From 894c5fe36a7755325407ccc745ce450fbf7c737a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 17:06:16 -0700 Subject: [PATCH 001/398] test(orchestration): fail loudly on an unexpected second detection call The mock overwrote resolveDetection on every call, so a second invocation would strand the first promise and hang to a 30s timeout instead of naming what changed. A test that hangs rather than fails is how a real bug gets mistaken for infrastructure noise. --- .../orchestration-legacy-coordinator-race.test.ts | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts b/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts index 9236d643e40..3040c37a9ef 100644 --- a/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts +++ b/src/main/runtime/rpc/orchestration-legacy-coordinator-race.test.ts @@ -656,9 +656,21 @@ describe('legacy coordinator takeover races', () => { const detectionStarted = new Promise((resolve) => { signalDetectionStarted = resolve }) + let detectionCalls = 0 vi.spyOn(harness.runtime, 'isTerminalRunningAgent').mockImplementation( () => - new Promise((resolve) => { + new Promise((resolve, reject) => { + detectionCalls += 1 + // Why reject instead of re-arming: a second call would overwrite resolveDetection and + // strand the first promise, hanging to a timeout instead of naming what changed. + if (detectionCalls > 1) { + reject( + new Error( + `isTerminalRunningAgent was called ${detectionCalls} times; this test drives exactly one detection.` + ) + ) + return + } resolveDetection = resolve signalDetectionStarted?.() }) From 401664298faf07f1e704d60f9ba689d7a962a2f8 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 16:11:12 -0700 Subject: [PATCH 002/398] fix(preload): make a dropped bridge key a compile error The split silently dropped jira.searchUsers and runtimeEnvironments.retryControlConnection. Neither failed typecheck: the bridge modules carried no satisfies annotation and the composed api object was unannotated, so a missing key was only a runtime TypeError in the renderer. Annotates each module against PreloadApi, the type window.api is already declared as, so the contract supplies the shape rather than a parallel copy. Deleting jira.searchUsers now fails with TS2741 naming the key. Turning this on surfaced 106 places where a bridge locally annotated Promise or unknown[] over a contract that declares concrete types -- the bridge was erasing types the renderer relied on. Those annotations are gone. Also exposes app.awaitBeforeUnloadCheckpoint, which was declared and called but never actually on the bridge, so the lazy-chunk recovery reload optional-chained to a no-op and navigated without joining the checkpoint. The missing key was caught by the new annotation rather than by hand. --- src/preload/api/agent-status-bridge.ts | 3 +- src/preload/api/agent-trust-bridge.ts | 3 +- src/preload/api/ai-vault-bridge.ts | 14 ++-- src/preload/api/app-bridge.ts | 3 +- src/preload/api/automations-bridge.ts | 4 +- src/preload/api/bitbucket-bridge.ts | 5 +- ...bridge-guest-registration-and-downloads.ts | 3 +- ...er-bridge-page-interaction-and-sessions.ts | 45 ++++-------- src/preload/api/browser-bridge.ts | 3 +- src/preload/api/claude-accounts-bridge.ts | 14 ++-- src/preload/api/claude-usage-bridge.ts | 6 +- src/preload/api/cli-bridge.ts | 3 +- src/preload/api/codex-accounts-bridge.ts | 18 +++-- src/preload/api/codex-config-sync-bridge.ts | 3 +- src/preload/api/codex-usage-bridge.ts | 6 +- .../api/computer-use-permissions-bridge.ts | 9 +-- src/preload/api/crash-reports-bridge.ts | 3 +- src/preload/api/dashboard-bridge.ts | 3 +- .../api/developer-permissions-bridge.ts | 10 +-- src/preload/api/diagnostics-bridge.ts | 9 +-- src/preload/api/doc-preview-bridge.ts | 3 +- src/preload/api/e2e-bridge.ts | 3 +- src/preload/api/emulator-bridge.ts | 3 +- src/preload/api/export-bridge.ts | 3 +- src/preload/api/feedback-bridge.ts | 3 +- src/preload/api/fs-bridge.ts | 3 +- .../api/gh-bridge-mutations-and-projects.ts | 25 +++---- .../gh-bridge-pull-requests-and-work-items.ts | 71 ++++++++++--------- src/preload/api/gh-bridge.ts | 6 +- src/preload/api/git-bash-bridge.ts | 3 +- src/preload/api/git-bridge.ts | 35 ++++----- src/preload/api/gl-bridge.ts | 3 +- src/preload/api/grok-accounts-bridge.ts | 3 +- src/preload/api/hooks-bridge.ts | 20 ++---- src/preload/api/hosted-review-bridge.ts | 12 ++-- src/preload/api/jira-bridge.ts | 66 ++++++----------- src/preload/api/keybindings-bridge.ts | 3 +- src/preload/api/linear-bridge.ts | 58 ++++++--------- src/preload/api/macos-tcc-prompts-bridge.ts | 11 +-- src/preload/api/memory-bridge.ts | 3 +- src/preload/api/minimax-credentials-bridge.ts | 3 +- src/preload/api/mobile-bridge.ts | 3 +- src/preload/api/native-chat-bridge.ts | 5 +- src/preload/api/notebook-bridge.ts | 3 +- src/preload/api/notifications-bridge.ts | 3 +- src/preload/api/onboarding-bridge.ts | 3 +- src/preload/api/open-code-usage-bridge.ts | 6 +- src/preload/api/pet-bridge.ts | 3 +- src/preload/api/plugins-bridge.ts | 7 +- src/preload/api/preflight-bridge.ts | 4 +- src/preload/api/pty-bridge-session-control.ts | 3 +- .../pty-bridge-stream-and-serialization.ts | 3 +- src/preload/api/pty-bridge.ts | 6 +- src/preload/api/pwsh-bridge.ts | 3 +- src/preload/api/rate-limits-bridge.ts | 3 +- src/preload/api/runtime-bridge.ts | 3 +- .../api/runtime-environments-bridge.ts | 3 +- src/preload/api/settings-bridge.ts | 16 ++--- src/preload/api/shell-bridge.ts | 3 +- src/preload/api/skills-bridge.ts | 3 +- src/preload/api/speech-bridge.ts | 3 +- src/preload/api/ssh-bridge.ts | 3 +- src/preload/api/star-nag-bridge.ts | 3 +- src/preload/api/stats-bridge.ts | 3 +- src/preload/api/terminal-preview-bridge.ts | 3 +- ...ui-bridge-clipboard-and-window-controls.ts | 3 +- .../api/ui-bridge-state-and-menu-commands.ts | 3 +- .../api/ui-bridge-tab-and-browser-commands.ts | 3 +- .../ui-bridge-terminal-and-session-tabs.ts | 3 +- src/preload/api/wsl-bridge.ts | 3 +- .../app-restart-checkpoint-routing.test.ts | 14 ++++ src/preload/gitlab.ts | 52 ++++++-------- src/preload/index.ts | 3 +- 73 files changed, 350 insertions(+), 339 deletions(-) diff --git a/src/preload/api/agent-status-bridge.ts b/src/preload/api/agent-status-bridge.ts index 5bc0757c5ca..b5ac5c8b5b2 100644 --- a/src/preload/api/agent-status-bridge.ts +++ b/src/preload/api/agent-status-bridge.ts @@ -6,6 +6,7 @@ import type { } from '../../shared/agent-status-types' import type { AgentInterruptInferenceRequest } from '../../shared/agent-interrupt-intent' import type { AgentQuestionAnsweredInferenceRequest } from '../../shared/agent-question-answered-intent' +import type { PreloadApi } from '../api-types' export const agentStatusApi = { /** Listen for agent status updates forwarded from native hook receivers. */ @@ -85,4 +86,4 @@ export const agentStatusApi = { }): void => { ipcRenderer.send('agentStatus:transferPaneAuthority', args) } -} +} satisfies PreloadApi['agentStatus'] diff --git a/src/preload/api/agent-trust-bridge.ts b/src/preload/api/agent-trust-bridge.ts index f27146d9cd3..5aca3fd805c 100644 --- a/src/preload/api/agent-trust-bridge.ts +++ b/src/preload/api/agent-trust-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const agentTrustApi = { markTrusted: (args: { @@ -6,4 +7,4 @@ export const agentTrustApi = { workspacePath: string connectionId?: string }): Promise => ipcRenderer.invoke('agentTrust:markTrusted', args) -} +} satisfies PreloadApi['agentTrust'] diff --git a/src/preload/api/ai-vault-bridge.ts b/src/preload/api/ai-vault-bridge.ts index ea9b2e1be14..917c9f02b63 100644 --- a/src/preload/api/ai-vault-bridge.ts +++ b/src/preload/api/ai-vault-bridge.ts @@ -10,19 +10,19 @@ import type { } from '../../shared/ai-vault-types' import type { AiVaultSessionTitlesArgs } from '../../shared/ai-vault-session-title' import type { AiVaultPrepareSessionResumeArgs } from '../../shared/ai-vault-resume-preparation' +import type { PreloadApi } from '../api-types' export const aiVaultApi = { - listSessions: (args?: AiVaultListArgs): Promise => - ipcRenderer.invoke('aiVault:listSessions', args), - resolveSessionTitles: (args: AiVaultSessionTitlesArgs): Promise => + listSessions: (args?: AiVaultListArgs) => ipcRenderer.invoke('aiVault:listSessions', args), + resolveSessionTitles: (args: AiVaultSessionTitlesArgs) => ipcRenderer.invoke('aiVault:resolveSessionTitles', args), cancelListSessions: (args: { requestToken: string }): Promise => ipcRenderer.invoke('aiVault:cancelListSessions', args), - prepareSessionResume: (args: AiVaultPrepareSessionResumeArgs): Promise => + prepareSessionResume: (args: AiVaultPrepareSessionResumeArgs) => ipcRenderer.invoke('aiVault:prepareSessionResume', args), - listSubagentSessions: (args: AiVaultSubagentListArgs): Promise => + listSubagentSessions: (args: AiVaultSubagentListArgs) => ipcRenderer.invoke('aiVault:listSubagentSessions', args), - getFirstUserPrompt: (args: AiVaultFirstUserPromptArgs): Promise => + getFirstUserPrompt: (args: AiVaultFirstUserPromptArgs) => ipcRenderer.invoke('aiVault:getFirstUserPrompt', args), deleteSession: (args: AiVaultDeleteSessionArgs): Promise => ipcRenderer.invoke('aiVault:deleteSession', args), @@ -31,4 +31,4 @@ export const aiVaultApi = { ipcRenderer.on('aiVault:windowFocused', listener) return () => ipcRenderer.removeListener('aiVault:windowFocused', listener) } -} +} satisfies PreloadApi['aiVault'] diff --git a/src/preload/api/app-bridge.ts b/src/preload/api/app-bridge.ts index 705d24e4bda..22d46cc40d2 100644 --- a/src/preload/api/app-bridge.ts +++ b/src/preload/api/app-bridge.ts @@ -40,6 +40,7 @@ export const appApi = { throw new Error('Failed to stage renderer state before unload.') } }, + awaitBeforeUnloadCheckpoint: () => awaitBeforeUnloadCheckpoint(), awaitFirstWindowStartupServices: (): Promise => ipcRenderer.invoke('app:awaitFirstWindowStartupServices'), prepareTerminalStartupRestoration: (): Promise => @@ -73,4 +74,4 @@ export const appApi = { ipcRenderer.invoke('app:pickFloatingWorkspaceDirectory'), writeTerminalRenderDesyncEvidence: (args: WriteTerminalRenderDesyncEvidenceArgs) => ipcRenderer.invoke('terminal:writeRenderDesyncEvidence', args) -} +} satisfies PreloadApi['app'] diff --git a/src/preload/api/automations-bridge.ts b/src/preload/api/automations-bridge.ts index 87bec12a8e9..43c3df528d8 100644 --- a/src/preload/api/automations-bridge.ts +++ b/src/preload/api/automations-bridge.ts @@ -1,5 +1,5 @@ import { ipcRenderer } from 'electron' -import type { ExternalAutomationManagerResult } from '../api-types' +import type { ExternalAutomationManagerResult, PreloadApi } from '../api-types' import type { AutomationDispatchRequest, AutomationDispatchResult, @@ -56,4 +56,4 @@ export const automationsApi = { ipcRenderer.on('automations:changed', listener) return () => ipcRenderer.removeListener('automations:changed', listener) } -} +} satisfies PreloadApi['automations'] diff --git a/src/preload/api/bitbucket-bridge.ts b/src/preload/api/bitbucket-bridge.ts index cf51ce453df..dd683b1387b 100644 --- a/src/preload/api/bitbucket-bridge.ts +++ b/src/preload/api/bitbucket-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const bitbucketApi = { connect: (args: { @@ -12,5 +13,5 @@ export const bitbucketApi = { disconnect: (): Promise => ipcRenderer.invoke('bitbucket:disconnect'), - status: (): Promise => ipcRenderer.invoke('bitbucket:status') -} + status: () => ipcRenderer.invoke('bitbucket:status') +} satisfies PreloadApi['bitbucket'] diff --git a/src/preload/api/browser-bridge-guest-registration-and-downloads.ts b/src/preload/api/browser-bridge-guest-registration-and-downloads.ts index 3714a3bca7c..9d782971d96 100644 --- a/src/preload/api/browser-bridge-guest-registration-and-downloads.ts +++ b/src/preload/api/browser-bridge-guest-registration-and-downloads.ts @@ -6,6 +6,7 @@ import type { } from '../../shared/browser-webauthn-account' import { readBrowserClientHostIdArgument } from '../../shared/browser-client-host-id-argument' import { browserClientPageRendererRequests } from '../preload-runtime-support' +import type { PreloadApi } from '../api-types' export const browserGuestRegistrationAndDownloadsApi = { onClientPageRendererRequest: browserClientPageRendererRequests.subscribe, @@ -194,4 +195,4 @@ export const browserGuestRegistrationAndDownloadsApi = { ipcRenderer.on('browser:download-finished', listener) return () => ipcRenderer.removeListener('browser:download-finished', listener) } -} +} satisfies Partial diff --git a/src/preload/api/browser-bridge-page-interaction-and-sessions.ts b/src/preload/api/browser-bridge-page-interaction-and-sessions.ts index 0e959e0af4f..93d6001e61c 100644 --- a/src/preload/api/browser-bridge-page-interaction-and-sessions.ts +++ b/src/preload/api/browser-bridge-page-interaction-and-sessions.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const browserPageInteractionAndSessionsApi = { onContextMenuRequested: ( @@ -81,23 +82,17 @@ export const browserPageInteractionAndSessionsApi = { }, cancelDownload: (args: { downloadId: string }): Promise => ipcRenderer.invoke('browser:cancelDownload', args), - setGrabMode: (args: { - browserPageId: string - enabled: boolean - }): Promise<{ ok: true } | { ok: false; reason: string }> => + setGrabMode: (args: { browserPageId: string; enabled: boolean }) => ipcRenderer.invoke('browser:setGrabMode', args), - awaitGrabSelection: (args: { browserPageId: string; opId: string }): Promise => + awaitGrabSelection: (args: { browserPageId: string; opId: string }) => ipcRenderer.invoke('browser:awaitGrabSelection', args), cancelGrab: (args: { browserPageId: string }): Promise => ipcRenderer.invoke('browser:cancelGrab', args), captureSelectionScreenshot: (args: { browserPageId: string rect: { x: number; y: number; width: number; height: number } - }): Promise<{ ok: true; screenshot: unknown } | { ok: false; reason: string }> => - ipcRenderer.invoke('browser:captureSelectionScreenshot', args), - extractHoverPayload: (args: { - browserPageId: string - }): Promise<{ ok: true; payload: unknown } | { ok: false; reason: string }> => + }) => ipcRenderer.invoke('browser:captureSelectionScreenshot', args), + extractHoverPayload: (args: { browserPageId: string }) => ipcRenderer.invoke('browser:extractHoverPayload', args), onGrabModeToggle: (callback: (browserPageId: string) => void): (() => void) => { const listener = (_event: Electron.IpcRendererEvent, browserPageId: string) => @@ -115,7 +110,7 @@ export const browserPageInteractionAndSessionsApi = { ipcRenderer.on('browser:grabActionShortcut', listener) return () => ipcRenderer.removeListener('browser:grabActionShortcut', listener) }, - sessionListProfiles: (): Promise => ipcRenderer.invoke('browser:session:listProfiles'), + sessionListProfiles: () => ipcRenderer.invoke('browser:session:listProfiles'), prepareSshWorkspacePartition: (args: { targetId: string browserProfileId?: string @@ -126,40 +121,28 @@ export const browserPageInteractionAndSessionsApi = { scope: 'default' | 'isolated' | 'imported' label: string userAgentMode?: 'clean' | 'native' - }): Promise => ipcRenderer.invoke('browser:session:createProfile', args), + }) => ipcRenderer.invoke('browser:session:createProfile', args), sessionDeleteProfile: (args: { profileId: string }): Promise => ipcRenderer.invoke('browser:session:deleteProfile', args), - sessionImportCookies: (args: { - profileId: string - }): Promise<{ ok: true; profileId: string; summary: unknown } | { ok: false; reason: string }> => + sessionImportCookies: (args: { profileId: string }) => ipcRenderer.invoke('browser:session:importCookies', args), sessionResolvePartition: (args: { profileId: string | null }): Promise => ipcRenderer.invoke('browser:session:resolvePartition', args), - sessionDetectBrowsers: (): Promise => - ipcRenderer.invoke('browser:session:detectBrowsers'), - sessionDetectBrowsersForClientHost: (args: { - environmentId: string - }): Promise => + sessionDetectBrowsers: () => ipcRenderer.invoke('browser:session:detectBrowsers'), + sessionDetectBrowsersForClientHost: (args: { environmentId: string }) => ipcRenderer.invoke('browser:session:detectBrowsersForClientHost', args), - sessionImportFromBrowser: (args: { - profileId: string - browserFamily: string - }): Promise<{ ok: true; profileId: string; summary: unknown } | { ok: false; reason: string }> => + sessionImportFromBrowser: (args: { profileId: string; browserFamily: string }) => ipcRenderer.invoke('browser:session:importFromBrowser', args), sessionImportFromBrowserForClientHost: (args: { environmentId: string profileId: string browserFamily: string browserProfile?: string - }): Promise< - { ok: true; profileId: string; summary: unknown } | { ok: false; reason: string } | null - > => ipcRenderer.invoke('browser:session:importFromBrowserForClientHost', args), - sessionClientRouteImportSources: (args: { - environmentId: string - }): Promise> => + }) => ipcRenderer.invoke('browser:session:importFromBrowserForClientHost', args), + sessionClientRouteImportSources: (args: { environmentId: string }) => ipcRenderer.invoke('browser:session:clientRouteImportSources', args), sessionClearDefaultCookies: (): Promise => ipcRenderer.invoke('browser:session:clearDefaultCookies'), notifyActiveTabChanged: (args: { browserPageId: string }): Promise => ipcRenderer.invoke('browser:activeTabChanged', args) -} +} satisfies Partial diff --git a/src/preload/api/browser-bridge.ts b/src/preload/api/browser-bridge.ts index dca222c5843..d29b7365dc2 100644 --- a/src/preload/api/browser-bridge.ts +++ b/src/preload/api/browser-bridge.ts @@ -1,7 +1,8 @@ import { browserGuestRegistrationAndDownloadsApi } from './browser-bridge-guest-registration-and-downloads' import { browserPageInteractionAndSessionsApi } from './browser-bridge-page-interaction-and-sessions' +import type { PreloadApi } from '../api-types' export const browserApi = { ...browserGuestRegistrationAndDownloadsApi, ...browserPageInteractionAndSessionsApi -} +} satisfies PreloadApi['browser'] diff --git a/src/preload/api/claude-accounts-bridge.ts b/src/preload/api/claude-accounts-bridge.ts index 8200c17791d..69586525962 100644 --- a/src/preload/api/claude-accounts-bridge.ts +++ b/src/preload/api/claude-accounts-bridge.ts @@ -1,18 +1,18 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const claudeAccountsApi = { - list: (): Promise => ipcRenderer.invoke('claudeAccounts:list'), - add: (args?: { runtime?: 'host' | 'wsl'; wslDistro?: string | null }): Promise => + list: () => ipcRenderer.invoke('claudeAccounts:list'), + add: (args?: { runtime?: 'host' | 'wsl'; wslDistro?: string | null }) => ipcRenderer.invoke('claudeAccounts:add', args), cancelPendingLogin: (): Promise => ipcRenderer.invoke('claudeAccounts:cancelPendingLogin'), - reauthenticate: (args: { accountId: string }): Promise => + reauthenticate: (args: { accountId: string }) => ipcRenderer.invoke('claudeAccounts:reauthenticate', args), - remove: (args: { accountId: string }): Promise => - ipcRenderer.invoke('claudeAccounts:remove', args), + remove: (args: { accountId: string }) => ipcRenderer.invoke('claudeAccounts:remove', args), select: (args: { accountId: string | null runtime?: 'host' | 'wsl' wslDistro?: string | null - }): Promise => ipcRenderer.invoke('claudeAccounts:select', args) -} + }) => ipcRenderer.invoke('claudeAccounts:select', args) +} satisfies PreloadApi['claudeAccounts'] diff --git a/src/preload/api/claude-usage-bridge.ts b/src/preload/api/claude-usage-bridge.ts index 0c81e35e3e8..1b98202d88c 100644 --- a/src/preload/api/claude-usage-bridge.ts +++ b/src/preload/api/claude-usage-bridge.ts @@ -1,4 +1,8 @@ import { ipcRenderer } from 'electron' import { createUsageProviderApi } from '../usage-provider-api' +import type { PreloadApi } from '../api-types' -export const claudeUsageApi = createUsageProviderApi(ipcRenderer, 'claudeUsage') +export const claudeUsageApi = createUsageProviderApi( + ipcRenderer, + 'claudeUsage' +) satisfies PreloadApi['claudeUsage'] diff --git a/src/preload/api/cli-bridge.ts b/src/preload/api/cli-bridge.ts index 8f811cbdd40..76b574a2f44 100644 --- a/src/preload/api/cli-bridge.ts +++ b/src/preload/api/cli-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' import type { CliInstallStatus } from '../../shared/cli-install-types' +import type { PreloadApi } from '../api-types' export const cliApi = { getInstallStatus: (): Promise => ipcRenderer.invoke('cli:getInstallStatus'), @@ -11,4 +12,4 @@ export const cliApi = { ipcRenderer.invoke('cli:installWsl', args), removeWsl: (args?: { distro?: string | null }): Promise => ipcRenderer.invoke('cli:removeWsl', args) -} +} satisfies PreloadApi['cli'] diff --git a/src/preload/api/codex-accounts-bridge.ts b/src/preload/api/codex-accounts-bridge.ts index ecd32e4b923..d085de45855 100644 --- a/src/preload/api/codex-accounts-bridge.ts +++ b/src/preload/api/codex-accounts-bridge.ts @@ -1,20 +1,18 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const codexAccountsApi = { - list: (): Promise => ipcRenderer.invoke('codexAccounts:list'), - add: (args?: { runtime?: 'host' | 'wsl'; wslDistro?: string | null }): Promise => + list: () => ipcRenderer.invoke('codexAccounts:list'), + add: (args?: { runtime?: 'host' | 'wsl'; wslDistro?: string | null }) => ipcRenderer.invoke('codexAccounts:add', args), - reauthenticate: (args: { - accountId: string - activateIfSelectionWasEmpty?: boolean - }): Promise => ipcRenderer.invoke('codexAccounts:reauthenticate', args), - remove: (args: { accountId: string }): Promise => - ipcRenderer.invoke('codexAccounts:remove', args), + reauthenticate: (args: { accountId: string; activateIfSelectionWasEmpty?: boolean }) => + ipcRenderer.invoke('codexAccounts:reauthenticate', args), + remove: (args: { accountId: string }) => ipcRenderer.invoke('codexAccounts:remove', args), select: (args: { accountId: string | null runtime?: 'host' | 'wsl' wslDistro?: string | null - }): Promise => ipcRenderer.invoke('codexAccounts:select', args), + }) => ipcRenderer.invoke('codexAccounts:select', args), listStalePanes: (args: { ptyIds: string[] }): Promise< @@ -29,4 +27,4 @@ export const codexAccountsApi = { ipcRenderer.invoke('codexAccounts:listRecordedPaneLanes', args), forgetStalePanes: (args: { ptyIds: string[] }): Promise => ipcRenderer.invoke('codexAccounts:forgetStalePanes', args) -} +} satisfies PreloadApi['codexAccounts'] diff --git a/src/preload/api/codex-config-sync-bridge.ts b/src/preload/api/codex-config-sync-bridge.ts index e6c8903a698..82eedfaf0ec 100644 --- a/src/preload/api/codex-config-sync-bridge.ts +++ b/src/preload/api/codex-config-sync-bridge.ts @@ -1,6 +1,7 @@ import { ipcRenderer } from 'electron' import type { CodexConfigSyncStatus } from '../../shared/codex-config-sync-types' +import type { PreloadApi } from '../api-types' export const codexConfigSyncApi = { status: (): Promise => ipcRenderer.invoke('codexConfigSync:status') -} +} satisfies PreloadApi['codexConfigSync'] diff --git a/src/preload/api/codex-usage-bridge.ts b/src/preload/api/codex-usage-bridge.ts index 9dba4b72f82..2575f2bb0e9 100644 --- a/src/preload/api/codex-usage-bridge.ts +++ b/src/preload/api/codex-usage-bridge.ts @@ -1,4 +1,8 @@ import { ipcRenderer } from 'electron' import { createUsageProviderApi } from '../usage-provider-api' +import type { PreloadApi } from '../api-types' -export const codexUsageApi = createUsageProviderApi(ipcRenderer, 'codexUsage') +export const codexUsageApi = createUsageProviderApi( + ipcRenderer, + 'codexUsage' +) satisfies PreloadApi['codexUsage'] diff --git a/src/preload/api/computer-use-permissions-bridge.ts b/src/preload/api/computer-use-permissions-bridge.ts index be441a8682e..bd36efbdcc3 100644 --- a/src/preload/api/computer-use-permissions-bridge.ts +++ b/src/preload/api/computer-use-permissions-bridge.ts @@ -1,8 +1,9 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const computerUsePermissionsApi = { - getStatus: (): Promise => ipcRenderer.invoke('computerUsePermissions:getStatus'), - openSetup: (args?: { id?: string }): Promise => + getStatus: () => ipcRenderer.invoke('computerUsePermissions:getStatus'), + openSetup: (args?: { id?: string }) => ipcRenderer.invoke('computerUsePermissions:openSetup', args), - reset: (): Promise => ipcRenderer.invoke('computerUsePermissions:reset') -} + reset: () => ipcRenderer.invoke('computerUsePermissions:reset') +} satisfies PreloadApi['computerUsePermissions'] diff --git a/src/preload/api/crash-reports-bridge.ts b/src/preload/api/crash-reports-bridge.ts index 19d7044601d..a77ad412eb7 100644 --- a/src/preload/api/crash-reports-bridge.ts +++ b/src/preload/api/crash-reports-bridge.ts @@ -11,6 +11,7 @@ import type { RendererHeapStatistics } from '../../shared/renderer-heap-statisti import type { RendererProcessMemory } from '../../shared/renderer-process-memory' import { readRendererHeapStatistics } from '../renderer-heap-statistics-reader' import { readRendererProcessMemory } from '../renderer-process-memory-reader' +import type { PreloadApi } from '../api-types' export const crashReportsApi = { getLatestPending: () => ipcRenderer.invoke('crashReports:getLatestPending'), @@ -28,4 +29,4 @@ export const crashReportsApi = { ipcRenderer.invoke('crashReports:copyLatestDiagnostics', args), readHeapStatistics: (): RendererHeapStatistics | null => readRendererHeapStatistics(), readProcessMemory: (): Promise => readRendererProcessMemory() -} +} satisfies PreloadApi['crashReports'] diff --git a/src/preload/api/dashboard-bridge.ts b/src/preload/api/dashboard-bridge.ts index 17241630857..e6862504da9 100644 --- a/src/preload/api/dashboard-bridge.ts +++ b/src/preload/api/dashboard-bridge.ts @@ -5,6 +5,7 @@ import type { DashboardSnapshot, DashboardSpawnAgentArgs } from '../../shared/dashboard-snapshot' +import type { PreloadApi } from '../api-types' export const dashboardApi = { // Open the pop-out dashboard window, or focus it if already open. @@ -71,4 +72,4 @@ export const dashboardApi = { ipcRenderer.invoke('dashboardPopout:spawnAgent', args), sleepWorkspace: (args: DashboardSleepWorkspaceArgs): Promise => ipcRenderer.invoke('dashboardPopout:sleepWorkspace', args) -} +} satisfies PreloadApi['dashboard'] diff --git a/src/preload/api/developer-permissions-bridge.ts b/src/preload/api/developer-permissions-bridge.ts index 1aaa51deed3..94158cd7818 100644 --- a/src/preload/api/developer-permissions-bridge.ts +++ b/src/preload/api/developer-permissions-bridge.ts @@ -1,11 +1,11 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const developerPermissionsApi = { - getStatus: (): Promise => ipcRenderer.invoke('developerPermissions:getStatus'), - request: (args: { id: string }): Promise => - ipcRenderer.invoke('developerPermissions:request', args), + getStatus: () => ipcRenderer.invoke('developerPermissions:getStatus'), + request: (args: { id: string }) => ipcRenderer.invoke('developerPermissions:request', args), openSettings: (args: { id: string }): Promise => ipcRenderer.invoke('developerPermissions:openSettings', args), - testLocalNetworkConnection: (args: { host: string; port: number }): Promise => + testLocalNetworkConnection: (args: { host: string; port: number }) => ipcRenderer.invoke('developerPermissions:testLocalNetworkConnection', args) -} +} satisfies PreloadApi['developerPermissions'] diff --git a/src/preload/api/diagnostics-bridge.ts b/src/preload/api/diagnostics-bridge.ts index 6d274f9b08a..bff79fe5817 100644 --- a/src/preload/api/diagnostics-bridge.ts +++ b/src/preload/api/diagnostics-bridge.ts @@ -1,15 +1,16 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const diagnosticsApi = { - getStatus: (): Promise => ipcRenderer.invoke('diagnostics:getStatus'), - collectBundle: (lookbackMinutes?: number): Promise => + getStatus: () => ipcRenderer.invoke('diagnostics:getStatus'), + collectBundle: (lookbackMinutes?: number) => ipcRenderer.invoke('diagnostics:collectBundle', lookbackMinutes), openBundlePreview: (bundleSubmissionId: string): Promise => ipcRenderer.invoke('diagnostics:openBundlePreview', bundleSubmissionId), discardBundlePreview: (bundleSubmissionId: string): Promise => ipcRenderer.invoke('diagnostics:discardBundlePreview', bundleSubmissionId), - uploadBundle: (bundleSubmissionId: string): Promise => + uploadBundle: (bundleSubmissionId: string) => ipcRenderer.invoke('diagnostics:uploadBundle', bundleSubmissionId), deleteBundle: (ticketId: string): Promise => ipcRenderer.invoke('diagnostics:deleteBundle', ticketId) -} +} satisfies PreloadApi['diagnostics'] diff --git a/src/preload/api/doc-preview-bridge.ts b/src/preload/api/doc-preview-bridge.ts index 68a099d5ed2..97fbb8da975 100644 --- a/src/preload/api/doc-preview-bridge.ts +++ b/src/preload/api/doc-preview-bridge.ts @@ -8,6 +8,7 @@ import { type DocPreviewFailure } from '../../shared/doc-preview-scheme' import type { DocPreviewGrantRequest } from '../api/doc-preview-api' +import type { PreloadApi } from '../api-types' export const docPreviewApi = { mintGrant: (request: DocPreviewGrantRequest): Promise<{ grantId: string; url: string }> => @@ -28,4 +29,4 @@ export const docPreviewApi = { ipcRenderer.on(DOC_PREVIEW_LOAD_FAILURE_CHANNEL, listener) return () => ipcRenderer.removeListener(DOC_PREVIEW_LOAD_FAILURE_CHANNEL, listener) } -} +} satisfies PreloadApi['docPreview'] diff --git a/src/preload/api/e2e-bridge.ts b/src/preload/api/e2e-bridge.ts index 2876b72a265..900b17a9fcf 100644 --- a/src/preload/api/e2e-bridge.ts +++ b/src/preload/api/e2e-bridge.ts @@ -1,5 +1,6 @@ import { preloadE2EConfig } from '../e2e-config' +import type { PreloadApi } from '../api-types' export const e2eApi = { getConfig: () => preloadE2EConfig -} +} satisfies PreloadApi['e2e'] diff --git a/src/preload/api/emulator-bridge.ts b/src/preload/api/emulator-bridge.ts index f13e99fc52a..8ab56471a42 100644 --- a/src/preload/api/emulator-bridge.ts +++ b/src/preload/api/emulator-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const emulatorApi = { startFrameStream: (args: { @@ -95,4 +96,4 @@ export const emulatorApi = { ipcRenderer.on('ui:emulatorAutoAttach', listener) return () => ipcRenderer.removeListener('ui:emulatorAutoAttach', listener) } -} +} satisfies PreloadApi['emulator'] diff --git a/src/preload/api/export-bridge.ts b/src/preload/api/export-bridge.ts index 637d4bea2ef..67ffbfae109 100644 --- a/src/preload/api/export-bridge.ts +++ b/src/preload/api/export-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const exportApi = { htmlToPdf: (args: { @@ -7,4 +8,4 @@ export const exportApi = { }): Promise< { success: true; filePath: string } | { success: false; cancelled?: boolean; error?: string } > => ipcRenderer.invoke('export:html-to-pdf', args) -} +} satisfies PreloadApi['export'] diff --git a/src/preload/api/feedback-bridge.ts b/src/preload/api/feedback-bridge.ts index 55b3b5fcaa7..4241c5cacf9 100644 --- a/src/preload/api/feedback-bridge.ts +++ b/src/preload/api/feedback-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const feedbackApi = { submit: (args: { @@ -10,4 +11,4 @@ export const feedbackApi = { }): Promise< { ok: true; imagesDelivered?: boolean } | { ok: false; status: number | null; error: string } > => ipcRenderer.invoke('feedback:submit', args) -} +} satisfies PreloadApi['feedback'] diff --git a/src/preload/api/fs-bridge.ts b/src/preload/api/fs-bridge.ts index 538480ce2f5..c67e9abd9e3 100644 --- a/src/preload/api/fs-bridge.ts +++ b/src/preload/api/fs-bridge.ts @@ -8,6 +8,7 @@ import type { LocalLogTailReadResult, LocalLogTailWatchArgs } from '../../shared/local-log-tail-types' +import type { PreloadApi } from '../api-types' export const fsApi = { readDir: (args: { @@ -216,4 +217,4 @@ export const fsApi = { ipcRenderer.on('fs:changed', listener) return () => ipcRenderer.removeListener('fs:changed', listener) } -} +} satisfies PreloadApi['fs'] diff --git a/src/preload/api/gh-bridge-mutations-and-projects.ts b/src/preload/api/gh-bridge-mutations-and-projects.ts index 3103cb432ec..80b746bfb79 100644 --- a/src/preload/api/gh-bridge-mutations-and-projects.ts +++ b/src/preload/api/gh-bridge-mutations-and-projects.ts @@ -35,11 +35,12 @@ import type { UpdateProjectItemFieldArgs } from '../../shared/github/project-request-types' import type { AppStarSource } from '../../shared/gh-star-source' +import type { PreloadApi } from '../api-types' export const ghMutationsAndProjectsApi = { setPRAutoMerge: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number enabled: boolean @@ -49,7 +50,7 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:setPRAutoMerge', args), updatePRState: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number updates: { state: 'open' | 'closed' } @@ -58,7 +59,7 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:updatePRState', args), markPRReadyForReview: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number prRepo?: GitHubOwnerRepo | null @@ -66,7 +67,7 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:markPRReadyForReview', args), requestPRReviewers: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number reviewers: string[] @@ -75,7 +76,7 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:requestPRReviewers', args), removePRReviewers: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number reviewers: string[] @@ -84,7 +85,7 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:removePRReviewers', args), updateIssue: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null number: number updates: unknown @@ -92,7 +93,7 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:updateIssue', args), addIssueComment: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null number: number body: string @@ -101,7 +102,7 @@ export const ghMutationsAndProjectsApi = { }): Promise => ipcRenderer.invoke('gh:addIssueComment', args), addPRReviewCommentReply: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number commentId: number @@ -113,7 +114,7 @@ export const ghMutationsAndProjectsApi = { }): Promise => ipcRenderer.invoke('gh:addPRReviewCommentReply', args), addPRReviewComment: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number prRepo?: GitHubOwnerRepo | null @@ -125,12 +126,12 @@ export const ghMutationsAndProjectsApi = { }): Promise => ipcRenderer.invoke('gh:addPRReviewComment', args), listLabels: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null }): Promise => ipcRenderer.invoke('gh:listLabels', args), listAssignableUsers: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null }): Promise => ipcRenderer.invoke('gh:listAssignableUsers', args), onWorkItemMutated: ( @@ -199,4 +200,4 @@ export const ghMutationsAndProjectsApi = { ipcRenderer.invoke('gh:listIssueTypesBySlug', args), updateIssueTypeBySlug: (args: UpdateIssueTypeBySlugArgs): Promise => ipcRenderer.invoke('gh:updateIssueTypeBySlug', args) -} +} satisfies Partial diff --git a/src/preload/api/gh-bridge-pull-requests-and-work-items.ts b/src/preload/api/gh-bridge-pull-requests-and-work-items.ts index a4f3e1d45ef..3e4a5f6ce2a 100644 --- a/src/preload/api/gh-bridge-pull-requests-and-work-items.ts +++ b/src/preload/api/gh-bridge-pull-requests-and-work-items.ts @@ -9,33 +9,34 @@ import type { GitHubOwnerRepo } from '../../shared/github/pull-request-types' import type { GitHubWorkItem, ListWorkItemsResult } from '../../shared/github/work-item-types' import type { GitHubCreateIssueResult } from '../../shared/issue-mutation-types' import type { TaskSourceContext } from '../../shared/task-source-context' +import type { PreloadApi } from '../api-types' export const ghPullRequestsAndWorkItemsApi = { - viewer: (): Promise => ipcRenderer.invoke('gh:viewer'), - repoSlug: (args: { repoPath: string; repoId?: string }): Promise => + viewer: () => ipcRenderer.invoke('gh:viewer'), + repoSlug: (args: { repoPath: string; repoId?: string }) => ipcRenderer.invoke('gh:repoSlug', args), - repoUpstream: (args: { repoPath: string; repoId?: string }): Promise => + repoUpstream: (args: { repoPath: string; repoId?: string }) => ipcRenderer.invoke('gh:repoUpstream', args), prForBranch: (args: { repoPath: string - repoId?: string + repoId?: string | null branch: string linkedPRNumber?: number | null fallbackPRNumber?: number | null acceptMergedFallbackPR?: boolean currentHeadOid?: string | null - }): Promise => ipcRenderer.invoke('gh:prForBranch', args), - refreshPRNow: (args: { candidate: GitHubPRRefreshCandidate }): Promise => + }) => ipcRenderer.invoke('gh:prForBranch', args), + refreshPRNow: (args: { candidate: GitHubPRRefreshCandidate }) => ipcRenderer.invoke('gh:refreshPRNow', args), enqueuePRRefresh: (args: { candidate: GitHubPRRefreshCandidate reason: GitHubPRRefreshReason priority?: number - }): Promise => ipcRenderer.invoke('gh:enqueuePRRefresh', args), + }) => ipcRenderer.invoke('gh:enqueuePRRefresh', args), reportVisiblePRRefreshCandidates: (args: { candidates: GitHubPRRefreshCandidate[] generation: number - }): Promise => ipcRenderer.invoke('gh:reportVisiblePRRefreshCandidates', args), + }) => ipcRenderer.invoke('gh:reportVisiblePRRefreshCandidates', args), onPRRefreshEvent: (callback: (event: GitHubPRRefreshEvent) => void): (() => void) => { const listener = (_event: Electron.IpcRendererEvent, event: GitHubPRRefreshEvent): void => callback(event) @@ -44,42 +45,42 @@ export const ghPullRequestsAndWorkItemsApi = { }, issue: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null number: number - }): Promise => ipcRenderer.invoke('gh:issue', args), + }) => ipcRenderer.invoke('gh:issue', args), workItem: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null number: number type?: 'issue' | 'pr' - }): Promise => ipcRenderer.invoke('gh:workItem', args), + }) => ipcRenderer.invoke('gh:workItem', args), workItemByOwnerRepo: (args: { repoPath: string - repoId?: string + repoId?: string | null owner: string repo: string host?: string number: number type: 'issue' | 'pr' - }): Promise => ipcRenderer.invoke('gh:workItemByOwnerRepo', args), + }) => ipcRenderer.invoke('gh:workItemByOwnerRepo', args), workItemDetails: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null number: number type?: 'issue' | 'pr' - }): Promise => ipcRenderer.invoke('gh:workItemDetails', args), + }) => ipcRenderer.invoke('gh:workItemDetails', args), notifyWorkItemMutated: (args: { repoPath: string - repoId?: string + repoId?: string | null type: 'issue' | 'pr' number: number }): Promise => ipcRenderer.invoke('gh:notifyWorkItemMutated', args), prFileContents: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number prRepo?: GitHubOwnerRepo | null @@ -88,12 +89,12 @@ export const ghPullRequestsAndWorkItemsApi = { status: string headSha: string baseSha: string - }): Promise => ipcRenderer.invoke('gh:prFileContents', args), - listIssues: (args: { repoPath: string; repoId?: string; limit?: number }): Promise => + }) => ipcRenderer.invoke('gh:prFileContents', args), + listIssues: (args: { repoPath: string; repoId?: string; limit?: number }) => ipcRenderer.invoke('gh:listIssues', args), createIssue: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null title: string body: string @@ -104,7 +105,7 @@ export const ghPullRequestsAndWorkItemsApi = { ipcRenderer.invoke('gh:countWorkItems', args), listWorkItems: (args: { repoPath: string - repoId?: string + repoId?: string | null limit?: number query?: string page?: number @@ -113,26 +114,26 @@ export const ghPullRequestsAndWorkItemsApi = { ipcRenderer.invoke('gh:listWorkItems', args), prChecks: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number headSha?: string prRepo?: GitHubOwnerRepo | null noCache?: boolean - }): Promise => ipcRenderer.invoke('gh:prChecks', args), + }) => ipcRenderer.invoke('gh:prChecks', args), prCheckDetails: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null checkRunId?: number workflowRunId?: number checkName?: string url?: string | null prRepo?: GitHubOwnerRepo | null - }): Promise => ipcRenderer.invoke('gh:prCheckDetails', args), + }) => ipcRenderer.invoke('gh:prCheckDetails', args), rerunPRChecks: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number headSha?: string @@ -142,15 +143,15 @@ export const ghPullRequestsAndWorkItemsApi = { ipcRenderer.invoke('gh:rerunPRChecks', args), prComments: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number prRepo?: GitHubOwnerRepo | null noCache?: boolean - }): Promise => ipcRenderer.invoke('gh:prComments', args), + }) => ipcRenderer.invoke('gh:prComments', args), setPRCommentReaction: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null reactionSubjectId: string content: GitHubReactionContent @@ -159,7 +160,7 @@ export const ghPullRequestsAndWorkItemsApi = { }): Promise => ipcRenderer.invoke('gh:setPRCommentReaction', args), resolveReviewThread: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null threadId: string resolve: boolean @@ -167,7 +168,7 @@ export const ghPullRequestsAndWorkItemsApi = { }): Promise => ipcRenderer.invoke('gh:resolveReviewThread', args), setPRFileViewed: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number prRepo?: GitHubOwnerRepo | null @@ -177,17 +178,17 @@ export const ghPullRequestsAndWorkItemsApi = { }): Promise => ipcRenderer.invoke('gh:setPRFileViewed', args), updatePRTitle: (args: { repoPath: string - repoId?: string + repoId?: string | null prNumber: number title: string prRepo?: GitHubOwnerRepo | null }): Promise => ipcRenderer.invoke('gh:updatePRTitle', args), mergePR: (args: { repoPath: string - repoId?: string + repoId?: string | null sourceContext?: TaskSourceContext | null prNumber: number method?: 'merge' | 'squash' | 'rebase' prRepo?: GitHubOwnerRepo | null }): Promise<{ ok: true } | { ok: false; error: string }> => ipcRenderer.invoke('gh:mergePR', args) -} +} satisfies Partial diff --git a/src/preload/api/gh-bridge.ts b/src/preload/api/gh-bridge.ts index 52c21a966a7..c7da93698e6 100644 --- a/src/preload/api/gh-bridge.ts +++ b/src/preload/api/gh-bridge.ts @@ -1,4 +1,8 @@ +import type { PreloadApi } from '../api-types' import { ghPullRequestsAndWorkItemsApi } from './gh-bridge-pull-requests-and-work-items' import { ghMutationsAndProjectsApi } from './gh-bridge-mutations-and-projects' -export const ghApi = { ...ghPullRequestsAndWorkItemsApi, ...ghMutationsAndProjectsApi } +export const ghApi = { + ...ghPullRequestsAndWorkItemsApi, + ...ghMutationsAndProjectsApi +} satisfies PreloadApi['gh'] diff --git a/src/preload/api/git-bash-bridge.ts b/src/preload/api/git-bash-bridge.ts index bd62dfc614a..c2186d12aa3 100644 --- a/src/preload/api/git-bash-bridge.ts +++ b/src/preload/api/git-bash-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const gitBashApi = { isAvailable: (): Promise => ipcRenderer.invoke('gitBash:isAvailable') -} +} satisfies PreloadApi['gitBash'] diff --git a/src/preload/api/git-bridge.ts b/src/preload/api/git-bridge.ts index 89dfc8db5ae..987abab14fc 100644 --- a/src/preload/api/git-bridge.ts +++ b/src/preload/api/git-bridge.ts @@ -3,6 +3,7 @@ import type { GitForkSyncExpectedUpstream, GitForkSyncResult } from '../../share import type { GitStagingArea, GitUpstreamStatus } from '../../shared/git-status-types' import type { GitPushTarget } from '../../shared/worktree/types' import type { GitHistoryOptions, GitHistoryResult } from '../../shared/git-history' +import type { PreloadApi } from '../api-types' export const gitApi = { status: (args: { @@ -13,7 +14,7 @@ export const gitApi = { reuseLineStats?: boolean branchLineTotalMergeBase?: string requestToken?: string - }): Promise => ipcRenderer.invoke('git:status', args), + }) => ipcRenderer.invoke('git:status', args), cancelStatus: (args: { requestToken: string }): Promise => ipcRenderer.invoke('git:cancelStatus', args), setStatusUpstreamRefWatch: (args: { @@ -29,7 +30,7 @@ export const gitApi = { submodulePath: string connectionId?: string area?: GitStagingArea - }): Promise => ipcRenderer.invoke('git:submoduleStatus', args), + }) => ipcRenderer.invoke('git:submoduleStatus', args), checkIgnored: (args: { worktreePath: string paths: string[] @@ -42,7 +43,7 @@ export const gitApi = { history: ( args: { worktreePath: string; connectionId?: string } & GitHistoryOptions ): Promise => ipcRenderer.invoke('git:history', args), - conflictOperation: (args: { worktreePath: string; connectionId?: string }): Promise => + conflictOperation: (args: { worktreePath: string; connectionId?: string }) => ipcRenderer.invoke('git:conflictOperation', args), abortMerge: (args: { worktreePath: string; connectionId?: string }): Promise => ipcRenderer.invoke('git:abortMerge', args), @@ -54,17 +55,11 @@ export const gitApi = { staged: boolean compareAgainstHead?: boolean connectionId?: string - }): Promise => ipcRenderer.invoke('git:diff', args), - branchCompare: (args: { - worktreePath: string - baseRef: string - connectionId?: string - }): Promise => ipcRenderer.invoke('git:branchCompare', args), - commitCompare: (args: { - worktreePath: string - commitId: string - connectionId?: string - }): Promise => ipcRenderer.invoke('git:commitCompare', args), + }) => ipcRenderer.invoke('git:diff', args), + branchCompare: (args: { worktreePath: string; baseRef: string; connectionId?: string }) => + ipcRenderer.invoke('git:branchCompare', args), + commitCompare: (args: { worktreePath: string; commitId: string; connectionId?: string }) => + ipcRenderer.invoke('git:commitCompare', args), upstreamStatus: (args: { worktreePath: string connectionId?: string @@ -108,7 +103,7 @@ export const gitApi = { filePath: string oldPath?: string connectionId?: string - }): Promise => ipcRenderer.invoke('git:branchDiff', args), + }) => ipcRenderer.invoke('git:branchDiff', args), commitDiff: (args: { worktreePath: string commitOid: string @@ -116,7 +111,7 @@ export const gitApi = { filePath: string oldPath?: string connectionId?: string - }): Promise => ipcRenderer.invoke('git:commitDiff', args), + }) => ipcRenderer.invoke('git:commitDiff', args), commit: (args: { worktreePath: string message: string @@ -130,12 +125,12 @@ export const gitApi = { sourceControlAiResolvedParams?: unknown sourceControlAi?: unknown agentCmdOverrides?: Record - }): Promise => ipcRenderer.invoke('git:generateCommitMessage', args), + }) => ipcRenderer.invoke('git:generateCommitMessage', args), discoverCommitMessageModels: (args: { agentId: string worktreePath?: string connectionId?: string - }): Promise => ipcRenderer.invoke('git:discoverCommitMessageModels', args), + }) => ipcRenderer.invoke('git:discoverCommitMessageModels', args), cancelGenerateCommitMessage: (args: { worktreePath: string connectionId?: string @@ -154,7 +149,7 @@ export const gitApi = { sourceControlAiResolvedParams?: unknown sourceControlAi?: unknown agentCmdOverrides?: Record - }): Promise => ipcRenderer.invoke('git:generatePullRequestFields', args), + }) => ipcRenderer.invoke('git:generatePullRequestFields', args), cancelGeneratePullRequestFields: (args: { worktreePath: string connectionId?: string @@ -197,4 +192,4 @@ export const gitApi = { sha: string connectionId?: string }): Promise => ipcRenderer.invoke('git:remoteCommitUrl', args) -} +} satisfies PreloadApi['git'] diff --git a/src/preload/api/gl-bridge.ts b/src/preload/api/gl-bridge.ts index c48794a4976..3977536a838 100644 --- a/src/preload/api/gl-bridge.ts +++ b/src/preload/api/gl-bridge.ts @@ -1,3 +1,4 @@ import { glApi } from '../gitlab' +import type { PreloadApi } from '../api-types' -export const glApiBridge = glApi +export const glApiBridge = glApi satisfies PreloadApi['gl'] diff --git a/src/preload/api/grok-accounts-bridge.ts b/src/preload/api/grok-accounts-bridge.ts index b4719905ec7..246fc97bc56 100644 --- a/src/preload/api/grok-accounts-bridge.ts +++ b/src/preload/api/grok-accounts-bridge.ts @@ -1,6 +1,7 @@ import { ipcRenderer } from 'electron' import type { GrokAccountStatus } from '../../shared/rate-limit-types' +import type { PreloadApi } from '../api-types' export const grokAccountsApi = { getStatus: (): Promise => ipcRenderer.invoke('grokAccounts:getStatus') -} +} satisfies PreloadApi['grokAccounts'] diff --git a/src/preload/api/hooks-bridge.ts b/src/preload/api/hooks-bridge.ts index 59a6fee71b6..75c48281b16 100644 --- a/src/preload/api/hooks-bridge.ts +++ b/src/preload/api/hooks-bridge.ts @@ -1,22 +1,14 @@ import { ipcRenderer } from 'electron' import type { WorktreeSetupLaunch } from '../../shared/worktree/launch-types' import type { ExecutionHostId } from '../../shared/execution-host' +import type { PreloadApi } from '../api-types' export const hooksApi = { - check: (args: { - repoId: string - hostId?: ExecutionHostId - }): Promise<{ - status?: 'ok' | 'error' - hasHooks: boolean - hooks: unknown - mayNeedUpdate: boolean - }> => ipcRenderer.invoke('hooks:check', args), + check: (args: { repoId: string; hostId?: ExecutionHostId }) => + ipcRenderer.invoke('hooks:check', args), - inspectSetupScriptImports: (args: { - repoId: string - hostId?: ExecutionHostId - }): Promise => ipcRenderer.invoke('hooks:inspectSetupScriptImports', args), + inspectSetupScriptImports: (args: { repoId: string; hostId?: ExecutionHostId }) => + ipcRenderer.invoke('hooks:inspectSetupScriptImports', args), createIssueCommandRunner: (args: { repoId: string @@ -41,4 +33,4 @@ export const hooksApi = { content: string hostId?: ExecutionHostId }): Promise => ipcRenderer.invoke('hooks:writeIssueCommand', args) -} +} satisfies PreloadApi['hooks'] diff --git a/src/preload/api/hosted-review-bridge.ts b/src/preload/api/hosted-review-bridge.ts index dc8323b5bf3..b7e0d12af5c 100644 --- a/src/preload/api/hosted-review-bridge.ts +++ b/src/preload/api/hosted-review-bridge.ts @@ -1,12 +1,12 @@ import { ipcRenderer } from 'electron' import type { HostedReviewForBranchArgs } from '../../shared/hosted-review' +import type { PreloadApi } from '../api-types' export const hostedReviewApi = { - forBranch: (args: HostedReviewForBranchArgs): Promise => + forBranch: (args: HostedReviewForBranchArgs) => ipcRenderer.invoke('hostedReview:forBranch', args), - getCreationEligibility: (args: unknown): Promise => + getCreationEligibility: (args: unknown) => ipcRenderer.invoke('hostedReview:getCreationEligibility', args), - create: (args: unknown): Promise => ipcRenderer.invoke('hostedReview:create', args), - createStacked: (args: unknown): Promise => - ipcRenderer.invoke('hostedReview:createStacked', args) -} + create: (args: unknown) => ipcRenderer.invoke('hostedReview:create', args), + createStacked: (args: unknown) => ipcRenderer.invoke('hostedReview:createStacked', args) +} satisfies PreloadApi['hostedReview'] diff --git a/src/preload/api/jira-bridge.ts b/src/preload/api/jira-bridge.ts index 47a27b77f68..4b6b991095d 100644 --- a/src/preload/api/jira-bridge.ts +++ b/src/preload/api/jira-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' import type { JiraProjectStatusOrder } from '../../shared/jira-types' +import type { PreloadApi } from '../api-types' export const jiraApi = { connect: (args: { @@ -7,30 +8,21 @@ export const jiraApi = { email: string apiToken: string authType?: 'cloud' | 'server' - }): Promise<{ ok: true; viewer: unknown } | { ok: false; error: string }> => - ipcRenderer.invoke('jira:connect', args), + }) => ipcRenderer.invoke('jira:connect', args), disconnect: (args?: { siteId?: string }): Promise => ipcRenderer.invoke('jira:disconnect', args), - selectSite: (args: { siteId: string }): Promise => - ipcRenderer.invoke('jira:selectSite', args), + selectSite: (args: { siteId: string }) => ipcRenderer.invoke('jira:selectSite', args), - status: (): Promise => ipcRenderer.invoke('jira:status'), + status: () => ipcRenderer.invoke('jira:status'), - readStatus: (): Promise => ipcRenderer.invoke('jira:readStatus'), + readStatus: () => ipcRenderer.invoke('jira:readStatus'), - testConnection: (args?: { - siteId?: string - }): Promise<{ ok: true; viewer: unknown } | { ok: false; error: string }> => - ipcRenderer.invoke('jira:testConnection', args), + testConnection: (args?: { siteId?: string }) => ipcRenderer.invoke('jira:testConnection', args), - searchIssues: (args: { - jql: string - limit?: number - siteId?: string - requestId?: string - }): Promise => ipcRenderer.invoke('jira:searchIssues', args), + searchIssues: (args: { jql: string; limit?: number; siteId?: string; requestId?: string }) => + ipcRenderer.invoke('jira:searchIssues', args), cancelSearchIssues: (args: { requestId: string }): Promise => ipcRenderer.invoke('jira:cancelSearchIssues', args), @@ -38,16 +30,12 @@ export const jiraApi = { filter?: 'assigned' | 'reported' | 'all' | 'done' limit?: number siteId?: string - }): Promise => ipcRenderer.invoke('jira:listIssues', args), + }) => ipcRenderer.invoke('jira:listIssues', args), - getIssue: (args: { key: string; siteId?: string }): Promise => - ipcRenderer.invoke('jira:getIssue', args), + getIssue: (args: { key: string; siteId?: string }) => ipcRenderer.invoke('jira:getIssue', args), - lookupIssueSummary: (args: { - key: string - siteId: string - requestId?: string - }): Promise => ipcRenderer.invoke('jira:lookupIssueSummary', args), + lookupIssueSummary: (args: { key: string; siteId: string; requestId?: string }) => + ipcRenderer.invoke('jira:lookupIssueSummary', args), cancelIssueSummary: (args: { requestId: string }): Promise => ipcRenderer.invoke('jira:cancelIssueSummary', args), @@ -75,36 +63,28 @@ export const jiraApi = { }): Promise<{ ok: true; id: string } | { ok: false; error: string }> => ipcRenderer.invoke('jira:addIssueComment', args), - issueComments: (args: { key: string; siteId?: string }): Promise => + issueComments: (args: { key: string; siteId?: string }) => ipcRenderer.invoke('jira:issueComments', args), - listProjects: (args?: { siteId?: string }): Promise => - ipcRenderer.invoke('jira:listProjects', args), + listProjects: (args?: { siteId?: string }) => ipcRenderer.invoke('jira:listProjects', args), - listIssueTypes: (args: { projectIdOrKey: string; siteId?: string }): Promise => + listIssueTypes: (args: { projectIdOrKey: string; siteId?: string }) => ipcRenderer.invoke('jira:listIssueTypes', args), - listCreateFields: (args: { - projectIdOrKey: string - issueTypeId: string - siteId?: string - }): Promise => ipcRenderer.invoke('jira:listCreateFields', args), + listCreateFields: (args: { projectIdOrKey: string; issueTypeId: string; siteId?: string }) => + ipcRenderer.invoke('jira:listCreateFields', args), - listPriorities: (args?: { siteId?: string }): Promise => - ipcRenderer.invoke('jira:listPriorities', args), + listPriorities: (args?: { siteId?: string }) => ipcRenderer.invoke('jira:listPriorities', args), - listAssignableUsers: (args: { - key: string - query?: string - siteId?: string - }): Promise => ipcRenderer.invoke('jira:listAssignableUsers', args), - searchUsers: (args?: { query?: string; siteId?: string }): Promise => + listAssignableUsers: (args: { key: string; query?: string; siteId?: string }) => + ipcRenderer.invoke('jira:listAssignableUsers', args), + searchUsers: (args?: { query?: string; siteId?: string }) => ipcRenderer.invoke('jira:searchUsers', args), - listTransitions: (args: { key: string; siteId?: string }): Promise => + listTransitions: (args: { key: string; siteId?: string }) => ipcRenderer.invoke('jira:listTransitions', args), getProjectStatusOrder: (args: { projectKey: string siteId?: string }): Promise => ipcRenderer.invoke('jira:getProjectStatusOrder', args) -} +} satisfies PreloadApi['jira'] diff --git a/src/preload/api/keybindings-bridge.ts b/src/preload/api/keybindings-bridge.ts index 3111ddd6676..e91e587313d 100644 --- a/src/preload/api/keybindings-bridge.ts +++ b/src/preload/api/keybindings-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' import type { KeybindingActionId, KeybindingFileSnapshot } from '../../shared/keybindings' +import type { PreloadApi } from '../api-types' export const keybindingsApi = { get: (): Promise => ipcRenderer.invoke('keybindings:get'), @@ -17,4 +18,4 @@ export const keybindingsApi = { ipcRenderer.on('keybindings:changed', listener) return () => ipcRenderer.removeListener('keybindings:changed', listener) } -} +} satisfies PreloadApi['keybindings'] diff --git a/src/preload/api/linear-bridge.ts b/src/preload/api/linear-bridge.ts index cd6ec0e9c15..8092a8e70e7 100644 --- a/src/preload/api/linear-bridge.ts +++ b/src/preload/api/linear-bridge.ts @@ -1,37 +1,30 @@ import { ipcRenderer } from 'electron' import type { LinearProjectDetail } from '../../shared/linear/project-types' +import type { PreloadApi } from '../api-types' export const linearApi = { - connect: (args: { - apiKey: string - }): Promise<{ ok: true; viewer: unknown } | { ok: false; error: string }> => - ipcRenderer.invoke('linear:connect', args), + connect: (args: { apiKey: string }) => ipcRenderer.invoke('linear:connect', args), disconnect: (args?: { workspaceId?: string }): Promise => ipcRenderer.invoke('linear:disconnect', args), - selectWorkspace: (args: { workspaceId: string }): Promise => + selectWorkspace: (args: { workspaceId: string }) => ipcRenderer.invoke('linear:selectWorkspace', args), - status: (): Promise => ipcRenderer.invoke('linear:status'), + status: () => ipcRenderer.invoke('linear:status'), - testConnection: (args?: { - workspaceId?: string - }): Promise<{ ok: true; viewer: unknown } | { ok: false; error: string }> => + testConnection: (args?: { workspaceId?: string }) => ipcRenderer.invoke('linear:testConnection', args), - searchIssues: (args: { - query: string - limit?: number - workspaceId?: string - }): Promise => ipcRenderer.invoke('linear:searchIssues', args), + searchIssues: (args: { query: string; limit?: number; workspaceId?: string }) => + ipcRenderer.invoke('linear:searchIssues', args), listIssues: (args?: { filter?: 'assigned' | 'created' | 'all' | 'completed' limit?: number workspaceId?: string attributeFilter?: unknown - }): Promise => ipcRenderer.invoke('linear:listIssues', args), + }) => ipcRenderer.invoke('linear:listIssues', args), createIssue: (args: { teamId: string @@ -49,7 +42,7 @@ export const linearApi = { | { ok: false; error: string } > => ipcRenderer.invoke('linear:createIssue', args), - getIssue: (args: { id: string; workspaceId?: string }): Promise => + getIssue: (args: { id: string; workspaceId?: string }) => ipcRenderer.invoke('linear:getIssue', args), updateIssue: (args: { @@ -66,18 +59,17 @@ export const linearApi = { }): Promise<{ ok: true; id: string } | { ok: false; error: string }> => ipcRenderer.invoke('linear:addIssueComment', args), - issueComments: (args: { issueId: string; workspaceId?: string }): Promise => + issueComments: (args: { issueId: string; workspaceId?: string }) => ipcRenderer.invoke('linear:issueComments', args), - listTeams: (args?: { workspaceId?: string }): Promise => - ipcRenderer.invoke('linear:listTeams', args), + listTeams: (args?: { workspaceId?: string }) => ipcRenderer.invoke('linear:listTeams', args), listProjects: (args?: { query?: string limit?: number workspaceId?: string force?: boolean - }): Promise => ipcRenderer.invoke('linear:listProjects', args), + }) => ipcRenderer.invoke('linear:listProjects', args), createProject: (args: { name: string @@ -94,7 +86,7 @@ export const linearApi = { }): Promise<{ ok: true; project: LinearProjectDetail } | { ok: false; error: string }> => ipcRenderer.invoke('linear:createProject', args), - getProject: (args: { id: string; workspaceId: string; force?: boolean }): Promise => + getProject: (args: { id: string; workspaceId: string; force?: boolean }) => ipcRenderer.invoke('linear:getProject', args), listProjectIssues: (args: { @@ -102,42 +94,38 @@ export const linearApi = { limit?: number workspaceId: string force?: boolean - }): Promise => ipcRenderer.invoke('linear:listProjectIssues', args), + }) => ipcRenderer.invoke('linear:listProjectIssues', args), listCustomViews: (args: { model: string limit?: number workspaceId?: string force?: boolean - }): Promise => ipcRenderer.invoke('linear:listCustomViews', args), + }) => ipcRenderer.invoke('linear:listCustomViews', args), - getCustomView: (args: { - viewId: string - model: string - workspaceId: string - force?: boolean - }): Promise => ipcRenderer.invoke('linear:getCustomView', args), + getCustomView: (args: { viewId: string; model: string; workspaceId: string; force?: boolean }) => + ipcRenderer.invoke('linear:getCustomView', args), listCustomViewIssues: (args: { viewId: string limit?: number workspaceId: string force?: boolean - }): Promise => ipcRenderer.invoke('linear:listCustomViewIssues', args), + }) => ipcRenderer.invoke('linear:listCustomViewIssues', args), listCustomViewProjects: (args: { viewId: string limit?: number workspaceId: string force?: boolean - }): Promise => ipcRenderer.invoke('linear:listCustomViewProjects', args), + }) => ipcRenderer.invoke('linear:listCustomViewProjects', args), - teamStates: (args: { teamId: string; workspaceId?: string }): Promise => + teamStates: (args: { teamId: string; workspaceId?: string }) => ipcRenderer.invoke('linear:teamStates', args), - teamLabels: (args: { teamId: string; workspaceId?: string }): Promise => + teamLabels: (args: { teamId: string; workspaceId?: string }) => ipcRenderer.invoke('linear:teamLabels', args), - teamMembers: (args: { teamId: string; workspaceId?: string }): Promise => + teamMembers: (args: { teamId: string; workspaceId?: string }) => ipcRenderer.invoke('linear:teamMembers', args) -} +} satisfies PreloadApi['linear'] diff --git a/src/preload/api/macos-tcc-prompts-bridge.ts b/src/preload/api/macos-tcc-prompts-bridge.ts index 0c6c6b184fc..ef24b9e203e 100644 --- a/src/preload/api/macos-tcc-prompts-bridge.ts +++ b/src/preload/api/macos-tcc-prompts-bridge.ts @@ -1,11 +1,14 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const macosTccPromptsApi = { - onThreshold: (callback: (payload: unknown) => void) => { - const listener = (_event: Electron.IpcRendererEvent, payload: unknown): void => + onThreshold: (callback: (payload: { promptCount: number }) => void) => { + const listener = (_event: Electron.IpcRendererEvent, payload: { promptCount: number }): void => callback(payload) ipcRenderer.on('macosTccPrompts:threshold', listener) - return () => ipcRenderer.removeListener('macosTccPrompts:threshold', listener) + return (): void => { + ipcRenderer.removeListener('macosTccPrompts:threshold', listener) + } }, consumePending: (): Promise<{ claimId: number; promptCount: number } | null> => ipcRenderer.invoke('macosTccPrompts:consumePending'), @@ -14,4 +17,4 @@ export const macosTccPromptsApi = { releasePending: (claimId: number): Promise => ipcRenderer.invoke('macosTccPrompts:releasePending', claimId), dismiss: (): Promise => ipcRenderer.invoke('macosTccPrompts:dismiss') -} +} satisfies PreloadApi['macosTccPrompts'] diff --git a/src/preload/api/memory-bridge.ts b/src/preload/api/memory-bridge.ts index 15af55cb09c..c1735f86e31 100644 --- a/src/preload/api/memory-bridge.ts +++ b/src/preload/api/memory-bridge.ts @@ -1,6 +1,7 @@ import { ipcRenderer } from 'electron' import type { MemorySnapshot } from '../../shared/process-stats-types' +import type { PreloadApi } from '../api-types' export const memoryApi = { getSnapshot: (): Promise => ipcRenderer.invoke('memory:getSnapshot') -} +} satisfies PreloadApi['memory'] diff --git a/src/preload/api/minimax-credentials-bridge.ts b/src/preload/api/minimax-credentials-bridge.ts index a758d9e2e5b..e99bd843909 100644 --- a/src/preload/api/minimax-credentials-bridge.ts +++ b/src/preload/api/minimax-credentials-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const minimaxCredentialsApi = { getStatus: (): Promise<{ configured: boolean }> => @@ -7,4 +8,4 @@ export const minimaxCredentialsApi = { ipcRenderer.invoke('minimaxCredentials:saveCookie', cookie), clearCookie: (): Promise<{ configured: boolean }> => ipcRenderer.invoke('minimaxCredentials:clearCookie') -} +} satisfies PreloadApi['minimaxCredentials'] diff --git a/src/preload/api/mobile-bridge.ts b/src/preload/api/mobile-bridge.ts index 836f27f6f37..a1ad9a4c716 100644 --- a/src/preload/api/mobile-bridge.ts +++ b/src/preload/api/mobile-bridge.ts @@ -3,6 +3,7 @@ import type { MobileRelayStatus } from '../../shared/mobile-relay-status' import type { MobilePairingConnectionMode } from '../../shared/mobile-pairing-connection-mode' import type { RuntimePairingReach } from '../../shared/runtime-pairing-reach' import type { MobileRelayMintFailure } from '../../shared/mobile-relay-mint-failure' +import type { PreloadApi } from '../api-types' export const mobileApi = { listNetworkInterfaces: (): Promise<{ @@ -92,4 +93,4 @@ export const mobileApi = { ipcRenderer.on('mobile:unpairedDeviceAuthFailure', listener) return () => ipcRenderer.removeListener('mobile:unpairedDeviceAuthFailure', listener) } -} +} satisfies PreloadApi['mobile'] diff --git a/src/preload/api/native-chat-bridge.ts b/src/preload/api/native-chat-bridge.ts index 16c2906349e..3a0a5d9161a 100644 --- a/src/preload/api/native-chat-bridge.ts +++ b/src/preload/api/native-chat-bridge.ts @@ -2,7 +2,8 @@ import { ipcRenderer } from 'electron' import type { NativeChatAppendedPayload, NativeChatReadSessionResult, - NativeChatSubscriptionFrame + NativeChatSubscriptionFrame, + PreloadApi } from '../api-types' import type { AgentType } from '../../shared/native-chat-types' @@ -37,4 +38,4 @@ export const nativeChatApi = { ipcRenderer.send('nativeChat:unsubscribe', { subscriptionId: args.subscriptionId }) } } -} +} satisfies PreloadApi['nativeChat'] diff --git a/src/preload/api/notebook-bridge.ts b/src/preload/api/notebook-bridge.ts index ee307b9f0e2..436726b7783 100644 --- a/src/preload/api/notebook-bridge.ts +++ b/src/preload/api/notebook-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const notebookApi = { runPythonCell: (args: { @@ -8,4 +9,4 @@ export const notebookApi = { connectionId?: string | null }): Promise<{ stdout: string; stderr: string; exitCode: number | null; error?: string }> => ipcRenderer.invoke('notebook:runPythonCell', args) -} +} satisfies PreloadApi['notebook'] diff --git a/src/preload/api/notifications-bridge.ts b/src/preload/api/notifications-bridge.ts index aa84fb1a8a6..70c64d4ce0d 100644 --- a/src/preload/api/notifications-bridge.ts +++ b/src/preload/api/notifications-bridge.ts @@ -8,6 +8,7 @@ import type { NotificationSoundPathResult, NotificationSoundResult } from '../../shared/notification-settings-types' +import type { PreloadApi } from '../api-types' // Why: cache one shared Audio + blob URL per sound path so notifications do not re-read large files. let cachedNotificationSound: { @@ -117,4 +118,4 @@ export const notificationsApi = { return { played: false, reason: 'playback-failed' } } } -} +} satisfies PreloadApi['notifications'] diff --git a/src/preload/api/onboarding-bridge.ts b/src/preload/api/onboarding-bridge.ts index 1b4393268ac..7939ab64810 100644 --- a/src/preload/api/onboarding-bridge.ts +++ b/src/preload/api/onboarding-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' import type { OnboardingState } from '../../shared/onboarding-state-types' +import type { PreloadApi } from '../api-types' export const onboardingApi = { get: (): Promise => ipcRenderer.invoke('onboarding:get'), @@ -8,4 +9,4 @@ export const onboardingApi = { checklist?: Partial } ): Promise => ipcRenderer.invoke('onboarding:update', updates) -} +} satisfies PreloadApi['onboarding'] diff --git a/src/preload/api/open-code-usage-bridge.ts b/src/preload/api/open-code-usage-bridge.ts index 5cc668e6e00..cd3564d2b32 100644 --- a/src/preload/api/open-code-usage-bridge.ts +++ b/src/preload/api/open-code-usage-bridge.ts @@ -1,4 +1,8 @@ import { ipcRenderer } from 'electron' import { createUsageProviderApi } from '../usage-provider-api' +import type { PreloadApi } from '../api-types' -export const openCodeUsageApi = createUsageProviderApi(ipcRenderer, 'openCodeUsage') +export const openCodeUsageApi = createUsageProviderApi( + ipcRenderer, + 'openCodeUsage' +) satisfies PreloadApi['openCodeUsage'] diff --git a/src/preload/api/pet-bridge.ts b/src/preload/api/pet-bridge.ts index 8a5d3309c5c..c51751b3161 100644 --- a/src/preload/api/pet-bridge.ts +++ b/src/preload/api/pet-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' import type { CustomPet } from '../../shared/pet-types' +import type { PreloadApi } from '../api-types' export const petApi = { import: (): Promise => ipcRenderer.invoke('pet:import'), @@ -8,4 +9,4 @@ export const petApi = { ipcRenderer.invoke('pet:read', id, fileName, kind), delete: (id: string, fileName: string, kind?: 'image' | 'bundle'): Promise => ipcRenderer.invoke('pet:delete', id, fileName, kind) -} +} satisfies PreloadApi['pet'] diff --git a/src/preload/api/plugins-bridge.ts b/src/preload/api/plugins-bridge.ts index 2488d4cfdb7..0e529e74d96 100644 --- a/src/preload/api/plugins-bridge.ts +++ b/src/preload/api/plugins-bridge.ts @@ -24,11 +24,8 @@ export const pluginsApi = { pluginKey: string panelId: string }): Promise => ipcRenderer.invoke('plugins:readPanelEntry', args), - invokeCommand: (args: { - pluginKey: string - commandId: string - args?: unknown - }): Promise => ipcRenderer.invoke('plugins:invokeCommand', args), + invokeCommand: (args: { pluginKey: string; commandId: string; args?: unknown }) => + ipcRenderer.invoke('plugins:invokeCommand', args), panelAction: (args: { sessionToken: string action: string diff --git a/src/preload/api/preflight-bridge.ts b/src/preload/api/preflight-bridge.ts index c05721a2d3b..64d317d139c 100644 --- a/src/preload/api/preflight-bridge.ts +++ b/src/preload/api/preflight-bridge.ts @@ -1,5 +1,5 @@ import { ipcRenderer } from 'electron' -import type { PreflightRuntimeContext, RefreshAgentsResult } from '../api-types' +import type { PreflightRuntimeContext, PreloadApi, RefreshAgentsResult } from '../api-types' export const preflightApi = { check: (args?: { @@ -40,4 +40,4 @@ export const preflightApi = { gitBashAvailable: boolean hostPlatform: NodeJS.Platform | null }> => ipcRenderer.invoke('preflight:detectRemoteWindowsTerminalCapabilities', args) -} +} satisfies PreloadApi['preflight'] diff --git a/src/preload/api/pty-bridge-session-control.ts b/src/preload/api/pty-bridge-session-control.ts index 7e761738256..ef6002e11c8 100644 --- a/src/preload/api/pty-bridge-session-control.ts +++ b/src/preload/api/pty-bridge-session-control.ts @@ -15,6 +15,7 @@ import type { import type { TerminalViewAttributes } from '../../shared/terminal-view-attributes' import type { PtyMainDeliveryDiagnostics } from '../../shared/pty-delivery-diagnostics' import type { AgentKind, LaunchSource, RequestKind } from '../../shared/telemetry-events' +import type { PreloadApi } from '../api-types' export const ptySessionControlApi = { spawn: (opts: { @@ -204,4 +205,4 @@ export const ptySessionControlApi = { ipcRenderer.invoke('pty:hasChildProcesses', { id }), getForegroundProcess: (id: string): Promise => ipcRenderer.invoke('pty:getForegroundProcess', { id }) -} +} satisfies Partial diff --git a/src/preload/api/pty-bridge-stream-and-serialization.ts b/src/preload/api/pty-bridge-stream-and-serialization.ts index 8fc9c48bce1..f751491c0e0 100644 --- a/src/preload/api/pty-bridge-stream-and-serialization.ts +++ b/src/preload/api/pty-bridge-stream-and-serialization.ts @@ -1,6 +1,7 @@ import { ipcRenderer } from 'electron' import type { PtyModelRestoreNeededEvent } from '../../shared/pty-model-restore-marker' import type { TerminalSideEffectBatch } from '../../shared/terminal-side-effect-facts' +import type { PreloadApi } from '../api-types' export const ptyStreamAndSerializationApi = { inspectProcess: ( @@ -138,4 +139,4 @@ export const ptyStreamAndSerializationApi = { restart: () => ipcRenderer.invoke('pty:management:restart'), macTccAttribution: () => ipcRenderer.invoke('pty:management:macTccAttribution') } -} +} satisfies Partial diff --git a/src/preload/api/pty-bridge.ts b/src/preload/api/pty-bridge.ts index df080692178..18f867e6426 100644 --- a/src/preload/api/pty-bridge.ts +++ b/src/preload/api/pty-bridge.ts @@ -1,4 +1,8 @@ +import type { PreloadApi } from '../api-types' import { ptySessionControlApi } from './pty-bridge-session-control' import { ptyStreamAndSerializationApi } from './pty-bridge-stream-and-serialization' -export const ptyApi = { ...ptySessionControlApi, ...ptyStreamAndSerializationApi } +export const ptyApi = { + ...ptySessionControlApi, + ...ptyStreamAndSerializationApi +} satisfies PreloadApi['pty'] diff --git a/src/preload/api/pwsh-bridge.ts b/src/preload/api/pwsh-bridge.ts index bd202277ad1..34ed9880ada 100644 --- a/src/preload/api/pwsh-bridge.ts +++ b/src/preload/api/pwsh-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const pwshApi = { isAvailable: (): Promise => ipcRenderer.invoke('pwsh:isAvailable') -} +} satisfies PreloadApi['pwsh'] diff --git a/src/preload/api/rate-limits-bridge.ts b/src/preload/api/rate-limits-bridge.ts index f403d79cdc6..37af13f0006 100644 --- a/src/preload/api/rate-limits-bridge.ts +++ b/src/preload/api/rate-limits-bridge.ts @@ -4,6 +4,7 @@ import type { RateLimitRuntimeTarget, RateLimitState } from '../../shared/rate-limit-types' +import type { PreloadApi } from '../api-types' export const rateLimitsApi = { get: (): Promise => ipcRenderer.invoke('rateLimits:get'), @@ -27,4 +28,4 @@ export const rateLimitsApi = { ipcRenderer.on('rateLimits:update', listener) return () => ipcRenderer.removeListener('rateLimits:update', listener) } -} +} satisfies PreloadApi['rateLimits'] diff --git a/src/preload/api/runtime-bridge.ts b/src/preload/api/runtime-bridge.ts index c58049bcbe3..8b31e6873f5 100644 --- a/src/preload/api/runtime-bridge.ts +++ b/src/preload/api/runtime-bridge.ts @@ -9,6 +9,7 @@ import type { } from '../../shared/runtime-types' import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' import type { RuntimeEnvironmentSubscriptionHandle } from '../runtime-environment-subscriptions' +import type { PreloadApi } from '../api-types' export const runtimeApi = { syncWindowGraph: (graph: RuntimeRendererSyncWindowGraph): Promise => @@ -141,4 +142,4 @@ export const runtimeApi = { ipcRenderer.on('runtime:clientHostedBrowserRowsChanged', listener) return () => ipcRenderer.removeListener('runtime:clientHostedBrowserRowsChanged', listener) } -} +} satisfies PreloadApi['runtime'] diff --git a/src/preload/api/runtime-environments-bridge.ts b/src/preload/api/runtime-environments-bridge.ts index d018c0574c8..ddfa498dc74 100644 --- a/src/preload/api/runtime-environments-bridge.ts +++ b/src/preload/api/runtime-environments-bridge.ts @@ -9,6 +9,7 @@ import { subscribeRuntimeEnvironmentFromPreload, type RuntimeEnvironmentSubscriptionHandle } from '../runtime-environment-subscriptions' +import type { PreloadApi } from '../api-types' export const runtimeEnvironmentsApi = { list: (): Promise => @@ -90,4 +91,4 @@ export const runtimeEnvironmentsApi = { } ): Promise => subscribeRuntimeEnvironmentFromPreload(ipcRenderer, args, callbacks) -} +} satisfies PreloadApi['runtimeEnvironments'] diff --git a/src/preload/api/settings-bridge.ts b/src/preload/api/settings-bridge.ts index d8aaa23b0b7..4e7105c00cb 100644 --- a/src/preload/api/settings-bridge.ts +++ b/src/preload/api/settings-bridge.ts @@ -4,22 +4,20 @@ import type { WarpThemeImportPreview, WarpThemeImportSource } from '../../shared/terminal-custom-themes' +import type { PreloadApi } from '../api-types' export const settingsApi = { - get: (): Promise => ipcRenderer.invoke('settings:get'), + get: () => ipcRenderer.invoke('settings:get'), // Why: blocking read for the few startup decisions (terminal side-effect authority) that can't wait for async hydration. Call sparingly. - getSync: (): unknown => ipcRenderer.sendSync('settings:get-sync'), + getSync: () => ipcRenderer.sendSync('settings:get-sync'), - set: (args: Record): Promise => - ipcRenderer.invoke('settings:set', args), + set: (args: Record) => ipcRenderer.invoke('settings:set', args), - setActiveRuntimeEnvironmentPreference: (args: { - environmentId: string | null - }): Promise => + setActiveRuntimeEnvironmentPreference: (args: { environmentId: string | null }) => ipcRenderer.invoke('settings:set-active-runtime-environment-preference', args), - updatePRBotAuthorOverride: (args: { author: string; isBot: boolean }): Promise => + updatePRBotAuthorOverride: (args: { author: string; isBot: boolean }) => ipcRenderer.invoke('settings:update-pr-bot-author-override', args), listFonts: (): Promise => ipcRenderer.invoke('settings:listFonts'), @@ -36,4 +34,4 @@ export const settingsApi = { ipcRenderer.on('settings:changed', listener) return () => ipcRenderer.removeListener('settings:changed', listener) } -} +} satisfies PreloadApi['settings'] diff --git a/src/preload/api/shell-bridge.ts b/src/preload/api/shell-bridge.ts index be34c7ff70e..21ccda3fd83 100644 --- a/src/preload/api/shell-bridge.ts +++ b/src/preload/api/shell-bridge.ts @@ -4,6 +4,7 @@ import type { ShellOpenExternalEditorResult, ShellOpenLocalPathResult } from '../../shared/shell-open-types' +import type { PreloadApi } from '../api-types' export const shellApi = { openPath: (path: string): Promise => ipcRenderer.invoke('shell:openPath', path), @@ -38,4 +39,4 @@ export const shellApi = { copyFile: (args: { srcPath: string; destPath: string }): Promise => ipcRenderer.invoke('shell:copyFile', args) -} +} satisfies PreloadApi['shell'] diff --git a/src/preload/api/skills-bridge.ts b/src/preload/api/skills-bridge.ts index eafaf2ce543..4bcde9a613c 100644 --- a/src/preload/api/skills-bridge.ts +++ b/src/preload/api/skills-bridge.ts @@ -37,6 +37,7 @@ import type { SkillUpdateRun, SkillUpdateStartResult } from '../../shared/skill-freshness' +import type { PreloadApi } from '../api-types' export const skillsApi = { discover: (target?: SkillDiscoveryTarget): Promise => @@ -126,4 +127,4 @@ export const skillsApi = { ipcRenderer.on('skills:updateRun', listener) return () => ipcRenderer.removeListener('skills:updateRun', listener) } -} +} satisfies PreloadApi['skills'] diff --git a/src/preload/api/speech-bridge.ts b/src/preload/api/speech-bridge.ts index 513b219f498..dbd7e0d26ab 100644 --- a/src/preload/api/speech-bridge.ts +++ b/src/preload/api/speech-bridge.ts @@ -6,6 +6,7 @@ import type { SpeechModelState, SpeechTranscriptEvent } from '../../shared/speech-types' +import type { PreloadApi } from '../api-types' export const speechApi = { getCatalog: (): Promise => ipcRenderer.invoke('speech:getCatalog'), @@ -78,4 +79,4 @@ export const speechApi = { ipcRenderer.on('speech:error', listener) return () => ipcRenderer.removeListener('speech:error', listener) } -} +} satisfies PreloadApi['speech'] diff --git a/src/preload/api/ssh-bridge.ts b/src/preload/api/ssh-bridge.ts index 2884e4ec8c1..b0f7b89ec3d 100644 --- a/src/preload/api/ssh-bridge.ts +++ b/src/preload/api/ssh-bridge.ts @@ -17,6 +17,7 @@ import { admitSshDetectedPorts } from '../../shared/ssh-retained-payload-admission' import type { FilesystemPathFlavor } from '../../shared/filesystem-entry-types' +import type { PreloadApi } from '../api-types' export const sshApi = { listTargets: (): Promise => ipcRenderer.invoke('ssh:listTargets'), @@ -180,4 +181,4 @@ export const sshApi = { submitCredential: (args: { requestId: string; value: string | null }): Promise => ipcRenderer.invoke('ssh:submitCredential', args) -} +} satisfies PreloadApi['ssh'] diff --git a/src/preload/api/star-nag-bridge.ts b/src/preload/api/star-nag-bridge.ts index b697739a52f..49e56e4aa91 100644 --- a/src/preload/api/star-nag-bridge.ts +++ b/src/preload/api/star-nag-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const starNagApi = { onShow: ( @@ -27,4 +28,4 @@ export const starNagApi = { ipcRenderer.invoke('star-nag:agentValueMoment'), showAgentValueMoment: (): Promise => ipcRenderer.invoke('star-nag:showAgentValueMoment'), onboardingCompleted: (): Promise => ipcRenderer.invoke('star-nag:onboardingCompleted') -} +} satisfies PreloadApi['starNag'] diff --git a/src/preload/api/stats-bridge.ts b/src/preload/api/stats-bridge.ts index 20bc823b86a..f435b9980f5 100644 --- a/src/preload/api/stats-bridge.ts +++ b/src/preload/api/stats-bridge.ts @@ -1,4 +1,5 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const statsApi = { getSummary: (): Promise<{ @@ -7,4 +8,4 @@ export const statsApi = { totalAgentTimeMs: number firstEventAt: number | null }> => ipcRenderer.invoke('stats:summary') -} +} satisfies PreloadApi['stats'] diff --git a/src/preload/api/terminal-preview-bridge.ts b/src/preload/api/terminal-preview-bridge.ts index c8bf5b38623..3bd94d4998b 100644 --- a/src/preload/api/terminal-preview-bridge.ts +++ b/src/preload/api/terminal-preview-bridge.ts @@ -3,6 +3,7 @@ import type { TerminalPreviewConnectResult, TerminalPreviewDataPayload } from '../../shared/terminal-preview' +import type { PreloadApi } from '../api-types' export const terminalPreviewApi = { connect: ( @@ -30,4 +31,4 @@ export const terminalPreviewApi = { ipcRenderer.on('terminalPreview:data', listener) return () => ipcRenderer.removeListener('terminalPreview:data', listener) } -} +} satisfies PreloadApi['terminalPreview'] diff --git a/src/preload/api/ui-bridge-clipboard-and-window-controls.ts b/src/preload/api/ui-bridge-clipboard-and-window-controls.ts index 867bd80026d..fdad19c2944 100644 --- a/src/preload/api/ui-bridge-clipboard-and-window-controls.ts +++ b/src/preload/api/ui-bridge-clipboard-and-window-controls.ts @@ -12,6 +12,7 @@ import { import type { NativeFileDropPayload } from '../../shared/native-file-drop' import type { ReadClipboardTextOptions } from '../../shared/clipboard-text' import { subscribeNativeFileDrop } from '../preload-runtime-support' +import type { PreloadApi } from '../api-types' export const uiClipboardAndWindowControlsApi = { onOpenDiffFromMobile: ( @@ -195,4 +196,4 @@ export const uiClipboardAndWindowControlsApi = { notifyWindowRevealed: (): void => { ipcRenderer.send('ui:window-revealed') } -} +} satisfies Partial diff --git a/src/preload/api/ui-bridge-state-and-menu-commands.ts b/src/preload/api/ui-bridge-state-and-menu-commands.ts index 34eb83886c8..246cebb1613 100644 --- a/src/preload/api/ui-bridge-state-and-menu-commands.ts +++ b/src/preload/api/ui-bridge-state-and-menu-commands.ts @@ -2,6 +2,7 @@ import type { MarkdownDocument } from '../../shared/filesystem-entry-types' import { ipcRenderer } from 'electron' import type { PersistedUIState } from '../../shared/persisted-ui-state-types' import type { KeybindingActionId } from '../../shared/keybindings' +import type { PreloadApi } from '../api-types' export const uiStateAndMenuCommandsApi = { get: () => ipcRenderer.invoke('ui:get'), @@ -173,4 +174,4 @@ export const uiStateAndMenuCommandsApi = { replyTabCreate: (reply: { requestId: string; browserPageId?: string; error?: string }): void => { ipcRenderer.send('browser:tabCreateReply', reply) } -} +} satisfies Partial diff --git a/src/preload/api/ui-bridge-tab-and-browser-commands.ts b/src/preload/api/ui-bridge-tab-and-browser-commands.ts index ccca9a24f5b..d275367b398 100644 --- a/src/preload/api/ui-bridge-tab-and-browser-commands.ts +++ b/src/preload/api/ui-bridge-tab-and-browser-commands.ts @@ -6,6 +6,7 @@ import type { WorktreeSetupLaunch } from '../../shared/worktree/launch-types' import { browserFindSubscriptions } from '../preload-runtime-support' +import type { PreloadApi } from '../api-types' export const uiTabAndBrowserCommandsApi = { onRequestTabSetProfile: ( @@ -200,4 +201,4 @@ export const uiTabAndBrowserCommandsApi = { ipcRenderer.on('ui:activateWorktree', listener) return () => ipcRenderer.removeListener('ui:activateWorktree', listener) } -} +} satisfies Partial diff --git a/src/preload/api/ui-bridge-terminal-and-session-tabs.ts b/src/preload/api/ui-bridge-terminal-and-session-tabs.ts index 9de47cceb0c..eaff9e8847a 100644 --- a/src/preload/api/ui-bridge-terminal-and-session-tabs.ts +++ b/src/preload/api/ui-bridge-terminal-and-session-tabs.ts @@ -11,6 +11,7 @@ import type { RuntimeTerminalCreateRequestPayload, RuntimeTerminalPresentation } from '../../shared/runtime-types' +import type { PreloadApi } from '../api-types' export const uiTerminalAndSessionTabsApi = { onCreateTerminal: ( @@ -211,4 +212,4 @@ export const uiTerminalAndSessionTabsApi = { ipcRenderer.on('ui:openFileFromMobile', listener) return () => ipcRenderer.removeListener('ui:openFileFromMobile', listener) } -} +} satisfies Partial diff --git a/src/preload/api/wsl-bridge.ts b/src/preload/api/wsl-bridge.ts index aeb32cdfa45..bc7869000e4 100644 --- a/src/preload/api/wsl-bridge.ts +++ b/src/preload/api/wsl-bridge.ts @@ -1,6 +1,7 @@ import { ipcRenderer } from 'electron' +import type { PreloadApi } from '../api-types' export const wslApi = { isAvailable: (): Promise => ipcRenderer.invoke('wsl:isAvailable'), listDistros: (): Promise => ipcRenderer.invoke('wsl:listDistros') -} +} satisfies PreloadApi['wsl'] diff --git a/src/preload/app-restart-checkpoint-routing.test.ts b/src/preload/app-restart-checkpoint-routing.test.ts index 19eb6948c9a..d794a86b9ab 100644 --- a/src/preload/app-restart-checkpoint-routing.test.ts +++ b/src/preload/app-restart-checkpoint-routing.test.ts @@ -92,6 +92,20 @@ describe('native preload destructive app actions', () => { }) } + it('exposes the durable checkpoint join the lazy-chunk recovery reload depends on', async () => { + const api = await loadApi() + invoke.mockResolvedValue({ ok: true }) + + await expect(api.app.awaitBeforeUnloadCheckpoint()).resolves.toBeUndefined() + expect(invoke).toHaveBeenCalledWith('app:await-before-unload-checkpoint') + + invoke.mockResolvedValue({ ok: false }) + + await expect(api.app.awaitBeforeUnloadCheckpoint()).rejects.toThrow( + 'Failed to persist renderer state before unload.' + ) + }) + it('preserves both macOS keyboard preload adapters', async () => { const api = await loadApi() invoke.mockResolvedValue(undefined) diff --git a/src/preload/gitlab.ts b/src/preload/gitlab.ts index d6f12367db0..154e139e4c4 100644 --- a/src/preload/gitlab.ts +++ b/src/preload/gitlab.ts @@ -12,23 +12,21 @@ type GitLabRepoSelectorArgs = { } export const glApi = { - viewer: (): Promise => ipcRenderer.invoke('gitlab:viewer'), - diagnoseAuth: (): Promise => ipcRenderer.invoke('gitlab:diagnoseAuth'), - rateLimit: (args?: { force?: boolean; host?: string | null }): Promise => + viewer: () => ipcRenderer.invoke('gitlab:viewer'), + diagnoseAuth: () => ipcRenderer.invoke('gitlab:diagnoseAuth'), + rateLimit: (args?: { force?: boolean; host?: string | null }) => ipcRenderer.invoke('gitlab:rateLimit', args), - projectSlug: (args: GitLabRepoSelectorArgs): Promise => - ipcRenderer.invoke('gitlab:projectSlug', args), + projectSlug: (args: GitLabRepoSelectorArgs) => ipcRenderer.invoke('gitlab:projectSlug', args), mrForBranch: ( args: GitLabRepoSelectorArgs & { branch: string linkedMRIid?: number | null } - ): Promise => ipcRenderer.invoke('gitlab:mrForBranch', args), + ) => ipcRenderer.invoke('gitlab:mrForBranch', args), - mr: (args: GitLabRepoSelectorArgs & { iid: number }): Promise => - ipcRenderer.invoke('gitlab:mr', args), + mr: (args: GitLabRepoSelectorArgs & { iid: number }) => ipcRenderer.invoke('gitlab:mr', args), listMRs: ( args: GitLabRepoSelectorArgs & { @@ -37,7 +35,7 @@ export const glApi = { perPage?: number query?: string } - ): Promise => ipcRenderer.invoke('gitlab:listMRs', args), + ) => ipcRenderer.invoke('gitlab:listMRs', args), listWorkItems: ( args: GitLabRepoSelectorArgs & { @@ -46,9 +44,9 @@ export const glApi = { perPage?: number query?: string } - ): Promise => ipcRenderer.invoke('gitlab:listWorkItems', args), + ) => ipcRenderer.invoke('gitlab:listWorkItems', args), - issue: (args: GitLabRepoSelectorArgs & { number: number }): Promise => + issue: (args: GitLabRepoSelectorArgs & { number: number }) => ipcRenderer.invoke('gitlab:issue', args), listIssues: ( @@ -58,8 +56,7 @@ export const glApi = { limit?: number page?: number } - ): Promise<{ items: unknown[]; totalPages?: number; error?: unknown }> => - ipcRenderer.invoke('gitlab:listIssues', args), + ) => ipcRenderer.invoke('gitlab:listIssues', args), createIssue: ( args: GitLabRepoSelectorArgs & { @@ -77,25 +74,23 @@ export const glApi = { ): Promise<{ ok: true } | { ok: false; error: string }> => ipcRenderer.invoke('gitlab:updateIssue', args), - addIssueComment: ( - args: GitLabRepoSelectorArgs & { number: number; body: string } - ): Promise => ipcRenderer.invoke('gitlab:addIssueComment', args), + addIssueComment: (args: GitLabRepoSelectorArgs & { number: number; body: string }) => + ipcRenderer.invoke('gitlab:addIssueComment', args), listLabels: (args: GitLabRepoSelectorArgs): Promise => ipcRenderer.invoke('gitlab:listLabels', args), - listAssignableUsers: (args: GitLabRepoSelectorArgs): Promise => + listAssignableUsers: (args: GitLabRepoSelectorArgs) => ipcRenderer.invoke('gitlab:listAssignableUsers', args), - todos: (args: GitLabRepoSelectorArgs): Promise => - ipcRenderer.invoke('gitlab:todos', args), + todos: (args: GitLabRepoSelectorArgs) => ipcRenderer.invoke('gitlab:todos', args), workItemDetails: ( args: GitLabRepoSelectorArgs & { iid: number type: 'issue' | 'mr' } - ): Promise => ipcRenderer.invoke('gitlab:workItemDetails', args), + ) => ipcRenderer.invoke('gitlab:workItemDetails', args), closeMR: ( args: GitLabRepoSelectorArgs & { @@ -133,9 +128,9 @@ export const glApi = { reviewerIds: number[] projectRef?: unknown } - ): Promise => ipcRenderer.invoke('gitlab:updateMRReviewers', args), + ) => ipcRenderer.invoke('gitlab:updateMRReviewers', args), - addMRComment: (args: GitLabRepoSelectorArgs & { iid: number; body: string }): Promise => + addMRComment: (args: GitLabRepoSelectorArgs & { iid: number; body: string }) => ipcRenderer.invoke('gitlab:addMRComment', args), addMRInlineComment: ( @@ -144,7 +139,7 @@ export const glApi = { input: unknown projectRef?: unknown } - ): Promise => ipcRenderer.invoke('gitlab:addMRInlineComment', args), + ) => ipcRenderer.invoke('gitlab:addMRInlineComment', args), resolveMRDiscussion: ( args: GitLabRepoSelectorArgs & { @@ -152,15 +147,14 @@ export const glApi = { discussionId: string resolved: boolean } - ): Promise => ipcRenderer.invoke('gitlab:resolveMRDiscussion', args), + ) => ipcRenderer.invoke('gitlab:resolveMRDiscussion', args), jobTrace: ( args: GitLabRepoSelectorArgs & { jobId: number; projectRef?: unknown; logExcerpt?: boolean } - ): Promise => ipcRenderer.invoke('gitlab:jobTrace', args), + ) => ipcRenderer.invoke('gitlab:jobTrace', args), - retryJob: ( - args: GitLabRepoSelectorArgs & { jobId: number; projectRef?: unknown } - ): Promise => ipcRenderer.invoke('gitlab:retryJob', args), + retryJob: (args: GitLabRepoSelectorArgs & { jobId: number; projectRef?: unknown }) => + ipcRenderer.invoke('gitlab:retryJob', args), workItemByPath: ( args: GitLabRepoSelectorArgs & { @@ -169,5 +163,5 @@ export const glApi = { iid: number type: 'issue' | 'mr' } - ): Promise => ipcRenderer.invoke('gitlab:workItemByPath', args) + ) => ipcRenderer.invoke('gitlab:workItemByPath', args) } diff --git a/src/preload/index.ts b/src/preload/index.ts index ad97a911fe5..27d3ca8e062 100644 --- a/src/preload/index.ts +++ b/src/preload/index.ts @@ -182,7 +182,7 @@ const api = { mobile: mobileApi, agentStatus: agentStatusApi, speech: speechApi -} +} satisfies PreloadApi if (process.contextIsolated) { try { @@ -193,6 +193,5 @@ if (process.contextIsolated) { } } else { window.electron = electronAPI - // @ts-expect-error (define in dts) window.api = api } From fd33f9b0f93ff03055a05c977496e93a1e2f6573 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Tue, 1 Sep 2026 21:52:36 -0400 Subject: [PATCH 003/398] fix(review-notes): classify send failures and mirrored tabs (#18023) * fix(review-notes): classify send failures * fix(review-notes): honor structured runtime error codes * test(review-notes): use full runtime error envelope * chore: remove unrelated merge formatting * refactor(review-notes): share runtime failure codes * fix(review-notes): classify structured runtime timeouts --- .../editor/ReviewNotesSendMenuContent.tsx | 15 +- .../lib/active-agent-note-send-delivery.ts | 152 +++++++++++ .../lib/active-agent-note-send-diagnostics.ts | 82 ++++++ ...ve-agent-note-send-explicit-target.test.ts | 147 ++++++++++- ...ve-agent-note-send-focused-session.test.ts | 14 +- .../src/lib/active-agent-note-send-result.ts | 57 +++- ...gent-note-send-runtime-error-codes.test.ts | 56 ++++ .../src/lib/active-agent-note-send.ts | 245 ++++++------------ .../src/lib/active-agent-note-target.ts | 5 +- .../active-agent-terminal-send-readiness.ts | 49 ++-- .../store/slices/ui/ui-slice-agent-actions.ts | 13 +- 11 files changed, 617 insertions(+), 218 deletions(-) create mode 100644 src/renderer/src/lib/active-agent-note-send-delivery.ts create mode 100644 src/renderer/src/lib/active-agent-note-send-diagnostics.ts create mode 100644 src/renderer/src/lib/active-agent-note-send-runtime-error-codes.test.ts diff --git a/src/renderer/src/components/editor/ReviewNotesSendMenuContent.tsx b/src/renderer/src/components/editor/ReviewNotesSendMenuContent.tsx index 5fbb92c7636..e0b1d70d5a2 100644 --- a/src/renderer/src/components/editor/ReviewNotesSendMenuContent.tsx +++ b/src/renderer/src/components/editor/ReviewNotesSendMenuContent.tsx @@ -124,17 +124,18 @@ export function ReviewNotesSendMenuContent({ toast.message( activeAgentNotesSendFailureMessage(result.status, { - explicitTarget: options.explicitTarget + explicitTarget: options.explicitTarget, + code: result.code }) ) }) - .catch((error) => { - console.error('Failed to send notes:', error) + .catch(() => { + console.error('Failed to send notes:', { code: 'runtime-unverifiable' }) toast.error( - translate( - 'auto.components.editor.ReviewNotesSendMenuContent.f5096c6e4e', - 'Could not send notes.' - ) + activeAgentNotesSendFailureMessage('status-unavailable', { + explicitTarget: options.explicitTarget, + code: 'runtime-unverifiable' + }) ) }) .finally(() => { diff --git a/src/renderer/src/lib/active-agent-note-send-delivery.ts b/src/renderer/src/lib/active-agent-note-send-delivery.ts new file mode 100644 index 00000000000..2d9e88ae507 --- /dev/null +++ b/src/renderer/src/lib/active-agent-note-send-delivery.ts @@ -0,0 +1,152 @@ +import type { RuntimeTerminalSend } from '../../../shared/runtime-types' +import { sanitizeTerminalPasteText } from '@/components/terminal-pane/terminal-bracketed-paste' +import { callRuntimeRpc } from '@/runtime/runtime-rpc-client' +import { + BRACKETED_PASTE_BEGIN, + BRACKETED_PASTE_END, + POST_PASTE_SUBMIT_DELAY_MS +} from './agent-paste-draft' +import type { ActiveAgentNotesSendResult } from './active-agent-note-send-result' +import { + ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS, + getTerminalAgentSendReadiness, + isRuntimeTerminalNotWritable, + isRuntimeTerminalUnavailable +} from './active-agent-terminal-send-readiness' +import { codeForReadinessStatus, runtimeFailureCode } from './active-agent-note-send-diagnostics' + +const ORCA_DESKTOP_TERMINAL_CLIENT = { id: 'orca-desktop', type: 'desktop' as const } + +export async function sendPromptWithLegacyCombinedSend( + runtimeTarget: Parameters[0], + terminalHandle: string, + prompt: string +): Promise { + try { + const { send } = await callRuntimeRpc<{ send: RuntimeTerminalSend }>( + runtimeTarget, + 'terminal.send', + { terminal: terminalHandle, text: prompt, enter: true, client: ORCA_DESKTOP_TERMINAL_CLIENT }, + { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } + ) + return send.accepted + ? { status: 'sent' } + : { status: 'not-writable', code: 'terminal-send-refused' } + } catch (error) { + if (isRuntimeTerminalUnavailable(error)) { + return { + status: 'no-active-terminal', + code: runtimeFailureCode(error) ?? 'runtime-unverifiable' + } + } + if (isRuntimeTerminalNotWritable(error)) { + return { status: 'not-writable', code: 'terminal_not_writable' } + } + throw error + } +} + +export async function sendPromptWithGuardedPasteAndEnter( + runtimeTarget: Parameters[0], + terminalHandle: string, + prompt: string, + options: { allowLegacyFallback: boolean } +): Promise { + const initialAgentStatus = await getTerminalAgentSendReadiness( + runtimeTarget, + terminalHandle, + options + ) + if ( + initialAgentStatus.status !== 'sendable' && + !(initialAgentStatus.status === 'no-agent' && initialAgentStatus.supportsGuardedSend) + ) { + return { + status: initialAgentStatus.status, + code: initialAgentStatus.code ?? codeForReadinessStatus(initialAgentStatus.status) + } + } + + const pastePayload = `${BRACKETED_PASTE_BEGIN}${sanitizeTerminalPasteText(prompt)}${BRACKETED_PASTE_END}` + try { + const { send } = await callRuntimeRpc<{ send: RuntimeTerminalSend }>( + runtimeTarget, + 'terminal.send', + { + terminal: terminalHandle, + text: pastePayload, + requireAgentStatus: 'sendable', + client: ORCA_DESKTOP_TERMINAL_CLIENT + }, + { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } + ) + if (!send.accepted) { + if (send.refusedReason === 'permission') { + return { status: 'permission', code: 'terminal-send-permission' } + } + if (send.refusedReason === 'no-agent') { + return { status: 'no-agent', code: 'no-agent' } + } + return { status: 'not-writable', code: 'terminal-send-refused' } + } + } catch (error) { + if (isRuntimeTerminalUnavailable(error)) { + return { + status: 'no-active-terminal', + code: runtimeFailureCode(error) ?? 'runtime-unverifiable' + } + } + if (isRuntimeTerminalNotWritable(error)) { + return { status: 'not-writable', code: 'terminal_not_writable' } + } + throw error + } + + await new Promise((resolve) => setTimeout(resolve, POST_PASTE_SUBMIT_DELAY_MS)) + try { + const submitAgentStatus = await getTerminalAgentSendReadiness( + runtimeTarget, + terminalHandle, + options + ) + if ( + submitAgentStatus.status !== 'sendable' && + !(submitAgentStatus.status === 'no-agent' && submitAgentStatus.supportsGuardedSend) + ) { + return { + status: 'partial-submit-failed', + code: submitAgentStatus.code ?? 'submit-readiness-lost' + } + } + } catch (error) { + if (isRuntimeTerminalUnavailable(error)) { + return { + status: 'partial-submit-failed', + code: runtimeFailureCode(error) ?? 'submit-terminal-unavailable' + } + } + throw error + } + + try { + const { send } = await callRuntimeRpc<{ send: RuntimeTerminalSend }>( + runtimeTarget, + 'terminal.send', + { + terminal: terminalHandle, + enter: true, + requireAgentStatus: 'sendable', + client: ORCA_DESKTOP_TERMINAL_CLIENT + }, + { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } + ) + return send.accepted + ? { status: 'sent' } + : { status: 'partial-submit-failed', code: 'submit-send-refused' } + } catch (error) { + if (isRuntimeTerminalUnavailable(error) || isRuntimeTerminalNotWritable(error)) { + return { status: 'partial-submit-failed', code: 'submit-send-error' } + } + throw error + } +} diff --git a/src/renderer/src/lib/active-agent-note-send-diagnostics.ts b/src/renderer/src/lib/active-agent-note-send-diagnostics.ts new file mode 100644 index 00000000000..155dec17e06 --- /dev/null +++ b/src/renderer/src/lib/active-agent-note-send-diagnostics.ts @@ -0,0 +1,82 @@ +import type { ActiveTerminalNoteTarget } from './active-agent-note-target' +import type { + ActiveAgentNotesSendFailureCode, + ActiveAgentNotesSendResult +} from './active-agent-note-send-result' +import { hasRuntimeRpcErrorCode } from '../../../shared/runtime-rpc-error-code' + +export const TERMINAL_RUNTIME_FAILURE_CODES = [ + 'terminal_handle_stale', + 'terminal_exited', + 'terminal_gone', + 'no_active_terminal' +] as const + +export function reportNoteSendFailure( + result: ActiveAgentNotesSendResult, + noteTarget: ActiveTerminalNoteTarget | null +): ActiveAgentNotesSendResult { + if (result.status === 'sent' || result.status === 'empty') { + return result + } + const code = result.code ?? codeForStatus(result.status) + console.warn('[review-notes] send failed', { + code, + status: result.status, + tabId: noteTarget?.tabId, + leafId: noteTarget?.leafId + }) + return { ...result, code } +} + +export function codeForReadinessStatus( + status: 'no-active-terminal' | 'no-agent' | 'permission' | 'status-unavailable' +): ActiveAgentNotesSendFailureCode { + switch (status) { + case 'no-active-terminal': + return 'no-inventory-match' + case 'no-agent': + return 'no-agent' + case 'permission': + return 'agent-permission' + case 'status-unavailable': + return 'status-unavailable' + } +} + +export function runtimeFailureCode(error: unknown): ActiveAgentNotesSendFailureCode | null { + return TERMINAL_RUNTIME_FAILURE_CODES.find((code) => hasRuntimeRpcErrorCode(error, code)) ?? null +} + +export function runtimeFailureFallbackCode(error: unknown): ActiveAgentNotesSendFailureCode { + return isTimeoutError(error) ? 'runtime-timeout' : 'runtime-unverifiable' +} + +function isTimeoutError(error: unknown): boolean { + if (hasRuntimeRpcErrorCode(error, 'runtime_timeout')) { + return true + } + const message = error instanceof Error ? error.message : String(error) + return message.includes('timeout') +} + +function codeForStatus( + status: Exclude +): ActiveAgentNotesSendFailureCode { + switch (status) { + case 'no-active-terminal': + return 'no-inventory-match' + case 'no-agent': + return 'no-agent' + case 'permission': + return 'agent-permission' + case 'status-unavailable': + return 'status-unavailable' + case 'not-ready': + return 'terminal_wait_timeout' + case 'not-writable': + return 'terminal-send-refused' + case 'partial-submit-failed': + return 'submit-send-error' + } +} diff --git a/src/renderer/src/lib/active-agent-note-send-explicit-target.test.ts b/src/renderer/src/lib/active-agent-note-send-explicit-target.test.ts index 96eb0306a40..098dd649f29 100644 --- a/src/renderer/src/lib/active-agent-note-send-explicit-target.test.ts +++ b/src/renderer/src/lib/active-agent-note-send-explicit-target.test.ts @@ -172,7 +172,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'status-unavailable' }) + ).resolves.toEqual({ status: 'status-unavailable', code: 'status-unavailable' }) expect(methods).toEqual(['terminal.list', 'terminal.agentStatus']) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( @@ -275,7 +275,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'permission' }) + ).resolves.toEqual({ status: 'permission', code: 'agent-permission' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( expect.anything(), @@ -359,7 +359,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'no-agent' }) + ).resolves.toEqual({ status: 'no-agent', code: 'no-agent' }) expect(methods).toEqual(['terminal.list', 'terminal.agentStatus', 'terminal.send']) }) @@ -401,7 +401,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'not-writable' }) + ).resolves.toEqual({ status: 'not-writable', code: 'terminal-send-refused' }) const sendCalls = testState.callRuntimeRpc.mock.calls.filter( (call) => call[1] === 'terminal.send' @@ -454,7 +454,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'permission' }) + ).resolves.toEqual({ status: 'permission', code: 'terminal-send-permission' }) const sendCalls = testState.callRuntimeRpc.mock.calls.filter( (call) => call[1] === 'terminal.send' @@ -508,7 +508,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'partial-submit-failed' }) + ).resolves.toEqual({ status: 'partial-submit-failed', code: 'submit-readiness-lost' }) const sendCalls = testState.callRuntimeRpc.mock.calls.filter( (call) => call[1] === 'terminal.send' @@ -562,7 +562,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'partial-submit-failed' }) + ).resolves.toEqual({ status: 'partial-submit-failed', code: 'submit-send-refused' }) }) it('maps explicit target guarded Enter permission refusal to partial-submit-failed', async () => { @@ -620,7 +620,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'partial-submit-failed' }) + ).resolves.toEqual({ status: 'partial-submit-failed', code: 'submit-send-refused' }) }) it('uses selected-target failure wording for explicit note targets', () => { @@ -647,8 +647,90 @@ describe('active agent note send', () => { expect(activeAgentNotesSendFailureMessage('partial-submit-failed')).toBe( 'The notes may already be pasted in the active terminal, but Orca could not submit them.' ) + expect( + activeAgentNotesSendFailureMessage('no-active-terminal', { + explicitTarget: true, + code: 'no-inventory-match' + }) + ).toBe('The selected terminal is no longer available. (no-inventory-match)') }) + it.each(['terminal_handle_stale', 'terminal_exited', 'terminal_gone'] as const)( + 'returns and logs the runtime terminal failure code %s without note contents', + async (runtimeCode) => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + testState.callRuntimeRpc.mockImplementation(async (_target, method) => { + if (method === 'terminal.list') { + return { + terminals: [ + { + handle: 'term-runtime-failure', + worktreeId: 'wt-1', + worktreePath: '/repo', + branch: 'main', + tabId: 'tab-9', + leafId: OTHER_LEAF_ID, + title: 'Codex', + connected: true, + writable: true, + lastOutputAt: 1, + preview: '' + } + ], + totalCount: 1, + truncated: false + } + } + if (method === 'terminal.agentStatus') { + throw new Error(runtimeCode) + } + throw new Error(`unexpected method ${method}`) + }) + + await expect( + sendNotesToActiveAgentSession({ + worktreeId: 'wt-1', + prompt: 'private note contents', + noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } + }) + ).resolves.toEqual({ status: 'no-active-terminal', code: runtimeCode }) + + expect(warn).toHaveBeenCalledWith('[review-notes] send failed', { + code: runtimeCode, + status: 'no-active-terminal', + tabId: 'tab-9', + leafId: OTHER_LEAF_ID + }) + expect(JSON.stringify(warn.mock.calls)).not.toContain('private note contents') + warn.mockRestore() + } + ) + + it.each([ + ['remote connection closed at /private/workspace', 'runtime-unverifiable'], + ['runtime request timeout', 'runtime-timeout'] + ] as const)( + 'classifies terminal inventory failure without logging raw error details: %s', + async (message, code) => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + testState.callRuntimeRpc.mockRejectedValue(new Error(message)) + + await expect( + sendNotesToActiveAgentSession({ + worktreeId: 'wt-1', + prompt: 'private note contents', + noteTarget: { tabId: 'tab-9', leafId: OTHER_LEAF_ID } + }) + ).resolves.toEqual({ status: 'status-unavailable', code }) + + const logged = JSON.stringify(warn.mock.calls) + expect(logged).toContain(code) + expect(logged).not.toContain(message) + expect(logged).not.toContain('private note contents') + warn.mockRestore() + } + ) + it('returns no-active-terminal when the explicit note target is absent from the runtime list', async () => { testState.callRuntimeRpc.mockImplementation(async (_target, method) => { if (method === 'terminal.list') { @@ -681,7 +763,7 @@ describe('active agent note send', () => { prompt: 'notes', noteTarget: { tabId: 'tab-1', leafId: OTHER_LEAF_ID } }) - ).resolves.toEqual({ status: 'no-active-terminal' }) + ).resolves.toEqual({ status: 'no-active-terminal', code: 'no-inventory-match' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( expect.anything(), @@ -690,4 +772,51 @@ describe('active agent note send', () => { expect.anything() ) }) + + it('matches mirrored renderer tab IDs to host runtime tab IDs', async () => { + testState.callRuntimeRpc.mockImplementation(async (_target, method, params) => { + if (method === 'terminal.list') { + return { + terminals: [ + { + handle: 'term-mirrored', + worktreeId: 'wt-1', + worktreePath: '/repo', + branch: 'main', + tabId: 'tab-9', + leafId: OTHER_LEAF_ID, + title: 'Codex', + connected: true, + writable: true, + lastOutputAt: 1, + preview: '' + } + ], + totalCount: 1, + truncated: false + } + } + if (method === 'terminal.agentStatus') { + return { agentStatus: { handle: 'term-mirrored', isRunningAgent: true, status: 'working' } } + } + if (method === 'terminal.send') { + return { + send: { + handle: 'term-mirrored', + accepted: true, + bytesWritten: typeof params.text === 'string' ? params.text.length : 1 + } + } + } + throw new Error(`unexpected method ${method}`) + }) + + await expect( + sendNotesToActiveAgentSession({ + worktreeId: 'wt-1', + prompt: 'notes', + noteTarget: { tabId: 'web-terminal-tab-9', leafId: OTHER_LEAF_ID } + }) + ).resolves.toEqual({ status: 'sent' }) + }) }) diff --git a/src/renderer/src/lib/active-agent-note-send-focused-session.test.ts b/src/renderer/src/lib/active-agent-note-send-focused-session.test.ts index 76fcf49c4e6..118bd134df5 100644 --- a/src/renderer/src/lib/active-agent-note-send-focused-session.test.ts +++ b/src/renderer/src/lib/active-agent-note-send-focused-session.test.ts @@ -187,7 +187,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'permission' }) + ).resolves.toEqual({ status: 'permission', code: 'terminal-send-permission' }) }) it('keeps active-focused sends compatible when an older runtime lacks agentStatus', async () => { @@ -286,7 +286,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'no-agent' }) + ).resolves.toEqual({ status: 'no-agent', code: 'no-agent' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( expect.anything(), @@ -330,7 +330,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'not-ready' }) + ).resolves.toEqual({ status: 'not-ready', code: 'terminal_wait_timeout' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( expect.anything(), @@ -382,7 +382,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'no-active-terminal' }) + ).resolves.toEqual({ status: 'no-active-terminal', code: 'terminal_wait_not_running' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( expect.anything(), @@ -435,7 +435,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'permission' }) + ).resolves.toEqual({ status: 'permission', code: 'terminal_wait_blocked' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( expect.anything(), @@ -495,7 +495,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'permission' }) + ).resolves.toEqual({ status: 'permission', code: 'agent-permission' }) expect(statusChecks).toBe(2) expect(testState.callRuntimeRpc).not.toHaveBeenCalledWith( @@ -512,7 +512,7 @@ describe('active agent note send', () => { await expect( sendNotesToActiveAgentSession({ worktreeId: 'wt-1', prompt: 'notes' }) - ).resolves.toEqual({ status: 'no-active-terminal' }) + ).resolves.toEqual({ status: 'no-active-terminal', code: 'no-note-target' }) expect(testState.callRuntimeRpc).not.toHaveBeenCalled() }) diff --git a/src/renderer/src/lib/active-agent-note-send-result.ts b/src/renderer/src/lib/active-agent-note-send-result.ts index 3c292d497e4..388037d7226 100644 --- a/src/renderer/src/lib/active-agent-note-send-result.ts +++ b/src/renderer/src/lib/active-agent-note-send-result.ts @@ -9,39 +9,76 @@ export type ActiveAgentNotesSendStatus = | 'not-writable' | 'partial-submit-failed' +export type ActiveAgentNotesSendFailureCode = + | 'empty' + | 'no-note-target' + | 'no-inventory-match' + | 'terminal_handle_stale' + | 'terminal_exited' + | 'terminal_gone' + | 'no_active_terminal' + | 'terminal_wait_not_running' + | 'terminal_wait_blocked' + | 'terminal_wait_unsatisfied' + | 'terminal_wait_timeout' + | 'no-agent' + | 'agent-permission' + | 'status-unavailable' + | 'terminal-send-permission' + | 'terminal-send-refused' + | 'terminal_not_writable' + | 'submit-readiness-lost' + | 'submit-terminal-unavailable' + | 'submit-send-refused' + | 'submit-send-error' + | 'runtime-unverifiable' + | 'runtime-timeout' + export type ActiveAgentNotesSendResult = { status: ActiveAgentNotesSendStatus + code?: ActiveAgentNotesSendFailureCode } export function activeAgentNotesSendFailureMessage( status: ActiveAgentNotesSendStatus, - options: { explicitTarget?: boolean } = {} + options: { explicitTarget?: boolean; code?: ActiveAgentNotesSendFailureCode } = {} ): string { const target = options.explicitTarget ? 'selected' : 'active' + let message: string switch (status) { case 'empty': - return 'No notes to send.' + message = 'No notes to send.' + break case 'no-active-terminal': - return options.explicitTarget + message = options.explicitTarget ? 'The selected terminal is no longer available.' : 'Open the agent terminal in this worktree, then send the notes again.' + break case 'no-agent': - return `The ${target} terminal is not a recognized agent session.` + message = `The ${target} terminal is not a recognized agent session.` + break case 'permission': - return options.explicitTarget + message = options.explicitTarget ? 'The selected agent needs permission.' : 'The active agent needs permission.' + break case 'status-unavailable': - return `The ${target} agent status could not be verified.` + message = `The ${target} agent status could not be verified.` + break case 'not-ready': - return `The ${target} agent was not ready for input yet.` + message = `The ${target} agent was not ready for input yet.` + break case 'not-writable': - return `The ${target} terminal did not accept the notes.` + message = `The ${target} terminal did not accept the notes.` + break case 'partial-submit-failed': - return options.explicitTarget + message = options.explicitTarget ? 'The notes may already be pasted in the selected terminal, but Orca could not submit them.' : 'The notes may already be pasted in the active terminal, but Orca could not submit them.' + break case 'sent': - return '' + message = '' + break } + return options.code ? `${message} (${options.code})` : message } diff --git a/src/renderer/src/lib/active-agent-note-send-runtime-error-codes.test.ts b/src/renderer/src/lib/active-agent-note-send-runtime-error-codes.test.ts new file mode 100644 index 00000000000..baa41a7d139 --- /dev/null +++ b/src/renderer/src/lib/active-agent-note-send-runtime-error-codes.test.ts @@ -0,0 +1,56 @@ +import { describe, expect, it } from 'vitest' +import { hasRuntimeRpcErrorCode, RuntimeRpcCallError } from '@/runtime/runtime-rpc-client' +import { + isRuntimeTerminalNotWritable, + isRuntimeTerminalUnavailable, + isRuntimeTimeout +} from './active-agent-terminal-send-readiness' +import { + runtimeFailureCode, + runtimeFailureFallbackCode +} from './active-agent-note-send-diagnostics' + +function runtimeError(code: string): RuntimeRpcCallError { + return new RuntimeRpcCallError({ + id: 'test-runtime-error', + ok: false, + error: { code, message: 'The terminal is no longer available' } + }) +} + +describe('active agent note runtime error codes', () => { + it.each(['terminal_handle_stale', 'terminal_exited', 'terminal_gone', 'no_active_terminal'])( + 'uses structured %s codes even with human-readable messages', + (code) => { + const error = runtimeError(code) + + expect(isRuntimeTerminalUnavailable(error)).toBe(true) + expect(runtimeFailureCode(error)).toBe(code) + } + ) + + it('uses structured terminal_not_writable with a human-readable message', () => { + const error = runtimeError('terminal_not_writable') + + expect(isRuntimeTerminalNotWritable(error)).toBe(true) + expect(hasRuntimeRpcErrorCode(error, 'terminal_not_writable')).toBe(true) + }) + + it('uses structured runtime_timeout with a human-readable message', () => { + const error = new RuntimeRpcCallError({ + id: 'test-runtime-timeout', + ok: false, + error: { code: 'runtime_timeout', message: 'Timed out waiting for the remote runtime.' } + }) + + expect(isRuntimeTimeout(error)).toBe(true) + expect(runtimeFailureFallbackCode(error)).toBe('runtime-timeout') + }) + + it('retains support for transport-rewrapped error tokens', () => { + const error = new Error("Error invoking remote method 'terminal.send': terminal_gone") + + expect(isRuntimeTerminalUnavailable(error)).toBe(true) + expect(runtimeFailureCode(error)).toBe('terminal_gone') + }) +}) diff --git a/src/renderer/src/lib/active-agent-note-send.ts b/src/renderer/src/lib/active-agent-note-send.ts index 97bb2418132..48cced68feb 100644 --- a/src/renderer/src/lib/active-agent-note-send.ts +++ b/src/renderer/src/lib/active-agent-note-send.ts @@ -1,26 +1,26 @@ -import type { RuntimeTerminalSend, RuntimeTerminalWait } from '../../../shared/runtime-types' -import { sanitizeTerminalPasteText } from '@/components/terminal-pane/terminal-bracketed-paste' +import type { RuntimeTerminalWait } from '../../../shared/runtime-types' import { useAppStore } from '@/store' import { callRuntimeRpc, getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' import { getSettingsForWorktreeRuntimeOwner } from '@/lib/worktree-runtime-owner' -import { - findActiveRuntimeTerminal, - getActiveTerminalNoteTarget, - type ActiveTerminalNoteTarget -} from './active-agent-note-target' -import { - BRACKETED_PASTE_BEGIN, - BRACKETED_PASTE_END, - POST_PASTE_SUBMIT_DELAY_MS -} from './agent-paste-draft' +import { findActiveRuntimeTerminal, getActiveTerminalNoteTarget } from './active-agent-note-target' +import type { ActiveTerminalNoteTarget } from './active-agent-note-target' import type { ActiveAgentNotesSendResult } from './active-agent-note-send-result' import { ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS, getTerminalAgentSendReadiness, - isRuntimeTerminalNotWritable, isRuntimeTerminalUnavailable, isRuntimeTimeout } from './active-agent-terminal-send-readiness' +import { + codeForReadinessStatus, + reportNoteSendFailure, + runtimeFailureCode, + runtimeFailureFallbackCode +} from './active-agent-note-send-diagnostics' +import { + sendPromptWithGuardedPasteAndEnter, + sendPromptWithLegacyCombinedSend +} from './active-agent-note-send-delivery' export { getActiveAgentNoteTarget, @@ -34,11 +34,24 @@ export { type ActiveAgentNotesSendResult, type ActiveAgentNotesSendStatus } from './active-agent-note-send-result' - const ACTIVE_AGENT_SEND_TIMEOUT_MS = 8000 -const ORCA_DESKTOP_TERMINAL_CLIENT = { id: 'orca-desktop', type: 'desktop' as const } -export async function sendNotesToActiveAgentSession({ +export async function sendNotesToActiveAgentSession(args: { + worktreeId: string + prompt: string + noteTarget?: ActiveTerminalNoteTarget + timeoutMs?: number +}): Promise { + try { + return await sendNotesToActiveAgentSessionInternal(args) + } catch (error) { + return reportNoteSendFailure( + { status: 'status-unavailable', code: runtimeFailureFallbackCode(error) }, + args.noteTarget ?? null + ) + } +} +async function sendNotesToActiveAgentSessionInternal({ worktreeId, prompt, noteTarget: explicitNoteTarget, @@ -51,20 +64,13 @@ export async function sendNotesToActiveAgentSession({ }): Promise { const trimmedPrompt = prompt.trim() if (!trimmedPrompt) { - return { status: 'empty' } + return { status: 'empty', code: 'empty' } } - const state = useAppStore.getState() - // Why: an explicit target lets the notes dropdown address ANY running agent of - // the worktree, not just the focused pane; omitted, fall back to the focused - // active terminal so existing callers keep their behavior. Routing below still - // resolves the worktree's owner host, so explicit targets stay SSH/remote-correct. const noteTarget = explicitNoteTarget ?? getActiveTerminalNoteTarget(state, worktreeId) if (!noteTarget) { - return { status: 'no-active-terminal' } + return reportNoteSendFailure({ status: 'no-active-terminal', code: 'no-note-target' }, null) } - // Route by the worktree's owner host so the agent terminal is found and driven - // on the host that actually runs it, not on the focused runtime. const runtimeTarget = getActiveRuntimeTarget( getSettingsForWorktreeRuntimeOwner(state, worktreeId) ) @@ -75,21 +81,30 @@ export async function sendNotesToActiveAgentSession({ ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS ) if (!terminal) { - return { status: 'no-active-terminal' } + return reportNoteSendFailure( + { status: 'no-active-terminal', code: 'no-inventory-match' }, + noteTarget + ) } - if (explicitNoteTarget) { - return await sendPromptToExplicitAgentTarget(runtimeTarget, terminal.handle, trimmedPrompt) + return reportNoteSendFailure( + await sendPromptToExplicitAgentTarget(runtimeTarget, terminal.handle, trimmedPrompt), + noteTarget + ) } - const effectiveTimeoutMs = timeoutMs ?? ACTIVE_AGENT_SEND_TIMEOUT_MS const initialAgentStatus = await getTerminalAgentSendReadiness(runtimeTarget, terminal.handle, { allowLegacyFallback: true }) if (initialAgentStatus.status !== 'sendable') { - return { status: initialAgentStatus.status } + return reportNoteSendFailure( + { + status: initialAgentStatus.status, + code: initialAgentStatus.code ?? codeForReadinessStatus(initialAgentStatus.status) + }, + noteTarget + ) } - try { const { wait } = await callRuntimeRpc<{ wait: RuntimeTerminalWait }>( runtimeTarget, @@ -98,160 +113,64 @@ export async function sendNotesToActiveAgentSession({ { timeoutMs: effectiveTimeoutMs + 5000 } ) if (wait.status !== 'running') { - return { status: 'no-active-terminal' } + return reportNoteSendFailure( + { status: 'no-active-terminal', code: 'terminal_wait_not_running' }, + noteTarget + ) } if (wait.blockedReason) { - return { status: 'permission' } + return reportNoteSendFailure( + { status: 'permission', code: 'terminal_wait_blocked' }, + noteTarget + ) } if (!wait.satisfied) { - return { status: 'not-ready' } + return reportNoteSendFailure( + { status: 'not-ready', code: 'terminal_wait_unsatisfied' }, + noteTarget + ) } } catch (error) { if (isRuntimeTerminalUnavailable(error)) { - return { status: 'no-active-terminal' } + return reportNoteSendFailure( + { status: 'no-active-terminal', code: runtimeFailureCode(error) ?? 'runtime-unverifiable' }, + noteTarget + ) } if (isRuntimeTimeout(error)) { - return { status: 'not-ready' } + return reportNoteSendFailure( + { status: 'not-ready', code: 'terminal_wait_timeout' }, + noteTarget + ) } throw error } - const finalAgentStatus = await getTerminalAgentSendReadiness(runtimeTarget, terminal.handle, { allowLegacyFallback: true }) if (finalAgentStatus.status !== 'sendable') { - return { status: finalAgentStatus.status } + return reportNoteSendFailure( + { + status: finalAgentStatus.status, + code: finalAgentStatus.code ?? codeForReadinessStatus(finalAgentStatus.status) + }, + noteTarget + ) } if (finalAgentStatus.supportsGuardedSend) { - return await sendPromptWithGuardedPasteAndEnter(runtimeTarget, terminal.handle, trimmedPrompt, { - allowLegacyFallback: false - }) - } - - // Why: protocol-compatible older SSH runtimes do not know the guarded send - // option. They already passed terminal.wait + legacy isRunningAgent checks, - // so preserve the old active-focused send path for remote compatibility. - return await sendPromptWithLegacyCombinedSend(runtimeTarget, terminal.handle, trimmedPrompt) -} - -async function sendPromptWithLegacyCombinedSend( - runtimeTarget: ReturnType, - terminalHandle: string, - prompt: string -): Promise { - try { - const { send } = await callRuntimeRpc<{ send: RuntimeTerminalSend }>( - runtimeTarget, - 'terminal.send', - { - terminal: terminalHandle, - text: prompt, - enter: true, - client: ORCA_DESKTOP_TERMINAL_CLIENT - }, - { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } + return reportNoteSendFailure( + await sendPromptWithGuardedPasteAndEnter(runtimeTarget, terminal.handle, trimmedPrompt, { + allowLegacyFallback: false + }), + noteTarget ) - return send.accepted ? { status: 'sent' } : { status: 'not-writable' } - } catch (error) { - if (isRuntimeTerminalUnavailable(error)) { - return { status: 'no-active-terminal' } - } - if (isRuntimeTerminalNotWritable(error)) { - return { status: 'not-writable' } - } - throw error - } -} - -async function sendPromptWithGuardedPasteAndEnter( - runtimeTarget: ReturnType, - terminalHandle: string, - prompt: string, - options: { allowLegacyFallback: boolean } -): Promise { - const initialAgentStatus = await getTerminalAgentSendReadiness(runtimeTarget, terminalHandle, { - allowLegacyFallback: options.allowLegacyFallback - }) - // Why: the readiness probe and write guard can observe different transient - // title/process snapshots; the guard owns the bounded no-agent recheck. - if ( - initialAgentStatus.status !== 'sendable' && - !(initialAgentStatus.status === 'no-agent' && initialAgentStatus.supportsGuardedSend) - ) { - return { status: initialAgentStatus.status } } - const pastePayload = `${BRACKETED_PASTE_BEGIN}${sanitizeTerminalPasteText(prompt)}${BRACKETED_PASTE_END}` - try { - const { send } = await callRuntimeRpc<{ send: RuntimeTerminalSend }>( - runtimeTarget, - 'terminal.send', - { - terminal: terminalHandle, - text: pastePayload, - requireAgentStatus: 'sendable', - client: ORCA_DESKTOP_TERMINAL_CLIENT - }, - { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } - ) - if (!send.accepted) { - if (send.refusedReason === 'permission') { - return { status: 'permission' } - } - if (send.refusedReason === 'no-agent') { - return { status: 'no-agent' } - } - return { status: 'not-writable' } - } - } catch (error) { - if (isRuntimeTerminalUnavailable(error)) { - return { status: 'no-active-terminal' } - } - if (isRuntimeTerminalNotWritable(error)) { - return { status: 'not-writable' } - } - throw error - } - - await new Promise((resolve) => setTimeout(resolve, POST_PASTE_SUBMIT_DELAY_MS)) - - try { - const submitAgentStatus = await getTerminalAgentSendReadiness(runtimeTarget, terminalHandle, { - allowLegacyFallback: options.allowLegacyFallback - }) - if ( - submitAgentStatus.status !== 'sendable' && - !(submitAgentStatus.status === 'no-agent' && submitAgentStatus.supportsGuardedSend) - ) { - return { status: 'partial-submit-failed' } - } - } catch (error) { - if (isRuntimeTerminalUnavailable(error)) { - return { status: 'partial-submit-failed' } - } - throw error - } - - try { - const { send } = await callRuntimeRpc<{ send: RuntimeTerminalSend }>( - runtimeTarget, - 'terminal.send', - { - terminal: terminalHandle, - enter: true, - requireAgentStatus: 'sendable', - client: ORCA_DESKTOP_TERMINAL_CLIENT - }, - { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } - ) - return send.accepted ? { status: 'sent' } : { status: 'partial-submit-failed' } - } catch (error) { - if (isRuntimeTerminalUnavailable(error) || isRuntimeTerminalNotWritable(error)) { - return { status: 'partial-submit-failed' } - } - throw error - } + return reportNoteSendFailure( + await sendPromptWithLegacyCombinedSend(runtimeTarget, terminal.handle, trimmedPrompt), + noteTarget + ) } async function sendPromptToExplicitAgentTarget( diff --git a/src/renderer/src/lib/active-agent-note-target.ts b/src/renderer/src/lib/active-agent-note-target.ts index dbacc97e4c1..2b114166d1f 100644 --- a/src/renderer/src/lib/active-agent-note-target.ts +++ b/src/renderer/src/lib/active-agent-note-target.ts @@ -1,4 +1,5 @@ import type { RuntimeTerminalListResult } from '../../../shared/runtime-types' +import { toHostSessionTabId } from '../../../shared/terminal-surface-id' import { AGENT_STATUS_STALE_AFTER_MS, type AgentStatusEntry @@ -174,9 +175,11 @@ export async function findActiveRuntimeTerminal( }, { timeoutMs } ) + // Why: paired renderer tabs wrap the host id with `web-terminal-*`. + const runtimeTabId = toHostSessionTabId(noteTarget.tabId) return ( terminals.find( - (terminal) => terminal.tabId === noteTarget.tabId && terminal.leafId === noteTarget.leafId + (terminal) => terminal.tabId === runtimeTabId && terminal.leafId === noteTarget.leafId ) ?? null ) } diff --git a/src/renderer/src/lib/active-agent-terminal-send-readiness.ts b/src/renderer/src/lib/active-agent-terminal-send-readiness.ts index fa049398e8b..b432366527f 100644 --- a/src/renderer/src/lib/active-agent-terminal-send-readiness.ts +++ b/src/renderer/src/lib/active-agent-terminal-send-readiness.ts @@ -1,6 +1,12 @@ import type { RuntimeTerminalAgentStatus } from '../../../shared/runtime-types' +import { hasRuntimeRpcErrorCode } from '../../../shared/runtime-rpc-error-code' +import type { ActiveAgentNotesSendFailureCode } from './active-agent-note-send-result' import { callRuntimeRpc, RuntimeRpcCallError } from '@/runtime/runtime-rpc-client' import type { getActiveRuntimeTarget } from '@/runtime/runtime-rpc-client' +import { + runtimeFailureCode, + TERMINAL_RUNTIME_FAILURE_CODES +} from './active-agent-note-send-diagnostics' export const ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS = 15000 @@ -14,6 +20,7 @@ export type TerminalAgentSendReadiness = export type TerminalAgentSendReadinessResult = { status: TerminalAgentSendReadiness supportsGuardedSend: boolean + code?: ActiveAgentNotesSendFailureCode } export async function getTerminalAgentSendReadiness( @@ -44,13 +51,14 @@ export async function getTerminalAgentSendReadiness( } // Why: active-focused sends still wait for tui-idle, preserving old // runtime compatibility without immediate selected-target risk. - return { - status: await getLegacyTerminalAgentSendStatus(runtimeTarget, terminalHandle), - supportsGuardedSend: false - } + return await getLegacyTerminalAgentSendStatus(runtimeTarget, terminalHandle) } if (isRuntimeTerminalUnavailable(error)) { - return { status: 'no-active-terminal', supportsGuardedSend: false } + return { + status: 'no-active-terminal', + supportsGuardedSend: false, + code: runtimeTerminalUnavailableCode(error) + } } throw error } @@ -59,7 +67,7 @@ export async function getTerminalAgentSendReadiness( async function getLegacyTerminalAgentSendStatus( runtimeTarget: ReturnType, terminalHandle: string -): Promise { +): Promise { try { const { isRunningAgent } = await callRuntimeRpc<{ isRunningAgent: boolean }>( runtimeTarget, @@ -67,31 +75,38 @@ async function getLegacyTerminalAgentSendStatus( { terminal: terminalHandle }, { timeoutMs: ACTIVE_AGENT_SEND_RPC_TIMEOUT_MS } ) - return isRunningAgent ? 'sendable' : 'no-agent' + return { + status: isRunningAgent ? 'sendable' : 'no-agent', + supportsGuardedSend: false + } } catch (error) { if (isRuntimeTerminalUnavailable(error)) { - return 'no-active-terminal' + return { + status: 'no-active-terminal', + supportsGuardedSend: false, + code: runtimeTerminalUnavailableCode(error) + } } throw error } } +function runtimeTerminalUnavailableCode(error: unknown): ActiveAgentNotesSendFailureCode { + return runtimeFailureCode(error) ?? 'runtime-unverifiable' +} + export function isRuntimeTimeout(error: unknown): boolean { + if (hasRuntimeRpcErrorCode(error, 'runtime_timeout')) { + return true + } const message = error instanceof Error ? error.message : String(error) return message.includes('timeout') } export function isRuntimeTerminalUnavailable(error: unknown): boolean { - const message = error instanceof Error ? error.message : String(error) - return ( - message.includes('terminal_handle_stale') || - message.includes('terminal_exited') || - message.includes('terminal_gone') || - message.includes('no_active_terminal') - ) + return TERMINAL_RUNTIME_FAILURE_CODES.some((code) => hasRuntimeRpcErrorCode(error, code)) } export function isRuntimeTerminalNotWritable(error: unknown): boolean { - const message = error instanceof Error ? error.message : String(error) - return message.includes('terminal_not_writable') + return hasRuntimeRpcErrorCode(error, 'terminal_not_writable') } diff --git a/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts index 9fa15ffb81e..ebd0f016b3f 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts @@ -149,9 +149,11 @@ export function createUiAgentActions( worktreeId: mode.worktreeId, prompt: mode.prompt, noteTarget: { tabId: target.tabId, leafId: target.leafId } - }).catch((error) => { - console.error('Failed to send notes to sidebar agent target:', error) - return { status: 'no-active-terminal' as const } + }).catch(() => { + console.error('Failed to send notes to sidebar agent target:', { + code: 'runtime-unverifiable' + }) + return { status: 'status-unavailable' as const, code: 'runtime-unverifiable' as const } }) const stillCurrent = (): boolean => { @@ -160,7 +162,10 @@ export function createUiAgentActions( } if (result.status !== 'sent') { - const message = activeAgentNotesSendFailureMessage(result.status, { explicitTarget: true }) + const message = activeAgentNotesSendFailureMessage(result.status, { + explicitTarget: true, + code: result.code + }) set((s) => s.agentSendPopoverTargetMode?.id === mode.id && s.agentSendPopoverTargetMode.instanceId === mode.instanceId From 3f5c54332d8832ca2ed54a93d323e7f5a019f909 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 03:00:52 -0700 Subject: [PATCH 004/398] fix(github-project): sort and group empty field values last in both directions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit compareSort early-returned 1 for a missing value — before the trailing DESC flip — but expressed the same idea as `cmp = 1` for an empty users/labels list, which that line then negated. Descending order therefore scattered empty cells across both ends of the table. getFieldValueForGrouping had the matching defect: an empty list fell through to deriveStringValue and produced a blank-label group that the header renders as the literal "All". Both paths now share one predicate, which also covers `text: ''` and `date: ''` — reachable because the view normalizer maps a null GitHub text/date to the empty string. Co-authored-by: kaluli123123 <295758798+kaluli123123@users.noreply.github.com> --- .../github-project/group-sort.test.ts | 106 ++++++++++++++++++ src/shared/github/project-group-sort.ts | 62 +++++----- 2 files changed, 137 insertions(+), 31 deletions(-) diff --git a/src/renderer/src/components/github-project/group-sort.test.ts b/src/renderer/src/components/github-project/group-sort.test.ts index cd5e77738bc..57b96267bae 100644 --- a/src/renderer/src/components/github-project/group-sort.test.ts +++ b/src/renderer/src/components/github-project/group-sort.test.ts @@ -33,6 +33,27 @@ const iterationField: GitHubProjectField = { ] } +const assigneesField: GitHubProjectField = { + kind: 'field', + id: 'F_assignees', + name: 'Assignees', + dataType: 'ASSIGNEES' +} + +const labelsField: GitHubProjectField = { + kind: 'field', + id: 'F_labels', + name: 'Labels', + dataType: 'LABELS' +} + +const textField: GitHubProjectField = { + kind: 'field', + id: 'F_text', + name: 'Notes', + dataType: 'TEXT' +} + function makeRow( id: string, position: number, @@ -206,6 +227,66 @@ describe('sortRows', () => { expect(sorted.map((r) => r.id)).toEqual(['rHas', 'rEmpty']) }) + it('sorts an empty user list last in both directions, like a missing value', () => { + // Why: the DESC flip negated the empty branch, sending unassigned rows to the top. + const rows = [ + makeRow('empty-list', 0, { + F_assignees: { kind: 'users', fieldId: 'F_assignees', users: [] } + }), + makeRow('alice', 1, { + F_assignees: { + kind: 'users', + fieldId: 'F_assignees', + users: [{ login: 'alice', name: null, avatarUrl: null }] + } + }), + makeRow('no-value', 2, {}) + ] + + for (const direction of ['ASC', 'DESC'] as const) { + const view = makeView(assigneesField, { direction, field: assigneesField }) + const sorted = sortRows(makeTable(view, rows), rows) + expect(sorted.map((r) => r.id)).toEqual(['alice', 'empty-list', 'no-value']) + } + }) + + it('sorts an empty label list last in both directions, like a missing value', () => { + const rows = [ + makeRow('empty-list', 0, { + F_labels: { kind: 'labels', fieldId: 'F_labels', labels: [] } + }), + makeRow('bug', 1, { + F_labels: { + kind: 'labels', + fieldId: 'F_labels', + labels: [{ name: 'bug', color: 'ff0000' }] + } + }), + makeRow('no-value', 2, {}) + ] + + for (const direction of ['ASC', 'DESC'] as const) { + const view = makeView(labelsField, { direction, field: labelsField }) + const sorted = sortRows(makeTable(view, rows), rows) + expect(sorted.map((r) => r.id)).toEqual(['bug', 'empty-list', 'no-value']) + } + }) + + it('sorts a blank text value last in both directions, like a missing value', () => { + // Why reachable: the normalizer turns a null GitHub text/date into ''. + const rows = [ + makeRow('blank', 0, { F_text: { kind: 'text', fieldId: 'F_text', text: '' } }), + makeRow('alpha', 1, { F_text: { kind: 'text', fieldId: 'F_text', text: 'alpha' } }), + makeRow('no-value', 2, {}) + ] + + for (const direction of ['ASC', 'DESC'] as const) { + const view = makeView(textField, { direction, field: textField }) + const sorted = sortRows(makeTable(view, rows), rows) + expect(sorted.map((r) => r.id)).toEqual(['alpha', 'blank', 'no-value']) + } + }) + it('keeps sort fallback finite when row positions are absent', () => { const view = makeView(singleSelectField) const rows = [ @@ -220,6 +301,31 @@ describe('sortRows', () => { }) describe('groupRows', () => { + it('groups a present-but-empty user list with the missing-value rows', () => { + // Why: an empty list fell through to a blank-label group, which renders as "All". + const view = { ...makeView(assigneesField), groupByFields: [assigneesField] } + const rows = [ + makeRow('empty-list', 0, { + F_assignees: { kind: 'users', fieldId: 'F_assignees', users: [] } + }), + makeRow('alice', 1, { + F_assignees: { + kind: 'users', + fieldId: 'F_assignees', + users: [{ login: 'alice', name: null, avatarUrl: null }] + } + }), + makeRow('no-value', 2, {}) + ] + + const groups = groupRows(makeTable(view, rows), rows) + + expect(groups.map((group) => [group.label, group.rows.map((r) => r.id)])).toEqual([ + ['alice', ['alice']], + ['No Assignees', ['empty-list', 'no-value']] + ]) + }) + it('places the empty group last', () => { const view = { ...makeView(singleSelectField), diff --git a/src/shared/github/project-group-sort.ts b/src/shared/github/project-group-sort.ts index 707fbfb35e4..0b91d92b267 100644 --- a/src/shared/github/project-group-sort.ts +++ b/src/shared/github/project-group-sort.ts @@ -24,6 +24,29 @@ export type ProjectGroup = { const EMPTY_GROUP_KEY = '__empty__' +type ProjectFieldValue = GitHubProjectRow['fieldValuesByFieldId'][string] + +/** False for anything that renders as an empty cell — absent, or present with a blank payload. */ +function hasNonEmptyFieldValue(value: ProjectFieldValue | undefined): boolean { + if (!value) { + return false + } + switch (value.kind) { + case 'users': + return Boolean(value.users[0]?.login) + case 'labels': + return Boolean(value.labels[0]?.name) + case 'text': + return value.text.trim().length > 0 + case 'date': + return value.date.trim().length > 0 + case 'iteration': + case 'number': + case 'single-select': + return true + } +} + // Why: use a finite sentinel instead of Infinity so subtractions in the sort // comparator stay finite. `Infinity - Infinity` is NaN, which makes // Array.sort's behavior implementation-defined and skips later tie-breaks. @@ -35,7 +58,7 @@ function getFieldValueForGrouping( field: GitHubProjectField ): { key: string; label: string; orderHint: number; iteration: ProjectGroup['iteration'] } { const value = row.fieldValuesByFieldId[field.id] - if (!value) { + if (!hasNonEmptyFieldValue(value)) { return { key: EMPTY_GROUP_KEY, label: labelForEmpty(field), @@ -144,14 +167,11 @@ function compareSort(a: GitHubProjectRow, b: GitHubProjectRow, sort: GitHubProje const field = sort.field const aValue = a.fieldValuesByFieldId[field.id] const bValue = b.fieldValuesByFieldId[field.id] - if (!aValue && !bValue) { - return 0 - } - if (!aValue) { - return 1 - } - if (!bValue) { - return -1 + // Why: return before the trailing DESC flip so empty sorts last in both directions. + const aFilled = hasNonEmptyFieldValue(aValue) + const bFilled = hasNonEmptyFieldValue(bValue) + if (!aFilled || !bFilled) { + return aFilled === bFilled ? 0 : aFilled ? -1 : 1 } let cmp = 0 @@ -182,29 +202,9 @@ function compareSort(a: GitHubProjectRow, b: GitHubProjectRow, sort: GitHubProje } else if (aValue.kind === 'text' && bValue.kind === 'text') { cmp = aValue.text.localeCompare(bValue.text) } else if (aValue.kind === 'users' && bValue.kind === 'users') { - const aLogin = aValue.users[0]?.login ?? '' - const bLogin = bValue.users[0]?.login ?? '' - if (!aLogin && !bLogin) { - cmp = 0 - } else if (!aLogin) { - cmp = 1 - } else if (!bLogin) { - cmp = -1 - } else { - cmp = aLogin.localeCompare(bLogin) - } + cmp = (aValue.users[0]?.login ?? '').localeCompare(bValue.users[0]?.login ?? '') } else if (aValue.kind === 'labels' && bValue.kind === 'labels') { - const aName = aValue.labels[0]?.name ?? '' - const bName = bValue.labels[0]?.name ?? '' - if (!aName && !bName) { - cmp = 0 - } else if (!aName) { - cmp = 1 - } else if (!bName) { - cmp = -1 - } else { - cmp = aName.localeCompare(bName) - } + cmp = (aValue.labels[0]?.name ?? '').localeCompare(bValue.labels[0]?.name ?? '') } else { // Why: unknown sort-field kind — ignore this sort field and fall through // to tie-breaks (and eventually row.position). From e89321192aebbd1ad842d2c5bb5a94a46b6d332b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:01:59 -0700 Subject: [PATCH 005/398] perf(worktree): batch remote conflict probes, re-arm the prepared checkout (#17829) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(worktree): batch remote conflict probes, re-arm the prepared checkout A repo with many remotes paid one `git show-ref --verify` subprocess per remote on every branch-conflict check during create. Ask one `git cat-file --batch-check` over stdin instead; it reports a missing ref as data rather than a failed exit, so a batch stays as decidable as the per-ref probe. Hosts that cannot feed stdin, and undecided batches, still fall back to the per-ref path. The prepared checkout was single-use, so the second create in a row paid the full cold `git worktree add`. Re-arm it in the background after one is consumed; the existing TTL and preparation limit still bound it. The create timing recorder existed but its phases were never emitted and did not cover preflight, leaving a multi-second gap in the trace with no attribution. Add `resolve_name`/`prepare_push_target` phases and record the breakdown, plus the unattributed remainder, on the create span. * fix(worktree): format the conflicting review number eagerly for the create error * perf(worktree): re-arm a prepared checkout only for a burst of creates Re-arming after every consumed preparation spends a full checkout and ~200MB of disk on a user who created one worktree and stopped, then pays an unexplained delete when the TTL expires five minutes later. Track when each preparation key was last consumed and only replace it when a second create lands inside the burst window, so the warm second create is still free and an isolated create costs nothing. * fix(worktree): address review findings on the create-path batching Three findings from PR review: The `batched.found` fallback in the remote-conflict probe was unreachable — a present ref is decisive, so `found` never survives with `unknown` set, and the guard above already returns that case. `rearmPreparation` checked for an existing preparation before recording the consume, so a prefetch that re-armed the key while create finalized swallowed the timestamp and made the next create look isolated when it was really mid-burst. Create runs some phases concurrently, so summing phase durations double-counted overlap and understated `unattributed_ms` — the one number that matters when a create is slow for no visible reason. Measure the union of the phase intervals instead. * refactor(worktree): move stale-preparation cleanup into its own module The preparation module crossed the 300-line budget. Crash recovery is a separate concern from the pool itself — it discards preparations another process left registered, single-flighted per repo and runtime so a burst of arming calls shares one worktree listing. * test(worktree): make the re-arm test able to fail The burst test armed a preparation manually after the second consume, so the third checkout appeared whether or not the re-arm produced it — the assertion passed with re-arming disabled. Drop that arming call so the third checkout can only come from the re-arm, and assert the consume results rather than discarding them. --- src/main/git/exact-ref-probe.ts | 49 ++++ .../git/repo-branch-conflict-real-git.test.ts | 49 ++++ src/main/git/repo-branch-conflict.test.ts | 120 +++++++++ src/main/git/repo-branch-conflict.ts | 53 +++- src/main/ipc/worktree-remote.ts | 250 +++++++++--------- .../register-worktree-create-handlers.ts | 10 +- .../observability/instrumentation.test.ts | 48 ++++ src/main/observability/instrumentation.ts | 59 ++++- src/main/worktree-create-preparation-burst.ts | 21 ++ ...rktree-create-preparation-stale-cleanup.ts | 79 ++++++ src/main/worktree-create-preparation.test.ts | 47 ++++ src/main/worktree-create-preparation.ts | 125 ++++----- 12 files changed, 702 insertions(+), 208 deletions(-) create mode 100644 src/main/git/repo-branch-conflict-real-git.test.ts create mode 100644 src/main/worktree-create-preparation-burst.ts create mode 100644 src/main/worktree-create-preparation-stale-cleanup.ts diff --git a/src/main/git/exact-ref-probe.ts b/src/main/git/exact-ref-probe.ts index 6b13cc718c5..96bb421b8e8 100644 --- a/src/main/git/exact-ref-probe.ts +++ b/src/main/git/exact-ref-probe.ts @@ -19,6 +19,8 @@ export type ExactRefProbeSetResult = { type ExactRefPresence = 'present' | 'absent' | 'unknown' const EXACT_REF_PROBE_CONCURRENCY = 8 +// SHA-1 and SHA-256 repositories both report a full object id here. +const OBJECT_ID_PATTERN = /^[0-9a-f]{40}(?:[0-9a-f]{24})?$/ export function isShowRefNoMatchError(error: unknown): boolean { const record = error && typeof error === 'object' ? (error as Record) : undefined @@ -126,3 +128,50 @@ export async function probeAnyExactRef( await Promise.all(Array.from({ length: workerCount }, () => probeNext())) return { found, unknown } } + +/** Runs Git with a stdin payload. Only hosts that can feed a child's stdin supply one. */ +export type ExactRefProbeStdinExec = ( + argv: string[], + options: ExactRefProbeExecOptions & { stdin: string } +) => Promise<{ stdout: string }> + +/** `cat-file --batch-check` reports every ref from one child, and reports a missing ref as data + * rather than a failed exit — so a batch stays as decidable as a per-ref `show-ref --verify`. + * A repo with many remotes otherwise pays one subprocess per remote on every conflict check. */ +export async function probeAnyExactRefBatched( + runGit: ExactRefProbeStdinExec, + refs: readonly string[], + options: ExactRefProbeExecOptions = {} +): Promise<{ found: boolean; unknown: boolean }> { + const uniqueRefs = [...new Set(refs)] + const safeRefs = uniqueRefs.filter((ref) => isSafeGitRefName(ref)) + if (safeRefs.length === 0) { + return { found: false, unknown: uniqueRefs.length > 0 } + } + let stdout: string + try { + ;({ stdout } = await runGit(['cat-file', '--batch-check'], { + ...options, + stdin: `${safeRefs.join('\n')}\n` + })) + } catch { + return { found: false, unknown: true } + } + const lines = stdout.split('\n').filter((line) => line.trim().length > 0) + // One line per input, in order; a short read means the batch never answered for the rest. + if (lines.length !== safeRefs.length) { + return { found: false, unknown: true } + } + let unknown = safeRefs.length !== uniqueRefs.length + for (const line of lines) { + const [head, type] = line.split(' ') + if (OBJECT_ID_PATTERN.test(head) && type !== undefined && type !== 'missing') { + return { found: true, unknown: false } + } + if (type !== 'missing') { + // `ambiguous`, or a spelling this Git reports differently; neither proves absence. + unknown = true + } + } + return { found: false, unknown } +} diff --git a/src/main/git/repo-branch-conflict-real-git.test.ts b/src/main/git/repo-branch-conflict-real-git.test.ts new file mode 100644 index 00000000000..34092273eda --- /dev/null +++ b/src/main/git/repo-branch-conflict-real-git.test.ts @@ -0,0 +1,49 @@ +import { execFileSync } from 'node:child_process' +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { getBranchConflictKind } from './repo-branch-conflict' + +describe('branch conflict real Git contract', () => { + const tempPaths: string[] = [] + + afterEach(() => { + for (const path of tempPaths.splice(0)) { + rmSync(path, { recursive: true, force: true }) + } + }) + + it('decides remote conflicts from one batched probe across many remotes', async () => { + const repoPath = mkdtempSync(join(tmpdir(), 'orca-branch-conflict-')) + tempPaths.push(repoPath) + const git = (...args: string[]): string => + execFileSync('git', args, { cwd: repoPath, encoding: 'utf8' }) + + git('init', '--quiet') + git('config', 'user.name', 'Orca Test') + git('config', 'user.email', 'orca@example.test') + git('config', 'commit.gpgSign', 'false') + git('config', 'core.hooksPath', '.git/no-hooks') + writeFileSync(join(repoPath, 'fixture.txt'), 'base\n') + git('add', 'fixture.txt') + git('commit', '--quiet', '-m', 'base') + const head = git('rev-parse', 'HEAD').trim() + + // Many remotes is the shape that used to cost one subprocess each. + for (let index = 0; index < 12; index += 1) { + git('remote', 'add', `remote${index}`, 'https://example.test/repo.git') + } + git('update-ref', 'refs/remotes/remote7/taken', head) + + await expect(getBranchConflictKind(repoPath, 'taken')).resolves.toBe('remote') + await expect(getBranchConflictKind(repoPath, 'free')).resolves.toBeNull() + // The allowed base ref is the one remote spelling that is not a conflict. + await expect( + getBranchConflictKind(repoPath, 'taken', 'refs/remotes/remote7/taken') + ).resolves.toBeNull() + + git('branch', 'local-only', head) + await expect(getBranchConflictKind(repoPath, 'local-only')).resolves.toBe('local') + }) +}) diff --git a/src/main/git/repo-branch-conflict.test.ts b/src/main/git/repo-branch-conflict.test.ts index dab8dd396d6..873c8bd3340 100644 --- a/src/main/git/repo-branch-conflict.test.ts +++ b/src/main/git/repo-branch-conflict.test.ts @@ -134,3 +134,123 @@ describe('getBranchConflictKindViaExec', () => { expect(exec).not.toHaveBeenCalled() }) }) + +describe('getBranchConflictKindViaExec batched remote probe', () => { + function remoteNames(count: number): string { + return `${Array.from({ length: count }, (_, index) => `remote${index}`).join('\n')}\n` + } + + function baseExec(calls: string[][]): (argv: string[]) => Promise<{ stdout: string }> { + return async (argv) => { + calls.push(argv) + if (argv[0] === 'rev-parse') { + throw new Error('local branch is absent') + } + if (argv[0] === 'remote') { + return { stdout: remoteNames(3) } + } + throw new Error(`unexpected git command: ${argv.join(' ')}`) + } + } + + it('asks one batched child instead of one probe per remote', async () => { + const calls: string[][] = [] + const stdinPayloads: (string | undefined)[] = [] + const exec = baseExec(calls) + const batched = async ( + argv: string[], + options: { stdin: string } + ): Promise<{ stdout: string }> => { + calls.push(argv) + stdinPayloads.push(options.stdin) + return { + stdout: [ + 'refs/remotes/remote0/feature missing', + 'refs/remotes/remote1/feature missing', + 'refs/remotes/remote2/feature missing' + ].join('\n') + } + } + + await expect( + getBranchConflictKindViaExec(exec, 'feature', undefined, {}, batched) + ).resolves.toBeNull() + expect(calls).toEqual([ + ['rev-parse', '--verify', 'refs/heads/feature'], + ['remote'], + ['cat-file', '--batch-check'] + ]) + expect(stdinPayloads).toEqual([ + 'refs/remotes/remote0/feature\nrefs/remotes/remote1/feature\nrefs/remotes/remote2/feature\n' + ]) + }) + + it('reports a remote conflict from the batched answer', async () => { + const calls: string[][] = [] + const exec = baseExec(calls) + const batched = async (): Promise<{ stdout: string }> => ({ + stdout: [ + 'refs/remotes/remote0/feature missing', + `${'a'.repeat(40)} commit 214`, + 'refs/remotes/remote2/feature missing' + ].join('\n') + }) + + await expect( + getBranchConflictKindViaExec(exec, 'feature', undefined, {}, batched) + ).resolves.toBe('remote') + }) + + it('falls back to per-ref probes when the batch cannot answer', async () => { + const calls: string[][] = [] + const exec = async (argv: string[]): Promise<{ stdout: string }> => { + calls.push(argv) + if (argv[0] === 'rev-parse') { + throw new Error('local branch is absent') + } + if (argv[0] === 'remote') { + return { stdout: remoteNames(3) } + } + if (argv[0] === 'show-ref') { + if (argv[4] === 'refs/remotes/remote1/feature') { + return { stdout: 'abc refs/remotes/remote1/feature\n' } + } + throw Object.assign(new Error('missing'), { code: 1, stderr: '' }) + } + throw new Error(`unexpected git command: ${argv.join(' ')}`) + } + const batched = async (): Promise<{ stdout: string }> => { + throw new Error('cat-file is unavailable') + } + + await expect( + getBranchConflictKindViaExec(exec, 'feature', undefined, {}, batched) + ).resolves.toBe('remote') + expect(calls.filter((argv) => argv[0] === 'show-ref')).toHaveLength(3) + }) + + it('treats a short batch read as undecided rather than as absence', async () => { + const calls: string[][] = [] + const exec = async (argv: string[]): Promise<{ stdout: string }> => { + calls.push(argv) + if (argv[0] === 'rev-parse') { + throw new Error('local branch is absent') + } + if (argv[0] === 'remote') { + return { stdout: remoteNames(3) } + } + if (argv[0] === 'show-ref') { + throw Object.assign(new Error('missing'), { code: 1, stderr: '' }) + } + throw new Error(`unexpected git command: ${argv.join(' ')}`) + } + const batched = async (): Promise<{ stdout: string }> => ({ + stdout: 'refs/remotes/remote0/feature missing' + }) + + await expect( + getBranchConflictKindViaExec(exec, 'feature', undefined, {}, batched) + ).resolves.toBeNull() + expect(calls.filter((argv) => argv[0] === 'show-ref')).toHaveLength(3) + }) +}) diff --git a/src/main/git/repo-branch-conflict.ts b/src/main/git/repo-branch-conflict.ts index 162d5ef53b3..c799d3770be 100644 --- a/src/main/git/repo-branch-conflict.ts +++ b/src/main/git/repo-branch-conflict.ts @@ -4,8 +4,10 @@ import { gitExecFileAsync } from './runner' import { isSafeGitRefName } from '../../shared/git-status-upstream-ref' import { probeAnyExactRef, + probeAnyExactRefBatched, type ExactRefProbeExec, - type ExactRefProbeExecOptions + type ExactRefProbeExecOptions, + type ExactRefProbeStdinExec } from './exact-ref-probe' export type BranchConflictKind = 'local' | 'remote' @@ -79,12 +81,31 @@ function buildRemoteBranchConflictRefs( return [...refs] } +/** One batched child answers for every remote; the per-ref probes only run when the host cannot + * feed stdin, or when the batch came back undecided. */ +async function probeAnyRemoteConflictRef( + exec: ExactRefProbeExec, + batchedExec: ExactRefProbeStdinExec | undefined, + candidateRefs: readonly string[], + probeOptions: ExactRefProbeExecOptions +): Promise<{ found: boolean }> { + if (batchedExec) { + // A present ref is always decisive, so `found` never survives with `unknown` set. + const batched = await probeAnyExactRefBatched(batchedExec, candidateRefs, probeOptions) + if (!batched.unknown) { + return { found: batched.found } + } + } + return probeAnyExactRef(exec, candidateRefs, probeOptions) +} + /** Run branch-conflict policy through the host that owns Git execution. */ export async function getBranchConflictKindViaExec( exec: ExactRefProbeExec, branchName: string, allowedBaseRef?: string, - options: ExactRefProbeExecOptions = {} + options: ExactRefProbeExecOptions = {}, + batchedExec?: ExactRefProbeStdinExec ): Promise { if (!canQueryRemoteBranchName(branchName)) { return null @@ -104,7 +125,12 @@ export async function getBranchConflictKindViaExec( return null } - const { found: hasRemoteConflict } = await probeAnyExactRef(exec, candidateRefs, probeOptions) + const { found: hasRemoteConflict } = await probeAnyRemoteConflictRef( + exec, + batchedExec, + candidateRefs, + probeOptions + ) return hasRemoteConflict ? 'remote' : null } catch { @@ -119,15 +145,22 @@ export function getBranchConflictKind( options: LocalGitExecOptions = {} ): Promise { const execOptions = gitExecOptions(path, options) + const runLocalGit = ( + argv: string[], + commandOptions?: ExactRefProbeExecOptions & { stdin?: string } + ): Promise<{ stdout: string }> => + gitExecFileAsync(argv, { + ...execOptions, + ...(commandOptions?.maxBuffer === undefined ? {} : { maxBuffer: commandOptions.maxBuffer }), + ...(commandOptions?.timeoutMs === undefined ? {} : { timeout: commandOptions.timeoutMs }), + ...(commandOptions?.stdin === undefined ? {} : { stdin: commandOptions.stdin }) + }) return getBranchConflictKindViaExec( - (argv, commandOptions) => - gitExecFileAsync(argv, { - ...execOptions, - ...(commandOptions?.maxBuffer === undefined ? {} : { maxBuffer: commandOptions.maxBuffer }), - ...(commandOptions?.timeoutMs === undefined ? {} : { timeout: commandOptions.timeoutMs }) - }), + runLocalGit, branchName, - allowedBaseRef + allowedBaseRef, + {}, + (argv, commandOptions) => runLocalGit(argv, commandOptions) ) } diff --git a/src/main/ipc/worktree-remote.ts b/src/main/ipc/worktree-remote.ts index e53ae264129..7d28f38f2db 100644 --- a/src/main/ipc/worktree-remote.ts +++ b/src/main/ipc/worktree-remote.ts @@ -2189,130 +2189,139 @@ export async function createLocalWorktree( let lastExistingReviewNumber: number | null = null const shouldRetireGeneratedName = args.nameWasGenerated === true && isGeneratedWorktreeCreateName(sanitizedName) - const retiredNameRegistry = shouldRetireGeneratedName - ? await getRetiredNameRegistryForRepo(store, repo, store.getRepos(), settings) - : null - const isRetiredName = retiredNameRegistry ? createRetiredNameLookup(retiredNameRegistry) : null - // Why: a create-from-review branch override may already exist locally; suffix both branch and path instead of blocking the user. - for (let suffix = 1, attempts = 0; attempts < WORKTREE_CREATE_MAX_SUFFIX_ATTEMPTS; suffix += 1) { - effectiveSanitizedName = shouldRetireGeneratedName - ? getGeneratedWorktreeCreateCandidate( - sanitizedName, - suffix, - retiredNameRegistry?.exhaustedTiers - ) - : getWorktreeCreateCandidate(sanitizedName, suffix) - effectiveRequestedName = shouldRetireGeneratedName - ? effectiveSanitizedName - : requestedName.trim() - ? getWorktreeCreateCandidate(requestedName, suffix) - : effectiveSanitizedName - if (isRetiredName?.(effectiveSanitizedName)) { - continue - } - attempts += 1 - lastExistingReviewNumber = null + await timing.time('resolve_name', async () => { + const retiredNameRegistry = shouldRetireGeneratedName + ? await getRetiredNameRegistryForRepo(store, repo, store.getRepos(), settings) + : null + const isRetiredName = retiredNameRegistry ? createRetiredNameLookup(retiredNameRegistry) : null + // Why: a create-from-review branch override may already exist locally; suffix both branch and path instead of blocking the user. + for ( + let suffix = 1, attempts = 0; + attempts < WORKTREE_CREATE_MAX_SUFFIX_ATTEMPTS; + suffix += 1 + ) { + effectiveSanitizedName = shouldRetireGeneratedName + ? getGeneratedWorktreeCreateCandidate( + sanitizedName, + suffix, + retiredNameRegistry?.exhaustedTiers + ) + : getWorktreeCreateCandidate(sanitizedName, suffix) + effectiveRequestedName = shouldRetireGeneratedName + ? effectiveSanitizedName + : requestedName.trim() + ? getWorktreeCreateCandidate(requestedName, suffix) + : effectiveSanitizedName + if (isRetiredName?.(effectiveSanitizedName)) { + continue + } + attempts += 1 + lastExistingReviewNumber = null - branchName = await resolveCreateBranchName( - repo.path, - selectedExistingLocalBranchName - ? selectedExistingLocalBranchName - : getBranchNameOverrideCandidate(args.branchNameOverride, suffix), - effectiveSanitizedName, - settings, - username, - localWorktreeGitOptions - ) - checkoutExistingBranch = await canCheckoutExistingLocalBranch( - repo.path, - branchName, - baseBranch, - localWorktreeGitOptions - ) - if (checkoutExistingBranch && !selectedExistingLocalBranchName) { - // Why: suffix retries may need a new path, but an existing-branch checkout must keep the user-selected branch, not a sibling. - selectedExistingLocalBranchName = branchName - } - lastBranchConflictKind = checkoutExistingBranch - ? null - : await getBranchConflictKind(repo.path, branchName, baseBranch, localWorktreeGitOptions) - const allowedPushTargetRemoteConflict = - lastBranchConflictKind && - isAllowedPushTargetRemoteConflict(lastBranchConflictKind, branchName, args) - if (lastBranchConflictKind) { - if (allowedPushTargetRemoteConflict) { - lastExistingPR = null - let lookupFailed = false - const selectedReview = getSelectedReviewBranch(args) - if (selectedReview?.provider === 'github') { - try { - lastExistingPR = await getLocalGitHubPrForBranch( - repo.path, - branchName, - localWorktreeGitOptions - ) - } catch { - lookupFailed = true - } - if (!lookupFailed && isMatchingSelectedGitHubPr(lastExistingPR, args, branchName)) { - lastBranchConflictKind = null - } else if (lastExistingPR) { - lastExistingReviewNumber = lastExistingPR.number - } - } else if (selectedReview) { - let hostedReview: Awaited> = null - try { - hostedReview = await getSelectedHostedReviewForBranch(repo, branchName, args) - } catch { - lookupFailed = true - } - if (!lookupFailed && hostedReview?.matchesSelected) { - lastBranchConflictKind = null - } else if (hostedReview) { - lastExistingReviewNumber = hostedReview.number + branchName = await resolveCreateBranchName( + repo.path, + selectedExistingLocalBranchName + ? selectedExistingLocalBranchName + : getBranchNameOverrideCandidate(args.branchNameOverride, suffix), + effectiveSanitizedName, + settings, + username, + localWorktreeGitOptions + ) + checkoutExistingBranch = await canCheckoutExistingLocalBranch( + repo.path, + branchName, + baseBranch, + localWorktreeGitOptions + ) + if (checkoutExistingBranch && !selectedExistingLocalBranchName) { + // Why: suffix retries may need a new path, but an existing-branch checkout must keep the user-selected branch, not a sibling. + selectedExistingLocalBranchName = branchName + } + lastBranchConflictKind = checkoutExistingBranch + ? null + : await getBranchConflictKind(repo.path, branchName, baseBranch, localWorktreeGitOptions) + const allowedPushTargetRemoteConflict = + lastBranchConflictKind && + isAllowedPushTargetRemoteConflict(lastBranchConflictKind, branchName, args) + if (lastBranchConflictKind) { + if (allowedPushTargetRemoteConflict) { + lastExistingPR = null + let lookupFailed = false + const selectedReview = getSelectedReviewBranch(args) + if (selectedReview?.provider === 'github') { + try { + lastExistingPR = await getLocalGitHubPrForBranch( + repo.path, + branchName, + localWorktreeGitOptions + ) + } catch { + lookupFailed = true + } + if (!lookupFailed && isMatchingSelectedGitHubPr(lastExistingPR, args, branchName)) { + lastBranchConflictKind = null + } else if (lastExistingPR) { + lastExistingReviewNumber = lastExistingPR.number + } + } else if (selectedReview) { + let hostedReview: Awaited> = null + try { + hostedReview = await getSelectedHostedReviewForBranch(repo, branchName, args) + } catch { + lookupFailed = true + } + if (!lookupFailed && hostedReview?.matchesSelected) { + lastBranchConflictKind = null + } else if (hostedReview) { + lastExistingReviewNumber = hostedReview.number + } } } } - } - if (lastBranchConflictKind) { - continue - } - - // Why: gh pr list is a ~1–3s network call; only probe PR conflicts after a branch collision (suffix > 1) so the common no-collision path skips it. - if (suffix > 1 && !checkoutExistingBranch) { - lastExistingPR = null - try { - lastExistingPR = await getLocalGitHubPrForBranch( - repo.path, - branchName, - localWorktreeGitOptions - ) - } catch { - // GitHub API may be unreachable, rate-limited, or token missing - } - if (lastExistingPR && !isMatchingSelectedGitHubPr(lastExistingPR, args, branchName)) { - lastExistingReviewNumber = lastExistingPR.number + if (lastBranchConflictKind) { continue } - } - worktreePath = ensurePathWithinWorkspace( - computeWorktreePath(effectiveSanitizedName, repo.path, worktreePathSettings), - workspaceRoot - ) - if (existsSync(worktreePath)) { - continue - } + // Why: gh pr list is a ~1–3s network call; only probe PR conflicts after a branch collision (suffix > 1) so the common no-collision path skips it. + if (suffix > 1 && !checkoutExistingBranch) { + lastExistingPR = null + try { + lastExistingPR = await getLocalGitHubPrForBranch( + repo.path, + branchName, + localWorktreeGitOptions + ) + } catch { + // GitHub API may be unreachable, rate-limited, or token missing + } + if (lastExistingPR && !isMatchingSelectedGitHubPr(lastExistingPR, args, branchName)) { + lastExistingReviewNumber = lastExistingPR.number + continue + } + } - resolved = true - break - } + worktreePath = ensurePathWithinWorkspace( + computeWorktreePath(effectiveSanitizedName, repo.path, worktreePathSettings), + workspaceRoot + ) + if (existsSync(worktreePath)) { + continue + } + + resolved = true + break + } + }) if (!resolved) { // Why: every suffix collided; reject with a specific reason so the user sees why create failed instead of a generic error or hung spinner. - if (lastExistingReviewNumber !== null) { + // Read once and format eagerly: the suffix loop assigns this from a callback, so the `let`'s + // narrowing does not reach the message. + const existingReviewNumber = lastExistingReviewNumber + if (existingReviewNumber !== null) { throw new Error( - `Branch "${branchName}" already has PR #${lastExistingReviewNumber}. Pick a different ${branchConflictSubject}.` + `Branch "${branchName}" already has PR #${String(existingReviewNumber)}. Pick a different ${branchConflictSubject}.` ) } if (lastBranchConflictKind) { @@ -2361,14 +2370,17 @@ export async function createLocalWorktree( emitCreateWorktreeProgress(mainWindow, 'creating', args.creationId) let preparedPushTarget: GitPushTarget | undefined - if (args.pushTarget) { + const requestedPushTarget = args.pushTarget + if (requestedPushTarget) { // Why: validate/fetch the contributor remote before create so a failure doesn't leave a half-created worktree with conflicts on retry. - preparedPushTarget = await prepareWorktreePushTarget( - repo.path, - args.pushTarget, - store, - repo.id, - localWorktreeGitOptions + preparedPushTarget = await timing.time('prepare_push_target', () => + prepareWorktreePushTarget( + repo.path, + requestedPushTarget, + store, + repo.id, + localWorktreeGitOptions + ) ) } diff --git a/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts b/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts index 775a0859790..f371529963c 100644 --- a/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts +++ b/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts @@ -4,7 +4,10 @@ import type { CreateWorktreeResult, AdoptProvisionedRootArgs } from '../../../../shared/worktree/create-types' -import { withWorktreeSpan } from '../../../observability/instrumentation' +import { + addWorktreeCreatePhaseAttributes, + withWorktreeSpan +} from '../../../observability/instrumentation' import { workspaceSourceSchema } from '../../../../shared/telemetry-events' import type { WorkspaceSource } from '../../../../shared/telemetry-events' import { @@ -36,7 +39,7 @@ export function registerWorktreeCreateHandlers(context: WorktreeIpcContext): voi async (_event, rawArgs: CreateWorktreeArgs): Promise => { const args = normalizeLinkedWorkItemFields(rawArgs) // Why span here: parent the child git spans for the trace tree; don't attach branch name/remote URL (user content) — repo ID is the safer correlator. - return withWorktreeSpan({ stage: 'create' }, async () => { + return withWorktreeSpan({ stage: 'create' }, async (span) => { const repo = store.getRepo(args.repoId) if (!repo) { throw new Error(`Repo not found: ${args.repoId}`) @@ -74,6 +77,9 @@ export function registerWorktreeCreateHandlers(context: WorktreeIpcContext): voi throw error } finishAutomationWorkspaceProvenanceRequest(args.automationProvenanceRequest) + if (result.timing) { + addWorktreeCreatePhaseAttributes(span, result.timing) + } // Why: reaching here means create succeeded (helpers throw); skip a separate workspace_initialized (telemetry-plan.md§Deferred); never send the branch name. track('workspace_created', { diff --git a/src/main/observability/instrumentation.test.ts b/src/main/observability/instrumentation.test.ts index 2f7d1da05dd..978854897e4 100644 --- a/src/main/observability/instrumentation.test.ts +++ b/src/main/observability/instrumentation.test.ts @@ -3,6 +3,7 @@ import { _resetTracerForTests, setActiveSink, type TracerSink } from './tracer' import { _gitSpanSamplingBucketCountForTests, _resetGitSpanSamplingForTests, + addWorktreeCreatePhaseAttributes, withGitSpan } from './instrumentation' @@ -167,3 +168,50 @@ describe('withGitSpan sampling', () => { expect(_gitSpanSamplingBucketCountForTests()).toBe(1) }) }) + +describe('addWorktreeCreatePhaseAttributes', () => { + function capture(): { + attributes: Record + span: Parameters[0] + } { + const attributes: Record = {} + const span = { + setAttribute: (key: string, value: unknown) => { + attributes[key] = value + } + } as unknown as Parameters[0] + return { attributes, span } + } + + it('counts concurrent phases once when measuring unattributed time', () => { + const { attributes, span } = capture() + // Create resolves shared directories and .worktreeinclude concurrently; summing their + // durations would claim 400ms of coverage for a 200ms window. + addWorktreeCreatePhaseAttributes(span, { + totalDurationMs: 1000, + phases: [ + { phase: 'resolve_shared_directories', startedAtMs: 100, durationMs: 200 }, + { phase: 'resolve_worktreeinclude', startedAtMs: 150, durationMs: 150 } + ] + }) + + expect(attributes['worktree.create.phase.resolve_shared_directories_ms']).toBe(200) + expect(attributes['worktree.create.phase.resolve_worktreeinclude_ms']).toBe(150) + // Covered wall clock is 100..300, so 800ms is genuinely unaccounted for. + expect(attributes['worktree.create.unattributed_ms']).toBe(800) + }) + + it('sums disjoint phases and never reports negative unattributed time', () => { + const { attributes, span } = capture() + addWorktreeCreatePhaseAttributes(span, { + totalDurationMs: 500, + phases: [ + { phase: 'resolve_name', startedAtMs: 0, durationMs: 100 }, + { phase: 'git_worktree_add', startedAtMs: 300, durationMs: 200 } + ] + }) + + expect(attributes['worktree.create.total_ms']).toBe(500) + expect(attributes['worktree.create.unattributed_ms']).toBe(200) + }) +}) diff --git a/src/main/observability/instrumentation.ts b/src/main/observability/instrumentation.ts index bab57b74f64..fb57aa5a870 100644 --- a/src/main/observability/instrumentation.ts +++ b/src/main/observability/instrumentation.ts @@ -202,10 +202,11 @@ export type WorktreeSpanArgs = { readonly path?: string } -/** Wrap a worktree-setup phase in a `worktree.` span. */ +/** Wrap a worktree-setup phase in a `worktree.` span. The callback receives the span so a + * create can attach its own phase breakdown; the git children alone leave the waits invisible. */ export async function withWorktreeSpan( meta: WorktreeSpanArgs, - fn: () => Promise + fn: (span: ActiveSpan) => Promise ): Promise { return withSpan( `worktree.${meta.stage}`, @@ -214,12 +215,64 @@ export async function withWorktreeSpan( if (meta.path) { span.setAttribute('worktree.path', meta.path) } - return await fn() + return await fn(span) }, { attributes: { kind: 'worktree' } } ) } +type WorktreeCreatePhaseTiming = { + readonly phase: string + readonly startedAtMs: number + readonly durationMs: number +} + +/** Wall-clock span covered by at least one phase. Create runs some phases concurrently, so summing + * durations double-counts and would report overlap as coverage the phases never had. */ +function measuredWallClockMs(phases: readonly WorktreePhaseInterval[]): number { + const intervals = [...phases] + .map((phase) => [phase.startedAtMs, phase.startedAtMs + phase.durationMs] as const) + .sort((left, right) => left[0] - right[0]) + let covered = 0 + let openedAt: number | null = null + let closesAt = 0 + for (const [start, end] of intervals) { + if (openedAt === null) { + openedAt = start + closesAt = end + continue + } + if (start <= closesAt) { + closesAt = Math.max(closesAt, end) + continue + } + covered += closesAt - openedAt + openedAt = start + closesAt = end + } + return openedAt === null ? 0 : covered + (closesAt - openedAt) +} + +type WorktreePhaseInterval = Pick + +/** Records a create's phase breakdown on its span. Phase names are already a closed vocabulary in + * the recorder, so they are safe to key on; nothing here carries a branch name or a path. */ +export function addWorktreeCreatePhaseAttributes( + span: ActiveSpan, + timing: { totalDurationMs: number; phases: readonly WorktreeCreatePhaseTiming[] } +): void { + span.setAttribute('worktree.create.total_ms', Math.round(timing.totalDurationMs)) + for (const phase of timing.phases) { + span.setAttribute(`worktree.create.phase.${phase.phase}_ms`, Math.round(phase.durationMs)) + } + // What the phases do not cover is the number that matters when create feels slow for no visible + // reason, so name it rather than leaving it to subtraction. + span.setAttribute( + 'worktree.create.unattributed_ms', + Math.max(0, Math.round(timing.totalDurationMs - measuredWallClockMs(timing.phases))) + ) +} + /** Closed set so a typo can't silently mint an orphan span name. */ export type WorktreeRemoveStage = | 'archive_hook' diff --git a/src/main/worktree-create-preparation-burst.ts b/src/main/worktree-create-preparation-burst.ts new file mode 100644 index 00000000000..d289927ffaa --- /dev/null +++ b/src/main/worktree-create-preparation-burst.ts @@ -0,0 +1,21 @@ +import { setBoundedMapEntry } from './runtime/runtime-async-boundaries' + +/** Two creates this close together mean more are likely; an isolated create earns no replacement. */ +export const WORKTREE_CREATE_BURST_MS = 5 * 60_000 +const WORKTREE_CREATE_PREPARATION_CONSUME_MAX = 64 + +/** When each preparation key was last consumed, so a burst can be told from an isolated create. */ +const lastConsumedAt = new Map() + +/** Records this consume and reports whether it continues a burst. A replacement checkout costs a + * full tree and holds disk until its TTL, so only a user who is already creating repeatedly earns + * one; the first create of a session pays nothing for a spare nobody claims. */ +export function recordPreparationConsume(key: string, now = Date.now()): boolean { + const previous = lastConsumedAt.get(key) + setBoundedMapEntry(lastConsumedAt, key, now, WORKTREE_CREATE_PREPARATION_CONSUME_MAX) + return previous !== undefined && now - previous <= WORKTREE_CREATE_BURST_MS +} + +export function resetPreparationConsumeHistoryForTests(): void { + lastConsumedAt.clear() +} diff --git a/src/main/worktree-create-preparation-stale-cleanup.ts b/src/main/worktree-create-preparation-stale-cleanup.ts new file mode 100644 index 00000000000..4a422f672ed --- /dev/null +++ b/src/main/worktree-create-preparation-stale-cleanup.ts @@ -0,0 +1,79 @@ +import { + isWorktreeCreatePreparation, + parseWorktreePreparationOwnerPid, + parseWorktreePreparationPathOwnerPid +} from '../shared/worktree/create-preparation' +import type { AddWorktreeOptions } from './git/worktree' +import { listWorktreeGraph } from './git/worktree' +import { discardPreparedWorktree, unlockPreparedWorktree } from './git/worktree-create-preparation' +import { retryPendingPreparationDiscards } from './worktree-preparation-discard-retry' + +const STALE_PREPARATION_CLEANUP_CONCURRENCY = 4 + +const staleCleanupInFlight = new Map>() + +function isProcessAlive(pid: number): boolean { + try { + process.kill(pid, 0) + return true + } catch (error) { + return (error as NodeJS.ErrnoException).code !== 'ESRCH' + } +} + +/** Reclaims preparations a crashed process left registered. Single-flighted per host key so a burst + * of arming calls shares one worktree listing. */ +export async function cleanupStalePreparations( + cleanupKey: string, + repoPath: string, + options: AddWorktreeOptions +): Promise { + const existing = staleCleanupInFlight.get(cleanupKey) + if (existing) { + await existing.catch(() => {}) + return + } + const cleanup = (async () => { + // Not awaited: the create path awaits this cleanup, and one stranded discard costs an unlock plus + // a `worktree remove --force` bounded at 30s each. Reclaiming leaked scratch must not delay create. + void retryPendingPreparationDiscards(cleanupKey) + const worktrees = await listWorktreeGraph(repoPath, { + ...options, + includeCreatePreparations: true + }) + const staleWorktrees = worktrees.filter(isWorktreeCreatePreparation) + let nextIndex = 0 + async function discardNextStalePreparation(): Promise { + while (nextIndex < staleWorktrees.length) { + const worktree = staleWorktrees[nextIndex] + nextIndex += 1 + const lockOwnerPid = parseWorktreePreparationOwnerPid(worktree.lockReason) + const pathOwnerPid = parseWorktreePreparationPathOwnerPid(worktree.path) + if (!lockOwnerPid || isProcessAlive(lockOwnerPid)) { + continue + } + // Preserve a branch-attached final path after a crash; only detached or + // still-hidden preparations are safe to discard automatically. + if (worktree.branch && pathOwnerPid === null) { + await unlockPreparedWorktree(repoPath, worktree.path, options).catch(() => {}) + } else if (pathOwnerPid === lockOwnerPid) { + await discardPreparedWorktree(repoPath, worktree.path, options).catch(() => {}) + } + } + } + const workerCount = Math.min(STALE_PREPARATION_CLEANUP_CONCURRENCY, staleWorktrees.length) + await Promise.all(Array.from({ length: workerCount }, () => discardNextStalePreparation())) + })() + staleCleanupInFlight.set(cleanupKey, cleanup) + try { + await cleanup.catch(() => {}) + } finally { + if (staleCleanupInFlight.get(cleanupKey) === cleanup) { + staleCleanupInFlight.delete(cleanupKey) + } + } +} + +export function resetStalePreparationCleanupForTests(): void { + staleCleanupInFlight.clear() +} diff --git a/src/main/worktree-create-preparation.test.ts b/src/main/worktree-create-preparation.test.ts index 3e03643d6a8..dababbef88d 100644 --- a/src/main/worktree-create-preparation.test.ts +++ b/src/main/worktree-create-preparation.test.ts @@ -466,4 +466,51 @@ describe('worktree create preparation registry', () => { expect(mocks.mkdir).toHaveBeenCalledWith('/workspace', { recursive: true }) expect(mocks.discard).toHaveBeenCalledTimes(1) }) + + function consumeOnce(name: string): ReturnType { + return consumePreparedWorktreeCreate({ + repoPath: repo.path, + workspaceRoot: '/workspace', + worktreePath: `/workspace/${name}`, + branch: `feature/${name}`, + baseBranch: 'origin/main' + }) + } + + it('does not re-arm after an isolated create', async () => { + await prepareWorktreeCreateForRepo(store, repo, 'origin/main') + await expect(consumeOnce('only')).resolves.toEqual({}) + + // Why: a lone create would otherwise leave a full spare checkout on disk for the whole TTL. + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(1) + }) + + it('re-arms a preparation once creates arrive in a burst', async () => { + await prepareWorktreeCreateForRepo(store, repo, 'origin/main') + await expect(consumeOnce('first')).resolves.toEqual({}) + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(1) + + await prepareWorktreeCreateForRepo(store, repo, 'origin/main') + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(2) + + // No arming call follows this consume: the third checkout can only come from the re-arm. + await expect(consumeOnce('second')).resolves.toEqual({}) + expect(mocks.prepareCheckout).toHaveBeenCalledTimes(3) + + // The replacement is claimable, so a third create still skips the cold add. + await expect(consumeOnce('third')).resolves.toEqual({}) + expect(mocks.finalize).toHaveBeenCalledTimes(3) + }) + + it('does not re-arm when finalization failed', async () => { + await prepareWorktreeCreateForRepo(store, repo, 'origin/main') + await expect(consumeOnce('first')).resolves.toEqual({}) + await prepareWorktreeCreateForRepo(store, repo, 'origin/main') + mocks.prepareCheckout.mockClear() + mocks.finalize.mockRejectedValueOnce(new Error('submodules prevent worktree move')) + + await expect(consumeOnce('second')).resolves.toBeNull() + + expect(mocks.prepareCheckout).not.toHaveBeenCalled() + }) }) diff --git a/src/main/worktree-create-preparation.ts b/src/main/worktree-create-preparation.ts index ff7478cb46e..22bcf5d1e2c 100644 --- a/src/main/worktree-create-preparation.ts +++ b/src/main/worktree-create-preparation.ts @@ -7,17 +7,12 @@ import { isFolderRepo } from '../shared/repo-kind' import { isWindowsAbsolutePathLike } from '../shared/cross-platform-path' import { WORKTREE_CREATE_PREPARATION_DIRECTORY, - createWorktreePreparationLockReason, - isWorktreeCreatePreparation, - parseWorktreePreparationOwnerPid, - parseWorktreePreparationPathOwnerPid + createWorktreePreparationLockReason } from '../shared/worktree/create-preparation' import type { AddWorktreeOptions, AddWorktreeResult } from './git/worktree' -import { listWorktreeGraph } from './git/worktree' import { discardPreparedWorktree, finalizePreparedWorktree, - unlockPreparedWorktree, prepareWorktreeCreateCheckout } from './git/worktree-create-preparation' import { @@ -25,17 +20,23 @@ import { getWorktreeMirrorDistro } from './project-runtime-git-options' import { computeWorkspaceRootAsync, getWorktreePathSettings } from './ipc/worktree-logic' +import { + recordPreparationConsume, + resetPreparationConsumeHistoryForTests +} from './worktree-create-preparation-burst' +import { + cleanupStalePreparations, + resetStalePreparationCleanupForTests +} from './worktree-create-preparation-stale-cleanup' import { toHostFilesystemPath } from './host-tree-removal' import { discardPreparationWithRetry, resetPendingPreparationDiscardsForTests, - retryPendingPreparationDiscards, trackPreparationDiscard } from './worktree-preparation-discard-retry' export const WORKTREE_CREATE_PREPARATION_TTL_MS = 5 * 60_000 export const WORKTREE_CREATE_PREPARATION_LIMIT = 3 -const STALE_PREPARATION_CLEANUP_CONCURRENCY = 4 type PreparationEntry = { key: string @@ -59,7 +60,6 @@ type ConsumePreparedWorktreeArgs = { } const preparations = new Map() -const staleCleanupInFlight = new Map>() function pathOps(path: string): Pick { return isWindowsAbsolutePathLike(path) ? win32 : posix @@ -79,15 +79,6 @@ function preparationKey( return `${pathKey(repoPath)}\0${pathKey(workspaceRoot)}\0${baseBranch}\0${options.wslDistro ?? ''}` } -function isProcessAlive(pid: number): boolean { - try { - process.kill(pid, 0) - return true - } catch (error) { - return (error as NodeJS.ErrnoException).code !== 'ESRCH' - } -} - function preparationHostKey(repoPath: string, options: AddWorktreeOptions): string { return `${pathKey(repoPath)}\0${options.wslDistro ?? ''}` } @@ -131,57 +122,6 @@ function enforcePreparationLimit(): void { } } -async function cleanupStalePreparations( - repoPath: string, - options: AddWorktreeOptions -): Promise { - const cleanupKey = preparationHostKey(repoPath, options) - const existing = staleCleanupInFlight.get(cleanupKey) - if (existing) { - await existing.catch(() => {}) - return - } - const cleanup = (async () => { - // Not awaited: the create path awaits this cleanup, and one stranded discard costs an unlock plus - // a `worktree remove --force` bounded at 30s each. Reclaiming leaked scratch must not delay create. - void retryPendingPreparationDiscards(cleanupKey) - const worktrees = await listWorktreeGraph(repoPath, { - ...options, - includeCreatePreparations: true - }) - const staleWorktrees = worktrees.filter(isWorktreeCreatePreparation) - let nextIndex = 0 - async function discardNextStalePreparation(): Promise { - while (nextIndex < staleWorktrees.length) { - const worktree = staleWorktrees[nextIndex] - nextIndex += 1 - const lockOwnerPid = parseWorktreePreparationOwnerPid(worktree.lockReason) - const pathOwnerPid = parseWorktreePreparationPathOwnerPid(worktree.path) - if (!lockOwnerPid || isProcessAlive(lockOwnerPid)) { - continue - } - // Preserve a branch-attached final path after a crash; only detached or - // still-hidden preparations are safe to discard automatically. - if (worktree.branch && pathOwnerPid === null) { - await unlockPreparedWorktree(repoPath, worktree.path, options).catch(() => {}) - } else if (pathOwnerPid === lockOwnerPid) { - await discardPreparedWorktree(repoPath, worktree.path, options).catch(() => {}) - } - } - } - const workerCount = Math.min(STALE_PREPARATION_CLEANUP_CONCURRENCY, staleWorktrees.length) - await Promise.all(Array.from({ length: workerCount }, () => discardNextStalePreparation())) - })() - staleCleanupInFlight.set(cleanupKey, cleanup) - try { - await cleanup.catch(() => {}) - } finally { - if (staleCleanupInFlight.get(cleanupKey) === cleanup) { - staleCleanupInFlight.delete(cleanupKey) - } - } -} - export async function prepareWorktreeCreateForRepo( store: Store, repo: Repo, @@ -205,6 +145,16 @@ export async function prepareWorktreeCreateForRepo( return existing.ready } + return startPreparation(key, repo.path, workspaceRoot, baseBranch, options) +} + +function startPreparation( + key: string, + repoPath: string, + workspaceRoot: string, + baseBranch: string, + options: AddWorktreeOptions +): Promise { enforcePreparationLimit() const preparationId = `${process.pid}-${randomUUID()}` const lockReason = createWorktreePreparationLockReason(preparationId) @@ -218,21 +168,21 @@ export async function prepareWorktreeCreateForRepo( expiration.unref() Object.assign(entry, { key, - repoPath: repo.path, + repoPath, workspaceRoot, preparedPath, options, createdAt: Date.now(), expiration, ready: (async () => { - await cleanupStalePreparations(repo.path, options) + await cleanupStalePreparations(preparationHostKey(repoPath, options), repoPath, options) await mkdir( toHostFilesystemPath( pathOps(workspaceRoot).join(workspaceRoot, WORKTREE_CREATE_PREPARATION_DIRECTORY) ), { recursive: true } ) - await prepareWorktreeCreateCheckout(repo.path, preparedPath, baseBranch, lockReason, options) + await prepareWorktreeCreateCheckout(repoPath, preparedPath, baseBranch, lockReason, options) })() } satisfies PreparationEntry) preparations.set(key, entry) @@ -266,6 +216,28 @@ async function claimPreparedWorktree( } } +/** Replaces a just-consumed preparation, but only once the user has shown they are creating in a + * burst. A replacement costs a full checkout and ~5 minutes of disk until its TTL, so arming one + * after an isolated create spends that on nobody. Never awaited: create has already returned by + * the time the replacement checkout finishes. */ +function rearmPreparation(entry: PreparationEntry, baseBranch: string): void { + // Record first: a prefetch that re-armed this key while we finalized would otherwise swallow the + // consume, and the next create would look isolated when it is really the middle of a burst. + const continuesBurst = recordPreparationConsume(entry.key) + if (preparations.has(entry.key) || !continuesBurst) { + return + } + void startPreparation( + entry.key, + entry.repoPath, + entry.workspaceRoot, + baseBranch, + entry.options + ).catch(() => { + // Why: a warm-up failure is recovered by the normal add on the next create. + }) +} + export async function consumePreparedWorktreeCreate( args: ConsumePreparedWorktreeArgs ): Promise { @@ -283,7 +255,7 @@ export async function consumePreparedWorktreeCreate( await mkdir(toHostFilesystemPath(pathOps(args.worktreePath).dirname(args.worktreePath)), { recursive: true }) - return await finalizePreparedWorktree( + const result = await finalizePreparedWorktree( args.repoPath, entry.preparedPath, args.worktreePath, @@ -292,6 +264,10 @@ export async function consumePreparedWorktreeCreate( args.refreshLocalBaseRef, options ) + // Consuming the only prepared checkout leaves the next create cold. Re-arm for a user who is + // creating in a burst; the TTL and the preparation limit still bound an unused replacement. + rearmPreparation(entry, args.baseBranch) + return result } catch (error) { await discardPreparedWorktree(args.repoPath, entry.preparedPath, options).catch(() => {}) console.warn( @@ -305,7 +281,8 @@ export async function consumePreparedWorktreeCreate( export async function _resetWorktreeCreatePreparationsForTests(): Promise { const entries = [...preparations.values()] preparations.clear() - staleCleanupInFlight.clear() + resetPreparationConsumeHistoryForTests() + resetStalePreparationCleanupForTests() await Promise.all( entries.map(async (entry) => { clearTimeout(entry.expiration) From 8b7d778a2ef5e3c0a17434b564486e8aa61138de Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:04:03 -0700 Subject: [PATCH 006/398] perf(git-common): bound the fs-stat fan-out in the worktree pollers (#17839) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(git-common): bound the fs-stat fan-out in the worktree pollers snapshotGitCommon and snapshotBase issued one fs op per candidate via Promise.all/a serial loop, unbounded by worktree count. At 973 live worktrees this queued ~6,800 concurrent stat calls (measured peak 6000 in a 1000-entry synthetic benchmark) onto libuv's 4-thread default pool, starving every other main-process fs operation for the scan's duration (~1s). Bound both to concurrency 8 via the existing forEachWithConcurrency helper, matching the precedent in exact-ref-probe.ts and worktree-head-identity-reader.ts. Peak concurrent stats dropped 6000 -> 48 in the benchmark; wall time was essentially unchanged (495ms -> 541ms), since the real bottleneck was never total scan time but pool starvation of unrelated work. Also make the no-native-watch and crash-fuse polling fallbacks in worktree-git-common-watch.ts / worktree-git-common-narrow-watch.ts self-calibrate their cadence: on platforms/paths where this poller is the sole change signal, a fixed 2s cadence at hundreds of worktrees approaches a permanent scan loop. Stretch the interval so a scan stays a bounded fraction (10%) of its own cadence, capped at 30s, floored at the configured base interval. Left the reconciliation backstop (fixed 30s cadence, already accepted) and checkPendingMarkers (bounded by concurrent-worktree-creation count, not total count) untouched. Fixes #17828 * perf(git-common): split the tripwire from the per-entry sweep cadence Review on #17839 found a real staleness trade-off: adaptiveCadence gated ALL detection (worktree add/remove, HEAD, dirty refs, AND per-entry commit signals) behind one stretched interval, so on the crash-fuse polling fallback the reviewer measured cadence sitting at 5.4-10s sustained and hitting the 30s cap once a single scan reached 3s at 973 worktrees -- worse than the pre-#17828 fixed ~2s+250ms baseline for signals users notice immediately (sidebar worktree list, branch labels). Split snapshotGitCommon into a cheap structural "tripwire" (readdir, worktreesDir signature, primary-file signatures, newly-appeared entries -- ~5-6 fs ops, O(1) in worktree count) that always runs on the fixed pollIntervalMs, and the O(n) per-entry sweep (commit/dirty detection) that alone is gated by the adaptive cadence via a nextSweepDueAt deadline. Existing, unchanged entries are carried over by reference on a tripwire-only tick (no re-stat), so diffing produces no spurious events; genuinely new entries are still stat'd immediately so worktree add remains real-time. This keeps everything on one ticking-flag-guarded loop (no new concurrency/race surface) -- scheduling stays fixed at pollIntervalMs; only nextSweepDueAt stretches. Also drop the adaptive-cadence seed heuristic entirely: nextSweepDueAt starts at 0, so the first regular tick after bootstrap sweeps unconditionally on its own schedule instead of guessing an initial interval from the bootstrap snapshot's duration (which could stretch the very first tick to 10-30s on a slow disk). Documented that worktree-git-common-watch.ts's adaptiveCadence call site is unreachable in production (Electron only ships darwin/linux/win32, both covered by NARROW_WATCH_PLATFORMS) rather than implying it protects real users. The reachable path is the narrow-watch crash-fuse fallback in worktree-git-common-narrow-watch.ts. Filed #17878 to track the real long-term fix: periodically retrying the upgrade back to the narrow watch after a crash-fuse trip, so the degraded/polling state doesn't need to be tuned at all once the underlying failure clears. * perf(git-common): gate per-entry structural stats on the entry-dir signature Every real git write inside a worktree admin entry (HEAD, index, config.worktree, locked) goes through a lock file + rename, which moves the entry directory's own mtime/ctime/size signature. Only `gitdir` (worktree move/repair) is rewritten in place, and that's already covered by the periodic ungated backstop (INDEX_BACKSTOP_TICKS). The previous comment claiming structural leaves "change in place every tick" was wrong; verified against git 2.55 across checkout, commit, amend, reset, ref updates, stash, worktree lock/unlock, config --worktree, and index writes. Gate all six per-entry stats behind the entry dir's own signature instead of stat-ing every leaf unconditionally every tick: an unchanged entry now costs one stat per tick instead of six, and a changed one still costs six (bounded by change rate, not worktree count). This also fixes the actual in-flight fan-out: forEachWithConcurrency(entries, 8) previously still issued 6 stats per in-flight entry (48 real concurrent ops); with the gate, warm ticks issue ~1 stat per entry, so true in-flight tracks the concurrency limit directly. This makes the follow-up adaptive-cadence machinery from the prior commit unnecessary: the crash-fuse and no-narrow-watch polling fallbacks no longer need to stretch their own cadence, since a warm sweep across hundreds of worktrees is now cheap regardless of interval. Revert both call sites to a fixed pollIntervalMs and delete the adaptive-cadence option, the split tripwire/sweep cadence, and the seed heuristic — none of it earns its complexity once the real per-entry cost is fixed at the source. Per-entry staleness on the crash-fuse path returns to a fixed 2s + 250ms debounce instead of the previous 5.4-30s adaptive stretch. Refs #17828 --- ...ase-directory-poller-marker-fanout.test.ts | 84 +++++++ .../ipc/worktree-base-directory-poller.ts | 37 +-- .../ipc/worktree-git-common-entry-snapshot.ts | 42 ++-- .../ipc/worktree-git-common-narrow-watch.ts | 5 + .../ipc/worktree-git-common-polling.test.ts | 237 ++++++++++++++++++ src/main/ipc/worktree-git-common-polling.ts | 35 +-- src/main/ipc/worktree-git-common-watch.ts | 2 + 7 files changed, 395 insertions(+), 47 deletions(-) create mode 100644 src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts create mode 100644 src/main/ipc/worktree-git-common-polling.test.ts diff --git a/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts b/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts new file mode 100644 index 00000000000..023a8390ccd --- /dev/null +++ b/src/main/ipc/worktree-base-directory-poller-marker-fanout.test.ts @@ -0,0 +1,84 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { mkdir, mkdtemp, realpath, rm, writeFile } from 'node:fs/promises' +import type * as NodeFsPromises from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { startWorktreeBaseDirectoryPoller } from './worktree-base-directory-poller' +import type { + WorktreeBaseRepoWatchConfig, + WorktreeBaseWatchTarget +} from './worktree-base-directory-event-filter' + +// Why: the backstop full scan stats a `.git` marker per candidate dir; an +// unbounded fan-out at hundreds of worktrees would queue thousands of `stat` +// calls on libuv's 4-thread pool (#17828). +const { concurrency } = vi.hoisted(() => ({ concurrency: { current: 0, peak: 0 } })) + +vi.mock('node:fs/promises', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + stat: async (...args: Parameters) => { + concurrency.current += 1 + concurrency.peak = Math.max(concurrency.peak, concurrency.current) + try { + return await actual.stat(...args) + } finally { + concurrency.current -= 1 + } + } + } +}) + +function makeTarget(path: string): WorktreeBaseWatchTarget { + const repoConfig: WorktreeBaseRepoWatchConfig = { + repoId: 'repo-1', + repoName: 'project', + nestWorkspaces: false + } + return { + key: `base:local:${path}`, + kind: 'base', + path, + repos: new Map([[repoConfig.repoId, repoConfig]]) + } +} + +describe('worktree base directory poller marker fan-out (#17828)', () => { + const cleanups: (() => Promise)[] = [] + + beforeEach(() => { + concurrency.current = 0 + concurrency.peak = 0 + }) + + afterEach(async () => { + await Promise.all(cleanups.splice(0).map((cleanup) => cleanup())) + }) + + it('bounds concurrent `.git`-marker stats regardless of candidate count', async () => { + const root = await realpath(await mkdtemp(join(tmpdir(), 'orca-base-poller-fanout-'))) + cleanups.push(() => rm(root, { recursive: true, force: true })) + const candidateCount = 200 + for (let i = 0; i < candidateCount; i++) { + const worktree = join(root, `wt-${i}`) + await mkdir(worktree) + await writeFile(join(worktree, '.git'), 'gitdir: elsewhere') + } + + const target = makeTarget(root) + const poller = await startWorktreeBaseDirectoryPoller( + target, + () => target.repos, + () => {}, + { pollIntervalMs: 100_000 } + ) + cleanups.push(() => poller.unsubscribe()) + + // 200 candidates stated unbounded would peak near 200 concurrent `stat` + // calls; bounding the marker probe keeps the peak independent of count — + // while still overlapping requests (not serialized one-at-a-time). + expect(concurrency.peak).toBeGreaterThan(1) + expect(concurrency.peak).toBeLessThan(20) + }) +}) diff --git a/src/main/ipc/worktree-base-directory-poller.ts b/src/main/ipc/worktree-base-directory-poller.ts index 42ee184b811..201d9782fba 100644 --- a/src/main/ipc/worktree-base-directory-poller.ts +++ b/src/main/ipc/worktree-base-directory-poller.ts @@ -1,6 +1,8 @@ import { readdir, stat } from 'node:fs/promises' +import type { Dirent } from 'node:fs' import { join } from 'node:path' import { normalizeRuntimePathForComparison } from '../../shared/cross-platform-path' +import { forEachWithConcurrency } from '../../shared/map-with-concurrency' import { isMainWindowVisible, onMainWindowBecameVisible } from '../window/main-window-visibility' import type { WorktreeBaseRepoWatchConfig, @@ -92,6 +94,11 @@ export const WORKTREE_BASE_BACKSTOP_TICKS = 15 // backstop scan cover the pathological case. const PENDING_MARKER_MAX_TICKS = 300 +// Why: matches the git-common poller's fan-out bound (#17828) — bounded +// concurrency turns hundreds of serial round trips into a handful of batches +// without dumping every candidate onto libuv's 4-thread pool at once. +const MARKER_PROBE_CONCURRENCY = 8 + function statSignature(s: { mtimeMs: number; ctimeMs: number; ino: number }): string { return `${s.mtimeMs}:${s.ctimeMs}:${s.ino}` } @@ -123,6 +130,14 @@ type BaseSnapshot = { gateSignatures: string[] } +async function readdirSafe(path: string): Promise { + try { + return await readdir(path, { withFileTypes: true }) + } catch { + return [] + } +} + // Depth-1 worktree dirs (flat layout), plus depth-2 dirs under each nested // repo's container, mirroring what worktree-base-directory-event-filter // matches: `/.git` completion markers and `` deletions. @@ -144,14 +159,9 @@ async function snapshotBase( .map((config) => normalizeRuntimePathForComparison(config.repoName)) ) - let rootEntries - try { - rootEntries = await readdir(rootPath, { withFileTypes: true }) - } catch { - // Root vanished: an empty snapshot diffs into delete events for every - // previously-known worktree dir, matching the old watcher's error path. - return { markers, gateDirs, gateSignatures } - } + // Root vanished or unreadable: readdirSafe yields [], producing the same + // empty markers/candidates result as the old watcher's error path. + const rootEntries = await readdirSafe(rootPath) const candidates: string[] = [] for (const entry of rootEntries) { @@ -165,12 +175,7 @@ async function snapshotBase( if (nestedRepoNames.has(normalizeRuntimePathForComparison(entry.name))) { gateDirs.push(entryPath) gateSignatures.push(await dirSignature(entryPath)) - let subEntries - try { - subEntries = await readdir(entryPath, { withFileTypes: true }) - } catch { - subEntries = [] - } + const subEntries = await readdirSafe(entryPath) for (const sub of subEntries) { if (sub.isDirectory() || sub.isSymbolicLink()) { candidates.push(join(entryPath, sub.name)) @@ -179,9 +184,9 @@ async function snapshotBase( } } - for (const dir of candidates) { + await forEachWithConcurrency(candidates, MARKER_PROBE_CONCURRENCY, async (dir) => { markers.set(dir, await hasGitMarker(dir)) - } + }) return { markers, gateDirs, gateSignatures } } diff --git a/src/main/ipc/worktree-git-common-entry-snapshot.ts b/src/main/ipc/worktree-git-common-entry-snapshot.ts index d14f4ec8a1d..6dad6e0a048 100644 --- a/src/main/ipc/worktree-git-common-entry-snapshot.ts +++ b/src/main/ipc/worktree-git-common-entry-snapshot.ts @@ -39,11 +39,33 @@ export async function snapshotGitCommonEntry( previous: GitCommonEntrySnapshot | undefined, forceFullScan: boolean ): Promise { - // Structural leaves change in place every tick; only index uses the entry-dir gate. + // Git writes HEAD/index/config.worktree/locked via a lock file + rename inside the + // entry dir, so the entry dir's own signature moves on every one of those writes + // (verified against git 2.55: checkout, commit, amend, reset, ref updates, stash, + // worktree lock/unlock, config --worktree, index writes all move it). The one + // in-place exception is `gitdir` (worktree move/repair), which the periodic + // forceFullScan backstop (INDEX_BACKSTOP_TICKS) below re-stats regardless of this + // gate. Gating all of these leaves on the entry-dir signature turns an unchanged + // entry into a single stat per tick instead of stat-ing every leaf every tick. + const nextDirSignature = await gitCommonDirectorySignature(entryPath) + if (nextDirSignature === 'missing') { + return ( + previous ?? { + dirSignature: nextDirSignature, + structuralSignatures: new Map(), + indexSignature: null, + headLogSignature: null + } + ) + } + const shouldRescan = forceFullScan || !previous || previous.dirSignature !== nextDirSignature + if (!shouldRescan) { + return previous + } const structuralSignatures = new Map() - const [nextDirSignature, headLogSignature] = await Promise.all([ - gitCommonDirectorySignature(entryPath), + const [headLogSignature, indexSignature] = await Promise.all([ gitCommonFileSignature(join(entryPath, HEAD_LOG_FILE)), + gitCommonFileSignature(join(entryPath, INDEX_FILE)), Promise.all( STRUCTURAL_METADATA_FILES.map(async (name) => { const signature = await gitCommonFileSignature(join(entryPath, name)) @@ -53,20 +75,6 @@ export async function snapshotGitCommonEntry( }) ) ]) - if (nextDirSignature === 'missing') { - return ( - previous ?? { - dirSignature: nextDirSignature, - structuralSignatures, - indexSignature: null, - headLogSignature - } - ) - } - const shouldReadIndex = forceFullScan || !previous || previous.dirSignature !== nextDirSignature - const indexSignature = shouldReadIndex - ? await gitCommonFileSignature(join(entryPath, INDEX_FILE)) - : previous.indexSignature return { dirSignature: nextDirSignature, structuralSignatures, diff --git a/src/main/ipc/worktree-git-common-narrow-watch.ts b/src/main/ipc/worktree-git-common-narrow-watch.ts index b60ac9877e8..99e245998a1 100644 --- a/src/main/ipc/worktree-git-common-narrow-watch.ts +++ b/src/main/ipc/worktree-git-common-narrow-watch.ts @@ -73,6 +73,11 @@ export async function startGitCommonNarrowWatch( .unsubscribe() .catch(() => {}) .then(() => + // Crash fuse tripped: this poller is now the sole change signal until a + // future existence-poll upgrade (follow-up: #17878). Its own per-entry + // dir-signature gate (worktree-git-common-entry-snapshot.ts) already keeps + // an unchanged entry to a single stat, so a fixed `pollIntervalMs` cadence + // stays cheap at high worktree counts without needing to stretch itself. startGitCommonPolling( target.path, onEvents, diff --git a/src/main/ipc/worktree-git-common-polling.test.ts b/src/main/ipc/worktree-git-common-polling.test.ts new file mode 100644 index 00000000000..179b73e2c33 --- /dev/null +++ b/src/main/ipc/worktree-git-common-polling.test.ts @@ -0,0 +1,237 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { mkdir, mkdtemp, rename, rm, writeFile } from 'node:fs/promises' +import type * as NodeFsPromises from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join, sep } from 'node:path' +import { startGitCommonPolling } from './worktree-git-common-polling' +import type { + WorktreeBasePollEvent, + WorktreePollerWindowVisibility +} from './worktree-base-directory-poller' + +// Why: measure the fan-out this poller issues per scan (peak concurrent `stat` +// calls, `readdir` call count as a proxy for "a tick ran") without depending on +// real disk timing (#17828). `entryZeroStatCalls` tracks every stat under a +// specific pre-existing entry (its dir plus every leaf), used to prove the +// entry-dir signature gate keeps an unchanged entry to one stat per tick. +const { statDelayMs, readdirCalls, concurrency, entryZeroStatCalls } = vi.hoisted(() => ({ + statDelayMs: { current: 0 }, + readdirCalls: { count: 0 }, + concurrency: { current: 0, peak: 0 }, + entryZeroStatCalls: { count: 0 } +})) + +vi.mock('node:fs/promises', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + readdir: (...args: Parameters) => { + readdirCalls.count += 1 + return actual.readdir(...args) + }, + stat: async (...args: Parameters) => { + concurrency.current += 1 + concurrency.peak = Math.max(concurrency.peak, concurrency.current) + const path = args[0] + const entryZeroSegment = `${sep}wt-0` + if ( + typeof path === 'string' && + (path.endsWith(entryZeroSegment) || path.includes(`${entryZeroSegment}${sep}`)) + ) { + entryZeroStatCalls.count += 1 + } + try { + if (statDelayMs.current > 0) { + await new Promise((resolve) => setTimeout(resolve, statDelayMs.current)) + } + return await actual.stat(...args) + } finally { + concurrency.current -= 1 + } + } + } +}) + +const alwaysVisible: WorktreePollerWindowVisibility = { + isWindowVisible: () => true, + onWindowBecameVisible: () => () => {} +} + +async function makeCommonDir(entryCount: number): Promise { + const root = await mkdtemp(join(tmpdir(), 'git-common-polling-test-')) + for (let i = 0; i < entryCount; i++) { + const entryPath = join(root, 'worktrees', `wt-${i}`) + await mkdir(join(entryPath, 'logs'), { recursive: true }) + await Promise.all([ + writeFile(join(entryPath, 'HEAD'), 'ref: refs/heads/main\n'), + writeFile(join(entryPath, 'gitdir'), `${join(root, `checkout-${i}`, '.git')}\n`), + writeFile(join(entryPath, 'index'), Buffer.from([0])), + writeFile(join(entryPath, 'logs', 'HEAD'), '0000 aaaa\n') + ]) + } + return root +} + +describe('startGitCommonPolling fan-out bounds (#17828)', () => { + const cleanups: (() => Promise)[] = [] + const dirsToRemove: string[] = [] + + beforeEach(() => { + statDelayMs.current = 0 + readdirCalls.count = 0 + concurrency.current = 0 + concurrency.peak = 0 + entryZeroStatCalls.count = 0 + }) + + afterEach(async () => { + await Promise.all(cleanups.splice(0).map((cleanup) => cleanup())) + await Promise.all( + dirsToRemove.splice(0).map((dir) => rm(dir, { recursive: true, force: true })) + ) + vi.useRealTimers() + }) + + it('bounds concurrent per-entry stat fan-out regardless of entry count', async () => { + const commonDir = await makeCommonDir(200) + dirsToRemove.push(commonDir) + const sub = await startGitCommonPolling(commonDir, () => {}, 100_000, alwaysVisible) + cleanups.push(() => sub.unsubscribe()) + // 200 entries x ~6 concurrent structural stats each would peak near 1,200 + // unbounded; bounding to 8 in-flight entries keeps the peak independent of + // entry count instead of scaling with it. + expect(concurrency.peak).toBeLessThan(80) + }) + + it('never overlaps a scan with itself even when ticks fire faster than a scan completes', async () => { + const commonDir = await makeCommonDir(10) + dirsToRemove.push(commonDir) + statDelayMs.current = 20 + const pollIntervalMs = 5 + const sub = await startGitCommonPolling(commonDir, () => {}, pollIntervalMs, alwaysVisible) + cleanups.push(() => sub.unsubscribe()) + readdirCalls.count = 0 + // ~60 would-be 5ms ticks elapse in this window while every stat takes 20ms; + // the ticking guard must serialize scans, not launch overlapping ones. + await new Promise((resolve) => setTimeout(resolve, 300)) + expect(readdirCalls.count).toBeLessThan(10) + }) + + it('costs exactly one stat per tick for an unchanged entry', async () => { + const commonDir = await makeCommonDir(1) + dirsToRemove.push(commonDir) + const pollIntervalMs = 20 + const sub = await startGitCommonPolling(commonDir, () => {}, pollIntervalMs, alwaysVisible) + cleanups.push(() => sub.unsubscribe()) + + // Let the bootstrap snapshot (which always fully reads every entry once) settle. + await new Promise((resolve) => setTimeout(resolve, pollIntervalMs)) + readdirCalls.count = 0 + entryZeroStatCalls.count = 0 + await vi.waitFor( + () => { + expect(readdirCalls.count).toBeGreaterThanOrEqual(5) + }, + { timeout: 2_000 } + ) + // Without the entry-dir signature gate, an unchanged entry still costs ~6 + // stats every tick (HEAD/gitdir/locked/config.worktree/logs/HEAD/index). + // With the gate, only the entry dir itself is stat'd once nothing changed — + // one stat per tick, in lockstep with the readdir tripwire. + expect(entryZeroStatCalls.count).toBeLessThanOrEqual(readdirCalls.count + 1) + expect(entryZeroStatCalls.count).toBeGreaterThanOrEqual(readdirCalls.count - 1) + }) + + it('detects a HEAD rewrite via lock+rename on the next tick', async () => { + const commonDir = await makeCommonDir(1) + dirsToRemove.push(commonDir) + const events: WorktreeBasePollEvent[][] = [] + const pollIntervalMs = 20 + const sub = await startGitCommonPolling( + commonDir, + (batch) => events.push(batch), + pollIntervalMs, + alwaysVisible + ) + cleanups.push(() => sub.unsubscribe()) + // Let the bootstrap snapshot settle before mutating. + await new Promise((resolve) => setTimeout(resolve, pollIntervalMs)) + + const entryDir = join(commonDir, 'worktrees', 'wt-0') + const headPath = join(entryDir, 'HEAD') + const headLockPath = join(entryDir, 'HEAD.lock') + // Every real git ref write goes through a lock file + rename inside the entry + // dir (never an in-place overwrite), which moves the entry dir's own signature. + await writeFile(headLockPath, 'ref: refs/heads/feature\n') + await rename(headLockPath, headPath) + + await vi.waitFor( + () => { + expect(events.flat()).toContainEqual({ type: 'update', path: headPath }) + }, + { timeout: pollIntervalMs * 10 } + ) + }) + + it('detects an in-place gitdir rewrite only once the periodic backstop rescans it', async () => { + const commonDir = await makeCommonDir(1) + dirsToRemove.push(commonDir) + const events: WorktreeBasePollEvent[][] = [] + const pollIntervalMs = 10 + const sub = await startGitCommonPolling( + commonDir, + (batch) => events.push(batch), + pollIntervalMs, + alwaysVisible + ) + cleanups.push(() => sub.unsubscribe()) + // Let the bootstrap snapshot settle before mutating. + await new Promise((resolve) => setTimeout(resolve, pollIntervalMs)) + + const entryDir = join(commonDir, 'worktrees', 'wt-0') + const gitdirPath = join(entryDir, 'gitdir') + // `gitdir` is the one structural leaf git rewrites in place (worktree move/repair), + // so the entry dir's own signature never moves — the periodic ungated backstop + // (INDEX_BACKSTOP_TICKS = 15) is the only thing that catches it. + await writeFile(gitdirPath, `${join(commonDir, 'checkout-moved', '.git')}\n`) + + // Not caught by the next several ticks: the gate stays closed since nothing + // moved the entry dir's own signature. + await new Promise((resolve) => setTimeout(resolve, pollIntervalMs * 5)) + expect(events.flat()).not.toContainEqual({ type: 'update', path: gitdirPath }) + + // Eventually caught regardless of the gate, once tick 15 forces the periodic backstop. + await vi.waitFor( + () => { + expect(events.flat()).toContainEqual({ type: 'update', path: gitdirPath }) + }, + { timeout: pollIntervalMs * 40 } + ) + }) + + it('still detects entry add/remove correctly with bounded concurrency', async () => { + const commonDir = await makeCommonDir(5) + dirsToRemove.push(commonDir) + const events: WorktreeBasePollEvent[][] = [] + const sub = await startGitCommonPolling( + commonDir, + (batch) => events.push(batch), + 20, + alwaysVisible + ) + cleanups.push(() => sub.unsubscribe()) + + const newEntry = join(commonDir, 'worktrees', 'wt-new') + await mkdir(join(newEntry, 'logs'), { recursive: true }) + await writeFile(join(newEntry, 'HEAD'), 'ref: refs/heads/main\n') + + await vi.waitFor(() => { + expect(events.flat()).toContainEqual({ type: 'create', path: newEntry }) + }) + + await rm(newEntry, { recursive: true }) + await vi.waitFor(() => { + expect(events.flat()).toContainEqual({ type: 'delete', path: newEntry }) + }) + }) +}) diff --git a/src/main/ipc/worktree-git-common-polling.ts b/src/main/ipc/worktree-git-common-polling.ts index 0418f4a90fd..4b43835a81f 100644 --- a/src/main/ipc/worktree-git-common-polling.ts +++ b/src/main/ipc/worktree-git-common-polling.ts @@ -1,5 +1,6 @@ import { readdir } from 'node:fs/promises' import { join } from 'node:path' +import { forEachWithConcurrency } from '../../shared/map-with-concurrency' import { PRIMARY_CHECKOUT_METADATA_FILES } from './worktree-git-common-metadata-files' import { diffGitCommon, @@ -23,18 +24,26 @@ import { // same way the base poller's backstop rescan does. const INDEX_BACKSTOP_TICKS = 15 +// Why: an unbounded fan-out across every worktree admin entry queues thousands +// of ops on libuv's 4-thread default pool, starving every other main-process +// fs call for the scan's duration (#17828). 8 mirrors the existing +// head-identity/exact-ref-probe pools — enough to saturate typical local +// disks without monopolizing the pool. Since snapshotGitCommonEntry's own +// entry-dir gate (see worktree-git-common-entry-snapshot.ts) keeps most ticks +// down to 1 stat per unchanged entry, real in-flight is now bounded by this +// limit rather than limit × per-entry stat count. +const GIT_COMMON_SNAPSHOT_CONCURRENCY = 8 + async function snapshotStatusRefSignatures( paths: ReadonlySet ): Promise> { const signatures = new Map() - await Promise.all( - [...paths].map(async (path) => { - const signature = await gitCommonFileSignature(path) - if (signature !== null) { - signatures.set(path, signature) - } - }) - ) + await forEachWithConcurrency([...paths], GIT_COMMON_SNAPSHOT_CONCURRENCY, async (path) => { + const signature = await gitCommonFileSignature(path) + if (signature !== null) { + signatures.set(path, signature) + } + }) return signatures } @@ -91,12 +100,10 @@ async function snapshotGitCommon( } const entries = new Map() - await Promise.all( - entryPaths.map(async (entryPath) => { - const previousEntry = previous?.entries.get(entryPath) - entries.set(entryPath, await snapshotGitCommonEntry(entryPath, previousEntry, forceFullScan)) - }) - ) + await forEachWithConcurrency(entryPaths, GIT_COMMON_SNAPSHOT_CONCURRENCY, async (entryPath) => { + const previousEntry = previous?.entries.get(entryPath) + entries.set(entryPath, await snapshotGitCommonEntry(entryPath, previousEntry, forceFullScan)) + }) // Why: the expensive per-entry `index` read stays gated on each entry's own dir signature; onFullScan // now reflects an ungated index-metadata backstop fan-out (forceFullScan) — the real periodic cost — // rather than the always-run worktrees-dir readdir. diff --git a/src/main/ipc/worktree-git-common-watch.ts b/src/main/ipc/worktree-git-common-watch.ts index 9de2c8c3392..8ed696872f7 100644 --- a/src/main/ipc/worktree-git-common-watch.ts +++ b/src/main/ipc/worktree-git-common-watch.ts @@ -60,6 +60,8 @@ export async function startGitCommonWatch( } } } + // Why: Electron only ships darwin/linux/win32, all covered by NARROW_WATCH_PLATFORMS + // above, so this branch is defensive dead code in production, not a reachable fallback. return startGitCommonPolling( target.path, onEvents, From fdfe354045680a01b3743600362fd37262031af4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 18:20:47 -0700 Subject: [PATCH 007/398] test(relay): bind test WebSocket servers to loopback A control-handshake test that expects a timeout was instead getting 'Unexpected server response: 401' about once in fourteen runs. A slow machine cannot turn a timeout into a 401 -- that needs a real HTTP response, so the connection was reaching a different server. new WebSocketServer({ port: 0 }) binds the wildcard address while the client dials 127.0.0.1. On macOS those differ, and with SO_REUSEADDR a foreign process can hold the more specific 127.0.0.1:P and win the connection. Caught live: a wildcard bind took port 52584, which a running Orca app already held on loopback, and Orca answered the probe. A listener that checks a token answers 401. Ten constructions across seven files now pass host: '127.0.0.1', so the reservation covers the address the client dials and a duplicate bind is refused. Adds a ratchet, because this is not authors forgetting a convention: all 30+ .listen(0, ...) sites already pass '127.0.0.1', while 7 of 7 ws constructions did not. ws accepts { port } alone and binds the wildcard silently, so nothing told them. The guard pins the wildcard count, and pins separately at zero the option shapes it cannot read -- spreads and variable option objects fail rather than being exempted, and a recognized-construction floor catches the matcher going blind, which otherwise reads exactly like a clean tree. mobile/scripts/mock-server.ts stays on the wildcard deliberately: a phone reaches it over the LAN. --- ...bsocket-server-wildcard-bind-allowlist.txt | 14 ++ config/scripts/call-site-option-keys.ts | 204 ++++++++++++++++++ config/scripts/websocket-server-bind-scan.ts | 172 +++++++++++++++ .../websocket-server-loopback-bind.test.ts | 106 +++++++++ .../rpc-client-live-recovery.test.ts | 3 +- .../mobile-relay-e2ee.integration.test.ts | 3 +- .../relay/relay-control-client.test.ts | 7 +- src/main/runtime/rpc/relay-transport.test.ts | 3 +- .../src/web/web-runtime-client.test.ts | 3 +- src/shared/remote-runtime-client.test.ts | 13 +- .../remote-runtime-outbound-admission.test.ts | 3 +- .../remote-runtime-request-connection.test.ts | 3 +- ...mote-runtime-shared-control-test-server.ts | 7 +- ...emote-runtime-subscription-request.test.ts | 3 +- 14 files changed, 529 insertions(+), 15 deletions(-) create mode 100644 config/scripts/__fixtures__/websocket-server-wildcard-bind-allowlist.txt create mode 100644 config/scripts/call-site-option-keys.ts create mode 100644 config/scripts/websocket-server-bind-scan.ts create mode 100644 config/scripts/websocket-server-loopback-bind.test.ts diff --git a/config/scripts/__fixtures__/websocket-server-wildcard-bind-allowlist.txt b/config/scripts/__fixtures__/websocket-server-wildcard-bind-allowlist.txt new file mode 100644 index 00000000000..4b76622a9f2 --- /dev/null +++ b/config/scripts/__fixtures__/websocket-server-wildcard-bind-allowlist.txt @@ -0,0 +1,14 @@ +# Files allowed to construct a `ws` server that binds a port without pinning `host`. +# +# `ws` accepts `{ port }` alone and silently binds the wildcard address. A server +# reached over 127.0.0.1 must pin `host: '127.0.0.1'`, or a foreign loopback +# listener can hold the same port and answer in its place -- which is how +# relay-control-client.test.ts came to fail with a real HTTP 401 in a test that +# was simulating silence. +# +# This list only shrinks. Adding a line also requires raising the pin in +# websocket-server-loopback-bind.test.ts, which is deliberate friction. + +# Deliberate, not drift: this mock is dialled by a phone on the LAN, so it has to +# be reachable on a real interface. A loopback bind would make it unreachable. +mobile/scripts/mock-server.ts diff --git a/config/scripts/call-site-option-keys.ts b/config/scripts/call-site-option-keys.ts new file mode 100644 index 00000000000..dff3a971c45 --- /dev/null +++ b/config/scripts/call-site-option-keys.ts @@ -0,0 +1,204 @@ +/** + * Read the top-level option keys of a call's object-literal argument out of raw + * source text. + * + * Text rather than an AST because typescript@7 no longer ships the classic + * compiler API and every installed parser is a transitive dependency. The + * tradeoff is handled by refusing to guess: any shape this cannot read comes + * back as `unreadable` with a reason, and callers must treat that as a failure + * rather than as an absence of keys. + */ + +export type CallOptionKeys = + | { readonly readable: true; readonly keys: readonly string[] } + | { readonly readable: false; readonly reason: string } + +type ScanState = 'code' | 'line' | 'block' | 'single' | 'double' | 'template' + +function closesString(state: ScanState, current: string): boolean { + return ( + (state === 'single' && current === "'") || + (state === 'double' && current === '"') || + (state === 'template' && current === '`') + ) +} + +function opensNonCode(current: string, next: string | undefined): ScanState | null { + if (current === '/' && next === '/') { + return 'line' + } + if (current === '/' && next === '*') { + return 'block' + } + if (current === "'") { + return 'single' + } + if (current === '"') { + return 'double' + } + if (current === '`') { + return 'template' + } + return null +} + +/** + * Text between an open paren and its match, tracking strings and comments so a + * brace inside either cannot unbalance the count. Null when it never closes. + */ +function balancedArguments(text: string, openIndex: number): string | null { + let depth = 0 + let state: ScanState = 'code' + for (let index = openIndex; index < text.length; index++) { + const current = text[index] + const next = text[index + 1] + if (state === 'code') { + const opened = opensNonCode(current, next) + if (opened) { + state = opened + if (opened === 'line' || opened === 'block') { + index++ + } + } else if (current === '(' || current === '{' || current === '[') { + depth++ + } else if (current === ')' || current === '}' || current === ']') { + depth-- + if (depth === 0) { + return text.slice(openIndex + 1, index) + } + if (depth < 0) { + return null + } + } + continue + } + if (state === 'line') { + if (current === '\n') { + state = 'code' + } + continue + } + if (state === 'block') { + if (current === '*' && next === '/') { + state = 'code' + index++ + } + continue + } + if (current === '\\') { + index++ + continue + } + // Brace tracking inside `${}` would need its own depth; templates never + // appear as options, so report one as unreadable instead of guessing. + if (state === 'template' && current === '$' && next === '{') { + return null + } + if (closesString(state, current)) { + state = 'code' + } + } + return null +} + +/** Keys at depth 0 of an object literal body, with anything non-identifier kept verbatim. */ +function objectLiteralKeys(body: string): string[] { + const keys: string[] = [] + let depth = 0 + let state: ScanState = 'code' + let inValue = false + let token = '' + const flush = (): void => { + const name = token.trim() + token = '' + if (name && depth === 0) { + keys.push(name) + } + } + for (let index = 0; index < body.length; index++) { + const current = body[index] + const next = body[index + 1] + if (state === 'code') { + const opened = opensNonCode(current, next) + if (opened) { + state = opened + if (opened === 'line' || opened === 'block') { + index++ + } + } else if (current === '(' || current === '{' || current === '[') { + depth++ + if (!inValue) { + token += current + } + } else if (current === ')' || current === '}' || current === ']') { + depth-- + if (!inValue) { + token += current + } + } else if (current === ':' && depth === 0 && !inValue) { + flush() + inValue = true + } else if (current === ',' && depth === 0) { + // A shorthand or a spread ends here having never seen a colon. + if (inValue) { + inValue = false + token = '' + } else { + flush() + } + } else if (!inValue) { + token += current + } + continue + } + if (state === 'line') { + if (current === '\n') { + state = 'code' + } + continue + } + if (state === 'block') { + if (current === '*' && next === '/') { + state = 'code' + index++ + } + continue + } + if (current === '\\') { + index++ + continue + } + if (closesString(state, current)) { + state = 'code' + } + } + if (!inValue) { + flush() + } + return keys +} + +/** + * Option keys of the call whose argument list opens at `parenIndex`, or the + * reason the shape could not be read. Spreads and computed keys land in the + * latter: either can carry a key this would otherwise report as absent. + */ +export function readCallOptionKeys(text: string, parenIndex: number): CallOptionKeys { + const args = balancedArguments(text, parenIndex) + if (args === null) { + return { readable: false, reason: 'argument list never closes' } + } + if (!args.trim()) { + return { readable: false, reason: 'called with no options argument' } + } + const trimmed = args.trim() + if (!trimmed.startsWith('{') || !trimmed.endsWith('}')) { + return { readable: false, reason: 'options are not an object literal' } + } + const keys = objectLiteralKeys(trimmed.slice(1, -1)) + const unreadable = keys.find((key) => !/^[A-Za-z_$][\w$]*$/.test(key)) + if (unreadable !== undefined) { + return { readable: false, reason: `unreadable option key \`${unreadable}\`` } + } + return { readable: true, keys } +} diff --git a/config/scripts/websocket-server-bind-scan.ts b/config/scripts/websocket-server-bind-scan.ts new file mode 100644 index 00000000000..9054f8cc38e --- /dev/null +++ b/config/scripts/websocket-server-bind-scan.ts @@ -0,0 +1,172 @@ +import { readFileSync, readdirSync } from 'node:fs' +import { join, relative } from 'node:path' +import { readCallOptionKeys } from './call-site-option-keys' + +/** + * Locate every `new WebSocketServer(...)` in the tree and say, for each, whether + * it pins a bind address. + * + * `ws` accepts `{ port }` alone and silently binds the wildcard address, so a + * server the caller then dials on 127.0.0.1 sits at a port a foreign loopback + * listener can also hold -- and the more specific listener wins the connection, + * answering in that server's place. + * + * Anything unreadable is reported as `opaque` rather than skipped. A matcher + * that silently exempts the shapes it fails to parse is worse than no matcher, + * because it reads as coverage. + */ + +export type BindSite = { path: string; line: number } +export type OpaqueSite = BindSite & { reason: string } + +export type WebSocketServerBindScan = { + filesScanned: number + /** Every construction recognized, however it was then classified. */ + constructions: number + /** Binds a port with no `host`: reachable at an address the dialer never named. */ + wildcardBound: BindSite[] + /** Shape that could not be read; never treated as safe. */ + opaque: OpaqueSite[] + /** Binds a port and pins `host`. */ + loopbackBound: BindSite[] + /** No `port`: attaches to a server that owns the bind itself. */ + attached: BindSite[] +} + +const IGNORED_DIRECTORIES = new Set([ + 'node_modules', + 'dist', + 'out', + 'build', + '.git', + '__fixtures__', + 'coverage', + // Full snapshots of older releases; their bind sites are not this tree's to fix. + '.cross-version-checkouts' +]) +const SCANNED_EXTENSIONS = /\.(?:ts|tsx|mts|cts)$/ +const SCANNED_ROOTS = ['src', 'mobile', 'config', 'tests'] +const WS_IMPORT_HINT = /from\s*['"]ws['"]/ + +function collectSourceFiles(root: string, found: string[] = []): string[] { + let entries: ReturnType> + try { + entries = readdirSync(root, { withFileTypes: true }) + } catch { + return found + } + for (const entry of entries) { + if (IGNORED_DIRECTORIES.has(entry.name)) { + continue + } + const full = join(root, entry.name) + if (entry.isDirectory()) { + collectSourceFiles(full, found) + } else if (SCANNED_EXTENSIONS.test(entry.name)) { + found.push(full) + } + } + return found +} + +/** Local names bound to ws's server class, following `as` aliases and namespace imports. */ +function webSocketServerNames(text: string): { direct: Set; namespaces: Set } { + const direct = new Set() + const namespaces = new Set() + // One statement at a time: a pattern reaching for `from 'ws'` would swallow + // every import above it and lose the specifier names in the blob. + for (const match of text.matchAll(/\bimport\b([\s\S]*?)\bfrom\s*(['"])([^'"]+)\2/g)) { + if (match[3] !== 'ws') { + continue + } + const clause = match[1] + if (/^\s*type\b/.test(clause)) { + continue + } + const namespace = clause.match(/\*\s+as\s+([A-Za-z_$][\w$]*)/) + if (namespace) { + namespaces.add(namespace[1]) + } + const named = clause.match(/\{([\s\S]*)\}/) + if (!named) { + continue + } + for (const specifier of named[1].split(',')) { + const trimmed = specifier.trim() + if (!trimmed || /^type\s/.test(trimmed)) { + continue + } + const parts = trimmed.split(/\s+as\s+/) + // `Server` is ws's own alias for WebSocketServer. + if (parts[0].trim() === 'WebSocketServer' || parts[0].trim() === 'Server') { + direct.add((parts[1] ?? parts[0]).trim()) + } + } + } + return { direct, namespaces } +} + +function classify( + scan: WebSocketServerBindScan, + site: BindSite, + text: string, + paren: number +): void { + const options = readCallOptionKeys(text, paren) + if (!options.readable) { + scan.opaque.push({ ...site, reason: options.reason }) + return + } + if (!options.keys.includes('port')) { + scan.attached.push(site) + return + } + if (!options.keys.includes('host')) { + scan.wildcardBound.push(site) + return + } + scan.loopbackBound.push(site) +} + +export function scanWebSocketServerBinds(repoRoot: string): WebSocketServerBindScan { + const files = SCANNED_ROOTS.flatMap((directory) => collectSourceFiles(join(repoRoot, directory))) + const scan: WebSocketServerBindScan = { + filesScanned: files.length, + constructions: 0, + wildcardBound: [], + opaque: [], + loopbackBound: [], + attached: [] + } + for (const file of files) { + const text = readFileSync(file, 'utf8') + // Filter on the import, not on the class name: `Server as Wss` never spells + // WebSocketServer, and keying on that name silently skipped the whole alias. + if (!WS_IMPORT_HINT.test(text)) { + continue + } + const { direct, namespaces } = webSocketServerNames(text) + if (!direct.size && !namespaces.size) { + continue + } + const path = relative(repoRoot, file).split('\\').join('/') + const patterns = [ + ...[...direct].map((name) => new RegExp(`\\bnew\\s+${name}\\s*\\(`, 'g')), + ...[...namespaces].map( + (name) => new RegExp(`\\bnew\\s+${name}\\.(?:WebSocketServer|Server)\\s*\\(`, 'g') + ) + ] + for (const pattern of patterns) { + for (const match of text.matchAll(pattern)) { + scan.constructions++ + const line = text.slice(0, match.index).split('\n').length + classify(scan, { path, line }, text, match.index + match[0].length - 1) + } + } + } + return scan +} + +export function formatSites(sites: readonly BindSite[]): string[] { + return sites.map((site) => `${site.path}:${site.line}`) +} diff --git a/config/scripts/websocket-server-loopback-bind.test.ts b/config/scripts/websocket-server-loopback-bind.test.ts new file mode 100644 index 00000000000..9f32f8eda1a --- /dev/null +++ b/config/scripts/websocket-server-loopback-bind.test.ts @@ -0,0 +1,106 @@ +import { readFileSync } from 'node:fs' +import { join, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { formatSites, scanWebSocketServerBinds } from './websocket-server-bind-scan' + +/** + * Hold the bind address at the tree level rather than per call site. + * + * Every one of the ~30 `.listen(0, ...)` calls in this repo already passes + * '127.0.0.1'; 7 of 7 `new WebSocketServer({ port })` calls did not. Authors know + * the convention -- `ws` just never asks, because `{ port }` alone binds the + * wildcard without a word. That silence is what this test replaces. + * + * The allowlist only shrinks. A new wildcard bind fails here even where it looks + * harmless today, because harmless-looking is exactly what the seven were. + */ +/** The ratchet, held as data so it reads as the list it is. */ +const WILDCARD_BIND_ALLOWLIST: readonly string[] = readFileSync( + join(__dirname, '__fixtures__', 'websocket-server-wildcard-bind-allowlist.txt'), + 'utf8' +) + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0 && !line.startsWith('#')) + +/** + * The true count of constructions that bind a port without pinning a host. + * + * May only ever be DECREASED, and only by pinning a host. Raising it is never + * the fix. + */ +const WILDCARD_BIND_PIN = 1 + +/** + * A floor under the constructions the scanner still recognizes. + * + * This is the guard against the scanner going blind: an import pattern it stops + * following reports zero offenders and reads exactly like a clean tree. During + * development a single wrong regex dropped this from 24 to 3. + */ +const RECOGNIZED_CONSTRUCTION_FLOOR = 20 + +describe('WebSocketServer loopback bind boundary', () => { + const repoRoot = resolve(__dirname, '..', '..') + const scan = scanWebSocketServerBinds(repoRoot) + const offenders = scan.wildcardBound.map((site) => site.path) + + it('scans a plausible number of files', () => { + // A broken root or extension list would make the guard silently vacuous. + expect(scan.filesScanned).toBeGreaterThan(5_000) + }) + + it('still recognizes the known construction sites', () => { + expect( + scan.constructions, + `Only ${scan.constructions} WebSocketServer constructions were recognized; the floor is ` + + `${RECOGNIZED_CONSTRUCTION_FLOOR}. The scanner has probably stopped following an import ` + + 'shape rather than the tree having lost that many servers.' + ).toBeGreaterThanOrEqual(RECOGNIZED_CONSTRUCTION_FLOOR) + }) + + it('can read the options of every construction it found', () => { + // An unreadable shape is never assumed safe: it could be hiding a host, or + // hiding the absence of one. Rewrite it as a plain object literal. + expect( + scan.opaque.map((site) => `${site.path}:${site.line} -- ${site.reason}`), + 'WebSocketServer options that this guard cannot read.' + ).toEqual([]) + }) + + it('has no wildcard-bound server outside the allowlist', () => { + const unlisted = scan.wildcardBound.filter( + (site) => !WILDCARD_BIND_ALLOWLIST.includes(site.path) + ) + expect( + formatSites(unlisted), + "New WebSocketServer that binds a port without a host. Pass host: '127.0.0.1' so a foreign " + + 'loopback listener cannot claim the port and answer in its place.' + ).toEqual([]) + }) + + it('has no stale allowlist entry', () => { + // Why this direction matters too: an entry left behind after the file was + // fixed hides the next regression in that same path. + const stale = WILDCARD_BIND_ALLOWLIST.filter((path) => !offenders.includes(path)) + expect(stale, 'Allowlist entry no longer binds the wildcard — delete the line.').toEqual([]) + }) + + it('holds the wildcard-bind count at the pin', () => { + // Bounding by the allowlist's own length would prove nothing: the two move + // together, so appending a line to silence a failure would keep the bound + // satisfied. The pin is a literal so that widening takes a second edit. + expect( + scan.wildcardBound.length, + `${scan.wildcardBound.length} constructions bind the wildcard; the pin is ` + + `${WILDCARD_BIND_PIN}. Never raise the pin -- pass host: '127.0.0.1' instead.` + ).toBeLessThanOrEqual(WILDCARD_BIND_PIN) + // A pin left above reality is how a ratchet rots: it re-opens room for the + // next wildcard bind to land for free. + expect( + scan.wildcardBound.length, + `Only ${scan.wildcardBound.length} constructions bind the wildcard. Lower ` + + `WILDCARD_BIND_PIN to ${scan.wildcardBound.length} to keep the ground you just took.` + ).toBeGreaterThanOrEqual(WILDCARD_BIND_PIN) + }) +}) diff --git a/mobile/src/transport/rpc-client-live-recovery.test.ts b/mobile/src/transport/rpc-client-live-recovery.test.ts index 471a53be740..bef276f918d 100644 --- a/mobile/src/transport/rpc-client-live-recovery.test.ts +++ b/mobile/src/transport/rpc-client-live-recovery.test.ts @@ -59,7 +59,8 @@ function e2eeDecrypt(encrypted: string, sharedKey: Uint8Array): string | null { // fail with EADDRINUSE; the full scenario restarts on the captured port // because the client keeps reconnecting to its original URL. function startServer(port = 0): Promise { - const wss = new WebSocketServer({ port }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const wss = new WebSocketServer({ host: '127.0.0.1', port }) wss.on('connection', (ws: ServerSocket) => { let sharedKey: Uint8Array | null = null let authenticated = false diff --git a/src/main/runtime/relay/mobile-relay-e2ee.integration.test.ts b/src/main/runtime/relay/mobile-relay-e2ee.integration.test.ts index 1f2b50e0648..bf9ec2ce431 100644 --- a/src/main/runtime/relay/mobile-relay-e2ee.integration.test.ts +++ b/src/main/runtime/relay/mobile-relay-e2ee.integration.test.ts @@ -61,7 +61,8 @@ describe('desktop relay E2EE integration', () => { }) it('splices a simulated phone through CloudRelayTransport with real NaCl E2EE v2', async () => { - const relay = new WebSocketServer({ port: 0, perMessageDeflate: false }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const relay = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) servers.push(relay) await new Promise((resolve) => relay.once('listening', resolve)) const address = relay.address() diff --git a/src/main/runtime/relay/relay-control-client.test.ts b/src/main/runtime/relay/relay-control-client.test.ts index a1ad5c03482..2975b67e631 100644 --- a/src/main/runtime/relay/relay-control-client.test.ts +++ b/src/main/runtime/relay/relay-control-client.test.ts @@ -99,7 +99,8 @@ describe('RelayControlClient', () => { }) it('rejects a control handshake that never receives a proof response', async () => { - const server = new WebSocketServer({ port: 0, perMessageDeflate: false }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) servers.push(server) await new Promise((resolve) => server.once('listening', resolve)) const address = server.address() @@ -129,7 +130,7 @@ describe('RelayControlClient', () => { }) it('settles an opening control immediately when ownership closes', async () => { - const server = new WebSocketServer({ port: 0, perMessageDeflate: false }) + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) servers.push(server) await new Promise((resolve) => server.once('listening', resolve)) const address = server.address() @@ -165,7 +166,7 @@ describe('RelayControlClient', () => { }) it('proves the host key and drives control/data commands without URL credentials', async () => { - const server = new WebSocketServer({ port: 0, perMessageDeflate: false }) + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) servers.push(server) await new Promise((resolve) => server.once('listening', resolve)) const address = server.address() diff --git a/src/main/runtime/rpc/relay-transport.test.ts b/src/main/runtime/rpc/relay-transport.test.ts index 8b7303ed805..21b520c8c9e 100644 --- a/src/main/runtime/rpc/relay-transport.test.ts +++ b/src/main/runtime/rpc/relay-transport.test.ts @@ -28,7 +28,8 @@ describe('CloudRelayTransport', () => { }) it('authenticates one query-free host-data socket and forwards messages verbatim', async () => { - const server = new WebSocketServer({ port: 0, perMessageDeflate: false }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const server = new WebSocketServer({ host: '127.0.0.1', port: 0, perMessageDeflate: false }) servers.push(server) await new Promise((resolve) => server.once('listening', resolve)) const address = server.address() diff --git a/src/renderer/src/web/web-runtime-client.test.ts b/src/renderer/src/web/web-runtime-client.test.ts index 1aa36f06d0f..7373ede6b8c 100644 --- a/src/renderer/src/web/web-runtime-client.test.ts +++ b/src/renderer/src/web/web-runtime-client.test.ts @@ -658,7 +658,8 @@ describe('WebRuntimeClient', () => { vi.stubGlobal('WebSocket', WebSocket) const serverKeys = generateKeyPair() const frame = new Uint8Array([9, 8, 7]) - const wss = new WebSocketServer({ port: 0 }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const wss = new WebSocketServer({ host: '127.0.0.1', port: 0 }) const sockets = new Set() wss.on('connection', (socket) => { sockets.add(socket) diff --git a/src/shared/remote-runtime-client.test.ts b/src/shared/remote-runtime-client.test.ts index 3e3bba921ea..d7a76417e84 100644 --- a/src/shared/remote-runtime-client.test.ts +++ b/src/shared/remote-runtime-client.test.ts @@ -539,7 +539,12 @@ async function createSubscriptionServer( const nextAuth = new Promise((resolve) => { resolveAuth = resolve }) - const wss = new WebSocketServer({ port: 0, autoPong: options.disableAutoPong !== true }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const wss = new WebSocketServer({ + host: '127.0.0.1', + port: 0, + autoPong: options.disableAutoPong !== true + }) servers.push(wss) wss.on('connection', (ws) => { @@ -625,7 +630,7 @@ async function createClosingServer( reason: string ): Promise<{ pairing: PairingOffer }> { const serverKeyPair = generateKeyPair() - const wss = new WebSocketServer({ port: 0 }) + const wss = new WebSocketServer({ host: '127.0.0.1', port: 0 }) servers.push(wss) wss.on('connection', (ws) => { ws.close(code, reason) @@ -649,7 +654,7 @@ async function createClosingServer( async function createInvalidHandshakeServer(): Promise<{ pairing: PairingOffer }> { const serverKeyPair = generateKeyPair() - const wss = new WebSocketServer({ port: 0 }) + const wss = new WebSocketServer({ host: '127.0.0.1', port: 0 }) servers.push(wss) wss.on('connection', (ws) => { ws.once('message', () => ws.send(JSON.stringify({ type: 'not_orca' }))) @@ -680,7 +685,7 @@ async function createOneShotServer( } = {} ): Promise<{ pairing: PairingOffer }> { const serverKeyPair = generateKeyPair() - const wss = new WebSocketServer({ port: 0 }) + const wss = new WebSocketServer({ host: '127.0.0.1', port: 0 }) servers.push(wss) wss.on('connection', (ws) => { diff --git a/src/shared/remote-runtime-outbound-admission.test.ts b/src/shared/remote-runtime-outbound-admission.test.ts index f536549ca46..eb5f902a16a 100644 --- a/src/shared/remote-runtime-outbound-admission.test.ts +++ b/src/shared/remote-runtime-outbound-admission.test.ts @@ -352,7 +352,8 @@ describe('remote runtime outbound admission', () => { async function createServer(): Promise<{ pairing: PairingOffer; server: WebSocketServer }> { const keyPair = generateKeyPair() - const server = new WebSocketServer({ port: 0 }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const server = new WebSocketServer({ host: '127.0.0.1', port: 0 }) servers.push(server) await new Promise((resolve) => server.once('listening', resolve)) const address = server.address() as AddressInfo diff --git a/src/shared/remote-runtime-request-connection.test.ts b/src/shared/remote-runtime-request-connection.test.ts index 4d812322ba0..eb8f0b06e89 100644 --- a/src/shared/remote-runtime-request-connection.test.ts +++ b/src/shared/remote-runtime-request-connection.test.ts @@ -98,7 +98,8 @@ async function createServer(): Promise { const requests: unknown[] = [] const auths: unknown[] = [] let connectionCount = 0 - const wss = new WebSocketServer({ port: 0 }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const wss = new WebSocketServer({ host: '127.0.0.1', port: 0 }) servers.push(wss) wss.on('connection', (ws) => { diff --git a/src/shared/remote-runtime-shared-control-test-server.ts b/src/shared/remote-runtime-shared-control-test-server.ts index 33b535c8405..0e4adb1a69c 100644 --- a/src/shared/remote-runtime-shared-control-test-server.ts +++ b/src/shared/remote-runtime-shared-control-test-server.ts @@ -60,7 +60,12 @@ export async function createSharedControlTestServer( const delayedResponses: (() => void)[] = [] let connectionCount = 0 let closedAfterFirstStreamingResponse = false - const wss = new WebSocketServer({ port: 0, autoPong: options.disableAutoPong !== true }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const wss = new WebSocketServer({ + host: '127.0.0.1', + port: 0, + autoPong: options.disableAutoPong !== true + }) servers.push(wss) wss.on('connection', (ws) => { diff --git a/src/shared/remote-runtime-subscription-request.test.ts b/src/shared/remote-runtime-subscription-request.test.ts index dabc51e6a24..14f904308cc 100644 --- a/src/shared/remote-runtime-subscription-request.test.ts +++ b/src/shared/remote-runtime-subscription-request.test.ts @@ -262,7 +262,8 @@ async function createServer(options: ServerOptions = {}): Promise<{ const nextRequest = new Promise((resolve) => { resolveRequest = resolve }) - const wss = new WebSocketServer({ port: 0 }) + // host must match the 127.0.0.1 clients dial: a wildcard bind lets a foreign loopback listener claim the port and answer here. + const wss = new WebSocketServer({ host: '127.0.0.1', port: 0 }) servers.push(wss) wss.on('connection', (ws) => { let sharedKey: Uint8Array | null = null From d7123591cebd103658c6d5c8f601eebe1dc0cb3e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:06:44 -0700 Subject: [PATCH 008/398] perf(git): pack the loose refs Orca's own fetches leave behind (#17857) * perf(git): pack the loose refs Orca's own fetches leave behind Orca strips git's auto-maintenance off every fetch it issues (GIT_FETCH_SKIP_AUTO_MAINTENANCE_CONFIG_ARGS) and never compensated, so nothing in an Orca-driven checkout ever packs refs. One real machine reached 36,574 loose refs, where `git show-ref -- main` costs 5.2s and every worktree create pays for it. Add an idle-time, per-repo `git pack-refs --all --prune`, armed by the fetches that create the debt. It runs only after ten minutes of quiet on that repo, only above 1000 loose refs (probed with a walk bounded by that threshold, not by the backlog), one at a time across the whole app, at the background admission tier, and never while an agent is working, a create is prepared or in flight, a worktree removal is deleting refs, the app is quitting, or the machine is on battery. A user who set `maintenance.auto=false` or `gc.auto=0` has opted out. Measured on a 36,001-loose-ref fixture (macOS/APFS, git 2.44): `show-ref` 5.5-12.2s -> 30-49ms, `for-each-ref` 4.0-10.8s -> 43-48ms. Also fixes a pre-existing bug the split exposed: `--path-format=absolute` is ignored before git 2.31, and taking rev-parse's stdout raw collapsed every repo on such a host onto one fetch-serialization key. Refs #17828 * perf(git): make idle ref maintenance preemptible and cheaper to probe The idle veto was one-directional: it stopped a pack from starting during a create, removal, or agent work, but nothing stopped those from starting during a pack. A user-clicked Fetch, a branch delete, or a worktree removal that needed `packed-refs.lock` mid-rewrite could fail with `unable to create packed-refs.lock` -- a git error with no visible cause. Make the pack cancellable end to end. An AbortSignal now reaches the `pack-refs` child and both pre-pack probes, and `pause()` aborts what is running, waits for it to actually stop, and holds a suspension count so nothing new starts until the caller releases. Every entry point that deletes a ref takes that pause: gitFetch, gitPull, gitFastForward, removeWorktree, forceDeleteLocalBranch, prepareWorktreeCreateCheckout, addWorktree. Five more triggers close the rest of the window: battery drop, window focus, quit, the attempt deadline, and any other git command queueing for an admission slot. Judge a pack by re-probing the backlog rather than by the child's exit code. Measured in the field: another Orca session moved a branch mid-pack, git reported `cannot lock ref`, skipped that ref and packed the rest -- 36,688 loose refs down to 3. On a machine running several sessions that is the normal case, and retrying it would be wrong. Probe with one batched `readdir` per directory instead of streaming `opendir`, which issues a thread-pool round trip every 32 entries: 177ms -> 23ms on a real 36,600-ref repository, with half the event-loop lag. The walk stays strictly sequential so it can never occupy more than one of libuv's four filesystem threads. `PackRefsLockOwnership` makes a lock left by SIGKILL attributable, and only reclaims one when a marker exists, the lock is older than any pack-refs could run for, and the recorded process is gone. Refs #17828 * fix(git): wait out the packed-refs lock instead of killing the pack Measured on Git 2.55/APFS with 37k loose refs: a full `pack-refs --all --prune` takes 23-32s but holds `packed-refs.lock` for only 0.03-1.37s of it. The other ~95% is the prune phase, during which a concurrent `fetch --prune`, `branch -D` or `update-ref` succeeds every time -- per-ref locks last microseconds and git retries for `core.filesRefLockTimeout`. So the abort-on-everything design was strictly harmful. SIGTERM into the prune loop strands an empty `refs/**/*.lock` about one time in five (9/30, 5/40, 6/30 kills): `tempfile.c` opens the lock O_EXCL before `activate_tempfile()` links it into the list the signal handler walks, and a pack does ~36k lock cycles. Afterwards `update-ref -d` on that ref fails with `cannot lock ref ... File exists`, permanently. On Windows `taskkill /f` never runs git's handlers at all, so an abort inside the rewrite strands `packed-refs.lock` every time. Never signal the child. `packRefs` no longer takes an abort signal; it polls `packed-refs.lock` and reports the window through a `PackedRefsLockReporter`. `pause()` resolves when the lock is released -- bounded, and free during the prune -- while the suspension counter still blocks new attempts. Battery and window-focus become do-not-start rather than stop-what-is-running, and quit waits for the lock and lets the child finish orphaned. For strands that already exist, `PackRefsLockOwnership` now also reclaims `refs/**/*.lock` under the same three conditions plus a 0-byte check, and a lock carrying our own not-yet-reclaimable marker records `locked` with a 30min retry instead of the 6h failure cooldown -- so a Windows strand self-heals in half an hour rather than six. Reverts the git admission-scheduler event bus, which existed only to drive the abort this removes. Refs #17828 * test(git): make the ref-maintenance waits survive a loaded runner CI shard 4/8 failed on `restarts every armed countdown when the user does ref work themselves`, which passes locally. The `until()` helper spun a fixed 200 event-loop turns and then returned silently, so on a contended runner the filesystem probe had not finished and the assertion that followed failed with an unrelated message. Bound the wait by wall clock instead and throw a named error, which immediately exposed a second latent bug: the single-flight test's second wait could never succeed, because the deferred repo's retry is on a faked `setTimeout` that spinning the real loop never advances. It had been passing only because the old helper gave up quietly. Add a timer-aware variant for those, and have the countdown test await a signal the fake pack resolves rather than polling at all. Verified stable across five sequential runs and once under load average 32 with six concurrent suites. Refs #17828 --- src/main/agent-awake-service.ts | 5 + src/main/cli/cli-command-installation.ts | 8 +- src/main/cli/cli-installer.ts | 8 +- src/main/cli/wsl-cli-installer.ts | 4 +- src/main/git/canonical-repo-key.test.ts | 78 +++ src/main/git/canonical-repo-key.ts | 72 +++ .../git/local-repo-ref-maintenance.test.ts | 182 ++++++ src/main/git/local-repo-ref-maintenance.ts | 272 +++++++++ src/main/git/pack-refs-lock-ownership.test.ts | 192 +++++++ src/main/git/pack-refs-lock-ownership.ts | 202 +++++++ src/main/git/remote.ts | 56 +- .../git/repo-ref-maintenance-real-git.test.ts | 290 ++++++++++ src/main/git/worktree-add.ts | 21 +- src/main/git/worktree-branch-removal.ts | 7 +- src/main/git/worktree-create-preparation.ts | 81 +-- src/main/git/worktree-removal.ts | 10 +- src/main/ipc/repos-create.test.ts | 5 +- src/main/ipc/worktrees.ts | 7 +- .../ipc/worktrees/worktree-ipc-context.ts | 15 + src/main/repo-maintenance-idle-gate.test.ts | 134 +++++ src/main/repo-maintenance-idle-gate.ts | 67 +++ src/main/runtime/fetch-remote-cache.test.ts | 8 +- .../runtime-remote-fetch-controller.ts | 56 +- ...ntime-remote-fetch-ref-maintenance.test.ts | 145 +++++ src/main/startup/main-process-observers.ts | 5 + src/main/startup/main-process-quit.ts | 18 + src/main/startup/main-process-state.ts | 2 + src/main/worktree-create-preparation.ts | 5 + src/shared/git-binary-compatibility.test.ts | 33 ++ src/shared/loose-ref-count.test.ts | 119 ++++ src/shared/loose-ref-count.ts | 80 +++ src/shared/packed-refs-lock-gate.ts | 43 ++ src/shared/repo-ref-maintenance-policy.ts | 166 ++++++ src/shared/repo-ref-maintenance.test.ts | 530 ++++++++++++++++++ src/shared/repo-ref-maintenance.ts | 366 ++++++++++++ 35 files changed, 3197 insertions(+), 95 deletions(-) create mode 100644 src/main/git/canonical-repo-key.test.ts create mode 100644 src/main/git/canonical-repo-key.ts create mode 100644 src/main/git/local-repo-ref-maintenance.test.ts create mode 100644 src/main/git/local-repo-ref-maintenance.ts create mode 100644 src/main/git/pack-refs-lock-ownership.test.ts create mode 100644 src/main/git/pack-refs-lock-ownership.ts create mode 100644 src/main/git/repo-ref-maintenance-real-git.test.ts create mode 100644 src/main/repo-maintenance-idle-gate.test.ts create mode 100644 src/main/repo-maintenance-idle-gate.ts create mode 100644 src/main/runtime/runtime-remote-fetch-ref-maintenance.test.ts create mode 100644 src/shared/loose-ref-count.test.ts create mode 100644 src/shared/loose-ref-count.ts create mode 100644 src/shared/packed-refs-lock-gate.ts create mode 100644 src/shared/repo-ref-maintenance-policy.ts create mode 100644 src/shared/repo-ref-maintenance.test.ts create mode 100644 src/shared/repo-ref-maintenance.ts diff --git a/src/main/agent-awake-service.ts b/src/main/agent-awake-service.ts index 29e866d27f2..b79612e2b9c 100644 --- a/src/main/agent-awake-service.ts +++ b/src/main/agent-awake-service.ts @@ -117,6 +117,11 @@ export class AgentAwakeService { } } + /** Agents this runtime has seen working recently, independent of the awake setting. */ + getWorkingAgentCount(): number { + return this.getEligibleRunningStatusCount() + } + subscribe(listener: (status: ComputerAwakeStatus) => void): () => void { this.statusListeners.add(listener) return () => this.statusListeners.delete(listener) diff --git a/src/main/cli/cli-command-installation.ts b/src/main/cli/cli-command-installation.ts index 5b277bd8c62..fbabc3bf6dd 100644 --- a/src/main/cli/cli-command-installation.ts +++ b/src/main/cli/cli-command-installation.ts @@ -35,7 +35,9 @@ export class CliCommandInstallation extends CliCommandInspection { const inspected = await this.inspectStableSymlink(commandPath, launcherPath) if (inspected.status.state === 'conflict') { - throw new Error(`Refusing to replace non-Orca command at ${commandPath}. Remove it and register again if it is no longer needed.`) + throw new Error( + `Refusing to replace non-Orca command at ${commandPath}. Remove it and register again if it is no longer needed.` + ) } if (inspected.status.state === 'installed') { return @@ -54,7 +56,9 @@ export class CliCommandInstallation extends CliCommandInspection { if (!(await capturedExpectedEntry(quarantine, inspected))) { await this.restoreQuarantinedCommand(quarantine, commandPath) - throw new Error(`Refusing to replace non-Orca command at ${commandPath}. Remove it and register again if it is no longer needed.`) + throw new Error( + `Refusing to replace non-Orca command at ${commandPath}. Remove it and register again if it is no longer needed.` + ) } try { diff --git a/src/main/cli/cli-installer.ts b/src/main/cli/cli-installer.ts index 3c832df5078..d95e1649ac0 100644 --- a/src/main/cli/cli-installer.ts +++ b/src/main/cli/cli-installer.ts @@ -116,7 +116,9 @@ export class CliInstaller extends CliPathRegistration { throw new Error(initialStatus.detail ?? 'CLI registration is unavailable on this build.') } if (initialStatus.state === 'conflict') { - throw new Error(`Refusing to replace non-Orca command at ${initialStatus.commandPath}. Remove it and register again if it is no longer needed.`) + throw new Error( + `Refusing to replace non-Orca command at ${initialStatus.commandPath}. Remove it and register again if it is no longer needed.` + ) } const extractedRoot = await this.ensureLinuxAppImagePayload() const status = extractedRoot @@ -126,7 +128,9 @@ export class CliInstaller extends CliPathRegistration { throw new Error(status.detail ?? 'CLI registration is unavailable on this build.') } if (status.state === 'conflict') { - throw new Error(`Refusing to replace non-Orca command at ${status.commandPath}. Remove it and register again if it is no longer needed.`) + throw new Error( + `Refusing to replace non-Orca command at ${status.commandPath}. Remove it and register again if it is no longer needed.` + ) } // eslint-disable-next-line unicorn/prefer-ternary -- Why: the install path performs async side effects and is easier to audit as an explicit branch than as an awaited ternary. diff --git a/src/main/cli/wsl-cli-installer.ts b/src/main/cli/wsl-cli-installer.ts index f8362fddc66..484ed4f9bc3 100644 --- a/src/main/cli/wsl-cli-installer.ts +++ b/src/main/cli/wsl-cli-installer.ts @@ -207,7 +207,9 @@ export class WslCliInstaller { throw new Error(status.detail ?? 'WSL CLI registration is unavailable.') } if (status.state === 'conflict') { - throw new Error(`Refusing to replace non-Orca command at ${status.commandPath}. Remove it and register again if it is no longer needed.`) + throw new Error( + `Refusing to replace non-Orca command at ${status.commandPath}. Remove it and register again if it is no longer needed.` + ) } await this.run( diff --git a/src/main/git/canonical-repo-key.test.ts b/src/main/git/canonical-repo-key.test.ts new file mode 100644 index 00000000000..87965bcaed6 --- /dev/null +++ b/src/main/git/canonical-repo-key.test.ts @@ -0,0 +1,78 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const gitExecFileAsyncMock = vi.hoisted(() => vi.fn()) + +vi.mock('./runner', async (importOriginal) => ({ + ...((await importOriginal()) as Record), + gitExecFileAsync: gitExecFileAsyncMock +})) + +import { + _resetCanonicalRepoKeyCacheForTests, + getCanonicalRepoKey, + readGitCommonDir +} from './canonical-repo-key' + +beforeEach(() => { + _resetCanonicalRepoKeyCacheForTests() + gitExecFileAsyncMock.mockReset() +}) + +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('readGitCommonDir', () => { + it('reads the absolute answer modern Git gives', () => { + expect(readGitCommonDir('/repo/.git\n', '/repo/worktrees/a')).toBe('/repo/.git') + }) + + it('drops the flag Git older than 2.31 echoes back, and resolves the relative answer', () => { + // Without this every repository on such a host would answer `.git` and collide. + expect(readGitCommonDir('--path-format=absolute\n.git\n', '/repo')).toBe('/repo/.git') + }) + + it('resolves a WSL answer in Git execution space, not against the UNC path', () => { + expect(readGitCommonDir('.git\n', '//wsl$/Ubuntu/home/dev/repo')).toBe('/home/dev/repo/.git') + }) + + it('tolerates CRLF and blank lines', () => { + expect(readGitCommonDir('\r\n/repo/.git\r\n', '/repo')).toBe('/repo/.git') + }) + + it('returns undefined when Git printed nothing usable', () => { + expect(readGitCommonDir('\n', '/repo')).toBeUndefined() + }) +}) + +describe('getCanonicalRepoKey', () => { + it('gives every worktree of one repository the same key', async () => { + gitExecFileAsyncMock.mockResolvedValue({ stdout: '/repo/.git\n', stderr: '' }) + + await expect(getCanonicalRepoKey('/repo')).resolves.toBe('local::/repo/.git') + await expect(getCanonicalRepoKey('/repo/worktrees/a')).resolves.toBe('local::/repo/.git') + }) + + it('scopes the key to the execution host', async () => { + gitExecFileAsyncMock.mockResolvedValue({ stdout: '/home/dev/repo/.git\n', stderr: '' }) + + await expect( + getCanonicalRepoKey('//wsl$/Ubuntu/home/dev/repo', { wslDistro: 'Ubuntu' }) + ).resolves.toBe('wsl:Ubuntu::/home/dev/repo/.git') + }) + + it('caches so repeated arming costs no subprocess', async () => { + gitExecFileAsyncMock.mockResolvedValue({ stdout: '/repo/.git\n', stderr: '' }) + + await getCanonicalRepoKey('/repo') + await getCanonicalRepoKey('/repo') + + expect(gitExecFileAsyncMock).toHaveBeenCalledTimes(1) + }) + + it('falls back to the caller path when Git cannot answer', async () => { + gitExecFileAsyncMock.mockRejectedValue(new Error('not a git repository')) + + await expect(getCanonicalRepoKey('/not-a-repo')).resolves.toBe('local::/not-a-repo') + }) +}) diff --git a/src/main/git/canonical-repo-key.ts b/src/main/git/canonical-repo-key.ts new file mode 100644 index 00000000000..1b06423bdc9 --- /dev/null +++ b/src/main/git/canonical-repo-key.ts @@ -0,0 +1,72 @@ +import { toWslExecutionSpace } from '../../shared/wsl-paths' +import { gitExecFileAsync } from './runner' +import { resolveRevParsePath } from './worktree-path-comparison' + +/** + * One repository on one execution host, named by its Git common dir. + * + * Shared by the fetch controller (which serializes fetches on it) and idle ref + * maintenance (which scopes all of its state to it), so both agree on what "the + * same repo" means across every worktree that points at it. + */ + +export type CanonicalRepoKeyOptions = { wslDistro?: string } + +const CACHE_MAX = 512 +const cache = new Map() + +/** + * Git < 2.31 ignores `--path-format=absolute`: it echoes the unrecognized flag, + * exits 0, and prints a relative `.git`. Taking the raw stdout there would give + * every repository on the host the same key. + */ +export function readGitCommonDir(stdout: string, repoPath: string): string | undefined { + const commonDir = stdout + .split('\n') + .map((line) => (line.endsWith('\r') ? line.slice(0, -1) : line)) + .findLast((line) => line.length > 0 && !line.startsWith('-')) + return commonDir ? resolveRevParsePath(toWslExecutionSpace(repoPath), commonDir) : undefined +} + +function remember(cacheKey: string, value: string): string { + cache.delete(cacheKey) + cache.set(cacheKey, value) + while (cache.size > CACHE_MAX) { + const oldest = cache.keys().next() + if (oldest.done) { + break + } + cache.delete(oldest.value) + } + return value +} + +/** `${runtimeKey}::${gitCommonDir}`, falling back to the caller's path. */ +export async function getCanonicalRepoKey( + repoPath: string, + options: CanonicalRepoKeyOptions = {} +): Promise { + const runtimeKey = options.wslDistro ? `wsl:${options.wslDistro}` : 'local' + const cacheKey = `${runtimeKey}::${repoPath}` + const cached = cache.get(cacheKey) + if (cached !== undefined) { + return remember(cacheKey, cached) + } + try { + const { stdout } = await gitExecFileAsync( + ['rev-parse', '--path-format=absolute', '--git-common-dir'], + { cwd: repoPath, ...options } + ) + const commonDir = readGitCommonDir(stdout, repoPath) + if (commonDir) { + return remember(cacheKey, `${runtimeKey}::${commonDir}`) + } + } catch { + // The caller path remains a safe serialization key when canonicalization fails. + } + return remember(cacheKey, cacheKey) +} + +export function _resetCanonicalRepoKeyCacheForTests(): void { + cache.clear() +} diff --git a/src/main/git/local-repo-ref-maintenance.test.ts b/src/main/git/local-repo-ref-maintenance.test.ts new file mode 100644 index 00000000000..f6a411733c3 --- /dev/null +++ b/src/main/git/local-repo-ref-maintenance.test.ts @@ -0,0 +1,182 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const gitExecFileAsyncMock = vi.hoisted(() => vi.fn()) +const readRepoCommonDirFromGitMock = vi.hoisted(() => vi.fn()) + +vi.mock('./runner', async (importOriginal) => ({ + ...((await importOriginal()) as Record), + gitExecFileAsync: gitExecFileAsyncMock +})) + +vi.mock('./worktree-list-reader', async (importOriginal) => ({ + ...((await importOriginal()) as Record), + readRepoCommonDirFromGit: readRepoCommonDirFromGitMock +})) + +import { _resetCanonicalRepoKeyCacheForTests } from './canonical-repo-key' +import { + _resetLocalRepoRefMaintenanceForTests, + armLocalRepoRefMaintenance, + createLocalRepoRefMaintenanceTarget, + getLocalRepoRefMaintenance, + setRepoMaintenanceActivityProbe, + withRepoRefMaintenancePaused +} from './local-repo-ref-maintenance' + +const NO_ABORT = new AbortController().signal + +function target(wslDistro?: string): ReturnType { + return createLocalRepoRefMaintenanceTarget({ + key: 'local::/repo/.git', + repoPath: wslDistro ? '//wsl$/Ubuntu/home/dev/repo' : '/repo', + ...(wslDistro ? { wslDistro } : {}) + }) +} + +beforeEach(() => { + gitExecFileAsyncMock.mockReset() + readRepoCommonDirFromGitMock.mockReset() + delete process.env.ORCA_DISABLE_REPO_REF_MAINTENANCE + _resetCanonicalRepoKeyCacheForTests() + _resetLocalRepoRefMaintenanceForTests() +}) + +afterEach(() => { + delete process.env.ORCA_DISABLE_REPO_REF_MAINTENANCE + _resetLocalRepoRefMaintenanceForTests() + vi.restoreAllMocks() +}) + +describe('local repo ref maintenance target', () => { + it('never hands the pack child an abort signal', async () => { + // Killing a `pack-refs` strands a `refs/**` lock about one time in five, and + // on Windows a force-kill inside the rewrite strands `packed-refs.lock` + // every time. The child must always be allowed to finish. + readRepoCommonDirFromGitMock.mockResolvedValue('/repo/.git') + gitExecFileAsyncMock.mockResolvedValue({ stdout: '', stderr: '' }) + + await target().packRefs({ setHeld: () => {} }) + + const packCall = gitExecFileAsyncMock.mock.calls.find( + ([argv]) => (argv as string[])[0] === 'pack-refs' + ) + expect(packCall?.[1]).not.toHaveProperty('signal') + }) + + it('runs pack-refs at the background tier with a long deadline', async () => { + gitExecFileAsyncMock.mockResolvedValue({ stdout: '', stderr: '' }) + + await target().packRefs({ setHeld: () => {} }) + + expect(gitExecFileAsyncMock).toHaveBeenCalledWith( + ['pack-refs', '--all', '--prune'], + expect.objectContaining({ cwd: '/repo', admissionTier: 'background', timeout: 15 * 60_000 }) + ) + }) + + it('reads either Git auto-maintenance opt-out, and unset keys as consent', async () => { + for (const stdout of [ + 'maintenance.auto false\n', + 'gc.auto 0\n', + 'gc.auto 6700\nmaintenance.auto false\n' + ]) { + gitExecFileAsyncMock.mockResolvedValue({ stdout, stderr: '' }) + await expect(target().isOptedOut?.(NO_ABORT)).resolves.toBe(true) + } + + gitExecFileAsyncMock.mockResolvedValue({ + stdout: 'maintenance.auto true\ngc.auto 6700\n', + stderr: '' + }) + await expect(target().isOptedOut?.(NO_ABORT)).resolves.toBe(false) + + // `git config --get-regexp` exits non-zero when nothing matches. + gitExecFileAsyncMock.mockRejectedValue(new Error('exit 1')) + await expect(target().isOptedOut?.(NO_ABORT)).resolves.toBe(false) + }) + + it('walks the POSIX refs directory for a native repo', async () => { + readRepoCommonDirFromGitMock.mockResolvedValue('/repo/.git') + + await expect(target().resolveRefsDirectory(NO_ABORT)).resolves.toBe('/repo/.git/refs') + }) + + it('translates a WSL repo answer back to the UNC path the main process can open', async () => { + // Git answers in its own execution space, which for WSL is a Linux path. + readRepoCommonDirFromGitMock.mockResolvedValue('/home/dev/repo/.git') + + await expect(target('Ubuntu').resolveRefsDirectory(NO_ABORT)).resolves.toBe( + '\\\\wsl.localhost\\Ubuntu\\home\\dev\\repo\\.git\\refs' + ) + }) + + it('reports an unresolvable repository rather than guessing a path', async () => { + readRepoCommonDirFromGitMock.mockResolvedValue(undefined) + + await expect(target().resolveRefsDirectory(NO_ABORT)).resolves.toBeUndefined() + }) +}) + +describe('local repo ref maintenance scheduling', () => { + it('schedules nothing when the kill switch is set', () => { + process.env.ORCA_DISABLE_REPO_REF_MAINTENANCE = '1' + const arm = vi.spyOn(getLocalRepoRefMaintenance(), 'arm') + + armLocalRepoRefMaintenance({ key: 'local::/repo/.git', repoPath: '/repo' }) + + expect(arm).not.toHaveBeenCalled() + }) + + it('arms through the shared single-flight instance otherwise', () => { + const arm = vi.spyOn(getLocalRepoRefMaintenance(), 'arm') + + armLocalRepoRefMaintenance({ key: 'local::/repo/.git', repoPath: '/repo' }) + + expect(arm).toHaveBeenCalledTimes(1) + }) + + it('is free when nothing has ever been armed', async () => { + // The common case by far: no timers, no instance, no reason to pay anything. + await expect(withRepoRefMaintenancePaused('git-fetch', async () => 'done')).resolves.toBe( + 'done' + ) + }) + + it('holds the window shut for the duration of ref-touching work', async () => { + readRepoCommonDirFromGitMock.mockResolvedValue('/repo/.git') + _resetLocalRepoRefMaintenanceForTests({ quietPeriodMs: 1, looseRefThreshold: 0 }) + setRepoMaintenanceActivityProbe(() => false) + const maintenance = getLocalRepoRefMaintenance() + const packRefs = vi.fn(async () => {}) + maintenance.arm({ + key: 'local::/repo/.git', + resolveRefsDirectory: async () => '/repo/.git/refs', + packRefs + }) + + await withRepoRefMaintenancePaused('branch-delete', async () => { + await new Promise((resolve) => setTimeout(resolve, 25)) + expect(packRefs).not.toHaveBeenCalled() + }) + + await vi.waitFor(() => expect(packRefs).toHaveBeenCalledTimes(1)) + }) + + it('routes the app activity probe into the shared instance', async () => { + readRepoCommonDirFromGitMock.mockResolvedValue('/repo/.git') + let busy = true + setRepoMaintenanceActivityProbe(() => busy) + const maintenance = getLocalRepoRefMaintenance() + const packRefs = vi.fn(async () => {}) + + maintenance.arm({ + key: 'local::/repo/.git', + resolveRefsDirectory: async () => '/repo/.git/refs', + packRefs + }) + await maintenance.whenAttemptSettled() + + expect(packRefs).not.toHaveBeenCalled() + busy = false + }) +}) diff --git a/src/main/git/local-repo-ref-maintenance.ts b/src/main/git/local-repo-ref-maintenance.ts new file mode 100644 index 00000000000..e84f7d1ae30 --- /dev/null +++ b/src/main/git/local-repo-ref-maintenance.ts @@ -0,0 +1,272 @@ +import { posix, win32 } from 'node:path' +import { isWindowsAbsolutePathLike } from '../../shared/cross-platform-path' +import { RepoRefMaintenance } from '../../shared/repo-ref-maintenance' +import { + PACK_REFS_ARGS, + PACK_REFS_TIMEOUT_MS, + RefMaintenanceRepoLocked, + type PackedRefsLockReporter, + type RepoRefMaintenanceOptions, + type RepoRefMaintenanceTarget +} from '../../shared/repo-ref-maintenance-policy' +import { isWslUncPath, toWindowsWslPath } from '../../shared/wsl-paths' +import { withSpan } from '../observability/tracer' +import { PackRefsLockOwnership } from './pack-refs-lock-ownership' +import { gitExecFileAsync } from './runner' +import { readRepoCommonDirFromGit } from './worktree-list-reader' + +/** + * Main-process wiring for idle loose-ref packing on the local execution host + * (native and WSL). + * + * SSH-hosted repos are deliberately out of scope: the execution host owns + * anything that touches execution, so maintaining them means running host-side + * on the relay, which today has neither admission control nor spans. Keying all + * state by execution host is what keeps this path from reaching across. + */ + +export type RepoMaintenanceActivityProbe = () => boolean + +const REPO_BUSY_PROBE_MAX = 64 + +let activityProbe: RepoMaintenanceActivityProbe | null = null +let shared: RepoRefMaintenance | null = null +// Why keyed here rather than captured in the target: a repo can be armed from +// the fetch controller or from a user-initiated fetch, and every arming must see +// the same "this repo has work in flight" answer, not whichever closure was last. +const repoBusyProbes = new Map boolean>() + +/** Register the owner of "this repo has a fetch in flight" for `key`. */ +export function setRepoRefMaintenanceBusyProbe(key: string, probe: () => boolean): void { + repoBusyProbes.delete(key) + repoBusyProbes.set(key, probe) + while (repoBusyProbes.size > REPO_BUSY_PROBE_MAX) { + const oldest = repoBusyProbes.keys().next() + if (oldest.done) { + break + } + repoBusyProbes.delete(oldest.value) + } +} + +/** + * Register the app-wide "do not start maintenance now" signal. Owned by the + * main entry point because the inputs (live agents, battery, quit) are not + * visible from the git layer. + */ +export function setRepoMaintenanceActivityProbe(probe: RepoMaintenanceActivityProbe | null): void { + activityProbe = probe +} + +/** Support escape hatch: kills the sweep without touching the user's git config. */ +function isDisabled(): boolean { + return process.env.ORCA_DISABLE_REPO_REF_MAINTENANCE === '1' +} + +function localMaintenanceOptions(): RepoRefMaintenanceOptions { + return { + // Fail closed: without the app-level gate installed we cannot see agents, + // creates, or battery, and running blind is worse than not running. + isBusy: () => activityProbe?.() ?? true, + observe: (attempt) => + withSpan('repo.ref_maintenance', (span) => attempt(span), { + attributes: { kind: 'git', 'repo.maintenance_host': 'local' } + }), + onError: (error) => { + console.warn('[repo-ref-maintenance] attempt failed:', error) + } + } +} + +export function getLocalRepoRefMaintenance(): RepoRefMaintenance { + shared ??= new RepoRefMaintenance(localMaintenanceOptions()) + return shared +} + +/** + * Cancels every armed timer and waits out any `packed-refs` rewrite in progress. + * + * Deliberately does not kill the child. A pack orphaned by the app quitting + * finishes on its own; a pack signalled mid-prune strands a ref lock about one + * time in five, and on Windows a force-kill inside the rewrite strands + * `packed-refs.lock` every time -- which blocks every later ref deletion. + */ +export function disposeLocalRepoRefMaintenance(): Promise { + const settling = shared?.awaitPackedRefsLockRelease() ?? Promise.resolve() + shared?.dispose() + shared = null + repoBusyProbes.clear() + return settling +} + +/** + * Hold every repository open while `run` touches refs. + * + * A ref deletion needs `packed-refs.lock`, which a running pack holds only while + * it rewrites the file -- 0.03-1.37s of a 23-32s run. Waiting that out turns the + * collision into a short pause. Cancelling the pack instead would strand a + * `refs/**` lock about one time in five, which Git never clears, so the ref + * stays undeletable indefinitely. + */ +export async function withRepoRefMaintenancePaused( + reason: string, + run: () => Promise +): Promise { + // Taken unconditionally rather than only when something is already armed: a + // fetch inside `run` can arm the sweep, and one counter bump against an idle + // instance costs a microtask. This can rebuild the instance after the + // quit-time dispose; harmless, because a fresh one has no armed timers and its + // activity probe is gone, so it fails closed. + const release = await getLocalRepoRefMaintenance().pause(reason) + try { + return await run() + } finally { + release() + } +} + +/** Wait out a `packed-refs` rewrite without holding the window open. For shutdown. */ +export function awaitPackedRefsLockRelease(): Promise { + return shared ? shared.awaitPackedRefsLockRelease() : Promise.resolve() +} + +/** + * Count user-initiated ref work as activity and restart every armed countdown. + * + * Deliberately not keyed to a repo: resolving one would cost a `rev-parse` on a + * path the user is waiting on, and a manual fetch or pull says the user is at + * the keyboard, which is a reason to defer every repository. + */ +export function postponeRepoRefMaintenance(): void { + shared?.postponeAll() +} + +/** `overrides` preseeds the shared instance so a test can shorten the quiet period. */ +export function _resetLocalRepoRefMaintenanceForTests( + overrides?: Partial +): void { + shared?.dispose() + shared = overrides ? new RepoRefMaintenance({ ...localMaintenanceOptions(), ...overrides }) : null + activityProbe = null + repoBusyProbes.clear() +} + +/** + * Git reports the common dir in its own execution space, so a WSL repo answers + * with a Linux path the Windows main process cannot open. Translate it back to + * the UNC spelling for the dirent walk; the walk reads directories, not files, + * so the handful of round trips stays cheap even over the share. + */ +function refsDirectoryForMainProcess(commonDir: string, wslDistro: string | undefined): string { + if (wslDistro && !isWslUncPath(commonDir) && !isWindowsAbsolutePathLike(commonDir)) { + return win32.join(toWindowsWslPath(commonDir, wslDistro), 'refs') + } + // Decided by path syntax, not by platform: `win32.isAbsolute` accepts POSIX paths too. + return (isWindowsAbsolutePathLike(commonDir) ? win32 : posix).join(commonDir, 'refs') +} + +/** + * `maintenance.auto=false` and `gc.auto=0` are the two knobs a user reaches for + * to tell Git to stop maintaining a repository on its own. Orca sets both on its + * own fetches, but only as per-invocation `-c` flags, so this probe sees the + * user's persisted config and never Orca's own suppression. + */ +export function isGitAutoMaintenanceDisabled(configOutput: string): boolean { + return configOutput + .split('\n') + .map((line) => line.trim()) + .some((line) => line === 'maintenance.auto false' || line === 'gc.auto 0') +} + +/** + * The common dir in the spelling the main process can open. + * + * Derived from the converted refs path, not the raw one: a WSL answer arrives as + * a Linux path but converts to a UNC path with no `/` in it, so choosing the + * path flavour before conversion collapses the whole thing to `.`. + */ +function gitCommonDirForMainProcess(commonDir: string, wslDistro: string | undefined): string { + const refs = refsDirectoryForMainProcess(commonDir, wslDistro) + return (isWindowsAbsolutePathLike(refs) ? win32 : posix).dirname(refs) +} + +export type LocalRepoRefMaintenanceTargetArgs = { + /** `${runtimeKey}::${gitCommonDir}` -- already scoped to the execution host. */ + readonly key: string + readonly repoPath: string + readonly wslDistro?: string +} + +/** + * Record a write to this repo and restart its quiet-period countdown. The only + * entry point callers need: the kill switch is honoured before anything is + * scheduled, so a disabled build arms no timers at all. + */ +export function armLocalRepoRefMaintenance(args: LocalRepoRefMaintenanceTargetArgs): void { + if (isDisabled()) { + return + } + getLocalRepoRefMaintenance().arm(createLocalRepoRefMaintenanceTarget(args)) +} + +export function createLocalRepoRefMaintenanceTarget( + args: LocalRepoRefMaintenanceTargetArgs +): RepoRefMaintenanceTarget { + const gitOptions = args.wslDistro ? { wslDistro: args.wslDistro } : {} + // The engine always probes before it packs, so the pack reuses this answer + // rather than spending a second rev-parse on the same repository. + let commonDir: string | undefined + const resolveCommonDir = async (signal?: AbortSignal): Promise => { + commonDir ??= await readRepoCommonDirFromGit(args.repoPath, { + ...gitOptions, + ...(signal ? { signal } : {}) + }) + return commonDir + } + return { + key: args.key, + isBusy: () => repoBusyProbes.get(args.key)?.() ?? false, + async resolveRefsDirectory(signal: AbortSignal) { + const resolved = await resolveCommonDir(signal) + return resolved ? refsDirectoryForMainProcess(resolved, args.wslDistro) : undefined + }, + async isOptedOut(signal: AbortSignal) { + try { + const { stdout } = await gitExecFileAsync( + ['config', '--get-regexp', '^(maintenance\\.auto|gc\\.auto)$'], + { cwd: args.repoPath, ...gitOptions, admissionTier: 'background', signal } + ) + return isGitAutoMaintenanceDisabled(stdout) + } catch { + // Neither key set is the common case and exits non-zero; that is consent. + return false + } + }, + async packRefs(lock: PackedRefsLockReporter) { + const resolved = await resolveCommonDir() + const owner = resolved + ? new PackRefsLockOwnership(gitCommonDirForMainProcess(resolved, args.wslDistro)) + : null + const claim = owner ? await owner.claim() : { ok: true as const } + if (!claim.ok) { + throw new RefMaintenanceRepoLocked(claim.reason) + } + // Report the rewrite window rather than accepting a signal. A pack that is + // killed mid-prune strands a `refs/**` lock about one time in five, and + // Git never clears those; waiting out the window costs at most ~1.4s. + const watch = owner?.watchLock((held) => lock.setHeld(held)) + try { + await gitExecFileAsync([...PACK_REFS_ARGS], { + cwd: args.repoPath, + ...gitOptions, + admissionTier: 'background', + timeout: PACK_REFS_TIMEOUT_MS + }) + } finally { + watch?.stop() + lock.setHeld(false) + await owner?.release() + } + } + } +} diff --git a/src/main/git/pack-refs-lock-ownership.test.ts b/src/main/git/pack-refs-lock-ownership.test.ts new file mode 100644 index 00000000000..e181feb9976 --- /dev/null +++ b/src/main/git/pack-refs-lock-ownership.test.ts @@ -0,0 +1,192 @@ +import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { PackRefsLockOwnership } from './pack-refs-lock-ownership' + +const roots: string[] = [] + +async function gitCommonDir(): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-pack-refs-lock-')) + roots.push(root) + return root +} + +function paths(commonDir: string): { lock: string; marker: string } { + return { + lock: join(commonDir, 'packed-refs.lock'), + marker: join(commonDir, 'packed-refs.orca-owner') + } +} + +async function exists(path: string): Promise { + try { + await stat(path) + return true + } catch { + return false + } +} + +/** A pid that cannot be running: the kernel rejects it outright. */ +const DEAD_PID = 0x7fffffff +const ABANDONED_LOCK_AGE_MS = 15 * 60_000 +const PID_REUSE_HORIZON_MS = 24 * 60 * 60_000 + +/** `claim` takes `now`, so age cases need no sleeping and no mtime forgery. */ +function laterBy(ms: number): number { + return Date.now() + ms +} + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('packed-refs lock ownership', () => { + it('claims a repository with no lock and records the owner', async () => { + const commonDir = await gitCommonDir() + const { marker } = paths(commonDir) + + await expect(new PackRefsLockOwnership(commonDir).claim()).resolves.toEqual({ ok: true }) + + await expect(readFile(marker, 'utf-8')).resolves.toContain(String(process.pid)) + }) + + it('drops the owner marker on release', async () => { + const commonDir = await gitCommonDir() + const ownership = new PackRefsLockOwnership(commonDir) + await ownership.claim() + + await ownership.release() + + await expect(exists(paths(commonDir).marker)).resolves.toBe(false) + }) + + it('refuses a lock it cannot prove is its own', async () => { + const commonDir = await gitCommonDir() + // A lock with no marker belongs to the user's own git, or to another tool. + await writeFile(paths(commonDir).lock, 'someone else') + + await expect(new PackRefsLockOwnership(commonDir).claim()).resolves.toMatchObject({ ok: false }) + await expect(exists(paths(commonDir).lock)).resolves.toBe(true) + }) + + it('refuses a lock whose recorded owner is still running', async () => { + const commonDir = await gitCommonDir() + const { lock, marker } = paths(commonDir) + await writeFile(lock, 'in progress') + await writeFile(marker, JSON.stringify({ pid: process.pid })) + + await expect( + new PackRefsLockOwnership(commonDir).claim(laterBy(ABANDONED_LOCK_AGE_MS + 1)) + ).resolves.toMatchObject({ ok: false }) + await expect(exists(lock)).resolves.toBe(true) + }) + + it('reclaims the lock its own dead process left behind', async () => { + // SIGKILL and power loss bypass git's cleanup, and git never clears this itself. + const commonDir = await gitCommonDir() + const { lock, marker } = paths(commonDir) + await writeFile(lock, 'abandoned mid-rewrite') + await writeFile(marker, JSON.stringify({ pid: DEAD_PID })) + + const claimed = await new PackRefsLockOwnership(commonDir).claim( + laterBy(ABANDONED_LOCK_AGE_MS + 1) + ) + + expect(claimed).toEqual({ ok: true }) + await expect(exists(lock)).resolves.toBe(false) + await expect(readFile(marker, 'utf-8')).resolves.toContain(String(process.pid)) + }) + + it('leaves a young lock alone even when the marker names a dead process', async () => { + // A marker outlives its lock, so a foreign lock can appear after our death. + // Age is the only thing separating our wreckage from somebody's live lock. + const commonDir = await gitCommonDir() + const { lock, marker } = paths(commonDir) + await writeFile(marker, JSON.stringify({ pid: DEAD_PID })) + await writeFile(lock, 'a different git process, started just now') + + await expect(new PackRefsLockOwnership(commonDir).claim()).resolves.toMatchObject({ ok: false }) + await expect(exists(lock)).resolves.toBe(true) + }) + + it('does not wedge a repository forever when the recorded pid was recycled', async () => { + const commonDir = await gitCommonDir() + const { lock, marker } = paths(commonDir) + await writeFile(lock, 'abandoned mid-rewrite') + // Our own pid stands in for a recycled one: alive, but not the process that wrote this. + await writeFile(marker, JSON.stringify({ pid: process.pid })) + + await expect( + new PackRefsLockOwnership(commonDir).claim(laterBy(ABANDONED_LOCK_AGE_MS + 1)) + ).resolves.toMatchObject({ ok: false }) + + await expect( + new PackRefsLockOwnership(commonDir).claim(laterBy(PID_REUSE_HORIZON_MS + 1)) + ).resolves.toEqual({ ok: true }) + await expect(exists(lock)).resolves.toBe(false) + }) + + it('refuses a lock whose marker is unreadable rather than guessing', async () => { + const commonDir = await gitCommonDir() + const { lock, marker } = paths(commonDir) + await writeFile(lock, 'in progress') + await writeFile(marker, 'not json') + + await expect( + new PackRefsLockOwnership(commonDir).claim(laterBy(PID_REUSE_HORIZON_MS + 1)) + ).resolves.toMatchObject({ ok: false }) + await expect(exists(lock)).resolves.toBe(true) + }) + + it('claims cleanly when a marker outlived its lock', async () => { + const commonDir = await gitCommonDir() + await writeFile(paths(commonDir).marker, JSON.stringify({ pid: DEAD_PID })) + + await expect(new PackRefsLockOwnership(commonDir).claim()).resolves.toEqual({ ok: true }) + }) +}) + +describe('stranded per-ref locks', () => { + it('clears the empty refs/**/*.lock files its own dead process left behind', async () => { + // `tempfile.c` opens the lock O_EXCL before linking it into the list the + // signal handler walks, so a kill in that window leaves a 0-byte file that + // Git never clears -- and `update-ref -d` on that ref then fails forever. + const commonDir = await gitCommonDir() + const namespace = join(commonDir, 'refs', 'remotes', 'origin') + await mkdir(namespace, { recursive: true }) + await writeFile(join(namespace, 'main.lock'), '') + await writeFile(join(namespace, 'main'), 'a'.repeat(40)) + await writeFile(paths(commonDir).marker, JSON.stringify({ pid: DEAD_PID })) + + await new PackRefsLockOwnership(commonDir).claim(laterBy(ABANDONED_LOCK_AGE_MS + 1)) + + await expect(exists(join(namespace, 'main.lock'))).resolves.toBe(false) + // The ref itself is untouched. + await expect(exists(join(namespace, 'main'))).resolves.toBe(true) + }) + + it('leaves a non-empty ref lock alone, because a live writer is mid-write', async () => { + const commonDir = await gitCommonDir() + const namespace = join(commonDir, 'refs', 'heads') + await mkdir(namespace, { recursive: true }) + await writeFile(join(namespace, 'busy.lock'), 'b'.repeat(40)) + await writeFile(paths(commonDir).marker, JSON.stringify({ pid: DEAD_PID })) + + await new PackRefsLockOwnership(commonDir).claim(laterBy(ABANDONED_LOCK_AGE_MS + 1)) + + await expect(exists(join(namespace, 'busy.lock'))).resolves.toBe(true) + }) + + it('leaves ref locks alone when there is no marker naming a dead process', async () => { + const commonDir = await gitCommonDir() + const namespace = join(commonDir, 'refs', 'heads') + await mkdir(namespace, { recursive: true }) + await writeFile(join(namespace, 'other.lock'), '') + + await new PackRefsLockOwnership(commonDir).claim(laterBy(ABANDONED_LOCK_AGE_MS + 1)) + + await expect(exists(join(namespace, 'other.lock'))).resolves.toBe(true) + }) +}) diff --git a/src/main/git/pack-refs-lock-ownership.ts b/src/main/git/pack-refs-lock-ownership.ts new file mode 100644 index 00000000000..9e7273b143b --- /dev/null +++ b/src/main/git/pack-refs-lock-ownership.ts @@ -0,0 +1,202 @@ +import { readFile, readdir, rm, stat, writeFile } from 'node:fs/promises' +import { posix, win32 } from 'node:path' +import { isWindowsAbsolutePathLike } from '../../shared/cross-platform-path' +import { + PACK_REFS_TIMEOUT_MS, + PACKED_REFS_LOCK_POLL_MS +} from '../../shared/repo-ref-maintenance-policy' + +/** No legitimate `pack-refs` outlives its own deadline, so an older lock is abandoned. */ +const ABANDONED_LOCK_AGE_MS = PACK_REFS_TIMEOUT_MS + +/** Beyond this a recorded pid may have been recycled, so it stops being evidence of life. */ +const PID_REUSE_HORIZON_MS = 24 * 60 * 60_000 + +/** The ref tree is wide but shallow; this only stops a pathological walk. */ +const REF_LOCK_SCAN_CEILING = 4096 + +/** + * Makes a `packed-refs.lock` Orca left behind attributable, and only that one. + * + * Git registers signal handlers that clean the lock up, but SIGKILL and power + * loss bypass them, and Git never removes a stale `packed-refs.lock` on its own + * -- every later ref deletion in that repository fails until someone deletes a + * file they have never heard of. Recording our pid beside the lock lets a later + * run recognise its own wreckage. + * + * Three independent conditions must all hold before anything is unlinked, + * because deleting a lock somebody else is holding is far worse than declining + * to pack: a marker must exist at all, the lock must be older than any + * `pack-refs` could legitimately run for, and the recorded process must be gone. + * A marker can outlive its lock, so age is what separates "our wreckage" from a + * foreign lock that happened to appear afterwards. + */ +export class PackRefsLockOwnership { + private readonly lockPath: string + private readonly markerPath: string + + constructor(gitCommonDir: string) { + const path = isWindowsAbsolutePathLike(gitCommonDir) ? win32 : posix + this.lockPath = path.join(gitCommonDir, 'packed-refs.lock') + this.markerPath = path.join(gitCommonDir, 'packed-refs.orca-owner') + } + + /** Refused when the lock belongs to something we cannot prove is our own wreckage. */ + async claim(now = Date.now()): Promise { + const reclaim = await this.reclaimAbandonedLock(now) + if (!reclaim.ok) { + return reclaim + } + // Per-ref strands outlive their pack and are invisible to Git, which never + // clears a `refs/**\/*.lock` it did not create in this process. + await this.reclaimStrandedRefLocks(now) + try { + await writeFile(this.markerPath, JSON.stringify({ pid: process.pid }), 'utf-8') + } catch { + // Losing the marker only costs attribution on the next run, never correctness. + } + return { ok: true } + } + + /** + * Poll `packed-refs.lock` so the scheduler knows when the exclusive rewrite + * window opens and closes. Cheap: one `stat` on a fixed path. + */ + watchLock(report: (held: boolean) => void): { stop: () => void } { + let stopped = false + let last = false + const tick = async (): Promise => { + if (stopped) { + return + } + const held = (await fileAgeMs(this.lockPath, Date.now())) !== null + if (!stopped && held !== last) { + last = held + report(held) + } + } + const timer = setInterval(() => void tick(), PACKED_REFS_LOCK_POLL_MS) + timer.unref?.() + void tick() + return { + stop: () => { + stopped = true + clearInterval(timer) + } + } + } + + async release(): Promise { + await rm(this.markerPath, { force: true }).catch(() => {}) + } + + private async reclaimAbandonedLock(now: number): Promise { + const lockAgeMs = await fileAgeMs(this.lockPath, now) + if (lockAgeMs === null) { + return { ok: true } + } + // No marker means the lock is not ours to reason about, let alone remove. + const marker = await readOwnerMarker(this.markerPath) + if (marker === null) { + return { ok: false, reason: 'held by another process' } + } + if (lockAgeMs < ABANDONED_LOCK_AGE_MS) { + // Ours, but too young to be certain the writer is gone. Worth retrying soon. + return { ok: false, reason: 'our own lock, not yet old enough to reclaim' } + } + // Past the pid-reuse horizon the pid proves nothing, and a lock this old is + // abandoned whoever wrote it -- otherwise a recycled pid would wedge the + // repository permanently. + if (isProcessAlive(marker.pid) && lockAgeMs < PID_REUSE_HORIZON_MS) { + return { ok: false, reason: 'the recorded owner is still running' } + } + await rm(this.lockPath, { force: true }).catch(() => {}) + await rm(this.markerPath, { force: true }).catch(() => {}) + return { ok: true } + } + + /** + * Clear `refs/**\/*.lock` files a dead pack of ours left behind. + * + * `tempfile.c` opens the lock `O_EXCL` before `activate_tempfile()` links it + * into the list the signal handler walks, so a kill inside that window leaves + * a 0-byte file. Afterwards `update-ref -d` and any fetch touching that ref + * fail with `cannot lock ref ... File exists`, forever. Same three conditions + * as the packed-refs lock, plus a size check: a live writer's lock is not empty. + */ + private async reclaimStrandedRefLocks(now: number): Promise { + const marker = await readOwnerMarker(this.markerPath) + if (marker === null || isProcessAlive(marker.pid)) { + return + } + const markerAgeMs = await fileAgeMs(this.markerPath, now) + if (markerAgeMs === null || markerAgeMs < ABANDONED_LOCK_AGE_MS) { + return + } + const path = isWindowsAbsolutePathLike(this.markerPath) ? win32 : posix + const pending = [path.join(path.dirname(this.markerPath), 'refs')] + let visited = 0 + while (pending.length > 0) { + const directory = pending.pop() + if (directory === undefined || (visited += 1) > REF_LOCK_SCAN_CEILING) { + return + } + let entries: { name: string; isDirectory: () => boolean }[] + try { + entries = await readdir(directory, { withFileTypes: true }) + } catch { + continue + } + for (const entry of entries) { + const full = path.join(directory, entry.name) + if (entry.isDirectory()) { + pending.push(full) + } else if (entry.name.endsWith('.lock') && (await isEmptyFile(full))) { + await rm(full, { force: true }).catch(() => {}) + } + } + } + } +} + +export type PackRefsLockClaim = { ok: true } | { ok: false; reason: string } + +/** A strand from the `O_EXCL` window is 0 bytes; a live writer's lock is not. */ +async function isEmptyFile(path: string): Promise { + try { + return (await stat(path)).size === 0 + } catch { + return false + } +} + +async function readOwnerMarker(path: string): Promise<{ pid: number } | null> { + try { + const raw = (await readFile(path, 'utf-8')).slice(0, 256) + const pid = (JSON.parse(raw) as { pid?: unknown }).pid + return typeof pid === 'number' && Number.isInteger(pid) && pid > 0 ? { pid } : null + } catch { + return null + } +} + +/** Null when the file does not exist. Uses stat: the lock holds a whole packed-refs. */ +async function fileAgeMs(path: string, now: number): Promise { + try { + return Math.max(0, now - (await stat(path)).mtimeMs) + } catch (error) { + return (error as NodeJS.ErrnoException).code === 'ENOENT' ? null : 0 + } +} + +function isProcessAlive(pid: number): boolean { + if (pid === process.pid) { + return true + } + try { + process.kill(pid, 0) + return true + } catch (error) { + return (error as NodeJS.ErrnoException).code !== 'ESRCH' + } +} diff --git a/src/main/git/remote.ts b/src/main/git/remote.ts index c1557bdd806..2baf3b77137 100644 --- a/src/main/git/remote.ts +++ b/src/main/git/remote.ts @@ -7,6 +7,10 @@ import { gitRefTargetsBranchOnRemote } from '../../shared/git-remote-branch-name import type { GitPushTarget } from '../../shared/worktree/types' import type { GitRuntimeOptions } from './git-runtime-options' import { gitOptionsForWorktree } from './git-runtime-options' +import { + postponeRepoRefMaintenance, + withRepoRefMaintenancePaused +} from './local-repo-ref-maintenance' import { validateGitPushTarget } from './push-target-validation' import { gitExecFileAsync } from './runner' import { fetchForkRemoteWithStaleRefspecRepair } from './fork-remote-stale-branch-refspec' @@ -260,8 +264,11 @@ export async function gitPull( // Why: plain `git pull` uses the user's configured pull strategy (merge by // default) so diverged branches reconcile instead of erroring out. Conflicts // surface through the existing conflict-resolution flow. - await runWithGitWorktreeOperationLock(worktreePath, options.signal, () => - runWithGitReadCacheInvalidation(() => gitPullWithArgs(worktreePath, [], pushTarget, options)) + postponeRepoRefMaintenance() + await withRepoRefMaintenancePaused('git-pull', () => + runWithGitWorktreeOperationLock(worktreePath, options.signal, () => + runWithGitReadCacheInvalidation(() => gitPullWithArgs(worktreePath, [], pushTarget, options)) + ) ) } @@ -270,9 +277,12 @@ export async function gitFastForward( pushTarget?: GitPushTarget, options: GitRuntimeOptions = {} ): Promise { - await runWithGitWorktreeOperationLock(worktreePath, options.signal, () => - runWithGitReadCacheInvalidation(() => - gitPullWithArgs(worktreePath, ['--ff-only'], pushTarget, options) + postponeRepoRefMaintenance() + await withRepoRefMaintenancePaused('git-fast-forward', () => + runWithGitWorktreeOperationLock(worktreePath, options.signal, () => + runWithGitReadCacheInvalidation(() => + gitPullWithArgs(worktreePath, ['--ff-only'], pushTarget, options) + ) ) ) } @@ -282,22 +292,28 @@ export async function gitFetch( pushTarget?: GitPushTarget, options: GitRuntimeOptions = {} ): Promise { + // `--prune` deletes remote-tracking refs, which needs the `packed-refs` lock a + // running idle pack holds while it rewrites -- ~1.4s at most. This is the user + // clicking Fetch, so wait that window out rather than letting it fail on the lock. + postponeRepoRefMaintenance() try { - if (pushTarget) { - const target = await validateGitPushTarget(worktreePath, pushTarget, options) - const runtimeOptions = gitOptionsForWorktree(worktreePath, options) - await fetchForkRemoteWithStaleRefspecRepair( - (args, cwd) => gitExecFileAsync(args, { ...runtimeOptions, cwd }), - worktreePath, - target.remoteName, - () => - gitExecFileAsync(['fetch', '--prune', target.remoteName], runtimeOptions).then( - () => undefined - ) - ) - return - } - await gitExecFileAsync(['fetch', '--prune'], gitOptionsForWorktree(worktreePath, options)) + await withRepoRefMaintenancePaused('git-fetch', async () => { + if (pushTarget) { + const target = await validateGitPushTarget(worktreePath, pushTarget, options) + const runtimeOptions = gitOptionsForWorktree(worktreePath, options) + await fetchForkRemoteWithStaleRefspecRepair( + (args, cwd) => gitExecFileAsync(args, { ...runtimeOptions, cwd }), + worktreePath, + target.remoteName, + () => + gitExecFileAsync(['fetch', '--prune', target.remoteName], runtimeOptions).then( + () => undefined + ) + ) + return + } + await gitExecFileAsync(['fetch', '--prune'], gitOptionsForWorktree(worktreePath, options)) + }) } catch (error) { throw new Error(normalizeGitErrorMessage(error, 'fetch')) } diff --git a/src/main/git/repo-ref-maintenance-real-git.test.ts b/src/main/git/repo-ref-maintenance-real-git.test.ts new file mode 100644 index 00000000000..30c67b0365a --- /dev/null +++ b/src/main/git/repo-ref-maintenance-real-git.test.ts @@ -0,0 +1,290 @@ +import { execFileSync } from 'node:child_process' +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { countLooseRefs } from '../../shared/loose-ref-count' +import { RepoRefMaintenance } from '../../shared/repo-ref-maintenance' +import { + _resetLocalRepoRefMaintenanceForTests, + createLocalRepoRefMaintenanceTarget, + getLocalRepoRefMaintenance, + setRepoMaintenanceActivityProbe +} from './local-repo-ref-maintenance' +import { forceDeleteLocalBranch } from './worktree-branch-removal' + +const roots: string[] = [] +// Large enough that the deferral ladder (1x, 2x, 4x ... capped at 8x) outlasts +// three real `pack-refs` runs before the deferral budget is spent. +const QUIET_MS = 25 +const THRESHOLD = 20 + +function git(cwd: string, args: string[]): string { + return execFileSync('git', args, { + cwd, + encoding: 'utf8', + stdio: ['pipe', 'pipe', 'pipe'] + }).trim() +} + +/** A repo whose only loose-ref backlog is the one the test asks for. */ +async function createRepo(looseRefs: number): Promise<{ repoPath: string; refsDir: string }> { + const root = await mkdtemp(join(tmpdir(), 'orca-ref-maintenance-git-')) + roots.push(root) + const repoPath = join(root, 'repo') + execFileSync('git', ['init', '--quiet', repoPath]) + git(repoPath, ['symbolic-ref', 'HEAD', 'refs/heads/main']) + git(repoPath, ['config', 'user.email', 'test@example.com']) + git(repoPath, ['config', 'user.name', 'Test User']) + await writeFile(join(repoPath, 'file.txt'), 'one\n') + git(repoPath, ['add', 'file.txt']) + git(repoPath, ['commit', '--quiet', '-m', 'initial']) + const head = git(repoPath, ['rev-parse', 'HEAD']) + // Written directly: `update-ref` for thousands of refs is the slow part of the fixture. + const namespace = join(repoPath, '.git', 'refs', 'remotes', 'origin') + await mkdir(namespace, { recursive: true }) + for (let index = 0; index < looseRefs; index += 1) { + await writeFile(join(namespace, `branch-${index}`), `${head}\n`) + } + return { repoPath, refsDir: join(repoPath, '.git', 'refs') } +} + +function createMaintenance(onPackRefs: () => void = () => {}): { + maintenance: RepoRefMaintenance + arm: (repoPath: string) => void +} { + const maintenance = new RepoRefMaintenance({ + quietPeriodMs: QUIET_MS, + looseRefThreshold: THRESHOLD + }) + return { + maintenance, + arm: (repoPath: string) => { + const target = createLocalRepoRefMaintenanceTarget({ + key: `local::${repoPath}`, + repoPath + }) + maintenance.arm({ + ...target, + packRefs: async (signal) => { + onPackRefs() + await target.packRefs(signal) + } + }) + } + } +} + +async function settle(maintenance: RepoRefMaintenance): Promise { + await new Promise((resolve) => setTimeout(resolve, QUIET_MS * 4)) + await maintenance.whenAttemptSettled() +} + +/** Deferred repos re-arm for another quiet period, so drain rather than count rounds. */ +async function settleUntil( + maintenance: RepoRefMaintenance, + done: () => Promise +): Promise { + for (let round = 0; round < 100; round += 1) { + if (await done()) { + return + } + await settle(maintenance) + } +} + +afterEach(async () => { + _resetLocalRepoRefMaintenanceForTests() + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('idle ref maintenance against real Git', () => { + it('packs a backlogged repository down to zero loose refs', async () => { + const { repoPath, refsDir } = await createRepo(THRESHOLD + 30) + const { maintenance, arm } = createMaintenance() + + await expect(countLooseRefs(refsDir, 10_000)).resolves.toMatchObject({ + count: THRESHOLD + 31 + }) + + arm(repoPath) + await settle(maintenance) + maintenance.dispose() + + await expect(countLooseRefs(refsDir, 10_000)).resolves.toEqual({ count: 0, saturated: false }) + // The refs survived the move into packed-refs; nothing was lost. + expect(git(repoPath, ['for-each-ref', '--format=%(refname)']).split('\n')).toHaveLength( + THRESHOLD + 31 + ) + expect(git(repoPath, ['rev-parse', '--verify', 'refs/remotes/origin/branch-0'])).toMatch( + /^[0-9a-f]{40}$/ + ) + }, 30_000) + + it('leaves a healthy repository untouched', async () => { + const { repoPath, refsDir } = await createRepo(2) + let packed = 0 + const { maintenance, arm } = createMaintenance(() => { + packed += 1 + }) + + arm(repoPath) + await settle(maintenance) + maintenance.dispose() + + expect(packed).toBe(0) + await expect(countLooseRefs(refsDir, 10_000)).resolves.toMatchObject({ count: 3 }) + }, 30_000) + + it('honours maintenance.auto=false in the repository config', async () => { + const { repoPath, refsDir } = await createRepo(THRESHOLD + 30) + git(repoPath, ['config', 'maintenance.auto', 'false']) + let packed = 0 + const { maintenance, arm } = createMaintenance(() => { + packed += 1 + }) + + arm(repoPath) + await settle(maintenance) + maintenance.dispose() + + expect(packed).toBe(0) + await expect(countLooseRefs(refsDir, 10_000)).resolves.toMatchObject({ + count: THRESHOLD + 31 + }) + }, 30_000) + + it('runs one repository at a time even when several go quiet together', async () => { + const repos = await Promise.all([ + createRepo(THRESHOLD + 5), + createRepo(THRESHOLD + 5), + createRepo(THRESHOLD + 5) + ]) + let concurrent = 0 + let peak = 0 + const maintenance = new RepoRefMaintenance({ + quietPeriodMs: QUIET_MS, + looseRefThreshold: THRESHOLD + }) + for (const { repoPath } of repos) { + const target = createLocalRepoRefMaintenanceTarget({ + key: `local::${repoPath}`, + repoPath + }) + maintenance.arm({ + ...target, + packRefs: async (signal) => { + concurrent += 1 + peak = Math.max(peak, concurrent) + try { + await target.packRefs(signal) + } finally { + concurrent -= 1 + } + } + }) + } + + const allPacked = async (): Promise => { + const counts = await Promise.all(repos.map(({ refsDir }) => countLooseRefs(refsDir, 10_000))) + return counts.every((scan) => scan.count === 0) + } + await settleUntil(maintenance, allPacked) + maintenance.dispose() + + expect(peak).toBe(1) + for (const { refsDir } of repos) { + await expect(countLooseRefs(refsDir, 10_000)).resolves.toEqual({ + count: 0, + saturated: false + }) + } + }, 60_000) +}) + +describe('yielding the repository to work that deletes refs', () => { + it('waits for the packed-refs lock and succeeds while the prune continues', async () => { + // The pack is never killed. `packed-refs.lock` is held for ~1.4s of a 30s + // run; the rest is the prune, during which a concurrent `update-ref -d` + // succeeds on its own because per-ref locks last microseconds. Signalling + // the child there strands a `refs/**` lock Git never clears. + const { repoPath } = await createRepo(0) + git(repoPath, ['branch', 'doomed']) + const head = git(repoPath, ['rev-parse', 'refs/heads/doomed']) + + let packing = false + let releaseLock: (() => void) | undefined + _resetLocalRepoRefMaintenanceForTests({ quietPeriodMs: QUIET_MS, looseRefThreshold: 1 }) + setRepoMaintenanceActivityProbe(() => false) + getLocalRepoRefMaintenance().arm({ + key: `local::${repoPath}`, + resolveRefsDirectory: async () => join(repoPath, '.git', 'refs'), + packRefs: async (lock) => { + packing = true + lock.setHeld(true) + // Stands in for the rewrite window, then the long prune that follows it. + await new Promise((resolve) => { + releaseLock = () => { + lock.setHeld(false) + resolve() + } + }) + } + }) + for (let attempt = 0; attempt < 200 && !packing; attempt += 1) { + await new Promise((resolve) => setTimeout(resolve, QUIET_MS)) + } + expect(packing).toBe(true) + + // The real deletion path, which routes through withRepoRefMaintenancePaused. + let deleted = false + const deletion = forceDeleteLocalBranch(repoPath, 'doomed', head).then(() => { + deleted = true + }) + + // It must still be waiting: the rewrite window is open. + await new Promise((resolve) => setTimeout(resolve, QUIET_MS * 4)) + expect(deleted).toBe(false) + expect(git(repoPath, ['branch', '--list', 'doomed'])).toContain('doomed') + + // Releasing the window is enough -- the pack is never cancelled. + releaseLock?.() + await deletion + expect(deleted).toBe(true) + expect(git(repoPath, ['branch', '--list', 'doomed'])).toBe('') + }, 30_000) + + it('does not block the caller once the rewrite window has closed', async () => { + // The prune phase is concurrency-safe, so a caller arriving during it pays + // nothing at all. + const { repoPath } = await createRepo(0) + git(repoPath, ['branch', 'doomed']) + const head = git(repoPath, ['rev-parse', 'refs/heads/doomed']) + + let pruning = false + let finishPrune: (() => void) | undefined + _resetLocalRepoRefMaintenanceForTests({ quietPeriodMs: QUIET_MS, looseRefThreshold: 1 }) + setRepoMaintenanceActivityProbe(() => false) + getLocalRepoRefMaintenance().arm({ + key: `local::${repoPath}`, + resolveRefsDirectory: async () => join(repoPath, '.git', 'refs'), + packRefs: async (lock) => { + lock.setHeld(true) + lock.setHeld(false) + pruning = true + await new Promise((resolve) => { + finishPrune = resolve + }) + } + }) + for (let attempt = 0; attempt < 200 && !pruning; attempt += 1) { + await new Promise((resolve) => setTimeout(resolve, QUIET_MS)) + } + + const startedAt = Date.now() + await expect(forceDeleteLocalBranch(repoPath, 'doomed', head)).resolves.toBeUndefined() + expect(Date.now() - startedAt).toBeLessThan(2_000) + + finishPrune?.() + }, 30_000) +}) diff --git a/src/main/git/worktree-add.ts b/src/main/git/worktree-add.ts index 3cd761f4d13..ea6ec704b46 100644 --- a/src/main/git/worktree-add.ts +++ b/src/main/git/worktree-add.ts @@ -4,6 +4,7 @@ import type { LocalBaseRefUpdateSuggestion } from '../../shared/worktree/base-ref-drift-types' import { windowsLongPathGitArgs } from '../../shared/windows-long-path-git-args' +import { withRepoRefMaintenancePaused } from './local-repo-ref-maintenance' import { gitExecFileAsync } from './runner' import { runWithGitReadCacheInvalidation } from './status' import { invalidateWslLinkedWorktreeGitRouting } from './wsl-linked-worktree-git-routing' @@ -149,15 +150,17 @@ export async function addWorktree( options: AddWorktreeOptions = {} ): Promise { try { - return await runWithGitReadCacheInvalidation(() => - performAddWorktree( - repoPath, - worktreePath, - branch, - baseBranch, - refreshLocalBaseRef, - noCheckout, - options + return await withRepoRefMaintenancePaused('worktree-add', () => + runWithGitReadCacheInvalidation(() => + performAddWorktree( + repoPath, + worktreePath, + branch, + baseBranch, + refreshLocalBaseRef, + noCheckout, + options + ) ) ) } finally { diff --git a/src/main/git/worktree-branch-removal.ts b/src/main/git/worktree-branch-removal.ts index 5e64b4a582d..08b6b21a1e3 100644 --- a/src/main/git/worktree-branch-removal.ts +++ b/src/main/git/worktree-branch-removal.ts @@ -4,6 +4,7 @@ import { } from '../../shared/git-branch-cleanup' import type { RemoveWorktreeResult } from '../../shared/worktree/create-types' import { withLocalGitCapabilityCacheForExecution } from './git-capability-state' +import { withRepoRefMaintenancePaused } from './local-repo-ref-maintenance' import { gitExecFileAsync } from './runner' import { parseWorktreeList } from './worktree-list-parser' import type { GitWorktreeExecOptions, RemoveWorktreeOptions } from './worktree-operation-options' @@ -152,7 +153,11 @@ export async function forceDeleteLocalBranch( } // Why: stale toast actions must not delete a branch that moved; `update-ref -d` deletes only if the ref still == expectedHead. try { - await runGit(['update-ref', '-d', `refs/heads/${branchName}`, expectedHead], repoPath) + // `update-ref -d` needs the packed-refs lock a running idle pack holds while + // it rewrites; waits it out rather than cancelling the pack. + await withRepoRefMaintenancePaused('branch-delete', () => + runGit(['update-ref', '-d', `refs/heads/${branchName}`, expectedHead], repoPath) + ) } catch { throw new Error( `Local branch "${branchName}" changed after the workspace was deleted. Review it before deleting it.` diff --git a/src/main/git/worktree-create-preparation.ts b/src/main/git/worktree-create-preparation.ts index 3be0fee1648..b60dc01ec33 100644 --- a/src/main/git/worktree-create-preparation.ts +++ b/src/main/git/worktree-create-preparation.ts @@ -10,6 +10,7 @@ import { WORKTREE_REMOVAL_REGISTRATION_TIMEOUT_MS } from './worktree' import { hasWorktreeBaseCommitRef } from './worktree-base-ref-probe' +import { withRepoRefMaintenancePaused } from './local-repo-ref-maintenance' import { gitExecFileAsync } from './runner' import { runWithGitReadCacheInvalidation } from './status' import { invalidateWslLinkedWorktreeGitRouting } from './wsl-linked-worktree-git-routing' @@ -69,46 +70,48 @@ export async function prepareWorktreeCreateCheckout( options: GitWorktreeExecOptions = {} ): Promise { try { - await runWithGitReadCacheInvalidation(async () => { - const effectiveBase = await resolveWorktreeAddBaseRef(baseBranch, (qualifiedRef) => - hasWorktreeBaseCommitRef(repoPath, qualifiedRef, options) - ) - try { - await gitExecFileAsync( - [ - ...windowsLongPathGitArgs(repoPath), - 'worktree', - 'add', - '--detach', - '--no-checkout', - worktreePath, - effectiveBase - ], - { ...gitExecOptions(repoPath, options), timeout: resolveWorktreeAddTimeoutMs() } + await withRepoRefMaintenancePaused('worktree-prepare', () => + runWithGitReadCacheInvalidation(async () => { + const effectiveBase = await resolveWorktreeAddBaseRef(baseBranch, (qualifiedRef) => + hasWorktreeBaseCommitRef(repoPath, qualifiedRef, options) ) - // The add just wrote the marker; drop any pre-create route before the reset routes Git. - invalidateWslLinkedWorktreeGitRouting(worktreePath) - // Why: reset materializes files without running user post-checkout hooks before submit. - await gitExecFileAsync( - [...windowsLongPathGitArgs(worktreePath), 'reset', '--hard', effectiveBase], - { ...gitExecOptions(worktreePath, options), timeout: resolveWorktreeAddTimeoutMs() } - ) - await gitExecFileAsync( - [ - ...windowsLongPathGitArgs(repoPath), - 'worktree', - 'lock', - '--reason', - lockReason, - worktreePath - ], - { ...gitExecOptions(repoPath, options), timeout: resolveWorktreeAddTimeoutMs() } - ) - } catch (error) { - await performDiscardPreparedWorktree(repoPath, worktreePath, options).catch(() => {}) - throw error - } - }) + try { + await gitExecFileAsync( + [ + ...windowsLongPathGitArgs(repoPath), + 'worktree', + 'add', + '--detach', + '--no-checkout', + worktreePath, + effectiveBase + ], + { ...gitExecOptions(repoPath, options), timeout: resolveWorktreeAddTimeoutMs() } + ) + // The add just wrote the marker; drop any pre-create route before the reset routes Git. + invalidateWslLinkedWorktreeGitRouting(worktreePath) + // Why: reset materializes files without running user post-checkout hooks before submit. + await gitExecFileAsync( + [...windowsLongPathGitArgs(worktreePath), 'reset', '--hard', effectiveBase], + { ...gitExecOptions(worktreePath, options), timeout: resolveWorktreeAddTimeoutMs() } + ) + await gitExecFileAsync( + [ + ...windowsLongPathGitArgs(repoPath), + 'worktree', + 'lock', + '--reason', + lockReason, + worktreePath + ], + { ...gitExecOptions(repoPath, options), timeout: resolveWorktreeAddTimeoutMs() } + ) + } catch (error) { + await performDiscardPreparedWorktree(repoPath, worktreePath, options).catch(() => {}) + throw error + } + }) + ) } finally { notifyPreparedWorktreeMutation(repoPath) } diff --git a/src/main/git/worktree-removal.ts b/src/main/git/worktree-removal.ts index c6743a4a7e2..afe1bf2a9c1 100644 --- a/src/main/git/worktree-removal.ts +++ b/src/main/git/worktree-removal.ts @@ -22,6 +22,7 @@ import { } from './worktree-operation-options' import { areWorktreePathsEqual } from './worktree-path-comparison' import { assertWorktreeCleanForRemoval } from './worktree-removal-preflight' +import { withRepoRefMaintenancePaused } from './local-repo-ref-maintenance' import { bumpWorktreeScanGeneration, listWorktrees } from './worktree-scan-cache' import { invalidateSparseCheckoutState } from './worktree-sparse-checkout-cache' @@ -36,8 +37,13 @@ export async function removeWorktree( options: RemoveWorktreeOptions = {} ): Promise { try { - return await runWithGitReadCacheInvalidation(() => - performRemoveWorktree(repoPath, worktreePath, force, options) + // Removal deletes branches, and a ref deletion needs the packed-refs lock a + // running idle pack holds while it rewrites. Waits that window out; the + // prune phase that follows it is concurrency-safe and is left to finish. + return await withRepoRefMaintenancePaused('worktree-remove', () => + runWithGitReadCacheInvalidation(() => + performRemoveWorktree(repoPath, worktreePath, force, options) + ) ) } finally { invalidateWslLinkedWorktreeGitRouting(worktreePath) diff --git a/src/main/ipc/repos-create.test.ts b/src/main/ipc/repos-create.test.ts index 86e2dc5ee10..44a16f8e10d 100644 --- a/src/main/ipc/repos-create.test.ts +++ b/src/main/ipc/repos-create.test.ts @@ -61,8 +61,11 @@ vi.mock('fs/promises', () => ({ rm: rmMock })) +// `availableParallelism` is read at module load by the git admission scheduler, +// which this module graph reaches; a partial `os` mock breaks that import. vi.mock('os', () => ({ - homedir: homedirMock + homedir: homedirMock, + availableParallelism: () => 8 })) vi.mock('../git/runner', () => ({ diff --git a/src/main/ipc/worktrees.ts b/src/main/ipc/worktrees.ts index ff58a561ee3..3fd2fb41908 100644 --- a/src/main/ipc/worktrees.ts +++ b/src/main/ipc/worktrees.ts @@ -17,7 +17,10 @@ import { registerSparseCheckoutCacheInvalidation } from './worktrees/listing/reg import { registerWorktreeMetadataHandlers } from './worktrees/metadata/register-worktree-metadata-handlers' import { registerWorktreeForgetHandlers } from './worktrees/removal/register-worktree-forget-handlers' import { registerWorktreeRemovalHandlers } from './worktrees/removal/register-worktree-removal-handlers' -import type { WorktreeIpcContext } from './worktrees/worktree-ipc-context' +import { + createWorktreeRemovalRegistry, + type WorktreeIpcContext +} from './worktrees/worktree-ipc-context' registerDetectedWorktreeScanInvalidation() @@ -66,7 +69,7 @@ export function registerWorktreeHandlers( runtime, ...(options ? { options } : {}), detectedWorktreeCancellations: createSenderScopedRequestCancellations(), - worktreeRemovalsInFlight: new Map() + worktreeRemovalsInFlight: createWorktreeRemovalRegistry() } // Remove all stale registrations before installing any replacement handler. diff --git a/src/main/ipc/worktrees/worktree-ipc-context.ts b/src/main/ipc/worktrees/worktree-ipc-context.ts index 5455a71d06e..153e4b723ec 100644 --- a/src/main/ipc/worktrees/worktree-ipc-context.ts +++ b/src/main/ipc/worktrees/worktree-ipc-context.ts @@ -14,3 +14,18 @@ export type WorktreeIpcContext = { detectedWorktreeCancellations: SenderScopedRequestCancellations worktreeRemovalsInFlight: Map } + +// Why: removal and forget both delete refs, and a ref deletion has to take the +// `packed-refs` lock. Idle ref maintenance needs a process-wide view of that +// registry so it never packs while one is running. +let activeWorktreeRemovals: ReadonlyMap | null = null + +export function createWorktreeRemovalRegistry(): Map { + const registry = new Map() + activeWorktreeRemovals = registry + return registry +} + +export function hasWorktreeRemovalsInFlight(): boolean { + return (activeWorktreeRemovals?.size ?? 0) > 0 +} diff --git a/src/main/repo-maintenance-idle-gate.test.ts b/src/main/repo-maintenance-idle-gate.test.ts new file mode 100644 index 00000000000..78728f3a43a --- /dev/null +++ b/src/main/repo-maintenance-idle-gate.test.ts @@ -0,0 +1,134 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const isOnBatteryPowerMock = vi.hoisted(() => vi.fn(() => false)) +const hasPendingPreparationsMock = vi.hoisted(() => vi.fn(() => false)) +const hasRemovalsInFlightMock = vi.hoisted(() => vi.fn(() => false)) +const setProbeMock = vi.hoisted(() => vi.fn()) +const disposeMock = vi.hoisted(() => vi.fn(async () => {})) +const postponeMock = vi.hoisted(() => vi.fn()) +const powerListeners = vi.hoisted(() => new Map void>()) +const appListeners = vi.hoisted(() => new Map void>()) + +vi.mock('electron', () => ({ + app: { + on: (event: string, listener: () => void) => appListeners.set(event, listener), + off: (event: string) => appListeners.delete(event) + }, + powerMonitor: { + isOnBatteryPower: isOnBatteryPowerMock, + on: (event: string, listener: () => void) => powerListeners.set(event, listener), + off: (event: string) => powerListeners.delete(event) + } +})) + +vi.mock('./worktree-create-preparation', () => ({ + hasPendingWorktreeCreatePreparations: hasPendingPreparationsMock +})) + +vi.mock('./ipc/worktrees/worktree-ipc-context', () => ({ + hasWorktreeRemovalsInFlight: hasRemovalsInFlightMock +})) + +vi.mock('./git/local-repo-ref-maintenance', () => ({ + setRepoMaintenanceActivityProbe: setProbeMock, + disposeLocalRepoRefMaintenance: disposeMock, + postponeRepoRefMaintenance: postponeMock +})) + +import { installRepoMaintenanceIdleGate } from './repo-maintenance-idle-gate' + +function installProbe( + overrides: Partial<{ isQuitting: () => boolean; getWorkingAgentCount: () => number }> = {} +): { probe: () => boolean; uninstall: () => Promise } { + const uninstall = installRepoMaintenanceIdleGate({ + isQuitting: () => false, + getWorkingAgentCount: () => 0, + ...overrides + }) + return { probe: setProbeMock.mock.calls.at(-1)?.[0] as () => boolean, uninstall } +} + +beforeEach(() => { + isOnBatteryPowerMock.mockReturnValue(false) + hasPendingPreparationsMock.mockReturnValue(false) + hasRemovalsInFlightMock.mockReturnValue(false) + postponeMock.mockClear() + powerListeners.clear() + appListeners.clear() + setProbeMock.mockClear() + disposeMock.mockClear() +}) + +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('repo maintenance idle gate', () => { + it('reports idle when nothing is happening', () => { + expect(installProbe().probe()).toBe(false) + }) + + it('vetoes while an agent is working', () => { + expect(installProbe({ getWorkingAgentCount: () => 1 }).probe()).toBe(true) + }) + + it('vetoes while a worktree create is prepared or in flight', () => { + hasPendingPreparationsMock.mockReturnValue(true) + + expect(installProbe().probe()).toBe(true) + }) + + it('vetoes while a worktree removal is deleting refs', () => { + // Removal deletes branches, and a ref deletion needs the same packed-refs lock. + hasRemovalsInFlightMock.mockReturnValue(true) + + expect(installProbe().probe()).toBe(true) + }) + + it('vetoes on battery power', () => { + isOnBatteryPowerMock.mockReturnValue(true) + + expect(installProbe().probe()).toBe(true) + }) + + it('vetoes during shutdown', () => { + expect(installProbe({ isQuitting: () => true }).probe()).toBe(true) + }) + + it('treats an unavailable power API as not-on-battery', () => { + isOnBatteryPowerMock.mockImplementation(() => { + throw new Error('unsupported') + }) + + expect(installProbe().probe()).toBe(false) + }) + + it('pushes the next attempt out when the machine drops onto battery', () => { + // Do-not-start, never stop-what-is-running: killing a pack to honour a + // battery change would strand a ref lock to save a little unlinking. + installProbe() + + powerListeners.get('on-battery')?.() + + expect(postponeMock).toHaveBeenCalledTimes(1) + }) + + it('pushes the next attempt out when the user comes back to the window', () => { + // A focus transition, not focus itself: a window left focused while the user + // walks away fires no event and blocks nothing. + installProbe() + + appListeners.get('browser-window-focus')?.() + + expect(postponeMock).toHaveBeenCalledTimes(1) + }) + + it('cancels armed timers, unsubscribes both sources, and clears the probe when uninstalled', async () => { + await installProbe().uninstall() + + expect(disposeMock).toHaveBeenCalledTimes(1) + expect(powerListeners.has('on-battery')).toBe(false) + expect(appListeners.has('browser-window-focus')).toBe(false) + expect(setProbeMock).toHaveBeenLastCalledWith(null) + }) +}) diff --git a/src/main/repo-maintenance-idle-gate.ts b/src/main/repo-maintenance-idle-gate.ts new file mode 100644 index 00000000000..78bb25fe68c --- /dev/null +++ b/src/main/repo-maintenance-idle-gate.ts @@ -0,0 +1,67 @@ +import { app, powerMonitor } from 'electron' +import { + disposeLocalRepoRefMaintenance, + postponeRepoRefMaintenance, + setRepoMaintenanceActivityProbe +} from './git/local-repo-ref-maintenance' +import { hasWorktreeRemovalsInFlight } from './ipc/worktrees/worktree-ipc-context' +import { hasPendingWorktreeCreatePreparations } from './worktree-create-preparation' + +/** + * The app-wide "not now" answer for idle repo maintenance. + * + * `pack-refs` holds a general git admission slot for its whole run, which on a + * large backlog is minutes, and takes the `packed-refs` lock while it writes. + * Any ref deletion needs that same lock and gives up after + * `core.packedRefsTimeout` (1s), so worktree removal in particular has to veto + * this -- as does a create in flight, an agent mid-run, and shutdown. Battery is + * a veto too: this is work the user did not ask for, and a plugged-in quiet + * window always comes along later. + */ +export type RepoMaintenanceIdleInputs = { + isQuitting: () => boolean + getWorkingAgentCount: () => number +} + +export function installRepoMaintenanceIdleGate( + inputs: RepoMaintenanceIdleInputs +): () => Promise { + setRepoMaintenanceActivityProbe( + () => + inputs.isQuitting() || + inputs.getWorkingAgentCount() > 0 || + hasPendingWorktreeCreatePreparations() || + hasWorktreeRemovalsInFlight() || + isOnBatteryPower() + ) + // Do-not-start, never stop-what-is-running. Killing a pack to honour a battery + // or focus change would strand a ref lock roughly one time in five to save at + // most a couple of minutes of background unlinking; pushing the next attempt + // out costs nothing and risks nothing. + const onBattery = (): void => { + postponeRepoRefMaintenance() + } + const onFocus = (): void => { + postponeRepoRefMaintenance() + } + powerMonitor.on('on-battery', onBattery) + app.on('browser-window-focus', onFocus) + return () => { + app.off('browser-window-focus', onFocus) + powerMonitor.off('on-battery', onBattery) + // Order matters: clearing the probe alone would leave armed timers running + // against a gate that can no longer see agents, creates, or shutdown. + const stopped = disposeLocalRepoRefMaintenance() + setRepoMaintenanceActivityProbe(null) + return stopped + } +} + +function isOnBatteryPower(): boolean { + try { + return powerMonitor.isOnBatteryPower() + } catch { + // Absence of the API is not evidence of battery; desktops answer false anyway. + return false + } +} diff --git a/src/main/runtime/fetch-remote-cache.test.ts b/src/main/runtime/fetch-remote-cache.test.ts index fc2d761c059..c97966c344e 100644 --- a/src/main/runtime/fetch-remote-cache.test.ts +++ b/src/main/runtime/fetch-remote-cache.test.ts @@ -133,9 +133,11 @@ describe('OrcaRuntimeService.fetchRemoteWithCache', () => { const first = runtime.fetchRemoteWithCache('/repo/c', 'origin') const second = runtime.fetchRemoteWithCache('/repo/c', 'origin') - // Allow both callers to register before we resolve. - await Promise.resolve() - await Promise.resolve() + // Allow both callers to register before we resolve. Each canonicalizes the + // repo key first, so the dispatch lands several microtasks in. + for (let tick = 0; tick < 8; tick += 1) { + await Promise.resolve() + } expect(fetchCallCount()).toBe(1) resolveFetch() diff --git a/src/main/runtime/runtime-remote-fetch-controller.ts b/src/main/runtime/runtime-remote-fetch-controller.ts index c48a8437a3d..b3b9df5dcba 100644 --- a/src/main/runtime/runtime-remote-fetch-controller.ts +++ b/src/main/runtime/runtime-remote-fetch-controller.ts @@ -1,4 +1,9 @@ import { GIT_FETCH_SKIP_AUTO_MAINTENANCE_CONFIG_ARGS } from '../../shared/git-fetch-auto-maintenance' +import { getCanonicalRepoKey } from '../git/canonical-repo-key' +import { + armLocalRepoRefMaintenance, + setRepoRefMaintenanceBusyProbe +} from '../git/local-repo-ref-maintenance' import { gitExecFileAsync } from '../git/runner' import { setBoundedMapEntry } from './runtime-async-boundaries' @@ -33,6 +38,11 @@ export class RuntimeRemoteFetchController { return this.fetchLastCompletedAt } + /** `${runtimeKey}::${gitCommonDir}` -- one repo on one execution host. */ + async getCanonicalRepoKey(repoPath: string, gitOptions: GitOptions = {}): Promise { + return getCanonicalRepoKey(repoPath, gitOptions) + } + async getCanonicalFetchKey( repoPath: string, remote: string, @@ -45,23 +55,41 @@ export class RuntimeRemoteFetchController { setBoundedMapEntry(this.canonicalFetchKeyCache, cacheKey, cached, REMOTE_FETCH_CACHE_MAX) return cached } - let resolved = cacheKey - try { - const { stdout } = await gitExecFileAsync( - ['rev-parse', '--path-format=absolute', '--git-common-dir'], - { cwd: repoPath, ...gitOptions } - ) - const commonDir = stdout.trim() - if (commonDir) { - resolved = `${runtimeKey}::${commonDir}::${remote}` - } - } catch { - // The caller path remains a safe serialization key when canonicalization fails. - } + const resolved = `${await this.getCanonicalRepoKey(repoPath, gitOptions)}::${remote}` setBoundedMapEntry(this.canonicalFetchKeyCache, cacheKey, resolved, REMOTE_FETCH_CACHE_MAX) return resolved } + /** + * Orca strips git's auto-maintenance off these fetches, so every one of them + * adds to a loose-ref backlog nothing else will ever pack. Arm the idle sweep + * that pays it back; each fetch pushes the attempt a further quiet period out. + */ + private armRefMaintenance(repoPath: string, gitOptions: GitOptions): void { + void this.getCanonicalRepoKey(repoPath, gitOptions) + .then((key) => { + setRepoRefMaintenanceBusyProbe(key, () => this.hasInflightFetchForRepo(key)) + armLocalRepoRefMaintenance({ + key, + repoPath, + ...(gitOptions.wslDistro ? { wslDistro: gitOptions.wslDistro } : {}) + }) + }) + .catch(() => { + // Maintenance is best effort; a repo we cannot name is a repo we skip. + }) + } + + private hasInflightFetchForRepo(repoKey: string): boolean { + const prefix = `${repoKey}::` + for (const key of this.fetchInflight.keys()) { + if (key.startsWith(prefix)) { + return true + } + } + return false + } + private enqueueRemoteFetch( remoteKey: string, runFetch: () => Promise @@ -123,6 +151,7 @@ export class RuntimeRemoteFetchController { }) ).finally(() => { this.fetchInflight.delete(key) + this.armRefMaintenance(repoPath, gitOptions) }) this.fetchInflight.set(key, promise) return promise @@ -178,6 +207,7 @@ export class RuntimeRemoteFetchController { }) }).finally(() => { this.fetchInflight.delete(key) + this.armRefMaintenance(repoPath, gitOptions) }) this.fetchInflight.set(key, promise) return promise diff --git a/src/main/runtime/runtime-remote-fetch-ref-maintenance.test.ts b/src/main/runtime/runtime-remote-fetch-ref-maintenance.test.ts new file mode 100644 index 00000000000..f577da24854 --- /dev/null +++ b/src/main/runtime/runtime-remote-fetch-ref-maintenance.test.ts @@ -0,0 +1,145 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// Why: Orca's fetches are what create the loose-ref backlog (they suppress +// git's auto-maintenance), so the fetch controller is where the idle sweep has +// to be armed. These tests pin that wiring and the per-repo busy signal it +// hands the sweep. + +const gitExecFileAsyncMock = vi.hoisted(() => vi.fn()) +const armMock = vi.hoisted(() => vi.fn()) +const busyProbeMock = vi.hoisted(() => vi.fn()) + +vi.mock('../git/runner', async (importOriginal) => ({ + ...((await importOriginal()) as Record), + gitExecFileAsync: gitExecFileAsyncMock +})) + +vi.mock('../git/local-repo-ref-maintenance', async (importOriginal) => ({ + ...((await importOriginal()) as Record), + armLocalRepoRefMaintenance: armMock, + setRepoRefMaintenanceBusyProbe: busyProbeMock +})) + +import { _resetCanonicalRepoKeyCacheForTests } from '../git/canonical-repo-key' +import { RuntimeRemoteFetchController } from './runtime-remote-fetch-controller' + +function armedTargets(): { key: string }[] { + return armMock.mock.calls.map(([args]) => args as { key: string }) +} + +/** The per-repo "a fetch is in flight" answer the controller registers for a key. */ +function busyProbeFor(key: string): (() => boolean) | undefined { + return busyProbeMock.mock.calls.findLast(([registered]) => registered === key)?.[1] as + | (() => boolean) + | undefined +} + +beforeEach(() => { + _resetCanonicalRepoKeyCacheForTests() + gitExecFileAsyncMock.mockReset() + armMock.mockReset() + busyProbeMock.mockReset() + gitExecFileAsyncMock.mockImplementation(async (argv: string[]) => + argv[0] === 'rev-parse' ? { stdout: '/repo/.git\n', stderr: '' } : { stdout: '', stderr: '' } + ) +}) + +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('fetch-armed ref maintenance', () => { + it('arms the sweep for the repo after a remote fetch, keyed by common dir', async () => { + const controller = new RuntimeRemoteFetchController() + + await controller.getOrStartRemoteFetch('/repo/worktrees/a', 'origin') + + expect(armedTargets().map((target) => target.key)).toEqual(['local::/repo/.git']) + }) + + it('gives every worktree of one repo the same maintenance key', async () => { + const controller = new RuntimeRemoteFetchController() + + await controller.getOrStartRemoteFetch('/repo/worktrees/a', 'origin') + await controller.getOrStartRemoteTrackingBaseRefresh('/repo/worktrees/b', { + remote: 'origin', + branch: 'main', + ref: 'refs/remotes/origin/main', + base: 'origin/main' + }) + + const keys = new Set(armedTargets().map((target) => target.key)) + expect(keys).toEqual(new Set(['local::/repo/.git'])) + }) + + it('scopes the key to the WSL distro that executes the repo', async () => { + const controller = new RuntimeRemoteFetchController() + + await controller.getOrStartRemoteFetch('//wsl$/Ubuntu/repo', 'origin', { + wslDistro: 'Ubuntu' + }) + + expect(armedTargets()[0]?.key).toBe('wsl:Ubuntu::/repo/.git') + }) + + it('does not collapse every repo onto one key on Git older than 2.31', async () => { + // Old Git echoes the unrecognized `--path-format` flag, exits 0, and prints a + // relative `.git`; taking that raw would name every repository identically. + gitExecFileAsyncMock.mockImplementation(async (argv: string[]) => + argv[0] === 'rev-parse' + ? { stdout: '--path-format=absolute\n.git\n', stderr: '' } + : { stdout: '', stderr: '' } + ) + const controller = new RuntimeRemoteFetchController() + + await controller.getOrStartRemoteFetch('/repo/one', 'origin') + await controller.getOrStartRemoteFetch('/repo/two', 'origin') + + expect(armedTargets().map((entry) => entry.key)).toEqual([ + 'local::/repo/one/.git', + 'local::/repo/two/.git' + ]) + }) + + it('reports the repo as busy while another fetch on it is in flight', async () => { + const controller = new RuntimeRemoteFetchController() + await controller.getOrStartRemoteFetch('/repo', 'first') + const isBusy = busyProbeFor('local::/repo/.git') + expect(isBusy?.()).toBe(false) + + let releaseFetch: (() => void) | undefined + gitExecFileAsyncMock.mockImplementation(async (argv: string[]) => { + if (argv[0] === 'rev-parse') { + return { stdout: '/repo/.git\n', stderr: '' } + } + await new Promise((resolve) => { + releaseFetch = resolve + }) + return { stdout: '', stderr: '' } + }) + const second = controller.getOrStartRemoteFetch('/repo', 'second') + await vi.waitFor(() => expect(releaseFetch).toBeDefined()) + expect(isBusy?.()).toBe(true) + + releaseFetch?.() + await second + expect(isBusy?.()).toBe(false) + }) + + it('arms even when the fetch fails, because a partial fetch still writes refs', async () => { + const controller = new RuntimeRemoteFetchController() + gitExecFileAsyncMock.mockImplementation(async (argv: string[]) => { + if (argv[0] === 'rev-parse') { + return { stdout: '/repo/.git\n', stderr: '' } + } + throw new Error('network is unreachable') + }) + vi.spyOn(console, 'warn').mockImplementation(() => {}) + + await expect(controller.getOrStartRemoteFetch('/repo', 'origin')).resolves.toEqual({ + ok: false, + errorKind: 'git_error' + }) + expect(armedTargets()).toHaveLength(1) + }) +}) diff --git a/src/main/startup/main-process-observers.ts b/src/main/startup/main-process-observers.ts index c3d83450b7c..ba37b0a312f 100644 --- a/src/main/startup/main-process-observers.ts +++ b/src/main/startup/main-process-observers.ts @@ -18,6 +18,7 @@ import { AgentSessionTransitionRecorder } from '../stats/agent-session-transitio import { ClaudeUsageStore } from '../claude-usage/store' import { CodexUsageStore } from '../codex-usage/store' import { OpenCodeUsageStore } from '../opencode-usage/store' +import { installRepoMaintenanceIdleGate } from '../repo-maintenance-idle-gate' import { mainProcessState as state } from './main-process-state' export function initializeMainProcessObservers(): void { @@ -35,6 +36,10 @@ export function initializeMainProcessObservers(): void { ) // Why: start from empty — disk-hydrated status rows are UI continuity only; only this runtime's hook events keep the computer awake. state.agentAwakeService.setStatuses([]) + state.uninstallRepoMaintenanceIdleGate = installRepoMaintenanceIdleGate({ + isQuitting: () => state.isQuitting, + getWorkingAgentCount: () => state.agentAwakeService?.getWorkingAgentCount() ?? 0 + }) const collectChangedProviderSessionWorktrees = createHookProviderSessionInvalidator() const publishProviderSessionChanges = (identities: AgentHookProviderSessionIdentity[]): void => { const ownedIdentities = identities.map((identity) => ({ diff --git a/src/main/startup/main-process-quit.ts b/src/main/startup/main-process-quit.ts index ec893456c5a..61203bc3164 100644 --- a/src/main/startup/main-process-quit.ts +++ b/src/main/startup/main-process-quit.ts @@ -14,6 +14,7 @@ import { clearRuntimeMetadataIfOwned } from '../runtime/runtime-metadata' import { shutdownPairedRuntimeBrowserClientHosts } from '../browser/paired-runtime-browser-client-host-runtime' import { browserManager } from '../browser/browser-manager' import { stopCodexStateDbBackfillRecoveries } from '../codex/codex-state-db-backfill-recovery' +import { awaitPackedRefsLockRelease } from '../git/local-repo-ref-maintenance' import { settleTeardownWithinDeadline, settleWithinMs } from '../quit-teardown-deadline' import { quitTeardownStartGate } from '../quit-teardown-start-gate' import { setUnreadDockBadgeCount } from '../dock/unread-badge' @@ -33,6 +34,8 @@ let daemonDisconnectDone = false let watcherShutdownPromise: Promise | null = null // Why 2s: a config delete is best-effort, not durable state. const GROK_HOOK_CLEANUP_DEADLINE_MS = 2_000 +// Why 2s: long enough for a `pack-refs` child to take SIGTERM and unlink its lock. +const REF_MAINTENANCE_QUIT_DEADLINE_MS = 2_000 function shutdownWatchersOnce(): Promise { if (state.watcherShutdownDone) { @@ -73,6 +76,10 @@ function installBeforeQuitHandler(): void { state.unsubscribeAgentAwakeStatusChanges = null state.agentAwakeService?.dispose() state.agentAwakeService = null + // Why wait but not uninstall: a renderer beforeunload can still veto this + // quit, and tearing the sweep down here would kill it for the rest of the + // session. `isQuitting` already vetoes new attempts; will-quit does the teardown. + state.repoMaintenanceShutdown = awaitPackedRefsLockRelease() // Why: defer PTY cleanup to will-quit so the renderer captures scrollback before PTY-exit events unmount TerminalPane (dropping its capture callbacks). state.rateLimits?.stop() }) @@ -123,6 +130,16 @@ function installWillQuitHandler(): void { const structuredAgentSessionShutdown = stopStructuredAgentSessionRuntime() state.pluginService = null setUnreadDockBadgeCount(0) + // Why wait rather than kill: the child finishes fine orphaned, and signalling + // it mid-prune strands a ref lock Git never clears. The wait is only for the + // short rewrite window, and is bounded so a quit can never hang on it. + const refMaintenanceShutdown = settleWithinMs( + Promise.all([state.repoMaintenanceShutdown, state.uninstallRepoMaintenanceIdleGate?.()]).then( + () => {} + ), + REF_MAINTENANCE_QUIT_DEADLINE_MS + ).then(() => {}) + state.uninstallRepoMaintenanceIdleGate = null agentHookServer.stop() // Why Windows only: POSIX hooks short-circuit on ORCA_PANE_KEY, while Windows must register a // bare script path that cannot express the guard and would otherwise keep spawning after quit. @@ -219,6 +236,7 @@ function installWillQuitHandler(): void { { name: 'plugin-hosts', promise: pluginHostShutdown }, { name: 'skill-uploads', promise: skillUploadShutdown }, { name: 'grok-hooks', promise: grokHookCleanup }, + { name: 'ref-maintenance', promise: refMaintenanceShutdown }, { name: 'codex-backfill-recovery', promise: codexBackfillRecoveryShutdown }, { name: 'structured-agent-session', promise: structuredAgentSessionShutdown }, { name: 'usage-cache', promise: usageCacheFlush }, diff --git a/src/main/startup/main-process-state.ts b/src/main/startup/main-process-state.ts index 05c45d5b567..d48219b461e 100644 --- a/src/main/startup/main-process-state.ts +++ b/src/main/startup/main-process-state.ts @@ -71,6 +71,8 @@ export const mainProcessState = { headlessBrowserDisplayAvailable: false, starNag: null as StarNagService | null, agentAwakeService: null as AgentAwakeService | null, + uninstallRepoMaintenanceIdleGate: null as (() => Promise) | null, + repoMaintenanceShutdown: Promise.resolve() as Promise, crashReports: null as CrashReportStore | null, unsubscribeAgentAwakeStatusChanges: null as (() => void) | null, publishProviderSessionChanges: null as diff --git a/src/main/worktree-create-preparation.ts b/src/main/worktree-create-preparation.ts index 22bcf5d1e2c..b4d86923490 100644 --- a/src/main/worktree-create-preparation.ts +++ b/src/main/worktree-create-preparation.ts @@ -61,6 +61,11 @@ type ConsumePreparedWorktreeArgs = { const preparations = new Map() +/** A prepared checkout is a create that is either in flight or imminent. */ +export function hasPendingWorktreeCreatePreparations(): boolean { + return preparations.size > 0 || staleCleanupInFlight.size > 0 +} + function pathOps(path: string): Pick { return isWindowsAbsolutePathLike(path) ? win32 : posix } diff --git a/src/shared/git-binary-compatibility.test.ts b/src/shared/git-binary-compatibility.test.ts index 3a8f562dff1..2861c447788 100644 --- a/src/shared/git-binary-compatibility.test.ts +++ b/src/shared/git-binary-compatibility.test.ts @@ -277,6 +277,39 @@ describeBinaryCompatibility('real Git binary compatibility', () => { ).rejects.toMatchObject({ code: 1 }) }) + it('packs loose refs and reads the maintenance opt-out at the baseline', async () => { + // Why: idle ref maintenance runs `pack-refs --all --prune` on every supported + // Git rather than the 2.45+ `--auto` form, and reads `maintenance.auto` to + // honour a user who disabled Git's own auto-maintenance. Both must work at 2.25. + const head = (await runGit(['rev-parse', 'HEAD'])).stdout.trim() + const packedRef = 'refs/remotes/origin/compat-pack-refs' + await runGit(['update-ref', packedRef, head]) + await expect(readFile(join(repoPath, '.git', packedRef), 'utf-8')).resolves.toContain(head) + + await expect(runGit(['pack-refs', '--all', '--prune'])).resolves.toBeDefined() + + // The loose file is gone and the ref still resolves through packed-refs. + await expect(readFile(join(repoPath, '.git', packedRef), 'utf-8')).rejects.toMatchObject({ + code: 'ENOENT' + }) + await expect(runGit(['rev-parse', '--verify', packedRef])).resolves.toMatchObject({ + stdout: `${head}\n` + }) + await expect(readFile(join(repoPath, '.git', 'packed-refs'), 'utf-8')).resolves.toContain( + packedRef + ) + + // `--get` exits 1 on an unset key; that absence must read as consent, not opt-out. + await expect(runGit(['config', '--bool', '--get', 'maintenance.auto'])).rejects.toMatchObject({ + code: 1 + }) + await runGit(['config', 'maintenance.auto', 'false']) + await expect(runGit(['config', '--bool', '--get', 'maintenance.auto'])).resolves.toMatchObject({ + stdout: 'false\n' + }) + await runGit(['config', '--unset', 'maintenance.auto']) + }) + it('fetches hosted review heads into dedicated refs', async () => { const head = (await runGit(['rev-parse', 'HEAD'])).stdout.trim() await runGit(['update-ref', 'refs/pull/42/head', head]) diff --git a/src/shared/loose-ref-count.test.ts b/src/shared/loose-ref-count.test.ts new file mode 100644 index 00000000000..c69f655121a --- /dev/null +++ b/src/shared/loose-ref-count.test.ts @@ -0,0 +1,119 @@ +import { mkdir, mkdtemp, rm, symlink, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' + +// Wraps the real `readdir` so the walk's concurrency is observable without +// changing what it reads. +const readdirCalls = vi.hoisted(() => ({ outstanding: 0, peak: 0, count: 0 })) + +vi.mock('node:fs/promises', async (importOriginal) => { + const actual = await importOriginal>() + const realReaddir = actual.readdir as (...args: unknown[]) => Promise + return { + ...actual, + readdir: async (...args: unknown[]) => { + readdirCalls.outstanding += 1 + readdirCalls.count += 1 + readdirCalls.peak = Math.max(readdirCalls.peak, readdirCalls.outstanding) + try { + return await realReaddir(...args) + } finally { + readdirCalls.outstanding -= 1 + } + } + } +}) + +import { countLooseRefs } from './loose-ref-count' + +const roots: string[] = [] + +async function makeRefsTree(counts: Record): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-loose-refs-')) + roots.push(root) + const refs = join(root, 'refs') + for (const [namespace, count] of Object.entries(counts)) { + const directory = join(refs, namespace) + await mkdir(directory, { recursive: true }) + for (let index = 0; index < count; index += 1) { + await writeFile(join(directory, `ref-${index}`), 'a'.repeat(40)) + } + } + await mkdir(refs, { recursive: true }) + return refs +} + +afterEach(async () => { + vi.restoreAllMocks() + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('countLooseRefs', () => { + it('counts files across nested namespaces', async () => { + const refs = await makeRefsTree({ heads: 3, 'remotes/origin': 4, 'remotes/fork/deep': 2 }) + + await expect(countLooseRefs(refs, 100)).resolves.toEqual({ count: 9, saturated: false }) + }) + + it('stops at the budget instead of walking the whole backlog', async () => { + const refs = await makeRefsTree({ 'remotes/origin': 500 }) + + const result = await countLooseRefs(refs, 10) + + expect(result).toEqual({ count: 10, saturated: true }) + }) + + it('reports zero for a repository with no refs directory', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-loose-refs-missing-')) + roots.push(root) + + await expect(countLooseRefs(join(root, 'refs'), 100)).resolves.toEqual({ + count: 0, + saturated: false + }) + }) + + it('never has more than one directory read outstanding', async () => { + // libuv's filesystem thread pool has four slots shared with the whole main + // process. A probe that fanned out would stall unrelated fs work, so this + // pins the walk as strictly sequential rather than merely bounded. + const refs = await makeRefsTree({ + 'remotes/a': 3, + 'remotes/b': 3, + 'remotes/c': 3, + 'remotes/d': 3, + 'remotes/e/deep': 3 + }) + readdirCalls.peak = 0 + readdirCalls.count = 0 + + await countLooseRefs(refs, 1000) + + expect(readdirCalls.count).toBeGreaterThan(1) + expect(readdirCalls.peak).toBe(1) + }) + + it('reads each directory once rather than streaming it in batches', async () => { + // One thread-pool round trip per directory is what makes the probe ~8x + // cheaper than the streaming form on a real degraded repository. + const refs = await makeRefsTree({ 'remotes/origin': 400 }) + readdirCalls.count = 0 + + await countLooseRefs(refs, 1000) + + // refs/ plus refs/remotes plus refs/remotes/origin. + expect(readdirCalls.count).toBe(3) + }) + + it('does not follow directory symlinks into a loop', async () => { + const refs = await makeRefsTree({ heads: 2 }) + await symlink(refs, join(refs, 'loop'), 'dir') + + const result = await countLooseRefs(refs, 100) + + expect(result.saturated).toBe(false) + // The symlink is one dirent, never a second traversal of the tree. + expect(result.count).toBe(3) + }) +}) diff --git a/src/shared/loose-ref-count.ts b/src/shared/loose-ref-count.ts new file mode 100644 index 00000000000..8f31d8b03a8 --- /dev/null +++ b/src/shared/loose-ref-count.ts @@ -0,0 +1,80 @@ +import { readdir } from 'node:fs/promises' +import { join } from 'node:path' + +export type LooseRefCount = { + /** Loose ref files seen, never above `budget`. */ + count: number + /** The walk stopped early, so `count` is a floor rather than the total. */ + saturated: boolean +} + +// Why: a ref tree is shallow and wide; this bounds both the directories visited +// and the queue holding those still to visit, so neither a symlink loop nor a +// pathological repo turns a gate probe into an unbounded walk. +const DIRECTORY_VISIT_CEILING = 4096 + +/** + * Count loose refs under a repository's `refs/` directory, stopping at `budget`. + * + * Deliberately budgeted: callers use this as an admission gate, so the cost has + * to be bounded by the threshold being tested and not by the size of the + * backlog it is testing for. + * + * One `readdir` per directory, dirents only -- no `stat` per entry, and no + * `opendir` streaming. Measured against a real 36,600-loose-ref repository, the + * batched form is ~8x faster (23ms vs 177ms median to reach a 1000 threshold) + * and holds the event loop for less than half as long, because streaming issues + * a thread-pool round trip every 32 entries where this issues one per + * directory. The cost is holding one directory's dirents at a time, which is + * bounded by the widest ref namespace rather than by the size of the tree. + * + * Strictly sequential on purpose: it awaits one directory before opening the + * next, so it can never occupy more than one of libuv's four thread-pool slots + * and cannot stall unrelated main-process filesystem work. + * + * `signal` stops the walk between directories. A single hung `readdir` is not + * interruptible, but it holds no Git lock, so it delays only maintenance. + */ +export async function countLooseRefs( + refsDirectory: string, + budget: number, + signal?: AbortSignal +): Promise { + const pending = [refsDirectory] + let count = 0 + let visited = 0 + while (pending.length > 0) { + const directory = pending.pop() + if (directory === undefined) { + break + } + visited += 1 + // A cancelled walk reports what it saw as a floor rather than throwing; callers + // already have to treat a saturated result as "not known to be clean". + if ( + signal?.aborted === true || + visited > DIRECTORY_VISIT_CEILING || + pending.length > DIRECTORY_VISIT_CEILING + ) { + return { count, saturated: true } + } + let entries: { name: string; isDirectory: () => boolean }[] + try { + entries = await readdir(directory, { withFileTypes: true }) + } catch { + // A missing or unreadable namespace contributes nothing to the count. + continue + } + for (const entry of entries) { + if (entry.isDirectory()) { + pending.push(join(directory, entry.name)) + continue + } + count += 1 + if (count >= budget) { + return { count, saturated: true } + } + } + } + return { count, saturated: false } +} diff --git a/src/shared/packed-refs-lock-gate.ts b/src/shared/packed-refs-lock-gate.ts new file mode 100644 index 00000000000..f5cff6b8c4a --- /dev/null +++ b/src/shared/packed-refs-lock-gate.ts @@ -0,0 +1,43 @@ +/** + * Tracks the one window in a pack that actually excludes anybody: the + * `packed-refs` rewrite. Callers about to touch refs wait this out rather than + * killing the child, because a signal delivered into the prune phase strands a + * `refs/**\/*.lock` roughly one time in five and Git never clears those. + */ +export class PackedRefsLockGate { + private held = false + private waiters: (() => void)[] = [] + + setHeld(held: boolean): void { + this.held = held + if (held) { + return + } + const waiting = this.waiters + this.waiters = [] + for (const resolve of waiting) { + resolve() + } + } + + /** Resolves on release, or on `timeoutMs` -- past which Git's own retry is the better bet. */ + whenReleased(timeoutMs: number): Promise { + if (!this.held) { + return Promise.resolve() + } + return new Promise((resolve) => { + let settled = false + const finish = (): void => { + if (settled) { + return + } + settled = true + clearTimeout(timer) + resolve() + } + const timer = setTimeout(finish, timeoutMs) + timer.unref?.() + this.waiters.push(finish) + }) + } +} diff --git a/src/shared/repo-ref-maintenance-policy.ts b/src/shared/repo-ref-maintenance-policy.ts new file mode 100644 index 00000000000..7dc8edd1de3 --- /dev/null +++ b/src/shared/repo-ref-maintenance-policy.ts @@ -0,0 +1,166 @@ +/** + * Idle-time loose-ref packing for repositories Orca itself degrades. + * + * Orca strips git's auto-maintenance off its own frequent fetches + * (`GIT_FETCH_SKIP_AUTO_MAINTENANCE_CONFIG_ARGS`) and never compensated, so an + * Orca-driven checkout accumulates loose refs forever and every ref + * enumeration -- `show-ref`, `for-each-ref`, worktree create -- pays for them. + * This is the compensation: after a repo goes quiet, probe it, and pack only + * when the backlog is real. + * + * The engine is host-agnostic on purpose. The execution host owns everything + * that touches execution, so each host supplies its own target (which git to + * run, which filesystem to walk) and all state here is keyed per host. + */ + +/** + * Below this, ref enumeration is already fast and `pack-refs` would cost more + * than it saves. + * + * Git's own files-backend auto heuristic (2.47+) packs at + * `max(16, log2(packed_refs_bytes / 100) * 5)` loose refs -- about 76 for the + * 4.1 MB `packed-refs` that motivated this work. A flat 1000 is roughly an + * order of magnitude more conservative on purpose: this runs unasked against a + * real checkout, and being late is cheap where being wrong is not. + */ +export const LOOSE_REF_PACK_THRESHOLD = 1000 + +/** No fetch, create, or other tracked write on the repo for this long. */ +export const REF_MAINTENANCE_QUIET_PERIOD_MS = 10 * 60_000 + +/** Packing empties the backlog; there is nothing to do again for a long while. */ +export const REF_MAINTENANCE_PACKED_COOLDOWN_MS = 12 * 60 * 60_000 + +/** A healthy or unresolvable repo should not be re-probed on every quiet window. */ +export const REF_MAINTENANCE_CLEAN_COOLDOWN_MS = 6 * 60 * 60_000 + +/** A failing repo (permissions, stale lock) must not be retried in a loop. */ +export const REF_MAINTENANCE_FAILURE_COOLDOWN_MS = 6 * 60 * 60_000 + +/** + * A repository whose `packed-refs.lock` is a strand from our own dead process + * becomes reclaimable at `PACK_REFS_TIMEOUT_MS`, so retry near that rather than + * serving the full failure cooldown -- otherwise a Windows force-kill leaves + * every ref deletion in that repo failing for six hours instead of thirty + * minutes. + */ +export const REF_MAINTENANCE_LOCKED_COOLDOWN_MS = 30 * 60_000 + +/** + * `pack-refs` holds `packed-refs.lock` only while it rewrites the file -- + * measured at 0.03-1.37s of a 23-32s run, the other ~95% being the prune phase + * unlinking loose refs. A caller about to touch refs waits out that window + * instead of killing the pack. + */ +export const PACKED_REFS_LOCK_POLL_MS = 50 + +/** + * Ceiling on that wait. Past this we stop blocking the user and let Git's own + * retry (`core.filesRefLockTimeout`, `core.packedRefsTimeout`) handle it, which + * is what happens today without any of this. + */ +export const PACKED_REFS_LOCK_WAIT_MS = 5_000 + +/** + * `pack-refs --prune` unlinks one file per loose ref. Paying off a 36k-ref + * backlog measured at ~83s on APFS, so the deadline has to clear a cold repo on + * a slow disk by a wide margin. A kill mid-run is safe -- git renames + * `packed-refs` into place atomically and the surviving loose refs stay + * authoritative -- but it wastes the work. + */ +export const PACK_REFS_TIMEOUT_MS = 15 * 60_000 + +/** + * Ancient, safe on the Git 2.25 baseline, and does exactly one thing. + * + * Not `pack-refs --auto`: that arrived in 2.45 and unconditionally rewrote + * `packed-refs` on the files backend until 2.47, so it is both unavailable at + * our baseline and wrong on two shipped releases. Not `git maintenance run` + * either -- newer, and it pulls in commit-graph and repack work we did not ask + * for. `--all` is required because the backlog is `refs/heads` and + * `refs/remotes`, which a bare `pack-refs` leaves alone. + */ +export const PACK_REFS_ARGS = ['pack-refs', '--all', '--prune'] as const + +/** + * Backstop on a whole attempt: aborts it, rather than abandoning it. Every Git + * child is already deadlined, but an admission wait is not, and the whole app + * shares one maintenance slot. Abandoning would release that slot while a pack + * that may still hold `packed-refs.lock` runs on, so the deadline cancels the + * work instead and the slot is held until it really stops. + */ +export const REF_MAINTENANCE_ATTEMPT_DEADLINE_MS = PACK_REFS_TIMEOUT_MS + 5 * 60_000 + +export type RefMaintenanceOutcome = + | 'packed' + | 'below_threshold' + | 'unresolved' + | 'opted_out' + | 'deferred' + | 'interrupted' + | 'locked' + | 'timed_out' + | 'failed' + +/** Structurally satisfied by the tracer's `ActiveSpan`. */ +export type RefMaintenanceSpan = { + setAttribute(key: string, value: unknown): void +} + +export type RepoRefMaintenanceTarget = { + /** Repo identity scoped to its execution host; all state here is keyed by it. */ + readonly key: string + /** Absolute `refs/` path *on the host that runs the walk*, or undefined if unresolvable. */ + resolveRefsDirectory(signal: AbortSignal): Promise + /** A user who told Git not to auto-maintain this repo has told Orca too. */ + isOptedOut?(signal: AbortSignal): Promise + /** True while work on *this repo* is in flight -- a fetch, a create, a removal. */ + isBusy?(): boolean + /** + * Runs `pack-refs` to completion. Deliberately takes no abort signal: killing + * a pack is measurably worse than waiting for it (see `PACKED_REFS_LOCK_*`). + * It must report `packed-refs.lock` transitions through `lock` so callers can + * wait for the short window that actually blocks them. + */ + packRefs(lock: PackedRefsLockReporter): Promise +} + +/** How `packRefs` tells the scheduler whether the exclusive write window is open. */ +export type PackedRefsLockReporter = { + setHeld(held: boolean): void +} + +export type RepoRefMaintenanceOptions = { + now?: () => number + /** True while app-wide work this must not race is in flight (create, live agent, battery, quit). */ + isBusy?: () => boolean + /** Wraps one attempt so a host can trace it; must invoke and await `attempt`. */ + observe?: (attempt: (span: RefMaintenanceSpan) => Promise) => Promise + quietPeriodMs?: number + looseRefThreshold?: number + onError?: (error: unknown) => void +} + +/** Marks an abort Orca asked for, so the attempt is retried rather than blamed on the repo. */ +export class RefMaintenanceInterrupted extends Error { + constructor( + reason: string, + /** True when the attempt ran out of time rather than yielding to real work. */ + readonly deadline = false + ) { + super(`Ref maintenance interrupted: ${reason}`) + this.name = 'RefMaintenanceInterrupted' + } +} + +/** + * The repository's `packed-refs.lock` is held by something we must not touch. + * Distinct from a failure so a strand our own dead process left can be retried + * once it ages into reclaimability, rather than parked for six hours. + */ +export class RefMaintenanceRepoLocked extends Error { + constructor(detail: string) { + super(`packed-refs.lock is held: ${detail}`) + this.name = 'RefMaintenanceRepoLocked' + } +} diff --git a/src/shared/repo-ref-maintenance.test.ts b/src/shared/repo-ref-maintenance.test.ts new file mode 100644 index 00000000000..f59b894748f --- /dev/null +++ b/src/shared/repo-ref-maintenance.test.ts @@ -0,0 +1,530 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { RepoRefMaintenance } from './repo-ref-maintenance' +import { + RefMaintenanceRepoLocked, + REF_MAINTENANCE_PACKED_COOLDOWN_MS, + PACKED_REFS_LOCK_WAIT_MS, + type PackedRefsLockReporter, + type RefMaintenanceSpan, + type RepoRefMaintenanceOptions, + type RepoRefMaintenanceTarget +} from './repo-ref-maintenance-policy' + +const QUIET_MS = 1000 +const THRESHOLD = 5 +const roots: string[] = [] + +async function refsDirectoryWith(looseRefs: number): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-ref-maintenance-')) + roots.push(root) + const refs = join(root, 'refs', 'remotes', 'origin') + await mkdir(refs, { recursive: true }) + for (let index = 0; index < looseRefs; index += 1) { + await writeFile(join(refs, `ref-${index}`), 'a') + } + return join(root, 'refs') +} + +/** More directories than `countLooseRefs` will visit, but very few files. */ +async function saturatingRefsDirectory(): Promise { + const root = await mkdtemp(join(tmpdir(), 'orca-ref-maintenance-wide-')) + roots.push(root) + const refs = join(root, 'refs') + for (let index = 0; index < 4200; index += 1) { + await mkdir(join(refs, `ns-${index}`), { recursive: true }) + } + return refs +} + +/** Stands in for a pack that moved the refs into packed-refs before erroring. */ +async function emptyRefsDirectory(refs: string): Promise { + await rm(refs, { recursive: true, force: true }) +} + +function attributesOf(span: RefMaintenanceSpan): Record { + return (span as unknown as { recorded: Record }).recorded +} + +function recordingSpan(): RefMaintenanceSpan { + const recorded: Record = {} + return { + recorded, + setAttribute(key: string, value: unknown) { + recorded[key] = value + } + } as unknown as RefMaintenanceSpan +} + +type Harness = { + maintenance: RepoRefMaintenance + spans: RefMaintenanceSpan[] + packRefs: ((lock: PackedRefsLockReporter) => Promise) & { mock: { calls: unknown[] } } +} + +function createHarness( + overrides: Partial & { + packRefs?: (lock: PackedRefsLockReporter) => Promise + } = {} +): Harness { + const spans: RefMaintenanceSpan[] = [] + const packRefs = vi.fn<(lock: PackedRefsLockReporter) => Promise>( + overrides.packRefs ?? (async () => {}) + ) + const maintenance = new RepoRefMaintenance({ + quietPeriodMs: QUIET_MS, + looseRefThreshold: THRESHOLD, + now: () => Date.now(), + observe: (attempt) => { + const span = recordingSpan() + spans.push(span) + return attempt(span) + }, + ...overrides + }) + return { maintenance, spans, packRefs } +} + +function target( + key: string, + refsDirectory: string, + packRefs: (lock: PackedRefsLockReporter) => Promise, + extra: Partial = {} +): RepoRefMaintenanceTarget { + return { + key, + resolveRefsDirectory: async () => refsDirectory, + packRefs, + ...extra + } +} + +/** Resolves the first time the pack starts, so tests never race real filesystem I/O. */ +function packStartSignal(): { + started: Promise + onStart: (lock: PackedRefsLockReporter) => void +} { + let onStart: (lock: PackedRefsLockReporter) => void = () => {} + const started = new Promise((resolve) => { + onStart = resolve + }) + return { started, onStart } +} + +function yieldToIo(): Promise { + return new Promise((resolve) => setImmediate(resolve)) +} + +/** + * Spins the real event loop until `predicate` holds, so filesystem completions + * can land while `setTimeout` is faked. Bounded by wall clock rather than by a + * turn count: a loaded CI runner exhausts a fixed number of turns long before + * the I/O finishes, which fails as a confusing assertion somewhere else. + */ +async function until(predicate: () => boolean, what: string): Promise { + const deadline = Date.now() + 10_000 + while (!predicate() && Date.now() < deadline) { + await yieldToIo() + } + if (!predicate()) { + throw new Error(`timed out after 10s waiting for ${what}`) + } +} + +/** + * Like `until`, but for conditions that also need a scheduled retry to fire: + * spinning the real loop alone can never satisfy them, because `setTimeout` is + * faked. Alternates advancing the fake clock with yielding to real I/O. + */ +async function untilWithTimers(predicate: () => boolean, what: string): Promise { + const deadline = Date.now() + 10_000 + while (!predicate() && Date.now() < deadline) { + await vi.advanceTimersByTimeAsync(QUIET_MS) + await yieldToIo() + } + if (!predicate()) { + throw new Error(`timed out after 10s waiting for ${what}`) + } +} + +/** Fires the quiet-period timer and waits for the attempt it starts. */ +async function elapseQuietPeriod(maintenance: RepoRefMaintenance, periods = 1): Promise { + await vi.advanceTimersByTimeAsync(QUIET_MS * periods) + await maintenance.whenAttemptSettled() +} + +/** Only the quiet-period timer is faked; real filesystem I/O still has to complete. */ +beforeEach(() => { + vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout'] }) +}) + +afterEach(async () => { + vi.useRealTimers() + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('RepoRefMaintenance gating', () => { + it('packs only after the repo has been quiet for the full period', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, spans, packRefs } = createHarness() + const repo = target('local::/repo/.git', refs, packRefs) + + maintenance.arm(repo) + await vi.advanceTimersByTimeAsync(QUIET_MS - 1) + expect(packRefs).not.toHaveBeenCalled() + + // A second write restarts the countdown rather than shortening it. + maintenance.arm(repo) + await vi.advanceTimersByTimeAsync(QUIET_MS - 1) + expect(packRefs).not.toHaveBeenCalled() + + await elapseQuietPeriod(maintenance) + expect(packRefs).toHaveBeenCalledTimes(1) + expect(attributesOf(spans[0])).toMatchObject({ + 'repo.maintenance_outcome': 'packed', + 'repo.maintenance_key': 'local::/repo/.git', + 'git.loose_ref_count': THRESHOLD + 1 + }) + }) + + it('leaves a healthy repository alone', async () => { + const refs = await refsDirectoryWith(THRESHOLD - 1) + const { maintenance, spans, packRefs } = createHarness() + + maintenance.arm(target('local::/healthy/.git', refs, packRefs)) + await elapseQuietPeriod(maintenance) + + expect(packRefs).not.toHaveBeenCalled() + expect(attributesOf(spans[0])).toMatchObject({ + 'repo.maintenance_outcome': 'below_threshold', + 'git.loose_ref_count': THRESHOLD - 1 + }) + }) + + it('does not run while the app is busy', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + let busy = true + const { maintenance, packRefs } = createHarness({ isBusy: () => busy }) + + maintenance.arm(target('local::/busy/.git', refs, packRefs)) + await elapseQuietPeriod(maintenance) + expect(packRefs).not.toHaveBeenCalled() + + // The deferral re-arms on a backed-off delay, so the next window picks it up. + busy = false + await elapseQuietPeriod(maintenance, 2) + expect(packRefs).toHaveBeenCalledTimes(1) + }) + + it('does not run while the repo itself has work in flight', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, packRefs } = createHarness() + + maintenance.arm(target('local::/fetching/.git', refs, packRefs, { isBusy: () => true })) + await elapseQuietPeriod(maintenance) + + expect(packRefs).not.toHaveBeenCalled() + }) + + it('honours a user who disabled Git auto-maintenance for the repo', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, spans, packRefs } = createHarness() + + maintenance.arm( + target('local::/opted-out/.git', refs, packRefs, { isOptedOut: async () => true }) + ) + await elapseQuietPeriod(maintenance) + + expect(packRefs).not.toHaveBeenCalled() + expect(attributesOf(spans[0])['repo.maintenance_outcome']).toBe('opted_out') + }) + + it('never reads a truncated walk as a clean repository', async () => { + const { maintenance, spans, packRefs } = createHarness() + + // A walk that stopped early reports a floor, so a low count is not evidence of health. + maintenance.arm({ + key: 'local::/saturated/.git', + resolveRefsDirectory: async () => saturatingRefsDirectory(), + packRefs + }) + await elapseQuietPeriod(maintenance) + + expect(packRefs).toHaveBeenCalledTimes(1) + expect(attributesOf(spans[0])['repo.maintenance_outcome']).toBe('packed') + }) + + it('records a repo whose packed-refs lock is held, and retries sooner than a failure', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, spans } = createHarness() + + maintenance.arm( + target('local::/locked/.git', refs, async () => { + throw new RefMaintenanceRepoLocked('our own lock, not yet old enough to reclaim') + }) + ) + await elapseQuietPeriod(maintenance) + + expect(attributesOf(spans[0])['repo.maintenance_outcome']).toBe('locked') + }) + + it('skips a repository whose common dir cannot be resolved', async () => { + const { maintenance, spans, packRefs } = createHarness() + + maintenance.arm({ + key: 'local::/gone/.git', + resolveRefsDirectory: async () => undefined, + packRefs + }) + await elapseQuietPeriod(maintenance) + + expect(packRefs).not.toHaveBeenCalled() + expect(attributesOf(spans[0])['repo.maintenance_outcome']).toBe('unresolved') + }) +}) + +describe('RepoRefMaintenance single-flight and backoff', () => { + it('runs one repository at a time', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + let concurrent = 0 + let peak = 0 + const releases: (() => void)[] = [] + const { maintenance } = createHarness() + const slowPack = async (): Promise => { + concurrent += 1 + peak = Math.max(peak, concurrent) + await new Promise((resolve) => releases.push(resolve)) + concurrent -= 1 + } + + maintenance.arm(target('local::/a/.git', refs, slowPack)) + maintenance.arm(target('local::/b/.git', refs, slowPack)) + await vi.advanceTimersByTimeAsync(QUIET_MS) + await until(() => concurrent === 1, 'a pack to start') + expect(concurrent).toBe(1) + + releases.shift()?.() + await maintenance.whenAttemptSettled() + // The second repo was deferred behind the first, so its retry is on a timer. + await untilWithTimers(() => concurrent === 1, 'the second repo to start') + releases.shift()?.() + await maintenance.whenAttemptSettled() + + expect(peak).toBe(1) + expect(concurrent).toBe(0) + }) + + it('waits out the rewrite window instead of killing the pack', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + let finished = false + let release: (() => void) | undefined + const { maintenance } = createHarness() + const started = packStartSignal() + + maintenance.arm( + target('local::/yield/.git', refs, async (lock) => { + lock.setHeld(true) + started.onStart(lock) + await new Promise((resolve) => { + release = () => { + lock.setHeld(false) + resolve() + } + }) + finished = true + }) + ) + await vi.advanceTimersByTimeAsync(QUIET_MS) + await started.started + + let paused = false + void maintenance.pause('worktree-remove').then(() => { + paused = true + }) + await vi.advanceTimersByTimeAsync(1) + // Blocked while the rewrite window is open... + expect(paused).toBe(false) + expect(finished).toBe(false) + + release?.() + await until(() => paused, 'pause() to resolve') + // ...and released without the pack ever being cancelled. + expect(paused).toBe(true) + expect(finished).toBe(true) + }) + + it('gives up waiting on the lock rather than blocking the user indefinitely', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance } = createHarness() + const started = packStartSignal() + + maintenance.arm( + target('local::/stuck-lock/.git', refs, async (lock) => { + lock.setHeld(true) + started.onStart(lock) + await new Promise(() => {}) + }) + ) + await vi.advanceTimersByTimeAsync(QUIET_MS) + await started.started + + let paused = false + void maintenance.pause('git-fetch').then(() => { + paused = true + }) + await vi.advanceTimersByTimeAsync(PACKED_REFS_LOCK_WAIT_MS) + await until(() => paused, 'pause() to resolve') + + expect(paused).toBe(true) + }) + + it('reopens the window only when the last overlapping caller releases', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, packRefs } = createHarness() + + const outer = await maintenance.pause('worktree-add') + const inner = await maintenance.pause('git-fetch') + maintenance.arm(target('local::/nested/.git', refs, packRefs)) + + await elapseQuietPeriod(maintenance, 8) + expect(packRefs).not.toHaveBeenCalled() + + inner() + await elapseQuietPeriod(maintenance, 8) + expect(packRefs).not.toHaveBeenCalled() + + outer() + await elapseQuietPeriod(maintenance, 8) + expect(packRefs).toHaveBeenCalledTimes(1) + }) + + it('restarts every armed countdown when the user does ref work themselves', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, packRefs } = createHarness() + + const firstPack = packStartSignal() + const observed = target('local::/a/.git', refs, async (lock) => { + firstPack.onStart(lock) + await packRefs(lock) + }) + maintenance.arm(observed) + maintenance.arm(target('local::/b/.git', refs, packRefs)) + await vi.advanceTimersByTimeAsync(QUIET_MS - 1) + + // A manual fetch says the user is at the keyboard, so nothing may fire yet. + maintenance.postponeAll() + await vi.advanceTimersByTimeAsync(QUIET_MS - 1) + expect(packRefs).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(QUIET_MS) + await firstPack.started + expect(packRefs).toHaveBeenCalled() + }) + + it('costs nothing when no pack is running', async () => { + const { maintenance } = createHarness() + + const release = await maintenance.pause('git-fetch') + release() + // Releasing twice must not leave the window wedged shut. + release() + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { packRefs } = createHarness() + maintenance.arm(target('local::/free/.git', refs, packRefs)) + await elapseQuietPeriod(maintenance) + + expect(packRefs).toHaveBeenCalledTimes(1) + }) + + it('does not re-pack a repository inside its cooldown', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + let clock = 0 + const { maintenance, packRefs } = createHarness({ now: () => clock }) + const repo = target('local::/cooldown/.git', refs, packRefs) + + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + expect(packRefs).toHaveBeenCalledTimes(1) + + clock = REF_MAINTENANCE_PACKED_COOLDOWN_MS - 1 + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + expect(packRefs).toHaveBeenCalledTimes(1) + + clock = REF_MAINTENANCE_PACKED_COOLDOWN_MS + 1 + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + expect(packRefs).toHaveBeenCalledTimes(2) + }) + + it('counts a pack that could not lock every ref as a success', async () => { + // Field-observed on a machine running several Orca sessions: a branch moved + // mid-pack, Git reported an error, and 36,688 loose refs still became 3. + // Retrying that aggressively would be wrong -- the backlog is gone. + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, spans } = createHarness() + const repo = target('local::/raced/.git', refs, async () => { + await emptyRefsDirectory(refs) + throw new Error("error: cannot lock ref 'refs/heads/moved'") + }) + + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + + expect(attributesOf(spans[0])).toMatchObject({ + 'repo.maintenance_outcome': 'packed', + 'git.pack_refs_partial': true, + 'git.loose_ref_count_after': 0 + }) + + // And it serves the full post-pack cooldown rather than retrying. + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + expect(spans).toHaveLength(1) + }) + + it('records a failure when the pack left the backlog in place', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, spans } = createHarness({ + packRefs: async () => { + throw new Error('permission denied') + } + }) + + maintenance.arm(target('local::/denied/.git', refs, () => Promise.reject(new Error('denied')))) + await elapseQuietPeriod(maintenance) + + expect(attributesOf(spans[0])['repo.maintenance_outcome']).toBe('failed') + }) + + it('records a failure instead of throwing, and backs off', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, spans, packRefs } = createHarness({ + packRefs: async () => { + throw new Error('packed-refs.lock exists') + } + }) + const repo = target('local::/failing/.git', refs, packRefs) + + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + expect(attributesOf(spans[0])['repo.maintenance_outcome']).toBe('failed') + + maintenance.arm(repo) + await elapseQuietPeriod(maintenance) + expect(packRefs).toHaveBeenCalledTimes(1) + }) + + it('stops scheduling once disposed', async () => { + const refs = await refsDirectoryWith(THRESHOLD + 2) + const { maintenance, packRefs } = createHarness() + + maintenance.arm(target('local::/disposed/.git', refs, packRefs)) + maintenance.dispose() + await elapseQuietPeriod(maintenance) + + expect(packRefs).not.toHaveBeenCalled() + }) +}) diff --git a/src/shared/repo-ref-maintenance.ts b/src/shared/repo-ref-maintenance.ts new file mode 100644 index 00000000000..97e228fab52 --- /dev/null +++ b/src/shared/repo-ref-maintenance.ts @@ -0,0 +1,366 @@ +import { countLooseRefs } from './loose-ref-count' +import { PackedRefsLockGate } from './packed-refs-lock-gate' +import { + LOOSE_REF_PACK_THRESHOLD, + REF_MAINTENANCE_ATTEMPT_DEADLINE_MS, + REF_MAINTENANCE_CLEAN_COOLDOWN_MS, + REF_MAINTENANCE_FAILURE_COOLDOWN_MS, + REF_MAINTENANCE_PACKED_COOLDOWN_MS, + REF_MAINTENANCE_QUIET_PERIOD_MS, + PACKED_REFS_LOCK_WAIT_MS, + REF_MAINTENANCE_LOCKED_COOLDOWN_MS, + RefMaintenanceInterrupted, + RefMaintenanceRepoLocked, + type RefMaintenanceOutcome, + type RefMaintenanceSpan, + type RepoRefMaintenanceOptions, + type RepoRefMaintenanceTarget +} from './repo-ref-maintenance-policy' + +/** + * The scheduler half of idle loose-ref packing: when to probe, when to pack, + * when to stand down. The thresholds and the host contract it works against + * live in `./repo-ref-maintenance-policy`. + */ + +/** Give up until the next real activity rather than re-arming forever. */ +const MAX_DEFERRALS = 6 +/** Each deferral doubles the wait, so a busy app is retried rarely, not hammered. */ +const MAX_DEFERRAL_BACKOFF_MULTIPLIER = 8 +/** Armed repos are evicted oldest-first past this; the next write on one re-arms it. */ +const MAX_TRACKED_REPOS = 64 + +type TrackedRepo = { + target: RepoRefMaintenanceTarget + timer: ReturnType | null + deferrals: number +} + +const noopSpan: RefMaintenanceSpan = { setAttribute: () => {} } + +/** A deadline means something is stuck: back off instead of retrying straight away. */ +function hitDeadline(signal: AbortSignal): boolean { + return signal.reason instanceof RefMaintenanceInterrupted && signal.reason.deadline +} + +export class RepoRefMaintenance { + private readonly tracked = new Map() + private readonly cooldownUntil = new Map() + private readonly now: () => number + private readonly isAppBusy: () => boolean + private readonly observe: NonNullable + private readonly quietPeriodMs: number + private readonly looseRefThreshold: number + private readonly onError: (error: unknown) => void + // Why: at most one pack-refs anywhere. It holds a general git admission slot + // for its whole run, and two at once would halve git throughput on a small host. + // The slot is never released while a pack that could hold `packed-refs.lock` + // is still running -- an interrupt cancels the work and waits for it to stop. + private inFlight: Promise | null = null + private inFlightAbort: AbortController | null = null + private readonly lockGate = new PackedRefsLockGate() + // Why a count, not a flag: several ref-touching operations overlap routinely + // (a create's fetch inside a create), and the last one out reopens the window. + private suspensions = 0 + private lastAttempt: Promise = Promise.resolve() + private disposed = false + + constructor(options: RepoRefMaintenanceOptions = {}) { + this.now = options.now ?? Date.now + this.isAppBusy = options.isBusy ?? (() => false) + this.observe = options.observe ?? ((attempt) => attempt(noopSpan)) + this.quietPeriodMs = options.quietPeriodMs ?? REF_MAINTENANCE_QUIET_PERIOD_MS + this.looseRefThreshold = options.looseRefThreshold ?? LOOSE_REF_PACK_THRESHOLD + this.onError = options.onError ?? (() => {}) + } + + /** + * Record a write to `target`'s repo and (re)start its quiet-period countdown. + * Every call pushes the attempt further out, so a burst of fetches or a + * worktree create can never be interrupted by maintenance it triggered. + */ + arm(target: RepoRefMaintenanceTarget): void { + if (this.disposed) { + return + } + const existing = this.tracked.get(target.key) + if (existing?.timer) { + clearTimeout(existing.timer) + } + const tracked: TrackedRepo = { target, timer: null, deferrals: existing?.deferrals ?? 0 } + this.tracked.delete(target.key) + this.evictOldestBeyondCap() + this.tracked.set(target.key, tracked) + this.schedule(target.key, tracked) + } + + /** Resolves once the attempt started by the most recent timer has settled. */ + whenAttemptSettled(): Promise { + return this.lastAttempt + } + + /** + * Wait out the `packed-refs` rewrite, if one is in progress. + * + * Deliberately not a kill. The lock is held for 0.03-1.37s of a 23-32s pack; + * the rest is the prune phase, during which a concurrent `fetch --prune`, + * `branch -D` or `update-ref` measurably succeeds because per-ref locks last + * microseconds and Git retries for `core.filesRefLockTimeout`. Signalling the + * child there buys nothing and strands a lock file about one time in five. + * + * Free when no pack is running, which is almost always. + */ + awaitPackedRefsLockRelease(): Promise { + return this.lockGate.whenReleased(PACKED_REFS_LOCK_WAIT_MS) + } + + /** + * Push every armed repository's attempt out by a full quiet period. + * + * User-initiated ref work is evidence the user is active in the app, not just + * in one repo, and it is free -- no key to resolve, no subprocess, nothing at + * all when nothing is armed. + */ + postponeAll(): void { + if (this.disposed) { + return + } + for (const [key, tracked] of this.tracked) { + if (tracked.timer) { + clearTimeout(tracked.timer) + } + tracked.deferrals = 0 + this.schedule(key, tracked) + } + } + + /** + * Hold the repository open for work that is about to touch refs. + * + * Two things at once: no *new* attempt can start for any repository until the + * returned release is called, and the caller waits out any `packed-refs` + * rewrite already in progress. A prune already running is left alone to + * finish -- it does not block the caller. + */ + async pause(_reason: string): Promise<() => void> { + this.suspensions += 1 + let released = false + try { + await this.awaitPackedRefsLockRelease() + } catch { + // The wait cannot reject, but a release must exist even if it did. + } + return () => { + if (!released) { + released = true + this.suspensions -= 1 + } + } + } + + dispose(): void { + this.disposed = true + this.inFlightAbort?.abort(new RefMaintenanceInterrupted('disposed')) + for (const tracked of this.tracked.values()) { + if (tracked.timer) { + clearTimeout(tracked.timer) + } + } + this.tracked.clear() + this.cooldownUntil.clear() + } + + private isBusy(tracked: TrackedRepo): boolean { + return this.isAppBusy() || (tracked.target.isBusy?.() ?? false) + } + + private schedule(key: string, tracked: TrackedRepo, delayMs = this.quietPeriodMs): void { + const timer = setTimeout(() => { + tracked.timer = null + this.lastAttempt = this.attempt(key).catch((error) => this.onError(error)) + }, delayMs) + // Never hold the process open for maintenance. + timer.unref?.() + tracked.timer = timer + } + + private evictOldestBeyondCap(): void { + while (this.tracked.size >= MAX_TRACKED_REPOS) { + const oldest = this.tracked.keys().next() + if (oldest.done) { + return + } + const evicted = this.tracked.get(oldest.value) + if (evicted?.timer) { + clearTimeout(evicted.timer) + } + this.tracked.delete(oldest.value) + } + } + + /** + * `counted` spends the give-up budget. Waiting behind another repository's + * pack, or yielding to work Orca asked us to yield to, does not: both end on + * their own, so charging for them would let a busy machine starve a repo + * until its next fetch. Only "the app is busy" is charged. + */ + private defer(key: string, tracked: TrackedRepo, counted: boolean): void { + // A fetch that landed while this attempt was probing already re-armed the + // repo; that entry is fresher, so the deferral must not overwrite it. + if (this.disposed || this.tracked.has(key)) { + return + } + if (counted) { + if (tracked.deferrals >= MAX_DEFERRALS) { + return + } + tracked.deferrals += 1 + } + this.tracked.set(key, tracked) + const multiplier = Math.min(2 ** tracked.deferrals, MAX_DEFERRAL_BACKOFF_MULTIPLIER) + this.schedule(key, tracked, this.quietPeriodMs * multiplier) + } + + private async attempt(key: string): Promise { + const tracked = this.tracked.get(key) + if (!tracked || this.disposed) { + return + } + this.tracked.delete(key) + const cooldownUntil = this.cooldownUntil.get(key) + if (cooldownUntil !== undefined && this.now() < cooldownUntil) { + return + } + if (this.inFlight !== null) { + this.defer(key, tracked, false) + return + } + if (this.suspensions > 0 || this.isBusy(tracked)) { + this.defer(key, tracked, true) + return + } + const abort = new AbortController() + const deadline = setTimeout( + () => abort.abort(new RefMaintenanceInterrupted('attempt deadline', true)), + REF_MAINTENANCE_ATTEMPT_DEADLINE_MS + ) + deadline.unref?.() + const run = this.observe((span) => this.packIfNeeded(key, tracked, span, abort.signal)) + this.inFlight = run + this.inFlightAbort = abort + try { + await run + } finally { + clearTimeout(deadline) + if (this.inFlight === run) { + this.inFlight = null + this.inFlightAbort = null + } + } + } + + private async packIfNeeded( + key: string, + tracked: TrackedRepo, + span: RefMaintenanceSpan, + signal: AbortSignal + ): Promise { + span.setAttribute('repo.maintenance_key', key) + // Every await below carries the signal, so a caller waiting in `pause()` is + // never stuck behind a probe that has already been told to stop. + if (await tracked.target.isOptedOut?.(signal)) { + this.settle(key, span, 'opted_out', REF_MAINTENANCE_CLEAN_COOLDOWN_MS) + return + } + if (signal.aborted) { + this.yieldTo(key, tracked, span, signal) + return + } + const refsDirectory = await tracked.target.resolveRefsDirectory(signal) + if (!refsDirectory) { + this.settle(key, span, 'unresolved', REF_MAINTENANCE_CLEAN_COOLDOWN_MS) + return + } + const budget = this.looseRefThreshold + 1 + const before = await countLooseRefs(refsDirectory, budget, signal) + if (signal.aborted) { + this.yieldTo(key, tracked, span, signal) + return + } + span.setAttribute('git.loose_ref_count', before.count) + span.setAttribute('git.loose_ref_threshold', this.looseRefThreshold) + // A saturated walk stopped early, so `count` is a floor -- never read it as "clean". + if (!before.saturated && before.count < this.looseRefThreshold) { + this.settle(key, span, 'below_threshold', REF_MAINTENANCE_CLEAN_COOLDOWN_MS) + return + } + // The quiet window can close while the probe walks; re-check before spending a git slot. + if (this.suspensions > 0 || this.isBusy(tracked)) { + span.setAttribute('repo.maintenance_outcome', 'deferred' satisfies RefMaintenanceOutcome) + this.defer(key, tracked, true) + return + } + const startedAt = this.now() + let partial = false + try { + // No signal: the pack runs to completion. Callers that need the refs wait + // out the rewrite window through `pause()` instead of killing it. + await tracked.target.packRefs(this.lockGate) + } catch (error) { + span.setAttribute('repo.maintenance_error', String(error)) + if (error instanceof RefMaintenanceRepoLocked) { + this.settle(key, span, 'locked', REF_MAINTENANCE_LOCKED_COOLDOWN_MS) + return + } + partial = true + } finally { + this.lockGate.setHeld(false) + } + span.setAttribute('git.pack_refs_ms', this.now() - startedAt) + // Judge by the backlog, not by the exit code. On a machine running several + // Orca sessions a branch moving mid-pack is the normal case, and Git's + // response -- leave that one ref loose, pack the rest -- is the correct one. + // Measured in the field: 36,688 loose refs down to 3, reported as an error. + const after = await countLooseRefs(refsDirectory, budget, signal) + span.setAttribute('git.loose_ref_count_after', after.count) + if (partial && (after.saturated || after.count >= this.looseRefThreshold)) { + this.settle(key, span, 'failed', REF_MAINTENANCE_FAILURE_COOLDOWN_MS) + return + } + span.setAttribute('git.pack_refs_partial', partial) + this.settle(key, span, 'packed', REF_MAINTENANCE_PACKED_COOLDOWN_MS) + } + + /** Record an aborted attempt: retry soon if Orca yielded, back off if it stalled. */ + private yieldTo( + key: string, + tracked: TrackedRepo, + span: RefMaintenanceSpan, + signal: AbortSignal + ): void { + if (hitDeadline(signal)) { + this.settle(key, span, 'timed_out', REF_MAINTENANCE_FAILURE_COOLDOWN_MS) + return + } + span.setAttribute('repo.maintenance_outcome', 'interrupted' satisfies RefMaintenanceOutcome) + this.defer(key, tracked, false) + } + + private settle( + key: string, + span: RefMaintenanceSpan, + outcome: RefMaintenanceOutcome, + cooldownMs: number + ): void { + span.setAttribute('repo.maintenance_outcome', outcome) + // Re-insert so Map order stays newest-last and the eviction below drops the oldest. + this.cooldownUntil.delete(key) + this.cooldownUntil.set(key, this.now() + cooldownMs) + if (this.cooldownUntil.size > MAX_TRACKED_REPOS * 4) { + const oldest = this.cooldownUntil.keys().next() + if (!oldest.done) { + this.cooldownUntil.delete(oldest.value) + } + } + } +} From 0352c239c249a355a26c132b9aeb071b55bc66fb Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:20:48 -0700 Subject: [PATCH 009/398] Add Copy Session ID menu item to terminal tabs (#18039) * Add Copy Session ID menu item to terminal tabs Adds a menu item to copy the active pane's agent session ID when available. The item only appears when the session is still live and has reported an ID. * Add Copy Session ID i18n strings and e2e test - Add localized strings for Session ID context menu item - Add e2e test coverage for copying session ID from terminal tabs - Fix dev build permissions when copying private Electron app bundles * Drop the Electron dev-bundle fix from this branch It landed on main as 519af49a58, which restores write permission inside copyPrivateTree itself rather than at the dev runner's call site, so every caller of the private-copy contract is covered and not just this one. That commit also fixes the test that should have caught the crash: the wrapper ran with stdio: 'ignore', so a hard failure presented as a bare timeout. This branch predated that commit and carried a narrower duplicate, mixed into an i18n/e2e commit where it did not belong. * refactor: use dedicated i18n keys for copy session ID toasts Replace auto-generated translation keys with specific, dedicated keys for copy session ID success and error messages. This improves maintainability and makes the strings easier to translate across all supported languages. --- .../tab-bar/SortableTabContextMenu.test.tsx | 59 +++++++ .../tab-bar/SortableTabContextMenu.tsx | 7 + .../TabAgentSessionIdMenuItem.test.tsx | 79 +++++++++ .../tab-bar/TabAgentSessionIdMenuItem.tsx | 48 ++++++ .../tab-bar/tab-agent-session-id.test.ts | 159 ++++++++++++++++++ .../tab-bar/tab-agent-session-id.ts | 32 ++++ .../tab-context-menu-consistency.test.tsx | 1 + src/renderer/src/i18n/locales/en.json | 5 +- src/renderer/src/i18n/locales/es.json | 5 +- src/renderer/src/i18n/locales/ja.json | 5 +- src/renderer/src/i18n/locales/ko.json | 5 +- src/renderer/src/i18n/locales/zh.json | 5 +- tests/e2e/tab-context-menu-session-id.spec.ts | 76 +++++++++ 13 files changed, 481 insertions(+), 5 deletions(-) create mode 100644 src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx create mode 100644 src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx create mode 100644 src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts create mode 100644 src/renderer/src/components/tab-bar/tab-agent-session-id.ts create mode 100644 tests/e2e/tab-context-menu-session-id.spec.ts diff --git a/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx b/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx index 76d5d8d145a..e51c599f669 100644 --- a/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx +++ b/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx @@ -55,6 +55,7 @@ vi.mock('lucide-react', () => ({ ArrowRight: () => null, ArrowUp: () => null, Columns2: () => null, + Copy: () => null, ListX: () => null, MessageSquare: () => null, PanelBottomClose: () => null, @@ -71,6 +72,8 @@ vi.mock('@/i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) +vi.mock('sonner', () => ({ toast: { success: vi.fn(), error: vi.fn() } })) + vi.mock('../../store', () => ({ useAppStore: Object.assign( (selector: (state: Record) => unknown) => selector(storeMock.state), @@ -289,4 +292,60 @@ describe('SortableTabContextMenu', () => { expect(container.textContent).not.toContain('Move Tab to Split') expect(container.textContent).toContain('Split terminal right') }) + + describe('copy session id', () => { + const LEAF = '11111111-1111-4111-8111-111111111111' + + function withLiveAgent(sessionId: string | null): void { + storeMock.state = { + ...storeMock.state, + terminalLayoutsByTabId: { + 'term-1': { root: { type: 'leaf', leafId: LEAF }, activeLeafId: LEAF } + }, + agentStatusByPaneKey: { + [`term-1:${LEAF}`]: { + state: 'done', + prompt: '', + updatedAt: 1, + stateStartedAt: 1, + paneKey: `term-1:${LEAF}`, + agentType: 'claude', + stateHistory: [], + ...(sessionId ? { providerSession: { key: 'session_id', id: sessionId } } : {}) + } + }, + paneForegroundAgentByPaneKey: {} + } + } + + it('omits the item for a tab with no agent', () => { + const { container } = renderMenu() + + expect(container.textContent).not.toContain('Copy Session ID') + }) + + it('omits the item until the active agent reports a session id', () => { + withLiveAgent(null) + const { container } = renderMenu() + + expect(container.textContent).not.toContain('Copy Session ID') + }) + + it('copies the active pane session id', async () => { + const writeClipboardText = vi.fn().mockResolvedValue(undefined) + Object.assign(window, { api: { ui: { writeClipboardText } } }) + withLiveAgent('session-abc') + const { container } = renderMenu() + + act(() => getButton(container, 'Copy Session ID').click()) + await vi.waitFor(() => expect(writeClipboardText).toHaveBeenCalledWith('session-abc')) + }) + + it('does not resolve a session id while the menu is closed', () => { + withLiveAgent('session-abc') + const { container } = renderMenu({ open: false }) + + expect(container.textContent).not.toContain('Copy Session ID') + }) + }) }) diff --git a/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx b/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx index 53b31012d39..b298072f9f6 100644 --- a/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx +++ b/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx @@ -11,6 +11,8 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import { useAppStore } from '../../store' import { formatShortcutLabel, useOptionalShortcutLabel } from '@/hooks/useShortcutLabel' import { translate } from '@/i18n/i18n' +import { TabAgentSessionIdMenuItem } from './TabAgentSessionIdMenuItem' +import { resolveTabAgentSessionId } from './tab-agent-session-id' import { TerminalTabSplitMenuSection } from './TerminalTabSplitMenuSection' import { TAB_CONTEXT_MENU_CONTENT_CLASS } from './tab-context-menu-sizing' @@ -121,6 +123,10 @@ export function SortableTabContextMenu({ onTogglePin }: SortableTabContextMenuProps): React.JSX.Element { const keybindings = useAppStore((state) => state.keybindings) + // The id is a primitive, so unchanged sessions stay referentially stable without a cache. + const agentSessionId = useAppStore((state) => + open ? resolveTabAgentSessionId(state, tab.id) : null + ) const splitRightShortcut = formatShortcutLabel('terminal.splitRight', keybindings) const splitDownShortcut = formatShortcutLabel('terminal.splitDown', keybindings) @@ -188,6 +194,7 @@ export function SortableTabContextMenu({ {translate('auto.components.tab.bar.SortableTabContextMenu.2f697b3c31', 'Change Title')} {renameShortcut ? {renameShortcut} : null} +
{translate('auto.components.tab.bar.SortableTabContextMenu.35e8892fd0', 'Tab Color')} diff --git a/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx b/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx new file mode 100644 index 00000000000..da857ec57bf --- /dev/null +++ b/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx @@ -0,0 +1,79 @@ +/** + * @vitest-environment happy-dom + */ +import { act, type ReactNode } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { TabAgentSessionIdMenuItem } from './TabAgentSessionIdMenuItem' + +const toastMock = vi.hoisted(() => ({ success: vi.fn(), error: vi.fn() })) + +vi.mock('@/components/ui/dropdown-menu', () => ({ + DropdownMenuItem: ({ + children, + disabled, + onSelect, + 'aria-label': ariaLabel + }: { + children?: ReactNode + disabled?: boolean + onSelect?: () => void + 'aria-label'?: string + }) => ( + + ) +})) + +vi.mock('lucide-react', () => ({ Copy: () => null })) +vi.mock('@/i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) +vi.mock('sonner', () => ({ toast: toastMock })) + +const mounted: { container: HTMLDivElement; root: Root }[] = [] + +function render(sessionId: string | null): HTMLDivElement { + const container = document.createElement('div') + document.body.appendChild(container) + const root = createRoot(container) + act(() => root.render()) + mounted.push({ container, root }) + return container +} + +afterEach(() => { + for (const { container, root } of mounted.splice(0)) { + act(() => root.unmount()) + container.remove() + } + toastMock.success.mockReset() + toastMock.error.mockReset() +}) + +describe('TabAgentSessionIdMenuItem', () => { + it('renders nothing when no session id is available', () => { + expect(render(null).textContent).toBe('') + }) + + it('copies on select when an id is known', async () => { + const writeClipboardText = vi.fn().mockResolvedValue(undefined) + Object.assign(window, { api: { ui: { writeClipboardText } } }) + const container = render('abc-123') + + const button = container.querySelector('button') + expect(button?.disabled).toBe(false) + act(() => button?.click()) + await vi.waitFor(() => expect(writeClipboardText).toHaveBeenCalledWith('abc-123')) + }) + + it('reports clipboard failures', async () => { + const writeClipboardText = vi.fn().mockRejectedValue(new Error('clipboard unavailable')) + Object.assign(window, { api: { ui: { writeClipboardText } } }) + const button = render('abc-123').querySelector('button') + + act(() => button?.click()) + await vi.waitFor(() => + expect(toastMock.error).toHaveBeenCalledWith('Failed to copy Session ID') + ) + }) +}) diff --git a/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx b/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx new file mode 100644 index 00000000000..73f3f128e5d --- /dev/null +++ b/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx @@ -0,0 +1,48 @@ +import { Copy } from 'lucide-react' +import { toast } from 'sonner' +import { DropdownMenuItem } from '@/components/ui/dropdown-menu' +import { translate } from '@/i18n/i18n' + +async function copySessionId(sessionId: string): Promise { + try { + await window.api.ui.writeClipboardText(sessionId) + toast.success( + translate( + 'components.tab.bar.SortableTabContextMenu.copySessionIdSuccess', + 'Session ID copied' + ) + ) + } catch { + toast.error( + translate( + 'components.tab.bar.SortableTabContextMenu.copySessionIdError', + 'Failed to copy Session ID' + ) + ) + } +} + +/** Copies the active pane's provider session id when one is available. */ +export function TabAgentSessionIdMenuItem({ + sessionId +}: { + sessionId: string | null +}): React.JSX.Element | null { + if (sessionId === null) { + return null + } + const label = translate( + 'components.tab.bar.SortableTabContextMenu.copySessionId', + 'Copy Session ID' + ) + return ( + { + void copySessionId(sessionId) + }} + > + + {label} + + ) +} diff --git a/src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts b/src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts new file mode 100644 index 00000000000..2f06f0a8db7 --- /dev/null +++ b/src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts @@ -0,0 +1,159 @@ +import { describe, expect, it } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { resolveTabAgentSessionId, type TabAgentSessionIdState } from './tab-agent-session-id' + +const LEAF_A = '11111111-1111-4111-8111-111111111111' +const LEAF_B = '22222222-2222-4222-8222-222222222222' + +function entry(overrides: Partial = {}): AgentStatusEntry { + return { + state: 'done', + prompt: '', + updatedAt: 1, + stateStartedAt: 1, + paneKey: `tab-1:${LEAF_A}`, + agentType: 'claude', + stateHistory: [], + ...overrides + } +} + +function state(overrides: Partial = {}): TabAgentSessionIdState { + return { + terminalLayoutsByTabId: { + 'tab-1': { + root: { type: 'leaf', leafId: LEAF_A }, + activeLeafId: LEAF_A, + expandedLeafId: null + } + }, + agentStatusByPaneKey: {}, + paneForegroundAgentByPaneKey: {}, + ...overrides + } +} + +describe('resolveTabAgentSessionId', () => { + it('is absent when the pane has no agent row', () => { + expect(resolveTabAgentSessionId(state(), 'tab-1')).toBeNull() + }) + + it('is absent for a tab with no layout', () => { + expect(resolveTabAgentSessionId(state(), 'tab-missing')).toBeNull() + }) + + it('reads the id reported by the active pane', () => { + const resolved = resolveTabAgentSessionId( + state({ + agentStatusByPaneKey: { + [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'abc-123' } }) + } + }), + 'tab-1' + ) + expect(resolved).toBe('abc-123') + }) + + it('is absent until the agent reports an id', () => { + const resolved = resolveTabAgentSessionId( + state({ agentStatusByPaneKey: { [`tab-1:${LEAF_A}`]: entry() } }), + 'tab-1' + ) + expect(resolved).toBeNull() + }) + + describe('liveness', () => { + it('is absent for a hydrated row with no live hook since restore', () => { + const resolved = resolveTabAgentSessionId( + state({ + agentStatusByPaneKey: { + [`tab-1:${LEAF_A}`]: entry({ + restoredUnconfirmed: true, + providerSession: { key: 'session_id', id: 'abc-123' } + }) + } + }), + 'tab-1' + ) + expect(resolved).toBeNull() + }) + + it('is absent once the pane is proven back at the shell', () => { + const resolved = resolveTabAgentSessionId( + state({ + agentStatusByPaneKey: { + [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'abc-123' } }) + }, + paneForegroundAgentByPaneKey: { + [`tab-1:${LEAF_A}`]: { agent: null, shellForeground: true } + } + }), + 'tab-1' + ) + expect(resolved).toBeNull() + }) + + it('keeps a session whose foreground evidence is only that an agent runs', () => { + const resolved = resolveTabAgentSessionId( + state({ + agentStatusByPaneKey: { + [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'abc-123' } }) + }, + paneForegroundAgentByPaneKey: { + [`tab-1:${LEAF_A}`]: { agent: 'claude', shellForeground: false } + } + }), + 'tab-1' + ) + expect(resolved).toBe('abc-123') + }) + + it('keeps a working session that reported a session boundary', () => { + // Why: sessionBoundary marks a resume/clear landing idle — a session start, + // not a session end, and exactly when the first id arrives. + const resolved = resolveTabAgentSessionId( + state({ + agentStatusByPaneKey: { + [`tab-1:${LEAF_A}`]: entry({ + sessionBoundary: true, + providerSession: { key: 'session_id', id: 'fresh-1' } + }) + } + }), + 'tab-1' + ) + expect(resolved).toBe('fresh-1') + }) + }) + + describe('split tabs', () => { + const splitState = (activeLeafId: string): TabAgentSessionIdState => + state({ + terminalLayoutsByTabId: { + 'tab-1': { + root: { + type: 'split', + direction: 'vertical', + first: { type: 'leaf', leafId: LEAF_A }, + second: { type: 'leaf', leafId: LEAF_B } + }, + activeLeafId, + expandedLeafId: null + } + }, + agentStatusByPaneKey: { + [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'left' } }), + [`tab-1:${LEAF_B}`]: entry({ providerSession: { key: 'session_id', id: 'right' } }) + } + }) + + it('reads the active pane, not a sibling', () => { + expect(resolveTabAgentSessionId(splitState(LEAF_B), 'tab-1')).toBe('right') + }) + + it('is absent when the active leaf id no longer exists in the layout', () => { + const stale = '33333333-3333-4333-8333-333333333333' + expect(resolveTabAgentSessionId(splitState(stale), 'tab-1')).toBeNull() + }) + }) +}) diff --git a/src/renderer/src/components/tab-bar/tab-agent-session-id.ts b/src/renderer/src/components/tab-bar/tab-agent-session-id.ts new file mode 100644 index 00000000000..e0d804bc4c5 --- /dev/null +++ b/src/renderer/src/components/tab-bar/tab-agent-session-id.ts @@ -0,0 +1,32 @@ +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import type { TerminalLayoutSnapshot } from '../../../../shared/terminal-tab-types' +import type { PaneForegroundAgentEntry } from '../../store/slices/pane-foreground-agent' +import { resolveNativeChatActiveLayoutLeafId } from '../native-chat/native-chat-leaf-routing' + +export type TabAgentSessionIdState = { + agentStatusByPaneKey?: Record + terminalLayoutsByTabId?: Record + paneForegroundAgentByPaneKey?: Record +} + +/** Returns the active pane's provider session id when its agent is still live. */ +export function resolveTabAgentSessionId( + state: TabAgentSessionIdState, + tabId: string +): string | null { + const leafId = resolveNativeChatActiveLayoutLeafId(state.terminalLayoutsByTabId?.[tabId]) + if (!leafId) { + return null + } + const paneKey = `${tabId}:${leafId}` + const entry = state.agentStatusByPaneKey?.[paneKey] + // Hydrated rows may describe a session that ended while no receiver was up. + if (!entry?.agentType || entry.restoredUnconfirmed === true) { + return null + } + // OSC 133;D proves the pane is back at the shell, regardless of the last hook state. + if (state.paneForegroundAgentByPaneKey?.[paneKey]?.shellForeground === true) { + return null + } + return entry.providerSession?.id ?? null +} diff --git a/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx b/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx index c81eae348dd..54f762d4611 100644 --- a/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx +++ b/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx @@ -13,6 +13,7 @@ import { const TAB_MENU_SOURCES = [ 'EditorFileTabContextMenu.tsx', 'SortableTabContextMenu.tsx', + 'TabAgentSessionIdMenuItem.tsx', 'BrowserTab.tsx', 'TabWorkspaceLayoutMenuSection.tsx', 'TerminalTabSplitMenuSection.tsx' diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 7fc9c837ebc..22b1695cc3d 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16724,7 +16724,10 @@ "SortableTabContextMenu": { "switchToTerminalView": "Switch to terminal view", "switchToChatView": "Switch to chat view", - "closeTabsToLeft": "Close Tabs To The Left" + "closeTabsToLeft": "Close Tabs To The Left", + "copySessionId": "Copy Session ID", + "copySessionIdSuccess": "Session ID copied", + "copySessionIdError": "Failed to copy Session ID" }, "BrowserTab": { "closeOthers": "Close Others", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 1c8a5c7325b..be6e087060d 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -14646,7 +14646,10 @@ "SortableTabContextMenu": { "switchToTerminalView": "Cambiar a la vista de terminal", "switchToChatView": "Cambiar a vista de chat", - "closeTabsToLeft": "Cerrar pestañas a la izquierda" + "closeTabsToLeft": "Cerrar pestañas a la izquierda", + "copySessionId": "Copiar ID de sesión", + "copySessionIdSuccess": "ID de sesión copiado", + "copySessionIdError": "No se pudo copiar el ID de sesión" }, "BrowserTab": { "closeOthers": "Cerrar otras", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 334e9e32d27..20e2fca6553 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -14646,7 +14646,10 @@ "SortableTabContextMenu": { "switchToTerminalView": "ターミナルビューに切り替える", "switchToChatView": "チャットビューに切り替える", - "closeTabsToLeft": "左側のタブを閉じる" + "closeTabsToLeft": "左側のタブを閉じる", + "copySessionId": "セッション ID をコピー", + "copySessionIdSuccess": "セッション ID をコピーしました", + "copySessionIdError": "セッション ID のコピーに失敗しました" }, "BrowserTab": { "closeOthers": "その他を閉じる", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 7ef6c53ca86..e9b9c48fcdf 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -14689,7 +14689,10 @@ "SortableTabContextMenu": { "switchToTerminalView": "terminal 보기로 전환", "switchToChatView": "채팅 보기로 전환", - "closeTabsToLeft": "왼쪽으로 탭 닫기" + "closeTabsToLeft": "왼쪽으로 탭 닫기", + "copySessionId": "세션 ID 복사", + "copySessionIdSuccess": "세션 ID를 복사했습니다", + "copySessionIdError": "세션 ID를 복사하지 못했습니다" }, "BrowserTab": { "closeOthers": "다른 탭 닫기", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 06fae6cca89..49560989924 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -14689,7 +14689,10 @@ "SortableTabContextMenu": { "switchToTerminalView": "切换到终端视图", "switchToChatView": "切换到聊天视图", - "closeTabsToLeft": "关闭左侧的选项卡" + "closeTabsToLeft": "关闭左侧的选项卡", + "copySessionId": "复制会话 ID", + "copySessionIdSuccess": "已复制会话 ID", + "copySessionIdError": "复制会话 ID 失败" }, "BrowserTab": { "closeOthers": "关闭其他", diff --git a/tests/e2e/tab-context-menu-session-id.spec.ts b/tests/e2e/tab-context-menu-session-id.spec.ts new file mode 100644 index 00000000000..cab18bace32 --- /dev/null +++ b/tests/e2e/tab-context-menu-session-id.spec.ts @@ -0,0 +1,76 @@ +/** + * E2E coverage for copying an agent provider session ID from a terminal tab's + * context menu. + */ + +import { test, expect } from './helpers/orca-app' +import { + ensureTerminalVisible, + getActiveTabId, + waitForActiveWorktree, + waitForSessionReady +} from './helpers/store' +import { waitForPaneIdentitySnapshot } from './helpers/terminal' + +const SESSION_ID = 'e2e-terminal-tab-session' + +test('terminal tab context menu copies the active agent session ID', async ({ orcaPage }) => { + await waitForSessionReady(orcaPage) + const worktreeId = await waitForActiveWorktree(orcaPage) + await ensureTerminalVisible(orcaPage) + + const tabId = await getActiveTabId(orcaPage) + if (!tabId) { + throw new Error('No active terminal tab') + } + const snapshot = await waitForPaneIdentitySnapshot(orcaPage, 1) + const leafId = snapshot.panes[0]?.leafId + if (!leafId) { + throw new Error('No active terminal pane') + } + const paneKey = `${tabId}:${leafId}` + + // Seed the same renderer state a live agent hook produces while keeping the + // test independent of an installed provider CLI. + await orcaPage.evaluate( + ({ paneKey, tabId, worktreeId, sessionId }) => { + const state = window.__store?.getState() + if (!state) { + throw new Error('Store unavailable') + } + state.setAgentStatus( + paneKey, + { state: 'working', prompt: 'copy session id', agentType: 'claude' }, + 'Claude', + undefined, + { tabId, worktreeId }, + { providerSession: { key: 'session_id', id: sessionId } } + ) + }, + { paneKey, tabId, worktreeId, sessionId: SESSION_ID } + ) + + await expect + .poll( + () => + orcaPage.evaluate( + ({ paneKey }) => + window.__store?.getState().agentStatusByPaneKey[paneKey]?.providerSession?.id, + { paneKey } + ), + { timeout: 3_000 } + ) + .toBe(SESSION_ID) + + const tab = orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${tabId}"]`) + await expect(tab).toBeVisible() + await tab.click({ button: 'right' }) + + const copyItem = orcaPage.getByRole('menuitem', { name: 'Copy Session ID', exact: true }) + await expect(copyItem).toBeVisible() + await copyItem.click() + + await expect + .poll(() => orcaPage.evaluate(() => window.api.ui.readClipboardText()), { timeout: 3_000 }) + .toBe(SESSION_ID) +}) From a7fda48fe3faa55b6248cd570165095be042e769 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Tue, 1 Sep 2026 22:33:39 -0400 Subject: [PATCH 010/398] feat(telemetry): measure macOS stale-daemon adoption and cwd denials (#18043) * feat(telemetry): measure macOS stale-daemon adoption and cwd denials Adds two enum-only PostHog events so #17696 can be sized instead of guessed at: - daemon_adopted: once per macOS launch that keeps a daemon an earlier app launch forked (invisible to daemon_lifecycle, which only sees replacements). Carries app-version match, spawner-path class (installed app / Squirrel ShipIt cache / other / missing), the existing TCC attribution verdict, and the bucketed live-session count. - daemon_pty_cwd_denied: the symptom itself. The daemon probes the requested cwd in its own process (only its TCC context counts) and returns an additive cwdReadableByDaemon field; the app emits only when the daemon was denied AND the app can read the same path, so a missing or genuinely unreadable cwd never counts. Non-permission errors read as readable on purpose. Both emitters swallow every failure; nothing here can delay or fail daemon startup or a PTY spawn. Off macOS neither event fires. The new wire field is optional, so older daemons and clients are unaffected. * fix(telemetry): keep cwd-denial classification inside the swallow guard Read the pid record at emit time (inside the try) rather than passing the adapter's startup snapshot: a throwing app-environment read can no longer escape spawn(), and a denial after a respawn is billed to the daemon that actually spawned the PTY. --- .../daemon-adoption-telemetry-event.test.ts | 168 ++++++++++++++++++ .../daemon/daemon-adoption-telemetry-event.ts | 81 +++++++++ .../daemon/daemon-create-or-attach-result.ts | 6 + .../daemon/daemon-init-dependency-mocks.ts | 9 +- src/main/daemon/daemon-init-fresh-import.ts | 4 +- src/main/daemon/daemon-init-mock-types.ts | 3 + .../daemon-init-provider-installation.test.ts | 45 +++++ src/main/daemon/daemon-init-test-harness.ts | 4 +- .../daemon/daemon-out-of-process-launcher.ts | 1 + src/main/daemon/daemon-provider-init.ts | 31 ++++ src/main/daemon/daemon-pty-session-spawn.ts | 4 + src/main/daemon/daemon-spawner.ts | 2 + src/main/daemon/daemon-terminal-admission.ts | 5 +- .../daemon/terminal-host-create-contract.ts | 2 + .../terminal-host-cwd-readability.test.ts | 75 ++++++++ .../daemon/terminal-host-session-create.ts | 17 ++ src/main/ipc/telemetry.ts | 2 + src/shared/daemon-adoption-telemetry.test.ts | 89 ++++++++++ src/shared/daemon-adoption-telemetry.ts | 67 +++++++ src/shared/telemetry-daemon-event-schemas.ts | 27 +++ src/shared/telemetry-event-registry.ts | 4 + 21 files changed, 642 insertions(+), 4 deletions(-) create mode 100644 src/main/daemon/daemon-adoption-telemetry-event.test.ts create mode 100644 src/main/daemon/daemon-adoption-telemetry-event.ts create mode 100644 src/main/daemon/terminal-host-cwd-readability.test.ts create mode 100644 src/shared/daemon-adoption-telemetry.test.ts create mode 100644 src/shared/daemon-adoption-telemetry.ts diff --git a/src/main/daemon/daemon-adoption-telemetry-event.test.ts b/src/main/daemon/daemon-adoption-telemetry-event.test.ts new file mode 100644 index 00000000000..5a5cd7406c4 --- /dev/null +++ b/src/main/daemon/daemon-adoption-telemetry-event.test.ts @@ -0,0 +1,168 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { ParsedDaemonPid } from './daemon-pid-file-parse' +import { validate } from '../telemetry/validator' + +const { trackMock, accessSyncMock, existsSyncMock, readFileSyncMock, getVersionMock } = vi.hoisted( + () => ({ + trackMock: vi.fn(), + accessSyncMock: vi.fn(), + existsSyncMock: vi.fn(() => true), + readFileSyncMock: vi.fn(), + getVersionMock: vi.fn(() => '1.4.191') + }) +) +vi.mock('../telemetry/client', () => ({ track: trackMock })) +vi.mock('node:fs', async (importOriginal) => ({ + ...(await importOriginal>()), + accessSync: accessSyncMock, + existsSync: existsSyncMock, + readFileSync: readFileSyncMock +})) +vi.mock('node:os', async (importOriginal) => ({ + ...(await importOriginal>()), + homedir: () => '/Users/alice' +})) +vi.mock('../../shared/app-environment', () => ({ + getAppEnvironment: () => ({ getVersion: getVersionMock }) +})) + +import { + classifyDaemonAdoptionOrigin, + trackDaemonAdopted, + trackDaemonPtyCwdDeniedIfDiverged +} from './daemon-adoption-telemetry-event' + +const stalePidRecord: ParsedDaemonPid = { + pid: 1530, + startedAtMs: 1, + entryPath: '/x/daemon-entry.js', + appVersion: '1.4.187', + launchNonce: 'n', + linuxStartTicks: null, + bootId: null, + spawnerExecPath: + '/Users/alice/Library/Caches/com.stablyai.orca.ShipIt/u/Orca.app/Contents/MacOS/Orca' +} +const origin = { app_version_match: 'different', spawner_path_class: 'updater-cache' } as const +const PID_PATH = '/fake/daemon.pid' + +beforeEach(() => { + trackMock.mockReset() + accessSyncMock.mockReset() + existsSyncMock.mockReset().mockReturnValue(true) + readFileSyncMock.mockReset().mockReturnValue(JSON.stringify(stalePidRecord)) + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') +}) + +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('classifyDaemonAdoptionOrigin', () => { + it('compares the recorded app version and classifies the spawner path', () => { + expect(classifyDaemonAdoptionOrigin(stalePidRecord)).toEqual(origin) + expect(classifyDaemonAdoptionOrigin({ ...stalePidRecord, appVersion: '1.4.191' })).toEqual({ + app_version_match: 'same', + spawner_path_class: 'updater-cache' + }) + expect(classifyDaemonAdoptionOrigin(null)).toEqual({ + app_version_match: 'unknown', + spawner_path_class: 'unknown' + }) + }) +}) + +describe('trackDaemonAdopted', () => { + it('emits a validator-accepted payload', () => { + trackDaemonAdopted(stalePidRecord, 'intact', 7) + expect(trackMock).toHaveBeenCalledTimes(1) + const [name, props] = trackMock.mock.calls[0] + expect(name).toBe('daemon_adopted') + expect(props).toEqual({ + ...origin, + tcc_attribution: 'intact', + live_session_count_bucket: '6+' + }) + expect(validate('daemon_adopted', props).ok).toBe(true) + }) + + it('swallows a throwing telemetry client', () => { + trackMock.mockImplementationOnce(() => { + throw new Error('posthog exploded') + }) + expect(() => trackDaemonAdopted(null, 'unknown', null)).not.toThrow() + }) +}) + +describe('trackDaemonPtyCwdDeniedIfDiverged', () => { + it('emits only when the daemon was denied and the app can read the same cwd', () => { + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', false, PID_PATH) + expect(accessSyncMock).toHaveBeenCalledWith('/Users/alice/Documents/repo', expect.any(Number)) + expect(trackMock).toHaveBeenCalledTimes(1) + const [name, props] = trackMock.mock.calls[0] + expect(name).toBe('daemon_pty_cwd_denied') + expect(props).toEqual({ cwd_class: 'documents', ...origin }) + expect(validate('daemon_pty_cwd_denied', props).ok).toBe(true) + }) + + // False positives would drown the signal this event exists to measure, so every + // non-divergent shape must stay silent. + it('stays silent when the daemon could read the cwd or did not report', () => { + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', true, PID_PATH) + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', undefined, PID_PATH) + trackDaemonPtyCwdDeniedIfDiverged(undefined, false, PID_PATH) + expect(accessSyncMock).not.toHaveBeenCalled() + expect(trackMock).not.toHaveBeenCalled() + }) + + it('stays silent when the app cannot read the cwd either (no divergence)', () => { + accessSyncMock.mockImplementation(() => { + throw Object.assign(new Error('EACCES'), { code: 'EACCES' }) + }) + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', false, PID_PATH) + expect(trackMock).not.toHaveBeenCalled() + }) + + it('attributes the denial to the daemon recorded right now, not a startup snapshot', () => { + readFileSyncMock.mockReturnValue( + JSON.stringify({ + ...stalePidRecord, + appVersion: '1.4.191', + spawnerExecPath: '/Applications/Orca.app/Contents/MacOS/Orca' + }) + ) + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', false, PID_PATH) + expect(readFileSyncMock).toHaveBeenCalledWith(PID_PATH, 'utf8') + expect(trackMock.mock.calls[0][1]).toEqual({ + cwd_class: 'documents', + app_version_match: 'same', + spawner_path_class: 'applications' + }) + }) + + it('swallows a throwing app environment or pid-record read instead of failing the spawn', () => { + getVersionMock.mockImplementationOnce(() => { + throw new Error('AppEnvironment not initialized') + }) + expect(() => + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', false, PID_PATH) + ).not.toThrow() + expect(trackMock).not.toHaveBeenCalled() + }) + + it('stays silent off macOS', () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('linux') + trackDaemonPtyCwdDeniedIfDiverged('/home/alice/Documents/repo', false, PID_PATH) + expect(accessSyncMock).not.toHaveBeenCalled() + expect(trackMock).not.toHaveBeenCalled() + }) + + it('swallows a throwing telemetry client', () => { + trackMock.mockImplementationOnce(() => { + throw new Error('posthog exploded') + }) + expect(() => + trackDaemonPtyCwdDeniedIfDiverged('/Users/alice/Documents/repo', false, PID_PATH) + ).not.toThrow() + }) +}) diff --git a/src/main/daemon/daemon-adoption-telemetry-event.ts b/src/main/daemon/daemon-adoption-telemetry-event.ts new file mode 100644 index 00000000000..47f554bf9bb --- /dev/null +++ b/src/main/daemon/daemon-adoption-telemetry-event.ts @@ -0,0 +1,81 @@ +// App-side emitters for `daemon_adopted` and `daemon_pty_cwd_denied` (#17696). Both sit on the +// daemon launch / PTY spawn path, so every failure dies here — telemetry can never cost a terminal. + +import { accessSync, constants as fsConstants, existsSync } from 'node:fs' +import { homedir } from 'node:os' +import { getAppEnvironment } from '../../shared/app-environment' +import { + classifyDaemonPtyCwd, + classifyDaemonSpawnerPath, + type DaemonAdoptedAppVersionMatch, + type DaemonSpawnerPathClass +} from '../../shared/daemon-adoption-telemetry' +import { bucketDaemonLiveSessionCount } from '../../shared/daemon-lifecycle-telemetry' +import type { EventProps } from '../../shared/telemetry-events' +import { track } from '../telemetry/client' +import { readDaemonPidRecord } from './daemon-endpoint-incarnation' +import type { ParsedDaemonPid } from './daemon-pid-file-parse' +import type { MacDaemonTccAttributionHealth } from './daemon-tcc-attribution' + +export type DaemonAdoptionOrigin = Pick< + EventProps<'daemon_pty_cwd_denied'>, + 'app_version_match' | 'spawner_path_class' +> + +/** Classifies the adopted daemon's pid record against the running app; enum-only by construction. */ +export function classifyDaemonAdoptionOrigin( + pidRecord: ParsedDaemonPid | null +): DaemonAdoptionOrigin { + const appVersionMatch: DaemonAdoptedAppVersionMatch = !pidRecord?.appVersion + ? 'unknown' + : pidRecord.appVersion === getAppEnvironment().getVersion() + ? 'same' + : 'different' + const spawnerPathClass: DaemonSpawnerPathClass = classifyDaemonSpawnerPath( + pidRecord?.spawnerExecPath ?? null, + existsSync + ) + return { app_version_match: appVersionMatch, spawner_path_class: spawnerPathClass } +} + +// Adopted a daemon that a previous app launch forked (macOS only; that is where attribution matters). +export function trackDaemonAdopted( + pidRecord: ParsedDaemonPid | null, + tccAttribution: MacDaemonTccAttributionHealth, + liveSessionCount: number | null +): void { + try { + track('daemon_adopted', { + ...classifyDaemonAdoptionOrigin(pidRecord), + tcc_attribution: tccAttribution, + live_session_count_bucket: bucketDaemonLiveSessionCount(liveSessionCount) + }) + } catch { + // Telemetry is best-effort; a dropped event must not fail daemon adoption. + } +} + +/** + * Emits only on proven divergence: the daemon reported the cwd unreadable AND this process can + * read it. A cwd neither can read (chmod, ENOENT, unmounted volume) is not the #17696 shape. + */ +export function trackDaemonPtyCwdDeniedIfDiverged( + cwd: string | undefined, + cwdReadableByDaemon: boolean | undefined, + pidPath: string | null +): void { + try { + if (process.platform !== 'darwin' || !cwd || cwdReadableByDaemon !== false) { + return + } + accessSync(cwd, fsConstants.R_OK | fsConstants.X_OK) + // Why read now, not the adapter's startup snapshot: a respawn swaps the daemon under a + // long-lived adapter, and the denial must be attributed to the daemon that just spawned. + track('daemon_pty_cwd_denied', { + cwd_class: classifyDaemonPtyCwd(cwd, homedir()), + ...classifyDaemonAdoptionOrigin(readDaemonPidRecord(pidPath)) + }) + } catch { + // Either the app cannot read it (no divergence) or telemetry failed; neither may reach the caller. + } +} diff --git a/src/main/daemon/daemon-create-or-attach-result.ts b/src/main/daemon/daemon-create-or-attach-result.ts index d6668342487..92a7e451a10 100644 --- a/src/main/daemon/daemon-create-or-attach-result.ts +++ b/src/main/daemon/daemon-create-or-attach-result.ts @@ -14,6 +14,12 @@ export type DaemonCreateOrAttachResult = { wslDistro?: string | null agentSessionEnsure?: AgentSessionClaimedSpawnResult incarnationId?: PtyIncarnationId + /** + * Whether the daemon process itself could read the requested cwd at spawn. Only the daemon's own + * verdict counts: macOS TCC scopes folder access per process tree, so the app's view of the same + * path proves nothing about the daemon's (#17696). Omitted by daemons predating this field. + */ + cwdReadableByDaemon?: boolean } export function getDaemonSessionResultMetadata(session: { diff --git a/src/main/daemon/daemon-init-dependency-mocks.ts b/src/main/daemon/daemon-init-dependency-mocks.ts index d920a13866e..d7217d4df63 100644 --- a/src/main/daemon/daemon-init-dependency-mocks.ts +++ b/src/main/daemon/daemon-init-dependency-mocks.ts @@ -50,7 +50,8 @@ export function createDaemonInitModuleFactories(state: DaemonInitMockState) { unbindLocalProviderListenersMock, rebindLocalProviderListenersMock, trackDaemonReplacedMock, - trackDaemonRetiredMock + trackDaemonRetiredMock, + trackDaemonAdoptedMock } = state // Why: both fakes are annotated with constructor types so the exported factories widen to @@ -82,6 +83,9 @@ export function createDaemonInitModuleFactories(state: DaemonInitMockState) { if (result.mode) { this.handle.mode = result.mode } + if (result.adopted) { + this.handle.adopted = true + } return { socketPath: result.socketPath, tokenPath: result.tokenPath @@ -199,6 +203,9 @@ export function createDaemonInitModuleFactories(state: DaemonInitMockState) { trackDaemonReplaced: trackDaemonReplacedMock, trackDaemonRetired: trackDaemonRetiredMock }), + daemonAdoptionTelemetryEvent: () => ({ + trackDaemonAdopted: trackDaemonAdoptedMock + }), daemonSpawner: () => ({ DaemonSpawner: MockDaemonSpawner, getDaemonSocketPath: (_dir: string, version?: number) => diff --git a/src/main/daemon/daemon-init-fresh-import.ts b/src/main/daemon/daemon-init-fresh-import.ts index e1cc0416337..39f3625c063 100644 --- a/src/main/daemon/daemon-init-fresh-import.ts +++ b/src/main/daemon/daemon-init-fresh-import.ts @@ -41,7 +41,8 @@ export async function importFreshDaemonInit(state: DaemonInitMockState) { unbindLocalProviderListenersMock, rebindLocalProviderListenersMock, trackDaemonReplacedMock, - trackDaemonRetiredMock + trackDaemonRetiredMock, + trackDaemonAdoptedMock } = state vi.resetModules() @@ -64,6 +65,7 @@ export async function importFreshDaemonInit(state: DaemonInitMockState) { rebindLocalProviderListenersMock.mockClear() trackDaemonReplacedMock.mockClear() trackDaemonRetiredMock.mockClear() + trackDaemonAdoptedMock.mockClear() checkDaemonHealthMock.mockClear() checkDaemonHealthMock.mockResolvedValue('healthy') healthCheckDaemonMock.mockClear() diff --git a/src/main/daemon/daemon-init-mock-types.ts b/src/main/daemon/daemon-init-mock-types.ts index 8c8b805740f..341fcd26df7 100644 --- a/src/main/daemon/daemon-init-mock-types.ts +++ b/src/main/daemon/daemon-init-mock-types.ts @@ -47,6 +47,7 @@ export type MockAdapterConstructor = new (opts: MockAdapter['options']) => MockA /** Handle the fake spawner hands back from ensureRunning/getHandle. */ export type MockSpawnerHandle = { mode?: 'degraded-new-pty-fallback' + adopted?: true releaseAdoptionLease?: () => void shutdown: () => Promise } @@ -95,6 +96,7 @@ export type EnsureRunningOverride = () => Promise<{ socketPath: string tokenPath: string mode?: 'degraded-new-pty-fallback' + adopted?: true }> /** Every stub daemon-init's suites share, plus the control knobs they mutate per test. */ @@ -143,6 +145,7 @@ export type DaemonInitMockState = { rebindLocalProviderListenersMock: Mock<(...args: unknown[]) => void> trackDaemonReplacedMock: Mock<(...args: unknown[]) => void> trackDaemonRetiredMock: Mock<(...args: unknown[]) => void> + trackDaemonAdoptedMock: Mock<(...args: unknown[]) => void> } /** net.connect stubs the suites install in beforeEach. */ diff --git a/src/main/daemon/daemon-init-provider-installation.test.ts b/src/main/daemon/daemon-init-provider-installation.test.ts index ceb7e9dfa69..423ee9ee34f 100644 --- a/src/main/daemon/daemon-init-provider-installation.test.ts +++ b/src/main/daemon/daemon-init-provider-installation.test.ts @@ -2,6 +2,8 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { isPackagedMock, + getMacDaemonTccAttributionHealthMock, + trackDaemonAdoptedMock, probeSocketExistsMock, readFileSyncMock, unlinkSyncMock, @@ -42,6 +44,7 @@ vi.mock('./daemon-process-start-time', () => moduleFactories.daemonProcessStartT vi.mock('./daemon-pid-file-parse', () => moduleFactories.daemonPidFileParse()) vi.mock('./client', () => moduleFactories.client()) vi.mock('./daemon-lifecycle-event', () => moduleFactories.daemonLifecycleEvent()) +vi.mock('./daemon-adoption-telemetry-event', () => moduleFactories.daemonAdoptionTelemetryEvent()) vi.mock('./daemon-spawner', () => moduleFactories.daemonSpawner()) vi.mock('./daemon-pty-adapter', () => moduleFactories.daemonPtyAdapter()) vi.mock('../ipc/pty', () => moduleFactories.ipcPty()) @@ -228,6 +231,48 @@ describe('daemon-init: runRestartDaemon (7-step sequence)', () => { expect(adapterInstances[1].disconnectOnly).toHaveBeenCalledOnce() }) + // #17696: adopting a daemon from an earlier app launch is invisible to daemon_lifecycle, so + // it gets its own event — macOS only, and only for adopted (not freshly forked) daemons. + it('reports a macOS daemon adoption with its TCC attribution and live session bucket', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + const mod = await importFresh() + ensureRunningOverrides.push(async () => ({ + socketPath: '/fake/adopted-socket', + tokenPath: '/fake/adopted-token', + adopted: true + })) + getMacDaemonTccAttributionHealthMock.mockResolvedValueOnce('severed') + defaultListSessionsSessions.push({ sessionId: 'wt-1@@a' }, { sessionId: 'wt-1@@b' }) + + await mod.initDaemonPtyProvider() + await vi.waitFor(() => expect(trackDaemonAdoptedMock).toHaveBeenCalledOnce()) + + // null pid record: the harness has no pid file, which the emitter classifies as 'unknown'. + expect(trackDaemonAdoptedMock).toHaveBeenCalledWith(null, 'severed', 2) + vi.restoreAllMocks() + }) + + it('does not report adoption for a freshly forked daemon or off macOS', async () => { + vi.spyOn(process, 'platform', 'get').mockReturnValue('darwin') + const mod = await importFresh() + await mod.initDaemonPtyProvider() + await new Promise((resolve) => setImmediate(resolve)) + expect(trackDaemonAdoptedMock).not.toHaveBeenCalled() + vi.restoreAllMocks() + + vi.spyOn(process, 'platform', 'get').mockReturnValue('linux') + const linuxMod = await importFresh() + ensureRunningOverrides.push(async () => ({ + socketPath: '/fake/adopted-socket', + tokenPath: '/fake/adopted-token', + adopted: true + })) + await linuxMod.initDaemonPtyProvider() + await new Promise((resolve) => setImmediate(resolve)) + expect(trackDaemonAdoptedMock).not.toHaveBeenCalled() + vi.restoreAllMocks() + }) + it('routes fresh PTYs to the local fallback when a preserved daemon cannot spawn new PTYs', async () => { const mod = await importFresh() ensureRunningOverrides.push(async () => ({ diff --git a/src/main/daemon/daemon-init-test-harness.ts b/src/main/daemon/daemon-init-test-harness.ts index 6c38720a40b..6060ed9b759 100644 --- a/src/main/daemon/daemon-init-test-harness.ts +++ b/src/main/daemon/daemon-init-test-harness.ts @@ -156,6 +156,7 @@ function createDaemonInitMockState(): DaemonInitMockState { const rebindLocalProviderListenersMock = vi.fn() const trackDaemonReplacedMock = vi.fn() const trackDaemonRetiredMock = vi.fn() + const trackDaemonAdoptedMock = vi.fn() return { getPathMock, @@ -197,7 +198,8 @@ function createDaemonInitMockState(): DaemonInitMockState { unbindLocalProviderListenersMock, rebindLocalProviderListenersMock, trackDaemonReplacedMock, - trackDaemonRetiredMock + trackDaemonRetiredMock, + trackDaemonAdoptedMock } } diff --git a/src/main/daemon/daemon-out-of-process-launcher.ts b/src/main/daemon/daemon-out-of-process-launcher.ts index b59535a249a..21ff31918ac 100644 --- a/src/main/daemon/daemon-out-of-process-launcher.ts +++ b/src/main/daemon/daemon-out-of-process-launcher.ts @@ -41,6 +41,7 @@ function createPreservedDaemonHandle( mode?: 'degraded-new-pty-fallback' ): DaemonProcessHandle { const handle: DaemonProcessHandle = { + adopted: true, shutdown: async () => { await cleanupDaemonForProtocol(runtimeDir, protocolVersion) } diff --git a/src/main/daemon/daemon-provider-init.ts b/src/main/daemon/daemon-provider-init.ts index 1881e276e97..fa794257bda 100644 --- a/src/main/daemon/daemon-provider-init.ts +++ b/src/main/daemon/daemon-provider-init.ts @@ -24,7 +24,10 @@ import { import type { DaemonProvider } from './daemon-provider-routing' import { installDaemonProvider } from './daemon-provider-state' import { DegradedDaemonPtyProvider } from './degraded-daemon-pty-provider' +import { trackDaemonAdopted } from './daemon-adoption-telemetry-event' +import { readDaemonPidRecord } from './daemon-endpoint-incarnation' import { trackDaemonRetired } from './daemon-lifecycle-event' +import { getMacDaemonTccAttributionHealth } from './daemon-tcc-attribution' import { DaemonPtyAdapter } from './daemon-pty-adapter' import type { DaemonRespawnReason } from './daemon-pty-runtime-state' import { DaemonPtyRouter } from './daemon-pty-router' @@ -156,9 +159,37 @@ export async function initDaemonPtyProvider( logDaemonMilestone('daemon-init-done', { legacyAdapters: legacyAdapters.length }) + if (process.platform === 'darwin' && newSpawner.getHandle()?.adopted) { + void reportDaemonAdoption(runtimeDir, info.socketPath, info.tokenPath, newAdapter) + } await reconcileSeededClaudeLivePtys(routedAdapter) } +// Why off the init path: this is measurement of an adopted daemon (#17696), and neither its probes nor their failure may delay or fail startup. +async function reportDaemonAdoption( + runtimeDir: string, + socketPath: string, + tokenPath: string, + adapter: DaemonPtyAdapter +): Promise { + try { + const [tccAttribution, liveSessionCount] = await Promise.all([ + getMacDaemonTccAttributionHealth(runtimeDir, socketPath, tokenPath), + adapter.listSessions().then( + (sessions) => sessions.length, + () => null + ) + ]) + trackDaemonAdopted( + readDaemonPidRecord(getDaemonPidPath(runtimeDir)), + tccAttribution, + liveSessionCount + ) + } catch { + // Best-effort measurement only. + } +} + // Why: release gate ids only for daemon-confirmed-dead sessions; keep seeds on listing failure since releasing early can rotate a live CLI's refresh token. async function reconcileSeededClaudeLivePtys(provider: DaemonProvider): Promise { if (!hasSeededUnconfirmedClaudePtys()) { diff --git a/src/main/daemon/daemon-pty-session-spawn.ts b/src/main/daemon/daemon-pty-session-spawn.ts index 62c073f403e..235b286b0bc 100644 --- a/src/main/daemon/daemon-pty-session-spawn.ts +++ b/src/main/daemon/daemon-pty-session-spawn.ts @@ -4,6 +4,7 @@ import type { HistoryRecoveryContext, PendingDaemonSpawnOperation } from './daemon-pty-runtime-state' +import { trackDaemonPtyCwdDeniedIfDiverged } from './daemon-adoption-telemetry-event' import { STABLE_PANE_ATTACH_ONLY_DAEMON_PROTOCOL_VERSION } from './daemon-protocol-version' import { TerminalKilledError } from './daemon-pty-lifecycle-errors' import { DaemonPtySpawnResult } from './daemon-pty-spawn-result' @@ -246,6 +247,9 @@ export abstract class DaemonPtySessionSpawn extends DaemonPtySpawnResult { } activeSpawnContext = context const result = await this.createOrAttachSpawn(context, context.historySeedSegments) + if (result.isNew && !attachOnly) { + trackDaemonPtyCwdDeniedIfDiverged(effectiveCwd, result.cwdReadableByDaemon, this.pidPath) + } return this.finishSpawn(context, result) } diff --git a/src/main/daemon/daemon-spawner.ts b/src/main/daemon/daemon-spawner.ts index a0376ef0fc0..8c50b764b05 100644 --- a/src/main/daemon/daemon-spawner.ts +++ b/src/main/daemon/daemon-spawner.ts @@ -31,6 +31,8 @@ export type DaemonPidFile = { export type DaemonProcessHandle = { mode?: 'degraded-new-pty-fallback' + /** Set when the launcher kept a daemon some earlier app launch forked, rather than forking one. */ + adopted?: true releaseAdoptionLease?(): void shutdown(): Promise } diff --git a/src/main/daemon/daemon-terminal-admission.ts b/src/main/daemon/daemon-terminal-admission.ts index b47b497fe43..83dadf5f5d2 100644 --- a/src/main/daemon/daemon-terminal-admission.ts +++ b/src/main/daemon/daemon-terminal-admission.ts @@ -161,7 +161,10 @@ export class DaemonTerminalAdmission { ...(result.launchAgent ? { launchAgent: result.launchAgent } : {}), wslDistro: result.wslDistro, ...(result.historySeeded !== undefined ? { historySeeded: result.historySeeded } : {}), - ...(result.agentSessionEnsure ? { agentSessionEnsure: result.agentSessionEnsure } : {}) + ...(result.agentSessionEnsure ? { agentSessionEnsure: result.agentSessionEnsure } : {}), + ...(result.cwdReadableByDaemon !== undefined + ? { cwdReadableByDaemon: result.cwdReadableByDaemon } + : {}) } } diff --git a/src/main/daemon/terminal-host-create-contract.ts b/src/main/daemon/terminal-host-create-contract.ts index aaaffaeb8e8..42f5bf457f4 100644 --- a/src/main/daemon/terminal-host-create-contract.ts +++ b/src/main/daemon/terminal-host-create-contract.ts @@ -54,4 +54,6 @@ export type CreateOrAttachResult = { attachToken: symbol incarnationId: PtyIncarnationId agentSessionEnsure?: AgentSessionClaimedSpawnResult + /** Daemon-process verdict on the spawn cwd; only set on a fresh spawn that was given a cwd. */ + cwdReadableByDaemon?: boolean } diff --git a/src/main/daemon/terminal-host-cwd-readability.test.ts b/src/main/daemon/terminal-host-cwd-readability.test.ts new file mode 100644 index 00000000000..aa9e08379d7 --- /dev/null +++ b/src/main/daemon/terminal-host-cwd-readability.test.ts @@ -0,0 +1,75 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { SubprocessHandle } from './session-subprocess-handle' +import { TerminalHost, type TerminalHostOptions } from './terminal-host' + +vi.mock('../pty-descendant-termination', () => ({ killWithDescendantSweep: vi.fn() })) + +function createMockSubprocess(): SubprocessHandle { + let onExitCb: ((code: number) => void) | null = null + return { + pid: 99999, + getForegroundProcess: vi.fn(() => null), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(() => { + setTimeout(() => onExitCb?.(0), 5) + }), + terminateOwnedTree: () => 'unavailable' as const, + forceKill: vi.fn(() => onExitCb?.(137)), + signal: vi.fn(), + onData() {}, + onExit(cb) { + onExitCb = cb + }, + dispose: vi.fn() + } +} + +// #17696: only the daemon process can say whether TCC lets it read the cwd, so its verdict +// rides on the create result. A non-permission failure must never read as denial. +describe('TerminalHost cwd readability verdict', () => { + let host: TerminalHost + let platformDescriptor: PropertyDescriptor | undefined + + beforeEach(() => { + platformDescriptor = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'linux' }) + const spawnSubprocess: TerminalHostOptions['spawnSubprocess'] = () => createMockSubprocess() + host = new TerminalHost({ spawnSubprocess }) + }) + + afterEach(async () => { + await host.dispose() + if (platformDescriptor) { + Object.defineProperty(process, 'platform', platformDescriptor) + } + }) + + const create = (sessionId: string, cwd?: string) => + host.createOrAttach({ + sessionId, + cols: 80, + rows: 24, + ...(cwd ? { cwd } : {}), + streamClient: { onData: vi.fn(), onExit: vi.fn() } + }) + + it('reports a readable cwd as readable', async () => { + expect((await create('readable', process.cwd())).cwdReadableByDaemon).toBe(true) + }) + + it('reports a missing cwd as readable — absence is not a permission denial', async () => { + expect((await create('missing', '/definitely/not/a/real/dir')).cwdReadableByDaemon).toBe(true) + }) + + it('omits the verdict when no cwd was requested', async () => { + expect((await create('no-cwd')).cwdReadableByDaemon).toBeUndefined() + }) + + it('omits the verdict on attach to an existing session', async () => { + await create('attach', process.cwd()) + const attached = await create('attach', process.cwd()) + expect(attached.isNew).toBe(false) + expect(attached.cwdReadableByDaemon).toBeUndefined() + }) +}) diff --git a/src/main/daemon/terminal-host-session-create.ts b/src/main/daemon/terminal-host-session-create.ts index e1c8a22e750..9ee51c9968d 100644 --- a/src/main/daemon/terminal-host-session-create.ts +++ b/src/main/daemon/terminal-host-session-create.ts @@ -1,3 +1,4 @@ +import { accessSync, constants as fsConstants } from 'node:fs' import { buildStartupCommandSubmission } from '../../shared/startup-command-submission' import { resolvePtyOwnerBackend } from '../../shared/pty-owner-backend' import { getDaemonSessionResultMetadata } from './daemon-create-or-attach-result' @@ -88,6 +89,8 @@ async function spawnAndPublishSession( ctx: { size: { cols: number; rows: number }; wslDistro: string | undefined } ): Promise { const { size, wslDistro } = ctx + // Why before the fork: the shell's own cwd may already have fallen back, so probe the requested path. + const cwdReadableByDaemon = opts.cwd && !wslDistro ? isCwdReadableByThisProcess(opts.cwd) : null const subprocess = await deps.spawnSubprocess({ sessionId: opts.sessionId, cols: size.cols, @@ -184,6 +187,20 @@ async function spawnAndPublishSession( shellState: session.shellState, incarnationId: session.incarnationId, ...getDaemonSessionResultMetadata(session), + ...(cwdReadableByDaemon !== null ? { cwdReadableByDaemon } : {}), attachToken: token } } + +// Why R_OK|X_OK: listing a directory needs read, and entering it needs search — both are what +// TCC withholds. A non-permission failure (ENOENT, ENOTDIR) reads as readable so it can never +// masquerade as a permission denial. +function isCwdReadableByThisProcess(cwd: string): boolean { + try { + accessSync(cwd, fsConstants.R_OK | fsConstants.X_OK) + return true + } catch (error) { + const code = (error as NodeJS.ErrnoException).code + return code !== 'EACCES' && code !== 'EPERM' + } +} diff --git a/src/main/ipc/telemetry.ts b/src/main/ipc/telemetry.ts index c1fe7d0f68d..5e649f51b84 100644 --- a/src/main/ipc/telemetry.ts +++ b/src/main/ipc/telemetry.ts @@ -24,7 +24,9 @@ let storeRef: Store | null = null const MAIN_OWNED_TELEMETRY_EVENTS = new Set([ 'app_starred_orca', + 'daemon_adopted', 'daemon_audit_eligibility', + 'daemon_pty_cwd_denied', 'star_nag_outcome', 'feature_interaction_usage_bucket_reached' ]) diff --git a/src/shared/daemon-adoption-telemetry.test.ts b/src/shared/daemon-adoption-telemetry.test.ts new file mode 100644 index 00000000000..f02aa431506 --- /dev/null +++ b/src/shared/daemon-adoption-telemetry.test.ts @@ -0,0 +1,89 @@ +import { describe, expect, it } from 'vitest' +import { classifyDaemonPtyCwd, classifyDaemonSpawnerPath } from './daemon-adoption-telemetry' +import { eventSchemas } from './telemetry-event-registry' + +describe('classifyDaemonSpawnerPath', () => { + const alwaysExists = () => true + + it('classifies the installed app, the ShipIt staging area, and everything else', () => { + expect( + classifyDaemonSpawnerPath('/Applications/Orca.app/Contents/MacOS/Orca', alwaysExists) + ).toBe('applications') + expect( + classifyDaemonSpawnerPath('/private/Applications/Orca.app/Contents/MacOS/Orca', alwaysExists) + ).toBe('applications') + expect( + classifyDaemonSpawnerPath( + '/Users/a/Library/Caches/com.stablyai.orca.ShipIt/update.abc/Orca.app/Contents/MacOS/Orca', + alwaysExists + ) + ).toBe('updater-cache') + expect( + classifyDaemonSpawnerPath('/Users/a/Applications/Orca.app/Contents/MacOS/Orca', alwaysExists) + ).toBe('other') + expect(classifyDaemonSpawnerPath('/tmp/OrcaA.app/Contents/MacOS/Orca', alwaysExists)).toBe( + 'other' + ) + }) + + it('reports a deleted spawner as missing and an unrecorded one as unknown', () => { + expect( + classifyDaemonSpawnerPath('/Applications/Orca.app/Contents/MacOS/Orca', () => false) + ).toBe('missing') + expect(classifyDaemonSpawnerPath(null, alwaysExists)).toBe('unknown') + }) +}) + +describe('classifyDaemonPtyCwd', () => { + it('maps the TCC-protected home folders and separates the rest of home from outside it', () => { + expect(classifyDaemonPtyCwd('/Users/a/Documents/repo', '/Users/a')).toBe('documents') + expect(classifyDaemonPtyCwd('/Users/a/Desktop', '/Users/a/')).toBe('desktop') + expect(classifyDaemonPtyCwd('/Users/a/Downloads/x/y', '/Users/a')).toBe('downloads') + expect(classifyDaemonPtyCwd('/Users/a/projects/repo', '/Users/a')).toBe('other-home') + expect(classifyDaemonPtyCwd('/Users/a', '/Users/a')).toBe('other-home') + expect(classifyDaemonPtyCwd('/Volumes/ext/repo', '/Users/a')).toBe('outside-home') + // A sibling home that merely shares the prefix is not inside this home. + expect(classifyDaemonPtyCwd('/Users/ab/Documents', '/Users/a')).toBe('outside-home') + }) +}) + +// Privacy invariant: enum-only. A raw path, version, or exact count must be rejected by .strict(). +describe('daemon_adopted / daemon_pty_cwd_denied schemas', () => { + const adopted = { + app_version_match: 'different', + spawner_path_class: 'updater-cache', + tcc_attribution: 'intact', + live_session_count_bucket: '2-5' + } + const denied = { + cwd_class: 'documents', + app_version_match: 'different', + spawner_path_class: 'updater-cache' + } + + it('accepts the enum payloads', () => { + expect(eventSchemas.daemon_adopted.safeParse(adopted).success).toBe(true) + expect(eventSchemas.daemon_pty_cwd_denied.safeParse(denied).success).toBe(true) + }) + + it('rejects leaked paths, versions, counts, and unknown enum values', () => { + for (const leak of [ + { spawner_exec_path: '/Users/alice/Library/Caches/ShipIt/Orca.app' }, + { app_version: '1.4.187' }, + { live_session_count: 3 }, + { cwd: '/Users/alice/Documents' } + ]) { + expect(eventSchemas.daemon_adopted.safeParse({ ...adopted, ...leak }).success).toBe(false) + expect(eventSchemas.daemon_pty_cwd_denied.safeParse({ ...denied, ...leak }).success).toBe( + false + ) + } + expect( + eventSchemas.daemon_adopted.safeParse({ ...adopted, spawner_path_class: '/Applications' }) + .success + ).toBe(false) + expect( + eventSchemas.daemon_pty_cwd_denied.safeParse({ ...denied, cwd_class: 'Documents' }).success + ).toBe(false) + }) +}) diff --git a/src/shared/daemon-adoption-telemetry.ts b/src/shared/daemon-adoption-telemetry.ts new file mode 100644 index 00000000000..72c21647cd3 --- /dev/null +++ b/src/shared/daemon-adoption-telemetry.ts @@ -0,0 +1,67 @@ +// Enums for the `daemon_adopted` and `daemon_pty_cwd_denied` telemetry events (#17696). +// Both exist to measure how often a macOS app runs on a daemon left behind by an earlier app +// bundle, and how often such a daemon actually spawns a terminal whose cwd it cannot read. +// Enum-only: no paths, versions, or exact counts ever reach the wire. + +/** How the adopted daemon's recorded app version compares to the running app. */ +export const DAEMON_ADOPTED_APP_VERSION_MATCH = ['same', 'different', 'unknown'] as const +export type DaemonAdoptedAppVersionMatch = (typeof DAEMON_ADOPTED_APP_VERSION_MATCH)[number] + +/** + * Where the binary that forked the adopted daemon lives now. `updater-cache` is the Squirrel + * ShipIt staging area — a daemon attributed there is the reported #17696 shape. + */ +export const DAEMON_SPAWNER_PATH_CLASSES = [ + 'applications', + 'updater-cache', + 'other', + 'missing', + 'unknown' +] as const +export type DaemonSpawnerPathClass = (typeof DAEMON_SPAWNER_PATH_CLASSES)[number] + +export const DAEMON_TCC_ATTRIBUTION_VALUES = ['intact', 'severed', 'unknown'] as const + +/** Which macOS-protected folder class the denied cwd falls under. */ +export const DAEMON_PTY_CWD_CLASSES = [ + 'documents', + 'desktop', + 'downloads', + 'other-home', + 'outside-home' +] as const +export type DaemonPtyCwdClass = (typeof DAEMON_PTY_CWD_CLASSES)[number] + +export function classifyDaemonSpawnerPath( + spawnerExecPath: string | null, + exists: (path: string) => boolean +): DaemonSpawnerPathClass { + if (!spawnerExecPath) { + return 'unknown' + } + if (!exists(spawnerExecPath)) { + return 'missing' + } + if (/\/Library\/Caches\/[^/]*ShipIt\//.test(spawnerExecPath)) { + return 'updater-cache' + } + return /^(?:\/private)?\/Applications\//.test(spawnerExecPath) ? 'applications' : 'other' +} + +export function classifyDaemonPtyCwd(cwd: string, homeDir: string): DaemonPtyCwdClass { + const home = homeDir.replace(/\/+$/, '') + if (!home || !(cwd === home || cwd.startsWith(`${home}/`))) { + return 'outside-home' + } + const topLevel = cwd.slice(home.length + 1).split('/')[0] + switch (topLevel) { + case 'Documents': + return 'documents' + case 'Desktop': + return 'desktop' + case 'Downloads': + return 'downloads' + default: + return 'other-home' + } +} diff --git a/src/shared/telemetry-daemon-event-schemas.ts b/src/shared/telemetry-daemon-event-schemas.ts index a0543d4d49b..c6b2795a333 100644 --- a/src/shared/telemetry-daemon-event-schemas.ts +++ b/src/shared/telemetry-daemon-event-schemas.ts @@ -14,6 +14,12 @@ import { DAEMON_AUDIT_TRIGGER_VALUES, DAEMON_EVIDENCE_SOURCE_VALUES } from './daemon-audit-eligibility' +import { + DAEMON_ADOPTED_APP_VERSION_MATCH, + DAEMON_PTY_CWD_CLASSES, + DAEMON_SPAWNER_PATH_CLASSES, + DAEMON_TCC_ATTRIBUTION_VALUES +} from './daemon-adoption-telemetry' import { errorClassSchema, settingsChangedKeySchema } from './telemetry-property-schemas' // Why: daemon start-failure signal (fleet-wide outage like v1.4.129-rc.1); enum-only so raw stderr never reaches the wire. @@ -50,6 +56,27 @@ export const mainThreadHangDetectedSchema = z }) .strict() +// Why: #17696 — a macOS app adopting a daemon from an earlier bundle is invisible to +// `daemon_lifecycle` (nothing is replaced). Once per macOS launch that adopts; enum-only. +export const daemonAdoptedSchema = z + .object({ + app_version_match: z.enum(DAEMON_ADOPTED_APP_VERSION_MATCH), + spawner_path_class: z.enum(DAEMON_SPAWNER_PATH_CLASSES), + tcc_attribution: z.enum(DAEMON_TCC_ATTRIBUTION_VALUES), + live_session_count_bucket: z.enum(DAEMON_LIFECYCLE_SESSION_BUCKETS) + }) + .strict() + +// Why: the #17696 symptom itself — the daemon spawned a terminal into a cwd it cannot read while +// the app can. Emitted only on that proven divergence, so a missing or app-unreadable cwd never counts. +export const daemonPtyCwdDeniedSchema = z + .object({ + cwd_class: z.enum(DAEMON_PTY_CWD_CLASSES), + app_version_match: z.enum(DAEMON_ADOPTED_APP_VERSION_MATCH), + spawner_path_class: z.enum(DAEMON_SPAWNER_PATH_CLASSES) + }) + .strict() + // Why: daemon replace/retire lifecycle signal — issue #7936 was undiagnosable without asking a user for daemon.log. // Enum-only + bucketed session count so no paths, raw versions, or exact counts reach the wire. // The union keeps each reason pinned to its transition, so a death can't be reported as a replace. diff --git a/src/shared/telemetry-event-registry.ts b/src/shared/telemetry-event-registry.ts index 29479f363cb..5a91640359a 100644 --- a/src/shared/telemetry-event-registry.ts +++ b/src/shared/telemetry-event-registry.ts @@ -14,8 +14,10 @@ import { agentHookTransportBlockedSchema, agentHookUnattributedSchema, codexTrustGrantSchema, + daemonAdoptedSchema, daemonAuditEligibilitySchema, daemonLifecycleSchema, + daemonPtyCwdDeniedSchema, daemonStartFailedSchema, mainThreadHangDetectedSchema, remoteOutboundBudgetCloseSchema, @@ -122,6 +124,8 @@ export const eventSchemas = { daemon_start_failed: daemonStartFailedSchema, main_thread_hang_detected: mainThreadHangDetectedSchema, daemon_lifecycle: daemonLifecycleSchema, + daemon_adopted: daemonAdoptedSchema, + daemon_pty_cwd_denied: daemonPtyCwdDeniedSchema, daemon_audit_eligibility: daemonAuditEligibilitySchema, runtime_rpc_start_failed: runtimeRpcStartFailedSchema, remote_outbound_budget_close: remoteOutboundBudgetCloseSchema, From 7f6cf271ceedb0030c280eae4db4458bdd892c28 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Tue, 1 Sep 2026 19:53:11 -0700 Subject: [PATCH 011/398] fix(terminal): preserve panes when restored PTY owner is unverifiable (#17860) * fix(terminal): preserve unverifiable restored pane bindings * test(terminal): cover unverifiable restored pane identity * fix(terminal): settle direct SSH retry on unverifiable owner * fix(terminal): make owner warning actionable * fix(terminal): harden owner warning recovery feedback * test(terminal): consolidate fixture imports --------- Co-authored-by: Merge Sim --- .../terminal-pane/TerminalErrorToast.test.ts | 84 ++++++++++++++++- .../terminal-pane/TerminalErrorToast.tsx | 84 ++++++++++++++--- .../terminal-pane/TerminalPaneSurface.tsx | 22 ++++- ...-connection-direct-ssh-spawn-retry.test.ts | 92 ++++++++++++++++++- .../pty-connection-session-liveness.test.ts | 65 +++++++++++++ .../deferred-session-reattach-connect.ts | 20 ++++ src/renderer/src/i18n/locales/en.json | 6 +- 7 files changed, 354 insertions(+), 19 deletions(-) diff --git a/src/renderer/src/components/terminal-pane/TerminalErrorToast.test.ts b/src/renderer/src/components/terminal-pane/TerminalErrorToast.test.ts index 069ed371ec0..220f968cd98 100644 --- a/src/renderer/src/components/terminal-pane/TerminalErrorToast.test.ts +++ b/src/renderer/src/components/terminal-pane/TerminalErrorToast.test.ts @@ -1,7 +1,7 @@ // @vitest-environment happy-dom import React from 'react' -import { cleanup, render, waitFor } from '@testing-library/react' +import { cleanup, fireEvent, render, waitFor } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const environmentMocks = vi.hoisted(() => ({ @@ -15,6 +15,7 @@ vi.mock('@/lib/client-environment-info', () => ({ import { TerminalErrorToast, humanizeTerminalError, + isPaneOwnerUnverifiedError, isExplainedTerminalError, isSshReconnectOwnedTerminalError, shouldOfferDaemonRestart, @@ -69,7 +70,15 @@ describe('humanizeTerminalError', () => { it('replaces the pane-owner-unverified code with actionable copy', () => { const humanized = humanizeTerminalError('terminal_pane_owner_unverified') expect(humanized).not.toContain('terminal_pane_owner_unverified') - expect(humanized).toContain('Reopen this pane to retry') + expect(humanized).toContain('Click Retry to try reconnecting now') + expect(humanized).toContain('Orca left the saved session unchanged') + expect(humanized).not.toContain('was not closed or deleted') + }) + + it('identifies the owner-unverified safety state', () => { + expect(isPaneOwnerUnverifiedError('terminal_pane_owner_unverified')).toBe(true) + expect(isPaneOwnerUnverifiedError('Paste failed.')).toBe(false) + expect(isPaneOwnerUnverifiedError('Paste failed.\nterminal_pane_owner_unverified')).toBe(false) }) it('humanizes an IPC-wrapped pane-owner-unverified error', () => { @@ -78,6 +87,22 @@ describe('humanizeTerminalError', () => { expect(humanizeTerminalError(wrapped)).not.toContain('terminal_pane_owner_unverified') }) + it('humanizes an owner marker without classifying mixed errors as safe warnings', () => { + const mixed = humanizeTerminalError('Paste failed.\nterminal_pane_owner_unverified') + expect(mixed).toContain('Paste failed.') + expect(mixed).toContain("Orca couldn't verify this terminal's owner.") + expect(mixed).not.toContain('terminal_pane_owner_unverified') + expect(isPaneOwnerUnverifiedError('Paste failed.\nterminal_pane_owner_unverified')).toBe(false) + }) + + it('humanizes every owner marker in an aggregated warning', () => { + const repeated = humanizeTerminalError( + "terminal_pane_owner_unverified\nError invoking remote method 'pty:spawn': Error: terminal_pane_owner_unverified" + ) + + expect(repeated).not.toContain('terminal_pane_owner_unverified') + }) + it('leaves other errors untouched', () => { expect(humanizeTerminalError('Paste failed.')).toBe('Paste failed.') }) @@ -302,4 +327,59 @@ describe('TerminalErrorToast environment footer', () => { await waitFor(() => expect(environmentMocks.resolveFooter).not.toHaveBeenCalled()) }) + + it('renders owner-unverified as a warning without an issue link', () => { + const onRetry = vi.fn().mockResolvedValue(true) + const view = render( + React.createElement(TerminalErrorToast, { + error: 'terminal_pane_owner_unverified', + onDismiss: vi.fn(), + onRetry + }) + ) + + const toast = view.container.querySelector('[data-terminal-error-toast]') + expect(toast?.getAttribute('data-terminal-error-kind')).toBe('owner-unverified') + expect(toast?.querySelector('a')).toBeNull() + expect(toast?.textContent).toContain('Orca left the saved session unchanged') + expect(view.getByRole('button', { name: 'Retry' }).getAttribute('data-slot')).toBe('button') + fireEvent.click(view.getByRole('button', { name: 'Retry' })) + expect(onRetry).toHaveBeenCalledTimes(1) + }) + + it('keeps Retry available when the recovery attempt rejects', async () => { + const onRetry = vi.fn().mockRejectedValue(new Error('recovery unavailable')) + const view = render( + React.createElement(TerminalErrorToast, { + error: 'terminal_pane_owner_unverified', + onDismiss: vi.fn(), + onRetry + }) + ) + + fireEvent.click(view.getByRole('button', { name: 'Retry' })) + + await waitFor(() => + expect((view.getByRole('button', { name: 'Retry' }) as HTMLButtonElement).disabled).toBe( + false + ) + ) + expect(onRetry).toHaveBeenCalledTimes(1) + }) + + it('explains when Retry is temporarily unavailable', async () => { + const onRetry = vi.fn().mockResolvedValue(false) + const view = render( + React.createElement(TerminalErrorToast, { + error: 'terminal_pane_owner_unverified', + onDismiss: vi.fn(), + onRetry + }) + ) + + fireEvent.click(view.getByRole('button', { name: 'Retry' })) + await waitFor(() => + expect(view.container.textContent).toContain('Retry could not reconnect yet') + ) + }) }) diff --git a/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx b/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx index ee876ae719e..2ba98a180d6 100644 --- a/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx @@ -1,6 +1,7 @@ import { useEffect, useState } from 'react' import { translate } from '@/i18n/i18n' import { resolveClientEnvironmentFooter } from '@/lib/client-environment-info' +import { Button } from '@/components/ui/button' import { hasClientEnvironmentFooter } from '../../../../shared/client-environment-info' const SSH_PREFIX = 'SSH connection is not active' @@ -78,6 +79,11 @@ export function isExplainedTerminalError(error: string): boolean { ) } +export function isPaneOwnerUnverifiedError(error: string): boolean { + const lines = error.split('\n').filter((line) => line.length > 0) + return lines.length > 0 && lines.every((line) => line.includes(PANE_OWNER_UNVERIFIED_MARKER)) +} + function humanizeUnreattachableSession(error: string): string { const explanation = translate( 'auto.components.terminal.pane.TerminalErrorToast.sessionUnavailable', @@ -94,13 +100,16 @@ function humanizeUnreattachableSession(error: string): string { export function humanizeTerminalError(error: string): string { let humanized = error if (humanized.includes(PANE_OWNER_UNVERIFIED_MARKER)) { - humanized = humanized.replace( - PANE_OWNER_UNVERIFIED_MARKER, - translate( - 'auto.components.terminal.pane.TerminalErrorToast.7ee11bc0db', - "Orca couldn't confirm whether this terminal's previous session is still running, so it left the session untouched. Reopen this pane to retry." - ) - ) + const explanation = isPaneOwnerUnverifiedError(humanized) + ? translate( + 'auto.components.terminal.pane.TerminalErrorToast.42b283ecfc', + "Orca couldn't safely reconnect this terminal because the host couldn't verify its saved session. Orca left the saved session unchanged. Click Retry to try reconnecting now. If it still cannot reconnect, open a new terminal." + ) + : translate( + 'auto.components.terminal.pane.TerminalErrorToast.ownerUnknown', + "Orca couldn't verify this terminal's owner." + ) + humanized = humanized.replaceAll(PANE_OWNER_UNVERIFIED_MARKER, () => explanation) } humanized = humanizeUnreattachableSession(humanized) if (!isExplainedTerminalError(humanized)) { @@ -127,17 +136,23 @@ export function humanizeTerminalError(error: string): string { export function TerminalErrorToast({ error, onDismiss, - onRestartDaemon + onRestartDaemon, + onRetry }: { error: string onDismiss: () => void onRestartDaemon?: () => void + onRetry?: () => Promise }): React.JSX.Element { const ssh = isSshError(error) + const paneOwnerUnverified = isPaneOwnerUnverifiedError(error) const showDaemonRestart = !ssh && onRestartDaemon && shouldOfferDaemonRestart(error) // Restart cannot recover a session after its owning daemon exits. - const showIssueLink = !ssh && !showDaemonRestart && !isExplainedTerminalError(error) + const showIssueLink = + !ssh && !paneOwnerUnverified && !showDaemonRestart && !isExplainedTerminalError(error) const displayError = humanizeTerminalError(error) + const [retrying, setRetrying] = useState(false) + const [retryFailed, setRetryFailed] = useState(false) const [environmentFooter, setEnvironmentFooter] = useState<{ error: string footer: string @@ -160,10 +175,26 @@ export function TerminalErrorToast({ }, [displayError, ssh]) const footer = environmentFooter?.error === displayError ? environmentFooter.footer : '' + const handleRetry = async (): Promise => { + if (!onRetry || retrying) { + return + } + setRetrying(true) + setRetryFailed(false) + try { + setRetryFailed(!(await onRetry())) + } catch { + // Keep the safety warning available when a best-effort remount cannot start. + setRetryFailed(true) + } finally { + setRetrying(false) + } + } return (
) : null} {!ssh && footer ? `\n\n${footer}` : null} + {paneOwnerUnverified && retryFailed + ? `\n${translate( + 'auto.components.terminal.pane.TerminalErrorToast.retryUnavailable', + 'Retry could not reconnect yet. Try again shortly.' + )}` + : null} {showDaemonRestart ? ( + ) : null} +
) diff --git a/src/renderer/src/components/editor/combined-diff/review-controls/use-combined-diff-view-preferences.ts b/src/renderer/src/components/editor/combined-diff/review-controls/use-combined-diff-view-preferences.ts index 1ff70d76026..b32c88d7806 100644 --- a/src/renderer/src/components/editor/combined-diff/review-controls/use-combined-diff-view-preferences.ts +++ b/src/renderer/src/components/editor/combined-diff/review-controls/use-combined-diff-view-preferences.ts @@ -11,6 +11,7 @@ export type CombinedDiffViewPreferences = { setFileTreeCollapsed: (collapsed: boolean) => void setSideBySide: React.Dispatch> sideBySide: boolean + toggleDiffShowWhitespace: () => void toggleDiffWordWrap: () => void toggleSideBySide: () => void } @@ -18,6 +19,7 @@ export type CombinedDiffViewPreferences = { export function useCombinedDiffViewPreferences({ combinedDiffFileTreeVisibleByDefault, diffDefaultView, + diffShowWhitespace, diffWordWrap, registry, setSections, @@ -25,10 +27,11 @@ export function useCombinedDiffViewPreferences({ }: { combinedDiffFileTreeVisibleByDefault: boolean | undefined diffDefaultView: string | undefined + diffShowWhitespace: boolean | undefined diffWordWrap: boolean | undefined registry: CombinedDiffSectionLoadRegistry setSections: React.Dispatch> - updateSettings: (patch: { diffWordWrap: boolean }) => unknown + updateSettings: (patch: { diffShowWhitespace?: boolean; diffWordWrap?: boolean }) => unknown }): CombinedDiffViewPreferences { const { loadSchedulerRef, loadedIndicesRef, sectionsRef } = registry const [sideBySide, setSideBySide] = useState( @@ -89,12 +92,17 @@ export function useCombinedDiffViewPreferences({ void updateSettings({ diffWordWrap: diffWordWrap !== true }) }, [diffWordWrap, updateSettings]) + const toggleDiffShowWhitespace = useCallback(() => { + void updateSettings({ diffShowWhitespace: diffShowWhitespace !== true }) + }, [diffShowWhitespace, updateSettings]) + return { fileTreeCollapsed, setAllSectionsCollapsed, setFileTreeCollapsed, setSideBySide, sideBySide, + toggleDiffShowWhitespace, toggleDiffWordWrap, toggleSideBySide } diff --git a/src/renderer/src/components/editor/diff-editor-whitespace-options.test.ts b/src/renderer/src/components/editor/diff-editor-whitespace-options.test.ts new file mode 100644 index 00000000000..793914d4799 --- /dev/null +++ b/src/renderer/src/components/editor/diff-editor-whitespace-options.test.ts @@ -0,0 +1,13 @@ +import { describe, expect, it } from 'vitest' +import { buildDiffEditorWhitespaceOptions } from './diff-editor-whitespace-options' + +describe('buildDiffEditorWhitespaceOptions', () => { + it('ignores trim whitespace by default', () => { + expect(buildDiffEditorWhitespaceOptions(undefined)).toEqual({ ignoreTrimWhitespace: true }) + expect(buildDiffEditorWhitespaceOptions(false)).toEqual({ ignoreTrimWhitespace: true }) + }) + + it('includes whitespace in the diff when the preference is on', () => { + expect(buildDiffEditorWhitespaceOptions(true)).toEqual({ ignoreTrimWhitespace: false }) + }) +}) diff --git a/src/renderer/src/components/editor/diff-editor-whitespace-options.ts b/src/renderer/src/components/editor/diff-editor-whitespace-options.ts new file mode 100644 index 00000000000..4c860fadeb7 --- /dev/null +++ b/src/renderer/src/components/editor/diff-editor-whitespace-options.ts @@ -0,0 +1,10 @@ +import type { editor } from 'monaco-editor' + +export function buildDiffEditorWhitespaceOptions( + diffShowWhitespace: boolean | undefined +): Pick { + return { + // Why: Monaco defaults this to true, which hides indentation-only diffs. + ignoreTrimWhitespace: diffShowWhitespace !== true + } +} diff --git a/src/renderer/src/components/editor/diff-section-item-props.ts b/src/renderer/src/components/editor/diff-section-item-props.ts index 29d327d100a..e5ef73860c2 100644 --- a/src/renderer/src/components/editor/diff-section-item-props.ts +++ b/src/renderer/src/components/editor/diff-section-item-props.ts @@ -13,6 +13,7 @@ export type DiffSectionItemProps = { terminalFontSize?: number terminalFontFamily?: string diffWordWrap?: boolean + diffShowWhitespace?: boolean } | null sectionHeight: number | undefined worktreeId?: string diff --git a/src/renderer/src/components/settings/DiffShowWhitespaceSetting.test.tsx b/src/renderer/src/components/settings/DiffShowWhitespaceSetting.test.tsx new file mode 100644 index 00000000000..294d627fc20 --- /dev/null +++ b/src/renderer/src/components/settings/DiffShowWhitespaceSetting.test.tsx @@ -0,0 +1,73 @@ +// @vitest-environment happy-dom + +import { join } from 'node:path' +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { getDefaultSettings } from '../../../../shared/constants' + +vi.mock('../../store', () => ({ + useAppStore: (selector: (state: { settingsSearchQuery: string }) => unknown) => + selector({ settingsSearchQuery: '' }) +})) + +import { DiffShowWhitespaceSetting } from './DiffShowWhitespaceSetting' + +let root: Root | null = null +let container: HTMLDivElement | null = null + +afterEach(() => { + if (root) { + act(() => root?.unmount()) + } + container?.remove() + root = null + container = null +}) + +function renderSetting(diffShowWhitespace: boolean, updateSettings = vi.fn()) { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + act(() => { + root?.render( + + ) + }) + return { container, updateSettings } +} + +describe('DiffShowWhitespaceSetting', () => { + it('defaults to off so whitespace-only diffs stay quiet', () => { + const { container } = renderSetting(false) + const off = [...container.querySelectorAll('[role="radio"]')].find( + (button) => button.textContent === 'Off' + ) + + expect(off?.getAttribute('aria-checked')).toBe('true') + }) + + it('shows on when the preference is enabled', () => { + const { container } = renderSetting(true) + const on = [...container.querySelectorAll('[role="radio"]')].find( + (button) => button.textContent === 'On' + ) + + expect(on?.getAttribute('aria-checked')).toBe('true') + }) + + it('persists the on choice', () => { + const updateSettings = vi.fn() + const { container } = renderSetting(false, updateSettings) + const on = [...container.querySelectorAll('[role="radio"]')].find( + (button) => button.textContent === 'On' + ) + + act(() => on?.click()) + + expect(updateSettings).toHaveBeenCalledWith({ diffShowWhitespace: true }) + }) +}) diff --git a/src/renderer/src/components/settings/DiffShowWhitespaceSetting.tsx b/src/renderer/src/components/settings/DiffShowWhitespaceSetting.tsx new file mode 100644 index 00000000000..998274b84a7 --- /dev/null +++ b/src/renderer/src/components/settings/DiffShowWhitespaceSetting.tsx @@ -0,0 +1,69 @@ +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { translate } from '@/i18n/i18n' +import { SearchableSetting } from './SearchableSetting' +import { Label } from '../ui/label' +import { SettingsSegmentedControl } from './SettingsFormControls' + +type DiffShowWhitespaceSettingProps = { + settings: GlobalSettings + updateSettings: (updates: Partial) => void +} + +export function DiffShowWhitespaceSetting({ + settings, + updateSettings +}: DiffShowWhitespaceSettingProps): React.JSX.Element { + return ( + +
+ +

+ {translate( + 'auto.components.settings.GeneralEditorSettingsSection.94a479cef3', + 'Show leading and trailing whitespace differences in diffs.' + )} +

+
+ updateSettings({ diffShowWhitespace: option === 'on' })} + options={[ + { + value: 'off', + label: translate( + 'auto.components.settings.GeneralEditorSettingsSection.bf16ef0af2', + 'Off' + ) + }, + { + value: 'on', + label: translate( + 'auto.components.settings.GeneralEditorSettingsSection.3f6892f307', + 'On' + ) + } + ]} + /> +
+ ) +} diff --git a/src/renderer/src/components/settings/GeneralEditorSettingsSection.tsx b/src/renderer/src/components/settings/GeneralEditorSettingsSection.tsx index 6b3b211f628..f4221c2823d 100644 --- a/src/renderer/src/components/settings/GeneralEditorSettingsSection.tsx +++ b/src/renderer/src/components/settings/GeneralEditorSettingsSection.tsx @@ -17,6 +17,7 @@ import { } from './SettingsFormControls' import { translate } from '@/i18n/i18n' import { RichMarkdownSpellcheckSetting } from './RichMarkdownSpellcheckSetting' +import { DiffShowWhitespaceSetting } from './DiffShowWhitespaceSetting' import { EditorWordWrapSetting } from './EditorWordWrapSetting' import { EditorFontFamilySetting } from './EditorFontFamilySetting' import { @@ -232,6 +233,8 @@ export function GeneralEditorSettingsSection({ + + Date: Wed, 2 Sep 2026 01:29:26 -0700 Subject: [PATCH 054/398] fix(relay): fail an over-budget RPC response, not the connection (#17968) The relay's control lane is a shared 1 MiB budget, and `sendResponse` admitted responses onto it with the fatal default: once the lane was full, admission closed the client. A ~900 KB `fs.listFiles` reply from a large remote workspace therefore took down the whole remote session -- every terminal on it -- rather than failing the one Quick Open request. The substitute `ResponseOverCapacity` frame already there only covered the `legacy-response` lane, because the fatal close beat it to the client. A JSON-RPC response is the droppable class of control frame: it carries an id, so one caller can be told and can retry. `pty.replay` and `notifyControl` keep the fatal default -- they are never re-sent, and a silent drop there desyncs the client with nothing to retry. Both response enqueues now pass `controlOverflow: 'reject'`, so the substitute error is what the caller sees; in the corner where even ~150 bytes will not fit, the caller's own 30s request timeout settles it and the session survives. Old clients are unaffected: they already decode this error code and message generically (`ssh-channel-multiplexer.handleResponse` rejects the pending promise with both), and the frame shape is unchanged. What changes is that a listing which used to drop the connection now returns an error on it. --- .../dispatcher-capacity-degradation.test.ts | 96 +++++++++++++++++++ src/relay/dispatcher-rpc-routing.ts | 14 ++- src/relay/fs-handler-file-range.test.ts | 8 +- src/shared/file-range-read.ts | 9 +- 4 files changed, 116 insertions(+), 11 deletions(-) diff --git a/src/relay/dispatcher-capacity-degradation.test.ts b/src/relay/dispatcher-capacity-degradation.test.ts index 7bcf81674dc..2c746168ecc 100644 --- a/src/relay/dispatcher-capacity-degradation.test.ts +++ b/src/relay/dispatcher-capacity-degradation.test.ts @@ -59,6 +59,14 @@ function makeBoundedClient(highWaterMark: number): BoundedClient { return client } +// Why: the sink accepts every write but never settles it, so control-lane bytes stay retained and the +// queue fills, while the writer keeps pumping the other lanes — the shape of a peer whose socket is behind. +function makeUnsettledWriteClient(highWaterMark: number): BoundedClient { + const client = makeBoundedClient(highWaterMark) + client.options = { ...client.options, supportsWriteCallback: true } + return client +} + function decodePayload(frame: Buffer): Record { const length = frame.readUInt32BE(9) return JSON.parse(frame.subarray(13, 13 + length).toString('utf-8')) @@ -464,4 +472,92 @@ describe('RelayDispatcher bounded-capacity degradation', () => { bounded.dispose() } }) + + it('answers an over-budget response with a capacity error instead of closing the connection', async () => { + const primary = makeUnsettledWriteClient(65536) + const bounded = new RelayDispatcher(primary.write, primary.options) + try { + const clientId = bounded.activeClientIds()[0] + bounded.onRequest('fs.listFiles', async () => ({ paths: 'x'.repeat(700 * 1024) })) + bounded.onRequest('workspace.get', async () => ({ name: 'workspace' })) + + bounded.feed(encodeJsonRpcFrame({ jsonrpc: '2.0', id: 91, method: 'fs.listFiles' }, 1, 0)) + await vi.advanceTimersByTimeAsync(0) + expect(primary.frames).toHaveLength(1) + + // The first reply still holds the shared control budget, so the second cannot fit under 1 MiB. + bounded.feed(encodeJsonRpcFrame({ jsonrpc: '2.0', id: 92, method: 'fs.listFiles' }, 2, 0)) + await vi.advanceTimersByTimeAsync(0) + + expect(primary.closes).toBe(0) + expect(bounded.isClientAttached(clientId)).toBe(true) + expect(primary.frames).toHaveLength(2) + const rejected = decodePayload(primary.frames[1]) as unknown as { + id: number + error: { code: number; message: string } + } + expect(rejected.id).toBe(92) + expect(rejected.error.code).toBe(RelayErrorCode.ResponseOverCapacity) + expect(rejected.error.message).toBe('Relay response exceeded the bounded transport capacity') + + // Every other pane and request on this connection keeps working. + bounded.feed(encodeJsonRpcFrame({ jsonrpc: '2.0', id: 93, method: 'workspace.get' }, 3, 0)) + await vi.advanceTimersByTimeAsync(0) + expect(decodePayload(primary.frames[2])).toMatchObject({ + id: 93, + result: { name: 'workspace' } + }) + + bounded.notify('pty.data', { paneId: 'pane-1', data: 'still-live' }) + expect(decodePayload(primary.frames[3])).toMatchObject({ method: 'pty.data' }) + expect(primary.closes).toBe(0) + } finally { + bounded.dispose() + } + }) + + it('still closes the client when a protocol-critical control frame overflows', () => { + const primary = makeUnsettledWriteClient(65536) + const bounded = new RelayDispatcher(primary.write, primary.options) + try { + const clientId = bounded.activeClientIds()[0] + bounded.notifyClient(clientId, 'workspace.stale', { blob: 'x'.repeat(700 * 1024) }) + expect(primary.closes).toBe(0) + + // Replay is never re-sent, so an unnoticed drop strands the pane: overflow here stays fatal. + bounded.notify('pty.replay', { paneKey: 'tab-1:pane-1', data: 'y'.repeat(700 * 1024) }) + expect(primary.closes).toBe(1) + } finally { + bounded.dispose() + } + }) + + it('drops an unsendable response without closing when even the capacity error will not fit', async () => { + const primary = makeUnsettledWriteClient(65536) + const bounded = new RelayDispatcher(primary.write, primary.options) + try { + const clientId = bounded.activeClientIds()[0] + const settlements: SinkWriteSettlement[] = [] + bounded.onRequest('workspace.get', async (_params, context) => { + context.onResponseSettled?.((result) => settlements.push(result)) + return { name: 'workspace' } + }) + for (let index = 0; index < DISPATCHER_CONTROL_QUEUE_MAX_FRAMES; index += 1) { + bounded.notifyClient(clientId, `control.${index}`) + } + const framesBefore = primary.frames.length + + bounded.feed(encodeJsonRpcFrame({ jsonrpc: '2.0', id: 94, method: 'workspace.get' }, 1, 0)) + await vi.advanceTimersByTimeAsync(0) + + // Nothing goes out, but the connection lives and the caller's own request timeout settles it. + expect(primary.closes).toBe(0) + expect(primary.frames).toHaveLength(framesBefore) + expect(settlements).toEqual([ + { ok: false, error: new Error('Relay response was not admitted') } + ]) + } finally { + bounded.dispose() + } + }) }) diff --git a/src/relay/dispatcher-rpc-routing.ts b/src/relay/dispatcher-rpc-routing.ts index 164a375e1c6..a6e6c24190c 100644 --- a/src/relay/dispatcher-rpc-routing.ts +++ b/src/relay/dispatcher-rpc-routing.ts @@ -203,12 +203,17 @@ export abstract class RelayDispatcherRpcRouting extends RelayDispatcherFrameCode const frame = this.prepareFrame(msg) const lane = frame.frameBytes > DISPATCHER_CONTROL_QUEUE_MAX_BYTES ? 'legacy-response' : 'control' - const accepted = this.enqueuePreparedFrame(client, frame, lane, onSettled) + // Why 'reject': the control lane is a shared budget, so a reply that fits the 1 MiB ceiling alone + // still overflows it under concurrent traffic. Fatal admission would close the connection — every + // pane on the host — over one listing. A response is the droppable class of control frame: it + // carries an id, so the substitute below tells that one caller, and pty.replay/notifyControl keep + // the fatal default because a silent drop there desyncs the client with nothing to retry. + const accepted = this.enqueuePreparedFrame(client, frame, lane, onSettled, 'reject') if (accepted) { return true } // Why: an oversized response must fail its own request; closing would kill every pane on the host. - // A rejected first enqueue either left onSettled untouched or closed the client, so exactly one settlement happens. + // A rejected first enqueue leaves onSettled untouched, so exactly one settlement happens. return this.enqueuePreparedFrame( client, this.prepareFrame({ @@ -227,7 +232,10 @@ export abstract class RelayDispatcherRpcRouting extends RelayDispatcherFrameCode settlement.ok ? { ok: false, error: new Error(RESPONSE_OVER_CAPACITY_MESSAGE) } : settlement - ) + ), + // Why 'reject': if even ~150 bytes will not fit, the caller's own request timeout settles it. + // Closing to report that one request failed is the outcome this whole path exists to avoid. + 'reject' ) } diff --git a/src/relay/fs-handler-file-range.test.ts b/src/relay/fs-handler-file-range.test.ts index 3ecca25ec4c..0d601b9b9fe 100644 --- a/src/relay/fs-handler-file-range.test.ts +++ b/src/relay/fs-handler-file-range.test.ts @@ -105,10 +105,10 @@ describe('readRelayFileRange', () => { // which the writer refuses once the producer queue is busy -- so an over-wide // cap fails with ResponseOverCapacity depending on unrelated load. // - // Fitting the lane once is not enough: the control queue is a SHARED budget - // and overflowing it closes the client, so a full-cap frame has to leave room - // for a second one. Anything wider lets two pipelined tail reads -- or one - // read racing an unrelated response -- kill the connection. + // Fitting the lane once is not enough: the control queue is a SHARED budget, + // so a full-cap frame has to leave room for a second one. Anything wider lets + // two pipelined tail reads -- or one read racing an unrelated response -- + // fail as ResponseOverCapacity on load that has nothing to do with them. it('leaves control-queue headroom for a second full-cap window', async () => { const contents = Buffer.allocUnsafe(MAX_FILE_RANGE_READ_BYTES) for (let i = 0; i < contents.length; i++) { diff --git a/src/shared/file-range-read.ts b/src/shared/file-range-read.ts index fe695e9a2c9..61401f6d40f 100644 --- a/src/shared/file-range-read.ts +++ b/src/shared/file-range-read.ts @@ -8,10 +8,11 @@ * frames to ~350 KB, so a full-cap response takes the control lane instead. * * The control lane is a shared budget, not a per-frame one: two full-cap - * responses fit alongside each other, and the third overflows -- which for a - * response is fatal, it closes the client. That is the same exposure every - * control-lane response already carries (`fs.readFile` frames any sub-1 MiB - * file the same way), and the two-deep headroom is pinned by a test. Widening + * responses fit alongside each other, and the third is refused. That refusal + * costs the one request -- `sendResponse` admits responses with + * `controlOverflow: 'reject'` and substitutes a `ResponseOverCapacity` error + * rather than closing the connection -- but it still turns on unrelated load, + * so the two-deep headroom that keeps it rare is pinned by a test. Widening * the cap spends that headroom, so bigger transfers belong on the ack-paced * bulk lane (`fs.readFileStream`) rather than on a wider window here. * From 7eb13c184c48d61d42f058eeddbe8e18cea0cba9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 01:29:30 -0700 Subject: [PATCH 055/398] fix(ssh): keep remote PowerShell commands inside what sshd's cmd.exe accepts (#17947) * fix(ssh): keep remote Windows commands inside cmd.exe's command-line limit Windows OpenSSH runs every exec request through sshd's DefaultShell, which is cmd.exe on a stock install, and cmd.exe refuses a line over 8191 characters with exit 1 and a localized "The command line is too long". `-EncodedCommand` spends 2.67 characters per script character, so five commands on the first-connect path were already over: the stale upload-stage recovery that opens a fresh install (23,210), promote (20,646), cleanup (19,798), the install-lock steal (11,798) and reserve (9,434). A Windows-to-Windows `ssh:connect` died on the first of them before the relay was ever uploaded (#16126). powerShellCommand now falls back to a gzip self-extracting bootstrap once the inline form passes the budget - these scripts are repetitive enough that the worst one lands at 6.5KB - and throws a message naming the limit if even that cannot fit, rather than letting cmd.exe answer in the host's locale. Commands that already fit are byte-identical. The real-binary PowerShell suite in ssh-relay-upload-stage-commands.test.ts exercises the bootstrap end to end, including `exit` and here-string semantics through Invoke-Expression. * fix(ssh): cite the real command-line budget and reuse the cmd.exe ceiling --- src/main/providers/windows-shell-args.ts | 3 +- .../ssh-relay-sftp-namespace-install.test.ts | 4 +- src/main/ssh/ssh-remote-commands.test.ts | 4 +- src/main/ssh/ssh-remote-powershell.ts | 50 ++++++++++++ ...-remote-windows-command-line-limit.test.ts | 78 +++++++++++++++++++ 5 files changed, 134 insertions(+), 5 deletions(-) create mode 100644 src/main/ssh/ssh-remote-windows-command-line-limit.test.ts diff --git a/src/main/providers/windows-shell-args.ts b/src/main/providers/windows-shell-args.ts index 74e5af534f2..70fd22a06ad 100644 --- a/src/main/providers/windows-shell-args.ts +++ b/src/main/providers/windows-shell-args.ts @@ -14,7 +14,8 @@ import { } from '../powershell-osc133-bootstrap' import { quoteStartupArg } from '../../shared/tui-agent-startup-shell' -const CMD_EXE_COMMAND_LINE_MAX_CHARS = 8191 +/** cmd.exe's own documented ceiling; callers that go through sshd budget below it. */ +export const CMD_EXE_COMMAND_LINE_MAX_CHARS = 8191 const STARTUP_COMMAND_TEXT_MAX_CHARS = 6000 const POWERSHELL_ENCODED_COMMAND_ARG_MAX_CHARS = 28_000 const CMD_UTF8_SETUP_COMMAND = 'chcp 65001 > nul' diff --git a/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts b/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts index 4c0874979b3..10ebce2dac4 100644 --- a/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts +++ b/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts @@ -85,6 +85,7 @@ import { RELAY_DEPLOY_TIMEOUT_MS } from './ssh-relay-deploy-timing' import { parseUnameToRelayPlatform } from './relay-protocol' +import { decodeRemotePowerShellScript } from './ssh-remote-powershell' import { abandonInstall, finalizeInstall, @@ -155,8 +156,7 @@ function issuedMarkerNames(): string[] { } function decodeCommand(command: string): string { - const match = command.match(/-EncodedCommand\s+([A-Za-z0-9+/=]+)/) - return match ? Buffer.from(match[1], 'base64').toString('utf16le') : command + return decodeRemotePowerShellScript(command) } function execCommands(): string[] { diff --git a/src/main/ssh/ssh-remote-commands.test.ts b/src/main/ssh/ssh-remote-commands.test.ts index b69715b23bf..363a87a645d 100644 --- a/src/main/ssh/ssh-remote-commands.test.ts +++ b/src/main/ssh/ssh-remote-commands.test.ts @@ -13,6 +13,7 @@ import { import { tmpdir } from 'node:os' import { join } from 'node:path' import { describe, expect, it } from 'vitest' +import { decodeRemotePowerShellScript } from './ssh-remote-powershell' import { lockAgeSecondsCommand, tryCreateInstallLockCommand, @@ -63,8 +64,7 @@ const powerShell51Executable = : undefined function decodePowerShellCommand(command: string): string { - const match = command.match(/-EncodedCommand\s+([A-Za-z0-9+/=]+)/) - return match ? Buffer.from(match[1], 'base64').toString('utf16le') : '' + return command.includes('-EncodedCommand ') ? decodeRemotePowerShellScript(command) : '' } function runShellCommand(command: string): Promise { diff --git a/src/main/ssh/ssh-remote-powershell.ts b/src/main/ssh/ssh-remote-powershell.ts index cbc3b4faddc..8c94fd3c483 100644 --- a/src/main/ssh/ssh-remote-powershell.ts +++ b/src/main/ssh/ssh-remote-powershell.ts @@ -1,9 +1,59 @@ +import { gunzipSync, gzipSync } from 'node:zlib' import { encodePowerShellCommand } from '../../shared/powershell-command-encoding' +import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args' export { quotePowerShellLiteral as powerShellLiteral, quotePowerShellNativeArgument as powerShellNativeArg } from '../../shared/powershell-native-argument' +// Why cmd.exe and not the 32767 CreateProcess cap: Windows OpenSSH runs every exec request +// through sshd's DefaultShell, cmd.exe on a stock install. Budget under cmd.exe's own ceiling +// to leave room for the `/c` wrapper sshd adds before cmd.exe counts the line. +const WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS = 8_000 + export function powerShellCommand(script: string): string { + const inline = encodedPowerShellCommand(script) + if (inline.length <= WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { + return inline + } + // Why: these scripts are repetitive enough that gzip beats the UTF-16LE tax by + // ~4x, which is the difference between a line cmd.exe runs and one it refuses. + const compressed = encodedPowerShellCommand(selfExtractingPowerShellScript(script)) + if (compressed.length > WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS) { + throw new Error( + `Remote Windows command needs ${compressed.length} characters; Orca budgets ${WINDOWS_REMOTE_COMMAND_LINE_BUDGET_CHARS} for a line sshd hands to cmd.exe, which itself refuses more than ${CMD_EXE_COMMAND_LINE_MAX_CHARS}.` + ) + } + return compressed +} + +function encodedPowerShellCommand(script: string): string { return `powershell.exe -NoProfile -NonInteractive -ExecutionPolicy Bypass -EncodedCommand ${encodePowerShellCommand(script)}` } + +/** Orca-prefixed names so the payload can never shadow the bootstrap's own state. */ +function selfExtractingPowerShellScript(script: string): string { + const payload = gzipSync(Buffer.from(script, 'utf-8'), { level: 9 }).toString('base64') + return [ + `$OrcaScriptBytes = [Convert]::FromBase64String('${payload}')`, + '$OrcaScriptMemory = New-Object System.IO.MemoryStream -ArgumentList (,$OrcaScriptBytes)', + '$OrcaScriptGzip = New-Object System.IO.Compression.GZipStream -ArgumentList $OrcaScriptMemory, ([System.IO.Compression.CompressionMode]::Decompress)', + '$OrcaScriptReader = New-Object System.IO.StreamReader -ArgumentList $OrcaScriptGzip, ([System.Text.Encoding]::UTF8)', + '$OrcaScriptText = $OrcaScriptReader.ReadToEnd()', + '$OrcaScriptReader.Dispose()', + 'Invoke-Expression $OrcaScriptText' + ].join('\n') +} + +/** Inverse of `powerShellCommand`: the script the host will actually run. */ +export function decodeRemotePowerShellScript(command: string): string { + const encoded = command.match(/-EncodedCommand\s+([A-Za-z0-9+/=]+)/u)?.[1] + if (!encoded) { + return command + } + const script = Buffer.from(encoded, 'base64').toString('utf16le') + const payload = script.match( + /^\$OrcaScriptBytes = \[Convert\]::FromBase64String\('([A-Za-z0-9+/=]+)'\)/u + )?.[1] + return payload ? gunzipSync(Buffer.from(payload, 'base64')).toString('utf-8') : script +} diff --git a/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts new file mode 100644 index 00000000000..c104078238e --- /dev/null +++ b/src/main/ssh/ssh-remote-windows-command-line-limit.test.ts @@ -0,0 +1,78 @@ +import { gunzipSync } from 'node:zlib' +import { describe, expect, it } from 'vitest' +import { CMD_EXE_COMMAND_LINE_MAX_CHARS } from '../providers/windows-shell-args' +import { getRemoteHostPlatform } from './ssh-remote-platform' +import { tryStealInstallLockCommand } from './ssh-relay-install-lock-commands' +import { decodeRemotePowerShellScript, powerShellCommand } from './ssh-remote-powershell' +import { + cleanupOwnedRelayUploadStageCommand, + promoteOwnedRelayUploadStageCommand, + recoverOneStaleRelayUploadStageCommand, + reserveRelayUploadStageCommand, + type RelayUploadStageSlot +} from './ssh-relay-upload-stage-commands' + +const windows = getRemoteHostPlatform('win32-x64') +const owner = '.sftp-namespace-123e4567e89b12d3a456426614174000' +const pool = 'C:\\Users\\orca\\.orca-remote\\.upload-stages' +const stage: RelayUploadStageSlot = { + poolDir: pool, + slotName: 'slot-0', + slotDir: `${pool}\\slot-0`, + claimDir: `${pool}\\claim-0`, + deleteDir: `${pool}\\delete-0` +} + +// Why: sshd runs an exec request through its DefaultShell, which is cmd.exe on a +// stock Windows OpenSSH install, and cmd.exe refuses a longer line with exit 1 +// and a localized "The command line is too long" — the whole connect dies there. +describe('Windows remote command line limit', () => { + it.each([ + ['recover stale upload stage', recoverOneStaleRelayUploadStageCommand(windows, pool)], + ['reserve upload stage', reserveRelayUploadStageCommand(windows, pool, owner)], + [ + 'promote upload stage', + promoteOwnedRelayUploadStageCommand(windows, stage, owner, 'C:\\Users\\orca\\.orca-remote') + ], + ['cleanup upload stage', cleanupOwnedRelayUploadStageCommand(windows, stage, owner)], + [ + 'steal stale install lock', + tryStealInstallLockCommand(windows, 'C:\\Users\\orca\\.orca-remote\\relay', 1_200) + ] + ])('keeps the %s command inside what sshd\u2019s cmd.exe accepts', (_name, command) => { + expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS) + }) + + it('leaves a command that already fits byte-identical', () => { + const script = "Write-Output ([Environment]::GetFolderPath('UserProfile'))" + expect(decodeRemotePowerShellScript(powerShellCommand(script))).toBe(script) + }) + + it('carries an oversized script through gzip without altering it', () => { + const script = Array.from( + { length: 200 }, + (_unused, index) => `Write-Output ${index}; $slot = 'C:\\Users\\orca\\stage-${index}'` + ).join('\n') + const command = powerShellCommand(script) + expect(command.length).toBeLessThanOrEqual(CMD_EXE_COMMAND_LINE_MAX_CHARS) + expect(decodeRemotePowerShellScript(command)).toBe(script) + const bootstrap = Buffer.from( + command.match(/-EncodedCommand\s+([A-Za-z0-9+/=]+)$/u)?.[1] ?? '', + 'base64' + ).toString('utf16le') + const payload = bootstrap.match(/FromBase64String\('([A-Za-z0-9+/=]+)'\)/u)?.[1] ?? '' + expect(gunzipSync(Buffer.from(payload, 'base64')).toString('utf-8')).toBe(script) + expect(bootstrap).toContain('Invoke-Expression $OrcaScriptText') + }) + + it('refuses a script no encoding can fit instead of letting cmd.exe reject it', () => { + let seed = 12345 + const incompressible = Array.from({ length: 60_000 }, () => { + seed = (seed * 1103515245 + 12345) % 2147483648 + return String.fromCharCode(97 + (seed % 26)) + }).join('') + expect(() => powerShellCommand(`Write-Output '${incompressible}'`)).toThrow( + /Orca budgets 8000 for a line sshd hands to cmd\.exe/u + ) + }) +}) From 5dc1195a47b6499ff1e099b6f1a5840e3e66dbd3 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 01:34:58 -0700 Subject: [PATCH 056/398] fix(native-chat): keep disabled CLI models out of the Claude picker (#18055) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(native-chat): keep disabled CLI models out of the Claude picker The Claude CLI advertises models it cannot run yet as disabled placeholder rows. On 2.1.237 `list_models` returns a sixth row alongside the four real models: {"value":"cc-update-required-1","displayName":"Fable 5.1 (disabled)", "description":"Update to 2.1.255+ to use Fable 5.1","disabled":true} `toListedModel` never read `disabled`, and for Claude the discovered list replaces the seed catalog verbatim, so the picker rendered that row as a selectable model and `/model cc-update-required-1` went to the CLI. It was also adoptable as a launch default, putting the sentinel behind `--model` on spawn. Drop disabled rows at the parse choke point, which both the native-chat picker and commit-message model discovery share. The two adjacent fixes are the same version-pinning bug the placeholder announces. `compactTerminalText` strips only whitespace, so a point release keeps its dot and the pinned consent literals (`fable5uses…`, `switchtofable5?`) stop matching a "Fable 5.1" prompt — the switch would degrade to `unknown` instead of `interaction-required`. Likewise the scoped weekly usage window matched `display_name === 'fable'` exactly, so it would disappear once the scope is named "Fable 5.1". Claude-Session: https://claude.ai/code/session_01SJy4XGrdre6YaU1wYNKak4 * fix(native-chat): make the Fable consent match version-optional Probing a 2.1.258 CLI shows the shipped Fable 5.1 row carries displayName "Fable" with the version only in the description: {"value":"claude-fable-5-1[1m]","resolvedModel":"claude-fable-5-1", "displayName":"Fable","description":"Fable 5.1 · Most capable for …"} So the consent prompt may name the model with no digits at all. Requiring a version would have missed that, the same way the old pinned literal missed "Fable 5.1". Accept both. Claude-Session: https://claude.ai/code/session_01SJy4XGrdre6YaU1wYNKak4 --------- Co-authored-by: Merge Sim Co-authored-by: Neil <4138956+nwparker@users.noreply.github.com> --- .../claude-fetcher-fable-usage.test.ts | 45 +++++++++++++++++++ .../rate-limits/claude-oauth-usage-request.ts | 6 ++- .../claude-model-switch-confirmation.test.ts | 42 +++++++++++++++++ .../claude-model-switch-confirmation.ts | 11 ++++- src/shared/claude-model-list-probe.test.ts | 26 +++++++++++ src/shared/claude-model-list-probe.ts | 6 ++- 6 files changed, 132 insertions(+), 4 deletions(-) diff --git a/src/main/rate-limits/claude-fetcher-fable-usage.test.ts b/src/main/rate-limits/claude-fetcher-fable-usage.test.ts index 2a84cc4f2af..50617d95bb4 100644 --- a/src/main/rate-limits/claude-fetcher-fable-usage.test.ts +++ b/src/main/rate-limits/claude-fetcher-fable-usage.test.ts @@ -144,6 +144,51 @@ describe('fetchClaudeRateLimits', () => { expect(fetchViaPty).not.toHaveBeenCalled() }) + it('maps a scoped Fable window whose display name carries a point release', async () => { + const configDir = '/Users/test/.claude' + const authPreparation: ClaudeRuntimeAuthPreparation = { + configDir, + envPatch: { CLAUDE_CONFIG_DIR: configDir }, + stripAuthEnv: false, + provenance: 'managed:account-1' + } + vi.mocked(readActiveClaudeKeychainCredentialsStrict).mockResolvedValueOnce( + JSON.stringify({ claudeAiOauth: { accessToken: 'oauth-token' } }) + ) + netFetchMock.mockResolvedValueOnce( + new Response( + JSON.stringify({ + five_hour: { utilization: 36 }, + seven_day: { utilization: 73 }, + // Discriminating: a passing scope match must beat this fallback. + fable_weekly: { utilization: 12 }, + limits: [ + { + kind: 'weekly_scoped', + percent: 64, + resets_at: '2026-07-17T20:00:00.099908+00:00', + is_active: true, + scope: { model: { display_name: 'Fable 5.1' } } + } + ] + }), + { status: 200 } + ) + ) + + await expect( + fetchClaudeRateLimits({ authPreparation, allowUsagePanelSupplement: true }) + ).resolves.toMatchObject({ + provider: 'claude', + status: 'ok', + fableWeekly: { + usedPercent: 64, + resetsAt: Date.parse('2026-07-17T20:00:00.099908+00:00') + } + }) + expect(fetchViaPty).not.toHaveBeenCalled() + }) + it('surfaces inactive scoped Fable usage over the legacy OAuth fallback', async () => { const configDir = '/Users/test/.claude' const authPreparation: ClaudeRuntimeAuthPreparation = { diff --git a/src/main/rate-limits/claude-oauth-usage-request.ts b/src/main/rate-limits/claude-oauth-usage-request.ts index 9565a064514..e29bef5f48e 100644 --- a/src/main/rate-limits/claude-oauth-usage-request.ts +++ b/src/main/rate-limits/claude-oauth-usage-request.ts @@ -32,13 +32,17 @@ async function ensureProxyFromEnvironment(): Promise { }).catch(() => {}) } +// Why: the scope name carries the shipped version once a point release exists +// ("Fable 5.1"), so exact equality would drop the window. +const FABLE_SCOPE_RE = /^fable\b/ + function mapFableWeeklyWindow(data: OAuthUsageResponse): RateLimitWindow | null { const scoped = Array.isArray(data.limits) ? data.limits.find( (limit) => limit?.kind === 'weekly_scoped' && Number.isFinite(limit.percent) && - limit.scope?.model?.display_name?.trim().toLowerCase() === 'fable' + FABLE_SCOPE_RE.test(limit.scope?.model?.display_name?.trim().toLowerCase() ?? '') ) : undefined return ( diff --git a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts index ed4d9f725b0..c0fb5d3aa3d 100644 --- a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts +++ b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts @@ -159,6 +159,48 @@ describe('Claude model switch confirmation detection', () => { await expect(observer.result).resolves.toBe('interaction-required') }) + it('requests interaction for a Fable point-release consent prompt', async () => { + const dataObserver = { current: (_data: string): void => {} } + const observer = createClaudeModelSwitchConfirmationObserver({ + ptyId: 'pty-1', + settings: {}, + expectedModelLabel: 'Fable 5.1', + subscribeToData: (watcher) => { + dataObserver.current = watcher + return vi.fn(() => {}) + }, + timeoutMs: 100 + }) + + await observer.ready + observer.arm() + // Only the versioned consent line: the generic "pick Fable from /model" + // sentence must not be what carries this case. + dataObserver.current('Fable 5.1 uses usage credits and needs a one-time consent') + + await expect(observer.result).resolves.toBe('interaction-required') + }) + + it('requests interaction for a versioned Fable switch prompt', async () => { + const dataObserver = { current: (_data: string): void => {} } + const observer = createClaudeModelSwitchConfirmationObserver({ + ptyId: 'pty-1', + settings: {}, + expectedModelLabel: 'Fable 5.1', + subscribeToData: (watcher) => { + dataObserver.current = watcher + return vi.fn(() => {}) + }, + timeoutMs: 100 + }) + + await observer.ready + observer.arm() + dataObserver.current('Switch to \u001b[1mFable 5.1\u001b[0m? This model uses usage credits.') + + await expect(observer.result).resolves.toBe('interaction-required') + }) + it('reports unknown when the PTY observer cannot be established', async () => { const observer = createClaudeModelSwitchConfirmationObserver({ ptyId: 'pty-1', diff --git a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts index f9664bacfa5..e97daf9cd23 100644 --- a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts +++ b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts @@ -62,12 +62,19 @@ function hasClaudeModelSwitchRejection(buffer: string): boolean { return compactTerminalText(buffer).includes('keptmodelas') } +// Why: compactTerminalText only strips whitespace, so a point release keeps its +// dot ("fable5.1uses..."). The version is optional because the CLI's own label +// for the newest Fable carries no number at all. +const FABLE_VERSION = String.raw`fable(?:\d+(?:\.\d+)*)?` +const FABLE_CONSENT_RE = new RegExp(`${FABLE_VERSION}usesusagecreditsandneedsaone-timeconsent`) +const FABLE_SWITCH_PROMPT_RE = new RegExp(`switchto${FABLE_VERSION}\\?`) + function hasClaudeModelSwitchInteraction(buffer: string): boolean { const text = compactTerminalText(buffer) return ( - text.includes('fable5usesusagecreditsandneedsaone-timeconsent') || + FABLE_CONSENT_RE.test(text) || text.includes('pickfablefrom/modelinaninteractivesessiontosetitup') || - (text.includes('switchtofable5?') && text.includes('usagecredits')) + (FABLE_SWITCH_PROMPT_RE.test(text) && text.includes('usagecredits')) ) } diff --git a/src/shared/claude-model-list-probe.test.ts b/src/shared/claude-model-list-probe.test.ts index 86305b45549..056b41e0a98 100644 --- a/src/shared/claude-model-list-probe.test.ts +++ b/src/shared/claude-model-list-probe.test.ts @@ -103,6 +103,32 @@ describe('parseClaudeModelList', () => { expect(parseClaudeModelList(hostile)).toEqual([]) }) + it('drops the disabled placeholder row the CLI advertises for a model it cannot run', () => { + // Captured from `claude` 2.1.237: Fable 5.1 is announced but gated on 2.1.255+. + const parsed = parseClaudeModelList( + controlResponseLine([ + { value: 'sonnet', displayName: 'Sonnet' }, + { + value: 'cc-update-required-1', + resolvedModel: 'cc-update-required-1', + displayName: 'Fable 5.1 (disabled)', + description: 'Update to 2.1.255+ to use Fable 5.1', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'], + disabled: true + } + ]) + ) + expect(parsed.map(({ id }) => id)).toEqual(['sonnet']) + }) + + it('keeps a model that reports disabled as anything other than true', () => { + const parsed = parseClaudeModelList( + controlResponseLine([{ value: 'fable', displayName: 'Fable', disabled: false }]) + ) + expect(parsed.map(({ id }) => id)).toEqual(['fable']) + }) + it('ignores effort levels when the model does not declare effort support', () => { const parsed = parseClaudeModelList( controlResponseLine([ diff --git a/src/shared/claude-model-list-probe.ts b/src/shared/claude-model-list-probe.ts index fa953d3da16..9be7f5c3a12 100644 --- a/src/shared/claude-model-list-probe.ts +++ b/src/shared/claude-model-list-probe.ts @@ -53,6 +53,7 @@ type RawListedModel = { supportsEffort?: unknown supportedEffortLevels?: unknown supportsFastMode?: unknown + disabled?: unknown } function toListedModel(value: unknown): ClaudeListedModel | null { @@ -61,7 +62,10 @@ function toListedModel(value: unknown): ClaudeListedModel | null { } const raw = value as RawListedModel const id = typeof raw.value === 'string' ? raw.value.trim() : '' - if (!id) { + // Why: the CLI advertises models it cannot run yet as disabled placeholder + // rows ("Fable 5.1 (disabled)", value `cc-update-required-1`); selecting one + // sends that sentinel straight to `--model`. + if (!id || raw.disabled === true) { return null } const label = typeof raw.displayName === 'string' && raw.displayName.trim() ? raw.displayName : id From 4bc20cb842b784f9202d1095575b7bc92b406a75 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 01:39:48 -0700 Subject: [PATCH 057/398] fix(wsl): name an explicit Windows cwd for wsl.exe spawns (#17834) * fix(wsl): name an explicit Windows cwd for wsl.exe spawns Removing the worktree Orca was launched from broke every wsl.exe spawn for the rest of the session. The WSL command builders passed `cwd: undefined` meaning "the directory is inside the command" -- but CreateProcessW reads NULL as "inherit the parent's", and the parent's was a \\wsl.localhost path Linux had just deleted. Fixes #16463 * fix(wsl): name the spawn directory at the six remaining wsl.exe sites The first commit fixed the WSL command builders. Six spawn sites were left inheriting the process cwd, which is the same deletable `\\wsl.localhost` worktree: `wsl-availability` (both probes), the WSL filesystem watcher, the agent-hook relay launch, the UNC delete, and the local worktree filesystem. `wsl-availability` is the one that matters most, and it turns the bug into a latching false negative. `isRetryableWslProbeFailure` returns false for ENOENT, so a spawn that failed only because the inherited cwd was gone is cached as "WSL is not installed" on the 10-minute definitive TTL with exponential backoff up to 30 minutes. Git keeps working and Orca reports WSL unavailable -- worse than the bug being fixed. ENOENT stays non-retryable. It is answer-shaped for the reason it is meant to be -- wsl.exe is not on PATH -- and naming the directory is what removes the one cause that was not. Making it retryable would instead re-probe every non-WSL Windows machine on the short window, and would leave the false ENOENT in place for the other five sites, which have no cache to correct. Three of these are also on the `runWslProcess` W3 migration allowlist; this is the interim until they move, and matches what #17837 does inside the runner. --- config/tsconfig.tc.web.json | 1 + .../agent-hooks/wsl-hook-relay-launch.test.ts | 25 ++++++ src/main/agent-hooks/wsl-hook-relay-launch.ts | 7 +- src/main/cli/wsl-cli-installer.test.ts | 8 +- src/main/cli/wsl-cli-scripts.ts | 10 ++- .../command-runner/wsl-command-resolution.ts | 14 +-- src/main/git/runner-command-exec.test.ts | 12 ++- src/main/git/runner-wsl-gh-fallback.test.ts | 17 +++- ...tlab-known-host-probe-wsl-fallback.test.ts | 6 +- src/main/ipc/filesystem-watcher-wsl.test.ts | 6 +- src/main/ipc/filesystem-watcher-wsl.ts | 7 +- src/main/local-worktree-filesystem.test.ts | 6 +- src/main/local-worktree-filesystem.ts | 5 ++ ...e-text-generation-local-subprocess.test.ts | 6 +- ...ge-text-generation-model-discovery.test.ts | 6 +- src/main/wsl-availability.ts | 18 +++- src/main/wsl-interop-spawn-directory.test.ts | 90 +++++++++++++++++++ src/main/wsl-interop-spawn-directory.ts | 70 +++++++++++++++ src/main/wsl-unc-delete.test.ts | 5 +- src/main/wsl-unc-delete.ts | 5 +- src/main/wsl.test.ts | 34 +++++++ src/main/wsl.ts | 21 +++-- .../wsl-probe-failure-swallow-allowlist.txt | 6 ++ 23 files changed, 357 insertions(+), 28 deletions(-) create mode 100644 src/main/agent-hooks/wsl-hook-relay-launch.test.ts create mode 100644 src/main/wsl-interop-spawn-directory.test.ts create mode 100644 src/main/wsl-interop-spawn-directory.ts diff --git a/config/tsconfig.tc.web.json b/config/tsconfig.tc.web.json index afe5e83024c..56253527c69 100644 --- a/config/tsconfig.tc.web.json +++ b/config/tsconfig.tc.web.json @@ -31,6 +31,7 @@ "../src/main/wsl-distro-retry.ts", "../src/main/wsl-running-distro-cache.ts", "../src/main/wsl.ts", + "../src/main/wsl-interop-spawn-directory.ts", "../src/main/persistence/applying-settings/ui-state-read.ts", "../src/main/persistence/applying-settings/ui-state-update.ts", "../src/main/persistence/applying-settings/ui-selection-normalization.ts", diff --git a/src/main/agent-hooks/wsl-hook-relay-launch.test.ts b/src/main/agent-hooks/wsl-hook-relay-launch.test.ts new file mode 100644 index 00000000000..6ed9cf2625a --- /dev/null +++ b/src/main/agent-hooks/wsl-hook-relay-launch.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it, vi } from 'vitest' + +const { spawnMock } = vi.hoisted(() => ({ + spawnMock: vi.fn((..._args: unknown[]) => ({ pid: 1 })) +})) + +vi.mock('node:child_process', () => ({ spawn: spawnMock })) + +import { spawnWslRelayProcess } from './wsl-hook-relay-launch' + +describe('spawnWslRelayProcess', () => { + it('names an explicit Windows directory rather than inheriting one', () => { + spawnWslRelayProcess('Ubuntu', {}, '1.2.3') + + // Why (#16463): the guest path is inside the `sh -c` command, so the Windows + // cwd only decides whether CreateProcessW succeeds. Omitting it inherits + // Orca's own — a `\\wsl.localhost` worktree the user can delete, after which + // every relay launch fails `spawn wsl.exe ENOENT` for the rest of the session. + expect(spawnMock).toHaveBeenCalledWith( + 'wsl.exe', + expect.arrayContaining(['-d', 'Ubuntu', '--exec']), + expect.objectContaining({ cwd: expect.any(String) }) + ) + }) +}) diff --git a/src/main/agent-hooks/wsl-hook-relay-launch.ts b/src/main/agent-hooks/wsl-hook-relay-launch.ts index 9f32de10bd5..ae1ea9c20d7 100644 --- a/src/main/agent-hooks/wsl-hook-relay-launch.ts +++ b/src/main/agent-hooks/wsl-hook-relay-launch.ts @@ -16,6 +16,7 @@ import { } from './wsl-hook-relay-sentinel' import { addOrcaWslInteropEnv } from '../pty/wsl-orca-env' import { runWslProcess } from '../wsl/wsl-runner' +import { resolveWslInteropSpawnCwd } from '../wsl-interop-spawn-directory' import { listRunningWslDistrosAsync } from '../wsl' import { WSL_HOOK_RELAY_BUNDLE_NAME, @@ -137,7 +138,11 @@ export function spawnWslRelayProcess( return spawn('wsl.exe', ['-d', distro, '--exec', 'sh', '-c', command], { env, stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true + windowsHide: true, + // Why explicit (#16463): the guest path is in `command`, so the Windows cwd + // only decides whether CreateProcessW succeeds -- and an inherited one is a + // worktree the user can delete, which kills every later relay launch. + cwd: resolveWslInteropSpawnCwd() }) } diff --git a/src/main/cli/wsl-cli-installer.test.ts b/src/main/cli/wsl-cli-installer.test.ts index 717232509ba..a728e0acafe 100644 --- a/src/main/cli/wsl-cli-installer.test.ts +++ b/src/main/cli/wsl-cli-installer.test.ts @@ -340,7 +340,13 @@ describe('WslCliInstaller', () => { expect(bridge).toContain('$ForwardArgs = @($args[$ForwardArgStart..($args.Count - 1)])') expect(bridge).toContain('if ([string]::IsNullOrEmpty($WslCwd))') expect(bridge).toContain('$env:ORCA_CLI_CWD = $WslCwd') - expect(bridge).toContain('Push-Location -LiteralPath (Split-Path -Parent $OrcaLauncher)') + expect(bridge).toContain('$LauncherDirectory = Split-Path -Parent $OrcaLauncher') + expect(bridge).toContain('Push-Location -LiteralPath $LauncherDirectory') + // Why (#16463): Push-Location moves only the PowerShell provider location. + // Without an explicit WorkingDirectory the started app inherits the caller's + // Win32 cwd — the user's worktree on \\wsl.localhost — and every wsl.exe + // spawn it makes dies with ENOENT once that worktree is removed. + expect(bridge).toContain('$StartInfo.WorkingDirectory = $LauncherDirectory') expect(bridge).toContain('function ConvertTo-NativeCommandLineArgument') expect(bridge).toContain("[void]$Quoted.Append([char]'\\', $BackslashCount * 2 + 1)") expect(bridge).toContain('$StartInfo.UseShellExecute = $false') diff --git a/src/main/cli/wsl-cli-scripts.ts b/src/main/cli/wsl-cli-scripts.ts index 8eda355cc93..8875a81d18e 100644 --- a/src/main/cli/wsl-cli-scripts.ts +++ b/src/main/cli/wsl-cli-scripts.ts @@ -88,7 +88,8 @@ try { } else { $env:ORCA_CLI_CWD = $WslCwd } - Push-Location -LiteralPath (Split-Path -Parent $OrcaLauncher) + $LauncherDirectory = Split-Path -Parent $OrcaLauncher + Push-Location -LiteralPath $LauncherDirectory # Why: Windows PowerShell 5.1 cannot losslessly splat strings to native argv. $StartInfo = [System.Diagnostics.ProcessStartInfo]::new() $StartInfo.FileName = $OrcaLauncher @@ -96,6 +97,13 @@ try { ConvertTo-NativeCommandLineArgument $_ }) -join ' ') $StartInfo.UseShellExecute = $false + # Why (#16463): Push-Location moves the PowerShell provider location, not the + # Win32 current directory, and an empty WorkingDirectory with UseShellExecute + # disabled means "inherit the caller's". Launched from a WSL shell that is the + # user's worktree on the 9P share, so without this the app stands in a + # directory Linux can delete -- after which every CreateProcessW it makes + # fails ERROR_PATH_NOT_FOUND, reported as: spawn wsl.exe ENOENT. + $StartInfo.WorkingDirectory = $LauncherDirectory $Process = [System.Diagnostics.Process]::Start($StartInfo) if ($null -eq $Process) { throw 'Unable to start the Orca Windows CLI launcher.' diff --git a/src/main/git/command-runner/wsl-command-resolution.ts b/src/main/git/command-runner/wsl-command-resolution.ts index a70c0ca2bc9..559e1821bd1 100644 --- a/src/main/git/command-runner/wsl-command-resolution.ts +++ b/src/main/git/command-runner/wsl-command-resolution.ts @@ -13,6 +13,7 @@ import { type WslProcessGroupTermination } from '../wsl-process-group-termination' import { translateArgForWsl, translateArgsForWsl } from './wsl-path-translation' +import { resolveWslInteropSpawnCwd } from '../../wsl-interop-spawn-directory' // Env-assignment prefix for WSL-routed git, where spawn env can't cross the wsl.exe boundary; values are shell-safe unquoted. const GIT_OUTPUT_LOCALE_SHELL_PREFIX = Object.entries(UNTRANSLATED_GIT_OUTPUT_ENV) @@ -111,7 +112,7 @@ export function resolveCommand( ...(linuxCwd ? ['-C', linuxCwd] : []), ...translatedArgs ], - cwd: undefined, + cwd: resolveWslInteropSpawnCwd(), wsl, wslMode: 'direct-git' }, @@ -130,7 +131,7 @@ export function resolveCommand( { binary: 'wsl.exe', args: buildWslExecArgs(wsl.distro, ['sh', '-lc', captured.command]), - cwd: undefined, + cwd: resolveWslInteropSpawnCwd(), wsl, wslMode: 'login-shell', captured @@ -142,7 +143,7 @@ export function resolveCommand( { binary: 'wsl.exe', args: buildWslExecArgs(wsl.distro, ['sh', '-lc', buildWslLoginShellCommand(shellCmd)]), - cwd: undefined, + cwd: resolveWslInteropSpawnCwd(), wsl, wslMode: 'login-shell' }, @@ -154,8 +155,11 @@ export function resolveCommand( { binary: 'wsl.exe', args: buildWslExecArgs(wsl.distro, ['bash', '-c', shellCmd]), - // Why: the `cd` inside bash -c handles the directory; a UNC cwd on the Node process is redundant and can break Node internals. - cwd: undefined, + // Why: the `cd` inside bash -c handles the Linux directory. This names an + // explicit Windows directory anyway, because `undefined` makes + // CreateProcessW inherit the parent's — which is a deletable WSL UNC path + // when Orca was launched from a worktree (#16463). + cwd: resolveWslInteropSpawnCwd(), wsl, wslMode: 'non-login-shell' }, diff --git a/src/main/git/runner-command-exec.test.ts b/src/main/git/runner-command-exec.test.ts index 1974c4e2384..27d7a72c747 100644 --- a/src/main/git/runner-command-exec.test.ts +++ b/src/main/git/runner-command-exec.test.ts @@ -626,7 +626,11 @@ describe('runner execFile timeout handling', () => { expect(execFileMock).toHaveBeenCalledWith( 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'sh', '-lc', expect.any(String)], - expect.objectContaining({ cwd: undefined }), + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. The Linux directory still rides inside the command (/mnt/c/repo, + // asserted below). + expect.objectContaining({ cwd: expect.any(String) }), expect.any(Function) ) // A read also warms the direct-git environment probe in the background, so @@ -659,7 +663,11 @@ describe('runner execFile timeout handling', () => { expect(execFileMock).toHaveBeenCalledWith( 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', expect.any(String)], - expect.objectContaining({ cwd: undefined }), + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. The Linux directory still rides inside the command (/mnt/c/repo, + // asserted below). + expect.objectContaining({ cwd: expect.any(String) }), expect.any(Function) ) const shellCommand = execFileMock.mock.calls[0]?.[1]?.[5] as string diff --git a/src/main/git/runner-wsl-gh-fallback.test.ts b/src/main/git/runner-wsl-gh-fallback.test.ts index 57dd86ba26c..5f933c60620 100644 --- a/src/main/git/runner-wsl-gh-fallback.test.ts +++ b/src/main/git/runner-wsl-gh-fallback.test.ts @@ -99,7 +99,10 @@ describe('ghExecFileAsync WSL fallback', () => { '-c', "cd '/home/jinwoo/stably/noqa' && 'gh' 'issue' 'list' '--repo' 'stablyhq/noqa' '--json' 'number,title'" ], - expect.objectContaining({ cwd: undefined }), + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. The Linux directory still rides inside the command. + expect.objectContaining({ cwd: expect.any(String) }), expect.any(Function) ) expect(execFileMock).toHaveBeenNthCalledWith( @@ -382,7 +385,11 @@ describe('ghExecFileAsync WSL fallback', () => { 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'gh' 'api' 'rate_limit'"], - expect.objectContaining({ cwd: undefined }), + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. This global call has no repo directory at all, so nothing about + // where it runs changes. + expect.objectContaining({ cwd: expect.any(String) }), expect.any(Function) ) }) @@ -501,7 +508,11 @@ describe('ghExecFileAsync WSL fallback', () => { 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'api' 'projects'"], - expect.objectContaining({ cwd: undefined }), + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. This global call has no repo directory at all, so nothing about + // where it runs changes. + expect.objectContaining({ cwd: expect.any(String) }), expect.any(Function) ) }) diff --git a/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts b/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts index a3fe62c9063..40da88f0c91 100644 --- a/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts +++ b/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts @@ -76,7 +76,11 @@ describe('glab known-hosts probe on Windows', () => { expect(execFileMock).toHaveBeenCalledWith( 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'auth' 'status'"], - expect.objectContaining({ cwd: undefined }), + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. This probe has no repo directory at all, so nothing about where + // it runs changes. The native `glab` assertion above keeps `undefined`. + expect.objectContaining({ cwd: expect.any(String) }), expect.any(Function) ) }) diff --git a/src/main/ipc/filesystem-watcher-wsl.test.ts b/src/main/ipc/filesystem-watcher-wsl.test.ts index 13346047d86..84d8d0a2306 100644 --- a/src/main/ipc/filesystem-watcher-wsl.test.ts +++ b/src/main/ipc/filesystem-watcher-wsl.test.ts @@ -95,7 +95,11 @@ describe('createWslWatcher', () => { ['-d', 'Ubuntu', '--exec', 'sh', '-s', '--', '/home/me/repo'], expect.objectContaining({ stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true + windowsHide: true, + // Why a concrete directory (#16463): the watched path rides in argv, so + // an omitted cwd only means CreateProcessW inherits Orca's -- a worktree + // that can be deleted, after which every watcher start is ENOENT. + cwd: expect.any(String) }) ) }) diff --git a/src/main/ipc/filesystem-watcher-wsl.ts b/src/main/ipc/filesystem-watcher-wsl.ts index 86f10c2ee68..a444113c1cf 100644 --- a/src/main/ipc/filesystem-watcher-wsl.ts +++ b/src/main/ipc/filesystem-watcher-wsl.ts @@ -14,6 +14,7 @@ import { parseWslUncPath } from '../../shared/wsl-paths' import { createWslWatcherProcessExit, createWslWatcherStartup } from './wsl-watcher-process-exit' import { reserveWatcherChild, WatcherChildCapacityError } from './parcel-watcher-child-registry' import { createDebouncedBatch, type DebouncedBatch } from './filesystem-watcher-batch-control' +import { resolveWslInteropSpawnCwd } from '../wsl-interop-spawn-directory' export type WatcherSubscription = { unsubscribe(): Promise @@ -244,7 +245,11 @@ export async function createWslWatcher( try { child = spawn('wsl.exe', ['-d', distro, '--exec', 'sh', '-s', '--', linuxPath], { stdio: ['pipe', 'pipe', 'pipe'], - windowsHide: true + windowsHide: true, + // Why explicit (#16463): the watched directory rides in argv, and an + // inherited cwd is a worktree that can be deleted -- after which every + // watcher start fails `spawn wsl.exe ENOENT`. + cwd: resolveWslInteropSpawnCwd() }) } catch (error) { releaseChildReservation() diff --git a/src/main/local-worktree-filesystem.test.ts b/src/main/local-worktree-filesystem.test.ts index 52b1d16f21a..c9e97f72e90 100644 --- a/src/main/local-worktree-filesystem.test.ts +++ b/src/main/local-worktree-filesystem.test.ts @@ -188,7 +188,11 @@ describe('local worktree filesystem runtime access', () => { 1, expect.objectContaining({ program: 'wsl.exe', - args: expect.arrayContaining(['-d', 'Ubuntu']) + args: expect.arrayContaining(['-d', 'Ubuntu']), + // Why a concrete directory (#16463): the guest path is inside the + // command, and these run while a worktree is being removed -- which is + // the cwd an omitted one would inherit. + cwd: expect.any(String) }) ) const removeArgs = runProcessMock.mock.calls[2]?.[0].args as string[] diff --git a/src/main/local-worktree-filesystem.ts b/src/main/local-worktree-filesystem.ts index f5d8714e196..1078b1f50bc 100644 --- a/src/main/local-worktree-filesystem.ts +++ b/src/main/local-worktree-filesystem.ts @@ -3,6 +3,7 @@ import { lstat, readFile } from 'node:fs/promises' import { buildWslExecArgs, quotePosixShell } from '../shared/wsl-login-shell-command' import { removeHostTree } from './host-tree-removal' import { toLinuxPath } from './wsl' +import { resolveWslInteropSpawnCwd } from './wsl-interop-spawn-directory' import type { ReadPath, StatPath } from './worktree-orphan-gitdir-proof' export { toHostFilesystemPath, toHostRemovalPath } from './host-tree-removal' @@ -36,6 +37,10 @@ async function runWslCommand(distro: string, command: string): Promise { const result = await runProcess({ program: 'wsl.exe', args: buildWslExecArgs(distro, ['sh', '-c', command]), + // Why explicit (#16463): the guest path is inside `command`, so this only + // decides whether CreateProcessW succeeds -- and these calls run while a + // worktree is being removed, which is the cwd an inherited one would be. + cwd: resolveWslInteropSpawnCwd(), timeoutMs: WSL_FILE_OPERATION_TIMEOUT_MS }) if (result.timedOut) { diff --git a/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts b/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts index fcdbe4719cc..c124cb85779 100644 --- a/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts +++ b/src/main/text-generation/commit-message-text-generation-local-subprocess.test.ts @@ -164,7 +164,11 @@ describe('generateCommitMessageFromContext', () => { 'wsl.exe', ['-d', 'Ubuntu 24.04', '--exec', 'sh', '-lc', expect.any(String)], expect.objectContaining({ - cwd: undefined, + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. The Linux directory still rides inside the command (/mnt/c/repo, + // asserted below), so the Windows-side cwd never decides where the agent runs. + cwd: expect.any(String), windowsHide: true, env: expect.objectContaining({ CODEX_HOME: '/home/tester/.codex' }) }) diff --git a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts index d54f1d995f6..7ef438c06ae 100644 --- a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts +++ b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts @@ -235,7 +235,11 @@ describe('discoverCommitMessageModelsLocal', () => { 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'sh', '-lc', expect.any(String)], expect.objectContaining({ - cwd: undefined, + // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit + // Orca's own cwd, a deletable WSL UNC path when it was launched from a + // worktree. The Linux directory still rides inside the command (/mnt/c/repo, + // asserted below), so the Windows-side cwd never decides where discovery runs. + cwd: expect.any(String), windowsHide: true }) ) diff --git a/src/main/wsl-availability.ts b/src/main/wsl-availability.ts index c44476c9d60..1d14f526413 100644 --- a/src/main/wsl-availability.ts +++ b/src/main/wsl-availability.ts @@ -1,4 +1,5 @@ import { execFile, execFileSync } from 'node:child_process' +import { resolveWslInteropSpawnCwd } from './wsl-interop-spawn-directory' type WslAvailabilityCache = | { available: true } @@ -35,6 +36,9 @@ function wslAvailabilityRetryDelayMs(cache: { retryable: boolean; failures: numb return Math.min(base * 2 ** (cache.failures - 1), WSL_AVAILABILITY_MAX_RETRY_DELAY_MS) } +// Why ENOENT stays definitive: it means wsl.exe is not on PATH. It used to also mean +// "the cwd this process inherited was deleted", which is not answer-shaped at all -- +// naming an explicit spawn directory below is what removes that source (#16463). // Why: a non-zero exit (wsl.exe ran and said no) or ENOENT (not installed) is answer-shaped, // so it earns a long window rather than the short one a timeout gets. execFileSync reports the // exit code as `status`, the execFile callback as a numeric `code`; both must count as @@ -95,7 +99,14 @@ function probeWslStatus(): Promise { execFile( 'wsl.exe', ['--status'], - { timeout: WSL_AVAILABILITY_PROBE_TIMEOUT_MS, windowsHide: true }, + { + timeout: WSL_AVAILABILITY_PROBE_TIMEOUT_MS, + windowsHide: true, + // Why explicit (#16463): inheriting a cwd the user deleted makes + // CreateProcessW fail ENOENT, which this cache reads as "WSL is not + // installed" and holds on the definitive TTL with backoff. + cwd: resolveWslInteropSpawnCwd() + }, (error: unknown) => { if (error) { reject(error) @@ -129,7 +140,10 @@ export function isWslAvailable(): boolean { try { execFileSync('wsl.exe', ['--status'], { stdio: ['pipe', 'pipe', 'pipe'], - timeout: WSL_AVAILABILITY_PROBE_TIMEOUT_MS + timeout: WSL_AVAILABILITY_PROBE_TIMEOUT_MS, + // Same reason as the async twin: they share one cache, so a false ENOENT + // from either poisons both. + cwd: resolveWslInteropSpawnCwd() }) return cacheWslAvailabilityProbeResult(null, startedAtGeneration) } catch (error) { diff --git a/src/main/wsl-interop-spawn-directory.test.ts b/src/main/wsl-interop-spawn-directory.test.ts new file mode 100644 index 00000000000..310e5d3a5ca --- /dev/null +++ b/src/main/wsl-interop-spawn-directory.test.ts @@ -0,0 +1,90 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' + +import { + resetWslInteropSpawnDirectoryCache, + resolveWslInteropSpawnCwd +} from './wsl-interop-spawn-directory' + +// Regression coverage for #16463 ("Removing the worktree Orca was launched from +// breaks every wsl.exe spawn for the rest of the session"). The WSL command +// builders passed `cwd: undefined` meaning "the directory is inside the +// command", but CreateProcessW reads NULL as "inherit the parent's" — and the +// parent's was a `\\wsl.localhost\...` worktree Linux had just deleted. 1805 of +// 1806 git calls then failed `spawn wsl.exe ENOENT` until the app restarted. + +const createdRoots: string[] = [] + +function makeExistingDirectory(): string { + const dir = mkdtempSync(join(tmpdir(), 'orca-wsl-spawn-cwd-')) + createdRoots.push(dir) + return dir +} + +const ENV_KEYS = ['ORCA_USER_DATA_PATH', 'USERPROFILE', 'HOMEDRIVE', 'HOMEPATH'] as const +const savedEnv = new Map() + +beforeEach(() => { + for (const key of ENV_KEYS) { + savedEnv.set(key, process.env[key]) + delete process.env[key] + } + resetWslInteropSpawnDirectoryCache() +}) + +afterEach(() => { + for (const key of ENV_KEYS) { + const saved = savedEnv.get(key) + if (saved === undefined) { + delete process.env[key] + } else { + process.env[key] = saved + } + } + resetWslInteropSpawnDirectoryCache() + while (createdRoots.length > 0) { + rmSync(createdRoots.pop()!, { recursive: true, force: true }) + } +}) + +describe('resolveWslInteropSpawnCwd', () => { + it('names the app-owned directory first, so no worktree can be the answer', () => { + const userData = makeExistingDirectory() + process.env.ORCA_USER_DATA_PATH = userData + process.env.USERPROFILE = makeExistingDirectory() + + expect(resolveWslInteropSpawnCwd()).toBe(userData) + }) + + it('skips a candidate that does not resolve instead of naming it', () => { + process.env.ORCA_USER_DATA_PATH = join(tmpdir(), 'orca-wsl-spawn-cwd-never-created') + const profile = makeExistingDirectory() + process.env.USERPROFILE = profile + + expect(resolveWslInteropSpawnCwd()).toBe(profile) + }) + + it('always names some directory rather than letting the spawn inherit one', () => { + // Why: inheriting is the failure mode. With no configured candidate at all + // the home directory and system root still stand between a spawn and the + // parent's cwd. + expect(resolveWslInteropSpawnCwd()).toEqual(expect.any(String)) + }) + + it('re-answers after the directory it memoized goes away mid-session', () => { + // This is the incident: the chosen directory was valid when the process + // started and was deleted underneath it hours later. A memo that is never + // re-validated reproduces the original bug one layer up. + const doomed = makeExistingDirectory() + process.env.ORCA_USER_DATA_PATH = doomed + const survivor = makeExistingDirectory() + expect(resolveWslInteropSpawnCwd()).toBe(doomed) + + rmSync(doomed, { recursive: true, force: true }) + process.env.ORCA_USER_DATA_PATH = survivor + + expect(resolveWslInteropSpawnCwd()).toBe(survivor) + }) +}) diff --git a/src/main/wsl-interop-spawn-directory.ts b/src/main/wsl-interop-spawn-directory.ts new file mode 100644 index 00000000000..daa8e08d830 --- /dev/null +++ b/src/main/wsl-interop-spawn-directory.ts @@ -0,0 +1,70 @@ +import { statSync } from 'node:fs' +import { homedir } from 'node:os' + +/** + * A Windows directory that is safe to hand `wsl.exe` as its working directory. + * + * Why this exists (#16463): the WSL command builders set `cwd: undefined`, + * meaning "the directory is already expressed inside the command" — but that is + * not what `undefined` means to `CreateProcessW`. libuv passes NULL for + * `lpCurrentDirectory`, and NULL means *inherit the parent's*. Orca launched by + * `orca-ide` from a WSL shell inherits `\\wsl.localhost\\...\` + * as its Win32 cwd; Linux can delete that directory out from under a Windows + * process across the 9P share, and from then on `CreateProcessW` fails + * `ERROR_PATH_NOT_FOUND` — surfaced by libuv as `spawn wsl.exe ENOENT`, for the + * rest of the process's life, for every repository. + * + * Naming an explicit directory removes the dependency on process-global state + * entirely, so a repaired or unrepaired `process.cwd()` cannot decide whether + * git works. It is never the cwd the command runs in: WSL invocations carry + * their Linux directory in `git -C`, a `cd` inside `bash -c`, or the `sh -c` + * wrapper `withGuestCwd` builds. + */ + +let cachedSpawnCwd: string | null = null + +function isExistingDirectory(path: string | undefined | null): path is string { + if (!path) { + return false + } + try { + return statSync(path).isDirectory() + } catch { + return false + } +} + +/** Test seam: forget the memoized directory so a later probe re-validates. */ +export function resetWslInteropSpawnDirectoryCache(): void { + cachedSpawnCwd = null +} + +export function resolveWslInteropSpawnCwd(): string | undefined { + // Why re-validate: the answer is only useful while it still resolves, and the + // user's profile directory can go away on a roaming/mapped-drive host. + if (isExistingDirectory(cachedSpawnCwd)) { + return cachedSpawnCwd + } + const env = process.env + // Why this order: an app-owned directory first (it outlives every worktree), + // then the user's profile, then the system root as a floor that always exists. + // A root is fine here — nothing scans this directory, it is only the value + // `CreateProcessW` receives for `lpCurrentDirectory`. + const candidates: (string | undefined)[] = [ + env.ORCA_USER_DATA_PATH, + env.USERPROFILE, + env.HOMEDRIVE && env.HOMEPATH ? `${env.HOMEDRIVE}${env.HOMEPATH}` : undefined, + homedir(), + env.SystemDrive ? `${env.SystemDrive}\\` : 'C:\\' + ] + for (const candidate of candidates) { + if (isExistingDirectory(candidate)) { + cachedSpawnCwd = candidate + return candidate + } + } + cachedSpawnCwd = null + // Why undefined rather than a guess: inheriting is still better than naming a + // directory we just proved does not exist. + return undefined +} diff --git a/src/main/wsl-unc-delete.test.ts b/src/main/wsl-unc-delete.test.ts index 4ca6c43312c..a902b4b6d53 100644 --- a/src/main/wsl-unc-delete.test.ts +++ b/src/main/wsl-unc-delete.test.ts @@ -50,8 +50,11 @@ describe('tryDeleteWslUncPath', () => { }) expect(execFileMock).toHaveBeenCalledTimes(1) - const [binary, spawnArgs] = execFileMock.mock.calls[0] + const [binary, spawnArgs, spawnOptions] = execFileMock.mock.calls[0] expect(binary).toBe('wsl.exe') + // Why a concrete directory (#16463): this deletes worktrees, so the cwd it + // would otherwise inherit is the very directory about to disappear. + expect(spawnOptions).toEqual(expect.objectContaining({ cwd: expect.any(String) })) expect(spawnArgs).toEqual([ '-d', 'Ubuntu', diff --git a/src/main/wsl-unc-delete.ts b/src/main/wsl-unc-delete.ts index 9cc62c9b236..b094a152beb 100644 --- a/src/main/wsl-unc-delete.ts +++ b/src/main/wsl-unc-delete.ts @@ -1,5 +1,6 @@ import { execFile } from 'node:child_process' import { parseWslPath } from './wsl' +import { resolveWslInteropSpawnCwd } from './wsl-interop-spawn-directory' import { containedDeleteCommand, rejectionFromWslDeleteStderr, @@ -69,7 +70,9 @@ function execFileWsl(distro: string, command: string[]): Promise { ['-d', distro, '--exec', ...command], // Why: a generous bound so deleting a large directory tree on the WSL fs // doesn't abort mid-delete, while still capping a wedged wsl.exe. - { encoding: 'utf-8', timeout: 30000 }, + // Why an explicit cwd (#16463): the target rides in argv, and this deletes + // worktrees -- so an inherited cwd is exactly the directory about to go. + { encoding: 'utf-8', timeout: 30000, cwd: resolveWslInteropSpawnCwd() }, (error, _stdout, stderr) => { if (error) { reject(wslDeleteError(error, stderr)) diff --git a/src/main/wsl.test.ts b/src/main/wsl.test.ts index bc3498b1145..6ef8adbbb61 100644 --- a/src/main/wsl.test.ts +++ b/src/main/wsl.test.ts @@ -392,6 +392,40 @@ describe('WSL availability cache', () => { }) }) + // Why this site matters more than the other wsl.exe spawns (#16463): ENOENT is + // deliberately non-retryable here, so a spawn that failed only because the + // inherited cwd had been deleted was cached as "WSL is not installed" on the + // 10-minute definitive TTL with exponential backoff. Git kept working and Orca + // reported WSL unavailable -- a worse state than the bug being fixed. Naming + // the directory is what keeps ENOENT meaning "wsl.exe is not on PATH". + it('names an explicit spawn directory on both probes, so no deleted cwd can read as ENOENT', async () => { + execFileSyncMock.mockReturnValueOnce('') + execFileMock.mockImplementation((_command, _args, _options, callback) => { + callback(null, '', '') + }) + + withPlatform('win32', () => { + expect(isWslAvailable()).toBe(true) + }) + expect(execFileSyncMock).toHaveBeenCalledWith( + 'wsl.exe', + ['--status'], + expect.objectContaining({ cwd: expect.any(String) }) + ) + + // The two probes share one cache, so a false ENOENT from either poisons both. + _resetWslCachesForTests() + await withPlatformAsync('win32', async () => { + await expect(isWslAvailableAsync()).resolves.toBe(true) + }) + expect(execFileMock).toHaveBeenCalledWith( + 'wsl.exe', + ['--status'], + expect.objectContaining({ cwd: expect.any(String) }), + expect.any(Function) + ) + }) + it('shares one wsl.exe spawn between concurrent async probes', async () => { execFileMock.mockImplementation((_command, _args, _options, callback) => { setTimeout(() => callback(null, '', ''), 0) diff --git a/src/main/wsl.ts b/src/main/wsl.ts index 59e74a7f4df..579031de934 100644 --- a/src/main/wsl.ts +++ b/src/main/wsl.ts @@ -7,6 +7,7 @@ import { _setWslAvailabilityCacheForTests, dropStaleWslAvailabilityFailure } from './wsl-availability' +import { resolveWslInteropSpawnCwd } from './wsl-interop-spawn-directory' import { _resetRunningWslDistroCacheForTests, resolveRunningWslDistros @@ -76,7 +77,8 @@ export function wslUncDirectoryExists(uncPath: string): boolean | null { const stdout = execFileSync('wsl.exe', getWslDirectoryProbeArgs(info), { stdio: ['pipe', 'pipe', 'pipe'], timeout: 5000, - encoding: 'utf8' + encoding: 'utf8', + cwd: resolveWslInteropSpawnCwd() }) return parseWslDirectoryProbeOutput(stdout) } catch { @@ -93,7 +95,8 @@ export function wslUncDirectoryExistsAsync(uncPath: string): Promise { - execFile('wsl.exe', getWslDirectoryProbeArgs(info), { timeout: 5000 }, (_error, stdout) => { + const probeOpts = { timeout: 5000, cwd: resolveWslInteropSpawnCwd() } + execFile('wsl.exe', getWslDirectoryProbeArgs(info), probeOpts, (_error, stdout) => { // Why: wsl.exe uses numeric exits for both guest results and host failures; only the guest marker is authoritative. resolve(parseWslDirectoryProbeOutput(stdout)) }) @@ -181,7 +184,8 @@ export function listWslDistros(): string[] { const output = execFileSync('wsl.exe', ['--list', '--quiet'], { encoding: 'utf-8', stdio: ['pipe', 'pipe', 'pipe'], - timeout: 5000 + timeout: 5000, + cwd: resolveWslInteropSpawnCwd() }) return cacheWslDistroList(parseWslDistros(output), probeSequence) } catch { @@ -283,7 +287,8 @@ export function getWslHome(distro: string): string | null { const home = execFileSync('wsl.exe', ['-d', distro, '--exec', 'bash', '-c', 'echo $HOME'], { encoding: 'utf-8', stdio: ['pipe', 'pipe', 'pipe'], - timeout: 5000 + timeout: 5000, + cwd: resolveWslInteropSpawnCwd() }).trim() if (!home || !home.startsWith('/')) { @@ -382,7 +387,13 @@ function execFileUtf8(command: string, args: string[], env?: NodeJS.ProcessEnv): execFile( command, args, - { encoding: 'utf-8', env, timeout: 5000, windowsHide: true }, + { + encoding: 'utf-8', + env, + timeout: 5000, + windowsHide: true, + cwd: resolveWslInteropSpawnCwd() + }, (error, stdout) => { if (error) { reject(error) diff --git a/src/main/wsl/__fixtures__/wsl-probe-failure-swallow-allowlist.txt b/src/main/wsl/__fixtures__/wsl-probe-failure-swallow-allowlist.txt index db6392e21f3..f0a4c4f1082 100644 --- a/src/main/wsl/__fixtures__/wsl-probe-failure-swallow-allowlist.txt +++ b/src/main/wsl/__fixtures__/wsl-probe-failure-swallow-allowlist.txt @@ -23,3 +23,9 @@ main/ipc/preflight-command-exec.ts main/ipc/preflight-test-harness.ts main/ipc/preflight-wsl-agent-detection.ts main/wsl.ts +# Scanned only because the filename starts with `wsl`; it answers nothing about a +# distro. The swallow is a `statSync` on a LOCAL WINDOWS directory, and only the +# positive answer is memoized -- and re-validated on every call, which is the +# point of the module (#16463). A failed stat drops to the next candidate for +# that one call and is re-asked on the next, so there is no value to pin. +main/wsl-interop-spawn-directory.ts From 81972689560b8af2759a9b67ac6f1c8029809b90 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 02:23:45 -0700 Subject: [PATCH 058/398] fix(pty,remote): close the pty master fd leak, and two remote-terminal defects (#17914) * fix(pty,remote): close the pty master fd leak and two remote-terminal defects so on Linux every later child of the process -- both later pty children and plain child_process spawns -- inherits it and keeps the /dev/pts device alive. Measured on Linux with stock node-pty 1.1.0: master fd flags 0404002 (cloexec=false), and 17 -> /dev/pts/ptmx present in both a later pty child's /proc/self/fd and a later child_process child's. Extend the existing node-pty patch with pty_cloexec() on both PtyFork spawn paths; after the patch the flags read 02404002 (cloexec=true) and neither child sees the master. This covers the app and terminal daemon only -- the SSH relay installs node-pty from npm on the remote host, so it stays exposed (see the report). rejecting inspection as a renderer-global unhandledrejection, which an unreachable runtime produced on every cadence tick. path cleared the close intent for it exactly like a dropped connection, so a host that keeps republishing the dead surface re-materialized the pane the user just closed. Keep that intent and drop its TTL. Also route the banner's "Remote terminal was closed." line through translate() so it stops mixing English into a localized banner. * test(pty,remote): make the fd-leak evidence positive and size the close intent to its RPC The Linux 'does not hand an earlier pty master to a later pty child' case only asserted that ptmx was absent from the captured listing, so any run that produced no listing passed without inspecting a single fd. Block the child on stdin, emit a sentinel, and assert both the sentinel and a real /dev/pts fd row before the negative assertion. Verified in node:24-bookworm: passes with the patch, and with pty_cloexec() reverted it fails on four inherited /dev/pts/ptmx rows. The close intent's TTL was a 10s literal while the close RPC that can still answer tab_not_found had its own 15s literal. A host that answered slowly while republishing the surface had its intent evicted by the republish path's own pending-check, so makeWebSessionCloseIntentDurable found nothing to flip and #9194 reproduced. Derive the TTL from the shared session.tabs RPC timeout so the two cannot cross, with an invariant test and a regression test for the slow answer. --- config/patches/node-pty@1.1.0.patch | 61 +++++++++- pnpm-lock.yaml | 6 +- src/main/daemon/node-pty-fd-leak.test.ts | 107 +++++++++++++++++- .../terminal-pane/TerminalErrorToast.tsx | 12 ++ ...ection-queue-rejection-containment.test.ts | 72 ++++++++++++ .../agent-process-inspection-queue.ts | 18 ++- ...l-error-remote-closed-localization.test.ts | 24 ++++ src/renderer/src/i18n/locales/en.json | 3 +- src/renderer/src/i18n/locales/es.json | 3 +- src/renderer/src/i18n/locales/ja.json | 3 +- src/renderer/src/i18n/locales/ko.json | 3 +- src/renderer/src/i18n/locales/zh.json | 3 +- ...runtime-session-tab-activate-close.test.ts | 79 +++++++++++++ .../web-runtime-session-tab-lifecycle.ts | 21 +++- .../runtime/web-session-close-intent.test.ts | 19 +++- .../src/runtime/web-session-close-intent.ts | 37 +++++- .../runtime/web-session-tab-rpc-timeout.ts | 5 + 17 files changed, 445 insertions(+), 31 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/agent-process-inspection-queue-rejection-containment.test.ts create mode 100644 src/renderer/src/components/terminal-pane/terminal-error-remote-closed-localization.test.ts create mode 100644 src/renderer/src/runtime/web-session-tab-rpc-timeout.ts diff --git a/config/patches/node-pty@1.1.0.patch b/config/patches/node-pty@1.1.0.patch index 9ee2ebd39b4..348ce6ef7ce 100644 --- a/config/patches/node-pty@1.1.0.patch +++ b/config/patches/node-pty@1.1.0.patch @@ -176,7 +176,7 @@ index 181ccabbbe9c4948a9725fb1db907a68e9de01fc..67f31facf85562b67adbfbd04ce28ddd process.send!({ consoleProcessList }); process.exit(0); diff --git a/src/unix/pty.cc b/src/unix/pty.cc -index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..383df0c9c48355547c65e6c9bbba593d15c4dd44 100644 +index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..2ae787c5bd4f3eba470584dc658a01a52c690e0a 100644 --- a/src/unix/pty.cc +++ b/src/unix/pty.cc @@ -23,7 +23,9 @@ @@ -215,7 +215,17 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..383df0c9c48355547c65e6c9bbba593d /* Some platforms name VWERASE and VDISCARD differently */ #if !defined(VWERASE) && defined(VWERSE) #define VWERASE VWERSE -@@ -237,13 +258,23 @@ pty_getproc(int, char *); +@@ -228,6 +249,9 @@ Napi::Value PtyGetProc(const Napi::CallbackInfo& info); + static int + pty_nonblock(int); + ++static int ++pty_cloexec(int); ++ + #if defined(__APPLE__) + static char * + pty_getproc(int); +@@ -237,13 +261,23 @@ pty_getproc(int, char *); #endif #if defined(__APPLE__) || defined(__OpenBSD__) @@ -240,7 +250,7 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..383df0c9c48355547c65e6c9bbba593d #endif struct DelBuf { -@@ -367,10 +398,11 @@ Napi::Value PtyFork(const Napi::CallbackInfo& info) { +@@ -367,14 +401,18 @@ Napi::Value PtyFork(const Napi::CallbackInfo& info) { argv[i + 3] = strdup(arg.c_str()); } @@ -256,7 +266,48 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..383df0c9c48355547c65e6c9bbba593d } if (pty_nonblock(master) == -1) { throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); -@@ -684,15 +716,73 @@ pty_getproc(int fd, char *tty) { + } ++ if (pty_cloexec(master) == -1) { ++ throw Napi::Error::New(napiEnv, "Could not set master fd to close-on-exec."); ++ } + #else + int argc = argv_.Length(); + int argl = argc + 2; +@@ -445,6 +483,9 @@ Napi::Value PtyFork(const Napi::CallbackInfo& info) { + if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } ++ if (pty_cloexec(master) == -1) { ++ throw Napi::Error::New(napiEnv, "Could not set master fd to close-on-exec."); ++ } + } + #endif + +@@ -586,6 +627,23 @@ pty_nonblock(int fd) { + return fcntl(fd, F_SETFL, flags | O_NONBLOCK); + } + ++/** ++ * Orca: close-on-exec FD ++ * ++ * forkpty()/posix_openpt() have no atomic O_CLOEXEC, so a master left without ++ * FD_CLOEXEC is inherited by every later child of this process -- including ++ * later pty children -- which keeps its /dev/pts device and buffers alive long ++ * after its own session ends (#8362). ++ */ ++ ++static int ++pty_cloexec(int fd) { ++ int flags = fcntl(fd, F_GETFD); ++ if (flags == -1) return -1; ++ if (flags & FD_CLOEXEC) return 0; ++ return fcntl(fd, F_SETFD, flags | FD_CLOEXEC); ++} ++ + /** + * pty_getproc + * Taken from tmux. +@@ -684,15 +742,73 @@ pty_getproc(int fd, char *tty) { #endif #if defined(__APPLE__) @@ -332,7 +383,7 @@ index 7b4b9e1f990fbf95b51528bb56dc9717f5b87532..383df0c9c48355547c65e6c9bbba593d for (; count < 3; count++) { low_fds[count] = posix_openpt(O_RDWR); -@@ -706,80 +796,118 @@ pty_posix_spawn(char** argv, char** env, +@@ -706,80 +822,118 @@ pty_posix_spawn(char** argv, char** env, POSIX_SPAWN_SETSID; *master = posix_openpt(O_RDWR); if (*master == -1) { diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 5b5b4c054a0..1abec6c0590 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -115,7 +115,7 @@ patchedDependencies: '@xterm/addon-webgl@0.20.0-beta.299': 94687e89a0115e6e6aa102837f986debdc029c091527ee5eb4a4e17ceaf9473e '@xterm/xterm@6.1.0-beta.303': 98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d lint-staged@16.4.0: 7333b3837f80a7fbd045964db6d76ba4fc118e49134bdbabb00585b6b7b60673 - node-pty@1.1.0: 572a46f539dd9da26e259702da974e1e693329e299625c97eb7c28e4e642500e + node-pty@1.1.0: 40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e importers: @@ -156,7 +156,7 @@ importers: version: 3.3.1 node-pty: specifier: ^1.1.0 - version: 1.1.0(patch_hash=572a46f539dd9da26e259702da974e1e693329e299625c97eb7c28e4e642500e) + version: 1.1.0(patch_hash=40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e) posthog-node: specifier: ^5.33.3 version: 5.33.3 @@ -12194,7 +12194,7 @@ snapshots: node-int64@0.4.0: {} - node-pty@1.1.0(patch_hash=572a46f539dd9da26e259702da974e1e693329e299625c97eb7c28e4e642500e): + node-pty@1.1.0(patch_hash=40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e): dependencies: node-addon-api: 7.1.1 diff --git a/src/main/daemon/node-pty-fd-leak.test.ts b/src/main/daemon/node-pty-fd-leak.test.ts index 91958b49975..f021f7d48df 100644 --- a/src/main/daemon/node-pty-fd-leak.test.ts +++ b/src/main/daemon/node-pty-fd-leak.test.ts @@ -1,5 +1,6 @@ -import { execFileSync } from 'node:child_process' -import { existsSync, renameSync } from 'node:fs' +import { execFileSync, spawn } from 'node:child_process' +import { once } from 'node:events' +import { existsSync, readdirSync, readFileSync, readlinkSync, renameSync } from 'node:fs' import { setTimeout as delay } from 'node:timers/promises' import * as pty from 'node-pty' import { describe, expect, it } from 'vitest' @@ -90,3 +91,105 @@ describeOnDarwin('node-pty macOS spawn fd handling', () => { expect(after - before).toBe(0) }, 15000) }) + +// Linux is the only platform where node-pty takes the forkpty() path, which has no atomic +// O_CLOEXEC. /proc is what makes the inheritance observable, so the assertions live here. +const describeOnLinux = process.platform === 'linux' ? describe : describe.skip + +const O_CLOEXEC = 0o2000000 + +const LISTING_READY = '__fd_listing_ready__' + +function ptyMasterFd(term: pty.IPty): number { + return (term as unknown as { fd: number }).fd +} + +function isCloseOnExec(fd: number): boolean { + const flags = /flags:\s*(\d+)/.exec(readFileSync(`/proc/self/fdinfo/${fd}`, 'utf8')) + expect(flags).toBeTruthy() + return (Number.parseInt(flags![1]!, 8) & O_CLOEXEC) !== 0 +} + +function openFdTargets(pid: number): string[] { + return readdirSync(`/proc/${pid}/fd`).map((entry) => { + try { + return readlinkSync(`/proc/${pid}/fd/${entry}`) + } catch { + return '' + } + }) +} + +describeOnLinux('node-pty Linux forkpty fd handling', () => { + it('marks pty masters close-on-exec so later children cannot inherit them', async () => { + const terms: pty.IPty[] = [] + let child: ReturnType | null = null + try { + for (let i = 0; i < 3; i++) { + terms.push( + pty.spawn('/bin/sh', ['-c', 'sleep 30'], { + name: 'xterm-256color', + cols: 80, + rows: 24, + cwd: process.cwd(), + env: { ...process.env, ORCA_FD_LEAK_TEST_INDEX: String(i) } + }) + ) + } + + // The masters this process owns must not survive an exec in any child it forks later. + expect(terms.map((term) => isCloseOnExec(ptyMasterFd(term)))).toEqual([true, true, true]) + + child = spawn('/bin/sh', ['-c', 'sleep 5'], { stdio: 'ignore' }) + await once(child, 'spawn') + const inherited = openFdTargets(child.pid!).filter((target) => target.includes('ptmx')) + expect(inherited).toEqual([]) + } finally { + child?.kill() + for (const term of terms) { + term.kill() + } + } + }, 15000) + + it('does not hand an earlier pty master to a later pty child', async () => { + const first = pty.spawn('/bin/sh', ['-c', 'sleep 30'], { + name: 'xterm-256color', + cols: 80, + rows: 24, + cwd: process.cwd(), + env: { ...process.env } + }) + try { + // Why the read: the child must not be able to run its listing before onData is armed, or an + // empty capture would satisfy the negative assertion without inspecting a single fd. + const second = pty.spawn( + '/bin/sh', + ['-c', `IFS= read -r _; printf '${LISTING_READY}\\n'; ls -l /proc/self/fd; exit 0`], + { + name: 'xterm-256color', + cols: 200, + rows: 24, + cwd: process.cwd(), + env: { ...process.env } + } + ) + let output = '' + second.onData((data) => { + output += data + }) + second.write('go\n') + await new Promise((resolve) => { + second.onExit(() => resolve()) + }) + await delay(100) + + // The listing is the evidence; assert it arrived before reading anything into its absence. + expect(output).toContain(LISTING_READY) + expect(output).toMatch(/\d+ -> \/dev\/pts\//) + expect(output).not.toMatch(/ptmx/) + } finally { + first.kill() + } + }, 15000) +}) diff --git a/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx b/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx index 11543b96ccc..dcb838007b9 100644 --- a/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx @@ -19,6 +19,10 @@ const STALE_DAEMON_CWD_MARKERS = [ ] // Thrown by ipc/pty.ts when a persisted pane owner can't be proven alive or dead (STA-3536). const PANE_OWNER_UNVERIFIED_MARKER = 'terminal_pane_owner_unverified' +// remote-runtime-pty-transport.ts surfaces this English literal as a wire-level marker, so it is +// translated here rather than at the source -- otherwise the banner mixes English with the +// localized chrome around it (#9194). +const REMOTE_TERMINAL_CLOSED_MARKER = 'Remote terminal was closed.' // Why one source: the test and replace forms must match the same token, and a lone /g regex carries // lastIndex state across .test() calls. Capture the leading boundary so replacement can restore it. const TERMINAL_HOST_GONE_SOURCE = '(^|[^a-z0-9_])terminal_host_gone(?=$|[^a-z0-9_])' @@ -127,6 +131,14 @@ export function humanizeTerminalError(error: string): string { 'Reconnecting this terminal — its output is being re-established. The session is still running.' ) ) + if (humanized.includes(REMOTE_TERMINAL_CLOSED_MARKER)) { + humanized = humanized.replaceAll(REMOTE_TERMINAL_CLOSED_MARKER, () => + translate( + 'auto.components.terminal.pane.TerminalErrorToast.remoteTerminalClosed', + 'Remote terminal was closed.' + ) + ) + } humanized = humanizeUnreattachableSession(humanized) if (!isExplainedTerminalError(humanized)) { return humanized diff --git a/src/renderer/src/components/terminal-pane/agent-process-inspection-queue-rejection-containment.test.ts b/src/renderer/src/components/terminal-pane/agent-process-inspection-queue-rejection-containment.test.ts new file mode 100644 index 00000000000..639c457016f --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-process-inspection-queue-rejection-containment.test.ts @@ -0,0 +1,72 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { + enqueueAgentProcessInspection, + resetAgentProcessInspectionQueueForTests +} from './agent-process-inspection-queue' + +// Node emits 'unhandledRejection' a turn after the microtask queue drains. +async function settleRejections(): Promise { + for (let index = 0; index < 4; index += 1) { + await new Promise((resolve) => setTimeout(resolve, 0)) + } +} + +async function collectUnhandledRejections(run: () => Promise): Promise { + const unhandled: unknown[] = [] + const onUnhandledRejection = (reason: unknown): void => { + unhandled.push(reason) + } + process.on('unhandledRejection', onUnhandledRejection) + try { + await run() + } finally { + process.off('unhandledRejection', onUnhandledRejection) + } + return unhandled +} + +describe('agent process inspection queue rejection containment', () => { + afterEach(() => { + resetAgentProcessInspectionQueueForTests() + }) + + it('contains an unreachable-runtime inspection failure instead of raising unhandledrejection', async () => { + const unhandled = await collectUnhandledRejections(async () => { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + run: () => + Promise.reject( + new Error( + "Error invoking remote method 'runtimeEnvironments:call': RemoteRuntimeClientError: Could not connect to the remote Orca runtime." + ) + ) + }) + await settleRejections() + }) + + expect(unhandled).toEqual([]) + }) + + it('keeps draining the queue after a rejecting inspection', async () => { + const ran: string[] = [] + const unhandled = await collectUnhandledRejections(async () => { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + run: () => Promise.reject(new Error('unreachable')) + }) + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + run: async () => { + ran.push('second') + } + }) + await settleRejections() + }) + + expect(unhandled).toEqual([]) + expect(ran).toEqual(['second']) + }) +}) diff --git a/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts b/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts index f7512ad8c60..6dc1dc4ccf2 100644 --- a/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts +++ b/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts @@ -63,12 +63,18 @@ function pumpInspectionQueue(): void { activeInspections += 1 inspectionStarts.push(now) - void next.run().finally(() => { - activeInspections = Math.max(0, activeInspections - 1) - if (inspectionQueue.length > 0) { - scheduleInspectionPump() - } - }) + // Why the catch before finally: an unreachable runtime rejects the inspection on a cadence, and a + // bare `.finally()` chain re-raises it as a renderer-global unhandledrejection. Coordinators own + // their own failure/backoff state, so the queue only has to keep its accounting running. + void next + .run() + .catch(() => {}) + .finally(() => { + activeInspections = Math.max(0, activeInspections - 1) + if (inspectionQueue.length > 0) { + scheduleInspectionPump() + } + }) if (inspectionQueue.length > 0) { scheduleInspectionPump() diff --git a/src/renderer/src/components/terminal-pane/terminal-error-remote-closed-localization.test.ts b/src/renderer/src/components/terminal-pane/terminal-error-remote-closed-localization.test.ts new file mode 100644 index 00000000000..12ff0dbda82 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/terminal-error-remote-closed-localization.test.ts @@ -0,0 +1,24 @@ +import { describe, expect, it, vi } from 'vitest' + +// Why a locale stand-in: the banner's own chrome is already translated, so the only way to see the +// mixed-language regression (#9194) is to render the message through a non-English catalog. +vi.mock('@/i18n/i18n', () => ({ + translate: (key: string, fallback: string) => + key === 'auto.components.terminal.pane.TerminalErrorToast.remoteTerminalClosed' + ? '远程终端已关闭。' + : fallback +})) + +import { humanizeTerminalError } from './TerminalErrorToast' + +describe('remote-closed terminal banner localization', () => { + it('translates the remote-closed line instead of pinning it to English', () => { + expect(humanizeTerminalError('Remote terminal was closed.')).toBe('远程终端已关闭。') + }) + + it('translates the line when it is accumulated with other errors', () => { + expect(humanizeTerminalError('Paste failed.\nRemote terminal was closed.')).toBe( + 'Paste failed.\n远程终端已关闭。' + ) + }) +}) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 99805af4c89..c7a66f2decc 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -3088,7 +3088,8 @@ "42b283ecfc": "Orca couldn't safely reconnect this terminal because the host couldn't verify its saved session. Orca left the saved session unchanged. Click Retry to try reconnecting now. If it still cannot reconnect, open a new terminal.", "e16012e31e": "The terminal daemon that owned this session exited, so the session and its scrollback could not be recovered. Open a new terminal to continue.", "sessionUnavailable": "Orca couldn't reattach to this pane's terminal session on the host. Open a new terminal to continue.", - "sourceRestoring": "Reconnecting this terminal — its output is being re-established. The session is still running." + "sourceRestoring": "Reconnecting this terminal — its output is being re-established. The session is still running.", + "remoteTerminalClosed": "Remote terminal was closed." }, "TerminalProcessExitOverlay": { "capacityTitle": "Git Bash console limit reached", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index be6e087060d..e55defa7518 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -2741,7 +2741,8 @@ "e4aa243f8c": "Reiniciar servicio", "a7e2fd2699": "abre un issue", "5c8ce20be6": "Si esto persiste, por favor", - "cc6d997c65": "Reinicia el servicio del terminal desde aquí para borrar el estado obsoleto." + "cc6d997c65": "Reinicia el servicio del terminal desde aquí para borrar el estado obsoleto.", + "remoteTerminalClosed": "La terminal remota se cerró." }, "TerminalProcessExitOverlay": { "capacityTitle": "Se alcanzó el límite de consolas de Git Bash", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 20e2fca6553..5d97966fca0 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -2741,7 +2741,8 @@ "e4aa243f8c": "デーモンを再起動します", "a7e2fd2699": "Issue を登録", "5c8ce20be6": "この状態が続く場合は、", - "cc6d997c65": "ここからターミナルデーモンを再起動して、古いデーモン状態をクリアします。" + "cc6d997c65": "ここからターミナルデーモンを再起動して、古いデーモン状態をクリアします。", + "remoteTerminalClosed": "リモートターミナルが閉じられました。" }, "TerminalProcessExitOverlay": { "capacityTitle": "Git Bash のコンソール上限に達しました", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index e9b9c48fcdf..795d359bce0 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -2746,7 +2746,8 @@ "e4aa243f8c": "데몬 재시작", "a7e2fd2699": "이슈 등록", "5c8ce20be6": "문제가 계속되면", - "cc6d997c65": "오래된 데몬 상태를 지우려면 여기에서 terminal 데몬을 다시 시작하세요." + "cc6d997c65": "오래된 데몬 상태를 지우려면 여기에서 terminal 데몬을 다시 시작하세요.", + "remoteTerminalClosed": "원격 터미널이 종료되었습니다." }, "TerminalProcessExitOverlay": { "capacityTitle": "Git Bash 콘솔 한도에 도달했습니다", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 49560989924..3aa8333b872 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -2756,7 +2756,8 @@ "e4aa243f8c": "重新启动守护进程", "a7e2fd2699": "提交议题", "5c8ce20be6": "如果这种情况持续存在,请", - "cc6d997c65": "从此处重新启动终端守护进程以清除失效的守护进程状态。" + "cc6d997c65": "从此处重新启动终端守护进程以清除失效的守护进程状态。", + "remoteTerminalClosed": "远程终端已关闭。" }, "TerminalProcessExitOverlay": { "capacityTitle": "已达到 Git Bash 控制台上限", diff --git a/src/renderer/src/runtime/web-runtime-session-tab-activate-close.test.ts b/src/renderer/src/runtime/web-runtime-session-tab-activate-close.test.ts index 6ed42c3c841..b535bd31836 100644 --- a/src/renderer/src/runtime/web-runtime-session-tab-activate-close.test.ts +++ b/src/renderer/src/runtime/web-runtime-session-tab-activate-close.test.ts @@ -9,6 +9,8 @@ import { recordWebSessionCloseIntent, resetWebSessionCloseIntentForTests } from './web-session-close-intent' +import { WEB_SESSION_TAB_RPC_TIMEOUT_MS } from './web-session-tab-rpc-timeout' +import { toHostSessionTabId } from './web-terminal-surface-id' import { ENVIRONMENT_ID, WORKTREE_ID, makeSnapshot } from './web-runtime-session-test-harness' const mocks = vi.hoisted(() => ({ @@ -305,6 +307,83 @@ describe('web runtime session tab actions', () => { ).resolves.toBe(outcome) }) + // #9194: a host can answer tab_not_found and still keep republishing the surface. The close + // intent is what hides the mirror, so letting it age out handed the user back a phantom pane + // whose handle is already gone -- and closing it again just restarted the same TTL loop. + it.each([ + ['tab_not_found', true], + ['runtime_rpc_timeout', false] + ])('keeps a %s close suppressed past the close-intent TTL: %s', async (code, stillPending) => { + const runtimeCall = vi + .fn() + .mockResolvedValueOnce({ id: 'close', ok: false, error: { code, message: code } }) + .mockResolvedValueOnce({ id: 'list', ok: true, result: makeSnapshot() }) + vi.stubGlobal('window', { api: { runtimeEnvironments: { call: runtimeCall } } }) + + await closeWebRuntimeSessionTab({ + worktreeId: WORKTREE_ID, + tabId: 'local-browser-unified', + reason: 'user' + }) + + const hostTabId = toHostSessionTabId('local-browser-unified') + expect( + isWebSessionCloseIntentPending( + { environmentId: ENVIRONMENT_ID }, + WORKTREE_ID, + hostTabId, + Date.now() + 60_000 + ) + ).toBe(stillPending) + }) + + // #9194, slow host: the close RPC can answer `tab_not_found` at any point up to its own timeout, + // and a host that still republishes the surface keeps querying the intent in the meantime. That + // query is what evicts an expired entry, so a TTL under the RPC timeout leaves nothing for the + // durable flip to reach and the pane the user closed comes back. + it('keeps a tab_not_found close suppressed when the host answers slower than the old TTL', async () => { + const startedAt = 1_700_000_000_000 + let clock = startedAt + const nowSpy = vi.spyOn(Date, 'now').mockImplementation(() => clock) + const owner = { environmentId: ENVIRONMENT_ID } + const hostTabIds = [toHostSessionTabId('local-browser-unified'), 'host-browser-unified'] + let pendingWhileHostRepublished: boolean[] = [] + try { + const runtimeCall = vi + .fn() + .mockImplementationOnce(() => { + clock = startedAt + WEB_SESSION_TAB_RPC_TIMEOUT_MS - 1 + pendingWhileHostRepublished = hostTabIds.map((hostTabId) => + isWebSessionCloseIntentPending(owner, WORKTREE_ID, hostTabId, clock) + ) + return Promise.resolve({ + id: 'close', + ok: false, + error: { code: 'tab_not_found', message: 'tab_not_found' } + }) + }) + .mockResolvedValueOnce({ id: 'list', ok: true, result: makeSnapshot() }) + vi.stubGlobal('window', { api: { runtimeEnvironments: { call: runtimeCall } } }) + + await expect( + closeWebRuntimeSessionTab({ + worktreeId: WORKTREE_ID, + tabId: 'local-browser-unified', + reason: 'user' + }) + ).resolves.toBe('unknown-tab') + + expect(pendingWhileHostRepublished).toEqual([true, true]) + expect( + hostTabIds.map((hostTabId) => + isWebSessionCloseIntentPending(owner, WORKTREE_ID, hostTabId, clock + 60_000) + ) + ).toEqual([true, true]) + } finally { + nowSpy.mockRestore() + } + }) + it('fails closed when reconnect routes a lifecycle close to an older host', async () => { const runtimeCall = vi .fn() diff --git a/src/renderer/src/runtime/web-runtime-session-tab-lifecycle.ts b/src/renderer/src/runtime/web-runtime-session-tab-lifecycle.ts index a7381f10d6a..56800f3cefb 100644 --- a/src/renderer/src/runtime/web-runtime-session-tab-lifecycle.ts +++ b/src/renderer/src/runtime/web-runtime-session-tab-lifecycle.ts @@ -6,11 +6,16 @@ import type { import { useAppStore } from '../store' import { hasRuntimeRpcErrorCode, unwrapRuntimeRpcResult } from './runtime-rpc-client' import { toRuntimeWorktreeSelector } from './runtime-worktree-selector' -import { clearWebSessionCloseIntent, recordWebSessionCloseIntent } from './web-session-close-intent' +import { + clearWebSessionCloseIntent, + makeWebSessionCloseIntentDurable, + recordWebSessionCloseIntent +} from './web-session-close-intent' import { clearWebSessionFocusIntentIfMatches, recordWebSessionFocusIntent } from './web-session-focus-intent' +import { WEB_SESSION_TAB_RPC_TIMEOUT_MS } from './web-session-tab-rpc-timeout' import { toHostSessionTabId } from './web-terminal-surface-id' import { captureRuntimeEnvironmentCall, @@ -134,7 +139,7 @@ async function callWebRuntimeSessionTabMethod( ? { reason: args.reason } : {}) }, - timeoutMs: 15_000 + timeoutMs: WEB_SESSION_TAB_RPC_TIMEOUT_MS }) const result = unwrapRuntimeRpcResult( response as RuntimeRpcResponse @@ -157,8 +162,16 @@ async function callWebRuntimeSessionTabMethod( if (activationHostTabId) { clearWebSessionFocusIntentIfMatches(intentOwner, args.worktreeId, activationHostTabId) } + // Why the split: only 'tab_not_found' is absence proof (see the outcome doc above). Restoring the + // mirror on it hands the user back a pane the host cannot close and whose handle is already gone + // (#9194), so keep the suppression and drop its TTL instead. Every other failure is a "not now". + const hostHasNoSuchTab = hasRuntimeRpcErrorCode(error, 'tab_not_found') for (const hostTabId of closeIntentTabIds) { - clearWebSessionCloseIntent(intentOwner, args.worktreeId, hostTabId) + if (hostHasNoSuchTab) { + makeWebSessionCloseIntentDurable(intentOwner, args.worktreeId, hostTabId) + } else { + clearWebSessionCloseIntent(intentOwner, args.worktreeId, hostTabId) + } } if (isLifecycleClose) { const { acceptReplayedWebSessionTabsSnapshot } = await import('./web-session-tabs-sync') @@ -171,6 +184,6 @@ async function callWebRuntimeSessionTabMethod( `[web-runtime-session] failed to ${isClose ? 'close' : 'activate'} tab:`, error instanceof Error ? error.message : String(error) ) - return hasRuntimeRpcErrorCode(error, 'tab_not_found') ? 'unknown-tab' : 'failed' + return hostHasNoSuchTab ? 'unknown-tab' : 'failed' } } diff --git a/src/renderer/src/runtime/web-session-close-intent.test.ts b/src/renderer/src/runtime/web-session-close-intent.test.ts index 92ab8dbca96..418a8544840 100644 --- a/src/renderer/src/runtime/web-session-close-intent.test.ts +++ b/src/renderer/src/runtime/web-session-close-intent.test.ts @@ -6,8 +6,10 @@ import { isWebSessionCloseIntentPending, reconcileWebSessionCloseIntents, recordWebSessionCloseIntent, - resetWebSessionCloseIntentForTests + resetWebSessionCloseIntentForTests, + WEB_SESSION_CLOSE_INTENT_TTL_MS } from './web-session-close-intent' +import { WEB_SESSION_TAB_RPC_TIMEOUT_MS } from './web-session-tab-rpc-timeout' const WT = 'repo::/wt' const OWNER = { environmentId: 'runtime-a', pairingRevision: 1 } @@ -26,7 +28,20 @@ describe('web session close intent', () => { it('expires a never-confirmed close', () => { recordWebSessionCloseIntent(OWNER, WT, 'host-tab-1', 1000) - expect(isWebSessionCloseIntentPending(OWNER, WT, 'host-tab-1', 12_000)).toBe(false) + expect( + isWebSessionCloseIntentPending( + OWNER, + WT, + 'host-tab-1', + 1000 + WEB_SESSION_CLOSE_INTENT_TTL_MS + 1 + ) + ).toBe(false) + }) + + // The durable flip in the tab_not_found path can only reach an entry that is still there, so the + // TTL must outlast the RPC that produces that answer. Two independent literals would drift. + it('outlives the tab RPC that can still answer tab_not_found', () => { + expect(WEB_SESSION_CLOSE_INTENT_TTL_MS).toBeGreaterThan(WEB_SESSION_TAB_RPC_TIMEOUT_MS) }) it('scopes intents by owner, pairing revision, and worktree', () => { diff --git a/src/renderer/src/runtime/web-session-close-intent.ts b/src/renderer/src/runtime/web-session-close-intent.ts index 012dfd81d26..5a80fdf87f3 100644 --- a/src/renderer/src/runtime/web-session-close-intent.ts +++ b/src/renderer/src/runtime/web-session-close-intent.ts @@ -1,10 +1,19 @@ // Why: closing a remote tab prunes the local mirror immediately for responsiveness, so stale pre-close snapshots must not rematerialize it. import { webSessionIntentOwnerKey, type WebSessionIntentOwner } from './web-session-intent-owner' +import { WEB_SESSION_TAB_RPC_TIMEOUT_MS } from './web-session-tab-rpc-timeout' -const CLOSE_INTENT_TTL_MS = 10_000 +/** + * Why derived rather than a literal: `makeWebSessionCloseIntentDurable` can only flip an entry that + * still exists, and the close RPC may answer `tab_not_found` at any point up to its own timeout. A + * TTL shorter than that timeout lets a republishing host's pending-check delete the entry mid-call, + * the durable flip then no-ops, and the pane the user closed comes back (#9194). + */ +const CLOSE_INTENT_ANSWER_GRACE_MS = 5_000 +export const WEB_SESSION_CLOSE_INTENT_TTL_MS = + WEB_SESSION_TAB_RPC_TIMEOUT_MS + CLOSE_INTENT_ANSWER_GRACE_MS -type CloseIntent = { recordedAt: number } +type CloseIntent = { recordedAt: number; durable: boolean } const pendingCloseByOwnerAndWorktree = new Map>() @@ -28,7 +37,27 @@ export function recordWebSessionCloseIntent( byTab = new Map() pendingCloseByOwnerAndWorktree.set(partitionKey, byTab) } - byTab.set(trimmed, { recordedAt: now }) + byTab.set(trimmed, { recordedAt: now, durable: byTab.get(trimmed)?.durable === true }) +} + +/** + * Why no TTL: `tab_not_found` is the host's definitive answer that it does not have this tab, yet a + * host can keep republishing the surface in its snapshot (#9194). Letting that intent age out + * re-materializes a pane whose handle is already gone, and the pane the user just closed comes back + * showing "Remote terminal was closed." with no way to dismiss it. The intent still clears the + * moment the surface leaves a snapshot, so a host that recovers the tab is never suppressed forever. + */ +export function makeWebSessionCloseIntentDurable( + owner: WebSessionIntentOwner, + worktreeId: string, + hostTabId: string +): void { + const intent = pendingCloseByOwnerAndWorktree + .get(closeIntentPartitionKey(owner, worktreeId)) + ?.get(hostTabId) + if (intent) { + intent.durable = true + } } export function isWebSessionCloseIntentPending( @@ -43,7 +72,7 @@ export function isWebSessionCloseIntentPending( if (!intent) { return false } - if (now - intent.recordedAt > CLOSE_INTENT_TTL_MS) { + if (!intent.durable && now - intent.recordedAt > WEB_SESSION_CLOSE_INTENT_TTL_MS) { byTab!.delete(hostTabId) if (byTab!.size === 0) { pendingCloseByOwnerAndWorktree.delete(partitionKey) diff --git a/src/renderer/src/runtime/web-session-tab-rpc-timeout.ts b/src/renderer/src/runtime/web-session-tab-rpc-timeout.ts new file mode 100644 index 00000000000..876ff56a9c0 --- /dev/null +++ b/src/renderer/src/runtime/web-session-tab-rpc-timeout.ts @@ -0,0 +1,5 @@ +/** + * The budget for a `session.tabs.*` RPC. Client-side suppression that has to outlive one of these + * calls (the close intent) derives its own lifetime from this, so the two cannot drift apart. + */ +export const WEB_SESSION_TAB_RPC_TIMEOUT_MS = 15_000 From 34999e328e03e42edd8f0ed2b78dde82edac221f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 02:49:09 -0700 Subject: [PATCH 059/398] fix(orcad): stop demanding a spawn-helper only macOS builds (#18122) node-pty declares the spawn-helper target inside binding.gyp's OS=="mac" block and pty.cc execs it only under __APPLE__. Asserting it on `!== 'win32'` made every Linux orcad boot degraded with spawn_helper_missing while its terminals worked fine. Route all four sites through one shared `usesNodePtySpawnHelper` predicate: the precondition verdict, the prebuilt slot install, the +x repair, and the prebuilds build script (which threw outright on a Linux slot build). Fixes #17844 --- config/scripts/build-orcad-prebuilds.mjs | 7 ++-- src/main/orcad/node-pty-prebuilt-slot.test.ts | 32 +++++++++++++--- src/main/orcad/node-pty-prebuilt-slot.ts | 5 ++- src/main/orcad/node-pty-precondition.test.ts | 37 ++++++++++++++++--- src/main/orcad/node-pty-precondition.ts | 5 ++- src/main/providers/local-pty-utils.ts | 4 +- src/shared/node-pty-spawn-helper.test.ts | 14 +++++++ src/shared/node-pty-spawn-helper.ts | 11 ++++++ 8 files changed, 96 insertions(+), 19 deletions(-) create mode 100644 src/shared/node-pty-spawn-helper.test.ts create mode 100644 src/shared/node-pty-spawn-helper.ts diff --git a/config/scripts/build-orcad-prebuilds.mjs b/config/scripts/build-orcad-prebuilds.mjs index efbafbe9a1b..2d818d5d255 100644 --- a/config/scripts/build-orcad-prebuilds.mjs +++ b/config/scripts/build-orcad-prebuilds.mjs @@ -177,10 +177,11 @@ function build() { copyFileSync(builtBinary, join(slotDir, 'pty.node')) console.log(`[orcad-prebuilds] stored ${slot}/pty.node`) - // Why spawn-helper ships too: on Unix node-pty posix_spawns build/Release/spawn-helper, + // Why spawn-helper ships too: on macOS node-pty posix_spawns build/Release/spawn-helper, // so a slot without it installs cleanly and then fails ENOENT the first time a user - // opens a terminal. Windows has no spawn-helper. - if (process.platform !== 'win32') { + // opens a terminal. binding.gyp builds the helper only under OS=="mac"; every other + // platform forks directly, so demanding one there fails a healthy Linux slot build. + if (process.platform === 'darwin') { const helperSource = join(dirname(builtBinary), 'spawn-helper') if (!existsSync(helperSource)) { throw new Error(`[orcad-prebuilds] spawn-helper missing at ${helperSource}`) diff --git a/src/main/orcad/node-pty-prebuilt-slot.test.ts b/src/main/orcad/node-pty-prebuilt-slot.test.ts index 00880eee1e5..85aafddc7a1 100644 --- a/src/main/orcad/node-pty-prebuilt-slot.test.ts +++ b/src/main/orcad/node-pty-prebuilt-slot.test.ts @@ -17,6 +17,14 @@ const LINUX_GLIBC: NativeHostAbi = { nodeAbi: '127' } +const DARWIN_ARM64: NativeHostAbi = { + platform: 'darwin', + arch: 'arm64', + libc: 'none', + glibcVersion: null, + nodeAbi: '127' +} + const dirs: string[] = [] const temp = (): string => { const dir = mkdtempSync(join(tmpdir(), 'orcad-slot-')) @@ -48,18 +56,32 @@ describe('resolveOrcadPrebuildsDir', () => { }) describe('installPrebuiltSlot', () => { - it('installs the slot binary and spawn-helper into build/Release', () => { + it('installs the slot binary and spawn-helper into build/Release on macOS', () => { + const prebuilds = temp() + const nodePtyDir = temp() + stageSlot(prebuilds, 'darwin-arm64') + + const outcome = installPrebuiltSlot({ abi: DARWIN_ARM64, nodePtyDir, prebuildsDir: prebuilds }) + + expect(outcome).toEqual({ installed: true, slot: 'darwin-arm64', spawnHelper: true }) + expect(existsSync(join(nodePtyDir, 'build', 'Release', 'pty.node'))).toBe(true) + // Without the executable bit every spawn fails EACCES at the moment a user opens a terminal. + const helper = statSync(join(nodePtyDir, 'build', 'Release', 'spawn-helper')) + expect(helper.mode & 0o111).not.toBe(0) + }) + + it('installs a Linux slot without claiming a spawn-helper it never execs', () => { + // node-pty builds spawn-helper only under binding.gyp's OS=="mac"; reporting one off + // macOS is what made every Linux orcad boot degraded on spawn_helper_missing (#17844). const prebuilds = temp() const nodePtyDir = temp() stageSlot(prebuilds, 'linux-x64-glibc') const outcome = installPrebuiltSlot({ abi: LINUX_GLIBC, nodePtyDir, prebuildsDir: prebuilds }) - expect(outcome).toEqual({ installed: true, slot: 'linux-x64-glibc', spawnHelper: true }) + expect(outcome).toEqual({ installed: true, slot: 'linux-x64-glibc', spawnHelper: false }) expect(existsSync(join(nodePtyDir, 'build', 'Release', 'pty.node'))).toBe(true) - // Without the executable bit every spawn fails EACCES at the moment a user opens a terminal. - const helper = statSync(join(nodePtyDir, 'build', 'Release', 'spawn-helper')) - expect(helper.mode & 0o111).not.toBe(0) + expect(existsSync(join(nodePtyDir, 'build', 'Release', 'spawn-helper'))).toBe(false) }) it('will not load a glibc slot on a musl host', () => { diff --git a/src/main/orcad/node-pty-prebuilt-slot.ts b/src/main/orcad/node-pty-prebuilt-slot.ts index d0623689902..7dcdebba3da 100644 --- a/src/main/orcad/node-pty-prebuilt-slot.ts +++ b/src/main/orcad/node-pty-prebuilt-slot.ts @@ -16,6 +16,7 @@ import { chmodSync, copyFileSync, existsSync, mkdirSync, readFileSync } from 'node:fs' import { dirname, join } from 'node:path' import process from 'node:process' +import { usesNodePtySpawnHelper } from '../../shared/node-pty-spawn-helper' import { nativeSlotName, type NativeHostAbi } from './native-host-abi' export type PrebuiltSlotManifest = { @@ -103,11 +104,11 @@ export function installPrebuiltSlot(options: { mkdirSync(releaseDir, { recursive: true }) copyFileSync(source, join(releaseDir, 'pty.node')) - // Why this matters as much as pty.node: on Unix node-pty posix_spawns + // Why this matters as much as pty.node: on macOS node-pty posix_spawns // build/Release/spawn-helper. Without it every spawn fails with ENOENT at the moment // a user opens a terminal, long after the "install succeeded" line. let spawnHelper = false - if (options.abi.platform !== 'win32') { + if (usesNodePtySpawnHelper(options.abi.platform)) { const helperSource = join(prebuildsDir, slot, 'spawn-helper') if (existsSync(helperSource)) { const helperDest = join(releaseDir, 'spawn-helper') diff --git a/src/main/orcad/node-pty-precondition.test.ts b/src/main/orcad/node-pty-precondition.test.ts index fadb144d1ff..76d842a9456 100644 --- a/src/main/orcad/node-pty-precondition.test.ts +++ b/src/main/orcad/node-pty-precondition.test.ts @@ -31,10 +31,10 @@ const realNodePtyLoads = ((): boolean => { if (!existsSync(REAL_PTY_NODE)) { return false } - // Why spawn-helper too: a slot without it is legitimately 'degraded', so a test that - // expects 'ok' has an unsatisfiable premise on a host that lacks it. CI has the - // binding but not the helper, which is what made the previous gate insufficient. - if (process.platform !== 'win32' && !existsSync(REAL_SPAWN_HELPER)) { + // Why spawn-helper too: on macOS a slot without it is legitimately 'degraded', so a + // test that expects 'ok' has an unsatisfiable premise on a host that lacks it. Only + // macOS builds the helper, so gating other platforms on it never lets them run. + if (process.platform === 'darwin' && !existsSync(REAL_SPAWN_HELPER)) { return false } const probe = spawnSync(process.execPath, ['-e', `require(${JSON.stringify(REAL_PTY_NODE)})`], { @@ -240,8 +240,8 @@ describe('checkNodePtyPrecondition', () => { // Why gated on the real binding: this asserts a LOAD outcome, so it needs a pty.node // built for the Node ABI. CI's shard never runs ensure-native-runtime, so the copy - // ENOENT'd there. - it.runIf(process.platform !== 'win32' && realNodePtyLoads)( + // ENOENT'd there. macOS only — it is the only platform that execs spawn-helper. + it.runIf(process.platform === 'darwin' && realNodePtyLoads)( 'degrades rather than blocks when only spawn-helper is missing', () => { // node-pty posix_spawns spawn-helper, so this host loads fine and then fails ENOENT @@ -258,6 +258,31 @@ describe('checkNodePtyPrecondition', () => { } ) + // Same staging as above, read through a Linux ABI: node-pty builds spawn-helper only + // under binding.gyp's OS=="mac", so demanding one here called every healthy Linux + // orcad degraded while its terminals worked (#17844). Gated on a loadable binding for + // the same reason as the macOS case; the ABI is what makes it a Linux verdict. + it.runIf(process.platform !== 'win32' && realNodePtyLoads)( + 'does not call a Linux host degraded over a spawn-helper it never execs', + () => { + const dir = stageNodePty() + cpSync( + join(REAL_NODE_PTY, 'build', 'Release', 'pty.node'), + join(dir, 'build', 'Release', 'pty.node') + ) + expect(existsSync(join(dir, 'build', 'Release', 'spawn-helper'))).toBe(false) + + const verdict = checkNodePtyPrecondition({ + nodePtyDir: dir, + prebuildsDir: null, + abi: { platform: 'linux', arch: 'x64', libc: 'glibc', glibcVersion: '2.31', nodeAbi: '127' } + }) + + expect(verdict).toMatchObject({ status: 'ok', slot: 'linux-x64-glibc' }) + expect(verdict.reason).toBeUndefined() + } + ) + // Why split: the "ok" half needs a REAL loadable pty.node, which only exists after // `ensure-native-runtime --runtime=node`. CI's shard runs vitest directly, so copying // from node_modules ENOENT'd there. Slot *placement* is the logic worth checking on diff --git a/src/main/orcad/node-pty-precondition.ts b/src/main/orcad/node-pty-precondition.ts index 7d633bd0f9a..ff41a14cf81 100644 --- a/src/main/orcad/node-pty-precondition.ts +++ b/src/main/orcad/node-pty-precondition.ts @@ -19,6 +19,7 @@ import { existsSync, accessSync, constants } from 'node:fs' import { dirname, join } from 'node:path' import process from 'node:process' import { runProcessSync, type ProcessResult } from '../../shared/child-process/run-process' +import { usesNodePtySpawnHelper } from '../../shared/node-pty-spawn-helper' import type { RuntimeTerminalUnavailableReason } from '../../shared/runtime-types' import { buildToolchainProbeCommand, @@ -302,12 +303,12 @@ export function checkNodePtyPrecondition( } } - // Loaded. The remaining way terminals fail is spawn-time: node-pty posix_spawns + // Loaded. The remaining way terminals fail is spawn-time: on macOS node-pty posix_spawns // build/Release/spawn-helper, and a missing one turns every terminal.create into ENOENT // on a host that otherwise looks healthy. That is a degradation, not a boot blocker. const outcome = readNodePtyProbeOutcome(result) const loadedDir = outcome.kind === 'loaded' ? outcome.loadedDir : null - if (abi.platform !== 'win32') { + if (usesNodePtySpawnHelper(abi.platform)) { const helper = join(loadedDir || join(nodePtyDir, 'build', 'Release'), 'spawn-helper') if (!isExecutableFile(helper)) { return { diff --git a/src/main/providers/local-pty-utils.ts b/src/main/providers/local-pty-utils.ts index 5649db65534..735393449f2 100644 --- a/src/main/providers/local-pty-utils.ts +++ b/src/main/providers/local-pty-utils.ts @@ -1,6 +1,7 @@ import { basename, isAbsolute, join } from 'node:path' import { existsSync, accessSync, statSync, chmodSync, constants as fsConstants } from 'node:fs' import type * as pty from 'node-pty' +import { usesNodePtySpawnHelper } from '../../shared/node-pty-spawn-helper' import { hostReportsChildExitStatus, wrapShellSpawnForMacosTccAttribution @@ -82,9 +83,10 @@ export function resolveUnixShellPath(shellPath: string): string { * Why: when Electron packages the app via asar, the native spawn-helper * binary may lose its +x permission. This function detects and repairs * that so pty.spawn() does not fail with EACCES on first launch. + * macOS only — no other platform builds or execs the helper. */ export function ensureNodePtySpawnHelperExecutable(): void { - if (didEnsureSpawnHelperExecutable || process.platform === 'win32') { + if (didEnsureSpawnHelperExecutable || !usesNodePtySpawnHelper(process.platform)) { return } didEnsureSpawnHelperExecutable = true diff --git a/src/shared/node-pty-spawn-helper.test.ts b/src/shared/node-pty-spawn-helper.test.ts new file mode 100644 index 00000000000..4562c2bc9b7 --- /dev/null +++ b/src/shared/node-pty-spawn-helper.test.ts @@ -0,0 +1,14 @@ +import { describe, expect, it } from 'vitest' +import { usesNodePtySpawnHelper } from './node-pty-spawn-helper' + +describe('usesNodePtySpawnHelper', () => { + it('is macOS only', () => { + // The predicate this file exists for: node-pty's binding.gyp declares the + // spawn-helper target inside OS=="mac". Reading it as "every non-Windows platform" + // is what reported spawn_helper_missing on healthy Linux hosts (#17844). + expect(usesNodePtySpawnHelper('darwin')).toBe(true) + for (const platform of ['linux', 'win32', 'freebsd', 'openbsd', 'sunos', 'aix']) { + expect(usesNodePtySpawnHelper(platform)).toBe(false) + } + }) +}) diff --git a/src/shared/node-pty-spawn-helper.ts b/src/shared/node-pty-spawn-helper.ts new file mode 100644 index 00000000000..c3e40df101a --- /dev/null +++ b/src/shared/node-pty-spawn-helper.ts @@ -0,0 +1,11 @@ +/** + * Whether node-pty execs its `spawn-helper` binary on a platform. + * + * Only macOS: binding.gyp declares the `spawn-helper` target inside `OS=="mac"`, and + * `src/unix/pty.cc` reads the helper path only under `#if defined(__APPLE__)`. Every + * other platform forks directly, so requiring the helper there calls a working host + * broken. + */ +export function usesNodePtySpawnHelper(platform: string): boolean { + return platform === 'darwin' +} From 0da52453a75cc75b369a0a71969780625921d485 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 02:55:38 -0700 Subject: [PATCH 060/398] fix(settings): surface why CLI registration failed (#18125) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Settings CLI panel treated every resolved `cli:install` as a success, so a refusal that arrives as data (conflict, missing launcher, unreadable Windows PATH) produced a green "Registered `orca` in PATH." toast while the switch stayed off. A thrown refusal fared little better: the raw Electron `Error invoking remote method 'cli:install': ...` string went into a toast that then disappeared, leaving the panel indistinguishable from "not yet installed". Inspect the returned status with the predicate the onboarding and agent-skill flows already use (`state !== 'installed'`), unwrap the IPC transport prefix off thrown installer messages, and persist the existing main-process reason inline per STYLEGUIDE (toasts disappear; errors the user must act on stay inline). No new error taxonomy — the reasons already carry path and remedy; a conflict status, which names the path but not the remedy, gets the installer's own remedy sentence. Closes #3952 --- .../CliSection.install-failure.test.tsx | 154 ++++++++++++++++++ .../src/components/settings/CliSection.tsx | 113 +++++-------- .../settings/cli-install-failure.test.ts | 108 ++++++++++++ .../settings/cli-install-failure.ts | 40 +++++ .../settings/use-cli-registration-actions.ts | 127 +++++++++++++++ src/renderer/src/i18n/locales/en.json | 4 +- 6 files changed, 472 insertions(+), 74 deletions(-) create mode 100644 src/renderer/src/components/settings/CliSection.install-failure.test.tsx create mode 100644 src/renderer/src/components/settings/cli-install-failure.test.ts create mode 100644 src/renderer/src/components/settings/cli-install-failure.ts create mode 100644 src/renderer/src/components/settings/use-cli-registration-actions.ts diff --git a/src/renderer/src/components/settings/CliSection.install-failure.test.tsx b/src/renderer/src/components/settings/CliSection.install-failure.test.tsx new file mode 100644 index 00000000000..414a5ace240 --- /dev/null +++ b/src/renderer/src/components/settings/CliSection.install-failure.test.tsx @@ -0,0 +1,154 @@ +// @vitest-environment happy-dom + +import { act, cleanup, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { getDefaultSettings } from '../../../../shared/constants' +import type { CliInstallStatus } from '../../../../shared/cli-install-types' +import { CliSection } from './CliSection' + +const toasts = vi.hoisted(() => ({ error: vi.fn(), success: vi.fn() })) +const dialog = vi.hoisted(() => ({ + props: null as null | { onInstall: () => Promise; open: boolean } +})) + +vi.mock('sonner', () => ({ toast: toasts })) + +vi.mock('@/hooks/useInstalledAgentSkills', () => ({ + GLOBAL_AGENT_SKILL_SOURCE_KINDS: ['global'], + useInstalledAgentSkill: () => ({ + installed: false, + loading: false, + error: null, + refresh: vi.fn() + }) +})) + +vi.mock('@/hooks/useActiveProjectSkillRuntime', () => ({ + useActiveProjectSkillRuntime: () => ({ canUseLocalSkillFreshness: true }) +})) + +vi.mock('./AgentSkillSetupPanel', () => ({ + AgentSkillSetupPanel: () =>
+})) + +vi.mock('./WslCliRegistration', () => ({ WslCliRegistration: () => null })) + +vi.mock('./CliRegistrationDialog', () => ({ + CliRegistrationDialog: function CliRegistrationDialog(props: { + onInstall: () => Promise + open: boolean + }) { + dialog.props = props + return null + } +})) + +function notInstalledStatus(overrides: Partial = {}): CliInstallStatus { + return { + platform: 'darwin', + commandName: 'orca', + commandPath: '/usr/local/bin/orca', + pathDirectory: '/usr/local/bin', + pathConfigured: true, + launcherPath: '/Applications/Orca.app/Contents/Resources/bin/orca', + installMethod: 'symlink', + supported: true, + state: 'not_installed', + currentTarget: null, + unsupportedReason: null, + detail: 'Register /usr/local/bin/orca to use Orca from the terminal.', + ...overrides + } +} + +async function renderCliSectionAndInstall(install: () => Promise): Promise { + Object.assign(window, { + api: { + cli: { + getInstallStatus: vi.fn().mockResolvedValue(notInstalledStatus()), + getWslInstallStatus: vi.fn(), + install: vi.fn(install), + remove: vi.fn() + }, + shell: { openPath: vi.fn() } + } + }) + + render() + await screen.findByRole('switch') + await act(async () => { + await dialog.props?.onInstall() + }) +} + +afterEach(() => { + cleanup() + dialog.props = null + toasts.error.mockReset() + toasts.success.mockReset() +}) + +describe('CliSection install failure surfacing', () => { + it('shows the thrown conflict reason and its remedy instead of a success toast', async () => { + await renderCliSectionAndInstall(async () => { + throw new Error( + "Error invoking remote method 'cli:install': Error: Refusing to replace non-Orca " + + 'command at /usr/local/bin/orca. Remove it and register again if it is no longer needed.' + ) + }) + + const alert = screen.getByRole('alert') + expect(alert.textContent).toContain('Failed to register `orca` in PATH.') + expect(alert.textContent).toContain( + 'Refusing to replace non-Orca command at /usr/local/bin/orca.' + ) + expect(alert.textContent).toContain('Remove it and register again if it is no longer needed.') + // The Electron transport wrapper must not leak into the panel. + expect(alert.textContent).not.toContain('invoking remote method') + expect(toasts.success).not.toHaveBeenCalled() + expect(toasts.error).toHaveBeenCalledTimes(1) + }) + + it('names the path and the remedy when install resolves with a conflict', async () => { + await renderCliSectionAndInstall(async () => + notInstalledStatus({ + state: 'conflict', + detail: '/usr/local/bin/orca exists but is not an Orca symlink.' + }) + ) + + const alert = screen.getByRole('alert') + expect(alert.textContent).toContain('/usr/local/bin/orca exists but is not an Orca symlink.') + expect(alert.textContent).toContain( + 'Remove /usr/local/bin/orca and register again if it is no longer needed.' + ) + expect(toasts.success).not.toHaveBeenCalled() + }) + + it('does not claim success when install resolves without registering', async () => { + await renderCliSectionAndInstall(async () => + notInstalledStatus({ + state: 'unsupported', + supported: false, + unsupportedReason: 'launcher_missing', + detail: 'The bundled CLI launcher is missing from this Orca build.' + }) + ) + + expect(screen.getByRole('alert').textContent).toContain( + 'The bundled CLI launcher is missing from this Orca build.' + ) + expect(toasts.success).not.toHaveBeenCalled() + expect(toasts.error).toHaveBeenCalledTimes(1) + }) + + it('keeps the success toast and shows no failure notice when registration lands', async () => { + await renderCliSectionAndInstall(async () => + notInstalledStatus({ state: 'installed', detail: null }) + ) + + expect(screen.queryByRole('alert')).toBeNull() + expect(toasts.success).toHaveBeenCalledTimes(1) + expect(toasts.error).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/settings/CliSection.tsx b/src/renderer/src/components/settings/CliSection.tsx index 32cbb3e5766..364a866df5b 100644 --- a/src/renderer/src/components/settings/CliSection.tsx +++ b/src/renderer/src/components/settings/CliSection.tsx @@ -33,6 +33,7 @@ import { getWslCliDistroRequest } from './CliSkillRuntimeSetup' import { WslCliRegistration } from './WslCliRegistration' +import { useCliRegistrationActions } from './use-cli-registration-actions' import { useLocalCliSkillFreshnessName } from './use-local-cli-skill-freshness-name' import { translate } from '@/i18n/i18n' @@ -81,7 +82,6 @@ export function CliSection({ const [status, setStatus] = useState(null) const [loading, setLoading] = useState(true) const [dialogOpen, setDialogOpen] = useState(false) - const [busyAction, setBusyAction] = useState<'install' | 'remove' | null>(null) const mountedRef = useMountedRef() const agentRuntime = useMemo( () => @@ -132,8 +132,19 @@ export function CliSection({ [mountedRef] ) + const closeDialog = useCallback((): void => setDialogOpen(false), []) + const commandName = status?.commandName ?? getFallbackCommandName(currentPlatform) + const { busyAction, installFailure, clearInstallFailure, install, remove } = + useCliRegistrationActions({ + commandName, + mountedRef, + onStatusChange: handleStatusChange, + onSettled: closeDialog + }) + const refreshStatus = useCallback(async (): Promise => { setLoading(true) + clearInstallFailure() try { handleStatusChange(await window.api.cli.getInstallStatus()) } catch (error) { @@ -152,7 +163,7 @@ export function CliSection({ setLoading(false) } } - }, [handleStatusChange, mountedRef]) + }, [clearInstallFailure, handleStatusChange, mountedRef]) useEffect(() => { void refreshStatus() @@ -163,78 +174,9 @@ export function CliSection({ const isSupported = status?.supported ?? false const isBrowserManaged = status?.unsupportedReason === 'launch_mode_unavailable' const revealLabel = getRevealLabel(currentPlatform) - const commandName = status?.commandName ?? getFallbackCommandName(currentPlatform) const canRevealCommandPath = status?.commandPath != null && ['installed', 'stale', 'conflict'].includes(status.state) - const handleInstall = async (): Promise => { - setBusyAction('install') - try { - const next = await window.api.cli.install() - if (mountedRef.current) { - setStatus(next) - setDialogOpen(false) - toast.success( - translate( - 'auto.components.settings.CliSection.9cbcd31338', - 'Registered `{{value0}}` in PATH.', - { value0: next.commandName } - ) - ) - } - } catch (error) { - if (mountedRef.current) { - toast.error( - error instanceof Error - ? error.message - : translate( - 'auto.components.settings.CliSection.a2b13efa94', - 'Failed to register `{{value0}}` in PATH.', - { value0: commandName } - ) - ) - } - } finally { - if (mountedRef.current) { - setBusyAction(null) - } - } - } - - const handleRemove = async (): Promise => { - setBusyAction('remove') - try { - const next = await window.api.cli.remove() - if (mountedRef.current) { - setStatus(next) - setDialogOpen(false) - toast.success( - translate( - 'auto.components.settings.CliSection.af5540930c', - 'Removed `{{value0}}` from PATH.', - { value0: next.commandName } - ) - ) - } - } catch (error) { - if (mountedRef.current) { - toast.error( - error instanceof Error - ? error.message - : translate( - 'auto.components.settings.CliSection.d77352f2df', - 'Failed to remove `{{value0}}` from PATH.', - { value0: commandName } - ) - ) - } - } finally { - if (mountedRef.current) { - setBusyAction(null) - } - } - } - return (
@@ -336,6 +278,31 @@ export function CliSection({

{status.detail}

) : null} + {installFailure ? ( +
+

+ {translate( + 'auto.components.settings.CliSection.a2b13efa94', + 'Failed to register `{{value0}}` in PATH.', + { value0: commandName } + )} +

+

{installFailure.reason}

+ {installFailure.conflictCommandPath ? ( +

+ {translate( + 'auto.components.settings.CliSection.installFailureConflictRemedy', + 'Remove {{value0}} and register again if it is no longer needed.', + { value0: installFailure.conflictCommandPath } + )} +

+ ) : null} +
+ ) : null} +
{status?.commandPath ? (
diff --git a/src/renderer/src/components/settings/cli-install-failure.test.ts b/src/renderer/src/components/settings/cli-install-failure.test.ts new file mode 100644 index 00000000000..dfd39f15649 --- /dev/null +++ b/src/renderer/src/components/settings/cli-install-failure.test.ts @@ -0,0 +1,108 @@ +import { describe, expect, it } from 'vitest' +import type { CliInstallStatus } from '../../../../shared/cli-install-types' +import { readCliInstallFailure, readCliInstallRejection } from './cli-install-failure' + +const FALLBACK = 'Orca could not finish CLI registration and reported no reason.' + +function cliStatus(overrides: Partial = {}): CliInstallStatus { + return { + platform: 'darwin', + commandName: 'orca', + commandPath: '/usr/local/bin/orca', + pathDirectory: '/usr/local/bin', + pathConfigured: true, + launcherPath: '/Applications/Orca.app/Contents/Resources/bin/orca', + installMethod: 'symlink', + supported: true, + state: 'installed', + currentTarget: null, + unsupportedReason: null, + detail: null, + ...overrides + } +} + +describe('readCliInstallFailure', () => { + it('reports no failure for a landed registration', () => { + expect(readCliInstallFailure(cliStatus(), FALLBACK)).toBeNull() + }) + + it('surfaces the main-process reason verbatim without re-classifying it', () => { + expect( + readCliInstallFailure( + cliStatus({ + state: 'unsupported', + supported: false, + unsupportedReason: 'launcher_missing', + detail: 'The bundled CLI launcher is missing from this Orca build.' + }), + FALLBACK + ) + ).toEqual({ + reason: 'The bundled CLI launcher is missing from this Orca build.', + conflictCommandPath: null + }) + }) + + it('names the conflicting path so the panel can offer the remedy', () => { + expect( + readCliInstallFailure( + cliStatus({ + state: 'conflict', + detail: '/usr/local/bin/orca exists but is not an Orca symlink.' + }), + FALLBACK + ) + ).toEqual({ + reason: '/usr/local/bin/orca exists but is not an Orca symlink.', + conflictCommandPath: '/usr/local/bin/orca' + }) + }) + + it('falls back when the main process reported no detail', () => { + expect(readCliInstallFailure(cliStatus({ state: 'not_installed' }), FALLBACK)).toEqual({ + reason: FALLBACK, + conflictCommandPath: null + }) + }) +}) + +describe('readCliInstallRejection', () => { + it('strips the Electron transport prefix off the installer message', () => { + expect( + readCliInstallRejection( + new Error( + "Error invoking remote method 'cli:install': Error: Refusing to replace non-Orca " + + 'command at /usr/local/bin/orca. Remove it and register again if it is no longer needed.' + ), + FALLBACK + ) + ).toEqual({ + reason: + 'Refusing to replace non-Orca command at /usr/local/bin/orca. ' + + 'Remove it and register again if it is no longer needed.', + conflictCommandPath: null + }) + }) + + it('keeps the registration-lock remedy that names the lock file', () => { + const failure = readCliInstallRejection( + new Error( + "Error invoking remote method 'cli:install': Error: Timed out waiting for another Orca " + + 'process to finish CLI registration (waited 330s). If no other Orca is running, remove ' + + '/home/u/.cache/orca/appimage/.cli-registration.lock and retry.' + ), + FALLBACK + ) + + expect(failure.reason).toContain('.cli-registration.lock and retry.') + expect(failure.reason.startsWith('Timed out waiting')).toBe(true) + }) + + it('falls back for a non-Error rejection with no message', () => { + expect(readCliInstallRejection(new Error(' '), FALLBACK)).toEqual({ + reason: FALLBACK, + conflictCommandPath: null + }) + }) +}) diff --git a/src/renderer/src/components/settings/cli-install-failure.ts b/src/renderer/src/components/settings/cli-install-failure.ts new file mode 100644 index 00000000000..8ca55a43299 --- /dev/null +++ b/src/renderer/src/components/settings/cli-install-failure.ts @@ -0,0 +1,40 @@ +import type { CliInstallStatus } from '../../../../shared/cli-install-types' + +// Why: Electron re-wraps a rejected `ipcMain.handle` as +// `Error invoking remote method '': Error: `, so the installer's +// own sentence is buried behind transport noise by the time it reaches the panel. +const IPC_INVOKE_PREFIX = /^Error invoking remote method '[^']*':\s*(?:Error:\s*)?/ + +export type CliInstallFailure = { + /** The main-process reason verbatim; installer throws already embed their own remedy. */ + reason: string + /** Set only for a conflict, whose status detail names the path but stops short of the remedy. */ + conflictCommandPath: string | null +} + +/** + * A registration call that resolved without landing. The main process already + * reported why in `detail`, so this only decides that it failed — it does not + * re-classify the reason. + */ +export function readCliInstallFailure( + status: CliInstallStatus, + fallbackReason: string +): CliInstallFailure | null { + if (status.state === 'installed') { + return null + } + return { + reason: status.detail?.trim() || fallbackReason, + conflictCommandPath: status.state === 'conflict' ? status.commandPath : null + } +} + +/** A registration call that threw: unwrap the transport prefix off the installer's message. */ +export function readCliInstallRejection(error: unknown, fallbackReason: string): CliInstallFailure { + const message = error instanceof Error ? error.message : String(error) + return { + reason: message.replace(IPC_INVOKE_PREFIX, '').trim() || fallbackReason, + conflictCommandPath: null + } +} diff --git a/src/renderer/src/components/settings/use-cli-registration-actions.ts b/src/renderer/src/components/settings/use-cli-registration-actions.ts new file mode 100644 index 00000000000..29737764e9c --- /dev/null +++ b/src/renderer/src/components/settings/use-cli-registration-actions.ts @@ -0,0 +1,127 @@ +import { useCallback, useState, type MutableRefObject } from 'react' +import { toast } from 'sonner' +import type { CliInstallStatus } from '../../../../shared/cli-install-types' +import { translate } from '@/i18n/i18n' +import { + readCliInstallFailure, + readCliInstallRejection, + type CliInstallFailure +} from './cli-install-failure' + +type CliRegistrationActionsOptions = { + commandName: string + mountedRef: MutableRefObject + onStatusChange: (status: CliInstallStatus) => void + onSettled: () => void +} + +export type CliRegistrationActions = { + busyAction: 'install' | 'remove' | null + installFailure: CliInstallFailure | null + clearInstallFailure: () => void + install: () => Promise + remove: () => Promise +} + +function unknownReason(): string { + return translate( + 'auto.components.settings.CliSection.installFailureUnknownReason', + 'Orca could not finish CLI registration and reported no reason.' + ) +} + +function failedTitle(commandName: string): string { + return translate( + 'auto.components.settings.CliSection.a2b13efa94', + 'Failed to register `{{value0}}` in PATH.', + { value0: commandName } + ) +} + +export function useCliRegistrationActions({ + commandName, + mountedRef, + onStatusChange, + onSettled +}: CliRegistrationActionsOptions): CliRegistrationActions { + const [busyAction, setBusyAction] = useState<'install' | 'remove' | null>(null) + const [installFailure, setInstallFailure] = useState(null) + const clearInstallFailure = useCallback((): void => setInstallFailure(null), []) + + const install = useCallback(async (): Promise => { + setBusyAction('install') + try { + const next = await window.api.cli.install() + if (!mountedRef.current) { + return + } + onStatusChange(next) + onSettled() + // Why: `install()` resolves with the post-registration status, so a refusal + // (conflict, unsupported build, unreadable PATH) arrives as data, not a throw. + const failure = readCliInstallFailure(next, unknownReason()) + setInstallFailure(failure) + if (failure) { + toast.error(failedTitle(next.commandName), { description: failure.reason }) + return + } + toast.success( + translate( + 'auto.components.settings.CliSection.9cbcd31338', + 'Registered `{{value0}}` in PATH.', + { value0: next.commandName } + ) + ) + } catch (error) { + if (!mountedRef.current) { + return + } + const failure = readCliInstallRejection(error, unknownReason()) + setInstallFailure(failure) + // Why: closing reveals the persistent notice the toast is only a preview of. + onSettled() + toast.error(failedTitle(commandName), { description: failure.reason }) + } finally { + if (mountedRef.current) { + setBusyAction(null) + } + } + }, [commandName, mountedRef, onSettled, onStatusChange]) + + const remove = useCallback(async (): Promise => { + setBusyAction('remove') + try { + const next = await window.api.cli.remove() + if (mountedRef.current) { + onStatusChange(next) + onSettled() + setInstallFailure(null) + toast.success( + translate( + 'auto.components.settings.CliSection.af5540930c', + 'Removed `{{value0}}` from PATH.', + { value0: next.commandName } + ) + ) + } + } catch (error) { + if (mountedRef.current) { + toast.error( + error instanceof Error + ? error.message + : translate( + 'auto.components.settings.CliSection.d77352f2df', + 'Failed to remove `{{value0}}` from PATH.', + { value0: commandName } + ) + ) + } + } finally { + if (mountedRef.current) { + setBusyAction(null) + } + } + }, [commandName, mountedRef, onSettled, onStatusChange]) + + return { busyAction, installFailure, clearInstallFailure, install, remove } +} diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index c7a66f2decc..a641c2cf7dd 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -6754,7 +6754,9 @@ "8a9b784c60": "stale", "d363e5929b": "Checking CLI registration…", "cliSkillTerminalTitle": "CLI skill setup", - "cliSkillTerminalAria": "CLI skill install terminal" + "cliSkillTerminalAria": "CLI skill install terminal", + "installFailureUnknownReason": "Orca could not finish CLI registration and reported no reason.", + "installFailureConflictRemedy": "Remove {{value0}} and register again if it is no longer needed." }, "CliSkillRuntimeSetup": { "04325573f8": "WSL", From 62e9949141f6de571f8808e0d819bc6af5565f80 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 03:01:07 -0700 Subject: [PATCH 061/398] perf(renderer): index worktree owner lookups instead of rescanning every workspace (#18130) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `worktreeUsesRemoteConnection`, `getRemoteConnectionIdForWorktree`, `worktreeUsesWslPath` and `rightSidebarShowsPullRequestData` each did `Object.values(state.worktreesByRepo).flat().find(...)` plus a linear `repos.find(...)`. They are called from unmemoized Zustand selectors (`use-tab-agent.ts:263`, `use-visible-review-refresh.ts:45`), so every store write re-ran the whole scan once per open tab. Measured on a real instance (10 repos / 423 worktrees / 382 tabs): the `.find()` predicate alone ran 1,320,424 times in 30s — 44,000 worktree visits/sec — while the app was idle. Switched to the existing WeakMap-cached `getIndexedWorktreeMap` / `getIndexedRepoMap` from `store/worktree-repo-index.ts`, matching what `connection-owner-resolution.ts` already does. Same duplicate-id and host-collision semantics; no behavior change. Benchmark at that scale, 200 store writes x 382 tabs x 3 lookups: before 2.762ms per store write after 0.167ms per store write (16.6x) At ~20 store writes/sec that is 55.2ms/sec of renderer CPU down to 3.3ms/sec. The new scale test counts worktree `id` reads: 160,000 before, 800 after. --- .../src/lib/right-sidebar-visibility.ts | 9 +- .../terminal-workspace-routing.scale.test.ts | 98 +++++++++++++++++++ .../terminals/terminal-workspace-routing.ts | 25 ++--- 3 files changed, 113 insertions(+), 19 deletions(-) create mode 100644 src/renderer/src/store/terminals/terminal-workspace-routing.scale.test.ts diff --git a/src/renderer/src/lib/right-sidebar-visibility.ts b/src/renderer/src/lib/right-sidebar-visibility.ts index 673fff6e01e..4eb9375efb4 100644 --- a/src/renderer/src/lib/right-sidebar-visibility.ts +++ b/src/renderer/src/lib/right-sidebar-visibility.ts @@ -1,4 +1,5 @@ import type { AppState } from '@/store/types' +import { getIndexedRepoMap, getIndexedWorktreeMap } from '@/store/worktree-repo-index' import { isFolderRepo } from '../../../shared/repo-kind' type ActiveView = AppState['activeView'] @@ -37,11 +38,11 @@ export function rightSidebarShowsPullRequestData( return false } - const activeWorktree = Object.values(state.worktreesByRepo) - .flat() - .find((worktree) => worktree.id === state.activeWorktreeId) + const activeWorktree = state.activeWorktreeId + ? getIndexedWorktreeMap(state.worktreesByRepo).get(state.activeWorktreeId) + : undefined const activeRepo = activeWorktree - ? state.repos.find((repo) => repo.id === activeWorktree.repoId) + ? getIndexedRepoMap(state.repos).get(activeWorktree.repoId) : null if (!activeRepo || isFolderRepo(activeRepo)) { return false diff --git a/src/renderer/src/store/terminals/terminal-workspace-routing.scale.test.ts b/src/renderer/src/store/terminals/terminal-workspace-routing.scale.test.ts new file mode 100644 index 00000000000..998ad871d86 --- /dev/null +++ b/src/renderer/src/store/terminals/terminal-workspace-routing.scale.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from 'vitest' +import type { AppState } from '../types' +import { + getRemoteConnectionIdForWorktree, + worktreeUsesRemoteConnection, + worktreeUsesWslPath +} from './terminal-workspace-routing' +import { rightSidebarShowsPullRequestData } from '@/lib/right-sidebar-visibility' + +const REPO_COUNT = 10 +const WORKTREE_COUNT = 400 + +/** Counts every `id` read so a rescan shows up as a multiple of the row count. */ +function buildCountingState(): { + state: AppState + reads: () => number + worktreeId: string +} { + let idReads = 0 + const repos = Array.from({ length: REPO_COUNT }, (_, index) => ({ + id: `repo-${index}`, + name: `repo-${index}`, + path: `/repos/repo-${index}`, + connectionId: null + })) + const worktreesByRepo: Record = {} + const perRepo = WORKTREE_COUNT / REPO_COUNT + for (let repoIndex = 0; repoIndex < REPO_COUNT; repoIndex++) { + worktreesByRepo[`repo-${repoIndex}`] = Array.from({ length: perRepo }, (_, index) => { + const id = `repo-${repoIndex}::/repos/repo-${repoIndex}/wt-${index}` + return { + get id() { + idReads++ + return id + }, + repoId: `repo-${repoIndex}`, + path: `/repos/repo-${repoIndex}/wt-${index}`, + branch: `branch-${index}`, + hostId: 'local' + } + }) + } + return { + state: { + repos, + worktreesByRepo, + folderWorkspaces: [], + projectGroups: [], + activeView: 'worktrees', + activeWorktreeId: 'repo-9::/repos/repo-9/wt-39', + rightSidebarOpen: true, + rightSidebarTab: 'checks' + } as unknown as AppState, + reads: () => idReads, + worktreeId: 'repo-9::/repos/repo-9/wt-39' + } +} + +describe('terminal workspace routing scales with tab count, not workspace count', () => { + it('answers repeated owner lookups without rescanning every worktree', () => { + const { state, reads, worktreeId } = buildCountingState() + const CALLS = 200 + + for (let call = 0; call < CALLS; call++) { + worktreeUsesRemoteConnection(state, worktreeId) + getRemoteConnectionIdForWorktree(state, worktreeId) + worktreeUsesWslPath(state, worktreeId) + rightSidebarShowsPullRequestData(state) + } + + // One index build over every row, then O(1) map hits. A per-call scan would + // read at least CALLS x WORKTREE_COUNT ids. + expect(reads()).toBeLessThanOrEqual(WORKTREE_COUNT * 2) + expect(reads()).toBeLessThan(CALLS * WORKTREE_COUNT) + }) + + it('still resolves the owning repo and its connection', () => { + const { state, worktreeId } = buildCountingState() + const remoteState = { + ...state, + repos: state.repos.map((repo) => + repo.id === 'repo-9' ? { ...repo, connectionId: 'ssh-host-1' } : repo + ) + } as AppState + + expect(worktreeUsesRemoteConnection(remoteState, worktreeId)).toBe(true) + expect(getRemoteConnectionIdForWorktree(remoteState, worktreeId)).toBe('ssh-host-1') + expect(worktreeUsesRemoteConnection(state, worktreeId)).toBe(false) + expect(getRemoteConnectionIdForWorktree(state, worktreeId)).toBeNull() + expect(worktreeUsesWslPath(state, worktreeId)).toBe(false) + }) + + it('returns null for a worktree id that no repo owns', () => { + const { state } = buildCountingState() + expect(getRemoteConnectionIdForWorktree(state, 'ghost::/nowhere')).toBeNull() + expect(worktreeUsesRemoteConnection(state, 'ghost::/nowhere')).toBe(false) + }) +}) diff --git a/src/renderer/src/store/terminals/terminal-workspace-routing.ts b/src/renderer/src/store/terminals/terminal-workspace-routing.ts index 9148b9a5f7c..ae07c3e7b8e 100644 --- a/src/renderer/src/store/terminals/terminal-workspace-routing.ts +++ b/src/renderer/src/store/terminals/terminal-workspace-routing.ts @@ -7,6 +7,7 @@ import { resolveLocalWindowsTerminalShellOverrideForTab } from '../../../../shar import { WINDOWS_GIT_BASH_SHELL } from '../../../../shared/windows-terminal-shell' import { getFolderWorkspaceConnectionId } from '@/lib/folder-workspace-connection' import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' +import { getIndexedRepoMap, getIndexedWorktreeMap } from '../worktree-repo-index' export function isWindowsRendererRuntime(): boolean { return typeof navigator !== 'undefined' && navigator.userAgent.includes('Windows') @@ -61,9 +62,7 @@ export function worktreeUsesWslPath( ) return folderWorkspace ? isWslUncPath(folderWorkspace.folderPath) : false } - const worktree = Object.values(state.worktreesByRepo) - .flat() - .find((entry) => entry.id === worktreeId) + const worktree = getIndexedWorktreeMap(state.worktreesByRepo).get(worktreeId) return worktree ? isWslUncPath(worktree.path) : false } @@ -75,15 +74,13 @@ export function worktreeUsesRemoteConnection( if (parsedWorkspaceKey?.type === 'folder') { return Boolean(getFolderWorkspaceConnectionId(state, parsedWorkspaceKey.folderWorkspaceId)) } - const directRepoId = getRepoIdFromWorktreeId(worktreeId) - const directRepo = state.repos.find((repo) => repo.id === directRepoId) + const repoMap = getIndexedRepoMap(state.repos) + const directRepo = repoMap.get(getRepoIdFromWorktreeId(worktreeId)) if (directRepo) { return Boolean(directRepo.connectionId) } - const worktree = Object.values(state.worktreesByRepo) - .flat() - .find((entry) => entry.id === worktreeId) - const repo = worktree ? state.repos.find((entry) => entry.id === worktree.repoId) : null + const worktree = getIndexedWorktreeMap(state.worktreesByRepo).get(worktreeId) + const repo = worktree ? repoMap.get(worktree.repoId) : null return Boolean(repo?.connectionId) } @@ -95,15 +92,13 @@ export function getRemoteConnectionIdForWorktree( if (parsedWorkspaceKey?.type === 'folder') { return getFolderWorkspaceConnectionId(state, parsedWorkspaceKey.folderWorkspaceId) ?? null } - const directRepoId = getRepoIdFromWorktreeId(worktreeId) - const directRepo = state.repos.find((repo) => repo.id === directRepoId) + const repoMap = getIndexedRepoMap(state.repos) + const directRepo = repoMap.get(getRepoIdFromWorktreeId(worktreeId)) if (directRepo) { return directRepo.connectionId?.trim() || null } - const worktree = Object.values(state.worktreesByRepo) - .flat() - .find((entry) => entry.id === worktreeId) - const repo = worktree ? state.repos.find((entry) => entry.id === worktree.repoId) : null + const worktree = getIndexedWorktreeMap(state.worktreesByRepo).get(worktreeId) + const repo = worktree ? repoMap.get(worktree.repoId) : null return repo?.connectionId?.trim() || null } From aa3ae6f56ec54d9c28c6b65f257c36d1e207fe0b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 03:02:27 -0700 Subject: [PATCH 062/398] fix(ssh): close the pty master fd leak on relay hosts too (#17920) * fix(ssh): close the pty master fd leak on Linux relay hosts The app gets the FD_CLOEXEC patch through pnpm patchedDependencies (#17914); the relay installs stock node-pty from npm, where no pnpm patch reaches. Linux is where that matters -- it is the only relay platform that takes forkpty()'s no-atomic-O_CLOEXEC path, and it is also the only one that already compiles node-pty at install time, so the fix costs a second compile rather than a first. Ships the patch as a relay asset applied like the existing Windows console-list one, and rebuilds only after the probe has proven node-pty loadable. The rebuild is non-fatal by construction: the working build is moved aside first and moved back on any failure, a failed attempt drops a skip marker so the compile is attempted at most once per relay directory, and the caller swallows the whole step. macOS and Windows relays never run it. Measured on node:22 with a relay-style npm install: before, the master is cloexec=false and shows up as `26 -> /dev/pts/ptmx` in both a later pty child and a later child_process child; after, cloexec=true and neither child sees it. Closes #17915. * test(ssh): feed the cloexec patch exec to the hand-rolled namespace fixtures These sequences are positional, so the new Linux-only patch exec swallowed the READY slot and every install/repair case timed out waiting for the relay. * fix(ssh): patch the pty master before publishing the shared native-deps tree * fix(ssh): refuse to publish a native-deps tree whose cloexec patch did not take --- .../node-pty-1.1.0-master-cloexec-patch.cjs | 318 +++++++ .../__fixtures__/node-pty-1.1.0-unix-pty.cc | 799 ++++++++++++++++++ config/scripts/build-relay.mjs | 11 + .../node-pty-master-cloexec-patch.test.mjs | 229 +++++ src/main/ssh/ssh-relay-deploy.ts | 122 ++- ...ssh-relay-native-deps-cache-deploy.test.ts | 7 + src/main/ssh/ssh-relay-native-deps-cache.ts | 6 + .../ssh-relay-native-deps-install-fixture.ts | 6 + .../ssh/ssh-relay-native-deps-install.test.ts | 1 + ...sh-relay-native-deps-probe-verdict.test.ts | 6 + .../ssh-relay-node-pty-spawn-repair.test.ts | 4 + ...h-relay-pty-master-cloexec-install.test.ts | 336 ++++++++ .../ssh-relay-sftp-namespace-install.test.ts | 8 + src/shared/relay-artifacts.ts | 5 + src/shared/relay-optional-artifacts.test.ts | 8 + 15 files changed, 1865 insertions(+), 1 deletion(-) create mode 100644 config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs create mode 100644 config/scripts/__fixtures__/node-pty-1.1.0-unix-pty.cc create mode 100644 config/scripts/node-pty-master-cloexec-patch.test.mjs create mode 100644 src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts diff --git a/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs b/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs new file mode 100644 index 00000000000..f4f4f87619a --- /dev/null +++ b/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs @@ -0,0 +1,318 @@ +/** + * Relay-side pty-master close-on-exec patch for node-pty 1.1.0 (#17915). + * + * The app gets this through pnpm `patchedDependencies`; the relay installs stock + * node-pty from npm onto the host, where no pnpm patch reaches. Without it every + * later child of the relay -- pty children, git helpers, probes, agent CLIs -- + * inherits each live master fd and keeps its /dev/pts device alive for the life + * of the relay (#8362). + * + * Linux only, deliberately: it is the only relay platform that takes forkpty()'s + * no-atomic-O_CLOEXEC path, and the only one that already compiles node-pty at + * install time, so the rebuild costs a second compile rather than a first one. + * macOS re-opens the tty through uv_tty_init's cloexec dup and Windows has no fds. + * + * Non-fatal by construction: the working build is moved aside before anything is + * touched and moved back on any failure, and a failed attempt drops a skip marker + * so the compile is attempted at most once per relay directory. + */ + +const { spawnSync } = require('node:child_process') +const { createHash } = require('node:crypto') +const { + existsSync, + mkdirSync, + readFileSync, + renameSync, + rmSync, + writeFileSync +} = require('node:fs') +const { dirname, join, resolve } = require('node:path') + +const EXPECTED_NODE_PTY_VERSION = '1.1.0' +const ORIGINAL_SOURCE_SHA256 = '5e1005d6bdcfbe97b486ee415419fe7adae99035047f07340fbad36419e0bae6' +const PATCHED_SOURCE_SHA256 = '97dea52199216c01b62070758f0f38621ae53adc16c221271dd35ae2d8ee3482' + +const STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' +const SKIP_MARKER_FILENAME = '.node-pty-cloexec-skip' +const BACKUP_DIRNAME = '.orca-cloexec-prepatch-release' +// Under the caller's 240s SSH command timeout, so the rollback below still runs. +const REBUILD_TIMEOUT_MS = 200000 +const VERIFY_TIMEOUT_MS = 15000 + +const FORWARD_DECLARATION = [ + 'static int\npty_nonblock(int);\n', + 'static int\npty_nonblock(int);\n\nstatic int\npty_cloexec(int);\n' +] + +const DEFINITION = [ + `static int +pty_nonblock(int fd) { + int flags = fcntl(fd, F_GETFL, 0); + if (flags == -1) return -1; + return fcntl(fd, F_SETFL, flags | O_NONBLOCK); +} +`, + `static int +pty_nonblock(int fd) { + int flags = fcntl(fd, F_GETFL, 0); + if (flags == -1) return -1; + return fcntl(fd, F_SETFL, flags | O_NONBLOCK); +} + +/** + * Orca: close-on-exec FD + * + * forkpty()/posix_openpt() have no atomic O_CLOEXEC, so a master left without + * FD_CLOEXEC is inherited by every later child of this process -- including + * later pty children -- which keeps its /dev/pts device and buffers alive long + * after its own session ends (#8362). + */ + +static int +pty_cloexec(int fd) { + int flags = fcntl(fd, F_GETFD); + if (flags == -1) return -1; + if (flags & FD_CLOEXEC) return 0; + return fcntl(fd, F_SETFD, flags | FD_CLOEXEC); +} +` +] + +const FORKPTY_CALL_SITE = [ + ` default: + if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } + } +`, + ` default: + if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } + if (pty_cloexec(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to close-on-exec."); + } + } +` +] + +const REPLACEMENTS = [FORWARD_DECLARATION, DEFINITION, FORKPTY_CALL_SITE] + +function sourceSha256(source) { + return createHash('sha256').update(source).digest('hex') +} + +function nodePtyDir(relayDir) { + return resolve(relayDir, 'node_modules', 'node-pty') +} + +function inspectNodePtyUnixSource(relayDir) { + const ptyDir = nodePtyDir(relayDir) + const sourcePath = join(ptyDir, 'src', 'unix', 'pty.cc') + const version = JSON.parse(readFileSync(join(ptyDir, 'package.json'), 'utf8')).version + if (version !== EXPECTED_NODE_PTY_VERSION) { + throw new Error(`Refusing to patch node-pty ${version}; expected ${EXPECTED_NODE_PTY_VERSION}`) + } + return { ptyDir, sourcePath, source: readFileSync(sourcePath, 'utf8') } +} + +function writeSourceAtomically(sourcePath, contents) { + const temporaryPath = `${sourcePath}.orca-patch-${process.pid}` + // Why: a terminated install must leave one of the two known source versions on disk. + try { + writeFileSync(temporaryPath, contents) + renameSync(temporaryPath, sourcePath) + } finally { + rmSync(temporaryPath, { force: true }) + } +} + +function rewriteSource(source, reverse) { + let rewritten = source + for (const [original, patched] of REPLACEMENTS) { + const from = reverse ? patched : original + const to = reverse ? original : patched + if (rewritten.split(from).length - 1 !== 1) { + throw new Error('Refusing to rewrite unexpected node-pty pty.cc source') + } + rewritten = rewritten.replace(from, to) + } + return rewritten +} + +/** True when the patch was applied, false when it was already installed. */ +function patchNodePtyMasterCloexecSource(relayDir = process.cwd()) { + const inspected = inspectNodePtyUnixSource(relayDir) + const hash = sourceSha256(inspected.source) + if (hash === PATCHED_SOURCE_SHA256) { + return false + } + if (hash !== ORIGINAL_SOURCE_SHA256) { + throw new Error('Refusing to patch unexpected node-pty pty.cc source') + } + writeSourceAtomically(inspected.sourcePath, rewriteSource(inspected.source, false)) + assertPatchedNodePtyMasterCloexecSource(relayDir) + return true +} + +function assertPatchedNodePtyMasterCloexecSource(relayDir = process.cwd()) { + const inspected = inspectNodePtyUnixSource(relayDir) + if (sourceSha256(inspected.source) !== PATCHED_SOURCE_SHA256) { + throw new Error('node-pty pty master close-on-exec patch is not installed') + } +} + +function revertNodePtyMasterCloexecSource(relayDir = process.cwd()) { + const inspected = inspectNodePtyUnixSource(relayDir) + if (sourceSha256(inspected.source) === ORIGINAL_SOURCE_SHA256) { + return false + } + writeSourceAtomically(inspected.sourcePath, rewriteSource(inspected.source, true)) + return true +} + +function rebuildNodePty(relayDir) { + const result = spawnSync('npm', ['rebuild', '--ignore-scripts=false', 'node-pty'], { + cwd: relayDir, + encoding: 'utf8', + timeout: REBUILD_TIMEOUT_MS, + windowsHide: true + }) + if (result.error) { + throw new Error(`npm rebuild node-pty failed: ${result.error.message}`) + } + if (result.status !== 0) { + const tail = `${result.stdout || ''}${result.stderr || ''}`.trim().slice(-300) + throw new Error(`npm rebuild node-pty exited ${result.status ?? result.signal}: ${tail}`) + } +} + +// Why a child: a bad build can abort the process on require, which would strand the +// moved-aside working build. Why the reachability check: a host without /proc cannot +// show inheritance, and an unobservable flag is not evidence the rebuild was wrong. +const VERIFY_SCRIPT = ` +const pty = require(process.argv[1]); +const term = pty.spawn('/bin/sh', ['-c', 'exit 0'], { + name: 'xterm-256color', cols: 80, rows: 24, cwd: process.cwd(), env: process.env +}); +const probe = require('node:child_process').spawnSync('/bin/sh', ['-c', 'ls -l /proc/self/fd'], { encoding: 'utf8' }); +try { term.kill() } catch {} +const listing = probe.stdout || ''; +if (probe.status !== 0 || !listing.includes('->')) { console.log('UNVERIFIED'); process.exit(0) } +console.log(listing.includes('ptmx') ? 'INHERITED' : 'ISOLATED'); +process.exit(0); +` + +/** 'isolated' when a later plain child no longer inherits the master, 'unverified' when /proc cannot say. */ +function verifyMasterNotInheritedByLaterChild(relayDir) { + const result = spawnSync(process.execPath, ['-e', VERIFY_SCRIPT, nodePtyDir(relayDir)], { + cwd: relayDir, + encoding: 'utf8', + timeout: VERIFY_TIMEOUT_MS, + windowsHide: true + }) + const output = `${result.stdout || ''}` + if (result.status !== 0 || result.error) { + const tail = `${output}${result.stderr || ''}`.trim().slice(-300) + throw new Error( + `rebuilt node-pty did not load: ${tail || result.error?.message || result.signal}` + ) + } + if (output.includes('INHERITED')) { + throw new Error('rebuilt node-pty still leaks the pty master into later children') + } + return output.includes('ISOLATED') ? 'isolated' : 'unverified' +} + +function rollback(relayDir, releaseDir, backupDir) { + rmSync(releaseDir, { recursive: true, force: true }) + try { + revertNodePtyMasterCloexecSource(relayDir) + } catch { + // The build that is about to be restored predates the patch either way. + } + if (existsSync(backupDir)) { + mkdirSync(dirname(releaseDir), { recursive: true }) + renameSync(backupDir, releaseDir) + } +} + +/** + * Patch and rebuild the host's node-pty, or leave it exactly as found. + * Never throws: the caller is on the connect path and a leaky relay beats no relay. + */ +function applyNodePtyMasterCloexecPatch(relayDir = process.cwd(), options = {}) { + const platform = options.platform || process.platform + const rebuild = options.rebuild || rebuildNodePty + const verify = options.verify || verifyMasterNotInheritedByLaterChild + if (platform !== 'linux') { + return 'skipped:not-linux' + } + const skipMarkerPath = join(relayDir, SKIP_MARKER_FILENAME) + if (existsSync(skipMarkerPath)) { + return 'skipped:earlier-attempt-failed' + } + const releaseDir = join(nodePtyDir(relayDir), 'build', 'Release') + const backupDir = join(nodePtyDir(relayDir), BACKUP_DIRNAME) + // A backup stranded by a connection that died mid-rebuild is stale by definition: + // whatever repaired node-pty since built from the source now on disk. + rmSync(backupDir, { recursive: true, force: true }) + + let inspected + try { + inspected = inspectNodePtyUnixSource(relayDir) + } catch (err) { + return `skipped:${err.message}` + } + const hash = sourceSha256(inspected.source) + if (hash === PATCHED_SOURCE_SHA256) { + return 'already-patched' + } + if (hash !== ORIGINAL_SOURCE_SHA256) { + return 'skipped:unexpected-source' + } + // No compiled build means the host runs a prebuild or nothing at all; rebuilding + // could only take away the artifact the probe just proved loadable. + if (!existsSync(join(releaseDir, 'pty.node'))) { + return 'skipped:no-compiled-build' + } + + try { + renameSync(releaseDir, backupDir) + } catch (err) { + return `skipped:${err.message}` + } + try { + patchNodePtyMasterCloexecSource(relayDir) + rebuild(relayDir) + const verdict = verify(relayDir) + rmSync(backupDir, { recursive: true, force: true }) + return verdict === 'isolated' ? 'patched' : 'patched-unverified' + } catch (err) { + rollback(relayDir, releaseDir, backupDir) + // Bounded on purpose: one compile attempt per relay directory, never a retry loop. + try { + writeFileSync(skipMarkerPath, `${new Date().toISOString()} ${err.message}\n`) + } catch { + // A relay dir we cannot write to will fail the cheap checks above next time anyway. + } + return `failed:${err.message}` + } +} + +if (require.main === module) { + console.log(`${STATUS_PREFIX}${applyNodePtyMasterCloexecPatch()}`) +} + +module.exports = { + EXPECTED_NODE_PTY_VERSION, + ORIGINAL_SOURCE_SHA256, + PATCHED_SOURCE_SHA256, + SKIP_MARKER_FILENAME, + STATUS_PREFIX, + applyNodePtyMasterCloexecPatch, + assertPatchedNodePtyMasterCloexecSource, + patchNodePtyMasterCloexecSource, + revertNodePtyMasterCloexecSource +} diff --git a/config/scripts/__fixtures__/node-pty-1.1.0-unix-pty.cc b/config/scripts/__fixtures__/node-pty-1.1.0-unix-pty.cc new file mode 100644 index 00000000000..7b4b9e1f990 --- /dev/null +++ b/config/scripts/__fixtures__/node-pty-1.1.0-unix-pty.cc @@ -0,0 +1,799 @@ +/** + * Copyright (c) 2012-2015, Christopher Jeffrey (MIT License) + * Copyright (c) 2017, Daniel Imms (MIT License) + * + * pty.cc: + * This file is responsible for starting processes + * with pseudo-terminal file descriptors. + * + * See: + * man pty + * man tty_ioctl + * man termios + * man forkpty + */ + +/** + * Includes + */ + +#define NODE_ADDON_API_DISABLE_DEPRECATED +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +/* forkpty */ +/* http://www.gnu.org/software/gnulib/manual/html_node/forkpty.html */ +#if defined(__linux__) +#include +#elif defined(__APPLE__) +#include +#elif defined(__FreeBSD__) +#include +#include +#elif defined(__OpenBSD__) +#include +#include +#endif + +/* Some platforms name VWERASE and VDISCARD differently */ +#if !defined(VWERASE) && defined(VWERSE) +#define VWERASE VWERSE +#endif +#if !defined(VDISCARD) && defined(VDISCRD) +#define VDISCARD VDISCRD +#endif + +/* for pty_getproc */ +#if defined(__linux__) +#include +#include +#elif defined(__APPLE__) +#include +#include +#include +#include +#include +#include +#include +#endif + +/* NSIG - macro for highest signal + 1, should be defined */ +#ifndef NSIG +#define NSIG 32 +#endif + +/* macOS 10.14 back does not define this constant */ +#ifndef POSIX_SPAWN_SETSID + #define POSIX_SPAWN_SETSID 1024 +#endif + +/* environ for execvpe */ +/* node/src/node_child_process.cc */ +#if !defined(__APPLE__) +extern char **environ; +#endif + +#if defined(__APPLE__) +extern "C" { +// Changes the current thread's directory to a path or directory file +// descriptor. libpthread only exposes a syscall wrapper starting in +// macOS 10.12, but the system call dates back to macOS 10.5. On older OSes, +// the syscall is issued directly. +int pthread_chdir_np(const char* dir) API_AVAILABLE(macosx(10.12)); +int pthread_fchdir_np(int fd) API_AVAILABLE(macosx(10.12)); +} + +#define HANDLE_EINTR(x) ({ \ + int eintr_wrapper_counter = 0; \ + decltype(x) eintr_wrapper_result; \ + do { \ + eintr_wrapper_result = (x); \ + } while (eintr_wrapper_result == -1 && errno == EINTR && \ + eintr_wrapper_counter++ < 100); \ + eintr_wrapper_result; \ +}) +#endif + +struct ExitEvent { + int exit_code = 0, signal_code = 0; +}; + +void SetupExitCallback(Napi::Env env, Napi::Function cb, pid_t pid) { + std::thread *th = new std::thread; + // Don't use Napi::AsyncWorker which is limited by UV_THREADPOOL_SIZE. + auto tsfn = Napi::ThreadSafeFunction::New( + env, + cb, // JavaScript function called asynchronously + "SetupExitCallback_resource", // Name + 0, // Unlimited queue + 1, // Only one thread will use this initially + [th](Napi::Env) { // Finalizer used to clean threads up + th->join(); + delete th; + }); + *th = std::thread([tsfn = std::move(tsfn), pid] { + auto callback = [](Napi::Env env, Napi::Function cb, ExitEvent *exit_event) { + cb.Call({Napi::Number::New(env, exit_event->exit_code), + Napi::Number::New(env, exit_event->signal_code)}); + delete exit_event; + }; + + int ret; + int stat_loc; +#if defined(__APPLE__) + // Based on + // https://source.chromium.org/chromium/chromium/src/+/main:base/process/kill_mac.cc;l=35-69? + int kq = HANDLE_EINTR(kqueue()); + struct kevent change = {0}; + EV_SET(&change, pid, EVFILT_PROC, EV_ADD, NOTE_EXIT, 0, NULL); + ret = HANDLE_EINTR(kevent(kq, &change, 1, NULL, 0, NULL)); + if (ret == -1) { + if (errno == ESRCH) { + // At this point, one of the following has occurred: + // 1. The process has died but has not yet been reaped. + // 2. The process has died and has already been reaped. + // 3. The process is in the process of dying. It's no longer + // kqueueable, but it may not be waitable yet either. Mark calls + // this case the "zombie death race". + ret = HANDLE_EINTR(waitpid(pid, &stat_loc, WNOHANG)); + if (ret == 0) { + ret = kill(pid, SIGKILL); + if (ret != -1) { + HANDLE_EINTR(waitpid(pid, &stat_loc, 0)); + } + } + } + } else { + struct kevent event = {0}; + ret = HANDLE_EINTR(kevent(kq, NULL, 0, &event, 1, NULL)); + if (ret == 1) { + if ((event.fflags & NOTE_EXIT) && + (event.ident == static_cast(pid))) { + // The process is dead or dying. This won't block for long, if at + // all. + HANDLE_EINTR(waitpid(pid, &stat_loc, 0)); + } + } + } +#else + while (true) { + errno = 0; + if ((ret = waitpid(pid, &stat_loc, 0)) != pid) { + if (ret == -1 && errno == EINTR) { + continue; + } + if (ret == -1 && errno == ECHILD) { + // XXX node v0.8.x seems to have this problem. + // waitpid is already handled elsewhere. + ; + } else { + assert(false); + } + } + break; + } +#endif + ExitEvent *exit_event = new ExitEvent; + if (WIFEXITED(stat_loc)) { + exit_event->exit_code = WEXITSTATUS(stat_loc); // errno? + } + if (WIFSIGNALED(stat_loc)) { + exit_event->signal_code = WTERMSIG(stat_loc); + } + auto status = tsfn.BlockingCall(exit_event, callback); // In main thread + switch (status) { + case napi_closing: + break; + + case napi_queue_full: + Napi::Error::Fatal("SetupExitCallback", "Queue was full"); + + case napi_ok: + if (tsfn.Release() != napi_ok) { + Napi::Error::Fatal("SetupExitCallback", "ThreadSafeFunction.Release() failed"); + } + break; + + default: + Napi::Error::Fatal("SetupExitCallback", "ThreadSafeFunction.BlockingCall() failed"); + } + }); +} + +/** + * Methods + */ + +Napi::Value PtyFork(const Napi::CallbackInfo& info); +Napi::Value PtyOpen(const Napi::CallbackInfo& info); +Napi::Value PtyResize(const Napi::CallbackInfo& info); +Napi::Value PtyGetProc(const Napi::CallbackInfo& info); + +/** + * Functions + */ + +static int +pty_nonblock(int); + +#if defined(__APPLE__) +static char * +pty_getproc(int); +#else +static char * +pty_getproc(int, char *); +#endif + +#if defined(__APPLE__) || defined(__OpenBSD__) +static void +pty_posix_spawn(char** argv, char** env, + const struct termios *termp, + const struct winsize *winp, + int* master, + pid_t* pid, + int* err); +#endif + +struct DelBuf { + int len; + DelBuf(int len) : len(len) {} + void operator()(char **p) { + if (p == nullptr) + return; + for (int i = 0; i < len; i++) + free(p[i]); + delete[] p; + } +}; + +Napi::Value PtyFork(const Napi::CallbackInfo& info) { + Napi::Env napiEnv(info.Env()); + Napi::HandleScope scope(napiEnv); + + if (info.Length() != 11 || + !info[0].IsString() || + !info[1].IsArray() || + !info[2].IsArray() || + !info[3].IsString() || + !info[4].IsNumber() || + !info[5].IsNumber() || + !info[6].IsNumber() || + !info[7].IsNumber() || + !info[8].IsBoolean() || + !info[9].IsString() || + !info[10].IsFunction()) { + throw Napi::Error::New(napiEnv, "Usage: pty.fork(file, args, env, cwd, cols, rows, uid, gid, utf8, helperPath, onexit)"); + } + + // file + std::string file = info[0].As(); + + // args + Napi::Array argv_ = info[1].As(); + + // env + Napi::Array env_ = info[2].As(); + int envc = env_.Length(); + std::unique_ptr env_unique_ptr(new char *[envc + 1], DelBuf(envc + 1)); + char **env = env_unique_ptr.get(); + env[envc] = NULL; + for (int i = 0; i < envc; i++) { + std::string pair = env_.Get(i).As(); + env[i] = strdup(pair.c_str()); + } + + // cwd + std::string cwd_ = info[3].As(); + + // size + struct winsize winp; + winp.ws_col = info[4].As().Int32Value(); + winp.ws_row = info[5].As().Int32Value(); + winp.ws_xpixel = 0; + winp.ws_ypixel = 0; + +#if !defined(__APPLE__) + // uid / gid + int uid = info[6].As().Int32Value(); + int gid = info[7].As().Int32Value(); +#endif + + // termios + struct termios t = termios(); + struct termios *term = &t; + term->c_iflag = ICRNL | IXON | IXANY | IMAXBEL | BRKINT; + if (info[8].As().Value()) { +#if defined(IUTF8) + term->c_iflag |= IUTF8; +#endif + } + term->c_oflag = OPOST | ONLCR; + term->c_cflag = CREAD | CS8 | HUPCL; + term->c_lflag = ICANON | ISIG | IEXTEN | ECHO | ECHOE | ECHOK | ECHOKE | ECHOCTL; + + term->c_cc[VEOF] = 4; + term->c_cc[VEOL] = -1; + term->c_cc[VEOL2] = -1; + term->c_cc[VERASE] = 0x7f; + term->c_cc[VWERASE] = 23; + term->c_cc[VKILL] = 21; + term->c_cc[VREPRINT] = 18; + term->c_cc[VINTR] = 3; + term->c_cc[VQUIT] = 0x1c; + term->c_cc[VSUSP] = 26; + term->c_cc[VSTART] = 17; + term->c_cc[VSTOP] = 19; + term->c_cc[VLNEXT] = 22; + term->c_cc[VDISCARD] = 15; + term->c_cc[VMIN] = 1; + term->c_cc[VTIME] = 0; + + #if (__APPLE__) + term->c_cc[VDSUSP] = 25; + term->c_cc[VSTATUS] = 20; + #endif + + cfsetispeed(term, B38400); + cfsetospeed(term, B38400); + + // helperPath + std::string helper_path = info[9].As(); + + pid_t pid; + int master; +#if defined(__APPLE__) + int argc = argv_.Length(); + int argl = argc + 4; + std::unique_ptr argv_unique_ptr(new char *[argl], DelBuf(argl)); + char **argv = argv_unique_ptr.get(); + argv[0] = strdup(helper_path.c_str()); + argv[1] = strdup(cwd_.c_str()); + argv[2] = strdup(file.c_str()); + argv[argl - 1] = NULL; + for (int i = 0; i < argc; i++) { + std::string arg = argv_.Get(i).As(); + argv[i + 3] = strdup(arg.c_str()); + } + + int err = -1; + pty_posix_spawn(argv, env, term, &winp, &master, &pid, &err); + if (err != 0) { + throw Napi::Error::New(napiEnv, "posix_spawnp failed."); + } + if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } +#else + int argc = argv_.Length(); + int argl = argc + 2; + std::unique_ptr argv_unique_ptr(new char *[argl], DelBuf(argl)); + char** argv = argv_unique_ptr.get(); + argv[0] = strdup(file.c_str()); + argv[argl - 1] = NULL; + for (int i = 0; i < argc; i++) { + std::string arg = argv_.Get(i).As(); + argv[i + 1] = strdup(arg.c_str()); + } + + sigset_t newmask, oldmask; + struct sigaction sig_action; + // temporarily block all signals + // this is needed due to a race condition in openpty + // and to avoid running signal handlers in the child + // before exec* happened + sigfillset(&newmask); + pthread_sigmask(SIG_SETMASK, &newmask, &oldmask); + + pid = forkpty(&master, nullptr, static_cast(term), static_cast(&winp)); + + if (!pid) { + // remove all signal handler from child + sig_action.sa_handler = SIG_DFL; + sig_action.sa_flags = 0; + sigemptyset(&sig_action.sa_mask); + for (int i = 0 ; i < NSIG ; i++) { // NSIG is a macro for all signals + 1 + sigaction(i, &sig_action, NULL); + } + } + + // reenable signals + pthread_sigmask(SIG_SETMASK, &oldmask, NULL); + + switch (pid) { + case -1: + throw Napi::Error::New(napiEnv, "forkpty(3) failed."); + case 0: + if (strlen(cwd_.c_str())) { + if (chdir(cwd_.c_str()) == -1) { + perror("chdir(2) failed."); + _exit(1); + } + } + + if (uid != -1 && gid != -1) { + if (setgid(gid) == -1) { + perror("setgid(2) failed."); + _exit(1); + } + if (setuid(uid) == -1) { + perror("setuid(2) failed."); + _exit(1); + } + } + + { + char **old = environ; + environ = env; + execvp(argv[0], argv); + environ = old; + perror("execvp(3) failed."); + _exit(1); + } + default: + if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } + } +#endif + + Napi::Object obj = Napi::Object::New(napiEnv); + obj.Set("fd", Napi::Number::New(napiEnv, master)); + obj.Set("pid", Napi::Number::New(napiEnv, pid)); + obj.Set("pty", Napi::String::New(napiEnv, ptsname(master))); + + // Set up process exit callback. + Napi::Function cb = info[10].As(); + SetupExitCallback(napiEnv, cb, pid); + return obj; +} + +Napi::Value PtyOpen(const Napi::CallbackInfo& info) { + Napi::Env env(info.Env()); + Napi::HandleScope scope(env); + + if (info.Length() != 2 || + !info[0].IsNumber() || + !info[1].IsNumber()) { + throw Napi::Error::New(env, "Usage: pty.open(cols, rows)"); + } + + // size + struct winsize winp; + winp.ws_col = info[0].As().Int32Value(); + winp.ws_row = info[1].As().Int32Value(); + winp.ws_xpixel = 0; + winp.ws_ypixel = 0; + + // pty + int master, slave; + int ret = openpty(&master, &slave, nullptr, NULL, static_cast(&winp)); + + if (ret == -1) { + throw Napi::Error::New(env, "openpty(3) failed."); + } + + if (pty_nonblock(master) == -1) { + throw Napi::Error::New(env, "Could not set master fd to nonblocking."); + } + + if (pty_nonblock(slave) == -1) { + throw Napi::Error::New(env, "Could not set slave fd to nonblocking."); + } + + Napi::Object obj = Napi::Object::New(env); + obj.Set("master", Napi::Number::New(env, master)); + obj.Set("slave", Napi::Number::New(env, slave)); + obj.Set("pty", Napi::String::New(env, ptsname(master))); + + return obj; +} + +Napi::Value PtyResize(const Napi::CallbackInfo& info) { + Napi::Env env(info.Env()); + Napi::HandleScope scope(env); + + if (info.Length() != 3 || + !info[0].IsNumber() || + !info[1].IsNumber() || + !info[2].IsNumber()) { + throw Napi::Error::New(env, "Usage: pty.resize(fd, cols, rows)"); + } + + int fd = info[0].As().Int32Value(); + + struct winsize winp; + winp.ws_col = info[1].As().Int32Value(); + winp.ws_row = info[2].As().Int32Value(); + winp.ws_xpixel = 0; + winp.ws_ypixel = 0; + + if (ioctl(fd, TIOCSWINSZ, &winp) == -1) { + switch (errno) { + case EBADF: + throw Napi::Error::New(env, "ioctl(2) failed, EBADF"); + case EFAULT: + throw Napi::Error::New(env, "ioctl(2) failed, EFAULT"); + case EINVAL: + throw Napi::Error::New(env, "ioctl(2) failed, EINVAL"); + case ENOTTY: + throw Napi::Error::New(env, "ioctl(2) failed, ENOTTY"); + } + throw Napi::Error::New(env, "ioctl(2) failed"); + } + + return env.Undefined(); +} + +/** + * Foreground Process Name + */ +Napi::Value PtyGetProc(const Napi::CallbackInfo& info) { + Napi::Env env(info.Env()); + Napi::HandleScope scope(env); + +#if defined(__APPLE__) + if (info.Length() != 1 || + !info[0].IsNumber()) { + throw Napi::Error::New(env, "Usage: pty.process(pid)"); + } + + int fd = info[0].As().Int32Value(); + char *name = pty_getproc(fd); +#else + if (info.Length() != 2 || + !info[0].IsNumber() || + !info[1].IsString()) { + throw Napi::Error::New(env, "Usage: pty.process(fd, tty)"); + } + + int fd = info[0].As().Int32Value(); + + std::string tty_ = info[1].As(); + char *tty = strdup(tty_.c_str()); + char *name = pty_getproc(fd, tty); + free(tty); +#endif + + if (name == NULL) { + return env.Undefined(); + } + + Napi::String name_ = Napi::String::New(env, name); + free(name); + return name_; +} + +/** + * Nonblocking FD + */ + +static int +pty_nonblock(int fd) { + int flags = fcntl(fd, F_GETFL, 0); + if (flags == -1) return -1; + return fcntl(fd, F_SETFL, flags | O_NONBLOCK); +} + +/** + * pty_getproc + * Taken from tmux. + */ + +// Taken from: tmux (http://tmux.sourceforge.net/) +// Copyright (c) 2009 Nicholas Marriott +// Copyright (c) 2009 Joshua Elsasser +// Copyright (c) 2009 Todd Carson +// +// Permission to use, copy, modify, and distribute this software for any +// purpose with or without fee is hereby granted, provided that the above +// copyright notice and this permission notice appear in all copies. +// +// THE SOFTWARE IS PROVIDED "AS IS" AND THE AUTHOR DISCLAIMS ALL WARRANTIES +// WITH REGARD TO THIS SOFTWARE INCLUDING ALL IMPLIED WARRANTIES OF +// MERCHANTABILITY AND FITNESS. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR +// ANY SPECIAL, DIRECT, INDIRECT, OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES +// WHATSOEVER RESULTING FROM LOSS OF MIND, USE, DATA OR PROFITS, WHETHER +// IN AN ACTION OF CONTRACT, NEGLIGENCE OR OTHER TORTIOUS ACTION, ARISING +// OUT OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. + +#if defined(__linux__) + +static char * +pty_getproc(int fd, char *tty) { + FILE *f; + char *path, *buf; + size_t len; + int ch; + pid_t pgrp; + int r; + + if ((pgrp = tcgetpgrp(fd)) == -1) { + return NULL; + } + + r = asprintf(&path, "/proc/%lld/cmdline", (long long)pgrp); + if (r == -1 || path == NULL) return NULL; + + if ((f = fopen(path, "r")) == NULL) { + free(path); + return NULL; + } + + free(path); + + len = 0; + buf = NULL; + while ((ch = fgetc(f)) != EOF) { + if (ch == '\0') break; + buf = (char *)realloc(buf, len + 2); + if (buf == NULL) return NULL; + buf[len++] = ch; + } + + if (buf != NULL) { + buf[len] = '\0'; + } + + fclose(f); + return buf; +} + +#elif defined(__APPLE__) + +static char * +pty_getproc(int fd) { + int mib[4] = { CTL_KERN, KERN_PROC, KERN_PROC_PID, 0 }; + size_t size; + struct kinfo_proc kp; + + if ((mib[3] = tcgetpgrp(fd)) == -1) { + return NULL; + } + + size = sizeof kp; + if (sysctl(mib, 4, &kp, &size, NULL, 0) == -1) { + return NULL; + } + + if (size != (sizeof kp) || *kp.kp_proc.p_comm == '\0') { + return NULL; + } + + return strdup(kp.kp_proc.p_comm); +} + +#else + +static char * +pty_getproc(int fd, char *tty) { + return NULL; +} + +#endif + +#if defined(__APPLE__) +static void +pty_posix_spawn(char** argv, char** env, + const struct termios *termp, + const struct winsize *winp, + int* master, + pid_t* pid, + int* err) { + int low_fds[3]; + size_t count = 0; + + for (; count < 3; count++) { + low_fds[count] = posix_openpt(O_RDWR); + if (low_fds[count] >= STDERR_FILENO) + break; + } + + int flags = POSIX_SPAWN_CLOEXEC_DEFAULT | + POSIX_SPAWN_SETSIGDEF | + POSIX_SPAWN_SETSIGMASK | + POSIX_SPAWN_SETSID; + *master = posix_openpt(O_RDWR); + if (*master == -1) { + return; + } + + int res = grantpt(*master) || unlockpt(*master); + if (res == -1) { + return; + } + + // Use TIOCPTYGNAME instead of ptsname() to avoid threading problems. + int slave; + char slave_pty_name[128]; + res = ioctl(*master, TIOCPTYGNAME, slave_pty_name); + if (res == -1) { + return; + } + + slave = open(slave_pty_name, O_RDWR | O_NOCTTY); + if (slave == -1) { + return; + } + + if (termp) { + res = tcsetattr(slave, TCSANOW, termp); + if (res == -1) { + return; + }; + } + + if (winp) { + res = ioctl(slave, TIOCSWINSZ, winp); + if (res == -1) { + return; + } + } + + posix_spawn_file_actions_t acts; + posix_spawn_file_actions_init(&acts); + posix_spawn_file_actions_adddup2(&acts, slave, STDIN_FILENO); + posix_spawn_file_actions_adddup2(&acts, slave, STDOUT_FILENO); + posix_spawn_file_actions_adddup2(&acts, slave, STDERR_FILENO); + posix_spawn_file_actions_addclose(&acts, slave); + posix_spawn_file_actions_addclose(&acts, *master); + + posix_spawnattr_t attrs; + posix_spawnattr_init(&attrs); + *err = posix_spawnattr_setflags(&attrs, flags); + if (*err != 0) { + goto done; + } + + sigset_t signal_set; + /* Reset all signal the child to their default behavior */ + sigfillset(&signal_set); + *err = posix_spawnattr_setsigdefault(&attrs, &signal_set); + if (*err != 0) { + goto done; + } + + /* Reset the signal mask for all signals */ + sigemptyset(&signal_set); + *err = posix_spawnattr_setsigmask(&attrs, &signal_set); + if (*err != 0) { + goto done; + } + + do + *err = posix_spawn(pid, argv[0], &acts, &attrs, argv, env); + while (*err == EINTR); +done: + posix_spawn_file_actions_destroy(&acts); + posix_spawnattr_destroy(&attrs); + + for (; count > 0; count--) { + close(low_fds[count]); + } +} +#endif + +/** + * Init + */ + +Napi::Object init(Napi::Env env, Napi::Object exports) { + exports.Set("fork", Napi::Function::New(env, PtyFork)); + exports.Set("open", Napi::Function::New(env, PtyOpen)); + exports.Set("resize", Napi::Function::New(env, PtyResize)); + exports.Set("process", Napi::Function::New(env, PtyGetProc)); + return exports; +} + +NODE_API_MODULE(NODE_GYP_MODULE_NAME, init) diff --git a/config/scripts/build-relay.mjs b/config/scripts/build-relay.mjs index 506036ede3a..289c7a957bd 100644 --- a/config/scripts/build-relay.mjs +++ b/config/scripts/build-relay.mjs @@ -57,6 +57,13 @@ const NODE_PTY_CONSOLE_LIST_PATCH_SOURCE = join( 'relay-assets', NODE_PTY_CONSOLE_LIST_PATCH_FILENAME ) +const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' +const NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE = join( + ROOT, + 'config', + 'relay-assets', + NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME +) // Written by build-windows-process-tree-relay-addon.mjs, which only runs on a // Windows machine. const WINDOWS_PROCESS_TREE_BUILD_DIR = join(ROOT, '.build', 'windows-process-tree') @@ -126,6 +133,10 @@ for (const platform of RELAY_BUILD_PLATFORMS) { join(outDir, NODE_PTY_CONSOLE_LIST_PATCH_FILENAME) ) } + copyFileSync( + NODE_PTY_MASTER_CLOEXEC_PATCH_SOURCE, + join(outDir, NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME) + ) stageWindowsProcessTreeAddon(platform, outDir) await build({ diff --git a/config/scripts/node-pty-master-cloexec-patch.test.mjs b/config/scripts/node-pty-master-cloexec-patch.test.mjs new file mode 100644 index 00000000000..16013cbf0a3 --- /dev/null +++ b/config/scripts/node-pty-master-cloexec-patch.test.mjs @@ -0,0 +1,229 @@ +import { existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { join, resolve } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { + SKIP_MARKER_FILENAME, + applyNodePtyMasterCloexecPatch, + assertPatchedNodePtyMasterCloexecSource, + patchNodePtyMasterCloexecSource, + revertNodePtyMasterCloexecSource +} = require('../relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs') + +// Byte-exact src/unix/pty.cc from the npm tarball the relay installs. The patch is keyed by its +// sha256, so a fixture that drifted from what npm ships would make every assertion below vacuous. +const STOCK_SOURCE = readFileSync( + resolve(import.meta.dirname, '__fixtures__', 'node-pty-1.1.0-unix-pty.cc'), + 'utf8' +) +const projectDir = resolve(import.meta.dirname, '..', '..') +const cleanupDirs = [] + +afterEach(() => { + for (const dir of cleanupDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +describe('SSH relay node-pty pty-master close-on-exec patch', () => { + it('adds the forkpty close-on-exec call and reverts to the published bytes', () => { + const fixture = writeRelayFixture() + + expect(patchNodePtyMasterCloexecSource(fixture.root)).toBe(true) + const patched = readFileSync(fixture.sourcePath, 'utf8') + expect(patched).toContain('pty_cloexec(int fd)') + expect(patched).toContain('if (pty_cloexec(master) == -1)') + expect(() => assertPatchedNodePtyMasterCloexecSource(fixture.root)).not.toThrow() + + expect(patchNodePtyMasterCloexecSource(fixture.root)).toBe(false) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(patched) + + expect(revertNodePtyMasterCloexecSource(fixture.root)).toBe(true) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + }) + + it('refuses a different node-pty version or an unrecognized source', () => { + const wrongVersion = writeRelayFixture({ version: '1.2.0-beta.4' }) + expect(() => patchNodePtyMasterCloexecSource(wrongVersion.root)).toThrow('expected 1.1.0') + + const drifted = writeRelayFixture({ + source: `${STOCK_SOURCE}\n// drift\n` + }) + expect(() => patchNodePtyMasterCloexecSource(drifted.root)).toThrow('unexpected node-pty') + + const tampered = writeRelayFixture() + patchNodePtyMasterCloexecSource(tampered.root) + writeFileSync(tampered.sourcePath, `${readFileSync(tampered.sourcePath, 'utf8')}\n// drift\n`) + expect(() => assertPatchedNodePtyMasterCloexecSource(tampered.root)).toThrow('not installed') + }) + + it('keeps the rebuilt addon once a later child no longer inherits the master', () => { + const fixture = writeRelayFixture() + const calls = [] + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => { + calls.push('rebuild') + writeBuild(fixture, 'patched-build') + }, + verify: () => 'isolated' + }) + + expect(status).toBe('patched') + expect(calls).toEqual(['rebuild']) + expect(readFileSync(fixture.buildPath, 'utf8')).toBe('patched-build') + expect(readFileSync(fixture.sourcePath, 'utf8')).not.toBe(STOCK_SOURCE) + expect(existsSync(fixture.backupDir)).toBe(false) + expect(existsSync(fixture.skipMarkerPath)).toBe(false) + }) + + it('keeps a rebuilt addon whose flag /proc could not confirm', () => { + const fixture = writeRelayFixture() + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => writeBuild(fixture, 'patched-build'), + verify: () => 'unverified' + }) + + expect(status).toBe('patched-unverified') + expect(readFileSync(fixture.buildPath, 'utf8')).toBe('patched-build') + }) + + it('restores the working build when the compile fails, and never retries it', () => { + const fixture = writeRelayFixture() + const calls = [] + + const failed = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => { + calls.push('rebuild') + throw new Error('npm rebuild node-pty exited 1: no C++ toolchain') + }, + verify: () => 'isolated' + }) + + expect(failed).toContain('failed:') + expect(failed).toContain('no C++ toolchain') + expect(readFileSync(fixture.buildPath, 'utf8')).toBe('stock-build') + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + expect(existsSync(fixture.backupDir)).toBe(false) + expect(existsSync(fixture.skipMarkerPath)).toBe(true) + + // Bounded, not backed off: a relay directory gets one compile attempt, ever. + const again = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => calls.push('rebuild'), + verify: () => 'isolated' + }) + expect(again).toBe('skipped:earlier-attempt-failed') + expect(calls).toEqual(['rebuild']) + }) + + it('restores the working build when the rebuilt addon still leaks the master', () => { + const fixture = writeRelayFixture() + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => writeBuild(fixture, 'still-leaky-build'), + verify: () => { + throw new Error('rebuilt node-pty still leaks the pty master into later children') + } + }) + + expect(status).toContain('still leaks') + expect(readFileSync(fixture.buildPath, 'utf8')).toBe('stock-build') + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + }) + + it('never compiles on a platform that does not leak', () => { + for (const platform of ['darwin', 'win32']) { + const fixture = writeRelayFixture() + const calls = [] + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform, + rebuild: () => calls.push('rebuild'), + verify: () => 'isolated' + }) + expect(status).toBe('skipped:not-linux') + expect(calls).toEqual([]) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + } + }) + + it('leaves an already patched install alone', () => { + const fixture = writeRelayFixture() + patchNodePtyMasterCloexecSource(fixture.root) + const calls = [] + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => calls.push('rebuild'), + verify: () => 'isolated' + }) + + expect(status).toBe('already-patched') + expect(calls).toEqual([]) + }) + + it('will not rebuild an install that has no compiled addon to fall back on', () => { + const fixture = writeRelayFixture({ build: false }) + const calls = [] + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => calls.push('rebuild'), + verify: () => 'isolated' + }) + + expect(status).toBe('skipped:no-compiled-build') + expect(calls).toEqual([]) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + }) + + it('discards a backup stranded by an interrupted rebuild', () => { + const fixture = writeRelayFixture() + mkdirSync(fixture.backupDir, { recursive: true }) + writeFileSync(join(fixture.backupDir, 'pty.node'), 'stranded-build') + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'linux', + rebuild: () => writeBuild(fixture, 'patched-build'), + verify: () => 'isolated' + }) + + expect(status).toBe('patched') + expect(existsSync(fixture.backupDir)).toBe(false) + expect(readFileSync(fixture.buildPath, 'utf8')).toBe('patched-build') + }) +}) + +function writeRelayFixture({ version = '1.1.0', source = STOCK_SOURCE, build = true } = {}) { + const root = mkdtempSync(join(projectDir, '.node-pty-cloexec-patch-test-')) + cleanupDirs.push(root) + const nodePtyDir = join(root, 'node_modules', 'node-pty') + const sourcePath = join(nodePtyDir, 'src', 'unix', 'pty.cc') + const buildPath = join(nodePtyDir, 'build', 'Release', 'pty.node') + mkdirSync(join(nodePtyDir, 'src', 'unix'), { recursive: true }) + writeFileSync(join(nodePtyDir, 'package.json'), JSON.stringify({ version })) + writeFileSync(sourcePath, source) + const fixture = { + root, + sourcePath, + buildPath, + backupDir: join(nodePtyDir, '.orca-cloexec-prepatch-release'), + skipMarkerPath: join(root, SKIP_MARKER_FILENAME) + } + if (build) { + writeBuild(fixture, 'stock-build') + } + return fixture +} + +function writeBuild(fixture, contents) { + mkdirSync(resolve(fixture.buildPath, '..'), { recursive: true }) + writeFileSync(fixture.buildPath, contents) +} diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index 54c64a80b20..fc65af35c0a 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -727,6 +727,31 @@ function uploadStageNamespaceIfSupported( const NODE_PTY_VERSION = '1.1.0' const NODE_PTY_CONSOLE_LIST_PATCH_FILENAME = 'node-pty-1.1.0-console-list-agent-patch.cjs' +const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' +const NODE_PTY_CLOEXEC_STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' +/** + * Whether the tree the patch left behind still leaks the pty master into every later child. + * `fixed` is the only outcome a shared cache entry may be published from. + */ +type NodePtyMasterCloexecOutcome = 'fixed' | 'unfixed' +/** + * The statuses that leave a non-leaking tree. Deliberately an allowlist, not a `failed:` denylist: + * the script's `skipped:` family is mixed. `skipped:not-linux` is a platform that never leaks, but + * `skipped:earlier-attempt-failed`, `skipped:no-compiled-build`, `skipped:unexpected-source` and + * the two `skipped:` forms all mean the patch was refused and the leaky build is still on + * disk -- indistinguishable from `failed:` as far as what gets published. + */ +const NODE_PTY_CLOEXEC_FIXED_STATUSES: ReadonlySet = new Set([ + 'patched', + // The rebuild ran from patched source; only the isolation check could not observe the result. + // An unobservable check is not a failed patch, and treating it as one would disable the shared + // cache on every host without `lsof`. + 'patched-unverified', + 'already-patched', + // Unreachable while the platform gate below short-circuits first, but it is the one `skipped:` + // that means "nothing to fix" rather than "would not fix it". + 'skipped:not-linux' +]) // Exported for the relay-native-dependency-coverage test, which asserts every // native addon the relay bundle imports is either installed here or explicitly // declared as degrading without it. @@ -1213,10 +1238,35 @@ async function installNativeDeps( } } + // Why this precedes promotion: the patch renames `node-pty/build/Release`, runs `npm rebuild` + // and rolls back inside `node_modules`, and promotion turns that directory into a symlink to a + // published -- and by contract immutable -- shared cache entry. Patching afterwards would write + // through the link, and `.deps-complete` would already have published an unpatched tree that + // every later host links and skips. + const cloexec = probe.available + ? await applyNodePtyMasterCloexecPatch( + conn, + remoteDir, + platform, + hostPlatform, + nodePath, + signal + ) + : 'unfixed' + // Why promotion is gated on the probe and not on npm's exit code: an entry is shared, so the // only evidence worth publishing is this host having loaded both addons out of that tree. + // Why it is gated on the patch too: a refused or rolled-back patch leaves the pre-patch leaky + // build in place, and the cache key hashes this patch's bytes -- so publishing it would hand + // every later host on the machine a tree that links, probes loadable, and skips patching. if (probe.available && cacheContext && cache) { - await promoteRelayNativeDepsCache(conn, cacheContext, cache.key) + if (cloexec === 'fixed') { + await promoteRelayNativeDepsCache(conn, cacheContext, cache.key) + } else { + console.warn( + `[ssh-relay][NPTY-CLOEXEC-UNSHARED] keeping the native deps at ${remoteDir} (${platform}) private; the tree still leaks the pty master, so it is not publishable as ${cache.key}` + ) + } } // MISSING is non-fatal by design: the relay still serves fs/git/preflight; only native-backed ops fail on hosts that can't build the addons. @@ -1227,6 +1277,76 @@ async function installNativeDeps( } } +/** + * Re-apply the pty-master FD_CLOEXEC patch the app gets from pnpm to the host's npm copy (#17915). + * + * Why it is safe to rebuild under a live relay: this only runs from installNativeDeps, so only on a + * freshly created directory or a locked repair, and a relay already serving PTYs has pty.node mapped + * -- replacing the file on disk does not touch the running process. It keeps the build it started + * with and picks up the patched one when it restarts. + * + * Why it is bounded: the remote script attempts the compile at most once per relay directory, and + * the directory is content-hashed over the relay manifest -- so at most one compile per bundle. + * + * Why a shared cache entry never reaches here: the caller returns as soon as a linked tree probes + * loadable, so this only ever rewrites a `node_modules` the relay directory still owns privately. + * + * Returns whether the tree that is left behind still leaks, which is what decides publishability. + * The script exits 0 on every outcome by design, so the status line is the only evidence there is. + */ +async function applyNodePtyMasterCloexecPatch( + conn: SshConnection, + remoteDir: string, + platform: RelayPlatform, + hostPlatform: RemoteHostPlatform, + nodePath: string, + signal?: AbortSignal +): Promise { + // Linux is the only relay platform that takes forkpty()'s no-O_CLOEXEC path; macOS and Windows + // ship prebuilds, so forcing a rebuild there would add a first compile to fix nothing. + if (isWindowsRemoteHost(hostPlatform) || !platform.startsWith('linux')) { + return 'fixed' + } + try { + const command = commandWithNodePath( + hostPlatform, + nodePath, + remoteDir, + `${shellEscape(nodePath)} ${shellEscape(NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME)} 2>&1` + ) + const output = await execHostCommand(conn, hostPlatform, command, { + timeoutMs: NATIVE_DEPS_COMMAND_TIMEOUT_MS, + signal + }) + const status = + output + .split(/\r?\n/) + .map((line) => line.trim()) + .find((line) => line.startsWith(NODE_PTY_CLOEXEC_STATUS_PREFIX)) + ?.slice(NODE_PTY_CLOEXEC_STATUS_PREFIX.length) ?? 'no-status' + if (!NODE_PTY_CLOEXEC_FIXED_STATUSES.has(status)) { + // Warn, not log: the script exits 0 on a refusal too, so this line is the only thing that + // says the relay directory will leak a master into every child for its whole life. + console.warn( + `[ssh-relay][NPTY-CLOEXEC-UNFIXED] pty master still leaks at ${remoteDir} (${platform}): ${status}` + ) + return 'unfixed' + } + console.log(`[ssh-relay][NPTY-CLOEXEC] ${remoteDir} (${platform}): ${status}`) + return 'fixed' + } catch (err) { + signal?.throwIfAborted() + // Never fatal: the script restores the working build itself, and a leaky relay beats none. An + // interrupted rebuild leaves node-pty unloadable, which the existing repair path reinstalls. + console.warn( + `[ssh-relay][NPTY-CLOEXEC-FAIL] pty master cloexec patch failed at ${remoteDir} (${platform}): ${(err as Error).message}` + ) + // An exec that never answered cannot say which build is on disk, and a tree nobody can vouch + // for is exactly the one not to share. + return 'unfixed' + } +} + /** * Drop a shared-cache symlink before anything writes into `node_modules`. * diff --git a/src/main/ssh/ssh-relay-native-deps-cache-deploy.test.ts b/src/main/ssh/ssh-relay-native-deps-cache-deploy.test.ts index 61e6133b9a5..cafc82960f3 100644 --- a/src/main/ssh/ssh-relay-native-deps-cache-deploy.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-cache-deploy.test.ts @@ -85,6 +85,9 @@ import { } from './ssh-relay-native-deps-cache-commands' // Everything after the probe on a healthy install: stderr cleanup, stage cleanup, launch. +// Stdout of the relay-side pty-master cloexec patch, which runs on Linux hosts once a +// freshly installed node-pty loads (#17915). +const NPTY_CLOEXEC_PATCHED = 'ORCA-NPTY-CLOEXEC:patched\n' const LAUNCH_TAIL: ExecResponse[] = ['', 'DEAD', '', 'READY'] describe('relay native-deps cache on the deploy path', () => { @@ -165,6 +168,7 @@ describe('relay native-deps cache on the deploy path', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, RELAY_NATIVE_CACHE_PROMOTED, ...LAUNCH_TAIL ]) @@ -214,6 +218,7 @@ describe('relay native-deps cache on the deploy path', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, '', // publication is attempted and answers nothing ...LAUNCH_TAIL ]) @@ -235,6 +240,7 @@ describe('relay native-deps cache on the deploy path', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, '', // promotion attempt (the entry already exists, so it is declined) ...LAUNCH_TAIL ]) @@ -264,6 +270,7 @@ describe('relay native-deps cache on the deploy path', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // publish the per-launch credential 'READY' diff --git a/src/main/ssh/ssh-relay-native-deps-cache.ts b/src/main/ssh/ssh-relay-native-deps-cache.ts index 934aaff63a6..8400a467a91 100644 --- a/src/main/ssh/ssh-relay-native-deps-cache.ts +++ b/src/main/ssh/ssh-relay-native-deps-cache.ts @@ -23,6 +23,12 @@ * Windows is deliberately excluded. node-pty's npm tarball ships win32 prebuilts, so there is no * compile to avoid there, and `node-pty-1.1.0-console-list-agent-patch.cjs` mutates the installed * tree in place — which rule 1 forbids for a shared one. + * + * Linux's `node-pty-1.1.0-master-cloexec-patch.cjs` also mutates in place, but it stays inside rule + * 1: the deploy path runs it before promotion, and returns early on a linked entry, so it only ever + * touches a private tree. Its bytes are in the key, so a patched build never links a pre-patch + * entry -- and a tree whose patch was refused or rolled back is not promoted at all, because under + * that same key it would publish the leak to every later host on the machine. */ import { createHash } from 'node:crypto' import { RELAY_REMOTE_DIR } from './relay-protocol' diff --git a/src/main/ssh/ssh-relay-native-deps-install-fixture.ts b/src/main/ssh/ssh-relay-native-deps-install-fixture.ts index 922b8bedf61..5cd20a03438 100644 --- a/src/main/ssh/ssh-relay-native-deps-install-fixture.ts +++ b/src/main/ssh/ssh-relay-native-deps-install-fixture.ts @@ -15,6 +15,9 @@ export type SftpWriteCapture = { type SftpCallback = (err: Error | null, resolved?: string) => void const NO_SUCH_SFTP_FILE = Object.assign(new Error('No such file'), { code: 2 }) +// Stdout of the relay-side pty-master cloexec patch; kept as a literal so the fixture states the +// wire token it is standing in for rather than importing the module under test. +const NODE_PTY_CLOEXEC_STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' export function makeMockConnection(capture: SftpWriteCapture): SshConnection { // Why: production attaches/removes real listeners (including prependOnceListener), so the fake must be an emitter. @@ -196,6 +199,9 @@ export function makeExecResponses(opts: { } // Publication is gated on the probe: only a tree this host actually loaded is shared. if (loadable) { + // The cloexec patch runs first, and publication is gated on its status, so `patched` is what + // makes the promote exec below reachable at all. + slots.push(`${NODE_PTY_CLOEXEC_STATUS_PREFIX}patched\n`) slots.push('') // promote the private tree into the shared native-deps cache } slots.push('', 'DEAD', '', 'READY') // clean stage root, launch, credential, readiness diff --git a/src/main/ssh/ssh-relay-native-deps-install.test.ts b/src/main/ssh/ssh-relay-native-deps-install.test.ts index 16144fb2cf3..01625816170 100644 --- a/src/main/ssh/ssh-relay-native-deps-install.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-install.test.ts @@ -630,6 +630,7 @@ describe('installNativeDeps (via deployAndLaunchRelay)', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + 'ORCA-NPTY-CLOEXEC:patched\n', // pty-master cloexec patch on the loadable node-pty 'DEAD', '', // publish the per-launch credential 'READY' diff --git a/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts b/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts index df946dfe346..3939c954f65 100644 --- a/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts @@ -85,6 +85,9 @@ import { type SftpWriteCapture } from './ssh-relay-native-deps-install-fixture' +// Stdout of the relay-side pty-master cloexec patch, which runs on Linux hosts once a +// freshly installed node-pty loads (#17915). +const NPTY_CLOEXEC_PATCHED = 'ORCA-NPTY-CLOEXEC:patched\n' const NODE_PTY_RESET = "rm -rf 'node_modules/node-pty'" const WATCHER_RESET = "rm -rf 'node_modules/@parcel/watcher'" @@ -225,6 +228,7 @@ describe('native-deps repair probe verdicts', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // publish the per-launch credential 'READY' @@ -281,6 +285,7 @@ describe('native-deps repair probe verdicts', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // publish the per-launch credential 'READY' @@ -324,6 +329,7 @@ describe('native-deps repair probe verdicts', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // publish the per-launch credential 'READY' diff --git a/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts b/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts index 12a51ce6bdd..8981b6319a8 100644 --- a/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts +++ b/src/main/ssh/ssh-relay-node-pty-spawn-repair.test.ts @@ -103,6 +103,9 @@ const ABI_MISMATCH: TerminalUnavailableCause = { // The relay dir is complete but node-pty will not load, which is exactly what the spawn-time cause // describes. @parcel/watcher is healthy, so only node-pty is reset and rebuilt. +// Stdout of the relay-side pty-master cloexec patch, which runs on Linux hosts once a +// freshly installed node-pty loads (#17915). +const NPTY_CLOEXEC_PATCHED = 'ORCA-NPTY-CLOEXEC:patched\n' const NODE_PTY_BROKEN = 'ORCA-NATIVE-DEPS-MISSING:node-pty\nMISSING' function repairSucceedsResponses(): ExecResponse[] { @@ -116,6 +119,7 @@ function repairSucceedsResponses(): ExecResponse[] { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', // node-pty loads again '', // rm -f probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // publish the per-launch credential 'READY' diff --git a/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts b/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts new file mode 100644 index 00000000000..66af01fa43c --- /dev/null +++ b/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts @@ -0,0 +1,336 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as RelayInstallMarkerModule from './ssh-relay-install-marker' + +vi.mock('electron', () => ({ + app: { getAppPath: () => '/mock/app' } +})) + +vi.mock('fs', () => ({ + existsSync: vi.fn().mockReturnValue(true), + readFileSync: vi.fn().mockReturnValue('0.1.0+testhash') +})) + +vi.mock('./relay-protocol', () => ({ + RELAY_VERSION: '0.1.0', + RELAY_REMOTE_DIR: '.orca-remote', + parseUnameToRelayPlatform: vi.fn().mockReturnValue('linux-x64'), + RELAY_SENTINEL: 'ORCA-RELAY v0.1.0 READY\n', + RELAY_SENTINEL_TIMEOUT_MS: 10_000 +})) + +vi.mock('./ssh-relay-deploy-helpers', () => ({ + uploadDirectory: vi.fn().mockResolvedValue(undefined), + waitForSentinel: vi.fn().mockResolvedValue({ + write: vi.fn(), + onData: vi.fn(), + onClose: vi.fn() + }), + isUnconfirmedSshCommandTermination: (error: unknown) => + error instanceof Error && + (error as Error & { sshChannelCloseConfirmed?: boolean }).sshChannelCloseConfirmed === false, + execCommand: vi.fn() +})) + +vi.mock('./ssh-remote-node-resolution', () => ({ + resolveRemoteNodePath: vi.fn().mockResolvedValue('/usr/bin/node') +})) + +vi.mock('./ssh-relay-install-marker', async (importOriginal) => ({ + ...(await importOriginal()), + createRelayInstallMarkerFileName: () => '.sftp-namespace-00000000000000000000000000000000' +})) + +vi.mock('./ssh-relay-versioned-install', () => ({ + readLocalFullVersion: vi.fn().mockReturnValue('0.1.0+testhash'), + computeRemoteRelayDir: (home: string, v: string) => `${home}/.orca-remote/relay-${v}`, + isRelayAlreadyInstalled: vi.fn().mockResolvedValue(false), + finalizeInstall: vi.fn().mockResolvedValue(undefined), + abandonInstall: vi.fn().mockResolvedValue(undefined), + gcOldRelayVersions: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-relay-install-lock', () => ({ + acquireInstallLock: vi.fn().mockResolvedValue(undefined), + RELAY_INSTALL_LOCK_NAME: '.install-lock' +})) + +vi.mock('./ssh-relay-repair-lock', () => ({ + tryAcquireRelayRepairLock: vi.fn().mockResolvedValue('acquired') +})) + +vi.mock('./ssh-relay-gc-claim', () => ({ + releaseRelayGcClaimWithRetry: vi.fn().mockResolvedValue('released'), + tryAcquireRelayGcClaim: vi.fn().mockResolvedValue('launch-token'), + waitForRelayGcClaimRelease: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-connection-utils', () => ({ + shellEscape: (s: string) => `'${s}'` +})) + +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import { execCommand } from './ssh-relay-deploy-helpers' +import { parseUnameToRelayPlatform } from './relay-protocol' +import { + makeExecResponses, + makeStagedFirstInstallExecPrefix, + makeMockConnection, + type ExecResponse, + type SftpWriteCapture +} from './ssh-relay-native-deps-install-fixture' +import { RELAY_NATIVE_CACHE_LINKED } from './ssh-relay-native-deps-cache-commands' +import { RELAY_ARTIFACTS } from '../../shared/relay-artifacts' +import { + computeRelayNativeDepsCacheKey, + RELAY_NATIVE_DEPS_PATCH_ARTIFACT_PATTERN +} from './ssh-relay-native-deps-cache' + +const PATCH_ASSET = 'node-pty-1.1.0-master-cloexec-patch.cjs' + +/** + * The relay installs stock node-pty from npm, so the app's pnpm patch never reaches it and every + * later child of the relay inherits a live pty master (#17915). The compile that closes it sits on + * the connect path, so what these specs pin is the blast radius, not the patch itself. + */ +describe('relay pty-master close-on-exec patch on the install path', () => { + const sftpCapture: SftpWriteCapture = { + paths: [], + contents: {}, + execCallCountAtWrite: {} + } + + beforeEach(() => { + vi.clearAllMocks() + vi.mocked(execCommand).mockReset().mockResolvedValue('') + sftpCapture.paths.length = 0 + vi.mocked(parseUnameToRelayPlatform).mockReturnValue('linux-x64') + }) + + function feed(execResponses: ExecResponse[]): void { + const mockExec = vi.mocked(execCommand) + for (const response of execResponses) { + if (typeof response === 'string') { + mockExec.mockResolvedValueOnce(response) + } else { + mockExec.mockRejectedValueOnce(new Error(response.reject)) + } + } + } + + function firstInstall(cacheAnswer: string, tail: ExecResponse[]): ExecResponse[] { + const prefix = makeStagedFirstInstallExecPrefix() + // The prefix's last slot is the shared native-deps cache probe. + prefix[prefix.length - 1] = cacheAnswer + return [...prefix, ...tail] + } + + function patchCommands(): string[] { + return vi + .mocked(execCommand) + .mock.calls.map(([, command]) => command) + .filter((command) => command.includes(PATCH_ASSET)) + } + + /** Whether this deploy elected itself publisher of the shared entry. */ + function promoted(): boolean { + return vi + .mocked(execCommand) + .mock.calls.some(([, command]) => command.includes('mkdir "$cache"')) + } + + /** + * A cache-miss first install whose patch reports `status`. The promote slot is fed either way, + * so a run that wrongly promotes reads a valid response rather than falling off the end -- the + * assertion has to be the absence of the command itself, not a downstream crash. + */ + function firstInstallReporting(status: string): ExecResponse[] { + return [ + ...makeStagedFirstInstallExecPrefix(), + '', // npm install native deps + '', // chmod prebuilds + 'ORCA-NPTY-PROBE-OK\n', + '', // rm probe stderr + `ORCA-NPTY-CLOEXEC:${status}\n`, + '', // promote into the shared native-deps cache, if this deploy still gets that far + '', // clean stage root + 'DEAD', + '', // publish the per-launch credential + 'READY' + ] + } + + it('runs the patch on a Linux relay once node-pty is proven loadable', async () => { + const conn = makeMockConnection(sftpCapture) + feed(makeExecResponses({ npmInstall: 'ok', probe: 'ok' })) + + await deployAndLaunchRelay(conn) + + expect(patchCommands()).toHaveLength(1) + expect(patchCommands()[0]).toContain("'/usr/bin/node'") + }) + + it('patches the private tree before it is published to the shared native-deps cache', async () => { + // Promotion moves `node_modules` into `~/.orca-remote/native/` and leaves a symlink + // behind, and a published entry is immutable by contract. Patching afterwards would rename, + // rebuild and roll back inside a tree every other relay on the host links -- and the + // `.deps-complete` written by promotion would have published an unpatched tree that every + // later host links and skips. The ordering is invisible in review, so pin it. + const conn = makeMockConnection(sftpCapture) + feed(makeExecResponses({ npmInstall: 'ok', probe: 'ok' })) + + await deployAndLaunchRelay(conn) + + const commands = vi.mocked(execCommand).mock.calls.map(([, command]) => command) + const patchAt = commands.findIndex((command) => command.includes(PATCH_ASSET)) + const promoteAt = commands.findIndex((command) => command.includes('mkdir "$cache"')) + expect(patchAt).toBeGreaterThan(-1) + expect(promoteAt).toBeGreaterThan(-1) + expect(patchAt).toBeLessThan(promoteAt) + }) + + it('does not publish a tree whose patch failed and rolled back', async () => { + // The script rolls `pty.cc` and `build/Release` back to the pre-patch, still-leaky build and + // reports `failed:` with exit 0, so nothing throws. Publishing that tree would be worse than + // the leak this PR closes: the key hashes the patch's bytes, so every later host on the + // machine links the entry, probes it loadable, and skips patching. Stay private instead. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + const conn = makeMockConnection(sftpCapture) + feed(firstInstallReporting('failed:npm rebuild node-pty failed: gyp ERR! not found: make')) + + await deployAndLaunchRelay(conn) + + expect(patchCommands()).toHaveLength(1) + expect(promoted()).toBe(false) + expect(warn.mock.calls.map((args) => String(args[0] ?? '')).join('\n')).toContain( + '[ssh-relay][NPTY-CLOEXEC-UNSHARED]' + ) + } finally { + warn.mockRestore() + } + }) + + it('does not publish a tree the patch refused to touch', async () => { + // `skipped:` is not one verdict. Every form except `skipped:not-linux` means the patch was + // declined and the leaky build is still on disk, which is indistinguishable from `failed:` + // as far as what would get published. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + const conn = makeMockConnection(sftpCapture) + feed(firstInstallReporting('skipped:earlier-attempt-failed')) + + await deployAndLaunchRelay(conn) + + expect(promoted()).toBe(false) + // A refusal exits 0, so the warn is the only signal that this host stayed leaky. + expect(warn.mock.calls.map((args) => String(args[0] ?? '')).join('\n')).toContain( + '[ssh-relay][NPTY-CLOEXEC-UNFIXED]' + ) + } finally { + warn.mockRestore() + } + }) + + it('still publishes a tree that was patched but whose isolation check could not run', async () => { + // `patched-unverified` rebuilt from patched source; only the check that watches a later child + // could not observe the result. An unobservable check is not a failed patch, and refusing to + // publish here would disable the shared cache on every host without `lsof`. + const conn = makeMockConnection(sftpCapture) + feed(firstInstallReporting('patched-unverified')) + + await deployAndLaunchRelay(conn) + + expect(promoted()).toBe(true) + }) + + it('never patches through a symlink into an entry another relay already published', async () => { + // A linked entry was built under a key that hashes this patch's bytes, so it is already + // patched; re-running the patch would rebuild inside the shared tree. + const conn = makeMockConnection(sftpCapture) + feed( + firstInstall(RELAY_NATIVE_CACHE_LINKED, [ + '', // chmod prebuilds, through the symlink + 'ORCA-NPTY-PROBE-OK\n', + '', // rm probe stderr + '', // clean stage root + 'DEAD', + '', // publish the per-launch credential + 'READY' + ]) + ) + + await deployAndLaunchRelay(conn) + + expect(patchCommands()).toEqual([]) + }) + + it('leaves an unloadable node-pty alone rather than rebuilding it blind', async () => { + // A relay that could not build node-pty has nothing to fall back to, and the existing + // reinstall path owns that repair. + const conn = makeMockConnection(sftpCapture) + feed( + makeExecResponses({ + npmInstall: 'ok', + probe: 'missing', + repairProbe: 'missing' + }) + ) + + await deployAndLaunchRelay(conn) + + expect(patchCommands()).toEqual([]) + }) + + it('never adds a compile to a macOS relay, which does not leak the master', async () => { + vi.mocked(parseUnameToRelayPlatform).mockReturnValue('darwin-arm64') + const conn = makeMockConnection(sftpCapture) + feed([ + ...makeStagedFirstInstallExecPrefix(), + '', // npm install native deps + '', // chmod prebuilds + 'ORCA-NPTY-PROBE-OK\n', + '', // rm probe stderr + '', // promote into the shared native-deps cache + '', // clean stage root + 'DEAD', + '', // publish the per-launch credential + 'READY' + ]) + + await deployAndLaunchRelay(conn) + + expect(patchCommands()).toEqual([]) + }) + + it('connects anyway when the patch command fails outright', async () => { + const conn = makeMockConnection(sftpCapture) + const responses = makeExecResponses({ npmInstall: 'ok', probe: 'ok' }) + const patchSlot = responses.findIndex( + (response) => typeof response === 'string' && response.includes('ORCA-NPTY-CLOEXEC:') + ) + expect(patchSlot).toBeGreaterThan(-1) + responses[patchSlot] = { reject: 'no such file or directory' } + feed(responses) + + await expect(deployAndLaunchRelay(conn)).resolves.toBeDefined() + }) +}) + +describe('the shipped patch is part of the shared native-deps cache key', () => { + it('mints a new entry, so a pre-fix unpatched tree is never linked by a patched build', () => { + const artifact = RELAY_ARTIFACTS.find((entry) => entry.filename === PATCH_ASSET) + expect(artifact).toBeDefined() + // A windowsOnly artifact never reaches a Linux relay dir, so it would drop out of the key. + expect(artifact?.windowsOnly).toBeFalsy() + expect(RELAY_NATIVE_DEPS_PATCH_ARTIFACT_PATTERN.test(PATCH_ASSET)).toBe(true) + + const deps = { 'node-pty': '1.1.0' } + expect( + computeRelayNativeDepsCacheKey({ + platform: 'linux-x64', + deps, + patchSources: [{ filename: PATCH_ASSET, contents: 'patch bytes' }] + }) + ).not.toBe(computeRelayNativeDepsCacheKey({ platform: 'linux-x64', deps })) + }) +}) diff --git a/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts b/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts index 10ebce2dac4..81142743cd1 100644 --- a/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts +++ b/src/main/ssh/ssh-relay-sftp-namespace-install.test.ts @@ -105,6 +105,9 @@ const RELAY_SUFFIX = '.orca-remote/relay-0.1.0+testhash' const SHELL_RELAY_DIR = `${SHELL_HOME}/${RELAY_SUFFIX}` const SFTP_RELAY_DIR = `${SFTP_HOME}/${RELAY_SUFFIX}` const MARKER_PATTERN = /\.sftp-namespace-[0-9a-f]{32}/ +// Stdout of the relay-side pty-master cloexec patch, which runs on Linux hosts once a +// freshly installed node-pty loads (#17915). +const NPTY_CLOEXEC_PATCHED = 'ORCA-NPTY-CLOEXEC:patched\n' const STAGE_OWNER = '.sftp-namespace-00000000000000000000000000000000' const STAGE_RESERVED = `__ORCA_UPLOAD_STAGE_SLOT__${STAGE_OWNER}:slot-0` const STAGE_PROMOTED = `__ORCA_UPLOAD_STAGE_PROMOTION__${STAGE_OWNER}:PROMOTED` @@ -256,6 +259,7 @@ const POSIX_FIRST_INSTALL = [ '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, '', // promote into the shared native-deps cache '', // clean stage root 'DEAD', @@ -275,6 +279,7 @@ const POSIX_SYSTEM_SSH_FIRST_INSTALL = [ '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, '', // promote into the shared native-deps cache '', // clean stage root 'DEAD', @@ -293,6 +298,7 @@ const POSIX_REPAIR = [ '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // publish the per-launch credential 'READY' @@ -703,6 +709,7 @@ describe('relay repair writes on a split SFTP namespace', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // remote credential generation 'READY' @@ -733,6 +740,7 @@ describe('relay repair writes on a split SFTP namespace', () => { '', // chmod prebuilds 'ORCA-NPTY-PROBE-OK\n', '', // rm probe stderr + NPTY_CLOEXEC_PATCHED, 'DEAD', '', // remote credential generation 'READY' diff --git a/src/shared/relay-artifacts.ts b/src/shared/relay-artifacts.ts index 18c88135a62..273f6e059b8 100644 --- a/src/shared/relay-artifacts.ts +++ b/src/shared/relay-artifacts.ts @@ -51,6 +51,11 @@ export const RELAY_ARTIFACTS: readonly RelayArtifact[] = [ // title request with no title and no error. { filename: 'wsl-transcript-fs-process-entry.js' }, { filename: 'node-pty-1.1.0-console-list-agent-patch.cjs', windowsOnly: true }, + // Only Linux relays run it, but it ships everywhere: the manifest's only + // platform axis is Windows, and a second one would buy nothing but a fork in + // the hash. Its presence is what moves a host to a fresh relay directory, and + // therefore to a re-install that can apply it. + { filename: 'node-pty-1.1.0-master-cloexec-patch.cjs' }, // Optional because only a Windows build machine can compile it. Without it the // relay reads the process table through a PowerShell scan instead -- slower, // but correct, so a relay built anywhere else is still shippable. diff --git a/src/shared/relay-optional-artifacts.test.ts b/src/shared/relay-optional-artifacts.test.ts index 74f3530b2d1..0b8b750dbf9 100644 --- a/src/shared/relay-optional-artifacts.test.ts +++ b/src/shared/relay-optional-artifacts.test.ts @@ -31,4 +31,12 @@ describe('optional relay artifacts', () => { expect(relayArtifactFilenames(true)).toContain('relay.js') expect(relayArtifactFilenames(true)).toContain('node-pty-1.1.0-console-list-agent-patch.cjs') }) + + it('ships the pty-master cloexec patch to every platform', () => { + // Only Linux runs it, but its bytes are what change the relay content hash, and therefore what + // moves an upgrading host to a fresh directory whose install can apply it (#17915). + for (const isWindows of [true, false]) { + expect(relayArtifactFilenames(isWindows)).toContain('node-pty-1.1.0-master-cloexec-patch.cjs') + } + }) }) From f37d2fec9711c9945600527a95cd605157ea7c6d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 03:08:01 -0700 Subject: [PATCH 063/398] fix(linux): land the reviewed Linux packaging stack on main (#18100) * fix(linux): give the CLI one entrypoint by extracting the AppImage once * refactor(linux): trim AppImage CLI registration seams * test(cli): assert registration lock serialization * fix(linux): fence AppImage terminal shim mounts * fix(linux): accept extracted AppImage runtimes with APPDIR only * docs(linux): make headless AppImage extraction runnable * refactor(linux): import bundled launcher directly * fix(linux): reclaim superseded AppImage payloads and packaged symlinks Pruning removed 3215 of 3216 files from a superseded generation and always stranded resources/app.asar, leaking ~105 MB per version update. Electron's asar shim reports a *.asar file as a directory, so the recursive remove tried to rmdir a real file and failed with ENOTEMPTY; the .catch(() => {}) hid it. Reproduced end to end on Ubuntu 24.04: 519M -> 623M across one update, and 519M again once the payload is actually reclaimed. removeExtractedAppImagePayload holds process.noAsar for the removal, counted so overlapping removals cannot hand the shim back early, and the prune site now warns with the path instead of swallowing the rejection. All three removal sites use it -- staging cleanup and displaced roots leaked the same way. Also reclaim symlinks left by a packaged deb/rpm install, which the extracted-cache-only rule turned into a hard conflict on a deb -> AppImage migration, and name the remedy in the conflict error. * fix(linux): bound the CLI registration lock wait `retries: 1000` caps the attempt count, not elapsed time, so at up to 1s per attempt an IPC-driven registration could hang ~16 minutes against a wedged holder with no feedback. A legitimate holder is bounded by the extraction timeout, so wait that plus slack and then fail with a message naming the lock file, rather than hanging. `maxRetryTime` is forwarded verbatim to the `retry` package by proper-lockfile. * fix(linux): stop re-extracting the AppImage on inode metadata churn The extracted-payload cache key hashed ctime alongside dev/ino/size/mtime. ctime moves on any inode metadata write -- `chmod +x`, which every AppImage user is told to run, plus `chown`, an ACL or SELinux relabel, and a backup restore -- none of which alter a byte of the payload. Measured on Ubuntu 24.04: `chmod +x` leaves dev, ino, size and mtime identical and moves ctime alone, so the key changed and the next launch paid a full ~519 MB re-extraction and a multi-second stall to rebuild a payload it already had, then pruned the old generation. Key on content identity instead. An in-place content change moves mtime and almost always size; a replacement moves the inode. The existing replace-in-place test still passes. * fix(linux): stop CLI commands from falling through to Chromium startup * refactor(cli): remove redundant command membership check * test(cli): cover command-named project selectors * fix(cli): redirect the open-url command before startup * test(linux): cover AUR serve wrapper flags * fix(linux): tighten CLI launch detection * fix(linux): respect CLI flag value boundaries * fix(linux): strip injected Chromium switches from CLI args * fix(linux): report a missing display instead of dying in uv_close * refactor(linux): read display locks without a preflight race * fix(linux): preserve unverified external displays * chore: format reliability gate manifest * test(packaging): split runtime resource checks * fix(linux): fail serve when no display is available * fix(linux): do not treat a lockless X socket as a dead display An X server writes its lock beside its socket and both survive a crash (verified against Xvfb under SIGKILL), so a socket with no lock was never left by a crashed server. It is an endpoint published from elsewhere: a container bind-mounting only /tmp/.X11-unix, WSLg, or a foreign PID namespace. Declaring those dead made the desktop gate exit(1) on displays that work, with no workaround, and the serve gate refuse to start. Liveness now splits by ownership. A foreign DISPLAY trusts a lockless socket; Orca's own :99 does not, because removeStaleDisplayArtifacts unlinks the lock before the socket and so manufactures that state itself -- adopting it would resurrect the orphan-socket bug and stop the cleanup from self-healing. The stale-lock rejection is unchanged. Also correct four doc statements this behaviour falsified. * fix(linux): fail closed when a stale socket blocks the Xvfb rebind Readiness only checked that /tmp/.X11-unix/X99 exists. A stale socket we could not unlink still exists after our own Xvfb refused to bind, so Orca set DISPLAY to a dead server and Chromium died in Ozone init. Measured on Ubuntu 24.04 against the pre-fix build: with a leftover :99 socket and no lock, serve exits 139 (SIGSEGV), the socket inode is unchanged before and after, and no lock is recreated -- it neither cleaned up nor respawned. To a user that is a crash, not a misconfiguration. This is reachable in the documented topology, where orca-xvfb.service has no User= and runs as root while serve runs as User=orca: /tmp is sticky, so the orca uid cannot unlink a root-owned socket, rmSync fails, and Xvfb exits with the display already active. Readiness now requires the display to actually be live -- our socket plus a lock naming a running process -- so the same state reports an unusable display and exits 1 with the existing diagnosis. * fix(linux): recognise abstract X sockets and inherited Wayland fds Two display setups this gate could not prove were refused outright, and on the desktop path that is app.exit(1) with no workaround. An X server may bind only the abstract namespace (`@/tmp/.X11-unix/X0`), which leaves no filesystem socket to stat. Abstract addresses are kernel-owned and vanish the moment the owner exits, so an entry in /proc/net/unix is proof of a live server -- no lock file needed and no stale entry possible. Verified on Ubuntu 24.04, where 139 such addresses were present. WAYLAND_SOCKET is an already-connected fd handed over by the compositor, so there is no path to stat and WAYLAND_DISPLAY may be unset entirely. Its presence is the display. Both are consulted only after the filesystem-socket check fails, so no existing verdict changes. * fix(linux): never treat Orca's own display number as a foreign endpoint Recognising a lockless X socket as live is correct for an endpoint published from elsewhere -- a container bind mount, WSLg -- because an X server writes its lock beside its socket and both survive a crash. It is wrong for VIRTUAL_DISPLAY_NUMBER, because Orca's own teardown unlinks the lock before the socket and so manufactures that exact state. The managed branch was already strict, but a caller that sets DISPLAY=:99 explicitly takes the foreign path and skipped it, accepting a dead display left by Orca's own interrupted cleanup. Route the managed number through the strict probe on both paths. Found by an adversarial audit of the asymmetry introduced earlier in this branch; the documented systemd topology is unaffected because its Xvfb writes a real lock. * test(linux): add a packaged-artifact contract for the CLI launch paths * test(linux): avoid buffered serve readiness detection * test(linux): signal AppImage serve owner directly * test(linux): tolerate readiness timeout boundary * test(linux): add startup margin to shutdown oracle * ci(linux): give package contracts timeout headroom * fix(ci): route all Linux packaging contract changes * test(linux): poll shutdown readiness without tail leaks * test(linux): bound shutdown cleanup grace * test(linux): assert on CLI output, not the harness's own control lines run-cli-case.sh echoes `RESULT status=N case=`, and the two cases named *-skills asserted `expectOutput: 'skills'`. That substring was satisfied by the case name in the harness's own line, so 2 of 8 cases asserted nothing about the command -- gutting `skills` entirely would still have gone green. Control lines are now excluded before matching, and both cases assert the rendered help header, which only real help output produces. Verified on an Ubuntu 24.04 host: 8/8 still pass against a stack-tip AppImage. Also register the gate in reliability-gates.jsonc, which #15085 added a CI Docker gate without. Red/green is recorded from a stock release AppImage failing 4 of 8, three of them at status 133 (SIGTRAP). * fix(linux): require static AppImage runtimes (#17319) * test(linux): reject a wrong-architecture native binary at packaging time Cross-building the arm64 slice on an x64 host silently packed an x86-64 `pty.node` -- the rebuild logged "Forcing native rebuild for linux-arm64" and shipped the host's binary anyway. Every gate here inspects symbol versions, which are perfectly valid on the wrong architecture, so nothing noticed. Observed on a Raspberry Pi 5: the packaged app loaded, then failed with "Failed to load native module: pty.node", and the launch contract reported 3 of 8 cases crashed rather than naming the cause. Swapping in the aarch64 `pty.node` took the same build to 8/8. Compare ELF `e_machine` against the slice being packaged and fail with the offending path. Checked before the glibc pass, because a wrong-architecture binary's symbol versions are valid but meaningless and would send the reader down the wrong path. Release CI builds arm64 on a native runner, so this guards local and future cross-builds rather than a shipped artifact. * test(linux): judge per-arch vendored binaries against their own path The first CI run of the architecture gate failed the x64 package job on `@parcel/watcher-linux-arm64-glibc/watcher.node`. That binary is arm64 on purpose: the package ships every architecture and its loader picks the match, so its presence in an x64 build is correct. Judge a binary against the architecture its own path names, falling back to the slice when the path names none. That keeps the case this gate exists for -- `bin/linux-arm64-*/node-pty.node` holding an x86-64 binary, which is what shipped to a Raspberry Pi 5 -- while letting multi-arch dependencies through. Dry-run over the real dependency tree flags nothing for either target arch. * fix(linux): move deb/rpm update installation outside Orca (#17318) * fix(linux): complete deb/rpm package metadata * fix(linux): preserve CLI link during package upgrades * docs(linux): document local RPM build prerequisites * fix(linux): move deb/rpm update installation outside Orca * fix(updater): preserve Linux recovery across stale events * fix(updater): fence stale downloaded events by active target * fix(updater): preserve active Linux package recovery * test(linux): keep workflow order assertion in scope * test(updater): assert stale recovery stays silent * fix(updater): preserve Linux package recovery after checks * refactor(updater): keep Linux marker message with status * fix(linux): describe the right manual update path for deb/rpm hosts A remote host installed from .deb or .rpm now reports manual-service-update-required, and the guidance told the operator to "update through the service manager that starts this server" -- which is correct for unsupported-headless-serve but wrong for a package install, where nothing about the remedy involves the service manager. Say both, keyed on how the host was installed. * docs(linux): document orcad update restart safety * docs(linux): scope restart census omissions * docs(linux): use absolute service CLI launcher * fix(serve): validate in-process serve options before startup (#17683) * fix(linux): stop offering updates a distro-managed install cannot apply (#17918) Closes #17702. The resources/package-type marker is authoritative but never checked against the host, so any repackager that unpacks Orca's .deb -- AUR, Nix, a container rebuild -- inherits `deb` verbatim. Install feasibility was then computed after a ~165 MB download, so those users got check -> download -> a card promising an install command -> a dead end. Validate the marker against the host: a deb/rpm marker with no matching package manager in the trusted directories means a package manager owns this install. This reuses the exact lists and resolver that buildLinuxPackageInstallCommand already loops over, so a false positive is impossible by construction -- any host flagged here would have failed with no-package-manager after the download anyway. The gate only moves that verdict earlier. Verified across Debian 12, Ubuntu 24.04, Arch, Fedora 40 and openSUSE Leap: no false positive on a real deb host, correct on every repackaging host. The release is still reported, because the user does want to know 1.4.194 exists and to update through their distro; only the download path is closed. `externallyManaged` is an additive optional field on the existing `available` status, so older paired clients decode it unchanged. downloadUpdate() refuses authoritatively, since main owns this verdict rather than the card, and unwinds any pinned-build state first -- a Linux pinned jump resolves to 'release', and stranding isPinnedBuildActive would silently kill every background check for the rest of the process. Note the fix the issue suggests cannot work: electron-updater builds a PacmanUpdater whose doDownloadUpdate looks for a .pacman asset Orca does not publish, then dereferences undefined. * style(cli): restore prettier wrapping on install error copy * test(linux): re-pin the child-process ratchets and the batch-shim allowlist after the merge --- .github/workflows/pr.yml | 35 +- config/docker/cli-launch-contract/Dockerfile | 34 + .../cli-launch-contract/run-cli-case.sh | 84 ++ config/docker/headless-pairing/Dockerfile | 1 - .../docker/headless-serve-shutdown/Dockerfile | 3 +- .../run-appimage-desktop-startup-case.sh | 265 ++++++ .../run-signal-case.sh | 92 +- config/electron-builder.config.cjs | 59 +- config/reliability-gates.jsonc | 132 ++- config/scripts/build-linux-local.mjs | 65 ++ config/scripts/build-linux-local.test.mjs | 85 ++ .../scripts/electron-builder-config.test.mjs | 337 +------ ...lectron-builder-runtime-resources.test.mjs | 308 +++++++ .../headless-serve-shutdown-workflow.test.mjs | 144 ++- .../linux-package-maintainer-scripts.test.mjs | 18 + .../orcad-operations-restart-safety.test.mjs | 42 + config/scripts/pr-code-change-scope.mjs | 7 + config/scripts/pr-code-change-scope.test.mjs | 22 + .../scripts/pr-workflow-parallelism.test.mjs | 13 +- .../run-headless-serve-shutdown-docker.mjs | 54 +- .../run-linux-cli-launch-contract-docker.mjs | 264 ++++++ .../static-appimage-package-contract.cjs | 260 ++++++ .../static-appimage-package-contract.test.mjs | 225 +++++ config/scripts/verify-cli-bin.mjs | 19 +- config/scripts/verify-linux-glibc-floor.cjs | 99 +++ .../scripts/verify-linux-glibc-floor.test.mjs | 87 ++ .../windows-cmd-shim-spawn-boundary.test.mjs | 4 +- config/tsconfig.cli.json | 2 + docs/reference/headless-linux-server.md | 56 +- docs/reference/linux-glibc-compatibility.md | 8 + docs/reference/orcad-operations.md | 66 +- docs/reference/ssh-execution-boundary.md | 2 +- package.json | 3 +- resources/linux/bin/orca-ide | 1 + resources/linux/packaging/after-remove.sh | 6 + src/cli/args.test.ts | 20 + src/cli/args.ts | 75 +- src/cli/cli-command-name-parity.test.ts | 25 + src/cli/cli-version.test.ts | 36 + src/cli/cli-version.ts | 14 + src/cli/command-suggestion.ts | 27 +- src/cli/handlers/core.ts | 51 +- src/cli/index.ts | 12 + src/cli/runtime/launch.test.ts | 144 ++- src/cli/runtime/launch.ts | 45 +- .../serve-signal-exit-diagnostic.test.ts | 30 + src/cli/runtime/serve-update-supervisor.ts | 17 +- src/cli/serve-electron-flag-parity.test.ts | 17 +- src/main/cli/cli-command-installation.ts | 49 +- src/main/cli/cli-installer.test.ts | 50 ++ src/main/cli/packaged-cli-assets.test.ts | 57 ++ src/main/linux-package-downloaded-status.ts | 88 ++ .../linux-package-install-command.test.ts | 51 ++ src/main/linux-package-install-command.ts | 10 + .../linux-package-install-diagnostic.test.ts | 271 +----- src/main/linux-package-install-diagnostic.ts | 146 +-- .../linux-package-update-recovery.test.ts | 121 +-- src/main/linux-package-update-recovery.ts | 79 +- src/main/linux-update-package-type.test.ts | 299 +++++-- src/main/linux-update-package-type.ts | 115 ++- .../startup/appimage-cli-redirect.test.ts | 211 ----- src/main/startup/appimage-cli-redirect.ts | 209 ----- src/main/startup/cli-command-names.ts | 73 ++ src/main/startup/cli-launch-redirect.test.ts | 357 ++++++++ src/main/startup/cli-launch-redirect.ts | 245 ++++++ .../startup/ensure-virtual-display.test.ts | 410 ++++++++- src/main/startup/ensure-virtual-display.ts | 217 ++++- src/main/startup/main-process-preflight.ts | 39 +- .../startup/main-process-runtime-launch.ts | 2 +- src/main/startup/main-process-serve.ts | 40 +- .../packaged-cli-entry-redirect.test.ts | 157 ---- .../startup/packaged-cli-entry-redirect.ts | 128 --- ...serve-mode-argv-cli-redirect-order.test.ts | 59 +- src/main/startup/serve-mode-argv.test.ts | 21 + src/main/startup/serve-mode-argv.ts | 15 +- src/main/startup/serve-options.test.ts | 161 ++++ src/main/startup/serve-options.ts | 134 +++ .../startup/serve-signal-handlers.test.ts | 4 +- src/main/startup/serve-signal-handlers.ts | 5 +- ...single-instance-lock-exit.electron.test.ts | 48 +- src/main/updater-events.test.ts | 274 +++++- src/main/updater-events.ts | 114 ++- src/main/updater-fallback.ts | 9 +- ...ter-linux-package-recovery-actions.test.ts | 193 +++- src/main/updater-mac-install.ts | 59 +- src/main/updater-test-harness.ts | 25 +- src/main/updater.fallback.test.ts | 17 + .../updater.headless-serve-install.test.ts | 6 + .../updater.install-failure-cause.test.ts | 7 + .../updater.linux-externally-managed.test.ts | 155 ++++ ...updater.linux-root-package-install.test.ts | 828 ++++-------------- src/main/updater.mac-install.test.ts | 10 +- src/main/updater.quit-and-install.test.ts | 1 + src/main/updater.startup-scheduling.test.ts | 1 + src/main/updater/updater-build-selection.ts | 4 +- src/main/updater/updater-check-failure.ts | 15 +- src/main/updater/updater-check-state.ts | 20 +- src/main/updater/updater-download-install.ts | 34 +- src/main/updater/updater-install-execution.ts | 107 +-- src/main/updater/updater-install-support.ts | 7 +- src/main/updater/updater-menu-checks.ts | 2 +- src/main/updater/updater-package-recovery.ts | 202 +---- src/main/updater/updater-remote-status.ts | 9 +- src/main/updater/updater-setup.ts | 6 +- src/main/updater/updater-state.ts | 6 - .../LinuxPackageInstallRecoveryCard.test.tsx | 361 +++----- .../LinuxPackageInstallRecoveryCard.tsx | 129 +-- .../components/UpdateCard.error-card.test.tsx | 104 ++- .../src/components/UpdateCard.test.ts | 31 + src/renderer/src/components/UpdateCard.tsx | 18 +- .../src/components/UpdateErrorCardContent.tsx | 2 +- .../UpdateAvailableCardContent.tsx | 58 +- .../update-card/UpdateCardStateContent.tsx | 11 +- .../update-card/update-card-error-model.ts | 28 +- .../update-card/update-card-visibility.ts | 12 +- .../GeneralUpdateSettingsSection.test.tsx | 37 + .../settings/GeneralUpdateSettingsSection.tsx | 60 +- .../settings/RemoteServerUpdateStatus.tsx | 2 +- .../status-bar/UpdateStatusSegment.tsx | 10 +- src/renderer/src/i18n/locales/en.json | 23 +- .../child-process-import-allowlist.txt | 2 - .../windows-console-visibility-allowlist.txt | 2 - .../child-process-import-boundary.test.ts | 2 +- .../windows-console-visibility.test.ts | 2 +- src/shared/cli-argument-boundary.test.ts | 33 + src/shared/cli-argument-boundary.ts | 96 ++ src/shared/edit-distance.ts | 23 + src/shared/serve-option-validation.test.ts | 72 ++ src/shared/serve-option-validation.ts | 89 ++ src/shared/update-status-types.ts | 15 +- 130 files changed, 7056 insertions(+), 3563 deletions(-) create mode 100644 config/docker/cli-launch-contract/Dockerfile create mode 100755 config/docker/cli-launch-contract/run-cli-case.sh create mode 100755 config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh create mode 100644 config/scripts/build-linux-local.mjs create mode 100644 config/scripts/build-linux-local.test.mjs create mode 100644 config/scripts/electron-builder-runtime-resources.test.mjs create mode 100644 config/scripts/linux-package-maintainer-scripts.test.mjs create mode 100644 config/scripts/orcad-operations-restart-safety.test.mjs create mode 100755 config/scripts/run-linux-cli-launch-contract-docker.mjs create mode 100644 config/scripts/static-appimage-package-contract.cjs create mode 100644 config/scripts/static-appimage-package-contract.test.mjs create mode 100644 src/cli/cli-command-name-parity.test.ts create mode 100644 src/cli/cli-version.test.ts create mode 100644 src/cli/cli-version.ts create mode 100644 src/main/linux-package-downloaded-status.ts delete mode 100644 src/main/startup/appimage-cli-redirect.test.ts delete mode 100644 src/main/startup/appimage-cli-redirect.ts create mode 100644 src/main/startup/cli-command-names.ts create mode 100644 src/main/startup/cli-launch-redirect.test.ts create mode 100644 src/main/startup/cli-launch-redirect.ts delete mode 100644 src/main/startup/packaged-cli-entry-redirect.test.ts delete mode 100644 src/main/startup/packaged-cli-entry-redirect.ts create mode 100644 src/main/startup/serve-options.test.ts create mode 100644 src/main/startup/serve-options.ts create mode 100644 src/main/updater.linux-externally-managed.test.ts create mode 100644 src/renderer/src/components/settings/GeneralUpdateSettingsSection.test.tsx create mode 100644 src/shared/cli-argument-boundary.test.ts create mode 100644 src/shared/cli-argument-boundary.ts create mode 100644 src/shared/edit-distance.ts create mode 100644 src/shared/serve-option-validation.test.ts create mode 100644 src/shared/serve-option-validation.ts diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index 80d3d42a8bb..c17ad60d5d3 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -623,6 +623,8 @@ jobs: needs: [code_paths] if: needs.code_paths.outputs.package == 'true' runs-on: ubuntu-latest + # Let the serial Docker gates reach their own deadlines and report cleanup failures. + timeout-minutes: 90 steps: - name: Checkout @@ -678,14 +680,45 @@ jobs: - name: Build native components run: pnpm run build:native + - name: Install Linux package tooling + run: sudo apt-get update && sudo apt-get install -y cpio rpm + - name: Package unpacked app env: ORCA_REUSE_PREPARED_NATIVE_RUNTIME: '1' - run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage --x64 --publish never + run: pnpm exec electron-builder --config config/electron-builder.config.cjs --linux AppImage deb rpm --x64 --publish never + + - name: Verify root-package marker payloads + run: | + set -euo pipefail + version="$(node -p "require('./package.json').version")" + deb="dist/orca-ide_${version}_amd64.deb" + rpm="dist/orca-ide-${version}.x86_64.rpm" + test -s "$deb" + test -s "$rpm" + deb_marker="$(dpkg-deb --fsys-tarfile "$deb" | tar -xOf - ./opt/Orca/resources/package-type)" + rpm_marker="$(rpm2cpio "$rpm" | cpio --quiet --extract --to-stdout ./opt/Orca/resources/package-type)" + [[ "$deb_marker" == deb ]] || { echo "Expected deb marker, got: $deb_marker"; exit 1; } + [[ "$rpm_marker" == rpm ]] || { echo "Expected rpm marker, got: $rpm_marker"; exit 1; } - name: Verify headless serve signal shutdown run: node config/scripts/run-headless-serve-shutdown-docker.mjs --appimage dist/orca-linux.AppImage + - name: Verify extracted launcher serve signal shutdown + run: >- + node config/scripts/run-headless-serve-shutdown-docker.mjs + --appimage dist/orca-linux.AppImage --entrypoint launcher + + - name: Verify AppImage CLI registration and serve signal shutdown + run: >- + node config/scripts/run-headless-serve-shutdown-docker.mjs + --appimage dist/orca-linux.AppImage --entrypoint appimage + --signal-target serving-electron --int-delivery pid + + # A default container reproduces the hostile AppImage launch environment. + - name: Verify Linux CLI launch contract + run: node config/scripts/run-linux-cli-launch-contract-docker.mjs --appimage dist/orca-linux.AppImage + - name: Smoke packaged CLI run: node config/scripts/smoke-packaged-cli.mjs --app-dir=dist/linux-unpacked diff --git a/config/docker/cli-launch-contract/Dockerfile b/config/docker/cli-launch-contract/Dockerfile new file mode 100644 index 00000000000..f6a618a8ece --- /dev/null +++ b/config/docker/cli-launch-contract/Dockerfile @@ -0,0 +1,34 @@ +ARG BASE_IMAGE=ubuntu:24.04 +FROM ${BASE_IMAGE} + +ARG LIBASOUND_PACKAGE=libasound2t64 + +ENV DEBIAN_FRONTEND=noninteractive + +# Install Electron's link-time libraries without adding a display server or FUSE. +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + bash \ + ca-certificates \ + coreutils \ + "${LIBASOUND_PACKAGE}" \ + libatk-bridge2.0-0 \ + libatspi2.0-0 \ + libdrm2 \ + libgbm1 \ + libgtk-3-0 \ + libnss3 \ + libxcomposite1 \ + libxdamage1 \ + libxfixes3 \ + libxkbcommon0 \ + libxrandr2 \ + procps \ + util-linux \ + && rm -rf /var/lib/apt/lists/* + +RUN useradd --create-home --shell /bin/bash orca + +COPY run-cli-case.sh /usr/local/bin/run-cli-case + +ENTRYPOINT ["/usr/local/bin/run-cli-case"] diff --git a/config/docker/cli-launch-contract/run-cli-case.sh b/config/docker/cli-launch-contract/run-cli-case.sh new file mode 100755 index 00000000000..3293601f22f --- /dev/null +++ b/config/docker/cli-launch-contract/run-cli-case.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# Print a parseable verdict; the host script owns expected statuses. +set -uo pipefail + +case_name=${1:?launch case is required} +extracted_root=${ORCA_TEST_EXTRACTED_ROOT:-/artifacts/squashfs-root} +launcher="$extracted_root/resources/bin/orca-ide" +command_timeout_seconds=${ORCA_TEST_COMMAND_TIMEOUT_SECONDS:-60} + +if ((EUID == 0)); then + # Reproduce extracted AppImage sandbox ownership as an unprivileged user. + exec runuser --user orca --preserve-environment -- "$0" "$@" +fi + +# Guard the restricted-userns precondition instead of accepting a false pass. +if [[ "$case_name" == *-userns-* ]]; then + if unshare -Ur true 2>/dev/null; then + echo "PRECONDITION_FAILED user namespaces are available; this case needs them restricted" + exit 90 + fi +fi +if [[ "$case_name" == nofuse-* && -e /dev/fuse ]]; then + echo "PRECONDITION_FAILED /dev/fuse is present; this case needs it absent" + exit 90 +fi + +unset DISPLAY WAYLAND_DISPLAY XDG_RUNTIME_DIR +if [[ "$case_name" == stale-display-* ]]; then + DISPLAY=:77 + export DISPLAY +fi + +case "$case_name" in + # The bundled launcher must stay in Electron's node mode. + nofuse-userns-bundled-help) command=("$launcher" --help) ;; + nofuse-userns-bundled-version) command=("$launcher" --version) ;; + nofuse-userns-bundled-status) command=("$launcher" status) ;; + nofuse-userns-bundled-skills) command=("$launcher" skills --help) ;; + nofuse-userns-bundled-worktree) command=("$launcher" worktree list) ;; + # Direct binaries must hand off before Ozone initializes. + nofuse-nosandbox-direct-binary-skills) + command=("$extracted_root/orca-ide" --no-sandbox skills --help) + ;; + nofuse-nosandbox-direct-binary-gui) + command=("$extracted_root/orca-ide" --no-sandbox) + ;; + stale-display-nosandbox-direct-binary-gui) + command=("$extracted_root/orca-ide" --no-sandbox) + ;; + *) + echo "UNKNOWN_CASE $case_name" + exit 91 + ;; +esac + +output=$(timeout --foreground --signal=TERM --kill-after=5s "${command_timeout_seconds}s" "${command[@]}" 2>&1) +status=$? + +if ((status == 124)); then + echo "TIMED_OUT seconds=$command_timeout_seconds case=$case_name" + printf '%s\n' "$output" | tail -30 + exit 94 +fi + +if [[ "$case_name" == nofuse-userns-bundled-version ]]; then + version_file="$extracted_root/resources/app.asar.unpacked/out/package.json" + expected_version=$(sed -n 's/.*"version"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$version_file") + if [[ -z "$expected_version" || "$output" != "$expected_version" ]]; then + output="VERSION_MISMATCH expected=${expected_version:-missing} got=$output" + status=93 + fi +fi + +# Shell signal exits are reported as 128 plus the signal number. +if ((status >= 128)); then + echo "CRASHED status=$status case=$case_name" + printf '%s\n' "$output" | tail -30 + exit 92 +fi + +echo "RESULT status=$status case=$case_name" +# Preserve the help header used by output assertions. +printf '%s\n' "$output" | head -200 +exit 0 diff --git a/config/docker/headless-pairing/Dockerfile b/config/docker/headless-pairing/Dockerfile index 8feafcc6e82..03664f68b0d 100644 --- a/config/docker/headless-pairing/Dockerfile +++ b/config/docker/headless-pairing/Dockerfile @@ -28,7 +28,6 @@ RUN apt-get update \ util-linux \ xauth \ xvfb \ - zlib1g-dev \ && rm -rf /var/lib/apt/lists/* RUN useradd --create-home --shell /bin/bash orca diff --git a/config/docker/headless-serve-shutdown/Dockerfile b/config/docker/headless-serve-shutdown/Dockerfile index 669bb02b00a..13b1ed2b69f 100644 --- a/config/docker/headless-serve-shutdown/Dockerfile +++ b/config/docker/headless-serve-shutdown/Dockerfile @@ -22,16 +22,15 @@ RUN apt-get update \ libxkbcommon0 \ libxrandr2 \ libxss1 \ - p7zip-full \ procps \ util-linux \ xauth \ xvfb \ - zlib1g-dev \ && rm -rf /var/lib/apt/lists/* RUN useradd --create-home --shell /bin/bash orca COPY run-signal-case.sh /usr/local/bin/run-signal-case +COPY run-appimage-desktop-startup-case.sh /usr/local/bin/run-appimage-desktop-startup-case ENTRYPOINT ["/usr/local/bin/run-signal-case"] diff --git a/config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh b/config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh new file mode 100755 index 00000000000..59a6bef0e5c --- /dev/null +++ b/config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh @@ -0,0 +1,265 @@ +#!/usr/bin/env bash +set -euo pipefail + +appimage=${1:-/input/orca.AppImage} +startup_timeout_seconds=90 +if [[ $# -gt 1 ]]; then + echo "usage: run-appimage-desktop-startup-case.sh [appimage]" >&2 + exit 64 +fi + +if ((EUID == 0)); then + if ! state_dir=$(mktemp -d /tmp/orca-appimage-startup.XXXXXX); then + echo 'FAIL: unable to create the AppImage startup state directory' >&2 + exit 1 + fi + if ! chown orca:orca "$state_dir"; then + echo "FAIL: unable to hand the AppImage startup state directory to orca: $state_dir" >&2 + rm -rf -- "$state_dir" || true + exit 1 + fi + exec runuser --user orca --preserve-environment -- env \ + ORCA_STARTUP_STATE_DIR="$state_dir" \ + ORCA_STARTUP_STATE_DIR_CLEANUP=1 \ + "$0" "$@" +fi + +remove_state_dir_on_exit=${ORCA_STARTUP_STATE_DIR_CLEANUP:-0} +if [[ -n "${ORCA_STARTUP_STATE_DIR:-}" ]]; then + state_dir=$ORCA_STARTUP_STATE_DIR +else + if ! state_dir=$(mktemp -d /tmp/orca-appimage-startup.XXXXXX); then + echo 'FAIL: unable to create the AppImage startup state directory' >&2 + exit 1 + fi + remove_state_dir_on_exit=1 +fi +stdout_log="$state_dir/stdout.log" +stderr_log="$state_dir/stderr.log" +launcher_pid= +launcher_start_ticks= +launcher_pgid= +launcher_status= +launcher_waited=false +tree_pids=() +declare -A tree_start_ticks=() + +read_start_ticks() { + local pid=$1 + [[ -r "/proc/$pid/stat" ]] || return 1 + awk '{print $22}' "/proc/$pid/stat" +} + +identity_alive() { + local pid=$1 + local expected_ticks=$2 + [[ -n "$expected_ticks" ]] || return 1 + [[ -r "/proc/$pid/stat" ]] || return 1 + [[ $(awk '{print $22}' "/proc/$pid/stat" 2>/dev/null || true) == "$expected_ticks" ]] || return 1 + local process_state + process_state=$(ps -o stat= -p "$pid" 2>/dev/null | tr -d '[:space:]' || true) + [[ -n "$process_state" && "$process_state" != Z* ]] +} + +collect_process_tree() { + tree_pids=() + tree_start_ticks=() + [[ -n "$launcher_pid" ]] || return + [[ -n "$launcher_start_ticks" ]] || return + tree_pids+=("$launcher_pid") + tree_start_ticks["$launcher_pid"]="$launcher_start_ticks" + local -a frontier=("$launcher_pid") + while ((${#frontier[@]})); do + local parent=${frontier[0]} + frontier=("${frontier[@]:1}") + while read -r child; do + [[ -n "$child" ]] || continue + [[ -z "${tree_start_ticks[$child]+present}" ]] || continue + local child_ticks + child_ticks=$(read_start_ticks "$child" 2>/dev/null || true) + [[ -n "$child_ticks" ]] || continue + tree_pids+=("$child") + tree_start_ticks["$child"]="$child_ticks" + frontier+=("$child") + done < <(ps -eo pid=,ppid= | awk -v parent="$parent" '$2 == parent {print $1}') + done +} + +process_is_xvfb() { + local pid=$1 + local command_name + command_name=$(ps -o comm= -p "$pid" 2>/dev/null || true) + [[ "$command_name" == Xvfb ]] && return 0 + local command_line + command_line=$(ps -o args= -p "$pid" 2>/dev/null || true) + [[ "$command_line" =~ (^|[[:space:]/])Xvfb([[:space:]]|$) ]] +} + +signal_process_group() { + local signal=$1 + identity_alive "$launcher_pid" "$launcher_start_ticks" || return 0 + [[ "$launcher_pgid" =~ ^[0-9]+$ ]] || return 0 + [[ "$launcher_pgid" != "$(ps -o pgid= -p "$$" | tr -d ' ')" ]] || return 0 + kill -s "$signal" -- "-$launcher_pgid" 2>/dev/null || true +} + +signal_owned_processes() { + local signal=$1 + local index pid ticks + for ((index = ${#tree_pids[@]} - 1; index >= 0; index--)); do + pid=${tree_pids[index]} + ticks=${tree_start_ticks[$pid]-} + if identity_alive "$pid" "$ticks"; then + kill -s "$signal" "$pid" 2>/dev/null || true + fi + done +} + +wait_for_owned_exit() { + local timeout_seconds=$1 + local deadline=$((SECONDS + timeout_seconds)) + local pid ticks alive + while ((SECONDS < deadline)); do + alive=0 + for pid in "${tree_pids[@]}"; do + ticks=${tree_start_ticks[$pid]-} + if identity_alive "$pid" "$ticks"; then + alive=1 + break + fi + done + if ((alive == 0)); then + return 0 + fi + sleep 0.2 + done + return 1 +} + +dump_logs() { + echo "--- desktop startup stdout ---" >&2 + cat "$stdout_log" >&2 2>/dev/null || true + echo "--- desktop startup stderr ---" >&2 + cat "$stderr_log" >&2 2>/dev/null || true +} + +cleanup_state_dir() { + [[ "$remove_state_dir_on_exit" == 1 ]] || return 0 + [[ "$state_dir" =~ ^/tmp/orca-appimage-startup\.[^/]+$ ]] || return 0 + [[ -d "$state_dir" && ! -L "$state_dir" && -O "$state_dir" ]] || return 0 + rm -rf -- "$state_dir" +} + +capture_launcher_status() { + [[ "$launcher_waited" == false ]] || return 0 + [[ -n "$launcher_pid" ]] || return 1 + if wait "$launcher_pid"; then + launcher_status=0 + else + launcher_status=$? + fi + launcher_waited=true +} + +report_launcher_exit() { + local reason=$1 + local observed_status=unknown + local exit_status=1 + if capture_launcher_status; then + observed_status=$launcher_status + if ((launcher_status != 0)); then + exit_status=$launcher_status + fi + fi + echo "FAIL: desktop launcher exited before ${reason} (status=${observed_status})" >&2 + exit "$exit_status" +} + +cleanup() { + local status=$? + trap - EXIT + signal_process_group TERM || true + signal_owned_processes TERM || true + if ! wait_for_owned_exit 10; then + signal_process_group KILL || true + signal_owned_processes KILL || true + wait_for_owned_exit 5 || status=1 + fi + capture_launcher_status || true + if ((status != 0)); then + dump_logs + else + if ! cleanup_state_dir; then + status=1 + dump_logs + fi + fi + exit "$status" +} +trap cleanup EXIT + +mkdir -p "$state_dir/home" "$state_dir/config" "$state_dir/cache" "$state_dir/runtime" +chmod 700 "$state_dir/runtime" +export HOME="$state_dir/home" +export XDG_CONFIG_HOME="$state_dir/config" +export XDG_CACHE_HOME="$state_dir/cache" +export XDG_RUNTIME_DIR="$state_dir/runtime" +export LIBGL_ALWAYS_SOFTWARE=1 +export ORCA_STARTUP_DIAGNOSTICS=1 +ulimit -c 0 + +[[ -r "$appimage" ]] || { echo "FAIL: AppImage is not readable: $appimage" >&2; exit 1; } +[[ -x "$appimage" ]] || { echo "FAIL: AppImage is not executable: $appimage" >&2; exit 1; } + +setsid --wait dbus-run-session -- xvfb-run -a "$appimage" --appimage-extract-and-run --no-sandbox \ + >"$stdout_log" 2>"$stderr_log" & +launcher_pid=$! +launcher_start_ticks=$(read_start_ticks "$launcher_pid" 2>/dev/null || true) +launcher_pgid=$(ps -o pgid= -p "$launcher_pid" 2>/dev/null | tr -d ' ' || true) +if [[ -z "$launcher_start_ticks" ]]; then + report_launcher_exit 'its identity could be recorded' +fi + +marker_seen=false +deadline=$((SECONDS + startup_timeout_seconds)) +while ((SECONDS < deadline)); do + if grep -Eq '^\[startup\] updater-setup-done t=[0-9]+$' "$stderr_log"; then + marker_seen=true + break + fi + if ! identity_alive "$launcher_pid" "$launcher_start_ticks"; then + report_launcher_exit 'the updater-setup-done marker' + fi + sleep 0.2 +done +if [[ "$marker_seen" != true ]]; then + if ! identity_alive "$launcher_pid" "$launcher_start_ticks"; then + report_launcher_exit 'the updater-setup-done marker' + fi + echo "FAIL: desktop AppImage did not emit updater-setup-done within ${startup_timeout_seconds}s" >&2 + exit 1 +fi +if ! identity_alive "$launcher_pid" "$launcher_start_ticks"; then + echo "FAIL: desktop launcher identity changed after startup marker" >&2 + exit 1 +fi + +collect_process_tree +xvfb_pids=() +for pid in "${tree_pids[@]}"; do + if process_is_xvfb "$pid"; then + xvfb_pids+=("$pid") + fi +done +if ((${#xvfb_pids[@]} == 0)); then + echo "FAIL: no launcher-owned Xvfb process was found after startup" >&2 + exit 1 +fi +for pid in "${xvfb_pids[@]}"; do + if ! identity_alive "$pid" "${tree_start_ticks[$pid]-}"; then + echo "FAIL: launcher-owned Xvfb identity changed before cleanup" >&2 + exit 1 + fi +done + +echo "Desktop AppImage startup validation passed (launcher=${launcher_pid}, xvfb=${xvfb_pids[*]})." diff --git a/config/docker/headless-serve-shutdown/run-signal-case.sh b/config/docker/headless-serve-shutdown/run-signal-case.sh index 3cbf594ae1e..2d561629170 100755 --- a/config/docker/headless-serve-shutdown/run-signal-case.sh +++ b/config/docker/headless-serve-shutdown/run-signal-case.sh @@ -6,7 +6,9 @@ app_root=${ORCA_TEST_APP_ROOT:-/artifacts/root} signal_target_kind=${ORCA_SIGNAL_TARGET:-app} entrypoint_kind=${ORCA_TEST_ENTRYPOINT:-app} int_delivery=${ORCA_INT_DELIVERY:-foreground-process-group} -startup_timeout_seconds=${ORCA_STARTUP_TIMEOUT_SECONDS:-90} +# Packaged Electron startup can approach 90s on a cold CI runner; leave room +# for the readiness line to reach the log before the observer deadline. +startup_timeout_seconds=${ORCA_STARTUP_TIMEOUT_SECONDS:-180} if ((EUID == 0)); then exec runuser --user orca --preserve-environment -- "$0" "$@" @@ -41,8 +43,8 @@ chmod 700 "$XDG_RUNTIME_DIR" case "$entrypoint_kind" in app) entrypoint=("$app_root/AppRun" --no-sandbox) ;; + appimage) entrypoint=(/input/orca.AppImage --appimage-extract-and-run --no-sandbox) ;; launcher) - export ELECTRON_DISABLE_SANDBOX=1 entrypoint=("$app_root/resources/bin/orca-ide") ;; *) echo "unsupported entrypoint: $entrypoint_kind" >&2; exit 64 ;; @@ -53,18 +55,50 @@ setsid env -u DISPLAY "${entrypoint[@]}" serve --port 0 --pairing-address 127.0. app_pid=$! app_start_ticks=$(awk '{print $22}' "/proc/$app_pid/stat") -# The inner shell expands its positional parameters. -# shellcheck disable=SC2016 -ready_line=$(timeout "$startup_timeout_seconds" bash -c ' - tail --pid="$1" -n +1 -F "$2" 2>/dev/null \ - | jq --unbuffered -nc '\''first(inputs | select(.type == "orca_server_ready" and .schemaVersion == 1))'\'' -' bash "$app_pid" "$stdout_log" || true) +# jq's `inputs` waits for EOF even when wrapped in `first`, so a tail -F +# observer can outlive the timeout and leak into the next signal case. Poll +# finite snapshots instead; each parser invocation has a definite EOF. +read_ready_line() { + sed -u -n 's/^[^{]*//p' "$stdout_log" \ + | jq --unbuffered -Rnc 'first(inputs | fromjson? | select(.type == "orca_server_ready" and .schemaVersion == 1))' +} + +ready_line='' +startup_deadline=$((SECONDS + startup_timeout_seconds)) +while (( SECONDS < startup_deadline )); do + ready_line=$(read_ready_line) + [[ -n "$ready_line" ]] && break + kill -0 "$app_pid" 2>/dev/null || break + sleep 1 +done +# A readiness event can land as the final poll races the write. +if [[ -z "$ready_line" ]]; then + ready_line=$(read_ready_line) +fi if [[ -z "$ready_line" ]]; then cat "$stdout_log" "$stderr_log" >&2 - echo "FAIL: AppRun exited or timed out before orca_server_ready" >&2 + echo "FAIL: entrypoint exited or timed out before orca_server_ready" >&2 exit 1 fi +registered_cli_verified=false +if [[ "$entrypoint_kind" == appimage ]]; then + registered_cli="$HOME/.local/bin/orca-ide" + expected_target="$XDG_CACHE_HOME/orca/appimage/launcher/orca-ide" + actual_target=$(readlink "$registered_cli" 2>/dev/null || true) + if [[ "$actual_target" != "$expected_target" ]]; then + echo "FAIL: registered CLI target is ${actual_target:-missing}; expected $expected_target" >&2 + exit 1 + fi + if ! registered_help=$("$registered_cli" --help 2>&1) \ + || [[ "$registered_help" != *'Usage: orca '* ]]; then + echo "FAIL: registered CLI did not execute the packaged help command" >&2 + printf '%s\n' "$registered_help" >&2 + exit 1 + fi + registered_cli_verified=true +fi + bound_endpoint=$(jq -r '.boundEndpoint' <<<"$ready_line") bound_port=${bound_endpoint##*:} listener_before=$(ss -H -ltnp "sport = :$bound_port" || true) @@ -72,6 +106,7 @@ if [[ -z "$listener_before" ]]; then echo "FAIL: ready listener has no socket owner at $bound_endpoint" >&2 exit 1 fi +listener_before_pids=$(grep -oE 'pid=[0-9]+' <<<"$listener_before" | cut -d= -f2 || true) tree_pids=() declare -A tree_start_ticks @@ -104,8 +139,15 @@ fi signal_target_pid=$app_pid if [[ "$signal_target_kind" == serving-electron ]]; then - signal_target_pid=$(awk '/\/orca-ide .* --serve / {print $1; exit}' <<<"$tree_snapshot") + # The ready socket identifies the serving Electron even when AppImage's + # extraction wrapper rewrites the command line before it reaches Chromium. + signal_target_pid=$(head -n1 <<<"$listener_before_pids") [[ -n "$signal_target_pid" ]] || { echo "FAIL: serving Electron process not found" >&2; exit 1; } + if [[ -z "${tree_start_ticks[$signal_target_pid]+present}" ]]; then + echo "FAIL: ready listener PID $signal_target_pid is outside the entrypoint process tree" >&2 + echo "listener: $listener_before" >&2 + exit 1 + fi elif [[ "$signal_target_kind" != app ]]; then echo "unsupported signal target: $signal_target_kind" >&2 exit 64 @@ -138,17 +180,25 @@ fi kill "$watchdog_pid" 2>/dev/null || true wait "$watchdog_pid" 2>/dev/null || true -listener_after=$(ss -H -ltnp "sport = :$bound_port" || true) -survivors=() -for pid in "${tree_pids[@]}"; do - if [[ -r "/proc/$pid/stat" ]] \ - && [[ $(awk '{print $22}' "/proc/$pid/stat" 2>/dev/null || true) == "${tree_start_ticks[$pid]}" ]] \ - && ps -o stat= -p "$pid" 2>/dev/null | grep -qv '^Z'; then - survivors+=("$pid") +# Crashpad can exit just after Electron; poll all owned shutdown state for up to 5s. +for shutdown_poll in {0..50}; do + listener_after=$(ss -H -ltnp "sport = :$bound_port" || true) + survivors=() + for pid in "${tree_pids[@]}"; do + if [[ -r "/proc/$pid/stat" ]] \ + && [[ $(awk '{print $22}' "/proc/$pid/stat" 2>/dev/null || true) == "${tree_start_ticks[$pid]}" ]] \ + && ps -o stat= -p "$pid" 2>/dev/null | grep -qv '^Z'; then + survivors+=("$pid") + fi + done + owned_residue=$(ps -eo pid=,ppid=,stat=,args= | awk -v state="$state_dir" \ + '($0 ~ state || $0 ~ /\/artifacts\/root\/orca-ide/ || $0 ~ /[X]vfb :99 /) && $0 !~ /awk -v state=/ {print}' || true) + if [[ -z "$listener_after" && -z "$owned_residue" ]] \ + && ((${#survivors[@]} == 0)); then + break fi + ((shutdown_poll < 50)) && sleep 0.1 done -owned_residue=$(ps -eo pid=,ppid=,stat=,args= | awk -v state="$state_dir" \ - '($0 ~ state || $0 ~ /\/artifacts\/root\/orca-ide/ || $0 ~ /[X]vfb :99 /) && $0 !~ /awk -v state=/ {print}' || true) canary_alive=false if kill -0 "$canary_pid" 2>/dev/null \ @@ -170,16 +220,18 @@ jq -nc \ --argjson signalTargetPid "$signal_target_pid" \ --arg endpoint "$bound_endpoint" \ --arg listenerBefore "$listener_before" \ + --arg listenerBeforePids "$listener_before_pids" \ --arg listenerAfter "$listener_after" \ --arg xvfbPids "$xvfb_pids" \ --arg treeBefore "$tree_snapshot" \ --argjson waitStatus "$wait_status" \ --argjson fatalEvidence "$fatal_evidence" \ --argjson canaryAlive "$canary_alive" \ + --argjson registeredCliVerified "$registered_cli_verified" \ --arg survivors "${survivors[*]:-}" \ --arg residue "$owned_residue" \ --arg corePattern "$(cat /proc/sys/kernel/core_pattern)" \ - '{signal:$signal,signalDelivery:$signalDelivery,entrypointKind:$entrypointKind,signalTargetKind:$signalTargetKind,appPid:$appPid,signalTargetPid:$signalTargetPid,boundEndpoint:$endpoint,listenerBefore:$listenerBefore,listenerAfter:$listenerAfter,xvfbPids:$xvfbPids,treeBefore:$treeBefore,waitStatus:$waitStatus,fatalEvidence:$fatalEvidence,canaryAlive:$canaryAlive,survivingTreePids:$survivors,ownedResidue:$residue,corePattern:$corePattern}' + '{signal:$signal,signalDelivery:$signalDelivery,entrypointKind:$entrypointKind,signalTargetKind:$signalTargetKind,appPid:$appPid,signalTargetPid:$signalTargetPid,boundEndpoint:$endpoint,listenerBefore:$listenerBefore,listenerBeforePids:$listenerBeforePids,listenerAfter:$listenerAfter,xvfbPids:$xvfbPids,treeBefore:$treeBefore,waitStatus:$waitStatus,fatalEvidence:$fatalEvidence,canaryAlive:$canaryAlive,registeredCliVerified:$registeredCliVerified,survivingTreePids:$survivors,ownedResidue:$residue,corePattern:$corePattern}' if ((wait_status != 0)) || [[ -n "$listener_after" ]] || [[ "$fatal_evidence" != false ]] \ || [[ "$canary_alive" != true ]] || ((${#survivors[@]})) || [[ -n "$owned_residue" ]]; then diff --git a/config/electron-builder.config.cjs b/config/electron-builder.config.cjs index 13633ed8e01..06d41bad344 100644 --- a/config/electron-builder.config.cjs +++ b/config/electron-builder.config.cjs @@ -1,4 +1,4 @@ -const { chmodSync, existsSync, readdirSync } = require('node:fs') +const { chmodSync, existsSync, readdirSync, readFileSync, writeFileSync } = require('node:fs') const { execFileSync } = require('node:child_process') const { join, resolve } = require('node:path') const electronBuilderNativeRebuild = require('./scripts/electron-builder-native-rebuild.cjs') @@ -18,6 +18,7 @@ const { verifyPackagedNodePtyJobOwnership } = require('./scripts/verify-packaged-node-pty-job-ownership.cjs') const { verifySkillsCliRuntime } = require('./scripts/verify-skills-cli-runtime.cjs') +const { verifyStaticAppImagePackage } = require('./scripts/static-appimage-package-contract.cjs') // Why: dev-channel builds must carry the *release* identity — same bundle id, // Developer ID signature, and notarization ticket — or Squirrel.Mac refuses to @@ -104,6 +105,29 @@ const winSpeechNativeResource = { from: 'node_modules/sherpa-onnx-win-x64', to: 'node_modules/sherpa-onnx-win-x64' } +// electron-builder replaces these defaults when `depends` is configured; retain +// Electron's loader requirements alongside Orca's headless-host dependencies. +const debElectronRuntimeDependencies = [ + 'libgtk-3-0', + 'libnotify4', + 'libnss3', + 'libxss1', + 'libxtst6', + 'xdg-utils', + 'libatspi2.0-0', + 'libuuid1', + 'libsecret-1-0' +] +const rpmElectronRuntimeDependencies = [ + 'gtk3', + 'libnotify', + 'nss', + 'libXScrnSaver', + '(libXtst or libXtst6)', + 'xdg-utils', + 'at-spi2-core', + '(libuuid or libuuid1)' +] // Why mirrored, not imported: this config is CJS loaded by electron-builder outside the TS build. // Keep in sync with isMarkdownDocumentName() in src/main/ipc/markdown-documents.ts and with @@ -115,6 +139,7 @@ module.exports = { appId, productName: 'Orca', protocols: [{ name: 'Orca', schemes: ['orca'] }], + toolsets: { appimage: '1.0.3' }, ...(devChannelBuildVersion ? { extraMetadata: { version: devChannelBuildVersion } } : localBuildVersion @@ -235,12 +260,21 @@ module.exports = { 'node_modules/zod/**', 'node_modules/yaml/**' ], + artifactBuildCompleted: ({ file, arch }) => { + if (file.endsWith('.AppImage')) { + verifyStaticAppImagePackage(file, arch) + } + }, afterPack: async (context) => { // Why: a Linux runner-image glibc bump silently shipped a node-pty pty.node // requiring GLIBC_2.34, crashing the app on startup on Ubuntu 20.04 (#9902). // Fail packaging if any bundled native binary exceeds the supported floor. if (context.electronPlatformName === 'linux') { - verifyLinuxGlibcFloor(context.appOutDir) + // Why the arch is passed: symbol-version checks pass happily on a wrong-architecture binary, + // so a cross-built slice could ship the host's pty.node and only fail at runtime. + verifyLinuxGlibcFloor(context.appOutDir, { + targetArch: { 1: 'x64', 3: 'arm64' }[context.arch] + }) } const resourcesDir = context.electronPlatformName === 'darwin' @@ -254,6 +288,10 @@ module.exports = { if (!existsSync(resourcesDir)) { throw new Error(`Missing packaged resources directory: ${resourcesDir}`) } + // FpmTarget replaces this with deb/rpm while building those artifacts from the shared app tree. + if (context.electronPlatformName === 'linux') { + writeFileSync(join(resourcesDir, 'package-type'), 'AppImage') + } if (context.electronPlatformName === 'darwin') { const architectureByEnum = { 1: 'x64', 3: 'arm64' } const architecture = architectureByEnum[context.arch] @@ -273,6 +311,7 @@ module.exports = { } writeMacBuildCompatibility(resourcesDir, { version, commit, architecture }) } + stampPackagedCliVersion(resourcesDir, context.packager.appInfo.version) prunePackagedRuntimeNodeModules(resourcesDir, context.electronPlatformName, context.arch) verifyPackagedMainRuntimeDeps(resourcesDir) // Why: boot the packaged daemon-entry under plain Node, but only for the @@ -522,7 +561,8 @@ module.exports = { }, featureWallResources ], - target: ['AppImage', 'deb'], + // Keep local artifacts aligned with the release pipeline. + target: ['AppImage', 'deb', 'rpm'], maintainer: 'stablyai', category: 'Utility' }, @@ -536,6 +576,7 @@ module.exports = { // Linux host — Chromium needs a display server even for offscreen rendering, // and serve starts Xvfb itself when present (see ensure-virtual-display.ts). depends: [ + ...debElectronRuntimeDependencies, 'python3', 'python3-gi', 'gir1.2-atspi-2.0', @@ -557,9 +598,9 @@ module.exports = { // Why: see deb depends. RPM distros ship Xvfb as xorg-x11-server-Xvfb (there // is no `xvfb` package), so the name differs from the deb here. depends: [ + ...rpmElectronRuntimeDependencies, 'python3', 'python3-gobject', - 'at-spi2-core', 'xdotool', 'xclip', 'xorg-x11-server-Xvfb' @@ -584,6 +625,16 @@ module.exports = { } } +// Stamp the effective channel version where node-mode CLI code can read it. +function stampPackagedCliVersion(resourcesDir, version) { + const packageJsonPath = join(resourcesDir, 'app.asar.unpacked', 'out', 'package.json') + if (!existsSync(packageJsonPath)) { + throw new Error(`Missing unpacked CLI package boundary: ${packageJsonPath}`) + } + const packageJson = JSON.parse(readFileSync(packageJsonPath, 'utf8')) + writeFileSync(packageJsonPath, `${JSON.stringify({ ...packageJson, version }, null, 2)}\n`) +} + function chmodUnixCliLaunchers(resourcesDir, electronPlatformName) { if (electronPlatformName === 'win32') { return diff --git a/config/reliability-gates.jsonc b/config/reliability-gates.jsonc index 0c2337ba36f..83de3128ecf 100644 --- a/config/reliability-gates.jsonc +++ b/config/reliability-gates.jsonc @@ -16701,16 +16701,17 @@ "providers": ["local-daemon"], "coveredPlatforms": ["linux"], "coveredProviders": ["local-daemon"], - "coverageNotes": "An Ubuntu 26.04 amd64 container extracts the packaged AppImage into disposable HOME and XDG directories, leaves APPDIR unset to preserve extracted-AppRun direct serve mode, waits for structured serve readiness, then exercises terminal-style foreground-process-group SIGINT and the documented systemd KillMode=mixed main-PID SIGTERM in separate containers. Local evidence runs under Rosetta on an arm64 Docker host; native amd64 PR CI repeats the same foreground AppRun identity contract.", + "coverageNotes": "An Ubuntu 26.04 amd64 container first launches the original AppImage through dbus-run-session and xvfb-run with startup diagnostics, then extracts the packaged AppImage into disposable HOME and XDG directories, leaves APPDIR unset to preserve extracted-AppRun direct serve mode, waits for structured serve readiness, and exercises terminal-style foreground-process-group SIGINT plus the documented systemd KillMode=mixed main-PID SIGTERM in separate containers. Local evidence runs under Rosetta on an arm64 Docker host; native amd64 PR CI repeats the same startup and foreground-AppRun identity contracts.", "motivatingLinks": [ "https://github.com/stablyai/orca/issues/14109", "https://linear.app/stably/issue/STA-4051" ], "invariant": "After packaged foreground headless serve publishes structured readiness, one SIGINT or SIGTERM exits successfully without an Electron fatal trap or core evidence, releases the exact listener and owned Xvfb/process tree, and leaves an unrelated process identity untouched.", - "oracle": "For each signal, start a fresh unprivileged Ubuntu 26.04 container with disposable profile and runtime directories, a random loopback port, DISPLAY unset, software GL, and the extracted AppImage in a fresh session. Wait for orca_server_ready schema version 1, record the listener owner, process tree, owned Xvfb, and unrelated canary identities, deliver SIGINT to the foreground process group or the documented KillMode=mixed graceful SIGTERM to the AppRun PID, then require wait status zero, no Failed to shutdown, SIGTRAP, core, listener, recorded descendant, profile/AppImage/Xvfb residue, or changed canary identity. The 30-second bounds are failure deadlines, never success conditions.", + "oracle": "First start the original, readable-and-executable AppImage once in a fresh restricted Ubuntu 26.04 container through dbus-run-session -- xvfb-run -a --appimage-extract-and-run with ORCA_STARTUP_DIAGNOSTICS=1, and require the exact updater-setup-done marker within 90 seconds while fencing the launcher and owned Xvfb by PID start ticks. For each signal, start a separate unprivileged container with disposable profile and runtime directories, a random loopback port, DISPLAY unset, software GL, and the extracted AppImage in a fresh session. Wait for orca_server_ready schema version 1, record the listener owner, process tree, owned Xvfb, and unrelated canary identities, deliver SIGINT to the foreground process group or the documented KillMode=mixed graceful SIGTERM to the AppRun PID, then require wait status zero, no Failed to shutdown, SIGTRAP, core, listener, recorded descendant, profile/AppImage/Xvfb residue, or changed canary identity. The 30-second bounds are failure deadlines, never success conditions.", "commands": [ "pnpm exec vitest run --config config/vitest.config.ts src/main/startup/ensure-virtual-display.test.ts config/scripts/headless-serve-shutdown-workflow.test.mjs --reporter=dot", "shellcheck config/docker/headless-serve-shutdown/run-signal-case.sh", + "shellcheck config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh", "node config/scripts/run-headless-serve-shutdown-docker.mjs --appimage dist/orca-linux.AppImage", "node config/scripts/run-headless-serve-shutdown-docker.mjs --appimage dist/orca-linux.AppImage --platform linux/amd64" ], @@ -16718,7 +16719,8 @@ "src/main/startup/ensure-virtual-display.test.ts", "config/scripts/headless-serve-shutdown-workflow.test.mjs", "config/scripts/run-headless-serve-shutdown-docker.mjs", - "config/docker/headless-serve-shutdown/run-signal-case.sh" + "config/docker/headless-serve-shutdown/run-signal-case.sh", + "config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh" ], "assertionRefs": [ { @@ -16735,15 +16737,19 @@ "file": "config/scripts/headless-serve-shutdown-workflow.test.mjs", "assertions": [ "PR CI builds an x64 AppImage before invoking the packaged shutdown oracle", + "the original AppImage desktop startup oracle is wired before extraction and signal cases", + "the bound AppImage is readable and executable before desktop launch and extraction", "the documented systemd unit uses KillMode=mixed so graceful TERM targets Orca before its owned Xvfb" ] }, { "file": "config/scripts/run-headless-serve-shutdown-docker.mjs", "assertions": [ + "the original AppImage startup runs through dbus-run-session and xvfb-run with a bounded diagnostics marker", "SIGINT and SIGTERM run in separate disposable containers", "both signal failures are reported before the oracle exits", "the exact AppImage SHA-256, entrypoint, and signal target are published", + "the read-only AppImage bind is checked for read and execute permissions before extraction", "the launcher exec overlay isolates the related STA-4017 signal boundary" ] }, @@ -16754,6 +16760,14 @@ "SIGTERM reaches the exact AppRun PID under the documented systemd KillMode=mixed policy", "target and descendant identities are fenced by PID start ticks before signaling and residue checks" ] + }, + { + "file": "config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh", + "assertions": [ + "the original AppImage emits the exact updater-setup-done startup marker within 90 seconds", + "launcher and owned Xvfb identities are fenced by PID start ticks", + "cleanup sends bounded TERM then KILL signals and preserves failure logs" + ] } ], "evidenceRuns": [ @@ -16774,11 +16788,20 @@ "result": "passed", "durationSeconds": 48, "summary": "The extracted candidate AppRun passed process-group SIGINT and systemd-mixed main-PID SIGTERM under Ubuntu 26.04 amd64 emulation with status zero, no fatal evidence, full listener/Xvfb/tree cleanup, and an unchanged canary identity." + }, + { + "date": "2026-08-31", + "runner": "ci", + "platform": "linux", + "command": "node config/scripts/run-headless-serve-shutdown-docker.mjs --appimage dist/orca-linux.AppImage --platform linux/amd64", + "result": "passed", + "durationSeconds": 1336, + "summary": "Native-amd64 PR package job https://github.com/stablyai/orca/actions/runs/33360129768/job/99389831915 built AppImage SHA-256 999d43bfe123e87a77fe917a5f46be1efd5c45a5205f99f02998a75136d8a793 and ran the restricted original-AppImage startup oracle before each of the three signal matrices. Each startup reached the exact updater-setup-done marker with stable launcher/Xvfb PID-start-tick identities and bounded TERM/KILL cleanup; SIGINT and SIGTERM then returned wait status 0 with no fatal evidence, listener, descendant, Xvfb, or canary residue. This is CI evidence only; no fresh local Docker oracle is claimed." } ], "runtimeBudget": { - "p95Seconds": 240, - "scope": "two fresh Ubuntu 26.04 containers, one per foreground signal" + "p95Seconds": 1800, + "scope": "three AppImage startup/extraction matrices, each with fresh Ubuntu 26.04 INT and TERM containers" }, "flakeHistory": { "status": "not-started", @@ -16805,6 +16828,105 @@ ], "demotionRule": "Keep experimental or demote if either signal traps, returns nonzero, retains its listener/Xvfb/run-owned process identity, touches the unrelated canary, or the focused gate flakes without an identified product or harness defect." }, + { + "id": "runtime.linux-cli-launch-contract", + "title": "Packaged Linux CLI commands run without FUSE, user namespaces, or a display", + "maturity": "experimental", + "protection": "partial", + "owner": "runtime-platform", + "layer": "appimage-cli-entrypoint", + "surfaces": [ + "packaged Linux AppImage", + "bundled CLI launcher", + "extracted direct binary", + "desktop launch diagnosis" + ], + "platforms": ["linux"], + "providers": ["local-daemon"], + "coveredPlatforms": ["linux"], + "coveredProviders": ["local-daemon"], + "coverageNotes": "A restricted Ubuntu container stages the extracted AppImage payload with no /dev/fuse and with unprivileged user namespaces denied, then runs eight CLI cases across the bundled launcher and the extracted direct binary. x64 only, because PR CI builds only --x64.", + "motivatingLinks": [ + "https://github.com/stablyai/orca/issues/13719", + "https://github.com/stablyai/orca/issues/14229" + ], + "invariant": "On a host without FUSE and without unprivileged user namespaces, every packaged CLI entrypoint either completes its command or reports a diagnosis, and never terminates on a signal.", + "oracle": "Build the image, stage the AppImage payload, and run each case in the restricted container. Assert the preconditions first: unshare -Ur must fail and /dev/fuse must be absent, so a relaxed runner fails the job rather than silently skipping. For each case require the exact expected exit status and an expected substring of the command's own output, with the harness RESULT/CRASHED/PRECONDITION_FAILED control lines excluded so a case name can never satisfy its own assertion. Any status of 128 or above is a crash and fails immediately. The per-case timeout is a failure deadline, never a success condition.", + "commands": [ + "node config/scripts/run-linux-cli-launch-contract-docker.mjs --appimage dist/orca-linux.AppImage", + "shellcheck config/docker/cli-launch-contract/run-cli-case.sh" + ], + "testFiles": [ + "config/scripts/run-linux-cli-launch-contract-docker.mjs", + "config/docker/cli-launch-contract/run-cli-case.sh" + ], + "assertionRefs": [ + { + "file": "config/scripts/run-linux-cli-launch-contract-docker.mjs", + "assertions": [ + "the bundled launcher serves --help, --version, status, skills --help, and worktree list without Chromium", + "a direct binary launch reaching JavaScript runs the command instead of booting a GUI", + "a desktop launch with no display reports the missing-display diagnosis instead of trapping", + "a stale DISPLAY is diagnosed rather than trusted", + "expected output is matched against the command's own output, not the harness control lines" + ] + }, + { + "file": "config/docker/cli-launch-contract/run-cli-case.sh", + "assertions": [ + "the container refuses to run unless unprivileged user namespaces are denied and /dev/fuse is absent", + "an exit status of 128 or above is reported as a crash rather than compared to the expected status" + ] + } + ], + "evidenceRuns": [ + { + "date": "2026-08-31", + "runner": "ci", + "platform": "linux", + "command": "node config/scripts/run-linux-cli-launch-contract-docker.mjs --appimage dist/orca-linux.AppImage", + "result": "passed", + "durationSeconds": 40, + "summary": "PR package job https://github.com/stablyai/orca/actions/runs/33360129768/job/99389831915 ran all eight cases to ok on ubuntu-latest. The unshare and /dev/fuse preconditions held under moby's default seccomp profile rather than tripping." + }, + { + "date": "2026-09-01", + "runner": "local", + "platform": "linux", + "command": "node config/scripts/run-linux-cli-launch-contract-docker.mjs --appimage dist/orca-linux.AppImage", + "result": "passed", + "durationSeconds": 60, + "summary": "All eight cases passed on Ubuntu 24.04 amd64 hardware against an AppImage built from the stack tip, invoked against a copy of that artifact outside dist/." + } + ], + "runtimeBudget": { + "p95Seconds": 300, + "scope": "eight CLI launch cases in one restricted Ubuntu container" + }, + "flakeHistory": { + "status": "not-started", + "evidence": "The harness is new; soak history is not yet available." + }, + "redGreenEvidence": { + "status": "complete", + "evidence": "The same harness run against a stock release AppImage failed four of eight cases: nofuse-userns-bundled-version at status 93, and all three direct-binary cases crashed at status 133 (SIGTRAP, the uv_close abort of #13719 and #14229). The stack-tip AppImage passed all eight. Corroborated at artifact level: the stack-tip runtime is a static-pie ELF with no PT_INTERP, the stock runtime is dynamically linked. Caveat: the two skills cases previously asserted a substring that the harness's own RESULT line contained, so they passed independently of command output; both now assert the rendered help header, and the green runs above predate that change." + }, + "performanceBudget": { + "required": false, + "evidence": "The gate is CI-only and adds no product code path." + }, + "promotionCriteria": [ + "Collect 30 consecutive CI passes or 14 days without an unexplained flake.", + "Extend the matrix to arm64 once PR CI builds that architecture.", + "Re-run red/green against a stock AppImage after any change to the launcher entrypoint." + ], + "knownGaps": [ + "The preconditions depend on moby's default seccomp profile denying unshare(CLONE_NEWUSER) and on /dev/fuse being absent. A runner with a relaxed profile or a mounted /dev/fuse trips PRECONDITION_FAILED and fails the job rather than skipping.", + "x64 only: PR CI builds only --x64, so the arm64 launcher path is unexercised.", + "The harness covers CLI entrypoints only; it does not exercise a full desktop session." + ], + "demotionRule": "Keep experimental or demote if any case terminates on a signal, the preconditions stop holding on the CI runner, or an assertion can be satisfied by anything other than the command's own output." + }, { "id": "ssh-managed-hooks.node18-runtime-compatibility", "title": "SSH managed-hook companions load and install hooks on Node 18", diff --git a/config/scripts/build-linux-local.mjs b/config/scripts/build-linux-local.mjs new file mode 100644 index 00000000000..2328f00bede --- /dev/null +++ b/config/scripts/build-linux-local.mjs @@ -0,0 +1,65 @@ +#!/usr/bin/env node + +import { execFileSync } from 'node:child_process' +import { resolve } from 'node:path' + +const SUPPORTED_ARCHES = new Set(['x64', 'arm64']) + +/** Select the local Linux package architecture without relying on builder defaults. */ +export function resolveLinuxBuildArch({ + platform = process.platform, + hostArch = process.arch, + requestedArch = process.env.ORCA_LINUX_BUILD_ARCH +} = {}) { + const arch = requestedArch ?? (platform === 'linux' ? hostArch : 'x64') + if (!SUPPORTED_ARCHES.has(arch)) { + throw new Error( + `Unsupported Linux build architecture: ${arch}. Use ORCA_LINUX_BUILD_ARCH=x64|arm64.` + ) + } + return arch +} + +export function buildLinuxElectronBuilderArgs(arch, extraArgs = []) { + if (!SUPPORTED_ARCHES.has(arch)) { + throw new Error(`Unsupported Linux build architecture: ${arch}`) + } + return [ + 'exec', + 'electron-builder', + '--config', + 'config/electron-builder.config.cjs', + '--linux', + 'AppImage', + 'deb', + 'rpm', + `--${arch}`, + ...extraArgs + ] +} + +export function runLocalLinuxBuild({ + arch = resolveLinuxBuildArch(), + extraArgs = [], + environment = process.env, + execFile = execFileSync, + platform = process.platform, + cwd = resolve(import.meta.dirname, '../..') +} = {}) { + const env = { ...environment } + if (arch === 'arm64') { + env.ORCA_LINUX_ARM64_RELEASE = '1' + } else { + delete env.ORCA_LINUX_ARM64_RELEASE + } + const pnpm = platform === 'win32' ? 'pnpm.cmd' : 'pnpm' + execFile(pnpm, buildLinuxElectronBuilderArgs(arch, extraArgs), { + cwd, + env, + stdio: 'inherit' + }) +} + +if (process.argv[1] && resolve(process.argv[1]) === resolve(import.meta.filename)) { + runLocalLinuxBuild({ extraArgs: process.argv.slice(2) }) +} diff --git a/config/scripts/build-linux-local.test.mjs b/config/scripts/build-linux-local.test.mjs new file mode 100644 index 00000000000..684c83f3265 --- /dev/null +++ b/config/scripts/build-linux-local.test.mjs @@ -0,0 +1,85 @@ +import { readFileSync } from 'node:fs' +import { resolve } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { + buildLinuxElectronBuilderArgs, + resolveLinuxBuildArch, + runLocalLinuxBuild +} from './build-linux-local.mjs' + +describe('local Linux build target', () => { + it('is the package script used by the local Linux build', () => { + const packageJson = JSON.parse( + readFileSync(resolve(import.meta.dirname, '../../package.json'), 'utf8') + ) + expect(packageJson.scripts['build:linux']).toContain( + 'node config/scripts/build-linux-local.mjs' + ) + }) + + it('follows a native Linux host architecture', () => { + expect(resolveLinuxBuildArch({ platform: 'linux', hostArch: 'arm64' })).toBe('arm64') + expect(resolveLinuxBuildArch({ platform: 'linux', hostArch: 'x64' })).toBe('x64') + }) + + it('defaults cross-platform Linux builds to x64 and allows an explicit override', () => { + expect(resolveLinuxBuildArch({ platform: 'darwin', hostArch: 'arm64' })).toBe('x64') + expect( + resolveLinuxBuildArch({ platform: 'darwin', hostArch: 'arm64', requestedArch: 'arm64' }) + ).toBe('arm64') + }) + + it('rejects unsupported architectures', () => { + expect(() => resolveLinuxBuildArch({ platform: 'linux', hostArch: 'ia32' })).toThrow( + 'Unsupported Linux build architecture' + ) + expect(() => buildLinuxElectronBuilderArgs('ia32')).toThrow( + 'Unsupported Linux build architecture' + ) + }) + + it('passes an explicit target and matching artifact-name environment', () => { + const execFile = vi.fn() + runLocalLinuxBuild({ + arch: 'arm64', + environment: { PATH: '/bin', ORCA_LINUX_ARM64_RELEASE: undefined }, + execFile, + platform: 'linux', + cwd: '/workspace' + }) + expect(execFile).toHaveBeenCalledWith( + 'pnpm', + buildLinuxElectronBuilderArgs('arm64'), + expect.objectContaining({ + cwd: '/workspace', + env: expect.objectContaining({ ORCA_LINUX_ARM64_RELEASE: '1' }), + stdio: 'inherit' + }) + ) + + expect(buildLinuxElectronBuilderArgs('x64')).toEqual( + expect.arrayContaining(['--linux', 'AppImage', 'deb', 'rpm', '--x64']) + ) + + runLocalLinuxBuild({ + arch: 'x64', + environment: { PATH: '/bin', ORCA_LINUX_ARM64_RELEASE: '1' }, + execFile, + platform: 'linux', + cwd: '/workspace' + }) + expect(execFile).toHaveBeenLastCalledWith( + 'pnpm', + buildLinuxElectronBuilderArgs('x64'), + expect.objectContaining({ + env: expect.not.objectContaining({ ORCA_LINUX_ARM64_RELEASE: expect.anything() }) + }) + ) + }) + + it('uses the Windows pnpm command name when cross-host packaging', () => { + const execFile = vi.fn() + runLocalLinuxBuild({ arch: 'x64', execFile, platform: 'win32', cwd: 'C:\\workspace' }) + expect(execFile.mock.calls[0]?.[0]).toBe('pnpm.cmd') + }) +}) diff --git a/config/scripts/electron-builder-config.test.mjs b/config/scripts/electron-builder-config.test.mjs index 347cae8fcfc..3f055ce222d 100644 --- a/config/scripts/electron-builder-config.test.mjs +++ b/config/scripts/electron-builder-config.test.mjs @@ -1,4 +1,4 @@ -import { cp, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from 'node:fs/promises' +import { chmod, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' import { createRequire } from 'node:module' import { tmpdir } from 'node:os' import { join } from 'node:path' @@ -10,17 +10,8 @@ const SRC_MAIN_DIR = join(REPO_ROOT, 'src', 'main') const require = createRequire(import.meta.url) const electronBuilderConfig = require('../electron-builder.config.cjs') const { FileMatcher } = require('app-builder-lib/out/fileMatcher') +const FpmTarget = require('app-builder-lib/out/targets/FpmTarget').default const electronBuilderNativeRebuild = require('./electron-builder-native-rebuild.cjs') -const { - createPackagedRuntimeNodeModuleResources, - findAsarEntry, - prunePackagedNodePty, - prunePackagedParcelWatcher, - prunePackagedSherpaOnnx, - prunePackagedRuntimeTypeAndSourceMapArtifacts, - prunePackagedZodSources, - verifyPackagedMainRuntimeDeps -} = require('../packaged-runtime-node-modules.cjs') describe('electron-builder config', () => { it('keeps the packaged app identity aligned with local-build validation', () => { @@ -280,8 +271,9 @@ describe('electron-builder config', () => { expect(electronBuilderConfig.linux.desktop.entry.StartupWMClass).toBe('orca') }) - it('uses AppImage and deb as local Linux targets without changing existing artifact names', () => { - expect(electronBuilderConfig.linux.target).toEqual(['AppImage', 'deb']) + it('uses the release artifact set as local Linux targets without changing existing names', () => { + expect(electronBuilderConfig.linux.target).toEqual(['AppImage', 'deb', 'rpm']) + expect(electronBuilderConfig.toolsets).toEqual({ appimage: '1.0.3' }) expect(electronBuilderConfig.appImage.artifactName).toBe('orca-linux.${ext}') expect(electronBuilderConfig.deb.artifactName).toBe('orca-ide_${version}_${arch}.${ext}') expect(electronBuilderConfig.rpm).toMatchObject({ @@ -290,6 +282,33 @@ describe('electron-builder config', () => { }) }) + it('retains electron-builder runtime dependencies in deb and rpm packages', () => { + for (const target of ['deb', 'rpm']) { + const dependencies = electronBuilderConfig[target].depends + expect(dependencies).toEqual( + expect.arrayContaining(FpmTarget.prototype.getDefaultDepends(target)) + ) + expect(new Set(dependencies).size).toBe(dependencies.length) + } + }) + + it('validates each AppImage before electron-builder publishes it', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-electron-builder-appimage-')) + try { + const appImage = join(root, 'orca-linux.AppImage') + await writeFile(appImage, 'not an ELF') + await chmod(appImage, 0o755) + + expect(() => + electronBuilderConfig.artifactBuildCompleted({ file: appImage, arch: 1 }) + ).toThrow(/ELF header is outside/) + expect(() => + electronBuilderConfig.artifactBuildCompleted({ file: join(root, 'orca-ide.deb') }) + ).not.toThrow() + } finally { + await rm(root, { recursive: true, force: true }) + } + }) it('uses a distinct AppImage name for Linux arm64 release uploads', () => { const configPath = require.resolve('../electron-builder.config.cjs') const original = process.env.ORCA_LINUX_ARM64_RELEASE @@ -367,286 +386,6 @@ describe('electron-builder config', () => { expect(electronBuilderConfig.npmRebuild).toBe(true) }) - it('verifies packaged main runtime deps from Windows-style asar entries', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-deps-')) - try { - await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') - await mkdir(join(resourcesDir, 'node_modules', 'yaml'), { recursive: true }) - await mkdir(join(resourcesDir, 'node_modules', 'zod'), { recursive: true }) - - const sources = new Map([ - ['out\\main\\index.js', 'const z = require("zod")'], - ['out\\main\\agent-hooks\\managed-agent-hook-controls.js', 'const YAML = require("yaml")'] - ]) - const asar = { - listPackage: () => [...sources.keys()].map((entry) => `\\${entry}`), - extractFile: (_asarPath, internalPath) => Buffer.from(sources.get(internalPath), 'utf8') - } - - expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).not.toThrow() - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('normalizes host-specific asar entry separators', () => { - expect(findAsarEntry(['\\out\\main\\index.js'], 'out/main/index.js')).toBe( - '\\out\\main\\index.js' - ) - expect(findAsarEntry(['/out/main/index.js'], 'out/main/index.js')).toBe('/out/main/index.js') - }) - - it('prunes non-target node-pty architecture outputs from packaged runtime resources', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-node-pty-prune-')) - try { - const nodePtyDir = join(resourcesDir, 'node_modules', 'node-pty') - const prebuildsDir = join(nodePtyDir, 'prebuilds') - const binDir = join(nodePtyDir, 'bin') - await mkdir(join(prebuildsDir, 'darwin-arm64'), { recursive: true }) - await mkdir(join(prebuildsDir, 'darwin-x64'), { recursive: true }) - await mkdir(join(prebuildsDir, 'linux-x64'), { recursive: true }) - await mkdir(join(prebuildsDir, 'win32-x64'), { recursive: true }) - await mkdir(join(binDir, 'darwin-arm64-148'), { recursive: true }) - await mkdir(join(binDir, 'darwin-x64-148'), { recursive: true }) - await mkdir(join(nodePtyDir, 'third_party', 'conpty'), { - recursive: true - }) - await mkdir(join(nodePtyDir, 'deps', 'winpty'), { recursive: true }) - - prunePackagedNodePty(resourcesDir, 'darwin', 3) - - await expect(readdir(prebuildsDir)).resolves.toEqual(['darwin-arm64']) - await expect(readdir(binDir)).resolves.toEqual(['darwin-arm64-148']) - await expect(readdir(join(nodePtyDir, 'third_party'))).resolves.toEqual([]) - await expect(readdir(join(nodePtyDir, 'deps'))).resolves.toEqual([]) - expect(() => prunePackagedNodePty(resourcesDir, 'darwin', 4)).toThrow( - 'Unsupported packaged runtime architecture: 4' - ) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('copies the Windows node-pty ConPTY runtime beside the rebuilt addon', async () => { - for (const [arch, electronArch] of [ - ['x64', 1], - ['arm64', 3] - ]) { - const resourcesDir = await mkdtemp(join(tmpdir(), `orca-node-pty-conpty-${arch}-`)) - try { - const nodePtyDir = join(resourcesDir, 'node_modules', 'node-pty') - const releaseDir = join(nodePtyDir, 'build', 'Release') - const conptyRoot = join(nodePtyDir, 'third_party', 'conpty', '0.1.0') - await mkdir(releaseDir, { recursive: true }) - await writeFile(join(releaseDir, 'conpty.node'), 'native addon placeholder', 'utf8') - for (const sourceArch of ['x64', 'arm64']) { - const sourceDir = join(conptyRoot, `win10-${sourceArch}`) - await mkdir(sourceDir, { recursive: true }) - await writeFile(join(sourceDir, 'conpty.dll'), `dll payload ${sourceArch}`, 'utf8') - await writeFile( - join(sourceDir, 'OpenConsole.exe'), - `console payload ${sourceArch}`, - 'utf8' - ) - } - - prunePackagedNodePty(resourcesDir, 'win32', electronArch) - - await expect(readFile(join(releaseDir, 'conpty', 'conpty.dll'), 'utf8')).resolves.toBe( - `dll payload ${arch}` - ) - await expect(readFile(join(releaseDir, 'conpty', 'OpenConsole.exe'), 'utf8')).resolves.toBe( - `console payload ${arch}` - ) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - } - }) - - it('includes external main dependencies in the packaged runtime closure', () => { - // Why: the main process imports '@parcel/watcher' for filesystem change - // events; if it is absent from the packaged closure the serve host silently - // stops propagating file changes to clients (regression guard for #4851). - const packaged = createPackagedRuntimeNodeModuleResources() - const packagedTargets = packaged.map((resource) => resource.to) - expect(packagedTargets).toContain(join('node_modules', '@parcel', 'watcher')) - expect( - packagedTargets.some((target) => - target.startsWith(join('node_modules', '@parcel', 'watcher-')) - ) - ).toBe(true) - expect(packagedTargets).toContain(join('node_modules', 'proper-lockfile')) - }) - - it('prunes non-target @parcel/watcher architecture subpackages', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-parcel-watcher-prune-')) - try { - const parcelDir = join(resourcesDir, 'node_modules', '@parcel') - await mkdir(join(parcelDir, 'watcher'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-darwin-arm64'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-darwin-x64'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-linux-x64-glibc'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-linux-arm64-glibc'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-win32-x64'), { recursive: true }) - - prunePackagedParcelWatcher(resourcesDir, 'linux', 'arm64') - - await expect(readdir(parcelDir).then((entries) => entries.sort())).resolves.toEqual([ - 'watcher', - 'watcher-linux-arm64-glibc' - ]) - expect(() => prunePackagedParcelWatcher(resourcesDir, 'linux', 'universal')).toThrow( - 'Unsupported packaged runtime architecture: universal' - ) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('leaves unrelated @parcel/* runtime deps untouched when pruning the watcher', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-parcel-watcher-prune-unrelated-')) - try { - const parcelDir = join(resourcesDir, 'node_modules', '@parcel') - await mkdir(join(parcelDir, 'watcher'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-darwin-arm64'), { recursive: true }) - await mkdir(join(parcelDir, 'watcher-linux-x64-glibc'), { recursive: true }) - // A hypothetical future @parcel/* runtime dep that is NOT a watcher subpackage. - await mkdir(join(parcelDir, 'transformer-js'), { recursive: true }) - - prunePackagedParcelWatcher(resourcesDir, 'linux', 1) - - await expect(readdir(parcelDir).then((entries) => entries.sort())).resolves.toEqual([ - 'transformer-js', - 'watcher', - 'watcher-linux-x64-glibc' - ]) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('prunes type declaration artifacts from packaged runtime node_modules', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-type-prune-')) - try { - const packageDir = join(resourcesDir, 'node_modules', 'example-package') - await mkdir(join(packageDir, 'dist'), { recursive: true }) - await writeFile(join(packageDir, 'dist', 'index.cjs'), 'module.exports = {}', 'utf8') - await writeFile(join(packageDir, 'dist', 'index.d.ts'), 'export type Value = string', 'utf8') - await writeFile(join(packageDir, 'dist', 'index.d.cts'), 'export type Value = string', 'utf8') - await writeFile(join(packageDir, 'dist', 'index.d.mts'), 'export type Value = string', 'utf8') - await writeFile(join(packageDir, 'dist', 'index.d.cts.map'), '{}', 'utf8') - await writeFile(join(packageDir, 'dist', 'index.d.mts.map'), '{}', 'utf8') - - prunePackagedRuntimeTypeAndSourceMapArtifacts(resourcesDir) - - await expect(readdir(join(packageDir, 'dist'))).resolves.toEqual(['index.cjs']) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('prunes duplicate darwin sherpa-onnx runtime dylib aliases', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-sherpa-prune-')) - try { - const packageDir = join(resourcesDir, 'node_modules', 'sherpa-onnx-darwin-arm64') - await mkdir(packageDir, { recursive: true }) - await writeFile(join(packageDir, 'sherpa-onnx.node'), '', 'utf8') - await writeFile(join(packageDir, 'libonnxruntime.1.23.2.dylib'), '', 'utf8') - await writeFile(join(packageDir, 'libonnxruntime.dylib'), '', 'utf8') - - prunePackagedSherpaOnnx(resourcesDir, 'darwin') - - await expect(readdir(packageDir).then((entries) => entries.sort())).resolves.toEqual([ - 'libonnxruntime.1.23.2.dylib', - 'sherpa-onnx.node' - ]) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('prunes zod TypeScript sources from packaged runtime resources', async () => { - const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-zod-prune-')) - try { - const packageDir = join(resourcesDir, 'node_modules', 'zod') - await mkdir(join(packageDir, 'src'), { recursive: true }) - await writeFile(join(packageDir, 'index.cjs'), 'module.exports = {}', 'utf8') - await writeFile(join(packageDir, 'src', 'index.ts'), 'export const value = true', 'utf8') - - prunePackagedZodSources(resourcesDir) - - await expect(readdir(packageDir)).resolves.toEqual(['index.cjs']) - } finally { - await rm(resourcesDir, { recursive: true, force: true }) - } - }) - - it('fails when the packaged resources directory is missing', async () => { - const root = await mkdtemp(join(tmpdir(), 'orca-electron-builder-config-')) - try { - await expect( - electronBuilderConfig.afterPack({ - appOutDir: root, - electronPlatformName: 'win32' - }) - ).rejects.toThrow(/Missing packaged resources directory/) - } finally { - await rm(root, { recursive: true, force: true }) - } - }) - - it.skipIf(process.platform === 'win32')( - 'marks packaged Unix CLI launchers executable', - async () => { - const root = await mkdtemp(join(tmpdir(), 'orca-electron-builder-config-')) - try { - const resourcesDir = join(root, 'linux-unpacked', 'resources') - const launcherPath = join(resourcesDir, 'bin', 'orca-ide') - await mkdir(join(resourcesDir, 'bin'), { recursive: true }) - await cp( - join(process.cwd(), 'resources', 'plugins', 'launch'), - join(resourcesDir, 'plugins', 'launch'), - { recursive: true } - ) - await mkdir(join(resourcesDir, 'node_modules', 'zod', 'src'), { recursive: true }) - // Why: afterPack now fails hard when the unpacked daemon entry is - // missing, so the fixture must carry one like a real package layout. - const unpackedMainDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'main') - await mkdir(unpackedMainDir, { recursive: true }) - await writeFile( - join(unpackedMainDir, 'daemon-entry.js'), - 'console.error("Usage: daemon-entry "); process.exit(1)\n', - 'utf8' - ) - const unpackedCliDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'cli') - await mkdir(join(unpackedCliDir, 'handlers'), { recursive: true }) - await writeFile(join(unpackedCliDir, 'handlers', 'skills.js'), '', 'utf8') - await writeFile( - join(unpackedCliDir, 'index.js'), - [ - 'const args = process.argv.slice(2)', - "if (args[1] === 'list') console.log(JSON.stringify({ topics: [{ name: 'orca-cli' }, { name: 'computer-use' }] }))", - "else if (args[1] === 'get') console.log(`---\\nname: ${args[2]}\\n---`)", - 'else console.log(JSON.stringify({ executed: false }))' - ].join('\n'), - 'utf8' - ) - await writeFile(launcherPath, '#!/usr/bin/env bash\n', { encoding: 'utf8', mode: 0o644 }) - - await electronBuilderConfig.afterPack({ - appOutDir: join(root, 'linux-unpacked'), - electronPlatformName: 'linux', - arch: 1 - }) - - expect((await stat(launcherPath)).mode & 0o111).not.toBe(0) - } finally { - await rm(root, { recursive: true, force: true }) - } - } - ) - // Why: the .deb/.rpm update-recovery path keys entirely off the resources/package-type marker that // app-builder-lib's FpmTarget writes. If packaging silently stops shipping an fpm target, or adds // one the recovery path does not cover, getLinuxRootPackageType() returns null, autoInstallOnAppQuit @@ -680,5 +419,17 @@ describe('electron-builder config', () => { expect(source).toContain(`value === '${target}'`) } }) + + it('keeps the pinned FpmTarget overwrite for configured deb and rpm artifacts', async () => { + const source = await readFile( + require.resolve('app-builder-lib/out/targets/FpmTarget'), + 'utf8' + ) + + expect(source).toContain('path.join(resourceDir, "package-type"), target') + for (const target of RECOVERABLE_TARGETS) { + expect(electronBuilderConfig[target]).toBeDefined() + } + }) }) }) diff --git a/config/scripts/electron-builder-runtime-resources.test.mjs b/config/scripts/electron-builder-runtime-resources.test.mjs new file mode 100644 index 00000000000..d2407776fa7 --- /dev/null +++ b/config/scripts/electron-builder-runtime-resources.test.mjs @@ -0,0 +1,308 @@ +import { cp, mkdir, mkdtemp, readFile, readdir, rm, stat, writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const electronBuilderConfig = require('../electron-builder.config.cjs') +const { + createPackagedRuntimeNodeModuleResources, + findAsarEntry, + prunePackagedNodePty, + prunePackagedParcelWatcher, + prunePackagedSherpaOnnx, + prunePackagedRuntimeTypeAndSourceMapArtifacts, + prunePackagedZodSources, + verifyPackagedMainRuntimeDeps +} = require('../packaged-runtime-node-modules.cjs') + +describe('packaged runtime resources', () => { + it('verifies packaged main runtime deps from Windows-style asar entries', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-deps-')) + try { + await writeFile(join(resourcesDir, 'app.asar'), '', 'utf8') + await mkdir(join(resourcesDir, 'node_modules', 'yaml'), { recursive: true }) + await mkdir(join(resourcesDir, 'node_modules', 'zod'), { recursive: true }) + + const sources = new Map([ + ['out\\main\\index.js', 'const z = require("zod")'], + ['out\\main\\agent-hooks\\managed-agent-hook-controls.js', 'const YAML = require("yaml")'] + ]) + const asar = { + listPackage: () => [...sources.keys()].map((entry) => `\\${entry}`), + extractFile: (_asarPath, internalPath) => Buffer.from(sources.get(internalPath), 'utf8') + } + + expect(() => verifyPackagedMainRuntimeDeps(resourcesDir, asar)).not.toThrow() + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('normalizes host-specific asar entry separators', () => { + expect(findAsarEntry(['\\out\\main\\index.js'], 'out/main/index.js')).toBe( + '\\out\\main\\index.js' + ) + expect(findAsarEntry(['/out/main/index.js'], 'out/main/index.js')).toBe('/out/main/index.js') + }) + + it('prunes non-target node-pty architecture outputs from packaged runtime resources', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-node-pty-prune-')) + try { + const nodePtyDir = join(resourcesDir, 'node_modules', 'node-pty') + const prebuildsDir = join(nodePtyDir, 'prebuilds') + const binDir = join(nodePtyDir, 'bin') + await mkdir(join(prebuildsDir, 'darwin-arm64'), { recursive: true }) + await mkdir(join(prebuildsDir, 'darwin-x64'), { recursive: true }) + await mkdir(join(prebuildsDir, 'linux-x64'), { recursive: true }) + await mkdir(join(prebuildsDir, 'win32-x64'), { recursive: true }) + await mkdir(join(binDir, 'darwin-arm64-148'), { recursive: true }) + await mkdir(join(binDir, 'darwin-x64-148'), { recursive: true }) + await mkdir(join(nodePtyDir, 'third_party', 'conpty'), { + recursive: true + }) + await mkdir(join(nodePtyDir, 'deps', 'winpty'), { recursive: true }) + + prunePackagedNodePty(resourcesDir, 'darwin', 3) + + await expect(readdir(prebuildsDir)).resolves.toEqual(['darwin-arm64']) + await expect(readdir(binDir)).resolves.toEqual(['darwin-arm64-148']) + await expect(readdir(join(nodePtyDir, 'third_party'))).resolves.toEqual([]) + await expect(readdir(join(nodePtyDir, 'deps'))).resolves.toEqual([]) + expect(() => prunePackagedNodePty(resourcesDir, 'darwin', 4)).toThrow( + 'Unsupported packaged runtime architecture: 4' + ) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('copies the Windows node-pty ConPTY runtime beside the rebuilt addon', async () => { + for (const [arch, electronArch] of [ + ['x64', 1], + ['arm64', 3] + ]) { + const resourcesDir = await mkdtemp(join(tmpdir(), `orca-node-pty-conpty-${arch}-`)) + try { + const nodePtyDir = join(resourcesDir, 'node_modules', 'node-pty') + const releaseDir = join(nodePtyDir, 'build', 'Release') + const conptyRoot = join(nodePtyDir, 'third_party', 'conpty', '0.1.0') + await mkdir(releaseDir, { recursive: true }) + await writeFile(join(releaseDir, 'conpty.node'), 'native addon placeholder', 'utf8') + for (const sourceArch of ['x64', 'arm64']) { + const sourceDir = join(conptyRoot, `win10-${sourceArch}`) + await mkdir(sourceDir, { recursive: true }) + await writeFile(join(sourceDir, 'conpty.dll'), `dll payload ${sourceArch}`, 'utf8') + await writeFile( + join(sourceDir, 'OpenConsole.exe'), + `console payload ${sourceArch}`, + 'utf8' + ) + } + + prunePackagedNodePty(resourcesDir, 'win32', electronArch) + + await expect(readFile(join(releaseDir, 'conpty', 'conpty.dll'), 'utf8')).resolves.toBe( + `dll payload ${arch}` + ) + await expect(readFile(join(releaseDir, 'conpty', 'OpenConsole.exe'), 'utf8')).resolves.toBe( + `console payload ${arch}` + ) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + } + }) + + it('includes external main dependencies in the packaged runtime closure', () => { + // Why: the main process imports '@parcel/watcher' for filesystem change + // events; if it is absent from the packaged closure the serve host silently + // stops propagating file changes to clients (regression guard for #4851). + const packaged = createPackagedRuntimeNodeModuleResources() + const packagedTargets = packaged.map((resource) => resource.to) + expect(packagedTargets).toContain(join('node_modules', '@parcel', 'watcher')) + expect( + packagedTargets.some((target) => + target.startsWith(join('node_modules', '@parcel', 'watcher-')) + ) + ).toBe(true) + expect(packagedTargets).toContain(join('node_modules', 'proper-lockfile')) + }) + + it('prunes non-target @parcel/watcher architecture subpackages', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-parcel-watcher-prune-')) + try { + const parcelDir = join(resourcesDir, 'node_modules', '@parcel') + await mkdir(join(parcelDir, 'watcher'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-darwin-arm64'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-darwin-x64'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-linux-x64-glibc'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-linux-arm64-glibc'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-win32-x64'), { recursive: true }) + + prunePackagedParcelWatcher(resourcesDir, 'linux', 'arm64') + + await expect(readdir(parcelDir).then((entries) => entries.sort())).resolves.toEqual([ + 'watcher', + 'watcher-linux-arm64-glibc' + ]) + expect(() => prunePackagedParcelWatcher(resourcesDir, 'linux', 'universal')).toThrow( + 'Unsupported packaged runtime architecture: universal' + ) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('leaves unrelated @parcel/* runtime deps untouched when pruning the watcher', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-parcel-watcher-prune-unrelated-')) + try { + const parcelDir = join(resourcesDir, 'node_modules', '@parcel') + await mkdir(join(parcelDir, 'watcher'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-darwin-arm64'), { recursive: true }) + await mkdir(join(parcelDir, 'watcher-linux-x64-glibc'), { recursive: true }) + // A hypothetical future @parcel/* runtime dep that is NOT a watcher subpackage. + await mkdir(join(parcelDir, 'transformer-js'), { recursive: true }) + + prunePackagedParcelWatcher(resourcesDir, 'linux', 1) + + await expect(readdir(parcelDir).then((entries) => entries.sort())).resolves.toEqual([ + 'transformer-js', + 'watcher', + 'watcher-linux-x64-glibc' + ]) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('prunes type declaration artifacts from packaged runtime node_modules', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-runtime-type-prune-')) + try { + const packageDir = join(resourcesDir, 'node_modules', 'example-package') + await mkdir(join(packageDir, 'dist'), { recursive: true }) + await writeFile(join(packageDir, 'dist', 'index.cjs'), 'module.exports = {}', 'utf8') + await writeFile(join(packageDir, 'dist', 'index.d.ts'), 'export type Value = string', 'utf8') + await writeFile(join(packageDir, 'dist', 'index.d.cts'), 'export type Value = string', 'utf8') + await writeFile(join(packageDir, 'dist', 'index.d.mts.map'), '{}', 'utf8') + + prunePackagedRuntimeTypeAndSourceMapArtifacts(resourcesDir) + + await expect(readdir(join(packageDir, 'dist'))).resolves.toEqual(['index.cjs']) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('prunes duplicate darwin sherpa-onnx runtime dylib aliases', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-sherpa-prune-')) + try { + const packageDir = join(resourcesDir, 'node_modules', 'sherpa-onnx-darwin-arm64') + await mkdir(packageDir, { recursive: true }) + await writeFile(join(packageDir, 'sherpa-onnx.node'), '', 'utf8') + await writeFile(join(packageDir, 'libonnxruntime.1.23.2.dylib'), '', 'utf8') + await writeFile(join(packageDir, 'libonnxruntime.dylib'), '', 'utf8') + + prunePackagedSherpaOnnx(resourcesDir, 'darwin') + + await expect(readdir(packageDir).then((entries) => entries.sort())).resolves.toEqual([ + 'libonnxruntime.1.23.2.dylib', + 'sherpa-onnx.node' + ]) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('prunes zod TypeScript sources from packaged runtime resources', async () => { + const resourcesDir = await mkdtemp(join(tmpdir(), 'orca-zod-prune-')) + try { + const packageDir = join(resourcesDir, 'node_modules', 'zod') + await mkdir(join(packageDir, 'src'), { recursive: true }) + await writeFile(join(packageDir, 'index.cjs'), 'module.exports = {}', 'utf8') + await writeFile(join(packageDir, 'src', 'index.ts'), 'export const value = true', 'utf8') + + prunePackagedZodSources(resourcesDir) + + await expect(readdir(packageDir)).resolves.toEqual(['index.cjs']) + } finally { + await rm(resourcesDir, { recursive: true, force: true }) + } + }) + + it('fails when the packaged resources directory is missing', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-electron-builder-config-')) + try { + await expect( + electronBuilderConfig.afterPack({ + appOutDir: root, + electronPlatformName: 'win32' + }) + ).rejects.toThrow(/Missing packaged resources directory/) + } finally { + await rm(root, { recursive: true, force: true }) + } + }) + + it.skipIf(process.platform === 'win32')( + 'marks packaged Unix CLI launchers executable', + async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-electron-builder-config-')) + try { + const resourcesDir = join(root, 'linux-unpacked', 'resources') + const launcherPath = join(resourcesDir, 'bin', 'orca-ide') + await mkdir(join(resourcesDir, 'bin'), { recursive: true }) + await cp( + join(process.cwd(), 'resources', 'plugins', 'launch'), + join(resourcesDir, 'plugins', 'launch'), + { recursive: true } + ) + await mkdir(join(resourcesDir, 'node_modules', 'zod', 'src'), { recursive: true }) + // Why: afterPack now fails hard when the unpacked daemon entry is + // missing, so the fixture must carry one like a real package layout. + const unpackedMainDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'main') + await mkdir(unpackedMainDir, { recursive: true }) + await writeFile( + join(unpackedMainDir, 'daemon-entry.js'), + 'console.error("Usage: daemon-entry "); process.exit(1)\n', + 'utf8' + ) + await writeFile( + join(resourcesDir, 'app.asar.unpacked', 'out', 'package.json'), + `${JSON.stringify({ name: 'orca-compiled-output', type: 'commonjs', private: true })}\n`, + 'utf8' + ) + const unpackedCliDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'cli') + await mkdir(join(unpackedCliDir, 'handlers'), { recursive: true }) + await writeFile(join(unpackedCliDir, 'handlers', 'skills.js'), '', 'utf8') + await writeFile( + join(unpackedCliDir, 'index.js'), + [ + 'const args = process.argv.slice(2)', + "if (args[1] === 'list') console.log(JSON.stringify({ topics: [{ name: 'orca-cli' }, { name: 'computer-use' }] }))", + "else if (args[1] === 'get') console.log(`---\\nname: ${args[2]}\\n---`)", + 'else console.log(JSON.stringify({ executed: false }))' + ].join('\n'), + 'utf8' + ) + await writeFile(launcherPath, '#!/usr/bin/env bash\n', { encoding: 'utf8', mode: 0o644 }) + + await electronBuilderConfig.afterPack({ + appOutDir: join(root, 'linux-unpacked'), + electronPlatformName: 'linux', + arch: 1, + packager: { appInfo: { version: '9.9.9' } } + }) + + expect((await stat(launcherPath)).mode & 0o111).not.toBe(0) + await expect( + readFile(join(resourcesDir, 'app.asar.unpacked', 'out', 'package.json'), 'utf8') + ).resolves.toContain('"version": "9.9.9"') + await expect(readFile(join(resourcesDir, 'package-type'), 'utf8')).resolves.toBe('AppImage') + } finally { + await rm(root, { recursive: true, force: true }) + } + } + ) +}) diff --git a/config/scripts/headless-serve-shutdown-workflow.test.mjs b/config/scripts/headless-serve-shutdown-workflow.test.mjs index 6590596e41c..90a3f73c77d 100644 --- a/config/scripts/headless-serve-shutdown-workflow.test.mjs +++ b/config/scripts/headless-serve-shutdown-workflow.test.mjs @@ -5,6 +5,17 @@ import { describe, expect, it } from 'vitest' const workflow = parse(readFileSync('.github/workflows/pr.yml', 'utf8')) const headlessLinuxGuide = readFileSync('docs/reference/headless-linux-server.md', 'utf8') +const signalCase = readFileSync('config/docker/headless-serve-shutdown/run-signal-case.sh', 'utf8') +const shutdownDockerRunner = readFileSync( + 'config/scripts/run-headless-serve-shutdown-docker.mjs', + 'utf8' +) +const shutdownDockerfile = readFileSync('config/docker/headless-serve-shutdown/Dockerfile', 'utf8') +const desktopStartupOracle = readFileSync( + 'config/docker/headless-serve-shutdown/run-appimage-desktop-startup-case.sh', + 'utf8' +) +const headlessLinuxProse = headlessLinuxGuide.replace(/\s+/g, ' ') function readSystemdUnitBlocks(doc, unitName) { const escapedUnitName = unitName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&') @@ -40,16 +51,112 @@ describe('headless serve shutdown PR gate', () => { ).toThrow('Missing closing code fence for orca-serve.service') }) - it('packages an x64 AppImage before running the Docker signal oracle', () => { + it('packages Linux artifacts before running the Docker signal oracle', () => { const steps = workflow.jobs.package.steps const packageStep = steps.find((step) => step.name === 'Package unpacked app') + const markerStep = steps.find((step) => step.name === 'Verify root-package marker payloads') const shutdownStep = steps.find((step) => step.name === 'Verify headless serve signal shutdown') + const launcherShutdownStep = steps.find( + (step) => step.name === 'Verify extracted launcher serve signal shutdown' + ) + const appImageShutdownStep = steps.find( + (step) => step.name === 'Verify AppImage CLI registration and serve signal shutdown' + ) - expect(packageStep.run).toContain('--linux AppImage --x64 --publish never') + expect(workflow.jobs.package['timeout-minutes']).toBe(90) + expect(packageStep.run).toContain('--linux AppImage deb rpm --x64 --publish never') + expect(markerStep.run).toContain('dpkg-deb --fsys-tarfile') + expect(markerStep.run).toContain('rpm2cpio') + expect(steps.indexOf(markerStep)).toBeGreaterThan(steps.indexOf(packageStep)) expect(shutdownStep.run).toBe( 'node config/scripts/run-headless-serve-shutdown-docker.mjs --appimage dist/orca-linux.AppImage' ) + expect(launcherShutdownStep.run).toContain( + 'node config/scripts/run-headless-serve-shutdown-docker.mjs' + ) + expect(launcherShutdownStep.run).toContain('--entrypoint launcher') + expect(appImageShutdownStep.run).toContain('--entrypoint appimage') + expect(appImageShutdownStep.run).toContain('--signal-target serving-electron') + expect(appImageShutdownStep.run).toContain('--int-delivery pid') expect(steps.indexOf(shutdownStep)).toBeGreaterThan(steps.indexOf(packageStep)) + expect(steps.indexOf(shutdownStep)).toBeGreaterThan(steps.indexOf(markerStep)) + expect(steps.indexOf(launcherShutdownStep)).toBeGreaterThan(steps.indexOf(shutdownStep)) + expect(steps.indexOf(appImageShutdownStep)).toBeGreaterThan(steps.indexOf(launcherShutdownStep)) + }) + + it('keeps readiness polling finite and leak-free', () => { + expect(signalCase).toContain('read_ready_line()') + expect(signalCase).toContain("sed -u -n 's/^[^{]*//p'") + expect(signalCase).toContain('startup_timeout_seconds=${ORCA_STARTUP_TIMEOUT_SECONDS:-180}') + expect(signalCase).toContain('startup_deadline=$((SECONDS + startup_timeout_seconds))') + expect(signalCase).toContain('while (( SECONDS < startup_deadline )); do') + expect(signalCase).toContain('kill -0 "$app_pid" 2>/dev/null || break') + expect(signalCase).toContain( + "jq's `inputs` waits for EOF even when wrapped in `first`, so a tail -F" + ) + expect(signalCase).not.toContain('tail --pid=') + }) + + it('gives owned shutdown state a bounded cleanup grace', () => { + expect(signalCase).toContain('for shutdown_poll in {0..50}; do') + expect(signalCase).toContain('[[ -z "$listener_after" && -z "$owned_residue" ]]') + expect(signalCase).toContain('((${#survivors[@]} == 0))') + expect(signalCase).toContain('((shutdown_poll < 50)) && sleep 0.1') + }) + + it('checks that a serving-electron signal target owns the ready socket', () => { + const ssRecord = + 'LISTEN 0 128 127.0.0.1:41235 0.0.0.0:* users:(("orca-ide",pid=23,fd=7),("orca-ide",pid=25,fd=8))' + expect([...ssRecord.matchAll(/pid=([0-9]+)/g)].map((match) => match[1])).toEqual(['23', '25']) + expect(signalCase).toContain( + 'listener_before_pids=$(grep -oE \'pid=[0-9]+\' <<<"$listener_before" | cut -d= -f2 || true)' + ) + expect(signalCase).toContain('signal_target_pid=$(head -n1 <<<"$listener_before_pids")') + expect(signalCase).toContain('outside the entrypoint process tree') + }) + + it('runs the original AppImage desktop startup oracle before extraction and signals', () => { + expect(shutdownDockerfile).toContain( + 'COPY run-appimage-desktop-startup-case.sh /usr/local/bin/run-appimage-desktop-startup-case' + ) + const startupCall = shutdownDockerRunner.indexOf( + 'runDesktopStartupOracle({ image, appImage, platform })' + ) + const extractionCall = shutdownDockerRunner.indexOf( + "'timeout --kill-after=10s 120s /input/orca.AppImage --appimage-extract" + ) + const signalLoop = shutdownDockerRunner.indexOf("for (const signal of ['INT', 'TERM'])") + expect(startupCall).toBeGreaterThan(-1) + expect(extractionCall).toBeGreaterThan(startupCall) + expect(signalLoop).toBeGreaterThan(startupCall) + expect(shutdownDockerRunner).toContain("'/usr/local/bin/run-appimage-desktop-startup-case'") + }) + + it('preserves startup logs when the launcher exits before its marker', () => { + expect(desktopStartupOracle).toContain('signal_process_group TERM || true') + expect(desktopStartupOracle).toContain('signal_process_group KILL || true') + expect(desktopStartupOracle).toContain('cat "$stdout_log" >&2 2>/dev/null || true') + expect(desktopStartupOracle).toContain('cat "$stderr_log" >&2 2>/dev/null || true') + expect(desktopStartupOracle).toContain( + 'FAIL: desktop launcher exited before ${reason} (status=${observed_status})' + ) + expect(desktopStartupOracle).toContain('ORCA_STARTUP_STATE_DIR_CLEANUP=1') + expect(desktopStartupOracle).toContain( + '[[ "$state_dir" =~ ^/tmp/orca-appimage-startup\\.[^/]+$ ]] || return 0' + ) + }) + + it('requires the bound AppImage to be executable before launch and extraction', () => { + expect(desktopStartupOracle).toContain( + '[[ -x "$appimage" ]] || { echo "FAIL: AppImage is not executable: $appimage" >&2; exit 1; }' + ) + expect(shutdownDockerRunner).toContain( + '\'test -r /input/orca.AppImage && test -x /input/orca.AppImage || { echo "FAIL: AppImage bind must be readable and executable" >&2; exit 1; }\'' + ) + }) + + it('gives the original AppImage enough bounded extraction space', () => { + expect(shutdownDockerRunner).toContain("'/tmp:rw,nosuid,nodev,exec,size=1g'") }) it('keeps owned Xvfb alive during the documented systemd graceful stop', () => { @@ -63,4 +170,37 @@ describe('headless serve shutdown PR gate', () => { expect(managedXvfbUnits).toHaveLength(1) expect(managedXvfbUnits[0]).not.toMatch(/^KillMode=/m) }) + + it('distinguishes persisted state from live work during a service restart', () => { + expect(headlessLinuxProse).toContain( + 'Every `systemctl stop` or `restart` therefore ends live terminals and agent processes' + ) + expect(headlessLinuxProse).toContain( + 'These guarantees do not preserve live processes. The service restart kills every terminal and agent in its cgroup' + ) + expect(headlessLinuxProse).toContain( + 'A separately paired runtime is outside that boundary; local execution and SSH hosts reached through this runtime are not. An affected or unknown omission, missing scope, failed request or lost connection is `unverifiable`' + ) + expect(headlessLinuxGuide).toContain( + 'sudo -Hu orca /home/orca/.local/bin/orca-ide terminal list --json' + ) + expect(headlessLinuxGuide).not.toContain('sudo -Hu orca orca-ide terminal list --json') + expect(headlessLinuxGuide).not.toContain('Two facts make this safe and predictable') + }) + + it('uses the registered CLI name from ordinary Linux shells', () => { + const commandRule = + 'The registered Linux CLI command is `orca-ide`, not `orca`, to avoid shadowing the GNOME Orca screen reader.' + const substitutionRule = + "From an ordinary shell outside that service user's managed environment, substitute `orca-ide` for `orca` in commands below." + const censusCommand = '`sudo -Hu orca /home/orca/.local/bin/orca-ide terminal list --json`' + + expect(headlessLinuxProse).toContain(commandRule) + expect(headlessLinuxProse).toContain(substitutionRule) + expect(headlessLinuxProse).toContain(censusCommand) + expect(headlessLinuxGuide).toContain('best-effort dispatcher at `$HOME/.local/bin/orca`') + expect(headlessLinuxProse.indexOf(substitutionRule)).toBeLessThan( + headlessLinuxProse.indexOf(censusCommand) + ) + }) }) diff --git a/config/scripts/linux-package-maintainer-scripts.test.mjs b/config/scripts/linux-package-maintainer-scripts.test.mjs new file mode 100644 index 00000000000..f315f2fd2ee --- /dev/null +++ b/config/scripts/linux-package-maintainer-scripts.test.mjs @@ -0,0 +1,18 @@ +import { readFileSync } from 'node:fs' +import { describe, expect, it } from 'vitest' + +describe('Linux package maintainer scripts', () => { + it('keeps upgrades from removing the installed CLI', () => { + const script = readFileSync( + new URL('../../resources/linux/packaging/after-remove.sh', import.meta.url), + 'utf8' + ) + const unlinkStart = script.indexOf('link="/usr/bin/orca-ide"') + const upgradeGuard = script.slice(0, unlinkStart) + + expect(unlinkStart).toBeGreaterThan(-1) + expect(upgradeGuard).toContain('case "${1-}" in') + expect(upgradeGuard).toContain('0 | remove | purge) ;;') + expect(upgradeGuard).toContain('*) exit 0 ;;') + }) +}) diff --git a/config/scripts/orcad-operations-restart-safety.test.mjs b/config/scripts/orcad-operations-restart-safety.test.mjs new file mode 100644 index 00000000000..60f9cb05524 --- /dev/null +++ b/config/scripts/orcad-operations-restart-safety.test.mjs @@ -0,0 +1,42 @@ +import { readFileSync } from 'node:fs' + +import { describe, expect, it } from 'vitest' + +const operationsGuide = readFileSync('docs/reference/orcad-operations.md', 'utf8') +const operationsProse = operationsGuide.replace(/\s+/g, ' ') + +describe('orcad operations restart safety', () => { + it('distinguishes PID-scoped preservation from systemd cgroup teardown', () => { + expect(operationsProse).toContain( + 'This makes a PID-scoped update, rollback or restart non-destructive to live work' + ) + expect(operationsProse).toContain( + 'The successor adopts the current endpoint and routes supported previous protocol versions through legacy adapters' + ) + expect(operationsProse).toContain('`KillMode=mixed` does **not** preserve them') + expect(operationsProse).toContain( + '`KillMode=process` leaves service-owned processes unmanaged and is not a supported preservation mechanism' + ) + }) + + it('fails closed before cgroup-wide maintenance', () => { + expect(operationsProse).toContain( + 'A safe empty census is untruncated, has an explicit `hostScope`, covers every execution host affected by the stop, and lists no terminals on those hosts' + ) + expect(operationsProse).toContain( + "Every `omittedHostIds` entry must be explicitly accounted for outside the target service's execution boundary" + ) + expect(operationsProse).toContain( + '`sudo -Hu orca /home/orca/.local/bin/orca-ide terminal list --json`' + ) + expect(operationsGuide).not.toContain('sudo -Hu orca orca-ide terminal list --json') + expect(operationsProse).toContain( + 'A separately paired runtime is outside that boundary; local execution and SSH hosts reached through this runtime are not. An affected or unknown omission, missing scope, truncation, a failed request or lost contact makes the result `unverifiable`' + ) + expect(operationsProse).toContain('Orca does not yet provide an atomic census-and-stop fence') + }) + + it('does not refer to the unavailable shipping design', () => { + expect(operationsGuide).not.toContain('docs/design/shipping-orcad.html') + }) +}) diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index c296ad9e893..67d4f565afa 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -167,6 +167,7 @@ const SHARED_PACKAGE_PREFIXES = [ 'config/scripts/smoke-packaged', 'config/scripts/install-electron-package-binary', 'config/scripts/verify-packaged', + 'config/scripts/verify-skills-cli-runtime', 'config/scripts/verify-linux-glibc', 'config/scripts/run-electron-vite', 'skills/', @@ -180,6 +181,12 @@ const SHARED_PACKAGE_PREFIXES = [ const LINUX_PACKAGE_PREFIXES = [ ...SHARED_PACKAGE_PREFIXES, + 'config/docker/cli-launch-contract/', + 'config/docker/headless-pairing/', + 'config/docker/headless-serve-shutdown/', + 'config/scripts/run-linux-cli-launch-contract', + 'config/scripts/run-headless-linux-pairing-docker', + 'config/scripts/static-appimage-package-contract', 'native/computer-use-linux/', 'resources/linux/', 'config/scripts/run-headless-serve' diff --git a/config/scripts/pr-code-change-scope.test.mjs b/config/scripts/pr-code-change-scope.test.mjs index f9411eed956..9a1c9e649b6 100644 --- a/config/scripts/pr-code-change-scope.test.mjs +++ b/config/scripts/pr-code-change-scope.test.mjs @@ -181,6 +181,28 @@ describe('per-job path classification', () => { expectClassification(['native/computer-use-macos/Package.swift'], {}) }) + it('runs Linux packaging when an artifact contract changes', () => { + for (const file of [ + 'config/docker/cli-launch-contract/Dockerfile', + 'config/docker/cli-launch-contract/run-cli-case.sh', + 'config/docker/headless-pairing/Dockerfile', + 'config/docker/headless-pairing/run-appimage-case.sh', + 'config/docker/headless-serve-shutdown/Dockerfile', + 'config/scripts/run-linux-cli-launch-contract-docker.mjs', + 'config/scripts/run-headless-linux-pairing-docker.mjs', + 'config/scripts/static-appimage-package-contract.cjs' + ]) { + expectClassification([file], { package: true }) + } + }) + + it('runs both package jobs when the shared skills runtime verifier changes', () => { + expectClassification(['config/scripts/verify-skills-cli-runtime.cjs'], { + package: true, + package_windows: true + }) + }) + it('runs shell contracts when live-shell inputs change', () => { expectClassification(['src/main/daemon/shell-ready.ts'], { shell_contracts: true, diff --git a/config/scripts/pr-workflow-parallelism.test.mjs b/config/scripts/pr-workflow-parallelism.test.mjs index c69b04d663f..92d4fe4c26b 100644 --- a/config/scripts/pr-workflow-parallelism.test.mjs +++ b/config/scripts/pr-workflow-parallelism.test.mjs @@ -102,8 +102,13 @@ describe('PR workflow parallelism', () => { .split(/\s+/) .filter((token) => !['apt-get', 'install', 'sudo', ''].includes(token)) .filter((token) => !token.startsWith('-')) - const jobsInstallingPackages = Object.entries(workflow.jobs) - .filter(([, job]) => (job.steps ?? []).some((step) => aptPackages(step).length > 0)) + const requiredShells = ['zsh', 'fish'] + const jobsInstallingShells = Object.entries(workflow.jobs) + .filter(([, job]) => + (job.steps ?? []).some((step) => + aptPackages(step).some((packageName) => requiredShells.includes(packageName)) + ) + ) .map(([name]) => name) expect(shellStep).toBeDefined() @@ -111,11 +116,11 @@ describe('PR workflow parallelism', () => { expect(shellStep.run.split(/\s+/)).toContain('--maxWorkers=1') // Why the whole workflow, not just the general shards: any other lane installing // these shells would silently start running the real-shell tests twice. - expect(jobsInstallingPackages).toEqual(['shell_contracts']) + expect(jobsInstallingShells).toEqual(['shell_contracts']) // Why each shell is asserted: the live tests skip themselves when the binary is // missing, so a dropped package silently empties this lane instead of failing it. const shellPackages = workflow.jobs.shell_contracts.steps.flatMap(aptPackages) - for (const shell of ['zsh', 'fish']) { + for (const shell of requiredShells) { expect(shellPackages).toContain(shell) } expect(shellInstall.with['native-runtime']).toBe('node') diff --git a/config/scripts/run-headless-serve-shutdown-docker.mjs b/config/scripts/run-headless-serve-shutdown-docker.mjs index f3852624854..184713c41a0 100755 --- a/config/scripts/run-headless-serve-shutdown-docker.mjs +++ b/config/scripts/run-headless-serve-shutdown-docker.mjs @@ -17,7 +17,7 @@ if (!appImageArg) { if (!['app', 'serving-electron'].includes(signalTarget)) { fail(`Unsupported --signal-target: ${signalTarget}`) } -if (!['app', 'launcher'].includes(entrypoint)) { +if (!['app', 'appimage', 'launcher'].includes(entrypoint)) { fail(`Unsupported --entrypoint: ${entrypoint}`) } if (!['pid', 'foreground-process-group'].includes(intDelivery)) { @@ -54,11 +54,19 @@ try { shutdownDockerDirectory ]) docker(['volume', 'create', artifactVolume]) + runDesktopStartupOracle({ image, appImage, platform }) docker([ 'run', '--rm', '--platform', platform, + '--network', + 'none', + '--read-only', + '--cap-drop', + 'ALL', + '--security-opt', + 'no-new-privileges', '--entrypoint', 'bash', '-v', @@ -68,11 +76,17 @@ try { image, '-lc', [ - '7z x /input/orca.AppImage -o/artifacts/root -y >/dev/null', + 'trap \'status=$?; if [ "$status" -ne 0 ]; then cat /artifacts/appimage-help.log /artifacts/appimage-extract.log 2>/dev/null || true; fi; exit "$status"\' EXIT', + 'test -r /input/orca.AppImage && test -x /input/orca.AppImage || { echo "FAIL: AppImage bind must be readable and executable" >&2; exit 1; }', + 'timeout --kill-after=5s 15s /input/orca.AppImage --appimage-help > /artifacts/appimage-help.log 2>&1', + 'cd /artifacts', + 'timeout --kill-after=10s 120s /input/orca.AppImage --appimage-extract > /artifacts/appimage-extract.log 2>&1', + 'mv squashfs-root root', launcherExecOverlay ? "sed -i 's/^ELECTRON_RUN_AS_NODE=1 /export ELECTRON_RUN_AS_NODE=1\\nexec /' /artifacts/root/resources/bin/orca-ide" : ':', - 'chmod -R a+rX /artifacts/root' + 'chmod -R a+rX /artifacts/root', + 'rm /artifacts/appimage-help.log /artifacts/appimage-extract.log' ].join(' && ') ]) @@ -108,6 +122,8 @@ try { '-e', `ORCA_INT_DELIVERY=${intDelivery}`, '-v', + `${appImage}:/input/orca.AppImage:ro`, + '-v', `${artifactVolume}:/artifacts:ro`, image, signal @@ -129,6 +145,38 @@ try { docker(['image', 'rm', image], { allowFailure: true }) } +function runDesktopStartupOracle({ image, appImage, platform }) { + console.log('Running original AppImage desktop startup oracle...') + docker([ + 'run', + '--rm', + '--init', + '--platform', + platform, + '--network', + 'none', + '--read-only', + '--tmpfs', + '/tmp:rw,nosuid,nodev,exec,size=1g', + '--shm-size', + '256m', + '--cap-drop', + 'ALL', + '--security-opt', + 'no-new-privileges', + '--user', + 'orca', + '--entrypoint', + '/usr/local/bin/run-appimage-desktop-startup-case', + '-e', + 'ORCA_STARTUP_DIAGNOSTICS=1', + '-v', + `${appImage}:/input/orca.AppImage:ro`, + image, + '/input/orca.AppImage' + ]) +} + function valueAfter(flag) { const index = args.indexOf(flag) return index === -1 ? null : (args[index + 1] ?? null) diff --git a/config/scripts/run-linux-cli-launch-contract-docker.mjs b/config/scripts/run-linux-cli-launch-contract-docker.mjs new file mode 100755 index 00000000000..901e0877e85 --- /dev/null +++ b/config/scripts/run-linux-cli-launch-contract-docker.mjs @@ -0,0 +1,264 @@ +#!/usr/bin/env node +// Exercise packaged CLI paths under the hostile Linux conditions from #11609/#12530/#13719/#14229. +import { execFileSync } from 'node:child_process' +import { existsSync } from 'node:fs' +import { resolve } from 'node:path' + +const commandArgs = process.argv.slice(2) +const appImageArg = valueAfter('--appimage') +const appImage = appImageArg ? resolve(appImageArg) : null +const platform = valueAfter('--platform') +const dockerPlatformArgs = platform ? ['--platform', platform] : [] + +const suffix = `${process.pid}-${Date.now()}` +const artifactVolume = `orca-cli-contract-artifact-${suffix}` +const tagArchitecture = platform?.split('/')[1] ?? process.arch +const tag = `orca-cli-launch-contract:ubuntu-24.04-${tagArchitecture}-${suffix}` +const base = 'ubuntu@sha256:4fbb8e6a8395de5a7550b33509421a2bafbc0aab6c06ba2cef9ebffbc7092d90' +const containers = new Set() +let artifactVolumeCreated = false +const CASE_TIMEOUT_MS = 90_000 +const BUILD_TIMEOUT_MS = 10 * 60_000 +const STAGING_TIMEOUT_MS = 5 * 60_000 +const DOCKER_TIMEOUT_MS = 2 * 60_000 +const CLEANUP_TIMEOUT_MS = 30_000 + +// Exact statuses reject silent no-op launches as well as crashes. +const CASES = [ + { + name: 'nofuse-userns-bundled-help', + expectStatus: 0, + expectOutput: 'Usage: orca ', + why: 'The bundled launcher must run with no FUSE, no display, and userns restricted (#11609, #12530).' + }, + { + name: 'nofuse-userns-bundled-version', + expectStatus: 0, + expectOutput: /^\d+\.\d+\.\d+/m, + why: 'A deployment must be able to read the installed version without a display (#13719).' + }, + { + name: 'nofuse-userns-bundled-status', + // No runtime is running; the CLI must report that itself. + expectStatus: 1, + expectOutput: 'appRunning', + why: 'A command that needs the runtime must report its absence, not abort.' + }, + { + name: 'nofuse-userns-bundled-skills', + expectStatus: 0, + // Why: the rendered help header, not a bare 'skills' — the case name contains that word. + expectOutput: 'Usage: orca skills', + why: 'skills is a pure-text command that must never need Chromium (#14229).' + }, + { + name: 'nofuse-userns-bundled-worktree', + expectStatus: 1, + expectOutput: "Orca is not running. Run 'orca open' first.", + why: 'A runtime-dependent command must report the missing runtime, not abort.' + }, + { + name: 'nofuse-nosandbox-direct-binary-skills', + expectStatus: 0, + expectOutput: 'Usage: orca skills', + why: 'A direct binary launch that reaches JavaScript must run the command, not boot a GUI (#14229).' + }, + { + name: 'nofuse-nosandbox-direct-binary-gui', + // A missing display is an expected diagnosis, not a crash. + expectStatus: 1, + expectOutput: 'needs a usable display server', + why: 'A desktop launch with no display must diagnose it instead of dying in uv_close (#13719).' + }, + { + name: 'stale-display-nosandbox-direct-binary-gui', + expectStatus: 1, + expectOutput: 'needs a usable display server', + why: 'A stale DISPLAY value must diagnose the unreachable endpoint instead of dying in uv_close (#13719).' + } +] + +try { + if (!appImage) { + fail( + 'Usage: run-linux-cli-launch-contract-docker.mjs --appimage /path/to/orca-linux.AppImage [--platform linux/amd64|linux/arm64]' + ) + } + if (commandArgs.includes('--platform') && !platform) { + fail('Missing value for --platform') + } + if (platform !== null && platform !== 'linux/amd64' && platform !== 'linux/arm64') { + fail(`Unsupported --platform: ${platform}`) + } + if (!existsSync(appImage)) { + fail(`AppImage not found: ${appImage}`) + } + docker(['volume', 'create', artifactVolume], { timeoutMs: DOCKER_TIMEOUT_MS }) + artifactVolumeCreated = true + buildImage() + stageArtifacts() + runContract() + console.log('\nLinux CLI launch contract passed.') +} catch (error) { + console.error(error instanceof Error ? error.message : String(error)) + process.exitCode = 1 +} finally { + for (const container of containers) { + docker(['rm', '-f', container], { allowFailure: true, timeoutMs: CLEANUP_TIMEOUT_MS }) + } + if (artifactVolumeCreated) { + docker(['volume', 'rm', artifactVolume], { + allowFailure: true, + timeoutMs: CLEANUP_TIMEOUT_MS + }) + } + docker(['image', 'rm', tag], { allowFailure: true, timeoutMs: CLEANUP_TIMEOUT_MS }) +} + +function runContract() { + const failures = [] + for (const testCase of CASES) { + const output = runCase(testCase.name) + const statusMatch = /^RESULT status=(\d+)/m.exec(output) + if (!statusMatch) { + failures.push(`${testCase.name}: ${firstLine(output)}\n ${testCase.why}`) + console.log(` FAIL ${testCase.name} — ${firstLine(output)}`) + continue + } + const status = Number(statusMatch[1]) + // Why: the harness echoes `RESULT status=N case=`, so a case whose name contains the + // expected substring would assert against the harness's own line instead of the CLI's output. + const commandOutput = output + .split('\n') + .filter((line) => !/^(?:RESULT|CRASHED|PRECONDITION_FAILED) /.test(line)) + .join('\n') + const matchesOutput = + typeof testCase.expectOutput === 'string' + ? commandOutput.includes(testCase.expectOutput) + : testCase.expectOutput.test(commandOutput) + if (status !== testCase.expectStatus || !matchesOutput) { + failures.push( + `${testCase.name}: expected status ${testCase.expectStatus} and ${testCase.expectOutput}, ` + + `got status ${status}\n ${testCase.why}` + ) + console.log(` FAIL ${testCase.name} — status ${status}`) + continue + } + console.log(` ok ${testCase.name} (status ${status})`) + } + if (failures.length > 0) { + fail(`Linux CLI launch contract failed:\n - ${failures.join('\n - ')}`) + } +} + +function runCase(caseName) { + const container = `orca-cli-contract-${caseName}-${suffix}` + containers.add(container) + // FUSE and extra capabilities would invalidate the test conditions. + return docker( + [ + 'run', + ...dockerPlatformArgs, + '--name', + container, + '--rm', + '-v', + `${artifactVolume}:/artifacts`, + tag, + caseName + ], + { allowFailure: true, capture: true, timeoutMs: CASE_TIMEOUT_MS } + ) +} + +function buildImage() { + console.log(`Building ${tag}…`) + docker( + [ + 'build', + ...dockerPlatformArgs, + '--build-arg', + `BASE_IMAGE=${base}`, + '-f', + 'config/docker/cli-launch-contract/Dockerfile', + '-t', + tag, + 'config/docker/cli-launch-contract' + ], + { timeoutMs: BUILD_TIMEOUT_MS } + ) +} + +// Extract unprivileged so chrome-sandbox is not root-owned setuid. +function stageArtifacts() { + console.log('Staging the AppImage payload…') + const container = `orca-cli-contract-stage-${suffix}` + containers.add(container) + docker( + [ + 'run', + ...dockerPlatformArgs, + '--name', + container, + '--rm', + '-v', + `${artifactVolume}:/artifacts`, + '-v', + `${appImage}:/input/orca-linux.AppImage:ro`, + '--entrypoint', + 'bash', + tag, + '-lc', + [ + 'set -euo pipefail', + 'cp /input/orca-linux.AppImage /artifacts/orca-linux.AppImage', + 'chmod +x /artifacts/orca-linux.AppImage', + 'chown -R orca:orca /artifacts', + // Use the AppImage runtime's no-FUSE extraction path. + 'cd /artifacts && runuser --user orca -- ./orca-linux.AppImage --appimage-extract >/dev/null', + 'test -x /artifacts/squashfs-root/resources/bin/orca-ide' + ].join(' && ') + ], + { timeoutMs: STAGING_TIMEOUT_MS } + ) +} + +function docker(args, options = {}) { + try { + const output = execFileSync('docker', args, { + encoding: 'utf8', + stdio: options.capture ? ['ignore', 'pipe', 'pipe'] : 'inherit', + timeout: options.timeoutMs ?? DOCKER_TIMEOUT_MS, + killSignal: 'SIGTERM' + }) + return output ?? '' + } catch (error) { + const timedOut = error instanceof Error && 'code' in error && error.code === 'ETIMEDOUT' + if (timedOut) { + const message = `docker ${args.join(' ')} timed out after ${options.timeoutMs ?? DOCKER_TIMEOUT_MS}ms` + if (!options.allowFailure) { + fail(message) + } + return message + } + if (!options.allowFailure) { + fail( + `docker ${args.join(' ')} failed: ${error instanceof Error ? error.message : String(error)}` + ) + } + return `${error?.stdout ?? ''}${error?.stderr ?? ''}` + } +} + +function firstLine(value) { + return (value ?? '').trim().split('\n')[0] || '(no output)' +} + +function valueAfter(flag) { + const index = commandArgs.indexOf(flag) + return index === -1 ? null : (commandArgs[index + 1] ?? null) +} + +function fail(message) { + throw new Error(message) +} diff --git a/config/scripts/static-appimage-package-contract.cjs b/config/scripts/static-appimage-package-contract.cjs new file mode 100644 index 00000000000..8a11cecf880 --- /dev/null +++ b/config/scripts/static-appimage-package-contract.cjs @@ -0,0 +1,260 @@ +const { closeSync, fstatSync, openSync, readSync } = require('node:fs') +const { basename } = require('node:path') + +const EXPECTED_ARCHITECTURE_BY_FILENAME = new Map([ + ['orca-linux.AppImage', 'x64'], + ['orca-linux-arm64.AppImage', 'arm64'] +]) +const APPIMAGE_MAGIC = Buffer.from([0x41, 0x49, 0x02]) +const RUNTIME_SOURCE = Buffer.from('https://github.com/AppImage/type2-runtime') +const TARGET_ARCHITECTURE_BY_ENUM = new Map([ + [1, 'x64'], + [3, 'arm64'] +]) +const RUNTIME_ARCHITECTURE_BY_MACHINE = new Map([ + [0x3e, 'x64'], + [0xb7, 'arm64'] +]) +const ELF_HEADER_BYTES = 64 +const PROGRAM_HEADER_BYTES = 56 +const DYNAMIC_ENTRY_BYTES = 16 +const MAX_PROGRAM_HEADERS = 128 +const MAX_LOAD_BYTES = 16 * 1024 * 1024 +const MAX_DYNAMIC_BYTES = 1024 * 1024 + +function verifyStaticAppImagePackage(filePath, targetArch) { + const filename = basename(filePath) + const filenameArchitecture = EXPECTED_ARCHITECTURE_BY_FILENAME.get(filename) + if (!filenameArchitecture) { + invalid( + filename, + `unsupported artifact name; expected ${[...EXPECTED_ARCHITECTURE_BY_FILENAME.keys()].join(' or ')}` + ) + } + const targetArchitecture = normalizeTargetArchitecture(targetArch, filename) + if (filenameArchitecture !== targetArchitecture) { + invalid( + filename, + `artifact filename targets ${filenameArchitecture}, but electron-builder target is ${targetArchitecture}` + ) + } + + const descriptor = openSync(filePath, 'r') + try { + const stats = fstatSync(descriptor, { bigint: true }) + if (process.platform !== 'win32' && (stats.mode & 0o111n) === 0n) { + invalid(filename, 'artifact is not executable') + } + const fileSize = stats.size + const header = readRange( + descriptor, + 0n, + BigInt(ELF_HEADER_BYTES), + fileSize, + filename, + 'ELF header' + ) + const { entry, machine } = verifyElfHeader(header, filename) + const runtimeArchitecture = RUNTIME_ARCHITECTURE_BY_MACHINE.get(machine) + if (runtimeArchitecture !== targetArchitecture) { + invalid( + filename, + `runtime architecture ${runtimeArchitecture ?? `machine 0x${machine.toString(16)}`} does not match electron-builder target ${targetArchitecture}` + ) + } + + const programHeaderOffset = header.readBigUInt64LE(32) + const programHeaderSize = header.readUInt16LE(54) + const programHeaderCount = header.readUInt16LE(56) + if (programHeaderSize !== PROGRAM_HEADER_BYTES) { + invalid(filename, `unexpected ELF program-header size ${programHeaderSize}`) + } + if (programHeaderCount === 0 || programHeaderCount > MAX_PROGRAM_HEADERS) { + invalid(filename, `invalid ELF program-header count ${programHeaderCount}`) + } + + const tableSize = BigInt(programHeaderSize * programHeaderCount) + const table = readRange( + descriptor, + programHeaderOffset, + tableSize, + fileSize, + filename, + 'ELF program-header table' + ) + const segments = parseProgramHeaders(table, programHeaderSize) + verifySegments(descriptor, segments, fileSize, filename, entry) + } finally { + closeSync(descriptor) + } +} + +function verifyElfHeader(header, filename) { + if (!header.subarray(0, 4).equals(Buffer.from([0x7f, 0x45, 0x4c, 0x46]))) { + invalid(filename, 'missing ELF magic') + } + if (header[4] !== 2 || header[5] !== 1 || header[6] !== 1) { + invalid(filename, 'runtime must be ELF64 little-endian version 1') + } + if (!header.subarray(8, 11).equals(APPIMAGE_MAGIC)) { + invalid(filename, 'missing type-2 AppImage marker') + } + if (header.readUInt16LE(16) !== 3) { + invalid(filename, 'runtime must be an ET_DYN static PIE') + } + const machine = header.readUInt16LE(18) + if (!RUNTIME_ARCHITECTURE_BY_MACHINE.has(machine)) { + invalid(filename, `unsupported ELF machine 0x${machine.toString(16)}`) + } + if (header.readUInt32LE(20) !== 1) { + invalid(filename, 'runtime has an unsupported ELF version') + } + if (header.readUInt16LE(52) !== ELF_HEADER_BYTES) { + invalid(filename, `unexpected ELF header size ${header.readUInt16LE(52)}`) + } + return { entry: header.readBigUInt64LE(24), machine } +} + +function parseProgramHeaders(table, entrySize) { + const segments = [] + for (let offset = 0; offset < table.length; offset += entrySize) { + segments.push({ + type: table.readUInt32LE(offset), + flags: table.readUInt32LE(offset + 4), + offset: table.readBigUInt64LE(offset + 8), + virtualAddress: table.readBigUInt64LE(offset + 16), + fileSize: table.readBigUInt64LE(offset + 32), + memorySize: table.readBigUInt64LE(offset + 40) + }) + } + return segments +} + +function verifySegments(descriptor, segments, fileSize, filename, entry) { + if (segments.some((segment) => segment.type === 3)) { + invalid(filename, 'runtime contains PT_INTERP') + } + + const loadSegments = segments.filter((segment) => segment.type === 1) + const totalLoadBytes = loadSegments.reduce((total, segment) => total + segment.fileSize, 0n) + if (loadSegments.length === 0 || totalLoadBytes > BigInt(MAX_LOAD_BYTES)) { + invalid(filename, `invalid or oversized PT_LOAD data (${totalLoadBytes} bytes)`) + } + if ( + !loadSegments.some( + (segment) => + segment.flags & 1 && + entry >= segment.virtualAddress && + entry - segment.virtualAddress < segment.memorySize + ) + ) { + invalid(filename, 'ELF entry point is outside an executable PT_LOAD segment') + } + let identifiesStaticRuntime = false + for (const segment of loadSegments) { + verifyFileBackedSegment(segment, fileSize, filename, 'PT_LOAD') + const data = readRange( + descriptor, + segment.offset, + segment.fileSize, + fileSize, + filename, + 'PT_LOAD data' + ) + identifiesStaticRuntime ||= data.includes(RUNTIME_SOURCE) + } + if (!identifiesStaticRuntime) { + invalid(filename, `runtime does not identify ${RUNTIME_SOURCE.toString()}`) + } + + for (const segment of segments.filter((entry) => entry.type === 2)) { + verifyDynamicSegment(descriptor, segment, fileSize, filename) + } +} + +function normalizeTargetArchitecture(targetArch, filename) { + const architecture = + typeof targetArch === 'number' ? TARGET_ARCHITECTURE_BY_ENUM.get(targetArch) : targetArch + if (architecture !== 'x64' && architecture !== 'arm64') { + invalid(filename, `unsupported electron-builder target architecture ${String(targetArch)}`) + } + return architecture +} + +function verifyFileBackedSegment(segment, fileSize, filename, label) { + if (segment.memorySize < segment.fileSize) { + invalid(filename, `${label} memory size is smaller than its file size`) + } + verifyRange(segment.offset, segment.fileSize, fileSize, filename, label) +} + +function verifyDynamicSegment(descriptor, segment, fileSize, filename) { + verifyFileBackedSegment(segment, fileSize, filename, 'PT_DYNAMIC') + if ( + segment.fileSize === 0n || + segment.fileSize > BigInt(MAX_DYNAMIC_BYTES) || + segment.fileSize % BigInt(DYNAMIC_ENTRY_BYTES) !== 0n + ) { + invalid(filename, `invalid PT_DYNAMIC size ${segment.fileSize}`) + } + const dynamic = readRange( + descriptor, + segment.offset, + segment.fileSize, + fileSize, + filename, + 'PT_DYNAMIC data' + ) + let terminated = false + for (let offset = 0; offset < dynamic.length; offset += DYNAMIC_ENTRY_BYTES) { + const tag = dynamic.readBigInt64LE(offset) + if (tag === 0n) { + terminated = true + break + } + if (tag === 1n) { + invalid(filename, 'runtime contains a DT_NEEDED dependency') + } + } + if (!terminated) { + invalid(filename, 'PT_DYNAMIC is missing DT_NULL') + } +} + +function readRange(descriptor, offset, size, fileSize, filename, label) { + verifyRange(offset, size, fileSize, filename, label) + const buffer = Buffer.alloc(Number(size)) + let bytesRead = 0 + while (bytesRead < buffer.length) { + const count = readSync( + descriptor, + buffer, + bytesRead, + buffer.length - bytesRead, + Number(offset) + bytesRead + ) + if (count === 0) { + throw new Error(`Unable to read complete ${label}`) + } + bytesRead += count + } + return buffer +} + +function verifyRange(offset, size, fileSize, filename, label) { + const maxSafeOffset = BigInt(Number.MAX_SAFE_INTEGER) + if ( + offset > fileSize || + size > fileSize - offset || + offset > maxSafeOffset || + size > maxSafeOffset - offset + ) { + invalid(filename, `${label} is outside the artifact`) + } +} + +function invalid(filename, reason) { + throw new Error(`Invalid static AppImage ${filename}: ${reason}`) +} + +module.exports = { verifyStaticAppImagePackage } diff --git a/config/scripts/static-appimage-package-contract.test.mjs b/config/scripts/static-appimage-package-contract.test.mjs new file mode 100644 index 00000000000..2addc675482 --- /dev/null +++ b/config/scripts/static-appimage-package-contract.test.mjs @@ -0,0 +1,225 @@ +import { chmod, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' + +const require = createRequire(import.meta.url) +const { verifyStaticAppImagePackage } = require('./static-appimage-package-contract.cjs') + +const RUNTIME_SOURCE = Buffer.from('https://github.com/AppImage/type2-runtime') +const LOAD_HEADER = 64 +const DYNAMIC_HEADER = 120 +const DYNAMIC_OFFSET = 320 +const FIXTURE_BYTES = 384 + +describe('static AppImage package contract', () => { + it.each([ + ['orca-linux.AppImage', 0x3e, 1], + ['orca-linux-arm64.AppImage', 0xb7, 'arm64'] + ])('accepts a dependency-free type-2 %s runtime', async (filename, machine, targetArch) => { + await withFixture(filename, createRuntime({ machine }), (path) => { + expect(() => verifyStaticAppImagePackage(path, targetArch)).not.toThrow() + }) + }) + + it.each([ + ['generic filename for an arm64 runtime and target', 'orca-linux.AppImage', 0xb7, 3], + ['arm64 filename for an x64 runtime and target', 'orca-linux-arm64.AppImage', 0x3e, 1], + ['generic x64 runtime for an arm64 target', 'orca-linux.AppImage', 0x3e, 3], + ['generic arm64 runtime for an x64 target', 'orca-linux.AppImage', 0xb7, 1], + ['arm64 artifact filename for an x64 target', 'orca-linux-arm64.AppImage', 0xb7, 1], + ['x64 runtime under an arm64 artifact filename', 'orca-linux-arm64.AppImage', 0x3e, 3] + ])('rejects %s', async (_label, filename, machine, targetArch) => { + await withFixture(filename, createRuntime({ machine }), (path) => { + expect(() => verifyStaticAppImagePackage(path, targetArch)).toThrow(/architecture|target/) + }) + }) + + it.each([undefined, 0, 'ia32'])( + 'rejects unsupported target architecture %s', + async (targetArch) => { + await withFixture('orca-linux.AppImage', createRuntime(), (path) => { + expect(() => verifyStaticAppImagePackage(path, targetArch)).toThrow(/target architecture/) + }) + } + ) + + it('accepts PT_DYNAMIC relocation metadata without dependencies', async () => { + const runtime = createRuntime() + runtime.writeBigInt64LE(7n, DYNAMIC_OFFSET) + await withFixture('orca-linux.AppImage', runtime, (path) => { + expect(() => verifyStaticAppImagePackage(path, 1)).not.toThrow() + }) + }) + + it('does not scan the appended AppImage payload as outer ELF data', async () => { + const payload = Buffer.concat([RUNTIME_SOURCE, Buffer.alloc(16, 1)]) + await withFixture('orca-linux.AppImage', Buffer.concat([createRuntime(), payload]), (path) => { + expect(() => verifyStaticAppImagePackage(path, 1)).not.toThrow() + }) + + const unidentifiedRuntime = createRuntime() + unidentifiedRuntime.fill(0, 192, 192 + RUNTIME_SOURCE.length) + await withFixture( + 'orca-linux.AppImage', + Buffer.concat([unidentifiedRuntime, payload]), + (path) => { + expect(() => verifyStaticAppImagePackage(path, 1)).toThrow(/does not identify/) + } + ) + }) + + it('rejects artifact names outside the release contract before reading them', () => { + expect(() => verifyStaticAppImagePackage('/missing/orca-preview.AppImage')).toThrow( + 'unsupported artifact name' + ) + }) + + it.skipIf(process.platform === 'win32')( + 'rejects a readable but non-executable AppImage', + async () => { + await withFixture( + 'orca-linux.AppImage', + createRuntime(), + (path) => { + expect(() => verifyStaticAppImagePackage(path, 1)).toThrow(/not executable/) + }, + { mode: 0o644 } + ) + } + ) + + it.each([ + [ + 'non-ELF64 runtimes', + (runtime) => { + runtime[4] = 1 + }, + /ELF64 little-endian/ + ], + [ + 'unsupported ELF versions', + (runtime) => runtime.writeUInt32LE(2, 20), + /unsupported ELF version/ + ], + [ + 'non-type-2 AppImages', + (runtime) => { + runtime[10] = 1 + }, + /type-2 AppImage marker/ + ], + ['non-PIE runtimes', (runtime) => runtime.writeUInt16LE(2, 16), /ET_DYN static PIE/], + [ + 'unsupported architectures', + (runtime) => runtime.writeUInt16LE(3, 18), + /unsupported ELF machine/ + ], + ['dynamic loaders', (runtime) => runtime.writeUInt32LE(3, DYNAMIC_HEADER), /PT_INTERP/], + [ + 'shared-library dependencies', + (runtime) => runtime.writeBigInt64LE(1n, DYNAMIC_OFFSET), + /DT_NEEDED/ + ], + [ + 'unidentified runtimes', + (runtime) => runtime.fill(0, 192, 192 + RUNTIME_SOURCE.length), + /does not identify/ + ], + [ + 'out-of-bounds load segments', + (runtime) => { + runtime.writeBigUInt64LE(1000n, LOAD_HEADER + 32) + runtime.writeBigUInt64LE(1000n, LOAD_HEADER + 40) + }, + /outside the artifact/ + ], + [ + 'oversized load claims', + (runtime) => { + runtime.writeBigUInt64LE(16n * 1024n * 1024n + 1n, LOAD_HEADER + 32) + runtime.writeBigUInt64LE(16n * 1024n * 1024n + 1n, LOAD_HEADER + 40) + }, + /oversized PT_LOAD/ + ], + [ + 'non-executable entry segments', + (runtime) => runtime.writeUInt32LE(4, LOAD_HEADER + 4), + /executable PT_LOAD/ + ], + [ + 'entry points outside load segments', + (runtime) => runtime.writeBigUInt64LE(4096n, 24), + /entry point/ + ] + ])('rejects %s', async (_label, mutate, expected) => { + const runtime = createRuntime() + mutate(runtime) + await withFixture('orca-linux.AppImage', runtime, (path) => { + expect(() => verifyStaticAppImagePackage(path, 1)).toThrow(expected) + }) + }) +}) + +function createRuntime({ machine = 0x3e } = {}) { + const runtime = Buffer.alloc(FIXTURE_BYTES) + Buffer.from([0x7f, 0x45, 0x4c, 0x46, 2, 1, 1]).copy(runtime) + Buffer.from([0x41, 0x49, 0x02]).copy(runtime, 8) + runtime.writeUInt16LE(3, 16) + runtime.writeUInt16LE(machine, 18) + runtime.writeUInt32LE(1, 20) + runtime.writeBigUInt64LE(0n, 24) + runtime.writeBigUInt64LE(64n, 32) + runtime.writeUInt16LE(64, 52) + runtime.writeUInt16LE(56, 54) + runtime.writeUInt16LE(2, 56) + + writeProgramHeader(runtime, LOAD_HEADER, { + type: 1, + flags: 5, + offset: 0, + virtualAddress: 0, + size: FIXTURE_BYTES, + memorySize: FIXTURE_BYTES, + alignment: 4096 + }) + writeProgramHeader(runtime, DYNAMIC_HEADER, { + type: 2, + flags: 4, + offset: DYNAMIC_OFFSET, + virtualAddress: DYNAMIC_OFFSET, + size: 32, + memorySize: 32, + alignment: 8 + }) + RUNTIME_SOURCE.copy(runtime, 192) + return runtime +} + +function writeProgramHeader( + runtime, + headerOffset, + { type, flags, offset, virtualAddress, size, memorySize = size, alignment } +) { + runtime.writeUInt32LE(type, headerOffset) + runtime.writeUInt32LE(flags, headerOffset + 4) + runtime.writeBigUInt64LE(BigInt(offset), headerOffset + 8) + runtime.writeBigUInt64LE(BigInt(virtualAddress), headerOffset + 16) + runtime.writeBigUInt64LE(BigInt(offset), headerOffset + 24) + runtime.writeBigUInt64LE(BigInt(size), headerOffset + 32) + runtime.writeBigUInt64LE(BigInt(memorySize), headerOffset + 40) + runtime.writeBigUInt64LE(BigInt(alignment), headerOffset + 48) +} + +async function withFixture(filename, contents, check, { mode = 0o755 } = {}) { + const root = await mkdtemp(join(tmpdir(), 'orca-static-appimage-contract-')) + try { + const path = join(root, filename) + await writeFile(path, contents) + await chmod(path, mode) + await check(path) + } finally { + await rm(root, { recursive: true, force: true }) + } +} diff --git a/config/scripts/verify-cli-bin.mjs b/config/scripts/verify-cli-bin.mjs index a9fa71ce2e7..cdc56401262 100755 --- a/config/scripts/verify-cli-bin.mjs +++ b/config/scripts/verify-cli-bin.mjs @@ -5,15 +5,14 @@ import { chmodSync, mkdirSync, readFileSync, statSync, writeFileSync } from 'nod import path from 'node:path' import { pathToFileURL } from 'node:url' -const OUT_COMMONJS_PACKAGE_JSON = `${JSON.stringify( - { - name: 'orca-compiled-output', - type: 'commonjs', - private: true - }, - null, - 2 -)}\n` +// Electron packaging restamps the channel-specific version after compilation. +function buildOutPackageJson(version) { + return `${JSON.stringify( + { name: 'orca-compiled-output', type: 'commonjs', private: true, version }, + null, + 2 + )}\n` +} /** * Verifies the published CLI entrypoint and the module-type boundary for the @@ -49,7 +48,7 @@ export function verifyPackageCliBin({ const outPackageJsonPath = path.join(projectDir, 'out', 'package.json') if (fixPackageJson) { mkdirSync(path.dirname(outPackageJsonPath), { recursive: true }) - writeFileSync(outPackageJsonPath, OUT_COMMONJS_PACKAGE_JSON, 'utf8') + writeFileSync(outPackageJsonPath, buildOutPackageJson(packageJson.version), 'utf8') } let outPackageJson try { diff --git a/config/scripts/verify-linux-glibc-floor.cjs b/config/scripts/verify-linux-glibc-floor.cjs index 55ec8ca7724..3a138ed894b 100644 --- a/config/scripts/verify-linux-glibc-floor.cjs +++ b/config/scripts/verify-linux-glibc-floor.cjs @@ -164,6 +164,78 @@ function findMissingProviderDeps(importedSymbols, neededLibraries) { return missing } +// ELF e_machine values for the Linux slices we package. Names match electron-builder's Arch enum. +const ELF_MACHINE_BY_ARCH = Object.freeze({ x64: 0x3e, arm64: 0xb7 }) +const ARCH_BY_ELF_MACHINE = Object.freeze({ 0x3e: 'x64', 0xb7: 'arm64' }) + +/** + * ELF `e_machine`, or null when the file is not a readable little-endian ELF. + * + * Why this is checked at all: cross-building an arm64 package on an x64 host can silently pack an + * x86-64 `pty.node` into the arm64 slice — the rebuild logs a forced arm64 rebuild and still ships + * the host's binary. Every other gate here inspects symbol versions, which are perfectly valid on + * the wrong architecture, so nothing noticed. Observed on a Raspberry Pi 5: the app loaded, then + * failed with "Failed to load native module: pty.node". + */ +function readElfMachine(filePath) { + let fd + try { + fd = openSync(filePath, 'r') + const header = Buffer.alloc(20) + if (readSync(fd, header, 0, 20, 0) !== 20) { + return null + } + // EI_DATA (offset 5) must be ELFDATA2LSB for a little-endian e_machine read. + if (header[5] !== 1) { + return null + } + return header.readUInt16LE(18) + } catch { + return null + } finally { + if (fd !== undefined) { + closeSync(fd) + } + } +} + +// Arch tokens that appear in vendored per-architecture package/directory names. +const ARCH_TOKEN_PATTERN = /(?:^|[^a-z0-9])(arm64|aarch64|x64|x86_64)(?:[^a-z0-9]|$)/i +const ARCH_BY_TOKEN = Object.freeze({ arm64: 'arm64', aarch64: 'arm64', x64: 'x64', x86_64: 'x64' }) + +/** + * The architecture a path advertises, or null when it advertises none. + * + * Why this matters: some dependencies ship every architecture and let their loader pick + * (`@parcel/watcher-linux-arm64-glibc/watcher.node` is arm64 on purpose inside an x64 build). Those + * must be judged against the arch their own path declares, not against the slice. + */ +function declaredArchFromPath(filePath) { + const match = ARCH_TOKEN_PATTERN.exec(filePath) + return match ? ARCH_BY_TOKEN[match[1].toLowerCase()] : null +} + +function findArchViolation(filePath, targetArch) { + // A path that names an architecture is judged against that name, so a per-arch vendored package + // is fine while `bin/linux-arm64-.../node-pty.node` holding an x86-64 binary is still caught. + const declared = declaredArchFromPath(filePath) + const expectedArch = declared ?? targetArch + const expected = ELF_MACHINE_BY_ARCH[expectedArch] + if (expected === undefined) { + return null + } + const machine = readElfMachine(filePath) + if (machine === null || machine === expected) { + return null + } + return { + machine, + actual: ARCH_BY_ELF_MACHINE[machine] ?? `0x${machine.toString(16)}`, + expectedArch, + declared: declared !== null + } +} + function isElfFile(filePath) { let fd try { @@ -312,6 +384,7 @@ function readImportedSymbols(filePath, objdumpPath) { */ function verifyLinuxGlibcFloor(rootDir, options = {}) { const binaries = collectNativeBinaries(rootDir) + const targetArch = options.targetArch if (binaries.length === 0) { console.log(`[verify-linux-glibc-floor] OK — no bundled native binaries under ${rootDir}`) return @@ -327,6 +400,28 @@ function verifyLinuxGlibcFloor(rootDir, options = {}) { ) } + // Why before the glibc pass: a wrong-architecture binary's symbol versions are valid but + // meaningless, so reporting a floor violation for it would send the reader down the wrong path. + const archOffenders = binaries + .map((filePath) => ({ filePath, violation: findArchViolation(filePath, targetArch) })) + .filter(({ violation }) => violation !== null) + if (archOffenders.length > 0) { + const detail = archOffenders + .map( + ({ filePath, violation }) => + ` ${relative(rootDir, filePath) || filePath} is ${violation.actual}, expected ` + + `${violation.expectedArch}${violation.declared ? ' (from its own path)' : ''}` + ) + .join('\n') + throw new Error( + `[verify-linux-glibc-floor] ${archOffenders.length} bundled native binar` + + `${archOffenders.length === 1 ? 'y is' : 'ies are'} built for the wrong architecture ` + + `(target ${targetArch}), so the app will fail to load them at runtime:\n${detail}\n` + + 'Cross-building a Linux slice can pack the host architecture despite a forced rebuild; ' + + 'build this slice on a native runner.' + ) + } + const offenders = [] for (const filePath of binaries) { const { versionNeeds, neededLibraries } = readDynamicInfo(filePath, objdumpPath) @@ -375,6 +470,10 @@ function verifyLinuxGlibcFloor(rootDir, options = {}) { module.exports = { MIN_GLIBC, + ELF_MACHINE_BY_ARCH, + readElfMachine, + declaredArchFromPath, + findArchViolation, VERSION_FLOORS, FLOOR_LABEL, RELOCATED_SYMBOL_PROVIDERS, diff --git a/config/scripts/verify-linux-glibc-floor.test.mjs b/config/scripts/verify-linux-glibc-floor.test.mjs index 603e4e85c00..d8d82165053 100644 --- a/config/scripts/verify-linux-glibc-floor.test.mjs +++ b/config/scripts/verify-linux-glibc-floor.test.mjs @@ -6,6 +6,10 @@ import { describe, expect, it } from 'vitest' const require = createRequire(import.meta.url) const { + readElfMachine, + declaredArchFromPath, + findArchViolation, + ELF_MACHINE_BY_ARCH, parseGlibcVersion, compareGlibcVersions, parseVersionNeeds, @@ -321,3 +325,86 @@ describe.skipIf(process.platform === 'win32')('verifyLinuxGlibcFloor', () => { } }) }) + +/** Minimal little-endian 64-bit ELF header with the given e_machine. */ +function elfHeader(machine) { + const header = Buffer.alloc(64) + header.write('\x7fELF', 0, 'latin1') + header[4] = 2 // ELFCLASS64 + header[5] = 1 // ELFDATA2LSB + header[6] = 1 // EV_CURRENT + header.writeUInt16LE(3, 16) // ET_DYN + header.writeUInt16LE(machine, 18) + return header +} + +describe('bundled native binary architecture', () => { + it('reads e_machine from a little-endian ELF', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const file = join(dir, 'pty.node') + await writeFile(file, elfHeader(ELF_MACHINE_BY_ARCH.arm64)) + expect(readElfMachine(file)).toBe(ELF_MACHINE_BY_ARCH.arm64) + await rm(dir, { recursive: true, force: true }) + }) + + // The observed failure: cross-building arm64 on an x64 host packed an x86-64 pty.node, whose + // symbol versions are valid, so every other gate here passed it. + // Real CI hit: @parcel/watcher ships every architecture and its loader picks the match, so the + // arm64 copy is present in an x64 build on purpose. + it('accepts a per-arch vendored package that matches its own path', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const pkg = join(dir, '@parcel', 'watcher-linux-arm64-glibc') + await mkdir(pkg, { recursive: true }) + const file = join(pkg, 'watcher.node') + await writeFile(file, elfHeader(ELF_MACHINE_BY_ARCH.arm64)) + expect(declaredArchFromPath(file)).toBe('arm64') + expect(findArchViolation(file, 'x64')).toBeNull() + await rm(dir, { recursive: true, force: true }) + }) + + // But a path that names an arch must actually hold it — this is the Pi 5 failure. + it('flags a binary that contradicts the architecture its own path names', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const nested = join(dir, 'bin', 'linux-arm64-148') + await mkdir(nested, { recursive: true }) + const file = join(nested, 'node-pty.node') + await writeFile(file, elfHeader(ELF_MACHINE_BY_ARCH.x64)) + expect(findArchViolation(file, 'arm64')).toMatchObject({ actual: 'x64', expectedArch: 'arm64' }) + // Still caught even when the slice being built is x64. + expect(findArchViolation(file, 'x64')).toMatchObject({ actual: 'x64', expectedArch: 'arm64' }) + await rm(dir, { recursive: true, force: true }) + }) + + it('flags an x86-64 binary in an arm64 slice', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const file = join(dir, 'pty.node') + await writeFile(file, elfHeader(ELF_MACHINE_BY_ARCH.x64)) + expect(findArchViolation(file, 'arm64')).toMatchObject({ actual: 'x64' }) + await rm(dir, { recursive: true, force: true }) + }) + + it('accepts a matching architecture', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const file = join(dir, 'pty.node') + await writeFile(file, elfHeader(ELF_MACHINE_BY_ARCH.x64)) + expect(findArchViolation(file, 'x64')).toBeNull() + await rm(dir, { recursive: true, force: true }) + }) + + it('stays silent when no target architecture is supplied', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const file = join(dir, 'pty.node') + await writeFile(file, elfHeader(ELF_MACHINE_BY_ARCH.x64)) + expect(findArchViolation(file, undefined)).toBeNull() + await rm(dir, { recursive: true, force: true }) + }) + + it('ignores a file that is not a readable little-endian ELF', async () => { + const dir = await mkdtemp(join(tmpdir(), 'orca-elf-arch-')) + const file = join(dir, 'not-elf.node') + await writeFile(file, Buffer.from('not an elf at all')) + expect(readElfMachine(file)).toBeNull() + expect(findArchViolation(file, 'arm64')).toBeNull() + await rm(dir, { recursive: true, force: true }) + }) +}) diff --git a/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs b/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs index 0c192382d7a..a8c2cb3f4e7 100644 --- a/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs +++ b/config/scripts/windows-cmd-shim-spawn-boundary.test.mjs @@ -47,8 +47,10 @@ const WINDOWS_SHIM_SPAWN_ALLOWLIST = [ 'config/scripts/ensure-native-runtime.test.mjs', 'config/scripts/live-remote-freeze-rpc.mjs', 'config/scripts/remote-agent-session-authority-repro.mjs', - // macOS-only build path; the win32 branch is dead code there. + // Platform-local build paths; the win32 branch is dead code on both. 'config/scripts/build-mac-local.mjs', + 'config/scripts/build-linux-local.mjs', + 'config/scripts/build-linux-local.test.mjs', // Benchmarks, repros and e2e drivers — developer-invoked or Linux-only in CI. 'config/scripts/build-orcad-prebuilds.mjs', 'config/scripts/run-ai-vault-typing-bench.mjs', diff --git a/config/tsconfig.cli.json b/config/tsconfig.cli.json index 93d556602f8..1b9600188f2 100644 --- a/config/tsconfig.cli.json +++ b/config/tsconfig.cli.json @@ -127,6 +127,8 @@ // Why: serve-electron-flag-parity.test.ts checks the Electron-side serve argv rewrite against this // project's serve spec; the module has no imports, so listing it pulls in nothing else. "../src/main/startup/serve-mode-argv.ts", + // The parity test keeps this import-free list aligned with COMMAND_SPECS. + "../src/main/startup/cli-command-names.ts", "../src/main/runtime/runtime-metadata.ts", "../src/main/sqlite/sync-database.ts", "../src/main/win32-utils.ts" diff --git a/docs/reference/headless-linux-server.md b/docs/reference/headless-linux-server.md index 3d7db834e8e..7368678c2e4 100644 --- a/docs/reference/headless-linux-server.md +++ b/docs/reference/headless-linux-server.md @@ -8,7 +8,11 @@ Linux, the packaged AppImage still needs the libraries that Electron expects at startup. Current Orca builds start Xvfb automatically for `orca serve` when no `DISPLAY` is set, but Xvfb must be installed first. A separate D-Bus session is not required. When `DISPLAY` is set, Orca uses that display instead of starting -a competing Xvfb process. +a competing Xvfb process, provided the display is usable: its socket must exist, +and if an X lock file is present it must name a running process. A `DISPLAY` +whose lock names a dead process is refused rather than replaced, and `orca serve` +exits — unset `DISPLAY` to let Orca start its own Xvfb. A socket published with +no lock at all (a container bind-mounting `/tmp/.X11-unix`, or WSLg) is accepted. The supported deployment matrix covers Ubuntu 20.04, 22.04, and 24.04 and current Debian stable — anything with glibc 2.31 or newer (see @@ -229,6 +233,10 @@ clients should use. `KillMode=mixed` sends the graceful stop signal only to Orca's main process, then retains systemd's cgroup-wide `SIGKILL` fallback if shutdown times out. This lets Orca keep its owned Xvfb alive until Electron disconnects cleanly. +It does **not** preserve the detached terminal daemon: the daemon and its PTYs +remain in `orca-serve.service`'s cgroup and are killed when the stop completes. +Every `systemctl stop` or `restart` therefore ends live terminals and agent +processes, even though their persisted layout and terminal history remain. Exit status `3` means another process already owns this userData profile, so `RestartPreventExitStatus=3` stops the unit instead of retrying a launch that @@ -324,6 +332,14 @@ sudo systemctl enable --now orca-xvfb.service orca-serve.service ## CLI Install Note +The registered Linux CLI command is `orca-ide`, not `orca`, to avoid shadowing +the GNOME Orca screen reader. Desktop-managed terminals receive a +terminal-scoped bare-`orca` shim. A packaged headless `orca serve` also makes a +best-effort dispatcher at `$HOME/.local/bin/orca` for the service user's own +shell, so the Claude Teams launcher can resolve its bare command; it does not +replace another user's `orca`. From an ordinary shell outside that service +user's managed environment, substitute `orca-ide` for `orca` in commands below. + On a headless host, you do not need to open the desktop UI just to run the server. Invoke the AppImage directly: @@ -376,7 +392,7 @@ at all — the built-in updater only runs in the desktop GUI, and no paired mobi or web client can trigger it remotely. Upgrading is always a deliberate step: replace the AppImage and restart the service. -Two facts make this safe and predictable: +Two facts make the persisted-state transition predictable: - **State lives in the service user's home, not next to the binary.** Persisted data is under `/home/orca/.config/` (Orca uses both an `orca` and an `Orca` @@ -388,15 +404,37 @@ Two facts make this safe and predictable: state into the current schema and writes it back in the current shape, so a forward upgrade needs no manual data step. +These guarantees do not preserve live processes. The service restart kills +every terminal and agent in its cgroup; an agent conversation may be resumable, +but its current process and any in-flight command are gone. + +Immediately before stopping the service, obtain a fresh census as the service's +OS account and home. Use the installer's absolute launcher path so `sudo`'s +`secure_path` cannot hide a per-user registration: +`sudo -Hu orca /home/orca/.local/bin/orca-ide terminal list --json`. +Replace both `orca` and `/home/orca` with the service account and home used by +your unit; for an extracted deployment, use its absolute `resources/bin/orca-ide` +launcher instead. Proceed only when the result is +untruncated, has an explicit `hostScope`, covers every execution host affected +by this service stop, and lists no terminals on those hosts. Every +`omittedHostIds` entry must be explicitly accounted for outside this service's +execution boundary. A separately paired runtime is outside that boundary; local +execution and SSH hosts reached through this runtime are not. An affected or +unknown omission, missing scope, failed request or lost connection is +`unverifiable`, so defer the restart. Do not allow new work between that census +and the stop; Orca does not yet provide an atomic census-and-stop fence. + Rolling back is the case that needs care — see [Roll back](#roll-back). ### Record the version you deploy -Orca has no headless version command: there is no `--version` flag or `version` -subcommand, and `orca serve` prints only its endpoint. Choose a release tag -explicitly instead of following the `latest` URL, and record it next to the -binary so upgrades are auditable. The steps below keep that record in -`/opt/orca/VERSION`. +The bundled CLI launcher prints the Orca build with `orca-ide --version`. For an +extracted deployment, that launcher is +`squashfs-root/resources/bin/orca-ide`; deb/rpm installs and CLI registration put +it on `PATH`. Do not use `orca-linux.AppImage --version` for this audit because +Electron owns the direct binary's version flags and may report its own runtime +version. For an AppImage service, choose a release tag explicitly and record it +next to the binary. The steps below keep that record in `/opt/orca/VERSION`. ### Upgrade steps @@ -877,8 +915,8 @@ refuse to run there and print the command to run on the machine you want. - `dlopen(): error loading libfuse.so.2`: install `libfuse2`. - `Missing X server or $DISPLAY`: install `xvfb`, or start the managed Xvfb service and set `DISPLAY=:99`. -- `Xvfb not found`: confirm `command -v Xvfb` and use that absolute path in the - systemd unit. +- `[serve] Xvfb failed to start` or `[serve] Could not start Xvfb`: confirm + `command -v Xvfb` and that it is on the service `PATH`. - GPU or DRI warnings on a VPS: keep `LIBGL_ALWAYS_SOFTWARE=1` in the service environment. - Chromium sandbox errors: confirm the service is running as the non-root diff --git a/docs/reference/linux-glibc-compatibility.md b/docs/reference/linux-glibc-compatibility.md index a11506235fe..20e9b38acb5 100644 --- a/docs/reference/linux-glibc-compatibility.md +++ b/docs/reference/linux-glibc-compatibility.md @@ -6,6 +6,14 @@ Packaging enforces this floor automatically; keep it in mind when adding or upgrading native dependencies. (The optional speech feature is the one exception — see below.) +## Local package build prerequisites + +`pnpm run build:linux` produces AppImage, deb, and RPM artifacts. The RPM target +requires `rpmbuild` on `PATH`; install `rpm` on Ubuntu/Debian, `rpm-build` on +Fedora/RHEL, or `rpm` through Homebrew on macOS, then verify it with +`rpmbuild --version` before packaging. Cross-host builds have the same +requirement. + ## Why this needs attention A native module (`.node`) links against the glibc of the machine that compiled diff --git a/docs/reference/orcad-operations.md b/docs/reference/orcad-operations.md index bbde9829514..2901a5bf0b6 100644 --- a/docs/reference/orcad-operations.md +++ b/docs/reference/orcad-operations.md @@ -4,8 +4,6 @@ whatever supervises it: what it binds, what it owns on disk, who restarts what, and what its readiness payload actually proves. -Design background: `docs/design/shipping-orcad.html` §00c and §04. - ## Two long-lived processes, not one A deployment is **orcad** plus **the terminal daemon**. @@ -14,18 +12,22 @@ A deployment is **orcad** plus **the terminal daemon**. | ---------- | -------------------------------- | ------------------------------------- | | Started by | the supervisor | orcad, detached | | Owns | RPC, git, worktrees, persistence | every local PTY | -| Lifetime | one supervised run | **outlives orcad** | +| Lifetime | one supervised run | detached from orcad, not its service | | Endpoint | `ws://:` | `/daemon/daemon-v.sock` | -The daemon outliving orcad is the property the whole peer model is recommended for -(`docs/reference/ssh-execution-boundary.md`): daemon-backed PTYs stay `live` across a runtime -restart, so a restart, an update or a rollback does not destroy running work. Everything -below exists to keep that true. +orcad detaches the daemon and calls `disconnectDaemon()`, never `shutdownDaemon()`. The +built-in remote deployment path stops only the recorded orcad PID, so the daemon and its PTYs +survive. The successor adopts the current endpoint and routes supported previous protocol +versions through legacy adapters. This makes a PID-scoped update, rollback or restart +non-destructive to live work. -**Consequence for supervision:** orcad's shutdown path calls `disconnectDaemon()`, never -`shutdownDaemon()`. A supervisor that reaps orcad's whole process group — systemd's -`KillMode=control-group` — kills the daemon too and turns every restart back into data loss. -Use `KillMode=mixed` (the default) or `process`, and never `--send-sigkill` on the group. +Process detachment is not service isolation. A daemon forked by orcad, and every PTY it owns, +remain in the same systemd service cgroup. `KillMode=mixed` does **not** preserve them: it +sends the graceful stop signal only to the main process, then sends `SIGKILL` to every process +remaining in the cgroup when the stop timeout expires. `KillMode=control-group` is destructive +too. `KillMode=process` leaves service-owned processes unmanaged and is not a supported +preservation mechanism. Service-restart survival requires separately supervised cgroups; the +current deployment does not provide them. ## Bind policy @@ -76,6 +78,25 @@ a live daemon makes worthwhile. ## Supervision +### Process-scoped and cgroup-wide stops + +The built-in remote updater performs a PID-scoped stop and keeps the daemon's install version +pinned while it owns sessions. A combined-unit systemd stop or restart is different: it reaps +the daemon and every live terminal after the graceful window. + +Before a cgroup-wide stop, obtain a fresh `orca-ide terminal list --json` result using the same OS +account and home as the daemon. Invoke the installer's absolute launcher path so `sudo`'s +`secure_path` cannot hide a per-user registration (for example, +`sudo -Hu orca /home/orca/.local/bin/orca-ide terminal list --json`). Replace both `orca` and +`/home/orca` with the service account and home used by the unit; an extracted deployment may use +its absolute `resources/bin/orca-ide` launcher instead. A safe empty census is untruncated, has an explicit `hostScope`, covers every +execution host affected by the stop, and lists no terminals on those hosts. Every +`omittedHostIds` entry must be explicitly accounted for outside the target service's execution +boundary. A separately paired runtime is outside that boundary; local execution and SSH hosts +reached through this runtime are not. An affected or unknown omission, missing scope, +truncation, a failed request or lost contact makes the result `unverifiable`: defer the stop. Do +not admit new work after the census. Orca does not yet provide an atomic census-and-stop fence. + ### Who supervises orcad An external supervisor (systemd, launchd, a process manager). orcad conforms to it: @@ -127,11 +148,11 @@ An external supervisor (systemd, launchd, a process manager). orcad conforms to ### Decommissioning -The daemon outliving orcad is deliberate, so stopping orcad does **not** leave the host with -zero Orca processes. A daemon that has been adopted stays resident after its runtime -disconnects — that is what makes the next start a reattach rather than a cold restore. To -retire a host completely, stop orcad and then stop the daemon named by -`health.terminalDaemon.pid`, or delete the data root and let the endpoint go stale. +After a PID-scoped stop, an adopted daemon stays resident so the next orcad can reattach. +A combined-unit systemd stop kills it instead. To retire a process-scoped deployment, apply +the census rule above, stop orcad, then stop the daemon named by `health.terminalDaemon.pid`. +Only report it `exited` after verification on the execution host; loss of contact is +`unverifiable`. ## Health @@ -145,7 +166,8 @@ nodeVersion / nodeAbi process.versions.node / .modules — the ABI native add platform / arch / pid terminalDaemon: state live | degraded | absent - ownsFreshSessions whether NEW terminals are daemon-owned, i.e. survive an orcad restart + ownsFreshSessions whether NEW terminals are daemon-owned; this supports PID-scoped + restart recovery, not supervisor or service-cgroup isolation pid the live daemon's pid, from its own PID record buildVersion the build the LIVE daemon was forked from (may legitimately predate this orcad after an update — reporting orcad's version for both would @@ -179,11 +201,13 @@ Named here so nothing reads as implemented that is not: - **A continuous health endpoint.** `health` is published once, in the readiness payload. A supervisor's periodic liveness/readiness probe needs an HTTP or RPC surface over the same `collectOrcadHealth()`; that surface does not exist yet. -- **libc slot.** §04 asks for it in the health payload. It belongs to the native strategy - (plan item 5), which owns libc detection; there is no honest value to publish until then. -- **`degradations[]`.** Plan item 2's contract, not this one. +- **Systemd-isolated daemon supervision.** orcad and its daemon currently share one service + cgroup, so a combined-unit stop cannot preserve live terminals. +- **libc slot.** There is no honest health value to publish until native libc detection owns + it. +- **`degradations[]`.** The readiness contract does not publish this collection yet. - **Credential administration** (list / revoke / rotate devices, expiring pending offers, - structured security logging) — §04, not delivered here. + structured security logging). - **Pinned-port fail-closed.** A pinned `--port` still falls back to an OS-assigned port on conflict. - **Reconciling `webClientUrl` with reachability** under the loopback default. diff --git a/docs/reference/ssh-execution-boundary.md b/docs/reference/ssh-execution-boundary.md index d85dadacccd..45b307411f6 100644 --- a/docs/reference/ssh-execution-boundary.md +++ b/docs/reference/ssh-execution-boundary.md @@ -82,4 +82,4 @@ A listing is only evidence about the hosts it actually covered. When a result do An SSH host and a paired runtime (`orca environment`) imply opposite boundaries: the first is a dumb execution host driven by your client, the second is a peer that owns its own control plane. Registering the same machine both ways splits its worktrees across two identities, makes `terminal list` return different sets depending on `--environment`, and reliably confuses both humans and agents. Pick one per machine. -For work that must continue while you are offline, use the peer/headless-runtime model on the remote host instead of the direct-SSH model. Its control plane is host-local, and its daemon-backed PTYs stay `live` across a normal runtime restart so the runtime can reattach; an explicit daemon shutdown can still make them `exited`. Do not register the same machine through both models. A detached agent process outside Orca can also survive a control-plane outage, but it has no stdin, so its instructions cannot be amended mid-run. +For work that must continue while you are offline, use the peer/headless-runtime model on the remote host instead of the direct-SSH model. Its control plane is host-local, and its daemon-backed PTYs can stay `live` across a PID-scoped runtime restart so the runtime can reattach. A service manager that reaps the runtime's cgroup, or an explicit daemon shutdown, makes them `exited`; see [Running orcad](./orcad-operations.md#process-scoped-and-cgroup-wide-stops). Do not register the same machine through both models. A detached agent process outside Orca can also survive a control-plane outage, but it has no stdin, so its instructions cannot be amended mid-run. diff --git a/package.json b/package.json index aab4c660c48..9846604fab6 100644 --- a/package.json +++ b/package.json @@ -75,6 +75,7 @@ "verify:localization-coverage": "node config/scripts/audit-localization-coverage.mjs --check", "audit:localization": "node config/scripts/audit-localization-coverage.mjs", "build:cli": "tsc -p config/tsconfig.cli.json --outDir out --composite false --incremental false && node config/scripts/verify-cli-bin.mjs --fix-executable --fix-package-json && node config/scripts/install-dev-cli.mjs", + "test:linux-cli-contract": "node config/scripts/run-linux-cli-launch-contract-docker.mjs --appimage dist/orca-linux.AppImage", "test:repro:skills-cli-runtime": "pnpm run build:cli && pnpm run build:electron-vite && pnpm run verify:built-skills-cli", "build:electron-vite": "node config/scripts/run-electron-vite-build.mjs", "build:electron-vite:parallel": "node config/scripts/run-electron-vite-targets-in-parallel.mjs", @@ -94,7 +95,7 @@ "build:icons": "bash resources/icon-source/generate.sh", "build:mac": "pnpm run build:desktop && pnpm run build:computer-macos && pnpm run build:keyboard-layout-macos && pnpm run build:notification-status-macos && pnpm run ensure:electron-runtime && node config/scripts/build-mac-local.mjs", "build:mac:release": "node config/scripts/verify-macos-release-env.mjs && ORCA_MAC_RELEASE=1 pnpm run build:desktop && ORCA_MAC_RELEASE=1 pnpm run build:computer-macos && ORCA_MAC_RELEASE=1 pnpm run build:keyboard-layout-macos && ORCA_MAC_RELEASE=1 pnpm run build:notification-status-macos && pnpm run ensure:electron-runtime && ORCA_MAC_RELEASE=1 electron-builder --config config/electron-builder.config.cjs --mac", - "build:linux": "pnpm run build:desktop && pnpm run ensure:electron-runtime && electron-builder --config config/electron-builder.config.cjs --linux AppImage deb", + "build:linux": "pnpm run build:desktop && pnpm run ensure:electron-runtime && node config/scripts/build-linux-local.mjs", "test:e2e": "pnpm run ensure:electron-runtime && npx playwright test --config tests/playwright.config.ts --project=electron-headless", "test:e2e:workspace-session-golden": "pnpm run ensure:electron-runtime && npx playwright test tests/e2e/golden-quit-relaunch-session.spec.ts tests/e2e/golden-terminal-file-link.spec.ts tests/e2e/golden-worktree-create-switch.spec.ts --config tests/playwright.config.ts --project electron-headless --workers=1", "test:e2e:multi-client-navigation": "node config/scripts/run-multi-client-navigation-e2e.mjs", diff --git a/resources/linux/bin/orca-ide b/resources/linux/bin/orca-ide index 88b04e9e47d..f191f27c770 100755 --- a/resources/linux/bin/orca-ide +++ b/resources/linux/bin/orca-ide @@ -38,4 +38,5 @@ export ORCA_NODE_REPL_EXTERNAL_MODULE="${NODE_REPL_EXTERNAL_MODULE-}" unset NODE_OPTIONS unset NODE_REPL_EXTERNAL_MODULE +# CLI commands run in Electron's Node mode and must never initialize Chromium. ELECTRON_RUN_AS_NODE=1 exec "$ELECTRON" "$CLI" "$@" diff --git a/resources/linux/packaging/after-remove.sh b/resources/linux/packaging/after-remove.sh index 0f497024613..a23426df446 100755 --- a/resources/linux/packaging/after-remove.sh +++ b/resources/linux/packaging/after-remove.sh @@ -4,6 +4,12 @@ # /usr/bin/orca-ide a user or other package may own. set -e +# RPM passes an instance count; dpkg passes the package lifecycle action. +case "${1-}" in + 0 | remove | purge) ;; + *) exit 0 ;; +esac + link="/usr/bin/orca-ide" if [ -L "$link" ]; then diff --git a/src/cli/args.test.ts b/src/cli/args.test.ts index eeedfbe0428..1ac86d99e12 100644 --- a/src/cli/args.test.ts +++ b/src/cli/args.test.ts @@ -93,6 +93,16 @@ describe('parseArgs', () => { expect(parsed.flags.get('repo')).toBe('id:abc') }) + it('preserves a project selector before the project command', () => { + const parsed = parseArgs( + ['--project', 'github:stablyai/orca', 'project', 'setups'], + [['project', 'setups']] + ) + + expect(parsed.commandPath).toEqual(['project', 'setups']) + expect(parsed.flags.get('project')).toBe('github:stablyai/orca') + }) + it('preserves a selector value that is also a registered command', () => { const parsed = parseArgs( ['--environment', 'status', 'worktree', 'list'], @@ -113,6 +123,16 @@ describe('parseArgs', () => { expect(parsed.flags.get('environment')).toBe('worktree') }) + it.each([ + ['--project', 'project', 'project', 'setups'], + ['--project=project', 'project', 'setups'] + ])('preserves a command-named project selector in %j', (...args) => { + const parsed = parseArgs(args, [['project', 'setups']]) + + expect(parsed.commandPath).toEqual(['project', 'setups']) + expect(parsed.flags.get('project')).toBe('project') + }) + it('parses emulator reinstall as a boolean flag', () => { const parsed = parseArgs(['emulator', 'install', 'app.apk', '--reinstall', '--device', 'emu']) diff --git a/src/cli/args.ts b/src/cli/args.ts index c4c373a814f..a934915655e 100644 --- a/src/cli/args.ts +++ b/src/cli/args.ts @@ -1,6 +1,12 @@ import { RuntimeClientError } from './runtime/types' import { unknownCommandData, unknownFlagData } from './command-suggestion' import { specPaths, type CommandSpec } from './command-spec' +import { + CLI_BOOLEAN_FLAGS, + CLI_GLOBAL_FLAGS, + CLI_GLOBAL_VALUE_FLAGS, + findCliCommandIndex +} from '../shared/cli-argument-boundary' export { specPaths } export type { CommandSpec } @@ -11,51 +17,9 @@ export type ParsedArgs = { positionalFlagConflicts?: string[] } -export const GLOBAL_FLAGS = ['help', 'json', 'pairing-code', 'environment'] -const GLOBAL_VALUE_FLAGS = new Set(['pairing-code', 'environment']) -export const BOOLEAN_FLAGS = new Set([ - 'all', - 'attachments', - 'children', - 'comments', - 'connect', - 'current', - 'dry-run', - 'enter', - 'focus', - 'force', - 'full', - 'help', - 'inject', - 'include-archived', - 'include-visual-layouts', - 'interrupt', - 'json', - 'local', - 'messages', - 'me', - 'mobile', - 'mobile-pairing', - 'no-pairing', - 'screen', - 'parent-current', - 'provision', - 'ready', - 'recipe-json', - 'relations', - 'reinstall', - 'restore-window', - 'return-preamble', - 'run-hooks', - 'show-profile', - 'staged', - 'tab', - 'tasks', - 'text-stdin', - 'unread', - 'value-stdin', - 'wait' -]) +export const GLOBAL_FLAGS = CLI_GLOBAL_FLAGS +const GLOBAL_VALUE_FLAGS = new Set(CLI_GLOBAL_VALUE_FLAGS) +export const BOOLEAN_FLAGS = CLI_BOOLEAN_FLAGS export const REPEATED_FLAG_SEPARATOR = '\u0000' const REPEATABLE_STRING_FLAGS = new Set(['label', 'skill']) @@ -69,25 +33,10 @@ function setFlagValue(flags: Map, name: string, value: flags.set(name, value) } -function commandPathStartsAt(argv: string[], tokenIndex: number, path: string[]): boolean { - let cursor = tokenIndex - for (const part of path) { - while (argv[cursor]?.startsWith('--')) { - const assignment = argv[cursor].slice(2) - const flag = assignment.split('=', 1)[0] - cursor += assignment.includes('=') || BOOLEAN_FLAGS.has(flag) ? 1 : 2 - } - if (argv[cursor] !== part) { - return false - } - cursor += 1 - } - return true -} - export function parseArgs(argv: string[], commandPaths?: readonly string[][]): ParsedArgs { const commandPath: string[] = [] const flags = new Map() + const commandIndex = findCliCommandIndex(argv, commandPaths ?? []) for (let i = 0; i < argv.length; i += 1) { const token = argv[i] @@ -112,9 +61,7 @@ export function parseArgs(argv: string[], commandPaths?: readonly string[][]): P continue } // Why: a pre-command flag must not consume a registry-resolvable command path. - const startsCommandAt = (tokenIndex: number): boolean => - commandPaths?.some((path) => commandPathStartsAt(argv, tokenIndex, path)) ?? false - if (commandPath.length === 0 && startsCommandAt(i + 1) && !startsCommandAt(i + 2)) { + if (commandPath.length === 0 && i + 1 === commandIndex) { flags.set(flag, true) continue } diff --git a/src/cli/cli-command-name-parity.test.ts b/src/cli/cli-command-name-parity.test.ts new file mode 100644 index 00000000000..32bbcc06add --- /dev/null +++ b/src/cli/cli-command-name-parity.test.ts @@ -0,0 +1,25 @@ +import { describe, expect, it } from 'vitest' +import { CLI_COMMAND_NAMES } from '../main/startup/cli-command-names' +import { COMMAND_SPECS } from './specs' + +const specCommandNames = [...new Set(COMMAND_SPECS.map((spec) => spec.path[0]))].sort() + +describe('CLI command-name parity between COMMAND_SPECS and the launch redirect', () => { + it('has commands to compare', () => { + expect(specCommandNames.length).toBeGreaterThan(0) + }) + + it('redirects every top-level CLI command', () => { + const redirected = new Set(CLI_COMMAND_NAMES) + expect(specCommandNames.filter((name) => !redirected.has(name))).toEqual([]) + }) + + it('lists no command that COMMAND_SPECS does not define', () => { + const specNames = new Set(specCommandNames) + expect([...CLI_COMMAND_NAMES].filter((name) => !specNames.has(name))).toEqual([]) + }) + + it('stays sorted and free of duplicates so additions are easy to review', () => { + expect([...CLI_COMMAND_NAMES]).toEqual([...new Set(CLI_COMMAND_NAMES)].sort()) + }) +}) diff --git a/src/cli/cli-version.test.ts b/src/cli/cli-version.test.ts new file mode 100644 index 00000000000..6ffe60692fb --- /dev/null +++ b/src/cli/cli-version.test.ts @@ -0,0 +1,36 @@ +import { mkdtemp, mkdir, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { readOrcaCliVersion } from './cli-version' + +const temporaryDirectories: string[] = [] + +afterEach(() => + Promise.all(temporaryDirectories.splice(0).map((path) => rm(path, { recursive: true }))) +) + +describe('CLI version', () => { + it('reads the package boundary beside the compiled CLI', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-cli-version-')) + const runtimeDir = join(root, 'cli') + temporaryDirectories.push(root) + await mkdir(runtimeDir) + await writeFile(join(root, 'package.json'), JSON.stringify({ version: '1.4.178-rc.2' })) + + expect(readOrcaCliVersion(runtimeDir)).toBe('1.4.178-rc.2') + }) + + it('rejects missing, malformed, and non-string versions', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-cli-version-invalid-')) + const runtimeDir = join(root, 'cli') + temporaryDirectories.push(root) + await mkdir(runtimeDir) + + expect(readOrcaCliVersion(runtimeDir)).toBeNull() + await writeFile(join(root, 'package.json'), '{') + expect(readOrcaCliVersion(runtimeDir)).toBeNull() + await writeFile(join(root, 'package.json'), JSON.stringify({ version: 178 })) + expect(readOrcaCliVersion(runtimeDir)).toBeNull() + }) +}) diff --git a/src/cli/cli-version.ts b/src/cli/cli-version.ts new file mode 100644 index 00000000000..0918ec763fd --- /dev/null +++ b/src/cli/cli-version.ts @@ -0,0 +1,14 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' + +// Node-mode CLI code cannot read the package metadata inside app.asar. +export function readOrcaCliVersion(runtimeDir = __dirname): string | null { + try { + const parsed = JSON.parse(readFileSync(join(runtimeDir, '..', 'package.json'), 'utf8')) as { + version?: unknown + } + return typeof parsed.version === 'string' && parsed.version.length > 0 ? parsed.version : null + } catch { + return null + } +} diff --git a/src/cli/command-suggestion.ts b/src/cli/command-suggestion.ts index db481de6ea8..6bc6d6eee0b 100644 --- a/src/cli/command-suggestion.ts +++ b/src/cli/command-suggestion.ts @@ -1,4 +1,7 @@ import { specPaths, type CommandSpec } from './command-spec' +import { levenshtein } from '../shared/edit-distance' + +export { levenshtein } from '../shared/edit-distance' // Why: rank the live registry so typo recovery cannot drift from accepted paths. @@ -46,30 +49,6 @@ export type CommandErrorData = { nextSteps: string[] } -export function levenshtein(a: string, b: string): number { - const m = a.length - const n = b.length - if (m === 0) { - return n - } - if (n === 0) { - return m - } - let prev = Array.from({ length: n + 1 }, (_, index) => index) - let curr = Array.from({ length: n + 1 }, () => 0) - for (let i = 1; i <= m; i += 1) { - curr[0] = i - for (let j = 1; j <= n; j += 1) { - const cost = a[i - 1] === b[j - 1] ? 0 : 1 - curr[j] = Math.min(prev[j] + 1, curr[j - 1] + 1, prev[j - 1] + cost) - } - const swap = prev - prev = curr - curr = swap - } - return prev[n] -} - // Why: one bounded near-match ranking keeps command and flag recovery consistent. function rankByDistance(scored: { label: string; distance: number }[]): string[] { return scored diff --git a/src/cli/handlers/core.ts b/src/cli/handlers/core.ts index 145540bb627..d4979ff2ae9 100644 --- a/src/cli/handlers/core.ts +++ b/src/cli/handlers/core.ts @@ -3,6 +3,7 @@ import type { CommandHandler } from '../dispatch' import { formatCliStatus, formatStatus, printResult } from '../format' import { RuntimeClientError, serveOrcaApp } from '../runtime-client' import { stripElectronRunAsNode } from '../runtime/launch' +import { getServeOptionValidationError } from '../../shared/serve-option-validation' function envRecord(): Record { // Why: the `orca` launcher runs Orca's Electron binary as Node, so this CLI @@ -92,43 +93,29 @@ export const CORE_HANDLERS: Record = { printResult(result, json, formatCliStatus) }, serve: async ({ flags, json }) => { - if (flags.get('no-pairing') === true && flags.get('mobile-pairing') === true) { - throw new RuntimeClientError( - 'invalid_argument', - 'Use either --mobile-pairing or --no-pairing, not both.' - ) - } - if (flags.get('recipe-json') === true && flags.get('no-pairing') === true) { - throw new RuntimeClientError( - 'invalid_argument', - 'Recipe JSON output requires runtime pairing; remove --no-pairing.' - ) - } - if (flags.get('recipe-json') === true && flags.get('mobile-pairing') === true) { - throw new RuntimeClientError( - 'invalid_argument', - 'Recipe JSON output requires runtime pairing; remove --mobile-pairing.' - ) - } - const projectRoot = - typeof flags.get('project-root') === 'string' ? (flags.get('project-root') as string) : null - if (flags.get('recipe-json') === true && !projectRoot) { - throw new RuntimeClientError( - 'invalid_argument', - 'Recipe JSON output requires --project-root.' - ) + const projectRootValue = flags.get('project-root') + const projectRoot = typeof projectRootValue === 'string' ? projectRootValue : null + const noPairing = flags.get('no-pairing') === true + const mobilePairing = flags.get('mobile-pairing') === true + const recipeJson = flags.get('recipe-json') === true + const validationError = getServeOptionValidationError({ + noPairing, + mobilePairing, + recipeJson, + projectRoot + }) + if (validationError) { + throw new RuntimeClientError('invalid_argument', validationError) } const port = getOptionalServePort(flags) + const pairingAddressValue = flags.get('pairing-address') const exitCode = await serveOrcaApp({ json, port, - pairingAddress: - typeof flags.get('pairing-address') === 'string' - ? (flags.get('pairing-address') as string) - : null, - noPairing: flags.get('no-pairing') === true, - mobilePairing: flags.get('mobile-pairing') === true, - recipeJson: flags.get('recipe-json') === true, + pairingAddress: typeof pairingAddressValue === 'string' ? pairingAddressValue : null, + noPairing, + mobilePairing, + recipeJson, projectRoot }) process.exitCode = exitCode diff --git a/src/cli/index.ts b/src/cli/index.ts index 2a8921a2cdb..c29bfff7060 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -8,6 +8,7 @@ import { specPaths, validateCommandAndFlags } from './args' +import { readOrcaCliVersion } from './cli-version' import { dispatch } from './dispatch' import { assertEnvironmentSelectorResolvable, @@ -59,6 +60,17 @@ export async function main( argv = process.argv.slice(2), cwd = resolveInvocationCwd() ): Promise { + // Why: version audits use the bundled launcher; Electron intercepts direct binary version flags. + if (argv.length === 1 && (argv[0] === '--version' || argv[0] === '-v')) { + const version = readOrcaCliVersion() + if (!version) { + process.stderr.write('Could not determine the Orca version for this build.\n') + process.exitCode = 1 + return + } + process.stdout.write(`${version}\n`) + return + } if (argv[0] === 'agent-teams-tmux') { await runAgentTeamsTmuxShim(argv.slice(1)) return diff --git a/src/cli/runtime/launch.test.ts b/src/cli/runtime/launch.test.ts index 235b2b2ed56..7931e489e3d 100644 --- a/src/cli/runtime/launch.test.ts +++ b/src/cli/runtime/launch.test.ts @@ -13,12 +13,14 @@ import { SERVE_REPLACEMENT_READY_TIMEOUT_MS } from './serve-update-supervisor' -const { spawnMock } = vi.hoisted(() => ({ - spawnMock: vi.fn() +const { spawnMock, spawnSyncMock } = vi.hoisted(() => ({ + spawnMock: vi.fn(), + spawnSyncMock: vi.fn() })) vi.mock('child_process', () => ({ - spawn: spawnMock + spawn: spawnMock, + spawnSync: spawnSyncMock })) import { launchOrcaApp, serveOrcaApp } from './launch' @@ -86,6 +88,7 @@ describe('serveOrcaApp', () => { beforeEach(() => { spawnMock.mockReset() + spawnSyncMock.mockReset() process.env.ORCA_APP_EXECUTABLE = '/Applications/Orca.app/Contents/MacOS/Orca' }) @@ -93,7 +96,6 @@ describe('serveOrcaApp', () => { vi.restoreAllMocks() delete process.env.ORCA_APP_EXECUTABLE delete process.env.ORCA_APP_EXECUTABLE_NEEDS_APP_ROOT - delete process.env.ORCA_APPIMAGE_NO_SANDBOX delete process.env.ORCA_USER_DATA_PATH return Promise.all( temporaryDirectories.splice(0).map((directory) => rm(directory, { recursive: true })) @@ -391,32 +393,6 @@ describe('serveOrcaApp', () => { ) }) - it('preserves an AppImage no-sandbox launch for the server child', async () => { - process.env.ORCA_APPIMAGE_NO_SANDBOX = '1' - const child = { - kill: vi.fn(), - once: vi.fn( - (event: string, handler: (code: number | null, signal: string | null) => void) => { - if (event === 'exit') { - queueMicrotask(() => handler(0, null)) - } - return child - } - ) - } - spawnMock.mockReturnValue(child) - - await expect(serveOrcaApp({ json: true })).resolves.toBe(0) - - expect(spawnMock).toHaveBeenCalledWith( - '/Applications/Orca.app/Contents/MacOS/Orca', - ['--no-sandbox', '--serve', '--serve-json'], - expect.any(Object) - ) - const spawnOptions = spawnMock.mock.calls[0]?.[2] as { env?: NodeJS.ProcessEnv } - expect(spawnOptions.env).not.toHaveProperty('ORCA_APPIMAGE_NO_SANDBOX') - }) - it('passes the app root before serve flags for dev Electron executables', async () => { process.env.ORCA_APP_EXECUTABLE = '/repo/node_modules/.bin/electron' process.env.ORCA_APP_EXECUTABLE_NEEDS_APP_ROOT = '1' @@ -444,6 +420,66 @@ describe('serveOrcaApp', () => { ) }) + it.each([ + { probe: 'exits nonzero', result: { status: 1 }, expectedPrefix: ['--no-sandbox'] }, + { probe: 'succeeds', result: { status: 0 }, expectedPrefix: [] }, + { + probe: 'times out', + result: { + status: null, + error: Object.assign(new Error('timed out'), { code: 'ETIMEDOUT' }) + }, + expectedPrefix: ['--no-sandbox'] + }, + { + probe: 'cannot start', + result: { status: null, error: Object.assign(new Error('missing'), { code: 'ENOENT' }) }, + expectedPrefix: ['--no-sandbox'] + } + ])( + 'uses the extracted AppImage sandbox fallback when the userns probe $probe', + async ({ result: userNamespaceResult, expectedPrefix }) => { + const platformDescriptor = Object.getOwnPropertyDescriptor(process, 'platform') + const getuidDescriptor = Object.getOwnPropertyDescriptor(process, 'getuid') + const root = await mkdtemp(join(tmpdir(), 'orca-extracted-appimage-')) + temporaryDirectories.push(root) + const executable = join(root, 'orca-ide') + await writeFile(join(root, 'AppRun'), '', { mode: 0o755 }) + process.env.ORCA_APP_EXECUTABLE = executable + Object.defineProperty(process, 'platform', { value: 'linux' }) + Object.defineProperty(process, 'getuid', { configurable: true, value: () => 1000 }) + spawnSyncMock.mockReturnValue(userNamespaceResult) + const child = new FakeChildProcess() + spawnMock.mockReturnValue(child) + + try { + const result = serveOrcaApp({ json: true }) + queueMicrotask(() => child.emit('exit', 0, null)) + await expect(result).resolves.toBe(0) + expect(spawnSyncMock).toHaveBeenCalledWith( + 'unshare', + ['-Ur', 'true'], + expect.objectContaining({ stdio: 'ignore', timeout: 2_000 }) + ) + expect(spawnMock).toHaveBeenCalledWith( + executable, + [...expectedPrefix, '--serve', '--serve-json'], + // Foreground serve must share POSIX job-control signals with its CLI supervisor. + expect.objectContaining({ detached: false }) + ) + } finally { + if (platformDescriptor) { + Object.defineProperty(process, 'platform', platformDescriptor) + } + if (getuidDescriptor) { + Object.defineProperty(process, 'getuid', getuidDescriptor) + } else { + Reflect.deleteProperty(process, 'getuid') + } + } + } + ) + it('prints recipe JSON from a detached server child and exits', async () => { const child = new FakeChildProcess() spawnMock.mockReturnValue(child) @@ -599,6 +635,7 @@ describe('serveOrcaApp', () => { describe('launchOrcaApp', () => { beforeEach(() => { spawnMock.mockReset() + spawnSyncMock.mockReset() }) afterEach(() => { @@ -618,4 +655,51 @@ describe('launchOrcaApp', () => { expect(child.unref).toHaveBeenCalled() }) + + it('adds the extracted-AppImage sandbox fallback for open launches', async () => { + const platformDescriptor = Object.getOwnPropertyDescriptor(process, 'platform') + const getuidDescriptor = Object.getOwnPropertyDescriptor(process, 'getuid') + const root = await mkdtemp(join(tmpdir(), 'orca-open-extracted-appimage-')) + const executable = join(root, 'orca-ide') + + try { + await writeFile(join(root, 'AppRun'), '') + process.env.ORCA_APP_EXECUTABLE = executable + process.env.ELECTRON_RUN_AS_NODE = '1' + Object.defineProperty(process, 'platform', { configurable: true, value: 'linux' }) + Object.defineProperty(process, 'getuid', { configurable: true, value: () => 1000 }) + spawnSyncMock.mockReturnValue({ status: 1 }) + const child = new FakeChildProcess() + spawnMock.mockReturnValue(child) + + launchOrcaApp() + + expect(spawnSyncMock).toHaveBeenCalledWith( + 'unshare', + ['-Ur', 'true'], + expect.objectContaining({ stdio: 'ignore', timeout: 2_000 }) + ) + expect(spawnMock).toHaveBeenCalledWith( + executable, + ['--no-sandbox'], + expect.objectContaining({ + detached: true, + stdio: 'ignore', + env: expect.not.objectContaining({ ELECTRON_RUN_AS_NODE: '1' }) + }) + ) + expect(child.unref).toHaveBeenCalledOnce() + } finally { + await rm(root, { recursive: true, force: true }) + delete process.env.ELECTRON_RUN_AS_NODE + if (platformDescriptor) { + Object.defineProperty(process, 'platform', platformDescriptor) + } + if (getuidDescriptor) { + Object.defineProperty(process, 'getuid', getuidDescriptor) + } else { + Reflect.deleteProperty(process, 'getuid') + } + } + }) }) diff --git a/src/cli/runtime/launch.ts b/src/cli/runtime/launch.ts index bd7d939be5a..a326ae333f5 100644 --- a/src/cli/runtime/launch.ts +++ b/src/cli/runtime/launch.ts @@ -1,6 +1,8 @@ import { spawn as spawnProcess, type SpawnOptions } from 'node:child_process' -import { resolve } from 'node:path' +import { existsSync } from 'node:fs' +import { dirname, join, resolve } from 'node:path' import { StringDecoder } from 'node:string_decoder' +import { runProcessSync } from '../../shared/child-process/run-process' import { SERVE_UPDATE_HANDOFF_PATH_ENV, getServeUpdateHandoffPath @@ -19,6 +21,7 @@ import { import { RuntimeClientError } from './types' const IGNORED_NON_RECIPE_STDOUT = '[serve] ignored non-recipe stdout' +const USER_NAMESPACE_PROBE_TIMEOUT_MS = 2_000 export function launchOrcaApp(): void { const overrideCommand = process.env.ORCA_OPEN_COMMAND @@ -29,7 +32,7 @@ export function launchOrcaApp(): void { const overrideExecutable = process.env.ORCA_APP_EXECUTABLE if (typeof overrideExecutable === 'string' && overrideExecutable.trim().length > 0) { - spawnDetached(overrideExecutable, getExecutableAppArgs(), { + spawnDetached(overrideExecutable, getExecutableAppArgs(overrideExecutable), { ...getExecutableSpawnOptions(overrideExecutable), env: stripElectronRunAsNode(process.env) }) @@ -50,7 +53,7 @@ export function launchOrcaApp(): void { } } - spawnDetached(process.execPath, [], { + spawnDetached(process.execPath, getExecutableAppArgs(process.execPath), { env: stripElectronRunAsNode(process.env) }) return @@ -86,10 +89,7 @@ export function serveOrcaApp( } = {} ): Promise { const executable = resolveForegroundOrcaExecutable() - const childArgs = [...getExecutableAppArgs()] - if (process.env.ORCA_APPIMAGE_NO_SANDBOX === '1') { - childArgs.push('--no-sandbox') - } + const childArgs = [...getExecutableAppArgs(executable)] childArgs.push('--serve') if (args.json) { childArgs.push('--serve-json') @@ -121,7 +121,6 @@ export function serveOrcaApp( ? getServeUpdateHandoffPath(getDefaultUserDataPath()) : null const childEnv = stripElectronRunAsNode(process.env) - delete childEnv.ORCA_APPIMAGE_NO_SANDBOX if (handoffPath) { childEnv[SERVE_UPDATE_HANDOFF_PATH_ENV] = handoffPath } @@ -256,8 +255,34 @@ function waitForRecipeJson(child: ReturnType): Promise { diff --git a/src/cli/runtime/serve-signal-exit-diagnostic.test.ts b/src/cli/runtime/serve-signal-exit-diagnostic.test.ts index f5d348798c3..cc47deec3f7 100644 --- a/src/cli/runtime/serve-signal-exit-diagnostic.test.ts +++ b/src/cli/runtime/serve-signal-exit-diagnostic.test.ts @@ -134,6 +134,36 @@ describe('superviseForegroundServe signal exits', () => { expect(vi.getTimerCount()).toBe(0) }) + it('forwards Linux terminal hangup and removes the listener after exit', async () => { + setPlatform('linux') + const listenersBefore = process.listeners('SIGHUP') + const child = new FakeChildProcess() + const supervised = superviseChild(child) + + expect(process.listeners('SIGHUP')).toHaveLength(listenersBefore.length + 1) + process.emit('SIGHUP', 'SIGHUP') + expect(child.kill).toHaveBeenCalledWith('SIGHUP') + + child.emit('exit', null, 'SIGHUP') + await expect(supervised).resolves.toBe(0) + expect(process.listeners('SIGHUP')).toEqual(listenersBefore) + + const killCallsAfterExit = child.kill.mock.calls.length + process.emit('SIGHUP', 'SIGHUP') + expect(child.kill).toHaveBeenCalledTimes(killCallsAfterExit) + }) + + it('treats a child exit through the caller-forwarded SIGINT as graceful', async () => { + setPlatform('linux') + const child = new FakeChildProcess() + const supervised = superviseChild(child) + + process.emit('SIGINT', 'SIGINT') + child.emit('exit', null, 'SIGINT') + + await expect(supervised).resolves.toBe(0) + }) + it('does not terminate an exited child when update handoff completion fails late', async () => { vi.useFakeTimers() const missingParent = await mkdtemp(join(tmpdir(), 'orca-serve-missing-handoff-')) diff --git a/src/cli/runtime/serve-update-supervisor.ts b/src/cli/runtime/serve-update-supervisor.ts index a5a303dc1a2..f791ef2ae6e 100644 --- a/src/cli/runtime/serve-update-supervisor.ts +++ b/src/cli/runtime/serve-update-supervisor.ts @@ -86,8 +86,8 @@ export async function superviseForegroundServe( handoff?.phase !== 'install-requested' || (child.pid !== undefined && handoff.servingPid !== child.pid) ) { - if (typeof result.code === 'number') { - return result.code + if (typeof result.code === 'number' || result.signalWasForwarded) { + return result.code ?? 0 } throw serveSignalExitError(result.signal) } @@ -114,8 +114,11 @@ function waitForForegroundChild( code: number | null signal: NodeJS.Signals | null readiness: ServeReadiness + signalWasForwarded: boolean }> { return new Promise((resolveWait, reject) => { + const forwardsHangup = process.platform === 'linux' + const forwardedSignals = new Set() let forceKillTimer: ReturnType | null = null let readyTimer: ReturnType | null = null let readiness: ServeReadiness = expected ? 'pending' : 'not-expected' @@ -155,6 +158,7 @@ function waitForForegroundChild( const forwardSignal = (signal: NodeJS.Signals): void => { // A Windows console delivers Ctrl-C to parent and child; child.kill would terminate the child mid-teardown. if (process.platform !== 'win32') { + forwardedSignals.add(signal) child.kill(signal) } forceKillTimer ??= setTimeout(() => child.kill('SIGKILL'), SERVE_CHILD_FORCE_KILL_GRACE_MS) @@ -188,6 +192,9 @@ function waitForForegroundChild( const cleanup = (): void => { process.off('SIGINT', forwardSignal) process.off('SIGTERM', forwardSignal) + if (forwardsHangup) { + process.off('SIGHUP', forwardSignal) + } if (typeof child.off === 'function') { child.off('message', handleMessage) } @@ -200,6 +207,9 @@ function waitForForegroundChild( } process.on('SIGINT', forwardSignal) process.on('SIGTERM', forwardSignal) + if (forwardsHangup) { + process.on('SIGHUP', forwardSignal) + } if (typeof child.on === 'function') { child.on('message', handleMessage) } @@ -213,7 +223,8 @@ function waitForForegroundChild( const handleExit = (code: number | null, signal: NodeJS.Signals | null): void => { childSettled = true cleanup() - void stateWrite.then(() => resolveWait({ code, signal, readiness })) + const signalWasForwarded = signal !== null && forwardedSignals.has(signal) + void stateWrite.then(() => resolveWait({ code, signal, readiness, signalWasForwarded })) } child.once('error', (error) => { childSettled = true diff --git a/src/cli/serve-electron-flag-parity.test.ts b/src/cli/serve-electron-flag-parity.test.ts index 4213d360a41..a4964e1a824 100644 --- a/src/cli/serve-electron-flag-parity.test.ts +++ b/src/cli/serve-electron-flag-parity.test.ts @@ -35,7 +35,7 @@ describe('serve flag parity between the CLI spec and the Electron argv rewrite', expect(normalizeServeModeArgv(argv)).toEqual(expected) if (takesValue) { - // The equals form is the other shape `orca serve` accepts, and getServeOptions only reads the next token. + // The equals form is the other shape `orca serve` accepts; normalize it to the internal shape. expect(normalizeServeModeArgv(['/AppRun', 'serve', `--${flag}=value`])).toEqual(expected) } else { // A boolean with an attached value is not a truthy assertion: the CLI reads these as @@ -53,20 +53,19 @@ describe('serve flag parity between the CLI spec and the Electron argv rewrite', }) it('emits the same --serve-* names the CLI spawns with and the main process reads', () => { - // Why source text: serveOrcaApp spawns a real process and getServeOptions is not exported, so - // both ends of the contract are only readable statically. Without this leg the rewrite could - // emit a name nothing reads and every behavioural assertion above would still pass. + // Why source text: serveOrcaApp spawns a real process; keeping both names visible here makes + // the rewrite/parser contract fail loudly if either side drifts. const launchSource = readFileSync(join(process.cwd(), 'src/cli/runtime/launch.ts'), 'utf8') - const mainSource = readFileSync( - join(process.cwd(), 'src/main/startup/main-process-serve.ts'), + const serveOptionsSource = readFileSync( + join(process.cwd(), 'src/main/startup/serve-options.ts'), 'utf8' ) - const start = mainSource.indexOf('export function getServeOptions(') + const start = serveOptionsSource.indexOf('export function getServeOptions(') // Why bound the anchor: an unresolved indexOf slices to EOF and passes vacuously. expect(start).toBeGreaterThanOrEqual(0) - const end = mainSource.indexOf('\n}', start) + const end = serveOptionsSource.indexOf('\n}', start) expect(end).toBeGreaterThan(start) - const getServeOptionsBody = mainSource.slice(start, end) + const getServeOptionsBody = serveOptionsSource.slice(start, end) for (const flag of translatedFlags) { expect(launchSource).toContain(`'--serve-${flag}'`) diff --git a/src/main/cli/cli-command-installation.ts b/src/main/cli/cli-command-installation.ts index fbabc3bf6dd..a3b5e9a0523 100644 --- a/src/main/cli/cli-command-installation.ts +++ b/src/main/cli/cli-command-installation.ts @@ -20,11 +20,9 @@ import { } from './cli-command-filesystem-transaction' import { DEV_LAUNCHER_DIR, LEGACY_LINUX_COMMAND_NAME } from './cli-install-constants' import { buildWindowsForwarder } from './cli-dev-launcher' -import { isMissingError, isPermissionError } from './cli-install-errors' +import { isPermissionError } from './cli-install-errors' import { isPathInsideOrEqual } from './cli-install-path-format' -const STABLE_LEGACY_INSPECTION_ATTEMPTS = 3 - export class CliCommandInstallation extends CliCommandInspection { protected async installSymlink(status: CliInstallStatus): Promise { const commandPath = status.commandPath @@ -198,34 +196,25 @@ export class CliCommandInstallation extends CliCommandInspection { }) | null > { - for (let attempt = 0; attempt < STABLE_LEGACY_INSPECTION_ATTEMPTS; attempt += 1) { - const before = await readEntrySnapshot(commandPath) - if (!before) { - return null - } - let target: string | null = null - try { - target = before.isSymbolicLink ? await readlink(commandPath) : null - } catch (error) { - if (isMissingError(error)) { - continue - } - throw error - } - const after = await readEntrySnapshot(commandPath) - if (after && hasSameSnapshot(before, after)) { - const resolvedTarget = target ? resolve(dirname(commandPath), target) : null - return { - fileSha256: null, - rawSymlinkTarget: target, - snapshot: after, - managed: Boolean( - resolvedTarget && this.isManagedLegacyLinuxTarget(resolvedTarget, launcherPath) - ) - } - } + const inspected = await inspectStableCommand(commandPath, () => + this.inspectSymlink(commandPath, launcherPath) + ) + if (!inspected.snapshot) { + return null + } + const resolvedTarget = inspected.rawSymlinkTarget + ? resolve(dirname(commandPath), inspected.rawSymlinkTarget) + : inspected.status.currentTarget + return { + fileSha256: inspected.fileSha256, + rawSymlinkTarget: inspected.rawSymlinkTarget, + snapshot: inspected.snapshot, + managed: Boolean( + resolvedTarget && + (this.isManagedLegacyLinuxTarget(resolvedTarget, launcherPath) || + (this.appImagePath && resolve(resolvedTarget) === resolve(this.appImagePath))) + ) } - throw new Error(`The command at ${commandPath} changed while Orca inspected it.`) } private async restoreQuarantinedCommand( diff --git a/src/main/cli/cli-installer.test.ts b/src/main/cli/cli-installer.test.ts index 1a5279644e9..51d5cf05e35 100644 --- a/src/main/cli/cli-installer.test.ts +++ b/src/main/cli/cli-installer.test.ts @@ -370,6 +370,56 @@ describe('CliInstaller', () => { } ) + it.skipIf(process.platform === 'win32')( + 'removes a legacy AppImage wrapper only when it names the current AppImage', + async () => { + const fixture = await makeFixture() + const homePath = join(fixture.root, 'home') + const commandDir = join(homePath, '.local', 'bin') + const legacyCommandPath = join(commandDir, 'orca') + const appImagePath = join(fixture.root, 'Orca.AppImage') + const foreignAppImagePath = join(fixture.root, 'Other.AppImage') + const cacheRootPath = join(fixture.root, 'cache') + await mkdir(commandDir, { recursive: true }) + await writeFile(appImagePath, '#!/usr/bin/env bash\n', { + encoding: 'utf8', + mode: 0o755 + }) + await writeFile(foreignAppImagePath, '#!/usr/bin/env bash\n', { + encoding: 'utf8', + mode: 0o755 + }) + await writeFile(legacyCommandPath, buildLegacyAppImageCliWrapper(appImagePath), { + encoding: 'utf8', + mode: 0o755 + }) + + const installer = new CliInstaller({ + platform: 'linux', + isPackaged: true, + userDataPath: fixture.userDataPath, + appPath: fixture.appPath, + appImagePath, + appImageCacheRootPath: cacheRootPath, + appImageExtractRunner: fakeAppImageExtractRunner, + homePath, + processPathEnv: commandDir + }) + + await installer.install() + await expect(lstat(legacyCommandPath)).rejects.toMatchObject({ code: 'ENOENT' }) + + await writeFile(legacyCommandPath, buildLegacyAppImageCliWrapper(foreignAppImagePath), { + encoding: 'utf8', + mode: 0o755 + }) + await installer.remove() + await expect(readFile(legacyCommandPath, 'utf8')).resolves.toBe( + buildLegacyAppImageCliWrapper(foreignAppImagePath) + ) + } + ) + // Why: the privilegedRunner is injectable so the EACCES→osascript path can be // exercised in integration without spawning osascript in unit tests. it.skipIf(process.platform === 'win32' || process.getuid?.() === 0)( diff --git a/src/main/cli/packaged-cli-assets.test.ts b/src/main/cli/packaged-cli-assets.test.ts index d5c17123ec1..1fb71e3e00f 100644 --- a/src/main/cli/packaged-cli-assets.test.ts +++ b/src/main/cli/packaged-cli-assets.test.ts @@ -305,6 +305,63 @@ node -e 'console.log(JSON.stringify({ await rm(root, { recursive: true, force: true }) } }) + + itRunsUnixShell('keeps Linux serve on the CLI entrypoint in node mode', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-linux-cli-serve-')) + try { + const appDir = join(root, 'Orca') + const resourcesDir = join(appDir, 'resources') + const launcherDir = join(resourcesDir, 'bin') + const cliDir = join(resourcesDir, 'app.asar.unpacked', 'out', 'cli') + const launcherPath = join(launcherDir, 'orca-ide') + const appRunPath = join(appDir, 'AppRun') + const electronPath = join(appDir, 'orca-ide') + const cliPath = join(cliDir, 'index.js') + const statePath = join(root, 'launch-state.json') + + await mkdir(launcherDir, { recursive: true }) + await mkdir(cliDir, { recursive: true }) + await copyFile(linuxLauncherAsset, launcherPath) + await writeFile(cliPath, '', 'utf8') + // An accidental AppRun handoff would skip CLI validation and fail this contract. + await writeFile( + appRunPath, + `#!/usr/bin/env bash +printf 'unexpected AppRun handoff\n' >&2 +exit 97 +`, + { encoding: 'utf8', mode: 0o755 } + ) + await writeFile( + electronPath, + `#!/usr/bin/env node +require('node:fs').writeFileSync(process.env.ORCA_TEST_LAUNCH_STATE, JSON.stringify({ + argv: process.argv.slice(2), + runAsNode: process.env.ELECTRON_RUN_AS_NODE ?? null +})) +`, + { encoding: 'utf8', mode: 0o755 } + ) + + await execFileAsync(launcherPath, ['serve', '--recipe-json', '--project-root', '/tmp/repo'], { + env: { ...process.env, ORCA_TEST_LAUNCH_STATE: statePath } + }) + const payload = JSON.parse(await readFile(statePath, 'utf8')) as { + argv: string[] + runAsNode: string | null + } + expect(payload.argv).toEqual([ + cliPath, + 'serve', + '--recipe-json', + '--project-root', + '/tmp/repo' + ]) + expect(payload.runAsNode).toBe('1') + } finally { + await rm(root, { recursive: true, force: true }) + } + }) }) async function waitForListenerState(path: string): Promise<{ pid: number; port: number }> { diff --git a/src/main/linux-package-downloaded-status.ts b/src/main/linux-package-downloaded-status.ts new file mode 100644 index 00000000000..df0c9b32e6d --- /dev/null +++ b/src/main/linux-package-downloaded-status.ts @@ -0,0 +1,88 @@ +import type { UpdateStatus } from '../shared/update-status-types' +import { + captureLinuxPackageArtifact, + clearTrackedLinuxPackageArtifact, + getTrackedLinuxPackageArtifact +} from './linux-package-update-recovery' +import { getLinuxPackageType } from './linux-update-package-type' +import type { LinuxPackageArtifact } from './linux-package-update-recovery' + +export const LINUX_PACKAGE_MARKER_UNUSABLE_MESSAGE = + 'Orca could not verify the installed Linux package format, so it will not install this update automatically. Download the update from the official release page and install it manually.' +export const LINUX_PACKAGE_EXTERNALLY_MANAGED_MESSAGE = + 'This copy of Orca is managed by your system package manager, so Orca cannot install updates itself. Update Orca through your distribution instead.' +export const LINUX_PACKAGE_MANUAL_INSTALL_MESSAGE = + 'Quit Orca before running the system package install command.' +const PACKAGE_METADATA_UNUSABLE_MESSAGE = + 'The downloaded package metadata could not be verified. Quit Orca before downloading and installing the update from the official release page.' + +export function createLinuxPackageManualInstallStatus( + artifact: Pick +): UpdateStatus { + return { + state: 'error', + message: LINUX_PACKAGE_MANUAL_INSTALL_MESSAGE, + recovery: { + kind: 'linux-package-install', + packageType: artifact.packageType, + reason: 'manual-install-required', + version: artifact.version + } + } +} + +export function getRetainedLinuxPackageManualInstallStatus(): UpdateStatus | null { + const artifact = getTrackedLinuxPackageArtifact() + return artifact ? createLinuxPackageManualInstallStatus(artifact) : null +} + +function getActiveDownloadVersion(status: UpdateStatus): string | null { + if (status.state === 'downloading' || status.state === 'downloaded') { + return status.version + } + if (status.state === 'error' && status.recovery?.kind === 'linux-package-install') { + return status.recovery.version + } + return null +} + +export function shouldIgnoreDownloadedUpdateEvent( + status: UpdateStatus, + infoVersion: string, + pendingVersion: string +): boolean { + const activeDownloadVersion = getActiveDownloadVersion(status) + return ( + activeDownloadVersion === null || + infoVersion !== activeDownloadVersion || + (pendingVersion !== '' && infoVersion !== pendingVersion) + ) +} + +export function resolveLinuxPackageDownloadedStatus(info: { + version: string +}): UpdateStatus | null { + const packageType = getLinuxPackageType() + if (packageType === 'non-root') { + return null + } + if (packageType === 'unusable') { + clearTrackedLinuxPackageArtifact() + return { + state: 'error', + message: LINUX_PACKAGE_MARKER_UNUSABLE_MESSAGE, + version: info.version, + retryable: false + } + } + const artifact = captureLinuxPackageArtifact(info) + if (!artifact) { + return { + state: 'error', + message: PACKAGE_METADATA_UNUSABLE_MESSAGE, + version: info.version, + retryable: false + } + } + return createLinuxPackageManualInstallStatus(artifact) +} diff --git a/src/main/linux-package-install-command.test.ts b/src/main/linux-package-install-command.test.ts index 368e6620fd5..c76c062bbce 100644 --- a/src/main/linux-package-install-command.test.ts +++ b/src/main/linux-package-install-command.test.ts @@ -252,3 +252,54 @@ describe('buildLinuxPackageInstallCommand', () => { }) }) }) + +describe('hasTrustedPackageManagerFor', () => { + it('accepts a deb host that has dpkg', async () => { + install('/usr/bin/dpkg') + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('deb')).toBe(true) + }) + + it('accepts a deb host that has only apt', async () => { + install('/usr/bin/apt') + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('deb')).toBe(true) + }) + + // The #17702 case: an Arch rebuild of the .deb inherits the marker but has no deb tooling. + it('rejects a deb marker on a host with only pacman', async () => { + install('/usr/bin/pacman') + install('/usr/bin/sudo') + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('deb')).toBe(false) + }) + + it('rejects an rpm marker on a host with only deb tooling', async () => { + install('/usr/bin/dpkg') + install('/usr/bin/apt') + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('rpm')).toBe(false) + }) + + it('accepts each rpm-family manager on its own', async () => { + for (const name of ['zypper', 'dnf', 'yum', 'rpm']) { + vi.resetModules() + executables = new Map() + install(`/usr/bin/${name}`) + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('rpm')).toBe(true) + } + }) + + it('ignores a package manager outside the trusted directories', async () => { + install('/home/user/.local/bin/dpkg') + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('deb')).toBe(false) + }) + + it('ignores a non-executable file at a trusted path', async () => { + install('/usr/bin/dpkg', { mode: 0o644 }) + const { hasTrustedPackageManagerFor } = await loadCommandModule() + expect(hasTrustedPackageManagerFor('deb')).toBe(false) + }) +}) diff --git a/src/main/linux-package-install-command.ts b/src/main/linux-package-install-command.ts index 60a8ecdca00..728ffcd1d92 100644 --- a/src/main/linux-package-install-command.ts +++ b/src/main/linux-package-install-command.ts @@ -52,6 +52,16 @@ export function resolveTrustedExecutable(name: string): string | null { return null } +/** + * Whether this host has any package manager able to install the marker's format. A repackaged + * install (AUR, Nix, a container rebuild) inherits the `package-type` marker from the .deb/.rpm it + * was built from, so the marker alone never proves the host can act on it. + */ +export function hasTrustedPackageManagerFor(packageType: LinuxRootPackageType): boolean { + const candidates = packageType === 'deb' ? DEB_PACKAGE_MANAGERS : RPM_PACKAGE_MANAGERS + return candidates.some((candidate) => resolveTrustedExecutable(candidate.name) !== null) +} + /** * Builds the interactive command the user pastes into their own terminal. Every token except the * package path is a fixed literal, and the path is POSIX-single-quoted — Orca never runs this. diff --git a/src/main/linux-package-install-diagnostic.test.ts b/src/main/linux-package-install-diagnostic.test.ts index 5b8e1b1c77b..14e82a0763b 100644 --- a/src/main/linux-package-install-diagnostic.test.ts +++ b/src/main/linux-package-install-diagnostic.test.ts @@ -3,7 +3,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as DiagnosticModule from './linux-package-install-diagnostic' const ESC = String.fromCharCode(27) - let diagnostic: typeof DiagnosticModule beforeEach(async () => { @@ -20,64 +19,35 @@ afterEach(() => { }) describe('redactLinuxPackageInstallText', () => { - it('strips ANSI escape sequences', () => { - const text = `${ESC}[31mdpkg: error${ESC}[0m processing` - expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('dpkg: error processing') + it('strips terminal escapes and control characters', () => { + const text = `${ESC}[?25l${ESC}[31mdpkg:\r\n\terror\u0000${ESC}[0m${ESC}[?25h` + expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('dpkg: error') }) - it('strips ANSI sequences with private and intermediate bytes', () => { - const text = `${ESC}[?25lworking${ESC}[?25h` - expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('working') + it('strips string escape payloads and remaining two-byte escapes', () => { + const BEL = String.fromCharCode(7) + const text = + `${ESC}]8;;https://tracker.invalid/report${BEL}dpkg${ESC}]8;;${BEL} ` + + `${ESC}P1;2|payload${ESC}\\failed${ESC}c` + expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('dpkg failed') }) - it('replaces control characters and collapses whitespace', () => { - const text = `line one\r\n\tline\u0000two spaced\u007f` - expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('line one line two spaced') - }) - - it('replaces the cached package path with a placeholder', () => { - const packagePath = '/home/user/.cache/orca-updater/Orca-1.2.3.deb' - const text = `dpkg: error processing ${packagePath} (--install)` - expect(diagnostic.redactLinuxPackageInstallText(text, packagePath)).toBe( - 'dpkg: error processing (--install)' - ) - }) - - it('replaces every occurrence of the package path', () => { - const packagePath = '/tmp/orca.deb' - const text = `${packagePath} failed; retry ${packagePath}` - expect(diagnostic.redactLinuxPackageInstallText(text, packagePath)).toBe( - ' failed; retry ' - ) - }) - - it('replaces the home directory with a placeholder', () => { - const home = os.homedir() - const text = `could not read ${home}/.config/orca/settings.json` - expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe( - 'could not read /.config/orca/settings.json' - ) - }) - - it('prefers the package placeholder for a path inside the home directory', () => { + it('replaces every cached package-path occurrence before the home directory', () => { const home = os.homedir() const packagePath = `${home}/.cache/orca-updater/Orca-1.2.3.deb` - expect(diagnostic.redactLinuxPackageInstallText(`install ${packagePath}`, packagePath)).toBe( - 'install ' + const text = `${packagePath} failed; retry ${packagePath}; config ${home}/.config/orca` + expect(diagnostic.redactLinuxPackageInstallText(text, packagePath)).toBe( + ' failed; retry ; config /.config/orca' ) }) - it('replaces the bare username with a placeholder', () => { - // Why: sudo names the user without any path around it, so the rule never sees it. + it('replaces a bare username without corrupting short names', () => { vi.spyOn(os, 'userInfo').mockReturnValue({ username: 'devuser' } as os.UserInfo) expect( diagnostic.redactLinuxPackageInstallText('devuser is not in the sudoers file', null) ).toBe(' is not in the sudoers file') - }) - it('leaves a username shorter than three characters alone', () => { - // Short names would corrupt unrelated words. - vi.spyOn(os, 'userInfo').mockReturnValue({ username: 'ci' } as os.UserInfo) + vi.mocked(os.userInfo).mockReturnValue({ username: 'ci' } as os.UserInfo) expect(diagnostic.redactLinuxPackageInstallText('ci: incident in circuit', null)).toBe( 'ci: incident in circuit' ) @@ -92,34 +62,23 @@ describe('redactLinuxPackageInstallText', () => { ) }) - it('truncates to 1024 characters', () => { - const result = diagnostic.redactLinuxPackageInstallText('a'.repeat(2000), null) - expect(result).toHaveLength(1024) + it('bounds the result at 1024 characters', () => { + expect(diagnostic.redactLinuxPackageInstallText('a'.repeat(2_000), null)).toHaveLength(1_024) + expect(diagnostic.redactLinuxPackageInstallText('a'.repeat(1_024), null)).toHaveLength(1_024) }) - it('keeps text at exactly the limit', () => { - const result = diagnostic.redactLinuxPackageInstallText('a'.repeat(1024), null) - expect(result).toHaveLength(1024) - }) - - it('returns null for empty and whitespace-only input', () => { + it('returns null when no visible text remains', () => { expect(diagnostic.redactLinuxPackageInstallText('', null)).toBeNull() expect(diagnostic.redactLinuxPackageInstallText(' \n\t ', null)).toBeNull() expect(diagnostic.redactLinuxPackageInstallText(`${ESC}[0m`, null)).toBeNull() - }) - - it('returns null for null and undefined', () => { expect(diagnostic.redactLinuxPackageInstallText(null, null)).toBeNull() expect(diagnostic.redactLinuxPackageInstallText(undefined, null)).toBeNull() }) - it('uses the message of an Error', () => { + it('normalizes errors, objects, and primitives', () => { expect(diagnostic.redactLinuxPackageInstallText(new Error('pkexec failed'), null)).toBe( 'pkexec failed' ) - }) - - it('serializes plain objects and other primitives', () => { expect(diagnostic.redactLinuxPackageInstallText({ code: 127 }, null)).toBe('{"code":127}') expect(diagnostic.redactLinuxPackageInstallText(127, null)).toBe('127') }) @@ -130,208 +89,22 @@ describe('redactLinuxPackageInstallText', () => { expect(diagnostic.redactLinuxPackageInstallText(circular, null)).toBeNull() }) - it('strips an OSC hyperlink along with its URL payload', () => { - const BEL = String.fromCharCode(7) - const text = `${ESC}]8;;https://tracker.invalid/report${BEL}dpkg: error${ESC}]8;;${BEL} processing` - expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('dpkg: error processing') - }) - - it('strips a string-terminated DCS sequence and a two-byte escape', () => { - const text = `${ESC}P1;2|payload${ESC}\\dpkg${ESC}c: error` - expect(diagnostic.redactLinuxPackageInstallText(text, null)).toBe('dpkg: error') - }) - it('ignores an empty package path', () => { expect(diagnostic.redactLinuxPackageInstallText('plain output', '')).toBe('plain output') }) }) describe('createUpdaterDiagnosticLogger', () => { - it('retains redacted error output while capturing', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture('/tmp/orca.deb') - logger.error(`${ESC}[31mpkexec: /tmp/orca.deb not authorized${ESC}[0m`) - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toEqual({ - message: 'pkexec: not authorized', - reason: 'authentication-denied' - }) - }) - - it('ignores non-error levels', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - logger.info('downloading') - logger.warn('retrying') - logger.debug('verbose') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toBeNull() - }) - - it('still forwards every level to the console', () => { + it('forwards every updater level to the matching console method', () => { const logger = diagnostic.createUpdaterDiagnosticLogger() logger.info('a') logger.warn('b') logger.error('c') logger.debug('d') + expect(console.info).toHaveBeenCalledWith('[autoUpdater]', 'a') expect(console.warn).toHaveBeenCalledWith('[autoUpdater]', 'b') expect(console.error).toHaveBeenCalledWith('[autoUpdater]', 'c') expect(console.debug).toHaveBeenCalledWith('[autoUpdater]', 'd') }) - - it('retains nothing outside a capture window', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - logger.error('unrelated failure') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toBeNull() - }) - - it('keeps the last usable error and ignores empty ones', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - logger.error('first failure') - logger.error('second failure') - logger.error('') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toEqual({ - message: 'second failure', - reason: 'package-install-failed' - }) - }) - - it('hands back and clears the diagnostic when capture ends', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - logger.error('request dismissed') - expect(diagnostic.endLinuxPackageInstallDiagnosticCapture()).toEqual({ - message: 'request dismissed', - reason: 'authentication-denied' - }) - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toBeNull() - logger.error('later noise') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toBeNull() - }) - - it('drops a previous attempt when a new capture begins', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture('/tmp/a.deb') - logger.error('old failure') - diagnostic.beginLinuxPackageInstallDiagnosticCapture('/tmp/b.deb') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toBeNull() - logger.error('new failure at /tmp/b.deb') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toEqual({ - message: 'new failure at ', - reason: 'package-install-failed' - }) - }) - - it('classifies the original output, not the redacted text', () => { - // A user named "age" turns "agent" into "nt", which would hide the missing polkit agent. - vi.spyOn(os, 'userInfo').mockReturnValue({ username: 'age' } as os.UserInfo) - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - logger.error('Error executing command as another user: No authentication agent found for age.') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toEqual({ - message: - 'Error executing command as another user: No authentication nt found for .', - reason: 'authentication-agent-unavailable' - }) - }) - - it('keeps a specific verdict when a generic line follows it', () => { - // electron-updater logs the polkit output first, then "Command failed, exited with code 126". - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - logger.error('polkit-agent-helper-1: no authentication agent found') - logger.error('Command failed, exited with code 126') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toEqual({ - message: 'polkit-agent-helper-1: no authentication agent found', - reason: 'authentication-agent-unavailable' - }) - }) - - it('lets a later specific line replace an earlier one', () => { - const logger = diagnostic.createUpdaterDiagnosticLogger() - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - logger.error('no authentication agent') - logger.error('request dismissed') - expect(diagnostic.getLinuxPackageInstallDiagnostic()).toEqual({ - message: 'request dismissed', - reason: 'authentication-denied' - }) - }) - - it('returns null from an empty capture window', () => { - diagnostic.beginLinuxPackageInstallDiagnosticCapture(null) - expect(diagnostic.endLinuxPackageInstallDiagnosticCapture()).toBeNull() - }) -}) - -describe('classifyLinuxPackageInstallFailure', () => { - it('reports a missing authentication agent', () => { - for (const text of [ - 'Error executing command as another user: No authentication agent found.', - 'polkit-agent-helper: agent not found', - 'polkit agent was not found' - ]) { - expect(diagnostic.classifyLinuxPackageInstallFailure(text)).toBe( - 'authentication-agent-unavailable' - ) - } - }) - - it('reports a denied authentication', () => { - for (const text of [ - 'Error executing command as another user: Request dismissed', - 'polkit: Authentication failed', - 'Error executing command as another user: Not authorized', - 'Authorization failed for org.freedesktop.policykit.exec', - 'pkexec: 3 incorrect password attempts' - ]) { - expect(diagnostic.classifyLinuxPackageInstallFailure(text)).toBe('authentication-denied') - } - }) - - it('prefers the agent reason when both patterns appear', () => { - expect( - diagnostic.classifyLinuxPackageInstallFailure( - 'No authentication agent found; authentication failed' - ) - ).toBe('authentication-agent-unavailable') - }) - - it('falls back to a generic failure for localized output', () => { - expect( - diagnostic.classifyLinuxPackageInstallFailure( - "Erreur lors de l'exécution : aucun agent d'authentification trouvé" - ) - ).toBe('package-install-failed') - }) - - it('falls back to a generic failure for unrecognized and missing output', () => { - expect(diagnostic.classifyLinuxPackageInstallFailure('dpkg: dependency problems')).toBe( - 'package-install-failed' - ) - expect(diagnostic.classifyLinuxPackageInstallFailure(null)).toBe('package-install-failed') - expect(diagnostic.classifyLinuxPackageInstallFailure('')).toBe('package-install-failed') - }) -}) - -describe('parseLinuxPackageInstallExitCode', () => { - it('parses electron-updater exit-code messages', () => { - expect( - diagnostic.parseLinuxPackageInstallExitCode( - new Error('Command /usr/bin/pkexec exited with code 127') - ) - ).toBe(127) - expect(diagnostic.parseLinuxPackageInstallExitCode('Command failed exited with code 0')).toBe(0) - }) - - it('parses a negative code and matches case-insensitively', () => { - expect(diagnostic.parseLinuxPackageInstallExitCode('Exited With Code -1')).toBe(-1) - }) - - it('returns null when no code is present', () => { - expect(diagnostic.parseLinuxPackageInstallExitCode(new Error('spawn ENOENT'))).toBeNull() - expect(diagnostic.parseLinuxPackageInstallExitCode('exited with code abc')).toBeNull() - expect(diagnostic.parseLinuxPackageInstallExitCode(null)).toBeNull() - expect(diagnostic.parseLinuxPackageInstallExitCode({ code: 127 })).toBeNull() - }) }) diff --git a/src/main/linux-package-install-diagnostic.ts b/src/main/linux-package-install-diagnostic.ts index bdef7c5450e..c65c62679fa 100644 --- a/src/main/linux-package-install-diagnostic.ts +++ b/src/main/linux-package-install-diagnostic.ts @@ -1,16 +1,7 @@ import os from 'node:os' -import type { LinuxPackageInstallFailureReason } from '../shared/update-status-types' - -/** The redacted text shown locally, paired with the reason classified from the ORIGINAL output. */ -export type LinuxPackageInstallDiagnostic = { - message: string - reason: LinuxPackageInstallFailureReason -} const MAX_DIAGNOSTIC_LENGTH = 1_024 -// Built via RegExp so the source carries no raw control bytes. Alternatives in order: CSI; then the -// string sequences (OSC/DCS/PM/APC/SOS), whose payload — an OSC 8 hyperlink URL, say — must be -// dropped with the introducer rather than left behind; then any remaining two-byte escape. +// Alternatives in order: CSI; string sequences whose payload must also be dropped; remaining two-byte escapes. const ANSI_ESCAPE = new RegExp( [ String.raw`\u001b\[[0-9;?]*[ -/]*[@-~]`, @@ -20,28 +11,7 @@ const ANSI_ESCAPE = new RegExp( 'g' ) const CONTROL_CHARACTERS = new RegExp(String.raw`[\u0000-\u001f\u007f]`, 'g') - -// Why: pkexec/polkit print these before any package manager runs; matching them keeps the UI from -// blaming dpkg for an authentication problem. Anything else stays generic on purpose. -const AGENT_UNAVAILABLE_PATTERNS = [ - /no authentication agent/i, - /polkit.{0,20}agent.{0,20}not found/i -] -const AUTHENTICATION_DENIED_PATTERNS = [ - /request dismissed/i, - /authentication failed/i, - /not authorized/i, - /authorization failed/i, - /incorrect password attempt/i -] - -let capturing = false -let retainedDiagnostic: string | null = null -// Why: classification must read the ORIGINAL text. Redaction can rewrite a pattern word — a user -// named "age" turns "No authentication agent found" into "No authentication nt" — which would -// silently downgrade a missing-agent failure to the generic reason. -let retainedReason: LinuxPackageInstallFailureReason | null = null -let redactedPackagePath: string | null = null +const MIN_REDACTED_USERNAME_LENGTH = 3 function stringifyLoggerValue(value: unknown): string { if (typeof value === 'string') { @@ -63,9 +33,6 @@ function stringifyLoggerValue(value: unknown): string { return String(value) } -// Short names would corrupt unrelated words, so they are left alone. -const MIN_REDACTED_USERNAME_LENGTH = 3 - function readUserName(): string | null { try { return os.userInfo().username || null @@ -75,16 +42,10 @@ function readUserName(): string | null { } function replaceAllLiteral(text: string, needle: string, replacement: string): string { - if (needle.length === 0) { - return text - } - return text.split(needle).join(replacement) + return needle.length === 0 ? text : text.split(needle).join(replacement) } -/** - * Turns arbitrary updater/child output into text safe to show locally: no ANSI, no control bytes, - * no home directory, no cached package path, bounded length. - */ +/** Removes terminal escapes and local identity from updater text before showing it in the UI. */ export function redactLinuxPackageInstallText( value: unknown, packagePath: string | null @@ -101,7 +62,7 @@ export function redactLinuxPackageInstallText( if (homeDir) { text = replaceAllLiteral(text, homeDir, '') } - // Why: sudo reports " is not in the sudoers file", which the home-directory rule cannot catch. + // Why: privilege tools can name the user without including their home directory. const userName = readUserName() if (userName && userName.length >= MIN_REDACTED_USERNAME_LENGTH) { text = replaceAllLiteral(text, userName, '') @@ -113,97 +74,16 @@ export function redactLinuxPackageInstallText( return text.length > MAX_DIAGNOSTIC_LENGTH ? text.slice(0, MAX_DIAGNOSTIC_LENGTH) : text } -/** Starts retaining redacted error output for one native root-package install attempt. */ -export function beginLinuxPackageInstallDiagnosticCapture(packagePath: string | null): void { - capturing = true - retainedDiagnostic = null - retainedReason = null - redactedPackagePath = packagePath -} - -/** Stops capture and hands back the retained diagnostic, clearing it for the next attempt. */ -export function endLinuxPackageInstallDiagnosticCapture(): LinuxPackageInstallDiagnostic | null { - const captured = getLinuxPackageInstallDiagnostic() - capturing = false - retainedDiagnostic = null - retainedReason = null - redactedPackagePath = null - return captured -} - -export function getLinuxPackageInstallDiagnostic(): LinuxPackageInstallDiagnostic | null { - return retainedDiagnostic === null - ? null - : { message: retainedDiagnostic, reason: retainedReason ?? 'package-install-failed' } -} - -function recordLinuxPackageInstallDiagnostic(value: unknown): void { - if (!capturing) { - return - } - const raw = stringifyLoggerValue(value) - const redacted = redactLinuxPackageInstallText(raw, redactedPackagePath) - if (!redacted) { - return - } - const reason = classifyLinuxPackageInstallFailure(raw) - // Why: electron-updater logs the polkit output first and a generic "exited with code N" line after, - // so a later generic line must not erase the specific verdict the card branches on. - if ( - reason === 'package-install-failed' && - retainedReason !== null && - retainedReason !== 'package-install-failed' - ) { - return - } - retainedDiagnostic = redacted - retainedReason = reason -} - -/** - * The `autoUpdater.logger`. Every level still reaches the same console method; only error output - * during an in-flight root-package install is retained, redacted, for the recovery card. - */ export function createUpdaterDiagnosticLogger(): { - info: (m: unknown) => void - warn: (m: unknown) => void - error: (m: unknown) => void - debug: (m: unknown) => void + info: (message: unknown) => void + warn: (message: unknown) => void + error: (message: unknown) => void + debug: (message: unknown) => void } { return { - info: (m: unknown) => console.info('[autoUpdater]', m), - warn: (m: unknown) => console.warn('[autoUpdater]', m), - error: (m: unknown) => { - recordLinuxPackageInstallDiagnostic(m) - console.error('[autoUpdater]', m) - }, - debug: (m: unknown) => console.debug('[autoUpdater]', m) + info: (message) => console.info('[autoUpdater]', message), + warn: (message) => console.warn('[autoUpdater]', message), + error: (message) => console.error('[autoUpdater]', message), + debug: (message) => console.debug('[autoUpdater]', message) } } - -export function classifyLinuxPackageInstallFailure( - diagnostic: string | null -): LinuxPackageInstallFailureReason { - if (!diagnostic) { - return 'package-install-failed' - } - if (AGENT_UNAVAILABLE_PATTERNS.some((pattern) => pattern.test(diagnostic))) { - return 'authentication-agent-unavailable' - } - if (AUTHENTICATION_DENIED_PATTERNS.some((pattern) => pattern.test(diagnostic))) { - return 'authentication-denied' - } - // Localized or unrecognized output must never be reported as a missing agent. - return 'package-install-failed' -} - -/** Parses the child exit status out of electron-updater's `Command exited with code `. */ -export function parseLinuxPackageInstallExitCode(error: unknown): number | null { - const message = error instanceof Error ? error.message : typeof error === 'string' ? error : '' - const match = /exited with code (-?\d{1,5})\b/i.exec(message) - if (!match) { - return null - } - const code = Number.parseInt(match[1], 10) - return Number.isFinite(code) ? code : null -} diff --git a/src/main/linux-package-update-recovery.test.ts b/src/main/linux-package-update-recovery.test.ts index ec2738b823b..859e80e74b2 100644 --- a/src/main/linux-package-update-recovery.test.ts +++ b/src/main/linux-package-update-recovery.test.ts @@ -5,18 +5,14 @@ import path from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type * as NodeFs from 'node:fs' import type { LinuxPackageInstallRecovery } from '../shared/update-status-types' +import type { LinuxPackageArtifact } from './linux-package-update-recovery' import type * as RecoveryModule from './linux-package-update-recovery' -const { showItemInFolderMock, getPackageTypeMock, buildCommandMock, hashPasses } = vi.hoisted( - () => ({ - showItemInFolderMock: vi.fn(), - getPackageTypeMock: vi.fn(), - buildCommandMock: vi.fn(), - hashPasses: { count: 0 } - }) -) - -vi.mock('electron', () => ({ shell: { showItemInFolder: showItemInFolderMock } })) +const { getPackageTypeMock, buildCommandMock, hashPasses } = vi.hoisted(() => ({ + getPackageTypeMock: vi.fn(), + buildCommandMock: vi.fn(), + hashPasses: { count: 0 } +})) vi.mock('./linux-update-package-type', () => ({ getLinuxRootPackageType: getPackageTypeMock })) @@ -67,9 +63,9 @@ async function writePackage(name: string, contents = PAYLOAD): Promise { } /** Captures a well-formed downloaded event unless a field is overridden. */ -function capture(overrides: Record = {}): void { +function capture(overrides: Record = {}): LinuxPackageArtifact | null { const downloadedFile = (overrides.downloadedFile ?? path.join(downloadDir, 'orca.deb')) as string - recovery.captureLinuxPackageArtifact({ + return recovery.captureLinuxPackageArtifact({ version: VERSION, files: [{ url: path.basename(downloadedFile), sha512: SHA512 }], ...overrides, @@ -80,7 +76,6 @@ function capture(overrides: Record = {}): void { beforeEach(async () => { vi.resetModules() hashPasses.count = 0 - showItemInFolderMock.mockReset() getPackageTypeMock.mockReset().mockReturnValue('deb') buildCommandMock.mockReset().mockReturnValue({ ok: true, command: 'installed command' }) tempRoot = await fsp.mkdtemp(path.join(os.tmpdir(), 'orca-recovery-')) @@ -103,13 +98,14 @@ afterEach(async () => { describe('captureLinuxPackageArtifact', () => { it('retains the downloaded package with the digest from the event metadata', () => { - capture() - expect(recovery.getTrackedLinuxPackageArtifact()).toEqual({ + const artifact = { packageType: 'deb', version: VERSION, path: path.join(downloadDir, 'orca.deb'), sha512: SHA512 - }) + } satisfies LinuxPackageArtifact + expect(capture()).toEqual(artifact) + expect(recovery.getTrackedLinuxPackageArtifact()).toEqual(artifact) }) it('ignores the event on a build that is not a root package', () => { @@ -221,7 +217,7 @@ describe('captureLinuxPackageArtifact', () => { it('keeps a retained artifact when a later event carries a malformed digest', () => { capture() - capture({ files: [{ url: 'orca.deb', sha512: 'not-a-digest' }] }) + expect(capture({ files: [{ url: 'orca.deb', sha512: 'not-a-digest' }] })).toBeNull() expect(recovery.getTrackedLinuxPackageArtifact()?.sha512).toBe(SHA512) }) @@ -596,7 +592,7 @@ describePosix('validation coalescing', () => { capture() const [first, second] = await Promise.all([ recovery.resolveLinuxPackageInstallInstructions(recoveryFor()), - recovery.revealLinuxPackage(recoveryFor()) + recovery.resolveLinuxPackageRevealTarget(recoveryFor()) ]) expect(first.ok).toBe(true) expect(second.ok).toBe(true) @@ -611,31 +607,6 @@ describePosix('validation coalescing', () => { expect(hashPasses.count).toBe(2) }) - // Why: a verdict handed to a root package manager must cover the bytes as of the click, not the - // bytes a Copy click started streaming seconds earlier. - it('never reuses an in-flight pass for a pre-install re-proof', async () => { - await writePackage('orca.deb') - capture() - const artifact = recovery.getTrackedLinuxPackageArtifact() - const copyPass = recovery.resolveLinuxPackageInstallInstructions(recoveryFor()) - const installPass = recovery.revalidateLinuxPackageForInstall(artifact!) - - await expect(installPass).resolves.toEqual({ ok: true }) - await expect(copyPass).resolves.toMatchObject({ ok: true }) - expect(hashPasses.count).toBe(2) - }) - - it('lets a later Copy click join the pre-install pass', async () => { - await writePackage('orca.deb') - capture() - const artifact = recovery.getTrackedLinuxPackageArtifact() - const installPass = recovery.revalidateLinuxPackageForInstall(artifact!) - const copyPass = recovery.resolveLinuxPackageInstallInstructions(recoveryFor()) - - await Promise.all([installPass, copyPass]) - expect(hashPasses.count).toBe(1) - }) - it('does not reuse an in-flight pass for a different artifact', async () => { await writePackage('orca.deb') await writePackage('orca-next.deb') @@ -652,77 +623,43 @@ describePosix('validation coalescing', () => { await Promise.all([first, second]) expect(hashPasses.count).toBe(2) }) -}) -describePosix('revalidateLinuxPackageForInstall', () => { - it('proves the retained package still matches its release digest', async () => { + it('starts a fresh proof when the same package is captured again', async () => { await writePackage('orca.deb') capture() - const artifact = recovery.getTrackedLinuxPackageArtifact() - await expect(recovery.revalidateLinuxPackageForInstall(artifact!)).resolves.toEqual({ - ok: true - }) - }) - - it('rejects a package swapped after the download was verified', async () => { - await writePackage('orca.deb') + const first = recovery.resolveLinuxPackageInstallInstructions(recoveryFor()) capture() - const artifact = recovery.getTrackedLinuxPackageArtifact() - await writePackage('orca.deb', 'attacker supplied package') - await expect(recovery.revalidateLinuxPackageForInstall(artifact!)).resolves.toEqual({ - ok: false, - reason: 'hash-mismatch' - }) - }) + const second = recovery.resolveLinuxPackageInstallInstructions(recoveryFor()) - it('reports a package deleted from the cache as missing', async () => { - const filePath = await writePackage('orca.deb') - capture() - const artifact = recovery.getTrackedLinuxPackageArtifact() - await fsp.rm(filePath) - await expect(recovery.revalidateLinuxPackageForInstall(artifact!)).resolves.toEqual({ - ok: false, - reason: 'missing' - }) + await Promise.all([first, second]) + expect(hashPasses.count).toBe(2) }) }) -describePosix('revealLinuxPackage', () => { - it('reveals a verified package on the machine that owns it', async () => { +describePosix('resolveLinuxPackageRevealTarget', () => { + it('returns the verified package path', async () => { const filePath = await writePackage('orca.deb') capture() - await expect(recovery.revealLinuxPackage(recoveryFor())).resolves.toEqual({ ok: true }) - expect(showItemInFolderMock).toHaveBeenCalledWith(filePath) + await expect(recovery.resolveLinuxPackageRevealTarget(recoveryFor())).resolves.toEqual({ + ok: true, + path: filePath + }) }) - it('does not reveal a package that fails validation', async () => { + it('rejects a package that fails validation', async () => { const filePath = await writePackage('orca.deb') capture() await fsp.writeFile(filePath, 'tampered payload') - await expect(recovery.revealLinuxPackage(recoveryFor())).resolves.toEqual({ + await expect(recovery.resolveLinuxPackageRevealTarget(recoveryFor())).resolves.toEqual({ ok: false, reason: 'hash-mismatch' }) - expect(showItemInFolderMock).not.toHaveBeenCalled() }) - it('reports read-failed when the desktop file manager throws', async () => { - await writePackage('orca.deb') - capture() - showItemInFolderMock.mockImplementation(() => { - throw new Error('no file manager available') - }) - await expect(recovery.revealLinuxPackage(recoveryFor())).resolves.toEqual({ - ok: false, - reason: 'read-failed' - }) - }) - - it('does not reveal anything without a retained artifact', async () => { - await expect(recovery.revealLinuxPackage(recoveryFor())).resolves.toEqual({ + it('returns missing without a retained artifact', async () => { + await expect(recovery.resolveLinuxPackageRevealTarget(recoveryFor())).resolves.toEqual({ ok: false, reason: 'missing' }) - expect(showItemInFolderMock).not.toHaveBeenCalled() }) }) diff --git a/src/main/linux-package-update-recovery.ts b/src/main/linux-package-update-recovery.ts index e5269a2077b..e0bf530bb65 100644 --- a/src/main/linux-package-update-recovery.ts +++ b/src/main/linux-package-update-recovery.ts @@ -3,7 +3,6 @@ import { createReadStream } from 'node:fs' import fsp from 'node:fs/promises' import os from 'node:os' import path from 'node:path' -import { shell } from 'electron' import type { LinuxPackageInstallRecovery, LinuxRootPackageType @@ -36,7 +35,7 @@ export type LinuxPackageInstructionsResult = | { ok: false; reason: LinuxPackageRecoveryUnavailableReason } export type LinuxPackageRevealResult = - | { ok: true } + | { ok: true; path: string } | { ok: false; reason: LinuxPackageRecoveryUnavailableReason } type ValidationResult = @@ -45,7 +44,7 @@ type ValidationResult = let trackedArtifact: LinuxPackageArtifact | null = null // Why: the renderer debounces clicks, but the IPC boundary must not allow parallel hashing of a 160 MB package. -let inFlightValidation: { key: string; promise: Promise } | null = null +const inFlightValidations = new WeakMap>() export function getTrackedLinuxPackageArtifact(): LinuxPackageArtifact | null { return trackedArtifact @@ -117,24 +116,24 @@ function resolveExpectedSha512( } /** - * Retains the verified download so a failed root-package install stays recoverable without paying - * for the 160 MB transfer again. Only the in-memory event metadata is trusted for the digest. + * Retains the downloaded package and its release digest so manual actions do not repeat the 160 MB + * transfer. Only the in-memory event metadata is trusted for the digest. */ -export function captureLinuxPackageArtifact(event: unknown): void { +export function captureLinuxPackageArtifact(event: unknown): LinuxPackageArtifact | null { const packageType = getLinuxRootPackageType() if (!packageType) { - return + return null } const downloadedFile = (event as { downloadedFile?: unknown })?.downloadedFile const version = (event as { version?: unknown })?.version if (typeof downloadedFile !== 'string' || !path.isAbsolute(downloadedFile)) { - return + return null } if (!downloadedFile.toLowerCase().endsWith(`.${packageType}`)) { - return + return null } if (typeof version !== 'string' || version.length === 0) { - return + return null } const sha512 = resolveExpectedSha512( (event as { files?: unknown })?.files, @@ -147,9 +146,11 @@ export function captureLinuxPackageArtifact(event: unknown): void { // Why: an unresolvable digest only means THIS event cannot arm recovery. A previously retained // artifact carries its own digest and is revalidated on every use, so dropping it would force a // needless 160 MB redownload of a file that is still on disk and still verifiable. - return + return null } - trackedArtifact = { packageType, version, path: downloadedFile, sha512 } + const artifact = { packageType, version, path: downloadedFile, sha512 } + trackedArtifact = artifact + return artifact } function isInsideDirectory(root: string, target: string): boolean { @@ -244,26 +245,18 @@ async function validateArtifact(artifact: LinuxPackageArtifact): Promise { - const key = `${artifact.packageType}:${artifact.version}:${artifact.path}:${artifact.sha512}` - if (!options?.fresh && inFlightValidation?.key === key) { - return inFlightValidation.promise +/** Hashes the exact captured artifact, joining only that capture's in-flight proof. */ +function runValidation(artifact: LinuxPackageArtifact): Promise { + const inFlight = inFlightValidations.get(artifact) + if (inFlight) { + return inFlight } const promise: Promise = validateArtifact(artifact).finally(() => { - // Why: identity, not key — a fresh install pass may already have replaced this entry. - if (inFlightValidation?.promise === promise) { - inFlightValidation = null + if (inFlightValidations.get(artifact) === promise) { + inFlightValidations.delete(artifact) } }) - inFlightValidation = { key, promise } + inFlightValidations.set(artifact, promise) return promise } @@ -305,38 +298,12 @@ export async function resolveLinuxPackageInstallInstructions( } } -/** - * Re-proves the retained package immediately before the privileged installer consumes it. - * - * The cache path is user-writable, so a digest checked when the download finished says nothing - * about the bytes `dpkg -i` will read minutes later. Re-hashing here does not close the race — - * only an immutable handoff would — but it shrinks the window from "since the download" to - * "since this call", and it catches the artifact being swapped or deleted outright. Takes the - * artifact rather than a recovery so both the retry and the plain "Restart to Update" install - * are covered. - */ -export async function revalidateLinuxPackageForInstall( - artifact: LinuxPackageArtifact -): Promise<{ ok: true } | { ok: false; reason: LinuxPackageRecoveryUnavailableReason }> { - const validation = await runValidation(artifact, { fresh: true }) - return validation.ok ? { ok: true } : { ok: false, reason: validation.reason } -} - -export async function revealLinuxPackage( +export async function resolveLinuxPackageRevealTarget( recovery: LinuxPackageInstallRecovery ): Promise { const validation = await validateTrackedArtifact(recovery) if (!validation.ok) { return validation } - // Why: this cache path must not travel through the workspace shell:openPath API, whose execution - // host can be an SSH or WSL machine rather than the one that owns the installed package. - try { - shell.showItemInFolder(validation.artifact.path) - } catch { - // Why: every other failure in this module reports through {ok:false}; a raw throw here would - // reject the IPC with an unredacted message and skip the lifecycle record. - return { ok: false, reason: 'read-failed' } - } - return { ok: true } + return { ok: true, path: validation.artifact.path } } diff --git a/src/main/linux-update-package-type.test.ts b/src/main/linux-update-package-type.test.ts index 2a2858c2c68..7453c727bb8 100644 --- a/src/main/linux-update-package-type.test.ts +++ b/src/main/linux-update-package-type.test.ts @@ -2,16 +2,38 @@ import fsp from 'node:fs/promises' import os from 'node:os' import path from 'node:path' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { LinuxPackageType, LinuxRootPackageType } from './linux-update-package-type' -const { appMock } = vi.hoisted(() => ({ appMock: { isPackaged: true } })) +const { appMock, hasTrustedPackageManagerForMock } = vi.hoisted(() => ({ + appMock: { isPackaged: true }, + hasTrustedPackageManagerForMock: vi.fn(() => true) +})) vi.mock('electron', () => ({ app: appMock })) +vi.mock('./linux-package-install-command', () => ({ + hasTrustedPackageManagerFor: hasTrustedPackageManagerForMock +})) const originalPlatform = process.platform const originalResourcesPath = process.resourcesPath as string | undefined +const originalExecPath = process.execPath +const originalAppImage = process.env.APPIMAGE +const originalAppDir = process.env.APPDIR let resourcesDir: string +type PackageTypeModule = { + getLinuxPackageType: () => LinuxPackageType + getLinuxRootPackageType: () => LinuxRootPackageType | null + isExternallyManagedLinuxInstall: () => boolean + isLegacyAppImageRuntimeIdentity: (identity: { + appImagePath: unknown + appDirPath: unknown + execPath: unknown + resourcesPath: unknown + }) => boolean +} + function setPlatform(platform: string): void { Object.defineProperty(process, 'platform', { configurable: true, value: platform }) } @@ -20,19 +42,25 @@ function setResourcesPath(value: unknown): void { Object.defineProperty(process, 'resourcesPath', { configurable: true, value }) } +function setExecPath(value: string): void { + Object.defineProperty(process, 'execPath', { configurable: true, value }) +} + async function writeMarker(contents: string): Promise { await fsp.writeFile(path.join(resourcesDir, 'package-type'), contents, 'utf8') } -async function loadPackageType(): Promise<() => 'deb' | 'rpm' | null> { - const module = await import('./linux-update-package-type') - return module.getLinuxRootPackageType +async function loadPackageType(): Promise { + return import('./linux-update-package-type') } beforeEach(async () => { vi.resetModules() + hasTrustedPackageManagerForMock.mockReset().mockReturnValue(true) vi.spyOn(console, 'warn').mockImplementation(() => {}) appMock.isPackaged = true + delete process.env.APPIMAGE + delete process.env.APPDIR setPlatform('linux') resourcesDir = await fsp.mkdtemp(path.join(os.tmpdir(), 'orca-package-type-')) setResourcesPath(resourcesDir) @@ -42,140 +70,297 @@ afterEach(async () => { vi.restoreAllMocks() setPlatform(originalPlatform) setResourcesPath(originalResourcesPath) + setExecPath(originalExecPath) + if (originalAppImage === undefined) { + delete process.env.APPIMAGE + } else { + process.env.APPIMAGE = originalAppImage + } + if (originalAppDir === undefined) { + delete process.env.APPDIR + } else { + process.env.APPDIR = originalAppDir + } await fsp.rm(resourcesDir, { recursive: true, force: true }) }) describe('getLinuxRootPackageType', () => { it('reads a deb marker', async () => { await writeMarker('deb') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBe('deb') + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('deb') + expect(module.getLinuxRootPackageType()).toBe('deb') }) it('reads an rpm marker', async () => { await writeMarker('rpm') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBe('rpm') + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('rpm') + expect(module.getLinuxRootPackageType()).toBe('rpm') }) it('trims surrounding whitespace', async () => { await writeMarker('\n rpm \t\n') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBe('rpm') + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('rpm') + expect(module.getLinuxRootPackageType()).toBe('rpm') }) it('treats the AppImage marker as not a root package', async () => { await writeMarker('AppImage') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('non-root') + expect(module.getLinuxRootPackageType()).toBeNull() }) - it('treats an unknown marker value as not a root package', async () => { + it('treats an unknown marker value as unusable', async () => { await writeMarker('snap') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() }) - it('treats a pacman marker as a recognized but unsupported target', async () => { + it('treats a pacman marker as unusable until recovery supports it', async () => { const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) await writeMarker('pacman') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() - expect(warn).not.toHaveBeenCalled() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() + expect(warn).toHaveBeenCalledTimes(1) }) it('rejects a marker that only differs by case', async () => { await writeMarker('DEB') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() }) - it('returns null when the marker is missing', async () => { - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + it('uses a legacy AppImage identity when its executable and resources are inside APPDIR', async () => { + process.env.APPIMAGE = '/opt/orca/orca.AppImage' + process.env.APPDIR = '/tmp/.mount_orca' + setExecPath('/tmp/.mount_orca/orca') + setResourcesPath('/tmp/.mount_orca/resources') + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('non-root') + expect(module.getLinuxRootPackageType()).toBeNull() }) - it('returns null when the marker is unreadable', async () => { + it.each([ + ['relative APPIMAGE', 'relative/orca.AppImage', '/tmp/.mount_orca'], + ['relative APPDIR', '/opt/orca/orca.AppImage', 'relative/.mount_orca'] + ])('rejects a legacy identity with %s', async (_label, appImagePath, appDirPath) => { + process.env.APPIMAGE = appImagePath + process.env.APPDIR = appDirPath + setExecPath('/tmp/.mount_orca/orca') + setResourcesPath('/tmp/.mount_orca/resources') + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() + }) + + it.each([ + ['executable', '/tmp/.mount_orca-shadow/orca', '/tmp/.mount_orca/resources'], + ['resources', '/tmp/.mount_orca/orca', '/tmp/.mount_orca-shadow/resources'] + ])( + 'rejects a prefix-collision outside APPDIR for %s', + async (_label, execPath, resourcesPath) => { + process.env.APPIMAGE = '/opt/orca/orca.AppImage' + process.env.APPDIR = '/tmp/.mount_orca' + setExecPath(execPath) + setResourcesPath(resourcesPath) + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() + } + ) + + it('rejects NULs in every legacy AppImage identity path', async () => { + const { isLegacyAppImageRuntimeIdentity } = await loadPackageType() + const identity = { + appImagePath: '/opt/orca/orca.AppImage', + appDirPath: '/tmp/.mount_orca', + execPath: '/tmp/.mount_orca/orca', + resourcesPath: '/tmp/.mount_orca/resources' + } + + for (const field of Object.keys(identity) as (keyof typeof identity)[]) { + expect( + isLegacyAppImageRuntimeIdentity({ ...identity, [field]: `${identity[field]}\0suffix` }) + ).toBe(false) + } + }) + + it('requires every legacy AppImage identity path to be absolute', async () => { + const { isLegacyAppImageRuntimeIdentity } = await loadPackageType() + const identity = { + appImagePath: '/opt/orca/orca.AppImage', + appDirPath: '/tmp/.mount_orca', + execPath: '/tmp/.mount_orca/orca', + resourcesPath: '/tmp/.mount_orca/resources' + } + + for (const field of Object.keys(identity) as (keyof typeof identity)[]) { + expect(isLegacyAppImageRuntimeIdentity({ ...identity, [field]: 'relative/path' })).toBe(false) + } + }) + + it('prefers a package marker over an invalid legacy AppImage identity', async () => { + await writeMarker('deb') + process.env.APPIMAGE = 'relative/orca.AppImage' + process.env.APPDIR = 'relative/.mount_orca' + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('deb') + expect(module.getLinuxRootPackageType()).toBe('deb') + }) + + it('returns unusable when the marker is missing without AppImage identity', async () => { + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() + }) + + it('returns unusable when the marker is unreadable', async () => { // A directory in the marker's place makes readFileSync fail with EISDIR. await fsp.mkdir(path.join(resourcesDir, 'package-type')) - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() }) - it('returns null when resourcesPath is unavailable', async () => { + it('returns unusable when resourcesPath is unavailable', async () => { setResourcesPath(undefined) - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() }) - it('returns null when resourcesPath is empty', async () => { + it('returns unusable when resourcesPath is empty', async () => { setResourcesPath('') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() }) it('ignores a readable marker in an unpackaged dev run', async () => { await writeMarker('deb') appMock.isPackaged = false - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('non-root') + expect(module.getLinuxRootPackageType()).toBeNull() }) it('ignores a readable marker off Linux', async () => { await writeMarker('deb') setPlatform('darwin') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('non-root') + expect(module.getLinuxRootPackageType()).toBeNull() }) it('caches the resolved type for the process lifetime', async () => { await writeMarker('deb') - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBe('deb') + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('deb') + expect(module.getLinuxRootPackageType()).toBe('deb') await writeMarker('rpm') - expect(getLinuxRootPackageType()).toBe('deb') + expect(module.getLinuxPackageType()).toBe('deb') + expect(module.getLinuxRootPackageType()).toBe('deb') }) - it('caches a resolved null so a later marker is not picked up', async () => { - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() + it('caches an unusable result so a later marker is not picked up', async () => { + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') await writeMarker('deb') - expect(getLinuxRootPackageType()).toBeNull() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() }) it('warns about an unknown marker value', async () => { const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) await writeMarker('snap') - const getLinuxRootPackageType = await loadPackageType() - getLinuxRootPackageType() + const module = await loadPackageType() + module.getLinuxPackageType() expect(warn).toHaveBeenCalledTimes(1) - expect(warn.mock.calls[0][0]).toContain('marker is not deb or rpm') + expect(warn.mock.calls[0][0]).toContain('marker is not AppImage, deb, or rpm') }) it('reads the marker once and warns once per process', async () => { const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) // A directory in place of the marker is readable-but-unusable, which is the case worth reporting. await fsp.mkdir(path.join(resourcesDir, 'package-type')) - const getLinuxRootPackageType = await loadPackageType() - getLinuxRootPackageType() - expect(getLinuxRootPackageType()).toBeNull() + const module = await loadPackageType() + module.getLinuxPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() expect(warn).toHaveBeenCalledTimes(1) expect(warn.mock.calls[0][0]).toContain('marker unreadable') }) - // Why: AppImage ships no marker at all, so the normal case must stay silent. - it('stays silent when no marker is present', async () => { + it('warns when a packaged marker is missing', async () => { const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) - const getLinuxRootPackageType = await loadPackageType() - expect(getLinuxRootPackageType()).toBeNull() - expect(warn).not.toHaveBeenCalled() + const module = await loadPackageType() + expect(module.getLinuxPackageType()).toBe('unusable') + expect(module.getLinuxRootPackageType()).toBeNull() + expect(warn).toHaveBeenCalledWith(expect.stringContaining('marker missing')) }) it('does not warn in a dev run', async () => { const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) appMock.isPackaged = false - const getLinuxRootPackageType = await loadPackageType() - getLinuxRootPackageType() + const module = await loadPackageType() + module.getLinuxPackageType() expect(warn).not.toHaveBeenCalled() }) }) + +describe('isExternallyManagedLinuxInstall', () => { + // The #17702 case: an AUR/Nix/container rebuild of the .deb inherits `package-type` verbatim. + it('reports a deb marker with no deb package manager as externally managed', async () => { + hasTrustedPackageManagerForMock.mockReturnValue(false) + await writeMarker('deb') + const module = await loadPackageType() + expect(module.isExternallyManagedLinuxInstall()).toBe(true) + expect(hasTrustedPackageManagerForMock).toHaveBeenCalledWith('deb') + }) + + it('reports an rpm marker with no rpm package manager as externally managed', async () => { + hasTrustedPackageManagerForMock.mockReturnValue(false) + await writeMarker('rpm') + const module = await loadPackageType() + expect(module.isExternallyManagedLinuxInstall()).toBe(true) + expect(hasTrustedPackageManagerForMock).toHaveBeenCalledWith('rpm') + }) + + it('leaves a real deb host self-updatable', async () => { + await writeMarker('deb') + const module = await loadPackageType() + expect(module.isExternallyManagedLinuxInstall()).toBe(false) + }) + + it('never probes the host for an AppImage install', async () => { + hasTrustedPackageManagerForMock.mockReturnValue(false) + await writeMarker('AppImage') + const module = await loadPackageType() + expect(module.isExternallyManagedLinuxInstall()).toBe(false) + expect(hasTrustedPackageManagerForMock).not.toHaveBeenCalled() + }) + + it('never probes the host for an unusable marker', async () => { + hasTrustedPackageManagerForMock.mockReturnValue(false) + await writeMarker('snap') + const module = await loadPackageType() + expect(module.isExternallyManagedLinuxInstall()).toBe(false) + expect(hasTrustedPackageManagerForMock).not.toHaveBeenCalled() + }) + + it('probes the host at most once per process', async () => { + hasTrustedPackageManagerForMock.mockReturnValue(false) + await writeMarker('deb') + const module = await loadPackageType() + module.isExternallyManagedLinuxInstall() + module.isExternallyManagedLinuxInstall() + module.isExternallyManagedLinuxInstall() + expect(hasTrustedPackageManagerForMock).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/linux-update-package-type.ts b/src/main/linux-update-package-type.ts index 5acc83a4cf3..f58c6cf16ad 100644 --- a/src/main/linux-update-package-type.ts +++ b/src/main/linux-update-package-type.ts @@ -1,57 +1,136 @@ import { readFileSync } from 'node:fs' import path from 'node:path' import { app } from 'electron' +import { hasTrustedPackageManagerFor } from './linux-package-install-command' import type { LinuxRootPackageType } from '../shared/update-status-types' export type { LinuxRootPackageType } -// Why: `undefined` means "not resolved yet"; `null` is a resolved "not a root package". -let cachedPackageType: LinuxRootPackageType | null | undefined +/** The packaged Linux format that controls how updates may be installed. */ +export type LinuxPackageType = LinuxRootPackageType | 'non-root' | 'unusable' + +// Why: `undefined` means "not resolved yet"; every other value is stable for this process. +let cachedPackageType: LinuxPackageType | undefined +let cachedExternallyManaged: boolean | undefined // Bounded by construction: the marker is read at most once per process. function warnMarkerUnusable(detail: string): void { console.warn(`[updater] linux package-type marker unusable: ${detail}`) } -function readPackageTypeMarker(): LinuxRootPackageType | null { +function isAbsolutePathString(value: unknown): value is string { + return ( + typeof value === 'string' && value.length > 0 && !value.includes('\0') && path.isAbsolute(value) + ) +} + +function isInsideDirectory(root: string, candidate: string): boolean { + const relative = path.relative(root, candidate) + return ( + relative.length > 0 && + relative !== '..' && + !relative.startsWith(`..${path.sep}`) && + !path.isAbsolute(relative) + ) +} + +export function isLegacyAppImageRuntimeIdentity(identity: { + appImagePath: unknown + appDirPath: unknown + execPath: unknown + resourcesPath: unknown +}): boolean { + if ( + !isAbsolutePathString(identity.appImagePath) || + !isAbsolutePathString(identity.appDirPath) || + !isAbsolutePathString(identity.execPath) || + !isAbsolutePathString(identity.resourcesPath) + ) { + return false + } + return ( + isInsideDirectory(identity.appDirPath, identity.execPath) && + isInsideDirectory(identity.appDirPath, identity.resourcesPath) + ) +} + +function hasLegacyAppImageRuntimeIdentity(resourcesPath: unknown): boolean { + return isLegacyAppImageRuntimeIdentity({ + appImagePath: process.env.APPIMAGE, + appDirPath: process.env.APPDIR, + execPath: process.execPath, + resourcesPath + }) +} + +function readPackageTypeMarker(): LinuxPackageType { if (process.platform !== 'linux' || !app.isPackaged) { - return null + return 'non-root' } const resourcesPath = process.resourcesPath if (typeof resourcesPath !== 'string' || resourcesPath.length === 0) { warnMarkerUnusable('resourcesPath unavailable') - return null + return 'unusable' } let raw: string try { raw = readFileSync(path.join(resourcesPath, 'package-type'), 'utf8') } catch (error) { - // Why: AppImage legitimately ships no marker, so only an unreadable one is worth reporting. - if ((error as NodeJS.ErrnoException)?.code !== 'ENOENT') { - warnMarkerUnusable('marker unreadable') + if ( + (error as NodeJS.ErrnoException)?.code === 'ENOENT' && + hasLegacyAppImageRuntimeIdentity(resourcesPath) + ) { + return 'non-root' } - return null + warnMarkerUnusable( + (error as NodeJS.ErrnoException)?.code === 'ENOENT' ? 'marker missing' : 'marker unreadable' + ) + return 'unusable' } const value = raw.trim() if (value === 'deb' || value === 'rpm') { return value } - // Why: electron-updater also supports pacman, but this recovery path covers deb/rpm only — a - // recognized marker is a deliberate scope cut, not a broken install. - if (value !== 'pacman') { - warnMarkerUnusable('marker is not deb or rpm') + if (value === 'AppImage') { + return 'non-root' } - return null + warnMarkerUnusable('marker is not AppImage, deb, or rpm') + return 'unusable' } /** - * The installed Linux package format, or null when this build does not install through a - * root package. Reads only the packaged marker `electron-updater` itself uses — never distro - * files, executable paths, or available package managers. + * Resolves the installed Linux package format. Packaged builds fail closed when their marker is + * missing or unusable unless the legacy APPIMAGE runtime identity is valid. Unpackaged, non-Linux, + * and identified AppImage runs are non-root. */ -export function getLinuxRootPackageType(): LinuxRootPackageType | null { +export function getLinuxPackageType(): LinuxPackageType { if (cachedPackageType === undefined) { cachedPackageType = readPackageTypeMarker() } return cachedPackageType } + +/** Returns a root package type when this build supports manual package recovery. */ +export function getLinuxRootPackageType(): LinuxRootPackageType | null { + const packageType = getLinuxPackageType() + return packageType === 'deb' || packageType === 'rpm' ? packageType : null +} + +/** + * Whether the marker claims a root package format this host cannot install. Repackagers (AUR, Nix, + * container rebuilds) unpack Orca's .deb and inherit its `package-type` verbatim, so the marker + * describes the artifact Orca was built as, never the system that now owns the install. Without a + * matching package manager no downloaded package can ever be applied here. + * + * A false positive is impossible by construction: this reuses the exact manager lists and resolver + * that `buildLinuxPackageInstallCommand` already loops over, so any host flagged here would have + * failed with `no-package-manager` after the download anyway. The gate only moves that verdict + * earlier — it never refuses a host that could have installed the update. + */ +export function isExternallyManagedLinuxInstall(): boolean { + if (cachedExternallyManaged === undefined) { + const packageType = getLinuxRootPackageType() + cachedExternallyManaged = packageType !== null && !hasTrustedPackageManagerFor(packageType) + } + return cachedExternallyManaged +} diff --git a/src/main/startup/appimage-cli-redirect.test.ts b/src/main/startup/appimage-cli-redirect.test.ts deleted file mode 100644 index 99d5024a517..00000000000 --- a/src/main/startup/appimage-cli-redirect.test.ts +++ /dev/null @@ -1,211 +0,0 @@ -import { mkdir, mkdtemp, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { describe, expect, it, vi } from 'vitest' -import { getAppImageCliArgs, maybeRedirectAppImageCliLaunch } from './appimage-cli-redirect' - -const commandNames = ['serve', 'status', 'terminal'] - -describe('AppImage CLI redirect', () => { - it('detects direct AppImage CLI commands', () => { - expect( - getAppImageCliArgs( - ['orca-linux.AppImage', 'status', '--json'], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toEqual(['status', '--json']) - }) - - it('allows CLI global flags before the command', () => { - expect( - getAppImageCliArgs( - ['orca-linux.AppImage', '--pairing-code', 'abc123', '--json', 'terminal', 'list'], - { - APPIMAGE: '/opt/orca' - }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toEqual(['--pairing-code', 'abc123', '--json', 'terminal', 'list']) - }) - - it('does not redirect normal desktop AppImage launches', () => { - expect( - getAppImageCliArgs( - ['AppRun', '--no-sandbox', 'file:///tmp/example.txt'], - { - APPIMAGE: '/opt/orca' - }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toBeNull() - }) - - it('keeps direct serve launches in Electron when Chromium switches are present', () => { - expect( - getAppImageCliArgs( - [ - 'AppRun', - '--no-sandbox', - '--disable-features=FedCm,DirectSockets', - 'serve', - '--port', - '6768' - ], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toBeNull() - }) - - it('keeps clean serve launches on the CLI path for validation', () => { - expect( - getAppImageCliArgs( - ['AppRun', '--no-sandbox', 'serve', '--port', '6768'], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toEqual(['serve', '--port', '6768']) - }) - - it('handles a space-separated Chromium switch value', () => { - expect( - getAppImageCliArgs( - ['AppRun', '--disable-features', 'FedCm', 'serve'], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toBeNull() - }) - - it('does not broaden the Electron-owned exception to switches after serve', () => { - expect( - getAppImageCliArgs( - ['AppRun', 'serve', '--disable-features=FedCm'], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toEqual(['serve', '--disable-features=FedCm']) - }) - - it('removes no-sandbox before forwarding CLI help', () => { - expect( - getAppImageCliArgs( - ['AppRun', '--no-sandbox', 'serve', '--help'], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toEqual(['serve', '--help']) - }) - - it('still redirects serve help even when Chromium switches are present', () => { - expect( - getAppImageCliArgs( - ['AppRun', '--disable-features=FedCm', 'serve', '--help'], - { APPIMAGE: '/opt/orca' }, - { - platform: 'linux', - isPackaged: true, - commandNames - } - ) - ).toEqual(['--disable-features=FedCm', 'serve', '--help']) - }) - - it('spawns the unpacked CLI entrypoint with Electron node mode', async () => { - const root = await mkdtemp(join(tmpdir(), 'orca-appimage-cli-redirect-')) - const cliEntryPath = join(root, 'app.asar.unpacked', 'out', 'cli', 'index.js') - await mkdir(join(root, 'app.asar.unpacked', 'out', 'cli'), { recursive: true }) - await writeFile(cliEntryPath, '', 'utf8') - const spawn = vi.fn((..._args: unknown[]) => ({ status: 0 })) - - const result = maybeRedirectAppImageCliLaunch({ - argv: ['orca-linux.AppImage', 'status', '--json'], - env: { - APPIMAGE: '/opt/orca/orca-linux.AppImage', - NODE_OPTIONS: '--inspect', - NODE_REPL_EXTERNAL_MODULE: '/tmp/repl.js' - }, - platform: 'linux', - isPackaged: true, - resourcesPath: root, - execPath: '/opt/orca/orca-ide', - commandNames, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: true, status: 0 }) - expect(spawn).toHaveBeenCalledWith('/opt/orca/orca-ide', [cliEntryPath, 'status', '--json'], { - env: expect.objectContaining({ - APPIMAGE: '/opt/orca/orca-linux.AppImage', - ELECTRON_RUN_AS_NODE: '1', - ORCA_NODE_OPTIONS: '--inspect', - ORCA_NODE_REPL_EXTERNAL_MODULE: '/tmp/repl.js' - }), - stdio: 'inherit' - }) - const spawnOptions = spawn.mock.calls[0]?.[2] as { env: NodeJS.ProcessEnv } | undefined - expect(spawnOptions?.env).not.toHaveProperty('NODE_OPTIONS') - expect(spawnOptions?.env).not.toHaveProperty('NODE_REPL_EXTERNAL_MODULE') - }) - - it('keeps a clean no-sandbox serve launch on the CLI path', async () => { - const root = await mkdtemp(join(tmpdir(), 'orca-appimage-cli-redirect-')) - const cliEntryPath = join(root, 'app.asar.unpacked', 'out', 'cli', 'index.js') - await mkdir(join(root, 'app.asar.unpacked', 'out', 'cli'), { recursive: true }) - await writeFile(cliEntryPath, '', 'utf8') - const spawn = vi.fn((..._args: unknown[]) => ({ status: 0 })) - - const result = maybeRedirectAppImageCliLaunch({ - argv: ['orca-linux.AppImage', '--no-sandbox', 'serve'], - env: { APPIMAGE: '/opt/orca/orca-linux.AppImage' }, - platform: 'linux', - isPackaged: true, - resourcesPath: root, - execPath: '/opt/orca/orca-ide', - commandNames, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: true, status: 0 }) - expect(spawn).toHaveBeenCalledWith( - '/opt/orca/orca-ide', - [cliEntryPath, 'serve'], - expect.objectContaining({ - env: expect.objectContaining({ ORCA_APPIMAGE_NO_SANDBOX: '1' }) - }) - ) - }) -}) diff --git a/src/main/startup/appimage-cli-redirect.ts b/src/main/startup/appimage-cli-redirect.ts deleted file mode 100644 index 2ca5f54e155..00000000000 --- a/src/main/startup/appimage-cli-redirect.ts +++ /dev/null @@ -1,209 +0,0 @@ -import { spawnSync, type SpawnSyncReturns } from 'node:child_process' -import { existsSync } from 'node:fs' -import { join } from 'node:path' - -type RedirectResult = - | { - redirected: false - } - | { - redirected: true - status: number - } - -type RedirectOptions = { - argv?: string[] - env?: NodeJS.ProcessEnv - platform?: NodeJS.Platform - isPackaged?: boolean - resourcesPath?: string - execPath?: string - commandNames?: readonly string[] - spawn?: typeof spawnSync -} - -const HELP_FLAGS = new Set(['--help', '-h', 'help']) -const APPIMAGE_DESKTOP_FLAGS = new Set(['--no-sandbox']) -const ELECTRON_LAUNCH_SWITCHES = new Set(['--disable-features']) -const CLI_FLAGS_WITH_VALUES = new Set(['--environment', '--pairing-code', '--disable-features']) -// Why: the main tsconfig cannot import the CLI project, but AppImage direct -// launches need a conservative allow-list before bypassing the GUI startup. -const APPIMAGE_CLI_COMMAND_NAMES = [ - 'agent', - 'automations', - 'back', - 'capture', - 'check', - 'clear', - 'click', - 'clipboard', - 'computer', - 'console', - 'cookie', - 'dblclick', - 'dialog', - 'download', - 'drag', - 'environment', - 'eval', - 'exec', - 'file', - 'fill', - 'find', - 'focus', - 'forward', - 'full-screenshot', - 'geolocation', - 'get', - 'goto', - 'highlight', - 'hover', - 'inserttext', - 'intercept', - 'is', - 'keypress', - 'mouse', - 'network', - 'open', - 'orchestration', - 'pdf', - 'reload', - 'repo', - 'screenshot', - 'scroll', - 'scrollintoview', - 'select', - 'select-all', - 'serve', - 'set', - 'snapshot', - 'status', - 'storage', - 'tab', - 'terminal', - 'type', - 'uncheck', - 'upload', - 'viewport', - 'wait', - 'worktree' -] - -export function maybeRedirectAppImageCliLaunch(options: RedirectOptions = {}): RedirectResult { - const argv = options.argv ?? process.argv - const env = options.env ?? process.env - const platform = options.platform ?? process.platform - const isPackaged = options.isPackaged ?? false - const resourcesPath = options.resourcesPath ?? process.resourcesPath - const execPath = options.execPath ?? process.execPath - const spawn = options.spawn ?? spawnSync - const cliArgs = getAppImageCliArgs(argv, env, { - platform, - isPackaged, - commandNames: options.commandNames ?? APPIMAGE_CLI_COMMAND_NAMES - }) - - if (!cliArgs) { - return { redirected: false } - } - - const cliEntryPath = join(resourcesPath, 'app.asar.unpacked', 'out', 'cli', 'index.js') - if (!existsSync(cliEntryPath)) { - process.stderr.write(`Unable to locate the Orca CLI entrypoint at ${cliEntryPath}\n`) - return { redirected: true, status: 1 } - } - - const childEnv = buildElectronRunAsNodeEnv(env) - if (argv.slice(1).includes('--no-sandbox')) { - // Why: the operator explicitly disabled Chromium's sandbox; preserve that choice when `serve` launches the Electron child. - childEnv.ORCA_APPIMAGE_NO_SANDBOX = '1' - } - const result = spawn(execPath, [cliEntryPath, ...cliArgs], { - env: childEnv, - stdio: 'inherit' - }) as SpawnSyncReturns - - if (result.error) { - process.stderr.write(`${result.error.message}\n`) - return { redirected: true, status: 1 } - } - - return { redirected: true, status: result.status ?? 1 } -} - -export function getAppImageCliArgs( - argv: string[], - env: NodeJS.ProcessEnv, - options: { - platform: NodeJS.Platform - isPackaged: boolean - commandNames: readonly string[] - } -): string[] | null { - if (options.platform !== 'linux' || !options.isPackaged) { - return null - } - if (!env.APPIMAGE && !env.APPDIR) { - return null - } - - const args = argv.slice(1) - if (args.length === 0) { - return null - } - const cliArgs = args.filter((arg) => !APPIMAGE_DESKTOP_FLAGS.has(arg)) - if (cliArgs.some((arg) => HELP_FLAGS.has(arg))) { - return cliArgs - } - - const commandNames = new Set(options.commandNames) - const firstPositional = findFirstCommandCandidate(cliArgs) - if (!firstPositional || !commandNames.has(firstPositional)) { - return null - } - // Keep serve in Electron only when an Electron launch switch is present. - // Forwarding it to the strict Node-mode CLI parser makes an otherwise valid - // serve launch fail, while clean serve invocations retain CLI validation. - if ( - firstPositional === 'serve' && - cliArgs - .slice(0, findFirstCommandCandidateIndex(cliArgs)) - .some((arg) => ELECTRON_LAUNCH_SWITCHES.has(flagName(arg))) - ) { - return null - } - return cliArgs -} - -function findFirstCommandCandidate(args: string[]): string | null { - const index = findFirstCommandCandidateIndex(args) - return index === -1 ? null : args[index]! -} - -function findFirstCommandCandidateIndex(args: string[]): number { - for (let index = 0; index < args.length; index += 1) { - const arg = args[index] - if (!arg.startsWith('-')) { - return index - } - if (CLI_FLAGS_WITH_VALUES.has(flagName(arg)) && !arg.includes('=')) { - index += 1 - } - } - return -1 -} - -function flagName(arg: string): string { - const equalsIndex = arg.indexOf('=') - return equalsIndex === -1 ? arg : arg.slice(0, equalsIndex) -} - -function buildElectronRunAsNodeEnv(env: NodeJS.ProcessEnv): NodeJS.ProcessEnv { - const childEnv = { ...env } - childEnv.ORCA_NODE_OPTIONS = env.NODE_OPTIONS ?? '' - childEnv.ORCA_NODE_REPL_EXTERNAL_MODULE = env.NODE_REPL_EXTERNAL_MODULE ?? '' - childEnv.ELECTRON_RUN_AS_NODE = '1' - delete childEnv.NODE_OPTIONS - delete childEnv.NODE_REPL_EXTERNAL_MODULE - return childEnv -} diff --git a/src/main/startup/cli-command-names.ts b/src/main/startup/cli-command-names.ts new file mode 100644 index 00000000000..2f1b0e4394d --- /dev/null +++ b/src/main/startup/cli-command-names.ts @@ -0,0 +1,73 @@ +// Kept import-free for main/CLI project isolation; a parity test prevents drift. +export const CLI_COMMAND_NAMES = [ + 'account', + 'agent', + 'agent-context', + 'artifacts', + 'automations', + 'back', + 'capture', + 'check', + 'claude-teams', + 'clear', + 'click', + 'clipboard', + 'computer', + 'console', + 'cookie', + 'dblclick', + 'diagnostics', + 'dialog', + 'download', + 'drag', + 'emulator', + 'environment', + 'eval', + 'exec', + 'file', + 'fill', + 'find', + 'focus', + 'forward', + 'full-screenshot', + 'geolocation', + 'get', + 'goto', + 'highlight', + 'host', + 'hover', + 'inserttext', + 'intercept', + 'is', + 'keypress', + 'linear', + 'mouse', + 'network', + 'open', + 'open-url', + 'orchestration', + 'pdf', + 'project', + 'reload', + 'repo', + 'screenshot', + 'scroll', + 'scrollintoview', + 'select', + 'select-all', + 'serve', + 'set', + 'skills', + 'snapshot', + 'status', + 'storage', + 'tab', + 'terminal', + 'type', + 'uncheck', + 'upload', + 'viewport', + 'vm', + 'wait', + 'worktree' +] as const diff --git a/src/main/startup/cli-launch-redirect.test.ts b/src/main/startup/cli-launch-redirect.test.ts new file mode 100644 index 00000000000..7a4b29fe60e --- /dev/null +++ b/src/main/startup/cli-launch-redirect.test.ts @@ -0,0 +1,357 @@ +import { posix, win32 } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { getCliLaunchArgs, maybeRedirectCliLaunch } from './cli-launch-redirect' + +const COMMAND_NAMES = ['project', 'serve', 'status', 'skills', 'worktree'] + +const linux = { + resourcesPath: '/opt/Orca/resources', + execPath: '/opt/Orca/orca-ide', + get cliEntryPath(): string { + return posix.join(this.resourcesPath, 'app.asar.unpacked', 'out', 'cli', 'index.js') + } +} +const windows = { + resourcesPath: 'C:\\Users\\me\\AppData\\Local\\Programs\\Orca\\resources', + execPath: 'C:\\Users\\me\\AppData\\Local\\Programs\\Orca\\Orca.exe', + get cliEntryPath(): string { + return win32.join(this.resourcesPath, 'app.asar.unpacked', 'out', 'cli', 'index.js') + } +} + +const linuxOptions = { platform: 'linux' as const, isPackaged: true, commandNames: COMMAND_NAMES } +const windowsOptions = { platform: 'win32' as const, isPackaged: true, commandNames: COMMAND_NAMES } + +describe('CLI launch redirect: entry-path form', () => { + it('detects a launch that received the unpacked CLI entrypoint', () => { + expect( + getCliLaunchArgs( + [windows.execPath, windows.cliEntryPath.toUpperCase(), 'status', '--json'], + windows.cliEntryPath, + windowsOptions + ) + ).toEqual(['status', '--json']) + }) + + it('ignores normal desktop launches', () => { + expect( + getCliLaunchArgs([windows.execPath, '--updated'], windows.cliEntryPath, windowsOptions) + ).toBeNull() + }) + + it('ignores the entrypoint when it is only the executable itself (argv[0])', () => { + expect( + getCliLaunchArgs([windows.cliEntryPath, 'status'], windows.cliEntryPath, windowsOptions) + ).toBeNull() + }) + + it('applies on Linux too', () => { + expect( + getCliLaunchArgs( + [linux.execPath, linux.cliEntryPath, 'status'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['status']) + }) + + it('strips injected Chromium switches before node-mode CLI arguments', () => { + expect( + getCliLaunchArgs( + [ + linux.execPath, + linux.cliEntryPath, + '--no-sandbox', + '--disable-gpu', + '--disable-features=Vulkan', + 'status', + '--json' + ], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['status', '--json']) + expect( + getCliLaunchArgs( + [linux.execPath, linux.cliEntryPath, '--disable-features', 'Vulkan', 'skills', 'get'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['skills', 'get']) + }) + + it('keeps user flags after the command and malformed boolean assignments', () => { + expect( + getCliLaunchArgs( + [linux.execPath, linux.cliEntryPath, 'status', '--disable-features=Vulkan'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['status', '--disable-features=Vulkan']) + expect( + getCliLaunchArgs( + [linux.execPath, linux.cliEntryPath, '--no-sandbox=true', 'status'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['--no-sandbox=true', 'status']) + }) + + it('does not treat a later positional entrypoint path as the launcher', () => { + expect( + getCliLaunchArgs( + [linux.execPath, 'file', 'open', '--path', linux.cliEntryPath], + linux.cliEntryPath, + linuxOptions + ) + ).toBeNull() + }) +}) + +describe('CLI launch redirect: command form', () => { + it('redirects a direct binary launch with no AppImage env at all', () => { + expect( + getCliLaunchArgs( + ['/home/u/.config/orca-runtime/versions/1.4.158/orca-ide', 'skills', 'get', '--full'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['skills', 'get', '--full']) + }) + + it('strips Chromium switches node mode would reject', () => { + expect( + getCliLaunchArgs( + [linux.execPath, '--no-sandbox', '--disable-gpu', 'status', '--json'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['status', '--json']) + }) + + it('preserves desktop-shaped switches after the command', () => { + expect( + getCliLaunchArgs( + [linux.execPath, 'skills', 'get', '--disable-gpu', '--no-sandbox'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['skills', 'get', '--disable-gpu', '--no-sandbox']) + }) + + it('leaves direct serve in-process but redirects its help', () => { + expect( + getCliLaunchArgs( + [linux.execPath, '--no-sandbox', 'serve', '--port', '6768'], + linux.cliEntryPath, + linuxOptions + ) + ).toBeNull() + expect( + getCliLaunchArgs( + [linux.execPath, '--no-sandbox', 'serve', '--help'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['serve', '--help']) + expect( + getCliLaunchArgs( + [linux.execPath, '--disable-features', 'Vulkan', 'serve', '--help'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['serve', '--help']) + }) + + it('treats help as a CLI launch even without a command', () => { + expect(getCliLaunchArgs([linux.execPath, '--help'], linux.cliEntryPath, linuxOptions)).toEqual([ + '--help' + ]) + }) + + it.each(['--version', '-v'])('treats %s as a CLI launch even without a command', (flag) => { + expect(getCliLaunchArgs([linux.execPath, flag], linux.cliEntryPath, linuxOptions)).toEqual([ + flag + ]) + }) + + it.each(['--user-data-dir', '--proxy-server', '--unknown-desktop-switch'])( + 'does not treat a value of %s as a CLI early-exit flag', + (flag) => { + expect( + getCliLaunchArgs([linux.execPath, flag, 'help'], linux.cliEntryPath, linuxOptions) + ).toBeNull() + } + ) + + it('does not treat a serve option value as a help request', () => { + expect( + getCliLaunchArgs( + [linux.execPath, 'serve', '--project-root', 'help'], + linux.cliEntryPath, + linuxOptions + ) + ).toBeNull() + }) + + it('does not reinterpret help after the argument terminator', () => { + expect( + getCliLaunchArgs([linux.execPath, 'serve', '--', '--help'], linux.cliEntryPath, linuxOptions) + ).toBeNull() + }) + + it('leaves a plain desktop launch alone', () => { + expect(getCliLaunchArgs([linux.execPath], linux.cliEntryPath, linuxOptions)).toBeNull() + expect( + getCliLaunchArgs([linux.execPath, '/home/u/project'], linux.cliEntryPath, linuxOptions) + ).toBeNull() + }) + + it('skips flag values when looking for the command positional', () => { + expect( + getCliLaunchArgs( + [linux.execPath, '--environment', 'status', 'worktree', 'ps'], + linux.cliEntryPath, + linuxOptions + ) + ).toEqual(['--environment', 'status', 'worktree', 'ps']) + expect( + getCliLaunchArgs( + [linux.execPath, '--environment', 'status'], + linux.cliEntryPath, + linuxOptions + ) + ).toBeNull() + }) + + it.each([ + ['--project', 'github:stablyai/orca', 'project', 'setups'], + ['--project=github:stablyai/orca', 'project', 'setups'], + ['--project', 'project', 'project', 'setups'], + ['--project=project', 'project', 'setups'] + ])('preserves a project selector in %j', (...args) => { + expect(getCliLaunchArgs([linux.execPath, ...args], linux.cliEntryPath, linuxOptions)).toEqual( + args + ) + }) + + it('does not apply the command form on macOS or Windows', () => { + for (const platform of ['darwin', 'win32'] as const) { + expect( + getCliLaunchArgs([linux.execPath, 'status'], linux.cliEntryPath, { + platform, + isPackaged: true, + commandNames: COMMAND_NAMES + }) + ).toBeNull() + } + }) + + it('never redirects an unpackaged build', () => { + expect( + getCliLaunchArgs([linux.execPath, 'status'], linux.cliEntryPath, { + ...linuxOptions, + isPackaged: false + }) + ).toBeNull() + }) +}) + +describe('CLI launch redirect: spawning', () => { + it('runs the in-package CLI in Electron node mode with sanitized env', () => { + const run = vi.fn((..._args: unknown[]) => ({ + code: 0, + signal: null, + stdout: '', + stderr: '', + timedOut: false + })) + + const result = maybeRedirectCliLaunch({ + argv: [linux.execPath, 'status', '--json'], + env: { NODE_OPTIONS: '--inspect', NODE_REPL_EXTERNAL_MODULE: 'external-loader' }, + platform: 'linux', + isPackaged: true, + resourcesPath: linux.resourcesPath, + execPath: linux.execPath, + commandNames: COMMAND_NAMES, + exists: () => true, + run: run as never + }) + + expect(result).toEqual({ redirected: true, status: 0 }) + expect(run).toHaveBeenCalledWith( + expect.objectContaining({ + program: linux.execPath, + args: [linux.cliEntryPath, 'status', '--json'], + stdio: 'inherit', + timeoutMs: null, + env: expect.objectContaining({ + ELECTRON_RUN_AS_NODE: '1', + ORCA_CLI_LAUNCH_REDIRECTED: '1', + ORCA_NODE_OPTIONS: '--inspect', + ORCA_NODE_REPL_EXTERNAL_MODULE: 'external-loader' + }) + }) + ) + const spawnedEnv = (run.mock.calls[0][0] as { env: NodeJS.ProcessEnv }).env + expect(spawnedEnv).not.toHaveProperty('NODE_OPTIONS') + expect(spawnedEnv).not.toHaveProperty('NODE_REPL_EXTERNAL_MODULE') + }) + + it('refuses to redirect twice so a dropped ELECTRON_RUN_AS_NODE cannot loop', () => { + const run = vi.fn() + + const result = maybeRedirectCliLaunch({ + argv: [linux.execPath, 'status'], + env: { ORCA_CLI_LAUNCH_REDIRECTED: '1' }, + platform: 'linux', + isPackaged: true, + resourcesPath: linux.resourcesPath, + execPath: linux.execPath, + commandNames: COMMAND_NAMES, + exists: () => true, + run: run as never + }) + + expect(result).toEqual({ redirected: true, status: 1 }) + expect(run).not.toHaveBeenCalled() + }) + + it('reports a missing CLI entrypoint instead of booting the desktop app', () => { + const run = vi.fn() + + const result = maybeRedirectCliLaunch({ + argv: [linux.execPath, 'status'], + env: {}, + platform: 'linux', + isPackaged: true, + resourcesPath: linux.resourcesPath, + execPath: linux.execPath, + commandNames: COMMAND_NAMES, + exists: () => false, + run: run as never + }) + + expect(result).toEqual({ redirected: true, status: 1 }) + expect(run).not.toHaveBeenCalled() + }) + + it('surfaces a spawn failure as a non-zero exit', () => { + const result = maybeRedirectCliLaunch({ + argv: [linux.execPath, 'status'], + env: {}, + platform: 'linux', + isPackaged: true, + resourcesPath: linux.resourcesPath, + execPath: linux.execPath, + commandNames: COMMAND_NAMES, + exists: () => true, + run: (() => { + throw new Error('spawn ENOENT') + }) as never + }) + + expect(result).toEqual({ redirected: true, status: 1 }) + }) +}) diff --git a/src/main/startup/cli-launch-redirect.ts b/src/main/startup/cli-launch-redirect.ts new file mode 100644 index 00000000000..88c0e8062cb --- /dev/null +++ b/src/main/startup/cli-launch-redirect.ts @@ -0,0 +1,245 @@ +import { existsSync } from 'node:fs' +import { posix, win32 } from 'node:path' +import { runProcessSync } from '../../shared/child-process/run-process' +import { CLI_BOOLEAN_FLAGS, findCliCommandIndex } from '../../shared/cli-argument-boundary' +import { CLI_COMMAND_NAMES } from './cli-command-names' +import { VALUE_TAKING_FLAGS } from './serve-mode-argv' + +export type CliLaunchRedirectResult = { redirected: false } | { redirected: true; status: number } + +export type CliLaunchRedirectOptions = { + argv?: string[] + env?: NodeJS.ProcessEnv + platform?: NodeJS.Platform + isPackaged?: boolean + resourcesPath?: string + execPath?: string + commandNames?: readonly string[] + exists?: typeof existsSync + run?: typeof runProcessSync +} + +const CLI_EARLY_EXIT_FLAGS = new Set(['--help', '-h', 'help', '--version', '-v']) +const DESKTOP_FLAGS = new Set(['--no-sandbox', '--disable-gpu']) +const DESKTOP_VALUE_FLAGS = new Set(['--disable-features']) +const CLI_LAUNCH_VALUE_FLAG_NAMES = [...VALUE_TAKING_FLAGS].map((flag) => flag.slice(2)) + +// Fence recursion if a wrapper drops ELECTRON_RUN_AS_NODE again. +const REDIRECT_ATTEMPT_ENV = 'ORCA_CLI_LAUNCH_REDIRECTED' + +// Redirect packaged CLI-shaped launches before Chromium initializes. +export function maybeRedirectCliLaunch( + options: CliLaunchRedirectOptions = {} +): CliLaunchRedirectResult { + const argv = options.argv ?? process.argv + const env = options.env ?? process.env + const platform = options.platform ?? process.platform + const isPackaged = options.isPackaged ?? false + const resourcesPath = options.resourcesPath ?? process.resourcesPath + const execPath = options.execPath ?? process.execPath + const exists = options.exists ?? existsSync + const run = options.run ?? runProcessSync + const cliEntryPath = buildPackagedCliEntryPath(platform, resourcesPath) + const cliArgs = getCliLaunchArgs(argv, cliEntryPath, { + platform, + isPackaged, + commandNames: options.commandNames ?? CLI_COMMAND_NAMES + }) + + if (!cliArgs) { + return { redirected: false } + } + if (env[REDIRECT_ATTEMPT_ENV] === '1') { + process.stderr.write('Unable to start the Orca CLI through Electron node mode.\n') + return { redirected: true, status: 1 } + } + if (!exists(cliEntryPath)) { + process.stderr.write(`Unable to locate the Orca CLI entrypoint at ${cliEntryPath}\n`) + return { redirected: true, status: 1 } + } + + const childEnv = buildElectronRunAsNodeEnv(env) + try { + const result = run({ + program: execPath, + args: [cliEntryPath, ...cliArgs], + env: childEnv, + stdio: 'inherit', + timeoutMs: null + }) + return { redirected: true, status: result.code ?? 1 } + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + return { redirected: true, status: 1 } + } +} + +export function getCliLaunchArgs( + argv: string[], + cliEntryPath: string, + options: { + platform: NodeJS.Platform + isPackaged: boolean + commandNames: readonly string[] + } +): string[] | null { + if (!options.isPackaged) { + return null + } + return getEntryPathLaunchArgs(argv, cliEntryPath, options) ?? getCommandLaunchArgs(argv, options) +} + +function getEntryPathLaunchArgs( + argv: string[], + cliEntryPath: string, + options: { platform: NodeJS.Platform; commandNames: readonly string[] } +): string[] | null { + const expectedCliPath = normalizePathForPlatform(cliEntryPath, options.platform) + // The packaged launcher always passes the entrypoint as Electron's first argument. + // Matching later positional arguments can mistake a normal desktop launch for the CLI. + if (!argv[1] || normalizePathForPlatform(argv[1], options.platform) !== expectedCliPath) { + return null + } + const args = argv.slice(2) + return stripDesktopFlags(args, findCommandIndex(args, options.commandNames)) +} + +function getCommandLaunchArgs( + argv: string[], + options: { platform: NodeJS.Platform; commandNames: readonly string[] } +): string[] | null { + if (options.platform !== 'linux') { + return null + } + const args = argv.slice(1) + if (args.length === 0) { + return null + } + const commandIndex = findCommandIndex(args, options.commandNames) + const cliArgs = stripDesktopFlags(args, commandIndex) + const command = commandIndex === -1 ? null : args[commandIndex] + // Keep direct serve in-process so signals reach its full child tree. + if (command && command !== 'serve') { + return cliArgs + } + return hasCliEarlyExitArg(args, commandIndex) ? cliArgs : null +} + +function findCommandIndex(args: readonly string[], commandNames: readonly string[]): number { + return findCliCommandIndex( + args, + commandNames.map((name) => [name]), + CLI_LAUNCH_VALUE_FLAG_NAMES + ) +} + +function stripDesktopFlags(args: readonly string[], commandIndex: number): string[] { + const boundary = commandIndex === -1 ? findLeadingFlagBoundary(args) : commandIndex + const cliArgs: string[] = [] + for (let index = 0; index < args.length; index += 1) { + const arg = args[index]! + if (index < boundary) { + if (DESKTOP_FLAGS.has(arg)) { + continue + } + if (DESKTOP_VALUE_FLAGS.has(flagName(arg))) { + if (!arg.includes('=') && args[index + 1] && !args[index + 1]!.startsWith('-')) { + index += 1 + } + continue + } + } + cliArgs.push(arg) + } + return cliArgs +} + +function findLeadingFlagBoundary(args: readonly string[]): number { + let index = 0 + while (index < args.length) { + const token = args[index]! + if (token === '--' || !token.startsWith('-')) { + return index + } + index += 1 + if (takesLaunchValue(token, args[index])) { + index += 1 + } + } + return index +} + +function hasCliEarlyExitArg(args: readonly string[], commandIndex: number): boolean { + let index = 0 + let positionalCount = 0 + while (index < args.length) { + const token = args[index]! + if (token === '--') { + return false + } + if ( + CLI_EARLY_EXIT_FLAGS.has(token) && + (token !== 'help' || positionalCount === 0 || commandIndex !== -1) + ) { + return true + } + if (!token.startsWith('-')) { + if (token === 'help' && (positionalCount === 0 || commandIndex !== -1)) { + return true + } + positionalCount += 1 + } + index += 1 + if (takesLaunchValue(token, args[index])) { + index += 1 + } + } + return false +} + +function takesLaunchValue(token: string, next: string | undefined): boolean { + if ( + !next || + next.startsWith('-') || + !token.startsWith('-') || + token.includes('=') || + CLI_EARLY_EXIT_FLAGS.has(token) || + DESKTOP_FLAGS.has(token) + ) { + return false + } + const name = flagName(token) + return DESKTOP_VALUE_FLAGS.has(name) || !CLI_BOOLEAN_FLAGS.has(name.replace(/^-+/, '')) +} + +function flagName(arg: string): string { + const equalsIndex = arg.indexOf('=') + return equalsIndex === -1 ? arg : arg.slice(0, equalsIndex) +} + +function buildPackagedCliEntryPath(platform: NodeJS.Platform, resourcesPath: string): string { + return getPathApi(platform).join(resourcesPath, 'app.asar.unpacked', 'out', 'cli', 'index.js') +} + +function normalizePathForPlatform(value: string, platform: NodeJS.Platform): string { + const pathApi = getPathApi(platform) + const normalized = pathApi.normalize(pathApi.isAbsolute(value) ? value : pathApi.resolve(value)) + // Windows path comparisons are case-insensitive. + return platform === 'win32' ? normalized.toLowerCase() : normalized +} + +function getPathApi(platform: NodeJS.Platform): typeof win32 | typeof posix { + return platform === 'win32' ? win32 : posix +} + +function buildElectronRunAsNodeEnv(env: NodeJS.ProcessEnv): NodeJS.ProcessEnv { + const childEnv = { ...env } + // Preserve user values without exposing them to Electron's bootstrap. + childEnv.ORCA_NODE_OPTIONS = env.NODE_OPTIONS ?? '' + childEnv.ORCA_NODE_REPL_EXTERNAL_MODULE = env.NODE_REPL_EXTERNAL_MODULE ?? '' + childEnv.ELECTRON_RUN_AS_NODE = '1' + childEnv[REDIRECT_ATTEMPT_ENV] = '1' + delete childEnv.NODE_OPTIONS + delete childEnv.NODE_REPL_EXTERNAL_MODULE + return childEnv +} diff --git a/src/main/startup/ensure-virtual-display.test.ts b/src/main/startup/ensure-virtual-display.test.ts index 93f4bb14f77..ded8360bce5 100644 --- a/src/main/startup/ensure-virtual-display.test.ts +++ b/src/main/startup/ensure-virtual-display.test.ts @@ -1,24 +1,25 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -const { spawnMock, spawnSyncMock, existsSyncMock, readFileSyncMock, rmSyncMock, appMock } = +const { spawnMock, existsSyncMock, readFileSyncMock, rmSyncMock, statSyncMock, appMock } = vi.hoisted(() => ({ spawnMock: vi.fn(), - spawnSyncMock: vi.fn(), existsSyncMock: vi.fn(), readFileSyncMock: vi.fn(), rmSyncMock: vi.fn(), + statSyncMock: vi.fn(), appMock: { disableHardwareAcceleration: vi.fn(), - commandLine: { appendSwitch: vi.fn() }, + commandLine: { appendSwitch: vi.fn(), getSwitchValue: vi.fn() }, once: vi.fn() } })) -vi.mock('child_process', () => ({ spawn: spawnMock, spawnSync: spawnSyncMock })) +vi.mock('child_process', () => ({ spawn: spawnMock })) vi.mock('fs', () => ({ existsSync: existsSyncMock, readFileSync: readFileSyncMock, - rmSync: rmSyncMock + rmSync: rmSyncMock, + statSync: statSyncMock })) vi.mock('electron', () => ({ app: appMock })) @@ -29,15 +30,39 @@ function setPlatform(platform: NodeJS.Platform): void { Object.defineProperty(process, 'platform', { value: platform, configurable: true }) } +function mockLiveXDisplay(pid = 4321): void { + statSyncMock.mockReturnValue({ isSocket: () => true }) + existsSyncMock.mockReturnValue(true) + readFileSyncMock.mockReturnValue(`${pid}\n`) + vi.spyOn(process, 'kill').mockImplementation(() => true) +} + +function mockXvfbTakesDisplay(pid = 1234): void { + let bound = false + statSyncMock.mockImplementation(() => ({ isSocket: () => true })) + readFileSyncMock.mockImplementation(() => { + if (!bound) { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + } + return `${pid}\n` + }) + vi.spyOn(process, 'kill').mockImplementation(() => true) + spawnMock.mockImplementation(() => { + bound = true + return { pid, once: vi.fn(), kill: vi.fn(), killed: false } + }) +} + describe('ensureVirtualDisplayForHeadlessServe', () => { beforeEach(() => { spawnMock.mockReset() - spawnSyncMock.mockReset() existsSyncMock.mockReset() readFileSyncMock.mockReset() rmSyncMock.mockReset() + statSyncMock.mockReset() appMock.disableHardwareAcceleration.mockReset() appMock.commandLine.appendSwitch.mockReset() + appMock.commandLine.getSwitchValue.mockReset().mockReturnValue('') appMock.once.mockReset() delete process.env.DISPLAY }) @@ -76,6 +101,7 @@ describe('ensureVirtualDisplayForHeadlessServe', () => { it('reuses an externally provided DISPLAY without starting Xvfb', async () => { setPlatform('linux') process.env.DISPLAY = ':0' + mockLiveXDisplay() const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(true) @@ -86,48 +112,75 @@ describe('ensureVirtualDisplayForHeadlessServe', () => { expect(appMock.commandLine.appendSwitch).toHaveBeenCalledWith('disable-gpu') }) - it('reports unsupported (no spawn) when Xvfb is not installed', async () => { + it('reports unsupported when Xvfb cannot be launched', async () => { setPlatform('linux') - spawnSyncMock.mockReturnValue({ status: 1 }) // `which Xvfb` fails + spawnMock.mockReturnValue({ pid: undefined, once: vi.fn(), kill: vi.fn(), killed: false }) + const { ensureVirtualDisplayForHeadlessServe, MISSING_LINUX_DISPLAY_MESSAGE } = + await import('./ensure-virtual-display') + + expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(false) + expect(spawnMock).toHaveBeenCalledWith('Xvfb', expect.any(Array), expect.any(Object)) + expect(MISSING_LINUX_DISPLAY_MESSAGE).toContain('endpoint is unavailable') + expect(MISSING_LINUX_DISPLAY_MESSAGE).toContain('XDG_RUNTIME_DIR') + expect(MISSING_LINUX_DISPLAY_MESSAGE).toContain('`xvfb` on Debian/Ubuntu') + expect(MISSING_LINUX_DISPLAY_MESSAGE).toContain('`xorg-x11-server-Xvfb`') + }) + + it('leaves an externally configured stale display untouched', async () => { + setPlatform('linux') + process.env.DISPLAY = ':77' + statSyncMock.mockReturnValue({ isSocket: () => true }) + existsSyncMock.mockReturnValue(true) + readFileSyncMock.mockImplementation(() => { + throw new Error('display lock is outside this namespace') + }) const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(false) expect(spawnMock).not.toHaveBeenCalled() + expect(rmSyncMock).not.toHaveBeenCalled() + expect(process.env.DISPLAY).toBe(':77') }) - it('starts Xvfb and switches to software rendering when none exists', async () => { + // #15084 review: a container that bind-mounts only /tmp/.X11-unix used to serve and would + // otherwise now exit(1) at index.ts, since the serve gate treats false as fatal. + it('serves on an externally configured display that has no lock file', async () => { setPlatform('linux') - spawnSyncMock.mockReturnValue({ status: 0 }) // `which Xvfb` succeeds - // First existsSync (stale-socket check) false; later (socket-ready poll) true. - existsSyncMock.mockReturnValueOnce(false).mockReturnValue(true) - spawnMock.mockReturnValue({ once: vi.fn(), kill: vi.fn(), killed: false }) - const processOnceSpy = vi.spyOn(process, 'once') - const processRemoveListenerSpy = vi.spyOn(process, 'removeListener') - const { ensureVirtualDisplayForHeadlessServe, stopVirtualDisplay } = - await import('./ensure-virtual-display') + process.env.DISPLAY = ':0' + statSyncMock.mockReturnValue({ isSocket: () => true }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(true) + expect(spawnMock).not.toHaveBeenCalled() + expect(rmSyncMock).not.toHaveBeenCalled() + expect(process.env.DISPLAY).toBe(':0') + }) + + // removeStaleDisplayArtifacts unlinks the lock before the socket, so a crash between the two + // leaves a lockless socket on Orca's OWN :99. Adopting it would resurrect the orphan-socket bug. + it('does not adopt its own :99 socket when the lock is missing', async () => { + setPlatform('linux') + existsSyncMock.mockReturnValue(true) + mockXvfbTakesDisplay() + const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') + + expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(true) + // Cleaned up and respawned rather than trusted. + expect(rmSyncMock).toHaveBeenCalled() expect(spawnMock).toHaveBeenCalledWith( 'Xvfb', - expect.arrayContaining([':99', '-terminate']), - expect.objectContaining({ detached: true }) + expect.arrayContaining([':99']), + expect.any(Object) ) - expect(process.env.DISPLAY).toBe(':99') - expect(appMock.disableHardwareAcceleration).toHaveBeenCalled() - expect(appMock.commandLine.appendSwitch).toHaveBeenCalledWith('disable-dev-shm-usage') - expect(appMock.commandLine.appendSwitch).toHaveBeenCalledWith('disable-gpu') - expect(processOnceSpy).toHaveBeenCalledWith('exit', stopVirtualDisplay) - const readyHandler = appMock.once.mock.calls.find(([event]) => event === 'ready')?.[1] - expect(readyHandler).toBeTypeOf('function') - readyHandler() - expect(processRemoveListenerSpy).toHaveBeenCalledWith('exit', stopVirtualDisplay) - expect(appMock.once.mock.calls.some(([event]) => event === 'will-quit')).toBe(false) }) it('reuses an existing virtual display only when its X server is alive', async () => { setPlatform('linux') - spawnSyncMock.mockReturnValue({ status: 0 }) - existsSyncMock.mockReturnValue(true) // :99 socket + lock present + statSyncMock.mockReturnValue({ isSocket: () => true }) // :99 socket present + existsSyncMock.mockReturnValue(true) readFileSyncMock.mockReturnValue('4321\n') // lock holds a PID const killSpy = vi.spyOn(process, 'kill').mockReturnValue(true as never) // PID alive const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') @@ -142,14 +195,21 @@ describe('ensureVirtualDisplayForHeadlessServe', () => { it('treats a stale socket (dead server) as no display and starts a fresh Xvfb', async () => { setPlatform('linux') - spawnSyncMock.mockReturnValue({ status: 0 }) - existsSyncMock.mockReturnValue(true) // orphan socket + lock present - readFileSyncMock.mockReturnValue('9999\n') - // PID is gone: process.kill throws ESRCH. - const killSpy = vi.spyOn(process, 'kill').mockImplementation(() => { - throw new Error('ESRCH') + existsSyncMock.mockReturnValue(true) // lock present + let bound = false + statSyncMock.mockImplementation(() => ({ isSocket: () => true })) + readFileSyncMock.mockImplementation(() => (bound ? '1234\n' : '9999\n')) + // The orphan lock names a dead PID; the freshly spawned Xvfb is alive. + const killSpy = vi.spyOn(process, 'kill').mockImplementation((pid) => { + if (pid === 9999) { + throw new Error('ESRCH') + } + return true as never + }) + spawnMock.mockImplementation(() => { + bound = true + return { pid: 1234, once: vi.fn(), kill: vi.fn(), killed: false } }) - spawnMock.mockReturnValue({ once: vi.fn(), kill: vi.fn(), killed: false }) const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(true) @@ -163,4 +223,278 @@ describe('ensureVirtualDisplayForHeadlessServe', () => { expect(process.env.DISPLAY).toBe(':99') killSpy.mockRestore() }) + + // A root-owned stale :99 socket (crashed system Xvfb, serve running as User=orca) cannot be + // unlinked, so our Xvfb refuses to bind and exits. Trusting the surviving socket set DISPLAY to a + // dead server and Chromium died in Ozone init with SIGSEGV. + it('reports failure when a stale socket blocks the Xvfb rebind', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => true }) + // Removal fails (foreign owner) and no lock ever appears, because Xvfb never took the display. + rmSyncMock.mockImplementation(() => { + throw Object.assign(new Error('permission denied'), { code: 'EACCES' }) + }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + spawnMock.mockReturnValue({ pid: 4242, once: vi.fn(), kill: vi.fn(), killed: false }) + const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') + + expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(false) + expect(process.env.DISPLAY).toBeUndefined() + }) + + it('accepts the display once the spawned Xvfb owns its lock', async () => { + setPlatform('linux') + let lockWritten = false + statSyncMock.mockImplementation(() => ({ isSocket: () => lockWritten })) + rmSyncMock.mockImplementation(() => {}) + readFileSyncMock.mockImplementation(() => { + if (!lockWritten) { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + } + return '4242\n' + }) + vi.spyOn(process, 'kill').mockImplementation(() => true) + spawnMock.mockImplementation(() => { + lockWritten = true + return { pid: 4242, once: vi.fn(), kill: vi.fn(), killed: false } + }) + const { ensureVirtualDisplayForHeadlessServe } = await import('./ensure-virtual-display') + + expect(ensureVirtualDisplayForHeadlessServe({ isServeMode: true })).toBe(true) + expect(process.env.DISPLAY).toBe(':99') + }) + + describe('hasUsableLinuxDisplay', () => { + it('accepts live local X11 and Wayland sockets', async () => { + setPlatform('linux') + mockLiveXDisplay() + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(true) + expect( + hasUsableLinuxDisplay({ + WAYLAND_DISPLAY: 'wayland-0', + XDG_RUNTIME_DIR: '/run/user/1000' + }) + ).toBe(true) + expect(statSyncMock).toHaveBeenCalledWith('/tmp/.X11-unix/X0') + expect(statSyncMock).toHaveBeenCalledWith('/run/user/1000/wayland-0') + }) + + // An X server may bind only the abstract namespace, leaving nothing to stat. Abstract addresses + // are kernel-owned and vanish when the owner exits, so an entry is proof of a live server. + it('accepts an abstract-namespace X socket with no filesystem socket', async () => { + setPlatform('linux') + statSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + readFileSyncMock.mockImplementation((path: string) => { + if (path === '/proc/net/unix') { + return [ + 'Num RefCount Protocol Flags Type St Inode Path', + '0000000000000000: 00000003 00000000 00000000 0001 03 12014 @/tmp/.X11-unix/X0', + '' + ].join('\n') + } + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(true) + }) + + it('does not confuse a different display number in the abstract table', async () => { + setPlatform('linux') + statSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + readFileSyncMock.mockImplementation((path: string) => { + if (path === '/proc/net/unix') { + return '0000000000000000: 00000003 00000000 00000000 0001 03 12014 @/tmp/.X11-unix/X10\n' + } + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':1' })).toBe(false) + }) + + it('accepts an inherited WAYLAND_SOCKET fd with no WAYLAND_DISPLAY', async () => { + setPlatform('linux') + statSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ WAYLAND_SOCKET: '7' })).toBe(true) + expect(hasUsableLinuxDisplay({ WAYLAND_SOCKET: 'not-an-fd' })).toBe(false) + }) + + // Orca's own teardown unlinks the lock before the socket, so a lockless :99 is our own + // half-finished cleanup — trusting it because DISPLAY names it would accept a dead display. + it('does not trust a lockless socket on its own managed display number', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => true }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':99' })).toBe(false) + // A foreign display number with the same shape is still accepted. + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(true) + }) + + it('rejects an orphaned local X11 socket whose server PID is gone', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => true }) + existsSyncMock.mockReturnValue(true) + readFileSyncMock.mockReturnValue('9999\n') + vi.spyOn(process, 'kill').mockImplementation(() => { + throw Object.assign(new Error('no such process'), { code: 'ESRCH' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':77' })).toBe(false) + expect(readFileSyncMock).toHaveBeenCalledWith('/tmp/.X77-lock', 'utf8') + }) + + // An X server writes its lock beside the socket and both survive a crash, so a lockless + // socket is an endpoint published from elsewhere (container bind mount, WSLg) — not an orphan. + it('accepts a local X11 socket published without a lock file', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => true }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const killSpy = vi.spyOn(process, 'kill') + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(true) + expect(killSpy).not.toHaveBeenCalled() + }) + + // WSLg with ELECTRON_OZONE_PLATFORM_HINT=x11 has no Wayland fallback to rescue it. + it('accepts a lockless X11 socket when x11 is pinned and Wayland is unavailable', async () => { + setPlatform('linux') + statSyncMock.mockImplementation((path: string) => ({ + isSocket: () => path === '/tmp/.X11-unix/X0' + })) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0', ELECTRON_OZONE_PLATFORM_HINT: 'x11' })).toBe( + true + ) + }) + + it('still rejects a missing socket even when no lock file exists', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => false }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(false) + }) + + it('rejects a lock that exists but cannot be read', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => true }) + readFileSyncMock.mockImplementation(() => { + throw Object.assign(new Error('permission denied'), { code: 'EACCES' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(false) + }) + + it('accepts a live local X11 server owned by another user', async () => { + setPlatform('linux') + statSyncMock.mockReturnValue({ isSocket: () => true }) + existsSyncMock.mockReturnValue(true) + readFileSyncMock.mockReturnValue('4321\n') + vi.spyOn(process, 'kill').mockImplementation(() => { + throw Object.assign(new Error('not permitted'), { code: 'EPERM' }) + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: ':0' })).toBe(true) + }) + + it('rejects absent, blank, and stale local displays', async () => { + setPlatform('linux') + statSyncMock.mockImplementation(() => { + throw new Error('ENOENT') + }) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({})).toBe(false) + expect(hasUsableLinuxDisplay({ DISPLAY: ' ', WAYLAND_DISPLAY: '' })).toBe(false) + expect(hasUsableLinuxDisplay({ DISPLAY: ':77' })).toBe(false) + expect( + hasUsableLinuxDisplay({ WAYLAND_DISPLAY: 'wayland-0', XDG_RUNTIME_DIR: '/run/user/1000' }) + ).toBe(false) + expect(hasUsableLinuxDisplay({ WAYLAND_DISPLAY: 'wayland-0' })).toBe(false) + }) + + it.each([ + ['localhost:10.0', true], + ['build-host.example:1', true], + ['[2001:db8::1]:2.0', true], + ['tcp/build-host.example:3', true], + ['garbage', false], + ['build host:1', false], + ['build-host.example:', false], + ['build-host.example:abc', false] + ])('validates remote X display syntax for %s', async (display, expected) => { + setPlatform('linux') + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({ DISPLAY: display })).toBe(expected) + expect(statSyncMock).not.toHaveBeenCalled() + }) + + it('honors forced X11 and Wayland platform selection', async () => { + setPlatform('linux') + statSyncMock.mockImplementation((path: string) => ({ + isSocket: () => path === '/run/user/1000/wayland-0' + })) + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + const env = { + DISPLAY: ':77', + WAYLAND_DISPLAY: 'wayland-0', + XDG_RUNTIME_DIR: '/run/user/1000' + } + + appMock.commandLine.getSwitchValue.mockReturnValue('x11') + expect(hasUsableLinuxDisplay(env)).toBe(false) + appMock.commandLine.getSwitchValue.mockReturnValue('wayland') + expect(hasUsableLinuxDisplay(env)).toBe(true) + + statSyncMock.mockImplementation((path: string) => ({ + isSocket: () => path === '/tmp/.X11-unix/X0' + })) + expect(hasUsableLinuxDisplay({ ...env, DISPLAY: ':0' })).toBe(false) + + appMock.commandLine.getSwitchValue.mockReturnValue('') + expect(hasUsableLinuxDisplay({ ...env, ELECTRON_OZONE_PLATFORM_HINT: 'x11' })).toBe(false) + }) + + it('never gates a non-Linux platform', async () => { + setPlatform('darwin') + const { hasUsableLinuxDisplay } = await import('./ensure-virtual-display') + + expect(hasUsableLinuxDisplay({})).toBe(true) + expect(statSyncMock).not.toHaveBeenCalled() + }) + }) }) diff --git a/src/main/startup/ensure-virtual-display.ts b/src/main/startup/ensure-virtual-display.ts index a9d517f9f99..6b78c05aa52 100644 --- a/src/main/startup/ensure-virtual-display.ts +++ b/src/main/startup/ensure-virtual-display.ts @@ -1,5 +1,6 @@ -import { spawn, spawnSync, type ChildProcess } from 'node:child_process' -import { existsSync, readFileSync, rmSync } from 'node:fs' +import { spawn, type ChildProcess } from 'node:child_process' +import { readFileSync, rmSync, statSync } from 'node:fs' +import { isAbsolute, join } from 'node:path' import { app } from 'electron' // Why: headless `orca serve` backs browser panes with offscreen BrowserWindows. @@ -12,6 +13,8 @@ const XVFB_STARTUP_TIMEOUT_MS = 5_000 const XVFB_POLL_INTERVAL_MS = 50 const VIRTUAL_DISPLAY_NUMBER = 99 const VIRTUAL_DISPLAY = `:${VIRTUAL_DISPLAY_NUMBER}` +const XVFB_INSTALL_GUIDANCE = + 'Install `xvfb` on Debian/Ubuntu or `xorg-x11-server-Xvfb` on RPM-based systems.' let xvfbProcess: ChildProcess | null = null @@ -33,32 +36,55 @@ function xDisplayLockPath(displayNumber: number): string { return `/tmp/.X${displayNumber}-lock` } -// Why: a socket file can outlive the X server that made it. The X lock file holds -// the server PID; if that process is gone, the display is dead despite the socket. -function isDisplayServerAlive(displayNumber: number): boolean { - const lockPath = xDisplayLockPath(displayNumber) - if (!existsSync(lockPath)) { - // No lock means no server claimed this display; the bare socket is stale. - return false - } +// Why: a socket file can outlive the X server that made it. The X lock file holds the server PID; +// if that process is gone, the display is dead despite the socket. `missing` is a third outcome the +// two callers must treat differently — see each call site. +type DisplayLockProbe = 'alive' | 'dead' | 'missing' + +function probeDisplayLock(displayNumber: number): DisplayLockProbe { let pid: number try { - pid = Number.parseInt(readFileSync(lockPath, 'utf8').trim(), 10) - } catch { - return false + pid = Number.parseInt(readFileSync(xDisplayLockPath(displayNumber), 'utf8').trim(), 10) + } catch (error) { + // An unreadable lock is a lock we cannot clear: treat it as dead, not absent. + return (error as NodeJS.ErrnoException)?.code === 'ENOENT' ? 'missing' : 'dead' } if (!Number.isInteger(pid) || pid <= 0) { - return false + return 'dead' } try { // signal 0 probes existence without affecting the process. process.kill(pid, 0) - return true - } catch { - return false + return 'alive' + } catch (error) { + // EPERM means the PID exists under another uid — a root-owned X server is still live. + return typeof error === 'object' && error !== null && 'code' in error && error.code === 'EPERM' + ? 'alive' + : 'dead' } } +/** + * Liveness for a display Orca did not create. An X server writes its lock beside the socket and + * both survive a crash (verified against Xvfb under SIGKILL), so a socket with no lock was never + * left by a crashed server — it is an endpoint published from elsewhere: a container bind-mounting + * only /tmp/.X11-unix, WSLg, or a foreign PID namespace. We cannot judge those, and refusing them + * blocks startup on displays that work. + */ +function isForeignDisplayServerAlive(displayNumber: number): boolean { + return probeDisplayLock(displayNumber) !== 'dead' +} + +/** + * Liveness for Orca's own VIRTUAL_DISPLAY_NUMBER. Stricter on purpose: `removeStaleDisplayArtifacts` + * unlinks the lock before the socket, so a lockless socket here is Orca's own half-finished + * teardown, not a foreign endpoint. Adopting it would resurrect the orphan-socket bug and stop the + * cleanup below from self-healing. + */ +function isManagedDisplayServerAlive(displayNumber: number): boolean { + return probeDisplayLock(displayNumber) === 'alive' +} + function removeStaleDisplayArtifacts(displayNumber: number): void { for (const path of [xDisplayLockPath(displayNumber), xvfbSocketPath(displayNumber)]) { try { @@ -69,30 +95,126 @@ function removeStaleDisplayArtifacts(displayNumber: number): void { } } -function hasXvfbBinary(): boolean { - // Why: spawnSync `which` is cheap and avoids spawning Xvfb only to fail; a - // clear up-front warning beats a cryptic ENOENT mid-startup. - const result = spawnSync('which', ['Xvfb'], { stdio: 'ignore' }) - return result.status === 0 -} - function sleepSync(ms: number): void { // Why: this runs in the synchronous pre-whenReady startup path, so block // without spinning the CPU or spawning a process. Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, ms) } -function waitForDisplaySocket(displayNumber: number, deadline: number): boolean { - const socket = xvfbSocketPath(displayNumber) - // Why: Xvfb creates its socket asynchronously after spawn; Electron must not - // boot before it exists or display init still fails. +// Why not socket presence alone: a stale socket we failed to unlink (a root-owned one left by a +// crashed system Xvfb, which `User=orca` serve cannot remove) still exists after our own Xvfb +// refused to bind the display. Treating that as ready sets DISPLAY to a dead server and Chromium +// dies in Ozone init with SIGSEGV instead of reporting an unusable display. +function waitForDisplayReady(displayNumber: number, deadline: number): boolean { + const isReady = (): boolean => + isUnixSocket(xvfbSocketPath(displayNumber)) && isManagedDisplayServerAlive(displayNumber) while (Date.now() < deadline) { - if (existsSync(socket)) { + if (isReady()) { return true } sleepSync(XVFB_POLL_INTERVAL_MS) } - return existsSync(socket) + return isReady() +} + +// Validate display syntax and local sockets before Chromium reaches Ozone initialization. +export function hasUsableLinuxDisplay(env: NodeJS.ProcessEnv = process.env): boolean { + if (process.platform !== 'linux') { + return true + } + + const ozonePlatform = app.commandLine.getSwitchValue('ozone-platform').trim().toLowerCase() + const ozonePlatformHint = env.ELECTRON_OZONE_PLATFORM_HINT?.trim().toLowerCase() + const selectedPlatform = + ozonePlatform === 'x11' || ozonePlatform === 'wayland' + ? ozonePlatform + : ozonePlatformHint === 'x11' || ozonePlatformHint === 'wayland' + ? ozonePlatformHint + : null + + if (selectedPlatform === 'x11') { + return hasUsableXDisplay(env.DISPLAY) + } + if (selectedPlatform === 'wayland') { + return hasUsableWaylandDisplay(env) + } + return hasUsableXDisplay(env.DISPLAY) || hasUsableWaylandDisplay(env) +} + +export const MISSING_LINUX_DISPLAY_MESSAGE = [ + 'Orca needs a usable display server, but the selected X11 or Wayland endpoint is unavailable.', + 'Check DISPLAY, WAYLAND_DISPLAY, XDG_RUNTIME_DIR, and any --ozone-platform override.', + `Use \`orca-ide serve\` to run headless. On a bare server, ${XVFB_INSTALL_GUIDANCE}` +].join('\n') + +// Why: an X server may bind only the abstract namespace (`@/tmp/.X11-unix/X0`), which leaves no +// filesystem socket to stat. Abstract addresses are kernel-owned and vanish the moment the owner +// exits, so an entry here is proof of a live server — no lock file needed, and no stale entry is +// possible. Refusing these was a hard startup failure with no workaround. +function hasAbstractXSocket(displayNumber: number): boolean { + let table: unknown + try { + table = readFileSync('/proc/net/unix', 'utf8') + } catch { + return false + } + if (typeof table !== 'string') { + return false + } + const address = `@${xvfbSocketPath(displayNumber)}` + return table + .split('\n') + .some((line) => line.slice(line.lastIndexOf(' ') + 1).trimEnd() === address) +} + +function isUnixSocket(path: string): boolean { + try { + return statSync(path).isSocket() + } catch { + return false + } +} + +function hasUsableXDisplay(value: string | undefined): boolean { + const display = value?.trim() + if (!display) { + return false + } + + const localDisplay = /^(?:unix\/?)?:(\d+)(?:\.\d+)?$/i.exec(display) + // Remote endpoints cannot be proven with local socket checks. + if (!localDisplay) { + return /^\S+:\d+(?:\.\d+)?$/.test(display) + } + const displayNumber = Number(localDisplay[1]) + if (isUnixSocket(xvfbSocketPath(displayNumber))) { + // Why the managed number is never treated as foreign: Orca's own teardown unlinks the lock + // before the socket, so a lockless socket on VIRTUAL_DISPLAY_NUMBER is our own half-finished + // cleanup even when DISPLAY names it explicitly. Trusting it there would accept a dead display. + return displayNumber === VIRTUAL_DISPLAY_NUMBER + ? isManagedDisplayServerAlive(displayNumber) + : isForeignDisplayServerAlive(displayNumber) + } + return hasAbstractXSocket(displayNumber) +} + +function hasUsableWaylandDisplay(env: NodeJS.ProcessEnv): boolean { + // Why: WAYLAND_SOCKET is an already-connected fd handed over by the compositor, so there is no + // path to stat and WAYLAND_DISPLAY may be unset entirely. Its presence IS the display. + const inheritedFd = env.WAYLAND_SOCKET?.trim() + if (inheritedFd && /^\d+$/.test(inheritedFd)) { + return true + } + const display = env.WAYLAND_DISPLAY?.trim() + if (!display) { + return false + } + if (isAbsolute(display)) { + return isUnixSocket(display) + } + + const runtimeDir = env.XDG_RUNTIME_DIR?.trim() + return Boolean(runtimeDir && isAbsolute(runtimeDir) && isUnixSocket(join(runtimeDir, display))) } /** @@ -107,16 +229,17 @@ export function ensureVirtualDisplayForHeadlessServe(options: { isServeMode: boo configureHeadlessServeChromiumFlags() - // Why: respect an externally provided display (a real X server, or the image - // already running its own Xvfb). Don't start a competing one. - if (process.env.DISPLAY && process.env.DISPLAY.trim().length > 0) { - return true - } - - if (!hasXvfbBinary()) { + // Offscreen serve windows require X11; Wayland alone still needs Xvfb. + // Never delete artifacts from an externally managed display: a container may + // expose its socket without the host lock/PID being visible here. + const configuredDisplay = process.env.DISPLAY?.trim() + if (configuredDisplay) { + if (hasUsableXDisplay(configuredDisplay)) { + return true + } console.warn( - '[serve] Xvfb not found; browser panes are unavailable on this headless Linux host. ' + - 'Install Xvfb (e.g. `apt-get install xvfb`) or set DISPLAY to enable them.' + `[serve] DISPLAY=${configuredDisplay} is not verifiably live; leaving it untouched. ` + + 'Unset DISPLAY to let Orca start its own Xvfb.' ) return false } @@ -124,8 +247,8 @@ export function ensureVirtualDisplayForHeadlessServe(options: { isServeMode: boo // Why: reuse an existing display ONLY if a live X server actually backs it. // A crashed prior run can leave an orphan socket; trusting it by path alone // would advertise browser support that then fails at tab creation. - if (existsSync(xvfbSocketPath(VIRTUAL_DISPLAY_NUMBER))) { - if (isDisplayServerAlive(VIRTUAL_DISPLAY_NUMBER)) { + if (isUnixSocket(xvfbSocketPath(VIRTUAL_DISPLAY_NUMBER))) { + if (isManagedDisplayServerAlive(VIRTUAL_DISPLAY_NUMBER)) { process.env.DISPLAY = VIRTUAL_DISPLAY return true } @@ -147,6 +270,11 @@ export function ensureVirtualDisplayForHeadlessServe(options: { isServeMode: boo xvfbProcess.once('error', (error) => { console.warn('[serve] Xvfb failed to start:', error instanceof Error ? error.message : error) }) + // PATH lookup failures emit asynchronously, but a successful spawn has a PID immediately. + if (xvfbProcess.pid === undefined) { + xvfbProcess = null + return false + } } catch (error) { console.warn( '[serve] Could not start Xvfb:', @@ -155,9 +283,12 @@ export function ensureVirtualDisplayForHeadlessServe(options: { isServeMode: boo return false } - const ready = waitForDisplaySocket(VIRTUAL_DISPLAY_NUMBER, Date.now() + XVFB_STARTUP_TIMEOUT_MS) + const ready = waitForDisplayReady(VIRTUAL_DISPLAY_NUMBER, Date.now() + XVFB_STARTUP_TIMEOUT_MS) if (!ready) { - console.warn('[serve] Xvfb did not become ready in time; browser panes may be unavailable.') + console.warn( + `[serve] Xvfb did not take ownership of ${VIRTUAL_DISPLAY}; browser panes are unavailable. ` + + 'A stale socket from another user can block the rebind.' + ) stopVirtualDisplay() return false } diff --git a/src/main/startup/main-process-preflight.ts b/src/main/startup/main-process-preflight.ts index eb2a51cb7ff..acbe42360e8 100644 --- a/src/main/startup/main-process-preflight.ts +++ b/src/main/startup/main-process-preflight.ts @@ -2,8 +2,7 @@ import { app, ipcMain, powerMonitor, session } from 'electron' import { is } from '@electron-toolkit/utils' import os from 'node:os' import { join } from 'node:path' -import { maybeRedirectAppImageCliLaunch } from './appimage-cli-redirect' -import { maybeRedirectPackagedCliEntryLaunch } from './packaged-cli-entry-redirect' +import { maybeRedirectCliLaunch } from './cli-launch-redirect' import { argvRequestsServeMode, normalizeServeModeArgv } from './serve-mode-argv' import { configureDevUserDataPath, @@ -78,7 +77,11 @@ import { recordCrashBreadcrumb } from '../crash-reporting/crash-breadcrumb-store import { recordDurableCrashBreadcrumb } from '../crash-reporting/durable-crash-breadcrumb' import { GpuCrashDiagnosticsRecorder } from '../crash-reporting/gpu-crash-diagnostics' import { getMainProcessLifecycleIdentity } from '../crash-reporting/main-process-lifecycle-identity' -import { ensureVirtualDisplayForHeadlessServe } from './ensure-virtual-display' +import { + ensureVirtualDisplayForHeadlessServe, + hasUsableLinuxDisplay, + MISSING_LINUX_DISPLAY_MESSAGE +} from './ensure-virtual-display' import { maybeApplyGpuFallbackForThisLaunch, registerGpuLifecycleHandlers } from './gpu-lifecycle' import { mainProcessState as state } from './main-process-state' import { initializeSyntheticTitleRuntime } from './synthetic-title-runtime' @@ -91,25 +94,15 @@ export type MainProcessPreflightOptions = { /** Performs all module-scope work that must happen before Electron's ready event. */ export function runMainProcessPreflight(options: MainProcessPreflightOptions): boolean { // Why: on Windows a CLI launch that lost ELECTRON_RUN_AS_NODE would boot the GUI and exit silently; redirect to node mode before the lock gate below. - // Both redirects run before the serve-argv rewrite so they still match on the launch argv verbatim. - // It is load-bearing for the AppImage one: rewriting first replaces the `serve` positional, so its - // command-name lookup finds a port number and strands the launch in an in-process serve. The - // packaged-CLI one matches on the entry path instead, so order cannot affect it either way. - const packagedRedirect = maybeRedirectPackagedCliEntryLaunch({ + // The redirect runs before the serve-argv rewrite so it still matches on the launch argv verbatim. + // Direct serve stays in-process so its signal handlers own all children. + const cliLaunchRedirect = maybeRedirectCliLaunch({ isPackaged: app.isPackaged, resourcesPath: process.resourcesPath, execPath: process.execPath }) - if (packagedRedirect.redirected) { - app.exit(packagedRedirect.status) - } - const appImageRedirect = maybeRedirectAppImageCliLaunch({ - isPackaged: app.isPackaged, - resourcesPath: process.resourcesPath, - execPath: process.execPath - }) - if (appImageRedirect.redirected) { - app.exit(appImageRedirect.status) + if (cliLaunchRedirect.redirected) { + app.exit(cliLaunchRedirect.status) } // Why: extracted AppRun / binary launches can land CLI-form `serve` args on the // Electron process without the CLI rewrite that injects `--serve` (#12677). @@ -118,6 +111,11 @@ export function runMainProcessPreflight(options: MainProcessPreflightOptions): b process.argv = normalizeServeModeArgv(process.argv) } state.isServeMode = process.argv.includes('--serve') + // Fail before Chromium's missing-display teardown can segfault (#13719). + if (app.isPackaged && !state.isServeMode && !hasUsableLinuxDisplay()) { + process.stderr.write(`${MISSING_LINUX_DISPLAY_MESSAGE}\n`) + app.exit(1) + } if (state.isServeMode) { reserveServeStdoutForReadiness() } @@ -317,6 +315,11 @@ export function runMainProcessPreflight(options: MainProcessPreflightOptions): b state.headlessBrowserDisplayAvailable = ensureVirtualDisplayForHeadlessServe({ isServeMode: state.isServeMode }) + // Why: continuing without Xvfb lets Ozone initialize without a display and SIGSEGV (#17615). + if (state.isServeMode && !state.headlessBrowserDisplayAvailable) { + process.stderr.write(`${MISSING_LINUX_DISPLAY_MESSAGE}\n`) + app.exit(1) + } initializeSyntheticTitleRuntime() registerGpuLifecycleHandlers() return true diff --git a/src/main/startup/main-process-runtime-launch.ts b/src/main/startup/main-process-runtime-launch.ts index 78061fb439c..7e5f278fdc5 100644 --- a/src/main/startup/main-process-runtime-launch.ts +++ b/src/main/startup/main-process-runtime-launch.ts @@ -280,7 +280,7 @@ export async function initializeMainProcessRuntimeLaunch( } let serveOptions: ReturnType | null = null try { - serveOptions = state.isServeMode ? getServeOptions() : null + serveOptions = state.isServeMode ? getServeOptions(process.argv) : null } catch (error) { console.error(error instanceof Error ? error.message : String(error)) app.exit(1) diff --git a/src/main/startup/main-process-serve.ts b/src/main/startup/main-process-serve.ts index 367bdf6d033..2be4d47d071 100644 --- a/src/main/startup/main-process-serve.ts +++ b/src/main/startup/main-process-serve.ts @@ -4,45 +4,9 @@ import { app } from 'electron' import { resolveAdvertisedPairingEndpoint } from '../runtime/pairing-endpoint' import { notifyServeSupervisorReady } from '../serve-update-handoff' import { mainProcessState as state } from './main-process-state' +import { getServeOptions, type ServeOptions } from './serve-options' -export type ServeOptions = { - json: boolean - wsPort?: number - pairingAddress: string | null - noPairing: boolean - mobilePairing: boolean - recipeJson: boolean - projectRoot: string | null -} - -export function getServeOptions(argv = process.argv): ServeOptions { - const valueAfter = (flag: string): string | null => { - const index = argv.indexOf(flag) - if (index === -1) { - return null - } - const value = argv[index + 1] - return value && !value.startsWith('--') ? value : null - } - const rawPort = valueAfter('--serve-port') - let wsPort: number | undefined - if (rawPort) { - const parsedPort = Number(rawPort) - if (!Number.isInteger(parsedPort) || parsedPort < 0 || parsedPort > 65535) { - throw new Error(`Invalid --serve-port value: ${rawPort}`) - } - wsPort = parsedPort - } - return { - json: argv.includes('--serve-json'), - ...(wsPort !== undefined ? { wsPort } : {}), - pairingAddress: valueAfter('--serve-pairing-address'), - noPairing: argv.includes('--serve-no-pairing'), - mobilePairing: argv.includes('--serve-mobile-pairing'), - recipeJson: argv.includes('--serve-recipe-json'), - projectRoot: valueAfter('--serve-project-root') - } -} +export { getServeOptions, type ServeOptions } export function getBundledWebClientRoot(): string | undefined { const appPath = app.getAppPath() diff --git a/src/main/startup/packaged-cli-entry-redirect.test.ts b/src/main/startup/packaged-cli-entry-redirect.test.ts deleted file mode 100644 index 4f91657eca0..00000000000 --- a/src/main/startup/packaged-cli-entry-redirect.test.ts +++ /dev/null @@ -1,157 +0,0 @@ -import { win32 } from 'node:path' -import { describe, expect, it, vi } from 'vitest' -import { - getPackagedCliEntryArgs, - maybeRedirectPackagedCliEntryLaunch -} from './packaged-cli-entry-redirect' - -const resourcesPath = 'C:\\Users\\me\\AppData\\Local\\Programs\\Orca\\resources' -const execPath = 'C:\\Users\\me\\AppData\\Local\\Programs\\Orca\\Orca.exe' -const cliEntryPath = win32.join(resourcesPath, 'app.asar.unpacked', 'out', 'cli', 'index.js') - -describe('packaged CLI entry redirect', () => { - it('detects Windows GUI launches that received the unpacked CLI entrypoint', () => { - expect( - getPackagedCliEntryArgs( - [execPath, cliEntryPath.toUpperCase(), 'status', '--json'], - cliEntryPath, - 'win32' - ) - ).toEqual(['status', '--json']) - }) - - it('ignores normal desktop launches', () => { - expect(getPackagedCliEntryArgs([execPath, '--updated'], cliEntryPath, 'win32')).toBeNull() - }) - - it('ignores the entrypoint when it is only the executable itself (argv[0])', () => { - expect(getPackagedCliEntryArgs([cliEntryPath, 'status'], cliEntryPath, 'win32')).toBeNull() - }) - - it('does not match the entrypoint on non-Windows platforms', () => { - expect( - getPackagedCliEntryArgs([execPath, cliEntryPath, 'status'], cliEntryPath, 'linux') - ).toBeNull() - }) - - it('spawns the in-package CLI in Electron node mode before the single-instance lock can win', () => { - const spawn = vi.fn((..._args: unknown[]) => ({ status: 0 })) - - const result = maybeRedirectPackagedCliEntryLaunch({ - argv: [execPath, cliEntryPath, 'status', '--json'], - env: { - NODE_OPTIONS: '--inspect', - NODE_REPL_EXTERNAL_MODULE: 'external-loader' - }, - platform: 'win32', - isPackaged: true, - resourcesPath, - execPath, - exists: () => true, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: true, status: 0 }) - expect(spawn).toHaveBeenCalledWith(execPath, [cliEntryPath, 'status', '--json'], { - env: expect.objectContaining({ - ELECTRON_RUN_AS_NODE: '1', - ORCA_PACKAGED_CLI_ENTRY_REDIRECTED: '1', - ORCA_NODE_OPTIONS: '--inspect', - ORCA_NODE_REPL_EXTERNAL_MODULE: 'external-loader' - }), - stdio: 'inherit' - }) - const spawnOptions = spawn.mock.calls[0]?.[2] as { env: NodeJS.ProcessEnv } | undefined - expect(spawnOptions?.env).not.toHaveProperty('NODE_OPTIONS') - expect(spawnOptions?.env).not.toHaveProperty('NODE_REPL_EXTERNAL_MODULE') - }) - - it('never spawns an attacker-supplied script — only the computed in-package entry', () => { - const spawn = vi.fn((..._args: unknown[]) => ({ status: 0 })) - const attackerScript = 'C:\\Users\\me\\evil.js' - - const result = maybeRedirectPackagedCliEntryLaunch({ - // An attacker placing some other script path in argv must not cause it to run. - argv: [execPath, attackerScript, 'status'], - platform: 'win32', - isPackaged: true, - resourcesPath, - execPath, - exists: () => true, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: false }) - expect(spawn).not.toHaveBeenCalled() - }) - - it('does not redirect development launches', () => { - const spawn = vi.fn() - - const result = maybeRedirectPackagedCliEntryLaunch({ - argv: ['C:\\dev\\Orca.exe', cliEntryPath, 'status'], - platform: 'win32', - isPackaged: false, - resourcesPath, - execPath: 'C:\\dev\\Orca.exe', - exists: () => true, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: false }) - expect(spawn).not.toHaveBeenCalled() - }) - - it('reports a clear failure instead of locating a missing entrypoint', () => { - const spawn = vi.fn() - const stderrWrite = vi.spyOn(process.stderr, 'write').mockImplementation(() => true) - - try { - const result = maybeRedirectPackagedCliEntryLaunch({ - argv: [execPath, cliEntryPath, 'status'], - platform: 'win32', - isPackaged: true, - resourcesPath, - execPath, - exists: () => false, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: true, status: 1 }) - expect(stderrWrite).toHaveBeenCalledWith( - `Unable to locate the Orca CLI entrypoint at ${cliEntryPath}\n` - ) - expect(spawn).not.toHaveBeenCalled() - } finally { - stderrWrite.mockRestore() - } - }) - - it('fails clearly instead of recursively redirecting when node mode already failed once', () => { - const spawn = vi.fn() - const stderrWrite = vi.spyOn(process.stderr, 'write').mockImplementation(() => true) - - try { - const result = maybeRedirectPackagedCliEntryLaunch({ - argv: [execPath, cliEntryPath, 'status', '--json'], - env: { - ORCA_PACKAGED_CLI_ENTRY_REDIRECTED: '1' - }, - platform: 'win32', - isPackaged: true, - resourcesPath, - execPath, - exists: () => true, - spawn: spawn as never - }) - - expect(result).toEqual({ redirected: true, status: 1 }) - expect(stderrWrite).toHaveBeenCalledWith( - 'Unable to start the Orca CLI through Electron node mode.\n' - ) - expect(spawn).not.toHaveBeenCalled() - } finally { - stderrWrite.mockRestore() - } - }) -}) diff --git a/src/main/startup/packaged-cli-entry-redirect.ts b/src/main/startup/packaged-cli-entry-redirect.ts deleted file mode 100644 index 333da2fcdd8..00000000000 --- a/src/main/startup/packaged-cli-entry-redirect.ts +++ /dev/null @@ -1,128 +0,0 @@ -import { spawnSync, type SpawnSyncReturns } from 'node:child_process' -import { existsSync } from 'node:fs' -import { posix, win32 } from 'node:path' - -type RedirectResult = - | { - redirected: false - } - | { - redirected: true - status: number - } - -type RedirectOptions = { - argv?: string[] - env?: NodeJS.ProcessEnv - platform?: NodeJS.Platform - isPackaged?: boolean - resourcesPath?: string - execPath?: string - exists?: typeof existsSync - spawn?: typeof spawnSync -} - -// Why: set on the re-spawned node-mode child so a failure to honor -// ELECTRON_RUN_AS_NODE can't make us redirect forever in a tight loop. -const REDIRECT_ATTEMPT_ENV = 'ORCA_PACKAGED_CLI_ENTRY_REDIRECTED' - -/** - * Why: on Windows the bundled native launcher runs `Orca.exe ` - * with ELECTRON_RUN_AS_NODE=1. When that env var is dropped (e.g. a wrapper or - * shell that resets it), Orca boots as a GUI, loses the single-instance lock to - * an already-running window, and exits silently with no stdout. This detects the - * CLI-shaped launch — argv carrying the known in-package CLI entry path — and - * re-runs it in Electron node mode BEFORE the lock gate, then exits with the - * CLI's status. - * - * Security: the spawned program is always `execPath` (Orca.exe) and the script - * is always `cliEntryPath`, derived solely from `resourcesPath` + a fixed - * relative path — never taken from argv. argv only contributes the trailing - * CLI arguments forwarded to the already-trusted in-package CLI, and the - * redirect only fires when an argv element exactly equals that computed path, - * so it cannot be coerced into spawning an arbitrary script. - */ -export function maybeRedirectPackagedCliEntryLaunch(options: RedirectOptions = {}): RedirectResult { - const argv = options.argv ?? process.argv - const env = options.env ?? process.env - const platform = options.platform ?? process.platform - const isPackaged = options.isPackaged ?? false - const resourcesPath = options.resourcesPath ?? process.resourcesPath - const execPath = options.execPath ?? process.execPath - const exists = options.exists ?? existsSync - const spawn = options.spawn ?? spawnSync - const cliEntryPath = buildPackagedCliEntryPath(platform, resourcesPath) - const cliArgs = getPackagedCliEntryArgs(argv, cliEntryPath, platform) - - if (!isPackaged || !cliArgs) { - return { redirected: false } - } - if (env[REDIRECT_ATTEMPT_ENV] === '1') { - process.stderr.write('Unable to start the Orca CLI through Electron node mode.\n') - return { redirected: true, status: 1 } - } - if (!exists(cliEntryPath)) { - process.stderr.write(`Unable to locate the Orca CLI entrypoint at ${cliEntryPath}\n`) - return { redirected: true, status: 1 } - } - - const result = spawn(execPath, [cliEntryPath, ...cliArgs], { - env: buildElectronRunAsNodeEnv(env), - stdio: 'inherit' - }) as SpawnSyncReturns - - if (result.error) { - process.stderr.write(`${result.error.message}\n`) - return { redirected: true, status: 1 } - } - - return { redirected: true, status: result.status ?? 1 } -} - -/** - * Returns the CLI arguments that follow the in-package CLI entrypoint in argv, - * or null when this is not a Windows CLI-shaped launch. Scoped to win32 because - * the AppImage redirect already covers the Linux equivalent. - */ -export function getPackagedCliEntryArgs( - argv: string[], - cliEntryPath: string, - platform: NodeJS.Platform -): string[] | null { - if (platform !== 'win32') { - return null - } - const expectedCliPath = normalizePathForPlatform(cliEntryPath, platform) - const cliEntryIndex = argv.findIndex( - (arg, index) => index > 0 && normalizePathForPlatform(arg, platform) === expectedCliPath - ) - return cliEntryIndex === -1 ? null : argv.slice(cliEntryIndex + 1) -} - -function buildPackagedCliEntryPath(platform: NodeJS.Platform, resourcesPath: string): string { - return getPathApi(platform).join(resourcesPath, 'app.asar.unpacked', 'out', 'cli', 'index.js') -} - -function normalizePathForPlatform(value: string, platform: NodeJS.Platform): string { - const pathApi = getPathApi(platform) - const normalized = pathApi.normalize(pathApi.isAbsolute(value) ? value : pathApi.resolve(value)) - // Why: Windows paths are case-insensitive, so compare case-folded. - return platform === 'win32' ? normalized.toLowerCase() : normalized -} - -function getPathApi(platform: NodeJS.Platform): typeof win32 | typeof posix { - return platform === 'win32' ? win32 : posix -} - -function buildElectronRunAsNodeEnv(env: NodeJS.ProcessEnv): NodeJS.ProcessEnv { - const childEnv = { ...env } - // Why: the CLI re-reads these from the ORCA_-prefixed copies; clearing the - // originals keeps Electron's own node bootstrap from inheriting them. - childEnv.ORCA_NODE_OPTIONS = env.NODE_OPTIONS ?? '' - childEnv.ORCA_NODE_REPL_EXTERNAL_MODULE = env.NODE_REPL_EXTERNAL_MODULE ?? '' - childEnv.ELECTRON_RUN_AS_NODE = '1' - childEnv[REDIRECT_ATTEMPT_ENV] = '1' - delete childEnv.NODE_OPTIONS - delete childEnv.NODE_REPL_EXTERNAL_MODULE - return childEnv -} diff --git a/src/main/startup/serve-mode-argv-cli-redirect-order.test.ts b/src/main/startup/serve-mode-argv-cli-redirect-order.test.ts index 3455c8c59ab..182bc94cc98 100644 --- a/src/main/startup/serve-mode-argv-cli-redirect-order.test.ts +++ b/src/main/startup/serve-mode-argv-cli-redirect-order.test.ts @@ -1,70 +1,67 @@ import { readFileSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it } from 'vitest' -import { getAppImageCliArgs } from './appimage-cli-redirect' +import { getCliLaunchArgs } from './cli-launch-redirect' import { argvRequestsServeMode, normalizeServeModeArgv } from './serve-mode-argv' -// Why: index.ts runs CLI redirects before rewriting argv. Direct AppImage serve -// stays in Electron so launch switches do not cross into the strict Node-mode -// CLI parser; other CLI commands still depend on redirect ordering (#12677). - +const CLI_ENTRY_PATH = '/opt/orca/resources/app.asar.unpacked/out/cli/index.js' const REDIRECT_OPTIONS = { platform: 'linux' as const, isPackaged: true, commandNames: ['serve', 'status'] } -// A mounted AppImage is the case where the runtime does export these. -const MOUNTED_APPIMAGE_ENV = { APPIMAGE: '/opt/orca/Orca.AppImage', APPDIR: '/tmp/.mount_ab12' } function rewriteAsIndexDoes(argv: string[]): string[] { return argvRequestsServeMode(argv) ? normalizeServeModeArgv(argv) : argv } -describe('serve argv rewrite vs AppImage CLI redirect ordering', () => { - const launchArgv = ['/opt/orca/orca-ide', '--no-sandbox', 'serve', '--port', '7777', '--json'] +describe('serve argv rewrite vs CLI launch redirect ordering', () => { + const launchArgv = [ + '/opt/orca/orca-ide', + '--disable-features=Vulkan', + 'serve', + '--port', + '7777', + '--json' + ] - it('keeps clean serve validation on the CLI path', () => { - expect(getAppImageCliArgs(launchArgv, MOUNTED_APPIMAGE_ENV, REDIRECT_OPTIONS)).toEqual([ - 'serve', - '--port', - '7777', - '--json' - ]) + it('leaves direct serve in the main process', () => { + expect(getCliLaunchArgs(launchArgv, CLI_ENTRY_PATH, REDIRECT_OPTIONS)).toBeNull() }) - it('keeps an injected Chromium switch in Electron before argv rewriting', () => { - const injected = [...launchArgv.slice(0, 2), '--disable-features=FedCm', ...launchArgv.slice(2)] - expect(getAppImageCliArgs(injected, MOUNTED_APPIMAGE_ENV, REDIRECT_OPTIONS)).toBeNull() - }) - - it('loses the redirect if the rewrite runs first', () => { + it('rewrites direct serve into the in-process flag shape', () => { const rewritten = rewriteAsIndexDoes(launchArgv) + expect(rewritten).toContain('--disable-features=Vulkan') expect(rewritten).toContain('--serve') - expect(getAppImageCliArgs(rewritten, MOUNTED_APPIMAGE_ENV, REDIRECT_OPTIONS)).toBeNull() + expect(rewritten).toContain('--serve-port') + expect(getCliLaunchArgs(rewritten, CLI_ENTRY_PATH, REDIRECT_OPTIONS)).toBeNull() }) it('leaves non-serve CLI commands redirectable either way', () => { const argv = ['/opt/orca/orca-ide', 'status'] expect(rewriteAsIndexDoes(argv)).toEqual(argv) - expect(getAppImageCliArgs(argv, MOUNTED_APPIMAGE_ENV, REDIRECT_OPTIONS)).toEqual(['status']) + expect(getCliLaunchArgs(argv, CLI_ENTRY_PATH, REDIRECT_OPTIONS)).toEqual(['status']) + }) + + it('redirects serve help instead of binding a server', () => { + const argv = ['/opt/orca/orca-ide', 'serve', '--help'] + expect(rewriteAsIndexDoes(argv)).toEqual(argv) + expect(getCliLaunchArgs(argv, CLI_ENTRY_PATH, REDIRECT_OPTIONS)).toEqual(['serve', '--help']) }) // Why source text: the ordering is the preflight phase's executable statement order, and the // cases above stay green if it is reversed — nothing else would catch the regression. - it('keeps the preflight running both CLI redirects before the argv rewrite', () => { + it('keeps the preflight running the CLI redirect before the argv rewrite', () => { const source = readFileSync( join(process.cwd(), 'src/main/startup/main-process-preflight.ts'), 'utf8' ) - const packagedRedirect = source.indexOf('maybeRedirectPackagedCliEntryLaunch({') - const appImageRedirect = source.indexOf('maybeRedirectAppImageCliLaunch({') + const cliRedirect = source.indexOf('maybeRedirectCliLaunch({') const rewrite = source.indexOf('process.argv = normalizeServeModeArgv(process.argv)') const serveModeCheck = source.indexOf("state.isServeMode = process.argv.includes('--serve')") - expect(packagedRedirect).toBeGreaterThanOrEqual(0) - expect(appImageRedirect).toBeGreaterThanOrEqual(0) - expect(rewrite).toBeGreaterThan(packagedRedirect) - expect(rewrite).toBeGreaterThan(appImageRedirect) + expect(cliRedirect).toBeGreaterThanOrEqual(0) + expect(rewrite).toBeGreaterThan(cliRedirect) // The rewrite is pointless unless it lands before the flag it exists to inject is read. expect(serveModeCheck).toBeGreaterThan(rewrite) }) diff --git a/src/main/startup/serve-mode-argv.test.ts b/src/main/startup/serve-mode-argv.test.ts index 13c340c8e10..53b0bb616a1 100644 --- a/src/main/startup/serve-mode-argv.test.ts +++ b/src/main/startup/serve-mode-argv.test.ts @@ -25,6 +25,19 @@ describe('serve-mode-argv', () => { expect(findServeSubcommandIndex(['app', '--user-data-dir', '/tmp/x', 'serve'])).toBe(3) }) + it('skips a space-separated Chromium switch value while locating serve', () => { + const argv = ['/AppRun', '--disable-features', 'Vulkan', 'serve', '--port', '6768'] + expect(findServeSubcommandIndex(argv)).toBe(3) + expect(normalizeServeModeArgv(argv)).toEqual([ + '/AppRun', + '--disable-features', + 'Vulkan', + '--serve', + '--serve-port', + '6768' + ]) + }) + it('refuses a help launch instead of binding a server', () => { // Why: `--help` is not a serve flag, so it used to be swallowed and the launch bound a // network-exposed runtime server with pairing on. The AppImage redirect routes help to the CLI. @@ -151,6 +164,14 @@ describe('serve-mode-argv', () => { ).toEqual(['/AppRun', '--serve', '--serve-port', '9090', '--serve-pairing-address', '0.0.0.0']) }) + it('keeps equals-form values that start with a flag marker intact', () => { + expect(normalizeServeModeArgv(['/AppRun', 'serve', '--pairing-address=--no-pairng'])).toEqual([ + '/AppRun', + '--serve', + '--serve-pairing-address=--no-pairng' + ]) + }) + it('translates serve flags in the mixed `--serve --port` form', () => { // Why: leaving these untranslated silently kept pairing enabled despite --no-pairing. expect(normalizeServeModeArgv(['orca', '--serve', '--port', '9090', '--no-pairing'])).toEqual([ diff --git a/src/main/startup/serve-mode-argv.ts b/src/main/startup/serve-mode-argv.ts index 6b406faf80e..8441f116805 100644 --- a/src/main/startup/serve-mode-argv.ts +++ b/src/main/startup/serve-mode-argv.ts @@ -26,13 +26,14 @@ const CLI_TO_SERVE_VALUE_FLAG = new Map([ * Residual class: a flag outside this list whose space-separated value is literally `serve` would * read as the subcommand. Include switches that may arrive in either argv shape. */ -const VALUE_TAKING_FLAGS = new Set([ +export const VALUE_TAKING_FLAGS = new Set([ ...CLI_TO_SERVE_VALUE_FLAG.keys(), '--serve-port', '--serve-pairing-address', '--serve-project-root', '--disable-features', '--user-data-dir', + '--proxy-server', '--environment', '--pairing-code' ]) @@ -135,8 +136,7 @@ export function normalizeServeModeArgv(argv: readonly string[]): string[] { next.push(...argv.slice(i)) break } - // Why: the CLI accepts `--port=6768` as well as `--port 6768`, but - // getServeOptions only reads the next token, so `=` must be split apart. + // Why: keep the internal argv shape canonical even though getServeOptions accepts both forms. const eq = token.indexOf('=') const name = eq === -1 ? token : token.slice(0, eq) // Why only the bare form: the CLI reads its serve booleans as `flags.get(name) === true` @@ -154,7 +154,14 @@ export function normalizeServeModeArgv(argv: readonly string[]): string[] { continue } if (eq !== -1) { - next.push(valueFlag, token.slice(eq + 1)) + const value = token.slice(eq + 1) + // Preserve the unambiguous `=` form when its value starts with `--`; splitting + // it would make the value look like a second option to the direct parser. + if (value.startsWith('--')) { + next.push(`${valueFlag}=${value}`) + } else { + next.push(valueFlag, value) + } continue } next.push(valueFlag) diff --git a/src/main/startup/serve-options.test.ts b/src/main/startup/serve-options.test.ts new file mode 100644 index 00000000000..9e2bb06919d --- /dev/null +++ b/src/main/startup/serve-options.test.ts @@ -0,0 +1,161 @@ +import { describe, expect, it } from 'vitest' +import { getServeOptions } from './serve-options' +import { normalizeServeModeArgv } from './serve-mode-argv' + +describe('getServeOptions', () => { + it('parses a valid launch', () => { + expect( + getServeOptions(['/AppRun', '--serve', '--serve-port', '6768', '--serve-no-pairing']) + ).toEqual({ + json: false, + wsPort: 6768, + pairingAddress: null, + noPairing: true, + mobilePairing: false, + recipeJson: false, + projectRoot: null + }) + }) + + it('accepts equals-form values in the normalized shape', () => { + expect( + getServeOptions([ + '/AppRun', + '--serve', + '--serve-port=6768', + '--serve-pairing-address=127.0.0.1', + '--serve-project-root=/tmp/repo' + ]) + ).toMatchObject({ + wsPort: 6768, + pairingAddress: '127.0.0.1', + projectRoot: '/tmp/repo' + }) + }) + + it('uses the final occurrence of each value flag', () => { + expect( + getServeOptions([ + '/AppRun', + '--serve', + '--serve-port', + '6768', + '--serve-port=6769', + '--serve-pairing-address', + 'first.example', + '--serve-pairing-address=last.example', + '--serve-project-root', + '/first', + '--serve-project-root=/last' + ]) + ).toMatchObject({ + wsPort: 6769, + pairingAddress: 'last.example', + projectRoot: '/last' + }) + }) + + it('applies missing or invalid values only to the final occurrence', () => { + expect( + getServeOptions(['/AppRun', '--serve', '--serve-port', '--serve-port', '6768']).wsPort + ).toBe(6768) + expect(() => + getServeOptions(['/AppRun', '--serve', '--serve-port', '6768', '--serve-port']) + ).toThrow('Missing value for --serve-port.') + expect(() => + getServeOptions(['/AppRun', '--serve', '--serve-port', '6768', '--serve-port=bad']) + ).toThrow('Invalid --serve-port value: bad') + }) + + it('uses the final value of mixed boolean aliases', () => { + expect( + getServeOptions(['/AppRun', '--serve', '--serve-no-pairing', '--no-pairing=false']).noPairing + ).toBe(false) + expect( + getServeOptions(['/AppRun', '--serve', '--no-pairing=false', '--serve-no-pairing']).noPairing + ).toBe(true) + expect( + getServeOptions(['/AppRun', '--serve', '--serve-mobile-pairing', '--mobile-pairing=0']) + .mobilePairing + ).toBe(false) + expect( + getServeOptions(['/AppRun', '--serve', '--serve-recipe-json', '--recipe-json=false']) + .recipeJson + ).toBe(false) + }) + + it('keeps JSON enabled for an equals-form global flag', () => { + expect(getServeOptions(['/AppRun', '--serve', '--json=false']).json).toBe(true) + }) + + it('accepts an equals-form value that resembles a pairing flag', () => { + const argv = normalizeServeModeArgv(['/AppRun', 'serve', '--pairing-address=--no-pairng']) + expect(getServeOptions(argv).pairingAddress).toBe('--no-pairng') + }) + + it('shares cross-flag validation with the CLI-form launch', () => { + const argv = normalizeServeModeArgv([ + '/opt/orca/orca-ide', + 'serve', + '--no-pairing', + '--mobile-pairing' + ]) + expect(() => getServeOptions(argv)).toThrow(/either --mobile-pairing or --no-pairing/i) + }) + + it('rejects recipe JSON without runtime pairing and a project root', () => { + expect(() => + getServeOptions([ + '/AppRun', + '--serve', + '--serve-recipe-json', + '--serve-no-pairing', + '--serve-project-root', + '/tmp/repo' + ]) + ).toThrow(/requires runtime pairing.*--no-pairing/i) + expect(() => getServeOptions(['/AppRun', '--serve', '--serve-recipe-json'])).toThrow( + /requires --project-root/i + ) + }) + + it('rejects a security-shaped typo while allowing Chromium switches', () => { + const normalized = normalizeServeModeArgv(['/AppRun', 'serve', '--no-pairng']) + expect(() => getServeOptions(normalized)).toThrow(/Unknown flag --no-pairng.*--no-pairing/i) + expect( + getServeOptions(['/AppRun', '--serve', '--disable-gpu', '--disable-features=Vulkan']) + .noPairing + ).toBe(false) + }) + + it('still rejects a flag-shaped space value, as the CLI does', () => { + expect(() => + getServeOptions(['/AppRun', '--serve', '--serve-pairing-address', '--no-pairng']) + ).toThrow(/Unknown flag --no-pairng.*--no-pairing/i) + }) + + it('ignores serve-looking arguments after the terminator', () => { + expect( + getServeOptions(['/AppRun', '--serve', '--', '--serve-port', '1', '--serve-no-pairing']) + ).toEqual({ + json: false, + pairingAddress: null, + noPairing: false, + mobilePairing: false, + recipeJson: false, + projectRoot: null + }) + }) + + it('requires a port value', () => { + expect(() => getServeOptions(['/AppRun', '--serve', '--serve-port'])).toThrow( + 'Missing value for --serve-port.' + ) + }) + + it.each(['', '--serve-json', '--'])('rejects an unusable port value %j', (value) => { + expect(() => getServeOptions(['/AppRun', '--serve', '--serve-port', value])).toThrow( + 'Missing value for --serve-port.' + ) + }) +}) diff --git a/src/main/startup/serve-options.ts b/src/main/startup/serve-options.ts new file mode 100644 index 00000000000..c0398123797 --- /dev/null +++ b/src/main/startup/serve-options.ts @@ -0,0 +1,134 @@ +import { + getServeFlagTypoError, + getServeOptionValidationError +} from '../../shared/serve-option-validation' + +export type ServeOptions = { + json: boolean + wsPort?: number + pairingAddress: string | null + noPairing: boolean + mobilePairing: boolean + recipeJson: boolean + projectRoot: string | null +} + +function optionsBeforeTerminator(argv: readonly string[]): readonly string[] { + const terminatorIndex = argv.indexOf('--') + return terminatorIndex === -1 ? argv : argv.slice(0, terminatorIndex) +} + +function optionName(token: string): string { + const equalsIndex = token.indexOf('=') + return equalsIndex === -1 ? token : token.slice(0, equalsIndex) +} + +function lastValueOccurrence( + argv: readonly string[], + flags: readonly string[] +): string | null | undefined { + const flagNames = new Set(flags) + let value: string | null | undefined + for (let index = 0; index < argv.length; index += 1) { + const token = argv[index]! + const name = optionName(token) + if (!flagNames.has(name)) { + continue + } + + const equalsIndex = token.indexOf('=') + if (equalsIndex !== -1) { + const assigned = token.slice(equalsIndex + 1) + value = assigned || null + continue + } + + const next = argv[index + 1] + if (next !== undefined && !next.startsWith('--')) { + value = next || null + index += 1 + } else { + value = null + } + } + return value +} + +function valueAfter( + argv: readonly string[], + flags: readonly string[], + required: boolean, + displayFlag: string +): string | null { + const value = lastValueOccurrence(argv, flags) + if (value === undefined || value === null) { + if (required && value !== undefined) { + throw new Error(`Missing value for ${displayFlag}.`) + } + return null + } + return value +} + +function lastBooleanValue(argv: readonly string[], flags: readonly string[]): boolean { + const flagNames = new Set(flags) + let value = false + for (const token of argv) { + const name = optionName(token) + if (!flagNames.has(name)) { + continue + } + // CLI boolean flags are true only in bare form; `--flag=...` is a string value. + value = !token.includes('=') + } + return value +} + +function hasFlag(argv: readonly string[], flags: readonly string[]): boolean { + const flagNames = new Set(flags) + return argv.some((token) => flagNames.has(optionName(token))) +} + +export function getServeOptions(argv: readonly string[]): ServeOptions { + const optionsArgv = optionsBeforeTerminator(argv) + const typoError = getServeFlagTypoError(optionsArgv) + if (typoError) { + throw new Error(typoError) + } + + const rawPort = valueAfter(optionsArgv, ['--serve-port', '--port'], true, '--serve-port') + let wsPort: number | undefined + if (rawPort) { + const parsedPort = Number(rawPort) + if (!Number.isInteger(parsedPort) || parsedPort < 0 || parsedPort > 65535) { + throw new Error(`Invalid --serve-port value: ${rawPort}`) + } + wsPort = parsedPort + } + + const options: ServeOptions = { + // The CLI uses `flags.has('json')`, so even `--json=false` enables JSON output. + json: hasFlag(optionsArgv, ['--serve-json', '--json']), + ...(wsPort !== undefined ? { wsPort } : {}), + pairingAddress: valueAfter( + optionsArgv, + ['--serve-pairing-address', '--pairing-address'], + false, + '--serve-pairing-address' + ), + noPairing: lastBooleanValue(optionsArgv, ['--serve-no-pairing', '--no-pairing']), + mobilePairing: lastBooleanValue(optionsArgv, ['--serve-mobile-pairing', '--mobile-pairing']), + recipeJson: lastBooleanValue(optionsArgv, ['--serve-recipe-json', '--recipe-json']), + projectRoot: valueAfter( + optionsArgv, + ['--serve-project-root', '--project-root'], + false, + '--serve-project-root' + ) + } + const validationError = getServeOptionValidationError(options) + if (validationError) { + throw new Error(validationError) + } + return options +} diff --git a/src/main/startup/serve-signal-handlers.test.ts b/src/main/startup/serve-signal-handlers.test.ts index 36250cd4146..35c70ef87de 100644 --- a/src/main/startup/serve-signal-handlers.test.ts +++ b/src/main/startup/serve-signal-handlers.test.ts @@ -11,9 +11,11 @@ describe('registerServeSignalHandlers', () => { signalSource.emit('SIGINT') signalSource.emit('SIGINT') signalSource.emit('SIGTERM') + signalSource.emit('SIGHUP') - expect(quitApplication).toHaveBeenCalledTimes(3) + expect(quitApplication).toHaveBeenCalledTimes(4) expect(signalSource.listenerCount('SIGINT')).toBe(1) expect(signalSource.listenerCount('SIGTERM')).toBe(1) + expect(signalSource.listenerCount('SIGHUP')).toBe(1) }) }) diff --git a/src/main/startup/serve-signal-handlers.ts b/src/main/startup/serve-signal-handlers.ts index 3022d935ac5..0df48d5ea35 100644 --- a/src/main/startup/serve-signal-handlers.ts +++ b/src/main/startup/serve-signal-handlers.ts @@ -1,12 +1,13 @@ type ServeSignalSource = { - on(event: 'SIGINT' | 'SIGTERM', listener: () => void): unknown + on(event: 'SIGINT' | 'SIGTERM' | 'SIGHUP', listener: () => void): unknown } export function registerServeSignalHandlers( signalSource: ServeSignalSource, quitApplication: () => void ): void { - // Keep both listeners installed so duplicate delivery cannot fall through to default termination. + // Keep every listener installed so duplicate delivery cannot fall through to default termination. signalSource.on('SIGINT', quitApplication) signalSource.on('SIGTERM', quitApplication) + signalSource.on('SIGHUP', quitApplication) } diff --git a/src/main/startup/single-instance-lock-exit.electron.test.ts b/src/main/startup/single-instance-lock-exit.electron.test.ts index 15623c2459f..8c60f276216 100644 --- a/src/main/startup/single-instance-lock-exit.electron.test.ts +++ b/src/main/startup/single-instance-lock-exit.electron.test.ts @@ -6,11 +6,8 @@ import { join } from 'node:path' import { afterAll, describe, expect, it } from 'vitest' import { SINGLE_INSTANCE_ALREADY_RUNNING_EXIT_CODE } from './single-instance-lock' -// Why #11935: the lock-loss gate runs before Electron `ready`, where `app.quit()` is deferred, so a -// duplicate headless `orca serve` kept executing the rest of startup, reached Linux Ozone/X11 init -// with no display, died with SIGSEGV, and systemd restarted it until the leaked AppImage FUSE mounts -// hit the kernel's 1000-mount ceiling. This runs the gate's own termination statement, lifted out of -// `src/main/index.ts`, under the real Electron binary. +// Why: `app.quit()` is deferred before Electron `ready`, so fatal startup gates must use the +// synchronous `app.exit()`. Run their shipped termination statements under the real binary. // // Why not a live lock race: Chromium's Linux ProcessSingleton only answers a second process once the // browser IO thread is up, which needs `ready` and therefore a display. On a display-less CI runner @@ -19,10 +16,10 @@ import { SINGLE_INSTANCE_ALREADY_RUNNING_EXIT_CODE } from './single-instance-loc // only a real process can settle is what the loser does next, which is what this file pins. const electronBinary = createRequire(import.meta.url)('electron') as string -const LOCK_LOST = 'LOCK_LOST' +const GATE_ENTERED = 'GATE_ENTERED' const CONTINUED_INTO_STARTUP = 'CONTINUED_INTO_STARTUP' const REACHED_TAIL = 'REACHED_TAIL' -const MARKER_ENV = 'ORCA_LOCK_FIXTURE_MARKER' +const MARKER_ENV = 'ORCA_PRE_READY_EXIT_FIXTURE_MARKER' const fixtureRoots: string[] = [] @@ -32,13 +29,13 @@ afterAll(() => { } }) -/** The `app.*` call the shipped lock-loss gate executes, so a revert to `app.quit()` fails here. */ -function readLockLossTermination(): string { +/** Read the `app.*` termination statement from a pre-ready gate in the shipped entrypoint. */ +function readPreReadyTermination(gate: string): string { const source = readFileSync( join(process.cwd(), 'src/main/startup/main-process-preflight.ts'), 'utf8' ) - const start = source.indexOf('if (!hasLock) {') + const start = source.indexOf(gate) expect(start).toBeGreaterThanOrEqual(0) const end = source.indexOf('\n }', start) expect(end).toBeGreaterThan(start) @@ -59,7 +56,7 @@ function buildFixtureMain(termination: string): string { `const marker = process.env.${MARKER_ENV}`, `const mark = (name) => appendFileSync(marker, name + '\\n')`, `const SINGLE_INSTANCE_ALREADY_RUNNING_EXIT_CODE = ${SINGLE_INSTANCE_ALREADY_RUNNING_EXIT_CODE}`, - `mark('${LOCK_LOST}')`, + `mark('${GATE_ENTERED}')`, termination, `mark('${CONTINUED_INTO_STARTUP}')`, // Why: stand in for the rest of `src/main/index.ts`, which on the reported host was display init. @@ -70,15 +67,15 @@ function buildFixtureMain(termination: string): string { type FixtureRun = { status: number | null; markers: string[] } -function runLockLossGate(termination: string): FixtureRun { - const root = mkdtempSync(join(tmpdir(), 'orca-lock-loss-')) +function runPreReadyGate(termination: string): FixtureRun { + const root = mkdtempSync(join(tmpdir(), 'orca-pre-ready-exit-')) fixtureRoots.push(root) const dir = join(root, 'fixture') const marker = join(root, 'markers.log') mkdirSync(dir, { recursive: true }) writeFileSync( join(dir, 'package.json'), - '{ "name": "orca-lock-loss-fixture", "main": "main.js" }' + '{ "name": "orca-pre-ready-exit-fixture", "main": "main.js" }' ) writeFileSync(join(dir, 'main.js'), buildFixtureMain(termination)) writeFileSync(marker, '') @@ -96,23 +93,34 @@ function runLockLossGate(termination: string): FixtureRun { } } -describe('#11935 pre-ready lock-loss termination under real Electron', () => { +describe('pre-ready termination under real Electron', () => { it('stops the duplicate launch before any further startup runs, with the already-running code', () => { - const termination = readLockLossTermination() + const termination = readPreReadyTermination('if (!hasLock) {') // Why: an empty slice would let the fixture fall through to its own exit and pass vacuously. expect(termination).not.toBe('') - const run = runLockLossGate(termination) + const run = runPreReadyGate(termination) - expect(run.markers).toEqual([LOCK_LOST]) + expect(run.markers).toEqual([GATE_ENTERED]) expect(run.status).toBe(SINGLE_INSTANCE_ALREADY_RUNNING_EXIT_CODE) }, 90_000) + it('#17615 stops serve when display setup fails instead of entering Chromium startup', () => { + const termination = readPreReadyTermination( + 'if (state.isServeMode && !state.headlessBrowserDisplayAvailable) {' + ) + + const run = runPreReadyGate(termination) + + expect(run.markers).toEqual([GATE_ENTERED]) + expect(run.status).toBe(1) + }, 90_000) + it('reproduces the deferred graceful quit that let the doomed launch keep booting', () => { - const run = runLockLossGate('app.quit()') + const run = runPreReadyGate('app.quit()') // Why: pins the Electron semantic the fix rests on — pre-`ready` `quit()` schedules, it does not stop. - expect(run.markers).toEqual([LOCK_LOST, CONTINUED_INTO_STARTUP, REACHED_TAIL]) + expect(run.markers).toEqual([GATE_ENTERED, CONTINUED_INTO_STARTUP, REACHED_TAIL]) expect(run.status).not.toBe(SINGLE_INSTANCE_ALREADY_RUNNING_EXIT_CODE) }, 90_000) }) diff --git a/src/main/updater-events.test.ts b/src/main/updater-events.test.ts index 7a24483ca70..6a643ae64a5 100644 --- a/src/main/updater-events.test.ts +++ b/src/main/updater-events.test.ts @@ -1,14 +1,23 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { UpdateStatus } from '../shared/update-status-types' import type { registerAutoUpdaterHandlers } from './updater-events' -const { appMock, nativeUpdaterMock, getLinuxRootPackageTypeMock } = vi.hoisted(() => ({ +const { + appMock, + nativeUpdaterMock, + getLinuxPackageTypeMock, + getLinuxRootPackageTypeMock, + isExternallyManagedLinuxInstallMock +} = vi.hoisted(() => ({ appMock: { isPackaged: true, getVersion: vi.fn(() => '1.0.51'), on: vi.fn() }, nativeUpdaterMock: { on: vi.fn() }, - getLinuxRootPackageTypeMock: vi.fn<() => 'deb' | 'rpm' | null>(() => 'deb') + getLinuxPackageTypeMock: vi.fn<() => 'deb' | 'rpm' | 'non-root' | 'unusable'>(() => 'deb'), + getLinuxRootPackageTypeMock: vi.fn<() => 'deb' | 'rpm' | null>(() => 'deb'), + isExternallyManagedLinuxInstallMock: vi.fn<() => boolean>(() => false) })) vi.mock('electron', () => ({ @@ -19,7 +28,9 @@ vi.mock('electron', () => ({ // Why: only the packaged-marker resolver is faked so the real artifact tracking runs. vi.mock('./linux-update-package-type', () => ({ - getLinuxRootPackageType: getLinuxRootPackageTypeMock + getLinuxPackageType: getLinuxPackageTypeMock, + getLinuxRootPackageType: getLinuxRootPackageTypeMock, + isExternallyManagedLinuxInstall: isExternallyManagedLinuxInstallMock })) vi.mock('./updater-changelog', () => ({ fetchChangelog: vi.fn().mockResolvedValue(null) })) @@ -59,7 +70,9 @@ function createContext(overrides?: Partial): HandlerContext { consumeMissingManifestPrereleaseFallbackResult: vi.fn(() => null), getPublishingWindowLastGoodCheck: vi.fn(() => null), getMissingManifestPrereleaseFallbackUserInitiated: vi.fn(() => null), - getCurrentStatus: vi.fn(() => ({ state: 'checking' }) as never), + getCurrentStatus: vi.fn( + () => ({ state: 'downloading', percent: 42, version: '1.0.61' }) as never + ), getActiveUpdateCheckEventAttemptId: vi.fn(() => 1), getKnownReleaseUrl: vi.fn(() => undefined), getPendingInstallVersion: vi.fn(() => '1.0.61'), @@ -107,7 +120,10 @@ describe('registerAutoUpdaterHandlers linux package artifact tracking', () => { appMock.on.mockReset() nativeUpdaterMock.on.mockReset() appMock.getVersion.mockReset().mockReturnValue('1.0.51') + getLinuxPackageTypeMock.mockReset().mockReturnValue('deb') getLinuxRootPackageTypeMock.mockReset().mockReturnValue('deb') + isExternallyManagedLinuxInstallMock.mockReset().mockReturnValue(false) + appMock.isPackaged = true }) const register = async ( @@ -142,6 +158,94 @@ describe('registerAutoUpdaterHandlers linux package artifact tracking', () => { }) }) + it.each(['deb', 'rpm'] as const)( + 'publishes manual-install recovery after a %s download', + async (packageType) => { + getLinuxPackageTypeMock.mockReturnValue(packageType) + getLinuxRootPackageTypeMock.mockReturnValue(packageType) + const { emit, context } = await register() + const fileName = packageType === 'deb' ? 'orca.deb' : 'orca.rpm' + + emit( + 'update-downloaded', + downloadedEvent({ + downloadedFile: `/home/tester/.cache/orca-updater/pending/${fileName}`, + files: [{ url: fileName, sha512: DEB_SHA512 }] + }) + ) + + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType, + reason: 'manual-install-required', + version: '1.0.61' + } + }) + } + ) + + it.each([ + ['missing', [{ url: 'orca-ide_1.0.61_amd64.deb' }]], + ['malformed', [{ url: 'orca-ide_1.0.61_amd64.deb', sha512: 'not-a-digest' }]] + ])('does not offer recovery when the package digest is %s', async (_kind, files) => { + const { emit, context, getArtifact } = await register() + + emit('update-downloaded', downloadedEvent({ files })) + + const status = { + state: 'error', + message: + 'The downloaded package metadata could not be verified. Quit Orca before downloading and installing the update from the official release page.', + version: '1.0.61', + retryable: false + } + expect(context.sendStatus).toHaveBeenLastCalledWith(status) + expect(context.sendStatus).not.toHaveBeenCalledWith( + expect.objectContaining({ recovery: expect.anything() }) + ) + expect(getArtifact()).toBeNull() + }) + + it('publishes the normal downloaded state for AppImage builds', async () => { + getLinuxPackageTypeMock.mockReturnValue('non-root') + getLinuxRootPackageTypeMock.mockReturnValue(null) + const { emit, context } = await register() + + emit('update-downloaded', downloadedEvent()) + if (process.platform === 'darwin') { + const handler = nativeUpdaterMock.on.mock.calls.find( + ([eventName]) => eventName === 'update-downloaded' + )?.[1] as (() => void) | undefined + handler?.() + } + + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'downloaded', + version: '1.0.61', + releaseUrl: undefined + }) + }) + + it('blocks downloaded-state handling when the packaged marker is unusable', async () => { + getLinuxPackageTypeMock.mockReturnValue('unusable') + getLinuxRootPackageTypeMock.mockReturnValue(null) + const { emit, context, getArtifact } = await register() + + emit('update-downloaded', downloadedEvent()) + + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'error', + message: + 'Orca could not verify the installed Linux package format, so it will not install this update automatically. Download the update from the official release page and install it manually.', + version: '1.0.61', + retryable: false + }) + expect(getArtifact()).toBeNull() + }) + it('passes the actual updater error into the install-failure handler', async () => { const handleQuitAndInstallFailure = vi.fn<(error?: unknown) => boolean>(() => true) const { emit, context } = await register({ handleQuitAndInstallFailure }) @@ -155,13 +259,64 @@ describe('registerAutoUpdaterHandlers linux package artifact tracking', () => { expect(context.sendErrorStatus).not.toHaveBeenCalled() }) - it('drops the artifact once the update resolves as not available', async () => { - const { emit, getArtifact } = await register() + it('keeps manual-install recovery when a later check finds no newer release', async () => { + const { emit, context, getArtifact } = await register() emit('update-downloaded', downloadedEvent()) + // Why: the download already produced this exact status, so the assertion below could pass + // on that call alone. Clear it so only the second emit can satisfy it. + vi.mocked(context.sendStatus).mockClear() + + emit('update-not-available') + + expect(getArtifact()).toEqual(expect.objectContaining({ version: '1.0.61' })) + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType: 'deb', + reason: 'manual-install-required', + version: '1.0.61' + } + }) + }) + + it('keeps manual-install recovery when a later check finds only the installed release', async () => { + const { emit, context, getArtifact } = await register() + emit('update-downloaded', downloadedEvent()) + + // Why: the download already produced this exact status, so the assertion below could pass + // on that call alone. Clear it so only the second emit can satisfy it. + vi.mocked(context.sendStatus).mockClear() + + emit('update-available', { version: '1.0.51' }) + + expect(getArtifact()).toEqual(expect.objectContaining({ version: '1.0.61' })) + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType: 'deb', + reason: 'manual-install-required', + version: '1.0.61' + } + }) + }) + + it('clears recovery when a newer update takes over before no-update settles', async () => { + const { emit, context, getArtifact } = await register() + emit('update-downloaded', downloadedEvent()) + + emit('update-available', { version: '1.0.62' }) emit('update-not-available') expect(getArtifact()).toBeNull() + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'not-available', + userInitiated: undefined + }) }) it('drops the artifact when another version takes over the cycle', async () => { @@ -173,6 +328,74 @@ describe('registerAutoUpdaterHandlers linux package artifact tracking', () => { expect(getArtifact()).toBeNull() }) + it('ignores a downloaded event for an older target', async () => { + const { emit, context, getArtifact } = await register({ + getCurrentStatus: vi.fn(() => ({ state: 'available', version: '1.0.62' }) as never), + getPendingInstallVersion: vi.fn(() => '1.0.62') + }) + + emit('update-downloaded', downloadedEvent()) + + expect(getArtifact()).toBeNull() + expect(context.sendStatus).not.toHaveBeenCalled() + }) + + it.each([ + ['idle', { state: 'idle' }], + ['not-available', { state: 'not-available' }], + ['check error', { state: 'error', message: 'check failed' }] + ] as const)( + 'ignores a downloaded event after the target is no longer active (%s)', + async (_name, status) => { + const { emit, context, getArtifact } = await register({ + getCurrentStatus: vi.fn(() => status as never), + getPendingInstallVersion: vi.fn(() => '') + }) + + emit('update-downloaded', downloadedEvent()) + + expect(getArtifact()).toBeNull() + expect(context.sendStatus).not.toHaveBeenCalled() + } + ) + + it('accepts a matching event when the pending cache target was cleared', async () => { + const { emit, context, getArtifact } = await register({ + getCurrentStatus: vi.fn( + () => ({ state: 'downloading', percent: 42, version: '1.0.61' }) as never + ), + getPendingInstallVersion: vi.fn(() => '') + }) + + emit('update-downloaded', downloadedEvent()) + + expect(getArtifact()).toEqual(expect.objectContaining({ version: '1.0.61' })) + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType: 'deb', + reason: 'manual-install-required', + version: '1.0.61' + } + }) + }) + + it('ignores a downloaded event when the active status and pending target disagree', async () => { + const { emit, context, getArtifact } = await register({ + getCurrentStatus: vi.fn( + () => ({ state: 'downloading', percent: 42, version: '1.0.62' }) as never + ), + getPendingInstallVersion: vi.fn(() => '1.0.62') + }) + + emit('update-downloaded', downloadedEvent()) + + expect(getArtifact()).toBeNull() + expect(context.sendStatus).not.toHaveBeenCalled() + }) + it('drops the artifact when progress reports a different pending version', async () => { const { emit, getArtifact } = await register({ getPendingInstallVersion: vi.fn(() => '1.0.62') @@ -184,13 +407,38 @@ describe('registerAutoUpdaterHandlers linux package artifact tracking', () => { expect(getArtifact()).toBeNull() }) - it('keeps the artifact through a same-version recheck', async () => { - const { emit, getArtifact } = await register() - emit('update-downloaded', downloadedEvent()) + // #17702: the externallyManaged flag is spread onto the fallback object only, so a retained + // manual-install status must still win. Cross-version case: the host could self-update when it + // downloaded, and cannot now. + it.each([false, true])( + 'keeps manual-install recovery through a same-version recheck (externallyManaged=%s)', + async (externallyManaged) => { + isExternallyManagedLinuxInstallMock.mockReturnValue(externallyManaged) + let status: UpdateStatus = { state: 'downloading', percent: 100, version: '1.0.61' } + const { emit, context, getArtifact } = await register({ + getCurrentStatus: vi.fn(() => status) + }) + emit('update-downloaded', downloadedEvent()) - emit('update-available', { version: '1.0.61' }) - emit('download-progress', { percent: 100 }) + // Why: the download already emitted the manual-install status, so waitFor would pass on that + // call alone. Clear it so the assertion can only be satisfied by the recheck. + vi.mocked(context.sendStatus).mockClear() + status = { state: 'checking' } + emit('update-available', { version: '1.0.61' }) - expect(getArtifact()).toEqual(expect.objectContaining({ version: '1.0.61', path: DEB_PATH })) - }) + expect(getArtifact()).toEqual(expect.objectContaining({ version: '1.0.61', path: DEB_PATH })) + await vi.waitFor(() => + expect(context.sendStatus).toHaveBeenLastCalledWith({ + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType: 'deb', + reason: 'manual-install-required', + version: '1.0.61' + } + }) + ) + } + ) }) diff --git a/src/main/updater-events.ts b/src/main/updater-events.ts index b77cd8a8cc5..d2b7f2a1e83 100644 --- a/src/main/updater-events.ts +++ b/src/main/updater-events.ts @@ -1,11 +1,8 @@ -import { app, autoUpdater as nativeUpdater } from 'electron' +import { app } from 'electron' import type { UpdateStatus } from '../shared/update-status-types' import { - consumeMacInstallGuardBypass, - deferMacQuitUntilInstallerReady, - handleMacInstallerReady, isMacInstallerReady, - isMacQuitAndInstallInFlight, + registerMacUpdaterEvents, resetMacInstallState } from './updater-mac-install' import { compareVersions } from './updater-fallback' @@ -13,10 +10,12 @@ import { fetchChangelog } from './updater-changelog' import type { ElectronAutoUpdater } from './electron-updater-loader' import { recordUpdaterLifecycle } from './updater-lifecycle-diagnostics' import { - captureLinuxPackageArtifact, - clearTrackedLinuxPackageArtifact, - clearTrackedLinuxPackageArtifactForOtherVersion -} from './linux-package-update-recovery' + getRetainedLinuxPackageManualInstallStatus, + resolveLinuxPackageDownloadedStatus, + shouldIgnoreDownloadedUpdateEvent +} from './linux-package-downloaded-status' +import { isExternallyManagedLinuxInstall } from './linux-update-package-type' +import * as linuxPackageRecovery from './linux-package-update-recovery' const AUTO_UPDATE_CHECK_INTERVAL_MS = 24 * 60 * 60 * 1000 const AUTO_UPDATE_RETRY_INTERVAL_MS = 60 * 60 * 1000 @@ -101,47 +100,14 @@ export function registerAutoUpdaterHandlers({ setAvailableVersion, setUserInitiatedCheck }: UpdaterHandlerContext): void { - // Why: electron-updater fires 'update-downloaded' before Squirrel.Mac finishes; track readiness to avoid a premature "ready". - if (process.platform === 'darwin') { - nativeUpdater.on('update-downloaded', () => { - const hasInstallableVersion = hasInstallableDownloadedVersion() - handleMacInstallerReady(hasInstallableVersion, performQuitAndInstall, () => { - // Send the held status only while its staged build is still installable. - sendStatus({ - state: 'downloaded', - version: getPendingInstallVersion(), - releaseUrl: getKnownReleaseUrl() - }) - }) - }) - } - - app.on('before-quit', (event) => { - if (!shouldDeferMacQuitForInstall()) { - return - } - if (consumeMacInstallGuardBypass()) { - recordUpdaterLifecycle('macos_before_quit_guard_bypassed') - return - } - if (isMacQuitAndInstallInFlight()) { - return - } - - // Why: quitting before Squirrel.Mac finishes staging leaves nothing to install; hold the quit until it's ready. - if ( - deferMacQuitUntilInstallerReady( - getCurrentStatus(), - hasInstallableDownloadedVersion(), - getPendingInstallVersion, - sendStatus - ) - ) { - recordUpdaterLifecycle('macos_before_quit_deferred', { - version: getPendingInstallVersion() - }) - event.preventDefault() - } + registerMacUpdaterEvents({ + getCurrentStatus, + hasInstallableDownloadedVersion, + getPendingInstallVersion, + getKnownReleaseUrl, + performQuitAndInstall, + shouldDeferMacQuitForInstall, + sendStatus }) autoUpdater.on('checking-for-update', () => { @@ -185,13 +151,18 @@ export function registerAutoUpdaterHandlers({ scheduleAutomaticUpdateCheck(AUTO_UPDATE_CHECK_INTERVAL_MS) } } - sendStatus({ state: 'not-available', userInitiated: wasUserInitiated || undefined }) + sendStatus( + getRetainedLinuxPackageManualInstallStatus() ?? { + state: 'not-available', + userInitiated: wasUserInitiated || undefined + } + ) return } // Why: only a genuinely newer offer supersedes the retained package; a publishing-window blip that // momentarily resolves an older tag must not destroy a still-valid recovery path. - clearTrackedLinuxPackageArtifactForOtherVersion(info.version) + linuxPackageRecovery.clearTrackedLinuxPackageArtifactForOtherVersion(info.version) // Why: fetch the changelog in main to avoid renderer-side CORS on onorca.dev. markUpdateAvailableEventPending(attemptId) @@ -228,7 +199,15 @@ export function registerAutoUpdaterHandlers({ } } - sendStatus({ state: 'available', version: info.version, changelog }) + sendStatus( + getRetainedLinuxPackageManualInstallStatus() ?? { + state: 'available', + version: info.version, + changelog, + // Why: the offer is real, but this host can never apply it — say so before a download is offered. + ...(isExternallyManagedLinuxInstall() ? { externallyManaged: true } : {}) + } + ) } finally { clearUpdateAvailableEventPending(attemptId) } @@ -241,7 +220,7 @@ export function registerAutoUpdaterHandlers({ } clearBackgroundCheckLaunchPending() resetMacInstallState() - clearTrackedLinuxPackageArtifact() + const retainedStatus = getRetainedLinuxPackageManualInstallStatus() const missingManifestFallback = consumeMissingManifestPrereleaseFallbackResult() const publishingWindowLastGoodCheck = getPublishingWindowLastGoodCheck() const wasUserInitiated = missingManifestFallback?.userInitiated ?? getUserInitiatedCheck() @@ -262,7 +241,11 @@ export function registerAutoUpdaterHandlers({ } } } - sendStatus({ state: 'not-available', userInitiated: wasUserInitiated || undefined }) + // Why: a later check can report no newer release while a verified deb/rpm is still waiting for + // the user to install it outside Orca. Keep both the artifact and its recovery card reachable. + sendStatus( + retainedStatus ?? { state: 'not-available', userInitiated: wasUserInitiated || undefined } + ) if (localBuildCheck || pinnedBuildCheck) { restoreReleaseUpdateSource() } @@ -271,7 +254,7 @@ export function registerAutoUpdaterHandlers({ autoUpdater.on('download-progress', (progress) => { clearBackgroundCheckLaunchPending() const version = getPendingInstallVersion() - clearTrackedLinuxPackageArtifactForOtherVersion(version) + linuxPackageRecovery.clearTrackedLinuxPackageArtifactForOtherVersion(version) sendStatus({ state: 'downloading', percent: Math.round(progress.percent), @@ -280,6 +263,16 @@ export function registerAutoUpdaterHandlers({ }) autoUpdater.on('update-downloaded', (info) => { + // Why: an earlier download can finish after a newer target replaced it; uncached pre-staged events have no target to compare. + if ( + shouldIgnoreDownloadedUpdateEvent( + getCurrentStatus(), + info.version, + getPendingInstallVersion() + ) + ) { + return + } clearBackgroundCheckLaunchPending() // Release downloads remain newer-only; the local source was validated before checking, and a pinned jump is explicit. if ( @@ -288,14 +281,17 @@ export function registerAutoUpdaterHandlers({ compareVersions(info.version, app.getVersion()) <= 0 ) { clearAvailableUpdateContext() - clearTrackedLinuxPackageArtifact() + linuxPackageRecovery.clearTrackedLinuxPackageArtifact() sendStatus({ state: 'not-available' }) return } - // Why: retain the verified artifact now — the 'error' event after a failed install no longer carries it. - captureLinuxPackageArtifact(info) const macInstallerReady = process.platform === 'darwin' ? isMacInstallerReady() : true recordUpdaterLifecycle('update_downloaded', { version: info.version, macInstallerReady }) + const linuxPackageStatus = resolveLinuxPackageDownloadedStatus(info) + if (linuxPackageStatus) { + sendStatus(linuxPackageStatus) + return + } // On macOS, defer 'downloaded' until Squirrel.Mac finishes processing; other platforms are ready immediately. if (process.platform === 'darwin' && !macInstallerReady) { // Keep the UI at 100% downloaded while Squirrel processes, to avoid a premature "ready to install". diff --git a/src/main/updater-fallback.ts b/src/main/updater-fallback.ts index 22cfb049555..62ec3ff3b8c 100644 --- a/src/main/updater-fallback.ts +++ b/src/main/updater-fallback.ts @@ -49,13 +49,12 @@ export function statusesEqual(left: UpdateStatus, right: UpdateStatus): boolean return ( right.state === 'error' && left.message === right.message && + left.version === right.version && + left.retryable === right.retryable && left.userInitiated === right.userInitiated && left.activeNudgeId === right.activeNudgeId && - // Why: clearing recovery must reach the renderer even when the message is unchanged, or dead actions stay enabled. - left.recovery?.kind === right.recovery?.kind && - left.recovery?.packageType === right.recovery?.packageType && - left.recovery?.reason === right.recovery?.reason && - left.recovery?.version === right.recovery?.version + // Recovery identity fences async actions, so same-valued recaptures must reach the renderer. + left.recovery === right.recovery ) } } diff --git a/src/main/updater-linux-package-recovery-actions.test.ts b/src/main/updater-linux-package-recovery-actions.test.ts index ff0b974bf7d..9a546dd8fac 100644 --- a/src/main/updater-linux-package-recovery-actions.test.ts +++ b/src/main/updater-linux-package-recovery-actions.test.ts @@ -10,8 +10,8 @@ const { getTrackedLinuxPackageArtifactMock, recordUpdaterLifecycleMock, resolveLinuxPackageInstallInstructionsMock, - revalidateLinuxPackageForInstallMock, - revealLinuxPackageMock, + resolveLinuxPackageRevealTargetMock, + showItemInFolderMock, resetHandlers } = vi.hoisted(() => { const updaterHandlers = new Map void)[]>() @@ -44,8 +44,8 @@ const { getTrackedLinuxPackageArtifactMock: vi.fn(), recordUpdaterLifecycleMock: vi.fn(), resolveLinuxPackageInstallInstructionsMock: vi.fn(), - revalidateLinuxPackageForInstallMock: vi.fn(), - revealLinuxPackageMock: vi.fn(), + resolveLinuxPackageRevealTargetMock: vi.fn(), + showItemInFolderMock: vi.fn(), resetHandlers: () => updaterHandlers.clear() } }) @@ -55,6 +55,7 @@ vi.mock('electron', () => ({ BrowserWindow: { getAllWindows: vi.fn(() => []) }, autoUpdater: { on: vi.fn() }, powerMonitor: { on: vi.fn() }, + shell: { showItemInFolder: showItemInFolderMock }, net: { fetch: vi.fn() } })) @@ -78,15 +79,18 @@ vi.mock('./update-install-exit-watchdog', () => ({ vi.mock('./updater-lifecycle-diagnostics', () => ({ recordUpdaterLifecycle: recordUpdaterLifecycleMock })) -vi.mock('./linux-update-package-type', () => ({ getLinuxRootPackageType: () => 'deb' })) +vi.mock('./linux-update-package-type', () => ({ + getLinuxPackageType: () => 'deb', + getLinuxRootPackageType: () => 'deb', + isExternallyManagedLinuxInstall: () => false +})) vi.mock('./linux-package-update-recovery', () => ({ - captureLinuxPackageArtifact: vi.fn(), + captureLinuxPackageArtifact: vi.fn(() => getTrackedLinuxPackageArtifactMock()), clearTrackedLinuxPackageArtifact: clearTrackedLinuxPackageArtifactMock, clearTrackedLinuxPackageArtifactForOtherVersion: vi.fn(), getTrackedLinuxPackageArtifact: getTrackedLinuxPackageArtifactMock, resolveLinuxPackageInstallInstructions: resolveLinuxPackageInstallInstructionsMock, - revalidateLinuxPackageForInstall: revalidateLinuxPackageForInstallMock, - revealLinuxPackage: revealLinuxPackageMock + resolveLinuxPackageRevealTarget: resolveLinuxPackageRevealTargetMock })) const ARTIFACT = { @@ -95,6 +99,16 @@ const ARTIFACT = { path: '/home/tester/.cache/orca-updater/pending/orca-ide_1.0.61_amd64.deb', sha512: 'LHlL7dKoqg98gS2nfQv878dK+UoktbAkm4M20/hoJ2Qr0Kqsa3MSL4VmWy/Lll/MYjQFkpvOxduQ/vswentozA==' } +const MANUAL_INSTALL_STATUS = { + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType: 'deb', + reason: 'manual-install-required', + version: '1.0.61' + } +} as const satisfies UpdateStatus warmUpdaterModule() @@ -108,7 +122,7 @@ describe('linux package recovery actions', () => { vi.useFakeTimers() resetHandlers() autoUpdaterMock.checkForUpdates.mockReset().mockResolvedValue(null) - autoUpdaterMock.downloadUpdate.mockReset() + autoUpdaterMock.downloadUpdate.mockReset().mockResolvedValue([]) autoUpdaterMock.quitAndInstall.mockReset() autoUpdaterMock.setFeedURL.mockReset() autoUpdaterMock.on.mockClear() @@ -119,8 +133,10 @@ describe('linux package recovery actions', () => { resolveLinuxPackageInstallInstructionsMock .mockReset() .mockResolvedValue({ ok: true, command: "sudo apt install -- ''", packageFileName: 'p' }) - revalidateLinuxPackageForInstallMock.mockReset().mockResolvedValue({ ok: true }) - revealLinuxPackageMock.mockReset().mockResolvedValue({ ok: true }) + resolveLinuxPackageRevealTargetMock + .mockReset() + .mockResolvedValue({ ok: true, path: ARTIFACT.path }) + showItemInFolderMock.mockReset() }) const startUpdater = async (): Promise<{ @@ -135,13 +151,19 @@ describe('linux package recovery actions', () => { return { send, updater } } - /** Drives a pre-commit install failure so the status carries the recovery discriminant. */ - const failInstall = async (updater: typeof UpdaterModule): Promise => { - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error('Command failed, exited with code 127')) + const activateRecovery = async ( + updater: typeof UpdaterModule, + version = '1.0.61' + ): Promise => { + autoUpdaterMock.checkForUpdates.mockImplementationOnce(() => { + autoUpdaterMock.emit('checking-for-update') + queueMicrotask(() => autoUpdaterMock.emit('update-available', { version })) + return Promise.resolve(null) }) - updater.quitAndInstall() - await vi.advanceTimersByTimeAsync(100) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + updater.downloadUpdate() + autoUpdaterMock.emit('update-downloaded', { version }) } type ErrorStatus = Extract @@ -162,12 +184,12 @@ describe('linux package recovery actions', () => { 'No package install recovery is available.' ) expect(resolveLinuxPackageInstallInstructionsMock).not.toHaveBeenCalled() - expect(revealLinuxPackageMock).not.toHaveBeenCalled() + expect(resolveLinuxPackageRevealTargetMock).not.toHaveBeenCalled() }) - it('revalidates the retained package on every invocation', async () => { + it('validates the retained package on every invocation', async () => { const { updater } = await startUpdater() - await failInstall(updater) + await activateRecovery(updater) await expect(updater.getLinuxPackageInstallInstructions()).resolves.toEqual({ ok: true, @@ -181,16 +203,74 @@ describe('linux package recovery actions', () => { const recovery = { kind: 'linux-package-install', packageType: 'deb', - reason: 'package-install-failed', + reason: 'manual-install-required', version: '1.0.61' } expect(resolveLinuxPackageInstallInstructionsMock.mock.calls).toEqual([[recovery], [recovery]]) - expect(revealLinuxPackageMock.mock.calls).toEqual([[recovery], [recovery]]) + expect(resolveLinuxPackageRevealTargetMock.mock.calls).toEqual([[recovery], [recovery]]) + expect(showItemInFolderMock).toHaveBeenCalledTimes(2) + }) + + it('restores recovery after a recheck resolves without a terminal event', async () => { + const { send, updater } = await startUpdater() + await activateRecovery(updater) + send.mockClear() + + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(1_000) + + expect(send).toHaveBeenLastCalledWith('updater:status', MANUAL_INSTALL_STATUS) + await expect(updater.showLinuxPackage()).resolves.toBeUndefined() + }) + + it('restores recovery after a recheck fails', async () => { + const { send, updater } = await startUpdater() + await activateRecovery(updater) + autoUpdaterMock.checkForUpdates.mockRejectedValueOnce(new Error('offline')) + send.mockClear() + + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + + expect(send).toHaveBeenLastCalledWith('updater:status', MANUAL_INSTALL_STATUS) + await expect(updater.getLinuxPackageInstallInstructions()).resolves.toEqual({ + ok: true, + command: "sudo apt install -- ''", + packageFileName: 'p' + }) + }) + + it('restores recovery when a pinned check resolves to the current version', async () => { + const { send, updater } = await startUpdater() + await activateRecovery(updater) + send.mockClear() + + updater.checkForUpdatesFromMenu({ channel: 'stable', targetTag: 'v1.0.51' }) + await vi.advanceTimersByTimeAsync(0) + + expect(send).toHaveBeenLastCalledWith('updater:status', MANUAL_INSTALL_STATUS) + await expect(updater.showLinuxPackage()).resolves.toBeUndefined() + }) + + it('restores recovery when resolving a pinned check fails', async () => { + const { send, updater } = await startUpdater() + await activateRecovery(updater) + send.mockClear() + + updater.checkForUpdatesFromMenu({ channel: 'stable', targetTag: 'not-a-release-tag' }) + await vi.advanceTimersByTimeAsync(0) + + expect(send).toHaveBeenLastCalledWith('updater:status', MANUAL_INSTALL_STATUS) + await expect(updater.getLinuxPackageInstallInstructions()).resolves.toEqual({ + ok: true, + command: "sudo apt install -- ''", + packageFileName: 'p' + }) }) it('replaces the structured status when revalidation fails so stale actions die', async () => { const { send, updater } = await startUpdater() - await failInstall(updater) + await activateRecovery(updater) resolveLinuxPackageInstallInstructionsMock.mockResolvedValue({ ok: false, reason: 'hash-mismatch' @@ -203,6 +283,7 @@ describe('linux package recovery actions', () => { expect(clearTrackedLinuxPackageArtifactMock).toHaveBeenCalledTimes(1) const latest = errorStatuses(send).at(-1) expect(latest?.state === 'error' && latest.recovery).toBeUndefined() + expect(latest?.version).toBe('1.0.61') expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith( 'linux_package_recovery_unavailable', { reason: 'hash-mismatch', packageType: 'deb', version: '1.0.61' }, @@ -212,13 +293,13 @@ describe('linux package recovery actions', () => { await expect(updater.showLinuxPackage()).rejects.toThrow( 'No package install recovery is available.' ) - expect(revealLinuxPackageMock).not.toHaveBeenCalled() + expect(resolveLinuxPackageRevealTargetMock).not.toHaveBeenCalled() }) it('clears recovery for both actions once the package is gone', async () => { const { updater } = await startUpdater() - await failInstall(updater) - revealLinuxPackageMock.mockResolvedValue({ ok: false, reason: 'missing' }) + await activateRecovery(updater) + resolveLinuxPackageRevealTargetMock.mockResolvedValue({ ok: false, reason: 'missing' }) await expect(updater.showLinuxPackage()).rejects.toThrow('no longer in the update cache') @@ -231,7 +312,7 @@ describe('linux package recovery actions', () => { 'resolves %s as a result and keeps the card usable instead of rejecting', async (reason) => { const { send, updater } = await startUpdater() - await failInstall(updater) + await activateRecovery(updater) resolveLinuxPackageInstallInstructionsMock.mockResolvedValue({ ok: false, reason }) const statusesBefore = errorStatuses(send).length @@ -256,8 +337,10 @@ describe('linux package recovery actions', () => { it('keeps recovery available after a transient read failure', async () => { const { send, updater } = await startUpdater() - await failInstall(updater) - revealLinuxPackageMock.mockResolvedValue({ ok: false, reason: 'read-failed' }) + await activateRecovery(updater) + showItemInFolderMock.mockImplementationOnce(() => { + throw new Error('no file manager available') + }) const statusesBefore = errorStatuses(send).length await expect(updater.showLinuxPackage()).rejects.toThrow( @@ -267,13 +350,12 @@ describe('linux package recovery actions', () => { // Why: a read error is not evidence the artifact is bad, so retrying must stay possible. expect(clearTrackedLinuxPackageArtifactMock).not.toHaveBeenCalled() expect(errorStatuses(send)).toHaveLength(statusesBefore) - revealLinuxPackageMock.mockResolvedValue({ ok: true }) await expect(updater.showLinuxPackage()).resolves.toBeUndefined() }) - it('ignores a stale mismatch verdict once a newer recovery replaced the card', async () => { + it('ignores a stale mismatch after the same package cycle is captured again', async () => { const { send, updater } = await startUpdater() - await failInstall(updater) + await activateRecovery(updater) let settleValidation!: (result: { ok: false; reason: 'hash-mismatch' }) => void resolveLinuxPackageInstallInstructionsMock.mockReturnValue( new Promise((resolve) => { @@ -283,12 +365,51 @@ describe('linux package recovery actions', () => { const pending = updater.getLinuxPackageInstallInstructions() // A 160 MB hash outlives the cycle it started in; a newer download takes over meanwhile. - getTrackedLinuxPackageArtifactMock.mockReturnValue({ ...ARTIFACT, version: '1.0.62' }) - await failInstall(updater) + getTrackedLinuxPackageArtifactMock.mockReturnValue({ ...ARTIFACT }) + await activateRecovery(updater) settleValidation({ ok: false, reason: 'hash-mismatch' }) - await expect(pending).rejects.toThrow('no longer matches the verified release') + await expect(pending).rejects.toThrow('Package install recovery is no longer current.') expect(clearTrackedLinuxPackageArtifactMock).not.toHaveBeenCalled() - expect(errorStatuses(send).at(-1)?.recovery?.version).toBe('1.0.62') + expect(errorStatuses(send).at(-1)?.recovery?.version).toBe('1.0.61') + }) + + it('does not return stale instructions after a same-version recapture', async () => { + const { send, updater } = await startUpdater() + await activateRecovery(updater) + let settleValidation!: (result: { ok: true; command: string; packageFileName: string }) => void + resolveLinuxPackageInstallInstructionsMock.mockReturnValue( + new Promise((resolve) => { + settleValidation = resolve + }) + ) + + const pending = updater.getLinuxPackageInstallInstructions() + getTrackedLinuxPackageArtifactMock.mockReturnValue({ ...ARTIFACT }) + await activateRecovery(updater) + const statusesBefore = errorStatuses(send).length + settleValidation({ ok: true, command: 'stale command', packageFileName: 'stale.deb' }) + + await expect(pending).rejects.toThrow('Package install recovery is no longer current.') + expect(errorStatuses(send)).toHaveLength(statusesBefore) + }) + + it('does not reveal a stale path after a same-version recapture', async () => { + const { updater } = await startUpdater() + await activateRecovery(updater) + let settleValidation!: (result: { ok: true; path: string }) => void + resolveLinuxPackageRevealTargetMock.mockReturnValue( + new Promise((resolve) => { + settleValidation = resolve + }) + ) + + const pending = updater.showLinuxPackage() + getTrackedLinuxPackageArtifactMock.mockReturnValue({ ...ARTIFACT }) + await activateRecovery(updater) + settleValidation({ ok: true, path: ARTIFACT.path }) + + await expect(pending).rejects.toThrow('Package install recovery is no longer current.') + expect(showItemInFolderMock).not.toHaveBeenCalled() }) }) diff --git a/src/main/updater-mac-install.ts b/src/main/updater-mac-install.ts index e2441d64180..234cdc0b2c4 100644 --- a/src/main/updater-mac-install.ts +++ b/src/main/updater-mac-install.ts @@ -1,9 +1,66 @@ -import { app } from 'electron' +import { app, autoUpdater as nativeUpdater } from 'electron' import type { UpdateStatus } from '../shared/update-status-types' import { recordUpdaterLifecycle } from './updater-lifecycle-diagnostics' const MAC_INSTALL_READY_TIMEOUT_MS = 15000 +export function registerMacUpdaterEvents({ + getCurrentStatus, + hasInstallableDownloadedVersion, + getPendingInstallVersion, + getKnownReleaseUrl, + performQuitAndInstall, + shouldDeferMacQuitForInstall, + sendStatus +}: { + getCurrentStatus: () => UpdateStatus + hasInstallableDownloadedVersion: () => boolean + getPendingInstallVersion: () => string + getKnownReleaseUrl: () => string | undefined + performQuitAndInstall: () => void | Promise + shouldDeferMacQuitForInstall: () => boolean + sendStatus: (status: UpdateStatus) => void +}): void { + if (process.platform === 'darwin') { + nativeUpdater.on('update-downloaded', () => { + const hasInstallableVersion = hasInstallableDownloadedVersion() + handleMacInstallerReady(hasInstallableVersion, performQuitAndInstall, () => { + sendStatus({ + state: 'downloaded', + version: getPendingInstallVersion(), + releaseUrl: getKnownReleaseUrl() + }) + }) + }) + } + + app.on('before-quit', (event) => { + if (!shouldDeferMacQuitForInstall()) { + return + } + if (consumeMacInstallGuardBypass()) { + recordUpdaterLifecycle('macos_before_quit_guard_bypassed') + return + } + if (isMacQuitAndInstallInFlight()) { + return + } + if ( + deferMacQuitUntilInstallerReady( + getCurrentStatus(), + hasInstallableDownloadedVersion(), + getPendingInstallVersion, + sendStatus + ) + ) { + recordUpdaterLifecycle('macos_before_quit_deferred', { + version: getPendingInstallVersion() + }) + event.preventDefault() + } + }) +} + /** Whether Squirrel.Mac has finished downloading the update from the localhost proxy. */ let squirrelReady = false /** Remembers a user/app quit request that arrived before Squirrel.Mac had a diff --git a/src/main/updater-test-harness.ts b/src/main/updater-test-harness.ts index 4a3315862f3..36687d0a79e 100644 --- a/src/main/updater-test-harness.ts +++ b/src/main/updater-test-harness.ts @@ -4,6 +4,7 @@ import { clearTrackedRealTimers, trackRealTimers } from './updater-test-timer-tr /** Loose spy signature for the electron/electron-updater calls the suites only assert on. */ type UpdaterSpy = Mock<(...args: unknown[]) => unknown> +type LinuxPackageType = 'deb' | 'rpm' | 'non-root' | 'unusable' type AutoUpdaterMock = { autoDownload: boolean @@ -44,7 +45,11 @@ type UpdaterModuleFactories = { electronUpdaterLoader: () => { loadElectronAutoUpdater: () => AutoUpdaterMock } electronToolkitUtils: () => { is: { dev: boolean } } ipcPty: () => { killAllPty: UpdaterSpy } - linuxUpdatePackageType: () => { getLinuxRootPackageType: Mock<() => 'deb' | 'rpm' | null> } + linuxUpdatePackageType: () => { + getLinuxPackageType: Mock<() => LinuxPackageType> + getLinuxRootPackageType: Mock<() => 'deb' | 'rpm' | null> + isExternallyManagedLinuxInstall: Mock<() => boolean> + } updaterLifecycleDiagnostics: () => { recordUpdaterLifecycle: UpdaterSpy } updaterChangelog: () => { fetchChangelog: UpdaterSpy } updaterNudge: () => { fetchNudge: UpdaterSpy; shouldApplyNudge: UpdaterSpy } @@ -68,7 +73,9 @@ export type UpdaterMocks = { isMock: { dev: boolean } killAllPtyMock: UpdaterSpy powerMonitorOnMock: UpdaterSpy + getLinuxPackageTypeMock: Mock<() => LinuxPackageType> getLinuxRootPackageTypeMock: Mock<() => 'deb' | 'rpm' | null> + isExternallyManagedLinuxInstallMock: Mock<() => boolean> recordUpdaterLifecycleMock: UpdaterSpy fetchChangelogMock: UpdaterSpy fetchNudgeMock: UpdaterSpy @@ -206,6 +213,10 @@ export function createUpdaterMocks(): UpdaterMocks { const killAllPtyMock = vi.fn() const powerMonitorOnMock = vi.fn() const getLinuxRootPackageTypeMock = vi.fn<() => 'deb' | 'rpm' | null>(() => null) + const getLinuxPackageTypeMock = vi.fn<() => LinuxPackageType>(() => { + return getLinuxRootPackageTypeMock() ?? 'non-root' + }) + const isExternallyManagedLinuxInstallMock = vi.fn<() => boolean>(() => false) const recordUpdaterLifecycleMock = vi.fn() const fetchChangelogMock = vi.fn() const fetchNudgeMock = vi.fn() @@ -232,7 +243,11 @@ export function createUpdaterMocks(): UpdaterMocks { electronToolkitUtils: () => ({ is: isMock }), ipcPty: () => ({ killAllPty: killAllPtyMock }), // Why: only the marker resolver is faked so the real artifact capture/redaction path stays under test. - linuxUpdatePackageType: () => ({ getLinuxRootPackageType: getLinuxRootPackageTypeMock }), + linuxUpdatePackageType: () => ({ + getLinuxPackageType: getLinuxPackageTypeMock, + getLinuxRootPackageType: getLinuxRootPackageTypeMock, + isExternallyManagedLinuxInstall: isExternallyManagedLinuxInstallMock + }), updaterLifecycleDiagnostics: () => ({ recordUpdaterLifecycle: recordUpdaterLifecycleMock }), updaterChangelog: () => ({ fetchChangelog: fetchChangelogMock }), updaterNudge: () => ({ fetchNudge: fetchNudgeMock, shouldApplyNudge: shouldApplyNudgeMock }), @@ -276,6 +291,10 @@ export function createUpdaterMocks(): UpdaterMocks { disarmExitWatchdogMock.mockReset() powerMonitorOnMock.mockReset() getLinuxRootPackageTypeMock.mockReset().mockReturnValue(null) + getLinuxPackageTypeMock.mockReset().mockImplementation(() => { + return getLinuxRootPackageTypeMock() ?? 'non-root' + }) + isExternallyManagedLinuxInstallMock.mockReset().mockReturnValue(false) recordUpdaterLifecycleMock.mockReset() fetchNudgeMock.mockReset().mockResolvedValue(null) shouldApplyNudgeMock.mockReset().mockReturnValue(false) @@ -306,7 +325,9 @@ export function createUpdaterMocks(): UpdaterMocks { isMock, killAllPtyMock, powerMonitorOnMock, + getLinuxPackageTypeMock, getLinuxRootPackageTypeMock, + isExternallyManagedLinuxInstallMock, recordUpdaterLifecycleMock, fetchChangelogMock, fetchNudgeMock, diff --git a/src/main/updater.fallback.test.ts b/src/main/updater.fallback.test.ts index 9cbcc97c85e..767fef421e3 100644 --- a/src/main/updater.fallback.test.ts +++ b/src/main/updater.fallback.test.ts @@ -86,6 +86,23 @@ describe('statusesEqual', () => { ).toBe(false) expect(statusesEqual(withRecovery, { ...withRecovery })).toBe(true) }) + + it('delivers a same-valued recovery recaptured for a new package cycle', () => { + expect(statusesEqual(withRecovery, { ...withRecovery, recovery: { ...recovery } })).toBe(false) + }) + + it('separates generic errors by version and retryability', () => { + const error: UpdateStatus = { + state: 'error', + message: 'package unavailable', + version: '1.0.61', + retryable: false + } + + expect(statusesEqual(error, { ...error, version: '1.0.62' })).toBe(false) + expect(statusesEqual(error, { ...error, retryable: true })).toBe(false) + expect(statusesEqual(error, { ...error })).toBe(true) + }) }) describe('isReleaseAssetsPublishingFailure', () => { diff --git a/src/main/updater.headless-serve-install.test.ts b/src/main/updater.headless-serve-install.test.ts index 08e2f519861..bae6474cc66 100644 --- a/src/main/updater.headless-serve-install.test.ts +++ b/src/main/updater.headless-serve-install.test.ts @@ -77,6 +77,11 @@ vi.mock('electron', () => ({ vi.mock('electron-updater', () => ({ autoUpdater: autoUpdaterMock })) vi.mock('./electron-updater-loader', () => ({ loadElectronAutoUpdater: () => autoUpdaterMock })) +vi.mock('./linux-update-package-type', () => ({ + getLinuxPackageType: () => 'non-root', + getLinuxRootPackageType: () => null, + isExternallyManagedLinuxInstall: () => false +})) vi.mock('@electron-toolkit/utils', () => ({ is: { dev: false } })) vi.mock('./ipc/pty', () => ({ killAllPty: killAllPtyMock })) vi.mock('./updater-changelog', () => ({ fetchChangelog: vi.fn().mockResolvedValue(null) })) @@ -166,6 +171,7 @@ describe('headless serve update install handoff', () => { checkForUpdatesFromMenu() await vi.advanceTimersByTimeAsync(0) + autoUpdaterMock.emit('download-progress', { percent: 100 }) autoUpdaterMock.emit('update-downloaded', { version: pendingInstaller.version }) const nativeReadyHandler = nativeUpdaterMock.on.mock.calls.find( ([event]) => event === 'update-downloaded' diff --git a/src/main/updater.install-failure-cause.test.ts b/src/main/updater.install-failure-cause.test.ts index 6344a79d13b..bef63d6912a 100644 --- a/src/main/updater.install-failure-cause.test.ts +++ b/src/main/updater.install-failure-cause.test.ts @@ -101,6 +101,11 @@ vi.mock('./updater-nudge', () => ({ vi.mock('./updater-lifecycle-diagnostics', () => ({ recordUpdaterLifecycle: recordUpdaterLifecycleMock })) +vi.mock('./linux-update-package-type', () => ({ + getLinuxPackageType: () => 'non-root', + getLinuxRootPackageType: () => null, + isExternallyManagedLinuxInstall: () => false +})) // The real electron-updater DebUpdater failure text when elevation is impossible. const DEB_ELEVATION_ERROR = @@ -155,6 +160,8 @@ async function reachDownloaded(): Promise { autoUpdaterMock.emit('checking-for-update') autoUpdaterMock.emit('update-available', { version: '1.4.163' }) await new Promise((resolve) => setTimeout(resolve, 0)) + autoUpdaterMock.downloadUpdate.mockResolvedValue([]) + updater.downloadUpdate() autoUpdaterMock.emit('update-downloaded', { version: '1.4.163' }) expect(updater.getUpdateStatus().state).toBe('downloaded') return updater diff --git a/src/main/updater.linux-externally-managed.test.ts b/src/main/updater.linux-externally-managed.test.ts new file mode 100644 index 00000000000..0b63a8ec9ed --- /dev/null +++ b/src/main/updater.linux-externally-managed.test.ts @@ -0,0 +1,155 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as UpdaterModule from './updater' +import { loadUpdaterModule, warmUpdaterModule } from './updater-test-module-loader' +import type { LinuxRootPackageType, UpdateStatus } from '../shared/update-status-types' + +const { + autoUpdaterMock, + getLinuxRootPackageTypeMock, + isExternallyManagedLinuxInstallMock, + recordUpdaterLifecycleMock, + fetchNewerReleaseTagsMock, + moduleFactories, + resetUpdaterMocks +} = await vi.hoisted(async () => (await import('./updater-test-harness')).createUpdaterMocks()) + +vi.mock('electron', () => moduleFactories.electron()) +vi.mock('electron-updater', () => moduleFactories.electronUpdater()) +vi.mock('./electron-updater-loader', () => moduleFactories.electronUpdaterLoader()) +vi.mock('@electron-toolkit/utils', () => moduleFactories.electronToolkitUtils()) +vi.mock('./ipc/pty', () => moduleFactories.ipcPty()) +vi.mock('./linux-update-package-type', () => moduleFactories.linuxUpdatePackageType()) +vi.mock('./updater-lifecycle-diagnostics', () => moduleFactories.updaterLifecycleDiagnostics()) +vi.mock('./updater-changelog', () => moduleFactories.updaterChangelog()) +vi.mock('./updater-nudge', () => moduleFactories.updaterNudge()) +vi.mock('./update-install-exit-watchdog', () => moduleFactories.updateInstallExitWatchdog()) +vi.mock('./updater-prerelease-feed', () => moduleFactories.updaterPrereleaseFeed()) +vi.mock('./local-builds/local-build-switch', () => moduleFactories.localBuildSwitch()) +vi.mock('./local-builds/local-build-feed-server', () => moduleFactories.localBuildFeedServer()) + +const EXTERNALLY_MANAGED_MESSAGE = + 'This copy of Orca is managed by your system package manager, so Orca cannot install updates itself. Update Orca through your distribution instead.' + +/** #17702: a repackaged install (AUR, Nix, container rebuild) inherits the .deb `package-type` + * marker but has no package manager that can apply an Orca-downloaded package. */ +warmUpdaterModule() + +describe('updater externally managed Linux installs', () => { + beforeEach(() => { + resetUpdaterMocks() + }) + + async function startUpdater(options: { + packageType: LinuxRootPackageType | null + externallyManaged: boolean + }): Promise<{ send: ReturnType; updater: typeof UpdaterModule }> { + getLinuxRootPackageTypeMock.mockReturnValue(options.packageType) + isExternallyManagedLinuxInstallMock.mockReturnValue(options.externallyManaged) + vi.useFakeTimers() + fetchNewerReleaseTagsMock.mockResolvedValue({ tags: ['v1.0.61'], state: 'ready' }) + autoUpdaterMock.checkForUpdates.mockImplementation(() => { + autoUpdaterMock.emit('checking-for-update') + queueMicrotask(() => autoUpdaterMock.emit('update-available', { version: '1.0.61' })) + return Promise.resolve(undefined) + }) + const send = vi.fn() + const updater = await loadUpdaterModule() + updater.setupAutoUpdater({ webContents: { send } } as never, { + getLastUpdateCheckAt: () => Date.now(), + installMode: 'interactive' + }) + return { send, updater } + } + + function lastStatus(send: ReturnType): UpdateStatus | undefined { + return send.mock.calls.findLast(([channel]) => channel === 'updater:status')?.[1] + } + + it('still reports the available release so the user can update through their distribution', async () => { + const { send, updater } = await startUpdater({ packageType: 'deb', externallyManaged: true }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + + expect(lastStatus(send)).toEqual({ + state: 'available', + version: '1.0.61', + changelog: null, + externallyManaged: true + }) + }) + + it('does not flag a real deb host', async () => { + const { send, updater } = await startUpdater({ packageType: 'deb', externallyManaged: false }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + + const status = lastStatus(send) + expect(status).toEqual({ state: 'available', version: '1.0.61', changelog: null }) + expect(status && 'externallyManaged' in status).toBe(false) + }) + + it('refuses the download instead of spending it on a package it can never install', async () => { + const { send, updater } = await startUpdater({ packageType: 'deb', externallyManaged: true }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + + updater.downloadUpdate() + await vi.advanceTimersByTimeAsync(0) + + expect(autoUpdaterMock.downloadUpdate).not.toHaveBeenCalled() + expect(lastStatus(send)).toEqual({ + state: 'error', + message: EXTERNALLY_MANAGED_MESSAGE, + version: '1.0.61', + retryable: false + }) + }) + + it('marks the refusal non-retryable so the card offers no Retry Download', async () => { + const { send, updater } = await startUpdater({ packageType: 'deb', externallyManaged: true }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + updater.downloadUpdate() + await vi.advanceTimersByTimeAsync(0) + + const status = lastStatus(send) + expect(status?.state === 'error' && status.retryable).toBe(false) + }) + + it('records the blocked download for field diagnosis', async () => { + const { updater } = await startUpdater({ packageType: 'deb', externallyManaged: true }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + updater.downloadUpdate() + await vi.advanceTimersByTimeAsync(0) + + expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith( + 'linux_package_externally_managed_download_blocked', + { version: '1.0.61' } + ) + }) + + it('leaves an ordinary deb host able to download', async () => { + const { updater } = await startUpdater({ packageType: 'deb', externallyManaged: false }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + autoUpdaterMock.downloadUpdate.mockResolvedValue([]) + + updater.downloadUpdate() + await vi.advanceTimersByTimeAsync(0) + + expect(autoUpdaterMock.downloadUpdate).toHaveBeenCalled() + }) + + it('leaves an AppImage host able to download', async () => { + const { updater } = await startUpdater({ packageType: null, externallyManaged: false }) + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + autoUpdaterMock.downloadUpdate.mockResolvedValue([]) + + updater.downloadUpdate() + await vi.advanceTimersByTimeAsync(0) + + expect(autoUpdaterMock.downloadUpdate).toHaveBeenCalled() + }) +}) diff --git a/src/main/updater.linux-root-package-install.test.ts b/src/main/updater.linux-root-package-install.test.ts index b52bf40f4c8..01f489bd8ef 100644 --- a/src/main/updater.linux-root-package-install.test.ts +++ b/src/main/updater.linux-root-package-install.test.ts @@ -1,12 +1,8 @@ -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { createHash } from 'node:crypto' -import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs' -import { tmpdir } from 'node:os' +import { beforeEach, describe, expect, it, vi } from 'vitest' import { join } from 'node:path' +import { tmpdir } from 'node:os' import type * as UpdaterModule from './updater' -import type * as RecoveryModule from './linux-package-update-recovery' -import type { UpdateStatus } from '../shared/update-status-types' -import { PRE_COMMIT_INSTALL_FAILURE } from './updater-test-harness' +import type { LinuxRootPackageType, UpdateStatus } from '../shared/update-status-types' import { loadUpdaterModule, warmUpdaterModule } from './updater-test-module-loader' const { @@ -14,10 +10,9 @@ const { nativeUpdaterMock, autoUpdaterMock, killAllPtyMock, + getLinuxPackageTypeMock, getLinuxRootPackageTypeMock, recordUpdaterLifecycleMock, - armExitWatchdogMock, - disarmExitWatchdogMock, fetchNewerReleaseTagsMock, moduleFactories, resetUpdaterMocks @@ -37,687 +32,222 @@ vi.mock('./updater-prerelease-feed', () => moduleFactories.updaterPrereleaseFeed vi.mock('./local-builds/local-build-switch', () => moduleFactories.localBuildSwitch()) vi.mock('./local-builds/local-build-feed-server', () => moduleFactories.localBuildFeedServer()) -type RevalidationVerdict = Awaited< - ReturnType -> +const packageSha512 = Buffer.alloc(64).toString('base64') -// Captured before any vi.useFakeTimers() call: the only handle left that still yields to libuv. -const realSetTimeout = globalThis.setTimeout - -type StagedLinuxPackages = { - cacheRoot: string - debPath: string - debSha512: string - rpmPath: string - rpmSha512: string -} - -/** - * Stages real packages inside a real updater cache: every install re-proves the retained digest by - * streaming the file off disk, so a path that never existed would abort before reaching the native - * updater. Returns the actual digests for the download events. - */ -function stageLinuxUpdateCache(): StagedLinuxPackages { - const cacheRoot = mkdtempSync(join(tmpdir(), 'orca-updater-cache-')) - const pendingDir = join(cacheRoot, 'orca-updater', 'pending') - mkdirSync(pendingDir, { recursive: true }) - const stagePackage = (fileName: string): { path: string; sha512: string } => { - const packagePath = join(pendingDir, fileName) - const bytes = Buffer.from(`orca test package ${fileName}`) - writeFileSync(packagePath, bytes) - return { path: packagePath, sha512: createHash('sha512').update(bytes).digest('base64') } - } - const deb = stagePackage('orca-ide_1.0.61_amd64.deb') - const rpm = stagePackage('orca-ide-1.0.61.x86_64.rpm') +function downloadedEvent(packageType: LinuxRootPackageType): Record { + const fileName = + packageType === 'deb' ? 'orca-ide_1.0.61_amd64.deb' : 'orca-ide-1.0.61.x86_64.rpm' return { - cacheRoot, - debPath: deb.path, - debSha512: deb.sha512, - rpmPath: rpm.path, - rpmSha512: rpm.sha512 - } -} - -type RevalidationProbe = { - /** Switch to held mode, where a verdict only lands when the test says so. Must precede startUpdater. */ - hold: () => void - settle: (verdict: RevalidationVerdict) => void - fail: (error: Error) => void - invocationCount: () => number - /** Resolves once every re-proof this test started has finished. */ - drain: () => Promise -} - -/** One outstanding re-proof; `awaitable` stays false while a held verdict has no way to settle. */ -type OutstandingRevalidation = { promise: Promise; awaitable: boolean } - -/** - * Wraps the pre-install re-proof so tests can await the real disk read instead of budgeting - * event-loop turns, and can hold a verdict open at an exact point in the cycle. Only that one call - * is wrapped — the artifact state stays real. - */ -function probeRevalidation(): RevalidationProbe { - type Artifact = Parameters[0] - let held = false - let invocationCount = 0 - let pending: { - resolve: (verdict: RevalidationVerdict) => void - reject: (error: Error) => void - entry: OutstandingRevalidation - } | null = null - const outstanding: OutstandingRevalidation[] = [] - - const track = (verdict: Promise, awaitable: boolean) => { - const noop = (): void => undefined - const entry: OutstandingRevalidation = { promise: verdict.then(noop, noop), awaitable } - outstanding.push(entry) - return entry - } - - vi.doMock('./linux-package-update-recovery', async () => { - const actual = await vi.importActual('./linux-package-update-recovery') - return { - ...actual, - revalidateLinuxPackageForInstall: vi.fn((artifact: Artifact) => { - invocationCount += 1 - if (!held) { - const verdict = actual.revalidateLinuxPackageForInstall(artifact) - track(verdict, true) - return verdict - } - let resolve!: (verdict: RevalidationVerdict) => void - let reject!: (error: Error) => void - const verdict = new Promise((res, rej) => { - resolve = res - reject = rej - }) - pending = { resolve, reject, entry: track(verdict, false) } - return verdict - }) - } - }) - - const release = (): typeof pending => { - const current = pending - if (current) { - current.entry.awaitable = true - pending = null - } - return current - } - - return { - hold: () => { - held = true - }, - settle: (verdict) => release()?.resolve(verdict), - fail: (error) => release()?.reject(error), - invocationCount: () => invocationCount, - drain: async () => { - // A verdict still held open can never settle on its own, so draining skips it. - let ready = outstanding.filter((entry) => entry.awaitable) - while (ready.length > 0) { - for (const entry of ready) { - outstanding.splice(outstanding.indexOf(entry), 1) - } - await Promise.all(ready.map((entry) => entry.promise)) - ready = outstanding.filter((entry) => entry.awaitable) - } - } + version: '1.0.61', + downloadedFile: join(tmpdir(), 'orca-updater', 'pending', fileName), + files: [{ url: fileName, sha512: packageSha512 }] } } warmUpdaterModule() -describe('updater', () => { +describe('updater Linux root packages', () => { beforeEach(() => { resetUpdaterMocks() }) - describe('linux root package install recovery', () => { - let staged: StagedLinuxPackages - let EXIT_127: string - let revalidation: RevalidationProbe + async function startUpdater( + packageType: LinuxRootPackageType | null, + installMode: UpdaterModule.UpdateInstallMode = 'interactive' + ): Promise<{ send: ReturnType; updater: typeof UpdaterModule }> { + getLinuxRootPackageTypeMock.mockReturnValue(packageType) + vi.useFakeTimers() + fetchNewerReleaseTagsMock.mockResolvedValue({ tags: ['v1.0.61'], state: 'ready' }) + autoUpdaterMock.checkForUpdates.mockImplementation(() => { + autoUpdaterMock.emit('checking-for-update') + queueMicrotask(() => autoUpdaterMock.emit('update-available', { version: '1.0.61' })) + return Promise.resolve(undefined) + }) + const send = vi.fn() + const updater = await loadUpdaterModule() + updater.setupAutoUpdater({ webContents: { send } } as never, { + getLastUpdateCheckAt: () => Date.now(), + installMode + }) + return { send, updater } + } - // Why: the quit timer needs fake time, and the work it starts needs real event-loop turns — - // fake timers never advance libuv. The re-proof itself is awaited rather than counted out - // (#15243): its disk read is wall-clock bound, so a loaded runner outlasts any turn budget and - // the tail lands in the next test. - const settleQuitAndInstall = async (): Promise => { - await vi.advanceTimersByTimeAsync(100) - await revalidation.drain() - // Full Node 26 shards can briefly starve the libuv poll phase while other workers transform - // tests; keep the operation alive long enough to avoid leaking it into the next test. - for (let turn = 0; turn < 200; turn += 1) { - await new Promise((resolve) => realSetTimeout(resolve, 0)) - } - await vi.advanceTimersByTimeAsync(0) + function lastStatus(send: ReturnType): UpdateStatus | undefined { + return send.mock.calls.findLast(([channel]) => channel === 'updater:status')?.[1] + } + + function markMacInstallerReady(): void { + if (process.platform !== 'darwin') { + return } + const handler = nativeUpdaterMock.on.mock.calls.find( + ([eventName]) => eventName === 'update-downloaded' + )?.[1] as (() => void) | undefined + handler?.() + } - beforeEach(() => { - staged = stageLinuxUpdateCache() - vi.stubEnv('XDG_CACHE_HOME', staged.cacheRoot) - EXIT_127 = `Command failed: /usr/bin/pkexec /usr/bin/dpkg -i ${staged.debPath}, exited with code 127` - revalidation = probeRevalidation() - }) - - afterEach(async () => { - // Why: an unfinished re-proof keeps running against this test's module instance, whose mocks - // are the same singletons the next test asserts on — it would double every install-path count. - await revalidation.drain() - vi.doUnmock('./linux-package-update-recovery') - vi.unstubAllEnvs() - rmSync(staged.cacheRoot, { recursive: true, force: true }) - }) - - const lastStatus = (send: ReturnType): UpdateStatus | undefined => - send.mock.calls.findLast(([channel]) => channel === 'updater:status')?.[1] - - const PRE_COMMIT_FAILURE_MESSAGE = PRE_COMMIT_INSTALL_FAILURE - const AGENT_STDERR = - 'pkexec: Error executing command as another user: No authentication agent found.' - - const downloadedEvent = (overrides?: Record): Record => ({ - version: '1.0.61', - downloadedFile: staged.debPath, - files: [{ url: 'orca-ide_1.0.61_amd64.deb', sha512: staged.debSha512 }], - ...overrides - }) - - const rpmDownloadedEvent = (): Record => - downloadedEvent({ - downloadedFile: staged.rpmPath, - files: [{ url: 'orca-ide-1.0.61.x86_64.rpm', sha512: staged.rpmSha512 }] - }) - - const startUpdater = async ( - packageType: 'deb' | 'rpm' | null - ): Promise<{ send: ReturnType; updater: typeof UpdaterModule }> => { - getLinuxRootPackageTypeMock.mockReturnValue(packageType) - vi.useFakeTimers() - fetchNewerReleaseTagsMock.mockResolvedValue({ tags: ['v1.0.61'], state: 'ready' }) - autoUpdaterMock.checkForUpdates.mockImplementation(() => { - autoUpdaterMock.emit('checking-for-update') - queueMicrotask(() => autoUpdaterMock.emit('update-available', { version: '1.0.61' })) - return Promise.resolve(undefined) - }) - const send = vi.fn() - const updater = await loadUpdaterModule() - updater.setupAutoUpdater({ webContents: { send } } as never, { - getLastUpdateCheckAt: () => Date.now() - }) - return { send, updater } + async function reachDownloaded( + updater: typeof UpdaterModule, + event: Record, + markInstallerReady = false + ): Promise { + updater.checkForUpdatesFromMenu() + await vi.advanceTimersByTimeAsync(0) + autoUpdaterMock.downloadUpdate.mockResolvedValue([]) + updater.downloadUpdate() + autoUpdaterMock.emit('update-downloaded', event) + if (markInstallerReady) { + markMacInstallerReady() } + await vi.advanceTimersByTimeAsync(0) + } - const reachDownloaded = async ( - updater: typeof UpdaterModule, - event: Record - ): Promise => { - updater.checkForUpdatesFromMenu() - await vi.advanceTimersByTimeAsync(0) - autoUpdaterMock.emit('update-downloaded', event) - if (process.platform === 'darwin') { - const nativeReady = nativeUpdaterMock.on.mock.calls.find( - ([eventName]) => eventName === 'update-downloaded' - )?.[1] as (() => void) | undefined - nativeReady?.() - } - await vi.advanceTimersByTimeAsync(0) - } - - it('disables install-on-quit for deb and rpm root packages', async () => { - for (const packageType of ['deb', 'rpm'] as const) { - vi.resetModules() - autoUpdaterMock.autoInstallOnAppQuit = true - getLinuxRootPackageTypeMock.mockReturnValue(packageType) - const { setupAutoUpdater } = await loadUpdaterModule() - - setupAutoUpdater({ webContents: { send: vi.fn() } } as never, { - getLastUpdateCheckAt: () => Date.now(), - installMode: 'interactive' - }) - - expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(false) - } - }) - - it('keeps interactive install-on-quit when no root-package marker is present', async () => { - autoUpdaterMock.autoInstallOnAppQuit = false - const { setupAutoUpdater } = await loadUpdaterModule() - - setupAutoUpdater({ webContents: { send: vi.fn() } } as never, { - getLastUpdateCheckAt: () => Date.now(), - installMode: 'interactive' - }) - - expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(true) - }) - - it('leaves headless serve installs supervisor-controlled', async () => { - for (const installMode of [ - 'supervised-headless-serve', - 'unsupported-headless-serve' - ] as const) { - vi.resetModules() - autoUpdaterMock.autoInstallOnAppQuit = true - const { setupAutoUpdater } = await loadUpdaterModule() - - setupAutoUpdater({ webContents: { send: vi.fn() } } as never, { - getLastUpdateCheckAt: () => Date.now(), - installMode - }) - - expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(false) - } - }) - - it('sends structured recovery when quitAndInstall throws synchronously', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.logger?.error(`${AGENT_STDERR} target ${staged.debPath}`) - throw new Error(EXIT_127) - }) - - updater.quitAndInstall() - await settleQuitAndInstall() - - // Why: the sync throw ends capture before the catch, so the stashed text must survive. - expect(send).toHaveBeenCalledWith('updater:status', { - state: 'error', - message: `${AGENT_STDERR} target `, - recovery: { - kind: 'linux-package-install', - packageType: 'deb', - reason: 'authentication-agent-unavailable', - version: '1.0.61' - } - }) - }) - - it('recovers an event-driven pre-commit failure without tearing down the session', async () => { + it.each(['deb', 'rpm'] as const)( + 'hands off %s installs without invoking the native updater', + async (packageType) => { const openWindow = { removeAllListeners: vi.fn() } browserWindowMock.getAllWindows.mockReturnValue([openWindow] as never) - const { send, updater } = await startUpdater('rpm') - await reachDownloaded(updater, rpmDownloadedEvent()) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error('Command failed, exited with code 1')) - }) + const { send, updater } = await startUpdater(packageType) - updater.quitAndInstall() - await settleQuitAndInstall() + await reachDownloaded(updater, downloadedEvent(packageType)) - expect(send).toHaveBeenCalledWith('updater:status', { + expect(lastStatus(send)).toEqual({ state: 'error', - message: 'Command failed, exited with code 1', + message: 'Quit Orca before running the system package install command.', recovery: { kind: 'linux-package-install', - packageType: 'rpm', - reason: 'package-install-failed', + packageType, + reason: 'manual-install-required', version: '1.0.61' } }) + + updater.quitAndInstall() + await vi.advanceTimersByTimeAsync(100) + + expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() expect(killAllPtyMock).not.toHaveBeenCalled() expect(openWindow.removeAllListeners).not.toHaveBeenCalled() expect(updater.isQuittingForUpdate()).toBe(false) - expect(disarmExitWatchdogMock).toHaveBeenCalled() - }) - - it('keeps the generic install-failure copy when no artifact was retained', async () => { - const { send, updater } = await startUpdater('deb') - // Release metadata without a digest must not enable cached-package recovery. - await reachDownloaded( - updater, - downloadedEvent({ files: [{ url: 'orca-ide_1.0.61_amd64.deb' }] }) - ) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error(EXIT_127)) - }) - - updater.quitAndInstall() - await settleQuitAndInstall() - - expect(send).toHaveBeenCalledWith('updater:status', { - state: 'error', - message: `${PRE_COMMIT_FAILURE_MESSAGE} (${EXIT_127})` - }) - expect(recordUpdaterLifecycleMock).not.toHaveBeenCalledWith( - 'linux_package_install_failed', - expect.anything(), - expect.anything() - ) - }) - - it('advises a restart only for a failure before the native invoke', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - // The source calls this out as the pre-native "cleanup/tracing exception" case. - recordUpdaterLifecycleMock.mockImplementation((event: unknown) => { - if (event === 'quit_and_install_invoking_native') { - throw new Error('tracing sink unavailable') - } - }) - - updater.quitAndInstall() - await settleQuitAndInstall() - - expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() - expect(send).toHaveBeenCalledWith('updater:status', { - state: 'error', - message: 'Could not restart to install the update. Quit and reopen Orca, then try again.' - }) - expect(updater.isQuittingForUpdate()).toBe(false) - }) - - it('keeps a committed install intact when post-commit cleanup throws', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - // Why: spawnSync already installed the package; a teardown throw must not be reported as failure. - killAllPtyMock.mockImplementation(() => { - throw new Error('pty teardown failed') - }) - send.mockClear() - - updater.quitAndInstall() - await settleQuitAndInstall() - - expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith( - 'post_commit_cleanup_failed', - { errorType: 'Error' }, - expect.objectContaining({ level: 'warn' }) - ) - expect(send).not.toHaveBeenCalled() - expect(updater.isQuittingForUpdate()).toBe(true) - expect(armExitWatchdogMock).toHaveBeenCalledTimes(1) - expect(disarmExitWatchdogMock).not.toHaveBeenCalled() - }) - - it('still suppresses late post-commit errors while an artifact is retained', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - - updater.quitAndInstall() - await settleQuitAndInstall() - expect(killAllPtyMock).toHaveBeenCalledTimes(1) - - send.mockClear() - autoUpdaterMock.emit('error', new Error(EXIT_127)) - - expect(send).not.toHaveBeenCalled() - expect(updater.isQuittingForUpdate()).toBe(true) - }) - - it('retries the automatic install without redownloading the package', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error(EXIT_127)) - }) - - updater.quitAndInstall() - await settleQuitAndInstall() - updater.quitAndInstall() - await settleQuitAndInstall() - - // Why: a retry usually fails identically; a deduped status would strand the preload restart relay. - expect( - send.mock.calls.filter( - ([channel, status]) => - channel === 'updater:status' && - (status as { recovery?: { kind?: string } })?.recovery?.kind === 'linux-package-install' - ) - ).toHaveLength(2) - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(2) - expect(autoUpdaterMock.downloadUpdate).not.toHaveBeenCalled() - expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith('linux_package_recovery_requested', { - action: 'retry-automatic', - packageType: 'deb', - version: '1.0.61' - }) - }) - - // Why: the cache path is user-writable, so the bytes verified when the recovery card - // rendered are not necessarily the bytes a root package manager would read on retry. - it('aborts the retry when the retained package no longer matches its digest', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - - // The escalation fails, which is what puts the recovery card (and its retry) on screen. - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error(EXIT_127)) - }) - updater.quitAndInstall() - await settleQuitAndInstall() - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) - - // A local process swaps the verified package for its own between failure and retry. - writeFileSync(staged.debPath, Buffer.from('attacker supplied package')) - send.mockClear() - killAllPtyMock.mockClear() - - updater.quitAndInstall() - await settleQuitAndInstall() - - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) - expect(killAllPtyMock).not.toHaveBeenCalled() - expect(updater.isQuittingForUpdate()).toBe(false) - expect(send).toHaveBeenCalledWith('updater:status', { - state: 'error', - message: - 'The downloaded package no longer matches the verified release, so Orca will not hand it to a package manager. Download the update again, or get it from the official release page.' - }) - expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith( - 'linux_package_revalidation_failed', - expect.objectContaining({ action: 'retry-automatic', reason: 'hash-mismatch' }), - expect.anything() - ) - }) - - // Why: "Restart to Update" is the common path and can sit unclicked for hours, so the same - // user-writable package reaches a root installer with a far longer window than any retry. - it('aborts the first install when the downloaded package was swapped', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - writeFileSync(staged.debPath, Buffer.from('attacker supplied package')) - send.mockClear() - - updater.quitAndInstall() - await settleQuitAndInstall() - - expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() - expect(killAllPtyMock).not.toHaveBeenCalled() - expect(updater.isQuittingForUpdate()).toBe(false) - expect(send).toHaveBeenCalledWith('updater:status', { - state: 'error', - message: - 'The downloaded package no longer matches the verified release, so Orca will not hand it to a package manager. Download the update again, or get it from the official release page.' - }) expect(send).toHaveBeenCalledWith('updater:quitAndInstallAborted') expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith( - 'linux_package_revalidation_failed', - expect.objectContaining({ action: 'restart-to-install', reason: 'hash-mismatch' }), - expect.anything() + 'linux_package_manual_install_required', + { packageType, version: '1.0.61' } ) - }) + } + ) - it('installs normally when the retained package still matches its digest', async () => { - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) + it('guards the native boundary even when no package artifact was retained', async () => { + const { send, updater } = await startUpdater('deb') - updater.quitAndInstall() - await settleQuitAndInstall() + updater.quitAndInstall() + await vi.advanceTimersByTimeAsync(100) - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) - expect(killAllPtyMock).toHaveBeenCalledTimes(1) - // Why: an abort push here would clear the restart flag mid-quit and re-arm the dirty-buffer - // prompt against the install that is already committed. - expect(send).not.toHaveBeenCalledWith('updater:quitAndInstallAborted') - expect(recordUpdaterLifecycleMock).not.toHaveBeenCalledWith( - 'linux_package_revalidation_failed', - expect.anything(), - expect.anything() - ) - }) + expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() + expect(killAllPtyMock).not.toHaveBeenCalled() + expect(send).toHaveBeenCalledWith('updater:quitAndInstallAborted') + }) - // Why: hashing 160 MB outlives the cycle it started in, and Check for Updates stays enabled - // while it runs — a verdict from the old cycle must not replace the card that took over. - it('drops an abort verdict once a newer check replaced the card', async () => { - revalidation.hold() - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - - updater.quitAndInstall() - await vi.advanceTimersByTimeAsync(100) - // The user gives up waiting and checks again; that check owns the card from here. - updater.checkForUpdatesFromMenu() - await vi.advanceTimersByTimeAsync(0) - expect(lastStatus(send)).toMatchObject({ state: 'available', version: '1.0.61' }) - - revalidation.settle({ ok: false, reason: 'hash-mismatch' }) - await settleQuitAndInstall() - - // The install is still abandoned — only the stale status is withheld. - expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() - expect(lastStatus(send)).toMatchObject({ state: 'available', version: '1.0.61' }) - // Withholding the status must not also withhold the abort: the renderer armed its restart and - // would otherwise skip its unsaved-work prompt for the rest of the session. - expect(send).toHaveBeenCalledWith('updater:quitAndInstallAborted') - expect(recordUpdaterLifecycleMock).toHaveBeenCalledWith( - 'linux_package_revalidation_failed', - expect.objectContaining({ reason: 'hash-mismatch' }), - expect.anything() - ) - }) - - // Why: EMFILE/EIO during the stream says nothing about the bytes, so the copy must not claim - // the package changed and the card must keep the actions that still work. - it('keeps the recovery card usable when the re-proof cannot read the package', async () => { - revalidation.hold() - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error(EXIT_127)) - }) - updater.quitAndInstall() - await vi.advanceTimersByTimeAsync(100) - revalidation.settle({ ok: true }) - await settleQuitAndInstall() - send.mockClear() - - updater.quitAndInstall() - await vi.advanceTimersByTimeAsync(100) - revalidation.settle({ ok: false, reason: 'read-failed' }) - await settleQuitAndInstall() - - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) - expect(lastStatus(send)).toEqual({ - state: 'error', - message: - 'Orca could not read the downloaded package. Download the update again, or get it from the official release page.', - recovery: { - kind: 'linux-package-install', - packageType: 'deb', - reason: 'package-install-failed', - version: '1.0.61' - } - }) - }) - - // Why: the re-proof runs before performQuitAndInstall's own error handling, so a rejection - // there would strand the quit timer and make every later install a silent no-op. - it('stays installable after a re-proof that rejects outright', async () => { - revalidation.hold() - const { send, updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - - updater.quitAndInstall() - await vi.advanceTimersByTimeAsync(100) - revalidation.fail(new Error('hash worker crashed')) - await settleQuitAndInstall() - - // Fails closed: an unprovable package is not handed to a root package manager. - expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() - expect(lastStatus(send)).toMatchObject({ - state: 'error', - message: - 'Orca could not read the downloaded package. Download the update again, or get it from the official release page.' - }) - - updater.quitAndInstall() - await vi.advanceTimersByTimeAsync(100) - revalidation.settle({ ok: true }) - await settleQuitAndInstall() - - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) - }) - - // Why: a second click during the multi-second hash must not schedule a parallel install. - it('ignores a second install request while the digest re-proof runs', async () => { - revalidation.hold() - const { updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - - updater.quitAndInstall() - // Fires the quit timer, which starts the re-proof; its verdict is still outstanding. - await vi.advanceTimersByTimeAsync(100) - expect(revalidation.invocationCount()).toBe(1) - updater.quitAndInstall() - // Advancing here proves the second request never scheduled its own quit timer. - await vi.advanceTimersByTimeAsync(100) - expect(revalidation.invocationCount()).toBe(1) - revalidation.settle({ ok: true }) - await settleQuitAndInstall() - - expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) - }) - - it('records classification-only lifecycle data for a package install failure', async () => { - const { updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.logger?.error(`${AGENT_STDERR} target ${staged.debPath}`) - autoUpdaterMock.emit('error', new Error(EXIT_127)) - }) - - updater.quitAndInstall() - await settleQuitAndInstall() - - const failure = recordUpdaterLifecycleMock.mock.calls.find( - ([event]) => event === 'linux_package_install_failed' - ) - expect(failure?.[1]).toEqual({ - packageType: 'deb', - reason: 'authentication-agent-unavailable', - exitCode: 127, + it('preserves the normal AppImage install path when no root-package marker is present', async () => { + const { send, updater } = await startUpdater(null) + await reachDownloaded( + updater, + { version: '1.0.61', - errorType: 'Error' - }) - const durable = JSON.stringify(recordUpdaterLifecycleMock.mock.calls) - expect(durable).not.toContain(staged.debPath) - expect(durable).not.toContain('authentication agent') - }) + downloadedFile: join(tmpdir(), 'Orca-1.0.61.AppImage'), + files: [] + }, + true + ) - it('omits exitCode from lifecycle data when the child status is unparseable', async () => { - const { updater } = await startUpdater('deb') - await reachDownloaded(updater, downloadedEvent()) - autoUpdaterMock.quitAndInstall.mockImplementation(() => { - autoUpdaterMock.emit('error', new Error('dpkg was interrupted')) - }) + updater.quitAndInstall() + await vi.advanceTimersByTimeAsync(100) - updater.quitAndInstall() - await settleQuitAndInstall() - - const failure = recordUpdaterLifecycleMock.mock.calls.find( - ([event]) => event === 'linux_package_install_failed' + expect(autoUpdaterMock.quitAndInstall).toHaveBeenCalledTimes(1) + expect(killAllPtyMock).toHaveBeenCalledTimes(1) + expect(send).not.toHaveBeenCalledWith('updater:quitAndInstallAborted') + expect( + send.mock.calls.some( + ([channel, status]) => + channel === 'updater:status' && + (status as UpdateStatus).state === 'error' && + (status as Extract).recovery?.kind === + 'linux-package-install' ) - // Why: an absent key, not an explicit null, keeps the breadcrumb schema honest. - expect(failure?.[1]).toEqual({ - packageType: 'deb', - reason: 'package-install-failed', - version: '1.0.61', - errorType: 'Error' + ).toBe(false) + }) + + it.each(['deb', 'rpm'] as const)( + 'disables install-on-quit and remote automatic control for %s builds', + async (packageType) => { + autoUpdaterMock.autoInstallOnAppQuit = true + const { updater } = await startUpdater(packageType) + + expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(false) + expect(updater.getRemoteServerUpdateSupport()).toEqual({ + installMode: 'interactive', + automatic: false, + reason: 'manual-service-update-required' }) - expect(Object.keys(failure?.[1] as object)).not.toContain('exitCode') + expect(() => updater.checkForRemoteServerUpdate('runtime-1')).toThrow( + 'remote_update_manual_required' + ) + } + ) + + it('keeps interactive install-on-quit and remote control for non-root packages', async () => { + autoUpdaterMock.autoInstallOnAppQuit = false + const { updater } = await startUpdater(null) + + expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(true) + expect(updater.getRemoteServerUpdateSupport()).toEqual({ + installMode: 'interactive', + automatic: true, + reason: 'available' }) }) + + it('fails closed for an unusable packaged marker', async () => { + getLinuxPackageTypeMock.mockReturnValue('unusable') + getLinuxRootPackageTypeMock.mockReturnValue(null) + autoUpdaterMock.autoInstallOnAppQuit = true + const { send, updater } = await startUpdater(null) + + expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(false) + expect(updater.getRemoteServerUpdateSupport()).toEqual({ + installMode: 'interactive', + automatic: false, + reason: 'manual-service-update-required' + }) + + await reachDownloaded(updater, { + version: '1.0.61', + downloadedFile: join(tmpdir(), 'orca-updater', 'pending', 'orca-ide_1.0.61_amd64.deb'), + files: [{ url: 'orca-ide_1.0.61_amd64.deb', sha512: packageSha512 }] + }) + expect(lastStatus(send)).toEqual({ + state: 'error', + message: + 'Orca could not verify the installed Linux package format, so it will not install this update automatically. Download the update from the official release page and install it manually.', + version: '1.0.61', + retryable: false + }) + + updater.quitAndInstall() + await vi.advanceTimersByTimeAsync(100) + expect(autoUpdaterMock.quitAndInstall).not.toHaveBeenCalled() + expect(send).toHaveBeenCalledWith('updater:quitAndInstallAborted') + }) + + it('leaves headless serve installs supervisor-controlled', async () => { + for (const installMode of [ + 'supervised-headless-serve', + 'unsupported-headless-serve' + ] as const) { + resetUpdaterMocks() + autoUpdaterMock.autoInstallOnAppQuit = true + await startUpdater(null, installMode) + expect(autoUpdaterMock.autoInstallOnAppQuit).toBe(false) + } + }) }) diff --git a/src/main/updater.mac-install.test.ts b/src/main/updater.mac-install.test.ts index e4fe26296a0..563b8562fd6 100644 --- a/src/main/updater.mac-install.test.ts +++ b/src/main/updater.mac-install.test.ts @@ -134,6 +134,7 @@ describe('updater mac install handoff', () => { appMock.isPackaged = true isMock.dev = false killAllPtyMock.mockReset() + autoUpdaterMock.downloadUpdate.mockResolvedValue([]) vi.unstubAllGlobals() vi.useRealTimers() }) @@ -145,7 +146,7 @@ describe('updater mac install handoff', () => { const mainWindow = { webContents: { send: sendMock } } autoUpdaterMock.checkForUpdates.mockResolvedValue(undefined) - const { setupAutoUpdater } = await loadUpdaterModule() + const { setupAutoUpdater, downloadUpdate } = await loadUpdaterModule() setupAutoUpdater(mainWindow as never) await vi.waitFor(() => { @@ -156,6 +157,7 @@ describe('updater mac install handoff', () => { // Why: the update-available handler is now async (it awaits fetchChangelog). // Flush microtasks so setAvailableVersion runs before update-downloaded fires. await new Promise((r) => setTimeout(r, 0)) + downloadUpdate() autoUpdaterMock.emit('update-downloaded', { version: '1.0.61' }) const preventDefault = vi.fn() @@ -197,7 +199,7 @@ describe('updater mac install handoff', () => { const mainWindow = { webContents: { send: vi.fn() } } autoUpdaterMock.checkForUpdates.mockResolvedValue(undefined) - const { setupAutoUpdater, quitAndInstall } = await loadUpdaterModule() + const { setupAutoUpdater, downloadUpdate, quitAndInstall } = await loadUpdaterModule() setupAutoUpdater(mainWindow as never, { onBeforeQuit }) await vi.waitFor(() => { @@ -206,6 +208,7 @@ describe('updater mac install handoff', () => { autoUpdaterMock.emit('checking-for-update') autoUpdaterMock.emit('update-available', { version: '1.0.61' }) await vi.advanceTimersByTimeAsync(0) + downloadUpdate() autoUpdaterMock.emit('update-downloaded', { version: '1.0.61' }) const preventDefault = vi.fn() @@ -279,7 +282,7 @@ describe('updater mac install handoff', () => { const mainWindow = { webContents: { send: sendMock } } autoUpdaterMock.checkForUpdates.mockResolvedValue(undefined) - const { setupAutoUpdater } = await loadUpdaterModule() + const { setupAutoUpdater, downloadUpdate } = await loadUpdaterModule() setupAutoUpdater(mainWindow as never) await vi.waitFor(() => { @@ -290,6 +293,7 @@ describe('updater mac install handoff', () => { // Why: the update-available handler is now async (it awaits fetchChangelog). // Flush microtasks so setAvailableVersion runs before update-downloaded fires. await vi.advanceTimersByTimeAsync(0) + downloadUpdate() autoUpdaterMock.emit('update-downloaded', { version: '1.0.61' }) const preventDefault = vi.fn() diff --git a/src/main/updater.quit-and-install.test.ts b/src/main/updater.quit-and-install.test.ts index 0d381ea473a..0080db58991 100644 --- a/src/main/updater.quit-and-install.test.ts +++ b/src/main/updater.quit-and-install.test.ts @@ -290,6 +290,7 @@ describe('updater', () => { }) }) + autoUpdaterMock.emit('download-progress', { percent: 100 }) autoUpdaterMock.emit('update-downloaded', { version: '1.0.61' }) // Why: on macOS install commits only once Squirrel is ready; mark it ready so this test covers the post-commit path on all platforms. diff --git a/src/main/updater.startup-scheduling.test.ts b/src/main/updater.startup-scheduling.test.ts index 46190de47da..ef72af27def 100644 --- a/src/main/updater.startup-scheduling.test.ts +++ b/src/main/updater.startup-scheduling.test.ts @@ -31,6 +31,7 @@ warmUpdaterModule() describe('updater', () => { beforeEach(() => { resetUpdaterMocks() + vi.useFakeTimers() }) it('does not load or configure electron-updater during dev setup', async () => { diff --git a/src/main/updater/updater-build-selection.ts b/src/main/updater/updater-build-selection.ts index 63222fb3101..a533b776a84 100644 --- a/src/main/updater/updater-build-selection.ts +++ b/src/main/updater/updater-build-selection.ts @@ -111,7 +111,7 @@ export abstract class UpdaterBuildSelection extends UpdaterMenuChecks { try { const target = resolveTargetBuild(channel, tag) if (compareVersions(target.version, app.getVersion()) === 0) { - this.sendStatus({ state: 'not-available', userInitiated: true }) + this.sendSettledCheckStatus({ state: 'not-available', userInitiated: true }) return } this.closeLocalBuildFeed() @@ -140,7 +140,7 @@ export abstract class UpdaterBuildSelection extends UpdaterMenuChecks { this.userInitiatedCheck = false this.clearAvailableUpdateContext() this.restoreReleaseUpdateSource() - this.sendStatus({ + this.sendSettledCheckStatus({ state: 'error', message: String((error as Error)?.message ?? error), userInitiated: true diff --git a/src/main/updater/updater-check-failure.ts b/src/main/updater/updater-check-failure.ts index 8ee2500c6a9..742897af6fd 100644 --- a/src/main/updater/updater-check-failure.ts +++ b/src/main/updater/updater-check-failure.ts @@ -34,7 +34,7 @@ export abstract class UpdaterCheckFailure extends UpdaterReleaseFeed { // Why: a failed pinned jump must hand the feed back before surfacing the error, or the pin blocks background checks for the process lifetime. this.clearAvailableUpdateContext() this.restoreReleaseUpdateSource() - this.sendStatus({ state: 'error', message, userInitiated }) + this.sendSettledCheckStatus({ state: 'error', message, userInitiated }) return } const failureKey = this.getCheckFailureKey(message, userInitiated) @@ -80,18 +80,19 @@ export abstract class UpdaterCheckFailure extends UpdaterReleaseFeed { this.scheduleAutomaticUpdateCheck(this.getAutomaticRetryInterval()) if (userInitiated) { // Why: a user click needs visible feedback (idle looks broken); distinguish incomplete releases from transport failures. - this.sendErrorStatus( - this.isStableReleaseNotReadyFailure(sourceError) + this.sendSettledCheckStatus({ + state: 'error', + message: this.isStableReleaseNotReadyFailure(sourceError) ? "A newer release isn't available for this device yet. Check again later." : "Couldn't reach the update server. Try again in a few minutes.", - true - ) + userInitiated: true + }) } else { if (this.isRetryableReleaseFeedPreflightFailure(sourceError)) { // Why: release probes can fail transiently; keep the campaign pending so the short retry can still show it. this.deferPendingUpdateNudgeUntilRetry() } - this.sendStatus({ state: 'idle' }) + this.sendSettledCheckStatus({ state: 'idle' }) } return } @@ -100,7 +101,7 @@ export abstract class UpdaterCheckFailure extends UpdaterReleaseFeed { if (!userInitiated) { this.scheduleAutomaticUpdateCheck(this.getAutomaticRetryInterval()) } - this.sendErrorStatus(message, userInitiated) + this.sendSettledCheckStatus({ state: 'error', message, userInitiated }) } this.pendingCheckFailureKey = failureKey diff --git a/src/main/updater/updater-check-state.ts b/src/main/updater/updater-check-state.ts index 8905029c75d..d7380c17305 100644 --- a/src/main/updater/updater-check-state.ts +++ b/src/main/updater/updater-check-state.ts @@ -1,6 +1,7 @@ import { writeMainThreadDiagnosticMarker } from '../diagnostics/main-thread-churn-probe' import { isWindowsSignatureCheckUnavailableFailure } from '../../shared/updater-windows-signature-check' import { recordUpdaterLifecycle } from '../updater-lifecycle-diagnostics' +import { getRetainedLinuxPackageManualInstallStatus } from '../linux-package-downloaded-status' import type { UpdateCheckOptions, UpdateStatus } from '../../shared/update-status-types' import type { UpdateCheckVariant } from './updater-types' import { UpdaterStatus } from './updater-status' @@ -229,7 +230,7 @@ export abstract class UpdaterCheckState extends UpdaterStatus { this.deferPendingUpdateNudgeUntilRetry() return } - this.sendStatus({ state: 'not-available', userInitiated }) + this.sendSettledCheckStatus({ state: 'not-available', userInitiated }) } } return @@ -239,7 +240,7 @@ export abstract class UpdaterCheckState extends UpdaterStatus { this.backgroundCheckPromotedToUserInitiated = false this.userInitiatedCheck = false this.completeSilentUpdateCheck(userInitiated) - this.sendStatus({ state: 'not-available', userInitiated }) + this.sendSettledCheckStatus({ state: 'not-available', userInitiated }) } protected handleSettledUpdateCheckPromise(attemptId: number): void { @@ -284,6 +285,21 @@ export abstract class UpdaterCheckState extends UpdaterStatus { this.sendStatus({ state: 'error', message, userInitiated }) } + /** + * Settles a check without discarding a retained manual-install card. A distro-managed host has a + * downloaded package it can still be told about, and the ordinary settle status would erase it. + */ + protected sendSettledCheckStatus(status: UpdateStatus): void { + const retainedStatus = getRetainedLinuxPackageManualInstallStatus() + if (retainedStatus) { + this.sendStatus(retainedStatus) + } else if (status.state === 'error') { + this.sendErrorStatus(status.message, status.userInitiated) + } else { + this.sendStatus(status) + } + } + protected abstract consumeMissingManifestPrereleaseFallbackResult(): { userInitiated: boolean } | null diff --git a/src/main/updater/updater-download-install.ts b/src/main/updater/updater-download-install.ts index a1627d3895c..455e56e6efa 100644 --- a/src/main/updater/updater-download-install.ts +++ b/src/main/updater/updater-download-install.ts @@ -1,5 +1,7 @@ import { beginMacUpdateDownload, deferMacQuitUntilInstallerReady } from '../updater-mac-install' import { recordUpdaterLifecycle } from '../updater-lifecycle-diagnostics' +import { isExternallyManagedLinuxInstall } from '../linux-update-package-type' +import { LINUX_PACKAGE_EXTERNALLY_MANAGED_MESSAGE } from '../linux-package-downloaded-status' import { QUIT_AND_INSTALL_DELAY_MS } from './updater-state' import { UpdaterRemoteStatus } from './updater-remote-status' @@ -10,22 +12,11 @@ export abstract class UpdaterDownloadInstall extends UpdaterRemoteStatus { this.localBuildSelectionInProgress || this.pinnedBuildSelectionInProgress || this.pendingQuitAndInstallTimer || - this.quitAndInstallInProgress || - // Why: the quit timer is already cleared while the pre-install digest re-proof streams, so without this a second click would schedule a parallel install of the same package. - this.linuxPackageRevalidationInFlight + this.quitAndInstallInProgress ) { return } - const retriedRecovery = this.getActiveLinuxPackageRecovery() - if (retriedRecovery) { - recordUpdaterLifecycle('linux_package_recovery_requested', { - action: 'retry-automatic', - packageType: retriedRecovery.packageType, - version: retriedRecovery.version - }) - } - if (this.deferHeadlessServeInstall('install', this.getPendingInstallVersion())) { return } @@ -66,6 +57,25 @@ export abstract class UpdaterDownloadInstall extends UpdaterRemoteStatus { if (!version) { return } + // Why: main owns this verdict, not the card — an older renderer or a direct IPC call must not be + // able to spend a package download that this host could never install. + if (isExternallyManagedLinuxInstall()) { + recordUpdaterLifecycle('linux_package_externally_managed_download_blocked', { + version + }) + // Why: a pinned jump resolves to 'release' on Linux (no dev-channel artifact is built for it), + // so refusing without unwinding would strand isPinnedBuildActive and silently kill every + // background check for the rest of the process. A no-op on the ordinary release path. + this.clearAvailableUpdateContext() + this.restoreReleaseUpdateSource() + this.sendStatus({ + state: 'error', + message: LINUX_PACKAGE_EXTERNALLY_MANAGED_MESSAGE, + version, + retryable: false + }) + return + } if (this.deferHeadlessServeInstall('download', version)) { return } diff --git a/src/main/updater/updater-install-execution.ts b/src/main/updater/updater-install-execution.ts index 4522e155865..257f8e4fa93 100644 --- a/src/main/updater/updater-install-execution.ts +++ b/src/main/updater/updater-install-execution.ts @@ -4,19 +4,15 @@ import { withUpdaterSpan } from '../observability/instrumentation' import { runWithLaunchPath } from '../startup/hydrate-shell-path' import { markMacQuitAndInstallInFlight, isMacInstallerReady } from '../updater-mac-install' import { armUpdateInstallExitWatchdog } from '../update-install-exit-watchdog' -import { getLinuxRootPackageType } from '../linux-update-package-type' -import { - beginLinuxPackageInstallDiagnosticCapture, - endLinuxPackageInstallDiagnosticCapture -} from '../linux-package-install-diagnostic' -import { getTrackedLinuxPackageArtifact } from '../linux-package-update-recovery' +import { getLinuxPackageType } from '../linux-update-package-type' +import { LINUX_PACKAGE_MARKER_UNUSABLE_MESSAGE } from '../linux-package-downloaded-status' import { recordUpdaterLifecycle } from '../updater-lifecycle-diagnostics' import { requestServeUpdateHandoff, failServeUpdateHandoff } from '../serve-update-handoff' import { UpdaterPackageRecovery } from './updater-package-recovery' export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { protected async performQuitAndInstall(): Promise { - if (this.quitAndInstallInProgress || this.linuxPackageRevalidationInFlight) { + if (this.quitAndInstallInProgress) { recordUpdaterLifecycle('quit_and_install_ignored', { reason: 'already-in-progress' }) return } @@ -30,20 +26,31 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { if (this.deferHeadlessServeInstall('install', pendingVersion)) { return } - // Why: the retained .deb/.rpm sits on a user-writable path that a root package manager is about - // to read, and nothing re-checks it after download. Re-prove it here — before any teardown — so a - // swapped or vanished package aborts instead of being installed as root. The synchronous guard - // keeps every non-Linux install on its existing timing. - if ( - getTrackedLinuxPackageArtifact() && - !(await this.proveRetainedLinuxPackage(pendingVersion)) - ) { - // Why: the renderer armed its restart before invoking, and it infers the abort from the error - // status — which a stale-cycle verdict deliberately withholds. Signal the abandon here, where - // it cannot depend on that decision, or the window keeps skipping its unsaved-work prompt. + const linuxPackageType = getLinuxPackageType() + if (linuxPackageType === 'deb' || linuxPackageType === 'rpm') { + recordUpdaterLifecycle('linux_package_manual_install_required', { + packageType: linuxPackageType, + version: pendingVersion || null + }) + // The preload prepares renderer state before invoking; explicitly release it when main refuses. this.mainWindowRef?.webContents.send('updater:quitAndInstallAborted') return } + if (linuxPackageType === 'unusable') { + recordUpdaterLifecycle( + 'linux_package_marker_unusable', + { version: pendingVersion || null }, + { level: 'warn', message: 'Linux package marker is unusable; native install blocked' } + ) + // The preload prepares renderer state before invoking; release it when the marker is unknown. + this.mainWindowRef?.webContents.send('updater:quitAndInstallAborted') + this.sendInstallFailureStatus({ + state: 'error', + message: LINUX_PACKAGE_MARKER_UNUSABLE_MESSAGE, + ...(pendingVersion ? { version: pendingVersion } : {}) + }) + return + } this.quitAndInstallInProgress = true markMacQuitAndInstallInFlight() @@ -98,22 +105,14 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { } // Why: mark before the call so a sync 'error' during quitAndInstall can recover; pre-native errors must not look like install failure. this.quitAndInstallNativeInvoked = true - // Why: invoke before killAllPty/removing close listeners so a sync 'error' (the "no filepath" path) can recover while windows and PTYs are intact. + // Why: invoke before killAllPty/removing close listeners so a sync 'error' can recover while windows and PTYs are intact. const supervisorOwnsRelaunch = this.updateInstallMode === 'supervised-headless-serve' - // Why: BaseUpdater logs child stderr but drops it from the 'error' event, so retain it for the span of this call. - beginLinuxPackageInstallDiagnosticCapture(getTrackedLinuxPackageArtifact()?.path ?? null) - try { - runWithLaunchPath(() => - this.getAutoUpdater().quitAndInstall(supervisorOwnsRelaunch, !supervisorOwnsRelaunch) - ) - } finally { - const diagnostic = endLinuxPackageInstallDiagnosticCapture() - // Why: a synchronous 'error' already consumed and reset this attempt; re-stashing would leak it into the next one. - this.lastInstallAttemptDiagnostic = this.quitAndInstallInProgress ? diagnostic : null - } + runWithLaunchPath(() => + this.getAutoUpdater().quitAndInstall(supervisorOwnsRelaunch, !supervisorOwnsRelaunch) + ) span.addEvent('native_quit_and_install_invoked') - // Why: quitAndInstall can synchronously clear quitAndInstallInProgress via recovery (Win/Linux dispatchError); skip destructive prep if it already ran. + // Why: quitAndInstall can synchronously clear quitAndInstallInProgress via dispatchError; skip destructive prep if it already ran. if (!this.quitAndInstallInProgress) { // Why: recovery already wrote the reason to currentStatus; a bare return would exit this span Success. span.fail( @@ -124,14 +123,6 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { return } - // Why: DebUpdater/RpmUpdater install through spawnSync, so a normal return already means the - // package is installed. Commit here or a throw in the cleanup below is reported as an install - // failure — offering a recovery card, and stale stderr, for an update that actually succeeded. - if (getLinuxRootPackageType() !== null) { - this.updateInstallCommitted = true - armUpdateInstallExitWatchdog() - } - killAllPty() span.addEvent('local_pty_kill_all') @@ -153,9 +144,7 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { } }) } catch (error) { - // Why: on Linux the package is already installed once quitAndInstall returns, and the installer is - // waiting for this process to exit. Tearing down here would disarm the exit watchdog (#4438), clear - // quittingForUpdate mid-quit, and tell the user an install failed that actually succeeded. + // Past commit the installer is waiting for this process to exit; keep the handoff and watchdog intact. if (this.updateInstallCommitted) { recordUpdaterLifecycle( 'post_commit_cleanup_failed', @@ -167,12 +156,7 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { ) return } - // Why: a pre-native cleanup/tracing exception is not a package install failure and must not be labelled as one. const quitAndInstallNativeInvokedBeforeReset = this.quitAndInstallNativeInvoked - const recoveryStatus = - quitAndInstallNativeInvokedBeforeReset && !this.updateInstallCommitted - ? this.buildLinuxPackageInstallFailureStatus(error) - : null failServeUpdateHandoff('Could not invoke the native updater.') this.resetQuitForUpdateState() recordUpdaterLifecycle( @@ -183,16 +167,13 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { message: 'Could not start update install' } ) - this.sendInstallFailureStatus( - recoveryStatus ?? { - state: 'error', - // Why: past the native invoke this is the same pre-commit failure the event path reports, so it gets the same copy; only a pre-native exception can be helped by a restart. - // A synchronous throw out of quitAndInstall carries the same installer text the 'error' event would have. - message: quitAndInstallNativeInvokedBeforeReset - ? this.withInstallFailureCause(this.getPreCommitInstallFailureMessage(), error) - : 'Could not restart to install the update. Quit and reopen Orca, then try again.' - } - ) + this.sendInstallFailureStatus({ + state: 'error', + // A synchronous throw carries the same installer text the 'error' event would have. + message: quitAndInstallNativeInvokedBeforeReset + ? this.withInstallFailureCause(this.getPreCommitInstallFailureMessage(), error) + : 'Could not restart to install the update. Quit and reopen Orca, then try again.' + }) } } @@ -205,10 +186,8 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { ) { return false } - const recoveryStatus = this.buildLinuxPackageInstallFailureStatus(error) failServeUpdateHandoff('The native updater rejected the install request.') this.resetQuitForUpdateState() - // Durable data carries classification only — the cause text stays on the status the user can read. recordUpdaterLifecycle( 'quit_and_install_failed_via_event', { errorType: error instanceof Error ? error.name : typeof error }, @@ -217,12 +196,10 @@ export abstract class UpdaterInstallExecution extends UpdaterPackageRecovery { message: 'Update install could not start; recovered app state' } ) - this.sendInstallFailureStatus( - recoveryStatus ?? { - state: 'error', - message: this.withInstallFailureCause(this.getPreCommitInstallFailureMessage(), error) - } - ) + this.sendInstallFailureStatus({ + state: 'error', + message: this.withInstallFailureCause(this.getPreCommitInstallFailureMessage(), error) + }) return true } } diff --git a/src/main/updater/updater-install-support.ts b/src/main/updater/updater-install-support.ts index 048f293f01e..175362dad05 100644 --- a/src/main/updater/updater-install-support.ts +++ b/src/main/updater/updater-install-support.ts @@ -35,6 +35,12 @@ export abstract class UpdaterInstallSupport extends UpdaterCheckState { if (this.currentStatus.state === 'downloading' || this.currentStatus.state === 'downloaded') { return this.currentStatus.version } + if ( + this.currentStatus.state === 'error' && + this.currentStatus.recovery?.kind === 'linux-package-install' + ) { + return this.currentStatus.recovery.version + } return '' } @@ -70,7 +76,6 @@ export abstract class UpdaterInstallSupport extends UpdaterCheckState { this.quittingForUpdate = false this.updateInstallCommitted = false this.quitAndInstallNativeInvoked = false - this.lastInstallAttemptDiagnostic = null disarmUpdateInstallExitWatchdog() resetMacInstallState() } diff --git a/src/main/updater/updater-menu-checks.ts b/src/main/updater/updater-menu-checks.ts index b76d5c19710..5e5294f29dc 100644 --- a/src/main/updater/updater-menu-checks.ts +++ b/src/main/updater/updater-menu-checks.ts @@ -74,7 +74,7 @@ export abstract class UpdaterMenuChecks extends UpdaterScheduling { this.userInitiatedCheck = false this.finishActiveUpdateCheckAttempt() this.recordCompletedUpdateCheck() - this.sendStatus({ state: 'not-available', userInitiated: true }) + this.sendSettledCheckStatus({ state: 'not-available', userInitiated: true }) return false } return launch() diff --git a/src/main/updater/updater-package-recovery.ts b/src/main/updater/updater-package-recovery.ts index 870d601a5f3..f33b5044e48 100644 --- a/src/main/updater/updater-package-recovery.ts +++ b/src/main/updater/updater-package-recovery.ts @@ -1,22 +1,16 @@ +import { shell } from 'electron' import { recordUpdaterLifecycle } from '../updater-lifecycle-diagnostics' import { getTrackedLinuxPackageArtifact, clearTrackedLinuxPackageArtifact, - revalidateLinuxPackageForInstall, resolveLinuxPackageInstallInstructions, - revealLinuxPackage, + resolveLinuxPackageRevealTarget, type LinuxPackageArtifact, type LinuxPackageRecoveryUnavailableReason } from '../linux-package-update-recovery' -import { - getLinuxPackageInstallDiagnostic, - parseLinuxPackageInstallExitCode, - redactLinuxPackageInstallText -} from '../linux-package-install-diagnostic' import type { LinuxPackageInstallInstructions, - LinuxPackageInstallRecovery, - UpdateStatus + LinuxPackageInstallRecovery } from '../../shared/update-status-types' import { UpdaterInstallSupport } from './updater-install-support' @@ -67,131 +61,47 @@ export abstract class UpdaterPackageRecovery extends UpdaterInstallSupport { ) } + /** Whether the card this action was invoked from still owns both the status and the artifact. */ + protected isCurrentLinuxPackageRecovery( + recovery: LinuxPackageInstallRecovery, + artifact: LinuxPackageArtifact | null + ): boolean { + return ( + this.getActiveLinuxPackageRecovery() === recovery && + getTrackedLinuxPackageArtifact() === artifact + ) + } + + protected assertCurrentLinuxPackageRecovery( + recovery: LinuxPackageInstallRecovery, + artifact: LinuxPackageArtifact | null + ): void { + if (!this.isCurrentLinuxPackageRecovery(recovery, artifact)) { + throw new Error('Package install recovery is no longer current.') + } + } + protected failLinuxPackageRecovery( recovery: LinuxPackageInstallRecovery, + artifact: LinuxPackageArtifact | null, reason: LinuxPackageRecoveryUnavailableReason ): never { + this.assertCurrentLinuxPackageRecovery(recovery, artifact) this.recordLinuxPackageRecoveryUnavailable(recovery, reason) const message = LINUX_PACKAGE_RECOVERY_MESSAGES[reason] - // Why: hashing 160 MB takes long enough for a new cycle to land. Acting on a stale verdict would - // destroy the newer artifact and clobber whatever card replaced this one. - const active = this.getActiveLinuxPackageRecovery() - const stillCurrent = - active?.version === recovery.version && active?.packageType === recovery.packageType - if (stillCurrent && RECOVERY_CLEARING_REASONS.includes(reason)) { + if (RECOVERY_CLEARING_REASONS.includes(reason)) { clearTrackedLinuxPackageArtifact() - this.sendStatus({ state: 'error', message }) + this.sendStatus({ state: 'error', message, version: recovery.version }) } throw new Error(message) } - /** - * Identifies the update cycle an install belongs to, so a verdict produced by a multi-second hash - * can be dropped when a newer cycle already replaced the card it would otherwise overwrite. - */ - protected getInstallCycleSignature(): string { - const recovery = this.getActiveLinuxPackageRecovery() - if (recovery) { - return `recovery:${recovery.packageType}:${recovery.version}` - } - return this.currentStatus.state === 'downloaded' - ? `downloaded:${this.currentStatus.version}` - : `state:${this.currentStatus.state}` - } - - /** - * Re-proves the retained package before the install starts. Returns false when the install must be - * abandoned; the artifact is only re-read here, so callers still own every teardown decision. - */ - protected async proveRetainedLinuxPackage(pendingVersion: string): Promise { - const artifact = getTrackedLinuxPackageArtifact() - if (!artifact) { - return true - } - // Why: an artifact retained from another cycle says nothing about the file electron-updater is - // about to install, so proving it would block a legitimate install on an unrelated digest. - if (pendingVersion && pendingVersion !== artifact.version) { - return true - } - const recovery = this.getActiveLinuxPackageRecovery() - const cycle = this.getInstallCycleSignature() - const reason = await this.revalidateRetainedLinuxPackage(artifact) - if (!reason) { - return true - } - this.reportLinuxPackageRevalidationFailure({ artifact, recovery, reason, cycle }) - return false - } - - /** The failing reason, or null when the retained package still matches its release digest. */ - protected async revalidateRetainedLinuxPackage( - artifact: LinuxPackageArtifact - ): Promise { - this.linuxPackageRevalidationInFlight = true - try { - const verdict = await revalidateLinuxPackageForInstall(artifact) - return verdict.ok ? null : verdict.reason - } catch (error) { - recordUpdaterLifecycle( - 'linux_package_revalidation_errored', - { errorType: error instanceof Error ? error.name : typeof error }, - { level: 'warn', message: 'Could not re-verify the retained update package' } - ) - // Why: fail closed — bytes we could not read are bytes we cannot hand to a root installer. - return 'read-failed' - } finally { - // Why: the invariant every install path depends on — a wedged flag would make quitAndInstall - // early-return for the rest of the session. - this.linuxPackageRevalidationInFlight = false - } - } - - protected reportLinuxPackageRevalidationFailure({ - artifact, - recovery, - reason, - cycle - }: { - artifact: LinuxPackageArtifact - recovery: LinuxPackageInstallRecovery | null - reason: LinuxPackageRecoveryUnavailableReason - cycle: string - }): void { - recordUpdaterLifecycle( - 'linux_package_revalidation_failed', - { - action: recovery ? 'retry-automatic' : 'restart-to-install', - packageType: artifact.packageType, - version: artifact.version, - reason - }, - { level: 'warn', message: 'Retained update package failed its pre-install digest check' } - ) - // Why: a package proven bad must not stay tracked, but a download that landed during the hash - // owns the slot now and destroying it would force a needless 160 MB redownload. - const clearsArtifact = RECOVERY_CLEARING_REASONS.includes(reason) - if (clearsArtifact && getTrackedLinuxPackageArtifact() === artifact) { - clearTrackedLinuxPackageArtifact() - } - // Why: same reasoning as failLinuxPackageRecovery — a verdict from a cycle that has since been - // replaced must not clobber whatever card the user is looking at now. - if (this.getInstallCycleSignature() !== cycle) { - return - } - this.sendInstallFailureStatus({ - state: 'error', - message: LINUX_PACKAGE_RECOVERY_MESSAGES[reason], - // Why: an unreadable file is not evidence the bytes changed, so the recovery card and its - // Copy/Show actions survive a transient I/O failure exactly as they do elsewhere. - ...(recovery && !clearsArtifact ? { recovery } : {}) - }) - } - protected async getLinuxPackageInstallInstructions(): Promise { const recovery = this.getActiveLinuxPackageRecovery() if (!recovery) { throw new Error('No package install recovery is available.') } + const artifact = getTrackedLinuxPackageArtifact() recordUpdaterLifecycle('linux_package_recovery_requested', { action: 'copy-command', packageType: recovery.packageType, @@ -202,6 +112,7 @@ export abstract class UpdaterPackageRecovery extends UpdaterInstallSupport { // Why: the renderer must distinguish "this machine has no package manager" (keep the card, promote // Show Package) from "the artifact is gone" (recovery is cleared and the card unmounts). if (result.reason === 'no-sudo' || result.reason === 'no-package-manager') { + this.assertCurrentLinuxPackageRecovery(recovery, artifact) this.recordLinuxPackageRecoveryUnavailable(recovery, result.reason) return { ok: false, @@ -209,8 +120,9 @@ export abstract class UpdaterPackageRecovery extends UpdaterInstallSupport { message: LINUX_PACKAGE_RECOVERY_MESSAGES[result.reason] } } - this.failLinuxPackageRecovery(recovery, result.reason) + this.failLinuxPackageRecovery(recovery, artifact, result.reason) } + this.assertCurrentLinuxPackageRecovery(recovery, artifact) return { ok: true, command: result.command, packageFileName: result.packageFileName } } @@ -219,56 +131,22 @@ export abstract class UpdaterPackageRecovery extends UpdaterInstallSupport { if (!recovery) { throw new Error('No package install recovery is available.') } + const artifact = getTrackedLinuxPackageArtifact() recordUpdaterLifecycle('linux_package_recovery_requested', { action: 'show-package', packageType: recovery.packageType, version: recovery.version }) - const result = await revealLinuxPackage(recovery) + const result = await resolveLinuxPackageRevealTarget(recovery) if (!result.ok) { - this.failLinuxPackageRecovery(recovery, result.reason) + this.failLinuxPackageRecovery(recovery, artifact, result.reason) } - } - - /** Builds a recoverable status when the native Linux package installer rejects a retained artifact. */ - protected buildLinuxPackageInstallFailureStatus(error: unknown): UpdateStatus | null { - const artifact = getTrackedLinuxPackageArtifact() - if (!artifact) { - return null - } - const pendingVersion = this.getPendingInstallVersion() - if (pendingVersion && pendingVersion !== artifact.version) { - return null - } - const diagnostic = getLinuxPackageInstallDiagnostic() ?? this.lastInstallAttemptDiagnostic - const reason = diagnostic?.reason ?? 'package-install-failed' - const exitCode = parseLinuxPackageInstallExitCode(error) - recordUpdaterLifecycle( - 'linux_package_install_failed', - { - packageType: artifact.packageType, - reason, - ...(exitCode === null ? {} : { exitCode }), - version: artifact.version, - errorType: error instanceof Error ? error.name : typeof error - }, - { level: 'warn', message: 'Linux package install failed; cached package retained' } - ) - const message = - diagnostic?.message ?? - (error instanceof Error - ? redactLinuxPackageInstallText(error.message, artifact.path) - : null) ?? - 'The system package installer did not start.' - return { - state: 'error', - message, - recovery: { - kind: 'linux-package-install', - packageType: artifact.packageType, - reason, - version: artifact.version - } + this.assertCurrentLinuxPackageRecovery(recovery, artifact) + // Why: this cache path belongs to the installed app host, not a workspace's SSH or WSL host. + try { + shell.showItemInFolder(result.path) + } catch { + this.failLinuxPackageRecovery(recovery, artifact, 'read-failed') } } } diff --git a/src/main/updater/updater-remote-status.ts b/src/main/updater/updater-remote-status.ts index 6e3db76b0c8..fc34d40b6d3 100644 --- a/src/main/updater/updater-remote-status.ts +++ b/src/main/updater/updater-remote-status.ts @@ -7,6 +7,7 @@ import type { RemoteServerUpdateSupport } from '../../shared/remote-server-update' import { hasServeUpdateSupervisor } from '../serve-update-handoff' +import { getLinuxPackageType } from '../linux-update-package-type' import { UpdaterNudge } from './updater-nudge' import type { UpdateInstallMode } from './updater-state' @@ -31,7 +32,13 @@ export abstract class UpdaterRemoteStatus extends UpdaterNudge { reason: 'updater-unavailable' } } - if (this.updateInstallMode === 'unsupported-headless-serve') { + const linuxPackageType = getLinuxPackageType() + if ( + this.updateInstallMode === 'unsupported-headless-serve' || + linuxPackageType === 'deb' || + linuxPackageType === 'rpm' || + linuxPackageType === 'unusable' + ) { return { installMode: this.updateInstallMode, automatic: false, diff --git a/src/main/updater/updater-setup.ts b/src/main/updater/updater-setup.ts index 21354f97af3..d60444d449a 100644 --- a/src/main/updater/updater-setup.ts +++ b/src/main/updater/updater-setup.ts @@ -12,7 +12,7 @@ import type { RemoteServerUpdaterSnapshot, RemoteServerUpdateSupport } from '../../shared/remote-server-update' -import { getLinuxRootPackageType } from '../linux-update-package-type' +import { getLinuxPackageType } from '../linux-update-package-type' import { createUpdaterDiagnosticLogger } from '../linux-package-install-diagnostic' import { registerAutoUpdaterHandlers } from '../updater-events' import { getServeUpdateHandoffFailure } from '../serve-update-handoff' @@ -143,9 +143,9 @@ export class UpdaterSetup extends UpdaterDownloadInstall { autoUpdater.disableDifferentialDownload = false } // Why: supervised serve installs require an explicit handoff; ordinary service quits must never install implicitly. - // Root Linux packages also opt out: an implicit quit-time escalation would fail after the UI is gone, leaving no recovery surface. + // Only an explicit AppImage/non-root marker may opt into electron-updater's implicit quit install. autoUpdater.autoInstallOnAppQuit = - this.updateInstallMode === 'interactive' && getLinuxRootPackageType() === null + this.updateInstallMode === 'interactive' && getLinuxPackageType() === 'non-root' // Why: MacUpdater ignores quitAndInstall arguments; the surviving CLI supervisor must be the only serve relaunch owner. autoUpdater.autoRunAppAfterInstall = this.updateInstallMode === 'interactive' // Why: our only on-machine window into electron-updater; otherwise an unexpected update-not-available or failed fetch is invisible. diff --git a/src/main/updater/updater-state.ts b/src/main/updater/updater-state.ts index a410c5e10b8..d15ce42520c 100644 --- a/src/main/updater/updater-state.ts +++ b/src/main/updater/updater-state.ts @@ -1,6 +1,5 @@ import type { BrowserWindow } from 'electron' import type { ElectronAutoUpdater } from '../electron-updater-loader' -import type { LinuxPackageInstallDiagnostic } from '../linux-package-install-diagnostic' import type { LocalBuildFeed } from '../local-builds/local-build-feed-server' import type { UpdateSource, UpdateStatus } from '../../shared/update-status-types' import type { ReleaseChannel } from '../../shared/release-channel' @@ -54,9 +53,6 @@ export abstract class UpdaterState { protected nudgeCheckTimer: ReturnType | null = null protected pendingQuitAndInstallTimer: ReturnType | null = null protected quitAndInstallInProgress = false - // Why: the pre-install digest re-proof streams the whole package, so a second install request can - // arrive while it runs — after the quit timer was cleared but before the handoff owns the process. - protected linuxPackageRevalidationInFlight = false protected updateInstallMode: UpdateInstallMode = 'interactive' protected lastInstallDeferralVersion = { download: null as string | null, @@ -66,8 +62,6 @@ export abstract class UpdaterState { protected updateInstallCommitted = false // Why: recovery must only run after the native quitAndInstall call; pre-native errors must not clear quittingForUpdate or look like install recovery. protected quitAndInstallNativeInvoked = false - // Why: a synchronous throw out of quitAndInstall ends diagnostic capture before the catch runs, so stash the redacted text for it. - protected lastInstallAttemptDiagnostic: LinuxPackageInstallDiagnostic | null = null protected persistLastUpdateCheckAt: ((timestamp: number) => void) | null = null protected _getLastUpdateCheckAt: (() => number | null) | null = null protected backgroundCheckLaunchPending = false diff --git a/src/renderer/src/components/LinuxPackageInstallRecoveryCard.test.tsx b/src/renderer/src/components/LinuxPackageInstallRecoveryCard.test.tsx index 0c44cf153e5..8e1bc86d729 100644 --- a/src/renderer/src/components/LinuxPackageInstallRecoveryCard.test.tsx +++ b/src/renderer/src/components/LinuxPackageInstallRecoveryCard.test.tsx @@ -3,16 +3,21 @@ import { act, cleanup, fireEvent, render, screen, type RenderResult } from '@tes import userEvent from '@testing-library/user-event' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { LinuxPackageInstallRecovery } from '../../../shared/update-status-types' + +const { toastSuccess } = vi.hoisted(() => ({ toastSuccess: vi.fn() })) +vi.mock('sonner', () => ({ toast: { success: toastSuccess } })) + import { LinuxPackageInstallRecoveryCard } from './LinuxPackageInstallRecoveryCard' const RELEASE_URL = 'https://github.com/stablyai/orca/releases/tag/v1.4.200' const DIAGNOSTIC = 'pkexec: no polkit authentication agent found' const INSTALL_COMMAND = 'sudo apt-get install -y /tmp/orca-updates/orca_1.4.200_amd64.deb' const PACKAGE_FILE_NAME = 'orca_1.4.200_amd64.deb' -const SUMMARY = 'Orca downloaded the update but could not install the system package automatically.' +const SUMMARY = + 'Orca downloaded the system package. Quit Orca before finishing the update from a terminal.' const COPIED_NOTE = - `Command copied. Run it in a system terminal to install ${PACKAGE_FILE_NAME}, ` + - 'then quit and reopen Orca.' + `Command copied. Quit Orca, run it in a system terminal to install ${PACKAGE_FILE_NAME}, ` + + 'then reopen Orca.' const INSTRUCTIONS = { ok: true as const, command: INSTALL_COMMAND, @@ -26,17 +31,16 @@ const NO_PACKAGE_MANAGER = { const getInstructions = vi.fn() const showLinuxPackage = vi.fn() -const quitAndInstall = vi.fn() const writeClipboardText = vi.fn() const openUrl = vi.fn() const onClose = vi.fn() const allMocks = [ getInstructions, showLinuxPackage, - quitAndInstall, writeClipboardText, openUrl, - onClose + onClose, + toastSuccess ] function makeRecovery( @@ -45,7 +49,7 @@ function makeRecovery( return { kind: 'linux-package-install', packageType: 'deb', - reason: 'package-install-failed', + reason: 'manual-install-required', version: '1.4.200', ...overrides } @@ -68,14 +72,6 @@ function renderCard(options: CardOptions = {}): RenderResult { return render(cardElement(options)) } -/** - * Main force-sends a new recovery object on every attempt, which re-renders this card rather than - * remounting it. That push is the only signal a retry failed — quitAndInstall already resolved. - */ -function pushFreshRecovery(view: RenderResult, options: CardOptions = {}): void { - view.rerender(cardElement({ ...options, recovery: makeRecovery(options.recovery) })) -} - // Why: each action chains several promises; drain them without depending on timer faking. async function flushActions(): Promise { await act(async () => { @@ -85,12 +81,18 @@ async function flushActions(): Promise { }) } -function deferred(): { promise: Promise; resolve: (value: T) => void } { +function deferred(): { + promise: Promise + resolve: (value: T) => void + reject: (reason?: unknown) => void +} { let resolve!: (value: T) => void - const promise = new Promise((res) => { + let reject!: (reason?: unknown) => void + const promise = new Promise((res, rej) => { resolve = res + reject = rej }) - return { promise, resolve } + return { promise, resolve, reject } } function button(name: string): HTMLElement { @@ -120,8 +122,8 @@ beforeEach(() => { allMocks.forEach((mock) => mock.mockReset()) getInstructions.mockResolvedValue(INSTRUCTIONS) showLinuxPackage.mockResolvedValue(undefined) - quitAndInstall.mockResolvedValue(undefined) writeClipboardText.mockResolvedValue(undefined) + openUrl.mockResolvedValue(undefined) Object.defineProperty(window, 'api', { configurable: true, value: { @@ -129,8 +131,7 @@ beforeEach(() => { ui: { writeClipboardText }, updater: { getLinuxPackageInstallInstructions: getInstructions, - showLinuxPackage, - quitAndInstall + showLinuxPackage } } }) @@ -142,27 +143,27 @@ afterEach(() => { }) describe('LinuxPackageInstallRecoveryCard copy', () => { - it('leads with the recovery copy and the three dedicated actions', () => { + it('leads with the manual-install copy and recovery actions', () => { renderCard() - expect(screen.getByText('Automatic Install Failed')).toBeTruthy() + expect(screen.getByText('Manual Install Required')).toBeTruthy() expect(screen.getByText(SUMMARY)).toBeTruthy() expect( screen.getByText(/a system terminal on the computer where Orca is installed/) ).toBeTruthy() - expect(screen.getByText(/quit and reopen Orca to run the new version/)).toBeTruthy() + expect(screen.getByText(/Copy the command, quit Orca/)).toBeTruthy() expect(button('Copy Install Command')).toBeTruthy() - expect(button('Try Automatic Install Again')).toBeTruthy() expect(button('Show Package')).toBeTruthy() + expect(button('Download Manually')).toBeTruthy() + expect(screen.queryByRole('button', { name: /Automatic Install/ })).toBeNull() }) it('never offers the generic Retry Download action', () => { renderCard() expect(screen.queryByRole('button', { name: 'Retry Download' })).toBeNull() - // The release fallback only appears once no command can be built. - expect(screen.queryByRole('button', { name: 'Download Manually' })).toBeNull() + expect(button('Download Manually')).toBeTruthy() }) it('minimizes to the status bar from the header control', () => { @@ -182,73 +183,12 @@ describe('LinuxPackageInstallRecoveryCard copy action', () => { expect(getInstructions).toHaveBeenCalledTimes(1) expect(writeClipboardText).toHaveBeenCalledWith(INSTALL_COMMAND) - // The confirmation names the artifact and never replaces the button's own label. - expect(footnoteText()).toBe(COPIED_NOTE) - expect(footnoteText()).toContain(PACKAGE_FILE_NAME) + expect(toastSuccess).toHaveBeenCalledWith(COPIED_NOTE) + expect(footnoteElement()).toBeNull() expect(button('Copy Install Command')).toBeTruthy() expect(screen.queryByRole('button', { name: 'Command copied' })).toBeNull() }) - it('announces through the card, not a nested live region', async () => { - renderCard() - - fireEvent.click(button('Copy Install Command')) - await flushActions() - - expect(footnoteText()).toBe(COPIED_NOTE) - expect(footnoteElement()?.hasAttribute('role')).toBe(false) - expect(screen.queryByRole('status')).toBeNull() - }) - - it('clears the confirmation when another action starts', async () => { - renderCard() - - fireEvent.click(button('Copy Install Command')) - await flushActions() - expect(footnoteText()).toBe(COPIED_NOTE) - - fireEvent.click(button('Show Package')) - await flushActions() - - expect(footnoteElement()).toBeNull() - }) - - it('clears the confirmation when the automatic install is retried', async () => { - renderCard() - - fireEvent.click(button('Copy Install Command')) - await flushActions() - expect(footnoteText()).toBe(COPIED_NOTE) - - fireEvent.click(button('Try Automatic Install Again')) - await flushActions() - - expect(quitAndInstall).toHaveBeenCalledTimes(1) - expect(footnoteElement()).toBeNull() - }) - - it('retires the copy confirmation after the transient window', async () => { - // Why: happy-dom's window timers escape Vitest's fake clock, so capture the scheduled callback. - const scheduled: { handler: () => void; delay?: number }[] = [] - vi.spyOn(window, 'setTimeout').mockImplementation(((handler: () => void, delay?: number) => { - scheduled.push({ handler, delay }) - return scheduled.length - }) as unknown as typeof window.setTimeout) - vi.spyOn(window, 'clearTimeout').mockImplementation(() => undefined) - renderCard() - - fireEvent.click(button('Copy Install Command')) - await flushActions() - expect(footnoteText()).toBe(COPIED_NOTE) - - const expiry = scheduled.find((entry) => entry.delay === 4_000) - expect(expiry).toBeTruthy() - act(() => expiry?.handler()) - - expect(footnoteElement()).toBeNull() - expect(button('Copy Install Command')).toBeTruthy() - }) - it('keeps the copy path when the instruction call rejects', async () => { getInstructions.mockRejectedValue( new Error( @@ -266,8 +206,8 @@ describe('LinuxPackageInstallRecoveryCard copy action', () => { expect(footnoteElement()?.className).toContain('text-destructive') // Why: only main can rule out a command; a rejection must not push the 160 MB redownload. expect(button('Copy Install Command').dataset.variant).toBe('default') - expect(screen.getByText(/Copy the command and run it/)).toBeTruthy() - expect(screen.queryByRole('button', { name: 'Download Manually' })).toBeNull() + expect(screen.getByText(/Copy the command, quit Orca/)).toBeTruthy() + expect(button('Download Manually')).toBeTruthy() expect(writeClipboardText).not.toHaveBeenCalled() }) @@ -281,9 +221,9 @@ describe('LinuxPackageInstallRecoveryCard copy action', () => { await flushActions() expect(footnoteText()).toBe('The downloaded package no longer matches the verified release.') - expect(screen.getByText('Automatic Install Failed')).toBeTruthy() + expect(screen.getByText('Manual Install Required')).toBeTruthy() expect(button('Copy Install Command').dataset.variant).toBe('default') - expect(button('Show Package').dataset.variant).toBe('link') + expect(button('Show Package').dataset.variant).toBe('outline') }) it('retries the instruction call after a rejection', async () => { @@ -297,7 +237,8 @@ describe('LinuxPackageInstallRecoveryCard copy action', () => { await flushActions() expect(getInstructions).toHaveBeenCalledTimes(2) - expect(footnoteText()).toBe(COPIED_NOTE) + expect(footnoteElement()).toBeNull() + expect(toastSuccess).toHaveBeenCalledWith(COPIED_NOTE) }) it('keeps the copy path when only the clipboard write fails', async () => { @@ -313,9 +254,9 @@ describe('LinuxPackageInstallRecoveryCard copy action', () => { const copyButton = button('Copy Install Command') expect(copyButton.dataset.variant).toBe('default') expect(isAriaDisabled(copyButton)).toBe(false) - expect(button('Show Package').dataset.variant).toBe('link') - expect(screen.getByText(/Copy the command and run it/)).toBeTruthy() - expect(screen.queryByRole('button', { name: 'Download Manually' })).toBeNull() + expect(button('Show Package').dataset.variant).toBe('outline') + expect(screen.getByText(/Copy the command, quit Orca/)).toBeTruthy() + expect(button('Download Manually')).toBeTruthy() }) it('recovers from a clipboard failure on the next copy attempt', async () => { @@ -329,7 +270,69 @@ describe('LinuxPackageInstallRecoveryCard copy action', () => { await flushActions() expect(writeClipboardText).toHaveBeenCalledTimes(2) - expect(footnoteText()).toBe(COPIED_NOTE) + expect(footnoteElement()).toBeNull() + expect(toastSuccess).toHaveBeenCalledWith(COPIED_NOTE) + }) + + it('does not write a command after the card unmounts', async () => { + const pending = deferred() + getInstructions.mockReturnValue(pending.promise) + const { unmount } = renderCard() + + fireEvent.click(button('Copy Install Command')) + unmount() + pending.resolve(INSTRUCTIONS) + await flushActions() + + expect(writeClipboardText).not.toHaveBeenCalled() + expect(toastSuccess).not.toHaveBeenCalled() + }) + + it('does not write a command for a recovery the card has replaced', async () => { + const pending = deferred() + getInstructions.mockReturnValue(pending.promise) + const view = renderCard({ recovery: makeRecovery() }) + + fireEvent.click(button('Copy Install Command')) + view.rerender(cardElement({ recovery: makeRecovery({ version: '1.4.201' }) })) + pending.resolve(INSTRUCTIONS) + await flushActions() + + expect(writeClipboardText).not.toHaveBeenCalled() + expect(toastSuccess).not.toHaveBeenCalled() + }) + + it('ignores an instruction rejection from a same-version recovery cycle', async () => { + const pending = deferred() + const recovery = makeRecovery() + getInstructions.mockReturnValue(pending.promise) + const view = renderCard({ recovery }) + + fireEvent.click(button('Copy Install Command')) + view.rerender(cardElement({ recovery: { ...recovery } })) + pending.reject(new Error('Package install recovery is no longer current.')) + await flushActions() + + expect(footnoteElement()).toBeNull() + expect(isAriaDisabled(button('Copy Install Command'))).toBe(false) + }) + + it('ignores a clipboard rejection from a replaced recovery cycle', async () => { + const pending = deferred() + const recovery = makeRecovery() + writeClipboardText.mockReturnValue(pending.promise) + const view = renderCard({ recovery }) + + fireEvent.click(button('Copy Install Command')) + await flushActions() + expect(writeClipboardText).toHaveBeenCalledWith(INSTALL_COMMAND) + + view.rerender(cardElement({ recovery: { ...recovery } })) + pending.reject(new Error('Clipboard is unavailable.')) + await flushActions() + + expect(footnoteElement()).toBeNull() + expect(toastSuccess).not.toHaveBeenCalled() }) }) @@ -344,21 +347,18 @@ describe('LinuxPackageInstallRecoveryCard hashing state', () => { const checking = button('Checking package...') expect(isAriaDisabled(checking)).toBe(true) expect(isAriaDisabled(button('Show Package'))).toBe(true) - expect(isAriaDisabled(button('Try Automatic Install Again'))).toBe(true) fireEvent.click(checking) fireEvent.click(button('Show Package')) - fireEvent.click(button('Try Automatic Install Again')) // Why: the buttons stay clickable for focus reasons, so the handlers must do the refusing. expect(getInstructions).toHaveBeenCalledTimes(1) expect(showLinuxPackage).not.toHaveBeenCalled() - expect(quitAndInstall).not.toHaveBeenCalled() pending.resolve(INSTRUCTIONS) await flushActions() - expect(footnoteText()).toBe(COPIED_NOTE) + expect(toastSuccess).toHaveBeenCalledWith(COPIED_NOTE) expect(isAriaDisabled(button('Show Package'))).toBe(false) }) @@ -370,7 +370,7 @@ describe('LinuxPackageInstallRecoveryCard hashing state', () => { fireEvent.click(button('Copy Install Command')) // Why: ui/button styles only `disabled:`, so without these an inert action looks fully live. - for (const name of ['Checking package...', 'Try Automatic Install Again', 'Show Package']) { + for (const name of ['Checking package...', 'Show Package']) { expect(button(name).className).toContain('aria-disabled:opacity-50') expect(button(name).className).toContain('aria-disabled:cursor-default') } @@ -436,8 +436,11 @@ describe('LinuxPackageInstallRecoveryCard details', () => { fireEvent.click(button('Show details')) + expect(screen.getByText('Details')).toBeTruthy() + expect(screen.queryByText('Last error')).toBeNull() // Why: the digest check is a point-in-time claim, not a standing guarantee about the file. const detail = screen.getByText(/Orca checks the downloaded file against the release metadata/) + expect(detail.textContent).not.toContain(DIAGNOSTIC) expect(detail.textContent).toContain('at the moment it builds this command') expect(detail.textContent).toContain( 'The system package itself is not signature-checked, and Orca cannot vouch for the file ' + @@ -456,7 +459,10 @@ describe('LinuxPackageInstallRecoveryCard details', () => { it('scrolls long diagnostics instead of widening the card', () => { const long = `${DIAGNOSTIC} ${'diagnostic-overflow '.repeat(400)}` - const { container } = renderCard({ diagnostic: long }) + const { container } = renderCard({ + diagnostic: long, + recovery: makeRecovery({ reason: 'authentication-denied' }) + }) fireEvent.click(button('Show details')) @@ -468,97 +474,12 @@ describe('LinuxPackageInstallRecoveryCard details', () => { }) }) -describe('LinuxPackageInstallRecoveryCard retry', () => { - it('retries the automatic install through quitAndInstall', () => { - renderCard() - - fireEvent.click(button('Try Automatic Install Again')) - - expect(quitAndInstall).toHaveBeenCalledTimes(1) - expect(getInstructions).not.toHaveBeenCalled() - }) - - it('holds the busy slot while the quit is in flight', async () => { - renderCard() - - fireEvent.click(button('Try Automatic Install Again')) - await flushActions() - - // Why: the retry now re-proves the package digest before quitting, so the click must report - // progress instead of leaving three inert buttons for the length of the hash. - expect(isAriaDisabled(button('Checking package...'))).toBe(true) - // Why: quitAndInstall resolves as soon as main schedules the install, so a resolved promise is - // not an outcome — the slot stays held until a real status arrives. - expect(isAriaDisabled(button('Copy Install Command'))).toBe(true) - expect(isAriaDisabled(button('Show Package'))).toBe(true) - - fireEvent.click(button('Copy Install Command')) - fireEvent.click(button('Show Package')) - await flushActions() - - expect(getInstructions).not.toHaveBeenCalled() - expect(showLinuxPackage).not.toHaveBeenCalled() - expect(quitAndInstall).toHaveBeenCalledTimes(1) - }) - - it('releases the busy slot when a fresh recovery status arrives', async () => { - const view = renderCard() - - fireEvent.click(button('Try Automatic Install Again')) - await flushActions() - expect(isAriaDisabled(button('Copy Install Command'))).toBe(true) - - // A failed retry never rejects — main pushes a new recovery status a moment later. - pushFreshRecovery(view) - - expect(isAriaDisabled(button('Copy Install Command'))).toBe(false) - expect(isAriaDisabled(button('Show Package'))).toBe(false) - expect(isAriaDisabled(button('Try Automatic Install Again'))).toBe(false) - - fireEvent.click(button('Copy Install Command')) - await flushActions() - - expect(getInstructions).toHaveBeenCalledTimes(1) - expect(writeClipboardText).toHaveBeenCalledWith(INSTALL_COMMAND) - expect(footnoteText()).toBe(COPIED_NOTE) - }) - - it('leaves an in-flight hash job busy when a fresh recovery status arrives', () => { - const pending = deferred() - getInstructions.mockReturnValue(pending.promise) - const view = renderCard() - - fireEvent.click(button('Copy Install Command')) - pushFreshRecovery(view) - - // Why: the release is scoped to the retry slot — a running hash must keep its busy state. - expect(button('Checking package...')).toBeTruthy() - expect(isAriaDisabled(button('Show Package'))).toBe(true) - - pending.resolve(INSTRUCTIONS) - }) - - it('also releases the busy slot if the preload call itself rejects', async () => { - quitAndInstall.mockRejectedValue(new Error('Error: updater is not initialized')) - renderCard() - - fireEvent.click(button('Try Automatic Install Again')) - await flushActions() - - expect(footnoteText()).toBe('updater is not initialized') - expect(isAriaDisabled(button('Copy Install Command'))).toBe(false) - expect(isAriaDisabled(button('Show Package'))).toBe(false) - }) -}) - describe('LinuxPackageInstallRecoveryCard reveal', () => { - it('reveals the retained package from the link-style action', async () => { + it('reveals the retained package from the secondary action', async () => { renderCard() const show = button('Show Package') - expect(show.dataset.variant).toBe('link') - // The link-style action still has to meet the touch-target floor. - expect(show.className).toContain('min-h-[44px]') + expect(show.dataset.variant).toBe('outline') fireEvent.click(show) await flushActions() @@ -578,6 +499,17 @@ describe('LinuxPackageInstallRecoveryCard reveal', () => { // Why: a reveal failure is not a command-build failure, so the copy path must survive it. expect(button('Copy Install Command')).toBeTruthy() }) + + it('reports a failed official-release open in place', async () => { + openUrl.mockRejectedValue(new Error('Could not open the release page.')) + renderCard() + + fireEvent.click(button('Download Manually')) + await flushActions() + + expect(footnoteText()).toBe('Could not open the release page.') + expect(footnoteElement()?.className).toContain('text-destructive') + }) }) describe('LinuxPackageInstallRecoveryCard without a usable command', () => { @@ -590,10 +522,8 @@ describe('LinuxPackageInstallRecoveryCard without a usable command', () => { expect(screen.queryByRole('button', { name: 'Copy Install Command' })).toBeNull() expect(button('Show Package').dataset.variant).toBe('default') - expect(button('Try Automatic Install Again')).toBeTruthy() expect(footnoteText()).toBe(NO_PACKAGE_MANAGER.message) - // The copy-and-run explainer would be dead advice with no command to copy. - expect(screen.queryByText(/Copy the command and run it/)).toBeNull() + expect(screen.getByText(/Quit Orca before finishing the update/)).toBeTruthy() fireEvent.click(button('Download Manually')) expect(openUrl).toHaveBeenCalledWith(RELEASE_URL) @@ -627,43 +557,6 @@ describe('LinuxPackageInstallRecoveryCard without a usable command', () => { expect(showLinuxPackage).toHaveBeenCalledTimes(1) }) - - it('restores the copy path when the automatic install is retried', async () => { - getInstructions.mockResolvedValueOnce(NO_PACKAGE_MANAGER) - renderCard() - - fireEvent.click(button('Copy Install Command')) - await flushActions() - expect(screen.queryByRole('button', { name: 'Copy Install Command' })).toBeNull() - - // Why: a retry re-evaluates the machine, so the earlier "no command" verdict must not stick. - fireEvent.click(button('Try Automatic Install Again')) - await flushActions() - - expect(quitAndInstall).toHaveBeenCalledTimes(1) - expect(button('Copy Install Command').dataset.variant).toBe('default') - expect(button('Show Package').dataset.variant).toBe('link') - expect(screen.getByText(/Copy the command and run it/)).toBeTruthy() - expect(screen.queryByRole('button', { name: 'Download Manually' })).toBeNull() - }) - - it('copies again once a fresh recovery status follows a failed retry', async () => { - getInstructions.mockResolvedValueOnce(NO_PACKAGE_MANAGER) - const view = renderCard() - - fireEvent.click(button('Copy Install Command')) - await flushActions() - - fireEvent.click(button('Try Automatic Install Again')) - await flushActions() - pushFreshRecovery(view) - - fireEvent.click(button('Copy Install Command')) - await flushActions() - - expect(writeClipboardText).toHaveBeenCalledWith(INSTALL_COMMAND) - expect(footnoteText()).toBe(COPIED_NOTE) - }) }) describe('LinuxPackageInstallRecoveryCard keyboard', () => { @@ -690,8 +583,8 @@ describe('LinuxPackageInstallRecoveryCard keyboard', () => { 'Minimize to status bar', 'Show details', 'Copy Install Command', - 'Try Automatic Install Again', - 'Show Package' + 'Show Package', + 'Download Manually' ] for (const name of order) { await user.tab() diff --git a/src/renderer/src/components/LinuxPackageInstallRecoveryCard.tsx b/src/renderer/src/components/LinuxPackageInstallRecoveryCard.tsx index 69a0f73d627..10980667055 100644 --- a/src/renderer/src/components/LinuxPackageInstallRecoveryCard.tsx +++ b/src/renderer/src/components/LinuxPackageInstallRecoveryCard.tsx @@ -1,17 +1,17 @@ -import { useCallback, useEffect, useRef, useState } from 'react' +import { useLayoutEffect, useRef, useState } from 'react' +import { toast } from 'sonner' import type { LinuxPackageInstallInstructions, LinuxPackageInstallRecovery } from '../../../shared/update-status-types' import { UpdateErrorCardContent } from './UpdateErrorCardContent' import { translate } from '@/i18n/i18n' - -const COPY_CONFIRMATION_MS = 4_000 +import { useMountedRef } from '@/hooks/useMountedRef' function copiedNote(packageFileName: string): string { return translate( 'auto.components.LinuxPackageInstallRecoveryCard.aa57fa4f80', - 'Command copied. Run it in a system terminal to install {{value0}}, then quit and reopen Orca.', + 'Command copied. Quit Orca, run it in a system terminal to install {{value0}}, then reopen Orca.', { value0: packageFileName } @@ -39,15 +39,15 @@ export function LinuxPackageInstallRecoveryCard({ // be resolved per render — at module scope they would freeze the whole card in English. const TITLE = translate( 'auto.components.LinuxPackageInstallRecoveryCard.53e1559f99', - 'Automatic Install Failed' + 'Manual Install Required' ) const SUMMARY = translate( 'auto.components.LinuxPackageInstallRecoveryCard.a7ac6ec78b', - 'Orca downloaded the update but could not install the system package automatically.' + 'Orca downloaded the system package. Quit Orca before finishing the update from a terminal.' ) const EXPLAINER = translate( 'auto.components.LinuxPackageInstallRecoveryCard.82c6dbea00', - 'Copy the command and run it in a system terminal on the computer where Orca is installed. After it finishes, quit and reopen Orca to run the new version.' + 'Copy the command, quit Orca, and run it in a system terminal on the computer where Orca is installed. Reopen Orca after it finishes.' ) const AGENT_NOTE = translate( 'auto.components.LinuxPackageInstallRecoveryCard.53c4b8e148', @@ -61,42 +61,23 @@ export function LinuxPackageInstallRecoveryCard({ 'auto.components.LinuxPackageInstallRecoveryCard.c732bcbf8f', 'Checking package...' ) - const [pendingAction, setPendingAction] = useState<'copy' | 'show' | 'retry' | null>(null) + const [pendingAction, setPendingAction] = useState<'copy' | 'show' | null>(null) const [actionError, setActionError] = useState(null) - const [copiedFileName, setCopiedFileName] = useState(null) // Why: the trusted system directories lack sudo or a package manager — no command can be offered at all. const [commandUnavailable, setCommandUnavailable] = useState(false) - const mountedRef = useRef(true) - - useEffect(() => { - mountedRef.current = true - return () => { - mountedRef.current = false - } - }, []) - - // Why: quitAndInstall resolves as soon as main schedules the install, so a failed retry never - // rejects — it arrives as a new recovery status. Without this the busy slot never clears and every - // action stays inert for the rest of the session. - useEffect(() => { - setPendingAction((current) => (current === 'retry' ? null : current)) + const mountedRef = useMountedRef() + const recoveryRef = useRef(recovery) + useLayoutEffect(() => { + recoveryRef.current = recovery }, [recovery]) + const isCurrentRecovery = (): boolean => mountedRef.current && recoveryRef.current === recovery - useEffect(() => { - if (!copiedFileName) { - return - } - const timer = window.setTimeout(() => setCopiedFileName(null), COPY_CONFIRMATION_MS) - return () => window.clearTimeout(timer) - }, [copiedFileName]) - - const handleCopyCommand = useCallback(() => { + const handleCopyCommand = (): void => { if (pendingAction) { return } setPendingAction('copy') setActionError(null) - setCopiedFileName(null) void (async () => { let instructions: LinuxPackageInstallInstructions try { @@ -104,26 +85,29 @@ export function LinuxPackageInstallRecoveryCard({ } catch (error) { // Why: only main knows whether the machine simply has no package manager; any other failure // (stale status, untrusted sender, invalid artifact) must not demote the copy path. - if (mountedRef.current) { + if (isCurrentRecovery()) { setActionError(toMessage(error)) } return } if (!instructions.ok) { - if (mountedRef.current) { + if (isCurrentRecovery()) { setCommandUnavailable(true) setActionError(instructions.message) } return } + if (!isCurrentRecovery()) { + return + } try { await window.api.ui.writeClipboardText(instructions.command) - if (mountedRef.current) { - setCopiedFileName(instructions.packageFileName) + if (isCurrentRecovery()) { + toast.success(copiedNote(instructions.packageFileName)) } } catch (error) { // Why: the command itself is valid — only the clipboard failed, so keep the copy action. - if (mountedRef.current) { + if (isCurrentRecovery()) { setActionError(toMessage(error)) } } @@ -132,19 +116,18 @@ export function LinuxPackageInstallRecoveryCard({ setPendingAction(null) } }) - }, [pendingAction]) + } - const handleShowPackage = useCallback(() => { + const handleShowPackage = (): void => { if (pendingAction) { return } setPendingAction('show') setActionError(null) - setCopiedFileName(null) void window.api.updater .showLinuxPackage() .catch((error: unknown) => { - if (mountedRef.current) { + if (isCurrentRecovery()) { setActionError(toMessage(error)) } }) @@ -153,29 +136,9 @@ export function LinuxPackageInstallRecoveryCard({ setPendingAction(null) } }) - }, [pendingAction]) + } - const handleRetryAutomatic = useCallback(() => { - // Why: guard in the handler, not only through the disabled prop, so no path can quit and install mid-hash. - if (pendingAction) { - return - } - // Why: the quit sequence owns the app from here; hold the busy slot so no other action starts - // work mid-quit. Released by the effect above when a fresh recovery status says Orca stayed open. - setPendingAction('retry') - setActionError(null) - setCopiedFileName(null) - // Why: a fresh install attempt re-evaluates the machine, so an earlier "no command" verdict must not stick. - setCommandUnavailable(false) - void window.api.updater.quitAndInstall().catch((error: unknown) => { - if (mountedRef.current) { - setActionError(toMessage(error)) - setPendingAction(null) - } - }) - }, [pendingAction]) - - // Why: the label keeps naming its action — the footnote below the buttons carries the confirmation. + // Why: the label keeps naming its action while the toast carries transient confirmation. const copyAction = { label: translate( 'auto.components.LinuxPackageInstallRecoveryCard.55c86654b7', @@ -193,40 +156,28 @@ export function LinuxPackageInstallRecoveryCard({ disabled: pendingAction !== null, onClick: handleShowPackage } - const retryAction = { - label: translate( - 'auto.components.LinuxPackageInstallRecoveryCard.3da99454c6', - 'Try Automatic Install Again' - ), - // Why: the retry re-proves the package digest before it quits, so the click is no longer - // instant — without this the card would just go inert for the length of the hash. - pendingLabel: CHECKING_LABEL, - isPending: pendingAction === 'retry', - disabled: pendingAction !== null, - onClick: handleRetryAutomatic - } - const officialReleaseAction = releaseUrl ? { label: translate('auto.components.UpdateCard.47126bcf57', 'Download Manually'), - onClick: () => void window.api.shell.openUrl(releaseUrl) + onClick: () => { + setActionError(null) + void window.api.shell.openUrl(releaseUrl).catch((error: unknown) => { + if (isCurrentRecovery()) { + setActionError(toMessage(error)) + } + }) + } } : undefined const detail = [ recovery.reason === 'authentication-agent-unavailable' ? AGENT_NOTE : null, - diagnostic, + recovery.reason === 'manual-install-required' ? null : diagnostic, TRUST_NOTE ] .filter(Boolean) .join(' ') - const footnote = actionError - ? { text: actionError, tone: 'destructive' as const } - : copiedFileName - ? { text: copiedNote(copiedFileName) } - : undefined - return ( ) diff --git a/src/renderer/src/components/UpdateCard.error-card.test.tsx b/src/renderer/src/components/UpdateCard.error-card.test.tsx index 4cb297b7ca4..1e6da6879f7 100644 --- a/src/renderer/src/components/UpdateCard.error-card.test.tsx +++ b/src/renderer/src/components/UpdateCard.error-card.test.tsx @@ -2,7 +2,7 @@ import { act, cleanup, fireEvent, render, screen, type RenderResult } from '@testing-library/react' import userEvent from '@testing-library/user-event' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import type { LinuxPackageInstallRecovery } from '../../../shared/update-status-types' +import type { LinuxPackageInstallRecovery, UpdateStatus } from '../../../shared/update-status-types' import { useAppStore } from '../store' import { UpdateCard } from './UpdateCard' @@ -19,17 +19,13 @@ const setSettings = vi.fn() const PACKAGE_RECOVERY: LinuxPackageInstallRecovery = { kind: 'linux-package-install', packageType: 'deb', - reason: 'authentication-agent-unavailable', + reason: 'manual-install-required', version: '1.4.200' } -function renderAfterAvailableStatus(): RenderResult { +function renderWithInitialStatus(updateStatus: UpdateStatus): RenderResult { useAppStore.setState({ - updateStatus: { - state: 'available', - version: '1.4.200', - changelog: null - }, + updateStatus, updateChangelog: null, dismissedUpdateVersion: null, updateCardCollapsed: false, @@ -38,6 +34,10 @@ function renderAfterAvailableStatus(): RenderResult { return render() } +function renderAfterAvailableStatus(): RenderResult { + return renderWithInitialStatus({ state: 'available', version: '1.4.200', changelog: null }) +} + function mockReducedMotion(matches: boolean): void { Object.defineProperty(window, 'matchMedia', { configurable: true, @@ -213,23 +213,58 @@ function showPackageRecovery(recovery = PACKAGE_RECOVERY): void { act(() => useAppStore.getState().setUpdateStatus({ state: 'error', - message: 'pkexec: no polkit authentication agent found', + message: 'Quit Orca before running the system package install command.', recovery }) ) } describe('UpdateCard Linux package-install recovery', () => { - it('routes package-install errors to the recovery card instead of the generic one', () => { + it('routes root-package downloads to the manual-install card instead of the generic one', () => { renderAfterAvailableStatus() showPackageRecovery() - expect(screen.getByText('Automatic Install Failed')).toBeTruthy() + expect(screen.getByText('Manual Install Required')).toBeTruthy() expect(screen.queryByText('Update Error')).toBeNull() expect(screen.queryByRole('button', { name: 'Retry Download' })).toBeNull() expect(screen.getByRole('button', { name: 'Copy Install Command' })).toBeTruthy() - expect(screen.getByRole('button', { name: 'Try Automatic Install Again' })).toBeTruthy() + expect(screen.getByRole('button', { name: 'Show Package' })).toBeTruthy() + expect(screen.getByRole('button', { name: 'Download Manually' })).toBeTruthy() + expect(quitAndInstall).not.toHaveBeenCalled() + }) + + it('renders an initial recovery snapshot with its versioned release fallback', () => { + renderWithInitialStatus({ + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: PACKAGE_RECOVERY + }) + + expect(screen.getByText('Manual Install Required')).toBeTruthy() + fireEvent.click(screen.getByRole('button', { name: 'Download Manually' })) + expect(openUrl).toHaveBeenCalledWith('https://github.com/stablyai/orca/releases/tag/v1.4.200') + }) + + it('uses the recovery version when cached update state is stale', () => { + renderWithInitialStatus({ state: 'available', version: '1.4.199', changelog: null }) + showPackageRecovery() + + fireEvent.click(screen.getByRole('button', { name: 'Download Manually' })) + expect(openUrl).toHaveBeenCalledWith('https://github.com/stablyai/orca/releases/tag/v1.4.200') + }) + + it.each([ + 'authentication-agent-unavailable', + 'authentication-denied', + 'package-install-failed' + ] as const)('keeps recovery usable for the legacy %s reason', (reason) => { + renderAfterAvailableStatus() + + showPackageRecovery({ ...PACKAGE_RECOVERY, reason }) + + expect(screen.getByText('Manual Install Required')).toBeTruthy() + expect(screen.getByRole('button', { name: 'Copy Install Command' })).toBeTruthy() expect(screen.getByRole('button', { name: 'Show Package' })).toBeTruthy() }) @@ -262,13 +297,52 @@ describe('UpdateCard Linux package-install recovery', () => { expect(openUrl).toHaveBeenCalledWith('https://github.com/stablyai/orca/releases/tag/v1.4.200') }) + it('resets command discovery when a newer package cycle replaces the recovery', async () => { + getInstructions.mockResolvedValueOnce({ + ok: false, + reason: 'no-package-manager', + message: 'No supported package manager was found.' + }) + renderAfterAvailableStatus() + showPackageRecovery() + + fireEvent.click(screen.getByRole('button', { name: 'Copy Install Command' })) + await flushActions() + expect(screen.queryByRole('button', { name: 'Copy Install Command' })).toBeNull() + + showPackageRecovery({ ...PACKAGE_RECOVERY, version: '1.4.201' }) + + fireEvent.click(screen.getByRole('button', { name: 'Copy Install Command' })) + await flushActions() + expect(getInstructions).toHaveBeenCalledTimes(2) + expect(writeClipboardText).toHaveBeenCalledTimes(1) + }) + + it('links unusable package metadata to the release without offering a futile retry', () => { + const message = + 'The downloaded package metadata could not be verified. Quit Orca before downloading and installing the update from the official release page.' + renderWithInitialStatus({ + state: 'error', + message, + version: '1.4.200', + retryable: false + }) + + expect(screen.getByText('Update Error')).toBeTruthy() + expect(screen.getByText(message)).toBeTruthy() + expect(screen.queryByText('Manual Install Required')).toBeNull() + expect(screen.queryByRole('button', { name: 'Retry Download' })).toBeNull() + fireEvent.click(screen.getByRole('button', { name: 'Download Manually' })) + expect(openUrl).toHaveBeenCalledWith('https://github.com/stablyai/orca/releases/tag/v1.4.200') + }) + it('keeps generic errors on the generic card when no recovery is attached', () => { renderAfterAvailableStatus() act(() => useAppStore.getState().setUpdateStatus({ state: 'error', message: 'ENOSPC' })) expect(screen.getByText('Update Error')).toBeTruthy() - expect(screen.queryByText('Automatic Install Failed')).toBeNull() + expect(screen.queryByText('Manual Install Required')).toBeNull() fireEvent.click(screen.getByRole('button', { name: 'Retry Download' })) expect(download).toHaveBeenCalledTimes(1) }) @@ -295,7 +369,7 @@ describe('UpdateCard Linux package-install recovery', () => { ) expect(screen.getByText('HTTP/2 Download Blocked')).toBeTruthy() - expect(screen.queryByText('Automatic Install Failed')).toBeNull() + expect(screen.queryByText('Manual Install Required')).toBeNull() expect(screen.getByRole('button', { name: 'Enable & Restart' })).toBeTruthy() }) @@ -360,7 +434,7 @@ describe('UpdateCard recovery keyboard and motion', () => { fireEvent.keyDown(screen.getByRole('complementary'), { key: 'Escape' }) expect(useAppStore.getState().updateCardCollapsed).toBe(true) - expect(screen.queryByText('Automatic Install Failed')).toBeNull() + expect(screen.queryByText('Manual Install Required')).toBeNull() }) it('plays the exit animation before minimizing when motion is allowed', () => { diff --git a/src/renderer/src/components/UpdateCard.test.ts b/src/renderer/src/components/UpdateCard.test.ts index a12146e9754..e3143835b6e 100644 --- a/src/renderer/src/components/UpdateCard.test.ts +++ b/src/renderer/src/components/UpdateCard.test.ts @@ -503,6 +503,37 @@ describe('UpdateCard visibility gates', () => { ).toBe('visible') }) + it('shows an initial package recovery before any version was cached', () => { + expect( + computeVisibility({ + status: { + state: 'error', + message: 'Quit Orca before running the system package install command.', + recovery: { + kind: 'linux-package-install', + packageType: 'deb', + reason: 'manual-install-required', + version: '1.2.0' + } + }, + dismissedVersion: null, + cachedVersion: null, + hasStartedDownload: false + }) + ).toBe('visible') + }) + + it('shows an initial versioned download error before any version was cached', () => { + expect( + computeVisibility({ + status: { state: 'error', message: 'invalid metadata', version: '1.2.0' }, + dismissedVersion: null, + cachedVersion: null, + hasStartedDownload: false + }) + ).toBe('visible') + }) + it('shows downloaded for card-initiated downloads', () => { expect( computeVisibility({ diff --git a/src/renderer/src/components/UpdateCard.tsx b/src/renderer/src/components/UpdateCard.tsx index 4bbf47fd492..bcee9e0f521 100644 --- a/src/renderer/src/components/UpdateCard.tsx +++ b/src/renderer/src/components/UpdateCard.tsx @@ -34,7 +34,6 @@ export function UpdateCard(): React.JSX.Element | null { const [installError, setInstallError] = useState(null) const [compatibilityRelaunching, setCompatibilityRelaunching] = useState(false) const [compatibilitySetupError, setCompatibilitySetupError] = useState(null) - const [errorDismissed, setErrorDismissed] = useState(false) const [autoDismissed, setAutoDismissed] = useState(false) const [exiting, setExiting] = useState(false) const isLocalBuild = status.source === 'local' @@ -65,9 +64,6 @@ export function UpdateCard(): React.JSX.Element | null { if (exiting) { setExiting(false) } - if (errorDismissed) { - setErrorDismissed(false) - } } const shouldAutoDismissLatest = @@ -115,7 +111,6 @@ export function UpdateCard(): React.JSX.Element | null { hasStartedDownload: hasStartedDownload.current, updateUserInitiatedCycle, autoDismissed, - errorDismissed, collapsed }) ) { @@ -130,13 +125,6 @@ export function UpdateCard(): React.JSX.Element | null { void window.api.updater.download() } const handleClose = (): void => { - if (status.state === 'error') { - setErrorDismissed(true) - if (cachedVersion) { - dismissUpdate(cachedVersion) - } - return - } dismissUpdate() } const handleInstallRetry = (): void => { @@ -177,7 +165,6 @@ export function UpdateCard(): React.JSX.Element | null { status.state === 'error' && status.recovery?.kind === 'linux-package-install' ? { recovery: status.recovery, diagnostic: status.message } : null - const handleDismissWithAnimation = (): void => { if (prefersReducedMotion) { handleClose() @@ -230,7 +217,6 @@ export function UpdateCard(): React.JSX.Element | null { errorCard={errorCard} linuxPackageRecovery={linuxPackageRecovery} isLocalBuild={isLocalBuild} - cachedVersion={cachedVersion} hasStartedDownload={hasStartedDownload.current} prefersReducedMotion={prefersReducedMotion} mediaFailed={mediaFailed} @@ -250,7 +236,9 @@ export function UpdateCard(): React.JSX.Element | null { ? 'animate-update-card-exit' : 'animate-update-card-enter' const showReassurance = - !reassuranceSeen && (status.state === 'available' || status.state === 'downloading') + !reassuranceSeen && + ((status.state === 'available' && !status.externallyManaged) || + status.state === 'downloading') return (

- {translate('auto.components.UpdateCard.3553a8672f', 'Last error')} + {translate('auto.components.UpdateCard.3553a8672f', 'Details')}

{detail} diff --git a/src/renderer/src/components/maintenance/update-card/UpdateAvailableCardContent.tsx b/src/renderer/src/components/maintenance/update-card/UpdateAvailableCardContent.tsx index b18c18537d1..ce45e1c69e5 100644 --- a/src/renderer/src/components/maintenance/update-card/UpdateAvailableCardContent.tsx +++ b/src/renderer/src/components/maintenance/update-card/UpdateAvailableCardContent.tsx @@ -7,6 +7,18 @@ function isAnimatedGif(url: string | undefined): boolean { return typeof url === 'string' && url.toLowerCase().endsWith('.gif') } +/** A package manager owns this install: the release is real but Orca can never apply it here. */ +function ExternallyManagedNote(): React.JSX.Element { + return ( +

+ {translate( + 'auto.components.UpdateCard.7f1a4c9e02', + 'Your system package manager installed Orca, so update it from there — Orca cannot install this release itself.' + )} +

+ ) +} + export function UpdateAvailableRichContent({ release, releasesBehind, @@ -16,7 +28,8 @@ export function UpdateAvailableRichContent({ onMediaError, onMediaLoad, onUpdate, - onClose + onClose, + externallyManaged = false }: { release: NonNullable releasesBehind: number | null @@ -27,6 +40,7 @@ export function UpdateAvailableRichContent({ onMediaLoad: () => void onUpdate: () => void onClose: () => void + externallyManaged?: boolean }): React.JSX.Element { const showMedia = release.mediaUrl && !mediaFailed && !(prefersReducedMotion && isAnimatedGif(release.mediaUrl)) @@ -87,9 +101,13 @@ export function UpdateAvailableRichContent({ > {translate('auto.components.UpdateCard.aad383aecc', 'Read the full release notes')} - + {externallyManaged ? ( + + ) : ( + + )}
) } @@ -98,12 +116,14 @@ export function UpdateAvailableSimpleContent({ version, releaseUrl, onUpdate, - onClose + onClose, + externallyManaged = false }: { version: string releaseUrl?: string onUpdate: () => void onClose: () => void + externallyManaged?: boolean }): React.JSX.Element { return (
@@ -126,9 +146,13 @@ export function UpdateAvailableSimpleContent({ value0: version })}

-

- {translate('auto.components.UpdateCard.fdd4a364fa', "Sessions won't be interrupted.")} -

+ {externallyManaged ? ( + + ) : ( +

+ {translate('auto.components.UpdateCard.fdd4a364fa', "Sessions won't be interrupted.")} +

+ )} {releaseUrl && ( + {!externallyManaged && ( + + )}
) } diff --git a/src/renderer/src/components/maintenance/update-card/UpdateCardStateContent.tsx b/src/renderer/src/components/maintenance/update-card/UpdateCardStateContent.tsx index 3a501f574af..8a67a762a50 100644 --- a/src/renderer/src/components/maintenance/update-card/UpdateCardStateContent.tsx +++ b/src/renderer/src/components/maintenance/update-card/UpdateCardStateContent.tsx @@ -20,7 +20,6 @@ export function UpdateCardStateContent({ errorCard, linuxPackageRecovery, isLocalBuild, - cachedVersion, hasStartedDownload, prefersReducedMotion, mediaFailed, @@ -40,7 +39,6 @@ export function UpdateCardStateContent({ diagnostic: string } | null isLocalBuild: boolean - cachedVersion: string | null hasStartedDownload: boolean prefersReducedMotion: boolean mediaFailed: boolean @@ -71,9 +69,14 @@ export function UpdateCardStateContent({ if (linuxPackageRecovery) { return ( ) @@ -129,6 +132,7 @@ export function UpdateCardStateContent({ onMediaLoad={onMediaLoad} onUpdate={onUpdate} onClose={onDismiss} + externallyManaged={status.externallyManaged} /> ) : ( ) } diff --git a/src/renderer/src/components/maintenance/update-card/update-card-error-model.ts b/src/renderer/src/components/maintenance/update-card/update-card-error-model.ts index 37f34f30b6a..1a6ab8672a0 100644 --- a/src/renderer/src/components/maintenance/update-card/update-card-error-model.ts +++ b/src/renderer/src/components/maintenance/update-card/update-card-error-model.ts @@ -117,17 +117,25 @@ export function buildUpdateCardErrorModel({ } return { title: cachedVersion ? 'Update Error' : 'Update Check Failed', - summary: cachedVersion ? 'Could not complete the update.' : 'Could not check for updates.', + summary: + cachedVersion && status.retryable === false + ? status.message + : cachedVersion + ? 'Could not complete the update.' + : 'Could not check for updates.', detail: status.message, releaseUrl: getReleaseNotesUrlForVersion(cachedVersion), - primaryAction: cachedVersion - ? { - label: translate('auto.components.UpdateCard.48565a32bc', 'Retry Download'), - onClick: onRetryDownload - } - : { - label: translate('auto.components.UpdateCard.6b0085010d', 'Re-check'), - onClick: onRecheck - } + primaryAction: + cachedVersion && status.retryable !== false + ? { + label: translate('auto.components.UpdateCard.48565a32bc', 'Retry Download'), + onClick: onRetryDownload + } + : !cachedVersion + ? { + label: translate('auto.components.UpdateCard.6b0085010d', 'Re-check'), + onClick: onRecheck + } + : undefined } } diff --git a/src/renderer/src/components/maintenance/update-card/update-card-visibility.ts b/src/renderer/src/components/maintenance/update-card/update-card-visibility.ts index 038bf1542c6..f18629a4fbe 100644 --- a/src/renderer/src/components/maintenance/update-card/update-card-visibility.ts +++ b/src/renderer/src/components/maintenance/update-card/update-card-visibility.ts @@ -7,7 +7,6 @@ export function isUpdateCardVisible({ hasStartedDownload, updateUserInitiatedCycle, autoDismissed = false, - errorDismissed = false, collapsed = false }: { status: UpdateStatus @@ -16,12 +15,15 @@ export function isUpdateCardVisible({ hasStartedDownload: boolean updateUserInitiatedCycle: boolean autoDismissed?: boolean - errorDismissed?: boolean collapsed?: boolean }): boolean { const isUserInitiated = 'userInitiated' in status && Boolean(status.userInitiated) const shouldShowDetailedErrorCard = - status.state === 'error' && (hasStartedDownload || cachedVersion !== null) + status.state === 'error' && + (hasStartedDownload || + cachedVersion !== null || + status.version !== undefined || + status.recovery?.kind === 'linux-package-install') if (status.state === 'checking' && !isUserInitiated) { return false @@ -35,10 +37,6 @@ export function isUpdateCardVisible({ if (status.state === 'error' && !shouldShowDetailedErrorCard && !isUserInitiated) { return false } - if (status.state === 'error' && errorDismissed) { - return false - } - if (cachedVersion && dismissedVersion === cachedVersion && !updateUserInitiatedCycle) { if (status.state !== 'downloading' && status.state !== 'error') { return false diff --git a/src/renderer/src/components/settings/GeneralUpdateSettingsSection.test.tsx b/src/renderer/src/components/settings/GeneralUpdateSettingsSection.test.tsx new file mode 100644 index 00000000000..f569c9f5114 --- /dev/null +++ b/src/renderer/src/components/settings/GeneralUpdateSettingsSection.test.tsx @@ -0,0 +1,37 @@ +// @vitest-environment happy-dom +import { cleanup, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, expect, it, vi } from 'vitest' +import { useAppStore } from '../../store' +import { GeneralUpdateSettingsSection } from './GeneralUpdateSettingsSection' + +vi.mock('./GeneralRemoteServerUpdates', () => ({ GeneralRemoteServerUpdates: () => null })) +vi.mock('./ReleaseChannelSection', () => ({ ReleaseChannelSection: () => null })) + +beforeEach(() => { + useAppStore.setState({ + updateStatus: { state: 'available', version: '1.4.200', changelog: null } + }) + Object.defineProperty(window, 'api', { + configurable: true, + value: { + updater: { + check: vi.fn(), + download: vi.fn(), + getVersion: vi.fn().mockResolvedValue('1.4.199') + } + } + }) +}) + +afterEach(() => { + cleanup() + useAppStore.setState({ updateStatus: { state: 'idle' } }) +}) + +it('describes the available action as a download', () => { + render() + + expect(screen.getByRole('button', { name: 'Download Update (1.4.200)' })).toBeTruthy() + expect(screen.getByText(/is available\. Click "Download Update" to download it\./)).toBeTruthy() + expect(screen.queryByText(/download and install it/)).toBeNull() +}) diff --git a/src/renderer/src/components/settings/GeneralUpdateSettingsSection.tsx b/src/renderer/src/components/settings/GeneralUpdateSettingsSection.tsx index 15677927ad3..1e59c200fca 100644 --- a/src/renderer/src/components/settings/GeneralUpdateSettingsSection.tsx +++ b/src/renderer/src/components/settings/GeneralUpdateSettingsSection.tsx @@ -14,19 +14,9 @@ import { getReleaseNotesUrlForVersion } from '../../../../shared/release-channel export function GeneralUpdateSettingsSection(): React.JSX.Element { const updateStatus = useAppStore((s) => s.updateStatus) - // Why: the 'error' variant of UpdateStatus does not carry a `version` field. - // The main process emits `{ state: 'error' }` for both check failures (no - // version known yet) and download/install failures (version was known from - // the preceding 'available'/'downloading'/'downloaded' state). Cache the - // last-known version so the error copy below can distinguish the two cases - // without adding IPC. Mirrors `versionRef` in UpdateCard.tsx. + // Why: older hosts omit `version` from errors, so retain the last target for correct copy. const updateVersionRef = useRef(null) - if ( - (updateStatus.state === 'available' || - updateStatus.state === 'downloading' || - updateStatus.state === 'downloaded') && - updateStatus.version - ) { + if ('version' in updateStatus && updateStatus.version) { updateVersionRef.current = updateStatus.version } else if ( updateStatus.state === 'checking' || @@ -122,7 +112,7 @@ export function GeneralUpdateSettingsSection(): React.JSX.Element { )} - {updateStatus.state === 'available' ? ( + {updateStatus.state === 'available' && !updateStatus.externallyManaged ? ( @@ -178,10 +168,15 @@ export function GeneralUpdateSettingsSection(): React.JSX.Element { 'Version' )}{' '} {updateStatus.version}{' '} - {translate( - 'auto.components.settings.GeneralUpdateSettingsSection.8311da27ba', - 'is available. Click "Install Update" to download and install it.' - )}{' '} + {updateStatus.externallyManaged + ? translate( + 'auto.components.settings.GeneralUpdateSettingsSection.e3b9d21c07', + 'is available. Update Orca through your system package manager — Orca cannot install this release itself.' + ) + : translate( + 'auto.components.settings.GeneralUpdateSettingsSection.8311da27ba', + 'is available. Click "Download Update" to download it.' + )}{' '} {updateStatus.source !== 'local' && (
)} {updateStatus.state === 'error' && - // Why: `{ state: 'error' }` is emitted for both check-time - // failures (no version cached) and download/install failures - // (version cached from a prior 'available'/'downloading'/ - // 'downloaded' state). Label accordingly so a download failure - // isn't mislabeled as a "check" failure. Mirrors UpdateCard.tsx. - (updateVersionRef.current - ? translate( - 'auto.components.settings.GeneralUpdateSettingsSection.b9ad70c30d', - 'Update error. {{value0}}', - { value0: updateStatus.message } - ) - : translate( - 'auto.components.settings.GeneralUpdateSettingsSection.bd79d412f0', - 'Update check failed. {{value0}}', - { value0: updateStatus.message } - ))} + (updateStatus.recovery?.kind === 'linux-package-install' + ? updateStatus.message + : updateVersionRef.current + ? translate( + 'auto.components.settings.GeneralUpdateSettingsSection.b9ad70c30d', + 'Update error. {{value0}}', + { value0: updateStatus.message } + ) + : translate( + 'auto.components.settings.GeneralUpdateSettingsSection.bd79d412f0', + 'Update check failed. {{value0}}', + { value0: updateStatus.message } + ))}

{channelSwitcherRevealed ? : null} diff --git a/src/renderer/src/components/settings/RemoteServerUpdateStatus.tsx b/src/renderer/src/components/settings/RemoteServerUpdateStatus.tsx index 40dc49dcc5f..5d02889fb7f 100644 --- a/src/renderer/src/components/settings/RemoteServerUpdateStatus.tsx +++ b/src/renderer/src/components/settings/RemoteServerUpdateStatus.tsx @@ -96,7 +96,7 @@ export function getRemoteServerManualUpdateHelp(entry: RemoteServerUpdateEntry): if (entry.support?.reason === 'manual-service-update-required') { return translate( 'auto.components.settings.RemoteServerUpdateStatus.serviceManagerHelp', - 'Update Orca through the service manager that starts this server.' + 'Update Orca on the server host — through its system package manager if it was installed from a .deb or .rpm, otherwise through the service manager that starts it.' ) } if (entry.support?.reason === 'unpackaged-build') { diff --git a/src/renderer/src/components/status-bar/UpdateStatusSegment.tsx b/src/renderer/src/components/status-bar/UpdateStatusSegment.tsx index af41d79ea48..0da77795798 100644 --- a/src/renderer/src/components/status-bar/UpdateStatusSegment.tsx +++ b/src/renderer/src/components/status-bar/UpdateStatusSegment.tsx @@ -21,6 +21,10 @@ export function UpdateStatusSegment({ return null } + const linuxPackageRecovery = + status.state === 'error' && status.recovery?.kind === 'linux-package-install' + ? status.recovery + : null const segment = (() => { if (status.state === 'downloading') { const pct = Math.max(0, Math.min(100, Math.round(status.percent))) @@ -39,7 +43,9 @@ export function UpdateStatusSegment({ ) } } - if (status.state === 'downloaded') { + const readyVersion = + status.state === 'downloaded' ? status.version : linuxPackageRecovery?.version + if (readyVersion !== undefined) { return { icon: , label: translate( @@ -49,7 +55,7 @@ export function UpdateStatusSegment({ tooltip: translate( 'auto.components.status.bar.UpdateStatusSegment.9d13213a56', 'Orca v{{value0}} ready to install', - { value0: status.version } + { value0: readyVersion } ), ariaLabel: translate( 'auto.components.status.bar.UpdateStatusSegment.962404f68e', diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index a641c2cf7dd..35d8dc50d42 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -2264,7 +2264,7 @@ "8acbdd3961": "Minimize to status bar", "17412483da": "Ready to Install", "47126bcf57": "Download Manually", - "3553a8672f": "Last error", + "3553a8672f": "Details", "90559b14e3": "This turns on a process-wide Electron networking switch after restart. Use it for corporate VPNs or proxies that reject HTTP/2 update downloads.", "6e45bfa2e0": "Downloading...", "558842597d": "Downloading Update", @@ -2303,7 +2303,8 @@ "a4650b0dc4": "Could Not Use Local Build", "b1e390250d": "Could not complete the local build switch.", "d29740d175": "The selected build could not be used.", - "37d45c9ec1": "Choose Another Build" + "37d45c9ec1": "Choose Another Build", + "7f1a4c9e02": "Your system package manager installed Orca, so update it from there — Orca cannot install this release itself." }, "WorktreeJumpPalette": { "ac037cfac2": "Move", @@ -7055,9 +7056,9 @@ "8a52ca1d02": "Release notes", "d89806cc89": "is ready to install.", "a6b37929dc": "Version", - "8311da27ba": "is available. Click \"Install Update\" to download and install it.", + "8311da27ba": "is available. Click \"Download Update\" to download it.", "f44299636f": "Restart to Update (", - "42717918f4": "Install Update (", + "42717918f4": "Download Update (", "02dc082e70": "Could not start the update download.", "e1a647adc5": "Check for Updates", "ceb579abaf": "Check for app updates and install a newer Orca version.", @@ -7075,7 +7076,8 @@ "31fd7150cf": "Checking for updates...", "3394d1f663": "checking", "d69a09b672": "Updates are checked automatically on launch.", - "7173352632": "idle" + "7173352632": "idle", + "e3b9d21c07": "is available. Update Orca through your system package manager — Orca cannot install this release itself." }, "GeneralWorkspaceSettingsSection": { "3d538a98f7": "Choose apps available from a workspace's Open in menu.", @@ -10963,7 +10965,7 @@ "restarting": "Restarting…", "updated": "Updated", "failed": "Update failed", - "serviceManagerHelp": "Update Orca through the service manager that starts this server.", + "serviceManagerHelp": "Update Orca on the server host — through its system package manager if it was installed from a .deb or .rpm, otherwise through the service manager that starts it.", "unpackedHelp": "Development builds must be updated from their source checkout.", "legacyHelp": "Update this server manually once to enable remote updates." }, @@ -16342,14 +16344,13 @@ }, "LinuxPackageInstallRecoveryCard": { "e3de29c86a": "Show Package", - "3da99454c6": "Try Automatic Install Again", "55c86654b7": "Copy Install Command", - "53e1559f99": "Automatic Install Failed", - "a7ac6ec78b": "Orca downloaded the update but could not install the system package automatically.", - "82c6dbea00": "Copy the command and run it in a system terminal on the computer where Orca is installed. After it finishes, quit and reopen Orca to run the new version.", + "53e1559f99": "Manual Install Required", + "a7ac6ec78b": "Orca downloaded the system package. Quit Orca before finishing the update from a terminal.", + "82c6dbea00": "Copy the command, quit Orca, and run it in a system terminal on the computer where Orca is installed. Reopen Orca after it finishes.", "53c4b8e148": "No usable authentication agent answered the privileged install request.", "c732bcbf8f": "Checking package...", - "aa57fa4f80": "Command copied. Run it in a system terminal to install {{value0}}, then quit and reopen Orca.", + "aa57fa4f80": "Command copied. Quit Orca, run it in a system terminal to install {{value0}}, then reopen Orca.", "b7e7c5bc95": "Orca checks the downloaded file against the release metadata at the moment it builds this command. The system package itself is not signature-checked, and Orca cannot vouch for the file after that point." }, "pr-check-counts": { diff --git a/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt b/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt index 213b9ebdd96..f5fae013b7e 100644 --- a/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt +++ b/src/shared/child-process/__fixtures__/child-process-import-allowlist.txt @@ -150,10 +150,8 @@ src/main/ssh/system-ssh-dynamic-forward-process.ts src/main/ssh/system-ssh-file-transfer.ts src/main/ssh/system-ssh-forward-process.ts src/main/ssh/system-ssh-operation-lifecycle.ts -src/main/startup/appimage-cli-redirect.ts src/main/startup/ensure-virtual-display.ts src/main/startup/hydrate-shell-path.ts -src/main/startup/packaged-cli-entry-redirect.ts src/main/startup/windows-install-dir-acl-probe.ts src/main/startup/windows-user-data-acl.ts src/main/win32-utils.ts diff --git a/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt b/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt index 752a762370a..a8878a182d0 100644 --- a/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt +++ b/src/shared/child-process/__fixtures__/windows-console-visibility-allowlist.txt @@ -43,9 +43,7 @@ main/pty/windows-environment-path.ts main/rate-limits/codex-fetcher.ts main/runtime/tls-certificate.ts main/ssh/ssh-connection.ts -main/startup/appimage-cli-redirect.ts main/startup/ensure-virtual-display.ts -main/startup/packaged-cli-entry-redirect.ts main/startup/windows-install-dir-acl-probe.ts main/win32-utils.ts main/window/clipboard-ipc-handlers.ts diff --git a/src/shared/child-process/child-process-import-boundary.test.ts b/src/shared/child-process/child-process-import-boundary.test.ts index 10ca4fd8522..32c6c9d890e 100644 --- a/src/shared/child-process/child-process-import-boundary.test.ts +++ b/src/shared/child-process/child-process-import-boundary.test.ts @@ -29,7 +29,7 @@ const CHILD_PROCESS_IMPORT_ALLOWLIST: readonly string[] = readFileSync( * May only ever be DECREASED, and only by migrating a file off * `node:child_process`. Raising it is never the fix. */ -const DIRECT_IMPORTER_PIN = 160 +const DIRECT_IMPORTER_PIN = 158 const IMPORT_PATTERN = /(?:from\s+['"]node:child_process['"]|from\s+['"]child_process['"]|require\(\s*['"]node:child_process['"]|require\(\s*['"]child_process['"])/ diff --git a/src/shared/child-process/windows-console-visibility.test.ts b/src/shared/child-process/windows-console-visibility.test.ts index 7985b0ee11a..596728fa245 100644 --- a/src/shared/child-process/windows-console-visibility.test.ts +++ b/src/shared/child-process/windows-console-visibility.test.ts @@ -34,7 +34,7 @@ const ALLOWLIST: readonly string[] = readAllowlist( * the allowlist does not bound this: a swap (one file fixed and delisted, one * new file added with its entry) satisfies both membership assertions. */ -const UNHIDDEN_SPAWNER_PIN = 68 +const UNHIDDEN_SPAWNER_PIN = 66 const CHILD_PROCESS_IMPORT = /from\s+['"](?:node:)?child_process['"]|require\(\s*['"](?:node:)?child_process['"]/ diff --git a/src/shared/cli-argument-boundary.test.ts b/src/shared/cli-argument-boundary.test.ts new file mode 100644 index 00000000000..ceb2e64432f --- /dev/null +++ b/src/shared/cli-argument-boundary.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { CLI_GLOBAL_VALUE_FLAGS, findCliCommandIndex } from './cli-argument-boundary' + +const COMMAND_PATHS = [['project'], ['serve'], ['status'], ['worktree']] as const + +describe('findCliCommandIndex', () => { + it.each([ + { argv: ['--json', 'status'], expected: 1, name: 'global boolean' }, + { argv: ['--environment', 'status'], expected: 1, name: 'missing global value' }, + { + argv: ['--environment', 'status', 'worktree', 'list'], + expected: 2, + name: 'command-named value' + }, + { + argv: ['--project', 'github:stablyai/orca', 'project', 'setups'], + expected: 2, + name: 'selector value' + }, + { argv: ['--project=github:stablyai/orca', 'project'], expected: 1, name: 'assignment' }, + { argv: ['--', 'status'], expected: 1, name: 'bare double dash' }, + { argv: ['workspace', 'status'], expected: -1, name: 'first non-command positional' }, + { argv: ['serve'], expected: 0, name: 'direct serve' } + ])('$name', ({ argv, expected }) => { + expect(findCliCommandIndex(argv, COMMAND_PATHS)).toBe(expected) + }) + + it('consumes known global values at the launch boundary', () => { + expect( + findCliCommandIndex(['--environment', 'status'], COMMAND_PATHS, CLI_GLOBAL_VALUE_FLAGS) + ).toBe(-1) + }) +}) diff --git a/src/shared/cli-argument-boundary.ts b/src/shared/cli-argument-boundary.ts new file mode 100644 index 00000000000..7b29db49c45 --- /dev/null +++ b/src/shared/cli-argument-boundary.ts @@ -0,0 +1,96 @@ +export const CLI_GLOBAL_VALUE_FLAGS: readonly string[] = ['pairing-code', 'environment'] +export const CLI_GLOBAL_FLAGS: readonly string[] = ['help', 'json', ...CLI_GLOBAL_VALUE_FLAGS] + +export const CLI_BOOLEAN_FLAGS = new Set([ + 'all', + 'attachments', + 'children', + 'comments', + 'connect', + 'current', + 'dry-run', + 'enter', + 'focus', + 'force', + 'full', + 'help', + 'inject', + 'include-archived', + 'include-visual-layouts', + 'interrupt', + 'json', + 'local', + 'messages', + 'me', + 'mobile', + 'mobile-pairing', + 'no-pairing', + 'screen', + 'parent-current', + 'provision', + 'ready', + 'recipe-json', + 'relations', + 'reinstall', + 'restore-window', + 'return-preamble', + 'run-hooks', + 'show-profile', + 'staged', + 'tab', + 'tasks', + 'text-stdin', + 'unread', + 'value-stdin', + 'wait' +]) + +function commandPathStartsAt( + argv: readonly string[], + tokenIndex: number, + path: readonly string[] +): boolean { + let cursor = tokenIndex + for (const part of path) { + while (argv[cursor]?.startsWith('--')) { + const assignment = argv[cursor].slice(2) + const flag = assignment.split('=', 1)[0] + cursor += assignment.includes('=') || CLI_BOOLEAN_FLAGS.has(flag) ? 1 : 2 + } + if (argv[cursor] !== part) { + return false + } + cursor += 1 + } + return true +} + +export function findCliCommandIndex( + argv: readonly string[], + commandPaths: readonly (readonly string[])[], + knownValueFlags: readonly string[] = [] +): number { + const startsCommandAt = (index: number): boolean => + commandPaths.some((path) => commandPathStartsAt(argv, index, path)) + + for (let index = 0; index < argv.length;) { + const token = argv[index] + if (!token.startsWith('--')) { + return startsCommandAt(index) ? index : -1 + } + + const assignment = token.slice(2) + const flag = assignment.split('=', 1)[0] + const next = argv[index + 1] + const takesNext = + !assignment.includes('=') && + !CLI_BOOLEAN_FLAGS.has(flag) && + next !== undefined && + !next.startsWith('--') && + (knownValueFlags.includes(flag) || + !(startsCommandAt(index + 1) && !startsCommandAt(index + 2))) + + index += takesNext ? 2 : 1 + } + return -1 +} diff --git a/src/shared/edit-distance.ts b/src/shared/edit-distance.ts new file mode 100644 index 00000000000..7a496712b89 --- /dev/null +++ b/src/shared/edit-distance.ts @@ -0,0 +1,23 @@ +export function levenshtein(a: string, b: string): number { + const m = a.length + const n = b.length + if (m === 0) { + return n + } + if (n === 0) { + return m + } + let previous = Array.from({ length: n + 1 }, (_, index) => index) + let current = Array.from({ length: n + 1 }, () => 0) + for (let i = 1; i <= m; i += 1) { + current[0] = i + for (let j = 1; j <= n; j += 1) { + const cost = a[i - 1] === b[j - 1] ? 0 : 1 + current[j] = Math.min(previous[j] + 1, current[j - 1] + 1, previous[j - 1] + cost) + } + const swap = previous + previous = current + current = swap + } + return previous[n] +} diff --git a/src/shared/serve-option-validation.test.ts b/src/shared/serve-option-validation.test.ts new file mode 100644 index 00000000000..70473c57477 --- /dev/null +++ b/src/shared/serve-option-validation.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it } from 'vitest' +import { getServeFlagTypoError, getServeOptionValidationError } from './serve-option-validation' + +const validOptions = { + noPairing: false, + mobilePairing: false, + recipeJson: false, + projectRoot: null +} + +describe('getServeOptionValidationError', () => { + it('accepts compatible options', () => { + expect(getServeOptionValidationError(validOptions)).toBeNull() + }) + + it.each([ + [{ noPairing: true, mobilePairing: true }, /either --mobile-pairing or --no-pairing/i], + [ + { recipeJson: true, noPairing: true, projectRoot: '/tmp/repo' }, + /requires runtime pairing.*--no-pairing/i + ], + [ + { recipeJson: true, mobilePairing: true, projectRoot: '/tmp/repo' }, + /requires runtime pairing.*--mobile-pairing/i + ], + [{ recipeJson: true }, /requires --project-root/i] + ])('rejects incompatible options', (override, expected) => { + expect( + getServeOptionValidationError({ ...validOptions, ...override } as typeof validOptions) + ).toMatch(expected) + }) +}) + +describe('getServeFlagTypoError', () => { + it('accepts exact serve flags and arbitrary Chromium switches', () => { + expect( + getServeFlagTypoError([ + '/opt/orca/orca-ide', + '--serve', + '--serve-no-pairing', + '--disable-gpu', + '--disable-features=Vulkan', + '--no-parent' + ]) + ).toBeNull() + }) + + it.each(['--no-pair', '--no-pairng', '--no-paring', '--mobile-pairng'])( + 'suggests the intended pairing flag for %s', + (flag) => { + expect(getServeFlagTypoError(['/opt/orca/orca-ide', '--serve', flag])).toMatch( + /Unknown flag .*Did you mean --(?:no-pairing|mobile-pairing)\?/i + ) + } + ) + + it('does not reinterpret tokens after --', () => { + expect(getServeFlagTypoError(['/opt/orca/orca-ide', '--serve', '--', '--no-pairng'])).toBeNull() + }) + + it('does not inspect an equals-form value as a flag', () => { + expect( + getServeFlagTypoError(['/opt/orca/orca-ide', '--serve-pairing-address=--no-pairng']) + ).toBeNull() + }) + + it('keeps flag-shaped space values subject to typo validation', () => { + expect( + getServeFlagTypoError(['/opt/orca/orca-ide', '--serve-pairing-address', '--no-pairng']) + ).toMatch(/Unknown flag --no-pairng.*--no-pairing/i) + }) +}) diff --git a/src/shared/serve-option-validation.ts b/src/shared/serve-option-validation.ts new file mode 100644 index 00000000000..36f844f322f --- /dev/null +++ b/src/shared/serve-option-validation.ts @@ -0,0 +1,89 @@ +import { levenshtein } from './edit-distance' + +export type ServeOptionValidationInput = { + noPairing: boolean + mobilePairing: boolean + recipeJson: boolean + projectRoot: string | null | undefined +} + +export function getServeOptionValidationError(options: ServeOptionValidationInput): string | null { + if (options.noPairing && options.mobilePairing) { + return 'Use either --mobile-pairing or --no-pairing, not both.' + } + if (options.recipeJson && options.noPairing) { + return 'Recipe JSON output requires runtime pairing; remove --no-pairing.' + } + if (options.recipeJson && options.mobilePairing) { + return 'Recipe JSON output requires runtime pairing; remove --mobile-pairing.' + } + if (options.recipeJson && !options.projectRoot) { + return 'Recipe JSON output requires --project-root.' + } + return null +} + +const SERVE_SECURITY_FLAG_NAMES = [ + '--no-pairing', + '--serve-no-pairing', + '--mobile-pairing', + '--serve-mobile-pairing', + '--recipe-json', + '--serve-recipe-json', + '--pairing-address', + '--serve-pairing-address' +] as const + +const SERVE_VALUE_FLAG_NAMES = new Set([ + '--port', + '--serve-port', + '--pairing-address', + '--serve-pairing-address', + '--project-root', + '--serve-project-root', + '--pairing-code', + '--environment' +]) + +function flagName(token: string): string { + const equalsIndex = token.indexOf('=') + return equalsIndex === -1 ? token : token.slice(0, equalsIndex) +} + +/** Reject only near-miss pairing flags; Electron/Chromium switches stay open-ended. */ +export function getServeFlagTypoError(argv: readonly string[]): string | null { + for (let index = 0; index < argv.length; index += 1) { + const token = argv[index]! + if (token === '--') { + break + } + if (!token.startsWith('--')) { + continue + } + const name = flagName(token) + let suggestion: string | null = null + let bestDistance = Number.POSITIVE_INFINITY + for (const candidate of SERVE_SECURITY_FLAG_NAMES) { + const distance = levenshtein(name, candidate) + const maxDistance = candidate.startsWith(name) ? 3 : 2 + if (distance > 0 && distance <= maxDistance && distance < bestDistance) { + suggestion = candidate + bestDistance = distance + } + } + if (suggestion) { + return `Unknown flag ${name}. Did you mean ${suggestion}?` + } + + // A value that is not flag-shaped belongs to the preceding known value flag. + // A `--`-prefixed space token remains a flag, matching parseArgs; use `=` when + // a value itself starts with `--`. + if (!token.includes('=') && SERVE_VALUE_FLAG_NAMES.has(name)) { + const value = argv[index + 1] + if (value !== undefined && !value.startsWith('--')) { + index += 1 + } + } + } + return null +} diff --git a/src/shared/update-status-types.ts b/src/shared/update-status-types.ts index 8d54837c9c5..74ee382ec58 100644 --- a/src/shared/update-status-types.ts +++ b/src/shared/update-status-types.ts @@ -37,11 +37,15 @@ export type LinuxPackageInstallFailureReason = | 'authentication-denied' | 'package-install-failed' -// Why: the renderer must not infer "no polkit agent" from copy alone — main classifies and the card branches on this discriminant. +export type LinuxPackageInstallRecoveryReason = + | 'manual-install-required' + | LinuxPackageInstallFailureReason + +// Older paired hosts can still publish classified install failures; the manual reason is additive. export type LinuxPackageInstallRecovery = { kind: 'linux-package-install' packageType: LinuxRootPackageType - reason: LinuxPackageInstallFailureReason + reason: LinuxPackageInstallRecoveryReason version: string } @@ -70,6 +74,9 @@ export type UpdateStatus = ( // three-state ambiguity (undefined vs null vs present) and makes exhaustive // checks straightforward. changelog: ChangelogData | null + /** Linux only: a package manager owns this install, so Orca cannot apply the update itself. + * Additive and optional — older clients simply keep offering their own download. */ + externallyManaged?: boolean } | { state: 'not-available'; userInitiated?: boolean } | { state: 'downloading'; percent: number; version: string; activeNudgeId?: string } @@ -77,6 +84,10 @@ export type UpdateStatus = ( | { state: 'error' message: string + /** Known download/install target; absent for check-time failures and older hosts. */ + version?: string + /** Omitted by older hosts and for failures whose retryability is unknown. */ + retryable?: boolean userInitiated?: boolean activeNudgeId?: string recovery?: LinuxPackageInstallRecovery From e5a1e79e8e4af790d38337cd5b8dd2fc10adadc5 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 03:49:40 -0700 Subject: [PATCH 064/398] docs(linux): say which package to install and how updates arrive (#18123) * docs(linux): say which package to install and how updates arrive Closes #5188. Closes #10987. The install guide's entire Linux section was "AppImage and `.deb` builds are available. See the Releases page for details." It named two of the three published packages, gave no basis for choosing between them, and said nothing about updating -- which is the one thing that actually differs between them. Separately, nothing human-facing said the Linux CLI is `orca-ide`; only skills/orca-cli/SKILL.md carried it, which agents read and humans do not. Install page now picks the package by update behaviour: the AppImage self-updates, deb/rpm report the new version and hand over the install command, and a repackaged build is not offered a download it cannot apply. Records that Orca never escalates privileges for the package install, and points at #18086 for the signed repo as planned, not shipped. Adds .rpm to the download list. Release CI builds it (release-cut.yml: `--linux AppImage deb rpm`) and verify-release-required-assets.mjs requires the artifact, so omitting it was just wrong. The CLI command name is now stated where humans hit it -- the CLI reference and overview -- with the GNOME Orca collision as the reason, plus the two places bare `orca` does work: inside Orca-managed terminals (PTY PATH shim) and on a packaged `orca serve` host (the ~/.local/bin dispatcher). The headless guide gains the same note, which is what makes its `orca skills install` lines correct rather than a typo. * docs(linux): fix install ordering, CLI verification, and serve bootstrap Readiness review found ten defects. Two would have had a reader run the wrong program, and one would have had them install a .deb over a live app. Install ordering was reversed. The page said "run it, then quit and reopen Orca"; the ref this is gated to land with says the opposite in four places (linux-package-downloaded-status.ts LINUX_PACKAGE_MANUAL_INSTALL_MESSAGE, "Quit Orca before running the system package install command", plus the recovery card's title, summary and explainer). That wording came from main's older run-then-quit card, which the stack deliberately reversed when it retitled the card to "Manual Install Required". Now: quit first. CLI verification put the Linux caveat *below* `command -v orca`. That check succeeds on any GNOME desktop and resolves to the screen reader, so the reader got a confident hit from the page's own verification step and then invoked the wrong program. Caveat moved above, and the block now spells `orca-ide` literally instead of asking the reader to substitute. The serve bootstrap was circular: the bare-`orca` dispatcher is written *during* serve startup (main-process-runtime-launch.ts), so it can never be the command that starts serve. First launch is `orca-ide serve`. Fixed here and in the two pages this links to. Accuracy: the install command now matches what the code emits -- absolute paths resolved from the trusted directories and a POSIX-single-quoted package path, as pinned by linux-package-install-command.test.ts -- and names the manager fallbacks (dpkg; zypper/dnf/yum/rpm) rather than presenting apt as the only form. The pending path honours XDG_CACHE_HOME. rpm arch tokens are x86_64 and aarch64, not deb's amd64/arm64. arm64 AppImage is linked. Dropped the container example: isExternallyManagedLinuxInstall() needs a root marker AND no trusted package manager, and a Debian-based container has apt, so it is not flagged. --- docs/reference/headless-linux-server.md | 15 ++++++ docs/site/content/docs/cli/overview.mdx | 2 +- docs/site/content/docs/cli/reference.mdx | 18 +++++++- docs/site/content/docs/install.mdx | 56 +++++++++++++++++++++-- docs/site/content/docs/remote-servers.mdx | 8 ++++ docs/site/content/docs/ways-to-run.mdx | 4 +- 6 files changed, 96 insertions(+), 7 deletions(-) diff --git a/docs/reference/headless-linux-server.md b/docs/reference/headless-linux-server.md index 7368678c2e4..50a38cf446e 100644 --- a/docs/reference/headless-linux-server.md +++ b/docs/reference/headless-linux-server.md @@ -357,6 +357,21 @@ the command: This disables a security boundary. Prefer a dedicated unprivileged service user, especially when the listener is reachable beyond localhost. +The Linux CLI is named `orca-ide`, not `orca`, so it never shadows the GNOME +Orca screen reader at `/usr/bin/orca`. The `.deb` and `.rpm` packages put +`orca-ide` on `PATH` themselves at install time; with the AppImage it arrives +as `~/.local/bin/orca-ide` when the CLI is registered. + +A packaged `orca serve` start also writes a bare `orca` into `~/.local/bin` +that execs the same launcher, which is why the skills commands below can be +typed as `orca`. It writes it while starting, so it is never the command that +starts the server — the first launch is `orca-ide serve`, or the AppImage +invoked directly as above. The write is best-effort: it is gated on a packaged +build, it is skipped when no bundled launcher resolves, and it is skipped when +a file Orca does not own already holds that name (ownership is a marker on the +second line of the file). A host that really does run the screen reader keeps +its own `orca`. + ## Pairing troubleshooting - A pairing offer is a capability containing a device credential and E2EE diff --git a/docs/site/content/docs/cli/overview.mdx b/docs/site/content/docs/cli/overview.mdx index 3248e89e1eb..1e966c6217c 100644 --- a/docs/site/content/docs/cli/overview.mdx +++ b/docs/site/content/docs/cli/overview.mdx @@ -14,7 +14,7 @@ import { Callout } from '@/components/docs/prose' The Orca CLI is the `orca` command-line interface for scripting a running Orca editor from any shell. Use it to create and inspect worktrees, drive agent terminals, open files and diffs, automate the built-in browser, run scheduled automations, share HTML/Markdown artifacts, and control Orca-native tools from scripts or AI agents. -It ships with the desktop app; register it under [Settings → General → Orca CLI](/docs/settings). +It ships with the desktop app; register it under [Settings → General → Orca CLI](/docs/settings). On Linux the command is `orca-ide`, because GNOME Orca's screen reader already owns `/usr/bin/orca` — see [Install → Linux](/docs/install#linux). Agents can install the matching Orca CLI skill with: diff --git a/docs/site/content/docs/cli/reference.mdx b/docs/site/content/docs/cli/reference.mdx index 3df0773a4a7..a13ec0fdde4 100644 --- a/docs/site/content/docs/cli/reference.mdx +++ b/docs/site/content/docs/cli/reference.mdx @@ -9,13 +9,29 @@ The `orca` CLI talks to a running Orca runtime. Use it when a shell script or ag ## Verify the runtime -Register the CLI under [Settings → General → Orca CLI](/docs/settings), then check that it can reach Orca: +Register the CLI under [Settings → General → Orca CLI](/docs/settings), then check that it can reach Orca. + + + GNOME Orca — the screen reader that ships with most GNOME desktops — already owns `/usr/bin/orca`, + so Orca's Linux CLI installs as `orca-ide`. Do not check for it with `command -v orca`: that + succeeds on a GNOME desktop and resolves to the screen reader, not to Orca. This page writes + `orca` throughout — read it as `orca-ide` on Linux. See [Install → Linux](/docs/install#linux). + + +On macOS and Windows: ```bash command -v orca orca status --json ``` +On Linux: + +```bash +command -v orca-ide +orca-ide status --json +``` + If Orca is not already running: ```bash diff --git a/docs/site/content/docs/install.mdx b/docs/site/content/docs/install.mdx index f710644d19e..9341cb0fa0d 100644 --- a/docs/site/content/docs/install.mdx +++ b/docs/site/content/docs/install.mdx @@ -31,8 +31,11 @@ import { Callout } from '@/components/docs/prose'
  • **Linux:** - [AppImage](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) · - [.deb](https://github.com/stablyai/orca/releases) + AppImage + [x64](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) · + [arm64](https://github.com/stablyai/orca/releases/latest/download/orca-linux-arm64.AppImage) · + [.deb](https://github.com/stablyai/orca/releases) · + [.rpm](https://github.com/stablyai/orca/releases) — see [Linux](#linux) for which to pick
  • Older versions: [GitHub Releases](https://github.com/stablyai/orca/releases).
  • @@ -59,6 +62,8 @@ On first launch Orca will: Orca auto-updates by default, tracking the **stable** channel. Stable releases are vetted; **RC (release candidate)** builds ship new features first, often daily. +On Linux, whether Orca can apply an update itself depends on which package you installed. See [Linux](#linux) before you pick one. + There is no permanent in-app opt-in for the RC channel. Modifier clicks on **Check for Updates** ([Settings → General → Updates](/docs/settings), or the app / Help menu): | Modifier | Effect | @@ -87,4 +92,49 @@ The default shell can be set to PowerShell or CMD under [Settings → Terminal]( ### Linux -AppImage and `.deb` builds are available. See the Releases page for details. +Each published release ships three Linux packages — an **AppImage**, a **`.deb`**, and an **`.rpm`** — for both x64 and arm64. They contain the same app. What differs is how updates reach you, so pick on that. + +| Package | Pick it when | Updates | +| ------------ | --------------------------------------------------------- | ----------------------------------------------------------------- | +| **AppImage** | You want Orca to update itself, like on macOS and Windows | Orca downloads and applies the update in place | +| **`.deb`** | You manage software with `apt` on Debian or Ubuntu | Orca tells you a version is out and hands you the install command | +| **`.rpm`** | You manage software with `dnf`, `yum`, or `zypper` | Same as `.deb` | + +The AppImage has a stable download link per architecture — [`orca-linux.AppImage`](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) for x64 and [`orca-linux-arm64.AppImage`](https://github.com/stablyai/orca/releases/latest/download/orca-linux-arm64.AppImage) for arm64 — and needs `chmod +x` before its first run, because GitHub release assets carry no permission bits. The `.deb` and `.rpm` filenames carry the version and architecture, and the two formats spell architecture differently (`orca-ide__amd64.deb` or `_arm64.deb`; `orca-ide-.x86_64.rpm` or `.aarch64.rpm`), so take those from the [Releases page](https://github.com/stablyai/orca/releases) rather than a fixed URL. + +#### How updating works + +**The AppImage self-updates.** Choose it if you want automatic updates. Orca checks for a new release, you click **Update**, and it replaces the AppImage in place — the same flow as macOS and Windows. + +**The `.deb` and `.rpm` do not self-update.** Orca still notices the new version and downloads the package, then gives you a **Copy Install Command** button. Copy it rather than retyping it: Orca resolves every program to an absolute path in a trusted system directory and single-quotes the package path, so what you paste looks like this: + +``` +/usr/bin/sudo /usr/bin/apt install -- '/home/you/.cache/orca-updater/pending/orca-ide_1.4.194_amd64.deb' +``` + +Which package manager appears depends on what your system actually has: `apt`, else `dpkg -i`, for a `.deb`; `zypper`, `dnf`, `yum`, then `rpm -Uvh` for an `.rpm`. The download directory follows `XDG_CACHE_HOME` when that is set and falls back to `~/.cache` when it is not. + +**Quit Orca before you run the command**, then reopen it once the install finishes. You are replacing the files of a running application, and the package manager cannot swap them safely underneath a live process. Orca deliberately never escalates privileges to do this for you: installing a system package needs root, `orca serve` runs as an unprivileged user, and a headless machine has no authentication agent to prompt. VS Code and Signal make the same call on `.deb`. + +**A distro-managed build is left alone.** If you are running a repackaged Orca — an AUR build, a Nix derivation — Orca sees that no package manager it can drive owns this install and stops offering a download it could never apply. It still reports that a new version exists, so you can update the way you normally would. + + + [#18086](https://github.com/stablyai/orca/issues/18086) tracks publishing a signed repository so + your OS package manager owns Orca updates the way it owns everything else. It does not exist yet — + today, `.deb` and `.rpm` updates are the manual step described above. + + +#### The CLI command is `orca-ide` + +On Linux the [Orca CLI](/docs/cli/reference) installs as **`orca-ide`**, not `orca`. GNOME Orca — the screen reader that ships by default on Ubuntu and other GNOME desktops — already owns `/usr/bin/orca`, and Orca will not shadow it. The `.deb` and `.rpm` packages are named `orca-ide` for the same reason. + +- The `.deb` and `.rpm` put `orca-ide` on your `PATH` at install time, as `/usr/bin/orca-ide`. +- With the AppImage, register the CLI from [Settings → General → Orca CLI](/docs/settings). That installs `~/.local/bin/orca-ide`. +- Inside Orca's own terminals, bare `orca` works. Orca puts a shim on the `PATH` of the terminals it manages, so agents and scripts running there use the same command as on macOS and Windows. +- On a headless host, a packaged `orca serve` writes a bare `orca` into `~/.local/bin` as it starts, unless a file it does not own already holds that name. It writes that *during* startup, so it is never what starts the server — the first launch is always [`orca-ide serve`](/docs/remote-servers). + +Do not verify with `command -v orca`: on a GNOME desktop that succeeds and resolves to the screen reader. Use `orca-ide` in your own shell and `orca` inside Orca. If you want the short name everywhere and you do not use the screen reader, link it yourself: + +``` +ln -s "$(command -v orca-ide)" ~/.local/bin/orca +``` diff --git a/docs/site/content/docs/remote-servers.mdx b/docs/site/content/docs/remote-servers.mdx index 63f94e35af8..37f86665d6e 100644 --- a/docs/site/content/docs/remote-servers.mdx +++ b/docs/site/content/docs/remote-servers.mdx @@ -126,6 +126,14 @@ Use `orca serve` when the host should run without the desktop window—for examp Install Orca and its bundled CLI on the server, then run: + + The Linux CLI is named `orca-ide`, because GNOME Orca's screen reader already owns + `/usr/bin/orca`. A packaged `orca serve` does write a bare `orca` into `~/.local/bin`, but only + while it is starting, so that shim can never be the command that starts the server. Read + `orca serve` as `orca-ide serve` throughout this page when the host is Linux. See + [Install → Linux](/docs/install#linux). + + ```bash orca serve --pairing-address ``` diff --git a/docs/site/content/docs/ways-to-run.mdx b/docs/site/content/docs/ways-to-run.mdx index 1a9c44ba1d1..e35b63eae90 100644 --- a/docs/site/content/docs/ways-to-run.mdx +++ b/docs/site/content/docs/ways-to-run.mdx @@ -52,10 +52,10 @@ Keep Orca running on a machine you control—an old laptop, Mac mini, home serve **Easiest setup:** install Orca and Tailscale on both computers. On the server, open **Settings → Remote Orca Servers → Advertise this app as a server → New Link**, choose its Tailscale address, and generate an access link. On the client, choose **Add Server** and paste that link. -For a headless Linux server or service-managed VM, use `orca serve` as the alternative: +For a headless Linux server or service-managed VM, use `orca serve` as the alternative. On Linux the CLI is named `orca-ide`, so the first launch is: ```bash -orca serve --pairing-address +orca-ide serve --pairing-address ``` Full detail: [Remote Orca Servers](/docs/remote-servers). From f737f3499f3f9194fc4984b110dd202e5d089856 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 05:36:54 -0700 Subject: [PATCH 065/398] fix(relay): stream an oversized fs.listFiles reply instead of refusing it (#17954) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Opening Orca's own checkout over SSH cannot list its files in one response frame. 22,617 tracked paths average 58 characters, so the 20,001-row page the client asks for serializes to 1,223,415 bytes — past `DISPATCHER_CONTROL_QUEUE_MAX_BYTES`, so `sendResponse` demotes it to the `legacy-response` lane, where an unrelated producer backlog can refuse it as an opaque `ResponseOverCapacity`. Break-even is around 49 characters of average path; any `packages//src/...` monorepo is over the line. Picking a ceiling to refuse at does not fix that, it just moves where it shows up and refuses listings that would have been delivered. `__streamResponse` already exists for exactly this on the git methods, and it is its own negotiation in both directions: an old client never sends it and gets the plain array on the legacy-response lane as before, and an old relay ignores it and answers plainly, which the client detects by the sentinel marker being absent. So fs.listFiles opts into it — no new method, no new opcode, nothing to advertise — and the size of a listing stops being a correctness question. The response-stream registry becomes one per relay, shared by FsHandler and GitHandler. A second registry is not an option and the header of git-response-stream.ts says why: a client keys reassembly on `streamId` alone, so two would hand out the same id and cross-feed chunks, and only the handler that registers `git.responseAck` can credit the window a pump parks on. Also declares `maxResults` on the runtime-RPC `files.listAll` and forwards it. The mechanism "the client names its cap, so a full page reads as truncation" was wired only on the Electron IPC hop; web and mobile were saved incidentally by `remoteFileContentBudget` defaulting the cap inside `listRuntimeFiles`. A new optional field is additive in both directions (wire rule 1). The new Docker-gated spec is claimed by run-ssh-docker-e2e.mjs. The sharded e2e lanes set no ORCA_E2E_SSH_DOCKER, so a Docker-gated spec that no runner names self-skips everywhere and still reports green — pr-e2e-gate-contract enforces that. Closes #12547 --- config/scripts/run-ssh-docker-e2e.mjs | 1 + .../providers/ssh-filesystem-provider.test.ts | 44 +++--- src/main/providers/ssh-filesystem-provider.ts | 7 +- .../methods/files-list-all-page-size.test.ts | 53 +++++++ src/main/runtime/rpc/methods/files.ts | 8 +- src/relay/fs-handler.ts | 25 +++- ...t-files-large-response.integration.test.ts | 130 +++++++++++++++++ src/relay/git-handler.ts | 22 ++- src/relay/git-response-stream.ts | 47 +++++- src/relay/relay-runtime-services.ts | 9 +- tests/e2e/helpers/docker-ssh-relay-target.ts | 10 ++ ...sh-docker-quick-open-large-listing.spec.ts | 134 ++++++++++++++++++ 12 files changed, 441 insertions(+), 49 deletions(-) create mode 100644 src/main/runtime/rpc/methods/files-list-all-page-size.test.ts create mode 100644 src/relay/fs-list-files-large-response.integration.test.ts create mode 100644 tests/e2e/ssh-docker-quick-open-large-listing.spec.ts diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs index 195811f7312..dddfa3e0148 100644 --- a/config/scripts/run-ssh-docker-e2e.mjs +++ b/config/scripts/run-ssh-docker-e2e.mjs @@ -71,6 +71,7 @@ const result = spawnSync( 'tests/e2e/ssh-ai-vault-session-history.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', 'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts', + 'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts', 'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts', 'tests/e2e/ssh-docker-transport-drop-recovery.spec.ts', 'tests/e2e/ssh-external-image-preview.spec.ts', diff --git a/src/main/providers/ssh-filesystem-provider.test.ts b/src/main/providers/ssh-filesystem-provider.test.ts index 65913f0c7e9..b9cb21c3354 100644 --- a/src/main/providers/ssh-filesystem-provider.test.ts +++ b/src/main/providers/ssh-filesystem-provider.test.ts @@ -486,14 +486,16 @@ describe('SshFilesystemProvider', () => { expect(result).toEqual(searchResult) }) - it('listFiles sends fs.listFiles request', async () => { + // Why #12547: a monorepo listing does not fit one control-lane frame, so the request opts into + // response streaming. An old relay ignores `__streamResponse` and answers plainly, which is the + // plain-array case each of these asserts. + it('listFiles sends a streamable fs.listFiles request', async () => { mux.request.mockResolvedValue(['src/index.ts', 'package.json']) const result = await provider.listFiles('/home/user/project') - expect(mux.request).toHaveBeenCalledWith( - 'fs.listFiles', - { rootPath: '/home/user/project' }, - { signal: undefined } - ) + expect(mux.request).toHaveBeenCalledWith('fs.listFiles', { + rootPath: '/home/user/project', + __streamResponse: true + }) expect(result).toEqual(['src/index.ts', 'package.json']) }) @@ -503,26 +505,22 @@ describe('SshFilesystemProvider', () => { maxResults: 20_000, searchQuery: 'target' }) - expect(mux.request).toHaveBeenCalledWith( - 'fs.listFiles', - { - rootPath: '/home/user/project', - excludePaths: ['/home/user/project/worktrees/b'], - maxResults: 20_000, - searchQuery: 'target' - }, - { signal: undefined } - ) + expect(mux.request).toHaveBeenCalledWith('fs.listFiles', { + rootPath: '/home/user/project', + excludePaths: ['/home/user/project/worktrees/b'], + maxResults: 20_000, + searchQuery: 'target', + __streamResponse: true + }) }) it('listFiles omits excludePaths when empty', async () => { mux.request.mockResolvedValue([]) await provider.listFiles('/home/user/project', { excludePaths: [] }) - expect(mux.request).toHaveBeenCalledWith( - 'fs.listFiles', - { rootPath: '/home/user/project' }, - { signal: undefined } - ) + expect(mux.request).toHaveBeenCalledWith('fs.listFiles', { + rootPath: '/home/user/project', + __streamResponse: true + }) }) it('listFiles forwards the cancellation signal to the mux request (#7721)', async () => { @@ -531,8 +529,8 @@ describe('SshFilesystemProvider', () => { await provider.listFiles('/home/user/project', { signal: controller.signal }) expect(mux.request).toHaveBeenCalledWith( 'fs.listFiles', - { rootPath: '/home/user/project' }, - { signal: controller.signal } + { rootPath: '/home/user/project', __streamResponse: true }, + { signal: controller.signal, timeoutMs: undefined } ) }) diff --git a/src/main/providers/ssh-filesystem-provider.ts b/src/main/providers/ssh-filesystem-provider.ts index 70bb06730f8..f6208ea00e9 100644 --- a/src/main/providers/ssh-filesystem-provider.ts +++ b/src/main/providers/ssh-filesystem-provider.ts @@ -1,6 +1,7 @@ import type { SshChannelMultiplexer } from '../ssh/ssh-channel-multiplexer' import { isMethodNotFoundError, readFileViaStream } from '../ssh/ssh-filesystem-stream-reader' import { uploadBuffer } from '../ssh/sftp-upload' +import { requestGitStreamable } from '../ssh/ssh-git-response-stream-reader' import { lstatViaSftp } from './ssh-filesystem-provider-sftp' import { downloadFileViaSftp, @@ -314,7 +315,11 @@ export class SshFilesystemProvider implements IFilesystemProvider { // Why #7721: the signal lets a workspace switch send rpc.cancel so the // relay aborts the full-tree scan instead of stacking abandoned scans // that starve interactive fs.readDir/fs.stat on the shared SSH channel. - return (await this.mux.request('fs.listFiles', params, { + // Why streamable: a monorepo listing serializes past the relay's 1 MiB control lane, and the + // lane it demotes to is refused under unrelated producer load. Opting in moves it to the bulk + // lane in chunks; an old relay ignores the flag and answers plainly, which the reader detects + // by the sentinel marker being absent. + return (await requestGitStreamable(this.mux, 'fs.listFiles', params, { signal: options?.signal })) as string[] } diff --git a/src/main/runtime/rpc/methods/files-list-all-page-size.test.ts b/src/main/runtime/rpc/methods/files-list-all-page-size.test.ts new file mode 100644 index 00000000000..826d0c0f3f0 --- /dev/null +++ b/src/main/runtime/rpc/methods/files-list-all-page-size.test.ts @@ -0,0 +1,53 @@ +/** + * #12547: `files.listAll` did not declare `maxResults`, so "the client names its cap and a full page + * means there is more" was wired only on the Electron IPC hop. Web and mobile were saved incidentally, + * by `remoteFileContentBudget` defaulting the cap inside `listRuntimeFiles`. + */ +import { describe, expect, it, vi } from 'vitest' +import { RpcDispatcher } from '../dispatcher' +import type { RpcRequest } from '../core' +import type { OrcaRuntimeService } from '../../orca-runtime' +import { FILE_METHODS } from './files' + +function makeRequest(method: string, params?: unknown): RpcRequest { + return { id: 'req-1', authToken: 'tok', method, params } +} + +describe('files.listAll page size', () => { + // Why #12547: `maxResults` was wired only on the Electron IPC hop, so "a full page means there is + // more" was true for a desktop client and incidental for web/mobile. Declaring it here is a new + // optional field (wire rule 1): an older host strips it and keeps its own default. + it('forwards a client-named page size for a selected worktree', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + listRuntimeFiles: vi.fn().mockResolvedValue(['src/index.ts']) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: FILE_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('files.listAll', { worktree: 'id:wt-1', maxResults: 20_001 }) + ) + + expect(runtime.listRuntimeFiles).toHaveBeenCalledWith('id:wt-1', { + excludePaths: undefined, + maxResults: 20_001 + }) + expect(response).toMatchObject({ ok: true, result: ['src/index.ts'] }) + }) + + // Why refuse rather than fall back: no released client sends this field, so a malformed value is a + // bug in the caller, not skew — the same call `files.search` already makes for its own maxResults. + it('refuses a malformed page size instead of silently picking one', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + listRuntimeFiles: vi.fn().mockResolvedValue(['src/index.ts']) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: FILE_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('files.listAll', { worktree: 'id:wt-1', maxResults: -3 }) + ) + + expect(response).toMatchObject({ ok: false }) + }) +}) diff --git a/src/main/runtime/rpc/methods/files.ts b/src/main/runtime/rpc/methods/files.ts index d4aaae455bc..ef349a22f84 100644 --- a/src/main/runtime/rpc/methods/files.ts +++ b/src/main/runtime/rpc/methods/files.ts @@ -93,8 +93,13 @@ const FileSearch = WorktreeSelector.extend({ maxResults: z.number().int().positive().optional() }) +// Why: `maxResults` is a new optional field (wire rule 1) — an older host strips it and keeps its +// own default. It existed only on the Electron IPC hop, so "the client names its cap and a full page +// means there is more" was true for desktop and merely incidental for web and mobile, which were +// saved by `remoteFileContentBudget` defaulting the cap inside `listRuntimeFiles`. const FileListAll = WorktreeSelector.extend({ - excludePaths: z.array(z.string()).optional() + excludePaths: z.array(z.string()).optional(), + maxResults: z.number().int().positive().optional() }) const FileUnwatch = z.object({ @@ -236,6 +241,7 @@ export const FILE_METHODS: RpcAnyMethod[] = [ const maxContentBytes = remoteFileContentBudget(clientKind, requestId) return runtime.listRuntimeFiles(params.worktree, { excludePaths: params.excludePaths, + ...(params.maxResults === undefined ? {} : { maxResults: params.maxResults }), ...(signal === undefined ? {} : { signal }), ...(maxContentBytes === undefined ? {} : { maxContentBytes }) }) diff --git a/src/relay/fs-handler.ts b/src/relay/fs-handler.ts index f8514a47191..ec11a806bb8 100644 --- a/src/relay/fs-handler.ts +++ b/src/relay/fs-handler.ts @@ -25,6 +25,7 @@ import { writeRelayFile } from './fs-path-mutation-requests' import { buildExcludePathPrefixes } from '../shared/quick-open-filter' +import { maybeStreamRpcResponse, type GitResponseStreamRegistry } from './git-response-stream' import { readRelayFileContent, readRelayFileStreamMetadata } from './fs-handler-file-read' import { readRelayFileRange } from './fs-handler-file-range' import { FileRangeReadRequestError } from '../shared/file-range-read' @@ -47,12 +48,19 @@ export class FsHandler { private watchRegistry: RelayFilesystemWatchRegistry private streamRegistry = new RelayStreamRegistry() private listFilesScans = new ListFilesScanCoordinator() + private readonly responseStreams: GitResponseStreamRegistry | undefined constructor( dispatcher: RelayDispatcher, _context: RelayContext, - watcherPool?: RelayWatcherProcessPool + watcherPool?: RelayWatcherProcessPool, + // Why passed in rather than owned: GitHandler registers the `git.responseAck` route every pump + // is credited through, and a client keys reassembly on `streamId` alone — see the header of + // git-response-stream.ts. Without one this handler answers plainly, which is the pre-streaming + // behavior rather than a stream nothing can credit. + responseStreams?: GitResponseStreamRegistry ) { + this.responseStreams = responseStreams this.dispatcher = dispatcher this.watchRegistry = new RelayFilesystemWatchRegistry(dispatcher, watcherPool) this.registerHandlers() @@ -204,7 +212,10 @@ export class FsHandler { } } - private listFiles(params: Record, context?: RequestContext): Promise { + private async listFiles( + params: Record, + context?: RequestContext + ): Promise { const rootPath = expandTilde(params.rootPath as string) const maxResults = typeof params.maxResults === 'number' && @@ -224,13 +235,21 @@ export class FsHandler { // Why #7721: full-tree scans are the relay's most expensive request; the // coordinator caps them at one per client, coalescing duplicates and // aborting a stale scan when the workspace changes or the host cancels. - return this.listFilesScans.run({ + const files = await this.listFilesScans.run({ clientId: context?.clientId ?? 0, key: JSON.stringify([rootPath, excludePathPrefixes, maxResults, searchQuery]), signal: context?.signal, start: (signal) => runListFilesScan(rootPath, excludePathPrefixes, signal, maxResults, searchQuery) }) + // Why: a full listing of a real monorepo serializes past the 1 MiB control lane — Orca's own + // checkout is 22.6k paths averaging 58 characters, so a 20,001-row page is ~1.2MB — and the + // legacy-response lane it demotes to is refused under unrelated producer load. Streaming makes + // size stop being a correctness question instead of picking a row or byte ceiling to refuse at. + // A client that did not opt in still gets the plain array, exactly as before. + return this.responseStreams + ? maybeStreamRpcResponse(files, params, context, this.responseStreams, this.dispatcher) + : files } private async workspaceSpaceScan(params: Record, context: RequestContext) { diff --git a/src/relay/fs-list-files-large-response.integration.test.ts b/src/relay/fs-list-files-large-response.integration.test.ts new file mode 100644 index 00000000000..27c7ee7e648 --- /dev/null +++ b/src/relay/fs-list-files-large-response.integration.test.ts @@ -0,0 +1,130 @@ +/** + * #12547: a full `fs.listFiles` reply for a real monorepo does not fit the relay's control lane. + * + * Orca's own checkout is ~22.6k tracked paths averaging 58 characters, so a 20,001-row page + * serializes to ~1.2MB — past `DISPATCHER_CONTROL_QUEUE_MAX_BYTES`, which demotes it to the + * `legacy-response` lane where an unrelated producer backlog can refuse it. Refusing at a fixed row + * or byte ceiling only moves where that shows up; streaming removes it, so these run the real + * dispatcher, the real FsHandler and the real client multiplexer over an in-memory pipe and assert + * an over-budget listing arrives intact — in both wire directions. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { runListFilesScanMock } = vi.hoisted(() => ({ runListFilesScanMock: vi.fn() })) + +vi.mock('./fs-list-files-fallback-chain', () => ({ runListFilesScan: runListFilesScanMock })) +vi.mock('@parcel/watcher', () => ({ subscribe: vi.fn() })) + +import { + SshChannelMultiplexer, + type MultiplexerTransport +} from '../main/ssh/ssh-channel-multiplexer' +import { requestGitStreamable } from '../main/ssh/ssh-git-response-stream-reader' +import { RelayContext } from './context' +import { RelayDispatcher } from './dispatcher' +import { DISPATCHER_CONTROL_QUEUE_MAX_BYTES } from './dispatcher-writer-admission' +import { FsHandler } from './fs-handler' +import { GitHandler } from './git-handler' +import { GitResponseStreamRegistry } from './git-response-stream' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../shared/quick-open-listing-limits' + +/** Shaped like this repository: `packages//src/...`, ~58 characters. */ +function monorepoPaths(count: number): string[] { + return Array.from( + { length: count }, + (_, index) => + `packages/pkg-${String(index % 64).padStart(2, '0')}/src/renderer/components/entry-${String(index).padStart(6, '0')}.tsx` + ) +} + +describe('Integration: an over-budget fs.listFiles reply (#12547)', () => { + let mux: SshChannelMultiplexer + let dispatcher: RelayDispatcher + let fsHandler: FsHandler + let gitHandler: GitHandler + let writtenFrames: number[] + + beforeEach(() => { + runListFilesScanMock.mockReset() + writtenFrames = [] + + let relayFeed: (data: Buffer) => void + const clientDataCallbacks: ((data: Buffer) => void)[] = [] + const clientTransport: MultiplexerTransport = { + write: (data: Buffer) => { + setImmediate(() => relayFeed?.(data)) + }, + onData: (cb) => { + clientDataCallbacks.push(cb) + }, + onClose: () => {} + } + dispatcher = new RelayDispatcher((data: Buffer) => { + writtenFrames.push(data.length) + setImmediate(() => { + for (const cb of clientDataCallbacks) { + cb(data) + } + }) + return true + }) + relayFeed = (data: Buffer) => dispatcher.feed(data) + // Why: the same single registry production wires, so `git.responseAck` — registered by + // GitHandler — credits the pump an fs.listFiles stream parks on. + const responseStreams = new GitResponseStreamRegistry() + const context = new RelayContext() + fsHandler = new FsHandler(dispatcher, context, undefined, responseStreams) + gitHandler = new GitHandler(dispatcher, context, undefined, responseStreams) + mux = new SshChannelMultiplexer(clientTransport) + }) + + afterEach(() => { + mux.dispose() + dispatcher.dispose() + fsHandler.dispose() + gitHandler.dispose() + }) + + it('delivers a page too large for the control lane, in chunks no frame has to carry', async () => { + const files = monorepoPaths(QUICK_OPEN_LISTING_MAX_RESULTS) + // Precondition, measured from the payload rather than asserted between two constants: this is + // the listing that does not fit, which is what makes the rest of the test mean anything. + expect(Buffer.byteLength(JSON.stringify(files), 'utf8')).toBeGreaterThan( + DISPATCHER_CONTROL_QUEUE_MAX_BYTES + ) + runListFilesScanMock.mockResolvedValue(files) + + const received = await requestGitStreamable(mux, 'fs.listFiles', { + rootPath: '/remote/root', + maxResults: QUICK_OPEN_LISTING_MAX_RESULTS + }) + + expect(received).toEqual(files) + expect(Math.max(...writtenFrames)).toBeLessThan(DISPATCHER_CONTROL_QUEUE_MAX_BYTES) + }) + + it('still answers a client that never opts into streaming, with the whole array', async () => { + const files = monorepoPaths(QUICK_OPEN_LISTING_MAX_RESULTS) + runListFilesScanMock.mockResolvedValue(files) + + // Why: an old client sends neither `__streamResponse` nor `maxResults`. It gets one plain frame + // on the legacy-response lane, as it did before this call ever learned to stream. + const received = await mux.request('fs.listFiles', { rootPath: '/remote/root' }) + + expect(received).toEqual(files) + expect(Math.max(...writtenFrames)).toBeGreaterThan(DISPATCHER_CONTROL_QUEUE_MAX_BYTES) + }) + + it('leaves a reply that fits on the plain response path', async () => { + const files = monorepoPaths(100) + runListFilesScanMock.mockResolvedValue(files) + + const received = await requestGitStreamable(mux, 'fs.listFiles', { + rootPath: '/remote/root', + maxResults: 100 + }) + + expect(received).toEqual(files) + expect(Math.max(...writtenFrames)).toBeLessThan(DISPATCHER_CONTROL_QUEUE_MAX_BYTES) + }) +}) diff --git a/src/relay/git-handler.ts b/src/relay/git-handler.ts index 58f47da9fce..66bbd6d3cd1 100644 --- a/src/relay/git-handler.ts +++ b/src/relay/git-handler.ts @@ -10,8 +10,7 @@ import { createSubmodulePathsCache, type SubmodulePathsCache } from './git-handler-submodule-ops' -import { GitResponseStreamRegistry } from './git-response-stream' -import { GIT_RESPONSE_STREAM_THRESHOLD } from './protocol' +import { GitResponseStreamRegistry, maybeStreamRpcResponse } from './git-response-stream' import { clearGitStatusLineStatsCache } from '../shared/git-status-line-stats-cache' import { invalidateGitBranchLineTotalInFlight } from '../shared/git-branch-line-total' import { buildRelayGitEnv, buildRelayUnattendedGitEnv } from './relay-command-env' @@ -68,9 +67,6 @@ export class GitHandler { private dispatcher: RelayDispatcher private readonly gitDiffReadDedupe = new InFlightPromiseDedupe() private readonly gitCapabilities = new GitCapabilityCache() - // Why: use the bulk lane so large responses do not block interactive PTY echo. - private readonly responseStreams = new GitResponseStreamRegistry() - // Why: cache .gitmodules per instance to avoid SSH reads and test leakage. private submodulePathsCache: SubmodulePathsCache = createSubmodulePathsCache() @@ -78,7 +74,12 @@ export class GitHandler { constructor( dispatcher: RelayDispatcher, _context: RelayContext, - private readonly watcherRegistry?: GitHandlerWatcherRegistry + private readonly watcherRegistry?: GitHandlerWatcherRegistry, + // Why: use the bulk lane so large responses do not block interactive PTY echo. This handler + // registers the `git.responseAck` route below, so in production it takes the relay's single + // registry and FsHandler is handed the same one — see the header of git-response-stream.ts for + // why a second registry both collides on stream ids and stalls on credit. + private readonly responseStreams: GitResponseStreamRegistry = new GitResponseStreamRegistry() ) { this.dispatcher = dispatcher const handlers = createGitHandlerOperationSet({ @@ -132,14 +133,7 @@ export class GitHandler { params: Record, context: RequestContext | undefined ): unknown { - if (params.__streamResponse !== true || !context) { - return result - } - const payload = Buffer.from(JSON.stringify(result ?? null), 'utf-8') - if (payload.length <= GIT_RESPONSE_STREAM_THRESHOLD) { - return result - } - return this.responseStreams.startStream(payload, this.dispatcher, context) + return maybeStreamRpcResponse(result, params, context, this.responseStreams, this.dispatcher) } private clearGitMutationReadCaches(): void { diff --git a/src/relay/git-response-stream.ts b/src/relay/git-response-stream.ts index 3ccbd66ee8d..c9d8d2ce290 100644 --- a/src/relay/git-response-stream.ts +++ b/src/relay/git-response-stream.ts @@ -1,11 +1,22 @@ -// Streams large git RPC responses (diff family + exec) onto the bulk lane in -// chunks instead of one JSON-RPC frame, so a big diff cannot head-of-line-block -// interactive pty.data echo on the shared SSH channel. Mirrors the fs -// read-stream credit-window pattern (see fs-handler-file-read.ts) but the -// payload is an in-memory serialized string rather than a file handle. +// Streams large RPC responses onto the bulk lane in chunks instead of one +// JSON-RPC frame, so a big reply cannot head-of-line-block interactive pty.data +// echo on the shared SSH channel. Mirrors the fs read-stream credit-window +// pattern (see fs-handler-file-read.ts) but the payload is an in-memory +// serialized string rather than a file handle. +// +// ONE REGISTRY PER RELAY. The `git.*` method names below are the shipped wire +// spelling and are permanent, the way an opcode number is, so a second handler +// that needs streaming (`fs.listFiles` is the first) shares this instance rather +// than minting its own. A second registry is not an option: a client keys +// reassembly on `streamId` alone, so two would hand out the same id and +// cross-feed each other's chunks, and only the handler that registers +// `git.responseAck` can credit the ack window a pump parks on — the other's +// streams would stall at STREAM_ACK_WINDOW_CHUNKS forever. See +// `relay-runtime-services.ts` for the wiring. import type { RelayDispatcher, RequestContext } from './dispatcher' import { GIT_RESPONSE_CHUNK_SIZE, + GIT_RESPONSE_STREAM_THRESHOLD, STREAM_ACK_WINDOW_CHUNKS, STREAM_ACK_STALL_RECHECK_MS, type GitResponseStreamMarker @@ -220,3 +231,29 @@ export class GitResponseStreamRegistry { this.streams.clear() } } + +/** + * Opt-in response streaming, shared by every handler that can answer with a + * payload too large for one control-lane frame. + * + * `__streamResponse` is its own negotiation in both directions: an old client + * never sends it and gets the plain result, and an old relay ignores it and + * answers plainly, which the client detects by the sentinel marker being absent. + * So there is no new method and no capability to advertise. + */ +export function maybeStreamRpcResponse( + result: unknown, + params: Record, + context: RequestContext | undefined, + registry: GitResponseStreamRegistry, + dispatcher: RelayDispatcher +): unknown { + if (params.__streamResponse !== true || !context) { + return result + } + const payload = Buffer.from(JSON.stringify(result ?? null), 'utf-8') + if (payload.length <= GIT_RESPONSE_STREAM_THRESHOLD) { + return result + } + return registry.startStream(payload, dispatcher, context) +} diff --git a/src/relay/relay-runtime-services.ts b/src/relay/relay-runtime-services.ts index 36276ed9b70..485c9e787d1 100644 --- a/src/relay/relay-runtime-services.ts +++ b/src/relay/relay-runtime-services.ts @@ -6,6 +6,7 @@ import { RelayContext, expandTilde } from './context' import { PtyHandler } from './pty-handler' import { FsHandler } from './fs-handler' import { GitHandler } from './git-handler' +import { GitResponseStreamRegistry } from './git-response-stream' import { PreflightHandler } from './preflight-handler' import { ExternalAutomationsHandler } from './external-automations-handler' import { PortScanHandler } from './port-scan-handler' @@ -51,13 +52,17 @@ export class RelayRuntimeServices { ) this.ptyHandler.setSourcePublication(this.ptySourcePublication) - this.fsHandler = new FsHandler(dispatcher, context) + // Why one instance for both handlers: a client reassembles a streamed reply by `streamId` alone, + // so two registries would hand out the same id, and only GitHandler routes the `git.responseAck` + // credit every pump waits on. A second registry is not an option — see git-response-stream.ts. + const responseStreams = new GitResponseStreamRegistry() + this.fsHandler = new FsHandler(dispatcher, context, undefined, responseStreams) const watchRegistry = this.fsHandler.getWatchRegistry() this.ptyHandler.setWorktreeRemovalCoordinator(watchRegistry) watchRegistry.setWorktreePtyTeardown((rootPath) => this.ptyHandler.shutdownForWorktreePath(rootPath) ) - this.gitHandler = new GitHandler(dispatcher, context, watchRegistry) + this.gitHandler = new GitHandler(dispatcher, context, watchRegistry, responseStreams) const preflightHandler = new PreflightHandler(dispatcher) this.skillInstallHandler = new SkillInstallHandler(dispatcher) const externalAutomationsHandler = new ExternalAutomationsHandler(dispatcher) diff --git a/tests/e2e/helpers/docker-ssh-relay-target.ts b/tests/e2e/helpers/docker-ssh-relay-target.ts index 78534499943..f535b22d073 100644 --- a/tests/e2e/helpers/docker-ssh-relay-target.ts +++ b/tests/e2e/helpers/docker-ssh-relay-target.ts @@ -211,6 +211,16 @@ export function writeDockerSshRelayTargetFile( ) } +/** Why not `writeDockerSshRelayTargetFile`: that one passes the contents as a shell argument, so a + * payload the size of a real repository's path list exceeds ARG_MAX before it reaches the shell. */ +export function copyFileIntoDockerSshRelayTarget( + target: DockerSshRelayTarget, + localPath: string, + remotePath: string +): void { + run('docker', ['cp', localPath, `${target.containerName}:${remotePath}`], { timeoutMs: 120_000 }) +} + export function startDockerSshRelayTarget(testInfo: TestInfo): DockerSshRelayTarget { const host = process.env.ORCA_E2E_SSH_TARGET_HOST?.trim() || '127.0.0.1' if (host === 'localhost' || host === '::1' || host.startsWith('127.')) { diff --git a/tests/e2e/ssh-docker-quick-open-large-listing.spec.ts b/tests/e2e/ssh-docker-quick-open-large-listing.spec.ts new file mode 100644 index 00000000000..ee814ab02db --- /dev/null +++ b/tests/e2e/ssh-docker-quick-open-large-listing.spec.ts @@ -0,0 +1,134 @@ +/** + * #12547 acceptance: open a repository the size of Orca's own checkout over SSH and list its files. + * + * The remote tree is seeded from this repository's real `git ls-files` output, so the payload has + * the shape that broke: ~22.6k paths averaging 58 characters, whose 20,001-row page serializes to + * ~1.2MB — past `DISPATCHER_CONTROL_QUEUE_MAX_BYTES`. Both wire directions are exercised over the + * real relay: a current client, and a client that names no `maxResults` at all. + */ +import { execFileSync } from 'node:child_process' +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import path from 'node:path' + +import { expect, test } from './helpers/orca-app' +import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' +import { + cleanupDockerSshRelayTarget, + copyFileIntoDockerSshRelayTarget, + execDockerSshRelayTargetCommand, + shellQuote, + startDockerSshRelayTarget, + type DockerSshRelayTarget +} from './helpers/docker-ssh-relay-target' +import { ensureDockerSshRelayImage } from './helpers/docker-ssh-relay-image' +import { waitForSessionReady } from './helpers/store' +import { shouldIncludeQuickOpenPath } from '../../src/shared/quick-open-filter' + +const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' +const REMOTE_REPO_PATH = '/tmp/orca-quick-open-large-listing-repo' +const REMOTE_PATH_LIST = '/tmp/orca-quick-open-large-listing-paths.txt' +/** What the desktop client asks for; a full page is what it reads as "there is more". */ +const CLIENT_PAGE_SIZE = 20_001 + +function thisRepositoryTrackedPaths(): string[] { + const root = execFileSync('git', ['rev-parse', '--show-toplevel'], { encoding: 'utf8' }).trim() + // Why -z: `git ls-files` C-quotes any path with a special character, which would seed a tree that + // does not match the one being measured. + return execFileSync('git', ['ls-files', '-z'], { + cwd: root, + encoding: 'utf8', + maxBuffer: 1024 * 1024 * 256 + }) + .split('\0') + .filter(Boolean) +} + +function seedRemoteTree(target: DockerSshRelayTarget, paths: string[]): void { + const stagingDir = mkdtempSync(path.join(tmpdir(), 'orca-quick-open-large-listing-')) + try { + const localList = path.join(stagingDir, 'paths.txt') + writeFileSync(localList, `${paths.join('\n')}\n`) + copyFileIntoDockerSshRelayTarget(target, localList, REMOTE_PATH_LIST) + } finally { + rmSync(stagingDir, { recursive: true, force: true }) + } + const seedScript = [ + "const fs = require('fs'), path = require('path')", + `const list = fs.readFileSync(${JSON.stringify(REMOTE_PATH_LIST)}, 'utf8').split('\\n').filter(Boolean)`, + 'const seen = new Set()', + 'for (const entry of list) {', + ' const dir = path.dirname(entry)', + ' if (dir !== "." && !seen.has(dir)) { fs.mkdirSync(dir, { recursive: true }); seen.add(dir) }', + " fs.writeFileSync(entry, '')", + '}' + ].join(';') + const encoded = Buffer.from(seedScript, 'utf8').toString('base64') + execDockerSshRelayTargetCommand( + target, + [ + `rm -rf ${shellQuote(REMOTE_REPO_PATH)}`, + `mkdir -p ${shellQuote(REMOTE_REPO_PATH)}`, + `cd ${shellQuote(REMOTE_REPO_PATH)}`, + 'git init -q', + 'git config user.email e2e@test.local', + 'git config user.name "Orca Docker SSH E2E"', + `node -e ${shellQuote(`eval(Buffer.from('${encoded}', 'base64').toString('utf8'))`)}`, + 'git add -A', + 'git commit -q -m "seed monorepo-shaped tree"' + ].join(' && ') + ) +} + +test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run the Docker SSH relay lane') + +test('lists a monorepo-sized remote workspace, with and without a client page size (#12547)', async ({ + orcaPage +}, testInfo) => { + test.setTimeout(420_000) + let target: DockerSshRelayTarget | null = null + try { + const trackedPaths = thisRepositoryTrackedPaths() + expect(trackedPaths.length).toBeGreaterThan(CLIENT_PAGE_SIZE) + // Why the real predicate rather than a copy of it: Quick Open prunes a few tracked paths on + // purpose (`.husky/` among them), and a hand-written expectation would go stale the first time + // that list changes and read as a transport bug. + const listablePaths = trackedPaths.filter(shouldIncludeQuickOpenPath) + // Precondition, measured rather than assumed: the page a current client asks for does not fit + // one control-lane frame, which is the listing that used to be refused outright. + expect( + Buffer.byteLength(JSON.stringify(trackedPaths.slice(0, CLIENT_PAGE_SIZE)), 'utf8') + ).toBeGreaterThan(1024 * 1024) + + ensureDockerSshRelayImage(process.cwd()) + target = startDockerSshRelayTarget(testInfo) + seedRemoteTree(target, trackedPaths) + + await waitForSessionReady(orcaPage) + const connected = await connectDockerSshRelayTarget(orcaPage, target, { + remotePath: REMOTE_REPO_PATH + }) + + const listFiles = async (maxResults?: number): Promise => + orcaPage.evaluate( + ({ connectionId, rootPath, maxResults }) => + window.api.fs.listFiles({ + rootPath, + connectionId, + ...(maxResults === undefined ? {} : { maxResults }) + }), + { connectionId: connected.targetId, rootPath: REMOTE_REPO_PATH, maxResults } + ) + + const currentClient = await listFiles(CLIENT_PAGE_SIZE) + expect(currentClient).toHaveLength(CLIENT_PAGE_SIZE) + + // Why: a client that predates `maxResults` on this call sends none at all, and it cannot + // reassemble a streamed reply either — it has to be answered on the plain response path. + const oldClient = await listFiles() + expect(oldClient).toHaveLength(listablePaths.length) + expect(new Set(oldClient)).toEqual(new Set(listablePaths)) + } finally { + cleanupDockerSshRelayTarget(target) + } +}) From 8dc3c1dd9787935bd14f9bab03f5ebdbe706a8e5 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:21:28 -0700 Subject: [PATCH 066/398] Display favicons for browser website entries (#18099) * Display favicons for browser website entries Capture favicons from pages as they load and persist them with browser history entries. Display favicons in tabs, tab creation search results, and palette searches to improve visual recognition of websites and help users identify pages at a glance. * Fix favicon retry on back navigation after load failure Reset the favicon failure cache when the favicon URL changes, enabling retry of a previously failed favicon when navigating back to the same URL. Distinguish between explicit null (clear cached favicon) and omitted (don't update history), so stale favicons don't persist incorrectly. --- .../src/components/browser-favicon.tsx | 59 +++++++++++++++++++ .../host-guest/attach-browser-page-webview.ts | 4 +- ...rowser-page-webview-navigation-handlers.ts | 6 +- .../components/tab-bar/BrowserTab.test.tsx | 34 ++++++++++- .../src/components/tab-bar/BrowserTab.tsx | 53 ++--------------- .../TabBarCreateEntry.history.test.tsx | 13 +++- .../tab-bar/TabBarCreateEntryRow.tsx | 5 +- .../tab-bar/open-tab-entry-dedupe.test.ts | 3 +- .../tab-bar/open-tab-search.test.ts | 10 +++- .../src/components/tab-bar/open-tab-search.ts | 4 +- .../open-tab-selection-routing.test.ts | 3 +- ...ee-jump-palette-browser-simulator-rows.tsx | 5 +- .../src/lib/browser-palette-search.test.ts | 20 +++++++ .../src/lib/browser-palette-search.ts | 2 + src/renderer/src/store/slices/browser.test.ts | 47 +++++++++++++++ .../slices/browser/browser-history-actions.ts | 11 +++- .../browser/browser-page-state-actions.ts | 12 ++++ .../slices/browser/browser-slice-contract.ts | 2 +- src/shared/browser-workspace-types.ts | 1 + .../workspace-session-browser-schema.ts | 1 + src/shared/workspace-session-schema.test.ts | 4 ++ 21 files changed, 233 insertions(+), 66 deletions(-) create mode 100644 src/renderer/src/components/browser-favicon.tsx diff --git a/src/renderer/src/components/browser-favicon.tsx b/src/renderer/src/components/browser-favicon.tsx new file mode 100644 index 00000000000..ecde4065f03 --- /dev/null +++ b/src/renderer/src/components/browser-favicon.tsx @@ -0,0 +1,59 @@ +import { useState } from 'react' +import { Globe } from 'lucide-react' +import { cn } from '@/lib/utils' + +function displayableFaviconUrl(faviconUrl: string | null | undefined): string | null { + const trimmed = faviconUrl?.trim() + if (!trimmed) { + return null + } + if (trimmed.startsWith('data:image/')) { + return trimmed + } + try { + const url = new URL(trimmed) + return url.protocol === 'http:' || url.protocol === 'https:' ? trimmed : null + } catch { + return null + } +} + +export function BrowserFavicon({ + faviconUrl, + className, + fallbackClassName +}: { + faviconUrl: string | null | undefined + className?: string + fallbackClassName?: string +}): React.JSX.Element { + const displayUrl = displayableFaviconUrl(faviconUrl) + const [failedUrl, setFailedUrl] = useState(null) + + // Why: reset during render on any favicon identity change — including a clear to null while + // a page loads — so navigating back to the same url retries instead of keeping the fallback. + if (failedUrl !== null && failedUrl !== displayUrl) { + setFailedUrl(null) + } + + if (displayUrl && failedUrl !== displayUrl) { + return ( + setFailedUrl(displayUrl)} + /> + ) + } + + return
    diff --git a/src/renderer/src/lib/browser-palette-search.test.ts b/src/renderer/src/lib/browser-palette-search.test.ts index 41cfd58f54d..759e1cc5421 100644 --- a/src/renderer/src/lib/browser-palette-search.test.ts +++ b/src/renderer/src/lib/browser-palette-search.test.ts @@ -102,6 +102,26 @@ describe('browser-palette-search', () => { expect(result.executionHostId).toBe('ssh:box') }) + it('carries the page favicon into palette results', () => { + const faviconUrl = 'https://example.com/favicon.ico' + const [result] = searchBrowserPages( + [ + makeEntry({ + page: makePage({ faviconUrl }), + workspace: makeWorkspace(), + worktree: makeWorktree(), + repoName: 'repo/one', + worktreeSortIndex: 0, + isCurrentPage: false, + isCurrentWorktree: false + }) + ], + '' + ) + + expect(result.faviconUrl).toBe(faviconUrl) + }) + it('keeps empty-query ordering deterministic and context-first', () => { const results = searchBrowserPages( [ diff --git a/src/renderer/src/lib/browser-palette-search.ts b/src/renderer/src/lib/browser-palette-search.ts index 1529c46860e..0cd9f7e0618 100644 --- a/src/renderer/src/lib/browser-palette-search.ts +++ b/src/renderer/src/lib/browser-palette-search.ts @@ -42,6 +42,7 @@ export type BrowserPaletteSearchResult = { workspaceId: string worktreeId: string title: string + faviconUrl: string | null /** Raw page URL, so callers can dedupe a row against another list of destinations. */ url: string secondaryText: string @@ -153,6 +154,7 @@ function baseResult(entry: SearchableBrowserPage): BrowserPaletteSearchResult { workspaceId: entry.workspace.id, worktreeId: entry.worktree.id, title: entry.page.title || formattedUrl, + faviconUrl: entry.page.faviconUrl, url: entry.page.url, secondaryText: formattedUrl, workspaceLabel: entry.workspace.label ?? null, diff --git a/src/renderer/src/store/slices/browser.test.ts b/src/renderer/src/store/slices/browser.test.ts index ba014db3aa3..fc1620871d5 100644 --- a/src/renderer/src/store/slices/browser.test.ts +++ b/src/renderer/src/store/slices/browser.test.ts @@ -302,6 +302,53 @@ describe('createBrowserSlice annotations', () => { expect(store.getState().browserTabsByWorktree).toBe(browserTabsByWorktree) }) + it('persists a captured favicon with history and refreshes it with the page state', () => { + const store = createTestStore() + const tab = store.getState().createBrowserTab('wt-1', 'https://example.com', { + title: 'Example' + }) + const pageId = tab.activePageId + if (!pageId) { + throw new Error('Expected a new browser page') + } + const initialFavicon = 'https://example.com/favicon.ico' + const refreshedFavicon = 'https://cdn.example.com/favicon.png' + + store.getState().addBrowserHistoryEntry('https://example.com', 'Example', initialFavicon) + expect(store.getState().browserUrlHistory[0]?.faviconUrl).toBe(initialFavicon) + + store.getState().updateBrowserPageState(pageId, { faviconUrl: refreshedFavicon }) + expect(store.getState().browserUrlHistory[0]?.faviconUrl).toBe(refreshedFavicon) + }) + + it('clears a stale history favicon when a page reports none, and keeps it when none is reported', () => { + const store = createTestStore() + const tab = store.getState().createBrowserTab('wt-1', 'https://example.com', { + title: 'Example' + }) + const pageId = tab.activePageId + if (!pageId) { + throw new Error('Expected a new browser page') + } + const favicon = 'https://example.com/favicon.ico' + + store.getState().addBrowserHistoryEntry('https://example.com', 'Example', favicon) + store.getState().updateBrowserPageState(pageId, { faviconUrl: favicon }) + + // An omitted favicon leaves the stored one alone; an explicit null clears it. + store.getState().addBrowserHistoryEntry('https://example.com', 'Example') + expect(store.getState().browserUrlHistory[0]?.faviconUrl).toBe(favicon) + + store.getState().updateBrowserPageState(pageId, { faviconUrl: null }) + expect(store.getState().browserUrlHistory[0]?.faviconUrl).toBeNull() + + store.getState().addBrowserHistoryEntry('https://example.com', 'Example', favicon) + expect(store.getState().browserUrlHistory[0]?.faviconUrl).toBe(favicon) + + store.getState().addBrowserHistoryEntry('https://example.com', 'Example', null) + expect(store.getState().browserUrlHistory[0]?.faviconUrl).toBeNull() + }) + it('repairs a stale active browser unified-tab label on an otherwise unchanged title update', () => { const store = createTestStore() const tab = store.getState().createBrowserTab('wt-1', 'https://example.com', { diff --git a/src/renderer/src/store/slices/browser/browser-history-actions.ts b/src/renderer/src/store/slices/browser/browser-history-actions.ts index 97ebf8b873b..85c5b4f5479 100644 --- a/src/renderer/src/store/slices/browser/browser-history-actions.ts +++ b/src/renderer/src/store/slices/browser/browser-history-actions.ts @@ -60,7 +60,7 @@ export function createBrowserHistoryActions( }) }, - addBrowserHistoryEntry: (url, title) => { + addBrowserHistoryEntry: (url, title, faviconUrl) => { const safeUrl = redactKagiSessionToken(url) if (safeUrl === ORCA_BROWSER_BLANK_URL || safeUrl === 'about:blank' || !safeUrl) { return @@ -71,7 +71,13 @@ export function createBrowserHistoryActions( let next: BrowserHistoryEntry[] = existing ? s.browserUrlHistory.map((entry) => entry === existing - ? { ...entry, title, lastVisitedAt: Date.now(), visitCount: entry.visitCount + 1 } + ? { + ...entry, + title, + ...(faviconUrl !== undefined ? { faviconUrl } : {}), + lastVisitedAt: Date.now(), + visitCount: entry.visitCount + 1 + } : entry ) : [ @@ -79,6 +85,7 @@ export function createBrowserHistoryActions( url: safeUrl, normalizedUrl: normalized, title, + ...(faviconUrl !== undefined ? { faviconUrl } : {}), lastVisitedAt: Date.now(), visitCount: 1 }, diff --git a/src/renderer/src/store/slices/browser/browser-page-state-actions.ts b/src/renderer/src/store/slices/browser/browser-page-state-actions.ts index 15e4892b351..83aa3c23673 100644 --- a/src/renderer/src/store/slices/browser/browser-page-state-actions.ts +++ b/src/renderer/src/store/slices/browser/browser-page-state-actions.ts @@ -9,6 +9,7 @@ import { normalizeBrowserTitle, normalizeUrl } from '../browser-page-records' +import { normalizeBrowserHistoryUrl } from '../../../../../shared/workspace-session-browser-history' export function createBrowserPageStateActions( set: BrowserSliceSet, @@ -100,6 +101,17 @@ export function createBrowserPageStateActions( [workspace.id]: nextPages } } + if (updates.faviconUrl !== undefined && updates.faviconUrl !== page.faviconUrl) { + const historyIndex = s.browserUrlHistory.findIndex( + (entry) => entry.normalizedUrl === normalizeBrowserHistoryUrl(page.url) + ) + const historyEntry = s.browserUrlHistory[historyIndex] + if (historyEntry && historyEntry.faviconUrl !== updates.faviconUrl) { + nextState.browserUrlHistory = s.browserUrlHistory.map((entry, index) => + index === historyIndex ? { ...entry, faviconUrl: updates.faviconUrl } : entry + ) + } + } if (!browserWorkspaceMirrorFieldsEqual(workspace, nextWorkspace)) { nextState.browserTabsByWorktree = { ...s.browserTabsByWorktree, diff --git a/src/renderer/src/store/slices/browser/browser-slice-contract.ts b/src/renderer/src/store/slices/browser/browser-slice-contract.ts index c5577aaf0c4..ee50fa1a087 100644 --- a/src/renderer/src/store/slices/browser/browser-slice-contract.ts +++ b/src/renderer/src/store/slices/browser/browser-slice-contract.ts @@ -239,7 +239,7 @@ export type BrowserSlice = { ) => Promise clearDefaultSessionCookies: () => Promise browserUrlHistory: BrowserHistoryEntry[] - addBrowserHistoryEntry: (url: string, title: string) => void + addBrowserHistoryEntry: (url: string, title: string, faviconUrl?: string | null) => void workspaceDocHistory: WorkspaceDocHistoryEntry[] /** A visit bumps recency and count; a title-only refresh (bump: false) renames the row. */ recordWorkspaceDocVisit: ( diff --git a/src/shared/browser-workspace-types.ts b/src/shared/browser-workspace-types.ts index 5b2a1e7f33d..ecab57c8783 100644 --- a/src/shared/browser-workspace-types.ts +++ b/src/shared/browser-workspace-types.ts @@ -2,6 +2,7 @@ export type BrowserHistoryEntry = { url: string normalizedUrl: string title: string + faviconUrl?: string | null lastVisitedAt: number visitCount: number } diff --git a/src/shared/workspace-session-browser-schema.ts b/src/shared/workspace-session-browser-schema.ts index 71f88fb3b0c..159905e5ab9 100644 --- a/src/shared/workspace-session-browser-schema.ts +++ b/src/shared/workspace-session-browser-schema.ts @@ -117,6 +117,7 @@ const browserHistoryEntrySchema = z.object({ url: z.string(), normalizedUrl: z.string(), title: z.string(), + faviconUrl: z.string().nullable().optional(), lastVisitedAt: z.number(), visitCount: z.number() }) diff --git a/src/shared/workspace-session-schema.test.ts b/src/shared/workspace-session-schema.test.ts index 66e2ddcc514..81bc37b6b65 100644 --- a/src/shared/workspace-session-schema.test.ts +++ b/src/shared/workspace-session-schema.test.ts @@ -504,6 +504,7 @@ describe('parseWorkspaceSession', () => { url: `https://example.com/${index}`, normalizedUrl: `https://example.com/${index}`, title: `Example ${index}`, + faviconUrl: index === 0 ? 'https://example.com/favicon.ico' : null, lastVisitedAt: 1_700_000_000_000 - index, visitCount: 1 })) @@ -512,6 +513,9 @@ describe('parseWorkspaceSession', () => { expect(result.ok).toBe(true) if (result.ok) { expect(result.value.browserUrlHistory).toHaveLength(MAX_BROWSER_HISTORY_ENTRIES) + expect(result.value.browserUrlHistory?.[0]?.faviconUrl).toBe( + 'https://example.com/favicon.ico' + ) expect(result.value.browserUrlHistory?.at(-1)?.url).toBe('https://example.com/199') } }) From 6062edf2962e1dc33c09d1be6bfd5a9469e969b8 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 10:39:46 -0700 Subject: [PATCH 067/398] test: simplify remote pane link routing to server-hosted placement (#18219) Remote-pane links are now explicitly server-hosted regardless of generic client-hosted preference. Remove client-hosted placement verification, placement-switching test acts, and related type definitions. Focus the test on verifying the core invariant: links stay server-hosted on their owning runtime. --- ...d-remote-browser-link-open-routing.spec.ts | 139 ++++-------------- 1 file changed, 29 insertions(+), 110 deletions(-) diff --git a/tests/e2e/paired-remote-browser-link-open-routing.spec.ts b/tests/e2e/paired-remote-browser-link-open-routing.spec.ts index 695d39c251b..a8b15d97a03 100644 --- a/tests/e2e/paired-remote-browser-link-open-routing.spec.ts +++ b/tests/e2e/paired-remote-browser-link-open-routing.spec.ts @@ -1,7 +1,6 @@ import { createServer, type IncomingMessage, type Server, type ServerResponse } from 'node:http' import type { AddressInfo } from 'node:net' import type { Page } from '@stablyai/playwright-test' -import { parseBrowserNetworkExecutionHostKey } from '../../src/main/browser/browser-network-execution-route' import { LOCAL_EXECUTION_HOST_ID } from '../../src/shared/execution-host' import { readOwnedPageUrls } from './helpers/client-hosted-browser-observer' import { @@ -16,11 +15,8 @@ import { } from './helpers/paired-electron-client' // The link is a dev-server URL on the pane runtime's network, so a client-local fallback would -// silently load a *different machine's* server. Which machine renders the pixels no longer answers -// that: under client-hosted placement the guest paints on this desktop while its network is still -// pinned to the host at creation. So each act below pins the placement it was written for and reads -// the host's own record — the page row's placement and executionHostKey — instead of inferring -// routing from where a appeared. +// silently load a *different machine's* server. Remote-pane links are explicitly server-hosted, so +// the acts below read the host's own record instead of inferring routing from a . const PANE_PATH = '/remote-pane' const LINK_PATH = '/remote-link-target' @@ -103,56 +99,6 @@ async function readHostServerPlacedBrowserUrls( return response.result.tabs.filter((tab) => tab.type === 'browser').map((tab) => tab.url ?? '') } -type HostBrowserRow = { - executionHostKey: string | null - placementKind: string | null - url: string -} - -/** - * The host's rows for one URL, asked through the paired client's connection. - * - * Why through the client: an Electron peer advertises the client-host capability, so the host - * answers it with client-placed pages intact and with the placement and network pin it minted at - * creation. The host still authors every field; the client is only the transport. - */ -async function readHostBrowserRows( - page: Page, - environmentId: string, - worktreeId: string, - urlPrefix: string -): Promise { - return page.evaluate( - async ({ environmentId, urlPrefix, worktreeId }) => { - const response = await window.api.runtimeEnvironments.call({ - selector: environmentId, - method: 'session.tabs.list', - params: { worktree: `id:${worktreeId}` }, - timeoutMs: 15_000 - }) - if (!response.ok) { - throw new Error('host session tab inventory unavailable') - } - const { tabs } = response.result as { - tabs: { - type: string - url?: string - executionHostKey?: string - placement?: { kind: string } - }[] - } - return tabs - .filter((tab) => tab.type === 'browser' && (tab.url ?? '').startsWith(urlPrefix)) - .map((tab) => ({ - executionHostKey: tab.executionHostKey ?? null, - placementKind: tab.placement?.kind ?? null, - url: tab.url ?? '' - })) - }, - { environmentId, urlPrefix, worktreeId } - ) -} - /** Under server placement the client renders nothing itself, so any is a local fallback. */ async function readLocalBrowserViewUrls(page: Page): Promise { return page.evaluate(() => @@ -183,7 +129,6 @@ async function findMirroredPage( ): Promise<{ handleEnvironmentId: string | null pageId: string - placementKind: string | null } | null> { return page.evaluate( ({ url, worktreeId }) => { @@ -194,8 +139,7 @@ async function findMirroredPage( const handle = state?.remoteBrowserPageHandlesByPageId[browserPage.id] return { handleEnvironmentId: handle?.environmentId ?? null, - pageId: browserPage.id, - placementKind: handle?.placement?.kind ?? null + pageId: browserPage.id } } } @@ -277,7 +221,7 @@ async function openLinkFromRemotePaneContextMenu(page: Page): Promise { await openInOrca.click() } -test('opens a remote pane link on the pane runtime under either placement and refuses to fall back to the client', async ({ +test('opens a remote pane link on the pane runtime and refuses to fall back to the client', async ({ testRepoPath }, testInfo) => { test.setTimeout(300_000) @@ -286,8 +230,7 @@ test('opens a remote pane link on the pane runtime under either placement and re let client: PairedElectronClient | null = null try { - const hostRuntimeId = (await host.client.call('repo.add', { path: testRepoPath, kind: 'git' })) - ._meta.runtimeId + await host.client.call('repo.add', { path: testRepoPath, kind: 'git' }) client = await launchPairedElectronClient(host.offer, testInfo, 'Remote browser link routing') const page = client.page const environmentId = client.environmentId @@ -382,89 +325,65 @@ test('opens a remote pane link on the pane runtime under either placement and re ) .toBe(0) await focusMirroredPage(page, worktreeId, pane.pageId) - const linkLoadsBeforeClientAct = fixture.linkLoadCount() + const linkLoadsBeforeSecondAct = fixture.linkLoadCount() - // Act 2: client-hosted placement, the default. The page is hosted by this desktop, so the - // proof of correct routing is the host's record of it, not where it painted. + // Act 2 still stays server-hosted even when the generic client-hosted preference is enabled: + // the remote pane explicitly pins links to its owning runtime. await pinClientHostedPlacement(page, true) await openLinkFromRemotePaneContextMenu(page) - await expect - .poll( - async () => (await findMirroredPage(page, worktreeId, fixture.linkUrl))?.placementKind, - { - timeout: 60_000, - message: 'the link never became a client-hosted browser page on this desktop' - } - ) - .toBe('client') - // The host's own connection, unprojected: browser.tabList reads the page registry, which is - // where a client-hosted page lives. await expect .poll( async () => - (await readHostBrowserPageUrls(host.client, worktreeSelector)).filter((url) => + (await readHostServerPlacedBrowserUrls(host, worktreeId)).filter((url) => url.startsWith(fixture.linkUrl) ).length, - { timeout: 60_000, message: 'the link never became a browser page on the host runtime' } + { + timeout: 60_000, + message: 'the owner-pinned link did not stay server-hosted' + } ) .toBe(1) - const hostRows = await readHostBrowserRows(page, environmentId, worktreeId, fixture.linkUrl) - expect(hostRows).toHaveLength(1) - expect(hostRows[0]?.placementKind).toBe('client') - // The routing invariant, structurally: the host pinned this page's network to its own runtime - // when it created it, so the dev server was reached through the runtime and not through this - // machine — which CI cannot tell apart by watching the fixture, since both ends are loopback. - const executionHostKey = hostRows[0]?.executionHostKey - if (!executionHostKey) { - // Narrowed before parsing: an unpinned page would otherwise surface as a parse crash rather - // than as the missing network pin it is. - throw new Error('the host minted no network pin for the client-hosted page') - } - expect(parseBrowserNetworkExecutionHostKey(executionHostKey)).toMatchObject({ - runtimeId: hostRuntimeId - }) - expect(fixture.linkLoadCount()).toBeGreaterThan(linkLoadsBeforeClientAct) - // Hosted here, streamed from nowhere: this desktop holds the page and the runtime holds none. + expect(fixture.linkLoadCount()).toBeGreaterThan(linkLoadsBeforeSecondAct) await expect - .poll(() => readOwnedPageUrls(client!.app, fixture.linkUrl), { + .poll(() => readOwnedPageUrls(host.app, fixture.linkUrl), { timeout: 60_000, - message: 'the client-hosted guest never loaded the link on this desktop' + message: 'the server-hosted guest never loaded the link on the pane runtime' }) .toHaveLength(1) - expect(await readOwnedPageUrls(host.app, fixture.linkUrl)).toHaveLength(0) - await expect(page.getByTestId('remote-browser-pane')).toHaveCount(paneCountBeforeOpen) + expect(await readOwnedPageUrls(client!.app, fixture.linkUrl)).toHaveLength(0) + expect(await readLocalBrowserViewUrls(page)).toHaveLength(0) + await expect(page.getByTestId('remote-browser-pane')).toHaveCount(paneCountBeforeOpen + 1) - // The store drops the tab synchronously and only then fires browser.tabClose, so the mirror - // going empty proves nothing about the host or the guest. Settle both before act 3 baselines - // them, or act 3 reads this teardown landing mid-act as its own doing. + // The store drops the tab synchronously and only then fires browser.tabClose, so settle the + // mirror, host inventory, and host guest before act 3 reads them as its own baseline. await closeBrowserTabsExceptPane(page, worktreeId, fixture.paneUrl) await expect .poll(() => findMirroredPage(page, worktreeId, fixture.linkUrl), { timeout: 60_000, - message: 'the client kept the closed link tab' + message: 'the client kept the closed owner-pinned link tab' }) .toBeNull() await expect .poll( async () => - (await readHostBrowserPageUrls(host.client, worktreeSelector)).filter((url) => + (await readHostServerPlacedBrowserUrls(host, worktreeId)).filter((url) => url.startsWith(fixture.linkUrl) ).length, - { timeout: 60_000, message: 'the runtime kept the closed client-hosted page' } + { timeout: 60_000, message: 'the runtime kept the closed server-hosted page' } ) .toBe(0) await expect - .poll(() => readOwnedPageUrls(client!.app, fixture.linkUrl), { + .poll(() => readOwnedPageUrls(host.app, fixture.linkUrl), { timeout: 60_000, - message: 'the client-hosted guest outlived the tab that owned it' + message: 'the server-hosted guest outlived the tab that owned it' }) .toHaveLength(0) await focusMirroredPage(page, worktreeId, pane.pageId) - // Act 3: the user moves this workspace onto their own machine while the runtime's page is - // still on screen. Opening the link must fail in the pane, not load the runtime's dev server - // here — the client has no business serving a page for a workspace it does not run. + // Act 3: the user moves this workspace onto their own machine. Opening the remote pane link + // must fail in the pane, not load the runtime's dev server here — the client has no business + // serving a page for a workspace it does not run. await page.evaluate( ({ localHostId, worktreeId }) => { window.__store?.getState().setActiveWorktree(worktreeId, localHostId) From 61e010079f769f40cff39aef09f9788c13c3257d Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:00:24 -0700 Subject: [PATCH 068/398] New agent dashboard (#18222) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * more obvious toggle * more obvious toggle * feat(activity): redesign thread rows and add child agent filtering - Emphasize task title and last activity in row layout over metadata - Add child agent toggle; hide orchestration workers by default - Support collapsible groups and ungrouped view mode - Improve orchestration worker message handling to surface replies - Add sidebar search and filter controls for agent activity * periodic checkin * feat(activity): add "Clear completed" action and performance improvement - Add "Clear completed" action for activity threads with undo window; clears completed and interrupted rows from view, persists across restart - Virtualize activity thread list to render only viewport-bounded rows - Cache activity thread search text to prevent recomputation on every keystroke - Cache dashboard bucket counts per-worktree for selective invalidation on unrelated changes - Use useDeferredValue for activity search filtering to keep input responsive - Make compact mode the default display for activity threads - Add activity-cleared-at persisted state tracking (per-pane cutoff timestamps) * improve style * minor change * feat(activity): add persisted host and project filters to agents view Agents scope filters are deliberately separate from workspace-nav filters so a monitoring surface never inherits workspace context silently. Filters survive restarts and always display an active-filter chips row with hidden count, making filtering visible and reversible. * Graduate Agents view from experimental, refine activity handling - Agents Dashboard moves from experimental to standard feature with showAgentsSidebar setting controlling visibility - Add identity-checked cache eviction (dropPersisted IPC) to prevent newer runs from being evicted when UI clears older status, fixing clear-completed safety - Extract ActivityThreadHoverCardSummary and ActivityThreadListToolbar components for better organization and reusability - Implement mark-thread-read as separate action from select with clickable bell icon - Add hasActivityThreadWorkspace helper for checking workspace availability across hosts (SSH/runtime targets) - Preserve scope filter array identity during hydration for memo optimization - Track manually-unread turns in auto-ack to prevent re-acknowledgement - Clean up activity cleared-at cutoffs on pane retirement - Remove activity-thread-hover-card max-lines lint override (code refactored below threshold) * Refactor agent cache identity to use timing fields only - Simplify AgentStatusCacheIdentity: keep only paneKey, receivedAt, stateStartedAt - This fixes silent no-ops where renderer-enriched fields diverged from main's cache - Add worktree-jump-navigation for navigating activity to workspaces - Add manual mark-unread protection separate from auto-ack - Optimize activity owner resolution with per-build memoization - Optimize detected worktree lookup with indexed search * Remove sticky header, add scroll position persistence Replace the floating sticky header overlay with scroll position memory via a ref. This preserves the user's scroll location when switching between threads or remounting the agents list, improving UX without requiring React state. * Implement sticky group headers in activity thread list Keep group headers visible at the top while scrolling when threads are grouped. Headers stick to the viewport while their section is in view, then unstick as the next header approaches. * add blue flash * update settings appearnce * Extracted activity acknowledgement/clearance actions from the oversized UI slice. - Removed dead sidebar search/menu props and the unused search ref. - Removed the unnecessary sidebar visibility bitmask. - Replaced hardcoded sidebar toggle colors with design-system tokens. - Removed duplicate “mark all read / clear completed” controls in the sidebar. - Preserved manual-unread state correctly across pane retire, transfer, and drop. - Made clear-completed cutoffs monotonic so clock skew cannot resurrect old activity. - Fixed blank workspace names in hover cards with the existing fallback helper. - Added missing localization entries and stabilized hydrated filter array identity. - Updated misleading Agents setting copy to describe both sidebar surfaces. * add onboarding guide for the new agents panel * Add activity clearance tracking and synced agent view settings Agent view filters and presentation settings now sync across paired clients. Preserves per-pane activity clearance cutoffs in persistent state. Improves activity thread row accessibility with proper ARIA roles, and preserves terminal host ownership after pane teardown via retained terminal handle. * rm html * Graduate Agents from experimental and improve activity visibility - Migrate `showAgentsSidebar` setting from legacy experimental flags; default new profiles to the agents sidebar - Replace scoped-thread filtering with visible-thread filtering so bulk actions (mark all read, clear completed) only affect rendered rows - Rewrite child agent classification as a set of visible pane keys to fix orphan promotion and parent-cycle handling - Improve activity cleared-at cutoff lifecycle: preserve on row dismissal (pane may still be live) but clear on pane removal - Add pagehide flush for pending clear-completed evictions so quit/reload cannot replay cleared activity - Polish agents sidebar: unread count badge, expand button, onboarding intro for migrated/new users - Extract shared time-ago formatting to a library module - Fix scroll restoration to defer until content can contain the saved offset - Improve stable message hold for compact agent rows using state instead of refs - Add worktree filter-visibility check to distinguish collapsed-but-unfiltered from filtered-hidden * Graduate Agents from experimental and improve activity visibility - Remove the deprecated full-page Agents view; fix settings navigation fallback - Refactor bulk action bindings and separate mark-all-read from visible threads - Preserve sidebar collapse state across remounts; fix child-agent badge filtering - Add safety window for scroll-restore and improve worktree host-qualified filtering * Graduate Agents from experimental and add manual unread tracking - Move Agents sidebar from experimental settings to standard feature with intro flow - Add persistent manual unread turn tracking for activity feed - Consolidate workspace activation through activateAndRevealWorkspace dispatcher - Improve sidebar view toggle with radio semantics and arrow-key navigation * Graduate Agents sidebar and separate dashboard experiment The Agents tab now has its own `showAgentsSidebar` setting (defaults on) independent from the dashboard popout experiment. Activity unread counting is simplified to count all events uniformly without mode-specific filtering. Dashboard visibility is now controlled solely by `experimentalAgentDashboardPopout`, with its own UI in the Experimental settings pane. Migration path updated: only `experimentalActivity=true` graduates to the sidebar; the dashboard experiment remains separate. * Add agent-session tab support to activity tracking Build activity event contexts from structured agent-session tabs and worktree-attributed status entries. When activating a thread, try agent-session tab activation before falling back to terminal pane. * • The workspace sidebar tab is now a static Spaces label—no grouping-based “Projects” label or hidden width-reservation span. * Show unread count badge and prioritize attention-needing agent threads Activity group order now surfaces threads needing attention (blocked, waiting, interrupted) before working/done so they're never buried. The Agents tab shows an unread count badge while viewing Spaces, since the open Agents list already highlights unread rows. Also improves UX text ("Hide Agents" vs "Maybe later"), accessibility with proper ARIA labels, and handles edge cases: preserves read state for retained panes on SSH reconnect and handles deleted worktrees gracefully in navigation. * Batch agent-status evictions and optimize activity pane rebuilds - Add dropPersistedStatusEntries batch API; consolidate evictions into one persist - Implement fallback timeout in clear-completed for unseen toast callbacks - Project only activity-relevant tabs; memoize terminal tab derivations - Stabilize activity virtualizer key to prevent unnecessary item measurements * Remove unread count badge from Agents sidebar tab Simplify useActivityUnreadCount by removing the enabled parameter and conditional logic, as the badge is no longer displayed in the UI. * Deduplicate activity unread counts across source overlaps Live pane status is the primary source; retained and migration entries serve as fallback caches that may briefly overlap it during lifecycle transitions. Count each pane only once by tracking seen keys, prioritizing the live status as the canonical source. Also fix monitoring state display: it's a distinct agent state, not a tool-running row state, so exclude it from tool preview checks. * Update activity pane tests to remove unread badge assertions - Remove ActivityPaneVisibility type and readActivityPaneVisibility() helper - Update agentsSidebarButton selector to match badge-less state - Simplify assertions to check pane focus instead of visibility isolation - Remove test for unread badge acknowledgement flow * Fix activity pane workspace resolution and localization handling - Thread defaultHostId through activity operations for correct host resolution - Add language-aware caching for standalone terminal names with cache invalidation - Fix scroll restoration bounds calculation for tall viewports - Add focus management to sidebar radio group keyboard navigation - Refresh localized sidebar content on language changes - Preserve activity state across heartbeats to prevent history loss - Improve host-id strictness in worktree jump navigation * Preserve activity view when settings fetch fails A failed window.api.settings.get() leaves settings null, which was incorrectly treated as opt-out. Add the missing null check so the activity-view gate only applies when settings are available. Includes tests for this scenario and related edge cases in keyboard navigation, worktree jumping, and session state handling. --- docs/reference/wsl-probe-failure-semantics.md | 10 +- .../server-status-listener-fanout.test.ts | 138 +++++ src/main/agent-hooks/server/server-cleanup.ts | 46 ++ src/main/ipc/agent-hooks.test.ts | 72 +++ src/main/ipc/agent-status-row-teardown-ipc.ts | 53 +- .../applying-settings/ui-state-read.ts | 1 + .../applying-settings/ui-state-update.ts | 4 + .../normalize-loaded-global-settings.test.ts | 80 +++ .../normalize-loaded-global-settings.ts | 11 + .../workspace-session-snapshot-publication.ts | 26 +- .../pane-key-remapping.test.ts | 33 ++ .../restoring-sessions/pane-key-remapping.ts | 75 +++ .../workspace-pane-normalization.ts | 86 ++-- .../client-ui-pairing-local-fields.test.ts | 8 +- .../runtime/rpc/methods/client-ui-schemas.ts | 6 + src/preload/api/agent-status-api.ts | 5 + src/preload/api/agent-status-bridge.ts | 7 + .../app-shell/use-app-startup-hydration.ts | 10 + .../src/app-shell/use-persisted-ui-writer.ts | 6 +- .../activity/ActivityPrototypePage.test.ts | 44 +- ...ivityPrototypePage.thread-grouping.test.ts | 13 + .../activity/ActivityPrototypePage.tsx | 160 +++--- .../ActivityThreadOptionsMenu.test.tsx | 150 ++++++ .../activity/ActivityTitlebarControls.tsx | 2 +- ...vity-auto-mark-read-loop.react185.test.tsx | 2 +- .../activity/activity-clear-completed.test.ts | 362 +++++++++++++ .../activity/activity-clear-completed.ts | 206 ++++++++ .../activity/activity-event-build-cache.ts | 129 +++++ ...tivity-event-builder-agent-context.test.ts | 165 ++++++ .../activity-event-builder-context.ts | 182 +++++++ .../activity-event-builder-sources.ts | 105 ++++ ...vity-event-builder.bounded-history.test.ts | 122 +++++ ...ivity-event-builder.host-ownership.test.ts | 206 ++++++++ ...ivity-event-builder.identity-reuse.test.ts | 225 ++++++++ .../activity/activity-event-builder.ts | 306 ++++------- .../components/activity/activity-event-cap.ts | 3 +- .../activity/activity-pane-events.ts | 108 ++++ .../activity-scope-filter-controls.tsx | 170 ++++++ .../activity/activity-scope-filter.test.ts | 130 +++++ .../activity/activity-scope-filter.ts | 80 +++ .../activity/activity-standalone-worktree.ts | 67 +++ .../activity/activity-tab-projection.ts | 66 +++ .../activity/activity-thread-actions.test.ts | 162 ++++++ .../activity/activity-thread-actions.ts | 101 +++- .../activity/activity-thread-builder.ts | 89 +++- .../activity-thread-child-agent.test.ts | 171 +++++++ .../activity/activity-thread-child-agent.ts | 90 ++++ .../activity-thread-collapse-context.ts | 13 + .../activity/activity-thread-controls.tsx | 130 +++-- ...ivity-thread-grouping.search-cache.test.ts | 53 ++ .../activity/activity-thread-grouping.ts | 41 +- .../activity-thread-hover-card-summary.tsx | 230 +++++++++ .../activity-thread-hover-card.test.tsx | 238 +++++++++ .../activity/activity-thread-hover-card.tsx | 407 +++++++++++++++ ...vity-thread-list-pane-collapsible.test.tsx | 292 +++++++++++ .../activity/activity-thread-list-pane.tsx | 484 ++++++++++++------ ...y-thread-list-pane.virtualization.test.tsx | 237 +++++++++ .../activity-thread-list-resize-handle.tsx | 37 ++ .../activity/activity-thread-list-toolbar.tsx | 227 ++++++++ .../activity/activity-thread-options-menu.tsx | 291 +++++++++++ .../activity-thread-presentation.test.ts | 97 ++++ .../activity/activity-thread-presentation.ts | 60 ++- .../activity/activity-thread-row.tsx | 318 ++++++------ .../activity/activity-thread-types.ts | 2 +- .../activity-thread-virtual-items.test.ts | 82 +++ .../activity/activity-thread-virtual-items.ts | 74 +++ .../activity/activity-thread-virtual-row.tsx | 71 +++ .../activity/dev-activity-fixture.test.ts | 39 ++ .../activity/dev-activity-fixture.ts | 127 +++++ .../event-time-clock-refresh.test.tsx | 54 ++ ...e-activity-thread-action-bindings.test.tsx | 113 ++++ .../use-activity-thread-action-bindings.ts | 71 +++ .../activity/use-agent-pane-threads.ts | 264 ++++++++++ .../activity/useActivityUnreadCount.test.ts | 69 ++- .../activity/useActivityUnreadCount.ts | 130 ++--- .../dashboard/agent-row-lineage-model.ts | 17 +- ...uild-dashboard-bucket-counts.cache.test.ts | 190 +++++++ .../build-dashboard-bucket-counts.ts | 86 ++-- ...uild-dashboard-snapshot.rows-cache.test.ts | 142 +++++ .../dashboard/build-dashboard-snapshot.ts | 75 ++- .../dashboard/useAgentBucketCounts.test.tsx | 11 +- .../dashboard/useAgentBucketCounts.ts | 16 +- .../dashboard/useDashboardPopoutBridge.ts | 10 +- .../dashboard/useLiveDashboardSnapshot.ts | 9 +- .../dashboard/worktree-agent-rows-cache.ts | 181 +++++++ .../AppearanceWindowSidebarSection.tsx | 2 - .../settings/ExperimentalPane.test.tsx | 19 +- .../components/settings/ExperimentalPane.tsx | 83 ++- .../settings/appearance-sidebar-search.ts | 21 + .../settings/experimental-search.ts | 61 +-- .../sidebar/AgentDashboardSidebarEntry.tsx | 5 +- .../src/components/sidebar/Sidebar.test.tsx | 69 ++- .../components/sidebar/SidebarAgentsList.tsx | 144 ++++++ .../components/sidebar/SidebarHeader.test.tsx | 339 +++++++++++- .../src/components/sidebar/SidebarHeader.tsx | 263 +++++++--- .../components/sidebar/SidebarNav.test.tsx | 35 +- .../src/components/sidebar/SidebarNav.tsx | 71 +-- .../SidebarRepositoryFilterSection.tsx | 14 +- .../sidebar/SidebarWorkspaceOptionsMenu.tsx | 223 +------- .../sidebar/agents-sidebar-visibility.test.ts | 15 + .../sidebar/agents-sidebar-visibility.ts | 11 + src/renderer/src/components/sidebar/index.tsx | 209 +++++++- .../sidebar/sidebar-count-badge.tsx | 24 + .../sidebar/sidebar-header-actions.tsx | 194 +++++++ .../sidebar/sidebar-view-toggle.test.tsx | 145 ++++++ .../sidebar/sidebar-view-toggle.tsx | 106 ++++ ...se-workspace-reveal-body-redirect.test.tsx | 77 +++ .../use-workspace-reveal-body-redirect.ts | 43 ++ .../components/sidebar/visible-worktrees.ts | 69 +-- .../sidebar/workspace-options-menu-items.tsx | 240 +++++++++ .../sidebar/worktree-agent-row-selectors.ts | 12 + ...-compact-agent-row.stable-message.test.tsx | 112 ++++ .../worktree-card-compact-agent-row.tsx | 49 +- .../worktree-filter-visibility.test.ts | 79 +++ .../sidebar/worktree-filter-visibility.ts | 43 ++ .../terminal-pane/stale-agent-row.ts | 2 +- .../use-terminal-pane-close-actions.ts | 2 +- .../src/components/ui/dropdown-menu.tsx | 2 +- src/renderer/src/components/ui/popover.tsx | 17 +- .../useAutoAckViewedAgent.clock-skew.test.ts | 10 + .../src/hooks/useAutoAckViewedAgent.test.ts | 45 ++ .../src/hooks/useAutoAckViewedAgent.ts | 66 ++- src/renderer/src/i18n/locales/en.json | 70 ++- src/renderer/src/i18n/locales/es.json | 29 +- src/renderer/src/i18n/locales/ja.json | 29 +- src/renderer/src/i18n/locales/ko.json | 52 +- src/renderer/src/i18n/locales/zh.json | 52 +- .../src/lib/activity-thread-display.test.ts | 79 +++ .../src/lib/activity-thread-display.ts | 64 ++- src/renderer/src/lib/short-time-ago.ts | 16 + src/renderer/src/lib/worktree-activation.ts | 43 +- .../src/lib/worktree-jump-navigation.test.ts | 151 ++++++ .../src/lib/worktree-jump-navigation.ts | 95 ++++ .../src/lib/worktree-runtime-owner-index.ts | 9 +- .../agent-status-primitives.ts | 8 +- .../runtime/web-session-tabs-sync/state.ts | 2 + .../store/slices/activity-cleared-at.test.ts | 169 ++++++ .../store/slices/agent-pane-authority.test.ts | 14 + .../slices/agent-status-ack-cleanup.test.ts | 61 +++ .../slices/agent-status-authority-actions.ts | 10 + .../slices/agent-status-cleanup-actions.ts | 52 +- .../src/store/slices/agent-status-contract.ts | 10 + .../store/slices/agent-status-drop-actions.ts | 58 ++- .../store/slices/agent-status-drop-reducer.ts | 35 +- .../slices/agent-status-pane-keyed-records.ts | 12 + .../agent-status-provider-session-actions.ts | 10 +- .../agent-status-provider-session.test.ts | 40 ++ .../slices/agent-status-retention-actions.ts | 30 ++ .../slices/agent-status-slice-contract.ts | 6 +- .../agent-status-worktree-drop-actions.ts | 52 +- .../persisted-ui-write-baseline.test.ts | 24 + .../slices/persisted-ui-write-baseline.ts | 13 +- .../retired-terminal-tab-state-sweep.ts | 8 +- .../store/slices/runtime-status-recheck.ts | 5 +- .../slices/ui-hydration-view-layout.test.ts | 35 +- ...ui-hydration-workspace-preferences.test.ts | 118 +++++ .../store/slices/ui-page-navigation.test.ts | 29 ++ .../slices/ui/ui-slice-activity-actions.ts | 155 ++++++ .../store/slices/ui/ui-slice-agent-actions.ts | 105 +--- .../store/slices/ui/ui-slice-contract-core.ts | 6 + .../ui/ui-slice-contract-preferences.ts | 12 + .../slices/ui/ui-slice-hydration-actions.ts | 94 ++-- .../ui/ui-slice-hydration-sanitizers.ts | 40 +- .../slices/ui/ui-slice-hydration-values.ts | 22 + .../slices/ui/ui-slice-preference-actions.ts | 25 + .../slices/ui/ui-slice-settings-actions.ts | 5 +- .../slices/ui/ui-slice-surface-actions.ts | 4 + .../store/slices/ui/ui-slice-view-actions.ts | 6 +- .../listing/detected-worktree-meta.ts | 30 +- .../teardown/worktree-purge-state.ts | 2 + .../web/preload-api/web-agent-status-api.ts | 2 + .../web-preference-normalization.ts | 8 +- .../src/web/web-preload-api-ui.test.ts | 16 +- src/shared/agent-status-ipc-payload.ts | 10 + src/shared/agent-status-types.ts | 1 + src/shared/agents-sidebar-visibility.ts | 12 + src/shared/constants.ts | 6 + src/shared/default-global-settings.ts | 1 + src/shared/global-settings-types.ts | 6 + src/shared/pairing-local-ui-fields.test.ts | 8 +- src/shared/pairing-local-ui-fields.ts | 9 +- src/shared/persisted-ui-state-types.ts | 12 + src/shared/telemetry-property-schemas.ts | 1 + .../e2e/activity-agent-pane-isolation.spec.ts | 129 +---- 184 files changed, 12459 insertions(+), 1960 deletions(-) create mode 100644 src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts create mode 100644 src/main/persistence/restoring-sessions/pane-key-remapping.test.ts create mode 100644 src/main/persistence/restoring-sessions/pane-key-remapping.ts create mode 100644 src/renderer/src/components/activity/activity-clear-completed.test.ts create mode 100644 src/renderer/src/components/activity/activity-clear-completed.ts create mode 100644 src/renderer/src/components/activity/activity-event-build-cache.ts create mode 100644 src/renderer/src/components/activity/activity-event-builder-agent-context.test.ts create mode 100644 src/renderer/src/components/activity/activity-event-builder-context.ts create mode 100644 src/renderer/src/components/activity/activity-event-builder-sources.ts create mode 100644 src/renderer/src/components/activity/activity-event-builder.bounded-history.test.ts create mode 100644 src/renderer/src/components/activity/activity-event-builder.host-ownership.test.ts create mode 100644 src/renderer/src/components/activity/activity-event-builder.identity-reuse.test.ts create mode 100644 src/renderer/src/components/activity/activity-pane-events.ts create mode 100644 src/renderer/src/components/activity/activity-scope-filter-controls.tsx create mode 100644 src/renderer/src/components/activity/activity-scope-filter.test.ts create mode 100644 src/renderer/src/components/activity/activity-scope-filter.ts create mode 100644 src/renderer/src/components/activity/activity-standalone-worktree.ts create mode 100644 src/renderer/src/components/activity/activity-tab-projection.ts create mode 100644 src/renderer/src/components/activity/activity-thread-actions.test.ts create mode 100644 src/renderer/src/components/activity/activity-thread-child-agent.test.ts create mode 100644 src/renderer/src/components/activity/activity-thread-child-agent.ts create mode 100644 src/renderer/src/components/activity/activity-thread-collapse-context.ts create mode 100644 src/renderer/src/components/activity/activity-thread-grouping.search-cache.test.ts create mode 100644 src/renderer/src/components/activity/activity-thread-hover-card-summary.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-hover-card.test.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-hover-card.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-list-pane-collapsible.test.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-list-pane.virtualization.test.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-list-resize-handle.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-list-toolbar.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-options-menu.tsx create mode 100644 src/renderer/src/components/activity/activity-thread-presentation.test.ts create mode 100644 src/renderer/src/components/activity/activity-thread-virtual-items.test.ts create mode 100644 src/renderer/src/components/activity/activity-thread-virtual-items.ts create mode 100644 src/renderer/src/components/activity/activity-thread-virtual-row.tsx create mode 100644 src/renderer/src/components/activity/dev-activity-fixture.test.ts create mode 100644 src/renderer/src/components/activity/dev-activity-fixture.ts create mode 100644 src/renderer/src/components/activity/event-time-clock-refresh.test.tsx create mode 100644 src/renderer/src/components/activity/use-activity-thread-action-bindings.test.tsx create mode 100644 src/renderer/src/components/activity/use-activity-thread-action-bindings.ts create mode 100644 src/renderer/src/components/activity/use-agent-pane-threads.ts create mode 100644 src/renderer/src/components/dashboard/build-dashboard-bucket-counts.cache.test.ts create mode 100644 src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts create mode 100644 src/renderer/src/components/dashboard/worktree-agent-rows-cache.ts create mode 100644 src/renderer/src/components/sidebar/SidebarAgentsList.tsx create mode 100644 src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts create mode 100644 src/renderer/src/components/sidebar/agents-sidebar-visibility.ts create mode 100644 src/renderer/src/components/sidebar/sidebar-count-badge.tsx create mode 100644 src/renderer/src/components/sidebar/sidebar-header-actions.tsx create mode 100644 src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx create mode 100644 src/renderer/src/components/sidebar/sidebar-view-toggle.tsx create mode 100644 src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.test.tsx create mode 100644 src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.ts create mode 100644 src/renderer/src/components/sidebar/workspace-options-menu-items.tsx create mode 100644 src/renderer/src/components/sidebar/worktree-card-compact-agent-row.stable-message.test.tsx create mode 100644 src/renderer/src/components/sidebar/worktree-filter-visibility.test.ts create mode 100644 src/renderer/src/components/sidebar/worktree-filter-visibility.ts create mode 100644 src/renderer/src/lib/short-time-ago.ts create mode 100644 src/renderer/src/lib/worktree-jump-navigation.test.ts create mode 100644 src/renderer/src/lib/worktree-jump-navigation.ts create mode 100644 src/renderer/src/store/slices/activity-cleared-at.test.ts create mode 100644 src/renderer/src/store/slices/ui/ui-slice-activity-actions.ts create mode 100644 src/shared/agents-sidebar-visibility.ts diff --git a/docs/reference/wsl-probe-failure-semantics.md b/docs/reference/wsl-probe-failure-semantics.md index cadcbf53364..dcaa67a5d4d 100644 --- a/docs/reference/wsl-probe-failure-semantics.md +++ b/docs/reference/wsl-probe-failure-semantics.md @@ -35,11 +35,11 @@ silent, and indistinguishable from the real thing. Three instances so far: -| Where | What the user saw | Status | -| --- | --- | --- | -| Preflight CLI probes | Caching the result would have pinned "git not installed" until relaunch | Bounded entry ([#17350](https://github.com/stablyai/orca/pull/17350)) | -| `glab auth status` fallback into WSL | Idle VM woken repeatedly for users who never touch GitLab | Open ([#8941](https://github.com/stablyai/orca/issues/8941)) | -| `listRunningWslDistrosAsync` | Fails closed to `[]` with no last-known-good, polled every 2s — a persistently broken `wsl.exe` makes every WSL session vanish app-wide | Open (PR #17072 review) | +| Where | What the user saw | Status | +| ------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------- | +| Preflight CLI probes | Caching the result would have pinned "git not installed" until relaunch | Bounded entry ([#17350](https://github.com/stablyai/orca/pull/17350)) | +| `glab auth status` fallback into WSL | Idle VM woken repeatedly for users who never touch GitLab | Open ([#8941](https://github.com/stablyai/orca/issues/8941)) | +| `listRunningWslDistrosAsync` | Fails closed to `[]` with no last-known-good, polled every 2s — a persistently broken `wsl.exe` makes every WSL session vanish app-wide | Open (PR #17072 review) | ## What to do instead diff --git a/src/main/agent-hooks/server-status-listener-fanout.test.ts b/src/main/agent-hooks/server-status-listener-fanout.test.ts index 9f977672239..b6633dbf6f1 100644 --- a/src/main/agent-hooks/server-status-listener-fanout.test.ts +++ b/src/main/agent-hooks/server-status-listener-fanout.test.ts @@ -205,6 +205,144 @@ describe('AgentHookServer listener replay', () => { expect(listener).toHaveBeenNthCalledWith(4, []) }) + it('evicts only the matching persisted status identity', () => { + const server = new AgentHookServer() + server.ingestRemote( + { + paneKey: PANE, + tabId: 'tab-1', + worktreeId: 'wt-1', + providerSession: { key: 'session_id', id: 'resume-me' }, + payload: { state: 'done', prompt: 'old run', agentType: 'claude' } + }, + 'conn-1' + ) + const old = server.getStatusSnapshot()[0] + expect(old).toBeDefined() + + server.ingestRemote( + { + paneKey: PANE, + tabId: 'tab-1', + worktreeId: 'wt-1', + payload: { state: 'working', prompt: 'new run', agentType: 'claude' } + }, + 'conn-1' + ) + server.dropPersistedStatusEntry({ + paneKey: old!.paneKey, + receivedAt: old!.receivedAt, + stateStartedAt: old!.stateStartedAt + }) + + expect(server.getStatusSnapshot()[0]).toMatchObject({ state: 'working', prompt: 'new run' }) + + // A matching eviction follows ordinary dismissal semantics, including + // preserving a resumable provider session for the still-live TUI. + const resumed = new AgentHookServer() + resumed.ingestRemote( + { + paneKey: PANE, + tabId: 'tab-1', + worktreeId: 'wt-1', + providerSession: { key: 'session_id', id: 'resume-me' }, + payload: { state: 'done', prompt: 'old run', agentType: 'claude' } + }, + 'conn-1' + ) + const resumedIdentity = resumed.getStatusSnapshot()[0]! + expect( + resumed.dropPersistedStatusEntry({ + paneKey: resumedIdentity.paneKey, + receivedAt: resumedIdentity.receivedAt, + stateStartedAt: resumedIdentity.stateStartedAt + }) + ).toBe(true) + expect(resumed.getStatusSnapshot()[0]).toMatchObject({ + providerSessionOnly: true, + providerSession: { id: 'resume-me' } + }) + }) + + it('evicts when the renderer identity was stamped after receipt but pins the same turn', () => { + // Runtime-sync and recovery entries stamp updatedAt with Date.now()/capturedAt, which is + // at or after main's receivedAt; the eviction must still land for those rows. + const server = new AgentHookServer() + server.ingestRemote( + { + paneKey: PANE, + tabId: 'tab-1', + worktreeId: 'wt-1', + payload: { state: 'done', prompt: 'run', agentType: 'claude' } + }, + 'conn-1' + ) + const entry = server.getStatusSnapshot()[0]! + expect( + server.dropPersistedStatusEntry({ + paneKey: entry.paneKey, + receivedAt: entry.receivedAt + 5_000, + stateStartedAt: entry.stateStartedAt + }) + ).toBe(true) + + // A different turn never matches, whatever the receivedAt relationship. + const other = new AgentHookServer() + other.ingestRemote( + { + paneKey: PANE, + tabId: 'tab-1', + worktreeId: 'wt-1', + payload: { state: 'done', prompt: 'run', agentType: 'claude' } + }, + 'conn-1' + ) + const otherEntry = other.getStatusSnapshot()[0]! + expect( + other.dropPersistedStatusEntry({ + paneKey: otherEntry.paneKey, + receivedAt: otherEntry.receivedAt + 5_000, + stateStartedAt: otherEntry.stateStartedAt + 1 + }) + ).toBe(false) + }) + + it('evicts a batch of persisted identities with one status-change notification', () => { + const server = new AgentHookServer() + const otherPane = makePaneKey('tab-2', '22222222-2222-4222-8222-222222222222') + for (const paneKey of [PANE, otherPane]) { + server.ingestRemote( + { + paneKey, + tabId: paneKey.split(':')[0]!, + worktreeId: 'wt-1', + payload: { state: 'done', prompt: 'run', agentType: 'claude' } + }, + 'conn-1' + ) + } + const listener = vi.fn() + server.subscribeStatusChanges(listener) + const dropped: string[] = [] + server.subscribeStatusDrop((paneKey) => dropped.push(paneKey)) + const identities = server.getStatusSnapshot().map((entry) => ({ + paneKey: entry.paneKey, + receivedAt: entry.receivedAt, + stateStartedAt: entry.stateStartedAt + })) + + const evicted = server.dropPersistedStatusEntries([ + ...identities, + // A stale identity never matches and never blocks the rest of the batch. + { ...identities[0]!, stateStartedAt: identities[0]!.stateStartedAt + 1 } + ]) + + expect(evicted.sort()).toEqual([PANE, otherPane].sort()) + expect(dropped.sort()).toEqual([PANE, otherPane].sort()) + expect(listener).toHaveBeenCalledTimes(1) + expect(server.getStatusSnapshot()).toEqual([]) + }) + it('notifies pane-status-clear listener when pane teardown evicts a cached status', () => { const server = new AgentHookServer() const listener = vi.fn() diff --git a/src/main/agent-hooks/server/server-cleanup.ts b/src/main/agent-hooks/server/server-cleanup.ts index 04acc058dea..4fcd0b5e58e 100644 --- a/src/main/agent-hooks/server/server-cleanup.ts +++ b/src/main/agent-hooks/server/server-cleanup.ts @@ -1,4 +1,5 @@ import { paneHasStateClaims } from '../../../shared/agent-hook-listener/listener-state' +import type { AgentStatusCacheIdentity } from '../../../shared/agent-status-types' import type { EnrichedAgentHookEventPayload } from './server-types' import { AgentHookServerAuthorityFences } from './server-authority-fences' @@ -35,6 +36,51 @@ export abstract class AgentHookServerCleanup extends AgentHookServerAuthorityFen this.emitStatusDropped(deleted.paneKey) } + /** Evict a UI-cleared status only if no newer status has replaced it. */ + dropPersistedStatusEntry(identity: AgentStatusCacheIdentity): boolean { + return this.dropPersistedStatusEntries([identity]).length > 0 + } + + /** Batch form: one persist and one listener notification for the whole set. Returns the + * pane keys that were actually evicted. */ + dropPersistedStatusEntries(identities: readonly AgentStatusCacheIdentity[]): string[] { + const evicted: string[] = [] + for (const identity of identities) { + const resolvedPaneKey = this.resolvePaneKeyAlias(identity.paneKey) + const existing = this.state.lastStatusByPaneKey.get(resolvedPaneKey) as + | EnrichedAgentHookEventPayload + | undefined + // Why: stateStartedAt pins the turn; the renderer's updatedAt is stamped at or after this + // receivedAt (runtime-sync and recovery paths use Date.now()/capturedAt), so a strictly + // newer cached event is the only replacement worth protecting. + if ( + !existing || + existing.stateStartedAt !== identity.stateStartedAt || + existing.receivedAt > identity.receivedAt + ) { + continue + } + const deleted = this.deleteStatusEntry(resolvedPaneKey, { preserveAuthority: true }) + if (!deleted) { + continue + } + const retained = this.toRetainedProviderSessionRow(deleted) + if (retained) { + this.state.lastStatusByPaneKey.set(deleted.paneKey, retained) + } + evicted.push(deleted.paneKey) + } + if (evicted.length === 0) { + return evicted + } + this.scheduleStatusPersist() + this.notifyStatusChangeListeners() + for (const paneKey of evicted) { + this.emitStatusDropped(paneKey) + } + return evicted + } + /** Retire panes whose owning process is certifiably dead. * * The ordinary teardown already does this: every attributable PTY exit reaches diff --git a/src/main/ipc/agent-hooks.test.ts b/src/main/ipc/agent-hooks.test.ts index e9e7a728e83..cd6c21526d8 100644 --- a/src/main/ipc/agent-hooks.test.ts +++ b/src/main/ipc/agent-hooks.test.ts @@ -8,6 +8,8 @@ import { makePaneKey } from '../../shared/stable-pane-id' // evicts the entry. const dropStatusEntry = vi.fn() +const dropPersistedStatusEntry = vi.fn() +const dropPersistedStatusEntries = vi.fn(() => [] as string[]) const dropStatusEntriesByTabPrefix = vi.fn() const retirePaneAuthority = vi.fn() const transferPaneAuthority = vi.fn() @@ -44,6 +46,8 @@ vi.mock('../agent-hooks/server', async () => { ...actual, agentHookServer: { dropStatusEntry, + dropPersistedStatusEntry, + dropPersistedStatusEntries, dropStatusEntriesByTabPrefix, retirePaneAuthority, transferPaneAuthority, @@ -105,6 +109,9 @@ vi.mock('../kimi/hook-service', () => ({ beforeEach(() => { dropStatusEntry.mockReset() + dropPersistedStatusEntry.mockReset() + dropPersistedStatusEntries.mockReset() + dropPersistedStatusEntries.mockReturnValue([]) dropStatusEntriesByTabPrefix.mockReset() retirePaneAuthority.mockReset() transferPaneAuthority.mockReset() @@ -279,6 +286,71 @@ describe('agentStatus:drop IPC', () => { }) }) +describe('agentStatus:dropPersisted IPC', () => { + it('forwards a validated cache identity without clearing pane state', async () => { + const { registerAgentHookHandlers } = await import('./agent-hooks') + registerAgentHookHandlers() + + const handler = onHandlers.get('agentStatus:dropPersisted') + expect(handler).toBeDefined() + const identity = { + paneKey: PANE_KEY, + receivedAt: 2_000, + stateStartedAt: 1_000 + } + handler!({}, identity) + expect(dropPersistedStatusEntry).toHaveBeenCalledWith(identity) + expect(dropStatusEntry).not.toHaveBeenCalled() + }) + + it('forwards a batch, keeping only valid identities, and clears migration state per evicted pane', async () => { + const { registerAgentHookHandlers } = await import('./agent-hooks') + registerAgentHookHandlers() + + const handler = onHandlers.get('agentStatus:dropPersistedBatch') + expect(handler).toBeDefined() + const good = { paneKey: PANE_KEY, receivedAt: 2_000, stateStartedAt: 1_000 } + const alsoGood = { paneKey: CHILD_PANE_KEY, receivedAt: 3_000, stateStartedAt: 2_500 } + dropPersistedStatusEntries.mockReturnValue([PANE_KEY]) + handler!({}, [good, { paneKey: 'not-a-pane-key', receivedAt: 1, stateStartedAt: 1 }, alsoGood]) + expect(dropPersistedStatusEntries).toHaveBeenCalledWith([good, alsoGood]) + expect(clearMigrationUnsupportedPtysForPaneKey).toHaveBeenCalledWith(PANE_KEY) + expect(clearMigrationUnsupportedPtysForPaneKey).not.toHaveBeenCalledWith(CHILD_PANE_KEY) + expect(dropPersistedStatusEntry).not.toHaveBeenCalled() + }) + + it('ignores a batch that is not an array or is empty after validation', async () => { + const { registerAgentHookHandlers } = await import('./agent-hooks') + registerAgentHookHandlers() + + const handler = onHandlers.get('agentStatus:dropPersistedBatch')! + for (const value of [null, {}, 'x', [], [{ paneKey: PANE_KEY }]]) { + expect(() => handler({}, value)).not.toThrow() + } + expect(dropPersistedStatusEntries).not.toHaveBeenCalled() + }) + + it('rejects malformed cache identities', async () => { + const { registerAgentHookHandlers } = await import('./agent-hooks') + registerAgentHookHandlers() + + const handler = onHandlers.get('agentStatus:dropPersisted')! + for (const value of [ + null, + undefined, + {}, + { paneKey: PANE_KEY }, + { paneKey: PANE_KEY, receivedAt: Number.NaN, stateStartedAt: 1 }, + { paneKey: PANE_KEY, receivedAt: 2, stateStartedAt: Number.POSITIVE_INFINITY }, + { paneKey: 'not-a-pane-key', receivedAt: 2, stateStartedAt: 1 }, + { paneKey: PANE_KEY, receivedAt: '2', stateStartedAt: 1 } + ]) { + expect(() => handler({}, value)).not.toThrow() + } + expect(dropPersistedStatusEntry).not.toHaveBeenCalled() + }) +}) + describe('agentStatus:dropByTabPrefix IPC', () => { it('forwards valid tab ids to tab-prefix cache eviction', async () => { const { registerAgentHookHandlers } = await import('./agent-hooks') diff --git a/src/main/ipc/agent-status-row-teardown-ipc.ts b/src/main/ipc/agent-status-row-teardown-ipc.ts index e33e1c93789..020cfa7259d 100644 --- a/src/main/ipc/agent-status-row-teardown-ipc.ts +++ b/src/main/ipc/agent-status-row-teardown-ipc.ts @@ -1,5 +1,6 @@ import { ipcMain } from 'electron' import { agentHookServer, isValidPaneKey } from '../agent-hooks/server' +import type { AgentStatusCacheIdentity } from '../../shared/agent-status-types' import { clearMigrationUnsupportedPtysByTabPrefix, clearMigrationUnsupportedPtysForPaneKey @@ -7,7 +8,7 @@ import { import { isValidAgentStatusDropTabId } from './agent-status-ipc-boundary' /** - * The three renderer-initiated ways a status row goes away. All fire-and-forget + * The renderer-initiated ways a status row goes away. All fire-and-forget * (`ipcRenderer.send` → `ipcMain.on`), so none round-trips a response; removing the * listeners first keeps re-registration safe. * @@ -15,8 +16,13 @@ import { isValidAgentStatusDropTabId } from './agent-status-ipc-boundary' * still be alive; a confirmed process exit must take them too, or a surviving Claude latch resolves * the pane's next event straight back to `working`. */ +// Why a cap: the renderer sends one batch per Clear-completed click, bounded by visible rows. +const MAX_DROP_PERSISTED_BATCH = 5_000 + export function registerAgentStatusRowTeardownIpcHandlers(): void { ipcMain.removeAllListeners('agentStatus:drop') + ipcMain.removeAllListeners('agentStatus:dropPersisted') + ipcMain.removeAllListeners('agentStatus:dropPersistedBatch') ipcMain.removeAllListeners('agentStatus:reconcileEndedProcess') ipcMain.removeAllListeners('agentStatus:dropByTabPrefix') @@ -36,6 +42,36 @@ export function registerAgentStatusRowTeardownIpcHandlers(): void { } }) + ipcMain.on('agentStatus:dropPersisted', (_event, request: unknown) => { + if (!isValidAgentStatusCacheIdentity(request)) { + return + } + try { + if (agentHookServer.dropPersistedStatusEntry(request)) { + clearMigrationUnsupportedPtysForPaneKey(request.paneKey) + } + } catch (err) { + console.warn('[agent-hooks] dropPersistedStatusEntry failed:', err) + } + }) + + ipcMain.on('agentStatus:dropPersistedBatch', (_event, request: unknown) => { + if (!Array.isArray(request) || request.length > MAX_DROP_PERSISTED_BATCH) { + return + } + const identities = request.filter(isValidAgentStatusCacheIdentity) + if (identities.length === 0) { + return + } + try { + for (const paneKey of agentHookServer.dropPersistedStatusEntries(identities)) { + clearMigrationUnsupportedPtysForPaneKey(paneKey) + } + } catch (err) { + console.warn('[agent-hooks] dropPersistedStatusEntries failed:', err) + } + }) + ipcMain.on('agentStatus:reconcileEndedProcess', (_event, paneKey: unknown) => { if (typeof paneKey !== 'string' || !isValidPaneKey(paneKey)) { return @@ -67,3 +103,18 @@ export function registerAgentStatusRowTeardownIpcHandlers(): void { } }) } + +function isValidAgentStatusCacheIdentity(value: unknown): value is AgentStatusCacheIdentity { + if (typeof value !== 'object' || value === null || Array.isArray(value)) { + return false + } + const request = value as Record + return ( + typeof request.paneKey === 'string' && + isValidPaneKey(request.paneKey) && + typeof request.receivedAt === 'number' && + Number.isFinite(request.receivedAt) && + typeof request.stateStartedAt === 'number' && + Number.isFinite(request.stateStartedAt) + ) +} diff --git a/src/main/persistence/applying-settings/ui-state-read.ts b/src/main/persistence/applying-settings/ui-state-read.ts index 7a249934260..a640f971f89 100644 --- a/src/main/persistence/applying-settings/ui-state-read.ts +++ b/src/main/persistence/applying-settings/ui-state-read.ts @@ -62,6 +62,7 @@ export function getPersistedUI( markdownTocPanelWidth: clampMarkdownTocPanelWidth(state.ui?.markdownTocPanelWidth), combinedDiffFileTreeWidth: clampCombinedDiffFileTreeWidth(state.ui?.combinedDiffFileTreeWidth), visibleWorkspaceHostIds: normalizeVisibleExecutionHostIds(state.ui?.visibleWorkspaceHostIds), + agentsVisibleHostIds: normalizeVisibleExecutionHostIds(state.ui?.agentsVisibleHostIds), workspaceHostOrder: normalizeExecutionHostOrder(state.ui?.workspaceHostOrder), manualRepoOrder: normalizeManualRepoOrder(state.ui?.manualRepoOrder), browserDefaultZoomLevel: normalizeBrowserPageZoomLevel(state.ui?.browserDefaultZoomLevel), diff --git a/src/main/persistence/applying-settings/ui-state-update.ts b/src/main/persistence/applying-settings/ui-state-update.ts index b93ddfb1296..db06a2738fd 100644 --- a/src/main/persistence/applying-settings/ui-state-update.ts +++ b/src/main/persistence/applying-settings/ui-state-update.ts @@ -152,6 +152,10 @@ export function updatePersistedUI( sanitizedUpdates.visibleWorkspaceHostIds !== undefined ? normalizeVisibleExecutionHostIds(sanitizedUpdates.visibleWorkspaceHostIds) : normalizeVisibleExecutionHostIds(operations.state.ui?.visibleWorkspaceHostIds), + agentsVisibleHostIds: + sanitizedUpdates.agentsVisibleHostIds !== undefined + ? normalizeVisibleExecutionHostIds(sanitizedUpdates.agentsVisibleHostIds) + : normalizeVisibleExecutionHostIds(operations.state.ui?.agentsVisibleHostIds), workspaceHostOrder: sanitizedUpdates.workspaceHostOrder !== undefined ? normalizeExecutionHostOrder(sanitizedUpdates.workspaceHostOrder) diff --git a/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts b/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts new file mode 100644 index 00000000000..6d1040f09ae --- /dev/null +++ b/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts @@ -0,0 +1,80 @@ +import { homedir } from 'node:os' +import { describe, expect, it } from 'vitest' +import { getDefaultPersistedState } from '../../../shared/constants' +import { normalizeLoadedGlobalSettings } from './normalize-loaded-global-settings' +import { prepareLoadedTerminalSettings } from './prepare-loaded-terminal-settings' +import { prepareLoadedProfileSettings } from './prepare-loaded-profile-settings' +import type { GlobalSettings } from '../../../shared/global-settings-types' +import type { PersistedState } from '../../../shared/persisted-state-types' + +// Simulates a profile created before the dedicated Experimental switch was persisted. +function normalizeLegacyProfile(overrides: Partial): PersistedState['settings'] { + const defaults = getDefaultPersistedState(homedir()) + const settings: Partial = { ...defaults.settings } + delete settings.showAgentsSidebar + delete settings.experimentalActivity + delete settings.experimentalAgentDashboardPopout + Object.assign(settings, overrides) + const parsed: PersistedState = { ...defaults, settings: settings as GlobalSettings } + const noop = (): void => {} + const terminal = prepareLoadedTerminalSettings(parsed, noop) + const profile = prepareLoadedProfileSettings(parsed, defaults, noop) + return normalizeLoadedGlobalSettings(parsed, terminal, profile) +} + +describe('showAgentsSidebar experimental-setting migration', () => { + it('keeps the sidebar for Agents-view opt-ins regardless of the dashboard experiment', () => { + const normalized = normalizeLegacyProfile({ + experimentalActivity: true, + experimentalAgentDashboardPopout: false + }) + expect(normalized.showAgentsSidebar).toBe(true) + expect(normalized.agentsSidebarMigratedFromExperimental).toBe(true) + }) + + it('carries the legacy Agents-view opt-in into the sidebar', () => { + expect(normalizeLegacyProfile({ experimentalActivity: true }).showAgentsSidebar).toBe(true) + }) + + it('does not show Agents migration copy for a dashboard-only opt-in', () => { + expect( + normalizeLegacyProfile({ experimentalAgentDashboardPopout: true }) + .agentsSidebarMigratedFromExperimental + ).toBe(false) + }) + + it('defaults profiles with no legacy signal to the sidebar', () => { + const normalized = normalizeLegacyProfile({}) + expect(normalized.showAgentsSidebar).toBe(true) + expect(normalized.agentsSidebarMigratedFromExperimental).toBe(false) + }) + + it('does not treat a dashboard opt-out as an Agents-tab opt-out', () => { + expect( + normalizeLegacyProfile({ experimentalAgentDashboardPopout: false }).showAgentsSidebar + ).toBe(true) + }) + + it('ignores a pre-stamp forced-default experimentalActivity true (not an opt-in)', () => { + const normalized = normalizeLegacyProfile({ + experimentalActivity: true, + experimentalActivityDefaultedOffForAllUsers: undefined + }) + expect(normalized.experimentalActivity).toBe(false) + expect(normalized.showAgentsSidebar).toBe(true) + expect(normalized.agentsSidebarMigratedFromExperimental).toBe(false) + }) + + it('preserves a stored showAgentsSidebar choice over legacy flags', () => { + expect( + normalizeLegacyProfile({ showAgentsSidebar: false, experimentalActivity: true }) + .showAgentsSidebar + ).toBe(false) + expect( + normalizeLegacyProfile({ + showAgentsSidebar: true, + experimentalAgentDashboardPopout: false + }).showAgentsSidebar + ).toBe(true) + }) +}) diff --git a/src/main/persistence/loading-store/normalize-loaded-global-settings.ts b/src/main/persistence/loading-store/normalize-loaded-global-settings.ts index fb5cc9f1fe8..9370ec4febc 100644 --- a/src/main/persistence/loading-store/normalize-loaded-global-settings.ts +++ b/src/main/persistence/loading-store/normalize-loaded-global-settings.ts @@ -1,4 +1,5 @@ import { getDefaultVoiceSettings } from '../../../shared/constants' +import { resolveAgentsSidebarVisible } from '../../../shared/agents-sidebar-visibility' import { normalizePRBotAuthorOverrides } from '../../../shared/pr-bot-author-overrides' import { normalizeTerminalQuickCommands } from '../../../shared/terminal-quick-commands' import { normalizeOpenInApplications } from '../../../shared/open-in-applications' @@ -85,6 +86,16 @@ export function normalizeLoadedGlobalSettings( ...migratedTerminalTuiScrollSensitivity.settings, experimentalActivity: migratedExperimentalActivity, experimentalActivityDefaultedOffForAllUsers: true, + // Keep the experimental Agents tab's rollout default for older profiles while + // preserving any choice made through its dedicated Experimental setting. + showAgentsSidebar: resolveAgentsSidebarVisible({ + showAgentsSidebar: parsed.settings?.showAgentsSidebar + }), + // Preserve the legacy opt-in before the experimental setting is normalized away. This + // drives the migration-specific introduction copy without changing runtime behavior. + agentsSidebarMigratedFromExperimental: + parsed.settings?.agentsSidebarMigratedFromExperimental === true || + migratedExperimentalActivity, // Why: compact worktree cards graduated from Experimental; preserve the old opt-in for rollout-era profiles. compactWorktreeCards: loadedCompactWorktreeCards, experimentalCompactWorktreeCards: undefined, diff --git a/src/main/persistence/loading-store/workspace-session-snapshot-publication.ts b/src/main/persistence/loading-store/workspace-session-snapshot-publication.ts index 97cd3b2b544..5b7f90cbb8f 100644 --- a/src/main/persistence/loading-store/workspace-session-snapshot-publication.ts +++ b/src/main/persistence/loading-store/workspace-session-snapshot-publication.ts @@ -15,6 +15,8 @@ import { registerPersistedPaneKeyAlias } from '../restoring-sessions/pane-alias- import { normalizeWorkspaceSessionPaneIdentities, remapAcknowledgedAgentPaneKeys, + remapActivityClearedAtPaneKeys, + remapManuallyUnreadTurnPaneKeys, remapSshRemotePtyLeaseLeafIds, type WorkspaceSessionPaneIdentityRemap } from '../restoring-sessions/workspace-pane-normalization' @@ -50,10 +52,30 @@ export function setLocalWorkspaceSession( context.runtime.state.ui?.acknowledgedAgentsByPaneKey, normalized.leafIdByInputLeafIdByTabId ) - if (remappedAcknowledgements.changed) { + const remappedActivityCutoffs = remapActivityClearedAtPaneKeys( + context.runtime.state.ui?.activityClearedAtByPaneKey, + normalized.leafIdByInputLeafIdByTabId + ) + const remappedManualUnread = remapManuallyUnreadTurnPaneKeys( + context.runtime.state.ui?.manuallyUnreadTurnsByPaneKey, + normalized.leafIdByInputLeafIdByTabId + ) + if ( + remappedAcknowledgements.changed || + remappedActivityCutoffs.changed || + remappedManualUnread.changed + ) { context.runtime.state.ui = { ...context.runtime.state.ui, - acknowledgedAgentsByPaneKey: remappedAcknowledgements.acknowledgements + ...(remappedAcknowledgements.changed + ? { acknowledgedAgentsByPaneKey: remappedAcknowledgements.acknowledgements } + : {}), + ...(remappedActivityCutoffs.changed + ? { activityClearedAtByPaneKey: remappedActivityCutoffs.cutoffs } + : {}), + ...(remappedManualUnread.changed + ? { manuallyUnreadTurnsByPaneKey: remappedManualUnread.turns } + : {}) } } for (const entry of normalized.legacyPaneKeyAliasEntries) { diff --git a/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts b/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts new file mode 100644 index 00000000000..c0359efd612 --- /dev/null +++ b/src/main/persistence/restoring-sessions/pane-key-remapping.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { makePaneKey } from '../../../shared/stable-pane-id' +import { + remapActivityClearedAtPaneKeys, + remapManuallyUnreadTurnPaneKeys +} from './pane-key-remapping' + +const STABLE_LEAF_ID = '00000000-0000-4000-8000-000000000001' + +describe('remapManuallyUnreadTurnPaneKeys', () => { + it('promotes legacy pane keys to the restored stable leaf like clear-completed cutoffs', () => { + const remap = new Map([['tab-1', new Map([['pane:1', STABLE_LEAF_ID]])]]) + const turns = { 'tab-1:pane:1': 42, 'tab-2:pane:9': 7 } + + const result = remapManuallyUnreadTurnPaneKeys(turns, remap) + + expect(result.changed).toBe(true) + expect(result.turns).toEqual({ [makePaneKey('tab-1', STABLE_LEAF_ID)]: 42, 'tab-2:pane:9': 7 }) + // Same remap contract as the cutoffs so the two never drift after a session restore. + expect(remapActivityClearedAtPaneKeys(turns, remap).cutoffs).toEqual(result.turns) + }) + + it('reports no change for empty or already-stable records', () => { + const remap = new Map([['tab-1', new Map([['pane:1', STABLE_LEAF_ID]])]]) + expect(remapManuallyUnreadTurnPaneKeys(undefined, remap).changed).toBe(false) + expect(remapManuallyUnreadTurnPaneKeys({}, remap).changed).toBe(false) + const stable = { [makePaneKey('tab-1', STABLE_LEAF_ID)]: 1 } + expect(remapManuallyUnreadTurnPaneKeys(stable, remap)).toEqual({ + turns: stable, + changed: false + }) + }) +}) diff --git a/src/main/persistence/restoring-sessions/pane-key-remapping.ts b/src/main/persistence/restoring-sessions/pane-key-remapping.ts new file mode 100644 index 00000000000..4f3440f087e --- /dev/null +++ b/src/main/persistence/restoring-sessions/pane-key-remapping.ts @@ -0,0 +1,75 @@ +import type { PersistedState } from '../../../shared/persisted-state-types' +import { isTerminalLeafId, makePaneKey, parsePaneKey } from '../../../shared/stable-pane-id' + +type PaneLeafRemap = Map> + +function remapPaneKeys( + values: Record | undefined, + leafIdByInputLeafIdByTabId: PaneLeafRemap +): { values: Record | undefined; changed: boolean } { + if (!values || Object.keys(values).length === 0) { + return { values, changed: false } + } + + let changed = false + const next: Record = {} + const setValue = (paneKey: string, value: T): void => { + const existing = next[paneKey] + next[paneKey] = existing === undefined ? value : (Math.max(existing, value) as T) + } + for (const [paneKey, value] of Object.entries(values)) { + const parsed = parsePaneKey(paneKey) + if (parsed) { + setValue(paneKey, value) + continue + } + + const delimiter = paneKey.indexOf(':') + if (delimiter <= 0 || delimiter === paneKey.length - 1) { + setValue(paneKey, value) + continue + } + + const tabId = paneKey.slice(0, delimiter) + const legacyLeafId = paneKey.slice(delimiter + 1) + const remappedLeafId = leafIdByInputLeafIdByTabId.get(tabId)?.get(legacyLeafId) + if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { + setValue(paneKey, value) + continue + } + + try { + // Carry values over when a legacy leaf is promoted to a UUID. + setValue(makePaneKey(tabId, remappedLeafId), value) + changed = true + } catch { + setValue(paneKey, value) + } + } + + return { values: next, changed } +} + +export function remapAcknowledgedAgentPaneKeys( + acknowledgements: PersistedState['ui']['acknowledgedAgentsByPaneKey'], + leafIdByInputLeafIdByTabId: PaneLeafRemap +): { acknowledgements: PersistedState['ui']['acknowledgedAgentsByPaneKey']; changed: boolean } { + const result = remapPaneKeys(acknowledgements, leafIdByInputLeafIdByTabId) + return { acknowledgements: result.values, changed: result.changed } +} + +export function remapManuallyUnreadTurnPaneKeys( + turns: PersistedState['ui']['manuallyUnreadTurnsByPaneKey'], + leafIdByInputLeafIdByTabId: PaneLeafRemap +): { turns: PersistedState['ui']['manuallyUnreadTurnsByPaneKey']; changed: boolean } { + const result = remapPaneKeys(turns, leafIdByInputLeafIdByTabId) + return { turns: result.values, changed: result.changed } +} + +export function remapActivityClearedAtPaneKeys( + cutoffs: PersistedState['ui']['activityClearedAtByPaneKey'], + leafIdByInputLeafIdByTabId: PaneLeafRemap +): { cutoffs: PersistedState['ui']['activityClearedAtByPaneKey']; changed: boolean } { + const result = remapPaneKeys(cutoffs, leafIdByInputLeafIdByTabId) + return { cutoffs: result.values, changed: result.changed } +} diff --git a/src/main/persistence/restoring-sessions/workspace-pane-normalization.ts b/src/main/persistence/restoring-sessions/workspace-pane-normalization.ts index 2bfafebae9f..cfc26c7bb8a 100644 --- a/src/main/persistence/restoring-sessions/workspace-pane-normalization.ts +++ b/src/main/persistence/restoring-sessions/workspace-pane-normalization.ts @@ -8,7 +8,7 @@ import { type ExecutionHostId } from '../../../shared/execution-host' import type { SshRemotePtyLease } from '../../../shared/ssh-types' -import { isTerminalLeafId, makePaneKey, parsePaneKey } from '../../../shared/stable-pane-id' +import { isTerminalLeafId, parsePaneKey } from '../../../shared/stable-pane-id' import { findCrossHostPaneTabIds, withoutPaneTabIds } from './cross-host-pane-tab-ids' import { createLazyTerminalTabLookup, @@ -22,6 +22,17 @@ import { migrationUnsupportedEntriesEqual, normalizeLegacyPaneKeyAliasEntries } from './pane-alias-normalization' +import { + remapAcknowledgedAgentPaneKeys, + remapActivityClearedAtPaneKeys, + remapManuallyUnreadTurnPaneKeys +} from './pane-key-remapping' + +export { + remapAcknowledgedAgentPaneKeys, + remapActivityClearedAtPaneKeys, + remapManuallyUnreadTurnPaneKeys +} from './pane-key-remapping' export function normalizeWorkspaceSessionPaneIdentities( session: WorkspaceSessionState, @@ -220,6 +231,14 @@ export function normalizePersistedPaneIdentityState(state: PersistedState): { state.ui?.acknowledgedAgentsByPaneKey, withoutPaneTabIds(acknowledgementLeafIdByInputLeafIdByTabId, crossHostTabIds) ) + const remappedActivityCutoffs = remapActivityClearedAtPaneKeys( + state.ui?.activityClearedAtByPaneKey, + withoutPaneTabIds(acknowledgementLeafIdByInputLeafIdByTabId, crossHostTabIds) + ) + const remappedManualUnread = remapManuallyUnreadTurnPaneKeys( + state.ui?.manuallyUnreadTurnsByPaneKey, + withoutPaneTabIds(acknowledgementLeafIdByInputLeafIdByTabId, crossHostTabIds) + ) const migrationUnsupportedChanged = !migrationUnsupportedEntriesEqual( state.migrationUnsupportedPtyEntries ?? [], mergedMigrationUnsupportedEntries @@ -234,7 +253,9 @@ export function normalizePersistedPaneIdentityState(state: PersistedState): { !remappedLeases.changed && !migrationUnsupportedChanged && !legacyAliasesChanged && - !remappedAcknowledgements.changed + !remappedAcknowledgements.changed && + !remappedActivityCutoffs.changed && + !remappedManualUnread.changed ) { return { state, @@ -251,11 +272,21 @@ export function normalizePersistedPaneIdentityState(state: PersistedState): { sshRemotePtyLeases: remappedLeases.leases, migrationUnsupportedPtyEntries: mergedMigrationUnsupportedEntries, legacyPaneKeyAliasEntries: mergedLegacyPaneKeyAliasEntries, - ...(remappedAcknowledgements.changed + ...(remappedAcknowledgements.changed || + remappedActivityCutoffs.changed || + remappedManualUnread.changed ? { ui: { ...state.ui, - acknowledgedAgentsByPaneKey: remappedAcknowledgements.acknowledgements + ...(remappedAcknowledgements.changed + ? { acknowledgedAgentsByPaneKey: remappedAcknowledgements.acknowledgements } + : {}), + ...(remappedActivityCutoffs.changed + ? { activityClearedAtByPaneKey: remappedActivityCutoffs.cutoffs } + : {}), + ...(remappedManualUnread.changed + ? { manuallyUnreadTurnsByPaneKey: remappedManualUnread.turns } + : {}) } } : {}) @@ -265,50 +296,3 @@ export function normalizePersistedPaneIdentityState(state: PersistedState): { legacyPaneKeyAliasEntries: mergedLegacyPaneKeyAliasEntries } } - -export function remapAcknowledgedAgentPaneKeys( - acknowledgements: PersistedState['ui']['acknowledgedAgentsByPaneKey'], - leafIdByInputLeafIdByTabId: Map> -): { acknowledgements: PersistedState['ui']['acknowledgedAgentsByPaneKey']; changed: boolean } { - if (!acknowledgements || Object.keys(acknowledgements).length === 0) { - return { acknowledgements, changed: false } - } - - let changed = false - const next: NonNullable = {} - const setAcknowledgement = (paneKey: string, acknowledgedAt: number): void => { - const existing = next[paneKey] - next[paneKey] = existing === undefined ? acknowledgedAt : Math.max(existing, acknowledgedAt) - } - for (const [paneKey, acknowledgedAt] of Object.entries(acknowledgements)) { - const parsed = parsePaneKey(paneKey) - if (parsed) { - setAcknowledgement(paneKey, acknowledgedAt) - continue - } - - const delimiter = paneKey.indexOf(':') - if (delimiter <= 0 || delimiter === paneKey.length - 1) { - setAcknowledgement(paneKey, acknowledgedAt) - continue - } - - const tabId = paneKey.slice(0, delimiter) - const legacyLeafId = paneKey.slice(delimiter + 1) - const remappedLeafId = leafIdByInputLeafIdByTabId.get(tabId)?.get(legacyLeafId) - if (!remappedLeafId || !isTerminalLeafId(remappedLeafId)) { - setAcknowledgement(paneKey, acknowledgedAt) - continue - } - - try { - // Why: when a legacy leaf is promoted to a UUID, carry the read marker over so seen rows don't come back unread. - setAcknowledgement(makePaneKey(tabId, remappedLeafId), acknowledgedAt) - changed = true - } catch { - setAcknowledgement(paneKey, acknowledgedAt) - } - } - - return { acknowledgements: next, changed } -} diff --git a/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts b/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts index 02c93c14f6b..d6c08ad44cf 100644 --- a/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts +++ b/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts @@ -44,7 +44,13 @@ describe('client UI RPC pairing-local field seams', () => { manualRepoOrder: [ { hostId: 'runtime:web-11111111-2222-3333-4444-555555555555', repoId: 'repo-a' } ], - workspaceHostOrder: ['runtime:web-11111111-2222-3333-4444-555555555555', 'local'] + workspaceHostOrder: ['runtime:web-11111111-2222-3333-4444-555555555555', 'local'], + agentsVisibleHostIds: ['runtime:web-11111111-2222-3333-4444-555555555555'], + agentsFilterRepoIds: ['repo-a'], + agentsShowChildAgents: true, + agentsCompactMode: false, + activityClearedAtByPaneKey: { 'tab-1:leaf-1': 123 }, + manuallyUnreadTurnsByPaneKey: { 'tab-1:leaf-1': 321 } } it.each(PAIRING_LOCAL_UI_FIELDS.map((field) => [field] as const))( diff --git a/src/main/runtime/rpc/methods/client-ui-schemas.ts b/src/main/runtime/rpc/methods/client-ui-schemas.ts index 79b923a55db..edb3f651c32 100644 --- a/src/main/runtime/rpc/methods/client-ui-schemas.ts +++ b/src/main/runtime/rpc/methods/client-ui-schemas.ts @@ -122,6 +122,10 @@ const UiUpdateFields = z showInactiveWorkspaces: z.boolean().optional(), workspaceHostScope: z.string().optional(), visibleWorkspaceHostIds: z.array(z.string()).nullable().optional(), + agentsVisibleHostIds: z.array(z.string()).nullable().optional(), + agentsFilterRepoIds: StringArray.optional(), + agentsShowChildAgents: z.boolean().optional(), + agentsCompactMode: z.boolean().optional(), workspaceHostOrder: z.array(z.string()).optional(), automationHostFilter: z .union([ @@ -171,6 +175,8 @@ const UiUpdateFields = z updateReassuranceSeen: z.boolean().optional(), osc52ClipboardDefaultOnNoticePending: z.boolean().optional(), acknowledgedAgentsByPaneKey: z.record(z.string(), z.number().finite()).optional(), + activityClearedAtByPaneKey: z.record(z.string(), z.number().finite()).optional(), + manuallyUnreadTurnsByPaneKey: z.record(z.string(), z.number().finite()).optional(), browserDefaultUrl: NullableString.optional(), browserDefaultSearchEngine: z .enum(['google', 'duckduckgo', 'bing', 'kagi']) diff --git a/src/preload/api/agent-status-api.ts b/src/preload/api/agent-status-api.ts index fd6da88e351..7aa6c21115d 100644 --- a/src/preload/api/agent-status-api.ts +++ b/src/preload/api/agent-status-api.ts @@ -1,4 +1,5 @@ import type { + AgentStatusCacheIdentity, AgentStatusClearIpcPayload, AgentStatusIpcPayload, MigrationUnsupportedPtyEntry @@ -30,6 +31,10 @@ export type AgentStatusApi = { getMigrationUnsupportedSnapshot: () => Promise /** Drop a paneKey from the main-process hook cache and on-disk last-status file. Fire-and-forget. */ drop: (paneKey: string) => void + /** Evict a previously-cleared status only when its identity still matches the main-process cache. */ + dropPersisted: (identity: AgentStatusCacheIdentity) => void + /** Same as dropPersisted for many identities in one IPC message and one listener notification. */ + dropPersistedBatch?: (identities: readonly AgentStatusCacheIdentity[]) => void /** Retire a pane whose agent process is proven gone — clears the row AND the per-pane caches a * dismissal deliberately keeps. Not `drop`: that one is a user dismissal of a live pane's row. */ reconcileEndedProcess: (paneKey: string) => void diff --git a/src/preload/api/agent-status-bridge.ts b/src/preload/api/agent-status-bridge.ts index b5ac5c8b5b2..3cc1654aaed 100644 --- a/src/preload/api/agent-status-bridge.ts +++ b/src/preload/api/agent-status-bridge.ts @@ -1,5 +1,6 @@ import { ipcRenderer } from 'electron' import type { + AgentStatusCacheIdentity, AgentStatusClearIpcPayload, AgentStatusIpcPayload, MigrationUnsupportedPtyEntry @@ -66,6 +67,12 @@ export const agentStatusApi = { drop: (paneKey: string): void => { ipcRenderer.send('agentStatus:drop', paneKey) }, + dropPersisted: (identity: AgentStatusCacheIdentity): void => { + ipcRenderer.send('agentStatus:dropPersisted', identity) + }, + dropPersistedBatch: (identities: readonly AgentStatusCacheIdentity[]): void => { + ipcRenderer.send('agentStatus:dropPersistedBatch', identities) + }, reconcileEndedProcess: (paneKey: string): void => { ipcRenderer.send('agentStatus:reconcileEndedProcess', paneKey) }, diff --git a/src/renderer/src/app-shell/use-app-startup-hydration.ts b/src/renderer/src/app-shell/use-app-startup-hydration.ts index 2b5272f440d..447fbfb3780 100644 --- a/src/renderer/src/app-shell/use-app-startup-hydration.ts +++ b/src/renderer/src/app-shell/use-app-startup-hydration.ts @@ -269,6 +269,16 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta // Why (issue #1158): unlock the session writer only after hydration and all dependent steps succeeded, so a mid-startup throw can't serialize partially-mutated state to disk. actions.setHydrationSucceeded(true) actions.setTerminalStartupRestorationReady(true) + // Why the explicit opt-in: unconditional seeding hijacks every empty dev + // profile's active workspace, making onboarding/empty-state flows untestable. + if ( + import.meta.env.DEV && + String(import.meta.env.VITE_ACTIVITY_DEV_FIXTURE).toLowerCase() === 'true' + ) { + const { seedDevActivityFixture } = + await import('../components/activity/dev-activity-fixture') + seedDevActivityFixture() + } logRendererStartupDiagnostic('startup-hydration-done', { durationMs: Math.round(performance.now() - startupStartedAt) }) diff --git a/src/renderer/src/app-shell/use-persisted-ui-writer.ts b/src/renderer/src/app-shell/use-persisted-ui-writer.ts index 857b55fc8a4..19109c25eb2 100644 --- a/src/renderer/src/app-shell/use-persisted-ui-writer.ts +++ b/src/renderer/src/app-shell/use-persisted-ui-writer.ts @@ -167,7 +167,11 @@ export function usePersistedUIWriter(): void { // paths in agent-status.ts (close/dismiss) flow to disk through map identity changes. // Without persisting, agent rows that survive restart come back bold even when the // user had already visited them. - acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey + acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey, + // Why: "Clear completed" must survive restart, or cleared done/interrupted rows return. + activityClearedAtByPaneKey: s.activityClearedAtByPaneKey, + // Why: an explicit "mark unread" must survive restart, or the row comes back read. + manuallyUnreadTurnsByPaneKey: s.manuallyUnreadTurnsByPaneKey })) ) useEffect(() => { diff --git a/src/renderer/src/components/activity/ActivityPrototypePage.test.ts b/src/renderer/src/components/activity/ActivityPrototypePage.test.ts index 48bdbb3a250..9ccf93248dc 100644 --- a/src/renderer/src/components/activity/ActivityPrototypePage.test.ts +++ b/src/renderer/src/components/activity/ActivityPrototypePage.test.ts @@ -399,7 +399,7 @@ describe('buildActivityEvents', () => { expect(threads[0].events[0].entry.prompt).toBe('Retained prior run') }) - it('groups visible threads by current status order', () => { + it('groups visible threads with attention states before working and done', () => { const repo = makeRepo() const worktree = makeWorktree() const workingTab = makeTab() @@ -442,11 +442,49 @@ describe('buildActivityEvents', () => { }) ) - expect(groups.map((group) => group.id)).toEqual(['working', 'blocked', 'done']) + expect(groups.map((group) => group.id)).toEqual(['blocked', 'working', 'done']) expect(groups.map((group) => group.threads.map((thread) => thread.paneKey))).toEqual([ - [PANE_KEY], [PANE_KEY_2], + [PANE_KEY], [PANE_KEY_3] ]) }) + + it('merges runtime orchestration context into activity events and entries', () => { + const repo = makeRepo() + const worktree = makeWorktree() + const tab1 = makeTabWithIds('tab-1', worktree.id) + const tab2 = makeTabWithIds('tab-2', worktree.id) + const result = buildActivityEvents({ + agentStatusByPaneKey: { + [PANE_KEY]: makeWorkingEntryWithoutHistory(), + [PANE_KEY_2]: { + ...makeWorkingEntryWithoutHistory(), + paneKey: PANE_KEY_2, + terminalHandle: 'terminal-child' + } + }, + runtimeAgentOrchestrationByPaneKey: { + [PANE_KEY_2]: { + parentPaneKey: PANE_KEY, + parentTerminalHandle: 'terminal-parent', + taskId: 'task-counsel', + dispatchId: 'ctx-counsel' + } + }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { + [worktree.id]: [tab1, tab2] + }, + worktreeMap: new Map([[worktree.id, worktree]]), + repoMap: new Map([[repo.id, repo]]), + acknowledgedAgentsByPaneKey: {}, + now: 5_000 + }) + + expect(result.liveAgentByPaneKey[PANE_KEY_2].entry.orchestration?.parentPaneKey).toBe(PANE_KEY) + expect(result.liveAgentByPaneKey[PANE_KEY_2].entry.orchestration?.parentTerminalHandle).toBe( + 'terminal-parent' + ) + }) }) diff --git a/src/renderer/src/components/activity/ActivityPrototypePage.thread-grouping.test.ts b/src/renderer/src/components/activity/ActivityPrototypePage.thread-grouping.test.ts index 72a1b6d3355..0435f2c61f2 100644 --- a/src/renderer/src/components/activity/ActivityPrototypePage.thread-grouping.test.ts +++ b/src/renderer/src/components/activity/ActivityPrototypePage.thread-grouping.test.ts @@ -207,4 +207,17 @@ describe('activity thread grouping', () => { it('returns no groups for empty thread input', () => { expect(buildActivityThreadGroups([], 'status')).toEqual([]) }) + + it('keeps all threads in one ungrouped list', () => { + const threads = makeThreads( + makeActivityResult({ + entries: { + [PANE_KEY]: makeWorkingEntryWithoutHistory(), + [PANE_KEY_2]: makeWorkingEntryWithoutHistory() + } + }) + ) + + expect(buildActivityThreadGroups(threads, 'none')).toEqual([{ key: 'all', label: '', threads }]) + }) }) diff --git a/src/renderer/src/components/activity/ActivityPrototypePage.tsx b/src/renderer/src/components/activity/ActivityPrototypePage.tsx index 59fa8a0f1a0..83581d4cf10 100644 --- a/src/renderer/src/components/activity/ActivityPrototypePage.tsx +++ b/src/renderer/src/components/activity/ActivityPrototypePage.tsx @@ -1,8 +1,7 @@ import React, { useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react' -import { useShallow } from 'zustand/react/shallow' -import { useSidebarResize } from '@/hooks/useSidebarResize' import { useAppStore } from '@/store' -import { getRepoMapFromState, getWorktreeMapFromState } from '@/store/selectors' +import { useSidebarResize } from '@/hooks/useSidebarResize' +import { ActivityScopeFilterChips } from './activity-scope-filter-controls' import { setActivityTerminalPortals, type ActivityTerminalPortalTarget @@ -11,15 +10,10 @@ import { reconcileActivityPortalThreads, resolveActivityPortalSwap } from './activity-portal-thread-reconciliation' -import { buildActivityEvents } from './activity-event-builder' -import { buildAgentPaneThreads } from './activity-thread-builder' -import { - activityThreadMatchesSearchQuery, - buildActivityThreadGroups, - isActivitySearchQueryTooLarge -} from './activity-thread-grouping' +import { useAgentPaneThreads } from './use-agent-pane-threads' import { handleActivityFilterFocusShortcut } from './activity-filter-focus-shortcut' -import { createActivityThreadActions } from './activity-thread-actions' +import { hasActivityThreadWorkspace } from './activity-thread-actions' +import { useActivityThreadActionBindings } from './use-activity-thread-action-bindings' import { ActivityThreadListPane } from './activity-thread-list-pane' import { ActivityThreadDetailPane } from './activity-thread-detail-pane' import { @@ -42,7 +36,11 @@ export default function ActivityPrototypePage(): React.JSX.Element { const activityFilterInputRef = useRef(null) // Why: bounds auto mark-read to one acknowledgement per selected thread turn. const autoAcknowledgedTurnRef = useRef(null) - const [compactMode, setCompactMode] = useState(false) + // Why store-backed: persisted preferences shared with the sidebar agents list. + const compactMode = useAppStore((s) => s.agentsCompactMode) + const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) + const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) + const setShowChildAgents = useAppStore((s) => s.setAgentsShowChildAgents) const [selectedPaneKey, setSelectedPaneKey] = useState(null) const [displayedPaneKey, setDisplayedPaneKey] = useState(null) const [activePortalSlotId, setActivePortalSlotId] = @@ -64,86 +62,30 @@ export default function ActivityPrototypePage(): React.JSX.Element { setWidth: setThreadListWidth }) - const storeData = useAppStore( - useShallow((s) => ({ - agentStatusByPaneKey: s.agentStatusByPaneKey, - migrationUnsupportedByPtyId: s.migrationUnsupportedByPtyId, - retainedAgentsByPaneKey: s.retainedAgentsByPaneKey, - tabsByWorktree: s.tabsByWorktree, - worktreeMap: getWorktreeMapFromState(s), - repoMap: getRepoMapFromState(s), - acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey, - acknowledgeAgents: s.acknowledgeAgents, - unacknowledgeAgents: s.unacknowledgeAgents, - generatedTitlesEnabled: s.settings?.tabAutoGenerateTitle === true - })) - ) - // Why: agentStatusEpoch is a dep (not used in the body) so the memo recomputes when freshness boundaries expire even without new PTY data. - const agentStatusEpoch = useAppStore((s) => s.agentStatusEpoch) - - const { events: allEvents, liveAgentByPaneKey } = useMemo( - () => - buildActivityEvents({ - agentStatusByPaneKey: storeData.agentStatusByPaneKey, - migrationUnsupportedByPtyId: storeData.migrationUnsupportedByPtyId, - retainedAgentsByPaneKey: storeData.retainedAgentsByPaneKey, - tabsByWorktree: storeData.tabsByWorktree, - worktreeMap: storeData.worktreeMap, - repoMap: storeData.repoMap, - acknowledgedAgentsByPaneKey: storeData.acknowledgedAgentsByPaneKey, - // Why: Date.now() is read in the memo body (not a dep) so stale-decay recomputes when agentStatusEpoch ticks, not on wall-clock time. - now: Date.now() - }), - // eslint-disable-next-line react-hooks/exhaustive-deps - [storeData, agentStatusEpoch] - ) - - const allThreads = useMemo( - () => - buildAgentPaneThreads({ - events: allEvents, - liveAgentByPaneKey, - generatedTitlesEnabled: storeData.generatedTitlesEnabled - }), - [allEvents, liveAgentByPaneKey, storeData.generatedTitlesEnabled] - ) - const selectedPaneKeyIsLive = - selectedPaneKey === null || allThreads.some((thread) => thread.paneKey === selectedPaneKey) - const effectiveSelectedPaneKey = selectedPaneKeyIsLive ? selectedPaneKey : null + const { + storeData, + allThreads, + selectedPaneKeyIsLive, + effectiveSelectedPaneKey, + visibleThreads, + markAllReadThreads, + visibleThreadGroups + } = useAgentPaneThreads({ query, readFilter, groupBy, selectedPaneKey, showChildAgents }) if (!selectedPaneKeyIsLive) { // Why: rows disappear when agent retention or tab state changes; clear stale selection before detail/portal rendering targets it. setSelectedPaneKey(null) } - const visibleThreads = useMemo(() => { - const normalizedQuery = isActivitySearchQueryTooLarge(query) ? null : query.trim().toLowerCase() - return allThreads.filter((thread) => { - // Why: keep the just-selected thread visible after auto-mark-read flips it to read, else unread-only mode makes the clicked row vanish from the list. - if ( - readFilter === 'unread' && - !thread.unread && - thread.paneKey !== effectiveSelectedPaneKey - ) { - return false - } - if (normalizedQuery === null) { - return false - } - return activityThreadMatchesSearchQuery({ thread, searchQuery: normalizedQuery }) - }) - }, [allThreads, readFilter, query, effectiveSelectedPaneKey]) - const visibleThreadGroups = useMemo( - () => buildActivityThreadGroups(visibleThreads, groupBy), - [visibleThreads, groupBy] - ) - const selectedThread = effectiveSelectedPaneKey ? (allThreads.find((thread) => thread.paneKey === effectiveSelectedPaneKey) ?? null) : null const selectedTabId = selectedThread?.tab.id ?? null + const selectedWorktreeAvailable = selectedThread + ? hasActivityThreadWorkspace(selectedThread, storeData) + : false // Why: repo-less terminal buckets can produce Activity rows, but the workspace Terminal tree only portals real worktrees. const selectedHasLiveTab = - selectedThread && selectedTabId && storeData.worktreeMap.has(selectedThread.worktree.id) + selectedThread && selectedTabId && selectedWorktreeAvailable ? (storeData.tabsByWorktree[selectedThread.worktree.id] ?? []).some( (tab) => tab.id === selectedTabId ) @@ -152,8 +94,11 @@ export default function ActivityPrototypePage(): React.JSX.Element { ? (allThreads.find((thread) => thread.paneKey === displayedPaneKey) ?? null) : null const displayedTabId = displayedThread?.tab.id ?? null + const displayedWorktreeAvailable = displayedThread + ? hasActivityThreadWorkspace(displayedThread, storeData) + : false const displayedHasLiveTab = - displayedThread && displayedTabId && storeData.worktreeMap.has(displayedThread.worktree.id) + displayedThread && displayedTabId && displayedWorktreeAvailable ? (storeData.tabsByWorktree[displayedThread.worktree.id] ?? []).some( (tab) => tab.id === displayedTabId ) @@ -298,13 +243,38 @@ export default function ActivityPrototypePage(): React.JSX.Element { return () => window.removeEventListener('keydown', focusActivityFilter, { capture: true }) }, [activePortalTargetEl, inactivePortalTargetEl]) - const { hasUnreadThreads, markThreadUnread, selectThread, jumpToWorkspace, markAllThreadsRead } = - createActivityThreadActions({ - allThreads, - acknowledgeAgents: storeData.acknowledgeAgents, - unacknowledgeAgents: storeData.unacknowledgeAgents, - setSelectedPaneKey - }) + const { + markThreadRead, + markThreadUnread, + selectThread, + jumpToWorkspace, + markAllThreadsRead, + hasUnreadThreads, + hasCompletedThreads, + handleClearCompleted + } = useActivityThreadActionBindings({ + visibleThreads, + markAllReadThreads, + acknowledgeAgents: storeData.acknowledgeAgents, + unacknowledgeAgents: storeData.unacknowledgeAgents, + setSelectedPaneKey + }) + + const canJumpToWorkspace = useCallback( + (thread: Parameters[0]) => + hasActivityThreadWorkspace(thread, { + worktreesByRepo: storeData.worktreesByRepo, + detectedWorktreesByRepo: storeData.detectedWorktreesByRepo, + folderWorkspaces: storeData.folderWorkspaces, + defaultHostId: storeData.defaultHostId + }), + [ + storeData.worktreesByRepo, + storeData.detectedWorktreesByRepo, + storeData.folderWorkspaces, + storeData.defaultHostId + ] + ) useEffect(() => { if ( @@ -356,25 +326,29 @@ export default function ActivityPrototypePage(): React.JSX.Element { readFilter={readFilter} onReadFilterChange={setReadFilter} compactMode={compactMode} + showChildAgents={showChildAgents} hasUnreadThreads={hasUnreadThreads} onCompactModeChange={setCompactMode} + onShowChildAgentsChange={setShowChildAgents} onMarkAllThreadsRead={markAllThreadsRead} + hasCompletedThreads={hasCompletedThreads} + onClearCompleted={handleClearCompleted} visibleThreadGroups={visibleThreadGroups} visibleThreadCount={visibleThreads.length} selectedPaneKey={selectedThread?.paneKey ?? null} onSelectThread={selectThread} onJumpToWorkspace={jumpToWorkspace} + onMarkThreadRead={markThreadRead} onMarkThreadUnread={markThreadUnread} - canJumpToWorkspace={(thread) => storeData.worktreeMap.has(thread.worktree.id)} + canJumpToWorkspace={canJumpToWorkspace} isThreadListResizing={isThreadListResizing} onResizeStart={onResizeStart} + scopeFilterRow={} /> void compactMode?: boolean + showChildAgents?: boolean + onShowChildAgentsChange?: (showChildAgents: boolean) => void hasUnreadThreads?: boolean }): ReactElement { return ( { expect(document.body.textContent).toContain('Compact mode') }) + + it('renders group by options when provided', async () => { + const onGroupByChange = vi.fn() + await act(async () => { + root.render() + }) + + const trigger = container.querySelector( + 'button[aria-label="Thread list options"]' + ) + + await act(async () => { + trigger?.dispatchEvent(new KeyboardEvent('keydown', { bubbles: true, key: 'Enter' })) + }) + + expect(document.body.textContent).toContain('Group by') + expect(document.body.textContent).toContain('Status') + + const subTrigger = document.querySelector( + '[data-slot="dropdown-menu-sub-trigger"]' + ) + expect(subTrigger).not.toBeNull() + + await act(async () => { + subTrigger?.dispatchEvent(new KeyboardEvent('keydown', { bubbles: true, key: 'ArrowRight' })) + }) + + expect(document.body.textContent).toContain('Project') + expect(document.body.textContent).toContain('Worktree') + expect(document.body.textContent).toContain('Agent') + }) + + it('explains compact mode on hover', async () => { + await act(async () => { + root.render() + }) + + const trigger = container.querySelector( + 'button[aria-label="Thread list options"]' + ) + await act(async () => { + trigger?.dispatchEvent(new KeyboardEvent('keydown', { bubbles: true, key: 'Enter' })) + }) + + const compactMode = document.querySelector('[role="menuitemcheckbox"]') + await act(async () => { + compactMode?.dispatchEvent(new Event('pointermove', { bubbles: true })) + }) + + expect(document.body.textContent).toContain( + 'Shows shorter thread rows with one-line titles and two-line status messages.' + ) + }) + + it('puts search and unread actions in the menu when header overflow handlers are provided', async () => { + const onSearch = vi.fn() + const onToggleUnread = vi.fn() + await act(async () => { + root.render( + + + + ) + }) + + const trigger = container.querySelector( + 'button[aria-label="Thread list options"]' + ) + await act(async () => { + trigger?.dispatchEvent(new KeyboardEvent('keydown', { bubbles: true, key: 'Enter' })) + }) + + expect(document.body.textContent).toContain('Search') + expect(document.body.textContent).toContain('Show unread only') + }) + + it('explains show unread threads only on hover and shows unread dot when hasUnreadThreads is true', async () => { + const onToggleUnread = vi.fn() + await act(async () => { + root.render( + + + + ) + }) + + const trigger = container.querySelector( + 'button[aria-label="Thread list options"]' + ) + await act(async () => { + trigger?.dispatchEvent(new KeyboardEvent('keydown', { bubbles: true, key: 'Enter' })) + }) + + const unreadItem = document.querySelector('[role="menuitemcheckbox"]') + await act(async () => { + unreadItem?.dispatchEvent(new Event('pointermove', { bubbles: true })) + }) + + expect(document.body.textContent).toContain( + 'Filters the activity list to show only threads with unread updates.' + ) + expect(document.querySelector('[data-unread-dot]')).not.toBeNull() + }) + + it('renders show child agents checkbox when onShowChildAgentsChange is provided', async () => { + const onShowChildAgentsChange = vi.fn() + await act(async () => { + root.render( + + ) + }) + + const trigger = container.querySelector( + 'button[aria-label="Thread list options"]' + ) + + await act(async () => { + trigger?.dispatchEvent(new KeyboardEvent('keydown', { bubbles: true, key: 'Enter' })) + }) + + expect(document.body.textContent).toContain('Show child agents') + }) }) diff --git a/src/renderer/src/components/activity/ActivityTitlebarControls.tsx b/src/renderer/src/components/activity/ActivityTitlebarControls.tsx index 93ca4c65c1c..c6194ffcc96 100644 --- a/src/renderer/src/components/activity/ActivityTitlebarControls.tsx +++ b/src/renderer/src/components/activity/ActivityTitlebarControls.tsx @@ -8,7 +8,7 @@ import { useActivityUnreadCount } from './useActivityUnreadCount' import { translate } from '@/i18n/i18n' export function ActivityTitlebarControls(): React.JSX.Element { - const unreadCount = useActivityUnreadCount(true, 'agent-events') + const unreadCount = useActivityUnreadCount() const closeActivityPage = useAppStore((s) => s.closeActivityPage) return ( diff --git a/src/renderer/src/components/activity/activity-auto-mark-read-loop.react185.test.tsx b/src/renderer/src/components/activity/activity-auto-mark-read-loop.react185.test.tsx index 9fdc4c0ecef..013afb58408 100644 --- a/src/renderer/src/components/activity/activity-auto-mark-read-loop.react185.test.tsx +++ b/src/renderer/src/components/activity/activity-auto-mark-read-loop.react185.test.tsx @@ -125,7 +125,7 @@ async function mountActivityPage(): Promise { } async function selectSeededThread(): Promise { - const row = Array.from(seededContainer.querySelectorAll('[role="button"]')).find( + const row = Array.from(seededContainer.querySelectorAll('[role="listitem"]')).find( (element) => element.textContent?.includes(PROMPT) ) expect(row).toBeDefined() diff --git a/src/renderer/src/components/activity/activity-clear-completed.test.ts b/src/renderer/src/components/activity/activity-clear-completed.test.ts new file mode 100644 index 00000000000..3723563d143 --- /dev/null +++ b/src/renderer/src/components/activity/activity-clear-completed.test.ts @@ -0,0 +1,362 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { RetainedAgentEntry } from '@/store/slices/agent-status' +import type { ActivityEvent, AgentPaneThread } from './activity-thread-types' +import { makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' + +const mockStore = vi.hoisted(() => { + const state = { + activityClearedAtByPaneKey: {} as Record, + agentStatusByPaneKey: {} as Record, + retainedAgentsByPaneKey: {} as Record, + retentionSuppressedPaneKeys: {} as Record, + applyActivityClearedAt: vi.fn((patch: Record) => { + const next = { ...state.activityClearedAtByPaneKey } + for (const [key, value] of Object.entries(patch)) { + if (value === null) { + delete next[key] + } else { + next[key] = value + } + } + state.activityClearedAtByPaneKey = next + }), + dismissRetainedAgents: vi.fn((paneKeys: readonly string[]) => { + const next = { ...state.retainedAgentsByPaneKey } + for (const key of paneKeys) { + if (state.agentStatusByPaneKey[key]) { + state.retentionSuppressedPaneKeys[key] = true + } + delete next[key] + } + state.retainedAgentsByPaneKey = next + }), + clearRetentionSuppressedPaneKeys: vi.fn((paneKeys: string[]) => { + for (const key of paneKeys) { + delete state.retentionSuppressedPaneKeys[key] + } + }), + retainAgents: vi.fn((entries: RetainedAgentEntry[]) => { + const next = { ...state.retainedAgentsByPaneKey } + for (const retained of entries) { + next[retained.entry.paneKey] = retained + } + state.retainedAgentsByPaneKey = next + }) + } + return state +}) + +const toastSpy = vi.hoisted(() => vi.fn()) + +vi.mock('@/store', () => ({ + useAppStore: { getState: () => mockStore } +})) +vi.mock('sonner', () => ({ toast: toastSpy })) + +import { + CLEAR_COMPLETED_EVICTION_FALLBACK_MS, + clearCompletedActivity, + flushPendingClearCompletedEvictions, + isClearableActivityThread, + planClearCompletedActivity +} from './activity-clear-completed' + +function makeThread(paneKey: string, overrides: Partial = {}): AgentPaneThread { + return { + paneKey, + tab: makeTab(), + worktree: makeWorktree(), + repo: null, + currentAgentState: null, + currentAgentEntry: null, + latestEvent: null, + latestTimestamp: 5_000, + agentType: 'claude', + unread: false, + paneTitle: `Agent ${paneKey}`, + responsePreview: '', + events: [], + ...overrides + } +} + +function doneEvent(interrupted: boolean): ActivityEvent { + return { + id: 'evt', + state: 'done', + timestamp: 5_000, + worktree: makeWorktree(), + repo: null, + entry: { interrupted } as ActivityEvent['entry'], + tab: makeTab(), + agentType: 'claude', + agentAlive: false, + unread: false + } +} + +const workingThread = makeThread('t-working:1', { currentAgentState: 'working' }) +const blockedThread = makeThread('t-blocked:1', { currentAgentState: 'blocked' }) +const waitingThread = makeThread('t-waiting:1', { currentAgentState: 'waiting' }) +const doneThread = makeThread('t-done:1', { latestEvent: doneEvent(false) }) +const interruptedThread = makeThread('t-interrupted:1', { latestEvent: doneEvent(true) }) + +function makeRetained(paneKey: string): RetainedAgentEntry { + return { + entry: { + state: 'done', + prompt: 'retained run', + updatedAt: 5_000, + stateStartedAt: 5_000, + paneKey, + stateHistory: [], + agentType: 'claude' + }, + worktreeId: 'wt-1', + tab: makeTab(), + agentType: 'claude', + startedAt: 5_000 + } +} + +describe('isClearableActivityThread', () => { + it('clears only completed and interrupted threads', () => { + expect(isClearableActivityThread(doneThread)).toBe(true) + expect(isClearableActivityThread(interruptedThread)).toBe(true) + expect(isClearableActivityThread(workingThread)).toBe(false) + expect(isClearableActivityThread(blockedThread)).toBe(false) + expect(isClearableActivityThread(waitingThread)).toBe(false) + }) +}) + +describe('clearCompletedActivity', () => { + beforeEach(() => { + mockStore.activityClearedAtByPaneKey = {} + mockStore.agentStatusByPaneKey = {} + mockStore.retainedAgentsByPaneKey = { 't-done:1': makeRetained('t-done:1') } + mockStore.retentionSuppressedPaneKeys = {} + vi.stubGlobal('window', { + api: { agentStatus: { dropPersisted: vi.fn(), dropPersistedBatch: vi.fn() } } + }) + }) + + afterEach(() => { + // Drain any eviction left pending by a test that never closed its toast. + flushPendingClearCompletedEvictions() + vi.clearAllMocks() + vi.unstubAllGlobals() + }) + + function lastToastOptions(): { + action: { label: string; onClick: () => void } + onDismiss: () => void + onAutoClose: () => void + } { + return toastSpy.mock.calls.at(-1)?.[1] + } + + it('plans cutoffs and retained removals for completed threads only', () => { + const plan = planClearCompletedActivity( + [workingThread, blockedThread, doneThread, interruptedThread], + mockStore + ) + expect(plan.clearedThreadCount).toBe(2) + expect(plan.cutoffPatch).toEqual({ 't-done:1': 5_000, 't-interrupted:1': 5_000 }) + expect(plan.restorePatch).toEqual({ 't-done:1': null, 't-interrupted:1': null }) + expect(plan.retainedSnapshots.map((r) => r.entry.paneKey)).toEqual(['t-done:1']) + }) + + it('stamps cutoffs, dismisses retained snapshots, and defers the disk drop to toast close', () => { + const cleared = clearCompletedActivity([workingThread, doneThread, interruptedThread]) + expect(cleared).toBe(true) + expect(mockStore.activityClearedAtByPaneKey).toEqual({ + 't-done:1': 5_000, + 't-interrupted:1': 5_000 + }) + expect(mockStore.dismissRetainedAgents).toHaveBeenCalledWith(['t-done:1']) + const drop = ( + window as unknown as { + api: { agentStatus: { dropPersistedBatch: ReturnType } } + } + ).api.agentStatus.dropPersistedBatch + expect(drop).not.toHaveBeenCalled() + + lastToastOptions().onAutoClose() + expect(drop).toHaveBeenCalledTimes(1) + expect(drop).toHaveBeenCalledWith([ + expect.objectContaining({ paneKey: 't-done:1', receivedAt: 5_000, stateStartedAt: 5_000 }) + ]) + // A later dismiss must not double-drop. + lastToastOptions().onDismiss() + expect(drop).toHaveBeenCalledTimes(1) + }) + + it('undo restores prior cutoffs and re-retains snapshots, and skips the disk drop', () => { + mockStore.activityClearedAtByPaneKey = { 't-done:1': 1_111 } + clearCompletedActivity([doneThread, interruptedThread]) + expect(mockStore.activityClearedAtByPaneKey).toEqual({ + 't-done:1': 5_000, + 't-interrupted:1': 5_000 + }) + expect(mockStore.retainedAgentsByPaneKey['t-done:1']).toBeUndefined() + + lastToastOptions().action.onClick() + expect(mockStore.activityClearedAtByPaneKey).toEqual({ 't-done:1': 1_111 }) + expect(mockStore.retainedAgentsByPaneKey['t-done:1']).toBeDefined() + + lastToastOptions().onAutoClose() + const drop = ( + window as unknown as { + api: { agentStatus: { dropPersistedBatch: ReturnType } } + } + ).api.agentStatus.dropPersistedBatch + expect(drop).not.toHaveBeenCalled() + }) + + it('undo restores a completed live row and removes the suppressor created by clear', () => { + const retained = mockStore.retainedAgentsByPaneKey['t-done:1'] + mockStore.agentStatusByPaneKey['t-done:1'] = retained.entry + mockStore.activityClearedAtByPaneKey = { 't-done:1': 1_111 } + + clearCompletedActivity([doneThread]) + expect(mockStore.activityClearedAtByPaneKey['t-done:1']).toBe(5_000) + expect(mockStore.retentionSuppressedPaneKeys['t-done:1']).toBe(true) + + lastToastOptions().action.onClick() + + expect(mockStore.activityClearedAtByPaneKey['t-done:1']).toBe(1_111) + expect(mockStore.retentionSuppressedPaneKeys['t-done:1']).toBeUndefined() + expect(mockStore.retainedAgentsByPaneKey['t-done:1']).toBeUndefined() + }) + + it('undo removes the suppressor even after an identity-only live entry replacement', () => { + // A runtime orchestration merge replaces the live entry object without a state + // change; the suppressor undo must key on the turn, not on object identity. + const retained = mockStore.retainedAgentsByPaneKey['t-done:1'] + mockStore.agentStatusByPaneKey['t-done:1'] = retained.entry + + clearCompletedActivity([doneThread]) + expect(mockStore.retentionSuppressedPaneKeys['t-done:1']).toBe(true) + mockStore.agentStatusByPaneKey['t-done:1'] = { ...retained.entry } + + lastToastOptions().action.onClick() + + expect(mockStore.retentionSuppressedPaneKeys['t-done:1']).toBeUndefined() + }) + + it('undo keeps the suppressor when the live row has moved to a new turn', () => { + const retained = mockStore.retainedAgentsByPaneKey['t-done:1'] + mockStore.agentStatusByPaneKey['t-done:1'] = retained.entry + + clearCompletedActivity([doneThread]) + mockStore.agentStatusByPaneKey['t-done:1'] = { + ...retained.entry, + stateStartedAt: retained.entry.stateStartedAt + 1 + } + + lastToastOptions().action.onClick() + + expect(mockStore.retentionSuppressedPaneKeys['t-done:1']).toBe(true) + }) + + it('does not restore a cleared snapshot over a newer retained run', () => { + clearCompletedActivity([doneThread]) + const newer = makeRetained('t-done:1') + newer.entry.prompt = 'newer run' + mockStore.retainedAgentsByPaneKey['t-done:1'] = newer + + lastToastOptions().action.onClick() + + expect(mockStore.retainedAgentsByPaneKey['t-done:1']).toBe(newer) + expect(mockStore.activityClearedAtByPaneKey['t-done:1']).toBeUndefined() + }) + + it('stamps a real cutoff for a thread with no usable timestamp', () => { + // A zero cutoff is dropped by the hydrate sanitizer and the clear would replay after restart. + const unstamped = makeThread('t-done:1', { latestEvent: doneEvent(false), latestTimestamp: 0 }) + const plan = planClearCompletedActivity([unstamped], mockStore, 42_000) + expect(plan.cutoffPatch).toEqual({ 't-done:1': 42_000 }) + }) + + it('evicts after the fallback window when no toast close callback ever fires', () => { + vi.useFakeTimers() + try { + clearCompletedActivity([doneThread]) + const drop = ( + window as unknown as { + api: { agentStatus: { dropPersistedBatch: ReturnType } } + } + ).api.agentStatus.dropPersistedBatch + expect(drop).not.toHaveBeenCalled() + + vi.advanceTimersByTime(CLEAR_COMPLETED_EVICTION_FALLBACK_MS) + expect(drop).toHaveBeenCalledTimes(1) + + lastToastOptions().onDismiss() + expect(drop).toHaveBeenCalledTimes(1) + } finally { + vi.useRealTimers() + } + }) + + it('undo cancels the fallback eviction timer', () => { + vi.useFakeTimers() + try { + clearCompletedActivity([doneThread]) + lastToastOptions().action.onClick() + vi.advanceTimersByTime(CLEAR_COMPLETED_EVICTION_FALLBACK_MS) + const drop = ( + window as unknown as { + api: { agentStatus: { dropPersistedBatch: ReturnType } } + } + ).api.agentStatus.dropPersistedBatch + expect(drop).not.toHaveBeenCalled() + } finally { + vi.useRealTimers() + } + }) + + it('falls back to per-identity drops when the batch API is absent', () => { + const dropOne = vi.fn() + vi.stubGlobal('window', { api: { agentStatus: { dropPersisted: dropOne } } }) + clearCompletedActivity([doneThread]) + lastToastOptions().onAutoClose() + expect(dropOne).toHaveBeenCalledTimes(1) + }) + + it('does nothing when no thread is clearable', () => { + expect(clearCompletedActivity([workingThread, blockedThread])).toBe(false) + expect(toastSpy).not.toHaveBeenCalled() + expect(mockStore.applyActivityClearedAt).not.toHaveBeenCalled() + }) + + it('pagehide flush evicts a clear whose undo toast is still open', () => { + clearCompletedActivity([doneThread]) + const drop = ( + window as unknown as { + api: { agentStatus: { dropPersistedBatch: ReturnType } } + } + ).api.agentStatus.dropPersistedBatch + expect(drop).not.toHaveBeenCalled() + + // Quit/reload path: the toast's close callbacks never fire. + flushPendingClearCompletedEvictions() + expect(drop).toHaveBeenCalledTimes(1) + + // The flushed eviction is consumed; later toast close must not double-drop. + lastToastOptions().onAutoClose() + expect(drop).toHaveBeenCalledTimes(1) + }) + + it('pagehide flush skips a clear that was undone', () => { + clearCompletedActivity([doneThread]) + lastToastOptions().action.onClick() + flushPendingClearCompletedEvictions() + const drop = ( + window as unknown as { + api: { agentStatus: { dropPersistedBatch: ReturnType } } + } + ).api.agentStatus.dropPersistedBatch + expect(drop).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/activity/activity-clear-completed.ts b/src/renderer/src/components/activity/activity-clear-completed.ts new file mode 100644 index 00000000000..72eee3ea4af --- /dev/null +++ b/src/renderer/src/components/activity/activity-clear-completed.ts @@ -0,0 +1,206 @@ +import { toast } from 'sonner' +import { useAppStore } from '@/store' +import { translate } from '@/i18n/i18n' +import type { RetainedAgentEntry } from '@/store/slices/agent-status' +import type { AgentStatusCacheIdentity } from '../../../../shared/agent-status-types' +import { threadStatusGroupId } from './activity-thread-grouping' +import type { AgentPaneThread } from './activity-thread-types' + +export type ClearCompletedActivityPlan = { + /** Panes whose activity gets a cleared-at cutoff stamped. */ + cutoffPatch: Record + /** Exact prior cutoff values (or null when absent) so undo restores byte-for-byte. */ + restorePatch: Record + /** Retained snapshots removed by the clear; undo re-retains them verbatim. */ + retainedSnapshots: RetainedAgentEntry[] + /** Exact status identities cleared so deferred disk eviction cannot remove a later run. */ + cacheIdentities: AgentStatusCacheIdentity[] + clearedThreadCount: number +} + +/** A thread is clearable when it needs nothing from the user: completed or interrupted, + * with no fresh live working/monitoring/blocked/waiting state. */ +export function isClearableActivityThread(thread: AgentPaneThread): boolean { + const groupId = threadStatusGroupId(thread) + return groupId === 'done' || groupId === 'interrupted' +} + +export function planClearCompletedActivity( + threads: readonly AgentPaneThread[], + state: { + activityClearedAtByPaneKey: Record + retainedAgentsByPaneKey: Record + }, + now: number = Date.now() +): ClearCompletedActivityPlan { + const cutoffPatch: Record = {} + const restorePatch: Record = {} + const retainedSnapshots: RetainedAgentEntry[] = [] + const cacheIdentities: AgentStatusCacheIdentity[] = [] + let clearedThreadCount = 0 + for (const thread of threads) { + if (!isClearableActivityThread(thread)) { + continue + } + clearedThreadCount += 1 + const previousCutoff = state.activityClearedAtByPaneKey[thread.paneKey] ?? null + const latestCutoff = Math.max(previousCutoff ?? 0, thread.latestTimestamp) + // Why `now` for an unstamped thread: the hydrate sanitizer drops non-positive cutoffs, so a + // zero cutoff would replay the cleared thread after restart. + cutoffPatch[thread.paneKey] = latestCutoff > 0 ? latestCutoff : now + restorePatch[thread.paneKey] = previousCutoff + const retained = state.retainedAgentsByPaneKey[thread.paneKey] + if (retained) { + retainedSnapshots.push(retained) + const entry = retained.entry + // updatedAt mirrors the wire receivedAt; renderer-enriched fields (connectionId, + // worktreeId) diverge from main's cache and are deliberately excluded. + cacheIdentities.push({ + paneKey: thread.paneKey, + receivedAt: entry.updatedAt, + stateStartedAt: entry.stateStartedAt + }) + } + } + return { cutoffPatch, restorePatch, retainedSnapshots, cacheIdentities, clearedThreadCount } +} + +// Deferred evictions whose undo toast is still open; flushed on pagehide because the toast's +// close callbacks never fire on quit/reload, which would let cleared rows replay next launch. +const pendingDiskEvictions = new Set<() => void>() +export function flushPendingClearCompletedEvictions(): void { + // Set iteration tolerates the self-delete each evict() performs. + for (const evict of pendingDiskEvictions) { + evict() + } +} +if (typeof window !== 'undefined') { + window.addEventListener('pagehide', flushPendingClearCompletedEvictions) +} + +// Why a fallback: sonner only fires onDismiss/onAutoClose for the toast's own close paths; a +// `toast.dismiss()` from another caller leaves the eviction pending until pagehide. +export const CLEAR_COMPLETED_EVICTION_FALLBACK_MS = 60_000 + +function evictPersistedStatuses(identities: readonly AgentStatusCacheIdentity[]): void { + const api = window.api?.agentStatus + if (!api || identities.length === 0) { + return + } + if (api.dropPersistedBatch) { + api.dropPersistedBatch(identities) + return + } + for (const identity of identities) { + api.dropPersisted?.(identity) + } +} + +/** + * Clear completed/interrupted activity threads with an undo window. + * + * Live agent status, resume identity, and attention/working rows are untouched: + * clearing stamps per-pane cutoffs (persisted UI) and removes retained completed + * snapshots. The identity-checked main-process cache eviction is deferred until + * the undo toast closes so Undo can restore everything losslessly. + */ +export function clearCompletedActivity(threads: readonly AgentPaneThread[]): boolean { + const state = useAppStore.getState() + const plan = planClearCompletedActivity(threads, state) + if (plan.clearedThreadCount === 0) { + return false + } + state.applyActivityClearedAt(plan.cutoffPatch) + // Why turn timestamps, not entry identity: a runtime orchestration merge replaces the live + // entry object without a state change (setRuntimeAgentOrchestrationByPaneKey), and an + // identity check would then strand the clear-planted suppressor past Undo, losing the run. + const introducedSuppressorLiveTurns = new Map( + plan.retainedSnapshots.flatMap((retained) => { + const paneKey = retained.entry.paneKey + const liveEntry = state.agentStatusByPaneKey[paneKey] + return liveEntry && !state.retentionSuppressedPaneKeys[paneKey] + ? ([[paneKey, liveEntry.stateStartedAt]] as const) + : [] + }) + ) + state.dismissRetainedAgents(plan.retainedSnapshots.map((retained) => retained.entry.paneKey)) + + let undone = false + let dropped = false + let fallbackTimer: ReturnType | null = null + const dropRetainedFromDiskCache = (): void => { + pendingDiskEvictions.delete(dropRetainedFromDiskCache) + if (fallbackTimer !== null) { + clearTimeout(fallbackTimer) + fallbackTimer = null + } + if (undone || dropped) { + return + } + dropped = true + evictPersistedStatuses(plan.cacheIdentities) + } + pendingDiskEvictions.add(dropRetainedFromDiskCache) + fallbackTimer = setTimeout(dropRetainedFromDiskCache, CLEAR_COMPLETED_EVICTION_FALLBACK_MS) + toast( + plan.clearedThreadCount === 1 + ? translate('auto.components.activity.clearCompleted.clearedOne', 'Cleared 1 completed agent') + : translate( + 'auto.components.activity.clearCompleted.clearedMany', + 'Cleared {{count}} completed agents', + { count: plan.clearedThreadCount } + ), + { + action: { + label: translate('auto.components.activity.clearCompleted.undo', 'Undo'), + onClick: () => { + undone = true + pendingDiskEvictions.delete(dropRetainedFromDiskCache) + if (fallbackTimer !== null) { + clearTimeout(fallbackTimer) + fallbackTimer = null + } + const current = useAppStore.getState() + const retainedByPaneKey = new Map( + plan.retainedSnapshots.map((retained) => [retained.entry.paneKey, retained]) + ) + const restorePatch: Record = {} + const snapshotsToRestore: RetainedAgentEntry[] = [] + const suppressorPaneKeysToClear: string[] = [] + for (const paneKey of Object.keys(plan.restorePatch)) { + const currentLive = current.agentStatusByPaneKey?.[paneKey] + const currentRetained = current.retainedAgentsByPaneKey[paneKey] + const clearedSnapshot = retainedByPaneKey.get(paneKey) + const cutoffStillOwned = + current.activityClearedAtByPaneKey[paneKey] === plan.cutoffPatch[paneKey] + if (cutoffStillOwned) { + restorePatch[paneKey] = plan.restorePatch[paneKey] ?? null + } + if ( + cutoffStillOwned && + introducedSuppressorLiveTurns.has(paneKey) && + currentLive?.stateStartedAt === introducedSuppressorLiveTurns.get(paneKey) && + current.retentionSuppressedPaneKeys[paneKey] + ) { + suppressorPaneKeysToClear.push(paneKey) + } + if (currentLive || (currentRetained && currentRetained !== clearedSnapshot)) { + continue + } + if (clearedSnapshot && !currentRetained) { + snapshotsToRestore.push(clearedSnapshot) + } + } + current.applyActivityClearedAt(restorePatch) + current.clearRetentionSuppressedPaneKeys(suppressorPaneKeysToClear) + if (snapshotsToRestore.length > 0) { + current.retainAgents(snapshotsToRestore) + } + } + }, + onDismiss: dropRetainedFromDiskCache, + onAutoClose: dropRetainedFromDiskCache + } + ) + return true +} diff --git a/src/renderer/src/components/activity/activity-event-build-cache.ts b/src/renderer/src/components/activity/activity-event-build-cache.ts new file mode 100644 index 00000000000..09f713c5170 --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-build-cache.ts @@ -0,0 +1,129 @@ +import { entryWithRuntimeOrchestration } from '../sidebar/worktree-agent-row-orchestration' +import type { + AgentStatusEntry, + AgentStatusOrchestrationContext, + AgentType +} from '../../../../shared/agent-status-types' +import type { Repo } from '../../../../shared/repo-types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type { + ActivityEvent, + ActivityLiveAgentSnapshot, + ActivityLiveAgentState +} from './activity-thread-types' +import { buildPaneActivityEvents } from './activity-pane-events' + +type PaneActivityCacheEntry = { + source: unknown + orchestration: AgentStatusOrchestrationContext | undefined + acknowledgedAt: number + clearedAt: number + worktree: Worktree + repo: Repo | null + tab: TerminalTab + events: ActivityEvent[] + live: ActivityLiveAgentSnapshot | null + rowEntry: AgentStatusEntry +} + +export type ActivityEventBuildCache = { + panes: Map +} + +export function createActivityEventBuildCache(): ActivityEventBuildCache { + return { panes: new Map() } +} + +export type PaneBuildRequest = { + cacheKey: string + source: unknown + entry: AgentStatusEntry + orchestration: AgentStatusOrchestrationContext | undefined + worktree: Worktree + repo: Repo | null + tab: TerminalTab + agentType: AgentType + agentAlive: boolean + acknowledgedAt: number + clearedAt: number + migrationUnsupportedPtyId?: string + liveState: ActivityLiveAgentState | null +} + +export function resolvePaneBuild( + request: PaneBuildRequest, + cache: ActivityEventBuildCache | undefined, + seenCacheKeys: Set | null +): { events: ActivityEvent[]; live: ActivityLiveAgentSnapshot | null } { + seenCacheKeys?.add(request.cacheKey) + const cached = cache?.panes.get(request.cacheKey) + const inputsUnchanged = + cached !== undefined && + cached.source === request.source && + cached.orchestration === request.orchestration && + cached.acknowledgedAt === request.acknowledgedAt && + cached.clearedAt === request.clearedAt && + cached.worktree === request.worktree && + cached.repo === request.repo && + cached.tab === request.tab + const rowEntry = inputsUnchanged + ? cached.rowEntry + : entryWithRuntimeOrchestration( + request.entry, + request.orchestration ? { [request.entry.paneKey]: request.orchestration } : undefined + ) + + const liveTimestamp = rowEntry.stateStartedAt + const liveMatchesCache = + inputsUnchanged && + (request.liveState === null + ? cached.live === null + : cached.live !== null && + cached.live.state === request.liveState && + cached.live.timestamp === liveTimestamp) + + if (inputsUnchanged && liveMatchesCache) { + return { events: cached.events, live: cached.live } + } + + const events = inputsUnchanged + ? cached.events + : buildPaneActivityEvents({ + entry: rowEntry, + worktree: request.worktree, + repo: request.repo, + tab: request.tab, + agentType: request.agentType, + agentAlive: request.agentAlive, + acknowledgedAt: request.acknowledgedAt, + clearedAt: request.clearedAt, + migrationUnsupportedPtyId: request.migrationUnsupportedPtyId + }) + const live: ActivityLiveAgentSnapshot | null = + request.liveState === null + ? null + : { + state: request.liveState, + timestamp: liveTimestamp, + worktree: request.worktree, + repo: request.repo, + entry: rowEntry, + tab: request.tab, + agentType: request.agentType + } + + cache?.panes.set(request.cacheKey, { + source: request.source, + orchestration: request.orchestration, + acknowledgedAt: request.acknowledgedAt, + clearedAt: request.clearedAt, + worktree: request.worktree, + repo: request.repo, + tab: request.tab, + events, + live, + rowEntry + }) + return { events, live } +} diff --git a/src/renderer/src/components/activity/activity-event-builder-agent-context.test.ts b/src/renderer/src/components/activity/activity-event-builder-agent-context.test.ts new file mode 100644 index 00000000000..4f876eb5824 --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-builder-agent-context.test.ts @@ -0,0 +1,165 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import type { Tab } from '../../../../shared/tab-types' +import { + makeRepo, + makeWorkingEntryWithoutHistory, + makeWorktree, + PANE_KEY +} from './ActivityPrototypePage-test-fixtures' +import { buildActivityEvents } from './activity-event-builder' + +function build(args: { + entry: AgentStatusEntry + unifiedTabs?: Tab[] +}): ReturnType { + const repo = makeRepo() + const worktree = makeWorktree() + return buildActivityEvents({ + agentStatusByPaneKey: { [PANE_KEY]: args.entry }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { [worktree.id]: [] }, + unifiedTabsByWorktree: { [worktree.id]: args.unifiedTabs ?? [] }, + worktreeMap: new Map([[worktree.id, worktree]]), + repoMap: new Map([[repo.id, repo]]), + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) +} + +describe('activity event agent contexts', () => { + it('builds a live thread context from a unified structured-agent tab', () => { + const structuredTab = { + id: 'tab-1', + entityId: 'session-1', + groupId: 'group-1', + worktreeId: 'wt-1', + executionHostId: 'local', + contentType: 'agent-session', + label: 'Codex chat', + customLabel: null, + color: null, + sortOrder: 0, + createdAt: 1, + agentSessionAgent: 'codex' + } satisfies Tab + + const result = build({ + entry: makeWorkingEntryWithoutHistory(), + unifiedTabs: [structuredTab] + }) + + expect(result.liveAgentByPaneKey[PANE_KEY]).toMatchObject({ + state: 'working', + worktree: { id: 'wt-1' }, + tab: { id: 'tab-1', ptyId: null, title: 'Codex chat' } + }) + }) + + it('uses direct worktree attribution before an agent tab reaches the renderer', () => { + const result = build({ + entry: { + ...makeWorkingEntryWithoutHistory(), + worktreeId: 'wt-1' + } + }) + + expect(result.liveAgentByPaneKey[PANE_KEY]).toMatchObject({ + state: 'working', + worktree: { id: 'wt-1' }, + tab: { id: 'tab-1', worktreeId: 'wt-1', ptyId: null } + }) + }) + + it('preserves a unified structured session remote-runtime owner', () => { + const localRepo = makeRepo() + const runtimeRepo = { + ...makeRepo(), + executionHostId: 'runtime:env-1' as const, + displayName: 'Runtime repo' + } + const localWorktree = makeWorktree() + const runtimeWorktree = { + ...makeWorktree(), + hostId: 'runtime:env-1' as const, + runtimeOwnerEnvironmentId: 'env-1', + displayName: 'Runtime worktree' + } + const structuredTab = { + id: 'tab-1', + entityId: 'session-1', + groupId: 'group-1', + worktreeId: 'wt-1', + executionHostId: 'runtime:env-1', + contentType: 'agent-session', + label: 'Remote Codex chat', + customLabel: null, + color: null, + sortOrder: 0, + createdAt: 1, + agentSessionAgent: 'codex' + } satisfies Tab + const resolveWorktree = vi.fn((_worktreeId, executionHostId) => + executionHostId === 'runtime:env-1' ? runtimeWorktree : localWorktree + ) + + const result = buildActivityEvents({ + agentStatusByPaneKey: { + [PANE_KEY]: { ...makeWorkingEntryWithoutHistory(), connectionId: null } + }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { [localWorktree.id]: [] }, + unifiedTabsByWorktree: { [localWorktree.id]: [structuredTab] }, + worktreeMap: new Map([[localWorktree.id, localWorktree]]), + repoMap: new Map([[localRepo.id, localRepo]]), + repos: [localRepo, runtimeRepo], + resolveWorktree, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(resolveWorktree).toHaveBeenCalledWith('wt-1', 'runtime:env-1') + expect(result.liveAgentByPaneKey[PANE_KEY]?.worktree).toBe(runtimeWorktree) + expect(result.liveAgentByPaneKey[PANE_KEY]?.repo).toBe(runtimeRepo) + }) + + it('preserves an early worktree-attributed SSH owner before its tab arrives', () => { + const localRepo = makeRepo() + const remoteRepo = { + ...makeRepo(), + connectionId: 'builder', + displayName: 'SSH repo' + } + const localWorktree = makeWorktree() + const remoteWorktree = { + ...makeWorktree(), + hostId: 'ssh:builder' as const, + displayName: 'SSH worktree' + } + const resolveWorktree = vi.fn((_worktreeId, executionHostId) => + executionHostId === 'ssh:builder' ? remoteWorktree : localWorktree + ) + + const result = buildActivityEvents({ + agentStatusByPaneKey: { + [PANE_KEY]: { + ...makeWorkingEntryWithoutHistory(), + worktreeId: 'wt-1', + connectionId: 'builder' + } + }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { [localWorktree.id]: [] }, + worktreeMap: new Map([[localWorktree.id, localWorktree]]), + repoMap: new Map([[localRepo.id, localRepo]]), + repos: [localRepo, remoteRepo], + resolveWorktree, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(resolveWorktree).toHaveBeenCalledWith('wt-1', 'ssh:builder') + expect(result.liveAgentByPaneKey[PANE_KEY]?.worktree).toBe(remoteWorktree) + expect(result.liveAgentByPaneKey[PANE_KEY]?.repo).toBe(remoteRepo) + }) +}) diff --git a/src/renderer/src/components/activity/activity-event-builder-context.ts b/src/renderer/src/components/activity/activity-event-builder-context.ts new file mode 100644 index 00000000000..3fdd2584fef --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-builder-context.ts @@ -0,0 +1,182 @@ +import { findIndexedRepoOwnerForHost } from '@/lib/worktree-runtime-owner-index' +import { getRemoteRuntimePtyEnvironmentId } from '@/runtime/runtime-terminal-stream' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { + LOCAL_EXECUTION_HOST_ID, + toRuntimeExecutionHostId, + toSshExecutionHostId, + type ExecutionHostId +} from '../../../../shared/execution-host' +import type { Repo } from '../../../../shared/repo-types' +import { parseAppSshPtyId } from '../../../../shared/ssh-pty-id' +import type { Tab } from '../../../../shared/tab-types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import { + effectiveWorktreeAgentRowStartedAt, + tabFromWorktreeAttributedStatusEntry +} from '../sidebar/worktree-agent-row-fallback-tab' +import type { BuildActivityEventsArgs } from './activity-event-builder' +import { standaloneActivityWorktree } from './activity-standalone-worktree' + +export type ActivityTabContext = { worktreeId: string; tab: TerminalTab } +export type ActivityEventOwner = { worktree: Worktree; repo: Repo | null; knownWorktree: boolean } +export type ActivityTabHostIndex = Map> + +// Why memoized on the source object: the pane build cache compares `tab` by identity, so a +// fresh derived object per rebuild would miss the cache for every agent-session and +// missing-tab row. Upstream keeps the source identity stable while its fields are unchanged. +const agentSessionTerminalTabs = new WeakMap() +const attributedTabContexts = new WeakMap() + +function terminalTabFromAgentSessionTab(tab: Tab): TerminalTab { + const cached = agentSessionTerminalTabs.get(tab) + if (cached) { + return cached + } + const derived: TerminalTab = { + id: tab.id, + ptyId: null, + worktreeId: tab.worktreeId, + title: tab.customLabel ?? tab.generatedLabel ?? tab.label, + customTitle: tab.customLabel, + color: tab.color, + isPinned: tab.isPinned, + sortOrder: tab.sortOrder, + createdAt: tab.createdAt + } + agentSessionTerminalTabs.set(tab, derived) + return derived +} + +export function buildActivityTabContext( + tabsByWorktree: Record, + unifiedTabsByWorktree?: Record +): Map { + const contexts = new Map() + for (const [worktreeId, tabs] of Object.entries(tabsByWorktree)) { + for (const tab of tabs) { + contexts.set(tab.id, { worktreeId, tab }) + } + } + for (const [worktreeId, tabs] of Object.entries(unifiedTabsByWorktree ?? {})) { + for (const tab of tabs) { + if (tab.contentType !== 'agent-session' || contexts.has(tab.id)) { + continue + } + contexts.set(tab.id, { worktreeId, tab: terminalTabFromAgentSessionTab(tab) }) + } + } + return contexts +} + +export function attributedActivityTabContext(entry: AgentStatusEntry): ActivityTabContext | null { + const cached = attributedTabContexts.get(entry) + if (cached !== undefined) { + return cached + } + const tab = tabFromWorktreeAttributedStatusEntry(entry, effectiveWorktreeAgentRowStartedAt(entry)) + const context = tab ? { worktreeId: tab.worktreeId, tab } : null + attributedTabContexts.set(entry, context) + return context +} + +export function buildActivityTabHostIndex( + unifiedTabsByWorktree?: Record +): ActivityTabHostIndex { + const index: ActivityTabHostIndex = new Map() + for (const [worktreeId, tabs] of Object.entries(unifiedTabsByWorktree ?? {})) { + for (const tab of tabs) { + if ( + (tab.contentType !== 'terminal' && tab.contentType !== 'agent-session') || + !tab.executionHostId + ) { + continue + } + let byTabId = index.get(worktreeId) + if (!byTabId) { + byTabId = new Map() + index.set(worktreeId, byTabId) + } + const contextTabId = tab.contentType === 'terminal' ? tab.entityId : tab.id + const existing = byTabId.get(contextTabId) + byTabId.set( + contextTabId, + existing === undefined || existing === tab.executionHostId ? tab.executionHostId : null + ) + } + } + return index +} + +function resolveActivityExecutionHostId( + context: ActivityTabContext, + entry: AgentStatusEntry, + terminalPtyId: string | null | undefined, + tabHostIndex: ActivityTabHostIndex +): ExecutionHostId | undefined { + const tabHostId = tabHostIndex.get(context.worktreeId)?.get(context.tab.id) + if (tabHostId) { + return tabHostId + } + // Why before connectionId: a runtime pane's status entry publishes connectionId: null, + // which would otherwise resolve to LOCAL (see dashboard-card-terminal-input's precedent). + const runtimeEnvironmentId = getRemoteRuntimePtyEnvironmentId(terminalPtyId ?? '') + if (runtimeEnvironmentId) { + return toRuntimeExecutionHostId(runtimeEnvironmentId) + } + if (entry.connectionId !== undefined) { + return entry.connectionId ? toSshExecutionHostId(entry.connectionId) : LOCAL_EXECUTION_HOST_ID + } + const connectionId = parseAppSshPtyId(terminalPtyId ?? '')?.connectionId + return connectionId ? toSshExecutionHostId(connectionId) : undefined +} + +export function resolveActivityEventOwner( + args: BuildActivityEventsArgs, + context: ActivityTabContext, + entry: AgentStatusEntry, + terminalPtyId: string | null | undefined, + tabHostIndex: ActivityTabHostIndex, + ownerCache: Map +): ActivityEventOwner { + const executionHostId = resolveActivityExecutionHostId( + context, + entry, + terminalPtyId, + tabHostIndex + ) + // Why: resolution runs per pane per rebuild and the miss path scans detected worktrees; + // everything below depends only on worktreeId + host, so memoize per build. + const ownerCacheKey = `${context.worktreeId}\0${executionHostId ?? ''}` + const cached = ownerCache.get(ownerCacheKey) + if (cached) { + return cached + } + const resolvedWorktree = args.resolveWorktree?.(context.worktreeId, executionHostId) + const mappedWorktree = args.worktreeMap.get(context.worktreeId) + const worktree = + resolvedWorktree ?? + mappedWorktree ?? + standaloneActivityWorktree(context.worktreeId, executionHostId) + let repo = + executionHostId && args.repos + ? findIndexedRepoOwnerForHost(args.repos, worktree.repoId, executionHostId) + : null + if (!repo && worktree.runtimeOwnerEnvironmentId && args.repos) { + repo = findIndexedRepoOwnerForHost( + args.repos, + worktree.repoId, + toRuntimeExecutionHostId(worktree.runtimeOwnerEnvironmentId) + ) + } + const owner: ActivityEventOwner = { + worktree, + repo: repo ?? args.repoMap.get(worktree.repoId) ?? null, + knownWorktree: Boolean( + resolvedWorktree || mappedWorktree || args.tabsByWorktree[context.worktreeId] + ) + } + ownerCache.set(ownerCacheKey, owner) + return owner +} diff --git a/src/renderer/src/components/activity/activity-event-builder-sources.ts b/src/renderer/src/components/activity/activity-event-builder-sources.ts new file mode 100644 index 00000000000..1049ed95acf --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-builder-sources.ts @@ -0,0 +1,105 @@ +import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' +import { parsePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import type { Repo } from '../../../../shared/repo-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type { ActivityLiveAgentSnapshot, ActivityEvent } from './activity-thread-types' +import type { ActivityEventBuildCache } from './activity-event-build-cache' +import { resolvePaneBuild } from './activity-event-build-cache' +import type { BuildActivityEventsArgs } from './activity-event-builder' + +export function appendUnsupportedAndRetainedEvents(context: { + args: BuildActivityEventsArgs + cache: ActivityEventBuildCache | undefined + seenCacheKeys: Set | null + liveAgentByPaneKey: Record + tabContext: Map + resolveOwner: ( + context: { worktreeId: string; tab: TerminalTab }, + entry: AgentStatusEntry, + terminalPtyId?: string | null + ) => { worktree: Worktree; repo: Repo | null; knownWorktree: boolean } + pushPaneEvents: (paneEvents: ActivityEvent[]) => void +}): void { + const { + args, + cache, + seenCacheKeys, + liveAgentByPaneKey, + tabContext, + resolveOwner, + pushPaneEvents + } = context + + for (const unsupported of Object.values(args.migrationUnsupportedByPtyId ?? {})) { + const cacheKey = `unsupported:${unsupported.paneKey ?? unsupported.ptyId}` + const cached = cache?.panes.get(cacheKey) + const entry = + cached?.source === unsupported + ? cached.rowEntry + : migrationUnsupportedToAgentStatusEntry(unsupported) + const parsed = entry ? parsePaneKey(entry.paneKey) : null + const tabEntry = parsed ? tabContext.get(parsed.tabId) : null + if (!entry || !tabEntry) { + continue + } + const owner = resolveOwner(tabEntry, entry, unsupported.ptyId) + const { events: paneEvents, live } = resolvePaneBuild( + { + cacheKey, + source: unsupported, + entry, + orchestration: undefined, + worktree: owner.worktree, + repo: owner.repo, + tab: tabEntry.tab, + agentType: entry.agentType ?? 'unknown', + agentAlive: false, + acknowledgedAt: args.acknowledgedAgentsByPaneKey[entry.paneKey] ?? 0, + clearedAt: args.activityClearedAtByPaneKey?.[entry.paneKey] ?? 0, + migrationUnsupportedPtyId: unsupported.ptyId, + liveState: 'blocked' + }, + cache, + seenCacheKeys + ) + if (live) { + liveAgentByPaneKey[entry.paneKey] = live + } + pushPaneEvents(paneEvents) + } + + for (const [paneKey, retained] of Object.entries(args.retainedAgentsByPaneKey)) { + if (!parsePaneKey(paneKey)) { + continue + } + const owner = resolveOwner( + { worktreeId: retained.worktreeId, tab: retained.tab }, + retained.entry, + retained.tab.ptyId ?? retained.entry.terminalHandle + ) + if (!owner.knownWorktree) { + continue + } + const { events: paneEvents } = resolvePaneBuild( + { + cacheKey: `retained:${paneKey}`, + source: retained, + entry: retained.entry, + orchestration: args.runtimeAgentOrchestrationByPaneKey?.[paneKey], + worktree: owner.worktree, + repo: owner.repo, + tab: retained.tab, + agentType: retained.agentType, + agentAlive: false, + acknowledgedAt: args.acknowledgedAgentsByPaneKey[paneKey] ?? 0, + clearedAt: args.activityClearedAtByPaneKey?.[paneKey] ?? 0, + liveState: null + }, + cache, + seenCacheKeys + ) + pushPaneEvents(paneEvents) + } +} diff --git a/src/renderer/src/components/activity/activity-event-builder.bounded-history.test.ts b/src/renderer/src/components/activity/activity-event-builder.bounded-history.test.ts new file mode 100644 index 00000000000..c6350555584 --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-builder.bounded-history.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it } from 'vitest' +import type { + AgentStateHistoryEntry, + AgentStatusEntry +} from '../../../../shared/agent-status-types' +import { buildActivityEvents, newestActivityHistoryEntries } from './activity-event-builder' +import { EVENTS_PER_PANE_CAP } from './activity-event-cap' +import { makeRepo, makeTab, makeWorktree, PANE_KEY } from './ActivityPrototypePage-test-fixtures' + +function historyEntry( + startedAt: number, + state: AgentStateHistoryEntry['state'] +): AgentStateHistoryEntry { + return { state, prompt: `prompt-${startedAt}`, startedAt } +} + +function build(args: { + entries?: Record + activityClearedAtByPaneKey?: Record + now?: number +}) { + const repo = makeRepo() + const worktree = makeWorktree() + const tab = makeTab() + return buildActivityEvents({ + agentStatusByPaneKey: args.entries ?? {}, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { [worktree.id]: [tab] }, + worktreeMap: new Map([[worktree.id, worktree]]), + repoMap: new Map([[repo.id, repo]]), + acknowledgedAgentsByPaneKey: {}, + activityClearedAtByPaneKey: args.activityClearedAtByPaneKey, + now: args.now ?? 100_000 + }) +} + +describe('newestActivityHistoryEntries', () => { + it('takes only the newest cap-many eligible entries without scanning results past the cap', () => { + const history: AgentStateHistoryEntry[] = [] + for (let i = 0; i < 10_000; i += 1) { + history.push(historyEntry(i + 1, i % 2 === 0 ? 'done' : 'working')) + } + const newest = newestActivityHistoryEntries(history, EVENTS_PER_PANE_CAP) + expect(newest).toHaveLength(EVENTS_PER_PANE_CAP) + // Only done/blocked/waiting are eligible; newest five eligible are the last five even-indexed rows, oldest-first. + expect(newest.map((entry) => entry.startedAt)).toEqual([9991, 9993, 9995, 9997, 9999]) + }) + + it('returns fewer entries when eligible history is short', () => { + const history = [historyEntry(1, 'working'), historyEntry(2, 'done')] + expect( + newestActivityHistoryEntries(history, EVENTS_PER_PANE_CAP).map((e) => e.startedAt) + ).toEqual([2]) + }) +}) + +describe('buildActivityEvents bounded history', () => { + it('produces identical visible events for a pane with unbounded history as the per-pane cap allows', () => { + const longHistory: AgentStateHistoryEntry[] = [] + for (let i = 0; i < 1_000; i += 1) { + longHistory.push(historyEntry(i + 1, 'done')) + } + const entry: AgentStatusEntry = { + state: 'done', + prompt: 'latest', + updatedAt: 5_000, + stateStartedAt: 5_000, + paneKey: PANE_KEY, + stateHistory: longHistory, + agentType: 'claude' + } + const { events } = build({ entries: { [PANE_KEY]: entry } }) + // Per-pane cap holds: newest events only, newest-first ordering preserved. + expect(events).toHaveLength(EVENTS_PER_PANE_CAP) + expect(events.map((event) => event.timestamp)).toEqual([5_000, 1_000, 999, 998, 997]) + }) +}) + +describe('buildActivityEvents cleared cutoff', () => { + const doneEntry: AgentStatusEntry = { + state: 'done', + prompt: 'finish it', + updatedAt: 2_000, + stateStartedAt: 2_000, + paneKey: PANE_KEY, + stateHistory: [historyEntry(1_000, 'done')], + agentType: 'claude' + } + + it('hides events stamped at or before the pane cutoff', () => { + const { events } = build({ + entries: { [PANE_KEY]: doneEntry }, + activityClearedAtByPaneKey: { [PANE_KEY]: 2_000 } + }) + expect(events).toHaveLength(0) + }) + + it('keeps events newer than the cutoff', () => { + const { events } = build({ + entries: { [PANE_KEY]: doneEntry }, + activityClearedAtByPaneKey: { [PANE_KEY]: 1_000 } + }) + expect(events.map((event) => event.timestamp)).toEqual([2_000]) + }) + + it('does not suppress a live working snapshot for a cleared pane', () => { + const workingEntry: AgentStatusEntry = { + ...doneEntry, + state: 'working', + updatedAt: 99_000, + stateStartedAt: 99_000 + } + const { events, liveAgentByPaneKey } = build({ + entries: { [PANE_KEY]: workingEntry }, + activityClearedAtByPaneKey: { [PANE_KEY]: 98_000 }, + now: 99_500 + }) + expect(liveAgentByPaneKey[PANE_KEY]?.state).toBe('working') + // The historical done at 1_000 stays hidden by the cutoff. + expect(events).toHaveLength(0) + }) +}) diff --git a/src/renderer/src/components/activity/activity-event-builder.host-ownership.test.ts b/src/renderer/src/components/activity/activity-event-builder.host-ownership.test.ts new file mode 100644 index 00000000000..51beb6cf1f9 --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-builder.host-ownership.test.ts @@ -0,0 +1,206 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { folderWorkspaceKey } from '../../../../shared/workspace-scope' +import { + LEAF_ID, + makeRepo, + makeRetainedDoneEntry, + makeTab, + makeWorktree +} from './ActivityPrototypePage-test-fixtures' +import { buildActivityEvents } from './activity-event-builder' + +const PANE_KEY = `tab-1:${LEAF_ID}` + +function doneEntry(connectionId: string | null): AgentStatusEntry { + return { + state: 'done', + prompt: 'Finished task', + updatedAt: 2_000, + stateStartedAt: 2_000, + paneKey: PANE_KEY, + tabId: 'tab-1', + connectionId, + stateHistory: [], + agentType: 'claude' + } +} + +describe('activity event host ownership', () => { + it('uses the status transport host when worktree and repo ids collide', () => { + const localRepo = makeRepo() + const remoteRepo = { ...makeRepo(), connectionId: 'builder', displayName: 'Remote repo' } + const localWorktree = makeWorktree() + const remoteWorktree = { + ...makeWorktree(), + hostId: 'ssh:builder' as const, + displayName: 'Remote worktree' + } + const tab = makeTab() + const resolveWorktree = vi.fn((_worktreeId, executionHostId) => + executionHostId === 'ssh:builder' ? remoteWorktree : localWorktree + ) + + const result = buildActivityEvents({ + agentStatusByPaneKey: { [PANE_KEY]: doneEntry('builder') }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { [localWorktree.id]: [tab] }, + worktreeMap: new Map([[localWorktree.id, localWorktree]]), + repoMap: new Map([[localRepo.id, localRepo]]), + repos: [localRepo, remoteRepo], + resolveWorktree, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(resolveWorktree).toHaveBeenCalledWith(localWorktree.id, 'ssh:builder') + expect(result.events[0]?.worktree).toBe(remoteWorktree) + expect(result.events[0]?.repo).toBe(remoteRepo) + }) + + it('uses the mirrored tab host when paired-runtime status is host-local', () => { + const localRepo = makeRepo() + const runtimeRepo = { + ...makeRepo(), + executionHostId: 'runtime:env-1' as const, + displayName: 'Runtime repo' + } + const localWorktree = makeWorktree() + const runtimeWorktree = { + ...makeWorktree(), + hostId: 'runtime:env-1' as const, + runtimeOwnerEnvironmentId: 'env-1', + displayName: 'Runtime worktree' + } + const tab = makeTab() + const resolveWorktree = vi.fn((_worktreeId, executionHostId) => + executionHostId === 'runtime:env-1' ? runtimeWorktree : localWorktree + ) + + const result = buildActivityEvents({ + agentStatusByPaneKey: { [PANE_KEY]: doneEntry(null) }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { [localWorktree.id]: [tab] }, + unifiedTabsByWorktree: { + [localWorktree.id]: [ + { + id: tab.id, + entityId: tab.id, + groupId: 'group-1', + worktreeId: localWorktree.id, + executionHostId: 'runtime:env-1', + contentType: 'terminal', + label: tab.title, + customLabel: null, + color: null, + sortOrder: 0, + createdAt: 1 + } + ] + }, + worktreeMap: new Map([[localWorktree.id, localWorktree]]), + repoMap: new Map([[localRepo.id, localRepo]]), + repos: [localRepo, runtimeRepo], + resolveWorktree, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(resolveWorktree).toHaveBeenCalledWith(localWorktree.id, 'runtime:env-1') + expect(result.events[0]?.worktree).toBe(runtimeWorktree) + expect(result.events[0]?.repo).toBe(runtimeRepo) + }) + + it('keeps retained folder-workspace activity after its terminal tab is gone', () => { + const folderWorktree = { + ...makeWorktree(), + id: folderWorkspaceKey('folder-1'), + repoId: 'folder-workspace:group-1', + hostId: 'local' as const, + displayName: 'Docs folder' + } + const tab = { ...makeTab(), worktreeId: folderWorktree.id } + const retained = makeRetainedDoneEntry(tab) + retained.worktreeId = folderWorktree.id + retained.entry = doneEntry(null) + + const result = buildActivityEvents({ + agentStatusByPaneKey: {}, + retainedAgentsByPaneKey: { [PANE_KEY]: retained }, + tabsByWorktree: {}, + worktreeMap: new Map(), + repoMap: new Map(), + resolveWorktree: (worktreeId, executionHostId) => + worktreeId === folderWorktree.id && executionHostId === 'local' + ? folderWorktree + : undefined, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(result.events[0]?.worktree).toBe(folderWorktree) + expect(result.events[0]?.worktree.displayName).toBe('Docs folder') + }) + + it('carries migrationUnsupportedPtyId on events built for un-migratable panes', () => { + const worktree = makeWorktree() + const repo = makeRepo() + const tab = makeTab() + + const result = buildActivityEvents({ + agentStatusByPaneKey: {}, + retainedAgentsByPaneKey: {}, + migrationUnsupportedByPtyId: { + 'pty-1': { + ptyId: 'pty-1', + paneKey: PANE_KEY, + tabId: tab.id, + reason: 'legacy-numeric-pane-key', + source: 'local', + updatedAt: 1_000 + } + }, + tabsByWorktree: { [worktree.id]: [tab] }, + worktreeMap: new Map([[worktree.id, worktree]]), + repoMap: new Map([[repo.id, repo]]), + resolveWorktree: () => worktree, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(result.events.length).toBeGreaterThan(0) + for (const event of result.events) { + expect(event.migrationUnsupportedPtyId).toBe('pty-1') + } + }) + + it('uses the retained terminal handle to preserve runtime host ownership after teardown', () => { + const localWorktree = makeWorktree() + const runtimeWorktree = { + ...makeWorktree(), + hostId: 'runtime:env-1' as const, + runtimeOwnerEnvironmentId: 'env-1', + displayName: 'Runtime worktree' + } + const tab = { ...makeTab(), ptyId: null } + const retained = makeRetainedDoneEntry(tab) + retained.entry = { ...doneEntry(null), terminalHandle: 'remote:env-1@@pty-1' } + const resolveWorktree = vi.fn((_worktreeId, executionHostId) => + executionHostId === 'runtime:env-1' ? runtimeWorktree : localWorktree + ) + + const result = buildActivityEvents({ + agentStatusByPaneKey: {}, + retainedAgentsByPaneKey: { [PANE_KEY]: retained }, + tabsByWorktree: {}, + worktreeMap: new Map([[localWorktree.id, localWorktree]]), + repoMap: new Map(), + resolveWorktree, + acknowledgedAgentsByPaneKey: {}, + now: 3_000 + }) + + expect(resolveWorktree).toHaveBeenCalledWith(localWorktree.id, 'runtime:env-1') + expect(result.events[0]?.worktree).toBe(runtimeWorktree) + }) +}) diff --git a/src/renderer/src/components/activity/activity-event-builder.identity-reuse.test.ts b/src/renderer/src/components/activity/activity-event-builder.identity-reuse.test.ts new file mode 100644 index 00000000000..1c1a76aff20 --- /dev/null +++ b/src/renderer/src/components/activity/activity-event-builder.identity-reuse.test.ts @@ -0,0 +1,225 @@ +import { describe, expect, it } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import type { RetainedAgentEntry } from '@/store/slices/agent-status' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import { + buildActivityEvents, + createActivityEventBuildCache, + type ActivityEventBuildCache +} from './activity-event-builder' +import { + buildAgentPaneThreads, + createAgentPaneThreadReuseCache, + type AgentPaneThreadReuseCache +} from './activity-thread-builder' +import { + LEAF_ID, + LEAF_ID_2, + makeRepo, + makeTab, + makeTabWithIds, + makeWorktree +} from './ActivityPrototypePage-test-fixtures' + +const PANE_A = makePaneKey('tab-1', LEAF_ID) +const PANE_B = makePaneKey('tab-2', LEAF_ID_2) +const NOW = 100_000 + +function entry(paneKey: string, overrides: Partial = {}): AgentStatusEntry { + return { + state: 'done', + prompt: `run ${paneKey}`, + updatedAt: 50_000, + stateStartedAt: 50_000, + paneKey, + stateHistory: [{ state: 'done', prompt: 'older', startedAt: 10_000 }], + agentType: 'claude', + ...overrides + } +} + +type BuildArgs = Parameters[0] + +function makeArgs(overrides: Partial = {}): BuildArgs { + const repo = makeRepo() + const worktree = makeWorktree() + return { + agentStatusByPaneKey: { + [PANE_A]: entry(PANE_A), + [PANE_B]: entry(PANE_B, { state: 'working', stateStartedAt: NOW - 1_000 }) + }, + retainedAgentsByPaneKey: {}, + tabsByWorktree: { + [worktree.id]: [makeTab(), makeTabWithIds('tab-2', worktree.id)] + }, + worktreeMap: new Map([[worktree.id, worktree]]), + repoMap: new Map([[repo.id, repo]]), + acknowledgedAgentsByPaneKey: {}, + now: NOW, + ...overrides + } +} + +function buildBoth( + args: BuildArgs, + eventCache: ActivityEventBuildCache, + threadCache: AgentPaneThreadReuseCache +) { + const result = buildActivityEvents(args, eventCache) + const threads = buildAgentPaneThreads( + { events: result.events, liveAgentByPaneKey: result.liveAgentByPaneKey }, + threadCache + ) + return { ...result, threads } +} + +function threadByPane(threads: T[], paneKey: string): T | undefined { + return threads.find((thread) => thread.paneKey === paneKey) +} + +describe('activity build identity reuse', () => { + it('returns identical event, snapshot, thread, and list identities for identical inputs', () => { + const eventCache = createActivityEventBuildCache() + const threadCache = createAgentPaneThreadReuseCache() + const args = makeArgs() + const first = buildBoth(args, eventCache, threadCache) + const second = buildBoth(args, eventCache, threadCache) + + expect(second.threads).toBe(first.threads) + expect(second.events.map((event) => event)).toEqual(first.events.map((event) => event)) + for (let i = 0; i < first.events.length; i += 1) { + expect(second.events[i]).toBe(first.events[i]) + } + expect(second.liveAgentByPaneKey[PANE_B]).toBe(first.liveAgentByPaneKey[PANE_B]) + }) + + it('changes only the written pane; every other thread keeps its identity', () => { + const eventCache = createActivityEventBuildCache() + const threadCache = createAgentPaneThreadReuseCache() + const args = makeArgs() + const first = buildBoth(args, eventCache, threadCache) + + const next = makeArgs({ + agentStatusByPaneKey: { + ...args.agentStatusByPaneKey, + [PANE_B]: entry(PANE_B, { + state: 'working', + stateStartedAt: NOW - 1_000, + prompt: 'streamed update' + }) + }, + tabsByWorktree: args.tabsByWorktree, + worktreeMap: args.worktreeMap, + repoMap: args.repoMap + }) + const second = buildBoth(next, eventCache, threadCache) + + expect(threadByPane(second.threads, PANE_A)).toBe(threadByPane(first.threads, PANE_A)) + expect(threadByPane(second.threads, PANE_B)).not.toBe(threadByPane(first.threads, PANE_B)) + expect(second.threads).not.toBe(first.threads) + }) + + it('an acknowledgement or cleared-cutoff change rebuilds only that pane', () => { + const eventCache = createActivityEventBuildCache() + const threadCache = createAgentPaneThreadReuseCache() + const args = makeArgs() + const first = buildBoth(args, eventCache, threadCache) + + const acked = buildBoth( + makeArgs({ + agentStatusByPaneKey: args.agentStatusByPaneKey, + tabsByWorktree: args.tabsByWorktree, + worktreeMap: args.worktreeMap, + repoMap: args.repoMap, + acknowledgedAgentsByPaneKey: { [PANE_A]: NOW } + }), + eventCache, + threadCache + ) + expect(threadByPane(acked.threads, PANE_B)).toBe(threadByPane(first.threads, PANE_B)) + expect(threadByPane(acked.threads, PANE_A)?.unread).toBe(false) + expect(threadByPane(first.threads, PANE_A)?.unread).toBe(true) + + const cleared = buildBoth( + makeArgs({ + agentStatusByPaneKey: args.agentStatusByPaneKey, + tabsByWorktree: args.tabsByWorktree, + worktreeMap: args.worktreeMap, + repoMap: args.repoMap, + acknowledgedAgentsByPaneKey: { [PANE_A]: NOW }, + activityClearedAtByPaneKey: { [PANE_A]: NOW } + }), + eventCache, + threadCache + ) + expect(threadByPane(cleared.threads, PANE_B)).toBe(threadByPane(first.threads, PANE_B)) + expect(threadByPane(cleared.threads, PANE_A)).toBeUndefined() + }) + + it('freshness decay refreshes the live snapshot without churning event identities', () => { + const eventCache = createActivityEventBuildCache() + const threadCache = createAgentPaneThreadReuseCache() + const args = makeArgs() + const first = buildBoth(args, eventCache, threadCache) + expect(first.liveAgentByPaneKey[PANE_B]?.state).toBe('working') + + // Same inputs much later: the working turn is stale now, so the snapshot drops. + const decayed = buildBoth( + makeArgs({ ...args, now: NOW + 60 * 60 * 1000 }), + eventCache, + threadCache + ) + expect(decayed.liveAgentByPaneKey[PANE_B]).toBeUndefined() + // PANE_A had no live snapshot; its thread survives untouched. + expect(threadByPane(decayed.threads, PANE_A)).toBe(threadByPane(first.threads, PANE_A)) + }) + + it('cached builds always equal a cold uncached build (no drift)', () => { + const eventCache = createActivityEventBuildCache() + const threadCache = createAgentPaneThreadReuseCache() + const scenarios: BuildArgs[] = [ + makeArgs(), + makeArgs({ acknowledgedAgentsByPaneKey: { [PANE_A]: NOW } }), + makeArgs({ activityClearedAtByPaneKey: { [PANE_A]: NOW } }), + makeArgs({ + runtimeAgentOrchestrationByPaneKey: { + [PANE_B]: { taskId: 't1', dispatchId: 'd1', parentPaneKey: PANE_A } + } + }), + makeArgs({ now: NOW + 60 * 60 * 1000 }) + ] + for (const scenario of scenarios) { + const cached = buildBoth(scenario, eventCache, threadCache) + const cold = buildActivityEvents(scenario) + const coldThreads = buildAgentPaneThreads({ + events: cold.events, + liveAgentByPaneKey: cold.liveAgentByPaneKey + }) + expect(cached.events).toEqual(cold.events) + expect(cached.liveAgentByPaneKey).toEqual(cold.liveAgentByPaneKey) + expect(cached.threads).toEqual(coldThreads) + } + }) + + it('keeps first-source-wins dedupe when a pane is both live and retained, and evicts gone panes', () => { + const eventCache = createActivityEventBuildCache() + const threadCache = createAgentPaneThreadReuseCache() + const retained: RetainedAgentEntry = { + entry: entry(PANE_A, { prompt: 'retained copy' }), + worktreeId: makeWorktree().id, + tab: makeTab(), + agentType: 'claude', + startedAt: 50_000 + } + const args = makeArgs({ retainedAgentsByPaneKey: { [PANE_A]: retained } }) + const cachedResult = buildBoth(args, eventCache, threadCache) + const cold = buildActivityEvents(args) + expect(cachedResult.events).toEqual(cold.events) + expect(eventCache.panes.has(`retained:${PANE_A}`)).toBe(true) + + // Retained entry dismissed: its cache row must not linger. + buildBoth(makeArgs(), eventCache, threadCache) + expect(eventCache.panes.has(`retained:${PANE_A}`)).toBe(false) + expect(eventCache.panes.has(`live:${PANE_A}`)).toBe(true) + }) +}) diff --git a/src/renderer/src/components/activity/activity-event-builder.ts b/src/renderer/src/components/activity/activity-event-builder.ts index 11e66f65bb2..5e1e3a112ae 100644 --- a/src/renderer/src/components/activity/activity-event-builder.ts +++ b/src/renderer/src/components/activity/activity-event-builder.ts @@ -1,33 +1,41 @@ import { isExplicitAgentStatusFresh } from '@/lib/agent-status' -import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' import type { RetainedAgentEntry } from '@/store/slices/agent-status' import { AGENT_STATUS_STALE_AFTER_MS, - type AgentStateHistoryEntry, type AgentStatusEntry, + type AgentStatusOrchestrationContext, type AgentStatusState, - type AgentType, type MigrationUnsupportedPtyEntry } from '../../../../shared/agent-status-types' -import { FLOATING_TERMINAL_WORKTREE_ID } from '../../../../shared/constants' +import type { ExecutionHostId } from '../../../../shared/execution-host' import type { Repo } from '../../../../shared/repo-types' import { parsePaneKey } from '../../../../shared/stable-pane-id' +import type { Tab } from '../../../../shared/tab-types' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { Worktree } from '../../../../shared/worktree/types' import type { ActivityEvent, - ActivityEventState, ActivityHookLiveAgentState, ActivityLiveAgentSnapshot, ActivityLiveAgentState } from './activity-thread-types' import { capActivityEvents } from './activity-event-cap' +import { newestActivityHistoryEntries } from './activity-pane-events' +import { + createActivityEventBuildCache, + resolvePaneBuild, + type ActivityEventBuildCache +} from './activity-event-build-cache' +import { appendUnsupportedAndRetainedEvents } from './activity-event-builder-sources' +import { + attributedActivityTabContext, + buildActivityTabContext, + buildActivityTabHostIndex, + resolveActivityEventOwner, + type ActivityEventOwner +} from './activity-event-builder-context' -const STANDALONE_ACTIVITY_WORKTREE_REPO_ID = '__activity_standalone__' - -function isActivityEventState(state: AgentStatusState): state is ActivityEventState { - return state === 'done' || state === 'blocked' || state === 'waiting' -} +export { createActivityEventBuildCache, type ActivityEventBuildCache, newestActivityHistoryEntries } function isActivityHookLiveAgentState( state: AgentStatusState @@ -50,142 +58,47 @@ function freshActivityLiveAgentState( : entry.state } -function standaloneActivityWorktree(worktreeId: string): Worktree { - const displayName = - worktreeId === FLOATING_TERMINAL_WORKTREE_ID ? 'Floating terminal' : 'Standalone terminal' - return { - id: worktreeId, - repoId: STANDALONE_ACTIVITY_WORKTREE_REPO_ID, - path: '', - head: '', - branch: displayName, - isBare: false, - isMainWorktree: false, - displayName, - comment: '', - linkedIssue: null, - linkedPR: null, - linkedLinearIssue: null, - isArchived: false, - isUnread: false, - isPinned: false, - sortOrder: 0, - lastActivityAt: 0 - } -} - -function historyEntrySnapshot( - entry: AgentStatusEntry, - history: AgentStateHistoryEntry -): AgentStatusEntry { - return { - ...entry, - state: history.state, - prompt: history.prompt, - updatedAt: history.startedAt, - stateStartedAt: history.startedAt, - stateHistory: [], - toolName: undefined, - toolInput: undefined, - lastAssistantMessage: undefined, - interrupted: history.interrupted - } -} - -function appendActivityEvent(args: { - events: ActivityEvent[] - seenEventIds: Set - state: ActivityEventState - timestamp: number - worktree: Worktree - repo: Repo | null - entry: AgentStatusEntry - tab: TerminalTab - agentType: AgentType - agentAlive: boolean - acknowledgedAt: number - migrationUnsupportedPtyId?: string -}): void { - const id = `agent:${args.entry.paneKey}:${args.state}:${args.timestamp}` - if (args.seenEventIds.has(id)) { - return - } - args.seenEventIds.add(id) - args.events.push({ - id, - state: args.state, - timestamp: args.timestamp, - worktree: args.worktree, - repo: args.repo, - entry: args.entry, - tab: args.tab, - agentType: args.agentType, - agentAlive: args.agentAlive, - migrationUnsupportedPtyId: args.migrationUnsupportedPtyId, - unread: args.acknowledgedAt < args.timestamp - }) -} - -function appendActivityEventsForEntry(args: { - events: ActivityEvent[] - seenEventIds: Set - entry: AgentStatusEntry - worktree: Worktree - repo: Repo | null - tab: TerminalTab - agentType: AgentType - agentAlive: boolean - acknowledgedAt: number - migrationUnsupportedPtyId?: string -}): void { - // Why: Activity is append-only; when a pane continues (done→working), stateHistory is the only record of the previous done/blocking event. - for (const history of args.entry.stateHistory) { - if (!isActivityEventState(history.state)) { - continue - } - appendActivityEvent({ - ...args, - state: history.state, - timestamp: history.startedAt, - entry: historyEntrySnapshot(args.entry, history) - }) - } - - // Why: SessionStart creates an idle row, not an "Agent finished" activity event (STA-3386). - if (!isActivityEventState(args.entry.state) || args.entry.sessionBoundary === true) { - return - } - appendActivityEvent({ - ...args, - state: args.entry.state, - timestamp: args.entry.stateStartedAt - }) -} - -type BuildActivityEventsArgs = { +export type BuildActivityEventsArgs = { agentStatusByPaneKey: Record + runtimeAgentOrchestrationByPaneKey?: Record migrationUnsupportedByPtyId?: Record retainedAgentsByPaneKey: Record tabsByWorktree: Record + unifiedTabsByWorktree?: Record worktreeMap: Map repoMap: Map + repos?: readonly Repo[] + resolveWorktree?: (worktreeId: string, executionHostId?: ExecutionHostId) => Worktree | undefined acknowledgedAgentsByPaneKey: Record + /** Per-pane "Clear completed" cutoffs; events stamped at or before the cutoff are hidden. */ + activityClearedAtByPaneKey?: Record now: number } -export function buildActivityEvents(args: BuildActivityEventsArgs): { +export function buildActivityEvents( + args: BuildActivityEventsArgs, + cache?: ActivityEventBuildCache +): { events: ActivityEvent[] liveAgentByPaneKey: Record } { const events: ActivityEvent[] = [] const seenEventIds = new Set() - const tabContext = new Map() + const tabContext = buildActivityTabContext(args.tabsByWorktree, args.unifiedTabsByWorktree) + const tabHostIndex = buildActivityTabHostIndex(args.unifiedTabsByWorktree) + const ownerCache = new Map() const liveAgentByPaneKey: Record = {} + const seenCacheKeys = cache ? new Set() : null - for (const [worktreeId, tabs] of Object.entries(args.tabsByWorktree)) { - const worktree = args.worktreeMap.get(worktreeId) ?? standaloneActivityWorktree(worktreeId) - for (const tab of tabs) { - tabContext.set(tab.id, { worktree, tab }) + const pushPaneEvents = (paneEvents: ActivityEvent[]): void => { + // Why: a paneKey can appear in more than one source (live + retained overlap); + // event ids stay globally unique so the first source wins, as before. + for (const event of paneEvents) { + if (seenEventIds.has(event.id)) { + continue + } + seenEventIds.add(event.id) + events.push(event) } } @@ -194,101 +107,64 @@ export function buildActivityEvents(args: BuildActivityEventsArgs): { if (!parsed) { continue } - const context = tabContext.get(parsed.tabId) + const context = tabContext.get(parsed.tabId) ?? attributedActivityTabContext(entry) if (!context) { continue } - const ackAt = args.acknowledgedAgentsByPaneKey[paneKey] ?? 0 + const owner = resolveActivityEventOwner( + args, + context, + entry, + context.tab.ptyId, + tabHostIndex, + ownerCache + ) + const orchestration = args.runtimeAgentOrchestrationByPaneKey?.[paneKey] // Why: live status is separate from history; a fresh working turn updates the thread without counting as an unread done/blocked/waiting event. + // The freshness check runs on the raw entry (orchestration merges never change state/timing fields). const liveState = freshActivityLiveAgentState(entry, args.now) - if (liveState) { - liveAgentByPaneKey[paneKey] = { - state: liveState, - timestamp: entry.stateStartedAt, - worktree: context.worktree, - repo: args.repoMap.get(context.worktree.repoId) ?? null, + const { events: paneEvents, live } = resolvePaneBuild( + { + cacheKey: `live:${paneKey}`, + source: entry, entry, + orchestration, + worktree: owner.worktree, + repo: owner.repo, tab: context.tab, - agentType: entry.agentType ?? 'unknown' + agentType: entry.agentType ?? 'unknown', + agentAlive: true, + acknowledgedAt: args.acknowledgedAgentsByPaneKey[paneKey] ?? 0, + clearedAt: args.activityClearedAtByPaneKey?.[paneKey] ?? 0, + liveState + }, + cache, + seenCacheKeys + ) + if (live) { + liveAgentByPaneKey[paneKey] = live + } + pushPaneEvents(paneEvents) + } + + appendUnsupportedAndRetainedEvents({ + args, + cache, + seenCacheKeys, + liveAgentByPaneKey, + tabContext, + resolveOwner: (context, entry, terminalPtyId) => + resolveActivityEventOwner(args, context, entry, terminalPtyId, tabHostIndex, ownerCache), + pushPaneEvents + }) + + // Why: evict panes gone from every source so the cache can't outgrow the live state maps. + if (cache && seenCacheKeys) { + for (const cacheKey of cache.panes.keys()) { + if (!seenCacheKeys.has(cacheKey)) { + cache.panes.delete(cacheKey) } } - appendActivityEventsForEntry({ - events, - seenEventIds, - worktree: context.worktree, - repo: args.repoMap.get(context.worktree.repoId) ?? null, - entry, - tab: context.tab, - agentType: entry.agentType ?? 'unknown', - agentAlive: true, - acknowledgedAt: ackAt - }) } - - appendUnsupportedAndRetainedEvents(args, events, seenEventIds, liveAgentByPaneKey, tabContext) return { events: capActivityEvents(events), liveAgentByPaneKey } } - -function appendUnsupportedAndRetainedEvents( - args: BuildActivityEventsArgs, - events: ActivityEvent[], - seenEventIds: Set, - liveAgentByPaneKey: Record, - tabContext: Map -): void { - for (const unsupported of Object.values(args.migrationUnsupportedByPtyId ?? {})) { - const entry = migrationUnsupportedToAgentStatusEntry(unsupported) - const parsed = entry ? parsePaneKey(entry.paneKey) : null - const context = parsed ? tabContext.get(parsed.tabId) : null - if (!entry || !context) { - continue - } - const ackAt = args.acknowledgedAgentsByPaneKey[entry.paneKey] ?? 0 - liveAgentByPaneKey[entry.paneKey] = { - state: 'blocked', - timestamp: entry.stateStartedAt, - worktree: context.worktree, - repo: args.repoMap.get(context.worktree.repoId) ?? null, - entry, - tab: context.tab, - agentType: entry.agentType ?? 'unknown' - } - appendActivityEventsForEntry({ - events, - seenEventIds, - worktree: context.worktree, - repo: args.repoMap.get(context.worktree.repoId) ?? null, - entry, - tab: context.tab, - agentType: entry.agentType ?? 'unknown', - agentAlive: false, - acknowledgedAt: ackAt, - migrationUnsupportedPtyId: unsupported.ptyId - }) - } - - for (const [paneKey, retained] of Object.entries(args.retainedAgentsByPaneKey)) { - if (!parsePaneKey(paneKey)) { - continue - } - const worktree = - args.worktreeMap.get(retained.worktreeId) ?? - (args.tabsByWorktree[retained.worktreeId] - ? standaloneActivityWorktree(retained.worktreeId) - : null) - if (!worktree) { - continue - } - appendActivityEventsForEntry({ - events, - seenEventIds, - worktree, - repo: args.repoMap.get(worktree.repoId) ?? null, - entry: retained.entry, - tab: retained.tab, - agentType: retained.agentType, - agentAlive: false, - acknowledgedAt: args.acknowledgedAgentsByPaneKey[paneKey] ?? 0 - }) - } -} diff --git a/src/renderer/src/components/activity/activity-event-cap.ts b/src/renderer/src/components/activity/activity-event-cap.ts index c6b80d53823..54d84c3039c 100644 --- a/src/renderer/src/components/activity/activity-event-cap.ts +++ b/src/renderer/src/components/activity/activity-event-cap.ts @@ -1,7 +1,8 @@ import type { ActivityEvent } from './activity-thread-types' // Why: per-pane cap guarantees each agent appears in the left list even when one pane has a long history. -const EVENTS_PER_PANE_CAP = 5 +// Exported so the event builder can skip building history entries the cap would drop anyway. +export const EVENTS_PER_PANE_CAP = 5 export function capActivityEvents(events: ActivityEvent[]): ActivityEvent[] { const sorted = events.sort((a, b) => b.timestamp - a.timestamp) diff --git a/src/renderer/src/components/activity/activity-pane-events.ts b/src/renderer/src/components/activity/activity-pane-events.ts new file mode 100644 index 00000000000..d3da82e5486 --- /dev/null +++ b/src/renderer/src/components/activity/activity-pane-events.ts @@ -0,0 +1,108 @@ +import type { + AgentStateHistoryEntry, + AgentStatusEntry +} from '../../../../shared/agent-status-types' +import type { Repo } from '../../../../shared/repo-types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type { ActivityEvent, ActivityEventState } from './activity-thread-types' +import { EVENTS_PER_PANE_CAP } from './activity-event-cap' + +export function isActivityEventState( + state: AgentStatusEntry['state'] +): state is ActivityEventState { + return state === 'done' || state === 'blocked' || state === 'waiting' +} + +function historyEntrySnapshot( + entry: AgentStatusEntry, + history: AgentStateHistoryEntry +): AgentStatusEntry { + return { + ...entry, + state: history.state, + prompt: history.prompt, + updatedAt: history.startedAt, + stateStartedAt: history.startedAt, + stateHistory: [], + toolName: undefined, + toolInput: undefined, + lastAssistantMessage: undefined, + interrupted: history.interrupted + } +} + +/** Newest activity-eligible history entries, at most `cap`, oldest-first. */ +export function newestActivityHistoryEntries( + history: readonly AgentStateHistoryEntry[], + cap: number +): AgentStateHistoryEntry[] { + const newest: AgentStateHistoryEntry[] = [] + for (let i = history.length - 1; i >= 0 && newest.length < cap; i -= 1) { + if (isActivityEventState(history[i].state)) { + newest.push(history[i]) + } + } + return newest.toReversed() +} + +type PaneEventInputs = { + entry: AgentStatusEntry + worktree: Worktree + repo: Repo | null + tab: TerminalTab + agentType: AgentStatusEntry['agentType'] + agentAlive: boolean + acknowledgedAt: number + clearedAt: number + migrationUnsupportedPtyId?: string +} + +/** Build one pane's activity events (bounded by the per-pane cap, cutoff applied). */ +export function buildPaneActivityEvents(args: PaneEventInputs): ActivityEvent[] { + const events: ActivityEvent[] = [] + const seenIds = new Set() + const append = (state: ActivityEventState, timestamp: number, entry: AgentStatusEntry): void => { + const id = `agent:${entry.paneKey}:${state}:${timestamp}` + if (seenIds.has(id)) { + return + } + seenIds.add(id) + events.push({ + id, + state, + timestamp, + worktree: args.worktree, + repo: args.repo, + entry, + tab: args.tab, + agentType: args.agentType ?? 'unknown', + agentAlive: args.agentAlive, + migrationUnsupportedPtyId: args.migrationUnsupportedPtyId, + unread: args.acknowledgedAt < timestamp + }) + } + + for (const history of newestActivityHistoryEntries( + args.entry.stateHistory, + EVENTS_PER_PANE_CAP + )) { + if (history.startedAt <= args.clearedAt) { + continue + } + append( + history.state as ActivityEventState, + history.startedAt, + historyEntrySnapshot(args.entry, history) + ) + } + + if (!isActivityEventState(args.entry.state) || args.entry.sessionBoundary === true) { + return events + } + if (args.entry.stateStartedAt <= args.clearedAt) { + return events + } + append(args.entry.state, args.entry.stateStartedAt, args.entry) + return events +} diff --git a/src/renderer/src/components/activity/activity-scope-filter-controls.tsx b/src/renderer/src/components/activity/activity-scope-filter-controls.tsx new file mode 100644 index 00000000000..6256595ba6b --- /dev/null +++ b/src/renderer/src/components/activity/activity-scope-filter-controls.tsx @@ -0,0 +1,170 @@ +import React, { useMemo } from 'react' +import { X } from 'lucide-react' +import { useAppStore } from '@/store' +import { DropdownMenuLabel, DropdownMenuSeparator } from '@/components/ui/dropdown-menu' +import SidebarRepositoryFilterSection from '@/components/sidebar/SidebarRepositoryFilterSection' +import { SidebarHostScopeMenuSection } from '@/components/sidebar/SidebarHostScopeMenuSection' +import { + getSidebarHostVisibilityLabel, + shouldShowHostScopeControls +} from '@/components/sidebar/sidebar-host-options' +import { useSidebarHostScopeOptions } from '@/components/sidebar/use-sidebar-host-scope-options' +import type { SidebarHostOption } from '@/components/sidebar/sidebar-host-options' +import { getExecutionHostLabel, type ExecutionHostId } from '../../../../shared/execution-host' +import { translate } from '@/i18n/i18n' + +/** + * Host/project scope controls for the Agents activity surfaces. State is the + * persisted agents-view scope (agentsVisibleHostIds / agentsFilterRepoIds), + * deliberately separate from the workspace-nav filters. + */ +export function ActivityScopeFilterMenuSections(): React.JSX.Element | null { + const repos = useAppStore((s) => s.repos) + const agentsVisibleHostIds = useAppStore((s) => s.agentsVisibleHostIds) + const setAgentsVisibleHostIds = useAppStore((s) => s.setAgentsVisibleHostIds) + const agentsFilterRepoIds = useAppStore((s) => s.agentsFilterRepoIds) + const setAgentsFilterRepoIds = useAppStore((s) => s.setAgentsFilterRepoIds) + const { hostOptions } = useSidebarHostScopeOptions() + const showHostScopeControls = shouldShowHostScopeControls(hostOptions) + + if (!showHostScopeControls && repos.length <= 1) { + return null + } + return ( + <> + + {translate('auto.components.sidebar.SidebarWorkspaceOptionsMenu.showSection', 'Show')} + + {showHostScopeControls ? ( + setAgentsVisibleHostIds(null)} + visibleWorkspaceHostIds={agentsVisibleHostIds} + setVisibleWorkspaceHostIds={setAgentsVisibleHostIds} + /> + ) : null} + + + + ) +} + +// Why not getSidebarHostVisibilityLabel: it collapses a full selection to "All +// hosts", but a chip only renders while a filter is set — name the selection, +// falling back to the raw host label for hosts no longer in the options list. +function getScopeHostChipLabel( + visibleHostIds: readonly ExecutionHostId[], + hostOptions: readonly SidebarHostOption[] +): string { + if (visibleHostIds.length === 1) { + const id = visibleHostIds[0] + return hostOptions.find((host) => host.id === id)?.label ?? getExecutionHostLabel(id) + } + return translate( + 'auto.components.sidebar.sidebarHostOptions.visibleHostsCount', + '{{value0}} hosts', + { value0: visibleHostIds.length } + ) +} + +function ScopeFilterChip({ + label, + clearLabel, + onClear +}: { + label: string + clearLabel: string + onClear: () => void +}): React.JSX.Element { + return ( + + {label} + + + ) +} + +/** + * Dismissible chips naming the active persisted scope. + * Why always shown while a scope is active: the filter survives restarts, so an + * invisible one would silently hide running agents from a monitoring surface. + * + * Why the outer/inner split: this stays mounted on every activity surface, so + * while no scope is set it must subscribe only to the two filter fields — the + * host-registry derivation (settings, SSH/runtime status churn) lives in the + * inner row and mounts only for an active filter. + */ +export function ActivityScopeFilterChips(): React.JSX.Element | null { + const agentsVisibleHostIds = useAppStore((s) => s.agentsVisibleHostIds) + const agentsFilterRepoIds = useAppStore((s) => s.agentsFilterRepoIds) + if (agentsVisibleHostIds === null && agentsFilterRepoIds.length === 0) { + return null + } + return +} + +function ActiveScopeFilterChipsRow(): React.JSX.Element | null { + const repos = useAppStore((s) => s.repos) + const agentsVisibleHostIds = useAppStore((s) => s.agentsVisibleHostIds) + const setAgentsVisibleHostIds = useAppStore((s) => s.setAgentsVisibleHostIds) + const agentsFilterRepoIds = useAppStore((s) => s.agentsFilterRepoIds) + const setAgentsFilterRepoIds = useAppStore((s) => s.setAgentsFilterRepoIds) + const { hostOptions } = useSidebarHostScopeOptions() + + const selectedRepoNames = useMemo( + () => + repos.filter((repo) => agentsFilterRepoIds.includes(repo.id)).map((repo) => repo.displayName), + [repos, agentsFilterRepoIds] + ) + const hasHostFilter = agentsVisibleHostIds !== null + const hasRepoFilter = selectedRepoNames.length > 0 + // Why: a repo filter of only-stale ids renders nothing; the outer gate is a fast path, not the authority. + if (!hasHostFilter && !hasRepoFilter) { + return null + } + const repoLabel = + selectedRepoNames.length === 1 + ? selectedRepoNames[0] + : translate( + 'auto.components.sidebar.SidebarRepositoryFilterSection.selectedProjectsCount', + '{{value0}} projects', + { value0: selectedRepoNames.length } + ) + return ( +
    + {agentsVisibleHostIds ? ( + setAgentsVisibleHostIds(null)} + /> + ) : null} + {hasRepoFilter ? ( + setAgentsFilterRepoIds([])} + /> + ) : null} +
    + ) +} diff --git a/src/renderer/src/components/activity/activity-scope-filter.test.ts b/src/renderer/src/components/activity/activity-scope-filter.test.ts new file mode 100644 index 00000000000..45817e50a7b --- /dev/null +++ b/src/renderer/src/components/activity/activity-scope-filter.test.ts @@ -0,0 +1,130 @@ +import { describe, expect, it } from 'vitest' +import { LOCAL_EXECUTION_HOST_ID, type ExecutionHostId } from '../../../../shared/execution-host' +import type { Repo } from '../../../../shared/repo-types' +import { + filterThreadsByActivityScope, + resolveActivityScopeRepoIds, + threadMatchesActivityScope, + type ActivityScopeFilter +} from './activity-scope-filter' +import type { AgentPaneThread } from './activity-thread-types' +import { + makeRepo, + makeTabWithIds, + makeWorktree, + PANE_KEY +} from './ActivityPrototypePage-test-fixtures' + +const SSH_HOST = 'ssh:devbox' as ExecutionHostId + +function makeThread(overrides: Partial = {}): AgentPaneThread { + const worktree = makeWorktree() + return { + paneKey: PANE_KEY, + paneTitle: 'Test Agent', + agentType: 'claude', + worktree, + repo: makeRepo(), + tab: makeTabWithIds('tab-1', worktree.id), + events: [], + latestEvent: null, + latestTimestamp: 1000, + currentAgentState: 'working', + currentAgentEntry: null, + unread: false, + responsePreview: '', + ...overrides + } +} + +function makeScope(overrides: Partial = {}): ActivityScopeFilter { + return { + visibleHostIds: null, + filterRepoIds: [], + defaultHostId: LOCAL_EXECUTION_HOST_ID, + ...overrides + } +} + +describe('threadMatchesActivityScope', () => { + it('matches everything when no scope is active', () => { + expect(threadMatchesActivityScope(makeThread(), makeScope())).toBe(true) + expect(threadMatchesActivityScope(makeThread({ repo: null }), makeScope())).toBe(true) + }) + + it('filters by execution host, falling back to the default host for local worktrees', () => { + const local = makeThread() + const remote = makeThread({ + worktree: { ...makeWorktree(), hostId: SSH_HOST } + }) + const localOnly = makeScope({ visibleHostIds: [LOCAL_EXECUTION_HOST_ID] }) + expect(threadMatchesActivityScope(local, localOnly)).toBe(true) + expect(threadMatchesActivityScope(remote, localOnly)).toBe(false) + const remoteOnly = makeScope({ visibleHostIds: [SSH_HOST] }) + expect(threadMatchesActivityScope(local, remoteOnly)).toBe(false) + expect(threadMatchesActivityScope(remote, remoteOnly)).toBe(true) + }) + + it('filters by project and hides repo-less threads under a project scope', () => { + const scope = makeScope({ filterRepoIds: ['repo-1'] }) + expect(threadMatchesActivityScope(makeThread(), scope)).toBe(true) + expect( + threadMatchesActivityScope(makeThread({ repo: { ...makeRepo(), id: 'repo-2' } }), scope) + ).toBe(false) + expect(threadMatchesActivityScope(makeThread({ repo: null }), scope)).toBe(false) + }) +}) + +describe('filterThreadsByActivityScope', () => { + it('returns the input array by identity when the scope is inactive', () => { + const threads = [makeThread(), makeThread({ paneKey: 'pane-2', repo: null })] + const result = filterThreadsByActivityScope({ + threads, + scope: makeScope(), + exemptPaneKey: null + }) + expect(result.threads).toBe(threads) + expect(result.matchingThreads).toBe(threads) + expect(result.hiddenCount).toBe(0) + }) + + it('returns the input array by identity when an active scope hides nothing', () => { + const threads = [makeThread()] + const result = filterThreadsByActivityScope({ + threads, + scope: makeScope({ visibleHostIds: [LOCAL_EXECUTION_HOST_ID] }), + exemptPaneKey: null + }) + expect(result.threads).toBe(threads) + expect(result.matchingThreads).toBe(threads) + expect(result.hiddenCount).toBe(0) + }) + + it('hides scoped-out threads but keeps the exempt pane, counting only real hides', () => { + const local = makeThread() + const remote = makeThread({ + paneKey: 'pane-remote', + worktree: { ...makeWorktree(), hostId: SSH_HOST } + }) + const exemptRemote = makeThread({ + paneKey: 'pane-exempt', + worktree: { ...makeWorktree(), hostId: SSH_HOST } + }) + const result = filterThreadsByActivityScope({ + threads: [local, remote, exemptRemote], + scope: makeScope({ visibleHostIds: [LOCAL_EXECUTION_HOST_ID] }), + exemptPaneKey: 'pane-exempt' + }) + expect(result.threads).toEqual([local, exemptRemote]) + expect(result.matchingThreads).toEqual([local]) + expect(result.hiddenCount).toBe(1) + }) +}) + +describe('resolveActivityScopeRepoIds', () => { + it('drops stale repo ids so they cannot count as an active filter', () => { + const repoMap = new Map([['repo-1', makeRepo()]]) + expect(resolveActivityScopeRepoIds(['repo-1', 'gone-repo'], repoMap)).toEqual(['repo-1']) + expect(resolveActivityScopeRepoIds(['gone-repo'], repoMap)).toEqual([]) + }) +}) diff --git a/src/renderer/src/components/activity/activity-scope-filter.ts b/src/renderer/src/components/activity/activity-scope-filter.ts new file mode 100644 index 00000000000..36dfafa8cf6 --- /dev/null +++ b/src/renderer/src/components/activity/activity-scope-filter.ts @@ -0,0 +1,80 @@ +import type { ExecutionHostId } from '../../../../shared/execution-host' +import { getWorktreeExecutionHostId } from '../../../../shared/execution-host' +import type { Repo } from '../../../../shared/repo-types' +import type { AgentPaneThread } from './activity-thread-types' + +/** Host/project scope for the Agents activity surfaces. Persisted and separate + * from the workspace-nav filters; hosts `null` = all, repoIds empty = all. */ +export type ActivityScopeFilter = { + visibleHostIds: readonly ExecutionHostId[] | null + filterRepoIds: readonly string[] + defaultHostId: ExecutionHostId +} + +/** Repo ids that still exist; stale persisted ids must not count as an active filter. */ +export function resolveActivityScopeRepoIds( + filterRepoIds: readonly string[], + repoMap: ReadonlyMap +): string[] { + return filterRepoIds.filter((repoId) => repoMap.has(repoId)) +} + +/** + * Apply the scope to a thread list. Returns the input array by identity when + * the scope is inactive or hides nothing, so downstream memos (visible threads, + * grouping) see an unchanged dep instead of re-running on every rebuild. + */ +export function filterThreadsByActivityScope(args: { + threads: AgentPaneThread[] + scope: ActivityScopeFilter + /** Kept visible even when scoped out, so changing scope can't vanish the open row. */ + exemptPaneKey: string | null +}): { + threads: AgentPaneThread[] + /** Strict matches for bulk actions; excludes a selected row kept visible only by the exemption. */ + matchingThreads: AgentPaneThread[] + hiddenCount: number +} { + const { threads, scope, exemptPaneKey } = args + if (!scope.visibleHostIds && scope.filterRepoIds.length === 0) { + return { threads, matchingThreads: threads, hiddenCount: 0 } + } + const matchingThreads: AgentPaneThread[] = [] + const visibleThreads: AgentPaneThread[] = [] + for (const thread of threads) { + if (threadMatchesActivityScope(thread, scope)) { + matchingThreads.push(thread) + visibleThreads.push(thread) + } else if (thread.paneKey === exemptPaneKey) { + visibleThreads.push(thread) + } + } + return { + threads: visibleThreads.length === threads.length ? threads : visibleThreads, + matchingThreads: matchingThreads.length === threads.length ? threads : matchingThreads, + hiddenCount: threads.length - visibleThreads.length + } +} + +export function threadMatchesActivityScope( + thread: AgentPaneThread, + scope: ActivityScopeFilter +): boolean { + if (scope.visibleHostIds) { + const hostId = getWorktreeExecutionHostId( + thread.worktree, + thread.repo ?? undefined, + scope.defaultHostId + ) + if (!scope.visibleHostIds.includes(hostId)) { + return false + } + } + // Why: repo-less terminal buckets have no project, so a project scope hides them. + if (scope.filterRepoIds.length > 0) { + if (!thread.repo || !scope.filterRepoIds.includes(thread.repo.id)) { + return false + } + } + return true +} diff --git a/src/renderer/src/components/activity/activity-standalone-worktree.ts b/src/renderer/src/components/activity/activity-standalone-worktree.ts new file mode 100644 index 00000000000..9844fbb9896 --- /dev/null +++ b/src/renderer/src/components/activity/activity-standalone-worktree.ts @@ -0,0 +1,67 @@ +import { i18n, translate } from '@/i18n/i18n' +import { FLOATING_TERMINAL_WORKTREE_ID } from '../../../../shared/constants' +import type { Worktree } from '../../../../shared/worktree/types' +import type { ExecutionHostId } from '../../../../shared/execution-host' + +const STANDALONE_ACTIVITY_WORKTREE_REPO_ID = '__activity_standalone__' +const STANDALONE_ACTIVITY_WORKTREES_CAP = 200 +const standaloneActivityWorktrees = new Map() +// The cached rows carry a localized displayName, so they cannot outlive a language switch. +let cachedDisplayNameLocale: string | undefined + +function buildStandaloneActivityWorktree( + worktreeId: string, + executionHostId?: ExecutionHostId +): Worktree { + const displayName = + worktreeId === FLOATING_TERMINAL_WORKTREE_ID + ? translate( + 'auto.components.activity.standaloneWorktree.floatingTerminal', + 'Floating terminal' + ) + : translate( + 'auto.components.activity.standaloneWorktree.standaloneTerminal', + 'Standalone terminal' + ) + return { + id: worktreeId, + ...(executionHostId ? { hostId: executionHostId } : {}), + repoId: STANDALONE_ACTIVITY_WORKTREE_REPO_ID, + path: '', + head: '', + branch: displayName, + isBare: false, + isMainWorktree: false, + displayName, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 0 + } +} + +/** Return a stable synthetic worktree for terminal-only activity. */ +export function standaloneActivityWorktree( + worktreeId: string, + executionHostId?: ExecutionHostId +): Worktree { + if (cachedDisplayNameLocale !== i18n.language) { + cachedDisplayNameLocale = i18n.language + standaloneActivityWorktrees.clear() + } + const cacheKey = `${worktreeId}\0${executionHostId ?? ''}` + let worktree = standaloneActivityWorktrees.get(cacheKey) + if (!worktree) { + if (standaloneActivityWorktrees.size >= STANDALONE_ACTIVITY_WORKTREES_CAP) { + standaloneActivityWorktrees.clear() + } + worktree = buildStandaloneActivityWorktree(worktreeId, executionHostId) + standaloneActivityWorktrees.set(cacheKey, worktree) + } + return worktree +} diff --git a/src/renderer/src/components/activity/activity-tab-projection.ts b/src/renderer/src/components/activity/activity-tab-projection.ts new file mode 100644 index 00000000000..028d474d519 --- /dev/null +++ b/src/renderer/src/components/activity/activity-tab-projection.ts @@ -0,0 +1,66 @@ +import type { Tab } from '../../../../shared/tab-types' + +/** + * Stable view of the unified tab map for the activity pipeline. + * + * Why: the store rewrites the focused tab object (lastFocusedAt) on every focus, which would + * rebuild every activity thread even though nothing the pipeline reads changed. Keep only the + * tab kinds the pipeline consults and reuse prior tab objects when their relevant fields match, + * so downstream identity-keyed caches keep hitting. + */ +export type ActivityTabProjection = Record + +function isActivityRelevantTab(tab: Tab): boolean { + return tab.contentType === 'terminal' || tab.contentType === 'agent-session' +} + +function activityTabFieldsEqual(a: Tab, b: Tab): boolean { + return ( + a.id === b.id && + a.entityId === b.entityId && + a.worktreeId === b.worktreeId && + a.executionHostId === b.executionHostId && + a.contentType === b.contentType && + a.label === b.label && + a.generatedLabel === b.generatedLabel && + a.customLabel === b.customLabel && + a.color === b.color && + a.isPinned === b.isPinned && + a.sortOrder === b.sortOrder && + a.createdAt === b.createdAt + ) +} + +export function projectActivityTabs( + unifiedTabsByWorktree: Record | undefined, + previous: ActivityTabProjection | null +): ActivityTabProjection { + const next: ActivityTabProjection = {} + let changed = previous === null + for (const [worktreeId, tabs] of Object.entries(unifiedTabsByWorktree ?? {})) { + const relevant = tabs.filter(isActivityRelevantTab) + if (relevant.length === 0) { + continue + } + const prior = previous?.[worktreeId] + let reusable = prior !== undefined && prior.length === relevant.length + const projected = relevant.map((tab, index) => { + const priorTab = prior?.[index] + if (priorTab && activityTabFieldsEqual(priorTab, tab)) { + return priorTab + } + reusable = false + return tab + }) + if (reusable && prior) { + next[worktreeId] = prior + } else { + next[worktreeId] = projected + changed = true + } + } + if (!changed && previous && Object.keys(previous).length === Object.keys(next).length) { + return previous + } + return next +} diff --git a/src/renderer/src/components/activity/activity-thread-actions.test.ts b/src/renderer/src/components/activity/activity-thread-actions.test.ts new file mode 100644 index 00000000000..ce2de01fe02 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-actions.test.ts @@ -0,0 +1,162 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { makeRepo, makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' +import type { AgentPaneThread } from './activity-thread-types' + +const mocks = vi.hoisted(() => ({ + getState: vi.fn(), + activateTabAndFocusPane: vi.fn(), + activateStructuredAgentSessionTab: vi.fn(), + activateAndRevealWorkspace: vi.fn() +})) + +vi.mock('@/store', () => ({ useAppStore: { getState: mocks.getState } })) +vi.mock('@/lib/activate-tab-and-focus-pane', () => ({ + activateTabAndFocusPane: mocks.activateTabAndFocusPane +})) +vi.mock('@/lib/structured-agent-session-tab-activation', () => ({ + activateStructuredAgentSessionTab: mocks.activateStructuredAgentSessionTab +})) +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorkspace: mocks.activateAndRevealWorkspace +})) + +import { createActivityThreadActions, hasActivityThreadWorkspace } from './activity-thread-actions' + +const REMOTE_HOST = 'ssh:devbox' as const + +function makeRemoteThread(): AgentPaneThread { + const worktree = { ...makeWorktree(), hostId: REMOTE_HOST } + return { + paneKey: 'tab-1:11111111-1111-4111-8111-111111111111', + paneTitle: 'Remote agent', + agentType: 'claude', + worktree, + repo: makeRepo(), + tab: makeTab(), + events: [], + latestEvent: null, + latestTimestamp: 1_000, + currentAgentState: 'working', + currentAgentEntry: null, + unread: true, + responsePreview: '' + } +} + +describe('activity thread host routing', () => { + const thread = makeRemoteThread() + const getKnownWorktreeById = vi.fn() + const setActiveWorktree = vi.fn() + const acknowledgeAgents = vi.fn() + const setSelectedPaneKey = vi.fn() + + beforeEach(() => { + vi.clearAllMocks() + mocks.activateStructuredAgentSessionTab.mockReturnValue(false) + getKnownWorktreeById.mockReturnValue(thread.worktree) + mocks.getState.mockReturnValue({ + getKnownWorktreeById, + worktreesByRepo: { [thread.worktree.repoId]: [thread.worktree] }, + detectedWorktreesByRepo: {}, + folderWorkspaces: [], + showSleepingWorkspaces: true, + filterRepoIds: [], + hideDefaultBranchWorkspace: false, + hideAutomationGeneratedWorkspaces: false, + hideCliCreatedWorkspaces: false, + hideDetachedHeadWorkspaces: false, + hideWorkspacesFromOtherDevices: false, + alwaysShowDefaultBranchWorkspace: true, + visibleWorkspaceHostIds: null, + workspaceHostScope: 'all', + tabsByWorktree: { [thread.worktree.id]: [thread.tab] }, + unifiedTabsByWorktree: {}, + activeRepoId: thread.worktree.repoId, + activeWorktreeId: thread.worktree.id, + activeWorkspaceExecutionHostId: 'local', + setActiveRepo: vi.fn(), + setActiveWorktree, + setActiveTabType: vi.fn() + }) + }) + + it('selects the matching host when the same workspace id is active elsewhere', () => { + const actions = createActivityThreadActions({ + getMarkAllReadThreads: () => [thread], + acknowledgeAgents, + unacknowledgeAgents: vi.fn(), + setSelectedPaneKey + }) + + actions.selectThread(thread) + + expect(getKnownWorktreeById).toHaveBeenCalledWith(thread.worktree.id, REMOTE_HOST) + expect(setActiveWorktree).toHaveBeenCalledWith(thread.worktree.id, REMOTE_HOST) + expect(mocks.activateTabAndFocusPane).toHaveBeenCalledWith( + thread.tab.id, + '11111111-1111-4111-8111-111111111111', + { flashFocusedPane: true, scrollToBottomIfOutputSinceLastView: true } + ) + }) + + it('activates a structured agent session instead of looking for a terminal pane', () => { + mocks.activateStructuredAgentSessionTab.mockReturnValue(true) + mocks.getState.mockReturnValue({ + ...mocks.getState(), + tabsByWorktree: { [thread.worktree.id]: [] }, + unifiedTabsByWorktree: { + [thread.worktree.id]: [{ id: thread.tab.id, contentType: 'agent-session' }] + } + }) + const actions = createActivityThreadActions({ + getMarkAllReadThreads: () => [thread], + acknowledgeAgents, + unacknowledgeAgents: vi.fn(), + setSelectedPaneKey + }) + + actions.selectThread(thread) + + expect(mocks.activateStructuredAgentSessionTab).toHaveBeenCalledWith({ + worktreeId: thread.worktree.id, + tabId: thread.tab.id + }) + expect(mocks.activateTabAndFocusPane).not.toHaveBeenCalled() + }) + + it('jumps to and probes the matching host-qualified workspace', () => { + expect(hasActivityThreadWorkspace(thread)).toBe(true) + const actions = createActivityThreadActions({ + getMarkAllReadThreads: () => [thread], + acknowledgeAgents, + unacknowledgeAgents: vi.fn(), + setSelectedPaneKey + }) + + actions.jumpToWorkspace(thread) + + expect(acknowledgeAgents).toHaveBeenCalledWith([thread.paneKey]) + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith(thread.worktree.id, { + executionHostId: REMOTE_HOST + }) + }) + + it('marks all unread threads in the mark-all set, reading it at call time', () => { + const readThread = { ...makeRemoteThread(), paneKey: 'tab-2:read', unread: false } + let markAllSet = [readThread] + const actions = createActivityThreadActions({ + getMarkAllReadThreads: () => markAllSet, + acknowledgeAgents, + unacknowledgeAgents: vi.fn(), + setSelectedPaneKey + }) + + actions.markAllThreadsRead() + expect(acknowledgeAgents).not.toHaveBeenCalled() + + // The handler keeps one identity while the set changes underneath it. + markAllSet = [thread, readThread] + actions.markAllThreadsRead() + expect(acknowledgeAgents).toHaveBeenCalledWith([thread.paneKey]) + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-actions.ts b/src/renderer/src/components/activity/activity-thread-actions.ts index f1e1ff6ae5b..b0971da7ea7 100644 --- a/src/renderer/src/components/activity/activity-thread-actions.ts +++ b/src/renderer/src/components/activity/activity-thread-actions.ts @@ -1,22 +1,65 @@ import { activateTabAndFocusPane } from '@/lib/activate-tab-and-focus-pane' -import { activateAndRevealWorktree } from '@/lib/worktree-activation' +import { activateStructuredAgentSessionTab } from '@/lib/structured-agent-session-tab-activation' +import { jumpToWorktreeFromSidebar } from '@/lib/worktree-jump-navigation' import { useAppStore } from '@/store' -import { getWorktreeMapFromState } from '@/store/selectors' +import { + getSettingsFocusedExecutionHostId, + getWorktreeExecutionHostId, + type ExecutionHostId +} from '../../../../shared/execution-host' import { parsePaneKey } from '../../../../shared/stable-pane-id' +import { findKnownWorktreeById } from '@/store/slices/worktrees/listing/detected-worktree-meta' +import type { AppState } from '@/store/types' import type { AgentPaneThread } from './activity-thread-types' +// Same focused-host fallback the Agents scope filter uses; defaulting to `local` here would +// look up a hostless runtime-owned workspace on the wrong host and silently drop the jump. +function getActivityThreadExecutionHostId( + thread: AgentPaneThread, + defaultHostId: ExecutionHostId +): ExecutionHostId { + return getWorktreeExecutionHostId(thread.worktree, thread.repo ?? undefined, defaultHostId) +} + +type ActivityThreadWorkspaceCatalog = Pick< + AppState, + 'worktreesByRepo' | 'detectedWorktreesByRepo' | 'folderWorkspaces' +> & { defaultHostId: ExecutionHostId } + +function readActivityThreadWorkspaceCatalog(): ActivityThreadWorkspaceCatalog { + const state = useAppStore.getState() + return { ...state, defaultHostId: getSettingsFocusedExecutionHostId(state.settings) } +} + +export function hasActivityThreadWorkspace( + thread: AgentPaneThread, + catalog: ActivityThreadWorkspaceCatalog = readActivityThreadWorkspaceCatalog() +): boolean { + return Boolean( + findKnownWorktreeById( + catalog, + thread.worktree.id, + getActivityThreadExecutionHostId(thread, catalog.defaultHostId) + ) + ) +} + export function createActivityThreadActions({ - allThreads, + getMarkAllReadThreads, acknowledgeAgents, unacknowledgeAgents, setSelectedPaneKey }: { - allThreads: AgentPaneThread[] + /** Getter (not a snapshot) so the handlers keep one identity for the row memo + * bail-outs while bulk actions still see the current thread set. This is the + * badge-coherent set (child-filter only), not the search/scope-narrowed one, + * so Mark all read always drives the Agents badge to zero. */ + getMarkAllReadThreads: () => AgentPaneThread[] acknowledgeAgents: (paneKeys: string[]) => void unacknowledgeAgents: (paneKeys: string[]) => void setSelectedPaneKey: (paneKey: string | null) => void }): { - hasUnreadThreads: boolean + markThreadRead: (thread: AgentPaneThread) => void markThreadUnread: (thread: AgentPaneThread) => void selectThread: (thread: AgentPaneThread) => void jumpToWorkspace: (thread: AgentPaneThread) => void @@ -30,51 +73,67 @@ export function createActivityThreadActions({ unacknowledgeAgents([thread.paneKey]) } - const activateThreadTerminal = (thread: AgentPaneThread): void => { + const activateThreadTarget = (thread: AgentPaneThread): void => { const state = useAppStore.getState() - const worktree = getWorktreeMapFromState(state).get(thread.worktree.id) + const executionHostId = getActivityThreadExecutionHostId( + thread, + getSettingsFocusedExecutionHostId(state.settings) + ) + const worktree = state.getKnownWorktreeById(thread.worktree.id, executionHostId) if (!worktree) { return } - // Why: retained-agent threads can outlive their tab; without a live tab, reorienting the workspace and focusing a dead tab id would just confuse the user. const liveTabs = state.tabsByWorktree[worktree.id] ?? [] - const hasLiveTab = liveTabs.some((t) => t.id === thread.tab.id) - if (!hasLiveTab) { + const hasLiveTerminal = liveTabs.some((tab) => tab.id === thread.tab.id) + const hasLiveAgentSession = (state.unifiedTabsByWorktree?.[worktree.id] ?? []).some( + (tab) => tab.id === thread.tab.id && tab.contentType === 'agent-session' + ) + // Why: retained threads can outlive their target; reorienting the workspace for a + // dead terminal or structured session would just confuse the user. + if (!hasLiveTerminal && !hasLiveAgentSession) { return } if (state.activeRepoId !== worktree.repoId) { state.setActiveRepo(worktree.repoId) } - if (state.activeWorktreeId !== worktree.id) { - state.setActiveWorktree(worktree.id) + if ( + state.activeWorktreeId !== worktree.id || + state.activeWorkspaceExecutionHostId !== executionHostId + ) { + state.setActiveWorktree(worktree.id, executionHostId) + } + if (activateStructuredAgentSessionTab({ worktreeId: worktree.id, tabId: thread.tab.id })) { + return } state.setActiveTabType('terminal') const parsed = parsePaneKey(thread.paneKey) activateTabAndFocusPane( thread.tab.id, parsed && parsed.tabId === thread.tab.id ? parsed.leafId : null, - { scrollToBottomIfOutputSinceLastView: true } + { flashFocusedPane: true, scrollToBottomIfOutputSinceLastView: true } ) } const selectThread = (thread: AgentPaneThread): void => { setSelectedPaneKey(thread.paneKey) - activateThreadTerminal(thread) + activateThreadTarget(thread) } const jumpToWorkspace = (thread: AgentPaneThread): void => { - const state = useAppStore.getState() - if (!getWorktreeMapFromState(state).has(thread.worktree.id)) { + const catalog = readActivityThreadWorkspaceCatalog() + if (!hasActivityThreadWorkspace(thread, catalog)) { return } markThreadRead(thread) - activateAndRevealWorktree(thread.worktree.id) + jumpToWorktreeFromSidebar(thread.worktree.id, { + executionHostId: getActivityThreadExecutionHostId(thread, catalog.defaultHostId) + }) } - const hasUnreadThreads = allThreads.some((thread) => thread.unread) - const markAllThreadsRead = (): void => { - const unreadKeys = allThreads.filter((t) => t.unread).map((t) => t.paneKey) + const unreadKeys = getMarkAllReadThreads() + .filter((t) => t.unread) + .map((t) => t.paneKey) if (unreadKeys.length === 0) { return } @@ -82,7 +141,7 @@ export function createActivityThreadActions({ } return { - hasUnreadThreads, + markThreadRead, markThreadUnread, selectThread, jumpToWorkspace, diff --git a/src/renderer/src/components/activity/activity-thread-builder.ts b/src/renderer/src/components/activity/activity-thread-builder.ts index 8ba6ab4f5fd..a1076ea0022 100644 --- a/src/renderer/src/components/activity/activity-thread-builder.ts +++ b/src/renderer/src/components/activity/activity-thread-builder.ts @@ -9,11 +9,68 @@ import type { AgentPaneThread } from './activity-thread-types' -export function buildAgentPaneThreads(args: { - events: ActivityEvent[] - liveAgentByPaneKey: Record - generatedTitlesEnabled?: boolean -}): AgentPaneThread[] { +/** + * Caller-owned reuse cache: threads whose derived content is unchanged keep their + * previous object (and the whole list keeps its array) identity, so memo'd rows and + * the search-text cache survive unrelated store writes. + */ +export type AgentPaneThreadReuseCache = { + previousByPaneKey: Map + previousList: AgentPaneThread[] +} + +export function createAgentPaneThreadReuseCache(): AgentPaneThreadReuseCache { + return { previousByPaneKey: new Map(), previousList: [] } +} + +function arrayItemsEqual(a: readonly T[], b: readonly T[]): boolean { + if (a.length !== b.length) { + return false + } + for (let i = 0; i < a.length; i += 1) { + if (a[i] !== b[i]) { + return false + } + } + return true +} + +// Why: event and live-snapshot identities are preserved upstream (activity-event-builder +// cache), so identity comparison on the referenced objects is a correct change detector. +function reuseThreadIfEqual( + previous: AgentPaneThread | undefined, + next: AgentPaneThread +): AgentPaneThread { + if ( + previous !== undefined && + previous.paneKey === next.paneKey && + previous.paneTitle === next.paneTitle && + previous.worktree === next.worktree && + previous.repo === next.repo && + previous.tab === next.tab && + previous.agentType === next.agentType && + previous.currentAgentState === next.currentAgentState && + previous.currentAgentEntry === next.currentAgentEntry && + previous.responsePreview === next.responsePreview && + previous.latestTimestamp === next.latestTimestamp && + previous.latestEvent === next.latestEvent && + previous.migrationUnsupportedPtyId === next.migrationUnsupportedPtyId && + previous.unread === next.unread && + arrayItemsEqual(previous.events, next.events) + ) { + return previous + } + return next +} + +export function buildAgentPaneThreads( + args: { + events: ActivityEvent[] + liveAgentByPaneKey: Record + generatedTitlesEnabled?: boolean + }, + reuseCache?: AgentPaneThreadReuseCache +): AgentPaneThread[] { const generatedTitlesEnabled = args.generatedTitlesEnabled === true const byPaneKey = new Map() for (const event of args.events) { @@ -92,10 +149,22 @@ export function buildAgentPaneThreads(args: { existing.latestTimestamp = liveAgent.timestamp } - return Array.from(byPaneKey.values()) - .map((thread) => ({ - ...thread, - events: [...thread.events].sort((a, b) => b.timestamp - a.timestamp) - })) + const built = Array.from(byPaneKey.values()) + .map((thread) => { + const next: AgentPaneThread = { + ...thread, + events: [...thread.events].sort((a, b) => b.timestamp - a.timestamp) + } + return reuseThreadIfEqual(reuseCache?.previousByPaneKey.get(thread.paneKey), next) + }) .sort((a, b) => b.latestTimestamp - a.latestTimestamp) + + if (!reuseCache) { + return built + } + // Why: keep the list's array identity too, so downstream memos keyed on the list bail out. + const result = arrayItemsEqual(reuseCache.previousList, built) ? reuseCache.previousList : built + reuseCache.previousList = result + reuseCache.previousByPaneKey = new Map(result.map((thread) => [thread.paneKey, thread])) + return result } diff --git a/src/renderer/src/components/activity/activity-thread-child-agent.test.ts b/src/renderer/src/components/activity/activity-thread-child-agent.test.ts new file mode 100644 index 00000000000..7c687c72037 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-child-agent.test.ts @@ -0,0 +1,171 @@ +import { describe, expect, it } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { collectChildAgentPaneKeys } from './activity-thread-child-agent' +import type { ActivityEvent, AgentPaneThread } from './activity-thread-types' +import { + makeRepo, + makeTabWithIds, + makeWorkingEntryWithoutHistory, + makeWorktree, + PANE_KEY, + PANE_KEY_2, + PANE_KEY_3 +} from './ActivityPrototypePage-test-fixtures' + +function makeTestEntry( + paneKey: string, + overrides: Partial = {} +): AgentStatusEntry { + return { + ...makeWorkingEntryWithoutHistory(), + paneKey, + state: 'done', + prompt: 'test prompt', + stateHistory: [], + ...overrides + } +} + +function makeTestThread( + paneKey: string, + overrides: Partial = {} +): AgentPaneThread { + const worktree = makeWorktree() + return { + paneKey, + paneTitle: 'Test Agent', + agentType: 'claude', + worktree, + repo: makeRepo(), + tab: makeTabWithIds('tab-1', worktree.id), + events: [], + latestEvent: null, + latestTimestamp: 1000, + currentAgentState: 'working', + currentAgentEntry: makeTestEntry(paneKey), + unread: false, + responsePreview: '', + ...overrides + } +} + +function makeEventFor(entry: AgentStatusEntry): ActivityEvent { + const worktree = makeWorktree() + return { + id: `event-${entry.paneKey}`, + state: 'done', + timestamp: 1000, + unread: false, + worktree, + repo: null, + tab: makeTabWithIds('tab-1', worktree.id), + agentType: 'claude', + agentAlive: true, + entry + } +} + +describe('collectChildAgentPaneKeys', () => { + it('returns an empty set when no thread carries orchestration', () => { + const threads = [makeTestThread(PANE_KEY), makeTestThread(PANE_KEY_2)] + expect(collectChildAgentPaneKeys(threads).size).toBe(0) + }) + + it('classifies a thread whose parent pane is listed as a child', () => { + const parent = makeTestThread(PANE_KEY) + const child = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2, { + orchestration: { parentPaneKey: PANE_KEY, taskId: 'task-1', dispatchId: 'ctx-1' } + }) + }) + expect(collectChildAgentPaneKeys([parent, child])).toEqual(new Set([PANE_KEY_2])) + }) + + it('promotes an orphan whose parent pane is no longer listed', () => { + const orphan = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2, { + orchestration: { parentPaneKey: PANE_KEY, taskId: 'task-1', dispatchId: 'ctx-1' } + }) + }) + expect(collectChildAgentPaneKeys([orphan]).size).toBe(0) + }) + + it('ignores a self-referencing parentPaneKey', () => { + const thread = makeTestThread(PANE_KEY, { + currentAgentEntry: makeTestEntry(PANE_KEY, { + orchestration: { parentPaneKey: PANE_KEY, taskId: 'task-1', dispatchId: 'ctx-1' } + }) + }) + expect(collectChildAgentPaneKeys([thread]).size).toBe(0) + }) + + it('resolves coordinatorHandle through a listed thread terminal handle', () => { + const coordinator = makeTestThread(PANE_KEY, { + currentAgentEntry: makeTestEntry(PANE_KEY, { terminalHandle: 'terminal-coord' }) + }) + const worker = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2, { + terminalHandle: 'terminal-worker', + orchestration: { + coordinatorHandle: 'terminal-coord', + taskId: 'task-1', + dispatchId: 'ctx-1' + } + }) + }) + expect(collectChildAgentPaneKeys([coordinator, worker])).toEqual(new Set([PANE_KEY_2])) + }) + + it('promotes a worker whose coordinator handle matches no listed thread', () => { + const worker = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2, { + terminalHandle: 'terminal-worker', + orchestration: { coordinatorHandle: 'terminal-gone', taskId: 'task-1', dispatchId: 'ctx-1' } + }) + }) + expect(collectChildAgentPaneKeys([worker]).size).toBe(0) + }) + + it('keeps child classification from an older event while the parent is listed', () => { + const parent = makeTestThread(PANE_KEY) + const childEntry = makeTestEntry(PANE_KEY_2, { + orchestration: { parentPaneKey: PANE_KEY, taskId: 'task-1', dispatchId: 'ctx-1' } + }) + const child = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2), + events: [makeEventFor(childEntry)] + }) + expect(collectChildAgentPaneKeys([parent, child])).toEqual(new Set([PANE_KEY_2])) + }) + + it('classifies a grandchild chained through a listed child', () => { + const root = makeTestThread(PANE_KEY) + const child = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2, { + orchestration: { parentPaneKey: PANE_KEY, taskId: 'task-1', dispatchId: 'ctx-1' } + }) + }) + const grandchild = makeTestThread(PANE_KEY_3, { + currentAgentEntry: makeTestEntry(PANE_KEY_3, { + orchestration: { parentPaneKey: PANE_KEY_2, taskId: 'task-2', dispatchId: 'ctx-2' } + }) + }) + expect(collectChildAgentPaneKeys([root, child, grandchild])).toEqual( + new Set([PANE_KEY_2, PANE_KEY_3]) + ) + }) + + it('promotes every member of a parent cycle instead of hiding them all', () => { + const a = makeTestThread(PANE_KEY, { + currentAgentEntry: makeTestEntry(PANE_KEY, { + orchestration: { parentPaneKey: PANE_KEY_2, taskId: 'task-1', dispatchId: 'ctx-1' } + }) + }) + const b = makeTestThread(PANE_KEY_2, { + currentAgentEntry: makeTestEntry(PANE_KEY_2, { + orchestration: { parentPaneKey: PANE_KEY, taskId: 'task-2', dispatchId: 'ctx-2' } + }) + }) + expect(collectChildAgentPaneKeys([a, b]).size).toBe(0) + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-child-agent.ts b/src/renderer/src/components/activity/activity-thread-child-agent.ts new file mode 100644 index 00000000000..4fd0a3dbd10 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-child-agent.ts @@ -0,0 +1,90 @@ +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { + buildAgentRowLineageTree, + resolveAgentRowParentPaneKey, + type AgentLineageSourceRow +} from '../dashboard/agent-row-lineage-model' + +type ChildAgentLineageEntry = Pick + +/** Minimal structural input: a full AgentPaneThread satisfies it (pinned by the + * classifier tests), and count/badge callers can feed synthetic rows uncast. */ +export type ChildAgentClassifiableThread = { + paneKey: string + currentAgentEntry?: ChildAgentLineageEntry | null + latestEvent?: { entry: ChildAgentLineageEntry } | null + events?: readonly { entry: ChildAgentLineageEntry }[] +} + +/** Every entry that can carry the pane's orchestration lineage, newest first. */ +function candidateEntries(thread: ChildAgentClassifiableThread): ChildAgentLineageEntry[] { + const entries: ChildAgentLineageEntry[] = [] + if (thread.currentAgentEntry) { + entries.push(thread.currentAgentEntry) + } + if (thread.latestEvent?.entry) { + entries.push(thread.latestEvent.entry) + } + for (const event of thread.events ?? []) { + entries.push(event.entry) + } + return entries +} + +function firstReportedTerminalHandle(thread: ChildAgentClassifiableThread): string | undefined { + for (const entry of candidateEntries(thread)) { + if (entry.terminalHandle) { + return entry.terminalHandle + } + } + return undefined +} + +/** + * Pane keys of threads that are children of another currently listed thread. + * Delegates to the dashboard's lineage model so both surfaces classify the same + * pane identically: a parent reference only counts while the parent thread is + * still listed, so orphaned workers (their coordinator pane closed) and cycle + * members are promoted to top level instead of staying hidden behind the + * child-agent filter. Classification is sticky across a thread's older events: + * the newest entry whose parent still resolves wins. + */ +export function collectChildAgentPaneKeys( + threads: readonly ChildAgentClassifiableThread[] +): Set { + const baseRows: AgentLineageSourceRow[] = threads.map((thread) => ({ + paneKey: thread.paneKey, + entry: { terminalHandle: firstReportedTerminalHandle(thread) } + })) + const rowsByPaneKey = new Map() + for (const row of baseRows) { + if (!rowsByPaneKey.has(row.paneKey)) { + rowsByPaneKey.set(row.paneKey, row) + } + } + const paneKeyByTerminalHandle = new Map() + for (const row of baseRows) { + if (row.entry.terminalHandle && !paneKeyByTerminalHandle.has(row.entry.terminalHandle)) { + paneKeyByTerminalHandle.set(row.entry.terminalHandle, row.paneKey) + } + } + + const rows = threads.map((thread, index) => { + const base = baseRows[index] + for (const entry of candidateEntries(thread)) { + if (!entry.orchestration) { + continue + } + const probe: AgentLineageSourceRow = { + paneKey: thread.paneKey, + entry: { terminalHandle: base.entry.terminalHandle, orchestration: entry.orchestration } + } + if (resolveAgentRowParentPaneKey(probe, rowsByPaneKey, paneKeyByTerminalHandle)) { + return probe + } + } + return base + }) + + return buildAgentRowLineageTree(rows).childPaneKeys +} diff --git a/src/renderer/src/components/activity/activity-thread-collapse-context.ts b/src/renderer/src/components/activity/activity-thread-collapse-context.ts new file mode 100644 index 00000000000..070b395174d --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-collapse-context.ts @@ -0,0 +1,13 @@ +import { createContext } from 'react' + +export type ActivityThreadCollapseState = { + collapsedGroupKeys: ReadonlySet + onToggleGroupCollapse: (groupKey: string) => void +} + +/** + * Caller-owned collapse state for ActivityThreadListPane hosts that unmount the + * pane (sidebar body switches) but should keep the user's collapsed groups. + * Explicit collapsedGroupKeys/onToggleGroupCollapse props take precedence. + */ +export const ActivityThreadCollapseContext = createContext(null) diff --git a/src/renderer/src/components/activity/activity-thread-controls.tsx b/src/renderer/src/components/activity/activity-thread-controls.tsx index f6cb249c9fc..ad52cd9c51c 100644 --- a/src/renderer/src/components/activity/activity-thread-controls.tsx +++ b/src/renderer/src/components/activity/activity-thread-controls.tsx @@ -1,17 +1,11 @@ import React from 'react' -import { MoreVertical } from 'lucide-react' +import { ChevronDown } from 'lucide-react' import { AgentStateDot } from '@/components/AgentStateDot' -import { Button } from '@/components/ui/button' -import { - DropdownMenu, - DropdownMenuCheckboxItem, - DropdownMenuContent, - DropdownMenuItem, - DropdownMenuSeparator, - DropdownMenuTrigger -} from '@/components/ui/dropdown-menu' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { translate } from '@/i18n/i18n' +import { useNow } from '@/hooks/use-now' +import { cn } from '@/lib/utils' +import { formatShortTimeAgo } from '@/lib/short-time-ago' import { RepoBadgeMark } from '@/components/repo/RepoBadgeLabel' import type { Repo } from '../../../../shared/repo-types' import { @@ -22,18 +16,33 @@ import { } from './activity-thread-presentation' import type { ActivityThreadGroup, AgentPaneThread } from './activity-thread-types' -export function EventTime({ timestamp }: { timestamp: number }): React.JSX.Element { +export { ActivityThreadOptionsMenu } from './activity-thread-options-menu' + +export function EventTime({ + timestamp, + compact = false +}: { + timestamp: number + compact?: boolean +}): React.JSX.Element { + // Why: rows reuse their identity across store writes, so this label can't rely on + // incidental re-renders to stay honest; the shared visibility-gated clock re-renders + // only this leaf (memo'd rows stay bailed out). 30s cadence matches WorktreeCardAgents. + const now = useNow(30_000) const absolute = formatAbsoluteDate(timestamp) return ( @@ -43,60 +52,6 @@ export function EventTime({ timestamp }: { timestamp: number }): React.JSX.Eleme ) } -export function ActivityThreadOptionsMenu({ - compactMode, - hasUnreadThreads, - onCompactModeChange, - onMarkAllThreadsRead -}: { - compactMode: boolean - hasUnreadThreads: boolean - onCompactModeChange: (compactMode: boolean) => void - onMarkAllThreadsRead: () => void -}): React.JSX.Element { - return ( - - - - {/* Why: keep Tooltip and Dropdown from composing refs onto the same button (Radix setRef crash loop). */} - - - - - - - - {translate('auto.components.activity.ActivityPrototypePage.a472a14700', 'More options')} - - - - onCompactModeChange(checked === true)} - onSelect={(event) => event.preventDefault()} - > - {translate('auto.components.activity.ActivityPrototypePage.f70e4bec47', 'Compact mode')} - - - - {translate('auto.components.activity.ActivityPrototypePage.023ff75afe', 'Mark all read')} - - - - ) -} - export function ActivityProjectLabel({ repo }: { repo: Repo | null }): React.JSX.Element { const label = repo?.displayName?.trim() || @@ -150,21 +105,54 @@ export function ThreadAgentStateIndicator({ } export function ActivityStatusGroupHeader({ - group + group, + collapsed = false, + onToggle, + className }: { group: ActivityThreadGroup + collapsed?: boolean + onToggle?: () => void + className?: string }): React.JSX.Element { + const isInteractive = Boolean(onToggle) return ( -
    +
    { + if (event.key === 'Enter' || event.key === ' ') { + event.preventDefault() + onToggle?.() + } + } + : undefined + } + className={cn( + 'group flex h-7 select-none items-center gap-1.5 rounded-md px-1.5 py-1 text-muted-foreground transition-colors', + isInteractive && 'cursor-pointer hover:bg-accent/60 hover:text-foreground', + className + )} + > + {group.state ? ( ) : null} - + {group.label} - + {group.threads.length}
    diff --git a/src/renderer/src/components/activity/activity-thread-grouping.search-cache.test.ts b/src/renderer/src/components/activity/activity-thread-grouping.search-cache.test.ts new file mode 100644 index 00000000000..4c5b003eac4 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-grouping.search-cache.test.ts @@ -0,0 +1,53 @@ +import { describe, expect, it } from 'vitest' +import { + activityThreadMatchesSearchQuery, + getThreadSearchTextComputeCount +} from './activity-thread-grouping' +import type { AgentPaneThread } from './activity-thread-types' +import { makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' + +function makeThread(paneKey: string, paneTitle: string): AgentPaneThread { + return { + paneKey, + tab: makeTab(), + worktree: makeWorktree(), + repo: null, + currentAgentState: null, + currentAgentEntry: null, + latestEvent: null, + latestTimestamp: 1_000, + agentType: 'claude', + unread: false, + paneTitle, + responsePreview: 'x'.repeat(2_000), + events: [] + } +} + +describe('activity thread search text cache', () => { + it('builds a thread searchable text once per thread identity across keystrokes', () => { + const thread = makeThread('tab-1:leaf-1', 'Refactor billing pipeline') + const before = getThreadSearchTextComputeCount() + // Simulate typing a query letter by letter against the same thread objects. + for (const searchQuery of ['r', 're', 'ref', 'refa', 'refac']) { + expect(activityThreadMatchesSearchQuery({ thread, searchQuery })).toBe(true) + } + expect(getThreadSearchTextComputeCount() - before).toBe(1) + }) + + it('recomputes when thread data changes (new thread identity)', () => { + const before = getThreadSearchTextComputeCount() + const first = makeThread('tab-1:leaf-1', 'First title') + const rebuilt = makeThread('tab-1:leaf-1', 'Second title') + expect(activityThreadMatchesSearchQuery({ thread: first, searchQuery: 'first' })).toBe(true) + expect(activityThreadMatchesSearchQuery({ thread: rebuilt, searchQuery: 'second' })).toBe(true) + expect(getThreadSearchTextComputeCount() - before).toBe(2) + }) + + it('keeps match semantics: state labels, workspace, and previews still match', () => { + const thread = makeThread('tab-1:leaf-1', 'My task') + expect(activityThreadMatchesSearchQuery({ thread, searchQuery: 'feature' })).toBe(true) + expect(activityThreadMatchesSearchQuery({ thread, searchQuery: 'zzz-no-match' })).toBe(false) + expect(activityThreadMatchesSearchQuery({ thread, searchQuery: '' })).toBe(true) + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-grouping.ts b/src/renderer/src/components/activity/activity-thread-grouping.ts index 1744c0b84f6..0730aa68b5d 100644 --- a/src/renderer/src/components/activity/activity-thread-grouping.ts +++ b/src/renderer/src/components/activity/activity-thread-grouping.ts @@ -18,19 +18,23 @@ import type { AgentPaneThread } from './activity-thread-types' +// Attention-needing groups first (interrupted included: it's stopped and awaiting the user) so they're never buried under Working/Done. const ACTIVITY_STATUS_GROUP_ORDER: ActivityStatusGroupId[] = [ + 'waiting', + 'blocked', + 'interrupted', 'working', 'monitoring', - 'blocked', - 'waiting', - 'done', - 'interrupted' + 'done' ] export function getActivityThreadGroup( thread: AgentPaneThread, groupBy: ActivityGroupBy ): { key: string; label: string } { + if (groupBy === 'none') { + return { key: 'all', label: '' } + } if (groupBy === 'status') { const state = threadAgentState(thread) if (!thread.currentAgentState && state === 'done' && thread.latestEvent?.entry.interrupted) { @@ -59,6 +63,9 @@ export function buildActivityThreadGroups( threads: AgentPaneThread[], groupBy: ActivityGroupBy ): ActivityThreadGroup[] { + if (groupBy === 'none') { + return threads.length > 0 ? [{ key: 'all', label: '', threads }] : [] + } const groups: ActivityThreadGroup[] = [] const groupIndexByKey = new Map() for (const thread of threads) { @@ -74,7 +81,7 @@ export function buildActivityThreadGroups( return groups } -function threadStatusGroupId(thread: AgentPaneThread): ActivityStatusGroupId { +export function threadStatusGroupId(thread: AgentPaneThread): ActivityStatusGroupId { const state = threadAgentState(thread) if (!thread.currentAgentState && state === 'done' && thread.latestEvent?.entry.interrupted) { return 'interrupted' @@ -118,7 +125,7 @@ export function groupActivityThreadsByStatus(threads: AgentPaneThread[]): Activi }) } -function threadSearchText(thread: AgentPaneThread): string { +function buildThreadSearchText(thread: AgentPaneThread): string { const latest = thread.latestEvent const stateLabel = threadAgentStateLabel(thread) const currentPrompt = thread.currentAgentEntry @@ -132,6 +139,28 @@ function threadSearchText(thread: AgentPaneThread): string { return `${thread.paneTitle} ${getActivityThreadWorkspaceTitle(thread.worktree)} ${thread.worktree.branch ?? ''} ${thread.repo?.displayName ?? ''} ${formatAgentTypeLabel(thread.agentType)} ${stateLabel} ${currentPrompt} ${rawCurrentPrompt} ${currentSummary} ${thread.responsePreview} ${latestEventText}`.toLowerCase() } +// Why: thread objects are rebuilt only when the underlying store data changes, so their +// identity is a correct cache key; without this every keystroke re-lowercases a large +// string per thread. WeakMap so dropped threads release their text. +const threadSearchTextCache = new WeakMap() +let threadSearchTextComputeCount = 0 + +/** Test hook: how many times search text was actually (re)built. */ +export function getThreadSearchTextComputeCount(): number { + return threadSearchTextComputeCount +} + +function threadSearchText(thread: AgentPaneThread): string { + const cached = threadSearchTextCache.get(thread) + if (cached !== undefined) { + return cached + } + threadSearchTextComputeCount += 1 + const text = buildThreadSearchText(thread) + threadSearchTextCache.set(thread, text) + return text +} + export const ACTIVITY_SEARCH_QUERY_MAX_BYTES = 2 * 1024 export function isActivitySearchQueryTooLarge( diff --git a/src/renderer/src/components/activity/activity-thread-hover-card-summary.tsx b/src/renderer/src/components/activity/activity-thread-hover-card-summary.tsx new file mode 100644 index 00000000000..37b0658f0d9 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-hover-card-summary.tsx @@ -0,0 +1,230 @@ +import React, { useCallback, useMemo } from 'react' +import { Cloud, Copy, FolderGit2, GitBranch, Laptop, LocateFixed, Server } from 'lucide-react' +import { toast } from 'sonner' +import { useAppStore } from '@/store' +import { RepoBadgeMark } from '@/components/repo/RepoBadgeLabel' +import { AgentIcon } from '@/lib/agent-catalog' +import { agentTypeToIconAgent, formatAgentTypeLabel } from '@/lib/agent-status' +import { getWorktreeGitIdentityDisplay } from '@/lib/worktree-git-identity-display' +import { jumpToWorktreeFromSidebar } from '@/lib/worktree-jump-navigation' +import { translate } from '@/i18n/i18n' +import { cn } from '@/lib/utils' +import { + getExecutionHostLabel, + getWorktreeExecutionHostId, + parseExecutionHostId +} from '../../../../shared/execution-host' +import type { GlobalSettings } from '../../../../shared/global-settings-types' +import { getHostDisplayLabelOverrides } from '../../../../shared/host-setting-overrides' +import CommentMarkdown from '../sidebar/CommentMarkdown' +import { DetailHeader, MetadataActionIcon } from '../sidebar/WorktreeCardMetadataControls' +import { + WorktreeCardDetailSection, + WorktreeCardDetailSectionContent +} from '../sidebar/WorktreeCardDetailSection' +import { EventTime, ThreadAgentStateIndicator } from './activity-thread-controls' +import { activityThreadRowCopy, threadAgentStateLabel } from './activity-thread-presentation' +import { getActivityThreadWorkspaceTitle } from '@/lib/activity-thread-display' +import type { AgentPaneThread } from './activity-thread-types' + +export function ActivityThreadHoverCardSummary({ + thread, + settings, + onJumpToWorkspace, + canJumpToWorkspace +}: { + thread: AgentPaneThread + settings: GlobalSettings | null | undefined + onJumpToWorkspace?: (event: React.MouseEvent) => void + canJumpToWorkspace?: boolean +}): React.JSX.Element { + const { worktree, repo } = thread + const sshTargetLabels = useAppStore((s) => s.sshTargetLabels) + const runtimeEnvironments = useAppStore((s) => s.runtimeEnvironments) + const executionHostId = getWorktreeExecutionHostId(worktree, repo ?? undefined) + const parsedHost = parseExecutionHostId(executionHostId) + const hostLabelOverrides = useMemo(() => getHostDisplayLabelOverrides(settings), [settings]) + const hostDisplayLabel = useMemo(() => { + const override = hostLabelOverrides.get(executionHostId) + if (override) { + return override + } + if (parsedHost?.kind === 'runtime') { + const environment = runtimeEnvironments.find((entry) => entry.id === parsedHost.environmentId) + if (environment?.name) { + return environment.name + } + } + if (parsedHost?.kind === 'ssh') { + const target = sshTargetLabels.get(parsedHost.targetId) + if (target) { + return target + } + } + return getExecutionHostLabel(executionHostId) + }, [executionHostId, hostLabelOverrides, parsedHost, runtimeEnvironments, sshTargetLabels]) + const branchIdentityDisplay = useMemo(() => getWorktreeGitIdentityDisplay(worktree), [worktree]) + const { taskTitle, needsAttention } = activityThreadRowCopy(thread) + const workspaceTitle = getActivityThreadWorkspaceTitle(worktree) + const copyPathLabel = translate( + 'auto.components.activity.ActivityThreadHoverCard.copyPath', + 'Copy path' + ) + + const isKnownWorktree = useAppStore((s) => + Boolean(s.getKnownWorktreeById(worktree.id, executionHostId)) + ) + const canJump = canJumpToWorkspace ?? isKnownWorktree + + const handleJumpToWorkspace = useCallback( + (event: React.MouseEvent) => { + event.stopPropagation() + if (onJumpToWorkspace) { + onJumpToWorkspace(event) + } else { + const state = useAppStore.getState() + if (state.getKnownWorktreeById(worktree.id, executionHostId)) { + state.acknowledgeAgents([thread.paneKey]) + jumpToWorktreeFromSidebar(worktree.id, { executionHostId }) + } + } + }, + [executionHostId, onJumpToWorkspace, thread.paneKey, worktree.id] + ) + + const handleCopyPath = useCallback(async () => { + if (!worktree.path) { + return + } + try { + await window.api.ui.writeClipboardText(worktree.path) + toast.success( + translate( + 'auto.components.activity.ActivityThreadHoverCard.pathCopied', + 'Path copied to clipboard' + ) + ) + } catch { + toast.error( + translate( + 'auto.components.activity.ActivityThreadHoverCard.copyPathFailed', + 'Failed to copy path' + ) + ) + } + }, [worktree.path]) + + return ( + <> +
    +
    +
    + + + + + {formatAgentTypeLabel(thread.agentType)} + +
    +
    + + + {threadAgentStateLabel(thread)} + + • + +
    +
    + +
    + {taskTitle} +
    + + {thread.responsePreview ? ( +
    + +
    + ) : null} +
    + + + } + label={translate( + 'auto.components.activity.ActivityThreadHoverCard.workspace', + 'Workspace' + )} + actions={ + canJump ? ( + + + + ) : null + } + /> + +
    + {repo ? ( +
    + + + {repo.displayName} + +
    + ) : null} + + {workspaceTitle} + +
    + + {branchIdentityDisplay ? ( +
    + + + {branchIdentityDisplay.kind === 'branch' + ? branchIdentityDisplay.branchName + : branchIdentityDisplay.sidebarLabel} + +
    + ) : null} + +
    + {parsedHost?.kind === 'runtime' ? ( + + ) : parsedHost?.kind === 'ssh' ? ( + + ) : ( + + )} + {hostDisplayLabel} +
    + + {worktree.path ? ( +
    + + {worktree.path} + + + + +
    + ) : null} +
    +
    + + ) +} diff --git a/src/renderer/src/components/activity/activity-thread-hover-card.test.tsx b/src/renderer/src/components/activity/activity-thread-hover-card.test.tsx new file mode 100644 index 00000000000..0e604a16098 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-hover-card.test.tsx @@ -0,0 +1,238 @@ +// @vitest-environment happy-dom + +import React, { act, type ReactElement, type ReactNode } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { TooltipProvider } from '@/components/ui/tooltip' +import type { Worktree } from '../../../../shared/worktree/types' +import { ActivityThreadHoverCard } from './activity-thread-hover-card' +import { ActivityThreadRow } from './activity-thread-row' +import type { AgentPaneThread } from './activity-thread-types' +import { makeRepo, makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' + +vi.mock('@/components/ui/hover-card', () => ({ + HoverCard: ({ + children, + onOpenChange + }: { + children: ReactNode + onOpenChange?: (open: boolean) => void + }) => { + React.useEffect(() => onOpenChange?.(true), [onOpenChange]) + return <>{children} + }, + HoverCardContent: ({ children, className }: { children: ReactNode; className?: string }) => ( +
    + {children} +
    + ), + HoverCardTrigger: ({ children }: { children: ReactNode }) => <>{children} +})) + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +function createTestThread(overrides: Partial = {}): AgentPaneThread { + const repo = makeRepo() + const worktree: Worktree = { + ...makeWorktree(), + displayName: 'm4air-audit', + branch: 'feat/m4air-performance', + path: '/Users/test/projects/orca/worktrees/m4air-audit', + comment: 'Notes for performance audit', + hostId: 'runtime:m4air-env-id' as const + } + const tab = makeTab() + + return { + paneKey: 'tab-1:leaf-1', + tab, + worktree, + repo, + agentType: 'claude', + latestEvent: null, + currentAgentState: 'working', + currentAgentEntry: null, + events: [], + unread: false, + paneTitle: 'Audit current HEAD on m4air environment', + responsePreview: 'Auditing terminal PTY and IME bug categories...', + latestTimestamp: 1700000000000, + ...overrides + } +} + +function Harness({ + thread, + selected = false, + onSelect = vi.fn(), + onJump = vi.fn(), + onMarkRead = vi.fn(), + onMarkUnread = vi.fn() +}: { + thread: AgentPaneThread + selected?: boolean + onSelect?: () => void + onJump?: () => void + onMarkRead?: () => void + onMarkUnread?: () => void +}): ReactElement { + return ( + + + + ) +} + +describe('ActivityThreadHoverCard and ActivityThreadRow', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + }) + + afterEach(() => { + act(() => root.unmount()) + container.remove() + document.body.replaceChildren() + }) + + it('renders the thread row with task title and workspace label', async () => { + const thread = createTestThread() + + await act(async () => { + root.render() + }) + + const card = container.querySelector('[data-worktree-card-surface="true"]') + expect(card).not.toBeNull() + expect(card?.getAttribute('role')).toBe('listitem') + expect( + card?.querySelector('button[aria-label="Audit current HEAD on m4air environment"]') + ).not.toBeNull() + expect(card?.textContent).toContain('Audit current HEAD on m4air environment') + expect(card?.textContent).toContain('m4air-audit') + }) + + it('marks an unread thread as read from its bell without selecting the row', async () => { + const thread = createTestThread({ unread: true }) + const onMarkRead = vi.fn() + const onSelect = vi.fn() + + await act(async () => { + root.render() + }) + + const markReadButton = container.querySelector( + 'button[aria-label="Mark thread as read"]' + ) + expect(markReadButton).not.toBeNull() + + act(() => { + markReadButton?.click() + }) + + expect(onMarkRead).toHaveBeenCalledWith(thread) + expect(onSelect).not.toHaveBeenCalled() + }) + + it('renders hover card content with workspace info, host, task details, and notes', async () => { + const thread = createTestThread({ + worktree: { + ...makeWorktree(), + displayName: 'm4air-audit', + branch: 'feat/m4air-performance', + path: '/Users/test/projects/orca/worktrees/m4air-audit', + comment: 'Performance investigation notes', + hostId: 'runtime:m4air-env' as const + } + }) + + await act(async () => { + root.render( + + +
    Hover Target
    +
    +
    + ) + }) + + const content = container.querySelector('[data-testid="hover-card-content"]')?.textContent ?? '' + + // Workspace & Host info + expect(content).toContain('Workspace') + expect(content).toContain('m4air-audit') + expect(content).toContain('feat/m4air-performance') + expect(content).toContain('/Users/test/projects/orca/worktrees/m4air-audit') + + // Agent & Task details + expect(content).toContain('Claude') + expect(content).toContain('Audit current HEAD on m4air environment') + expect(content).toContain('Auditing terminal PTY and IME bug categories...') + + // Notes + expect(content).toContain('Notes') + expect(content).toContain('Performance investigation notes') + }) + + it('allows clicking row while preventing inner hover interactions from bubbling', async () => { + const onSelect = vi.fn() + const thread = createTestThread() + + await act(async () => { + root.render() + }) + + const card = container.querySelector('[data-worktree-card-surface="true"]') + expect(card).not.toBeNull() + + await act(async () => { + card?.dispatchEvent(new MouseEvent('click', { bubbles: true })) + }) + + expect(onSelect).toHaveBeenCalledTimes(1) + }) + + it('triggers jump to workspace from the hover card crosshair locator button', async () => { + const onJump = vi.fn() + const thread = createTestThread() + + await act(async () => { + root.render( + + +
    Hover Target
    +
    +
    + ) + }) + + const jumpButton = container.querySelector( + 'button[aria-label="Jump to workspace"]' + ) + expect(jumpButton).not.toBeNull() + + act(() => { + jumpButton?.click() + }) + + expect(onJump).toHaveBeenCalledWith(thread) + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-hover-card.tsx b/src/renderer/src/components/activity/activity-thread-hover-card.tsx new file mode 100644 index 00000000000..9bb4ed83887 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-hover-card.tsx @@ -0,0 +1,407 @@ +import React, { useCallback } from 'react' +import { ExternalLink, MonitorUp, Pencil, StickyNote } from 'lucide-react' +import { toast } from 'sonner' +import { Badge } from '@/components/ui/badge' +import { HoverCard, HoverCardContent, HoverCardTrigger } from '@/components/ui/hover-card' +import { LinearIcon } from '@/components/icons/LinearIcon' +import { JiraIcon } from '@/components/icons/JiraIcon' +import { SelectedTextCopyMenu } from '@/components/SelectedTextCopyMenu' +import { translate } from '@/i18n/i18n' +import CommentMarkdown from '../sidebar/CommentMarkdown' +import { DetailHeader, MetadataActionIcon } from '../sidebar/WorktreeCardMetadataControls' +import { + WorktreeCardDetailSection, + WorktreeCardDetailSectionContent +} from '../sidebar/WorktreeCardDetailSection' +import { LinearStateBadge } from '../sidebar/WorktreeCardMetadataStatusBadges' +import { WorktreeCardIssueDetailSection } from '../sidebar/WorktreeCardIssueDetailSection' +import { WorktreeCardReviewDetailSection } from '../sidebar/WorktreeCardReviewDetailSection' +import { WorktreeCardAutomationDetailSection } from '../sidebar/WorktreeCardAutomationDetailSection' +import { WorktreeCardCliDetailSection } from '../sidebar/WorktreeCardCliDetailSection' +import { WorktreeCardPortsDetails } from '../sidebar/WorktreeCardPorts' +import { WORKTREE_NATIVE_CONTEXT_MENU_ATTR } from '../sidebar/WorktreeContextMenu' +import { useWorktreeCardDetailsHoverControl } from '../sidebar/worktree-card-details-hover-state' +import { useWorktreeCardFoundation } from '../sidebar/use-worktree-card-foundation' +import { useWorktreeCardReviewDetails } from '../sidebar/use-worktree-card-review-details' +import { useWorktreeCardLinkedDetails } from '../sidebar/use-worktree-card-linked-details' +import { useWorktreeCardLifecycleEffects } from '../sidebar/use-worktree-card-lifecycle-effects' +import { useWorktreeCardSecondaryDetails } from '../sidebar/use-worktree-card-secondary-details' +import { getReviewLabel } from '../sidebar/worktree-review-helpers' +import { ActivityThreadHoverCardSummary } from './activity-thread-hover-card-summary' +import type { AgentPaneThread } from './activity-thread-types' + +export type ActivityThreadHoverCardProps = { + thread: AgentPaneThread + children: React.ReactElement + openDelay?: number + closeDelay?: number + onJumpToWorkspace?: (thread: AgentPaneThread) => void + canJumpToWorkspace?: boolean +} + +export function ActivityThreadHoverCard({ + thread, + children, + openDelay = 200, + closeDelay = 120, + onJumpToWorkspace, + canJumpToWorkspace +}: ActivityThreadHoverCardProps): React.JSX.Element { + const detailsHoverControl = useWorktreeCardDetailsHoverControl() + + return ( + + {children} + {detailsHoverControl.hoverOpen ? ( + + ) : null} + + ) +} + +function ActivityThreadHoverCardContent({ + thread, + detailsHoverControl, + onJumpToWorkspace, + canJumpToWorkspace +}: { + thread: AgentPaneThread + detailsHoverControl: ReturnType + onJumpToWorkspace?: (thread: AgentPaneThread) => void + canJumpToWorkspace?: boolean +}): React.JSX.Element { + const { worktree, repo } = thread + const foundation = useWorktreeCardFoundation({ worktree, repo: repo ?? undefined }) + const review = useWorktreeCardReviewDetails({ + worktree, + repo: repo ?? undefined, + settings: foundation.settings, + projectGroups: foundation.projectGroups, + cardProps: foundation.cardProps, + newCardStyle: foundation.newCardStyle + }) + const linked = useWorktreeCardLinkedDetails({ + worktree, + newCardStyle: foundation.newCardStyle, + deleteState: foundation.deleteState, + branch: review.branch, + issueEntry: review.issueEntry, + linearIssueEntry: review.linearIssueEntry, + linearIssueFallbackEntry: review.linearIssueFallbackEntry, + prDisplay: review.prDisplay + }) + + const hoverDetailsOpen = detailsHoverControl.hoverOpen + + useWorktreeCardLifecycleEffects({ + worktree, + repo: repo ?? undefined, + isFolder: review.isFolder, + hostedReviewCacheKey: review.hostedReviewCacheKey, + cachedBranchFallbackGitHubPRNumber: review.cachedBranchFallbackGitHubPRNumber, + linkedGitLabMR: review.linkedGitLabMR, + linkedBitbucketPR: review.linkedBitbucketPR, + linkedAzureDevOpsPR: review.linkedAzureDevOpsPR, + linkedGiteaPR: review.linkedGiteaPR, + branch: review.branch, + fetchHostedReviewForBranch: foundation.fetchHostedReviewForBranch, + shouldRefreshHostedReview: false, + newCardStyle: true, + hoverDetailsOpen, + showIssue: true, + issueCacheKey: review.issueCacheKey, + fetchIssue: foundation.fetchIssue, + showLinearIssue: true, + fetchLinearIssue: foundation.fetchLinearIssue + }) + + const secondary = useWorktreeCardSecondaryDetails({ + worktree, + repo: repo ?? undefined, + statusPrDisplay: null, + showStatus: true, + showIssue: true, + showLinearIssue: true, + showJiraIssue: true, + showPR: true, + showAutomation: true, + showCli: true, + showComment: true, + showPorts: true, + issueDisplay: linked.issueDisplay, + linearIssue: linked.linearIssue, + linearIssueDisplay: linked.linearIssueDisplay, + jiraIssueDisplay: linked.jiraIssueDisplay, + prDisplay: review.prDisplay, + linkedGitLabMR: review.linkedGitLabMR, + linkedBitbucketPR: review.linkedBitbucketPR, + linkedAzureDevOpsPR: review.linkedAzureDevOpsPR, + linkedGiteaPR: review.linkedGiteaPR, + cardProps: foundation.cardProps, + newCardStyle: foundation.newCardStyle, + compactCards: foundation.compactCards, + agentActivityDisplayMode: foundation.agentActivityDisplayMode, + workspacePorts: foundation.workspacePorts, + openTaskPage: foundation.openTaskPage, + updateWorktreeMeta: foundation.updateWorktreeMeta, + settings: foundation.settings + }) + + const copyLinkedWorkItemLink = useCallback(async (url: string, label: string) => { + try { + await window.api.ui.writeClipboardText(url) + toast.success( + translate('auto.components.sidebar.WorktreeCardMeta.copyLinkSuccess', '{{value0}} copied', { + value0: label + }) + ) + } catch { + toast.error( + translate('auto.components.sidebar.WorktreeCardMeta.copyLinkFailure', 'Failed to copy link') + ) + } + }, []) + + const handleCopyIssueLink = useCallback(() => { + if (!secondary.hoverIssue?.url) { + return + } + detailsHoverControl.closeHover() + void copyLinkedWorkItemLink( + secondary.hoverIssue.url, + translate('auto.components.sidebar.WorktreeCardMeta.issueLinkLabel', 'Issue link') + ) + }, [copyLinkedWorkItemLink, detailsHoverControl, secondary.hoverIssue?.url]) + + const handleCopyReviewLink = useCallback(() => { + if (!secondary.hoverReview?.url) { + return + } + void copyLinkedWorkItemLink( + secondary.hoverReview.url, + translate('auto.components.sidebar.WorktreeCardMeta.reviewLinkLabel', '{{value0}} link', { + value0: getReviewLabel(secondary.hoverReview) + }) + ) + }, [copyLinkedWorkItemLink, secondary.hoverReview]) + + const dismissAndRun = useCallback( + (handler: ((event: React.MouseEvent) => void) | undefined) => (event: React.MouseEvent) => { + detailsHoverControl.closeHover() + handler?.(event) + }, + [detailsHoverControl] + ) + + return ( + event.stopPropagation()} + onDoubleClick={(event) => event.stopPropagation()} + > + + onJumpToWorkspace(thread)) : undefined + } + canJumpToWorkspace={canJumpToWorkspace} + /> + + {/* GitHub / GitLab Issue */} + { + detailsHoverControl.closeHover() + secondary.handleOpenIssueInBrowser(url) + } + : undefined + } + /> + + {/* Linear Issue */} + {secondary.hoverLinearIssue && ( + + } + label={translate( + 'auto.components.sidebar.WorktreeCardMeta.5e982e6128', + 'Linear {{value0}}', + { value0: secondary.hoverLinearIssue.identifier } + )} + actions={ + <> + {secondary.hoverLinearIssue.url && secondary.handleOpenLinearIssueInOrca && ( + + + + )} + {secondary.hoverLinearIssue.url && ( + + + + )} + + } + /> + +
    + {secondary.hoverLinearIssue.title} +
    + {((secondary.hoverLinearIssue.labels && + secondary.hoverLinearIssue.labels.length > 0) || + secondary.hoverLinearIssue.stateName) && ( +
    + {secondary.hoverLinearIssue.stateName && ( + + )} + {(secondary.hoverLinearIssue.labels ?? []).map((label) => ( + + {label} + + ))} +
    + )} +
    +
    + )} + + {/* Jira Issue */} + {secondary.hoverJiraIssue && ( + + } + label={translate( + 'auto.components.sidebar.WorktreeCardMeta.jiraIssue', + 'Jira {{value0}}', + { value0: secondary.hoverJiraIssue.identifier } + )} + actions={ + + + + } + /> + +
    + {secondary.hoverJiraIssue.title} +
    +
    +
    + )} + + {/* Pull Request / Review */} + + + {/* Automation Provenance */} + {secondary.metaAutomationProvenance && ( + + )} + + {/* CLI Provenance */} + {secondary.metaCliProvenance && ( + + )} + + {/* Notes / Comment */} + {(secondary.hoverComment ?? '').trim().length > 0 && ( + + } + label={translate('auto.components.sidebar.WorktreeCardMeta.93cbea12c2', 'Notes')} + actions={ + + + + } + /> + + + + + )} + + {/* Ports */} + {foundation.workspacePorts.length > 0 && ( + + )} +
    +
    + ) +} diff --git a/src/renderer/src/components/activity/activity-thread-list-pane-collapsible.test.tsx b/src/renderer/src/components/activity/activity-thread-list-pane-collapsible.test.tsx new file mode 100644 index 00000000000..6db25785dc6 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-list-pane-collapsible.test.tsx @@ -0,0 +1,292 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { TooltipProvider } from '@/components/ui/tooltip' +import { ActivityStatusGroupHeader } from './activity-thread-controls' +import { ActivityThreadListPane } from './activity-thread-list-pane' +import type { ActivityThreadGroup, AgentPaneThread } from './activity-thread-types' +import { makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +const mockThread: AgentPaneThread = { + paneKey: 'tab-1:agent-1', + tab: makeTab(), + worktree: makeWorktree(), + repo: null, + currentAgentState: null, + currentAgentEntry: null, + latestEvent: null, + latestTimestamp: 1000, + agentType: 'claude', + unread: false, + paneTitle: 'Test agent', + responsePreview: 'Done testing', + events: [] +} + +const mockGroup: ActivityThreadGroup = { + key: 'done', + label: 'Done', + state: 'done', + threads: [mockThread] +} + +describe('ActivityStatusGroupHeader', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + }) + + afterEach(() => { + act(() => root.unmount()) + container.remove() + document.body.replaceChildren() + }) + + it('renders group label, count, and expanded state', () => { + const onToggle = vi.fn() + act(() => { + root.render( + + + + ) + }) + + const header = container.querySelector('[role="button"]') + expect(header).not.toBeNull() + expect(header?.getAttribute('aria-expanded')).toBe('true') + expect(header?.textContent).toContain('Done') + expect(header?.textContent).toContain('1') + }) + + it('triggers onToggle on click and on keydown (Enter / Space)', () => { + const onToggle = vi.fn() + act(() => { + root.render( + + + + ) + }) + + const header = container.querySelector('[role="button"]') as HTMLElement + act(() => { + header.dispatchEvent(new MouseEvent('click', { bubbles: true })) + }) + expect(onToggle).toHaveBeenCalledTimes(1) + + act(() => { + header.dispatchEvent(new KeyboardEvent('keydown', { key: 'Enter', bubbles: true })) + }) + expect(onToggle).toHaveBeenCalledTimes(2) + + act(() => { + header.dispatchEvent(new KeyboardEvent('keydown', { key: ' ', bubbles: true })) + }) + expect(onToggle).toHaveBeenCalledTimes(3) + }) + + it('renders collapsed state with aria-expanded false', () => { + const onToggle = vi.fn() + act(() => { + root.render( + + + + ) + }) + + const header = container.querySelector('[role="button"]') + expect(header?.getAttribute('aria-expanded')).toBe('false') + }) + + it('supports custom className and uses accessible contrast tokens without hardcoded white background', () => { + act(() => { + root.render( + + + + ) + }) + + const header = container.querySelector('div') + expect(header?.className).toContain('custom-header-class') + expect(header?.className).not.toContain('bg-background/95') + }) +}) + +describe('ActivityThreadListPane collapsible sections', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + }) + + afterEach(() => { + act(() => root.unmount()) + container.remove() + document.body.replaceChildren() + }) + + it('toggles thread visibility when header is clicked in uncontrolled mode', () => { + const inputRef = { current: null } + act(() => { + root.render( + + true} + showFilterControls={false} + showOptionsMenu={false} + /> + + ) + }) + + // Initially open: thread row should be visible + expect(container.textContent).toContain('Test agent') + + // Click header to collapse + const header = container.querySelector('[role="button"]') as HTMLElement + act(() => { + header.dispatchEvent(new MouseEvent('click', { bubbles: true })) + }) + + // Thread row should now be hidden + expect(container.textContent).not.toContain('Test agent') + expect(header.getAttribute('aria-expanded')).toBe('false') + + // Click header again to re-expand + act(() => { + header.dispatchEvent(new MouseEvent('click', { bubbles: true })) + }) + + expect(container.textContent).toContain('Test agent') + expect(header.getAttribute('aria-expanded')).toBe('true') + }) + + it('respects controlled collapsedGroupKeys and invokes onToggleGroupCollapse', () => { + const inputRef = { current: null } + const onToggleGroup = vi.fn() + act(() => { + root.render( + + true} + showFilterControls={false} + showOptionsMenu={false} + collapsedGroupKeys={new Set(['done'])} + onToggleGroupCollapse={onToggleGroup} + /> + + ) + }) + + // Controlled as collapsed: thread row should not be rendered + expect(container.textContent).not.toContain('Test agent') + + // Click header + const header = container.querySelector('[role="button"]') as HTMLElement + act(() => { + header.dispatchEvent(new MouseEvent('click', { bubbles: true })) + }) + + expect(onToggleGroup).toHaveBeenCalledWith('done') + }) + + it('keeps mark-unread enabled for the thread whose terminal pane is selected', () => { + const inputRef = { current: null } + const onMarkThreadUnread = vi.fn() + act(() => { + root.render( + + true} + allowMarkUnreadWhenSelected + showFilterControls={false} + showOptionsMenu={false} + /> + + ) + }) + + const markUnreadButton = container.querySelector( + 'button[aria-label="Mark thread unread"]' + ) as HTMLButtonElement | null + expect(markUnreadButton).not.toBeNull() + expect(markUnreadButton?.disabled).toBe(false) + + act(() => { + markUnreadButton?.click() + }) + expect(onMarkThreadUnread).toHaveBeenCalledWith(mockThread) + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-list-pane.tsx b/src/renderer/src/components/activity/activity-thread-list-pane.tsx index 4411d385e0b..9c2beee6e0c 100644 --- a/src/renderer/src/components/activity/activity-thread-list-pane.tsx +++ b/src/renderer/src/components/activity/activity-thread-list-pane.tsx @@ -1,19 +1,29 @@ -import React from 'react' -import { BellDot, Search } from 'lucide-react' -import { Input } from '@/components/ui/input' +import React, { useCallback, useContext, useEffect, useMemo, useRef, useState } from 'react' import { - Select, - SelectContent, - SelectItem, - SelectTrigger, - SelectValue -} from '@/components/ui/select' -import { Toggle } from '@/components/ui/toggle' -import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' + defaultRangeExtractor, + measureElement as measureVirtualElementSize, + observeElementRect, + useVirtualizer, + type Range +} from '@tanstack/react-virtual' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' -import { ActivityStatusGroupHeader, ActivityThreadOptionsMenu } from './activity-thread-controls' -import { ActivityThreadRow } from './activity-thread-row' +import { ActivityThreadListToolbar } from './activity-thread-list-toolbar' +import { + getActiveStickyHeaderIndex, + getActiveStickyHeaderIndexForScroll, + getPreviousStickyHeaderIndex +} from '../sidebar/worktree-list/viewport/virtual-rows' +import { ActivityThreadVirtualRow } from './activity-thread-virtual-row' +import { ActivityThreadListResizeHandle } from './activity-thread-list-resize-handle' +import { + buildActivityVirtualItems, + estimateActivityVirtualItemSize, + findActivityThreadItemIndex, + getActivityHeaderItemIndexes, + getActivityVirtualItemKey +} from './activity-thread-virtual-items' +import { ActivityThreadCollapseContext } from './activity-thread-collapse-context' import type { ActivityGroupBy, ActivityThreadGroup, @@ -21,6 +31,16 @@ import type { ThreadReadFilter } from './activity-thread-types' +const ZERO_RECT_FALLBACK_VIEWPORT = { width: 320, height: 600 } +const observeActivityListRect: typeof observeElementRect = (instance, cb) => + observeElementRect(instance, (rect) => { + cb(rect.height > 0 ? rect : ZERO_RECT_FALLBACK_VIEWPORT) + }) + +// A saved offset the content cannot contain yet is restored once it can; past +// this window it is stale (the list shrank) and restoring would yank the viewport. +const DEFERRED_SCROLL_RESTORE_WINDOW_MS = 3000 + export function ActivityThreadListPane({ threadListRef, threadListWidth, @@ -32,21 +52,34 @@ export function ActivityThreadListPane({ readFilter, onReadFilterChange, compactMode, + showChildAgents, hasUnreadThreads, onCompactModeChange, + onShowChildAgentsChange, onMarkAllThreadsRead, + hasCompletedThreads, + onClearCompleted, visibleThreadGroups, visibleThreadCount, selectedPaneKey, onSelectThread, onJumpToWorkspace, + onMarkThreadRead, onMarkThreadUnread, canJumpToWorkspace, + allowMarkUnreadWhenSelected = false, + showJumpAction = true, isThreadListResizing, - onResizeStart + onResizeStart, + showFilterControls = true, + showOptionsMenu = true, + scopeFilterRow, + collapsedGroupKeys, + onToggleGroupCollapse, + scrollTopRef }: { - threadListRef: React.RefObject - threadListWidth: number + threadListRef?: React.RefObject + threadListWidth?: number activityFilterInputRef: React.RefObject query: string onQueryChange: (query: string) => void @@ -55,163 +88,308 @@ export function ActivityThreadListPane({ readFilter: ThreadReadFilter onReadFilterChange: (readFilter: ThreadReadFilter) => void compactMode: boolean + showChildAgents?: boolean hasUnreadThreads: boolean onCompactModeChange: (compactMode: boolean) => void - onMarkAllThreadsRead: () => void + onShowChildAgentsChange?: (showChildAgents: boolean) => void + onMarkAllThreadsRead?: () => void + hasCompletedThreads?: boolean + onClearCompleted?: () => void visibleThreadGroups: ActivityThreadGroup[] visibleThreadCount: number selectedPaneKey: string | null onSelectThread: (thread: AgentPaneThread) => void onJumpToWorkspace: (thread: AgentPaneThread) => void + onMarkThreadRead: (thread: AgentPaneThread) => void onMarkThreadUnread: (thread: AgentPaneThread) => void canJumpToWorkspace: (thread: AgentPaneThread) => boolean - isThreadListResizing: boolean - onResizeStart: React.MouseEventHandler + allowMarkUnreadWhenSelected?: boolean + showJumpAction?: boolean + isThreadListResizing?: boolean + onResizeStart?: React.MouseEventHandler + showFilterControls?: boolean + showOptionsMenu?: boolean + /** Rendered between the toolbar and the list; carries the active-scope chips row. */ + scopeFilterRow?: React.ReactNode + collapsedGroupKeys?: ReadonlySet + onToggleGroupCollapse?: (groupKey: string) => void + /** Optional view-local scroll memory; updated without triggering React renders. */ + scrollTopRef?: React.MutableRefObject }): React.JSX.Element { + const [internalCollapsedGroupKeys, setInternalCollapsedGroupKeys] = useState>( + () => new Set() + ) + // Precedence: explicit props, then a caller-owned context (hosts that unmount + // the pane on body switches), then pane-local state. + const contextCollapse = useContext(ActivityThreadCollapseContext) + const isControlled = collapsedGroupKeys !== undefined && onToggleGroupCollapse !== undefined + const effectiveCollapsedGroupKeys = isControlled + ? collapsedGroupKeys + : (contextCollapse?.collapsedGroupKeys ?? internalCollapsedGroupKeys) + const handleToggleGroup = isControlled + ? onToggleGroupCollapse + : (contextCollapse?.onToggleGroupCollapse ?? + ((groupKey: string) => { + setInternalCollapsedGroupKeys((prev) => { + const next = new Set(prev) + if (next.has(groupKey)) { + next.delete(groupKey) + } else { + next.add(groupKey) + } + return next + }) + })) + + const scrollContainerRef = useRef(null) + const hasRestoredScrollRef = useRef(false) + const handleScroll = useCallback( + (event: React.UIEvent) => { + if (!scrollTopRef) { + return + } + const scrollTop = event.currentTarget.scrollTop + // A clamp-to-0 fired before the deferred restore must not wipe the saved offset. + if (!hasRestoredScrollRef.current) { + if (scrollTop === 0) { + return + } + hasRestoredScrollRef.current = true + } + scrollTopRef.current = scrollTop + }, + [scrollTopRef] + ) + const virtualItems = useMemo( + () => + buildActivityVirtualItems({ + groups: visibleThreadGroups, + groupBy, + collapsedGroupKeys: effectiveCollapsedGroupKeys + }), + [visibleThreadGroups, groupBy, effectiveCollapsedGroupKeys] + ) + const headerItemIndexes = useMemo( + () => getActivityHeaderItemIndexes(virtualItems), + [virtualItems] + ) + const selectedItemIndex = useMemo( + () => findActivityThreadItemIndex(virtualItems, selectedPaneKey), + [virtualItems, selectedPaneKey] + ) + + // Why keyed on virtualItems: getItemKey identity is a measurement-memo input in tanstack + // virtual. A per-render closure recomputes every row on unrelated re-renders; a fully stable + // one would miss same-count reorders. Changing exactly with the items is the correct middle. + const getItemKey = useCallback( + (index: number) => { + const item = virtualItems[index] + return item ? getActivityVirtualItemKey(item) : `__stale_${index}` + }, + [virtualItems] + ) + const virtualizer = useVirtualizer({ + count: virtualItems.length, + getScrollElement: () => scrollContainerRef.current, + estimateSize: (index) => estimateActivityVirtualItemSize(virtualItems[index], compactMode), + getItemKey, + measureElement: (element, entry, instance) => { + const measured = measureVirtualElementSize(element, entry, instance) + if (measured > 0) { + return measured + } + const index = Number.parseInt(element.getAttribute('data-index') ?? '', 10) + return estimateActivityVirtualItemSize( + Number.isNaN(index) ? undefined : virtualItems[index], + compactMode + ) + }, + rangeExtractor: useCallback( + (range: Range) => { + const activeStickyIndex = + groupBy !== 'none' + ? getActiveStickyHeaderIndex(headerItemIndexes, range.startIndex) + : null + const previousStickyIndex = + activeStickyIndex !== null + ? getPreviousStickyHeaderIndex(headerItemIndexes, activeStickyIndex) + : null + const indexSet = new Set(defaultRangeExtractor(range)) + if (activeStickyIndex !== null) { + indexSet.add(activeStickyIndex) + } + if (previousStickyIndex !== null) { + indexSet.add(previousStickyIndex) + } + if (selectedItemIndex !== null && selectedItemIndex >= 0) { + indexSet.add(selectedItemIndex) + } + return Array.from(indexSet).sort((a, b) => a - b) + }, + [groupBy, headerItemIndexes, selectedItemIndex] + ), + overscan: 8, + observeElementRect: observeActivityListRect, + useFlushSync: false + }) + + // Row heights differ between densities; drop stale measurements on toggle (not on mount). + const measuredCompactModeRef = useRef(compactMode) + useEffect(() => { + if (measuredCompactModeRef.current === compactMode) { + return + } + measuredCompactModeRef.current = compactMode + virtualizer.measure() + }, [virtualizer, compactMode]) + + // Restore only once the (estimated) content can contain the saved offset, so a + // pre-hydration mount doesn't clamp the restore to 0. + const totalSize = virtualizer.getTotalSize() + const restoreArmedAtRef = useRef(null) + useEffect(() => { + if (!scrollTopRef || hasRestoredScrollRef.current) { + return + } + if (restoreArmedAtRef.current === null) { + restoreArmedAtRef.current = Date.now() + } else if (Date.now() - restoreArmedAtRef.current > DEFERRED_SCROLL_RESTORE_WINDOW_MS) { + // The list stayed too small for the saved offset (it shrank); firing the + // restore on some later growth would yank the viewport out from the user. + hasRestoredScrollRef.current = true + scrollTopRef.current = 0 + return + } + const scrollContainer = scrollContainerRef.current + if (!scrollContainer) { + return + } + // Against the max offset, not the content height: a viewport taller than the + // remaining content clamps the assignment to 0 and burns the one restore. + const maxScrollTop = Math.max(0, totalSize - scrollContainer.clientHeight) + if (scrollTopRef.current > maxScrollTop) { + return + } + scrollContainer.scrollTop = scrollTopRef.current + hasRestoredScrollRef.current = true + }, [scrollTopRef, totalSize]) + + const scrollOffset = virtualizer.scrollOffset ?? 0 + const activeStickyHeaderIndex = + groupBy !== 'none' + ? getActiveStickyHeaderIndexForScroll({ + rangeStartIndex: virtualizer.range?.startIndex ?? 0, + scrollOffset, + stickyHeaderIndexes: headerItemIndexes, + virtualItems: virtualizer.getVirtualItems() + }) + : null + + const resizable = onResizeStart !== undefined return (
    -
    - {visibleThreadGroups.map((group) => ( -
    - - {group.threads.map((thread) => ( - onSelectThread(thread)} - onJump={() => onJumpToWorkspace(thread)} - onMarkUnread={() => onMarkThreadUnread(thread)} - canJump={canJumpToWorkspace(thread)} - compactMode={compactMode} - /> - ))} -
    - ))} - {visibleThreadCount === 0 ? ( -
    - {translate( - 'auto.components.activity.ActivityPrototypePage.7cd632006b', - 'No agent activity matches these filters.' - )} -
    - ) : null} -
    -
    -
    -
    + ) : null} ) } diff --git a/src/renderer/src/components/activity/activity-thread-list-pane.virtualization.test.tsx b/src/renderer/src/components/activity/activity-thread-list-pane.virtualization.test.tsx new file mode 100644 index 00000000000..5b531570444 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-list-pane.virtualization.test.tsx @@ -0,0 +1,237 @@ +// @vitest-environment happy-dom + +import { act, type MutableRefObject } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { TooltipProvider } from '@/components/ui/tooltip' +import { ActivityThreadListPane } from './activity-thread-list-pane' +import { + ActivityThreadCollapseContext, + type ActivityThreadCollapseState +} from './activity-thread-collapse-context' +import type { ActivityThreadGroup, AgentPaneThread } from './activity-thread-types' +import { makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +const THREAD_COUNT = 300 + +function makeThread(index: number): AgentPaneThread { + return { + paneKey: `tab-${index}:leaf-${index}`, + tab: makeTab(), + worktree: makeWorktree(), + repo: null, + currentAgentState: null, + currentAgentEntry: null, + latestEvent: null, + latestTimestamp: 1_000_000 - index, + agentType: 'claude', + unread: false, + paneTitle: `Virtual agent ${index}`, + responsePreview: '', + events: [] + } +} + +function makeManyThreads(): AgentPaneThread[] { + return Array.from({ length: THREAD_COUNT }, (_, index) => makeThread(index)) +} + +function makeGroups(threads: AgentPaneThread[]): ActivityThreadGroup[] { + return [{ key: 'done', label: 'Done', state: 'done', threads }] +} + +function renderPane( + root: Root, + args: { + threads: AgentPaneThread[] + selectedPaneKey?: string | null + scrollTopRef?: MutableRefObject + collapseState?: ActivityThreadCollapseState + } +): void { + act(() => { + root.render( + + + true} + showFilterControls={false} + showOptionsMenu={false} + scrollTopRef={args.scrollTopRef} + /> + + + ) + }) +} + +describe('ActivityThreadListPane virtualization', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + }) + + afterEach(() => { + act(() => root.unmount()) + container.remove() + document.body.replaceChildren() + }) + + function mountedRowCount(): number { + return container.querySelectorAll('[data-worktree-card-surface="true"]').length + } + + it('mounts a viewport-bounded number of rows, not one per thread', () => { + renderPane(root, { threads: makeManyThreads() }) + const mounted = mountedRowCount() + expect(mounted).toBeGreaterThan(0) + // Viewport (600px fallback) / ~96px rows + 2x overscan(8); far below THREAD_COUNT. + expect(mounted).toBeLessThanOrEqual(40) + // Off-screen rows are not in the DOM at all. + expect(container.textContent).not.toContain('Virtual agent 250') + }) + + it('keeps the selected off-screen row mounted so activation stays accessible', () => { + renderPane(root, { + threads: makeManyThreads(), + selectedPaneKey: 'tab-250:leaf-250' + }) + const selected = container.querySelector('[data-worktree-card-active="primary"]') + expect(selected).not.toBeNull() + expect(selected?.textContent).toContain('Virtual agent 250') + // Still virtualized: pinning the selection must not mount the rest of the list. + expect(mountedRowCount()).toBeLessThanOrEqual(41) + }) + + it('does not mount an off-screen row when its thread data updates', () => { + const threads = makeManyThreads() + renderPane(root, { threads }) + const before = mountedRowCount() + + const updated = [...threads] + updated[250] = { ...threads[250], paneTitle: 'Virtual agent 250 UPDATED', unread: true } + renderPane(root, { threads: updated }) + + expect(container.textContent).not.toContain('Virtual agent 250 UPDATED') + expect(mountedRowCount()).toBe(before) + }) + + it('renders every row for a short list', () => { + renderPane(root, { threads: [makeThread(0), makeThread(1), makeThread(2)] }) + expect(mountedRowCount()).toBe(3) + expect(container.textContent).toContain('Virtual agent 0') + expect(container.textContent).toContain('Virtual agent 2') + }) + + it('keeps the saved scroll offset when the pane mounts before threads hydrate', () => { + const scrollTopRef = { current: 360 } + renderPane(root, { threads: [], scrollTopRef }) + const scrollContainer = container.querySelector('.overflow-y-auto') + // Empty list cannot contain the offset: restore is deferred, not clamped to 0. + act(() => { + if (scrollContainer) { + scrollContainer.scrollTop = 0 + scrollContainer.dispatchEvent(new Event('scroll', { bubbles: true })) + } + }) + expect(scrollTopRef.current).toBe(360) + + renderPane(root, { threads: makeManyThreads(), scrollTopRef }) + expect(container.querySelector('.overflow-y-auto')?.scrollTop).toBe(360) + }) + + it('restores caller-held collapse state across remounts', () => { + // Models the sidebar: the caller owns the Set and provides it via context. + let collapsed: ReadonlySet = new Set() + const collapseState = (): ActivityThreadCollapseState => ({ + collapsedGroupKeys: collapsed, + onToggleGroupCollapse: (groupKey: string) => { + const next = new Set(collapsed) + if (next.has(groupKey)) { + next.delete(groupKey) + } else { + next.add(groupKey) + } + collapsed = next + } + }) + renderPane(root, { threads: makeManyThreads(), collapseState: collapseState() }) + const header = container.querySelector('[role="button"]') as HTMLElement + act(() => { + header.dispatchEvent(new MouseEvent('click', { bubbles: true })) + }) + renderPane(root, { threads: makeManyThreads(), collapseState: collapseState() }) + expect(container.querySelector('[role="button"]')?.getAttribute('aria-expanded')).toBe('false') + + act(() => root.unmount()) + root = createRoot(container) + renderPane(root, { threads: makeManyThreads(), collapseState: collapseState() }) + const remountedHeader = container.querySelector('[role="button"]') + expect(remountedHeader?.getAttribute('aria-expanded')).toBe('false') + }) + + it('disarms a stale deferred restore instead of yanking the viewport on late growth', () => { + vi.useFakeTimers() + vi.setSystemTime(1_000_000) + try { + const scrollTopRef = { current: 360 } + // Mounts with a list too small to contain the saved offset (it shrank). + renderPane(root, { threads: [makeThread(0)], scrollTopRef }) + const scrollContainer = container.querySelector('.overflow-y-auto') + expect(scrollContainer?.scrollTop).toBe(0) + + // Well past the restore window, the list grows beyond the saved offset. + vi.setSystemTime(1_010_000) + renderPane(root, { threads: makeManyThreads(), scrollTopRef }) + + expect(container.querySelector('.overflow-y-auto')?.scrollTop).toBe(0) + expect(scrollTopRef.current).toBe(0) + } finally { + vi.useRealTimers() + } + }) + + it('restores the Agents scroll position without storing it in React state', () => { + const scrollTopRef = { current: 240 } + renderPane(root, { threads: makeManyThreads(), scrollTopRef }) + const scrollContainer = container.querySelector('.overflow-y-auto') + expect(scrollContainer?.scrollTop).toBe(240) + + act(() => { + if (scrollContainer) { + scrollContainer.scrollTop = 360 + scrollContainer.dispatchEvent(new Event('scroll', { bubbles: true })) + } + }) + expect(scrollTopRef.current).toBe(360) + + act(() => root.unmount()) + root = createRoot(container) + renderPane(root, { threads: makeManyThreads(), scrollTopRef }) + expect(container.querySelector('.overflow-y-auto')?.scrollTop).toBe(360) + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-list-resize-handle.tsx b/src/renderer/src/components/activity/activity-thread-list-resize-handle.tsx new file mode 100644 index 00000000000..a97a7ce223b --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-list-resize-handle.tsx @@ -0,0 +1,37 @@ +import type React from 'react' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' + +export function ActivityThreadListResizeHandle({ + isResizing, + onResizeStart +}: { + isResizing?: boolean + onResizeStart?: React.MouseEventHandler +}): React.JSX.Element { + return ( +
    +
    +
    + ) +} diff --git a/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx b/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx new file mode 100644 index 00000000000..f1ad0fb9057 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx @@ -0,0 +1,227 @@ +import React from 'react' +import { BellDot, CheckCheck, Search, Trash2, X } from 'lucide-react' +import { Button } from '@/components/ui/button' +import { Input } from '@/components/ui/input' +import { + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue +} from '@/components/ui/select' +import { Toggle } from '@/components/ui/toggle' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import { ActivityThreadOptionsMenu } from './activity-thread-controls' +import type { ActivityGroupBy, ThreadReadFilter } from './activity-thread-types' + +export function ActivityThreadListToolbar({ + activityFilterInputRef, + query, + onQueryChange, + groupBy, + onGroupByChange, + readFilter, + onReadFilterChange, + compactMode, + showChildAgents, + hasUnreadThreads, + onCompactModeChange, + onShowChildAgentsChange, + onMarkAllThreadsRead, + hasCompletedThreads, + onClearCompleted, + resizable, + showFilterControls, + showOptionsMenu +}: { + activityFilterInputRef: React.RefObject + query: string + onQueryChange: (query: string) => void + groupBy: ActivityGroupBy + onGroupByChange: (groupBy: ActivityGroupBy) => void + readFilter: ThreadReadFilter + onReadFilterChange: (readFilter: ThreadReadFilter) => void + compactMode: boolean + showChildAgents?: boolean + hasUnreadThreads: boolean + onCompactModeChange: (compactMode: boolean) => void + onShowChildAgentsChange?: (showChildAgents: boolean) => void + onMarkAllThreadsRead?: () => void + hasCompletedThreads?: boolean + onClearCompleted?: () => void + resizable: boolean + showFilterControls: boolean + showOptionsMenu: boolean +}): React.JSX.Element | null { + const showToolbar = showFilterControls || showOptionsMenu + if (!showToolbar) { + return null + } + + return ( + <> +
    +
    + {showFilterControls ? ( +
    + + onQueryChange(event.target.value)} + placeholder={translate( + 'auto.components.activity.ActivityPrototypePage.795cbf26e2', + 'Filter...' + )} + className={cn('h-7 w-full pl-6 text-[11px]', query ? 'pr-6' : '')} + /> + {query ? ( + + ) : null} +
    + ) : null} + {resizable ? ( + + ) : null} + {showFilterControls ? ( + + + onReadFilterChange(pressed ? 'unread' : 'all')} + size="sm" + className={cn( + 'size-7 shrink-0 p-0 rounded-md transition-all', + readFilter === 'unread' + ? '!border border-primary/50 !bg-primary/20 !text-primary shadow-xs hover:!bg-primary/30' + : 'text-muted-foreground hover:text-foreground hover:bg-muted/50' + )} + aria-label={translate( + 'auto.components.activity.ActivityPrototypePage.d1a88df9a8', + 'Show unread threads only' + )} + > + + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.d1a88df9a8', + 'Show unread threads only' + )} + + + ) : null} + {showOptionsMenu ? ( + + ) : null} +
    +
    + {onMarkAllThreadsRead || onClearCompleted ? ( +
    + {onMarkAllThreadsRead ? ( + + ) : null} + {onClearCompleted ? ( + + ) : null} +
    + ) : null} + + ) +} diff --git a/src/renderer/src/components/activity/activity-thread-options-menu.tsx b/src/renderer/src/components/activity/activity-thread-options-menu.tsx new file mode 100644 index 00000000000..a7d50a108ac --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-options-menu.tsx @@ -0,0 +1,291 @@ +import React from 'react' +import { + BellDot, + Check, + CheckCheck, + GitFork, + Layers, + MoreVertical, + Rows3, + Search, + Trash2 +} from 'lucide-react' +import { Button } from '@/components/ui/button' +import { + DropdownMenu, + DropdownMenuCheckboxItem, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuRadioGroup, + DropdownMenuRadioItem, + DropdownMenuSeparator, + DropdownMenuSub, + DropdownMenuSubContent, + DropdownMenuSubTrigger, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { translate } from '@/i18n/i18n' +import { ActivityScopeFilterMenuSections } from './activity-scope-filter-controls' +import type { ActivityGroupBy } from './activity-thread-types' + +const ALIGNED_CHECKBOX_ITEM_CLASS = 'pl-2 [&>span.absolute]:hidden' + +function getActivityGroupByLabel(groupBy: ActivityGroupBy): string { + switch (groupBy) { + case 'none': + return translate('auto.components.activity.ActivityPrototypePage.none', 'None') + case 'status': + return translate('auto.components.activity.ActivityPrototypePage.4a3986b200', 'Status') + case 'project': + return translate('auto.components.activity.ActivityPrototypePage.8c3b621ddf', 'Project') + case 'worktree': + return translate('auto.components.activity.ActivityPrototypePage.b29191b3e0', 'Worktree') + case 'agent': + return translate('auto.components.activity.ActivityPrototypePage.f6396e1f85', 'Agent') + } +} + +export function ActivityThreadOptionsMenu({ + groupBy, + onGroupByChange, + compactMode, + showChildAgents = false, + hasUnreadThreads, + hasCompletedThreads = false, + onCompactModeChange, + onShowChildAgentsChange, + onMarkAllThreadsRead, + onClearCompleted, + onSearch, + unreadOnly = false, + onToggleUnread +}: { + groupBy?: ActivityGroupBy + onGroupByChange?: (groupBy: ActivityGroupBy) => void + compactMode: boolean + showChildAgents?: boolean + hasUnreadThreads: boolean + hasCompletedThreads?: boolean + onCompactModeChange: (compactMode: boolean) => void + onShowChildAgentsChange?: (showChildAgents: boolean) => void + onMarkAllThreadsRead?: () => void + onClearCompleted?: () => void + onSearch?: () => void + unreadOnly?: boolean + onToggleUnread?: () => void +}): React.JSX.Element { + const skipCloseAutoFocusRef = React.useRef(false) + + return ( + + + + {/* Why: keep Tooltip and Dropdown from composing refs onto the same button (Radix setRef crash loop). */} + + + + + + + + {translate('auto.components.activity.ActivityPrototypePage.a472a14700', 'More options')} + + + { + if (skipCloseAutoFocusRef.current) { + event.preventDefault() + skipCloseAutoFocusRef.current = false + } + }} + > + {onSearch || onToggleUnread ? ( + <> + {onSearch ? ( + { + skipCloseAutoFocusRef.current = true + onSearch() + }} + > + + + {translate('auto.components.activity.ActivityPrototypePage.search', 'Search')} + + + ) : null} + {onToggleUnread ? ( + + + onToggleUnread()} + onSelect={(event) => event.preventDefault()} + > + + + {translate( + 'auto.components.activity.ActivityPrototypePage.showUnreadOnly', + 'Show unread only' + )} + + {hasUnreadThreads ? ( + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.unreadOnlyDescription', + 'Filters the activity list to show only threads with unread updates.' + )} + + + ) : null} + + + ) : null} + + {groupBy && onGroupByChange ? ( + + + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.770d458144', + 'Group by' + )} + + + {getActivityGroupByLabel(groupBy)} + + + + + onGroupByChange(value as ActivityGroupBy)} + > + {[ + ['none', 'None', 'auto.components.activity.ActivityPrototypePage.none'], + ['status', 'Status', 'auto.components.activity.ActivityPrototypePage.4a3986b200'], + [ + 'project', + 'Project', + 'auto.components.activity.ActivityPrototypePage.8c3b621ddf' + ], + [ + 'worktree', + 'Worktree', + 'auto.components.activity.ActivityPrototypePage.b29191b3e0' + ], + ['agent', 'Agent', 'auto.components.activity.ActivityPrototypePage.f6396e1f85'] + ].map(([value, label, key]) => ( + event.preventDefault()} + > + {translate(key, label)} + + ))} + + + + ) : null} + + + onCompactModeChange(checked === true)} + onSelect={(event) => event.preventDefault()} + > + + + {translate( + 'auto.components.activity.ActivityPrototypePage.f70e4bec47', + 'Compact mode' + )} + + {compactMode ? : null} + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.compactModeDescription', + 'Shows shorter thread rows with one-line titles and two-line status messages.' + )} + + + {onShowChildAgentsChange ? ( + onShowChildAgentsChange(checked === true)} + onSelect={(event) => event.preventDefault()} + > + + + {translate( + 'auto.components.activity.ActivityPrototypePage.showChildAgents', + 'Show child agents' + )} + + {showChildAgents ? : null} + + ) : null} + {onMarkAllThreadsRead || onClearCompleted ? ( + <> + + {onMarkAllThreadsRead ? ( + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.023ff75afe', + 'Mark all read' + )} + + + ) : null} + {onClearCompleted ? ( + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.clearCompleted', + 'Clear completed' + )} + + + ) : null} + + ) : null} + + + ) +} diff --git a/src/renderer/src/components/activity/activity-thread-presentation.test.ts b/src/renderer/src/components/activity/activity-thread-presentation.test.ts new file mode 100644 index 00000000000..6aa3efc1a32 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-presentation.test.ts @@ -0,0 +1,97 @@ +import { describe, expect, it } from 'vitest' +import type { AgentPaneThread } from './activity-thread-types' +import { activityThreadRowCopy } from './activity-thread-presentation' +import { formatShortTimeAgo } from '@/lib/short-time-ago' +import { + makeRepo, + makeTabWithIds, + makeWorktree, + PANE_KEY +} from './ActivityPrototypePage-test-fixtures' + +function makeThread(overrides: Partial = {}): AgentPaneThread { + const worktree = makeWorktree() + return { + paneKey: PANE_KEY, + paneTitle: 'low hanging issues', + agentType: 'codex', + worktree, + repo: makeRepo(), + tab: makeTabWithIds('tab-1', worktree.id), + events: [], + latestEvent: null, + latestTimestamp: 1_000, + currentAgentState: null, + currentAgentEntry: null, + unread: false, + responsePreview: '', + ...overrides + } +} + +describe('formatShortTimeAgo', () => { + it('uses short units', () => { + const now = 1_000_000 + expect(formatShortTimeAgo(now - 10_000, now)).toBe('now') + expect(formatShortTimeAgo(now - 5 * 60_000, now)).toBe('5m') + expect(formatShortTimeAgo(now - 20 * 60 * 60_000, now)).toBe('20h') + expect(formatShortTimeAgo(now - 2 * 24 * 60 * 60_000, now)).toBe('2d') + }) +}) + +describe('activityThreadRowCopy', () => { + it('leads with the task and the last activity, not project or workspace', () => { + const copy = activityThreadRowCopy( + makeThread({ + responsePreview: 'Filed 8 issues from the audit.' + }) + ) + expect(copy.taskTitle).toBe('low hanging issues') + expect(copy.statusLine).toBe('Filed 8 issues from the audit.') + expect(copy.statusKind).toBe('message') + expect(copy.needsAttention).toBe(false) + expect(copy.workspaceLabel).toBe('feature') + }) + + it('names the live tool while working', () => { + const copy = activityThreadRowCopy( + makeThread({ + paneTitle: 'Fix checkout race', + currentAgentState: 'working', + responsePreview: 'Edit src/checkout/session.ts' + }) + ) + expect(copy.statusKind).toBe('tool') + expect(copy.statusLine).toBe('Edit src/checkout/session.ts') + }) + + it('falls back to a state label when a live agent has no preview', () => { + const copy = activityThreadRowCopy( + makeThread({ + paneTitle: 'Review PR 1842', + currentAgentState: 'waiting', + responsePreview: '' + }) + ) + expect(copy.statusKind).toBe('state') + expect(copy.statusLine).toBe('Waiting for input') + expect(copy.needsAttention).toBe(true) + }) + + it('does not invent a status line for a finished agent with no reply', () => { + const copy = activityThreadRowCopy(makeThread({ currentAgentState: null, responsePreview: '' })) + expect(copy.statusKind).toBe('none') + expect(copy.statusLine).toBe('') + }) + + it('does not repeat the task title as the last-message line', () => { + const copy = activityThreadRowCopy( + makeThread({ + paneTitle: 'low hanging issues', + responsePreview: 'low hanging issues' + }) + ) + expect(copy.statusKind).toBe('none') + expect(copy.statusLine).toBe('') + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-presentation.ts b/src/renderer/src/components/activity/activity-thread-presentation.ts index 167d2f30954..93d69c7688a 100644 --- a/src/renderer/src/components/activity/activity-thread-presentation.ts +++ b/src/renderer/src/components/activity/activity-thread-presentation.ts @@ -1,8 +1,10 @@ import { agentStateLabel, type AgentDotState } from '@/components/AgentStateDot' import { formatAgentTypeLabel } from '@/lib/agent-status' import { getAgentRowPrimaryText } from '@/lib/agent-row-primary-text' +import { showsAgentToolPreview } from '@/lib/agent-row-tool-preview' import { getActivityThreadTaskTitle, + getActivityThreadWorkspaceTitle, resolveActivityThreadStatusPreview } from '@/lib/activity-thread-display' import { formatUiRelativeTime } from '@/i18n/relative-time-format' @@ -24,8 +26,8 @@ export function formatAbsoluteDate(timestamp: number): string { return absoluteDateFormatter.format(new Date(timestamp)) } -export function formatRelativeTime(timestamp: number): string { - return formatUiRelativeTime(timestamp - Date.now()) +export function formatRelativeTime(timestamp: number, now = Date.now()): string { + return formatUiRelativeTime(timestamp - now) } function truncatePreservingSurrogates(value: string, maxLength: number): string { @@ -111,3 +113,57 @@ export function threadAgentStateLabel(thread: AgentPaneThread): string { } return agentStateLabel(state) } + +export type ActivityThreadStatusKind = 'tool' | 'message' | 'state' | 'none' + +export type ActivityThreadRowCopy = { + taskTitle: string + statusLine: string + statusKind: ActivityThreadStatusKind + needsAttention: boolean + workspaceLabel: string +} + +function normalizeScanLabel(value: string): string { + return value.trim().toLowerCase().replace(/[-_]+/g, ' ').replace(/\s+/g, ' ') +} + +function previewDuplicatesIdentity(preview: string, title: string, workspace: string): boolean { + const normalized = normalizeScanLabel(preview) + if (!normalized) { + return true + } + return normalized === normalizeScanLabel(title) || normalized === normalizeScanLabel(workspace) +} + +export function activityThreadRowCopy(thread: AgentPaneThread): ActivityThreadRowCopy { + const workspaceLabel = getActivityThreadWorkspaceTitle(thread.worktree) + const taskTitle = thread.paneTitle.trim() || workspaceLabel + const renderedPreview = activityThreadResponseRenderPreview({ + responsePreview: thread.responsePreview + }) + const liveState = thread.currentAgentState ?? thread.latestEvent?.state ?? null + const toolPreviewState = liveState === 'monitoring' ? null : liveState + const state = threadAgentState(thread) + const needsAttention = state === 'waiting' || state === 'blocked' || state === 'permission' + if (renderedPreview && !previewDuplicatesIdentity(renderedPreview, taskTitle, workspaceLabel)) { + return { + taskTitle, + statusLine: renderedPreview, + // Monitoring is a distinct live state, not a tool-running row state. + statusKind: showsAgentToolPreview(toolPreviewState) ? 'tool' : 'message', + needsAttention, + workspaceLabel + } + } + if (state !== 'done' && state !== 'idle') { + return { + taskTitle, + statusLine: threadAgentStateLabel(thread), + statusKind: 'state', + needsAttention, + workspaceLabel + } + } + return { taskTitle, statusLine: '', statusKind: 'none', needsAttention, workspaceLabel } +} diff --git a/src/renderer/src/components/activity/activity-thread-row.tsx b/src/renderer/src/components/activity/activity-thread-row.tsx index 77750f6c390..93d4d5ec18f 100644 --- a/src/renderer/src/components/activity/activity-thread-row.tsx +++ b/src/renderer/src/components/activity/activity-thread-row.tsx @@ -2,207 +2,213 @@ import React from 'react' import { Bell, ExternalLink } from 'lucide-react' import { AgentIcon } from '@/lib/agent-catalog' import { agentTypeToIconAgent, formatAgentTypeLabel } from '@/lib/agent-status' -import { getActivityThreadWorkspaceTitle } from '@/lib/activity-thread-display' import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' import { FilledBellIcon } from '../sidebar/WorktreeCardHelpers' import CommentMarkdown from '../sidebar/CommentMarkdown' -import { - ActivityProjectLabel, - EventTime, - ThreadAgentStateIndicator -} from './activity-thread-controls' -import { activityThreadResponseRenderPreview } from './activity-thread-presentation' +import { EventTime, ThreadAgentStateIndicator } from './activity-thread-controls' +import { ActivityThreadHoverCard } from './activity-thread-hover-card' +import { activityThreadRowCopy } from './activity-thread-presentation' import type { AgentPaneThread } from './activity-thread-types' -function isEventFromNestedInteractiveElement( - target: EventTarget | null, - currentTarget: HTMLElement -): boolean { - if (!(target instanceof HTMLElement)) { - return false - } - const interactiveTarget = target.closest( - 'a, button, input, select, textarea, [role="button"], [role="link"], [tabindex]:not([tabindex="-1"])' - ) - return ( - interactiveTarget instanceof HTMLElement && - interactiveTarget !== currentTarget && - currentTarget.contains(interactiveTarget) - ) -} - -export function ActivityThreadRow({ +// Why React.memo: rows are pure functions of these props; thread identity is stable across +// query/selection/group re-renders, so memo keeps a keystroke or selection change from +// re-rendering every mounted row. Callbacks take the thread so parents can pass stable handlers. +export const ActivityThreadRow = React.memo(function ActivityThreadRow({ thread, selected, onSelect, onJump, + onMarkRead, onMarkUnread, canJump, - compactMode + compactMode, + disableMarkUnread = false, + showJumpAction = true }: { thread: AgentPaneThread selected: boolean - onSelect: () => void - onJump: () => void - onMarkUnread: () => void + onSelect: (thread: AgentPaneThread) => void + onJump: (thread: AgentPaneThread) => void + onMarkRead: (thread: AgentPaneThread) => void + onMarkUnread: (thread: AgentPaneThread) => void canJump: boolean compactMode: boolean + disableMarkUnread?: boolean + showJumpAction?: boolean }): React.JSX.Element { - const renderedResponsePreview = activityThreadResponseRenderPreview({ - responsePreview: thread.responsePreview - }) - const workspaceTitle = getActivityThreadWorkspaceTitle(thread.worktree) - const taskTitle = thread.paneTitle + const { taskTitle, statusLine, statusKind, needsAttention, workspaceLabel } = + activityThreadRowCopy(thread) + const showMarkdownStatus = statusKind === 'message' const agentLabel = formatAgentTypeLabel(thread.agentType) - const showStatusPreview = - !compactMode && - renderedResponsePreview.length > 0 && - renderedResponsePreview !== taskTitle && - renderedResponsePreview !== workspaceTitle + return ( -
    { - // Why: markdown responses can contain links; keyboard activation on a nested link follows the link instead of selecting the row. - if (isEventFromNestedInteractiveElement(event.target, event.currentTarget)) { - return - } - if (event.key === 'Enter' || event.key === ' ') { - event.preventDefault() - onSelect() - } - }} - className={cn( - // Why (WorktreeCard cues): selected = tint+shadow, beats hover; unread = weight + left bar only; stacking all three confused selected vs unread on hover. - // Why (asymmetric padding): title leading-snug adds ~3px above cap-height; smaller top pad evens the row. - 'group relative flex w-full cursor-pointer flex-col gap-1 border-b border-border px-3 pt-2.5 pb-3 text-left transition-colors', - selected - ? 'bg-black/[0.08] shadow-[0_1px_2px_rgba(0,0,0,0.04)] dark:bg-white/[0.10] dark:shadow-[0_1px_2px_rgba(0,0,0,0.03)]' - : 'hover:bg-accent/40' - )} + - {thread.unread ? ( - - ) : null} -
    - - - - +
    onSelect(thread)} + role="listitem" + aria-label={taskTitle} + aria-current={selected ? 'true' : undefined} + className={cn( + 'group relative flex w-full cursor-pointer flex-col gap-1.5 rounded-lg border border-transparent px-1.5 py-2.5 text-left transition-[background-color,border-color,opacity,box-shadow] duration-200 outline-none select-none worktree-sidebar-card-hover focus-visible:ring-1 focus-visible:ring-ring', + selected && 'border-transparent' + )} + > +
    + + - -
    -
    -
    - -
    - {workspaceTitle} -
    - {taskTitle !== workspaceTitle ? ( -
    - {taskTitle} -
    - ) : null} - {showStatusPreview ? ( +
    + {/* Keep the activation target separate from markdown links and row actions. */} + + + {statusLine ? ( + showMarkdownStatus ? ( - ) : null} -
    - {agentLabel} - {canJump ? ( - - - - - - - {translate( + ) : ( +
    + {statusLine} +
    + ) + ) : null} + +
    + + + + + {workspaceLabel} + + {canJump && showJumpAction ? ( + + + +
    -
    - - + onClick={(event) => { + event.stopPropagation() + onJump(thread) + }} + onMouseDown={(event) => event.stopPropagation()} + > + + + + + {translate( + 'auto.components.activity.ActivityPrototypePage.4616ea39fd', + 'Jump to workspace' + )} + + + + ) : null} + {thread.unread ? ( - - ) : ( + + + {translate( + 'auto.components.activity.ActivityPrototypePage.markThreadRead', + 'Mark thread as read' + )} + + + ) : ( + + + @@ -214,11 +220,11 @@ export function ActivityThreadRow({ )} - - + +
    -
    + ) -} +}) diff --git a/src/renderer/src/components/activity/activity-thread-types.ts b/src/renderer/src/components/activity/activity-thread-types.ts index 7609d416581..7225a71b935 100644 --- a/src/renderer/src/components/activity/activity-thread-types.ts +++ b/src/renderer/src/components/activity/activity-thread-types.ts @@ -10,7 +10,7 @@ import type { Worktree } from '../../../../shared/worktree/types' import type { ActivityPortalReadinessStatus } from './activity-portal-readiness-oscillation' export type ThreadReadFilter = 'all' | 'unread' -export type ActivityGroupBy = 'status' | 'project' | 'worktree' | 'agent' +export type ActivityGroupBy = 'none' | 'status' | 'project' | 'worktree' | 'agent' export type ActivityEventState = Extract export type ActivityHookLiveAgentState = Extract< AgentStatusState, diff --git a/src/renderer/src/components/activity/activity-thread-virtual-items.test.ts b/src/renderer/src/components/activity/activity-thread-virtual-items.test.ts new file mode 100644 index 00000000000..041841bcad4 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-virtual-items.test.ts @@ -0,0 +1,82 @@ +import { describe, expect, it } from 'vitest' +import { + buildActivityVirtualItems, + findActivityThreadItemIndex, + getActivityHeaderItemIndexes, + getActivityVirtualItemKey +} from './activity-thread-virtual-items' +import type { ActivityThreadGroup, AgentPaneThread } from './activity-thread-types' +import { makeTab, makeWorktree } from './ActivityPrototypePage-test-fixtures' + +function makeThread(paneKey: string): AgentPaneThread { + return { + paneKey, + tab: makeTab(), + worktree: makeWorktree(), + repo: null, + currentAgentState: null, + currentAgentEntry: null, + latestEvent: null, + latestTimestamp: 1000, + agentType: 'claude', + unread: false, + paneTitle: `Agent ${paneKey}`, + responsePreview: '', + events: [] + } +} + +function makeGroup(key: string, threadKeys: string[]): ActivityThreadGroup { + return { key, label: key, threads: threadKeys.map(makeThread) } +} + +describe('buildActivityVirtualItems', () => { + it('flattens headers and threads in group order', () => { + const items = buildActivityVirtualItems({ + groups: [makeGroup('working', ['a', 'b']), makeGroup('done', ['c'])], + groupBy: 'status', + collapsedGroupKeys: new Set() + }) + expect(items.map((item) => getActivityVirtualItemKey(item))).toEqual([ + 'h:working', + 't:a', + 't:b', + 'h:done', + 't:c' + ]) + expect(getActivityHeaderItemIndexes(items)).toEqual([0, 3]) + }) + + it('omits header rows entirely when ungrouped', () => { + const items = buildActivityVirtualItems({ + groups: [{ key: 'all', label: '', threads: [makeThread('a'), makeThread('b')] }], + groupBy: 'none', + collapsedGroupKeys: new Set() + }) + expect(items.map((item) => getActivityVirtualItemKey(item))).toEqual(['t:a', 't:b']) + }) + + it('keeps a collapsed group header but drops its thread rows', () => { + const items = buildActivityVirtualItems({ + groups: [makeGroup('working', ['a', 'b']), makeGroup('done', ['c'])], + groupBy: 'status', + collapsedGroupKeys: new Set(['working']) + }) + expect(items.map((item) => getActivityVirtualItemKey(item))).toEqual([ + 'h:working', + 'h:done', + 't:c' + ]) + }) + + it('locates the selected thread row by paneKey', () => { + const items = buildActivityVirtualItems({ + groups: [makeGroup('working', ['a', 'b', 'c'])], + groupBy: 'status', + collapsedGroupKeys: new Set() + }) + expect(findActivityThreadItemIndex(items, 'c')).toBe(3) + expect(findActivityThreadItemIndex(items, 'missing')).toBeNull() + expect(findActivityThreadItemIndex(items, null)).toBeNull() + }) +}) diff --git a/src/renderer/src/components/activity/activity-thread-virtual-items.ts b/src/renderer/src/components/activity/activity-thread-virtual-items.ts new file mode 100644 index 00000000000..63ccf56de38 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-virtual-items.ts @@ -0,0 +1,74 @@ +import type { ActivityGroupBy, ActivityThreadGroup, AgentPaneThread } from './activity-thread-types' + +/** One row of the virtualized Activity list: a group header or a thread. */ +export type ActivityVirtualItemDescriptor = + | { type: 'header'; group: ActivityThreadGroup } + | { type: 'thread'; thread: AgentPaneThread; groupKey: string } + +export const ACTIVITY_HEADER_ROW_ESTIMATE = 32 +export const ACTIVITY_THREAD_ROW_COMPACT_ESTIMATE = 96 +export const ACTIVITY_THREAD_ROW_FULL_ESTIMATE = 116 + +/** + * Flatten grouped threads into a single virtualizable row list, honoring + * collapsed groups (their thread rows are omitted entirely). + */ +export function buildActivityVirtualItems(args: { + groups: readonly ActivityThreadGroup[] + groupBy: ActivityGroupBy + collapsedGroupKeys: ReadonlySet +}): ActivityVirtualItemDescriptor[] { + const items: ActivityVirtualItemDescriptor[] = [] + for (const group of args.groups) { + if (args.groupBy !== 'none') { + items.push({ type: 'header', group }) + if (args.collapsedGroupKeys.has(group.key)) { + continue + } + } + for (const thread of group.threads) { + items.push({ type: 'thread', thread, groupKey: group.key }) + } + } + return items +} + +/** Stable per-row key: group key for headers, paneKey for threads. */ +export function getActivityVirtualItemKey(item: ActivityVirtualItemDescriptor): string { + return item.type === 'header' ? `h:${item.group.key}` : `t:${item.thread.paneKey}` +} + +export function estimateActivityVirtualItemSize( + item: ActivityVirtualItemDescriptor | undefined, + compactMode: boolean +): number { + if (!item || item.type === 'header') { + return ACTIVITY_HEADER_ROW_ESTIMATE + } + return compactMode ? ACTIVITY_THREAD_ROW_COMPACT_ESTIMATE : ACTIVITY_THREAD_ROW_FULL_ESTIMATE +} + +/** Index of the selected thread's row, or null; kept mounted so activation and focus survive scrolling. */ +export function findActivityThreadItemIndex( + items: readonly ActivityVirtualItemDescriptor[], + paneKey: string | null +): number | null { + if (paneKey === null) { + return null + } + const index = items.findIndex((item) => item.type === 'thread' && item.thread.paneKey === paneKey) + return index === -1 ? null : index +} + +/** Indexes of header rows, for sticky-header resolution. */ +export function getActivityHeaderItemIndexes( + items: readonly ActivityVirtualItemDescriptor[] +): number[] { + const indexes: number[] = [] + items.forEach((item, index) => { + if (item.type === 'header') { + indexes.push(index) + } + }) + return indexes +} diff --git a/src/renderer/src/components/activity/activity-thread-virtual-row.tsx b/src/renderer/src/components/activity/activity-thread-virtual-row.tsx new file mode 100644 index 00000000000..4b6d7ac7d48 --- /dev/null +++ b/src/renderer/src/components/activity/activity-thread-virtual-row.tsx @@ -0,0 +1,71 @@ +import type React from 'react' +import { translate } from '@/i18n/i18n' +import { ActivityStatusGroupHeader } from './activity-thread-controls' +import { ActivityThreadRow } from './activity-thread-row' +import type { ActivityVirtualItemDescriptor } from './activity-thread-virtual-items' + +export function ActivityThreadVirtualRow({ + item, + collapsed, + onToggleGroup, + selectedPaneKey, + onSelectThread, + onJumpToWorkspace, + onMarkThreadRead, + onMarkThreadUnread, + canJumpToWorkspace, + compactMode, + allowMarkUnreadWhenSelected, + showJumpAction +}: { + item: ActivityVirtualItemDescriptor + collapsed: boolean + onToggleGroup: (groupKey: string) => void + selectedPaneKey: string | null + onSelectThread: Parameters[0]['onSelect'] + onJumpToWorkspace: Parameters[0]['onJump'] + onMarkThreadRead: Parameters[0]['onMarkRead'] + onMarkThreadUnread: Parameters[0]['onMarkUnread'] + canJumpToWorkspace: ( + thread: Extract['thread'] + ) => boolean + compactMode: boolean + allowMarkUnreadWhenSelected: boolean + showJumpAction: boolean +}): React.JSX.Element { + if (item.type === 'header') { + return ( +
    + onToggleGroup(item.group.key)} + /> +
    + ) + } + return ( +
    + +
    + ) +} diff --git a/src/renderer/src/components/activity/dev-activity-fixture.test.ts b/src/renderer/src/components/activity/dev-activity-fixture.test.ts new file mode 100644 index 00000000000..3895c01eeb4 --- /dev/null +++ b/src/renderer/src/components/activity/dev-activity-fixture.test.ts @@ -0,0 +1,39 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { parsePaneKey } from '../../../../shared/stable-pane-id' + +const mocks = vi.hoisted(() => ({ + getState: vi.fn(), + setState: vi.fn(), + setAgentStatuses: vi.fn() +})) + +vi.mock('@/store', () => ({ + useAppStore: { + getState: mocks.getState, + setState: mocks.setState + } +})) + +import { seedDevActivityFixture } from './dev-activity-fixture' + +describe('seedDevActivityFixture', () => { + beforeEach(() => { + vi.clearAllMocks() + mocks.getState.mockReturnValue({ + repos: [], + agentStatusByPaneKey: {}, + setAgentStatuses: mocks.setAgentStatuses + }) + }) + + it('seeds locally-owned rows with valid stable pane keys', () => { + seedDevActivityFixture() + + const updates = mocks.setAgentStatuses.mock.calls[0]?.[0] ?? [] + expect(updates).toHaveLength(3) + for (const update of updates) { + expect(parsePaneKey(update.paneKey)).not.toBeNull() + expect(update.routing.connectionId).toBeNull() + } + }) +}) diff --git a/src/renderer/src/components/activity/dev-activity-fixture.ts b/src/renderer/src/components/activity/dev-activity-fixture.ts new file mode 100644 index 00000000000..9c2f2fc45f9 --- /dev/null +++ b/src/renderer/src/components/activity/dev-activity-fixture.ts @@ -0,0 +1,127 @@ +import { useAppStore } from '@/store' +import type { Repo } from '../../../../shared/repo-types' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' + +const FIXTURE_REPO_ID = 'dev-fixture-repo' +const FIXTURE_WORKTREE_ID = `${FIXTURE_REPO_ID}::/dev/orca-sample` +const FIXTURE_LEAF_IDS = [ + '11111111-1111-4111-8111-111111111111', + '22222222-2222-4222-8222-222222222222', + '33333333-3333-4333-8333-333333333333' +] as const + +function fixtureRepo(): Repo { + return { + id: FIXTURE_REPO_ID, + path: '/dev/orca-sample', + displayName: 'Orca Sample App', + badgeColor: '#8b5cf6', + addedAt: Date.now(), + kind: 'git', + executionHostId: 'local' + } +} + +function fixtureWorktree(): Worktree { + return { + id: FIXTURE_WORKTREE_ID, + repoId: FIXTURE_REPO_ID, + path: '/dev/orca-sample', + head: 'dev-fixture-head', + branch: 'feature/activity-dashboard', + isBare: false, + isMainWorktree: false, + displayName: 'Activity dashboard', + comment: 'Development fixture workspace', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: true, + isPinned: false, + sortOrder: 0, + lastActivityAt: Date.now() + } +} + +function fixtureTab(id: string, title: string, sortOrder: number): TerminalTab { + return { + id, + ptyId: null, + worktreeId: FIXTURE_WORKTREE_ID, + title, + customTitle: null, + color: null, + sortOrder, + createdAt: Date.now(), + launchAgent: id === 'dev-fixture-tab-1' ? 'codex' : 'claude' + } +} + +/** Populate a fresh development profile with representative activity rows. */ +export function seedDevActivityFixture(): void { + const state = useAppStore.getState() + if (state.repos.length > 0 || Object.keys(state.agentStatusByPaneKey).length > 0) { + return + } + + const now = Date.now() + const repo = fixtureRepo() + const worktree = fixtureWorktree() + const tabs = [ + fixtureTab('dev-fixture-tab-1', 'Refactor activity filters', 0), + fixtureTab('dev-fixture-tab-2', 'Review empty-state copy', 1), + fixtureTab('dev-fixture-tab-3', 'Add keyboard shortcut', 2) + ] + + useAppStore.setState({ + repos: [repo], + activeRepoId: repo.id, + worktreesByRepo: { [repo.id]: [worktree] }, + activeWorktreeId: worktree.id, + activeWorkspaceKey: `worktree:${worktree.id}`, + tabsByWorktree: { [worktree.id]: tabs } + }) + + state.setAgentStatuses([ + { + paneKey: makePaneKey(tabs[0].id, FIXTURE_LEAF_IDS[0]), + payload: { + state: 'working', + prompt: 'Refactor the activity filters and keep the list responsive.', + agentType: 'codex', + model: 'gpt-5-codex', + toolName: 'Edit', + toolInput: 'activity-scope-filter.ts' + }, + timing: { updatedAt: now - 12_000, stateStartedAt: now - 90_000 }, + routing: { tabId: tabs[0].id, worktreeId: worktree.id, connectionId: null } + }, + { + paneKey: makePaneKey(tabs[1].id, FIXTURE_LEAF_IDS[1]), + payload: { + state: 'waiting', + prompt: 'Review the new empty-state copy before merging.', + agentType: 'claude', + model: 'claude-sonnet-4', + lastAssistantMessage: 'The copy is ready for your review.' + }, + timing: { updatedAt: now - 45_000, stateStartedAt: now - 120_000 }, + routing: { tabId: tabs[1].id, worktreeId: worktree.id, connectionId: null } + }, + { + paneKey: makePaneKey(tabs[2].id, FIXTURE_LEAF_IDS[2]), + payload: { + state: 'done', + prompt: 'Add a shortcut to focus the activity search field.', + agentType: 'codex', + model: 'gpt-5-codex', + lastAssistantMessage: 'Added the shortcut and covered it with a test.' + }, + timing: { updatedAt: now - 5 * 60_000, stateStartedAt: now - 8 * 60_000 }, + routing: { tabId: tabs[2].id, worktreeId: worktree.id, connectionId: null } + } + ]) +} diff --git a/src/renderer/src/components/activity/event-time-clock-refresh.test.tsx b/src/renderer/src/components/activity/event-time-clock-refresh.test.tsx new file mode 100644 index 00000000000..d0350fef867 --- /dev/null +++ b/src/renderer/src/components/activity/event-time-clock-refresh.test.tsx @@ -0,0 +1,54 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { TooltipProvider } from '@/components/ui/tooltip' + +const clock = vi.hoisted(() => ({ now: 1_000_000 })) + +vi.mock('@/hooks/use-now', () => ({ + useNow: () => clock.now +})) + +import { EventTime } from './activity-thread-controls' + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +describe('EventTime shared-clock refresh', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + }) + + afterEach(() => { + act(() => root.unmount()) + container.remove() + document.body.replaceChildren() + }) + + it('derives the label from the shared clock, not a frozen render-time Date.now()', () => { + // Why: memo'd rows no longer re-render on unrelated store writes, so the label + // must follow the injected clock or "2m" would freeze at whatever render saw. + const timestamp = clock.now - 2 * 60_000 + const render = (): void => { + act(() => { + root.render( + + + + ) + }) + } + render() + expect(container.textContent).toContain('2m') + + clock.now += 8 * 60_000 + render() + expect(container.textContent).toContain('10m') + }) +}) diff --git a/src/renderer/src/components/activity/use-activity-thread-action-bindings.test.tsx b/src/renderer/src/components/activity/use-activity-thread-action-bindings.test.tsx new file mode 100644 index 00000000000..ea21dc82cbf --- /dev/null +++ b/src/renderer/src/components/activity/use-activity-thread-action-bindings.test.tsx @@ -0,0 +1,113 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentPaneThread } from './activity-thread-types' + +const mocks = vi.hoisted(() => ({ + clearCompletedActivity: vi.fn() +})) + +vi.mock('./activity-clear-completed', () => ({ + clearCompletedActivity: mocks.clearCompletedActivity, + // Done threads carry no live state; the null marker stands in for the real group-id predicate. + isClearableActivityThread: (thread: AgentPaneThread) => thread.currentAgentState === null +})) + +vi.mock('@/store', () => ({ useAppStore: { getState: () => ({}) } })) + +import { useActivityThreadActionBindings } from './use-activity-thread-action-bindings' + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +function makeThread(paneKey: string, overrides: Partial = {}): AgentPaneThread { + return { paneKey, unread: false, currentAgentState: 'working', ...overrides } as AgentPaneThread +} + +type HookResult = ReturnType + +let container: HTMLDivElement +let root: Root +let latest: HookResult | null + +function Probe(props: Parameters[0]): null { + latest = useActivityThreadActionBindings(props) + return null +} + +function renderProbe(props: Parameters[0]): HookResult { + act(() => { + root.render() + }) + if (!latest) { + throw new Error('hook did not render') + } + return latest +} + +beforeEach(() => { + mocks.clearCompletedActivity.mockClear() + latest = null + container = document.createElement('div') + document.body.append(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() +}) + +describe('useActivityThreadActionBindings', () => { + const acknowledgeAgents = vi.fn() + const baseProps = { + acknowledgeAgents, + unacknowledgeAgents: vi.fn(), + setSelectedPaneKey: vi.fn() + } + + beforeEach(() => { + acknowledgeAgents.mockClear() + }) + + it('enables and applies Mark all read from the badge set, not the narrowed visible set', () => { + const hiddenUnread = makeThread('tab-1:hidden', { unread: true }) + const bindings = renderProbe({ + ...baseProps, + // Search/scope narrowing hid the only unread thread from the list… + visibleThreads: [makeThread('tab-2:read')], + // …but it is still in the badge-coherent set, so it must stay clearable. + markAllReadThreads: [makeThread('tab-2:read'), hiddenUnread] + }) + + expect(bindings.hasUnreadThreads).toBe(true) + bindings.markAllThreadsRead() + expect(acknowledgeAgents).toHaveBeenCalledWith([hiddenUnread.paneKey]) + }) + + it('clears completed strictly from the visible set', () => { + const visibleDone = makeThread('tab-1:done', { currentAgentState: null }) + const hiddenDone = makeThread('tab-2:hidden-done', { currentAgentState: null }) + const bindings = renderProbe({ + ...baseProps, + visibleThreads: [visibleDone], + markAllReadThreads: [visibleDone, hiddenDone] + }) + + expect(bindings.hasCompletedThreads).toBe(true) + bindings.handleClearCompleted() + expect(mocks.clearCompletedActivity).toHaveBeenCalledWith([visibleDone]) + }) + + it('disables clear-completed when completions are only outside the visible set', () => { + const hiddenDone = makeThread('tab-2:hidden-done', { currentAgentState: null }) + const bindings = renderProbe({ + ...baseProps, + visibleThreads: [makeThread('tab-1:working')], + markAllReadThreads: [makeThread('tab-1:working'), hiddenDone] + }) + + expect(bindings.hasCompletedThreads).toBe(false) + }) +}) diff --git a/src/renderer/src/components/activity/use-activity-thread-action-bindings.ts b/src/renderer/src/components/activity/use-activity-thread-action-bindings.ts new file mode 100644 index 00000000000..0004db75737 --- /dev/null +++ b/src/renderer/src/components/activity/use-activity-thread-action-bindings.ts @@ -0,0 +1,71 @@ +import { useCallback, useEffect, useMemo, useRef } from 'react' +import { clearCompletedActivity, isClearableActivityThread } from './activity-clear-completed' +import { createActivityThreadActions } from './activity-thread-actions' +import type { AgentPaneThread } from './activity-thread-types' + +type ActivityThreadActionBindings = { + markThreadRead: (thread: AgentPaneThread) => void + markThreadUnread: (thread: AgentPaneThread) => void + selectThread: (thread: AgentPaneThread) => void + jumpToWorkspace: (thread: AgentPaneThread) => void + markAllThreadsRead: () => void + hasUnreadThreads: boolean + hasCompletedThreads: boolean + handleClearCompleted: () => void +} + +/** + * Bulk-action wiring shared by the sidebar Agents list and the full Activity + * page, so their semantics can never drift apart: + * - Mark all read acts on the badge-coherent set (`markAllReadThreads`), so the + * Agents-tab badge always reaches zero even when search/scope hides rows. + * - Clear completed is destructive and acts only on `visibleThreads` — never on + * rows the user cannot currently see. + */ +export function useActivityThreadActionBindings({ + visibleThreads, + markAllReadThreads, + acknowledgeAgents, + unacknowledgeAgents, + setSelectedPaneKey +}: { + visibleThreads: AgentPaneThread[] + markAllReadThreads: AgentPaneThread[] + acknowledgeAgents: (paneKeys: string[]) => void + unacknowledgeAgents: (paneKeys: string[]) => void + setSelectedPaneKey: (paneKey: string | null) => void +}): ActivityThreadActionBindings { + // Why refs: rows are React.memo'd on these handlers; recreating them whenever a + // thread array identity changes (every status ping) would re-render every mounted row. + const visibleThreadsRef = useRef(visibleThreads) + const markAllReadThreadsRef = useRef(markAllReadThreads) + useEffect(() => { + visibleThreadsRef.current = visibleThreads + markAllReadThreadsRef.current = markAllReadThreads + }, [visibleThreads, markAllReadThreads]) + + const actions = useMemo( + () => + createActivityThreadActions({ + getMarkAllReadThreads: () => markAllReadThreadsRef.current, + acknowledgeAgents, + unacknowledgeAgents, + setSelectedPaneKey + }), + [acknowledgeAgents, unacknowledgeAgents, setSelectedPaneKey] + ) + + const hasUnreadThreads = useMemo( + () => markAllReadThreads.some((t) => t.unread), + [markAllReadThreads] + ) + const hasCompletedThreads = useMemo( + () => visibleThreads.some(isClearableActivityThread), + [visibleThreads] + ) + const handleClearCompleted = useCallback(() => { + clearCompletedActivity(visibleThreadsRef.current) + }, []) + + return { ...actions, hasUnreadThreads, hasCompletedThreads, handleClearCompleted } +} diff --git a/src/renderer/src/components/activity/use-agent-pane-threads.ts b/src/renderer/src/components/activity/use-agent-pane-threads.ts new file mode 100644 index 00000000000..01eb0475a02 --- /dev/null +++ b/src/renderer/src/components/activity/use-agent-pane-threads.ts @@ -0,0 +1,264 @@ +import { useDeferredValue, useMemo, useRef } from 'react' +import { useShallow } from 'zustand/react/shallow' +import { useAppStore } from '@/store' +import { getRepoMapFromState, getWorktreeMapFromState } from '@/store/selectors' +import type { AppState } from '@/store/types' +import { + getSettingsFocusedExecutionHostId, + type ExecutionHostId +} from '../../../../shared/execution-host' +import { buildActivityEvents, createActivityEventBuildCache } from './activity-event-builder' +import { projectActivityTabs, type ActivityTabProjection } from './activity-tab-projection' +import { buildAgentPaneThreads, createAgentPaneThreadReuseCache } from './activity-thread-builder' +import { collectChildAgentPaneKeys } from './activity-thread-child-agent' + +const EMPTY_PANE_KEYS: ReadonlySet = new Set() +import { filterThreadsByActivityScope, resolveActivityScopeRepoIds } from './activity-scope-filter' +import { + activityThreadMatchesSearchQuery, + buildActivityThreadGroups, + isActivitySearchQueryTooLarge +} from './activity-thread-grouping' +import type { + ActivityGroupBy, + ActivityThreadGroup, + AgentPaneThread, + ThreadReadFilter +} from './activity-thread-types' + +export type AgentPaneThreadsStoreData = Pick< + AppState, + | 'agentStatusByPaneKey' + | 'runtimeAgentOrchestrationByPaneKey' + | 'migrationUnsupportedByPtyId' + | 'retainedAgentsByPaneKey' + | 'tabsByWorktree' + | 'repos' + | 'worktreesByRepo' + | 'folderWorkspaces' + | 'detectedWorktreesByRepo' + | 'getKnownWorktreeById' + | 'acknowledgedAgentsByPaneKey' + | 'activityClearedAtByPaneKey' + | 'acknowledgeAgents' + | 'unacknowledgeAgents' +> & { + /** Terminal and agent-session tabs only, identity-stable across focus writes. */ + activityTabs: ActivityTabProjection + worktreeMap: ReturnType + repoMap: ReturnType + generatedTitlesEnabled: boolean + /** Focused-host fallback for hostless worktrees, shared by the scope filter and the row actions. */ + defaultHostId: ExecutionHostId +} + +/** The Activity thread pipeline (store read -> events -> threads -> filter -> + * groups), shared by the Activity page and the sidebar agents list. */ +export function useAgentPaneThreads(args: { + query: string + readFilter: ThreadReadFilter + groupBy: ActivityGroupBy + selectedPaneKey: string | null + showChildAgents?: boolean +}): { + storeData: AgentPaneThreadsStoreData + allThreads: AgentPaneThread[] + selectedPaneKeyIsLive: boolean + effectiveSelectedPaneKey: string | null + /** Threads shown after every active filter; Clear completed operates on exactly + * this set so it never destroys rows the user cannot see. */ + visibleThreads: AgentPaneThread[] + /** Threads after the child-agent classification only — the set the Agents-tab + * badge counts and Mark all read clears; transient search/scope/read narrowing + * is ignored so the badge is always clearable. */ + markAllReadThreads: AgentPaneThread[] + visibleThreadGroups: ActivityThreadGroup[] +} { + const { query, readFilter, groupBy, selectedPaneKey, showChildAgents = false } = args + const agentsVisibleHostIds = useAppStore((s) => s.agentsVisibleHostIds) + const agentsFilterRepoIds = useAppStore((s) => s.agentsFilterRepoIds) + // Why project: the unified tab map is rewritten on every tab focus; the projection keeps + // its identity (and each tab's) unless a field this pipeline reads actually changed. + const tabProjectionRef = useRef<{ + raw: AppState['unifiedTabsByWorktree'] | null + projected: ActivityTabProjection | null + }>({ raw: null, projected: null }) + const storeData = useAppStore( + useShallow((s) => ({ + agentStatusByPaneKey: s.agentStatusByPaneKey, + runtimeAgentOrchestrationByPaneKey: s.runtimeAgentOrchestrationByPaneKey, + migrationUnsupportedByPtyId: s.migrationUnsupportedByPtyId, + retainedAgentsByPaneKey: s.retainedAgentsByPaneKey, + tabsByWorktree: s.tabsByWorktree, + activityTabs: (() => { + const cache = tabProjectionRef.current + if (cache.raw === s.unifiedTabsByWorktree && cache.projected) { + return cache.projected + } + const projected = projectActivityTabs(s.unifiedTabsByWorktree, cache.projected) + tabProjectionRef.current = { raw: s.unifiedTabsByWorktree, projected } + return projected + })(), + repos: s.repos, + worktreesByRepo: s.worktreesByRepo, + folderWorkspaces: s.folderWorkspaces, + detectedWorktreesByRepo: s.detectedWorktreesByRepo, + getKnownWorktreeById: s.getKnownWorktreeById, + worktreeMap: getWorktreeMapFromState(s), + repoMap: getRepoMapFromState(s), + acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey, + activityClearedAtByPaneKey: s.activityClearedAtByPaneKey, + acknowledgeAgents: s.acknowledgeAgents, + unacknowledgeAgents: s.unacknowledgeAgents, + generatedTitlesEnabled: s.settings?.tabAutoGenerateTitle === true, + defaultHostId: getSettingsFocusedExecutionHostId(s.settings) + })) + ) + // Why: agentStatusEpoch is a dep (not used in the body) so the memo recomputes when freshness boundaries expire even without new PTY data. + const agentStatusEpoch = useAppStore((s) => s.agentStatusEpoch) + + // Why per-hook caches: unchanged panes keep their exact event/snapshot/thread object + // identities across rebuilds, so a status write to one agent leaves every other row's + // memo bail-out and cached search text intact. Rebuilds are deterministic, so a repeated + // (StrictMode/deferred) memo invocation returns identical objects from the cache. + const eventBuildCacheRef = useRef(createActivityEventBuildCache()) + const threadReuseCacheRef = useRef(createAgentPaneThreadReuseCache()) + + const { events: allEvents, liveAgentByPaneKey } = useMemo( + () => + buildActivityEvents( + { + agentStatusByPaneKey: storeData.agentStatusByPaneKey, + runtimeAgentOrchestrationByPaneKey: storeData.runtimeAgentOrchestrationByPaneKey, + migrationUnsupportedByPtyId: storeData.migrationUnsupportedByPtyId, + retainedAgentsByPaneKey: storeData.retainedAgentsByPaneKey, + tabsByWorktree: storeData.tabsByWorktree, + unifiedTabsByWorktree: storeData.activityTabs, + worktreeMap: storeData.worktreeMap, + repoMap: storeData.repoMap, + repos: storeData.repos, + resolveWorktree: storeData.getKnownWorktreeById, + acknowledgedAgentsByPaneKey: storeData.acknowledgedAgentsByPaneKey, + activityClearedAtByPaneKey: storeData.activityClearedAtByPaneKey, + // Why: Date.now() is read in the memo body (not a dep) so stale-decay recomputes when agentStatusEpoch ticks, not on wall-clock time. + now: Date.now() + }, + eventBuildCacheRef.current + ), + // eslint-disable-next-line react-hooks/exhaustive-deps + [storeData, agentStatusEpoch] + ) + + const allThreads = useMemo( + () => + buildAgentPaneThreads( + { + events: allEvents, + liveAgentByPaneKey, + generatedTitlesEnabled: storeData.generatedTitlesEnabled + }, + threadReuseCacheRef.current + ), + [allEvents, liveAgentByPaneKey, storeData.generatedTitlesEnabled] + ) + + const selectedPaneKeyIsLive = + selectedPaneKey === null || allThreads.some((thread) => thread.paneKey === selectedPaneKey) + const effectiveSelectedPaneKey = selectedPaneKeyIsLive ? selectedPaneKey : null + + // Why scope runs before the per-view filters: host/project scope must stay separate + // from unread/search narrowing. + const { threads: scopeVisibleThreads } = useMemo( + () => + filterThreadsByActivityScope({ + threads: allThreads, + scope: { + visibleHostIds: agentsVisibleHostIds, + filterRepoIds: resolveActivityScopeRepoIds(agentsFilterRepoIds, storeData.repoMap), + defaultHostId: storeData.defaultHostId + }, + exemptPaneKey: effectiveSelectedPaneKey + }), + [ + allThreads, + agentsVisibleHostIds, + agentsFilterRepoIds, + storeData.repoMap, + storeData.defaultHostId, + effectiveSelectedPaneKey + ] + ) + + // Why over allThreads (not the scoped list): child classification asks whether the + // parent pane still exists at all, and a scope filter hiding the parent must not + // reclassify its workers as orphans. + // Skipped entirely when children are shown: nothing reads the set then. + const childAgentPaneKeys = useMemo( + () => (showChildAgents ? EMPTY_PANE_KEYS : collectChildAgentPaneKeys(allThreads)), + [allThreads, showChildAgents] + ) + + // Why deferred: filtering hundreds of threads is interruptible background work; the input + // echoes the keystroke at full priority while the list catches up on the deferred value. + const deferredQuery = useDeferredValue(query) + const visibleThreads = useMemo(() => { + const normalizedQuery = isActivitySearchQueryTooLarge(deferredQuery) + ? null + : deferredQuery.trim().toLowerCase() + return scopeVisibleThreads.filter((thread) => { + // Why: keep the just-selected thread visible after auto-mark-read flips it to read, else unread-only mode makes the clicked row vanish from the list. + if ( + readFilter === 'unread' && + !thread.unread && + thread.paneKey !== effectiveSelectedPaneKey + ) { + return false + } + // Why: child agents (e.g. dispatched orchestration workers) are hidden by default to keep top-level agent views focused on root tasks. + if ( + !showChildAgents && + childAgentPaneKeys.has(thread.paneKey) && + thread.paneKey !== effectiveSelectedPaneKey + ) { + return false + } + if (normalizedQuery === null) { + return false + } + return activityThreadMatchesSearchQuery({ thread, searchQuery: normalizedQuery }) + }) + }, [ + scopeVisibleThreads, + readFilter, + deferredQuery, + effectiveSelectedPaneKey, + showChildAgents, + childAgentPaneKeys + ]) + + const markAllReadThreads = useMemo( + () => + allThreads.filter( + (thread) => + showChildAgents || + !childAgentPaneKeys.has(thread.paneKey) || + thread.paneKey === effectiveSelectedPaneKey + ), + [allThreads, showChildAgents, childAgentPaneKeys, effectiveSelectedPaneKey] + ) + + const visibleThreadGroups = useMemo( + () => buildActivityThreadGroups(visibleThreads, groupBy), + [visibleThreads, groupBy] + ) + + return { + storeData, + allThreads, + selectedPaneKeyIsLive, + effectiveSelectedPaneKey, + visibleThreads, + markAllReadThreads, + visibleThreadGroups + } +} diff --git a/src/renderer/src/components/activity/useActivityUnreadCount.test.ts b/src/renderer/src/components/activity/useActivityUnreadCount.test.ts index 5c28a132762..62c423ce1bd 100644 --- a/src/renderer/src/components/activity/useActivityUnreadCount.test.ts +++ b/src/renderer/src/components/activity/useActivityUnreadCount.test.ts @@ -22,29 +22,26 @@ function makeSource(entry: AgentStatusEntry, ackAt = 0) { acknowledgedAgentsByPaneKey: { [PANE]: ackAt }, agentStatusByPaneKey: { [PANE]: entry }, migrationUnsupportedByPtyId: {}, - retainedAgentsByPaneKey: {}, - worktreesByRepo: {} + retainedAgentsByPaneKey: {} } } describe('countActivityUnread session-boundary rows (STA-3386)', () => { - it('does not count a session-boundary done as unread in either mode', () => { + it('does not count a session-boundary done as unread', () => { const source = makeSource(makeEntry({ sessionBoundary: true })) - expect(countActivityUnread(source, 'sidebar-badge')).toBe(0) - expect(countActivityUnread(source, 'agent-events')).toBe(0) + expect(countActivityUnread(source)).toBe(0) }) it('keeps counting a real completion displaced into history by a session boundary', () => { // Why: agent finished (unacknowledged), then the user resumed the session — the - // boundary row replaces the live done but the finish must stay unread in both badges. + // boundary row replaces the live done but the finish must stay unread. const source = makeSource( makeEntry({ sessionBoundary: true, stateHistory: [{ state: 'done', prompt: 'fix bug', startedAt: 1_000 }] }) ) - expect(countActivityUnread(source, 'sidebar-badge')).toBe(1) - expect(countActivityUnread(source, 'agent-events')).toBe(1) + expect(countActivityUnread(source)).toBe(1) }) it('stops counting the displaced completion once acknowledged', () => { @@ -55,12 +52,60 @@ describe('countActivityUnread session-boundary rows (STA-3386)', () => { }), 1_500 ) - expect(countActivityUnread(source, 'sidebar-badge')).toBe(0) - expect(countActivityUnread(source, 'agent-events')).toBe(0) + expect(countActivityUnread(source)).toBe(0) }) - it('still counts an ordinary unacknowledged done in sidebar-badge mode', () => { + it('still counts an ordinary unacknowledged done', () => { const source = makeSource(makeEntry({})) - expect(countActivityUnread(source, 'sidebar-badge')).toBe(1) + expect(countActivityUnread(source)).toBe(1) + }) +}) + +describe('countActivityUnread with Clear completed cutoffs', () => { + it('does not count events hidden by the pane cutoff', () => { + const source = { + ...makeSource( + makeEntry({ + stateHistory: [{ state: 'done', prompt: 'older run', startedAt: 1_000 }] + }) + ), + activityClearedAtByPaneKey: { [PANE]: 2_000 } + } + // Both the history event (1_000) and the live done (2_000) are at or before the cutoff. + expect(countActivityUnread(source)).toBe(0) + }) + + it('keeps counting turns newer than the cutoff', () => { + const source = { + ...makeSource( + makeEntry({ + stateStartedAt: 3_000, + stateHistory: [{ state: 'done', prompt: 'older run', startedAt: 1_000 }] + }) + ), + activityClearedAtByPaneKey: { [PANE]: 2_000 } + } + expect(countActivityUnread(source)).toBe(1) + }) +}) + +describe('countActivityUnread source overlap', () => { + it('counts an overlapping live and retained pane only once', () => { + const entry = makeEntry({}) + const source = { + acknowledgedAgentsByPaneKey: { [PANE]: 0 }, + agentStatusByPaneKey: { [PANE]: entry }, + retainedAgentsByPaneKey: { + [PANE]: { + entry, + worktreeId: 'wt-1', + tab: {} as never, + agentType: 'claude', + startedAt: 1_000 + } + }, + migrationUnsupportedByPtyId: {} + } + expect(countActivityUnread(source)).toBe(1) }) }) diff --git a/src/renderer/src/components/activity/useActivityUnreadCount.ts b/src/renderer/src/components/activity/useActivityUnreadCount.ts index 91b51c4455e..3a3074d24e9 100644 --- a/src/renderer/src/components/activity/useActivityUnreadCount.ts +++ b/src/renderer/src/components/activity/useActivityUnreadCount.ts @@ -12,85 +12,60 @@ type ActivityUnreadCountSource = Pick< | 'agentStatusByPaneKey' | 'migrationUnsupportedByPtyId' | 'retainedAgentsByPaneKey' - | 'worktreesByRepo' -> - -type ActivityUnreadCountMode = 'agent-events' | 'sidebar-badge' - -const EMPTY_WORKTREES_BY_REPO: AppState['worktreesByRepo'] = {} -const EMPTY_MIGRATION_UNSUPPORTED: AppState['migrationUnsupportedByPtyId'] = {} -const EMPTY_RETAINED_AGENTS: AppState['retainedAgentsByPaneKey'] = {} -const EMPTY_ACKNOWLEDGED_AGENTS: AppState['acknowledgedAgentsByPaneKey'] = {} - -const DISABLED_ACTIVITY_UNREAD_INPUTS = { - sortEpoch: 0, - worktreesByRepo: EMPTY_WORKTREES_BY_REPO, - migrationUnsupportedByPtyId: EMPTY_MIGRATION_UNSUPPORTED, - retainedAgentsByPaneKey: EMPTY_RETAINED_AGENTS, - acknowledgedAgentsByPaneKey: EMPTY_ACKNOWLEDGED_AGENTS +> & { + /** Per-pane "Clear completed" cutoffs; hidden events must not count as unread. */ + activityClearedAtByPaneKey?: Record } function isUnreadAgentState(state: AgentStatusState): boolean { return state === 'done' || state === 'blocked' || state === 'waiting' } -export function countActivityUnread( - source: ActivityUnreadCountSource, - mode: ActivityUnreadCountMode -): number { +/** Counts unread done/blocked/waiting events for the Activity page titlebar badge. */ +export function countActivityUnread(source: ActivityUnreadCountSource): number { let count = 0 + const seenPaneKeys = new Set() - if (mode === 'sidebar-badge') { - for (const worktrees of Object.values(source.worktreesByRepo)) { - for (const worktree of worktrees) { - if (worktree.createdAt && worktree.isUnread) { - count += 1 - } - } - } - } - + // Why no worktree.isUnread here: Activity lists only agent threads, so a worktree + // unread would light a badge with no row to read and no way to clear it. const countEntry = (entry: AgentStatusEntry, ackAt: number): void => { - if (mode === 'agent-events') { - // Why: Activity feed surfaces historical done/blocked/waiting events - // from stateHistory, so the titlebar badge must mirror that event count. - for (const history of entry.stateHistory) { - if (isUnreadAgentState(history.state) && ackAt < history.startedAt) { - count += 1 - } + // Why: "Clear completed" hides events at or before the pane's cutoff from the feed, + // so a hidden event must not keep the badge lit; treat the cutoff like an ack floor. + const clearedAt = source.activityClearedAtByPaneKey?.[entry.paneKey] ?? 0 + const mutedAt = Math.max(ackAt, clearedAt) + // Why: Activity feed surfaces historical done/blocked/waiting events + // from stateHistory, so the titlebar badge must mirror that event count. + for (const history of entry.stateHistory) { + if (isUnreadAgentState(history.state) && mutedAt < history.startedAt) { + count += 1 } } // Why: a session-boundary done is an idle connect (STA-3386), not an event to read. - // History never contains a boundary, but it DOES keep the real completion a boundary - // displaced (the slice pushes it on done→done), so sidebar-badge mode — which skips the - // history loop above — must still count that displaced completion or the badge silently - // drops an unacknowledged finish the moment its session is resumed. if ( isUnreadAgentState(entry.state) && entry.sessionBoundary !== true && - ackAt < entry.stateStartedAt + mutedAt < entry.stateStartedAt ) { count += 1 - } else if (mode === 'sidebar-badge' && entry.state === 'done' && entry.sessionBoundary) { - const displaced = entry.stateHistory.at(-1) - if (displaced && isUnreadAgentState(displaced.state) && ackAt < displaced.startedAt) { - count += 1 - } } } for (const [paneKey, entry] of Object.entries(source.agentStatusByPaneKey)) { + seenPaneKeys.add(paneKey) countEntry(entry, source.acknowledgedAgentsByPaneKey[paneKey] ?? 0) } for (const [paneKey, retained] of Object.entries(source.retainedAgentsByPaneKey)) { - if (mode === 'sidebar-badge' && retained.entry.state !== 'done') { + // Live status is the primary source; retained is a handoff cache and may briefly overlap it. + if (seenPaneKeys.has(paneKey)) { continue } + seenPaneKeys.add(paneKey) countEntry(retained.entry, source.acknowledgedAgentsByPaneKey[paneKey] ?? 0) } for (const unsupported of Object.values(source.migrationUnsupportedByPtyId)) { const entry = migrationUnsupportedToAgentStatusEntry(unsupported) - if (entry) { + if (entry && !seenPaneKeys.has(entry.paneKey)) { + seenPaneKeys.add(entry.paneKey) countEntry(entry, source.acknowledgedAgentsByPaneKey[entry.paneKey] ?? 0) } } @@ -98,53 +73,40 @@ export function countActivityUnread( return count } -export function useActivityUnreadCount(enabled: boolean, mode: ActivityUnreadCountMode): number { +export function useActivityUnreadCount(): number { const { sortEpoch, - worktreesByRepo, migrationUnsupportedByPtyId, retainedAgentsByPaneKey, - acknowledgedAgentsByPaneKey + acknowledgedAgentsByPaneKey, + activityClearedAtByPaneKey } = useAppStore( - useShallow((state) => { - if (!enabled) { - return DISABLED_ACTIVITY_UNREAD_INPUTS - } - return { - // Why: live status prompt/tool updates churn agentStatusByPaneKey but - // cannot change unread count unless a sort-relevant state transition - // or removal occurred. sortEpoch is the cheap invalidation signal. - sortEpoch: state.sortEpoch, - worktreesByRepo: state.worktreesByRepo, - migrationUnsupportedByPtyId: state.migrationUnsupportedByPtyId, - retainedAgentsByPaneKey: state.retainedAgentsByPaneKey, - acknowledgedAgentsByPaneKey: state.acknowledgedAgentsByPaneKey - } - }) + useShallow((state) => ({ + // Why: live status prompt/tool updates churn agentStatusByPaneKey but + // cannot change unread count unless a sort-relevant state transition + // or removal occurred. sortEpoch is the cheap invalidation signal. + sortEpoch: state.sortEpoch, + migrationUnsupportedByPtyId: state.migrationUnsupportedByPtyId, + retainedAgentsByPaneKey: state.retainedAgentsByPaneKey, + acknowledgedAgentsByPaneKey: state.acknowledgedAgentsByPaneKey, + activityClearedAtByPaneKey: state.activityClearedAtByPaneKey + })) ) return useMemo(() => { - if (!enabled) { - return 0 - } void sortEpoch - return countActivityUnread( - { - agentStatusByPaneKey: useAppStore.getState().agentStatusByPaneKey, - migrationUnsupportedByPtyId, - retainedAgentsByPaneKey, - worktreesByRepo, - acknowledgedAgentsByPaneKey - }, - mode - ) + return countActivityUnread({ + agentStatusByPaneKey: useAppStore.getState().agentStatusByPaneKey, + migrationUnsupportedByPtyId, + retainedAgentsByPaneKey, + acknowledgedAgentsByPaneKey, + activityClearedAtByPaneKey + }) }, [ acknowledgedAgentsByPaneKey, - enabled, + activityClearedAtByPaneKey, migrationUnsupportedByPtyId, - mode, retainedAgentsByPaneKey, - sortEpoch, - worktreesByRepo + sortEpoch ]) } diff --git a/src/renderer/src/components/dashboard/agent-row-lineage-model.ts b/src/renderer/src/components/dashboard/agent-row-lineage-model.ts index cb1cf336346..bcac6ddfd4b 100644 --- a/src/renderer/src/components/dashboard/agent-row-lineage-model.ts +++ b/src/renderer/src/components/dashboard/agent-row-lineage-model.ts @@ -1,12 +1,19 @@ -import type { DashboardAgentRow } from './useDashboardData' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' -export type AgentRowLineageTree = { +/** Minimal row shape the lineage rules need; DashboardAgentRow satisfies it, + * and other surfaces (e.g. the Agents thread list) can feed synthetic rows. */ +export type AgentLineageSourceRow = { + paneKey: string + entry: Pick +} + +export type AgentRowLineageTree = { rootRows: T[] childrenByParentPaneKey: Map childPaneKeys: Set } -function buildPaneKeyByTerminalHandle( +function buildPaneKeyByTerminalHandle( rows: readonly T[] ): Map { const paneKeyByTerminalHandle = new Map() @@ -18,7 +25,7 @@ function buildPaneKeyByTerminalHandle( return paneKeyByTerminalHandle } -export function resolveAgentRowParentPaneKey( +export function resolveAgentRowParentPaneKey( row: T, rowsByPaneKey: ReadonlyMap, paneKeyByTerminalHandle: ReadonlyMap @@ -48,7 +55,7 @@ export function resolveAgentRowParentPaneKey( return undefined } -export function buildAgentRowLineageTree( +export function buildAgentRowLineageTree( rows: readonly T[] ): AgentRowLineageTree { const rowsByPaneKey = new Map() diff --git a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.cache.test.ts b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.cache.test.ts new file mode 100644 index 00000000000..31aa10458f0 --- /dev/null +++ b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.cache.test.ts @@ -0,0 +1,190 @@ +import { describe, expect, it } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type { DashboardSnapshotState } from './build-dashboard-snapshot' +import { + buildDashboardBucketCounts, + createDashboardBucketCountsCache +} from './build-dashboard-bucket-counts' + +const NOW = 1_000_000_000 +// Freshness decay boundaries are exercised via generation bumps; the exact stale +// window belongs to the shared agent-status constants. +const AGENT_STALE_STEP = 60_000 +const LEAF_1 = '11111111-1111-4111-8111-111111111111' +const LEAF_2 = '22222222-2222-4222-8222-222222222222' +const PANE_1 = makePaneKey('tab1', LEAF_1) +const PANE_2 = makePaneKey('tab2', LEAF_2) + +function worktree(id: string): Worktree { + return { + id, + repoId: 'r1', + path: `/r1/${id}`, + head: 'abc123', + branch: 'main', + isBare: false, + isMainWorktree: false, + displayName: id, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: NOW + } +} + +function tab(id: string, worktreeId: string): TerminalTab { + return { + id, + ptyId: `pty-${id}`, + worktreeId, + title: 'shell', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: NOW + } +} + +function entry(paneKey: string, tabId: string, worktreeId: string): AgentStatusEntry { + return { + paneKey, + state: 'working', + prompt: 'do the thing', + updatedAt: NOW, + stateStartedAt: NOW - 5_000, + stateHistory: [], + agentType: 'claude', + tabId, + worktreeId + } +} + +function baseState(): DashboardSnapshotState { + return { + repos: [{ id: 'r1', path: '/r1', displayName: 'Repo One', badgeColor: '#000', addedAt: 1 }], + worktreesByRepo: { r1: [worktree('w1'), worktree('w2')] }, + tabsByWorktree: { w1: [tab('tab1', 'w1')], w2: [tab('tab2', 'w2')] }, + agentStatusByPaneKey: { + [PANE_1]: entry(PANE_1, 'tab1', 'w1'), + [PANE_2]: entry(PANE_2, 'tab2', 'w2') + }, + retainedAgentsByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + runtimeAgentOrchestrationByPaneKey: {}, + terminalLayoutsByTabId: { + tab1: { + root: { type: 'leaf', leafId: LEAF_1 }, + activeLeafId: LEAF_1, + expandedLeafId: null, + ptyIdsByLeafId: { [LEAF_1]: 'pty-tab1' } + }, + tab2: { + root: { type: 'leaf', leafId: LEAF_2 }, + activeLeafId: LEAF_2, + expandedLeafId: null, + ptyIdsByLeafId: { [LEAF_2]: 'pty-tab2' } + } + }, + ptyIdsByTabId: { tab1: ['pty-tab1'], tab2: ['pty-tab2'] }, + runtimePaneTitlesByTabId: { tab1: { 0: 'shell' }, tab2: { 0: 'shell' } }, + acknowledgedAgentsByPaneKey: {}, + settings: null + } +} + +describe('buildDashboardBucketCounts per-worktree cache', () => { + it('computes every worktree on the first call and none when nothing changed', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + const first = buildDashboardBucketCounts(state, NOW, cache, 1) + expect(cache.lastComputedWorktreeIds.sort()).toEqual(['w1', 'w2']) + expect(first.working).toBe(2) + + const second = buildDashboardBucketCounts(state, NOW + 1_000, cache, 1) + expect(cache.lastComputedWorktreeIds).toEqual([]) + expect(second).toEqual(first) + }) + + it('recomputes only the worktree affected by an unrelated title write', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + buildDashboardBucketCounts(state, NOW, cache, 1) + + // A pane-title frame for w2's tab: new top-level map identity, only tab2 changed. + const next: DashboardSnapshotState = { + ...state, + runtimePaneTitlesByTabId: { ...state.runtimePaneTitlesByTabId, tab2: { 0: 'sh' } } + } + const counts = buildDashboardBucketCounts(next, NOW + 500, cache, 1) + expect(cache.lastComputedWorktreeIds).toEqual(['w2']) + // Correctness: identical to a cold, uncached run over the same state. + expect(counts).toEqual(buildDashboardBucketCounts(next, NOW + 500)) + }) + + it('recomputes only the worktree affected by a status write', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + buildDashboardBucketCounts(state, NOW, cache, 1) + + const next: DashboardSnapshotState = { + ...state, + agentStatusByPaneKey: { + ...state.agentStatusByPaneKey, + [PANE_1]: { ...entry(PANE_1, 'tab1', 'w1'), prompt: 'new streamed prompt' } + } + } + const counts = buildDashboardBucketCounts(next, NOW + 500, cache, 1) + expect(cache.lastComputedWorktreeIds).toEqual(['w1']) + expect(counts).toEqual(buildDashboardBucketCounts(next, NOW + 500)) + }) + + it('recomputes every worktree when the freshness generation changes', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + buildDashboardBucketCounts(state, NOW, cache, 1) + buildDashboardBucketCounts(state, NOW + AGENT_STALE_STEP, cache, 2) + expect(cache.lastComputedWorktreeIds.sort()).toEqual(['w1', 'w2']) + }) + + it('drops cache rows for worktrees that leave the active set', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + buildDashboardBucketCounts(state, NOW, cache, 1) + expect([...cache.byWorktree.keys()].sort()).toEqual(['w1', 'w2']) + + const next: DashboardSnapshotState = { + ...state, + worktreesByRepo: { r1: [worktree('w1')] } + } + buildDashboardBucketCounts(next, NOW + 500, cache, 1) + expect([...cache.byWorktree.keys()]).toEqual(['w1']) + }) + + it('matches the uncached result for acknowledgement changes', () => { + const cache = createDashboardBucketCountsCache() + const state: DashboardSnapshotState = { + ...baseState(), + agentStatusByPaneKey: { + [PANE_1]: { ...entry(PANE_1, 'tab1', 'w1'), state: 'done' }, + [PANE_2]: entry(PANE_2, 'tab2', 'w2') + } + } + buildDashboardBucketCounts(state, NOW, cache, 1) + const acked: DashboardSnapshotState = { + ...state, + acknowledgedAgentsByPaneKey: { [PANE_1]: NOW } + } + const counts = buildDashboardBucketCounts(acked, NOW + 500, cache, 1) + expect(counts).toEqual(buildDashboardBucketCounts(acked, NOW + 500)) + // Rows don't depend on acks — the recount must not rebuild any row pipeline. + expect(cache.lastComputedWorktreeIds).toEqual([]) + }) +}) diff --git a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts index d077b447a3a..ea21ff33a45 100644 --- a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts +++ b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts @@ -1,22 +1,16 @@ import type { DashboardBucket } from '../../../../shared/dashboard-snapshot' -import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' -import { applyAgentRowLineage } from './agent-row-lineage' import type { DashboardSnapshotState } from './build-dashboard-snapshot' import { collectActiveDashboardWorkspaces } from './dashboard-snapshot-workspaces' import { selectDashboardOrchestration } from './dashboard-orchestration-selection' import { dashboardRowBucketProjection } from './dashboard-row-bucket' -import { buildWorktreeAgentRows } from '../sidebar/worktree-agent-rows' -import { - selectLiveAgentStatusEntriesForWorktree, - selectMigrationUnsupportedEntriesForWorktree, - selectRetainedAgentEntriesForWorktree, - selectTerminalLayoutsForWorktree -} from '../sidebar/worktree-agent-row-selectors' import { EMPTY_WORKTREE_AGENT_ORCHESTRATION } from '../sidebar/worktree-agent-orchestration-batch' import { - selectLivePtyIdsForWorktree, - selectRuntimePaneTitlesForWorktree -} from '../sidebar/worktree-card-status-inputs' + createWorktreeAgentRowsCache, + finishWorktreeAgentRowsCachePass, + selectWorktreeAgentRowsCached, + startWorktreeAgentRowsCachePass, + type WorktreeAgentRowsCache +} from './worktree-agent-rows-cache' const EMPTY_COUNTS: Record = { attention: 0, @@ -25,10 +19,26 @@ const EMPTY_COUNTS: Record = { idle: 0 } -/** Derive sidebar counts without allocating dashboard cards or metadata. */ +export type DashboardBucketCountsCache = WorktreeAgentRowsCache + +export function createDashboardBucketCountsCache(): DashboardBucketCountsCache { + return createWorktreeAgentRowsCache() +} + +/** + * Derive sidebar counts without allocating dashboard cards or metadata. + * + * With a cache, each worktree's row pipeline reruns only when one of its own + * inputs changed (see worktree-agent-rows-cache). Counting over the (possibly + * reused) rows happens on every call, so acknowledgement changes recount + * without rebuilding any rows. `generation` must change whenever time-based + * freshness decay may have shifted a bucket (agentStatusEpoch). + */ export function buildDashboardBucketCounts( state: DashboardSnapshotState, - now: number + now: number, + cache?: DashboardBucketCountsCache, + generation?: unknown ): Record { const counts = { attention: 0, @@ -41,39 +51,23 @@ export function buildDashboardBucketCounts( state, activeWorktrees ) + if (cache) { + startWorktreeAgentRowsCachePass(cache) + } for (const { worktree } of activeWorktrees) { const worktreeId = worktree.id - const liveEntries = selectLiveAgentStatusEntriesForWorktree(state, worktreeId) - const migrationUnsupported = selectMigrationUnsupportedEntriesForWorktree(state, worktreeId) - const entries = - migrationUnsupported.length > 0 - ? [ - ...liveEntries, - ...migrationUnsupported.flatMap((unsupported) => { - const entry = migrationUnsupportedToAgentStatusEntry(unsupported) - return entry ? [entry] : [] - }) - ] - : liveEntries - const terminalLayoutsByTabId = selectTerminalLayoutsForWorktree(state, worktreeId) - const paneTitlesByTabId = selectRuntimePaneTitlesForWorktree(state, worktreeId) - const rows = applyAgentRowLineage( - buildWorktreeAgentRows({ - tabs: state.tabsByWorktree[worktreeId] ?? [], - entries, - retained: selectRetainedAgentEntriesForWorktree(state, worktreeId), - runtimePaneTitlesByTabId: paneTitlesByTabId, - ptyIdsByTabId: selectLivePtyIdsForWorktree(state, worktreeId), - terminalLayoutsByTabId, - runtimeAgentOrchestrationByPaneKey: - singletonOrchestration ?? - orchestrationByWorktree?.get(worktreeId) ?? - EMPTY_WORKTREE_AGENT_ORCHESTRATION, - now - }) - ) - + const rows = selectWorktreeAgentRowsCached({ + state, + worktreeId, + orchestration: + singletonOrchestration ?? + orchestrationByWorktree?.get(worktreeId) ?? + EMPTY_WORKTREE_AGENT_ORCHESTRATION, + now, + generation, + cache + }) for (const row of rows) { if (row.rowSource === 'subagent') { continue @@ -82,6 +76,10 @@ export function buildDashboardBucketCounts( } } + if (cache) { + finishWorktreeAgentRowsCachePass(cache) + } + return counts.attention === 0 && counts.working === 0 && counts.done === 0 && counts.idle === 0 ? EMPTY_COUNTS : counts diff --git a/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts b/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts new file mode 100644 index 00000000000..f99738a5c4a --- /dev/null +++ b/src/renderer/src/components/dashboard/build-dashboard-snapshot.rows-cache.test.ts @@ -0,0 +1,142 @@ +import { describe, expect, it } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import { buildDashboardSnapshot, type DashboardSnapshotState } from './build-dashboard-snapshot' +import { createWorktreeAgentRowsCache } from './worktree-agent-rows-cache' + +const NOW = 1_000_000_000 +const LEAF_1 = '11111111-1111-4111-8111-111111111111' +const LEAF_2 = '22222222-2222-4222-8222-222222222222' +const PANE_1 = makePaneKey('tab1', LEAF_1) +const PANE_2 = makePaneKey('tab2', LEAF_2) + +function worktree(id: string): Worktree { + return { + id, + repoId: 'r1', + path: `/r1/${id}`, + head: 'abc123', + branch: 'main', + isBare: false, + isMainWorktree: false, + displayName: id, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: NOW + } +} + +function tab(id: string, worktreeId: string): TerminalTab { + return { + id, + ptyId: `pty-${id}`, + worktreeId, + title: 'shell', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: NOW + } +} + +function entry(paneKey: string, tabId: string, worktreeId: string): AgentStatusEntry { + return { + paneKey, + state: 'done', + prompt: 'finish the task', + updatedAt: NOW, + stateStartedAt: NOW - 5_000, + stateHistory: [], + agentType: 'claude', + tabId, + worktreeId + } +} + +function baseState(): DashboardSnapshotState { + return { + repos: [{ id: 'r1', path: '/r1', displayName: 'Repo One', badgeColor: '#000', addedAt: 1 }], + worktreesByRepo: { r1: [worktree('w1'), worktree('w2')] }, + tabsByWorktree: { w1: [tab('tab1', 'w1')], w2: [tab('tab2', 'w2')] }, + agentStatusByPaneKey: { + [PANE_1]: entry(PANE_1, 'tab1', 'w1'), + [PANE_2]: entry(PANE_2, 'tab2', 'w2') + }, + retainedAgentsByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + runtimeAgentOrchestrationByPaneKey: {}, + terminalLayoutsByTabId: { + tab1: { + root: { type: 'leaf', leafId: LEAF_1 }, + activeLeafId: LEAF_1, + expandedLeafId: null, + ptyIdsByLeafId: { [LEAF_1]: 'pty-tab1' } + }, + tab2: { + root: { type: 'leaf', leafId: LEAF_2 }, + activeLeafId: LEAF_2, + expandedLeafId: null, + ptyIdsByLeafId: { [LEAF_2]: 'pty-tab2' } + } + }, + ptyIdsByTabId: { tab1: ['pty-tab1'], tab2: ['pty-tab2'] }, + runtimePaneTitlesByTabId: { tab1: { 0: 'shell' }, tab2: { 0: 'shell' } }, + acknowledgedAgentsByPaneKey: {}, + settings: null + } +} + +describe('buildDashboardSnapshot rows cache', () => { + it('matches the uncached snapshot exactly across unrelated and targeted writes', () => { + const cache = createWorktreeAgentRowsCache() + const first = baseState() + expect(buildDashboardSnapshot(first, NOW, { rowsCache: cache, rowsGeneration: 1 })).toEqual( + buildDashboardSnapshot(first, NOW) + ) + + const titleWrite: DashboardSnapshotState = { + ...first, + runtimePaneTitlesByTabId: { ...first.runtimePaneTitlesByTabId, tab2: { 0: 'sh' } } + } + const cached = buildDashboardSnapshot(titleWrite, NOW + 500, { + rowsCache: cache, + rowsGeneration: 1 + }) + expect(cache.lastComputedWorktreeIds).toEqual(['w2']) + expect(cached).toEqual(buildDashboardSnapshot(titleWrite, NOW + 500)) + }) + + it('keeps card-level fields fresh (acks, workspace statuses) without recomputing rows', () => { + const cache = createWorktreeAgentRowsCache() + const state = baseState() + const before = buildDashboardSnapshot(state, NOW, { rowsCache: cache, rowsGeneration: 1 }) + expect(before.cards.find((card) => card.paneKey === PANE_1)?.unseen).toBe(true) + + const acked: DashboardSnapshotState = { + ...state, + acknowledgedAgentsByPaneKey: { [PANE_1]: NOW } + } + const after = buildDashboardSnapshot(acked, NOW + 500, { rowsCache: cache, rowsGeneration: 1 }) + // Why asserted: acks and statuses are card-assembly inputs the rows cache deliberately + // does not key — they must still flow into every rebuild. + expect(cache.lastComputedWorktreeIds).toEqual([]) + expect(after.cards.find((card) => card.paneKey === PANE_1)?.unseen).toBe(false) + expect(after).toEqual(buildDashboardSnapshot(acked, NOW + 500)) + }) + + it('recomputes all worktrees when the freshness generation ticks', () => { + const cache = createWorktreeAgentRowsCache() + const state = baseState() + buildDashboardSnapshot(state, NOW, { rowsCache: cache, rowsGeneration: 1 }) + buildDashboardSnapshot(state, NOW + 60_000, { rowsCache: cache, rowsGeneration: 2 }) + expect(cache.lastComputedWorktreeIds.sort()).toEqual(['w1', 'w2']) + }) +}) diff --git a/src/renderer/src/components/dashboard/build-dashboard-snapshot.ts b/src/renderer/src/components/dashboard/build-dashboard-snapshot.ts index 4b336a56a88..fbdf01fe551 100644 --- a/src/renderer/src/components/dashboard/build-dashboard-snapshot.ts +++ b/src/renderer/src/components/dashboard/build-dashboard-snapshot.ts @@ -12,21 +12,17 @@ import { type DashboardCardTerminalInputState } from './dashboard-card-terminal-input' import { readDashboardClientHost } from './dashboard-client-host' -import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' -import { applyAgentRowLineage, dashboardCardParentPaneKey } from './agent-row-lineage' +import { dashboardCardParentPaneKey } from './agent-row-lineage' import { lastEnteredDoneAt } from './agent-finished-timestamp' -import { buildWorktreeAgentRows } from '../sidebar/worktree-agent-rows' -import { - selectLiveAgentStatusEntriesForWorktree, - selectMigrationUnsupportedEntriesForWorktree, - selectRetainedAgentEntriesForWorktree, - selectTerminalLayoutsForWorktree -} from '../sidebar/worktree-agent-row-selectors' +import { selectTerminalLayoutsForWorktree } from '../sidebar/worktree-agent-row-selectors' import { EMPTY_WORKTREE_AGENT_ORCHESTRATION } from '../sidebar/worktree-agent-orchestration-batch' import { - selectLivePtyIdsForWorktree, - selectRuntimePaneTitlesForWorktree -} from '../sidebar/worktree-card-status-inputs' + finishWorktreeAgentRowsCachePass, + selectWorktreeAgentRowsCached, + startWorktreeAgentRowsCachePass, + type WorktreeAgentRowsCache +} from './worktree-agent-rows-cache' +import { selectRuntimePaneTitlesForWorktree } from '../sidebar/worktree-card-status-inputs' import { resolveDashboardCardContext, type DashboardCardContextState @@ -85,7 +81,15 @@ export type DashboardSnapshotState = Pick< export function buildDashboardSnapshot( state: DashboardSnapshotState, now: number, - options: { includeCardDetails?: boolean; includeFilterOptions?: boolean } = {} + options: { + includeCardDetails?: boolean + includeFilterOptions?: boolean + /** Optional per-worktree row-pipeline reuse; card assembly always runs fresh + * because cards also read review/host/status slices the cache does not key. */ + rowsCache?: WorktreeAgentRowsCache + /** Freshness token for rowsCache (agentStatusEpoch); required for the cache to be safe. */ + rowsGeneration?: unknown + } = {} ): DashboardSnapshot { const cards: DashboardCard[] = [] const workspaces: DashboardWorkspace[] | undefined = @@ -104,41 +108,28 @@ export function buildDashboardSnapshot( state, activeWorktrees ) + if (options.rowsCache) { + startWorktreeAgentRowsCachePass(options.rowsCache) + } for (const workspace of activeWorktrees) { const { repo, worktree } = workspace const worktreeId = worktree.id const parentWorktreeId = worktree.parentWorktreeId - const liveEntries = selectLiveAgentStatusEntriesForWorktree(state, worktreeId) - const migrationUnsupported = selectMigrationUnsupportedEntriesForWorktree(state, worktreeId) - const entries = - migrationUnsupported.length > 0 - ? [ - ...liveEntries, - ...migrationUnsupported.flatMap((unsupported) => { - const entry = migrationUnsupportedToAgentStatusEntry(unsupported) - return entry ? [entry] : [] - }) - ] - : liveEntries const terminalLayoutsByTabId = selectTerminalLayoutsForWorktree(state, worktreeId) const paneTitlesByTabId = selectRuntimePaneTitlesForWorktree(state, worktreeId) - const rows = applyAgentRowLineage( - buildWorktreeAgentRows({ - tabs: state.tabsByWorktree[worktreeId] ?? [], - entries, - retained: selectRetainedAgentEntriesForWorktree(state, worktreeId), - runtimePaneTitlesByTabId: paneTitlesByTabId, - ptyIdsByTabId: selectLivePtyIdsForWorktree(state, worktreeId), - terminalLayoutsByTabId, - runtimeAgentOrchestrationByPaneKey: - singletonOrchestration ?? - orchestrationByWorktree?.get(worktreeId) ?? - EMPTY_WORKTREE_AGENT_ORCHESTRATION, - now - }) - ) + const rows = selectWorktreeAgentRowsCached({ + state, + worktreeId, + orchestration: + singletonOrchestration ?? + orchestrationByWorktree?.get(worktreeId) ?? + EMPTY_WORKTREE_AGENT_ORCHESTRATION, + now, + generation: options.rowsGeneration, + cache: options.rowsCache + }) const subagentsByParentPaneKey = includeCardDetails ? groupSubagentsByParentPaneKey(rows) : undefined @@ -270,6 +261,10 @@ export function buildDashboardSnapshot( } } + if (options.rowsCache) { + finishWorktreeAgentRowsCachePass(options.rowsCache) + } + return { generatedAt: now, cards, diff --git a/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx b/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx index f45dcd3b6e7..541fbd716c1 100644 --- a/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx +++ b/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx @@ -29,7 +29,12 @@ vi.mock('@/store', () => ({ })) vi.mock('./build-dashboard-bucket-counts', () => ({ - buildDashboardBucketCounts: mocks.buildDashboardBucketCounts + buildDashboardBucketCounts: mocks.buildDashboardBucketCounts, + createDashboardBucketCountsCache: () => ({ + byWorktree: new Map(), + computeCount: 0, + lastComputedWorktreeIds: [] + }) })) import { useAgentBucketCounts } from './useAgentBucketCounts' @@ -60,7 +65,9 @@ describe('useAgentBucketCounts', () => { folderWorkspaces: mocks.state.folderWorkspaces, unifiedTabsByWorktree: mocks.state.unifiedTabsByWorktree }), - expect.any(Number) + expect.any(Number), + expect.objectContaining({ byWorktree: expect.any(Map) }), + expect.anything() ) }) diff --git a/src/renderer/src/components/dashboard/useAgentBucketCounts.ts b/src/renderer/src/components/dashboard/useAgentBucketCounts.ts index 58b7a444e37..60df4995e56 100644 --- a/src/renderer/src/components/dashboard/useAgentBucketCounts.ts +++ b/src/renderer/src/components/dashboard/useAgentBucketCounts.ts @@ -1,8 +1,11 @@ -import { useMemo } from 'react' +import { useMemo, useRef } from 'react' import { useAppStore } from '@/store' import { useShallow } from 'zustand/react/shallow' import type { DashboardBucket } from '../../../../shared/dashboard-snapshot' -import { buildDashboardBucketCounts } from './build-dashboard-bucket-counts' +import { + buildDashboardBucketCounts, + createDashboardBucketCountsCache +} from './build-dashboard-bucket-counts' export type AgentBucketCounts = Record @@ -46,6 +49,9 @@ export function useAgentBucketCounts(): AgentBucketCounts { })) ) + // Why a per-hook cache: unrelated status/title writes change one worktree's inputs; + // the cache keeps every other worktree's counts without rerunning its row pipeline. + const cacheRef = useRef(createDashboardBucketCountsCache()) return useMemo(() => { return buildDashboardBucketCounts( { @@ -66,7 +72,11 @@ export function useAgentBucketCounts(): AgentBucketCounts { // generated-title gate is moot and the sidebar stays off settings. settings: null }, - Date.now() + Date.now(), + cacheRef.current, + // Why: time-based freshness decay is signaled by agentStatusEpoch; it invalidates + // every cached worktree so stale-decayed buckets recount. + agentStatusEpoch ) // Why: Date.now() is read inside the memo (not a dep) so idle-decay tracks // agentStatusEpoch ticks, matching useDashboardData. diff --git a/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts b/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts index c2446f4e11f..e6163a70773 100644 --- a/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts +++ b/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts @@ -4,6 +4,7 @@ import { activateTabAndFocusPane } from '@/lib/activate-tab-and-focus-pane' import { runSleepWorktree } from '../sidebar/sleep-worktree-flow' import type { RepoIcon } from '../../../../shared/repo-icon' import { buildDashboardSnapshot, type DashboardSnapshotState } from './build-dashboard-snapshot' +import { createWorktreeAgentRowsCache } from './worktree-agent-rows-cache' import { launchDashboardAgent } from './launch-dashboard-agent' // Why: cap snapshot rebuilds during bursts of agent-status pings. The board is a @@ -164,9 +165,16 @@ export function useDashboardPopoutBridge(enabled: boolean): void { // `withIcons` is forced whenever the pop-out could be starting from nothing — // it opened, or it mounted and asked. Throttled republishes omit an unchanged // icon map and the pop-out keeps the one it already has. + // Why effect-scoped: one cache per popout-bridge lifecycle; unchanged worktrees + // reuse their row pipeline across the up-to-4Hz republish stream. + const rowsCache = createWorktreeAgentRowsCache() const publishNow = (withIcons: boolean): void => { lastPublishAt = Date.now() - const snapshot = buildDashboardSnapshot(useAppStore.getState(), lastPublishAt) + const state = useAppStore.getState() + const snapshot = buildDashboardSnapshot(state, lastPublishAt, { + rowsCache, + rowsGeneration: state.agentStatusEpoch + }) const icons = snapshot.repoIconsByRepoId ?? {} if (!withIcons && repoIconsUnchanged(icons, lastPublishedRepoIcons)) { const { repoIconsByRepoId: _omitted, ...withoutIcons } = snapshot diff --git a/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts b/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts index 07317083d7e..227cf4e3f8b 100644 --- a/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts +++ b/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts @@ -1,7 +1,8 @@ -import { useMemo } from 'react' +import { useMemo, useRef } from 'react' import { useAppStore } from '@/store' import type { DashboardSnapshot } from '../../../../shared/dashboard-snapshot' import { buildDashboardSnapshot } from './build-dashboard-snapshot' +import { createWorktreeAgentRowsCache } from './worktree-agent-rows-cache' /** * Builds the dashboard snapshot directly from the live renderer store for the @@ -10,6 +11,7 @@ import { buildDashboardSnapshot } from './build-dashboard-snapshot' * is no relay, so we derive it here from the same builder the bridge uses. */ export function useLiveDashboardSnapshot(): DashboardSnapshot { + const rowsCacheRef = useRef(createWorktreeAgentRowsCache()) const repos = useAppStore((s) => s.repos) const worktreesByRepo = useAppStore((s) => s.worktreesByRepo) const tabsByWorktree = useAppStore((s) => s.tabsByWorktree) @@ -102,7 +104,10 @@ export function useLiveDashboardSnapshot(): DashboardSnapshot { // rebuild the board. Matches the bridge's republish gate. agentLaunchConfigByPaneKey: useAppStore.getState().agentLaunchConfigByPaneKey }, - Date.now() + Date.now(), + // Why: unchanged worktrees reuse their row pipeline; card assembly still runs + // fresh against the review/host/status slices this memo subscribes to. + { rowsCache: rowsCacheRef.current, rowsGeneration: agentStatusEpoch } ), // eslint-disable-next-line react-hooks/exhaustive-deps [ diff --git a/src/renderer/src/components/dashboard/worktree-agent-rows-cache.ts b/src/renderer/src/components/dashboard/worktree-agent-rows-cache.ts new file mode 100644 index 00000000000..a00cae7af19 --- /dev/null +++ b/src/renderer/src/components/dashboard/worktree-agent-rows-cache.ts @@ -0,0 +1,181 @@ +import type { + AgentStatusEntry, + AgentStatusOrchestrationContext, + MigrationUnsupportedPtyEntry +} from '../../../../shared/agent-status-types' +import type { TerminalLayoutSnapshot, TerminalTab } from '../../../../shared/terminal-tab-types' +import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' +import type { RetainedAgentEntry } from '@/store/slices/agent-status' +import type { AppState } from '@/store/types' +import { applyAgentRowLineage, type DashboardAgentRowWithLineage } from './agent-row-lineage' +import { buildWorktreeAgentRows } from '../sidebar/worktree-agent-rows' +import { + selectLiveAgentStatusEntriesForWorktree, + selectMigrationUnsupportedEntriesForWorktree, + selectRetainedAgentEntriesForWorktree, + selectTerminalLayoutsForWorktree +} from '../sidebar/worktree-agent-row-selectors' +import { + selectLivePtyIdsForWorktree, + selectRuntimePaneTitlesForWorktree +} from '../sidebar/worktree-card-status-inputs' + +type WorktreeAgentRowsCacheEntry = { + /** Caller-provided invalidation token for time-based freshness (agentStatusEpoch). */ + generation: unknown + liveEntries: AgentStatusEntry[] + migrationUnsupported: MigrationUnsupportedPtyEntry[] + retained: RetainedAgentEntry[] + tabs: TerminalTab[] | undefined + orchestration: Record + terminalLayoutsByTabId: Record + paneTitlesByTabId: Record> + ptyIdsByTabId: Record + rows: DashboardAgentRowWithLineage[] +} + +/** + * Per-worktree cache for the lineage-applied agent-row pipeline shared by the + * dashboard snapshot and the sidebar bucket counts. Rows rerun only when one of + * that worktree's own inputs changed — the indexed selectors keep untouched + * worktrees' arrays referentially stable — so an unrelated status/title write + * recomputes one worktree instead of all of them. + * + * Rows depend on wall-clock freshness, so `generation` (agentStatusEpoch) must + * change whenever decay may have shifted a row state. + */ +export type WorktreeAgentRowsCache = { + byWorktree: Map + /** Test instrumentation: cumulative per-worktree row-pipeline (re)computations. */ + computeCount: number + /** Test instrumentation: worktree ids recomputed since the last startPass. */ + lastComputedWorktreeIds: string[] + /** Worktree ids requested since the last startPass; finishPass evicts the rest. */ + seenWorktreeIds: Set +} + +export function createWorktreeAgentRowsCache(): WorktreeAgentRowsCache { + return { + byWorktree: new Map(), + computeCount: 0, + lastComputedWorktreeIds: [], + seenWorktreeIds: new Set() + } +} + +/** Begin one full pass over the active workspaces (resets pass-scoped tracking). */ +export function startWorktreeAgentRowsCachePass(cache: WorktreeAgentRowsCache): void { + cache.lastComputedWorktreeIds = [] + cache.seenWorktreeIds.clear() +} + +/** Drop cache rows for workspaces the pass did not visit so the map stays bounded. */ +export function finishWorktreeAgentRowsCachePass(cache: WorktreeAgentRowsCache): void { + for (const worktreeId of cache.byWorktree.keys()) { + if (!cache.seenWorktreeIds.has(worktreeId)) { + cache.byWorktree.delete(worktreeId) + } + } +} + +export type WorktreeAgentRowsState = Pick< + AppState, + | 'agentStatusByPaneKey' + | 'migrationUnsupportedByPtyId' + | 'retainedAgentsByPaneKey' + | 'tabsByWorktree' + | 'terminalLayoutsByTabId' + | 'ptyIdsByTabId' + | 'runtimePaneTitlesByTabId' +> & + Partial> + +// Why not identity: the per-worktree layout/title/ptyId selectors build a fresh top-level +// record per call while preserving per-tab value references, so shallow equality is the +// correct (and cheap — bounded by tabs per worktree) comparison. +function shallowRecordEqual(a: Record, b: Record): boolean { + if (a === b) { + return true + } + const aKeys = Object.keys(a) + if (aKeys.length !== Object.keys(b).length) { + return false + } + return aKeys.every((key) => Object.is(a[key], b[key])) +} + +/** The lineage-applied rows for one worktree, reused by identity when its inputs are unchanged. */ +export function selectWorktreeAgentRowsCached(args: { + state: WorktreeAgentRowsState + worktreeId: string + orchestration: Record + now: number + generation: unknown + cache?: WorktreeAgentRowsCache +}): DashboardAgentRowWithLineage[] { + const { state, worktreeId, orchestration, now, generation, cache } = args + cache?.seenWorktreeIds.add(worktreeId) + const liveEntries = selectLiveAgentStatusEntriesForWorktree(state, worktreeId) + const migrationUnsupported = selectMigrationUnsupportedEntriesForWorktree(state, worktreeId) + const retained = selectRetainedAgentEntriesForWorktree(state, worktreeId) + const tabs = state.tabsByWorktree[worktreeId] + const terminalLayoutsByTabId = selectTerminalLayoutsForWorktree(state, worktreeId) + const paneTitlesByTabId = selectRuntimePaneTitlesForWorktree(state, worktreeId) + const ptyIdsByTabId = selectLivePtyIdsForWorktree(state, worktreeId) + + const cached = cache?.byWorktree.get(worktreeId) + if ( + cached && + cached.generation === generation && + cached.liveEntries === liveEntries && + cached.migrationUnsupported === migrationUnsupported && + cached.retained === retained && + cached.tabs === tabs && + cached.orchestration === orchestration && + shallowRecordEqual(cached.terminalLayoutsByTabId, terminalLayoutsByTabId) && + shallowRecordEqual(cached.paneTitlesByTabId, paneTitlesByTabId) && + shallowRecordEqual(cached.ptyIdsByTabId, ptyIdsByTabId) + ) { + return cached.rows + } + + const entries = + migrationUnsupported.length > 0 + ? [ + ...liveEntries, + ...migrationUnsupported.flatMap((unsupported) => { + const entry = migrationUnsupportedToAgentStatusEntry(unsupported) + return entry ? [entry] : [] + }) + ] + : liveEntries + const rows = applyAgentRowLineage( + buildWorktreeAgentRows({ + tabs: tabs ?? [], + entries, + retained, + runtimePaneTitlesByTabId: paneTitlesByTabId, + ptyIdsByTabId, + terminalLayoutsByTabId, + runtimeAgentOrchestrationByPaneKey: orchestration, + now + }) + ) + if (cache) { + cache.computeCount += 1 + cache.lastComputedWorktreeIds.push(worktreeId) + cache.byWorktree.set(worktreeId, { + generation, + liveEntries, + migrationUnsupported, + retained, + tabs, + orchestration, + terminalLayoutsByTabId, + paneTitlesByTabId, + ptyIdsByTabId, + rows + }) + } + return rows +} diff --git a/src/renderer/src/components/settings/AppearanceWindowSidebarSection.tsx b/src/renderer/src/components/settings/AppearanceWindowSidebarSection.tsx index c8d63b3866a..1d82af8ed7d 100644 --- a/src/renderer/src/components/settings/AppearanceWindowSidebarSection.tsx +++ b/src/renderer/src/components/settings/AppearanceWindowSidebarSection.tsx @@ -203,8 +203,6 @@ export function AppearanceWindowSidebarSection({ title={translate('auto.components.settings.AppearancePane.dc29f3cc0d', 'Sidebar')} />
    - {/* Why: this setting lives with the sidebar layout controls; Settings only - names that ownership so we do not create a second stateful control. */} ({ useAppStore: (selector: (state: { settingsSearchQuery: string }) => unknown) => @@ -132,6 +132,23 @@ describe('ExperimentalPane', () => { ) }) + it('renders the Agents sidebar switch in Experimental without stale Appearance copy', () => { + const settings = getDefaultSettings('/tmp') + const markup = renderToStaticMarkup( + + ) + + expect(settings.experimentalAgentDashboardPopout).toBeUndefined() + expect(markup).toContain('Show Agents Button') + // The visible copy is the search entry's own description, so settings search can't + // advertise text the page doesn't show. + expect(markup).toContain(getExperimentalSearchEntry().agentsSidebar.description) + expect(markup).not.toContain('Window & Sidebar') + expect(getExperimentalPaneSearchEntries().map((entry) => entry.title)).toContain( + 'Show Agents Button' + ) + }) + it('renders the agent dashboard as an off-by-default searchable experiment', () => { const settings = getDefaultSettings('/tmp') const markup = renderToStaticMarkup( diff --git a/src/renderer/src/components/settings/ExperimentalPane.tsx b/src/renderer/src/components/settings/ExperimentalPane.tsx index b54e7ce75f2..4c3536c78ed 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.tsx @@ -36,15 +36,15 @@ export function ExperimentalPane({ }: ExperimentalPaneProps): React.JSX.Element { const searchQuery = useAppStore((s) => s.settingsSearchQuery) const showPet = matchesSettingsSearch(searchQuery, [getExperimentalSearchEntry().pet]) - const showAgentsView = matchesSettingsSearch(searchQuery, [ - getExperimentalSearchEntry().agentsView + const showNativeChat = matchesSettingsSearch(searchQuery, [ + getExperimentalSearchEntry().nativeChat + ]) + const showAgentsSidebar = matchesSettingsSearch(searchQuery, [ + getExperimentalSearchEntry().agentsSidebar ]) const showAgentDashboard = matchesSettingsSearch(searchQuery, [ getExperimentalSearchEntry().agentDashboard ]) - const showNativeChat = matchesSettingsSearch(searchQuery, [ - getExperimentalSearchEntry().nativeChat - ]) const showTerminalAttention = matchesSettingsSearch(searchQuery, [ getExperimentalSearchEntry().terminalAttention ]) @@ -64,6 +64,37 @@ export function ExperimentalPane({ return (
    + {showAgentsSidebar ? ( + +
    +
    + + {/* Same string the search entry advertises, so search results match the page. */} +

    + {getExperimentalSearchEntry().agentsSidebar.description} +

    +
    + + updateSettings({ showAgentsSidebar: settings.showAgentsSidebar === false }) + } + /> +
    +
    + ) : null} + + {showAgentDashboard ? ( + + ) : null} + {showPet ? ( ) : null} - {showAgentsView ? ( - -
    -
    - -

    - {translate( - 'auto.components.settings.ExperimentalPane.0277901cf7', - 'Adds an Agents entry to the left sidebar with a threaded worktree feed for completed agents, blocking questions, unread state, and worktree creation events. Experimental — the event model and UI may change.' - )} -

    -
    - - updateSettings({ - experimentalActivity: checked - }) - } - /> -
    -
    - ) : null} - - {showAgentDashboard ? ( - - ) : null} - {showNativeChat ? ( ) : null} diff --git a/src/renderer/src/components/settings/appearance-sidebar-search.ts b/src/renderer/src/components/settings/appearance-sidebar-search.ts index ab77d2a0ad0..2b01f7704a7 100644 --- a/src/renderer/src/components/settings/appearance-sidebar-search.ts +++ b/src/renderer/src/components/settings/appearance-sidebar-search.ts @@ -123,6 +123,27 @@ export const getShowPinnedWorktreesInGroupsEntry = createLocalizedCatalog( }) ) +export const getAgentsSidebarEntry = createLocalizedCatalog((): SettingsSearchEntry => ({ + title: translate('settings.appearance.agentsSidebar.title', 'Show Agents Button'), + description: translate( + 'settings.appearance.agentsSidebar.description', + 'Control whether the Agents tab appears in the left sidebar so you can monitor agent activity.' + ), + keywords: [ + ...translateSearchKeyword('auto.components.settings.general.search.baa263d6d8', 'agents'), + ...translateSearchKeyword('auto.components.settings.agents.search.96ba2373b6', 'agent'), + // Why: the sidebar row is labeled "Agent Dashboard"; searching that name must find this. + ...translateSearchKeyword( + 'auto.components.settings.experimental.search.agentDashboard.dashboard', + 'dashboard' + ), + ...translateSearchKeyword('auto.components.settings.appearance.search.5bff6a2ef0', 'sidebar'), + ...translateSearchKeyword('auto.components.settings.general.search.2a254b725e', 'tab'), + ...translateSearchKeyword('auto.components.settings.appearance.search.648eeada79', 'hide'), + ...translateSearchKeyword('auto.components.settings.appearance.search.ac79fe4a04', 'show') + ] +})) + export const getSidebarEntries = createLocalizedCatalog((): SettingsSearchEntry[] => [ { title: translate('auto.components.settings.appearance.search.155a1e7438', 'Show Tasks Button'), diff --git a/src/renderer/src/components/settings/experimental-search.ts b/src/renderer/src/components/settings/experimental-search.ts index e4ffd8f1a28..821de798301 100644 --- a/src/renderer/src/components/settings/experimental-search.ts +++ b/src/renderer/src/components/settings/experimental-search.ts @@ -5,6 +5,7 @@ import { translateSearchKeyword } from './settings-search-keywords' import { getNewWorktreeCardStyleSearchEntry } from './new-worktree-card-style-search-entry' import { getNativeChatExperimentalSearchEntry } from './native-chat-experimental-search-entry' import { getEphemeralVmsSearchEntry } from './ephemeral-vms-search' +import { getAgentsSidebarEntry } from './appearance-sidebar-search' export const getExperimentalPaneSearchEntries = createLocalizedCatalog( (): SettingsSearchEntry[] => [ @@ -46,55 +47,8 @@ export const getExperimentalPaneSearchEntries = createLocalizedCatalog( ) ] }, - { - title: translate('auto.components.settings.experimental.search.ccc5548ac5', 'Agents View'), - description: translate( - 'auto.components.settings.experimental.search.4d63251595', - 'Threaded left-sidebar feed for agent completions and blocking states.' - ), - keywords: [ - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.0d24759f14', - 'experimental' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.fa72e71f05', - 'agents' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.92a9357d1f', - 'agents view' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.244a0ecd3d', - 'activity' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.d01b3882ba', - 'notifications' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.10b52f79c1', - 'worktrees' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.ca5d1f3f46', - 'timeline' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.7b79081695', - 'unread' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.8facf10138', - 'bell' - ), - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.fe5688b761', - 'sidebar' - ) - ] - }, + getNativeChatExperimentalSearchEntry(), + getAgentsSidebarEntry(), { title: translate( 'auto.components.settings.experimental.search.agentDashboard.title', @@ -139,7 +93,6 @@ export const getExperimentalPaneSearchEntries = createLocalizedCatalog( ) ] }, - getNativeChatExperimentalSearchEntry(), { title: translate( 'auto.components.settings.experimental.search.9e4ddf776d', @@ -247,18 +200,16 @@ function findEntry(title: string): SettingsSearchEntry { export function getExperimentalSearchEntry() { return { pet: findEntry(translate('auto.components.settings.experimental.search.87d99e634b', 'Pet')), - agentsView: findEntry( - translate('auto.components.settings.experimental.search.ccc5548ac5', 'Agents View') + nativeChat: findEntry( + translate('auto.components.settings.experimental.search.nativeChat.title', 'Chat UI') ), + agentsSidebar: getAgentsSidebarEntry(), agentDashboard: findEntry( translate( 'auto.components.settings.experimental.search.agentDashboard.title', 'Agent Dashboard' ) ), - nativeChat: findEntry( - translate('auto.components.settings.experimental.search.nativeChat.title', 'Chat UI') - ), terminalAttention: findEntry( translate('auto.components.settings.experimental.search.9e4ddf776d', 'Terminal attention') ), diff --git a/src/renderer/src/components/sidebar/AgentDashboardSidebarEntry.tsx b/src/renderer/src/components/sidebar/AgentDashboardSidebarEntry.tsx index c67d8f4477f..4b997ca2cc0 100644 --- a/src/renderer/src/components/sidebar/AgentDashboardSidebarEntry.tsx +++ b/src/renderer/src/components/sidebar/AgentDashboardSidebarEntry.tsx @@ -68,6 +68,7 @@ export default function AgentDashboardSidebarEntry(): React.JSX.Element { return ( ) diff --git a/src/renderer/src/components/sidebar/Sidebar.test.tsx b/src/renderer/src/components/sidebar/Sidebar.test.tsx index 098ed7f203c..343c3f10c3c 100644 --- a/src/renderer/src/components/sidebar/Sidebar.test.tsx +++ b/src/renderer/src/components/sidebar/Sidebar.test.tsx @@ -3,7 +3,7 @@ import type { CSSProperties, ReactNode } from 'react' import { renderToStaticMarkup } from 'react-dom/server' import { tmpdir } from 'node:os' -import { cleanup, render } from '@testing-library/react' +import { cleanup, fireEvent, render, waitFor } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { getDefaultSettings } from '../../../../shared/constants' import type { GlobalSettings } from '../../../../shared/global-settings-types' @@ -32,11 +32,31 @@ vi.mock('@/hooks/useSidebarResize', () => ({ })) vi.mock('@/components/ui/tooltip', () => ({ - TooltipProvider: ({ children }: { children: ReactNode }) => <>{children} + TooltipProvider: ({ children }: { children: ReactNode }) => <>{children}, + Tooltip: ({ children }: { children: ReactNode }) => <>{children}, + TooltipTrigger: ({ children }: { children: ReactNode }) => <>{children}, + TooltipContent: ({ children }: { children: ReactNode }) => <>{children} })) vi.mock('./SidebarHeader', () => ({ - default: () =>
    + default: ({ + agentToolbar, + agentSearchRow + }: { + agentToolbar?: ReactNode + agentSearchRow?: ReactNode + }) => ( +
    + {agentToolbar} + {agentSearchRow} +
    + ) +})) + +vi.mock('./SidebarAgentsList', () => ({ + default: ({ query }: { query: string }) => ( +
    + ) })) vi.mock('./SidebarNav', () => ({ @@ -219,4 +239,47 @@ describe('Sidebar', () => { expect(fetchAllWorktrees).not.toHaveBeenCalled() }) + + it('clears the agents search query when the search row closes', async () => { + setSidebarState(getDefaultSettings(tmpdir())) + mocks.state = { ...mocks.state, sidebarBody: 'agents' } + const view = render(sidebarElement()) + const agentsList = await view.findByTestId('sidebar-agents-list') + + const searchToggle = view.getAllByRole('button', { name: 'Search' })[0] + fireEvent.click(searchToggle) + const searchInput = view.getByPlaceholderText('Filter...') + fireEvent.change(searchInput, { target: { value: 'deploy' } }) + expect(agentsList.getAttribute('data-query')).toBe('deploy') + + // Escape hides the row; a lingering query would silently keep filtering the list. + fireEvent.keyDown(searchInput, { key: 'Escape' }) + expect(view.queryByPlaceholderText('Filter...')).toBeNull() + expect(agentsList.getAttribute('data-query')).toBe('') + + fireEvent.click(searchToggle) + fireEvent.change(view.getByPlaceholderText('Filter...'), { target: { value: 'again' } }) + expect(agentsList.getAttribute('data-query')).toBe('again') + fireEvent.click(searchToggle) + expect(view.queryByPlaceholderText('Filter...')).toBeNull() + expect(agentsList.getAttribute('data-query')).toBe('') + }) + + it('closes the dashboard drawer when the dashboard experiment is disabled', async () => { + setSidebarState({ + ...getDefaultSettings(tmpdir()), + showAgentsSidebar: true, + experimentalAgentDashboardPopout: false + }) + const setAgentDashboardDrawerOpen = vi.fn() + mocks.state = { + ...mocks.state, + agentDashboardDrawerOpen: true, + setAgentDashboardDrawerOpen + } + + render(sidebarElement()) + + await waitFor(() => expect(setAgentDashboardDrawerOpen).toHaveBeenCalledWith(false)) + }) }) diff --git a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx new file mode 100644 index 00000000000..75c0c441bad --- /dev/null +++ b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx @@ -0,0 +1,144 @@ +import React, { useCallback, useEffect, useRef, useState } from 'react' +import { createPortal } from 'react-dom' +import { useAppStore } from '@/store' +import { ActivityScopeFilterChips } from '@/components/activity/activity-scope-filter-controls' +import { hasActivityThreadWorkspace } from '@/components/activity/activity-thread-actions' +import { useActivityThreadActionBindings } from '@/components/activity/use-activity-thread-action-bindings' +import { ActivityThreadListPane } from '@/components/activity/activity-thread-list-pane' +import { ActivityThreadOptionsMenu } from '@/components/activity/activity-thread-controls' +import { useAgentPaneThreads } from '@/components/activity/use-agent-pane-threads' +import type { ActivityGroupBy, ThreadReadFilter } from '@/components/activity/activity-thread-types' + +/** + * The Activity thread list, hosted in the sidebar as a navigator: selecting a + * row reveals that agent's pane in the workbench instead of swapping the view. + * Threads whose pane is gone stay listed but inert — activateThreadTerminal + * already no-ops without a live tab. + */ +export type SidebarAgentsListProps = { + readFilter: ThreadReadFilter + setReadFilter: (filter: ThreadReadFilter) => void + groupBy: ActivityGroupBy + setGroupBy: (groupBy: ActivityGroupBy) => void + query: string + setQuery: (query: string) => void + optionsTarget?: HTMLElement | null + scrollTopRef?: React.MutableRefObject +} + +export default function SidebarAgentsList({ + readFilter, + setReadFilter, + groupBy, + setGroupBy, + query, + setQuery, + optionsTarget, + scrollTopRef +}: SidebarAgentsListProps): React.JSX.Element { + // Why store-backed: these are persisted preferences (agents* UI fields), unlike the momentary read filter/search. + const compactMode = useAppStore((s) => s.agentsCompactMode) + const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) + const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) + const setShowChildAgents = useAppStore((s) => s.setAgentsShowChildAgents) + const [selectedPaneKey, setSelectedPaneKey] = useState(null) + const activityFilterInputRef = useRef(null) + + const { + storeData, + selectedPaneKeyIsLive, + effectiveSelectedPaneKey, + visibleThreads, + markAllReadThreads, + visibleThreadGroups + } = useAgentPaneThreads({ query, readFilter, groupBy, selectedPaneKey, showChildAgents }) + + useEffect(() => { + if (!selectedPaneKeyIsLive) { + setSelectedPaneKey(null) + } + }, [selectedPaneKeyIsLive]) + + const { + markThreadRead, + markThreadUnread, + selectThread, + jumpToWorkspace, + markAllThreadsRead, + hasUnreadThreads, + hasCompletedThreads, + handleClearCompleted + } = useActivityThreadActionBindings({ + visibleThreads, + markAllReadThreads, + acknowledgeAgents: storeData.acknowledgeAgents, + unacknowledgeAgents: storeData.unacknowledgeAgents, + setSelectedPaneKey + }) + + const canJumpToWorkspace = useCallback( + (thread: Parameters[0]) => + hasActivityThreadWorkspace(thread, { + worktreesByRepo: storeData.worktreesByRepo, + detectedWorktreesByRepo: storeData.detectedWorktreesByRepo, + folderWorkspaces: storeData.folderWorkspaces, + defaultHostId: storeData.defaultHostId + }), + [ + storeData.worktreesByRepo, + storeData.detectedWorktreesByRepo, + storeData.folderWorkspaces, + storeData.defaultHostId + ] + ) + + return ( + <> + } + scrollTopRef={scrollTopRef} + /> + {optionsTarget + ? createPortal( + , + optionsTarget + ) + : null} + + ) +} diff --git a/src/renderer/src/components/sidebar/SidebarHeader.test.tsx b/src/renderer/src/components/sidebar/SidebarHeader.test.tsx index 19af555be04..5df9ee619c5 100644 --- a/src/renderer/src/components/sidebar/SidebarHeader.test.tsx +++ b/src/renderer/src/components/sidebar/SidebarHeader.test.tsx @@ -8,23 +8,51 @@ import SidebarHeader from './SidebarHeader' ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true const mocks = vi.hoisted(() => ({ - openWorkspaceCreationComposerWithTourHandoff: vi.fn() + openWorkspaceCreationComposerWithTourHandoff: vi.fn(), + popoverContentProps: { current: null as Record | null }, + toast: vi.fn() })) type MockState = { repos: { id: string }[] groupBy: string + sidebarBody: 'workspaces' | 'agents' + sidebarWidth: number + setSidebarBody: (body: 'workspaces' | 'agents') => void openModal: (modal: string, data?: unknown) => void + updateSettings: (patch: Record) => void + activeContextualTourId: string | null + settings?: { + showAgentsSidebar?: boolean + experimentalAgentDashboardPopout?: boolean + agentsSidebarIntroShown?: boolean + agentsSidebarMigratedFromExperimental?: boolean + } } let mockState: MockState -vi.mock('@/store', () => ({ - useAppStore: (selector: (state: MockState) => unknown) => selector(mockState) +vi.mock('@/store', () => { + const useAppStore = (selector: (state: MockState) => unknown) => selector(mockState) + useAppStore.getState = () => mockState + return { useAppStore } +}) + +vi.mock('@/components/dashboard/useAgentBucketCounts', () => ({ + useAgentBucketCounts: () => ({ attention: 0, working: 0, done: 0, idle: 0 }) })) vi.mock('./SidebarWorkspaceOptionsMenu', () => ({ default: () => null })) +vi.mock('./workspace-options-menu-items', () => ({ + useWorkspaceOptionsFilterBadge: () => ({ + hasAnyFilter: false, + activeFilterCount: 0, + activeFilterLabel: '0 filters' + }), + WorkspaceOptionsMenuItems: () => null +})) + vi.mock('@/hooks/useShortcutLabel', () => ({ useShortcutLabel: () => '⌘N' })) vi.mock('@/components/ui/tooltip', () => ({ @@ -37,6 +65,21 @@ vi.mock('../contextual-tours/workspace-creation-tour-handoff', () => ({ openWorkspaceCreationComposerWithTourHandoff: mocks.openWorkspaceCreationComposerWithTourHandoff })) +vi.mock('sonner', () => ({ toast: mocks.toast })) + +// Deterministic popover: expose the open flag instead of relying on radix portals. +vi.mock('@/components/ui/popover', () => ({ + Popover: ({ children, open }: { children: React.ReactNode; open?: boolean }) => ( +
    {children}
    + ), + PopoverAnchor: ({ children }: { children: React.ReactNode }) => <>{children}, + PopoverArrow: () =>
    , + PopoverContent: ({ children, ...props }: { children: React.ReactNode }) => { + mocks.popoverContentProps.current = props + return <>{children} + } +})) + let container: HTMLDivElement let root: Root @@ -48,9 +91,32 @@ function newWorkspaceButton(): HTMLButtonElement { return button } +function workspaceViewLabel(): string { + const button = container.querySelector( + 'button[data-sidebar-section-title="projects"]' + ) + const label = button?.querySelector('span:not([aria-hidden])')?.textContent + if (!label) { + throw new Error('Workspace view label not rendered') + } + return label +} + beforeEach(() => { mocks.openWorkspaceCreationComposerWithTourHandoff.mockClear() - mockState = { repos: [], groupBy: 'repo', openModal: vi.fn() } + mocks.toast.mockClear() + mockState = { + repos: [], + groupBy: 'repo', + sidebarBody: 'workspaces', + sidebarWidth: 280, + setSidebarBody: vi.fn(), + openModal: vi.fn(), + updateSettings: vi.fn(), + activeContextualTourId: null, + // Hydrated settings: the Agents tab is hidden until settings load. + settings: { showAgentsSidebar: true } + } container = document.createElement('div') document.body.append(container) root = createRoot(container) @@ -90,4 +156,269 @@ describe('SidebarHeader', () => { expect(newWorkspaceButton().disabled).toBe(false) expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1) }) + + it('switches sidebar body to agents when clicking the agents tab in workspaces mode', () => { + act(() => { + root.render() + }) + + const agentTab = container.querySelector( + 'button[data-sidebar-section-title="agents"]' + ) + expect(agentTab).toBeTruthy() + + act(() => { + agentTab?.click() + }) + + expect(mockState.setSidebarBody).toHaveBeenCalledWith('agents') + }) + + it('switches sidebar body to workspaces when clicking the projects tab in agents mode', () => { + mockState.sidebarBody = 'agents' + act(() => { + root.render() + }) + + const projectsTab = container.querySelector( + 'button[data-sidebar-section-title="projects"]' + ) + expect(projectsTab).toBeTruthy() + + act(() => { + projectsTab?.click() + }) + + expect(mockState.setSidebarBody).toHaveBeenCalledWith('workspaces') + }) + + it('always labels the workspace view as Spaces', () => { + act(() => { + root.render() + }) + + expect(workspaceViewLabel()).toBe('Spaces') + + mockState.groupBy = 'workspace-status' + act(() => { + root.render() + }) + expect(workspaceViewLabel()).toBe('Spaces') + + mockState.sidebarBody = 'agents' + act(() => { + root.render() + }) + expect(workspaceViewLabel()).toBe('Spaces') + + mockState.sidebarBody = 'workspaces' + mockState.groupBy = 'none' + act(() => { + root.render() + }) + expect(workspaceViewLabel()).toBe('Spaces') + + mockState.sidebarBody = 'agents' + act(() => { + root.render() + }) + expect(workspaceViewLabel()).toBe('Spaces') + }) + + it('omits the Agents tab and returns to workspaces when it is disabled', () => { + mockState.sidebarBody = 'agents' + mockState.settings = { showAgentsSidebar: false } + + act(() => { + root.render() + }) + + expect(container.querySelector('button[data-sidebar-section-title="agents"]')).toBeNull() + expect(mockState.setSidebarBody).toHaveBeenCalledWith('workspaces') + }) + + it('does not render workspace action buttons in agents mode', () => { + mockState.sidebarBody = 'agents' + act(() => { + root.render() + }) + + expect(container.querySelector('[aria-label="New workspace"]')).toBeNull() + expect(container.querySelector('[aria-label="Add Project"]')).toBeNull() + }) + + it('keeps the view toggle and actions on one row at the default sidebar width', () => { + act(() => { + root.render() + }) + + const headerRow = container.querySelector('[role="radiogroup"]')?.parentElement + const headerClasses = new Set(headerRow?.className.split(/\s+/) ?? []) + expect(headerClasses.has('flex-wrap')).toBe(false) + expect(headerClasses.has('h-9')).toBe(true) + expect(container.querySelector('[aria-label="Add Project"]')).toBeTruthy() + expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy() + }) + + it('keeps New workspace and a more menu on one row at compact width', () => { + mockState.sidebarWidth = 220 + act(() => { + root.render() + }) + + expect(container.querySelector('[aria-label="Add Project"]')).toBeNull() + expect(container.querySelector('[aria-label="New workspace"]')).toBeTruthy() + expect(container.querySelector('[aria-label="More workspace actions"]')).toBeTruthy() + + act(() => { + newWorkspaceButton().click() + }) + expect(mocks.openWorkspaceCreationComposerWithTourHandoff).toHaveBeenCalledTimes(1) + }) + + it('keeps the intro closed and unstamped before settings hydrate', () => { + mockState.settings = undefined + act(() => { + root.render() + }) + + expect(container.querySelector('[data-intro-open]')).toBeNull() + + const projectsTab = container.querySelector( + 'button[data-sidebar-section-title="projects"]' + ) + act(() => { + projectsTab?.click() + }) + expect(mockState.updateSettings).not.toHaveBeenCalled() + }) + + it('does not reset a persisted agents body before settings hydrate', () => { + mockState.settings = undefined + mockState.sidebarBody = 'agents' + act(() => { + root.render() + }) + + expect(mockState.setSidebarBody).not.toHaveBeenCalled() + }) + + it('opens the intro once hydrated and stamps it only while it is on screen', () => { + act(() => { + root.render() + }) + + expect(container.querySelector('[data-intro-open]')).toBeTruthy() + + const agentTab = container.querySelector( + 'button[data-sidebar-section-title="agents"]' + ) + act(() => { + agentTab?.click() + }) + expect(mockState.updateSettings).toHaveBeenCalledWith({ agentsSidebarIntroShown: true }) + }) + + it('hides the Agents tab when a new user chooses Hide Agents', () => { + act(() => { + root.render() + }) + + const deferButton = Array.from(container.querySelectorAll('button')).find( + (button) => button.textContent?.trim() === 'Hide Agents' + ) + expect(deferButton).toBeTruthy() + act(() => { + deferButton?.click() + }) + + expect(mockState.updateSettings).toHaveBeenCalledWith({ + agentsSidebarIntroShown: true, + showAgentsSidebar: false + }) + expect(mocks.toast).toHaveBeenCalledWith( + 'Agents tab hidden. Re-enable it in Settings → Experimental.' + ) + }) + + it('shows only Open Agents for migrated users', () => { + mockState.settings = { + showAgentsSidebar: true, + agentsSidebarMigratedFromExperimental: true + } + act(() => { + root.render() + }) + + const introButtons = Array.from(container.querySelectorAll('button')).filter((button) => + ['Open Agents', 'Hide Agents'].includes(button.textContent?.trim() ?? '') + ) + expect(introButtons).toHaveLength(1) + expect(introButtons[0]?.textContent?.trim()).toBe('Open Agents') + }) + + it('prevents auto-focus and outside focus transfers from dismissing the intro popover', () => { + act(() => { + root.render() + }) + + const props = mocks.popoverContentProps.current as { + onOpenAutoFocus?: (event: Event) => void + onFocusOutside?: (event: Event) => void + } | null + + expect(props).toBeTruthy() + + const openEvent = new Event('openAutoFocus') + const openPreventDefault = vi.spyOn(openEvent, 'preventDefault') + props?.onOpenAutoFocus?.(openEvent) + expect(openPreventDefault).toHaveBeenCalled() + + const focusOutsideEvent = new Event('focusOutside') + const focusOutsidePreventDefault = vi.spyOn(focusOutsideEvent, 'preventDefault') + props?.onFocusOutside?.(focusOutsideEvent) + expect(focusOutsidePreventDefault).toHaveBeenCalled() + }) + + it('never re-stamps the intro after it was acknowledged', () => { + mockState.settings = { showAgentsSidebar: true, agentsSidebarIntroShown: true } + act(() => { + root.render() + }) + + expect(container.querySelector('[data-intro-open]')).toBeNull() + + const agentTab = container.querySelector( + 'button[data-sidebar-section-title="agents"]' + ) + act(() => { + agentTab?.click() + }) + expect(mockState.updateSettings).not.toHaveBeenCalled() + }) + + it('does not expose the deprecated full Agents view in agents mode', () => { + mockState.settings = { showAgentsSidebar: true, agentsSidebarIntroShown: true } + mockState.sidebarBody = 'agents' + act(() => { + root.render() + }) + + expect(container.querySelector('[aria-label="Open full Agents view"]')).toBeNull() + }) + + it('switches to compact actions only below the wide-layout breakpoint', () => { + mockState.sidebarWidth = 234 + act(() => { + root.render() + }) + expect(container.querySelector('[aria-label="More workspace actions"]')).toBeTruthy() + + mockState.sidebarWidth = 235 + act(() => { + root.render() + }) + expect(container.querySelector('[aria-label="More workspace actions"]')).toBeNull() + expect(container.querySelector('[aria-label="Add Project"]')).toBeTruthy() + }) }) diff --git a/src/renderer/src/components/sidebar/SidebarHeader.tsx b/src/renderer/src/components/sidebar/SidebarHeader.tsx index b31eb5d378b..dfc22c8b1bc 100644 --- a/src/renderer/src/components/sidebar/SidebarHeader.tsx +++ b/src/renderer/src/components/sidebar/SidebarHeader.tsx @@ -1,88 +1,207 @@ -import React from 'react' -import { FolderPlus, Plus } from 'lucide-react' +import React, { useEffect, useId } from 'react' +import { useTranslation } from 'react-i18next' import { useAppStore } from '@/store' -import { Button } from '@/components/ui/button' -import { Tooltip, TooltipTrigger, TooltipContent } from '@/components/ui/tooltip' -import SidebarWorkspaceOptionsMenu from './SidebarWorkspaceOptionsMenu' -import { useShortcutLabel } from '@/hooks/useShortcutLabel' -import { openWorkspaceCreationComposerWithTourHandoff } from '../contextual-tours/workspace-creation-tour-handoff' import { translate } from '@/i18n/i18n' +import { SidebarViewToggle } from './sidebar-view-toggle' +import { SidebarHeaderActions } from './sidebar-header-actions' +import { shouldShowAgentsSidebar } from './agents-sidebar-visibility' +import { Popover, PopoverAnchor, PopoverArrow, PopoverContent } from '@/components/ui/popover' +import { Button } from '@/components/ui/button' +import { Sparkles } from 'lucide-react' +import { toast } from 'sonner' type SidebarHeaderProps = { onWorkspaceBoardMenuOpenChange: (open: boolean) => void + agentToolbar?: React.ReactNode + agentSearchRow?: React.ReactNode + showAgentsSidebar?: boolean } const SidebarHeader = React.memo(function SidebarHeader({ - onWorkspaceBoardMenuOpenChange + onWorkspaceBoardMenuOpenChange, + agentToolbar, + agentSearchRow, + showAgentsSidebar: showAgentsSidebarProp }: SidebarHeaderProps) { - const openModal = useAppStore((s) => s.openModal) - const newWorktreeShortcutLabel = useShortcutLabel('workspace.create') - const groupBy = useAppStore((s) => s.groupBy) - const sidebarTitle = groupBy === 'repo' ? 'Projects' : 'Workspaces' + // Subscribe this memoized header to locale changes before using translate(). + useTranslation() + const sidebarBody = useAppStore((s) => s.sidebarBody ?? 'workspaces') + // Why the derived boolean, not s.settings: the settings object gets a new identity on + // every write, which would re-render this memoized header subtree each time. + const showAgentsSidebarFromStore = useAppStore((s) => shouldShowAgentsSidebar(s.settings)) + const showAgentsSidebar = showAgentsSidebarProp ?? showAgentsSidebarFromStore + const setSidebarBody = useAppStore((s) => s.setSidebarBody) + const updateSettings = useAppStore((s) => s.updateSettings) + const agentsSidebarIntroShown = useAppStore((s) => s.settings?.agentsSidebarIntroShown === true) + const migratedFromExperimental = useAppStore( + (s) => s.settings?.agentsSidebarMigratedFromExperimental === true + ) + const introTitleId = useId() + const introDescriptionId = useId() + // Why: settings are null until hydration; deriving intro visibility from the + // null default would flash the popover open (and stamp it shown) every launch. + const settingsHydrated = useAppStore((s) => s.settings != null) + const agentsViewActive = showAgentsSidebar && sidebarBody === 'agents' + const introOpen = settingsHydrated && showAgentsSidebar && !agentsSidebarIntroShown + const acknowledgeIntro = React.useCallback(() => { + void updateSettings?.({ agentsSidebarIntroShown: true }) + }, [updateSettings]) + const deferAgentsIntro = React.useCallback(() => { + // Hide the new tab; users can re-enable it in Settings. + void updateSettings?.({ agentsSidebarIntroShown: true, showAgentsSidebar: false }) + toast( + translate( + 'agentsSidebarIntro.new.hiddenToast', + 'Agents tab hidden. Re-enable it in Settings → Experimental.' + ) + ) + }, [updateSettings]) + + useEffect(() => { + // Wait for hydration: settings null must not clobber a persisted 'agents' body. + if (settingsHydrated && !showAgentsSidebar && sidebarBody === 'agents') { + setSidebarBody?.('workspaces') + } + }, [setSidebarBody, settingsHydrated, showAgentsSidebar, sidebarBody]) return ( -
    -
    - +
    + { + if (!open) { + acknowledgeIntro() + } + }} > - {sidebarTitle} - +
    + { + // Only stamp the intro as seen when it is actually on screen. + if (introOpen) { + acknowledgeIntro() + } + setSidebarBody?.(value as 'workspaces' | 'agents') + }} + options={[ + { + value: 'workspaces', + label: translate('auto.components.sidebar.SidebarHeader.spaces', 'Spaces'), + sectionTitle: 'projects' + }, + ...(showAgentsSidebar + ? [ + { + value: 'agents' as const, + label: translate('dashboard.sidebar.label', 'Agents'), + sectionTitle: 'agents' as const, + renderWrapper: (button: React.ReactNode) => ( + {button} + ) + } + ] + : []) + ]} + /> +
    + {/* Why: prevent startup terminal/editor auto-focus from dismissing the intro popover. */} + event.preventDefault()} + onFocusOutside={(event) => event.preventDefault()} + aria-labelledby={introTitleId} + aria-describedby={introDescriptionId} + > + + + + + + +
    + +
    +
    + +

    + {migratedFromExperimental + ? translate('agentsSidebarIntro.migrated.title', 'Agents are easier to find') + : translate('agentsSidebarIntro.new.title', 'Meet your Agents tab')} +

    +
    +

    + {migratedFromExperimental + ? translate( + 'agentsSidebarIntro.migrated.description', + 'Your Agents view is now a dedicated sidebar tab. Your activity and filters are preserved.' + ) + : translate( + 'agentsSidebarIntro.new.description', + 'See what your agents are working on, what is done, and where you need to step in.' + )} +

    +
    +
    + {!migratedFromExperimental ? ( + + ) : null} + +
    +
    +
    +
    + {agentsViewActive ? ( +
    + {/* Do not add an expand action: the full Agents view is deprecated and must not open. */} + {agentToolbar} +
    + ) : null} + {!agentsViewActive ? ( + + ) : null}
    -
    - - - - - - - - {translate('auto.components.sidebar.SidebarHeader.25a95899c9', 'Add Project')} - - - - - - - - - {translate( - 'auto.components.sidebar.SidebarHeader.ca6f729da2', - 'New workspace ({{value0}})', - { value0: newWorktreeShortcutLabel } - )} - - -
    -
    + {agentsViewActive ? agentSearchRow : null} + ) }) diff --git a/src/renderer/src/components/sidebar/SidebarNav.test.tsx b/src/renderer/src/components/sidebar/SidebarNav.test.tsx index 19c93a2d1c4..e3612db3858 100644 --- a/src/renderer/src/components/sidebar/SidebarNav.test.tsx +++ b/src/renderer/src/components/sidebar/SidebarNav.test.tsx @@ -89,8 +89,6 @@ vi.mock('@/components/ui/context-menu', () => ({ import SidebarNav, { getSetupGuideSidebarEntryReady, - shouldShowAgentDashboardButton, - shouldShowAgentsButton, shouldShowAutomationsButton, shouldShowArtifactsButton, shouldShowMobileButton, @@ -220,45 +218,18 @@ describe('SidebarNav', () => { setSidebarState() }) - it('hides the Agents entry while settings are loading', () => { - expect(shouldShowAgentsButton(null)).toBe(false) - }) - - it('hides the Agents entry while the experimental Agents view is off', () => { - expect( - shouldShowAgentsButton({ - ...getDefaultSettings('/tmp'), - experimentalActivity: false - }) - ).toBe(false) - }) - - it('shows the Agents entry when the experimental Agents view is on', () => { - expect( - shouldShowAgentsButton({ - ...getDefaultSettings('/tmp'), - experimentalActivity: true - }) - ).toBe(true) - }) - - it('shows the Agent Dashboard entry only when its experiment is enabled', () => { - expect(shouldShowAgentDashboardButton(null)).toBe(false) - expect(shouldShowAgentDashboardButton({ experimentalAgentDashboardPopout: false })).toBe(false) - expect(shouldShowAgentDashboardButton({ experimentalAgentDashboardPopout: true })).toBe(true) - }) - - it('keeps the Agent Dashboard row unmounted by default', async () => { + it('keeps the Agent Dashboard row unmounted while its experiment is off', async () => { const container = await renderSidebarNav() expect(queryButtonByText(container, 'Agent Dashboard')).toBeNull() expect(mocks.getAgentBucketCounts).not.toHaveBeenCalled() }) - it('mounts the Agent Dashboard row after opt-in', async () => { + it('mounts the Agent Dashboard row only when its experiment is enabled', async () => { setSidebarState({ settings: { ...getDefaultSettings('/tmp'), + showAgentsSidebar: false, experimentalAgentDashboardPopout: true } }) diff --git a/src/renderer/src/components/sidebar/SidebarNav.tsx b/src/renderer/src/components/sidebar/SidebarNav.tsx index bdb98ead14f..08fc1c941a5 100644 --- a/src/renderer/src/components/sidebar/SidebarNav.tsx +++ b/src/renderer/src/components/sidebar/SidebarNav.tsx @@ -1,10 +1,8 @@ import React from 'react' -import { Bell, BookOpen, CalendarClock, EyeOff, Files, Search, Smartphone } from 'lucide-react' +import { BookOpen, CalendarClock, EyeOff, Files, Search, Smartphone } from 'lucide-react' import { useTranslation } from 'react-i18next' import { useAppStore } from '@/store' import { cn } from '@/lib/utils' -import type { GlobalSettings } from '../../../../shared/global-settings-types' -import { useActivityUnreadCount } from '@/components/activity/useActivityUnreadCount' import { useShortcutKeyComboDetails } from '@/hooks/useShortcutLabel' import { ShortcutKeyCombo } from '@/components/ShortcutKeyCombo' import { useMobileSidebarOnboardingBadge } from './mobile-sidebar-onboarding-badge' @@ -16,45 +14,40 @@ import { SidebarTaskNavButton } from './SidebarTaskNavButton' import { HideSidebarMenu } from './sidebar-nav-controls' import { translate } from '@/i18n/i18n' import { lazyWithRetry } from '@/lib/lazy-with-retry' +import type { GlobalSettings } from '../../../../shared/global-settings-types' export { getSetupGuideSidebarEntryReady, shouldShowSetupGuideEntry } from './SetupGuideSidebarEntry' -export function shouldShowAgentsButton( - settings: Pick | null | undefined -): boolean { - return settings?.experimentalActivity === true -} - -export function shouldShowAgentDashboardButton( - settings: Pick | null | undefined -): boolean { - return settings?.experimentalAgentDashboardPopout === true -} - export function shouldShowMobileButton( - settings: Pick | null | undefined + settings: Partial> | null | undefined ): boolean { return settings?.showMobileButton !== false } export function shouldShowAutomationsButton( - settings: Pick | null | undefined + settings: Partial> | null | undefined ): boolean { return settings?.showAutomationsButton !== false } export function shouldShowArtifactsButton( - settings: Pick | null | undefined + settings: Partial> | null | undefined ): boolean { return settings?.showArtifactsButton === true } export function shouldShowSkillsButton( - settings: Pick | null | undefined + settings: Partial> | null | undefined ): boolean { return settings?.showSkillsButton === true } +export function shouldShowAgentDashboardButton( + settings: Partial> | null | undefined +): boolean { + return settings?.experimentalAgentDashboardPopout === true +} + const AgentDashboardSidebarEntry = lazyWithRetry(() => import('./AgentDashboardSidebarEntry')) const SidebarNav = React.memo(function SidebarNav() { @@ -63,30 +56,21 @@ const SidebarNav = React.memo(function SidebarNav() { useTranslation() const worktreePaletteShortcutCombos = useShortcutKeyComboDetails('worktree.palette') const openAutomationsPage = useAppStore((s) => s.openAutomationsPage) - const openActivityPage = useAppStore((s) => s.openActivityPage) const openMobilePage = useAppStore((s) => s.openMobilePage) const openArtifactsPage = useAppStore((s) => s.openArtifactsPage) const openSkillsPage = useAppStore((s) => s.openSkillsPage) const openModal = useAppStore((s) => s.openModal) const updateSettings = useAppStore((s) => s.updateSettings) const activeView = useAppStore((s) => s.activeView) - const experimentalSidebarButtons = useAppStore( - (s) => - (shouldShowAgentsButton(s.settings) ? 1 : 0) | - (shouldShowAgentDashboardButton(s.settings) ? 2 : 0) - ) - const showAgentsButton = (experimentalSidebarButtons & 1) !== 0 - const showAgentDashboardButton = (experimentalSidebarButtons & 2) !== 0 + const showAgentDashboardButton = useAppStore((s) => shouldShowAgentDashboardButton(s.settings)) const showAutomationsButton = useAppStore((s) => shouldShowAutomationsButton(s.settings)) const showMobileButton = useAppStore((s) => shouldShowMobileButton(s.settings)) const showArtifactsButton = useAppStore((s) => shouldShowArtifactsButton(s.settings)) const showSkillsButton = useAppStore((s) => shouldShowSkillsButton(s.settings)) const automationsActive = activeView === 'automations' - const activityActive = activeView === 'activity' const mobileActive = activeView === 'mobile' const artifactsActive = activeView === 'artifacts' const skillsActive = activeView === 'skills' - const activityUnreadCount = useActivityUnreadCount(showAgentsButton, 'sidebar-badge') const mobileOnboardingBadge = useMobileSidebarOnboardingBadge(showMobileButton) const hideAutomationsButton = React.useCallback(() => { void updateSettings({ showAutomationsButton: false }) @@ -229,35 +213,6 @@ const SidebarNav = React.memo(function SidebarNav() { ) : null} - {showAgentsButton ? ( - - ) : null} {showMobileButton ? ( diff --git a/src/renderer/src/components/sidebar/SidebarRepositoryFilterSection.tsx b/src/renderer/src/components/sidebar/SidebarRepositoryFilterSection.tsx index 65623ee8983..1db46d84660 100644 --- a/src/renderer/src/components/sidebar/SidebarRepositoryFilterSection.tsx +++ b/src/renderer/src/components/sidebar/SidebarRepositoryFilterSection.tsx @@ -34,14 +34,22 @@ function getProjectFilterVisibilityLabel({ type SidebarRepositoryFilterSectionProps = { preserveWorkspaceBoardOpen?: boolean + // Why: the Agents view reuses this section with its own persisted filter; + // absent props fall back to the workspace-nav filter state. + filterRepoIds?: readonly string[] + setFilterRepoIds?: (ids: string[]) => void } const SidebarRepositoryFilterSection = React.memo(function SidebarRepositoryFilterSection({ - preserveWorkspaceBoardOpen = false + preserveWorkspaceBoardOpen = false, + filterRepoIds: filterRepoIdsProp, + setFilterRepoIds: setFilterRepoIdsProp }: SidebarRepositoryFilterSectionProps) { - const filterRepoIds = useAppStore((s) => s.filterRepoIds) - const setFilterRepoIds = useAppStore((s) => s.setFilterRepoIds) + const workspaceFilterRepoIds = useAppStore((s) => s.filterRepoIds) + const setWorkspaceFilterRepoIds = useAppStore((s) => s.setFilterRepoIds) const repos = useAppStore((s) => s.repos) + const filterRepoIds = filterRepoIdsProp ?? workspaceFilterRepoIds + const setFilterRepoIds = setFilterRepoIdsProp ?? setWorkspaceFilterRepoIds const canFilterRepos = repos.length > 1 // Why: derive from current repos so stale ids (e.g. lingering after a repo diff --git a/src/renderer/src/components/sidebar/SidebarWorkspaceOptionsMenu.tsx b/src/renderer/src/components/sidebar/SidebarWorkspaceOptionsMenu.tsx index 302d654faea..9f230e94f95 100644 --- a/src/renderer/src/components/sidebar/SidebarWorkspaceOptionsMenu.tsx +++ b/src/renderer/src/components/sidebar/SidebarWorkspaceOptionsMenu.tsx @@ -1,31 +1,17 @@ -import React, { useCallback, useMemo, useState } from 'react' +import React, { useCallback, useState } from 'react' import { SlidersHorizontal } from 'lucide-react' -import { useAppStore } from '@/store' import { Button } from '@/components/ui/button' import { DropdownMenu, DropdownMenuContent, - DropdownMenuLabel, - DropdownMenuRadioGroup, - DropdownMenuRadioItem, - DropdownMenuSeparator, - DropdownMenuSub, - DropdownMenuSubContent, - DropdownMenuSubTrigger, DropdownMenuTrigger } from '@/components/ui/dropdown-menu' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' -import { DEFAULT_SHOW_SLEEPING_WORKSPACES } from '../../../../shared/constants' -import { isSleepingSweepExemptionNarrowingList } from './visible-worktrees' -import SidebarRepositoryFilterSection from './SidebarRepositoryFilterSection' -import SidebarWorkspaceFilterSection from './SidebarWorkspaceFilterSection' -import { getSidebarHostVisibilityLabel, shouldShowHostScopeControls } from './sidebar-host-options' -import { useSidebarHostScopeOptions } from './use-sidebar-host-scope-options' -import { SidebarHostScopeMenuSection } from './SidebarHostScopeMenuSection' -import { PROJECT_ORDER_OPTIONS, SORT_OPTIONS } from './sidebar-workspace-option-items' -import { WorktreeCardDisplayMenuSection } from './WorktreeCardDisplayMenuSection' import { translate } from '@/i18n/i18n' -import { SidebarGroupByToggle } from './SidebarGroupByToggle' +import { + useWorkspaceOptionsFilterBadge, + WorkspaceOptionsMenuItems +} from './workspace-options-menu-items' type SidebarWorkspaceOptionsMenuProps = { preserveWorkspaceBoardOpen?: boolean @@ -36,28 +22,8 @@ const SidebarWorkspaceOptionsMenu = React.memo(function SidebarWorkspaceOptionsM preserveWorkspaceBoardOpen = false, onMenuOpenChange }: SidebarWorkspaceOptionsMenuProps) { - const showSleepingWorkspaces = useAppStore((s) => s.showSleepingWorkspaces) - const hideDefaultBranchWorkspace = useAppStore((s) => s.hideDefaultBranchWorkspace) - const hideAutomationGeneratedWorkspaces = useAppStore((s) => s.hideAutomationGeneratedWorkspaces) - const hideCliCreatedWorkspaces = useAppStore((s) => s.hideCliCreatedWorkspaces) - const hideDetachedHeadWorkspaces = useAppStore((s) => s.hideDetachedHeadWorkspaces) - const hideWorkspacesFromOtherDevices = useAppStore((s) => s.hideWorkspacesFromOtherDevices) - const alwaysShowDefaultBranchWorkspace = useAppStore((s) => s.alwaysShowDefaultBranchWorkspace) - const filterRepoIds = useAppStore((s) => s.filterRepoIds) - const repos = useAppStore((s) => s.repos) - const setWorkspaceHostScope = useAppStore((s) => s.setWorkspaceHostScope) - const visibleWorkspaceHostIds = useAppStore((s) => s.visibleWorkspaceHostIds) - const setVisibleWorkspaceHostIds = useAppStore((s) => s.setVisibleWorkspaceHostIds) - const sortBy = useAppStore((s) => s.sortBy) - const setSortBy = useAppStore((s) => s.setSortBy) - const groupBy = useAppStore((s) => s.groupBy) - const setGroupBy = useAppStore((s) => s.setGroupBy) - const projectOrderBy = useAppStore((s) => s.projectOrderBy) - const setProjectOrderBy = useAppStore((s) => s.setProjectOrderBy) - const [open, setOpen] = useState(false) - const { hostOptions } = useSidebarHostScopeOptions() - const showHostScopeControls = shouldShowHostScopeControls(hostOptions) + const { hasAnyFilter, activeFilterCount, activeFilterLabel } = useWorkspaceOptionsFilterBadge() const handleOpenChange = useCallback( (next: boolean) => { @@ -67,52 +33,6 @@ const SidebarWorkspaceOptionsMenu = React.memo(function SidebarWorkspaceOptionsM [onMenuOpenChange] ) - // Why: derive from current repos so stale ids (e.g. lingering after a repo - // is removed) don't inflate counts or falsely signal an applied filter. - const selectedCount = useMemo(() => { - let count = 0 - for (const repo of repos) { - if (filterRepoIds.includes(repo.id)) { - count += 1 - } - } - return count - }, [repos, filterRepoIds]) - const hasRepoFilter = selectedCount > 0 - const hasSleepingFilter = showSleepingWorkspaces !== DEFAULT_SHOW_SLEEPING_WORKSPACES - const hasHostVisibilityFilter = visibleWorkspaceHostIds !== null - // Why gated on the parent row: the exemption only narrows the list during the - // "Hide sleeping" sweep, which is also the only time its row is rendered. - const hasSleepingExemptionFilter = isSleepingSweepExemptionNarrowingList( - showSleepingWorkspaces, - alwaysShowDefaultBranchWorkspace - ) - const hasAnyFilter = - hasSleepingFilter || - hideDefaultBranchWorkspace || - hideAutomationGeneratedWorkspaces || - hideCliCreatedWorkspaces || - hideDetachedHeadWorkspaces || - hideWorkspacesFromOtherDevices || - hasSleepingExemptionFilter || - hasRepoFilter || - hasHostVisibilityFilter - const activeFilterCount = - (hasSleepingFilter ? 1 : 0) + - (hideDefaultBranchWorkspace ? 1 : 0) + - (hideAutomationGeneratedWorkspaces ? 1 : 0) + - (hideCliCreatedWorkspaces ? 1 : 0) + - (hideDetachedHeadWorkspaces ? 1 : 0) + - (hideWorkspacesFromOtherDevices ? 1 : 0) + - (hasSleepingExemptionFilter ? 1 : 0) + - (hasHostVisibilityFilter ? 1 : 0) + - selectedCount - const activeFilterLabel = `${activeFilterCount} ${activeFilterCount === 1 ? 'filter' : 'filters'}` - const sortLabel = SORT_OPTIONS.find((opt) => opt.id === sortBy)?.label ?? 'Sort' - const projectOrderLabel = - PROJECT_ORDER_OPTIONS.find((opt) => opt.id === projectOrderBy)?.label ?? 'Manual' - const hostVisibilityLabel = getSidebarHostVisibilityLabel(visibleWorkspaceHostIds, hostOptions) - return ( @@ -171,136 +91,7 @@ const SidebarWorkspaceOptionsMenu = React.memo(function SidebarWorkspaceOptionsM className="w-72 pb-2" data-workspace-board-preserve-open={preserveWorkspaceBoardOpen ? '' : undefined} > - {/* Why: host + project filters share one section and the same single-row - shell as Sort by (label left, value right) so the menu stays flat. */} - {(showHostScopeControls || repos.length > 1) && ( - <> - - {translate('auto.components.sidebar.SidebarWorkspaceOptionsMenu.showSection', 'Show')} - - {showHostScopeControls && ( - - )} - - - - )} - - - {translate('auto.components.sidebar.SidebarWorkspaceOptionsMenu.dc0bb670bc', 'Group by')} - -
    - -
    - - - - - - - {translate( - 'auto.components.sidebar.SidebarWorkspaceOptionsMenu.7bada3b1ab', - 'Sort by' - )} - - {sortLabel} - - - - setSortBy(v as typeof sortBy)} - > - {SORT_OPTIONS.map((opt) => { - const radioItem = ( - e.preventDefault()} - > - {opt.label} - - ) - if (!opt.description) { - return radioItem - } - return ( - - {radioItem} - - {opt.description} - - - ) - })} - - - - - {/* Why: project order only has a visible effect when grouping by - project; hide it in none/status/PR modes to avoid a dead control. */} - {groupBy === 'repo' && ( - - - - - {translate( - 'auto.components.sidebar.SidebarWorkspaceOptionsMenu.09faabd875', - 'Project order' - )} - - - {projectOrderLabel} - - - - - setProjectOrderBy(v as typeof projectOrderBy)} - > - {PROJECT_ORDER_OPTIONS.map((opt) => ( - - - e.preventDefault()} - > - {opt.label} - - - - {opt.description} - - - ))} - - - - )} - - - - - +
    ) diff --git a/src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts b/src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts new file mode 100644 index 00000000000..21639737994 --- /dev/null +++ b/src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts @@ -0,0 +1,15 @@ +import { describe, expect, it } from 'vitest' +import { shouldShowAgentsSidebar } from './agents-sidebar-visibility' + +describe('shouldShowAgentsSidebar', () => { + it('hides while settings are not yet hydrated', () => { + expect(shouldShowAgentsSidebar(null)).toBe(false) + expect(shouldShowAgentsSidebar(undefined)).toBe(false) + }) + + it('defaults on and honors only the dedicated setting', () => { + expect(shouldShowAgentsSidebar({})).toBe(true) + expect(shouldShowAgentsSidebar({ showAgentsSidebar: true })).toBe(true) + expect(shouldShowAgentsSidebar({ showAgentsSidebar: false })).toBe(false) + }) +}) diff --git a/src/renderer/src/components/sidebar/agents-sidebar-visibility.ts b/src/renderer/src/components/sidebar/agents-sidebar-visibility.ts new file mode 100644 index 00000000000..f0e978a4d2c --- /dev/null +++ b/src/renderer/src/components/sidebar/agents-sidebar-visibility.ts @@ -0,0 +1,11 @@ +import { + resolveAgentsSidebarVisible, + type AgentsSidebarVisibilitySettings +} from '../../../../shared/agents-sidebar-visibility' + +export function shouldShowAgentsSidebar( + settings: Partial | null | undefined +): boolean { + // Settings hydrate after first render; avoid flashing UI for opted-out profiles. + return settings ? resolveAgentsSidebarVisible(settings) : false +} diff --git a/src/renderer/src/components/sidebar/index.tsx b/src/renderer/src/components/sidebar/index.tsx index bed75283e38..20d2fef711f 100644 --- a/src/renderer/src/components/sidebar/index.tsx +++ b/src/renderer/src/components/sidebar/index.tsx @@ -1,21 +1,33 @@ import React, { useEffect, useMemo } from 'react' +import { useTranslation } from 'react-i18next' import { useAppStore } from '@/store' -import { TooltipProvider } from '@/components/ui/tooltip' +import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from '@/components/ui/tooltip' import { useSidebarResize } from '@/hooks/useSidebarResize' import SidebarHeader from './SidebarHeader' import SidebarNav from './SidebarNav' +import { shouldShowAgentsSidebar } from './agents-sidebar-visibility' import SetupScriptPromptCard from './SetupScriptPromptCard' import WorktreeList from './WorktreeList' import SidebarToolbar from './SidebarToolbar' import WorkspaceKanbanDrawer from './WorkspaceKanbanDrawer' import type { VirtualizedScrollAnchor } from '@/hooks/useVirtualizedScrollAnchor' import { cn } from '@/lib/utils' -import { FolderPlus, Loader2 } from 'lucide-react' +import { BellDot, FolderPlus, Loader2, Search } from 'lucide-react' +import { Button } from '@/components/ui/button' +import { Input } from '@/components/ui/input' +import type { ActivityGroupBy, ThreadReadFilter } from '@/components/activity/activity-thread-types' +import { ActivityThreadCollapseContext } from '@/components/activity/activity-thread-collapse-context' import { useSidebarProjectDrop } from './useSidebarProjectDrop' import { useWorkspaceBoardPanel } from './useWorkspaceBoardPanel' +import { useWorkspaceRevealBodyRedirect } from './use-workspace-reveal-body-redirect' import { resolveLeftSidebarStyleVariables } from '@/lib/left-sidebar-appearance' import { useSystemPrefersDark } from '@/components/terminal-pane/use-system-prefers-dark' import { lazyWithRetry } from '@/lib/lazy-with-retry' +import { translate } from '@/i18n/i18n' + +// Why lazy: the Agents list pulls the whole activity pipeline (virtualizer, markdown +// previews, thread derivation); users on the workspace view should not load or render any of it. +const SidebarAgentsList = lazyWithRetry(() => import('./SidebarAgentsList')) const WorktreeMetaDialog = lazyWithRetry(() => import('./WorktreeMetaDialog')) const RemoveFolderDialog = lazyWithRetry(() => import('./RemoveFolderDialog')) @@ -42,12 +54,57 @@ function Sidebar({ worktreeScrollOffsetRef, worktreeScrollAnchorRef }: SidebarProps): React.JSX.Element { + // Why: the memoized toolbar/search JSX below is localized, so it needs both a + // language subscription here and the locale as a memo dep to refresh on a switch. + const { i18n } = useTranslation() + const locale = i18n.resolvedLanguage ?? i18n.language + const sidebarTranslate = React.useCallback( + (key: string, fallback: string): string => translate(key, fallback, { lng: locale }), + [locale] + ) const sidebarOpen = useAppStore((s) => s.sidebarOpen) const sidebarWidth = useAppStore((s) => s.sidebarWidth) const setSidebarWidth = useAppStore((s) => s.setSidebarWidth) const repos = useAppStore((s) => s.repos) const startupWorktreeRefreshCompleted = useAppStore((s) => s.startupWorktreeRefreshCompleted) const settings = useAppStore((s) => s.settings) + const sidebarBody = useAppStore((s) => s.sidebarBody ?? 'workspaces') + const showAgentsSidebar = shouldShowAgentsSidebar(settings) + const showAgentDashboard = settings?.experimentalAgentDashboardPopout === true + const agentDashboardDrawerOpen = useAppStore((s) => s.agentDashboardDrawerOpen) + const setAgentDashboardDrawerOpen = useAppStore((s) => s.setAgentDashboardDrawerOpen) + const [agentReadFilter, setAgentReadFilter] = React.useState('all') + const [agentGroupBy, setAgentGroupBy] = React.useState('status') + const [agentQuery, setAgentQuery] = React.useState('') + const [agentSearchOpen, setAgentSearchOpen] = React.useState(false) + // Why clear on close: the hidden input's query would keep filtering the list with no visible indicator. + const closeAgentSearch = React.useCallback(() => { + setAgentSearchOpen(false) + setAgentQuery('') + }, []) + const [agentOptionsTarget, setAgentOptionsTarget] = React.useState(null) + const agentsScrollTopRef = React.useRef(0) + // Held here so collapsed groups (and the layout the saved scrollTop assumes) + // survive the Agents list unmounting on sidebar body switches. + const [agentsCollapsedGroupKeys, setAgentsCollapsedGroupKeys] = React.useState< + ReadonlySet + >(() => new Set()) + const agentsCollapseState = useMemo( + () => ({ + collapsedGroupKeys: agentsCollapsedGroupKeys, + onToggleGroupCollapse: (groupKey: string) => + setAgentsCollapsedGroupKeys((prev) => { + const next = new Set(prev) + if (next.has(groupKey)) { + next.delete(groupKey) + } else { + next.add(groupKey) + } + return next + }) + }), + [agentsCollapsedGroupKeys] + ) const fetchAllWorktrees = useAppStore((s) => s.fetchAllWorktrees) const activeModal = useAppStore((s) => s.activeModal) const statusBarVisible = useAppStore((s) => s.statusBarVisible) @@ -92,6 +149,12 @@ function Sidebar({ } }, [closeWorkspaceBoard, sidebarOpen, workspaceBoardRenderedOpen]) + useEffect(() => { + if (!showAgentDashboard && agentDashboardDrawerOpen) { + setAgentDashboardDrawerOpen(false) + } + }, [agentDashboardDrawerOpen, setAgentDashboardDrawerOpen, showAgentDashboard]) + const { containerRef, onResizeStart, isResizing } = useSidebarResize({ isOpen: sidebarOpen, width: sidebarWidth, @@ -102,6 +165,107 @@ function Sidebar({ onDraftWidthChange: setLiveSidebarWidth }) + // Why memoized: SidebarHeader is React.memo; fresh JSX here on every Sidebar render would + // defeat that memo and re-render the header subtree on unrelated store churn. + const agentToolbar = useMemo( + () => ( +
    + + + + + + {sidebarTranslate('auto.components.activity.ActivityPrototypePage.search', 'Search')} + + + + + + + + {sidebarTranslate( + 'auto.components.activity.ActivityPrototypePage.d1a88df9a8', + 'Show unread threads only' + )} + + +
    +
    + ), + [agentReadFilter, agentSearchOpen, closeAgentSearch, sidebarTranslate] + ) + const agentSearchRow = useMemo( + () => + agentSearchOpen ? ( +
    + setAgentQuery(event.target.value)} + onKeyDown={(event) => { + if (event.key === 'Escape') { + closeAgentSearch() + } + }} + placeholder={sidebarTranslate( + 'auto.components.activity.ActivityPrototypePage.795cbf26e2', + 'Filter...' + )} + className="h-7 w-full text-[11px]" + aria-label={sidebarTranslate( + 'auto.components.activity.ActivityPrototypePage.search', + 'Search' + )} + /> +
    + ) : null, + [agentQuery, agentSearchOpen, closeAgentSearch, sidebarTranslate] + ) + + useWorkspaceRevealBodyRedirect(sidebarOpen && sidebarBody === 'agents' && showAgentsSidebar) + return (
    {/* Fixed controls */} - - - + {sidebarBody === 'agents' && showAgentsSidebar ? ( + }> + + + + + ) : ( + + )}
    @@ -195,7 +380,7 @@ function Sidebar({ onMenuOpenChange={setWorkspaceBoardMenuOpen} /> ) : null} - {settings?.experimentalAgentDashboardPopout === true ? ( + {showAgentDashboard ? ( + {count > 9 ? '9+' : count} + + ) +} diff --git a/src/renderer/src/components/sidebar/sidebar-header-actions.tsx b/src/renderer/src/components/sidebar/sidebar-header-actions.tsx new file mode 100644 index 00000000000..fda4f3ebb8b --- /dev/null +++ b/src/renderer/src/components/sidebar/sidebar-header-actions.tsx @@ -0,0 +1,194 @@ +import React, { useCallback, useState } from 'react' +import { Ellipsis, FolderPlus, Plus } from 'lucide-react' +import { useAppStore } from '@/store' +import { Button } from '@/components/ui/button' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { useShortcutLabel } from '@/hooks/useShortcutLabel' +import { translate } from '@/i18n/i18n' +import { openWorkspaceCreationComposerWithTourHandoff } from '../contextual-tours/workspace-creation-tour-handoff' +import SidebarWorkspaceOptionsMenu from './SidebarWorkspaceOptionsMenu' +import { SidebarCountBadge } from './sidebar-count-badge' +import { + useWorkspaceOptionsFilterBadge, + WorkspaceOptionsMenuItems +} from './workspace-options-menu-items' + +export const SIDEBAR_HEADER_WIDE_MIN_WIDTH = 235 + +function AddProjectButton(): React.JSX.Element { + const openModal = useAppStore((s) => s.openModal) + return ( + + + + + + {translate('auto.components.sidebar.SidebarHeader.25a95899c9', 'Add Project')} + + + ) +} + +function CompactWorkspaceOverflow({ + preserveWorkspaceBoardOpen, + onMenuOpenChange +}: { + preserveWorkspaceBoardOpen: boolean + onMenuOpenChange?: (open: boolean) => void +}): React.JSX.Element { + const openModal = useAppStore((s) => s.openModal) + const [open, setOpen] = useState(false) + const { hasAnyFilter, activeFilterCount } = useWorkspaceOptionsFilterBadge() + const boardAttr = preserveWorkspaceBoardOpen ? '' : undefined + + const handleOpenChange = useCallback( + (next: boolean) => { + setOpen(next) + onMenuOpenChange?.(next) + }, + [onMenuOpenChange] + ) + + return ( + + + + + + + + + {translate('auto.components.sidebar.SidebarHeader.moreActions', 'More workspace actions')} + + + + + + openModal('add-repo')}> + + {translate('auto.components.sidebar.SidebarHeader.25a95899c9', 'Add Project')} + + + + ) +} + +export function SidebarHeaderActions({ + onWorkspaceBoardMenuOpenChange +}: { + onWorkspaceBoardMenuOpenChange: (open: boolean) => void +}): React.JSX.Element { + const sidebarWidth = useAppStore((s) => s.sidebarWidth) + const newWorktreeShortcutLabel = useShortcutLabel('workspace.create') + const compact = sidebarWidth < SIDEBAR_HEADER_WIDE_MIN_WIDTH + + if (compact) { + return ( +
    + + + + + + {translate( + 'auto.components.sidebar.SidebarHeader.ca6f729da2', + 'New workspace ({{value0}})', + { value0: newWorktreeShortcutLabel } + )} + + + +
    + ) + } + + return ( +
    + + + + + + + + + + {translate( + 'auto.components.sidebar.SidebarHeader.ca6f729da2', + 'New workspace ({{value0}})', + { value0: newWorktreeShortcutLabel } + )} + + +
    + ) +} diff --git a/src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx b/src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx new file mode 100644 index 00000000000..38893fb8f2d --- /dev/null +++ b/src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx @@ -0,0 +1,145 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { SidebarViewToggle } from './sidebar-view-toggle' + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +let container: HTMLDivElement +let root: Root + +beforeEach(() => { + container = document.createElement('div') + document.body.append(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() +}) + +describe('SidebarViewToggle', () => { + it('exposes radio semantics with a roving tabindex so arrow keys move between tabs', () => { + act(() => { + root.render( + undefined} + options={[ + { value: 'workspaces', label: 'Spaces', sectionTitle: 'projects' }, + { value: 'agents', label: 'Agents', sectionTitle: 'agents' } + ]} + /> + ) + }) + + const items = [...container.querySelectorAll('[role="radio"]')] + expect(items).toHaveLength(2) + const agents = container.querySelector('[data-sidebar-section-title="agents"]') + const projects = container.querySelector('[data-sidebar-section-title="projects"]') + expect(agents?.getAttribute('aria-checked')).toBe('true') + expect(projects?.getAttribute('aria-checked')).toBe('false') + // Only one tab is in the tab order (roving tabindex); arrow keys reach the other. + const tabStops = items.map((item) => item.getAttribute('tabindex')) + expect(tabStops.filter((stop) => stop === '0')).toHaveLength(1) + expect(tabStops.filter((stop) => stop === '-1')).toHaveLength(1) + }) + + it('never deselects when the active tab is clicked again', () => { + const onSelect = vi.fn() + act(() => { + root.render( + + ) + }) + const agents = container.querySelector( + '[data-sidebar-section-title="agents"]' + ) + act(() => agents?.click()) + expect(onSelect).not.toHaveBeenCalled() + const projects = container.querySelector( + '[data-sidebar-section-title="projects"]' + ) + act(() => projects?.click()) + expect(onSelect).toHaveBeenCalledWith('workspaces') + }) + + it('moves focus to the radio an arrow key selects', () => { + // Without the focus move, every later arrow press steps from the old index and + // keeps re-selecting the same neighbour. + const onSelect = vi.fn() + act(() => { + root.render( + + ) + }) + + const projects = container.querySelector( + '[data-sidebar-section-title="projects"]' + ) + act(() => { + projects?.focus() + projects?.dispatchEvent( + new KeyboardEvent('keydown', { key: 'ArrowRight', bubbles: true, cancelable: true }) + ) + }) + + expect(onSelect).toHaveBeenCalledWith('agents') + expect(document.activeElement).toBe( + container.querySelector('[data-sidebar-section-title="agents"]') + ) + }) + + it('keeps the visible label on one line', () => { + act(() => { + root.render( + undefined} + options={[ + { + value: 'workspaces', + label: 'Spaces', + sectionTitle: 'projects' + }, + { value: 'agents', label: 'Agents', sectionTitle: 'agents' } + ]} + /> + ) + }) + + const group = container.querySelector('[role="radiogroup"]') + const groupClasses = new Set(group?.className.split(/\s+/) ?? []) + expect(groupClasses.has('inline-flex')).toBe(true) + expect(groupClasses.has('shrink-0')).toBe(true) + expect(groupClasses.has('flex-1')).toBe(false) + + const spacesTab = container.querySelector('[data-sidebar-section-title="projects"]') + const visibleLabel = [...(spacesTab?.querySelectorAll('span') ?? [])].find( + (span) => span.getAttribute('aria-hidden') == null && span.textContent === 'Spaces' + ) + expect(visibleLabel?.className).toContain('whitespace-nowrap') + expect(visibleLabel?.className.includes('truncate')).toBe(false) + }) +}) diff --git a/src/renderer/src/components/sidebar/sidebar-view-toggle.tsx b/src/renderer/src/components/sidebar/sidebar-view-toggle.tsx new file mode 100644 index 00000000000..888f4f851c7 --- /dev/null +++ b/src/renderer/src/components/sidebar/sidebar-view-toggle.tsx @@ -0,0 +1,106 @@ +import React from 'react' +import { cn } from '@/lib/utils' + +type SidebarViewToggleOption = { + value: string + label: string + /** Every label this slot can ever show; reserves width so switching never resizes the tab. */ + widthLabels?: readonly string[] + sectionTitle?: string + renderWrapper?: (button: React.ReactNode) => React.ReactNode +} + +type SidebarViewToggleProps = { + ariaLabel: string + value: string + options: readonly SidebarViewToggleOption[] + onSelect: (value: string) => void + className?: string +} + +/** Two-up segmented control; tab widths stay frozen so nothing reflows on toggle. */ +export function SidebarViewToggle({ + ariaLabel, + value, + options, + onSelect, + className +}: SidebarViewToggleProps): React.JSX.Element { + const buttonRefs = React.useRef<(HTMLButtonElement | null)[]>([]) + // Arrow keys must carry focus to the newly checked radio, or every later press + // would still step from the old index and re-select the same neighbour. + const selectAndFocus = (index: number): void => { + const option = options[index] + if (!option) { + return + } + if (option.value !== value) { + onSelect(option.value) + } + buttonRefs.current[index]?.focus() + } + return ( +
    + {options.map((option, index) => { + const active = option.value === value + const button = ( + + ) + + return option.renderWrapper ? ( + {option.renderWrapper(button)} + ) : ( + button + ) + })} +
    + ) +} diff --git a/src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.test.tsx b/src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.test.tsx new file mode 100644 index 00000000000..2c230a8f98c --- /dev/null +++ b/src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.test.tsx @@ -0,0 +1,77 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT } from '@/lib/scroll-to-current-workspace-status' +import { useWorkspaceRevealBodyRedirect } from './use-workspace-reveal-body-redirect' + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +const mocks = vi.hoisted(() => ({ setSidebarBody: vi.fn() })) + +vi.mock('@/store', () => ({ + useAppStore: (selector: (state: { setSidebarBody: typeof mocks.setSidebarBody }) => unknown) => + selector({ setSidebarBody: mocks.setSidebarBody }) +})) + +function Host({ agentsBodyShowing }: { agentsBodyShowing: boolean }): null { + useWorkspaceRevealBodyRedirect(agentsBodyShowing) + return null +} + +let container: HTMLDivElement +let root: Root + +beforeEach(() => { + mocks.setSidebarBody.mockClear() + container = document.createElement('div') + document.body.append(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() +}) + +describe('useWorkspaceRevealBodyRedirect', () => { + it('switches the body to Spaces and replays the request once the list is mounted', () => { + act(() => { + root.render() + }) + const seen: unknown[] = [] + const listener = (event: Event): void => { + seen.push(event instanceof CustomEvent ? event.detail : null) + } + + act(() => { + window.dispatchEvent( + new CustomEvent(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, { + detail: { target: { type: 'active-workspace' }, beginRename: true } + }) + ) + }) + expect(mocks.setSidebarBody).toHaveBeenCalledWith('workspaces') + expect(seen).toEqual([]) + + // The worktree list mounts (and registers its listener) when the body flips. + window.addEventListener(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, listener) + act(() => { + root.render() + }) + window.removeEventListener(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, listener) + + expect(seen).toEqual([{ target: { type: 'active-workspace' }, beginRename: true }]) + }) + + it('does not intercept requests while Spaces is already showing', () => { + act(() => { + root.render() + }) + act(() => { + window.dispatchEvent(new CustomEvent(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT)) + }) + expect(mocks.setSidebarBody).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.ts b/src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.ts new file mode 100644 index 00000000000..7b2e889b859 --- /dev/null +++ b/src/renderer/src/components/sidebar/use-workspace-reveal-body-redirect.ts @@ -0,0 +1,43 @@ +import { useEffect, useRef } from 'react' +import { useAppStore } from '@/store' +import { SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT } from '@/lib/scroll-to-current-workspace-status' + +/** + * Reveal requests (rename shortcut, reveal-active-workspace button) are handled inside the + * worktree list, which is unmounted while the Agents body is showing. Capture the request, + * switch the body to Spaces, and replay it once the list's listener is registered. + */ +export function useWorkspaceRevealBodyRedirect(agentsBodyShowing: boolean): void { + const pendingDetailRef = useRef<{ detail: unknown } | null>(null) + const setSidebarBody = useAppStore((s) => s.setSidebarBody) + + useEffect(() => { + if (!agentsBodyShowing) { + return + } + const onRequest = (event: Event): void => { + pendingDetailRef.current = { detail: event instanceof CustomEvent ? event.detail : undefined } + setSidebarBody('workspaces') + } + window.addEventListener(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, onRequest) + return () => { + window.removeEventListener(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, onRequest) + } + }, [agentsBodyShowing, setSidebarBody]) + + useEffect(() => { + if (agentsBodyShowing) { + return + } + const pending = pendingDetailRef.current + if (!pending) { + return + } + pendingDetailRef.current = null + // Why safe to replay synchronously: the worktree list is a child of the sidebar, so its + // listener effect ran earlier in this same commit. + window.dispatchEvent( + new CustomEvent(SCROLL_TO_CURRENT_WORKSPACE_REVEAL_REQUEST_EVENT, { detail: pending.detail }) + ) + }, [agentsBodyShowing]) +} diff --git a/src/renderer/src/components/sidebar/visible-worktrees.ts b/src/renderer/src/components/sidebar/visible-worktrees.ts index 18e01106495..bb22f5f6086 100644 --- a/src/renderer/src/components/sidebar/visible-worktrees.ts +++ b/src/renderer/src/components/sidebar/visible-worktrees.ts @@ -265,6 +265,41 @@ export function setVisibleWorktreeShortcutTargets( * recomputes the order the sidebar *would* render from the same row pipeline, * so a closed sidebar numbers workspaces the same way an open one does (#9497). */ +export function buildVisibleWorktreeOptionsFromState( + state: ReturnType, + repoMap: Map +): VisibleWorktreeOptions { + return { + filterRepoIds: state.filterRepoIds, + showSleepingWorkspaces: state.showSleepingWorkspaces, + tabsByWorktree: state.tabsByWorktree, + ptyIdsByTabId: state.ptyIdsByTabId, + browserTabsByWorktree: state.browserTabsByWorktree, + worktreeIdsWithLiveAgent: getWorktreeIdsWithLiveAgent( + state.agentStatusByPaneKey, + state.tabsByWorktree, + Date.now() + ), + hideDefaultBranchWorkspace: state.hideDefaultBranchWorkspace, + hideAutomationGeneratedWorkspaces: state.hideAutomationGeneratedWorkspaces, + hideCliCreatedWorkspaces: state.hideCliCreatedWorkspaces, + hideDetachedHeadWorkspaces: state.hideDetachedHeadWorkspaces, + hideWorkspacesFromOtherDevices: state.hideWorkspacesFromOtherDevices, + pairedDeviceIdsByEnvironment: state.hideWorkspacesFromOtherDevices + ? getPairedDeviceIdsByEnvironment( + state.runtimeEnvironments, + state.runtimeStatusByEnvironmentId + ) + : EMPTY_PAIRED_DEVICE_IDS_BY_ENVIRONMENT, + alwaysShowDefaultBranchWorkspace: state.alwaysShowDefaultBranchWorkspace, + repoMap, + workspaceHostScope: state.workspaceHostScope, + visibleWorkspaceHostIds: state.visibleWorkspaceHostIds, + defaultHostId: getSettingsFocusedExecutionHostId(state.settings), + worktreeLineageById: state.worktreeLineageById + } +} + export function getVisibleWorktreeIds(): string[] { // Prefer the published IDs that mirror the rendered sidebar order. if (_publishedVisibleIds) { @@ -299,35 +334,11 @@ export function getVisibleWorktreeIds(): string[] { sortedIds = sorted.map((w) => w.id) } - const visibleIds = computeVisibleWorktreeIds(state.worktreesByRepo, sortedIds, { - filterRepoIds: state.filterRepoIds, - showSleepingWorkspaces: state.showSleepingWorkspaces, - tabsByWorktree: state.tabsByWorktree, - ptyIdsByTabId: state.ptyIdsByTabId, - browserTabsByWorktree: state.browserTabsByWorktree, - worktreeIdsWithLiveAgent: getWorktreeIdsWithLiveAgent( - state.agentStatusByPaneKey, - state.tabsByWorktree, - Date.now() - ), - hideDefaultBranchWorkspace: state.hideDefaultBranchWorkspace, - hideAutomationGeneratedWorkspaces: state.hideAutomationGeneratedWorkspaces, - hideCliCreatedWorkspaces: state.hideCliCreatedWorkspaces, - hideDetachedHeadWorkspaces: state.hideDetachedHeadWorkspaces, - hideWorkspacesFromOtherDevices: state.hideWorkspacesFromOtherDevices, - pairedDeviceIdsByEnvironment: state.hideWorkspacesFromOtherDevices - ? getPairedDeviceIdsByEnvironment( - state.runtimeEnvironments, - state.runtimeStatusByEnvironmentId - ) - : EMPTY_PAIRED_DEVICE_IDS_BY_ENVIRONMENT, - alwaysShowDefaultBranchWorkspace: state.alwaysShowDefaultBranchWorkspace, - repoMap, - workspaceHostScope: state.workspaceHostScope, - visibleWorkspaceHostIds: state.visibleWorkspaceHostIds, - defaultHostId: getSettingsFocusedExecutionHostId(state.settings), - worktreeLineageById: state.worktreeLineageById - }) + const visibleIds = computeVisibleWorktreeIds( + state.worktreesByRepo, + sortedIds, + buildVisibleWorktreeOptionsFromState(state, repoMap) + ) const visibleIdRank = new Map(visibleIds.map((id, index) => [id, index])) const visibleHostIds = getVisibleWorkspaceHostIdSet(state) diff --git a/src/renderer/src/components/sidebar/workspace-options-menu-items.tsx b/src/renderer/src/components/sidebar/workspace-options-menu-items.tsx new file mode 100644 index 00000000000..c4318449bee --- /dev/null +++ b/src/renderer/src/components/sidebar/workspace-options-menu-items.tsx @@ -0,0 +1,240 @@ +import { useMemo, type JSX } from 'react' +import { useAppStore } from '@/store' +import { + DropdownMenuLabel, + DropdownMenuRadioGroup, + DropdownMenuRadioItem, + DropdownMenuSeparator, + DropdownMenuSub, + DropdownMenuSubContent, + DropdownMenuSubTrigger +} from '@/components/ui/dropdown-menu' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { DEFAULT_SHOW_SLEEPING_WORKSPACES } from '../../../../shared/constants' +import { isSleepingSweepExemptionNarrowingList } from './visible-worktrees' +import SidebarRepositoryFilterSection from './SidebarRepositoryFilterSection' +import SidebarWorkspaceFilterSection from './SidebarWorkspaceFilterSection' +import { getSidebarHostVisibilityLabel, shouldShowHostScopeControls } from './sidebar-host-options' +import { useSidebarHostScopeOptions } from './use-sidebar-host-scope-options' +import { SidebarHostScopeMenuSection } from './SidebarHostScopeMenuSection' +import { PROJECT_ORDER_OPTIONS, SORT_OPTIONS } from './sidebar-workspace-option-items' +import { WorktreeCardDisplayMenuSection } from './WorktreeCardDisplayMenuSection' +import { translate } from '@/i18n/i18n' +import { SidebarGroupByToggle } from './SidebarGroupByToggle' + +export function useWorkspaceOptionsFilterBadge(): { + hasAnyFilter: boolean + activeFilterCount: number + activeFilterLabel: string +} { + const showSleepingWorkspaces = useAppStore((s) => s.showSleepingWorkspaces) + const hideDefaultBranchWorkspace = useAppStore((s) => s.hideDefaultBranchWorkspace) + const hideAutomationGeneratedWorkspaces = useAppStore((s) => s.hideAutomationGeneratedWorkspaces) + const hideCliCreatedWorkspaces = useAppStore((s) => s.hideCliCreatedWorkspaces) + const hideDetachedHeadWorkspaces = useAppStore((s) => s.hideDetachedHeadWorkspaces) + const hideWorkspacesFromOtherDevices = useAppStore((s) => s.hideWorkspacesFromOtherDevices) + const alwaysShowDefaultBranchWorkspace = useAppStore((s) => s.alwaysShowDefaultBranchWorkspace) + const filterRepoIds = useAppStore((s) => s.filterRepoIds) + const repos = useAppStore((s) => s.repos) + const visibleWorkspaceHostIds = useAppStore((s) => s.visibleWorkspaceHostIds) + + const selectedCount = useMemo(() => { + let count = 0 + for (const repo of repos) { + if (filterRepoIds.includes(repo.id)) { + count += 1 + } + } + return count + }, [repos, filterRepoIds]) + + const hasSleepingFilter = showSleepingWorkspaces !== DEFAULT_SHOW_SLEEPING_WORKSPACES + const hasSleepingExemptionFilter = isSleepingSweepExemptionNarrowingList( + showSleepingWorkspaces, + alwaysShowDefaultBranchWorkspace + ) + const hasRepoFilter = selectedCount > 0 + const hasHostVisibilityFilter = visibleWorkspaceHostIds !== null + const hasAnyFilter = + hasSleepingFilter || + hideDefaultBranchWorkspace || + hideAutomationGeneratedWorkspaces || + hideCliCreatedWorkspaces || + hideDetachedHeadWorkspaces || + hideWorkspacesFromOtherDevices || + hasSleepingExemptionFilter || + hasRepoFilter || + hasHostVisibilityFilter + const activeFilterCount = + (hasSleepingFilter ? 1 : 0) + + (hideDefaultBranchWorkspace ? 1 : 0) + + (hideAutomationGeneratedWorkspaces ? 1 : 0) + + (hideCliCreatedWorkspaces ? 1 : 0) + + (hideDetachedHeadWorkspaces ? 1 : 0) + + (hideWorkspacesFromOtherDevices ? 1 : 0) + + (hasSleepingExemptionFilter ? 1 : 0) + + (hasHostVisibilityFilter ? 1 : 0) + + selectedCount + + return { + hasAnyFilter, + activeFilterCount, + activeFilterLabel: `${activeFilterCount} ${activeFilterCount === 1 ? 'filter' : 'filters'}` + } +} + +export function WorkspaceOptionsMenuItems({ + preserveWorkspaceBoardOpen = false +}: { + preserveWorkspaceBoardOpen?: boolean +}): JSX.Element { + const repos = useAppStore((s) => s.repos) + const setWorkspaceHostScope = useAppStore((s) => s.setWorkspaceHostScope) + const visibleWorkspaceHostIds = useAppStore((s) => s.visibleWorkspaceHostIds) + const setVisibleWorkspaceHostIds = useAppStore((s) => s.setVisibleWorkspaceHostIds) + const sortBy = useAppStore((s) => s.sortBy) + const setSortBy = useAppStore((s) => s.setSortBy) + const groupBy = useAppStore((s) => s.groupBy) + const setGroupBy = useAppStore((s) => s.setGroupBy) + const projectOrderBy = useAppStore((s) => s.projectOrderBy) + const setProjectOrderBy = useAppStore((s) => s.setProjectOrderBy) + const { hostOptions } = useSidebarHostScopeOptions() + const showHostScopeControls = shouldShowHostScopeControls(hostOptions) + const sortLabel = SORT_OPTIONS.find((opt) => opt.id === sortBy)?.label ?? 'Sort' + const projectOrderLabel = + PROJECT_ORDER_OPTIONS.find((opt) => opt.id === projectOrderBy)?.label ?? 'Manual' + const hostVisibilityLabel = getSidebarHostVisibilityLabel(visibleWorkspaceHostIds, hostOptions) + const boardAttr = preserveWorkspaceBoardOpen ? '' : undefined + + return ( + <> + + {translate( + 'auto.components.sidebar.SidebarWorkspaceOptionsMenu.workspaceOptions', + 'Workspace options' + )} + + {/* Why: host + project filters share one section and the same single-row + shell as Sort by (label left, value right) so the menu stays flat. */} + {(showHostScopeControls || repos.length > 1) && ( + <> + + {translate('auto.components.sidebar.SidebarWorkspaceOptionsMenu.showSection', 'Show')} + + {showHostScopeControls && ( + + )} + + + + )} + + + {translate('auto.components.sidebar.SidebarWorkspaceOptionsMenu.dc0bb670bc', 'Group by')} + +
    + +
    + + + + + + + {translate( + 'auto.components.sidebar.SidebarWorkspaceOptionsMenu.7bada3b1ab', + 'Sort by' + )} + + {sortLabel} + + + + setSortBy(v as typeof sortBy)} + > + {SORT_OPTIONS.map((opt) => { + const radioItem = ( + e.preventDefault()} + > + {opt.label} + + ) + if (!opt.description) { + return radioItem + } + return ( + + {radioItem} + + {opt.description} + + + ) + })} + + + + + {/* Why: project order only has a visible effect when grouping by + project; hide it in none/status/PR modes to avoid a dead control. */} + {groupBy === 'repo' && ( + + + + + {translate( + 'auto.components.sidebar.SidebarWorkspaceOptionsMenu.09faabd875', + 'Project order' + )} + + + {projectOrderLabel} + + + + + setProjectOrderBy(v as typeof projectOrderBy)} + > + {PROJECT_ORDER_OPTIONS.map((opt) => ( + + + e.preventDefault()} + > + {opt.label} + + + + {opt.description} + + + ))} + + + + )} + + + + + + ) +} diff --git a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts index 4947fce433f..84588c4296c 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts @@ -37,6 +37,10 @@ type TabWorktreeIndexCache = { tabIdToWorktreeId: Map } +type LiveTabWorktreeIndexCache = TabWorktreeIndexCache & { + unifiedTabsByWorktree: WorktreeAgentRowsState['unifiedTabsByWorktree'] +} + type MigrationUnsupportedByWorktreeCache = { tabsByWorktree: WorktreeAgentRowsState['tabsByWorktree'] migrationUnsupportedByPtyId: WorktreeAgentRowsState['migrationUnsupportedByPtyId'] @@ -49,6 +53,7 @@ type RetainedEntriesByWorktreeCache = { } let tabWorktreeIndexCache: TabWorktreeIndexCache | null = null +let liveTabWorktreeIndexCache: LiveTabWorktreeIndexCache | null = null let liveEntriesByWorktreeCache: LiveEntriesByWorktreeCache | null = null let migrationUnsupportedByWorktreeCache: MigrationUnsupportedByWorktreeCache | null = null let retainedEntriesByWorktreeCache: RetainedEntriesByWorktreeCache | null = null @@ -88,6 +93,12 @@ function getLiveTabIdToWorktreeId( tabsByWorktree: WorktreeAgentRowsState['tabsByWorktree'], unifiedTabsByWorktree: WorktreeAgentRowsState['unifiedTabsByWorktree'] ): Map { + if ( + liveTabWorktreeIndexCache?.tabsByWorktree === tabsByWorktree && + liveTabWorktreeIndexCache.unifiedTabsByWorktree === unifiedTabsByWorktree + ) { + return liveTabWorktreeIndexCache.tabIdToWorktreeId + } const tabIdToWorktreeId = new Map(getTabIdToWorktreeId(tabsByWorktree)) for (const [worktreeId, tabs] of Object.entries(unifiedTabsByWorktree ?? {})) { for (const tab of tabs) { @@ -96,6 +107,7 @@ function getLiveTabIdToWorktreeId( } } } + liveTabWorktreeIndexCache = { tabsByWorktree, unifiedTabsByWorktree, tabIdToWorktreeId } return tabIdToWorktreeId } diff --git a/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.stable-message.test.tsx b/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.stable-message.test.tsx new file mode 100644 index 00000000000..c9510cfef4f --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.stable-message.test.tsx @@ -0,0 +1,112 @@ +/** @vitest-environment happy-dom */ +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { DashboardAgentRow as DashboardAgentRowData } from '@/components/dashboard/useDashboardData' +import { TooltipProvider } from '@/components/ui/tooltip' +import { CompactAgentRow } from './worktree-card-compact-agent-row' + +;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + +vi.mock('@/components/dashboard/use-agent-row-conversation-name', () => ({ + useAgentRowConversationName: () => null +})) + +vi.mock('./CacheTimer', () => ({ + default: () => null, + usePromptCacheCountdownForPane: () => null +})) + +function makeAgent({ + stateStartedAt, + lastAssistantMessage, + state = 'working' +}: { + stateStartedAt: number + lastAssistantMessage?: string + state?: string +}): DashboardAgentRowData { + return { + paneKey: 'tab-1:leaf-1', + tab: { id: 'tab-1' }, + agentType: 'claude', + state, + startedAt: 500, + entry: { + prompt: 'do the task', + state, + stateStartedAt, + lastAssistantMessage, + paneKey: 'tab-1:leaf-1', + updatedAt: stateStartedAt + } + } as unknown as DashboardAgentRowData +} + +let root: Root | undefined + +afterEach(() => { + act(() => root?.unmount()) + document.body.replaceChildren() +}) + +function renderRow(agent: DashboardAgentRowData): HTMLElement { + const container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + act(() => { + root!.render( + + {}} /> + + ) + }) + return container +} + +function rerenderRow(agent: DashboardAgentRowData): void { + act(() => { + root!.render( + + {}} /> + + ) + }) +} + +describe('CompactAgentRow stable assistant message', () => { + it('holds the last assistant line when a same-turn ping omits it', () => { + const container = renderRow( + makeAgent({ stateStartedAt: 1000, lastAssistantMessage: 'First reply' }) + ) + expect(container.textContent).toContain('First reply') + + rerenderRow(makeAgent({ stateStartedAt: 1000 })) + expect(container.textContent).toContain('First reply') + }) + + it('drops the held line when a new turn starts', () => { + const container = renderRow( + makeAgent({ stateStartedAt: 1000, lastAssistantMessage: 'First reply' }) + ) + rerenderRow(makeAgent({ stateStartedAt: 3000 })) + expect(container.textContent).not.toContain('First reply') + }) + + it('never holds across pings for entries without a turn identity (stateStartedAt 0)', () => { + const container = renderRow(makeAgent({ stateStartedAt: 0, lastAssistantMessage: 'Turn one' })) + expect(container.textContent).toContain('Turn one') + + rerenderRow(makeAgent({ stateStartedAt: 0 })) + expect(container.textContent).not.toContain('Turn one') + }) + + it('drops the held line when the agent leaves working', () => { + const container = renderRow( + makeAgent({ stateStartedAt: 1000, lastAssistantMessage: 'First reply' }) + ) + rerenderRow(makeAgent({ stateStartedAt: 1000, state: 'done' })) + rerenderRow(makeAgent({ stateStartedAt: 1000, state: 'working' })) + expect(container.textContent).not.toContain('First reply') + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.tsx b/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.tsx index f66b54adeb2..b698982a7dd 100644 --- a/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.tsx +++ b/src/renderer/src/components/sidebar/worktree-card-compact-agent-row.tsx @@ -1,4 +1,4 @@ -import React, { useCallback } from 'react' +import React, { useCallback, useEffect, useRef } from 'react' import { ChevronRight } from 'lucide-react' import { AgentStateDot, agentStateLabel } from '@/components/AgentStateDot' import type { DashboardAgentRow as DashboardAgentRowData } from '@/components/dashboard/useDashboardData' @@ -13,22 +13,7 @@ import { agentNoUpdateLabel } from '@/lib/agent-row-decay-state' import { useAgentRowConversationName } from '@/components/dashboard/use-agent-row-conversation-name' import { lastEnteredDoneAt } from '@/components/dashboard/agent-finished-timestamp' import CacheTimer, { usePromptCacheCountdownForPane } from './CacheTimer' - -function formatShortTimeAgo(ts: number, now: number): string { - const delta = now - ts - if (delta < 60_000) { - return 'now' - } - const minutes = Math.floor(delta / 60_000) - if (minutes < 60) { - return `${minutes}m` - } - const hours = Math.floor(minutes / 60) - if (hours < 24) { - return `${hours}h` - } - return `${Math.floor(hours / 24)}d` -} +import { formatShortTimeAgo } from '@/lib/short-time-ago' function getCompactAgentPrimary( agent: DashboardAgentRowData, @@ -38,7 +23,11 @@ function getCompactAgentPrimary( return prompt || agentStateLabel(getAgentDotState(agent)) } -export function getCompactAgentSecondary(agent: DashboardAgentRowData, now: number): string { +export function getCompactAgentSecondary( + agent: DashboardAgentRowData, + now: number, + lastAssistantMessageOverride?: string +): string { if (agent.entry.interrupted === true) { return 'Interrupted by user' } @@ -55,7 +44,8 @@ export function getCompactAgentSecondary(agent: DashboardAgentRowData, now: numb if (toolPreview) { return toolPreview } - const lastAssistantMessage = agent.entry.lastAssistantMessage?.trim() + const lastAssistantMessage = + lastAssistantMessageOverride ?? agent.entry.lastAssistantMessage?.trim() if (lastAssistantMessage) { return lastAssistantMessage } @@ -128,7 +118,26 @@ export const CompactAgentRow = React.memo(function CompactAgentRow({ const conversationName = useAgentRowConversationName(agent) const primary = getCompactAgentPrimary(agent, conversationName) const isLineageChild = agent.lineage?.depth === 1 - const secondary = getCompactAgentSecondary(agent, now) + // Keep a live row's last assistant line stable while status/tool payloads + // briefly omit the hook-only field between updates. Committed in an effect so a + // discarded concurrent render can't pin an uncommitted message and no extra render + // pass runs per streaming ping; a zero stateStartedAt has no per-turn identity, so + // those rows never cache. + const turn = agent.entry.stateStartedAt + const currentMessage = agent.entry.lastAssistantMessage?.trim() ?? '' + const turnHoldable = agent.state === 'working' && turn > 0 + const heldMessageRef = useRef<{ turn: number; message: string } | null>(null) + useEffect(() => { + if (turnHoldable && currentMessage) { + heldMessageRef.current = { turn, message: currentMessage } + } else if (!turnHoldable) { + heldMessageRef.current = null + } + }, [turnHoldable, turn, currentMessage]) + const held = heldMessageRef.current + const stableMessage = + turnHoldable && !currentMessage && held?.turn === turn ? held.message : undefined + const secondary = getCompactAgentSecondary(agent, now, stableMessage) // Why: sidebar truncation must preserve the passive-vs-active distinction. const leadingText = dotState === 'monitoring' ? secondary : primary const trailingText = diff --git a/src/renderer/src/components/sidebar/worktree-filter-visibility.test.ts b/src/renderer/src/components/sidebar/worktree-filter-visibility.test.ts new file mode 100644 index 00000000000..0bb2283843b --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-filter-visibility.test.ts @@ -0,0 +1,79 @@ +import { describe, expect, it, vi } from 'vitest' +import type { Worktree } from '../../../../shared/worktree/types' + +const mocks = vi.hoisted(() => ({ getState: vi.fn() })) +vi.mock('@/store', () => ({ useAppStore: { getState: mocks.getState } })) + +import { worktreePassesSidebarFilters } from './worktree-filter-visibility' + +const TWIN_ID = 'repo-1::/projects/app' + +function makeTwin(hostId?: string): Worktree { + return { + id: TWIN_ID, + repoId: 'repo-1', + path: '/projects/app', + head: 'abc', + branch: 'main', + isBare: false, + isMainWorktree: false, + displayName: 'app', + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 1, + ...(hostId ? { hostId } : {}) + } as Worktree +} + +// STA-4343: the same worktree id names one workspace per host; only the local +// twin passes a local-only host scope. +function stateWithLocalScopedTwins(): unknown { + return { + repos: [ + { id: 'repo-1', path: '/projects/app', displayName: 'app', badgeColor: '', addedAt: 1 } + ], + worktreesByRepo: { 'repo-1': [makeTwin(), makeTwin('ssh:beta')] }, + filterRepoIds: [], + showSleepingWorkspaces: true, + tabsByWorktree: {}, + ptyIdsByTabId: {}, + browserTabsByWorktree: {}, + agentStatusByPaneKey: {}, + hideDefaultBranchWorkspace: false, + hideAutomationGeneratedWorkspaces: false, + hideCliCreatedWorkspaces: false, + hideDetachedHeadWorkspaces: false, + hideWorkspacesFromOtherDevices: false, + alwaysShowDefaultBranchWorkspace: true, + workspaceHostScope: 'local', + visibleWorkspaceHostIds: ['local'], + settings: null, + worktreeLineageById: {} + } +} + +describe('worktreePassesSidebarFilters', () => { + it('does not let a visible local twin vouch for a host-filtered remote target', () => { + mocks.getState.mockReturnValue(stateWithLocalScopedTwins()) + + expect(worktreePassesSidebarFilters(TWIN_ID, 'ssh:beta')).toBe(false) + expect(worktreePassesSidebarFilters(TWIN_ID, 'local')).toBe(true) + // Host unknown to the caller: id-only match keeps prior behavior. + expect(worktreePassesSidebarFilters(TWIN_ID)).toBe(true) + }) + + it('reports the remote twin visible when its host is in scope', () => { + const state = stateWithLocalScopedTwins() as { visibleWorkspaceHostIds: string[] } + state.visibleWorkspaceHostIds = ['ssh:beta'] + mocks.getState.mockReturnValue(state) + + expect(worktreePassesSidebarFilters(TWIN_ID, 'ssh:beta')).toBe(true) + expect(worktreePassesSidebarFilters(TWIN_ID, 'local')).toBe(false) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-filter-visibility.ts b/src/renderer/src/components/sidebar/worktree-filter-visibility.ts new file mode 100644 index 00000000000..321111f2392 --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-filter-visibility.ts @@ -0,0 +1,43 @@ +import { useAppStore } from '@/store' +import { getRepoMapFromState } from '@/store/selectors' +import { + getSettingsFocusedExecutionHostId, + getWorktreeExecutionHostId, + normalizeExecutionHostId, + type ExecutionHostId +} from '../../../../shared/execution-host' +import { buildVisibleWorktreeOptionsFromState, computeVisibleWorktrees } from './visible-worktrees' + +/** + * Filter-only visibility for one worktree id: runs the sidebar filter pipeline + * without collapse elision or rendered order, so a target inside a collapsed + * group is not misreported as hidden by filters. + * + * Worktree ids are not host-qualified (STA-4343): the same id can name a + * workspace on two hosts, so an id-only match would let a host-filtered remote + * target pass on the strength of its visible local twin. When the caller knows + * the target's host, the visible twin must resolve to that host too. + */ +export function worktreePassesSidebarFilters( + worktreeId: string, + executionHostId?: ExecutionHostId +): boolean { + const state = useAppStore.getState() + const repoMap = getRepoMapFromState(state) + const requestedHostId = executionHostId ? normalizeExecutionHostId(executionHostId) : null + const defaultHostId = getSettingsFocusedExecutionHostId(state.settings) + return computeVisibleWorktrees( + state.worktreesByRepo, + [], + buildVisibleWorktreeOptionsFromState(state, repoMap) + ).some((worktree) => { + if (worktree.id !== worktreeId) { + return false + } + if (!requestedHostId) { + return true + } + const hostId = getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId) + return normalizeExecutionHostId(hostId) === requestedHostId + }) +} diff --git a/src/renderer/src/components/terminal-pane/stale-agent-row.ts b/src/renderer/src/components/terminal-pane/stale-agent-row.ts index c08f176691e..b781f76940d 100644 --- a/src/renderer/src/components/terminal-pane/stale-agent-row.ts +++ b/src/renderer/src/components/terminal-pane/stale-agent-row.ts @@ -7,7 +7,7 @@ export function dismissStaleAgentRowByKey(paneKey: string): void { const store = useAppStore.getState() const liveExisted = paneKey in store.agentStatusByPaneKey const retainedExisted = paneKey in store.retainedAgentsByPaneKey - store.dropAgentStatus(paneKey) + store.dropAgentStatus(paneKey, { paneRemoved: true }) store.dismissRetainedAgent(paneKey) if (liveExisted || retainedExisted) { toast.info( diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-close-actions.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-close-actions.ts index 83f2e565c92..886c19fdf0d 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-pane-close-actions.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-close-actions.ts @@ -47,7 +47,7 @@ export function useTerminalPaneCloseActions(controller: TerminalPaneBindingContr const leafId = manager.getLeafId(paneId) if (leafId) { useAppStore.getState().setCacheTimerStartedAt(makePaneKey(tabId, leafId), null) - useAppStore.getState().dropAgentStatus(makePaneKey(tabId, leafId)) + useAppStore.getState().dropAgentStatus(makePaneKey(tabId, leafId), { paneRemoved: true }) } setTerminalErrorsByPaneId((current) => clearPaneTerminalError(current, paneId)) if (leafId) { diff --git a/src/renderer/src/components/ui/dropdown-menu.tsx b/src/renderer/src/components/ui/dropdown-menu.tsx index 193c5324fd3..b0e1d67a77c 100644 --- a/src/renderer/src/components/ui/dropdown-menu.tsx +++ b/src/renderer/src/components/ui/dropdown-menu.tsx @@ -86,7 +86,7 @@ function DropdownMenuCheckboxItem({ > - + {children} diff --git a/src/renderer/src/components/ui/popover.tsx b/src/renderer/src/components/ui/popover.tsx index 01365ba665b..ffd2c221e85 100644 --- a/src/renderer/src/components/ui/popover.tsx +++ b/src/renderer/src/components/ui/popover.tsx @@ -189,4 +189,19 @@ function PopoverContent({ ) } -export { Popover, PopoverAnchor, PopoverContent, PopoverTrigger } +function PopoverArrow({ + className, + style, + ...props +}: React.ComponentProps) { + return ( + + ) +} + +export { Popover, PopoverAnchor, PopoverArrow, PopoverContent, PopoverTrigger } diff --git a/src/renderer/src/hooks/useAutoAckViewedAgent.clock-skew.test.ts b/src/renderer/src/hooks/useAutoAckViewedAgent.clock-skew.test.ts index d81c0c49ffa..2c69593563e 100644 --- a/src/renderer/src/hooks/useAutoAckViewedAgent.clock-skew.test.ts +++ b/src/renderer/src/hooks/useAutoAckViewedAgent.clock-skew.test.ts @@ -109,4 +109,14 @@ describe('useAutoAckViewedAgent — clock-skewed execution host', () => { expect(calls).toEqual([[PANE_KEY]]) }) + + it('keeps an explicitly marked-unread visible turn unread', () => { + seedFutureStampedTurn(NOW - 5_000) + useAppStore.getState().acknowledgeAgents([PANE_KEY]) + + renderHook(() => useAutoAckViewedAgent(false)) + useAppStore.getState().unacknowledgeAgents([PANE_KEY]) + + expect(useAppStore.getState().acknowledgedAgentsByPaneKey[PANE_KEY]).toBeUndefined() + }) }) diff --git a/src/renderer/src/hooks/useAutoAckViewedAgent.test.ts b/src/renderer/src/hooks/useAutoAckViewedAgent.test.ts index d461dd13b90..370205299d6 100644 --- a/src/renderer/src/hooks/useAutoAckViewedAgent.test.ts +++ b/src/renderer/src/hooks/useAutoAckViewedAgent.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { acknowledgeViewedAgentAttention, computeAutoAckTargets, + computeLapsedManualUnreadProtections, computeViewedAgentCompletionPaneKey, resolveAutoAckTabTargets, shouldClearViewedAgentWorktreeUnread @@ -503,3 +504,47 @@ describe('floating workspace auto-ack against the attention dot', () => { expect(selectFloatingWorkspaceHasUnread(store.getState())).toBe(false) }) }) + +describe('computeLapsedManualUnreadProtections', () => { + const paneKey = makePaneKey('tab-1', CODEX_LEAF_ID) + const otherPaneKey = makePaneKey('tab-1', OTHER_LEAF_ID) + + it('keeps an active pane whose status row has not arrived yet (startup race)', () => { + // Persisted UI (manual-unread stamps) hydrates before the agent-status snapshot; a focused + // scan in that window must not treat "no row yet" as "the agent moved on". + const lapsed = computeLapsedManualUnreadProtections( + { + agentStatusByPaneKey: {}, + retainedAgentsByPaneKey: {}, + manuallyUnreadTurnsByPaneKey: { [paneKey]: 1_000 } + }, + new Set([paneKey]) + ) + expect(lapsed).toEqual([]) + }) + + it('lapses a pane that is no longer active or whose agent took a new turn', () => { + const store = createTestStore() + store.getState().setAgentStatus(paneKey, { state: 'done', prompt: 'p', agentType: 'claude' }) + const turn = store.getState().agentStatusByPaneKey[paneKey]!.stateStartedAt + const lapsed = computeLapsedManualUnreadProtections( + { + ...store.getState(), + manuallyUnreadTurnsByPaneKey: { [paneKey]: turn - 1, [otherPaneKey]: 5 } + }, + new Set([paneKey]) + ) + expect(lapsed.sort()).toEqual([paneKey, otherPaneKey].sort()) + }) + + it('keeps an active pane whose turn is unchanged', () => { + const store = createTestStore() + store.getState().setAgentStatus(paneKey, { state: 'done', prompt: 'p', agentType: 'claude' }) + const turn = store.getState().agentStatusByPaneKey[paneKey]!.stateStartedAt + const lapsed = computeLapsedManualUnreadProtections( + { ...store.getState(), manuallyUnreadTurnsByPaneKey: { [paneKey]: turn } }, + new Set([paneKey]) + ) + expect(lapsed).toEqual([]) + }) +}) diff --git a/src/renderer/src/hooks/useAutoAckViewedAgent.ts b/src/renderer/src/hooks/useAutoAckViewedAgent.ts index 67f7be66742..40a3591e2b0 100644 --- a/src/renderer/src/hooks/useAutoAckViewedAgent.ts +++ b/src/renderer/src/hooks/useAutoAckViewedAgent.ts @@ -67,6 +67,20 @@ export function computeViewedAgentCompletionPaneKey( return state.unreadAgentCompletionPanes[targetKey] ? targetKey : null } +function getAgentTurnTimestamp( + state: { + agentStatusByPaneKey: Record + retainedAgentsByPaneKey: Record + }, + paneKey: string +): number | null { + return ( + state.agentStatusByPaneKey[paneKey]?.stateStartedAt ?? + state.retainedAgentsByPaneKey[paneKey]?.entry.stateStartedAt ?? + null + ) +} + export function shouldClearViewedAgentWorktreeUnread( state: { tabsByWorktree: Record @@ -108,6 +122,35 @@ export function shouldClearViewedAgentWorktreeUnread( return true } +/** + * Manual mark-unread protections that no longer apply: the user moved to another pane, or the + * agent took a new turn. Exported for the startup-race test. + */ +export function computeLapsedManualUnreadProtections( + state: { + agentStatusByPaneKey: Record + retainedAgentsByPaneKey: Record + manuallyUnreadTurnsByPaneKey: Record + }, + activePaneKeys: ReadonlySet +): string[] { + const lapsed: string[] = [] + for (const [paneKey, turnTimestamp] of Object.entries(state.manuallyUnreadTurnsByPaneKey)) { + if (!activePaneKeys.has(paneKey)) { + lapsed.push(paneKey) + continue + } + const currentTurn = getAgentTurnTimestamp(state, paneKey) + // Why keep on null: persisted UI hydrates before the status snapshot lands, so an active + // pane with no row yet is "not known", not "moved on"; wiping it would lose the mark-unread + // the user made before relaunch. + if (currentTurn !== null && currentTurn !== turnTimestamp) { + lapsed.push(paneKey) + } + } + return lapsed +} + type ViewedAgentAttentionActions = { acknowledgeAgents: (paneKeys: string[]) => void clearWorktreeUnread: (worktreeId: string) => void @@ -230,6 +273,9 @@ export function useAutoAckViewedAgent(floatingPanelVisible: boolean): void { const targets = resolveAutoAckTabTargets(s, { floatingPanelVisible: floatingPanelVisibleRef.current }) + // Why no protection reset here: zero targets just means nothing is on screen + // (Settings, browser, an overlay) — a transient view switch must not lapse an + // explicit mark-unread the user just made. if (targets.length === 0) { return } @@ -243,13 +289,31 @@ export function useAutoAckViewedAgent(floatingPanelVisible: boolean): void { lastLayouts = s.terminalLayoutsByTabId lastUnreadAgentCompletionPanes = s.unreadAgentCompletionPanes + const activePaneKeys = new Set() + for (const target of targets) { + const activeLeafId = resolveActiveLeafId(s, target.tabId) + if (activeLeafId) { + activePaneKeys.add(makePaneKey(target.tabId, activeLeafId)) + } + } + // Protection lapses when the user moves on to another pane or the agent takes a new + // turn; a still-active pane with an unchanged turn keeps its explicit mark-unread. + const lapsedProtections = computeLapsedManualUnreadProtections(s, activePaneKeys) + if (lapsedProtections.length > 0) { + s.clearManuallyUnreadTurns(lapsedProtections) + } + for (const target of targets) { // Why re-read: acking target[0] writes to the store, which re-enters this scan synchronously // and may already have handled target[1]; `s` is a pre-write snapshot that would re-ack it. const current = useAppStore.getState() const tabId = target.tabId const activeLeafId = resolveActiveLeafId(current, tabId) - const toAck = computeAutoAckTargets(current, tabId, activeLeafId) + const toAck = computeAutoAckTargets(current, tabId, activeLeafId).filter( + (paneKey) => + current.manuallyUnreadTurnsByPaneKey[paneKey] !== + getAgentTurnTimestamp(current, paneKey) + ) const activePaneKey = computeViewedAgentCompletionPaneKey(current, tabId, activeLeafId) if (toAck.length > 0 || activePaneKey) { const paneKeysToClear = new Set(toAck) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 35d8dc50d42..7d621d77751 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -137,6 +137,10 @@ "menuBarIcon": { "title": "Show Menu Bar Icon", "description": "Keep an Orca shortcut and activity indicator in the macOS menu bar." + }, + "agentsSidebar": { + "title": "Show Agents Button", + "description": "Control whether the Agents tab appears in the left sidebar so you can monitor agent activity." } }, "browser": { @@ -945,6 +949,9 @@ "ephemeralVmWorktreeCreation": { "sparseCheckoutUnsupported": "Provisioned-root recipes do not support sparse checkout." }, + "worktreeJumpNavigation": { + "filteredNotice": "This worktree is hidden by sidebar filters. The workspace was opened, but it is not shown in Spaces." + }, "file": { "preview": { "pairedOutsideWorktree": "Files outside the workspace can't be previewed on a paired server yet." @@ -5231,12 +5238,16 @@ "keepDefaultBranchAria": "Keep the default branch visible while hiding sleeping workspaces" }, "SidebarHeader": { + "projects": "Projects", + "spaces": "Spaces", "25a95899c9": "Add Project", "92154beb7e": "New workspace", "49f62c5665": "Workspace board", "5c9c7c16aa": "Add a project to create workspaces", "ca6f729da2": "New workspace ({{value0}})", - "a30e34eb5c": "Close workspace board" + "a30e34eb5c": "Close workspace board", + "views": "Sidebar view", + "moreActions": "More workspace actions" }, "SidebarNav": { "80611a8b10": "Search", @@ -5368,7 +5379,8 @@ "65a9820bd1": "Agent statuses", "219ebf1961": "Branch name", "folderPathIdentity": "Branch / folder path", - "cli": "Orca CLI" + "cli": "Orca CLI", + "workspaceOptions": "Workspace options" }, "SshTargetRow": { "4677394048": "Connecting…", @@ -9305,7 +9317,8 @@ }, "agentDashboard": { "title": "Agent Dashboard", - "description": "Kanban board for monitoring agents across worktrees, in-window or as a pop-out." + "description": "Kanban board for monitoring agents across worktrees, in-window or as a pop-out.", + "dashboard": "dashboard" } } }, @@ -15909,19 +15922,48 @@ "b29191b3e0": "Worktree", "8c3b621ddf": "Project", "4a3986b200": "Status", - "770d458144": "Group agent activity by", + "770d458144": "Group by", "795cbf26e2": "Filter...", "4616ea39fd": "Jump to workspace", + "markThreadRead": "Mark thread as read", "59b131fbd9": "Mark thread unread", "beb2c19173": "Unread", "5651b216c6": "Unknown project", "22b22034bc": "Standalone terminal unavailable in Activity.", - "afdc2139a8": "Agent terminal closed. Open a new terminal in this workspace to continue." + "afdc2139a8": "Agent terminal closed. Open a new terminal in this workspace to continue.", + "compactModeDescription": "Shows shorter thread rows with one-line titles and two-line status messages.", + "unreadOnlyDescription": "Filters the activity list to show only threads with unread updates.", + "clearCompleted": "Clear completed", + "none": "None", + "search": "Search", + "showUnreadOnly": "Show unread only", + "showChildAgents": "Show child agents" + }, + "clearCompleted": { + "clearedOne": "Cleared 1 completed agent", + "clearedMany": "Cleared {{count}} completed agents", + "undo": "Undo" + }, + "standaloneWorktree": { + "floatingTerminal": "Floating terminal", + "standaloneTerminal": "Standalone terminal" }, "ActivityTitlebarControls": { "f915168c8e": "unread", "d6a8de3934": "agents", "dc708f3eff": "Close agents" + }, + "ActivityScopeFilterControls": { + "clearHostFilter": "Show all hosts", + "clearProjectFilter": "Show all projects", + "hiddenCount": "{{value0}} hidden" + }, + "ActivityThreadHoverCard": { + "pathCopied": "Path copied to clipboard", + "copyPathFailed": "Failed to copy path", + "workspace": "Workspace", + "copyPath": "Copy path", + "jumpToWorkspace": "Jump to workspace" } }, "confirmation": { @@ -17186,7 +17228,8 @@ }, "dashboard": { "sidebar": { - "label": "Agent Dashboard" + "label": "Agents", + "dashboardLabel": "Agent Dashboard" } }, "runtimeRpc": { @@ -17233,5 +17276,20 @@ "unlinkedPr": { "status": "PR #{{number}} unlinked" } + }, + "agentsSidebarIntro": { + "migrated": { + "title": "Agents are easier to find", + "description": "Your Agents view is now a dedicated sidebar tab. Your activity and filters are preserved.", + "dismiss": "Got it", + "action": "Open Agents" + }, + "new": { + "title": "Meet your Agents tab", + "description": "See what your agents are working on, what is done, and where you need to step in.", + "hide": "Hide Agents", + "action": "Try Agents", + "hiddenToast": "Agents tab hidden. Re-enable it in Settings → Experimental." + } } } diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index e55defa7518..4736cdbbbd4 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -4404,12 +4404,15 @@ "keepDefaultBranchAria": "Mantener visible la rama predeterminada al ocultar los espacios de trabajo en reposo" }, "SidebarHeader": { + "projects": "Proyectos", "92154beb7e": "Nuevo espacio de trabajo", "49f62c5665": "Tablero de espacios de trabajo", "5c9c7c16aa": "Agregar un proyecto para crear espacios de trabajo", "ca6f729da2": "Nuevo espacio de trabajo ({{value0}})", "a30e34eb5c": "Cerrar tablero del espacio de trabajo", - "25a95899c9": "Agregar proyecto" + "25a95899c9": "Agregar proyecto", + "spaces": "Espacios", + "views": "Vista de la barra lateral" }, "SidebarNav": { "80611a8b10": "Buscar", @@ -8231,7 +8234,8 @@ }, "agentDashboard": { "title": "Panel de agentes", - "description": "Tablero Kanban para monitorear agentes en diferentes worktrees, en ventana o como ventana emergente." + "description": "Tablero Kanban para monitorear agentes en diferentes worktrees, en ventana o como ventana emergente.", + "dashboard": "panel" } } }, @@ -14143,9 +14147,10 @@ "b29191b3e0": "worktree", "8c3b621ddf": "Proyecto", "4a3986b200": "Estado", - "770d458144": "Agrupar actividad del agente por", + "770d458144": "Agrupar por", "795cbf26e2": "Filtrar...", "4616ea39fd": "Ir al workspace", + "markThreadRead": "Marcar hilo como leído", "59b131fbd9": "Marcar hilo como no leído", "beb2c19173": "No leído", "5651b216c6": "Proyecto desconocido", @@ -14795,7 +14800,8 @@ }, "dashboard": { "sidebar": { - "label": "Panel de agentes" + "label": "Agentes", + "dashboardLabel": "Panel de agentes" } }, "browser": { @@ -14836,5 +14842,20 @@ "unknown": "Restart Orca to try again." } } + }, + "agentsSidebarIntro": { + "migrated": { + "title": "Los agentes son más fáciles de encontrar", + "description": "Tu vista de Agentes ahora es una pestaña dedicada de la barra lateral. Tu actividad y filtros se conservan.", + "dismiss": "Entendido", + "action": "Abrir Agentes" + }, + "new": { + "title": "Conoce tu pestaña de Agentes", + "description": "Ve en qué están trabajando tus agentes, qué está terminado y dónde necesitas intervenir.", + "hide": "Ocultar Agentes", + "action": "Probar Agentes", + "hiddenToast": "La pestaña Agentes está oculta. Vuelve a activarla en Configuración → Experimental." + } } } diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 5d97966fca0..315e5775442 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -4385,12 +4385,15 @@ "keepDefaultBranchAria": "スリープ中のワークスペースを非表示にしてもデフォルトのブランチは表示したままにする" }, "SidebarHeader": { + "projects": "プロジェクト", "92154beb7e": "新規ワークスペース", "49f62c5665": "ワークスペースボード", "5c9c7c16aa": "プロジェクトを追加してワークスペースを作成する", "ca6f729da2": "新規ワークスペース ({{value0}})", "a30e34eb5c": "ワークスペースボードを閉じる", - "25a95899c9": "プロジェクトを追加" + "25a95899c9": "プロジェクトを追加", + "spaces": "スペース", + "views": "サイドバービュー" }, "SidebarNav": { "80611a8b10": "検索", @@ -8253,7 +8256,8 @@ }, "agentDashboard": { "title": "Agent ダッシュボード", - "description": "ワークツリー Agent を監視するカンバンボード。ウィンドウ内またはポップアウトで表示できます。" + "description": "ワークツリー Agent を監視するカンバンボード。ウィンドウ内またはポップアウトで表示できます。", + "dashboard": "ダッシュボード" } } }, @@ -14143,9 +14147,10 @@ "b29191b3e0": "ワークツリー", "8c3b621ddf": "プロジェクト", "4a3986b200": "状態", - "770d458144": "Agent のアクティビティをグループ化する", + "770d458144": "グループ化", "795cbf26e2": "フィルター…", "4616ea39fd": "ワークスペースにジャンプ", + "markThreadRead": "スレッドを既読としてマーク", "59b131fbd9": "スレッドを未読としてマークする", "beb2c19173": "未読", "5651b216c6": "不明なプロジェクト", @@ -14795,7 +14800,8 @@ }, "dashboard": { "sidebar": { - "label": "Agent ダッシュボード" + "label": "Agent", + "dashboardLabel": "Agent ダッシュボード" } }, "browser": { @@ -14836,5 +14842,20 @@ "unknown": "Orca を再起動してから、もう一度お試しください。" } } + }, + "agentsSidebarIntro": { + "migrated": { + "title": "Agent が見つけやすくなりました", + "description": "Agent ビューはサイドバーの専用タブになりました。アクティビティとフィルターはそのまま引き継がれます。", + "dismiss": "OK", + "action": "Agent を開く" + }, + "new": { + "title": "Agent タブのご紹介", + "description": "Agent が何に取り組んでいるか、何が完了したか、どこで対応が必要かを確認できます。", + "hide": "Agent を隠す", + "action": "Agent を試す", + "hiddenToast": "Agent タブを非表示にしました。設定 → 実験的機能で再度有効にできます。" + } } } diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 795d359bce0..4a196c62304 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -3912,7 +3912,9 @@ "searchLinks": "링크 검색", "deleteSkills": "스킬 삭제…" }, - "SkillShareSelectionControls": { "01c5a15e02": "스킬 공유" }, + "SkillShareSelectionControls": { + "01c5a15e02": "스킬 공유" + }, "SkillRow": { "updatedUnknown": "날짜 없음", "pathCopied": "경로 복사됨", @@ -3922,13 +3924,17 @@ "viewDetails": "세부 정보 보기", "deleteSkill": "삭제…" }, - "SkillsList": { "listLabel": "스킬" }, + "SkillsList": { + "listLabel": "스킬" + }, "sourceStatus": { "missing": "폴더를 찾을 수 없음", "remoteRepo": "원격 리포지토리 — 검색 안 됨", "unavailable": "검색 안 됨" }, - "sources": { "heading": "스킬 폴더" }, + "sources": { + "heading": "스킬 폴더" + }, "sourceKind": { "home": "홈", "workspace": "워크스페이스", @@ -3950,7 +3956,10 @@ "linkOne": "링크 {{count}}개", "linkOther": "링크 {{count}}개" }, - "filter": { "allAgents": "모든 에이전트", "sharedAgent": "공유됨 (.agents)" }, + "filter": { + "allAgents": "모든 에이전트", + "sharedAgent": "공유됨 (.agents)" + }, "SkillsSelectionHeader": { "exit": "선택 나가기", "exitTooltip": "선택 나가기 · Esc", @@ -3959,7 +3968,11 @@ "clear": "지우기", "deleteTitle": "삭제할 스킬 선택" }, - "SkillDetailDialog": { "agents": "에이전트", "updated": "업데이트됨", "copy": "복사" }, + "SkillDetailDialog": { + "agents": "에이전트", + "updated": "업데이트됨", + "copy": "복사" + }, "SkillFreshnessNudge": { "titleOne": "설치된 Orca 스킬이 오래되었습니다", "titleMany": "설치된 Orca 스킬 {{value0}}개가 오래되었습니다", @@ -4377,12 +4390,15 @@ "keepDefaultBranchAria": "슬립 중인 워크스페이스를 숨겨도 기본 브랜치는 계속 표시" }, "SidebarHeader": { + "projects": "프로젝트", "92154beb7e": "새로운 워크스페이스", "49f62c5665": "워크스페이스 보드", "5c9c7c16aa": "워크스페이스를 만들려면 프로젝트를 추가하세요.", "ca6f729da2": "새 워크스페이스({{value0}})", "a30e34eb5c": "워크스페이스 보드 닫기", - "25a95899c9": "프로젝트 추가" + "25a95899c9": "프로젝트 추가", + "spaces": "스페이스", + "views": "사이드바 보기" }, "SidebarNav": { "80611a8b10": "검색", @@ -8208,7 +8224,8 @@ }, "agentDashboard": { "title": "에이전트 대시보드", - "description": "워크트리 전반에 걸친 에이전트를 모니터링하는 칸반 보드. 윈도우 내 또는 팝업으로 표시됩니다." + "description": "워크트리 전반에 걸친 에이전트를 모니터링하는 칸반 보드. 윈도우 내 또는 팝업으로 표시됩니다.", + "dashboard": "대시보드" } } }, @@ -14186,9 +14203,10 @@ "b29191b3e0": "워크트리", "8c3b621ddf": "프로젝트", "4a3986b200": "상태", - "770d458144": "agent 활동 그룹화 기준", + "770d458144": "그룹화 기준", "795cbf26e2": "필터...", "4616ea39fd": "워크스페이스로 이동", + "markThreadRead": "스레드를 읽은 것으로 표시", "59b131fbd9": "스레드를 읽지 않은 것으로 표시", "beb2c19173": "읽지 않음", "5651b216c6": "알 수 없는 프로젝트", @@ -14899,7 +14917,8 @@ }, "dashboard": { "sidebar": { - "label": "에이전트 대시보드" + "label": "에이전트", + "dashboardLabel": "에이전트 대시보드" } }, "browser": { @@ -14940,5 +14959,20 @@ "unknown": "Restart Orca to try again." } } + }, + "agentsSidebarIntro": { + "migrated": { + "title": "에이전트를 더 쉽게 찾을 수 있습니다", + "description": "에이전트 보기가 이제 사이드바 전용 탭이 되었습니다. 활동과 필터는 그대로 유지됩니다.", + "dismiss": "확인", + "action": "에이전트 열기" + }, + "new": { + "title": "에이전트 탭을 만나보세요", + "description": "에이전트가 무엇을 작업 중인지, 무엇이 완료되었는지, 어디에 개입이 필요한지 확인하세요.", + "hide": "에이전트 숨기기", + "action": "에이전트 사용해 보기", + "hiddenToast": "에이전트 탭이 숨겨졌습니다. 설정 → 실험 기능에서 다시 활성화하세요." + } } } diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 3aa8333b872..edbd96e2623 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -3922,7 +3922,9 @@ "searchLinks": "搜索链接", "deleteSkills": "删除技能…" }, - "SkillShareSelectionControls": { "01c5a15e02": "共享技能" }, + "SkillShareSelectionControls": { + "01c5a15e02": "共享技能" + }, "SkillRow": { "updatedUnknown": "无日期", "pathCopied": "路径已复制", @@ -3932,13 +3934,17 @@ "viewDetails": "查看详情", "deleteSkill": "删除…" }, - "SkillsList": { "listLabel": "技能" }, + "SkillsList": { + "listLabel": "技能" + }, "sourceStatus": { "missing": "未找到文件夹", "remoteRepo": "远程仓库 — 未扫描", "unavailable": "未扫描" }, - "sources": { "heading": "技能文件夹" }, + "sources": { + "heading": "技能文件夹" + }, "sourceKind": { "home": "主目录", "workspace": "工作区", @@ -3968,7 +3974,10 @@ "deleteLinkOne": "{{count}} 个链接", "deleteLinkOther": "{{count}} 个链接" }, - "filter": { "allAgents": "所有 Agent", "sharedAgent": "共享 (.agents)" }, + "filter": { + "allAgents": "所有 Agent", + "sharedAgent": "共享 (.agents)" + }, "SkillsSelectionHeader": { "exit": "退出选择", "exitTooltip": "退出选择 · Esc", @@ -3977,7 +3986,11 @@ "clear": "清除", "deleteTitle": "选择要删除的技能" }, - "SkillDetailDialog": { "agents": "Agent", "updated": "已更新", "copy": "复制" }, + "SkillDetailDialog": { + "agents": "Agent", + "updated": "已更新", + "copy": "复制" + }, "SkillFreshnessNudge": { "titleOne": "已安装的 Orca 技能已过期", "titleMany": "{{value0}} 个已安装的 Orca 技能已过期", @@ -4420,12 +4433,15 @@ "keepDefaultBranchAria": "隐藏休眠工作区时仍显示默认分支" }, "SidebarHeader": { + "projects": "项目", "92154beb7e": "新工作区", "49f62c5665": "工作区板", "5c9c7c16aa": "添加项目以创建工作区", "ca6f729da2": "新工作区 ({{value0}})", "a30e34eb5c": "关闭工作区板", - "25a95899c9": "添加项目" + "25a95899c9": "添加项目", + "spaces": "空间", + "views": "侧边栏视图" }, "SidebarNav": { "80611a8b10": "搜索", @@ -8251,7 +8267,8 @@ }, "agentDashboard": { "title": "智能体仪表盘", - "description": "用于监控跨工作树的智能体的看板,支持窗口内或弹出窗口显示。" + "description": "用于监控跨工作树的智能体的看板,支持窗口内或弹出窗口显示。", + "dashboard": "仪表盘" } } }, @@ -14186,9 +14203,10 @@ "b29191b3e0": "工作树", "8c3b621ddf": "项目", "4a3986b200": "状态", - "770d458144": "对智能体活动进行分组", + "770d458144": "分组方式", "795cbf26e2": "筛选...", "4616ea39fd": "跳转到工作区", + "markThreadRead": "将话题标记为已读", "59b131fbd9": "将话题标记为未读", "beb2c19173": "未读", "5651b216c6": "未知项目", @@ -14899,7 +14917,8 @@ }, "dashboard": { "sidebar": { - "label": "智能体仪表盘" + "label": "智能体", + "dashboardLabel": "智能体仪表盘" } }, "browser": { @@ -14940,5 +14959,20 @@ "unknown": "请重启 Orca 以重试。" } } + }, + "agentsSidebarIntro": { + "migrated": { + "title": "智能体更容易找到了", + "description": "智能体视图现在是侧边栏的专用标签页。你的活动和筛选条件都会保留。", + "dismiss": "知道了", + "action": "打开智能体" + }, + "new": { + "title": "认识你的智能体标签页", + "description": "查看智能体正在做什么、哪些已完成,以及哪些需要你介入。", + "hide": "隐藏智能体", + "action": "试用智能体", + "hiddenToast": "智能体标签页已隐藏。可在设置 → 实验性功能中重新启用。" + } } } diff --git a/src/renderer/src/lib/activity-thread-display.test.ts b/src/renderer/src/lib/activity-thread-display.test.ts index 6beb4387c13..03d5e788331 100644 --- a/src/renderer/src/lib/activity-thread-display.test.ts +++ b/src/renderer/src/lib/activity-thread-display.test.ts @@ -12,6 +12,8 @@ describe('isTerseAgentFollowUpPrompt', () => { expect(isTerseAgentFollowUpPrompt('yes')).toBe(true) expect(isTerseAgentFollowUpPrompt('ok proceed')).toBe(true) expect(isTerseAgentFollowUpPrompt('Looks good.')).toBe(true) + expect(isTerseAgentFollowUpPrompt('hi')).toBe(true) + expect(isTerseAgentFollowUpPrompt('hello')).toBe(true) }) it('keeps substantive prompts', () => { @@ -238,6 +240,69 @@ describe('getActivityThreadStatusPreview', () => { }) ).toBe('') }) + + it('surfaces the completed-turn reply on a finished row', () => { + expect( + getActivityThreadStatusPreview({ + state: 'done', + prompt: 'Audit the repo', + lastCompletedAssistantMessage: 'Filed 8 issues from the audit.' + }) + ).toBe('Filed 8 issues from the audit.') + }) + + it('does not show a prior completed reply while the agent is working', () => { + expect( + getActivityThreadStatusPreview({ + state: 'working', + prompt: 'Next turn', + lastCompletedAssistantMessage: 'Filed 8 issues from the audit.' + }) + ).toBe('') + }) + + it('skips orchestration worker_done wrap-up so the card shows the reply', () => { + expect( + getActivityThreadStatusPreview({ + state: 'done', + prompt: 'On the m4air environment, use the terminal', + lastAssistantMessage: + 'Task complete — worker_done sent. Summary of what happened: Verdict: The two PRs were merged.' + }) + ).toBe('The two PRs were merged.') + }) + + it('unwraps worker_done wrap-up that uses ascii dashes or markdown verdict labels', () => { + expect( + getActivityThreadStatusPreview({ + state: 'done', + prompt: 'On the m4air environment, use the terminal', + lastAssistantMessage: + 'Task complete -- worker_done sent. Summary of what happened: **Verdict:** The two PRs were merged.' + }) + ).toBe('The two PRs were merged.') + }) + + it('drops a worker_done report that has no assistant reply', () => { + expect( + getActivityThreadStatusPreview({ + state: 'done', + prompt: 'Fix checkout', + lastAssistantMessage: 'Done — worker_done sent with outcome succeeded.' + }) + ).toBe('') + }) + + it('falls through to the completed-turn reply when the live preview is only worker_done', () => { + expect( + getActivityThreadStatusPreview({ + state: 'done', + prompt: 'Fix checkout', + lastAssistantMessage: 'Done — worker_done sent with outcome succeeded.', + lastCompletedAssistantMessage: 'I updated the tests and checked the activity row.' + }) + ).toBe('I updated the tests and checked the activity row.') + }) }) describe('resolveActivityThreadStatusPreview', () => { @@ -254,4 +319,18 @@ describe('resolveActivityThreadStatusPreview', () => { ) ).toBe('Implemented the skill creator port.') }) + + it('keeps the previous recap after a greeting follow-up with no new assistant preview', () => { + expect( + resolveActivityThreadStatusPreview( + { + state: 'done', + prompt: 'hi', + lastAssistantMessage: '' + }, + 'done', + 'The two PRs were merged or superseded.' + ) + ).toBe('The two PRs were merged or superseded.') + }) }) diff --git a/src/renderer/src/lib/activity-thread-display.ts b/src/renderer/src/lib/activity-thread-display.ts index 7dd5f663370..c330edcb6f0 100644 --- a/src/renderer/src/lib/activity-thread-display.ts +++ b/src/renderer/src/lib/activity-thread-display.ts @@ -15,7 +15,7 @@ import { formatAgentToolPreview } from './agent-row-tool-preview' // Why: follow-up replies ("yes", "ok proceed") are valid hook prompts but are // terrible scan labels for a cross-worktree agent list — treat them as non-titles. const TERSE_FOLLOW_UP_PATTERN = - /^(yes|no|ok|yep|nope|sure|thanks|thank you|please|proceed|continue|go ahead|lgtm|done|looks good|ok proceed)\.?$/i + /^(yes|no|ok|yep|nope|sure|thanks|thank you|please|proceed|continue|go ahead|lgtm|done|looks good|ok proceed|hi|hey|hello|yo)\.?$/i export function isTerseAgentFollowUpPrompt(prompt: string): boolean { const trimmed = prompt.trim() @@ -153,11 +153,48 @@ function isMislabeledUserPrompt(text: string, entry: Pick): string { + const trimmed = text.trim() + if (!trimmed || isMislabeledUserPrompt(trimmed, entry)) { + return '' + } + const unwrapped = unwrapOrchestrationAssistantPreview(trimmed) + if (!unwrapped || /^[.\s…]+$/.test(unwrapped) || isMislabeledUserPrompt(unwrapped, entry)) { + return '' + } + return unwrapped +} + /** Latest agent activity line — tool step while working, assistant reply otherwise. */ export function getActivityThreadStatusPreview( entry: Pick< AgentStatusEntry, - 'state' | 'toolName' | 'toolInput' | 'lastAssistantMessage' | 'interrupted' | 'prompt' + | 'state' + | 'toolName' + | 'toolInput' + | 'lastAssistantMessage' + | 'lastCompletedAssistantMessage' + | 'interrupted' + | 'prompt' >, agentState?: AgentStatusState | null ): string { @@ -169,10 +206,15 @@ export function getActivityThreadStatusPreview( if (toolPreview) { return toolPreview } - const assistant = entry.lastAssistantMessage?.trim() ?? '' - if (assistant && !isMislabeledUserPrompt(assistant, entry)) { + const assistant = usefulAssistantReply(entry.lastAssistantMessage ?? '', entry) + if (assistant) { return assistant } + // Why: live working/waiting pings clear lastAssistantMessage; the completed-turn + // snapshot is the last useful reply once the agent is no longer in-flight. + if (state !== 'working' && state !== 'waiting') { + return usefulAssistantReply(entry.lastCompletedAssistantMessage ?? '', entry) + } return '' } @@ -180,7 +222,13 @@ export function getActivityThreadStatusPreview( export function resolveActivityThreadStatusPreview( entry: Pick< AgentStatusEntry, - 'state' | 'toolName' | 'toolInput' | 'lastAssistantMessage' | 'interrupted' | 'prompt' + | 'state' + | 'toolName' + | 'toolInput' + | 'lastAssistantMessage' + | 'lastCompletedAssistantMessage' + | 'interrupted' + | 'prompt' >, agentState: AgentStatusState | null | undefined, previousPreview?: string @@ -195,9 +243,5 @@ export function resolveActivityThreadStatusPreview( if (!isTerseAgentFollowUpPrompt(entry.prompt)) { return '' } - const previous = previousPreview?.trim() ?? '' - if (previous && !isMislabeledUserPrompt(previous, entry)) { - return previous - } - return '' + return usefulAssistantReply(previousPreview ?? '', entry) } diff --git a/src/renderer/src/lib/short-time-ago.ts b/src/renderer/src/lib/short-time-ago.ts new file mode 100644 index 00000000000..731f9cbc136 --- /dev/null +++ b/src/renderer/src/lib/short-time-ago.ts @@ -0,0 +1,16 @@ +/** Compact "now / 5m / 3h / 2d" age label shared by agent rows and activity threads. */ +export function formatShortTimeAgo(ts: number, now = Date.now()): string { + const delta = now - ts + if (delta < 60_000) { + return 'now' + } + const minutes = Math.floor(delta / 60_000) + if (minutes < 60) { + return `${minutes}m` + } + const hours = Math.floor(minutes / 60) + if (hours < 24) { + return `${hours}h` + } + return `${Math.floor(hours / 24)}d` +} diff --git a/src/renderer/src/lib/worktree-activation.ts b/src/renderer/src/lib/worktree-activation.ts index 1717cba7408..e7d3f9cae9c 100644 --- a/src/renderer/src/lib/worktree-activation.ts +++ b/src/renderer/src/lib/worktree-activation.ts @@ -187,6 +187,8 @@ export function activateAndRevealWorktree( * runtime-owned workspace with a live web session the host owns terminal creation, * so ensureWebRuntimeWorktreeTerminalAfterWake may still seed one (matches main). */ providesInitialSurface?: boolean + /** Keep sidebar filters intact when navigating to a hidden target. */ + clearSidebarFilters?: boolean } ): ActivateAndRevealResult | false { const state = useAppStore.getState() @@ -283,20 +285,22 @@ export function activateAndRevealWorktree( } // 5. Clear sidebar filters hiding the target — reveal needs the card rendered, else it silently no-ops. - if (state.filterRepoIds.length > 0 && !state.filterRepoIds.includes(wt.repoId)) { - state.setFilterRepoIds([]) - } - if ( - state.hideAutomationGeneratedWorkspaces && - wt.automationProvenance?.kind === 'created-by-automation' - ) { - state.setHideAutomationGeneratedWorkspaces(false) - } - if (state.hideCliCreatedWorkspaces && wt.cliProvenance?.kind === 'created-by-cli') { - state.setHideCliCreatedWorkspaces(false) - } - if (state.hideDetachedHeadWorkspaces && isDetachedHeadWorkspace(wt)) { - state.setHideDetachedHeadWorkspaces(false) + if (opts?.clearSidebarFilters !== false) { + if (state.filterRepoIds.length > 0 && !state.filterRepoIds.includes(wt.repoId)) { + state.setFilterRepoIds([]) + } + if ( + state.hideAutomationGeneratedWorkspaces && + wt.automationProvenance?.kind === 'created-by-automation' + ) { + state.setHideAutomationGeneratedWorkspaces(false) + } + if (state.hideCliCreatedWorkspaces && wt.cliProvenance?.kind === 'created-by-cli') { + state.setHideCliCreatedWorkspaces(false) + } + if (state.hideDetachedHeadWorkspaces && isDetachedHeadWorkspace(wt)) { + state.setHideDetachedHeadWorkspaces(false) + } } // 6. Reveal in sidebar @@ -326,11 +330,18 @@ export function activateAndRevealWorktree( */ export function activateAndRevealWorkspace( workspaceId: string, - opts?: { executionHostId?: ExecutionHostId; providesInitialSurface?: boolean } + opts?: { + executionHostId?: ExecutionHostId + providesInitialSurface?: boolean + /** Worktree-only: folder workspaces are never filter-hidden, so these are dropped there. */ + revealInSidebar?: boolean + clearSidebarFilters?: boolean + } ): ActivateAndRevealResult | false { const workspaceScope = parseWorkspaceKey(workspaceId) if (workspaceScope?.type === 'folder') { - return activateAndRevealFolderWorkspace(workspaceScope.folderWorkspaceId, opts) + const { revealInSidebar: _reveal, clearSidebarFilters: _clear, ...folderOpts } = opts ?? {} + return activateAndRevealFolderWorkspace(workspaceScope.folderWorkspaceId, folderOpts) } return activateAndRevealWorktree(workspaceId, opts) } diff --git a/src/renderer/src/lib/worktree-jump-navigation.test.ts b/src/renderer/src/lib/worktree-jump-navigation.test.ts new file mode 100644 index 00000000000..4c76316556c --- /dev/null +++ b/src/renderer/src/lib/worktree-jump-navigation.test.ts @@ -0,0 +1,151 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const mocks = vi.hoisted(() => ({ + getState: vi.fn(), + activateAndRevealWorkspace: vi.fn(), + getVisibleWorktreeShortcutTargets: vi.fn(), + worktreePassesSidebarFilters: vi.fn(), + warning: vi.fn() +})) + +vi.mock('@/store', () => ({ useAppStore: { getState: mocks.getState } })) +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorkspace: mocks.activateAndRevealWorkspace +})) +vi.mock('@/components/sidebar/visible-worktrees', () => ({ + getVisibleWorktreeShortcutTargets: mocks.getVisibleWorktreeShortcutTargets +})) +vi.mock('@/components/sidebar/worktree-filter-visibility', () => ({ + worktreePassesSidebarFilters: mocks.worktreePassesSidebarFilters +})) +vi.mock('sonner', () => ({ toast: { warning: mocks.warning } })) + +import { jumpToWorktreeFromSidebar } from './worktree-jump-navigation' + +describe('worktree jump navigation', () => { + beforeEach(() => { + vi.clearAllMocks() + mocks.activateAndRevealWorkspace.mockReturnValue({ primaryTabId: null }) + mocks.getVisibleWorktreeShortcutTargets.mockReturnValue([]) + mocks.worktreePassesSidebarFilters.mockReturnValue(false) + mocks.getState.mockReturnValue({ + sidebarBody: 'agents', + setSidebarBody: vi.fn(), + worktreesByRepo: { repo: [] }, + showSleepingWorkspaces: true, + filterRepoIds: ['other-repo'], + hideDefaultBranchWorkspace: false, + hideAutomationGeneratedWorkspaces: false, + hideCliCreatedWorkspaces: false, + hideDetachedHeadWorkspaces: false, + hideWorkspacesFromOtherDevices: false, + alwaysShowDefaultBranchWorkspace: true, + visibleWorkspaceHostIds: null, + workspaceHostScope: 'all', + revealWorktreeInSidebar: vi.fn(), + getKnownWorktreeById: vi.fn(() => ({ id: 'known' })) + }) + }) + + it('does not blame filters for a worktree that no longer exists', () => { + // A retained agent row can outlive its (deleted) worktree; every filter check fails for + // an unknown id, so without the existence guard any active filter would toast. + const state = mocks.getState() + state.getKnownWorktreeById.mockReturnValue(undefined) + + expect(jumpToWorktreeFromSidebar('repo::/deleted')).toBe(true) + + expect(mocks.worktreePassesSidebarFilters).not.toHaveBeenCalled() + expect(mocks.warning).not.toHaveBeenCalled() + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('repo::/deleted', {}) + }) + + it('switches the left sidebar to Spaces and warns when filters hide the target', () => { + const state = mocks.getState() + + expect(jumpToWorktreeFromSidebar('repo::/target')).toBe(true) + + expect(state.setSidebarBody).toHaveBeenCalledWith('workspaces') + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('repo::/target', { + revealInSidebar: false, + clearSidebarFilters: false + }) + expect(mocks.warning).toHaveBeenCalledOnce() + }) + + it('does not warn when the target is visible', () => { + mocks.getVisibleWorktreeShortcutTargets.mockReturnValue([{ id: 'wt-1' }]) + + jumpToWorktreeFromSidebar('wt-1') + + expect(mocks.warning).not.toHaveBeenCalled() + }) + + it('reveals without warning when activation wakes a target hidden only by Hide sleeping', () => { + const state = mocks.getState() + mocks.worktreePassesSidebarFilters.mockReturnValueOnce(false).mockReturnValueOnce(true) + + expect(jumpToWorktreeFromSidebar('wt-sleeping')).toBe(true) + + expect(state.revealWorktreeInSidebar).toHaveBeenCalledWith('wt-sleeping', {}) + expect(mocks.warning).not.toHaveBeenCalled() + }) + + it('reveals instead of warning when the target is only inside a collapsed group', () => { + // Absent from the rendered list (collapse elision) but not excluded by filters. + mocks.getVisibleWorktreeShortcutTargets.mockReturnValue([]) + mocks.worktreePassesSidebarFilters.mockReturnValue(true) + + expect(jumpToWorktreeFromSidebar('wt-collapsed')).toBe(true) + + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('wt-collapsed', {}) + expect(mocks.warning).not.toHaveBeenCalled() + }) + + it('passes the target execution host to the filter check so a local twin cannot vouch', () => { + mocks.getVisibleWorktreeShortcutTargets.mockReturnValue([]) + mocks.worktreePassesSidebarFilters.mockReturnValue(false) + + expect(jumpToWorktreeFromSidebar('repo::/target', { executionHostId: 'ssh:beta' })).toBe(true) + + expect(mocks.worktreePassesSidebarFilters).toHaveBeenCalledWith('repo::/target', 'ssh:beta') + expect(mocks.warning).toHaveBeenCalledOnce() + }) + + it('does not let a hostless legacy target vouch for a host-scoped one', () => { + // Legacy rows publish without executionHostId; treating that as a match would clear the + // user's filters instead of preserving them and warning. + mocks.getVisibleWorktreeShortcutTargets.mockReturnValue([{ id: 'repo::/target' }]) + mocks.worktreePassesSidebarFilters.mockReturnValue(false) + + expect(jumpToWorktreeFromSidebar('repo::/target', { executionHostId: 'ssh:beta' })).toBe(true) + + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('repo::/target', { + revealInSidebar: false, + clearSidebarFilters: false, + executionHostId: 'ssh:beta' + }) + expect(mocks.warning).toHaveBeenCalledOnce() + }) + + it('routes folder workspaces through the workspace dispatcher without a filter check', () => { + const state = mocks.getState() + + expect(jumpToWorktreeFromSidebar('folder:folder-1', { executionHostId: 'local' })).toBe(true) + + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('folder:folder-1', { + executionHostId: 'local' + }) + // Folder workspaces never get the filter-hidden treatment. + expect(mocks.worktreePassesSidebarFilters).not.toHaveBeenCalled() + expect(state.setSidebarBody).toHaveBeenCalledWith('workspaces') + }) + + it('propagates a blocked folder-workspace activation as failure', () => { + const state = mocks.getState() + mocks.activateAndRevealWorkspace.mockReturnValue(false) + + expect(jumpToWorktreeFromSidebar('folder:folder-1')).toBe(false) + expect(state.setSidebarBody).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/lib/worktree-jump-navigation.ts b/src/renderer/src/lib/worktree-jump-navigation.ts new file mode 100644 index 00000000000..bfc0d6dd936 --- /dev/null +++ b/src/renderer/src/lib/worktree-jump-navigation.ts @@ -0,0 +1,95 @@ +import { toast } from 'sonner' +import { translate } from '@/i18n/i18n' +import { useAppStore } from '@/store' +import { activateAndRevealWorkspace } from '@/lib/worktree-activation' +import { getVisibleWorktreeShortcutTargets } from '@/components/sidebar/visible-worktrees' +import { worktreePassesSidebarFilters } from '@/components/sidebar/worktree-filter-visibility' +import { sidebarHasActiveFilters } from '@/components/sidebar/sidebar-filter-actions' +import { parseWorkspaceKey } from '../../../shared/workspace-scope' +import { normalizeExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' + +function wasHiddenBySidebarFilters(worktreeId: string, executionHostId?: ExecutionHostId): boolean { + const state = useAppStore.getState() + // Some lightweight callers/tests provide only the activation slice of state. + if (!state.worktreesByRepo || !sidebarHasActiveFilters(state)) { + return false + } + + const inRenderedTargets = getVisibleWorktreeShortcutTargets().some((target) => { + if (target.id !== worktreeId) { + return false + } + if (!executionHostId) { + return true + } + // Why strict: legacy rows publish without a host, and a hostless twin must not vouch for a + // filtered ssh:*/runtime:* target — that would clear the user's filters instead of warning. + if (!target.executionHostId) { + return false + } + return ( + normalizeExecutionHostId(target.executionHostId) === normalizeExecutionHostId(executionHostId) + ) + }) + if (inRenderedTargets) { + return false + } + // Why: a retained agent can outlive its worktree; a deleted worktree fails every filter + // pass, so without this check any active filter would blame itself for the missing row. + if (!state.getKnownWorktreeById?.(worktreeId, executionHostId)) { + return false + } + // Absent from the rendered list can mean a collapsed group, not a filter: + // collapsed-but-unfiltered targets should be revealed, not toasted. The host + // matters: an id-only check would pass on a filtered target's same-id twin + // from another execution host. + return !worktreePassesSidebarFilters(worktreeId, executionHostId) +} + +/** Navigate from a worktree reference in either sidebar back to the workspace surface. */ +export function jumpToWorktreeFromSidebar( + worktreeId: string, + options?: { executionHostId?: ExecutionHostId } +): boolean { + const state = useAppStore.getState() + + // Folder workspaces aren't in the worktree filter pipeline; only git worktrees can be filter-hidden. + const hiddenBeforeActivation = + parseWorkspaceKey(worktreeId)?.type !== 'folder' && + wasHiddenBySidebarFilters(worktreeId, options?.executionHostId) + + // Why the workspace dispatcher: it owns the folder-vs-worktree split and the folder path-status gate. + const activated = activateAndRevealWorkspace(worktreeId, { + ...(hiddenBeforeActivation ? { revealInSidebar: false, clearSidebarFilters: false } : {}), + ...(options?.executionHostId ? { executionHostId: options.executionHostId } : {}) + }) + if (activated === false) { + return false + } + + // The worktree list is the Spaces/Projects sidebar body; jump actions should always expose it. + state.setSidebarBody?.('workspaces') + + const hiddenAfterActivation = + hiddenBeforeActivation && wasHiddenBySidebarFilters(worktreeId, options?.executionHostId) + if (hiddenBeforeActivation && !hiddenAfterActivation) { + // Activation can seed a terminal, making a workspace excluded only by Hide sleeping visible. + // Queue the reveal after that state transition instead of reporting a filter conflict. + useAppStore + .getState() + .revealWorktreeInSidebar( + worktreeId, + options?.executionHostId ? { executionHostId: options.executionHostId } : {} + ) + } + + if (hiddenAfterActivation) { + toast.warning( + translate( + 'auto.lib.worktreeJumpNavigation.filteredNotice', + 'This worktree is hidden by sidebar filters. The workspace was opened, but it is not shown in Spaces.' + ) + ) + } + return true +} diff --git a/src/renderer/src/lib/worktree-runtime-owner-index.ts b/src/renderer/src/lib/worktree-runtime-owner-index.ts index 7bd0ebf2eba..cddb3b81d90 100644 --- a/src/renderer/src/lib/worktree-runtime-owner-index.ts +++ b/src/renderer/src/lib/worktree-runtime-owner-index.ts @@ -265,17 +265,18 @@ export function findIndexedRepoOwner( return resolution.kind === 'resolved' ? resolution.owner : null } -export function findIndexedRepoOwnerForHost( - repos: readonly RepoOwnerRecord[] | undefined, +export function findIndexedRepoOwnerForHost( + repos: readonly T[] | undefined, repoId: string, executionHostId: ExecutionHostId -): RepoOwnerRecord | null { +): T | null { if (!repos) { return null } resolveIndexedRepoOwner(repos, repoId) const resolution = repoOwnerIndexCache.get(repos)?.get(`${repoId}\0${executionHostId}`) - return resolution?.kind === 'resolved' ? resolution.owner : null + // The cache is keyed by this exact array, so its owner retains the caller's row type. + return resolution?.kind === 'resolved' ? (resolution.owner as T) : null } export function findIndexedFolderWorkspaceOwner( diff --git a/src/renderer/src/runtime/web-session-tabs-sync/agent-status-primitives.ts b/src/renderer/src/runtime/web-session-tabs-sync/agent-status-primitives.ts index 30cb1ac7632..f0eb938f00e 100644 --- a/src/renderer/src/runtime/web-session-tabs-sync/agent-status-primitives.ts +++ b/src/renderer/src/runtime/web-session-tabs-sync/agent-status-primitives.ts @@ -167,10 +167,12 @@ export function buildRetractedMirroredTabSweepPatch( } const sweepState: RetiredTerminalTabSweepState = { acknowledgedAgentsByPaneKey: state.acknowledgedAgentsByPaneKey ?? {}, + activityClearedAtByPaneKey: state.activityClearedAtByPaneKey ?? {}, agentLaunchConfigByPaneKey: state.agentLaunchConfigByPaneKey ?? {}, agentStatusByPaneKey: agentStatusPatch?.agentStatusByPaneKey ?? state.agentStatusByPaneKey, agentStatusEpoch: agentStatusPatch?.agentStatusEpoch ?? state.agentStatusEpoch, migrationUnsupportedByPtyId: state.migrationUnsupportedByPtyId ?? {}, + manuallyUnreadTurnsByPaneKey: state.manuallyUnreadTurnsByPaneKey ?? {}, paneForegroundAgentByPaneKey: state.paneForegroundAgentByPaneKey ?? {}, recentlyClosedAgentStatusTabIds: state.recentlyClosedAgentStatusTabIds ?? {}, recentlyRetiredAgentStatusPaneKeys: state.recentlyRetiredAgentStatusPaneKeys ?? {}, @@ -181,7 +183,11 @@ export function buildRetractedMirroredTabSweepPatch( // so it must see the post-removal tab list, not the one the snapshot replaced. tabsByWorktree: nextTabsByWorktree } - const sweep = buildRetiredTerminalTabStateSweepPatch(sweepState, retractedTabIds, worktreeId) + // Why: a retraction can be a reconnect re-key, not pane death (ssh-execution-boundary); keeping + // cutoffs means a republished pane cannot replay activity the user cleared on this client. + const sweep = buildRetiredTerminalTabStateSweepPatch(sweepState, retractedTabIds, worktreeId, { + preserveActivityClearedState: true + }) if (!sweep?.agentStatusByPaneKey || !batchContext) { return sweep ?? null } diff --git a/src/renderer/src/runtime/web-session-tabs-sync/state.ts b/src/renderer/src/runtime/web-session-tabs-sync/state.ts index d8be13bde50..f44d4a3d18c 100644 --- a/src/renderer/src/runtime/web-session-tabs-sync/state.ts +++ b/src/renderer/src/runtime/web-session-tabs-sync/state.ts @@ -199,9 +199,11 @@ export type WebSessionTabsSyncState = Pick< Pick< AppState, | 'acknowledgedAgentsByPaneKey' + | 'activityClearedAtByPaneKey' | 'agentLaunchConfigByPaneKey' | 'automaticAgentResumeClaimsByTabId' | 'migrationUnsupportedByPtyId' + | 'manuallyUnreadTurnsByPaneKey' | 'paneForegroundAgentByPaneKey' | 'pendingStartupByTabId' | 'recentlyClosedAgentStatusTabIds' diff --git a/src/renderer/src/store/slices/activity-cleared-at.test.ts b/src/renderer/src/store/slices/activity-cleared-at.test.ts new file mode 100644 index 00000000000..3a59ee3a11b --- /dev/null +++ b/src/renderer/src/store/slices/activity-cleared-at.test.ts @@ -0,0 +1,169 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { RetainedAgentEntry } from './agent-status' +import { createTestStore } from './store-test-helpers' +import { + sanitizeAcknowledgedAgentsByPaneKey, + sanitizeActivityClearedAtByPaneKey +} from './ui/ui-slice-hydration-sanitizers' + +function makeRetained(paneKey: string, worktreeId = 'wt-1'): RetainedAgentEntry { + const entry: AgentStatusEntry = { + state: 'done', + prompt: 'run', + updatedAt: 1_000, + stateStartedAt: 1_000, + paneKey, + stateHistory: [], + agentType: 'claude' + } + const tab: TerminalTab = { + id: paneKey.split(':')[0], + ptyId: 'pty-1', + worktreeId, + title: 'agent', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 1 + } + return { entry, worktreeId, tab, agentType: 'claude', startedAt: 1_000 } +} + +describe('applyActivityClearedAt', () => { + it('merges stamps, deletes on null, and no-ops on identical patches', () => { + const store = createTestStore() + store.getState().applyActivityClearedAt({ 'a:1': 100, 'b:2': 200 }) + expect(store.getState().activityClearedAtByPaneKey).toEqual({ 'a:1': 100, 'b:2': 200 }) + + const before = store.getState().activityClearedAtByPaneKey + store.getState().applyActivityClearedAt({ 'a:1': 100 }) + // Identity preserved when nothing changed, so subscribers don't churn. + expect(store.getState().activityClearedAtByPaneKey).toBe(before) + + store.getState().applyActivityClearedAt({ 'a:1': null }) + expect(store.getState().activityClearedAtByPaneKey).toEqual({ 'b:2': 200 }) + + const afterDelete = store.getState().activityClearedAtByPaneKey + store.getState().applyActivityClearedAt({ missing: null }) + expect(store.getState().activityClearedAtByPaneKey).toBe(afterDelete) + }) +}) + +describe('dismissRetainedAgents', () => { + it('removes the named retained entries in one update and leaves others intact', () => { + const store = createTestStore() + store + .getState() + .retainAgents([makeRetained('tab-a:1'), makeRetained('tab-b:2'), makeRetained('tab-c:3')]) + store.getState().dismissRetainedAgents(['tab-a:1', 'tab-c:3', 'tab-unknown:9']) + expect(Object.keys(store.getState().retainedAgentsByPaneKey)).toEqual(['tab-b:2']) + }) + + it('plants a retention suppressor only for panes that still have a live entry', () => { + const store = createTestStore() + store.getState().retainAgents([makeRetained('tab-a:1'), makeRetained('tab-b:2')]) + store.setState({ + agentStatusByPaneKey: { + 'tab-a:1': { + state: 'done', + prompt: 'live', + updatedAt: 2_000, + stateStartedAt: 2_000, + paneKey: 'tab-a:1', + stateHistory: [], + agentType: 'claude' + } + } + }) + store.getState().dismissRetainedAgents(['tab-a:1', 'tab-b:2']) + expect(store.getState().retainedAgentsByPaneKey).toEqual({}) + // Live pane gets a one-shot suppressor; the gone pane must NOT (undo re-retains it cleanly). + expect(store.getState().retentionSuppressedPaneKeys['tab-a:1']).toBe(true) + expect(store.getState().retentionSuppressedPaneKeys['tab-b:2']).toBeUndefined() + }) + + it('no-ops without reallocation when nothing matches', () => { + const store = createTestStore() + store.getState().retainAgents([makeRetained('tab-a:1')]) + const before = store.getState().retainedAgentsByPaneKey + store.getState().dismissRetainedAgents(['tab-zz:9']) + expect(store.getState().retainedAgentsByPaneKey).toBe(before) + }) +}) + +describe('dropAgentStatus cleared-at/manual-unread lifecycle', () => { + // Why: setAgentStatus schedules a real 30-minute freshness setTimeout. + afterEach(() => { + vi.useRealTimers() + }) + + function seedLiveWithClearState(store: ReturnType): void { + vi.useFakeTimers() + store.getState().setAgentStatus('tab-a:1', { state: 'done', prompt: 'p', agentType: 'claude' }) + store.getState().applyActivityClearedAt({ 'tab-a:1': 5_000 }) + store.getState().unacknowledgeAgents(['tab-a:1']) + expect(store.getState().activityClearedAtByPaneKey['tab-a:1']).toBe(5_000) + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-a:1']).toBeGreaterThan(0) + } + + it('row dismissal keeps the cutoff and manual-unread stamp for a still-live pane', () => { + const store = createTestStore() + seedLiveWithClearState(store) + store.getState().dropAgentStatus('tab-a:1') + expect(store.getState().agentStatusByPaneKey['tab-a:1']).toBeUndefined() + // The pane may republish its full stateHistory; without the cutoff every + // cleared event would flood back as unread (the Clear-completed undo bug). + expect(store.getState().activityClearedAtByPaneKey['tab-a:1']).toBe(5_000) + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-a:1']).toBeGreaterThan(0) + }) + + it('paneRemoved drop clears the cutoff and manual-unread stamp with the pane', () => { + const store = createTestStore() + seedLiveWithClearState(store) + store.getState().dropAgentStatus('tab-a:1', { paneRemoved: true }) + expect(store.getState().activityClearedAtByPaneKey['tab-a:1']).toBeUndefined() + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-a:1']).toBeUndefined() + }) +}) + +describe('dropAgentStatusByTabPrefix preserveActivityClearedState', () => { + afterEach(() => { + vi.useRealTimers() + }) + + it('keeps cutoffs and manual-unread stamps for a mirrored-tab retraction sweep', () => { + vi.useFakeTimers() + const store = createTestStore() + store.getState().setAgentStatus('tab-a:1', { state: 'done', prompt: 'p', agentType: 'claude' }) + store.getState().applyActivityClearedAt({ 'tab-a:1': 5_000 }) + store.getState().unacknowledgeAgents(['tab-a:1']) + + store.getState().dropAgentStatusByTabPrefix('tab-a', { preserveActivityClearedState: true }) + + expect(store.getState().agentStatusByPaneKey['tab-a:1']).toBeUndefined() + // Loss of contact is not pane death: the host republishes the same panes on reconnect, + // and the preserved cutoff keeps cleared activity from replaying. + expect(store.getState().activityClearedAtByPaneKey['tab-a:1']).toBe(5_000) + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-a:1']).toBeGreaterThan(0) + + store.getState().dropAgentStatusByTabPrefix('tab-a') + expect(store.getState().activityClearedAtByPaneKey['tab-a:1']).toBeUndefined() + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-a:1']).toBeUndefined() + }) +}) + +describe('sanitizeActivityClearedAtByPaneKey hydration TTL', () => { + it('keeps cutoffs past the 7-day ack TTL so they outlive the persisted entries they guard', () => { + const eightDaysAgo = Date.now() - 8 * 24 * 60 * 60 * 1000 + const record = { 'tab-a:1': eightDaysAgo } + // Main prunes persisted entries at 7d from receivedAt; a same-aged cutoff must survive + // hydration or the entry it shadows replays as unread on restart. + expect(sanitizeAcknowledgedAgentsByPaneKey(record)).toEqual({}) + expect(sanitizeActivityClearedAtByPaneKey(record)).toEqual(record) + + const fifteenDaysAgo = Date.now() - 15 * 24 * 60 * 60 * 1000 + expect(sanitizeActivityClearedAtByPaneKey({ 'tab-a:1': fifteenDaysAgo })).toEqual({}) + }) +}) diff --git a/src/renderer/src/store/slices/agent-pane-authority.test.ts b/src/renderer/src/store/slices/agent-pane-authority.test.ts index a5df08cd4a8..5e97e3c9064 100644 --- a/src/renderer/src/store/slices/agent-pane-authority.test.ts +++ b/src/renderer/src/store/slices/agent-pane-authority.test.ts @@ -74,6 +74,20 @@ describe('agent pane authority', () => { expect(retirePaneAuthority).toHaveBeenCalledWith(TARGET) }) + it('retires the pane activity cutoff with the rest of its pane-owned state', () => { + const store = createTestStore() + store.setState({ + activityClearedAtByPaneKey: { + [TARGET]: 1_000, + [SIBLING]: 2_000 + } + }) + + store.getState().retireAgentPaneAuthority(TARGET) + + expect(store.getState().activityClearedAtByPaneKey).toEqual({ [SIBLING]: 2_000 }) + }) + // STA-4114: the renderer tombstone outlived the detach/reattach cycle, so a pane // that was still running never showed status again for the rest of its life. it('lifts the retirement fence on re-attach so an in-flight turn can still report done', () => { diff --git a/src/renderer/src/store/slices/agent-status-ack-cleanup.test.ts b/src/renderer/src/store/slices/agent-status-ack-cleanup.test.ts index 4ca35b9bdc2..773e3879aba 100644 --- a/src/renderer/src/store/slices/agent-status-ack-cleanup.test.ts +++ b/src/renderer/src/store/slices/agent-status-ack-cleanup.test.ts @@ -20,11 +20,17 @@ describe('acknowledgedAgentsByPaneKey cleanup on teardown', () => { .getState() .setAgentStatus('tab-1:0', { state: 'working', prompt: 'p', agentType: 'claude' }) store.getState().acknowledgeAgents(['tab-1:0']) + store.setState({ + activityClearedAtByPaneKey: { 'tab-1:0': 100 }, + manuallyUnreadTurnsByPaneKey: { 'tab-1:0': 200 } + }) expect(store.getState().acknowledgedAgentsByPaneKey['tab-1:0']).toBeGreaterThan(0) store.getState().removeAgentStatus('tab-1:0') expect(store.getState().acknowledgedAgentsByPaneKey['tab-1:0']).toBeUndefined() + expect(store.getState().activityClearedAtByPaneKey['tab-1:0']).toBeUndefined() + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-1:0']).toBeUndefined() }) it('removeAgentStatusByTabPrefix drops every ack entry whose paneKey starts with the tab prefix', () => { @@ -40,6 +46,10 @@ describe('acknowledgedAgentsByPaneKey cleanup on teardown', () => { .getState() .setAgentStatus('tab-10:0', { state: 'working', prompt: 'p', agentType: 'claude' }) store.getState().acknowledgeAgents(['tab-1:0', 'tab-1:1', 'tab-10:0']) + store.setState({ + activityClearedAtByPaneKey: { 'tab-1:0': 100, 'tab-10:0': 300 }, + manuallyUnreadTurnsByPaneKey: { 'tab-1:1': 200, 'tab-10:0': 400 } + }) store.getState().removeAgentStatusByTabPrefix('tab-1') @@ -49,6 +59,29 @@ describe('acknowledgedAgentsByPaneKey cleanup on teardown', () => { // Why: the ":" delimiter on the prefix guards against false-prefix matches // across tab ids that share a leading substring (tab-1 vs tab-10). expect(ack['tab-10:0']).toBeGreaterThan(0) + expect(store.getState().activityClearedAtByPaneKey).toEqual({ 'tab-10:0': 300 }) + expect(store.getState().manuallyUnreadTurnsByPaneKey).toEqual({ 'tab-10:0': 400 }) + }) + + it('removeAgentStatus leaves read state alone for a retained-only pane (unverified SSH exit)', () => { + vi.useFakeTimers() + const store = createTestStore() + // Why: PTY exit calls removeAgentStatus unconditionally, including a synthetic exit from a + // lost SSH link. The live row is already gone; the retained/persisted read state must + // survive so acked rows do not re-bold and cleared history does not replay on reconnect. + store.getState().acknowledgeAgents(['tab-6:0']) + store.setState({ + activityClearedAtByPaneKey: { 'tab-6:0': 100 }, + manuallyUnreadTurnsByPaneKey: { 'tab-6:0': 200 } + }) + const epochBefore = store.getState().agentStatusEpoch + + store.getState().removeAgentStatus('tab-6:0') + + expect(store.getState().acknowledgedAgentsByPaneKey['tab-6:0']).toBeGreaterThan(0) + expect(store.getState().activityClearedAtByPaneKey['tab-6:0']).toBe(100) + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-6:0']).toBe(200) + expect(store.getState().agentStatusEpoch).toBe(epochBefore) }) it('dropAgentStatus drops the ack entry even when the pane had no live entry', () => { @@ -81,6 +114,34 @@ describe('acknowledgedAgentsByPaneKey cleanup on teardown', () => { expect(ack['tab-3:1']).toBeUndefined() }) + it('dropAgentStatusByTabPrefix clears pane-keyed activity maps without live rows', () => { + vi.useFakeTimers() + const store = createTestStore() + store.setState({ + activityClearedAtByPaneKey: { 'tab-4:0': 100 }, + manuallyUnreadTurnsByPaneKey: { 'tab-4:0': 200 } + }) + + store.getState().dropAgentStatusByTabPrefix('tab-4') + + expect(store.getState().activityClearedAtByPaneKey['tab-4:0']).toBeUndefined() + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-4:0']).toBeUndefined() + }) + + it('dropHibernatedAgentStatusPane clears pane-keyed activity maps without completion evidence', () => { + vi.useFakeTimers() + const store = createTestStore() + store.setState({ + activityClearedAtByPaneKey: { 'tab-5:0': 100 }, + manuallyUnreadTurnsByPaneKey: { 'tab-5:0': 200 } + }) + + store.getState().dropHibernatedAgentStatusPane('wt-1', 'tab-5:0') + + expect(store.getState().activityClearedAtByPaneKey['tab-5:0']).toBeUndefined() + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-5:0']).toBeUndefined() + }) + it('a paneKey reused after teardown reads as unvisited (no leaked ack suppresses the signal)', () => { vi.useFakeTimers() vi.setSystemTime(new Date('2026-04-29T12:00:00.000Z')) diff --git a/src/renderer/src/store/slices/agent-status-authority-actions.ts b/src/renderer/src/store/slices/agent-status-authority-actions.ts index 0d96ca524ed..c0d9cae2cb0 100644 --- a/src/renderer/src/store/slices/agent-status-authority-actions.ts +++ b/src/renderer/src/store/slices/agent-status-authority-actions.ts @@ -72,6 +72,14 @@ export function createAgentStatusAuthorityActions( s.acknowledgedAgentsByPaneKey, retiredPaneKeySet ), + activityClearedAtByPaneKey: removePaneKeys( + s.activityClearedAtByPaneKey, + retiredPaneKeySet + ), + manuallyUnreadTurnsByPaneKey: removePaneKeys( + s.manuallyUnreadTurnsByPaneKey, + retiredPaneKeySet + ), paneForegroundAgentByPaneKey: removePaneKeys( s.paneForegroundAgentByPaneKey, retiredPaneKeySet @@ -199,6 +207,8 @@ export function createAgentStatusAuthorityActions( }) ), acknowledgedAgentsByPaneKey: movePaneKeyedRecord(s.acknowledgedAgentsByPaneKey, from, to), + activityClearedAtByPaneKey: movePaneKeyedRecord(s.activityClearedAtByPaneKey, from, to), + manuallyUnreadTurnsByPaneKey: movePaneKeyedRecord(s.manuallyUnreadTurnsByPaneKey, from, to), paneForegroundAgentByPaneKey: movePaneKeyedRecord(s.paneForegroundAgentByPaneKey, from, to), unreadTerminalPanes: movePaneKeyedRecord(s.unreadTerminalPanes, from, to), unreadAgentCompletionPanes: movePaneKeyedRecord(s.unreadAgentCompletionPanes, from, to), diff --git a/src/renderer/src/store/slices/agent-status-cleanup-actions.ts b/src/renderer/src/store/slices/agent-status-cleanup-actions.ts index 352a86526df..e47369e791c 100644 --- a/src/renderer/src/store/slices/agent-status-cleanup-actions.ts +++ b/src/renderer/src/store/slices/agent-status-cleanup-actions.ts @@ -2,6 +2,7 @@ import type { AgentStatusSlice } from './agent-status-slice-contract' import type { AgentStatusRuntime } from './agent-status-runtime' import { collectWorktreeIdsForConnection } from './agent-status-connection-worktree-scope' import { pruneMigrationUnsupportedEntries } from './agent-status-migration-unsupported-entries' +import { removePaneKeys, removePaneKeysByTabPrefix } from './agent-status-pane-keyed-records' /** Actions for removing transient rows and migration-era cache entries. */ export function createAgentStatusCleanupActions( @@ -50,6 +51,9 @@ export function createAgentStatusCleanupActions( removeAgentStatus: (paneKey) => { const current = get() + // Why no ack/cleared-at/manual-unread in the guard: PTY exit calls this unconditionally, + // including unverified exits from a lost SSH link. A retained-only pane keeps its read + // state (see preserveActivityClearedState); only a row that is actually here gets swept. if ( !(paneKey in current.agentStatusByPaneKey) && !(paneKey in current.agentLaunchConfigByPaneKey) && @@ -76,12 +80,10 @@ export function createAgentStatusCleanupActions( s.migrationUnsupportedByPtyId, (entry) => entry.paneKey === paneKey ) - // Ack entries belong to the pane lifecycle; never let a reused key inherit one. - let nextAck = s.acknowledgedAgentsByPaneKey - if (paneKey in nextAck) { - nextAck = { ...nextAck } - delete nextAck[paneKey] - } + const paneKeys = new Set([paneKey]) + const nextAck = removePaneKeys(s.acknowledgedAgentsByPaneKey, paneKeys) + const nextClearedAt = removePaneKeys(s.activityClearedAtByPaneKey, paneKeys) + const nextManualUnread = removePaneKeys(s.manuallyUnreadTurnsByPaneKey, paneKeys) return { agentStatusByPaneKey: next, agentLaunchConfigByPaneKey: nextLaunchConfigs, @@ -89,6 +91,12 @@ export function createAgentStatusCleanupActions( ...(nextAck !== s.acknowledgedAgentsByPaneKey ? { acknowledgedAgentsByPaneKey: nextAck } : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}), agentStatusEpoch: s.agentStatusEpoch + 1, sortEpoch: s.sortEpoch + 1 } @@ -108,7 +116,17 @@ export function createAgentStatusCleanupActions( const hasMigrationUnsupported = Object.values(current.migrationUnsupportedByPtyId).some( (entry) => entry.paneKey?.startsWith(prefix) ) - if (toRemove.length === 0 && launchConfigKeys.length === 0 && !hasMigrationUnsupported) { + const hasPaneActivityState = [ + current.acknowledgedAgentsByPaneKey, + current.activityClearedAtByPaneKey, + current.manuallyUnreadTurnsByPaneKey + ].some((record) => Object.keys(record).some((key) => key.startsWith(prefix))) + if ( + toRemove.length === 0 && + launchConfigKeys.length === 0 && + !hasMigrationUnsupported && + !hasPaneActivityState + ) { return } set((s) => { @@ -124,14 +142,12 @@ export function createAgentStatusCleanupActions( s.migrationUnsupportedByPtyId, (entry) => entry.paneKey?.startsWith(prefix) ?? false ) - let nextAck = s.acknowledgedAgentsByPaneKey - const ackKeys = Object.keys(nextAck).filter((key) => key.startsWith(prefix)) - if (ackKeys.length > 0) { - nextAck = { ...nextAck } - for (const key of ackKeys) { - delete nextAck[key] - } - } + const nextAck = removePaneKeysByTabPrefix(s.acknowledgedAgentsByPaneKey, tabIdPrefix) + const nextClearedAt = removePaneKeysByTabPrefix(s.activityClearedAtByPaneKey, tabIdPrefix) + const nextManualUnread = removePaneKeysByTabPrefix( + s.manuallyUnreadTurnsByPaneKey, + tabIdPrefix + ) return { agentStatusByPaneKey: next, agentLaunchConfigByPaneKey: nextLaunchConfigs, @@ -139,6 +155,12 @@ export function createAgentStatusCleanupActions( ...(nextAck !== s.acknowledgedAgentsByPaneKey ? { acknowledgedAgentsByPaneKey: nextAck } : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}), agentStatusEpoch: s.agentStatusEpoch + 1, sortEpoch: s.sortEpoch + 1 } diff --git a/src/renderer/src/store/slices/agent-status-contract.ts b/src/renderer/src/store/slices/agent-status-contract.ts index 65b91e4074b..64dcb7f6919 100644 --- a/src/renderer/src/store/slices/agent-status-contract.ts +++ b/src/renderer/src/store/slices/agent-status-contract.ts @@ -42,8 +42,18 @@ export type DropHibernatedAgentPaneOptions = { retainedCompletionEvidence?: readonly RetainedAgentEntry[] } +export type DropAgentStatusOptions = { + /** The pane itself is gone (pane close, stale-row teardown). Row-only dismissals leave the + * cleared-at cutoff and manual-unread stamp in place so a still-live pane's next hook event + * cannot resurrect activity the user already cleared. */ + paneRemoved?: boolean +} + export type DropAgentStatusByTabPrefixOptions = { worktreeId?: string + /** Keep cleared-at cutoffs and manual-unread stamps: a mirrored-tab retraction is loss of + * contact, not pane death, and the host republishes the same panes on reconnect. */ + preserveActivityClearedState?: boolean } export type AgentLaunchConfigRegistrationMetadata = { diff --git a/src/renderer/src/store/slices/agent-status-drop-actions.ts b/src/renderer/src/store/slices/agent-status-drop-actions.ts index c918793a303..cff59fdc329 100644 --- a/src/renderer/src/store/slices/agent-status-drop-actions.ts +++ b/src/renderer/src/store/slices/agent-status-drop-actions.ts @@ -1,6 +1,7 @@ import type { RetainedAgentEntry, DropAgentStatusByTabPrefixOptions, + DropAgentStatusOptions, DropHibernatedAgentPaneOptions } from './agent-status-contract' import type { AgentStatusSlice } from './agent-status-slice-contract' @@ -33,7 +34,7 @@ export function createAgentStatusDropActions( > { const { set, freshness } = runtime return { - dropAgentStatus: (paneKey) => { + dropAgentStatus: (paneKey, opts?: DropAgentStatusOptions) => { let liveExisted = false set((s) => { const hasLive = paneKey in s.agentStatusByPaneKey @@ -44,6 +45,14 @@ export function createAgentStatusDropActions( (entry) => entry.paneKey === paneKey ) const nextAck = removeAcknowledgement(s.acknowledgedAgentsByPaneKey, paneKey) + // Row dismissal keeps cutoff/manual-unread: the pane may still be live, and its next + // hook event would replay every cleared stateHistory event as unread without them. + const nextClearedAt = opts?.paneRemoved + ? removeAcknowledgement(s.activityClearedAtByPaneKey, paneKey) + : s.activityClearedAtByPaneKey + const nextManualUnread = opts?.paneRemoved + ? removeAcknowledgement(s.manuallyUnreadTurnsByPaneKey, paneKey) + : s.manuallyUnreadTurnsByPaneKey const hasLaunchConfig = paneKey in s.agentLaunchConfigByPaneKey const nextLaunchConfigs = hasLaunchConfig ? { ...s.agentLaunchConfigByPaneKey } @@ -52,17 +61,19 @@ export function createAgentStatusDropActions( delete nextLaunchConfigs[paneKey] } if (!hasLive && !hasRetained && !migrationUnsupported.changed) { - if (hasLaunchConfig) { - return { - agentLaunchConfigByPaneKey: nextLaunchConfigs, - ...(nextAck !== s.acknowledgedAgentsByPaneKey - ? { acknowledgedAgentsByPaneKey: nextAck } - : {}) - } + const cleanupPatch = { + ...(hasLaunchConfig ? { agentLaunchConfigByPaneKey: nextLaunchConfigs } : {}), + ...(nextAck !== s.acknowledgedAgentsByPaneKey + ? { acknowledgedAgentsByPaneKey: nextAck } + : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}) } - return nextAck !== s.acknowledgedAgentsByPaneKey - ? { acknowledgedAgentsByPaneKey: nextAck } - : s + return Object.keys(cleanupPatch).length > 0 ? cleanupPatch : s } const nextLive = hasLive ? { ...s.agentStatusByPaneKey } : s.agentStatusByPaneKey if (hasLive) { @@ -83,6 +94,8 @@ export function createAgentStatusDropActions( ...(nextAck !== s.acknowledgedAgentsByPaneKey ? { acknowledgedAgentsByPaneKey: nextAck } : {}), + activityClearedAtByPaneKey: nextClearedAt, + manuallyUnreadTurnsByPaneKey: nextManualUnread, ...(needsSuppressor ? { retentionSuppressedPaneKeys: { @@ -160,6 +173,12 @@ export function createAgentStatusDropActions( const nextAck = !keepsCompletionEvidence ? removeAcknowledgement(s.acknowledgedAgentsByPaneKey, paneKey) : s.acknowledgedAgentsByPaneKey + const nextClearedAt = !keepsCompletionEvidence + ? removeAcknowledgement(s.activityClearedAtByPaneKey, paneKey) + : s.activityClearedAtByPaneKey + const nextManualUnread = !keepsCompletionEvidence + ? removeAcknowledgement(s.manuallyUnreadTurnsByPaneKey, paneKey) + : s.manuallyUnreadTurnsByPaneKey if ( !hasLive && !hasRetained && @@ -167,9 +186,18 @@ export function createAgentStatusDropActions( !migrationUnsupported.changed && !keepsCompletionEvidence ) { - return nextAck !== s.acknowledgedAgentsByPaneKey - ? { acknowledgedAgentsByPaneKey: nextAck } - : s + const cleanupPatch = { + ...(nextAck !== s.acknowledgedAgentsByPaneKey + ? { acknowledgedAgentsByPaneKey: nextAck } + : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}) + } + return Object.keys(cleanupPatch).length > 0 ? cleanupPatch : s } hadLive = hasLive const nextLive = hasLive ? { ...s.agentStatusByPaneKey } : s.agentStatusByPaneKey @@ -204,6 +232,8 @@ export function createAgentStatusDropActions( ...(nextAck !== s.acknowledgedAgentsByPaneKey ? { acknowledgedAgentsByPaneKey: nextAck } : {}), + activityClearedAtByPaneKey: nextClearedAt, + manuallyUnreadTurnsByPaneKey: nextManualUnread, ...(needsSuppressor ? { retentionSuppressedPaneKeys: { diff --git a/src/renderer/src/store/slices/agent-status-drop-reducer.ts b/src/renderer/src/store/slices/agent-status-drop-reducer.ts index 5d8387e9bf0..ca48b195238 100644 --- a/src/renderer/src/store/slices/agent-status-drop-reducer.ts +++ b/src/renderer/src/store/slices/agent-status-drop-reducer.ts @@ -3,7 +3,8 @@ import type { DropAgentStatusByTabPrefixOptions } from './agent-status-contract' import { pruneMigrationUnsupportedEntries } from './agent-status-migration-unsupported-entries' import { boundRecentlyClosedAgentStatusTabIds, - boundRecentlyRetiredAgentStatusPaneKeys + boundRecentlyRetiredAgentStatusPaneKeys, + removePaneKeysByTabPrefix } from './agent-status-pane-keyed-records' import { findCompletedOrphanPaneKeysForTabClose } from './agent-status-pane-key-tab-binding' @@ -26,10 +27,12 @@ export function buildAgentStatusBatchPatch( export type AgentStatusTabPrefixDropState = Pick< AppState, | 'acknowledgedAgentsByPaneKey' + | 'activityClearedAtByPaneKey' | 'agentLaunchConfigByPaneKey' | 'agentStatusByPaneKey' | 'agentStatusEpoch' | 'migrationUnsupportedByPtyId' + | 'manuallyUnreadTurnsByPaneKey' | 'recentlyClosedAgentStatusTabIds' | 'recentlyRetiredAgentStatusPaneKeys' | 'retainedAgentsByPaneKey' @@ -86,6 +89,16 @@ export function buildAgentStatusTabPrefixDropPatch( s.recentlyRetiredAgentStatusPaneKeys, retiredAliasPaneKeys ) + const nextClearedAt = opts?.preserveActivityClearedState + ? s.activityClearedAtByPaneKey + : removePaneKeysByTabPrefix(s.activityClearedAtByPaneKey, tabIdPrefix, completedOrphanKeySet) + const nextManualUnread = opts?.preserveActivityClearedState + ? s.manuallyUnreadTurnsByPaneKey + : removePaneKeysByTabPrefix( + s.manuallyUnreadTurnsByPaneKey, + tabIdPrefix, + completedOrphanKeySet + ) if ( liveKeys.length === 0 && @@ -96,13 +109,25 @@ export function buildAgentStatusTabPrefixDropPatch( if (nextAck !== s.acknowledgedAgentsByPaneKey) { return { acknowledgedAgentsByPaneKey: nextAck, + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}), recentlyClosedAgentStatusTabIds: nextClosedTabs, recentlyRetiredAgentStatusPaneKeys: nextRetiredPaneKeys } } return { recentlyClosedAgentStatusTabIds: nextClosedTabs, - recentlyRetiredAgentStatusPaneKeys: nextRetiredPaneKeys + recentlyRetiredAgentStatusPaneKeys: nextRetiredPaneKeys, + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}) } } hadLive = liveKeys.length > 0 @@ -148,6 +173,12 @@ export function buildAgentStatusTabPrefixDropPatch( ...(nextAck !== s.acknowledgedAgentsByPaneKey ? { acknowledgedAgentsByPaneKey: nextAck } : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}), // Why: mirrors removeAgentStatusByTabPrefix — only bump epochs when the live map changed; retained-only sweeps don't affect sort/freshness. agentStatusEpoch: hadLive || migrationUnsupported.changed ? s.agentStatusEpoch + 1 : s.agentStatusEpoch, diff --git a/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts b/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts index 1e28710c564..af5cc4e83c7 100644 --- a/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts +++ b/src/renderer/src/store/slices/agent-status-pane-keyed-records.ts @@ -74,3 +74,15 @@ export function removePaneKeys( } return next } + +export function removePaneKeysByTabPrefix( + record: Record, + tabPrefix: string, + extraPaneKeys: ReadonlySet = new Set() +): Record { + const prefix = `${tabPrefix}:` + const matchingKeys = Object.keys(record).filter( + (key) => key.startsWith(prefix) || extraPaneKeys.has(key) + ) + return removePaneKeys(record, new Set(matchingKeys)) +} diff --git a/src/renderer/src/store/slices/agent-status-provider-session-actions.ts b/src/renderer/src/store/slices/agent-status-provider-session-actions.ts index b8a88b9bca6..32e5378fe11 100644 --- a/src/renderer/src/store/slices/agent-status-provider-session-actions.ts +++ b/src/renderer/src/store/slices/agent-status-provider-session-actions.ts @@ -134,6 +134,7 @@ export function createAgentStatusProviderSessionActions( if (nextRetained !== s.retainedAgentsByPaneKey) { delete nextRetained[paneKey] } + const retiredPaneKeys = new Set([paneKey]) // Why: on identity mismatch the sleeping record drops its launch config, so clear the stale // registry entry too, else a later return to the old identity reuses stale args/env. let nextLaunchConfigs = s.agentLaunchConfigByPaneKey @@ -159,12 +160,11 @@ export function createAgentStatusProviderSessionActions( agentLaunchConfigByPaneKey: nextLaunchConfigs, acknowledgedAgentsByPaneKey: removePaneKeys( s.acknowledgedAgentsByPaneKey, - new Set([paneKey]) - ), - unreadAgentCompletionPanes: removePaneKeys( - s.unreadAgentCompletionPanes, - new Set([paneKey]) + retiredPaneKeys ), + // Why the cleared-at/manual-unread maps stay: the pane survives this transition, so a + // repeat heartbeat would resurrect cleared history. They are swept on pane retirement. + unreadAgentCompletionPanes: removePaneKeys(s.unreadAgentCompletionPanes, retiredPaneKeys), agentStatusEpoch: removedLiveStatus ? s.agentStatusEpoch + 1 : s.agentStatusEpoch, sortEpoch: removedLiveStatus ? s.sortEpoch + 1 : s.sortEpoch } diff --git a/src/renderer/src/store/slices/agent-status-provider-session.test.ts b/src/renderer/src/store/slices/agent-status-provider-session.test.ts index 2ad173eac6d..f854517d94f 100644 --- a/src/renderer/src/store/slices/agent-status-provider-session.test.ts +++ b/src/renderer/src/store/slices/agent-status-provider-session.test.ts @@ -222,6 +222,46 @@ describe('recordAgentProviderSession', () => { }) }) + it('keeps the activity cutoff and manual-unread stamp across a heartbeat for the same pane', () => { + const store = createTestStore() + store.setState({ + tabsByWorktree: { 'wt-1': [makeTab({ id: 'tab-1', worktreeId: 'wt-1' })] } + } as Partial) + const providerSession = { + key: 'session_id' as const, + id: 'pi-session-1', + transcriptPath: '/tmp/pi-session-1.jsonl' + } + store + .getState() + .setAgentStatus( + 'tab-1:leaf-1', + { state: 'done', prompt: 'finish', agentType: 'pi' }, + 'Pi', + { updatedAt: 10, stateStartedAt: 10 }, + { tabId: 'tab-1', worktreeId: 'wt-1' } + ) + store.getState().applyActivityClearedAt({ 'tab-1:leaf-1': 5_000 }) + store.getState().unacknowledgeAgents(['tab-1:leaf-1']) + + store.getState().recordAgentProviderSession( + 'tab-1:leaf-1', + 'pi', + providerSession, + { updatedAt: 20 }, + { + tabId: 'tab-1', + worktreeId: 'wt-1', + connectionId: null + } + ) + + // The pane is not retired here — only its live status becomes a sleeping record. Dropping + // these maps would replay history the user already cleared on the next heartbeat. + expect(store.getState().activityClearedAtByPaneKey['tab-1:leaf-1']).toBe(5_000) + expect(store.getState().manuallyUnreadTurnsByPaneKey['tab-1:leaf-1']).toBeGreaterThan(0) + }) + it('does not reuse Pi launch config when the session file identity changes', () => { const store = createTestStore() store.setState({ diff --git a/src/renderer/src/store/slices/agent-status-retention-actions.ts b/src/renderer/src/store/slices/agent-status-retention-actions.ts index ac7cb5b7b48..737a7231551 100644 --- a/src/renderer/src/store/slices/agent-status-retention-actions.ts +++ b/src/renderer/src/store/slices/agent-status-retention-actions.ts @@ -10,6 +10,7 @@ export function createAgentStatusRetentionActions( AgentStatusSlice, | 'retainAgents' | 'dismissRetainedAgent' + | 'dismissRetainedAgents' | 'dismissRetainedAgentsByWorktree' | 'pruneRetainedAgents' | 'clearRetentionSuppressedPaneKeys' @@ -69,6 +70,35 @@ export function createAgentStatusRetentionActions( }) }, + dismissRetainedAgents: (paneKeys) => { + set((s) => { + let next: Record | null = null + let nextSuppressed: Record | null = null + for (const paneKey of paneKeys) { + if (!(paneKey in (next ?? s.retainedAgentsByPaneKey))) { + continue + } + if (next === null) { + next = { ...s.retainedAgentsByPaneKey } + } + delete next[paneKey] + if (paneKey in s.agentStatusByPaneKey && !(paneKey in s.retentionSuppressedPaneKeys)) { + if (nextSuppressed === null) { + nextSuppressed = { ...s.retentionSuppressedPaneKeys } + } + nextSuppressed[paneKey] = true + } + } + if (next === null) { + return s + } + return { + retainedAgentsByPaneKey: next, + ...(nextSuppressed ? { retentionSuppressedPaneKeys: nextSuppressed } : {}) + } + }) + }, + dismissRetainedAgentsByWorktree: (worktreeId) => { const dismissedPaneKeys: string[] = [] set((s) => { diff --git a/src/renderer/src/store/slices/agent-status-slice-contract.ts b/src/renderer/src/store/slices/agent-status-slice-contract.ts index c1598f77a5a..9762207091b 100644 --- a/src/renderer/src/store/slices/agent-status-slice-contract.ts +++ b/src/renderer/src/store/slices/agent-status-slice-contract.ts @@ -14,6 +14,7 @@ import type { AgentProviderSessionMetadata, DropAgentStatusByTabPrefixOptions, DropAgentStatusByWorktreeOptions, + DropAgentStatusOptions, DropHibernatedAgentPaneOptions, RetainedAgentEntry, AllAgentSessionCaptureMode @@ -133,7 +134,7 @@ export type AgentStatusSlice = { clearTransientAgentStatuses: (connectionId: string, clearedAt: number) => void /** Remove a single entry AND suppress re-retention on its next disappearance (user-initiated teardown: X button, pane close). */ - dropAgentStatus: (paneKey: string) => void + dropAgentStatus: (paneKey: string, opts?: DropAgentStatusOptions) => void /** Remove all entries under a tab AND suppress re-retention for each (tab close — no rows may reappear). */ dropAgentStatusByTabPrefix: ( @@ -167,6 +168,9 @@ export type AgentStatusSlice = { /** Dismiss a retained entry by its paneKey. */ dismissRetainedAgent: (paneKey: string) => void + /** Dismiss several retained entries in one set (Activity "Clear completed"). */ + dismissRetainedAgents: (paneKeys: readonly string[]) => void + /** Dismiss all retained entries belonging to a worktree. */ dismissRetainedAgentsByWorktree: (worktreeId: string) => void diff --git a/src/renderer/src/store/slices/agent-status-worktree-drop-actions.ts b/src/renderer/src/store/slices/agent-status-worktree-drop-actions.ts index 4bafe7dc381..b7d88675f6b 100644 --- a/src/renderer/src/store/slices/agent-status-worktree-drop-actions.ts +++ b/src/renderer/src/store/slices/agent-status-worktree-drop-actions.ts @@ -8,6 +8,7 @@ import { retainedAgentEntryFromLive, shouldReplaceRetainedWithLive } from './agent-status-pane-key-tab-binding' +import { removePaneKeys } from './agent-status-pane-keyed-records' export function createAgentStatusWorktreeDropActions( runtime: AgentStatusRuntime @@ -72,21 +73,23 @@ export function createAgentStatusWorktreeDropActions( } } const retainedEvidenceKeys = new Set(retainedEvidence.keys()) - // Keep acknowledgement for completion evidence so a slept card does not turn bold again. - let nextAck = s.acknowledgedAgentsByPaneKey - const ackKeys = Object.keys(nextAck).filter( - (key) => - !retainedEvidenceKeys.has(key) && - (paneKeyMatchesAnyTabPrefix(key, tabPrefixes) || - liveKeySet.has(key) || - retainedKeySet.has(key)) + // Completion evidence keeps its read/clear state; every fully retired pane drops all three maps. + const activityStateKeys = new Set( + [ + ...Object.keys(s.acknowledgedAgentsByPaneKey), + ...Object.keys(s.activityClearedAtByPaneKey), + ...Object.keys(s.manuallyUnreadTurnsByPaneKey) + ].filter( + (key) => + !retainedEvidenceKeys.has(key) && + (paneKeyMatchesAnyTabPrefix(key, tabPrefixes) || + liveKeySet.has(key) || + retainedKeySet.has(key)) + ) ) - if (ackKeys.length > 0) { - nextAck = { ...nextAck } - for (const key of ackKeys) { - delete nextAck[key] - } - } + const nextAck = removePaneKeys(s.acknowledgedAgentsByPaneKey, activityStateKeys) + const nextClearedAt = removePaneKeys(s.activityClearedAtByPaneKey, activityStateKeys) + const nextManualUnread = removePaneKeys(s.manuallyUnreadTurnsByPaneKey, activityStateKeys) if ( liveKeys.length === 0 && launchConfigKeys.length === 0 && @@ -94,9 +97,18 @@ export function createAgentStatusWorktreeDropActions( retainedEvidence.size === 0 && !migrationUnsupported.changed ) { - return nextAck !== s.acknowledgedAgentsByPaneKey - ? { acknowledgedAgentsByPaneKey: nextAck } - : s + const cleanupPatch = { + ...(nextAck !== s.acknowledgedAgentsByPaneKey + ? { acknowledgedAgentsByPaneKey: nextAck } + : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}) + } + return Object.keys(cleanupPatch).length > 0 ? cleanupPatch : s } hadLive = liveKeys.length > 0 const nextLive = @@ -144,6 +156,12 @@ export function createAgentStatusWorktreeDropActions( ...(nextAck !== s.acknowledgedAgentsByPaneKey ? { acknowledgedAgentsByPaneKey: nextAck } : {}), + ...(nextClearedAt !== s.activityClearedAtByPaneKey + ? { activityClearedAtByPaneKey: nextClearedAt } + : {}), + ...(nextManualUnread !== s.manuallyUnreadTurnsByPaneKey + ? { manuallyUnreadTurnsByPaneKey: nextManualUnread } + : {}), agentStatusEpoch: hadLive || migrationUnsupported.changed ? s.agentStatusEpoch + 1 : s.agentStatusEpoch, sortEpoch: hadLive || migrationUnsupported.changed ? s.sortEpoch + 1 : s.sortEpoch diff --git a/src/renderer/src/store/slices/persisted-ui-write-baseline.test.ts b/src/renderer/src/store/slices/persisted-ui-write-baseline.test.ts index e38823f5344..300d0fbabc5 100644 --- a/src/renderer/src/store/slices/persisted-ui-write-baseline.test.ts +++ b/src/renderer/src/store/slices/persisted-ui-write-baseline.test.ts @@ -29,6 +29,8 @@ function makeBaseline(overrides: Partial = {}): Persis showDotfilesByWorktree: {}, filterRepoIds: [], acknowledgedAgentsByPaneKey: {}, + activityClearedAtByPaneKey: {}, + manuallyUnreadTurnsByPaneKey: {}, ...overrides } } @@ -95,6 +97,28 @@ describe('diffPersistedUIWriteFields', () => { }) }) +describe('manuallyUnreadTurnsByPaneKey write round-trip', () => { + it('is writer-owned and diffs by record content like the other pane-key records', () => { + expect(PERSISTED_UI_WRITE_BASELINE_FIELDS).toContain('manuallyUnreadTurnsByPaneKey') + const baseline = makeBaseline({ manuallyUnreadTurnsByPaneKey: { p1: 5 } }) + expect( + diffPersistedUIWriteFields( + makeBaseline({ manuallyUnreadTurnsByPaneKey: { p1: 5 } }), + baseline + ) + ).toEqual({}) + expect( + diffPersistedUIWriteFields( + makeBaseline({ manuallyUnreadTurnsByPaneKey: { p1: 7 } }), + baseline + ) + ).toEqual({ manuallyUnreadTurnsByPaneKey: { p1: 7 } }) + expect(persistedUIWriteFieldsToWireUpdate({ manuallyUnreadTurnsByPaneKey: { p1: 7 } })).toEqual( + { manuallyUnreadTurnsByPaneKey: { p1: 7 } } + ) + }) +}) + describe('persistedUIWriteFieldsToWireUpdate', () => { it('inverts showSleepingWorkspaces to the durable hide form', () => { expect(persistedUIWriteFieldsToWireUpdate({ showSleepingWorkspaces: true })).toEqual({ diff --git a/src/renderer/src/store/slices/persisted-ui-write-baseline.ts b/src/renderer/src/store/slices/persisted-ui-write-baseline.ts index d7a73c78b62..2d9da5d1ea4 100644 --- a/src/renderer/src/store/slices/persisted-ui-write-baseline.ts +++ b/src/renderer/src/store/slices/persisted-ui-write-baseline.ts @@ -29,6 +29,8 @@ export type PersistedUIWriteBaseline = { showDotfilesByWorktree: Record filterRepoIds: readonly string[] acknowledgedAgentsByPaneKey: Record + activityClearedAtByPaneKey: Record + manuallyUnreadTurnsByPaneKey: Record } // Why `satisfies Record<...>` rather than a keyof[] annotation: a plain `satisfies @@ -55,7 +57,9 @@ const PERSISTED_UI_WRITE_BASELINE_FIELD_SET = { alwaysShowDefaultBranchWorkspace: true, showDotfilesByWorktree: true, filterRepoIds: true, - acknowledgedAgentsByPaneKey: true + acknowledgedAgentsByPaneKey: true, + activityClearedAtByPaneKey: true, + manuallyUnreadTurnsByPaneKey: true } satisfies Record export const PERSISTED_UI_WRITE_BASELINE_FIELDS = Object.keys( @@ -95,7 +99,12 @@ function writeFieldEqual(field: keyof PersistedUIWriteBaseline, a: unknown, b: u if (field === 'filterRepoIds') { return stringArrayEqual(a as readonly string[], b as readonly string[]) } - if (field === 'showDotfilesByWorktree' || field === 'acknowledgedAgentsByPaneKey') { + if ( + field === 'showDotfilesByWorktree' || + field === 'acknowledgedAgentsByPaneKey' || + field === 'activityClearedAtByPaneKey' || + field === 'manuallyUnreadTurnsByPaneKey' + ) { return shallowRecordEqual( a as Record | undefined, b as Record | undefined diff --git a/src/renderer/src/store/slices/retired-terminal-tab-state-sweep.ts b/src/renderer/src/store/slices/retired-terminal-tab-state-sweep.ts index cfdefcf87e0..9985a0f7406 100644 --- a/src/renderer/src/store/slices/retired-terminal-tab-state-sweep.ts +++ b/src/renderer/src/store/slices/retired-terminal-tab-state-sweep.ts @@ -58,7 +58,8 @@ export function sweepRetiredTerminalTabState( export function buildRetiredTerminalTabStateSweepPatch( state: RetiredTerminalTabSweepState, tabIds: readonly string[], - worktreeId?: string | null + worktreeId?: string | null, + opts?: { preserveActivityClearedState?: boolean } ): Partial | null { if (tabIds.length === 0) { return null @@ -73,7 +74,10 @@ export function buildRetiredTerminalTabStateSweepPatch( swept, tabId, retireAgentPaneAuthorityAliasesByOwnerTab(tabId), - worktreeId ? { worktreeId } : undefined + { + ...(worktreeId ? { worktreeId } : {}), + ...(opts?.preserveActivityClearedState ? { preserveActivityClearedState: true } : {}) + } ) const foreground = buildPaneForegroundAgentTabPrefixClearPatch( swept.paneForegroundAgentByPaneKey, diff --git a/src/renderer/src/store/slices/runtime-status-recheck.ts b/src/renderer/src/store/slices/runtime-status-recheck.ts index 65635157a18..303e5b4318d 100644 --- a/src/renderer/src/store/slices/runtime-status-recheck.ts +++ b/src/renderer/src/store/slices/runtime-status-recheck.ts @@ -19,10 +19,7 @@ type RecheckState = { type RuntimeStatusStore = { runtimeEnvironments: readonly { id: string }[] - setRuntimeEnvironmentStatus: ( - environmentId: string, - status: RuntimeEnvironmentStatus - ) => void + setRuntimeEnvironmentStatus: (environmentId: string, status: RuntimeEnvironmentStatus) => void } const rechecks = new Map() diff --git a/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts b/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts index d0104d8d51b..a9a6f7aed0a 100644 --- a/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts +++ b/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts @@ -49,6 +49,22 @@ beforeEach(() => { mocks.toastError.mockReset() }) +describe('sidebar reveal actions', () => { + it('switch the sidebar body back to Spaces so the worktree list can consume the reveal', () => { + const store = createUIStore() + store.getState().setSidebarBody('agents') + + store.getState().revealWorktreeInSidebar('wt-1', { highlight: true }) + expect(store.getState().sidebarBody).toBe('workspaces') + expect(store.getState().pendingRevealWorktree?.worktreeId).toBe('wt-1') + + store.getState().setSidebarBody('agents') + store.getState().revealSidebarRow('repo:r1') + expect(store.getState().sidebarBody).toBe('workspaces') + expect(store.getState().pendingRevealSidebarRow?.rowKey).toBe('repo:r1') + }) +}) + describe('createUISlice hydratePersistedUI', () => { it('defaults persisted right sidebar visibility to open', () => { expect(getDefaultUIState().rightSidebarOpen).toBe(true) @@ -148,10 +164,10 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().activeView).toBe('terminal') }) - it('drops a persisted activity view when experimental activity is disabled', () => { + it('drops a persisted activity view when the Agents sidebar is hidden', () => { const store = createUIStore() store.setState({ - settings: { experimentalActivity: false } as AppState['settings'] + settings: { showAgentsSidebar: false } as AppState['settings'] }) store.getState().hydratePersistedUI(makePersistedUI({ activeView: 'activity' }), 'startup') @@ -159,10 +175,21 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().activeView).toBe('terminal') }) - it('restores a persisted activity view when experimental activity is enabled', () => { + it('keeps a persisted activity view when the settings fetch failed', () => { + // A failed window.api.settings.get() leaves settings null; downgrading here would let the + // persisted-UI writer overwrite the saved view with terminal. + const store = createUIStore() + store.setState({ settings: null as unknown as AppState['settings'] }) + + store.getState().hydratePersistedUI(makePersistedUI({ activeView: 'activity' }), 'startup') + + expect(store.getState().activeView).toBe('activity') + }) + + it('restores a persisted activity view when the Agents sidebar is shown', () => { const store = createUIStore() store.setState({ - settings: { experimentalActivity: true } as AppState['settings'] + settings: { showAgentsSidebar: true } as AppState['settings'] }) store.getState().hydratePersistedUI(makePersistedUI({ activeView: 'activity' }), 'startup') diff --git a/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts b/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts index a591ce4c40a..642297bb109 100644 --- a/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts +++ b/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts @@ -518,6 +518,94 @@ describe('createUISlice hydratePersistedUI', () => { } }) + it('keeps agents scope filter array identities stable across unchanged re-hydrations', () => { + const store = createUIStore() + const persisted = makePersistedUI({ + agentsVisibleHostIds: ['ssh:devbox'], + agentsFilterRepoIds: ['repo-1'] + }) + + store.getState().hydratePersistedUI(persisted) + const firstHostIds = store.getState().agentsVisibleHostIds + const firstRepoIds = store.getState().agentsFilterRepoIds + + // Why identity (toBe): sync hydration fires on every ui:changed broadcast, and a + // fresh array would invalidate the agents-view scope memos each time. + store.getState().hydratePersistedUI(makePersistedUI({ ...persisted })) + expect(store.getState().agentsVisibleHostIds).toBe(firstHostIds) + expect(store.getState().agentsFilterRepoIds).toBe(firstRepoIds) + + store + .getState() + .hydratePersistedUI(makePersistedUI({ ...persisted, agentsVisibleHostIds: ['ssh:otherbox'] })) + expect(store.getState().agentsVisibleHostIds).toEqual(['ssh:otherbox']) + expect(store.getState().agentsFilterRepoIds).toBe(firstRepoIds) + }) + + it('hydrates agents view preferences with safe defaults for absent fields', () => { + const store = createUIStore() + + store.getState().hydratePersistedUI(makePersistedUI({})) + + expect(store.getState().agentsVisibleHostIds).toBeNull() + expect(store.getState().agentsFilterRepoIds).toEqual([]) + expect(store.getState().agentsShowChildAgents).toBe(false) + expect(store.getState().agentsCompactMode).toBe(true) + }) + + it('sanitizes malformed agents repo filters before the repo catalog loads', () => { + const store = createUIStore() + + expect(() => + store.getState().hydratePersistedUI( + makePersistedUI({ + agentsFilterRepoIds: 'repo-1' as unknown as PersistedUIState['agentsFilterRepoIds'] + }) + ) + ).not.toThrow() + expect(store.getState().agentsFilterRepoIds).toEqual([]) + + store.getState().hydratePersistedUI( + makePersistedUI({ + agentsFilterRepoIds: [ + 'repo-1', + 42, + 'missing' + ] as unknown as PersistedUIState['agentsFilterRepoIds'] + }) + ) + expect(store.getState().agentsFilterRepoIds).toEqual(['repo-1', 'missing']) + }) + + it('sanitizes and prunes agents repo filters against a loaded repo catalog', () => { + const store = createUIStore() + store.setState({ + repos: [ + { + id: 'repo-1', + path: '/tmp/repo-1', + displayName: 'Repo 1', + badgeColor: 'gray', + addedAt: 1, + kind: 'git' + } + ] + }) + + expect(() => + store.getState().hydratePersistedUI( + makePersistedUI({ + agentsFilterRepoIds: [ + 'repo-1', + null, + 'missing' + ] as unknown as PersistedUIState['agentsFilterRepoIds'] + }) + ) + ).not.toThrow() + expect(store.getState().agentsFilterRepoIds).toEqual(['repo-1']) + }) + it('prunes acknowledgedAgentsByPaneKey entries older than the 7-day TTL during hydration', () => { // HYDRATE_MAX_AGE_MS lives in src/renderer/src/store/slices/ui.ts and matches // the constant in src/main/agent-hooks/server.ts. @@ -693,3 +781,33 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().agentActivityDisplayMode).toBe('compact') }) }) + +describe('createUISlice hydratePersistedUI manual unread turns', () => { + it('restores manual unread stamps and prunes malformed or stale ones', () => { + const store = createUIStore() + const fresh = Date.now() - 60_000 + const stale = Date.now() - 8 * 24 * 60 * 60 * 1000 + + store.getState().hydratePersistedUI( + makePersistedUI({ + manuallyUnreadTurnsByPaneKey: { + 'tab-a:1': fresh, + 'tab-b:1': stale, + 'tab-c:1': 'bogus' as unknown as number + } + }) + ) + + expect(store.getState().manuallyUnreadTurnsByPaneKey).toEqual({ 'tab-a:1': fresh }) + }) + + it('hydrates to an empty record when the field is absent from an older profile', () => { + const store = createUIStore() + + store + .getState() + .hydratePersistedUI(makePersistedUI({ manuallyUnreadTurnsByPaneKey: undefined })) + + expect(store.getState().manuallyUnreadTurnsByPaneKey).toEqual({}) + }) +}) diff --git a/src/renderer/src/store/slices/ui-page-navigation.test.ts b/src/renderer/src/store/slices/ui-page-navigation.test.ts index eb1da4f6dd4..9a355223c27 100644 --- a/src/renderer/src/store/slices/ui-page-navigation.test.ts +++ b/src/renderer/src/store/slices/ui-page-navigation.test.ts @@ -298,6 +298,35 @@ describe('createUISlice settings navigation', () => { expect(store.getState().activeView).toBe('tasks') }) + it('returns to the graduated Agents view after visiting settings', () => { + const store = createUIStore() + + store.setState({ + settings: { showAgentsSidebar: true } as unknown as AppState['settings'] + } as unknown as Partial) + store.getState().openActivityPage() + expect(store.getState().activeView).toBe('activity') + + store.getState().openSettingsPage() + store.getState().closeSettingsPage() + + expect(store.getState().activeView).toBe('activity') + }) + + it('falls back to terminal when closing settings with the Agents surfaces hidden', () => { + const store = createUIStore() + + store.setState({ + settings: { showAgentsSidebar: false } as unknown as AppState['settings'], + activeView: 'settings', + previousViewBeforeSettings: 'activity' + } as unknown as Partial) + + store.getState().closeSettingsPage() + + expect(store.getState().activeView).toBe('terminal') + }) + it('clears transient settings search when opening settings', () => { const store = createUIStore() diff --git a/src/renderer/src/store/slices/ui/ui-slice-activity-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-activity-actions.ts new file mode 100644 index 00000000000..0ae704a5702 --- /dev/null +++ b/src/renderer/src/store/slices/ui/ui-slice-activity-actions.ts @@ -0,0 +1,155 @@ +import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' +import { + collectAcknowledgedAgentNotificationId, + latestAgentTurnTimestamp, + resolvePaneKeyWorktreeIdFromTabs, + usableTimestamp +} from './ui-slice-agent-notification-acknowledgement' + +type ActivityActions = Pick< + UISlice, + | 'acknowledgedAgentsByPaneKey' + | 'acknowledgeAgents' + | 'unacknowledgeAgents' + | 'activityClearedAtByPaneKey' + | 'applyActivityClearedAt' + | 'manuallyUnreadTurnsByPaneKey' + | 'clearManuallyUnreadTurns' +> + +export function createUiActivityActions(set: UISliceSet, _get: UISliceGet): ActivityActions { + return { + acknowledgedAgentsByPaneKey: {}, + acknowledgeAgents: (paneKeys) => { + const notificationIdsToDismiss = new Set() + set((s) => { + if (paneKeys.length === 0) { + return s + } + const now = Date.now() + const migrationUnsupported = Object.values(s.migrationUnsupportedByPtyId ?? {}) + let next: Record | null = null + let nextUnreadCompletions: Record | null = null + for (const key of paneKeys) { + if (s.unreadAgentCompletionPanes[key]) { + nextUnreadCompletions ??= { ...s.unreadAgentCompletionPanes } + delete nextUnreadCompletions[key] + } + const prev = s.acknowledgedAgentsByPaneKey[key] ?? 0 + let stamp = now + const liveEntry = s.agentStatusByPaneKey?.[key] + if (liveEntry) { + collectAcknowledgedAgentNotificationId({ + ids: notificationIdsToDismiss, + worktreeId: resolvePaneKeyWorktreeIdFromTabs(s, key) ?? liveEntry.worktreeId, + paneKey: key, + stateStartedAt: liveEntry.stateStartedAt, + previousAckAt: prev + }) + stamp = Math.max(stamp, latestAgentTurnTimestamp(liveEntry)) + } + const retained = s.retainedAgentsByPaneKey?.[key] + if (retained) { + collectAcknowledgedAgentNotificationId({ + ids: notificationIdsToDismiss, + worktreeId: retained.worktreeId, + paneKey: key, + stateStartedAt: retained.entry.stateStartedAt, + previousAckAt: prev + }) + stamp = Math.max(stamp, latestAgentTurnTimestamp(retained.entry)) + } + for (const unsupported of migrationUnsupported) { + if (unsupported.paneKey === key) { + stamp = Math.max(stamp, usableTimestamp(unsupported.updatedAt)) + } + } + if (prev < stamp) { + next ??= { ...s.acknowledgedAgentsByPaneKey } + next[key] = stamp + } + } + let nextManual: Record | null = null + for (const key of paneKeys) { + if (s.manuallyUnreadTurnsByPaneKey[key] !== undefined) { + nextManual ??= { ...s.manuallyUnreadTurnsByPaneKey } + delete nextManual[key] + } + } + if (!next && !nextUnreadCompletions && !nextManual) { + return s + } + return { + ...(next ? { acknowledgedAgentsByPaneKey: next } : {}), + ...(nextUnreadCompletions ? { unreadAgentCompletionPanes: nextUnreadCompletions } : {}), + ...(nextManual ? { manuallyUnreadTurnsByPaneKey: nextManual } : {}) + } + }) + const ids = [...notificationIdsToDismiss] + if (ids.length > 0 && typeof window !== 'undefined') { + void window.api?.notifications?.dismiss?.(ids) + } + }, + unacknowledgeAgents: (paneKeys) => + set((s) => { + if (paneKeys.length === 0) { + return s + } + let next: Record | null = null + let nextManual: Record | null = null + for (const key of paneKeys) { + if (s.acknowledgedAgentsByPaneKey[key] !== undefined) { + next ??= { ...s.acknowledgedAgentsByPaneKey } + delete next[key] + } + const turnTimestamp = + s.agentStatusByPaneKey?.[key]?.stateStartedAt ?? + s.retainedAgentsByPaneKey?.[key]?.entry.stateStartedAt + if ( + turnTimestamp !== undefined && + s.manuallyUnreadTurnsByPaneKey[key] !== turnTimestamp + ) { + nextManual ??= { ...s.manuallyUnreadTurnsByPaneKey } + nextManual[key] = turnTimestamp + } + } + if (!next && !nextManual) { + return s + } + return { + ...(next ? { acknowledgedAgentsByPaneKey: next } : {}), + ...(nextManual ? { manuallyUnreadTurnsByPaneKey: nextManual } : {}) + } + }), + manuallyUnreadTurnsByPaneKey: {}, + clearManuallyUnreadTurns: (paneKeys) => + set((s) => { + let next: Record | null = null + for (const key of paneKeys) { + if (s.manuallyUnreadTurnsByPaneKey[key] !== undefined) { + next ??= { ...s.manuallyUnreadTurnsByPaneKey } + delete next[key] + } + } + return next ? { manuallyUnreadTurnsByPaneKey: next } : s + }), + activityClearedAtByPaneKey: {}, + applyActivityClearedAt: (patch) => + set((s) => { + let next: Record | null = null + for (const [key, value] of Object.entries(patch)) { + const previous = s.activityClearedAtByPaneKey[key] + if (value === null ? previous === undefined : previous === value) { + continue + } + next ??= { ...s.activityClearedAtByPaneKey } + if (value === null) { + delete next[key] + } else { + next[key] = value + } + } + return next ? { activityClearedAtByPaneKey: next } : s + }) + } +} diff --git a/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts index ebd0f016b3f..9c34d49eb38 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-agent-actions.ts @@ -5,12 +5,7 @@ import { resolveRunningAgentSendTarget } from '../../../lib/running-agent-targets' import { translate } from '@/i18n/i18n' -import { - collectAcknowledgedAgentNotificationId, - latestAgentTurnTimestamp, - resolvePaneKeyWorktreeIdFromTabs, - usableTimestamp -} from './ui-slice-agent-notification-acknowledgement' +import { createUiActivityActions } from './ui-slice-activity-actions' let agentSendTargetModeInstanceCounter = 0 @@ -39,8 +34,13 @@ export function createUiAgentActions( | 'acknowledgedAgentsByPaneKey' | 'acknowledgeAgents' | 'unacknowledgeAgents' + | 'activityClearedAtByPaneKey' + | 'applyActivityClearedAt' + | 'manuallyUnreadTurnsByPaneKey' + | 'clearManuallyUnreadTurns' > { return { + ...createUiActivityActions(set, get), sidebarOpen: true, sidebarWidth: 280, toggleSidebar: () => set((s) => ({ sidebarOpen: !s.sidebarOpen })), @@ -209,97 +209,6 @@ export function createUiAgentActions( ) get().closeAgentSendPopoverTargetMode(mode.id, mode.instanceId) return true - }, - - acknowledgedAgentsByPaneKey: {}, - acknowledgeAgents: (paneKeys) => { - const notificationIdsToDismiss = new Set() - set((s) => { - if (paneKeys.length === 0) { - return s - } - const now = Date.now() - const migrationUnsupported = Object.values(s.migrationUnsupportedByPtyId ?? {}) - // Why: only reallocate if an ack advances; compare prev | null = null - // Why: one ack, two records — leaving the completion marker set keeps the tab dot, - // the ⌘J row and the floating-workspace dot lit with nothing left to read. - let nextUnreadCompletions: Record | null = null - for (const key of paneKeys) { - if (s.unreadAgentCompletionPanes[key]) { - if (nextUnreadCompletions === null) { - nextUnreadCompletions = { ...s.unreadAgentCompletionPanes } - } - delete nextUnreadCompletions[key] - } - const prev = s.acknowledgedAgentsByPaneKey[key] ?? 0 - // Why not plain Date.now(): a remote/SSH execution host can stamp a turn ahead of this clock, - // and every unread rule is `ackAt < turnTimestamp`. A behind-the-turn ack can never clear the - // row, so its auto-ack effect re-fires on each new millisecond forever (React #185). - let stamp = now - const liveEntry = s.agentStatusByPaneKey?.[key] - if (liveEntry) { - collectAcknowledgedAgentNotificationId({ - ids: notificationIdsToDismiss, - worktreeId: resolvePaneKeyWorktreeIdFromTabs(s, key) ?? liveEntry.worktreeId, - paneKey: key, - stateStartedAt: liveEntry.stateStartedAt, - previousAckAt: prev - }) - stamp = Math.max(stamp, latestAgentTurnTimestamp(liveEntry)) - } - const retained = s.retainedAgentsByPaneKey?.[key] - if (retained) { - collectAcknowledgedAgentNotificationId({ - ids: notificationIdsToDismiss, - worktreeId: retained.worktreeId, - paneKey: key, - stateStartedAt: retained.entry.stateStartedAt, - previousAckAt: prev - }) - stamp = Math.max(stamp, latestAgentTurnTimestamp(retained.entry)) - } - for (const unsupported of migrationUnsupported) { - // Why: Activity synthesizes a blocked row from this entry, stamped by the pane's host like any turn. - if (unsupported.paneKey === key) { - stamp = Math.max(stamp, usableTimestamp(unsupported.updatedAt)) - } - } - if (prev < stamp) { - if (next === null) { - next = { ...s.acknowledgedAgentsByPaneKey } - } - next[key] = stamp - } - } - if (!next && !nextUnreadCompletions) { - return s - } - return { - ...(next ? { acknowledgedAgentsByPaneKey: next } : {}), - ...(nextUnreadCompletions ? { unreadAgentCompletionPanes: nextUnreadCompletions } : {}) - } - }) - const notificationIds = [...notificationIdsToDismiss] - if (notificationIds.length > 0 && typeof window !== 'undefined') { - void window.api?.notifications?.dismiss?.(notificationIds) - } - }, - unacknowledgeAgents: (paneKeys) => - set((s) => { - if (paneKeys.length === 0) { - return s - } - let next: Record | null = null - for (const key of paneKeys) { - if (s.acknowledgedAgentsByPaneKey[key] !== undefined) { - if (next === null) { - next = { ...s.acknowledgedAgentsByPaneKey } - } - delete next[key] - } - } - return next ? { acknowledgedAgentsByPaneKey: next } : s - }) + } } } diff --git a/src/renderer/src/store/slices/ui/ui-slice-contract-core.ts b/src/renderer/src/store/slices/ui/ui-slice-contract-core.ts index cd2af73b123..2506e5dd071 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-contract-core.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-contract-core.ts @@ -133,6 +133,12 @@ export type UISliceCore = { acknowledgedAgentsByPaneKey: Record acknowledgeAgents: (paneKeys: string[]) => void unacknowledgeAgents: (paneKeys: string[]) => void + /** Per-pane cutoffs used to hide activity entries cleared by the user. */ + activityClearedAtByPaneKey: Record + applyActivityClearedAt: (patch: Record) => void + /** Session-local protection for turns explicitly marked unread. */ + manuallyUnreadTurnsByPaneKey: Record + clearManuallyUnreadTurns: (paneKeys: string[]) => void activeView: TopLevelView previousViewBeforeTasks: Exclude previousViewBeforeSettings: Exclude diff --git a/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts b/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts index 97ab0cb484b..2e764e5cd79 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts @@ -22,6 +22,9 @@ import type { PersistedUIWriteBaseline } from '../persisted-ui-write-baseline' import type { UISliceCore } from './ui-slice-contract-core' export type UISlicePreferences = { + /** Which list the sidebar body shows. Navigator-only; does not change the active view. */ + sidebarBody: 'workspaces' | 'agents' + setSidebarBody: (body: UISlicePreferences['sidebarBody']) => void groupBy: 'none' | 'workspace-status' | 'repo' | 'pr-status' setGroupBy: (g: UISlicePreferences['groupBy']) => void sortBy: 'name' | 'smart' | 'recent' | 'repo' | 'manual' @@ -59,6 +62,15 @@ export type UISlicePreferences = { toggleShowDotfilesForWorktree: (worktreeId: string) => void filterRepoIds: readonly string[] setFilterRepoIds: (ids: readonly string[]) => void + /** Agents-view scope filters, independent from workspace navigation filters. */ + agentsVisibleHostIds: VisibleWorkspaceHostIds + setAgentsVisibleHostIds: (ids: VisibleWorkspaceHostIds) => void + agentsFilterRepoIds: readonly string[] + setAgentsFilterRepoIds: (ids: readonly string[]) => void + agentsShowChildAgents: boolean + setAgentsShowChildAgents: (v: boolean) => void + agentsCompactMode: boolean + setAgentsCompactMode: (v: boolean) => void collapsedGroups: Set toggleCollapsedGroup: (key: string) => void worktreeCardProperties: WorktreeCardProperty[] diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts index 7902d1bbba8..64858db11a0 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts @@ -1,4 +1,6 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' +import type { AppState } from '../../types' +import type { PersistedUIState } from '../../../../../shared/persisted-ui-state-types' import { normalizeRightSidebarRoute } from '../../right-sidebar-route' import { applyManualRepoOrder, @@ -7,7 +9,8 @@ import { import { normalizeWorkspaceCleanupBrowseState } from '../../../../../shared/workspace-cleanup-browse-state' import { normalizeExecutionHostScope, - normalizeExecutionHostOrder + normalizeExecutionHostOrder, + normalizeVisibleExecutionHostIds } from '../../../../../shared/execution-host' import { normalizeFeatureInteractions } from '../../../../../shared/feature-interactions' import { normalizeContextualTourIds } from '../../../../../shared/contextual-tours' @@ -46,7 +49,7 @@ import { import { hydrateTrustedOrcaHooks, normalizeHydratedVisibleWorkspaceHostIds, - sanitizeAcknowledgedAgentsByPaneKey, + preserveStringArrayIdentity, sanitizeHydratedActiveView, sanitizePersistedRepoIds, sanitizeShowDotfilesByWorktree, @@ -56,7 +59,7 @@ import { migrateStatusBarItems, clampPetSize } from './ui-slice-hydration-sanitizers' -import { sanitizeTaskResumeState } from './ui-slice-hydration-values' +import { hydrateAgentReadState, sanitizeTaskResumeState } from './ui-slice-hydration-values' const MAX_LEFT_SIDEBAR_WIDTH = 500 const MAX_RIGHT_SIDEBAR_WIDTH = 4000 @@ -66,6 +69,28 @@ const DEFAULT_ON_MINIMAX_STATUS_BAR_ITEM: StatusBarItem = 'minimax' const DEFAULT_ON_ANTIGRAVITY_STATUS_BAR_ITEM: StatusBarItem = 'antigravity' const DEFAULT_ON_GROK_STATUS_BAR_ITEM: StatusBarItem = 'grok' +function hydrateStatusBarItems(ui: PersistedUIState): StatusBarItem[] { + let items = migrateStatusBarItems(ui.statusBarItems) + const defaults = [ + ['_portsStatusBarDefaultAdded', DEFAULT_ON_PORTS_STATUS_BAR_ITEM], + ['_kimiStatusBarDefaultAdded', DEFAULT_ON_KIMI_STATUS_BAR_ITEM], + ['_minimaxStatusBarDefaultAdded', DEFAULT_ON_MINIMAX_STATUS_BAR_ITEM], + ['_antigravityStatusBarDefaultAdded', DEFAULT_ON_ANTIGRAVITY_STATUS_BAR_ITEM], + ['_grokStatusBarDefaultAdded', DEFAULT_ON_GROK_STATUS_BAR_ITEM] + ] as const + for (const [flag, item] of defaults) { + if (!ui[flag] && !items.includes(item)) { + items = [...items, item] + } + } + if (typeof window !== 'undefined' && defaults.some(([flag]) => !ui[flag])) { + window.api.ui + .set({ statusBarItems: items, ...Object.fromEntries(defaults.map(([flag]) => [flag, true])) }) + .catch(console.error) + } + return items +} + export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Partial { return { hydratePersistedUI: (ui, source = 'sync') => @@ -75,6 +100,7 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par const validRepoIds = new Set(s.repos.map((repo) => repo.id)) const validRepoHostIdentities = new Set(s.repos.map(getRepoHostIdentity)) const persistedFilterRepoIds = sanitizePersistedRepoIds(ui.filterRepoIds) + const persistedAgentsFilterRepoIds = sanitizePersistedRepoIds(ui.agentsFilterRepoIds) // Why: pre-rename builds used sidekick* keys; read as fallback only so new pet* writes win after upgrade. const customPets = Array.isArray(ui.customPets) ? ui.customPets @@ -84,46 +110,7 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par const petId = ui.petId ?? ui.sidekickId // Migration: one-shot old-'recent'→'smart' runs in main (_sortBySmartMigrated), not here, so a deliberate 'recent' choice survives restart. const sortBy = ui.sortBy - const migratedStatusBarItems = migrateStatusBarItems(ui.statusBarItems) - const statusBarItemsWithPorts: StatusBarItem[] = - ui._portsStatusBarDefaultAdded || migratedStatusBarItems.includes('ports') - ? migratedStatusBarItems - : [...migratedStatusBarItems, DEFAULT_ON_PORTS_STATUS_BAR_ITEM] - const statusBarItems: StatusBarItem[] = - ui._kimiStatusBarDefaultAdded || statusBarItemsWithPorts.includes('kimi') - ? statusBarItemsWithPorts - : [...statusBarItemsWithPorts, DEFAULT_ON_KIMI_STATUS_BAR_ITEM] - const statusBarItemsWithMiniMax: StatusBarItem[] = - ui._minimaxStatusBarDefaultAdded || statusBarItems.includes('minimax') - ? statusBarItems - : [...statusBarItems, DEFAULT_ON_MINIMAX_STATUS_BAR_ITEM] - const statusBarItemsWithAntigravity: StatusBarItem[] = - ui._antigravityStatusBarDefaultAdded || statusBarItemsWithMiniMax.includes('antigravity') - ? statusBarItemsWithMiniMax - : [...statusBarItemsWithMiniMax, DEFAULT_ON_ANTIGRAVITY_STATUS_BAR_ITEM] - const statusBarItemsWithGrok: StatusBarItem[] = - ui._grokStatusBarDefaultAdded || statusBarItemsWithAntigravity.includes('grok') - ? statusBarItemsWithAntigravity - : [...statusBarItemsWithAntigravity, DEFAULT_ON_GROK_STATUS_BAR_ITEM] - if ( - (!ui._portsStatusBarDefaultAdded || - !ui._kimiStatusBarDefaultAdded || - !ui._minimaxStatusBarDefaultAdded || - !ui._antigravityStatusBarDefaultAdded || - !ui._grokStatusBarDefaultAdded) && - typeof window !== 'undefined' - ) { - window.api.ui - .set({ - statusBarItems: statusBarItemsWithGrok, - _portsStatusBarDefaultAdded: true, - _kimiStatusBarDefaultAdded: true, - _minimaxStatusBarDefaultAdded: true, - _antigravityStatusBarDefaultAdded: true, - _grokStatusBarDefaultAdded: true - }) - .catch(console.error) - } + const statusBarItemsWithGrok = hydrateStatusBarItems(ui) const rightSidebarRoute = normalizeRightSidebarRoute( ui.rightSidebarTab, ui.rightSidebarExplorerView @@ -183,6 +170,18 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par validRepoIds.size === 0 ? persistedFilterRepoIds : persistedFilterRepoIds.filter((repoId) => validRepoIds.has(repoId)), + agentsVisibleHostIds: preserveStringArrayIdentity( + s.agentsVisibleHostIds, + normalizeVisibleExecutionHostIds(ui.agentsVisibleHostIds) + ), + agentsFilterRepoIds: preserveStringArrayIdentity( + s.agentsFilterRepoIds, + validRepoIds.size === 0 + ? persistedAgentsFilterRepoIds + : persistedAgentsFilterRepoIds.filter((repoId) => validRepoIds.has(repoId)) + ), + agentsShowChildAgents: ui.agentsShowChildAgents === true, + agentsCompactMode: ui.agentsCompactMode !== false, collapsedGroups: new Set(ui.collapsedGroups ?? []), uiZoomLevel: ui.uiZoomLevel ?? 0, editorFontZoomLevel: ui.editorFontZoomLevel ?? 0, @@ -262,10 +261,7 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par ui.usagePercentageDisplayChangeNoticeDismissed === true, // Why: default false so existing users still see the CTA; only explicit dismissal persists true. usageEmptyStateDismissed: ui.usageEmptyStateDismissed === true, - // Why: stale acks are inert (paneKey reuse beats them via stateStartedAt); sanitizer bounds growth past HYDRATE_MAX_AGE_MS. - acknowledgedAgentsByPaneKey: sanitizeAcknowledgedAgentsByPaneKey( - ui.acknowledgedAgentsByPaneKey - ), + ...hydrateAgentReadState(ui), workspaceCleanupDismissals: sanitizeWorkspaceCleanupDismissals( ui.workspaceCleanup?.dismissals ), @@ -280,7 +276,7 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par // Why: restore only on startup; on 'sync' broadcasts it would clobber the window's current per-window view. activeView: source === 'startup' - ? sanitizeHydratedActiveView(ui.activeView, s.settings?.experimentalActivity === true) + ? sanitizeHydratedActiveView(ui.activeView, s.settings) : s.activeView, persistedUIReady: true } @@ -337,7 +333,7 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par ...hydrated, persistedUIWriteBaseline: nextWriteBaseline, persistedUIWriteBaselineGeneration: nextWriteBaselineGeneration - } + } as Partial }) } } diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts index 8d81fdd4d53..edb79511cd8 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts @@ -15,11 +15,26 @@ import { normalizeExecutionHostScope } from '../../../../../shared/execution-host' import { persistedUIValuesEqual } from '../../../../../shared/persisted-ui-equality' +// Pure predicate over GlobalSettings; safe to share with the store layer. +import { shouldShowAgentsSidebar } from '@/components/sidebar/agents-sidebar-visibility' import { DEFAULT_STATUS_BAR_ITEMS } from '../../../../../shared/constants' import type { UISlice } from './ui-slice-contract' const MIN_SIDEBAR_WIDTH = 220 const HYDRATE_MAX_AGE_MS = 7 * 24 * 60 * 60 * 1000 + +export function preserveStringArrayIdentity( + current: readonly T[] | null, + next: T[] | null +): T[] | null { + if (!current || !next) { + return next + } + return current.length === next.length && current.every((value, index) => value === next[index]) + ? (current as T[]) + : next +} + export function isPlainPersistedRecord(value: unknown): value is Record { return Boolean(value) && typeof value === 'object' && !Array.isArray(value) } @@ -91,11 +106,14 @@ export function sanitizePersistedSidebarWidth( return Math.min(maxWidth, Math.max(MIN_SIDEBAR_WIDTH, width)) } -export function sanitizeAcknowledgedAgentsByPaneKey(value: unknown): Record { +export function sanitizePaneKeyTimestampRecord( + value: unknown, + maxAgeMs: number = HYDRATE_MAX_AGE_MS +): Record { if (value === null || typeof value !== 'object' || Array.isArray(value)) { return {} } - const cutoff = Date.now() - HYDRATE_MAX_AGE_MS + const cutoff = Date.now() - maxAgeMs const out: Record = {} for (const [key, ackAt] of Object.entries(value as Record)) { if (!isSafePersistedRecordKey(key)) { @@ -109,6 +127,16 @@ export function sanitizeAcknowledgedAgentsByPaneKey(value: unknown): Record { + return sanitizePaneKeyTimestampRecord(value, 2 * HYDRATE_MAX_AGE_MS) +} + export function sanitizeWorkspaceCleanupDismissals( value: unknown ): Record { @@ -147,14 +175,16 @@ export function sanitizeWorkspaceCleanupDismissals( export function sanitizeHydratedActiveView( value: PersistedUIState['activeView'], - experimentalActivityEnabled: boolean + settings: Parameters[0] ): TopLevelView { // Why: older data (pre-activeView) or a view a different build doesn't have falls back to terminal rather than rendering nothing. if (!isTopLevelView(value)) { return 'terminal' } - // Why: activity is hidden when its setting is off, so gate only it (mobile/automations stay functional when hidden). - if (value === 'activity' && !experimentalActivityEnabled) { + // Why: activity is hidden when its entry points are, so gate only it (mobile/automations stay functional when hidden). + // Why the null check: a failed settings fetch is not an opt-out. Downgrading on absent settings + // would let the persisted-UI writer overwrite the user's saved `activity` with `terminal`. + if (value === 'activity' && settings && !shouldShowAgentsSidebar(settings)) { return 'terminal' } return value diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-values.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-values.ts index 7bb8376fb27..5b891d59b01 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-values.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-values.ts @@ -4,6 +4,12 @@ import type { FeatureInteractionState } from '../../../../../shared/feature-inte import type { ContextualTourId } from '../../../../../shared/contextual-tours' import { normalizeFeatureInteractions } from '../../../../../shared/feature-interactions' import { normalizeContextualTourIds } from '../../../../../shared/contextual-tours' +import type { UISlice } from './ui-slice-contract' +import { + sanitizeAcknowledgedAgentsByPaneKey, + sanitizeActivityClearedAtByPaneKey, + sanitizePaneKeyTimestampRecord +} from './ui-slice-hydration-sanitizers' const VALID_TASK_PRESETS = new Set([ 'all', @@ -132,3 +138,19 @@ export function mergeContextualTourSeenIds( } return [...merged] } + +/** Stale acks/marks are inert (paneKey reuse beats them via stateStartedAt); the sanitizers only bound growth past HYDRATE_MAX_AGE_MS. */ +export function hydrateAgentReadState( + ui: PersistedUIState +): Pick< + UISlice, + 'acknowledgedAgentsByPaneKey' | 'activityClearedAtByPaneKey' | 'manuallyUnreadTurnsByPaneKey' +> { + return { + acknowledgedAgentsByPaneKey: sanitizeAcknowledgedAgentsByPaneKey( + ui.acknowledgedAgentsByPaneKey + ), + activityClearedAtByPaneKey: sanitizeActivityClearedAtByPaneKey(ui.activityClearedAtByPaneKey), + manuallyUnreadTurnsByPaneKey: sanitizePaneKeyTimestampRecord(ui.manuallyUnreadTurnsByPaneKey) + } +} diff --git a/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts index 3859ebfc75a..971c64dfa52 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts @@ -36,6 +36,9 @@ import { export function createUiPreferenceActions(set: UISliceSet, get: UISliceGet): Partial { return { + sidebarBody: 'workspaces', + setSidebarBody: (body) => set({ sidebarBody: body }), + groupBy: 'repo', // Why: group keys are mode-specific, so clear collapsed state on mode switch — stale keys are meaningless and accumulate. setGroupBy: (g) => { @@ -146,6 +149,28 @@ export function createUiPreferenceActions(set: UISliceSet, get: UISliceGet): Par filterRepoIds: [], setFilterRepoIds: (ids) => set({ filterRepoIds: ids }), + agentsVisibleHostIds: null, + setAgentsVisibleHostIds: (ids) => { + const agentsVisibleHostIds = normalizeVisibleExecutionHostIds(ids) + set({ agentsVisibleHostIds }) + window.api.ui.set({ agentsVisibleHostIds }).catch(console.error) + }, + agentsFilterRepoIds: [], + setAgentsFilterRepoIds: (ids) => { + set({ agentsFilterRepoIds: ids }) + window.api.ui.set({ agentsFilterRepoIds: [...ids] }).catch(console.error) + }, + agentsShowChildAgents: false, + setAgentsShowChildAgents: (v) => { + set({ agentsShowChildAgents: v }) + window.api.ui.set({ agentsShowChildAgents: v }).catch(console.error) + }, + agentsCompactMode: true, + setAgentsCompactMode: (v) => { + set({ agentsCompactMode: v }) + window.api.ui.set({ agentsCompactMode: v }).catch(console.error) + }, + collapsedGroups: new Set(), toggleCollapsedGroup: (key) => set((s) => { diff --git a/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts index 93a4d0069fb..c693c900c4a 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts @@ -1,5 +1,7 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' import { isSettingsNavigationTarget } from '../../../lib/settings-navigation-types' +// Pure predicate over GlobalSettings; safe to share with the store layer. +import { shouldShowAgentsSidebar } from '@/components/sidebar/agents-sidebar-visibility' export function createUiSettingsActions(set: UISliceSet, get: UISliceGet): Partial { return { @@ -15,9 +17,10 @@ export function createUiSettingsActions(set: UISliceSet, get: UISliceGet): Parti }, closeSettingsPage: () => set((state) => { + // Agents graduated from experimentalActivity; match openActivityPage's gate. const previousView = state.previousViewBeforeSettings === 'activity' && - state.settings?.experimentalActivity !== true + !shouldShowAgentsSidebar(state.settings) ? 'terminal' : state.previousViewBeforeSettings return { activeView: previousView } diff --git a/src/renderer/src/store/slices/ui/ui-slice-surface-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-surface-actions.ts index fb83ff6fc9b..26aab77d0cb 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-surface-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-surface-actions.ts @@ -143,8 +143,11 @@ export function createUiSurfaceActions(set: UISliceSet, _get: UISliceGet): Parti pendingRevealWorktree: null, pendingRevealSidebarRow: null, + // Why sidebarBody here: the worktree list (and its reveal consumer) is unmounted while the + // Agents body is showing, so a reveal that does not switch bodies silently no-ops. revealWorktreeInSidebar: (worktreeId, options) => set({ + sidebarBody: 'workspaces', pendingRevealWorktree: { worktreeId, ...(options?.executionHostId ? { executionHostId: options.executionHostId } : {}), @@ -155,6 +158,7 @@ export function createUiSurfaceActions(set: UISliceSet, _get: UISliceGet): Parti }), revealSidebarRow: (rowKey, options) => set({ + sidebarBody: 'workspaces', pendingRevealSidebarRow: { rowKey, behavior: options?.behavior ?? 'smooth', diff --git a/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts index 432572b799f..d61efbb598d 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts @@ -1,10 +1,14 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' import { rewindHistoryIndexPastView } from '../worktree-nav-history' +// Pure predicate over GlobalSettings; safe to share with the store layer. +import { shouldShowAgentsSidebar } from '@/components/sidebar/agents-sidebar-visibility' export function createUiViewActions(set: UISliceSet, get: UISliceGet): Partial { return { openActivityPage: () => { - if (get().settings?.experimentalActivity !== true) { + // Agents graduated from experimentalActivity; gate on the same visibility + // rule as the sidebar entry points so the view is reachable iff shown. + if (!shouldShowAgentsSidebar(get().settings)) { return } set((state) => ({ diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-meta.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-meta.ts index 64f4f4162ca..f2a5306e683 100644 --- a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-meta.ts +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-meta.ts @@ -10,7 +10,10 @@ import { } from '../../../../../../shared/execution-host' import { parseWorkspaceKey } from '../../../../../../shared/workspace-scope' import { folderWorkspaceToWorktree } from '../../../../../../shared/folder-workspace-worktree' -import { findIndexedWorktreeOwnerForHost } from '@/lib/worktree-runtime-owner-index' +import { + findIndexedDetectedWorktrees, + findIndexedWorktreeOwnerForHost +} from '@/lib/worktree-runtime-owner-index' import { findWorktreeById, withoutErasedRequiredWorktreeFields } from '../../worktree-helpers' import { worktreeMatchesHost } from './worktree-host-ownership' @@ -99,16 +102,21 @@ export function findKnownWorktreeById( if (visible) { return visible } - for (const result of Object.values(state.detectedWorktreesByRepo)) { - const detected = result.worktrees.find( - (worktree) => - worktree.id === worktreeId && - (!executionHostId || - worktreeMatchesHost(worktree, executionHostId, { - unhostedWorktreesMatchHost: executionHostId === LOCAL_EXECUTION_HOST_ID - })) - ) - if (detected) { + // Why the index: this miss path runs per activity row for exactly the worktrees the + // feature targets (retained agents on deleted worktrees); the cached index replaces a + // full scan of every repo's detected worktrees. The index holds the same row objects, + // so the cast restores the listing's row type. + const detectedCandidates = findIndexedDetectedWorktrees( + state.detectedWorktreesByRepo, + worktreeId + ) as DetectedWorktreeListResult['worktrees'] + for (const detected of detectedCandidates) { + if ( + !executionHostId || + worktreeMatchesHost(detected, executionHostId, { + unhostedWorktreesMatchHost: executionHostId === LOCAL_EXECUTION_HOST_ID + }) + ) { return detected } } diff --git a/src/renderer/src/store/slices/worktrees/teardown/worktree-purge-state.ts b/src/renderer/src/store/slices/worktrees/teardown/worktree-purge-state.ts index 87becf6b343..618ae38815c 100644 --- a/src/renderer/src/store/slices/worktrees/teardown/worktree-purge-state.ts +++ b/src/renderer/src/store/slices/worktrees/teardown/worktree-purge-state.ts @@ -104,6 +104,8 @@ export function buildWorktreePurgeState( : {}), agentLaunchConfigByPaneKey: omitByPaneKeyTabPrefix(s.agentLaunchConfigByPaneKey), acknowledgedAgentsByPaneKey: omitByPaneKeyTabPrefix(s.acknowledgedAgentsByPaneKey), + activityClearedAtByPaneKey: omitByPaneKeyTabPrefix(s.activityClearedAtByPaneKey), + manuallyUnreadTurnsByPaneKey: omitByPaneKeyTabPrefix(s.manuallyUnreadTurnsByPaneKey), paneForegroundAgentByPaneKey: omitByPaneKeyTabPrefix(s.paneForegroundAgentByPaneKey), sleepingAgentSessionsByPaneKey: omitByPaneKeyTabPrefix(s.sleepingAgentSessionsByPaneKey), unreadTerminalTabs: omitByTabId(s.unreadTerminalTabs), diff --git a/src/renderer/src/web/preload-api/web-agent-status-api.ts b/src/renderer/src/web/preload-api/web-agent-status-api.ts index ed4636ab5fb..d7c9740018c 100644 --- a/src/renderer/src/web/preload-api/web-agent-status-api.ts +++ b/src/renderer/src/web/preload-api/web-agent-status-api.ts @@ -14,6 +14,8 @@ export function createWebAgentStatusApi(): Partial { onLegacyWorkerTerminalRecovery: () => noopUnsubscribe, getMigrationUnsupportedSnapshot: () => Promise.resolve([]), drop: () => {}, + dropPersisted: () => {}, + dropPersistedBatch: () => {}, reconcileEndedProcess: () => {}, dropByTabPrefix: () => {}, retirePaneAuthority: () => {}, diff --git a/src/renderer/src/web/preload-api/web-preference-normalization.ts b/src/renderer/src/web/preload-api/web-preference-normalization.ts index a5e99c34788..9ae3ad00d4a 100644 --- a/src/renderer/src/web/preload-api/web-preference-normalization.ts +++ b/src/renderer/src/web/preload-api/web-preference-normalization.ts @@ -68,7 +68,13 @@ export function mergeHostWebUIState( automationHostFilter: local.automationHostFilter, hideWorkspacesFromOtherDevices: local.hideWorkspacesFromOtherDevices === true, manualRepoOrder: local.manualRepoOrder, - workspaceHostOrder: local.workspaceHostOrder + workspaceHostOrder: local.workspaceHostOrder, + agentsVisibleHostIds: local.agentsVisibleHostIds, + agentsFilterRepoIds: local.agentsFilterRepoIds, + agentsShowChildAgents: local.agentsShowChildAgents, + agentsCompactMode: local.agentsCompactMode, + activityClearedAtByPaneKey: local.activityClearedAtByPaneKey, + manuallyUnreadTurnsByPaneKey: local.manuallyUnreadTurnsByPaneKey } satisfies Record & Partial return { ...mergeWebUIState(local, incoming), ...pinned } } diff --git a/src/renderer/src/web/web-preload-api-ui.test.ts b/src/renderer/src/web/web-preload-api-ui.test.ts index 9222fdbc42a..94e9ba8e1eb 100644 --- a/src/renderer/src/web/web-preload-api-ui.test.ts +++ b/src/renderer/src/web/web-preload-api-ui.test.ts @@ -464,13 +464,25 @@ describe('web UI preload API', () => { automationHostFilter: { kind: 'host', hostKey: 'browser-local-host-key' }, hideWorkspacesFromOtherDevices: true, manualRepoOrder: [{ hostId: 'runtime:web-env-1', repoId: 'repo-b' }], - workspaceHostOrder: ['runtime:web-env-1', 'local'] + workspaceHostOrder: ['runtime:web-env-1', 'local'], + agentsVisibleHostIds: ['runtime:web-env-1'], + agentsFilterRepoIds: ['repo-b'], + agentsShowChildAgents: true, + agentsCompactMode: false, + activityClearedAtByPaneKey: { 'tab-1:leaf-1': 123 }, + manuallyUnreadTurnsByPaneKey: { 'tab-1:leaf-1': 321 } } const hostUiSamples: Record = { automationHostFilter: { kind: 'all' }, hideWorkspacesFromOtherDevices: false, manualRepoOrder: [{ hostId: 'local', repoId: 'repo-a' }], - workspaceHostOrder: ['local', 'ssh:box'] + workspaceHostOrder: ['local', 'ssh:box'], + agentsVisibleHostIds: ['local'], + agentsFilterRepoIds: ['repo-a'], + agentsShowChildAgents: false, + agentsCompactMode: true, + activityClearedAtByPaneKey: { 'tab-2:leaf-2': 456 }, + manuallyUnreadTurnsByPaneKey: { 'tab-2:leaf-2': 654 } } it.each(PAIRING_LOCAL_UI_FIELDS.map((field) => [field] as const))( diff --git a/src/shared/agent-status-ipc-payload.ts b/src/shared/agent-status-ipc-payload.ts index 6a469b31491..09d8eb284ac 100644 --- a/src/shared/agent-status-ipc-payload.ts +++ b/src/shared/agent-status-ipc-payload.ts @@ -51,6 +51,16 @@ export type AgentStatusIpcPayload = ParsedAgentStatusPayload & { restoredUnconfirmed?: boolean } & WithAgentStatusObservation +/** Identity used by UI-only cleanup to evict exactly the status it cleared. + * Deliberately minimal — receivedAt + stateStartedAt pin the exact event instance + * (the same baseline the interrupt-inference guard uses). Renderer-enriched fields + * (connectionId, worktreeId) diverge from main's cache and must not participate. */ +export type AgentStatusCacheIdentity = { + paneKey: string + receivedAt: number + stateStartedAt: number +} + /** Wire shape for ordinary pane teardown or a stamped SSH disconnect batch. */ export type AgentStatusClearIpcPayload = | { paneKey: string } diff --git a/src/shared/agent-status-types.ts b/src/shared/agent-status-types.ts index 36bf7cbe09c..d2446051115 100644 --- a/src/shared/agent-status-types.ts +++ b/src/shared/agent-status-types.ts @@ -15,6 +15,7 @@ import { assertJsonTextStructureWithinLimits } from './json-text-structure-limit export { AGENT_STATUS_MAX_FIELD_LENGTH } from './agent-status-field-normalization' export type { + AgentStatusCacheIdentity, AgentStatusClearIpcPayload, AgentStatusIpcPayload, MigrationUnsupportedPtyEntry diff --git a/src/shared/agents-sidebar-visibility.ts b/src/shared/agents-sidebar-visibility.ts new file mode 100644 index 00000000000..c05f8e106ed --- /dev/null +++ b/src/shared/agents-sidebar-visibility.ts @@ -0,0 +1,12 @@ +export type AgentsSidebarVisibilitySettings = { + showAgentsSidebar?: boolean +} + +export function resolveAgentsSidebarVisible( + settings: Partial | null | undefined +): boolean { + if (!settings) { + return true + } + return settings.showAgentsSidebar !== false +} diff --git a/src/shared/constants.ts b/src/shared/constants.ts index 96c7901cc08..9d26b140dec 100644 --- a/src/shared/constants.ts +++ b/src/shared/constants.ts @@ -269,6 +269,10 @@ export function getDefaultUIState(): PersistedUIState { alwaysShowDefaultBranchWorkspace: true, showDotfilesByWorktree: {}, filterRepoIds: [], + agentsVisibleHostIds: null, + agentsFilterRepoIds: [], + agentsShowChildAgents: false, + agentsCompactMode: true, collapsedGroups: [], uiZoomLevel: 0, editorFontZoomLevel: 0, @@ -292,6 +296,8 @@ export function getDefaultUIState(): PersistedUIState { trustedOrcaHooks: {}, setupScriptPromptDismissedRepoIds: [], acknowledgedAgentsByPaneKey: {}, + activityClearedAtByPaneKey: {}, + manuallyUnreadTurnsByPaneKey: {}, setupGuideSidebarDismissed: false, setupGuideBrowserMilestoneMigrated: true, setupGuideBrowserMilestoneLegacyComplete: false, diff --git a/src/shared/default-global-settings.ts b/src/shared/default-global-settings.ts index 313e9fc6a9a..260fb5d6846 100644 --- a/src/shared/default-global-settings.ts +++ b/src/shared/default-global-settings.ts @@ -225,6 +225,7 @@ export function buildDefaultSettings(args: { // Why: off keeps the cosmetic overlay unmounted for users who never opt in. experimentalPet: false, experimentalActivity: false, + showAgentsSidebar: true, experimentalActivityDefaultedOffForAllUsers: true, experimentalTerminalAttention: false, experimentalAgentHibernation: false, diff --git a/src/shared/global-settings-types.ts b/src/shared/global-settings-types.ts index 49f9be8baad..f650e4f4a77 100644 --- a/src/shared/global-settings-types.ts +++ b/src/shared/global-settings-types.ts @@ -427,6 +427,12 @@ export type GlobalSettings = { experimentalActivity: boolean /** Experimental: pop-out Kanban dashboard for monitoring and opening agent terminals across worktrees. */ experimentalAgentDashboardPopout?: boolean + /** Experimental: whether the Agents tab is shown in the left sidebar. Defaults on. */ + showAgentsSidebar?: boolean + /** Set after the experimental Agents tab introduction has been acknowledged. */ + agentsSidebarIntroShown?: boolean + /** True when the profile previously opted into the legacy experimental Agents view. */ + agentsSidebarMigratedFromExperimental?: boolean /** How the Agent Dashboard opens: an in-window companion board or a separate pop-out window. Defaults to in-window. */ experimentalAgentDashboardMode?: AgentDashboardMode /** Includes stale quiet agents as a fourth Agent Dashboard column. */ diff --git a/src/shared/pairing-local-ui-fields.test.ts b/src/shared/pairing-local-ui-fields.test.ts index ab837782783..ffc7baafe07 100644 --- a/src/shared/pairing-local-ui-fields.test.ts +++ b/src/shared/pairing-local-ui-fields.test.ts @@ -9,7 +9,13 @@ describe('pairing-local UI fields', () => { 'automationHostFilter', 'hideWorkspacesFromOtherDevices', 'manualRepoOrder', - 'workspaceHostOrder' + 'workspaceHostOrder', + 'agentsVisibleHostIds', + 'agentsFilterRepoIds', + 'agentsShowChildAgents', + 'agentsCompactMode', + 'activityClearedAtByPaneKey', + 'manuallyUnreadTurnsByPaneKey' ]) }) diff --git a/src/shared/pairing-local-ui-fields.ts b/src/shared/pairing-local-ui-fields.ts index 55865416c01..d925c3094d8 100644 --- a/src/shared/pairing-local-ui-fields.ts +++ b/src/shared/pairing-local-ui-fields.ts @@ -11,7 +11,14 @@ export const PAIRING_LOCAL_UI_FIELDS = [ 'automationHostFilter', 'hideWorkspacesFromOtherDevices', 'manualRepoOrder', - 'workspaceHostOrder' + 'workspaceHostOrder', + // Agent View filters and presentation belong to each client's host catalog and viewport. + 'agentsVisibleHostIds', + 'agentsFilterRepoIds', + 'agentsShowChildAgents', + 'agentsCompactMode', + 'activityClearedAtByPaneKey', + 'manuallyUnreadTurnsByPaneKey' ] as const satisfies readonly (keyof PersistedUIState)[] export type PairingLocalUiField = (typeof PAIRING_LOCAL_UI_FIELDS)[number] diff --git a/src/shared/persisted-ui-state-types.ts b/src/shared/persisted-ui-state-types.ts index 2d5f635477c..14e97bc34fc 100644 --- a/src/shared/persisted-ui-state-types.ts +++ b/src/shared/persisted-ui-state-types.ts @@ -73,6 +73,14 @@ export type PersistedUIState = { /** Per-worktree Explorer dotfile visibility. Missing entries inherit the default: show. */ showDotfilesByWorktree?: Record filterRepoIds: string[] + /** Agents-view host scope; deliberately separate from visibleWorkspaceHostIds so a monitoring surface never inherits nav filters silently. `null` = all hosts. */ + agentsVisibleHostIds?: VisibleWorkspaceHostIds + /** Agents-view project filter; empty = all projects. Separate from filterRepoIds (workspace nav). */ + agentsFilterRepoIds?: string[] + /** Agents-view: include child (orchestration-dispatched) agent threads. Absent means off. */ + agentsShowChildAgents?: boolean + /** Agents-view compact thread rows. Absent means on. */ + agentsCompactMode?: boolean collapsedGroups: string[] uiZoomLevel: number editorFontZoomLevel: number @@ -120,6 +128,10 @@ export type PersistedUIState = { updateReassuranceSeen?: boolean /** Per-paneKey "row visited" timestamps that mute seen inline-agent rows; persisted because rows survive restart, else acked rows return bold. Renderer-owned via ui:set. */ acknowledgedAgentsByPaneKey?: Record + /** Per-paneKey "Clear completed" cutoffs hiding activity events stamped at or before the cutoff; persisted so cleared rows stay cleared across restart. Renderer-owned via ui:set. */ + activityClearedAtByPaneKey?: Record + /** Per-paneKey turn stamps the user explicitly marked unread; persisted so a manual unread survives restart the way acks and cutoffs do. Renderer-owned via ui:set. */ + manuallyUnreadTurnsByPaneKey?: Record /** User-hidden setup-guide sidebar entry; a reversible declutter pref (Help menu stays available), not completion. */ setupGuideSidebarDismissed?: boolean /** One-shot marker for the browser setup-guide milestone; profiles missing it are evaluated once in the renderer (completion needs runtime probes). */ diff --git a/src/shared/telemetry-property-schemas.ts b/src/shared/telemetry-property-schemas.ts index 3346ea73320..b1e8d2ed573 100644 --- a/src/shared/telemetry-property-schemas.ts +++ b/src/shared/telemetry-property-schemas.ts @@ -192,6 +192,7 @@ export const SETTINGS_CHANGED_WHITELIST = [ 'experimentalNativeChat', 'experimentalStructuredNativeChat', 'experimentalActivity', + 'showAgentsSidebar', 'experimentalAgentDashboardPopout', 'experimentalTerminalAttention', 'experimentalAgentHibernation', diff --git a/tests/e2e/activity-agent-pane-isolation.spec.ts b/tests/e2e/activity-agent-pane-isolation.spec.ts index fe6600f3f76..817044bd127 100644 --- a/tests/e2e/activity-agent-pane-isolation.spec.ts +++ b/tests/e2e/activity-agent-pane-isolation.spec.ts @@ -16,12 +16,6 @@ type SeededActivityThread = { prompt: string } -type ActivityPaneVisibility = { - slotId: string | null - allLeafIds: string[] - visibleLeafIds: string[] -} - type ActivePaneSelection = { activeWorktreeId: string | null activeGroupId: string | null @@ -38,7 +32,7 @@ type SplitGroupTerminal = { } function agentsSidebarButton(page: Page) { - return page.getByRole('button', { name: /^Agents(?:\s+\d+)?$/ }).first() + return page.getByRole('radio', { name: /^Agents$/ }).first() } async function seedActivityThread( @@ -113,40 +107,6 @@ async function seedActivityThreadsForSplitPanes( return [first, second] } -async function readActivityPaneVisibility(page: Page): Promise { - return page.evaluate(() => { - const slot = document.querySelector( - '[data-activity-terminal-slot-id]:not([aria-hidden="true"])' - ) - if (!slot) { - return { slotId: null, allLeafIds: [], visibleLeafIds: [] } - } - - const hasInlineDisplayNoneBetween = (element: HTMLElement, root: HTMLElement): boolean => { - let current: HTMLElement | null = element - while (current) { - if (current.style.display === 'none') { - return true - } - if (current === root) { - return false - } - current = current.parentElement - } - return false - } - - const panes = Array.from(slot.querySelectorAll('[data-leaf-id]')) - return { - slotId: slot.dataset.activityTerminalSlotId ?? null, - allLeafIds: panes.map((pane) => pane.dataset.leafId ?? ''), - visibleLeafIds: panes - .filter((pane) => !hasInlineDisplayNoneBetween(pane, slot)) - .map((pane) => pane.dataset.leafId ?? '') - } - }) -} - async function enableInlineAgentCards(page: Page): Promise { await page.evaluate(() => { const store = window.__store @@ -164,9 +124,12 @@ async function enableInlineAgentCards(page: Page): Promise { async function enableActivityAgentsView(page: Page): Promise { await page.evaluate(async () => { - const settings = await window.api.settings.set({ experimentalActivity: true }) - // Why: these specs exercise the experimental Agents page. E2E profiles use - // production defaults, where the sidebar entry is hidden unless enabled. + // Why: the Agents tab is on by default, but a fresh profile opens the intro popover + // over it; stamping it as shown keeps the toggle clickable without dismissing it first. + const settings = await window.api.settings.set({ + showAgentsSidebar: true, + agentsSidebarIntroShown: true + }) window.__store?.setState({ settings }) }) } @@ -267,7 +230,7 @@ test.describe('Activity Agent Pane Isolation', () => { await waitForPaneCount(orcaPage, 1, 30_000) }) - test('selecting agent rows isolates the matching split pane by stable leaf id', async ({ + test('selecting agent rows focuses the matching split pane by stable leaf id', async ({ orcaPage }) => { await splitActiveTerminalPane(orcaPage, 'vertical') @@ -281,87 +244,27 @@ test.describe('Activity Agent Pane Isolation', () => { await orcaPage.getByRole('button').filter({ hasText: first.prompt }).first().click() await expect - .poll(async () => readActivityPaneVisibility(orcaPage), { + .poll(async () => readActivePaneSelection(orcaPage), { timeout: 10_000, - message: 'Activity did not isolate the first selected split pane' + message: 'Agents sidebar row did not focus the first selected split pane' }) .toMatchObject({ - allLeafIds: expect.arrayContaining([first.leafId, second.leafId]), - visibleLeafIds: [first.leafId] + activeTabId: snapshot.tabId, + activeLeafId: first.leafId }) await orcaPage.getByRole('button').filter({ hasText: second.prompt }).first().click() await expect - .poll(async () => readActivityPaneVisibility(orcaPage), { + .poll(async () => readActivePaneSelection(orcaPage), { timeout: 10_000, - message: 'Activity did not switch isolation to the second selected split pane' + message: 'Agents sidebar row did not focus the second selected split pane' }) .toMatchObject({ - allLeafIds: expect.arrayContaining([first.leafId, second.leafId]), - visibleLeafIds: [second.leafId] + activeTabId: snapshot.tabId, + activeLeafId: second.leafId }) }) - test('acknowledged stable pane keys clear the Agents unread badge', async ({ orcaPage }) => { - await splitActiveTerminalPane(orcaPage, 'vertical') - await waitForPaneCount(orcaPage, 2) - const snapshot = await waitForPaneIdentitySnapshot(orcaPage, 2) - // Why: useAutoAckViewedAgent (App.tsx) auto-acknowledges the agent on the - // store's *active* visible terminal leaf the instant its status lands, which - // clears the unread badge before we can assert it (flaky on focused xvfb CI - // windows). Seed on the non-active split pane — auto-ack only ever targets the - // active leaf — so the badge stays unread until the explicit acknowledgeAgents() - // call under test. - const activeLeafId = await orcaPage.evaluate( - (tabId) => window.__store?.getState().terminalLayoutsByTabId[tabId]?.activeLeafId ?? null, - snapshot.tabId - ) - const targetPane = - snapshot.panes.find((pane) => pane.leafId !== activeLeafId) ?? snapshot.panes[0] - if (!targetPane) { - throw new Error('Activity acknowledgement test needs a split pane') - } - const now = Date.now() - const thread: SeededActivityThread = { - paneKey: `${snapshot.tabId}:${targetPane.leafId}`, - leafId: targetPane.leafId, - prompt: `ACTIVITY_ACK_STABLE_PANE_${now}` - } - - await orcaPage.evaluate(() => { - const store = window.__store - if (!store) { - throw new Error('window.__store is not available') - } - const state = store.getState() - for (const worktree of Object.values(state.worktreesByRepo).flat()) { - state.markWorktreeVisited(worktree.id) - } - }) - - await seedActivityThread( - orcaPage, - thread, - 'Codex acknowledged pane', - 'blocked', - 'Waiting for acknowledgement migration coverage.', - now - 5_000 - ) - - await expect(agentsSidebarButton(orcaPage)).toHaveAccessibleName(/^Agents\s+1$/) - - await orcaPage.evaluate((paneKey) => { - const store = window.__store - if (!store) { - throw new Error('window.__store is not available') - } - store.getState().acknowledgeAgents([paneKey]) - }, thread.paneKey) - - await expect(agentsSidebarButton(orcaPage)).toHaveAccessibleName(/^Agents$/) - await expect(orcaPage.getByRole('button', { name: /^Agents\s+1$/ })).toHaveCount(0) - }) - test('workspace card agent rows focus the matching terminal split pane', async ({ orcaPage }) => { await splitActiveTerminalPane(orcaPage, 'vertical') await waitForPaneCount(orcaPage, 2) From 6c66487fca0ba5afa8c26fe86527a17c95c62a8d Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:20:15 -0700 Subject: [PATCH 069/398] ci: checkout PR head for reusable E2E (#18230) --- .github/workflows/pr.yml | 3 +++ config/scripts/pr-e2e-gate-contract.test.mjs | 1 + 2 files changed, 4 insertions(+) diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index c17ad60d5d3..50062161da6 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -897,6 +897,9 @@ jobs: contents: read uses: ./.github/workflows/e2e.yml with: + # The synthetic pull-request merge ref can disappear while this reusable + # workflow is queued. The head SHA is immutable and works for every PR. + ref: ${{ github.event.pull_request.head.sha }} test_files: ${{ needs.e2e-paths.outputs.test_files }} ssh_source_changed: ${{ needs.e2e-paths.outputs.ssh_source_changed }} diff --git a/config/scripts/pr-e2e-gate-contract.test.mjs b/config/scripts/pr-e2e-gate-contract.test.mjs index 232780c6941..ceac6b8cc6e 100644 --- a/config/scripts/pr-e2e-gate-contract.test.mjs +++ b/config/scripts/pr-e2e-gate-contract.test.mjs @@ -114,6 +114,7 @@ describe('PR E2E gate contract', () => { expect(prWorkflow.jobs['e2e-paths'].outputs.test_files).toBe( '${{ steps.filter.outputs.test_files }}' ) + expect(prWorkflow.jobs.e2e.with.ref).toBe('${{ github.event.pull_request.head.sha }}') expect(prWorkflow.jobs.e2e.with.test_files).toBe('${{ needs.e2e-paths.outputs.test_files }}') }) From 616fa751da7790afb6c152e45beeb4b0b46a47d3 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 11:20:53 -0700 Subject: [PATCH 070/398] revert(native-chat): drop the speculative Fable model-switch detections (#18215) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Both changes shipped in #18055 were written against strings never observed in a real session, and neither fixed a reported problem. Guessing at agent output we have not seen is how the picker got a row that silently no-ops. Fable consent detection is removed outright. It watched the session for "Fable N uses usage credits and needs a one-time consent" and answered `interaction-required`. No consent prompt appeared in any validation run — the test account had already consented — so the matched wording was never confirmed. With the detector gone nothing produces `interaction-required`, so the outcome leaves the union and its unreachable handler goes with it. A real consent prompt now reports the switch as unverified, which is the honest failure mode for output we cannot recognize. The weekly usage scope goes back to exact `display_name === 'fable'`. It had been widened to `/^fable\b/` against a hypothetical rename of Anthropic's own usage window; the API still reports "Fable", so the match was insurance against a scenario with no evidence behind it. Tests covering the removed behavior are deleted rather than rewritten, including the two pre-existing `interaction-required` cases that asserted the terminal is revealed. The disabled-row filter from #18055 is deliberately untouched. Claude-Session: https://claude.ai/code/session_01SJy4XGrdre6YaU1wYNKak4 Co-authored-by: Merge Sim --- .../claude-fetcher-fable-usage.test.ts | 45 ------------- .../rate-limits/claude-oauth-usage-request.ts | 6 +- .../native-chat/NativeChatComposer.test.tsx | 31 +-------- .../claude-model-switch-confirmation.test.ts | 63 ------------------- .../claude-model-switch-confirmation.ts | 22 +------ .../native-chat-pty-session-options.test.ts | 22 ------- .../native-chat-session-option-apply.ts | 6 -- 7 files changed, 4 insertions(+), 191 deletions(-) diff --git a/src/main/rate-limits/claude-fetcher-fable-usage.test.ts b/src/main/rate-limits/claude-fetcher-fable-usage.test.ts index 50617d95bb4..2a84cc4f2af 100644 --- a/src/main/rate-limits/claude-fetcher-fable-usage.test.ts +++ b/src/main/rate-limits/claude-fetcher-fable-usage.test.ts @@ -144,51 +144,6 @@ describe('fetchClaudeRateLimits', () => { expect(fetchViaPty).not.toHaveBeenCalled() }) - it('maps a scoped Fable window whose display name carries a point release', async () => { - const configDir = '/Users/test/.claude' - const authPreparation: ClaudeRuntimeAuthPreparation = { - configDir, - envPatch: { CLAUDE_CONFIG_DIR: configDir }, - stripAuthEnv: false, - provenance: 'managed:account-1' - } - vi.mocked(readActiveClaudeKeychainCredentialsStrict).mockResolvedValueOnce( - JSON.stringify({ claudeAiOauth: { accessToken: 'oauth-token' } }) - ) - netFetchMock.mockResolvedValueOnce( - new Response( - JSON.stringify({ - five_hour: { utilization: 36 }, - seven_day: { utilization: 73 }, - // Discriminating: a passing scope match must beat this fallback. - fable_weekly: { utilization: 12 }, - limits: [ - { - kind: 'weekly_scoped', - percent: 64, - resets_at: '2026-07-17T20:00:00.099908+00:00', - is_active: true, - scope: { model: { display_name: 'Fable 5.1' } } - } - ] - }), - { status: 200 } - ) - ) - - await expect( - fetchClaudeRateLimits({ authPreparation, allowUsagePanelSupplement: true }) - ).resolves.toMatchObject({ - provider: 'claude', - status: 'ok', - fableWeekly: { - usedPercent: 64, - resetsAt: Date.parse('2026-07-17T20:00:00.099908+00:00') - } - }) - expect(fetchViaPty).not.toHaveBeenCalled() - }) - it('surfaces inactive scoped Fable usage over the legacy OAuth fallback', async () => { const configDir = '/Users/test/.claude' const authPreparation: ClaudeRuntimeAuthPreparation = { diff --git a/src/main/rate-limits/claude-oauth-usage-request.ts b/src/main/rate-limits/claude-oauth-usage-request.ts index e29bef5f48e..9565a064514 100644 --- a/src/main/rate-limits/claude-oauth-usage-request.ts +++ b/src/main/rate-limits/claude-oauth-usage-request.ts @@ -32,17 +32,13 @@ async function ensureProxyFromEnvironment(): Promise { }).catch(() => {}) } -// Why: the scope name carries the shipped version once a point release exists -// ("Fable 5.1"), so exact equality would drop the window. -const FABLE_SCOPE_RE = /^fable\b/ - function mapFableWeeklyWindow(data: OAuthUsageResponse): RateLimitWindow | null { const scoped = Array.isArray(data.limits) ? data.limits.find( (limit) => limit?.kind === 'weekly_scoped' && Number.isFinite(limit.percent) && - FABLE_SCOPE_RE.test(limit.scope?.model?.display_name?.trim().toLowerCase() ?? '') + limit.scope?.model?.display_name?.trim().toLowerCase() === 'fable' ) : undefined return ( diff --git a/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx b/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx index be54ac6ac00..b1cf1a5879e 100644 --- a/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx @@ -27,10 +27,10 @@ const mocks = vi.hoisted(() => ({ sessionOptionsSnapshot?: SessionOptionDescriptor[] attachDisabled?: boolean } | null, - modelSwitchOutcome: 'applied' as 'applied' | 'rejected' | 'interaction-required' | 'unknown', + modelSwitchOutcome: 'applied' as 'applied' | 'rejected' | 'unknown', confirmationObserver: null as { ready: Promise - result: Promise<'applied' | 'rejected' | 'interaction-required' | 'unknown'> + result: Promise<'applied' | 'rejected' | 'unknown'> arm: ReturnType startDetection: ReturnType dispose: ReturnType @@ -731,33 +731,6 @@ describe('NativeChatComposer', () => { expect(onSwitchToTerminal).not.toHaveBeenCalled() }) - it('reveals Claude interaction only when the model switch needs user input', async () => { - mocks.sendHandle.settleAfterMs = 0 - mocks.modelSwitchOutcome = 'interaction-required' - const onSwitchToTerminal = vi.fn() - render( - - ) - - await act(async () => { - await mocks.fieldProps?.sessionOptionsSurface?.setOption('model', 'fable') - }) - - expect(mocks.sendNativeChatMessageVerified).toHaveBeenCalledWith( - {}, - 'pty-1', - '/model fable', - expect.any(AbortSignal) - ) - expect(onSwitchToTerminal).toHaveBeenCalledOnce() - }) - it('types the Codex picker command and switches to the terminal', async () => { mocks.sendHandle.settleAfterMs = 0 const onSwitchToTerminal = vi.fn() diff --git a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts index c0fb5d3aa3d..5697dc3d2be 100644 --- a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts +++ b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.test.ts @@ -138,69 +138,6 @@ describe('Claude model switch confirmation detection', () => { await expect(observer.result).resolves.toBe('rejected') }) - it('requests interaction for Fable one-time usage-credit consent', async () => { - const dataObserver = { current: (_data: string): void => {} } - const observer = createClaudeModelSwitchConfirmationObserver({ - ptyId: 'pty-1', - settings: {}, - expectedModelLabel: 'Fable 5', - subscribeToData: (watcher) => { - dataObserver.current = watcher - return vi.fn(() => {}) - }, - timeoutMs: 100 - }) - - await observer.ready - observer.arm() - dataObserver.current('Fable 5 uses usage credits and needs a one-time consent — ') - dataObserver.current('pick Fable from /model in an interactive session to set it up') - - await expect(observer.result).resolves.toBe('interaction-required') - }) - - it('requests interaction for a Fable point-release consent prompt', async () => { - const dataObserver = { current: (_data: string): void => {} } - const observer = createClaudeModelSwitchConfirmationObserver({ - ptyId: 'pty-1', - settings: {}, - expectedModelLabel: 'Fable 5.1', - subscribeToData: (watcher) => { - dataObserver.current = watcher - return vi.fn(() => {}) - }, - timeoutMs: 100 - }) - - await observer.ready - observer.arm() - // Only the versioned consent line: the generic "pick Fable from /model" - // sentence must not be what carries this case. - dataObserver.current('Fable 5.1 uses usage credits and needs a one-time consent') - - await expect(observer.result).resolves.toBe('interaction-required') - }) - - it('requests interaction for a versioned Fable switch prompt', async () => { - const dataObserver = { current: (_data: string): void => {} } - const observer = createClaudeModelSwitchConfirmationObserver({ - ptyId: 'pty-1', - settings: {}, - expectedModelLabel: 'Fable 5.1', - subscribeToData: (watcher) => { - dataObserver.current = watcher - return vi.fn(() => {}) - }, - timeoutMs: 100 - }) - - await observer.ready - observer.arm() - dataObserver.current('Switch to \u001b[1mFable 5.1\u001b[0m? This model uses usage credits.') - - await expect(observer.result).resolves.toBe('interaction-required') - }) - it('reports unknown when the PTY observer cannot be established', async () => { const observer = createClaudeModelSwitchConfirmationObserver({ ptyId: 'pty-1', diff --git a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts index e97daf9cd23..eedf8506e1c 100644 --- a/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts +++ b/src/renderer/src/components/native-chat/claude-model-switch-confirmation.ts @@ -10,7 +10,7 @@ const MAX_OBSERVED_BYTES = 64 * 1024 type SubscribeToData = (watcher: (data: string) => void) => Promise<() => void> | (() => void) -export type ClaudeModelSwitchOutcome = 'applied' | 'rejected' | 'interaction-required' | 'unknown' +export type ClaudeModelSwitchOutcome = 'applied' | 'rejected' | 'unknown' export type ClaudeModelSwitchConfirmationObserver = { ready: Promise @@ -62,22 +62,6 @@ function hasClaudeModelSwitchRejection(buffer: string): boolean { return compactTerminalText(buffer).includes('keptmodelas') } -// Why: compactTerminalText only strips whitespace, so a point release keeps its -// dot ("fable5.1uses..."). The version is optional because the CLI's own label -// for the newest Fable carries no number at all. -const FABLE_VERSION = String.raw`fable(?:\d+(?:\.\d+)*)?` -const FABLE_CONSENT_RE = new RegExp(`${FABLE_VERSION}usesusagecreditsandneedsaone-timeconsent`) -const FABLE_SWITCH_PROMPT_RE = new RegExp(`switchto${FABLE_VERSION}\\?`) - -function hasClaudeModelSwitchInteraction(buffer: string): boolean { - const text = compactTerminalText(buffer) - return ( - FABLE_CONSENT_RE.test(text) || - text.includes('pickfablefrom/modelinaninteractivesessiontosetitup') || - (FABLE_SWITCH_PROMPT_RE.test(text) && text.includes('usagecredits')) - ) -} - function subscribeToClaudeModelSwitchData(args: { ptyId: string settings: Pick | null | undefined @@ -156,10 +140,6 @@ export function createClaudeModelSwitchConfirmationObserver(args: { finish('rejected') return } - if (hasClaudeModelSwitchInteraction(observed)) { - finish('interaction-required') - return - } if (!confirmationSubmitted && hasClaudeModelSwitchConfirmation(observed)) { confirmationSubmitted = true try { diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts index e653a29258f..9dae22da197 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts @@ -159,28 +159,6 @@ describe('native chat PTY session options', () => { }) }) - it('reveals the terminal only when Claude actually requires model-switch interaction', async () => { - seedNativeChatAppliedSessionOptions('pty-1', 'claude', { model: 'sonnet' }) - const dispatch = vi.fn().mockResolvedValue({ outcome: 'interaction-required' }) - const onAgentPicker = vi.fn() - const surface = createNativeChatPtySessionOptions({ - agent: 'claude', - scopeKey: 'pty-1', - mode: 'live', - dispatchCommand: dispatch, - onAgentPicker - })! - - const result = await surface.setOption('model', 'haiku') - - expect(dispatch).toHaveBeenCalledWith('/model haiku', { - detectAgentInteraction: 'claude-model-switch-confirmation', - expectedChoiceLabel: 'Haiku' - }) - expect(onAgentPicker).toHaveBeenCalledOnce() - expect(result.snapshot[0]).toMatchObject({ valueSource: 'unknown' }) - }) - it('keeps the prior model and persistence when Claude rejects the switch', async () => { seedNativeChatAppliedSessionOptions('pty-1', 'claude', { model: 'fable', diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-apply.ts b/src/renderer/src/components/native-chat/native-chat-session-option-apply.ts index 5585512633a..7e31e1a17cc 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-apply.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-apply.ts @@ -171,12 +171,6 @@ function applyDispatchOutcome( ctx.publish() throw new Error('Could not verify the model change; open the terminal to check.') } - if (dispatchResult?.outcome === 'interaction-required') { - ctx.clearModelTruth() - const snapshot = ctx.publish() - ctx.onAgentPicker?.() - return { snapshot } - } return null } From 1d94ebee3f65940e1a6d5d608c6ffe84fc17733e Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:49:15 -0700 Subject: [PATCH 071/398] fix(agents): stop a deeper vendor helper from stealing a pane's agent identity (#18062) * fix(agents): keep outer agent identity over vendor helpers * fix(agents): preserve outer identity across relay scans --------- Co-authored-by: Merge Sim --- .../__fixtures__/real-agent-rows.json.gz | Bin 0 -> 569 bytes .../agent-foreground-process-batch.ts | 27 ++---- ...agent-foreground-process-real-rows.test.ts | 37 ++++++++ .../providers/agent-foreground-process.ts | 36 ++++---- src/relay/pty-shell-utils.ts | 22 ++--- .../foreground-process-selection.test.ts | 51 +++++++++++ src/shared/foreground-process-selection.ts | 80 ++++++++++++++++++ 7 files changed, 201 insertions(+), 52 deletions(-) create mode 100644 src/main/providers/__fixtures__/real-agent-rows.json.gz create mode 100644 src/main/providers/agent-foreground-process-real-rows.test.ts create mode 100644 src/shared/foreground-process-selection.test.ts create mode 100644 src/shared/foreground-process-selection.ts diff --git a/src/main/providers/__fixtures__/real-agent-rows.json.gz b/src/main/providers/__fixtures__/real-agent-rows.json.gz new file mode 100644 index 0000000000000000000000000000000000000000..3bd62f9eda36e9bbf3a7e3c07ad27cca35f5c0e9 GIT binary patch literal 569 zcmV-90>=FxiwFP!000026U~>)PQx$|Mfd%RC|e3?V#g0wd;+375JIJ~N+F4(rUfL# zzvHBjlyspNq|xf^)ihxvG(Kbfg~>0f&OG@Yyx}Rau_b_54lB z=kOSo!&m}g&wjveJV5Yd$Y34^c(-TPWtw_8EO(6MIIM7t6+3MGeLvU;IG9SEsCQ^6 zNlZC*D4X0)L%G&sw~NFw0&p!A_F=^1IEa(c>A2%v(S^#Ze5f&$#uVF_Cbv^#c5>`y zQOY2*T0*R59T1QEHB;FI$5Qv46bHc z&r%hvC7(~zdGNT(lU?K@doB?^8?7;*wY3DR&pm0C64o03KdnLvQ0uoP;E&>}9(ITq z`UM(cCMkM^o7_$#TuZT=#U!}dTOzED{YJAKj9CF$B*N`=!ERBL>*fYk)kAnC!U*!J zrN(|R8Uqk8G7=yxf}W5!)+w&);jj{q9b>Q(g$+k|>@5Ntj!;+%JzQJfJG+J$9P1Y+ z`ovM@Vs%XqO+6e|;;}c~S1Ee!AqC&s{tU = null - for (const candidate of candidates) { - const recognized = recognizeAgentProcessFromCommandLine(candidate.command) - if ( - recognized && - (bestCandidate === null || - scoreForegroundCandidateRow(candidate) > scoreForegroundCandidateRow(bestCandidate)) - ) { - bestCandidate = candidate - bestName = recognized - } - } - if (bestCandidate && bestName) { + const selected = selectForegroundProcessCandidate(candidates, allCandidates) + if (selected) { return { available: true, - processName: resolveOuterWrapperForegroundProcess(bestName, bestCandidate, allCandidates) + processName: resolveOuterWrapperForegroundProcess( + selected.recognized, + selected.candidate, + allCandidates + ) } } return { available: true, processName: null } diff --git a/src/main/providers/agent-foreground-process-real-rows.test.ts b/src/main/providers/agent-foreground-process-real-rows.test.ts new file mode 100644 index 00000000000..fd86e207398 --- /dev/null +++ b/src/main/providers/agent-foreground-process-real-rows.test.ts @@ -0,0 +1,37 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { gunzipSync } from 'node:zlib' +import { describe, expect, it } from 'vitest' +import type { ProcessTableRow } from '../../shared/process-table-snapshot' +import { resolveAgentForegroundProcessFromPs } from './agent-foreground-process' + +type CapturedRun = { + agent: string + shellPid: number + rows: ProcessTableRow[] +} + +describe('real foreground process captures', () => { + it('resolves all six agents, including omp over its deeper vendor helpers', () => { + const captured = JSON.parse( + gunzipSync(readFileSync(join(__dirname, '__fixtures__', 'real-agent-rows.json.gz'))).toString( + 'utf8' + ) + ) as CapturedRun[] + + expect(captured).toHaveLength(6) + expect( + captured.map(({ agent, shellPid, rows }) => ({ + agent, + processName: resolveAgentForegroundProcessFromPs(rows, shellPid) + })) + ).toEqual([ + { agent: 'claude', processName: 'claude' }, + { agent: 'codex', processName: 'codex' }, + { agent: 'opencode', processName: 'opencode' }, + { agent: 'gemini', processName: 'gemini' }, + { agent: 'grok', processName: 'grok' }, + { agent: 'omp', processName: 'omp' } + ]) + }) +}) diff --git a/src/main/providers/agent-foreground-process.ts b/src/main/providers/agent-foreground-process.ts index d171e39a18e..d244e0100dc 100644 --- a/src/main/providers/agent-foreground-process.ts +++ b/src/main/providers/agent-foreground-process.ts @@ -11,6 +11,7 @@ import { type AgentForegroundResolutionOptions } from './windows-agent-foreground-process' import { isShellProcess } from '../../shared/shell-process-detection' +import { selectForegroundProcessCandidate } from '../../shared/foreground-process-selection' export type { AgentForegroundResolutionOptions } from './windows-agent-foreground-process' export { @@ -120,13 +121,6 @@ export async function confirmShellForegroundProcess( } } -function candidateScore(row: ProcessTableRow & { depth: number }): number { - // Why: foreground descendants carry `+` in `ps stat` on Unix PTYs. Prefer - // them, then prefer leaf/deeper wrappers so `node /path/bin/codex` beats the - // parent shell but still lets the native child confirm the same identity. - return (row.stat.includes('+') ? 10_000 : 0) + row.depth -} - export async function resolveAgentForegroundProcess( shellPid: number | null | undefined, fallbackProcess: string | null, @@ -191,30 +185,30 @@ export async function resolveAgentForegroundProcessWithAvailability( } } -function resolveAgentForegroundProcessFromPs( +export function resolveAgentForegroundProcessFromPs( rows: ProcessTableRow[], shellPid: number ): string | null { const shellRow = rows.find((row) => row.pid === shellPid) - const candidates = collectDescendants(rows, shellPid).sort( - (a, b) => candidateScore(b) - candidateScore(a) - ) + const candidates = collectDescendants(rows, shellPid) // Why: `+` in `ps stat` marks the process holding the terminal foreground. // The root shell can hold it after Ctrl-Z, so use the whole PTY tree as the // foreground gate; otherwise a stopped agent child still masquerades as live. const foregroundIsKnown = shellRow?.stat.includes('+') === true || candidates.some((candidate) => candidate.stat.includes('+')) - for (const candidate of candidates) { - if (foregroundIsKnown && !candidate.stat.includes('+')) { - continue - } - const recognized = recognizeAgentProcessFromCommandLine(candidate.command) - if (recognized) { - // Why: return the outer wrapper (omp) rather than the deeper wrapped child - // (pi) of a shell→omp→pi tree — see resolveOuterWrapperForegroundProcess. - return resolveOuterWrapperForegroundProcess(recognized, candidate, candidates) - } + const foregroundCandidates = foregroundIsKnown + ? candidates.filter((candidate) => candidate.stat.includes('+')) + : candidates + // Keep the complete process tree for ancestry checks. A recognized agent can + // sit above a non-foreground helper before another recognized process; the + // helper is filtered from selection but must remain traversable. + const ancestryCandidates = shellRow ? [{ ...shellRow, depth: 0 }, ...candidates] : candidates + const selected = selectForegroundProcessCandidate(foregroundCandidates, ancestryCandidates) + if (selected) { + // Why: return the outer wrapper (omp) rather than the deeper wrapped child + // (pi) of a shell→omp→pi tree — see resolveOuterWrapperForegroundProcess. + return resolveOuterWrapperForegroundProcess(selected.recognized, selected.candidate, candidates) } return null } diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index accccb9e702..e06edaeabc7 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -6,17 +6,16 @@ import { promisify } from 'node:util' import { isAgentForegroundWrapperProcess, isExpectedAgentProcess, - recognizeAgentProcess, - recognizeAgentProcessFromCommandLine + recognizeAgentProcess } from '../shared/agent-process-recognition' import { getFirstCommandToken } from '../shared/command-token-scanner' import { getProcessTableIndex, getProcessTableSnapshot, - scoreForegroundCandidateRow, type ProcessTableIndex, type ProcessTableRow } from '../shared/process-table-snapshot' +import { selectForegroundProcessCandidate } from '../shared/foreground-process-selection' import { resolveOuterWrapperForegroundProcess, shouldInspectOuterWrapperForegroundProcess @@ -240,9 +239,7 @@ function getForegroundProcessNameFromProcessTable( // snapshot no longer each rebuild the parent/child map over every row. const index = getProcessTableIndex(rows) const root = index.byPid.get(pid) - const candidates = collectDescendants(index, pid).sort( - (a, b) => scoreForegroundCandidateRow(b) - scoreForegroundCandidateRow(a) - ) + const candidates = collectDescendants(index, pid) // Why: SSH relays do not have the daemon's async wrapper cache. Inspect the // remote process tree so node/python agent entrypoints become real agents. const foregroundIsKnown = @@ -264,13 +261,12 @@ function getForegroundProcessNameFromProcessTable( ) { return null } - for (const candidate of inspectionCandidates) { - const recognized = recognizeAgentProcessFromCommandLine(candidate.command) - if (recognized) { - // Why: return the outer wrapper (omp) rather than the deeper wrapped child - // (pi) of a shell→omp→pi tree — see resolveOuterWrapperForegroundProcess. - return resolveOuterWrapperForegroundProcess(recognized, candidate, candidates) - } + const ancestryCandidates = root ? [{ ...root, depth: 0 }, ...candidates] : candidates + const selected = selectForegroundProcessCandidate(inspectionCandidates, ancestryCandidates) + if (selected) { + // Why: return the outer wrapper (omp) rather than a deeper recognized helper + // in the same process lineage. + return resolveOuterWrapperForegroundProcess(selected.recognized, selected.candidate, candidates) } return null } diff --git a/src/shared/foreground-process-selection.test.ts b/src/shared/foreground-process-selection.test.ts new file mode 100644 index 00000000000..5fc45b186fb --- /dev/null +++ b/src/shared/foreground-process-selection.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it } from 'vitest' +import { selectForegroundProcessCandidate } from './foreground-process-selection' + +describe('selectForegroundProcessCandidate', () => { + it('keeps a recognized ancestor over a different agent helper below a non-agent', () => { + const candidates = [ + { pid: 101, ppid: 100, depth: 1, stat: 'S+', command: 'omp' }, + { pid: 102, ppid: 101, depth: 2, stat: 'S+', command: 'vendor-ui' }, + { pid: 103, ppid: 102, depth: 3, stat: 'S+', command: 'codex' } + ] + + expect(selectForegroundProcessCandidate(candidates)).toMatchObject({ + candidate: { pid: 101 }, + recognized: { agent: 'omp' } + }) + }) + + it('traverses non-foreground helpers when checking ancestry', () => { + const all = [ + { pid: 101, ppid: 100, depth: 1, stat: 'S+', command: 'omp' }, + { pid: 102, ppid: 101, depth: 2, stat: 'S', command: 'vendor-helper' }, + { pid: 103, ppid: 102, depth: 3, stat: 'S+', command: 'codex' } + ] + + expect(selectForegroundProcessCandidate([all[0], all[2]], all)).toMatchObject({ + candidate: { pid: 101 }, + recognized: { agent: 'omp' } + }) + }) + + it('refuses different recognized agents on sibling lineages', () => { + const candidates = [ + { pid: 101, ppid: 100, depth: 1, stat: 'S+', command: 'codex' }, + { pid: 102, ppid: 100, depth: 1, stat: 'S+', command: 'gemini' } + ] + + expect(selectForegroundProcessCandidate(candidates)).toBeNull() + }) + + it('keeps the deepest process when one recognized agent owns the lineage', () => { + const candidates = [ + { pid: 101, ppid: 100, depth: 1, stat: 'S+', command: 'node /opt/bin/codex' }, + { pid: 102, ppid: 101, depth: 2, stat: 'S+', command: '/opt/vendor/bin/codex' } + ] + + expect(selectForegroundProcessCandidate(candidates)).toMatchObject({ + candidate: { pid: 102 }, + recognized: { agent: 'codex' } + }) + }) +}) diff --git a/src/shared/foreground-process-selection.ts b/src/shared/foreground-process-selection.ts new file mode 100644 index 00000000000..294bb40e5d9 --- /dev/null +++ b/src/shared/foreground-process-selection.ts @@ -0,0 +1,80 @@ +import { + recognizeAgentProcessFromCommandLine, + type RecognizedAgentProcess +} from './agent-process-recognition' + +export type ForegroundProcessCandidate = { + pid: number + ppid: number + command: string + depth: number + stat?: string +} + +export type SelectedForegroundProcess = { + candidate: ForegroundProcessCandidate + recognized: RecognizedAgentProcess +} + +/** + * Select a foreground agent without letting a vendor helper steal an outer + * agent's identity when both names occur in one process lineage. + */ +export function selectForegroundProcessCandidate( + candidates: readonly ForegroundProcessCandidate[], + ancestryCandidates: readonly ForegroundProcessCandidate[] = candidates +): SelectedForegroundProcess | null { + const recognized = candidates.flatMap((candidate) => { + const agent = recognizeAgentProcessFromCommandLine(candidate.command) + return agent ? [{ candidate, recognized: agent }] : [] + }) + if (recognized.length === 0) { + return null + } + + const agentNames = new Set(recognized.map(({ recognized: agent }) => agent.agent)) + if (agentNames.size > 1) { + const candidatesByPid = new Map( + ancestryCandidates.map((candidate) => [candidate.pid, candidate]) + ) + const outer = [...recognized].sort( + (left, right) => left.candidate.depth - right.candidate.depth + )[0] + if ( + !outer || + !recognized.every((entry) => + isAncestorOrSelf(outer.candidate, entry.candidate, candidatesByPid) + ) + ) { + // Distinct sibling agents do not provide a trustworthy identity. + return null + } + return outer + } + + return recognized.reduce((best, current) => + foregroundCandidateScore(current.candidate) > foregroundCandidateScore(best.candidate) + ? current + : best + ) +} + +function foregroundCandidateScore(candidate: ForegroundProcessCandidate): number { + return (candidate.stat?.includes('+') ? 10_000 : 0) + candidate.depth +} + +function isAncestorOrSelf( + ancestor: ForegroundProcessCandidate, + descendant: ForegroundProcessCandidate, + candidatesByPid: ReadonlyMap +): boolean { + let currentPid = descendant.pid + while (currentPid !== ancestor.pid) { + const current = candidatesByPid.get(currentPid) + if (!current) { + return false + } + currentPid = current.ppid + } + return true +} From 104f9655e43540dacdf3c5189681e06a280ec64e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:53:48 -0700 Subject: [PATCH 072/398] perf(git): answer remote-URL questions from one subprocess, not one per remote (#18158) Four copies of the same loop ran `git remote` and then a serial `git remote get-url ` per remote to answer "which remote has this URL". On a repo with 58 remotes that is 59 subprocesses -- measured at 1083 ms -- for one question, and worktree create asks it several times. `git remote -v` answers for every remote from one child, reporting the same insteadOf-expanded first fetch URL `get-url` prints. The batched `cat-file --batch-check` branch-conflict probe decides from stdout, but its WSL route was unfenced, so a login-shell fallback printed the distro banner onto the stream it parses. That broke the one-line-per-ref contract, made every batch undecided, and fell straight back to one `show-ref` per remote -- the cost the batch exists to remove. Measured at 58 remotes / 4346 branches, spawns and wall time: push-target remote scan 59 -> 1 (1083 ms -> 8 ms) branch-conflict probe 60 -> 3 (984 ms -> 43 ms) configured push target 123 -> 6 (2707 ms -> 157 ms) --- src/main/git/exact-ref-probe.ts | 6 +- src/main/git/remote.test.ts | 59 ++++++ src/main/git/remote.ts | 38 ++-- ...repo-branch-conflict-batched-probe.test.ts | 102 ++++++++++ src/main/git/repo-branch-conflict.ts | 12 +- src/main/git/upstream.test.ts | 10 + .../worktree-push-target-reconciliation.ts | 25 +-- .../worktree-push-target-remote-scan.test.ts | 178 ++++++++++++++++++ .../ipc/worktree-push-target-setup.test.ts | 10 + src/main/ipc/worktree-push-target-setup.ts | 53 ++---- ...remote-push-target-materialization.test.ts | 11 ++ src/relay/git-handler-push-target.test.ts | 11 ++ src/relay/git-handler-push-target.ts | 33 ++-- src/shared/git-binary-compatibility.test.ts | 36 ++++ .../git-configured-branch-target.test.ts | 175 +++++++++++++++++ src/shared/git-configured-branch-target.ts | 34 ++-- src/shared/git-remote-url-index.test.ts | 112 +++++++++++ src/shared/git-remote-url-index.ts | 64 +++++++ 18 files changed, 841 insertions(+), 128 deletions(-) create mode 100644 src/main/git/repo-branch-conflict-batched-probe.test.ts create mode 100644 src/main/ipc/worktree-push-target-remote-scan.test.ts create mode 100644 src/shared/git-configured-branch-target.test.ts create mode 100644 src/shared/git-remote-url-index.test.ts create mode 100644 src/shared/git-remote-url-index.ts diff --git a/src/main/git/exact-ref-probe.ts b/src/main/git/exact-ref-probe.ts index 96bb421b8e8..ce89fac3f69 100644 --- a/src/main/git/exact-ref-probe.ts +++ b/src/main/git/exact-ref-probe.ts @@ -157,7 +157,11 @@ export async function probeAnyExactRefBatched( } catch { return { found: false, unknown: true } } - const lines = stdout.split('\n').filter((line) => line.trim().length > 0) + // Trim per line so a CRLF-translating host's `\r` does not become part of the type. + const lines = stdout + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0) // One line per input, in order; a short read means the batch never answered for the rest. if (lines.length !== safeRefs.length) { return { found: false, unknown: true } diff --git a/src/main/git/remote.test.ts b/src/main/git/remote.test.ts index 11ac2c21264..feb237eb18a 100644 --- a/src/main/git/remote.test.ts +++ b/src/main/git/remote.test.ts @@ -193,6 +193,17 @@ describe('git remote operations', () => { if (args[0] === 'remote' && args[1] === 'get-url' && args[2] === 'pr-pynickle-orca') { return { stdout: 'https://github.com/pynickle/orca.git\n', stderr: '' } } + if (args[0] === 'remote' && args[1] === '-v') { + return { + stdout: [ + 'origin\thttps://github.com/stablyai/orca.git (fetch)', + 'origin\thttps://github.com/stablyai/orca.git (push)', + 'pr-pynickle-orca\thttps://github.com/pynickle/orca.git (fetch)', + 'pr-pynickle-orca\thttps://github.com/pynickle/orca.git (push)' + ].join('\n'), + stderr: '' + } + } if (args[0] === 'remote') { return { stdout: 'origin\npr-pynickle-orca\n', stderr: '' } } @@ -207,6 +218,54 @@ describe('git remote operations', () => { ) }) + // Regression: normalizing a URL-valued push remote used to run `git remote` and then a + // serial `git remote get-url` per remote -- 59 subprocesses on a 58-remote repo. + it('normalizes a URL-valued push remote from one remote table read at 58 remotes', async () => { + const remotes = [ + { name: 'origin', url: 'https://github.com/stablyai/orca.git' }, + ...Array.from({ length: 56 }, (_, index) => ({ + name: `pr-user${index}-orca`, + url: `https://github.com/user${index}/orca.git` + })), + { name: 'pr-pynickle-orca', url: 'https://github.com/pynickle/orca.git' } + ] + gitExecFileAsyncMock.mockImplementation(async (args: string[]) => { + if (args[0] === 'symbolic-ref') { + return { stdout: 'imp/chinese-translation\n', stderr: '' } + } + if (args[0] === 'config' && args.includes('branch.imp/chinese-translation.remote')) { + return { stdout: 'https://github.com/pynickle/orca.git\n', stderr: '' } + } + if (args[0] === 'config' && args.includes('branch.imp/chinese-translation.merge')) { + return { stdout: 'refs/heads/imp/chinese-translation\n', stderr: '' } + } + if (args[0] === 'config') { + throw new Error(`config key is not set: ${args.join(' ')}`) + } + if (args[0] === 'remote' && args[1] === '-v') { + return { + stdout: remotes + .flatMap(({ name, url }) => [`${name}\t${url} (fetch)`, `${name}\t${url} (push)`]) + .join('\n'), + stderr: '' + } + } + if (args[0] === 'remote') { + throw new Error(`unexpected remote scan: ${args.join(' ')}`) + } + return { stdout: '', stderr: '' } + }) + + await gitPush('/repo', false) + + const remoteReads = gitExecFileAsyncMock.mock.calls.filter(([args]) => args[0] === 'remote') + expect(remoteReads.map(([args]) => args)).toEqual([['remote', '-v']]) + expect(gitExecFileAsyncMock).toHaveBeenLastCalledWith( + ['push', '--set-upstream', 'pr-pynickle-orca', 'HEAD:imp/chinese-translation'], + { cwd: '/repo' } + ) + }) + it('uses an explicit push target even when it differs from the local branch name', async () => { gitExecFileAsyncMock .mockResolvedValueOnce({ stdout: '', stderr: '' }) diff --git a/src/main/git/remote.ts b/src/main/git/remote.ts index 2baf3b77137..20cf8415d04 100644 --- a/src/main/git/remote.ts +++ b/src/main/git/remote.ts @@ -4,6 +4,7 @@ import { } from '../../shared/git-remote-error' import { resolveEffectiveGitUpstream } from '../../shared/git-effective-upstream' import { gitRefTargetsBranchOnRemote } from '../../shared/git-remote-branch-name' +import { findGitRemoteNameByFetchUrl } from '../../shared/git-remote-url-index' import type { GitPushTarget } from '../../shared/worktree/types' import type { GitRuntimeOptions } from './git-runtime-options' import { gitOptionsForWorktree } from './git-runtime-options' @@ -84,6 +85,8 @@ type ConfiguredPushRemote = { branchRemote: string | null } +// One `git remote -v` instead of `git remote` plus a serial `git remote get-url` +// per remote; both print the same insteadOf-expanded fetch URL. async function findRemoteNameForUrl( worktreePath: string, remoteUrl: string, @@ -91,30 +94,13 @@ async function findRemoteNameForUrl( ): Promise { try { const { stdout } = await gitExecFileAsync( - ['remote'], + ['remote', '-v'], gitOptionsForWorktree(worktreePath, options) ) - const remotes = stdout - .split(/\r?\n/) - .map((line) => line.trim()) - .filter(Boolean) - for (const remoteName of remotes) { - try { - const { stdout: urlStdout } = await gitExecFileAsync( - ['remote', 'get-url', remoteName], - gitOptionsForWorktree(worktreePath, options) - ) - if (urlStdout.trim() === remoteUrl) { - return remoteName - } - } catch { - // Ignore a remote that disappeared or has no fetch URL. - } - } + return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => candidateUrl === remoteUrl) } catch { return null } - return null } async function normalizePushRemote( @@ -141,11 +127,17 @@ async function getConfiguredPushRemote( if (!remote) { return null } + const normalizedRemote = await normalizePushRemote(worktreePath, remote, options) + // The two usually name the same URL; resolving it twice reads the remote table twice. + if (!branchRemote) { + return { remote: normalizedRemote, branchRemote: null } + } return { - remote: await normalizePushRemote(worktreePath, remote, options), - branchRemote: branchRemote - ? await normalizePushRemote(worktreePath, branchRemote, options) - : null + remote: normalizedRemote, + branchRemote: + branchRemote === remote + ? normalizedRemote + : await normalizePushRemote(worktreePath, branchRemote, options) } } diff --git a/src/main/git/repo-branch-conflict-batched-probe.test.ts b/src/main/git/repo-branch-conflict-batched-probe.test.ts new file mode 100644 index 00000000000..e6e672c65a2 --- /dev/null +++ b/src/main/git/repo-branch-conflict-batched-probe.test.ts @@ -0,0 +1,102 @@ +// Why: the batched `cat-file --batch-check` conflict probe decides from stdout, so a +// WSL login-shell fallback that prints the distro banner onto that stream desynchronizes +// the one-line-per-ref contract. Every batch then came back undecided and fell through to +// one `show-ref` subprocess per remote -- the cost the batch exists to remove. These tests +// pin the fence request and the resulting subprocess count at 58 remotes. + +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const { gitExecFileAsyncMock } = vi.hoisted(() => ({ gitExecFileAsyncMock: vi.fn() })) + +vi.mock('./runner', () => ({ gitExecFileAsync: gitExecFileAsyncMock })) + +import { getBranchConflictKind } from './repo-branch-conflict' + +const REMOTES = Array.from({ length: 58 }, (_, index) => `r${index}`) +const BRANCH = 'user/feature' +const WSL_BANNER = + 'Welcome to Ubuntu 24.04.1 LTS (GNU/Linux 5.15.167.4-microsoft-standard-WSL2 x86_64)\n' + + 'To run a command as administrator (user "root"), use "sudo ".\n' + +type GitExecOptions = { stdin?: string; captureWslLoginShellOutput?: boolean } + +/** + * Stand-in for a WSL-routed runner: the login shell prepends its banner to stdout unless + * the caller asked for the fenced form, which slices the payload back out. + */ +function installLoginShellRunner(): { argv: string[][] } { + const argv: string[][] = [] + gitExecFileAsyncMock.mockImplementation(async (args: string[], options: GitExecOptions = {}) => { + argv.push(args) + if (args[0] === 'rev-parse') { + throw new Error('local branch is absent') + } + if (args[0] === 'remote') { + return { stdout: `${WSL_BANNER}${REMOTES.join('\n')}\n`, stderr: '' } + } + if (args[0] === 'show-ref') { + throw Object.assign(new Error('missing ref'), { code: 1, stderr: '' }) + } + if (args[0] === 'cat-file') { + const payload = `${(options.stdin ?? '') + .split('\n') + .filter(Boolean) + .map((ref) => `${ref} missing`) + .join('\n')}\n` + return { + stdout: options.captureWslLoginShellOutput ? payload : `${WSL_BANNER}${payload}`, + stderr: '' + } + } + throw new Error(`unexpected git command: ${args.join(' ')}`) + }) + return { argv } +} + +function countSubcommand(argv: readonly string[][], subcommand: string): number { + return argv.filter((args) => args[0] === subcommand).length +} + +describe('getBranchConflictKind batched remote probe', () => { + beforeEach(() => { + gitExecFileAsyncMock.mockReset() + }) + + it('asks the WSL login shell to fence the batch payload it parses', async () => { + installLoginShellRunner() + + await getBranchConflictKind('/repo', BRANCH) + + const batchCall = gitExecFileAsyncMock.mock.calls.find(([args]) => args[0] === 'cat-file') + expect(batchCall?.[1]).toMatchObject({ captureWslLoginShellOutput: true }) + }) + + it('answers from one batched subprocess instead of one show-ref per remote', async () => { + const { argv } = installLoginShellRunner() + + await expect(getBranchConflictKind('/repo', BRANCH)).resolves.toBeNull() + + expect(countSubcommand(argv, 'cat-file')).toBe(1) + expect(countSubcommand(argv, 'show-ref')).toBe(0) + }) + + it('still falls back to per-ref probes when the batch itself fails', async () => { + gitExecFileAsyncMock.mockImplementation(async (args: string[]) => { + if (args[0] === 'rev-parse') { + throw new Error('local branch is absent') + } + if (args[0] === 'remote') { + return { stdout: `${REMOTES.join('\n')}\n`, stderr: '' } + } + if (args[0] === 'cat-file') { + throw new Error('cat-file is unavailable on this host') + } + if (args[0] === 'show-ref') { + return { stdout: '', stderr: '' } + } + throw new Error(`unexpected git command: ${args.join(' ')}`) + }) + + await expect(getBranchConflictKind('/repo', BRANCH)).resolves.toBe('remote') + }) +}) diff --git a/src/main/git/repo-branch-conflict.ts b/src/main/git/repo-branch-conflict.ts index c799d3770be..5c6e03b94b0 100644 --- a/src/main/git/repo-branch-conflict.ts +++ b/src/main/git/repo-branch-conflict.ts @@ -147,10 +147,12 @@ export function getBranchConflictKind( const execOptions = gitExecOptions(path, options) const runLocalGit = ( argv: string[], - commandOptions?: ExactRefProbeExecOptions & { stdin?: string } + commandOptions?: ExactRefProbeExecOptions & { stdin?: string }, + captureWslLoginShellOutput = false ): Promise<{ stdout: string }> => gitExecFileAsync(argv, { ...execOptions, + ...(captureWslLoginShellOutput ? { captureWslLoginShellOutput: true } : {}), ...(commandOptions?.maxBuffer === undefined ? {} : { maxBuffer: commandOptions.maxBuffer }), ...(commandOptions?.timeoutMs === undefined ? {} : { timeout: commandOptions.timeoutMs }), ...(commandOptions?.stdin === undefined ? {} : { stdin: commandOptions.stdin }) @@ -160,7 +162,13 @@ export function getBranchConflictKind( branchName, allowedBaseRef, {}, - (argv, commandOptions) => runLocalGit(argv, commandOptions) + // Why fenced: the batch decides from stdout, and a WSL login-shell fallback writes + // the distro's rc/motd banner to that same stream. The extra lines break the + // one-line-per-ref contract, so every batch came back undecided and fell through to + // one `show-ref` subprocess per remote -- the exact cost the batch exists to remove. + // `show-ref --verify --quiet` prints nothing and is read by exit code, so it needs + // no fence; the capture wrapper preserves the payload's exit status either way. + (argv, commandOptions) => runLocalGit(argv, commandOptions, true) ) } diff --git a/src/main/git/upstream.test.ts b/src/main/git/upstream.test.ts index 701808a8a31..b7644ad5543 100644 --- a/src/main/git/upstream.test.ts +++ b/src/main/git/upstream.test.ts @@ -363,6 +363,16 @@ describe('getUpstreamStatus', () => { if (args[0] === 'remote' && args[1] === 'get-url' && args[2] === 'pr-pynickle-orca') { return Promise.resolve({ stdout: 'https://github.com/pynickle/orca.git\n' }) } + if (args[0] === 'remote' && args[1] === '-v') { + return Promise.resolve({ + stdout: [ + 'origin\thttps://github.com/stablyai/orca.git (fetch)', + 'origin\thttps://github.com/stablyai/orca.git (push)', + 'pr-pynickle-orca\thttps://github.com/pynickle/orca.git (fetch)', + 'pr-pynickle-orca\thttps://github.com/pynickle/orca.git (push)' + ].join('\n') + }) + } if (args[0] === 'remote') { return Promise.resolve({ stdout: 'origin\npr-pynickle-orca\n' }) } diff --git a/src/main/ipc/worktree-push-target-reconciliation.ts b/src/main/ipc/worktree-push-target-reconciliation.ts index e59da7bdfe7..b28ae36b6f4 100644 --- a/src/main/ipc/worktree-push-target-reconciliation.ts +++ b/src/main/ipc/worktree-push-target-reconciliation.ts @@ -13,7 +13,7 @@ import { listWorktrees } from '../git/worktree' import type { SshGitProvider } from '../providers/ssh-git-provider' import type { GitPushTarget } from '../../shared/worktree/types' import { WORKTREE_ID_SEPARATOR, worktreeIdComparisonKey } from '../../shared/worktree/id' -import { iterateProcessOutputLines } from '../../shared/process-output-field-scanner' +import { parseGitRemoteFetchUrls } from '../../shared/git-remote-url-index' import { findWorktreeMetaReferencingRemote, hasBranchConfigUsingRemote, @@ -44,26 +44,9 @@ async function listPrRemoteCandidates( } catch { return [] } - const candidates = new Map() - for (const line of iterateProcessOutputLines(stdout)) { - const parsed = parseRemoteVerboseLine(line) - if (parsed?.direction === 'fetch' && isOrcaGeneratedPrRemoteName(parsed.name)) { - candidates.set(parsed.name, parsed.url) - } - } - return [...candidates.entries()].map(([name, url]) => ({ name, url })) -} - -function parseRemoteVerboseLine( - line: string -): { name: string; url: string; direction: 'fetch' | 'push' } | null { - const tabIndex = line.indexOf('\t') - if (tabIndex === -1) { - return null - } - const name = line.slice(0, tabIndex) - const match = /^(.*) \((fetch|push)\)$/.exec(line.slice(tabIndex + 1).trim()) - return match ? { name, url: match[1], direction: match[2] as 'fetch' | 'push' } : null + return [...parseGitRemoteFetchUrls(stdout)] + .filter(([name]) => isOrcaGeneratedPrRemoteName(name)) + .map(([name, url]) => ({ name, url })) } async function shouldReclaimPrRemote( diff --git a/src/main/ipc/worktree-push-target-remote-scan.test.ts b/src/main/ipc/worktree-push-target-remote-scan.test.ts new file mode 100644 index 00000000000..f2392c5e71a --- /dev/null +++ b/src/main/ipc/worktree-push-target-remote-scan.test.ts @@ -0,0 +1,178 @@ +// Why: `findRemoteForUrl` used to run `git remote` and then one serial +// `git remote get-url` per remote. These tests pin both halves of the fix: the +// subprocess count at 58 remotes, and result-for-result parity with the old scan +// across the remote shapes a real repo produces. + +import { describe, expect, it } from 'vitest' +import { parseGitHubOwnerRepo } from '../github/gh-utils' +import { findRemoteForUrl } from './worktree-push-target-setup' +import type { GitRemoteExec } from './worktree-push-target-cleanup' + +const SSH_FORK = 'git@github.com:contributor/orca.git' +const HTTPS_FORK = 'https://github.com/contributor/orca.git' +const GITLAB_FORK = 'https://gitlab.com/contributor/orca.git' +const UPSTREAM = 'https://github.com/stablyai/orca.git' + +type RemoteRow = { name: string; fetchUrl: string; pushUrl?: string } + +type CountingExec = GitRemoteExec & { spawns: string[][] } + +function makeExec(remotes: readonly RemoteRow[]): CountingExec { + const spawns: string[][] = [] + const exec: GitRemoteExec = async (args: string[]) => { + spawns.push(args) + if (args[0] === 'remote' && args.length === 1) { + return { stdout: `${remotes.map((remote) => remote.name).join('\n')}\n` } + } + if (args[0] === 'remote' && args[1] === '-v') { + return { + stdout: remotes + .flatMap((remote) => [ + `${remote.name}\t${remote.fetchUrl} (fetch)`, + `${remote.name}\t${remote.pushUrl ?? remote.fetchUrl} (push)` + ]) + .join('\n') + } + } + if (args[0] === 'remote' && args[1] === 'get-url') { + const match = remotes.find((remote) => remote.name === args[2]) + if (!match) { + throw new Error(`No such remote ${args[2]}`) + } + return { stdout: `${match.fetchUrl}\n` } + } + throw new Error(`unexpected git command: ${args.join(' ')}`) + } + return Object.assign(exec, { spawns }) +} + +/** The pre-fix scan, kept as the oracle the batched form must reproduce exactly. */ +async function findRemoteForUrlPerRemote( + execGit: GitRemoteExec, + repoPath: string, + remoteUrl: string +): Promise { + const target = parseGitHubOwnerRepo(remoteUrl) + try { + const { stdout } = await execGit(['remote'], repoPath) + for (const remote of stdout + .split(/\r?\n/) + .map((line) => line.trim()) + .filter(Boolean)) { + try { + const { stdout: urlStdout } = await execGit(['remote', 'get-url', remote], repoPath) + const candidateUrl = urlStdout.trim() + const candidate = parseGitHubOwnerRepo(candidateUrl) + if ( + target && + candidate && + target.owner.toLowerCase() === candidate.owner.toLowerCase() && + target.repo.toLowerCase() === candidate.repo.toLowerCase() + ) { + return remote + } + if (candidateUrl === remoteUrl) { + return remote + } + } catch { + // Ignore a remote that disappeared or has no fetch URL. + } + } + } catch { + return null + } + return null +} + +const fiftyEightRemotes: RemoteRow[] = [ + { name: 'origin', fetchUrl: UPSTREAM }, + ...Array.from({ length: 56 }, (_, index) => ({ + name: `pr-user${index}-orca`, + fetchUrl: `https://github.com/user${index}/orca.git` + })), + { name: 'pr-contributor-orca', fetchUrl: SSH_FORK } +] + +const matrix: { name: string; remotes: RemoteRow[]; lookupUrl: string }[] = [ + { name: 'no remotes', remotes: [], lookupUrl: SSH_FORK }, + { + name: 'one matching remote', + remotes: [{ name: 'origin', fetchUrl: SSH_FORK }], + lookupUrl: SSH_FORK + }, + { + name: 'one non-matching remote', + remotes: [{ name: 'origin', fetchUrl: UPSTREAM }], + lookupUrl: SSH_FORK + }, + { name: '58 remotes, match last', remotes: fiftyEightRemotes, lookupUrl: SSH_FORK }, + { + name: '58 remotes, no match', + remotes: fiftyEightRemotes, + lookupUrl: 'https://github.com/nobody/other.git' + }, + { + name: 'duplicate URLs on two remotes', + remotes: [ + { name: 'origin', fetchUrl: UPSTREAM }, + { name: 'fork-a', fetchUrl: SSH_FORK }, + { name: 'fork-b', fetchUrl: SSH_FORK } + ], + lookupUrl: SSH_FORK + }, + { + name: 'fetch and push URLs differ', + remotes: [{ name: 'split', fetchUrl: SSH_FORK, pushUrl: HTTPS_FORK }], + lookupUrl: SSH_FORK + }, + { + name: 'SSH-form lookup against an HTTPS-form remote', + remotes: [ + { name: 'origin', fetchUrl: UPSTREAM }, + { name: 'fork', fetchUrl: HTTPS_FORK } + ], + lookupUrl: SSH_FORK + }, + { + name: 'HTTPS-form lookup against an SSH-form remote', + remotes: [ + { name: 'origin', fetchUrl: UPSTREAM }, + { name: 'fork', fetchUrl: SSH_FORK } + ], + lookupUrl: HTTPS_FORK + }, + { + name: 'non-GitHub provider matches only on the exact URL', + remotes: [{ name: 'gitlab-fork', fetchUrl: GITLAB_FORK }], + lookupUrl: GITLAB_FORK + }, + { + name: 'non-GitHub provider with a different host does not match', + remotes: [{ name: 'gitlab-fork', fetchUrl: GITLAB_FORK }], + lookupUrl: 'https://bitbucket.org/contributor/orca.git' + } +] + +describe('findRemoteForUrl', () => { + it.each(matrix)('matches the per-remote scan for $name', async ({ remotes, lookupUrl }) => { + const expected = await findRemoteForUrlPerRemote(makeExec(remotes), '/repo', lookupUrl) + await expect(findRemoteForUrl(makeExec(remotes), '/repo', lookupUrl)).resolves.toBe(expected) + }) + + it('answers from one subprocess at 58 remotes instead of one per remote', async () => { + const legacyExec = makeExec(fiftyEightRemotes) + await findRemoteForUrlPerRemote(legacyExec, '/repo', 'https://github.com/nobody/other.git') + expect(legacyExec.spawns).toHaveLength(fiftyEightRemotes.length + 1) + + const exec = makeExec(fiftyEightRemotes) + await findRemoteForUrl(exec, '/repo', 'https://github.com/nobody/other.git') + expect(exec.spawns).toEqual([['remote', '-v']]) + }) + + it('returns null when the remote table cannot be read', async () => { + const failing: GitRemoteExec = async () => { + throw new Error('not a git repository') + } + await expect(findRemoteForUrl(failing, '/repo', SSH_FORK)).resolves.toBeNull() + }) +}) diff --git a/src/main/ipc/worktree-push-target-setup.test.ts b/src/main/ipc/worktree-push-target-setup.test.ts index 2d9670c2f9b..ef1e44aa923 100644 --- a/src/main/ipc/worktree-push-target-setup.test.ts +++ b/src/main/ipc/worktree-push-target-setup.test.ts @@ -16,6 +16,13 @@ const REPO = '/repo-root' const FORK_SSH = 'git@github.com:contributor/orca.git' const FORK_HTTPS = 'https://github.com/contributor/orca.git' +/** Real `git remote -v` shape: a fetch row and a push row per remote, tab-separated. */ +export function renderRemoteVerbose(remotes: Record): string { + return Object.entries(remotes) + .flatMap(([name, url]) => [`${name}\t${url} (fetch)`, `${name}\t${url} (push)`]) + .join('\n') +} + // A stateful fake git: `remotes` maps name -> url. `remote add` mutates it so // later lookups see the new remote, matching real git behavior. Defaults // `symbolic-ref --short HEAD` to a real branch name, since a worktree's HEAD @@ -31,6 +38,9 @@ function makeRepoExec( if (args[0] === 'remote' && args.length === 1) { return { stdout: Object.keys(remotes).join('\n'), stderr: '' } } + if (args[0] === 'remote' && args[1] === '-v' && args.length === 2) { + return { stdout: renderRemoteVerbose(remotes), stderr: '' } + } if (args[0] === 'remote' && args[1] === 'get-url') { const url = remotes[args[2]!] if (!url) { diff --git a/src/main/ipc/worktree-push-target-setup.ts b/src/main/ipc/worktree-push-target-setup.ts index 59ef4c8fea5..c4a82d94933 100644 --- a/src/main/ipc/worktree-push-target-setup.ts +++ b/src/main/ipc/worktree-push-target-setup.ts @@ -5,53 +5,33 @@ // repo. The store-aware ownership decision stays with the caller via a predicate. import type { GitPushTarget } from '../../shared/worktree/types' -import { parseGitHubOwnerRepo } from '../github/gh-utils' -import type { GitRemoteExec } from './worktree-push-target-cleanup' +import { findGitRemoteNameByFetchUrl } from '../../shared/git-remote-url-index' +import { sameGitHubRemoteUrl, type GitRemoteExec } from './worktree-push-target-cleanup' import { buildNarrowForkFetchRefspec, ensureRemoteTracksBranchNarrowly } from '../git/fork-remote-refspec' +// One `git remote -v` replaces `git remote` plus a serial `git remote get-url` per +// remote -- 59 subprocesses at 58 remotes, on every push-target resolution (#17914). export async function findRemoteForUrl( execGit: GitRemoteExec, repoPath: string, remoteUrl: string ): Promise { - const target = parseGitHubOwnerRepo(remoteUrl) try { - const { stdout } = await execGit(['remote'], repoPath) - for (const remote of stdout - .split(/\r?\n/) - .map((line) => line.trim()) - .filter(Boolean)) { - try { - const { stdout: urlStdout } = await execGit(['remote', 'get-url', remote], repoPath) - const candidateUrl = urlStdout.trim() - const candidate = parseGitHubOwnerRepo(candidateUrl) - if ( - target && - candidate && - target.owner.toLowerCase() === candidate.owner.toLowerCase() && - target.repo.toLowerCase() === candidate.repo.toLowerCase() - ) { - return remote - } - if (candidateUrl === remoteUrl) { - return remote - } - } catch { - // Ignore a remote that disappeared or has no fetch URL. - } - } + const { stdout } = await execGit(['remote', '-v'], repoPath) + return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => + sameGitHubRemoteUrl(candidateUrl, remoteUrl) + ) } catch { return null } - return null } // O(1) probe used before materializing on demand (push/pull/fetch/fast-forward): -// a single `remote get-url ` avoids the O(remotes) `findRemoteForUrl` scan -// once a fork remote already exists under its expected name (#17828). +// a single `remote get-url ` skips the whole-remote-table read once a fork +// remote already exists under its expected name (#17828). export async function remoteAlreadyMatchesUrl( execGit: GitRemoteExec, repoPath: string, @@ -60,18 +40,7 @@ export async function remoteAlreadyMatchesUrl( ): Promise { try { const { stdout } = await execGit(['remote', 'get-url', remoteName], repoPath) - const candidateUrl = stdout.trim() - if (candidateUrl === remoteUrl) { - return true - } - const target = parseGitHubOwnerRepo(remoteUrl) - const candidate = parseGitHubOwnerRepo(candidateUrl) - return Boolean( - target && - candidate && - target.owner.toLowerCase() === candidate.owner.toLowerCase() && - target.repo.toLowerCase() === candidate.repo.toLowerCase() - ) + return sameGitHubRemoteUrl(stdout.trim(), remoteUrl) } catch { return false } diff --git a/src/main/ipc/worktree-remote-push-target-materialization.test.ts b/src/main/ipc/worktree-remote-push-target-materialization.test.ts index 68ed98ab35c..695baa10dec 100644 --- a/src/main/ipc/worktree-remote-push-target-materialization.test.ts +++ b/src/main/ipc/worktree-remote-push-target-materialization.test.ts @@ -553,6 +553,17 @@ describe('materializeWorktreePushTargetRemoteSsh', () => { } throw new Error('No such remote') } + if (args[0] === 'remote' && args[1] === '-v') { + return { + stdout: [ + 'origin\thttps://github.com/stablyai/orca.git (fetch)', + 'origin\thttps://github.com/stablyai/orca.git (push)', + `${SIBLING_REMOTE}\t${FORK_URL} (fetch)`, + `${SIBLING_REMOTE}\t${FORK_URL} (push)` + ].join('\n'), + stderr: '' + } + } if (args[0] === 'remote' && args.length === 1) { return { stdout: `origin\n${SIBLING_REMOTE}\n`, stderr: '' } } diff --git a/src/relay/git-handler-push-target.test.ts b/src/relay/git-handler-push-target.test.ts index c040a645a16..b6fe5e96ad0 100644 --- a/src/relay/git-handler-push-target.test.ts +++ b/src/relay/git-handler-push-target.test.ts @@ -46,6 +46,17 @@ function gitForConfig(config: { } return { stdout: `${config.base ?? ''}\n`, stderr: '' } } + if (args[0] === 'remote' && args[1] === '-v') { + return { + stdout: (config.remotes ?? []) + .flatMap((name) => { + const url = config.remoteUrls?.[name] ?? '' + return [`${name}\t${url} (fetch)`, `${name}\t${url} (push)`] + }) + .join('\n'), + stderr: '' + } + } if (args[0] === 'remote' && args.length === 1) { return { stdout: `${config.remotes?.join('\n') ?? ''}\n`, stderr: '' } } diff --git a/src/relay/git-handler-push-target.ts b/src/relay/git-handler-push-target.ts index 40765491c56..d81111d45d3 100644 --- a/src/relay/git-handler-push-target.ts +++ b/src/relay/git-handler-push-target.ts @@ -1,5 +1,6 @@ import { assertGitPushTargetShape } from '../shared/git-push-target-validation' import { gitRefTargetsBranchOnRemote } from '../shared/git-remote-branch-name' +import { findGitRemoteNameByFetchUrl } from '../shared/git-remote-url-index' import type { GitPushTarget } from '../shared/worktree/types' type RelayGit = (args: string[], cwd: string) => Promise<{ stdout: string; stderr: string }> @@ -67,31 +68,19 @@ type ConfiguredPushRemote = { branchRemote: string | null } +// Host-side twin of `src/main/git/remote.ts`: one `git remote -v` instead of +// `git remote` plus a serial `git remote get-url` per remote. async function findRemoteNameForUrl( git: RelayGit, worktreePath: string, remoteUrl: string ): Promise { try { - const { stdout } = await git(['remote'], worktreePath) - const remotes = stdout - .split(/\r?\n/) - .map((line) => line.trim()) - .filter(Boolean) - for (const remoteName of remotes) { - try { - const { stdout: urlStdout } = await git(['remote', 'get-url', remoteName], worktreePath) - if (urlStdout.trim() === remoteUrl) { - return remoteName - } - } catch { - // Ignore a remote that disappeared or has no fetch URL. - } - } + const { stdout } = await git(['remote', '-v'], worktreePath) + return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => candidateUrl === remoteUrl) } catch { return null } - return null } async function normalizePushRemote( @@ -120,9 +109,17 @@ async function getConfiguredPushRemote( if (!remote) { return null } + const normalizedRemote = await normalizePushRemote(git, worktreePath, remote) + // The two usually name the same URL; resolving it twice reads the remote table twice. + if (!branchRemote) { + return { remote: normalizedRemote, branchRemote: null } + } return { - remote: await normalizePushRemote(git, worktreePath, remote), - branchRemote: branchRemote ? await normalizePushRemote(git, worktreePath, branchRemote) : null + remote: normalizedRemote, + branchRemote: + branchRemote === remote + ? normalizedRemote + : await normalizePushRemote(git, worktreePath, branchRemote) } } diff --git a/src/shared/git-binary-compatibility.test.ts b/src/shared/git-binary-compatibility.test.ts index a61e64464da..5ad37398d14 100644 --- a/src/shared/git-binary-compatibility.test.ts +++ b/src/shared/git-binary-compatibility.test.ts @@ -15,6 +15,7 @@ import { isUnsupportedWorktreeListZError } from './git-worktree-command-capabilities' import { gitCredentialPromptGuardEnv } from './git-credential-prompt-env' +import { parseGitRemoteFetchUrls } from './git-remote-url-index' import { GIT_HISTORY_COMMIT_FORMAT, parseGitHistoryLog } from './git-history-log-parser' import { githubPullRequestHeadLocalRef, @@ -214,6 +215,41 @@ describeBinaryCompatibility('real Git binary compatibility', () => { await expect(runGit(['merge-base', '--end-of-options', head, unrelated])).rejects.toBeDefined() }) + // Why pin this: Orca answers "which remote has this URL" from one `git remote -v` + // instead of one `git remote get-url` per remote. That is only equivalent if both + // commands report the same URL — the insteadOf-expanded first `remote..url`, + // which a raw config read does not produce — on every supported Git. + it('reports the same fetch URL from remote -v as from remote get-url', async () => { + await runGit(['config', 'url.git@example.invalid:.insteadOf', 'https://example.invalid/']) + await runGit(['remote', 'add', 'compat-single', 'https://example.invalid/a/repo.git']) + await runGit(['remote', 'add', 'compat-multi', 'https://example.invalid/b/repo.git']) + await runGit([ + 'config', + '--add', + 'remote.compat-multi.url', + 'https://example.invalid/b2/repo.git' + ]) + await runGit([ + 'config', + 'remote.compat-multi.pushurl', + 'https://push.example.invalid/b/repo.git' + ]) + try { + const fetchUrls = parseGitRemoteFetchUrls((await runGit(['remote', '-v'])).stdout) + for (const name of ['compat-single', 'compat-multi']) { + const getUrl = (await runGit(['remote', 'get-url', name])).stdout.trim() + expect(fetchUrls.get(name)).toBe(getUrl) + } + expect(fetchUrls.get('compat-single')).toBe('git@example.invalid:a/repo.git') + // A `pushurl` must not displace the fetch URL the scan compares against. + expect(fetchUrls.get('compat-multi')).toBe('git@example.invalid:b/repo.git') + } finally { + await runGit(['remote', 'remove', 'compat-single']) + await runGit(['remote', 'remove', 'compat-multi']) + await runGit(['config', '--unset-all', 'url.git@example.invalid:.insteadOf']) + } + }) + it('recognizes ref and merge-tree compatibility boundaries', async () => { const fetchHeadPath = join(repoPath, '.git', 'FETCH_HEAD') await writeFile(fetchHeadPath, 'sentinel\n') diff --git a/src/shared/git-configured-branch-target.test.ts b/src/shared/git-configured-branch-target.test.ts new file mode 100644 index 00000000000..a599a53bbb8 --- /dev/null +++ b/src/shared/git-configured-branch-target.test.ts @@ -0,0 +1,175 @@ +// Why: resolving a URL-valued `branch..remote` (or `remote.pushDefault`) to a +// remote name used to cost `git remote` plus one serial `git remote get-url` per remote. +// `hasConfiguredBranchPushTarget` resolves up to two of them, so a 58-remote repo paid +// up to 118 subprocesses for one question. These tests pin the count and result parity. + +import { describe, expect, it } from 'vitest' +import { + getConfiguredBranchRemoteUpstream, + hasConfiguredBranchPushTarget +} from './git-configured-branch-target' + +const BRANCH = 'imp/translation' +const FORK_URL = 'https://github.com/contributor/orca.git' +const UPSTREAM_URL = 'https://github.com/stablyai/orca.git' + +type RemoteRow = { name: string; fetchUrl: string; pushUrl?: string } + +type Fixture = { + remotes: readonly RemoteRow[] + config: Readonly> +} + +function makeRunner(fixture: Fixture): { + runGit: (args: string[]) => Promise<{ stdout: string }> + spawns: string[][] +} { + const spawns: string[][] = [] + const runGit = async (args: string[]): Promise<{ stdout: string }> => { + spawns.push(args) + if (args[0] === 'config' && args[1] === '--get') { + const value = fixture.config[args[2]] + if (value === undefined) { + throw Object.assign(new Error('config key is not set'), { code: 1 }) + } + return { stdout: `${value}\n` } + } + if (args[0] === 'remote' && args[1] === '-v') { + return { + stdout: fixture.remotes + .flatMap((remote) => [ + `${remote.name}\t${remote.fetchUrl} (fetch)`, + `${remote.name}\t${remote.pushUrl ?? remote.fetchUrl} (push)` + ]) + .join('\n') + } + } + if (args[0] === 'remote' && args.length === 1) { + return { stdout: `${fixture.remotes.map((remote) => remote.name).join('\n')}\n` } + } + if (args[0] === 'remote' && args[1] === 'get-url') { + const match = fixture.remotes.find((remote) => remote.name === args[2]) + if (!match) { + throw new Error(`No such remote ${args[2]}`) + } + return { stdout: `${match.fetchUrl}\n` } + } + throw new Error(`unexpected git command: ${args.join(' ')}`) + } + return { runGit, spawns } +} + +const fiftyEightRemotes: RemoteRow[] = [ + { name: 'origin', fetchUrl: UPSTREAM_URL }, + ...Array.from({ length: 56 }, (_, index) => ({ + name: `pr-user${index}-orca`, + fetchUrl: `https://github.com/user${index}/orca.git` + })), + { name: 'pr-contributor-orca', fetchUrl: FORK_URL } +] + +describe('hasConfiguredBranchPushTarget', () => { + it('resolves both URL-valued remotes from one remote table read at 58 remotes', async () => { + const { runGit, spawns } = makeRunner({ + remotes: fiftyEightRemotes, + config: { + [`branch.${BRANCH}.pushRemote`]: FORK_URL, + [`branch.${BRANCH}.remote`]: FORK_URL, + [`branch.${BRANCH}.merge`]: `refs/heads/${BRANCH}` + } + }) + + await expect(hasConfiguredBranchPushTarget(runGit, BRANCH)).resolves.toBe(true) + + // Both the push remote and the branch remote name the same URL, so one table read answers. + expect(spawns.filter((args) => args[0] === 'remote')).toEqual([['remote', '-v']]) + expect(spawns.filter((args) => args[1] === 'get-url')).toEqual([]) + }) + + it('keeps the not-set case false when no remote is configured', async () => { + const { runGit } = makeRunner({ remotes: fiftyEightRemotes, config: {} }) + await expect(hasConfiguredBranchPushTarget(runGit, BRANCH)).resolves.toBe(false) + }) + + it('keeps the URL itself as the remote name when nothing matches', async () => { + const { runGit } = makeRunner({ + remotes: [{ name: 'origin', fetchUrl: UPSTREAM_URL }], + config: { + [`branch.${BRANCH}.pushRemote`]: FORK_URL, + [`branch.${BRANCH}.remote`]: FORK_URL, + [`branch.${BRANCH}.merge`]: 'refs/heads/other' + } + }) + // Unchanged no-match fallback: both remotes stay the raw URL, so they still agree + // and the differently named merge branch is still pushable. + await expect(hasConfiguredBranchPushTarget(runGit, BRANCH)).resolves.toBe(true) + }) +}) + +describe('getConfiguredBranchRemoteUpstream', () => { + const remoteTrackingRefExists = async (): Promise => true + + it('picks the first remote holding a duplicated URL', async () => { + const { runGit, spawns } = makeRunner({ + remotes: [ + { name: 'origin', fetchUrl: UPSTREAM_URL }, + { name: 'fork-a', fetchUrl: FORK_URL }, + { name: 'fork-b', fetchUrl: FORK_URL } + ], + config: { + [`branch.${BRANCH}.remote`]: FORK_URL, + [`branch.${BRANCH}.merge`]: `refs/heads/${BRANCH}` + } + }) + + await expect( + getConfiguredBranchRemoteUpstream(runGit, BRANCH, remoteTrackingRefExists) + ).resolves.toEqual({ + upstreamName: `fork-a/${BRANCH}`, + remoteName: 'fork-a', + branchName: BRANCH, + isConfiguredUpstream: false + }) + expect(spawns.filter((args) => args[0] === 'remote')).toEqual([['remote', '-v']]) + }) + + it('ignores a push URL when fetch and push differ', async () => { + const { runGit } = makeRunner({ + remotes: [{ name: 'split', fetchUrl: UPSTREAM_URL, pushUrl: FORK_URL }], + config: { + [`branch.${BRANCH}.remote`]: FORK_URL, + [`branch.${BRANCH}.merge`]: `refs/heads/${BRANCH}` + } + }) + await expect( + getConfiguredBranchRemoteUpstream(runGit, BRANCH, remoteTrackingRefExists) + ).resolves.toBeNull() + }) + + it('returns null with no remotes at all', async () => { + const { runGit } = makeRunner({ + remotes: [], + config: { + [`branch.${BRANCH}.remote`]: FORK_URL, + [`branch.${BRANCH}.merge`]: `refs/heads/${BRANCH}` + } + }) + await expect( + getConfiguredBranchRemoteUpstream(runGit, BRANCH, remoteTrackingRefExists) + ).resolves.toBeNull() + }) + + it('keeps a plain named remote untouched', async () => { + const { runGit, spawns } = makeRunner({ + remotes: fiftyEightRemotes, + config: { + [`branch.${BRANCH}.remote`]: 'origin', + [`branch.${BRANCH}.merge`]: `refs/heads/${BRANCH}` + } + }) + await expect( + getConfiguredBranchRemoteUpstream(runGit, BRANCH, remoteTrackingRefExists) + ).resolves.toMatchObject({ remoteName: 'origin' }) + expect(spawns.filter((args) => args[0] === 'remote')).toEqual([]) + }) +}) diff --git a/src/shared/git-configured-branch-target.ts b/src/shared/git-configured-branch-target.ts index 5c28f023814..9faca91dce7 100644 --- a/src/shared/git-configured-branch-target.ts +++ b/src/shared/git-configured-branch-target.ts @@ -1,4 +1,5 @@ import { gitRefTargetsBranchOnRemote } from './git-remote-branch-name' +import { findGitRemoteNameByFetchUrl } from './git-remote-url-index' type GitCommandRunner = (args: string[]) => Promise<{ stdout: string }> @@ -25,30 +26,18 @@ function isUrlValuedRemote(remote: string): boolean { return /^[A-Za-z][A-Za-z0-9+.-]*:\/\//.test(remote) || /^[^@/:]+@[^:]+:.+/.test(remote) } +// `hasConfiguredBranchPushTarget` resolves up to two URL-valued remotes, so the old +// per-remote `get-url` scan cost up to 2 x (1 + remotes) subprocesses per call. async function findRemoteNameForUrl( runGit: GitCommandRunner, remoteUrl: string ): Promise { try { - const { stdout } = await runGit(['remote']) - const remotes = stdout - .split(/\r?\n/) - .map((line) => line.trim()) - .filter(Boolean) - for (const remoteName of remotes) { - try { - const { stdout: urlStdout } = await runGit(['remote', 'get-url', remoteName]) - if (urlStdout.trim() === remoteUrl) { - return remoteName - } - } catch { - // Ignore a remote that disappeared or has no fetch URL. - } - } + const { stdout } = await runGit(['remote', '-v']) + return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => candidateUrl === remoteUrl) } catch { return null } - return null } export async function getConfiguredBranchRemoteUpstream( @@ -101,11 +90,14 @@ export async function hasConfiguredBranchPushTarget( const pushRemoteName = isUrlValuedRemote(remote) ? ((await findRemoteNameForUrl(runGit, remote)) ?? remote) : remote - const branchRemoteName = branchRemote - ? isUrlValuedRemote(branchRemote) - ? ((await findRemoteNameForUrl(runGit, branchRemote)) ?? branchRemote) - : branchRemote - : null + // The two usually name the same URL; resolving it twice reads the remote table twice. + const branchRemoteName = !branchRemote + ? null + : branchRemote === remote + ? pushRemoteName + : isUrlValuedRemote(branchRemote) + ? ((await findRemoteNameForUrl(runGit, branchRemote)) ?? branchRemote) + : branchRemote if (gitRefTargetsBranchOnRemote(baseRef, pushRemoteName, branchName)) { return false } diff --git a/src/shared/git-remote-url-index.test.ts b/src/shared/git-remote-url-index.test.ts new file mode 100644 index 00000000000..e798e136ae4 --- /dev/null +++ b/src/shared/git-remote-url-index.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it } from 'vitest' +import { + findGitRemoteNameByFetchUrl, + parseGitRemoteFetchUrls, + parseGitRemoteVerboseLine +} from './git-remote-url-index' + +const SSH_URL = 'git@github.com:contributor/orca.git' +const HTTPS_URL = 'https://github.com/contributor/orca.git' + +function verbose(rows: readonly (readonly [string, string])[]): string { + return rows.map(([name, url]) => `${name}\t${url}`).join('\n') +} + +describe('parseGitRemoteVerboseLine', () => { + it('reads the name, URL and direction', () => { + expect(parseGitRemoteVerboseLine(`origin\t${HTTPS_URL} (fetch)`)).toEqual({ + name: 'origin', + url: HTTPS_URL, + direction: 'fetch' + }) + }) + + it('keeps a URL that itself contains spaces and parentheses', () => { + const url = '/tmp/my repo (mirror)' + expect(parseGitRemoteVerboseLine(`local\t${url} (push)`)).toEqual({ + name: 'local', + url, + direction: 'push' + }) + }) + + it('rejects the URL-less row git prints for a pushurl-only remote', () => { + expect(parseGitRemoteVerboseLine('pushonly\t')).toBeNull() + expect(parseGitRemoteVerboseLine('not a remote row')).toBeNull() + }) +}) + +describe('parseGitRemoteFetchUrls', () => { + it('returns nothing for a repo with no remotes', () => { + expect([...parseGitRemoteFetchUrls('')]).toEqual([]) + }) + + it('keeps only fetch rows, in git remote order', () => { + const stdout = verbose([ + ['a', `${SSH_URL} (fetch)`], + ['a', `${SSH_URL} (push)`], + ['b', `${HTTPS_URL} (fetch)`], + ['b', 'https://github.com/contributor/other.git (push)'] + ]) + expect([...parseGitRemoteFetchUrls(stdout)]).toEqual([ + ['a', SSH_URL], + ['b', HTTPS_URL] + ]) + }) + + it('parses CRLF output', () => { + const stdout = `a\t${SSH_URL} (fetch)\r\na\t${SSH_URL} (push)\r\n` + expect([...parseGitRemoteFetchUrls(stdout)]).toEqual([['a', SSH_URL]]) + }) + + it('takes the first URL of a multi-URL remote, matching remote get-url', () => { + const stdout = verbose([ + ['multi', `${SSH_URL} (fetch)`], + ['multi', `${SSH_URL} (push)`], + ['multi', `${HTTPS_URL} (push)`] + ]) + expect(parseGitRemoteFetchUrls(stdout).get('multi')).toBe(SSH_URL) + }) + + it('scales to 58 remotes without losing order', () => { + const rows = Array.from({ length: 58 }, (_, index) => [ + `r${index}`, + `https://example.com/o${index}/repo.git` + ]) + const stdout = rows + .flatMap(([name, url]) => [`${name}\t${url} (fetch)`, `${name}\t${url} (push)`]) + .join('\n') + const parsed = [...parseGitRemoteFetchUrls(stdout)] + expect(parsed).toHaveLength(58) + expect(parsed[0]).toEqual(['r0', 'https://example.com/o0/repo.git']) + expect(parsed[57]).toEqual(['r57', 'https://example.com/o57/repo.git']) + }) +}) + +describe('findGitRemoteNameByFetchUrl', () => { + const stdout = verbose([ + ['origin', 'https://github.com/stablyai/orca.git (fetch)'], + ['origin', 'https://github.com/stablyai/orca.git (push)'], + ['first-fork', `${SSH_URL} (fetch)`], + ['first-fork', `${SSH_URL} (push)`], + ['second-fork', `${SSH_URL} (fetch)`], + ['second-fork', `${SSH_URL} (push)`] + ]) + + it('returns the first remote holding a duplicated URL', () => { + expect(findGitRemoteNameByFetchUrl(stdout, (url) => url === SSH_URL)).toBe('first-fork') + }) + + it('returns null when nothing matches', () => { + expect(findGitRemoteNameByFetchUrl(stdout, (url) => url === HTTPS_URL)).toBeNull() + }) + + it('ignores push URLs when fetch and push differ', () => { + const split = verbose([ + ['split', `${SSH_URL} (fetch)`], + ['split', `${HTTPS_URL} (push)`] + ]) + expect(findGitRemoteNameByFetchUrl(split, (url) => url === SSH_URL)).toBe('split') + expect(findGitRemoteNameByFetchUrl(split, (url) => url === HTTPS_URL)).toBeNull() + }) +}) diff --git a/src/shared/git-remote-url-index.ts b/src/shared/git-remote-url-index.ts new file mode 100644 index 00000000000..ec4f2b4a509 --- /dev/null +++ b/src/shared/git-remote-url-index.ts @@ -0,0 +1,64 @@ +// Why: "which remote has this URL?" was answered with one `git remote get-url` +// subprocess per remote, awaited serially -- 58 spawns on a repo with 58 remotes, +// on every push-target resolution. `git remote -v` answers for every remote from +// one child. +// +// `remote -v` is the faithful one-command form, not `config --get-regexp '^remote\.'`: +// both `remote -v` and `remote get-url` print the URL *after* `url..insteadOf` +// expansion and pick the first of several `remote..url` values, while raw config +// reads return the unexpanded value and the last of the multiple values. +// +// `remote -v` also predates `remote get-url` (2.7), so this lowers rather than raises +// the Git floor and needs no capability gate. + +import { iterateProcessOutputLines } from './process-output-field-scanner' + +export type GitRemoteVerboseEntry = { + name: string + url: string + direction: 'fetch' | 'push' +} + +// Greedy prefix so a URL containing spaces or parentheses keeps them. +const REMOTE_VERBOSE_URL_PATTERN = /^(.*) \((fetch|push)\)$/ + +/** Parse one `\t (fetch|push)` row. */ +export function parseGitRemoteVerboseLine(line: string): GitRemoteVerboseEntry | null { + const tabIndex = line.indexOf('\t') + if (tabIndex === -1) { + return null + } + const name = line.slice(0, tabIndex) + const match = REMOTE_VERBOSE_URL_PATTERN.exec(line.slice(tabIndex + 1).trim()) + return match ? { name, url: match[1], direction: match[2] as 'fetch' | 'push' } : null +} + +/** + * Fetch URL per remote in `git remote` order -- the value `git remote get-url ` + * prints. A remote configured with only a `pushurl` has no fetch row and is absent + * here; `get-url` echoed the remote's own name for it, which no caller can match. + */ +export function parseGitRemoteFetchUrls(stdout: string): Map { + const fetchUrls = new Map() + for (const line of iterateProcessOutputLines(stdout)) { + const parsed = parseGitRemoteVerboseLine(line) + // First wins: `get-url` without `--all` prints the first `remote..url`. + if (parsed?.direction === 'fetch' && !fetchUrls.has(parsed.name)) { + fetchUrls.set(parsed.name, parsed.url) + } + } + return fetchUrls +} + +/** First remote whose fetch URL matches, in the order the per-remote scan visited them. */ +export function findGitRemoteNameByFetchUrl( + stdout: string, + matchesUrl: (url: string) => boolean +): string | null { + for (const [name, url] of parseGitRemoteFetchUrls(stdout)) { + if (matchesUrl(url)) { + return name + } + } + return null +} From 5510ad0cad7bcb73b5f76aa8b0f3ffe5ff2d41de Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:56:23 -0700 Subject: [PATCH 073/398] fix(diff-comments): stop review refreshes from blanking inline comment cards (#18142) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit commentableLineSet was memoized on array identity. Review surfaces hand the decorator a fresh-but-equal number[] on every PR/MR data refresh, so the set churned, tore down the overlay+zone effect (unmounting every comment card's React root and clearing the zone map) while the zone-creating effect — which does not depend on the set — never re-ran. Monaco kept the view zones as untracked blank gaps, and the next refresh stacked more on top. - memoize the set on a joined value key so equal refreshes are a no-op - split the add-button overlay (needs the set) from the zone teardown (must not), so the teardown's deps stay a subset of the zone-creating effect's - have the teardown actually removeZone what it stops tracking --- ...ommentDecorator.commentable-lines.test.tsx | 284 ++++++++++++++++++ .../diff-comments/useDiffCommentDecorator.tsx | 33 +- 2 files changed, 312 insertions(+), 5 deletions(-) create mode 100644 src/renderer/src/components/diff-comments/useDiffCommentDecorator.commentable-lines.test.tsx diff --git a/src/renderer/src/components/diff-comments/useDiffCommentDecorator.commentable-lines.test.tsx b/src/renderer/src/components/diff-comments/useDiffCommentDecorator.commentable-lines.test.tsx new file mode 100644 index 00000000000..d954bafa62b --- /dev/null +++ b/src/renderer/src/components/diff-comments/useDiffCommentDecorator.commentable-lines.test.tsx @@ -0,0 +1,284 @@ +// @vitest-environment happy-dom +import { renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { editor as MonacoEditor, IDisposable } from 'monaco-editor' +import type { DecoratedDiffComment } from './decorated-diff-comment' +import type * as ReactDomClientModule from 'react-dom/client' +import type * as DiffCommentZoneCardModule from './diff-comment-zone-card' + +const storeFixture = vi.hoisted(() => ({ + activeGroupIdByWorktree: {}, + clearDeliveredDiffComments: vi.fn() +})) + +vi.mock('@/store', () => ({ + useAppStore: (selector: (state: typeof storeFixture) => unknown) => selector(storeFixture) +})) + +// Stub only the card render: this suite is about zone/root lifecycle, not card markup. +vi.mock('./diff-comment-zone-card', async (importOriginal) => ({ + ...(await importOriginal()), + renderDiffCommentZoneCard: vi.fn() +})) + +const rootCounts = vi.hoisted(() => ({ created: 0, unmounted: 0 })) + +// Count only roots the decorator creates for its zones — @testing-library/react creates its own. +vi.mock('react-dom/client', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + createRoot: (container: Element, options?: Parameters[1]) => { + const isZoneRoot = container.classList?.contains('orca-diff-comment-inline') ?? false + if (isZoneRoot) { + rootCounts.created += 1 + } + const root = actual.createRoot(container, options) + return { + render: (node: Parameters[0]) => root.render(node), + unmount: () => { + if (isZoneRoot) { + rootCounts.unmounted += 1 + } + root.unmount() + } + } + } + } +}) + +import { useDiffCommentDecorator } from './useDiffCommentDecorator' + +type FakeEditor = { + editor: MonacoEditor.ICodeEditor + domNode: HTMLElement + zones: Map + emitMouseMove: (lineNumber: number) => void +} + +function createFakeEditor(): FakeEditor { + const domNode = document.createElement('div') + document.body.appendChild(domNode) + const zones = new Map() + let nextZoneId = 0 + const mouseMoveListeners: ((e: { target: { position: { lineNumber: number } } }) => void)[] = [] + const noopDisposable: IDisposable = { dispose: () => {} } + + const editor = { + getDomNode: () => domNode, + getModel: () => ({}), + getOption: () => 19, + getTopForLineNumber: () => 0, + getScrollTop: () => 0, + getLayoutInfo: () => ({ height: 400 }), + setScrollTop: () => {}, + deltaDecorations: () => [], + getTargetAtClientPoint: () => null, + onMouseMove: (listener: (e: { target: { position: { lineNumber: number } } }) => void) => { + mouseMoveListeners.push(listener) + return noopDisposable + }, + onMouseLeave: () => noopDisposable, + onDidScrollChange: () => noopDisposable, + changeViewZones: (callback: (accessor: MonacoEditor.IViewZoneChangeAccessor) => void) => + callback({ + addZone: (zone: MonacoEditor.IViewZone) => { + const id = `zone-${(nextZoneId += 1)}` + zones.set(id, zone) + return id + }, + removeZone: (id: string) => { + zones.delete(id) + }, + layoutZone: () => {} + } as unknown as MonacoEditor.IViewZoneChangeAccessor) + } as unknown as MonacoEditor.ICodeEditor + + return { + editor, + domNode, + zones, + emitMouseMove: (lineNumber) => { + for (const listener of mouseMoveListeners) { + listener({ target: { position: { lineNumber } } }) + } + } + } +} + +const FILE_PATH = 'src/index.ts' +const REVIEW_SURFACE_ID = 'pr:acme/widgets:42' + +function reviewNote(index: number): DecoratedDiffComment { + return { + id: `review-note-${index}`, + worktreeId: REVIEW_SURFACE_ID, + filePath: FILE_PATH, + lineNumber: 10 + index, + body: `Please rename this (${index}).`, + createdAt: index, + side: 'modified', + author: 'octocat' + } +} + +// Every refresh of remote review data yields a fresh-but-equal array, exactly as the main process ships it. +function freshCommentableLines(): readonly number[] { + return [10, 11, 12, 13, 14, 15, 16] +} + +// Root teardown is deferred through queueMicrotask, so drain before asserting on unmount counts. +async function flushDeferredUnmounts(): Promise { + await Promise.resolve() + await Promise.resolve() +} + +// Asserted as one object so a failure reports every lifecycle number at once. +function lifecycleTotals(fake: FakeEditor): Record { + return { + createRootCalls: rootCounts.created, + rootUnmounts: rootCounts.unmounted, + monacoViewZones: fake.zones.size + } +} + +function isAddButtonVisible(domNode: HTMLElement): boolean { + const button = domNode.querySelector('.orca-diff-comment-add-btn') + return button != null && button.style.display !== 'none' +} + +type DecoratorProps = { + commentableLineNumbers: readonly number[] + comments: readonly DecoratedDiffComment[] +} + +function renderDecorator(fake: FakeEditor, initialProps: DecoratorProps) { + return renderHook( + ({ commentableLineNumbers, comments }: DecoratorProps) => + useDiffCommentDecorator({ + editor: fake.editor, + filePath: FILE_PATH, + worktreeId: REVIEW_SURFACE_ID, + comments, + commentableLineNumbers, + onAddCommentClick: vi.fn(), + onDeleteComment: vi.fn() + }), + { initialProps } + ) +} + +beforeEach(() => { + rootCounts.created = 0 + rootCounts.unmounted = 0 +}) + +afterEach(() => { + document.body.replaceChildren() + vi.clearAllMocks() +}) + +describe('useDiffCommentDecorator commentable-line churn', () => { + it('keeps every comment root and view zone alive across value-equal review refreshes', async () => { + const fake = createFakeEditor() + const comments = [reviewNote(1), reviewNote(2), reviewNote(3)] + const hook = renderDecorator(fake, { + commentableLineNumbers: freshCommentableLines(), + comments + }) + + expect(rootCounts.created).toBe(3) + expect(fake.zones.size).toBe(3) + + for (let refresh = 0; refresh < 5; refresh += 1) { + hook.rerender({ commentableLineNumbers: freshCommentableLines(), comments }) + } + await flushDeferredUnmounts() + + expect(lifecycleTotals(fake)).toEqual({ + createRootCalls: 3, + rootUnmounts: 0, + monacoViewZones: 3 + }) + + // Still tracked, so dropping a note reclaims its vertical space instead of leaving a blank gap. + hook.rerender({ commentableLineNumbers: freshCommentableLines(), comments: comments.slice(1) }) + expect(fake.zones.size).toBe(2) + }) + + it('does not accumulate orphan zones when refreshes interleave with new review comments', async () => { + const fake = createFakeEditor() + let comments = [reviewNote(1)] + const hook = renderDecorator(fake, { + commentableLineNumbers: freshCommentableLines(), + comments + }) + + for (let refresh = 2; refresh <= 6; refresh += 1) { + comments = [...comments, reviewNote(refresh)] + hook.rerender({ commentableLineNumbers: freshCommentableLines(), comments }) + } + await flushDeferredUnmounts() + + // 6 live notes => 6 roots, 6 zones, nothing stranded. + expect(lifecycleTotals(fake)).toEqual({ + createRootCalls: 6, + rootUnmounts: 0, + monacoViewZones: 6 + }) + }) + + it('rebuilds the add-button overlay when the commentable lines really change', async () => { + const fake = createFakeEditor() + const comments = [reviewNote(1)] + const hook = renderDecorator(fake, { + commentableLineNumbers: [10, 11, 12] as readonly number[], + comments + }) + + fake.emitMouseMove(40) + expect(isAddButtonVisible(fake.domNode)).toBe(false) + + hook.rerender({ commentableLineNumbers: [10, 11, 12, 40], comments }) + await flushDeferredUnmounts() + + // Decorator is not stale: the widened set is live... + fake.emitMouseMove(40) + expect(isAddButtonVisible(fake.domNode)).toBe(true) + // ...and the existing note's zone/root was neither orphaned nor rebuilt. + expect(lifecycleTotals(fake)).toEqual({ + createRootCalls: 1, + rootUnmounts: 0, + monacoViewZones: 1 + }) + }) + + it('removes its zones from Monaco when the model swaps under a retained editor', async () => { + const fake = createFakeEditor() + const comments = [reviewNote(1)] + const hook = renderHook( + ({ monacoModelIdentity }) => + useDiffCommentDecorator({ + editor: fake.editor, + monacoModelIdentity, + filePath: FILE_PATH, + worktreeId: REVIEW_SURFACE_ID, + comments, + commentableLineNumbers: freshCommentableLines(), + onAddCommentClick: vi.fn(), + onDeleteComment: vi.fn() + }), + { initialProps: { monacoModelIdentity: 'modified-v1' } } + ) + const firstZoneIds = [...fake.zones.keys()] + + hook.rerender({ monacoModelIdentity: 'modified-v2' }) + await flushDeferredUnmounts() + + // Stale zone ids are gone rather than left as untracked blank gaps, and the note was rebuilt. + expect(fake.zones.size).toBe(1) + expect([...fake.zones.keys()]).not.toEqual(firstZoneIds) + expect(rootCounts.created).toBe(2) + expect(rootCounts.unmounted).toBe(1) + }) +}) diff --git a/src/renderer/src/components/diff-comments/useDiffCommentDecorator.tsx b/src/renderer/src/components/diff-comments/useDiffCommentDecorator.tsx index 8e7c484bb09..d1f26d24765 100644 --- a/src/renderer/src/components/diff-comments/useDiffCommentDecorator.tsx +++ b/src/renderer/src/components/diff-comments/useDiffCommentDecorator.tsx @@ -80,11 +80,17 @@ export function useDiffCommentDecorator({ scrollToZoneFrameRef.current = null }, []) + // Key on the values, not the array identity: review surfaces re-fetch PR/MR file data on every + // refresh and hand us a fresh-but-equal number[], which would otherwise churn every consumer. + const commentableLineKey = commentableLineNumbers?.join(',') const commentableLineSet = useMemo( () => (commentableLineNumbers ? new Set(commentableLineNumbers) : null), - [commentableLineNumbers] + // eslint-disable-next-line react-hooks/exhaustive-deps + [commentableLineKey] ) + // Add-button overlay only: it captures commentableLineSet/addButtonLabel, so it must be rebuilt when + // either changes. Kept apart from the zone teardown below, whose deps must mirror the zone-creating effect. useEffect(() => { if (!editor) { return @@ -95,8 +101,7 @@ export function useDiffCommentDecorator({ return } - const zones = zonesRef.current - const disposeAddButtonOverlay = installDiffCommentAddButtonOverlay({ + return installDiffCommentAddButtonOverlay({ editor, editorDomNode, addButtonLabel, @@ -105,15 +110,33 @@ export function useDiffCommentDecorator({ disposablesRef, onAddCommentClickRef }) + }, [addButtonLabel, commentableLineSet, editor, monacoModelIdentity]) + + // Deps must stay a subset of the zone-creating effect's, or a teardown here is never followed by a rebuild. + useEffect(() => { + if (!editor) { + return + } + + const zones = zonesRef.current return () => { - disposeAddButtonOverlay() // Editor swapped/torn down: unmount roots and clear tracking so the next mount starts known-empty. // Defer unmount via queueMicrotask: a sync unmount during React's commit triggers React 19's "unmount while rendering" warning; clear zones synchronously. const rootsToUnmount = Array.from(zones.values(), (z) => { z.disposeMouseDownStopper() return z.root }) + // Drop the zones from Monaco too: clearing our map alone would strand them as untracked blank gaps + // in a still-live editor. No-op when the model already swapped (Monaco dropped them) or the editor is disposed. + if (zones.size > 0) { + const zoneIds = Array.from(zones.values(), (z) => z.zoneId) + editor.changeViewZones((accessor) => { + for (const zoneId of zoneIds) { + accessor.removeZone(zoneId) + } + }) + } zones.clear() if (rootsToUnmount.length > 0) { queueMicrotask(() => { @@ -127,7 +150,7 @@ export function useDiffCommentDecorator({ pendingScrollRef.current = null scrollToZoneRef.current = null } - }, [addButtonLabel, cancelScrollToZoneFrame, commentableLineSet, editor, monacoModelIdentity]) + }, [cancelScrollToZoneFrame, editor, monacoModelIdentity]) useEffect(() => { if (!editor) { From 3d67f2be2c5d5fb308c426964470133729241b20 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:56:26 -0700 Subject: [PATCH 074/398] perf(editor,source-control): batch closed-tab model sweeps, drop 40 store subscriptions (#18144) Closing N diff tabs scanned the global Monaco model registry 2N times and rendered both URI forms for every retained model on each scan. The Source Control panel opened 42 store subscriptions from one hook, 40 of which watched action identities that are fixed at store construction and can never change. --- .../closed-editor-tab-cache-sweep.test.ts | 10 +- .../editor/closed-editor-tab-cache-sweep.ts | 45 +++- .../editor/closed-editor-tab-disposal.test.ts | 235 ++++++++++++++++++ .../editor/closed-editor-tab-disposal.ts | 86 +++++++ .../editor/diff-monaco-model-disposal.ts | 69 ++++- .../editor/useClosedEditorTabCleanup.ts | 70 +----- ...store-actions.store-subscriptions.test.tsx | 166 +++++++++++++ .../listing/use-store-actions.ts | 153 +++++------- 8 files changed, 654 insertions(+), 180 deletions(-) create mode 100644 src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts create mode 100644 src/renderer/src/components/editor/closed-editor-tab-disposal.ts create mode 100644 src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.store-subscriptions.test.tsx diff --git a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts index 1b975d81339..726795a31dd 100644 --- a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts +++ b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts @@ -7,7 +7,7 @@ const position = (pageNumber: number): PdfViewPosition => ({ pageNumber, top: 0, describe('sweepClosedPdfViewPositions', () => { it('deletes the unscoped :pdf entry', () => { const cache = new Map([['/a.pdf:pdf', position(4)]]) - sweepClosedPdfViewPositions(cache, '/a.pdf') + sweepClosedPdfViewPositions(cache, ['/a.pdf']) expect(cache.size).toBe(0) }) @@ -17,7 +17,7 @@ describe('sweepClosedPdfViewPositions', () => { ['/a.pdf::tab-2:pdf', position(9)], ['/a.pdf::tab-3:pdf', position(11)] ]) - sweepClosedPdfViewPositions(cache, '/a.pdf') + sweepClosedPdfViewPositions(cache, ['/a.pdf']) expect(cache.size).toBe(0) }) @@ -27,7 +27,7 @@ describe('sweepClosedPdfViewPositions', () => { ['/b.pdf:pdf', position(7)], ['/b.pdf::tab-2:pdf', position(8)] ]) - sweepClosedPdfViewPositions(cache, '/a.pdf') + sweepClosedPdfViewPositions(cache, ['/a.pdf']) expect([...cache.keys()]).toEqual(['/b.pdf:pdf', '/b.pdf::tab-2:pdf']) }) @@ -36,13 +36,13 @@ describe('sweepClosedPdfViewPositions', () => { ['/report.pdf:pdf', position(2)], ['/report.pdf.bak:pdf', position(3)] ]) - sweepClosedPdfViewPositions(cache, '/report.pdf') + sweepClosedPdfViewPositions(cache, ['/report.pdf']) expect([...cache.keys()]).toEqual(['/report.pdf.bak:pdf']) }) it('is a no-op when the file has no cached position', () => { const cache = new Map([['/b.pdf:pdf', position(7)]]) - sweepClosedPdfViewPositions(cache, '/a.pdf') + sweepClosedPdfViewPositions(cache, ['/a.pdf']) expect(cache.size).toBe(1) }) }) diff --git a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts index d53f0212723..b9769fb9dc8 100644 --- a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts +++ b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts @@ -1,23 +1,50 @@ import type { PdfViewPosition } from '@/lib/scroll-cache' -function deleteCacheEntriesByPrefix(cache: Map, prefix: string): void { +/** + * Drops every pane-scoped (`::…`) entry belonging to any of `owners` in one pass over + * the cache, rather than one pass per owner. Split out from the cleanup hook so it is testable + * without pulling in the hook's `monaco-editor` import. + */ +export function deletePaneScopedCacheEntries( + cache: Map, + owners: readonly string[] +): void { + if (owners.length === 0) { + return + } + + const ownerSet = new Set(owners) for (const key of cache.keys()) { - if (key.startsWith(prefix)) { + if (hasPaneScopeOwner(key, ownerSet)) { cache.delete(key) } } } -/** - * Release the PDF positions a closed edit tab owns. Split out from the cleanup - * hook so it is testable without pulling in the hook's `monaco-editor` import. - */ +/** Equivalent to `key.startsWith(`${owner}::`)` for any owner in the set, probing `::` boundaries. */ +function hasPaneScopeOwner(key: string, owners: ReadonlySet): boolean { + for ( + let boundary = key.indexOf('::'); + boundary !== -1; + boundary = key.indexOf('::', boundary + 1) + ) { + if (owners.has(key.slice(0, boundary))) { + return true + } + } + + return false +} + +/** Release the PDF positions closed edit tabs own. */ export function sweepClosedPdfViewPositions( cache: Map, - filePath: string + filePaths: readonly string[] ): void { // Why: the `::`-scoped sweep does not cover the single-colon suffix, so the // unscoped key needs its own delete (same shape as :rich / :preview). - cache.delete(`${filePath}:pdf`) - deleteCacheEntriesByPrefix(cache, `${filePath}::`) + for (const filePath of filePaths) { + cache.delete(`${filePath}:pdf`) + } + deletePaneScopedCacheEntries(cache, filePaths) } diff --git a/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts b/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts new file mode 100644 index 00000000000..0223a4394b0 --- /dev/null +++ b/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts @@ -0,0 +1,235 @@ +import { beforeEach, describe, expect, it } from 'vitest' +import { + diffViewStateCache, + editorSelectionCache, + pdfViewPositionCache, + scrollTopCache +} from '@/lib/scroll-cache' +import type { OpenFile } from '@/store/slices/editor' +import { disposeClosedEditorTabs } from './closed-editor-tab-disposal' +import { + getDiffViewerMonacoModelPaths, + getDiffViewerMonacoModelPathPrefixes, + type MonacoModelRegistry +} from './diff-monaco-model-disposal' + +const CLOSED_DIFF_TAB_COUNT = 100 +const RETAINED_MODEL_COUNT = 320 + +type FakeModel = { + path: string + attached: boolean + disposed: boolean + dispose: () => void + isAttachedToEditor: () => boolean + uri: { toString: (skipEncoding?: boolean) => string } +} + +type FakeRegistry = MonacoModelRegistry & { + models: FakeModel[] + counters: { getModelsCalls: number; uriToStringCalls: number } +} + +function createRegistry(models: FakeModel[]): FakeRegistry { + const counters = { getModelsCalls: 0, uriToStringCalls: 0 } + const byPath = new Map(models.map((model) => [model.path, model])) + for (const model of models) { + model.uri.toString = () => { + counters.uriToStringCalls += 1 + return model.path + } + } + return { + models, + counters, + Uri: { parse: (value: string) => value }, + editor: { + getModel: (uri: unknown) => byPath.get(String(uri)) ?? null, + getModels: () => { + counters.getModelsCalls += 1 + return models + } + } + } +} + +function createModel(path: string, attached = false): FakeModel { + const model: FakeModel = { + path, + attached, + disposed: false, + dispose: () => { + model.disposed = true + }, + isAttachedToEditor: () => model.attached, + uri: { toString: () => path } + } + return model +} + +function diffTab(id: string): OpenFile { + return { id, mode: 'diff', filePath: `/repo/${id}.ts` } as OpenFile +} + +/** The pre-fix shape: one full registry scan, with both URI renderings, per owned prefix. */ +function disposeByPrefixPerTab(registry: FakeRegistry, prefixes: readonly string[]): void { + for (const prefix of prefixes) { + for (const model of registry.editor.getModels()) { + const uriString = model.uri.toString(true) + const encodedUriString = model.uri.toString() + if ( + uriString === prefix || + uriString.startsWith(`${prefix}:`) || + encodedUriString === prefix || + encodedUriString.startsWith(`${prefix}:`) + ) { + if (!model.isAttachedToEditor()) { + model.dispose() + } + } + } + } +} + +/** + * 100 closed diff tabs, of which 60 still hold retained models (some with a large-diff generation + * suffix, some attached), plus 200 unrelated retained models from other tabs. + */ +function buildScenario(): { + closedTabs: OpenFile[] + models: FakeModel[] + prefixes: string[] +} { + const closedTabs = Array.from({ length: CLOSED_DIFF_TAB_COUNT }, (_, i) => diffTab(`tab-${i}`)) + const models: FakeModel[] = [] + + for (let i = 0; i < 60; i += 1) { + const base = getDiffViewerMonacoModelPaths({ + modelKey: `tab-${i}`, + generationSuffix: '' + }) + models.push(createModel(base.originalModelPath, i % 10 === 0)) + models.push(createModel(base.modifiedModelPath)) + if (i % 3 === 0) { + const regenerated = getDiffViewerMonacoModelPaths({ + modelKey: `tab-${i}`, + generationSuffix: ':large-diff-generation:2' + }) + models.push(createModel(regenerated.originalModelPath)) + } + } + + // Still-open tabs and plain edit models the sweep must not touch. + for (let i = 0; models.length < RETAINED_MODEL_COUNT; i += 1) { + const stillOpen = getDiffViewerMonacoModelPaths({ + modelKey: `open-tab-${i}`, + generationSuffix: '' + }) + models.push(createModel(stillOpen.originalModelPath)) + models.push(createModel(`/repo/src/file-${i}.ts`)) + } + + const prefixes = closedTabs.flatMap((tab) => { + const { originalModelPathPrefix, modifiedModelPathPrefix } = + getDiffViewerMonacoModelPathPrefixes(tab.id) + return [originalModelPathPrefix, modifiedModelPathPrefix] + }) + + return { closedTabs, models, prefixes } +} + +beforeEach(() => { + scrollTopCache.clear() + editorSelectionCache.clear() + diffViewStateCache.clear() + pdfViewPositionCache.clear() +}) + +describe('disposeClosedEditorTabs', () => { + it('scans the model registry once per batch instead of twice per closed diff tab', () => { + const batched = buildScenario() + const batchedRegistry = createRegistry(batched.models) + disposeClosedEditorTabs(batchedRegistry, batched.closedTabs) + + const perTab = buildScenario() + const perTabRegistry = createRegistry(perTab.models) + disposeByPrefixPerTab(perTabRegistry, perTab.prefixes) + + // Pre-fix: 2 scans per closed tab, each rendering both URI forms for every retained model. + expect(perTabRegistry.counters.getModelsCalls).toBe(CLOSED_DIFF_TAB_COUNT * 2) + expect(perTabRegistry.counters.uriToStringCalls).toBe( + CLOSED_DIFF_TAB_COUNT * 2 * perTab.models.length * 2 + ) + + expect(batchedRegistry.counters.getModelsCalls).toBe(1) + expect(batchedRegistry.counters.uriToStringCalls).toBeLessThanOrEqual(batched.models.length * 2) + }) + + it('disposes exactly the models the per-tab sweep disposed', () => { + const batched = buildScenario() + disposeClosedEditorTabs(createRegistry(batched.models), batched.closedTabs) + + const perTab = buildScenario() + disposeByPrefixPerTab(createRegistry(perTab.models), perTab.prefixes) + + const disposedPaths = (models: FakeModel[]): string[] => + models + .filter((m) => m.disposed) + .map((m) => m.path) + .sort() + + expect(disposedPaths(batched.models)).toEqual(disposedPaths(perTab.models)) + expect(disposedPaths(batched.models).length).toBeGreaterThan(0) + // Attached models survive, as does everything owned by a still-open tab. + expect(batched.models.filter((m) => m.attached).every((m) => !m.disposed)).toBe(true) + expect( + batched.models.filter((m) => m.path.includes('open-tab-')).every((m) => !m.disposed) + ).toBe(true) + }) + + it('sweeps pane-scoped cache entries for closed edit tabs in one pass per cache', () => { + scrollTopCache.set('/repo/a.ts', 10) + scrollTopCache.set('/repo/a.ts::pane-1', 20) + scrollTopCache.set('/repo/a.ts:rich', 30) + scrollTopCache.set('/repo/b.ts::pane-1', 40) + editorSelectionCache.set('/repo/a.ts::pane-2', [] as never) + pdfViewPositionCache.set('/repo/a.ts:pdf', { + pageNumber: 1, + top: 0, + left: 0 + }) + pdfViewPositionCache.set('/repo/a.ts::pane-1:pdf', { + pageNumber: 2, + top: 0, + left: 0 + }) + + disposeClosedEditorTabs(createRegistry([]), [ + { id: '/repo/a.ts', mode: 'edit', filePath: '/repo/a.ts' } as OpenFile + ]) + + expect([...scrollTopCache.keys()]).toEqual(['/repo/b.ts::pane-1']) + expect(editorSelectionCache.size).toBe(0) + expect(pdfViewPositionCache.size).toBe(0) + }) + + it('drops diff view state and preview scroll entries for closed diff tabs', () => { + diffViewStateCache.set('tab-1', {} as never) + diffViewStateCache.set('tab-1::pane-1', {} as never) + diffViewStateCache.set('tab-10', {} as never) + scrollTopCache.set('tab-1:preview', 5) + scrollTopCache.set('tab-1::pane-1', 6) + + disposeClosedEditorTabs(createRegistry([]), [diffTab('tab-1')]) + + expect([...diffViewStateCache.keys()]).toEqual(['tab-10']) + expect(scrollTopCache.size).toBe(0) + }) + + it('is a no-op when nothing closed', () => { + const registry = createRegistry([createModel('diff:original:tab-1:tab-1')]) + disposeClosedEditorTabs(registry, []) + expect(registry.counters.getModelsCalls).toBe(0) + expect(registry.models[0].disposed).toBe(false) + }) +}) diff --git a/src/renderer/src/components/editor/closed-editor-tab-disposal.ts b/src/renderer/src/components/editor/closed-editor-tab-disposal.ts new file mode 100644 index 00000000000..ddde3a74502 --- /dev/null +++ b/src/renderer/src/components/editor/closed-editor-tab-disposal.ts @@ -0,0 +1,86 @@ +import type { OpenFile } from '@/store/slices/editor' +import { + editorSelectionCache, + diffViewStateCache, + pdfViewPositionCache, + scrollTopCache +} from '@/lib/scroll-cache' +import { + disposeUnattachedMonacoModelsByPathPrefixes, + getDiffViewerMonacoModelPathPrefixes, + type MonacoModelRegistry +} from './diff-monaco-model-disposal' +import { + deletePaneScopedCacheEntries, + sweepClosedPdfViewPositions +} from './closed-editor-tab-cache-sweep' + +/** + * Releases the Monaco models and view-state cache entries owned by a batch of closed tabs. + * + * Why the batch shape: every prefix sweep here is a full scan of a shared registry or cache, so + * doing one per closed tab makes "close all"/worktree-switch quadratic in retained models. Takes + * the monaco namespace as an argument so it stays testable without importing `monaco-editor`. + */ +export function disposeClosedEditorTabs( + monacoRegistry: MonacoModelRegistry, + closedFiles: readonly OpenFile[] +): void { + if (closedFiles.length === 0) { + return + } + + const diffModelPathPrefixes: string[] = [] + const scrollTopOwners: string[] = [] + const editorSelectionOwners: string[] = [] + const diffViewStateOwners: string[] = [] + const closedPdfFilePaths: string[] = [] + + for (const closedFile of closedFiles) { + switch (closedFile.mode) { + case 'edit': + // Why: the edit model URI is constructed via monaco.Uri.parse(filePath) + // to match @monaco-editor/react's `path` prop convention. + monacoRegistry.editor.getModel(monacoRegistry.Uri.parse(closedFile.filePath))?.dispose() + scrollTopCache.delete(closedFile.filePath) + // Why: markdown and mermaid surfaces keep mode-scoped scroll positions. + scrollTopCache.delete(`${closedFile.filePath}:rich`) + scrollTopCache.delete(`${closedFile.filePath}:preview`) + scrollTopCache.delete(`${closedFile.filePath}:mermaid-diagram`) + editorSelectionCache.delete(closedFile.filePath) + scrollTopOwners.push(closedFile.filePath) + editorSelectionOwners.push(closedFile.filePath) + // Why: only 'edit' tabs ever get a PDF scroll key (see EditorContent). + closedPdfFilePaths.push(closedFile.filePath) + break + case 'markdown-preview': + // Why: preview tabs own pane-scoped preview scroll cache entries even + // though they do not retain Monaco models. + scrollTopCache.delete(`${closedFile.id}:preview`) + scrollTopOwners.push(closedFile.id) + break + case 'diff': { + // Why: kept diff models are keyed by tab id, and fallback recovery can + // append generation suffixes; closing the tab owns that whole namespace. + const { originalModelPathPrefix, modifiedModelPathPrefix } = + getDiffViewerMonacoModelPathPrefixes(closedFile.id) + diffModelPathPrefixes.push(originalModelPathPrefix, modifiedModelPathPrefix) + diffViewStateCache.delete(closedFile.id) + diffViewStateOwners.push(closedFile.id) + scrollTopCache.delete(`${closedFile.id}:preview`) + scrollTopOwners.push(closedFile.id) + break + } + case 'conflict-review': + break + case 'check-details': + break + } + } + + disposeUnattachedMonacoModelsByPathPrefixes(monacoRegistry, diffModelPathPrefixes) + deletePaneScopedCacheEntries(scrollTopCache, scrollTopOwners) + deletePaneScopedCacheEntries(editorSelectionCache, editorSelectionOwners) + deletePaneScopedCacheEntries(diffViewStateCache, diffViewStateOwners) + sweepClosedPdfViewPositions(pdfViewPositionCache, closedPdfFilePaths) +} diff --git a/src/renderer/src/components/editor/diff-monaco-model-disposal.ts b/src/renderer/src/components/editor/diff-monaco-model-disposal.ts index 8252abc8381..f833e53ec5b 100644 --- a/src/renderer/src/components/editor/diff-monaco-model-disposal.ts +++ b/src/renderer/src/components/editor/diff-monaco-model-disposal.ts @@ -16,7 +16,7 @@ type DisposableMonacoModel = Pick, + bounds: { shortestPrefixLength: number; longestPrefixLength: number } +): boolean { + if (ownedPrefixes.has(uriString)) { + return true + } + + for ( + let boundary = uriString.indexOf(':'); + boundary !== -1 && boundary <= bounds.longestPrefixLength; + boundary = uriString.indexOf(':', boundary + 1) + ) { + if ( + boundary >= bounds.shortestPrefixLength && + ownedPrefixes.has(uriString.slice(0, boundary)) + ) { + return true + } + } + + return false +} + function disposeUnattachedMonacoModel(model: DisposableMonacoModel | null): void { if (!model || model.isAttachedToEditor()) { return diff --git a/src/renderer/src/components/editor/useClosedEditorTabCleanup.ts b/src/renderer/src/components/editor/useClosedEditorTabCleanup.ts index 56e25045353..e86fbbc43f6 100644 --- a/src/renderer/src/components/editor/useClosedEditorTabCleanup.ts +++ b/src/renderer/src/components/editor/useClosedEditorTabCleanup.ts @@ -1,80 +1,22 @@ import { useEffect, useRef } from 'react' import * as monaco from 'monaco-editor' import type { OpenFile } from '@/store/slices/editor' -import { - editorSelectionCache, - diffViewStateCache, - pdfViewPositionCache, - scrollTopCache -} from '@/lib/scroll-cache' -import { - disposeUnattachedMonacoModelsByPathPrefix, - getDiffViewerMonacoModelPathPrefixes -} from './diff-monaco-model-disposal' -import { sweepClosedPdfViewPositions } from './closed-editor-tab-cache-sweep' - -function deleteCacheEntriesByPrefix(cache: Map, prefix: string): void { - for (const key of cache.keys()) { - if (key.startsWith(prefix)) { - cache.delete(key) - } - } -} +import { disposeClosedEditorTabs } from './closed-editor-tab-disposal' export function useClosedEditorTabCleanup(openFiles: OpenFile[]): void { const prevOpenFilesRef = useRef>(new Map()) useEffect(() => { const currentFilesById = new Map(openFiles.map((f) => [f.id, f])) + const closedFiles: OpenFile[] = [] for (const [prevId, prevFile] of prevOpenFilesRef.current) { if (!currentFilesById.has(prevId)) { - disposeClosedEditorTab(prevId, prevFile) + closedFiles.push(prevFile) } } + // Why one call for the whole removal batch: each sweep scans a shared registry/cache, so + // per-tab sweeps make a "close all" quadratic in retained models. + disposeClosedEditorTabs(monaco, closedFiles) prevOpenFilesRef.current = currentFilesById }, [openFiles]) } - -function disposeClosedEditorTab(prevId: string, prevFile: OpenFile): void { - switch (prevFile.mode) { - case 'edit': - // Why: the edit model URI is constructed via monaco.Uri.parse(filePath) - // to match @monaco-editor/react's `path` prop convention. - monaco.editor.getModel(monaco.Uri.parse(prevFile.filePath))?.dispose() - scrollTopCache.delete(prevFile.filePath) - deleteCacheEntriesByPrefix(scrollTopCache, `${prevFile.filePath}::`) - // Why: markdown and mermaid surfaces keep mode-scoped scroll positions. - scrollTopCache.delete(`${prevFile.filePath}:rich`) - scrollTopCache.delete(`${prevFile.filePath}:preview`) - scrollTopCache.delete(`${prevFile.filePath}:mermaid-diagram`) - editorSelectionCache.delete(prevFile.filePath) - deleteCacheEntriesByPrefix(editorSelectionCache, `${prevFile.filePath}::`) - // Why: only 'edit' tabs ever get a PDF scroll key (see EditorContent). - sweepClosedPdfViewPositions(pdfViewPositionCache, prevFile.filePath) - break - case 'markdown-preview': - // Why: preview tabs own pane-scoped preview scroll cache entries even - // though they do not retain Monaco models. - scrollTopCache.delete(`${prevFile.id}:preview`) - deleteCacheEntriesByPrefix(scrollTopCache, `${prevFile.id}::`) - break - case 'diff': - // Why: kept diff models are keyed by tab id, and fallback recovery can - // append generation suffixes; closing the tab owns that whole namespace. - { - const { originalModelPathPrefix, modifiedModelPathPrefix } = - getDiffViewerMonacoModelPathPrefixes(prevId) - disposeUnattachedMonacoModelsByPathPrefix(monaco, originalModelPathPrefix) - disposeUnattachedMonacoModelsByPathPrefix(monaco, modifiedModelPathPrefix) - } - diffViewStateCache.delete(prevId) - deleteCacheEntriesByPrefix(diffViewStateCache, `${prevId}::`) - scrollTopCache.delete(`${prevId}:preview`) - deleteCacheEntriesByPrefix(scrollTopCache, `${prevId}::`) - break - case 'conflict-review': - break - case 'check-details': - break - } -} diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.store-subscriptions.test.tsx b/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.store-subscriptions.test.tsx new file mode 100644 index 00000000000..dca92f903f1 --- /dev/null +++ b/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.store-subscriptions.test.tsx @@ -0,0 +1,166 @@ +// @vitest-environment happy-dom + +import { act, useState, type ReactNode } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, describe, expect, it } from 'vitest' +import { useAppStore } from '@/store' +import { readStoreListenerCount } from '@/store/store-listener-census' +import { useSourceControlStoreActions, type SourceControlStoreActions } from './use-store-actions' + +const originalState = useAppStore.getState() + +let root: Root | null = null +let container: HTMLDivElement | null = null + +function mount(node: ReactNode): void { + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + act(() => root?.render(node)) +} + +function unmount(): void { + if (root) { + act(() => root?.unmount()) + } + root = null + container?.remove() + container = null +} + +function listenerCount(): number { + const count = readStoreListenerCount() + if (count === null) { + throw new Error('store listener census unavailable') + } + return count +} + +afterEach(() => { + unmount() + useAppStore.setState(originalState, true) +}) + +/** Everything the hook returns that is a store action rather than subscribed state. */ +const ACTION_KEYS = Object.keys(originalState).filter( + (key) => typeof (originalState as Record)[key] === 'function' +) + +describe('useSourceControlStoreActions store subscriptions', () => { + it('keeps only the two generation-record maps subscribed', () => { + const baseline = listenerCount() + + function Probe(): null { + useSourceControlStoreActions() + return null + } + mount() + + // Why 2: `pullRequestGenerationRecords` and `commitMessageGenerationRecords` are the only + // entries that are state; the other 40 are actions read through getState(). + expect(listenerCount() - baseline).toBe(2) + + unmount() + expect(listenerCount()).toBe(baseline) + }) + + it('returns the same object across an unrelated store write and re-render', () => { + let latest: SourceControlStoreActions | null = null + let rerender: (() => void) | null = null + + function Probe(): null { + const [, setTick] = useState(0) + rerender = () => setTick((t) => t + 1) + latest = useSourceControlStoreActions() + return null + } + mount() + + const first = latest + expect(first).not.toBeNull() + + act(() => { + useAppStore.setState({ + rightSidebarOpen: !originalState.rightSidebarOpen + }) + }) + act(() => rerender?.()) + + expect(latest).toBe(first) + }) + + it('still tracks the generation-record maps it subscribes to', () => { + let latest: SourceControlStoreActions | null = null + function Probe(): null { + latest = useSourceControlStoreActions() + return null + } + function read(): SourceControlStoreActions { + if (!latest) { + throw new Error('probe did not render') + } + return latest + } + mount() + + const before = read() + const record = { status: 'pending' } as never + act(() => { + useAppStore.setState({ + pullRequestGenerationRecords: { 'wt-1': record } + }) + }) + + expect(read()).not.toBe(before) + expect(read().prGenerationRecords).toEqual({ 'wt-1': record }) + + const afterPr = read() + act(() => { + useAppStore.setState({ + commitMessageGenerationRecords: { 'wt-1': record } + }) + }) + expect(read()).not.toBe(afterPr) + expect(read().commitMessageGenerationRecords).toEqual({ 'wt-1': record }) + }) + + it('hands back the live store action references', () => { + let latest: SourceControlStoreActions | null = null + function Probe(): null { + latest = useSourceControlStoreActions() + return null + } + mount() + + const state = useAppStore.getState() as unknown as Record + const returned = latest as unknown as Record + const returnedActionKeys = Object.keys(returned).filter( + (key) => typeof returned[key] === 'function' + ) + + expect(returnedActionKeys.length).toBe(40) + for (const key of returnedActionKeys) { + expect(returned[key]).toBe(state[key]) + } + }) + + it('never reassigns a store action, which is what makes getState() safe here', () => { + const before = useAppStore.getState() as unknown as Record + const snapshot = new Map(ACTION_KEYS.map((key) => [key, before[key]])) + + // Drive real writes through several slices, then confirm no action identity moved. + act(() => { + useAppStore.getState().setRightSidebarOpen(true) + useAppStore.getState().setRightSidebarTab('source-control') + useAppStore.getState().allocatePullRequestGenerationRequestId() + useAppStore.getState().setPullRequestGenerationRecord('wt-1', { status: 'pending' } as never) + useAppStore.getState().setCommitMessageGenerationRecord('wt-1', { + status: 'pending' + } as never) + }) + + const after = useAppStore.getState() as unknown as Record + const moved = ACTION_KEYS.filter((key) => after[key] !== snapshot.get(key)) + expect(moved).toEqual([]) + }) +}) diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.ts b/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.ts index 230cc4aebeb..d85c7cd04d8 100644 --- a/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.ts +++ b/src/renderer/src/components/right-sidebar/source-control/listing/use-store-actions.ts @@ -1,105 +1,70 @@ +import { useMemo } from 'react' import { useAppStore } from '@/store' /** - * Binds every store action the Source Control panel dispatches. Each entry keeps its own selector so - * the returned references stay stable and can be used directly in downstream dependency arrays. + * Binds every store action the Source Control panel dispatches. + * + * Why `getState()` and not one selector each: zustand action identities are fixed when the store is + * built and no slice ever puts one in a `set()` payload, so subscribing to them can never fire. The + * 40 action subscriptions only added 40 live listeners and 40 selector runs to every store write + * while the panel was mounted. The two generation-record maps are real state, so they stay + * subscribed. + * + * Reference stability is preserved and slightly stronger than before: each action keeps the single + * identity it was created with, and the returned object itself is now stable until one of the two + * subscribed maps changes, so downstream dependency arrays keep working. */ export function useSourceControlStoreActions() { - const updateSettings = useAppStore((s) => s.updateSettings) - const openSettingsTarget = useAppStore((s) => s.openSettingsTarget) - const openSettingsPage = useAppStore((s) => s.openSettingsPage) - const fetchHostedReviewForBranch = useAppStore((s) => s.fetchHostedReviewForBranch) - const getHostedReviewCreationEligibility = useAppStore( - (s) => s.getHostedReviewCreationEligibility - ) - const createHostedReview = useAppStore((s) => s.createHostedReview) - const createStackedHostedReview = useAppStore((s) => s.createStackedHostedReview) - const updateWorktreeMeta = useAppStore((s) => s.updateWorktreeMeta) - const openModal = useAppStore((s) => s.openModal) - const fetchPRForBranch = useAppStore((s) => s.fetchPRForBranch) - const enqueueGitHubPRRefresh = useAppStore((s) => s.enqueueGitHubPRRefresh) - const updateRepo = useAppStore((s) => s.updateRepo) - const setGitStatus = useAppStore((s) => s.setGitStatus) - const updateWorktreeGitIdentity = useAppStore((s) => s.updateWorktreeGitIdentity) - const beginGitBranchCompareRequest = useAppStore((s) => s.beginGitBranchCompareRequest) - const setGitBranchCompareResult = useAppStore((s) => s.setGitBranchCompareResult) - const fetchUpstreamStatus = useAppStore((s) => s.fetchUpstreamStatus) - const ensureHostedReviewPushTarget = useAppStore((s) => s.ensureHostedReviewPushTarget) - const setUpstreamStatus = useAppStore((s) => s.setUpstreamStatus) - const pushBranch = useAppStore((s) => s.pushBranch) - const pullBranch = useAppStore((s) => s.pullBranch) - const fastForwardBranch = useAppStore((s) => s.fastForwardBranch) - const syncBranch = useAppStore((s) => s.syncBranch) - const rebaseFromBase = useAppStore((s) => s.rebaseFromBase) - const fetchBranch = useAppStore((s) => s.fetchBranch) - const revealInExplorer = useAppStore((s) => s.revealInExplorer) - const openConflictReview = useAppStore((s) => s.openConflictReview) - const openAllDiffs = useAppStore((s) => s.openAllDiffs) - const openBranchAllDiffs = useAppStore((s) => s.openBranchAllDiffs) - const deleteDiffComment = useAppStore((s) => s.deleteDiffComment) - const clearDiffComments = useAppStore((s) => s.clearDiffComments) - const clearDiffCommentsForFile = useAppStore((s) => s.clearDiffCommentsForFile) - const setRightSidebarOpen = useAppStore((s) => s.setRightSidebarOpen) - const setRightSidebarTab = useAppStore((s) => s.setRightSidebarTab) const prGenerationRecords = useAppStore((s) => s.pullRequestGenerationRecords) - const allocatePullRequestGenerationRequestId = useAppStore( - (s) => s.allocatePullRequestGenerationRequestId - ) - const setPullRequestGenerationRecord = useAppStore((s) => s.setPullRequestGenerationRecord) - const updatePullRequestGenerationRecord = useAppStore((s) => s.updatePullRequestGenerationRecord) const commitMessageGenerationRecords = useAppStore((s) => s.commitMessageGenerationRecords) - const allocateCommitMessageGenerationRequestId = useAppStore( - (s) => s.allocateCommitMessageGenerationRequestId - ) - const setCommitMessageGenerationRecord = useAppStore((s) => s.setCommitMessageGenerationRecord) - const updateCommitMessageGenerationRecord = useAppStore( - (s) => s.updateCommitMessageGenerationRecord - ) - return { - allocateCommitMessageGenerationRequestId, - allocatePullRequestGenerationRequestId, - beginGitBranchCompareRequest, - clearDiffComments, - clearDiffCommentsForFile, - commitMessageGenerationRecords, - createHostedReview, - createStackedHostedReview, - deleteDiffComment, - enqueueGitHubPRRefresh, - ensureHostedReviewPushTarget, - fastForwardBranch, - fetchBranch, - fetchHostedReviewForBranch, - fetchPRForBranch, - fetchUpstreamStatus, - getHostedReviewCreationEligibility, - openAllDiffs, - openBranchAllDiffs, - openConflictReview, - openModal, - openSettingsPage, - openSettingsTarget, - prGenerationRecords, - pullBranch, - pushBranch, - rebaseFromBase, - revealInExplorer, - setCommitMessageGenerationRecord, - setGitBranchCompareResult, - setGitStatus, - setPullRequestGenerationRecord, - setRightSidebarOpen, - setRightSidebarTab, - setUpstreamStatus, - syncBranch, - updateCommitMessageGenerationRecord, - updatePullRequestGenerationRecord, - updateRepo, - updateSettings, - updateWorktreeGitIdentity, - updateWorktreeMeta - } + return useMemo(() => { + const state = useAppStore.getState() + return { + allocateCommitMessageGenerationRequestId: state.allocateCommitMessageGenerationRequestId, + allocatePullRequestGenerationRequestId: state.allocatePullRequestGenerationRequestId, + beginGitBranchCompareRequest: state.beginGitBranchCompareRequest, + clearDiffComments: state.clearDiffComments, + clearDiffCommentsForFile: state.clearDiffCommentsForFile, + commitMessageGenerationRecords, + createHostedReview: state.createHostedReview, + createStackedHostedReview: state.createStackedHostedReview, + deleteDiffComment: state.deleteDiffComment, + enqueueGitHubPRRefresh: state.enqueueGitHubPRRefresh, + ensureHostedReviewPushTarget: state.ensureHostedReviewPushTarget, + fastForwardBranch: state.fastForwardBranch, + fetchBranch: state.fetchBranch, + fetchHostedReviewForBranch: state.fetchHostedReviewForBranch, + fetchPRForBranch: state.fetchPRForBranch, + fetchUpstreamStatus: state.fetchUpstreamStatus, + getHostedReviewCreationEligibility: state.getHostedReviewCreationEligibility, + openAllDiffs: state.openAllDiffs, + openBranchAllDiffs: state.openBranchAllDiffs, + openConflictReview: state.openConflictReview, + openModal: state.openModal, + openSettingsPage: state.openSettingsPage, + openSettingsTarget: state.openSettingsTarget, + prGenerationRecords, + pullBranch: state.pullBranch, + pushBranch: state.pushBranch, + rebaseFromBase: state.rebaseFromBase, + revealInExplorer: state.revealInExplorer, + setCommitMessageGenerationRecord: state.setCommitMessageGenerationRecord, + setGitBranchCompareResult: state.setGitBranchCompareResult, + setGitStatus: state.setGitStatus, + setPullRequestGenerationRecord: state.setPullRequestGenerationRecord, + setRightSidebarOpen: state.setRightSidebarOpen, + setRightSidebarTab: state.setRightSidebarTab, + setUpstreamStatus: state.setUpstreamStatus, + syncBranch: state.syncBranch, + updateCommitMessageGenerationRecord: state.updateCommitMessageGenerationRecord, + updatePullRequestGenerationRecord: state.updatePullRequestGenerationRecord, + updateRepo: state.updateRepo, + updateSettings: state.updateSettings, + updateWorktreeGitIdentity: state.updateWorktreeGitIdentity, + updateWorktreeMeta: state.updateWorktreeMeta + } + }, [commitMessageGenerationRecords, prGenerationRecords]) } export type SourceControlStoreActions = ReturnType From f03043544a12511b8d56e73b8bb998209c7272db Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:59:42 -0700 Subject: [PATCH 075/398] perf(keybindings): stop recomputing shortcut labels on every render (#18145) Shortcut labels were rebuilt from scratch in the render body of every component that shows one, which kept parseKeybinding running ~120x/sec in a fully idle app. - Cache the label layer per overrides object (WeakMap), so a keybinding edit hands out a new object and therefore a fresh cache. - Memoize parseKeybinding behind a bounded cache; binding strings come from a fixed definition set plus user overrides. - Hoist the per-call token/label object literals in normalizeKeyToken and formatKeyToken to module constants. --- .../src/hooks/shortcut-label-cache.test.tsx | 174 ++++++++++++++++++ src/renderer/src/hooks/useShortcutLabel.ts | 69 +++++-- src/shared/keybindings-parse-cache.test.ts | 95 ++++++++++ src/shared/keybindings/formatting.ts | 65 +++---- src/shared/keybindings/parser.ts | 71 ++++--- 5 files changed, 403 insertions(+), 71 deletions(-) create mode 100644 src/renderer/src/hooks/shortcut-label-cache.test.tsx create mode 100644 src/shared/keybindings-parse-cache.test.ts diff --git a/src/renderer/src/hooks/shortcut-label-cache.test.tsx b/src/renderer/src/hooks/shortcut-label-cache.test.tsx new file mode 100644 index 00000000000..9c7f32a632e --- /dev/null +++ b/src/renderer/src/hooks/shortcut-label-cache.test.tsx @@ -0,0 +1,174 @@ +// @vitest-environment happy-dom +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { act, cleanup, render, screen } from '@testing-library/react' +import type * as KeybindingsModule from '../../../shared/keybindings' +import type { KeybindingOverrides } from '../../../shared/keybindings' + +const counters = vi.hoisted(() => ({ effective: 0, formatList: 0, formatBinding: 0 })) +const platformRef = vi.hoisted(() => ({ current: 'darwin' as NodeJS.Platform })) + +vi.mock('../lib/shortcut-platform', () => ({ + getShortcutPlatform: () => platformRef.current +})) + +vi.mock('../../../shared/keybindings', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + getEffectiveKeybindingsForAction: ( + ...args: Parameters + ) => { + counters.effective++ + return actual.getEffectiveKeybindingsForAction(...args) + }, + formatKeybindingList: (...args: Parameters) => { + counters.formatList++ + return actual.formatKeybindingList(...args) + }, + formatKeybinding: (...args: Parameters) => { + counters.formatBinding++ + return actual.formatKeybinding(...args) + } + } +}) + +const { + formatOptionalShortcutLabel, + formatPrimaryShortcutLabel, + formatShortcutKeyComboDetails, + formatShortcutLabel, + useShortcutLabel +} = await import('./useShortcutLabel') +const { useAppStore } = await import('@/store') + +function resetCounters(): void { + counters.effective = 0 + counters.formatList = 0 + counters.formatBinding = 0 +} + +// A fresh overrides object per test keeps each case on its own cache entry, exactly as a real edit does. +function overridesFor(binding: string): KeybindingOverrides { + return { 'tab.close': [binding] } +} + +describe('shortcut label memoization', () => { + beforeEach(() => { + platformRef.current = 'darwin' + resetCounters() + }) + + it('computes a label once no matter how many times it is asked for', () => { + const overrides = overridesFor('Mod+Shift+K') + const first = formatShortcutLabel('tab.close', overrides) + resetCounters() + for (let index = 0; index < 200; index++) { + expect(formatShortcutLabel('tab.close', overrides)).toBe(first) + } + expect(counters.effective).toBe(0) + expect(counters.formatList).toBe(0) + }) + + it('memoizes each label shape separately and keeps their values correct', () => { + const overrides = overridesFor('Mod+Shift+K') + expect(formatShortcutLabel('tab.close', overrides)).toBe( + formatShortcutLabel('tab.close', overrides) + ) + expect(formatPrimaryShortcutLabel('tab.close', overrides)).toBe( + formatPrimaryShortcutLabel('tab.close', overrides) + ) + expect(formatOptionalShortcutLabel('tab.close', overrides)).toBe( + formatOptionalShortcutLabel('tab.close', overrides) + ) + expect(formatShortcutKeyComboDetails('tab.close', overrides)).toBe( + formatShortcutKeyComboDetails('tab.close', overrides) + ) + expect(formatShortcutKeyComboDetails('tab.close', overrides)[0]?.keys).toEqual(['⌘', '⇧', 'K']) + expect(formatShortcutLabel('tab.close', overrides)).toBe('⌘⇧K') + }) + + it('returns null rather than a cached sentinel for a disabled action', () => { + const overrides: KeybindingOverrides = { 'tab.close': [] } + expect(formatOptionalShortcutLabel('tab.close', overrides)).toBe(null) + expect(formatOptionalShortcutLabel('tab.close', overrides)).toBe(null) + expect(formatShortcutLabel('tab.close', overrides)).toBe('Unassigned') + expect(formatPrimaryShortcutLabel('tab.close', overrides)).toBe('Unassigned') + }) + + it('recomputes as soon as a different overrides object arrives', () => { + expect(formatShortcutLabel('tab.close', overridesFor('Mod+Shift+K'))).toBe('⌘⇧K') + expect(formatShortcutLabel('tab.close', overridesFor('Mod+Shift+L'))).toBe('⌘⇧L') + expect(formatShortcutLabel('tab.close', overridesFor('Mod+Shift+K'))).toBe('⌘⇧K') + }) + + it('does not let one action id serve another', () => { + const overrides = overridesFor('Mod+Shift+K') + expect(formatShortcutLabel('tab.close', overrides)).toBe('⌘⇧K') + expect(formatShortcutLabel('tab.rename', overrides)).not.toBe('⌘⇧K') + }) + + it('keys the cache by platform so Mac and Windows glyphs never cross', () => { + const overrides = overridesFor('Mod+Shift+K') + expect(formatShortcutLabel('tab.close', overrides)).toBe('⌘⇧K') + platformRef.current = 'win32' + expect(formatShortcutLabel('tab.close', overrides)).toBe('Ctrl+Shift+K') + platformRef.current = 'darwin' + expect(formatShortcutLabel('tab.close', overrides)).toBe('⌘⇧K') + }) + + it('caches the undefined-overrides case without leaking into the override case', () => { + const defaultLabel = formatShortcutLabel('tab.close') + expect(formatShortcutLabel('tab.close')).toBe(defaultLabel) + expect(formatShortcutLabel('tab.close', overridesFor('Mod+Shift+K'))).toBe('⌘⇧K') + expect(formatShortcutLabel('tab.close')).toBe(defaultLabel) + }) +}) + +function CloseLabel(): React.JSX.Element { + return {useShortcutLabel('tab.close')} +} + +describe('useShortcutLabel', () => { + beforeEach(() => { + platformRef.current = 'darwin' + resetCounters() + }) + + afterEach(() => { + cleanup() + useAppStore.setState({ keybindings: {} }) + }) + + it('does not recompute the label on re-render', () => { + useAppStore.setState({ keybindings: overridesFor('Mod+Shift+K') }) + const { rerender } = render() + expect(screen.getByTestId('close-label').textContent).toBe('⌘⇧K') + resetCounters() + for (let index = 0; index < 25; index++) { + rerender() + } + expect(screen.getByTestId('close-label').textContent).toBe('⌘⇧K') + expect(counters.effective).toBe(0) + expect(counters.formatList).toBe(0) + }) + + it('shows an edited keybinding immediately, with no stale-cache window', () => { + useAppStore.setState({ keybindings: overridesFor('Mod+Shift+K') }) + render() + expect(screen.getByTestId('close-label').textContent).toBe('⌘⇧K') + + // Mirrors the store update a Settings edit performs: a brand new overrides object. + act(() => useAppStore.setState({ keybindings: overridesFor('Mod+Shift+L') })) + expect(screen.getByTestId('close-label').textContent).toBe('⌘⇧L') + + act(() => useAppStore.setState({ keybindings: { 'tab.close': ['Mod+Alt+Backspace'] } })) + expect(screen.getByTestId('close-label').textContent).toBe('⌘⌥⌫') + + act(() => useAppStore.setState({ keybindings: { 'tab.close': [] } })) + expect(screen.getByTestId('close-label').textContent).toBe('Unassigned') + + // Back to the first binding: a revert must not resurrect the entry cached for the old object. + act(() => useAppStore.setState({ keybindings: overridesFor('Mod+Shift+K') })) + expect(screen.getByTestId('close-label').textContent).toBe('⌘⇧K') + }) +}) diff --git a/src/renderer/src/hooks/useShortcutLabel.ts b/src/renderer/src/hooks/useShortcutLabel.ts index 1d96ad8745a..fe78739f687 100644 --- a/src/renderer/src/hooks/useShortcutLabel.ts +++ b/src/renderer/src/hooks/useShortcutLabel.ts @@ -16,14 +16,48 @@ export type ShortcutKeyComboDetails = { doubleTap: boolean } +// Why: these run in render bodies of components that re-render constantly, and every call is two full keybinding parses. +// The store hands out a new overrides object on every keybinding edit, so keying the cache on that object gives exact +// invalidation: an edit can never be served from a stale entry, and the old entry is dropped with the old object. +const cachesByOverrides = new WeakMap>() +const defaultOverridesCache = new Map() + +function labelCache(overrides: KeybindingOverrides | undefined): Map { + if (!overrides) { + return defaultOverridesCache + } + let cache = cachesByOverrides.get(overrides) + if (!cache) { + cache = new Map() + cachesByOverrides.set(overrides, cache) + } + return cache +} + +function memoizeShortcut( + kind: string, + actionId: KeybindingActionId, + platform: NodeJS.Platform, + overrides: KeybindingOverrides | undefined, + compute: () => T +): T { + const cache = labelCache(overrides) + const key = `${platform}\u0000${kind}\u0000${actionId}` + if (cache.has(key)) { + return cache.get(key) as T + } + const value = compute() + cache.set(key, value) + return value +} + export function formatShortcutLabel( actionId: KeybindingActionId, overrides?: KeybindingOverrides ): string { const platform = getShortcutPlatform() - return formatKeybindingList( - getEffectiveKeybindingsForAction(actionId, platform, overrides), - platform + return memoizeShortcut('label', actionId, platform, overrides, () => + formatKeybindingList(getEffectiveKeybindingsForAction(actionId, platform, overrides), platform) ) } @@ -32,8 +66,10 @@ export function formatPrimaryShortcutLabel( overrides?: KeybindingOverrides ): string { const platform = getShortcutPlatform() - const [binding] = getEffectiveKeybindingsForAction(actionId, platform, overrides) - return binding ? formatKeybindingList([binding], platform) : 'Unassigned' + return memoizeShortcut('primary', actionId, platform, overrides, () => { + const [binding] = getEffectiveKeybindingsForAction(actionId, platform, overrides) + return binding ? formatKeybindingList([binding], platform) : 'Unassigned' + }) } export function useShortcutLabel(actionId: KeybindingActionId): string { @@ -49,11 +85,13 @@ export function formatOptionalShortcutLabel( overrides?: KeybindingOverrides ): string | null { const platform = getShortcutPlatform() - const bindings = getEffectiveKeybindingsForAction(actionId, platform, overrides) - if (bindings.length === 0) { - return null - } - return formatKeybindingList(bindings, platform) + return memoizeShortcut('optional', actionId, platform, overrides, () => { + const bindings = getEffectiveKeybindingsForAction(actionId, platform, overrides) + if (bindings.length === 0) { + return null + } + return formatKeybindingList(bindings, platform) + }) } export function useOptionalShortcutLabel(actionId: KeybindingActionId): string | null { @@ -66,10 +104,13 @@ export function formatShortcutKeyComboDetails( overrides?: KeybindingOverrides ): ShortcutKeyComboDetails[] { const platform = getShortcutPlatform() - return getEffectiveKeybindingsForAction(actionId, platform, overrides).map((binding) => ({ - keys: formatKeybinding(binding, platform), - doubleTap: isDoubleTapBinding(binding) - })) + // The returned array is shared across callers now, so treat it as read-only (every current caller does). + return memoizeShortcut('combo', actionId, platform, overrides, () => + getEffectiveKeybindingsForAction(actionId, platform, overrides).map((binding) => ({ + keys: formatKeybinding(binding, platform), + doubleTap: isDoubleTapBinding(binding) + })) + ) } export function useShortcutKeyComboDetails( diff --git a/src/shared/keybindings-parse-cache.test.ts b/src/shared/keybindings-parse-cache.test.ts new file mode 100644 index 00000000000..d2eb899b309 --- /dev/null +++ b/src/shared/keybindings-parse-cache.test.ts @@ -0,0 +1,95 @@ +import { describe, expect, it } from 'vitest' +import { KEYBINDING_DEFINITIONS } from './keybindings/definitions' +import { getDefaultBindings } from './keybindings/effective' +import { formatKeybindingList } from './keybindings/formatting' +import { normalizeKeyToken, parseKeybinding } from './keybindings/parser' + +describe('parseKeybinding memoization', () => { + it('reuses the parsed result for a repeated binding string', () => { + const first = parseKeybinding('Mod+Shift+K') + const second = parseKeybinding('Mod+Shift+K') + expect(first).not.toBeNull() + expect(second).toBe(first) + }) + + it('caches rejections without re-parsing', () => { + expect(parseKeybinding('K+J')).toBeNull() + expect(parseKeybinding('K+J')).toBeNull() + }) + + it('never hands out an entry a caller can corrupt', () => { + const parsed = parseKeybinding('Mod+P') + expect(Object.isFrozen(parsed)).toBe(true) + expect(parseKeybinding('Mod+P')?.key).toBe('P') + }) + + it('stays bounded and correct when fed far more strings than the cache holds', () => { + for (let index = 0; index < 2000; index++) { + expect(parseKeybinding(`Mod+Alt+F${(index % 24) + 1}`)?.key).toBe(`F${(index % 24) + 1}`) + } + // A cleared cache must still return the right answer, not a stale neighbour. + expect(parseKeybinding('Mod+Shift+K')?.key).toBe('K') + expect(parseKeybinding('DoubleTap+Shift')?.doubleTapModifier).toBe('Shift') + }) + + it('parses every distinct token form identically across repeat calls', () => { + const tokens = [ + ' ', + 'a', + '7', + 'f7', + '[', + '}', + '-', + '_', + '=', + '+', + ',', + '.', + '/', + '\\', + ';', + "'", + '`', + 'return', + 'esc', + 'spacebar', + 'pgup', + 'pgdn', + 'arrowleft', + 'left', + 'down', + 'backspace', + 'del', + 'ins', + 'numpadadd', + 'subtract', + 'nonsense', + '' + ] + for (const token of tokens) { + expect(normalizeKeyToken(token)).toBe(normalizeKeyToken(token)) + } + expect(normalizeKeyToken(' ')).toBe('Space') + expect(normalizeKeyToken('pgdn')).toBe('PageDown') + expect(normalizeKeyToken('subtract')).toBe('NumpadSubtract') + expect(normalizeKeyToken('nonsense')).toBe(null) + expect(normalizeKeyToken('')).toBe(null) + // Object.prototype keys must not leak through the token table. + expect(normalizeKeyToken('constructor')).toBe(null) + expect(normalizeKeyToken('__proto__')).toBe(null) + }) +}) + +describe('shortcut label output', () => { + it('formats every default binding identically on repeat calls, on both glyph platforms', () => { + for (const platform of ['darwin', 'win32'] as const) { + for (const definition of KEYBINDING_DEFINITIONS) { + const bindings = getDefaultBindings(definition, platform) + const label = formatKeybindingList(bindings, platform) + expect(formatKeybindingList(bindings, platform)).toBe(label) + expect(label.length).toBeGreaterThan(0) + } + } + }) +}) diff --git a/src/shared/keybindings/formatting.ts b/src/shared/keybindings/formatting.ts index 1c3c75ba8fd..c9e2b35ac43 100644 --- a/src/shared/keybindings/formatting.ts +++ b/src/shared/keybindings/formatting.ts @@ -90,38 +90,41 @@ export function findKeybindingActionsForBinding( ).map((definition) => definition.id) } +const KEY_TOKEN_LABELS: Record = { + BracketLeft: '[', + BracketRight: ']', + Minus: '-', + Underscore: '_', + Equal: '=', + Plus: '+', + ArrowLeft: '←', + ArrowRight: '→', + ArrowUp: '↑', + ArrowDown: '↓', + PageUp: 'PageUp', + PageDown: 'PageDown', + NumpadAdd: 'Numpad +', + NumpadSubtract: 'Numpad -', + Comma: ',', + Period: '.', + Slash: '/', + Backslash: '\\', + Semicolon: ';', + Quote: "'", + Backquote: '`', + Enter: 'Enter', + Backspace: 'Backspace', + Delete: 'Delete', + Insert: 'Insert', + Tab: 'Tab', + Escape: 'Esc', + Space: 'Space' +} + +const MAC_KEY_TOKEN_LABELS: Record = { ...KEY_TOKEN_LABELS, Backspace: '⌫' } + function formatKeyToken(token: string, isMac: boolean): string { - const labels: Record = { - BracketLeft: '[', - BracketRight: ']', - Minus: '-', - Underscore: '_', - Equal: '=', - Plus: '+', - ArrowLeft: '←', - ArrowRight: '→', - ArrowUp: '↑', - ArrowDown: '↓', - PageUp: 'PageUp', - PageDown: 'PageDown', - NumpadAdd: 'Numpad +', - NumpadSubtract: 'Numpad -', - Comma: ',', - Period: '.', - Slash: '/', - Backslash: '\\', - Semicolon: ';', - Quote: "'", - Backquote: '`', - Enter: 'Enter', - Backspace: isMac ? '⌫' : 'Backspace', - Delete: 'Delete', - Insert: 'Insert', - Tab: 'Tab', - Escape: 'Esc', - Space: 'Space' - } - return labels[token] ?? token + return (isMac ? MAC_KEY_TOKEN_LABELS : KEY_TOKEN_LABELS)[token] ?? token } export function findKeybindingConflicts( diff --git a/src/shared/keybindings/parser.ts b/src/shared/keybindings/parser.ts index 022aa4bec29..0c3180d9d00 100644 --- a/src/shared/keybindings/parser.ts +++ b/src/shared/keybindings/parser.ts @@ -16,31 +16,8 @@ export function hasModifier( return Boolean(input.shift ?? input.shiftKey) } -function isFunctionKeyToken(key: string): boolean { - return /^F([1-9]|1[0-9]|2[0-4])$/.test(key) -} - -export function normalizeKeyToken(token: string): string | null { - if (token === ' ') { - return 'Space' - } - const trimmed = token.trim() - if (!trimmed) { - return null - } - const upper = trimmed.toUpperCase() - if (upper.length === 1 && upper >= 'A' && upper <= 'Z') { - return upper - } - if (upper.length === 1 && upper >= '0' && upper <= '9') { - return upper - } - // Function keys F1–F24 (event.key/event.code report them verbatim, e.g. F7). - if (isFunctionKeyToken(upper)) { - return upper - } - - const simple: Record = { +const SIMPLE_KEY_TOKENS = new Map( + Object.entries({ '[': 'BracketLeft', ']': 'BracketRight', '{': 'BracketLeft', @@ -97,9 +74,34 @@ export function normalizeKeyToken(token: string): string | null { SEMICOLON: 'Semicolon', QUOTE: 'Quote', BACKQUOTE: 'Backquote' + }) +) + +function isFunctionKeyToken(key: string): boolean { + return /^F([1-9]|1[0-9]|2[0-4])$/.test(key) +} + +export function normalizeKeyToken(token: string): string | null { + if (token === ' ') { + return 'Space' + } + const trimmed = token.trim() + if (!trimmed) { + return null + } + const upper = trimmed.toUpperCase() + if (upper.length === 1 && upper >= 'A' && upper <= 'Z') { + return upper + } + if (upper.length === 1 && upper >= '0' && upper <= '9') { + return upper + } + // Function keys F1–F24 (event.key/event.code report them verbatim, e.g. F7). + if (isFunctionKeyToken(upper)) { + return upper } - return simple[upper] ?? null + return SIMPLE_KEY_TOKENS.get(upper) ?? null } export function parseModifierToken(rawPart: string): ModifierToken | null { @@ -177,7 +179,24 @@ export function parseDoubleTapKeybinding(rawParts: string[]): ParsedKeybinding | return parsed } +// Binding strings come from a fixed definition set plus user overrides, so the live set is tiny; the cap only guards a caller feeding arbitrary strings. +const PARSE_CACHE_LIMIT = 512 +const parseCache = new Map() + export function parseKeybinding(binding: string): ParsedKeybinding | null { + if (parseCache.has(binding)) { + return parseCache.get(binding) ?? null + } + const parsed = parseKeybindingUncached(binding) + if (parseCache.size >= PARSE_CACHE_LIMIT) { + parseCache.clear() + } + // Frozen so a caller can never corrupt the shared entry; every current caller spread-copies before changing a field. + parseCache.set(binding, parsed ? Object.freeze(parsed) : null) + return parsed +} + +function parseKeybindingUncached(binding: string): ParsedKeybinding | null { const rawParts = binding .split('+') .map((part) => part.trim()) From c0e5b189aa08844aa71f855edbc9d1be5765d17b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 12:59:46 -0700 Subject: [PATCH 076/398] perf(sidebar,terminal): memoize terminal-title agent classification and lineage projections (#18148) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Idle-app CPU profiling showed `titleHasAgentName` running 11,771x/sec and the legacy any-agent regex 4,399x/sec, roughly once per zustand subscriber notify. The regexes were already precompiled; the problem was call volume — every store write re-classified every unchanged pane title through the whole agent-name ladder. Every title classifier is pure in the title string, so memoize them on it (bounded FIFO, 1024 entries). A new title is a new key, so there is no staleness window. The same profile showed the sidebar lineage projection re-scanning all worktrees several times per pass; cache it on the identity pair of its two immutable inputs, mirroring store/worktree-repo-index.ts. --- .../worktree-lineage-projection.test.ts | 126 ++++++++ .../sidebar/worktree-lineage-projection.ts | 48 ++- src/shared/agent-title-core.ts | 8 +- src/shared/agent-title-identity.ts | 13 +- src/shared/agent-title-status.ts | 10 +- src/shared/terminal-title-agent-type.ts | 19 +- ...rminal-title-classification-corpus.test.ts | 293 ++++++++++++++++++ .../terminal-title-classification-corpus.ts | 86 +++++ ...terminal-title-classification-memo.test.ts | 98 ++++++ .../terminal-title-classification-memo.ts | 43 +++ 10 files changed, 736 insertions(+), 8 deletions(-) create mode 100644 src/renderer/src/components/sidebar/worktree-lineage-projection.test.ts create mode 100644 src/shared/terminal-title-classification-corpus.test.ts create mode 100644 src/shared/terminal-title-classification-corpus.ts create mode 100644 src/shared/terminal-title-classification-memo.test.ts create mode 100644 src/shared/terminal-title-classification-memo.ts diff --git a/src/renderer/src/components/sidebar/worktree-lineage-projection.test.ts b/src/renderer/src/components/sidebar/worktree-lineage-projection.test.ts new file mode 100644 index 00000000000..8d2ebcdb517 --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-lineage-projection.test.ts @@ -0,0 +1,126 @@ +import { describe, expect, it } from 'vitest' +import type { WorktreeLineage } from '../../../../shared/worktree/lineage-types' +import type { Worktree } from '../../../../shared/worktree/types' +import { + getCyclicProjectedWorktreeLineageIds, + getLineageRenderInfo, + getProjectedWorktreeLineageChildrenByParentId +} from './worktree-lineage-projection' + +function makeWorktree(id: string): Worktree { + return { + id, + repoId: 'repo-1', + instanceId: `${id}-instance`, + path: `/tmp/${id}`, + branch: id, + isMainWorktree: false + } as unknown as Worktree +} + +function makeLineage(childId: string, parentId: string): WorktreeLineage { + return { + worktreeId: childId, + worktreeInstanceId: `${childId}-instance`, + parentWorktreeId: parentId, + parentWorktreeInstanceId: `${parentId}-instance` + } as unknown as WorktreeLineage +} + +/** + * A sidebar-scale fixture: one root with many children, mirroring the shape the + * row builder scans on every store write. + */ +function buildFixture(childCount: number): { + lineageById: Record + worktreeMap: Map +} { + const worktreeMap = new Map() + const lineageById: Record = {} + worktreeMap.set('root', makeWorktree('root')) + for (let index = 0; index < childCount; index += 1) { + const id = `child-${index}` + worktreeMap.set(id, makeWorktree(id)) + lineageById[id] = makeLineage(id, 'root') + } + return { lineageById, worktreeMap } +} + +describe('worktree lineage projection cache', () => { + it('reuses the cyclic-id scan for an unchanged input pair', () => { + const { lineageById, worktreeMap } = buildFixture(8) + const first = getCyclicProjectedWorktreeLineageIds(lineageById, worktreeMap) + const second = getCyclicProjectedWorktreeLineageIds(lineageById, worktreeMap) + expect(second).toBe(first) + }) + + it('reuses the children projection for an unchanged input pair', () => { + const { lineageById, worktreeMap } = buildFixture(8) + const first = getProjectedWorktreeLineageChildrenByParentId(lineageById, worktreeMap) + const second = getProjectedWorktreeLineageChildrenByParentId(lineageById, worktreeMap) + expect(second).toBe(first) + expect(first.get('root')?.map((worktree) => worktree.id)).toEqual([ + 'child-0', + 'child-1', + 'child-2', + 'child-3', + 'child-4', + 'child-5', + 'child-6', + 'child-7' + ]) + }) + + it('rescans when either input is replaced', () => { + const { lineageById, worktreeMap } = buildFixture(4) + const baseline = getProjectedWorktreeLineageChildrenByParentId(lineageById, worktreeMap) + + const replacedLineage = { ...lineageById } + expect(getProjectedWorktreeLineageChildrenByParentId(replacedLineage, worktreeMap)).not.toBe( + baseline + ) + + const replacedWorktrees = new Map(worktreeMap) + expect(getProjectedWorktreeLineageChildrenByParentId(lineageById, replacedWorktrees)).not.toBe( + baseline + ) + }) + + it('reflects a removed lineage edge as soon as the record is replaced', () => { + const { lineageById, worktreeMap } = buildFixture(2) + expect( + getProjectedWorktreeLineageChildrenByParentId(lineageById, worktreeMap).get('root') + ).toHaveLength(2) + + const withoutFirstChild = { ...lineageById } + delete withoutFirstChild['child-0'] + const reprojected = getProjectedWorktreeLineageChildrenByParentId( + withoutFirstChild, + worktreeMap + ) + expect(reprojected.get('root')?.map((worktree) => worktree.id)).toEqual(['child-1']) + expect( + getLineageRenderInfo( + worktreeMap.get('child-0') as Worktree, + withoutFirstChild, + worktreeMap, + getCyclicProjectedWorktreeLineageIds(withoutFirstChild, worktreeMap) + ).state + ).toBe('none') + }) + + it('still reports cycles from the cached scan', () => { + const worktreeMap = new Map([ + ['a', makeWorktree('a')], + ['b', makeWorktree('b')] + ]) + const lineageById: Record = { + a: makeLineage('a', 'b'), + b: makeLineage('b', 'a') + } + const cyclic = getCyclicProjectedWorktreeLineageIds(lineageById, worktreeMap) + expect([...cyclic].sort()).toEqual(['a', 'b']) + expect(getCyclicProjectedWorktreeLineageIds(lineageById, worktreeMap)).toBe(cyclic) + expect(getProjectedWorktreeLineageChildrenByParentId(lineageById, worktreeMap).size).toBe(0) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-lineage-projection.ts b/src/renderer/src/components/sidebar/worktree-lineage-projection.ts index 337140f586c..b8748f287c5 100644 --- a/src/renderer/src/components/sidebar/worktree-lineage-projection.ts +++ b/src/renderer/src/components/sidebar/worktree-lineage-projection.ts @@ -22,10 +22,49 @@ export function getProjectedWorktreeLineage( return (worktree as WorktreeWithResolvedLineage).lineage } +type LineageProjection = { + cyclicLineageIds?: Set + childrenByParentId?: Map +} + +/** + * Why: both projections are O(worktrees) scans that the sidebar row builder and + * the pinned/attached-children readers re-run several times per pass, and + * zustand re-runs those on every store write. Both are pure in the two inputs, + * and both inputs are immutable store-derived collections that are REPLACED + * rather than mutated, so their identity pair is a sound cache key. Weak on both + * levels so a superseded lineage record or worktree index is not pinned. + */ +const projectionByLineageAndWorktreeMap = new WeakMap< + Readonly>, + WeakMap, LineageProjection> +>() + +function getLineageProjection( + lineageById: Readonly>, + worktreeMap: ReadonlyMap +): LineageProjection { + let byWorktreeMap = projectionByLineageAndWorktreeMap.get(lineageById) + if (!byWorktreeMap) { + byWorktreeMap = new WeakMap() + projectionByLineageAndWorktreeMap.set(lineageById, byWorktreeMap) + } + let projection = byWorktreeMap.get(worktreeMap) + if (!projection) { + projection = {} + byWorktreeMap.set(worktreeMap, projection) + } + return projection +} + export function getCyclicProjectedWorktreeLineageIds( lineageById: Readonly>, worktreeMap: ReadonlyMap ): Set { + const projection = getLineageProjection(lineageById, worktreeMap) + if (projection.cyclicLineageIds) { + return projection.cyclicLineageIds + } const validLineageByChildId = new Map() for (const worktree of worktreeMap.values()) { const lineage = getProjectedWorktreeLineage(worktree, lineageById) @@ -37,7 +76,9 @@ export function getCyclicProjectedWorktreeLineageIds( validLineageByChildId.set(worktree.id, lineage) } } - return getCyclicWorktreeLineageChildIds(validLineageByChildId) + const cyclicLineageIds = getCyclicWorktreeLineageChildIds(validLineageByChildId) + projection.cyclicLineageIds = cyclicLineageIds + return cyclicLineageIds } export function getLineageRenderInfo( @@ -65,6 +106,10 @@ export function getProjectedWorktreeLineageChildrenByParentId( lineageById: Readonly>, worktreeMap: ReadonlyMap ): Map { + const projection = getLineageProjection(lineageById, worktreeMap) + if (projection.childrenByParentId) { + return projection.childrenByParentId + } const cyclicLineageIds = getCyclicProjectedWorktreeLineageIds(lineageById, worktreeMap) const childrenByParentId = new Map() for (const worktree of worktreeMap.values()) { @@ -76,6 +121,7 @@ export function getProjectedWorktreeLineageChildrenByParentId( children.push(worktree) childrenByParentId.set(lineage.parent.id, children) } + projection.childrenByParentId = childrenByParentId return childrenByParentId } diff --git a/src/shared/agent-title-core.ts b/src/shared/agent-title-core.ts index 5d2c688b7d2..3d0bdbc0feb 100644 --- a/src/shared/agent-title-core.ts +++ b/src/shared/agent-title-core.ts @@ -7,6 +7,7 @@ import { } from './agent-name-token-match' import { stripLeadingAgentTitleDecorationOrEmpty } from './agent-title-decoration' import { isLegacyPiCompatibleTitle } from './pi-compatible-synthetic-title' +import { memoizeTitleClassification } from './terminal-title-classification-memo' import { getWrapperTitleSegments } from './terminal-title-wrapper-segments' export { AGY_AGENT_NAME_RE, DROID_AGENT_NAME_RE, HERMES_AGENT_NAME_RE, titleHasAgentName } @@ -54,7 +55,7 @@ export const BRAILLE_SPINNER_RE = /[\u2800-\u28ff]/g // Reserve the whole quarter-circle block so a later frame addition cannot regress this. export const QUARTER_CIRCLE_SPINNER_RE = /[\u25d0-\u25d3]/g -export function isGeminiTerminalTitle(title: string): boolean { +function computeIsGeminiTerminalTitle(title: string): boolean { // Why: Gemini OSC glyphs are stronger evidence than any cwd/session text. if ( title.includes(GEMINI_PERMISSION) || @@ -80,6 +81,11 @@ export function isGeminiTerminalTitle(title: string): boolean { return titleHasAgentName(title, 'gemini') } +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const isGeminiTerminalTitle: (title: string) => boolean = memoizeTitleClassification( + computeIsGeminiTerminalTitle +) + export function isPiTerminalTitle(title: string): boolean { return isLegacyPiCompatibleTitle(title) && !containsBrailleSpinner(title) } diff --git a/src/shared/agent-title-identity.ts b/src/shared/agent-title-identity.ts index cdfc6946777..2b5194bfda8 100644 --- a/src/shared/agent-title-identity.ts +++ b/src/shared/agent-title-identity.ts @@ -12,12 +12,13 @@ import { } from './agent-title-core' import { isOpenCodeNativeTitle } from './opencode-terminal-title' import { getPiCompatibleSyntheticAgentLabel } from './pi-compatible-synthetic-title' +import { memoizeTitleClassification } from './terminal-title-classification-memo' /** * Returns true when the terminal title matches Claude Code's title conventions. * Used to scope prompt-cache-timer behavior to Claude sessions only. */ -export function isClaudeAgent(title: string): boolean { +function computeIsClaudeAgent(title: string): boolean { if (!title || isClaudeManagementTitle(title) || isOpenCodeNativeTitle(title)) { return false } @@ -43,7 +44,11 @@ export function isClaudeAgent(title: string): boolean { ) } -export function getAgentLabel(title: string): string | null { +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const isClaudeAgent: (title: string) => boolean = + memoizeTitleClassification(computeIsClaudeAgent) + +function computeAgentLabel(title: string): string | null { if (isClaudeManagementTitle(title)) { return null } @@ -119,3 +124,7 @@ export function getAgentLabel(title: string): string | null { return null } + +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const getAgentLabel: (title: string) => string | null = + memoizeTitleClassification(computeAgentLabel) diff --git a/src/shared/agent-title-status.ts b/src/shared/agent-title-status.ts index 423af759281..fa1e35652e2 100644 --- a/src/shared/agent-title-status.ts +++ b/src/shared/agent-title-status.ts @@ -32,6 +32,7 @@ import { import { clearPiStateWorkingMarker, getPiStateTitleStatus } from './pi-state-title-marker' import { getWrapperTitleSegments } from './terminal-title-wrapper-segments' import { isGrokRotatingWorkingTitle } from './terminal-title-agent-type' +import { memoizeTitleClassification } from './terminal-title-classification-memo' /** * Strip working-status indicators so stale exit titles stop reporting working. @@ -178,7 +179,7 @@ function canonicalizeBrailleSpinnerFrame(title: string): string { return canonical } -export function detectAgentStatusFromTitle(title: string): AgentStatus | null { +function computeAgentStatusFromTitle(title: string): AgentStatus | null { if (!title || isClaudeManagementTitle(title)) { return null } @@ -262,6 +263,13 @@ export function detectAgentStatusFromTitle(title: string): AgentStatus | null { return 'idle' } +/** + * Pure in `title`, so it is memoized on the title string: sidebar/tab selectors + * re-ask for the same unchanged titles on every store write. + */ +export const detectAgentStatusFromTitle: (title: string) => AgentStatus | null = + memoizeTitleClassification(computeAgentStatusFromTitle) + /** * True when a quarter-circle spinner frame is the only agent evidence a title carries. * Any TUI animates those glyphs, so they prove activity, not identity — callers that diff --git a/src/shared/terminal-title-agent-type.ts b/src/shared/terminal-title-agent-type.ts index 4a078fde877..cdbce788806 100644 --- a/src/shared/terminal-title-agent-type.ts +++ b/src/shared/terminal-title-agent-type.ts @@ -10,6 +10,7 @@ import { getPiCompatibleSyntheticAgentLabel, isLegacyPiCompatibleTitle } from './pi-compatible-synthetic-title' +import { memoizeTitleClassification } from './terminal-title-classification-memo' import type { TuiAgent } from './tui-agent' export const CLAUDE_IDLE = '\u2733' // ✳ (eight-spoked asterisk — Claude Code idle prefix) @@ -84,7 +85,7 @@ export function isPiAgentTitle(title: string): boolean { * Used to scope prompt-cache-timer behavior to Claude sessions only — other * agents have different (or no) caching semantics. */ -export function isClaudeAgent(title: string): boolean { +function computeIsClaudeAgent(title: string): boolean { if (!title || isClaudeManagementTitle(title) || isOpenCodeNativeTitle(title)) { return false } @@ -121,11 +122,15 @@ export function isClaudeAgent(title: string): boolean { return false } +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const isClaudeAgent: (title: string) => boolean = + memoizeTitleClassification(computeIsClaudeAgent) + export function isClaudeManagementTitle(title: string): boolean { return CLAUDE_MANAGEMENT_TITLE_RE.test(title) } -export function getAgentLabel(title: string): string | null { +function computeAgentLabel(title: string): string | null { if (isClaudeManagementTitle(title)) { return null } @@ -235,6 +240,10 @@ const TITLE_LABEL_TO_AGENT: Partial> = { OMP: 'omp' } +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const getAgentLabel: (title: string) => string | null = + memoizeTitleClassification(computeAgentLabel) + function hasGenericClaudeStatusPrefix(title: string): boolean { return ( containsAgentSpinnerGlyph(title) || @@ -266,10 +275,14 @@ export function resolveTerminalTitleAgentType(title: string): TuiAgent | null { * that something is running, not proof the agent is Claude — so a task or * worktree title cannot become Claude without an explicit "Claude Code" name. */ -export function resolveExplicitTerminalTitleAgentType(title: string): TuiAgent | null { +function computeExplicitTerminalTitleAgentType(title: string): TuiAgent | null { const titleAgent = resolveTerminalTitleAgentType(title) if (isGenericClaudeStatusClaim(title, titleAgent)) { return null } return titleAgent } + +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const resolveExplicitTerminalTitleAgentType: (title: string) => TuiAgent | null = + memoizeTitleClassification(computeExplicitTerminalTitleAgentType) diff --git a/src/shared/terminal-title-classification-corpus.test.ts b/src/shared/terminal-title-classification-corpus.test.ts new file mode 100644 index 00000000000..ae41d451b20 --- /dev/null +++ b/src/shared/terminal-title-classification-corpus.test.ts @@ -0,0 +1,293 @@ +import { describe, expect, it } from 'vitest' +import { isGeminiTerminalTitle } from './agent-title-core' +import { getAgentLabel, isClaudeAgent } from './agent-title-identity' +import { detectAgentStatusFromTitle } from './agent-title-status' +import { TERMINAL_TITLE_CLASSIFICATION_CORPUS } from './terminal-title-classification-corpus' +import { + getAgentLabel as getExplicitAgentLabel, + isClaudeAgent as isExplicitClaudeAgent, + resolveExplicitTerminalTitleAgentType, + resolveTerminalTitleAgentType +} from './terminal-title-agent-type' + +/** + * Pins the exact verdict every title classifier returns for a realistic corpus. + * + * Why: these classifiers are now memoized on the title string, and a caching bug + * here would repaint a pane under the wrong agent. This table is the proof that + * memoization is transparent — it was generated from the pre-memo implementation + * and must keep matching byte-for-byte. + */ +type PinnedRow = [ + title: string, + status: string | null, + label: string | null, + claude: boolean, + gemini: boolean, + explicitLabel: string | null, + explicitClaude: boolean, + titleAgent: string | null, + explicitTitleAgent: string | null +] + +const PINNED_CLASSIFICATIONS: readonly PinnedRow[] = [ + ['', null, null, false, false, null, false, null, null], + ['zsh', null, null, false, false, null, false, null, null], + ['bash', null, null, false, false, null, false, null, null], + ['nwparker@mac: ~/orca', null, null, false, false, null, false, null, null], + ['npm run dev', null, null, false, false, null, false, null, null], + ['opencode-blinker', null, null, false, false, null, false, null, null], + [ + 'openclaude', + 'idle', + 'OpenClaude', + false, + false, + 'OpenClaude', + false, + 'openclaude', + 'openclaude' + ], + ['openclaude-scratch', null, null, false, false, null, false, null, null], + ['claude-scratch', null, null, false, false, null, false, null, null], + ['~/codex/ready', null, null, false, false, null, false, null, null], + ['review-14600-codex', null, null, false, false, null, false, null, null], + ['timestamp ready', null, null, false, false, null, false, null, null], + ['android build running', null, null, false, false, null, false, null, null], + ['~/hermes/working', null, null, false, false, null, false, null, null], + ['C:\\tools\\codex\\run', null, null, false, false, null, false, null, null], + ['/usr/local/bin/claude/notes', null, null, false, false, null, false, null, null], + ['agy-nightly', null, null, false, false, null, false, null, null], + ['codex.exe', 'idle', 'Codex', false, false, 'Codex', false, 'codex', 'codex'], + [ + 'openclaude.cmd', + 'idle', + 'OpenClaude', + false, + false, + 'OpenClaude', + false, + 'openclaude', + 'openclaude' + ], + [ + 'claude.bat working', + 'working', + 'Claude Code', + true, + false, + 'Claude Code', + true, + 'claude', + 'claude' + ], + ['aider.ps1 ready', 'idle', 'Aider', false, false, 'Aider', false, 'aider', 'aider'], + [ + 'copilot.exe - action required', + 'permission', + 'GitHub Copilot', + false, + false, + 'GitHub Copilot', + false, + 'copilot', + 'copilot' + ], + ['droid.exe', null, null, false, false, null, false, null, null], + ['\u2733', 'idle', 'Claude Code', true, false, 'Claude Code', true, 'claude', null], + [ + '\u2733 Claude Code', + 'idle', + 'Claude Code', + true, + false, + 'Claude Code', + true, + 'claude', + 'claude' + ], + ['\u2733 ready', 'idle', 'Claude Code', true, false, 'Claude Code', true, 'claude', null], + ['. building the parser', null, 'Claude Code', true, false, 'Claude Code', true, 'claude', null], + ['* done', null, 'Claude Code', true, false, 'Claude Code', true, 'claude', null], + ['Claude Code', 'idle', 'Claude Code', true, false, 'Claude Code', true, 'claude', 'claude'], + [ + 'claude - action required', + 'permission', + 'Claude Code', + true, + false, + 'Claude Code', + true, + 'claude', + 'claude' + ], + ['Claude ready', 'idle', 'Claude Code', true, false, 'Claude Code', true, 'claude', 'claude'], + ['claude agents', null, null, false, false, null, false, null, null], + ['"/usr/local/bin/claude" agents', null, null, false, false, null, false, null, null], + [ + '\u280b Claude Code', + 'working', + 'Claude Code', + true, + false, + 'Claude Code', + true, + 'claude', + 'claude' + ], + [ + '\u2809 Codex \u2014 refactoring', + 'working', + 'Codex', + true, + false, + 'Codex', + true, + 'codex', + 'codex' + ], + ['\u25d0 working', 'working', 'Claude Code', true, false, 'Claude Code', true, 'claude', null], + ['\u25d3 Grok', 'working', 'Grok', true, false, 'Grok', true, 'grok', 'grok'], + ['\u280b Cursor Agent', 'working', 'Cursor', false, false, 'Cursor', false, 'cursor', 'cursor'], + ['\u280b Droid', 'working', 'Droid', true, false, 'Droid', true, 'droid', 'droid'], + ['\u280b Hermes', 'working', 'Hermes', true, false, 'Hermes', true, 'hermes', 'hermes'], + ['\u2726 gemini', 'working', 'Gemini CLI', false, true, 'Gemini CLI', false, 'gemini', 'gemini'], + [ + '\u23f2 Gemini CLI', + 'working', + 'Gemini CLI', + false, + true, + 'Gemini CLI', + false, + 'gemini', + 'gemini' + ], + ['\u25c7 Gemini CLI', 'idle', 'Gemini CLI', false, true, 'Gemini CLI', false, 'gemini', 'gemini'], + [ + '\u270b Gemini CLI', + 'permission', + 'Gemini CLI', + false, + true, + 'Gemini CLI', + false, + 'gemini', + 'gemini' + ], + ['gemini', 'idle', 'Gemini CLI', false, true, 'Gemini CLI', false, 'gemini', 'gemini'], + [ + 'antigravity gemini 3 pro', + 'idle', + 'Antigravity', + false, + false, + 'Antigravity', + false, + 'antigravity', + 'antigravity' + ], + [ + 'agy - gemini 2 flash', + 'idle', + 'Antigravity', + false, + false, + 'Antigravity', + false, + 'antigravity', + 'antigravity' + ], + ['codex working', 'working', 'Codex', false, false, 'Codex', false, 'codex', 'codex'], + ['codex ready', 'idle', 'Codex', false, false, 'Codex', false, 'codex', 'codex'], + [ + 'copilot waiting', + 'permission', + 'GitHub Copilot', + false, + false, + 'GitHub Copilot', + false, + 'copilot', + 'copilot' + ], + ['devin thinking', 'working', 'Devin', false, false, 'Devin', false, 'devin', 'devin'], + ['mimo idle', 'idle', 'MiMo Code', false, false, 'MiMo Code', false, 'mimo-code', 'mimo-code'], + ['aider running', 'working', 'Aider', false, false, 'Aider', false, 'aider', 'aider'], + ['grok done', 'idle', 'Grok', false, false, 'Grok', false, 'grok', 'grok'], + ['opencode ready', 'idle', 'OpenCode', false, false, 'OpenCode', false, 'opencode', 'opencode'], + ['hermes ready', 'idle', 'Hermes', false, false, 'Hermes', false, 'hermes', 'hermes'], + ['droid ready', 'idle', 'Droid', false, false, 'Droid', false, 'droid', 'droid'], + ['cursor agent', null, 'Cursor', false, false, 'Cursor', false, 'cursor', 'cursor'], + ['cursor ready', 'idle', 'Cursor', false, false, 'Cursor', false, 'cursor', 'cursor'], + [ + 'cursor - action required', + 'permission', + 'Cursor', + false, + false, + 'Cursor', + false, + 'cursor', + 'cursor' + ], + ['cursor position reset', 'idle', null, false, false, null, false, null, null], + ['\u03c0 > session - ~/orca', 'idle', 'Pi', false, false, 'Pi', false, 'pi', 'pi'], + ['\u03c0 ! blocked-session', 'permission', 'Pi', false, false, 'Pi', false, 'pi', 'pi'], + ['\u280b \u03c0 - session - ~/orca', 'working', 'Pi', true, false, 'Pi', true, 'pi', 'pi'], + ['zsh | \u280b Codex', 'working', 'Codex', true, false, 'Codex', true, 'codex', 'codex'], + ['tmux | claude - action required', 'permission', null, false, false, null, false, null, null], + [ + 'ssh host | opencode ready', + 'idle', + 'OpenCode', + false, + false, + 'OpenCode', + false, + 'opencode', + 'opencode' + ] +] + +describe('terminal title classification', () => { + it('covers every corpus title exactly once', () => { + expect(PINNED_CLASSIFICATIONS.map(([title]) => title)).toEqual([ + ...TERMINAL_TITLE_CLASSIFICATION_CORPUS + ]) + }) + + it.each(PINNED_CLASSIFICATIONS)( + 'classifies %j identically', + ( + title, + status, + label, + claude, + gemini, + explicitLabel, + explicitClaude, + titleAgent, + explicitTitleAgent + ) => { + expect(detectAgentStatusFromTitle(title)).toBe(status) + expect(getAgentLabel(title)).toBe(label) + expect(isClaudeAgent(title)).toBe(claude) + expect(isGeminiTerminalTitle(title)).toBe(gemini) + expect(getExplicitAgentLabel(title)).toBe(explicitLabel) + expect(isExplicitClaudeAgent(title)).toBe(explicitClaude) + expect(resolveTerminalTitleAgentType(title)).toBe(titleAgent) + expect(resolveExplicitTerminalTitleAgentType(title)).toBe(explicitTitleAgent) + } + ) + + it('returns the same verdict on the second read of every title', () => { + for (const title of TERMINAL_TITLE_CLASSIFICATION_CORPUS) { + expect(detectAgentStatusFromTitle(title)).toBe(detectAgentStatusFromTitle(title)) + expect(getAgentLabel(title)).toBe(getAgentLabel(title)) + expect(resolveExplicitTerminalTitleAgentType(title)).toBe( + resolveExplicitTerminalTitleAgentType(title) + ) + } + }) +}) diff --git a/src/shared/terminal-title-classification-corpus.ts b/src/shared/terminal-title-classification-corpus.ts new file mode 100644 index 00000000000..e1eb6e833a7 --- /dev/null +++ b/src/shared/terminal-title-classification-corpus.ts @@ -0,0 +1,86 @@ +/** + * Realistic terminal-title corpus for pinning agent classification. + * + * Why a shared const: the corpus is the contract the memoized classifiers must + * reproduce byte-for-byte, so the pinning test and the memo regression test + * read the same titles. + */ +export const TERMINAL_TITLE_CLASSIFICATION_CORPUS: readonly string[] = [ + // Plain shell / directory titles — must classify as nothing. + '', + 'zsh', + 'bash', + 'nwparker@mac: ~/orca', + 'npm run dev', + // Boundary-guard cases from agent-name-token-match.ts's header comment. + 'opencode-blinker', + 'openclaude', + 'openclaude-scratch', + 'claude-scratch', + '~/codex/ready', + 'review-14600-codex', + 'timestamp ready', + 'android build running', + '~/hermes/working', + 'C:\\tools\\codex\\run', + '/usr/local/bin/claude/notes', + 'agy-nightly', + // Windows launcher suffixes. + 'codex.exe', + 'openclaude.cmd', + 'claude.bat working', + 'aider.ps1 ready', + 'copilot.exe - action required', + 'droid.exe', + // Claude Code prefixes and identity frames. + '\u2733', + '\u2733 Claude Code', + '\u2733 ready', + '. building the parser', + '* done', + 'Claude Code', + 'claude - action required', + 'Claude ready', + 'claude agents', + '"/usr/local/bin/claude" agents', + // Leading spinner glyphs (braille + quarter circle). + '\u280b Claude Code', + '\u2809 Codex \u2014 refactoring', + '\u25d0 working', + '\u25d3 Grok', + '\u280b Cursor Agent', + '\u280b Droid', + '\u280b Hermes', + // Gemini glyph vocabulary. + '\u2726 gemini', + '\u23f2 Gemini CLI', + '\u25c7 Gemini CLI', + '\u270b Gemini CLI', + 'gemini', + 'antigravity gemini 3 pro', + 'agy - gemini 2 flash', + // Named agents with status words. + 'codex working', + 'codex ready', + 'copilot waiting', + 'devin thinking', + 'mimo idle', + 'aider running', + 'grok done', + 'opencode ready', + 'hermes ready', + 'droid ready', + // Cursor's closed identity set. + 'cursor agent', + 'cursor ready', + 'cursor - action required', + 'cursor position reset', + // Pi / OMP compatible titles. + '\u03c0 > session - ~/orca', + '\u03c0 ! blocked-session', + '\u280b \u03c0 - session - ~/orca', + // Wrapper/multiplexer prefixes. + 'zsh | \u280b Codex', + 'tmux | claude - action required', + 'ssh host | opencode ready' +] diff --git a/src/shared/terminal-title-classification-memo.test.ts b/src/shared/terminal-title-classification-memo.test.ts new file mode 100644 index 00000000000..e2606a58291 --- /dev/null +++ b/src/shared/terminal-title-classification-memo.test.ts @@ -0,0 +1,98 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import * as AgentNameTokenMatchModule from './agent-name-token-match' +import { getAgentLabel } from './agent-title-identity' +import { detectAgentStatusFromTitle } from './agent-title-status' +import { memoizeTitleClassification } from './terminal-title-classification-memo' +import { resolveExplicitTerminalTitleAgentType } from './terminal-title-agent-type' + +// Why a module mock: `titleHasAgentName` is the leaf regex test every title +// classifier funnels into, so counting its invocations is the direct measure of +// what one store write costs when no title has changed. +vi.mock('./agent-name-token-match', async (importOriginal) => { + const actual = await importOriginal() + return { ...actual, titleHasAgentName: vi.fn(actual.titleHasAgentName) } +}) + +const classifierCalls = vi.mocked(AgentNameTokenMatchModule.titleHasAgentName) + +// Titles a real sidebar holds steady while unrelated agent-status writes churn. +const UNCHANGED_TITLES = [ + 'codex working', + 'opencode-blinker', + 'zsh', + '✳ Claude Code', + 'copilot.exe - action required', + 'gemini', + 'cursor agent' +] +const STORE_WRITES = 50 + +function classifyEveryTitle(): void { + for (const title of UNCHANGED_TITLES) { + getAgentLabel(title) + detectAgentStatusFromTitle(title) + resolveExplicitTerminalTitleAgentType(title) + } +} + +describe('terminal title classification memo', () => { + beforeEach(() => { + classifierCalls.mockClear() + }) + + it('classifies each distinct title once across repeated store writes', () => { + // Warm the caches the way the first render would, then measure steady state. + classifyEveryTitle() + classifierCalls.mockClear() + + for (let write = 0; write < STORE_WRITES; write += 1) { + classifyEveryTitle() + } + + // Unmemoized this is STORE_WRITES x titles x the whole regex ladder — 4,350 + // leaf matches for this fixture. Memoized, an unchanged title costs nothing. + expect(classifierCalls).not.toHaveBeenCalled() + }) + + it('classifies a title once no matter how many readers ask', () => { + const title = 'aider running' + getAgentLabel(title) + const firstReadCalls = classifierCalls.mock.calls.length + expect(firstReadCalls).toBeGreaterThan(0) + + for (let read = 0; read < 20; read += 1) { + getAgentLabel(title) + } + expect(classifierCalls.mock.calls.length).toBe(firstReadCalls) + }) + + it('reclassifies as soon as the title changes', () => { + expect(getAgentLabel('codex ready')).toBe('Codex') + expect(getAgentLabel('grok ready')).toBe('Grok') + expect(detectAgentStatusFromTitle('codex ready')).toBe('idle') + expect(detectAgentStatusFromTitle('codex working')).toBe('working') + }) + + it('caches null and false verdicts, not just truthy ones', () => { + const classify = vi.fn((): string | null => null) + const memoized = memoizeTitleClassification(classify) + expect(memoized('zsh')).toBeNull() + expect(memoized('zsh')).toBeNull() + expect(classify).toHaveBeenCalledTimes(1) + }) + + it('evicts oldest entries instead of growing without bound', () => { + const classify = vi.fn((title: string) => title.length) + const memoized = memoizeTitleClassification(classify) + // Cap is 1024; overflow it and confirm the newest key still hits while the + // oldest was evicted. + for (let index = 0; index < 1030; index += 1) { + memoized(`title-${index}`) + } + const afterFill = classify.mock.calls.length + memoized('title-1029') + expect(classify.mock.calls.length).toBe(afterFill) + memoized('title-0') + expect(classify.mock.calls.length).toBe(afterFill + 1) + }) +}) diff --git a/src/shared/terminal-title-classification-memo.ts b/src/shared/terminal-title-classification-memo.ts new file mode 100644 index 00000000000..c712ca65c96 --- /dev/null +++ b/src/shared/terminal-title-classification-memo.ts @@ -0,0 +1,43 @@ +/** + * Bounded memo for pure `(title: string) => T` terminal-title classifiers. + * + * Why: the sidebar cards and tab strip re-derive agent identity/status from + * every pane title inside zustand selectors and render bodies, so an UNCHANGED + * title was re-tested against every agent-name regex on every store write — + * thousands of classifications per second while the app sat idle. Every + * classifier below depends on nothing but the title string, so the verdict is + * reusable until the title itself changes; a new title is simply a new key, so + * there is no staleness window and no invalidation signal to miss. + */ + +/** + * Cap: comfortably above the live working set (one title per open pane plus + * retained rows) so steady-state hit rate stays ~100%, small enough that the + * map cannot grow with session length. Entries hold a reference to a string the + * store already retains, so the marginal cost is the map entry itself. + */ +const MAX_MEMOIZED_TITLES = 1024 + +export function memoizeTitleClassification( + classify: (title: string) => T +): (title: string) => T { + // Boxed values so `undefined`/`null` verdicts are still cache hits. + const cache = new Map() + return (title: string): T => { + const cached = cache.get(title) + if (cached) { + return cached.value + } + const value = classify(title) + // Insertion-ordered FIFO eviction: a pane's superseded title frames are the + // oldest keys and the least likely to be asked for again. + if (cache.size >= MAX_MEMOIZED_TITLES) { + const oldest = cache.keys().next() + if (!oldest.done) { + cache.delete(oldest.value) + } + } + cache.set(title, { value }) + return value + } +} From 93cb1074b74ea1d6d177f1cf63529949fd7c3cd9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:04:02 -0700 Subject: [PATCH 077/398] perf(renderer): stop the mobile sync key rehashing every dirty file on each keystroke (#18154) `useRuntimeGraphSync` is mounted unconditionally, and its projection layer runs on every store write. Four of those projections did work proportional to the whole slice rather than to what changed: - `buildRuntimeMobileEditorDraftsProjection` FNV-hashed every open dirty draft on every `setEditorDraft`, which Monaco fires per keystroke with no debounce. - `buildRuntimeMobileOpenFilesProjection` and the browser projection rebuilt and re-stringified everything on any `isDirty`/title/url/loading change. - The agent-status sort built an ICU collation per comparison for a string that is only ever compared with `===`. Each now memoizes per entry against the previous build, mirroring the tabs and agent-status projections that already did. The duplicated draft-hash loop in `mobile-session-inputs` is gone; both consumers share one memo. The session-write subscriber also identity-scans SESSION_RELEVANT_FIELDS before allocating its 35-field snapshot and changed-field array. Projections are byte-identical apart from the agent-status sort order, which is never displayed. --- ...ession-write-subscriber-allocation.test.ts | 196 ++++++ .../src/lib/session-write-subscriber.ts | 49 ++ ...time-graph-agent-status-projection.test.ts | 33 +- ...-runtime-graph-projection-hot-path.test.ts | 576 ++++++++++++++++++ .../src/runtime/sync-runtime-graph.ts | 2 + .../agent-status-projection.ts | 5 +- .../sync-runtime-graph/editor-draft-hash.ts | 15 + .../runtime/sync-runtime-graph/graph-state.ts | 10 +- .../mobile-session-inputs.ts | 23 - .../mobile-session-snapshots.ts | 2 +- .../sync-runtime-graph/sync-projections.ts | 244 ++++++-- .../src/runtime/sync-runtime-graph/types.ts | 42 ++ 12 files changed, 1108 insertions(+), 89 deletions(-) create mode 100644 src/renderer/src/lib/session-write-subscriber-allocation.test.ts create mode 100644 src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts create mode 100644 src/renderer/src/runtime/sync-runtime-graph/editor-draft-hash.ts diff --git a/src/renderer/src/lib/session-write-subscriber-allocation.test.ts b/src/renderer/src/lib/session-write-subscriber-allocation.test.ts new file mode 100644 index 00000000000..9681af5a973 --- /dev/null +++ b/src/renderer/src/lib/session-write-subscriber-allocation.test.ts @@ -0,0 +1,196 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AppState } from '@/store' +import { createSessionWriteSubscriber } from './session-write-subscriber' +import { SESSION_RELEVANT_FIELDS } from './workspace-session' +import { buildWorkspaceSessionPatch } from './workspace-session-patch' + +/** + * Why a hand-built store: the subscriber allocates a 35-field snapshot plus a changed-field array + * on every fire that reaches its body, and both are invisible from outside. Driving it through an + * injected store lets `Array.prototype.filter` stand in as the allocation counter — the changed + * list is built 1:1 with the snapshot, in the same block, so one count measures both. + */ +function makeSessionState(overrides: Partial = {}): AppState { + const base: Record = { + workspaceSessionReady: true, + hydrationSucceeded: true, + // Non-session state the subscriber must ignore. + agentStatusByPaneKey: {}, + runtimePaneTitlesByTabId: {} + } + const arrayFields = new Set([ + 'repos', + 'openFiles', + 'browserUrlHistory', + 'workspaceDocHistory' + ]) + const mapFields = new Set(['sshConnectionStates']) + for (const key of SESSION_RELEVANT_FIELDS) { + base[key] = arrayFields.has(key) ? [] : mapFields.has(key) ? new Map() : {} + } + base.activeRepoId = null + base.activeWorkspaceKey = null + base.activeWorktreeId = null + base.activeTabId = null + base.activeWorkspaceExecutionHostId = null + return { ...base, ...overrides } as AppState +} + +function createHarness() { + let state = makeSessionState() + const listeners: ((next: AppState) => void)[] = [] + const persisted: unknown[] = [] + const dispose = createSessionWriteSubscriber({ + store: { + subscribe: (listener) => { + listeners.push(listener) + return () => { + listeners.splice(listeners.indexOf(listener), 1) + } + }, + getState: () => state + }, + persist: (payload) => persisted.push(payload) + }) + return { + dispose, + persisted, + write(mutate?: (previous: AppState) => Partial) { + state = { ...state, ...mutate?.(state) } as AppState + for (const listener of listeners.slice()) { + listener(state) + } + } + } +} + +function countFilterCalls(run: () => T): number { + const original = Array.prototype.filter + let calls = 0 + const spy = vi.spyOn(Array.prototype, 'filter').mockImplementation(function filterCounting( + this: unknown[], + ...args: never[] + ) { + calls += 1 + return (original as (...a: never[]) => unknown[]).apply(this, args) + } as typeof Array.prototype.filter) + try { + run() + return calls + } finally { + spy.mockRestore() + } +} + +beforeEach(() => { + vi.useFakeTimers() +}) +afterEach(() => { + vi.useRealTimers() +}) + +describe('session write subscriber allocation', () => { + it('allocates nothing for store writes that touch no session field', () => { + const harness = createHarness() + try { + // Prime `prev` on the first fire. + harness.write() + vi.advanceTimersByTime(500) + + const writes = 200 + const calls = countFilterCalls(() => { + for (let write = 0; write < writes; write += 1) { + harness.write((previous) => ({ + agentStatusByPaneKey: { ...previous.agentStatusByPaneKey, [`p-${write}`]: {} } as never + })) + } + }) + // Before: one changed-field array (and one 35-field snapshot) per write. + expect(calls).toBe(0) + } finally { + harness.dispose() + } + }) + + it('still allocates and persists when a session field really changes', () => { + const harness = createHarness() + try { + harness.write() + vi.advanceTimersByTime(500) + harness.persisted.length = 0 + + const writes = 20 + const calls = countFilterCalls(() => { + for (let write = 0; write < writes; write += 1) { + harness.write(() => ({ activeTabId: `tab-${write}` })) + } + }) + expect(calls).toBe(writes) + vi.advanceTimersByTime(500) + expect(harness.persisted).toHaveLength(1) + } finally { + harness.dispose() + } + }) + + it('still wakes a deferred write when an unrelated field changes', () => { + let state = makeSessionState() + const listeners: ((next: AppState) => void)[] = [] + const persisted: unknown[] = [] + let gateOpen = false + const dispose = createSessionWriteSubscriber({ + store: { + subscribe: (listener) => { + listeners.push(listener) + return () => {} + }, + getState: () => state + }, + persist: (payload) => persisted.push(payload), + shouldSchedulePersist: () => gateOpen, + subscribeToPersistGateOpen: () => () => {} + }) + const write = (patch: Partial): void => { + state = { ...state, ...patch } as AppState + for (const listener of listeners.slice()) { + listener(state) + } + } + try { + // A real session change lands while the gate is closed, so the write is owed but deferred. + write({ activeTabId: 'tab-1' }) + vi.advanceTimersByTime(500) + expect(persisted).toHaveLength(0) + + // An unrelated write with the gate open must re-arm it even though nothing session-relevant + // moved — the identity-scan fast path must not swallow this wake-up. + gateOpen = true + write({ agentStatusByPaneKey: { a: {} } as never }) + vi.advanceTimersByTime(500) + expect(persisted).toHaveLength(1) + } finally { + dispose() + } + }) + + it('gates on a strict superset of the fields the patch builder reads', () => { + // A gate that misses a projection input persists stale state, so this is a correctness lock, + // not a perf one. Record what the builder actually touches rather than trusting the types. + const read = new Set() + const snapshot = new Proxy(makeSessionState() as unknown as Record, { + get(target, property, receiver) { + if (typeof property === 'string') { + read.add(property) + } + return Reflect.get(target, property, receiver) + } + }) + buildWorkspaceSessionPatch( + snapshot as never, + SESSION_RELEVANT_FIELDS as unknown as Iterable + ) + const gated = new Set(SESSION_RELEVANT_FIELDS) + expect([...read].filter((field) => !gated.has(field))).toEqual([]) + expect(read.size).toBeGreaterThan(0) + }) +}) diff --git a/src/renderer/src/lib/session-write-subscriber.ts b/src/renderer/src/lib/session-write-subscriber.ts index f01b7b25011..e0e366ef6ba 100644 --- a/src/renderer/src/lib/session-write-subscriber.ts +++ b/src/renderer/src/lib/session-write-subscriber.ts @@ -124,6 +124,10 @@ export function createSessionWriteSubscriber({ // reuse the prior identity while a real session change keeps fresh tabs for // the eventual getState() patch build. `null` makes the first fire proceed. let prev: Record | null = null + // Why held separately from `prev`: `prev` stores the *projected* tab maps, so the raw slice + // identity is the only thing the pre-allocation scan below can compare them against. + let prevTabsSource: TabsByWorktree | null = null + let prevUnifiedTabsSource: UnifiedTabsByWorktree | null = null // Why: this set is the only record that a mutation still owes a write — `prev` has already // advanced past it, and change detection is identity-based, so a field dropped from here can // never be re-detected. It is retired only by a flush that reached `persist` (or found nothing @@ -168,10 +172,53 @@ export function createSessionWriteSubscriber({ timer = setTimeout(flushPendingWrite, debounceMs) } + /** + * Identity-only scan over exactly SESSION_RELEVANT_FIELDS, allocating nothing. + * + * Why sound: for the two projected fields an unchanged raw slice is strictly stronger than an + * unchanged projection (the projection is a function of the slice), so a `false` here always + * implies the full comparison below would have found no changed field. A changed raw slice + * falls through to that comparison, where the projection can still collapse it. + */ + const hasSessionFieldIdentityChange = (state: AppState): boolean => { + if (prev === null) { + return true + } + for (const key of SESSION_RELEVANT_FIELDS) { + const unchanged = + key === 'tabsByWorktree' + ? state.tabsByWorktree === prevTabsSource + : key === 'unifiedTabsByWorktree' + ? state.unifiedTabsByWorktree === prevUnifiedTabsSource + : prev[key] === state[key] + if (!unchanged) { + return true + } + } + return false + } + const unsub = store.subscribe((state) => { if (!shouldPersistWorkspaceSession(state)) { return } + // Why: this fires on every store write and almost none of them touch a session field. Scan + // identities first so the common case never allocates the 35-field snapshot or the changed + // list; only a real identity change pays for them. + if (!hasSessionFieldIdentityChange(state)) { + if (pendingChangedFields.size === 0) { + return + } + if (shouldSchedulePersist && !shouldSchedulePersist()) { + return + } + // An unrelated update may wake a deferred write but must never reset an armed debounce. + if (timer !== null) { + return + } + armFlushTimer() + return + } const next: Record = {} for (const key of SESSION_RELEVANT_FIELDS) { const value = state[key] @@ -190,6 +237,8 @@ export function createSessionWriteSubscriber({ return } prev = next + prevTabsSource = state.tabsByWorktree + prevUnifiedTabsSource = state.unifiedTabsByWorktree for (const field of changedFields) { pendingChangedFields.add(field) } diff --git a/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts b/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts index 724b8743dde..3ae6769bb15 100644 --- a/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts +++ b/src/renderer/src/runtime/sync-runtime-graph-agent-status-projection.test.ts @@ -6,14 +6,17 @@ import { resetRuntimeMobileAgentStatusProjectionCacheForTests } from './sync-runtime-graph' -// Reference: the pre-change whole-array serialization, kept verbatim. The bucket -// width is read from the module under test so a drifted constant cannot make this -// reference silently disagree for a reason unrelated to the change. +// Reference: the pre-change whole-array serialization, kept verbatim apart from the sort, which +// is now a code-unit compare — the projection is only ever `===`-compared, never displayed, so it +// needs to be deterministic rather than locale-correct. `localeCompareOrderedEntries` below pins +// that the two orders still serialize the same entries. The bucket width is read from the module +// under test so a drifted constant cannot make this reference silently disagree for a reason +// unrelated to the change. const BUCKET_MS = AGENT_STATUS_SYNC_UPDATED_AT_BUCKET_MS_FOR_TESTS function referenceProjection(map: AppState['agentStatusByPaneKey']): string { return JSON.stringify( Object.entries(map) - .sort(([a], [b]) => a.localeCompare(b)) + .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)) .map(([paneKey, entry]) => ({ paneKey, entryPaneKey: entry.paneKey, @@ -120,4 +123,26 @@ describe('mobile agent-status projection equivalence', () => { }).toEqual({ round, projection: referenceProjection(current) }) } }) + + it('serializes the same entries the localeCompare order did', () => { + resetRuntimeMobileAgentStatusProjectionCacheForTests() + // Keys where locale and code-unit order disagree: 'tab-1:' sorts after 'tab-10:' by code unit + // (':' > '0') and before it by locale, and case/accent handling differs too. + const map: AppState['agentStatusByPaneKey'] = {} + for (const paneKey of ['tab-1:leaf-0', 'tab-10:leaf-0', 'B:leaf-0', 'a:leaf-0', 'á:leaf-0']) { + map[paneKey] = makeEntry(0, { paneKey }) + } + const localeCompareOrderedEntries = JSON.parse( + JSON.stringify( + Object.entries(map) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([paneKey]) => paneKey) + ) + ) as string[] + const projected = ( + JSON.parse(buildRuntimeMobileAgentStatusProjectionForTests(map)) as { paneKey: string }[] + ).map((entry) => entry.paneKey) + expect([...projected].sort()).toEqual([...localeCompareOrderedEntries].sort()) + expect(projected).toHaveLength(localeCompareOrderedEntries.length) + }) }) diff --git a/src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts b/src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts new file mode 100644 index 00000000000..ed1991e60d2 --- /dev/null +++ b/src/renderer/src/runtime/sync-runtime-graph-projection-hot-path.test.ts @@ -0,0 +1,576 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { AppState } from '../store/types' +import { makeAgentStatusEntry, makeState } from './sync-runtime-graph-test-harness' +import type * as EditorDraftHashModule from './sync-runtime-graph/editor-draft-hash' + +// Why the mock: the draft hash is the only per-keystroke cost that scales with file size, so the +// regression this file guards is "how many characters were hashed", not "how long did it take". +// Instrumenting the real function through its own module keeps the counter out of shipped code. +const draftHashCounter = { calls: 0, chars: 0 } +vi.mock('./sync-runtime-graph/editor-draft-hash', async (importOriginal) => { + const actual = await importOriginal() + return { + stableHashString: (value: string): string => { + draftHashCounter.calls += 1 + draftHashCounter.chars += value.length + return actual.stableHashString(value) + } + } +}) + +const { stableHashString } = await import('./sync-runtime-graph/editor-draft-hash') +const { + buildRuntimeMobileAgentStatusProjectionForTests, + getRuntimeMobileSessionSyncKey, + resetRuntimeMobileAgentStatusProjectionCacheForTests, + resetRuntimeMobileSyncProjectionCachesForTests, + runtimeMobileSessionSyncKeysEqual +} = await import('./sync-runtime-graph') +const { + buildRuntimeMobileBrowserProjection, + buildRuntimeMobileEditorDraftsProjection, + buildRuntimeMobileOpenFilesProjection +} = await import('./sync-runtime-graph/sync-projections') +const { AGENT_STATUS_SYNC_UPDATED_AT_BUCKET_MS } = await import('./sync-runtime-graph/graph-state') + +// ── Reference implementations: the pre-change bodies, kept verbatim ──────────────────── + +function referenceEditorDraftsProjection(editorDrafts: AppState['editorDrafts']): string { + return JSON.stringify( + Object.fromEntries( + Object.entries(editorDrafts).map(([fileId, content]) => [fileId, stableHashString(content)]) + ) + ) +} + +function referenceOpenFilesProjection(openFiles: AppState['openFiles']): string { + return JSON.stringify( + openFiles.map((file) => ({ + id: file.id, + filePath: file.filePath, + relativePath: file.relativePath, + worktreeId: file.worktreeId, + language: file.language, + mode: file.mode, + diffSource: file.diffSource, + isDirty: file.isDirty, + isUntitled: file.isUntitled, + deleteUntouchedOnClose: file.deleteUntouchedOnClose, + markdownPreviewSourceFileId: file.markdownPreviewSourceFileId + })) + ) +} + +function referenceBrowserProjection(state: AppState): string { + return JSON.stringify({ + workspacesByWorktree: Object.fromEntries( + Object.entries(state.browserTabsByWorktree ?? {}).map(([worktreeId, workspaces]) => [ + worktreeId, + workspaces.map((workspace) => ({ + id: workspace.id, + activePageId: workspace.activePageId, + title: workspace.title, + url: workspace.url, + loading: workspace.loading, + canGoBack: workspace.canGoBack, + canGoForward: workspace.canGoForward + })) + ]) + ), + pagesByWorkspace: Object.fromEntries( + Object.entries(state.browserPagesByWorkspace ?? {}).map(([workspaceId, pages]) => [ + workspaceId, + pages.map((page) => ({ + id: page.id, + title: page.title, + url: page.url, + loading: page.loading, + canGoBack: page.canGoBack, + canGoForward: page.canGoForward + })) + ]) + ) + }) +} + +/** The pre-change agent-status serialization, including its `localeCompare` sort. */ +function referenceAgentStatusProjection(map: AppState['agentStatusByPaneKey']): string { + return JSON.stringify( + Object.entries(map) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([paneKey, entry]) => ({ + paneKey, + entryPaneKey: entry.paneKey, + state: entry.state, + workingMode: entry.workingMode ?? null, + prompt: entry.prompt, + updatedAtBucket: Math.floor(entry.updatedAt / AGENT_STATUS_SYNC_UPDATED_AT_BUCKET_MS), + stateStartedAt: entry.stateStartedAt, + agentType: entry.agentType ?? null, + terminalTitle: entry.terminalTitle ?? null, + stateHistory: entry.stateHistory.map((history) => ({ + state: history.state, + prompt: history.prompt, + startedAt: history.startedAt, + interrupted: history.interrupted ?? null + })), + toolName: entry.toolName ?? null, + toolInput: entry.toolInput ?? null, + interactivePrompt: entry.interactivePrompt ?? null, + lastAssistantMessage: entry.lastAssistantMessage ?? null, + lastAssistantMessageIsToolOutput: entry.lastAssistantMessageIsToolOutput ?? null, + interrupted: entry.interrupted ?? null + })) + ) +} + +/** Same entries, code-unit ordered — proves the new sort changes order only, never content. */ +function sortedProjectionEntries(projection: string): unknown[] { + return (JSON.parse(projection) as unknown[]).slice().sort((a, b) => { + const left = JSON.stringify(a) + const right = JSON.stringify(b) + return left < right ? -1 : left > right ? 1 : 0 + }) +} + +// ── Fixtures ────────────────────────────────────────────────────────────────────────── + +const DRAFT_FILE_COUNT = 5 +const DRAFT_CHARS_PER_FILE = 40_000 + +function makeDrafts(): Record { + const drafts: Record = {} + for (let index = 0; index < DRAFT_FILE_COUNT; index += 1) { + drafts[`file-${index}`] = 'x'.repeat(DRAFT_CHARS_PER_FILE) + } + return drafts +} + +function makeOpenFile(index: number, overrides: Record = {}): never { + return { + id: `file-${index}`, + filePath: `/repo/src/file-${index}.ts`, + relativePath: `src/file-${index}.ts`, + worktreeId: 'wt-1', + language: 'typescript', + mode: 'edit', + isDirty: false, + isUntitled: false, + ...overrides + } as never +} + +function makeBrowserWorkspace(index: number, overrides: Record = {}): never { + return { + id: `ws-${index}`, + activePageId: `page-${index}`, + title: `tab ${index}`, + url: `https://example.test/${index}`, + loading: false, + canGoBack: false, + canGoForward: false, + ...overrides + } as never +} + +function makeBrowserPage(index: number, overrides: Record = {}): never { + return { + id: `page-${index}`, + title: `page ${index}`, + url: `https://example.test/${index}`, + loading: false, + canGoBack: false, + canGoForward: false, + ...overrides + } as never +} + +/** + * Characters handed back by `JSON.stringify`, which is the allocation these projections dominate. + * Counting bytes rather than calls keeps the comparison fair: the old code made one big call per + * rebuild, the new code makes one small call per changed entry. + */ +function countSerializedChars(run: () => void): number { + const original = JSON.stringify + let chars = 0 + const spy = vi.spyOn(JSON, 'stringify').mockImplementation(((...args: never[]) => { + const serialized = (original as (...a: never[]) => string)(...args) + chars += serialized?.length ?? 0 + return serialized + }) as typeof JSON.stringify) + try { + run() + return chars + } finally { + spy.mockRestore() + } +} + +beforeEach(() => { + draftHashCounter.calls = 0 + draftHashCounter.chars = 0 + resetRuntimeMobileSyncProjectionCachesForTests() + resetRuntimeMobileAgentStatusProjectionCacheForTests() +}) + +describe('editor draft projection on the typing path', () => { + it('hashes only the edited file per keystroke, not every open dirty file', () => { + const typedCharacters = 100 + const totalDraftChars = DRAFT_FILE_COUNT * DRAFT_CHARS_PER_FILE + + // Baseline: the pre-change uncached projection, driven by the same counter. + let drafts = makeDrafts() + referenceEditorDraftsProjection(drafts) + draftHashCounter.calls = 0 + draftHashCounter.chars = 0 + for (let keystroke = 0; keystroke < typedCharacters; keystroke += 1) { + drafts = { ...drafts, 'file-0': `${drafts['file-0']}a` } + referenceEditorDraftsProjection(drafts) + } + const before = { calls: draftHashCounter.calls, chars: draftHashCounter.chars } + + resetRuntimeMobileSyncProjectionCachesForTests() + let memoDrafts = makeDrafts() + buildRuntimeMobileEditorDraftsProjection(memoDrafts) + draftHashCounter.calls = 0 + draftHashCounter.chars = 0 + for (let keystroke = 0; keystroke < typedCharacters; keystroke += 1) { + memoDrafts = { ...memoDrafts, 'file-0': `${memoDrafts['file-0']}a` } + buildRuntimeMobileEditorDraftsProjection(memoDrafts) + } + const after = { calls: draftHashCounter.calls, chars: draftHashCounter.chars } + + // Keystroke k has grown file-0 by k characters, so the exact totals are closed form. + const growth = (typedCharacters * (typedCharacters + 1)) / 2 + // Before: every keystroke rehashes all five drafts. + expect(before).toEqual({ + calls: typedCharacters * DRAFT_FILE_COUNT, + chars: typedCharacters * totalDraftChars + growth + }) + // After: one hash of one draft per keystroke. + expect(after).toEqual({ + calls: typedCharacters, + chars: typedCharacters * DRAFT_CHARS_PER_FILE + growth + }) + expect(before.chars / after.chars).toBeGreaterThan(DRAFT_FILE_COUNT - 0.1) + }) + + it('matches the uncached projection byte for byte across draft shapes', () => { + const shapes: Record[] = [ + {}, + { 'file-a': '' }, + { 'file-a': 'hello' }, + { 'file-a': 'hello', 'file-b': 'world' }, + { 'file-b': 'world', 'file-a': 'hello' }, + { '2': 'numeric-like key', 'file-a': 'hello', '1': 'other' }, + { 'quote"and\\slash': 'body with "quotes" and \\ and \u{1f389}' }, + { 'file-a': 'hello', 'file-b': 'world', 'file-c': 'third' }, + { 'file-a': 'HELLO', 'file-c': 'third' } + ] + for (const [index, shape] of shapes.entries()) { + expect({ index, projection: buildRuntimeMobileEditorDraftsProjection(shape) }).toEqual({ + index, + projection: referenceEditorDraftsProjection(shape) + }) + } + }) +}) + +describe('agent-status projection sort', () => { + it('constructs no ICU collator, and keeps the same entries as the localeCompare order', () => { + // MAX_LIVE_AGENT_STATUSES — the cap a busy session actually reaches. Real pane keys are + // `:`, so model them as unordered hex rather than a sorted + // `tab-` run that would let TimSort skip most comparisons. + let seed = 0x2f6e2b1 + const nextHex = (): string => { + seed = (seed * 1103515245 + 12345) & 0x7fffffff + return seed.toString(16).padStart(8, '0') + } + const paneKeys = Array.from({ length: 500 }, () => `${nextHex()}-${nextHex()}:${nextHex()}`) + const map: AppState['agentStatusByPaneKey'] = {} + for (const [index, paneKey] of paneKeys.entries()) { + map[paneKey] = makeAgentStatusEntry({ paneKey, prompt: `prompt ${index}` }) + } + // One ping replaces one entry and re-spreads the map, so the sort runs in full again. + const pinged = { ...map, [paneKeys[0]]: makeAgentStatusEntry({ paneKey: paneKeys[0] }) } + + const localeCompareSpy = vi.spyOn(String.prototype, 'localeCompare') + let projection = '' + let beforeCalls = 0 + let afterCalls = 0 + try { + referenceAgentStatusProjection(map) + beforeCalls = localeCompareSpy.mock.calls.length + localeCompareSpy.mockClear() + resetRuntimeMobileAgentStatusProjectionCacheForTests() + projection = buildRuntimeMobileAgentStatusProjectionForTests(map) + buildRuntimeMobileAgentStatusProjectionForTests(pinged) + afterCalls = localeCompareSpy.mock.calls.length + } finally { + // `mockRestore` clears the recorded calls, so read the counts first. + localeCompareSpy.mockRestore() + } + // Before: thousands of ICU collator comparisons for a single ping. + expect(beforeCalls).toBeGreaterThan(3000) + expect(afterCalls).toBe(0) + + // The projection is identical up to ordering, and ordering is only ever `===`-compared. + expect(sortedProjectionEntries(projection)).toEqual( + sortedProjectionEntries(referenceAgentStatusProjection(map)) + ) + }) + + it('is deterministic for keys where locale and code-unit order disagree', () => { + const map: AppState['agentStatusByPaneKey'] = {} + for (const paneKey of ['b:leaf', 'A:leaf', 'a:leaf', 'á:leaf', 'B:leaf']) { + map[paneKey] = makeAgentStatusEntry({ paneKey }) + } + resetRuntimeMobileAgentStatusProjectionCacheForTests() + const first = buildRuntimeMobileAgentStatusProjectionForTests(map) + resetRuntimeMobileAgentStatusProjectionCacheForTests() + const second = buildRuntimeMobileAgentStatusProjectionForTests({ ...map }) + expect(second).toBe(first) + expect(sortedProjectionEntries(first)).toEqual( + sortedProjectionEntries(referenceAgentStatusProjection(map)) + ) + }) +}) + +describe('open-files and browser projections', () => { + it('re-serializes only the changed entry per store write', () => { + const files = Array.from({ length: 20 }, (_value, index) => makeOpenFile(index)) + const writes = 50 + const driveWrites = (project: (openFiles: AppState['openFiles']) => string): void => { + let openFiles = files as unknown as AppState['openFiles'] + project(openFiles) + for (let write = 0; write < writes; write += 1) { + const next = [...openFiles] + next[0] = makeOpenFile(0, { isDirty: write % 2 === 0 }) + openFiles = next as unknown as AppState['openFiles'] + project(openFiles) + } + } + + const before = countSerializedChars(() => { + driveWrites(referenceOpenFilesProjection) + }) + resetRuntimeMobileSyncProjectionCachesForTests() + const after = countSerializedChars(() => { + driveWrites(buildRuntimeMobileOpenFilesProjection) + }) + // Only the flipped file re-serializes; the other 19 are reused by identity. + expect(before / after).toBeGreaterThan(14) + }) + + it('re-serializes only the changed browser bucket per store write', () => { + const workspaces = Array.from({ length: 8 }, (_value, index) => makeBrowserWorkspace(index)) + const pagesByWorkspace: Record = {} + for (let index = 0; index < 8; index += 1) { + pagesByWorkspace[`ws-${index}`] = [makeBrowserPage(index)] as never[] + } + const initial = makeState({ + browserTabsByWorktree: { 'wt-1': workspaces, 'wt-2': workspaces } as never, + browserPagesByWorkspace: pagesByWorkspace as never + }) + const writes = 50 + const driveWrites = (project: (state: AppState) => string): void => { + let state = initial + project(state) + for (let write = 0; write < writes; write += 1) { + const nextWorkspaces = [...(state.browserTabsByWorktree['wt-1'] ?? [])] + nextWorkspaces[0] = makeBrowserWorkspace(0, { title: `tab 0 (${write})` }) + state = makeState({ + ...state, + browserTabsByWorktree: { + ...state.browserTabsByWorktree, + 'wt-1': nextWorkspaces + } as never + }) + project(state) + } + } + + const before = countSerializedChars(() => { + driveWrites(referenceBrowserProjection) + }) + resetRuntimeMobileSyncProjectionCachesForTests() + const after = countSerializedChars(() => { + driveWrites(buildRuntimeMobileBrowserProjection) + }) + // Only the 'wt-1' bucket re-serializes; 'wt-2' and every page bucket are reused. + expect(before / after).toBeGreaterThan(2.5) + }) + + it('matches the uncached projections byte for byte across shapes', () => { + const openFileShapes: AppState['openFiles'][] = [ + [] as unknown as AppState['openFiles'], + [makeOpenFile(0)] as unknown as AppState['openFiles'], + [makeOpenFile(0, { isDirty: true })] as unknown as AppState['openFiles'], + [ + makeOpenFile(0, { isDirty: true, diffSource: 'working' }), + makeOpenFile(1, { mode: 'diff', markdownPreviewSourceFileId: 'file-0' }), + makeOpenFile(2, { isUntitled: true, deleteUntouchedOnClose: true, language: undefined }) + ] as unknown as AppState['openFiles'] + ] + for (const [index, shape] of openFileShapes.entries()) { + expect({ index, projection: buildRuntimeMobileOpenFilesProjection(shape) }).toEqual({ + index, + projection: referenceOpenFilesProjection(shape) + }) + } + + const browserShapes: AppState[] = [ + makeState({}), + makeState({ browserTabsByWorktree: { 'wt-1': [makeBrowserWorkspace(0)] } as never }), + makeState({ + browserTabsByWorktree: { + 'wt-1': [makeBrowserWorkspace(0, { title: undefined, loading: true })], + '3': [makeBrowserWorkspace(1)] + } as never, + browserPagesByWorkspace: { + 'ws-0': [makeBrowserPage(0), makeBrowserPage(1, { canGoBack: true })], + 'ws-1': [] + } as never + }), + makeState({ + browserTabsByWorktree: {} as never, + browserPagesByWorkspace: { 'ws-9': [makeBrowserPage(9, { url: 'a"b\\c' })] } as never + }) + ] + for (const [index, shape] of browserShapes.entries()) { + expect({ index, projection: buildRuntimeMobileBrowserProjection(shape) }).toEqual({ + index, + projection: referenceBrowserProjection(shape) + }) + } + }) +}) + +describe('sync key transitions', () => { + it('fires on exactly the transitions the uncached projections would have fired on', () => { + const drafts = { 'file-0': 'aaa', 'file-1': 'bbb' } + const files = [makeOpenFile(0), makeOpenFile(1)] as unknown as AppState['openFiles'] + const workspaces = [makeBrowserWorkspace(0)] as never + const status = { + 'tab-0:leaf-0': makeAgentStatusEntry({ paneKey: 'tab-0:leaf-0' }) + } as AppState['agentStatusByPaneKey'] + + const base = makeState({ + editorDrafts: drafts, + openFiles: files, + browserTabsByWorktree: { 'wt-1': workspaces } as never, + browserPagesByWorkspace: { 'ws-0': [makeBrowserPage(0)] } as never, + agentStatusByPaneKey: status + }) + + // Each step returns the next state; the flag is whether a mobile-visible input really moved. + const steps: { name: string; next: (from: AppState) => AppState }[] = [ + { name: 'no-op re-spread', next: (from) => makeState({ ...from }) }, + { + name: 'keystroke in one draft', + next: (from) => + makeState({ ...from, editorDrafts: { ...from.editorDrafts, 'file-0': 'aaab' } }) + }, + { + name: 'draft reverted to the same text', + next: (from) => + makeState({ ...from, editorDrafts: { ...from.editorDrafts, 'file-0': 'aaab' } }) + }, + { + name: 'draft removed', + next: (from) => makeState({ ...from, editorDrafts: { 'file-1': 'bbb' } }) + }, + { + name: 'isDirty flip', + next: (from) => + makeState({ + ...from, + openFiles: [makeOpenFile(0, { isDirty: true }), from.openFiles[1]] as never + }) + }, + { + name: 'open-files re-spread with identical content', + next: (from) => makeState({ ...from, openFiles: [...from.openFiles] as never }) + }, + { + name: 'browser title tick', + next: (from) => + makeState({ + ...from, + browserTabsByWorktree: { + 'wt-1': [makeBrowserWorkspace(0, { title: 'new title' })] + } as never + }) + }, + { + name: 'browser page loading flip', + next: (from) => + makeState({ + ...from, + browserPagesByWorkspace: { + 'ws-0': [makeBrowserPage(0, { loading: true })] + } as never + }) + }, + { + name: 'agent-status ping with an unchanged payload', + next: (from) => + makeState({ + ...from, + agentStatusByPaneKey: { + 'tab-0:leaf-0': makeAgentStatusEntry({ paneKey: 'tab-0:leaf-0' }) + } as never + }) + }, + { + name: 'agent-status prompt change', + next: (from) => + makeState({ + ...from, + agentStatusByPaneKey: { + 'tab-0:leaf-0': makeAgentStatusEntry({ paneKey: 'tab-0:leaf-0', prompt: 'new' }) + } as never + }) + }, + { + name: 'second agent added', + next: (from) => + makeState({ + ...from, + agentStatusByPaneKey: { + ...from.agentStatusByPaneKey, + 'tab-1:leaf-0': makeAgentStatusEntry({ paneKey: 'tab-1:leaf-0' }) + } as never + }) + } + ] + + const referenceTuple = (state: AppState): string[] => [ + referenceEditorDraftsProjection(state.editorDrafts), + referenceOpenFilesProjection(state.openFiles), + referenceBrowserProjection(state), + referenceAgentStatusProjection(state.agentStatusByPaneKey ?? {}) + ] + + resetRuntimeMobileSyncProjectionCachesForTests() + resetRuntimeMobileAgentStatusProjectionCacheForTests() + let previousState = base + let previousKey = getRuntimeMobileSessionSyncKey(base, undefined, undefined, false) + let previousReference = referenceTuple(base) + + for (const step of steps) { + const state = step.next(previousState) + const key = getRuntimeMobileSessionSyncKey(state, previousState, previousKey, false) + const reference = referenceTuple(state) + const referenceChanged = reference.some((part, index) => part !== previousReference[index]) + const keyChanged = !runtimeMobileSessionSyncKeysEqual(key, previousKey) + expect({ step: step.name, changed: keyChanged }).toEqual({ + step: step.name, + changed: referenceChanged + }) + previousState = state + previousKey = key + previousReference = reference + } + }) +}) diff --git a/src/renderer/src/runtime/sync-runtime-graph.ts b/src/renderer/src/runtime/sync-runtime-graph.ts index dd1bdb58d5f..b836a037134 100644 --- a/src/renderer/src/runtime/sync-runtime-graph.ts +++ b/src/renderer/src/runtime/sync-runtime-graph.ts @@ -21,6 +21,7 @@ import { runtimeMobileSessionSyncKeysEqual } from './sync-runtime-graph/sync-key' import { buildMobileSessionTabSnapshots } from './sync-runtime-graph/mobile-session-snapshots' +import { resetRuntimeMobileSyncProjectionCachesForTests } from './sync-runtime-graph/sync-projections' import type { RegisteredTerminalTab } from './sync-runtime-graph/types' export type { RegisteredTerminalTab, RuntimeMobileSessionSyncKey } from './sync-runtime-graph/types' @@ -28,6 +29,7 @@ export { AGENT_STATUS_SYNC_UPDATED_AT_BUCKET_MS_FOR_TESTS, buildRuntimeMobileAgentStatusProjectionForTests, resetRuntimeMobileAgentStatusProjectionCacheForTests, + resetRuntimeMobileSyncProjectionCachesForTests, canSkipRuntimeMobileSessionSyncKeyBuild, getRuntimeMobileSessionSyncKey, runtimeMobileSessionSyncKeysEqual, diff --git a/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts b/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts index f11e58a388f..5e14bd4a8ae 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/agent-status-projection.ts @@ -43,8 +43,11 @@ export function buildRuntimeMobileAgentStatusProjection( // A status ping replaces one entry and re-spreads the map; reuse every other entry. const entries = new Map() const parts: string[] = [] + // Code-unit order, not `localeCompare`: this projection is only ever compared with `===`, so it + // must be deterministic, not locale-correct — and an ICU collator per comparison is ~4.5k calls + // per ping at the 500-entry cap. for (const [paneKey, entry] of Object.entries(agentStatusByPaneKey).sort(([a], [b]) => - a.localeCompare(b) + a < b ? -1 : a > b ? 1 : 0 )) { const previous = cached?.entries.get(paneKey) const entryCache = diff --git a/src/renderer/src/runtime/sync-runtime-graph/editor-draft-hash.ts b/src/renderer/src/runtime/sync-runtime-graph/editor-draft-hash.ts new file mode 100644 index 00000000000..a38ebcf9f35 --- /dev/null +++ b/src/renderer/src/runtime/sync-runtime-graph/editor-draft-hash.ts @@ -0,0 +1,15 @@ +/** + * FNV-1a stamp for one editor draft's text. + * + * Why its own module: it is the single hash shared by the mobile sync key and the mobile session + * snapshot, and its cost is proportional to the draft's length — so callers must memoize it per + * file rather than re-running it whenever the drafts record is re-spread. + */ +export function stableHashString(value: string): string { + let hash = 2166136261 + for (let i = 0; i < value.length; i += 1) { + hash ^= value.charCodeAt(i) + hash = Math.imul(hash, 16777619) + } + return `draft:${value.length}:${(hash >>> 0).toString(16)}` +} diff --git a/src/renderer/src/runtime/sync-runtime-graph/graph-state.ts b/src/renderer/src/runtime/sync-runtime-graph/graph-state.ts index 8aa0d87064a..4869479bfaa 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/graph-state.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/graph-state.ts @@ -7,8 +7,12 @@ import type { } from '../../../../shared/runtime-types' import type { AgentStatusProjectionCache, + BrowserPagesProjectionCache, + BrowserWorkspacesProjectionCache, + EditorDraftHashCache, MobileSessionWorktreeInputs, OpenFileIndexes, + OpenFilesProjectionCache, RegisteredTerminalTab, TabsProjectionCache } from './types' @@ -54,10 +58,12 @@ export const graphState = { publishedMobileSessionSnapshotByWorktree: new Map(), cachedTabsProjection: null as TabsProjectionCache | null, cachedAgentStatusProjection: null as AgentStatusProjectionCache | null, + cachedOpenFilesProjection: null as OpenFilesProjectionCache | null, + cachedBrowserWorkspacesProjection: null as BrowserWorkspacesProjectionCache | null, + cachedBrowserPagesProjection: null as BrowserPagesProjectionCache | null, cachedOpenFileIndexesSource: null as AppState['openFiles'] | null, cachedOpenFileIndexes: null as OpenFileIndexes | null, - cachedEditorDraftsSource: null as AppState['editorDrafts'] | null, - cachedEditorDraftVersionByFileId: null as Map | null, + cachedEditorDraftHashes: null as EditorDraftHashCache | null, cachedMobileTerminalThemeSettings: null as AppState['settings'] | null, cachedMobileTerminalThemeSystemPrefersDark: null as boolean | null, cachedMobileTerminalTheme: undefined as RuntimeMobileTerminalTheme | undefined, diff --git a/src/renderer/src/runtime/sync-runtime-graph/mobile-session-inputs.ts b/src/renderer/src/runtime/sync-runtime-graph/mobile-session-inputs.ts index 0d6157fe75d..d692a5b83d3 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/mobile-session-inputs.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/mobile-session-inputs.ts @@ -51,29 +51,6 @@ export function getOpenFileIndexes(openFiles: AppState['openFiles']): OpenFileIn return graphState.cachedOpenFileIndexes } -export function getEditorDraftVersionByFileId( - editorDrafts: AppState['editorDrafts'] -): Map { - if ( - graphState.cachedEditorDraftsSource === editorDrafts && - graphState.cachedEditorDraftVersionByFileId - ) { - return graphState.cachedEditorDraftVersionByFileId - } - const versions = new Map() - for (const [fileId, content] of Object.entries(editorDrafts)) { - let hash = 2166136261 - for (let index = 0; index < content.length; index += 1) { - hash ^= content.charCodeAt(index) - hash = Math.imul(hash, 16777619) - } - versions.set(fileId, `draft:${content.length}:${(hash >>> 0).toString(16)}`) - } - graphState.cachedEditorDraftsSource = editorDrafts - graphState.cachedEditorDraftVersionByFileId = versions - return versions -} - export function buildMobileSessionAgentStatusByWorktree( agentStatusByPaneKey: AppState['agentStatusByPaneKey'], tabsByWorktree: AppState['tabsByWorktree'] diff --git a/src/renderer/src/runtime/sync-runtime-graph/mobile-session-snapshots.ts b/src/renderer/src/runtime/sync-runtime-graph/mobile-session-snapshots.ts index f6ba7f49cfb..7dc106a6cd4 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/mobile-session-snapshots.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/mobile-session-snapshots.ts @@ -14,9 +14,9 @@ import { import { buildMobileSessionAgentStatusByWorktree, buildMobileSessionWorktreeInputs, - getEditorDraftVersionByFileId, getOpenFileIndexes } from './mobile-session-inputs' +import { getEditorDraftVersionByFileId } from './sync-projections' import { getMobileTerminalTheme } from './mobile-terminal-theme' import { isMobilePublishableBrowserWorkspace, diff --git a/src/renderer/src/runtime/sync-runtime-graph/sync-projections.ts b/src/renderer/src/runtime/sync-runtime-graph/sync-projections.ts index 0627bccee06..44a1606b056 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/sync-projections.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/sync-projections.ts @@ -1,11 +1,19 @@ import type { AppState } from '@/store/types' import { resolveTerminalTabTitle } from '../../../../shared/tab-title-resolution' +import { stableHashString } from './editor-draft-hash' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import { EMPTY_BROWSER_PAGES_BY_WORKSPACE, EMPTY_BROWSER_TABS_BY_WORKTREE, graphState } from './graph-state' +import type { + BrowserPagesProjectionCacheEntry, + BrowserWorkspacesProjectionCacheEntry, + EditorDraftHashCache, + EditorDraftHashCacheEntry, + OpenFilesProjectionCacheEntry +} from './types' export function getBrowserTabsByWorktree(state: AppState): AppState['browserTabsByWorktree'] { // Some callers/tests build partial pre-browser states; treat missing slices as empty. @@ -76,72 +84,192 @@ export function buildRuntimeMobileTabsProjection( } export function buildRuntimeMobileOpenFilesProjection(openFiles: AppState['openFiles']): string { - return JSON.stringify( - openFiles.map((file) => ({ - id: file.id, - filePath: file.filePath, - relativePath: file.relativePath, - worktreeId: file.worktreeId, - language: file.language, - mode: file.mode, - diffSource: file.diffSource, - isDirty: file.isDirty, - isUntitled: file.isUntitled, - deleteUntouchedOnClose: file.deleteUntouchedOnClose, - markdownPreviewSourceFileId: file.markdownPreviewSourceFileId - })) - ) + const cached = graphState.cachedOpenFilesProjection + if (cached?.source === openFiles) { + return cached.projection + } + + // An isDirty flip replaces one file and re-spreads the array; reuse every other file. + const previousEntries = cached?.entries + const entries = new Map() + const parts: string[] = [] + for (const file of openFiles) { + const previous = previousEntries?.get(file.id) + const entry = + previous?.file === file + ? previous + : { + file, + projection: JSON.stringify({ + id: file.id, + filePath: file.filePath, + relativePath: file.relativePath, + worktreeId: file.worktreeId, + language: file.language, + mode: file.mode, + diffSource: file.diffSource, + isDirty: file.isDirty, + isUntitled: file.isUntitled, + deleteUntouchedOnClose: file.deleteUntouchedOnClose, + markdownPreviewSourceFileId: file.markdownPreviewSourceFileId + }) + } + entries.set(file.id, entry) + parts.push(entry.projection) + } + const projection = `[${parts.join(',')}]` + graphState.cachedOpenFilesProjection = { source: openFiles, entries, projection } + return projection +} + +function buildBrowserWorkspacesProjection( + browserTabsByWorktree: AppState['browserTabsByWorktree'] +): string { + const cached = graphState.cachedBrowserWorkspacesProjection + if (cached?.source === browserTabsByWorktree) { + return cached.projection + } + + const previousEntries = cached?.entries + const entries = new Map() + const parts: string[] = [] + for (const [worktreeId, workspaces] of Object.entries(browserTabsByWorktree)) { + const previous = previousEntries?.get(worktreeId) + const entry = + previous?.workspaces === workspaces + ? previous + : { + workspaces, + keyJson: previous?.keyJson ?? JSON.stringify(worktreeId), + projection: JSON.stringify( + workspaces.map((workspace) => ({ + id: workspace.id, + activePageId: workspace.activePageId, + title: workspace.title, + url: workspace.url, + loading: workspace.loading, + canGoBack: workspace.canGoBack, + canGoForward: workspace.canGoForward + })) + ) + } + entries.set(worktreeId, entry) + parts.push(`${entry.keyJson}:${entry.projection}`) + } + const projection = `{${parts.join(',')}}` + graphState.cachedBrowserWorkspacesProjection = { + source: browserTabsByWorktree, + entries, + projection + } + return projection +} + +function buildBrowserPagesProjection( + browserPagesByWorkspace: AppState['browserPagesByWorkspace'] +): string { + const cached = graphState.cachedBrowserPagesProjection + if (cached?.source === browserPagesByWorkspace) { + return cached.projection + } + + const previousEntries = cached?.entries + const entries = new Map() + const parts: string[] = [] + for (const [workspaceId, pages] of Object.entries(browserPagesByWorkspace)) { + const previous = previousEntries?.get(workspaceId) + const entry = + previous?.pages === pages + ? previous + : { + pages, + keyJson: previous?.keyJson ?? JSON.stringify(workspaceId), + projection: JSON.stringify( + pages.map((page) => ({ + id: page.id, + title: page.title, + url: page.url, + loading: page.loading, + canGoBack: page.canGoBack, + canGoForward: page.canGoForward + })) + ) + } + entries.set(workspaceId, entry) + parts.push(`${entry.keyJson}:${entry.projection}`) + } + const projection = `{${parts.join(',')}}` + graphState.cachedBrowserPagesProjection = { + source: browserPagesByWorkspace, + entries, + projection + } + return projection } export function buildRuntimeMobileBrowserProjection(state: AppState): string { - const browserTabsByWorktree = getBrowserTabsByWorktree(state) - const browserPagesByWorkspace = getBrowserPagesByWorkspace(state) - return JSON.stringify({ - workspacesByWorktree: Object.fromEntries( - Object.entries(browserTabsByWorktree).map(([worktreeId, workspaces]) => [ - worktreeId, - workspaces.map((workspace) => ({ - id: workspace.id, - activePageId: workspace.activePageId, - title: workspace.title, - url: workspace.url, - loading: workspace.loading, - canGoBack: workspace.canGoBack, - canGoForward: workspace.canGoForward - })) - ]) - ), - pagesByWorkspace: Object.fromEntries( - Object.entries(browserPagesByWorkspace).map(([workspaceId, pages]) => [ - workspaceId, - pages.map((page) => ({ - id: page.id, - title: page.title, - url: page.url, - loading: page.loading, - canGoBack: page.canGoBack, - canGoForward: page.canGoForward - })) - ]) - ) - }) + // A title/url/loading tick replaces one worktree or workspace bucket; reuse the rest. + return `{"workspacesByWorktree":${buildBrowserWorkspacesProjection( + getBrowserTabsByWorktree(state) + )},"pagesByWorkspace":${buildBrowserPagesProjection(getBrowserPagesByWorkspace(state))}}` +} + +/** + * Why memoized per file id: `setEditorDraft` fires on every Monaco keystroke and re-spreads + * `editorDrafts`, so an unmemoized rebuild re-hashed every open dirty file's full text on the + * input path. Only the typed file's draft string changes identity, so only it needs rehashing. + */ +function getEditorDraftHashCache(editorDrafts: AppState['editorDrafts']): EditorDraftHashCache { + const cached = graphState.cachedEditorDraftHashes + if (cached?.source === editorDrafts) { + return cached + } + + const previousEntries = cached?.entries + const entries = new Map() + const hashByFileId = new Map() + const parts: string[] = [] + for (const [fileId, content] of Object.entries(editorDrafts)) { + const previous = previousEntries?.get(fileId) + let entry: EditorDraftHashCacheEntry + if (previous?.content === content) { + entry = previous + } else { + const fileIdJson = previous?.fileIdJson ?? JSON.stringify(fileId) + const hash = stableHashString(content) + entry = { content, hash, fileIdJson, projection: `${fileIdJson}:${JSON.stringify(hash)}` } + } + entries.set(fileId, entry) + hashByFileId.set(fileId, entry.hash) + parts.push(entry.projection) + } + const next: EditorDraftHashCache = { + source: editorDrafts, + entries, + hashByFileId, + projection: `{${parts.join(',')}}` + } + graphState.cachedEditorDraftHashes = next + return next } export function buildRuntimeMobileEditorDraftsProjection( editorDrafts: AppState['editorDrafts'] ): string { - return JSON.stringify( - Object.fromEntries( - Object.entries(editorDrafts).map(([fileId, content]) => [fileId, stableHashString(content)]) - ) - ) + return getEditorDraftHashCache(editorDrafts).projection } -export function stableHashString(value: string): string { - let hash = 2166136261 - for (let i = 0; i < value.length; i += 1) { - hash ^= value.charCodeAt(i) - hash = Math.imul(hash, 16777619) - } - return `draft:${value.length}:${(hash >>> 0).toString(16)}` +/** Per-file draft version stamps for the mobile session snapshot; shares the keystroke memo. */ +export function getEditorDraftVersionByFileId( + editorDrafts: AppState['editorDrafts'] +): ReadonlyMap { + return getEditorDraftHashCache(editorDrafts).hashByFileId +} + +export function resetRuntimeMobileSyncProjectionCachesForTests(): void { + graphState.cachedTabsProjection = null + graphState.cachedOpenFilesProjection = null + graphState.cachedBrowserWorkspacesProjection = null + graphState.cachedBrowserPagesProjection = null + graphState.cachedEditorDraftHashes = null } diff --git a/src/renderer/src/runtime/sync-runtime-graph/types.ts b/src/renderer/src/runtime/sync-runtime-graph/types.ts index b51b8cd8afa..448a3652a5d 100644 --- a/src/renderer/src/runtime/sync-runtime-graph/types.ts +++ b/src/renderer/src/runtime/sync-runtime-graph/types.ts @@ -60,6 +60,48 @@ export type TabsProjectionCache = { entries: Map projection: string } +export type OpenFilesProjectionCacheEntry = { + file: AppState['openFiles'][number] + projection: string +} +export type OpenFilesProjectionCache = { + source: AppState['openFiles'] + entries: Map + projection: string +} +export type BrowserWorkspacesProjectionCacheEntry = { + workspaces: NonNullable + keyJson: string + projection: string +} +export type BrowserWorkspacesProjectionCache = { + source: AppState['browserTabsByWorktree'] + entries: Map + projection: string +} +export type BrowserPagesProjectionCacheEntry = { + pages: NonNullable + keyJson: string + projection: string +} +export type BrowserPagesProjectionCache = { + source: AppState['browserPagesByWorkspace'] + entries: Map + projection: string +} +/** One dirty file's FNV draft stamp plus its pre-serialized projection fragment. */ +export type EditorDraftHashCacheEntry = { + content: string + hash: string + fileIdJson: string + projection: string +} +export type EditorDraftHashCache = { + source: AppState['editorDrafts'] + entries: Map + hashByFileId: Map + projection: string +} export type AgentStatusProjectionCacheEntry = { entry: AppState['agentStatusByPaneKey'][string] projection: string From 21a706a9323a9b527b1e5bdff52262df2226548f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:04:06 -0700 Subject: [PATCH 078/398] perf(terminal): stop shipping every agent spinner title frame to the renderer (#18155) Main re-asserts a working OSC title per pane every 80ms (12.5/sec) while an agent works, and every frame became its own pty:sideEffect IPC message. Both renderer store writes already discard those frames via isDecorativeAgentTitleFrameChange, and paired remote clients already never see them (RuntimeClientEventBus's per-listener title gate). Only the local desktop renderer was still paying for them. Apply the same decorative gate main already computes for the mobile fan-out one hop earlier, keeping a 500ms heartbeat so the renderer's 1500ms hook-done quiet window still sees a working title and can cancel a Pi/OMP milestone 'done'. --- .../decorative-title-fact-emission.test.ts | 49 ++++++++ .../runtime/decorative-title-fact-emission.ts | 38 ++++++ ...e-get-unpersisted-tracked-title-for-pty.ts | 34 ++++-- .../decorative-title-fact-throttle.spec.ts | 110 ++++++++++++++++++ .../terminal-side-effect-facts.spec.ts | 102 +++++++++------- src/main/runtime/orca-runtime.test.ts | 1 + .../runtime/runtime-terminal-state-records.ts | 2 + 7 files changed, 284 insertions(+), 52 deletions(-) create mode 100644 src/main/runtime/decorative-title-fact-emission.test.ts create mode 100644 src/main/runtime/decorative-title-fact-emission.ts create mode 100644 src/main/runtime/orca-runtime-tests/decorative-title-fact-throttle.spec.ts diff --git a/src/main/runtime/decorative-title-fact-emission.test.ts b/src/main/runtime/decorative-title-fact-emission.test.ts new file mode 100644 index 00000000000..609f2e06084 --- /dev/null +++ b/src/main/runtime/decorative-title-fact-emission.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from 'vitest' +import { + DECORATIVE_TITLE_FACT_HEARTBEAT_MS, + shouldEmitTitleFactForFrame +} from './decorative-title-fact-emission' + +const base = { + decorativeOnly: true, + staleWorkingTitleClear: false, + lastEmittedAtMs: 1_000, + nowMs: 1_000 +} + +describe('shouldEmitTitleFactForFrame', () => { + it('always emits a frame that is not a decorative repeat', () => { + expect(shouldEmitTitleFactForFrame({ ...base, decorativeOnly: false })).toBe(true) + }) + + it('emits the first frame of a pane', () => { + expect(shouldEmitTitleFactForFrame({ ...base, lastEmittedAtMs: null })).toBe(true) + }) + + it('suppresses a decorative repeat inside the heartbeat window', () => { + expect( + shouldEmitTitleFactForFrame({ ...base, nowMs: 1_000 + DECORATIVE_TITLE_FACT_HEARTBEAT_MS - 1 }) + ).toBe(false) + }) + + it('lets a decorative repeat through once the heartbeat window elapses', () => { + expect( + shouldEmitTitleFactForFrame({ ...base, nowMs: 1_000 + DECORATIVE_TITLE_FACT_HEARTBEAT_MS }) + ).toBe(true) + }) + + it('never throttles a timer-synthesized stale-working clear', () => { + // Why: it carries a staleWorkingTitleClear flag no earlier repeat can stand in for. + expect(shouldEmitTitleFactForFrame({ ...base, staleWorkingTitleClear: true })).toBe(true) + }) + + it('emits after a backwards clock step instead of parking until it catches up', () => { + expect(shouldEmitTitleFactForFrame({ ...base, nowMs: 900 })).toBe(true) + }) + + it('keeps at least three frames inside the renderer hook-done quiet window', () => { + // Why: observeTitle's arriving working title is what cancels a Pi/OMP milestone `done` + // scheduled with HOOK_DONE_QUIET_MS = 1500. Losing that would mint a false completion. + expect(DECORATIVE_TITLE_FACT_HEARTBEAT_MS * 3).toBeLessThanOrEqual(1_500) + }) +}) diff --git a/src/main/runtime/decorative-title-fact-emission.ts b/src/main/runtime/decorative-title-fact-emission.ts new file mode 100644 index 00000000000..d8248dc12c5 --- /dev/null +++ b/src/main/runtime/decorative-title-fact-emission.ts @@ -0,0 +1,38 @@ +/** + * Why: an agent spinner re-emits a semantically identical OSC title ~12.5x/sec (Orca's own + * synthetic frame timer, Pi/OMP, Claude Code, Grok), and main ships every frame to the renderer + * as its own `pty:sideEffect` message. Both renderer store writes already discard those frames + * via `isDecorativeAgentTitleFrameChange`, so the message is pure cross-process cost. + * + * Why a heartbeat and not a hard drop: `agentCompletionCoordinator.observeTitle` treats an + * arriving *working* title as "still working" and cancels a scheduled hook-`done` completion + * inside `HOOK_DONE_QUIET_MS` (1500ms). That is exactly how a Pi/OMP milestone `done` emitted + * mid-turn is stopped from minting a completion notification, and the frames that carry it are + * decorative repeats. 500ms keeps 3 frames inside that window. + */ +export const DECORATIVE_TITLE_FACT_HEARTBEAT_MS = 500 + +export type DecorativeTitleFactEmissionInput = { + /** The frame's decorative gate key matches the previous frame's. */ + decorativeOnly: boolean + /** Timer-synthesized stale-working clear — carries a flag no repeat can stand in for. */ + staleWorkingTitleClear: boolean + lastEmittedAtMs: number | null + nowMs: number +} + +export function shouldEmitTitleFactForFrame({ + decorativeOnly, + staleWorkingTitleClear, + lastEmittedAtMs, + nowMs +}: DecorativeTitleFactEmissionInput): boolean { + if (!decorativeOnly || staleWorkingTitleClear) { + return true + } + if (lastEmittedAtMs === null) { + return true + } + // A backwards clock step must not park the heartbeat until it catches up. + return nowMs < lastEmittedAtMs || nowMs - lastEmittedAtMs >= DECORATIVE_TITLE_FACT_HEARTBEAT_MS +} diff --git a/src/main/runtime/orca-runtime-get-unpersisted-tracked-title-for-pty.ts b/src/main/runtime/orca-runtime-get-unpersisted-tracked-title-for-pty.ts index 38bae372176..3ece144bc24 100644 --- a/src/main/runtime/orca-runtime-get-unpersisted-tracked-title-for-pty.ts +++ b/src/main/runtime/orca-runtime-get-unpersisted-tracked-title-for-pty.ts @@ -1,6 +1,7 @@ // @ts-nocheck -- mechanically split from OrcaRuntimeService; behavior is covered by AST equivalence and characterization tests. import { OrcaRuntimeWithEmitDaemonPtyTransientFact } from './orca-runtime-emit-daemon-pty-transient-fact' import { getDecorativeAgentTitleSignature } from '../../shared/agent-decorative-title-signature' +import { shouldEmitTitleFactForFrame } from './decorative-title-fact-emission' import type { RuntimePtyTitleTrackerEntry } from './runtime-terminal-state-records' import { createTerminalTitleTracker } from '../../shared/terminal-output-side-effects' import { detectAgentStatusFromTitle } from '../../shared/agent-detection' @@ -64,20 +65,36 @@ export class OrcaRuntimeWithGetUnpersistedTrackedTitleForPty extends OrcaRuntime const tracker = createTerminalTitleTracker( { onTitle: (normalizedTitle, rawTitle, meta) => { - this.recordTerminalSideEffectFact(ptyId, { - kind: 'title', - normalizedTitle, - rawTitle, - ...(meta?.staleWorkingTitleClear ? { staleWorkingTitleClear: true } : {}) - }) - const changed = this.applyTrackedPtyTitle(ptyId, rawTitle, normalizedTitle, meta) - const identityOnlyTitle = this.isLiveCursorNativeTitle(rawTitle, meta) const live = this.ptyTitleTrackersByPtyId.get(ptyId) const gateKey = this.makeDecorativeTitleGateKey(rawTitle, normalizedTitle) const decorativeOnly = live?.lastMobileTitleGateKey === gateKey if (live) { live.lastMobileTitleGateKey = gateKey } + // Why: the same gate the mobile fan-out below already uses, applied one hop earlier — + // a spinner frame the renderer store discards should not cost a pty:sideEffect message + // at all. See decorative-title-fact-emission.ts for why repeats still heartbeat. + const nowMs = Date.now() + if ( + shouldEmitTitleFactForFrame({ + decorativeOnly, + staleWorkingTitleClear: meta?.staleWorkingTitleClear === true, + lastEmittedAtMs: live?.lastTitleFactAtMs ?? null, + nowMs + }) + ) { + if (live) { + live.lastTitleFactAtMs = nowMs + } + this.recordTerminalSideEffectFact(ptyId, { + kind: 'title', + normalizedTitle, + rawTitle, + ...(meta?.staleWorkingTitleClear ? { staleWorkingTitleClear: true } : {}) + }) + } + const changed = this.applyTrackedPtyTitle(ptyId, rawTitle, normalizedTitle, meta) + const identityOnlyTitle = this.isLiveCursorNativeTitle(rawTitle, meta) const tracksReplicatedStatus = live?.applyingChunk === true && this.mobileSessionTabListeners.size > 0 const titleStatus = tracksReplicatedStatus ? detectAgentStatusFromTitle(rawTitle) : null @@ -151,6 +168,7 @@ export class OrcaRuntimeWithGetUnpersistedTrackedTitleForPty extends OrcaRuntime tracker, applyingChunk: false, lastMobileTitleGateKey: null, + lastTitleFactAtMs: null, chunkTouchedSessionTabs: false, pendingFacts: [], // Why: command-code facts exist only for the pty:sideEffect channel — diff --git a/src/main/runtime/orca-runtime-tests/decorative-title-fact-throttle.spec.ts b/src/main/runtime/orca-runtime-tests/decorative-title-fact-throttle.spec.ts new file mode 100644 index 00000000000..a2e35baa4aa --- /dev/null +++ b/src/main/runtime/orca-runtime-tests/decorative-title-fact-throttle.spec.ts @@ -0,0 +1,110 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { TerminalSideEffectBatch } from '../../../shared/terminal-side-effect-facts' +import { syncSinglePty } from '../orca-runtime-test-fixtures.spec' +import { createSideEffectRuntime } from '../orca-runtime-test-scenario-builders.spec' +import { DECORATIVE_TITLE_FACT_HEARTBEAT_MS } from '../decorative-title-fact-emission' + +// Orca's own synthetic agent spinner: one frame per pane every 80ms while an agent works. +const SPINNER_FRAMES = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'] +const SPINNER_INTERVAL_MS = 80 +const EPOCH = 1_700_000_000_000 + +type TitleFact = { kind: 'title'; normalizedTitle: string; rawTitle: string } + +function titleFacts(batches: TerminalSideEffectBatch[]): TitleFact[] { + return batches.flatMap((batch) => + batch.facts.filter((fact): fact is TitleFact => fact.kind === 'title') + ) +} + +describe('decorative title fact throttle', () => { + beforeEach(() => { + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(new Date(EPOCH)) + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('collapses spinner ticks with an unchanged underlying title to the heartbeat rate', () => { + const { runtime, batches } = createSideEffectRuntime() + syncSinglePty(runtime) + + const ticks = 125 // 10s of Orca's 80ms synthetic spinner timer + for (let tick = 0; tick < ticks; tick += 1) { + vi.setSystemTime(new Date(EPOCH + tick * SPINNER_INTERVAL_MS)) + runtime.ingestSyntheticTitleFrame( + 'pty-1', + `\x1b]0;${SPINNER_FRAMES[tick % SPINNER_FRAMES.length]} Claude Code\x07` + ) + } + + const facts = titleFacts(batches) + // Every frame carried the same underlying title, so the renderer learns nothing new past + // the heartbeat: 125 pty:sideEffect messages collapse to one per heartbeat window. + const elapsedMs = ticks * SPINNER_INTERVAL_MS + expect(facts.length).toBeLessThanOrEqual( + Math.ceil(elapsedMs / DECORATIVE_TITLE_FACT_HEARTBEAT_MS) + ) + expect(facts.length).toBeLessThan(ticks / 5) + // The heartbeat must not thin out below what the renderer's 1500ms hook-done quiet window + // needs to cancel a milestone `done` — three working frames per window. + expect(facts.length).toBeGreaterThanOrEqual(Math.floor(elapsedMs / 1_500) * 3) + for (const fact of facts) { + expect(fact.normalizedTitle.endsWith('Claude Code')).toBe(true) + } + }) + + it('propagates a real title change on the tick it arrives, mid-heartbeat', () => { + const { runtime, batches } = createSideEffectRuntime() + syncSinglePty(runtime) + + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠋ Claude Code\x07') + // Two more decorative ticks — still well inside the heartbeat window, so they are dropped. + vi.setSystemTime(new Date(EPOCH + SPINNER_INTERVAL_MS)) + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠙ Claude Code\x07') + vi.setSystemTime(new Date(EPOCH + 2 * SPINNER_INTERVAL_MS)) + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠹ Claude Code\x07') + expect(titleFacts(batches)).toHaveLength(1) + + const beforeChange = batches.length + vi.setSystemTime(new Date(EPOCH + 3 * SPINNER_INTERVAL_MS)) + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;✳ Claude Code\x07') + + expect(batches.length).toBeGreaterThan(beforeChange) + expect(titleFacts(batches.slice(beforeChange))).toEqual([ + { kind: 'title', normalizedTitle: '✳ Claude Code', rawTitle: '✳ Claude Code' } + ]) + }) + + it('propagates a changed working label immediately even while the spinner rotates', () => { + // Why: only the spinner glyph is decoration. Grok/Pi-style label churn is real content. + const { runtime, batches } = createSideEffectRuntime() + syncSinglePty(runtime) + + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠋ Claude Code\x07') + vi.setSystemTime(new Date(EPOCH + SPINNER_INTERVAL_MS)) + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠙ Reviewing diff — Claude Code\x07') + + expect(titleFacts(batches).map((fact) => fact.rawTitle)).toEqual([ + '⠋ Claude Code', + '⠙ Reviewing diff — Claude Code' + ]) + }) + + it('keeps main-side tracked title state current for every suppressed frame', () => { + // Why: mobile/remote snapshots read the tracked record, not the fact stream — suppressing + // the fact must not freeze what a phone or a paired client is shown. + const { runtime } = createSideEffectRuntime() + syncSinglePty(runtime) + + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠋ Claude Code\x07') + vi.setSystemTime(new Date(EPOCH + SPINNER_INTERVAL_MS)) + runtime.ingestSyntheticTitleFrame('pty-1', '\x1b]0;⠙ Claude Code\x07') + + expect(runtime.getTerminalSideEffectSnapshot('pty-1')?.facts).toEqual([ + { kind: 'title', normalizedTitle: '⠙ Claude Code', rawTitle: '⠙ Claude Code' } + ]) + }) +}) diff --git a/src/main/runtime/orca-runtime-tests/terminal-side-effect-facts.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-side-effect-facts.spec.ts index fb07f8a3d0e..79feca3268c 100644 --- a/src/main/runtime/orca-runtime-tests/terminal-side-effect-facts.spec.ts +++ b/src/main/runtime/orca-runtime-tests/terminal-side-effect-facts.spec.ts @@ -8,6 +8,7 @@ import { syncSinglePty } from '../orca-runtime-test-fixtures.spec' import { createSideEffectRuntime } from '../orca-runtime-test-scenario-builders.spec' +import { DECORATIVE_TITLE_FACT_HEARTBEAT_MS } from '../decorative-title-fact-emission' describe('terminal side-effect fact channel', () => { it('defers desktop-only output scanners until a headless runtime is promoted', () => { @@ -70,53 +71,66 @@ describe('terminal side-effect fact channel', () => { expect(events).toHaveLength(1) }) - it('bounds decorative title delivery per paired client without reducing local frames', () => { - const { runtime, batches } = createSideEffectRuntime() - const firstClientEvents: RuntimeClientEvent[] = [] - runtime.attachWindow(1) - runtime.syncWindowGraph(1, { tabs: [], leaves: [] }) - runtime.onClientEvent((event) => firstClientEvents.push(event)) + it('bounds decorative title delivery per paired client below the local heartbeat', () => { + // Why the clock steps: main throttles decorative repeats on the local fact stream, so each + // round must clear that heartbeat for the per-client gate to be what collapses them here. + vi.useFakeTimers({ toFake: ['Date'] }) + try { + const { runtime, batches } = createSideEffectRuntime() + const firstClientEvents: RuntimeClientEvent[] = [] + runtime.attachWindow(1) + runtime.syncWindowGraph(1, { tabs: [], leaves: [] }) + runtime.onClientEvent((event) => firstClientEvents.push(event)) - const ptyIds = Array.from({ length: 64 }, (_, index) => `pty-remote-${index}`) - const frames = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'] - for (const ptyId of ptyIds) { - runtime.ingestSyntheticTitleFrame(ptyId, `\x1b]0;${frames[0]} Cursor Agent\x07`) - } - firstClientEvents.length = 0 - - for (const frame of frames.slice(1)) { - for (const ptyId of ptyIds) { - runtime.ingestSyntheticTitleFrame(ptyId, `\x1b]0;${frame} Cursor Agent\x07`) + const ptyIds = Array.from({ length: 64 }, (_, index) => `pty-remote-${index}`) + const frames = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'] + const stepPastHeartbeat = (): void => { + vi.setSystemTime(new Date(Date.now() + DECORATIVE_TITLE_FACT_HEARTBEAT_MS)) } + for (const ptyId of ptyIds) { + runtime.ingestSyntheticTitleFrame(ptyId, `\x1b]0;${frames[0]} Cursor Agent\x07`) + } + firstClientEvents.length = 0 + + for (const frame of frames.slice(1)) { + stepPastHeartbeat() + for (const ptyId of ptyIds) { + runtime.ingestSyntheticTitleFrame(ptyId, `\x1b]0;${frame} Cursor Agent\x07`) + } + } + + expect(firstClientEvents).toEqual([]) + expect(batches).toHaveLength(ptyIds.length * frames.length) + + const bellChunk = `\x1b]0;${frames.at(-1)} Cursor Agent\x07\x07` + runtime.onPtyData(ptyIds[0], bellChunk, 1) + expect(firstClientEvents).toEqual([ + expect.objectContaining({ + type: 'terminalSideEffects', + batch: expect.objectContaining({ facts: [{ kind: 'bell' }] }) + }) + ]) + firstClientEvents.length = 0 + + const secondClientEvents: RuntimeClientEvent[] = [] + runtime.onClientEvent((event) => secondClientEvents.push(event)) + stepPastHeartbeat() + for (const ptyId of ptyIds) { + runtime.ingestSyntheticTitleFrame(ptyId, `\x1b]0;${frames[0]} Cursor Agent\x07`) + } + + expect(firstClientEvents).toEqual([]) + expect(secondClientEvents).toHaveLength(ptyIds.length) + + // A real title change is never throttled — no clock step needed. + for (const ptyId of ptyIds) { + runtime.ingestSyntheticTitleFrame(ptyId, '\x1b]0;Cursor ready\x07') + } + expect(firstClientEvents).toHaveLength(ptyIds.length) + expect(secondClientEvents).toHaveLength(ptyIds.length * 2) + } finally { + vi.useRealTimers() } - - expect(firstClientEvents).toEqual([]) - expect(batches).toHaveLength(ptyIds.length * frames.length) - - const bellChunk = `\x1b]0;${frames.at(-1)} Cursor Agent\x07\x07` - runtime.onPtyData(ptyIds[0], bellChunk, 1) - expect(firstClientEvents).toEqual([ - expect.objectContaining({ - type: 'terminalSideEffects', - batch: expect.objectContaining({ facts: [{ kind: 'bell' }] }) - }) - ]) - firstClientEvents.length = 0 - - const secondClientEvents: RuntimeClientEvent[] = [] - runtime.onClientEvent((event) => secondClientEvents.push(event)) - for (const ptyId of ptyIds) { - runtime.ingestSyntheticTitleFrame(ptyId, `\x1b]0;${frames[0]} Cursor Agent\x07`) - } - - expect(firstClientEvents).toEqual([]) - expect(secondClientEvents).toHaveLength(ptyIds.length) - - for (const ptyId of ptyIds) { - runtime.ingestSyntheticTitleFrame(ptyId, '\x1b]0;Cursor ready\x07') - } - expect(firstClientEvents).toHaveLength(ptyIds.length) - expect(secondClientEvents).toHaveLength(ptyIds.length * 2) }) it('omits terminalSideEffects from non-consuming listeners while other events still flow', () => { diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index 054d7f09316..f8b12ba4a9d 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -30,6 +30,7 @@ await import('./orca-runtime-tests/pty-title-status.spec') await import('./orca-runtime-tests/terminal-side-effect-facts.spec') await import('./orca-runtime-tests/terminal-side-effect-facts-part-02.spec') await import('./orca-runtime-tests/terminal-side-effect-facts-part-03.spec') +await import('./orca-runtime-tests/decorative-title-fact-throttle.spec') await import('./orca-runtime-tests/headless-snapshots.spec') await import('./orca-runtime-tests/headless-snapshots-part-02.spec') await import('./orca-runtime-tests/agent-status-and-waits.spec') diff --git a/src/main/runtime/runtime-terminal-state-records.ts b/src/main/runtime/runtime-terminal-state-records.ts index 68be05b8ad1..6ccb4ed82bb 100644 --- a/src/main/runtime/runtime-terminal-state-records.ts +++ b/src/main/runtime/runtime-terminal-state-records.ts @@ -89,6 +89,8 @@ export type RuntimePtyTitleTrackerEntry = { tracker: TerminalTitleTracker applyingChunk: boolean lastMobileTitleGateKey: string | null + /** When the last title fact was emitted — throttles decorative-only repeats. */ + lastTitleFactAtMs: number | null chunkTouchedSessionTabs: boolean pendingFacts: TerminalSideEffectFact[] commandCodeDetector: { observe: (data: string) => boolean } | null From f19a860a0f611453ffa34a637dcfa35f9fa1705b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:04:22 -0700 Subject: [PATCH 079/398] perf(terminal): sweep the parked-watcher registries once per pass, not once per workspace (#18157) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The always-mounted terminal controller looped every workspace surface (423 on a large profile) and called syncParkedTerminalTabWatchers per surface; that function scans both module-level registries in full, so one effect fire cost surfaces x registry — 323,172 map-row visits at 423 workspaces / 382 tabs. Add syncParkedTerminalTabWatchersForWorkspaces, which walks each registry once and then runs the per-tab start/reconcile pass; the single-worktree entry point delegates to it. Registry rows are tab-id keyed and a tab belongs to exactly one worktree, so hoisting the dispose and capture sweeps ahead of the start passes only reorders work across disjoint tab sets. Also derive workspaceSurfaceIds/workspaceSurfaceIdSet once in the workspace foundation (through the existing useReusedArrayIdentity) and key the watcher, parking and browser-retention effects on the id array instead of the surface array, which is re-identified on every worktree write. And pass the sidebar's already-computed defaultHostId into useVisibleSidebarWorktrees so an unrelated settings write stops re-running the 423-worktree visibility scan. --- .../src/components/sidebar/WorktreeList.tsx | 2 +- .../listing/use-visible-worktrees.test.tsx | 78 ++- .../listing/use-visible-worktrees.ts | 13 +- .../components/terminal-cold-activation.ts | 8 +- ...nal-parked-tab-watchers-batch-sync.test.ts | 540 ++++++++++++++++++ .../terminal-parked-tab-watchers.ts | 132 ++++- .../terminal-parking-pass-candidates.ts | 9 +- .../terminal-workspace-surface-ids.test.tsx | 184 ++++++ .../use-terminal-browser-retention.ts | 10 +- .../components/use-terminal-parking-pass.ts | 4 +- .../use-terminal-watcher-effects.ts | 43 +- .../use-terminal-workspace-foundation.ts | 14 + 12 files changed, 958 insertions(+), 79 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers-batch-sync.test.ts create mode 100644 src/renderer/src/components/terminal-workspace-surface-ids.test.tsx diff --git a/src/renderer/src/components/sidebar/WorktreeList.tsx b/src/renderer/src/components/sidebar/WorktreeList.tsx index 5d27ec0f23f..9153ff4e39e 100644 --- a/src/renderer/src/components/sidebar/WorktreeList.tsx +++ b/src/renderer/src/components/sidebar/WorktreeList.tsx @@ -116,7 +116,7 @@ const WorktreeList = React.memo(function WorktreeList({ sortedIds, repoMap, worktreeLineageById, - settings, + defaultHostId, agentSendTargetWorktreeId }) const effectiveCollapsedGroups = useEffectiveCollapsedGroups({ diff --git a/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.test.tsx b/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.test.tsx index d385f29d6c4..422a14b3693 100644 --- a/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.test.tsx +++ b/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.test.tsx @@ -1,11 +1,27 @@ // @vitest-environment happy-dom -import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { cleanup, renderHook } from '@testing-library/react' import { useAppStore } from '@/store' +import { LOCAL_EXECUTION_HOST_ID } from '../../../../../../shared/execution-host' import { getWorktreeHostIdentity } from '../../../../../../shared/worktree/host-qualified-identity' import { makeRepo, makeWorktree } from '../../../worktree-jump-palette-test-fixtures' import { useVisibleSidebarWorktrees } from './use-visible-worktrees' +import type * as visibleWorktreesModule from '../../visible-worktrees' + +const computeVisibleWorktreesCalls = { count: 0 } +vi.mock('../../visible-worktrees', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + computeVisibleWorktrees: ( + ...args: Parameters + ): ReturnType => { + computeVisibleWorktreesCalls.count += 1 + return actual.computeVisibleWorktrees(...args) + } + } +}) const initialState = useAppStore.getInitialState() @@ -43,7 +59,7 @@ describe('useVisibleSidebarWorktrees', () => { sortedIds: [local.id, ssh.id], repoMap: new Map([[repo.id, repo]]), worktreeLineageById: {}, - settings: useAppStore.getState().settings, + defaultHostId: LOCAL_EXECUTION_HOST_ID, agentSendTargetWorktreeId: null }) ) @@ -78,7 +94,7 @@ describe('useVisibleSidebarWorktrees', () => { sortedIds: [local.id, ssh.id], repoMap: new Map([[repo.id, repo]]), worktreeLineageById: {}, - settings: useAppStore.getState().settings, + defaultHostId: LOCAL_EXECUTION_HOST_ID, agentSendTargetWorktreeId: null }) ) @@ -87,4 +103,60 @@ describe('useVisibleSidebarWorktrees', () => { getWorktreeHostIdentity(ssh) ]) }) + it('does not rescan every worktree when a settings write leaves the focused host unchanged', () => { + const repo = makeRepo() + const worktree = makeWorktree('alpha', 'Alpha workspace', { hostId: 'local' }) + useAppStore.setState({ worktreesByRepo: { [repo.id]: [worktree] } }) + + const baseArgs = { + filterState: { + showSleepingWorkspaces: true, + filterRepoIds: [], + hideDefaultBranchWorkspace: false, + hideAutomationGeneratedWorkspaces: false, + hideCliCreatedWorkspaces: false, + hideDetachedHeadWorkspaces: false, + hideWorkspacesFromOtherDevices: false, + alwaysShowDefaultBranchWorkspace: true, + visibleWorkspaceHostIds: null, + workspaceHostScope: 'all' + }, + sortBy: 'recent', + sortedIds: [worktree.id], + repoMap: new Map([[repo.id, repo]]), + worktreeLineageById: {}, + defaultHostId: LOCAL_EXECUTION_HOST_ID, + agentSendTargetWorktreeId: null + } as Parameters[0] + // Why the extra `settings`: it is the pre-fix memo key. Passing it keeps + // this test red against the old hook, which re-keyed the whole scan on the + // settings object identity. + const withSettings = ( + settings: ReturnType['settings'] + ): Parameters[0] => Object.assign({}, baseArgs, { settings }) + + computeVisibleWorktreesCalls.count = 0 + const { result, rerender } = renderHook( + (args: Parameters[0]) => useVisibleSidebarWorktrees(args), + { initialProps: withSettings(useAppStore.getState().settings) } + ) + const initialVisible = result.current.visibleWorktrees + const callsAfterFirstRender = computeVisibleWorktreesCalls.count + expect(callsAfterFirstRender).toBe(1) + + // A settings write that does not move the focused execution host. + const nextSettings = { + ...useAppStore.getState().settings, + sidebarWidth: 321 + } as ReturnType['settings'] + useAppStore.setState({ settings: nextSettings }) + rerender(withSettings(nextSettings)) + + expect(computeVisibleWorktreesCalls.count).toBe(callsAfterFirstRender) + expect(result.current.visibleWorktrees).toBe(initialVisible) + + // A write that does move it still recomputes. + rerender(Object.assign({}, withSettings(nextSettings), { defaultHostId: 'runtime:other' })) + expect(computeVisibleWorktreesCalls.count).toBe(callsAfterFirstRender + 1) + }) }) diff --git a/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.ts b/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.ts index e04d57ccff9..804d5118605 100644 --- a/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.ts +++ b/src/renderer/src/components/sidebar/worktree-list/listing/use-visible-worktrees.ts @@ -2,10 +2,9 @@ import { useMemo } from 'react' import { useAppStore } from '@/store' import { getAgentStatusEpochNow } from '@/lib/agent-status-epoch-clock' import { getWorktreeIdsWithLiveAgent } from '@/lib/worktree-activity-state' -import type { AppState } from '@/store/types' import type { Repo } from '../../../../../../shared/repo-types' import type { WorktreeLineage } from '../../../../../../shared/worktree/lineage-types' -import { getSettingsFocusedExecutionHostId } from '../../../../../../shared/execution-host' +import type { ExecutionHostId } from '../../../../../../shared/execution-host' import { computeVisibleWorktrees } from '../../visible-worktrees' import { EMPTY_PAIRED_DEVICE_IDS_BY_ENVIRONMENT, @@ -29,10 +28,12 @@ export function useVisibleSidebarWorktrees(args: { sortedIds: string[] repoMap: Map worktreeLineageById: Record - settings: AppState['settings'] + /** Pre-derived focused host; the whole `settings` object would re-key this + * 423-workspace scan on every unrelated settings write. */ + defaultHostId: ExecutionHostId agentSendTargetWorktreeId: string | null }) { - const { filterState, sortBy, sortedIds, repoMap, worktreeLineageById, settings } = args + const { filterState, sortBy, sortedIds, repoMap, worktreeLineageById, defaultHostId } = args const { showSleepingWorkspaces, filterRepoIds, @@ -98,7 +99,7 @@ export function useVisibleSidebarWorktrees(args: { repoMap, workspaceHostScope, visibleWorkspaceHostIds, - defaultHostId: getSettingsFocusedExecutionHostId(settings), + defaultHostId, worktreeLineageById, forcedVisibleWorktreeIds: args.agentSendTargetWorktreeId ? [args.agentSendTargetWorktreeId] @@ -118,7 +119,7 @@ export function useVisibleSidebarWorktrees(args: { alwaysShowDefaultBranchWorkspace, workspaceHostScope, visibleWorkspaceHostIds, - settings, + defaultHostId, repoMap, tabsByWorktree, ptyIdsByTabId, diff --git a/src/renderer/src/components/terminal-cold-activation.ts b/src/renderer/src/components/terminal-cold-activation.ts index 8273e1536e6..d57cb82d766 100644 --- a/src/renderer/src/components/terminal-cold-activation.ts +++ b/src/renderer/src/components/terminal-cold-activation.ts @@ -36,7 +36,8 @@ export function applyTerminalColdActivation(controller: TerminalParkingFoundatio terminalParkingEnabled, terminalTitleSnapshotAuthorityEnabled, workspaceSessionReady, - workspaceSurfaces + workspaceSurfaceIds, + workspaceSurfaceIdSet } = controller if ( renderedActiveWorktreeId && @@ -150,16 +151,15 @@ export function applyTerminalColdActivation(controller: TerminalParkingFoundatio tabsByWorktree, activationDeferredMountTabIdsByWorktreeRef.current ) - const allWorktreeIds = new Set(workspaceSurfaces.map((workspace) => workspace.id)) for (const id of mountedWorktreeIdsRef.current) { - if (!allWorktreeIds.has(id)) { + if (!workspaceSurfaceIdSet.has(id)) { mountedWorktreeIdsRef.current.delete(id) backgroundMountTabIdsByWorktreeRef.current.delete(id) activationDeferredMountTabIdsByWorktreeRef.current.delete(id) } } const anyMountedWorktreeHasLayout = computeAnyMountedWorktreeHasLayout( - workspaceSurfaces.map((workspace) => workspace.id), + workspaceSurfaceIds, mountedWorktreeIdsRef.current, layoutByWorktree, groupsByWorktree, diff --git a/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers-batch-sync.test.ts b/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers-batch-sync.test.ts new file mode 100644 index 00000000000..8bb0dca14ff --- /dev/null +++ b/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers-batch-sync.test.ts @@ -0,0 +1,540 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { ParkedTerminalByteWatcherOptions } from './parked-terminal-byte-watcher' + +type StartEvent = { + kind: 'start' + worktreeId: string + tabId: string + ptyId: string + paneId: number + leafId: string + drivesTabTitle: boolean + restoreTitleOnRegister: boolean +} +type DisposeEvent = { kind: 'dispose'; worktreeId: string; tabId: string; ptyId: string } +type ClearTitleEvent = { kind: 'clearTitle'; tabId: string; paneId: number } +type DiscardEvent = { kind: 'discard'; ptyId: string } +type WatcherEvent = StartEvent | DisposeEvent | ClearTitleEvent | DiscardEvent + +let events: WatcherEvent[] = [] + +const startParkedTerminalByteWatcher = vi.fn((options: ParkedTerminalByteWatcherOptions) => { + events.push({ + kind: 'start', + worktreeId: options.worktreeId, + tabId: options.tabId, + ptyId: options.ptyId, + paneId: options.paneId, + leafId: options.leafId, + drivesTabTitle: options.drivesTabTitle === true, + restoreTitleOnRegister: options.restoreTitleOnRegister === true + }) + return () => { + events.push({ + kind: 'dispose', + worktreeId: options.worktreeId, + tabId: options.tabId, + ptyId: options.ptyId + }) + } +}) + +vi.mock('./parked-terminal-byte-watcher', () => ({ + startParkedTerminalByteWatcher: (options: ParkedTerminalByteWatcherOptions) => + startParkedTerminalByteWatcher(options) +})) + +vi.mock('./pty-dispatcher', () => ({ + subscribeToPtyExit: () => () => {} +})) + +vi.mock('./pty-pre-handler-buffer', () => ({ + discardPreHandlerPtyState: (ptyId: string) => { + events.push({ kind: 'discard', ptyId }) + }, + hasPreHandlerPtyExit: () => false +})) + +vi.mock('../terminal/terminal-tab-actions', () => ({ + closeTerminalTab: () => {} +})) + +type TabModel = { id: string; ptyId: string | null } +type MockStoreState = { + tabsByWorktree: Record + terminalLayoutsByTabId: Record< + string, + { + root: unknown + activeLeafId: string | null + expandedLeafId: string | null + ptyIdsByLeafId?: Record + } + > + runtimePaneTitlesByTabId: Record> + settings: { terminalSshViewParking?: boolean } | null + runtimeStatusByEnvironmentId: Map + clearRuntimePaneTitle: (tabId: string, paneId: number) => void + setRuntimePaneTitle: () => void + clearTabLaunchAgent: () => void + setTabLayout: () => void + updateTabTitle: () => void + markUnverifiedPtyLoss: () => void + isPtyShutdownPending: () => boolean + suppressedPtyExitIds: Record +} + +let mockStoreState: MockStoreState + +vi.mock('@/store', () => ({ + useAppStore: { getState: () => mockStoreState } +})) + +import { + clearTerminalProviderSnapshotCapabilities, + synchronizeTerminalProviderSnapshotCapabilities +} from '../terminal/terminal-provider-snapshot-capability' +import { + captureParkedTerminalPaneCandidates, + pruneParkedTerminalWatchers, + syncParkedTerminalTabWatchers, + syncParkedTerminalTabWatchersForWorkspaces, + type ParkedTerminalTabWatcherSyncEntry +} from './terminal-parked-tab-watchers' +import { capturedPanesByTabId, parkedWatchersByTabId } from './terminal-parked-watcher-registry' + +const leafId = (index: number): string => + `${index.toString(16).padStart(8, '0')}-1111-4111-8111-111111111111` + +function makeStore(): MockStoreState { + return { + tabsByWorktree: {}, + terminalLayoutsByTabId: {}, + runtimePaneTitlesByTabId: {}, + settings: null, + runtimeStatusByEnvironmentId: new Map(), + clearRuntimePaneTitle: (tabId: string, paneId: number) => { + events.push({ kind: 'clearTitle', tabId, paneId }) + }, + setRuntimePaneTitle: () => {}, + clearTabLaunchAgent: () => {}, + setTabLayout: () => {}, + updateTabTitle: () => {}, + markUnverifiedPtyLoss: () => {}, + isPtyShutdownPending: () => false, + suppressedPtyExitIds: {} + } +} + +/** Counts entries visited by `for...of` over the module-level registries. */ +function instrumentRegistryIteration(): { count: number; restore: () => void } { + const counter = { count: 0, restore: () => {} } + const patched: Map[] = [ + parkedWatchersByTabId as Map, + capturedPanesByTabId as Map + ] + for (const map of patched) { + Object.defineProperty(map, Symbol.iterator, { + configurable: true, + writable: true, + value: function* (this: Map) { + for (const entry of Map.prototype.entries.call(this)) { + counter.count += 1 + yield entry + } + } + }) + } + counter.restore = () => { + for (const map of patched) { + Reflect.deleteProperty(map, Symbol.iterator) + } + } + return counter +} + +describe('parked terminal watcher batch synchronization', () => { + beforeEach(() => { + events = [] + mockStoreState = makeStore() + clearTerminalProviderSnapshotCapabilities() + }) + + afterEach(() => { + pruneParkedTerminalWatchers(new Set()) + capturedPanesByTabId.clear() + events = [] + vi.clearAllMocks() + clearTerminalProviderSnapshotCapabilities() + }) + + describe('registry scan cost at the real-profile scale', () => { + // The user's profile: 423 workspace surfaces, 382 terminal tabs. + const WORKSPACE_COUNT = 423 + const TAB_COUNT = 382 + + function seedRegistries(): { + workspaceIds: string[] + tabsByWorktreeId: Map + } { + const workspaceIds = Array.from( + { length: WORKSPACE_COUNT }, + (_, index) => `repo::/worktree-${index}` + ) + const tabsByWorktreeId = new Map( + workspaceIds.map((workspaceId) => [workspaceId, [] as TabModel[]]) + ) + for (let index = 0; index < TAB_COUNT; index += 1) { + const worktreeId = workspaceIds[index % WORKSPACE_COUNT] + const tabId = `tab-${index}` + const ptyId = `${worktreeId}@@session-${index}` + tabsByWorktreeId.get(worktreeId)!.push({ id: tabId, ptyId }) + // A live, already-parked tab: present in both registries, nothing to + // dispose and nothing to start, so the pass is a pure registry scan. + parkedWatchersByTabId.set(tabId, { + worktreeId, + tabPtyId: ptyId, + paneIdByPtyId: new Map([[ptyId, 1]]), + disposersByPtyId: new Map() + }) + captureParkedTerminalPaneCandidates(tabId, worktreeId, [ + { ptyId, paneId: 1, leafId: leafId(index), drivesTabTitle: true } + ]) + } + return { workspaceIds, tabsByWorktreeId } + } + + it('collapses surfaces x registry scans into one scan of each registry', () => { + const { workspaceIds, tabsByWorktreeId } = seedRegistries() + const registryRows = parkedWatchersByTabId.size + capturedPanesByTabId.size + expect(registryRows).toBe(TAB_COUNT * 2) + + const perSurface = instrumentRegistryIteration() + for (const workspaceId of workspaceIds) { + syncParkedTerminalTabWatchers({ + worktreeId: workspaceId, + tabs: tabsByWorktreeId.get(workspaceId)!, + parkedTabIds: new Set() + }) + } + const perSurfaceVisits = perSurface.count + perSurface.restore() + + const batched = instrumentRegistryIteration() + const entries = new Map( + workspaceIds.map((workspaceId) => [ + workspaceId, + { tabs: tabsByWorktreeId.get(workspaceId)!, parkedTabIds: new Set() } + ]) + ) + syncParkedTerminalTabWatchersForWorkspaces(entries) + const batchedVisits = batched.count + batched.restore() + + // Old shape: every surface re-walks both registries in full. + expect(perSurfaceVisits).toBe(WORKSPACE_COUNT * registryRows) + // New shape: each registry is walked exactly once for the whole pass. + expect(batchedVisits).toBe(registryRows) + expect(perSurfaceVisits / batchedVisits).toBeGreaterThan(400) + }) + }) + + describe('start/dispose decisions match the per-surface path', () => { + type Scenario = { + name: string + workspaces: { + worktreeId: string + tabs: TabModel[] + parkedTabIds: string[] + restoreTitleOnStartTabIds?: string[] + }[] + /** Watcher rows already in the registry when the pass runs. */ + preParkedTabs: { worktreeId: string; tabId: string; ptyId: string; withDisposer: boolean }[] + /** Captures for tabs that may or may not still be live. */ + preCapturedTabs: { worktreeId: string; tabId: string; ptyId: string | null }[] + } + + const WORKTREE_A = 'repo::/alpha' + const WORKTREE_B = 'repo::/beta' + const WORKTREE_C = 'repo::/gamma' + + const pty = (worktreeId: string, index: number): string => `${worktreeId}@@session-${index}` + + const scenarios: Scenario[] = [ + { + name: 'cold start: nothing parked yet, two workspaces park every tab', + workspaces: [ + { + worktreeId: WORKTREE_A, + tabs: [ + { id: 'a1', ptyId: pty(WORKTREE_A, 1) }, + { id: 'a2', ptyId: pty(WORKTREE_A, 2) } + ], + parkedTabIds: ['a1', 'a2'] + }, + { + worktreeId: WORKTREE_B, + tabs: [{ id: 'b1', ptyId: pty(WORKTREE_B, 1) }], + parkedTabIds: ['b1'] + }, + { worktreeId: WORKTREE_C, tabs: [], parkedTabIds: [] } + ], + preParkedTabs: [], + preCapturedTabs: [] + }, + { + name: 'reveal: a parked workspace drops out of the parked set', + workspaces: [ + { + worktreeId: WORKTREE_A, + tabs: [ + { id: 'a1', ptyId: pty(WORKTREE_A, 1) }, + { id: 'a2', ptyId: pty(WORKTREE_A, 2) } + ], + parkedTabIds: [] + }, + { + worktreeId: WORKTREE_B, + tabs: [{ id: 'b1', ptyId: pty(WORKTREE_B, 1) }], + parkedTabIds: ['b1'] + } + ], + preParkedTabs: [ + { worktreeId: WORKTREE_A, tabId: 'a1', ptyId: pty(WORKTREE_A, 1), withDisposer: true }, + { worktreeId: WORKTREE_A, tabId: 'a2', ptyId: pty(WORKTREE_A, 2), withDisposer: true } + ], + preCapturedTabs: [ + { worktreeId: WORKTREE_A, tabId: 'a1', ptyId: pty(WORKTREE_A, 1) }, + { worktreeId: WORKTREE_A, tabId: 'a2', ptyId: pty(WORKTREE_A, 2) } + ] + }, + { + name: 'closed tabs: registry rows and captures outlive their tabs', + workspaces: [ + { + worktreeId: WORKTREE_A, + tabs: [{ id: 'a1', ptyId: pty(WORKTREE_A, 1) }], + parkedTabIds: ['a1'] + }, + { worktreeId: WORKTREE_B, tabs: [], parkedTabIds: [] } + ], + preParkedTabs: [ + { + worktreeId: WORKTREE_A, + tabId: 'a-closed', + ptyId: pty(WORKTREE_A, 9), + withDisposer: true + }, + { + worktreeId: WORKTREE_B, + tabId: 'b-closed', + ptyId: pty(WORKTREE_B, 9), + withDisposer: true + } + ], + preCapturedTabs: [ + { worktreeId: WORKTREE_A, tabId: 'a-closed', ptyId: pty(WORKTREE_A, 9) }, + { worktreeId: WORKTREE_B, tabId: 'b-closed', ptyId: pty(WORKTREE_B, 9) }, + { worktreeId: WORKTREE_A, tabId: 'a1', ptyId: pty(WORKTREE_A, 1) } + ] + }, + { + name: 're-minted pty: a parked tab wakes with a fresh pty id', + workspaces: [ + { + worktreeId: WORKTREE_A, + tabs: [{ id: 'a1', ptyId: pty(WORKTREE_A, 2) }], + parkedTabIds: ['a1'], + restoreTitleOnStartTabIds: ['a1'] + }, + { + worktreeId: WORKTREE_B, + tabs: [{ id: 'b1', ptyId: pty(WORKTREE_B, 1) }], + parkedTabIds: ['b1'] + } + ], + preParkedTabs: [ + { worktreeId: WORKTREE_A, tabId: 'a1', ptyId: pty(WORKTREE_A, 1), withDisposer: true } + ], + preCapturedTabs: [{ worktreeId: WORKTREE_A, tabId: 'a1', ptyId: pty(WORKTREE_A, 2) }] + }, + { + name: 'unknown worktree rows: registry holds a workspace absent from this pass', + workspaces: [ + { + worktreeId: WORKTREE_A, + tabs: [{ id: 'a1', ptyId: pty(WORKTREE_A, 1) }], + parkedTabIds: ['a1'] + } + ], + preParkedTabs: [ + { worktreeId: WORKTREE_C, tabId: 'c1', ptyId: pty(WORKTREE_C, 1), withDisposer: true } + ], + preCapturedTabs: [{ worktreeId: WORKTREE_C, tabId: 'c1', ptyId: pty(WORKTREE_C, 1) }] + }, + { + name: 'tombstone rows: a pinned-close entry with no live disposers', + workspaces: [ + { + worktreeId: WORKTREE_A, + tabs: [{ id: 'a1', ptyId: pty(WORKTREE_A, 1) }], + parkedTabIds: [] + }, + { + worktreeId: WORKTREE_B, + tabs: [{ id: 'b1', ptyId: pty(WORKTREE_B, 1) }], + parkedTabIds: ['b1'] + } + ], + preParkedTabs: [ + { worktreeId: WORKTREE_A, tabId: 'a1', ptyId: pty(WORKTREE_A, 1), withDisposer: false } + ], + preCapturedTabs: [{ worktreeId: WORKTREE_B, tabId: 'b1', ptyId: pty(WORKTREE_B, 1) }] + } + ] + + type Outcome = { + eventsByWorktree: Record + registry: readonly (readonly [ + string, + { worktreeId: string; tabPtyId: string | null; ptyIds: string[] } + ])[] + captures: string[] + } + + async function seedAndRun( + scenario: Scenario, + run: (scenario: Scenario) => void + ): Promise { + pruneParkedTerminalWatchers(new Set()) + capturedPanesByTabId.clear() + clearTerminalProviderSnapshotCapabilities() + mockStoreState = makeStore() + const allPtyIds = new Set() + for (const workspace of scenario.workspaces) { + for (const tab of workspace.tabs) { + if (tab.ptyId) { + allPtyIds.add(tab.ptyId) + } + } + } + for (const row of [...scenario.preParkedTabs, ...scenario.preCapturedTabs]) { + if (row.ptyId) { + allPtyIds.add(row.ptyId) + } + } + await synchronizeTerminalProviderSnapshotCapabilities(Array.from(allPtyIds), async (ids) => + ids.map((id) => ({ id, authoritative: true })) + ) + let paneOrdinal = 0 + for (const row of scenario.preParkedTabs) { + paneOrdinal += 1 + parkedWatchersByTabId.set(row.tabId, { + worktreeId: row.worktreeId, + tabPtyId: row.ptyId, + paneIdByPtyId: new Map([[row.ptyId, paneOrdinal]]), + disposersByPtyId: row.withDisposer + ? new Map([ + [ + row.ptyId, + () => { + events.push({ + kind: 'dispose', + worktreeId: row.worktreeId, + tabId: row.tabId, + ptyId: row.ptyId + }) + } + ] + ]) + : new Map() + }) + } + for (const [index, row] of scenario.preCapturedTabs.entries()) { + captureParkedTerminalPaneCandidates(row.tabId, row.worktreeId, [ + { ptyId: row.ptyId, paneId: 100 + index, leafId: leafId(index + 1), drivesTabTitle: true } + ]) + } + events = [] + run(scenario) + + const eventsByWorktree: Record = {} + for (const event of events) { + const key = 'worktreeId' in event ? event.worktreeId : 'shared' + ;(eventsByWorktree[key] ??= []).push(event) + } + // Why per-worktree, not one global sequence: batching deliberately hoists + // every workspace's dispose sweep ahead of every workspace's start pass. + // Registry rows are tab-id keyed and a tab belongs to exactly one + // worktree, so that reorder crosses only disjoint tab sets. `shared` + // events (pane-title clears, pre-handler discards) are compared as a + // multiset for the same reason. + for (const key of Object.keys(eventsByWorktree)) { + if (key === 'shared') { + eventsByWorktree[key] = [...eventsByWorktree[key]].sort((left, right) => + JSON.stringify(left).localeCompare(JSON.stringify(right)) + ) + } + } + return { + eventsByWorktree, + registry: Array.from( + parkedWatchersByTabId, + ([tabId, entry]) => + [ + tabId, + { + worktreeId: entry.worktreeId, + tabPtyId: entry.tabPtyId, + ptyIds: Array.from(entry.disposersByPtyId.keys()).sort() + } + ] as const + ).sort((left, right) => left[0].localeCompare(right[0])), + captures: Array.from(capturedPanesByTabId.keys()).sort() + } + } + + const runPerSurface = (scenario: Scenario): void => { + for (const workspace of scenario.workspaces) { + syncParkedTerminalTabWatchers({ + worktreeId: workspace.worktreeId, + tabs: workspace.tabs, + parkedTabIds: new Set(workspace.parkedTabIds), + ...(workspace.restoreTitleOnStartTabIds + ? { restoreTitleOnStartTabIds: new Set(workspace.restoreTitleOnStartTabIds) } + : {}) + }) + } + } + + const runBatched = (scenario: Scenario): void => { + syncParkedTerminalTabWatchersForWorkspaces( + new Map( + scenario.workspaces.map((workspace) => [ + workspace.worktreeId, + { + tabs: workspace.tabs, + parkedTabIds: new Set(workspace.parkedTabIds), + ...(workspace.restoreTitleOnStartTabIds + ? { restoreTitleOnStartTabIds: new Set(workspace.restoreTitleOnStartTabIds) } + : {}) + } + ]) + ) + ) + } + + for (const scenario of scenarios) { + it(`decides identically — ${scenario.name}`, async () => { + const perSurface = await seedAndRun(scenario, runPerSurface) + const batched = await seedAndRun(scenario, runBatched) + expect(batched).toEqual(perSurface) + // Guard against a vacuous comparison of two empty outcomes. + expect( + Object.values(perSurface.eventsByWorktree).some((list) => list.length > 0) || + perSurface.registry.length > 0 + ).toBe(true) + }) + } + }) +}) diff --git a/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers.ts b/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers.ts index f834944607b..d088fbe41ad 100644 --- a/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers.ts +++ b/src/renderer/src/components/terminal-pane/terminal-parked-tab-watchers.ts @@ -244,6 +244,93 @@ function disposeClosedParkedTabWatchers( disposeParkedTabWatchers(tabId) } +/** One workspace's rendered parked verdict, as the batch pass consumes it. */ +export type ParkedTerminalTabWatcherSyncEntry = { + tabs: readonly ParkableTerminalTabModel[] + parkedTabIds: ReadonlySet + /** Parked-equivalent tabs whose pane has not restored the current title. */ + restoreTitleOnStartTabIds?: ReadonlySet +} + +function startOrReconcileParkedTabWatchers( + worktreeId: string, + entry: ParkedTerminalTabWatcherSyncEntry +): void { + for (const tab of entry.tabs) { + if (!entry.parkedTabIds.has(tab.id)) { + continue + } + const watcherEntry = parkedWatchersByTabId.get(tab.id) + const restoreTitleOnRegister = entry.restoreTitleOnStartTabIds?.has(tab.id) === true + if (watcherEntry) { + reconcileParkedTabWatchers(worktreeId, tab, watcherEntry, restoreTitleOnRegister) + } else { + startParkedTabWatchers(worktreeId, tab, restoreTitleOnRegister) + } + } +} + +/** + * Reconciles watchers for every rendered workspace in one pass. + * + * Why batched: the per-worktree entry point scans both registries in full, so + * the terminal host calling it once per workspace made a single effect fire + * cost surfaces x registry map iterations. Walking each registry once and then + * doing the per-tab start/reconcile pass is O(registry + tabs) instead. + * + * The dispose/start decisions are identical: registry entries are keyed by tab + * id and every tab belongs to exactly one worktree, so hoisting the dispose and + * capture-cleanup sweeps ahead of every start only reorders work across + * disjoint tab sets. + */ +export function syncParkedTerminalTabWatchersForWorkspaces( + entriesByWorktreeId: ReadonlyMap +): void { + if (entriesByWorktreeId.size === 0) { + return + } + // Why lazy: only worktrees that actually own a registry row need the id set, + // so an idle profile allocates none of them. + const liveTabIdsByWorktreeId = new Map>() + const liveTabIdsFor = (worktreeId: string): ReadonlySet | null => { + const cached = liveTabIdsByWorktreeId.get(worktreeId) + if (cached) { + return cached + } + const entry = entriesByWorktreeId.get(worktreeId) + if (!entry) { + return null + } + const liveTabIds = new Set(entry.tabs.map((tab) => tab.id)) + liveTabIdsByWorktreeId.set(worktreeId, liveTabIds) + return liveTabIds + } + for (const [tabId, watcherEntry] of parkedWatchersByTabId) { + const entry = entriesByWorktreeId.get(watcherEntry.worktreeId) + if (!entry) { + continue + } + const liveTabIds = liveTabIdsFor(watcherEntry.worktreeId) + if (!liveTabIds?.has(tabId)) { + disposeClosedParkedTabWatchers(tabId, watcherEntry) + continue + } + if (!entry.parkedTabIds.has(tabId) && watcherEntry.disposersByPtyId.size > 0) { + disposeParkedTabWatchers(tabId) + } + } + // Why: closed tabs never park/reveal again; drop captures to keep the registry bounded. + for (const [tabId, capture] of capturedPanesByTabId) { + const liveTabIds = liveTabIdsFor(capture.worktreeId) + if (liveTabIds && !liveTabIds.has(tabId)) { + capturedPanesByTabId.delete(tabId) + } + } + for (const [worktreeId, entry] of entriesByWorktreeId) { + startOrReconcileParkedTabWatchers(worktreeId, entry) + } +} + /** * Reconciles watchers for one worktree against its rendered parked set. * Run from an effect keyed on committed render state so disposal shares the @@ -256,35 +343,18 @@ export function syncParkedTerminalTabWatchers(args: { /** Parked-equivalent tabs whose pane has not restored the current title. */ restoreTitleOnStartTabIds?: ReadonlySet }): void { - const liveTabIds = new Set(args.tabs.map((tab) => tab.id)) - for (const [tabId, entry] of parkedWatchersByTabId) { - if (entry.worktreeId !== args.worktreeId) { - continue - } - if (!liveTabIds.has(tabId)) { - disposeClosedParkedTabWatchers(tabId, entry) - continue - } - if (!args.parkedTabIds.has(tabId) && entry.disposersByPtyId.size > 0) { - disposeParkedTabWatchers(tabId) - } - } - // Why: closed tabs never park/reveal again; drop captures to keep the registry bounded. - for (const [tabId, capture] of capturedPanesByTabId) { - if (capture.worktreeId === args.worktreeId && !liveTabIds.has(tabId)) { - capturedPanesByTabId.delete(tabId) - } - } - for (const tab of args.tabs) { - if (!args.parkedTabIds.has(tab.id)) { - continue - } - const entry = parkedWatchersByTabId.get(tab.id) - const restoreTitleOnRegister = args.restoreTitleOnStartTabIds?.has(tab.id) === true - if (entry) { - reconcileParkedTabWatchers(args.worktreeId, tab, entry, restoreTitleOnRegister) - } else { - startParkedTabWatchers(args.worktreeId, tab, restoreTitleOnRegister) - } - } + syncParkedTerminalTabWatchersForWorkspaces( + new Map([ + [ + args.worktreeId, + { + tabs: args.tabs, + parkedTabIds: args.parkedTabIds, + ...(args.restoreTitleOnStartTabIds + ? { restoreTitleOnStartTabIds: args.restoreTitleOnStartTabIds } + : {}) + } + ] + ]) + ) } diff --git a/src/renderer/src/components/terminal-parking-pass-candidates.ts b/src/renderer/src/components/terminal-parking-pass-candidates.ts index 1f7d85d89de..98f9ff5e5cd 100644 --- a/src/renderer/src/components/terminal-parking-pass-candidates.ts +++ b/src/renderer/src/components/terminal-parking-pass-candidates.ts @@ -27,7 +27,8 @@ export function collectTerminalParkingPassCandidates(controller: TerminalParking terminalWorktreeHiddenSinceRef, terminalWorktreeParkCooldownUntilRef, terminalWorktreeParkingTimersRef, - workspaceSurfaces + workspaceSurfaceIds, + workspaceSurfaceIdSet } = controller const parkingTimers = terminalWorktreeParkingTimersRef.current for (const timer of parkingTimers.values()) { @@ -38,9 +39,8 @@ export function collectTerminalParkingPassCandidates(controller: TerminalParking const nowMs = Date.now() const overrides = getTerminalParkingPolicyOverrides() const portalWorktreeIds = new Set(activityTerminalPortals.map((portal) => portal.worktreeId)) - const currentWorktreeIds = new Set(workspaceSurfaces.map((workspace) => workspace.id)) for (const worktreeId of Array.from(terminalWorktreeHiddenSinceRef.current.keys())) { - if (!currentWorktreeIds.has(worktreeId) || !mountedWorktreeIdsRef.current.has(worktreeId)) { + if (!workspaceSurfaceIdSet.has(worktreeId) || !mountedWorktreeIdsRef.current.has(worktreeId)) { terminalWorktreeHiddenSinceRef.current.delete(worktreeId) measuringTerminalWorktreeIdsRef.current.delete(worktreeId) terminalWorktreeParkCooldownUntilRef.current.delete(worktreeId) @@ -48,8 +48,7 @@ export function collectTerminalParkingPassCandidates(controller: TerminalParking } const retentionCandidates: TerminalWorktreeColdParkCandidate[] = [] - for (const workspace of workspaceSurfaces) { - const worktreeId = workspace.id + for (const worktreeId of workspaceSurfaceIds) { if (!mountedWorktreeIdsRef.current.has(worktreeId)) { terminalWorktreeHiddenSinceRef.current.delete(worktreeId) measuringTerminalWorktreeIdsRef.current.delete(worktreeId) diff --git a/src/renderer/src/components/terminal-workspace-surface-ids.test.tsx b/src/renderer/src/components/terminal-workspace-surface-ids.test.tsx new file mode 100644 index 00000000000..770b1cfc47b --- /dev/null +++ b/src/renderer/src/components/terminal-workspace-surface-ids.test.tsx @@ -0,0 +1,184 @@ +// @vitest-environment happy-dom + +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { act, cleanup, renderHook } from '@testing-library/react' +import { useEffect } from 'react' +import { useAppStore } from '@/store' +import { makeRepo, makeWorktree } from './worktree-jump-palette-test-fixtures' +import { useTerminalWorkspaceFoundation } from './use-terminal-workspace-foundation' +import { applyTerminalColdActivation } from './terminal-cold-activation' +import { collectTerminalParkingPassCandidates } from './terminal-parking-pass-candidates' +import type { WorkspaceSurface } from './workspace-surface-projection' +import type { TerminalParkingFoundation } from './use-terminal-parking-foundation' + +const initialState = useAppStore.getInitialState() + +/** An id-only surface array that reports how often a caller re-derived its ids. */ +function countingSurfaces(ids: string[]): { surfaces: WorkspaceSurface[]; mapCalls: () => number } { + let mapCalls = 0 + const surfaces = ids.map((id) => ({ id, path: `/tmp/${id}` })) + Object.defineProperty(surfaces, 'map', { + configurable: true, + value(this: WorkspaceSurface[], ...args: Parameters) { + mapCalls += 1 + return Array.prototype.map.apply(this, args) + } + }) + return { surfaces, mapCalls: () => mapCalls } +} + +describe('workspace surface ids', () => { + beforeEach(() => { + useAppStore.setState(initialState, true) + }) + + afterEach(() => { + cleanup() + useAppStore.setState(initialState, true) + vi.restoreAllMocks() + }) + + it('keeps the id array and id set stable when a worktree write re-identifies the surfaces', () => { + const repo = makeRepo() + const first = makeWorktree('alpha', 'Alpha') + const second = makeWorktree('beta', 'Beta') + useAppStore.setState({ worktreesByRepo: { [repo.id]: [first, second] } }) + + const { result, rerender } = renderHook(() => useTerminalWorkspaceFoundation()) + const initialSurfaces = result.current.workspaceSurfaces + const initialIds = result.current.workspaceSurfaceIds + const initialIdSet = result.current.workspaceSurfaceIdSet + expect(initialIds).toEqual([first.id, second.id]) + + // A worktree write that changes no id: fresh row objects, fresh array. + useAppStore.setState({ + worktreesByRepo: { + [repo.id]: [ + { ...first, lastActivityAt: Date.now() }, + { ...second, lastActivityAt: Date.now() } + ] + } + }) + rerender() + + expect(result.current.workspaceSurfaces).not.toBe(initialSurfaces) + expect(result.current.workspaceSurfaceIds).toBe(initialIds) + expect(result.current.workspaceSurfaceIdSet).toBe(initialIdSet) + + // A real membership change still re-identifies both. + useAppStore.setState({ worktreesByRepo: { [repo.id]: [first] } }) + rerender() + expect(result.current.workspaceSurfaceIds).not.toBe(initialIds) + expect(result.current.workspaceSurfaceIdSet).not.toBe(initialIdSet) + expect(result.current.workspaceSurfaceIds).toEqual([first.id]) + }) + + it('does not rebuild a surface-id array or set on a cold-activation render', () => { + const ids = Array.from({ length: 423 }, (_, index) => `repo::/worktree-${index}`) + const { surfaces, mapCalls } = countingSurfaces(ids) + const controller = { + activationDeferredMountTabIdsByWorktreeRef: { current: new Map() }, + activeGroupIdByWorktree: {}, + activeTabId: null, + activeTabIdByWorktree: {}, + activeWorktreeDeferralHostId: null, + activityTerminalPortals: [], + backgroundMountTabIdsByWorktreeRef: { current: new Map() }, + groupsByWorktree: {}, + hydrationSucceeded: false, + lastActivationWorktreeIdRef: { current: null }, + layoutByWorktree: {}, + mountedWorktreeIdsRef: { current: new Set(['repo::/worktree-0', 'repo::/gone']) }, + pairedRuntimeParkingEnvironmentIds: new Set(), + pendingStartupByTabId: {}, + renderedActiveWorktreeId: null, + startupWorktreeRefreshCompleted: false, + tabsByWorktree: {}, + terminalParkingEnabled: true, + terminalTitleSnapshotAuthorityEnabled: true, + workspaceSessionReady: false, + workspaceSurfaces: surfaces, + workspaceSurfaceIds: ids, + workspaceSurfaceIdSet: new Set(ids) + } as unknown as TerminalParkingFoundation + + applyTerminalColdActivation(controller) + applyTerminalColdActivation(controller) + + // Pre-fix this was two 423-element id arrays (plus a 423-entry Set) per render. + expect(mapCalls()).toBe(0) + expect(controller.mountedWorktreeIdsRef.current.has('repo::/gone')).toBe(false) + expect(controller.mountedWorktreeIdsRef.current.has('repo::/worktree-0')).toBe(true) + }) + + it('does not rebuild a surface-id array or set on a parking pass', () => { + const ids = Array.from({ length: 423 }, (_, index) => `repo::/worktree-${index}`) + const { surfaces, mapCalls } = countingSurfaces(ids) + const hiddenSince = new Map([ + ['repo::/worktree-0', 1], + ['repo::/stale', 2] + ]) + const controller = { + activeView: 'terminal', + activityTerminalPortals: [], + measurableBackgroundWorktreeIdsRef: { current: new Set() }, + measuringTerminalWorktreeIdsRef: { current: new Set() }, + mountedWorktreeIdsRef: { current: new Set(['repo::/worktree-0']) }, + pairedRuntimeParkingEnvironmentIds: new Set(), + pendingStartupByTabId: {}, + renderedActiveWorktreeId: 'repo::/worktree-0', + tabsByWorktree: {}, + terminalParkingEnabled: true, + terminalSshParkingEnabled: true, + terminalWorktreeHiddenSinceRef: { current: hiddenSince }, + terminalWorktreeParkCooldownUntilRef: { current: new Map() }, + terminalWorktreeParkingTimersRef: { current: new Map() }, + workspaceSurfaces: surfaces, + workspaceSurfaceIds: ids, + workspaceSurfaceIdSet: new Set(ids) + } as unknown as TerminalParkingFoundation + + const pass = collectTerminalParkingPassCandidates(controller) + + // Pre-fix this allocated a 423-element id array and a 423-entry Set per fire. + expect(mapCalls()).toBe(0) + expect(pass.retentionCandidates.map((candidate) => candidate.worktreeId)).toEqual([ + 'repo::/worktree-0' + ]) + expect(hiddenSince.has('repo::/stale')).toBe(false) + }) + it('stops idle worktree writes from re-firing surface-keyed terminal effects', () => { + const repo = makeRepo() + const worktrees = Array.from({ length: 20 }, (_, index) => + makeWorktree(`wt-${index}`, `Workspace ${index}`) + ) + useAppStore.setState({ worktreesByRepo: { [repo.id]: worktrees } }) + + const fires = { surfaceKeyed: 0, idKeyed: 0 } + renderHook(() => { + const foundation = useTerminalWorkspaceFoundation() + useEffect(() => { + fires.surfaceKeyed += 1 + }, [foundation.workspaceSurfaces]) + useEffect(() => { + fires.idKeyed += 1 + }, [foundation.workspaceSurfaceIds]) + }) + expect(fires).toEqual({ surfaceKeyed: 1, idKeyed: 1 }) + + // 100 worktree writes that touch no workspace id — the idle shape (activity + // timestamps, status refreshes) that dominates a large profile. + for (let write = 0; write < 100; write += 1) { + act(() => { + useAppStore.setState({ + worktreesByRepo: { + [repo.id]: worktrees.map((worktree) => ({ ...worktree, lastActivityAt: write })) + } + }) + }) + } + + expect(fires.surfaceKeyed).toBe(101) + expect(fires.idKeyed).toBe(1) + }) +}) diff --git a/src/renderer/src/components/use-terminal-browser-retention.ts b/src/renderer/src/components/use-terminal-browser-retention.ts index cbc2dff078c..34d3b0e1a80 100644 --- a/src/renderer/src/components/use-terminal-browser-retention.ts +++ b/src/renderer/src/components/use-terminal-browser-retention.ts @@ -20,7 +20,8 @@ export function useTerminalBrowserRetention(controller: TerminalParkingFoundatio mountedWorktreeIdsRef, renderedActiveWorktreeId, setBrowserGuestRetentionRevision, - workspaceSurfaces + workspaceSurfaceIds, + workspaceSurfaceIdSet } = controller useEffect(() => { @@ -42,9 +43,8 @@ export function useTerminalBrowserRetention(controller: TerminalParkingFoundatio } const recency = browserGuestWorktreeRecencyRef.current touchBrowserGuestWorktreeRecency(recency, renderedActiveWorktreeId) - const surfaceIds = new Set(workspaceSurfaces.map((workspace) => workspace.id)) for (let index = recency.length - 1; index >= 0; index--) { - if (!surfaceIds.has(recency[index])) { + if (!workspaceSurfaceIdSet.has(recency[index])) { recency.splice(index, 1) } } @@ -55,7 +55,7 @@ export function useTerminalBrowserRetention(controller: TerminalParkingFoundatio const recencyIds = new Set(recency) const orderedWorktreeIds = [ ...recency, - ...workspaceSurfaces.map((workspace) => workspace.id).filter((id) => !recencyIds.has(id)) + ...workspaceSurfaceIds.filter((id) => !recencyIds.has(id)) ] const evictedWorktreeIds = selectBrowserGuestEvictionWorktreeIds({ orderedWorktreeIds, @@ -82,7 +82,7 @@ export function useTerminalBrowserRetention(controller: TerminalParkingFoundatio // oxlint-disable-next-line react-hooks/exhaustive-deps -- controller refs preserve their original stable identities. }, [ renderedActiveWorktreeId, - workspaceSurfaces, + workspaceSurfaceIds, browserGuestRetentionBudgetEnabled, browserGuestRetentionRevision ]) diff --git a/src/renderer/src/components/use-terminal-parking-pass.ts b/src/renderer/src/components/use-terminal-parking-pass.ts index 01733c3c09c..1ce5eab03ff 100644 --- a/src/renderer/src/components/use-terminal-parking-pass.ts +++ b/src/renderer/src/components/use-terminal-parking-pass.ts @@ -41,7 +41,7 @@ export function useTerminalParkingPass(controller: TerminalParkingFoundation): v terminalProviderSnapshotCapabilityRevision, terminalRetentionBudgetEnabled, terminalSshParkingEnabled, - workspaceSurfaces + workspaceSurfaceIds } = controller useEffect(() => { @@ -182,6 +182,6 @@ export function useTerminalParkingPass(controller: TerminalParkingFoundation): v terminalProviderSnapshotCapabilityRevision, terminalRetentionBudgetEnabled, terminalSshParkingEnabled, - workspaceSurfaces + workspaceSurfaceIds ]) } diff --git a/src/renderer/src/components/use-terminal-watcher-effects.ts b/src/renderer/src/components/use-terminal-watcher-effects.ts index 9b80e964747..9e82b821c31 100644 --- a/src/renderer/src/components/use-terminal-watcher-effects.ts +++ b/src/renderer/src/components/use-terminal-watcher-effects.ts @@ -5,8 +5,9 @@ import { canWatcherCoverParkedTerminalTab, disposeAllParkedTerminalWatchers, pruneParkedTerminalWatchers, - syncParkedTerminalTabWatchers, - terminalWatcherLiveWorkspaceIds + syncParkedTerminalTabWatchersForWorkspaces, + terminalWatcherLiveWorkspaceIds, + type ParkedTerminalTabWatcherSyncEntry } from './terminal-pane/terminal-parked-tab-watchers' import { useAppStore } from '@/store' import { gateWorktreeAgentActivation } from '@/lib/worktree-agent-activation-gate' @@ -40,36 +41,35 @@ export function useTerminalWatcherEffects(controller: TerminalColdActivationCont terminalStartupRestorationReady, terminalTitleSnapshotAuthorityEnabled, workspaceSessionReady, - workspaceSurfaces + workspaceSurfaceIds } = controller useEffect(() => { - pruneParkedTerminalWatchers( - terminalWatcherLiveWorkspaceIds(workspaceSurfaces.map((workspace) => workspace.id)) - ) - for (const workspace of workspaceSurfaces) { + pruneParkedTerminalWatchers(terminalWatcherLiveWorkspaceIds(workspaceSurfaceIds)) + const syncEntriesByWorktreeId = new Map() + for (const workspaceId of workspaceSurfaceIds) { if ( anyMountedWorktreeHasLayout && - mountedWorktreeIdsRef.current.has(workspace.id) && - getEffectiveLayoutForWorktree(workspace.id) + mountedWorktreeIdsRef.current.has(workspaceId) && + getEffectiveLayoutForWorktree(workspaceId) ) { continue } - const tabs = tabsByWorktree[workspace.id] ?? [] + const tabs = tabsByWorktree[workspaceId] ?? [] const parkedTabIds = new Set() let deferredTabIds: ReadonlySet | null = null - if (!anyMountedWorktreeHasLayout && mountedWorktreeIdsRef.current.has(workspace.id)) { - const isVisible = activeView === 'terminal' && workspace.id === renderedActiveWorktreeId + if (!anyMountedWorktreeHasLayout && mountedWorktreeIdsRef.current.has(workspaceId)) { + const isVisible = activeView === 'terminal' && workspaceId === renderedActiveWorktreeId const shouldMeasureHiddenWorktree = - !isVisible && measurableBackgroundWorktreeIdsRef.current.has(workspace.id) + !isVisible && measurableBackgroundWorktreeIdsRef.current.has(workspaceId) const parked = !isVisible && !shouldMeasureHiddenWorktree && - effectiveParkedTerminalWorktreeIds.has(workspace.id) + effectiveParkedTerminalWorktreeIds.has(workspaceId) if (parked) { for (const tab of tabs) { const activityTerminalPortal = findActivityTerminalPortal(activityTerminalPortals, { - worktreeId: workspace.id, + worktreeId: workspaceId, tabId: tab.id }) if (!activityTerminalPortal && !evictionExemptTerminalTabIds.has(tab.id)) { @@ -77,15 +77,14 @@ export function useTerminalWatcherEffects(controller: TerminalColdActivationCont } } } - deferredTabIds = - activationDeferredMountTabIdsByWorktreeRef.current.get(workspace.id) ?? null + deferredTabIds = activationDeferredMountTabIdsByWorktreeRef.current.get(workspaceId) ?? null for (const tab of tabs) { if ( deferredTabIds?.has(tab.id) && !parkedTabIds.has(tab.id) && - canWatcherCoverParkedTerminalTab(workspace.id, tab) && + canWatcherCoverParkedTerminalTab(workspaceId, tab) && !findActivityTerminalPortal(activityTerminalPortals, { - worktreeId: workspace.id, + worktreeId: workspaceId, tabId: tab.id }) ) { @@ -93,13 +92,13 @@ export function useTerminalWatcherEffects(controller: TerminalColdActivationCont } } } - syncParkedTerminalTabWatchers({ - worktreeId: workspace.id, + syncEntriesByWorktreeId.set(workspaceId, { tabs, parkedTabIds, ...(deferredTabIds ? { restoreTitleOnStartTabIds: deferredTabIds } : {}) }) } + syncParkedTerminalTabWatchersForWorkspaces(syncEntriesByWorktreeId) // oxlint-disable-next-line react-hooks/exhaustive-deps -- controller refs preserve their original stable identities. }, [ activeTabId, @@ -118,7 +117,7 @@ export function useTerminalWatcherEffects(controller: TerminalColdActivationCont terminalParkingEnabled, terminalTitleSnapshotAuthorityEnabled, workspaceSessionReady, - workspaceSurfaces + workspaceSurfaceIds ]) useEffect(() => () => disposeAllParkedTerminalWatchers(), []) diff --git a/src/renderer/src/components/use-terminal-workspace-foundation.ts b/src/renderer/src/components/use-terminal-workspace-foundation.ts index 9b365e9d885..61726fa9643 100644 --- a/src/renderer/src/components/use-terminal-workspace-foundation.ts +++ b/src/renderer/src/components/use-terminal-workspace-foundation.ts @@ -6,6 +6,7 @@ import { useWorktreeMap } from '../store/selectors' import { getResolvedExecutionHostIdForWorktree } from '@/lib/resolved-worktree-execution-host' import type { WorktreeTabBucketProjection } from '@/lib/worktree-tab-bucket-projection' import { projectWorkspaceSurfaces } from './workspace-surface-projection' +import { useReusedArrayIdentity } from './sidebar/worktree-list/listing/use-reused-array-identity' import { selectPairedRuntimeParkingEnvironmentIds } from './terminal-pane/terminal-hidden-view-parking' import { createTerminalWorktreeTopologyProjection } from './terminal-pane/terminal-hidden-worktree-retention' import { isMainTerminalSideEffectAuthorityForPty } from './terminal-pane/terminal-side-effect-facts-handler' @@ -46,6 +47,17 @@ export function useTerminalWorkspaceFoundation() { }), [worktreesById, folderWorkspaces, renderedActiveWorktreeId, activeFolderSurfaceHostId] ) + // Why split the ids out: every mount/park/activation pass reads only `.id`, but + // the surface array is re-identified on any worktree write. Reusing the previous + // id-array identity keeps those effects and their per-fire Sets from re-firing + // when the workspace set itself did not change. + const workspaceSurfaceIds = useReusedArrayIdentity( + useMemo(() => workspaceSurfaces.map((workspace) => workspace.id), [workspaceSurfaces]) + ) + const workspaceSurfaceIdSet = useMemo>( + () => new Set(workspaceSurfaceIds), + [workspaceSurfaceIds] + ) const activeView = useAppStore((state) => state.activeView) // Why: terminal titles are leaf chrome. The root host only subscribes to // mount/parking semantics; a real transition publishes fresh tab objects, @@ -88,6 +100,8 @@ export function useTerminalWorkspaceFoundation() { terminalWorktreeParkingTimersRef, folderWorkspaces, workspaceSurfaces, + workspaceSurfaceIds, + workspaceSurfaceIdSet, activeWorktreeId, renderedActiveWorktreeId, activeWorktreeDeferralHostId, From 084dbbc3b3d0b29e5edc8b7a0a344d9d0508bd01 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:05:16 -0700 Subject: [PATCH 080/398] perf(persistence): build the state file once per save instead of seven times (#18161) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every debounced save stringified the full persisted state, then ran two `String.replace` passes per secret sentinel — one for the on-disk payload, one for the guard hash. Each replace returns a rope the next one has to flatten before it can search, so three sentinels cost seven flattened copies of a 4.65 MB state (a two-byte V8 string, ~8.9 MB each), and the state was then UTF-8 encoded twice more: once inside `sha1.update(string)` and again inside `handle.writeFile(payload, 'utf-8')`. `applySecretSentinelSubstitutions` walks the state once with a single alternation regex, encodes each literal run to a Buffer exactly once, and feeds those same buffers to both the payload and the hash. Measured on the author's 4.65 MB store with three live secret slots: 48.8 MB -> 17.9 MB allocated per save, 26.6 MB -> 0 of large_object_space churn, and 22.1 -> 15.1 ms (min) / 32.3 -> 16.9 ms (median) for build+hash+encode. Bytes on disk and the guard hash are proven identical to the previous loop. Separately, non-local host session partitions carried stale replicas of the `browserUrlHistory` global — 589,807 bytes, 12.7% of the file — that neither the split (which writes globals only to 'local') nor the merge (which reads them only from 'local' unless local has none) can ever reach. The load path now drops them when the local slice already holds the field. Only the two history globals are dropped: the rest are read out of every partition by the worktree ownership sweep or the mobile/runtime projections. --- src/main/durable-file-write.ts | 9 +- .../normalize-loaded-profile-state.ts | 11 +- .../normalize-loaded-state-collections.ts | 4 +- .../loading-store/primary-state-writes.ts | 3 +- .../secret-sentinel-substitution.test.ts | 184 ++++++++++++++++++ .../secret-sentinel-substitution.ts | 78 ++++++++ .../state-serialization-secret-handling.ts | 33 ++-- .../state-write-round-trip.test.ts | 131 +++++++++++++ .../workspace-session-partitions.test.ts | 141 ++++++++++++++ .../workspace-session-partitions.ts | 41 +++- .../lib/workspace-session-host-contention.ts | 2 +- .../lib/workspace-session-host-split.test.ts | 43 ++++ .../src/lib/workspace-session-host-split.ts | 2 +- .../workspace-session-host-field-ownership.ts | 2 +- 14 files changed, 656 insertions(+), 28 deletions(-) create mode 100644 src/main/persistence/loading-store/secret-sentinel-substitution.test.ts create mode 100644 src/main/persistence/loading-store/secret-sentinel-substitution.ts create mode 100644 src/main/persistence/loading-store/state-write-round-trip.test.ts create mode 100644 src/main/persistence/loading-store/workspace-session-partitions.test.ts rename src/{renderer/src/lib => shared}/workspace-session-host-field-ownership.ts (97%) diff --git a/src/main/durable-file-write.ts b/src/main/durable-file-write.ts index 2f40b0881e8..83b0eabc5ed 100644 --- a/src/main/durable-file-write.ts +++ b/src/main/durable-file-write.ts @@ -187,10 +187,15 @@ export async function removeStaleDurableWriteTempFiles( } /** Synchronous counterpart for quit and crash paths that cannot await. */ -export function writeFileDurableSync(tmpPath: string, finalPath: string, payload: string): void { +export function writeFileDurableSync( + tmpPath: string, + finalPath: string, + payload: string | Uint8Array +): void { let renamed = false try { - writeFileSync(tmpPath, payload, 'utf-8') + // A Uint8Array payload is written verbatim; a string still defaults to UTF-8. + writeFileSync(tmpPath, payload) const fd = openSync(tmpPath, 'r+') try { fsyncSync(fd) diff --git a/src/main/persistence/loading-store/normalize-loaded-profile-state.ts b/src/main/persistence/loading-store/normalize-loaded-profile-state.ts index 2286e40d54b..39d625c525a 100644 --- a/src/main/persistence/loading-store/normalize-loaded-profile-state.ts +++ b/src/main/persistence/loading-store/normalize-loaded-profile-state.ts @@ -35,6 +35,8 @@ export function normalizeLoadedProfileState( const { defaults, migratedExternalVisibility, osc52ClipboardNoticePending } = terminal const { normalizedOnboarding, normalizedProjectGroups, loadedCompactWorktreeCards } = profile const projectCatalog = normalizeLoadedProjectCatalog(parsed, markNeedsSave) + // Ordered: the host partitions drop the global fields this slice already owns. + const workspaceSession = normalizeLoadedLocalSession(parsed, defaults, markNeedsSave) return { ...defaults, @@ -69,9 +71,14 @@ export function normalizeLoadedProfileState( markNeedsSave ), // Why: volatile schema; zod-validate workspaceSession at read so a bad payload falls to defaults, not a renderer crash. - workspaceSession: normalizeLoadedLocalSession(parsed, defaults, markNeedsSave), + workspaceSession, // Why: per-host session partitions, validated independently; 'local' stays in workspaceSession for downgrade compat. - workspaceSessionsByHostId: normalizeLoadedHostSessions(parsed, defaults, markNeedsSave), + workspaceSessionsByHostId: normalizeLoadedHostSessions( + parsed, + defaults, + workspaceSession, + markNeedsSave + ), sshTargets: (parsed.sshTargets ?? []).map(normalizeSshTarget), deletedSshConfigAliases: Array.isArray(parsed.deletedSshConfigAliases) ? parsed.deletedSshConfigAliases.filter((alias): alias is string => typeof alias === 'string') diff --git a/src/main/persistence/loading-store/normalize-loaded-state-collections.ts b/src/main/persistence/loading-store/normalize-loaded-state-collections.ts index 2b133d486b7..16c5b424def 100644 --- a/src/main/persistence/loading-store/normalize-loaded-state-collections.ts +++ b/src/main/persistence/loading-store/normalize-loaded-state-collections.ts @@ -41,11 +41,13 @@ export function normalizeLoadedLocalSession( export function normalizeLoadedHostSessions( parsed: PersistedState, defaults: PersistedState, + localSession: WorkspaceSessionState, markNeedsSave: () => void ): PersistedState['workspaceSessionsByHostId'] { const { partitions, repaired } = parseWorkspaceSessionsByHostId( parsed.workspaceSessionsByHostId, - defaults.workspaceSession + defaults.workspaceSession, + localSession ) if (repaired) { // Why: salvage repairs only the in-memory partitions; without a save the corrupt entries stay on disk and get re-dropped every launch. diff --git a/src/main/persistence/loading-store/primary-state-writes.ts b/src/main/persistence/loading-store/primary-state-writes.ts index 9d6e476ebe5..c112b4a9ba0 100644 --- a/src/main/persistence/loading-store/primary-state-writes.ts +++ b/src/main/persistence/loading-store/primary-state-writes.ts @@ -161,7 +161,8 @@ export async function writeToDiskAsync(owner: PrimaryStateWriteOperations): Prom // Why: fsync before rename, then fsync the directory; see writeFileDurable. const handle = await open(tmpFile, 'w') try { - await handle.writeFile(payload, 'utf-8') + // Already UTF-8 bytes: passing the string here would re-encode the whole state on the main thread. + await handle.writeFile(payload) await handle.sync() } finally { await handle.close() diff --git a/src/main/persistence/loading-store/secret-sentinel-substitution.test.ts b/src/main/persistence/loading-store/secret-sentinel-substitution.test.ts new file mode 100644 index 00000000000..e58d0280bae --- /dev/null +++ b/src/main/persistence/loading-store/secret-sentinel-substitution.test.ts @@ -0,0 +1,184 @@ +/** + * The bar for this change is "the bytes on disk did not move". Every case below runs the exact + * loop `applySecretSentinelSubstitutions` replaced — reproduced in `previousImplementation` — and + * compares payload bytes and guard hash, because a drifting hash silently disables the no-op write + * guard and a drifting payload is corrupted persisted state. + */ +import { createHash, randomUUID } from 'node:crypto' +import { describe, expect, it } from 'vitest' +import { + applySecretSentinelSubstitutions, + type SecretSentinelSubstitution +} from './secret-sentinel-substitution' + +/** Verbatim from state-serialization-secret-handling.ts before this change. */ +function previousImplementation( + serialized: string, + secretSubs: readonly SecretSentinelSubstitution[], + degradedPrefix: string +): { payload: Buffer; stateHash: string } { + let payload = serialized + let hashInput = serialized + for (const { sentinel, blob, hashValue } of secretSubs) { + const escapedSentinel = JSON.stringify(sentinel).slice(1, -1) + payload = payload.replace(escapedSentinel, () => JSON.stringify(blob).slice(1, -1)) + hashInput = hashInput.replace(escapedSentinel, () => JSON.stringify(hashValue).slice(1, -1)) + } + const stateHash = createHash('sha1').update(degradedPrefix).update(hashInput).digest('hex') + // `handle.writeFile(payload, 'utf-8')` is what turned the string into bytes. + return { payload: Buffer.from(payload, 'utf8'), stateHash } +} + +function expectIdenticalToPrevious( + serialized: string, + subs: readonly SecretSentinelSubstitution[], + degradedPrefix = '' +): void { + const before = previousImplementation(serialized, subs, degradedPrefix) + const after = applySecretSentinelSubstitutions(serialized, subs, degradedPrefix) + expect(after.payload.equals(before.payload)).toBe(true) + expect(after.stateHash).toBe(before.stateHash) +} + +function sentinel(): string { + return `orca-secret-slot-${randomUUID()}` +} + +describe('applySecretSentinelSubstitutions', () => { + it('produces bytes and a hash identical to the previous implementation', () => { + const subs: SecretSentinelSubstitution[] = [ + { sentinel: sentinel(), blob: 'djEwY2lwaGVy', hashValue: 'cookie-value' }, + { + sentinel: sentinel(), + // Regex-special *and* JSON-escapable, which is the pair that breaks a naive rewrite: + // `$&` would splice the match back in under string-form replace, and the backslash and + // quote have to survive `JSON.stringify(...).slice(1, -1)` unchanged. + blob: 'A+/=$&$1$`\\x "quoted" |.*?[](){}^', + hashValue: 'http://proxy.example:8080/?a=b&c=$&' + }, + { sentinel: sentinel(), blob: '', hashValue: 'https://kagi.com/session?t=abc' } + ] + const state = { + settings: { opencodeSessionCookie: subs[0].sentinel, httpProxyUrl: subs[1].sentinel }, + ui: { browserKagiSessionLink: subs[2].sentinel }, + // Adjacent content that must not shift: a near-miss prefix, and JSON escapes either side. + noise: ['orca-secret-slot-', 'a\\b"c\n\t', subs[0].sentinel.slice(0, -1)] + } + expectIdenticalToPrevious(JSON.stringify(state), subs) + }) + + it('stays identical when the state holds multi-byte and escaped characters', () => { + const subs: SecretSentinelSubstitution[] = [ + { sentinel: sentinel(), blob: 'blob-é', hashValue: 'plain-é' }, + { sentinel: sentinel(), blob: '😀', hashValue: '中文' } + ] + const state = { + // Segment boundaries land next to these, so a wrong split would corrupt the encode. + before: 'é中文😀', + a: subs[0].sentinel, + between: '😀

', + b: subs[1].sentinel, + after: '😀' + } + expectIdenticalToPrevious(JSON.stringify(state), subs) + }) + + it('stays identical with no substitutions and with the degraded-storage prefix', () => { + const state = JSON.stringify({ settings: { httpProxyUrl: '' }, big: 'x'.repeat(4096) }) + expectIdenticalToPrevious(state, []) + expectIdenticalToPrevious(state, [], 'safeStorage-degraded\0') + + const subs = [{ sentinel: sentinel(), blob: 'b', hashValue: 'h' }] + expectIdenticalToPrevious( + JSON.stringify({ s: subs[0].sentinel }), + subs, + 'safeStorage-degraded\0' + ) + }) + + it('escapes regex metacharacters in the sentinel itself', () => { + // Not reachable from a UUID sentinel, but the alternation must not be able to become a pattern. + const subs = [{ sentinel: 'a.b*c(d)|e[f]', blob: 'BLOB', hashValue: 'HASH' }] + const serialized = JSON.stringify({ real: subs[0].sentinel, decoy: 'axbxxcXdX_eXfX' }) + expectIdenticalToPrevious(serialized, subs) + expect( + applySecretSentinelSubstitutions(serialized, subs, '').payload.toString('utf8') + ).toContain('axbxxcXdX_eXfX') + }) + + it('substitutes every occurrence when a sentinel repeats', () => { + // Cannot happen today (a sentinel is a UUID minted after the state is assembled, so it appears + // exactly once), but the old first-match-only `String.replace` would have written a raw + // sentinel to disk in place of a secret if it ever did. The alternation is global instead. + const subs = [{ sentinel: sentinel(), blob: 'CIPHER', hashValue: 'PLAIN' }] + const serialized = JSON.stringify({ a: subs[0].sentinel, b: subs[0].sentinel }) + const { payload } = applySecretSentinelSubstitutions(serialized, subs, '') + expect(payload.toString('utf8')).toBe(JSON.stringify({ a: 'CIPHER', b: 'CIPHER' })) + expect(payload.toString('utf8')).not.toContain(subs[0].sentinel) + }) + + it('copies and UTF-8 encodes the full state once, not once per sentinel per side', () => { + const subs: SecretSentinelSubstitution[] = Array.from({ length: 3 }, () => ({ + sentinel: sentinel(), + blob: 'CIPHERTEXT', + hashValue: 'plaintext' + })) + const serialized = JSON.stringify({ + pad: 'x'.repeat(200_000), + a: subs[0].sentinel, + b: subs[1].sentinel, + c: subs[2].sentinel + }) + const FULL_STATE = 100_000 + + // Both costs are observable at their sources: a `String.replace` whose receiver is the whole + // state allocates another copy of it, and every string handed to `Buffer.from` or `hash.update` + // is one full UTF-8 encode pass on the main thread. + const counted = (run: () => unknown): { fullStateReplaces: number; encodedChars: number } => { + const realReplace = String.prototype.replace + const realBufferFrom = Buffer.from + const hashProto = Object.getPrototypeOf(createHash('sha1')) as { + update: (...args: unknown[]) => unknown + } + const realUpdate = hashProto.update + const counts = { fullStateReplaces: 0, encodedChars: 0 } + String.prototype.replace = function (this: string, ...args: unknown[]) { + if (this.length >= FULL_STATE) { + counts.fullStateReplaces++ + } + return realReplace.apply(this, args as never) + } as typeof String.prototype.replace + Buffer.from = function (...args: unknown[]) { + if (typeof args[0] === 'string') { + counts.encodedChars += args[0].length + } + return (realBufferFrom as (...a: unknown[]) => Buffer).apply(Buffer, args) + } as typeof Buffer.from + hashProto.update = function (this: unknown, ...args: unknown[]) { + if (typeof args[0] === 'string') { + counts.encodedChars += args[0].length + } + return realUpdate.apply(this, args) + } + try { + run() + } finally { + String.prototype.replace = realReplace + Buffer.from = realBufferFrom + hashProto.update = realUpdate + } + return counts + } + + const before = counted(() => previousImplementation(serialized, subs, '')) + const after = counted(() => applySecretSentinelSubstitutions(serialized, subs, '')) + + // Two `String.replace` calls over the whole state per sentinel — payload and hash input. + expect(before.fullStateReplaces).toBe(subs.length * 2) + expect(after.fullStateReplaces).toBe(0) + // The old path encoded the state twice: once for sha1, once for the file write. + expect(before.encodedChars).toBeGreaterThan(serialized.length * 1.9) + expect(after.encodedChars).toBeLessThan(serialized.length * 1.1) + expect(after.encodedChars).toBeGreaterThan(serialized.length * 0.9) + }) +}) diff --git a/src/main/persistence/loading-store/secret-sentinel-substitution.ts b/src/main/persistence/loading-store/secret-sentinel-substitution.ts new file mode 100644 index 00000000000..afcdddafa77 --- /dev/null +++ b/src/main/persistence/loading-store/secret-sentinel-substitution.ts @@ -0,0 +1,78 @@ +import { createHash } from 'node:crypto' +import { escapeRegex } from '../../../shared/string-utils' + +export type SecretSentinelSubstitution = { + /** The `orca-secret-slot-` placeholder standing in the serialized state. */ + sentinel: string + /** What the on-disk payload gets: the ciphertext. */ + blob: string + /** What the guard hash gets: a value stable across non-deterministic encryption. */ + hashValue: string +} + +/** + * Replace every secret sentinel in `serialized` in ONE pass, producing the on-disk bytes and the + * guard hash from the same encoded segments. + * + * Why not the obvious `payload.replace(...)` / `hashInput.replace(...)` loop it replaces: each + * `String.replace` returns a rope that the *next* `replace` has to flatten before it can search, so + * N sentinels cost 2N-1 flattened copies of the whole multi-MB state, plus one more per side when + * `hash.update` and the file write finally consume them. Measured on a 4.65 MB store with three + * sentinels: 7 full-state string allocations, 62 MB of V8 heap, 27 MB of it in large_object_space. + * + * Here the state is walked once, each literal run is UTF-8 encoded exactly once, and those same + * buffers feed both the payload and the hash — 1 full-state string, 1 encode. + * + * Byte-for-byte identical output to the loop: both sides read the sentinel in its JSON-escaped + * form, the replacements are the JSON-escaped `blob`/`hashValue`, and the hash sees the same byte + * sequence it saw when it was handed one concatenated string. + */ +export function applySecretSentinelSubstitutions( + serialized: string, + substitutions: readonly SecretSentinelSubstitution[], + degradedPrefix: string +): { payload: Buffer; stateHash: string } { + const hash = createHash('sha1').update(degradedPrefix) + if (substitutions.length === 0) { + const payload = Buffer.from(serialized, 'utf8') + return { payload, stateHash: hash.update(payload).digest('hex') } + } + + const replacementBySentinel = new Map() + const alternatives: string[] = [] + for (const { sentinel, blob, hashValue } of substitutions) { + // Preserved from the loop this replaces: both the search key and the replacements are the + // JSON-escaped forms, because that is what `serialized` actually contains. + const escapedSentinel = JSON.stringify(sentinel).slice(1, -1) + if (replacementBySentinel.has(escapedSentinel)) { + continue + } + alternatives.push(escapeRegex(escapedSentinel)) + replacementBySentinel.set(escapedSentinel, { + blob: Buffer.from(JSON.stringify(blob).slice(1, -1), 'utf8'), + hashValue: Buffer.from(JSON.stringify(hashValue).slice(1, -1), 'utf8') + }) + } + + // Global, though a sentinel is a UUID minted after the state was assembled and so occurs exactly + // once: a single pass that substitutes every occurrence cannot leave one behind on disk. + const pattern = new RegExp(alternatives.join('|'), 'g') + const chunks: Buffer[] = [] + let cursor = 0 + let match: RegExpExecArray | null + while ((match = pattern.exec(serialized)) !== null) { + // Non-null: the alternation is built from exactly the map's keys. + const replacement = replacementBySentinel.get(match[0])! + // A sliced substring, so this does not copy the state; the encode below is its only pass. + const literal = Buffer.from(serialized.slice(cursor, match.index), 'utf8') + chunks.push(literal, replacement.blob) + hash.update(literal) + hash.update(replacement.hashValue) + cursor = match.index + match[0].length + } + const tail = Buffer.from(serialized.slice(cursor), 'utf8') + chunks.push(tail) + hash.update(tail) + + return { payload: Buffer.concat(chunks), stateHash: hash.digest('hex') } +} diff --git a/src/main/persistence/loading-store/state-serialization-secret-handling.ts b/src/main/persistence/loading-store/state-serialization-secret-handling.ts index c9d81f071a5..61689ea0f13 100644 --- a/src/main/persistence/loading-store/state-serialization-secret-handling.ts +++ b/src/main/persistence/loading-store/state-serialization-secret-handling.ts @@ -1,4 +1,4 @@ -import { createHash, randomUUID } from 'node:crypto' +import { randomUUID } from 'node:crypto' import type { PersistedState } from '../../../shared/persisted-state-types' import { collectFolderWorkspaceDiffComments } from '../../folder-workspace-diff-comments' import { @@ -8,6 +8,10 @@ import { } from '../../protected-secret-persistence' import { stripRetiredGlobalSettings } from '../applying-settings/terminal-settings-migrations' +import { + applySecretSentinelSubstitutions, + type SecretSentinelSubstitution +} from './secret-sentinel-substitution' import type { StoreRuntimeState } from './store-runtime-state' type StateSerializationSecretHandlingOperationsRuntime = Pick< @@ -24,7 +28,7 @@ export class StateSerializationSecretHandlingOperations { } buildStateToSave(): { - payload: string + payload: Buffer stateHash: string protectedSecretUpdates: ProtectedSecretRetentionUpdate[] } { @@ -37,7 +41,7 @@ export class StateSerializationSecretHandlingOperations { // on deterministic-IV platforms (macOS/legacy-Linux OSCrypt). A per-slot // random UUID can't occur anywhere else in the serialized state (the user // sets their data before it is minted), so it appears exactly once. - const secretSubs: { sentinel: string; blob: string; hashValue: string }[] = [] + const secretSubs: SecretSentinelSubstitution[] = [] const protectedSecretUpdates: ProtectedSecretRetentionUpdate[] = [] let protectedStorageDegraded = false const encryptToSentinel = (slot: string, plaintext: string): string => { @@ -105,21 +109,14 @@ export class StateSerializationSecretHandlingOperations { // Why compact: ~20% fewer bytes and less serialize time; all readers JSON.parse so formatting is irrelevant. // One full-state stringify; secret slots currently hold sentinels. const serialized = JSON.stringify(stateToSave) - // Substitute each unique sentinel exactly once: ciphertext for the on-disk - // payload, a stable normalized value for the guard hash. Function-form - // replacement keeps `$` inert; both sides read the sentinel as JSON-escaped - // in `serialized`, so each replace is byte-for-byte position-exact. - let payload = serialized - let hashInput = serialized - for (const { sentinel, blob, hashValue } of secretSubs) { - const escapedSentinel = JSON.stringify(sentinel).slice(1, -1) - payload = payload.replace(escapedSentinel, () => JSON.stringify(blob).slice(1, -1)) - hashInput = hashInput.replace(escapedSentinel, () => JSON.stringify(hashValue).slice(1, -1)) - } - const stateHash = createHash('sha1') - .update(protectedStorageDegraded ? 'safeStorage-degraded\0' : '') - .update(hashInput) - .digest('hex') + // Substitute each unique sentinel: ciphertext for the on-disk payload, a stable normalized + // value for the guard hash. One pass builds both, so the multi-MB state is never copied per + // sentinel and never encoded twice. + const { payload, stateHash } = applySecretSentinelSubstitutions( + serialized, + secretSubs, + protectedStorageDegraded ? 'safeStorage-degraded\0' : '' + ) return { payload, stateHash, protectedSecretUpdates } } } diff --git a/src/main/persistence/loading-store/state-write-round-trip.test.ts b/src/main/persistence/loading-store/state-write-round-trip.test.ts new file mode 100644 index 00000000000..ee7f26feb3e --- /dev/null +++ b/src/main/persistence/loading-store/state-write-round-trip.test.ts @@ -0,0 +1,131 @@ +/** + * The write path now hands the file a Buffer it built in one pass instead of a string it rebuilt + * per secret. Drives the real `Store` end to end — encrypted settings, a local session and a remote + * host partition — and reloads from the file it actually wrote, because the failure this guards + * against (a mis-sliced segment, a re-encoded payload, a dropped sentinel) is invisible until + * something reads the bytes back. + */ +import { mkdtempSync, readFileSync, realpathSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { WorkspaceSessionState } from '../../../shared/workspace-session-state-types' + +vi.mock('electron', () => ({ + app: { + getPath: () => tmpdir(), + getName: () => 'orca-test', + getVersion: () => '0.0.0-test', + isPackaged: false, + on: () => {}, + whenReady: () => Promise.resolve() + }, + safeStorage: { + // Encryption ON, so the secret slots really do mint sentinels and the substitution pass runs. + isEncryptionAvailable: () => true, + encryptString: (value: string) => Buffer.from(`enc:${value}`), + decryptString: (value: Buffer) => value.toString().slice(4) + }, + ipcMain: { on: () => {}, handle: () => {} }, + BrowserWindow: { getAllWindows: () => [] } +})) + +const { Store } = await import('./store') + +const HOST_ID = 'ssh:user@host' + +const stores: InstanceType[] = [] +afterEach(() => { + for (const store of stores.splice(0)) { + store.flush() + } + vi.restoreAllMocks() +}) + +function openStore(dataFile: string): InstanceType { + const store = new Store({ dataFile }) + stores.push(store) + return store +} + +function session(activeTabId: string): WorkspaceSessionState { + return { + activeRepoId: 'repo-1', + // Left null: the load path's deregistered-repo sweep nulls an active worktree whose repo is + // not registered, which would mask what this test is actually about. + activeWorktreeId: null, + activeTabId, + tabsByWorktree: {}, + terminalLayoutsByTabId: {}, + // Non-ASCII on purpose: a byte-offset mistake in the encode shows up here first. + browserUrlHistory: [ + { + url: 'https://example.test/é😀', + normalizedUrl: 'https://example.test/é😀', + title: '中文 title', + lastVisitedAt: 17, + visitCount: 3 + } + ] + } as WorkspaceSessionState +} + +describe('persisted state survives a save/load round trip', () => { + it('reloads settings, secrets and both session partitions unchanged', () => { + const dataFile = join( + realpathSync(mkdtempSync(join(tmpdir(), 'orca-store-round-trip-'))), + 'orca-data.json' + ) + const written = openStore(dataFile) + written.updateSettings({ + // Three secret slots, i.e. three sentinels in one save — the case the old loop paid 7 copies for. + opencodeSessionCookie: 'cookie-é-value', + httpProxyUrl: 'http://proxy.example:8080/?a=b&c=$&' + }) + written.updateUI({ browserKagiSessionLink: 'https://kagi.com/session?t=abc' }) + written.setWorkspaceSession(session('local-tab')) + written.setWorkspaceSession(session('remote-tab'), HOST_ID) + written.flush() + + const before = { + settings: written.getSettings(), + ui: written.getUI(), + local: written.getWorkspaceSession(), + remote: written.getWorkspaceSession(HOST_ID) + } + + // The file is valid UTF-8 JSON and holds ciphertext, not the plaintext secrets. + const bytes = readFileSync(dataFile) + const onDisk = JSON.parse(bytes.toString('utf8')) + expect(onDisk.settings.opencodeSessionCookie).not.toBe('cookie-é-value') + expect(Buffer.from(onDisk.settings.opencodeSessionCookie, 'base64').toString('utf8')).toContain( + 'cookie-é-value' + ) + expect(bytes.toString('utf8')).not.toContain('orca-secret-slot-') + + const reloaded = openStore(dataFile) + expect(reloaded.getSettings().opencodeSessionCookie).toBe(before.settings.opencodeSessionCookie) + expect(reloaded.getSettings().httpProxyUrl).toBe(before.settings.httpProxyUrl) + expect(reloaded.getUI().browserKagiSessionLink).toBe(before.ui.browserKagiSessionLink) + // `toMatchObject`: the load path spreads session defaults over what was written, so the + // reloaded slice is a superset. Exact deep equality is asserted on the second trip below. + expect(reloaded.getWorkspaceSession()).toMatchObject(before.local) + // The remote partition keeps everything it owns; only globals local already holds are dropped, + // and `browserUrlHistory` comes back at its default from the same spread as before. + expect(reloaded.getWorkspaceSession(HOST_ID).activeTabId).toBe('remote-tab') + expect(reloaded.getWorkspaceSession(HOST_ID).browserUrlHistory).toEqual([]) + + // Deep equality of the whole reloaded state, taken across a second round trip so the assertion + // is not comparing against the first load's one-time settings migrations. + reloaded.flush() + const bytesAfterReload = readFileSync(dataFile) + const again = openStore(dataFile) + expect(again.getSettings()).toEqual(reloaded.getSettings()) + expect(again.getUI()).toEqual(reloaded.getUI()) + expect(again.getWorkspaceSession()).toEqual(reloaded.getWorkspaceSession()) + expect(again.getWorkspaceSession(HOST_ID)).toEqual(reloaded.getWorkspaceSession(HOST_ID)) + // ...and the bytes are stable, so a quiet app is not rewriting a 4 MB file with new content. + again.flush() + expect(readFileSync(dataFile).equals(bytesAfterReload)).toBe(true) + }) +}) diff --git a/src/main/persistence/loading-store/workspace-session-partitions.test.ts b/src/main/persistence/loading-store/workspace-session-partitions.test.ts new file mode 100644 index 00000000000..f9697f77997 --- /dev/null +++ b/src/main/persistence/loading-store/workspace-session-partitions.test.ts @@ -0,0 +1,141 @@ +/** + * Global session fields live in the 'local' slice. Copies of them inside a non-local host partition + * are legacy residue: the split never writes them there and the merge never reads them from there + * unless local has nothing. These tests pin the drop to exactly that condition, keep the renderer's + * merge landing on the same value either way, and re-check the two safety gates that decide which + * global fields may be dropped at all. + */ +import { describe, expect, it } from 'vitest' +import { getDefaultWorkspaceSession } from '../../../shared/constants' +import type { BrowserHistoryEntry } from '../../../shared/browser-workspace-types' +import type { WorkspaceDocHistoryEntry } from '../../../shared/workspace-doc-history' +import type { WorkspaceSessionState } from '../../../shared/workspace-session-state-types' +import { WORKSPACE_SESSION_FIELD_OWNERSHIP } from '../../../shared/workspace-session-host-field-ownership' +import { WORKSPACE_SESSION_WORKTREE_REFERENCE_KIND } from '../restoring-sessions/session-worktree-ownership' +import { + HOST_PARTITION_REDUNDANT_GLOBAL_FIELDS, + parseWorkspaceSessionsByHostId +} from './workspace-session-partitions' + +const HOST = 'ssh:target-1' + +function history(url: string): BrowserHistoryEntry[] { + return [{ url, normalizedUrl: url, title: url, lastVisitedAt: 1, visitCount: 1 }] +} + +function docEntry(filePath: string): WorkspaceDocHistoryEntry { + return { + docLocation: { kind: 'workspace-doc', worktreeId: 'repo-1::/tmp/a', filePath }, + title: filePath, + lastVisitedAt: 2, + visitCount: 1 + } +} + +function localSession(overrides: Partial): WorkspaceSessionState { + return { ...getDefaultWorkspaceSession(), ...overrides } +} + +function parse( + raw: Record, + local?: WorkspaceSessionState +): Partial> { + return parseWorkspaceSessionsByHostId(raw, getDefaultWorkspaceSession(), local).partitions +} + +describe('HOST_PARTITION_REDUNDANT_GLOBAL_FIELDS', () => { + it('only lists fields that are global AND that no worktree-ownership pass follows', () => { + for (const field of HOST_PARTITION_REDUNDANT_GLOBAL_FIELDS) { + // Gate 1: the renderer's split/merge treat it as local-owned, so a non-local copy is dead. + expect(WORKSPACE_SESSION_FIELD_OWNERSHIP[field]).toBe('global') + // Gate 2: `collectPersistedSessionWorktreeOwners` and the deregistered-repo residue sweep + // walk EVERY partition through this table. Anything but 'none' means dropping the field + // could un-own a worktree and get its metadata pruned. + expect(WORKSPACE_SESSION_WORKTREE_REFERENCE_KIND[field]).toBe('none') + } + }) +}) + +describe('parseWorkspaceSessionsByHostId global-field residue', () => { + it('drops a non-local global field the local slice already owns', () => { + const local = localSession({ browserUrlHistory: history('https://local.test') }) + const partitions = parse( + { + [HOST]: { + ...getDefaultWorkspaceSession(), + browserUrlHistory: history('https://stale.test') + } + }, + local + ) + // Back to the default from the spread, not the 65 KB stale replica. The merge reads this field + // from local whenever local has it, so the renderer still sees `https://local.test` + // (`workspace-session-host-split.test.ts` pins that half of the contract). + expect(partitions[HOST]?.browserUrlHistory).toEqual([]) + }) + + it('retains a non-local global field the local slice does NOT have', () => { + // `workspaceDocHistory` is optional and absent from the defaults, so local can genuinely lack + // it and the merge's fallback to another slice is live. + const local = localSession({}) + expect(local.workspaceDocHistory).toBeUndefined() + const docs = [docEntry('/repo/remote.md')] + const partitions = parse( + { [HOST]: { ...getDefaultWorkspaceSession(), workspaceDocHistory: docs } }, + local + ) + // Retained, so the merge's "fall back to any slice that has it" path still finds a value. + expect(partitions[HOST]?.workspaceDocHistory).toEqual(docs) + }) + + it('drops that same field once the local slice does have it', () => { + const localDocs = [docEntry('/repo/local.md')] + const local = localSession({ workspaceDocHistory: localDocs }) + const partitions = parse( + { + [HOST]: { + ...getDefaultWorkspaceSession(), + workspaceDocHistory: [docEntry('/repo/stale.md')] + } + }, + local + ) + expect(partitions[HOST]).not.toHaveProperty('workspaceDocHistory') + expect(local.workspaceDocHistory).toEqual(localDocs) + }) + + it('leaves worktree-referencing globals and host-owned fields alone', () => { + const local = localSession({ + browserUrlHistory: history('https://local.test'), + activeWorktreeId: 'repo-1::/tmp/local', + activeTabId: 'local-tab' + }) + const tabs = { 'repo-1::/tmp/a': [] } + const partitions = parse( + { + [HOST]: { + ...getDefaultWorkspaceSession(), + // A `'direct'` worktree reference the residue sweep reads out of every partition. + activeWorktreeId: 'repo-1::/tmp/a', + // Read on a partition by the mobile terminal projection. + activeTabId: 'remote-tab', + tabsByWorktree: tabs, + terminalTopologyRevisionByRepoId: { 'repo-1': 4 } + } + }, + local + ) + expect(partitions[HOST]?.activeWorktreeId).toBe('repo-1::/tmp/a') + expect(partitions[HOST]?.activeTabId).toBe('remote-tab') + expect(partitions[HOST]?.tabsByWorktree).toEqual(tabs) + expect(partitions[HOST]?.terminalTopologyRevisionByRepoId).toEqual({ 'repo-1': 4 }) + }) + + it('is a no-op when no local slice is supplied', () => { + const stale = history('https://stale.test') + const partitions = parse({ + [HOST]: { ...getDefaultWorkspaceSession(), browserUrlHistory: stale } + }) + expect(partitions[HOST]?.browserUrlHistory).toEqual(stale) + }) +}) diff --git a/src/main/persistence/loading-store/workspace-session-partitions.ts b/src/main/persistence/loading-store/workspace-session-partitions.ts index 50e9d4ae8cc..b6f78479dbf 100644 --- a/src/main/persistence/loading-store/workspace-session-partitions.ts +++ b/src/main/persistence/loading-store/workspace-session-partitions.ts @@ -17,11 +17,49 @@ export function workspaceSessionSalvageLogDetails(result: { } } +/** + * Global fields belong to the 'local' slice: the split writes them only there and the merge reads + * them only from there. A copy inside a non-local partition is legacy residue no read can reach — + * stale `browserUrlHistory` replicas alone were 589 KB, 12.7% of a 4.65 MB store, rewritten on + * every save and reparsed on every launch. + * + * Deliberately NOT every field in `GLOBAL_WORKSPACE_SESSION_FIELDS`. Two separate gates disqualify + * the rest, and both are load-bearing: + * - `activeWorktreeId` and `activeWorkspaceKey` are `'direct'` in + * `WORKSPACE_SESSION_WORKTREE_REFERENCE_KIND`, and both `collectPersistedSessionWorktreeOwners` + * and the deregistered-repo residue sweep read them out of EVERY partition. Dropping one + * un-owns a worktree, and an un-owned worktree gets its metadata pruned. + * - `activeTabId`, `activeConnectionIdsAtShutdown` and `activeRepoId` have live main-side readers + * on a partition: `isPersistedTerminalLeafActive` falls back to `activeTabId` for the mobile + * projection, and the runtime attach-window handoff unions `activeConnectionIdsAtShutdown`. + * + * `workspace-session-partitions.test.ts` re-checks both gates for every field listed here. + */ +export const HOST_PARTITION_REDUNDANT_GLOBAL_FIELDS = [ + 'browserUrlHistory', + 'workspaceDocHistory' +] as const satisfies readonly (keyof WorkspaceSessionState)[] + +/** Dropped only where local already holds the field — exactly when the merge's fallback to another + * slice cannot fire. Runs before the defaults spread, so a field the type requires comes back at + * its default rather than going missing. */ +function dropRedundantGlobalFields( + slice: Partial, + local: WorkspaceSessionState | undefined +): void { + for (const field of HOST_PARTITION_REDUNDANT_GLOBAL_FIELDS) { + if (local?.[field] !== undefined) { + delete slice[field] + } + } +} + /** Normalize non-'local' host partitions; 'local' (the legacy workspaceSession blob) is dropped so the two surfaces never diverge. * Each partition is zod-validated independently, so one corrupt host drops to defaults without taking out the others. Idempotent. */ export function parseWorkspaceSessionsByHostId( raw: unknown, - defaults: WorkspaceSessionState + defaults: WorkspaceSessionState, + localSession?: WorkspaceSessionState ): { partitions: Partial>; repaired: boolean } { if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { return { partitions: {}, repaired: raw !== undefined } @@ -50,6 +88,7 @@ export function parseWorkspaceSessionsByHostId( ) repaired = true } + dropRedundantGlobalFields(result.value, localSession) partitions[hostId] = { ...defaults, ...result.value } } return { partitions, repaired } diff --git a/src/renderer/src/lib/workspace-session-host-contention.ts b/src/renderer/src/lib/workspace-session-host-contention.ts index 51d09027200..652678e24a1 100644 --- a/src/renderer/src/lib/workspace-session-host-contention.ts +++ b/src/renderer/src/lib/workspace-session-host-contention.ts @@ -10,7 +10,7 @@ import { getWorktreeIdFromHostIdentity, isWorktreeHostIdentity } from '../../../shared/worktree/host-qualified-identity' -import { WORKSPACE_SESSION_FIELD_OWNERSHIP } from './workspace-session-host-field-ownership' +import { WORKSPACE_SESSION_FIELD_OWNERSHIP } from '../../../shared/workspace-session-host-field-ownership' import { isWorkspaceSessionRecord, type WorkspaceSessionRecord diff --git a/src/renderer/src/lib/workspace-session-host-split.test.ts b/src/renderer/src/lib/workspace-session-host-split.test.ts index de359cfb12e..ad540d2af04 100644 --- a/src/renderer/src/lib/workspace-session-host-split.test.ts +++ b/src/renderer/src/lib/workspace-session-host-split.test.ts @@ -392,3 +392,46 @@ describe('split → merge round trip', () => { expect(roundTrip(state)).toEqual(state) }) }) + +/** + * The main-process load path drops a global field from a non-local partition when the local slice + * already has it, on the strength of exactly these two rules. If either moves, that prune starts + * discarding a value the renderer would otherwise have read. + */ +describe('mergeWorkspaceSessionsFromHosts global-field precedence', () => { + const localEntry = { + url: 'local', + normalizedUrl: 'local', + title: 'l', + lastVisitedAt: 2, + visitCount: 1 + } + const hostEntry = { + url: 'host', + normalizedUrl: 'host', + title: 'h', + lastVisitedAt: 1, + visitCount: 1 + } + + it("takes a global field from 'local' whenever local has one, ignoring every other slice", () => { + const merged = mergeWorkspaceSessionsFromHosts({ + [LOCAL_EXECUTION_HOST_ID]: { + ...getDefaultWorkspaceSession(), + browserUrlHistory: [localEntry] + }, + [RUNTIME_A]: { ...getDefaultWorkspaceSession(), browserUrlHistory: [hostEntry] } + }) + expect(merged.browserUrlHistory).toEqual([localEntry]) + }) + + it('falls back to another slice only when local does not have the field', () => { + const local = getDefaultWorkspaceSession() + delete local.browserUrlHistory + const merged = mergeWorkspaceSessionsFromHosts({ + [LOCAL_EXECUTION_HOST_ID]: local, + [RUNTIME_A]: { ...getDefaultWorkspaceSession(), browserUrlHistory: [hostEntry] } + }) + expect(merged.browserUrlHistory).toEqual([hostEntry]) + }) +}) diff --git a/src/renderer/src/lib/workspace-session-host-split.ts b/src/renderer/src/lib/workspace-session-host-split.ts index 817b87eebeb..ea03cc6f3f8 100644 --- a/src/renderer/src/lib/workspace-session-host-split.ts +++ b/src/renderer/src/lib/workspace-session-host-split.ts @@ -8,7 +8,7 @@ import { isWorktreeHostIdentity } from '../../../shared/worktree/host-qualified- import { GLOBAL_WORKSPACE_SESSION_FIELDS, WORKSPACE_SESSION_FIELD_OWNERSHIP -} from './workspace-session-host-field-ownership' +} from '../../../shared/workspace-session-host-field-ownership' import { buildWorktreeIdByFileId, buildWorktreeIdByTabId, diff --git a/src/renderer/src/lib/workspace-session-host-field-ownership.ts b/src/shared/workspace-session-host-field-ownership.ts similarity index 97% rename from src/renderer/src/lib/workspace-session-host-field-ownership.ts rename to src/shared/workspace-session-host-field-ownership.ts index a44c7669bda..6397c2ad239 100644 --- a/src/renderer/src/lib/workspace-session-host-field-ownership.ts +++ b/src/shared/workspace-session-host-field-ownership.ts @@ -1,4 +1,4 @@ -import type { WorkspaceSessionState } from '../../../shared/workspace-session-state-types' +import type { WorkspaceSessionState } from './workspace-session-state-types' export type WorkspaceSessionFieldOwnership = | 'global' From ae1dab40d6a7b290a59fbbcb94ac26fcabdf7b4e Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:05:24 -0700 Subject: [PATCH 081/398] Sta 6308 add copy session id option to terminal tab context menu (#18070) * Move Copy Session ID from tab to terminal pane context menu - Relocates session ID copy to the exact pane that owns it, not the tab's active pane - Adds support for durable sleeping agent sessions as fallback for cleared live status - Generalizes copy-rejection guards to handle any identity type, not just pane IDs - Updates e2e test to verify pane-specific session ID copying * Gate session ID liveness by shell foreground state Once OSC 133;D proves a pane is back at the shell, don't return the session ID even if a durable record survived the exit. This prevents treating exited sessions as still active when the user is typing at the prompt. * Update hook order parity test for session-ID projection hook The pane session-ID projection adds a render hook to TerminalPane. Update the expected hook count from 229 to 230 and the corresponding SHA256 hash. --- .../use-native-chat-context-menu.tsx | 13 ++ .../tab-bar/SortableTabContextMenu.test.tsx | 56 ------ .../tab-bar/SortableTabContextMenu.tsx | 7 - .../TabAgentSessionIdMenuItem.test.tsx | 79 --------- .../tab-bar/TabAgentSessionIdMenuItem.tsx | 48 ------ .../tab-bar/tab-agent-session-id.test.ts | 159 ------------------ .../tab-bar/tab-agent-session-id.ts | 32 ---- .../tab-context-menu-consistency.test.tsx | 1 - .../TerminalContextMenu.test.tsx | 25 +++ .../terminal-pane/TerminalContextMenu.tsx | 15 +- .../TerminalPaneNativeChatPortal.tsx | 10 ++ .../terminal-pane/TerminalPaneSurface.tsx | 3 + .../pane-agent-session-id.test.ts | 110 ++++++++++++ .../terminal-pane/pane-agent-session-id.ts | 27 +++ .../terminal-copy-rejection-guards.ts | 10 +- .../terminal-copy-rejection-handling.test.ts | 12 +- .../terminal-pane-hook-order-parity.test.ts | 6 +- .../terminal-pane-menu-copy-actions.ts | 38 ++++- .../use-terminal-pane-context-menu.ts | 14 ++ .../use-terminal-pane-projection.ts | 10 ++ src/renderer/src/i18n/locales/en.json | 12 +- src/renderer/src/i18n/locales/es.json | 12 +- src/renderer/src/i18n/locales/ja.json | 12 +- src/renderer/src/i18n/locales/ko.json | 12 +- src/renderer/src/i18n/locales/zh.json | 12 +- ... terminal-context-menu-session-id.spec.ts} | 37 ++-- 26 files changed, 335 insertions(+), 437 deletions(-) delete mode 100644 src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx delete mode 100644 src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx delete mode 100644 src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts delete mode 100644 src/renderer/src/components/tab-bar/tab-agent-session-id.ts create mode 100644 src/renderer/src/components/terminal-pane/pane-agent-session-id.test.ts create mode 100644 src/renderer/src/components/terminal-pane/pane-agent-session-id.ts rename tests/e2e/{tab-context-menu-session-id.spec.ts => terminal-context-menu-session-id.spec.ts} (59%) diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx index 9aed41a0cdc..8e939401a47 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx @@ -56,6 +56,8 @@ export type NativeChatContextMenuActions = { onSetTitle: () => void onCopyTerminalId: () => void onCopyPaneId: () => void + canCopyAgentSessionId: boolean + onCopyAgentSessionId: () => void canClosePane: boolean onClosePane: () => void } @@ -75,6 +77,8 @@ export const emptyNativeChatContextMenuActions: Omit {}, onCopyTerminalId: () => {}, onCopyPaneId: () => {}, + canCopyAgentSessionId: false, + onCopyAgentSessionId: () => {}, canClosePane: false, onClosePane: () => {} } @@ -219,6 +223,15 @@ export function useNativeChatContextMenu({ rootRef, actions }: UseNativeChatCont 'Set Title…' )} + {actions.canCopyAgentSessionId ? ( + + + {translate( + 'components.terminalPane.TerminalContextMenu.copySessionId', + 'Copy Session ID' + )} + + ) : null} {translate( diff --git a/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx b/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx index e51c599f669..9d3e506e3f1 100644 --- a/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx +++ b/src/renderer/src/components/tab-bar/SortableTabContextMenu.test.tsx @@ -292,60 +292,4 @@ describe('SortableTabContextMenu', () => { expect(container.textContent).not.toContain('Move Tab to Split') expect(container.textContent).toContain('Split terminal right') }) - - describe('copy session id', () => { - const LEAF = '11111111-1111-4111-8111-111111111111' - - function withLiveAgent(sessionId: string | null): void { - storeMock.state = { - ...storeMock.state, - terminalLayoutsByTabId: { - 'term-1': { root: { type: 'leaf', leafId: LEAF }, activeLeafId: LEAF } - }, - agentStatusByPaneKey: { - [`term-1:${LEAF}`]: { - state: 'done', - prompt: '', - updatedAt: 1, - stateStartedAt: 1, - paneKey: `term-1:${LEAF}`, - agentType: 'claude', - stateHistory: [], - ...(sessionId ? { providerSession: { key: 'session_id', id: sessionId } } : {}) - } - }, - paneForegroundAgentByPaneKey: {} - } - } - - it('omits the item for a tab with no agent', () => { - const { container } = renderMenu() - - expect(container.textContent).not.toContain('Copy Session ID') - }) - - it('omits the item until the active agent reports a session id', () => { - withLiveAgent(null) - const { container } = renderMenu() - - expect(container.textContent).not.toContain('Copy Session ID') - }) - - it('copies the active pane session id', async () => { - const writeClipboardText = vi.fn().mockResolvedValue(undefined) - Object.assign(window, { api: { ui: { writeClipboardText } } }) - withLiveAgent('session-abc') - const { container } = renderMenu() - - act(() => getButton(container, 'Copy Session ID').click()) - await vi.waitFor(() => expect(writeClipboardText).toHaveBeenCalledWith('session-abc')) - }) - - it('does not resolve a session id while the menu is closed', () => { - withLiveAgent('session-abc') - const { container } = renderMenu({ open: false }) - - expect(container.textContent).not.toContain('Copy Session ID') - }) - }) }) diff --git a/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx b/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx index b298072f9f6..53b31012d39 100644 --- a/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx +++ b/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx @@ -11,8 +11,6 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import { useAppStore } from '../../store' import { formatShortcutLabel, useOptionalShortcutLabel } from '@/hooks/useShortcutLabel' import { translate } from '@/i18n/i18n' -import { TabAgentSessionIdMenuItem } from './TabAgentSessionIdMenuItem' -import { resolveTabAgentSessionId } from './tab-agent-session-id' import { TerminalTabSplitMenuSection } from './TerminalTabSplitMenuSection' import { TAB_CONTEXT_MENU_CONTENT_CLASS } from './tab-context-menu-sizing' @@ -123,10 +121,6 @@ export function SortableTabContextMenu({ onTogglePin }: SortableTabContextMenuProps): React.JSX.Element { const keybindings = useAppStore((state) => state.keybindings) - // The id is a primitive, so unchanged sessions stay referentially stable without a cache. - const agentSessionId = useAppStore((state) => - open ? resolveTabAgentSessionId(state, tab.id) : null - ) const splitRightShortcut = formatShortcutLabel('terminal.splitRight', keybindings) const splitDownShortcut = formatShortcutLabel('terminal.splitDown', keybindings) @@ -194,7 +188,6 @@ export function SortableTabContextMenu({ {translate('auto.components.tab.bar.SortableTabContextMenu.2f697b3c31', 'Change Title')} {renameShortcut ? {renameShortcut} : null} -
    {translate('auto.components.tab.bar.SortableTabContextMenu.35e8892fd0', 'Tab Color')} diff --git a/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx b/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx deleted file mode 100644 index da857ec57bf..00000000000 --- a/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.test.tsx +++ /dev/null @@ -1,79 +0,0 @@ -/** - * @vitest-environment happy-dom - */ -import { act, type ReactNode } from 'react' -import { createRoot, type Root } from 'react-dom/client' -import { afterEach, describe, expect, it, vi } from 'vitest' -import { TabAgentSessionIdMenuItem } from './TabAgentSessionIdMenuItem' - -const toastMock = vi.hoisted(() => ({ success: vi.fn(), error: vi.fn() })) - -vi.mock('@/components/ui/dropdown-menu', () => ({ - DropdownMenuItem: ({ - children, - disabled, - onSelect, - 'aria-label': ariaLabel - }: { - children?: ReactNode - disabled?: boolean - onSelect?: () => void - 'aria-label'?: string - }) => ( - - ) -})) - -vi.mock('lucide-react', () => ({ Copy: () => null })) -vi.mock('@/i18n/i18n', () => ({ translate: (_key: string, fallback: string) => fallback })) -vi.mock('sonner', () => ({ toast: toastMock })) - -const mounted: { container: HTMLDivElement; root: Root }[] = [] - -function render(sessionId: string | null): HTMLDivElement { - const container = document.createElement('div') - document.body.appendChild(container) - const root = createRoot(container) - act(() => root.render()) - mounted.push({ container, root }) - return container -} - -afterEach(() => { - for (const { container, root } of mounted.splice(0)) { - act(() => root.unmount()) - container.remove() - } - toastMock.success.mockReset() - toastMock.error.mockReset() -}) - -describe('TabAgentSessionIdMenuItem', () => { - it('renders nothing when no session id is available', () => { - expect(render(null).textContent).toBe('') - }) - - it('copies on select when an id is known', async () => { - const writeClipboardText = vi.fn().mockResolvedValue(undefined) - Object.assign(window, { api: { ui: { writeClipboardText } } }) - const container = render('abc-123') - - const button = container.querySelector('button') - expect(button?.disabled).toBe(false) - act(() => button?.click()) - await vi.waitFor(() => expect(writeClipboardText).toHaveBeenCalledWith('abc-123')) - }) - - it('reports clipboard failures', async () => { - const writeClipboardText = vi.fn().mockRejectedValue(new Error('clipboard unavailable')) - Object.assign(window, { api: { ui: { writeClipboardText } } }) - const button = render('abc-123').querySelector('button') - - act(() => button?.click()) - await vi.waitFor(() => - expect(toastMock.error).toHaveBeenCalledWith('Failed to copy Session ID') - ) - }) -}) diff --git a/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx b/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx deleted file mode 100644 index 73f3f128e5d..00000000000 --- a/src/renderer/src/components/tab-bar/TabAgentSessionIdMenuItem.tsx +++ /dev/null @@ -1,48 +0,0 @@ -import { Copy } from 'lucide-react' -import { toast } from 'sonner' -import { DropdownMenuItem } from '@/components/ui/dropdown-menu' -import { translate } from '@/i18n/i18n' - -async function copySessionId(sessionId: string): Promise { - try { - await window.api.ui.writeClipboardText(sessionId) - toast.success( - translate( - 'components.tab.bar.SortableTabContextMenu.copySessionIdSuccess', - 'Session ID copied' - ) - ) - } catch { - toast.error( - translate( - 'components.tab.bar.SortableTabContextMenu.copySessionIdError', - 'Failed to copy Session ID' - ) - ) - } -} - -/** Copies the active pane's provider session id when one is available. */ -export function TabAgentSessionIdMenuItem({ - sessionId -}: { - sessionId: string | null -}): React.JSX.Element | null { - if (sessionId === null) { - return null - } - const label = translate( - 'components.tab.bar.SortableTabContextMenu.copySessionId', - 'Copy Session ID' - ) - return ( - { - void copySessionId(sessionId) - }} - > - - {label} - - ) -} diff --git a/src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts b/src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts deleted file mode 100644 index 2f06f0a8db7..00000000000 --- a/src/renderer/src/components/tab-bar/tab-agent-session-id.test.ts +++ /dev/null @@ -1,159 +0,0 @@ -import { describe, expect, it } from 'vitest' -import type { AgentStatusEntry } from '../../../../shared/agent-status-types' -import { resolveTabAgentSessionId, type TabAgentSessionIdState } from './tab-agent-session-id' - -const LEAF_A = '11111111-1111-4111-8111-111111111111' -const LEAF_B = '22222222-2222-4222-8222-222222222222' - -function entry(overrides: Partial = {}): AgentStatusEntry { - return { - state: 'done', - prompt: '', - updatedAt: 1, - stateStartedAt: 1, - paneKey: `tab-1:${LEAF_A}`, - agentType: 'claude', - stateHistory: [], - ...overrides - } -} - -function state(overrides: Partial = {}): TabAgentSessionIdState { - return { - terminalLayoutsByTabId: { - 'tab-1': { - root: { type: 'leaf', leafId: LEAF_A }, - activeLeafId: LEAF_A, - expandedLeafId: null - } - }, - agentStatusByPaneKey: {}, - paneForegroundAgentByPaneKey: {}, - ...overrides - } -} - -describe('resolveTabAgentSessionId', () => { - it('is absent when the pane has no agent row', () => { - expect(resolveTabAgentSessionId(state(), 'tab-1')).toBeNull() - }) - - it('is absent for a tab with no layout', () => { - expect(resolveTabAgentSessionId(state(), 'tab-missing')).toBeNull() - }) - - it('reads the id reported by the active pane', () => { - const resolved = resolveTabAgentSessionId( - state({ - agentStatusByPaneKey: { - [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'abc-123' } }) - } - }), - 'tab-1' - ) - expect(resolved).toBe('abc-123') - }) - - it('is absent until the agent reports an id', () => { - const resolved = resolveTabAgentSessionId( - state({ agentStatusByPaneKey: { [`tab-1:${LEAF_A}`]: entry() } }), - 'tab-1' - ) - expect(resolved).toBeNull() - }) - - describe('liveness', () => { - it('is absent for a hydrated row with no live hook since restore', () => { - const resolved = resolveTabAgentSessionId( - state({ - agentStatusByPaneKey: { - [`tab-1:${LEAF_A}`]: entry({ - restoredUnconfirmed: true, - providerSession: { key: 'session_id', id: 'abc-123' } - }) - } - }), - 'tab-1' - ) - expect(resolved).toBeNull() - }) - - it('is absent once the pane is proven back at the shell', () => { - const resolved = resolveTabAgentSessionId( - state({ - agentStatusByPaneKey: { - [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'abc-123' } }) - }, - paneForegroundAgentByPaneKey: { - [`tab-1:${LEAF_A}`]: { agent: null, shellForeground: true } - } - }), - 'tab-1' - ) - expect(resolved).toBeNull() - }) - - it('keeps a session whose foreground evidence is only that an agent runs', () => { - const resolved = resolveTabAgentSessionId( - state({ - agentStatusByPaneKey: { - [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'abc-123' } }) - }, - paneForegroundAgentByPaneKey: { - [`tab-1:${LEAF_A}`]: { agent: 'claude', shellForeground: false } - } - }), - 'tab-1' - ) - expect(resolved).toBe('abc-123') - }) - - it('keeps a working session that reported a session boundary', () => { - // Why: sessionBoundary marks a resume/clear landing idle — a session start, - // not a session end, and exactly when the first id arrives. - const resolved = resolveTabAgentSessionId( - state({ - agentStatusByPaneKey: { - [`tab-1:${LEAF_A}`]: entry({ - sessionBoundary: true, - providerSession: { key: 'session_id', id: 'fresh-1' } - }) - } - }), - 'tab-1' - ) - expect(resolved).toBe('fresh-1') - }) - }) - - describe('split tabs', () => { - const splitState = (activeLeafId: string): TabAgentSessionIdState => - state({ - terminalLayoutsByTabId: { - 'tab-1': { - root: { - type: 'split', - direction: 'vertical', - first: { type: 'leaf', leafId: LEAF_A }, - second: { type: 'leaf', leafId: LEAF_B } - }, - activeLeafId, - expandedLeafId: null - } - }, - agentStatusByPaneKey: { - [`tab-1:${LEAF_A}`]: entry({ providerSession: { key: 'session_id', id: 'left' } }), - [`tab-1:${LEAF_B}`]: entry({ providerSession: { key: 'session_id', id: 'right' } }) - } - }) - - it('reads the active pane, not a sibling', () => { - expect(resolveTabAgentSessionId(splitState(LEAF_B), 'tab-1')).toBe('right') - }) - - it('is absent when the active leaf id no longer exists in the layout', () => { - const stale = '33333333-3333-4333-8333-333333333333' - expect(resolveTabAgentSessionId(splitState(stale), 'tab-1')).toBeNull() - }) - }) -}) diff --git a/src/renderer/src/components/tab-bar/tab-agent-session-id.ts b/src/renderer/src/components/tab-bar/tab-agent-session-id.ts deleted file mode 100644 index e0d804bc4c5..00000000000 --- a/src/renderer/src/components/tab-bar/tab-agent-session-id.ts +++ /dev/null @@ -1,32 +0,0 @@ -import type { AgentStatusEntry } from '../../../../shared/agent-status-types' -import type { TerminalLayoutSnapshot } from '../../../../shared/terminal-tab-types' -import type { PaneForegroundAgentEntry } from '../../store/slices/pane-foreground-agent' -import { resolveNativeChatActiveLayoutLeafId } from '../native-chat/native-chat-leaf-routing' - -export type TabAgentSessionIdState = { - agentStatusByPaneKey?: Record - terminalLayoutsByTabId?: Record - paneForegroundAgentByPaneKey?: Record -} - -/** Returns the active pane's provider session id when its agent is still live. */ -export function resolveTabAgentSessionId( - state: TabAgentSessionIdState, - tabId: string -): string | null { - const leafId = resolveNativeChatActiveLayoutLeafId(state.terminalLayoutsByTabId?.[tabId]) - if (!leafId) { - return null - } - const paneKey = `${tabId}:${leafId}` - const entry = state.agentStatusByPaneKey?.[paneKey] - // Hydrated rows may describe a session that ended while no receiver was up. - if (!entry?.agentType || entry.restoredUnconfirmed === true) { - return null - } - // OSC 133;D proves the pane is back at the shell, regardless of the last hook state. - if (state.paneForegroundAgentByPaneKey?.[paneKey]?.shellForeground === true) { - return null - } - return entry.providerSession?.id ?? null -} diff --git a/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx b/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx index 54f762d4611..c81eae348dd 100644 --- a/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx +++ b/src/renderer/src/components/tab-bar/tab-context-menu-consistency.test.tsx @@ -13,7 +13,6 @@ import { const TAB_MENU_SOURCES = [ 'EditorFileTabContextMenu.tsx', 'SortableTabContextMenu.tsx', - 'TabAgentSessionIdMenuItem.tsx', 'BrowserTab.tsx', 'TabWorkspaceLayoutMenuSection.tsx', 'TerminalTabSplitMenuSection.tsx' diff --git a/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx b/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx index bd2ca330f5c..d8dc54c954b 100644 --- a/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx @@ -92,6 +92,8 @@ function renderMenu(overrides: Record = {}): string { canClearPaneTitle: false, onCopyTerminalId: vi.fn(), onCopyPaneId: vi.fn(), + canCopyAgentSessionId: false, + onCopyAgentSessionId: vi.fn(), ...overrides } return renderToStaticMarkup(React.createElement(TerminalContextMenu, props)) @@ -146,6 +148,29 @@ describe('TerminalContextMenu', () => { expect(items.list.some((item) => childrenText(item.children).includes('Switch to'))).toBe(false) }) + it('shows Copy Session ID only for panes with provider identity', () => { + const onCopyAgentSessionId = vi.fn() + renderMenu({ canCopyAgentSessionId: true, onCopyAgentSessionId }) + + const item = items.list.find( + (candidate) => childrenText(candidate.children) === 'Copy Session ID' + ) + expect(item).toBeDefined() + expect( + items.list + .map((candidate) => childrenText(candidate.children)) + .filter((label) => ['Copy Session ID', 'Copy Terminal ID', 'Copy Pane ID'].includes(label)) + ).toEqual(['Copy Session ID', 'Copy Terminal ID', 'Copy Pane ID']) + item?.onSelect?.() + expect(onCopyAgentSessionId).toHaveBeenCalledTimes(1) + + items.list = [] + renderMenu({ canCopyAgentSessionId: false }) + expect( + items.list.some((candidate) => childrenText(candidate.children) === 'Copy Session ID') + ).toBe(false) + }) + it('shows one shortcut per terminal menu action on Windows', () => { vi.stubGlobal('navigator', { userAgent: 'Mozilla/5.0 (Windows NT 10.0; Win64; x64)' diff --git a/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx b/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx index 178ab0a3247..8311fb3df9d 100644 --- a/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx @@ -66,6 +66,8 @@ type TerminalContextMenuProps = { canClearPaneTitle: boolean onCopyTerminalId: () => void onCopyPaneId: () => void + canCopyAgentSessionId: boolean + onCopyAgentSessionId: () => void } export default function TerminalContextMenu({ @@ -101,7 +103,9 @@ export default function TerminalContextMenu({ onClearPaneTitle, canClearPaneTitle, onCopyTerminalId, - onCopyPaneId + onCopyPaneId, + canCopyAgentSessionId, + onCopyAgentSessionId }: TerminalContextMenuProps): React.JSX.Element { // Why: one primary binding prevents Windows/Linux shortcut labels from forcing row wraps. const shortcuts = useMemo( @@ -276,6 +280,15 @@ export default function TerminalContextMenu({ ) : null} ) : null} + {canCopyAgentSessionId ? ( + + + {translate( + 'components.terminalPane.TerminalContextMenu.copySessionId', + 'Copy Session ID' + )} + + ) : null} {translate( diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx index 73c4220dec5..1b5e94b4a11 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneNativeChatPortal.tsx @@ -3,6 +3,8 @@ import NativeChatView from '../native-chat/NativeChatView' import { makePaneKey } from '../../../../shared/stable-pane-id' import { canContinueAgentSessionInNewSession } from './terminal-agent-session-continuation' import type { TerminalPaneController } from './use-terminal-pane-controller' +import { useAppStore } from '@/store' +import { resolvePaneAgentSessionId } from './pane-agent-session-id' export function TerminalPaneNativeChatPortal({ controller @@ -30,6 +32,11 @@ export function TerminalPaneNativeChatPortal({ tabId, unifiedTabId } = controller + const chatPaneSessionId = useAppStore((state) => + effectiveChatViewMode && chatPane + ? resolvePaneAgentSessionId(state, makePaneKey(tabId, chatPane.leafId)) + : null + ) if (!effectiveChatViewMode || !chatPane?.container) { return null } @@ -78,6 +85,9 @@ export function TerminalPaneNativeChatPortal({ onCopyTerminalId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyTerminalId), onCopyPaneId: () => void contextMenu.runForPane(chatPane.id, contextMenu.onCopyPaneId), + canCopyAgentSessionId: chatPaneSessionId !== null, + onCopyAgentSessionId: () => + void contextMenu.runForPane(chatPane.id, contextMenu.onCopyAgentSessionId), canClosePane: managedPanes.length > 1, onClosePane: () => contextMenu.runForPane(chatPane.id, contextMenu.onClosePane) }} diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx index fff316dabce..fc33cab3f18 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx @@ -59,6 +59,7 @@ export function TerminalPaneSurface({ keybindings, managedPanes, managerRef, + menuAgentSessionId, menuPaneHasCustomTitle, openDiskSpaceAnalyzer, openQuickCommandEditor, @@ -247,6 +248,8 @@ export function TerminalPaneSurface({ canClearPaneTitle={menuPaneHasCustomTitle} onCopyTerminalId={() => void contextMenu.onCopyTerminalId()} onCopyPaneId={contextMenu.onCopyPaneId} + canCopyAgentSessionId={menuAgentSessionId !== null} + onCopyAgentSessionId={() => void contextMenu.onCopyAgentSessionId()} /> { + it('returns the live provider session for the exact pane', () => { + expect(resolvePaneAgentSessionId(state(live('live-session')), PANE_KEY)).toBe('live-session') + }) + + it('returns the pane-owned durable session after its live status row is cleared', () => { + expect( + resolvePaneAgentSessionId(state(undefined, sleeping('sleeping-session')), PANE_KEY) + ).toBe('sleeping-session') + }) + + it('does not reuse an older durable session while a newer live row lacks identity', () => { + expect(resolvePaneAgentSessionId(state(live(), sleeping('old-session')), PANE_KEY)).toBeNull() + }) + + it('falls back from an unconfirmed restored row to durable pane identity', () => { + expect( + resolvePaneAgentSessionId( + state(live('unconfirmed-session', true), sleeping('confirmed-session')), + PANE_KEY + ) + ).toBe('confirmed-session') + }) + + describe('liveness', () => { + it('is absent once the pane is proven back at the shell', () => { + expect( + resolvePaneAgentSessionId(state(live('live-session'), undefined, true), PANE_KEY) + ).toBe(null) + }) + + it('is absent at the shell even when a durable record survives the exit', () => { + expect( + resolvePaneAgentSessionId(state(undefined, sleeping('sleeping-session'), true), PANE_KEY) + ).toBeNull() + }) + + it('keeps a session whose foreground evidence is only that an agent runs', () => { + expect( + resolvePaneAgentSessionId(state(live('live-session'), undefined, false), PANE_KEY) + ).toBe('live-session') + }) + + it('keeps a session for a pane with no foreground evidence at all', () => { + expect( + resolvePaneAgentSessionId( + { + agentStatusByPaneKey: { [PANE_KEY]: live('live-session') }, + sleepingAgentSessionsByPaneKey: {}, + paneForegroundAgentByPaneKey: {} + }, + PANE_KEY + ) + ).toBe('live-session') + }) + }) + + it('does not read identity from a sibling pane', () => { + const sibling = 'tab-1:22222222-2222-4222-8222-222222222222' + expect(resolvePaneAgentSessionId(state(undefined, sleeping('session-1')), sibling)).toBeNull() + }) +}) diff --git a/src/renderer/src/components/terminal-pane/pane-agent-session-id.ts b/src/renderer/src/components/terminal-pane/pane-agent-session-id.ts new file mode 100644 index 00000000000..59ab6f38e86 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/pane-agent-session-id.ts @@ -0,0 +1,27 @@ +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import type { SleepingAgentSessionRecord } from '../../../../shared/agent-session-resume' +import type { PaneForegroundAgentEntry } from '../../store/slices/pane-foreground-agent' + +export type PaneAgentSessionIdState = { + agentStatusByPaneKey: Record + sleepingAgentSessionsByPaneKey: Record + paneForegroundAgentByPaneKey: Record +} + +/** Resolves the provider session owned by one exact terminal pane, while its agent is still live. */ +export function resolvePaneAgentSessionId( + state: PaneAgentSessionIdState, + paneKey: string +): string | null { + // OSC 133;D proves the pane is back at the shell. The durable record outlives that exit on + // purpose (cold restore resumes from it), so gate it here too — otherwise the gate would only + // hold for panes whose agent has no resumable record. + if (state.paneForegroundAgentByPaneKey[paneKey]?.shellForeground === true) { + return null + } + const live = state.agentStatusByPaneKey[paneKey] + if (live && live.restoredUnconfirmed !== true) { + return live.providerSession?.id ?? null + } + return state.sleepingAgentSessionsByPaneKey[paneKey]?.providerSession.id ?? null +} diff --git a/src/renderer/src/components/terminal-pane/terminal-copy-rejection-guards.ts b/src/renderer/src/components/terminal-pane/terminal-copy-rejection-guards.ts index 6debb3e3d5f..9a13d026d08 100644 --- a/src/renderer/src/components/terminal-pane/terminal-copy-rejection-guards.ts +++ b/src/renderer/src/components/terminal-pane/terminal-copy-rejection-guards.ts @@ -1,7 +1,7 @@ // Why: writeClipboardText resolved unconditionally in the web client until the // insecure-context copy fallback landed. It can now reject (insecure origin with -// no live user gesture), so these two menu actions need explicit outcomes: -// the copy must never leave the pane unfocused, and Copy Pane ID must not toast +// no live user gesture), so identity-copy actions need explicit outcomes: +// the copy must never leave the pane unfocused, and identity actions must not toast // success for a copy that did not happen. Extracted so both are testable without // mounting the whole context-menu hook. @@ -22,15 +22,15 @@ export async function runTerminalCopy(args: { } } -export async function runCopyPaneId(args: { - paneKey: string +export async function runTerminalIdentityCopy(args: { + text: string writeClipboardText: (text: string) => Promise onSuccess: () => void onError: () => void focus: () => void }): Promise { try { - await args.writeClipboardText(args.paneKey) + await args.writeClipboardText(args.text) args.onSuccess() } catch { args.onError() diff --git a/src/renderer/src/components/terminal-pane/terminal-copy-rejection-handling.test.ts b/src/renderer/src/components/terminal-pane/terminal-copy-rejection-handling.test.ts index d6aac273ed8..1b14b379f2a 100644 --- a/src/renderer/src/components/terminal-pane/terminal-copy-rejection-handling.test.ts +++ b/src/renderer/src/components/terminal-pane/terminal-copy-rejection-handling.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it, vi } from 'vitest' -import { runTerminalCopy, runCopyPaneId } from './terminal-copy-rejection-guards' +import { runTerminalCopy, runTerminalIdentityCopy } from './terminal-copy-rejection-guards' // Why this file exists: web-preload-api's writeClipboardText used to resolve // unconditionally, so the terminal copy surfaces call it without a rejection @@ -42,15 +42,15 @@ describe('runTerminalCopy', () => { }) }) -describe('runCopyPaneId', () => { +describe('runTerminalIdentityCopy', () => { it('reports failure instead of claiming success when the write rejects', async () => { const onSuccess = vi.fn() const onError = vi.fn() const focus = vi.fn() await expect( - runCopyPaneId({ - paneKey: 'tab:leaf', + runTerminalIdentityCopy({ + text: 'tab:leaf', writeClipboardText: vi.fn().mockRejectedValue(REJECTION), onSuccess, onError, @@ -68,8 +68,8 @@ describe('runCopyPaneId', () => { const onError = vi.fn() const focus = vi.fn() - await runCopyPaneId({ - paneKey: 'tab:leaf', + await runTerminalIdentityCopy({ + text: 'tab:leaf', writeClipboardText: vi.fn().mockResolvedValue(undefined), onSuccess, onError, diff --git a/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts b/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts index 206a31e7177..c42bbd31919 100644 --- a/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts +++ b/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts @@ -8,9 +8,9 @@ import { describe, expect, it } from 'vitest' const TERMINAL_PANE_HOOK_SOURCE_PATTERN = /^(?:TerminalPane\.tsx|use-terminal-pane-(?:chat-state|close-actions|context-actions|controller|foundation|global-listeners|layout-bindings|layout-persistence|lifecycle-stage|mobile-actions|paste-listeners|process-exit-actions|projection|reconciliation|startup-actions|store-bindings|title-effects|title-state)\.ts)$/ // Rebased onto main after the workbench surface-per-workspace and deferred -// split-cwd changes; this hash is from that main's pre-split TerminalPane (229 hooks). +// split-cwd changes; the pane session-ID projection adds one render hook (230 hooks). const PRE_REFACTOR_HOOK_ORDER_SHA256 = - 'be2366fffb992e082fd9e4641b7543a0db75cb7bc4ef6e0eb0deda41beaac358' + '77adcf8272ddc6f903920f12b2b652ec26c570987f452076ae6460c67d3741f3' const sourceFiles = readdirSync(__dirname) .filter((name) => TERMINAL_PANE_HOOK_SOURCE_PATTERN.test(name)) @@ -75,7 +75,7 @@ function readFlattenedHookOrder(): string[] { describe('TerminalPane refactor hook parity', () => { it('preserves the recursively flattened render hook order', () => { const hooks = readFlattenedHookOrder() - expect(hooks).toHaveLength(229) + expect(hooks).toHaveLength(230) expect(hooks.filter((hook) => hook === 'useMemo')).toHaveLength(4) expect(createHash('sha256').update(hooks.join('\n')).digest('hex')).toBe( PRE_REFACTOR_HOOK_ORDER_SHA256 diff --git a/src/renderer/src/components/terminal-pane/terminal-pane-menu-copy-actions.ts b/src/renderer/src/components/terminal-pane/terminal-pane-menu-copy-actions.ts index 62051f84f26..417f920af5c 100644 --- a/src/renderer/src/components/terminal-pane/terminal-pane-menu-copy-actions.ts +++ b/src/renderer/src/components/terminal-pane/terminal-pane-menu-copy-actions.ts @@ -3,7 +3,7 @@ import type { ManagedPane } from '@/lib/pane-manager/pane-manager' import { makePaneKey } from '../../../../shared/stable-pane-id' import { translate } from '@/i18n/i18n' import { copyTerminalHandleForPane } from './terminal-handle-copy' -import { runCopyPaneId, runTerminalCopy } from './terminal-copy-rejection-guards' +import { runTerminalCopy, runTerminalIdentityCopy } from './terminal-copy-rejection-guards' export const copyTerminalPaneMenuSelection = async (pane: ManagedPane | null): Promise => { if (!pane) { @@ -27,10 +27,10 @@ export const copyTerminalPaneMenuPaneId = async ( if (!pane) { return } - await runCopyPaneId({ + await runTerminalIdentityCopy({ // Why: orchestration targets use ORCA_PANE_KEY, which survives renderer // remounts; the numeric PaneManager id is only a local runtime handle. - paneKey: makePaneKey(tabId, pane.leafId), + text: makePaneKey(tabId, pane.leafId), writeClipboardText: window.api.ui.writeTerminalClipboardText, onSuccess: () => toast.success( @@ -83,3 +83,35 @@ export const copyTerminalPaneMenuTerminalId = async ( pane.terminal.focus() } } + +export const copyTerminalPaneMenuAgentSessionId = async ( + pane: ManagedPane | null, + sessionId: string | null +): Promise => { + if (!pane) { + return + } + if (!sessionId) { + pane.terminal.focus() + return + } + await runTerminalIdentityCopy({ + text: sessionId, + writeClipboardText: window.api.ui.writeTerminalClipboardText, + onSuccess: () => + toast.success( + translate( + 'components.terminalPane.TerminalContextMenu.copySessionIdSuccess', + 'Session ID copied' + ) + ), + onError: () => + toast.error( + translate( + 'components.terminalPane.TerminalContextMenu.copySessionIdError', + 'Unable to copy session ID' + ) + ), + focus: () => pane.terminal.focus() + }) +} diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-context-menu.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-context-menu.ts index a6f04fca5f3..155482f2163 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-pane-context-menu.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-context-menu.ts @@ -11,6 +11,7 @@ import type { PreparedAgentSessionFork } from './terminal-agent-session-fork' import type { AgentSessionContinuationRequest } from '@/lib/agent-session-continuation' import { pasteTerminalPaneMenuClipboard } from './terminal-pane-menu-paste' import { + copyTerminalPaneMenuAgentSessionId, copyTerminalPaneMenuPaneId, copyTerminalPaneMenuSelection, copyTerminalPaneMenuTerminalId @@ -22,6 +23,9 @@ import { } from './terminal-pane-menu-agent-session-actions' import { useTerminalPaneSplitActions } from './use-terminal-pane-split-actions' import { useTerminalContextMenuTrigger } from './use-terminal-context-menu-trigger' +import { useAppStore } from '@/store' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import { resolvePaneAgentSessionId } from './pane-agent-session-id' type UseTerminalPaneContextMenuDeps = { managerRef: React.RefObject @@ -57,6 +61,7 @@ type TerminalMenuState = { onSelectAll: () => void onCopyTerminalId: () => Promise onCopyPaneId: () => Promise + onCopyAgentSessionId: () => Promise onPaste: () => Promise onSplitRight: () => void onSplitDown: () => void @@ -170,6 +175,14 @@ export function useTerminalPaneContextMenu({ const onCopyTerminalId = async (): Promise => copyTerminalPaneMenuTerminalId(resolveMenuPane(), tabId) + const onCopyAgentSessionId = async (): Promise => { + const pane = resolveMenuPane() + const sessionId = pane + ? resolvePaneAgentSessionId(useAppStore.getState(), makePaneKey(tabId, pane.leafId)) + : null + return copyTerminalPaneMenuAgentSessionId(pane, sessionId) + } + const onPaste = async (): Promise => pasteResolvedPane('context-menu') const onEqualizePaneSizes = (): void => { @@ -275,6 +288,7 @@ export function useTerminalPaneContextMenu({ onSelectAll, onCopyTerminalId, onCopyPaneId, + onCopyAgentSessionId, onPaste, onSplitRight, onSplitDown, diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts index 3e4ff6f6efc..ea179b6925e 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts @@ -16,6 +16,9 @@ import { } from '../native-chat/native-chat-leaf-routing' import { canContinueAgentSessionInNewSession } from './terminal-agent-session-continuation' import type { TerminalPaneMobileController } from './use-terminal-pane-mobile-actions' +import { useAppStore } from '@/store' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import { resolvePaneAgentSessionId } from './pane-agent-session-id' export function useTerminalPaneProjection(controller: TerminalPaneMobileController) { const { @@ -40,6 +43,7 @@ export function useTerminalPaneProjection(controller: TerminalPaneMobileControll shouldMeasureHiddenStartup, structuredSessionAgent, structuredSessionId, + tabId, sshReconnectOwnsTerminalErrors, systemPrefersDark, tabAgentTypeByLeaf, @@ -100,6 +104,11 @@ export function useTerminalPaneProjection(controller: TerminalPaneMobileControll ) const menuPaneHasCustomTitle = contextMenu.menuPaneId !== null && Boolean(paneTitles[contextMenu.menuPaneId]) + const menuAgentSessionId = useAppStore((state) => + contextMenu.open && contextMenuLeafId + ? resolvePaneAgentSessionId(state, makePaneKey(tabId, contextMenuLeafId)) + : null + ) const chatLeafStillMounted = chatLeafId ? managedPanes.some((pane) => pane.leafId === chatLeafId) : false @@ -181,6 +190,7 @@ export function useTerminalPaneProjection(controller: TerminalPaneMobileControll showSshReconnectOverlay, visibleTerminalError, menuPaneHasCustomTitle, + menuAgentSessionId, chatLeafStillMounted, chatPane, chatPanePtyId, diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 7d621d77751..f1489d6754d 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16781,10 +16781,7 @@ "SortableTabContextMenu": { "switchToTerminalView": "Switch to terminal view", "switchToChatView": "Switch to chat view", - "closeTabsToLeft": "Close Tabs To The Left", - "copySessionId": "Copy Session ID", - "copySessionIdSuccess": "Session ID copied", - "copySessionIdError": "Failed to copy Session ID" + "closeTabsToLeft": "Close Tabs To The Left" }, "BrowserTab": { "closeOthers": "Close Others", @@ -17073,6 +17070,13 @@ "days": "{{value0}}d", "underOneMinute": "<1m" } + }, + "terminalPane": { + "TerminalContextMenu": { + "copySessionId": "Copy Session ID", + "copySessionIdSuccess": "Session ID copied", + "copySessionIdError": "Unable to copy session ID" + } } }, "dashboardPopout": { diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 4736cdbbbd4..79731f3b096 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -14652,10 +14652,7 @@ "SortableTabContextMenu": { "switchToTerminalView": "Cambiar a la vista de terminal", "switchToChatView": "Cambiar a vista de chat", - "closeTabsToLeft": "Cerrar pestañas a la izquierda", - "copySessionId": "Copiar ID de sesión", - "copySessionIdSuccess": "ID de sesión copiado", - "copySessionIdError": "No se pudo copiar el ID de sesión" + "closeTabsToLeft": "Cerrar pestañas a la izquierda" }, "BrowserTab": { "closeOthers": "Cerrar otras", @@ -14703,6 +14700,13 @@ "sent": "El contexto de la sesión se envió a {{agent}} en una sesión nueva.", "deliveryFailed": "La nueva sesión de {{agent}} se inició, pero no se pudo enviar su contexto.", "launchFailed": "No se pudo iniciar una sesión nueva de {{agent}}." + }, + "terminalPane": { + "TerminalContextMenu": { + "copySessionId": "Copiar ID de sesión", + "copySessionIdSuccess": "ID de sesión copiado", + "copySessionIdError": "No se pudo copiar el ID de sesión" + } } }, "dashboardPopout": { diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 315e5775442..6afc506b12b 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -14652,10 +14652,7 @@ "SortableTabContextMenu": { "switchToTerminalView": "ターミナルビューに切り替える", "switchToChatView": "チャットビューに切り替える", - "closeTabsToLeft": "左側のタブを閉じる", - "copySessionId": "セッション ID をコピー", - "copySessionIdSuccess": "セッション ID をコピーしました", - "copySessionIdError": "セッション ID のコピーに失敗しました" + "closeTabsToLeft": "左側のタブを閉じる" }, "BrowserTab": { "closeOthers": "その他を閉じる", @@ -14703,6 +14700,13 @@ "sent": "セッションコンテキストを新規 {{agent}} セッションに送信しました。", "deliveryFailed": "新規 {{agent}} セッションは開始しましたが、コンテキストを送信できませんでした。", "launchFailed": "新規 {{agent}} セッションを開始できませんでした。" + }, + "terminalPane": { + "TerminalContextMenu": { + "copySessionId": "セッション ID をコピー", + "copySessionIdSuccess": "セッション ID をコピーしました", + "copySessionIdError": "セッション ID のコピーに失敗しました" + } } }, "dashboardPopout": { diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 4a196c62304..c2836df4c40 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -14708,10 +14708,7 @@ "SortableTabContextMenu": { "switchToTerminalView": "terminal 보기로 전환", "switchToChatView": "채팅 보기로 전환", - "closeTabsToLeft": "왼쪽으로 탭 닫기", - "copySessionId": "세션 ID 복사", - "copySessionIdSuccess": "세션 ID를 복사했습니다", - "copySessionIdError": "세션 ID를 복사하지 못했습니다" + "closeTabsToLeft": "왼쪽으로 탭 닫기" }, "BrowserTab": { "closeOthers": "다른 탭 닫기", @@ -14820,6 +14817,13 @@ "sent": "세션 컨텍스트를 새 {{agent}} 세션으로 보냈습니다.", "deliveryFailed": "새 {{agent}} 세션은 시작되었지만 컨텍스트를 보내지 못했습니다.", "launchFailed": "새 {{agent}} 세션을 시작할 수 없습니다." + }, + "terminalPane": { + "TerminalContextMenu": { + "copySessionId": "세션 ID 복사", + "copySessionIdSuccess": "세션 ID를 복사했습니다", + "copySessionIdError": "세션 ID를 복사하지 못했습니다" + } } }, "dashboardPopout": { diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index edbd96e2623..77f56e805ae 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -14708,10 +14708,7 @@ "SortableTabContextMenu": { "switchToTerminalView": "切换到终端视图", "switchToChatView": "切换到聊天视图", - "closeTabsToLeft": "关闭左侧的选项卡", - "copySessionId": "复制会话 ID", - "copySessionIdSuccess": "已复制会话 ID", - "copySessionIdError": "复制会话 ID 失败" + "closeTabsToLeft": "关闭左侧的选项卡" }, "BrowserTab": { "closeOthers": "关闭其他", @@ -14820,6 +14817,13 @@ "sent": "已将会话上下文发送到新的 {{agent}} 会话。", "deliveryFailed": "新的 {{agent}} 会话已启动,但无法发送上下文。", "launchFailed": "无法启动新的 {{agent}} 会话。" + }, + "terminalPane": { + "TerminalContextMenu": { + "copySessionId": "复制会话 ID", + "copySessionIdSuccess": "已复制会话 ID", + "copySessionIdError": "复制会话 ID 失败" + } } }, "dashboardPopout": { diff --git a/tests/e2e/tab-context-menu-session-id.spec.ts b/tests/e2e/terminal-context-menu-session-id.spec.ts similarity index 59% rename from tests/e2e/tab-context-menu-session-id.spec.ts rename to tests/e2e/terminal-context-menu-session-id.spec.ts index cab18bace32..dc58567cefa 100644 --- a/tests/e2e/tab-context-menu-session-id.spec.ts +++ b/tests/e2e/terminal-context-menu-session-id.spec.ts @@ -1,7 +1,4 @@ -/** - * E2E coverage for copying an agent provider session ID from a terminal tab's - * context menu. - */ +/** E2E coverage for copying provider identity from the exact terminal pane. */ import { test, expect } from './helpers/orca-app' import { @@ -11,10 +8,11 @@ import { waitForSessionReady } from './helpers/store' import { waitForPaneIdentitySnapshot } from './helpers/terminal' +import { openTerminalContextMenu } from './helpers/terminal-pane-title-actions' -const SESSION_ID = 'e2e-terminal-tab-session' +const SESSION_ID = 'e2e-terminal-pane-session' -test('terminal tab context menu copies the active agent session ID', async ({ orcaPage }) => { +test('terminal pane context menu copies its agent session ID', async ({ orcaPage }) => { await waitForSessionReady(orcaPage) const worktreeId = await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) @@ -30,21 +28,20 @@ test('terminal tab context menu copies the active agent session ID', async ({ or } const paneKey = `${tabId}:${leafId}` - // Seed the same renderer state a live agent hook produces while keeping the - // test independent of an installed provider CLI. + // Keep this independent of an installed provider CLI while exercising the + // durable pane identity used when transient live status has been cleared. await orcaPage.evaluate( ({ paneKey, tabId, worktreeId, sessionId }) => { const state = window.__store?.getState() if (!state) { throw new Error('Store unavailable') } - state.setAgentStatus( + state.recordAgentProviderSession( paneKey, - { state: 'working', prompt: 'copy session id', agentType: 'claude' }, - 'Claude', + 'claude', + { key: 'session_id', id: sessionId }, undefined, - { tabId, worktreeId }, - { providerSession: { key: 'session_id', id: sessionId } } + { tabId, worktreeId } ) }, { paneKey, tabId, worktreeId, sessionId: SESSION_ID } @@ -55,16 +52,22 @@ test('terminal tab context menu copies the active agent session ID', async ({ or () => orcaPage.evaluate( ({ paneKey }) => - window.__store?.getState().agentStatusByPaneKey[paneKey]?.providerSession?.id, + window.__store?.getState().sleepingAgentSessionsByPaneKey[paneKey]?.providerSession.id, { paneKey } ), { timeout: 3_000 } ) .toBe(SESSION_ID) - const tab = orcaPage.locator(`[data-testid="sortable-tab"][data-tab-id="${tabId}"]`) - await expect(tab).toBeVisible() - await tab.click({ button: 'right' }) + await openTerminalContextMenu(orcaPage) + + const identityItems = await orcaPage.getByRole('menuitem').allInnerTexts() + const sessionIdIndex = identityItems.indexOf('Copy Session ID') + expect(identityItems.slice(sessionIdIndex, sessionIdIndex + 3)).toEqual([ + 'Copy Session ID', + 'Copy Terminal ID', + 'Copy Pane ID' + ]) const copyItem = orcaPage.getByRole('menuitem', { name: 'Copy Session ID', exact: true }) await expect(copyItem).toBeVisible() From 28cb372559a9a1be8076198569457c9f4049e46a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:05:30 -0700 Subject: [PATCH 082/398] perf(platform): resolve the immutable platform payload once (#18135) `window.api.platform.get()` runs ~19x/sec while the app is idle. Every call recomputed a payload whose fields are all fixed for the process lifetime (`process.platform`, `process.getSystemVersion()`, `process.arch`, the shell env vars, and the env-derived Linux display server), allocated a fresh object, and crossed the context bridge. Memoize the payload lazily at preload module scope and freeze it, and cache the resolved platform in `getRendererAppPlatform()` so the 32 renderer call sites stop crossing the bridge on every render. The user-agent fallback stays uncached because the web client installs its platform API after boot. --- src/preload/api/platform-bridge.test.ts | 55 +++++++++++++++++++ src/preload/api/platform-bridge.ts | 15 ++++- ...ebar-titlebar-drag-regions.render.test.tsx | 2 + .../settings/AppearancePane.test.tsx | 2 + .../src/lib/renderer-app-platform.test.ts | 52 ++++++++++++++++++ src/renderer/src/lib/renderer-app-platform.ts | 14 +++++ .../src/store/slices/preflight.test.ts | 2 + 7 files changed, 140 insertions(+), 2 deletions(-) create mode 100644 src/preload/api/platform-bridge.test.ts create mode 100644 src/renderer/src/lib/renderer-app-platform.test.ts diff --git a/src/preload/api/platform-bridge.test.ts b/src/preload/api/platform-bridge.test.ts new file mode 100644 index 00000000000..5c8bdde34f9 --- /dev/null +++ b/src/preload/api/platform-bridge.test.ts @@ -0,0 +1,55 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { platformApi } from './platform-bridge' + +const mocks = vi.hoisted(() => ({ getLinuxDisplayServer: vi.fn(() => null) })) + +vi.mock('../preload-runtime-support', () => ({ + getLinuxDisplayServer: mocks.getLinuxDisplayServer +})) + +// Electron declares getSystemVersion as required on NodeJS.Process; Node does not have it. +const mutableProcess = process as unknown as { getSystemVersion?: () => string } + +async function loadPlatformApi(): Promise { + vi.resetModules() + return (await import('./platform-bridge')).platformApi +} + +describe('platformApi.get', () => { + beforeEach(() => { + mocks.getLinuxDisplayServer.mockClear() + }) + + afterEach(() => { + delete mutableProcess.getSystemVersion + }) + + it('resolves the immutable payload once and returns the identical object', async () => { + const platformApi = await loadPlatformApi() + const getSystemVersion = vi.fn(() => '25.3.0') + mutableProcess.getSystemVersion = getSystemVersion + + const first = platformApi.get() + for (let index = 0; index < 100; index += 1) { + expect(platformApi.get()).toBe(first) + } + + expect(getSystemVersion).toHaveBeenCalledTimes(1) + expect(mocks.getLinuxDisplayServer).toHaveBeenCalledTimes(1) + expect(first.platform).toBe(process.platform) + expect(first.arch).toBe(process.arch) + expect(first.osRelease).toBe('25.3.0') + }) + + it('freezes the payload so no consumer can corrupt the shared instance', async () => { + const platformApi = await loadPlatformApi() + + expect(Object.isFrozen(platformApi.get())).toBe(true) + }) + + it('resolves nothing before the first get, keeping preload startup free', async () => { + await loadPlatformApi() + + expect(mocks.getLinuxDisplayServer).not.toHaveBeenCalled() + }) +}) diff --git a/src/preload/api/platform-bridge.ts b/src/preload/api/platform-bridge.ts index 8e47a4386cf..fb1aad5de4e 100644 --- a/src/preload/api/platform-bridge.ts +++ b/src/preload/api/platform-bridge.ts @@ -1,8 +1,14 @@ import { getLinuxDisplayServer } from '../preload-runtime-support' import type { PreloadApi } from '../api-types' -export const platformApi = { - get: () => ({ +type PlatformInfo = ReturnType + +// Why: the renderer reads this on its render cadence, and every field below is fixed +// for the process lifetime, so resolve once and hand back the same frozen payload. +let platformInfo: PlatformInfo | undefined + +function resolvePlatformInfo(): PlatformInfo { + return Object.freeze({ platform: process.platform, // Why: sandboxed preload cannot require node:os; Electron exposes the OS // version on process.getSystemVersion when available. @@ -14,4 +20,9 @@ export const platformApi = { shell: process.env.SHELL?.trim() || process.env.ComSpec?.trim() || '', displayServer: getLinuxDisplayServer() }) +} + +export const platformApi = { + // Why: resolved lazily so preload startup keeps paying nothing for it. + get: () => (platformInfo ??= resolvePlatformInfo()) } satisfies PreloadApi['platform'] diff --git a/src/renderer/src/components/right-sidebar/right-sidebar-titlebar-drag-regions.render.test.tsx b/src/renderer/src/components/right-sidebar/right-sidebar-titlebar-drag-regions.render.test.tsx index 6869daaa89c..40a3966047f 100644 --- a/src/renderer/src/components/right-sidebar/right-sidebar-titlebar-drag-regions.render.test.tsx +++ b/src/renderer/src/components/right-sidebar/right-sidebar-titlebar-drag-regions.render.test.tsx @@ -11,6 +11,7 @@ import { RIGHT_SIDEBAR_WINDOWS_TOP_ACTIVITY_STRIP_CLASS_NAME } from './right-sidebar-titlebar-drag-regions' import type { ActiveRightSidebarTab } from '@/store/slices/editor' +import { resetRendererAppPlatformCacheForTests } from '@/lib/renderer-app-platform' const mockAppState = vi.hoisted(() => ({ rightSidebarOpen: true, @@ -198,6 +199,7 @@ function expectNoDrag(tag: string): void { } function setRendererPlatform(platform: NodeJS.Platform): void { + resetRendererAppPlatformCacheForTests() Object.defineProperty(window, 'api', { configurable: true, value: { diff --git a/src/renderer/src/components/settings/AppearancePane.test.tsx b/src/renderer/src/components/settings/AppearancePane.test.tsx index 4a89d66cd03..01573d65b49 100644 --- a/src/renderer/src/components/settings/AppearancePane.test.tsx +++ b/src/renderer/src/components/settings/AppearancePane.test.tsx @@ -5,6 +5,7 @@ import { createRoot, type Root } from 'react-dom/client' import { I18nextProvider } from 'react-i18next' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { i18n } from '@/i18n/i18n' +import { resetRendererAppPlatformCacheForTests } from '@/lib/renderer-app-platform' import { getDefaultSettings } from '../../../../shared/constants' import type { GlobalSettings } from '../../../../shared/global-settings-types' import type { StatusBarItem } from '../../../../shared/ui-chrome-types' @@ -216,6 +217,7 @@ describe('AppearancePane', () => { beforeEach(() => { vi.clearAllMocks() + resetRendererAppPlatformCacheForTests() mocks.state.availableStatusBarToggles = [] mocks.state.appPlatform = 'linux' mocks.state.settingsSearchQuery = 'automations' diff --git a/src/renderer/src/lib/renderer-app-platform.test.ts b/src/renderer/src/lib/renderer-app-platform.test.ts new file mode 100644 index 00000000000..22c0388084a --- /dev/null +++ b/src/renderer/src/lib/renderer-app-platform.test.ts @@ -0,0 +1,52 @@ +// @vitest-environment happy-dom +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + getRendererAppPlatform, + resetRendererAppPlatformCacheForTests +} from './renderer-app-platform' + +function stubPlatformApi(get: () => { platform: NodeJS.Platform }): void { + ;(window as unknown as { api: unknown }).api = { platform: { get } } +} + +describe('getRendererAppPlatform', () => { + beforeEach(() => { + resetRendererAppPlatformCacheForTests() + }) + + afterEach(() => { + delete (window as unknown as { api?: unknown }).api + resetRendererAppPlatformCacheForTests() + }) + + it('crosses the preload bridge once no matter how many renders ask', () => { + const get = vi.fn(() => ({ platform: 'darwin' as NodeJS.Platform })) + stubPlatformApi(get) + + for (let index = 0; index < 500; index += 1) { + expect(getRendererAppPlatform()).toBe('darwin') + } + + expect(get).toHaveBeenCalledTimes(1) + }) + + it.each(['darwin', 'win32', 'linux'] as const)( + 'reports the preload platform verbatim on %s', + (platform) => { + stubPlatformApi(() => ({ platform })) + + expect(getRendererAppPlatform()).toBe(platform) + } + ) + + // Why: the web client injects its platform API after boot, so an early caller must + // not pin the user-agent guess for the rest of the session. + it('does not cache the user-agent fallback', () => { + // 'freebsd' is never a user-agent fallback answer, so the swap is unambiguous. + expect(getRendererAppPlatform()).not.toBe('freebsd') + + stubPlatformApi(() => ({ platform: 'freebsd' })) + + expect(getRendererAppPlatform()).toBe('freebsd') + }) +}) diff --git a/src/renderer/src/lib/renderer-app-platform.ts b/src/renderer/src/lib/renderer-app-platform.ts index e036d0690e3..f94b77ed2e9 100644 --- a/src/renderer/src/lib/renderer-app-platform.ts +++ b/src/renderer/src/lib/renderer-app-platform.ts @@ -1,7 +1,16 @@ +// Why: hot render paths call this per render; the preload answer never changes, so +// cache it. The user-agent fallback stays uncached because window.api can still be +// installing (the web client injects its own platform API after boot). +let cachedAppPlatform: NodeJS.Platform | undefined + export function getRendererAppPlatform(): NodeJS.Platform { + if (cachedAppPlatform) { + return cachedAppPlatform + } const preloadPlatform = typeof window === 'undefined' ? undefined : window.api?.platform?.get?.()?.platform if (preloadPlatform) { + cachedAppPlatform = preloadPlatform return preloadPlatform } const userAgent = typeof navigator === 'undefined' ? '' : navigator.userAgent @@ -16,3 +25,8 @@ export function getRendererAppPlatform(): NodeJS.Platform { } return 'win32' } + +/** Tests swap the window.api platform stub between cases; the real value never changes. */ +export function resetRendererAppPlatformCacheForTests(): void { + cachedAppPlatform = undefined +} diff --git a/src/renderer/src/store/slices/preflight.test.ts b/src/renderer/src/store/slices/preflight.test.ts index 386bdfdc59a..168f0099fcc 100644 --- a/src/renderer/src/store/slices/preflight.test.ts +++ b/src/renderer/src/store/slices/preflight.test.ts @@ -6,6 +6,7 @@ import type { Worktree } from '../../../../shared/worktree/types' import type { AppState } from '../types' import { createPreflightSlice } from './preflight' import { createRuntimeStatusSlice } from './runtime-status' +import { resetRendererAppPlatformCacheForTests } from '@/lib/renderer-app-platform' const preflightCheck = vi.fn() const callRuntimeRpc = vi.fn() @@ -62,6 +63,7 @@ function resetPreflightMocks(): void { preflightCheck.mockReset() callRuntimeRpc.mockReset() platformGet.mockReset().mockReturnValue({ platform: 'linux' }) + resetRendererAppPlatformCacheForTests() } function makeStatus(glabInstalled: boolean): PreflightStatus { From dff7f914016cde2f27174a7bedddd9e9602b05af Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:05:35 -0700 Subject: [PATCH 083/398] perf(sidebar): stop rebuilding per-card selector records on every store write (#18133) Zustand re-runs every mounted subscriber's selector on every store write. The per-worktree sidebar selectors built a fresh Record per call, so 15 visible cards x 6 reads x every write allocated a record each time even when nothing they read had changed. - Add createWorktreeRecordSelector: gates the build on the source slice identities, memoizes per worktree id, and carries the previous generation forward so a rebuild with equal contents keeps its reference. - Route the pane-title, live-PTY, layout-root and terminal-layout selectors through it, and return a shared frozen empty when a worktree has no tabs. - Swap useWorktreeAgentRows' inactive-branch `[]`/`{}` literals for the shared frozen constants so the `active` gate actually short-circuits on identity. - Identity-cache the sidebar pending-worktree-creation key list, which ran Object.values(...).map(...) from an always-mounted subscriber. - Drop `key={text}` from TruncatedSidebarLabel so a label change remeasures in place instead of remounting the span and rebuilding its ResizeObserver. - Remove the non-compositable `width` from the board drop indicator's will-change hint. --- src/renderer/src/assets/main.css | 4 +- .../sidebar/truncated-sidebar-label.test.tsx | 36 ++++++ .../sidebar/truncated-sidebar-label.tsx | 19 +-- .../sidebar/useWorktreeAgentRows.ts | 47 ++++++-- .../worktree-agent-row-selectors.test.ts | 113 ++++++++++++++++-- .../sidebar/worktree-agent-row-selectors.ts | 38 ++++-- .../worktree-card-status-inputs.test.ts | 74 ++++++++++++ .../sidebar/worktree-card-status-inputs.ts | 84 ++++++++----- .../pending-worktree-creation-keys.test.ts | 53 ++++++++ .../listing/pending-worktree-creation-keys.ts | 39 ++++++ .../worktree-list/listing/use-section-rows.ts | 13 +- .../sidebar/worktree-record-selector-cache.ts | 63 ++++++++++ 12 files changed, 507 insertions(+), 76 deletions(-) create mode 100644 src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.test.ts create mode 100644 src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.ts create mode 100644 src/renderer/src/components/sidebar/worktree-record-selector-cache.ts diff --git a/src/renderer/src/assets/main.css b/src/renderer/src/assets/main.css index 1b5f40ccdb4..8187fe496c7 100644 --- a/src/renderer/src/assets/main.css +++ b/src/renderer/src/assets/main.css @@ -1956,7 +1956,9 @@ html.native-shell .app-layout { transform 120ms cubic-bezier(0.2, 0.8, 0.2, 1), width 120ms cubic-bezier(0.2, 0.8, 0.2, 1), opacity 80ms ease-out; - will-change: transform, width, opacity; + /* Why no `width`: it is not compositable, so hinting it only pins a layer that + has to be re-rastered every frame of the transition anyway. */ + will-change: transform, opacity; } [data-workspace-board-card-drop-indicator='true']::before, diff --git a/src/renderer/src/components/sidebar/truncated-sidebar-label.test.tsx b/src/renderer/src/components/sidebar/truncated-sidebar-label.test.tsx index 27ed816c53a..1de1fac0140 100644 --- a/src/renderer/src/components/sidebar/truncated-sidebar-label.test.tsx +++ b/src/renderer/src/components/sidebar/truncated-sidebar-label.test.tsx @@ -105,4 +105,40 @@ describe('TruncatedSidebarLabel', () => { expect(container.textContent).toContain('feature/really-long-branch-name') expect(container.querySelector('[data-tooltip-content]')).toBeNull() }) + + // Why: worktree titles change on the hot store-write path. Remounting the + // span per text change tore down and rebuilt its ResizeObserver every time. + it('keeps one ResizeObserver across a label text change', async () => { + const originalResizeObserver = globalThis.ResizeObserver + let constructed = 0 + let disconnected = 0 + class CountingResizeObserver { + constructor(_callback: ResizeObserverCallback) { + constructed += 1 + } + observe(): void {} + unobserve(): void {} + disconnect(): void { + disconnected += 1 + } + } + globalThis.ResizeObserver = CountingResizeObserver as unknown as typeof ResizeObserver + + try { + await act(async () => { + root.render() + }) + expect(constructed).toBe(1) + + await act(async () => { + root.render() + }) + + expect(container.textContent).toBe('fix/short') + expect(constructed).toBe(1) + expect(disconnected).toBe(0) + } finally { + globalThis.ResizeObserver = originalResizeObserver + } + }) }) diff --git a/src/renderer/src/components/sidebar/truncated-sidebar-label.tsx b/src/renderer/src/components/sidebar/truncated-sidebar-label.tsx index c7afaa2e9b0..80955b5bfe8 100644 --- a/src/renderer/src/components/sidebar/truncated-sidebar-label.tsx +++ b/src/renderer/src/components/sidebar/truncated-sidebar-label.tsx @@ -1,4 +1,4 @@ -import React, { useCallback, useState } from 'react' +import React, { useCallback, useLayoutEffect, useState } from 'react' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { cn } from '@/lib/utils' @@ -23,6 +23,7 @@ export function TruncatedSidebarLabel({ tooltipSide = 'right', tooltipSideOffset = 8 }: TruncatedSidebarLabelProps): React.JSX.Element { + const nodeRef = React.useRef(null) const resizeObserverRef = React.useRef(null) const removeResizeListenerRef = React.useRef<(() => void) | null>(null) const [truncated, setTruncated] = useState(false) @@ -39,6 +40,7 @@ export function TruncatedSidebarLabel({ removeResizeListenerRef.current?.() removeResizeListenerRef.current = null + nodeRef.current = node if (!node) { measureTruncated(null) return @@ -60,14 +62,15 @@ export function TruncatedSidebarLabel({ [measureTruncated] ) + // Why: ResizeObserver does not fire when only the rendered text changes, but + // scrollWidth can. Remeasure in place rather than remounting the span, which + // would tear down and rebuild the observer on every title update. + useLayoutEffect(() => { + measureTruncated(nodeRef.current) + }, [measureTruncated, text]) + const label = ( - + {text} ) diff --git a/src/renderer/src/components/sidebar/useWorktreeAgentRows.ts b/src/renderer/src/components/sidebar/useWorktreeAgentRows.ts index b2f45a8a9a0..366bce5e5ed 100644 --- a/src/renderer/src/components/sidebar/useWorktreeAgentRows.ts +++ b/src/renderer/src/components/sidebar/useWorktreeAgentRows.ts @@ -5,22 +5,33 @@ import { applyAgentRowLineage } from '@/components/dashboard/agent-row-lineage' import { migrationUnsupportedToAgentStatusEntry } from '@/lib/migration-unsupported-agent-entry' import { useAppStore } from '@/store' import { + EMPTY_LIVE_PTY_IDS, + EMPTY_RUNTIME_PANE_TITLES, selectLivePtyIdsForWorktree, selectRuntimePaneTitlesForWorktree } from './worktree-card-status-inputs' import { buildWorktreeAgentRows } from './worktree-agent-rows' import { + EMPTY_LIVE_ENTRIES, + EMPTY_MIGRATION_UNSUPPORTED_ENTRIES, + EMPTY_RETAINED, + EMPTY_TERMINAL_LAYOUTS, selectLiveAgentStatusEntriesForWorktree, selectMigrationUnsupportedEntriesForWorktree, selectRuntimeAgentOrchestrationForWorktree, selectRetainedAgentEntriesForWorktree, selectTerminalLayoutsForWorktree } from './worktree-agent-row-selectors' +import { EMPTY_WORKTREE_AGENT_ORCHESTRATION } from './worktree-agent-orchestration-index' +import { EMPTY_TABS } from './WorktreeCardHelpers' import { createWorktreeAgentFreshnessSelector, EMPTY_WORKTREE_AGENT_FRESHNESS_SIGNATURE } from './worktree-agent-freshness-selector' +// Why frozen: shared by every inactive card, and callers only read these rows. +const EMPTY_AGENT_ROWS = Object.freeze([]) as unknown as DashboardAgentRow[] + export { buildWorktreeAgentRows } from './worktree-agent-rows' export { selectLiveAgentStatusEntriesForWorktree, @@ -44,35 +55,51 @@ export function useWorktreeAgentRows(worktreeId: string, active = true): Dashboa () => createWorktreeAgentFreshnessSelector(worktreeId), [worktreeId] ) - const tabs = useAppStore((s) => (active ? s.tabsByWorktree[worktreeId] : undefined)) + const tabs = useAppStore((s) => (active ? s.tabsByWorktree[worktreeId] : EMPTY_TABS)) // Why: narrow the subscriptions to only THIS worktree's entries via // useShallow. Subscribing to the whole agentStatusByPaneKey map would make // every on-screen card re-render on any agent-status update anywhere — // O(worktrees²) render amplification. Pre-filtering here means the card // only re-renders when something relevant to THIS worktree changes. const liveEntries = useAppStore( - useShallow((s) => (active ? selectLiveAgentStatusEntriesForWorktree(s, worktreeId) : [])) + useShallow((s) => + active ? selectLiveAgentStatusEntriesForWorktree(s, worktreeId) : EMPTY_LIVE_ENTRIES + ) ) // Why: keep the store selector limited to stable raw records. Converting // migration entries creates fresh objects with Date.now(), which breaks // useSyncExternalStore's cached-snapshot contract and can blank Electron. const migrationUnsupported = useAppStore( - useShallow((s) => (active ? selectMigrationUnsupportedEntriesForWorktree(s, worktreeId) : [])) + useShallow((s) => + active + ? selectMigrationUnsupportedEntriesForWorktree(s, worktreeId) + : EMPTY_MIGRATION_UNSUPPORTED_ENTRIES + ) ) const retained = useAppStore( - useShallow((s) => (active ? selectRetainedAgentEntriesForWorktree(s, worktreeId) : [])) + useShallow((s) => + active ? selectRetainedAgentEntriesForWorktree(s, worktreeId) : EMPTY_RETAINED + ) ) const runtimePaneTitlesByTabId = useAppStore( - useShallow((s) => (active ? selectRuntimePaneTitlesForWorktree(s, worktreeId) : {})) + useShallow((s) => + active ? selectRuntimePaneTitlesForWorktree(s, worktreeId) : EMPTY_RUNTIME_PANE_TITLES + ) ) const ptyIdsByTabId = useAppStore( - useShallow((s) => (active ? selectLivePtyIdsForWorktree(s, worktreeId) : {})) + useShallow((s) => (active ? selectLivePtyIdsForWorktree(s, worktreeId) : EMPTY_LIVE_PTY_IDS)) ) const terminalLayoutsByTabId = useAppStore( - useShallow((s) => (active ? selectTerminalLayoutsForWorktree(s, worktreeId) : {})) + useShallow((s) => + active ? selectTerminalLayoutsForWorktree(s, worktreeId) : EMPTY_TERMINAL_LAYOUTS + ) ) const runtimeAgentOrchestrationByPaneKey = useAppStore( - useShallow((s) => (active ? selectRuntimeAgentOrchestrationForWorktree(s, worktreeId) : {})) + useShallow((s) => + active + ? selectRuntimeAgentOrchestrationForWorktree(s, worktreeId) + : EMPTY_WORKTREE_AGENT_ORCHESTRATION + ) ) const agentFreshnessSignature = useAppStore((s) => active ? selectAgentFreshness(s) : EMPTY_WORKTREE_AGENT_FRESHNESS_SIGNATURE @@ -80,7 +107,7 @@ export function useWorktreeAgentRows(worktreeId: string, active = true): Dashboa return useMemo(() => { if (!active) { - return [] + return EMPTY_AGENT_ROWS } // Why: Date.now() is read inside the memo so stale-decay recalculates when // this worktree's freshness signature changes, even without new PTY data. @@ -97,7 +124,7 @@ export function useWorktreeAgentRows(worktreeId: string, active = true): Dashboa : liveEntries return applyAgentRowLineage( buildWorktreeAgentRows({ - tabs: tabs ?? [], + tabs: tabs ?? EMPTY_TABS, entries, retained, runtimePaneTitlesByTabId, diff --git a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.test.ts b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.test.ts index cbd3da64212..3e4e6defc6d 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.test.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.test.ts @@ -9,10 +9,15 @@ import type { RetainedAgentEntry } from '@/store/slices/agent-status' import { makePaneKey } from '../../../../shared/stable-pane-id' import { getLiveEntriesFullRebuildCountForTests } from './worktree-agent-live-index-patch' import { + EMPTY_LIVE_ENTRIES, + EMPTY_MIGRATION_UNSUPPORTED_ENTRIES, + EMPTY_RETAINED, + EMPTY_TERMINAL_LAYOUTS, selectLiveAgentStatusEntriesForWorktree, selectMigrationUnsupportedEntriesForWorktree, selectRuntimeAgentOrchestrationForWorktree, - selectRetainedAgentEntriesForWorktree + selectRetainedAgentEntriesForWorktree, + selectTerminalLayoutsForWorktree } from './worktree-agent-row-selectors' const PANE_KEY_1 = makePaneKey('tab-1', '22222222-2222-4222-8222-222222222222') @@ -94,8 +99,14 @@ describe('selectMigrationUnsupportedEntriesForWorktree', () => { describe('selectLiveAgentStatusEntriesForWorktree', () => { it('reuses unaffected worktree arrays when another worktree receives a same-state ping', () => { - const wt1Entry = makeEntry(PANE_KEY_1, 1000, { state: 'working', prompt: 'first' }) - const wt2Entry = makeEntry(PANE_KEY_2, 1000, { state: 'working', prompt: 'first' }) + const wt1Entry = makeEntry(PANE_KEY_1, 1000, { + state: 'working', + prompt: 'first' + }) + const wt2Entry = makeEntry(PANE_KEY_2, 1000, { + state: 'working', + prompt: 'first' + }) const state = { tabsByWorktree: { 'wt-1': [makeTab('tab-1')], @@ -192,8 +203,14 @@ describe('selectLiveAgentStatusEntriesForWorktree', () => { }) it('patches instead of full-rebuilding across within-state pings, and stays correct on transitions', () => { - const wt1Entry = makeEntry(PANE_KEY_1, 1000, { state: 'working', prompt: 'wt1 prompt' }) - const wt2Entry = makeEntry(PANE_KEY_2, 1000, { state: 'working', prompt: 'wt2 prompt' }) + const wt1Entry = makeEntry(PANE_KEY_1, 1000, { + state: 'working', + prompt: 'wt1 prompt' + }) + const wt2Entry = makeEntry(PANE_KEY_2, 1000, { + state: 'working', + prompt: 'wt2 prompt' + }) const baseState = { tabsByWorktree: { 'wt-1': [makeTab('tab-1')], @@ -256,7 +273,10 @@ describe('selectLiveAgentStatusEntriesForWorktree', () => { }) it('falls back to a full rebuild when a within-map update changes worktree attribution', () => { - const entry = makeEntry(PANE_KEY_1, 1000, { state: 'working', worktreeId: 'wt-1' }) + const entry = makeEntry(PANE_KEY_1, 1000, { + state: 'working', + worktreeId: 'wt-1' + }) const state = { // No tab membership: bucketing comes from entry.worktreeId attribution. tabsByWorktree: { 'wt-1': [], 'wt-2': [] }, @@ -276,7 +296,10 @@ describe('selectLiveAgentStatusEntriesForWorktree', () => { }) it('falls back to a full rebuild when a live entry completes with its tab gone', () => { - const entry = makeEntry(PANE_KEY_1, 1000, { state: 'working', worktreeId: 'wt-1' }) + const entry = makeEntry(PANE_KEY_1, 1000, { + state: 'working', + worktreeId: 'wt-1' + }) const state = { tabsByWorktree: { 'wt-1': [] }, agentStatusByPaneKey: { [PANE_KEY_1]: entry }, @@ -427,3 +450,79 @@ describe('selectRetainedAgentEntriesForWorktree', () => { expect(secondWt2[0]?.startedAt).toBe(1100) }) }) + +describe('selectTerminalLayoutsForWorktree', () => { + const layout = { + root: { type: 'leaf', leafId: '44444444-4444-4444-8444-444444444444' }, + activeLeafId: '44444444-4444-4444-8444-444444444444', + expandedLeafId: null, + ptyIdsByLeafId: {} + } as const + + // Why: every visible card re-runs this selector on every store write; a fresh + // record per call is an allocation per card per write. + it('returns one identity per store generation', () => { + const state = { + tabsByWorktree: { 'wt-1': [makeTab('tab-1')] }, + terminalLayoutsByTabId: { 'tab-1': layout } + } + + expect(selectTerminalLayoutsForWorktree(state, 'wt-1')).toBe( + selectTerminalLayoutsForWorktree(state, 'wt-1') + ) + }) + + it('returns the shared frozen empty for a worktree with no tabs', () => { + const state = { tabsByWorktree: {}, terminalLayoutsByTabId: {} } + + expect(selectTerminalLayoutsForWorktree(state, 'missing')).toBe(EMPTY_TERMINAL_LAYOUTS) + }) +}) + +// Why: the inline-agents hook short-circuits on `active`; the off branch has to +// hand back a shared identity or the gate allocates on every store write. +describe('inactive-card empty constants', () => { + it('are frozen and shared', () => { + for (const empty of [ + EMPTY_LIVE_ENTRIES, + EMPTY_MIGRATION_UNSUPPORTED_ENTRIES, + EMPTY_RETAINED, + EMPTY_TERMINAL_LAYOUTS + ]) { + expect(Object.isFrozen(empty)).toBe(true) + } + expect( + selectLiveAgentStatusEntriesForWorktree( + { + agentStatusByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + retainedAgentsByPaneKey: {}, + tabsByWorktree: {} + }, + 'missing' + ) + ).toBe(EMPTY_LIVE_ENTRIES) + expect( + selectMigrationUnsupportedEntriesForWorktree( + { + agentStatusByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + retainedAgentsByPaneKey: {}, + tabsByWorktree: {} + }, + 'missing' + ) + ).toBe(EMPTY_MIGRATION_UNSUPPORTED_ENTRIES) + expect( + selectRetainedAgentEntriesForWorktree( + { + agentStatusByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + retainedAgentsByPaneKey: {}, + tabsByWorktree: {} + }, + 'missing' + ) + ).toBe(EMPTY_RETAINED) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts index 84588c4296c..9fbbd882e03 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts @@ -13,11 +13,18 @@ import { recordLiveEntriesFullRebuild } from './worktree-agent-live-index-patch' import { selectWorktreeAgentOrchestration } from './worktree-agent-orchestration-index' +import { createWorktreeRecordSelector } from './worktree-record-selector-cache' import type { TerminalLayoutSnapshot } from '../../../../shared/terminal-tab-types' -const EMPTY_LIVE_ENTRIES: AgentStatusEntry[] = [] -const EMPTY_MIGRATION_UNSUPPORTED_ENTRIES: MigrationUnsupportedPtyEntry[] = [] -const EMPTY_RETAINED: RetainedAgentEntry[] = [] +// Why frozen and exported: card hooks return these from their inactive branch, +// so the identity has to be shared app-wide and safe from stray writes. +export const EMPTY_LIVE_ENTRIES = Object.freeze([]) as unknown as AgentStatusEntry[] +export const EMPTY_MIGRATION_UNSUPPORTED_ENTRIES = Object.freeze( + [] +) as unknown as MigrationUnsupportedPtyEntry[] +export const EMPTY_RETAINED = Object.freeze([]) as unknown as RetainedAgentEntry[] +export const EMPTY_TERMINAL_LAYOUTS: Record = + Object.freeze({}) // Why: selector unit tests often pass partial store mocks; production state // owns these maps, but missing mock maps should behave like empty slices. const EMPTY_RECORD = {} @@ -280,13 +287,20 @@ export function selectRuntimeAgentOrchestrationForWorktree( return selectWorktreeAgentOrchestration(state, worktreeId) } -export function selectTerminalLayoutsForWorktree( - state: Pick, - worktreeId: string -): Record { - const out: Record = {} - for (const tab of (state.tabsByWorktree ?? EMPTY_RECORD)[worktreeId] ?? []) { - out[tab.id] = (state.terminalLayoutsByTabId ?? EMPTY_RECORD)[tab.id] +export const selectTerminalLayoutsForWorktree = createWorktreeRecordSelector< + Pick, + Record +>({ + readSources: (state) => [ + state.tabsByWorktree ?? EMPTY_RECORD, + state.terminalLayoutsByTabId ?? EMPTY_RECORD + ], + empty: EMPTY_TERMINAL_LAYOUTS, + build: (state, worktreeId) => { + const out: Record = {} + for (const tab of (state.tabsByWorktree ?? EMPTY_RECORD)[worktreeId] ?? []) { + out[tab.id] = (state.terminalLayoutsByTabId ?? EMPTY_RECORD)[tab.id] + } + return out } - return out -} +}) diff --git a/src/renderer/src/components/sidebar/worktree-card-status-inputs.test.ts b/src/renderer/src/components/sidebar/worktree-card-status-inputs.test.ts index 7d366b845e8..59675017b2a 100644 --- a/src/renderer/src/components/sidebar/worktree-card-status-inputs.test.ts +++ b/src/renderer/src/components/sidebar/worktree-card-status-inputs.test.ts @@ -6,6 +6,9 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import { + EMPTY_LIVE_PTY_IDS, + EMPTY_RUNTIME_PANE_TITLES, + EMPTY_TERMINAL_LAYOUT_ROOTS, selectLivePtyIdsForWorktree, selectTerminalLayoutRootsForWorktree, selectTerminalLayoutRootsForWorktrees, @@ -145,4 +148,75 @@ describe('worktree card status input selectors', () => { ) ).toBe(true) }) + + // Why: zustand re-runs every mounted card's selector on every store write, so + // a fresh record per call multiplies by (visible cards x writes/sec). + it('returns one identity per store generation instead of rebuilding per call', () => { + const worktreeId = 'repo1::/path/wt1' + const state: SelectorState & LayoutRootSelectorState = { + tabsByWorktree: { + [worktreeId]: [makeTab('tab-1', worktreeId)] + }, + runtimePaneTitlesByTabId: { 'tab-1': { 0: 'codex [working]' } }, + ptyIdsByTabId: { 'tab-1': ['pty-1'] }, + terminalLayoutsByTabId: { + 'tab-1': makeLayout( + { type: 'leaf', leafId: '11111111-1111-4111-8111-111111111111' }, + 'pty-1' + ) + } + } + + expect(selectRuntimePaneTitlesForWorktree(state, worktreeId)).toBe( + selectRuntimePaneTitlesForWorktree(state, worktreeId) + ) + expect(selectLivePtyIdsForWorktree(state, worktreeId)).toBe( + selectLivePtyIdsForWorktree(state, worktreeId) + ) + expect(selectTerminalLayoutRootsForWorktree(state, worktreeId)).toBe( + selectTerminalLayoutRootsForWorktree(state, worktreeId) + ) + }) + + it('carries the same identity across unrelated pane-title and PTY churn', () => { + const worktreeId = 'repo1::/path/wt1' + const state: SelectorState = { + tabsByWorktree: { + [worktreeId]: [makeTab('tab-1', worktreeId)] + }, + runtimePaneTitlesByTabId: { 'tab-1': { 0: 'codex [working]' } }, + ptyIdsByTabId: { 'tab-1': ['pty-1'] } + } + const unrelatedUpdate: SelectorState = { + ...state, + runtimePaneTitlesByTabId: { + ...state.runtimePaneTitlesByTabId, + 'other-tab': { 0: 'claude [permission]' } + }, + ptyIdsByTabId: { ...state.ptyIdsByTabId, 'other-tab': ['pty-other'] } + } + + expect(selectRuntimePaneTitlesForWorktree(state, worktreeId)).toBe( + selectRuntimePaneTitlesForWorktree(unrelatedUpdate, worktreeId) + ) + expect(selectLivePtyIdsForWorktree(state, worktreeId)).toBe( + selectLivePtyIdsForWorktree(unrelatedUpdate, worktreeId) + ) + }) + + it('returns the shared frozen empty for a worktree with no tabs', () => { + const state: SelectorState & LayoutRootSelectorState = { + tabsByWorktree: {}, + runtimePaneTitlesByTabId: {}, + ptyIdsByTabId: {}, + terminalLayoutsByTabId: {} + } + + expect(selectRuntimePaneTitlesForWorktree(state, 'missing')).toBe(EMPTY_RUNTIME_PANE_TITLES) + expect(selectLivePtyIdsForWorktree(state, 'missing')).toBe(EMPTY_LIVE_PTY_IDS) + expect(selectTerminalLayoutRootsForWorktree(state, 'missing')).toBe(EMPTY_TERMINAL_LAYOUT_ROOTS) + expect(Object.isFrozen(EMPTY_RUNTIME_PANE_TITLES)).toBe(true) + expect(Object.isFrozen(EMPTY_LIVE_PTY_IDS)).toBe(true) + expect(Object.isFrozen(EMPTY_TERMINAL_LAYOUT_ROOTS)).toBe(true) + }) }) diff --git a/src/renderer/src/components/sidebar/worktree-card-status-inputs.ts b/src/renderer/src/components/sidebar/worktree-card-status-inputs.ts index 128afafe03d..cd9f5c4f7c5 100644 --- a/src/renderer/src/components/sidebar/worktree-card-status-inputs.ts +++ b/src/renderer/src/components/sidebar/worktree-card-status-inputs.ts @@ -1,9 +1,19 @@ import type { AppState } from '@/store/types' import type { TerminalPaneLayoutNode } from '../../../../shared/terminal-tab-types' +import { createWorktreeRecordSelector } from './worktree-record-selector-cache' // Why: these selectors return fresh maps whose top-level values preserve // underlying per-tab references, so callers must compare them shallowly. +// Why frozen: one instance is shared by every card, so a stray write would leak +// across worktrees instead of failing locally. +export const EMPTY_RUNTIME_PANE_TITLES: Record> = Object.freeze({}) +export const EMPTY_LIVE_PTY_IDS: Record = Object.freeze({}) +export const EMPTY_TERMINAL_LAYOUT_ROOTS: Record< + string, + TerminalPaneLayoutNode | null | undefined +> = Object.freeze({}) + type WorktreeCardStatusInputState = Pick & { tabsByWorktree: Record } @@ -12,44 +22,56 @@ type WorktreeCardLayoutRootInputState = Pick tabsByWorktree: Record } -export function selectRuntimePaneTitlesForWorktree( - state: WorktreeCardStatusInputState, - worktreeId: string -): Record> { - const out: Record> = {} - for (const tab of state.tabsByWorktree[worktreeId] ?? []) { - const paneTitles = state.runtimePaneTitlesByTabId[tab.id] - if (paneTitles) { - out[tab.id] = paneTitles +export const selectRuntimePaneTitlesForWorktree = createWorktreeRecordSelector< + WorktreeCardStatusInputState, + Record> +>({ + readSources: (state) => [state.tabsByWorktree, state.runtimePaneTitlesByTabId], + empty: EMPTY_RUNTIME_PANE_TITLES, + build: (state, worktreeId) => { + const out: Record> = {} + for (const tab of state.tabsByWorktree[worktreeId] ?? []) { + const paneTitles = state.runtimePaneTitlesByTabId[tab.id] + if (paneTitles) { + out[tab.id] = paneTitles + } } + return out } - return out -} +}) -export function selectLivePtyIdsForWorktree( - state: WorktreeCardStatusInputState, - worktreeId: string -): Record { - const out: Record = {} - for (const tab of state.tabsByWorktree[worktreeId] ?? []) { - const ids = state.ptyIdsByTabId[tab.id] - if (ids && ids.length > 0) { - out[tab.id] = ids +export const selectLivePtyIdsForWorktree = createWorktreeRecordSelector< + WorktreeCardStatusInputState, + Record +>({ + readSources: (state) => [state.tabsByWorktree, state.ptyIdsByTabId], + empty: EMPTY_LIVE_PTY_IDS, + build: (state, worktreeId) => { + const out: Record = {} + for (const tab of state.tabsByWorktree[worktreeId] ?? []) { + const ids = state.ptyIdsByTabId[tab.id] + if (ids && ids.length > 0) { + out[tab.id] = ids + } } + return out } - return out -} +}) -export function selectTerminalLayoutRootsForWorktree( - state: WorktreeCardLayoutRootInputState, - worktreeId: string -): Record { - const out: Record = {} - for (const tab of state.tabsByWorktree[worktreeId] ?? []) { - out[tab.id] = state.terminalLayoutsByTabId[tab.id]?.root +export const selectTerminalLayoutRootsForWorktree = createWorktreeRecordSelector< + WorktreeCardLayoutRootInputState, + Record +>({ + readSources: (state) => [state.tabsByWorktree, state.terminalLayoutsByTabId], + empty: EMPTY_TERMINAL_LAYOUT_ROOTS, + build: (state, worktreeId) => { + const out: Record = {} + for (const tab of state.tabsByWorktree[worktreeId] ?? []) { + out[tab.id] = state.terminalLayoutsByTabId[tab.id]?.root + } + return out } - return out -} +}) export function selectTerminalLayoutRootsForWorktrees( state: WorktreeCardLayoutRootInputState, diff --git a/src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.test.ts b/src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.test.ts new file mode 100644 index 00000000000..d45a7bf392f --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.test.ts @@ -0,0 +1,53 @@ +import { describe, expect, it } from 'vitest' +import type { AppState } from '@/store/types' +import { + EMPTY_PENDING_WORKTREE_CREATION_KEYS, + selectPendingWorktreeCreationKeys +} from './pending-worktree-creation-keys' + +type PendingCreations = AppState['pendingWorktreeCreations'] + +function makePending(creationId: string, repoId: string): PendingCreations[string] { + return { + creationId, + request: { repoId } + } as unknown as PendingCreations[string] +} + +describe('selectPendingWorktreeCreationKeys', () => { + // Why: this runs inside an always-mounted sidebar subscriber, so zustand + // re-evaluates it on every store write in the app. + it('returns the shared frozen empty when nothing is pending', () => { + const empty: PendingCreations = {} + + expect(selectPendingWorktreeCreationKeys(empty)).toBe(EMPTY_PENDING_WORKTREE_CREATION_KEYS) + expect(selectPendingWorktreeCreationKeys({})).toBe(EMPTY_PENDING_WORKTREE_CREATION_KEYS) + expect(selectPendingWorktreeCreationKeys(undefined)).toBe(EMPTY_PENDING_WORKTREE_CREATION_KEYS) + expect(Object.isFrozen(EMPTY_PENDING_WORKTREE_CREATION_KEYS)).toBe(true) + }) + + it('builds the key list once per slice identity', () => { + const pending: PendingCreations = { + 'creation-1': makePending('creation-1', 'repo with space') + } + + const first = selectPendingWorktreeCreationKeys(pending) + expect(first).toEqual(['creation-1 repo with space']) + expect(selectPendingWorktreeCreationKeys(pending)).toBe(first) + }) + + it('rebuilds when the slice is replaced', () => { + const before: PendingCreations = { + 'creation-1': makePending('creation-1', 'repo-1') + } + const after: PendingCreations = { + ...before, + 'creation-2': makePending('creation-2', 'repo-2') + } + + expect(selectPendingWorktreeCreationKeys(after)).toEqual([ + 'creation-1 repo-1', + 'creation-2 repo-2' + ]) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.ts b/src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.ts new file mode 100644 index 00000000000..857edd9c3c9 --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-list/listing/pending-worktree-creation-keys.ts @@ -0,0 +1,39 @@ +import type { AppState } from '@/store/types' + +// Why frozen: the sidebar row model is always mounted and this list is empty +// almost always, so one shared identity serves every read. +export const EMPTY_PENDING_WORKTREE_CREATION_KEYS: string[] = Object.freeze( + [] +) as unknown as string[] + +const keysBySource = new WeakMap() + +/** + * Flat `" "` keys for the pending-creation sidebar rows. + * + * Why identity-cached: this runs inside an always-mounted subscriber, so an + * unmemoized `Object.values(...).map(...)` allocated an array plus one template + * string per pending creation on every store write in the app. Keyed on the + * slice reference, so it only rebuilds when the slice itself is replaced. + * + * Split on the first space — creationId is a UUID (no space) so a + * space-containing repoId stays intact. + */ +export function selectPendingWorktreeCreationKeys( + pendingWorktreeCreations: AppState['pendingWorktreeCreations'] | undefined +): string[] { + if (!pendingWorktreeCreations) { + return EMPTY_PENDING_WORKTREE_CREATION_KEYS + } + const cached = keysBySource.get(pendingWorktreeCreations) + if (cached) { + return cached + } + const creations = Object.values(pendingWorktreeCreations) + const keys = + creations.length === 0 + ? EMPTY_PENDING_WORKTREE_CREATION_KEYS + : creations.map((creation) => `${creation.creationId} ${creation.request.repoId}`) + keysBySource.set(pendingWorktreeCreations, keys) + return keys +} diff --git a/src/renderer/src/components/sidebar/worktree-list/listing/use-section-rows.ts b/src/renderer/src/components/sidebar/worktree-list/listing/use-section-rows.ts index 16899dbc146..ddd91a23653 100644 --- a/src/renderer/src/components/sidebar/worktree-list/listing/use-section-rows.ts +++ b/src/renderer/src/components/sidebar/worktree-list/listing/use-section-rows.ts @@ -19,6 +19,7 @@ import { getEmptyProjectPlaceholderRepoIds } from '../../empty-project-placehold import { addHostSectionRows } from '../../host-section-rows' import { orderHostSectionOptions } from '../../host-section-order' import { buildSidebarHostOptions } from '../../sidebar-host-options' +import { selectPendingWorktreeCreationKeys } from './pending-worktree-creation-keys' type SectionRowsArgs = { groupBy: WorktreeGroupBy @@ -96,19 +97,17 @@ export function useSidebarSectionRows(args: SectionRowsArgs) { ) // Why: subscribe on a flat key array (useShallow) so progress ticks don't rebuild the whole row model. - // Split on first space — creationId is a UUID (no space) so a space-containing repoId stays intact. const pendingCreationKeys = useAppStore( - useShallow((s) => - Object.values(s.pendingWorktreeCreations ?? {}).map( - (creation) => `${creation.creationId} ${creation.request.repoId}` - ) - ) + useShallow((s) => selectPendingWorktreeCreationKeys(s.pendingWorktreeCreations)) ) const pendingCreations = useMemo( () => pendingCreationKeys.map((key) => { const separator = key.indexOf(' ') - return { creationId: key.slice(0, separator), repoId: key.slice(separator + 1) } + return { + creationId: key.slice(0, separator), + repoId: key.slice(separator + 1) + } }), [pendingCreationKeys] ) diff --git a/src/renderer/src/components/sidebar/worktree-record-selector-cache.ts b/src/renderer/src/components/sidebar/worktree-record-selector-cache.ts new file mode 100644 index 00000000000..220d2d074bf --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-record-selector-cache.ts @@ -0,0 +1,63 @@ +import { shallow } from 'zustand/shallow' + +type WorktreeRecordGeneration = { + sources: readonly unknown[] + carried: ReadonlyMap | null + byWorktreeId: Map +} + +function sameSources(previous: readonly unknown[], next: readonly unknown[]): boolean { + if (previous.length !== next.length) { + return false + } + for (let index = 0; index < next.length; index += 1) { + if (previous[index] !== next[index]) { + return false + } + } + return true +} + +/** + * Wraps a per-worktree record selector in a store-identity-keyed cache. + * + * Zustand re-runs every mounted subscriber's selector on every store write, so + * an unmemoized build allocates one record per visible card per write even when + * nothing it reads changed. Gating on the source slice identities collapses that + * to one build per worktree per generation; carrying the previous generation + * forward keeps the reference stable when a rebuild produces equal contents, so + * downstream `useShallow`/`useMemo` gates short-circuit on identity. + * + * The returned records are shared by every caller and must never be mutated. + */ +export function createWorktreeRecordSelector(options: { + readSources: (state: TState) => readonly unknown[] + build: (state: TState, worktreeId: string) => TValue + empty: TValue +}): (state: TState, worktreeId: string) => TValue { + let generation: WorktreeRecordGeneration | null = null + return (state, worktreeId) => { + const sources = options.readSources(state) + if (!generation || !sameSources(generation.sources, sources)) { + generation = { + sources, + carried: generation?.byWorktreeId ?? null, + byWorktreeId: new Map() + } + } + const cached = generation.byWorktreeId.get(worktreeId) + if (cached !== undefined) { + return cached + } + const built = options.build(state, worktreeId) + const carried = generation.carried?.get(worktreeId) + let value = built + if (Object.keys(built).length === 0) { + value = options.empty + } else if (carried && shallow(carried, built)) { + value = carried + } + generation.byWorktreeId.set(worktreeId, value) + return value + } +} From 89bd18990dc38bd7aef18a1b60e1ec5ec23fa89d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:05:41 -0700 Subject: [PATCH 084/398] perf(renderer): stop three always-mounted selectors rescanning the store (#18136) Zustand reruns every subscriber's selector on each store write. Three selectors did an O(N) scan of a store collection inside that path, so at 10 repos / 423 worktrees / 382 tabs they were paid thousands of times a second while the app sat idle. - getLocalWorktree / getLocalRuntimeRepoForWorktree now read the shared WeakMap indexes (getIndexedWorktreeById, getIndexedRepoMap) instead of `Object.values(worktreesByRepo).flat().find(...)` and `repos.find(...)`. SidebarTaskNavButton is always mounted and calls this on every write. - selectRepoByIdForActiveWorkspace caches its host-scoped resolution in a WeakMap keyed on the `repos` array, mirroring getIndexedRepoMap. - getProjectRuntimeSessionSummary memoizes per (tabsByWorktree, ptyIdsByTabId, agentStatusByPaneKey, repoId) and reuses the existing identity-cached getTabIdToWorktreeId index. --- .../repository-runtime-session-summary.ts | 45 ++- .../sidebar/worktree-agent-row-selectors.ts | 5 +- .../src/lib/local-preflight-context.ts | 22 +- .../always-mounted-selector-scan-cost.test.ts | 277 ++++++++++++++++++ src/renderer/src/store/selectors.ts | 70 +++-- 5 files changed, 389 insertions(+), 30 deletions(-) create mode 100644 src/renderer/src/store/always-mounted-selector-scan-cost.test.ts diff --git a/src/renderer/src/components/settings/repository-runtime-session-summary.ts b/src/renderer/src/components/settings/repository-runtime-session-summary.ts index 5dffcd3b1e2..62bfd7053e8 100644 --- a/src/renderer/src/components/settings/repository-runtime-session-summary.ts +++ b/src/renderer/src/components/settings/repository-runtime-session-summary.ts @@ -1,5 +1,6 @@ import type { AppState } from '../../store/types' import { getRepoIdFromWorktreeId } from '../../../../shared/worktree/id' +import { getTabIdToWorktreeId } from '../sidebar/worktree-agent-row-selectors' export type ProjectRuntimeSessionSummary = { liveTerminalCount: number @@ -11,16 +12,26 @@ type RuntimeSessionSummaryState = Pick< 'tabsByWorktree' | 'ptyIdsByTabId' | 'agentStatusByPaneKey' > +type SessionSummaryCache = { + ptyIdsByTabId: AppState['ptyIdsByTabId'] + agentStatusByPaneKey: AppState['agentStatusByPaneKey'] + byRepoId: Map +} + +// Why: one RepositoryPane per project reruns this on every store write, and each +// run walked every worktree bucket plus every agent-status pane. Key on the three +// input slices so unrelated writes reuse the answer instead of rescanning. +const sessionSummaryCache = new WeakMap() + function getTabIdFromPaneKey(paneKey: string): string | null { const separator = paneKey.indexOf(':') return separator > 0 ? paneKey.slice(0, separator) : null } -export function getProjectRuntimeSessionSummary( +function computeProjectRuntimeSessionSummary( state: RuntimeSessionSummaryState, repoId: string ): ProjectRuntimeSessionSummary { - const tabWorktreeIds = new Map() const projectWorktreeIds = new Set() let liveTerminalCount = 0 @@ -31,7 +42,6 @@ export function getProjectRuntimeSessionSummary( projectWorktreeIds.add(worktreeId) for (const tab of tabs) { - tabWorktreeIds.set(tab.id, worktreeId) const livePtyIds = new Set(state.ptyIdsByTabId[tab.id] ?? []) if (tab.ptyId) { livePtyIds.add(tab.ptyId) @@ -40,6 +50,9 @@ export function getProjectRuntimeSessionSummary( } } + // Rows outside this project resolve to a worktree the checks below reject, so + // the shared index answers the same question the repo-scoped map used to. + const tabWorktreeIds = getTabIdToWorktreeId(state.tabsByWorktree) let activeTaskCount = 0 for (const [paneKey, entry] of Object.entries(state.agentStatusByPaneKey)) { if (entry.state === 'done') { @@ -57,3 +70,29 @@ export function getProjectRuntimeSessionSummary( return { liveTerminalCount, activeTaskCount } } + +export function getProjectRuntimeSessionSummary( + state: RuntimeSessionSummaryState, + repoId: string +): ProjectRuntimeSessionSummary { + let cache = sessionSummaryCache.get(state.tabsByWorktree) + if ( + !cache || + cache.ptyIdsByTabId !== state.ptyIdsByTabId || + cache.agentStatusByPaneKey !== state.agentStatusByPaneKey + ) { + cache = { + ptyIdsByTabId: state.ptyIdsByTabId, + agentStatusByPaneKey: state.agentStatusByPaneKey, + byRepoId: new Map() + } + sessionSummaryCache.set(state.tabsByWorktree, cache) + } + const cached = cache.byRepoId.get(repoId) + if (cached) { + return cached + } + const summary = computeProjectRuntimeSessionSummary(state, repoId) + cache.byRepoId.set(repoId, summary) + return summary +} diff --git a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts index 9fbbd882e03..06ba3979ffe 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-row-selectors.ts @@ -80,7 +80,10 @@ export function reuseArrayIfEqual(previous: T[] | undefined, next: T[]): T[] return previous } -function getTabIdToWorktreeId( +// Why exported: the Settings -> Repositories runtime summary needs the same +// tab -> worktree index, and rebuilding it there would re-walk every tab bucket +// on each store write. +export function getTabIdToWorktreeId( tabsByWorktree: WorktreeAgentRowsState['tabsByWorktree'] ): Map { if (tabWorktreeIndexCache?.tabsByWorktree === tabsByWorktree) { diff --git a/src/renderer/src/lib/local-preflight-context.ts b/src/renderer/src/lib/local-preflight-context.ts index 98ac6c1059e..7988e4656a7 100644 --- a/src/renderer/src/lib/local-preflight-context.ts +++ b/src/renderer/src/lib/local-preflight-context.ts @@ -9,6 +9,7 @@ import { import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../../../shared/execution-host' import type { Repo } from '../../../shared/repo-types' import type { Worktree } from '../../../shared/worktree/types' +import { getIndexedRepoMap, getIndexedWorktreeById } from '@/store/worktree-repo-index' import { getProviderRuntimeContextKey } from './provider-runtime-context' import { getRendererAppPlatform } from './renderer-app-platform' import { @@ -36,6 +37,11 @@ type LocalProjectRuntimeState = Pick< 'activeRepoId' | 'activeWorktreeId' | 'projects' | 'repos' | 'settings' | 'worktreesByRepo' > +// Why: the shared indexes are WeakMap-keyed on slice identity, so a fresh `{}` +// or `[]` fallback would miss the cache on every read. +const EMPTY_WORKTREES_BY_REPO: AppState['worktreesByRepo'] = {} +const EMPTY_REPOS: AppState['repos'] = [] + type LocalProjectRuntimeWslContext = { wslAvailable?: boolean availableWslDistros?: readonly string[] | null @@ -120,7 +126,7 @@ export function getLocalRepoProjectExecutionRuntimeContext( return undefined } - const repo = (state.repos ?? []).find((entry) => entry.id === repoId) + const repo = getIndexedRepoMap(state.repos ?? EMPTY_REPOS).get(repoId) if (!isLocalRuntimeRepo(repo)) { return undefined } @@ -270,7 +276,7 @@ function getLocalRuntimeRepoForWorktree( worktree?: Pick | null ): Pick | undefined { const repoId = worktree?.repoId ?? state.activeRepoId - return repoId ? (state.repos ?? []).find((repo) => repo.id === repoId) : undefined + return repoId ? getIndexedRepoMap(state.repos ?? EMPTY_REPOS).get(repoId) : undefined } function isLocalRuntimeRepo( @@ -302,11 +308,13 @@ function getLocalWorktree( worktreeId?: string | null ): Pick | null { const targetWorktreeId = worktreeId ?? state.activeWorktreeId - return targetWorktreeId - ? (Object.values(state.worktreesByRepo ?? {}) - .flat() - .find((worktree) => worktree.id === targetWorktreeId) ?? null) - : null + if (!targetWorktreeId) { + return null + } + return ( + getIndexedWorktreeById(state.worktreesByRepo ?? EMPTY_WORKTREES_BY_REPO, targetWorktreeId) ?? + null + ) } function getLocalPreflightProjectId( diff --git a/src/renderer/src/store/always-mounted-selector-scan-cost.test.ts b/src/renderer/src/store/always-mounted-selector-scan-cost.test.ts new file mode 100644 index 00000000000..2baeb5958ef --- /dev/null +++ b/src/renderer/src/store/always-mounted-selector-scan-cost.test.ts @@ -0,0 +1,277 @@ +/** + * Zustand reruns every subscriber's selector on every store write, so an O(N) + * scan inside an always-mounted selector is paid thousands of times per second + * while the app is idle. These tests count property reads on the store rows to + * prove each selector builds its index once per snapshot instead of per read. + */ +import { describe, expect, it } from 'vitest' +import type { Repo } from '../../../shared/repo-types' +import type { TerminalTab } from '../../../shared/terminal-tab-types' +import type { Worktree } from '../../../shared/worktree/types' +import { WORKTREE_ID_SEPARATOR } from '../../../shared/worktree/id' +import { getLocalPreflightContext } from '../lib/local-preflight-context' +import { getProjectRuntimeSessionSummary } from '../components/settings/repository-runtime-session-summary' +import type { AppState } from './types' +import { selectRepoByIdForActiveWorkspace } from './selectors' + +// The user scale that motivated this: 10 repos, 423 worktrees, 382 open tabs. +const REPO_COUNT = 10 +const WORKTREES_PER_REPO = 42 +const TABS_PER_WORKTREE = 1 +const STORE_WRITES = 200 + +type ReadCounter = { count: number } + +function makeRepoRows(counter: ReadCounter): Repo[] { + return Array.from({ length: REPO_COUNT }, (_unused, index) => { + const id = `repo-${index}` + return { + get id() { + counter.count += 1 + return id + }, + path: `/tmp/repo-${index}`, + displayName: `repo-${index}`, + badgeColor: '#737373', + addedAt: 100, + kind: 'git' + } as Repo + }) +} + +function makeWorktreesByRepo(counter: ReadCounter): AppState['worktreesByRepo'] { + const worktreesByRepo: Record = {} + for (let repoIndex = 0; repoIndex < REPO_COUNT; repoIndex += 1) { + const repoId = `repo-${repoIndex}` + worktreesByRepo[repoId] = Array.from({ length: WORKTREES_PER_REPO }, (_unused, index) => { + const path = String.raw`\\wsl.localhost\Ubuntu\home\alice\wt-${repoIndex}-${index}` + const id = `${repoId}${WORKTREE_ID_SEPARATOR}${path}` + return { + get id() { + counter.count += 1 + return id + }, + repoId, + path + } as Worktree + }) + } + return worktreesByRepo +} + +/** The worst case for a first-wins linear scan: the last row of the last repo. */ +function lastWorktreeId(worktreesByRepo: AppState['worktreesByRepo']): string { + const lastBucket = Object.values(worktreesByRepo).at(-1) ?? [] + return (lastBucket.at(-1) as Worktree).id +} + +describe('local preflight context worktree lookup', () => { + it('builds the worktree index once instead of rescanning per store write', () => { + const worktreeReads: ReadCounter = { count: 0 } + const repoReads: ReadCounter = { count: 0 } + const worktreesByRepo = makeWorktreesByRepo(worktreeReads) + const activeWorktreeId = lastWorktreeId(worktreesByRepo) + const state = { + activeRepoId: `repo-${REPO_COUNT - 1}`, + activeWorktreeId, + repos: makeRepoRows(repoReads), + worktreesByRepo, + projects: [] + } as unknown as AppState + const rowCount = REPO_COUNT * WORKTREES_PER_REPO + worktreeReads.count = 0 + repoReads.count = 0 + + for (let write = 0; write < STORE_WRITES; write += 1) { + expect(getLocalPreflightContext(state, 'darwin')).toEqual({ + wslDistro: 'Ubuntu' + }) + } + + // One index build per snapshot, not one scan per store write. + expect(worktreeReads.count).toBeLessThanOrEqual(rowCount) + expect(repoReads.count).toBeLessThanOrEqual(REPO_COUNT) + }) + + it('rebuilds against a replacement snapshot', () => { + const counter: ReadCounter = { count: 0 } + const worktreesByRepo = makeWorktreesByRepo(counter) + const repos = makeRepoRows({ count: 0 }) + const activeWorktreeId = lastWorktreeId(worktreesByRepo) + const before = getLocalPreflightContext( + { + activeRepoId: 'repo-0', + activeWorktreeId, + repos, + worktreesByRepo + } as unknown as AppState, + 'darwin' + ) + expect(before).toEqual({ wslDistro: 'Ubuntu' }) + + const movedWorktree = { + id: activeWorktreeId, + repoId: `repo-${REPO_COUNT - 1}`, + path: String.raw`\\wsl.localhost\Debian\home\alice\moved` + } as Worktree + const after = getLocalPreflightContext( + { + activeRepoId: 'repo-0', + activeWorktreeId, + repos, + worktreesByRepo: { [`repo-${REPO_COUNT - 1}`]: [movedWorktree] } + } as unknown as AppState, + 'darwin' + ) + + expect(after).toEqual({ wslDistro: 'Debian' }) + }) +}) + +describe('selectRepoByIdForActiveWorkspace', () => { + function makeActiveWorkspaceState(counter: ReadCounter): AppState { + return { + repos: makeRepoRows(counter), + activeRepoId: 'repo-0', + // No repo row carries this host, so the fallback branch runs every time. + activeWorkspaceExecutionHostId: 'ssh:host-a' + } as unknown as AppState + } + + it('resolves the active-workspace host once per repos snapshot', () => { + const counter: ReadCounter = { count: 0 } + const state = makeActiveWorkspaceState(counter) + counter.count = 0 + + for (let write = 0; write < STORE_WRITES; write += 1) { + expect(selectRepoByIdForActiveWorkspace(state, 'repo-0')).toBeNull() + } + + // Worst case: the id-keyed map build plus one host-filter pass. + expect(counter.count).toBeLessThanOrEqual(REPO_COUNT * 3) + }) + + it('still prefers the row that carries the active workspace host', () => { + const localRepo = { + id: 'repo-0', + path: '/tmp/a', + displayName: 'a' + } as Repo + const sshRepo = { + id: 'repo-0', + path: '/tmp/a', + displayName: 'a', + connectionId: 'host-a' + } as Repo + const state = { + repos: [localRepo, sshRepo], + activeRepoId: 'repo-0', + activeWorkspaceExecutionHostId: 'ssh:host-a' + } as unknown as AppState + + expect(selectRepoByIdForActiveWorkspace(state, 'repo-0')).toBe(sshRepo) + expect(selectRepoByIdForActiveWorkspace(state, 'repo-0')).toBe(sshRepo) + }) + + it('returns an identical result for repeated reads of one snapshot', () => { + const state = makeActiveWorkspaceState({ count: 0 }) + expect(selectRepoByIdForActiveWorkspace(state, 'repo-0')).toBe( + selectRepoByIdForActiveWorkspace(state, 'repo-0') + ) + expect(selectRepoByIdForActiveWorkspace(state, 'repo-1')).toBe( + selectRepoByIdForActiveWorkspace(state, 'repo-1') + ) + }) +}) + +describe('project runtime session summary', () => { + function makeRuntimeSessionState(counter: ReadCounter): AppState { + const tabsByWorktree: Record = {} + for (let repoIndex = 0; repoIndex < REPO_COUNT; repoIndex += 1) { + for (let index = 0; index < WORKTREES_PER_REPO; index += 1) { + const worktreeId = `repo-${repoIndex}${WORKTREE_ID_SEPARATOR}/tmp/wt-${repoIndex}-${index}` + tabsByWorktree[worktreeId] = Array.from( + { length: TABS_PER_WORKTREE }, + (_unused, tabIndex) => { + const id = `tab-${repoIndex}-${index}-${tabIndex}` + return { + get id() { + counter.count += 1 + return id + }, + ptyId: `pty-${id}`, + worktreeId + } as TerminalTab + } + ) + } + } + return { + tabsByWorktree, + ptyIdsByTabId: {}, + agentStatusByPaneKey: {} + } as unknown as AppState + } + + it('reuses the tab index across repos and store writes', () => { + const counter: ReadCounter = { count: 0 } + const state = makeRuntimeSessionState(counter) + const tabCount = REPO_COUNT * WORKTREES_PER_REPO * TABS_PER_WORKTREE + counter.count = 0 + + // One RepositoryPane per project, all re-running on every store write. + for (let write = 0; write < STORE_WRITES; write += 1) { + for (let repoIndex = 0; repoIndex < REPO_COUNT; repoIndex += 1) { + expect(getProjectRuntimeSessionSummary(state, `repo-${repoIndex}`)).toEqual({ + liveTerminalCount: WORKTREES_PER_REPO * TABS_PER_WORKTREE, + activeTaskCount: 0 + }) + } + } + + // The shared tab index plus one pass over each repo's own tabs. + expect(counter.count).toBeLessThanOrEqual(tabCount * 3) + }) + + it('returns an identical summary for repeated reads of one snapshot', () => { + const state = makeRuntimeSessionState({ count: 0 }) + expect(getProjectRuntimeSessionSummary(state, 'repo-0')).toBe( + getProjectRuntimeSessionSummary(state, 'repo-0') + ) + }) + + it('recomputes when a tab slice is replaced', () => { + const state = makeRuntimeSessionState({ count: 0 }) + const first = getProjectRuntimeSessionSummary(state, 'repo-0') + const worktreeId = `repo-0${WORKTREE_ID_SEPARATOR}/tmp/wt-0-0` + const next = getProjectRuntimeSessionSummary( + { + ...state, + tabsByWorktree: { + [worktreeId]: [{ id: 'tab-new', ptyId: 'pty-new', worktreeId } as TerminalTab] + } + } as unknown as AppState, + 'repo-0' + ) + + expect(first.liveTerminalCount).toBe(WORKTREES_PER_REPO * TABS_PER_WORKTREE) + expect(next.liveTerminalCount).toBe(1) + }) + + it('counts running agents against the owning project only', () => { + const state = makeRuntimeSessionState({ count: 0 }) + const summary = getProjectRuntimeSessionSummary( + { + ...state, + agentStatusByPaneKey: { + 'tab-0-0-0:leaf': { state: 'working', tabId: 'tab-0-0-0' }, + 'tab-1-0-0:leaf': { state: 'working', tabId: 'tab-1-0-0' }, + 'tab-0-1-0:leaf': { state: 'done', tabId: 'tab-0-1-0' } + } + } as unknown as AppState, + 'repo-0' + ) + + expect(summary.activeTaskCount).toBe(1) + }) +}) diff --git a/src/renderer/src/store/selectors.ts b/src/renderer/src/store/selectors.ts index 77c124f562c..b30673a0b39 100644 --- a/src/renderer/src/store/selectors.ts +++ b/src/renderer/src/store/selectors.ts @@ -228,33 +228,65 @@ export const useActiveRepo = () => useAppStore(useShallow((s) => selectRepoByIdForActiveWorkspace(s, s.activeRepoId))) export const useRepoMap = () => useAppStore((s) => getCachedRepoMap(s.repos)) +type ActiveWorkspaceRepoState = Pick< + AppState, + 'repos' | 'activeRepoId' | 'activeWorkspaceExecutionHostId' +> + +// Why: mirrors getIndexedRepoMap above — the host-scoped branch re-filtered every +// repo on each store write even though its answer only moves when `repos` or the +// active workspace host does. +const activeWorkspaceRepoCache = new WeakMap>() + +function resolveRepoOnActiveWorkspaceHost( + state: ActiveWorkspaceRepoState, + repoId: string, + activeWorkspaceExecutionHostId: ExecutionHostId +): Repo | null { + const repoCandidates = state.repos.filter((candidate) => candidate.id === repoId) + const hostMatch = repoCandidates.find( + (candidate) => getRepoExecutionHostId(candidate) === activeWorkspaceExecutionHostId + ) + if (hostMatch) { + return hostMatch + } + // Why: withRepoHostOwnership keeps a paired-hub worktree on its own SSH host while the repo + // stays hub-owned, so that one mismatch still names the right repo; every other stays closed. + if (parseExecutionHostId(activeWorkspaceExecutionHostId)?.kind !== 'ssh') { + return null + } + const pairedHubRepos = repoCandidates.filter( + (candidate) => parseExecutionHostId(getRepoExecutionHostId(candidate))?.kind === 'runtime' + ) + return pairedHubRepos.length === 1 ? pairedHubRepos[0] : null +} + export function selectRepoByIdForActiveWorkspace( - state: Pick, + state: ActiveWorkspaceRepoState, repoId: string | null ): Repo | null { if (!repoId) { return null } const repo = getCachedRepoMap(state.repos).get(repoId) ?? null - if (repoId === state.activeRepoId && state.activeWorkspaceExecutionHostId) { - const repoCandidates = state.repos.filter((candidate) => candidate.id === repoId) - const hostMatch = repoCandidates.find( - (candidate) => getRepoExecutionHostId(candidate) === state.activeWorkspaceExecutionHostId - ) - if (hostMatch) { - return hostMatch - } - // Why: withRepoHostOwnership keeps a paired-hub worktree on its own SSH host while the repo - // stays hub-owned, so that one mismatch still names the right repo; every other stays closed. - if (parseExecutionHostId(state.activeWorkspaceExecutionHostId)?.kind !== 'ssh') { - return null - } - const pairedHubRepos = repoCandidates.filter( - (candidate) => parseExecutionHostId(getRepoExecutionHostId(candidate))?.kind === 'runtime' - ) - return pairedHubRepos.length === 1 ? pairedHubRepos[0] : null + const activeWorkspaceExecutionHostId = state.activeWorkspaceExecutionHostId + if (repoId !== state.activeRepoId || !activeWorkspaceExecutionHostId) { + return repo } - return repo + // The branch below only fires for the active repo, so the host id fully keys it. + let byHost = activeWorkspaceRepoCache.get(state.repos) + if (!byHost) { + byHost = new Map() + activeWorkspaceRepoCache.set(state.repos, byHost) + } + const cacheKey = `${activeWorkspaceExecutionHostId}\u0000${repoId}` + const cached = byHost.get(cacheKey) + if (cached !== undefined) { + return cached + } + const resolved = resolveRepoOnActiveWorkspaceHost(state, repoId, activeWorkspaceExecutionHostId) + byHost.set(cacheKey, resolved) + return resolved } export const useRepoById = (repoId: string | null) => From 9377214b4b135fb562775fad8a093a81293dbf70 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:07:58 -0700 Subject: [PATCH 085/398] perf(renderer): reconcile hydrated workspaces in one store write (#18150) * perf(renderer): reconcile hydrated workspaces in one store write Session hydration reconciled each workspace with its own set(), so a 193-workspace session fanned 193 writes out to every non-React store subscriber and re-spread three whole workspace-keyed maps per workspace. Fold the whole session into one patch, release the string-keyed terminal scroll-intent entries on pane close, and drop the per-workspace/per-tab reconnect debug logs. * fix(test): make the hydration fixture bucket switch exhaustive oxlint --type-aware flags the default arm; naming the editor case clears it. --- ...cile-hydrated-workspace-tab-models.test.ts | 5 +- ...reconcile-hydrated-workspace-tab-models.ts | 11 +- .../app-shell/use-app-startup-hydration.ts | 2 +- .../src/lib/pane-manager/pane-split-close.ts | 7 + .../terminal-scroll-intent-dom-tracking.ts | 3 +- ...rminal-scroll-intent-key-retention.test.ts | 159 ++++++++++++ .../terminal-scroll-intent-key-store.ts | 63 +++++ .../pane-manager/terminal-scroll-intent.ts | 35 ++- .../store-session-terminal-reconnect.test.ts | 7 +- ...ted-workspace-reconciliation-batch.test.ts | 148 +++++++++++ ...drated-workspace-reconciliation-fixture.ts | 242 ++++++++++++++++++ .../slices/tabs/tabs-reconciliation-batch.ts | 55 ++++ .../store/slices/tabs/tabs-reconciliation.ts | 69 +++-- .../store/slices/tabs/tabs-session-actions.ts | 42 ++- .../store/slices/tabs/tabs-slice-contract.ts | 2 + .../terminals/workspace-terminal-reconnect.ts | 19 +- 16 files changed, 805 insertions(+), 64 deletions(-) create mode 100644 src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-retention.test.ts create mode 100644 src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-store.ts create mode 100644 src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-batch.test.ts create mode 100644 src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-fixture.ts create mode 100644 src/renderer/src/store/slices/tabs/tabs-reconciliation-batch.ts diff --git a/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.test.ts b/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.test.ts index f4940db6ab5..cf33aed8e3e 100644 --- a/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.test.ts +++ b/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.test.ts @@ -2,13 +2,14 @@ import { describe, expect, it, vi } from 'vitest' import { reconcileHydratedWorkspaceTabModels } from './reconcile-hydrated-workspace-tab-models' describe('reconcileHydratedWorkspaceTabModels', () => { - it('reconciles every workspace the session hydrated, in session order', () => { + it('reconciles every workspace the session hydrated, in session order, in one call', () => { const reconcile = vi.fn() const reconciled = reconcileHydratedWorkspaceTabModels( { tabsByWorktree: { 'wt-a': [], 'wt-b': [], 'wt-c': [] } }, reconcile ) - expect(reconcile.mock.calls.map((call) => call[0])).toEqual(['wt-a', 'wt-b', 'wt-c']) + expect(reconcile).toHaveBeenCalledTimes(1) + expect(reconcile.mock.calls[0]?.[0]).toEqual(['wt-a', 'wt-b', 'wt-c']) expect(reconciled).toEqual(['wt-a', 'wt-b', 'wt-c']) }) diff --git a/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.ts b/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.ts index 84704b3eb2d..3da45244660 100644 --- a/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.ts +++ b/src/renderer/src/app-shell/reconcile-hydrated-workspace-tab-models.ts @@ -3,12 +3,13 @@ import type { WorkspaceSessionState } from '../../../shared/workspace-session-st /** Reconcile every workspace loaded during boot so stale unified-tab subsets converge. */ export function reconcileHydratedWorkspaceTabModels( session: Pick, - reconcileWorktreeTabModel: (worktreeId: string) => unknown + // Why batched: one store write for the whole session instead of one per + // workspace, each fanning out to every non-React store subscriber. + reconcileWorktreeTabModels: (worktreeIds: readonly string[]) => void ): string[] { - const reconciled: string[] = [] - for (const worktreeId of Object.keys(session.tabsByWorktree)) { - reconcileWorktreeTabModel(worktreeId) - reconciled.push(worktreeId) + const reconciled = Object.keys(session.tabsByWorktree) + if (reconciled.length > 0) { + reconcileWorktreeTabModels(reconciled) } return reconciled } diff --git a/src/renderer/src/app-shell/use-app-startup-hydration.ts b/src/renderer/src/app-shell/use-app-startup-hydration.ts index 447fbfb3780..b50ac86705e 100644 --- a/src/renderer/src/app-shell/use-app-startup-hydration.ts +++ b/src/renderer/src/app-shell/use-app-startup-hydration.ts @@ -198,7 +198,7 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta actions.hydrateBrowserSession(sessionRead.session, sessionHydrationOptions) reconcileHydratedWorkspaceTabModels( sessionRead.session, - useAppStore.getState().reconcileWorktreeTabModel + useAppStore.getState().reconcileWorktreeTabModels ) }) await timeRendererStartupStep('prepare-terminal-startup-restoration', () => diff --git a/src/renderer/src/lib/pane-manager/pane-split-close.ts b/src/renderer/src/lib/pane-manager/pane-split-close.ts index 80e35a18171..df725156655 100644 --- a/src/renderer/src/lib/pane-manager/pane-split-close.ts +++ b/src/renderer/src/lib/pane-manager/pane-split-close.ts @@ -21,6 +21,7 @@ import { disposeWebgl } from './pane-webgl-renderer' import { clearPendingSplitScrollRestore, scheduleSplitScrollRestore } from './pane-split-scroll' import { reattachWebglIfNeeded } from './pane-webgl-reattach' import { toPublicPane } from './pane-public-view' +import { releaseTerminalScrollIntentKey } from './terminal-scroll-intent-key-store' type MovedPaneSplitState = { pane: ManagedPaneInternal @@ -179,6 +180,12 @@ function teardownManagedPane( const closedLeafId = pane.leafId args.releasePaneIdentity(args.paneId) removePaneContainer(args, pane) + if (reason === 'close') { + // Leaf ids are minted UUIDs and never reused, so a closed leaf's scroll + // intent is unreachable. Detach/retire hand the leaf to a new host, which + // must still be able to restore it. + releaseTerminalScrollIntentKey(closedLeafId) + } const nextActivePaneId = activateReplacementPane(args) applyPaneOpacity(args.panes.values(), nextActivePaneId, args.styleOptions) for (const p of args.panes.values()) { diff --git a/src/renderer/src/lib/pane-manager/terminal-scroll-intent-dom-tracking.ts b/src/renderer/src/lib/pane-manager/terminal-scroll-intent-dom-tracking.ts index 9545615705c..81436039d88 100644 --- a/src/renderer/src/lib/pane-manager/terminal-scroll-intent-dom-tracking.ts +++ b/src/renderer/src/lib/pane-manager/terminal-scroll-intent-dom-tracking.ts @@ -8,7 +8,8 @@ import { syncTerminalScrollIntentFromViewport } from './terminal-scroll-intent' import { syncTerminalScrollIntentSoon } from './terminal-scroll-intent-settle' -import type { TerminalScrollIntentKey, TerminalScrollIntentTarget } from './terminal-scroll-intent' +import type { TerminalScrollIntentTarget } from './terminal-scroll-intent' +import type { TerminalScrollIntentKey } from './terminal-scroll-intent-key-store' import { isTerminalScrollIntentRebuildInFlight, onTerminalScrollIntentBufferRebuildComplete diff --git a/src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-retention.test.ts b/src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-retention.test.ts new file mode 100644 index 00000000000..e7da03090a1 --- /dev/null +++ b/src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-retention.test.ts @@ -0,0 +1,159 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { ManagedPaneInternal } from './pane-manager-types' +import type { TerminalLeafId } from '../../../../shared/stable-pane-id' + +const disposePane = vi.hoisted(() => + vi.fn((pane: ManagedPaneInternal, panes: Map) => { + panes.delete(pane.id) + }) +) + +vi.mock('./pane-tree-ops', () => ({ + captureScrollState: vi.fn(), + findPaneChildren: vi.fn(() => []), + promoteSibling: vi.fn(), + removeDividers: vi.fn(), + safeFit: vi.fn(), + wrapInSplit: vi.fn() +})) +vi.mock('./pane-lifecycle', () => ({ disposePane, openTerminal: vi.fn() })) +vi.mock('./pane-webgl-renderer', () => ({ disposeWebgl: vi.fn() })) +vi.mock('./pane-split-scroll', () => ({ + clearPendingSplitScrollRestore: vi.fn(), + scheduleSplitScrollRestore: vi.fn() +})) +vi.mock('./pane-drag-reorder', () => ({ updateMultiPaneState: vi.fn() })) +vi.mock('./pane-divider', () => ({ applyDividerStyles: vi.fn(), applyPaneOpacity: vi.fn() })) + +import { + closeManagedPane, + detachManagedPaneForExternalMove, + retireManagedPanePreservingPty +} from './pane-split-close' +import { + bindTerminalScrollIntentKey, + markTerminalPinnedViewport, + type TerminalScrollIntentTarget +} from './terminal-scroll-intent' +import { + readTerminalScrollIntentKeyRetention, + releaseTerminalScrollIntentKey +} from './terminal-scroll-intent-key-store' + +function leafIdAt(index: number): TerminalLeafId { + return `11111111-1111-4111-8111-${String(index).padStart(12, '0')}` as TerminalLeafId +} + +/** A pinned (non-bottom) viewport so a keyed intent is actually retained. */ +function createPinnedTerminal(): TerminalScrollIntentTarget { + return { + buffer: { active: { type: 'normal', viewportY: 3, baseY: 40 } } as never, + scrollToBottom: vi.fn(), + scrollToLine: vi.fn() + } +} + +function createPane(id: number, leafId: TerminalLeafId): ManagedPaneInternal { + const container = { + classList: { contains: (className: string) => className === 'pane' }, + dataset: { paneId: String(id), leafId }, + parentElement: null, + remove: vi.fn() + } + return { + id, + leafId, + stablePaneId: leafId, + terminal: { focus: vi.fn() } as never, + container: container as unknown as HTMLElement, + xtermContainer: {} as never, + linkTooltip: {} as never, + terminalGpuAcceleration: 'auto', + gpuRenderingEnabled: false, + webglAttachmentDeferred: false, + webglDisabledAfterContextLoss: false, + hasComplexScriptOutput: false, + webglAddon: null, + ligaturesAddon: null, + fitResizeObserver: null, + pendingObservedFitRafId: null, + fitAddon: {} as never, + searchAddon: {} as never, + serializeAddon: {} as never, + unicode11Addon: {} as never, + webLinksAddon: {} as never, + compositionHandler: null, + pendingSplitScrollState: null, + debugLabel: null + } +} + +function openPane(id: number, leafId: TerminalLeafId): ManagedPaneInternal { + const pane = createPane(id, leafId) + const terminal = createPinnedTerminal() + bindTerminalScrollIntentKey(terminal, leafId) + markTerminalPinnedViewport(terminal) + return pane +} + +function closeArgs(pane: ManagedPaneInternal, panes: Map) { + return { + paneId: pane.id, + activePaneId: null, + panes, + root: {} as HTMLElement, + styleOptions: {}, + managerOptions: { linkOpenHint: () => '' }, + getDragCallbacks: () => ({}) as never, + releasePaneIdentity: vi.fn(), + setActivePaneId: vi.fn() + } +} + +describe('terminal scroll intent key retention', () => { + beforeEach(() => { + vi.clearAllMocks() + }) + + it('returns both keyed maps to baseline after 200 pane open/close cycles', () => { + const baseline = readTerminalScrollIntentKeyRetention() + + for (let index = 0; index < 200; index += 1) { + const leafId = leafIdAt(index) + const pane = openPane(index + 1, leafId) + const panes = new Map([[pane.id, pane]]) + // A pinned pane must actually retain its keyed intent while it is open. + expect(readTerminalScrollIntentKeyRetention()).toEqual({ + intents: baseline.intents + 1, + bindings: baseline.bindings + 1 + }) + // Keep a second pane so close is never the last-pane no-op path. + const survivor = createPane(10_000 + index, leafIdAt(10_000 + index)) + panes.set(survivor.id, survivor) + closeManagedPane(closeArgs(pane, panes)) + } + + expect(readTerminalScrollIntentKeyRetention()).toEqual(baseline) + }) + + it('keeps the keyed intent when the leaf is handed to a new host', () => { + const baseline = readTerminalScrollIntentKeyRetention() + + for (const teardown of [detachManagedPaneForExternalMove, retireManagedPanePreservingPty]) { + const leafId = leafIdAt(9000 + Number(teardown === retireManagedPanePreservingPty)) + const pane = openPane(9000, leafId) + const survivor = createPane(9001, leafIdAt(9001)) + const panes = new Map([ + [pane.id, pane], + [survivor.id, survivor] + ]) + + expect(teardown(closeArgs(pane, panes))).toBe(true) + expect(readTerminalScrollIntentKeyRetention().intents).toBe(baseline.intents + 1) + + releaseTerminalScrollIntentKey(leafId) + } + + expect(readTerminalScrollIntentKeyRetention()).toEqual(baseline) + }) +}) diff --git a/src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-store.ts b/src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-store.ts new file mode 100644 index 00000000000..23d2d5d8af1 --- /dev/null +++ b/src/renderer/src/lib/pane-manager/terminal-scroll-intent-key-store.ts @@ -0,0 +1,63 @@ +import type { TerminalScrollBufferType } from './terminal-scroll-buffer-snapshot' + +export type TerminalScrollIntentKind = 'followOutput' | 'pinnedViewport' +export type TerminalScrollIntentKey = string + +export type TerminalScrollIntent = { + kind: TerminalScrollIntentKind + bufferType: TerminalScrollBufferType + viewportY: number + baseY: number + revision: number +} + +// Keyed by stable leaf id, so a pin outlives the xterm instance that recorded +// it (workspace switch, keyed remount). Unlike the WeakMap-keyed siblings in +// terminal-scroll-intent.ts these hold a strong string key, so a leaf that is +// gone for good must be released explicitly — see releaseTerminalScrollIntentKey. +const terminalScrollIntentByKey = new Map() +const terminalScrollIntentBindingByKey = new Map() + +export function readKeyedTerminalScrollIntent( + key: TerminalScrollIntentKey +): TerminalScrollIntent | undefined { + return terminalScrollIntentByKey.get(key) +} + +export function writeKeyedTerminalScrollIntent( + key: TerminalScrollIntentKey, + intent: TerminalScrollIntent +): void { + terminalScrollIntentByKey.set(key, intent) +} + +export function readKeyedTerminalScrollIntentBinding( + key: TerminalScrollIntentKey +): number | undefined { + return terminalScrollIntentBindingByKey.get(key) +} + +export function writeKeyedTerminalScrollIntentBinding( + key: TerminalScrollIntentKey, + binding: number +): void { + terminalScrollIntentBindingByKey.set(key, binding) +} + +/** + * Drops the keyed intent for a leaf that is gone for good. Only safe on a real + * close: plain disposal (workspace switch, keyed remount, manager destroy) + * relies on these entries to restore the pin when the leaf mounts again. + */ +export function releaseTerminalScrollIntentKey(key: TerminalScrollIntentKey): void { + terminalScrollIntentByKey.delete(key) + terminalScrollIntentBindingByKey.delete(key) +} + +/** Retention probe for the leak tests. */ +export function readTerminalScrollIntentKeyRetention(): { intents: number; bindings: number } { + return { + intents: terminalScrollIntentByKey.size, + bindings: terminalScrollIntentBindingByKey.size + } +} diff --git a/src/renderer/src/lib/pane-manager/terminal-scroll-intent.ts b/src/renderer/src/lib/pane-manager/terminal-scroll-intent.ts index bd0b63611a6..9e1e32fcb81 100644 --- a/src/renderer/src/lib/pane-manager/terminal-scroll-intent.ts +++ b/src/renderer/src/lib/pane-manager/terminal-scroll-intent.ts @@ -3,6 +3,17 @@ import { notifyTerminalFollowOutputWaiters } from './terminal-follow-output-waiters' import { isTerminalScrollIntentRebuildInFlight } from './terminal-scroll-intent-rebuild' +import { + readKeyedTerminalScrollIntent, + readKeyedTerminalScrollIntentBinding, + writeKeyedTerminalScrollIntent, + writeKeyedTerminalScrollIntentBinding +} from './terminal-scroll-intent-key-store' +import type { + TerminalScrollIntent, + TerminalScrollIntentKey, + TerminalScrollIntentKind +} from './terminal-scroll-intent-key-store' import { clampTerminalViewportY, isTerminalViewportAtBottom, @@ -11,24 +22,12 @@ import { type TerminalScrollBufferType } from './terminal-scroll-buffer-snapshot' -type TerminalScrollIntentKind = 'followOutput' | 'pinnedViewport' - export type TerminalScrollIntentTarget = { buffer?: Parameters[0]['buffer'] scrollToBottom?: () => void scrollToLine?: (line: number) => void } -export type TerminalScrollIntentKey = string - -type TerminalScrollIntent = { - kind: TerminalScrollIntentKind - bufferType: TerminalScrollBufferType - viewportY: number - baseY: number - revision: number -} - export type TerminalStructuralScrollIntentSnapshot = { kind: TerminalScrollIntentKind bufferType: TerminalScrollBufferType @@ -53,8 +52,6 @@ const terminalScrollIntentKeyByTerminal = new WeakMap< TerminalScrollIntentKey >() const terminalScrollIntentKeyBindingByTerminal = new WeakMap() -const terminalScrollIntentByKey = new Map() -const terminalScrollIntentBindingByKey = new Map() let nextTerminalScrollIntentRevision = 1 let nextTerminalScrollIntentKeyBinding = 1 @@ -92,7 +89,7 @@ function writeIntentSnapshot( terminalScrollIntentByTerminal.set(terminal, intent) const key = terminalScrollIntentKeyByTerminal.get(terminal) if (key) { - terminalScrollIntentByKey.set(key, intent) + writeKeyedTerminalScrollIntent(key, intent) } if (kind === 'followOutput') { notifyTerminalFollowOutputWaiters(terminal) @@ -106,7 +103,7 @@ function readStoredIntent(terminal: TerminalScrollIntentTarget): TerminalScrollI return terminalIntent } const key = terminalScrollIntentKeyByTerminal.get(terminal) - return key ? terminalScrollIntentByKey.get(key) : undefined + return key ? readKeyedTerminalScrollIntent(key) : undefined } export function bindTerminalScrollIntentKey( @@ -120,8 +117,8 @@ export function bindTerminalScrollIntentKey( const binding = nextTerminalScrollIntentKeyBinding nextTerminalScrollIntentKeyBinding += 1 terminalScrollIntentKeyBindingByTerminal.set(terminal, binding) - terminalScrollIntentBindingByKey.set(key, binding) - const existing = terminalScrollIntentByKey.get(key) + writeKeyedTerminalScrollIntentBinding(key, binding) + const existing = readKeyedTerminalScrollIntent(key) if (existing) { terminalScrollIntentByTerminal.set(terminal, existing) } @@ -137,7 +134,7 @@ export function isTerminalScrollIntentKeyBindingCurrent( } return ( terminalScrollIntentKeyBindingByTerminal.get(terminal) === - terminalScrollIntentBindingByKey.get(key) + readKeyedTerminalScrollIntentBinding(key) ) } diff --git a/src/renderer/src/store/slices/store-session-terminal-reconnect.test.ts b/src/renderer/src/store/slices/store-session-terminal-reconnect.test.ts index 81c6350680d..c0f05124eda 100644 --- a/src/renderer/src/store/slices/store-session-terminal-reconnect.test.ts +++ b/src/renderer/src/store/slices/store-session-terminal-reconnect.test.ts @@ -599,9 +599,12 @@ describe('reconnectPersistedTerminals', () => { } const sshConnectionStates = new Map([[targetId, currentAuthorityState]]) const originalGet = sshConnectionStates.get.bind(sshConnectionStates) + // Rotate from the first read after the entry authority check, so the + // pre-publication check sees the new generation whatever number of + // intermediate reads the reconnect pass happens to make. sshConnectionStates.get = ((key: string) => { authorityReads += 1 - return authorityReads >= 4 ? rotatedAuthorityState : originalGet(key) + return authorityReads >= 2 ? rotatedAuthorityState : originalGet(key) }) as typeof sshConnectionStates.get store.setState({ repos: [ @@ -636,7 +639,7 @@ describe('reconnectPersistedTerminals', () => { }) const after = store.getState() - expect(authorityReads).toBeGreaterThanOrEqual(4) + expect(authorityReads).toBeGreaterThanOrEqual(2) expect(after.tabsByWorktree).toBe(before.tabsByWorktree) expect(after.ptyIdsByTabId).toBe(before.ptyIdsByTabId) expect(after.pendingReconnectWorktreeIds).toBe(before.pendingReconnectWorktreeIds) diff --git a/src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-batch.test.ts b/src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-batch.test.ts new file mode 100644 index 00000000000..e6a7bfe7002 --- /dev/null +++ b/src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-batch.test.ts @@ -0,0 +1,148 @@ +import { describe, it, expect, vi } from 'vitest' +import type * as AgentStatusModule from '@/lib/agent-status' +import { createTabsSliceMockApi } from '../tabs-slice-test-harness' +import { createTestStore } from '../store-test-helpers' +import { + buildHydratedWorkspaceFixture, + HYDRATED_TAB_COUNT, + HYDRATED_WORKSPACE_COUNT, + RECONCILIATION_WRITABLE_KEYS +} from './hydrated-workspace-reconciliation-fixture' + +vi.mock('sonner', () => ({ toast: { info: vi.fn(), success: vi.fn(), error: vi.fn() } })) +vi.mock('@/lib/agent-status', async (importOriginal) => { + const actual = await importOriginal() + return { ...actual, detectAgentStatusFromTitle: vi.fn().mockReturnValue(null) } +}) + +createTabsSliceMockApi() + +// Approximates the renderer's live non-React store-subscriber population. +const SUBSCRIBER_COUNT = 1200 + +type Store = ReturnType + +function hydrateFixtureStore(): { store: Store; workspaceIds: string[] } { + const fixture = buildHydratedWorkspaceFixture() + expect(fixture.workspaceIds).toHaveLength(HYDRATED_WORKSPACE_COUNT) + expect(fixture.tabCount).toBe(HYDRATED_TAB_COUNT) + const store = createTestStore() + store.setState(fixture.state) + return { store, workspaceIds: fixture.workspaceIds } +} + +function countNotifications(store: Store, run: () => void): number { + let notifications = 0 + const unsubscribes = Array.from({ length: SUBSCRIBER_COUNT }, () => + store.subscribe(() => { + notifications += 1 + }) + ) + try { + run() + } finally { + for (const unsubscribe of unsubscribes) { + unsubscribe() + } + } + return notifications +} + +/** Group ids for restored legacy terminals are minted, so pin them to compare states. */ +function withDeterministicUuids(run: () => T): T { + let counter = 0 + const spy = vi.spyOn(globalThis.crypto, 'randomUUID').mockImplementation(() => { + counter += 1 + return `00000000-0000-4000-8000-${String(counter).padStart(12, '0')}` + }) + try { + return run() + } finally { + spy.mockRestore() + } +} + +function reconciliationSnapshot(store: Store): Record { + const state = store.getState() as unknown as Record + return Object.fromEntries(RECONCILIATION_WRITABLE_KEYS.map((key) => [key, state[key]])) +} + +describe('whole-session workspace tab-model reconciliation', () => { + it('collapses a 193-workspace hydration to one store write', () => { + const perWorkspace = hydrateFixtureStore() + const perWorkspaceNotifications = countNotifications(perWorkspace.store, () => { + for (const worktreeId of perWorkspace.workspaceIds) { + perWorkspace.store.getState().reconcileWorktreeTabModel(worktreeId) + } + }) + + const batched = hydrateFixtureStore() + const batchedNotifications = countNotifications(batched.store, () => { + batched.store.getState().reconcileWorktreeTabModels(batched.workspaceIds) + }) + + // The fixture must actually be write-heavy, or the collapse proves nothing. + expect(perWorkspaceNotifications / SUBSCRIBER_COUNT).toBeGreaterThan(100) + expect(batchedNotifications).toBe(SUBSCRIBER_COUNT) + }) + + it('leaves the store byte-identical to the per-workspace path', () => { + const perWorkspace = hydrateFixtureStore() + withDeterministicUuids(() => { + for (const worktreeId of perWorkspace.workspaceIds) { + perWorkspace.store.getState().reconcileWorktreeTabModel(worktreeId) + } + }) + const expected = reconciliationSnapshot(perWorkspace.store) + + const batched = hydrateFixtureStore() + withDeterministicUuids(() => { + batched.store.getState().reconcileWorktreeTabModels(batched.workspaceIds) + }) + const actual = reconciliationSnapshot(batched.store) + + expect(actual).toEqual(expected) + // Object key order is observable through Object.keys/entries iteration in + // selectors and session serialization, so equal values are not enough. + for (const key of RECONCILIATION_WRITABLE_KEYS) { + const actualValue = actual[key] + const expectedValue = expected[key] + if (actualValue && typeof actualValue === 'object' && !Array.isArray(actualValue)) { + expect([key, Object.keys(actualValue)]).toEqual([key, Object.keys(expectedValue as object)]) + } + } + }) + + it('re-reconciling the batched result is a no-op, as it is for the per-workspace path', () => { + const { store, workspaceIds } = hydrateFixtureStore() + store.getState().reconcileWorktreeTabModels(workspaceIds) + const settled = reconciliationSnapshot(store) + + const notifications = countNotifications(store, () => { + store.getState().reconcileWorktreeTabModels(workspaceIds) + }) + + expect(notifications).toBe(0) + expect(reconciliationSnapshot(store)).toEqual(settled) + }) + + it('indexes openFiles once instead of rescanning it per workspace', () => { + const { store, workspaceIds } = hydrateFixtureStore() + const openFiles = store.getState().openFiles + let scans = 0 + store.setState({ + openFiles: new Proxy(openFiles, { + get(target, property, receiver) { + if (property === 'filter') { + scans += 1 + } + return Reflect.get(target, property, receiver) + } + }) + }) + + store.getState().reconcileWorktreeTabModels(workspaceIds) + + expect(scans).toBe(0) + }) +}) diff --git a/src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-fixture.ts b/src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-fixture.ts new file mode 100644 index 00000000000..30ba7be1633 --- /dev/null +++ b/src/renderer/src/store/slices/tabs/hydrated-workspace-reconciliation-fixture.ts @@ -0,0 +1,242 @@ +import type { AppState } from '../../types' +import type { Tab, TabGroup } from '../../../../../shared/tab-types' +import type { TerminalTab } from '../../../../../shared/terminal-tab-types' +import type { OpenFile } from '../editor' + +/** + * Session shape measured on a real heavy profile: 193 workspaces / 382 tabs, + * spread over every reconciliation outcome (no-op, dropped tab, restored legacy + * runtime terminal, orphan sweep, editor tab) so a whole-session fold exercises + * both the workspace-scoped maps and the store-global ones workspaces share. + */ +export const HYDRATED_WORKSPACE_COUNT = 193 +export const HYDRATED_TAB_COUNT = 382 + +const BUCKETS = ['stable', 'stale', 'legacy', 'orphan', 'editor'] as const +type Bucket = (typeof BUCKETS)[number] + +export type HydratedWorkspaceFixture = { + workspaceIds: string[] + tabCount: number + state: Partial +} + +type FixtureDraft = { + unifiedTabsByWorktree: Record + groupsByWorktree: Record + activeGroupIdByWorktree: Record + tabsByWorktree: Record + activeTabIdByWorktree: Record + tabBarOrderByWorktree: Record + ptyIdsByTabId: Record + pendingReconnectPtyIdByTabId: Record + unreadTerminalTabs: Record + cacheTimerByKey: Record + openFiles: OpenFile[] +} + +function unifiedTab(id: string, worktreeId: string, groupId: string, over: Partial): Tab { + return { + id, + entityId: id, + groupId, + worktreeId, + contentType: 'terminal', + label: id, + customLabel: null, + color: null, + sortOrder: 0, + createdAt: 1, + ...over + } +} + +function runtimeTab(id: string, worktreeId: string, over: Partial): TerminalTab { + return { + id, + ptyId: `pty-${id}`, + worktreeId, + title: id, + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 1, + ...over + } +} + +function setGroup(draft: FixtureDraft, worktreeId: string, groupId: string, tabs: Tab[]): void { + draft.unifiedTabsByWorktree[worktreeId] = tabs + draft.groupsByWorktree[worktreeId] = [ + { + id: groupId, + worktreeId, + activeTabId: tabs[0]?.id ?? null, + tabOrder: tabs.map((tab) => tab.id) + } + ] +} + +/** Live terminal rows that reconcile to a no-op patch. */ +function buildStableWorkspace(draft: FixtureDraft, worktreeId: string, groupId: string): number { + const live = `${worktreeId}#live` + setGroup(draft, worktreeId, groupId, [unifiedTab(live, worktreeId, groupId, {})]) + draft.tabsByWorktree[worktreeId] = [runtimeTab(live, worktreeId, {})] + return 1 +} + +/** A unified row with no runtime backing: dropped, and its store-global unread flag with it. */ +function buildStaleWorkspace(draft: FixtureDraft, worktreeId: string, groupId: string): number { + const live = `${worktreeId}#live` + const stale = `${worktreeId}#stale` + setGroup(draft, worktreeId, groupId, [ + unifiedTab(live, worktreeId, groupId, {}), + unifiedTab(stale, worktreeId, groupId, { sortOrder: 1 }) + ]) + draft.tabsByWorktree[worktreeId] = [runtimeTab(live, worktreeId, {})] + draft.unreadTerminalTabs[stale] = true + return 2 +} + +/** Live only in a reconnect map: restored into the unified model, minting a group and layout. */ +function buildLegacyWorkspace(draft: FixtureDraft, worktreeId: string): number { + const legacy = `${worktreeId}#legacy` + draft.tabsByWorktree[worktreeId] = [runtimeTab(legacy, worktreeId, { ptyId: null })] + draft.pendingReconnectPtyIdByTabId[legacy] = `session-${legacy}` + draft.ptyIdsByTabId[legacy] = [] + draft.unifiedTabsByWorktree[worktreeId] = [] + draft.groupsByWorktree[worktreeId] = [] + draft.activeTabIdByWorktree[worktreeId] = legacy + return 1 +} + +/** No PTY and no unified row: swept, which writes the store-global per-tab maps. */ +function buildOrphanWorkspace(draft: FixtureDraft, worktreeId: string): number { + const orphan = `${worktreeId}#orphan` + draft.tabsByWorktree[worktreeId] = [runtimeTab(orphan, worktreeId, { ptyId: null })] + draft.tabBarOrderByWorktree[worktreeId] = [orphan] + draft.activeTabIdByWorktree[worktreeId] = orphan + draft.cacheTimerByKey[`${orphan}:git`] = 1 + draft.unifiedTabsByWorktree[worktreeId] = [] + draft.groupsByWorktree[worktreeId] = [] + return 1 +} + +/** One editor tab backed by an open file, one that is not — the openFiles rescan path. */ +function buildEditorWorkspace(draft: FixtureDraft, worktreeId: string, groupId: string): number { + const fileId = `${worktreeId}#file` + const ghost = `${worktreeId}#ghost-file` + setGroup(draft, worktreeId, groupId, [ + unifiedTab(fileId, worktreeId, groupId, { contentType: 'editor' }), + unifiedTab(ghost, worktreeId, groupId, { contentType: 'editor', sortOrder: 1 }) + ]) + draft.tabsByWorktree[worktreeId] = [] + draft.openFiles.push({ + id: fileId, + filePath: `/tmp/${worktreeId}/${fileId}`, + relativePath: fileId, + worktreeId, + language: 'typescript', + isDirty: false, + mode: 'edit' + }) + return 2 +} + +function buildWorkspace( + draft: FixtureDraft, + bucket: Bucket, + worktreeId: string, + groupId: string +): number { + switch (bucket) { + case 'stable': + return buildStableWorkspace(draft, worktreeId, groupId) + case 'stale': + return buildStaleWorkspace(draft, worktreeId, groupId) + case 'legacy': + return buildLegacyWorkspace(draft, worktreeId) + case 'orphan': + return buildOrphanWorkspace(draft, worktreeId) + case 'editor': + return buildEditorWorkspace(draft, worktreeId, groupId) + } +} + +/** Tops the fixture up to the measured tab count with extra live rows. */ +function padToTabCount(draft: FixtureDraft, workspaceIds: string[], missing: number): number { + let added = 0 + for (let slot = 0; added < missing; slot += 1) { + const worktreeId = workspaceIds[slot % workspaceIds.length] + const group = draft.groupsByWorktree[worktreeId]?.[0] + if (!group || draft.tabsByWorktree[worktreeId] == null) { + continue + } + const id = `${worktreeId}#extra-${slot}` + draft.unifiedTabsByWorktree[worktreeId].push( + unifiedTab(id, worktreeId, group.id, { sortOrder: 10 + slot }) + ) + draft.tabsByWorktree[worktreeId].push(runtimeTab(id, worktreeId, { sortOrder: 10 + slot })) + group.tabOrder.push(id) + added += 1 + } + return added +} + +export function buildHydratedWorkspaceFixture( + workspaceCount = HYDRATED_WORKSPACE_COUNT, + tabCountTarget = HYDRATED_TAB_COUNT +): HydratedWorkspaceFixture { + const draft: FixtureDraft = { + unifiedTabsByWorktree: {}, + groupsByWorktree: {}, + activeGroupIdByWorktree: {}, + tabsByWorktree: {}, + activeTabIdByWorktree: {}, + tabBarOrderByWorktree: {}, + ptyIdsByTabId: {}, + pendingReconnectPtyIdByTabId: {}, + unreadTerminalTabs: {}, + cacheTimerByKey: {}, + openFiles: [] + } + const workspaceIds: string[] = [] + let tabCount = 0 + + for (let index = 0; index < workspaceCount; index += 1) { + const worktreeId = `repo1::/tmp/w${index}` + const groupId = `g-${index}` + workspaceIds.push(worktreeId) + draft.activeGroupIdByWorktree[worktreeId] = groupId + tabCount += buildWorkspace(draft, BUCKETS[index % BUCKETS.length], worktreeId, groupId) + } + tabCount += padToTabCount(draft, workspaceIds, Math.max(0, tabCountTarget - tabCount)) + + return { workspaceIds, tabCount, state: { ...draft } } +} + +/** Every top-level key the two reconciliation patch producers can write. */ +export const RECONCILIATION_WRITABLE_KEYS = [ + 'unifiedTabsByWorktree', + 'groupsByWorktree', + 'activeGroupIdByWorktree', + 'layoutByWorktree', + 'unreadTerminalTabs', + 'tabsByWorktree', + 'ptyIdsByTabId', + 'runtimePaneTitlesByTabId', + 'expandedPaneByTabId', + 'canExpandPaneByTabId', + 'terminalLayoutsByTabId', + 'pendingStartupByTabId', + 'pendingInitialCwdByTabId', + 'pendingSetupSplitByTabId', + 'pendingIssueCommandSplitByTabId', + 'automaticAgentResumeClaimsByTabId', + 'nativeChatLaunchPromptByTabId', + 'nativeChatLaunchDraftByTabId', + 'tabBarOrderByWorktree', + 'cacheTimerByKey', + 'activeTabIdByWorktree', + 'activeTabId' +] as const satisfies readonly (keyof AppState)[] diff --git a/src/renderer/src/store/slices/tabs/tabs-reconciliation-batch.ts b/src/renderer/src/store/slices/tabs/tabs-reconciliation-batch.ts new file mode 100644 index 00000000000..1eb11064203 --- /dev/null +++ b/src/renderer/src/store/slices/tabs/tabs-reconciliation-batch.ts @@ -0,0 +1,55 @@ +import type { AppState } from '../../types' + +/** + * Scratch shared by one multi-workspace reconciliation fold. + * + * Reconciling N workspaces one `set()` at a time re-spreads the same + * workspace-keyed maps N times and rescans `openFiles` N times. A batch lets + * the fold clone each map once and then write into its own private draft, and + * index `openFiles` once — `openFiles` is never written by reconciliation, so + * the index stays valid for the whole fold. + */ +export type WorktreeTabModelReconciliationBatch = { + /** Top-level `AppState` keys this fold already cloned and therefore owns. */ + readonly ownedStateKeys: Set + readonly liveEditorIdsByWorktree: ReadonlyMap> +} + +export const EMPTY_LIVE_EDITOR_IDS: ReadonlySet = new Set() + +export function createWorktreeTabModelReconciliationBatch( + state: Pick +): WorktreeTabModelReconciliationBatch { + const liveEditorIdsByWorktree = new Map>() + for (const file of state.openFiles) { + let ids = liveEditorIdsByWorktree.get(file.worktreeId) + if (!ids) { + ids = new Set() + liveEditorIdsByWorktree.set(file.worktreeId, ids) + } + ids.add(file.id) + } + return { ownedStateKeys: new Set(), liveEditorIdsByWorktree } +} + +/** + * Sets one workspace entry, mutating the batch's own draft once it owns the + * map. Insertion order matches the spread it replaces: an existing key keeps + * its slot, a new key is appended. + */ +export function writeBatchedWorkspaceRecordEntry( + current: Record, + stateKey: string, + worktreeId: string, + // `undefined` is accepted because the spread this replaces also stored it. + value: T | undefined, + batch: WorktreeTabModelReconciliationBatch | undefined +): Record { + if (batch?.ownedStateKeys.has(stateKey)) { + ;(current as Record)[worktreeId] = value + return current + } + const next = { ...current, [worktreeId]: value } as Record + batch?.ownedStateKeys.add(stateKey) + return next +} diff --git a/src/renderer/src/store/slices/tabs/tabs-reconciliation.ts b/src/renderer/src/store/slices/tabs/tabs-reconciliation.ts index 1be6199b55c..72d7be7195e 100644 --- a/src/renderer/src/store/slices/tabs/tabs-reconciliation.ts +++ b/src/renderer/src/store/slices/tabs/tabs-reconciliation.ts @@ -7,6 +7,11 @@ import { getOrphanTerminalIds, terminalTabHasReconnectablePty } from '../terminal-orphan-helpers' +import { + EMPTY_LIVE_EDITOR_IDS, + writeBatchedWorkspaceRecordEntry, + type WorktreeTabModelReconciliationBatch +} from './tabs-reconciliation-batch' export type WorktreeTabModelReconciliation = { patch: Partial @@ -14,9 +19,15 @@ export type WorktreeTabModelReconciliation = { activeRenderableTabId: string | null } +/** + * Pure projection of one workspace's reconciliation patch. Passing `batch` + * lets a multi-workspace fold reuse its own map drafts and `openFiles` index; + * the projected values are identical either way. + */ export function projectWorktreeTabModelReconciliation( state: AppState, - worktreeId: string + worktreeId: string, + batch?: WorktreeTabModelReconciliationBatch ): WorktreeTabModelReconciliation { const unifiedTabs = state.unifiedTabsByWorktree[worktreeId] ?? [] const groups = state.groupsByWorktree[worktreeId] ?? [] @@ -93,9 +104,13 @@ export function projectWorktreeTabModelReconciliation( const liveTerminalIds = new Set( runtimeTerminalTabs.filter((tab) => !orphanTerminalIds.has(tab.id)).map((tab) => tab.id) ) - const liveEditorIds = new Set( - state.openFiles.filter((file) => file.worktreeId === worktreeId).map((file) => file.id) - ) + // Why batched: the unbatched scan is O(openFiles) per workspace, so a + // whole-session reconcile is O(workspaces x openFiles). + const liveEditorIds: ReadonlySet = batch + ? (batch.liveEditorIdsByWorktree.get(worktreeId) ?? EMPTY_LIVE_EDITOR_IDS) + : new Set( + state.openFiles.filter((file) => file.worktreeId === worktreeId).map((file) => file.id) + ) const liveBrowserIds = new Set( (state.browserTabsByWorktree[worktreeId] ?? []).map((browserTab) => browserTab.id) ) @@ -177,7 +192,10 @@ export function projectWorktreeTabModelReconciliation( ) let nextUnreadTerminalTabs = state.unreadTerminalTabs if (droppedTerminalEntityIds.length > 0) { - const copy = { ...state.unreadTerminalTabs } + // A batch that already owns this map published it in an earlier patch, so + // draining further entries in place needs no second patch entry. + const owned = batch?.ownedStateKeys.has('unreadTerminalTabs') === true + const copy = owned ? state.unreadTerminalTabs : { ...state.unreadTerminalTabs } let changed = false for (const entityId of droppedTerminalEntityIds) { if (copy[entityId]) { @@ -187,25 +205,44 @@ export function projectWorktreeTabModelReconciliation( } if (changed) { nextUnreadTerminalTabs = copy + batch?.ownedStateKeys.add('unreadTerminalTabs') } } patch = { - unifiedTabsByWorktree: { ...state.unifiedTabsByWorktree, [worktreeId]: validTabs }, - groupsByWorktree: { ...state.groupsByWorktree, [worktreeId]: nextGroups }, - activeGroupIdByWorktree: { - ...state.activeGroupIdByWorktree, - [worktreeId]: nextActiveGroupId - }, + unifiedTabsByWorktree: writeBatchedWorkspaceRecordEntry( + state.unifiedTabsByWorktree, + 'unifiedTabsByWorktree', + worktreeId, + validTabs, + batch + ), + groupsByWorktree: writeBatchedWorkspaceRecordEntry( + state.groupsByWorktree, + 'groupsByWorktree', + worktreeId, + nextGroups, + batch + ), + activeGroupIdByWorktree: writeBatchedWorkspaceRecordEntry( + state.activeGroupIdByWorktree, + 'activeGroupIdByWorktree', + worktreeId, + nextActiveGroupId, + batch + ), ...(nextUnreadTerminalTabs !== state.unreadTerminalTabs ? { unreadTerminalTabs: nextUnreadTerminalTabs } : {}), ...(nextLayout && layoutChanged ? { - layoutByWorktree: { - ...state.layoutByWorktree, - // Why: restored runtime terminals need a concrete leaf before activation. - [worktreeId]: nextLayout - } + // Why: restored runtime terminals need a concrete leaf before activation. + layoutByWorktree: writeBatchedWorkspaceRecordEntry( + state.layoutByWorktree, + 'layoutByWorktree', + worktreeId, + nextLayout, + batch + ) } : {}), ...(orphanTerminalIds.size > 0 diff --git a/src/renderer/src/store/slices/tabs/tabs-session-actions.ts b/src/renderer/src/store/slices/tabs/tabs-session-actions.ts index 1c2d22c0ce7..e33f1f533cd 100644 --- a/src/renderer/src/store/slices/tabs/tabs-session-actions.ts +++ b/src/renderer/src/store/slices/tabs/tabs-session-actions.ts @@ -8,6 +8,8 @@ import { } from '../degraded-repo-worktree-validity' import { buildHydratedTabState } from '../tabs-hydration' import { projectWorktreeTabModelReconciliation } from './tabs-reconciliation' +import { createWorktreeTabModelReconciliationBatch } from './tabs-reconciliation-batch' +import type { AppState } from '../../types' function replaceWorkspaceRecordKeys( current: Record, @@ -20,11 +22,49 @@ function replaceWorkspaceRecordKeys( } } +/** + * Folds every workspace's reconciliation into one patch. Equivalent to + * applying each patch with its own `set()`: each projection reads the state + * left by its predecessors (they share `unreadTerminalTabs` and the orphan + * cleanup maps), only the store write and subscriber fanout are deferred. + */ +function projectWorktreeTabModelReconciliations( + state: AppState, + worktreeIds: readonly string[] +): Partial { + const batch = createWorktreeTabModelReconciliationBatch(state) + // Private working copy so batch-owned maps can be written in place. + const working = { ...state } + const merged: Partial = {} + for (const worktreeId of worktreeIds) { + const { patch } = projectWorktreeTabModelReconciliation(working, worktreeId, batch) + if (Object.keys(patch).length === 0) { + continue + } + Object.assign(merged, patch) + Object.assign(working, patch) + } + return merged +} + export function createTabsSessionActions( set: TabsSliceSet, get: TabsSliceGet -): Pick { +): Pick< + TabsSlice, + 'reconcileWorktreeTabModel' | 'reconcileWorktreeTabModels' | 'hydrateTabsSession' +> { return { + reconcileWorktreeTabModels: (worktreeIds) => { + if (worktreeIds.length === 0) { + return + } + const patch = projectWorktreeTabModelReconciliations(get(), worktreeIds) + if (Object.keys(patch).length > 0) { + set(patch) + } + }, + reconcileWorktreeTabModel: (worktreeId) => { const reconciliation = projectWorktreeTabModelReconciliation(get(), worktreeId) if (Object.keys(reconciliation.patch).length > 0) { diff --git a/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts b/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts index 456d5892255..81d27ff54d5 100644 --- a/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts +++ b/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts @@ -152,6 +152,8 @@ export type TabsSlice = { renderableTabCount: number activeRenderableTabId: string | null } + /** Reconciles many workspaces through one store write instead of one per workspace. */ + reconcileWorktreeTabModels: (worktreeIds: readonly string[]) => void hydrateTabsSession: ( session: WorkspaceSessionState, options?: WorkspaceSessionHydrationOptions diff --git a/src/renderer/src/store/terminals/workspace-terminal-reconnect.ts b/src/renderer/src/store/terminals/workspace-terminal-reconnect.ts index 5b1e0d39f93..ea0c0545998 100644 --- a/src/renderer/src/store/terminals/workspace-terminal-reconnect.ts +++ b/src/renderer/src/store/terminals/workspace-terminal-reconnect.ts @@ -50,18 +50,6 @@ export function createWorkspaceTerminalReconnectActions( const repoById = buildByIdIndex(get().repos) for (const worktreeId of ids) { const tabs = tabsByWorktree[worktreeId] ?? [] - const worktree = worktreeById.get(worktreeId) - const repo = worktree ? (repoById.get(worktree.repoId) ?? null) : null - // Why: only allow deferred reattach when the SSH connection is active; reattaching to a not-yet-connected relay (deferred/passphrase targets) would fail. - const sshTargetId = options?.directSshAuthority.targetId ?? repo?.connectionId ?? null - const sshState = sshTargetId ? get().sshConnectionStates.get(sshTargetId) : null - const sshConnected = sshTargetId != null && sshState?.status === 'connected' - const supportsDeferredReattach = options - ? sshConnected - : !repo?.connectionId || sshConnected - console.debug( - `[reconnect-terminals] worktree=${worktreeId} connectionId=${repo?.connectionId} sshStatus=${sshState?.status} supportsDeferredReattach=${supportsDeferredReattach}` - ) const targetTabIds = pendingReconnectTabByWorktree[worktreeId] ?? [] const tabsToReconnect: TerminalTab[] = targetTabIds.length > 0 @@ -84,11 +72,8 @@ export function createWorkspaceTerminalReconnectActions( ? undefined : pendingPtyId const hasLeafMappings = Object.keys(leafPtyMap).length > 0 - // Why: publish live PTY hints before mount; pty-connection reattaches later. - console.debug( - `[reconnect-terminals] tab=${tabId} tabLevelPtyId=${tabLevelPtyId} supportsDeferredReattach=${supportsDeferredReattach} hasLeafMappings=${hasLeafMappings}` - ) - // Why: populate ptyIdsByTabId so the sessions status segment maps daemon IDs to tabs; otherwise all sessions look like orphans until the pane mounts. + // Why: publish live PTY hints before mount (pty-connection reattaches later) so the + // sessions status segment maps daemon IDs to tabs; otherwise all sessions look like orphans until the pane mounts. // A row whose tab.ptyId went to the canonical row has no tab-level id left, but its own leaf PTYs still need advertising. const allPtyIds = hasLeafMappings ? (Object.values(leafPtyMap).filter(Boolean) as string[]) From 53f105827bc1949c3c36838052ecd227b962d7f4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:08:01 -0700 Subject: [PATCH 086/398] perf(windows): stop asking the process table for memory, and share one projection per snapshot (#18151) Two costs on the Windows process-table hot path, plus the EDR doc that described neither of them accurately. 1. The snapshot set `ProcessDataFlag.Memory` and surfaced `memoryBytes`, which nothing read. The addon serves that flag with a second `OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ)` and a `GetProcessMemoryInfo` per process (process.cc:47-63), so the flag was one wasted handle per process per snapshot. 2. The shared TTL cache gave every pane the same native rows array, but each pane still ran `native.map(toProcessRow)` over the whole table, rebuilt a `childrenByPpid` Map from scratch, and did two linear scans. The `.map()` also handed `getProcessTableIndex` a new array each call, defeating the POSIX memo by construction. Both now cache per snapshot identity, and the POSIX resolver drops its duplicate descendant walk. `getProcessTableIndex` / `buildProcessTableIndex` are generic over the row shape so the Windows rows reuse the existing pass instead of a parallel one. No behavior change: same rows in, same rows out, same descendant ordering and same has-children answers. --- docs/reference/windows-edr-posture.md | 64 ++++--- docs/reference/windows-process-enumeration.md | 15 +- .../providers/agent-foreground-process.ts | 43 ++--- ...foreground-process-inspection-cost.test.ts | 156 ++++++++++++++++++ .../windows-foreground-process-rows.ts | 71 ++++---- .../windows/windows-process-table-cim-scan.ts | 2 - .../windows/windows-process-table.test.ts | 23 +-- src/main/windows/windows-process-table.ts | 34 ++-- src/shared/process-table-snapshot.ts | 65 ++++++-- 9 files changed, 346 insertions(+), 127 deletions(-) create mode 100644 src/main/providers/windows-foreground-process-inspection-cost.test.ts diff --git a/docs/reference/windows-edr-posture.md b/docs/reference/windows-edr-posture.md index d44aaf33e38..06eb2d5ff9b 100644 --- a/docs/reference/windows-edr-posture.md +++ b/docs/reference/windows-edr-posture.md @@ -72,12 +72,20 @@ behavioural engine can be expected to score it low. ### Every process gets a handle, on a timer -`src/main/windows/windows-process-table.ts` takes a Toolhelp32 snapshot under one -of two flag sets. Identity (`None | CreationTime`) answers pid/ppid/name from the -snapshot alone and opens nothing; the detailed set adds `CommandLine`, which -costs one `OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION)` per process. `Memory` -is retired — it took a second handle carrying `PROCESS_VM_READ` and never read -through it. +`src/main/windows/windows-process-table.ts` takes a Toolhelp32 snapshot under +**one** flag set, `CommandLine | CreationTime`, shared by every caller. pid, ppid +and name come out of the snapshot itself and open nothing. `CommandLine` is what +opens a handle: the addon calls `GetProcessCommandLine` per process, which opens +`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` and walks the PEB with three +`ReadProcessMemory` calls (`src/process_commandline.cc:32,41-47` in the vendored +`@vscode/windows-process-tree` 0.8.0 source that `config/patches/` patches). + +`Memory` is retired as of this change, and that is a real reduction: it made +`GetProcessMemoryUsage` open a **second** `PROCESS_QUERY_INFORMATION | +PROCESS_VM_READ` handle per process for a `GetProcessMemoryInfo` call whose +result no caller read (`src/process.cc:47-63`). Dropping it halves the handles +opened per snapshot. It does not remove the remote memory read, because the +command line still performs one. It exists because seven independent readers used to fork `powershell.exe` for a `Get-CimInstance Win32_Process` scan. That cost, measured: a PowerShell @@ -90,23 +98,41 @@ panes multiplied it (#15036). The native snapshot answers the same question in See [`windows-process-enumeration.md`](./windows-process-enumeration.md). -Asking for fewer fields is cheaper, and since the split the module does: one -cache per flag set, so teardown identity and the session owner probe open no -handle at all (6.3 ms p50) while only the callers that read a command line pay -for one (12.3 ms p50, at 492 processes). Each cache still single-flights within -itself, and one gate serializes the native reads because the vendored wrapper -coalesces the flags of two overlapping calls. +Asking for fewer fields is cheaper, and the module now asks for the smallest set +that still answers every caller. There is **no** per-flag-set cache split: one +TTL-cached snapshot serves everyone, deliberately, because a split would restore +the per-pane fan-out the cache exists to remove — a 32-wide teardown has to +collapse into one scan. So the cheap identity-only read is not something any +caller can select; every read pays for `CommandLine`. An earlier revision of this +file described a two-cache design with 6.3 ms / 12.3 ms p50 figures at 492 +processes. That design is not in the tree and those numbers describe no code +path here; the figures that do apply are the module's own, in +[`windows-process-enumeration.md`](./windows-process-enumeration.md). **How an EDR reads it:** a cross-process handle plus a remote memory read against every process on the box, repeating on a cadence, is the read half of the telemetry that credential dumping and process injection produce. MDE surfaced it -as "suspicious memory activity". The memory read is gone: the command line now -comes from the kernel, through `NtQueryInformationProcess`'s -`ProcessCommandLineInformation` class, which needs only -`PROCESS_QUERY_LIMITED_INFORMATION`. `ReadProcessMemory` is absent from the -compiled addon, asserted against the binary's import table because the published -prebuild loads fine and emits byte-identical strings. What is left to declare to -administrators is the per-process handle itself. +as "suspicious memory activity". + +**That signal is still present.** An earlier revision of this file claimed the +command line "now comes from the kernel" through `NtQueryInformationProcess`'s +`ProcessCommandLineInformation` class, needing only +`PROCESS_QUERY_LIMITED_INFORMATION`, and that `ReadProcessMemory` was absent from +the compiled addon. None of that is true of the code we ship. +`process_commandline.cc` calls `NtQueryInformationProcess` with +`ProcessBasicInformation` only — to locate the PEB — and then issues three +`ReadProcessMemory` calls against a `PROCESS_VM_READ` handle to read the PEB, the +`RTL_USER_PROCESS_PARAMETERS`, and the command-line buffer. Nothing asserts an +import table, and no such assertion would pass. + +What this change did remove is the `Memory` flag's second handle and its +`GetProcessMemoryInfo` call, so the per-process handle count per snapshot halves. +What remains to declare to administrators is unchanged in kind: one +`PROCESS_QUERY_INFORMATION | PROCESS_VM_READ` handle and a PEB read against every +process on the box, at the shared snapshot's cadence. Moving to +`ProcessCommandLineInformation` (Windows 8.1+, `PROCESS_QUERY_LIMITED_INFORMATION` +only) would genuinely retire the remote read, but it is an addon patch nobody has +written; treat it as unclaimed work, not as shipped. ### Encoded, policy-bypassing PowerShell diff --git a/docs/reference/windows-process-enumeration.md b/docs/reference/windows-process-enumeration.md index ef1faf5237c..87ac2a97fb1 100644 --- a/docs/reference/windows-process-enumeration.md +++ b/docs/reference/windows-process-enumeration.md @@ -40,6 +40,15 @@ Measured on Windows 11 with 1050 processes (p50 / p95): | + memory + command line | 30.6 ms | 33.7 ms | | `Get-CimInstance` via PowerShell | 706 ms | 723 ms | +Those are the module's published figures. The flag set this module actually +requests is `CommandLine | CreationTime` — **not** `Memory`, which cost a second +`OpenProcess(PROCESS_QUERY_INFORMATION | PROCESS_VM_READ)` plus +`GetProcessMemoryInfo` per process (`src/process.cc:47-63`) for a value nothing +read. Dropping it halves the handles a snapshot opens. The remaining set sits +between the two rows above and has not been measured separately; on a real +Windows host, `Get-Counter '\Process(Orca)\Handle Count'` sampled across a +snapshot cadence is the check. + Those CIM numbers are from a 1050-process host. The scan scales with process count: on a 1486-process Windows SSH host it measured **1.36 s** and produced **4.8 MiB** of JSON, against the fallback's 3 s and 8 MiB limits. Both limits @@ -220,11 +229,13 @@ ownership, and CPU accounting in the memory collector — still reads it through its own query. Those callers are not migrated. Committed private bytes have no equivalent either, and the one memory value the -snapshot does carry is unusable for the sizes Orca now sees: `process.cc` stores +snapshot _can_ carry is unusable for the sizes Orca now sees: `process.cc` stores `pmc.WorkingSetSize` into a `DWORD`, so anything above 4 GB wraps. That is the second reason `windows-process-resource-collector.ts` still runs its own `Get-CimInstance` sweep — it needs `PageFileUsage` (commit) and the CPU-time -counters in the same pass. Migrating it to the native table would cost both. +counters in the same pass. Migrating it to the native table would cost both, and +it is why this module no longer sets the `Memory` flag at all: the field had no +reader, and asking for it opened a handle per process on every snapshot. Start time is a proxy for identity, not identity. The durable answer for the process trees Orca itself spawns is an inherited handle: a job object names the diff --git a/src/main/providers/agent-foreground-process.ts b/src/main/providers/agent-foreground-process.ts index d244e0100dc..2000d35261c 100644 --- a/src/main/providers/agent-foreground-process.ts +++ b/src/main/providers/agent-foreground-process.ts @@ -1,7 +1,9 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' import { resolveOuterWrapperForegroundProcess } from '../../shared/foreground-wrapper-agent' import { + collectDescendantsFromIndex, getFreshProcessTableSnapshot, + getProcessTableIndex, getProcessTableSnapshot, type ProcessTableRow } from '../../shared/process-table-snapshot' @@ -43,29 +45,6 @@ type ShellForegroundConfirmationOptions = { | Promise | null> } -function collectDescendants( - rows: Row[], - rootPid: number -): (Row & { depth: number })[] { - const childrenByParent = new Map() - for (const row of rows) { - const children = childrenByParent.get(row.ppid) ?? [] - children.push(row) - childrenByParent.set(row.ppid, children) - } - - const descendants: (Row & { depth: number })[] = [] - const stack = (childrenByParent.get(rootPid) ?? []).map((row) => ({ row, depth: 1 })) - while (stack.length > 0) { - const { row, depth } = stack.pop()! - descendants.push({ ...row, depth }) - for (const child of childrenByParent.get(row.pid) ?? []) { - stack.push({ row: child, depth: depth + 1 }) - } - } - return descendants -} - function commandExecutable(command: string): string { const trimmed = command.trim().replace(/^[-]/, '') if (trimmed.startsWith('"') || trimmed.startsWith("'")) { @@ -97,12 +76,12 @@ export async function confirmShellForegroundProcess( } } try { - const rows = await getFreshProcessTableSnapshot() - if (!rows.some((row) => row.pid === shellPid)) { + const index = getProcessTableIndex(await getFreshProcessTableSnapshot()) + const root = index.byPid.get(shellPid) + if (!root) { return false } - const root = rows.find((row) => row.pid === shellPid)! - const tree = [{ ...root, depth: 0 }, ...collectDescendants(rows, shellPid)] + const tree = [{ ...root, depth: 0 }, ...collectDescendantsFromIndex(index, shellPid)] const spawnedShellBasename = executableBasename(spawnedShellProcess) const foregroundShell = tree .filter( @@ -172,7 +151,7 @@ export async function resolveAgentForegroundProcessWithAvailability( const rows = options.fresh ? await getFreshProcessTableSnapshot() : await getProcessTableSnapshot() - if (options.fresh && !rows.some((row) => row.pid === shellPid)) { + if (options.fresh && !getProcessTableIndex(rows).byPid.has(shellPid)) { return { available: false, processName: fallbackProcess } } return { @@ -186,11 +165,13 @@ export async function resolveAgentForegroundProcessWithAvailability( } export function resolveAgentForegroundProcessFromPs( - rows: ProcessTableRow[], + rows: readonly ProcessTableRow[], shellPid: number ): string | null { - const shellRow = rows.find((row) => row.pid === shellPid) - const candidates = collectDescendants(rows, shellPid) + // Memoized per snapshot identity, so the caller's own index build is reused. + const index = getProcessTableIndex(rows) + const shellRow = index.byPid.get(shellPid) + const candidates = collectDescendantsFromIndex(index, shellPid) // Why: `+` in `ps stat` marks the process holding the terminal foreground. // The root shell can hold it after Ctrl-Z, so use the whole PTY tree as the // foreground gate; otherwise a stopped agent child still masquerades as live. diff --git a/src/main/providers/windows-foreground-process-inspection-cost.test.ts b/src/main/providers/windows-foreground-process-inspection-cost.test.ts new file mode 100644 index 00000000000..f08d1b690bb --- /dev/null +++ b/src/main/providers/windows-foreground-process-inspection-cost.test.ts @@ -0,0 +1,156 @@ +// Regression guard on the per-inspection cost of Windows agent foreground +// inspection — the Windows analogue of the POSIX index memo (#6288). +// +// The shared TTL cache already collapses N panes into one Toolhelp32 snapshot +// (windows-agent-foreground-process-scan-volume.test.ts). What it never +// collapsed is the work each pane does ON that snapshot: a full +// `native.map(toProcessRow)` projection, a `childrenByPpid` Map rebuilt from +// scratch, and two linear scans. This file counts that work at a realistic +// table size and pane count, and pins the flag set the snapshot asks for. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { __setWindowsProcessTreeLoaderForTests } from '../windows/windows-process-table' +import { + queryWindowsPaneProcessInventory, + resetWindowsProcessRowsSnapshotForTests +} from './windows-foreground-process-rows' + +// 1050 processes is the host measured in windows-process-enumeration.md; 11 +// panes is the fan-out the shared snapshot exists to serve. +const TABLE_SIZE = 1050 +const PANE_COUNT = 11 + +const SELF_ROW = { pid: process.pid, ppid: 0, name: 'vitest.exe', commandLine: 'vitest' } + +const shellPid = (pane: number): number => 10_000 + pane * 10 +const agentPid = (pane: number): number => shellPid(pane) + 1 +/** A row every pane can look up, so distinct results == distinct projections. */ +const PROBE_PID = 900_000 + TABLE_SIZE - 1 + +/** One shell + one agent child per pane, padded out to a real table size. */ +function buildNativeTable(): { pid: number; ppid: number; name: string; commandLine: string }[] { + const rows = [SELF_ROW] + for (let pane = 0; pane < PANE_COUNT; pane += 1) { + rows.push({ pid: shellPid(pane), ppid: 4, name: 'cmd.exe', commandLine: 'cmd.exe' }) + rows.push({ + pid: agentPid(pane), + ppid: shellPid(pane), + name: 'node.exe', + commandLine: 'node C:/Users/dev/AppData/codex/bin/codex.js' + }) + } + for (let filler = rows.length; filler < TABLE_SIZE; filler += 1) { + rows.push({ pid: 900_000 + filler, ppid: 4, name: 'svchost.exe', commandLine: 'svchost.exe' }) + } + return rows +} + +const NATIVE_TABLE = buildNativeTable() + +/** + * Count `Map.prototype.set` calls — the primitive both the old per-call + * `childrenByPpid` rebuild and the shared index build are made of. Patched for + * one awaited region and restored in `finally`, so nothing else observes it. + */ +async function countMapInsertions(run: () => Promise): Promise { + const original = Map.prototype.set + let insertions = 0 + Map.prototype.set = function patched(this: Map, key: unknown, value: unknown) { + insertions += 1 + return original.call(this, key, value) + } as typeof Map.prototype.set + try { + await run() + } finally { + Map.prototype.set = original + } + return insertions +} + +describe('windows foreground inspection cost per pane', () => { + const getAllProcesses = vi.fn() + let platform: PropertyDescriptor | undefined + let flagsSeen: number[] = [] + + beforeEach(() => { + flagsSeen = [] + getAllProcesses.mockReset() + getAllProcesses.mockImplementation((cb: (rows: unknown) => void, flags: number) => { + flagsSeen.push(flags) + cb(NATIVE_TABLE) + }) + platform = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: 'win32' }) + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2, CreationTime: 4 }, + getAllProcesses + })) + resetWindowsProcessRowsSnapshotForTests() + vi.useFakeTimers({ toFake: ['Date'] }) + vi.setSystemTime(0) + }) + + afterEach(() => { + vi.useRealTimers() + __setWindowsProcessTreeLoaderForTests() + if (platform) { + Object.defineProperty(process, 'platform', platform) + } + }) + + async function sweepPanes(): Promise<(number | undefined)[]> { + const resolved: (number | undefined)[] = [] + for (let pane = 0; pane < PANE_COUNT; pane += 1) { + const inventory = await queryWindowsPaneProcessInventory(shellPid(pane), { + anchorPid: agentPid(pane) + }) + expect(inventory?.candidates).toHaveLength(1) + resolved.push(inventory?.candidates[0]?.pid) + } + return resolved + } + + it('never sets the Memory flag on the snapshot', async () => { + await queryWindowsPaneProcessInventory(shellPid(0)) + expect(flagsSeen).toHaveLength(1) + // Memory is bit 0, and it costs the addon a second OpenProcess per process + // carrying PROCESS_VM_READ (process.cc `GetProcessMemoryUsage`). + expect(flagsSeen[0]! & 1).toBe(0) + // CommandLine (2) | CreationTime (4). + expect(flagsSeen[0]).toBe(6) + }) + + it('projects the shared snapshot once for the whole pane fan-out', async () => { + const probeRows: unknown[] = [] + for (let pane = 0; pane < PANE_COUNT; pane += 1) { + const inventory = await queryWindowsPaneProcessInventory(shellPid(pane), { + anchorPid: PROBE_PID + }) + probeRows.push(inventory?.anchorRow) + } + expect(probeRows.filter(Boolean)).toHaveLength(PANE_COUNT) + // One projection produced every pane's row object. Pre-fix each pane ran + // its own `native.map(toProcessRow)` over all 1050 rows, so this set held + // PANE_COUNT distinct objects and the sweep allocated PANE_COUNT * 1050. + expect(new Set(probeRows).size).toBe(1) + }) + + it('indexes the shared snapshot once for the whole pane fan-out', async () => { + // Prime the TTL cache and the index so the snapshot read is not in the count. + await queryWindowsPaneProcessInventory(shellPid(0), { anchorPid: agentPid(0) }) + + const insertions = await countMapInsertions(async () => { + await sweepPanes() + }) + + // Pre-fix every pane rebuilt a whole-table `childrenByPpid`, so this was + // >= PANE_COUNT * (rows with a distinct ppid). One shared index makes the + // whole sweep cost no table-sized Map build at all. + expect(insertions).toBeLessThan(TABLE_SIZE) + }) + + it('resolves the same foreground child for every pane as an unshared scan would', async () => { + const resolved = await sweepPanes() + expect(resolved).toEqual(Array.from({ length: PANE_COUNT }, (_, pane) => agentPid(pane))) + }) +}) diff --git a/src/main/providers/windows-foreground-process-rows.ts b/src/main/providers/windows-foreground-process-rows.ts index f5a74bd0fe2..5be4734dddc 100644 --- a/src/main/providers/windows-foreground-process-rows.ts +++ b/src/main/providers/windows-foreground-process-rows.ts @@ -1,3 +1,7 @@ +import { + collectDescendantsFromIndex, + getProcessTableIndex +} from '../../shared/process-table-snapshot' import { readWindowsProcessTable, readWindowsProcessTableFresh, @@ -25,15 +29,40 @@ function toProcessRow(row: NativeWindowsProcessRow): WindowsProcessRow { } } +/** + * One projection per snapshot identity, mirroring `getProcessTableIndex`. + * + * The TTL cache already gives every pane the same native rows array; without + * this each of them still rebuilt ~1050 row objects, which also handed + * `getProcessTableIndex` a new array each time and defeated its memo by + * construction. Keyed weakly, so a projection dies with its snapshot. Rows are + * shared, never mutated: descendants are copied with their depth, and + * `anchorRow` is read-only to every caller. + */ +const projectedRows = new WeakMap() + +function projectProcessRows(native: readonly NativeWindowsProcessRow[]): WindowsProcessRow[] { + const cached = projectedRows.get(native) + if (cached) { + return cached + } + const rows = native.map(toProcessRow) + projectedRows.set(native, rows) + return rows +} + /** * Rows from a scan that starts after this call. * * PID-identity checks in teardown must not reuse a cached row — it can predate * the very recycle it is meant to detect. Rejects when the table is unreadable, * so "unavailable" stays distinguishable from "nothing is running". + * + * `readonly` because the projection is shared with every other reader of the + * same snapshot. */ -export async function queryWindowsProcessRowsFresh(): Promise { - return (await readWindowsProcessTableFresh()).map(toProcessRow) +export async function queryWindowsProcessRowsFresh(): Promise { + return projectProcessRows(await readWindowsProcessTableFresh()) } export async function queryWindowsProcessDescendants( @@ -63,21 +92,22 @@ export async function queryWindowsPaneProcessInventory( options.fresh === true ? await readWindowsProcessTableFresh() : await readWindowsProcessTable() - rows = native.map(toProcessRow) + rows = projectProcessRows(native) } catch { return null } + // One index per snapshot, shared by every pane inspecting inside the TTL + // window: `byPid` answers both lookups that used to be linear scans, and + // `childrenByPpid` replaces a per-call Map rebuild over the whole table. + const index = getProcessTableIndex(rows) // Why: a snapshot that omitted the PTY root may be stale or permission- // filtered; only an observed root can authoritatively have no descendants. - if (!rows.some((row) => row.pid === rootPid)) { + if (!index.byPid.has(rootPid)) { return null } return { - candidates: collectDescendants(rows, rootPid).sort((a, b) => b.depth - a.depth), - anchorRow: - options.anchorPid !== undefined - ? (rows.find((row) => row.pid === options.anchorPid) ?? null) - : null + candidates: collectDescendantsFromIndex(index, rootPid).sort((a, b) => b.depth - a.depth), + anchorRow: options.anchorPid !== undefined ? (index.byPid.get(options.anchorPid) ?? null) : null } } @@ -85,26 +115,3 @@ export async function queryWindowsPaneProcessInventory( export function resetWindowsProcessRowsSnapshotForTests(): void { resetWindowsProcessTableForTests() } - -function collectDescendants( - rows: Row[], - rootPid: number -): (Row & { depth: number })[] { - const childrenByParent = new Map() - for (const row of rows) { - const children = childrenByParent.get(row.ppid) ?? [] - children.push(row) - childrenByParent.set(row.ppid, children) - } - - const descendants: (Row & { depth: number })[] = [] - const stack = (childrenByParent.get(rootPid) ?? []).map((row) => ({ row, depth: 1 })) - while (stack.length > 0) { - const { row, depth } = stack.pop()! - descendants.push({ ...row, depth }) - for (const child of childrenByParent.get(row.pid) ?? []) { - stack.push({ row: child, depth: depth + 1 }) - } - } - return descendants -} diff --git a/src/main/windows/windows-process-table-cim-scan.ts b/src/main/windows/windows-process-table-cim-scan.ts index 898f0c36f33..213f157f63b 100644 --- a/src/main/windows/windows-process-table-cim-scan.ts +++ b/src/main/windows/windows-process-table-cim-scan.ts @@ -75,8 +75,6 @@ export function parseWindowsCimProcessRows(stdout: string): WindowsProcessRow[] return [] } const name = fieldAsString(row.Name) - // memoryBytes stays undefined: Win32_Process reports WorkingSetSize, but no - // caller reads it off this table and asking widens an already costly scan. return [{ pid, ppid, name, command: fieldAsString(row.CommandLine) || name }] }) } diff --git a/src/main/windows/windows-process-table.test.ts b/src/main/windows/windows-process-table.test.ts index 609de009820..360dae4ec14 100644 --- a/src/main/windows/windows-process-table.test.ts +++ b/src/main/windows/windows-process-table.test.ts @@ -52,21 +52,24 @@ describe('windows process table', () => { it('maps native rows, defaulting an unreadable command line to empty', async () => { const rows = await readWindowsProcessTableFresh() expect(rows).toEqual([ - { pid: process.pid, ppid: 0, name: 'vitest.exe', command: '', memoryBytes: undefined }, + { pid: process.pid, ppid: 0, name: 'vitest.exe', command: '' }, { pid: 100, ppid: 4, name: 'orca.exe', command: '"C:/a b/orca.exe" --x', - memoryBytes: 4096, creationTimeMs: 1_700_000_000_000 } ]) }) - it('requests memory and command line together', async () => { + it('requests the command line and creation time, never memory', async () => { await readWindowsProcessTableFresh() - expect(getAllProcesses.mock.calls[0]?.[1]).toBe(7) + // CommandLine (2) | CreationTime (4). The Memory bit (1) stays clear: the + // addon opens a second PROCESS_VM_READ handle per process to serve it and + // nothing reads a working set off this table. + expect(getAllProcesses.mock.calls[0]?.[1]).toBe(6) + expect((getAllProcesses.mock.calls[0]?.[1] as number) & 1).toBe(0) }) it('only advertises PID-safe ownership when the native creation-time field exists', () => { @@ -400,20 +403,19 @@ describe('resolving the native reader', () => { }) const rows = await readWindowsProcessTableFresh() expect(rows).toEqual([ - { pid: process.pid, ppid: 0, name: 'vitest.exe', command: '', memoryBytes: undefined }, + { pid: process.pid, ppid: 0, name: 'vitest.exe', command: '' }, { pid: 100, ppid: 4, name: 'orca.exe', command: '"C:/a b/orca.exe" --x', - memoryBytes: 4096, creationTimeMs: 1_700_000_000_000 } ]) expect(isWindowsProcessTableAvailable()).toBe(true) }) - it('asks the addon for memory and command line, as the package path does', async () => { + it('asks the addon for the command line but not memory, as the package path does', async () => { const addon = addonReturning(NATIVE) __setWindowsProcessTreeRequireForTests((specifier: string) => { if (specifier === ADDON_SPECIFIER) { @@ -422,9 +424,10 @@ describe('resolving the native reader', () => { throw new Error('MODULE_NOT_FOUND') }) await readWindowsProcessTableFresh() - // Memory | CommandLine. A bare snapshot would silently drop the command - // line every agent-recognition caller matches on first. - expect(addon.getProcessList).toHaveBeenCalledWith(expect.any(Function), 3) + // CommandLine only: a bare snapshot would silently drop the command line + // every agent-recognition caller matches on first, and the relay addon + // exposes no CreationTime bit to add. + expect(addon.getProcessList).toHaveBeenCalledWith(expect.any(Function), 2) }) it('reaches the CIM scan when neither the package nor the addon is present', async () => { diff --git a/src/main/windows/windows-process-table.ts b/src/main/windows/windows-process-table.ts index 9308c64a9f0..42d48f4fc79 100644 --- a/src/main/windows/windows-process-table.ts +++ b/src/main/windows/windows-process-table.ts @@ -23,6 +23,10 @@ import { readWindowsProcessRowsWithCim } from './windows-process-table-cim-scan' * pid+ppid+name 15.9 / 17.5 ms * +memory +commandLine 30.6 / 33.7 ms * PowerShell CIM 706 / 723 ms + * + * Those are the module's published figures for both extra fields together; the + * only flag set this module asks for is `CommandLine` (+ `CreationTime`, free), + * which sits between the two rows and has not been separately measured. */ export type WindowsProcessRow = { @@ -31,8 +35,6 @@ export type WindowsProcessRow = { name: string /** Full command line. Empty when the process denied a query handle. */ command: string - /** Working set in bytes, or undefined when not requested/queryable. */ - memoryBytes?: number /** Process creation time in Unix milliseconds, when the native snapshot provides it. */ creationTimeMs?: number } @@ -41,7 +43,6 @@ type NativeProcessInfo = { pid: number ppid: number name: string - memory?: number commandLine?: string creationTimeMs?: number } @@ -49,7 +50,6 @@ type NativeProcessInfo = { type WindowsProcessTreeModule = { ProcessDataFlag: { None: number - Memory: number CommandLine: number CreationTime?: number } @@ -82,7 +82,10 @@ type WindowsProcessTreeAddon = { ) => void } -/** Mirrors the package's enum; the addon takes the raw bit field. */ +/** + * Mirrors the package's enum; the addon takes the raw bit field. `Memory` (1) + * is listed for completeness and is deliberately never set — see `flags` below. + */ const PROCESS_DATA_FLAG = { None: 0, Memory: 1, CommandLine: 2 } as const /** Staged beside the relay bundle by build-relay; see RELAY_ARTIFACTS. */ @@ -190,16 +193,16 @@ function readNativeRows(): Promise { } const readId = ++readSequence const readerEpoch = nativeReaderEpoch - // Why always both flags: each adds an OpenProcess per process (Memory a - // GetProcessMemoryInfo, CommandLine a PEB read), so asking for less would be - // cheaper -- 15.9ms p50 versus 30.6ms at 1050 processes. But every read shares - // one snapshot so a 32-wide teardown collapses into a single scan, and that - // snapshot has to satisfy every caller. Splitting the cache per field set - // would restore exactly the fan-out it exists to prevent. - const flags = - native.ProcessDataFlag.Memory | - native.ProcessDataFlag.CommandLine | - (native.ProcessDataFlag.CreationTime ?? 0) + // Why CommandLine but not Memory: each flag costs one OpenProcess per process + // inside the addon (process.cc), and every caller of this table matches on + // `command`, while nothing reads a working set off it -- the Resource Manager + // runs its own CIM sweep because it needs commit and CPU time in one pass, and + // `process.cc` truncates the working set into a DWORD anyway. Dropping Memory + // halves the per-snapshot handle count; the remaining flags stay in ONE flag + // set because every read shares one snapshot, so a 32-wide teardown collapses + // into a single scan. Splitting the cache per field set would restore exactly + // the fan-out it exists to prevent. + const flags = native.ProcessDataFlag.CommandLine | (native.ProcessDataFlag.CreationTime ?? 0) return new Promise((resolve, reject) => { // Hoisted so a synchronous throw from getAllProcesses can clear it. An // orphaned timer would otherwise fire later and wedge a reader that had @@ -241,7 +244,6 @@ function readNativeRows(): Promise { ppid: row.ppid, name: row.name, command: row.commandLine ?? '', - memoryBytes: row.memory, ...(typeof row.creationTimeMs === 'number' ? { creationTimeMs: row.creationTimeMs } : {}) diff --git a/src/shared/process-table-snapshot.ts b/src/shared/process-table-snapshot.ts index 2d65fc3f324..54764767740 100644 --- a/src/shared/process-table-snapshot.ts +++ b/src/shared/process-table-snapshot.ts @@ -124,28 +124,36 @@ export type ProcessTableIndexStats = { indexLookups: number } -export type ProcessTableIndex = { - rows: readonly ProcessTableRow[] - byPid: ReadonlyMap - childrenByPpid: ReadonlyMap +/** The parent/child fields every process-table row shape shares. */ +export type ProcessIdentityRow = { pid: number; ppid: number } + +export type ProcessTableIndexOf = { + rows: readonly Row[] + byPid: ReadonlyMap + childrenByPpid: ReadonlyMap stats?: ProcessTableIndexStats } +export type ProcessTableIndex = ProcessTableIndexOf + /** * Build the correlation indexes in one linear pass over a capture. Only the * indexes a resolver actually reads are materialized: group indexes would cost * two more maps plus a per-row array allocation on every capture, and foreground * membership is derived from each row's own `pgid` against the root's `tpgid`. + * + * Generic over the row shape so the Windows snapshot (`pid`/`ppid`/`name`/ + * `command`) shares this pass rather than carrying a parallel one. */ -export function buildProcessTableIndex( - rows: readonly ProcessTableRow[], +export function buildProcessTableIndex( + rows: readonly Row[], stats?: ProcessTableIndexStats -): ProcessTableIndex { +): ProcessTableIndexOf { if (stats) { stats.indexBuilds += 1 } - const byPid = new Map() - const childrenByPpid = new Map() + const byPid = new Map() + const childrenByPpid = new Map() for (const row of rows) { if (stats) { stats.rowVisits += 1 @@ -161,6 +169,29 @@ export function buildProcessTableIndex( return { rows, byPid, childrenByPpid, stats } } +/** + * Depth-first descendants of `rootPid`, deepest-last, off a prebuilt index. + * + * Each row is copied with its depth, so callers may not mutate the index's rows + * through the result. Ordering matches a per-call `childrenByPpid` walk exactly: + * children keep capture order and the stack pops last-pushed first. + */ +export function collectDescendantsFromIndex( + index: ProcessTableIndexOf, + rootPid: number +): (Row & { depth: number })[] { + const descendants: (Row & { depth: number })[] = [] + const stack = (index.childrenByPpid.get(rootPid) ?? []).map((row) => ({ row, depth: 1 })) + while (stack.length > 0) { + const { row, depth } = stack.pop()! + descendants.push({ ...row, depth }) + for (const child of index.childrenByPpid.get(row.pid) ?? []) { + stack.push({ row: child, depth: depth + 1 }) + } + } + return descendants +} + /** * Rank a descendant row as a foreground candidate: a `+` (foreground process * group) row always outranks a background one, then the deepest wins. @@ -169,9 +200,9 @@ export function scoreForegroundCandidateRow(row: ProcessTableRow & { depth: numb return (row.stat.includes('+') ? 10_000 : 0) + row.depth } -export function lookupProcessTableIndex( - index: ProcessTableIndex, - lookup: (index: ProcessTableIndex) => T, +export function lookupProcessTableIndex( + index: ProcessTableIndexOf, + lookup: (index: ProcessTableIndexOf) => T, stats = index.stats ): T { if (stats) { @@ -180,7 +211,9 @@ export function lookupProcessTableIndex( return lookup(index) } -const processTableIndexes = new WeakMap() +// Keyed by array identity, which also pins the row shape the entry was built +// for, so the one cast below cannot hand a caller another row type's index. +const processTableIndexes = new WeakMap() /** * Memoize one index per snapshot identity, so the panes that share a TTL-cached @@ -195,8 +228,10 @@ const processTableIndexes = new WeakMap( + rows: readonly Row[] +): ProcessTableIndexOf { + const cached = processTableIndexes.get(rows) as ProcessTableIndexOf | undefined if (cached) { return cached } From d22d5c80a69ed7341074c388128dadabd43f5b26 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:08:27 -0700 Subject: [PATCH 087/398] fix(terminal): stop a hidden pane's cursor blink deterministically (#18152) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A retained hidden pane keeps a live WebglRenderer, and the only thing that stops its 600 ms cursor-blink timer is a real DOM blur event. Today that arrives incidentally from display:none/visibility:hidden; under a hide mode that keeps focus (opacity:0 without inert) it never fires and the pane blinks — redrawing its whole cursor row per toggle — until the 5-minute idle timeout. Park terminal.options.cursorBlink on suspend and restore the parked value on resume, so the property holds regardless of which CSS hid the pane. Settings writes land on the parked value while hidden, so a mid-hide settings change cannot re-arm the timer behind the surface, and a user who disabled blink never gets it back. --- .../terminal-pane/terminal-appearance.ts | 5 +- .../pane-cursor-blink-suspension.test.ts | 300 ++++++++++++++++++ .../pane-cursor-blink-suspension.ts | 57 ++++ .../pane-manager-pane-creation.ts | 5 + .../pane-manager/pane-rendering-control.ts | 13 + .../pane-webgl-context-recovery.test.ts | 1 + .../pane-webgl-refresh-lifecycle.test.ts | 1 + .../terminal-webgl-hidden-retention.test.ts | 3 +- 8 files changed, 383 insertions(+), 2 deletions(-) create mode 100644 src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.test.ts create mode 100644 src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.ts diff --git a/src/renderer/src/components/terminal-pane/terminal-appearance.ts b/src/renderer/src/components/terminal-pane/terminal-appearance.ts index bb61a50fbbc..a5aa36568ca 100644 --- a/src/renderer/src/components/terminal-pane/terminal-appearance.ts +++ b/src/renderer/src/components/terminal-pane/terminal-appearance.ts @@ -21,6 +21,7 @@ import { resolveTerminalCursorInactiveStyle } from '@/lib/pane-manager/pane-terminal-options' import { getFitOverrideForPty } from '@/lib/pane-manager/mobile-fit-overrides' +import { setTerminalCursorBlinkOption } from '@/lib/pane-manager/pane-cursor-blink-suspension' import type { PtyTransport } from './pty-transport' import type { EffectiveMacOptionAsAlt } from '@/lib/keyboard-layout/detect-option-as-alt' import { HEX_COLOR_RE } from '../../../../shared/color-validation' @@ -180,7 +181,9 @@ export function applyTerminalAppearance( const cursorStyle = settings.terminalCursorStyle ?? 'block' pane.terminal.options.cursorStyle = cursorStyle pane.terminal.options.cursorInactiveStyle = resolveTerminalCursorInactiveStyle(cursorStyle) - pane.terminal.options.cursorBlink = settings.terminalCursorBlink + // Why not a direct write: a suspended (hidden) pane parks the value instead, so a + // settings change mid-hide cannot re-arm its blink timer behind the hidden surface. + setTerminalCursorBlinkOption(pane.terminal, settings.terminalCursorBlink) const paneSize = paneFontSizes.get(pane.id) const metricOptions = { fontSize: paneSize ?? settings.terminalFontSize, diff --git a/src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.test.ts b/src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.test.ts new file mode 100644 index 00000000000..66dc3fc872b --- /dev/null +++ b/src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.test.ts @@ -0,0 +1,300 @@ +// @vitest-environment happy-dom + +import { Terminal } from '@xterm/xterm' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { ManagedPaneInternal } from './pane-manager-types' +import { resumePaneRendering, suspendPaneRendering } from './pane-rendering-control' +import { resetHiddenWebglRetentionForTest } from './terminal-webgl-hidden-retention' +import { isTerminalCursorBlinkSuspended } from './pane-cursor-blink-suspension' + +/** + * Invariant (`terminal-render.hidden-pane-idle-cost`): a hidden pane must not drive + * the cursor-blink redraw loop, and revealing it must put back exactly the blink + * state it had — never a frozen, missing, or unexpectedly blinking cursor. + * + * Oracle: the cursor xterm actually renders. The DOM renderer stamps + * `xterm-cursor` / `xterm-cursor-blink` on the cursor cell (DomRendererRowFactory), + * so these assertions read rendered output, not the option value — an assertion on + * `options.cursorBlink` alone would pass with the bug present. + * + * The hidden-side cost is proved separately by the redraw-count test at the bottom: + * the DOM renderer already gates its blink class on per-terminal focus, whereas the + * WebGL renderer's blink timer arms on `document.hasFocus()`, which is what leaks. + */ + +const COLS = 80 +const ROWS = 24 + +type TestPane = ManagedPaneInternal & { host: HTMLElement } + +function nextFrame(): Promise { + return new Promise((resolve) => requestAnimationFrame(() => resolve())) +} + +async function flushRender(): Promise { + await nextFrame() + await nextFrame() + await new Promise((resolve) => setTimeout(resolve, 0)) +} + +function write(terminal: Terminal, data: string): Promise { + return new Promise((resolve) => terminal.write(data, resolve)) +} + +function createTestPane(options: { cursorBlink?: boolean; webglAddon?: boolean } = {}): TestPane { + const host = document.createElement('div') + document.body.appendChild(host) + const terminal = new Terminal({ + cols: COLS, + rows: ROWS, + cursorBlink: options.cursorBlink ?? true, + allowProposedApi: true + }) + terminal.open(host) + terminal.focus() + const pane = { + id: 1, + terminal, + host, + container: host, + xtermContainer: host, + // Off so resume's reattachWebglIfNeeded leaves the DOM renderer in place; + // the rendered-cursor oracle needs a renderer that paints in happy-dom. + gpuRenderingEnabled: false, + terminalGpuAcceleration: 'off', + webglAddon: options.webglAddon ? ({ dispose: vi.fn() } as never) : null, + webglAttachmentDeferred: false, + webglDisabledAfterContextLoss: false, + webglAttachFailedSinceRecovery: false, + hasComplexScriptOutput: false, + pendingWebglRefreshRafId: null, + pendingObservedFitRafId: null + } as unknown as TestPane + return pane +} + +/** The cursor cell as the renderer painted it, or null when no cursor is rendered. */ +function renderedCursor(pane: TestPane): HTMLElement | null { + return pane.host.querySelector('.xterm-cursor') +} + +function rendersBlinkingCursor(pane: TestPane): boolean { + return renderedCursor(pane)?.classList.contains('xterm-cursor-blink') === true +} + +function renderedText(pane: TestPane): string { + return pane.host.querySelector('.xterm-rows')?.textContent ?? '' +} + +/** Reveal = the manager's resume pass, then the terminal regains real DOM focus. */ +async function reveal(panes: TestPane[], owner?: object): Promise { + resumePaneRendering(panes, owner) + for (const pane of panes) { + pane.terminal.focus() + } +} + +describe('hidden-pane cursor blink suspension', () => { + beforeEach(() => { + resetHiddenWebglRetentionForTest() + // happy-dom has no canvas text metrics; xterm measures glyphs on open(). + vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue({ + measureText: () => ({ width: 10 }) + } as unknown as CanvasRenderingContext2D) + }) + + afterEach(() => { + vi.restoreAllMocks() + document.body.replaceChildren() + }) + + it('renders a blinking cursor again one frame after reveal', async () => { + const pane = createTestPane() + await write(pane.terminal, 'ready$ ') + await flushRender() + expect(rendersBlinkingCursor(pane)).toBe(true) + + suspendPaneRendering([pane]) + expect(isTerminalCursorBlinkSuspended(pane.terminal)).toBe(true) + + await reveal([pane]) + await nextFrame() + expect(renderedCursor(pane), 'reveal must not leave the pane without a cursor').not.toBeNull() + expect(rendersBlinkingCursor(pane)).toBe(true) + expect(isTerminalCursorBlinkSuspended(pane.terminal)).toBe(false) + }) + + it('keeps a user who disabled cursor blink un-blinking across hide and reveal', async () => { + const pane = createTestPane({ cursorBlink: false }) + await write(pane.terminal, 'ready$ ') + await flushRender() + expect(rendersBlinkingCursor(pane)).toBe(false) + + suspendPaneRendering([pane]) + await reveal([pane]) + await flushRender() + + expect(renderedCursor(pane), 'a steady cursor must still be rendered').not.toBeNull() + expect(rendersBlinkingCursor(pane), 'blink must stay off for this user').toBe(false) + }) + + it('does not drop output written while hidden, and echoes typing after reveal', async () => { + const pane = createTestPane() + await write(pane.terminal, 'before-hide\r\n') + await flushRender() + + suspendPaneRendering([pane]) + await write(pane.terminal, 'while-hidden\r\n') + + await reveal([pane]) + await flushRender() + expect(renderedText(pane)).toContain('before-hide') + expect(renderedText(pane)).toContain('while-hidden') + + // Real key input through xterm's textarea must still route to the PTY. + const typed: string[] = [] + pane.terminal.onData((data) => typed.push(data)) + const keydown = new KeyboardEvent('keydown', { + key: 'x', + code: 'KeyX', + bubbles: true, + cancelable: true + }) + Object.defineProperty(keydown, 'keyCode', { value: 88 }) + pane.terminal.textarea?.dispatchEvent(keydown) + expect(typed, 'keystrokes must still reach the PTY after a hide/reveal').toEqual(['x']) + + await write(pane.terminal, 'x') + await flushRender() + expect(renderedText(pane)).toContain('x') + expect(rendersBlinkingCursor(pane)).toBe(true) + }) + + it('restores blink on the retained-hidden-WebGL path, which skips dispose', async () => { + const owner = {} + const pane = createTestPane({ webglAddon: true }) + const addon = pane.webglAddon + await write(pane.terminal, 'ready$ ') + await flushRender() + + suspendPaneRendering([pane], { owner, livePanes: () => [pane] }) + // The retention branch returns before disposeWebgl — this is the path where + // nothing else could have stopped the blink timer. + expect(addon?.dispose).not.toHaveBeenCalled() + expect(isTerminalCursorBlinkSuspended(pane.terminal)).toBe(true) + + await reveal([pane], owner) + await nextFrame() + expect(rendersBlinkingCursor(pane)).toBe(true) + }) + + it('restores every pane of a split, including one revealed a second time', async () => { + const owner = {} + const left = createTestPane() + const right = createTestPane() + const panes = [left, right] + await write(left.terminal, 'left$ ') + await write(right.terminal, 'right$ ') + await flushRender() + + for (let cycle = 0; cycle < 2; cycle++) { + suspendPaneRendering(panes, { owner, livePanes: () => panes }) + resumePaneRendering(panes, owner) + // One at a time: the DOM renderer only paints a blinking cursor in the pane + // that currently holds real focus, so focus each split half in turn. + for (const [name, pane] of [ + ['left', left], + ['right', right] + ] as const) { + pane.terminal.focus() + await nextFrame() + expect(renderedCursor(pane), `${name} cursor, cycle ${cycle}`).not.toBeNull() + expect(rendersBlinkingCursor(pane), `${name} blink, cycle ${cycle}`).toBe(true) + } + } + expect(renderedText(left)).toContain('left$') + expect(renderedText(right)).toContain('right$') + }) + + it('does not re-arm a hidden pane when the blink setting changes mid-hide', async () => { + const pane = createTestPane({ cursorBlink: false }) + suspendPaneRendering([pane]) + + // What applyTerminalAppearance does when the user flips the setting. + const { setTerminalCursorBlinkOption } = await import('./pane-cursor-blink-suspension') + setTerminalCursorBlinkOption(pane.terminal, true) + expect( + pane.terminal.options.cursorBlink, + 'a hidden pane must not start blinking behind the surface' + ).toBe(false) + + await reveal([pane]) + await nextFrame() + expect(rendersBlinkingCursor(pane), 'the new setting applies on reveal').toBe(true) + }) + + it('resume is a no-op for a pane that was never suspended', async () => { + const pane = createTestPane() + await flushRender() + await reveal([pane]) + await nextFrame() + expect(rendersBlinkingCursor(pane)).toBe(true) + }) +}) + +/** + * Deterministic cost gate for the hidden side. + * + * Model, taken from `@xterm/addon-webgl@0.20` sources: `CursorBlinkStateManager` + * runs a 600 ms interval whenever `ICoreBrowserService.isFocused` is true — a + * DOCUMENT-level fact — and every toggle calls `WebglRenderer._requestRedrawCursor()`, + * which redraws one full row (`_updateModel` loops `x < cols` per row). + * `WebglRenderer._updateCursorBlink()` decides whether the manager exists from + * `decPrivateModes.cursorBlink ?? terminal.options.cursorBlink`, which is read from + * the real Terminal below. + */ +const BLINK_INTERVAL_MS = 600 + +function blinkCellsRedrawnPerWindow(terminal: Terminal, windowMs: number): number { + const decPrivateBlink = ( + terminal as unknown as { + _core: { coreService: { decPrivateModes: { cursorBlink?: boolean } } } + } + )._core.coreService.decPrivateModes.cursorBlink + const blinking = decPrivateBlink ?? terminal.options.cursorBlink === true + if (!blinking) { + return 0 + } + return Math.floor(windowMs / BLINK_INTERVAL_MS) * terminal.cols +} + +describe('hidden-pane cursor blink redraw cost', () => { + beforeEach(() => { + resetHiddenWebglRetentionForTest() + vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue({ + measureText: () => ({ width: 10 }) + } as unknown as CanvasRenderingContext2D) + }) + + afterEach(() => { + vi.restoreAllMocks() + document.body.replaceChildren() + }) + + it('drops retained hidden panes to zero blink redraws and restores them on reveal', async () => { + const owner = {} + // The retention cap: MAX_RETAINED_HIDDEN_WEBGL_CONTEXTS. + const panes = Array.from({ length: 6 }, () => createTestPane({ webglAddon: true })) + const windowMs = 30_000 + const cells = () => + panes.reduce((sum, pane) => sum + blinkCellsRedrawnPerWindow(pane.terminal, windowMs), 0) + + expect(cells()).toBe(6 * 50 * COLS) + + suspendPaneRendering(panes, { owner, livePanes: () => panes }) + expect(cells(), 'hidden panes must cost nothing while the window is visible').toBe(0) + + await reveal(panes, owner) + expect(cells()).toBe(6 * 50 * COLS) + }) +}) diff --git a/src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.ts b/src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.ts new file mode 100644 index 00000000000..13f90af696c --- /dev/null +++ b/src/renderer/src/lib/pane-manager/pane-cursor-blink-suspension.ts @@ -0,0 +1,57 @@ +import type { Terminal } from '@xterm/xterm' + +/** + * Park `cursorBlink` while a pane is hidden. + * + * Why: a retained hidden pane (`terminal-webgl-hidden-retention.ts`) keeps a live + * WebglRenderer, and the only thing that stops its 600 ms blink timer is + * `WebglRenderer.handleBlur()` — reached only from a real DOM blur event. Today + * that arrives incidentally, because `display:none`/`visibility:hidden` move focus; + * under `opacity:0` without `inert` (TerminalOverlaySlot's startup probe) it never + * fires, `Terminal.blur()` is a no-op on a textarea that is not the active element, + * and the pane blinks — redrawing its whole cursor row through + * `WebglRenderer._updateModel` — until the 5-minute idle timeout. + * + * `cursorBlink` is the public option that tears the timer down deterministically + * (`RenderService.handleOptionsChanged` -> `WebglRenderer._updateCursorBlink`), so + * "a hidden pane does not blink" stops depending on which CSS hid it. + * + * Resume restores the parked value rather than the settings value, so a pane that + * was not blinking before the hide never comes back blinking. + */ +const parkedCursorBlink = new WeakMap() + +export function suspendTerminalCursorBlink(terminal: Terminal): void { + if (parkedCursorBlink.has(terminal)) { + return + } + parkedCursorBlink.set(terminal, terminal.options.cursorBlink === true) + terminal.options.cursorBlink = false +} + +/** Restores the pre-suspend blink state. No-op on a terminal that was never suspended. */ +export function resumeTerminalCursorBlink(terminal: Terminal): void { + if (!parkedCursorBlink.has(terminal)) { + return + } + const restored = parkedCursorBlink.get(terminal) === true + parkedCursorBlink.delete(terminal) + terminal.options.cursorBlink = restored +} + +/** + * Settings-driven blink writes land on the parked value while a pane is suspended; + * writing the option directly would re-arm the blink timer behind a hidden surface + * until the next reveal. + */ +export function setTerminalCursorBlinkOption(terminal: Terminal, enabled: boolean): void { + if (parkedCursorBlink.has(terminal)) { + parkedCursorBlink.set(terminal, enabled) + return + } + terminal.options.cursorBlink = enabled +} + +export function isTerminalCursorBlinkSuspended(terminal: Terminal): boolean { + return parkedCursorBlink.has(terminal) +} diff --git a/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts b/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts index 11efdff06c0..483f49a1f55 100644 --- a/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts +++ b/src/renderer/src/lib/pane-manager/pane-manager-pane-creation.ts @@ -2,6 +2,7 @@ import type { ManagedPane, ManagedPaneInternal, PaneManagerOptions } from './pan import type { PaneManagerHost } from './pane-manager-host' import { applyPaneOpacity } from './pane-divider' import { createPaneDOM, openTerminal } from './pane-lifecycle' +import { suspendTerminalCursorBlink } from './pane-cursor-blink-suspension' import { shouldFollowMouseFocus } from './focus-follows-mouse' import { toPublicPane } from './pane-public-view' @@ -53,6 +54,10 @@ export function createManagedPaneInternal( } ) pane.webglAttachmentDeferred = host.isRenderingSuspended() + if (host.isRenderingSuspended()) { + // A pane that mounts behind a hidden surface never sees suspendPaneRendering(). + suspendTerminalCursorBlink(pane.terminal) + } host.panes.set(id, pane) host.identities.register(id, leafId) return pane diff --git a/src/renderer/src/lib/pane-manager/pane-rendering-control.ts b/src/renderer/src/lib/pane-manager/pane-rendering-control.ts index 6709a7a1f98..219f17f4be6 100644 --- a/src/renderer/src/lib/pane-manager/pane-rendering-control.ts +++ b/src/renderer/src/lib/pane-manager/pane-rendering-control.ts @@ -1,4 +1,8 @@ import type { ManagedPaneInternal } from './pane-manager-types' +import { + resumeTerminalCursorBlink, + suspendTerminalCursorBlink +} from './pane-cursor-blink-suspension' import { safeFit } from './pane-tree-ops' import { attachWebgl, @@ -66,6 +70,12 @@ export function suspendPaneRendering( for (const pane of suspended) { pane.webglAttachmentDeferred = true pane.terminal.blur() + // Why here, above the retention return: the retention branch keeps a live + // WebglRenderer, whose blink timer only stops on a real DOM blur event. blur() + // above is a no-op unless that pane's textarea held focus, so under a hide mode + // that keeps focus the pane would blink — redrawing its cursor row — until the + // 5-minute idle timeout. Parking the option makes it unconditional. + suspendTerminalCursorBlink(pane.terminal) } // Keep recent hidden worktrees on live WebGL so switch-back never presents // DOM-fallback frames; evicted/over-cap owners fall back to dispose. @@ -86,6 +96,9 @@ export function resumePaneRendering( } for (const pane of panes) { clearTerminalWebglAttachBackoff(pane) + // Before the attach below so a freshly constructed WebglRenderer already samples + // the restored option and blinks on its first frame. + resumeTerminalCursorBlink(pane.terminal) const rebuildDeferred = pane.webglRebuildDeferred === true pane.webglAttachmentDeferred = false // Reveal can retry before the next resume, so both paths share the bounded loss policy. diff --git a/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts b/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts index 2897c1627de..fef2c9f5d1e 100644 --- a/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts +++ b/src/renderer/src/lib/pane-manager/pane-webgl-context-recovery.test.ts @@ -14,6 +14,7 @@ function createPane(options: { loadAddon?: () => void } = {}): ManagedPaneIntern leafId, stablePaneId: leafId, terminal: { + options: { cursorBlink: true }, cols: 80, rows: 24, refresh: vi.fn(), diff --git a/src/renderer/src/lib/pane-manager/pane-webgl-refresh-lifecycle.test.ts b/src/renderer/src/lib/pane-manager/pane-webgl-refresh-lifecycle.test.ts index d6e30bb57f8..09f2ef98c84 100644 --- a/src/renderer/src/lib/pane-manager/pane-webgl-refresh-lifecycle.test.ts +++ b/src/renderer/src/lib/pane-manager/pane-webgl-refresh-lifecycle.test.ts @@ -17,6 +17,7 @@ function createPane( leafId, stablePaneId: leafId, terminal: { + options: { cursorBlink: true }, element: null, cols: 80, rows: 24, diff --git a/src/renderer/src/lib/pane-manager/terminal-webgl-hidden-retention.test.ts b/src/renderer/src/lib/pane-manager/terminal-webgl-hidden-retention.test.ts index e630e88d149..7d6bfdf26c2 100644 --- a/src/renderer/src/lib/pane-manager/terminal-webgl-hidden-retention.test.ts +++ b/src/renderer/src/lib/pane-manager/terminal-webgl-hidden-retention.test.ts @@ -10,7 +10,8 @@ import { function createPane(withAddon = true): ManagedPaneInternal { return { - terminal: { blur: vi.fn() }, + // options mirrors the real Terminal: suspend parks cursorBlink here. + terminal: { blur: vi.fn(), options: { cursorBlink: true } }, webglAddon: withAddon ? ({ dispose: vi.fn() } as unknown as ManagedPaneInternal['webglAddon']) : null, From b00ec2073133fb4e965e93640988966eccad1d01 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:10:02 -0700 Subject: [PATCH 088/398] perf(startup): stop an unreachable SSH host from gating local terminal restore (#18164) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(startup): stop an unreachable SSH host from gating local terminal restore An asleep or unreachable SSH target held the terminal-restoration gate for the full 15s reconnect timeout, so no terminal restored — local ones included. Startup now awaits only the target that owns the active workspace's tabs and lets the rest connect in the background, folded into the existing deferred path that reattaches their PTYs on tab focus. Also splits the renderer's git-environment fence out of the first-window PTY services barrier: worktree hydration needs shell-PATH generation and the managed WSL CLI registration, not a daemon PTY spawn or a hook-server bind. Terminal restoration still fences on the first-window services via app:prepareTerminalStartupRestoration. Measured with tests/tools/benchmarks/startup-time-bench.mjs (382 restored tabs, 28k-file profile, medians of 3): unreachable SSH host: 17.27s -> 1.34s to renderer-startup-hydration-done all-local: 1.98s -> 1.33s * fix(startup): restore the startup-ordering oracle and keep a connected background SSH target undeferred app-startup-routing.test.ts pinned the old step names, so the two ordering cases went vacuous-then-red when the barrier split. Repoint them at the steps that now carry the same fences: 'git-environment-barrier-await' (shell PATH + managed WSL, the fence host Git needs) before hydration worktrees, and 'prepare-terminal-startup-restoration' (which awaits firstWindowStartupServicesReady in main) before terminal reconnect. Both still fail against main's hydration source. Also: the timed-out-eager rewrite of the deferred list re-added background targets that had already connected, undoing removeDeferredSshReconnectTarget and sending fresh panes on a reachable host down the cold-restore path. --- .../startup/desktop-startup-ordering.test.ts | 42 ++++ .../startup/main-process-ipc-bootstrap.ts | 7 + .../startup/main-process-runtime-launch.ts | 3 + src/preload/api/app-api.ts | 3 + src/preload/api/app-bridge.ts | 2 + .../startup-actions-selector.test.ts | 1 + .../src/app-shell/startup-actions-selector.ts | 4 + .../app-shell/use-app-startup-hydration.ts | 30 ++- src/renderer/src/app-startup-routing.test.ts | 25 ++- .../active-workspace-ssh-targets.test.ts | 112 +++++++++ .../startup/active-workspace-ssh-targets.ts | 49 ++++ .../src/startup/ssh-startup-reconnect.ts | 18 +- .../startup-ssh-connection-restore.test.ts | 208 +++++++++++++++++ .../startup/startup-ssh-connection-restore.ts | 66 +++++- .../src/web/preload-api/web-app-api.ts | 1 + .../startup-bench-state-fixture.mjs | 193 ++++++++++++++++ tests/tools/benchmarks/startup-time-bench.mjs | 212 ++++-------------- 17 files changed, 783 insertions(+), 193 deletions(-) create mode 100644 src/renderer/src/startup/active-workspace-ssh-targets.test.ts create mode 100644 src/renderer/src/startup/active-workspace-ssh-targets.ts create mode 100644 src/renderer/src/startup/startup-ssh-connection-restore.test.ts create mode 100644 tests/tools/benchmarks/startup-bench-state-fixture.mjs diff --git a/src/main/startup/desktop-startup-ordering.test.ts b/src/main/startup/desktop-startup-ordering.test.ts index e432ed383d2..fc381f15714 100644 --- a/src/main/startup/desktop-startup-ordering.test.ts +++ b/src/main/startup/desktop-startup-ordering.test.ts @@ -183,6 +183,48 @@ describe('startup ordering', () => { ) }) + it('keeps the git-environment barrier off the PTY startup services', () => { + const barrierSource = readFileSync( + join(process.cwd(), 'src/main/startup/main-process-ipc-bootstrap.ts'), + 'utf8' + ) + const launchSource = readFileSync( + join(process.cwd(), 'src/main/startup/main-process-runtime-launch.ts'), + 'utf8' + ) + const gitBarrierStart = barrierSource.indexOf( + "ipcMain.handle('app:awaitGitEnvironmentStartupBarrier'" + ) + const gitBarrierEnd = barrierSource.indexOf( + "'app:prepareTerminalStartupRestoration'", + gitBarrierStart + ) + expect(gitBarrierStart).toBeGreaterThanOrEqual(0) + expect(gitBarrierEnd).toBeGreaterThan(gitBarrierStart) + const gitBarrier = barrierSource.slice(gitBarrierStart, gitBarrierEnd) + // The git environment fence is shell PATH + WSL registration; a daemon PTY provider or a + // hook-server bind here puts terminal startup back in front of worktree hydration. + expect(gitBarrier).toContain('state.shellPathReady') + expect(gitBarrier).toContain('state.managedWslCliStartupBarrierReady') + expect(gitBarrier).not.toContain('firstWindowStartupServicesReady') + // The published promise must be the same one the terminal startup services wait on. + expect(launchSource).toContain('state.shellPathReady = shellPathReady') + expect(launchSource.indexOf('state.shellPathReady = shellPathReady')).toBeLessThan( + launchSource.indexOf('await launchDesktopMode(') + ) + // Terminal restoration itself must still fence on the first-window services. + const restorationStart = barrierSource.indexOf( + "ipcMain.handle('app:prepareTerminalStartupRestoration'" + ) + const restorationEnd = barrierSource.indexOf( + "'app:recoverLegacyWorkerTerminalsForRendererStartup'", + restorationStart + ) + expect(barrierSource.slice(restorationStart, restorationEnd)).toContain( + 'state.firstWindowStartupServicesReady' + ) + }) + it('reconciles retained Codex homes after authoritative daemon inventory', () => { const source = readFileSync( join(process.cwd(), 'src/main/startup/main-process-pty-startup.ts'), diff --git a/src/main/startup/main-process-ipc-bootstrap.ts b/src/main/startup/main-process-ipc-bootstrap.ts index 89be84d2119..918fd844364 100644 --- a/src/main/startup/main-process-ipc-bootstrap.ts +++ b/src/main/startup/main-process-ipc-bootstrap.ts @@ -11,6 +11,13 @@ export function registerMainProcessIpcHandlers(): void { state.managedWslCliStartupBarrierReady ]) }) + // Why separate from the first-window barrier: host Git needs the shell-PATH + // generation and the managed WSL CLI registration, not a daemon PTY provider + // or a hook-server bind. Bundling them made worktree hydration wait on a + // terminal service it never calls. + ipcMain.handle('app:awaitGitEnvironmentStartupBarrier', async () => { + await Promise.all([state.shellPathReady, state.managedWslCliStartupBarrierReady]) + }) ipcMain.handle('app:prepareTerminalStartupRestoration', async () => { await Promise.all([ state.firstWindowStartupServicesReady, diff --git a/src/main/startup/main-process-runtime-launch.ts b/src/main/startup/main-process-runtime-launch.ts index 7e5f278fdc5..9df28ecea8c 100644 --- a/src/main/startup/main-process-runtime-launch.ts +++ b/src/main/startup/main-process-runtime-launch.ts @@ -289,6 +289,9 @@ export async function initializeMainProcessRuntimeLaunch( state.serveOptions = serveOptions const runtimeRpc = installRuntimeRpc(runtime, serveOptions) const shellPathReady = shellPathHydration.whenReady() + // Why published: the renderer's git-environment barrier must fence on the same + // generation the terminal startup services wait for, not a later re-read. + state.shellPathReady = shellPathReady let desktopWindow: BrowserWindow | null = null if (process.platform === 'win32' && app.isPackaged && !serveOptions) { const desktopStartup = startWindowsDesktopBeforeShellPathReady({ diff --git a/src/preload/api/app-api.ts b/src/preload/api/app-api.ts index 88cb86dc32b..b2d6eed966c 100644 --- a/src/preload/api/app-api.ts +++ b/src/preload/api/app-api.ts @@ -38,6 +38,9 @@ export type AppApi = { /** Resolves when the daemon PTY provider and hook receiver have either * started or failed open for the first BrowserWindow. */ awaitFirstWindowStartupServices: () => Promise + /** Resolves when host Git can run: shell-PATH generation is published and the + * managed WSL CLI registration has reconciled. Does not wait on PTY services. */ + awaitGitEnvironmentStartupBarrier: () => Promise /** Inventories retained PTYs and restores durable structured ownership before renderer adoption. */ prepareTerminalStartupRestoration: () => Promise /** Reconciles legacy worker authority around persisted terminal reconnect. */ diff --git a/src/preload/api/app-bridge.ts b/src/preload/api/app-bridge.ts index 22d46cc40d2..2a0d50e9de4 100644 --- a/src/preload/api/app-bridge.ts +++ b/src/preload/api/app-bridge.ts @@ -43,6 +43,8 @@ export const appApi = { awaitBeforeUnloadCheckpoint: () => awaitBeforeUnloadCheckpoint(), awaitFirstWindowStartupServices: (): Promise => ipcRenderer.invoke('app:awaitFirstWindowStartupServices'), + awaitGitEnvironmentStartupBarrier: (): Promise => + ipcRenderer.invoke('app:awaitGitEnvironmentStartupBarrier'), prepareTerminalStartupRestoration: (): Promise => ipcRenderer.invoke('app:prepareTerminalStartupRestoration'), recoverLegacyWorkerTerminalsForRendererStartup: (): Promise => diff --git a/src/renderer/src/app-shell/startup-actions-selector.test.ts b/src/renderer/src/app-shell/startup-actions-selector.test.ts index 1d559eb0dcb..45a1b9cf5df 100644 --- a/src/renderer/src/app-shell/startup-actions-selector.test.ts +++ b/src/renderer/src/app-shell/startup-actions-selector.test.ts @@ -30,6 +30,7 @@ function makeActions(): StartupActions { reconnectPersistedTerminals: vi.fn(), setTerminalStartupRestorationReady: vi.fn(), setDeferredSshReconnectTargets: vi.fn(), + removeDeferredSshReconnectTarget: vi.fn(), setSshConnectionState: vi.fn(), hydratePersistedUI: vi.fn(), setHydrationSucceeded: vi.fn(), diff --git a/src/renderer/src/app-shell/startup-actions-selector.ts b/src/renderer/src/app-shell/startup-actions-selector.ts index c7dca311a03..ac18349909a 100644 --- a/src/renderer/src/app-shell/startup-actions-selector.ts +++ b/src/renderer/src/app-shell/startup-actions-selector.ts @@ -22,6 +22,7 @@ export type StartupActions = Pick< | 'reconnectPersistedTerminals' | 'setTerminalStartupRestorationReady' | 'setDeferredSshReconnectTargets' + | 'removeDeferredSshReconnectTarget' | 'setSshConnectionState' | 'hydratePersistedUI' | 'setHydrationSucceeded' @@ -59,6 +60,8 @@ export function selectStartupActions(state: StartupActions): StartupActions { cachedStartupActions.setTerminalStartupRestorationReady === state.setTerminalStartupRestorationReady && cachedStartupActions.setDeferredSshReconnectTargets === state.setDeferredSshReconnectTargets && + cachedStartupActions.removeDeferredSshReconnectTarget === + state.removeDeferredSshReconnectTarget && cachedStartupActions.setSshConnectionState === state.setSshConnectionState && cachedStartupActions.hydratePersistedUI === state.hydratePersistedUI && cachedStartupActions.setHydrationSucceeded === state.setHydrationSucceeded && @@ -91,6 +94,7 @@ export function selectStartupActions(state: StartupActions): StartupActions { reconnectPersistedTerminals: state.reconnectPersistedTerminals, setTerminalStartupRestorationReady: state.setTerminalStartupRestorationReady, setDeferredSshReconnectTargets: state.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: state.removeDeferredSshReconnectTarget, setSshConnectionState: state.setSshConnectionState, hydratePersistedUI: state.hydratePersistedUI, setHydrationSucceeded: state.setHydrationSucceeded, diff --git a/src/renderer/src/app-shell/use-app-startup-hydration.ts b/src/renderer/src/app-shell/use-app-startup-hydration.ts index b50ac86705e..91db232081c 100644 --- a/src/renderer/src/app-shell/use-app-startup-hydration.ts +++ b/src/renderer/src/app-shell/use-app-startup-hydration.ts @@ -19,6 +19,7 @@ import { } from '../startup/startup-diagnostics' import { recoverFromDegradedStartup } from '../startup/startup-degraded-recovery' import { restoreSshConnectionsForStartup } from '../startup/startup-ssh-connection-restore' +import { collectActiveWorkspaceSshTargetIds } from '../startup/active-workspace-ssh-targets' import { publishTerminalViewAttributesAtAppStart } from '../components/terminal-pane/terminal-appearance' import { getSystemPrefersDark } from '../lib/terminal-theme' import { @@ -155,9 +156,12 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta // Why: disconnected SSH repos hydrate from local metadata; only runtime-owned repos use placeholders. parseExecutionHostId(getRepoExecutionHostId(repo))?.kind !== 'runtime' ) - // Why: worktree refresh can spawn host Git; wait for main's shell-PATH generation fence first. - await timeRendererStartupStep('first-window-services-await', () => - window.api.app.awaitFirstWindowStartupServices() + // Why this barrier and not the first-window one: worktree refresh can spawn host Git, + // which needs the shell-PATH generation and the managed WSL CLI registration. It never + // needs the daemon PTY provider or the hook-server bind, and `prepare-terminal-startup-restoration` + // below still fences those before any terminal is restored. + await timeRendererStartupStep('git-environment-barrier-await', () => + window.api.app.awaitGitEnvironmentStartupBarrier() ) await timeRendererStartupStep('fetch-hydration-worktrees', () => mapWithConcurrency(hydrationRepos, WORKTREE_REFRESH_CONCURRENCY, (repo) => @@ -213,9 +217,14 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta actions.pruneLastVisitedTimestamps() actions.seedActiveWorktreeLastVisitedIfMissing() }) - await timeRendererStartupStep('fetch-browser-session-profiles', () => + // Why started here but not awaited: on a remote runtime this is an RPC with a 15s + // timeout, and nothing between here and terminal restoration reads the profile list — + // awaiting it put that timeout on the terminal-restoration gate. Starting it at the + // original point keeps the profiles landing no later than they did before; the action + // swallows its own failures, so the `.catch` only marks the timing wrapper handled. + void timeRendererStartupStep('fetch-browser-session-profiles', () => actions.fetchBrowserSessionProfiles() - ) + ).catch(() => {}) const onboardingState = await onboardingPromise if (!cancelled) { onOnboardingLoadedRef.current(onboardingState) @@ -228,9 +237,17 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta ) if (connectionIds.length > 0) { try { + // Why scoped: an unreachable host used to hold every restored terminal — local ones + // included — for the full reconnect timeout. Only the targets whose panes mount as + // soon as the gate opens are worth waiting for; the rest reattach on tab focus. + const blockingConnectionIds = collectActiveWorkspaceSshTargetIds( + useAppStore.getState() + ) await restoreSshConnectionsForStartup({ connectionIds, + blockingConnectionIds, setDeferredSshReconnectTargets: actions.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: actions.removeDeferredSshReconnectTarget, publishSshConnectionState: actions.setSshConnectionState }) } catch (err) { @@ -240,7 +257,8 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta logRendererStartupDiagnostic('ssh-reconnect-skipped', { connectionIds: 0 }) } - // first-window-services-await already fenced worktree hydration; terminal recovery reuses that ready state. + // Why no explicit barrier here: prepare-terminal-startup-restoration above already awaited + // the first-window services, and main re-awaits them inside this handler anyway. await timeRendererStartupStep('recover-legacy-worker-terminals-pre-reconnect', () => window.api.app.recoverLegacyWorkerTerminalsForRendererStartup() ) diff --git a/src/renderer/src/app-startup-routing.test.ts b/src/renderer/src/app-startup-routing.test.ts index 24a9cc75120..d1aad35829a 100644 --- a/src/renderer/src/app-startup-routing.test.ts +++ b/src/renderer/src/app-startup-routing.test.ts @@ -68,8 +68,11 @@ describe('renderer startup runtime routing', () => { const hydrationWorktreesIndex = source.indexOf( "timeRendererStartupStep('fetch-hydration-worktrees'" ) - const servicesIndex = source.indexOf( - "timeRendererStartupStep('first-window-services-await'", + // Why this barrier: worktree hydration can spawn host Git, so it must sit behind the + // shell-PATH + managed-WSL fence. On packaged Windows the window opens before + // shellPathReady resolves, so this really is the fence, not a formality. + const gitEnvironmentBarrierIndex = source.indexOf( + "timeRendererStartupStep('git-environment-barrier-await'", sessionIndex ) const fullWorktreesIndex = source.indexOf('await actions.fetchAllWorktrees()') @@ -89,8 +92,11 @@ describe('renderer startup runtime routing', () => { expect(localReposIndex).toBeLessThan(localGroupsIndex) expect(localGroupsIndex).toBeLessThan(localFoldersIndex) expect(localReposIndex).toBeLessThan(sessionIndex) - expect(sessionIndex).toBeLessThan(servicesIndex) - expect(servicesIndex).toBeLessThan(hydrationWorktreesIndex) + expect(sessionIndex).toBeLessThan(gitEnvironmentBarrierIndex) + expect(gitEnvironmentBarrierIndex).toBeLessThan(hydrationWorktreesIndex) + expect(source.slice(gitEnvironmentBarrierIndex, hydrationWorktreesIndex)).toContain( + 'window.api.app.awaitGitEnvironmentStartupBarrier()' + ) const hydrationWorktreeBlock = source.slice( hydrationWorktreesIndex, source.indexOf('await keybindingsPromise') @@ -180,7 +186,13 @@ describe('renderer startup runtime routing', () => { it('waits for first-window startup services before terminal reconnect', () => { const source = readSource(STARTUP_HYDRATION_PATH) - const servicesIndex = source.indexOf("timeRendererStartupStep('first-window-services-await'") + // Why this step: `app:prepareTerminalStartupRestoration` awaits + // firstWindowStartupServicesReady + managedWslCliStartupBarrierReady in main before it + // does anything else, so it is the renderer-side position of that fence. + // `desktop-startup-ordering.test.ts` pins the main-side await itself. + const servicesIndex = source.indexOf( + "timeRendererStartupStep('prepare-terminal-startup-restoration'" + ) const preReconnectRecoveryIndex = source.indexOf( "timeRendererStartupStep('recover-legacy-worker-terminals-pre-reconnect'" ) @@ -193,6 +205,9 @@ describe('renderer startup runtime routing', () => { ) expect(servicesIndex).toBeGreaterThanOrEqual(0) + expect(source.slice(servicesIndex)).toContain( + 'window.api.app.prepareTerminalStartupRestoration()' + ) expect(preReconnectRecoveryIndex).toBeGreaterThan(servicesIndex) expect(capabilityRefreshIndex).toBeGreaterThan(preReconnectRecoveryIndex) expect(reconnectIndex).toBeGreaterThan(capabilityRefreshIndex) diff --git a/src/renderer/src/startup/active-workspace-ssh-targets.test.ts b/src/renderer/src/startup/active-workspace-ssh-targets.test.ts new file mode 100644 index 00000000000..f36a186b0ac --- /dev/null +++ b/src/renderer/src/startup/active-workspace-ssh-targets.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it } from 'vitest' +import { toAppSshPtyId } from '../../../shared/ssh-pty-id' +import type { TerminalTab } from '../../../shared/terminal-tab-types' +import { collectActiveWorkspaceSshTargetIds } from './active-workspace-ssh-targets' + +function tab(id: string, ptyId: string | null = null): TerminalTab { + return { + id, + ptyId, + worktreeId: 'repo-a::/w/a', + title: id, + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 1 + } as TerminalTab +} + +const emptyInput = { + activeWorktreeId: null as string | null, + tabsByWorktree: {} as Record, + pendingReconnectPtyIdByTabId: {} as Record, + terminalLayoutsByTabId: {} as Record }>, + repos: [] as { id: string; connectionId?: string | null }[] +} + +describe('collectActiveWorkspaceSshTargetIds', () => { + it('returns nothing when no workspace is active', () => { + expect(collectActiveWorkspaceSshTargetIds(emptyInput)).toEqual([]) + }) + + it('returns nothing for a purely local active workspace', () => { + expect( + collectActiveWorkspaceSshTargetIds({ + ...emptyInput, + activeWorktreeId: 'repo-a::/w/a', + tabsByWorktree: { 'repo-a::/w/a': [tab('t1', 'local-pty-1')] }, + repos: [{ id: 'repo-a', connectionId: null }] + }) + ).toEqual([]) + }) + + it('names the target from the active workspace repo connection', () => { + expect( + collectActiveWorkspaceSshTargetIds({ + ...emptyInput, + activeWorktreeId: 'repo-remote::/srv/w', + repos: [{ id: 'repo-remote', connectionId: 'ssh-1' }] + }) + ).toEqual(['ssh-1']) + }) + + it('names the target from a restored PTY id when the repo catalog has no connection', () => { + // SSH worktrees are absent from worktreesByRepo at cold start; the PTY id is the durable name. + expect( + collectActiveWorkspaceSshTargetIds({ + ...emptyInput, + activeWorktreeId: 'repo-a::/w/a', + tabsByWorktree: { 'repo-a::/w/a': [tab('t1')] }, + pendingReconnectPtyIdByTabId: { t1: toAppSshPtyId('ssh-2', 'pty-9') }, + repos: [{ id: 'repo-a' }] + }) + ).toEqual(['ssh-2']) + }) + + it('names split-leaf targets on the active workspace', () => { + expect( + collectActiveWorkspaceSshTargetIds({ + ...emptyInput, + activeWorktreeId: 'repo-a::/w/a', + tabsByWorktree: { 'repo-a::/w/a': [tab('t1')] }, + terminalLayoutsByTabId: { + t1: { + ptyIdsByLeafId: { + leaf1: toAppSshPtyId('ssh-3', 'pty-1'), + leaf2: null, + leaf3: 'local-pty-2' + } + } + }, + repos: [{ id: 'repo-a' }] + }) + ).toEqual(['ssh-3']) + }) + + it('ignores targets that only own an inactive workspace', () => { + expect( + collectActiveWorkspaceSshTargetIds({ + ...emptyInput, + activeWorktreeId: 'repo-a::/w/a', + tabsByWorktree: { + 'repo-a::/w/a': [tab('t1', 'local-pty-1')], + 'repo-b::/w/b': [tab('t2', toAppSshPtyId('ssh-other', 'pty-1'))] + }, + repos: [{ id: 'repo-a' }, { id: 'repo-b', connectionId: 'ssh-other' }] + }) + ).toEqual([]) + }) + + it('deduplicates a target named by both the repo and its PTY ids', () => { + expect( + collectActiveWorkspaceSshTargetIds({ + ...emptyInput, + activeWorktreeId: 'repo-remote::/srv/w', + tabsByWorktree: { + 'repo-remote::/srv/w': [tab('t1', toAppSshPtyId('ssh-1', 'pty-1'))] + }, + repos: [{ id: 'repo-remote', connectionId: 'ssh-1' }] + }) + ).toEqual(['ssh-1']) + }) +}) diff --git a/src/renderer/src/startup/active-workspace-ssh-targets.ts b/src/renderer/src/startup/active-workspace-ssh-targets.ts new file mode 100644 index 00000000000..c7514abf4a1 --- /dev/null +++ b/src/renderer/src/startup/active-workspace-ssh-targets.ts @@ -0,0 +1,49 @@ +import { parseAppSshPtyId } from '../../../shared/ssh-pty-id' +import { getRepoIdFromWorktreeId } from '../../../shared/worktree/id' + +type ActiveWorkspaceSshTargetInput = { + activeWorktreeId: string | null + tabsByWorktree: Readonly> + /** Restored tab-level PTY ids, keyed by tab id. */ + pendingReconnectPtyIdByTabId: Readonly> + terminalLayoutsByTabId: Readonly< + Record> } | undefined> + > + repos: readonly { id: string; connectionId?: string | null }[] +} + +/** + * SSH targets that own terminals the user sees the moment the startup gate opens: the ones + * whose reconnect must still be awaited. Everything else can connect in the background and + * reattach on tab focus. + * + * Derived from the restored PTY ids rather than the repo catalog alone, because SSH worktrees + * are absent from `worktreesByRepo` at cold start — the PTY id is the durable name of the + * target the pane will reattach. + */ +export function collectActiveWorkspaceSshTargetIds(input: ActiveWorkspaceSshTargetInput): string[] { + const { activeWorktreeId } = input + if (!activeWorktreeId) { + return [] + } + const targetIds = new Set() + const repoId = getRepoIdFromWorktreeId(activeWorktreeId) + const connectionId = input.repos.find((repo) => repo.id === repoId)?.connectionId + if (connectionId) { + targetIds.add(connectionId) + } + for (const tab of input.tabsByWorktree[activeWorktreeId] ?? []) { + const ptyIds = [ + tab.ptyId, + input.pendingReconnectPtyIdByTabId[tab.id], + ...Object.values(input.terminalLayoutsByTabId[tab.id]?.ptyIdsByLeafId ?? {}) + ] + for (const ptyId of ptyIds) { + const parsed = ptyId ? parseAppSshPtyId(ptyId) : null + if (parsed) { + targetIds.add(parsed.connectionId) + } + } + } + return [...targetIds] +} diff --git a/src/renderer/src/startup/ssh-startup-reconnect.ts b/src/renderer/src/startup/ssh-startup-reconnect.ts index d312239a878..e31bd963d6f 100644 --- a/src/renderer/src/startup/ssh-startup-reconnect.ts +++ b/src/renderer/src/startup/ssh-startup-reconnect.ts @@ -6,7 +6,9 @@ export type SshStartupReconnectResult = { export async function reconnectSshTargetForRendererStartup(args: { targetId: string - timeoutMs: number + /** Omitted for a connect nobody is waiting on — no timer, so it cannot report + * a timeout the caller has no use for. */ + timeoutMs?: number connect: (targetId: string) => Promise publishState: (targetId: string, state: SshConnectionState) => void onFailure: (targetId: string, error: unknown) => void @@ -14,10 +16,16 @@ export async function reconnectSshTargetForRendererStartup(args: { const { targetId, timeoutMs, connect, publishState, onFailure } = args let timeoutId: ReturnType | null = null try { - const timeout = new Promise((_resolve, reject) => { - timeoutId = setTimeout(() => reject(new Error('SSH reconnect timeout')), timeoutMs) - }) - const state = await Promise.race([connect(targetId), timeout]) + const connected = connect(targetId) + const state = + timeoutMs === undefined + ? await connected + : await Promise.race([ + connected, + new Promise((_resolve, reject) => { + timeoutId = setTimeout(() => reject(new Error('SSH reconnect timeout')), timeoutMs) + }) + ]) // Why: the state-change IPC can trail connect's resolution. Publish the // authoritative result before restored terminals inspect renderer state. if (state) { diff --git a/src/renderer/src/startup/startup-ssh-connection-restore.test.ts b/src/renderer/src/startup/startup-ssh-connection-restore.test.ts new file mode 100644 index 00000000000..3464816d0d8 --- /dev/null +++ b/src/renderer/src/startup/startup-ssh-connection-restore.test.ts @@ -0,0 +1,208 @@ +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import type { SshConnectionState, SshProviderEpoch, SshTarget } from '../../../shared/ssh-types' +import { restoreSshConnectionsForStartup } from './startup-ssh-connection-restore' + +function connectedState(targetId: string): SshConnectionState { + return { + targetId, + status: 'connected', + error: null, + reconnectAttempt: 0, + providerEpoch: 'epoch' as SshProviderEpoch, + connectionGeneration: 1, + remotePlatform: 'linux' + } +} + +function target(id: string, lastRequiredPassphrase = false): SshTarget { + return { + id, + label: id, + host: `${id}.example`, + port: 22, + username: 'orca', + lastRequiredPassphrase + } +} + +type Harness = { + connect: Mock<(targetId: string) => Promise> + getState: Mock<(targetId: string) => Promise> + setDeferredSshReconnectTargets: Mock<(targetIds: string[]) => void> + removeDeferredSshReconnectTarget: Mock<(targetId: string) => void> + publishSshConnectionState: Mock<(targetId: string, state: SshConnectionState) => void> +} + +let harness: Harness + +function installWindowApi(targets: SshTarget[]): void { + harness = { + connect: vi.fn(), + getState: vi.fn().mockResolvedValue(null), + setDeferredSshReconnectTargets: vi.fn(), + removeDeferredSshReconnectTarget: vi.fn(), + publishSshConnectionState: vi.fn() + } + vi.stubGlobal('window', { + api: { + app: { startupDiagnostic: undefined }, + ssh: { + listTargets: vi.fn().mockResolvedValue(targets), + connect: (args: { targetId: string }) => harness.connect(args.targetId), + getState: (args: { targetId: string }) => harness.getState(args.targetId) + } + } + }) +} + +beforeEach(() => { + vi.spyOn(console, 'warn').mockImplementation(() => {}) +}) + +afterEach(() => { + vi.useRealTimers() + vi.unstubAllGlobals() + vi.restoreAllMocks() +}) + +describe('restoreSshConnectionsForStartup', () => { + it('does not wait on a target that owns no immediately-mounted pane', async () => { + vi.useFakeTimers() + installWindowApi([target('ssh-active'), target('ssh-asleep')]) + // The asleep host never answers — the old code awaited it for the full timeout. + harness.connect.mockImplementation((targetId: string) => + targetId === 'ssh-active' + ? Promise.resolve(connectedState(targetId)) + : new Promise(() => {}) + ) + + let settled = false + const restore = restoreSshConnectionsForStartup({ + connectionIds: ['ssh-active', 'ssh-asleep'], + blockingConnectionIds: ['ssh-active'], + setDeferredSshReconnectTargets: harness.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: harness.removeDeferredSshReconnectTarget, + publishSshConnectionState: harness.publishSshConnectionState + }).then(() => { + settled = true + }) + + await vi.advanceTimersByTimeAsync(0) + await restore + expect(settled).toBe(true) + // Both were dialled; only the active one gated restoration. + expect(harness.connect).toHaveBeenCalledWith('ssh-active') + expect(harness.connect).toHaveBeenCalledWith('ssh-asleep') + expect(harness.publishSshConnectionState).toHaveBeenCalledWith( + 'ssh-active', + connectedState('ssh-active') + ) + // The unreachable host is deferred, so its panes reattach on tab focus. + expect(harness.setDeferredSshReconnectTargets).toHaveBeenCalledWith(['ssh-asleep']) + }) + + it('awaits the target that owns the active workspace', async () => { + vi.useFakeTimers() + installWindowApi([target('ssh-active')]) + harness.connect.mockReturnValue(new Promise(() => {})) + + let settled = false + const restore = restoreSshConnectionsForStartup({ + connectionIds: ['ssh-active'], + blockingConnectionIds: ['ssh-active'], + setDeferredSshReconnectTargets: harness.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: harness.removeDeferredSshReconnectTarget, + publishSshConnectionState: harness.publishSshConnectionState + }).then(() => { + settled = true + }) + + await vi.advanceTimersByTimeAsync(14_000) + expect(settled).toBe(false) + await vi.advanceTimersByTimeAsync(1_000) + await restore + expect(settled).toBe(true) + expect(harness.setDeferredSshReconnectTargets).toHaveBeenLastCalledWith(['ssh-active']) + }) + + it('clears the deferred flag once a background target connects', async () => { + installWindowApi([target('ssh-bg')]) + harness.connect.mockResolvedValue(connectedState('ssh-bg')) + + await restoreSshConnectionsForStartup({ + connectionIds: ['ssh-bg'], + blockingConnectionIds: [], + setDeferredSshReconnectTargets: harness.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: harness.removeDeferredSshReconnectTarget, + publishSshConnectionState: harness.publishSshConnectionState + }) + await vi.waitFor(() => + expect(harness.removeDeferredSshReconnectTarget).toHaveBeenCalledWith('ssh-bg') + ) + expect(harness.publishSshConnectionState).toHaveBeenCalledWith( + 'ssh-bg', + connectedState('ssh-bg') + ) + }) + + it('does not push a connected background target back into the deferred list', async () => { + installWindowApi([target('ssh-active'), target('ssh-bg')]) + // The active host never answers and times out; the background host connects first. + harness.connect.mockImplementation((targetId: string) => + targetId === 'ssh-bg' + ? Promise.resolve(connectedState(targetId)) + : new Promise(() => {}) + ) + + await restoreSshConnectionsForStartup({ + connectionIds: ['ssh-active', 'ssh-bg'], + blockingConnectionIds: ['ssh-active'], + setDeferredSshReconnectTargets: harness.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: harness.removeDeferredSshReconnectTarget, + publishSshConnectionState: harness.publishSshConnectionState + }) + + expect(harness.removeDeferredSshReconnectTarget).toHaveBeenCalledWith('ssh-bg') + // The timed-out rewrite must not resurrect the reachable background target: a deferred + // connected target sends fresh panes down the cold-restore path instead of the normal one. + expect(harness.setDeferredSshReconnectTargets).toHaveBeenLastCalledWith(['ssh-active']) + }, 30_000) + + it('keeps passphrase targets deferred and never dials them', async () => { + installWindowApi([target('ssh-key', true), target('ssh-bg')]) + harness.connect.mockResolvedValue(connectedState('ssh-bg')) + + await restoreSshConnectionsForStartup({ + connectionIds: ['ssh-key', 'ssh-bg'], + blockingConnectionIds: [], + setDeferredSshReconnectTargets: harness.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: harness.removeDeferredSshReconnectTarget, + publishSshConnectionState: harness.publishSshConnectionState + }) + + expect(harness.connect).not.toHaveBeenCalledWith('ssh-key') + expect(harness.setDeferredSshReconnectTargets).toHaveBeenCalledWith(['ssh-key', 'ssh-bg']) + }) + + it('awaits every target when no blocking set is supplied', async () => { + vi.useFakeTimers() + installWindowApi([target('ssh-a'), target('ssh-b')]) + harness.connect.mockReturnValue(new Promise(() => {})) + + let settled = false + const restore = restoreSshConnectionsForStartup({ + connectionIds: ['ssh-a', 'ssh-b'], + setDeferredSshReconnectTargets: harness.setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget: harness.removeDeferredSshReconnectTarget, + publishSshConnectionState: harness.publishSshConnectionState + }).then(() => { + settled = true + }) + + await vi.advanceTimersByTimeAsync(14_000) + expect(settled).toBe(false) + await vi.advanceTimersByTimeAsync(1_000) + await restore + expect(harness.setDeferredSshReconnectTargets).toHaveBeenLastCalledWith(['ssh-a', 'ssh-b']) + }) +}) diff --git a/src/renderer/src/startup/startup-ssh-connection-restore.ts b/src/renderer/src/startup/startup-ssh-connection-restore.ts index b3ad1a42d61..35efaed2bfa 100644 --- a/src/renderer/src/startup/startup-ssh-connection-restore.ts +++ b/src/renderer/src/startup/startup-ssh-connection-restore.ts @@ -8,13 +8,27 @@ const SSH_RECONNECT_TIMEOUT_MS = 15_000 * Re-establishes the SSH targets that were live at shutdown before terminal reconnect, so * SSH-backed tabs route through pty.attach. Passphrase-protected and timed-out targets are * handed back as deferred so their PTYs reattach on tab focus instead of stacking dialogs. + * + * Only `blockingConnectionIds` are awaited. Every other target connects in the background and + * is registered as deferred up front, so an unreachable host cannot hold local terminal + * restoration for the reconnect timeout. Background connects keep running in main; the pane's + * deferred flow joins the same in-flight `ssh.connect` on tab focus. */ export async function restoreSshConnectionsForStartup(args: { connectionIds: string[] + /** Targets whose panes mount as soon as the startup gate opens. Omitted = await all. */ + blockingConnectionIds?: readonly string[] setDeferredSshReconnectTargets: (targetIds: string[]) => void + removeDeferredSshReconnectTarget: (targetId: string) => void publishSshConnectionState: (targetId: string, state: SshConnectionState) => void }): Promise { - const { connectionIds, setDeferredSshReconnectTargets, publishSshConnectionState } = args + const { + connectionIds, + blockingConnectionIds, + setDeferredSshReconnectTargets, + removeDeferredSshReconnectTarget, + publishSshConnectionState + } = args const allTargets = await timeRendererStartupStep('ssh-list-targets', () => window.api.ssh.listTargets() ) @@ -24,11 +38,42 @@ export async function restoreSshConnectionsForStartup(args: { needsPassphrase: targetMap.get(targetId)?.lastRequiredPassphrase ?? false })) - const eagerTargets = targets.filter((t) => !t.needsPassphrase) - const deferredTargets = targets.filter((t) => t.needsPassphrase) + const passphraseTargetIds = targets.filter((t) => t.needsPassphrase).map((t) => t.targetId) + const blocking = blockingConnectionIds ? new Set(blockingConnectionIds) : null + const eagerTargets = targets.filter( + (t) => !t.needsPassphrase && (blocking === null || blocking.has(t.targetId)) + ) + const backgroundTargets = targets.filter( + (t) => !t.needsPassphrase && blocking !== null && !blocking.has(t.targetId) + ) - if (deferredTargets.length > 0) { - setDeferredSshReconnectTargets(deferredTargets.map((t) => t.targetId)) + const deferredTargetIds = [...passphraseTargetIds, ...backgroundTargets.map((t) => t.targetId)] + if (deferredTargetIds.length > 0) { + setDeferredSshReconnectTargets(deferredTargetIds) + } + + // Why tracked: the timed-out branch below rewrites the whole deferred list, and a + // background target that already connected must not be pushed back into it. + const connectedBackgroundTargetIds = new Set() + // Why fired before the awaited group: a background target that lands before terminal + // reconnect reads as an ordinary connected target, exactly as it does today. + for (const { targetId } of backgroundTargets) { + void reconnectSshTargetForRendererStartup({ + targetId, + connect: (id) => window.api.ssh.connect({ targetId: id }), + publishState: (id, state) => { + publishSshConnectionState(id, state) + if (state.status === 'connected') { + // Why: a still-deferred connected target sends fresh panes down the deferred + // spawn path instead of the normal one. Clear it as soon as it is reachable. + connectedBackgroundTargetIds.add(id) + removeDeferredSshReconnectTarget(id) + } + }, + onFailure: (id, error) => { + console.warn(`SSH background auto-reconnect failed for ${id}:`, error) + } + }) } // Why: treat timed-out eager targets as deferred so their PTYs reattach on tab focus (ssh.connect keeps running in main and likely finishes by then). @@ -52,10 +97,17 @@ export async function restoreSshConnectionsForStartup(args: { } }) ), - { eagerTargets: eagerTargets.length, deferredTargets: deferredTargets.length } + { + eagerTargets: eagerTargets.length, + deferredTargets: passphraseTargetIds.length, + backgroundTargets: backgroundTargets.length + } ) if (timedOutTargets.length > 0) { - setDeferredSshReconnectTargets([...deferredTargets.map((t) => t.targetId), ...timedOutTargets]) + setDeferredSshReconnectTargets([ + ...deferredTargetIds.filter((id) => !connectedBackgroundTargetIds.has(id)), + ...timedOutTargets + ]) } // Why: older/wrapped providers may return no state from connect; poll main once as a compatibility fallback before terminal restoration. diff --git a/src/renderer/src/web/preload-api/web-app-api.ts b/src/renderer/src/web/preload-api/web-app-api.ts index f182b8a7790..28fd9f99176 100644 --- a/src/renderer/src/web/preload-api/web-app-api.ts +++ b/src/renderer/src/web/preload-api/web-app-api.ts @@ -33,6 +33,7 @@ export function createWebAppApi(): Partial { // Staging already wrote through to browser storage, so there is nothing left to join. awaitBeforeUnloadCheckpoint: () => Promise.resolve(), awaitFirstWindowStartupServices: () => Promise.resolve(), + awaitGitEnvironmentStartupBarrier: () => Promise.resolve(), prepareTerminalStartupRestoration: () => Promise.resolve(), recoverLegacyWorkerTerminalsForRendererStartup: () => Promise.resolve(), startupDiagnostic: () => Promise.resolve(), diff --git a/tests/tools/benchmarks/startup-bench-state-fixture.mjs b/tests/tools/benchmarks/startup-bench-state-fixture.mjs new file mode 100644 index 00000000000..87f353bcfce --- /dev/null +++ b/tests/tools/benchmarks/startup-bench-state-fixture.mjs @@ -0,0 +1,193 @@ +/** + * Persisted-state fixtures for the startup benchmark: the git repos, GitHub + * remotes, restored terminal tabs, and unreachable SSH targets that `orca-data.json` + * must contain for a run to exercise the corresponding startup path. + */ +import { spawnSync } from 'node:child_process' +import { existsSync, mkdirSync, realpathSync, unlinkSync, writeFileSync } from 'node:fs' +import { join } from 'node:path' + +function initFixtureGitRepo(repoDir) { + mkdirSync(repoDir, { recursive: true }) + if (!existsSync(join(repoDir, '.git'))) { + const init = spawnSync('git', ['init', repoDir], { stdio: 'ignore' }) + if (init.status !== 0) { + throw new Error(`Failed to create git repo fixture at ${repoDir}`) + } + } + return realpathSync(repoDir) +} + +/** + * Seed repos whose hydration reaches the `gh` login probe: a GitHub `origin` + * remote and no github.user/user.username config (the bench also points + * GIT_CONFIG_GLOBAL away from the developer's real config at launch). + */ +function buildGithubRepoFixtures(fixtureDir, githubRepos) { + const repos = [] + for (let i = 0; i < githubRepos; i++) { + const repoPath = initFixtureGitRepo(join(fixtureDir, `bench-gh-repo-${i}`)) + const remote = spawnSync( + 'git', + [ + '-C', + repoPath, + 'remote', + 'add', + 'origin', + `https://github.com/orca-bench/bench-gh-repo-${i}.git` + ], + { stdio: 'ignore' } + ) + // Exit 3 (remote exists) is fine on fixture reuse; anything else is not. + if (remote.status !== 0 && remote.status !== 3) { + throw new Error(`Failed to add GitHub remote to ${repoPath}`) + } + repos.push({ + id: `bench-gh-repo-${i}`, + path: repoPath, + displayName: `Bench GH Repo ${i}`, + badgeColor: '#000000', + addedAt: 1, + externalWorktreeVisibility: 'show' + }) + } + return repos +} + +/** + * SSH targets on TEST-NET-3 (RFC 5737). The address is guaranteed unroutable, + * so the TCP handshake never completes and never gets a reset — the wire + * behaviour of a host that is asleep or behind a dropped VPN. + */ +function buildUnreachableSshTargets(count) { + const targets = [] + for (let i = 0; i < count; i++) { + targets.push({ + id: `bench-ssh-unreachable-${i}`, + label: `Unreachable Host ${i}`, + host: `203.0.113.${i + 1}`, + port: 22, + username: 'orca', + source: 'manual', + lastRequiredPassphrase: false + }) + } + return targets +} + +export function writePersistedStateFixture( + fixtureDir, + { stateProfile, sessionTabs, githubRepos, sshUnreachableTargets = 0 } +) { + const dataPath = join(fixtureDir, 'orca-data.json') + if (stateProfile === 'none' && githubRepos === 0 && sshUnreachableTargets === 0) { + try { + unlinkSync(dataPath) + } catch { + // no persisted state fixture + } + return 0 + } + if (!['none', 'restored-local-tabs'].includes(stateProfile)) { + throw new Error(`Unknown state profile: ${stateProfile}`) + } + + const githubRepoEntries = buildGithubRepoFixtures(fixtureDir, githubRepos) + const sshTargets = buildUnreachableSshTargets(sshUnreachableTargets) + if (stateProfile === 'none') { + const state = { + schemaVersion: 1, + ...(sshTargets.length > 0 ? { sshTargets } : {}), + repos: githubRepoEntries, + settings: { + telemetry: { + installId: 'startup-bench', + optedIn: false, + existedBeforeTelemetryRelease: true + } + } + } + const json = JSON.stringify(state, null, 2) + writeFileSync(dataPath, json, 'utf-8') + return Buffer.byteLength(json) + } + + const repoPath = initFixtureGitRepo(join(fixtureDir, 'bench-repo')) + const repoId = 'bench-repo' + const worktreeId = `${repoId}::${repoPath}` + const tabCount = Math.max(1, sessionTabs) + const tabs = [] + const terminalLayoutsByTabId = {} + const activeTabIdByWorktree = {} + for (let i = 0; i < tabCount; i++) { + const tabId = `bench-tab-${String(i).padStart(5, '0')}` + const ptyId = `bench-pty-${String(i).padStart(5, '0')}` + tabs.push({ + id: tabId, + ptyId, + worktreeId, + title: `Terminal ${i + 1}`, + customTitle: null, + color: null, + sortOrder: i, + createdAt: 1 + }) + terminalLayoutsByTabId[tabId] = { + root: null, + activeLeafId: null, + expandedLeafId: null + } + } + activeTabIdByWorktree[worktreeId] = tabs[0]?.id ?? null + const state = { + schemaVersion: 1, + repos: [ + { + id: repoId, + path: repoPath, + displayName: 'Bench Repo', + badgeColor: '#000000', + addedAt: 1, + externalWorktreeVisibility: 'show' + }, + ...githubRepoEntries + ], + settings: { + telemetry: { + installId: 'startup-bench', + optedIn: false, + existedBeforeTelemetryRelease: true + } + }, + ui: { + lastActiveRepoId: repoId, + lastActiveWorktreeId: worktreeId + }, + workspaceSession: { + activeRepoId: repoId, + activeWorktreeId: worktreeId, + activeTabId: tabs[0]?.id ?? null, + tabsByWorktree: { + [worktreeId]: tabs + }, + terminalLayoutsByTabId, + activeTabIdByWorktree, + activeWorktreeIdsOnShutdown: [worktreeId], + defaultTerminalTabsAppliedByWorktreeId: { + [worktreeId]: true + }, + // Why on the session and not just the target list: startup reconnect only + // dials targets that were connected at shutdown. + ...(sshTargets.length > 0 + ? { activeConnectionIdsAtShutdown: sshTargets.map((target) => target.id) } + : {}) + } + } + if (sshTargets.length > 0) { + state.sshTargets = sshTargets + } + const json = JSON.stringify(state, null, 2) + writeFileSync(dataPath, json, 'utf-8') + return Buffer.byteLength(json) +} diff --git a/tests/tools/benchmarks/startup-time-bench.mjs b/tests/tools/benchmarks/startup-time-bench.mjs index 29363432c7a..e39f2d1d133 100644 --- a/tests/tools/benchmarks/startup-time-bench.mjs +++ b/tests/tools/benchmarks/startup-time-bench.mjs @@ -12,6 +12,7 @@ * node tests/tools/benchmarks/startup-time-bench.mjs --label baseline * [--iterations 5] [--files 28000] [--fixture-dir ] * [--state-profile none|restored-local-tabs] [--session-tabs 200] + * [--ssh-unreachable-targets 1] * [--github-repos 3] [--gh-hang-ms 30000] * [--wait-for-event renderer-startup-hydration-done] * [--exe ] [--timeout-ms 240000] @@ -27,20 +28,14 @@ * Results: tests/tools/benchmarks/results/startup-
    diff --git a/src/renderer/src/components/landing-github-star-state.ts b/src/renderer/src/components/landing-github-star-state.ts new file mode 100644 index 00000000000..098646b00f5 --- /dev/null +++ b/src/renderer/src/components/landing-github-star-state.ts @@ -0,0 +1,39 @@ +import { useEffect, useState, type Dispatch, type SetStateAction } from 'react' + +export type LandingStarState = 'loading' | 'starred' | 'not-starred' | 'web-fallback' | 'hidden' + +/** + * Resolve the viewer's Orca star state once per Landing mount. + * + * Why it lives here and not in the star button: the button renders inside a + * footer that is conditionally mounted on whether `repos` currently carries a + * GitHub provider identity, and `repos` is rewritten wholesale on every + * repo-catalog push. Every flicker of that condition re-ran the button's mount + * effect and forked another `gh api user/starred/...` (#18234). Landing itself + * only mounts when the user navigates, so the check runs once per visit. + */ +export function useLandingOrcaStarState(): [ + LandingStarState, + Dispatch> +] { + const [state, setState] = useState('loading') + + useEffect(() => { + let cancelled = false + void window.api.gh.checkOrcaStarred().then((result) => { + if (cancelled) { + return + } + if (result === null) { + setState('web-fallback') + } else { + setState(result ? 'starred' : 'not-starred') + } + }) + return () => { + cancelled = true + } + }, []) + + return [state, setState] +} From efe330549028f8fa5984a5a30311c399b3a88cb9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:31:02 -0700 Subject: [PATCH 094/398] test(editor): isolate prefix bleed in the batched model sweep, drop a dead export (#18240) Readiness-review follow-ups to #18144. The parity test in closed-editor-tab-disposal.test.ts cannot see prefix bleed: buildScenario closes tab-0..tab-99, so tab-10 is in the closed batch too and the per-tab oracle disposes its models via tab-10's own prefix. Batched and oracle agree and the assertion passes even with a bleeding predicate. Verified by mutation: replacing the boundary probe with a naive startsWith leaves all five of that file's tests green. Adds a test through the batched disposeClosedEditorTabs entry point with a still-OPEN tab-10 alongside a closed tab-1, which does fail under that mutation. Also records the `boundary + 1` advance in hasPaneScopeOwner as load-bearing for `:::` runs, with a test that fails under a `+ 2` "tidy-up", and removes disposeUnattachedMonacoModelsByPathPrefix, which #18144 left with zero production callers and a comment claiming it was kept for callers that do not exist. Finally, documents why title-derived rows carry `startedAt: 0`, which is the sole reason their `now` stamps cannot move a dashboard bucket. --- .../closed-editor-tab-cache-sweep.test.ts | 45 ++++++++++++++++++- .../editor/closed-editor-tab-cache-sweep.ts | 3 ++ .../editor/closed-editor-tab-disposal.test.ts | 24 ++++++++++ .../editor/diff-monaco-model-disposal.test.ts | 6 +-- .../editor/diff-monaco-model-disposal.ts | 7 --- .../worktree-title-derived-agent-rows.ts | 6 +++ 6 files changed, 80 insertions(+), 11 deletions(-) diff --git a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts index 726795a31dd..9d3b7ff042d 100644 --- a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts +++ b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.test.ts @@ -1,9 +1,52 @@ import { describe, expect, it } from 'vitest' import type { PdfViewPosition } from '@/lib/scroll-cache' -import { sweepClosedPdfViewPositions } from './closed-editor-tab-cache-sweep' +import { + deletePaneScopedCacheEntries, + sweepClosedPdfViewPositions +} from './closed-editor-tab-cache-sweep' const position = (pageNumber: number): PdfViewPosition => ({ pageNumber, top: 0, left: 0 }) +describe('deletePaneScopedCacheEntries', () => { + it('does not delete an owner whose id merely extends a closed owner id', () => { + const cache = new Map([ + ['tab-1::pane-1', 1], + ['tab-10::pane-1', 2], + ['tab-1x::pane-1', 3] + ]) + deletePaneScopedCacheEntries(cache, ['tab-1']) + expect([...cache.keys()]).toEqual(['tab-10::pane-1', 'tab-1x::pane-1']) + }) + + it('matches an owner that ends at the second `::` of a `:::` run', () => { + // Locks the `boundary + 1` advance in hasPaneScopeOwner: `a:` ends at index 2, which only the + // second `::` of the run exposes. A `+ 2` advance would skip it and leak the entry. + const cache = new Map([ + ['a:::b', 1], + ['a::b', 2], + ['ab:::c', 3] + ]) + deletePaneScopedCacheEntries(cache, ['a:']) + expect([...cache.keys()]).toEqual(['a::b', 'ab:::c']) + }) + + it('sweeps every owner in the batch in one pass', () => { + const cache = new Map([ + ['tab-1::pane-1', 1], + ['tab-2::pane-1', 2], + ['tab-3::pane-1', 3] + ]) + deletePaneScopedCacheEntries(cache, ['tab-1', 'tab-3']) + expect([...cache.keys()]).toEqual(['tab-2::pane-1']) + }) + + it('is a no-op for an empty owner batch', () => { + const cache = new Map([['tab-1::pane-1', 1]]) + deletePaneScopedCacheEntries(cache, []) + expect(cache.size).toBe(1) + }) +}) + describe('sweepClosedPdfViewPositions', () => { it('deletes the unscoped :pdf entry', () => { const cache = new Map([['/a.pdf:pdf', position(4)]]) diff --git a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts index b9769fb9dc8..a553625c838 100644 --- a/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts +++ b/src/renderer/src/components/editor/closed-editor-tab-cache-sweep.ts @@ -26,6 +26,9 @@ function hasPaneScopeOwner(key: string, owners: ReadonlySet): boolean { for ( let boundary = key.indexOf('::'); boundary !== -1; + // `+ 1`, not `+ 2`: in a `:::` run the second `::` starts one char after the first, and it can + // be the only boundary an owner ends at (owner `a:` against key `a:::b`). Skipping to `+ 2` + // steps over it and silently leaks that entry. boundary = key.indexOf('::', boundary + 1) ) { if (owners.has(key.slice(0, boundary))) { diff --git a/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts b/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts index 0223a4394b0..f23876fcf35 100644 --- a/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts +++ b/src/renderer/src/components/editor/closed-editor-tab-disposal.test.ts @@ -226,6 +226,30 @@ describe('disposeClosedEditorTabs', () => { expect(scrollTopCache.size).toBe(0) }) + // Why this is not covered by the parity test above: `buildScenario` closes tab-0..tab-99, so + // tab-10 is in the closed batch too. Prefix bleed from tab-1 would dispose tab-10's models, but + // the per-tab oracle disposes them as well via tab-10's own prefix, so the two agree and the + // assertion still passes. Isolating it needs a still-OPEN tab whose id extends a closed one. + it('does not dispose a still-open tab whose id extends a closed tab id', () => { + const closed = getDiffViewerMonacoModelPaths({ modelKey: 'tab-1', generationSuffix: '' }) + const stillOpen = getDiffViewerMonacoModelPaths({ modelKey: 'tab-10', generationSuffix: '' }) + const models = [ + createModel(closed.originalModelPath), + createModel(closed.modifiedModelPath), + createModel(stillOpen.originalModelPath), + createModel(stillOpen.modifiedModelPath) + ] + + // Batched entry point on purpose: the owned prefixes become a Set probed at the URI's own `:` + // boundaries, which is a different predicate from the pre-batch per-prefix `startsWith`. + disposeClosedEditorTabs(createRegistry(models), [diffTab('tab-1'), diffTab('tab-2')]) + + expect(models.filter((m) => m.disposed).map((m) => m.path)).toEqual([ + closed.originalModelPath, + closed.modifiedModelPath + ]) + }) + it('is a no-op when nothing closed', () => { const registry = createRegistry([createModel('diff:original:tab-1:tab-1')]) disposeClosedEditorTabs(registry, []) diff --git a/src/renderer/src/components/editor/diff-monaco-model-disposal.test.ts b/src/renderer/src/components/editor/diff-monaco-model-disposal.test.ts index dcefc61195d..75ccc46eff2 100644 --- a/src/renderer/src/components/editor/diff-monaco-model-disposal.test.ts +++ b/src/renderer/src/components/editor/diff-monaco-model-disposal.test.ts @@ -2,7 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { disposeUnattachedDiffViewerMonacoModels, disposeUnattachedMonacoModelPaths, - disposeUnattachedMonacoModelsByPathPrefix, + disposeUnattachedMonacoModelsByPathPrefixes, getDiffViewerMonacoModelPathPrefixes, getDiffViewerMonacoModelPaths } from './diff-monaco-model-disposal' @@ -139,7 +139,7 @@ describe('diff Monaco model disposal', () => { const monacoRegistry = createRegistry(models) const { originalModelPathPrefix } = getDiffViewerMonacoModelPathPrefixes('tab-1') - disposeUnattachedMonacoModelsByPathPrefix(monacoRegistry, originalModelPathPrefix) + disposeUnattachedMonacoModelsByPathPrefixes(monacoRegistry, [originalModelPathPrefix]) expect(baseDispose).toHaveBeenCalledOnce() expect(generatedDispose).toHaveBeenCalledOnce() @@ -177,7 +177,7 @@ describe('diff Monaco model disposal', () => { const monacoRegistry = createRegistry(models) const { originalModelPathPrefix } = getDiffViewerMonacoModelPathPrefixes('foo') - disposeUnattachedMonacoModelsByPathPrefix(monacoRegistry, originalModelPathPrefix) + disposeUnattachedMonacoModelsByPathPrefixes(monacoRegistry, [originalModelPathPrefix]) expect(ownedDispose).toHaveBeenCalledOnce() expect(siblingDispose).not.toHaveBeenCalled() diff --git a/src/renderer/src/components/editor/diff-monaco-model-disposal.ts b/src/renderer/src/components/editor/diff-monaco-model-disposal.ts index f833e53ec5b..2a1b5ceed5d 100644 --- a/src/renderer/src/components/editor/diff-monaco-model-disposal.ts +++ b/src/renderer/src/components/editor/diff-monaco-model-disposal.ts @@ -79,13 +79,6 @@ export function disposeUnattachedMonacoModelPaths( } } -export function disposeUnattachedMonacoModelsByPathPrefix( - monacoRegistry: MonacoModelRegistry, - modelPathPrefix: string -): void { - disposeUnattachedMonacoModelsByPathPrefixes(monacoRegistry, [modelPathPrefix]) -} - /** * Sweeps every owned prefix in a single scan of the global model registry. * diff --git a/src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts b/src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts index 25d87c50801..08c0d27b9e7 100644 --- a/src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts +++ b/src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts @@ -215,6 +215,12 @@ function buildTitleDerivedAgentRow(args: { agentType, rowSource: 'live', state: rowState, + // Load-bearing zero, not a placeholder: `dashboardRowBucketProjection` reads `startedAt === 0` + // as "title-derived" and short-circuits `unseen`. That is the ONLY reason the `args.now` stamps + // on `entry` above (updatedAt / stateStartedAt / observation) cannot move this row's bucket. + // Dashboard bucket caches key their invalidation on the freshness boundary in + // `isExplicitAgentStatusFresh` alone; give this a real timestamp and every one of them starts + // serving stale counts, with no test failing at the point of the change. startedAt: 0 } } From fda9cae5e27a0d8533a1ee76ee70436f59856a16 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:32:06 -0700 Subject: [PATCH 095/398] perf(renderer): build useRef seeds once instead of every render (#18159) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(renderer): stop re-running useRef initializers and ref-mirror effects every render React evaluates the argument you pass to `useRef` on every render and discards every result after the first. 30 renderer sites did real work in there — walking every browser page/tab across all worktrees, building activation-order maps, and minting `crypto.randomUUID()` per render on browser pages and the AI vault. Also moves 12 verbatim ref-mirror Effects to render-phase assignment, and routes the `tab.rename` shortcut straight to the focused tab instead of through a store field every mounted tab subscribed to. * perf(renderer): drop Fix 2 (render-phase ref mirrors) to satisfy no-ref-current-in-render * perf(renderer): convert the four lazy-useRef sites that landed on main --- .../src/app-shell/app-command-handlers.ts | 3 +- .../activity/use-agent-pane-threads.ts | 6 +- .../use-browser-page-webview-lifecycle.ts | 3 +- .../use-browser-page-zoom-feedback.ts | 4 +- .../use-remote-browser-page-input.ts | 3 +- ...-doc-preview-guest-tools.lazy-ref.test.tsx | 76 ++++++++++ .../use-doc-preview-guest-tools.ts | 3 +- .../dashboard/useAgentBucketCounts.ts | 3 +- .../dashboard/useLiveDashboardSnapshot.ts | 3 +- .../combined-diff-section-load-registry.ts | 9 +- .../use-combined-diff-view-restore.ts | 19 +-- .../use-emulator-pane-session.ts | 6 +- .../use-feature-wall-session-depth.ts | 5 +- .../use-feature-wall-tour-telemetry.ts | 3 +- .../FloatingTerminalPanel.shortcuts.test.tsx | 17 ++- .../floating-terminal-panel-test-fixtures.ts | 2 - .../floating-terminal-panel-test-harness.ts | 5 +- .../use-floating-terminal-panel-shortcuts.ts | 3 +- .../use-native-chat-live-session.ts | 3 +- .../use-native-chat-retained-session.ts | 3 +- .../right-sidebar/ai-vault-session-refresh.ts | 6 +- .../source-control/sync/git-history-panel.tsx | 3 +- .../useCreatePullRequestDialogFields.ts | 5 +- .../right-sidebar/useFileExplorerTree.ts | 3 +- .../settings/CommitMessageAiPane.tsx | 3 +- .../repository-source-control-ai-global-ux.ts | 35 +++-- .../use-settings-interaction-controller.ts | 3 +- .../use-workspace-kanban-column-resize.ts | 4 +- .../viewport/use-scroll-to-top.ts | 3 +- .../SortableTab.rename-shortcut.test.tsx | 42 ++++-- .../src/components/tab-bar/SortableTab.tsx | 29 ++-- .../SortableTab.update-depth-probe.test.tsx | 62 +++++++-- .../tab-bar/terminal-tab-rename-request.ts | 19 +++ .../use-terminal-tab-cold-parking.ts | 5 +- .../TerminalQuickCommandDialog.tsx | 5 +- ...erminal-window-lifecycle.lazy-ref.test.tsx | 86 ++++++++++++ .../use-terminal-window-lifecycle.ts | 11 +- .../composer-state/async-composer-state.ts | 31 +++-- .../composer-state/composer-drop-listener.ts | 3 +- src/renderer/src/hooks/use-audio-capture.ts | 3 +- src/renderer/src/lazy-use-ref-ratchet.test.ts | 130 ++++++++++++++++++ .../slices/tabs-label-and-pin-state.test.ts | 8 -- .../store/slices/tabs/create-tabs-slice.ts | 1 - .../store/slices/tabs/tabs-label-actions.ts | 5 - .../store/slices/tabs/tabs-slice-contract.ts | 3 - 45 files changed, 544 insertions(+), 143 deletions(-) create mode 100644 src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.lazy-ref.test.tsx create mode 100644 src/renderer/src/components/tab-bar/terminal-tab-rename-request.ts create mode 100644 src/renderer/src/components/use-terminal-window-lifecycle.lazy-ref.test.tsx create mode 100644 src/renderer/src/lazy-use-ref-ratchet.test.ts diff --git a/src/renderer/src/app-shell/app-command-handlers.ts b/src/renderer/src/app-shell/app-command-handlers.ts index 99915597e67..bb60f898130 100644 --- a/src/renderer/src/app-shell/app-command-handlers.ts +++ b/src/renderer/src/app-shell/app-command-handlers.ts @@ -5,6 +5,7 @@ import { requestScrollToCurrentWorkspaceRevealAndRename } from '@/lib/scroll-to- import { showTerminalShortcutCaptureNotification } from '@/lib/terminal-shortcut-capture-notification' import { shouldShowWorktreeHistoryControls } from '../lib/titlebar-worktree-history-controls' import { TOGGLE_WORKSPACE_BOARD_EVENT } from '../components/sidebar/useWorkspaceBoardPanel' +import { requestTerminalTabRename } from '../components/tab-bar/terminal-tab-rename-request' import { deleteHoveredWorkspaceImmediately, resolveHoveredWorkspaceDeleteTarget @@ -180,7 +181,7 @@ export function createAppCommandHandlers( ) { return false } - return claim('tab.rename', () => store.setRenamingTabId(store.activeTabId!)) + return claim('tab.rename', () => requestTerminalTabRename(store.activeTabId!)) } ], [ diff --git a/src/renderer/src/components/activity/use-agent-pane-threads.ts b/src/renderer/src/components/activity/use-agent-pane-threads.ts index 01eb0475a02..1934488439e 100644 --- a/src/renderer/src/components/activity/use-agent-pane-threads.ts +++ b/src/renderer/src/components/activity/use-agent-pane-threads.ts @@ -121,8 +121,10 @@ export function useAgentPaneThreads(args: { // identities across rebuilds, so a status write to one agent leaves every other row's // memo bail-out and cached search text intact. Rebuilds are deterministic, so a repeated // (StrictMode/deferred) memo invocation returns identical objects from the cache. - const eventBuildCacheRef = useRef(createActivityEventBuildCache()) - const threadReuseCacheRef = useRef(createAgentPaneThreadReuseCache()) + const eventBuildCacheRef = useRef>(undefined!) + eventBuildCacheRef.current ??= createActivityEventBuildCache() + const threadReuseCacheRef = useRef>(undefined!) + threadReuseCacheRef.current ??= createAgentPaneThreadReuseCache() const { events: allEvents, liveAgentByPaneKey } = useMemo( () => diff --git a/src/renderer/src/components/browser-pane/host-guest/use-browser-page-webview-lifecycle.ts b/src/renderer/src/components/browser-pane/host-guest/use-browser-page-webview-lifecycle.ts index a8de0e1e254..72859a415b5 100644 --- a/src/renderer/src/components/browser-pane/host-guest/use-browser-page-webview-lifecycle.ts +++ b/src/renderer/src/components/browser-pane/host-guest/use-browser-page-webview-lifecycle.ts @@ -132,7 +132,8 @@ export function useBrowserPageWebviewLifecycle({ const addBrowserHistoryEntryRef = useRef(addBrowserHistoryEntry) const createBrowserTab = useAppStore((s) => s.createBrowserTab) const isPaintableRef = useRef(isPaintable) - const annotationViewportBridgeTokenRef = useRef(createBrowserUuid().replaceAll('-', '')) + const annotationViewportBridgeTokenRef = useRef(undefined!) + annotationViewportBridgeTokenRef.current ??= createBrowserUuid().replaceAll('-', '') const isActiveRef = useRef(isActive) const pendingAnnotationPayloadRef = useRef(pendingAnnotationPayload) const browserAnnotations = useAppStore( diff --git a/src/renderer/src/components/browser-pane/host-guest/use-browser-page-zoom-feedback.ts b/src/renderer/src/components/browser-pane/host-guest/use-browser-page-zoom-feedback.ts index 368d66d5e8d..09474d21418 100644 --- a/src/renderer/src/components/browser-pane/host-guest/use-browser-page-zoom-feedback.ts +++ b/src/renderer/src/components/browser-pane/host-guest/use-browser-page-zoom-feedback.ts @@ -28,9 +28,9 @@ export function useBrowserPageZoomFeedback(browserTabId: string): { // tab's zoom through the shared setting. Why the module-level lookup: the guest webview outlives // this component (worktree switch, Settings visit), so re-seeding on remount would let a later // Settings change retroactively hijack a tab the user already zoomed. - const paneZoomLevelRef = useRef( + const paneZoomLevelRef = useRef(undefined!) + paneZoomLevelRef.current ??= getExplicitBrowserPageZoomLevel(browserTabId) ?? normalizedBrowserDefaultZoomLevel - ) const [browserZoomPercent, setBrowserZoomPercent] = useState(browserDefaultZoomPercent) const [browserZoomFeedbackVisible, setBrowserZoomFeedbackVisible] = useState(false) const browserZoomFeedbackTimerRef = useRef>(undefined) diff --git a/src/renderer/src/components/browser-pane/stream-remote/use-remote-browser-page-input.ts b/src/renderer/src/components/browser-pane/stream-remote/use-remote-browser-page-input.ts index 64002a1208a..c58a5c56a1a 100644 --- a/src/renderer/src/components/browser-pane/stream-remote/use-remote-browser-page-input.ts +++ b/src/renderer/src/components/browser-pane/stream-remote/use-remote-browser-page-input.ts @@ -28,7 +28,8 @@ export function useRemoteBrowserPageInputQueue(): { remoteWheelFrameRef: React.MutableRefObject remoteWheelInFlightRef: React.MutableRefObject } { - const remoteInputQueueRef = useRef>(Promise.resolve()) + const remoteInputQueueRef = useRef>(undefined!) + remoteInputQueueRef.current ??= Promise.resolve() const pendingRemoteWheelRef = useRef(null) const remoteWheelFrameRef = useRef(null) const remoteWheelInFlightRef = useRef(false) diff --git a/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.lazy-ref.test.tsx b/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.lazy-ref.test.tsx new file mode 100644 index 00000000000..d68cd200c21 --- /dev/null +++ b/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.lazy-ref.test.tsx @@ -0,0 +1,76 @@ +// @vitest-environment happy-dom + +/** + * The annotation viewport bridge token used to sit in a `useRef(...)` argument, so every render of + * a doc preview minted a fresh `crypto.randomUUID()` and threw it away — only the mount-time token + * was ever read. Pin the mint count to the mount count. + */ +import { useState } from 'react' +import { act, cleanup, render } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +const browserUuidCalls = vi.hoisted(() => ({ count: 0 })) + +vi.mock('@/lib/browser-uuid', () => ({ + createBrowserUuid: () => { + browserUuidCalls.count += 1 + return `00000000-0000-4000-8000-${String(browserUuidCalls.count).padStart(12, '0')}` + } +})) + +vi.mock('@/hooks/useShortcutLabel', () => ({ useShortcutLabel: () => 'Cmd+G' })) +vi.mock('@/components/browser-pane/annotate/guest-annotation-viewport-bridge', () => ({ + syncGuestAnnotationViewportBridge: vi.fn() +})) +vi.mock('@/components/browser-pane/annotate/use-browser-page-annotation-send', () => ({ + useBrowserPageAnnotationSend: () => ({ + browserAnnotations: [], + setBrowserAnnotationTrayOpen: vi.fn() + }) +})) +vi.mock('@/components/browser-pane/annotate/use-browser-page-grab-annotations', () => ({ + useBrowserPageGrabAnnotations: () => ({}) +})) +vi.mock('@/components/browser-pane/annotate/use-browser-page-markup-capture', () => ({ + useBrowserPageMarkupCapture: () => ({}) +})) +vi.mock('@/components/browser-pane/annotate/useGrabMode', () => ({ + useGrabMode: () => ({ active: false }) +})) + +const { useDocPreviewGuestTools } = await import('./use-doc-preview-guest-tools') + +let bumpRender: (() => void) | null = null + +function Host(): null { + const [, setTick] = useState(0) + bumpRender = () => setTick((tick) => tick + 1) + useDocPreviewGuestTools({ + previewId: 'preview-1', + worktreeId: 'wt-1', + grantId: 'grant-1', + webviewRef: { current: null }, + containerRef: { current: null }, + toolsReady: true + } as unknown as Parameters[0]) + return null +} + +afterEach(() => { + cleanup() + browserUuidCalls.count = 0 + bumpRender = null +}) + +describe('useDocPreviewGuestTools annotation bridge token', () => { + it('mints the bridge token once per mount, not once per render', () => { + render() + expect(browserUuidCalls.count).toBe(1) + + for (let i = 0; i < 20; i += 1) { + act(() => bumpRender?.()) + } + + expect(browserUuidCalls.count).toBe(1) + }) +}) diff --git a/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.ts b/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.ts index 9eec9b8affe..83706aa3fcd 100644 --- a/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.ts +++ b/src/renderer/src/components/browser-pane/workspace-doc/use-doc-preview-guest-tools.ts @@ -41,7 +41,8 @@ export function useDocPreviewGuestTools({ // Why still empty before the first grant: the page is only a tool target once a document is on // screen, and useGrabMode needs a stable identity every render rather than one to guess with. const toolTargetId = grantId === null ? '' : previewId - const annotationViewportBridgeTokenRef = useRef(createBrowserUuid().replaceAll('-', '')) + const annotationViewportBridgeTokenRef = useRef(undefined!) + annotationViewportBridgeTokenRef.current ??= createBrowserUuid().replaceAll('-', '') const [browserOverlayViewport, setBrowserOverlayViewport] = useState({ scrollX: 0, scrollY: 0, diff --git a/src/renderer/src/components/dashboard/useAgentBucketCounts.ts b/src/renderer/src/components/dashboard/useAgentBucketCounts.ts index 60df4995e56..0e8c2881244 100644 --- a/src/renderer/src/components/dashboard/useAgentBucketCounts.ts +++ b/src/renderer/src/components/dashboard/useAgentBucketCounts.ts @@ -51,7 +51,8 @@ export function useAgentBucketCounts(): AgentBucketCounts { // Why a per-hook cache: unrelated status/title writes change one worktree's inputs; // the cache keeps every other worktree's counts without rerunning its row pipeline. - const cacheRef = useRef(createDashboardBucketCountsCache()) + const cacheRef = useRef>(undefined!) + cacheRef.current ??= createDashboardBucketCountsCache() return useMemo(() => { return buildDashboardBucketCounts( { diff --git a/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts b/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts index 227cf4e3f8b..52270167ef9 100644 --- a/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts +++ b/src/renderer/src/components/dashboard/useLiveDashboardSnapshot.ts @@ -11,7 +11,8 @@ import { createWorktreeAgentRowsCache } from './worktree-agent-rows-cache' * is no relay, so we derive it here from the same builder the bridge uses. */ export function useLiveDashboardSnapshot(): DashboardSnapshot { - const rowsCacheRef = useRef(createWorktreeAgentRowsCache()) + const rowsCacheRef = useRef>(undefined!) + rowsCacheRef.current ??= createWorktreeAgentRowsCache() const repos = useAppStore((s) => s.repos) const worktreesByRepo = useAppStore((s) => s.worktreesByRepo) const tabsByWorktree = useAppStore((s) => s.tabsByWorktree) diff --git a/src/renderer/src/components/editor/combined-diff/load-sections/combined-diff-section-load-registry.ts b/src/renderer/src/components/editor/combined-diff/load-sections/combined-diff-section-load-registry.ts index c74d03d7692..370dbd6ebf0 100644 --- a/src/renderer/src/components/editor/combined-diff/load-sections/combined-diff-section-load-registry.ts +++ b/src/renderer/src/components/editor/combined-diff/load-sections/combined-diff-section-load-registry.ts @@ -47,11 +47,10 @@ export function useCombinedDiffSectionLoadRegistry( const loadSectionRef = useRef<(index: number) => Promise>(async () => {}) const retrySectionRef = useRef<(index: number) => void>(() => {}) const requestSectionReloadRef = useRef<(index: number) => void>(() => {}) - const loadSchedulerRef = useRef( - createCombinedDiffLoadScheduler({ - loadSection: (index) => loadSectionRef.current(index) - }) - ) + const loadSchedulerRef = useRef>(undefined!) + loadSchedulerRef.current ??= createCombinedDiffLoadScheduler({ + loadSection: (index) => loadSectionRef.current(index) + }) sectionsRef.current = sections useEffect(() => { diff --git a/src/renderer/src/components/editor/combined-diff/remember-view/use-combined-diff-view-restore.ts b/src/renderer/src/components/editor/combined-diff/remember-view/use-combined-diff-view-restore.ts index 9f657e3abd0..7fcc833df91 100644 --- a/src/renderer/src/components/editor/combined-diff/remember-view/use-combined-diff-view-restore.ts +++ b/src/renderer/src/components/editor/combined-diff/remember-view/use-combined-diff-view-restore.ts @@ -1,4 +1,4 @@ -import { useCallback, useLayoutEffect, useRef } from 'react' +import { useCallback, useLayoutEffect, useRef, useState } from 'react' import type React from 'react' import type { VirtualizedScrollAnchor } from '@/hooks/useVirtualizedScrollAnchor' import type { GitStatusEntry } from '../../../../../../shared/git-status-types' @@ -65,13 +65,16 @@ export function useCombinedDiffViewRestore({ sectionLoadTokensRef } = registry - const scrollOffsetRef = useRef(combinedDiffScrollTopCache.get(viewStateKey) ?? 0) - const scrollAnchorRef = useRef( - combinedDiffScrollAnchorCache.get(viewStateKey) ?? null - ) - const latestDomScrollAnchorRef = useRef( - combinedDiffScrollAnchorCache.get(viewStateKey) ?? null - ) + // Why useState and not `useRef(expr)`: the latter re-reads all three caches on every render and + // throws the result away, and an anchor seeds legitimately to null so a nullish guard would keep + // re-reading. useState's initializer runs once without writing a ref during render. + const [restoreSeed] = useState<{ offset: number; anchor: VirtualizedScrollAnchor }>(() => ({ + offset: combinedDiffScrollTopCache.get(viewStateKey) ?? 0, + anchor: combinedDiffScrollAnchorCache.get(viewStateKey) ?? null + })) + const scrollOffsetRef = useRef(restoreSeed.offset) + const scrollAnchorRef = useRef(restoreSeed.anchor) + const latestDomScrollAnchorRef = useRef(restoreSeed.anchor) // Why: tab/worktree switches unmount this viewer; cache by pane key so remount restores sections+scroll before repaint. const initializedEntryStateRef = useRef<{ diff --git a/src/renderer/src/components/emulator-pane/use-emulator-pane-session.ts b/src/renderer/src/components/emulator-pane/use-emulator-pane-session.ts index 2ac04a9ca99..c696374526e 100644 --- a/src/renderer/src/components/emulator-pane/use-emulator-pane-session.ts +++ b/src/renderer/src/components/emulator-pane/use-emulator-pane-session.ts @@ -39,11 +39,13 @@ export function useEmulatorPaneSession({ const configuredDefaultUdid = useAppStore( (state) => state.settings?.mobileEmulatorDefaultDeviceUdid ?? null ) - const prelaunchedSessionRef = useRef( + // Why the lazy initializer: the consume deletes the handoff entry, and a `useRef(expr)` argument + // re-runs every render — so a prelaunch registered after mount was consumed and then discarded. + const [prelaunchedSession] = useState(() => consumePrelaunchedSimulatorSession(worktreeId) ) const prelaunchedState = buildPrelaunchedEmulatorSessionState( - prelaunchedSessionRef.current, + prelaunchedSession, configuredDefaultUdid ) const [selectedUdid, setSelectedUdid] = useState(prelaunchedState.selectedUdid) diff --git a/src/renderer/src/components/feature-wall/use-feature-wall-session-depth.ts b/src/renderer/src/components/feature-wall/use-feature-wall-session-depth.ts index c4d57043410..9ac9fcaa48f 100644 --- a/src/renderer/src/components/feature-wall/use-feature-wall-session-depth.ts +++ b/src/renderer/src/components/feature-wall/use-feature-wall-session-depth.ts @@ -47,13 +47,14 @@ export function useFeatureWallSessionDepth( visitedWorkbenchSteps: Set visitedReviewSteps: Set lastGroupId: FeatureWallWorkflowId | null - }>({ + }>(undefined!) + sessionDepthRef.current ??= { visitedWorkflows: new Set(), visitedAgentSteps: new Set(), visitedWorkbenchSteps: new Set(), visitedReviewSteps: new Set(), lastGroupId: null - }) + } const getTourDepthSummary = useCallback((): FeatureWallTourDepthSummary => { const session = sessionDepthRef.current diff --git a/src/renderer/src/components/feature-wall/use-feature-wall-tour-telemetry.ts b/src/renderer/src/components/feature-wall/use-feature-wall-tour-telemetry.ts index 7f98c90520a..23134e2a4a6 100644 --- a/src/renderer/src/components/feature-wall/use-feature-wall-tour-telemetry.ts +++ b/src/renderer/src/components/feature-wall/use-feature-wall-tour-telemetry.ts @@ -77,7 +77,8 @@ export function useFeatureWallTourTelemetry(args: { getDepthSummary: () => FeatureWallTourDepthSummary }): { markExitAction: (exitAction: FeatureWallExitAction) => void } { const { isOpen, source, getDepthSummary } = args - const telemetryRef = useRef(createFeatureWallTourTelemetryState()) + const telemetryRef = useRef(undefined!) + telemetryRef.current ??= createFeatureWallTourTelemetryState() const sourceRef = useRef(source) const getDepthSummaryRef = useRef(getDepthSummary) // Why: close telemetry may emit from stable callbacks; keep the payload diff --git a/src/renderer/src/components/floating-terminal/FloatingTerminalPanel.shortcuts.test.tsx b/src/renderer/src/components/floating-terminal/FloatingTerminalPanel.shortcuts.test.tsx index 3593643b50c..3e4de4227e7 100644 --- a/src/renderer/src/components/floating-terminal/FloatingTerminalPanel.shortcuts.test.tsx +++ b/src/renderer/src/components/floating-terminal/FloatingTerminalPanel.shortcuts.test.tsx @@ -11,6 +11,10 @@ import { type FloatingPanelStoreState } from './floating-terminal-panel-test-fixtures' import { mocks, setupFloatingTerminalPanelTest } from './floating-terminal-panel-test-harness' +import { + RENAME_TERMINAL_TAB_EVENT, + type RenameTerminalTabDetail +} from '@/components/tab-bar/terminal-tab-rename-request' import { attachRef, bindFocusedFloatingPanelKeydown, @@ -159,6 +163,15 @@ vi.mock('@/components/ShortcutKeyCombo', async () => { return (await import('./floating-terminal-panel-component-stubs')).createShortcutKeyComboModule() }) +/** Tab ids the panel asked to rename, in dispatch order. */ +function dispatchedRenameTabIds(): string[] { + return vi + .mocked(window.dispatchEvent) + .mock.calls.map(([event]) => event as CustomEvent) + .filter((event) => event.type === RENAME_TERMINAL_TAB_EVENT) + .map((event) => event.detail.tabId) +} + describe('FloatingTerminalPanel close behavior', () => { beforeEach(setupFloatingTerminalPanelTest) @@ -434,7 +447,7 @@ describe('FloatingTerminalPanel close behavior', () => { expect(preventDefault).toHaveBeenCalledWith() expect(stopPropagation).toHaveBeenCalledWith() expect(stopImmediatePropagation).toHaveBeenCalledWith() - expect(mocks.setRenamingTabId).toHaveBeenCalledWith('tab-1') + expect(dispatchedRenameTabIds()).toEqual(['tab-1']) expect(mocks.setTabCustomTitle).not.toHaveBeenCalled() }) @@ -607,7 +620,7 @@ describe('FloatingTerminalPanel close behavior', () => { }) ) - expect(mocks.setRenamingTabId).not.toHaveBeenCalled() + expect(dispatchedRenameTabIds()).toEqual([]) }) it('leaves focused floating xterm tab index shortcuts to terminal-first terminals', async () => { diff --git a/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-fixtures.ts b/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-fixtures.ts index ac6fc0c1cf4..b51cb9a6d78 100644 --- a/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-fixtures.ts +++ b/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-fixtures.ts @@ -15,7 +15,6 @@ export type FloatingPanelStoreState = { activeGroupIdByWorktree: Record activeTabIdByWorktree: Record expandedPaneByTabId: Record - renamingTabId: string | null createTab: ( worktreeId: string, groupId?: string, @@ -42,7 +41,6 @@ export type FloatingPanelStoreState = { activateTab: (tabId: string) => void setActiveTab: (tabId: string) => void setTabCustomTitle: (tabId: string, title: string | null) => void - setRenamingTabId: (tabId: string | null) => void setTabColor: (tabId: string, color: string | null) => void setTabPaneExpanded: (tabId: string, expanded: boolean) => void makePreviewFilePermanent: (fileId: string, tabId?: string) => void diff --git a/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-harness.ts b/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-harness.ts index 392a7bd224c..183508fa9aa 100644 --- a/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-harness.ts +++ b/src/renderer/src/components/floating-terminal/floating-terminal-panel-test-harness.ts @@ -59,7 +59,6 @@ export type FloatingTerminalPanelMocks = { pinFile: Mock setFloatingFocus: Mock<(state: { panelFocused: boolean; terminalFocused: boolean }) => void> setActiveTab: Mock - setRenamingTabId: Mock setTabColor: Mock setTabCustomTitle: Mock setTabPaneExpanded: Mock @@ -106,7 +105,6 @@ export const mocks: FloatingTerminalPanelMocks = { pinFile: vi.fn(), setFloatingFocus: vi.fn(), setActiveTab: vi.fn(), - setRenamingTabId: vi.fn(), setTabColor: vi.fn(), setTabCustomTitle: vi.fn(), setTabPaneExpanded: vi.fn(), @@ -133,7 +131,6 @@ function resetStore(tabs: TerminalTab[] = []): void { activeGroupIdByWorktree: {}, activeTabIdByWorktree: { [FLOATING_TERMINAL_WORKTREE_ID]: tabs[0]?.id ?? null }, expandedPaneByTabId: {}, - renamingTabId: null, activateTab: mocks.activateTab, closeBrowserTab: mocks.closeBrowserTab, closeFile: mocks.closeFile, @@ -147,7 +144,6 @@ function resetStore(tabs: TerminalTab[] = []): void { pinFile: mocks.pinFile, setActiveTab: mocks.setActiveTab, setTabCustomTitle: mocks.setTabCustomTitle, - setRenamingTabId: mocks.setRenamingTabId, setTabColor: mocks.setTabColor, setTabPaneExpanded: mocks.setTabPaneExpanded, browserDefaultUrl: 'about:blank', @@ -196,6 +192,7 @@ export async function setupFloatingTerminalPanelTest(): Promise { } vi.stubGlobal('window', { addEventListener: vi.fn(), + dispatchEvent: vi.fn(), api: { app: { getFloatingMarkdownDirectory: mocks.getFloatingMarkdownDirectory, diff --git a/src/renderer/src/components/floating-terminal/use-floating-terminal-panel-shortcuts.ts b/src/renderer/src/components/floating-terminal/use-floating-terminal-panel-shortcuts.ts index b597810c0fd..ee1979451cf 100644 --- a/src/renderer/src/components/floating-terminal/use-floating-terminal-panel-shortcuts.ts +++ b/src/renderer/src/components/floating-terminal/use-floating-terminal-panel-shortcuts.ts @@ -7,6 +7,7 @@ import { } from '@/lib/floating-workspace-shortcut-policy' import { isFloatingWorkspaceTerminalInputTarget } from '@/lib/floating-workspace-terminal-actions' import { getShortcutPlatform } from '@/lib/shortcut-platform' +import { requestTerminalTabRename } from '@/components/tab-bar/terminal-tab-rename-request' import { useAppStore } from '@/store' import { FLOATING_TERMINAL_WORKTREE_ID } from '../../../../shared/constants' import type { KeybindingContext, KeybindingMatchOptions } from '../../../../shared/keybindings' @@ -168,7 +169,7 @@ export function useFloatingTerminalPanelShortcuts({ return 'unmatched' } consume() - useAppStore.getState().setRenamingTabId(activeTab.id) + requestTerminalTabRename(activeTab.id) return 'handled' } consume() diff --git a/src/renderer/src/components/native-chat/use-native-chat-live-session.ts b/src/renderer/src/components/native-chat/use-native-chat-live-session.ts index 59c4b80e389..9335965bdb8 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-live-session.ts +++ b/src/renderer/src/components/native-chat/use-native-chat-live-session.ts @@ -117,7 +117,8 @@ export function useNativeChatLiveSession( // Appended messages accumulate separately from the snapshot so pagination doesn't lose in-flight appends; merged by id and capped to the read window (#6). const [appended, setAppended] = useState([]) // Id-dedup merger backing `appended`; caches the id→index map so each live frame costs O(incoming), not O(existing) (#18). - const appendMergerRef = useRef(createNativeChatMerger(NATIVE_CHAT_SOURCE_PRIORITY)) + const appendMergerRef = useRef>(undefined!) + appendMergerRef.current ??= createNativeChatMerger(NATIVE_CHAT_SOURCE_PRIORITY) const [hookState, hookStateStartedAt, hookHasWorkingSubagents] = useNativeChatHookStatus(paneKey) diff --git a/src/renderer/src/components/native-chat/use-native-chat-retained-session.ts b/src/renderer/src/components/native-chat/use-native-chat-retained-session.ts index 6d176f2ac55..eda95b074cc 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-retained-session.ts +++ b/src/renderer/src/components/native-chat/use-native-chat-retained-session.ts @@ -22,7 +22,8 @@ export function useNativeChatRetainedSession( args.transcriptPath ?? null ]) const activeIdentityRef = useRef(identity) - const retentionRef = useRef(createNativeChatTranscriptRetention()) + const retentionRef = useRef>(undefined!) + retentionRef.current ??= createNativeChatTranscriptRetention() const sessionMatchesIdentity = activeIdentityRef.current === identity const readPhase = sessionMatchesIdentity ? session.readPhase : 'loading' diff --git a/src/renderer/src/components/right-sidebar/ai-vault-session-refresh.ts b/src/renderer/src/components/right-sidebar/ai-vault-session-refresh.ts index b1f72178b24..d8655a560d3 100644 --- a/src/renderer/src/components/right-sidebar/ai-vault-session-refresh.ts +++ b/src/renderer/src/components/right-sidebar/ai-vault-session-refresh.ts @@ -61,7 +61,8 @@ export function useAiVaultSessionRefresh( const sessions = scanResult?.sessions ?? EMPTY_AI_VAULT_SESSIONS const [loading, setLoading] = useState(false) const [error, setError] = useState(null) - const requestTokenRef = useRef(crypto.randomUUID()) + const requestTokenRef = useRef(undefined!) + requestTokenRef.current ??= crypto.randomUUID() const refreshIdRef = useRef(0) const refreshInFlightRef = useRef(false) const pendingRefreshRef = useRef(false) @@ -69,7 +70,8 @@ export function useAiVaultSessionRefresh( const pendingBackgroundRef = useRef(true) const lastAppliedScanRef = useRef<{ scopeKey: string; scannedAt: string } | null>(null) const mountedRef = useRef(true) - const publicationGateRef = useRef(new AiVaultSessionPublicationGate()) + const publicationGateRef = useRef(undefined!) + publicationGateRef.current ??= new AiVaultSessionPublicationGate() const scanScopeKey = `${aiVaultSessionResultCacheKey(executionHostScope, scopePaths)}\n${sessionLimit}` const scopePathsRef = useRef(scopePaths) scopePathsRef.current = scopePaths diff --git a/src/renderer/src/components/right-sidebar/source-control/sync/git-history-panel.tsx b/src/renderer/src/components/right-sidebar/source-control/sync/git-history-panel.tsx index 35751a428d1..71de9872649 100644 --- a/src/renderer/src/components/right-sidebar/source-control/sync/git-history-panel.tsx +++ b/src/renderer/src/components/right-sidebar/source-control/sync/git-history-panel.tsx @@ -92,7 +92,8 @@ export function GitHistoryPanel({ const loadedCommitsRef = useRef<{ result: GitHistoryResult | undefined ids: Set - }>({ result, ids: new Set() }) + }>(undefined!) + loadedCommitsRef.current ??= { result, ids: new Set() } // A new history result can reorder or replace commits, so drop any expansion // and cached file lists rather than risk showing stale files under a row. diff --git a/src/renderer/src/components/right-sidebar/useCreatePullRequestDialogFields.ts b/src/renderer/src/components/right-sidebar/useCreatePullRequestDialogFields.ts index 18e1c4c999d..fbb0508ee24 100644 --- a/src/renderer/src/components/right-sidebar/useCreatePullRequestDialogFields.ts +++ b/src/renderer/src/components/right-sidebar/useCreatePullRequestDialogFields.ts @@ -58,9 +58,8 @@ export function useCreatePullRequestDialogFields({ const generationRequestIdRef = useRef(0) const generationSeedRef = useRef(null) const restoredExternalGenerationSeedRef = useRef(null) - const fieldRevisionsRef = useRef( - createInitialPullRequestFieldRevisions() - ) + const fieldRevisionsRef = useRef(undefined!) + fieldRevisionsRef.current ??= createInitialPullRequestFieldRevisions() const [base, setBase] = useState('') const [title, setTitle] = useState('') const [body, setBody] = useState('') diff --git a/src/renderer/src/components/right-sidebar/useFileExplorerTree.ts b/src/renderer/src/components/right-sidebar/useFileExplorerTree.ts index d6f695c2ac3..4c8311f8dfa 100644 --- a/src/renderer/src/components/right-sidebar/useFileExplorerTree.ts +++ b/src/renderer/src/components/right-sidebar/useFileExplorerTree.ts @@ -45,7 +45,8 @@ export function useFileExplorerTree( const [rootError, setRootError] = useState(null) const dirCacheRef = useRef(dirCache) dirCacheRef.current = dirCache - const dirLoadTrackerRef = useRef(createFileExplorerDirLoadTracker()) + const dirLoadTrackerRef = useRef>(undefined!) + dirLoadTrackerRef.current ??= createFileExplorerDirLoadTracker() // Why: a ref, not state — the expansion effect must read the mark set by a refresh that landed // after the effect's render, and a state write would only be visible one render too late. const staleDirsRef = useRef(new Set()) diff --git a/src/renderer/src/components/settings/CommitMessageAiPane.tsx b/src/renderer/src/components/settings/CommitMessageAiPane.tsx index d14ff189c72..096ccd34d51 100644 --- a/src/renderer/src/components/settings/CommitMessageAiPane.tsx +++ b/src/renderer/src/components/settings/CommitMessageAiPane.tsx @@ -104,7 +104,8 @@ export function CommitMessageAiPane({ const searchQuery = settingsSearchQuery ?? storeSearchQuery const config = readSettings(settings) const ownership = getSettingOwnershipSummary('sourceControlAiDefaults') - const settingsWriteQueueRef = useRef>(Promise.resolve()) + const settingsWriteQueueRef = useRef>(undefined!) + settingsWriteQueueRef.current ??= Promise.resolve() const localWriteConfig = (patch: SourceControlAiSettingsPatch): Promise => { const next = settingsWriteQueueRef.current diff --git a/src/renderer/src/components/settings/repository-source-control-ai-global-ux.ts b/src/renderer/src/components/settings/repository-source-control-ai-global-ux.ts index a27db245e6d..3a13f6186ca 100644 --- a/src/renderer/src/components/settings/repository-source-control-ai-global-ux.ts +++ b/src/renderer/src/components/settings/repository-source-control-ai-global-ux.ts @@ -83,25 +83,24 @@ export function useRepositorySourceControlAiGlobalUx({ const lastSyncedRepoIdRef = useRef(repoId) const pendingWritesRef = useRef(0) - const queueRef = useRef( - createRepoAiPersistQueue({ - getRepoId: () => repoIdRef.current, - getPersisted: () => persistedRef.current, - setPersisted: (value) => { - persistedRef.current = value - if (mountedRef.current) { - setBaselineRepoAiRef.current(value) - } - }, - updateRepo: (id, updates) => updateRepoRef.current(id, updates), - isMounted: () => mountedRef.current, - onError: (message) => { - if (mountedRef.current) { - setSaveError(message) - } + const queueRef = useRef>(undefined!) + queueRef.current ??= createRepoAiPersistQueue({ + getRepoId: () => repoIdRef.current, + getPersisted: () => persistedRef.current, + setPersisted: (value) => { + persistedRef.current = value + if (mountedRef.current) { + setBaselineRepoAiRef.current(value) } - }) - ) + }, + updateRepo: (id, updates) => updateRepoRef.current(id, updates), + isMounted: () => mountedRef.current, + onError: (message) => { + if (mountedRef.current) { + setSaveError(message) + } + } + }) useEffect(() => { const repoChanged = lastSyncedRepoIdRef.current !== repoId diff --git a/src/renderer/src/components/settings/use-settings-interaction-controller.ts b/src/renderer/src/components/settings/use-settings-interaction-controller.ts index bfd24c7e5d8..6d949004749 100644 --- a/src/renderer/src/components/settings/use-settings-interaction-controller.ts +++ b/src/renderer/src/components/settings/use-settings-interaction-controller.ts @@ -39,7 +39,8 @@ export function useSettingsInteractionController(model: SettingsStoreModel) { const pendingScrollTargetWatchRef = useRef(null) const repoHooksRequestSeqRef = useRef(0) const shortcutsEscapeConfirmUntilRef = useRef(0) - const sourceControlAiWriteQueueRef = useRef>(Promise.resolve()) + const sourceControlAiWriteQueueRef = useRef>(undefined!) + sourceControlAiWriteQueueRef.current ??= Promise.resolve() const hasUnsavedSourceControlAiPromptChanges = hasUnsavedCommitPromptChanges || hasUnsavedBranchPromptChanges diff --git a/src/renderer/src/components/sidebar/use-workspace-kanban-column-resize.ts b/src/renderer/src/components/sidebar/use-workspace-kanban-column-resize.ts index a48e940a3db..17547d5cd56 100644 --- a/src/renderer/src/components/sidebar/use-workspace-kanban-column-resize.ts +++ b/src/renderer/src/components/sidebar/use-workspace-kanban-column-resize.ts @@ -20,7 +20,8 @@ export function useWorkspaceKanbanColumnResize( clampWorkspaceBoardColumnWidth(committedWidth) ) const [isResizingColumn, setIsResizingColumn] = useState(false) - const committedWidthRef = useRef(clampWorkspaceBoardColumnWidth(committedWidth)) + const nextCommittedWidth = clampWorkspaceBoardColumnWidth(committedWidth) + const committedWidthRef = useRef(nextCommittedWidth) const commitWidthRef = useRef(onCommitWidth) const resizingRef = useRef(false) const startXRef = useRef(0) @@ -29,7 +30,6 @@ export function useWorkspaceKanbanColumnResize( const frameRef = useRef(null) commitWidthRef.current = onCommitWidth - const nextCommittedWidth = clampWorkspaceBoardColumnWidth(committedWidth) if (committedWidthRef.current !== nextCommittedWidth) { committedWidthRef.current = nextCommittedWidth if (!resizingRef.current) { diff --git a/src/renderer/src/components/sidebar/worktree-list/viewport/use-scroll-to-top.ts b/src/renderer/src/components/sidebar/worktree-list/viewport/use-scroll-to-top.ts index d732aa71cec..8c88d812746 100644 --- a/src/renderer/src/components/sidebar/worktree-list/viewport/use-scroll-to-top.ts +++ b/src/renderer/src/components/sidebar/worktree-list/viewport/use-scroll-to-top.ts @@ -42,7 +42,8 @@ export function useWorktreeListScrollToTop({ showScrollToTop: boolean scrollToTop: () => void } { - const detectorRef = useRef(createHardScrollUpDetectorState()) + const detectorRef = useRef(undefined!) + detectorRef.current ??= createHardScrollUpDetectorState() const [showScrollToTop, setShowScrollToTop] = useState(false) const showScrollToTopRef = useRef(false) const idleTimerRef = useRef(null) diff --git a/src/renderer/src/components/tab-bar/SortableTab.rename-shortcut.test.tsx b/src/renderer/src/components/tab-bar/SortableTab.rename-shortcut.test.tsx index fd275935f77..2e04247dd9b 100644 --- a/src/renderer/src/components/tab-bar/SortableTab.rename-shortcut.test.tsx +++ b/src/renderer/src/components/tab-bar/SortableTab.rename-shortcut.test.tsx @@ -1,4 +1,7 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' +import { requestTerminalTabRename } from './terminal-tab-rename-request' + +const windowListeners = new Map void>>() const reactHookRuntime = vi.hoisted(() => ({ states: [] as unknown[], @@ -11,10 +14,8 @@ const storeState = vi.hoisted( clearTabLaunchAgent: ReturnType ptyIdsByTabId: Record retainedAgentsByPaneKey: Record - renamingTabId: string | null keybindings: Record repos: unknown[] - setRenamingTabId: ReturnType terminalLayoutsByTabId: Record worktreesByRepo: Record unreadTerminalTabs: Record @@ -23,12 +24,8 @@ const storeState = vi.hoisted( clearTabLaunchAgent: vi.fn(), ptyIdsByTabId: {} as Record, retainedAgentsByPaneKey: {}, - renamingTabId: null as string | null, keybindings: {}, repos: [], - setRenamingTabId: vi.fn((tabId: string | null) => { - storeState.renamingTabId = tabId - }), terminalLayoutsByTabId: {}, worktreesByRepo: {}, unreadTerminalTabs: {} as Record @@ -340,13 +337,24 @@ describe('SortableTab rename shortcut signal', () => { beforeEach(() => { reactHookRuntime.states = [] reactHookRuntime.index = 0 - storeState.renamingTabId = 'terminal-tab-1' storeState.unreadTerminalTabs = {} storeState.clearTabLaunchAgent.mockClear() - storeState.setRenamingTabId.mockClear() + windowListeners.clear() vi.stubGlobal('window', { - addEventListener: vi.fn(), - removeEventListener: vi.fn() + addEventListener: vi.fn((type: string, listener: (event: Event) => void) => { + const listeners = windowListeners.get(type) ?? new Set<(event: Event) => void>() + listeners.add(listener) + windowListeners.set(type, listeners) + }), + removeEventListener: vi.fn((type: string, listener: (event: Event) => void) => { + windowListeners.get(type)?.delete(listener) + }), + dispatchEvent: vi.fn((event: Event) => { + for (const listener of windowListeners.get(event.type) ?? []) { + listener(event) + } + return true + }) }) vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) => { callback(0) @@ -355,22 +363,30 @@ describe('SortableTab rename shortcut signal', () => { vi.stubGlobal('cancelAnimationFrame', vi.fn()) }) - it('opens the inline rename input and consumes the matching store signal', async () => { + it('opens the inline rename input for a request that targets this tab', async () => { await renderSortableTab() + requestTerminalTabRename('terminal-tab-1') const rerender = expandNode(await renderSortableTab()) const inputs = findElementsByType(rerender, 'input') - expect(storeState.setRenamingTabId).toHaveBeenCalledWith(null) - expect(storeState.renamingTabId).toBeNull() expect(inputs).toHaveLength(1) expect(inputs[0].props.value).toBe('Runtime terminal title') expect(inputs[0].props['data-tab-rename-input']).toBe('true') }) + it('ignores a rename request aimed at a different tab', async () => { + await renderSortableTab() + requestTerminalTabRename('terminal-tab-2') + const rerender = expandNode(await renderSortableTab()) + + expect(findElementsByType(rerender, 'input')).toHaveLength(0) + }) + it('ignores IME composition Enter before committing the custom tab title', async () => { const onSetCustomTitle = vi.fn() await renderSortableTab({ onSetCustomTitle }) + requestTerminalTabRename('terminal-tab-1') let rerender = expandNode(await renderSortableTab({ onSetCustomTitle })) let input = findElementsByType(rerender, 'input')[0] ;(input.props.onChange as (event: { target: { value: string } }) => void)({ diff --git a/src/renderer/src/components/tab-bar/SortableTab.tsx b/src/renderer/src/components/tab-bar/SortableTab.tsx index 70935abb29a..cb5966d1e23 100644 --- a/src/renderer/src/components/tab-bar/SortableTab.tsx +++ b/src/renderer/src/components/tab-bar/SortableTab.tsx @@ -17,6 +17,10 @@ import { type DropIndicator } from './drop-indicator' import { preventMiddleButtonDefault } from './middle-button-default-guard' +import { + RENAME_TERMINAL_TAB_EVENT, + type RenameTerminalTabDetail +} from './terminal-tab-rename-request' import { SortableTabContextMenu } from './SortableTabContextMenu' import { translate } from '@/i18n/i18n' import { TAB_CONTAINER_WIDTH_CLASSES, TAB_LABEL_WIDTH_CLASSES } from './tab-width-rules' @@ -97,8 +101,6 @@ export default function SortableTab({ terminalLayout: s.terminalLayoutsByTabId?.[tab.id] }) ) - const renamingTabId = useAppStore((s) => s.renamingTabId) - const setRenamingTabId = useAppStore((s) => s.setRenamingTabId) // Why: shellOverride is stamped at create time, so changing the default shell later won't repaint existing tabs. const shellForIcon = tab.shellOverride @@ -166,14 +168,25 @@ export default function SortableTab({ }) }, []) - // Why: the tab.rename shortcut routes through store renamingTabId; open the editor and clear it so it fires once. + // Why the ref: keeps the listener subscribed to tab.id alone, so OSC title churn can't + // resubscribe it mid-edit. Written from an Effect, not in render -- a render React discards + // must not leave a stale handler behind for the next commit to fire. + const handleRenameOpenRef = useRef(handleRenameOpen) useEffect(() => { - if (renamingTabId !== tab.id) { - return + handleRenameOpenRef.current = handleRenameOpen + }, [handleRenameOpen]) + + useEffect(() => { + const onRenameRequest = (event: Event): void => { + const detail = (event as CustomEvent).detail + if (detail?.tabId !== tab.id) { + return + } + handleRenameOpenRef.current() } - handleRenameOpen() - setRenamingTabId(null) - }, [renamingTabId, tab.id, handleRenameOpen, setRenamingTabId]) + window.addEventListener(RENAME_TERMINAL_TAB_EVENT, onRenameRequest) + return () => window.removeEventListener(RENAME_TERMINAL_TAB_EVENT, onRenameRequest) + }, [tab.id]) useEffect(() => { const closeMenu = (): void => setMenuOpen(false) diff --git a/src/renderer/src/components/tab-bar/SortableTab.update-depth-probe.test.tsx b/src/renderer/src/components/tab-bar/SortableTab.update-depth-probe.test.tsx index acbd7ce38bc..39cacfb7825 100644 --- a/src/renderer/src/components/tab-bar/SortableTab.update-depth-probe.test.tsx +++ b/src/renderer/src/components/tab-bar/SortableTab.update-depth-probe.test.tsx @@ -18,6 +18,7 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { TabDragItemData } from '../tab-group/useTabDragSplit' import { useAppStore } from '../../store' import SortableTab from './SortableTab' +import { requestTerminalTabRename } from './terminal-tab-rename-request' type ProbeState = { unreadTerminalTabs: Record @@ -27,9 +28,7 @@ type ProbeState = { runtimePaneTitlesByTabId: Record> ptyIdsByTabId: Record terminalLayoutsByTabId: Record - renamingTabId: string | null keybindings: Record - setRenamingTabId: (tabId: string | null) => void } type StoreApiWithHook = { @@ -44,7 +43,7 @@ async function createProbeStore(): Promise { const globals = globalThis as Record if (!globals[globalKey]) { const { create } = await import('zustand') - globals[globalKey] = create((set) => ({ + globals[globalKey] = create(() => ({ unreadTerminalTabs: {}, unreadAgentCompletionPanes: {}, agentStatusByPaneKey: {}, @@ -52,9 +51,7 @@ async function createProbeStore(): Promise { runtimePaneTitlesByTabId: {}, ptyIdsByTabId: {}, terminalLayoutsByTabId: {}, - renamingTabId: null, - keybindings: {}, - setRenamingTabId: (tabId) => set({ renamingTabId: tabId }) + keybindings: {} })) } return globals[globalKey] as StoreApiWithHook @@ -101,6 +98,15 @@ vi.mock('@/components/ui/input', () => ({ Input: (props: Record) => })) +// Counts SortableTab's own renders: it is rendered unconditionally inside the tab body, so one +// stub render == one SortableTab render, which the Harness counter above cannot see. +vi.mock('./TerminalTabLeadingIcon', () => ({ + TerminalTabLeadingIcon: () => { + tabRenderCount += 1 + return + } +})) + vi.mock('./shell-icons', () => ({ ShellIcon: () => })) vi.mock('@/lib/agent-catalog', () => ({ AgentIcon: () => })) vi.mock('../sidebar/WorktreeCardHelpers', () => ({ FilledBellIcon: () => })) @@ -127,6 +133,7 @@ const dragData: TabDragItemData = { const probeStore = useAppStore as unknown as StoreApiWithHook let renderCount = 0 +let tabRenderCount = 0 function Harness({ tab }: { tab: TerminalTab }): ReactElement { renderCount += 1 @@ -170,25 +177,52 @@ function ChurningHarness({ titles }: { titles: string[] }): ReactElement { afterEach(() => { cleanup() - probeStore.setState({ renamingTabId: null, unreadTerminalTabs: {}, agentStatusEpoch: 0 }) + probeStore.setState({ unreadTerminalTabs: {}, agentStatusEpoch: 0 }) renderCount = 0 + tabRenderCount = 0 }) describe('SortableTab update-depth probe', () => { - it('settles when the rename shortcut arms renamingTabId', () => { - probeStore.setState({ renamingTabId: 'terminal-tab-1' }) + it('settles when the rename shortcut targets this tab', () => { const { container } = render() + act(() => requestTerminalTabRename('terminal-tab-1')) expect(container.querySelector('[data-tab-rename-input]')).not.toBeNull() - expect(probeStore.getState().renamingTabId).toBeNull() expect(renderCount).toBeLessThan(20) }) it('settles under title churn while the rename editor is open', () => { - probeStore.setState({ renamingTabId: 'terminal-tab-1' }) render() + act(() => requestTerminalTabRename('terminal-tab-1')) expect(renderCount).toBeLessThan(40) }) + // Regression: the shortcut used to arm a store field every tab subscribed to, so one rename + // re-rendered every mounted tab twice — once to notice the id, once when the tab cleared it. + it('re-renders only the targeted tab, once, per rename request', () => { + render( + <> + + + + + ) + const mountRenders = tabRenderCount + act(() => requestTerminalTabRename('terminal-tab-1')) + expect(tabRenderCount - mountRenders).toBe(1) + }) + + it('does not re-render any tab for a rename request no mounted tab owns', () => { + render( + <> + + + + ) + const mountRenders = tabRenderCount + act(() => requestTerminalTabRename('terminal-tab-9')) + expect(tabRenderCount).toBe(mountRenders) + }) + it('settles under a store write storm', () => { render() const before = renderCount @@ -201,12 +235,12 @@ describe('SortableTab update-depth probe', () => { }) it('settles in StrictMode double-invoked effects', () => { - probeStore.setState({ renamingTabId: 'terminal-tab-1' }) - render( + const { container } = render( ) - expect(probeStore.getState().renamingTabId).toBeNull() + act(() => requestTerminalTabRename('terminal-tab-1')) + expect(container.querySelector('[data-tab-rename-input]')).not.toBeNull() }) }) diff --git a/src/renderer/src/components/tab-bar/terminal-tab-rename-request.ts b/src/renderer/src/components/tab-bar/terminal-tab-rename-request.ts new file mode 100644 index 00000000000..4559ad0ea95 --- /dev/null +++ b/src/renderer/src/components/tab-bar/terminal-tab-rename-request.ts @@ -0,0 +1,19 @@ +/** + * Targeted rename dispatch for the `tab.rename` shortcut. + * + * Why an event and not store state: a store field is read by every mounted tab, so arming it + * re-rendered the whole strip, and the consuming tab then had to clear it — a second pass over + * every tab. The event reaches only the tab that owns the id. Mirrors the terminal-pane + * TOGGLE_TERMINAL_PANE_EXPAND_EVENT / FOCUS_TERMINAL_PANE_EVENT dispatch. + */ +export const RENAME_TERMINAL_TAB_EVENT = 'orca-rename-terminal-tab' + +export type RenameTerminalTabDetail = { + tabId: string +} + +export function requestTerminalTabRename(tabId: string): void { + window.dispatchEvent( + new CustomEvent(RENAME_TERMINAL_TAB_EVENT, { detail: { tabId } }) + ) +} diff --git a/src/renderer/src/components/terminal-pane/use-terminal-tab-cold-parking.ts b/src/renderer/src/components/terminal-pane/use-terminal-tab-cold-parking.ts index e28637097f7..dfc6ffc9614 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-tab-cold-parking.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-tab-cold-parking.ts @@ -120,7 +120,10 @@ export function useTerminalTabColdParking(args: { ) const terminalTabHiddenSinceRef = useRef(new Map()) // Why: view switches hide every tab at once, so the park clock cannot rank them. - const terminalTabActivationOrderRef = useRef(createTerminalTabActivationOrder()) + const terminalTabActivationOrderRef = useRef>( + undefined! + ) + terminalTabActivationOrderRef.current ??= createTerminalTabActivationOrder() // Why (shared measure-clock contract with Terminal.tsx): tab hiddenSince // survives a background-measure window so per-tab park deadlines stay in // sync with the worktree retention/TTL clock, and a post-measure cool-down diff --git a/src/renderer/src/components/terminal-quick-commands/TerminalQuickCommandDialog.tsx b/src/renderer/src/components/terminal-quick-commands/TerminalQuickCommandDialog.tsx index c0d5a26c2e1..9593c23d276 100644 --- a/src/renderer/src/components/terminal-quick-commands/TerminalQuickCommandDialog.tsx +++ b/src/renderer/src/components/terminal-quick-commands/TerminalQuickCommandDialog.tsx @@ -76,7 +76,10 @@ export function TerminalQuickCommandDialog({ const [draft, setDraft] = useState(command) const wasOpenRef = useRef(open) const syncedCommandRef = useRef(command) - const draftMemoryRef = useRef(createTerminalQuickCommandDialogDraftMemory(command, fallbackAgent)) + const draftMemoryRef = useRef>( + undefined! + ) + draftMemoryRef.current ??= createTerminalQuickCommandDialogDraftMemory(command, fallbackAgent) const initialScope = getTerminalQuickCommandScope(command) const lastRepoScopeIdRef = useRef( initialScope.type === 'repo' ? initialScope.repoId : null diff --git a/src/renderer/src/components/use-terminal-window-lifecycle.lazy-ref.test.tsx b/src/renderer/src/components/use-terminal-window-lifecycle.lazy-ref.test.tsx new file mode 100644 index 00000000000..ce99e46db27 --- /dev/null +++ b/src/renderer/src/components/use-terminal-window-lifecycle.lazy-ref.test.tsx @@ -0,0 +1,86 @@ +// @vitest-environment happy-dom + +/** + * `collectBrowserWebviewIds` walks every browser page and tab across every worktree. It used to sit + * in a `useRef(...)` argument, so `Terminal` paid for the whole walk on every render and threw the + * result away. Pin the invocation count to the mount count, not the render count. + */ +import { useState } from 'react' +import { act, cleanup, render } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' + +const collectBrowserWebviewIdsCalls = vi.hoisted(() => ({ count: 0 })) + +vi.mock('../store', () => { + const state = { + browserTabsByWorktree: {}, + browserPagesByWorkspace: {}, + openFiles: [] + } + const useAppStore = Object.assign(() => undefined, { + getState: () => state, + subscribe: () => () => {} + }) + return { useAppStore } +}) + +vi.mock('../store/slices/browser-webview-cleanup', () => ({ + collectBrowserWebviewIds: (...args: unknown[]) => { + collectBrowserWebviewIdsCalls.count += 1 + void args + return new Set() + }, + destroyRemovedBrowserWebview: vi.fn() +})) + +vi.mock('@/lib/updater-beforeunload', () => ({ + isIntentionalAppRestartInProgress: () => false +})) +vi.mock('@/lib/shutdown-checkpoint-guard', () => ({ + preventUnloadAndScheduleShutdownCheckpointReset: vi.fn() +})) +vi.mock('./window-close-request-coordinator', () => ({ + setWindowCloseRequestHandler: vi.fn() +})) + +const { useTerminalWindowLifecycle } = await import('./use-terminal-window-lifecycle') + +const controller = { + activeBrowserTabId: null, + activeTabType: 'terminal', + activeWorktreeBrowserTabIdsKey: '', + proceedToNativeWindowClose: () => {}, + queueEditorCloseRequests: () => {}, + renderedActiveWorktreeId: null, + setActiveBrowserTab: () => {}, + setActiveTabType: () => {}, + windowCloseAfterDirtyRef: { current: false } +} as unknown as Parameters[0] + +let bumpRender: (() => void) | null = null + +function Host(): null { + const [, setTick] = useState(0) + bumpRender = () => setTick((tick) => tick + 1) + useTerminalWindowLifecycle(controller) + return null +} + +afterEach(() => { + cleanup() + collectBrowserWebviewIdsCalls.count = 0 + bumpRender = null +}) + +describe('useTerminalWindowLifecycle browser-webview id seed', () => { + it('collects the id set once per mount, not once per render', () => { + render() + expect(collectBrowserWebviewIdsCalls.count).toBe(1) + + for (let i = 0; i < 20; i += 1) { + act(() => bumpRender?.()) + } + + expect(collectBrowserWebviewIdsCalls.count).toBe(1) + }) +}) diff --git a/src/renderer/src/components/use-terminal-window-lifecycle.ts b/src/renderer/src/components/use-terminal-window-lifecycle.ts index a252422de9f..23cdb930535 100644 --- a/src/renderer/src/components/use-terminal-window-lifecycle.ts +++ b/src/renderer/src/components/use-terminal-window-lifecycle.ts @@ -58,11 +58,12 @@ export function useTerminalWindowLifecycle(controller: TerminalActivationControl // oxlint-disable-next-line react-hooks/exhaustive-deps -- controller refs preserve their original stable identities. }, [proceedToNativeWindowClose, queueEditorCloseRequests]) - const prevBrowserWebviewIdsRef = useRef>( - collectBrowserWebviewIds( - useAppStore.getState().browserTabsByWorktree, - useAppStore.getState().browserPagesByWorkspace - ) + // Why lazy: a `useRef(expr)` argument re-runs on every render and is thrown away, and this + // walks every browser page and tab across all worktrees on a component that renders constantly. + const prevBrowserWebviewIdsRef = useRef>(undefined!) + prevBrowserWebviewIdsRef.current ??= collectBrowserWebviewIds( + useAppStore.getState().browserTabsByWorktree, + useAppStore.getState().browserPagesByWorkspace ) useEffect(() => { let prevBrowserTabs = useAppStore.getState().browserTabsByWorktree diff --git a/src/renderer/src/hooks/composer-state/async-composer-state.ts b/src/renderer/src/hooks/composer-state/async-composer-state.ts index 2980576e3f7..f84baa51469 100644 --- a/src/renderer/src/hooks/composer-state/async-composer-state.ts +++ b/src/renderer/src/hooks/composer-state/async-composer-state.ts @@ -128,14 +128,13 @@ export function useComposerAsyncState(input: ComposerAsyncStateInput) { const [linkDirectLoading, setLinkDirectLoading] = useState(false) - const lastAutoNameRef = useRef( - getInitialAutoManagedWorkspaceName({ - draftName: persistDraft ? newWorkspaceDraft?.name : null, - draftLinkedWorkItem: persistDraft ? draftLinkedWorkItemSeed : null, - initialName, - initialLinkedWorkItem: initialLinkedWorkItemSeed - }) - ) + const lastAutoNameRef = useRef(undefined!) + lastAutoNameRef.current ??= getInitialAutoManagedWorkspaceName({ + draftName: persistDraft ? newWorkspaceDraft?.name : null, + draftLinkedWorkItem: persistDraft ? draftLinkedWorkItemSeed : null, + initialName, + initialLinkedWorkItem: initialLinkedWorkItemSeed + }) const nameRef = useRef(name) @@ -153,12 +152,18 @@ export function useComposerAsyncState(input: ComposerAsyncStateInput) { }, [name, note]) // Why: PR checkout refs resolve async, so submit can still see the linked PR as a checkout source if Create fires before the resolver settles. + // Why useState and not `useRef(expr)`: the seed legitimately resolves to null, so a nullish + // guard would keep re-running it; useState's initializer runs once without writing in render. + const [initialSmartGitHubPrStartPointSelection] = + useState(() => + getInitialGitHubPrStartPointSelection({ + item: initialGitHubWorkItem, + linkedWorkItem: initialLinkedWorkItemSeed, + repoId: selectedRepo?.id ?? initialRepoId + }) + ) const smartGitHubPrStartPointSelectionRef = useRef( - getInitialGitHubPrStartPointSelection({ - item: initialGitHubWorkItem, - linkedWorkItem: initialLinkedWorkItemSeed, - repoId: selectedRepo?.id ?? initialRepoId - }) + initialSmartGitHubPrStartPointSelection ) useEffect(() => { diff --git a/src/renderer/src/hooks/composer-state/composer-drop-listener.ts b/src/renderer/src/hooks/composer-state/composer-drop-listener.ts index 145fe1a3c35..5cca0d4b073 100644 --- a/src/renderer/src/hooks/composer-state/composer-drop-listener.ts +++ b/src/renderer/src/hooks/composer-state/composer-drop-listener.ts @@ -10,7 +10,8 @@ export function useComposerDropListener( useEffect(() => { applyDropRef.current = applyDrop }, [applyDrop]) - const instanceIdRef = useRef(Symbol('composer')) + const instanceIdRef = useRef(undefined!) + instanceIdRef.current ??= Symbol('composer') useEffect(() => { const instanceId = instanceIdRef.current diff --git a/src/renderer/src/hooks/use-audio-capture.ts b/src/renderer/src/hooks/use-audio-capture.ts index 702640ae1c9..31e851a58ee 100644 --- a/src/renderer/src/hooks/use-audio-capture.ts +++ b/src/renderer/src/hooks/use-audio-capture.ts @@ -55,7 +55,8 @@ export function useAudioCapture(publishMeter?: DictationMeterPublisher) { const capturedChunkCountRef = useRef(0) const sessionIdRef = useRef('desktop') const trackLostCleanupRef = useRef<(() => void) | null>(null) - const meterAnalyzerRef = useRef(createDictationMeterAnalyzerState()) + const meterAnalyzerRef = useRef>(undefined!) + meterAnalyzerRef.current ??= createDictationMeterAnalyzerState() const publishedMeterRef = useRef(DEFAULT_DICTATION_METER) const lastMeterPublishedAtRef = useRef(Number.NEGATIVE_INFINITY) diff --git a/src/renderer/src/lazy-use-ref-ratchet.test.ts b/src/renderer/src/lazy-use-ref-ratchet.test.ts new file mode 100644 index 00000000000..c0b23821bcb --- /dev/null +++ b/src/renderer/src/lazy-use-ref-ratchet.test.ts @@ -0,0 +1,130 @@ +import { readFileSync, readdirSync } from 'node:fs' +import path from 'node:path' +import { describe, expect, it } from 'vitest' + +/** + * `useRef(create())` evaluates its argument on EVERY render and discards every result after the + * first, so any real work there is pure waste. The lazy form keeps the same value with none of the + * churn: + * + * const ref = useRef(undefined!) + * ref.current ??= create() + * + * Use a `useState` lazy initializer instead when the seed can legitimately be null/undefined, or + * an explicit `seededRef` guard when both are true. + */ +const RENDERER_ROOT = import.meta.dirname + +function collectSourceFiles(dir: string, out: string[] = []): string[] { + for (const entry of readdirSync(dir, { withFileTypes: true })) { + const full = path.join(dir, entry.name) + if (entry.isDirectory()) { + if (entry.name === 'node_modules' || entry.name === 'dist') { + continue + } + collectSourceFiles(full, out) + continue + } + if (!/\.tsx?$/.test(entry.name) || /\.(test|spec)\.tsx?$/.test(entry.name)) { + continue + } + out.push(full) + } + return out +} + +/** Reads the balanced argument text of the `useRef(...)` starting at `from`. */ +function readUseRefArgument(source: string, from: number): { arg: string; end: number } | null { + let i = from + while (source[i] === ' ') { + i++ + } + if (source[i] === '<') { + let depth = 0 + while (i < source.length) { + if (source[i] === '<') { + depth++ + } else if (source[i] === '>') { + depth-- + if (depth === 0) { + i++ + break + } + } + i++ + } + } + while (source[i] === ' ') { + i++ + } + if (source[i] !== '(') { + return null + } + const argStart = i + 1 + let depth = 0 + while (i < source.length) { + if (source[i] === '(') { + depth++ + } else if (source[i] === ')') { + depth-- + if (depth === 0) { + return { arg: source.slice(argStart, i).trim(), end: i + 1 } + } + } + i++ + } + return null +} + +// Allowed because none of these is work that a render repeats for nothing: +// - an empty collection literal, which the repo keeps in the direct form +// - `undefined!`, the lazy-seed marker +// - a function literal, which is the ref's payload rather than its initialization +// - a primitive coercion of an already-computed value +const ALLOWED_ARGUMENT = new RegExp( + [ + '^new (Map|Set|WeakMap|WeakSet)(<[\\s\\S]*>)?\\(\\)$', + '^undefined!$', + '^(async )?\\([\\s\\S]*?\\)\\s*(:[^=]*)?=>[\\s\\S]*$', + '^(Boolean|Number|String)\\([\\s\\S]*\\)$', + '^[A-Za-z_$][\\w$.?\\[\\]\'"]*$' + ].join('|') +) + +function findNonLazyUseRefs(file: string): string[] { + const source = readFileSync(file, 'utf8') + const findings: string[] = [] + let index = 0 + while ((index = source.indexOf('useRef', index)) !== -1) { + const start = index + index += 'useRef'.length + if (/[\w$.]/.test(source[start - 1] ?? '')) { + continue + } + const parsed = readUseRefArgument(source, index) + if (!parsed) { + continue + } + index = parsed.end + const arg = parsed.arg + if (arg === '' || ALLOWED_ARGUMENT.test(arg)) { + continue + } + // Only a call or constructor invocation actually burns work per render. + if (!/\(/.test(arg) && !/\bnew\b/.test(arg)) { + continue + } + const line = source.slice(0, start).split('\n').length + findings.push( + `${path.relative(RENDERER_ROOT, file)}:${line} useRef(${arg.replace(/\s+/g, ' ').slice(0, 90)})` + ) + } + return findings +} + +describe('renderer useRef initializers', () => { + it('never does work in the useRef argument', () => { + const findings = collectSourceFiles(RENDERER_ROOT).flatMap(findNonLazyUseRefs) + expect(findings).toEqual([]) + }) +}) diff --git a/src/renderer/src/store/slices/tabs-label-and-pin-state.test.ts b/src/renderer/src/store/slices/tabs-label-and-pin-state.test.ts index ef78f34f954..ea2a9bbb385 100644 --- a/src/renderer/src/store/slices/tabs-label-and-pin-state.test.ts +++ b/src/renderer/src/store/slices/tabs-label-and-pin-state.test.ts @@ -26,14 +26,6 @@ describe('TabsSlice', () => { store = createTestStore() }) - it('setRenamingTabId sets and clears the tab rename signal', () => { - expect(store.getState().renamingTabId).toBeNull() - store.getState().setRenamingTabId('terminal-tab-1') - expect(store.getState().renamingTabId).toBe('terminal-tab-1') - store.getState().setRenamingTabId(null) - expect(store.getState().renamingTabId).toBeNull() - }) - // ─── setTabLabel / setTabCustomLabel / setUnifiedTabColor ───────── describe('tab property setters', () => { diff --git a/src/renderer/src/store/slices/tabs/create-tabs-slice.ts b/src/renderer/src/store/slices/tabs/create-tabs-slice.ts index 799cd257339..adbea2f1726 100644 --- a/src/renderer/src/store/slices/tabs/create-tabs-slice.ts +++ b/src/renderer/src/store/slices/tabs/create-tabs-slice.ts @@ -14,7 +14,6 @@ import { createTabsSessionActions } from './tabs-session-actions' export const createTabsSlice: StateCreator = (set, get) => ({ unifiedTabsByWorktree: {}, - renamingTabId: null, groupsByWorktree: {}, activeGroupIdByWorktree: {}, layoutByWorktree: {}, diff --git a/src/renderer/src/store/slices/tabs/tabs-label-actions.ts b/src/renderer/src/store/slices/tabs/tabs-label-actions.ts index 012e1e8c0c0..8acb925bfae 100644 --- a/src/renderer/src/store/slices/tabs/tabs-label-actions.ts +++ b/src/renderer/src/store/slices/tabs/tabs-label-actions.ts @@ -18,7 +18,6 @@ export function createTabsLabelActions( | 'setTabLabel' | 'setTabViewMode' | 'toggleTabViewMode' - | 'setRenamingTabId' | 'setTabCustomLabel' | 'setUnifiedTabColor' | 'pinTab' @@ -101,10 +100,6 @@ export function createTabsLabelActions( } }, - setRenamingTabId: (tabId) => { - set({ renamingTabId: tabId }) - }, - setTabCustomLabel: (tabId, label, opts) => { const exists = get().getTab(tabId) !== null set((state) => patchTab(state.unifiedTabsByWorktree, tabId, { customLabel: label }) ?? {}) diff --git a/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts b/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts index 81d27ff54d5..2835d5e46d6 100644 --- a/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts +++ b/src/renderer/src/store/slices/tabs/tabs-slice-contract.ts @@ -13,8 +13,6 @@ export type TabSplitDirection = 'left' | 'right' | 'up' | 'down' export type TabsSlice = { unifiedTabsByWorktree: Record - // Why: id of the tab whose inline title editor should open; shortcut (tab.rename) sets it, the tab clears it on consume. - renamingTabId: string | null groupsByWorktree: Record activeGroupIdByWorktree: Record layoutByWorktree: Record @@ -101,7 +99,6 @@ export type TabsSlice = { opts?: { recordInteraction?: boolean } ) => void setUnifiedTabColor: (tabId: string, color: string | null) => void - setRenamingTabId: (tabId: string | null) => void pinTab: (tabId: string) => void unpinTab: (tabId: string) => void closeOtherTabs: (tabId: string) => string[] From 974f3164a0380505611b19668612341e0bd3f742 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:36:00 -0700 Subject: [PATCH 096/398] Make terminal error overlays opaque (#18231) * Make terminal error overlays opaque * fix(terminal): keep opaque error toast text readable * fix(terminal): keep toast fallback opaque on older browsers --------- Co-authored-by: Merge Sim --- .../terminal-pane/TerminalErrorToast.tsx | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx b/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx index dcb838007b9..8b627a9651f 100644 --- a/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalErrorToast.tsx @@ -179,6 +179,11 @@ export function TerminalErrorToast({ const showIssueLink = !ssh && !paneOwnerUnverified && !showDaemonRestart && !isExplainedTerminalError(error) const displayError = humanizeTerminalError(error) + const tint = paneOwnerUnverified + ? null + : ssh + ? 'color-mix(in srgb, var(--color-amber-500) 20%, var(--popover))' + : 'color-mix(in srgb, var(--destructive) 20%, var(--popover))' const [retrying, setRetrying] = useState(false) const [retryFailed, setRetryFailed] = useState(false) const [environmentFooter, setEnvironmentFooter] = useState<{ @@ -231,17 +236,14 @@ export function TerminalErrorToast({ zIndex: 50, padding: '10px 14px', borderRadius: 6, - background: paneOwnerUnverified - ? 'var(--popover)' - : ssh - ? 'rgba(234, 179, 8, 0.12)' - : 'rgba(220, 38, 38, 0.15)', + background: 'var(--popover)', + backgroundImage: tint ? `linear-gradient(${tint}, ${tint})` : undefined, border: paneOwnerUnverified ? '1px solid var(--color-amber-500)' : ssh ? '1px solid rgba(234, 179, 8, 0.35)' : '1px solid rgba(220, 38, 38, 0.4)', - color: paneOwnerUnverified ? 'var(--popover-foreground)' : ssh ? '#fde68a' : '#fca5a5', + color: 'var(--popover-foreground)', fontSize: 12, fontFamily: 'monospace', whiteSpace: 'pre-wrap', @@ -268,7 +270,7 @@ export function TerminalErrorToast({ )}{' '} {translate( 'auto.components.terminal.pane.TerminalErrorToast.a7e2fd2699', @@ -326,7 +328,7 @@ export function TerminalErrorToast({ style={{ background: 'none', border: 'none', - color: paneOwnerUnverified ? 'var(--popover-foreground)' : ssh ? '#fde68a' : '#fca5a5', + color: 'inherit', cursor: 'pointer', fontSize: 14, padding: '0 0 0 8px', From a0de2fde0b802961721e6cfe5df277825e6afac2 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:39:56 -0400 Subject: [PATCH 097/398] fix(terminal): confirm an unrecognized foreground before downgrading agent prompts (#18238) --- ...rca-runtime-get-pty-record-for-pane-key.ts | 20 ++++- ...led-prompt-foreground-confirmation.spec.ts | 78 +++++++++++++++++++ src/main/runtime/orca-runtime.test.ts | 1 + .../runtime-terminal-agent-presence.ts | 16 +++- src/relay/pty-shell-utils.test.ts | 30 ++++++- src/relay/pty-shell-utils.ts | 8 +- 6 files changed, 140 insertions(+), 13 deletions(-) create mode 100644 src/main/runtime/orca-runtime-tests/terminal-settled-prompt-foreground-confirmation.spec.ts diff --git a/src/main/runtime/orca-runtime-get-pty-record-for-pane-key.ts b/src/main/runtime/orca-runtime-get-pty-record-for-pane-key.ts index 37fac094067..32dd82fd48c 100644 --- a/src/main/runtime/orca-runtime-get-pty-record-for-pane-key.ts +++ b/src/main/runtime/orca-runtime-get-pty-record-for-pane-key.ts @@ -112,13 +112,25 @@ export class OrcaRuntimeWithGetPtyRecordForPaneKey extends OrcaRuntimeWithPruneM if (!ptyId || !trackedPty || !this.ptyController) { return false } - const agent = recognizeAgentProcess( - await this.ptyController.getForegroundProcess(ptyId) - )?.agent + let foregroundProcess = await this.ptyController.getForegroundProcess(ptyId) + let agent = recognizeAgentProcess(foregroundProcess)?.agent + // Why: the cached foreground name can be an executable basename nothing recognizes + // (macOS p_comm reports the native Claude installer as `2.1.258`), and treating that + // as "no agent" silently downgrades the prompt to unframed chunks, which Claude's + // composer truncates. A fresh process-table scan reads the real command line. + if (agent === undefined && this.ptyController.confirmForegroundProcess) { + foregroundProcess = await this.ptyController.confirmForegroundProcess(ptyId) + agent = recognizeAgentProcess(foregroundProcess)?.agent + } if (agent !== 'claude' && agent !== 'codex') { return false } - if (!(await this.isTerminalRunningAgent(handle, { retryForegroundWrappers: false }))) { + if ( + !(await this.isTerminalRunningAgent(handle, { + retryForegroundWrappers: false, + foregroundProcess + })) + ) { return false } trackedPty.foregroundAgent = agent diff --git a/src/main/runtime/orca-runtime-tests/terminal-settled-prompt-foreground-confirmation.spec.ts b/src/main/runtime/orca-runtime-tests/terminal-settled-prompt-foreground-confirmation.spec.ts new file mode 100644 index 00000000000..84d55a7e1d2 --- /dev/null +++ b/src/main/runtime/orca-runtime-tests/terminal-settled-prompt-foreground-confirmation.spec.ts @@ -0,0 +1,78 @@ +import { describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from '../orca-runtime-test-mocks.spec' +import { store, syncSinglePty } from '../orca-runtime-test-fixtures.spec' + +// Why: node-pty's cached foreground name is p_comm on macOS, which reports the native Claude +// install as its version directory (`2.1.258`). Reading that as "no agent" silently downgraded +// `terminal.send --enter` from the atomic bracketed-paste route to unframed 16 KiB chunks, +// which Claude's composer truncates for large prompts (STA-4577). +describe('isTerminalRunningSettledPromptAgent foreground confirmation', () => { + // `2.1.258`: macOS p_comm for the native Claude install. `bash.exe`: the Windows daemon + // tracker answers with the shell fallback until its async scan lands. + it.each(['2.1.258', 'bash.exe'])( + 'confirms an unrecognized foreground (%s) before refusing the settled route', + async (cachedForeground) => { + const getForegroundProcess = vi.fn(async () => cachedForeground) + const confirmForegroundProcess = vi.fn(async () => 'claude') + const runtime = new OrcaRuntimeService(store) + runtime.setPtyController({ + write: () => true, + kill: () => true, + getForegroundProcess, + confirmForegroundProcess + }) + syncSinglePty(runtime, 'pty-1', { paneTitle: 'bash' }) + const [terminal] = (await runtime.listTerminals()).terminals + + await expect(runtime.isTerminalRunningSettledPromptAgent(terminal.handle)).resolves.toBe(true) + expect(confirmForegroundProcess).toHaveBeenCalledWith('pty-1') + // The confirmed identity is reused; the cached read must not be re-consulted and win. + expect(getForegroundProcess).toHaveBeenCalledTimes(1) + } + ) + + it('keeps legacy delivery when confirmation also finds no target agent', async () => { + const confirmForegroundProcess = vi.fn(async () => 'vim') + const runtime = new OrcaRuntimeService(store) + runtime.setPtyController({ + write: () => true, + kill: () => true, + getForegroundProcess: async () => '2.1.258', + confirmForegroundProcess + }) + syncSinglePty(runtime, 'pty-1', { paneTitle: 'bash' }) + const [terminal] = (await runtime.listTerminals()).terminals + + await expect(runtime.isTerminalRunningSettledPromptAgent(terminal.handle)).resolves.toBe(false) + expect(confirmForegroundProcess).toHaveBeenCalledOnce() + }) + + it('does not confirm when the cached foreground already names a target agent', async () => { + const confirmForegroundProcess = vi.fn(async () => 'claude') + const runtime = new OrcaRuntimeService(store) + runtime.setPtyController({ + write: () => true, + kill: () => true, + getForegroundProcess: async () => 'claude', + confirmForegroundProcess + }) + syncSinglePty(runtime, 'pty-1', { paneTitle: 'bash' }) + const [terminal] = (await runtime.listTerminals()).terminals + + await expect(runtime.isTerminalRunningSettledPromptAgent(terminal.handle)).resolves.toBe(true) + expect(confirmForegroundProcess).not.toHaveBeenCalled() + }) + + it('refuses the settled route when the provider cannot confirm', async () => { + const runtime = new OrcaRuntimeService(store) + runtime.setPtyController({ + write: () => true, + kill: () => true, + getForegroundProcess: async () => '2.1.258' + }) + syncSinglePty(runtime, 'pty-1', { paneTitle: 'bash' }) + const [terminal] = (await runtime.listTerminals()).terminals + + await expect(runtime.isTerminalRunningSettledPromptAgent(terminal.handle)).resolves.toBe(false) + }) +}) diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index f8b12ba4a9d..aabfeb899c8 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -57,6 +57,7 @@ await import('./orca-runtime-tests/terminal-output-and-worker-recovery-part-06.s await import('./orca-runtime-tests/terminal-output-and-worker-recovery-part-07.spec') await import('./orca-runtime-tests/terminal-handles-and-agent-status.spec') await import('./orca-runtime-tests/terminal-handles-and-agent-status-part-02.spec') +await import('./orca-runtime-tests/terminal-settled-prompt-foreground-confirmation.spec') await import('./orca-runtime-tests/terminal-handles-and-agent-status-part-03.spec') await import('./orca-runtime-tests/terminal-handles-and-agent-status-part-04.spec') await import('./orca-runtime-tests/terminal-handles-and-agent-status-part-05.spec') diff --git a/src/main/runtime/runtime-terminal-agent-presence.ts b/src/main/runtime/runtime-terminal-agent-presence.ts index 86681e97688..e87520fccc8 100644 --- a/src/main/runtime/runtime-terminal-agent-presence.ts +++ b/src/main/runtime/runtime-terminal-agent-presence.ts @@ -30,6 +30,8 @@ type RuntimeTerminalAgentPresenceDependencies = { export type RuntimeTerminalAgentPresenceOptions = { retryForegroundWrappers?: boolean + /** Foreground identity the caller already confirmed; skips the provider's cached read. */ + foregroundProcess?: string | null } export class RuntimeTerminalAgentPresence { @@ -75,7 +77,7 @@ export class RuntimeTerminalAgentPresence { if (!leaf.ptyId) { return false } - const foreground = await this.deps.getForegroundProcess(leaf.ptyId) + const foreground = await this.readForegroundProcess(leaf.ptyId, options) if (!foreground) { return false } @@ -138,7 +140,7 @@ export class RuntimeTerminalAgentPresence { ) { return true } - const foreground = await this.deps.getForegroundProcess(pty.ptyId) + const foreground = await this.readForegroundProcess(pty.ptyId, options) if (!foreground) { return false } @@ -157,6 +159,16 @@ export class RuntimeTerminalAgentPresence { ) } + private async readForegroundProcess( + ptyId: string, + options: RuntimeTerminalAgentPresenceOptions + ): Promise { + if (options.foregroundProcess !== undefined) { + return options.foregroundProcess + } + return await this.deps.getForegroundProcess(ptyId) + } + private async isRecognizedForegroundAgentProcess( ptyId: string, foregroundProcess: string, diff --git a/src/relay/pty-shell-utils.test.ts b/src/relay/pty-shell-utils.test.ts index c806faa7978..6bfeab1a56a 100644 --- a/src/relay/pty-shell-utils.test.ts +++ b/src/relay/pty-shell-utils.test.ts @@ -299,10 +299,34 @@ describe('resolveDefaultCwd', () => { }) describe('getForegroundProcessName', () => { - it('returns clear non-wrapper foregrounds without process-table enrichment', async () => { - await expect(getForegroundProcessName(100, 'vim')).resolves.toBe('vim') + it('keeps a non-agent foreground name when the process table shows no agent', async () => { + await withProcessPlatform('darwin', async () => { + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: ['100 99 Ss zsh -l', '101 100 S+ vim notes.md'].join('\n') } + } + return new Error('unexpected command') + }) - expect(execFileMock).not.toHaveBeenCalled() + await expect(getForegroundProcessName(100, 'vim')).resolves.toBe('vim') + }) + }) + + it('resolves a macOS p_comm basename to the agent that owns the foreground', async () => { + // Why: node-pty reports the native Claude binary as its version directory (`2.1.258`); + // answering with that name downgrades agent prompts to unframed chunks (STA-4577). + await withProcessPlatform('darwin', async () => { + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { + stdout: ['100 99 Ss zsh -l', '101 100 S+ claude --model haiku'].join('\n') + } + } + return new Error('unexpected command') + }) + + await expect(getForegroundProcessName(100, '2.1.258')).resolves.toBe('claude') + }) }) it('recognizes SSH relay node-wrapped agents from descendant command lines', async () => { diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index e06edaeabc7..febf025654c 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -20,7 +20,6 @@ import { resolveOuterWrapperForegroundProcess, shouldInspectOuterWrapperForegroundProcess } from '../shared/foreground-wrapper-agent' -import { isShellProcess } from '../shared/shell-process-detection' import { resolveWindowsAgentForegroundProcess, shouldInspectWindowsAgentForeground @@ -305,10 +304,11 @@ export async function getForegroundProcessName( (await resolveWindowsAgentForegroundProcess(pid, fallbackProcess, {})) ?? fallbackProcess ) } - if (!isShellProcess(fallbackProcess) && !isAgentForegroundWrapperProcess(fallbackProcess)) { - return fallbackProcess - } } + // Why: an unrecognized name is not proof of a non-agent foreground -- macOS p_comm truncates + // to the executable basename, which for the native Claude install is its version directory + // (`2.1.258`). The TTL-cached table read resolves the real command line; a foreground that + // is genuinely not an agent still answers with its own name below. const recognized = await getRecognizedForegroundDescendant(pid, fallbackProcess) if (recognized) { return recognized From e3de6b2ce871cd4b094af82c216d1eecef2072e3 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:42:14 -0700 Subject: [PATCH 098/398] Add automation runs dashboard with pagination and filtering (#18226) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add automation runs dashboard with pagination and filtering Adds a new Runs view in the Automations page that lets users browse all runs across automations with status/host filtering, search, and pagination support. Includes virtualized table rendering for efficient handling of large run histories and summary cards showing 24h/7d success/failure counts. * Fix missing dependencies in useCallback hooks and imports Missing dependencies in useCallback can cause stale closure bugs. This adds missing state setters to dependency arrays and consolidates type imports for consistency. * Use keyset pagination for stable automation runs pages Pagination now uses createdAt:id boundaries instead of offsets, so new runs arriving between pages don't shift the window. Maintains backwards compatibility with legacy offset cursors. Move pagination to shared module, fix outcome counting for future-dated runs, and improve hook state tracking on authority re-pairing or target changes. * Extract automation run details to top-level page view Moves run display from detail pane to dedicated page, establishing three-level navigation (Automations → Runs → Run Details) and simplifying the detail pane component. * Fix pagination stability when automation runs share createdAt - Define a stable total order with createdAt and id tiebreaker to prevent runs tied on createdAt from being dropped when the boundary run is pruned between page requests - Retain cursor on failed pagination so pages remain retryable - Update ownerNotice type to AutomationActionNotice * Extract automations list panel and worktree map logic Split AutomationsPageSurface into smaller, focused modules for better maintainability and reusability. Move list panel UI rendering to AutomationsPageListPanel component and worktree map selection logic to a standalone utility function. * Add i18n strings for automation runs dashboard Adds localized strings for the automation runs dashboard view, including search, filtering by host and status, run counts for 24h/7d windows, and empty state messaging across all supported languages. * fix missing translation * fix missing translation --- README.md | 1 + docs/reference/git-compatibility.md | 8 +- .../content/docs/review/annotate-ai-diff.mdx | 7 +- .../daemon/terminal-attach-cancellation.ts | 5 +- .../loading-store/automation-persistence.ts | 10 + .../automation-run-operations.test.ts | 95 ++++++ .../automation-run-operations.ts | 22 +- .../orca-runtime-fence-automation-owner.ts | 17 ++ .../runtime/rpc/methods/automation-schemas.ts | 4 +- .../runtime/rpc/methods/automations.test.ts | 19 ++ src/main/runtime/rpc/methods/automations.ts | 14 +- .../runtime/runtime-automation-controller.ts | 51 ++-- .../runtime/runtime-automation-run-context.ts | 19 ++ .../runtime-automation-update-value.ts | 8 + src/main/runtime/runtime-store-contract.ts | 1 + src/main/worktree-create-preparation-pool.ts | 8 +- .../src/components/GitLabItemDialog.tsx | 2 +- .../automations/AutomationListToolbar.tsx | 38 ++- .../automations/AutomationRunDetailsPage.tsx | 99 ++++++ .../automations/AutomationRunPageFrame.tsx | 2 +- .../automations/AutomationRunsDashboard.tsx | 282 ++++++++++++++++++ .../AutomationRunsDashboardSurface.test.tsx | 77 +++++ .../AutomationRunsDashboardSurface.tsx | 71 +++++ .../automations/AutomationRunsTable.test.tsx | 87 ++++++ .../automations/AutomationRunsTable.tsx | 156 ++++++++++ .../AutomationsDetailPane.run-count.test.tsx | 8 - .../AutomationsDetailPane.test.tsx | 16 - .../automations/AutomationsDetailPane.tsx | 105 +------ .../automations/AutomationsListPanel.test.tsx | 1 + .../automations/AutomationsListPanel.tsx | 5 +- .../AutomationsPageBreadcrumb.test.tsx | 84 ++++++ .../automations/AutomationsPageBreadcrumb.tsx | 76 +++++ .../AutomationsPageDeleteDialogs.tsx | 65 ++++ .../automations/AutomationsPageListPanel.tsx | 125 ++++++++ .../automations/AutomationsPageSurface.tsx | 245 ++++++--------- .../automations/AutomationsPageTopBar.tsx | 65 ++++ .../automations/automation-host-client.ts | 15 + .../automation-owner-action-runner.ts | 16 + .../automations/automation-page-state.ts | 5 + .../automation-row-action-dispatch.ts | 24 +- .../automation-runs-dashboard-model.test.ts | 105 +++++++ .../automation-runs-dashboard-model.ts | 138 +++++++++ .../automation-scoped-list-client.ts | 33 +- .../use-automation-run-page-state.ts | 15 + .../use-automation-runs-dashboard.test.tsx | 144 +++++++++ .../use-automation-runs-dashboard.ts | 188 ++++++++++++ .../use-automations-page-controller.ts | 10 + .../use-automations-page-destination-state.ts | 18 ++ .../use-automations-page-escape.ts | 23 +- .../use-automations-page-local-state.ts | 13 +- .../use-selected-automation-run-history.ts | 20 +- .../use-gitlab-review-actions.ts | 27 +- src/renderer/src/i18n/locales/en.json | 29 ++ src/renderer/src/i18n/locales/es.json | 29 ++ src/renderer/src/i18n/locales/ja.json | 29 ++ src/renderer/src/i18n/locales/ko.json | 29 ++ src/renderer/src/i18n/locales/zh.json | 29 ++ src/shared/automation-run-cursor.ts | 75 +++++ src/shared/automations-types.ts | 6 + tests/e2e/automation-runs-dashboard.spec.ts | 44 +++ 60 files changed, 2612 insertions(+), 350 deletions(-) create mode 100644 src/main/persistence/scheduling-automations/automation-run-operations.test.ts create mode 100644 src/main/runtime/runtime-automation-run-context.ts create mode 100644 src/main/runtime/runtime-automation-update-value.ts create mode 100644 src/renderer/src/components/automations/AutomationRunDetailsPage.tsx create mode 100644 src/renderer/src/components/automations/AutomationRunsDashboard.tsx create mode 100644 src/renderer/src/components/automations/AutomationRunsDashboardSurface.test.tsx create mode 100644 src/renderer/src/components/automations/AutomationRunsDashboardSurface.tsx create mode 100644 src/renderer/src/components/automations/AutomationRunsTable.test.tsx create mode 100644 src/renderer/src/components/automations/AutomationRunsTable.tsx create mode 100644 src/renderer/src/components/automations/AutomationsPageBreadcrumb.test.tsx create mode 100644 src/renderer/src/components/automations/AutomationsPageBreadcrumb.tsx create mode 100644 src/renderer/src/components/automations/AutomationsPageDeleteDialogs.tsx create mode 100644 src/renderer/src/components/automations/AutomationsPageListPanel.tsx create mode 100644 src/renderer/src/components/automations/AutomationsPageTopBar.tsx create mode 100644 src/renderer/src/components/automations/automation-runs-dashboard-model.test.ts create mode 100644 src/renderer/src/components/automations/automation-runs-dashboard-model.ts create mode 100644 src/renderer/src/components/automations/use-automation-runs-dashboard.test.tsx create mode 100644 src/renderer/src/components/automations/use-automation-runs-dashboard.ts create mode 100644 src/shared/automation-run-cursor.ts create mode 100644 tests/e2e/automation-runs-dashboard.spec.ts diff --git a/README.md b/README.md index b486a9a4925..0f99bfc877f 100644 --- a/README.md +++ b/README.md @@ -261,6 +261,7 @@ Want to contribute or run locally? See our [CONTRIBUTING.md](.github/CONTRIBUTIN

    ## Signed Builds + Windows code signing sponored/provided by [SignPath.io](https://signpath.io), certificate by [SignPath Foundation](https://signpath.org). ## License diff --git a/docs/reference/git-compatibility.md b/docs/reference/git-compatibility.md index 0e8b1f257d3..1e19860385e 100644 --- a/docs/reference/git-compatibility.md +++ b/docs/reference/git-compatibility.md @@ -44,14 +44,14 @@ authority. ### Placeholders That Fail Open -`GitCapabilityCache` records commands Git *rejects*. A `git log --format` +`GitCapabilityCache` records commands Git _rejects_. A `git log --format` placeholder Git does not know is not rejected: Git echoes it verbatim and exits zero, so there is no error to remember and no probe to cache. Ask for both forms in one record and pick at parse time. -| Placeholder | Preferred behavior | Compatibility behavior | -| ---------------- | ------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------ | -| `%(decorate:…)` | Git 2.43 separates commit decorations with `\x1f`, so ref names containing commas survive | The same record also carries `%D` (Git 2.10); an unexpanded `%(decorate` placeholder selects it, at the cost of comma-splitting | +| Placeholder | Preferred behavior | Compatibility behavior | +| --------------- | ----------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------- | +| `%(decorate:…)` | Git 2.43 separates commit decorations with `\x1f`, so ref names containing commas survive | The same record also carries `%D` (Git 2.10); an unexpanded `%(decorate` placeholder selects it, at the cost of comma-splitting | ## Why Not `simple-git` diff --git a/docs/site/content/docs/review/annotate-ai-diff.mdx b/docs/site/content/docs/review/annotate-ai-diff.mdx index 6ac3fe4e6a1..0744b2c5a9f 100644 --- a/docs/site/content/docs/review/annotate-ai-diff.mdx +++ b/docs/site/content/docs/review/annotate-ai-diff.mdx @@ -2,11 +2,14 @@ title: Annotate AI Diff --- -import { ImagePlaceholder } from '@/components/docs/prose'; +import { ImagePlaceholder } from '@/components/docs/prose' Annotate AI Diff is Orca's inline review loop for agent-generated code. You leave comments on any line of any AI-generated hunk, then send them back to the agent as a single batch for revision — no copying line numbers, no context-switching. - + ## Leave a comment diff --git a/src/main/daemon/terminal-attach-cancellation.ts b/src/main/daemon/terminal-attach-cancellation.ts index 52f773b1bf2..9e87cbdb1bc 100644 --- a/src/main/daemon/terminal-attach-cancellation.ts +++ b/src/main/daemon/terminal-attach-cancellation.ts @@ -1,10 +1,7 @@ import { TerminalAttachCanceledError } from './daemon-errors' /** Never resolves; only rejects, so it can bound a wait without settling it. */ -export function rejectOnAbort( - signal: AbortSignal | undefined, - sessionId: string -): Promise { +export function rejectOnAbort(signal: AbortSignal | undefined, sessionId: string): Promise { if (!signal) { return new Promise(() => {}) } diff --git a/src/main/persistence/loading-store/automation-persistence.ts b/src/main/persistence/loading-store/automation-persistence.ts index 3e56f2aaf8f..9f33f508fd9 100644 --- a/src/main/persistence/loading-store/automation-persistence.ts +++ b/src/main/persistence/loading-store/automation-persistence.ts @@ -29,6 +29,7 @@ import { import { createAutomationRun as createAutomationRunOperation, listAutomationRuns as listAutomationRunsOperation, + listAutomationRunsPage as listAutomationRunsOperationPage, recordRepeatedAutomationSkip as recordRepeatedAutomationSkipOperation, snapshotAutomationRunWorkspaceDisplayName as snapshotAutomationRunWorkspaceDisplayNameOperation, updateAutomationRun as updateAutomationRunOperation, @@ -125,6 +126,15 @@ export class AutomationPersistence { ) } + listAutomationRunsPage(automationId?: string, limit?: number, cursor?: string) { + return listAutomationRunsOperationPage( + this[automationPersistenceContext].runtime.state, + automationId, + limit, + cursor + ) + } + createAutomation( input: AutomationCreateInput, options?: { destination?: AutomationDestination } diff --git a/src/main/persistence/scheduling-automations/automation-run-operations.test.ts b/src/main/persistence/scheduling-automations/automation-run-operations.test.ts new file mode 100644 index 00000000000..c2ae00c02b6 --- /dev/null +++ b/src/main/persistence/scheduling-automations/automation-run-operations.test.ts @@ -0,0 +1,95 @@ +import { describe, expect, it } from 'vitest' +import type { PersistedState } from '../../../shared/persisted-state-types' +import { listAutomationRunsPage } from './automation-run-operations' + +function stateWithRuns(runs: { id: string; createdAt: number }[]): PersistedState { + return { + automationRuns: runs.map((run) => ({ ...run, automationId: 'a1' })) + } as PersistedState +} + +describe('listAutomationRunsPage', () => { + it('returns a bounded, newest-first page and an opaque continuation cursor', () => { + const state = stateWithRuns([ + { id: 'old', createdAt: 1 }, + { id: 'new', createdAt: 3 }, + { id: 'middle', createdAt: 2 } + ]) + + const first = listAutomationRunsPage(state, 'a1', 2) + expect(first.runs.map((run) => run.id)).toEqual(['new', 'middle']) + expect(first.nextCursor).not.toBeNull() + + expect(listAutomationRunsPage(state, 'a1', 2, first.nextCursor ?? undefined)).toEqual( + expect.objectContaining({ + runs: [expect.objectContaining({ id: 'old' })], + nextCursor: null + }) + ) + }) + + it('keeps the window stable when a newer run lands between pages', () => { + const state = stateWithRuns([ + { id: 'r1', createdAt: 1 }, + { id: 'r2', createdAt: 2 }, + { id: 'r3', createdAt: 3 } + ]) + + const first = listAutomationRunsPage(state, 'a1', 2) + expect(first.runs.map((run) => run.id)).toEqual(['r3', 'r2']) + + state.automationRuns = [ + ...state.automationRuns, + { id: 'r4', automationId: 'a1', createdAt: 4 } as PersistedState['automationRuns'][number] + ] + + const second = listAutomationRunsPage(state, 'a1', 2, first.nextCursor ?? undefined) + expect(second.runs.map((run) => run.id)).toEqual(['r1']) + expect(second.nextCursor).toBeNull() + }) + + it('resumes after a pruned boundary run instead of restarting the page', () => { + const state = stateWithRuns([ + { id: 'r1', createdAt: 1 }, + { id: 'r2', createdAt: 2 }, + { id: 'r3', createdAt: 3 } + ]) + const first = listAutomationRunsPage(state, 'a1', 2) + + state.automationRuns = state.automationRuns.filter((run) => run.id !== 'r2') + + expect( + listAutomationRunsPage(state, 'a1', 2, first.nextCursor ?? undefined).runs.map( + (run) => run.id + ) + ).toEqual(['r1']) + }) + + it('keeps runs tied on createdAt when the boundary run is pruned', () => { + const state = stateWithRuns([ + { id: 'r2', createdAt: 10 }, + { id: 'r1', createdAt: 10 }, + { id: 'r0', createdAt: 5 } + ]) + const first = listAutomationRunsPage(state, 'a1', 1) + expect(first.runs.map((run) => run.id)).toEqual(['r1']) + + state.automationRuns = state.automationRuns.filter((run) => run.id !== 'r1') + + expect( + listAutomationRunsPage(state, 'a1', 2, first.nextCursor ?? undefined).runs.map( + (run) => run.id + ) + ).toEqual(['r2', 'r0']) + }) + + it('still honours a legacy offset cursor issued before the upgrade', () => { + const state = stateWithRuns([ + { id: 'r1', createdAt: 1 }, + { id: 'r2', createdAt: 2 }, + { id: 'r3', createdAt: 3 } + ]) + + expect(listAutomationRunsPage(state, 'a1', 2, '2').runs.map((run) => run.id)).toEqual(['r1']) + }) +}) diff --git a/src/main/persistence/scheduling-automations/automation-run-operations.ts b/src/main/persistence/scheduling-automations/automation-run-operations.ts index 73e3be63e8b..0dc9e61731a 100644 --- a/src/main/persistence/scheduling-automations/automation-run-operations.ts +++ b/src/main/persistence/scheduling-automations/automation-run-operations.ts @@ -5,6 +5,7 @@ import type { Automation, AutomationDispatchResult, AutomationRun, + AutomationRunsPage, AutomationRunTrigger } from '../../../shared/automations-types' import type { PersistedState } from '../../../shared/persisted-state-types' @@ -12,6 +13,10 @@ import { nextAutomationRunNumber, pruneAutomationRuns } from '../../../shared/automation-run-retention' +import { + compareAutomationRunsNewestFirst, + paginateAutomationRuns +} from '../../../shared/automation-run-cursor' import { normalizeAutomationPrecheckResult, normalizeAutomationRunOutputSnapshot, @@ -36,14 +41,27 @@ function touchAutomation(state: PersistedState, automationId: string, now: numbe ) } -export function listAutomationRuns(state: PersistedState, automationId?: string): AutomationRun[] { +function sortedAutomationRuns(state: PersistedState, automationId?: string): AutomationRun[] { const runs = state.automationRuns ?? [] return [...(automationId ? runs.filter((run) => run.automationId === automationId) : runs)] .map((run) => ({ ...run, precheckResult: normalizeAutomationPrecheckResult(run.precheckResult) })) - .sort((left, right) => right.createdAt - left.createdAt) + .sort(compareAutomationRunsNewestFirst) +} + +export function listAutomationRuns(state: PersistedState, automationId?: string): AutomationRun[] { + return sortedAutomationRuns(state, automationId) +} + +export function listAutomationRunsPage( + state: PersistedState, + automationId: string | undefined, + limit = 100, + cursor?: string +): AutomationRunsPage { + return paginateAutomationRuns(sortedAutomationRuns(state, automationId), limit, cursor) } export function createAutomationRun( diff --git a/src/main/runtime/orca-runtime-fence-automation-owner.ts b/src/main/runtime/orca-runtime-fence-automation-owner.ts index 626d697716a..a90730c7886 100644 --- a/src/main/runtime/orca-runtime-fence-automation-owner.ts +++ b/src/main/runtime/orca-runtime-fence-automation-owner.ts @@ -53,6 +53,23 @@ export class OrcaRuntimeWithFenceAutomationOwner extends OrcaRuntimeWithPtyForeg }) } + listAutomationRunsPage( + automationId?: string, + expectedOwner?: AutomationOwnerPrecondition, + limit?: number, + cursor?: string + ) { + if (expectedOwner && !automationId) { + throw new Error('An expected owner requires an automation id.') + } + return this.automation.withExternalProbePriority(() => { + if (automationId) { + this.fenceAutomationOwner(automationId, expectedOwner, 'read') + } + return this.automation.listRunsPage(automationId, limit, cursor) + }) + } + showAutomation(id: string, expectedOwner?: AutomationOwnerPrecondition): Automation { const automation = this.automation.show(id) this.fenceAutomationOwner(id, expectedOwner, 'read') diff --git a/src/main/runtime/rpc/methods/automation-schemas.ts b/src/main/runtime/rpc/methods/automation-schemas.ts index 5e5dbe4e4ad..f2c829c1a9d 100644 --- a/src/main/runtime/rpc/methods/automation-schemas.ts +++ b/src/main/runtime/rpc/methods/automation-schemas.ts @@ -137,7 +137,9 @@ export const AutomationId = z.object({ export const AutomationRuns = z.object({ automationId: OptionalString, - expectedOwner: ExpectedOwner + expectedOwner: ExpectedOwner, + limit: OptionalPositiveInt, + cursor: OptionalString }) export const AutomationCreate = z.object({ diff --git a/src/main/runtime/rpc/methods/automations.test.ts b/src/main/runtime/rpc/methods/automations.test.ts index d972bf6a553..ab768559749 100644 --- a/src/main/runtime/rpc/methods/automations.test.ts +++ b/src/main/runtime/rpc/methods/automations.test.ts @@ -105,6 +105,25 @@ describe('automation RPC methods', () => { expect(runtime.listAutomationRuns).toHaveBeenCalledWith('auto-1', undefined) }) + it('returns a cursor page when the caller requests a bounded run history', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + listAutomationRunsPage: vi.fn().mockReturnValue({ + runs: [{ id: 'run-100', automationId: 'auto-1' }], + nextCursor: '100' + }) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: AUTOMATION_METHODS }) + + await expect( + dispatcher.dispatch(makeRequest('automation.runs', { automationId: 'auto-1', limit: 100 })) + ).resolves.toMatchObject({ + ok: true, + result: { nextCursor: '100' } + }) + expect(runtime.listAutomationRunsPage).toHaveBeenCalledWith('auto-1', undefined, 100, undefined) + }) + it('rejects unknown providers and invalid schedules', async () => { const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/rpc/methods/automations.ts b/src/main/runtime/rpc/methods/automations.ts index 3645df3c149..ff5daca315c 100644 --- a/src/main/runtime/rpc/methods/automations.ts +++ b/src/main/runtime/rpc/methods/automations.ts @@ -83,8 +83,16 @@ export const AUTOMATION_METHODS: RpcMethod[] = [ defineMethod({ name: 'automation.runs', params: AutomationRuns, - handler: (params, { runtime }) => ({ - runs: runtime.listAutomationRuns(params.automationId, params.expectedOwner) - }) + handler: (params, { runtime }) => { + if (params.limit !== undefined || params.cursor !== undefined) { + return runtime.listAutomationRunsPage( + params.automationId, + params.expectedOwner, + params.limit, + params.cursor + ) + } + return { runs: runtime.listAutomationRuns(params.automationId, params.expectedOwner) } + } }) ] diff --git a/src/main/runtime/runtime-automation-controller.ts b/src/main/runtime/runtime-automation-controller.ts index c8c8b1504eb..d3ac55df438 100644 --- a/src/main/runtime/runtime-automation-controller.ts +++ b/src/main/runtime/runtime-automation-controller.ts @@ -15,6 +15,9 @@ import type { AutomationDestination } from '../../shared/automation-owner-precondition' import { runAutomationNowFenced } from '../automations/refused-manual-run' +import { paginateAutomationRuns } from '../../shared/automation-run-cursor' +import { hasRuntimeAutomationUpdateValue } from './runtime-automation-update-value' +import { assertAutomationRunContextMatchesTarget } from './runtime-automation-run-context' export type RuntimeAutomationCreateInput = Omit< AutomationCreateInput, @@ -72,6 +75,13 @@ export class RuntimeAutomationController { return this.store.listAutomationRuns(automationId) } + listRunsPage(automationId?: string, limit?: number, cursor?: string) { + if (this.store?.listAutomationRunsPage) { + return this.store.listAutomationRunsPage(automationId, limit, cursor) + } + return paginateAutomationRuns(this.listRuns(automationId), limit, cursor) + } + listForScope(params: AutomationListParams = {}): AutomationListResult { if (!this.store?.listAutomationsForScope) { throw new Error('runtime_unavailable') @@ -99,7 +109,7 @@ export class RuntimeAutomationController { throw new Error('runtime_unavailable') } const target = await this.resolveTarget(input) - this.assertRunContextMatchesTarget(input.runContext, target.repo) + assertAutomationRunContextMatchesTarget(input.runContext, target.repo) if (input.reuseSession && target.workspaceMode !== 'existing') { throw new Error('Session reuse requires an existing workspace target.') } @@ -142,12 +152,12 @@ export class RuntimeAutomationController { const patch: AutomationUpdateInput = {} this.copyPatchValues(updates, patch) const targetChanged = - hasUpdateValue(updates, 'repo') || - hasUpdateValue(updates, 'workspace') || - hasUpdateValue(updates, 'workspaceMode') + hasRuntimeAutomationUpdateValue(updates, 'repo') || + hasRuntimeAutomationUpdateValue(updates, 'workspace') || + hasRuntimeAutomationUpdateValue(updates, 'workspaceMode') if (targetChanged) { const target = await this.resolveTarget(updates, current) - this.assertRunContextMatchesTarget(updates.runContext, target.repo) + assertAutomationRunContextMatchesTarget(updates.runContext, target.repo) if (patch.reuseSession === true && target.workspaceMode !== 'existing') { throw new Error('Session reuse requires an existing workspace target.') } @@ -158,9 +168,13 @@ export class RuntimeAutomationController { patch.reuseSession = false } } - if (!targetChanged && hasUpdateValue(updates, 'runContext') && current.projectId) { + if ( + !targetChanged && + hasRuntimeAutomationUpdateValue(updates, 'runContext') && + current.projectId + ) { const repo = await this.resolvers.showRepo(`id:${current.projectId}`) - this.assertRunContextMatchesTarget(updates.runContext, repo) + assertAutomationRunContextMatchesTarget(updates.runContext, repo) } if (!targetChanged && patch.reuseSession && current.workspaceMode !== 'existing') { throw new Error('Session reuse requires an existing workspace target.') @@ -225,7 +239,7 @@ export class RuntimeAutomationController { 'missedRunGraceMinutes' ] as const for (const key of keys) { - if (hasUpdateValue(updates, key)) { + if (hasRuntimeAutomationUpdateValue(updates, key)) { Object.assign(patch, { [key]: updates[key] }) } } @@ -292,25 +306,4 @@ export class RuntimeAutomationController { } return { projectId, workspaceMode: 'new_per_run', workspaceId: null, repo } } - - private assertRunContextMatchesTarget( - runContext: - | RuntimeAutomationCreateInput['runContext'] - | RuntimeAutomationUpdateInput['runContext'], - repo: Repo | null - ): void { - if (!runContext || !repo) { - return - } - if (runContext.repoId !== repo.id || runContext.path !== repo.path) { - throw new Error('Automation project does not match its run context.') - } - } -} - -function hasUpdateValue( - updates: RuntimeAutomationUpdateInput, - key: K -): boolean { - return Object.hasOwn(updates, key) && updates[key] !== undefined } diff --git a/src/main/runtime/runtime-automation-run-context.ts b/src/main/runtime/runtime-automation-run-context.ts new file mode 100644 index 00000000000..f69f1771156 --- /dev/null +++ b/src/main/runtime/runtime-automation-run-context.ts @@ -0,0 +1,19 @@ +import type { Repo } from '../../shared/repo-types' +import type { + RuntimeAutomationCreateInput, + RuntimeAutomationUpdateInput +} from './runtime-automation-controller' + +export function assertAutomationRunContextMatchesTarget( + runContext: + | RuntimeAutomationCreateInput['runContext'] + | RuntimeAutomationUpdateInput['runContext'], + repo: Repo | null +): void { + if (!runContext || !repo) { + return + } + if (runContext.repoId !== repo.id || runContext.path !== repo.path) { + throw new Error('Automation project does not match its run context.') + } +} diff --git a/src/main/runtime/runtime-automation-update-value.ts b/src/main/runtime/runtime-automation-update-value.ts new file mode 100644 index 00000000000..84f5a166c7b --- /dev/null +++ b/src/main/runtime/runtime-automation-update-value.ts @@ -0,0 +1,8 @@ +import type { RuntimeAutomationUpdateInput } from './runtime-automation-controller' + +export function hasRuntimeAutomationUpdateValue( + updates: RuntimeAutomationUpdateInput, + key: K +): boolean { + return Object.hasOwn(updates, key) && updates[key] !== undefined +} diff --git a/src/main/runtime/runtime-store-contract.ts b/src/main/runtime/runtime-store-contract.ts index e160e039533..aece6a8e7ad 100644 --- a/src/main/runtime/runtime-store-contract.ts +++ b/src/main/runtime/runtime-store-contract.ts @@ -61,6 +61,7 @@ export type RuntimeStore = { automationOwnerPrecondition?: Store['automationOwnerPrecondition'] automationChangeSelector?: Store['automationChangeSelector'] listAutomationRuns?: Store['listAutomationRuns'] + listAutomationRunsPage?: Store['listAutomationRunsPage'] createAutomation?: Store['createAutomation'] updateAutomation?: Store['updateAutomation'] deleteAutomation?: Store['deleteAutomation'] diff --git a/src/main/worktree-create-preparation-pool.ts b/src/main/worktree-create-preparation-pool.ts index 26ee1c41aad..7539c6076e4 100644 --- a/src/main/worktree-create-preparation-pool.ts +++ b/src/main/worktree-create-preparation-pool.ts @@ -183,7 +183,13 @@ export function startPreparation({ await cleanupStalePreparations(preparationHostKey(repoPathKey, wslDistro), repoPath, options) await mkdir(toHostFilesystemPath(preparationRoot), { recursive: true }) // Already canonical, so the add re-resolves nothing. - await prepareWorktreeCreateCheckout(repoPath, preparedPath, canonicalBase, lockReason, options) + await prepareWorktreeCreateCheckout( + repoPath, + preparedPath, + canonicalBase, + lockReason, + options + ) })() } satisfies PreparationEntry) preparations.set(key, entry) diff --git a/src/renderer/src/components/GitLabItemDialog.tsx b/src/renderer/src/components/GitLabItemDialog.tsx index 1cae8854a08..4c911520ed1 100644 --- a/src/renderer/src/components/GitLabItemDialog.tsx +++ b/src/renderer/src/components/GitLabItemDialog.tsx @@ -47,7 +47,7 @@ export default function GitLabItemDialog({ const handleRefresh = useCallback(() => { setRefreshNonce((n) => n + 1) - }, []) + }, [setRefreshNonce]) const detailsEditing = useGitLabDetailsEditing(item, repoSelector, state) const pipelineActions = useGitLabPipelineActions(item, repoSelector, state, handleRefresh) const reviewActions = useGitLabReviewActions(item, repoSelector, state) diff --git a/src/renderer/src/components/automations/AutomationListToolbar.tsx b/src/renderer/src/components/automations/AutomationListToolbar.tsx index c5efed566f6..259f58fba79 100644 --- a/src/renderer/src/components/automations/AutomationListToolbar.tsx +++ b/src/renderer/src/components/automations/AutomationListToolbar.tsx @@ -1,5 +1,5 @@ import React from 'react' -import { Plus, RefreshCw } from 'lucide-react' +import { History, Plus, RefreshCw } from 'lucide-react' import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { translate } from '@/i18n/i18n' @@ -25,6 +25,7 @@ type AutomationListToolbarProps = { hostEntries: readonly AutomationHostCatalogEntry[] onRefresh: () => void isRefreshing: boolean + onOpenRuns: () => void openCreateDialog: (template?: AutomationTemplate) => void canCreateAutomation: boolean } @@ -41,6 +42,7 @@ export function AutomationListToolbar({ hostEntries, onRefresh, isRefreshing, + onOpenRuns, openCreateDialog, canCreateAutomation }: AutomationListToolbarProps): React.JSX.Element { @@ -88,17 +90,29 @@ export function AutomationListToolbar({
    - +
    + + +
    ) } diff --git a/src/renderer/src/components/automations/AutomationRunDetailsPage.tsx b/src/renderer/src/components/automations/AutomationRunDetailsPage.tsx new file mode 100644 index 00000000000..63bdf933434 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationRunDetailsPage.tsx @@ -0,0 +1,99 @@ +import React from 'react' +import { Eye, RefreshCw } from 'lucide-react' +import { Button } from '@/components/ui/button' +import CommentMarkdown from '@/components/sidebar/CommentMarkdown' +import { cn } from '@/lib/utils' +import { translate } from '@/i18n/i18n' +import type { Automation, AutomationRun } from '../../../../shared/automations-types' +import { AutomationRunPageFrame } from './AutomationRunPageFrame' +import { getAutomationRunContent } from './automation-run-content' +import type { AutomationRunViewState } from './automation-run-view-state' +import type { AutomationRunWorkspaceDisplay } from './automation-run-workspace-display' +import { + formatAutomationDateTimeWithRelative, + getAutomationRunStatusLabel, + getAutomationRunStatusVariant +} from './automation-page-parts' + +export function AutomationRunDetailsPage({ + automation, + run, + relativeNow, + workspaceDisplay, + viewState, + canRerun, + isRerunPending, + onRerun, + onOpenWorkspace, + onBack +}: { + automation: Automation | null + run: AutomationRun + relativeNow: number + workspaceDisplay: AutomationRunWorkspaceDisplay | null + viewState: AutomationRunViewState | null + canRerun: boolean + isRerunPending: boolean + onRerun: () => void + onOpenWorkspace: () => void + onBack: () => void +}): React.JSX.Element { + return ( +
    + + {canRerun && automation ? ( + + ) : null} + {viewState ? ( + + ) : null} + + } + onBack={onBack} + > + + +
    + ) +} diff --git a/src/renderer/src/components/automations/AutomationRunPageFrame.tsx b/src/renderer/src/components/automations/AutomationRunPageFrame.tsx index 30c323d6dc8..e51c94155e7 100644 --- a/src/renderer/src/components/automations/AutomationRunPageFrame.tsx +++ b/src/renderer/src/components/automations/AutomationRunPageFrame.tsx @@ -26,7 +26,7 @@ export function AutomationRunPageFrame({ onBack }: AutomationRunPageFrameProps): React.JSX.Element { return ( -
    +
    diff --git a/src/renderer/src/components/automations/AutomationRunsDashboard.tsx b/src/renderer/src/components/automations/AutomationRunsDashboard.tsx new file mode 100644 index 00000000000..1271a1d5e7e --- /dev/null +++ b/src/renderer/src/components/automations/AutomationRunsDashboard.tsx @@ -0,0 +1,282 @@ +import React, { useDeferredValue, useMemo, useState } from 'react' +import { AlertCircle, ListFilter, RefreshCw, Search } from 'lucide-react' +import { Button } from '@/components/ui/button' +import { + DropdownMenu, + DropdownMenuCheckboxItem, + DropdownMenuContent, + DropdownMenuRadioGroup, + DropdownMenuRadioItem, + DropdownMenuSeparator, + DropdownMenuSub, + DropdownMenuSubContent, + DropdownMenuSubTrigger, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Input } from '@/components/ui/input' +import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' +import { translate } from '@/i18n/i18n' +import type { AutomationListRow } from './automation-list-row-identity' +import { + countAutomationRunOutcomes, + filterAutomationRunsDashboardEntries, + getAutomationRunsHostKey, + getAutomationRunsScope, + type AutomationRunsDashboardEntry, + type AutomationRunsDashboardFailure, + type AutomationRunsStatusFilter +} from './automation-runs-dashboard-model' +import { AutomationRunsTable } from './AutomationRunsTable' + +const STATUS_LABELS: Record = { + all: { key: 'allStatuses', fallback: 'All statuses' }, + successful: { key: 'successful', fallback: 'Successful' }, + failed: { key: 'failed', fallback: 'Failed' }, + active: { key: 'active', fallback: 'In progress' }, + skipped: { key: 'skipped', fallback: 'Skipped' } +} + +function SummaryCard({ label, value }: { label: string; value: number }): React.JSX.Element { + return ( +
    +
    {label}
    +
    {value}
    +
    + ) +} + +export function AutomationRunsDashboard({ + rows, + entries, + failures, + loading, + hasMore, + onLoadMore, + now, + onRefresh, + onOpenRun +}: { + rows: readonly AutomationListRow[] + entries: readonly AutomationRunsDashboardEntry[] + failures: readonly AutomationRunsDashboardFailure[] + loading: boolean + hasMore: boolean + onLoadMore: () => void + now: number + onRefresh: () => void + onOpenRun: (entry: AutomationRunsDashboardEntry) => void +}): React.JSX.Element { + const [query, setQuery] = useState('') + const deferredQuery = useDeferredValue(query) + const [status, setStatus] = useState('all') + const [hostKeys, setHostKeys] = useState([]) + const hostOptions = useMemo(() => { + const options = new Map() + for (const row of rows) { + const scope = getAutomationRunsScope(row) + options.set( + getAutomationRunsHostKey(row), + row.hostLabel || + translate( + `auto.components.automations.AutomationRunsDashboard.${scope}`, + scope === 'local' ? 'Local' : 'Remote' + ) + ) + } + return [...options].map(([key, label]) => ({ key, label })) + }, [rows]) + const hostEntries = useMemo( + () => filterAutomationRunsDashboardEntries({ entries, status: 'all', query: '', hostKeys }), + [entries, hostKeys] + ) + const visibleEntries = useMemo( + () => filterAutomationRunsDashboardEntries({ entries, status, query: deferredQuery, hostKeys }), + [deferredQuery, entries, hostKeys, status] + ) + const counts = useMemo(() => countAutomationRunOutcomes(hostEntries, now), [hostEntries, now]) + const visibleFailures = failures.filter( + (failure) => hostKeys.length === 0 || hostKeys.includes(getAutomationRunsHostKey(failure.row)) + ) + const activeFilterCount = (status === 'all' ? 0 : 1) + (hostKeys.length > 0 ? 1 : 0) + + const toggleHost = (hostKey: string): void => { + setHostKeys((current) => + current.includes(hostKey) + ? current.filter((candidate) => candidate !== hostKey) + : [...current, hostKey] + ) + } + + return ( +
    +
    +
    +
    + + setQuery(event.target.value)} + placeholder={translate( + 'auto.components.automations.AutomationRunsDashboard.search', + 'Search runs…' + )} + aria-label={translate( + 'auto.components.automations.AutomationRunsDashboard.search', + 'Search runs…' + )} + className="h-8 pl-8 text-xs" + /> +
    + + + + + + + + {translate('auto.components.automations.AutomationRunsDashboard.host', 'Host')} + + + setHostKeys([])} + > + {translate('auto.components.automations.hostPicker.allHosts', 'All hosts')} + + {hostOptions.map((host) => ( + toggleHost(host.key)} + onSelect={(event) => event.preventDefault()} + > + {host.label} + + ))} + + + + + + {translate( + 'auto.components.automations.AutomationRunsDashboard.status', + 'Status' + )} + + + setStatus(value as AutomationRunsStatusFilter)} + > + {Object.entries(STATUS_LABELS).map(([value, label]) => ( + + {translate( + `auto.components.automations.AutomationRunsDashboard.status.${label.key}`, + label.fallback + )} + + ))} + + + + + + + + + + + {translate( + 'auto.components.automations.AutomationRunsDashboard.refresh', + 'Refresh runs' + )} + + +
    + +
    +
    + + + + +
    + + {visibleFailures.length > 0 ? ( +
    + + + {visibleFailures.length === 1 + ? translate( + 'auto.components.automations.AutomationRunsDashboard.historyUnavailableOne', + 'Run history is unavailable for 1 automation. Counts include available history only.' + ) + : translate( + 'auto.components.automations.AutomationRunsDashboard.historyUnavailableMany', + 'Run history is unavailable for {{count}} automations. Counts include available history only.', + { count: visibleFailures.length } + )} + +
    + ) : null} + + +
    +
    +
    + ) +} diff --git a/src/renderer/src/components/automations/AutomationRunsDashboardSurface.test.tsx b/src/renderer/src/components/automations/AutomationRunsDashboardSurface.test.tsx new file mode 100644 index 00000000000..8da19eca24a --- /dev/null +++ b/src/renderer/src/components/automations/AutomationRunsDashboardSurface.test.tsx @@ -0,0 +1,77 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot } from 'react-dom/client' +import { describe, expect, it, vi } from 'vitest' +import type { AutomationRunsDashboardEntry } from './automation-runs-dashboard-model' +import { makeAutomationListRow, makeRun } from './automations-page-fixtures' + +let openRun: ((entry: AutomationRunsDashboardEntry) => void) | null = null + +vi.mock('./AutomationRunsDashboard', () => ({ + AutomationRunsDashboard: (props: { + entries: readonly AutomationRunsDashboardEntry[] + onOpenRun: (entry: AutomationRunsDashboardEntry) => void + }) => { + openRun = props.onOpenRun + return null + } +})) + +import { AutomationRunsDashboardSurface } from './AutomationRunsDashboardSurface' + +describe('AutomationRunsDashboardSurface', () => { + it('opens a run as a top-level page', () => { + const row = makeAutomationListRow() + const run = makeRun() + const entry: AutomationRunsDashboardEntry = { + key: `${row.key}:${run.id}`, + hostKey: 'desktop:self', + searchText: 'nightly', + row, + run, + scope: 'local' + } + const setPageView = vi.fn() + const setRunPageOrigin = vi.fn() + const selectAutomationRow = vi.fn() + const setPendingAutomationRunNavigation = vi.fn() + const setIsDetailOpen = vi.fn() + const container = document.createElement('div') + const root = createRoot(container) + + act(() => { + root.render( + + ) + }) + + act(() => openRun?.(entry)) + + expect(setPageView).toHaveBeenCalledWith('run') + expect(setRunPageOrigin).toHaveBeenCalledWith('runs') + expect(selectAutomationRow).toHaveBeenCalledWith(row.key) + expect(setPendingAutomationRunNavigation).toHaveBeenCalledWith({ + automationId: row.automation.id, + runId: run.id, + hostId: undefined + }) + expect(setIsDetailOpen).toHaveBeenCalledWith(true) + + act(() => root.unmount()) + }) +}) diff --git a/src/renderer/src/components/automations/AutomationRunsDashboardSurface.tsx b/src/renderer/src/components/automations/AutomationRunsDashboardSurface.tsx new file mode 100644 index 00000000000..2cfe6edc2a9 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationRunsDashboardSurface.tsx @@ -0,0 +1,71 @@ +import React from 'react' +import { toRuntimeExecutionHostId, type ExecutionHostId } from '../../../../shared/execution-host' +import type { AutomationListRow } from './automation-list-row-identity' +import type { + AutomationRunsDashboardEntry, + AutomationRunsDashboardFailure +} from './automation-runs-dashboard-model' +import { AutomationRunsDashboard } from './AutomationRunsDashboard' +import type { AutomationsPageView } from './automation-page-state' + +export function AutomationRunsDashboardSurface({ + rows, + entries, + failures, + loading, + hasMore, + onLoadMore, + now, + onRefresh, + setPageView, + setRunPageOrigin, + selectAutomationRow, + setPendingAutomationRunNavigation, + setIsDetailOpen +}: { + rows: readonly AutomationListRow[] + entries: readonly AutomationRunsDashboardEntry[] + failures: readonly AutomationRunsDashboardFailure[] + loading: boolean + hasMore: boolean + onLoadMore: () => void + now: number + onRefresh: () => void + setPageView: (view: AutomationsPageView) => void + setRunPageOrigin: (origin: 'runs' | 'automation') => void + selectAutomationRow: (rowKey: string | null) => void + setPendingAutomationRunNavigation: (navigation: { + automationId: string + runId: string | null + hostId?: ExecutionHostId + }) => void + setIsDetailOpen: (open: boolean) => void +}): React.JSX.Element { + return ( + { + const authority = entry.row.catalogRef?.authority + setRunPageOrigin('runs') + setPageView('run') + selectAutomationRow(entry.row.key) + setPendingAutomationRunNavigation({ + automationId: entry.row.automation.id, + runId: entry.run.id, + hostId: + authority?.kind === 'runtime' + ? toRuntimeExecutionHostId(authority.environmentId) + : undefined + }) + setIsDetailOpen(true) + }} + /> + ) +} diff --git a/src/renderer/src/components/automations/AutomationRunsTable.test.tsx b/src/renderer/src/components/automations/AutomationRunsTable.test.tsx new file mode 100644 index 00000000000..b5c8a70aa8b --- /dev/null +++ b/src/renderer/src/components/automations/AutomationRunsTable.test.tsx @@ -0,0 +1,87 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { Automation, AutomationRun } from '../../../../shared/automations-types' +import type { AutomationRunsDashboardEntry } from './automation-runs-dashboard-model' +import { AutomationRunsTable } from './AutomationRunsTable' + +vi.mock('@tanstack/react-virtual', () => ({ + useVirtualizer: ({ + count, + getItemKey + }: { + count: number + getItemKey: (index: number) => string + }) => ({ + getTotalSize: () => count * 59, + getVirtualItems: () => + Array.from({ length: Math.min(count, 21) }, (_, index) => ({ + index, + key: getItemKey(index), + start: index * 59 + })), + measureElement: () => undefined + }) +})) + +function entries(count: number): AutomationRunsDashboardEntry[] { + const automation = { id: 'automation', name: 'Daily check' } as Automation + const row = { + key: 'row', + automation, + catalogRef: { authority: { kind: 'desktop' }, selector: { kind: 'self' } }, + hostLabel: 'Local Mac', + usageSummary: null + } as const + return Array.from({ length: count }, (_, index) => ({ + key: `row:run-${index}`, + hostKey: 'desktop:self', + searchText: `daily check run ${index} local mac`, + row, + run: { + id: `run-${index}`, + automationId: automation.id, + title: `Run ${index}`, + scheduledFor: index, + trigger: 'scheduled', + status: 'completed' + } as AutomationRun, + scope: 'local' + })) +} + +describe('AutomationRunsTable virtualization', () => { + let container: HTMLDivElement + let root: Root + + beforeEach(() => { + globalThis.IS_REACT_ACT_ENVIRONMENT = true + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) + }) + + afterEach(() => { + act(() => root.unmount()) + container.remove() + }) + + it('keeps a 10,000-run history to a bounded number of mounted rows', () => { + act(() => { + root.render( + {}} + onOpenRun={() => {}} + /> + ) + }) + + const mountedRows = container.querySelectorAll('[data-testid="automation-runs-row"]') + expect(mountedRows).toHaveLength(21) + }) +}) diff --git a/src/renderer/src/components/automations/AutomationRunsTable.tsx b/src/renderer/src/components/automations/AutomationRunsTable.tsx new file mode 100644 index 00000000000..e4ec4f565eb --- /dev/null +++ b/src/renderer/src/components/automations/AutomationRunsTable.tsx @@ -0,0 +1,156 @@ +import React, { useEffect, useRef } from 'react' +import { useVirtualizer } from '@tanstack/react-virtual' +import { ChevronRight, Loader2 } from 'lucide-react' +import { Badge } from '@/components/ui/badge' +import { translate } from '@/i18n/i18n' +import type { AutomationRunsDashboardEntry } from './automation-runs-dashboard-model' +import { + formatAutomationDateTimeWithRelative, + getAutomationRunStatusLabel, + getAutomationRunStatusVariant +} from './automation-page-parts' + +const RUN_ROW_HEIGHT_PX = 59 +const RUN_ROW_OVERSCAN = 10 +const RUNS_VIEWPORT_INITIAL_RECT = { width: 1024, height: 600 } + +export function AutomationRunsTable({ + entries, + loading, + hasMore, + onLoadMore, + onOpenRun +}: { + entries: readonly AutomationRunsDashboardEntry[] + loading: boolean + hasMore: boolean + onLoadMore: () => void + onOpenRun: (entry: AutomationRunsDashboardEntry) => void +}): React.JSX.Element { + const scrollRef = useRef(null) + const loadMoreRequestedRef = useRef(false) + useEffect(() => { + if (!loading) { + loadMoreRequestedRef.current = false + } + }, [loading]) + const virtualizer = useVirtualizer({ + count: entries.length, + getScrollElement: () => scrollRef.current, + estimateSize: () => RUN_ROW_HEIGHT_PX, + overscan: RUN_ROW_OVERSCAN, + initialRect: RUNS_VIEWPORT_INITIAL_RECT, + getItemKey: (index) => entries[index]?.key ?? index + }) + + return ( +
    +
    +
    + {translate( + 'auto.components.automations.AutomationRunsDashboard.automation', + 'Automation' + )} +
    +
    + {translate('auto.components.automations.AutomationRunsDashboard.triggered', 'Triggered')} +
    +
    + {translate('auto.components.automations.AutomationRunsDashboard.trigger', 'Trigger')} +
    +
    {translate('auto.components.automations.AutomationRunsDashboard.host', 'Host')}
    +
    + {translate('auto.components.automations.AutomationRunsDashboard.status', 'Status')} +
    +
    +
    { + const { clientHeight, scrollHeight, scrollTop } = event.currentTarget + const nearEnd = scrollHeight - scrollTop - clientHeight < RUN_ROW_HEIGHT_PX * 10 + if (hasMore && !loading && nearEnd && !loadMoreRequestedRef.current) { + loadMoreRequestedRef.current = true + onLoadMore() + } + }} + > + {loading && entries.length === 0 ? ( +
    + + {translate( + 'auto.components.automations.AutomationRunsDashboard.loading', + 'Loading runs…' + )} +
    + ) : entries.length === 0 ? ( +
    +
    + {translate( + 'auto.components.automations.AutomationRunsDashboard.noRuns', + 'No runs yet' + )} +
    +
    + {translate( + 'auto.components.automations.AutomationRunsDashboard.emptyDescription', + 'Runs appear here after an automation is triggered.' + )} +
    +
    + ) : ( +
    + {virtualizer.getVirtualItems().map((virtualRow) => { + const entry = entries[virtualRow.index] + if (!entry) { + return null + } + return ( +
    + +
    + ) + })} +
    + )} +
    +
    + ) +} diff --git a/src/renderer/src/components/automations/AutomationsDetailPane.run-count.test.tsx b/src/renderer/src/components/automations/AutomationsDetailPane.run-count.test.tsx index 740f1130405..3d52b2335d6 100644 --- a/src/renderer/src/components/automations/AutomationsDetailPane.run-count.test.tsx +++ b/src/renderer/src/components/automations/AutomationsDetailPane.run-count.test.tsx @@ -28,7 +28,6 @@ async function renderRunsTab(historyUnavailable: boolean): Promise []} onActivePaneTabChange={onActivePaneTabChange} onClearExternalRunPage={() => undefined} - onClearAutomationRunPage={() => undefined} requestExternalAction={() => undefined} openExternalRunPage={() => undefined} openEditExternalDialog={() => undefined} @@ -75,8 +69,6 @@ function renderDetailPane(options: { openEditDialog={() => undefined} toggleAutomation={() => undefined} requestDeleteAutomation={() => undefined} - rerunAutomationRun={() => undefined} - openRunWorkspace={() => undefined} openAutomationRunPage={() => undefined} onBackToList={() => undefined} recoverSelectedRuns={() => undefined} @@ -178,7 +170,6 @@ describe('AutomationsDetailPane tab keyboard navigation', () => { selected={selected} selectedExternal={null} selectedExternalRunPage={null} - selectedAutomationRunPage={null} selectedRuns={[]} selectedRunsNotice={null} activePaneTab="overview" @@ -190,15 +181,10 @@ describe('AutomationsDetailPane tab keyboard navigation', () => { selectedHostEntry={null} hostLabelById={new Map()} selectedRunNowAvailability={null} - selectedAutomationRunPageWorkspaceDisplay={null} - selectedAutomationRunPageViewState={null} - canRerunSelectedAutomationRunPage={false} - isSelectedAutomationRunPageRerunPending={false} worktreeMap={new Map()} fetchExternalAutomationRuns={async () => []} onActivePaneTabChange={() => undefined} onClearExternalRunPage={() => undefined} - onClearAutomationRunPage={() => undefined} requestExternalAction={() => undefined} openExternalRunPage={() => undefined} openEditExternalDialog={() => undefined} @@ -206,8 +192,6 @@ describe('AutomationsDetailPane tab keyboard navigation', () => { openEditDialog={() => undefined} toggleAutomation={() => undefined} requestDeleteAutomation={() => undefined} - rerunAutomationRun={() => undefined} - openRunWorkspace={() => undefined} openAutomationRunPage={() => undefined} onBackToList={onBackToList} recoverSelectedRuns={() => undefined} diff --git a/src/renderer/src/components/automations/AutomationsDetailPane.tsx b/src/renderer/src/components/automations/AutomationsDetailPane.tsx index e0547174f33..72c60463ea3 100644 --- a/src/renderer/src/components/automations/AutomationsDetailPane.tsx +++ b/src/renderer/src/components/automations/AutomationsDetailPane.tsx @@ -1,8 +1,7 @@ import React from 'react' -import { ArrowLeft, Eye, RefreshCw } from 'lucide-react' +import { ArrowLeft } from 'lucide-react' import { Button } from '@/components/ui/button' import { Tabs, TabsContent, TabsList, TabsTrigger } from '@/components/ui/tabs' -import { cn } from '@/lib/utils' import type { Automation, ExternalAutomationAction, @@ -12,7 +11,6 @@ import type { AutomationRun } from '../../../../shared/automations-types' import type { Worktree } from '../../../../shared/worktree/types' -import CommentMarkdown from '@/components/sidebar/CommentMarkdown' import { AutomationDetail } from './AutomationDetail' import { HermesCronOutputView } from './HermesCronOutputView' import { AutomationRunPageFrame } from './AutomationRunPageFrame' @@ -28,18 +26,10 @@ import { getExternalRunStatusLabel, getExternalRunStatusVariant } from './external-automation-display' -import { - formatAutomationDateTimeWithRelative, - getAutomationRunStatusLabel, - getAutomationRunStatusVariant -} from './automation-page-parts' -import { getAutomationRunContent } from './automation-run-content' import type { AutomationActionNotice } from './automation-row-action-dispatch' import type { AutomationHostRecoveryAction } from './automation-host-status-descriptors' import type { AutomationHostCatalogEntry } from './automation-host-catalog-types' import type { AutomationTargetAvailability } from './automation-target-availability' -import type { AutomationRunViewState } from './automation-run-view-state' -import type { AutomationRunWorkspaceDisplay } from './automation-run-workspace-display' import type { AutomationPaneTab, SelectedExternalRunPage } from './automation-page-state' import { getAutomationDetailNextTab, @@ -52,7 +42,6 @@ type AutomationsDetailPaneProps = { selected: Automation | null selectedExternal: ExternalAutomationListEntry | null selectedExternalRunPage: SelectedExternalRunPage | null - selectedAutomationRunPage: AutomationRun | null selectedRuns: AutomationRun[] /** Set when the selected automation's history read failed; its runs are unknown. */ selectedRunsNotice: AutomationActionNotice | null @@ -66,15 +55,10 @@ type AutomationsDetailPaneProps = { selectedHostEntry: AutomationHostCatalogEntry | null hostLabelById: ReadonlyMap selectedRunNowAvailability: AutomationTargetAvailability | null - selectedAutomationRunPageWorkspaceDisplay: AutomationRunWorkspaceDisplay | null - selectedAutomationRunPageViewState: AutomationRunViewState | null - canRerunSelectedAutomationRunPage: boolean - isSelectedAutomationRunPageRerunPending: boolean worktreeMap: ReadonlyMap fetchExternalAutomationRuns: FetchExternalAutomationRuns onActivePaneTabChange: (tab: AutomationPaneTab) => void onClearExternalRunPage: () => void - onClearAutomationRunPage: () => void requestExternalAction: ( manager: ExternalAutomationManager, job: ExternalAutomationJob, @@ -95,8 +79,6 @@ type AutomationsDetailPaneProps = { openEditDialog: (automation: Automation) => void toggleAutomation: (automation: Automation) => void requestDeleteAutomation: (automation: Automation) => void - rerunAutomationRun: (automation: Automation, run: AutomationRun) => void - openRunWorkspace: (run: AutomationRun) => void openAutomationRunPage: (run: AutomationRun) => void onBackToList: () => void recoverSelectedRuns: (action: AutomationHostRecoveryAction) => void @@ -106,7 +88,6 @@ export function AutomationsDetailPane({ selected, selectedExternal, selectedExternalRunPage, - selectedAutomationRunPage, selectedRuns, selectedRunsNotice, activePaneTab, @@ -118,15 +99,10 @@ export function AutomationsDetailPane({ selectedHostEntry, hostLabelById, selectedRunNowAvailability, - selectedAutomationRunPageWorkspaceDisplay, - selectedAutomationRunPageViewState, - canRerunSelectedAutomationRunPage, - isSelectedAutomationRunPageRerunPending, worktreeMap, fetchExternalAutomationRuns, onActivePaneTabChange, onClearExternalRunPage, - onClearAutomationRunPage, requestExternalAction, openExternalRunPage, openEditExternalDialog, @@ -134,8 +110,6 @@ export function AutomationsDetailPane({ openEditDialog, toggleAutomation, requestDeleteAutomation, - rerunAutomationRun, - openRunWorkspace, openAutomationRunPage, onBackToList, recoverSelectedRuns @@ -148,10 +122,6 @@ export function AutomationsDetailPane({ onClearExternalRunPage() return } - if (selectedAutomationRunPage) { - onClearAutomationRunPage() - return - } onBackToList() return } @@ -179,10 +149,8 @@ export function AutomationsDetailPane({ activePaneTab, onActivePaneTabChange, onBackToList, - onClearAutomationRunPage, onClearExternalRunPage, selected, - selectedAutomationRunPage, selectedExternal, selectedExternalRunPage ]) @@ -291,76 +259,7 @@ export function AutomationsDetailPane({ - {selectedAutomationRunPage ? ( - - {canRerunSelectedAutomationRunPage && selected ? ( - - ) : null} - {selectedAutomationRunPageViewState ? ( - - ) : null} - - } - onBack={onClearAutomationRunPage} - > - - - ) : selected ? ( + {selected ? ( undefined} externalManagersUncheckedNotice={uncheckedNotice} onSelectHost={() => undefined} onRecoverHost={() => undefined} diff --git a/src/renderer/src/components/automations/AutomationsListPanel.tsx b/src/renderer/src/components/automations/AutomationsListPanel.tsx index fa6cb5891c4..5943096756a 100644 --- a/src/renderer/src/components/automations/AutomationsListPanel.tsx +++ b/src/renderer/src/components/automations/AutomationsListPanel.tsx @@ -107,6 +107,7 @@ type AutomationsListPanelProps = { onOpenDetail: () => void onRefresh: () => void isRefreshing: boolean + onOpenRuns: () => void } export function AutomationsListPanel(props: AutomationsListPanelProps): React.JSX.Element { @@ -153,7 +154,8 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS canCreateAutomation, onOpenDetail, onRefresh, - isRefreshing + isRefreshing, + onOpenRuns } = props const listRef = useRef(null) // Hosts moved into the Filters menu, so its toolbar row is the focus fallback now. @@ -293,6 +295,7 @@ export function AutomationsListPanel(props: AutomationsListPanelProps): React.JS hostEntries={hostCatalog.entries} onRefresh={onRefresh} isRefreshing={isRefreshing} + onOpenRuns={onOpenRuns} openCreateDialog={openCreateDialog} canCreateAutomation={canCreateAutomation} /> diff --git a/src/renderer/src/components/automations/AutomationsPageBreadcrumb.test.tsx b/src/renderer/src/components/automations/AutomationsPageBreadcrumb.test.tsx new file mode 100644 index 00000000000..2ce0a77fc56 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationsPageBreadcrumb.test.tsx @@ -0,0 +1,84 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot } from 'react-dom/client' +import { describe, expect, it, vi } from 'vitest' +import { AutomationsPageBreadcrumb } from './AutomationsPageBreadcrumb' + +describe('AutomationsPageBreadcrumb', () => { + it('renders the selected automation as the detail-page breadcrumb', () => { + const container = document.createElement('div') + const root = createRoot(container) + + act(() => { + root.render( + + ) + }) + + expect(container.textContent).toContain('Automations') + expect(container.textContent).toContain('Nightly sync') + expect(container.querySelector('[aria-current="page"]')?.textContent).toBe('Nightly sync') + + act(() => root.unmount()) + }) + + it('links a run detail page back through Runs and Automations', () => { + const container = document.createElement('div') + const root = createRoot(container) + const onBackToAutomations = vi.fn() + const onBackToRuns = vi.fn() + + act(() => { + root.render( + + ) + }) + + expect(container.textContent).toContain('Automations') + expect(container.textContent).toContain('Runs') + expect(container.querySelector('[aria-current="page"]')?.textContent).toBe('Run details') + + const buttons = container.querySelectorAll('button') + act(() => buttons[0]?.click()) + act(() => buttons[1]?.click()) + expect(onBackToAutomations).toHaveBeenCalledOnce() + expect(onBackToRuns).toHaveBeenCalledOnce() + + act(() => root.unmount()) + }) + + it('links an automation-origin run back to its automation details', () => { + const container = document.createElement('div') + const root = createRoot(container) + const onBackToAutomation = vi.fn() + + act(() => { + root.render( + + ) + }) + + expect(container.textContent).toContain('Nightly sync') + expect(container.textContent).toContain('Run details') + + const buttons = container.querySelectorAll('button') + act(() => buttons[1]?.click()) + expect(onBackToAutomation).toHaveBeenCalledOnce() + + act(() => root.unmount()) + }) +}) diff --git a/src/renderer/src/components/automations/AutomationsPageBreadcrumb.tsx b/src/renderer/src/components/automations/AutomationsPageBreadcrumb.tsx new file mode 100644 index 00000000000..fa98e91f490 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationsPageBreadcrumb.tsx @@ -0,0 +1,76 @@ +import React from 'react' +import { ChevronRight } from 'lucide-react' +import { Button } from '@/components/ui/button' +import { translate } from '@/i18n/i18n' + +export function AutomationsPageBreadcrumb({ + current, + onBackToAutomations, + onBackToRuns, + automationName, + onBackToAutomation +}: { + current: 'runs' | 'run' | 'automation' + onBackToAutomations: () => void + onBackToRuns?: () => void + automationName?: string + onBackToAutomation?: () => void +}): React.JSX.Element { + return ( + + ) +} diff --git a/src/renderer/src/components/automations/AutomationsPageDeleteDialogs.tsx b/src/renderer/src/components/automations/AutomationsPageDeleteDialogs.tsx new file mode 100644 index 00000000000..e5bb6cef3bc --- /dev/null +++ b/src/renderer/src/components/automations/AutomationsPageDeleteDialogs.tsx @@ -0,0 +1,65 @@ +import React from 'react' +import { AutomationDeleteDialog, ExternalAutomationDeleteDialog } from './AutomationDeleteDialogs' + +type Props = { + deleteTarget: React.ComponentProps['deleteTarget'] + dontAskDeleteAgain: boolean + deleteConfirmButtonRef: React.ComponentProps['confirmButtonRef'] + setDeleteTarget: (target: null) => void + setDontAskDeleteAgain: (value: boolean) => void + confirmDeleteAutomation: () => void + externalDeleteTarget: React.ComponentProps< + typeof ExternalAutomationDeleteDialog + >['externalDeleteTarget'] + externalDeleteConfirmButtonRef: React.ComponentProps< + typeof ExternalAutomationDeleteDialog + >['confirmButtonRef'] + setExternalDeleteTarget: (target: null) => void + confirmDeleteExternalAutomation: () => void +} + +export function AutomationsPageDeleteDialogs({ + deleteTarget, + dontAskDeleteAgain, + deleteConfirmButtonRef, + setDeleteTarget, + setDontAskDeleteAgain, + confirmDeleteAutomation, + externalDeleteTarget, + externalDeleteConfirmButtonRef, + setExternalDeleteTarget, + confirmDeleteExternalAutomation +}: Props): React.JSX.Element { + return ( + <> + { + if (!open) { + setDeleteTarget(null) + setDontAskDeleteAgain(false) + } + }} + onDontAskAgainToggle={() => setDontAskDeleteAgain(!dontAskDeleteAgain)} + onCancel={() => { + setDeleteTarget(null) + setDontAskDeleteAgain(false) + }} + onConfirm={confirmDeleteAutomation} + /> + { + if (!open) { + setExternalDeleteTarget(null) + } + }} + onCancel={() => setExternalDeleteTarget(null)} + onConfirm={confirmDeleteExternalAutomation} + /> + + ) +} diff --git a/src/renderer/src/components/automations/AutomationsPageListPanel.tsx b/src/renderer/src/components/automations/AutomationsPageListPanel.tsx new file mode 100644 index 00000000000..25c7ff7b88c --- /dev/null +++ b/src/renderer/src/components/automations/AutomationsPageListPanel.tsx @@ -0,0 +1,125 @@ +import React from 'react' +import type { AutomationsPageController } from './use-automations-page-controller' +import { AutomationsListPanel } from './AutomationsListPanel' + +export function AutomationsPageListPanel({ + controller, + onOpenDetail +}: { + controller: AutomationsPageController + onOpenDetail: () => void +}): React.JSX.Element { + const { + store, + local, + list, + destination, + sourceAvailability, + pageRefresh, + runActions, + editorActions, + managementActions, + externalActions, + presentation + } = controller + const { + projectHostSetups, + repoMap, + worktreeMap, + sshConnectionStates, + runtimeStatusByEnvironmentId + } = store + const { + listSearchQuery, + setListSearchQuery, + listFilter, + setListFilter, + relativeNow, + externalActionKey, + setActivePaneTab, + isLoading, + setPageView + } = local + const { + hostCatalog, + hasListItems, + hasFilteredListItems, + isListSearchQueryTooLarge, + filteredRows, + filteredExternalAutomationEntries, + selectedRow, + selectedExternal, + searchCounts + } = list + const onListFilterChange = (next: typeof listFilter): void => { + setListFilter(next) + if ((next.hostStableKeys?.length ?? 0) > 0 && hostCatalog.resolution.effective.kind !== 'all') { + hostCatalog.selectHost({ kind: 'all' }) + } + } + return ( + { + hostCatalog.recover(action, entry) + if (action === 'retry') { + void pageRefresh.refresh() + } + }} + filteredRows={filteredRows} + filteredExternalAutomationEntries={filteredExternalAutomationEntries} + selectedRowKey={selectedRow?.key ?? null} + selectedExternalKey={local.selectedExternalKey} + selectedExternal={selectedExternal} + relativeNow={relativeNow} + repoMap={repoMap} + worktreeMap={worktreeMap} + repoForRow={store.repoForRow} + worktreeForRow={store.worktreeForRow} + projectHostSetups={projectHostSetups} + sshConnectionStates={sshConnectionStates} + runtimeStatusByEnvironmentId={runtimeStatusByEnvironmentId} + hostTargetFor={destination.automationHostTargetFor} + automationSourceHostAvailabilityByRowKey={ + sourceAvailability.automationSourceHostAvailabilityByRowKey + } + hostLabelById={presentation.hostLabelById} + isActionEnabled={destination.isAutomationRowActionEnabled} + externalActionKey={externalActionKey} + selectAutomationRow={list.selectAutomationRow} + selectExternalKey={local.selectExternalKey} + setActivePaneTab={setActivePaneTab} + runNow={(row) => void runActions.runNow(row)} + openEditDialog={(row) => void editorActions.openEditDialog(row)} + toggleAutomation={(row) => void managementActions.toggleAutomation(row)} + requestDeleteAutomation={managementActions.requestDeleteAutomation} + requestExternalAction={externalActions.requestExternalAction} + openEditExternalDialog={editorActions.openEditExternalDialog} + openCreateDialog={editorActions.openCreateDialog} + canCreateAutomation={destination.canCreateAutomation} + onOpenDetail={onOpenDetail} + onRefresh={() => { + hostCatalog.refreshHosts() + void pageRefresh.refresh() + }} + isRefreshing={isLoading} + onOpenRuns={() => { + hostCatalog.selectHost({ kind: 'all' }) + setPageView('runs') + }} + /> + ) +} diff --git a/src/renderer/src/components/automations/AutomationsPageSurface.tsx b/src/renderer/src/components/automations/AutomationsPageSurface.tsx index 512add800a3..1e46b5f0283 100644 --- a/src/renderer/src/components/automations/AutomationsPageSurface.tsx +++ b/src/renderer/src/components/automations/AutomationsPageSurface.tsx @@ -1,17 +1,17 @@ import React, { useMemo } from 'react' import { translate } from '@/i18n/i18n' -import { AutomationDeleteDialog, ExternalAutomationDeleteDialog } from './AutomationDeleteDialogs' import { AutomationEditorDialog } from './AutomationEditorDialog' -import { AutomationOwnerConflictNotice } from './AutomationOwnerConflictNotice' import { AutomationsDetailPane } from './AutomationsDetailPane' -import { AutomationsListPanel } from './AutomationsListPanel' import { AutomationsPageSkeleton } from './AutomationsPageSkeleton' import { getAutomationAuthorityTarget } from './automation-host-client' import type { AutomationListRow } from './automation-list-row-identity' import type { AutomationHostRecoveryAction } from './automation-host-status-descriptors' +import { AutomationsPageTopBar } from './AutomationsPageTopBar' import type { AutomationsPageController } from './use-automations-page-controller' - -/** Renders the page from controller state; host/query side effects stay in hooks. */ +import { AutomationRunsDashboardSurface } from './AutomationRunsDashboardSurface' +import { AutomationRunDetailsPage } from './AutomationRunDetailsPage' +import { AutomationsPageDeleteDialogs } from './AutomationsPageDeleteDialogs' +import { AutomationsPageListPanel } from './AutomationsPageListPanel' export function AutomationsPageSurface({ controller }: { @@ -22,10 +22,10 @@ export function AutomationsPageSurface({ local, list, destination, + runsDashboard, destinationForm, setup, runPage, - sourceAvailability, presentation, pageRefresh, draftEffects, @@ -41,10 +41,9 @@ export function AutomationsPageSurface({ repoMap, worktreeMap, settings, - sshConnectionStates, - runtimeStatusByEnvironmentId, repoForRow, - worktreeForRow + worktreeForRow, + setPendingAutomationRunNavigation } = store const { createOpen, @@ -63,10 +62,6 @@ export function AutomationsPageSurface({ externalDeleteTarget, externalDeleteConfirmButtonRef, setExternalDeleteTarget, - listSearchQuery, - setListSearchQuery, - listFilter, - setListFilter, relativeNow, externalActionKey, activePaneTab, @@ -84,21 +79,14 @@ export function AutomationsPageSurface({ setEditorNotice, editorNoticeHost, setEditorNoticeHost, - isLoading + isLoading, + pageView, + setPageView, + runPageOrigin, + setRunPageOrigin } = local - const { - hostCatalog, - hasListItems, - hasFilteredListItems, - isListSearchQueryTooLarge, - filteredRows, - filteredExternalAutomationEntries, - selected, - selectedRow, - selectedExternal, - searchCounts - } = list - + const { hostCatalog, hasListItems, selected, selectedRow, selectedExternal } = list + const selectedAutomationRunPage = setup.selectedAutomationRunPage const selectedRunWorktreeMap = useMemo(() => { if (!selectedRow) { return worktreeMap @@ -111,7 +99,6 @@ export function AutomationsPageSurface({ }) ) }, [repoForRow, selectedRow, setup.selectedRuns, worktreeForRow, worktreeMap]) - const runSelectedRowAction = (action: (row: AutomationListRow) => void): void => { if (selectedRow) { action(selectedRow) @@ -127,32 +114,43 @@ export function AutomationsPageSurface({ void pageRefresh.refresh() } } - const onListFilterChange = (next: typeof listFilter): void => { - setListFilter(next) - if ((next.hostStableKeys?.length ?? 0) > 0 && hostCatalog.resolution.effective.kind !== 'all') { - // Host narrowing now lives in the Filters menu; clear the old single-host scope. - hostCatalog.selectHost({ kind: 'all' }) - } + const openAutomationRunPage = (run: (typeof setup.selectedRuns)[number]): void => { + externalActions.openAutomationRunPage(run) + setRunPageOrigin('automation') + setPageView('run') + } + const showAutomationsList = (): void => { + setPageView('automations') + setSelectedAutomationRunPageId(null) + setIsDetailOpen(false) + setActivePaneTab('overview') + } + const showRunsDashboard = (): void => { + setPageView('runs') + setSelectedAutomationRunPageId(null) + setIsDetailOpen(false) + setActivePaneTab('overview') + } + const showAutomationDetails = (): void => { + setPageView('automations') + setSelectedAutomationRunPageId(null) + setIsDetailOpen(true) + setActivePaneTab('runs') } - return (
    -
    -

    - {translate('auto.components.automations.AutomationsPage.77c2778945', 'Automations')} -

    -
    - - recoverOwnerAction(action)} - onDismiss={() => setOwnerAction(null)} + setOwnerAction(null)} + showAutomationsList={showAutomationsList} + showRunsDashboard={showRunsDashboard} + showAutomationDetails={showAutomationDetails} /> - void saveAutomation()} /> - - { - if (!open) { - setDeleteTarget(null) - setDontAskDeleteAgain(false) - } - }} - onDontAskAgainToggle={() => setDontAskDeleteAgain((previous) => !previous)} - onCancel={() => { - setDeleteTarget(null) - setDontAskDeleteAgain(false) - }} - onConfirm={() => void managementActions.confirmDeleteAutomation()} - /> - - void managementActions.confirmDeleteAutomation()} externalDeleteTarget={externalDeleteTarget} - confirmButtonRef={externalDeleteConfirmButtonRef} - onOpenChange={(open) => { - if (!open) { - setExternalDeleteTarget(null) - } - }} - onCancel={() => setExternalDeleteTarget(null)} - onConfirm={() => void externalActions.confirmDeleteExternalAutomation()} + externalDeleteConfirmButtonRef={externalDeleteConfirmButtonRef} + setExternalDeleteTarget={setExternalDeleteTarget} + confirmDeleteExternalAutomation={() => + void externalActions.confirmDeleteExternalAutomation() + } /> - - {isLoading && !hasListItems ? ( + {pageView === 'runs' ? ( + setRunHistoryReloadToken((token) => token + 1)} + setPageView={setPageView} + setRunPageOrigin={setRunPageOrigin} + selectAutomationRow={list.selectAutomationRow} + setPendingAutomationRunNavigation={setPendingAutomationRunNavigation} + setIsDetailOpen={setIsDetailOpen} + /> + ) : pageView === 'run' && selectedAutomationRunPage ? ( + + runSelectedRowAction((row) => + runActions.rerunAutomationRun(row, selectedAutomationRunPage) + ) + } + onOpenWorkspace={() => openRunWorkspace(selectedAutomationRunPage)} + onBack={runPageOrigin === 'automation' ? showAutomationDetails : showRunsDashboard} + /> + ) : pageView === 'run' ? ( + + ) : isLoading && !hasListItems ? ( ) : isDetailOpen && (selected || selectedExternal) ? ( setSelectedExternalRunPage(null)} - onClearAutomationRunPage={() => setSelectedAutomationRunPageId(null)} requestExternalAction={externalActions.requestExternalAction} openExternalRunPage={externalActions.openExternalRunPage} openEditExternalDialog={editorActions.openEditExternalDialog} @@ -299,11 +307,7 @@ export function AutomationsPageSurface({ requestDeleteAutomation={() => runSelectedRowAction(managementActions.requestDeleteAutomation) } - rerunAutomationRun={(_automation, run) => - runSelectedRowAction((row) => runActions.rerunAutomationRun(row, run)) - } - openRunWorkspace={openRunWorkspace} - openAutomationRunPage={externalActions.openAutomationRunPage} + openAutomationRunPage={openAutomationRunPage} onBackToList={() => { setIsDetailOpen(false) setSelectedAutomationRunPageId(null) @@ -312,64 +316,9 @@ export function AutomationsPageSurface({ }} /> ) : ( - { - hostCatalog.recover(action, entry) - if (action === 'retry') { - void pageRefresh.refresh() - } - }} - filteredRows={filteredRows} - filteredExternalAutomationEntries={filteredExternalAutomationEntries} - selectedRowKey={selectedRow?.key ?? null} - selectedExternalKey={local.selectedExternalKey} - selectedExternal={selectedExternal} - relativeNow={relativeNow} - repoMap={repoMap} - worktreeMap={worktreeMap} - repoForRow={repoForRow} - worktreeForRow={worktreeForRow} - projectHostSetups={projectHostSetups} - sshConnectionStates={sshConnectionStates} - runtimeStatusByEnvironmentId={runtimeStatusByEnvironmentId} - hostTargetFor={destination.automationHostTargetFor} - automationSourceHostAvailabilityByRowKey={ - sourceAvailability.automationSourceHostAvailabilityByRowKey - } - hostLabelById={presentation.hostLabelById} - isActionEnabled={destination.isAutomationRowActionEnabled} - externalActionKey={externalActionKey} - selectAutomationRow={list.selectAutomationRow} - selectExternalKey={local.selectExternalKey} - setActivePaneTab={setActivePaneTab} - runNow={(row) => void runActions.runNow(row)} - openEditDialog={(row) => void editorActions.openEditDialog(row)} - toggleAutomation={(row) => void managementActions.toggleAutomation(row)} - requestDeleteAutomation={managementActions.requestDeleteAutomation} - requestExternalAction={externalActions.requestExternalAction} - openEditExternalDialog={editorActions.openEditExternalDialog} - openCreateDialog={editorActions.openCreateDialog} - canCreateAutomation={destination.canCreateAutomation} + setIsDetailOpen(true)} - onRefresh={() => { - hostCatalog.refreshHosts() - void pageRefresh.refresh() - }} - isRefreshing={isLoading} /> )}
    diff --git a/src/renderer/src/components/automations/AutomationsPageTopBar.tsx b/src/renderer/src/components/automations/AutomationsPageTopBar.tsx new file mode 100644 index 00000000000..ec48aced278 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationsPageTopBar.tsx @@ -0,0 +1,65 @@ +import React from 'react' +import { translate } from '@/i18n/i18n' +import { AutomationOwnerConflictNotice } from './AutomationOwnerConflictNotice' +import { AutomationsPageBreadcrumb } from './AutomationsPageBreadcrumb' +import type { AutomationHostRecoveryAction } from './automation-host-status-descriptors' +import type { AutomationActionNotice } from './automation-row-action-dispatch' + +export function AutomationsPageTopBar({ + pageView, + isDetailOpen, + selectedAutomationName, + runPageOrigin, + ownerNotice, + recoverOwnerAction, + dismissOwnerAction, + showAutomationsList, + showRunsDashboard, + showAutomationDetails +}: { + pageView: 'automations' | 'runs' | 'run' + isDetailOpen: boolean + selectedAutomationName?: string + runPageOrigin: 'automation' | 'runs' + ownerNotice: AutomationActionNotice | null + recoverOwnerAction: (action: AutomationHostRecoveryAction) => void + dismissOwnerAction: () => void + showAutomationsList: () => void + showRunsDashboard: () => void + showAutomationDetails: () => void +}): React.JSX.Element { + return ( + <> +
    + {pageView === 'runs' || pageView === 'run' ? ( + + ) : isDetailOpen && selectedAutomationName ? ( + + ) : ( +

    + {translate('auto.components.automations.AutomationsPage.77c2778945', 'Automations')} +

    + )} +
    + + + ) +} diff --git a/src/renderer/src/components/automations/automation-host-client.ts b/src/renderer/src/components/automations/automation-host-client.ts index e7ef5a44770..b34efbb180e 100644 --- a/src/renderer/src/components/automations/automation-host-client.ts +++ b/src/renderer/src/components/automations/automation-host-client.ts @@ -3,6 +3,7 @@ import type { Automation, AutomationCreateInput, AutomationRun, + AutomationRunsPage, AutomationUpdateInput } from '../../../../shared/automations-types' import type { AutomationAuthorityRef } from '../../../../shared/automation-owner-ref' @@ -131,6 +132,20 @@ export async function listAutomationRunsForTarget( return result.runs } +export async function listAutomationRunsPageForTarget( + target: AutomationHostTarget, + automationId: string, + options: { limit?: number; cursor?: string } = {} +): Promise { + const result = await callRuntimeRpc( + target, + 'automation.runs', + { automationId, ...options }, + { timeoutMs: 15_000 } + ) + return { runs: result.runs, nextCursor: 'nextCursor' in result ? result.nextCursor : null } +} + export async function updateAutomationForTarget( automation: Automation, updates: AutomationUpdateInput, diff --git a/src/renderer/src/components/automations/automation-owner-action-runner.ts b/src/renderer/src/components/automations/automation-owner-action-runner.ts index b25baa16d08..ae4920252df 100644 --- a/src/renderer/src/components/automations/automation-owner-action-runner.ts +++ b/src/renderer/src/components/automations/automation-owner-action-runner.ts @@ -16,6 +16,7 @@ import type { Automation, AutomationCreateInput, AutomationRun, + AutomationRunsPage, AutomationUpdateInput } from '../../../../shared/automations-types' import { @@ -40,8 +41,10 @@ import { deleteAutomationForOwner, deleteOrphanAutomation, listAutomationRunsForOwner, + listAutomationRunsPageForOwner, listAutomationsForOwner, listOrphanAutomationRuns, + listOrphanAutomationRunsPage, matchAutomationOwnerConflict, runAutomationNowForOwner, updateAutomationForOwner, @@ -213,6 +216,19 @@ export async function listOwnedAutomationRuns( ) } +export async function listOwnedAutomationRunsPage( + availability: AutomationActionAvailability, + authority: AutomationAuthorityRef, + automationId: string, + options: { limit?: number; cursor?: string } +): Promise> { + return await attempt( + availability, + (owner) => listAutomationRunsPageForOwner(owner, automationId, options), + () => listOrphanAutomationRunsPage(authority, automationId, options) + ) +} + /** Create is destination-keyed rather than owner-keyed: the row does not exist yet. */ export async function createAutomationAtDestination( authority: AutomationAuthorityRef, diff --git a/src/renderer/src/components/automations/automation-page-state.ts b/src/renderer/src/components/automations/automation-page-state.ts index b4eb2a31cc0..c273c778038 100644 --- a/src/renderer/src/components/automations/automation-page-state.ts +++ b/src/renderer/src/components/automations/automation-page-state.ts @@ -7,6 +7,11 @@ import type { /** Detail-pane tab shared by the page, its list panel, and the detail pane. */ export type AutomationPaneTab = 'overview' | 'runs' +/** Top-level surface within Automations. */ +export type AutomationsPageView = 'automations' | 'runs' | 'run' + +export type AutomationRunPageOrigin = 'runs' | 'automation' + /** External run opened as a full page inside the detail pane. */ export type SelectedExternalRunPage = { manager: ExternalAutomationManager diff --git a/src/renderer/src/components/automations/automation-row-action-dispatch.ts b/src/renderer/src/components/automations/automation-row-action-dispatch.ts index 564f6cd5d74..e75e0ce70f7 100644 --- a/src/renderer/src/components/automations/automation-row-action-dispatch.ts +++ b/src/renderer/src/components/automations/automation-row-action-dispatch.ts @@ -12,6 +12,7 @@ import type { Automation, AutomationRun, + AutomationRunsPage, AutomationUpdateInput } from '../../../../shared/automations-types' import type { AutomationAuthorityRef } from '../../../../shared/automation-owner-ref' @@ -25,6 +26,7 @@ import { import { deleteOwnedAutomation, listOwnedAutomationRuns, + listOwnedAutomationRunsPage, runOwnedAutomationNow, showOwnedAutomation, updateOwnedAutomation, @@ -187,13 +189,31 @@ export async function dispatchAutomationReread( export async function dispatchAutomationRunHistory( context: AutomationDispatchContext, row: AutomationDispatchRow, - legacy: () => Promise + legacy: () => Promise, + authority: AutomationAuthorityRef = context.authority ): Promise> { return await dispatch( context, row, 'history', - (availability) => listOwnedAutomationRuns(availability, context.authority, row.automationId), + (availability) => listOwnedAutomationRuns(availability, authority, row.automationId), + legacy + ) +} + +export async function dispatchAutomationRunHistoryPage( + context: AutomationDispatchContext, + row: AutomationDispatchRow, + options: { limit?: number; cursor?: string }, + legacy: () => Promise, + authority: AutomationAuthorityRef = context.authority +): Promise> { + return await dispatch( + context, + row, + 'history', + (availability) => + listOwnedAutomationRunsPage(availability, authority, row.automationId, options), legacy ) } diff --git a/src/renderer/src/components/automations/automation-runs-dashboard-model.test.ts b/src/renderer/src/components/automations/automation-runs-dashboard-model.test.ts new file mode 100644 index 00000000000..43175dc3415 --- /dev/null +++ b/src/renderer/src/components/automations/automation-runs-dashboard-model.test.ts @@ -0,0 +1,105 @@ +import { describe, expect, it } from 'vitest' +import type { Automation, AutomationRun } from '../../../../shared/automations-types' +import type { AutomationListRow } from './automation-list-row-identity' +import { + buildAutomationRunsDashboardEntries, + countAutomationRunOutcomes, + filterAutomationRunsDashboardEntries, + getAutomationRunsHostKey, + getAutomationRunsScope +} from './automation-runs-dashboard-model' + +function row( + key: string, + hostLabel: string, + catalogRef: NonNullable +): AutomationListRow { + return { + key, + hostLabel, + catalogRef, + usageSummary: null, + automation: { + id: key, + name: `Automation ${key}`, + executionTargetType: catalogRef.selector.kind === 'ssh' ? 'ssh' : 'local' + } as Automation + } +} + +function run( + id: string, + automationId: string, + scheduledFor: number, + status: AutomationRun['status'] +) { + return { id, automationId, scheduledFor, status, title: `Run ${id}` } as AutomationRun +} + +describe('automation runs dashboard model', () => { + const local = row('local-row', 'Local Mac', { + authority: { kind: 'desktop' }, + selector: { kind: 'self' } + }) + const ssh = row('ssh-row', 'Build host', { + authority: { kind: 'desktop' }, + selector: { kind: 'ssh', targetId: 'build' } + }) + const runtime = row('runtime-row', 'Cloud runtime', { + authority: { kind: 'runtime', environmentId: 'cloud' }, + selector: { kind: 'self' } + }) + + it('keeps local, SSH, and runtime hosts in one chronologically sorted list', () => { + const entries = buildAutomationRunsDashboardEntries( + [local, ssh, runtime], + new Map([ + [local.key, [run('local', local.automation.id, 10, 'completed')]], + [ssh.key, [run('ssh', ssh.automation.id, 30, 'dispatch_failed')]], + [runtime.key, [run('runtime', runtime.automation.id, 20, 'completed')]] + ]) + ) + + expect(entries.map((entry) => entry.run.id)).toEqual(['ssh', 'runtime', 'local']) + expect(getAutomationRunsScope(local)).toBe('local') + expect(getAutomationRunsScope(ssh)).toBe('remote') + expect(getAutomationRunsScope(runtime)).toBe('remote') + }) + + it('filters by host without splitting the dashboard into local and remote views', () => { + const entries = buildAutomationRunsDashboardEntries( + [local, ssh], + new Map([ + [local.key, [run('local', local.automation.id, 10, 'completed')]], + [ssh.key, [run('ssh', ssh.automation.id, 20, 'dispatch_failed')]] + ]) + ) + + expect( + filterAutomationRunsDashboardEntries({ + entries, + status: 'all', + query: '', + hostKeys: [getAutomationRunsHostKey(ssh)] + }).map((entry) => entry.run.id) + ).toEqual(['ssh']) + expect(countAutomationRunOutcomes(entries, 30).successful7d).toBe(1) + }) + + it('keeps future-dated runs out of the outcome windows', () => { + const entries = buildAutomationRunsDashboardEntries( + [local, ssh], + new Map([ + [local.key, [run('ahead', local.automation.id, 40, 'completed')]], + [ssh.key, [run('ahead-failed', ssh.automation.id, 40, 'dispatch_failed')]] + ]) + ) + + expect(countAutomationRunOutcomes(entries, 30)).toEqual({ + successful24h: 0, + failed24h: 0, + successful7d: 0, + failed7d: 0 + }) + }) +}) diff --git a/src/renderer/src/components/automations/automation-runs-dashboard-model.ts b/src/renderer/src/components/automations/automation-runs-dashboard-model.ts new file mode 100644 index 00000000000..fb177a55312 --- /dev/null +++ b/src/renderer/src/components/automations/automation-runs-dashboard-model.ts @@ -0,0 +1,138 @@ +import type { AutomationRun, AutomationRunStatus } from '../../../../shared/automations-types' +import { parseExecutionHostId } from '../../../../shared/execution-host' +import type { AutomationActionNotice } from './automation-row-action-dispatch' +import type { AutomationListRow } from './automation-list-row-identity' + +export type AutomationRunsScope = 'local' | 'remote' +export type AutomationRunsStatusFilter = 'all' | 'successful' | 'failed' | 'active' | 'skipped' + +export type AutomationRunsDashboardEntry = { + key: string + hostKey: string + searchText: string + row: AutomationListRow + run: AutomationRun + scope: AutomationRunsScope +} + +export type AutomationRunsDashboardFailure = { + row: AutomationListRow + scope: AutomationRunsScope + notice: AutomationActionNotice +} + +export function getAutomationRunsHostKey(row: AutomationListRow): string { + const authority = row.catalogRef?.authority + const selector = row.catalogRef?.selector + const authorityKey = + authority?.kind === 'runtime' ? `runtime:${authority.environmentId}` : 'desktop' + const selectorKey = + selector?.kind === 'ssh' + ? `ssh:${selector.targetId}` + : selector?.kind === 'orphan' + ? 'orphan' + : 'self' + return `${authorityKey}:${selectorKey}` +} + +export function getAutomationRunsScope(row: AutomationListRow): AutomationRunsScope { + const authority = row.catalogRef?.authority + const selector = row.catalogRef?.selector + if (authority?.kind === 'runtime' || selector?.kind === 'ssh') { + return 'remote' + } + if (selector?.kind === 'self') { + return 'local' + } + const runHost = parseExecutionHostId(row.automation.runContext?.hostId) + return row.automation.executionTargetType === 'ssh' || runHost?.kind === 'runtime' + ? 'remote' + : 'local' +} + +export function buildAutomationRunsDashboardEntries( + rows: readonly AutomationListRow[], + runsByRowKey: ReadonlyMap +): AutomationRunsDashboardEntry[] { + return rows + .flatMap((row) => + (runsByRowKey.get(row.key) ?? []).map((run) => ({ + key: `${row.key}:${run.id}`, + hostKey: getAutomationRunsHostKey(row), + searchText: [row.automation.name, run.title, row.hostLabel].join('\n').toLocaleLowerCase(), + row, + run, + scope: getAutomationRunsScope(row) + })) + ) + .sort((left, right) => right.run.scheduledFor - left.run.scheduledFor) +} + +function matchesStatus(status: AutomationRunStatus, filter: AutomationRunsStatusFilter): boolean { + if (filter === 'all') { + return true + } + if (filter === 'successful') { + return status === 'completed' + } + if (filter === 'failed') { + return status === 'dispatch_failed' + } + if (filter === 'skipped') { + return status.startsWith('skipped') + } + return status === 'pending' || status === 'dispatching' || status === 'dispatched' +} + +export function filterAutomationRunsDashboardEntries({ + entries, + status, + query, + hostKeys +}: { + entries: readonly AutomationRunsDashboardEntry[] + status: AutomationRunsStatusFilter + query: string + hostKeys: readonly string[] +}): AutomationRunsDashboardEntry[] { + const normalizedQuery = query.trim().toLocaleLowerCase() + const selectedHosts = new Set(hostKeys) + return entries.filter((entry) => { + if ( + !matchesStatus(entry.run.status, status) || + (selectedHosts.size > 0 && !selectedHosts.has(entry.hostKey)) + ) { + return false + } + if (!normalizedQuery) { + return true + } + return entry.searchText.includes(normalizedQuery) + }) +} + +export function countAutomationRunOutcomes( + entries: readonly AutomationRunsDashboardEntry[], + now: number +): { successful24h: number; failed24h: number; successful7d: number; failed7d: number } { + const dayAgo = now - 24 * 60 * 60 * 1000 + const weekAgo = now - 7 * 24 * 60 * 60 * 1000 + const counts = { successful24h: 0, failed24h: 0, successful7d: 0, failed7d: 0 } + for (const entry of entries) { + // A clock-skewed or future-dated run has not happened inside either window yet. + if (entry.run.scheduledFor > now) { + continue + } + const successful = entry.run.status === 'completed' + const failed = entry.run.status === 'dispatch_failed' + if (entry.run.scheduledFor >= weekAgo) { + counts.successful7d += successful ? 1 : 0 + counts.failed7d += failed ? 1 : 0 + } + if (entry.run.scheduledFor >= dayAgo) { + counts.successful24h += successful ? 1 : 0 + counts.failed24h += failed ? 1 : 0 + } + } + return counts +} diff --git a/src/renderer/src/components/automations/automation-scoped-list-client.ts b/src/renderer/src/components/automations/automation-scoped-list-client.ts index 825329a1812..7c4cf6fddad 100644 --- a/src/renderer/src/components/automations/automation-scoped-list-client.ts +++ b/src/renderer/src/components/automations/automation-scoped-list-client.ts @@ -14,6 +14,7 @@ import type { Automation, AutomationCreateInput, AutomationRun, + AutomationRunsPage, AutomationUpdateInput } from '../../../../shared/automations-types' import type { @@ -173,11 +174,13 @@ export async function listAutomationsForOwner( async function listRunsFenced( authority: AutomationAuthorityRef, automationId: string, - expectedOwner: AutomationOwnerPrecondition + expectedOwner: AutomationOwnerPrecondition, + options: { limit?: number; cursor?: string } = {} ): Promise { const result = await callAuthority<{ runs: AutomationRun[] }>(authority, 'automation.runs', { automationId, - expectedOwner + expectedOwner, + ...options }) return result.runs } @@ -189,6 +192,19 @@ export async function listAutomationRunsForOwner( return await listRunsFenced(owner.authority, automationId, ownerPrecondition(owner)) } +export async function listAutomationRunsPageForOwner( + owner: AutomationOwnerRef, + automationId: string, + options: { limit?: number; cursor?: string } = {} +): Promise { + const result = await callAuthority( + owner.authority, + 'automation.runs', + { automationId, expectedOwner: ownerPrecondition(owner), ...options } + ) + return { runs: result.runs, nextCursor: 'nextCursor' in result ? result.nextCursor : null } +} + /** * The one fenced-mutation path every authority shares. Owned and orphan rows * differ in the precondition they fence with and in nothing else, so they @@ -269,6 +285,19 @@ export async function listOrphanAutomationRuns( return await listRunsFenced(authority, automationId, ORPHAN_OWNER_PRECONDITION) } +export async function listOrphanAutomationRunsPage( + authority: AutomationAuthorityRef, + automationId: string, + options: { limit?: number; cursor?: string } = {} +): Promise { + const result = await callAuthority( + authority, + 'automation.runs', + { automationId, expectedOwner: ORPHAN_OWNER_PRECONDITION, ...options } + ) + return { runs: result.runs, nextCursor: 'nextCursor' in result ? result.nextCursor : null } +} + export async function runAutomationNowForOwner( owner: AutomationOwnerRef, id: string diff --git a/src/renderer/src/components/automations/use-automation-run-page-state.ts b/src/renderer/src/components/automations/use-automation-run-page-state.ts index e99ec438cbf..ef970dff5a7 100644 --- a/src/renderer/src/components/automations/use-automation-run-page-state.ts +++ b/src/renderer/src/components/automations/use-automation-run-page-state.ts @@ -48,6 +48,8 @@ export function useAutomationRunPageState({ selectedExternalKey, selectExternalKey, setSelectedExternalRunPage, + pageView, + setPageView, isDetailOpen, setIsDetailOpen } = local @@ -105,6 +107,9 @@ export function useAutomationRunPageState({ setSelectedId(pending.automationId) setSelectedAutomationRunPageId(null) setPendingAutomationRunNavigation(null) + if (pageView === 'run') { + setPageView('runs') + } toast.message( translate( 'auto.components.automations.AutomationsPage.pendingAutomationMissing', @@ -123,6 +128,7 @@ export function useAutomationRunPageState({ setActivePaneTab('overview') setSelectedAutomationRunPageId(null) setPendingAutomationRunNavigation(null) + setPageView('automations') return } if ( @@ -133,6 +139,9 @@ export function useAutomationRunPageState({ setActivePaneTab('runs') setSelectedAutomationRunPageId(null) setPendingAutomationRunNavigation(null) + if (pageView === 'run') { + setPageView('runs') + } return } if (selectedAutomationRuns.automationId !== pending.automationId) { @@ -144,10 +153,14 @@ export function useAutomationRunPageState({ if (pendingRun) { setSelectedAutomationRunPageId(pending.runId) setPendingAutomationRunNavigation(null) + setPageView('run') return } setSelectedAutomationRunPageId(null) setPendingAutomationRunNavigation(null) + if (pageView === 'run') { + setPageView('runs') + } toast.message( translate( 'auto.components.automations.AutomationsPage.pendingAutomationRunMissing', @@ -159,6 +172,7 @@ export function useAutomationRunPageState({ automationHostTargetKey, isLoading, pendingAutomationRunNavigation, + pageView, selectExternalKey, selectedAutomationRuns.automationId, selectedAutomationRuns.notice, @@ -168,6 +182,7 @@ export function useAutomationRunPageState({ setActivePaneTab, setIsDetailOpen, setPendingAutomationRunNavigation, + setPageView, setSelectedAutomationRunPageId, setSelectedId ]) diff --git a/src/renderer/src/components/automations/use-automation-runs-dashboard.test.tsx b/src/renderer/src/components/automations/use-automation-runs-dashboard.test.tsx new file mode 100644 index 00000000000..519904a1e90 --- /dev/null +++ b/src/renderer/src/components/automations/use-automation-runs-dashboard.test.tsx @@ -0,0 +1,144 @@ +// @vitest-environment happy-dom + +import { act, useCallback } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { AutomationAuthorityRef } from '../../../../shared/automation-owner-ref' +import type { AutomationRunsPage } from '../../../../shared/automations-types' +import type { AutomationHostTarget } from './automation-host-client' +import { makeAutomationListRow, makeRun } from './automations-page-fixtures' +import * as dispatch from './automation-row-action-dispatch' +import { useAutomationRunsDashboard } from './use-automation-runs-dashboard' + +vi.mock('./automation-row-action-dispatch', async (importOriginal) => ({ + ...(await importOriginal()), + dispatchAutomationRunHistoryPage: vi.fn() +})) + +const historySpy = vi.mocked(dispatch.dispatchAutomationRunHistoryPage) + +const PAGES: Record = { + head: { runs: [makeRun({ id: 'run-2', createdAt: 2 })], nextCursor: '2:run-2' }, + '2:run-2': { runs: [makeRun({ id: 'run-1', createdAt: 1 })], nextCursor: null } +} + +const row = makeAutomationListRow() +const rows = [row] +const context = { capturedOwners: new Map(), authority: { kind: 'desktop' as const } } + +type DashboardResult = ReturnType + +let container: HTMLDivElement +let root: Root +let latest: DashboardResult | null = null + +type HarnessProps = { + enabled: boolean + authority?: AutomationAuthorityRef + target?: AutomationHostTarget | null +} + +function Harness({ enabled, authority, target }: HarnessProps): null { + latest = useAutomationRunsDashboard({ + enabled, + rows, + context, + legacyTarget: useCallback(() => target ?? null, [target]), + authorityForRow: useCallback(() => authority ?? { kind: 'desktop' }, [authority]), + reloadToken: 0 + }) + return null +} + +async function render(enabled: boolean, props: Omit = {}): Promise { + await act(async () => { + root.render() + }) +} + +beforeEach(() => { + historySpy.mockImplementation(async (_context, _row, options) => ({ + ok: true, + value: PAGES[options.cursor ?? 'head'] + })) + container = document.createElement('div') + document.body.append(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() + latest = null + historySpy.mockReset() +}) + +describe('useAutomationRunsDashboard', () => { + it('appends the next page when load more fires', async () => { + await render(true) + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2']) + + await act(async () => latest?.loadMore()) + + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2', 'run-1']) + expect(historySpy.mock.calls.at(-1)?.[2].cursor).toBe('2:run-2') + }) + + it('keeps the cursor when a load more fails so the page stays retryable', async () => { + await render(true) + historySpy.mockImplementationOnce(async () => ({ + ok: false, + notice: { message: 'offline', recovery: null, severity: 'failure' } + })) + + await act(async () => latest?.loadMore()) + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2']) + expect(latest?.hasMore).toBe(true) + + await act(async () => latest?.loadMore()) + + expect(historySpy.mock.calls.at(-1)?.[2].cursor).toBe('2:run-2') + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2', 'run-1']) + }) + + it('refetches the head when the view is re-entered after a load more', async () => { + await render(true) + await act(async () => latest?.loadMore()) + await render(false) + historySpy.mockClear() + + await render(true) + + expect(historySpy.mock.calls.at(-1)?.[2].cursor).toBeUndefined() + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2']) + }) + + it('reloads from the head when the authority is re-paired', async () => { + const paired = (pairingRevision: number): AutomationAuthorityRef => ({ + kind: 'runtime', + environmentId: 'env-1', + pairingRevision + }) + await render(true, { authority: paired(1) }) + await act(async () => latest?.loadMore()) + expect(latest?.entries).toHaveLength(2) + historySpy.mockClear() + + await render(true, { authority: paired(2) }) + + expect(historySpy.mock.calls.at(-1)?.[2].cursor).toBeUndefined() + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2']) + }) + + it('reloads from the head when an uncaptured row moves to another fallback target', async () => { + await render(true, { target: { kind: 'local' } }) + await act(async () => latest?.loadMore()) + expect(latest?.entries).toHaveLength(2) + historySpy.mockClear() + + await render(true, { target: { kind: 'environment', environmentId: 'env-1' } }) + + expect(historySpy.mock.calls.at(-1)?.[2].cursor).toBeUndefined() + expect(latest?.entries.map((entry) => entry.run.id)).toEqual(['run-2']) + }) +}) diff --git a/src/renderer/src/components/automations/use-automation-runs-dashboard.ts b/src/renderer/src/components/automations/use-automation-runs-dashboard.ts new file mode 100644 index 00000000000..60588b61fbf --- /dev/null +++ b/src/renderer/src/components/automations/use-automation-runs-dashboard.ts @@ -0,0 +1,188 @@ +import { useCallback, useEffect, useMemo, useRef, useState } from 'react' +import type { AutomationRun } from '../../../../shared/automations-types' +import type { AutomationAuthorityRef } from '../../../../shared/automation-owner-ref' +import { ownerKey } from '../../../../shared/automation-owner-key' +import { capturedAutomationOwner, capturedAutomationOwnerKey } from './automation-captured-owner' +import { + getAutomationHostTargetKey, + listAutomationRunsForTarget, + type AutomationHostTarget +} from './automation-host-client' +import type { AutomationListRow } from './automation-list-row-identity' +import { + dispatchAutomationRunHistoryPage, + type AutomationDispatchContext +} from './automation-row-action-dispatch' +import { + buildAutomationRunsDashboardEntries, + getAutomationRunsScope, + type AutomationRunsDashboardFailure +} from './automation-runs-dashboard-model' + +const FETCH_CONCURRENCY = 4 +// The persistence contract retains at most 100 final runs per automation; +// fetching that bound keeps summary cards complete without an extra scan. +const RUNS_PAGE_SIZE = 100 + +type DashboardState = { + entries: ReturnType + failures: AutomationRunsDashboardFailure[] + loading: boolean + nextCursors: ReadonlyMap + hasMore: boolean + loadMore: () => void +} + +const EMPTY_STATE: DashboardState = { + entries: [], + failures: [], + loading: false, + nextCursors: new Map(), + hasMore: false, + loadMore: () => undefined +} + +export function useAutomationRunsDashboard({ + enabled, + rows, + context, + legacyTarget, + authorityForRow, + reloadToken +}: { + enabled: boolean + rows: readonly AutomationListRow[] + context: AutomationDispatchContext + legacyTarget: (row: AutomationListRow) => AutomationHostTarget | null + authorityForRow: (row: AutomationListRow) => AutomationAuthorityRef + reloadToken: number +}): DashboardState { + const inputRef = useRef({ rows, context, legacyTarget, authorityForRow }) + useEffect(() => { + inputRef.current = { rows, context, legacyTarget, authorityForRow } + }, [authorityForRow, context, legacyTarget, rows]) + // Keys the effective request, not just the row: a re-pair bumps the authority's + // pairing revision and an uncaptured row's fallback target can move, and either + // makes the entries and cursors already on screen belong to a different host. + const queryKey = useMemo( + () => + rows + .map((row) => + [ + row.key, + row.automation.updatedAt, + capturedAutomationOwnerKey(capturedAutomationOwner(context.capturedOwners, row.key)), + ownerKey({ authority: authorityForRow(row), selector: { kind: 'self' } }), + getAutomationHostTargetKey(legacyTarget(row) ?? { kind: 'local' }) + ].join(':') + ) + .join('|'), + [authorityForRow, context.capturedOwners, legacyTarget, rows] + ) + const [state, setState] = useState(EMPTY_STATE) + const stateRef = useRef(state) + useEffect(() => { + stateRef.current = state + }, [state]) + const [loadMoreToken, setLoadMoreToken] = useState(0) + const loadMore = useCallback(() => setLoadMoreToken((token) => token + 1), []) + // Null while disabled: a fresh re-entry must never resume from the previous + // session's cursors, however many times load-more fired before it. + const generationRef = useRef<{ + queryKey: string + reloadToken: number + loadMoreToken: number + } | null>(null) + + useEffect(() => { + if (!enabled) { + generationRef.current = null + return + } + const input = inputRef.current + let cancelled = false + const previous = generationRef.current + const loadingMore = + previous !== null && + previous.queryKey === queryKey && + previous.reloadToken === reloadToken && + previous.loadMoreToken !== loadMoreToken && + stateRef.current.entries.length > 0 + generationRef.current = { queryKey, reloadToken, loadMoreToken } + const runsByRowKey = new Map() + if (loadingMore) { + for (const entry of stateRef.current.entries) { + const current = runsByRowKey.get(entry.row.key) ?? [] + current.push(entry.run) + runsByRowKey.set(entry.row.key, current) + } + } + const nextCursors = new Map() + setState((current) => + loadingMore ? { ...current, loading: true } : { ...EMPTY_STATE, loading: true, loadMore } + ) + const failures: AutomationRunsDashboardFailure[] = loadingMore + ? [...stateRef.current.failures] + : [] + let nextIndex = 0 + const fetchNext = async (): Promise => { + while (!cancelled && nextIndex < input.rows.length) { + const row = input.rows[nextIndex++] + const cursor = loadingMore ? stateRef.current.nextCursors.get(row.key) : undefined + if (loadingMore && !cursor) { + continue + } + const result = await dispatchAutomationRunHistoryPage( + input.context, + { rowKey: row.key, automationId: row.automation.id }, + { limit: RUNS_PAGE_SIZE, ...(cursor ? { cursor } : {}) }, + async () => ({ + runs: await listAutomationRunsForTarget( + input.legacyTarget(row) ?? { kind: 'local' }, + row.automation.id + ), + nextCursor: null + }), + input.authorityForRow(row) + ) + if (result.ok) { + const current = runsByRowKey.get(row.key) ?? [] + const seen = new Set(current.map((run) => run.id)) + runsByRowKey.set(row.key, [ + ...current, + ...result.value.runs.filter((run) => !seen.has(run.id)) + ]) + if (result.value.nextCursor) { + nextCursors.set(row.key, result.value.nextCursor) + } + } else { + // Only a successful terminal page retires a cursor; keeping it here + // leaves the row's remaining history reachable through `loadMore`. + if (cursor) { + nextCursors.set(row.key, cursor) + } + failures.push({ row, scope: getAutomationRunsScope(row), notice: result.notice }) + } + } + } + void Promise.all( + Array.from({ length: Math.min(FETCH_CONCURRENCY, input.rows.length) }, fetchNext) + ).then(() => { + if (!cancelled) { + setState({ + entries: buildAutomationRunsDashboardEntries(input.rows, runsByRowKey), + failures, + loading: false, + nextCursors, + hasMore: nextCursors.size > 0, + loadMore + }) + } + }) + return () => { + cancelled = true + } + }, [enabled, loadMore, loadMoreToken, queryKey, reloadToken]) + + return enabled ? { ...state, loadMore } : { ...EMPTY_STATE, loadMore } +} diff --git a/src/renderer/src/components/automations/use-automations-page-controller.ts b/src/renderer/src/components/automations/use-automations-page-controller.ts index f47e2378bb4..70214a367b4 100644 --- a/src/renderer/src/components/automations/use-automations-page-controller.ts +++ b/src/renderer/src/components/automations/use-automations-page-controller.ts @@ -16,12 +16,21 @@ import { useAutomationsPageRefresh } from './use-automations-page-refresh' import { useAutomationsPageSetupState } from './use-automations-page-setup-state' import { useAutomationsPageStoreState } from './use-automations-page-store-state' import { useExternalAutomationActions } from './use-external-automation-actions' +import { useAutomationRunsDashboard } from './use-automation-runs-dashboard' export function useAutomationsPageController() { const store = useAutomationsPageStoreState() const local = useAutomationsPageLocalState(store) const list = useAutomationsPageListState({ store, local }) const destination = useAutomationsPageDestinationState({ store, local, list }) + const runsDashboard = useAutomationRunsDashboard({ + enabled: local.pageView === 'runs', + rows: list.visibleRows, + context: destination.automationDispatchContext, + legacyTarget: destination.automationHostTargetFor, + authorityForRow: destination.automationAuthorityForRow, + reloadToken: local.runHistoryReloadToken + }) const destinationForm = useAutomationsPageDestinationForm({ store, local, @@ -90,6 +99,7 @@ export function useAutomationsPageController() { local, list, destination, + runsDashboard, destinationForm, setup, runPage, diff --git a/src/renderer/src/components/automations/use-automations-page-destination-state.ts b/src/renderer/src/components/automations/use-automations-page-destination-state.ts index f69374308f2..947d28182ef 100644 --- a/src/renderer/src/components/automations/use-automations-page-destination-state.ts +++ b/src/renderer/src/components/automations/use-automations-page-destination-state.ts @@ -123,6 +123,23 @@ export function useAutomationsPageDestinationState({ (row: { key: string }): AutomationHostTarget | null => automationHostTargetForRowKey(row.key), [automationHostTargetForRowKey] ) + const automationAuthorityForRow = useCallback( + (row: { catalogRef?: StableAutomationCatalogRef | null }): AutomationAuthorityRef => { + const authority = row.catalogRef?.authority + if (authority?.kind === 'runtime') { + return { + kind: 'runtime', + environmentId: authority.environmentId, + pairingRevision: automationRuntimePairingRevision( + runtimeEnvironments, + authority.environmentId + ) + } + } + return { kind: 'desktop' } + }, + [runtimeEnvironments] + ) const automationDispatchContext = useMemo( () => ({ capturedOwners: capturedAutomationOwners, authority: automationAuthority }), [automationAuthority, capturedAutomationOwners] @@ -198,6 +215,7 @@ export function useAutomationsPageDestinationState({ createDestinationHostId, automationHostTargetForRowKey, automationHostTargetFor, + automationAuthorityForRow, automationDispatchContext, rowRecoveryHost, reportOwnerAction, diff --git a/src/renderer/src/components/automations/use-automations-page-escape.ts b/src/renderer/src/components/automations/use-automations-page-escape.ts index 03f1f239f73..be40b8b3fe0 100644 --- a/src/renderer/src/components/automations/use-automations-page-escape.ts +++ b/src/renderer/src/components/automations/use-automations-page-escape.ts @@ -18,6 +18,9 @@ export function useAutomationsPageEscape({ isDetailOpen, selectedAutomationRunPageId, selectedExternalRunPage, + pageView, + runPageOrigin, + setPageView, setActivePaneTab, setIsDetailOpen, setSelectedAutomationRunPageId, @@ -61,6 +64,15 @@ export function useAutomationsPageEscape({ } } + if (pageView === 'run') { + event.preventDefault() + setSelectedAutomationRunPageId(null) + setPageView(runPageOrigin === 'automation' ? 'automations' : 'runs') + setIsDetailOpen(runPageOrigin === 'automation') + setActivePaneTab(runPageOrigin === 'automation' ? 'runs' : 'overview') + return + } + if (isDetailOpen) { event.preventDefault() if (selectedExternalRunPage) { @@ -76,6 +88,12 @@ export function useAutomationsPageEscape({ return } + if (pageView === 'runs') { + event.preventDefault() + setPageView('automations') + return + } + event.preventDefault() closeAutomationsPage() } @@ -89,11 +107,14 @@ export function useAutomationsPageEscape({ deleteTarget, externalDeleteTarget, isDetailOpen, + pageView, + runPageOrigin, selectedAutomationRunPageId, selectedExternalRunPage, setActivePaneTab, setIsDetailOpen, setSelectedAutomationRunPageId, - setSelectedExternalRunPage + setSelectedExternalRunPage, + setPageView ]) } diff --git a/src/renderer/src/components/automations/use-automations-page-local-state.ts b/src/renderer/src/components/automations/use-automations-page-local-state.ts index 25f4fa30444..e92f666cb22 100644 --- a/src/renderer/src/components/automations/use-automations-page-local-state.ts +++ b/src/renderer/src/components/automations/use-automations-page-local-state.ts @@ -13,7 +13,12 @@ import type { AutomationHostCatalogEntry } from './automation-host-catalog-types import type { AutomationCreateDestination } from './automation-create-destination' import type { AutomationListRow } from './automation-list-row-identity' import { EMPTY_AUTOMATION_LIST_FILTER, type AutomationListFilter } from './automation-list-view' -import type { AutomationPaneTab, SelectedExternalRunPage } from './automation-page-state' +import type { + AutomationPaneTab, + AutomationRunPageOrigin, + AutomationsPageView, + SelectedExternalRunPage +} from './automation-page-state' import type { ExternalAutomationScope } from './external-automation-scope-client' import type { SelectedAutomationRunHistoryOutcome } from './use-selected-automation-run-history' import type { AutomationsPageStoreState } from './use-automations-page-store-state' @@ -60,6 +65,8 @@ export function useAutomationsPageLocalState(store: AutomationsPageStoreState) { const [editingHostStableKey, setEditingHostStableKey] = useState(null) const moveCreationKeysRef = useRef(new Map()) const [relativeNow, setRelativeNow] = useState(() => Date.now()) + const [pageView, setPageView] = useState('automations') + const [runPageOrigin, setRunPageOrigin] = useState('runs') const [activePaneTab, setActivePaneTab] = useState('overview') const [selectedAutomationRunPageId, setSelectedAutomationRunPageId] = useState( null @@ -186,6 +193,10 @@ export function useAutomationsPageLocalState(store: AutomationsPageStoreState) { moveCreationKeysRef, relativeNow, setRelativeNow, + pageView, + setPageView, + runPageOrigin, + setRunPageOrigin, activePaneTab, setActivePaneTab, selectedAutomationRunPageId, diff --git a/src/renderer/src/components/automations/use-selected-automation-run-history.ts b/src/renderer/src/components/automations/use-selected-automation-run-history.ts index 2f1608f639a..a7b7dae6799 100644 --- a/src/renderer/src/components/automations/use-selected-automation-run-history.ts +++ b/src/renderer/src/components/automations/use-selected-automation-run-history.ts @@ -8,8 +8,9 @@ * the host a navigation named — so those, and only those, are the key. */ -import { useEffect, useRef } from 'react' +import { useEffect, useMemo, useRef } from 'react' import type { AutomationRun } from '../../../../shared/automations-types' +import type { AutomationAuthorityRef } from '../../../../shared/automation-owner-ref' import { capturedAutomationOwner, capturedAutomationOwnerKey } from './automation-captured-owner' import type { AutomationListRow } from './automation-list-row-identity' import { @@ -75,6 +76,14 @@ export function useSelectedAutomationRunHistory(input: SelectedAutomationRunHist navigationHostId(input.navigation, automationId) ?? '' ].join('|') : '' + const captured = input.selected + ? capturedAutomationOwner(input.context.capturedOwners, input.selected.key).owner + : null + const authority = captured?.authority ?? input.context.authority + const rowAuthority = useMemo( + () => (authority?.kind === 'runtime' ? authority : { kind: 'desktop' }), + [authority] + ) useEffect(() => { const { selected, context, legacyTarget, navigation, onSettled } = inputRef.current @@ -90,8 +99,11 @@ export function useSelectedAutomationRunHistory(input: SelectedAutomationRunHist const target = navigationHost ? getAutomationTargetFromHostId(navigationHost) : (legacyTarget(selected) ?? { kind: 'local' }) - void dispatchAutomationRunHistory(context, { rowKey, automationId }, () => - listAutomationRunsForTarget(target, automationId) + void dispatchAutomationRunHistory( + context, + { rowKey, automationId }, + () => listAutomationRunsForTarget(target, automationId), + rowAuthority ).then((result) => { if (cancelled) { return @@ -107,5 +119,5 @@ export function useSelectedAutomationRunHistory(input: SelectedAutomationRunHist return () => { cancelled = true } - }, [automationId, rowKey, fetchKey, reloadToken]) + }, [automationId, rowAuthority, rowKey, fetchKey, reloadToken]) } diff --git a/src/renderer/src/components/gitlab-item-dialog/use-gitlab-review-actions.ts b/src/renderer/src/components/gitlab-item-dialog/use-gitlab-review-actions.ts index 9b56234ffc3..e560fa16799 100644 --- a/src/renderer/src/components/gitlab-item-dialog/use-gitlab-review-actions.ts +++ b/src/renderer/src/components/gitlab-item-dialog/use-gitlab-review-actions.ts @@ -49,7 +49,14 @@ export function useGitLabReviewActions( setReviewerOptionsLoading(false) } } - }, [mountedRef, repoSelector, reviewerOptions, reviewerOptionsLoading]) + }, [ + mountedRef, + repoSelector, + reviewerOptions, + reviewerOptionsLoading, + setReviewerOptions, + setReviewerOptionsLoading + ]) const handleSetReviewers = useCallback( async (nextReviewers: GitLabAssignableUser[]): Promise => { @@ -101,7 +108,16 @@ export function useGitLabReviewActions( } } }, - [details, item, mountedRef, repoSelector] + [ + details, + item, + mountedRef, + repoSelector, + setDetails, + setReviewerDraftId, + setReviewerOptions, + setReviewerUpdating + ] ) const handleSubmitInlineComment = useCallback(async (): Promise => { @@ -185,7 +201,10 @@ export function useGitLabReviewActions( inlineCommentLine, item, mountedRef, - repoSelector + repoSelector, + setDetails, + setInlineCommentBody, + setInlineCommentSubmitting ]) const handleResolveDiscussion = useCallback( @@ -228,7 +247,7 @@ export function useGitLabReviewActions( } } }, - [item, repoSelector, mountedRef] + [item, repoSelector, mountedRef, setDetails, setResolvingThreadId] ) return { diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index f1489d6754d..93ba2bbd96f 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -15887,6 +15887,35 @@ "noProjects": "No projects are set up on {host}. Add one there, or choose another host.", "updateRequired": "Update the Orca server on {hosts} to store automations there.", "move": "Saving creates this automation on {host} and deletes the original and its run history." + }, + "AutomationListToolbar": { + "runs": "Runs" + }, + "AutomationRunsDashboard": { + "search": "Search runs…", + "filters": "Filters", + "host": "Host", + "status": "Status", + "refresh": "Refresh runs", + "successful24h": "Successful · 24h", + "failed24h": "Failed · 24h", + "successful7d": "Successful · 7d", + "failed7d": "Failed · 7d", + "historyUnavailableOne": "Run history is unavailable for 1 automation. Counts include available history only.", + "historyUnavailableMany": "Run history is unavailable for {{count}} automations. Counts include available history only.", + "automation": "Automation", + "triggered": "Triggered", + "trigger": "Trigger", + "loading": "Loading runs…", + "noRuns": "No runs yet", + "emptyDescription": "Runs appear here after an automation is triggered.", + "runs": "Runs", + "local": "Local", + "remote": "Remote" + }, + "AutomationsPageBreadcrumb": { + "ariaLabel": "Automations breadcrumb", + "runDetails": "Run details" } }, "agent": { diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 79731f3b096..f4a4fd931de 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -13719,6 +13719,35 @@ } }, "automations": { + "AutomationListToolbar": { + "runs": "Ejecuciones" + }, + "AutomationRunsDashboard": { + "search": "Buscar ejecuciones…", + "filters": "Filtros", + "host": "Host", + "status": "Estado", + "refresh": "Actualizar ejecuciones", + "successful24h": "Exitosas · 24 h", + "failed24h": "Fallidas · 24 h", + "successful7d": "Exitosas · 7 días", + "failed7d": "Fallidas · 7 días", + "historyUnavailableOne": "El historial de ejecuciones no está disponible para 1 automatización. Los recuentos solo incluyen el historial disponible.", + "historyUnavailableMany": "El historial de ejecuciones no está disponible para {{count}} automatizaciones. Los recuentos solo incluyen el historial disponible.", + "automation": "Automatización", + "triggered": "Activada", + "trigger": "Activar", + "loading": "Cargando ejecuciones…", + "noRuns": "Aún no hay ejecuciones", + "emptyDescription": "Las ejecuciones aparecerán aquí después de activar una automatización.", + "runs": "Ejecuciones", + "local": "Local", + "remote": "Remoto" + }, + "AutomationsPageBreadcrumb": { + "ariaLabel": "Ruta de navegación de automatizaciones", + "runDetails": "Detalles de la ejecución" + }, "AutomationCustomCronPanel": { "3e3b2c369f": "expresión cron", "e81a02d61b": "Ingrese un cron válido de cinco campos antes de guardar.", diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index 6afc506b12b..ee6e4440f79 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -13719,6 +13719,35 @@ } }, "automations": { + "AutomationListToolbar": { + "runs": "実行" + }, + "AutomationRunsDashboard": { + "search": "実行を検索…", + "filters": "フィルター", + "host": "ホスト", + "status": "ステータス", + "refresh": "実行を更新", + "successful24h": "成功 · 24時間", + "failed24h": "失敗 · 24時間", + "successful7d": "成功 · 7日間", + "failed7d": "失敗 · 7日間", + "historyUnavailableOne": "1件の自動化で実行履歴を利用できません。カウントには利用可能な履歴のみが含まれます。", + "historyUnavailableMany": "{{count}}件の自動化で実行履歴を利用できません。カウントには利用可能な履歴のみが含まれます。", + "automation": "自動化", + "triggered": "トリガー済み", + "trigger": "トリガー", + "loading": "実行を読み込み中…", + "noRuns": "実行はまだありません", + "emptyDescription": "自動化がトリガーされると、ここに実行が表示されます。", + "runs": "実行", + "local": "ローカル", + "remote": "リモート" + }, + "AutomationsPageBreadcrumb": { + "ariaLabel": "自動化のパンくずリスト", + "runDetails": "実行の詳細" + }, "AutomationCustomCronPanel": { "3e3b2c369f": "クロン式", "e81a02d61b": "保存する前に、有効な 5 フィールドの cron を入力してください。", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index c2836df4c40..76ed151ca2c 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -13775,6 +13775,35 @@ } }, "automations": { + "AutomationListToolbar": { + "runs": "실행" + }, + "AutomationRunsDashboard": { + "search": "실행 검색…", + "filters": "필터", + "host": "호스트", + "status": "상태", + "refresh": "실행 새로 고침", + "successful24h": "성공 · 24시간", + "failed24h": "실패 · 24시간", + "successful7d": "성공 · 7일", + "failed7d": "실패 · 7일", + "historyUnavailableOne": "자동화 1개의 실행 기록을 사용할 수 없습니다. 집계에는 사용 가능한 기록만 포함됩니다.", + "historyUnavailableMany": "자동화 {{count}}개의 실행 기록을 사용할 수 없습니다. 집계에는 사용 가능한 기록만 포함됩니다.", + "automation": "자동화", + "triggered": "트리거됨", + "trigger": "트리거", + "loading": "실행 로드 중…", + "noRuns": "아직 실행이 없습니다", + "emptyDescription": "자동화를 트리거하면 여기에 실행이 표시됩니다.", + "runs": "실행", + "local": "로컬", + "remote": "원격" + }, + "AutomationsPageBreadcrumb": { + "ariaLabel": "자동화 탐색경로", + "runDetails": "실행 세부정보" + }, "AutomationCustomCronPanel": { "3e3b2c369f": "Cron 식", "e81a02d61b": "저장하기 전에 유효한 5개 필드 크론을 입력하세요.", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 77f56e805ae..5c1324afa12 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -13775,6 +13775,35 @@ } }, "automations": { + "AutomationListToolbar": { + "runs": "运行" + }, + "AutomationRunsDashboard": { + "search": "搜索运行…", + "filters": "筛选条件", + "host": "主机", + "status": "状态", + "refresh": "刷新运行", + "successful24h": "成功 · 24 小时", + "failed24h": "失败 · 24 小时", + "successful7d": "成功 · 7 天", + "failed7d": "失败 · 7 天", + "historyUnavailableOne": "1 个自动化的运行历史不可用。计数仅包含可用的历史记录。", + "historyUnavailableMany": "{{count}} 个自动化的运行历史不可用。计数仅包含可用的历史记录。", + "automation": "自动化", + "triggered": "已触发", + "trigger": "触发", + "loading": "正在加载运行…", + "noRuns": "尚无运行", + "emptyDescription": "触发自动化后,运行记录会显示在这里。", + "runs": "运行", + "local": "本地", + "remote": "远程" + }, + "AutomationsPageBreadcrumb": { + "ariaLabel": "自动化面包屑导航", + "runDetails": "运行详情" + }, "AutomationCustomCronPanel": { "3e3b2c369f": "cron", "e81a02d61b": "保存前输入有效的五字段 cron。", diff --git a/src/shared/automation-run-cursor.ts b/src/shared/automation-run-cursor.ts new file mode 100644 index 00000000000..23bcaccc343 --- /dev/null +++ b/src/shared/automation-run-cursor.ts @@ -0,0 +1,75 @@ +import type { AutomationRun, AutomationRunsPage } from './automations-types' + +const MAX_PAGE_SIZE = 100 + +/** A keyset boundary (`createdAt:id` of the previous page's last run), or the + * bare offset older hosts emitted — still read so a cursor issued before an + * upgrade keeps working. */ +type AutomationRunCursor = + | { kind: 'key'; createdAt: number; id: string } + | { kind: 'offset'; offset: number } + +function decodeAutomationRunCursor(cursor: string | undefined): AutomationRunCursor | null { + if (!cursor) { + return null + } + const separator = cursor.indexOf(':') + if (separator === -1) { + const offset = Number.parseInt(cursor, 10) + return Number.isFinite(offset) && offset > 0 ? { kind: 'offset', offset } : null + } + const createdAt = Number.parseInt(cursor.slice(0, separator), 10) + const id = cursor.slice(separator + 1) + return Number.isFinite(createdAt) && id ? { kind: 'key', createdAt, id } : null +} + +/** The single total order pages and cursors agree on. Ties on `createdAt` fall + * back to `id`, so a pruned boundary cannot take the runs tied with it. */ +export function compareAutomationRunsNewestFirst( + left: Pick, + right: Pick +): number { + return right.createdAt - left.createdAt || left.id.localeCompare(right.id) +} + +function pageStartIndex( + runs: readonly AutomationRun[], + cursor: AutomationRunCursor | null +): number { + if (!cursor) { + return 0 + } + if (cursor.kind === 'offset') { + return Math.min(cursor.offset, runs.length) + } + const boundary = runs.findIndex( + (run) => run.id === cursor.id && run.createdAt === cursor.createdAt + ) + if (boundary !== -1) { + return boundary + 1 + } + // Boundary run pruned between pages: resume at the first run the total order + // places after it, so runs tied on `createdAt` are not dropped with it. + const older = runs.findIndex((run) => compareAutomationRunsNewestFirst(cursor, run) < 0) + return older === -1 ? runs.length : older +} + +/** + * Pages `runs`, which must already be sorted by `compareAutomationRunsNewestFirst`. + * The cursor names the previous page's last run rather than an index, so runs + * created between two page requests cannot shift the window and drop a run. + */ +export function paginateAutomationRuns( + runs: readonly AutomationRun[], + limit?: number, + cursor?: string +): AutomationRunsPage { + const start = pageStartIndex(runs, decodeAutomationRunCursor(cursor)) + const boundedLimit = Math.min(Math.max(1, limit ?? 100), MAX_PAGE_SIZE) + const page = runs.slice(start, start + boundedLimit) + const last = page.at(-1) + return { + runs: page, + nextCursor: last && start + page.length < runs.length ? `${last.createdAt}:${last.id}` : null + } +} diff --git a/src/shared/automations-types.ts b/src/shared/automations-types.ts index 24d9345242d..80a56e3cdeb 100644 --- a/src/shared/automations-types.ts +++ b/src/shared/automations-types.ts @@ -170,6 +170,12 @@ export type AutomationRun = { lastOccurrenceAt?: number } +/** A bounded history response; older hosts may continue returning `runs` only. */ +export type AutomationRunsPage = { + runs: AutomationRun[] + nextCursor: string | null +} + export type AutomationCreateInput = { /** Optional idempotency key; repeated creates return the original record. */ creationKey?: string diff --git a/tests/e2e/automation-runs-dashboard.spec.ts b/tests/e2e/automation-runs-dashboard.spec.ts new file mode 100644 index 00000000000..ac0154701b8 --- /dev/null +++ b/tests/e2e/automation-runs-dashboard.spec.ts @@ -0,0 +1,44 @@ +/** + * End-to-end coverage for the Automations runs surface. + * + * The test intentionally does not depend on seeded run history: a fresh E2E + * profile may have no automations, but the Runs navigation and empty state must + * still be usable. + */ + +import { test, expect } from './helpers/orca-app' +import { waitForSessionReady } from './helpers/store' + +test('opens the runs dashboard and returns to automations', async ({ orcaPage }) => { + await waitForSessionReady(orcaPage) + + await orcaPage.evaluate(() => { + const store = window.__store + if (!store) { + throw new Error('window.__store is not available') + } + store.getState().openAutomationsPage() + }) + + const runsButton = orcaPage.getByRole('button', { name: 'Runs' }) + await expect(runsButton).toBeVisible() + await runsButton.click() + + await expect(orcaPage.getByRole('navigation', { name: 'Automations breadcrumb' })).toBeVisible() + await expect(orcaPage.getByText('Successful · 24h')).toBeVisible() + await expect(orcaPage.getByText('Failed · 24h')).toBeVisible() + await expect(orcaPage.getByText('Successful · 7d')).toBeVisible() + await expect(orcaPage.getByText('Failed · 7d')).toBeVisible() + await expect(orcaPage.getByRole('button', { name: 'Filters' })).toBeVisible() + await expect(orcaPage.getByRole('button', { name: 'Refresh runs' })).toBeVisible() + await expect(orcaPage.getByText('Automation', { exact: true })).toBeVisible() + await expect(orcaPage.getByText('Triggered', { exact: true })).toBeVisible() + await expect(orcaPage.getByText('Status', { exact: true })).toBeVisible() + + await orcaPage + .getByRole('navigation', { name: 'Automations breadcrumb' }) + .getByRole('button', { name: 'Automations' }) + .click() + await expect(orcaPage.getByRole('heading', { name: 'Automations' })).toBeVisible() + await expect(runsButton).toBeVisible() +}) From c2fce80289622e965a281094e5b21cf0e5ddb28a Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 13:43:06 -0700 Subject: [PATCH 099/398] Fix agent dashboard setting configure (#18245) * Make agents activity always-on; toggle via bell icon - Remove optional showAgentsSidebar setting - Replace sidebar view-toggle with bell-button for activity access - Agents activity now always accessible in sidebar - Preserve migration flag for introduction to existing users - Remove visibility inference utilities * Simplify sidebar when agents view active: hide workspace options, add to - Hide workspace options menu and add project button when agents view is active, reducing UI clutter in that mode - Add tooltip to the activity bell button for better discoverability - Localize sidebar search field text - Move search and filter toggles to local state in SidebarAgentsList, removing unused callbacks from thread list components - Manage search input focus properly when opening --- .../terminal-settings-migrations.ts | 3 + .../normalize-loaded-global-settings.test.ts | 61 +---- .../normalize-loaded-global-settings.ts | 9 +- .../prepare-loaded-profile-settings.ts | 2 +- .../activity-scope-filter-controls.tsx | 5 +- .../activity/activity-thread-list-pane.tsx | 3 + .../activity/activity-thread-list-toolbar.tsx | 14 +- .../activity/activity-thread-options-menu.tsx | 13 +- .../settings/ExperimentalPane.test.tsx | 19 +- .../components/settings/ExperimentalPane.tsx | 30 --- .../settings/appearance-sidebar-search.ts | 21 -- .../settings/experimental-search.ts | 3 - .../src/components/sidebar/Sidebar.test.tsx | 43 +--- .../components/sidebar/SidebarAgentsList.tsx | 54 ++++- .../components/sidebar/SidebarHeader.test.tsx | 228 ++++-------------- .../src/components/sidebar/SidebarHeader.tsx | 222 ++++++----------- .../components/sidebar/SidebarNav.test.tsx | 1 - .../sidebar/agents-sidebar-visibility.test.ts | 15 -- .../sidebar/agents-sidebar-visibility.ts | 11 - src/renderer/src/components/sidebar/index.tsx | 131 +--------- .../sidebar/sidebar-header-actions.tsx | 49 ++-- .../sidebar/sidebar-view-toggle.test.tsx | 145 ----------- .../sidebar/sidebar-view-toggle.tsx | 106 -------- src/renderer/src/i18n/locales/en.json | 4 - .../slices/ui-hydration-view-layout.test.ts | 22 -- .../store/slices/ui-page-navigation.test.ts | 17 -- .../slices/ui/ui-slice-hydration-actions.ts | 4 +- .../ui/ui-slice-hydration-sanitizers.ts | 13 +- .../slices/ui/ui-slice-settings-actions.ts | 10 +- .../store/slices/ui/ui-slice-view-actions.ts | 7 - src/shared/agents-sidebar-visibility.ts | 12 - src/shared/default-global-settings.ts | 1 - src/shared/global-settings-types.ts | 6 +- src/shared/telemetry-property-schemas.ts | 1 - .../e2e/activity-agent-pane-isolation.spec.ts | 1 - 35 files changed, 236 insertions(+), 1050 deletions(-) delete mode 100644 src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts delete mode 100644 src/renderer/src/components/sidebar/agents-sidebar-visibility.ts delete mode 100644 src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx delete mode 100644 src/renderer/src/components/sidebar/sidebar-view-toggle.tsx delete mode 100644 src/shared/agents-sidebar-visibility.ts diff --git a/src/main/persistence/applying-settings/terminal-settings-migrations.ts b/src/main/persistence/applying-settings/terminal-settings-migrations.ts index f8edf0ee466..699189d1390 100644 --- a/src/main/persistence/applying-settings/terminal-settings-migrations.ts +++ b/src/main/persistence/applying-settings/terminal-settings-migrations.ts @@ -59,6 +59,7 @@ export function readLegacyTerminalScrollbackSettings( type RetiredGlobalSettings = { terminalScrollbackBytes?: unknown enableGitHubAttribution?: unknown + showAgentsSidebar?: unknown } export function stripRetiredGlobalSettings( @@ -67,10 +68,12 @@ export function stripRetiredGlobalSettings( const { terminalScrollbackBytes: _legacyScrollbackBytes, enableGitHubAttribution: _legacyGitHubAttribution, + showAgentsSidebar: _legacyShowAgentsSidebar, ...rest } = (settings ?? {}) as Partial & RetiredGlobalSettings void _legacyScrollbackBytes void _legacyGitHubAttribution + void _legacyShowAgentsSidebar return rest } diff --git a/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts b/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts index 6d1040f09ae..4c463959bba 100644 --- a/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts +++ b/src/main/persistence/loading-store/normalize-loaded-global-settings.test.ts @@ -8,10 +8,9 @@ import type { GlobalSettings } from '../../../shared/global-settings-types' import type { PersistedState } from '../../../shared/persisted-state-types' // Simulates a profile created before the dedicated Experimental switch was persisted. -function normalizeLegacyProfile(overrides: Partial): PersistedState['settings'] { +function normalizeLegacyProfile(overrides: Record): PersistedState['settings'] { const defaults = getDefaultPersistedState(homedir()) const settings: Partial = { ...defaults.settings } - delete settings.showAgentsSidebar delete settings.experimentalActivity delete settings.experimentalAgentDashboardPopout Object.assign(settings, overrides) @@ -22,59 +21,17 @@ function normalizeLegacyProfile(overrides: Partial): PersistedSt return normalizeLoadedGlobalSettings(parsed, terminal, profile) } -describe('showAgentsSidebar experimental-setting migration', () => { - it('keeps the sidebar for Agents-view opt-ins regardless of the dashboard experiment', () => { +describe('retired Agents sidebar setting', () => { + it('does not mark new profiles as migrated', () => { + expect(normalizeLegacyProfile({}).agentsSidebarMigratedFromExperimental).toBe(false) + }) + + it('drops the old visibility setting while preserving migration metadata', () => { const normalized = normalizeLegacyProfile({ experimentalActivity: true, - experimentalAgentDashboardPopout: false + showAgentsSidebar: false }) - expect(normalized.showAgentsSidebar).toBe(true) + expect('showAgentsSidebar' in normalized).toBe(false) expect(normalized.agentsSidebarMigratedFromExperimental).toBe(true) }) - - it('carries the legacy Agents-view opt-in into the sidebar', () => { - expect(normalizeLegacyProfile({ experimentalActivity: true }).showAgentsSidebar).toBe(true) - }) - - it('does not show Agents migration copy for a dashboard-only opt-in', () => { - expect( - normalizeLegacyProfile({ experimentalAgentDashboardPopout: true }) - .agentsSidebarMigratedFromExperimental - ).toBe(false) - }) - - it('defaults profiles with no legacy signal to the sidebar', () => { - const normalized = normalizeLegacyProfile({}) - expect(normalized.showAgentsSidebar).toBe(true) - expect(normalized.agentsSidebarMigratedFromExperimental).toBe(false) - }) - - it('does not treat a dashboard opt-out as an Agents-tab opt-out', () => { - expect( - normalizeLegacyProfile({ experimentalAgentDashboardPopout: false }).showAgentsSidebar - ).toBe(true) - }) - - it('ignores a pre-stamp forced-default experimentalActivity true (not an opt-in)', () => { - const normalized = normalizeLegacyProfile({ - experimentalActivity: true, - experimentalActivityDefaultedOffForAllUsers: undefined - }) - expect(normalized.experimentalActivity).toBe(false) - expect(normalized.showAgentsSidebar).toBe(true) - expect(normalized.agentsSidebarMigratedFromExperimental).toBe(false) - }) - - it('preserves a stored showAgentsSidebar choice over legacy flags', () => { - expect( - normalizeLegacyProfile({ showAgentsSidebar: false, experimentalActivity: true }) - .showAgentsSidebar - ).toBe(false) - expect( - normalizeLegacyProfile({ - showAgentsSidebar: true, - experimentalAgentDashboardPopout: false - }).showAgentsSidebar - ).toBe(true) - }) }) diff --git a/src/main/persistence/loading-store/normalize-loaded-global-settings.ts b/src/main/persistence/loading-store/normalize-loaded-global-settings.ts index 9370ec4febc..90678e9f822 100644 --- a/src/main/persistence/loading-store/normalize-loaded-global-settings.ts +++ b/src/main/persistence/loading-store/normalize-loaded-global-settings.ts @@ -1,5 +1,4 @@ import { getDefaultVoiceSettings } from '../../../shared/constants' -import { resolveAgentsSidebarVisible } from '../../../shared/agents-sidebar-visibility' import { normalizePRBotAuthorOverrides } from '../../../shared/pr-bot-author-overrides' import { normalizeTerminalQuickCommands } from '../../../shared/terminal-quick-commands' import { normalizeOpenInApplications } from '../../../shared/open-in-applications' @@ -86,13 +85,7 @@ export function normalizeLoadedGlobalSettings( ...migratedTerminalTuiScrollSensitivity.settings, experimentalActivity: migratedExperimentalActivity, experimentalActivityDefaultedOffForAllUsers: true, - // Keep the experimental Agents tab's rollout default for older profiles while - // preserving any choice made through its dedicated Experimental setting. - showAgentsSidebar: resolveAgentsSidebarVisible({ - showAgentsSidebar: parsed.settings?.showAgentsSidebar - }), - // Preserve the legacy opt-in before the experimental setting is normalized away. This - // drives the migration-specific introduction copy without changing runtime behavior. + // Preserve the legacy opt-in so the one-time introduction copy can target existing users. agentsSidebarMigratedFromExperimental: parsed.settings?.agentsSidebarMigratedFromExperimental === true || migratedExperimentalActivity, diff --git a/src/main/persistence/loading-store/prepare-loaded-profile-settings.ts b/src/main/persistence/loading-store/prepare-loaded-profile-settings.ts index 2410e68be54..69e0d28d8c5 100644 --- a/src/main/persistence/loading-store/prepare-loaded-profile-settings.ts +++ b/src/main/persistence/loading-store/prepare-loaded-profile-settings.ts @@ -62,7 +62,7 @@ export function prepareLoadedProfileSettings( ): PreparedLoadedProfileSettings { const experimentalActivityDefaultedOffForAllUsers = parsed.settings?.experimentalActivityDefaultedOffForAllUsers === true - // Why: the Agents view moved back behind Experimental; flip pre-migration profiles off once, then preserve opt-ins. + // Why: preserve the legacy rollout boundary while loading profiles created before the Agents tab graduated. const migratedExperimentalActivity = experimentalActivityDefaultedOffForAllUsers ? (parsed.settings?.experimentalActivity ?? false) : false diff --git a/src/renderer/src/components/activity/activity-scope-filter-controls.tsx b/src/renderer/src/components/activity/activity-scope-filter-controls.tsx index 6256595ba6b..7da07182062 100644 --- a/src/renderer/src/components/activity/activity-scope-filter-controls.tsx +++ b/src/renderer/src/components/activity/activity-scope-filter-controls.tsx @@ -1,7 +1,7 @@ import React, { useMemo } from 'react' import { X } from 'lucide-react' import { useAppStore } from '@/store' -import { DropdownMenuLabel, DropdownMenuSeparator } from '@/components/ui/dropdown-menu' +import { DropdownMenuSeparator } from '@/components/ui/dropdown-menu' import SidebarRepositoryFilterSection from '@/components/sidebar/SidebarRepositoryFilterSection' import { SidebarHostScopeMenuSection } from '@/components/sidebar/SidebarHostScopeMenuSection' import { @@ -32,9 +32,6 @@ export function ActivityScopeFilterMenuSections(): React.JSX.Element | null { } return ( <> - - {translate('auto.components.sidebar.SidebarWorkspaceOptionsMenu.showSection', 'Show')} - {showHostScopeControls ? ( showFilterControls?: boolean showOptionsMenu?: boolean + showInlineActions?: boolean /** Rendered between the toolbar and the list; carries the active-scope chips row. */ scopeFilterRow?: React.ReactNode collapsedGroupKeys?: ReadonlySet @@ -314,6 +316,7 @@ export function ActivityThreadListPane({ resizable={resizable} showFilterControls={showFilterControls} showOptionsMenu={showOptionsMenu} + showInlineActions={showInlineActions} /> {scopeFilterRow}
    diff --git a/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx b/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx index f1ad0fb9057..5bca25ecec9 100644 --- a/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx +++ b/src/renderer/src/components/activity/activity-thread-list-toolbar.tsx @@ -1,5 +1,5 @@ import React from 'react' -import { BellDot, CheckCheck, Search, Trash2, X } from 'lucide-react' +import { CheckCheck, ListChecks, Search, Trash2, X } from 'lucide-react' import { Button } from '@/components/ui/button' import { Input } from '@/components/ui/input' import { @@ -34,7 +34,8 @@ export function ActivityThreadListToolbar({ onClearCompleted, resizable, showFilterControls, - showOptionsMenu + showOptionsMenu, + showInlineActions = true }: { activityFilterInputRef: React.RefObject query: string @@ -54,6 +55,7 @@ export function ActivityThreadListToolbar({ resizable: boolean showFilterControls: boolean showOptionsMenu: boolean + showInlineActions?: boolean }): React.JSX.Element | null { const showToolbar = showFilterControls || showOptionsMenu if (!showToolbar) { @@ -63,7 +65,7 @@ export function ActivityThreadListToolbar({ return ( <>
    -
    +
    {showFilterControls ? (
    @@ -147,7 +149,7 @@ export function ActivityThreadListToolbar({ className={cn( 'size-7 shrink-0 p-0 rounded-md transition-all', readFilter === 'unread' - ? '!border border-primary/50 !bg-primary/20 !text-primary shadow-xs hover:!bg-primary/30' + ? '!border border-primary/30 !bg-primary/10 !text-primary/90 shadow-xs hover:!bg-primary/15 hover:!text-primary' : 'text-muted-foreground hover:text-foreground hover:bg-muted/50' )} aria-label={translate( @@ -155,7 +157,7 @@ export function ActivityThreadListToolbar({ 'Show unread threads only' )} > - + @@ -182,7 +184,7 @@ export function ActivityThreadListToolbar({ ) : null}
    - {onMarkAllThreadsRead || onClearCompleted ? ( + {showInlineActions && (onMarkAllThreadsRead || onClearCompleted) ? (
    {onMarkAllThreadsRead ? ( - {translate('auto.components.activity.ActivityPrototypePage.a472a14700', 'More options')} + {translate( + 'auto.components.activity.ActivityPrototypePage.activityOptions', + 'Activity options' + )} onToggleUnread()} onSelect={(event) => event.preventDefault()} > - + {translate( 'auto.components.activity.ActivityPrototypePage.showUnreadOnly', diff --git a/src/renderer/src/components/settings/ExperimentalPane.test.tsx b/src/renderer/src/components/settings/ExperimentalPane.test.tsx index 17abf3154a0..b421b77bfc2 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.test.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.test.tsx @@ -7,7 +7,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import type { GlobalSettings } from '../../../../shared/global-settings-types' import { getDefaultSettings } from '../../../../shared/constants' import { ExperimentalPane } from './ExperimentalPane' -import { getExperimentalPaneSearchEntries, getExperimentalSearchEntry } from './experimental-search' +import { getExperimentalPaneSearchEntries } from './experimental-search' vi.mock('../../store', () => ({ useAppStore: (selector: (state: { settingsSearchQuery: string }) => unknown) => @@ -132,23 +132,6 @@ describe('ExperimentalPane', () => { ) }) - it('renders the Agents sidebar switch in Experimental without stale Appearance copy', () => { - const settings = getDefaultSettings('/tmp') - const markup = renderToStaticMarkup( - - ) - - expect(settings.experimentalAgentDashboardPopout).toBeUndefined() - expect(markup).toContain('Show Agents Button') - // The visible copy is the search entry's own description, so settings search can't - // advertise text the page doesn't show. - expect(markup).toContain(getExperimentalSearchEntry().agentsSidebar.description) - expect(markup).not.toContain('Window & Sidebar') - expect(getExperimentalPaneSearchEntries().map((entry) => entry.title)).toContain( - 'Show Agents Button' - ) - }) - it('renders the agent dashboard as an off-by-default searchable experiment', () => { const settings = getDefaultSettings('/tmp') const markup = renderToStaticMarkup( diff --git a/src/renderer/src/components/settings/ExperimentalPane.tsx b/src/renderer/src/components/settings/ExperimentalPane.tsx index 4c3536c78ed..2ef34d943c8 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.tsx @@ -39,9 +39,6 @@ export function ExperimentalPane({ const showNativeChat = matchesSettingsSearch(searchQuery, [ getExperimentalSearchEntry().nativeChat ]) - const showAgentsSidebar = matchesSettingsSearch(searchQuery, [ - getExperimentalSearchEntry().agentsSidebar - ]) const showAgentDashboard = matchesSettingsSearch(searchQuery, [ getExperimentalSearchEntry().agentDashboard ]) @@ -64,33 +61,6 @@ export function ExperimentalPane({ return (
    - {showAgentsSidebar ? ( - -
    -
    - - {/* Same string the search entry advertises, so search results match the page. */} -

    - {getExperimentalSearchEntry().agentsSidebar.description} -

    -
    - - updateSettings({ showAgentsSidebar: settings.showAgentsSidebar === false }) - } - /> -
    -
    - ) : null} - {showAgentDashboard ? ( ) : null} diff --git a/src/renderer/src/components/settings/appearance-sidebar-search.ts b/src/renderer/src/components/settings/appearance-sidebar-search.ts index 2b01f7704a7..ab77d2a0ad0 100644 --- a/src/renderer/src/components/settings/appearance-sidebar-search.ts +++ b/src/renderer/src/components/settings/appearance-sidebar-search.ts @@ -123,27 +123,6 @@ export const getShowPinnedWorktreesInGroupsEntry = createLocalizedCatalog( }) ) -export const getAgentsSidebarEntry = createLocalizedCatalog((): SettingsSearchEntry => ({ - title: translate('settings.appearance.agentsSidebar.title', 'Show Agents Button'), - description: translate( - 'settings.appearance.agentsSidebar.description', - 'Control whether the Agents tab appears in the left sidebar so you can monitor agent activity.' - ), - keywords: [ - ...translateSearchKeyword('auto.components.settings.general.search.baa263d6d8', 'agents'), - ...translateSearchKeyword('auto.components.settings.agents.search.96ba2373b6', 'agent'), - // Why: the sidebar row is labeled "Agent Dashboard"; searching that name must find this. - ...translateSearchKeyword( - 'auto.components.settings.experimental.search.agentDashboard.dashboard', - 'dashboard' - ), - ...translateSearchKeyword('auto.components.settings.appearance.search.5bff6a2ef0', 'sidebar'), - ...translateSearchKeyword('auto.components.settings.general.search.2a254b725e', 'tab'), - ...translateSearchKeyword('auto.components.settings.appearance.search.648eeada79', 'hide'), - ...translateSearchKeyword('auto.components.settings.appearance.search.ac79fe4a04', 'show') - ] -})) - export const getSidebarEntries = createLocalizedCatalog((): SettingsSearchEntry[] => [ { title: translate('auto.components.settings.appearance.search.155a1e7438', 'Show Tasks Button'), diff --git a/src/renderer/src/components/settings/experimental-search.ts b/src/renderer/src/components/settings/experimental-search.ts index 821de798301..594469fa200 100644 --- a/src/renderer/src/components/settings/experimental-search.ts +++ b/src/renderer/src/components/settings/experimental-search.ts @@ -5,7 +5,6 @@ import { translateSearchKeyword } from './settings-search-keywords' import { getNewWorktreeCardStyleSearchEntry } from './new-worktree-card-style-search-entry' import { getNativeChatExperimentalSearchEntry } from './native-chat-experimental-search-entry' import { getEphemeralVmsSearchEntry } from './ephemeral-vms-search' -import { getAgentsSidebarEntry } from './appearance-sidebar-search' export const getExperimentalPaneSearchEntries = createLocalizedCatalog( (): SettingsSearchEntry[] => [ @@ -48,7 +47,6 @@ export const getExperimentalPaneSearchEntries = createLocalizedCatalog( ] }, getNativeChatExperimentalSearchEntry(), - getAgentsSidebarEntry(), { title: translate( 'auto.components.settings.experimental.search.agentDashboard.title', @@ -203,7 +201,6 @@ export function getExperimentalSearchEntry() { nativeChat: findEntry( translate('auto.components.settings.experimental.search.nativeChat.title', 'Chat UI') ), - agentsSidebar: getAgentsSidebarEntry(), agentDashboard: findEntry( translate( 'auto.components.settings.experimental.search.agentDashboard.title', diff --git a/src/renderer/src/components/sidebar/Sidebar.test.tsx b/src/renderer/src/components/sidebar/Sidebar.test.tsx index 343c3f10c3c..0de26d72270 100644 --- a/src/renderer/src/components/sidebar/Sidebar.test.tsx +++ b/src/renderer/src/components/sidebar/Sidebar.test.tsx @@ -3,7 +3,7 @@ import type { CSSProperties, ReactNode } from 'react' import { renderToStaticMarkup } from 'react-dom/server' import { tmpdir } from 'node:os' -import { cleanup, fireEvent, render, waitFor } from '@testing-library/react' +import { cleanup, render, waitFor } from '@testing-library/react' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { getDefaultSettings } from '../../../../shared/constants' import type { GlobalSettings } from '../../../../shared/global-settings-types' @@ -38,20 +38,7 @@ vi.mock('@/components/ui/tooltip', () => ({ TooltipContent: ({ children }: { children: ReactNode }) => <>{children} })) -vi.mock('./SidebarHeader', () => ({ - default: ({ - agentToolbar, - agentSearchRow - }: { - agentToolbar?: ReactNode - agentSearchRow?: ReactNode - }) => ( -
    - {agentToolbar} - {agentSearchRow} -
    - ) -})) +vi.mock('./SidebarHeader', () => ({ default: () =>
    })) vi.mock('./SidebarAgentsList', () => ({ default: ({ query }: { query: string }) => ( @@ -240,35 +227,9 @@ describe('Sidebar', () => { expect(fetchAllWorktrees).not.toHaveBeenCalled() }) - it('clears the agents search query when the search row closes', async () => { - setSidebarState(getDefaultSettings(tmpdir())) - mocks.state = { ...mocks.state, sidebarBody: 'agents' } - const view = render(sidebarElement()) - const agentsList = await view.findByTestId('sidebar-agents-list') - - const searchToggle = view.getAllByRole('button', { name: 'Search' })[0] - fireEvent.click(searchToggle) - const searchInput = view.getByPlaceholderText('Filter...') - fireEvent.change(searchInput, { target: { value: 'deploy' } }) - expect(agentsList.getAttribute('data-query')).toBe('deploy') - - // Escape hides the row; a lingering query would silently keep filtering the list. - fireEvent.keyDown(searchInput, { key: 'Escape' }) - expect(view.queryByPlaceholderText('Filter...')).toBeNull() - expect(agentsList.getAttribute('data-query')).toBe('') - - fireEvent.click(searchToggle) - fireEvent.change(view.getByPlaceholderText('Filter...'), { target: { value: 'again' } }) - expect(agentsList.getAttribute('data-query')).toBe('again') - fireEvent.click(searchToggle) - expect(view.queryByPlaceholderText('Filter...')).toBeNull() - expect(agentsList.getAttribute('data-query')).toBe('') - }) - it('closes the dashboard drawer when the dashboard experiment is disabled', async () => { setSidebarState({ ...getDefaultSettings(tmpdir()), - showAgentsSidebar: true, experimentalAgentDashboardPopout: false }) const setAgentDashboardDrawerOpen = vi.fn() diff --git a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx index 75c0c441bad..5d82b56253a 100644 --- a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx +++ b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx @@ -1,12 +1,15 @@ import React, { useCallback, useEffect, useRef, useState } from 'react' import { createPortal } from 'react-dom' +import { useTranslation } from 'react-i18next' +import { Input } from '@/components/ui/input' import { useAppStore } from '@/store' +import { translate } from '@/i18n/i18n' import { ActivityScopeFilterChips } from '@/components/activity/activity-scope-filter-controls' import { hasActivityThreadWorkspace } from '@/components/activity/activity-thread-actions' import { useActivityThreadActionBindings } from '@/components/activity/use-activity-thread-action-bindings' import { ActivityThreadListPane } from '@/components/activity/activity-thread-list-pane' -import { ActivityThreadOptionsMenu } from '@/components/activity/activity-thread-controls' import { useAgentPaneThreads } from '@/components/activity/use-agent-pane-threads' +import { ActivityThreadOptionsMenu } from '@/components/activity/activity-thread-controls' import type { ActivityGroupBy, ThreadReadFilter } from '@/components/activity/activity-thread-types' /** @@ -36,14 +39,27 @@ export default function SidebarAgentsList({ optionsTarget, scrollTopRef }: SidebarAgentsListProps): React.JSX.Element { + // The search row is owned here and mounts conditionally, so subscribe this host to locale changes. + useTranslation() // Why store-backed: these are persisted preferences (agents* UI fields), unlike the momentary read filter/search. const compactMode = useAppStore((s) => s.agentsCompactMode) const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) const setShowChildAgents = useAppStore((s) => s.setAgentsShowChildAgents) const [selectedPaneKey, setSelectedPaneKey] = useState(null) + const [searchOpen, setSearchOpen] = useState(false) const activityFilterInputRef = useRef(null) + useEffect(() => { + if (!searchOpen) { + return + } + // Radix restores focus to the menu trigger after selection; focus on the + // next frame so the newly mounted search field wins that race. + const frame = requestAnimationFrame(() => activityFilterInputRef.current?.focus()) + return () => cancelAnimationFrame(frame) + }, [searchOpen]) + const { storeData, selectedPaneKeyIsLive, @@ -93,7 +109,32 @@ export default function SidebarAgentsList({ ) return ( - <> +
    + {searchOpen ? ( +
    + setQuery(event.target.value)} + onKeyDown={(event) => { + if (event.key === 'Escape') { + setSearchOpen(false) + setQuery('') + } + }} + placeholder={translate( + 'auto.components.activity.ActivityPrototypePage.795cbf26e2', + 'Filter...' + )} + className="h-7 w-full text-[11px]" + aria-label={translate( + 'auto.components.activity.ActivityPrototypePage.search', + 'Search' + )} + /> +
    + ) : null} } scrollTopRef={scrollTopRef} /> @@ -135,10 +180,13 @@ export default function SidebarAgentsList({ onShowChildAgentsChange={setShowChildAgents} onMarkAllThreadsRead={markAllThreadsRead} onClearCompleted={handleClearCompleted} + onSearch={() => setSearchOpen(true)} + unreadOnly={readFilter === 'unread'} + onToggleUnread={() => setReadFilter(readFilter === 'unread' ? 'all' : 'unread')} />, optionsTarget ) : null} - +
    ) } diff --git a/src/renderer/src/components/sidebar/SidebarHeader.test.tsx b/src/renderer/src/components/sidebar/SidebarHeader.test.tsx index 5df9ee619c5..8fff14e1846 100644 --- a/src/renderer/src/components/sidebar/SidebarHeader.test.tsx +++ b/src/renderer/src/components/sidebar/SidebarHeader.test.tsx @@ -23,7 +23,6 @@ type MockState = { updateSettings: (patch: Record) => void activeContextualTourId: string | null settings?: { - showAgentsSidebar?: boolean experimentalAgentDashboardPopout?: boolean agentsSidebarIntroShown?: boolean agentsSidebarMigratedFromExperimental?: boolean @@ -42,7 +41,9 @@ vi.mock('@/components/dashboard/useAgentBucketCounts', () => ({ useAgentBucketCounts: () => ({ attention: 0, working: 0, done: 0, idle: 0 }) })) -vi.mock('./SidebarWorkspaceOptionsMenu', () => ({ default: () => null })) +vi.mock('./SidebarWorkspaceOptionsMenu', () => ({ + default: () => + + + + + {activityLabel} + + event.preventDefault()} - onFocusOutside={(event) => event.preventDefault()} aria-labelledby={introTitleId} aria-describedby={introDescriptionId} > - - - - - - +
    - -
    -
    - -

    - {migratedFromExperimental - ? translate('agentsSidebarIntro.migrated.title', 'Agents are easier to find') - : translate('agentsSidebarIntro.new.title', 'Meet your Agents tab')} -

    -
    -

    - {migratedFromExperimental - ? translate( - 'agentsSidebarIntro.migrated.description', - 'Your Agents view is now a dedicated sidebar tab. Your activity and filters are preserved.' - ) - : translate( - 'agentsSidebarIntro.new.description', - 'See what your agents are working on, what is done, and where you need to step in.' - )} -

    +
    +
    -
    - {!migratedFromExperimental ? ( - - ) : null} -
    {agentsViewActive ? ( -
    - {/* Do not add an expand action: the full Agents view is deprecated and must not open. */} - {agentToolbar} -
    - ) : null} - {!agentsViewActive ? ( - +
    ) : null} +
    - {agentsViewActive ? agentSearchRow : null} - +
    ) }) diff --git a/src/renderer/src/components/sidebar/SidebarNav.test.tsx b/src/renderer/src/components/sidebar/SidebarNav.test.tsx index e3612db3858..5d72c8c1706 100644 --- a/src/renderer/src/components/sidebar/SidebarNav.test.tsx +++ b/src/renderer/src/components/sidebar/SidebarNav.test.tsx @@ -229,7 +229,6 @@ describe('SidebarNav', () => { setSidebarState({ settings: { ...getDefaultSettings('/tmp'), - showAgentsSidebar: false, experimentalAgentDashboardPopout: true } }) diff --git a/src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts b/src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts deleted file mode 100644 index 21639737994..00000000000 --- a/src/renderer/src/components/sidebar/agents-sidebar-visibility.test.ts +++ /dev/null @@ -1,15 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { shouldShowAgentsSidebar } from './agents-sidebar-visibility' - -describe('shouldShowAgentsSidebar', () => { - it('hides while settings are not yet hydrated', () => { - expect(shouldShowAgentsSidebar(null)).toBe(false) - expect(shouldShowAgentsSidebar(undefined)).toBe(false) - }) - - it('defaults on and honors only the dedicated setting', () => { - expect(shouldShowAgentsSidebar({})).toBe(true) - expect(shouldShowAgentsSidebar({ showAgentsSidebar: true })).toBe(true) - expect(shouldShowAgentsSidebar({ showAgentsSidebar: false })).toBe(false) - }) -}) diff --git a/src/renderer/src/components/sidebar/agents-sidebar-visibility.ts b/src/renderer/src/components/sidebar/agents-sidebar-visibility.ts deleted file mode 100644 index f0e978a4d2c..00000000000 --- a/src/renderer/src/components/sidebar/agents-sidebar-visibility.ts +++ /dev/null @@ -1,11 +0,0 @@ -import { - resolveAgentsSidebarVisible, - type AgentsSidebarVisibilitySettings -} from '../../../../shared/agents-sidebar-visibility' - -export function shouldShowAgentsSidebar( - settings: Partial | null | undefined -): boolean { - // Settings hydrate after first render; avoid flashing UI for opted-out profiles. - return settings ? resolveAgentsSidebarVisible(settings) : false -} diff --git a/src/renderer/src/components/sidebar/index.tsx b/src/renderer/src/components/sidebar/index.tsx index 20d2fef711f..90c31999bd3 100644 --- a/src/renderer/src/components/sidebar/index.tsx +++ b/src/renderer/src/components/sidebar/index.tsx @@ -1,20 +1,16 @@ import React, { useEffect, useMemo } from 'react' -import { useTranslation } from 'react-i18next' import { useAppStore } from '@/store' -import { Tooltip, TooltipContent, TooltipProvider, TooltipTrigger } from '@/components/ui/tooltip' +import { TooltipProvider } from '@/components/ui/tooltip' import { useSidebarResize } from '@/hooks/useSidebarResize' import SidebarHeader from './SidebarHeader' import SidebarNav from './SidebarNav' -import { shouldShowAgentsSidebar } from './agents-sidebar-visibility' import SetupScriptPromptCard from './SetupScriptPromptCard' import WorktreeList from './WorktreeList' import SidebarToolbar from './SidebarToolbar' import WorkspaceKanbanDrawer from './WorkspaceKanbanDrawer' import type { VirtualizedScrollAnchor } from '@/hooks/useVirtualizedScrollAnchor' import { cn } from '@/lib/utils' -import { BellDot, FolderPlus, Loader2, Search } from 'lucide-react' -import { Button } from '@/components/ui/button' -import { Input } from '@/components/ui/input' +import { FolderPlus, Loader2 } from 'lucide-react' import type { ActivityGroupBy, ThreadReadFilter } from '@/components/activity/activity-thread-types' import { ActivityThreadCollapseContext } from '@/components/activity/activity-thread-collapse-context' import { useSidebarProjectDrop } from './useSidebarProjectDrop' @@ -23,7 +19,6 @@ import { useWorkspaceRevealBodyRedirect } from './use-workspace-reveal-body-redi import { resolveLeftSidebarStyleVariables } from '@/lib/left-sidebar-appearance' import { useSystemPrefersDark } from '@/components/terminal-pane/use-system-prefers-dark' import { lazyWithRetry } from '@/lib/lazy-with-retry' -import { translate } from '@/i18n/i18n' // Why lazy: the Agents list pulls the whole activity pipeline (virtualizer, markdown // previews, thread derivation); users on the workspace view should not load or render any of it. @@ -54,14 +49,6 @@ function Sidebar({ worktreeScrollOffsetRef, worktreeScrollAnchorRef }: SidebarProps): React.JSX.Element { - // Why: the memoized toolbar/search JSX below is localized, so it needs both a - // language subscription here and the locale as a memo dep to refresh on a switch. - const { i18n } = useTranslation() - const locale = i18n.resolvedLanguage ?? i18n.language - const sidebarTranslate = React.useCallback( - (key: string, fallback: string): string => translate(key, fallback, { lng: locale }), - [locale] - ) const sidebarOpen = useAppStore((s) => s.sidebarOpen) const sidebarWidth = useAppStore((s) => s.sidebarWidth) const setSidebarWidth = useAppStore((s) => s.setSidebarWidth) @@ -69,19 +56,12 @@ function Sidebar({ const startupWorktreeRefreshCompleted = useAppStore((s) => s.startupWorktreeRefreshCompleted) const settings = useAppStore((s) => s.settings) const sidebarBody = useAppStore((s) => s.sidebarBody ?? 'workspaces') - const showAgentsSidebar = shouldShowAgentsSidebar(settings) const showAgentDashboard = settings?.experimentalAgentDashboardPopout === true const agentDashboardDrawerOpen = useAppStore((s) => s.agentDashboardDrawerOpen) const setAgentDashboardDrawerOpen = useAppStore((s) => s.setAgentDashboardDrawerOpen) const [agentReadFilter, setAgentReadFilter] = React.useState('all') const [agentGroupBy, setAgentGroupBy] = React.useState('status') const [agentQuery, setAgentQuery] = React.useState('') - const [agentSearchOpen, setAgentSearchOpen] = React.useState(false) - // Why clear on close: the hidden input's query would keep filtering the list with no visible indicator. - const closeAgentSearch = React.useCallback(() => { - setAgentSearchOpen(false) - setAgentQuery('') - }, []) const [agentOptionsTarget, setAgentOptionsTarget] = React.useState(null) const agentsScrollTopRef = React.useRef(0) // Held here so collapsed groups (and the layout the saved scrollTop assumes) @@ -165,106 +145,7 @@ function Sidebar({ onDraftWidthChange: setLiveSidebarWidth }) - // Why memoized: SidebarHeader is React.memo; fresh JSX here on every Sidebar render would - // defeat that memo and re-render the header subtree on unrelated store churn. - const agentToolbar = useMemo( - () => ( -
    - - - - - - {sidebarTranslate('auto.components.activity.ActivityPrototypePage.search', 'Search')} - - - - - - - - {sidebarTranslate( - 'auto.components.activity.ActivityPrototypePage.d1a88df9a8', - 'Show unread threads only' - )} - - -
    -
    - ), - [agentReadFilter, agentSearchOpen, closeAgentSearch, sidebarTranslate] - ) - const agentSearchRow = useMemo( - () => - agentSearchOpen ? ( -
    - setAgentQuery(event.target.value)} - onKeyDown={(event) => { - if (event.key === 'Escape') { - closeAgentSearch() - } - }} - placeholder={sidebarTranslate( - 'auto.components.activity.ActivityPrototypePage.795cbf26e2', - 'Filter...' - )} - className="h-7 w-full text-[11px]" - aria-label={sidebarTranslate( - 'auto.components.activity.ActivityPrototypePage.search', - 'Search' - )} - /> -
    - ) : null, - [agentQuery, agentSearchOpen, closeAgentSearch, sidebarTranslate] - ) - - useWorkspaceRevealBodyRedirect(sidebarOpen && sidebarBody === 'agents' && showAgentsSidebar) + useWorkspaceRevealBodyRedirect(sidebarOpen && sidebarBody === 'agents') return ( @@ -281,11 +162,9 @@ function Sidebar({ - {sidebarBody === 'agents' && showAgentsSidebar ? ( + {sidebarBody === 'agents' ? ( }> s.openModal) - return ( - - - - - - {translate('auto.components.sidebar.SidebarHeader.25a95899c9', 'Add Project')} - - - ) -} - function CompactWorkspaceOverflow({ preserveWorkspaceBoardOpen, onMenuOpenChange @@ -109,9 +86,11 @@ function CompactWorkspaceOverflow({ } export function SidebarHeaderActions({ - onWorkspaceBoardMenuOpenChange + onWorkspaceBoardMenuOpenChange, + hideWorkspaceOptions = false }: { onWorkspaceBoardMenuOpenChange: (open: boolean) => void + hideWorkspaceOptions?: boolean }): React.JSX.Element { const sidebarWidth = useAppStore((s) => s.sidebarWidth) const newWorktreeShortcutLabel = useShortcutLabel('workspace.create') @@ -145,22 +124,24 @@ export function SidebarHeaderActions({ )} - + {hideWorkspaceOptions ? null : ( + + )}
    ) } return (
    - - - + {hideWorkspaceOptions ? null : ( + + )} diff --git a/src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx b/src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx deleted file mode 100644 index 38893fb8f2d..00000000000 --- a/src/renderer/src/components/sidebar/sidebar-view-toggle.test.tsx +++ /dev/null @@ -1,145 +0,0 @@ -// @vitest-environment happy-dom - -import { act } from 'react' -import { createRoot, type Root } from 'react-dom/client' -import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { SidebarViewToggle } from './sidebar-view-toggle' - -;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true - -let container: HTMLDivElement -let root: Root - -beforeEach(() => { - container = document.createElement('div') - document.body.append(container) - root = createRoot(container) -}) - -afterEach(() => { - act(() => root.unmount()) - container.remove() -}) - -describe('SidebarViewToggle', () => { - it('exposes radio semantics with a roving tabindex so arrow keys move between tabs', () => { - act(() => { - root.render( - undefined} - options={[ - { value: 'workspaces', label: 'Spaces', sectionTitle: 'projects' }, - { value: 'agents', label: 'Agents', sectionTitle: 'agents' } - ]} - /> - ) - }) - - const items = [...container.querySelectorAll('[role="radio"]')] - expect(items).toHaveLength(2) - const agents = container.querySelector('[data-sidebar-section-title="agents"]') - const projects = container.querySelector('[data-sidebar-section-title="projects"]') - expect(agents?.getAttribute('aria-checked')).toBe('true') - expect(projects?.getAttribute('aria-checked')).toBe('false') - // Only one tab is in the tab order (roving tabindex); arrow keys reach the other. - const tabStops = items.map((item) => item.getAttribute('tabindex')) - expect(tabStops.filter((stop) => stop === '0')).toHaveLength(1) - expect(tabStops.filter((stop) => stop === '-1')).toHaveLength(1) - }) - - it('never deselects when the active tab is clicked again', () => { - const onSelect = vi.fn() - act(() => { - root.render( - - ) - }) - const agents = container.querySelector( - '[data-sidebar-section-title="agents"]' - ) - act(() => agents?.click()) - expect(onSelect).not.toHaveBeenCalled() - const projects = container.querySelector( - '[data-sidebar-section-title="projects"]' - ) - act(() => projects?.click()) - expect(onSelect).toHaveBeenCalledWith('workspaces') - }) - - it('moves focus to the radio an arrow key selects', () => { - // Without the focus move, every later arrow press steps from the old index and - // keeps re-selecting the same neighbour. - const onSelect = vi.fn() - act(() => { - root.render( - - ) - }) - - const projects = container.querySelector( - '[data-sidebar-section-title="projects"]' - ) - act(() => { - projects?.focus() - projects?.dispatchEvent( - new KeyboardEvent('keydown', { key: 'ArrowRight', bubbles: true, cancelable: true }) - ) - }) - - expect(onSelect).toHaveBeenCalledWith('agents') - expect(document.activeElement).toBe( - container.querySelector('[data-sidebar-section-title="agents"]') - ) - }) - - it('keeps the visible label on one line', () => { - act(() => { - root.render( - undefined} - options={[ - { - value: 'workspaces', - label: 'Spaces', - sectionTitle: 'projects' - }, - { value: 'agents', label: 'Agents', sectionTitle: 'agents' } - ]} - /> - ) - }) - - const group = container.querySelector('[role="radiogroup"]') - const groupClasses = new Set(group?.className.split(/\s+/) ?? []) - expect(groupClasses.has('inline-flex')).toBe(true) - expect(groupClasses.has('shrink-0')).toBe(true) - expect(groupClasses.has('flex-1')).toBe(false) - - const spacesTab = container.querySelector('[data-sidebar-section-title="projects"]') - const visibleLabel = [...(spacesTab?.querySelectorAll('span') ?? [])].find( - (span) => span.getAttribute('aria-hidden') == null && span.textContent === 'Spaces' - ) - expect(visibleLabel?.className).toContain('whitespace-nowrap') - expect(visibleLabel?.className.includes('truncate')).toBe(false) - }) -}) diff --git a/src/renderer/src/components/sidebar/sidebar-view-toggle.tsx b/src/renderer/src/components/sidebar/sidebar-view-toggle.tsx deleted file mode 100644 index 888f4f851c7..00000000000 --- a/src/renderer/src/components/sidebar/sidebar-view-toggle.tsx +++ /dev/null @@ -1,106 +0,0 @@ -import React from 'react' -import { cn } from '@/lib/utils' - -type SidebarViewToggleOption = { - value: string - label: string - /** Every label this slot can ever show; reserves width so switching never resizes the tab. */ - widthLabels?: readonly string[] - sectionTitle?: string - renderWrapper?: (button: React.ReactNode) => React.ReactNode -} - -type SidebarViewToggleProps = { - ariaLabel: string - value: string - options: readonly SidebarViewToggleOption[] - onSelect: (value: string) => void - className?: string -} - -/** Two-up segmented control; tab widths stay frozen so nothing reflows on toggle. */ -export function SidebarViewToggle({ - ariaLabel, - value, - options, - onSelect, - className -}: SidebarViewToggleProps): React.JSX.Element { - const buttonRefs = React.useRef<(HTMLButtonElement | null)[]>([]) - // Arrow keys must carry focus to the newly checked radio, or every later press - // would still step from the old index and re-select the same neighbour. - const selectAndFocus = (index: number): void => { - const option = options[index] - if (!option) { - return - } - if (option.value !== value) { - onSelect(option.value) - } - buttonRefs.current[index]?.focus() - } - return ( -
    - {options.map((option, index) => { - const active = option.value === value - const button = ( - - ) - - return option.renderWrapper ? ( - {option.renderWrapper(button)} - ) : ( - button - ) - })} -
    - ) -} diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 93ba2bbd96f..b893f1e8bdf 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -137,10 +137,6 @@ "menuBarIcon": { "title": "Show Menu Bar Icon", "description": "Keep an Orca shortcut and activity indicator in the macOS menu bar." - }, - "agentsSidebar": { - "title": "Show Agents Button", - "description": "Control whether the Agents tab appears in the left sidebar so you can monitor agent activity." } }, "browser": { diff --git a/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts b/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts index a9a6f7aed0a..d08e291d982 100644 --- a/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts +++ b/src/renderer/src/store/slices/ui-hydration-view-layout.test.ts @@ -164,17 +164,6 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().activeView).toBe('terminal') }) - it('drops a persisted activity view when the Agents sidebar is hidden', () => { - const store = createUIStore() - store.setState({ - settings: { showAgentsSidebar: false } as AppState['settings'] - }) - - store.getState().hydratePersistedUI(makePersistedUI({ activeView: 'activity' }), 'startup') - - expect(store.getState().activeView).toBe('terminal') - }) - it('keeps a persisted activity view when the settings fetch failed', () => { // A failed window.api.settings.get() leaves settings null; downgrading here would let the // persisted-UI writer overwrite the saved view with terminal. @@ -186,17 +175,6 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().activeView).toBe('activity') }) - it('restores a persisted activity view when the Agents sidebar is shown', () => { - const store = createUIStore() - store.setState({ - settings: { showAgentsSidebar: true } as AppState['settings'] - }) - - store.getState().hydratePersistedUI(makePersistedUI({ activeView: 'activity' }), 'startup') - - expect(store.getState().activeView).toBe('activity') - }) - it('restores a default-on view (mobile) even when its nav button is hidden', () => { const store = createUIStore() store.setState({ diff --git a/src/renderer/src/store/slices/ui-page-navigation.test.ts b/src/renderer/src/store/slices/ui-page-navigation.test.ts index 9a355223c27..c61a0e69a48 100644 --- a/src/renderer/src/store/slices/ui-page-navigation.test.ts +++ b/src/renderer/src/store/slices/ui-page-navigation.test.ts @@ -301,9 +301,6 @@ describe('createUISlice settings navigation', () => { it('returns to the graduated Agents view after visiting settings', () => { const store = createUIStore() - store.setState({ - settings: { showAgentsSidebar: true } as unknown as AppState['settings'] - } as unknown as Partial) store.getState().openActivityPage() expect(store.getState().activeView).toBe('activity') @@ -313,20 +310,6 @@ describe('createUISlice settings navigation', () => { expect(store.getState().activeView).toBe('activity') }) - it('falls back to terminal when closing settings with the Agents surfaces hidden', () => { - const store = createUIStore() - - store.setState({ - settings: { showAgentsSidebar: false } as unknown as AppState['settings'], - activeView: 'settings', - previousViewBeforeSettings: 'activity' - } as unknown as Partial) - - store.getState().closeSettingsPage() - - expect(store.getState().activeView).toBe('terminal') - }) - it('clears transient settings search when opening settings', () => { const store = createUIStore() diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts index 64858db11a0..f44774b3770 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts @@ -275,9 +275,7 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par : s.workspaceCleanupBrowse, // Why: restore only on startup; on 'sync' broadcasts it would clobber the window's current per-window view. activeView: - source === 'startup' - ? sanitizeHydratedActiveView(ui.activeView, s.settings) - : s.activeView, + source === 'startup' ? sanitizeHydratedActiveView(ui.activeView) : s.activeView, persistedUIReady: true } // The incoming payload is authoritative for the writer-owned fields, so it becomes the diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts index edb79511cd8..bfed25a75cc 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-sanitizers.ts @@ -15,8 +15,6 @@ import { normalizeExecutionHostScope } from '../../../../../shared/execution-host' import { persistedUIValuesEqual } from '../../../../../shared/persisted-ui-equality' -// Pure predicate over GlobalSettings; safe to share with the store layer. -import { shouldShowAgentsSidebar } from '@/components/sidebar/agents-sidebar-visibility' import { DEFAULT_STATUS_BAR_ITEMS } from '../../../../../shared/constants' import type { UISlice } from './ui-slice-contract' @@ -173,20 +171,11 @@ export function sanitizeWorkspaceCleanupDismissals( return out } -export function sanitizeHydratedActiveView( - value: PersistedUIState['activeView'], - settings: Parameters[0] -): TopLevelView { +export function sanitizeHydratedActiveView(value: PersistedUIState['activeView']): TopLevelView { // Why: older data (pre-activeView) or a view a different build doesn't have falls back to terminal rather than rendering nothing. if (!isTopLevelView(value)) { return 'terminal' } - // Why: activity is hidden when its entry points are, so gate only it (mobile/automations stay functional when hidden). - // Why the null check: a failed settings fetch is not an opt-out. Downgrading on absent settings - // would let the persisted-UI writer overwrite the user's saved `activity` with `terminal`. - if (value === 'activity' && settings && !shouldShowAgentsSidebar(settings)) { - return 'terminal' - } return value } diff --git a/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts index c693c900c4a..229a3860fbc 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-settings-actions.ts @@ -1,7 +1,5 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' import { isSettingsNavigationTarget } from '../../../lib/settings-navigation-types' -// Pure predicate over GlobalSettings; safe to share with the store layer. -import { shouldShowAgentsSidebar } from '@/components/sidebar/agents-sidebar-visibility' export function createUiSettingsActions(set: UISliceSet, get: UISliceGet): Partial { return { @@ -17,13 +15,7 @@ export function createUiSettingsActions(set: UISliceSet, get: UISliceGet): Parti }, closeSettingsPage: () => set((state) => { - // Agents graduated from experimentalActivity; match openActivityPage's gate. - const previousView = - state.previousViewBeforeSettings === 'activity' && - !shouldShowAgentsSidebar(state.settings) - ? 'terminal' - : state.previousViewBeforeSettings - return { activeView: previousView } + return { activeView: state.previousViewBeforeSettings } }), settingsNavigationTarget: null, openSettingsTarget: (target) => { diff --git a/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts index d61efbb598d..b97f77c89a3 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-view-actions.ts @@ -1,16 +1,9 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' import { rewindHistoryIndexPastView } from '../worktree-nav-history' -// Pure predicate over GlobalSettings; safe to share with the store layer. -import { shouldShowAgentsSidebar } from '@/components/sidebar/agents-sidebar-visibility' export function createUiViewActions(set: UISliceSet, get: UISliceGet): Partial { return { openActivityPage: () => { - // Agents graduated from experimentalActivity; gate on the same visibility - // rule as the sidebar entry points so the view is reachable iff shown. - if (!shouldShowAgentsSidebar(get().settings)) { - return - } set((state) => ({ activeView: 'activity', previousViewBeforeActivity: diff --git a/src/shared/agents-sidebar-visibility.ts b/src/shared/agents-sidebar-visibility.ts deleted file mode 100644 index c05f8e106ed..00000000000 --- a/src/shared/agents-sidebar-visibility.ts +++ /dev/null @@ -1,12 +0,0 @@ -export type AgentsSidebarVisibilitySettings = { - showAgentsSidebar?: boolean -} - -export function resolveAgentsSidebarVisible( - settings: Partial | null | undefined -): boolean { - if (!settings) { - return true - } - return settings.showAgentsSidebar !== false -} diff --git a/src/shared/default-global-settings.ts b/src/shared/default-global-settings.ts index 260fb5d6846..313e9fc6a9a 100644 --- a/src/shared/default-global-settings.ts +++ b/src/shared/default-global-settings.ts @@ -225,7 +225,6 @@ export function buildDefaultSettings(args: { // Why: off keeps the cosmetic overlay unmounted for users who never opt in. experimentalPet: false, experimentalActivity: false, - showAgentsSidebar: true, experimentalActivityDefaultedOffForAllUsers: true, experimentalTerminalAttention: false, experimentalAgentHibernation: false, diff --git a/src/shared/global-settings-types.ts b/src/shared/global-settings-types.ts index f650e4f4a77..bc590a19a2e 100644 --- a/src/shared/global-settings-types.ts +++ b/src/shared/global-settings-types.ts @@ -427,11 +427,9 @@ export type GlobalSettings = { experimentalActivity: boolean /** Experimental: pop-out Kanban dashboard for monitoring and opening agent terminals across worktrees. */ experimentalAgentDashboardPopout?: boolean - /** Experimental: whether the Agents tab is shown in the left sidebar. Defaults on. */ - showAgentsSidebar?: boolean - /** Set after the experimental Agents tab introduction has been acknowledged. */ + /** Set after the one-time legacy Agents tab introduction has been acknowledged. */ agentsSidebarIntroShown?: boolean - /** True when the profile previously opted into the legacy experimental Agents view. */ + /** True when the profile previously opted into the legacy Agents view. */ agentsSidebarMigratedFromExperimental?: boolean /** How the Agent Dashboard opens: an in-window companion board or a separate pop-out window. Defaults to in-window. */ experimentalAgentDashboardMode?: AgentDashboardMode diff --git a/src/shared/telemetry-property-schemas.ts b/src/shared/telemetry-property-schemas.ts index b1e8d2ed573..3346ea73320 100644 --- a/src/shared/telemetry-property-schemas.ts +++ b/src/shared/telemetry-property-schemas.ts @@ -192,7 +192,6 @@ export const SETTINGS_CHANGED_WHITELIST = [ 'experimentalNativeChat', 'experimentalStructuredNativeChat', 'experimentalActivity', - 'showAgentsSidebar', 'experimentalAgentDashboardPopout', 'experimentalTerminalAttention', 'experimentalAgentHibernation', diff --git a/tests/e2e/activity-agent-pane-isolation.spec.ts b/tests/e2e/activity-agent-pane-isolation.spec.ts index 817044bd127..12ac2f68dc6 100644 --- a/tests/e2e/activity-agent-pane-isolation.spec.ts +++ b/tests/e2e/activity-agent-pane-isolation.spec.ts @@ -127,7 +127,6 @@ async function enableActivityAgentsView(page: Page): Promise { // Why: the Agents tab is on by default, but a fresh profile opens the intro popover // over it; stamping it as shown keeps the toggle clickable without dismissing it first. const settings = await window.api.settings.set({ - showAgentsSidebar: true, agentsSidebarIntroShown: true }) window.__store?.setState({ settings }) From 623d58e386170cb6c15b7afa6178d7c850f2b351 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:00:17 -0700 Subject: [PATCH 100/398] fix(native-chat): show pasted images while they save, and make them previewable (#18118) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(native-chat): show pasted images while they save, and make them previewable Pasting an image into the native chat composer showed nothing until the clipboard image finished being written to disk, and the resulting chip could never render the image at all. Preview was blocked by path authorization, not by rendering. Clipboard pastes are written to the OS temp dir, which sits outside every allowed root, so the composer's own `fs:readFile` of the file Orca had just written was denied. `saveClipboardImageBufferAsTempFile` now authorizes the path it writes, the same way other Orca-produced external files are handled. The delay is the macOS paste route: Cmd+V is intercepted in main and delivered through the app-menu paste channel, which has no clipboard blob in hand, so the composer only learned an image existed after the save round-trip. A new `clipboard:readImageThumbnail` probe reads the clipboard in memory and returns a downscaled preview; it runs alongside the save rather than before it, so text paste gains no latency. The DOM-paste route needs no probe — it mints a blob URL from the clipboard file on the same tick. Attachments now carry `pending` and `previewUrl`: the chip appears immediately with the real image dimmed under a spinner, then settles in place on the saved path. Send is blocked while anything is pending, because a pending chip has no agent-readable path yet. Pending chips are kept out of the pane attachment cache so a mid-save unmount cannot strand one, and blob previews are revoked on remove/clear. SSH pastes now carry their connectionId onto the chip so remote previews read over SFTP. Verified in a real Codex native chat under an isolated dev instance: the chip appears in 42-61ms with a spinner, settles at ~141ms, three rapid pastes produce three independent chips with Send disabled throughout, and the lightbox opens the full 5120x2880 image read from disk. Ablation confirms the authorization fix: the written path reads back, an unauthorized sibling in the same temp dir does not. Claude-Session: https://claude.ai/code/session_01NnEfY8NpfFtVnboLKnmgdW * fix(native-chat): avoid stale image attachments and preview cache growth --------- Co-authored-by: Merge Sim --- .../window/clipboard-image-temp-file.test.ts | 50 ++++ src/main/window/clipboard-image-temp-file.ts | 4 + .../window/clipboard-image-thumbnail.test.ts | 45 ++++ src/main/window/clipboard-image-thumbnail.ts | 45 ++++ .../window/clipboard-ipc-handlers.test.ts | 30 +-- src/main/window/clipboard-ipc-handlers.ts | 11 +- ...ui-bridge-clipboard-and-window-controls.ts | 3 + src/preload/api/ui-window-api.ts | 2 + .../native-chat/NativeChatComposer.test.tsx | 62 ++++- .../native-chat/NativeChatComposer.tsx | 86 +++--- .../native-chat/NativeChatComposerField.tsx | 6 + .../NativeChatImageAttachmentPreview.test.tsx | 55 ++++ .../NativeChatImageAttachmentPreview.tsx | 63 +++-- .../native-chat-image-paste.test.ts | 29 +-- .../native-chat/native-chat-image-paste.ts | 16 -- ...-native-chat-composer-attachments.test.tsx | 85 ++++++ .../use-native-chat-composer-attachments.ts | 140 ++++++++-- .../use-native-chat-composer-paste.test.tsx | 245 +++++++++++++++--- .../use-native-chat-composer-paste.ts | 160 ++++++++++-- ...se-native-chat-structured-composer-send.ts | 73 ++++++ src/renderer/src/i18n/locales/en.json | 1 + .../src/web/preload-api/web-clipboard-api.ts | 48 +++- .../src/web/preload-api/web-ui-api.ts | 2 + src/shared/clipboard-image.ts | 26 ++ 24 files changed, 1071 insertions(+), 216 deletions(-) create mode 100644 src/main/window/clipboard-image-temp-file.test.ts create mode 100644 src/main/window/clipboard-image-thumbnail.test.ts create mode 100644 src/main/window/clipboard-image-thumbnail.ts create mode 100644 src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.test.tsx create mode 100644 src/renderer/src/components/native-chat/use-native-chat-structured-composer-send.ts diff --git a/src/main/window/clipboard-image-temp-file.test.ts b/src/main/window/clipboard-image-temp-file.test.ts new file mode 100644 index 00000000000..71ca0804c0e --- /dev/null +++ b/src/main/window/clipboard-image-temp-file.test.ts @@ -0,0 +1,50 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const { authorizeExternalPathMock, writeFileMock, getPathMock, writeFileBase64Mock } = vi.hoisted( + () => ({ + authorizeExternalPathMock: vi.fn(), + writeFileMock: vi.fn(), + getPathMock: vi.fn(() => '/var/folders/ab/T'), + writeFileBase64Mock: vi.fn() + }) +) + +vi.mock('node:fs/promises', () => ({ default: { writeFile: writeFileMock } })) +vi.mock('node:crypto', () => ({ randomUUID: () => 'uuid-1' })) +vi.mock('../../shared/app-environment', () => ({ + getAppEnvironment: () => ({ getPath: getPathMock }) +})) +vi.mock('../providers/ssh-filesystem-dispatch', () => ({ + requireSshFilesystemProvider: () => ({ + getTempDir: async () => '/remote/tmp', + writeFileBase64: writeFileBase64Mock + }) +})) +vi.mock('../ipc/filesystem-auth', () => ({ authorizeExternalPath: authorizeExternalPathMock })) + +import { saveClipboardImageBufferAsTempFile } from './clipboard-image-temp-file' + +beforeEach(() => { + vi.clearAllMocks() +}) + +describe('saveClipboardImageBufferAsTempFile', () => { + it('authorizes the local temp file so the composer can preview what it just wrote', async () => { + const savedPath = await saveClipboardImageBufferAsTempFile(Buffer.from([1, 2, 3])) + + expect(writeFileMock).toHaveBeenCalledWith(savedPath, Buffer.from([1, 2, 3])) + // The OS temp dir is outside every allowed root, so an unauthorized path + // makes fs:readFile deny the preview read of Orca's own file. + expect(authorizeExternalPathMock).toHaveBeenCalledWith(savedPath) + }) + + it('does not authorize a local path for an SSH save', async () => { + const savedPath = await saveClipboardImageBufferAsTempFile(Buffer.from([1]), { + connectionId: 'conn-1' + }) + + expect(savedPath.startsWith('/remote/tmp/')).toBe(true) + expect(writeFileBase64Mock).toHaveBeenCalled() + expect(authorizeExternalPathMock).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/window/clipboard-image-temp-file.ts b/src/main/window/clipboard-image-temp-file.ts index cadbd677650..0024c4b6d1f 100644 --- a/src/main/window/clipboard-image-temp-file.ts +++ b/src/main/window/clipboard-image-temp-file.ts @@ -6,6 +6,7 @@ import { getAppEnvironment } from '../../shared/app-environment' import { requireSshFilesystemProvider } from '../providers/ssh-filesystem-dispatch' import { isWindowsAbsolutePathLike } from '../../shared/cross-platform-path' import { assertClipboardImageByteLengthWithinLimit } from '../../shared/clipboard-image' +import { authorizeExternalPath } from '../ipc/filesystem-auth' export type SaveClipboardImageAsTempFileArgs = { connectionId?: string | null @@ -41,5 +42,8 @@ export async function saveClipboardImageBufferAsTempFile( const tempPath = path.join(getAppEnvironment().getPath('temp'), fileName) await fs.writeFile(tempPath, buffer) + // Why: the OS temp dir is outside every allowed root, so without this the + // composer's own thumbnail/preview read of the file it just wrote is denied. + authorizeExternalPath(tempPath) return tempPath } diff --git a/src/main/window/clipboard-image-thumbnail.test.ts b/src/main/window/clipboard-image-thumbnail.test.ts new file mode 100644 index 00000000000..e3cd0608972 --- /dev/null +++ b/src/main/window/clipboard-image-thumbnail.test.ts @@ -0,0 +1,45 @@ +import { describe, expect, it, vi } from 'vitest' +import { buildClipboardImageThumbnail } from './clipboard-image-thumbnail' + +function fakeImage(overrides: Partial[0]>) { + return { + isEmpty: () => false, + getSize: () => ({ height: 10, width: 10 }), + resize: vi.fn(() => ({ toDataURL: () => 'data:image/png;base64,SMALL' })), + toDataURL: () => 'data:image/png;base64,FULL', + ...overrides + } +} + +describe('buildClipboardImageThumbnail', () => { + it('downscales to the thumbnail budget but reports the source dimensions', () => { + const image = fakeImage({ getSize: () => ({ height: 1600, width: 3200 }) }) + + expect(buildClipboardImageThumbnail(image)).toEqual({ + dataUrl: 'data:image/png;base64,SMALL', + height: 1600, + width: 3200 + }) + expect(image.resize).toHaveBeenCalledWith({ height: 160, quality: 'good', width: 320 }) + }) + + it('skips the resize for an image that already fits', () => { + const image = fakeImage({ getSize: () => ({ height: 200, width: 320 }) }) + + expect(buildClipboardImageThumbnail(image)?.dataUrl).toBe('data:image/png;base64,FULL') + expect(image.resize).not.toHaveBeenCalled() + }) + + it('reports no thumbnail for an empty clipboard so text paste falls through', () => { + expect(buildClipboardImageThumbnail(fakeImage({ isEmpty: () => true }))).toBeNull() + }) + + it('reports no thumbnail rather than throwing for an oversized image', () => { + // The save call still surfaces the real too-large error; the probe only + // decides whether a placeholder chip is worth showing. + const image = fakeImage({ getSize: () => ({ height: 100_000, width: 100_000 }) }) + + expect(buildClipboardImageThumbnail(image)).toBeNull() + expect(image.resize).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/window/clipboard-image-thumbnail.ts b/src/main/window/clipboard-image-thumbnail.ts new file mode 100644 index 00000000000..b2108e25f13 --- /dev/null +++ b/src/main/window/clipboard-image-thumbnail.ts @@ -0,0 +1,45 @@ +import { + assertClipboardImageDimensionsWithinLimit, + clipboardImageThumbnailSize, + type ClipboardImageDimensions, + type ClipboardImageThumbnail +} from '../../shared/clipboard-image' + +/** The slice of Electron's NativeImage this module needs, so the decision logic + * is testable without an Electron runtime. */ +export type ClipboardImageLike = { + isEmpty: () => boolean + getSize: () => ClipboardImageDimensions + resize: (options: { height: number; width: number; quality: 'good' | 'better' | 'best' }) => { + toDataURL: () => string + } + toDataURL: () => string +} + +/** + * In-memory preview of whatever image the clipboard holds. Writing the image to + * disk (or uploading it over SFTP) takes long enough that a composer with no + * feedback reads as a dropped paste, so this answers "is there an image, and + * what does it look like" without touching the filesystem. + */ +export function buildClipboardImageThumbnail( + image: ClipboardImageLike +): ClipboardImageThumbnail | null { + if (image.isEmpty()) { + return null + } + const size = image.getSize() + try { + assertClipboardImageDimensionsWithinLimit(size) + } catch { + // Oversized images still report through the save call; the probe only + // decides whether to show a placeholder, so degrade to "no preview". + return null + } + const thumbnailSize = clipboardImageThumbnailSize(size) + const thumbnail = + thumbnailSize.width === size.width && thumbnailSize.height === size.height + ? image + : image.resize({ ...thumbnailSize, quality: 'good' }) + return { dataUrl: thumbnail.toDataURL(), height: size.height, width: size.width } +} diff --git a/src/main/window/clipboard-ipc-handlers.test.ts b/src/main/window/clipboard-ipc-handlers.test.ts index ce84c68e327..ea337747421 100644 --- a/src/main/window/clipboard-ipc-handlers.test.ts +++ b/src/main/window/clipboard-ipc-handlers.test.ts @@ -13,6 +13,7 @@ const { spawnMock, childStdinEndMock, resolveAuthorizedPathMock, + authorizeExternalPathMock, fsAccessMock, fsLstatMock, fsMkdirMock, @@ -48,6 +49,7 @@ const { return child }), resolveAuthorizedPathMock: vi.fn(), + authorizeExternalPathMock: vi.fn(), fsAccessMock: vi.fn(), fsLstatMock: vi.fn(), fsMkdirMock: vi.fn(), @@ -90,7 +92,8 @@ vi.mock('node:fs/promises', () => ({ vi.mock('../ipc/filesystem-auth', () => ({ PATH_ACCESS_DENIED_MESSAGE: 'Access denied: path resolves outside allowed directories. If this blocks a legitimate workflow, please file a GitHub issue.', - resolveAuthorizedPath: resolveAuthorizedPathMock + resolveAuthorizedPath: resolveAuthorizedPathMock, + authorizeExternalPath: authorizeExternalPathMock })) vi.mock('node:crypto', () => ({ @@ -548,30 +551,7 @@ describe('registerClipboardHandlers', () => { expect(removeHandlerMock).toHaveBeenCalledWith('clipboard:writeImage') expect(removeHandlerMock).toHaveBeenCalledWith('clipboard:writeFile') expect(removeHandlerMock).toHaveBeenCalledWith('clipboard:saveImageAsTempFile') - }) - - it('saves clipboard images to a local temp file when no connection is provided', async () => { - const png = Buffer.from([0, 1, 2, 3]) - const expectedPath = join( - '/tmp', - 'orca-paste-1760000000000-00000000-0000-4000-8000-000000000000.png' - ) - clipboardReadImageMock.mockReturnValue({ - getSize: () => ({ height: 1, width: 1 }), - isEmpty: () => false, - toPNG: () => png - }) - - registerClipboardHandlers({} as never) - - const handlers = getRegisteredHandlers() - await expect( - handlers.get('clipboard:saveImageAsTempFile')?.(makeClipboardEvent(), undefined) - ).resolves.toBe(expectedPath) - expect(fsWriteFileMock).toHaveBeenCalledWith(expectedPath, png) - expect(clipboardReadBufferMock).not.toHaveBeenCalled() - expect(fsOpenMock).not.toHaveBeenCalled() - expect(getSshFilesystemProviderMock).not.toHaveBeenCalled() + expect(removeHandlerMock).toHaveBeenCalledWith('clipboard:readImageThumbnail') }) it('does not inspect FileNameW when an empty image clipboard is read outside Windows', async () => { diff --git a/src/main/window/clipboard-ipc-handlers.ts b/src/main/window/clipboard-ipc-handlers.ts index 6322715ad77..9528958c25f 100644 --- a/src/main/window/clipboard-ipc-handlers.ts +++ b/src/main/window/clipboard-ipc-handlers.ts @@ -23,7 +23,8 @@ import { import { assertClipboardImageBase64LengthWithinLimit, assertClipboardImageByteLengthWithinLimit, - assertClipboardImageDimensionsWithinLimit + assertClipboardImageDimensionsWithinLimit, + type ClipboardImageThumbnail } from '../../shared/clipboard-image' import { writeFileToClipboard, @@ -37,6 +38,7 @@ import { } from './clipboard-remote-file-copy' import { saveClipboardImageBufferInRuntime } from './clipboard-runtime-image-upload' import { readWindowsClipboardImageFileAsPng } from './clipboard-windows-image-file' +import { buildClipboardImageThumbnail } from './clipboard-image-thumbnail' import { writeClipboardTextAndVerify } from './clipboard-text-write-verify' import { isDashboardPopoutRenderer } from './dashboard-popout-window' @@ -85,6 +87,7 @@ export function registerClipboardHandlers(store: Store): void { ipcMain.removeHandler('clipboard:writeImage') ipcMain.removeHandler('clipboard:writeFile') ipcMain.removeHandler('clipboard:saveImageAsTempFile') + ipcMain.removeHandler('clipboard:readImageThumbnail') void cleanupExpiredRemoteClipboardFiles() scheduleLegacyRemoteClipboardFileCleanup() @@ -100,6 +103,12 @@ export function registerClipboardHandlers(store: Store): void { return assertClipboardTextWithinLimitWithYield(clipboard.readText('selection'), options) } ) + // Why: an unanswered paste reads as a dropped paste, so the composer probes + // the clipboard in memory before the (slower) save lands. + ipcMain.handle('clipboard:readImageThumbnail', (event): ClipboardImageThumbnail | null => { + assertTrustedClipboardSender(event) + return buildClipboardImageThumbnail(clipboard.readImage()) + }) // Why: terminals need to detect clipboard images to support tools like Claude // Code that accept image input via paste. Writes the clipboard image to a // temp file and returns the path, or null if the clipboard has no image. diff --git a/src/preload/api/ui-bridge-clipboard-and-window-controls.ts b/src/preload/api/ui-bridge-clipboard-and-window-controls.ts index fdad19c2944..1738cb46d2a 100644 --- a/src/preload/api/ui-bridge-clipboard-and-window-controls.ts +++ b/src/preload/api/ui-bridge-clipboard-and-window-controls.ts @@ -10,6 +10,7 @@ import { type RichMarkdownContextMenuTableTarget } from '../../shared/rich-markdown-context-menu' import type { NativeFileDropPayload } from '../../shared/native-file-drop' +import type { ClipboardImageThumbnail } from '../../shared/clipboard-image' import type { ReadClipboardTextOptions } from '../../shared/clipboard-text' import { subscribeNativeFileDrop } from '../preload-runtime-support' import type { PreloadApi } from '../api-types' @@ -93,6 +94,8 @@ export const uiClipboardAndWindowControlsApi = { connectionId?: string | null runtimeEnvironmentId?: string | null }): Promise => ipcRenderer.invoke('clipboard:saveImageAsTempFile', args), + readClipboardImageThumbnail: (): Promise => + ipcRenderer.invoke('clipboard:readImageThumbnail'), writeClipboardText: (text: string): Promise => ipcRenderer.invoke('clipboard:writeText', text), writeTerminalClipboardText: (text: string): Promise => diff --git a/src/preload/api/ui-window-api.ts b/src/preload/api/ui-window-api.ts index fc6aef6991b..b0fa905efa7 100644 --- a/src/preload/api/ui-window-api.ts +++ b/src/preload/api/ui-window-api.ts @@ -1,3 +1,4 @@ +import type { ClipboardImageThumbnail } from '../../shared/clipboard-image' import type { ReadClipboardTextOptions } from '../../shared/clipboard-text' import type { NativeFileDropPayload } from '../../shared/native-file-drop' import type { @@ -12,6 +13,7 @@ export type UiWindowApi = { connectionId?: string | null runtimeEnvironmentId?: string | null }) => Promise + readClipboardImageThumbnail: () => Promise writeClipboardText: (text: string) => Promise writeTerminalClipboardText: (text: string) => Promise writeSelectionClipboardText: (text: string) => Promise diff --git a/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx b/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx index b1cf1a5879e..a1aecd23419 100644 --- a/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatComposer.test.tsx @@ -26,6 +26,7 @@ const mocks = vi.hoisted(() => ({ sessionOptionsSurface?: SessionOptionsSurface | null sessionOptionsSnapshot?: SessionOptionDescriptor[] attachDisabled?: boolean + sendButtonDisabled?: boolean } | null, modelSwitchOutcome: 'applied' as 'applied' | 'rejected' | 'unknown', confirmationObserver: null as { @@ -38,10 +39,11 @@ const mocks = vi.hoisted(() => ({ createClaudeModelSwitchConfirmationObserver: vi.fn(), discoverCommitMessageModels: vi.fn(), draft: 'hello', - imageAttachments: [] as { id: string; path: string }[], + imageAttachments: [] as { id: string; path: string; pending?: boolean }[], getMainBufferSnapshot: vi.fn(), sendHandle: { cancel: vi.fn(), settleAfterMs: 500 }, sendNativeChatMessage: vi.fn(), + sendNativeChatMessageWithImageAttachments: vi.fn(), sendNativeChatTypedCommand: vi.fn(), sendNativeChatMessageVerified: vi.fn(), typeNativeChatCommand: vi.fn(), @@ -81,7 +83,8 @@ vi.mock('./native-chat-runtime-send', () => ({ submitNativeChatPrompt: vi.fn() })) vi.mock('./native-chat-runtime-image-send', () => ({ - sendNativeChatMessageWithImageAttachments: vi.fn() + sendNativeChatMessageWithImageAttachments: (...args: unknown[]) => + mocks.sendNativeChatMessageWithImageAttachments(...args) })) vi.mock('./claude-model-switch-confirmation', () => ({ createClaudeModelSwitchConfirmationObserver: (...args: unknown[]) => @@ -206,6 +209,7 @@ describe('NativeChatComposer', () => { ] }) mocks.sendNativeChatMessage.mockReturnValue(mocks.sendHandle) + mocks.sendNativeChatMessageWithImageAttachments.mockReturnValue(mocks.sendHandle) mocks.sendNativeChatTypedCommand.mockReturnValue(mocks.sendHandle) mocks.sendNativeChatMessageVerified.mockResolvedValue(true) mocks.typeNativeChatCommand.mockResolvedValue(true) @@ -341,6 +345,60 @@ describe('NativeChatComposer', () => { expect(mocks.setDraft).toHaveBeenCalledWith('') }) + it('disables Send and blocks a send while an image attachment is still pending', () => { + mocks.imageAttachments = [{ id: 'image-1', path: '', pending: true }] + render( + + ) + + expect(mocks.fieldProps?.sendButtonDisabled).toBe(true) + + act(() => mocks.fieldProps?.onSend?.()) + + expect(mocks.sendNativeChatMessage).not.toHaveBeenCalled() + expect(mocks.sendNativeChatTypedCommand).not.toHaveBeenCalled() + expect(mocks.sendNativeChatMessageWithImageAttachments).not.toHaveBeenCalled() + }) + + it('enables Send and dispatches once a pending attachment resolves', () => { + mocks.imageAttachments = [{ id: 'image-1', path: '', pending: true }] + const view = render( + + ) + expect(mocks.fieldProps?.sendButtonDisabled).toBe(true) + + mocks.imageAttachments = [{ id: 'image-1', path: '/tmp/pasted.png' }] + view.rerender( + + ) + expect(mocks.fieldProps?.sendButtonDisabled).toBe(false) + + act(() => mocks.fieldProps?.onSend?.()) + + expect(mocks.sendNativeChatMessageWithImageAttachments).toHaveBeenCalledWith( + {}, + 'pty-1', + 'hello', + ['/tmp/pasted.png'], + undefined + ) + }) + it('types Codex slash composer sends instead of pasting them', () => { mocks.draft = '/status' render( diff --git a/src/renderer/src/components/native-chat/NativeChatComposer.tsx b/src/renderer/src/components/native-chat/NativeChatComposer.tsx index 9081f6adf65..e0b1d97a057 100644 --- a/src/renderer/src/components/native-chat/NativeChatComposer.tsx +++ b/src/renderer/src/components/native-chat/NativeChatComposer.tsx @@ -3,15 +3,10 @@ import { useAppStore } from '../../store' import { sendRuntimePtyInput } from '@/runtime/runtime-terminal-inspection' import { getSettingsForAgentTabRuntimeOwner } from '@/lib/agent-paste-draft' import { getVerifiedNativeChatCommands } from '../../../../shared/native-chat-agent-profiles' -import { - isStructuredAgentSessionComposerCommand, - STRUCTURED_AGENT_SESSION_SLASH_COMMANDS -} from '../../../../shared/structured-agent-session-composer' -import { emitNativeChatMessageSent } from '@/lib/native-chat-telemetry' +import { STRUCTURED_AGENT_SESSION_SLASH_COMMANDS } from '../../../../shared/structured-agent-session-composer' import { applyMentionSuggestion, EMPTY_HISTORY, - pushHistory, type HistoryState } from './native-chat-composer-state' import { useNativeChatDraft } from './use-native-chat-draft' @@ -34,8 +29,8 @@ import type { NativeChatComposerHandle, NativeChatComposerProps } from './native-chat-composer-types' -import { dispatchNativeChatStructuredComposerText } from './native-chat-structured-composer-dispatch' import { useNativeChatPtyComposerSend } from './use-native-chat-pty-composer-send' +import { useNativeChatStructuredComposerSend } from './use-native-chat-structured-composer-send' import { useImeEnterGestureOwnership } from '@/lib/ime-composition-keyboard-event' import { useNativeChatComposerAppMenuSelection } from './use-native-chat-composer-app-menu-selection' @@ -171,11 +166,21 @@ const NativeChatComposerPane = forwardRef attachment.pending) const sendButtonDisabled = isWorking ? !hasPty || !onStop - : disabled || (draft.trim() === '' && imageAttachments.length === 0) + : disabled || hasPendingAttachment || (draft.trim() === '' && imageAttachments.length === 0) const { insertTypedText, focus } = useNativeChatTypedInsertion({ textareaRef, @@ -201,6 +206,9 @@ const NativeChatComposerPane = forwardRef { - if (!structuredTransport) { - return - } - if (attachments.length > 0 && isStructuredAgentSessionComposerCommand(text, agent)) { - structuredTransport.onError('Remove attachments before using a chat-session command.') - return - } - void dispatchNativeChatStructuredComposerText(structuredTransport, text, attachments) - .then(({ accepted, error }) => { - structuredTransport.onError(error) - if (!accepted) { - return - } - emitNativeChatMessageSent({ agent, runtime: structuredTransport.runtime }) - setHistory((previous) => pushHistory(previous, text)) - setDraft('') - setCaret(0) - clearSkillOrigin() - clearImageAttachments() - }) - .catch((error) => - structuredTransport.onError(error instanceof Error ? error.message : String(error)) - ) - }, - [ - agent, - clearImageAttachments, - clearSkillOrigin, - imageAttachments, - setDraft, - structuredTransport - ] - ) + const sendStructured = useNativeChatStructuredComposerSend({ + agent, + imageAttachments, + structuredTransport, + clearImageAttachments, + clearSkillOrigin, + setHistory, + setDraft, + setCaret + }) const sendPty = useNativeChatPtyComposerSend({ agent, @@ -296,12 +279,23 @@ const NativeChatComposerPane = forwardRef { + if (hasPendingAttachment) { + return + } if (!structuredTransport) { sendPty() } else if ((draft.trim() !== '' || imageAttachments.length > 0) && !disabled) { sendStructured(draft, imageAttachments) } - }, [disabled, draft, imageAttachments, sendPty, sendStructured, structuredTransport]) + }, [ + disabled, + draft, + hasPendingAttachment, + imageAttachments, + sendPty, + sendStructured, + structuredTransport + ]) const interrupt = useCallback(() => { cancelPendingSends() diff --git a/src/renderer/src/components/native-chat/NativeChatComposerField.tsx b/src/renderer/src/components/native-chat/NativeChatComposerField.tsx index 346e44e8e54..51484d5d8c5 100644 --- a/src/renderer/src/components/native-chat/NativeChatComposerField.tsx +++ b/src/renderer/src/components/native-chat/NativeChatComposerField.tsx @@ -55,8 +55,14 @@ export type NativeChatComposerFieldProps = { export type NativeChatComposerImageAttachment = { id: string + /** Empty while `pending`: the clipboard image has no agent-readable path yet. */ path: string connectionId?: string + /** Clipboard thumbnail (blob/data URL) rendered before — and after — the file + * lands, so the chip never waits on a disk round-trip to show something. */ + previewUrl?: string + /** True while the pasted image is still being written to disk or uploaded. */ + pending?: boolean } /** diff --git a/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.test.tsx b/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.test.tsx new file mode 100644 index 00000000000..2b5924e9826 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.test.tsx @@ -0,0 +1,55 @@ +// @vitest-environment happy-dom + +import { afterEach, describe, expect, it, vi } from 'vitest' +import { cleanup, render, screen } from '@testing-library/react' +import { NativeChatImageAttachmentPreview } from './NativeChatImageAttachmentPreview' +import type { NativeChatComposerImageAttachment } from './NativeChatComposerField' + +const mocks = vi.hoisted(() => ({ + useLocalImageSrc: vi.fn() +})) + +vi.mock('@/i18n/i18n', () => ({ + translate: (_key: string, fallback: string) => fallback +})) + +vi.mock('@/components/editor/useLocalImageSrc', () => ({ + useLocalImageSrc: mocks.useLocalImageSrc +})) + +afterEach(() => { + cleanup() + vi.unstubAllGlobals() + mocks.useLocalImageSrc.mockReset() +}) + +function renderPreview(attachment: NativeChatComposerImageAttachment): void { + vi.stubGlobal('IntersectionObserver', undefined) + render() +} + +describe('NativeChatImageAttachmentPreview', () => { + it('shows the clipboard thumbnail and a spinner while pending', () => { + mocks.useLocalImageSrc.mockReturnValue(undefined) + renderPreview({ id: 'a1', path: '', previewUrl: 'blob:clipboard-1', pending: true }) + + expect(document.querySelector('.animate-spin')).toBeTruthy() + expect(screen.getByRole('img', { name: 'Saving pasted image…' }).getAttribute('src')).toBe( + 'blob:clipboard-1' + ) + }) + + it('renders no spinner once the attachment has settled', () => { + mocks.useLocalImageSrc.mockReturnValue('blob:on-disk-1') + renderPreview({ id: 'a1', path: '/tmp/example.png' }) + + expect(document.querySelector('.animate-spin')).toBeFalsy() + }) + + it('does not read the on-disk file while the attachment is pending', () => { + mocks.useLocalImageSrc.mockReturnValue(undefined) + renderPreview({ id: 'a1', path: '', previewUrl: 'blob:clipboard-1', pending: true }) + + expect(mocks.useLocalImageSrc).toHaveBeenCalledWith(undefined, '', undefined) + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.tsx b/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.tsx index 27d6cbd42cf..52e9a5f7c3a 100644 --- a/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.tsx +++ b/src/renderer/src/components/native-chat/NativeChatImageAttachmentPreview.tsx @@ -1,5 +1,5 @@ import { useEffect, useRef, useState } from 'react' -import { Image as ImageIcon, X } from 'lucide-react' +import { Image as ImageIcon, Loader2, X } from 'lucide-react' import { Dialog, DialogContent, DialogDescription, DialogTitle } from '@/components/ui/dialog' import { translate } from '@/i18n/i18n' import { basename } from '@/lib/path' @@ -41,31 +41,55 @@ export function NativeChatImageAttachmentPreview({ observer.observe(element) return () => observer.disconnect() }, []) - const previewSrc = useLocalImageSrc( - isNearViewport || isOpen ? attachment.path : undefined, + const isPending = attachment.pending === true + const localSrc = useLocalImageSrc( + !isPending && (isNearViewport || isOpen) ? attachment.path : undefined, attachment.path, attachment.connectionId ) + // The clipboard thumbnail is already in this process, so it renders with no + // round-trip; the on-disk file only wins for the full-size dialog. + const thumbnailSrc = attachment.previewUrl ?? localSrc + const fullSizeSrc = localSrc ?? attachment.previewUrl const filename = isNativeChatPastedImagePath(attachment.path) ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') : basename(attachment.path) + const pendingLabel = translate( + 'components.native-chat.composer.imageSaving', + 'Saving pasted image…' + ) + const label = isPending ? pendingLabel : filename return ( <>
    + {isPending ? ( + + + + ) : null}
    - {filename} + {label} {translate('components.native-chat.composer.imagePreview', 'Full-size image preview')}
    - {previewSrc ? ( + {fullSizeSrc ? ( {filename} ) : (
    - - {translate( - 'components.native-chat.composer.imagePreviewUnavailable', - 'Preview unavailable' + {isPending ? ( + <> + + {pendingLabel} + + ) : ( + <> + + {translate( + 'components.native-chat.composer.imagePreviewUnavailable', + 'Preview unavailable' + )} + )}
    )} diff --git a/src/renderer/src/components/native-chat/native-chat-image-paste.test.ts b/src/renderer/src/components/native-chat/native-chat-image-paste.test.ts index 35fdc107695..a331f2c0bea 100644 --- a/src/renderer/src/components/native-chat/native-chat-image-paste.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-image-paste.test.ts @@ -1,38 +1,15 @@ import { describe, expect, it } from 'vitest' -import { - getAgentImageHandling, - isNativeChatPastedImagePath, - resolveImagePaste -} from './native-chat-image-paste' +import { getAgentImageHandling, isNativeChatPastedImagePath } from './native-chat-image-paste' describe('image paste agent map', () => { - it('known image-capable agent attaches the temp file path', () => { + it('vision-capable TUIs take image attachments', () => { expect(getAgentImageHandling('claude')).toBe('attachment') - const result = resolveImagePaste('claude', '/tmp/orca-img-123.png') - expect(result).toEqual({ kind: 'attach', path: '/tmp/orca-img-123.png' }) - }) - - it('codex also attaches image paths', () => { - expect(resolveImagePaste('codex', '/tmp/x.png')).toEqual({ - kind: 'attach', - path: '/tmp/x.png' - }) - }) - - it('grok attaches image paths like other vision-capable TUIs', () => { + expect(getAgentImageHandling('codex')).toBe('attachment') expect(getAgentImageHandling('grok')).toBe('attachment') - expect(resolveImagePaste('grok', '/tmp/orca-paste-1.png')).toEqual({ - kind: 'attach', - path: '/tmp/orca-paste-1.png' - }) }) it('unknown/custom agent is unsupported', () => { expect(getAgentImageHandling('some-custom-agent')).toBe('unsupported') - expect(resolveImagePaste('some-custom-agent', '/tmp/x.png')).toEqual({ - kind: 'unsupported', - agent: 'some-custom-agent' - }) }) }) diff --git a/src/renderer/src/components/native-chat/native-chat-image-paste.ts b/src/renderer/src/components/native-chat/native-chat-image-paste.ts index 792d10445ed..5b05ae2d849 100644 --- a/src/renderer/src/components/native-chat/native-chat-image-paste.ts +++ b/src/renderer/src/components/native-chat/native-chat-image-paste.ts @@ -30,22 +30,6 @@ export function getAgentImageHandling(agent: AgentType): AgentImageHandling { return IMAGE_ATTACHMENT_AGENTS.has(agent) ? 'attachment' : 'unsupported' } -export type ImagePasteResult = - | { kind: 'attach'; path: string } - | { kind: 'unsupported'; agent: AgentType } - -/** - * Given the agent and the temp-file path the image was written to, decide what - * (if anything) to attach. Attachment-capable agents receive the path through - * the same bracketed image-paste channel as the terminal TUI. - */ -export function resolveImagePaste(agent: AgentType, tempFilePath: string): ImagePasteResult { - if (getAgentImageHandling(agent) === 'attachment') { - return { kind: 'attach', path: tempFilePath } - } - return { kind: 'unsupported', agent } -} - export function isNativeChatImageAttachmentPath(path: string): boolean { return isImageDropPath(path) } diff --git a/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.test.tsx b/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.test.tsx index 7a3c1c6ad8b..1825a0db92a 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.test.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.test.tsx @@ -260,4 +260,89 @@ describe('useNativeChatComposerAttachments', () => { ) act(() => probe.root.unmount()) }) + + it('settles a pending image attachment in place', async () => { + const probe = await renderProbe('pty-1') + let id: string | null = null + act(() => { + id = probe.latest().beginPendingImageAttachment('blob:preview-1') + }) + expect(id).toBeTruthy() + expect(probe.latest().imageAttachments).toMatchObject([ + { id, path: '', previewUrl: 'blob:preview-1', pending: true } + ]) + + act(() => { + probe.latest().resolvePendingImageAttachment(id as string, '/tmp/resolved.png', 'conn-1') + }) + + expect(probe.latest().imageAttachments).toMatchObject([ + { id, path: '/tmp/resolved.png', previewUrl: 'blob:preview-1', connectionId: 'conn-1' } + ]) + expect(probe.latest().imageAttachments[0]?.pending).toBeUndefined() + act(() => probe.root.unmount()) + }) + + it('drops just the targeted pending chip', async () => { + const probe = await renderProbe('pty-1') + let firstId: string | null = null + let secondId: string | null = null + act(() => { + firstId = probe.latest().beginPendingImageAttachment('blob:preview-1') + }) + act(() => { + secondId = probe.latest().beginPendingImageAttachment('blob:preview-2') + }) + + act(() => { + probe.latest().dropPendingImageAttachment(firstId as string) + }) + + expect(probe.latest().imageAttachments).toMatchObject([ + { id: secondId, previewUrl: 'blob:preview-2', pending: true } + ]) + act(() => probe.root.unmount()) + }) + + it('excludes a pending chip from the scope cache while a settled chip persists', async () => { + const probe = await renderProbe('pty-1') + let pendingId: string | null = null + act(() => { + pendingId = probe.latest().beginPendingImageAttachment('blob:preview-1') + }) + await act(async () => { + probe.latest().attachResolvedPaths(['/tmp/settled.png']) + }) + + const cached = readNativeChatAttachmentCache('pty-1') + expect(cached.some((attachment) => attachment.id === pendingId)).toBe(false) + expect(cached).toMatchObject([{ path: '/tmp/settled.png' }]) + expect(cached[0]?.previewUrl).toBeUndefined() + act(() => probe.root.unmount()) + }) + + it('revokes a blob: preview URL on removal but not a data: preview URL', async () => { + const probe = await renderProbe('pty-1') + const revoke = vi.spyOn(URL, 'revokeObjectURL') + let blobId: string | null = null + act(() => { + blobId = probe.latest().beginPendingImageAttachment('blob:preview-1') + }) + act(() => { + probe.latest().beginPendingImageAttachment('data:image/png;base64,AAAA') + }) + + act(() => { + probe.latest().dropPendingImageAttachment(blobId as string) + }) + expect(revoke).toHaveBeenCalledWith('blob:preview-1') + + // Only the remaining data: chip is left to clear; revoke must not fire again. + revoke.mockClear() + act(() => { + probe.latest().clearImageAttachments() + }) + expect(revoke).not.toHaveBeenCalled() + act(() => probe.root.unmount()) + }) }) diff --git a/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.ts b/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.ts index d9a0844cbd4..c7a8c7ab2fc 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.ts +++ b/src/renderer/src/components/native-chat/use-native-chat-composer-attachments.ts @@ -40,6 +40,9 @@ export function useNativeChatComposerAttachments({ clearImageAttachments: () => void flushPendingAttachments: () => void removeImageAttachment: (id: string) => void + beginPendingImageAttachment: (previewUrl?: string) => string | null + resolvePendingImageAttachment: (id: string, path: string, connectionId?: string | null) => void + dropPendingImageAttachment: (id: string) => void } { const [imageAttachments, setImageAttachments] = useState( () => readNativeChatAttachmentCache(attachmentScopeKey) @@ -72,6 +75,30 @@ export function useNativeChatComposerAttachments({ [attachmentScopeKey] ) + const nextAttachmentId = useCallback((): string => { + imageAttachmentCounter.current += 1 + return `${Date.now()}-${imageAttachmentCounter.current}` + }, []) + + // Local paths are only attachable when the composer's target runs locally; + // remote-runtime panes read a different filesystem than the one we resolved. + const attachmentTargetBlocked = useCallback((): boolean => { + const target = resolveTarget() + return ( + (!target && !allowWithoutTarget) || + Boolean(target && nativeChatComposerTargetIsRemote(target.ptyId)) + ) + }, [allowWithoutTarget, resolveTarget]) + + const noteAttachmentTargetBlocked = useCallback(() => { + setNotice( + translate( + 'components.native-chat.composer.localAttachmentUnsupported', + 'Local attachments are not available for remote sessions.' + ) + ) + }, [setNotice]) + const appendImageAttachments = useCallback( (paths: { path: string; connectionId?: string | null }[]) => { if (paths.length === 0) { @@ -79,16 +106,56 @@ export function useNativeChatComposerAttachments({ } updateImageAttachments((prev) => [ ...prev, - ...paths.map(({ path, connectionId }) => { - imageAttachmentCounter.current += 1 - return { - id: `${Date.now()}-${imageAttachmentCounter.current}`, - path, - connectionId: connectionId ?? undefined - } - }) + ...paths.map(({ path, connectionId }) => ({ + id: nextAttachmentId(), + path, + connectionId: connectionId ?? undefined + })) ]) }, + [nextAttachmentId, updateImageAttachments] + ) + + // Placeholder chip shown the instant a paste starts, so a clipboard image that + // takes a beat to save (or upload over SSH) never reads as a dropped paste. + const beginPendingImageAttachment = useCallback( + (previewUrl?: string): string | null => { + if (disabledRef.current) { + return null + } + if (attachmentTargetBlocked()) { + noteAttachmentTargetBlocked() + return null + } + const id = nextAttachmentId() + updateImageAttachments((prev) => [...prev, { id, path: '', previewUrl, pending: true }]) + return id + }, + [attachmentTargetBlocked, nextAttachmentId, noteAttachmentTargetBlocked, updateImageAttachments] + ) + + const resolvePendingImageAttachment = useCallback( + (id: string, path: string, connectionId?: string | null) => { + updateImageAttachments((prev) => + prev.map((attachment) => + attachment.id === id + ? { + ...attachment, + path, + connectionId: connectionId ?? undefined, + pending: undefined + } + : attachment + ) + ) + }, + [updateImageAttachments] + ) + + const dropPendingImageAttachment = useCallback( + (id: string) => { + updateImageAttachments((prev) => removeAttachmentById(prev, id)) + }, [updateImageAttachments] ) @@ -120,17 +187,8 @@ export function useNativeChatComposerAttachments({ focus: boolean, preserveNotice = false ) => { - const target = resolveTarget() - if ( - (!target && !allowWithoutTarget) || - (target && nativeChatComposerTargetIsRemote(target.ptyId)) - ) { - setNotice( - translate( - 'components.native-chat.composer.localAttachmentUnsupported', - 'Local attachments are not available for remote sessions.' - ) - ) + if (attachmentTargetBlocked()) { + noteAttachmentTargetBlocked() return } const imagePaths = resolvedPaths.filter(({ path }) => isNativeChatImageAttachmentPath(path)) @@ -150,10 +208,10 @@ export function useNativeChatComposerAttachments({ } }, [ - allowWithoutTarget, appendImageAttachments, + attachmentTargetBlocked, insertFileReferences, - resolveTarget, + noteAttachmentTargetBlocked, setNotice, textareaRef ] @@ -201,13 +259,37 @@ export function useNativeChatComposerAttachments({ return { imageAttachments, attachResolvedPaths, - clearImageAttachments: () => updateImageAttachments(() => []), + clearImageAttachments: () => + updateImageAttachments((prev) => { + prev.forEach(releaseAttachmentPreview) + return [] + }), flushPendingAttachments, - removeImageAttachment: (id) => - updateImageAttachments((prev) => prev.filter((attachment) => attachment.id !== id)) + removeImageAttachment: (id) => updateImageAttachments((prev) => removeAttachmentById(prev, id)), + beginPendingImageAttachment, + resolvePendingImageAttachment, + dropPendingImageAttachment } } +/** Object URLs minted from a clipboard blob leak until revoked; data URLs don't. */ +function releaseAttachmentPreview(attachment: NativeChatComposerImageAttachment): void { + if (attachment.previewUrl?.startsWith('blob:')) { + URL.revokeObjectURL(attachment.previewUrl) + } +} + +function removeAttachmentById( + attachments: readonly NativeChatComposerImageAttachment[], + id: string +): NativeChatComposerImageAttachment[] { + const removed = attachments.find((attachment) => attachment.id === id) + if (removed) { + releaseAttachmentPreview(removed) + } + return attachments.filter((attachment) => attachment.id !== id) +} + const attachmentCache = new Map() export function readNativeChatAttachmentCache( @@ -218,8 +300,16 @@ export function readNativeChatAttachmentCache( function writeNativeChatAttachmentCache( scopeKey: string, - attachments: readonly NativeChatComposerImageAttachment[] + cacheable: readonly NativeChatComposerImageAttachment[] ): void { + // A pending chip's save resolves into THIS hook instance; restoring one into a + // remount would strand it pending forever, so only settled chips are cached. + const attachments = cacheable + .filter((attachment) => !attachment.pending) + // Preview URLs can retain the full clipboard Blob (or a large data URL) for + // the lifetime of the scope cache. Settled attachments reload from their + // authorized path after a remount, so never retain the transient preview. + .map(({ previewUrl: _previewUrl, ...attachment }) => attachment) if (attachments.length === 0) { attachmentCache.delete(scopeKey) return diff --git a/src/renderer/src/components/native-chat/use-native-chat-composer-paste.test.tsx b/src/renderer/src/components/native-chat/use-native-chat-composer-paste.test.tsx index 7f5d0e2f1b9..111b8369f12 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-composer-paste.test.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-composer-paste.test.tsx @@ -6,7 +6,8 @@ import type { NativeChatAttachmentOwner } from './native-chat-attachment-upload' const mocks = vi.hoisted(() => ({ saveClipboardImageAsTempFile: vi.fn(), - readClipboardText: vi.fn() + readClipboardText: vi.fn(), + readClipboardImageThumbnail: vi.fn() })) vi.mock('@/i18n/i18n', () => ({ @@ -27,42 +28,73 @@ vi.stubGlobal('window', { api: { ui: { saveClipboardImageAsTempFile: mocks.saveClipboardImageAsTempFile, - readClipboardText: mocks.readClipboardText + readClipboardText: mocks.readClipboardText, + readClipboardImageThumbnail: mocks.readClipboardImageThumbnail } } }) +vi.stubGlobal('URL', { + createObjectURL: () => 'blob:clipboard-image', + revokeObjectURL: () => {} +}) import { useNativeChatComposerPaste } from './use-native-chat-composer-paste' type HookApi = ReturnType -function Probe({ - disabled, - resolveAttachmentOwner, - attachResolvedPaths, - insertTypedText, - setNotice, - onReady -}: { +/** Mirrors the composer's attachment list so tests can assert what the user sees. */ +type FakeChip = { id: string; path: string; previewUrl?: string; pending: boolean } + +function createChipStore(): { + chips: FakeChip[] + begin: (previewUrl?: string) => string | null + resolve: (id: string, path: string, connectionId?: string | null) => void + drop: (id: string) => void + connectionIds: (string | null | undefined)[] +} { + const chips: FakeChip[] = [] + const connectionIds: (string | null | undefined)[] = [] + let counter = 0 + return { + chips, + connectionIds, + begin: (previewUrl) => { + counter += 1 + const id = `chip-${counter}` + chips.push({ id, path: '', previewUrl, pending: true }) + return id + }, + resolve: (id, path, connectionId) => { + const chip = chips.find((candidate) => candidate.id === id) + if (chip) { + chip.path = path + chip.pending = false + } + connectionIds.push(connectionId) + }, + drop: (id) => { + const index = chips.findIndex((candidate) => candidate.id === id) + if (index !== -1) { + chips.splice(index, 1) + } + } + } +} + +type ProbeArgs = { disabled: boolean resolveAttachmentOwner: () => NativeChatAttachmentOwner - attachResolvedPaths: (paths: string[]) => void + attachResolvedPaths: (paths: string[], connectionId?: string | null) => void + beginPendingImageAttachment: (previewUrl?: string) => string | null + resolvePendingImageAttachment: (id: string, path: string, connectionId?: string | null) => void + dropPendingImageAttachment: (id: string) => void insertTypedText: (text: string) => boolean setNotice: (notice: string | null) => void onReady: (api: HookApi) => void -}): null { - onReady( - useNativeChatComposerPaste({ - agent: 'claude', - disabled, - caret: 0, - resolveAttachmentOwner, - attachResolvedPaths, - insertTypedText, - setCaret: () => {}, - setNotice - }) - ) +} + +function Probe({ onReady, ...args }: ProbeArgs): null { + onReady(useNativeChatComposerPaste({ agent: 'claude', caret: 0, setCaret: () => {}, ...args })) return null } @@ -71,12 +103,14 @@ let root: Root | null = null async function renderProbe(args: { disabled?: boolean resolveAttachmentOwner: () => NativeChatAttachmentOwner - attachResolvedPaths?: (paths: string[]) => void + attachResolvedPaths?: (paths: string[], connectionId?: string | null) => void + store?: ReturnType insertTypedText?: (text: string) => boolean setNotice?: (notice: string | null) => void }): Promise<{ latest: () => HookApi; setDisabled: (disabled: boolean) => Promise }> { const container = document.createElement('div') document.body.append(container) + const store = args.store ?? createChipStore() let api: HookApi | null = null root = createRoot(container) const render = async (disabled: boolean): Promise => { @@ -86,6 +120,9 @@ async function renderProbe(args: { disabled, resolveAttachmentOwner: args.resolveAttachmentOwner, attachResolvedPaths: args.attachResolvedPaths ?? (() => {}), + beginPendingImageAttachment: store.begin, + resolvePendingImageAttachment: store.resolve, + dropPendingImageAttachment: store.drop, insertTypedText: args.insertTypedText ?? (() => true), setNotice: args.setNotice ?? (() => {}), onReady: (next) => { @@ -113,7 +150,9 @@ function imagePasteEvent(): { defaultPrevented: boolean } { return { - clipboardData: { items: [{ type: 'image/png' }] } as unknown as DataTransfer, + clipboardData: { + items: [{ type: 'image/png', getAsFile: () => new Blob([], { type: 'image/png' }) }] + } as unknown as DataTransfer, preventDefault: vi.fn(), defaultPrevented: false } @@ -137,10 +176,10 @@ afterEach(() => { describe('useNativeChatComposerPaste', () => { it('does not save a clipboard image locally for a remote runtime', async () => { const setNotice = vi.fn() - const attachResolvedPaths = vi.fn() + const store = createChipStore() const probe = await renderProbe({ resolveAttachmentOwner: () => ({ kind: 'runtime' }), - attachResolvedPaths, + store, setNotice }) @@ -150,18 +189,19 @@ describe('useNativeChatComposerPaste', () => { 'Local attachments are not available for remote sessions.' ) expect(mocks.saveClipboardImageAsTempFile).not.toHaveBeenCalled() - expect(attachResolvedPaths).not.toHaveBeenCalled() + expect(mocks.readClipboardImageThumbnail).not.toHaveBeenCalled() + expect(store.chips).toHaveLength(0) }) it('surfaces a failed SSH image save through the composer notice', async () => { mocks.saveClipboardImageAsTempFile.mockRejectedValue( new Error('Remote connection dropped. Click Reconnect on the SSH target before retrying.') ) - const attachResolvedPaths = vi.fn() + const store = createChipStore() const setNotice = vi.fn() const probe = await renderProbe({ resolveAttachmentOwner: () => sshOwner, - attachResolvedPaths, + store, setNotice }) await act(async () => { @@ -170,24 +210,138 @@ describe('useNativeChatComposerPaste', () => { expect(setNotice).toHaveBeenCalledWith( 'Remote connection dropped. Click Reconnect on the SSH target before retrying.' ) - expect(attachResolvedPaths).not.toHaveBeenCalled() + // The optimistic chip must not outlive a failed save. + expect(store.chips).toHaveLength(0) }) - it('saves on the SSH host and attaches the returned remote path', async () => { + it('saves on the SSH host and settles the chip on the returned remote path', async () => { mocks.saveClipboardImageAsTempFile.mockResolvedValue('/remote/tmp/orca-paste-1.png') + const store = createChipStore() const attachResolvedPaths = vi.fn() const probe = await renderProbe({ resolveAttachmentOwner: () => sshOwner, + store, attachResolvedPaths }) await act(async () => { probe.latest().handlePaste(imagePasteEvent()) }) expect(mocks.saveClipboardImageAsTempFile).toHaveBeenCalledWith({ connectionId: 'conn-1' }) - expect(attachResolvedPaths).toHaveBeenCalledWith(['/remote/tmp/orca-paste-1.png']) + expect(store.chips).toEqual([ + { + id: 'chip-1', + path: '/remote/tmp/orca-paste-1.png', + previewUrl: 'blob:clipboard-image', + pending: false + } + ]) + // The chip carries the SSH connection so its preview reads over SFTP. + expect(store.connectionIds).toEqual(['conn-1']) + expect(attachResolvedPaths).not.toHaveBeenCalled() + }) + + it('shows a pending chip before the save resolves', async () => { + let resolveSave: (path: string) => void = () => {} + mocks.saveClipboardImageAsTempFile.mockReturnValue( + new Promise((resolve) => { + resolveSave = resolve + }) + ) + const store = createChipStore() + const probe = await renderProbe({ + resolveAttachmentOwner: () => ({ kind: 'local' }), + store + }) + await act(async () => { + probe.latest().handlePaste(imagePasteEvent()) + }) + expect(store.chips).toEqual([ + { id: 'chip-1', path: '', previewUrl: 'blob:clipboard-image', pending: true } + ]) + await act(async () => { + resolveSave('/tmp/orca-paste-1.png') + }) + expect(store.chips[0]).toMatchObject({ path: '/tmp/orca-paste-1.png', pending: false }) + }) + + it('does not settle a local path after the attachment owner changes', async () => { + let resolveSave: (path: string) => void = () => {} + let owner: NativeChatAttachmentOwner = { kind: 'local' } + mocks.saveClipboardImageAsTempFile.mockReturnValue( + new Promise((resolve) => { + resolveSave = resolve + }) + ) + const store = createChipStore() + const setNotice = vi.fn() + const probe = await renderProbe({ + resolveAttachmentOwner: () => owner, + store, + setNotice + }) + + await act(async () => { + probe.latest().handlePaste(imagePasteEvent()) + }) + expect(store.chips).toHaveLength(1) + + owner = sshOwner + await act(async () => { + resolveSave('/tmp/orca-paste-owner-changed.png') + }) + + expect(store.chips).toHaveLength(0) + expect(setNotice).toHaveBeenCalledWith('Worktree not ready — try again in a moment.') + }) + + it('shows a pending chip for menu paste from the clipboard thumbnail probe', async () => { + mocks.readClipboardImageThumbnail.mockResolvedValue({ + dataUrl: 'data:image/png;base64,AAA', + width: 1200, + height: 800 + }) + let resolveSave: (path: string) => void = () => {} + mocks.saveClipboardImageAsTempFile.mockReturnValue( + new Promise((resolve) => { + resolveSave = resolve + }) + ) + const store = createChipStore() + const probe = await renderProbe({ + resolveAttachmentOwner: () => ({ kind: 'local' }), + store + }) + await act(async () => { + probe.latest().pasteFromClipboard() + }) + expect(store.chips).toEqual([ + { id: 'chip-1', path: '', previewUrl: 'data:image/png;base64,AAA', pending: true } + ]) + await act(async () => { + resolveSave('/tmp/orca-paste-2.png') + }) + expect(store.chips[0]).toMatchObject({ path: '/tmp/orca-paste-2.png', pending: false }) + }) + + it('attaches directly when no clipboard preview was available', async () => { + mocks.readClipboardImageThumbnail.mockResolvedValue(null) + mocks.saveClipboardImageAsTempFile.mockResolvedValue('C:\\Temp\\orca-paste-3.png') + const store = createChipStore() + const attachResolvedPaths = vi.fn() + const probe = await renderProbe({ + resolveAttachmentOwner: () => ({ kind: 'local' }), + store, + attachResolvedPaths + }) + await act(async () => { + probe.latest().pasteFromClipboard() + }) + expect(store.chips).toHaveLength(0) + expect(attachResolvedPaths).toHaveBeenCalledWith(['C:\\Temp\\orca-paste-3.png'], null) }) it('stops pasteFromClipboard on a failed save instead of falling through to text', async () => { + mocks.readClipboardImageThumbnail.mockResolvedValue(null) mocks.saveClipboardImageAsTempFile.mockRejectedValue(new Error('sftp down')) const insertTypedText = vi.fn() const setNotice = vi.fn() @@ -205,17 +359,40 @@ describe('useNativeChatComposerPaste', () => { }) it('still falls through to text when the clipboard holds no image', async () => { + mocks.readClipboardImageThumbnail.mockResolvedValue(null) mocks.saveClipboardImageAsTempFile.mockResolvedValue(null) mocks.readClipboardText.mockResolvedValue('hello') const insertTypedText = vi.fn() + const store = createChipStore() const probe = await renderProbe({ resolveAttachmentOwner: () => ({ kind: 'local' }), + store, insertTypedText }) await act(async () => { probe.latest().pasteFromClipboard() }) expect(insertTypedText).toHaveBeenCalledWith('hello') + expect(store.chips).toHaveLength(0) + }) + + it('drops the pending chip when the clipboard changed between probe and save', async () => { + mocks.readClipboardImageThumbnail.mockResolvedValue({ + dataUrl: 'data:image/png;base64,AAA', + width: 10, + height: 10 + }) + mocks.saveClipboardImageAsTempFile.mockResolvedValue(null) + mocks.readClipboardText.mockResolvedValue('hello') + const store = createChipStore() + const probe = await renderProbe({ + resolveAttachmentOwner: () => ({ kind: 'local' }), + store + }) + await act(async () => { + probe.latest().pasteFromClipboard() + }) + expect(store.chips).toHaveLength(0) }) it('suppresses the failure notice when the composer became disabled mid-save', async () => { diff --git a/src/renderer/src/components/native-chat/use-native-chat-composer-paste.ts b/src/renderer/src/components/native-chat/use-native-chat-composer-paste.ts index c92c071209e..06bff8eb2f8 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-composer-paste.ts +++ b/src/renderer/src/components/native-chat/use-native-chat-composer-paste.ts @@ -2,7 +2,7 @@ import { useCallback, useRef } from 'react' import { translate } from '@/i18n/i18n' import { extractIpcErrorMessage } from '@/lib/ipc-error' import type { AgentType } from '../../../../shared/agent-status-types' -import { resolveImagePaste } from './native-chat-image-paste' +import { getAgentImageHandling } from './native-chat-image-paste' import { NATIVE_CHAT_CONTEXT_PASTE_MAX_BYTES } from './native-chat-composer-target' import { nativeChatLocalAttachmentUnsupportedNotice, @@ -19,7 +19,10 @@ export type UseNativeChatComposerPasteArgs = { /** Resolved at paste time: SSH panes must save the clipboard image on the * remote host, or the attached path names a file the agent cannot read. */ resolveAttachmentOwner: () => NativeChatAttachmentOwner - attachResolvedPaths: (paths: string[]) => void + attachResolvedPaths: (paths: string[], connectionId?: string | null) => void + beginPendingImageAttachment: (previewUrl?: string) => string | null + resolvePendingImageAttachment: (id: string, path: string, connectionId?: string | null) => void + dropPendingImageAttachment: (id: string) => void insertTypedText: (text: string) => boolean setCaret: (caret: number) => void setNotice: (notice: string | null) => void @@ -33,12 +36,39 @@ type ClipboardEventLike = { defaultPrevented: boolean } -function clipboardEventHasImage(event: ClipboardEventLike): boolean { +function clipboardEventImageFile(event: ClipboardEventLike): File | null { const data = event.clipboardData if (!data) { + return null + } + const item = Array.from(data.items).find((candidate) => candidate.type.startsWith('image/')) + return item?.getAsFile() ?? null +} + +/** Owners whose attachment path is a file this client can write right now. */ +function ownerAcceptsClipboardImage( + owner: NativeChatAttachmentOwner +): owner is Extract { + return owner.kind === 'local' || owner.kind === 'ssh' +} + +function ownerConnectionId(owner: NativeChatAttachmentOwner): string | null { + return owner.kind === 'ssh' ? owner.connectionId : null +} + +/** A save can outlive a worktree/connection switch; never settle its path into + * a composer whose backing host changed while the clipboard was in flight. */ +function attachmentOwnerStillMatches( + original: NativeChatAttachmentOwner, + current: NativeChatAttachmentOwner +): boolean { + if (original.kind !== current.kind) { return false } - return Array.from(data.items).some((item) => item.type.startsWith('image/')) + if (original.kind !== 'ssh') { + return true + } + return current.kind === 'ssh' && original.connectionId === current.connectionId } /** @@ -47,7 +77,12 @@ function clipboardEventHasImage(event: ClipboardEventLike): boolean { * `handlePaste` consumes a paste event (the textarea's onPaste *or* the * pane-level capture listener — the OS often retargets the event off the * focused textarea, so the pane listener is the reliable path); - * `pasteFromClipboard` is the menu-driven path with no event in hand. + * `pasteFromClipboard` is the menu-driven path with no event in hand (on macOS + * Cmd+V is routed here, so it must feel just as immediate). + * + * Both paths show a pending attachment chip before the image is saved: writing + * the file (or uploading it over SFTP) takes long enough that silence reads as + * a dropped paste and invites duplicate pastes. */ export function useNativeChatComposerPaste({ agent, @@ -55,6 +90,9 @@ export function useNativeChatComposerPaste({ caret, resolveAttachmentOwner, attachResolvedPaths, + beginPendingImageAttachment, + resolvePendingImageAttachment, + dropPendingImageAttachment, insertTypedText, setCaret, setNotice @@ -67,6 +105,7 @@ export function useNativeChatComposerPaste({ // the captured closure would otherwise attach/insert into a guarded composer. const disabledRef = useRef(disabled) disabledRef.current = disabled + const acceptsImages = getAgentImageHandling(agent) === 'attachment' // Distinguishes 'empty' (no image on the clipboard — text may fall through) // from 'failed' (save errored — the flow must stop and say why). @@ -102,22 +141,45 @@ export function useNativeChatComposerPaste({ [setNotice] ) - const attachClipboardImageTempFile = useCallback( - (tempPath: string) => { - const result = resolveImagePaste(agent, tempPath) - if (result.kind === 'unsupported') { - setNotice( - translate( - 'components.native-chat.composer.imageUnsupported', - 'Image paste is not supported for this agent.' - ) - ) + const noteImagesUnsupported = useCallback(() => { + setNotice( + translate( + 'components.native-chat.composer.imageUnsupported', + 'Image paste is not supported for this agent.' + ) + ) + }, [setNotice]) + + /** Settle the chip started at paste time, or attach directly when the paste + * produced no placeholder (no clipboard preview was available). */ + const settleImagePaste = useCallback( + ( + pendingId: string | null, + path: string, + connectionId: string | null, + originalOwner: NativeChatAttachmentOwner + ) => { + if (!attachmentOwnerStillMatches(originalOwner, resolveAttachmentOwner())) { + if (pendingId) { + dropPendingImageAttachment(pendingId) + } + setNotice(nativeChatWorktreeNotReadyNotice()) return } - attachResolvedPaths([result.path]) + if (pendingId) { + resolvePendingImageAttachment(pendingId, path, connectionId) + } else { + attachResolvedPaths([path], connectionId) + } setNotice(null) }, - [agent, attachResolvedPaths, setNotice] + [ + attachResolvedPaths, + dropPendingImageAttachment, + resolveAttachmentOwner, + resolvePendingImageAttachment, + setNotice + ] ) const handlePaste = useCallback( @@ -131,7 +193,8 @@ export function useNativeChatComposerPaste({ // textarea's native paste keeps its caret/undo behavior when it is the // event target. (When the OS retargets the paste off the textarea the // pane listener still routes text via pasteFromClipboard.) - if (!clipboardEventHasImage(event)) { + const imageFile = clipboardEventImageFile(event) + if (!imageFile) { return } event.preventDefault() @@ -140,25 +203,45 @@ export function useNativeChatComposerPaste({ setNotice(nativeChatWorktreeNotReadyNotice()) return } + if (!acceptsImages) { + noteImagesUnsupported() + return + } // Why: snapshot the caret before the async temp-file round-trip — `caret` // state can move (further typing/selection) while the await is in flight. const caretAtPaste = caret + // The clipboard blob is already in this process, so the chip can show the + // real image on the same tick the paste happens — no round-trip at all. + const previewUrl = ownerAcceptsClipboardImage(owner) + ? URL.createObjectURL(imageFile) + : undefined + const pendingId = previewUrl ? beginPendingImageAttachment(previewUrl) : null + if (previewUrl && !pendingId) { + URL.revokeObjectURL(previewUrl) + } void (async () => { const saved = await saveClipboardImageForOwner(owner) if (saved.status !== 'saved' || disabledRef.current) { + if (pendingId) { + dropPendingImageAttachment(pendingId) + } return } - attachClipboardImageTempFile(saved.tempPath) + settleImagePaste(pendingId, saved.tempPath, ownerConnectionId(owner), owner) setCaret(caretAtPaste) })() }, [ - attachClipboardImageTempFile, + acceptsImages, + beginPendingImageAttachment, caret, + dropPendingImageAttachment, + noteImagesUnsupported, resolveAttachmentOwner, saveClipboardImageForOwner, setCaret, - setNotice + setNotice, + settleImagePaste ] ) @@ -169,8 +252,23 @@ export function useNativeChatComposerPaste({ // way to LEARN whether the clipboard holds an image. An image then gets // the not-ready notice (never a local-path attach for a possibly-remote // worktree); plain text falls through unaffected. - const saved = await saveClipboardImageForOwner(owner) + // + // The in-memory thumbnail probe runs alongside the save rather than before + // it: it answers first (it never touches disk or the network), so the chip + // appears while the save is still in flight and text paste stays as fast. + const wantsPlaceholder = acceptsImages && ownerAcceptsClipboardImage(owner) + const thumbnailPromise = wantsPlaceholder + ? window.api.ui.readClipboardImageThumbnail().catch(() => null) + : Promise.resolve(null) + const savePromise = saveClipboardImageForOwner(owner) + const thumbnail = await thumbnailPromise + const pendingId = + thumbnail && !disabledRef.current ? beginPendingImageAttachment(thumbnail.dataUrl) : null + const saved = await savePromise if (disabledRef.current || saved.status === 'failed') { + if (pendingId) { + dropPendingImageAttachment(pendingId) + } return } if (saved.status === 'saved') { @@ -178,9 +276,17 @@ export function useNativeChatComposerPaste({ setNotice(nativeChatWorktreeNotReadyNotice()) return } - attachClipboardImageTempFile(saved.tempPath) + if (!acceptsImages) { + noteImagesUnsupported() + return + } + settleImagePaste(pendingId, saved.tempPath, ownerConnectionId(owner), owner) return } + // Clipboard changed between the probe and the save: no image to attach. + if (pendingId) { + dropPendingImageAttachment(pendingId) + } const text = await window.api.ui .readClipboardText({ maxBytes: NATIVE_CHAT_CONTEXT_PASTE_MAX_BYTES }) .catch(() => '') @@ -192,11 +298,15 @@ export function useNativeChatComposerPaste({ } })() }, [ - attachClipboardImageTempFile, + acceptsImages, + beginPendingImageAttachment, + dropPendingImageAttachment, insertTypedText, + noteImagesUnsupported, resolveAttachmentOwner, saveClipboardImageForOwner, - setNotice + setNotice, + settleImagePaste ]) return { handlePaste, pasteFromClipboard } diff --git a/src/renderer/src/components/native-chat/use-native-chat-structured-composer-send.ts b/src/renderer/src/components/native-chat/use-native-chat-structured-composer-send.ts new file mode 100644 index 00000000000..e871d65c62c --- /dev/null +++ b/src/renderer/src/components/native-chat/use-native-chat-structured-composer-send.ts @@ -0,0 +1,73 @@ +import { useCallback } from 'react' +import { emitNativeChatMessageSent } from '@/lib/native-chat-telemetry' +import { isStructuredAgentSessionComposerCommand } from '../../../../shared/structured-agent-session-composer' +import type { AgentType } from '../../../../shared/agent-status-types' +import { dispatchNativeChatStructuredComposerText } from './native-chat-structured-composer-dispatch' +import { pushHistory, type HistoryState } from './native-chat-composer-state' +import type { NativeChatStructuredComposerTransport } from './native-chat-composer-types' +import type { NativeChatComposerImageAttachment } from './NativeChatComposerField' + +export type UseNativeChatStructuredComposerSendArgs = { + agent: AgentType + imageAttachments: readonly NativeChatComposerImageAttachment[] + structuredTransport?: NativeChatStructuredComposerTransport + clearImageAttachments: () => void + clearSkillOrigin: () => void + setHistory: (updater: (previous: HistoryState) => HistoryState) => void + setDraft: (value: string) => void + setCaret: (caret: number) => void +} + +/** Send through the structured journal transport, clearing the composer only + * once the transport accepts (the PTY path has its own sibling hook). */ +export function useNativeChatStructuredComposerSend({ + agent, + imageAttachments, + structuredTransport, + clearImageAttachments, + clearSkillOrigin, + setHistory, + setDraft, + setCaret +}: UseNativeChatStructuredComposerSendArgs): ( + text: string, + attachments?: readonly NativeChatComposerImageAttachment[] +) => void { + return useCallback( + (text: string, attachments = imageAttachments): void => { + if (!structuredTransport) { + return + } + if (attachments.length > 0 && isStructuredAgentSessionComposerCommand(text, agent)) { + structuredTransport.onError('Remove attachments before using a chat-session command.') + return + } + void dispatchNativeChatStructuredComposerText(structuredTransport, text, attachments) + .then(({ accepted, error }) => { + structuredTransport.onError(error) + if (!accepted) { + return + } + emitNativeChatMessageSent({ agent, runtime: structuredTransport.runtime }) + setHistory((previous) => pushHistory(previous, text)) + setDraft('') + setCaret(0) + clearSkillOrigin() + clearImageAttachments() + }) + .catch((error) => + structuredTransport.onError(error instanceof Error ? error.message : String(error)) + ) + }, + [ + agent, + clearImageAttachments, + clearSkillOrigin, + imageAttachments, + setCaret, + setDraft, + setHistory, + structuredTransport + ] + ) +} diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index b893f1e8bdf..b546a930830 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16684,6 +16684,7 @@ "removeAttachment": "Remove attachment", "pastedImageLabel": "Pasted image", "imagePasteFailed": "Image paste failed.", + "imageSaving": "Saving pasted image…", "worktreeNotReady": "Worktree not ready — try again in a moment.", "uploadingAttachments": "Uploading {{value0}} file(s) to remote…", "model": "Model", diff --git a/src/renderer/src/web/preload-api/web-clipboard-api.ts b/src/renderer/src/web/preload-api/web-clipboard-api.ts index d86ac8c45b3..ffd955c2d0b 100644 --- a/src/renderer/src/web/preload-api/web-clipboard-api.ts +++ b/src/renderer/src/web/preload-api/web-clipboard-api.ts @@ -4,7 +4,9 @@ import { CLIPBOARD_IMAGE_MAX_SOURCE_BYTES, CLIPBOARD_IMAGE_TOO_LARGE_ERROR, assertClipboardImageByteLengthWithinLimit, - assertClipboardImageDimensionsWithinLimit + assertClipboardImageDimensionsWithinLimit, + clipboardImageThumbnailSize, + type ClipboardImageThumbnail } from '../../../../shared/clipboard-image' import { assertClipboardTextWriteWithinLimitWithYield } from '../../../../shared/clipboard-text' import { copyClipboardTextViaExecCommand } from '../web-clipboard-copy-fallback' @@ -72,6 +74,50 @@ export async function convertImageBlobToPng(blob: Blob): Promise { } } +async function readClipboardImageBlob(): Promise { + const clipboard = navigator.clipboard as + | (Clipboard & { read?: () => Promise }) + | undefined + if (!clipboard?.read) { + return null + } + const items = await clipboard.read() + for (const item of items) { + const imageType = item.types.find((type) => type.startsWith('image/')) + if (imageType) { + return item.getType(imageType) + } + } + return null +} + +/** Web counterpart of the main-process clipboard probe: decodes the clipboard + * image once and returns a small preview so the composer can show a chip while + * the full image is still being uploaded to the runtime. */ +export async function readClipboardImageThumbnail(): Promise { + const blob = await readClipboardImageBlob() + if (!blob) { + return null + } + assertClipboardImageBlobWithinLimit(blob) + const bitmap = await createImageBitmap(blob) + try { + assertClipboardImageDimensionsWithinLimit(bitmap) + const thumbnailSize = clipboardImageThumbnailSize(bitmap) + const canvas = document.createElement('canvas') + canvas.width = thumbnailSize.width + canvas.height = thumbnailSize.height + const context = canvas.getContext('2d') + if (!context) { + return null + } + context.drawImage(bitmap, 0, 0, thumbnailSize.width, thumbnailSize.height) + return { dataUrl: canvas.toDataURL('image/png'), height: bitmap.height, width: bitmap.width } + } finally { + bitmap.close() + } +} + export async function readClipboardImagePngBase64(): Promise { const clipboard = navigator.clipboard as | (Clipboard & { read?: () => Promise }) diff --git a/src/renderer/src/web/preload-api/web-ui-api.ts b/src/renderer/src/web/preload-api/web-ui-api.ts index ca67f5664a9..2755fc5663b 100644 --- a/src/renderer/src/web/preload-api/web-ui-api.ts +++ b/src/renderer/src/web/preload-api/web-ui-api.ts @@ -7,6 +7,7 @@ import { omitPairingLocalUiFields } from '../../../../shared/pairing-local-ui-fi import type { PairedUiState } from '../../../../shared/pairing-local-ui-fields' import { readClipboardImagePngBase64, + readClipboardImageThumbnail, saveClipboardImageAsTempFileInRuntime, writeWebClipboardText } from './web-clipboard-api' @@ -131,6 +132,7 @@ export function createWebUiApi(): NonNullable['ui']> { } return saveClipboardImageAsTempFileInRuntime(contentBase64, args) }, + readClipboardImageThumbnail: () => readClipboardImageThumbnail().catch(() => null), writeClipboardText: writeWebClipboardText, writeTerminalClipboardText: writeWebClipboardText, writeSelectionClipboardText: () => diff --git a/src/shared/clipboard-image.ts b/src/shared/clipboard-image.ts index 092467a9de2..28c1a276e46 100644 --- a/src/shared/clipboard-image.ts +++ b/src/shared/clipboard-image.ts @@ -36,3 +36,29 @@ export function assertClipboardImageDimensionsWithinLimit({ throw new Error(CLIPBOARD_IMAGE_TOO_LARGE_ERROR) } } + +/** Longest edge of the thumbnail the composer shows while the full clipboard + * image is still being written to disk. Small enough to cross IPC instantly. */ +export const CLIPBOARD_IMAGE_THUMBNAIL_MAX_EDGE = 320 + +export type ClipboardImageThumbnail = ClipboardImageDimensions & { + /** `data:image/png;base64,...` preview of the clipboard image. */ + dataUrl: string +} + +/** Scale `size` down so its longest edge fits the thumbnail budget. Returns the + * input unchanged when it already fits, so small images skip the resize. */ +export function clipboardImageThumbnailSize({ + height, + width +}: ClipboardImageDimensions): ClipboardImageDimensions { + const longestEdge = Math.max(width, height) + if (longestEdge <= CLIPBOARD_IMAGE_THUMBNAIL_MAX_EDGE) { + return { height, width } + } + const scale = CLIPBOARD_IMAGE_THUMBNAIL_MAX_EDGE / longestEdge + return { + height: Math.max(1, Math.round(height * scale)), + width: Math.max(1, Math.round(width * scale)) + } +} From 0886db2b90e1487f58cf0df7fc0ad5dfe6c788e9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:01:24 -0700 Subject: [PATCH 101/398] refactor(process-table): extract the correlation indexes into their own module (#18246) `src/shared/process-table-snapshot.ts` is 308 code lines against the 300 cap for `**/*.ts`, so `static analysis` is red on `main` and every open PR inherits it. Neither PR that grew the file crossed the cap alone. #18151 took it to 427 raw lines; #18166 added ~35 more. #18166's branch predated #18151, so the head CI linted was 428 raw lines and passed, while the squash onto main is 463 -> 308 code lines. The gate lints the PR head, not the merge result, so nothing linted the sum until it was on main. Pure move, no behaviour change: the generic index machinery (ProcessIdentityRow, ProcessTableIndexOf, buildProcessTableIndex, collectDescendantsFromIndex, lookupProcessTableIndex, getProcessTableIndex and its WeakMap) moves to process-table-index.ts. `ProcessTableIndex` and `scoreForegroundCandidateRow` stay behind because they need `ProcessTableRow`, which keeps the new module free of any import back and so introduces no cycle. --- .../agent-foreground-process-batch.test.ts | 4 +- .../agent-foreground-process-batch.ts | 8 +- .../providers/agent-foreground-process.ts | 3 +- .../windows-foreground-process-rows.ts | 5 +- src/relay/pty-shell-utils.ts | 2 +- src/shared/process-table-index.ts | 117 ++++++++++++++++++ src/shared/process-table-snapshot.test.ts | 10 +- src/shared/process-table-snapshot.ts | 114 +---------------- 8 files changed, 134 insertions(+), 129 deletions(-) create mode 100644 src/shared/process-table-index.ts diff --git a/src/main/providers/agent-foreground-process-batch.test.ts b/src/main/providers/agent-foreground-process-batch.test.ts index 784a62c3376..8baa37a7499 100644 --- a/src/main/providers/agent-foreground-process-batch.test.ts +++ b/src/main/providers/agent-foreground-process-batch.test.ts @@ -1,9 +1,9 @@ import { describe, expect, it } from 'vitest' +import { parseStrictProcessTableRows } from '../../shared/process-table-snapshot' import { buildProcessTableIndex, - parseStrictProcessTableRows, type ProcessTableIndexStats -} from '../../shared/process-table-snapshot' +} from '../../shared/process-table-index' import { resolveAgentForegroundProcessesBatch, resolveAgentForegroundProcessesFromIndex diff --git a/src/main/providers/agent-foreground-process-batch.ts b/src/main/providers/agent-foreground-process-batch.ts index ac9223a955f..2e0036bb603 100644 --- a/src/main/providers/agent-foreground-process-batch.ts +++ b/src/main/providers/agent-foreground-process-batch.ts @@ -7,13 +7,15 @@ import { resolveOuterWrapperForegroundProcess } from '../../shared/foreground-wr import { selectForegroundProcessCandidate } from '../../shared/foreground-process-selection' import type { ForegroundProcessEvidence } from '../../shared/foreground-process-evidence' import { - buildProcessTableIndex, getStrictProcessTableSnapshot, - lookupProcessTableIndex, type ProcessTableIndex, - type ProcessTableIndexStats, type ProcessTableRow } from '../../shared/process-table-snapshot' +import { + buildProcessTableIndex, + lookupProcessTableIndex, + type ProcessTableIndexStats +} from '../../shared/process-table-index' export type BatchedForegroundProcessRequest = { rootPid: number diff --git a/src/main/providers/agent-foreground-process.ts b/src/main/providers/agent-foreground-process.ts index 2000d35261c..017f88a7f9b 100644 --- a/src/main/providers/agent-foreground-process.ts +++ b/src/main/providers/agent-foreground-process.ts @@ -1,12 +1,11 @@ import { recognizeAgentProcessFromCommandLine } from '../../shared/agent-process-recognition' import { resolveOuterWrapperForegroundProcess } from '../../shared/foreground-wrapper-agent' import { - collectDescendantsFromIndex, getFreshProcessTableSnapshot, - getProcessTableIndex, getProcessTableSnapshot, type ProcessTableRow } from '../../shared/process-table-snapshot' +import { collectDescendantsFromIndex, getProcessTableIndex } from '../../shared/process-table-index' import { resolveWindowsAgentForegroundProcessWithAvailability, shouldInspectWindowsAgentForeground, diff --git a/src/main/providers/windows-foreground-process-rows.ts b/src/main/providers/windows-foreground-process-rows.ts index 5be4734dddc..5f462649e6c 100644 --- a/src/main/providers/windows-foreground-process-rows.ts +++ b/src/main/providers/windows-foreground-process-rows.ts @@ -1,7 +1,4 @@ -import { - collectDescendantsFromIndex, - getProcessTableIndex -} from '../../shared/process-table-snapshot' +import { collectDescendantsFromIndex, getProcessTableIndex } from '../../shared/process-table-index' import { readWindowsProcessTable, readWindowsProcessTableFresh, diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index febf025654c..e89ac60a7ed 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -10,11 +10,11 @@ import { } from '../shared/agent-process-recognition' import { getFirstCommandToken } from '../shared/command-token-scanner' import { - getProcessTableIndex, getProcessTableSnapshot, type ProcessTableIndex, type ProcessTableRow } from '../shared/process-table-snapshot' +import { getProcessTableIndex } from '../shared/process-table-index' import { selectForegroundProcessCandidate } from '../shared/foreground-process-selection' import { resolveOuterWrapperForegroundProcess, diff --git a/src/shared/process-table-index.ts b/src/shared/process-table-index.ts new file mode 100644 index 00000000000..d67d2fa8ada --- /dev/null +++ b/src/shared/process-table-index.ts @@ -0,0 +1,117 @@ +/** + * Correlation indexes over a process-table capture, generic over the row shape so the POSIX + * `ps` snapshot and the Windows process table share one pass instead of parallel ones. + */ + +export type ProcessTableIndexStats = { + captures?: number + indexBuilds: number + rowVisits: number + indexLookups: number +} + +/** The parent/child fields every process-table row shape shares. */ +export type ProcessIdentityRow = { pid: number; ppid: number } + +export type ProcessTableIndexOf = { + rows: readonly Row[] + byPid: ReadonlyMap + childrenByPpid: ReadonlyMap + stats?: ProcessTableIndexStats +} + +/** + * Build the correlation indexes in one linear pass over a capture. Only the + * indexes a resolver actually reads are materialized: group indexes would cost + * two more maps plus a per-row array allocation on every capture, and foreground + * membership is derived from each row's own `pgid` against the root's `tpgid`. + * + * Generic over the row shape so the Windows snapshot (`pid`/`ppid`/`name`/ + * `command`) shares this pass rather than carrying a parallel one. + */ +export function buildProcessTableIndex( + rows: readonly Row[], + stats?: ProcessTableIndexStats +): ProcessTableIndexOf { + if (stats) { + stats.indexBuilds += 1 + } + const byPid = new Map() + const childrenByPpid = new Map() + for (const row of rows) { + if (stats) { + stats.rowVisits += 1 + } + // Preserve rows.find() semantics if a malformed table repeats a pid + if (!byPid.has(row.pid)) { + byPid.set(row.pid, row) + } + const children = childrenByPpid.get(row.ppid) ?? [] + children.push(row) + childrenByPpid.set(row.ppid, children) + } + return { rows, byPid, childrenByPpid, stats } +} + +/** + * Depth-first descendants of `rootPid`, deepest-last, off a prebuilt index. + * + * Each row is copied with its depth, so callers may not mutate the index's rows + * through the result. Ordering matches a per-call `childrenByPpid` walk exactly: + * children keep capture order and the stack pops last-pushed first. + */ +export function collectDescendantsFromIndex( + index: ProcessTableIndexOf, + rootPid: number +): (Row & { depth: number })[] { + const descendants: (Row & { depth: number })[] = [] + const stack = (index.childrenByPpid.get(rootPid) ?? []).map((row) => ({ row, depth: 1 })) + while (stack.length > 0) { + const { row, depth } = stack.pop()! + descendants.push({ ...row, depth }) + for (const child of index.childrenByPpid.get(row.pid) ?? []) { + stack.push({ row: child, depth: depth + 1 }) + } + } + return descendants +} + +export function lookupProcessTableIndex( + index: ProcessTableIndexOf, + lookup: (index: ProcessTableIndexOf) => T, + stats = index.stats +): T { + if (stats) { + stats.indexLookups += 1 + } + return lookup(index) +} + +// Keyed by array identity, which also pins the row shape the entry was built +// for, so the one cast below cannot hand a caller another row type's index. +const processTableIndexes = new WeakMap() + +/** + * Memoize one index per snapshot identity, so the panes that share a TTL-cached + * capture walk its rows once instead of once each. Keyed weakly by the rows + * array, so an index dies with the snapshot that produced it. The shared build + * materializes only `byPid` and `childrenByPpid`, so a one-pane relay pays for + * two maps per capture rather than four indexes no resolver queries. + * + * Deliberately stats-free: `buildProcessTableIndex` mutates the caller's counter + * bag and stores it on the index, so a shared index would hand one caller's bag + * to an unrelated later caller and let a cache hit satisfy an `indexBuilds` + * measurement without building anything. Measured callers keep calling + * `buildProcessTableIndex(rows, stats)` directly. + */ +export function getProcessTableIndex( + rows: readonly Row[] +): ProcessTableIndexOf { + const cached = processTableIndexes.get(rows) as ProcessTableIndexOf | undefined + if (cached) { + return cached + } + const index = buildProcessTableIndex(rows) + processTableIndexes.set(rows, index) + return index +} diff --git a/src/shared/process-table-snapshot.test.ts b/src/shared/process-table-snapshot.test.ts index 5cf693dfdbf..acb1ba321da 100644 --- a/src/shared/process-table-snapshot.test.ts +++ b/src/shared/process-table-snapshot.test.ts @@ -7,18 +7,20 @@ const { execFileMock } = vi.hoisted(() => ({ execFileMock: vi.fn() })) vi.mock('node:child_process', () => ({ execFile: execFileMock })) import { - buildProcessTableIndex, createProcessTableSnapshotReader, - getProcessTableIndex, getProcessTableSnapshot, getStrictProcessTableSnapshot, parseProcessTableRows, parseStrictProcessTableRows, ProcessTableCaptureError, PS_MAX_BUFFER_BYTES, - resetProcessTableSnapshotForTests, - type ProcessTableIndexStats + resetProcessTableSnapshotForTests } from './process-table-snapshot' +import { + buildProcessTableIndex, + getProcessTableIndex, + type ProcessTableIndexStats +} from './process-table-index' function deferred(): { promise: Promise diff --git a/src/shared/process-table-snapshot.ts b/src/shared/process-table-snapshot.ts index 8340ae37a1b..17243d17b96 100644 --- a/src/shared/process-table-snapshot.ts +++ b/src/shared/process-table-snapshot.ts @@ -1,5 +1,6 @@ import { execFile as execFileCb } from 'node:child_process' import { promisify } from 'node:util' +import type { ProcessTableIndexOf } from './process-table-index' const execFile = promisify(execFileCb) @@ -122,81 +123,8 @@ export function parseStrictProcessTableRows(stdout: string): ProcessTableRow[] { return rows } -export type ProcessTableIndexStats = { - captures?: number - indexBuilds: number - rowVisits: number - indexLookups: number -} - -/** The parent/child fields every process-table row shape shares. */ -export type ProcessIdentityRow = { pid: number; ppid: number } - -export type ProcessTableIndexOf = { - rows: readonly Row[] - byPid: ReadonlyMap - childrenByPpid: ReadonlyMap - stats?: ProcessTableIndexStats -} - export type ProcessTableIndex = ProcessTableIndexOf -/** - * Build the correlation indexes in one linear pass over a capture. Only the - * indexes a resolver actually reads are materialized: group indexes would cost - * two more maps plus a per-row array allocation on every capture, and foreground - * membership is derived from each row's own `pgid` against the root's `tpgid`. - * - * Generic over the row shape so the Windows snapshot (`pid`/`ppid`/`name`/ - * `command`) shares this pass rather than carrying a parallel one. - */ -export function buildProcessTableIndex( - rows: readonly Row[], - stats?: ProcessTableIndexStats -): ProcessTableIndexOf { - if (stats) { - stats.indexBuilds += 1 - } - const byPid = new Map() - const childrenByPpid = new Map() - for (const row of rows) { - if (stats) { - stats.rowVisits += 1 - } - // Preserve rows.find() semantics if a malformed table repeats a pid - if (!byPid.has(row.pid)) { - byPid.set(row.pid, row) - } - const children = childrenByPpid.get(row.ppid) ?? [] - children.push(row) - childrenByPpid.set(row.ppid, children) - } - return { rows, byPid, childrenByPpid, stats } -} - -/** - * Depth-first descendants of `rootPid`, deepest-last, off a prebuilt index. - * - * Each row is copied with its depth, so callers may not mutate the index's rows - * through the result. Ordering matches a per-call `childrenByPpid` walk exactly: - * children keep capture order and the stack pops last-pushed first. - */ -export function collectDescendantsFromIndex( - index: ProcessTableIndexOf, - rootPid: number -): (Row & { depth: number })[] { - const descendants: (Row & { depth: number })[] = [] - const stack = (index.childrenByPpid.get(rootPid) ?? []).map((row) => ({ row, depth: 1 })) - while (stack.length > 0) { - const { row, depth } = stack.pop()! - descendants.push({ ...row, depth }) - for (const child of index.childrenByPpid.get(row.pid) ?? []) { - stack.push({ row: child, depth: depth + 1 }) - } - } - return descendants -} - /** * Rank a descendant row as a foreground candidate: a `+` (foreground process * group) row always outranks a background one, then the deepest wins. @@ -205,46 +133,6 @@ export function scoreForegroundCandidateRow(row: ProcessTableRow & { depth: numb return (row.stat.includes('+') ? 10_000 : 0) + row.depth } -export function lookupProcessTableIndex( - index: ProcessTableIndexOf, - lookup: (index: ProcessTableIndexOf) => T, - stats = index.stats -): T { - if (stats) { - stats.indexLookups += 1 - } - return lookup(index) -} - -// Keyed by array identity, which also pins the row shape the entry was built -// for, so the one cast below cannot hand a caller another row type's index. -const processTableIndexes = new WeakMap() - -/** - * Memoize one index per snapshot identity, so the panes that share a TTL-cached - * capture walk its rows once instead of once each. Keyed weakly by the rows - * array, so an index dies with the snapshot that produced it. The shared build - * materializes only `byPid` and `childrenByPpid`, so a one-pane relay pays for - * two maps per capture rather than four indexes no resolver queries. - * - * Deliberately stats-free: `buildProcessTableIndex` mutates the caller's counter - * bag and stores it on the index, so a shared index would hand one caller's bag - * to an unrelated later caller and let a cache hit satisfy an `indexBuilds` - * measurement without building anything. Measured callers keep calling - * `buildProcessTableIndex(rows, stats)` directly. - */ -export function getProcessTableIndex( - rows: readonly Row[] -): ProcessTableIndexOf { - const cached = processTableIndexes.get(rows) as ProcessTableIndexOf | undefined - if (cached) { - return cached - } - const index = buildProcessTableIndex(rows) - processTableIndexes.set(rows, index) - return index -} - type Snapshot = { value: T; capturedAtMs: number } type ProcessTableSnapshotReaderDeps = { From ef4e9c40ab05a58f714451b8ecb3b86dd334f1ca Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:03:21 -0700 Subject: [PATCH 102/398] fix(terminal): replay paired-runtime snapshots at the host's grid (#18132) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(terminal): replay paired-runtime snapshots at the host's grid A paired remote pane parsed the host's authoritative terminal image at whatever grid its own xterm happened to have. The host dimensions every snapshot it publishes, but only the REQUESTED snapshot path ever read `cols`/`rows` back — both PUSH paths (initial subscribe and server recovery) dropped them, so `onSnapshot` handed the transport an image with no grid and the drain wrote it as-is. Serialized frames are grid-relative: rows are newline-fed and the frame ends in an absolute CUP. Parsed at a different grid they re-wrap and clip, and because an alternate-screen TUI has no scrollback the rows scrolled off the top are gone. An idle agent never repaints, so the pane stays wrong until the next byte arrives — which for a finished Claude Code session is never. Carry the grid the host already publishes through the multiplexer and transport, then reuse the choreography the reattach payload already follows: resize to the source grid, replay, fit back to the pane, and push the resulting grid to the PTY. A host that publishes no dimensions reads as unknown and keeps today's behaviour, so no wire change and no capability negotiation is involved. * fix(terminal): keep the source-grid fit correct under mobile fit overrides Two follow-ups on the source-grid replay: - A mobile fit override skipped the post-replay fit entirely, stranding the pane at the host's replay grid. Fit without the PTY grid push instead, matching applyMainBufferSnapshot. - Reset the source-grid flag when a drain is scheduled: a transaction whose restore was skipped never runs afterRestore, and the stale flag would fit a later drain that never left the pane's own grid. * perf(terminal): clear the replay buffer before the source-grid resize The drain resized xterm to the host's serialization grid and only then wrote the clearing `2J`/`3J`/`H`. `clearBeforeReplay` is true for every pushed remote snapshot, so a column change reflowed a full scrollback that the next sequence discarded microseconds later — on the recovery push that lands under output flood, when the renderer is already loaded. The clear is grid-independent, so running it first is equivalent: the resize then operates on an empty buffer. Verified identical end state (content, cursor, buffer type, baseY) across cols-change, rows-change, alt-screen, no-scrollback and equal-grid shapes. Interleaved 25-run medians on a 10k-line scrollback: 6.19ms -> 2.48ms on the normal buffer, unchanged on the alternate screen (where `3J` cannot free the normal buffer's history, so the reflow is paid either way). --- ...ection-remote-snapshot-source-grid.test.ts | 242 ++++++++++++++++++ .../deferred-cold-restore-and-snapshot.ts | 5 +- .../pty-connection/replay-data-drain.ts | 91 ++++++- .../terminal-pane/pty-output-processor.ts | 7 +- .../terminal-pane/pty-transport-types.ts | 4 + ...e-runtime-pty-snapshot-source-grid.test.ts | 138 ++++++++++ ...time-pty-transport-snapshot-replay.test.ts | 6 +- .../remote-runtime-pty-transport.ts | 6 + ...emote-runtime-terminal-binary-snapshots.ts | 11 +- ...mote-runtime-terminal-multiplexer-types.ts | 4 + .../runtime/runtime-terminal-stream.test.ts | 13 +- 11 files changed, 516 insertions(+), 11 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/pty-connection-remote-snapshot-source-grid.test.ts create mode 100644 src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-source-grid.test.ts diff --git a/src/renderer/src/components/terminal-pane/pty-connection-remote-snapshot-source-grid.test.ts b/src/renderer/src/components/terminal-pane/pty-connection-remote-snapshot-source-grid.test.ts new file mode 100644 index 00000000000..f4e8d8bd2bb --- /dev/null +++ b/src/renderer/src/components/terminal-pane/pty-connection-remote-snapshot-source-grid.test.ts @@ -0,0 +1,242 @@ +import type * as React from 'react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { flushAsyncTicks, renderHeadlessBuffer } from './pty-connection-test-async' +import { createMockTransport, createPane, createManager } from './pty-connection-test-pane-fixtures' +import type { ConnectCallbacks, MockTransport } from './pty-connection-test-pane-fixtures' +import { buildPaneConnectionDeps } from './pty-connection-test-deps' +import { + createInitialStoreState, + buildActiveRuntimeEnvironmentState +} from './pty-connection-test-store-fixtures' +import type { StoreState } from './pty-connection-test-store-state' +import { + installTerminalTestGlobals, + restoreTerminalTestGlobals +} from './pty-connection-test-environment' + +const { + resetAndRefreshAllTerminalWebglAtlases, + scheduleTerminalWebglAtlasRecovery, + scheduleRuntimeGraphSync, + shouldSeedCacheTimerOnInitialTitle, + toastInfo, + notifyCodexPaneBoundForStaleSweep +} = vi.hoisted(() => ({ + resetAndRefreshAllTerminalWebglAtlases: vi.fn(), + scheduleTerminalWebglAtlasRecovery: vi.fn(), + scheduleRuntimeGraphSync: vi.fn(), + shouldSeedCacheTimerOnInitialTitle: vi.fn(() => false), + toastInfo: vi.fn(), + notifyCodexPaneBoundForStaleSweep: vi.fn() +})) + +let mockStoreState: StoreState +let transportFactoryQueue: MockTransport[] = [] +let createdTransportOptions: Record[] = [] +let storeSubscribers: ((state: StoreState) => void)[] = [] + +vi.mock('@/runtime/sync-runtime-graph', () => ({ + scheduleRuntimeGraphSync +})) + +vi.mock('@/lib/pane-manager/pane-manager-registry', async (importOriginal) => ({ + ...(await importOriginal>()), + resetAndRefreshAllTerminalWebglAtlases +})) + +vi.mock('./terminal-webgl-atlas-recovery', () => ({ + scheduleTerminalWebglAtlasRecovery +})) + +vi.mock('@/store', () => ({ + useAppStore: { + getState: () => mockStoreState, + subscribe: (listener: (state: StoreState) => void) => { + storeSubscribers.push(listener) + return () => { + storeSubscribers = storeSubscribers.filter((candidate) => candidate !== listener) + } + } + } +})) + +vi.mock('@/lib/agent-status', async (importOriginal) => { + const { buildAgentStatusModuleMock } = await import('./pty-connection-test-environment') + return buildAgentStatusModuleMock(await importOriginal>()) +}) + +vi.mock('./cache-timer-seeding', () => ({ + shouldSeedCacheTimerOnInitialTitle +})) + +vi.mock('sonner', () => ({ + toast: { info: toastInfo } +})) + +vi.mock('@/lib/codex-stale-pane-sweep', () => ({ + notifyCodexPaneBoundForStaleSweep +})) + +vi.mock('react', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + useCallback: unknown>(fn: T): T => fn + } +}) + +vi.mock('./pty-transport', () => ({ + createIpcPtyTransport: vi.fn((options: Record) => { + createdTransportOptions.push(options) + const nextTransport = transportFactoryQueue.shift() + if (!nextTransport) { + throw new Error('No mock transport queued') + } + return nextTransport + }) +})) + +vi.mock('./remote-runtime-pty-transport', () => ({ + createRemoteRuntimePtyTransport: vi.fn( + (_environmentId: string, options: Record) => { + createdTransportOptions.push(options) + const nextTransport = transportFactoryQueue.shift() + if (!nextTransport) { + throw new Error('No mock transport queued') + } + return nextTransport + } + ) +})) + +vi.mock('./pty-dispatcher', async (importOriginal) => { + const actual = await importOriginal>() + return { + ...actual, + getEagerPtyBufferHandle: vi.fn(() => undefined) + } +}) + +const HOST_COLS = 143 +const HOST_ROWS = 12 +const PANE_COLS = 120 +const PANE_ROWS = 40 + +// A serialized TUI frame the way @xterm/addon-serialize emits one: newline-fed +// rows plus a trailing absolute CUP. Both are grid-relative. +const HOST_FRAME = `\x1b[?1049h\x1b[2J\x1b[H${Array.from( + { length: HOST_ROWS }, + (_unused, index) => `host row ${index + 1}` +).join('\r\n')}\x1b[${HOST_ROWS};3H` + +function createDeps(overrides: Record = {}) { + return buildPaneConnectionDeps(() => mockStoreState, overrides) +} + +async function connectRemotePane(): Promise<{ + operations: { kind: 'resize' | 'write'; value: string }[] + pane: ReturnType + transport: MockTransport + replay: (data: string, meta?: Record) => void + dispose: () => void +}> { + const { connectPanePty } = await import('./pty-connection') + mockStoreState = buildActiveRuntimeEnvironmentState(mockStoreState, 'env-1') + const transport = createMockTransport('remote:env-1@@terminal-1') + const captured: { current: ConnectCallbacks['onReplayData'] | null } = { current: null } + transport.connect.mockImplementation(async ({ callbacks }: { callbacks: ConnectCallbacks }) => { + captured.current = callbacks.onReplayData ?? null + return { id: 'remote:env-1@@terminal-1', replay: '' } + }) + transportFactoryQueue.push(transport) + + const pane = createPane(1) + pane.terminal.cols = PANE_COLS + pane.terminal.rows = PANE_ROWS + const operations: { kind: 'resize' | 'write'; value: string }[] = [] + pane.terminal.write = vi.fn((data: string, callback?: () => void) => { + operations.push({ kind: 'write', value: data }) + callback?.() + }) + pane.terminal.resize = vi.fn((cols: number, rows: number) => { + operations.push({ kind: 'resize', value: `${cols}x${rows}` }) + pane.terminal.cols = cols + pane.terminal.rows = rows + }) + pane.fitAddon.proposeDimensions = vi.fn(() => ({ cols: PANE_COLS, rows: PANE_ROWS })) + pane.fitAddon.fit = vi.fn(() => { + pane.terminal.resize(PANE_COLS, PANE_ROWS) + }) + + const manager = createManager(1) + const disposable = connectPanePty(pane as never, manager as never, createDeps() as never) + await flushAsyncTicks(6) + transport.resize.mockClear() + + return { + operations, + pane, + transport, + replay: (data, meta) => captured.current?.(data, meta as never), + dispose: () => disposable.dispose() + } +} + +describe('pushed remote snapshot replay grid', () => { + beforeEach(() => { + vi.resetModules() + vi.clearAllMocks() + transportFactoryQueue = [] + createdTransportOptions = [] + storeSubscribers = [] + mockStoreState = createInitialStoreState(() => mockStoreState) + installTerminalTestGlobals() + }) + + afterEach(async () => { + await restoreTerminalTestGlobals() + }) + + it('replays at the host grid and then pushes the pane grid back to the PTY', async () => { + const session = await connectRemotePane() + + session.replay(HOST_FRAME, { snapshotCols: HOST_COLS, snapshotRows: HOST_ROWS }) + await flushAsyncTicks(20) + + const frameWriteIndex = session.operations.findIndex( + (operation) => operation.kind === 'write' && operation.value === HOST_FRAME + ) + const sourceResizeIndex = session.operations.findIndex( + (operation) => operation.kind === 'resize' && operation.value === `${HOST_COLS}x${HOST_ROWS}` + ) + expect(sourceResizeIndex).toBeGreaterThanOrEqual(0) + expect(frameWriteIndex).toBeGreaterThan(sourceResizeIndex) + // Why the PTY push matters: the pane must not be left driving the host at + // the replay geometry once the destination fit has run. + expect(session.transport.resize).toHaveBeenCalledWith(PANE_COLS, PANE_ROWS) + expect(session.transport.resize).not.toHaveBeenCalledWith(HOST_COLS, HOST_ROWS) + session.dispose() + }) + + it('keeps the pane grid when the host published no snapshot dimensions', async () => { + const session = await connectRemotePane() + + session.replay(HOST_FRAME) + await flushAsyncTicks(20) + + expect(session.pane.terminal.resize).not.toHaveBeenCalledWith(HOST_COLS, HOST_ROWS) + session.dispose() + }) + + it('only reproduces the host frame when it is parsed at the host grid', async () => { + const atHostGrid = await renderHeadlessBuffer([HOST_FRAME], HOST_COLS, HOST_ROWS) + const atPaneGrid = await renderHeadlessBuffer([HOST_FRAME], PANE_COLS, HOST_ROWS - 4) + + // Why this is the user-visible failure: the alternate screen has no + // scrollback, so rows scrolled off by a shorter grid are gone for good and + // an idle TUI never repaints them. + expect(atHostGrid.filter((line) => line.startsWith('host row'))).toHaveLength(HOST_ROWS) + expect(atPaneGrid.filter((line) => line.startsWith('host row')).length).toBeLessThan(HOST_ROWS) + expect(atPaneGrid).not.toContain('host row 1') + }) +}) diff --git a/src/renderer/src/components/terminal-pane/pty-connection/deferred-cold-restore-and-snapshot.ts b/src/renderer/src/components/terminal-pane/pty-connection/deferred-cold-restore-and-snapshot.ts index 28e0f95b9f7..6c3a0628f79 100644 --- a/src/renderer/src/components/terminal-pane/pty-connection/deferred-cold-restore-and-snapshot.ts +++ b/src/renderer/src/components/terminal-pane/pty-connection/deferred-cold-restore-and-snapshot.ts @@ -156,7 +156,10 @@ export function bindDeferredColdRestoreAndSnapshot(session: ConnectPanePtySessio } : {}), ...(meta.terminalOwner ? { terminalOwner: meta.terminalOwner } : {}), - ...(meta.alternateScreen !== undefined ? { alternateScreen: meta.alternateScreen } : {}) + ...(meta.alternateScreen !== undefined ? { alternateScreen: meta.alternateScreen } : {}), + ...(meta.snapshotCols !== undefined && meta.snapshotRows !== undefined + ? { snapshotCols: meta.snapshotCols, snapshotRows: meta.snapshotRows } + : {}) } session.scheduleReplayDataDrain() } diff --git a/src/renderer/src/components/terminal-pane/pty-connection/replay-data-drain.ts b/src/renderer/src/components/terminal-pane/pty-connection/replay-data-drain.ts index 87a2b6039ce..81ac85ed752 100644 --- a/src/renderer/src/components/terminal-pane/pty-connection/replay-data-drain.ts +++ b/src/renderer/src/components/terminal-pane/pty-connection/replay-data-drain.ts @@ -1,4 +1,8 @@ import { waitForTerminalOutputParsed } from '@/lib/pane-manager/pane-terminal-output-scheduler' +import { safeFit, safeFitAndThen } from '@/lib/pane-manager/pane-tree-ops' +import { getFitOverrideForPty } from '@/lib/pane-manager/mobile-fit-overrides' + +import { resolvePositiveTerminalDimensions } from '../terminal-snapshot-replay-paint' import { CURSOR_SHOW_SEQUENCE, @@ -66,6 +70,9 @@ export function bindReplayDataDrain(session: ConnectPanePtySession): void { session.pendingReplayData = null session.replayPayloadGeneration = 0 let replayDrainQueued = false + // Why: a payload replayed at a foreign grid leaves xterm sized to the source, + // so the destination fit belongs after the whole transaction parses. + let replayedAtSourceGrid = false const drainReplayDataQueue = async ( expectedPtyId: string | null, expectedStreamGeneration: number @@ -86,8 +93,15 @@ export function bindReplayDataDrain(session: ConnectPanePtySession): void { return false } const payload = session.pendingReplayData - const { data, clearBeforeReplay, pendingEscapeTailAnsi, alternateScreen, terminalOwner } = - payload + const { + data, + clearBeforeReplay, + pendingEscapeTailAnsi, + alternateScreen, + terminalOwner, + snapshotCols, + snapshotRows + } = payload session.pendingReplayData = null const isCurrentPayload = (): boolean => !session.disposed && @@ -100,12 +114,35 @@ export function bindReplayDataDrain(session: ConnectPanePtySession): void { // Relay replay buffers may overlap with content already rendered in // xterm. Local eager replay decides this earlier so metadata-only frames // can keep restored scrollback while still using the replay guard. + // Why ahead of the source-grid resize: the clear is grid-independent, so + // dropping the scrollback first spares a reflow of history the very next + // sequence discards (see use-terminal-container-fit-sync.ts on its cost). if (clearBeforeReplay) { await session.writeReplayDataAsync('\x1b[2J\x1b[3J\x1b[H') if (!isCurrentPayload()) { continue } } + // Why before the frame: the payload's wraps and cursor moves are relative + // to the grid the host serialized it at. Parsing it at the pane's own grid + // clips or re-wraps the image, and an idle TUI never repaints to correct + // it — the pane stays blank until the next byte arrives. + const sourceGrid = resolvePositiveTerminalDimensions(snapshotCols, snapshotRows) + if ( + sourceGrid && + (session.pane.terminal.cols !== sourceGrid.cols || + session.pane.terminal.rows !== sourceGrid.rows) + ) { + // Why suppressed: this resize is a layout step for parsing, not the + // pane's real geometry — the destination fit below owns the PTY grid. + session.suppressStructuralReplayPtyResize = true + try { + session.pane.terminal.resize(sourceGrid.cols, sourceGrid.rows) + } finally { + session.suppressStructuralReplayPtyResize = false + } + replayedAtSourceGrid = true + } if (clearBeforeReplay || data.length > 0) { // Why: an empty clearing frame is still an authoritative repaint and // must clear a stale agent signal from an earlier payload. @@ -148,12 +185,59 @@ export function bindReplayDataDrain(session: ConnectPanePtySession): void { } return appliedCurrentPayload } + // Why the same helper the reattach payload uses: a source-grid replay leaves + // xterm at the host's geometry, so the pane must fit back and push the + // resulting grid to the PTY before live bytes resume. + const fitAfterSourceGridReplay = async ( + scheduledPtyId: string | null, + scheduledStreamGeneration: number + ): Promise => { + if (!replayedAtSourceGrid) { + return + } + replayedAtSourceGrid = false + if ( + session.disposed || + !scheduledPtyId || + session.transport.getPtyId() !== scheduledPtyId || + session.transportStreamGeneration !== scheduledStreamGeneration + ) { + return + } + if (getFitOverrideForPty(scheduledPtyId)) { + // Why fit without the grid push: a mobile driver owns the PTY geometry, + // but the pane must still leave the host's replay grid. + safeFit(session.pane) + return + } + const gridPush = session.createReattachGridPush(scheduledStreamGeneration, scheduledPtyId) + const fit = safeFitAndThen(session.pane, 'replay-source-grid-fit', gridPush.continuation, { + shouldContinue: gridPush.shouldContinue, + retryIfUnmeasurable: true, + // Why: a hidden or parked pane must still leave the source grid once it + // is revealed, or the PTY stays pinned to the host's replay geometry. + deferIfHidden: true + }) + session.pendingReattachFit = fit + try { + await fit.completion + } finally { + if (session.pendingReattachFit === fit) { + session.pendingReattachFit = null + } + } + } + session.scheduleReplayDataDrain = (): void => { if (replayDrainQueued) { return } const scheduledPtyId = session.pendingReplayData?.ptyId ?? null replayDrainQueued = true + // Why reset here: a transaction whose restore was skipped never ran its + // afterRestore, and a stale flag would fit a later drain that never left + // the pane's own grid. + replayedAtSourceGrid = false // Why: live bytes are newer than the authoritative replay frame. Hold // them until clear + replay + reset have all parsed, or replay can erase them. const scheduledStreamGeneration = @@ -171,7 +255,8 @@ export function bindReplayDataDrain(session: ConnectPanePtySession): void { shouldRestore: () => !session.disposed && session.transport.getPtyId() === scheduledPtyId && - session.transportStreamGeneration === scheduledStreamGeneration + session.transportStreamGeneration === scheduledStreamGeneration, + afterRestore: () => fitAfterSourceGridReplay(scheduledPtyId, scheduledStreamGeneration) } ) ) diff --git a/src/renderer/src/components/terminal-pane/pty-output-processor.ts b/src/renderer/src/components/terminal-pane/pty-output-processor.ts index ac611e8ed0c..983b619b3bd 100644 --- a/src/renderer/src/components/terminal-pane/pty-output-processor.ts +++ b/src/renderer/src/components/terminal-pane/pty-output-processor.ts @@ -37,6 +37,8 @@ export type ProcessPtyOutputOptions = { snapshotSeq?: number alternateScreen?: boolean terminalOwner?: 'shell' + snapshotCols?: number + snapshotRows?: number } function removeSuppressedCursorNativeTitles( @@ -217,7 +219,10 @@ export function createPtyOutputProcessor({ ...(options.alternateScreen !== undefined ? { alternateScreen: options.alternateScreen } : {}), - ...(options.terminalOwner ? { terminalOwner: options.terminalOwner } : {}) + ...(options.terminalOwner ? { terminalOwner: options.terminalOwner } : {}), + ...(options.snapshotCols !== undefined && options.snapshotRows !== undefined + ? { snapshotCols: options.snapshotCols, snapshotRows: options.snapshotRows } + : {}) } if (Object.keys(replayMeta).length > 0) { callbacks.onReplayData(data, replayMeta) diff --git a/src/renderer/src/components/terminal-pane/pty-transport-types.ts b/src/renderer/src/components/terminal-pane/pty-transport-types.ts index f041f495c5c..b3b3d0700d8 100644 --- a/src/renderer/src/components/terminal-pane/pty-transport-types.ts +++ b/src/renderer/src/components/terminal-pane/pty-transport-types.ts @@ -60,6 +60,10 @@ export type PtyReplayDataMeta = { snapshotSeq?: number alternateScreen?: boolean terminalOwner?: 'shell' + /** Grid the payload was serialized at. Present only when the producer proved + * it; the drain replays there and fits back to the pane afterwards. */ + snapshotCols?: number + snapshotRows?: number } export type LocalPtySessionMetadata = { diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-source-grid.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-source-grid.test.ts new file mode 100644 index 00000000000..3f49e6e0b7c --- /dev/null +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-snapshot-source-grid.test.ts @@ -0,0 +1,138 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + TerminalStreamOpcode, + decodeTerminalStreamFrame, + decodeTerminalStreamJson, + encodeTerminalStreamFrame, + encodeTerminalStreamJson, + encodeTerminalStreamText +} from '../../../../shared/terminal-stream-protocol' + +// Client-side wire regression: the host dimensions every snapshot it publishes, +// but only the REQUESTED snapshot path ever read `cols`/`rows` back. Both PUSH +// paths (initial subscribe, server recovery) dropped them, so the pane parsed a +// host-grid image at its own grid and an idle TUI never repainted the damage. +// This drives REAL binary frames through the REAL multiplexer +// (decodeSnapshotInfo → onSnapshot meta) into the REAL transport +// (processData → onReplayData meta). One stream carries every case: the +// multiplexer is a module-level singleton, so separate cases would need +// separate module registries. + +describe('remote transport snapshot source-grid threading', () => { + const runtimeCall = vi.fn() + const runtimeSubscribe = vi.fn() + const subscriptionSendBinary = vi.fn() + let subscriptionCallbacks: { + onResponse: (response: unknown) => void + onBinary?: (bytes: Uint8Array) => void + onError?: (error: { code: string; message: string }) => void + onClose?: () => void + } | null = null + + beforeEach(() => { + vi.resetModules() + vi.doUnmock('../../runtime/remote-runtime-terminal-multiplexer') + vi.clearAllMocks() + subscriptionCallbacks = null + subscriptionSendBinary.mockReset() + runtimeCall.mockResolvedValue({ + ok: true, + result: { + terminal: { + handle: 'terminal-1', + tabId: 'tab-1', + leafId: 'pane:1', + worktreeId: 'wt-1' + } + } + }) + runtimeSubscribe.mockImplementation( + async (_args: unknown, callbacks: typeof subscriptionCallbacks) => { + subscriptionCallbacks = callbacks + return { unsubscribe: vi.fn(), sendBinary: subscriptionSendBinary } + } + ) + vi.stubGlobal('window', { + api: { + runtimeEnvironments: { call: runtimeCall, subscribe: runtimeSubscribe } + } + }) + }) + + it('carries the host grid on pushed snapshots and omits it when the host has none', async () => { + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + const onReplayData = vi.fn() + transport.attach({ + existingPtyId: 'remote:env-1@@terminal-1', + cols: 80, + rows: 24, + callbacks: { onReplayData } + }) + + await expect.poll(() => subscriptionCallbacks !== null, { timeout: 5000 }).toBe(true) + subscriptionCallbacks?.onResponse({ ok: true, result: { type: 'ready' } }) + await expect + .poll(() => subscriptionSendBinary.mock.calls.length, { timeout: 5000 }) + .toBeGreaterThan(0) + const subscribeFrame = subscriptionSendBinary.mock.calls + .map((call) => decodeTerminalStreamFrame(call[0] as Uint8Array)) + .find((frame) => frame?.opcode === TerminalStreamOpcode.Subscribe) + expect(subscribeFrame).toBeDefined() + const streamId = decodeTerminalStreamJson<{ streamId: number }>( + subscribeFrame!.payload + )!.streamId + + const deliverSnapshot = (start: Record, body: string): void => { + for (const frame of [ + encodeTerminalStreamFrame({ + opcode: TerminalStreamOpcode.SnapshotStart, + streamId, + seq: 0, + payload: encodeTerminalStreamJson(start) + }), + encodeTerminalStreamFrame({ + opcode: TerminalStreamOpcode.SnapshotChunk, + streamId, + seq: 0, + payload: encodeTerminalStreamText(body) + }), + encodeTerminalStreamFrame({ + opcode: TerminalStreamOpcode.SnapshotEnd, + streamId, + seq: 0, + payload: new Uint8Array(0) + }) + ]) { + subscriptionCallbacks?.onBinary?.(frame) + } + } + + // Initial subscribe push: the host's 143x43 grid must reach the restorer. + deliverSnapshot({ cols: 143, rows: 43, seq: 7, source: 'headless' }, 'restored TUI frame') + await expect.poll(() => onReplayData.mock.calls.length, { timeout: 5000 }).toBe(1) + expect(onReplayData).toHaveBeenLastCalledWith( + 'restored TUI frame', + expect.objectContaining({ snapshotCols: 143, snapshotRows: 43 }) + ) + + // Server-pushed recovery: untagged, after the initial snapshot landed. + deliverSnapshot({ cols: 154, rows: 68, seq: 9, source: 'headless' }, 'recovered') + await expect.poll(() => onReplayData.mock.calls.length, { timeout: 5000 }).toBe(2) + expect(onReplayData).toHaveBeenLastCalledWith( + '\x1b[2J\x1b[3J\x1b[Hrecovered', + expect.objectContaining({ snapshotCols: 154, snapshotRows: 68 }) + ) + + // A host that publishes no dimensions must read as unknown, not as a grid. + deliverSnapshot({ seq: 11, source: 'headless' }, 'undimensioned') + await expect.poll(() => onReplayData.mock.calls.length, { timeout: 5000 }).toBe(3) + const [, meta] = onReplayData.mock.calls[2] as [string, Record | undefined] + expect(meta?.snapshotCols).toBeUndefined() + expect(meta?.snapshotRows).toBeUndefined() + }) +}) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-snapshot-replay.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-snapshot-replay.test.ts index c6d1b8508f7..b6fc15f6413 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-snapshot-replay.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-snapshot-replay.test.ts @@ -127,7 +127,11 @@ describe('createRemoteRuntimePtyTransport', () => { emitOutput(streamId, liveOutput, liveSeq) expect(onReplayData).toHaveBeenCalledOnce() - expect(onReplayData).toHaveBeenCalledWith('AUTHORITATIVE_INITIAL_MARKER') + expect(onReplayData).toHaveBeenCalledWith( + 'AUTHORITATIVE_INITIAL_MARKER', + // The host's grid rides the snapshot so the pane replays it there. + expect.objectContaining({ snapshotCols: 80, snapshotRows: 24 }) + ) expect(onConnect).toHaveBeenCalledOnce() expect(onData).toHaveBeenCalledWith(liveOutput, expect.objectContaining({ seq: liveSeq })) await vi.waitFor(() => { diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts index a4290c5dbd5..1e5782fe496 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts @@ -1922,6 +1922,12 @@ export function createRemoteRuntimePtyTransport( : {}), ...(meta?.alternateScreen !== undefined && meta.seq !== undefined ? { alternateScreen: meta.alternateScreen } + : {}), + // Why unconditional on seq: the grid describes the image itself, + // not a stream boundary, so it is valid for every snapshot the + // host dimensions. Absent/zero degrades to the pane's own grid. + ...(meta?.cols !== undefined && meta.rows !== undefined + ? { snapshotCols: meta.cols, snapshotRows: meta.rows } : {}) }) } diff --git a/src/renderer/src/runtime/remote-runtime-terminal-binary-snapshots.ts b/src/renderer/src/runtime/remote-runtime-terminal-binary-snapshots.ts index c90c00fe00b..61cb242a856 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-binary-snapshots.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-binary-snapshots.ts @@ -93,7 +93,12 @@ export abstract class RemoteRuntimeTerminalBinarySnapshots extends RemoteRuntime seq: info?.seq, kittyKeyboardFlags: info?.kittyKeyboardFlags, alternateScreen: info?.alternateScreen, - terminalOwner: info?.terminalOwner + terminalOwner: info?.terminalOwner, + // Why: the image encodes wraps and cursor moves against the host's + // grid, so the restorer must replay it there — the request path has + // always carried these; the pushes silently dropped them. + cols: info?.cols, + rows: info?.rows }) } else if (target === 'recovery') { // Why: a server-pushed recovery snapshot replaces terminal state @@ -105,7 +110,9 @@ export abstract class RemoteRuntimeTerminalBinarySnapshots extends RemoteRuntime seq: info?.seq, kittyKeyboardFlags: info?.kittyKeyboardFlags, alternateScreen: info?.alternateScreen, - terminalOwner: info?.terminalOwner + terminalOwner: info?.terminalOwner, + cols: info?.cols, + rows: info?.rows }) } } else if (matchesPendingRequest) { diff --git a/src/renderer/src/runtime/remote-runtime-terminal-multiplexer-types.ts b/src/renderer/src/runtime/remote-runtime-terminal-multiplexer-types.ts index e8c6a286718..4b3a94aaf97 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-multiplexer-types.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-multiplexer-types.ts @@ -41,6 +41,10 @@ export type RemoteRuntimeMultiplexedTerminalCallbacks = { kittyKeyboardFlags?: number alternateScreen?: boolean terminalOwner?: 'shell' + /** Grid the host serialized this image at. Absent from hosts that omit + * it, which must read as unknown so replay keeps the pane's own grid. */ + cols?: number + rows?: number } ) => void onSubscribed?: () => void diff --git a/src/renderer/src/runtime/runtime-terminal-stream.test.ts b/src/renderer/src/runtime/runtime-terminal-stream.test.ts index 39cd8ce829a..2fcc8b2e293 100644 --- a/src/renderer/src/runtime/runtime-terminal-stream.test.ts +++ b/src/renderer/src/runtime/runtime-terminal-stream.test.ts @@ -572,7 +572,10 @@ describe('remote runtime terminal multiplex ACK gate', () => { injectSnapshot({ kind: 'scrollback', cols: 120, rows: 40, truncated: false }, 'initial state') expect(onSnapshot).toHaveBeenCalledWith('initial state', { - pendingEscapeTailAnsi: undefined + pendingEscapeTailAnsi: undefined, + // The host's serialization grid; the restorer replays there, not at the pane's own. + cols: 120, + rows: 40 }) expect(onSubscribed).toHaveBeenCalledTimes(1) @@ -590,7 +593,9 @@ describe('remote runtime terminal multiplex ACK gate', () => { // clears screen and scrollback first and must not replay the subscribe // lifecycle. expect(onSnapshot).toHaveBeenCalledWith(`\x1b[2J\x1b[3J\x1b[H${'recovered state'}`, { - pendingEscapeTailAnsi: undefined + pendingEscapeTailAnsi: undefined, + cols: 120, + rows: 40 }) expect(onSubscribed).toHaveBeenCalledTimes(1) @@ -607,7 +612,9 @@ describe('remote runtime terminal multiplex ACK gate', () => { '' ) expect(onSnapshot).toHaveBeenCalledWith('\x1b[2J\x1b[3J\x1b[H', { - pendingEscapeTailAnsi: undefined + pendingEscapeTailAnsi: undefined, + cols: 120, + rows: 40 }) expect(onSubscribed).toHaveBeenCalledTimes(1) From d8ca420cddc0a49df629484436ac7a54b137e74d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:03:46 -0700 Subject: [PATCH 103/398] perf(remote): apply the no-evidence inspection cadence to remote panes (#18146) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A visible remote/SSH terminal that has never run an agent inspected its execution host every ~2s forever for a strictly negative answer — ~30 RPC round trips per minute per pane, each a network hop plus a host-side foreground process scan. The `no-evidence` 15s cadence tier exists to bound exactly that volume, but `isProcessInspectionCostly` gated it on local Windows only and explicitly excluded remote-execution-host PTYs — the most expensive inspection shape in the codebase. Extract the predicate to `agent-process-inspection-cost.ts` and treat a remote-execution-host PTY as costly on every client platform. The local branch (Windows costly, POSIX cheap) is byte-identical. Client-side timer choice only: no wire change, no new field, no opcode. Activity (output/title/hook) re-arms the 2s cadence, agent evidence returns the tier to active/idle, and the `unavailable` branch and its error backoff are untouched. --- .../agent-completion-coordinator-types.ts | 8 +- ...ent-completion-no-evidence-cadence.test.ts | 85 +++++++++++++++++-- .../agent-process-inspection-cost.ts | 27 ++++++ .../pty-connection/terminal-keydown-fit.ts | 15 +--- 4 files changed, 112 insertions(+), 23 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/agent-process-inspection-cost.ts diff --git a/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts b/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts index c65fec024ad..43389f59ce2 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-coordinator-types.ts @@ -46,10 +46,10 @@ export type AgentCompletionCoordinatorOptions = { // this renderer CONSUMES that evidence and can tell "no evidence published" // from "host too old to publish it" — mixed-version hosts omit the field. shouldPollNoEvidenceProcessCadence?: () => boolean - // Why: on hosts where one inspection forks a whole-process-table scan (local - // Windows PowerShell/CIM), panes without agent evidence relax to a slow - // cadence; remote authorities can disable no-evidence polling entirely and - // re-arm from output/title activity instead. + // Why: where one inspection is a whole-process-table scan (local Windows + // PowerShell/CIM) or a host round trip plus a host-side scan (remote/SSH), + // panes without agent evidence relax to a slow cadence and re-arm from + // output/title/hook activity. See agent-process-inspection-cost.ts. isProcessInspectionCostly?: () => boolean shouldSuppressHookCompletion?: (payload: AgentCompletionStatusSnapshot) => boolean } diff --git a/src/renderer/src/components/terminal-pane/agent-completion-no-evidence-cadence.test.ts b/src/renderer/src/components/terminal-pane/agent-completion-no-evidence-cadence.test.ts index a0a35c36057..238f073fd38 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-no-evidence-cadence.test.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-no-evidence-cadence.test.ts @@ -1,20 +1,27 @@ // Regression guard: bound the volume of cadence process inspections a visible, // idle terminal with NO agent evidence drives on hosts where each inspection is -// a whole-process-table scan (local Windows forks powershell.exe/CIM — the -// scan-cost analogue of #6288). Pre-fix a single visible idle shell inspected -// every 2s forever (~30 scans/min); with the no-evidence tier it inspects every -// 15s, and pane activity (output/title/hook) or agent evidence re-arms the hot -// cadence so agent-start detection stays event-driven and agent-finish -// detection is unchanged. +// expensive — local Windows forks a powershell.exe/CIM whole-process-table scan +// (the scan-cost analogue of #6288), and a remote/SSH pane pays a host round +// trip plus a host-side foreground scan. Pre-fix a single visible idle shell +// inspected every 2s forever (~30 scans/min); with the no-evidence tier it +// inspects every 15s, and pane activity (output/title/hook) or agent evidence +// re-arms the hot cadence so agent-start detection stays event-driven and +// agent-finish detection is unchanged. import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { createAgentCompletionCoordinator, resetAgentCompletionCoordinatorIdentitiesForTest } from './agent-completion-coordinator' import { resetAgentProcessInspectionQueueForTests } from './agent-process-inspection-queue' +import { isAgentProcessInspectionCostly } from './agent-process-inspection-cost' +import { toRemoteRuntimePtyId } from '../../../../shared/remote-runtime-pty-id' +import { toAppSshPtyId } from '../../../../shared/ssh-pty-id' import type { RuntimeTerminalProcessInspection } from '@/runtime/runtime-terminal-inspection' import type { AgentCompletionCoordinatorOptions } from './agent-completion-coordinator-types' +const MAC_UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7)' +const WINDOWS_UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64)' + function processResult( foregroundProcess: string | null, hasChildProcesses = foregroundProcess !== null @@ -67,6 +74,43 @@ describe('agent completion no-evidence inspection cadence', () => { expect(inspectProcess).toHaveBeenCalledTimes(4) }) + it('bounds a visible idle remote pane through the shipped cost predicate', async () => { + // Why: a remote inspection is an RPC round trip to the execution host plus a + // host-side foreground scan — the costliest inspection shape here — yet it + // was excluded from the no-evidence tier on every client platform. + const sshPtyId = toAppSshPtyId('target-1', 'pty-1') + const inspectProcess = vi.fn(async () => processResult(null, false)) + const { coordinator } = createCoordinator(inspectProcess, { + getPtyId: () => sshPtyId, + isProcessInspectionCostly: () => isAgentProcessInspectionCostly(MAC_UA, sshPtyId) + }) + + coordinator.startProcessTracking() + await vi.advanceTimersByTimeAsync(60_000) + + // 60s / 15s = 4 host round trips. Pre-fix (2s idle cadence) this was 30. + expect(inspectProcess).toHaveBeenCalledTimes(4) + }) + + it('re-arms the remote pane to the 2s cadence on the first byte of PTY output', async () => { + // Why: agent-start detection on a remote pane must stay event-driven, not + // wait out the relaxed interval. + const runtimePtyId = toRemoteRuntimePtyId('term_1', 'env-a') + const inspectProcess = vi.fn(async () => processResult(null, false)) + const { coordinator } = createCoordinator(inspectProcess, { + getPtyId: () => runtimePtyId, + isProcessInspectionCostly: () => isAgentProcessInspectionCostly(MAC_UA, runtimePtyId) + }) + + coordinator.startProcessTracking() + await vi.advanceTimersByTimeAsync(14_000) + expect(inspectProcess).not.toHaveBeenCalled() + + coordinator.observeOutputActivity() + await vi.advanceTimersByTimeAsync(2_000) + expect(inspectProcess).toHaveBeenCalledTimes(1) + }) + it('keeps the full 2s idle cadence on hosts where inspection is cheap', async () => { const inspectProcess = vi.fn(async () => processResult(null, false)) const { coordinator } = createCoordinator(inspectProcess, { @@ -76,7 +120,7 @@ describe('agent completion no-evidence inspection cadence', () => { coordinator.startProcessTracking() await vi.advanceTimersByTimeAsync(60_000) - // 60s / 2s = 30: POSIX/SSH/remote panes must not be relaxed. + // 60s / 2s = 30: local POSIX panes (cheap `ps`) must not be relaxed. expect(inspectProcess).toHaveBeenCalledTimes(30) }) @@ -275,3 +319,30 @@ describe('agent completion no-evidence inspection cadence', () => { }) }) }) + +describe('isAgentProcessInspectionCostly', () => { + it('treats remote-execution-host ptys as costly on every client platform', () => { + for (const userAgent of [MAC_UA, WINDOWS_UA]) { + expect(isAgentProcessInspectionCostly(userAgent, toAppSshPtyId('target-1', 'pty-1'))).toBe( + true + ) + expect( + isAgentProcessInspectionCostly(userAgent, toRemoteRuntimePtyId('term_1', 'env-a')) + ).toBe(true) + expect(isAgentProcessInspectionCostly(userAgent, toRemoteRuntimePtyId('term_1'))).toBe(true) + } + }) + + it('leaves the local branch unchanged: Windows costly, POSIX cheap', () => { + expect(isAgentProcessInspectionCostly(WINDOWS_UA, 'worktree-1|pane-1')).toBe(true) + expect(isAgentProcessInspectionCostly(WINDOWS_UA, null)).toBe(false) + expect(isAgentProcessInspectionCostly(MAC_UA, 'worktree-1|pane-1')).toBe(false) + expect(isAgentProcessInspectionCostly(MAC_UA, null)).toBe(false) + }) + + // Why: a bare "ssh:" id names no connection, so it is not evidence the + // inspection crosses a link (see remote-execution-host-pty.test.ts). + it('does not relax a POSIX pane for an ssh-prefixed id carrying no relay pty id', () => { + expect(isAgentProcessInspectionCostly(MAC_UA, 'ssh:target-1')).toBe(false) + }) +}) diff --git a/src/renderer/src/components/terminal-pane/agent-process-inspection-cost.ts b/src/renderer/src/components/terminal-pane/agent-process-inspection-cost.ts new file mode 100644 index 00000000000..aeeea606bc0 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-process-inspection-cost.ts @@ -0,0 +1,27 @@ +import { isRemoteExecutionHostPtyId } from './remote-execution-host-pty' + +/** + * Whether one cadence process inspection for this pane is expensive enough that + * a pane with no agent evidence should relax to the `no-evidence` tier. + * + * Why remote first: a remote inspection is a `terminal.inspectProcess` / + * `pty.inspectProcess` round trip to the execution host plus a host-side + * foreground scan there — the costliest shape in this codebase, on every client + * platform. Local Windows is costly for a different reason: it forks a + * powershell.exe whole-process-table CIM scan per poll (~10-40x POSIX `ps`). + * Local POSIX (and daemon/WSL panes on it) stays on the full cadence. + * + * Relaxing is the interim measure: once this renderer consumes the batched + * foreground evidence direct-SSH/remote authorities already publish with their + * PTY inventory (#17525), those panes can drop to `shouldPollNoEvidenceProcessCadence` + * and stop scheduling idle host reads altogether. + */ +export function isAgentProcessInspectionCostly(userAgent: string, ptyId: string | null): boolean { + if (ptyId !== null && isRemoteExecutionHostPtyId(ptyId)) { + return true + } + if (!userAgent.includes('Windows')) { + return false + } + return ptyId !== null +} diff --git a/src/renderer/src/components/terminal-pane/pty-connection/terminal-keydown-fit.ts b/src/renderer/src/components/terminal-pane/pty-connection/terminal-keydown-fit.ts index 947ead70c31..387e4362e4b 100644 --- a/src/renderer/src/components/terminal-pane/pty-connection/terminal-keydown-fit.ts +++ b/src/renderer/src/components/terminal-pane/pty-connection/terminal-keydown-fit.ts @@ -11,7 +11,7 @@ import { resolveCompatibleAgentTypeForOwner } from '../../../../../shared/agent- import { registerTerminalSideEffectFactConsumer } from '../terminal-side-effect-facts-handler' import { isAgentTaskCompleteTrackingEnabled } from './agent-task-complete-settings' -import { isRemoteExecutionHostPtyId } from '../remote-execution-host-pty' +import { isAgentProcessInspectionCostly } from '../agent-process-inspection-cost' import { isRemoteRuntimePtyId } from './paired-parked-terminal-restore' import type { ConnectPanePtySession } from './connect-pane-pty-session' @@ -229,17 +229,8 @@ export function installTerminalKeydownFit(session: ConnectPanePtySession): void }), shouldPollProcessCadence: () => isAgentTaskCompleteTrackingEnabled() && session.deps.isVisibleRef.current, - isProcessInspectionCostly: () => { - // Why: local Windows inspection forks a powershell.exe whole-process-table - // CIM scan per poll (~10-40x heavier than POSIX `ps`). Keep the no-evidence - // cadence enabled until inventory evidence is consumed by this renderer; - // mixed-version relays may omit the optional field. - if (!navigator.userAgent.includes('Windows')) { - return false - } - const ptyId = session.transport.getPtyId() - return ptyId !== null && !isRemoteExecutionHostPtyId(ptyId) - }, + isProcessInspectionCostly: () => + isAgentProcessInspectionCostly(navigator.userAgent, session.transport.getPtyId()), isLive: () => { if (session.disposed) { return false From 4ff96df2b527b74e2d1001d4ef7c5073743ce099 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:05:29 -0700 Subject: [PATCH 104/398] Auto e2e tests autofix scheduled ci 1h run 32 20260902T0700 (#18227) * Fix flaky e2e tests with improved locators and synchronization Add explicit waits, use more robust element selectors, and simplify test setup to reduce race conditions. Replace file-based fixtures with programmatic browser creation, use parent-scoped locators for menu interactions, and poll for stable state before assertions. * Add E2E failure triage report for run 33564563164 - Reconciles 14 failed tests against job logs and trace artifacts - Categorizes failures: 8 product bugs, 2 flaky tests, 4 test updates - Documents test-maintenance fixes and diagnostic findings - Files 8 Linear issues with owners and fresh recurrence evidence - Provides next actions for product owners and repository maintenance * rm artifact notes * Refactor browser creation E2E test to use UI interactions - Click through menu instead of manipulating internal store state - Use Playwright's locator and toBeVisible() assertion patterns * Record E2E browser creation pageId before barrier check Move createdPageId assignment before the barrier arm/fire checks. This ensures the pageId is recorded unconditionally when tracking is enabled, allowing tests to distinguish between creations rejected before the host attempt vs those that failed after creation. * Remove browser page reclamation assertion from restart test Simplifies test by removing page ID tracking and poll checking if pages persist after paired runtime restart. --- .../web-runtime-browser-creation-e2e-fault.ts | 7 +- .../issue-12656-terminal-link-tooltip.spec.ts | 9 ++- ...er-creation-reconciliation-failure.spec.ts | 78 +++++++++---------- .../source-control-large-file-count.spec.ts | 4 +- tests/e2e/tabs.spec.ts | 4 +- ...tree-active-delete-scroll-position.spec.ts | 22 +----- 6 files changed, 58 insertions(+), 66 deletions(-) diff --git a/src/renderer/src/runtime/web-runtime-browser-creation-e2e-fault.ts b/src/renderer/src/runtime/web-runtime-browser-creation-e2e-fault.ts index db26cfa3d67..44a422a5d6a 100644 --- a/src/renderer/src/runtime/web-runtime-browser-creation-e2e-fault.ts +++ b/src/renderer/src/runtime/web-runtime-browser-creation-e2e-fault.ts @@ -179,10 +179,15 @@ export function throwIfE2eWebRuntimeBrowserCapabilityUnavailable(): void { } export async function pauseAfterE2eWebRuntimeBrowserCreate(remotePageId: string): Promise { - if (!e2eConfig.exposeStore || !armed || !createdPageBarrier) { + if (!e2eConfig.exposeStore) { return } + // Recorded before the arm check so a journey that never arms the barrier can still prove no host + // page was created — a null id is only evidence if a real create would have set one. createdPageId = remotePageId + if (!armed || !createdPageBarrier) { + return + } await createdPageBarrier } diff --git a/tests/e2e/issue-12656-terminal-link-tooltip.spec.ts b/tests/e2e/issue-12656-terminal-link-tooltip.spec.ts index 06fc725ee36..879c8305266 100644 --- a/tests/e2e/issue-12656-terminal-link-tooltip.spec.ts +++ b/tests/e2e/issue-12656-terminal-link-tooltip.spec.ts @@ -147,8 +147,13 @@ test.describe('Issue #12656 terminal link tooltip', () => { expect(Math.abs(idle.paneBottom - idle.terminalBottom)).toBeLessThanOrEqual(1) await expect .poll(async () => { - await moveToLink(orcaPage, probe) - return readTooltipState(orcaPage, probe.tabId) + const currentProbe = await locateUrl(orcaPage, url) + if (!currentProbe) { + return { display: 'none', text: '' } + } + probe = currentProbe + await moveToLink(orcaPage, currentProbe) + return readTooltipState(orcaPage, currentProbe.tabId) }) .toMatchObject({ display: '', text: expect.stringContaining(url) }) diff --git a/tests/e2e/paired-browser-creation-reconciliation-failure.spec.ts b/tests/e2e/paired-browser-creation-reconciliation-failure.spec.ts index 27bbfff03fb..be280fcff13 100644 --- a/tests/e2e/paired-browser-creation-reconciliation-failure.spec.ts +++ b/tests/e2e/paired-browser-creation-reconciliation-failure.spec.ts @@ -1,10 +1,7 @@ -import { writeFileSync } from 'node:fs' -import path from 'node:path' import type { Page, TestInfo } from '@stablyai/playwright-test' import { RuntimeClient } from '../../src/cli/runtime/client' import { expect, test } from './helpers/orca-app' import { readHostBrowserPageIds, readHostTabs } from './helpers/host-session-tabs' -import { openFileExplorer } from './helpers/file-explorer' import { launchHeadlessPairedRuntimeHost, type HeadlessPairedRuntimeHost @@ -17,8 +14,6 @@ import { } from './helpers/paired-electron-client' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' -const FIXTURE_NAME = 'paired-browser-reconcile-failure.html' - type FaultSnapshot = { armed: boolean capabilityRejectionArmed: boolean @@ -36,6 +31,22 @@ type FaultWindow = Window & { } } +// Drives the real create menu so the failure surfaces through handleNewBrowserTab's toast. +async function startBrowserCreate(page: Page): Promise { + await page.evaluate(() => window.__store?.getState().setBrowserDefaultUrl('about:blank')) + await page.getByRole('button', { name: 'New tab' }).first().click() + const newBrowserTab = page.getByRole('menuitem', { name: /New Browser Tab/i }) + await expect(newBrowserTab).toBeVisible({ timeout: 30_000 }) + await newBrowserTab.click() +} + +async function readStableHostTabs(hostClient: RuntimeClient, repoPath: string) { + const { publicationEpoch, snapshotVersion, ...state } = await readHostTabs(hostClient, repoPath) + expect(publicationEpoch).not.toBe('') + expect(snapshotVersion).toBeGreaterThan(0) + return state +} + type ClientTabState = { browserTabIds: string[] browserWorkspaceIds: string[] @@ -115,14 +126,7 @@ async function runReconciliationFailureJourney(args: { }) .toMatchObject({ terminalTabIds: expect.arrayContaining([expect.any(String)]) }) - await openFileExplorer(page) - const fixtureRow = page.locator('[data-file-explorer-row]').filter({ hasText: FIXTURE_NAME }) - await expect(fixtureRow).toBeVisible({ timeout: 30_000 }) - await fixtureRow.click() - const openPreviewToSide = page.getByRole('button', { name: 'Open Preview to the Side' }) - await expect(openPreviewToSide).toBeVisible({ timeout: 30_000 }) const baselineClient = await readClientTabs(page, worktreeId) - expect(baselineClient.editorTabIds).not.toHaveLength(0) expect(baselineClient.terminalTabIds).not.toHaveLength(0) const baselineHostBrowserIds = await readHostBrowserPageIds(args.hostClient, args.repoPath) @@ -133,7 +137,7 @@ async function runReconciliationFailureJourney(args: { } fault.arm() }) - await openPreviewToSide.click() + await startBrowserCreate(page) const faultSnapshot = await expect .poll( @@ -155,9 +159,8 @@ async function runReconciliationFailureJourney(args: { } expect(await readHostBrowserPageIds(args.hostClient, args.repoPath)).toContain(createdPageId) - // Why: the tab is staged on click, so while the create is held the user already sees it — - // exactly one of it, in the new split. The rollback assertions after release are what prove - // the optimism is unwound rather than stranded. + // The managed-browser action stages one tab in the active group while the host create is held. + // The rollback assertions prove that optimism is unwound rather than stranded. const heldClient = await readClientTabs(page, worktreeId) const addedSince = (baseline: string[], held: string[]): string[] => { expect(held).toEqual(expect.arrayContaining(baseline)) @@ -169,7 +172,7 @@ async function runReconciliationFailureJourney(args: { ).toHaveLength(1) expect(heldClient.editorTabIds).toEqual(baselineClient.editorTabIds) expect(heldClient.terminalTabIds).toEqual(baselineClient.terminalTabIds) - expect(addedSince(baselineClient.groupIds, heldClient.groupIds)).toHaveLength(1) + expect(heldClient.groupIds).toEqual(baselineClient.groupIds) await page.screenshot({ path: args.testInfo.outputPath(`${args.topology}-browser-reconciliation-held.png`), @@ -181,9 +184,9 @@ async function runReconciliationFailureJourney(args: { ) ).toBe(true) - await expect(page.getByText('Unable to open this file in Orca Browser.')).toBeVisible({ - timeout: 30_000 - }) + await expect( + page.getByText('The paired runtime could not create a managed browser tab.') + ).toBeVisible({ timeout: 30_000 }) await expect .poll(() => readHostBrowserPageIds(args.hostClient, args.repoPath), { timeout: 30_000, @@ -261,14 +264,8 @@ async function runCapabilityFailureJourney(args: { }) .toMatchObject({ terminalTabIds: expect.arrayContaining([expect.any(String)]) }) - await openFileExplorer(page) - const fixtureRow = page.locator('[data-file-explorer-row]').filter({ hasText: FIXTURE_NAME }) - await expect(fixtureRow).toBeVisible({ timeout: 30_000 }) - await fixtureRow.click() - const openPreviewToSide = page.getByRole('button', { name: 'Open Preview to the Side' }) - await expect(openPreviewToSide).toBeVisible({ timeout: 30_000 }) const baselineClient = await readClientTabs(page, worktreeId) - const baselineHost = await readHostTabs(args.hostClient, args.repoPath) + const baselineHost = await readStableHostTabs(args.hostClient, args.repoPath) await page.evaluate(() => { const fault = (window as FaultWindow).__webRuntimeBrowserCreationFault @@ -277,18 +274,25 @@ async function runCapabilityFailureJourney(args: { } fault.armCapabilityRejection() }) - await openPreviewToSide.click() + await startBrowserCreate(page) - await expect(page.getByText('Unable to open this file in Orca Browser.')).toBeVisible({ + await expect(page.getByText(/E2E forced browser capability rejection/)).toBeVisible({ timeout: 30_000 }) + // Why: baseline equality alone also holds for a create that was rolled back. A null page id is + // what separates rejecting before the host create from undoing one afterwards. + expect( + await page.evaluate( + () => (window as FaultWindow).__webRuntimeBrowserCreationFault?.snapshot() ?? null + ) + ).toMatchObject({ createdPageId: null }) await expect .poll(() => readClientTabs(page, worktreeId), { timeout: 30_000, message: 'client split state did not settle after capability rejection' }) .toEqual(baselineClient) - expect(await readHostTabs(args.hostClient, args.repoPath)).toEqual(baselineHost) + expect(await readStableHostTabs(args.hostClient, args.repoPath)).toEqual(baselineHost) await page.screenshot({ path: args.testInfo.outputPath(`${args.topology}-browser-capability-rejected.png`), fullPage: true @@ -305,10 +309,6 @@ test('rolls back a headed-host browser when client reconciliation times out @hea testRepoPath }, testInfo) => { test.setTimeout(300_000) - writeFileSync( - path.join(testRepoPath, FIXTURE_NAME), - '

    browser reconciliation fault

    \n' - ) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) @@ -326,16 +326,12 @@ test('rolls back a headed-host browser when client reconciliation times out @hea }) }) -test('cleans up a headed-host preview when capability rejects after preflight @headful', async ({ +test('cleans up a headed-host browser when capability rejects before create @headful', async ({ electronApp, orcaPage, testRepoPath }, testInfo) => { test.setTimeout(300_000) - writeFileSync( - path.join(testRepoPath, FIXTURE_NAME), - '

    browser capability fault

    \n' - ) await waitForSessionReady(orcaPage) await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) @@ -355,10 +351,6 @@ test('cleans up a headed-host preview when capability rejects after preflight @h test('keeps browser failure cleanup on a headless host', async ({ testRepoPath }, testInfo) => { test.setTimeout(300_000) - writeFileSync( - path.join(testRepoPath, FIXTURE_NAME), - '

    headless capability fault

    \n' - ) const host: HeadlessPairedRuntimeHost = await launchHeadlessPairedRuntimeHost() try { await host.client.call('repo.add', { path: testRepoPath, kind: 'git' }) diff --git a/tests/e2e/source-control-large-file-count.spec.ts b/tests/e2e/source-control-large-file-count.spec.ts index fb98eb9a123..28a5e7758f2 100644 --- a/tests/e2e/source-control-large-file-count.spec.ts +++ b/tests/e2e/source-control-large-file-count.spec.ts @@ -438,7 +438,9 @@ test.describe('Source Control large file count (#8013)', () => { // explicit recovery path after the underlying change count drops. removeLargeFileCountUntrackedTree(fixture.repoPath) await expect(tooManyChangesBanner).toBeVisible() - await orcaPage.getByRole('button', { name: 'Retry' }).click() + const retryButton = tooManyChangesBanner.locator('..').getByRole('button', { name: 'Retry' }) + await expect(retryButton).toBeVisible() + await retryButton.click() await expect(tooManyChangesBanner).not.toBeVisible() await expect .poll(() => diff --git a/tests/e2e/tabs.spec.ts b/tests/e2e/tabs.spec.ts index b89e47d4079..2faabafc1cc 100644 --- a/tests/e2e/tabs.spec.ts +++ b/tests/e2e/tabs.spec.ts @@ -29,7 +29,8 @@ import { getActiveTabType, getWorktreeTabs, getTabBarOrder, - ensureTerminalVisible + ensureTerminalVisible, + waitForStartupWorktreeRefresh } from './helpers/store' const SORTABLE_TAB = '[data-testid="sortable-tab"]' @@ -69,6 +70,7 @@ async function getFocusedTerminalTabId(page: Page): Promise { test.describe('Tabs', () => { test.beforeEach(async ({ orcaPage }) => { await waitForSessionReady(orcaPage) + await waitForStartupWorktreeRefresh(orcaPage) await waitForActiveWorktree(orcaPage) await ensureTerminalVisible(orcaPage) }) diff --git a/tests/e2e/worktree-active-delete-scroll-position.spec.ts b/tests/e2e/worktree-active-delete-scroll-position.spec.ts index 1b9d4bc3283..5e2f15f5b75 100644 --- a/tests/e2e/worktree-active-delete-scroll-position.spec.ts +++ b/tests/e2e/worktree-active-delete-scroll-position.spec.ts @@ -237,24 +237,10 @@ test('deleting the active scrolled worktree preserves position and closes the ro `[data-worktree-sidebar] [data-worktree-id=${JSON.stringify(belowId)}]` ) await pauseForVisualProof(orcaPage) - await target.evaluate((element) => { - const scope = element.querySelector( - '[data-worktree-context-menu-scope="worktree"]' - ) - if (!scope) { - throw new Error('Worktree context-menu scope is unavailable') - } - scope.dispatchEvent( - new MouseEvent('contextmenu', { - bubbles: true, - button: 2, - cancelable: true, - clientX: scope.getBoundingClientRect().left + 10, - clientY: scope.getBoundingClientRect().top + 10 - }) - ) - }) - const deleteItem = orcaPage.getByRole('menuitem', { name: 'Delete', exact: true }) + const contextMenuScope = target.locator('[data-worktree-context-menu-scope="worktree"]') + await expect(contextMenuScope).toBeVisible() + await contextMenuScope.click({ button: 'right' }) + const deleteItem = orcaPage.getByRole('menuitem', { name: /^Delete(?:\s|$)/ }) await expect(deleteItem).toBeVisible() await expect(deleteItem).toBeInViewport() await pauseForVisualProof(orcaPage) From 817827be5b98d593184d62b9561e8b17acf87498 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:06:45 -0700 Subject: [PATCH 105/398] perf(renderer): drop react-markdown and the emoji catalog off the boot path (#18149) The sidebar pulled react-markdown, remark/rehype and DOMPurify onto the eager module graph through two static importers -- WorktreeCardMeta's hover-card notes and DashboardAgentRowMessage's inline agent preview -- and built a 3,979-key emoji shortcode catalog at module scope in both the renderer and the main process. Neither is needed before first paint. Route both markdown surfaces through one shared lazyWithRetry boundary that preloads on pointer-enter (250ms hover open delay) and on agent-row mount, with a same-box raw-text Suspense fallback so a pre-load paint cannot shift layout. Memoize the emoji catalog behind loadCatalog() so import costs nothing. Eager renderer JS: 5,569,446 B / 331 chunks -> 5,198,787 B / 325 chunks (-370,659 B, -6.7%). Emoji catalog module eval: ~19.6 ms median -> 0 ms, paid once on renderer boot and once on main boot. --- .../dashboard/DashboardAgentRowMessage.tsx | 11 +- .../components/sidebar/WorktreeCardMeta.tsx | 18 ++- .../sidebar/comment-markdown-lazy.tsx | 49 +++++++++ .../worktree-card-markdown-isolation.test.ts | 103 ++++++++++++++++++ .../workspace-emoji-shortcodes.lazy.test.ts | 30 +++++ .../src/lib/workspace-emoji-shortcodes.ts | 40 ++++--- .../emoji-shortcode-catalog.lazy.test.ts | 44 ++++++++ src/shared/emoji-shortcode-catalog.ts | 63 ++++++++--- 8 files changed, 320 insertions(+), 38 deletions(-) create mode 100644 src/renderer/src/components/sidebar/comment-markdown-lazy.tsx create mode 100644 src/renderer/src/components/sidebar/worktree-card-markdown-isolation.test.ts create mode 100644 src/renderer/src/lib/workspace-emoji-shortcodes.lazy.test.ts create mode 100644 src/shared/emoji-shortcode-catalog.lazy.test.ts diff --git a/src/renderer/src/components/dashboard/DashboardAgentRowMessage.tsx b/src/renderer/src/components/dashboard/DashboardAgentRowMessage.tsx index 6b107086523..74d36300436 100644 --- a/src/renderer/src/components/dashboard/DashboardAgentRowMessage.tsx +++ b/src/renderer/src/components/dashboard/DashboardAgentRowMessage.tsx @@ -1,5 +1,9 @@ +import { useEffect } from 'react' import { cn } from '@/lib/utils' -import CommentMarkdown from '@/components/sidebar/CommentMarkdown' +import { + CommentMarkdownAsync, + preloadCommentMarkdown +} from '@/components/sidebar/comment-markdown-lazy' import { translate } from '@/i18n/i18n' type DashboardAgentRowMessageProps = { @@ -13,6 +17,9 @@ export function DashboardAgentRowMessage({ isInterrupted, lastAssistantMessage }: DashboardAgentRowMessageProps): React.JSX.Element | null { + // These rows are the sidebar's only boot-visible markdown, so warm the chunk as + // soon as one mounts rather than waiting for text to arrive. + useEffect(preloadCommentMarkdown, []) // Why: message slot is always reserved in collapsed view so the row height // stays fixed as assistant text arrives or clears. if (!isInterrupted && !lastAssistantMessage) { @@ -38,7 +45,7 @@ export function DashboardAgentRowMessage({ ) : null} {lastAssistantMessage ? ( - - {children} + + {children} + - diff --git a/src/renderer/src/components/sidebar/comment-markdown-lazy.tsx b/src/renderer/src/components/sidebar/comment-markdown-lazy.tsx new file mode 100644 index 00000000000..81de8362505 --- /dev/null +++ b/src/renderer/src/components/sidebar/comment-markdown-lazy.tsx @@ -0,0 +1,49 @@ +import React from 'react' +import { cn } from '@/lib/utils' +import { lazyWithRetry } from '@/lib/lazy-with-retry' + +// Boot-path split: react-markdown + remark/rehype/DOMPurify is ~356 KB of JS that +// only the sidebar's two markdown surfaces pull onto the eager graph. One lazy() +// identity for both, so they share a component type and a single chunk fetch. +const LazyCommentMarkdown = lazyWithRetry(() => import('./CommentMarkdown'), { + reloadKey: 'comment-markdown' +}) + +/** Warms the chunk ahead of render so the fallback is never actually shown. */ +export function preloadCommentMarkdown(): void { + void import('./CommentMarkdown') +} + +type CommentMarkdownAsyncProps = React.ComponentProps & { + /** Extra classes for the pre-load fallback only, e.g. to mirror remark-breaks. */ + fallbackClassName?: string +} + +/** + * Renders the markdown body, falling back to the raw text in an identically + * classed box while the chunk loads — same width and wrapping constraints, so a + * paint before the chunk lands cannot shift layout. + */ +export function CommentMarkdownAsync({ + fallbackClassName, + ...props +}: CommentMarkdownAsyncProps): React.JSX.Element { + return ( + + {props.content} +
    + } + > + + + ) +} diff --git a/src/renderer/src/components/sidebar/worktree-card-markdown-isolation.test.ts b/src/renderer/src/components/sidebar/worktree-card-markdown-isolation.test.ts new file mode 100644 index 00000000000..28925fc89dc --- /dev/null +++ b/src/renderer/src/components/sidebar/worktree-card-markdown-isolation.test.ts @@ -0,0 +1,103 @@ +import { readFileSync, existsSync, statSync } from 'node:fs' +import { dirname, join, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' + +const rendererSrc = join(__dirname, '../..') +const entry = join(rendererSrc, 'main.tsx') +const COMMENT_MARKDOWN = join(rendererSrc, 'components/sidebar/CommentMarkdown.tsx') + +function source(relativePath: string): string { + return readFileSync(join(rendererSrc, relativePath), 'utf8') +} + +const MODULE_EXTENSIONS = ['.ts', '.tsx', '.js', '.jsx'] + +function resolveImport(specifier: string, fromFile: string): string | null { + const base = specifier.startsWith('@/') + ? join(rendererSrc, specifier.slice(2)) + : specifier.startsWith('.') + ? resolve(dirname(fromFile), specifier) + : null + if (base === null) { + return null + } + for (const extension of ['', ...MODULE_EXTENSIONS]) { + const candidate = base + extension + if (existsSync(candidate) && statSync(candidate).isFile()) { + return candidate + } + } + for (const extension of MODULE_EXTENSIONS) { + const candidate = join(base, `index${extension}`) + if (existsSync(candidate) && statSync(candidate).isFile()) { + return candidate + } + } + return null +} + +// Static `from '...'` edges only; `import('...')` and `import type` do not ship +// code onto the eager graph. +const STATIC_IMPORT = + /(?:^|[\n;])\s*(?:import|export)(?:(?!\bfrom\b)[\s\S])*?\bfrom\s*['"]([^'"]+)['"]/g + +/** Walks the renderer entry's static import graph, recording how each module was reached. */ +function eagerModuleGraph(): Map { + const parents = new Map([[entry, null]]) + const queue = [entry] + while (queue.length > 0) { + const current = queue.shift() as string + const contents = readFileSync(current, 'utf8') + for (const match of contents.matchAll(STATIC_IMPORT)) { + if (/^\s*(?:import|export)\s+type\b/.test(match[0].replace(/^[\n;]/, ''))) { + continue + } + const resolved = resolveImport(match[1], current) + if (resolved === null || parents.has(resolved)) { + continue + } + parents.set(resolved, current) + queue.push(resolved) + } + } + return parents +} + +function importChain(parents: Map, module: string): string[] { + const chain: string[] = [] + let cursor: string | null | undefined = module + while (cursor) { + chain.push(cursor.slice(rendererSrc.length + 1)) + cursor = parents.get(cursor) + } + return chain.toReversed() +} + +describe('worktree card markdown performance isolation', () => { + it('keeps CommentMarkdown off the renderer boot graph entirely', () => { + const parents = eagerModuleGraph() + + // Names the offending chain when this regresses, instead of a bare boolean. + const chain = parents.has(COMMENT_MARKDOWN) ? importChain(parents, COMMENT_MARKDOWN) : [] + expect(chain).toEqual([]) + expect(parents.size).toBeGreaterThan(1000) + }) + + it('routes both sidebar markdown surfaces through the shared lazy boundary', () => { + const lazyBoundary = source('components/sidebar/comment-markdown-lazy.tsx') + expect(lazyBoundary).toContain("import('./CommentMarkdown')") + // A fallback in the same box keeps first paint from shifting layout. + expect(lazyBoundary).toContain('React.Suspense') + + for (const file of [ + 'components/sidebar/WorktreeCardMeta.tsx', + 'components/dashboard/DashboardAgentRowMessage.tsx' + ]) { + const contents = source(file) + expect(contents).not.toMatch(/^import CommentMarkdown from/m) + expect(contents).toContain('CommentMarkdownAsync') + // The chunk must be warmed before the surface renders, not on demand. + expect(contents).toContain('preloadCommentMarkdown') + } + }) +}) diff --git a/src/renderer/src/lib/workspace-emoji-shortcodes.lazy.test.ts b/src/renderer/src/lib/workspace-emoji-shortcodes.lazy.test.ts new file mode 100644 index 00000000000..eafc3e3b2b8 --- /dev/null +++ b/src/renderer/src/lib/workspace-emoji-shortcodes.lazy.test.ts @@ -0,0 +1,30 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { beforeEach, describe, expect, it, vi } from 'vitest' + +describe('workspace emoji shortcode index laziness', () => { + beforeEach(() => { + vi.resetModules() + }) + + it('does not build the shared catalog when the renderer index is imported', async () => { + const shortcodeIndex = await import('./workspace-emoji-shortcodes') + const catalog = await import('../../../shared/emoji-shortcode-catalog') + + expect(catalog.isEmojiShortcodeCatalogBuiltForTest()).toBe(false) + + // Cursor/regex-only paths must stay off the catalog too. + expect(shortcodeIndex.getActiveWorkspaceEmojiShortcode('hi :tad', 7)).not.toBeNull() + expect(catalog.isEmojiShortcodeCatalogBuiltForTest()).toBe(false) + + expect(shortcodeIndex.searchWorkspaceEmojiShortcodes('tada')[0]?.emoji).toBe('🎉') + expect(catalog.isEmojiShortcodeCatalogBuiltForTest()).toBe(true) + }) + + it('keeps the exact-shortcode index out of module scope', () => { + const indexSource = readFileSync(join(__dirname, 'workspace-emoji-shortcodes.ts'), 'utf8') + + expect(indexSource).not.toMatch(/^const \w+ = new Map\(/m) + expect(indexSource).not.toContain('STANDARD_EMOJI_SHORTCODE_ENTRIES') + }) +}) diff --git a/src/renderer/src/lib/workspace-emoji-shortcodes.ts b/src/renderer/src/lib/workspace-emoji-shortcodes.ts index 29e06dc74c9..a3d65c18907 100644 --- a/src/renderer/src/lib/workspace-emoji-shortcodes.ts +++ b/src/renderer/src/lib/workspace-emoji-shortcodes.ts @@ -1,5 +1,5 @@ import { - STANDARD_EMOJI_SHORTCODE_ENTRIES, + getStandardEmojiShortcodeEntries, type StandardEmojiShortcodeEntry } from '../../../shared/emoji-shortcode-catalog' @@ -16,9 +16,19 @@ export type WorkspaceEmojiReplacement = { value: string } -const EXACT_SHORTCODE = new Map( - STANDARD_EMOJI_SHORTCODE_ENTRIES.map(({ emoji, shortcode }) => [shortcode, { emoji, shortcode }]) -) +// Lazy for the same reason as the shared catalog it indexes: nothing needs it +// until a `:` shortcode is completed. +let exactShortcode: ReadonlyMap | null = null + +function exactShortcodeIndex(): ReadonlyMap { + exactShortcode ??= new Map( + getStandardEmojiShortcodeEntries().map(({ emoji, shortcode }) => [ + shortcode, + { emoji, shortcode } + ]) + ) + return exactShortcode +} // Lower tiers rank first, so `korea` surfaces `south_korea` above `dishwasher`-style incidental hits. const MATCH_TIER = { exact: 0, prefix: 1, wordStart: 2, substring: 3 } as const @@ -43,15 +53,17 @@ export function searchWorkspaceEmojiShortcodes( return [] } - const matches = STANDARD_EMOJI_SHORTCODE_ENTRIES.flatMap((entry) => { - const tier = matchTier(entry.shortcode, normalizedQuery) - return tier === null ? [] : [{ ...entry, tier }] - }).sort( - (left, right) => - left.tier - right.tier || - left.shortcode.length - right.shortcode.length || - left.shortcode.localeCompare(right.shortcode) - ) + const matches = getStandardEmojiShortcodeEntries() + .flatMap((entry) => { + const tier = matchTier(entry.shortcode, normalizedQuery) + return tier === null ? [] : [{ ...entry, tier }] + }) + .sort( + (left, right) => + left.tier - right.tier || + left.shortcode.length - right.shortcode.length || + left.shortcode.localeCompare(right.shortcode) + ) const seenEmoji = new Set() const suggestions: WorkspaceEmojiSuggestion[] = [] for (const { emoji, shortcode } of matches) { @@ -96,7 +108,7 @@ export function replaceCompletedWorkspaceEmojiShortcode( if (!match) { return null } - const suggestion = EXACT_SHORTCODE.get(match[2].toLowerCase()) + const suggestion = exactShortcodeIndex().get(match[2].toLowerCase()) if (!suggestion) { return null } diff --git a/src/shared/emoji-shortcode-catalog.lazy.test.ts b/src/shared/emoji-shortcode-catalog.lazy.test.ts new file mode 100644 index 00000000000..e838eef765e --- /dev/null +++ b/src/shared/emoji-shortcode-catalog.lazy.test.ts @@ -0,0 +1,44 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { beforeEach, describe, expect, it, vi } from 'vitest' + +describe('emoji shortcode catalog laziness', () => { + beforeEach(() => { + vi.resetModules() + }) + + it('does not build the catalog when the shared module is imported', async () => { + const catalog = await import('./emoji-shortcode-catalog.js') + + expect(catalog.isEmojiShortcodeCatalogBuiltForTest()).toBe(false) + + expect(catalog.getStandardEmojiShortcodeEntries().length).toBeGreaterThan(1000) + expect(catalog.isEmojiShortcodeCatalogBuiltForTest()).toBe(true) + }) + + it('builds on first use and keeps the main process off the eager path', async () => { + const catalog = await import('./emoji-shortcode-catalog.js') + + expect(catalog.replaceKnownEmojiWithShortcodes('ship \u{1F389}')).toBe('ship party ') + expect(catalog.isEmojiShortcodeCatalogBuiltForTest()).toBe(true) + }) + + it('leaves the main-process worktree namer importing only the deferred entry point', () => { + // A cross-project import would drag src/main into the shared tsconfig, so assert on source. + const worktreeLogic = readFileSync(join(__dirname, '../main/ipc/worktree-logic.ts'), 'utf8') + const catalogImport = worktreeLogic.match( + /import \{([^}]*)\} from '[^']*emoji-shortcode-catalog'/ + ) + + expect(catalogImport?.[1].trim()).toBe('replaceKnownEmojiWithShortcodes') + }) + + it('keeps the catalog build out of module scope', () => { + const sharedSource = readFileSync(join(__dirname, 'emoji-shortcode-catalog.ts'), 'utf8') + + // A module-scope `const X = ` is the regression this guards. + expect(sharedSource).not.toMatch(/^const \w+ = Object\.entries\(/m) + expect(sharedSource).not.toMatch(/^const \w+ = new (?:Map|Intl\.Segmenter)\(/m) + expect(sharedSource).toContain('function loadCatalog()') + }) +}) diff --git a/src/shared/emoji-shortcode-catalog.ts b/src/shared/emoji-shortcode-catalog.ts index 63477e2069a..b583a0c617a 100644 --- a/src/shared/emoji-shortcode-catalog.ts +++ b/src/shared/emoji-shortcode-catalog.ts @@ -8,22 +8,50 @@ export type StandardEmojiShortcodeEntry = { // Skin-tone aliases (`wave_tone3`) are ~40% of the dataset and would drown the suggestion list. const SKIN_TONE_SHORTCODE = /_tone\d(?:-\d)?$/ -const CATALOG = Object.entries(emojiShortcodes).flatMap(([hexcode, value]) => { - const shortcodes = (typeof value === 'string' ? [value] : value).filter( - (shortcode) => !SKIN_TONE_SHORTCODE.test(shortcode) - ) - return shortcodes.length > 0 ? [{ emoji: hexcodeToEmoji(hexcode), shortcodes }] : [] -}) +type EmojiShortcodeCatalog = { + entries: readonly StandardEmojiShortcodeEntry[] + primaryShortcodeByEmoji: ReadonlyMap + segmenter: Intl.Segmenter +} -export const STANDARD_EMOJI_SHORTCODE_ENTRIES: readonly StandardEmojiShortcodeEntry[] = - CATALOG.flatMap(({ emoji, shortcodes }) => shortcodes.map((shortcode) => ({ emoji, shortcode }))) +let catalog: EmojiShortcodeCatalog | null = null -const PRIMARY_SHORTCODE_BY_EMOJI = new Map( - CATALOG.map(({ emoji, shortcodes }) => [ - normalizeEmojiLookup(emoji), - primaryShortcode(shortcodes) - ]) -) +// Why lazy: this walks ~3,900 shortcodes and is only needed once a `:` is typed +// or a worktree name is sanitized, but at module scope every renderer and main +// boot paid for it. Memoized so the first caller builds it exactly once. +function loadCatalog(): EmojiShortcodeCatalog { + if (catalog) { + return catalog + } + const grouped = Object.entries(emojiShortcodes).flatMap(([hexcode, value]) => { + const shortcodes = (typeof value === 'string' ? [value] : value).filter( + (shortcode) => !SKIN_TONE_SHORTCODE.test(shortcode) + ) + return shortcodes.length > 0 ? [{ emoji: hexcodeToEmoji(hexcode), shortcodes }] : [] + }) + catalog = { + entries: grouped.flatMap(({ emoji, shortcodes }) => + shortcodes.map((shortcode) => ({ emoji, shortcode })) + ), + primaryShortcodeByEmoji: new Map( + grouped.map(({ emoji, shortcodes }) => [ + normalizeEmojiLookup(emoji), + primaryShortcode(shortcodes) + ]) + ), + segmenter: new Intl.Segmenter('en', { granularity: 'grapheme' }) + } + return catalog +} + +export function getStandardEmojiShortcodeEntries(): readonly StandardEmojiShortcodeEntry[] { + return loadCatalog().entries +} + +/** Test-only probe for the lazy-boundary guard; never branch on this in product code. */ +export function isEmojiShortcodeCatalogBuiltForTest(): boolean { + return catalog !== null +} /** * Pick the alias that reads best as a branch or directory name: skip `+1`/`-1` so the name @@ -40,11 +68,10 @@ function primaryShortcode(shortcodes: readonly string[]): string { ) } -const EMOJI_SEGMENTER = new Intl.Segmenter('en', { granularity: 'grapheme' }) - export function replaceKnownEmojiWithShortcodes(input: string): string { - return Array.from(EMOJI_SEGMENTER.segment(input), ({ segment }) => { - const shortcode = PRIMARY_SHORTCODE_BY_EMOJI.get(normalizeEmojiLookup(segment)) + const { primaryShortcodeByEmoji, segmenter } = loadCatalog() + return Array.from(segmenter.segment(input), ({ segment }) => { + const shortcode = primaryShortcodeByEmoji.get(normalizeEmojiLookup(segment)) return shortcode ? ` ${shortcode.replaceAll('_', '-')} ` : segment }).join('') } From 1d34d76f28414c774564ffc7158a515e1dc3a03a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:29:21 -0700 Subject: [PATCH 106/398] fix(i18n): add the three activity keys #18245 left out of en.json (#18250) verify:localization-catalog and verify:localization-extraction both exited 1 on main. The failure was masked: the Lint step failed first on max-lines, so every later static-analysis step was skipped. --- src/renderer/src/i18n/locales/en.json | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index b546a930830..1ffa6eaf9ce 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -15962,7 +15962,8 @@ "none": "None", "search": "Search", "showUnreadOnly": "Show unread only", - "showChildAgents": "Show child agents" + "showChildAgents": "Show child agents", + "activityOptions": "Activity options" }, "clearCompleted": { "clearedOne": "Cleared 1 completed agent", @@ -17259,7 +17260,9 @@ "dashboard": { "sidebar": { "label": "Agents", - "dashboardLabel": "Agent Dashboard" + "dashboardLabel": "Agent Dashboard", + "openActivity": "View activity", + "closeActivity": "Turn off activity view" } }, "runtimeRpc": { From 6d8dc0c97a2895491d301a7f3da2333922b7f2d9 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:35:37 -0700 Subject: [PATCH 107/398] perf(diff): window the combined-diff file tree rows on large reviews (#18236) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `combined-diff-file-tree.tsx` had three unvirtualized `rows.map(...)` sites, so a 900-file review mounted all 931 tree rows at once. Route the three through a `CombinedDiffFileTreeRows` wrapper over the existing `SourceControlVirtualFileList`, reusing its `SOURCE_CONTROL_VIRTUALIZE_MIN_ROWS = 50` threshold and scroll-margin machinery, with the tree's own 24px row estimate. `SourceControlVirtualFileList` gains one optional `estimateRowHeightPx` prop that defaults to its current constant, so source control is unchanged. Below the threshold the rows stay in natural flow and the markup is unchanged. Above it, find-in-page, select-all-copy and Tab order see only the mounted window — the same trade already accepted for the source-control panel. --- .../combined-diff-file-tree-row.tsx | 3 + .../combined-diff-file-tree-rows.tsx | 63 ++++++++ ...combined-diff-file-tree-windowing.test.tsx | 146 ++++++++++++++++++ .../browse-files/combined-diff-file-tree.tsx | 74 ++++----- .../listing/virtual-file-list.tsx | 8 +- 5 files changed, 248 insertions(+), 46 deletions(-) create mode 100644 src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-rows.tsx create mode 100644 src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-windowing.test.tsx diff --git a/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-row.tsx b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-row.tsx index b71aab643b5..dfb417e2518 100644 --- a/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-row.tsx +++ b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-row.tsx @@ -24,6 +24,9 @@ export type CombinedDiffTreeNode = SourceControlTreeNode< GitStagingArea | CombinedDiffBranchTreeArea > +// Why: every row is a single `py-1 text-xs` line (16px line box + 8px padding); measureElement +// still corrects, but a wrong estimate makes the virtualized tree's scrollbar jump on first paint. +export const COMBINED_DIFF_TREE_ROW_HEIGHT_PX = 24 const COMBINED_DIFF_TREE_INDENT_PX = 12 const COMBINED_DIFF_TREE_DIRECTORY_PADDING_PX = 8 const COMBINED_DIFF_TREE_FILE_PADDING_PX = 20 diff --git a/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-rows.tsx b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-rows.tsx new file mode 100644 index 00000000000..229f9561ac0 --- /dev/null +++ b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-rows.tsx @@ -0,0 +1,63 @@ +import React from 'react' +import { SourceControlVirtualFileList } from '@/components/right-sidebar/source-control/listing/virtual-file-list' +import type { + CombinedDiffFileTreeEntry, + CombinedDiffFileTreeMode +} from '../resolve-changes/combined-diff-section-identity' +import { + CombinedDiffFileTreeRow, + COMBINED_DIFF_TREE_ROW_HEIGHT_PX +} from './combined-diff-file-tree-row' +import type { CombinedDiffTreeNode } from './combined-diff-file-tree-model' + +/** + * One flattened tree section, windowed inside the file tree's scroller. A 900-file review flattens + * to over a thousand rows; below the virtualize threshold the rows stay in natural flow so small + * diffs keep byte-identical markup. + */ +export function CombinedDiffFileTreeRows({ + rows, + mode, + worktreePath, + activeSectionKey, + sectionIndexByKey, + collapsedDirectoryKeys, + visibleFileCounts, + scrollElement, + onToggleDirectory, + onNavigate +}: { + rows: readonly CombinedDiffTreeNode[] + mode: CombinedDiffFileTreeMode + worktreePath: string + activeSectionKey: string | null + sectionIndexByKey: ReadonlyMap + collapsedDirectoryKeys: ReadonlySet + visibleFileCounts: ReadonlyMap | undefined + scrollElement: HTMLDivElement | null + onToggleDirectory: (key: string) => void + onNavigate: (entry: CombinedDiffFileTreeEntry) => void +}): React.JSX.Element { + return ( + node.key} + renderRow={(node) => ( + + )} + /> + ) +} diff --git a/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-windowing.test.tsx b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-windowing.test.tsx new file mode 100644 index 00000000000..2989f0bdeb7 --- /dev/null +++ b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree-windowing.test.tsx @@ -0,0 +1,146 @@ +// @vitest-environment happy-dom + +import React, { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { SOURCE_CONTROL_VIRTUALIZE_MIN_ROWS } from '@/components/right-sidebar/source-control/listing/virtual-file-list' +import type { GitBranchChangeEntry } from '../../../../../../shared/git-diff-compare-types' +import type { CombinedDiffFileTreeRow as CombinedDiffFileTreeRowComponent } from './combined-diff-file-tree-row' + +const mountedRows = vi.hoisted(() => ({ count: 0 })) + +vi.mock('@/store', () => ({ + useAppStore: (selector: (state: Record) => unknown) => + selector({ combinedDiffFileTreeWidth: 420, setCombinedDiffFileTreeWidth: () => {} }) +})) + +vi.mock('./combined-diff-file-tree-row', async (importOriginal) => { + const actual = (await importOriginal()) as { + CombinedDiffFileTreeRow: typeof CombinedDiffFileTreeRowComponent + } + const react = await import('react') + const Row = actual.CombinedDiffFileTreeRow + const CountingRow = react.memo((props: React.ComponentProps) => { + react.useEffect(() => { + mountedRows.count += 1 + return () => { + mountedRows.count -= 1 + } + }, []) + return react.createElement(Row, props) + }) + return { ...actual, CombinedDiffFileTreeRow: CountingRow } +}) + +const { CombinedDiffFileTree } = await import('./combined-diff-file-tree') +const { createCombinedDiffSectionIndexMap } = + await import('../resolve-changes/combined-diff-section-identity') +const { getCombinedDiffBranchEntriesInTreeOrder } = await import('./combined-diff-file-tree-filter') + +const VIEWPORT_HEIGHT_PX = 600 +const TREE_ROW_HEIGHT_PX = 24 +const EMPTY_VIEWED_KEYS: ReadonlySet = new Set() + +class NoopResizeObserver implements ResizeObserver { + observe(): void {} + unobserve(): void {} + disconnect(): void {} +} + +let host: HTMLDivElement +let root: Root + +beforeEach(() => { + globalThis.IS_REACT_ACT_ENVIRONMENT = true + mountedRows.count = 0 + host = document.createElement('div') + document.body.appendChild(host) + root = createRoot(host) + vi.stubGlobal('ResizeObserver', NoopResizeObserver) + vi.spyOn(HTMLElement.prototype, 'offsetHeight', 'get').mockImplementation( + function (this: HTMLElement) { + return this.classList.contains('overflow-auto') ? VIEWPORT_HEIGHT_PX : TREE_ROW_HEIGHT_PX + } + ) + vi.spyOn(Element.prototype, 'getBoundingClientRect').mockImplementation(function (this: Element) { + const height = this.classList.contains('overflow-auto') + ? VIEWPORT_HEIGHT_PX + : TREE_ROW_HEIGHT_PX + return { + top: 0, + bottom: height, + height, + left: 0, + right: 240, + width: 240, + x: 0, + y: 0, + toJSON: () => ({}) + } as DOMRect + }) +}) + +afterEach(() => { + act(() => root.unmount()) + host.remove() + vi.unstubAllGlobals() + vi.restoreAllMocks() +}) + +/** `fileCount` files spread over `directoryCount` directories, in the viewer's own tree order. */ +function buildEntries(fileCount: number, directoryCount: number): GitBranchChangeEntry[] { + const raw: GitBranchChangeEntry[] = Array.from({ length: fileCount }, (_, index) => ({ + path: `src/dir${String(index % directoryCount).padStart(2, '0')}/file-${String(index).padStart(4, '0')}.ts`, + status: 'modified' + })) + return getCombinedDiffBranchEntriesInTreeOrder('commit', raw) +} + +function renderTree(entries: readonly GitBranchChangeEntry[]): void { + const sectionIndexByKey = createCombinedDiffSectionIndexMap( + entries.map((entry) => ({ key: `combined-commit:${entry.path}` })) + ) + act(() => { + root.render( + {}} + onNavigate={() => {}} + /> + ) + }) +} + +describe('combined diff file tree row windowing', () => { + it('mounts every row below the virtualize threshold', () => { + const fileCount = SOURCE_CONTROL_VIRTUALIZE_MIN_ROWS - 10 + const directoryCount = 4 + renderTree(buildEntries(fileCount, directoryCount)) + + // `src` plus one directory row per leaf directory, plus one row per file. + const totalRows = 1 + directoryCount + fileCount + expect(totalRows).toBeLessThan(SOURCE_CONTROL_VIRTUALIZE_MIN_ROWS) + expect(mountedRows.count).toBe(totalRows) + expect(host.querySelector('[data-testid="source-control-virtual-list"]')).toBeNull() + // Natural flow: no absolutely positioned wrappers, exactly the pre-virtualization markup. + expect(host.querySelectorAll('[data-index]').length).toBe(0) + }) + + it('mounts only a window of rows for a large review', () => { + const fileCount = 900 + const directoryCount = 30 + renderTree(buildEntries(fileCount, directoryCount)) + + const totalRows = 1 + directoryCount + fileCount + expect(host.querySelector('[data-testid="source-control-virtual-list"]')).not.toBeNull() + expect(mountedRows.count).toBeGreaterThan(0) + // A 600px viewport plus overscan: bounded by the window, not by the review size. + expect(mountedRows.count).toBeLessThan(totalRows / 10) + }) +}) diff --git a/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree.tsx b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree.tsx index dc08942f8d5..c0eb5204c73 100644 --- a/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree.tsx +++ b/src/renderer/src/components/editor/combined-diff/browse-files/combined-diff-file-tree.tsx @@ -12,7 +12,7 @@ import type { CombinedDiffFileTreeEntry, CombinedDiffFileTreeMode } from '../resolve-changes/combined-diff-section-identity' -import { CombinedDiffFileTreeRow } from './combined-diff-file-tree-row' +import { CombinedDiffFileTreeRows } from './combined-diff-file-tree-rows' import { useCombinedDiffFileTreeResize } from './use-combined-diff-file-tree-resize' import { translate } from '@/i18n/i18n' import { @@ -50,6 +50,8 @@ export function CombinedDiffFileTree({ const [query, setQuery] = React.useState('') const [excludedExtensions, setExcludedExtensions] = React.useState>(() => new Set()) const [includeViewed, setIncludeViewed] = React.useState(true) + // Why: state, not a ref — the virtualized row lists need the scroller on their own mount pass. + const [listScrollElement, setListScrollElement] = React.useState(null) const { handleResizeKeyDown, handleResizeStart, maxWidth, minWidth, treeRef, width } = useCombinedDiffFileTreeResize(collapsed) const toggleDirectory = React.useCallback((key: string) => { @@ -171,6 +173,17 @@ export function CombinedDiffFileTree({ return null } + const sharedRowProps = { + mode, + worktreePath, + activeSectionKey, + sectionIndexByKey, + collapsedDirectoryKeys, + scrollElement: listScrollElement, + onToggleDirectory: toggleDirectory, + onNavigate + } + return ( // Why: this column must be height-bounded so the file list, not the page, // owns overflow when review diffs have more files than fit on screen. @@ -283,7 +296,7 @@ export function CombinedDiffFileTree({
    -
    +
    {visibleEntryCount === 0 ? (
    {translate( @@ -309,20 +322,11 @@ export function CombinedDiffFileTree({
    {group.label}
    - {rows.map((node) => ( - - ))} +
    ) })} @@ -334,38 +338,20 @@ export function CombinedDiffFileTree({ 'Committed on Branch' )}
    - {(branchVisibleRows?.rows ?? branchRows).map((node) => ( - - ))} +
    ) : null} ) : ( - (branchVisibleRows?.rows ?? branchRows).map((node) => ( - - )) + )}
    ({ rows, getRowKey, renderRow, - scrollElement + scrollElement, + estimateRowHeightPx = SOURCE_CONTROL_FILE_ROW_HEIGHT_PX }: { rows: readonly TRow[] getRowKey: (row: TRow) => string @@ -92,6 +93,9 @@ export function SourceControlVirtualFileList({ // yet when this component's mount effects run, so a ref would leave the // virtualizer unobserved until some unrelated re-render. scrollElement: HTMLDivElement | null + // Why: callers outside source control have their own row paddings; measureElement + // still corrects, but a wrong estimate makes the initial scrollbar jump. + estimateRowHeightPx?: number }): React.JSX.Element { const containerRef = useRef(null) const [scrollMargin, setScrollMargin] = useState(0) @@ -124,7 +128,7 @@ export function SourceControlVirtualFileList({ count: rows.length, enabled: virtualize && scrollElement !== null, getScrollElement: () => scrollElement, - estimateSize: () => SOURCE_CONTROL_FILE_ROW_HEIGHT_PX, + estimateSize: () => estimateRowHeightPx, overscan: SOURCE_CONTROL_FILE_ROW_OVERSCAN, scrollMargin, // Why: stable row keys let the virtualizer carry item identity across From 02a417c04b495c2c3a2daea4a83b5fe3fa8fc688 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:35:48 -0700 Subject: [PATCH 108/398] perf(renderer): stop six timers from ticking behind a hidden window (#18134) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(renderer): stop six timers from ticking behind a hidden window IntensiveWakeUpThrottling is disabled in this app, so a renderer interval really does fire at full rate with the window hidden. Six of them had nothing to observe them: - NativeChatWorkingStatus ran a 1s interval + setState per in-flight turn purely to advance an elapsed-seconds counter. Deleted the effect and derived elapsed during render from the shared, visibility-gated useNow(1_000) clock, so N turns collapse onto one tick. - The chromium-error fallback poll (250ms) kept probing a stuck-loading guest to write a loadError nobody could see. - The contextual-tour full-pass interval (500ms) woke twice a second to queue a rAF a hidden window never paints. - Three feature-wall animation timers (3600/2400/2400ms) kept committing React renders for animations nobody was watching. - The landing preflight poll (30s) kept forcing IPC refreshes. All five gated timers reuse installWindowVisibilityInterval. Each either resumes where it left off (animations) or re-derives from durable state on the becoming-visible run, so hiding and re-showing is observationally identical to never hiding. * test(git): stop two empty commits in the divergence fixture from hashing alike `counts drift in both directions` builds 100 empty commits, resets to the fork point, then adds one more — expecting 100 ahead + 1 behind to clear the cap of 100. An empty commit's hash covers only parent, tree, message and a one-second-granularity timestamp, and every commit in the fixture reuses `commit ${index}` starting from 0. On a runner fast enough to finish the whole build inside one wall-clock second (CI: 1059ms for the case, ~7ms per commit), the post-reset `commit 0` hashed identically to the first `commit 0` of the chain, so Git handed back that same object and left the branch 99/0 apart instead of 100/1 — `within`, not `exceeded`. Numbering the empty commits across calls makes the fixture build the 101 distinct commits it already claimed to. Reproduced deterministically by pinning GIT_AUTHOR_DATE/GIT_COMMITTER_DATE, which forces the timestamp collision the fast runner hits by chance: fails with the exact CI assertion before, passes after. --- .../worktree-base-divergence-real-git.test.ts | 10 +- ...hromium-error-page-poll-visibility.test.ts | 97 +++++++++++++++ .../use-browser-page-webview-url-sync.ts | 9 +- .../ContextualTourOverlay.tsx | 11 +- .../ContextualTourOverlay.visibility.test.tsx | 106 ++++++++++++++++ .../feature-wall/WorkspacesAnimatedVisual.tsx | 30 +++-- .../agents-orchestration/StatusesPage.tsx | 28 +++-- ...feature-wall-animation-visibility.test.tsx | 115 ++++++++++++++++++ .../use-workbench-terminal-storyboard.ts | 12 +- ...landing-preflight-runtime-boundary.test.ts | 30 +++++ .../components/landing-preflight-runtime.ts | 15 ++- .../native-chat/NativeChatWorkingStatus.tsx | 26 ++-- ...-chat-working-status-shared-clock.test.tsx | 94 ++++++++++++++ 13 files changed, 534 insertions(+), 49 deletions(-) create mode 100644 src/renderer/src/components/browser-pane/navigate/chromium-error-page-poll-visibility.test.ts create mode 100644 src/renderer/src/components/contextual-tours/ContextualTourOverlay.visibility.test.tsx create mode 100644 src/renderer/src/components/feature-wall/feature-wall-animation-visibility.test.tsx create mode 100644 src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx diff --git a/src/main/git/worktree-base-divergence-real-git.test.ts b/src/main/git/worktree-base-divergence-real-git.test.ts index e17192cf914..377b63b48f9 100644 --- a/src/main/git/worktree-base-divergence-real-git.test.ts +++ b/src/main/git/worktree-base-divergence-real-git.test.ts @@ -32,9 +32,17 @@ async function createRepo(): Promise { return repoPath } +// Why unique across calls: an empty commit's hash covers only parent, tree, message and a +// one-second-granularity timestamp. On a fast runner the whole 100-commit build finishes inside +// one second, so a post-reset `commit 0` off the same fork point hashed identically to the first +// `commit 0` of the chain and Git handed back that same object — leaving the branch 99/0 apart +// instead of 100/1. +let emptyCommitSequence = 0 + function commitEmpty(repoPath: string, count: number): void { for (let index = 0; index < count; index += 1) { - git(repoPath, ['commit', '--quiet', '--allow-empty', '-m', `commit ${index}`]) + emptyCommitSequence += 1 + git(repoPath, ['commit', '--quiet', '--allow-empty', '-m', `commit ${emptyCommitSequence}`]) } } diff --git a/src/renderer/src/components/browser-pane/navigate/chromium-error-page-poll-visibility.test.ts b/src/renderer/src/components/browser-pane/navigate/chromium-error-page-poll-visibility.test.ts new file mode 100644 index 00000000000..2256c4171f7 --- /dev/null +++ b/src/renderer/src/components/browser-pane/navigate/chromium-error-page-poll-visibility.test.ts @@ -0,0 +1,97 @@ +// @vitest-environment happy-dom + +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { BrowserTabPageState } from '../describe-page/browser-page-types' +import { useBrowserPageWebviewUrlSync } from './use-browser-page-webview-url-sync' + +const CHROMIUM_ERROR_URL = 'chrome-error://chromewebdata/' + +function setDocumentVisibility(state: 'visible' | 'hidden'): void { + Object.defineProperty(document, 'visibilityState', { configurable: true, get: () => state }) + act(() => { + document.dispatchEvent(new Event('visibilitychange')) + }) +} + +function renderUrlSync(guestUrl: () => string): { + updates: [string, BrowserTabPageState][] + unmount: () => void +} { + const updates: [string, BrowserTabPageState][] = [] + const webview = { getURL: () => guestUrl(), src: '' } as unknown as Electron.WebviewTag + const view = renderHook(() => + useBrowserPageWebviewUrlSync({ + browserTabId: 'tab-1', + browserTabUrl: 'https://example.test/slow', + browserTabLoading: true, + isActive: true, + isPaintable: true, + slotViewport: null, + webviewRef: { current: webview }, + chromeHeaderRef: { current: null }, + lastKnownWebviewUrlRef: { current: 'https://example.test/slow' }, + trackNextLoadingEventRef: { current: false }, + keepAddressBarFocusRef: { current: false }, + addressBarInputRef: { current: null }, + browserTabUrlRef: { current: 'https://example.test/slow' }, + addressBarValueRef: { current: 'https://example.test/slow' }, + onUpdatePageStateRef: { + current: (tabId, patch) => { + updates.push([tabId, patch]) + } + }, + focusWebviewNow: () => false + }) + ) + return { updates, unmount: view.unmount } +} + +beforeEach(() => { + vi.useFakeTimers() + setDocumentVisibility('visible') +}) + +afterEach(() => { + cleanup() + setDocumentVisibility('visible') + vi.useRealTimers() +}) + +describe('chromium error page poll visibility gate', () => { + it('stops the 250ms poll while hidden and re-detects the error page on return', () => { + let guestUrl = 'https://example.test/slow' + const { updates, unmount } = renderUrlSync(() => guestUrl) + + // Visible and still loading: the fallback poll is armed. + expect(vi.getTimerCount()).toBe(1) + act(() => vi.advanceTimersByTime(1_000)) + expect(updates).toHaveLength(0) + + setDocumentVisibility('hidden') + expect(vi.getTimerCount()).toBe(0) + + // The guest lands on a chrome-error page while nobody can see the surface. + guestUrl = CHROMIUM_ERROR_URL + act(() => vi.advanceTimersByTime(10_000)) + expect(updates).toHaveLength(0) + + // Returning re-reads the durable guest URL, so the loadError is not lost. + setDocumentVisibility('visible') + expect(updates).toHaveLength(1) + expect(updates[0]?.[1].loadError?.validatedUrl).toBe('https://example.test/slow') + expect(updates[0]?.[1].loading).toBe(false) + + unmount() + expect(vi.getTimerCount()).toBe(0) + }) + + it('polls unchanged while the window stays visible', () => { + let guestUrl = 'https://example.test/slow' + const { updates } = renderUrlSync(() => guestUrl) + + guestUrl = CHROMIUM_ERROR_URL + act(() => vi.advanceTimersByTime(250)) + expect(updates).toHaveLength(1) + }) +}) diff --git a/src/renderer/src/components/browser-pane/navigate/use-browser-page-webview-url-sync.ts b/src/renderer/src/components/browser-pane/navigate/use-browser-page-webview-url-sync.ts index 0c75ddefce0..bfd93702440 100644 --- a/src/renderer/src/components/browser-pane/navigate/use-browser-page-webview-url-sync.ts +++ b/src/renderer/src/components/browser-pane/navigate/use-browser-page-webview-url-sync.ts @@ -9,6 +9,7 @@ import { applyBrowserPageViewportLayout, syncBrowserPageChromeInset } from '../host-guest/browser-page-viewport' +import { installWindowVisibilityInterval } from '@/lib/window-visibility-interval' import { shouldPollChromiumErrorPage } from './chromium-error-page-polling' import { isChromiumErrorPage } from '../describe-page/browser-page-url-display' import type { BrowserTabPageState } from '../describe-page/browser-page-types' @@ -152,9 +153,11 @@ export function useBrowserPageWebviewUrlSync({ } // Why: some Electron builds paint chrome-error pages without a did-fail-load event; poll only while the active tab loads as a fallback. - detectChromiumErrorPage() - const intervalId = window.setInterval(detectChromiumErrorPage, 250) - return () => window.clearInterval(intervalId) + // Why gated: a page stuck loading would otherwise poll 4x/sec forever behind a hidden window. The guest URL is durable state, so the becoming-visible run re-derives anything a hidden window skipped. + return installWindowVisibilityInterval({ + run: detectChromiumErrorPage, + intervalMs: 250 + }) }, [ addressBarValueRef, browserTabId, diff --git a/src/renderer/src/components/contextual-tours/ContextualTourOverlay.tsx b/src/renderer/src/components/contextual-tours/ContextualTourOverlay.tsx index 243bfcc48c4..119ca73d451 100644 --- a/src/renderer/src/components/contextual-tours/ContextualTourOverlay.tsx +++ b/src/renderer/src/components/contextual-tours/ContextualTourOverlay.tsx @@ -24,6 +24,7 @@ import { handleContextualTourOverlayKeyDown, type ActiveTourRenderState } from './ContextualTourOverlaySurface' +import { installWindowVisibilityInterval } from '@/lib/window-visibility-interval' import { requestActiveTerminalPaneSplit } from '@/components/tab-bar/request-active-terminal-pane-split' import { performContextualTourStepAction } from './contextual-tour-step-actions' import { openWorkspaceCreationComposerWithTourHandoff } from './workspace-creation-tour-handoff' @@ -228,14 +229,20 @@ export function ContextualTourOverlay(): JSX.Element | null { const scheduleFullMeasure = (): void => scheduleMeasure(true) window.addEventListener('resize', scheduleFullMeasure) window.addEventListener('scroll', scheduleTargetMeasure, true) - const interval = window.setInterval(scheduleFullMeasure, 500) + // Why gated: a hidden window paints no frames, so the queued rAF never runs + // and the pass is pure wakeup. The becoming-visible run re-queues it, and + // the layout effect measures on every render, so nothing is missed. + const stopFullPassInterval = installWindowVisibilityInterval({ + run: scheduleFullMeasure, + intervalMs: 500 + }) return () => { if (frame !== null) { window.cancelAnimationFrame(frame) } window.removeEventListener('resize', scheduleFullMeasure) window.removeEventListener('scroll', scheduleTargetMeasure, true) - window.clearInterval(interval) + stopFullPassInterval() } }, [activeTourId, measureTourOverlay]) diff --git a/src/renderer/src/components/contextual-tours/ContextualTourOverlay.visibility.test.tsx b/src/renderer/src/components/contextual-tours/ContextualTourOverlay.visibility.test.tsx new file mode 100644 index 00000000000..f8b6fe0094d --- /dev/null +++ b/src/renderer/src/components/contextual-tours/ContextualTourOverlay.visibility.test.tsx @@ -0,0 +1,106 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { ContextualTourOverlay } from './ContextualTourOverlay' +import { useAppStore } from '@/store' + +let container: HTMLDivElement +let root: Root + +function tourTarget(name: string, top: number): { moveTo: (top: number) => void } { + let currentTop = top + const element = document.createElement('div') + element.setAttribute('data-contextual-tour-target', name) + Object.defineProperty(element, 'getBoundingClientRect', { + configurable: true, + value: () => ({ + left: 100, + right: 220, + top: currentTop, + bottom: currentTop + 40, + width: 120, + height: 40, + x: 100, + y: currentTop + }) + }) + document.body.appendChild(element) + return { + moveTo: (next) => { + currentTop = next + } + } +} + +async function settle(ms: number): Promise { + await act(async () => { + await new Promise((resolve) => setTimeout(resolve, ms)) + }) +} + +async function setDocumentVisibility(state: 'visible' | 'hidden'): Promise { + Object.defineProperty(document, 'visibilityState', { configurable: true, get: () => state }) + await act(async () => { + document.dispatchEvent(new Event('visibilitychange')) + await new Promise((resolve) => setTimeout(resolve, 0)) + }) +} + +function ringsTop(): string | undefined { + return container.querySelector('[data-contextual-tour-target-rings]')?.style.top +} + +beforeEach(async () => { + ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + ;(window as unknown as { api: unknown }).api = { ui: { set: () => Promise.resolve() } } + Object.defineProperty(window, 'innerWidth', { configurable: true, value: 1280 }) + Object.defineProperty(window, 'innerHeight', { configurable: true, value: 960 }) + await setDocumentVisibility('visible') + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(async () => { + act(() => root.unmount()) + container.remove() + document.querySelectorAll('[data-contextual-tour-target]').forEach((node) => node.remove()) + await setDocumentVisibility('visible') + useAppStore.setState({ activeContextualTourId: null, activeContextualTourStepIndex: 0 }) +}) + +describe('ContextualTourOverlay full-pass interval visibility gate', () => { + it('pauses the 500ms full pass while hidden and re-measures on return', async () => { + const target = tourTarget('workspace-create-control', 300) + useAppStore.setState({ + activeContextualTourId: 'workspace-agent-sessions', + activeContextualTourStepIndex: 1, + activeModal: 'none', + contextualToursOnboardingVisible: false, + contextualToursBlockingSurfaceVisible: false, + activeContextualTourSuppressed: false + }) + await act(async () => { + root.render() + await new Promise((resolve) => setTimeout(resolve, 50)) + }) + expect(ringsTop()).toBe('300px') + + // Baseline: while visible, the periodic full pass follows a silent move. + target.moveTo(640) + await settle(700) + expect(ringsTop()).toBe('640px') + + await setDocumentVisibility('hidden') + target.moveTo(900) + await settle(1_500) + expect(ringsTop()).toBe('640px') + + // Returning runs the pass immediately, so the overlay is never left stale. + await setDocumentVisibility('visible') + await settle(50) + expect(ringsTop()).toBe('900px') + }) +}) diff --git a/src/renderer/src/components/feature-wall/WorkspacesAnimatedVisual.tsx b/src/renderer/src/components/feature-wall/WorkspacesAnimatedVisual.tsx index 7240684aeb8..bc4260612fe 100644 --- a/src/renderer/src/components/feature-wall/WorkspacesAnimatedVisual.tsx +++ b/src/renderer/src/components/feature-wall/WorkspacesAnimatedVisual.tsx @@ -1,6 +1,7 @@ import { useEffect, useMemo, useState } from 'react' import type { JSX } from 'react' import { AgentStateDot } from '@/components/AgentStateDot' +import { installWindowVisibilityInterval } from '@/lib/window-visibility-interval' import { ClaudeIcon, OpenCodeGoIcon } from '../status-bar/icons' type AgentKind = 'claude' | 'codex' | 'opencode' @@ -87,18 +88,23 @@ export function WorkspacesAnimatedVisual(props: { reducedMotion: boolean }): JSX if (reducedMotion) { return } - const id = window.setInterval(() => { - setVisualState((current) => { - const next = current.order.slice() - const finishing = next.pop() - if (!finishing) { - return current - } - next.unshift(finishing) - return { order: next, promotedWorkspaceId: finishing.id } - }) - }, STEP_MS) - return () => window.clearInterval(id) + // Why: nobody watches an animation in a hidden window. `runOnVisible` is a + // no-op so revealing the window resumes the cycle instead of skipping a card. + return installWindowVisibilityInterval({ + run: () => { + setVisualState((current) => { + const next = current.order.slice() + const finishing = next.pop() + if (!finishing) { + return current + } + next.unshift(finishing) + return { order: next, promotedWorkspaceId: finishing.id } + }) + }, + runOnVisible: () => {}, + intervalMs: STEP_MS + }) }, [reducedMotion]) // Why: reduced-motion mode should display the static stack without a // post-render repair; only the animated interval needs promoted z-order. diff --git a/src/renderer/src/components/feature-wall/agents-orchestration/StatusesPage.tsx b/src/renderer/src/components/feature-wall/agents-orchestration/StatusesPage.tsx index acbfcb889ff..dd4412cf282 100644 --- a/src/renderer/src/components/feature-wall/agents-orchestration/StatusesPage.tsx +++ b/src/renderer/src/components/feature-wall/agents-orchestration/StatusesPage.tsx @@ -5,6 +5,7 @@ import { Wrench } from 'lucide-react' import { AgentStateDot } from '@/components/AgentStateDot' import { getAgentCatalog, AgentIcon, type AgentCatalogEntry } from '@/lib/agent-catalog' import { ClaudeIcon, OpenAIIcon } from '../../status-bar/icons' +import { installWindowVisibilityInterval } from '@/lib/window-visibility-interval' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' @@ -48,18 +49,25 @@ export function StatusesPage(props: { active: boolean; reducedMotion: boolean }) schedule(() => setRevealed((r) => ({ ...r, codex: true })), 1900) let idx = 0 - const cycleId = window.setInterval(() => { - setClaudeFading(true) - const swap = window.setTimeout(() => { - idx = (idx + 1) % CLAUDE_ACTIVITIES.length - setClaudeIdx(idx) - setClaudeFading(false) - }, 280) - timeouts.push(swap) - }, 2400) + // Why: nobody watches an animation in a hidden window. `runOnVisible` is a + // no-op so revealing the window resumes the cycle instead of skipping an + // activity; `idx` lives outside the timer, so the reveal picks up where it left off. + const stopCycle = installWindowVisibilityInterval({ + run: () => { + setClaudeFading(true) + const swap = window.setTimeout(() => { + idx = (idx + 1) % CLAUDE_ACTIVITIES.length + setClaudeIdx(idx) + setClaudeFading(false) + }, 280) + timeouts.push(swap) + }, + runOnVisible: () => {}, + intervalMs: 2400 + }) return () => { timeouts.forEach((id) => window.clearTimeout(id)) - window.clearInterval(cycleId) + stopCycle() } }, [active, reducedMotion]) diff --git a/src/renderer/src/components/feature-wall/feature-wall-animation-visibility.test.tsx b/src/renderer/src/components/feature-wall/feature-wall-animation-visibility.test.tsx new file mode 100644 index 00000000000..334356ef1a4 --- /dev/null +++ b/src/renderer/src/components/feature-wall/feature-wall-animation-visibility.test.tsx @@ -0,0 +1,115 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { TooltipProvider } from '@/components/ui/tooltip' +import { StatusesPage } from './agents-orchestration/StatusesPage' +import { useWorkbenchTerminalStoryboard } from './use-workbench-terminal-storyboard' +import { WorkspacesAnimatedVisual } from './WorkspacesAnimatedVisual' + +let container: HTMLDivElement +let root: Root + +function setDocumentVisibility(state: 'visible' | 'hidden'): void { + Object.defineProperty(document, 'visibilityState', { configurable: true, get: () => state }) + act(() => { + document.dispatchEvent(new Event('visibilitychange')) + }) +} + +function cardOrder(): string[] { + return Array.from(container.querySelectorAll('[data-ws-id]')) + .map((node) => ({ + id: node.dataset.wsId ?? '', + top: Number.parseFloat(node.style.transform.replace(/[^\d.-]/g, '')) || 0 + })) + .sort((left, right) => left.top - right.top) + .map((card) => card.id) +} + +beforeEach(() => { + ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + vi.useFakeTimers() + setDocumentVisibility('visible') + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() + setDocumentVisibility('visible') + vi.useRealTimers() +}) + +describe('feature wall animation timers', () => { + it('pauses the workspaces card rotation while hidden and resumes without skipping', () => { + act(() => + root.render( + + + + ) + ) + const initialOrder = cardOrder() + expect(vi.getTimerCount()).toBe(1) + + act(() => vi.advanceTimersByTime(3_600)) + const afterOneStep = cardOrder() + expect(afterOneStep).not.toEqual(initialOrder) + + setDocumentVisibility('hidden') + expect(vi.getTimerCount()).toBe(0) + act(() => vi.advanceTimersByTime(3_600 * 10)) + expect(cardOrder()).toEqual(afterOneStep) + + // Revealing resumes the cycle rather than jumping a card forward. + setDocumentVisibility('visible') + expect(cardOrder()).toEqual(afterOneStep) + expect(vi.getTimerCount()).toBe(1) + act(() => vi.advanceTimersByTime(3_600)) + expect(cardOrder()).not.toEqual(afterOneStep) + }) + + it('pauses the workbench run queue while hidden and resumes from the same entry', () => { + const view = renderHook(() => useWorkbenchTerminalStoryboard('tour', false)) + const first = view.result.current.running + act(() => vi.advanceTimersByTime(2_400)) + const second = view.result.current.running + expect(second).not.toBe(first) + + setDocumentVisibility('hidden') + act(() => vi.advanceTimersByTime(2_400 * 10)) + expect(view.result.current.running).toBe(second) + + setDocumentVisibility('visible') + expect(view.result.current.running).toBe(second) + act(() => vi.advanceTimersByTime(2_400)) + expect(view.result.current.running).not.toBe(second) + view.unmount() + }) + + it('pauses the agent-status activity cycle while hidden and resumes in place', () => { + act(() => + root.render( + + + + ) + ) + act(() => vi.advanceTimersByTime(2_400 + 280)) + const afterOneCycle = container.textContent ?? '' + + setDocumentVisibility('hidden') + act(() => vi.advanceTimersByTime((2_400 + 280) * 10)) + expect(container.textContent).toBe(afterOneCycle) + + setDocumentVisibility('visible') + expect(container.textContent).toBe(afterOneCycle) + act(() => vi.advanceTimersByTime(2_400 + 280)) + expect(container.textContent).not.toBe(afterOneCycle) + }) +}) diff --git a/src/renderer/src/components/feature-wall/use-workbench-terminal-storyboard.ts b/src/renderer/src/components/feature-wall/use-workbench-terminal-storyboard.ts index 919d3928500..4ee95a9d428 100644 --- a/src/renderer/src/components/feature-wall/use-workbench-terminal-storyboard.ts +++ b/src/renderer/src/components/feature-wall/use-workbench-terminal-storyboard.ts @@ -1,4 +1,5 @@ import { useEffect, useState } from 'react' +import { installWindowVisibilityInterval } from '@/lib/window-visibility-interval' import { WORKBENCH_RUN_QUEUE, WORKBENCH_RUN_TICK_MS, @@ -38,10 +39,13 @@ export function useWorkbenchTerminalStoryboard( if (reducedMotion || isTwoAgentsChecklist) { return } - const id = window.setInterval(() => { - setRunIdx((index) => (index + 1) % WORKBENCH_RUN_QUEUE.length) - }, WORKBENCH_RUN_TICK_MS) - return () => window.clearInterval(id) + // Why: nobody watches an animation in a hidden window. `runOnVisible` is a + // no-op so revealing the window resumes the queue instead of skipping an entry. + return installWindowVisibilityInterval({ + run: () => setRunIdx((index) => (index + 1) % WORKBENCH_RUN_QUEUE.length), + runOnVisible: () => {}, + intervalMs: WORKBENCH_RUN_TICK_MS + }) }, [isTwoAgentsChecklist, reducedMotion]) useEffect(() => { diff --git a/src/renderer/src/components/landing-preflight-runtime-boundary.test.ts b/src/renderer/src/components/landing-preflight-runtime-boundary.test.ts index f282ca5f15a..f6426c9e3cf 100644 --- a/src/renderer/src/components/landing-preflight-runtime-boundary.test.ts +++ b/src/renderer/src/components/landing-preflight-runtime-boundary.test.ts @@ -15,6 +15,11 @@ const status = (overrides: Partial = {}): PreflightStatus => ({ ...overrides }) +function setDocumentVisibility(state: 'visible' | 'hidden'): void { + Object.defineProperty(document, 'visibilityState', { configurable: true, get: () => state }) + document.dispatchEvent(new Event('visibilitychange')) +} + const githubRepo: Repo = { id: 'github', path: '/repos/github', @@ -27,6 +32,7 @@ const githubRepo: Repo = { beforeEach(() => { vi.useFakeTimers() + setDocumentVisibility('visible') refresh.mockClear() invalidate.mockClear() useAppStore.setState(useAppStore.getInitialState(), true) @@ -35,6 +41,7 @@ beforeEach(() => { afterEach(() => { cleanup() + setDocumentVisibility('visible') vi.useRealTimers() useAppStore.setState(useAppStore.getInitialState(), true) }) @@ -108,6 +115,29 @@ describe('landing preflight runtime boundary', () => { view.unmount() }) + it('stops the 30s preflight poll while hidden and lets the reveal refresh instead', () => { + useAppStore.setState({ repos: [githubRepo], preflightStatus: status() }) + const view = renderHook(() => useLandingPreflightRuntime()) + refresh.mockClear() + + expect(vi.getTimerCount()).toBe(1) + act(() => setDocumentVisibility('hidden')) + expect(vi.getTimerCount()).toBe(0) + + // Five poll windows pass behind a hidden window with no IPC at all. + act(() => vi.advanceTimersByTime(150_000)) + expect(refresh).not.toHaveBeenCalled() + + // The sibling visibilitychange handler force-refreshes on reveal, so the + // banner is current the moment it can be seen; the poll re-arms behind it. + act(() => setDocumentVisibility('visible')) + expect(refresh).toHaveBeenCalledTimes(1) + expect(refresh).toHaveBeenCalledWith({ force: true }) + expect(vi.getTimerCount()).toBe(1) + + view.unmount() + }) + it('keeps one active interval and removes listeners and polling on cleanup', () => { useAppStore.setState({ repos: [githubRepo], preflightStatus: status() }) const addEventListener = vi.spyOn(document, 'addEventListener') diff --git a/src/renderer/src/components/landing-preflight-runtime.ts b/src/renderer/src/components/landing-preflight-runtime.ts index 57cc9d1c31d..6df10613668 100644 --- a/src/renderer/src/components/landing-preflight-runtime.ts +++ b/src/renderer/src/components/landing-preflight-runtime.ts @@ -1,5 +1,6 @@ import { useEffect, useMemo } from 'react' import { useAppStore } from '../store' +import { installWindowVisibilityInterval } from '@/lib/window-visibility-interval' import { getLandingPreflightIssues, hasGitHubBackedProject, @@ -60,10 +61,16 @@ export function useLandingPreflightRuntime(): { preflightIssues: PreflightIssue[ if (preflightIssues.length === 0) { return } - const intervalId = window.setInterval(() => { - void refreshPreflightStatus({ force: true }) - }, 30000) - return () => window.clearInterval(intervalId) + // Why gated: the effect above already force-refreshes on visibilitychange + // and focus, so a revealed window has fresh data without this poll firing + // while hidden — hence the no-op `runOnVisible`. + return installWindowVisibilityInterval({ + run: () => { + void refreshPreflightStatus({ force: true }) + }, + runOnVisible: () => {}, + intervalMs: 30000 + }) }, [preflightIssues.length, refreshPreflightStatus]) return { preflightIssues } diff --git a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx index 5aef81ba51d..a883429f658 100644 --- a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx @@ -1,6 +1,7 @@ -import { useEffect, useState } from 'react' +import { useState } from 'react' import { ChevronRight } from 'lucide-react' import { translate } from '@/i18n/i18n' +import { useNow } from '@/hooks/use-now' export function NativeChatWorkingStatus({ startedAt, @@ -15,18 +16,17 @@ export function NativeChatWorkingStatus({ expanded?: boolean onToggleExpanded?: () => void }): React.JSX.Element { - const [elapsedSeconds, setElapsedSeconds] = useState(0) - - useEffect(() => { - if (thinking || workedSeconds != null) { - return - } - const epoch = startedAt ?? Date.now() - setElapsedSeconds(Math.max(0, Math.floor((Date.now() - epoch) / 1000))) - const update = () => setElapsedSeconds(Math.max(0, Math.floor((Date.now() - epoch) / 1000))) - const timer = window.setInterval(update, 1000) - return () => window.clearInterval(timer) - }, [startedAt, thinking, workedSeconds]) + // Why: elapsed seconds is ordinary render dataflow, not an external system. + // The shared 1s clock is visibility-gated and collapses every in-flight turn + // onto one tick, instead of one interval plus one commit per turn. + const counting = !thinking && workedSeconds == null + const now = useNow(1_000, counting) + // Why: preserves the old effect's `startedAt ?? Date.now()` epoch for the + // single frame before the turn's startedAt lands. + const [mountedAt] = useState(() => Date.now()) + const elapsedSeconds = counting + ? Math.max(0, Math.floor((now - (startedAt ?? mountedAt)) / 1000)) + : 0 const label = workedSeconds != null diff --git a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx new file mode 100644 index 00000000000..dfea16ae0e4 --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx @@ -0,0 +1,94 @@ +// @vitest-environment happy-dom + +import { act } from 'react' +import { createRoot, type Root } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' + +let container: HTMLDivElement +let root: Root + +function setDocumentVisibility(state: 'visible' | 'hidden'): void { + Object.defineProperty(document, 'visibilityState', { configurable: true, get: () => state }) + act(() => { + document.dispatchEvent(new Event('visibilitychange')) + }) +} + +function renderTurns(count: number, startedAt: number): void { + act(() => { + root.render( + <> + {Array.from({ length: count }, (_, index) => ( + + ))} + + ) + }) +} + +function elapsedLabels(): string[] { + return Array.from(container.querySelectorAll('[aria-label]'), (node) => node.textContent ?? '') +} + +beforeEach(() => { + ;(globalThis as { IS_REACT_ACT_ENVIRONMENT?: boolean }).IS_REACT_ACT_ENVIRONMENT = true + vi.useFakeTimers() + vi.setSystemTime(1_000_000) + setDocumentVisibility('visible') + container = document.createElement('div') + document.body.appendChild(container) + root = createRoot(container) +}) + +afterEach(() => { + act(() => root.unmount()) + container.remove() + setDocumentVisibility('visible') + vi.useRealTimers() +}) + +describe('native chat working status elapsed clock', () => { + it('collapses every in-flight turn onto one shared visibility-gated timer', () => { + renderTurns(3, 1_000_000) + + // One shared 1s clock for all three turns, not one interval per turn. + expect(vi.getTimerCount()).toBe(1) + act(() => vi.advanceTimersByTime(3_000)) + expect(elapsedLabels()).toEqual([ + 'Working for 3 seconds', + 'Working for 3 seconds', + 'Working for 3 seconds' + ]) + }) + + it('stops ticking while hidden and re-syncs the elapsed value on return', () => { + renderTurns(1, 1_000_000) + act(() => vi.advanceTimersByTime(3_000)) + expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + + setDocumentVisibility('hidden') + expect(vi.getTimerCount()).toBe(0) + + // A minute of hidden wall-clock: no callbacks, no commits, label frozen. + act(() => vi.advanceTimersByTime(60_000)) + expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + + // Returning re-derives elapsed from startedAt, so nothing was lost. + setDocumentVisibility('visible') + expect(elapsedLabels()).toEqual(['Working for 63 seconds']) + expect(vi.getTimerCount()).toBe(1) + }) + + it('holds no timer for a thinking turn or a completed turn', () => { + act(() => { + root.render( + <> + + + + ) + }) + expect(vi.getTimerCount()).toBe(0) + }) +}) From 437615499337cd963180923ac2d56c7d89e06505 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 14:46:52 -0700 Subject: [PATCH 109/398] perf(sidebar): stop re-allocating 423 workspace descriptors and 193 bucket projections per recompute (#18241) Layers three identity memos onto the row cache #18222 landed, without changing what the four sidebar numbers say in any state. - The active-workspace descriptor list is memoized on the four slices `collectActiveDashboardWorkspaces(state, false)` actually reads. - Each worktree's bucket tally is memoized on its rows plus the acknowledgement map, so a ping that rebuilds one worktree no longer re-projects the board. - The `useShallow` selector becomes a module-level 14-identity gate, which allocates nothing on the unchanged path. - The counts object is reused by identity when all four totals hold. --- ...dashboard-bucket-counts.allocation.test.ts | 250 ++++++++++++ ...ashboard-bucket-counts.equivalence.test.ts | 371 ++++++++++++++++++ .../build-dashboard-bucket-counts.ts | 169 +++++++- .../dashboard-snapshot-workspaces.ts | 7 +- .../useAgentBucketCounts.gate.test.ts | 158 ++++++++ .../dashboard/useAgentBucketCounts.test.tsx | 3 +- .../dashboard/useAgentBucketCounts.ts | 164 ++++---- 7 files changed, 1023 insertions(+), 99 deletions(-) create mode 100644 src/renderer/src/components/dashboard/build-dashboard-bucket-counts.allocation.test.ts create mode 100644 src/renderer/src/components/dashboard/build-dashboard-bucket-counts.equivalence.test.ts create mode 100644 src/renderer/src/components/dashboard/useAgentBucketCounts.gate.test.ts diff --git a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.allocation.test.ts b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.allocation.test.ts new file mode 100644 index 00000000000..9145a397854 --- /dev/null +++ b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.allocation.test.ts @@ -0,0 +1,250 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { AgentStatusEntry } from '../../../../shared/agent-status-types' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import type * as DashboardSnapshotWorkspaces from './dashboard-snapshot-workspaces' +import type * as DashboardRowBucket from './dashboard-row-bucket' +import type { ActiveDashboardWorkspace } from './dashboard-snapshot-workspaces' + +const collected = vi.hoisted(() => ({ + calls: 0, + descriptors: 0, + projections: 0 +})) + +vi.mock('./dashboard-snapshot-workspaces', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + collectActiveDashboardWorkspaces: ( + ...args: Parameters + ): ActiveDashboardWorkspace[] => { + const workspaces = actual.collectActiveDashboardWorkspaces(...args) + collected.calls += 1 + collected.descriptors += workspaces.length + return workspaces + } + } +}) + +vi.mock('./dashboard-row-bucket', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + dashboardRowBucketProjection: ( + ...args: Parameters + ) => { + collected.projections += 1 + return actual.dashboardRowBucketProjection(...args) + } + } +}) + +import type { DashboardSnapshotState as SnapshotState } from './build-dashboard-snapshot' +import { + buildDashboardBucketCounts, + createDashboardBucketCountsCache +} from './build-dashboard-bucket-counts' + +const NOW = 1_700_000_000_000 +const WORKSPACE_COUNT = 400 + +function leafId(index: number): string { + return `${String(index).padStart(8, '0')}-1111-4111-8111-111111111111` +} + +function worktree(index: number): Worktree { + return { + id: `w${index}`, + repoId: 'r1', + path: `/r1/w${index}`, + head: 'abc123', + branch: 'main', + isBare: false, + isMainWorktree: false, + displayName: `w${index}`, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: index, + lastActivityAt: NOW + } +} + +function tab(index: number): TerminalTab { + return { + id: `tab${index}`, + ptyId: `pty-tab${index}`, + worktreeId: `w${index}`, + title: 'shell', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: NOW + } +} + +function entry(index: number, prompt: string): AgentStatusEntry { + return { + paneKey: makePaneKey(`tab${index}`, leafId(index)), + state: 'working', + prompt, + updatedAt: NOW, + stateStartedAt: NOW - 5_000, + stateHistory: [], + agentType: 'claude', + tabId: `tab${index}`, + worktreeId: `w${index}` + } +} + +function largeState(): SnapshotState { + const worktrees: Worktree[] = [] + const tabsByWorktree: Record = {} + const agentStatusByPaneKey: Record = {} + const terminalLayoutsByTabId: Record = {} + const ptyIdsByTabId: Record = {} + for (let index = 0; index < WORKSPACE_COUNT; index += 1) { + worktrees.push(worktree(index)) + tabsByWorktree[`w${index}`] = [tab(index)] + agentStatusByPaneKey[makePaneKey(`tab${index}`, leafId(index))] = entry(index, 'do the thing') + terminalLayoutsByTabId[`tab${index}`] = { + root: { type: 'leaf', leafId: leafId(index) }, + activeLeafId: leafId(index), + expandedLeafId: null, + ptyIdsByLeafId: { [leafId(index)]: `pty-tab${index}` } + } + ptyIdsByTabId[`tab${index}`] = [`pty-tab${index}`] + } + return { + repos: [ + { + id: 'r1', + path: '/r1', + displayName: 'Repo One', + badgeColor: '#000', + addedAt: 1 + } + ], + worktreesByRepo: { r1: worktrees }, + folderWorkspaces: [], + projectGroups: [], + tabsByWorktree, + agentStatusByPaneKey, + retainedAgentsByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + runtimeAgentOrchestrationByPaneKey: {}, + terminalLayoutsByTabId, + ptyIdsByTabId, + runtimePaneTitlesByTabId: {}, + acknowledgedAgentsByPaneKey: {}, + settings: null + } as unknown as SnapshotState +} + +beforeEach(() => { + collected.calls = 0 + collected.descriptors = 0 + collected.projections = 0 +}) + +describe('bucket-count work reuse', () => { + it('allocates the descriptor list once across recomputes that leave the workspace slices alone', () => { + const cache = createDashboardBucketCountsCache() + const state = largeState() + + // Five recomputes driven by agent traffic: prompt streaming on one pane, + // which is what actually invalidates the counts memo in the sidebar. + for (let pass = 0; pass < 5; pass += 1) { + const next: SnapshotState = { + ...state, + agentStatusByPaneKey: { + ...state.agentStatusByPaneKey, + [makePaneKey('tab0', leafId(0))]: entry(0, `streamed ${pass}`) + } + } + buildDashboardBucketCounts(next, NOW + pass, cache, 1) + } + + expect(collected.calls).toBe(1) + expect(collected.descriptors).toBe(WORKSPACE_COUNT) + }) + + it('projects only the rows of the worktree whose inputs moved', () => { + const cache = createDashboardBucketCountsCache() + const state = largeState() + buildDashboardBucketCounts(state, NOW, cache, 1) + // Cold pass: one row per workspace. + expect(collected.projections).toBe(WORKSPACE_COUNT) + + collected.projections = 0 + for (let pass = 0; pass < 5; pass += 1) { + buildDashboardBucketCounts( + { + ...state, + agentStatusByPaneKey: { + ...state.agentStatusByPaneKey, + [makePaneKey('tab0', leafId(0))]: entry(0, `streamed ${pass}`) + } + }, + NOW + pass, + cache, + 1 + ) + } + // One rebuilt worktree per pass, not the whole board. + expect(collected.projections).toBe(5) + }) + + it('recounts every worktree without rebuilding rows when acknowledgements change', () => { + const cache = createDashboardBucketCountsCache() + const state = largeState() + buildDashboardBucketCounts(state, NOW, cache, 1) + + collected.projections = 0 + const acked: SnapshotState = { + ...state, + acknowledgedAgentsByPaneKey: { [makePaneKey('tab0', leafId(0))]: NOW } + } + buildDashboardBucketCounts(acked, NOW + 1, cache, 1) + expect(cache.lastComputedWorktreeIds).toEqual([]) + expect(collected.projections).toBe(WORKSPACE_COUNT) + }) + + it('rebuilds the descriptor list when any slice it reads changes identity', () => { + const cache = createDashboardBucketCountsCache() + const state = largeState() + buildDashboardBucketCounts(state, NOW, cache, 1) + expect(collected.calls).toBe(1) + + const slices: (keyof SnapshotState | 'projectGroups' | 'folderWorkspaces')[] = [ + 'repos', + 'worktreesByRepo', + 'folderWorkspaces', + 'projectGroups' + ] + let previous: Record = state as unknown as Record + for (const [index, slice] of slices.entries()) { + const source = previous[slice] + const next = { + ...previous, + [slice]: Array.isArray(source) ? [...source] : { ...(source as object) } + } + buildDashboardBucketCounts(next as unknown as SnapshotState, NOW, cache, 1) + expect(collected.calls, `re-collects after ${slice} changes`).toBe(index + 2) + previous = next + } + }) + + it('does not memoize when the caller passes no cache', () => { + const state = largeState() + buildDashboardBucketCounts(state, NOW) + buildDashboardBucketCounts(state, NOW) + expect(collected.calls).toBe(2) + }) +}) diff --git a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.equivalence.test.ts b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.equivalence.test.ts new file mode 100644 index 00000000000..2bca8fdf0e0 --- /dev/null +++ b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.equivalence.test.ts @@ -0,0 +1,371 @@ +import { describe, expect, it } from 'vitest' +import { + AGENT_STATUS_STALE_AFTER_MS, + type AgentStatusEntry +} from '../../../../shared/agent-status-types' +import type { DashboardBucket } from '../../../../shared/dashboard-snapshot' +import type { FolderWorkspace } from '../../../../shared/folder-workspace-types' +import type { ProjectGroup } from '../../../../shared/project-group-types' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { Worktree } from '../../../../shared/worktree/types' +import { folderWorkspaceKey } from '../../../../shared/workspace-scope' +import { EMPTY_WORKTREE_AGENT_ORCHESTRATION } from '../sidebar/worktree-agent-orchestration-batch' +import type { DashboardSnapshotState } from './build-dashboard-snapshot' +import { + buildDashboardBucketCounts, + createDashboardBucketCountsCache +} from './build-dashboard-bucket-counts' +import { selectDashboardOrchestration } from './dashboard-orchestration-selection' +import { dashboardRowBucketProjection } from './dashboard-row-bucket' +import { collectActiveDashboardWorkspaces } from './dashboard-snapshot-workspaces' +import { selectWorktreeAgentRowsCached } from './worktree-agent-rows-cache' + +const BASE = 1_700_000_000_000 +const STALE = AGENT_STATUS_STALE_AFTER_MS +const LEAF_1 = '11111111-1111-4111-8111-111111111111' +const LEAF_2 = '22222222-2222-4222-8222-222222222222' +const LEAF_3 = '33333333-3333-4333-8333-333333333333' +const PANE_1 = makePaneKey('tab1', LEAF_1) +const PANE_2 = makePaneKey('tab2', LEAF_2) +const FOLDER_WORKSPACE_ID = folderWorkspaceKey('folder-1') +const PANE_3 = makePaneKey('tab3', LEAF_3) + +/** + * Unmemoized reference walk, composed from the same shared primitives the sidebar + * counts are defined by. Every assertion below pins the memoized builder to this. + */ +function oracleBucketCounts( + state: DashboardSnapshotState, + now: number +): Record { + const counts: Record = { + attention: 0, + working: 0, + done: 0, + idle: 0 + } + const activeWorktrees = collectActiveDashboardWorkspaces(state, false) + const { singletonOrchestration, orchestrationByWorktree } = selectDashboardOrchestration( + state, + activeWorktrees + ) + for (const { worktree } of activeWorktrees) { + const rows = selectWorktreeAgentRowsCached({ + state, + worktreeId: worktree.id, + orchestration: + singletonOrchestration ?? + orchestrationByWorktree?.get(worktree.id) ?? + EMPTY_WORKTREE_AGENT_ORCHESTRATION, + now, + generation: undefined + }) + for (const row of rows) { + if (row.rowSource === 'subagent') { + continue + } + counts[dashboardRowBucketProjection(row, state.acknowledgedAgentsByPaneKey).bucket] += 1 + } + } + return counts +} + +function worktree(id: string): Worktree { + return { + id, + repoId: 'r1', + path: `/r1/${id}`, + head: 'abc123', + branch: 'main', + isBare: false, + isMainWorktree: false, + displayName: id, + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: BASE + } +} + +function tab(id: string, worktreeId: string): TerminalTab { + return { + id, + ptyId: `pty-${id}`, + worktreeId, + title: 'shell', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: BASE + } +} + +function entry( + paneKey: string, + tabId: string, + worktreeId: string, + overrides: Partial = {} +): AgentStatusEntry { + return { + paneKey, + state: 'working', + prompt: 'do the thing', + updatedAt: BASE, + stateStartedAt: BASE - 5_000, + stateHistory: [], + agentType: 'claude', + tabId, + worktreeId, + ...overrides + } +} + +function leafLayout(tabId: string, leafId: string) { + return { + root: { type: 'leaf', leafId } as const, + activeLeafId: leafId, + expandedLeafId: null, + ptyIdsByLeafId: { [leafId]: `pty-${tabId}` } + } +} + +function folderWorkspace(): FolderWorkspace { + return { + id: 'folder-1', + projectGroupId: 'group-1', + name: 'Docs workspace', + folderPath: '/workspace/docs', + connectionId: null, + linkedTask: null, + comment: '', + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: BASE, + createdAt: BASE, + updatedAt: BASE + } +} + +function projectGroup(): ProjectGroup { + return { + id: 'group-1', + name: 'Documentation', + parentPath: '/workspace', + connectionId: null, + parentGroupId: null, + createdFrom: 'folder-scan', + tabOrder: 0, + isCollapsed: false, + color: null, + createdAt: BASE, + updatedAt: BASE + } +} + +function baseState(): DashboardSnapshotState { + return { + repos: [ + { + id: 'r1', + path: '/r1', + displayName: 'Repo One', + badgeColor: '#000', + addedAt: 1 + } + ], + worktreesByRepo: { r1: [worktree('w1'), worktree('w2')] }, + folderWorkspaces: [folderWorkspace()], + projectGroups: [projectGroup()], + tabsByWorktree: { + w1: [tab('tab1', 'w1')], + w2: [tab('tab2', 'w2')], + [FOLDER_WORKSPACE_ID]: [tab('tab3', FOLDER_WORKSPACE_ID)] + }, + agentStatusByPaneKey: { + [PANE_1]: entry(PANE_1, 'tab1', 'w1'), + [PANE_2]: entry(PANE_2, 'tab2', 'w2', { state: 'done' }), + [PANE_3]: entry(PANE_3, 'tab3', FOLDER_WORKSPACE_ID, { + state: 'waiting' + }) + }, + retainedAgentsByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + runtimeAgentOrchestrationByPaneKey: {}, + terminalLayoutsByTabId: { + tab1: leafLayout('tab1', LEAF_1), + tab2: leafLayout('tab2', LEAF_2), + tab3: leafLayout('tab3', LEAF_3) + }, + ptyIdsByTabId: { + tab1: ['pty-tab1'], + tab2: ['pty-tab2'], + tab3: ['pty-tab3'] + }, + runtimePaneTitlesByTabId: { tab1: { 0: 'shell' } }, + acknowledgedAgentsByPaneKey: {}, + settings: null + } as unknown as DashboardSnapshotState +} + +/** + * One agent's clock crossing, walked in order through a single cache. + * + * `generation` models the store's `agentStatusEpoch`: it stays put until decay + * may have shifted a bucket. `BASE + STALE` is the last instant an entry is + * still fresh (`now - observedAt <= STALE`) and the only point where a cache + * *hit* — not a recompute — has to return a freshness verdict. + */ +const CLOCK_WALK: { label: string; now: number; generation: number }[] = [ + { label: 'cold', now: BASE, generation: 1 }, + { label: 'last fresh instant (cache hit)', now: BASE + STALE, generation: 1 }, + { label: 'first stale instant', now: BASE + STALE + 1, generation: 2 }, + { label: 'long stale (cache hit)', now: BASE + STALE * 4, generation: 2 }, + { label: 'long stale (recomputed)', now: BASE + STALE * 4, generation: 3 } +] + +type Mutation = { + label: string + apply: (state: DashboardSnapshotState) => DashboardSnapshotState +} + +const MUTATIONS: Mutation[] = [ + { label: 'unchanged', apply: (state) => state }, + { + label: 'acknowledgement written', + apply: (state) => ({ + ...state, + acknowledgedAgentsByPaneKey: { [PANE_2]: BASE + STALE * 8 } + }) + }, + { + label: 'unrelated pane-title frame', + apply: (state) => ({ + ...state, + runtimePaneTitlesByTabId: { + ...state.runtimePaneTitlesByTabId, + tab2: { 0: 'sh' } + } + }) + }, + { + label: 'status write on one worktree', + apply: (state) => ({ + ...state, + agentStatusByPaneKey: { + ...state.agentStatusByPaneKey, + [PANE_1]: entry(PANE_1, 'tab1', 'w1', { + prompt: 'streamed', + state: 'blocked' + }) + } + }) + }, + { + label: 'worktree leaves the active set', + apply: (state) => ({ ...state, worktreesByRepo: { r1: [worktree('w1')] } }) + }, + { + label: 'folder workspace archived', + apply: (state) => ({ + ...state, + folderWorkspaces: [{ ...folderWorkspace(), isArchived: true }] + }) + }, + { + label: 'project group renamed', + apply: (state) => ({ + ...state, + projectGroups: [{ ...projectGroup(), name: 'Renamed' }] + }) + }, + { + label: 'pty goes away', + apply: (state) => ({ + ...state, + ptyIdsByTabId: { tab2: ['pty-tab2'], tab3: ['pty-tab3'] } + }) + } +] + +describe('buildDashboardBucketCounts equivalence with the unmemoized walk', () => { + it('matches the oracle for every mutation at every clock, sharing one cache', () => { + for (const mutation of MUTATIONS) { + const cache = createDashboardBucketCountsCache() + for (const step of CLOCK_WALK) { + const state = mutation.apply(baseState()) + expect( + buildDashboardBucketCounts(state, step.now, cache, step.generation), + `${mutation.label} @ ${step.label}` + ).toEqual(oracleBucketCounts(state, step.now)) + } + } + }) + + it('matches the oracle when mutations are applied cumulatively through one cache', () => { + const cache = createDashboardBucketCountsCache() + let state = baseState() + for (const mutation of MUTATIONS) { + state = mutation.apply(state) + for (const step of CLOCK_WALK) { + expect( + buildDashboardBucketCounts(state, step.now, cache, step.generation), + `${mutation.label} @ ${step.label}` + ).toEqual(oracleBucketCounts(state, step.now)) + } + } + }) + + it('serves the last-fresh instant from a cache hit and still decays one ms later', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + + expect(buildDashboardBucketCounts(state, BASE, cache, 1).working).toBe(1) + expect(cache.lastComputedWorktreeIds.length).toBe(3) + + // The boundary the earlier matrix never covered: a cache hit answering at the + // exact instant `now - observedAt === STALE`, where the entry is still fresh. + const atBoundary = buildDashboardBucketCounts(state, BASE + STALE, cache, 1) + expect(cache.lastComputedWorktreeIds).toEqual([]) + expect(atBoundary.working).toBe(1) + expect(atBoundary).toEqual(oracleBucketCounts(state, BASE + STALE)) + + const pastBoundary = buildDashboardBucketCounts(state, BASE + STALE + 1, cache, 2) + expect(pastBoundary.working).toBe(0) + expect(pastBoundary).toEqual(oracleBucketCounts(state, BASE + STALE + 1)) + }) + + it('returns the previous counts object when the four totals are unchanged', () => { + const cache = createDashboardBucketCountsCache() + const state = baseState() + const first = buildDashboardBucketCounts(state, BASE, cache, 1) + + // A prompt stream on one pane: rows rebuild, totals do not move. + const streamed: DashboardSnapshotState = { + ...state, + agentStatusByPaneKey: { + ...state.agentStatusByPaneKey, + [PANE_1]: entry(PANE_1, 'tab1', 'w1', { prompt: 'more output' }) + } + } + const second = buildDashboardBucketCounts(streamed, BASE + 1_000, cache, 1) + expect(cache.lastComputedWorktreeIds).toEqual(['w1']) + expect(second).toBe(first) + + const moved: DashboardSnapshotState = { + ...state, + agentStatusByPaneKey: { + ...state.agentStatusByPaneKey, + [PANE_1]: entry(PANE_1, 'tab1', 'w1', { state: 'blocked' }) + } + } + expect(buildDashboardBucketCounts(moved, BASE + 2_000, cache, 1)).not.toBe(first) + }) +}) diff --git a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts index ea21ff33a45..e673ad73298 100644 --- a/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts +++ b/src/renderer/src/components/dashboard/build-dashboard-bucket-counts.ts @@ -1,8 +1,13 @@ import type { DashboardBucket } from '../../../../shared/dashboard-snapshot' import type { DashboardSnapshotState } from './build-dashboard-snapshot' -import { collectActiveDashboardWorkspaces } from './dashboard-snapshot-workspaces' +import { + collectActiveDashboardWorkspaces, + type ActiveDashboardWorkspace, + type DashboardWorkspaceState +} from './dashboard-snapshot-workspaces' import { selectDashboardOrchestration } from './dashboard-orchestration-selection' import { dashboardRowBucketProjection } from './dashboard-row-bucket' +import type { DashboardAgentRowWithLineage } from './agent-row-lineage' import { EMPTY_WORKTREE_AGENT_ORCHESTRATION } from '../sidebar/worktree-agent-orchestration-batch' import { createWorktreeAgentRowsCache, @@ -19,20 +24,133 @@ const EMPTY_COUNTS: Record = { idle: 0 } -export type DashboardBucketCountsCache = WorktreeAgentRowsCache +type ActiveWorkspacesMemo = { + repos: unknown + worktreesByRepo: unknown + folderWorkspaces: unknown + projectGroups: unknown + workspaces: ActiveDashboardWorkspace[] +} + +type WorktreeTallyMemo = { + rows: readonly unknown[] + acknowledgedAgentsByPaneKey: unknown + tally: Record +} + +export type DashboardBucketCountsCache = WorktreeAgentRowsCache & { + /** Memo over the metadata-free workspace collection; see selectActiveDashboardWorkspaces. */ + activeWorkspaces: ActiveWorkspacesMemo | null + /** Per-worktree bucket tallies; see tallyWorktreeRows. */ + tallyByWorktree: Map + /** Previously returned totals, reused by identity when all four are unchanged. */ + lastCounts: Record | null +} export function createDashboardBucketCountsCache(): DashboardBucketCountsCache { - return createWorktreeAgentRowsCache() + return { + ...createWorktreeAgentRowsCache(), + activeWorkspaces: null, + tallyByWorktree: new Map(), + lastCounts: null + } +} + +/** + * The workspace descriptor list, reused by identity while its inputs hold. + * + * `collectActiveDashboardWorkspaces(state, false)` allocates one descriptor per + * workspace (hundreds, in a large install) and, with metadata off, reads only the + * four slices keyed here — see the read-set note on `DashboardWorkspaceState`. + * Every other slice it can touch sits behind an `includeMapMetadata` gate. + */ +function selectActiveDashboardWorkspaces( + state: DashboardWorkspaceState, + cache: DashboardBucketCountsCache | undefined +): ActiveDashboardWorkspace[] { + const memo = cache?.activeWorkspaces + if ( + memo && + memo.repos === state.repos && + memo.worktreesByRepo === state.worktreesByRepo && + memo.folderWorkspaces === state.folderWorkspaces && + memo.projectGroups === state.projectGroups + ) { + return memo.workspaces + } + const workspaces = collectActiveDashboardWorkspaces(state, false) + if (cache) { + cache.activeWorkspaces = { + repos: state.repos, + worktreesByRepo: state.worktreesByRepo, + folderWorkspaces: state.folderWorkspaces, + projectGroups: state.projectGroups, + workspaces + } + } + return workspaces +} + +function countsEqual( + a: Record, + b: Record +): boolean { + return ( + a.attention === b.attention && a.working === b.working && a.done === b.done && a.idle === b.idle + ) +} + +/** + * One worktree's bucket tally, reused while its rows and the acknowledgement map + * both hold their identity. + * + * `dashboardRowBucketProjection` reads nothing but the row and + * `acknowledgedAgentsByPaneKey[row.paneKey]`, so those two identities are the + * whole input. Keying on the ack slice rather than folding it into the row cache + * keeps main's property that an acknowledgement recounts without rebuilding rows. + */ +function tallyWorktreeRows( + rows: DashboardAgentRowWithLineage[], + acknowledgedAgentsByPaneKey: Record | undefined, + worktreeId: string, + cache: DashboardBucketCountsCache | undefined +): Record { + const memo = cache?.tallyByWorktree.get(worktreeId) + if ( + memo && + memo.rows === rows && + memo.acknowledgedAgentsByPaneKey === acknowledgedAgentsByPaneKey + ) { + return memo.tally + } + const tally = { attention: 0, working: 0, done: 0, idle: 0 } satisfies Record< + DashboardBucket, + number + > + for (const row of rows) { + if (row.rowSource === 'subagent') { + continue + } + tally[dashboardRowBucketProjection(row, acknowledgedAgentsByPaneKey).bucket] += 1 + } + cache?.tallyByWorktree.set(worktreeId, { + rows, + acknowledgedAgentsByPaneKey, + tally + }) + return tally } /** * Derive sidebar counts without allocating dashboard cards or metadata. * - * With a cache, each worktree's row pipeline reruns only when one of its own - * inputs changed (see worktree-agent-rows-cache). Counting over the (possibly - * reused) rows happens on every call, so acknowledgement changes recount - * without rebuilding any rows. `generation` must change whenever time-based - * freshness decay may have shifted a bucket (agentStatusEpoch). + * With a cache, three layers reuse work independently: the workspace descriptor + * list while its four slices hold, each worktree's row pipeline while its own + * inputs hold (see worktree-agent-rows-cache), and each worktree's bucket tally + * while its rows and the acknowledgement map hold. An acknowledgement write + * therefore recounts without rebuilding any rows, as before. `generation` must + * change whenever time-based freshness decay may have shifted a bucket + * (agentStatusEpoch). */ export function buildDashboardBucketCounts( state: DashboardSnapshotState, @@ -46,7 +164,7 @@ export function buildDashboardBucketCounts( done: 0, idle: 0 } satisfies Record - const activeWorktrees = collectActiveDashboardWorkspaces(state, false) + const activeWorktrees = selectActiveDashboardWorkspaces(state, cache) const { singletonOrchestration, orchestrationByWorktree } = selectDashboardOrchestration( state, activeWorktrees @@ -68,19 +186,34 @@ export function buildDashboardBucketCounts( generation, cache }) - for (const row of rows) { - if (row.rowSource === 'subagent') { - continue - } - counts[dashboardRowBucketProjection(row, state.acknowledgedAgentsByPaneKey).bucket] += 1 - } + const tally = tallyWorktreeRows(rows, state.acknowledgedAgentsByPaneKey, worktreeId, cache) + counts.attention += tally.attention + counts.working += tally.working + counts.done += tally.done + counts.idle += tally.idle } if (cache) { + for (const worktreeId of cache.tallyByWorktree.keys()) { + if (!cache.seenWorktreeIds.has(worktreeId)) { + cache.tallyByWorktree.delete(worktreeId) + } + } finishWorktreeAgentRowsCachePass(cache) } - return counts.attention === 0 && counts.working === 0 && counts.done === 0 && counts.idle === 0 - ? EMPTY_COUNTS - : counts + const totals = + counts.attention === 0 && counts.working === 0 && counts.done === 0 && counts.idle === 0 + ? EMPTY_COUNTS + : counts + if (!cache) { + return totals + } + // Why: most recomputes are triggered by agent traffic that leaves all four + // totals where they were; a fresh object there would re-render the sidebar + // entry and miss every downstream memo keyed on this result. + const stable = + cache.lastCounts && countsEqual(cache.lastCounts, totals) ? cache.lastCounts : totals + cache.lastCounts = stable + return stable } diff --git a/src/renderer/src/components/dashboard/dashboard-snapshot-workspaces.ts b/src/renderer/src/components/dashboard/dashboard-snapshot-workspaces.ts index fcb34581414..0d8f2a33d93 100644 --- a/src/renderer/src/components/dashboard/dashboard-snapshot-workspaces.ts +++ b/src/renderer/src/components/dashboard/dashboard-snapshot-workspaces.ts @@ -30,7 +30,12 @@ export type ActiveDashboardWorkspace = { hostLabel?: string } -type DashboardWorkspaceState = Pick & +/** With `includeMapMetadata: false`, `collectActiveDashboardWorkspaces` reads only + * `repos`, `worktreesByRepo`, `folderWorkspaces` and `projectGroups` — every other + * slice here sits behind a metadata gate. Callers memoize the result on exactly those + * four (see `build-dashboard-bucket-counts`), so widening the metadata-free read set + * means widening that key too. */ +export type DashboardWorkspaceState = Pick & Partial< Pick< AppState, diff --git a/src/renderer/src/components/dashboard/useAgentBucketCounts.gate.test.ts b/src/renderer/src/components/dashboard/useAgentBucketCounts.gate.test.ts new file mode 100644 index 00000000000..618d04a9169 --- /dev/null +++ b/src/renderer/src/components/dashboard/useAgentBucketCounts.gate.test.ts @@ -0,0 +1,158 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { shallow } from 'zustand/shallow' +import type { AppState } from '@/store/types' +import { + resetAgentBucketCountStateForTests, + selectAgentBucketCountState +} from './useAgentBucketCounts' + +vi.mock('@/store', () => ({ useAppStore: () => undefined })) + +const STORE_WRITES = 2_000 + +function storeState(): AppState { + return { + repos: [], + worktreesByRepo: {}, + tabsByWorktree: {}, + unifiedTabsByWorktree: {}, + agentStatusByPaneKey: {}, + retainedAgentsByPaneKey: {}, + migrationUnsupportedByPtyId: {}, + runtimeAgentOrchestrationByPaneKey: {}, + terminalLayoutsByTabId: {}, + ptyIdsByTabId: {}, + runtimePaneTitlesByTabId: {}, + folderWorkspaces: [], + acknowledgedAgentsByPaneKey: {}, + agentStatusEpoch: 0, + // A slice the counts never read: writing it is what "unrelated store write" means. + unreadCountsByWorktree: {} + } as unknown as AppState +} + +// What `useShallow` did before the gate, unwrapped from the hook so it can be +// driven directly: allocate the 14-key object, then `shallow()` it against the +// previous one. Kept here as the comparison baseline. +let previousShallow: Record | null = null +const shallowInputs = (s: AppState): Record => ({ + repos: s.repos, + worktreesByRepo: s.worktreesByRepo, + tabsByWorktree: s.tabsByWorktree, + unifiedTabsByWorktree: s.unifiedTabsByWorktree, + agentStatusByPaneKey: s.agentStatusByPaneKey, + retainedAgentsByPaneKey: s.retainedAgentsByPaneKey, + migrationUnsupportedByPtyId: s.migrationUnsupportedByPtyId, + runtimeAgentOrchestrationByPaneKey: s.runtimeAgentOrchestrationByPaneKey, + terminalLayoutsByTabId: s.terminalLayoutsByTabId, + ptyIdsByTabId: s.ptyIdsByTabId, + runtimePaneTitlesByTabId: s.runtimePaneTitlesByTabId, + folderWorkspaces: s.folderWorkspaces, + acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey, + agentStatusEpoch: s.agentStatusEpoch +}) +const shallowSelector = (s: AppState): Record => { + const next = shallowInputs(s) + if (previousShallow !== null && shallow(previousShallow, next)) { + return previousShallow + } + previousShallow = next + return next +} + +function countAllocations(run: () => void): { entries: number; maps: number } { + const realEntries = Object.entries + const RealMap = globalThis.Map + let entries = 0 + let maps = 0 + Object.entries = ((target: object) => { + entries += 1 + return realEntries(target) + }) as typeof Object.entries + class CountingMap extends RealMap { + constructor(init?: readonly (readonly [K, V])[] | null) { + super(init as never) + maps += 1 + } + } + globalThis.Map = CountingMap as unknown as MapConstructor + try { + run() + } finally { + Object.entries = realEntries + globalThis.Map = RealMap + } + return { entries, maps } +} + +afterEach(() => { + resetAgentBucketCountStateForTests() + previousShallow = null +}) + +describe('agent bucket count input gate', () => { + it('allocates nothing on a store write that leaves all fourteen slices alone', () => { + const state = storeState() + // Prime the gate, then replay the writes an unrelated slice would trigger. + selectAgentBucketCountState(state) + + const gated = countAllocations(() => { + for (let write = 0; write < STORE_WRITES; write += 1) { + selectAgentBucketCountState(state) + } + }) + const shallowBaseline = countAllocations(() => { + shallowSelector(state) + for (let write = 0; write < STORE_WRITES; write += 1) { + shallowSelector(state) + } + }) + + expect(gated).toEqual({ entries: 0, maps: 0 }) + // zustand v5's shallow() takes the compareEntries path on a plain object: + // two Object.entries arrays and two Maps per write, to conclude nothing moved. + expect(shallowBaseline.entries).toBe(STORE_WRITES * 2) + expect(shallowBaseline.maps).toBe(STORE_WRITES * 2) + }) + + it('returns the identical inputs object until a read slice changes identity', () => { + const state = storeState() + const first = selectAgentBucketCountState(state) + expect(selectAgentBucketCountState(state)).toBe(first) + + const unrelated = { + ...state, + unreadCountsByWorktree: {} + } as unknown as AppState + expect(selectAgentBucketCountState(unrelated)).toBe(first) + + const moved = { ...state, agentStatusEpoch: 1 } as unknown as AppState + const second = selectAgentBucketCountState(moved) + expect(second).not.toBe(first) + expect(second.agentStatusEpoch).toBe(1) + }) + + it('carries every slice the counts read, and settings pinned to null', () => { + const state = storeState() + const inputs = selectAgentBucketCountState(state) + expect(inputs.settings).toBeNull() + for (const key of [ + 'repos', + 'worktreesByRepo', + 'tabsByWorktree', + 'unifiedTabsByWorktree', + 'agentStatusByPaneKey', + 'retainedAgentsByPaneKey', + 'migrationUnsupportedByPtyId', + 'runtimeAgentOrchestrationByPaneKey', + 'terminalLayoutsByTabId', + 'ptyIdsByTabId', + 'runtimePaneTitlesByTabId', + 'folderWorkspaces', + 'acknowledgedAgentsByPaneKey', + 'agentStatusEpoch' + ] as const) { + expect(inputs[key], key).toBe(state[key]) + } + }) +}) diff --git a/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx b/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx index 541fbd716c1..5c2ac278724 100644 --- a/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx +++ b/src/renderer/src/components/dashboard/useAgentBucketCounts.test.tsx @@ -37,11 +37,12 @@ vi.mock('./build-dashboard-bucket-counts', () => ({ }) })) -import { useAgentBucketCounts } from './useAgentBucketCounts' +import { resetAgentBucketCountStateForTests, useAgentBucketCounts } from './useAgentBucketCounts' afterEach(() => { cleanup() vi.clearAllMocks() + resetAgentBucketCountStateForTests() mocks.state.acknowledgedAgentsByPaneKey = {} mocks.state.unrelatedEpoch = 0 }) diff --git a/src/renderer/src/components/dashboard/useAgentBucketCounts.ts b/src/renderer/src/components/dashboard/useAgentBucketCounts.ts index 0e8c2881244..d003991cecf 100644 --- a/src/renderer/src/components/dashboard/useAgentBucketCounts.ts +++ b/src/renderer/src/components/dashboard/useAgentBucketCounts.ts @@ -1,6 +1,6 @@ import { useMemo, useRef } from 'react' import { useAppStore } from '@/store' -import { useShallow } from 'zustand/react/shallow' +import type { AppState } from '@/store/types' import type { DashboardBucket } from '../../../../shared/dashboard-snapshot' import { buildDashboardBucketCounts, @@ -9,93 +9,99 @@ import { export type AgentBucketCounts = Record +/** The bucket-count inputs, shaped so it doubles as the snapshot state passed to the builder. */ +export type AgentBucketCountState = Pick< + AppState, + | 'repos' + | 'worktreesByRepo' + | 'tabsByWorktree' + | 'unifiedTabsByWorktree' + | 'agentStatusByPaneKey' + | 'retainedAgentsByPaneKey' + | 'migrationUnsupportedByPtyId' + | 'runtimeAgentOrchestrationByPaneKey' + | 'terminalLayoutsByTabId' + | 'ptyIdsByTabId' + | 'runtimePaneTitlesByTabId' + | 'folderWorkspaces' + | 'acknowledgedAgentsByPaneKey' + | 'agentStatusEpoch' +> & { + // Why null: counts never render a card's conversation name, so the + // generated-title gate is moot and the sidebar stays off settings. + settings: null +} + +// Why module scope rather than useShallow: zustand runs this selector on every +// store write, and shallow() on a plain object takes the compareEntries path — +// two Object.entries arrays, 28 tuples and two Maps allocated per write just to +// conclude nothing moved. Fourteen `===` against the previous slices allocates +// nothing on the unchanged path, and the result is a pure function of the state +// so one gate can serve every mounted consumer. +let previousState: AgentBucketCountState | null = null + +/** Test-only: drop the cross-render identity gate so a case starts cold. */ +export function resetAgentBucketCountStateForTests(): void { + previousState = null +} + +export function selectAgentBucketCountState(s: AppState): AgentBucketCountState { + const previous = previousState + if ( + previous !== null && + previous.repos === s.repos && + previous.worktreesByRepo === s.worktreesByRepo && + previous.tabsByWorktree === s.tabsByWorktree && + previous.unifiedTabsByWorktree === s.unifiedTabsByWorktree && + previous.agentStatusByPaneKey === s.agentStatusByPaneKey && + previous.retainedAgentsByPaneKey === s.retainedAgentsByPaneKey && + previous.migrationUnsupportedByPtyId === s.migrationUnsupportedByPtyId && + previous.runtimeAgentOrchestrationByPaneKey === s.runtimeAgentOrchestrationByPaneKey && + previous.terminalLayoutsByTabId === s.terminalLayoutsByTabId && + previous.ptyIdsByTabId === s.ptyIdsByTabId && + previous.runtimePaneTitlesByTabId === s.runtimePaneTitlesByTabId && + previous.folderWorkspaces === s.folderWorkspaces && + previous.acknowledgedAgentsByPaneKey === s.acknowledgedAgentsByPaneKey && + previous.agentStatusEpoch === s.agentStatusEpoch + ) { + return previous + } + previousState = { + repos: s.repos, + worktreesByRepo: s.worktreesByRepo, + tabsByWorktree: s.tabsByWorktree, + unifiedTabsByWorktree: s.unifiedTabsByWorktree, + agentStatusByPaneKey: s.agentStatusByPaneKey, + retainedAgentsByPaneKey: s.retainedAgentsByPaneKey, + migrationUnsupportedByPtyId: s.migrationUnsupportedByPtyId, + runtimeAgentOrchestrationByPaneKey: s.runtimeAgentOrchestrationByPaneKey, + terminalLayoutsByTabId: s.terminalLayoutsByTabId, + ptyIdsByTabId: s.ptyIdsByTabId, + runtimePaneTitlesByTabId: s.runtimePaneTitlesByTabId, + folderWorkspaces: s.folderWorkspaces, + acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey, + agentStatusEpoch: s.agentStatusEpoch, + settings: null + } + return previousState +} + /** * Per-state agent counts for the sidebar dashboard entry, using the same row * and bucket derivation as the pop-out board without allocating its cards. * Recomputes only when an input slice changes. */ export function useAgentBucketCounts(): AgentBucketCounts { - const { - repos, - worktreesByRepo, - tabsByWorktree, - unifiedTabsByWorktree, - agentStatusByPaneKey, - retainedAgentsByPaneKey, - migrationUnsupportedByPtyId, - runtimeAgentOrchestrationByPaneKey, - terminalLayoutsByTabId, - ptyIdsByTabId, - runtimePaneTitlesByTabId, - folderWorkspaces, - acknowledgedAgentsByPaneKey, - agentStatusEpoch - } = useAppStore( - useShallow((s) => ({ - repos: s.repos, - worktreesByRepo: s.worktreesByRepo, - tabsByWorktree: s.tabsByWorktree, - unifiedTabsByWorktree: s.unifiedTabsByWorktree, - agentStatusByPaneKey: s.agentStatusByPaneKey, - retainedAgentsByPaneKey: s.retainedAgentsByPaneKey, - migrationUnsupportedByPtyId: s.migrationUnsupportedByPtyId, - runtimeAgentOrchestrationByPaneKey: s.runtimeAgentOrchestrationByPaneKey, - terminalLayoutsByTabId: s.terminalLayoutsByTabId, - ptyIdsByTabId: s.ptyIdsByTabId, - runtimePaneTitlesByTabId: s.runtimePaneTitlesByTabId, - folderWorkspaces: s.folderWorkspaces, - acknowledgedAgentsByPaneKey: s.acknowledgedAgentsByPaneKey, - agentStatusEpoch: s.agentStatusEpoch - })) - ) - + const state = useAppStore(selectAgentBucketCountState) // Why a per-hook cache: unrelated status/title writes change one worktree's inputs; - // the cache keeps every other worktree's counts without rerunning its row pipeline. + // the cache keeps every other worktree's rows without rerunning its row pipeline. const cacheRef = useRef>(undefined!) cacheRef.current ??= createDashboardBucketCountsCache() return useMemo(() => { - return buildDashboardBucketCounts( - { - repos, - worktreesByRepo, - tabsByWorktree, - unifiedTabsByWorktree, - agentStatusByPaneKey, - retainedAgentsByPaneKey, - migrationUnsupportedByPtyId, - runtimeAgentOrchestrationByPaneKey, - terminalLayoutsByTabId, - ptyIdsByTabId, - runtimePaneTitlesByTabId, - folderWorkspaces, - acknowledgedAgentsByPaneKey, - // Same: counts never render a card's conversation name, so the - // generated-title gate is moot and the sidebar stays off settings. - settings: null - }, - Date.now(), - cacheRef.current, - // Why: time-based freshness decay is signaled by agentStatusEpoch; it invalidates - // every cached worktree so stale-decayed buckets recount. - agentStatusEpoch - ) - // Why: Date.now() is read inside the memo (not a dep) so idle-decay tracks - // agentStatusEpoch ticks, matching useDashboardData. + // Why Date.now() is read here and not a dep: idle-decay tracks agentStatusEpoch + // ticks (carried in `state`), matching useDashboardData. That epoch doubles as the + // cache generation, so a stale-boundary tick recounts every decayed bucket. + return buildDashboardBucketCounts(state, Date.now(), cacheRef.current, state.agentStatusEpoch) // eslint-disable-next-line react-hooks/exhaustive-deps - }, [ - repos, - worktreesByRepo, - tabsByWorktree, - unifiedTabsByWorktree, - agentStatusByPaneKey, - retainedAgentsByPaneKey, - migrationUnsupportedByPtyId, - runtimeAgentOrchestrationByPaneKey, - terminalLayoutsByTabId, - ptyIdsByTabId, - runtimePaneTitlesByTabId, - folderWorkspaces, - acknowledgedAgentsByPaneKey, - agentStatusEpoch - ]) + }, [state]) } From fb48a9771b4f9eb87c4a6bfe1aadaf021034257c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:12:21 -0700 Subject: [PATCH 110/398] fix(gh): reap the whole gh/glab process tree at the deadline on POSIX (#18258) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `gh` and `glab` on PATH are routinely shims — mise, asdf, volta, or a hand-written wrapper — so a timed-out invocation has a chain to stop, not one process. `execFileCapture`'s POSIX kill path signals only the direct child; the descendants are orphaned to init and keep running. #18234 is exactly that shape: `bash ~/.local/bin/gh` -> `mise x gh` -> `gh`, where the reporter found the tail reparented to `systemd --user` and still at 100% CPU nearly two hours later. The 15s deadline #18239 added bounds Orca's semaphore slot and its promise; it does not bound the CPU burn. Route both CLIs through `execFileCaptureToTermination`, the primitive git's barrier path already uses: POSIX children spawn `detached`, the deadline signals `-pgid` and escalates to SIGKILL, and the promise waits for verified termination. Windows behaviour is unchanged (`taskkill /t` either way). Switching primitives also swapped execFile's hard maxBuffer failure for `runProcess`'s silent clipping, which would have turned an oversized gh response into a shorter valid-looking one. `ProcessResult` now reports truncation and the capture rejects on it, restoring the old contract and closing the same latent gap on git's barrier path. --- .../git/command-runner/exec-file-capture.ts | 22 +- .../gh-exec-file-deadline.test.ts | 102 ++-- src/main/git/command-runner/gh-exec-file.ts | 30 +- src/main/git/command-runner/glab-exec-file.ts | 25 +- src/main/git/runner-command-exec.test.ts | 160 ++++-- .../git/runner-gh-rate-limit-breaker.test.ts | 26 +- src/main/git/runner-wsl-gh-fallback.test.ts | 461 ++++++------------ ...tlab-known-host-probe-wsl-fallback.test.ts | 33 +- .../__fixtures__/fake-spawned-child.ts | 75 +++ .../child-process/bounded-output-sink.ts | 7 +- src/shared/child-process/process-spec.ts | 2 + src/shared/child-process/run-process.test.ts | 21 + src/shared/child-process/run-process.ts | 12 +- 13 files changed, 508 insertions(+), 468 deletions(-) create mode 100644 src/shared/child-process/__fixtures__/fake-spawned-child.ts diff --git a/src/main/git/command-runner/exec-file-capture.ts b/src/main/git/command-runner/exec-file-capture.ts index b1b9c664176..e9ae815ff34 100644 --- a/src/main/git/command-runner/exec-file-capture.ts +++ b/src/main/git/command-runner/exec-file-capture.ts @@ -25,7 +25,11 @@ export async function execFileCaptureToTermination( options: ExecFileCaptureOptions, termination?: WslProcessGroupTermination ): Promise<{ stdout: string | Buffer; stderr: string | Buffer }> { - const result = await runProcess({ + // Why measured here: runProcess spawns inside its promise executor, which runs + // synchronously, so this brackets exactly the main-thread block execFileCapture + // reports for its own spawns. + const spawnStartedAt = performance.now() + const pending = runProcess({ program: command, args, cwd: typeof options.cwd === 'string' ? options.cwd : undefined, @@ -37,10 +41,17 @@ export async function execFileCaptureToTermination( onChildTerminated: options.onChildTerminated, ...(options.stdin === undefined ? {} : { input: options.stdin }) }) + recordSubprocessSpawn(command, args, performance.now() - spawnStartedAt) + const result = await pending const stdout = options.encoding === 'buffer' ? Buffer.from(result.stdout) : result.stdout const cleanStderr = termination?.stripControlOutput(result.stderr) ?? result.stderr const stderr = options.encoding === 'buffer' ? Buffer.from(cleanStderr) : cleanStderr - if (result.code === 0 && !result.timedOut && !options.signal?.aborted) { + if ( + result.code === 0 && + !result.timedOut && + !result.outputTruncated && + !options.signal?.aborted + ) { return { stdout, stderr } } const error = result.timedOut @@ -48,7 +59,12 @@ export async function execFileCaptureToTermination( : new Error( options.signal?.aborted ? 'The operation was aborted.' - : cleanStderr.trim() || `${command} exited with ${result.code}.` + : result.outputTruncated + ? // Why fail instead of returning the clipped text: callers parse this + // as JSON or JSONL, where a clipped answer reads as a shorter valid + // one. execFile's own maxBuffer overrun errored for the same reason. + `${command} produced more than ${options.maxBuffer ?? DEFAULT_GIT_MAX_BUFFER} bytes of output.` + : cleanStderr.trim() || `${command} exited with ${result.code}.` ) if (options.signal?.aborted) { error.name = 'AbortError' diff --git a/src/main/git/command-runner/gh-exec-file-deadline.test.ts b/src/main/git/command-runner/gh-exec-file-deadline.test.ts index 3775b67a7ed..fa07dec32be 100644 --- a/src/main/git/command-runner/gh-exec-file-deadline.test.ts +++ b/src/main/git/command-runner/gh-exec-file-deadline.test.ts @@ -2,20 +2,15 @@ import { EventEmitter } from 'node:events' import type { ChildProcess } from 'node:child_process' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -const { execFileMock, spawnMock, killSpawnedCommandTreeMock } = vi.hoisted(() => ({ - execFileMock: vi.fn(), +const { spawnMock, processKillMock } = vi.hoisted(() => ({ spawnMock: vi.fn(), - killSpawnedCommandTreeMock: vi.fn().mockResolvedValue(undefined) + processKillMock: vi.fn() })) vi.mock('node:child_process', async (importOriginal) => ({ ...(await importOriginal()), - execFile: execFileMock, spawn: spawnMock })) -vi.mock('./spawned-command-tree-kill', () => ({ - killSpawnedCommandTree: killSpawnedCommandTreeMock -})) import { ghExecFileAsync } from './gh-exec-file' @@ -29,66 +24,87 @@ function mockChild(pid = 4321): ChildProcess { return child as unknown as ChildProcess } +function settleChild(child: ChildProcess, stdout: string): void { + child.stdout?.emit('data', Buffer.from(stdout)) + child.emit('exit', 0, null) + child.emit('close', 0, null) +} + /** * The contract the star check depends on after #18234: a `gh` that never exits - * is killed at the deadline, tree and all, rather than running forever. + * is killed at the deadline, and the kill reaches the whole chain. On the + * reporter's box `gh` was a shell wrapper calling `mise x gh`, so signalling + * only the direct child left the rest of the chain running under init. */ describe('gh exec deadline', () => { beforeEach(() => { vi.useFakeTimers() - execFileMock.mockReset() spawnMock.mockReset() - killSpawnedCommandTreeMock.mockClear() + processKillMock.mockReset() + vi.spyOn(process, 'kill').mockImplementation(processKillMock as unknown as typeof process.kill) }) afterEach(() => { vi.useRealTimers() + vi.restoreAllMocks() }) - it('kills the process tree and rejects when gh never exits', async () => { - const child = mockChild() - // Why never invoking the callback: this is exactly the stuck child from - // #18234 — spawned, spinning, and never reporting an exit. - execFileMock.mockReturnValue(child) + it.runIf(process.platform !== 'win32')( + 'signals the whole process group, not just the child, when gh never exits', + async () => { + const child = mockChild() + // Why never emitting exit: this is exactly the stuck child from #18234 — + // spawned, spinning, and never reporting an exit. + spawnMock.mockReturnValue(child) - const pending = ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { - timeout: 15_000 - }) - const rejection = expect(pending).rejects.toThrow('timed out') - await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledOnce()) + const pending = ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { + timeout: 15_000 + }) + const rejection = expect(pending).rejects.toThrow('timed out') + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledOnce()) - // Not yet: the deadline has not elapsed. - expect(killSpawnedCommandTreeMock).not.toHaveBeenCalled() + // The child must be its own group leader, or the signal below would go to + // whatever group it inherited — Orca's own. + expect(spawnMock.mock.calls[0][2].detached).toBe(true) + expect(processKillMock).not.toHaveBeenCalled() - await vi.advanceTimersByTimeAsync(15_000) - await rejection + await vi.advanceTimersByTimeAsync(15_000) + await vi.advanceTimersByTimeAsync(15_000) + await rejection - expect(killSpawnedCommandTreeMock).toHaveBeenCalledWith(child) - }) + expect(processKillMock).toHaveBeenCalledWith(-4321, undefined) + } + ) it('spawns with hidden console and captured stdio, never an inherited or shell stdio', async () => { const child = mockChild() - execFileMock.mockImplementation( - ( - _command: string, - _args: string[], - _options: unknown, - callback: (error: Error | null, stdout: string, stderr: string) => void - ) => { - callback(null, 'HTTP/2.0 204 No Content\r\n', '') - return child - } - ) + spawnMock.mockImplementation(() => { + queueMicrotask(() => settleChild(child, 'HTTP/2.0 204 No Content\r\n')) + return child + }) - await ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { timeout: 15_000 }) + const result = await ghExecFileAsync(['api', '--include', 'user/starred/stablyai/orca'], { + timeout: 15_000 + }) - const [command, args, options] = execFileMock.mock.calls[0] + expect(result.stdout).toContain('204 No Content') + const [command, args, options] = spawnMock.mock.calls[0] expect(command).toBe('gh') expect(args).toEqual(['api', '--include', 'user/starred/stablyai/orca']) - // `execFile` captures stdout/stderr over pipes and never inherits Orca's; - // `shell` is never set, and the console stays hidden on Windows. expect(options.windowsHide).toBe(true) - expect(options.stdio).toBeUndefined() - expect(options.shell).toBeUndefined() + expect(options.stdio).toEqual(['pipe', 'pipe', 'pipe']) + expect(options.shell).toBe(false) + }) + + it('fails rather than returning a clipped answer when gh overruns maxBuffer', async () => { + const child = mockChild() + spawnMock.mockImplementation(() => { + queueMicrotask(() => settleChild(child, '['.padEnd(64, 'x'))) + return child + }) + + await expect( + ghExecFileAsync(['api', 'repos/stablyai/orca/issues'], { timeout: 15_000, maxBuffer: 8 }) + ).rejects.toThrow('more than 8 bytes') }) }) diff --git a/src/main/git/command-runner/gh-exec-file.ts b/src/main/git/command-runner/gh-exec-file.ts index b8f13be5d5e..e8308a8e4b4 100644 --- a/src/main/git/command-runner/gh-exec-file.ts +++ b/src/main/git/command-runner/gh-exec-file.ts @@ -19,7 +19,7 @@ import { isHostCommandMissing, resolveHostGitHubCli } from './github-cli-host-fallback' -import { execFileCapture } from './exec-file-capture' +import { execFileCaptureToTermination } from './exec-file-capture' import type { GitExecOptions } from './git-exec-options' import { argsLookIdempotent } from './gh-idempotency' import { applyGhHostToArgs, explicitGhHostname, explicitGhRepoHostname } from './gh-host-args' @@ -115,15 +115,25 @@ export async function ghExecFileAsync( let attemptedDefaultWslFallback = false for (let attempt = 0; attempt <= GH_RETRY_DELAYS_MS.length; attempt++) { try { - const { stdout, stderr } = await execFileCapture(resolved.binary, resolved.args, { - cwd: resolved.cwd, - encoding: (options.encoding ?? 'utf-8') as BufferEncoding, - maxBuffer: options.maxBuffer, - // Why: bound gh so one stuck child fails visibly instead of wedging the IPC lane. - timeout: options.timeout ?? defaultGhExecTimeoutMs(options.env), - env: nonInteractiveGhEnv(options.env), - signal: options.signal - }) + // Why to-termination and not execFileCapture: `gh` on PATH is routinely a + // shim (mise, asdf, volta, a hand-written wrapper), so the deadline below + // has a chain to reap, not one process. execFileCapture's POSIX kill only + // signals the direct child, which orphans the rest to init — a wedged + // helper then outlives the timeout that was supposed to bound it (#18234). + const { stdout, stderr } = await execFileCaptureToTermination( + resolved.binary, + resolved.args, + { + cwd: resolved.cwd, + encoding: (options.encoding ?? 'utf-8') as BufferEncoding, + maxBuffer: options.maxBuffer, + // Why: bound gh so one stuck child fails visibly instead of wedging the IPC lane. + timeout: options.timeout ?? defaultGhExecTimeoutMs(options.env), + env: nonInteractiveGhEnv(options.env), + signal: options.signal + }, + resolved.termination + ) return { stdout: stdout as string, stderr: stderr as string } } catch (err) { lastError = err diff --git a/src/main/git/command-runner/glab-exec-file.ts b/src/main/git/command-runner/glab-exec-file.ts index 3a4b4467ba9..3257dd9e818 100644 --- a/src/main/git/command-runner/glab-exec-file.ts +++ b/src/main/git/command-runner/glab-exec-file.ts @@ -2,7 +2,7 @@ import { addWslEnvKeys } from '../../wsl-env' import { extractExecError, parseRetryAfterMs } from '../exec-error' import { resolveCommand, resolveDefaultWslCli } from './wsl-command-resolution' import { isHostCommandMissing } from './github-cli-host-fallback' -import { execFileCapture } from './exec-file-capture' +import { execFileCaptureToTermination } from './exec-file-capture' import type { GitExecOptions } from './git-exec-options' import { argsLookIdempotent } from './gh-idempotency' import { @@ -63,14 +63,21 @@ export async function glabExecFileAsync( let attemptedDefaultWslFallback = false for (let attempt = 0; attempt <= GH_RETRY_DELAYS_MS.length; attempt++) { try { - const { stdout, stderr } = await execFileCapture(resolved.binary, resolved.args, { - cwd: resolved.cwd, - encoding: (options.encoding ?? 'utf-8') as BufferEncoding, - maxBuffer: options.maxBuffer, - timeout: options.timeout ?? DEFAULT_GLAB_EXEC_TIMEOUT_MS, - env: options.env, - signal: options.signal - }) + // Why to-termination: same shim chain as gh — the deadline has to reap the + // whole tree, not just the wrapper that spawned it (#18234). + const { stdout, stderr } = await execFileCaptureToTermination( + resolved.binary, + resolved.args, + { + cwd: resolved.cwd, + encoding: (options.encoding ?? 'utf-8') as BufferEncoding, + maxBuffer: options.maxBuffer, + timeout: options.timeout ?? DEFAULT_GLAB_EXEC_TIMEOUT_MS, + env: options.env, + signal: options.signal + }, + resolved.termination + ) return { stdout: stdout as string, stderr: stderr as string } } catch (err) { lastError = err diff --git a/src/main/git/runner-command-exec.test.ts b/src/main/git/runner-command-exec.test.ts index 27d7a72c747..89bcc4dca15 100644 --- a/src/main/git/runner-command-exec.test.ts +++ b/src/main/git/runner-command-exec.test.ts @@ -46,6 +46,31 @@ function createMockChildProcess(pid: number): MockChildProcess { return child } +/** + * Spawn stand-in for the gh/glab deadline tests: the CLI hangs, while the `ps` + * quiescence probe the tree termination runs answers immediately. + */ +function mockWedgedCliSpawn(child: MockChildProcess): void { + spawnMock.mockImplementation((program: string) => { + if (program !== 'ps') { + return child + } + const probe = createMockChildProcess(9100) + queueMicrotask(() => probe.emit('close', 0, null)) + return probe + }) +} + +/** Signals succeed; the existence probe reports the group already gone. */ +function mockProcessGroupSignals(): ReturnType { + return vi.spyOn(process, 'kill').mockImplementation(((_pid: number, signal?: unknown) => { + if (signal === 0) { + throw Object.assign(new Error('ESRCH'), { code: 'ESRCH' }) + } + return true + }) as typeof process.kill) +} + function createMockTaskkillProcess(): MockChildProcess { const child = createMockChildProcess(9000) child.unref = vi.fn() @@ -271,32 +296,46 @@ describe('runner execFile timeout handling', () => { } ) - it('rejects gh executions that never call back using the default timeout', async () => { + // Why the group and not the child (#18234): `gh` and `glab` on PATH are often + // shims, so the deadline has a chain to reap. Signalling only the direct child + // leaves the rest of it running under init long after the deadline passed. + it('signals the whole gh process group when gh never calls back', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { + cwd: '/repo' + }) + const rejection = expect(promise).rejects.toThrow('gh timed out.') + await vi.advanceTimersByTimeAsync(30_000) + expect(spawnMock.mock.calls[0][2].detached).toBe(true) + await vi.advanceTimersByTimeAsync(2_000) - const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { - cwd: '/repo' - }) - const rejection = expect(promise).rejects.toThrow('gh timed out.') - await vi.advanceTimersByTimeAsync(30_000) - - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) - it('rejects glab executions that never call back using the default timeout', async () => { + it('signals the whole glab process group when glab never calls back', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const promise = glabExecFileAsync(['api', 'projects/stablyai%2Forca/issues'], { + cwd: '/repo' + }) + const rejection = expect(promise).rejects.toThrow('glab timed out.') + await vi.advanceTimersByTimeAsync(30_000) + await vi.advanceTimersByTimeAsync(2_000) - const promise = glabExecFileAsync(['api', 'projects/stablyai%2Forca/issues'], { - cwd: '/repo' - }) - const rejection = expect(promise).rejects.toThrow('glab timed out.') - await vi.advanceTimersByTimeAsync(30_000) - - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) it('aborts glab retry backoff instead of starting another attempt', async () => { @@ -304,9 +343,14 @@ describe('runner execFile timeout handling', () => { const transient = Object.assign(new Error('glab failed'), { stderr: 'HTTP 503 Service Unavailable' }) - execFileMock.mockImplementationOnce((_command, _args, _options, callback) => { - callback(transient) - return createMockChildProcess(1234) + spawnMock.mockImplementationOnce(() => { + const child = createMockChildProcess(1234) + queueMicrotask(() => { + child.stderr.emit('data', Buffer.from(transient.stderr)) + child.emit('exit', 1, null) + child.emit('close', 1, null) + }) + return child }) const promise = glabExecFileAsync(['api', 'projects'], { @@ -314,52 +358,68 @@ describe('runner execFile timeout handling', () => { signal: controller.signal }) const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) - await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledTimes(1)) + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledTimes(1)) controller.abort() await rejection - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('kills an active gh execution when its caller aborts', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) - const controller = new AbortController() - const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { - cwd: '/repo', - signal: controller.signal - }) - const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const controller = new AbortController() + const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { + cwd: '/repo', + signal: controller.signal + }) + const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) - controller.abort() + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalled()) + controller.abort() + await vi.advanceTimersByTimeAsync(2_000) - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) it('honors explicit gh timeouts', async () => { const child = createMockChildProcess(1234) - execFileMock.mockReturnValue(child) + mockWedgedCliSpawn(child) + const processKill = mockProcessGroupSignals() + try { + const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { + cwd: '/repo', + timeout: 1234 + }) + const rejection = expect(promise).rejects.toThrow('gh timed out.') + await vi.advanceTimersByTimeAsync(1233) + expect(processKill).not.toHaveBeenCalled() + await vi.advanceTimersByTimeAsync(1) + await vi.advanceTimersByTimeAsync(2_000) - const promise = ghExecFileAsync(['api', 'repos/stablyai/orca/issues/5388'], { - cwd: '/repo', - timeout: 1234 - }) - const rejection = expect(promise).rejects.toThrow('gh timed out.') - await vi.advanceTimersByTimeAsync(1233) - expect(child.kill).not.toHaveBeenCalled() - await vi.advanceTimersByTimeAsync(1) - - await rejection - expect(child.kill).toHaveBeenCalled() + await rejection + expect(processKill).toHaveBeenCalledWith(-1234, undefined) + } finally { + processKill.mockRestore() + } }) it('runs gh non-interactively while preserving explicit env', async () => { - const child = createMockChildProcess(1234) let capturedEnv: NodeJS.ProcessEnv | undefined - execFileMock.mockImplementation((_cmd, _args, opts, cb) => { + spawnMock.mockImplementation((_cmd, _args, opts) => { capturedEnv = opts.env - cb(null, 'ok', '') + const child = createMockChildProcess(1234) + queueMicrotask(() => { + child.stdout.emit('data', Buffer.from('ok')) + child.emit('exit', 0, null) + child.emit('close', 0, null) + }) return child }) diff --git a/src/main/git/runner-gh-rate-limit-breaker.test.ts b/src/main/git/runner-gh-rate-limit-breaker.test.ts index 17e14b8e7d6..e450afd7939 100644 --- a/src/main/git/runner-gh-rate-limit-breaker.test.ts +++ b/src/main/git/runner-gh-rate-limit-breaker.test.ts @@ -12,6 +12,7 @@ vi.mock('child_process', () => ({ spawn: spawnMock })) +import { fakeSpawnReturning } from '../../shared/child-process/__fixtures__/fake-spawned-child' import { ghExecFileAsync } from './runner' import { _resetGhRateLimitBreaker, @@ -23,25 +24,16 @@ const PRIMARY_RATE_LIMIT_STDERR = 'gh: API rate limit exceeded for user ID 1775218. Please wait. (HTTP 403)' function mockGhFailure(stderr: string): void { - execFileMock.mockImplementation((_binary, _args, options, callback) => { - const done = typeof options === 'function' ? options : callback - queueMicrotask(() => - done(Object.assign(new Error(`Command failed: gh\n${stderr}`), { stderr }), '', stderr) - ) - return { once: vi.fn() } - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr, code: 1 })) } function mockGhSuccess(stdout: string): void { - execFileMock.mockImplementation((_binary, _args, options, callback) => { - const done = typeof options === 'function' ? options : callback - queueMicrotask(() => done(null, stdout, '')) - return { once: vi.fn() } - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stdout })) } beforeEach(() => { execFileMock.mockReset() + spawnMock.mockReset() }) afterEach(() => { @@ -54,14 +46,14 @@ describe('ghExecFileAsync rate-limit breaker', () => { await expect( ghExecFileAsync(['api', '--cache', '120s', 'search/issues?q=repo:a/b&per_page=1']) ).rejects.toThrow('rate limit') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) // The 90-repo storm case: every further search-bucket call must fail fast // without a subprocess. await expect( ghExecFileAsync(['api', '--cache', '120s', 'search/issues?q=repo:c/d&per_page=1']) ).rejects.toMatchObject({ ghRateLimitBlocked: true }) - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('keeps other buckets working while one bucket is blocked', async () => { @@ -72,7 +64,7 @@ describe('ghExecFileAsync rate-limit breaker', () => { stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('keeps other GitHub hosts and WSL runtimes working when github.com is blocked', async () => { @@ -105,7 +97,7 @@ describe('ghExecFileAsync rate-limit breaker', () => { value: originalPlatform }) } - expect(execFileMock).toHaveBeenCalledTimes(5) + expect(spawnMock).toHaveBeenCalledTimes(5) }) it.each([ @@ -195,7 +187,7 @@ describe('ghExecFileAsync rate-limit breaker', () => { ).resolves.toMatchObject({ stdout: '{"resources":{}}' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('does not trip the breaker on secondary rate limits', async () => { diff --git a/src/main/git/runner-wsl-gh-fallback.test.ts b/src/main/git/runner-wsl-gh-fallback.test.ts index 5f933c60620..9f4566f56ae 100644 --- a/src/main/git/runner-wsl-gh-fallback.test.ts +++ b/src/main/git/runner-wsl-gh-fallback.test.ts @@ -1,16 +1,19 @@ -import { EventEmitter } from 'node:events' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + createFakeSpawnedChild, + fakeSpawnDispatch, + fakeSpawnReturning +} from '../../shared/child-process/__fixtures__/fake-spawned-child' import type * as WslModule from '../wsl' -const { execFileMock, execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ - execFileMock: vi.fn(), +const { execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ execFileSyncMock: vi.fn(), spawnMock: vi.fn(), getDefaultWslDistroMock: vi.fn() })) vi.mock('child_process', () => ({ - execFile: execFileMock, + execFile: vi.fn(), execFileSync: execFileSyncMock, spawn: spawnMock })) @@ -26,25 +29,18 @@ import { _resetGhRateLimitBreaker } from './gh-rate-limit-breaker' const PRIMARY_RATE_LIMIT_STDERR = 'gh: API rate limit exceeded for user ID 1775218. Please wait. (HTTP 403)' -type MockChildProcess = EventEmitter & { - pid: number - kill: ReturnType - unref: ReturnType -} +// What the distro prints when the CLI is absent inside WSL but present on the host. +const WSL_GH_MISSING = 'bash: line 1: gh: command not found\n' +const TRANSIENT_502 = 'HTTP 502 Bad Gateway' -function createMockChildProcess(pid: number): MockChildProcess { - const child = new EventEmitter() as MockChildProcess - child.pid = pid - child.kill = vi.fn() - child.unref = vi.fn() - return child +function spawnEnoent(command: string): { spawnError: Error } { + return { spawnError: Object.assign(new Error(`spawn ${command} ENOENT`), { code: 'ENOENT' }) } } describe('ghExecFileAsync WSL fallback', () => { const originalPlatform = process.platform beforeEach(() => { - execFileMock.mockReset() spawnMock.mockReset() getDefaultWslDistroMock.mockReset() getDefaultWslDistroMock.mockReturnValue(null) @@ -66,21 +62,11 @@ describe('ghExecFileAsync WSL fallback', () => { }) it('falls back to host gh for explicit-repo WSL calls when gh is missing in the distro', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '[]' } + ) + ) await expect( ghExecFileAsync(['issue', 'list', '--repo', 'stablyhq/noqa', '--json', 'number,title'], { @@ -88,7 +74,7 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 1, 'wsl.exe', [ @@ -102,53 +88,34 @@ describe('ghExecFileAsync WSL fallback', () => { // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. The Linux directory still rides inside the command. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['issue', 'list', '--repo', 'stablyhq/noqa', '--json', 'number,title'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('does not fall back for repo-context gh calls without explicit repo context', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: WSL_GH_MISSING, code: 1 })) await expect( ghExecFileAsync(['issue', 'list'], { cwd: String.raw`\\wsl.localhost\Ubuntu\home\jinwoo\stably\noqa` }) - ).rejects.toThrow('Command failed: wsl.exe') + ).rejects.toThrow('gh: command not found') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('falls back for short-form explicit repo flags used by gh', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '[]' } + ) + ) await expect( ghExecFileAsync(['issue', 'list', '-R', 'stablyhq/noqa'], { @@ -156,31 +123,20 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['issue', 'list', '-R', 'stablyhq/noqa'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('falls back for compact short-form repo flags used by gh', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '[]' } + ) + ) await expect( ghExecFileAsync(['issue', 'list', '-Rstablyhq/noqa'], { @@ -188,31 +144,20 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['issue', 'list', '-Rstablyhq/noqa'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('falls back for repo view with an explicit positional repository', async () => { - execFileMock.mockImplementation((binary, _args, options, callback) => { - if (typeof options === 'function') { - callback = options - } - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback(null, { stdout: '{"isFork":false}', stderr: '' }) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' ? { stderr: WSL_GH_MISSING, code: 1 } : { stdout: '{"isFork":false}' } + ) + ) await expect( ghExecFileAsync( @@ -224,68 +169,42 @@ describe('ghExecFileAsync WSL fallback', () => { ) ).resolves.toEqual({ stdout: '{"isFork":false}', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'gh', ['repo', 'view', 'github.acme-corp.com/stablyhq/noqa', '--json', 'isFork,parent'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('does not fall back for gh api calls that depend on repo-context placeholders', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: WSL_GH_MISSING, code: 1 })) await expect( ghExecFileAsync(['api', 'repos/stablyhq/noqa/branches/{branch}'], { cwd: String.raw`\\wsl.localhost\Ubuntu\home\jinwoo\stably\noqa` }) - ).rejects.toThrow('Command failed: wsl.exe') + ).rejects.toThrow('gh: command not found') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('retries idempotent gh GraphQL query transient failures', async () => { - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '{"data":{}}', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '{"data":{}}' })) await expect( ghExecFileAsync(['api', 'graphql', '-f', 'query=query { viewer { login } }']) ).resolves.toEqual({ stdout: '{"data":{}}', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('retries a host-pinned idempotent gh GraphQL query after host injection', async () => { - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '{"data":{}}', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '{"data":{}}' })) await expect( ghExecFileAsync(['api', 'graphql', '-f', 'query=query { viewer { login } }'], { @@ -293,8 +212,8 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '{"data":{}}', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenNthCalledWith( 1, 'gh', [ @@ -305,37 +224,22 @@ describe('ghExecFileAsync WSL fallback', () => { '-f', 'query=query { viewer { login } }' ], - expect.any(Object), - expect.any(Function) + expect.any(Object) ) }) it('does not retry non-idempotent gh API transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( ghExecFileAsync(['api', '-X', 'POST', 'repos/stablyai/orca/issues']) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('does not retry gh GraphQL mutation transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( ghExecFileAsync([ @@ -346,42 +250,31 @@ describe('ghExecFileAsync WSL fallback', () => { ]) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('does not retry high-level gh edit transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( ghExecFileAsync(['issue', 'edit', '5', '--repo', 'stablyai/orca']) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('retries cwd-less gh calls through the default WSL distro when host gh is missing', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '{"resources":{}}', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('gh'))) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '{"resources":{}}' })) await expect(ghExecFileAsync(['api', 'rate_limit'])).resolves.toEqual({ stdout: '{"resources":{}}', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'gh' 'api' 'rate_limit'"], @@ -389,53 +282,35 @@ describe('ghExecFileAsync WSL fallback', () => { // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. This global call has no repo directory at all, so nothing about // where it runs changes. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) }) it('checks a blocked WSL scope before repeating a native-to-WSL fallback', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock.mockImplementation((binary, _args, _options, callback) => { - if (binary === 'gh') { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT', stderr: '' })) - return - } - callback( - Object.assign(new Error(PRIMARY_RATE_LIMIT_STDERR), { - stdout: '', - stderr: PRIMARY_RATE_LIMIT_STDERR - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'gh' ? spawnEnoent('gh') : { stderr: PRIMARY_RATE_LIMIT_STDERR, code: 1 } ) - }) + ) await expect(ghExecFileAsync(['api', 'repos/acme/widgets/pulls'])).rejects.toThrow('rate limit') await expect(ghExecFileAsync(['api', 'repos/acme/widgets/pulls'])).rejects.toMatchObject({ ghRateLimitBlocked: true }) - expect(execFileMock).toHaveBeenCalledTimes(3) - expect(execFileMock.mock.calls.map(([binary]) => binary)).toEqual(['gh', 'wsl.exe', 'gh']) + expect(spawnMock).toHaveBeenCalledTimes(3) + expect(spawnMock.mock.calls.map(([binary]) => binary)).toEqual(['gh', 'wsl.exe', 'gh']) }) it('checks a blocked native scope before repeating a WSL-to-native fallback', async () => { - execFileMock.mockImplementation((binary, _args, _options, callback) => { - if (binary === 'wsl.exe') { - callback( - Object.assign(new Error('Command failed: wsl.exe'), { - stdout: '', - stderr: 'bash: line 1: gh: command not found\n' - }) - ) - return - } - callback( - Object.assign(new Error(PRIMARY_RATE_LIMIT_STDERR), { - stdout: '', - stderr: PRIMARY_RATE_LIMIT_STDERR - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' + ? { stderr: WSL_GH_MISSING, code: 1 } + : { stderr: PRIMARY_RATE_LIMIT_STDERR, code: 1 } ) - }) + ) const options = { cwd: String.raw`\\wsl.localhost\Ubuntu\home\jinwoo\stably\noqa` @@ -447,19 +322,12 @@ describe('ghExecFileAsync WSL fallback', () => { ghExecFileAsync(['api', 'repos/acme/widgets/pulls'], options) ).rejects.toMatchObject({ ghRateLimitBlocked: true }) - expect(execFileMock).toHaveBeenCalledTimes(3) - expect(execFileMock.mock.calls.map(([binary]) => binary)).toEqual(['wsl.exe', 'gh', 'wsl.exe']) + expect(spawnMock).toHaveBeenCalledTimes(3) + expect(spawnMock.mock.calls.map(([binary]) => binary)).toEqual(['wsl.exe', 'gh', 'wsl.exe']) }) it('does not retry non-idempotent glab transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( glabExecFileAsync(['api', '-X', 'POST', 'projects/stablyai%2Forca/issues/5/notes'], { @@ -467,18 +335,11 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('does not retry high-level glab update transient failures', async () => { - execFileMock.mockImplementation((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) + spawnMock.mockImplementation(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) await expect( glabExecFileAsync(['issue', 'update', '5', '-R', 'stablyai/orca'], { @@ -486,25 +347,21 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).rejects.toThrow('HTTP 502 Bad Gateway') - expect(execFileMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledTimes(1) }) it('retries cwd-less glab calls through the default WSL distro when host glab is missing', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '[]' })) await expect(glabExecFileAsync(['api', 'projects'])).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'api' 'projects'"], @@ -512,24 +369,21 @@ describe('ghExecFileAsync WSL fallback', () => { // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. This global call has no repo directory at all, so nothing about // where it runs changes. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) }) it('times out the default-WSL glab fallback and waits for full tree cleanup', async () => { vi.useFakeTimers() getDefaultWslDistroMock.mockReturnValue('Ubuntu') - const nativeChild = createMockChildProcess(1200) - const wslChild = createMockChildProcess(2400) - const taskkill = createMockChildProcess(3600) - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - return nativeChild - }) - .mockReturnValueOnce(wslChild) - spawnMock.mockReturnValue(taskkill) + const wslChild = createFakeSpawnedChild(2400) + const taskkill = createFakeSpawnedChild(3600) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + // Why a child that never exits: this is the wedged WSL helper the deadline + // has to reap, so nothing must settle the promise before taskkill reports. + .mockImplementationOnce(() => wslChild) + .mockImplementation(() => taskkill) const promise = glabExecFileAsync(['auth', 'status'], { timeout: 1000 }) const rejection = expect(promise).rejects.toThrow('wsl.exe timed out.') @@ -539,7 +393,7 @@ describe('ghExecFileAsync WSL fallback', () => { }) await vi.advanceTimersByTimeAsync(999) - expect(spawnMock).not.toHaveBeenCalled() + expect(spawnMock).toHaveBeenCalledTimes(2) await vi.advanceTimersByTimeAsync(1) expect(spawnMock).toHaveBeenCalledWith( 'taskkill', @@ -556,29 +410,24 @@ describe('ghExecFileAsync WSL fallback', () => { it('aborts the default-WSL glab fallback with full process-tree cleanup', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - const nativeChild = createMockChildProcess(1200) - const wslChild = createMockChildProcess(2400) - const taskkill = createMockChildProcess(3600) - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - return nativeChild - }) - .mockReturnValueOnce(wslChild) - spawnMock.mockReturnValue(taskkill) + const wslChild = createFakeSpawnedChild(2400) + const taskkill = createFakeSpawnedChild(3600) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + .mockImplementationOnce(() => wslChild) + .mockImplementation(() => taskkill) const controller = new AbortController() const promise = glabExecFileAsync(['auth', 'status'], { signal: controller.signal }) const rejection = expect(promise).rejects.toMatchObject({ name: 'AbortError' }) - await vi.waitFor(() => expect(execFileMock).toHaveBeenCalledTimes(2)) + await vi.waitFor(() => expect(spawnMock).toHaveBeenCalledTimes(2)) controller.abort() - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'auth' 'status'"], - expect.not.objectContaining({ signal: controller.signal }), - expect.any(Function) + expect.not.objectContaining({ signal: controller.signal }) ) expect(spawnMock).toHaveBeenCalledWith( 'taskkill', @@ -593,40 +442,26 @@ describe('ghExecFileAsync WSL fallback', () => { it('does not wake the default WSL distro for host-only GitLab diagnostics', async () => { getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: 'Logged in to gitlab.com', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('glab'))) + .mockImplementationOnce(fakeSpawnReturning({ stdout: 'Logged in to gitlab.com' })) await expect( glabExecFileAsync(['auth', 'status'], { allowDefaultWslFallback: false }) ).rejects.toThrow('spawn glab ENOENT') - expect(execFileMock).toHaveBeenCalledTimes(1) - expect(execFileMock).toHaveBeenCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledWith( 'glab', ['auth', 'status'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) it('still retries idempotent glab transient failures', async () => { - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback( - Object.assign(new Error('HTTP 502 Bad Gateway'), { - stdout: '', - stderr: 'HTTP 502 Bad Gateway' - }) - ) - }) - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(null, { stdout: '[]', stderr: '' }) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning({ stderr: TRANSIENT_502, code: 1 })) + .mockImplementationOnce(fakeSpawnReturning({ stdout: '[]' })) await expect( glabExecFileAsync(['api', 'projects/stablyai%2Forca/issues'], { @@ -634,7 +469,7 @@ describe('ghExecFileAsync WSL fallback', () => { }) ).resolves.toEqual({ stdout: '[]', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenCalledTimes(2) }) it('resolves fallback to the overridden distro if configured, and falls back to default WSL distro otherwise', async () => { @@ -642,60 +477,54 @@ describe('ghExecFileAsync WSL fallback', () => { setDefaultWslDistroOverride('Debian') getDefaultWslDistroMock.mockReturnValue('Ubuntu') - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((binary, args, _options, callback) => { - if (binary === 'wsl.exe' && args.includes('Debian')) { - callback(null, { stdout: 'Logged in to github.com as override', stderr: '' }) - return - } - callback(new Error('Wrong distro fallback')) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('gh'))) + .mockImplementationOnce( + fakeSpawnDispatch((program, args) => + program === 'wsl.exe' && args.includes('Debian') + ? { stdout: 'Logged in to github.com as override' } + : { stderr: 'Wrong distro fallback', code: 1 } + ) + ) await expect(ghExecFileAsync(['auth', 'status'])).resolves.toEqual({ stdout: 'Logged in to github.com as override', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Debian', '--exec', 'bash', '-c', "'gh' 'auth' 'status'"], - expect.any(Object), - expect.any(Function) + expect.any(Object) ) // 2) Test without override (should use default 'Ubuntu') - execFileMock.mockClear() + spawnMock.mockClear() setDefaultWslDistroOverride(null) - execFileMock - .mockImplementationOnce((_binary, _args, _options, callback) => { - callback(Object.assign(new Error('spawn gh ENOENT'), { code: 'ENOENT' })) - }) - .mockImplementationOnce((binary, args, _options, callback) => { - if (binary === 'wsl.exe' && args.includes('Ubuntu')) { - callback(null, { stdout: 'Logged in to github.com as default', stderr: '' }) - return - } - callback(new Error('Wrong distro fallback')) - }) + spawnMock + .mockImplementationOnce(fakeSpawnReturning(spawnEnoent('gh'))) + .mockImplementationOnce( + fakeSpawnDispatch((program, args) => + program === 'wsl.exe' && args.includes('Ubuntu') + ? { stdout: 'Logged in to github.com as default' } + : { stderr: 'Wrong distro fallback', code: 1 } + ) + ) await expect(ghExecFileAsync(['auth', 'status'])).resolves.toEqual({ stdout: 'Logged in to github.com as default', stderr: '' }) - expect(execFileMock).toHaveBeenCalledTimes(2) - expect(execFileMock).toHaveBeenNthCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(2) + expect(spawnMock).toHaveBeenNthCalledWith( 2, 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'gh' 'auth' 'status'"], - expect.any(Object), - expect.any(Function) + expect.any(Object) ) }) }) diff --git a/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts b/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts index 40da88f0c91..f9f3c573629 100644 --- a/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts +++ b/src/main/gitlab/gitlab-known-host-probe-wsl-fallback.test.ts @@ -1,15 +1,15 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { fakeSpawnDispatch } from '../../shared/child-process/__fixtures__/fake-spawned-child' import type * as WslModule from '../wsl' -const { execFileMock, execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ - execFileMock: vi.fn(), +const { execFileSyncMock, spawnMock, getDefaultWslDistroMock } = vi.hoisted(() => ({ execFileSyncMock: vi.fn(), spawnMock: vi.fn(), getDefaultWslDistroMock: vi.fn() })) vi.mock('child_process', () => ({ - execFile: execFileMock, + execFile: vi.fn(), execFileSync: execFileSyncMock, spawn: spawnMock })) @@ -26,17 +26,16 @@ describe('glab known-hosts probe on Windows', () => { const originalPlatform = process.platform const hostGlabMissingWslLoggedIn = (): void => { - execFileMock.mockImplementation((binary, _args, _options, callback) => { - if (binary === 'wsl.exe') { - callback(null, { stdout: 'Logged in to gitlab.wsl.test as user', stderr: '' }) - return - } - callback(Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' })) - }) + spawnMock.mockImplementation( + fakeSpawnDispatch((program) => + program === 'wsl.exe' + ? { stdout: 'Logged in to gitlab.wsl.test as user' } + : { spawnError: Object.assign(new Error('spawn glab ENOENT'), { code: 'ENOENT' }) } + ) + ) } beforeEach(() => { - execFileMock.mockReset() spawnMock.mockReset() getDefaultWslDistroMock.mockReset() getDefaultWslDistroMock.mockReturnValue('Ubuntu') @@ -56,12 +55,11 @@ describe('glab known-hosts probe on Windows', () => { await expect(getGlabKnownHosts()).resolves.toEqual(['gitlab.com']) - expect(execFileMock).toHaveBeenCalledTimes(1) - expect(execFileMock).toHaveBeenCalledWith( + expect(spawnMock).toHaveBeenCalledTimes(1) + expect(spawnMock).toHaveBeenCalledWith( 'glab', ['auth', 'status'], - expect.objectContaining({ cwd: undefined }), - expect.any(Function) + expect.objectContaining({ cwd: undefined }) ) }) @@ -73,15 +71,14 @@ describe('glab known-hosts probe on Windows', () => { await expect(getGlabKnownHosts('conn-1')).resolves.toEqual(['gitlab.com', 'gitlab.wsl.test']) - expect(execFileMock).toHaveBeenCalledWith( + expect(spawnMock).toHaveBeenCalledWith( 'wsl.exe', ['-d', 'Ubuntu', '--exec', 'bash', '-c', "'glab' 'auth' 'status'"], // Why a concrete directory (#16463): `undefined` makes CreateProcessW inherit // Orca's own cwd, a deletable WSL UNC path when it was launched from a // worktree. This probe has no repo directory at all, so nothing about where // it runs changes. The native `glab` assertion above keeps `undefined`. - expect.objectContaining({ cwd: expect.any(String) }), - expect.any(Function) + expect.objectContaining({ cwd: expect.any(String) }) ) }) }) diff --git a/src/shared/child-process/__fixtures__/fake-spawned-child.ts b/src/shared/child-process/__fixtures__/fake-spawned-child.ts new file mode 100644 index 00000000000..7183bc12e07 --- /dev/null +++ b/src/shared/child-process/__fixtures__/fake-spawned-child.ts @@ -0,0 +1,75 @@ +import { EventEmitter } from 'node:events' +import type { ChildProcess } from 'node:child_process' +import { vi } from 'vitest' + +/** + * A `child_process.spawn` stand-in for suites that drive gh/glab/git runners. + * + * Those runners capture output through `runProcess`, which reads the streams + * and waits for `close`, so a bare EventEmitter is not enough — a test child + * has to carry stdio and report an exit or the promise never settles. + */ +export function createFakeSpawnedChild(pid = 4321): ChildProcess { + const child = new EventEmitter() as EventEmitter & Record + child.pid = pid + child.kill = vi.fn(() => true) + child.stdin = Object.assign(new EventEmitter(), { end: vi.fn() }) + child.stdout = new EventEmitter() + child.stderr = new EventEmitter() + return child as unknown as ChildProcess +} + +/** Emit output and a clean exit, the way a CLI that answered would. */ +export function completeFakeSpawn( + child: ChildProcess, + result: { stdout?: string; stderr?: string; code?: number } = {} +): void { + if (result.stdout) { + child.stdout?.emit('data', Buffer.from(result.stdout)) + } + if (result.stderr) { + child.stderr?.emit('data', Buffer.from(result.stderr)) + } + const code = result.code ?? 0 + child.emit('exit', code, null) + child.emit('close', code, null) +} + +/** What a faked spawn should do: answer, or fail to start at all. */ +export type FakeSpawnOutcome = + | { stdout?: string; stderr?: string; code?: number } + | { spawnError: Error } + +function settleFakeSpawn(child: ChildProcess, outcome: FakeSpawnOutcome): void { + if ('spawnError' in outcome) { + // Why an event and not a throw: an unresolvable program fails asynchronously + // in libuv, which is what makes ENOENT reach callers as a rejection. + child.emit('error', outcome.spawnError) + return + } + completeFakeSpawn(child, outcome) +} + +/** + * Build a `spawn` implementation that answers every call the same way. + * + * Why a fresh child per call: the runners retry and fall back, and a shared + * emitter would replay the first call's exit into the second's listeners. + */ +export function fakeSpawnReturning( + outcome: FakeSpawnOutcome = {} +): (program: string, args: readonly string[]) => ChildProcess { + return fakeSpawnDispatch(() => outcome) +} + +/** Build a `spawn` implementation that answers per invoked program and argv. */ +export function fakeSpawnDispatch( + resolve: (program: string, args: readonly string[]) => FakeSpawnOutcome +): (program: string, args: readonly string[]) => ChildProcess { + return (program, args) => { + const child = createFakeSpawnedChild() + const outcome = resolve(program, args) + queueMicrotask(() => settleFakeSpawn(child, outcome)) + return child + } +} diff --git a/src/shared/child-process/bounded-output-sink.ts b/src/shared/child-process/bounded-output-sink.ts index c231234ef1a..195e466fdf6 100644 --- a/src/shared/child-process/bounded-output-sink.ts +++ b/src/shared/child-process/bounded-output-sink.ts @@ -10,6 +10,7 @@ import { Buffer } from 'node:buffer' export function createOutputSink(maxBytes: number): { write: (chunk: Buffer | string) => void text: () => string + truncated: () => boolean } { const chunks: Buffer[] = [] let bytes = 0 @@ -18,11 +19,15 @@ export function createOutputSink(maxBytes: number): { const chunk = Buffer.isBuffer(raw) ? raw : Buffer.from(raw) const remaining = maxBytes - bytes if (remaining <= 0) { + bytes += chunk.length return } chunks.push(chunk.length > remaining ? chunk.subarray(0, remaining) : chunk) bytes += chunk.length }, - text: () => Buffer.concat(chunks).toString('utf8') + text: () => Buffer.concat(chunks).toString('utf8'), + // Why: callers that parse the output need to tell a short answer from a + // clipped one -- truncated JSON or JSONL parses as a smaller valid result. + truncated: () => bytes > maxBytes } } diff --git a/src/shared/child-process/process-spec.ts b/src/shared/child-process/process-spec.ts index ac705974efe..2acfe82d61d 100644 --- a/src/shared/child-process/process-spec.ts +++ b/src/shared/child-process/process-spec.ts @@ -65,6 +65,8 @@ export type ProcessResult = { stderr: string /** True when the process was killed by `timeoutMs` rather than exiting. */ timedOut: boolean + /** True when stdout or stderr exceeded `maxOutputBytes` and was clipped. */ + outputTruncated?: boolean } export const DEFAULT_PROCESS_TIMEOUT_MS = 30_000 diff --git a/src/shared/child-process/run-process.test.ts b/src/shared/child-process/run-process.test.ts index 5fad9b5dc70..d36de3c688a 100644 --- a/src/shared/child-process/run-process.test.ts +++ b/src/shared/child-process/run-process.test.ts @@ -99,6 +99,27 @@ describe('runProcessSync', () => { }) }) +describe('bounded output', () => { + it('reports a clipped answer instead of passing it off as the whole one', async () => { + const result = await runProcess({ + program: process.execPath, + args: ['-e', 'process.stdout.write("x".repeat(64))'], + maxOutputBytes: 8 + }) + expect(result.stdout).toBe('xxxxxxxx') + expect(result.outputTruncated).toBe(true) + }) + + it('does not call output that exactly fills the cap truncated', async () => { + const result = await runProcess({ + program: process.execPath, + args: ['-e', 'process.stdout.write("x".repeat(8))'], + maxOutputBytes: 8 + }) + expect(result.outputTruncated).toBe(false) + }) +}) + describe('unkillable children', () => { it('settles after the grace period rather than outliving its own deadline', async () => { // `close` only fires once the child is gone, so a child that ignores the diff --git a/src/shared/child-process/run-process.ts b/src/shared/child-process/run-process.ts index ec83c5beed0..67bd48f5b81 100644 --- a/src/shared/child-process/run-process.ts +++ b/src/shared/child-process/run-process.ts @@ -186,7 +186,14 @@ export function runProcess(spec: ProcessSpec): Promise { const resolveFromClose = (code: number | null, signal: NodeJS.Signals | null): void => settle(() => - resolve({ code, signal, stdout: stdout.text(), stderr: stderr.text(), timedOut }) + resolve({ + code, + signal, + stdout: stdout.text(), + stderr: stderr.text(), + timedOut, + outputTruncated: stdout.truncated() || stderr.truncated() + }) ) const settleBarrierOutcome = (): void => { @@ -381,6 +388,9 @@ export function runProcessSync(spec: ProcessSpec): ProcessResult { signal: result.signal, stdout: result.stdout?.toString('utf8') ?? '', stderr: result.stderr?.toString('utf8') ?? '', + // Why always false: spawnSync reports an overrun as an ENOBUFS error, and + // the guard above rethrows it, so no truncated result reaches this point. + outputTruncated: false, // Why ETIMEDOUT and not the signal: a timeout kills with SIGTERM, but so // does anything else that terminates the child, and only a timeout also // sets this error. Reading the signal alone reports a deliberately From 31007c0d86cb4d5c2ab6f8621621bf8bf1800eb4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:14 -0700 Subject: [PATCH 111/398] fix(ssh): reclaim relay PTYs the client has provably lost, on host attestation only (#17831) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): reclaim relay PTYs the host attests this client orphaned (#9819) Orca could lose track of terminals running on an SSH relay until the 50-slot cap refused to open any more. This reclaims them, and the whole design is built around the fact that getting it wrong destroys a user's running process on their remote machine: the failure mode is leak, never kill. A stop requires all nine of: 1. the relay published an `ownerClientInstanceId` read from the live authenticated consumer grant of the connection that requested the spawn — never from a spawn parameter, since an echoed claim is no evidence; absent means skip 2. that id equals this client's persisted consumer identity 3. this connection holds the negotiated `session-owner` grant 4. `paneBound === true`, host-published 5. no `agentSessionOwners` — the host still advertises it as adoptable 6. `hostAgeMs >= 30s`, measured on the host's clock 7. this client has no route: not reattached, no lease outside terminated/expired, no pending kill, and no `expired` lease either — an expired lease is the record of a process deliberately left running, never a licence to kill it 8. every stop is fenced on the incarnation the same listing published, and on the owner identity, both re-checked by the host 9. a pass wanting to stop more than 8 refuses entirely Absence from a client-side set is `unverifiable` by construction (docs/reference/ssh-execution-boundary.md): a second machine attaches to the same relay and displaces the session owner, and its live agents are missing from this client's store for exactly the reason a genuine orphan is. So the host has to attest ownership, and the host has to attest that nothing is running. That second attestation is measured over the pane's whole tty, not its foreground process group. `tpgid == pgid` is foreground-only: on a real `bash -i` on a real pty, a shell holding `sleep 300 &` and a shell holding a Ctrl-Z'd job both read `pgid == tpgid`, `Ss+` — byte-identical to an idle prompt, with only the job's own row differing. A foreground-only gate therefore attests `pnpm build &` and a suspended editor as idle, and the stop that follows SIGKILLs every process group on the tty. `shellOwnsEveryTtyProcessGroup` is measured over that same set of groups, so the evidence and the kill describe the same thing. No new probe: `tpgid` already identifies the terminal, because a process group belongs to one session and a session to at most one controlling terminal. The freshness field is real rather than decorative. `capturedAgeMs` is stamped from when the capture was taken, deliberately as an upper bound since the process table is TTL-shared, and the sweep refuses an observation older than its own pass budget, counting its own elapsed time since the listing arrived. Stale evidence degrades to "do not sweep", never to "sweep". The display consumer of the same measurement keeps no age budget, as a stated decision: a stale pane title costs a redraw and self-corrects. `pty.shutdown` is authorized on the host that owns the process. `pty.spawn` and `pty.attach` both take a request context and check it; the one irreversible call took none, so the rule above lived entirely on the client that decided to make the call. It gains an optional `expectedOwnerClientInstanceId` and refuses unless the connection still authenticates as that identity AND this host recorded it at spawn. Finally, a reattach refusal now says whether it observed the process. Three refusals carry the same `SSH_SESSION_EXPIRED` text and only one is absence; `restoreRequired` means the PTY is live and only its source stream is not. Testing that text with `.includes()` expired the lease and deleted ownership for a running process, erasing this client's only record of it — and a PTY with no record is one the sweep may stop. Wire compatibility: four new optional fields and one new optional param on existing methods, no new method and no new stream opcode (Rule 1, and Rule 2 does not apply). Rule 1's caveat is discharged explicitly — no reader requires any of them, each absence is a named skip reason, and an ordinary pane teardown must omit the owner fence because a revived PTY carries no attested owner at all. New client plus old relay stops zero PTYs; old client plus new relay never reads the fields. Windows relay hosts publish no evidence and therefore never sweep. Verified by joining the real publisher to the real client reader over `ps` captured verbatim from a Linux container, and by driving a real group-for-group SIGKILL against a real pty: backgrounded and suspended jobs survive by pid, and an idle shell is still reclaimed, so the narrowed predicate is not a silent no-op. Squashed deliberately. The sweep is unsafe at every intermediate commit of its own history — before the foreground gate it reaps a hand-launched `claude`, and with a foreground-only gate it reaps a backgrounded build — so this ships as one commit with no bisectable state that kills live work. Refs #9819. Folds in #17939. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- .../pty-daemon-spawn-session-identity.test.ts | 67 +++- ...ty-runtime-ssh-binding-persistence.test.ts | 107 +++++- src/main/ipc/pty/ipc/spawn-execute.ts | 16 +- src/main/ipc/pty/runtime/spawn-execute.ts | 16 +- .../agent-foreground-process-batch.test.ts | 23 +- .../agent-foreground-process-batch.ts | 76 ++++- src/main/providers/pty-process-info.ts | 9 + src/main/providers/pty-provider-contract.ts | 6 + src/main/providers/ssh-pty-provider.ts | 8 +- ...ty-reattach-absence-discrimination.test.ts | 75 +++++ .../ssh/ssh-orphan-relay-pty-sweep.test.ts | 310 ++++++++++++++++++ src/main/ssh/ssh-orphan-relay-pty-sweep.ts | 164 +++++++++ ...h-orphan-sweep-pane-state-verdicts.test.ts | 238 ++++++++++++++ .../ssh/ssh-pty-consumer-recovery.test.ts | 15 + src/main/ssh/ssh-pty-consumer-recovery.ts | 9 + .../ssh-relay-session-orphan-sweep.test.ts | 293 +++++++++++++++++ src/main/ssh/ssh-relay-session.ts | 12 + ...handler-inventory-process-evidence.test.ts | 44 ++- .../pty-handler-ownership-attestation.test.ts | 283 ++++++++++++++++ src/relay/pty-handler.ts | 87 ++++- src/relay/relay-runtime-services.ts | 5 + src/relay/ssh-pty-consumer-session-adapter.ts | 6 + src/shared/foreground-process-evidence.ts | 33 +- src/shared/process-table-snapshot.ts | 5 + src/shared/pty-consumer-session.ts | 9 + .../ssh-relay-pty-ownership-proof.test.ts | 265 +++++++++++++++ src/shared/ssh-relay-pty-ownership-proof.ts | 226 +++++++++++++ 27 files changed, 2371 insertions(+), 36 deletions(-) create mode 100644 src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts create mode 100644 src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts create mode 100644 src/main/ssh/ssh-orphan-relay-pty-sweep.ts create mode 100644 src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts create mode 100644 src/main/ssh/ssh-relay-session-orphan-sweep.test.ts create mode 100644 src/relay/pty-handler-ownership-attestation.test.ts create mode 100644 src/shared/ssh-relay-pty-ownership-proof.test.ts create mode 100644 src/shared/ssh-relay-pty-ownership-proof.ts diff --git a/src/main/ipc/pty-daemon-spawn-session-identity.test.ts b/src/main/ipc/pty-daemon-spawn-session-identity.test.ts index b32ad6426f1..90ea56049fa 100644 --- a/src/main/ipc/pty-daemon-spawn-session-identity.test.ts +++ b/src/main/ipc/pty-daemon-spawn-session-identity.test.ts @@ -8,6 +8,7 @@ import { import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { createDaemonActiveProviderFixtures } from './pty-ipc-daemon-provider-fixtures' import { makePaneKey } from '../../shared/stable-pane-id' +import { SshPtyAbsentFromRelayError } from '../providers/ssh-pty-errors' import { registerPtyHandlers, registerSshPtyProvider, @@ -459,7 +460,9 @@ describe('registerPtyHandlers', () => { }) it('marks a caller-supplied SSH session expired when remote reattach is gone', async () => { const sshSpawn = vi.fn(async () => { - throw new Error('SSH_SESSION_EXPIRED: remote-pty') + // The class, not the message: `SSH_SESSION_EXPIRED` is also what a live PTY whose + // source stream needs restoring refuses with, and only this one is host-reported absence. + throw new SshPtyAbsentFromRelayError('SSH_SESSION_EXPIRED: remote-pty') }) const store = { markSshRemotePtyLease: vi.fn(), @@ -509,10 +512,70 @@ describe('registerPtyHandlers', () => { expect(store.markSshRemotePtyLease).toHaveBeenCalledWith('ssh-1', 'remote-pty', 'expired') }) + it('leaves the lease alone when the refusal did not observe the process', async () => { + // A `restoreRequired` reattach is refused with the SAME `SSH_SESSION_EXPIRED` text, and it + // means the opposite: the PTY is live, only its source stream could not be resumed. The + // lease and the in-memory ownership are between them this client's only record that the + // remote process exists, and #9819's sweep reads a PTY it has no record of as one it may + // SIGKILL on the next connect. Erasing them here is how a live shell gets reaped. + const sshSpawn = vi.fn(async () => { + throw new Error('SSH_SESSION_EXPIRED: remote-pty') + }) + const store = { + markSshRemotePtyLease: vi.fn(), + clearSshRemotePtyKillIntent: vi.fn() + } + registerSshPtyProvider('ssh-1', { + spawn: sshSpawn, + write: vi.fn(), + resize: vi.fn(), + shutdown: vi.fn(), + sendSignal: vi.fn(), + getCwd: vi.fn(), + getInitialCwd: vi.fn(), + clearBuffer: vi.fn(), + acknowledgeDataEvent: vi.fn(), + hasChildProcesses: vi.fn(), + getForegroundProcess: vi.fn(), + serialize: vi.fn(), + revive: vi.fn(), + onData: vi.fn(() => () => {}), + onReplay: vi.fn(() => () => {}), + onExit: vi.fn(() => () => {}), + listProcesses: vi.fn(async () => []), + attach: vi.fn(), + getDefaultShell: vi.fn(), + getProfiles: vi.fn() + } as never) + handlers.clear() + registerPtyHandlers( + mainWindow as never, + undefined, + undefined, + undefined, + undefined, + store as never + ) + + await expect( + handlers.get('pty:spawn')!(null, { + cols: 80, + rows: 24, + env: {}, + connectionId: 'ssh-1', + sessionId: 'remote-pty' + }) + ).rejects.toThrow('SSH_SESSION_EXPIRED: remote-pty') + + // The spawn still fails; what must not happen is the destructive bookkeeping. + expect(store.markSshRemotePtyLease).not.toHaveBeenCalled() + }) it('marks a scoped SSH session expired using the raw relay lease id', async () => { const scopedPtyId = 'ssh:ssh-1@@remote-pty' const sshSpawn = vi.fn(async () => { - throw new Error('SSH_SESSION_EXPIRED: remote-pty') + // The class, not the message: `SSH_SESSION_EXPIRED` is also what a live PTY whose + // source stream needs restoring refuses with, and only this one is host-reported absence. + throw new SshPtyAbsentFromRelayError('SSH_SESSION_EXPIRED: remote-pty') }) const store = { markSshRemotePtyLease: vi.fn(), diff --git a/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts b/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts index 708d41c3e08..3fd1f89b11d 100644 --- a/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts +++ b/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts @@ -2,7 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { spawnMock, openCodeClearPtyMock, piClearPtyMock } from './pty-ipc-mock-registry' import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { makePaneKey } from '../../shared/stable-pane-id' -import { SSH_SESSION_EXPIRED_ERROR } from '../providers/ssh-pty-errors' +import { SSH_SESSION_EXPIRED_ERROR, SshPtyAbsentFromRelayError } from '../providers/ssh-pty-errors' import { registerPtyHandlers, registerSshPtyProvider, @@ -446,7 +446,9 @@ describe('registerPtyHandlers', () => { const remoteWrite = vi.fn() registerSshPtyProvider('ssh-expired-runtime', { spawn: vi.fn(async () => { - throw new Error(`${SSH_SESSION_EXPIRED_ERROR}: relay-pty`) + // The class, not the message: `SSH_SESSION_EXPIRED` is also what a live PTY whose source + // stream needs restoring refuses with, and only this one is host-reported absence. + throw new SshPtyAbsentFromRelayError(`${SSH_SESSION_EXPIRED_ERROR}: relay-pty`) }), write: remoteWrite, resize: vi.fn(), @@ -531,4 +533,105 @@ describe('registerPtyHandlers', () => { unregisterSshPtyProvider('ssh-expired-runtime') } }) + it('leaves a runtime-owned lease alone when the refusal did not observe the process', async () => { + // The `restoreRequired` twin of the case above: same `SSH_SESSION_EXPIRED` text, opposite + // meaning — the PTY is live and only its source stream needs rebuilding. Expiring the lease + // and dropping ownership erases this client's only record of a running remote process, and + // #9819's sweep reads a PTY it has no record of as one it may SIGKILL on the next connect. + type RuntimeSpawnController = { + spawn(args: { + cols: number + rows: number + worktreeId?: string + connectionId?: string + tabId?: string + leafId?: string + sessionId?: string + persistHostSessionBinding?: boolean + }): Promise<{ id: string }> + } + const appPtyId = 'ssh:ssh-live-runtime@@relay-pty' + const remoteWrite = vi.fn() + registerSshPtyProvider('ssh-live-runtime', { + spawn: vi.fn(async () => { + throw new Error(`${SSH_SESSION_EXPIRED_ERROR}: relay-pty`) + }), + write: remoteWrite, + resize: vi.fn(), + shutdown: vi.fn(), + sendSignal: vi.fn(), + getCwd: vi.fn(), + getInitialCwd: vi.fn(), + clearBuffer: vi.fn(), + acknowledgeDataEvent: vi.fn(), + onData: vi.fn(() => () => {}), + onReplay: vi.fn(() => () => {}), + onExit: vi.fn(() => () => {}), + listProcesses: vi.fn(), + hasChildProcesses: vi.fn(), + getForegroundProcess: vi.fn(), + serialize: vi.fn(), + revive: vi.fn(), + getDefaultShell: vi.fn(), + getProfiles: vi.fn() + } as never) + const store = { + upsertSshRemotePtyLease: vi.fn(), + persistPtyBinding: vi.fn(), + removeSshRemotePtyLease: vi.fn(), + markSshRemotePtyLease: vi.fn(), + clearSshRemotePtyKillIntent: vi.fn() + } + let controller: RuntimeSpawnController | null = null + const runtime = { + setPtyController: vi.fn((value) => { + controller = value + }), + createPreAllocatedTerminalHandle: vi.fn(() => 'term_remote'), + registerPreAllocatedHandleForPty: vi.fn(), + registerPty: vi.fn(), + noteTerminalSpawnCommand: vi.fn(), + getDriver: vi.fn(() => ({ kind: 'host' })), + onPtySpawned: vi.fn(), + onPtyExit: vi.fn(), + onPtyData: vi.fn() + } + + try { + setPtyOwnership(appPtyId, 'ssh-live-runtime') + registerPtyHandlers( + mainWindow as never, + runtime as never, + undefined, + undefined, + undefined, + store as never + ) + const spawnController = controller as unknown as RuntimeSpawnController + const leafId = '11111111-1111-4111-8111-111111111111' + + await expect( + spawnController.spawn({ + cols: 80, + rows: 24, + connectionId: 'ssh-live-runtime', + worktreeId: 'wt-remote', + tabId: 'tab-remote', + leafId, + sessionId: appPtyId, + persistHostSessionBinding: true + }) + ).rejects.toThrow(SSH_SESSION_EXPIRED_ERROR) + + expect(store.markSshRemotePtyLease).not.toHaveBeenCalled() + expect(store.upsertSshRemotePtyLease).not.toHaveBeenCalled() + expect(store.persistPtyBinding).not.toHaveBeenCalled() + // Still routable: the client kept its handle on a process that is still running. + getPtyWriteListener()(mainWindowIpcEvent, { id: appPtyId, data: 'echo still-here' }) + expect(remoteWrite).toHaveBeenCalledWith(appPtyId, 'echo still-here') + } finally { + deletePtyOwnership(appPtyId) + unregisterSshPtyProvider('ssh-live-runtime') + } + }) }) diff --git a/src/main/ipc/pty/ipc/spawn-execute.ts b/src/main/ipc/pty/ipc/spawn-execute.ts index 58525e162aa..2d9de676c39 100644 --- a/src/main/ipc/pty/ipc/spawn-execute.ts +++ b/src/main/ipc/pty/ipc/spawn-execute.ts @@ -1,6 +1,7 @@ import { ensureWslHookRelayForReattach } from '../../../agent-hooks/wsl-hook-relay-reattach' import { SSH_SESSION_EXPIRED_ERROR, + isSshPtyAbsentFromRelayError, isSshPtyIdentityMismatchError } from '../../../providers/ssh-pty-errors' import { classifyError } from '../../../telemetry/classify-error' @@ -144,6 +145,15 @@ export async function executePtyIpcSpawn(ctx: PtyIpcSpawnState): Promise { Boolean(args.connectionId) && (spawnError.message.includes(SSH_SESSION_EXPIRED_ERROR) || rawMessage.includes(SSH_SESSION_EXPIRED_ERROR)) + // The message alone cannot carry this decision. All three reattach refusals are minted with the + // same `SSH_SESSION_EXPIRED` text, and only one of them observed the process: `restoreRequired` + // means the PTY is LIVE and only its source stream needs rebuilding, which + // `ssh-pty-errors.ts` states outright. Expiring its lease and deleting its ownership erases + // this client's last record of a running remote process, and #9819's sweep reads a PTY it has + // no record of as one it may SIGKILL on the next connect. Only positive host-reported absence + // may reach that bookkeeping; being too strict here merely leaves a dead lease for the next + // reattach to retire on real host evidence. + const relayReportedSessionAbsent = isExpiredSshSession && isSshPtyAbsentFromRelayError(err) const exitedBeforeSpawnReply = ctx.rejectedRegistrationCandidate?.exitedBeforeSpawnReply === true if (ctx.effectiveSessionAppId !== undefined) { @@ -157,7 +167,11 @@ export async function executePtyIpcSpawn(ctx: PtyIpcSpawnState): Promise { ptySizes.delete(ctx.effectiveSessionAppId) } } - if (args.connectionId && ctx.effectiveSessionRelayId !== undefined && isExpiredSshSession) { + if ( + args.connectionId && + ctx.effectiveSessionRelayId !== undefined && + relayReportedSessionAbsent + ) { // Why: expired remote reattach = relay already dropped the PTY; clear the lease so writes can't restore the stale binding. if (ctx.effectiveSessionAppId !== undefined && !isIdentityMismatch) { clearProviderPtyState(ctx.effectiveSessionAppId) diff --git a/src/main/ipc/pty/runtime/spawn-execute.ts b/src/main/ipc/pty/runtime/spawn-execute.ts index 99829f119bb..05a9ca82d15 100644 --- a/src/main/ipc/pty/runtime/spawn-execute.ts +++ b/src/main/ipc/pty/runtime/spawn-execute.ts @@ -13,6 +13,7 @@ import { clearProviderPtyState } from '../provider/state-cleanup' import { isProviderAgentSessionOwnerLive, normalizeNodePtySpawnError } from '../provider/liveness' import { SSH_SESSION_EXPIRED_ERROR, + isSshPtyAbsentFromRelayError, isSshPtyIdentityMismatchError } from '../../../providers/ssh-pty-errors' import type { RuntimePtySpawnState } from './spawn-state' @@ -224,6 +225,15 @@ export async function executeRuntimePtySpawn(ctx: RuntimePtySpawnState): Promise Boolean(args.connectionId) && (spawnError.message.includes(SSH_SESSION_EXPIRED_ERROR) || rawMessage.includes(SSH_SESSION_EXPIRED_ERROR)) + // The message alone cannot carry this decision. All three reattach refusals are minted with the + // same `SSH_SESSION_EXPIRED` text, and only one of them observed the process: `restoreRequired` + // means the PTY is LIVE and only its source stream needs rebuilding, which + // `ssh-pty-errors.ts` states outright. Expiring its lease and deleting its ownership erases + // this client's last record of a running remote process, and #9819's sweep reads a PTY it has + // no record of as one it may SIGKILL on the next connect. Only positive host-reported absence + // may reach that bookkeeping; being too strict here merely leaves a dead lease for the next + // reattach to retire on real host evidence. + const relayReportedSessionAbsent = isExpiredSshSession && isSshPtyAbsentFromRelayError(err) const exitedBeforeSpawnReply = ctx.rejectedRegistrationCandidate?.exitedBeforeSpawnReply === true if (ctx.effectiveSessionAppId !== undefined) { @@ -237,7 +247,11 @@ export async function executeRuntimePtySpawn(ctx: RuntimePtySpawnState): Promise ptySizes.delete(ctx.effectiveSessionAppId) } } - if (args.connectionId && ctx.effectiveSessionRelayId !== undefined && isExpiredSshSession) { + if ( + args.connectionId && + ctx.effectiveSessionRelayId !== undefined && + relayReportedSessionAbsent + ) { if (ctx.effectiveSessionAppId !== undefined && !isIdentityMismatch) { clearProviderPtyState(ctx.effectiveSessionAppId) deletePtyOwnership(ctx.effectiveSessionAppId) diff --git a/src/main/providers/agent-foreground-process-batch.test.ts b/src/main/providers/agent-foreground-process-batch.test.ts index 8baa37a7499..4107a96c2f1 100644 --- a/src/main/providers/agent-foreground-process-batch.test.ts +++ b/src/main/providers/agent-foreground-process-batch.test.ts @@ -22,7 +22,28 @@ describe('batched foreground process correlation', () => { resolveAgentForegroundProcessesFromIndex(buildProcessTableIndex(rows), [ { rootPid: 100, fallbackProcess: 'zsh' } ]) - ).toEqual([{ available: true, processName: 'codex' }]) + ).toEqual([{ available: true, processName: 'codex', shellOwnsEveryTtyProcessGroup: false }]) + }) + + it('reports whether the shell itself owns the terminal, named process or not', () => { + // The only host-observable "nothing is running here". pid 200's own pgid owns the terminal; + // pid 300 has an unrecognized command in the foreground, which nothing else here can see. + const rows = parseStrictProcessTableRows( + [ + '200 1 200 200 Ss /bin/zsh', + '300 1 300 301 Ss /bin/zsh', + '301 300 301 301 S+ vim notes.md' + ].join('\n') + ) + expect( + resolveAgentForegroundProcessesFromIndex(buildProcessTableIndex(rows), [ + { rootPid: 200, fallbackProcess: 'zsh' }, + { rootPid: 300, fallbackProcess: 'zsh' } + ]) + ).toEqual([ + { available: true, processName: null, shellOwnsEveryTtyProcessGroup: true }, + { available: true, processName: null, shellOwnsEveryTtyProcessGroup: false } + ]) }) it('returns unverifiable for a missing root or no controlling tty', () => { diff --git a/src/main/providers/agent-foreground-process-batch.ts b/src/main/providers/agent-foreground-process-batch.ts index 2e0036bb603..e2d9e5a1372 100644 --- a/src/main/providers/agent-foreground-process-batch.ts +++ b/src/main/providers/agent-foreground-process-batch.ts @@ -26,6 +26,9 @@ export type BatchedForegroundProcessResult = { available: boolean processName: string | null reason?: string + /** Set only when the table was readable: every process group attached to this PTY's terminal is + * the shell's own, and none of them is stopped. Left absent when we could not observe it. */ + shellOwnsEveryTtyProcessGroup?: boolean } export type BatchedForegroundProcessOptions = { @@ -34,6 +37,48 @@ export type BatchedForegroundProcessOptions = { stats?: ProcessTableIndexStats } +/** Which process groups occupy each controlling terminal, and which terminals hold a stopped + * process. */ +type TtyOccupancy = { + processGroupsByTty: ReadonlyMap> + stoppedTtys: ReadonlySet +} + +const ttyOccupancyByCapture = new WeakMap() + +/** Index the capture by controlling terminal. + * + * Keyed on `tpgid` because the snapshot carries no tty column and does not need one: a process + * group belongs to exactly one session, a session to at most one controlling terminal, so two + * rows reporting the same live `tpgid` are on the same tty. Memoized per capture, since the + * per-pane cadence poll and `pty.listProcesses` share one TTL-cached table. */ +function getTtyOccupancy(rows: readonly ProcessTableRow[]): TtyOccupancy { + const cached = ttyOccupancyByCapture.get(rows) + if (cached) { + return cached + } + const processGroupsByTty = new Map>() + const stoppedTtys = new Set() + for (const row of rows) { + if (row.pgid === undefined || row.tpgid === undefined || row.tpgid <= 0) { + continue + } + let groups = processGroupsByTty.get(row.tpgid) + if (!groups) { + groups = new Set() + processGroupsByTty.set(row.tpgid, groups) + } + groups.add(row.pgid) + // `T` is a job-control stop (Ctrl-Z), `t` a tracing stop. Both are work the pane still holds. + if (row.stat.startsWith('T') || row.stat.startsWith('t')) { + stoppedTtys.add(row.tpgid) + } + } + const occupancy: TtyOccupancy = { processGroupsByTty, stoppedTtys } + ttyOccupancyByCapture.set(rows, occupancy) + return occupancy +} + export async function resolveAgentForegroundProcessesBatch( requests: readonly BatchedForegroundProcessRequest[], options: BatchedForegroundProcessOptions = {} @@ -91,6 +136,7 @@ export function resolveAgentForegroundProcessesFromIndex( } } + const occupancy = getTtyOccupancy(index.rows) return requests.map((request) => { const root = lookupProcessTableIndex(index, (value) => value.byPid.get(request.rootPid)) if (!root) { @@ -114,6 +160,20 @@ export function resolveAgentForegroundProcessesFromIndex( reason: 'no_controlling_tty' } } + // The only host-observable "nothing is running here" signal, and it has to be read off the + // whole tty rather than off `tpgid === pgid`. A backgrounded `pnpm build &` and a Ctrl-Z'd + // editor both leave the shell owning the foreground group, byte-identical to an idle prompt; + // what separates them is a second process group attached to the pane's terminal. That is also + // exactly the blast radius of the stop this attests to — `forceKillPosixPtyProcessGroups` + // SIGKILLs every process group on the tty — so the evidence and the kill now measure the same + // thing. A reader may treat `false` as "busy" and must never treat absence as "idle". + const ttyProcessGroups = occupancy.processGroupsByTty.get(root.tpgid) + const shellOwnsEveryTtyProcessGroup = + root.tpgid === root.pgid && + ttyProcessGroups !== undefined && + ttyProcessGroups.size === 1 && + ttyProcessGroups.has(root.pgid) && + !occupancy.stoppedTtys.has(root.tpgid) const allCandidates = rowsByOwner.get(root.pid) ?? [] const foregroundCandidates = allCandidates.filter((row) => row.pgid === root.tpgid) const fallbackProcess = request.fallbackProcess @@ -125,7 +185,7 @@ export function resolveAgentForegroundProcessesFromIndex( ) : foregroundCandidates if (wrapperFallback && candidates.length !== 1) { - return { available: true, processName: null } + return { available: true, processName: null, shellOwnsEveryTtyProcessGroup } } const selected = selectForegroundProcessCandidate(candidates, allCandidates) if (selected) { @@ -135,10 +195,11 @@ export function resolveAgentForegroundProcessesFromIndex( selected.recognized, selected.candidate, allCandidates - ) + ), + shellOwnsEveryTtyProcessGroup } } - return { available: true, processName: null } + return { available: true, processName: null, shellOwnsEveryTtyProcessGroup } }) } @@ -147,7 +208,14 @@ export function toForegroundProcessEvidence( metadata: { authorityGeneration: string; observationEpoch: number; capturedAgeMs: number } ): ForegroundProcessEvidence { return result.available - ? { ...metadata, verdict: 'live', processName: result.processName } + ? { + ...metadata, + verdict: 'live', + processName: result.processName, + ...(result.shellOwnsEveryTtyProcessGroup !== undefined + ? { shellOwnsEveryTtyProcessGroup: result.shellOwnsEveryTtyProcessGroup } + : {}) + } : { ...metadata, verdict: 'unverifiable', diff --git a/src/main/providers/pty-process-info.ts b/src/main/providers/pty-process-info.ts index 848ff07c78a..942cab84f73 100644 --- a/src/main/providers/pty-process-info.ts +++ b/src/main/providers/pty-process-info.ts @@ -18,4 +18,13 @@ export type PtyProcessInfo = { /** Optional host-side process evidence attached to an inventory seed. */ foregroundProcessEvidence?: ForegroundProcessEvidence agentSessionOwners?: AgentSessionOwnerBinding[] + /** Age measured on the OWNING host's clock. Absent means the host did not measure it, which is + * not the same as "new" or "old" — a reader that needs an age must defer instead of assuming. */ + hostAgeMs?: number + /** True when the host spawned this PTY for an Orca pane, false for a bare host shell. Absent from + * a host that never published it; absence is neither value. */ + paneBound?: boolean + /** The client identity the OWNING host recorded as having asked it to create this PTY. Absent + * whenever the host could not attest one, and absence must never be read as "unowned". */ + ownerClientInstanceId?: string } diff --git a/src/main/providers/pty-provider-contract.ts b/src/main/providers/pty-provider-contract.ts index 35fca5b0b34..cec4dbcb30a 100644 --- a/src/main/providers/pty-provider-contract.ts +++ b/src/main/providers/pty-provider-contract.ts @@ -204,6 +204,12 @@ export type IPtyProvider = { keepHistory?: boolean deadlineMs?: number expectedIncarnationId?: PtyIncarnationId + /** Ask the execution host to refuse this stop unless it recorded this exact client identity + * as the PTY's creator AND this connection still authenticates as it. Optional because a + * host that predates it ignores the field, and because most stops are ordinary teardown of a + * pane whose owner the host may never have attested (a revived PTY carries none). Set it + * wherever the caller's authority to destroy comes from that attestation. */ + expectedOwnerClientInstanceId?: string } ): Promise sendSignal(id: string, signal: string): Promise diff --git a/src/main/providers/ssh-pty-provider.ts b/src/main/providers/ssh-pty-provider.ts index f218cc43eca..d48e09bba06 100644 --- a/src/main/providers/ssh-pty-provider.ts +++ b/src/main/providers/ssh-pty-provider.ts @@ -225,15 +225,17 @@ export class SshPtyProvider implements IPtyProvider { } async shutdown(id: string, opts: Parameters[1]): Promise { + // Both fences are omitted rather than sent undefined: a host that predates either must see no + // key at all, and the owner fence in particular must never reach it as a falsy claim. + const { expectedIncarnationId, expectedOwnerClientInstanceId } = opts await this.mux.request( 'pty.shutdown', { id: this.toRelayPtyId(id), immediate: opts.immediate ?? false, keepHistory: opts.keepHistory ?? false, - ...(opts.expectedIncarnationId === undefined - ? {} - : { expectedIncarnationId: opts.expectedIncarnationId }) + ...(expectedIncarnationId === undefined ? {} : { expectedIncarnationId }), + ...(expectedOwnerClientInstanceId === undefined ? {} : { expectedOwnerClientInstanceId }) }, relayTimeoutOptions(opts.deadlineMs) ) diff --git a/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts b/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts new file mode 100644 index 00000000000..13289416491 --- /dev/null +++ b/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts @@ -0,0 +1,75 @@ +// Three different refusals leave `reattachSshPtySessionForSpawn` carrying the same +// `SSH_SESSION_EXPIRED` text, and only one of them observed the process. That text is therefore not +// a verdict, and a caller that tests it with `.includes()` cannot tell "the host says this PTY is +// gone" from "the PTY is fine, its source stream needs rebuilding". +// +// It matters because the callers that DO test it act destructively: `spawn-execute.ts` expires the +// lease and deletes the in-memory ownership, which between them are the client's only record that a +// remote process exists. Erase both for a live PTY and #9819's sweep finds a host-attested, +// route-less, pane-bound shell on the next connect and SIGKILLs it. +// +// So this pins the discriminator the destructive branch keys on: the type, not the message. +import { describe, expect, it, vi } from 'vitest' +import { isSshPtyAbsentFromRelayError, SSH_SESSION_EXPIRED_ERROR } from './ssh-pty-errors' +import { reattachSshPtySessionForSpawn } from './ssh-pty-session-reattach' +import type { SshChannelMultiplexer } from '../ssh/ssh-channel-multiplexer' + +const CONNECTION = 'conn-1' +const SESSION = 'pty-1' + +function reattachAgainst(attach: () => Promise): Promise { + return reattachSshPtySessionForSpawn({ + mux: { request: vi.fn(attach) } as unknown as SshChannelMultiplexer, + connectionId: CONNECTION, + sessionId: SESSION, + options: { cols: 80, rows: 24 }, + exitRaceTracker: { + begin: () => 1, + didMatchingExitArrive: () => false, + finish: () => {} + } as never, + acceptLivePty: () => {} + }) +} + +async function refusalFrom(attach: () => Promise): Promise { + try { + await reattachAgainst(attach) + } catch (error) { + return error as Error + } + throw new Error('expected the reattach to be refused') +} + +describe('an SSH reattach refusal says whether the host observed the PTY', () => { + it('marks a relay that answered "not found" as positive evidence of absence', async () => { + const error = await refusalFrom(async () => { + throw new Error(`PTY "${SESSION}" not found`) + }) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(true) + }) + + it('does not mark a restoreRequired refusal as absence, though it reads identically', async () => { + // The PTY attached. The relay answered about it. It is running. Only the source stream could + // not be resumed — see the `restoreRequired` carve-out in ssh-pty-errors.ts. + const error = await refusalFrom(async () => ({ + incarnationId: '11111111-1111-4111-8111-111111111111', + sourceRecovery: { status: 'restoreRequired', reason: 'checkpoint_unavailable' } + })) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(false) + }) + + it('does not mark an identity mismatch as absence either', async () => { + // The id names a LIVE PTY that belongs to a different pane. + const error = await refusalFrom(async () => { + throw new Error(`PTY "${SESSION}" not found (identity mismatch)`) + }) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(false) + }) +}) diff --git a/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts b/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts new file mode 100644 index 00000000000..e2d8ef6c9dc --- /dev/null +++ b/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts @@ -0,0 +1,310 @@ +// #9819, the client half: what the sweep actually asks the store and the host, and what it does +// with the answers. The rule itself is covered in ssh-relay-pty-ownership-proof.test.ts. +import { describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import type { IPtyProvider } from '../providers/types' +import type { PtyProcessInfo } from '../providers/pty-process-info' +import type { SshRemotePtyLease } from '../../shared/ssh-types' +import type { ForegroundProcessEvidence } from '../../shared/foreground-process-evidence' +import type { PersistedState } from '../../shared/persisted-state-types' +import { + upsertSshRemotePtyLease, + type SshPtyLeaseOperations +} from '../persistence/leasing-ssh-ptys/ssh-pty-lease-operations' +import { + RELAY_PTY_SWEEP_PASS_BUDGET_MS, + sweepOrphanedRelayPtys +} from './ssh-orphan-relay-pty-sweep' +import { RELAY_PTY_SWEEP_MIN_AGE_MS } from '../../shared/ssh-relay-pty-ownership-proof' + +const TARGET = 'target-1' +const OURS = 'client-instance-ours' +// A stable pane id is a UUID; anything else is stripped before supersede can match on it. +const LEAF = '11111111-2222-4333-8444-555555555555' + +const OBSERVATION = { authorityGeneration: 'gen-1', observationEpoch: 1, capturedAgeMs: 0 } + +/** The host looked and saw its own shell owning the terminal: nothing is running in the pane. */ +function idleShell(): ForegroundProcessEvidence { + return { ...OBSERVATION, verdict: 'live', processName: null, shellOwnsEveryTtyProcessGroup: true } +} + +function hostEntry(overrides: Partial = {}): PtyProcessInfo { + return { + id: `ssh:${TARGET}@@pty-1`, + incarnationId: 'inc-1', + cwd: '/home/user', + title: 'zsh', + ownerClientInstanceId: OURS, + hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS * 2, + paneBound: true, + foregroundProcessEvidence: idleShell(), + ...overrides + } +} + +function createHarness( + processes: PtyProcessInfo[], + leases: SshRemotePtyLease[] = [] +): { provider: IPtyProvider; store: Store; shutdown: ReturnType } { + const shutdown = vi.fn().mockResolvedValue(undefined) + const provider = { + listProcesses: vi.fn().mockResolvedValue(processes), + shutdown + } as unknown as IPtyProvider + const store = { + getSshRemotePtyLeases: vi.fn().mockReturnValue(leases) + } as unknown as Store + return { provider, store, shutdown } +} + +function run( + harness: ReturnType, + overrides: Partial[0]> = {} +): Promise { + return sweepOrphanedRelayPtys({ + targetId: TARGET, + store: harness.store, + provider: harness.provider, + clientInstanceId: OURS, + isSessionOwner: true, + routedPtyIds: [], + shouldContinue: () => true, + ...overrides + }) +} + +function lease(ptyId: string, state: SshRemotePtyLease['state']): SshRemotePtyLease { + return { ptyId, state } as SshRemotePtyLease +} + +describe('sweepOrphanedRelayPtys', () => { + it('stops an attested orphan, fenced on the incarnation the same listing published', async () => { + const harness = createHarness([hostEntry()]) + + await run(harness) + + expect(harness.shutdown).toHaveBeenCalledWith( + `ssh:${TARGET}@@pty-1`, + expect.objectContaining({ immediate: true, expectedIncarnationId: 'inc-1' }) + ) + }) + + it('asks the host to re-check ownership on the one call that cannot be undone', async () => { + // The stop is the only irreversible step in this flow, and until now the whole nine-condition + // rule was enforced only here, on the client that decided to make it. Naming the owner makes + // the host re-decide where the processes actually live. + const harness = createHarness([hostEntry()]) + + await run(harness) + + expect(harness.shutdown).toHaveBeenCalledWith( + `ssh:${TARGET}@@pty-1`, + expect.objectContaining({ expectedOwnerClientInstanceId: OURS }) + ) + }) + + it('does not act on an observation that aged out between the listing and the plan', async () => { + // The listing answered inside the budget, but this pass then spent longer than the evidence is + // good for. Staleness has to degrade to "leave it running". + let clock = 1_000_000 + const harness = createHarness([hostEntry()]) + harness.provider.listProcesses = vi.fn().mockImplementation(async () => { + clock += 1 + return [hostEntry()] + }) + + const pass = (maximumEvidenceAgeMs: number): Promise => + run(harness, { + now: () => clock, + maximumEvidenceAgeMs, + passBudgetMs: 60_000, + shouldContinue: () => { + clock += 20 + return true + } + }) + + await pass(10) + expect(harness.shutdown).not.toHaveBeenCalled() + + // Positive control: the same entry, the same elapsed time, a budget that covers it. + await pass(10_000) + expect(harness.shutdown).toHaveBeenCalledTimes(1) + }) + + it('leaves a PTY the caller just reattached alone', async () => { + const harness = createHarness([hostEntry()]) + + await run(harness, { routedPtyIds: ['pty-1'] }) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it.each([['attached'], ['detached']] as const)( + 'leaves a PTY holding a live %s lease alone', + async (state) => { + const harness = createHarness([hostEntry()], [lease('pty-1', state)]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + } + ) + + it('leaves a PTY with an undelivered stop to the replay pass', async () => { + // The kill-intent journal owns those: it re-fences and retries them, and a second stop issued + // from here would race that decision with weaker evidence. + const tombstoned = { + ...lease('pty-1', 'terminated'), + pendingKill: { requestedAt: 1, incarnationId: 'inc-1', attempts: 0 } + } as SshRemotePtyLease + const harness = createHarness([hostEntry()], [tombstoned]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('leaves a PTY whose lease this client expired alone', async () => { + // The reversal this guards: supersedeSiblingLeasesForPane, dropStalePty and the missing-surface + // refusal all write `expired` precisely BECAUSE they will not stop the remote process. + const harness = createHarness([hostEntry()], [lease('pty-1', 'expired')]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('leaves alone a lease the real supersede path expired when a pane re-leased', async () => { + // Drives the actual persistence operation rather than asserting the state by hand, so this + // stays true only while supersede really does leave the predecessor's process running. + const state: PersistedState = { + sshRemotePtyLeases: [ + { + targetId: TARGET, + ptyId: 'pty-1', + state: 'attached', + worktreeId: 'wt-1', + leafId: LEAF, + createdAt: 1, + updatedAt: 1 + } + ] + } as unknown as PersistedState + const operations: SshPtyLeaseOperations = { + state, + toStoredPtyId: (_targetId, ptyId) => ptyId, + toComparablePtyId: (_targetId, ptyId) => ptyId, + clearBindingsForTarget: () => {}, + clearBindingsForLeases: () => false, + flush: () => {}, + flushDurableStateOrThrowAsync: async () => {} + } + // The same pane re-leases under a new relay id; pty-1 is expired, never terminated. + upsertSshRemotePtyLease(operations, { + targetId: TARGET, + ptyId: 'pty-2', + state: 'attached', + worktreeId: 'wt-1', + leafId: LEAF + }) + expect(state.sshRemotePtyLeases?.find((entry) => entry.ptyId === 'pty-1')?.state).toBe( + 'expired' + ) + const harness = createHarness([hostEntry()], state.sshRemotePtyLeases ?? []) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('forwards the host foreground observation, so a busy pane is never swept', async () => { + // A `claude` the user launched by hand: Orca registered no agent session, so the entry carries + // no agentSessionOwners and only the host's own observation can save it. + const harness = createHarness([ + hostEntry({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: 'claude', + shellOwnsEveryTtyProcessGroup: false + } + }) + ]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('bounds the listing and every stop with one connect budget', async () => { + const harness = createHarness([hostEntry()]) + const start = 1_000_000 + + await run(harness, { now: () => start }) + + const deadline = vi.mocked(harness.provider.listProcesses).mock.calls[0]?.[0]?.deadlineMs + expect(deadline).toBe(start + RELAY_PTY_SWEEP_PASS_BUDGET_MS) + expect(harness.shutdown).toHaveBeenCalledWith( + `ssh:${TARGET}@@pty-1`, + expect.objectContaining({ deadlineMs: deadline }) + ) + }) + + it('does sweep a PTY whose lease this client already tombstoned without an order', async () => { + const harness = createHarness([hostEntry()], [lease('pty-1', 'terminated')]) + + await run(harness) + + expect(harness.shutdown).toHaveBeenCalledTimes(1) + }) + + it('asks the host nothing when this connection is not the session owner', async () => { + const harness = createHarness([hostEntry()]) + + await run(harness, { isSessionOwner: false }) + + expect(harness.provider.listProcesses).not.toHaveBeenCalled() + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('stops nothing against a host that publishes no attestation', async () => { + const legacy = hostEntry() + delete legacy.ownerClientInstanceId + delete legacy.hostAgeMs + delete legacy.paneBound + const harness = createHarness([legacy]) + + await run(harness) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) + + it('swallows a failed listing rather than failing the connect it runs on', async () => { + const harness = createHarness([]) + vi.mocked(harness.provider.listProcesses).mockRejectedValue(new Error('relay went away')) + + await expect(run(harness)).resolves.toBeUndefined() + }) + + it('swallows a failed stop and leaves the order to the next connect', async () => { + const harness = createHarness([hostEntry()]) + harness.shutdown.mockRejectedValue(new Error('connection lost')) + + await expect(run(harness)).resolves.toBeUndefined() + }) + + it('abandons the pass when the attempt is superseded mid-flight', async () => { + const harness = createHarness([hostEntry()]) + let alive = true + vi.mocked(harness.provider.listProcesses).mockImplementation(async () => { + alive = false + return [hostEntry()] + }) + + await run(harness, { shouldContinue: () => alive }) + + expect(harness.shutdown).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/ssh/ssh-orphan-relay-pty-sweep.ts b/src/main/ssh/ssh-orphan-relay-pty-sweep.ts new file mode 100644 index 00000000000..fbc72bf5ec8 --- /dev/null +++ b/src/main/ssh/ssh-orphan-relay-pty-sweep.ts @@ -0,0 +1,164 @@ +import type { Store } from '../persistence' +import type { IPtyProvider } from '../providers/types' +import { toAppSshPtyId, toRelaySshPtyId } from '../providers/ssh-pty-id' +import { + planRelayPtySweep, + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + RELAY_PTY_SWEEP_MIN_AGE_MS, + type RelayPtyOwnershipEvidence +} from '../../shared/ssh-relay-pty-ownership-proof' + +export type SshOrphanRelayPtySweepArgs = { + targetId: string + store: Store + provider: IPtyProvider + /** This client's persisted consumer identity for the target. */ + clientInstanceId: string + /** True only when the relay granted this connection the negotiated `session-owner` role. */ + isSessionOwner: boolean + /** Relay PTY ids this connect just reattached, plus any the caller otherwise knows are live. */ + routedPtyIds: Iterable + shouldContinue: () => boolean + now?: () => number + minimumHostAgeMs?: number + /** Absolute budget for the whole pass, in ms from its start. */ + passBudgetMs?: number + maximumEvidenceAgeMs?: number +} + +/** The two client-side claims the plan needs, read in one pass over the leases. + * + * `routed` is every relay PTY id this client still has a route to: a live lease, an id this + * connect reattached, or a stop it recorded and has not delivered. + * + * `expired` is separate on purpose. It is written by four paths, and every one of them + * deliberately leaves the remote process running: a pane re-leasing under a new relay id, a pane + * surface missing from the layout, a retired reattach, and a reattach the HOST answered "not + * found" for. Only the second of those is reached with the process provably alive — the layout + * refusal runs after `pty.attach` already succeeded (`restoreReattachedPtyRuntime`), which is + * precisely why its own comment reads "topology absence alone is not authority to kill a + * process". A reattach that failed on the transport writes nothing at all: it early-returns as + * `reattachAttemptsExhausted` and the lease stays `attached`, hence routed. + * + * Folding it into `routed` would work, but it would also lose the reason in the skip log, and this + * is the distinction the sweep most needs to be able to explain. */ +function clientClaims(args: SshOrphanRelayPtySweepArgs): { + routed: Set + expired: Set +} { + const routed = new Set(args.routedPtyIds) + const expired = new Set() + for (const lease of args.store.getSshRemotePtyLeases(args.targetId)) { + if (lease.state === 'expired') { + expired.add(lease.ptyId) + } else if (lease.state !== 'terminated') { + routed.add(lease.ptyId) + } + if (lease.pendingKill) { + routed.add(lease.ptyId) + } + } + return { routed, expired } +} + +function toEvidence( + targetId: string, + process: Awaited>[number] +): RelayPtyOwnershipEvidence { + return { + ptyId: toRelaySshPtyId(targetId, process.id), + ...(process.incarnationId ? { incarnationId: process.incarnationId } : {}), + ...(process.ownerClientInstanceId + ? { ownerClientInstanceId: process.ownerClientInstanceId } + : {}), + ...(typeof process.hostAgeMs === 'number' ? { hostAgeMs: process.hostAgeMs } : {}), + ...(typeof process.paneBound === 'boolean' ? { paneBound: process.paneBound } : {}), + ...(process.agentSessionOwners ? { agentSessionOwners: process.agentSessionOwners } : {}), + ...(process.foregroundProcessEvidence + ? { foregroundProcessEvidence: process.foregroundProcessEvidence } + : {}) + } +} + +/** One budget for the whole pass, because this is opportunistic cleanup bolted onto the most + * latency-sensitive and most failure-prone path in the app (#14830, #17830). Without it the + * listing and up to eight stops inherit the mux default and connect waits on all of them. + * Overrunning it yields an empty pass — the same outcome as finding nothing, never a failed + * connect. */ +export const RELAY_PTY_SWEEP_PASS_BUDGET_MS = 5_000 + +/** Stops the relay PTYs this client can prove it created and has since lost every route to. + * + * Runs after reattach, so a PTY this connect reclaimed is already routed and can never be a + * candidate. Best-effort and never throws: it is opportunistic cleanup on the connect path, and a + * failed connection is a much worse outcome than a slot left leaked for another session. + * + * Costs one `pty.listProcesses` per connect. That is the price of reconciling at all — there is no + * cheaper question than asking the authoritative host what it is holding. */ +export async function sweepOrphanedRelayPtys(args: SshOrphanRelayPtySweepArgs): Promise { + if (!args.isSessionOwner || !args.clientInstanceId || !args.shouldContinue()) { + return + } + const now = args.now ?? Date.now + const deadlineMs = now() + (args.passBudgetMs ?? RELAY_PTY_SWEEP_PASS_BUDGET_MS) + try { + const processes = await args.provider.listProcesses({ deadlineMs }) + // The instant the host's observations reached this client. Every later step — reading the + // leases, planning, issuing the stops — ages them, and the plan has to see that age. + const listedAtMs = now() + if (!args.shouldContinue() || now() >= deadlineMs) { + return + } + const claims = clientClaims(args) + const plan = planRelayPtySweep( + processes.map((process) => toEvidence(args.targetId, process)), + { + clientInstanceId: args.clientInstanceId, + isSessionOwner: args.isSessionOwner, + routedPtyIds: claims.routed, + expiredLeasePtyIds: claims.expired, + minimumHostAgeMs: args.minimumHostAgeMs ?? RELAY_PTY_SWEEP_MIN_AGE_MS, + evidenceAgeSinceListingMs: Math.max(0, now() - listedAtMs), + maximumEvidenceAgeMs: args.maximumEvidenceAgeMs ?? RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS + } + ) + if (plan.sweep.length === 0) { + return + } + await Promise.all( + plan.sweep.map(async (target) => { + if (!args.shouldContinue()) { + return + } + try { + // Two fences, both enforced by the host that owns the process. The incarnation stops a + // relay that renumbered its ids between the read and this call from hitting a stranger; + // the owner id makes the host re-check the ownership rule itself, so the one irreversible + // call in this flow is not authorized by the client alone. + await args.provider.shutdown(toAppSshPtyId(args.targetId, target.ptyId), { + immediate: true, + deadlineMs, + expectedIncarnationId: target.incarnationId, + expectedOwnerClientInstanceId: args.clientInstanceId + }) + console.log( + `[ssh-orphan-sweep] stopped orphaned relay PTY ${args.targetId}/${target.ptyId}` + ) + } catch (err) { + // Unverifiable, not failed: the next connect re-reads the inventory and decides again. + console.warn( + `[ssh-orphan-sweep] stop for ${args.targetId}/${target.ptyId} is unverifiable: ${ + err instanceof Error ? err.message : String(err) + }` + ) + } + }) + ) + } catch (err) { + console.warn( + `[ssh-orphan-sweep] pass on ${args.targetId} stopped early: ${ + err instanceof Error ? err.message : String(err) + }` + ) + } +} diff --git a/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts b/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts new file mode 100644 index 00000000000..ab069240681 --- /dev/null +++ b/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts @@ -0,0 +1,238 @@ +// The one test that spans both halves of the sweep. Every other test in this feature asserts on a +// hand-written `ForegroundProcessEvidence` literal, which is exactly how a foreground-only idle +// predicate survived review: the literals said `shellIsForeground: true` for an idle shell because +// that is what the author believed, and nothing ever produced one from a real process table. +// +// So this runs the REAL publisher (`resolveAgentForegroundProcessesBatch` -> +// `toForegroundProcessEvidence`, what `pty.listProcesses` calls) against the REAL client reader +// (`planRelayPtySweep`), over `ps` output captured verbatim from a Linux container driving a real +// `bash -i` on a real pty. The fixtures below are transcripts, not constructions. +// +// Read the shell's own row in each fixture. In `background` and `ctrlz` it is +// `pgid == tpgid`, `Ss+` — byte-identical to `idle`. That is the defect: a foreground-only +// predicate cannot see a job the user backgrounded or suspended, and the stop it authorizes +// SIGKILLs every process group on the tty. +import { describe, expect, it } from 'vitest' +import { + resolveAgentForegroundProcessesBatch, + toForegroundProcessEvidence +} from '../providers/agent-foreground-process-batch' +import { parseStrictProcessTableRows } from '../../shared/process-table-snapshot' +import { + planRelayPtySweep, + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + RELAY_PTY_SWEEP_MIN_AGE_MS, + type RelayPtySweepContext +} from '../../shared/ssh-relay-pty-ownership-proof' + +const OURS = 'client-instance-ours' + +/** `ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=` on debian:bookworm-slim, one capture per pane + * state, each with a `bash -i` on a pty forked by the harness. */ +const CAPTURES = { + /** Nothing running. The only sweepable state. */ + idle: { + rootPid: 3150, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3150 1 3150 3150 Ss+ bash -i', + ' 3151 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** `sleep 300` in the foreground. The shell's tpgid moved off its own pgid. */ + foreground: { + rootPid: 3155, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3155 1 3155 3156 Ss bash -i', + ' 3156 3155 3156 3156 S+ sleep 300', + ' 3157 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** `sleep 300 &`. The shell reads IDENTICALLY to `idle`; only the job's own row differs. */ + background: { + rootPid: 3152, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3152 1 3152 3152 Ss+ bash -i', + ' 3153 3152 3153 3152 S sleep 300', + ' 3154 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** `sleep 300` then Ctrl-Z. The shell again reads IDENTICALLY to `idle`. */ + ctrlz: { + rootPid: 3158, + table: [ + ' 1 0 1 -1 Ss python3 /work/.ptycap.py', + ' 3158 1 3158 3158 Ss+ bash -i', + ' 3159 3158 3159 3158 T sleep 300', + ' 3160 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + } +} as const + +function context(overrides: Partial = {}): RelayPtySweepContext { + return { + clientInstanceId: OURS, + isSessionOwner: true, + routedPtyIds: new Set(), + expiredLeasePtyIds: new Set(), + minimumHostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS, + evidenceAgeSinceListingMs: 0, + maximumEvidenceAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + ...overrides + } +} + +/** Everything the host does between reading `ps` and putting a record on the wire. */ +async function publish( + capture: { rootPid: number; table: readonly string[] }, + capturedAgeMs = 0 +): Promise> { + const rows = parseStrictProcessTableRows(capture.table.join('\n')) + const [result] = await resolveAgentForegroundProcessesBatch( + [{ rootPid: capture.rootPid, fallbackProcess: 'bash' }], + { rows } + ) + return toForegroundProcessEvidence(result, { + authorityGeneration: 'relay-generation-1', + observationEpoch: 1, + capturedAgeMs + }) +} + +/** Everything the client does with that record. Returns the plan for one orphan entry. */ +async function planFor( + capture: { rootPid: number; table: readonly string[] }, + overrides: { capturedAgeMs?: number; context?: Partial } = {} +): Promise> { + return planRelayPtySweep( + [ + { + ptyId: 'pty-1', + incarnationId: 'inc-1', + ownerClientInstanceId: OURS, + hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS * 2, + paneBound: true, + foregroundProcessEvidence: await publish(capture, overrides.capturedAgeMs) + } + ], + context(overrides.context) + ) +} + +function skipReason(plan: ReturnType): string | undefined { + return plan.skipped.find((entry) => entry.ptyId === 'pty-1')?.reason +} + +describe('what the host publishes about a pane, read by the sweep', () => { + it('records that a backgrounded and a suspended shell are indistinguishable at tpgid/pgid', () => { + // The premise of the whole file. If this ever fails, the fixtures drifted and every verdict + // below is testing something other than the defect. Pids differ between captures, so the + // comparison is of the shell row's shape: who its parent is, whether it leads its own process + // group, whether that group owns the terminal, and its state flags. + const shellShape = (capture: { rootPid: number; table: readonly string[] }): string => { + const row = parseStrictProcessTableRows(capture.table.join('\n')).find( + (candidate) => candidate.pid === capture.rootPid + )! + return [ + `ppid=${row.ppid}`, + `leadsOwnGroup=${row.pgid === row.pid}`, + `ownsTerminal=${row.tpgid === row.pgid}`, + `stat=${row.stat}` + ].join(' ') + } + + expect(shellShape(CAPTURES.idle)).toBe('ppid=1 leadsOwnGroup=true ownsTerminal=true stat=Ss+') + expect(shellShape(CAPTURES.background)).toBe(shellShape(CAPTURES.idle)) + expect(shellShape(CAPTURES.ctrlz)).toBe(shellShape(CAPTURES.idle)) + expect(shellShape(CAPTURES.foreground)).not.toBe(shellShape(CAPTURES.idle)) + }) + + it('sweeps an idle shell', async () => { + const evidence = await publish(CAPTURES.idle) + expect(evidence).toMatchObject({ + verdict: 'live', + processName: null, + shellOwnsEveryTtyProcessGroup: true + }) + + const plan = await planFor(CAPTURES.idle) + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) + + it('never sweeps a pane running a foreground job', async () => { + const evidence = await publish(CAPTURES.foreground) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.foreground) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('never sweeps a pane holding a backgrounded job', async () => { + // `sleep 300 &`, i.e. `pnpm build &` or `npm run dev &`. The shell handed the terminal back, + // so the pane looks idle; the job is alive in its own process group on the same tty and a + // stop would SIGKILL it. + const evidence = await publish(CAPTURES.background) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.background) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('never sweeps a pane holding a Ctrl-Z suspended job', async () => { + const evidence = await publish(CAPTURES.ctrlz) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.ctrlz) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('refuses an observation older than the pass it would authorize', async () => { + // Same idle capture that sweeps above; only its age differs. Staleness degrades to "leave it + // running", never to "stop it". + const stale = await planFor(CAPTURES.idle, { + capturedAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS + 1 + }) + expect(stale.sweep).toEqual([]) + expect(skipReason(stale)).toBe('host foreground observation is too old to authorize a stop') + + // And the client's own share of the age counts: a host stamp inside the budget still ages out + // while this pass reads leases and plans. + const agedOnTheClient = await planFor(CAPTURES.idle, { + capturedAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + context: { evidenceAgeSinceListingMs: 1 } + }) + expect(agedOnTheClient.sweep).toEqual([]) + expect(skipReason(agedOnTheClient)).toBe( + 'host foreground observation is too old to authorize a stop' + ) + }) + + it('never sweeps when the host itself is too degraded to answer', async () => { + // `main` has since added `recoverRemoteTerminalRuntime`, a self-driven reconnect on relay + // node-pty failure — a sweep trigger that fires exactly when the host is unwell. The publisher + // has to fail closed there: a capture that cannot locate the shell is `unverifiable`, which is + // its own verdict and never collapses into "idle" (docs/reference/ssh-execution-boundary.md). + const evidence = await publish({ rootPid: 999_999, table: CAPTURES.idle.table }) + expect(evidence).toMatchObject({ verdict: 'unverifiable', reason: 'root_missing' }) + + const plan = await planFor({ rootPid: 999_999, table: CAPTURES.idle.table }) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host could not observe the pane foreground process') + }) + + it('still reclaims a shell whose work really did finish', async () => { + // The feature must not degrade into a no-op. The background fixture's job is gone; what is + // left is the same orphaned shell, and it is swept. + const finished = { + rootPid: CAPTURES.background.rootPid, + table: CAPTURES.background.table.filter((line) => !line.includes('sleep 300')) + } + const plan = await planFor(finished) + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) +}) diff --git a/src/main/ssh/ssh-pty-consumer-recovery.test.ts b/src/main/ssh/ssh-pty-consumer-recovery.test.ts index 09405f9ee0b..d77e8c3c8dc 100644 --- a/src/main/ssh/ssh-pty-consumer-recovery.test.ts +++ b/src/main/ssh/ssh-pty-consumer-recovery.test.ts @@ -8,6 +8,21 @@ import { } from './ssh-pty-consumer-recovery' describe('SSH PTY consumer recovery', () => { + it('says out loud when it mints a new identity, because the sweep goes quiet after', () => { + // The id the host attests on every PTY it holds for us. Minting a new one is fail-safe — it + // can only under-sweep — but it makes the reaper silently stop working, so it must be visible. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const store = { + getSshPtyConsumerRecovery: vi.fn().mockReturnValue(null), + upsertSshPtyConsumerRecovery: vi.fn() + } as unknown as Store + + claimSshPtyConsumerRecovery('mint-logs-the-loss', store) + + expect(warn).toHaveBeenCalledWith(expect.stringContaining('minting a new consumer identity')) + warn.mockRestore() + }) + it('keeps a detached identity when a concurrent open finishes late', async () => { const targetId = 'remember-detach-race' const store = { diff --git a/src/main/ssh/ssh-pty-consumer-recovery.ts b/src/main/ssh/ssh-pty-consumer-recovery.ts index 1c357d32a12..db49a45c399 100644 --- a/src/main/ssh/ssh-pty-consumer-recovery.ts +++ b/src/main/ssh/ssh-pty-consumer-recovery.ts @@ -37,6 +37,15 @@ export function claimSshPtyConsumerRecovery( return current } const persisted = current ? null : store.getSshPtyConsumerRecovery(targetId) + if (!persisted) { + // Deliberately not stabilized: this id is also the consumer session's ownership identity, and + // reusing it without the generations the same dropped record carried would replay a stale + // owner generation at the relay. The cost is visible instead of silent — every relay PTY the + // host still attributes to the previous identity becomes permanently unsweepable (#9819). + console.warn( + `[ssh-pty-consumer] no recovery record for ${targetId}; minting a new consumer identity. Relay PTYs the host attributes to this client's previous identity can no longer be swept.` + ) + } const created: SshPtyConsumerRecoveryState = { clientInstanceId: persisted?.clientInstanceId ?? randomUUID(), detached: false, diff --git a/src/main/ssh/ssh-relay-session-orphan-sweep.test.ts b/src/main/ssh/ssh-relay-session-orphan-sweep.test.ts new file mode 100644 index 00000000000..fd4cbea8c34 --- /dev/null +++ b/src/main/ssh/ssh-relay-session-orphan-sweep.test.ts @@ -0,0 +1,293 @@ +// #9819 end to end on the client: the sweep runs only after reattach, only under a negotiated +// session-owner grant, and only against PTYs this relay itself attributes to this client. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { SshRelaySession } from './ssh-relay-session' +import { createMockDeps, mockDeploySuccess } from './ssh-relay-session-test-fixtures' + +const { muxRequestMock, openConsumerSessionMock } = vi.hoisted(() => ({ + muxRequestMock: vi.fn(), + openConsumerSessionMock: vi.fn(async (_mux: unknown, options: { clientInstanceId: string }) => ({ + state: { + mode: 'negotiated' as const, + clientInstanceId: options.clientInstanceId, + clientGeneration: 1, + ownerGeneration: 1, + ownerLease: 'test-owner-lease' + }, + resumed: false + })) +})) + +vi.mock('./ssh-relay-deploy', () => ({ deployAndLaunchRelay: vi.fn() })) +vi.mock('./ssh-pty-consumer-session', () => ({ + openSshPtyConsumerSession: openConsumerSessionMock +})) +vi.mock('../ipc/ssh-pty-output-intake-registry', () => ({ + acceptSshPtyOutputData: vi.fn().mockResolvedValue(undefined), + acceptSshPtyOutputExit: vi.fn().mockResolvedValue(undefined), + allocateSshPtyProviderGeneration: vi.fn(() => 17), + beginSshPtyOutputGenerationMigration: vi.fn(() => ({ + byPty: new Map(), + completion: Promise.resolve() + })), + closeSshPtyOutputGeneration: vi.fn(), + getSshPtyAcceptedSourceCheckpoints: vi.fn(() => []), + applySshPtySourceCancellationProof: vi.fn(() => true), + applySshPtySourceRecoveryCancellationProof: vi.fn(() => true), + installSshPtySourceAckPublisher: vi.fn(() => () => {}), + installSshPtySourceCancellationPublisher: vi.fn(() => () => {}) +})) +vi.mock('./ssh-relay-deploy-helpers', () => ({ execCommand: vi.fn().mockResolvedValue('') })) +vi.mock('./ssh-remote-orca-cli', () => ({ + runRemoteOrcaCli: vi.fn().mockResolvedValue({ exitCode: 0, stdout: '', stderr: '' }) +})) +vi.mock('./ssh-channel-multiplexer', () => ({ + SshChannelMultiplexer: class MockSshChannelMultiplexer { + notify = vi.fn() + notifyWithSettlement = vi.fn() + request = muxRequestMock + onNotification = vi.fn().mockReturnValue(() => {}) + onNotificationByMethod = vi.fn().mockReturnValue(() => {}) + onRequest = vi.fn().mockReturnValue(() => {}) + onDispose = vi.fn().mockReturnValue(() => {}) + dispose = vi.fn() + isDisposed = vi.fn().mockReturnValue(false) + } +})) +vi.mock('../agent-hooks/remote-managed-hook-installers', () => ({ + installRemoteManagedAgentHooks: vi.fn() +})) +vi.mock('../providers/ssh-pty-provider', () => ({ + SshPtyProvider: class MockSshPtyProvider { + onData = vi.fn().mockReturnValue(() => {}) + onReplay = vi.fn().mockReturnValue(() => {}) + onExit = vi.fn().mockReturnValue(() => {}) + attach = vi.fn().mockResolvedValue(undefined) + attachForReconnect = vi.fn().mockResolvedValue({}) + dispose = vi.fn() + } +})) +vi.mock('../providers/ssh-filesystem-provider', () => ({ + SshFilesystemProvider: class MockSshFilesystemProvider { + dispose = vi.fn() + } +})) +vi.mock('../providers/ssh-git-provider', () => ({ + SshGitProvider: class MockSshGitProvider {} +})) +vi.mock('../ipc/pty', () => ({ + registerSshPtyProvider: vi.fn(), + unregisterSshPtyProvider: vi.fn(), + getSshPtyProvider: vi.fn(), + getPtyIdsForConnection: vi.fn().mockReturnValue([]), + clearPtyOwnershipForConnection: vi.fn(), + clearProviderPtyState: vi.fn(), + deletePtyOwnership: vi.fn(), + setPtyOwnership: vi.fn(), + restorePtyIncarnation: vi.fn(), + isCurrentPtyExit: vi.fn(() => true), + answerStartupTerminalColorQueriesForPty: vi.fn((_id: string, data: string) => data) +})) +vi.mock('../providers/ssh-filesystem-dispatch', () => ({ + registerSshFilesystemProvider: vi.fn(), + unregisterSshFilesystemProvider: vi.fn(), + getSshFilesystemProvider: vi.fn().mockReturnValue({ dispose: vi.fn() }) +})) +vi.mock('../providers/ssh-git-dispatch', () => ({ + registerSshGitProvider: vi.fn(), + unregisterSshGitProvider: vi.fn() +})) + +const { getSshPtyProvider, getPtyIdsForConnection } = await import('../ipc/pty') + +const OUR_CLIENT = 'client-instance-1' + +// One target per test. `claimSshPtyConsumerRecovery` keeps a module-level map keyed on target and +// mints a FRESH clientInstanceId whenever it is asked for a target it already holds a live entry +// for — so a second `establish` on a shared target silently stops matching the host attestation and +// every assertion after the first passes for the wrong reason. +let targetSeq = 0 +function nextTarget(): string { + targetSeq += 1 + return `target-${targetSeq}` +} + +const OBSERVATION = { authorityGeneration: 'gen-1', observationEpoch: 1, capturedAgeMs: 0 } + +function hostEntry( + target: string, + overrides: Record = {} +): Record { + return { + id: `ssh:${target}@@pty-orphan`, + incarnationId: 'inc-orphan', + cwd: '/home/user', + title: 'zsh', + // The relay stamps this from the live consumer grant, so it names THIS client. + ownerClientInstanceId: OUR_CLIENT, + hostAgeMs: 120_000, + paneBound: true, + // The same listing's host observation: the shell owns the terminal, nothing is running. + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: null, + shellOwnsEveryTtyProcessGroup: true + }, + ...overrides + } +} + +describe('SshRelaySession orphaned relay PTY sweep', () => { + let warn: ReturnType + + beforeEach(() => { + vi.clearAllMocks() + warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + muxRequestMock.mockReset() + muxRequestMock.mockResolvedValue([]) + mockDeploySuccess() + vi.mocked(getPtyIdsForConnection).mockReturnValue([]) + }) + + afterEach(() => { + warn.mockRestore() + }) + + async function establish( + target: string, + processes: Record[], + leases: { ptyId: string; state: string }[] = [] + ): Promise<{ shutdown: ReturnType; listProcesses: ReturnType }> { + const deps = createMockDeps() + // Why the recovery row: it pins this session's clientInstanceId, and the comparison is + // meaningless unless the id it uses is the persisted one. + vi.mocked(deps.mockStore.getSshPtyConsumerRecovery).mockReturnValue({ + targetId: target, + clientInstanceId: OUR_CLIENT, + serverBuildId: 'build-1', + clientGeneration: 1, + ownerGeneration: 1, + ownerLease: 'test-owner-lease' + } as ReturnType) + vi.mocked(deps.mockStore.getSshRemotePtyLeases).mockReturnValue( + leases.map((lease) => ({ targetId: target, ...lease })) as ReturnType< + typeof deps.mockStore.getSshRemotePtyLeases + > + ) + const shutdown = vi.fn().mockResolvedValue(undefined) + const listProcesses = vi.fn().mockResolvedValue(processes) + vi.mocked(getSshPtyProvider).mockReturnValue({ + attachForReconnect: vi.fn().mockResolvedValue({}), + listProcesses, + shutdown, + dispose: vi.fn() + } as unknown as ReturnType) + + const session = new SshRelaySession( + target, + deps.getMainWindow, + deps.mockStore, + deps.mockPortForward + ) + await session.establish(deps.mockConn) + // Both guards exist because every "never stops" case below is trivially satisfiable. The pass + // has to have run, and it has to have run under the identity the host attests — a session that + // minted a fresh one compares against nothing and skips everything for the wrong reason. + expect(listProcesses).toHaveBeenCalledTimes(1) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('minting a new consumer identity') + ) + return { shutdown, listProcesses } + } + + it('stops an attested orphan the client has no lease for', async () => { + const target = nextTarget() + const { shutdown } = await establish(target, [hostEntry(target)]) + + expect(shutdown).toHaveBeenCalledWith( + `ssh:${target}@@pty-orphan`, + expect.objectContaining({ immediate: true, expectedIncarnationId: 'inc-orphan' }) + ) + }) + + it('never stops a PTY that still holds a live lease', async () => { + const target = nextTarget() + const { shutdown } = await establish( + target, + [hostEntry(target, { id: `ssh:${target}@@pty-live`, incarnationId: 'inc-live' })], + [{ ptyId: 'pty-live', state: 'detached' }] + ) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a PTY whose lease this client expired rather than ordered stopped', async () => { + // What reaches this state in the field: a pane re-leased under a new relay id, a reattach that + // failed on the transport (dropStalePty), or a pane surface missing from the layout. All three + // leave the remote process running on purpose. + const target = nextTarget() + const { shutdown } = await establish( + target, + [hostEntry(target, { id: `ssh:${target}@@pty-gone`, incarnationId: 'inc-gone' })], + [{ ptyId: 'pty-gone', state: 'expired' }] + ) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a pane this relay observes running a foreground process', async () => { + // agentSessionOwners is empty here — the user typed `claude` themselves — so the only thing + // between a live agent and a stop is the host's own foreground observation. + const target = nextTarget() + const { shutdown } = await establish(target, [ + hostEntry(target, { + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: 'claude', + shellOwnsEveryTtyProcessGroup: false + } + }) + ]) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a pane whose foreground observation the relay could not make', async () => { + const target = nextTarget() + const { shutdown } = await establish(target, [ + hostEntry(target, { + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'unverifiable', + reason: 'table_unreadable' + } + }) + ]) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops a PTY this relay attributes to a different client', async () => { + const target = nextTarget() + const { shutdown } = await establish(target, [ + hostEntry(target, { ownerClientInstanceId: 'someone-elses-laptop' }) + ]) + + expect(shutdown).not.toHaveBeenCalled() + }) + + it('never stops anything a relay predating the attestation lists', async () => { + const target = nextTarget() + const legacy = hostEntry(target) + delete legacy.ownerClientInstanceId + delete legacy.hostAgeMs + delete legacy.paneBound + delete legacy.foregroundProcessEvidence + + const { shutdown } = await establish(target, [legacy]) + + expect(shutdown).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/ssh/ssh-relay-session.ts b/src/main/ssh/ssh-relay-session.ts index 68254fbb108..04fe6020749 100644 --- a/src/main/ssh/ssh-relay-session.ts +++ b/src/main/ssh/ssh-relay-session.ts @@ -10,6 +10,7 @@ import { isRelayEndpointHeldError } from './ssh-relay-endpoint-incumbent' import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair' import type { TerminalUnavailableCause } from '../../shared/terminal-unavailable-cause' import { replayPendingSshPtyKills } from './ssh-pending-pty-kill-replay' +import { sweepOrphanedRelayPtys } from './ssh-orphan-relay-pty-sweep' import { SshChannelMultiplexer } from './ssh-channel-multiplexer' import { SshPtyProvider } from '../providers/ssh-pty-provider' import type { SshPtyAttachResult } from '../providers/ssh-pty-session-reattach' @@ -2398,6 +2399,17 @@ export class SshRelaySession { Array.from(attachedLeaseIds) ) } + // Why last: reclaiming comes first, so every PTY this connect could route to is routed before + // anything asks which ones are unreachable (#9819). + await sweepOrphanedRelayPtys({ + targetId: this.targetId, + store: this.store, + provider: ptyProvider, + clientInstanceId: this.ptyConsumerClientInstanceId, + isSessionOwner: this.activePtyConsumerOwner() !== null, + routedPtyIds: ptyIds, + shouldContinue + }) } private async reattachKnownPty(args: { diff --git a/src/relay/pty-handler-inventory-process-evidence.test.ts b/src/relay/pty-handler-inventory-process-evidence.test.ts index 17896f67c72..1c8b59bc6de 100644 --- a/src/relay/pty-handler-inventory-process-evidence.test.ts +++ b/src/relay/pty-handler-inventory-process-evidence.test.ts @@ -157,22 +157,32 @@ describe('PtyHandler inventory foreground evidence', () => { expect((await listProcesses())[0].title).toBe('node') }) - it.each([1, 8])('visits the host table exactly once for %s panes', async (paneCount) => { - const table = Array.from({ length: paneCount }, (_, index) => - paneRows(10_000 + index * 10, ['node /opt/codex']) - ).flat() - const { rows, reads } = countingRows(table) - mockGetStrictProcessTableSnapshot.mockResolvedValue(rows) - for (let index = 0; index < paneCount; index += 1) { - await spawnPane(10_000 + index * 10, 'zsh') + // The cost that matters is per-CAPTURE, not per-pane: the defect this guards against is a + // full-table walk for every pane, which is what an O(PTY x rows) inventory looked like. Two + // linear passes build the two indexes the resolver reads — parent/child correlation, and which + // process groups occupy each controlling terminal — and neither grows with the pane count. + const CAPTURE_PASSES = 2 + + it.each([1, 8])( + 'walks the host table a fixed number of times for %s panes', + async (paneCount) => { + const table = Array.from({ length: paneCount }, (_, index) => + paneRows(10_000 + index * 10, ['node /opt/codex']) + ).flat() + const { rows, reads } = countingRows(table) + mockGetStrictProcessTableSnapshot.mockResolvedValue(rows) + for (let index = 0; index < paneCount; index += 1) { + await spawnPane(10_000 + index * 10, 'zsh') + } + + const listed = await listProcesses() + + expect(listed).toHaveLength(paneCount) + expect(listed.every((entry) => entry.title === 'codex')).toBe(true) + expect(mockGetStrictProcessTableSnapshot).toHaveBeenCalledTimes(1) + // Linear in the capture — NOT one full-table walk per pane, which would be + // `table.length * paneCount` here. + expect(reads()).toBe(table.length * CAPTURE_PASSES) } - - const listed = await listProcesses() - - expect(listed).toHaveLength(paneCount) - expect(listed.every((entry) => entry.title === 'codex')).toBe(true) - expect(mockGetStrictProcessTableSnapshot).toHaveBeenCalledTimes(1) - // One linear index pass — NOT one full-table walk per pane. - expect(reads()).toBe(table.length) - }) + ) }) diff --git a/src/relay/pty-handler-ownership-attestation.test.ts b/src/relay/pty-handler-ownership-attestation.test.ts new file mode 100644 index 00000000000..ec5f1beb5e3 --- /dev/null +++ b/src/relay/pty-handler-ownership-attestation.test.ts @@ -0,0 +1,283 @@ +// The host half of #9819: a client may only reap a relay PTY it can prove it created, so the relay +// has to say who created each one. The attestation is read from the live consumer grant, never from +// a spawn parameter — otherwise it would just echo the caller's claim back at it. +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' + +const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ + mockPtySpawn: vi.fn(), + mockCreateShellPromptReadinessProbe: vi.fn(), + mockPtyInstance: { + pid: process.pid, + onData: vi.fn(), + onExit: vi.fn(), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(), + clear: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + } +})) + +vi.mock('node-pty', () => ({ spawn: mockPtySpawn })) +vi.mock('../main/pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: vi.fn((_pid: number, fallback: () => void) => fallback()) +})) +vi.mock('../main/shell-prompt-readiness-probe', () => ({ + createShellPromptReadinessProbe: mockCreateShellPromptReadinessProbe +})) + +import type { PtyHandler } from './pty-handler' +import { + beginPtyHandlerTest, + endPtyHandlerTest, + type MockDispatcher +} from './pty-handler-test-harness' +import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../shared/process-table-snapshot' + +const PANE_KEY = 'tab-agent:22222222-2222-4222-8222-222222222222' + +type Summary = { + id: string + paneBound?: boolean + hostAgeMs?: number + ownerClientInstanceId?: string + foregroundProcessEvidence?: { capturedAgeMs: number } +} + +describe('PtyHandler publishes host-attested PTY ownership', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + + async function spawnFrom( + clientId: number, + params: Record = {} + ): Promise<{ id: string }> { + mockPtySpawn.mockReturnValue({ ...mockPtyInstance, onData: vi.fn(), onExit: vi.fn() }) + return (await dispatcher.callRequest('pty.spawn', params, { + clientId, + isStale: () => false + } as never)) as { id: string } + } + + async function listProcesses(): Promise { + return (await dispatcher.callRequest('pty.listProcesses', {})) as Summary[] + } + + beforeEach(() => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + handler.setConsumerIdentityResolver((clientId) => (clientId === 7 ? 'client-A' : null)) + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('attributes a pane spawn to the identity the consumer grant names', async () => { + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + vi.advanceTimersByTime(45_000) + + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(entry?.ownerClientInstanceId).toBe('client-A') + expect(entry?.paneBound).toBe(true) + expect(entry?.hostAgeMs).toBeGreaterThanOrEqual(45_000) + }) + + it('omits the attestation entirely when the connection holds no active grant', async () => { + const { id } = await spawnFrom(9, { env: { ORCA_PANE_KEY: PANE_KEY } }) + + const entry = (await listProcesses()).find((process) => process.id === id) + + // Absent, not empty-string or null: a reader must be able to tell "unattested" from any value. + expect(entry).not.toHaveProperty('ownerClientInstanceId') + }) + + it('never attests a revived PTY, so a restored session is not sweepable', async () => { + // The load-bearing invariant of #9819's host half, and the one most likely to be "helpfully" + // broken later: revive replays state a client serialized, which is not this host observing who + // asked for the shell. A revived PTY *does* get paneBound: true (paneKey is restored) and a + // fresh createdAt, so the omitted attestation is the only thing standing between a relay + // restart and a sweep of the entire restored session. + const revivedId = 'pty-revived-1' + await dispatcher.callRequest( + 'pty.revive', + { + state: JSON.stringify([ + { + id: revivedId, + pid: process.pid, + cwd: process.cwd(), + paneKey: PANE_KEY, + cols: 80, + rows: 24 + } + ]) + }, + { clientId: 7, isStale: () => false } as never + ) + + const entry = (await listProcesses()).find((process) => process.id === revivedId) + expect(entry, 'revive should have produced a live PTY entry').toBeDefined() + // paneBound is true, which is exactly why the missing attestation has to be asserted: + // every other sweep precondition is satisfied by a revived pane. + expect(entry?.paneBound).toBe(true) + expect(entry?.ownerClientInstanceId).toBeUndefined() + }) + + it('reports a bare shell as not pane-bound', async () => { + const { id } = await spawnFrom(7, {}) + + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(entry?.paneBound).toBe(false) + expect(entry?.ownerClientInstanceId).toBe('client-A') + }) + + it('dates the foreground observation instead of stamping it fresh', async () => { + // `capturedAgeMs` used to be a hardcoded 0 with no reader anywhere, so the one field that + // exists to bound staleness asserted the evidence was never stale. It now carries the + // worst-case age of the TTL-shared capture the record was derived from. + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBe( + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + ) + }) +}) + +// CodeRabbit's unaddressed note, and the asymmetry behind it: `pty.spawn` and `pty.attach` both +// take a request context and both check it, while `pty.shutdown` — the one call that irreversibly +// destroys a user's running process — took none, so the entire ownership rule was enforced only on +// the client that decided to make the call. +describe('PtyHandler authorizes a fenced stop against its own attestation', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + + async function spawnFrom(clientId: number): Promise<{ id: string }> { + mockPtySpawn.mockReturnValue({ ...mockPtyInstance, onData: vi.fn(), onExit: vi.fn() }) + return (await dispatcher.callRequest('pty.spawn', { env: { ORCA_PANE_KEY: PANE_KEY } }, { + clientId, + isStale: () => false + } as never)) as { id: string } + } + + async function isStillHeld(id: string): Promise { + const entries = (await dispatcher.callRequest('pty.listProcesses', {})) as Summary[] + return entries.some((entry) => entry.id === id) + } + + /** What actually reaches the process. A refusal has to leave this untouched — the point of the + * check is the process, not the error. */ + function killSignals(): unknown[][] { + return mockPtyInstance.kill.mock.calls + } + + function stop(id: string, params: Record, clientId: number): Promise { + return dispatcher.callRequest('pty.shutdown', { id, immediate: false, ...params }, { + clientId, + isStale: () => false + } as never) + } + + beforeEach(() => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + handler.setConsumerIdentityResolver((clientId) => + clientId === 7 ? 'client-A' : clientId === 8 ? 'client-B' : null + ) + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('stops a PTY when the connection and the host agree on the owner', async () => { + const { id } = await spawnFrom(7) + + await expect( + stop(id, { expectedOwnerClientInstanceId: 'client-A' }, 7) + ).resolves.toBeUndefined() + expect(killSignals()).toEqual([['SIGTERM']]) + }) + + it('refuses when another client asserts our identity, and leaves the process running', async () => { + // The claim is a parameter, so a confused or displaced client can send any value it likes. + // What it cannot do is authenticate as that identity on this connection. + const { id } = await spawnFrom(7) + + await expect(stop(id, { expectedOwnerClientInstanceId: 'client-A' }, 8)).rejects.toThrow( + /requester is not the attested owner/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(id)).toBe(true) + }) + + it('refuses when the connection holds no grant at all', async () => { + const { id } = await spawnFrom(7) + + await expect(stop(id, { expectedOwnerClientInstanceId: 'client-A' }, 99)).rejects.toThrow( + /requester is not the attested owner/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(id)).toBe(true) + }) + + it('refuses a PTY this host never attested, even to the client that asked', async () => { + // The revived-PTY case. Both sweep preconditions a client can see are satisfied — pane-bound, + // old enough — and only the host knows it never recorded a creator for it. + const revivedId = 'pty-revived-fence' + await dispatcher.callRequest( + 'pty.revive', + { + state: JSON.stringify([ + { + id: revivedId, + pid: process.pid, + cwd: process.cwd(), + paneKey: PANE_KEY, + cols: 80, + rows: 24 + } + ]) + }, + { clientId: 7, isStale: () => false } as never + ) + + await expect(stop(revivedId, { expectedOwnerClientInstanceId: 'client-A' }, 7)).rejects.toThrow( + /this host attested no such owner/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(revivedId)).toBe(true) + }) + + it('leaves an ordinary teardown that names no owner exactly as it was', async () => { + // Rule 1's obligation: an old client, and every non-sweep caller on a current one, omits the + // field. The host must not start refusing a stop it is obliged to honour. + const { id } = await spawnFrom(7) + + await expect(stop(id, {}, 7)).resolves.toBeUndefined() + expect(killSignals()).toEqual([['SIGTERM']]) + }) + + it('rejects a malformed owner claim rather than ignoring it', async () => { + const { id } = await spawnFrom(7) + + await expect(stop(id, { expectedOwnerClientInstanceId: '' }, 7)).rejects.toThrow( + /Invalid expectedOwnerClientInstanceId/ + ) + expect(killSignals()).toEqual([]) + expect(await isStillHeld(id)).toBe(true) + }) +}) diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index f8bd2351cf2..25945203878 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -70,6 +70,7 @@ import { } from '../main/providers/agent-foreground-process' import { getStrictProcessTableSnapshot, + PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS, type ProcessTableRow } from '../shared/process-table-snapshot' import type { ForegroundProcessEvidence } from '../shared/foreground-process-evidence' @@ -190,6 +191,13 @@ type ManagedPty = { startupIngressIntent?: ReturnType ownerBackend: PtyOwnerBackend agentSessionOwners?: AgentSessionOwnerBinding[] + /** Host clock, host-relative only: published as an age so no client has to trust our wall clock. */ + createdAt: number + /** The authenticated consumer identity that asked this host to create this PTY, read from the + * live grant rather than from a spawn parameter. Absent whenever the host could not attest one + * (no consumer session, or a revive replaying state some other client serialized), and absence + * must never be read as "nobody owns it". */ + ownerClientInstanceId?: string } type RelayAgentSessionCreateResult = { @@ -369,6 +377,14 @@ type PtyProcessSummary = { terminalHandle?: string foregroundProcessEvidence?: ForegroundProcessEvidence agentSessionOwners?: AgentSessionOwnerBinding[] + /** Age on the HOST's clock. Published instead of a creation timestamp so a client with a skewed + * clock cannot compute a negative or enormous age and act on it. */ + hostAgeMs?: number + /** True when this PTY was spawned for an Orca pane (`ORCA_PANE_KEY`). False means a bare relay + * shell. Absent from a host that predates the field — which is neither. */ + paneBound?: boolean + /** See {@link ManagedPty.ownerClientInstanceId}. Omitted when this host cannot attest one. */ + ownerClientInstanceId?: string } type SerializedPtyEntry = { @@ -456,6 +472,7 @@ export class PtyHandler { private consumerPausedOutputPtys = new Set() private removeLegacyCapacityListener: (() => void) | null = null private sourcePublication: RelayPtySourcePublication | null = null + private consumerIdentityResolver: ((clientId: number) => string | null) | null = null private lastInputAtByPty = new Map() private interactiveOutputCharsByPty = new Map() private pendingSpawnCount = 0 @@ -510,6 +527,12 @@ export class PtyHandler { this.sourcePublication = publication } + /** Supplies the authenticated client identity behind a transport connection, so a spawn can be + * attributed to the consumer session that requested it. */ + setConsumerIdentityResolver(resolve: ((clientId: number) => string | null) | null): void { + this.consumerIdentityResolver = resolve + } + handleSourceCreditAvailable(id: string): void { this.sourcePublication?.onCreditAvailable(id) } @@ -991,7 +1014,7 @@ export class PtyHandler { private registerHandlers(): void { this.dispatcher.onRequest('pty.spawn', (p, context) => this.spawn(p, context)) this.dispatcher.onRequest('pty.attach', (p, context) => this.attach(p, context)) - this.dispatcher.onRequest('pty.shutdown', (p) => this.shutdown(p)) + this.dispatcher.onRequest('pty.shutdown', (p, context) => this.shutdown(p, context)) this.dispatcher.onRequest('pty.sendSignal', (p) => this.sendSignal(p)) this.dispatcher.onRequest('pty.getCwd', (p) => this.getCwd(p)) this.dispatcher.onRequest('pty.getInitialCwd', (p) => this.getInitialCwd(p)) @@ -1871,11 +1894,15 @@ export class PtyHandler { params.startupIngressVersion === PTY_STARTUP_INGRESS_VERSION ? parsePtyStartupIngressIntent(params.startupIngress) : undefined + const ownerClientInstanceId = + context === undefined ? null : (this.consumerIdentityResolver?.(context.clientId) ?? null) const managed: ManagedPty = { id, incarnationId: randomUUID(), pty: term, initialCwd: cwd, + createdAt: Date.now(), + ...(ownerClientInstanceId ? { ownerClientInstanceId } : {}), buffered: new RecentPtyOutputBuffer({ preserveChunkBoundaries: false, limit: REPLAY_BUFFER_MAX @@ -2097,7 +2124,7 @@ export class PtyHandler { return { cols: managed.pty.cols, rows: managed.pty.rows } } - private async shutdown(params: Record): Promise { + private async shutdown(params: Record, context?: RequestContext): Promise { const id = params.id as string const immediate = params.immediate as boolean const expectedIncarnationId = params.expectedIncarnationId @@ -2107,6 +2134,14 @@ export class PtyHandler { ) { throw new Error('Invalid expectedIncarnationId') } + const expectedOwnerClientInstanceId = params.expectedOwnerClientInstanceId + if ( + expectedOwnerClientInstanceId !== undefined && + (typeof expectedOwnerClientInstanceId !== 'string' || + expectedOwnerClientInstanceId.length === 0) + ) { + throw new Error('Invalid expectedOwnerClientInstanceId') + } const managed = this.ptys.get(id) if (!managed) { return @@ -2114,6 +2149,9 @@ export class PtyHandler { if (expectedIncarnationId !== undefined && expectedIncarnationId !== managed.incarnationId) { throw new Error(`PTY incarnation mismatch for ${id}`) } + if (expectedOwnerClientInstanceId !== undefined) { + this.assertShutdownOwnership(id, managed, expectedOwnerClientInstanceId, context) + } // Why: `pty.shutdown` is the only authoritative statement this host ever gets that a tab is // gone. Record it before the kill request, because the kill is the part that can fail: an agent // that survives teardown otherwise keeps posting hooks the relay forwards as a live agent pane @@ -2138,6 +2176,34 @@ export class PtyHandler { } } + /** Re-decide, on the host, whether the caller may destroy this PTY. + * + * `pty.shutdown` is irreversible and its siblings `pty.spawn`/`pty.attach` already take a + * request context; without this the whole ownership rule lived on the client, on the one call + * that cannot be taken back. Both halves are checked here because either alone is an echo: the + * connection must still authenticate as that consumer identity (so a claim cannot be asserted), + * and this host must have recorded that same identity as the PTY's creator at spawn (so the + * caller cannot reach a PTY it never made). + * + * Only callers that opt in are checked. An ordinary pane teardown does not pass the field, and + * must not: a revived PTY carries no attested owner at all, and a host predating the attestation + * would refuse stops it is obliged to honour. */ + private assertShutdownOwnership( + id: string, + managed: ManagedPty, + expectedOwnerClientInstanceId: string, + context: RequestContext | undefined + ): void { + const requester = + context === undefined ? null : (this.consumerIdentityResolver?.(context.clientId) ?? null) + if (requester !== expectedOwnerClientInstanceId) { + throw new Error(`PTY "${id}" stop refused: requester is not the attested owner`) + } + if (managed.ownerClientInstanceId !== expectedOwnerClientInstanceId) { + throw new Error(`PTY "${id}" stop refused: this host attested no such owner`) + } + } + /** Record that this pane's client surface is gone, and tell the hook server so the pane's cached * agent status stops being replayed to reconnecting clients. Returns false when there is no pane * surface to retire. */ @@ -2406,9 +2472,15 @@ export class PtyHandler { let evidenceRows: readonly ProcessTableRow[] | null = null let evidenceResults: BatchedForegroundProcessResult[] = [] const evidenceEpoch = ++this.foregroundEvidenceEpoch + // Worst-case capture time for the snapshot below, not the instant its await settled: the + // reader may serve a TTL-cached table, so the observation can already be one window old. The + // loop that follows can await per entry, so each record is stamped against this rather than + // carrying a shared constant. + let evidenceCapturedAtMs = Date.now() if (process.platform !== 'win32' && managedEntries.length > 0) { try { evidenceRows = await getStrictProcessTableSnapshot() + evidenceCapturedAtMs = Date.now() - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS evidenceResults = await resolveAgentForegroundProcessesBatch( managedEntries.map(([, managed]) => ({ rootPid: managed.pty.pid, @@ -2442,7 +2514,7 @@ export class PtyHandler { { authorityGeneration: this.ptyIdMintEpoch, observationEpoch: evidenceEpoch, - capturedAgeMs: 0 + capturedAgeMs: Math.max(0, Date.now() - evidenceCapturedAtMs) } ) : undefined @@ -2451,6 +2523,11 @@ export class PtyHandler { incarnationId: managed.incarnationId, cwd: managed.initialCwd, title, + hostAgeMs: Math.max(0, Date.now() - managed.createdAt), + paneBound: Boolean(managed.paneKey ?? managed.attachIdentity?.paneKey), + ...(managed.ownerClientInstanceId + ? { ownerClientInstanceId: managed.ownerClientInstanceId } + : {}), ...(managed.worktreeId ? { worktreeId: managed.worktreeId } : {}), ...(managed.terminalHandle ? { terminalHandle: managed.terminalHandle } : {}), ...(foregroundProcessEvidence ? { foregroundProcessEvidence } : {}), @@ -2624,6 +2701,10 @@ export class PtyHandler { incarnationId: randomUUID(), pty: term, initialCwd: entry.cwd, + createdAt: Date.now(), + // Deliberately no ownerClientInstanceId: revive replays state a client serialized, which is + // not this host observing who asked for the shell. Unattested means never swept. + buffered: new RecentPtyOutputBuffer({ preserveChunkBoundaries: false, limit: REPLAY_BUFFER_MAX diff --git a/src/relay/relay-runtime-services.ts b/src/relay/relay-runtime-services.ts index 485c9e787d1..73e03242af6 100644 --- a/src/relay/relay-runtime-services.ts +++ b/src/relay/relay-runtime-services.ts @@ -45,6 +45,11 @@ export class RelayRuntimeServices { (id, paused) => this.ptyHandler.setConsumerDeliveryPaused(id, paused), (id) => this.ptyHandler.handleSourceCreditAvailable(id) ) + // Why wired after construction: the handler is built first, but PTY ownership has to be + // attested from the consumer grant the adapter holds. + this.ptyHandler.setConsumerIdentityResolver((clientId) => + this.ptyConsumerSessionAdapter.clientInstanceIdFor(clientId) + ) this.ptySourcePublication = new RelayPtySourcePublication( dispatcher, this.ptyConsumerSessionAdapter, diff --git a/src/relay/ssh-pty-consumer-session-adapter.ts b/src/relay/ssh-pty-consumer-session-adapter.ts index 83e61f17fa3..906edd47ff8 100644 --- a/src/relay/ssh-pty-consumer-session-adapter.ts +++ b/src/relay/ssh-pty-consumer-session-adapter.ts @@ -112,6 +112,12 @@ export class SshPtyConsumerSessionAdapter { }) } + /** The authenticated client identity behind a transport connection, or null when it holds no + * active grant. Used to stamp host-attested ownership on a PTY at spawn. */ + clientInstanceIdFor(clientId: number): string | null { + return this.session.activeClientInstanceId(String(clientId)) + } + openDelivery( clientId: number, id: string, diff --git a/src/shared/foreground-process-evidence.ts b/src/shared/foreground-process-evidence.ts index 08f9f208dbb..bbcfd091d60 100644 --- a/src/shared/foreground-process-evidence.ts +++ b/src/shared/foreground-process-evidence.ts @@ -2,12 +2,35 @@ export type ForegroundEvidenceObservation = { authorityGeneration: string observationEpoch: number - /** Age at serialization; receivers rebase this onto their monotonic clock. */ + /** How old the underlying process-table capture was when this record was serialized, measured on + * the OBSERVING host's clock so no clock skew enters it. Receivers rebase it onto their own + * monotonic clock by adding the time since the carrying response arrived. + * + * It is an upper bound, not an estimate: the capture is TTL-shared, so a reader may be served + * one up to `PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS` older than its own await, and the + * producer stamps for that worst case. Erring old is the safe direction for every consumer — + * the only one that acts destructively refuses stale evidence. */ capturedAgeMs: number } export type ForegroundProcessEvidence = - | ({ verdict: 'live'; processName: string | null } & ForegroundEvidenceObservation) + | ({ + verdict: 'live' + processName: string | null + /** True only when the host observed every process group attached to this PTY's terminal to be + * the shell's own, with none of them stopped — i.e. nothing is running in the pane, in the + * foreground OR the background, and nothing sits suspended. + * + * Deliberately not `tpgid === pgid`: a job the user backgrounded with `&` and a job the user + * suspended with Ctrl-Z both hand the terminal back to the shell, so a foreground-only + * predicate reads them as idle. This one is measured against the same set of process groups + * a forced stop would SIGKILL. + * + * False means something IS running, named or not. Absent from a host that predates the + * field, which is neither: a reader deciding whether the pane is idle must require `true` + * and defer on anything else. */ + shellOwnsEveryTtyProcessGroup?: boolean + } & ForegroundEvidenceObservation) | ({ verdict: 'unverifiable'; reason: string } & ForegroundEvidenceObservation) export function isForegroundProcessEvidence(value: unknown): value is ForegroundProcessEvidence { @@ -30,6 +53,12 @@ export function isForegroundProcessEvidence(value: unknown): value is Foreground return false } if (input.verdict === 'live') { + if ( + input.shellOwnsEveryTtyProcessGroup !== undefined && + typeof input.shellOwnsEveryTtyProcessGroup !== 'boolean' + ) { + return false + } return input.processName === null || typeof input.processName === 'string' } return ( diff --git a/src/shared/process-table-snapshot.ts b/src/shared/process-table-snapshot.ts index 17243d17b96..599489f0c99 100644 --- a/src/shared/process-table-snapshot.ts +++ b/src/shared/process-table-snapshot.ts @@ -323,6 +323,11 @@ const processTableReader = createProcessTableSnapshotReader now: () => Date.now() }) +/** How much older than its own await a snapshot from the shared reader may be. The reader serves a + * capture from its TTL cache, so a caller that needs to state the observation's age must assume + * this whole window rather than the instant its await settled. */ +export const PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS = DEFAULT_SNAPSHOT_TTL_MS + /** * Run (or reuse a recent) `ps -axo` process-table scan and return * its parsed rows. Per-process singleton: the relay and local main processes diff --git a/src/shared/pty-consumer-session.ts b/src/shared/pty-consumer-session.ts index 6b040be47cf..7c5b29aacfd 100644 --- a/src/shared/pty-consumer-session.ts +++ b/src/shared/pty-consumer-session.ts @@ -153,6 +153,15 @@ export class PtyConsumerSession { return client?.state === 'active' ? client.grant : null } + /** The authenticated client identity behind an active connection, or null. + * + * Why the host reads it here instead of taking a spawn parameter: this is what makes a later + * "this PTY belongs to you" attestation evidence rather than an echo of what a caller claimed. */ + activeClientInstanceId(connectionId: string): string | null { + const client = this.clients.get(connectionId) + return client?.state === 'active' ? client.clientInstanceId : null + } + private admissionFor( client: ClientRecord, displacedOwner?: Readonly diff --git a/src/shared/ssh-relay-pty-ownership-proof.test.ts b/src/shared/ssh-relay-pty-ownership-proof.test.ts new file mode 100644 index 00000000000..1abfc0ef7f5 --- /dev/null +++ b/src/shared/ssh-relay-pty-ownership-proof.test.ts @@ -0,0 +1,265 @@ +// #9819. Every case here is the same question asked from a different angle: can this client PROVE +// the host is holding a process nobody can reach? A "no" has to mean "leave it running". +import { describe, expect, it } from 'vitest' +import type { ForegroundProcessEvidence } from './foreground-process-evidence' +import { + planRelayPtySweep, + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + RELAY_PTY_SWEEP_MAX_PER_PASS, + RELAY_PTY_SWEEP_MIN_AGE_MS, + type RelayPtyOwnershipEvidence, + type RelayPtySweepContext +} from './ssh-relay-pty-ownership-proof' + +const OURS = 'client-instance-ours' + +const OBSERVATION = { authorityGeneration: 'gen-1', observationEpoch: 1, capturedAgeMs: 0 } + +/** The host looked at the pane and saw its own shell owning the terminal: nothing is running. */ +function idleShell(): ForegroundProcessEvidence { + return { ...OBSERVATION, verdict: 'live', processName: null, shellOwnsEveryTtyProcessGroup: true } +} + +function orphan(overrides: Partial = {}): RelayPtyOwnershipEvidence { + return { + ptyId: 'pty-1', + incarnationId: 'inc-1', + ownerClientInstanceId: OURS, + hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS * 2, + paneBound: true, + foregroundProcessEvidence: idleShell(), + ...overrides + } +} + +function context(overrides: Partial = {}): RelayPtySweepContext { + return { + clientInstanceId: OURS, + isSessionOwner: true, + routedPtyIds: new Set(), + expiredLeasePtyIds: new Set(), + minimumHostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS, + evidenceAgeSinceListingMs: 0, + maximumEvidenceAgeMs: RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS, + ...overrides + } +} + +function reasonFor(plan: ReturnType, ptyId: string): string | undefined { + return plan.skipped.find((entry) => entry.ptyId === ptyId)?.reason +} + +describe('planRelayPtySweep', () => { + it('sweeps a pane PTY this host attests we created and we have lost every route to', () => { + const plan = planRelayPtySweep([orphan()], context()) + + expect(plan.sweep).toEqual([{ ptyId: 'pty-1', incarnationId: 'inc-1' }]) + }) + + it('never sweeps a PTY this client still routes to', () => { + const plan = planRelayPtySweep([orphan()], context({ routedPtyIds: new Set(['pty-1']) })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('this client still has a route to it') + }) + + it('never sweeps a PTY whose lease this client expired without ordering a stop', () => { + // `expired` is what supersedeSiblingLeasesForPane, a reattach that failed on the transport, and + // a pane whose surface left the layout all write, and every one of them deliberately leaves the + // remote process running. Losing our handle is `unverifiable`; it is not abandonment. + const plan = planRelayPtySweep([orphan()], context({ expiredLeasePtyIds: new Set(['pty-1']) })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('this client expired its lease without ordering a stop') + }) + + it('never sweeps a pane the host observes running a named foreground process', () => { + // The hand-launched agent: the user typed `claude` in a pane, so Orca registered no agent + // session and agentSessionOwners is empty. Only the host's own observation can see it. + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: 'claude', + shellOwnsEveryTtyProcessGroup: false + } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host observes a named foreground process') + }) + + it('never sweeps a pane whose foreground group is not the shell, even unnamed', () => { + // A build, a test run, an editor: nothing recognizes it, but the host can still see that the + // terminal's foreground process group is not the shell's own. + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'live', + processName: null, + shellOwnsEveryTtyProcessGroup: false + } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host does not attest an idle shell') + }) + + it('never sweeps when the host could not observe the pane at all', () => { + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...OBSERVATION, + verdict: 'unverifiable', + reason: 'table_unreadable' + } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host could not observe the pane foreground process') + }) + + it('never sweeps a PTY the host attributes to another client instance', () => { + // The case that makes local absence useless as evidence: a second machine on the same build + // connects to the same relay, and its live agents are missing from our store exactly like an + // orphan is. + const plan = planRelayPtySweep( + [orphan({ ownerClientInstanceId: 'client-instance-theirs' })], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host attests another client created it') + }) + + it('never sweeps a PTY younger than the floor', () => { + const plan = planRelayPtySweep( + [orphan({ hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS - 1 })], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('younger than the sweep floor') + }) + + it('never sweeps a bare host shell', () => { + const plan = planRelayPtySweep([orphan({ paneBound: false })], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('not a pane-bound PTY') + }) + + it('never sweeps a PTY whose agent session the host still advertises as adoptable', () => { + const plan = planRelayPtySweep( + [orphan({ agentSessionOwners: [{ ptyId: 'pty-1' }] })], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host still advertises an adoptable agent session') + }) + + it('never sweeps without the negotiated session-owner grant', () => { + const plan = planRelayPtySweep([orphan()], context({ isSessionOwner: false })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('this client does not hold the relay session-owner grant') + }) + + it('refuses a pass larger than the per-pass ceiling instead of truncating it', () => { + const entries = Array.from({ length: RELAY_PTY_SWEEP_MAX_PER_PASS + 1 }, (_, index) => + orphan({ ptyId: `pty-${index}`, incarnationId: `inc-${index}` }) + ) + + const plan = planRelayPtySweep(entries, context()) + + expect(plan.sweep).toEqual([]) + expect(plan.skipped).toHaveLength(entries.length) + }) + + describe('against a host that predates the attestation', () => { + // Mixed versions: every new field is optional, and an older host publishes none of them. The + // sweep has to read each absence as "unknown", never as a permissive default. + it('skips an entry with no owner attestation', () => { + const { ownerClientInstanceId: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host attested no owning client') + }) + + it('skips an entry with no published age', () => { + const { hostAgeMs: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host published no age') + }) + + it('skips an entry with no paneBound field', () => { + const { paneBound: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('not a pane-bound PTY') + }) + + it('skips an entry with no foreground observation', () => { + const { foregroundProcessEvidence: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host published no foreground-process observation') + }) + + it('skips an entry from a host that observes the pane but cannot say the shell is idle', () => { + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { ...OBSERVATION, verdict: 'live', processName: null } + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host does not attest an idle shell') + }) + + it('skips an entry with no incarnation, so no stop is ever unfenced', () => { + const { incarnationId: _absent, ...legacy } = orphan() + + const plan = planRelayPtySweep([legacy], context()) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host published no PTY incarnation') + }) + + it('sweeps nothing at all when the whole listing predates the fields', () => { + const legacy = [ + { ptyId: 'pty-1', incarnationId: 'inc-1' }, + { ptyId: 'pty-2', incarnationId: 'inc-2' } + ] + + expect(planRelayPtySweep(legacy, context()).sweep).toEqual([]) + }) + }) +}) diff --git a/src/shared/ssh-relay-pty-ownership-proof.ts b/src/shared/ssh-relay-pty-ownership-proof.ts new file mode 100644 index 00000000000..ed87be13610 --- /dev/null +++ b/src/shared/ssh-relay-pty-ownership-proof.ts @@ -0,0 +1,226 @@ +import type { ForegroundProcessEvidence } from './foreground-process-evidence' + +/** Which relay PTYs a client may prove it orphaned, and therefore may stop (#9819). + * + * A relay PTY is a child of the detached relay daemon. Stopping one destroys a running process — + * often a running agent — on the user's remote machine, and the relay's 50-slot cap is a far + * cheaper failure than that. So every rule below is written to answer "can this client PROVE + * nobody owns this?" and to answer "no" whenever it cannot. + * + * The rule #9819 proposed — "pane-bound and the app no longer owns or leases it" — is not that + * proof. Absence from a client-side set is `unverifiable` by construction + * (`docs/reference/ssh-execution-boundary.md`): a second machine running the same Orca build + * connects to the SAME relay and displaces the session owner, and its PTYs are missing from THIS + * client's store for exactly the same reason a genuine orphan is. Sweeping on local absence alone + * would let one laptop reap another laptop's live agents. + * + * What replaces it: the host itself records which authenticated consumer identity asked it to + * create each PTY, and publishes that back. A PTY is sweepable only when the OWNING HOST names + * this client as its creator and this client's own durable state has no route to it. Both halves + * are required; either alone is a guess. + */ + +/** One `pty.listProcesses` entry, as far as this decision is concerned. Every field a host may + * omit is optional here, because a host predating it publishes nothing rather than a default. */ +export type RelayPtyOwnershipEvidence = { + /** Relay-scoped PTY id. */ + ptyId: string + incarnationId?: string + ownerClientInstanceId?: string + hostAgeMs?: number + paneBound?: boolean + /** Non-empty when the host still advertises an adoptable agent session on this PTY. */ + agentSessionOwners?: readonly unknown[] + /** What the OWNING host saw in the pane on the same listing. This answers a different question + * from `agentSessionOwners`: that one asks whether Orca REGISTERED an agent session here, this + * one asks whether anything at all is running. A `claude` the user typed by hand registers + * nothing, so only this can see it. */ + foregroundProcessEvidence?: ForegroundProcessEvidence +} + +export type RelayPtySweepContext = { + /** This client's persisted consumer identity for the target. */ + clientInstanceId: string + /** Whether the relay granted THIS connection the `session-owner` role. A subscriber, or a client + * that fell back to the unnegotiated legacy path, never sweeps. */ + isSessionOwner: boolean + /** Every relay PTY id this client still has any route to: a live provider PTY, a lease it has + * not tombstoned, an id it just reattached, or a stop it has recorded and not yet delivered. */ + routedPtyIds: ReadonlySet + /** Relay PTY ids this client holds an `expired` lease for. + * + * Separate from {@link routedPtyIds} because it is a different fact with the same verdict: an + * expired lease records that THIS CLIENT lost its handle — a pane re-leased under a new relay + * id, a pane surface that is no longer in the layout, a retired reattach. The layout case is the + * one that matters most, because it is reached only AFTER `pty.attach` succeeded: the process is + * not merely unproven, it is known to be alive. Every one of those writers deliberately declines + * to stop it, and client-side absence is `unverifiable` by construction + * (`docs/reference/ssh-execution-boundary.md`). So an expired lease is the record of a process + * left running on purpose, never a licence to kill it. */ + expiredLeasePtyIds: ReadonlySet + /** Host-measured age a PTY must exceed. Guards a spawn that is in flight from another window of + * this same client and has not written its lease yet. */ + minimumHostAgeMs: number + /** How long ago, on THIS client's clock, the listing that carried the evidence arrived. Added to + * each entry's host-stamped `capturedAgeMs` so {@link maximumEvidenceAgeMs} bounds staleness at + * the moment of the decision rather than at the moment of serialization. The transit itself is + * unmeasured — the two clocks are not synchronized — but it is bounded by the listing's own RPC + * deadline, and both halves that ARE measurable are counted. */ + evidenceAgeSinceListingMs: number + /** Oldest foreground observation that may authorize a stop. Stale evidence degrades to "do not + * sweep", never to "sweep". */ + maximumEvidenceAgeMs: number +} + +export type RelayPtySweepTarget = { ptyId: string; incarnationId: string } + +export type RelayPtySweepSkip = { ptyId: string; reason: string } + +export type RelayPtySweepPlan = { + sweep: RelayPtySweepTarget[] + skipped: RelayPtySweepSkip[] +} + +/** Deliberately longer than any single connect round trip. A PTY younger than this is never worth + * the risk: the leak it represents costs one slot for 30 more seconds, and reaping a shell that a + * concurrent spawn is still recording costs the user a terminal. */ +export const RELAY_PTY_SWEEP_MIN_AGE_MS = 30_000 + +/** Bounds one pass. A relay is capped at 50 PTYs, so a pass that wants to stop more than this is + * not reclaiming a leak — it is a disagreement about ownership, and stopping is the wrong move. */ +export const RELAY_PTY_SWEEP_MAX_PER_PASS = 8 + +/** The oldest foreground observation this sweep will treat as authorization to SIGKILL. + * + * Sized to the pass budget rather than to the 30s spawn floor: those answer different questions. + * The floor guards a concurrent spawn this client has not recorded yet; this one guards the pane + * the user started working in AFTER the host looked. An observation older than the whole pass it + * is meant to authorize cannot have been taken for this pass, so it is not evidence about now. + * + * It does not remove the race — nothing can, the host cannot re-check between the answer and the + * signal — it bounds it. The display consumer of the same measurement deliberately keeps NO age + * budget: a stale pane title costs a redraw and self-corrects on the next poll, so one truthful + * number carries two explicit budgets rather than one implicit one. */ +export const RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS = 5_000 + +/** The host's own answer to "is anything running in this pane?". Only a positive "no" clears the + * sweep; every other shape — an older host, an unreadable process table, an observation too old to + * describe now, a named foreground process, any other process group on the pane's terminal — is a + * reason to leave the process alone. */ +function foregroundSkipReason( + evidence: ForegroundProcessEvidence | undefined, + context: RelayPtySweepContext +): string | null { + if (evidence === undefined) { + // A host that never published it, or a Windows host where it is not collected. Absence of the + // observation is not the observation of absence. + return 'host published no foreground-process observation' + } + // Before anything is read out of it: an observation is only a claim about the instant it was + // taken. Age is checked on both verdicts because a stale `unverifiable` is no better. + if (evidence.capturedAgeMs + context.evidenceAgeSinceListingMs > context.maximumEvidenceAgeMs) { + return 'host foreground observation is too old to authorize a stop' + } + if (evidence.verdict !== 'live') { + return 'host could not observe the pane foreground process' + } + if (evidence.processName !== null) { + // The host named something running in the pane. It registered no agent session, which is + // exactly the hand-launched `claude`/`codex` case agentSessionOwners cannot see. + return 'host observes a named foreground process' + } + if (evidence.shellOwnsEveryTtyProcessGroup !== true) { + // Something other than the shell's own process group is attached to the pane's terminal — a + // foreground command, a job backgrounded with `&`, a Ctrl-Z'd editor — or this host predates + // the field. The stop would SIGKILL that group, so none of those is a pane to reclaim. + return 'host does not attest an idle shell' + } + return null +} + +function skipReason( + entry: RelayPtyOwnershipEvidence, + context: RelayPtySweepContext +): string | null { + if (typeof entry.incarnationId !== 'string' || entry.incarnationId.length === 0) { + // Without the host's own incarnation there is no fence, and an unfenced stop aimed at a relay + // id can hit whatever holds that id by the time it lands. + return 'host published no PTY incarnation' + } + if (typeof entry.ownerClientInstanceId !== 'string' || entry.ownerClientInstanceId.length === 0) { + return 'host attested no owning client' + } + if (entry.ownerClientInstanceId !== context.clientInstanceId) { + return 'host attests another client created it' + } + if (entry.paneBound !== true) { + // Covers both a bare host shell (a remote CLI terminal nobody's pane owns) and a host that + // never published the field. Neither is a pane this client lost. + return 'not a pane-bound PTY' + } + if (entry.agentSessionOwners !== undefined && entry.agentSessionOwners.length > 0) { + // The host still advertises this session as adoptable, so a later spawn can reclaim the running + // agent. Reaping it converts a recoverable session into a destroyed one. + return 'host still advertises an adoptable agent session' + } + const foregroundSkip = foregroundSkipReason(entry.foregroundProcessEvidence, context) + if (foregroundSkip !== null) { + return foregroundSkip + } + if (typeof entry.hostAgeMs !== 'number' || !Number.isFinite(entry.hostAgeMs)) { + return 'host published no age' + } + if (entry.hostAgeMs < context.minimumHostAgeMs) { + return 'younger than the sweep floor' + } + if (context.routedPtyIds.has(entry.ptyId)) { + return 'this client still has a route to it' + } + if (context.expiredLeasePtyIds.has(entry.ptyId)) { + return 'this client expired its lease without ordering a stop' + } + return null +} + +/** Plans one sweep pass. Pure: every input is evidence the caller already gathered, so the rule can + * be tested without a relay, and the irreversible call sits with the caller. */ +export function planRelayPtySweep( + entries: readonly RelayPtyOwnershipEvidence[], + context: RelayPtySweepContext +): RelayPtySweepPlan { + if (!context.isSessionOwner || !context.clientInstanceId) { + return { + sweep: [], + skipped: entries.map((entry) => ({ + ptyId: entry.ptyId, + reason: 'this client does not hold the relay session-owner grant' + })) + } + } + const sweep: RelayPtySweepTarget[] = [] + const skipped: RelayPtySweepSkip[] = [] + for (const entry of entries) { + const reason = skipReason(entry, context) + if (reason !== null) { + skipped.push({ ptyId: entry.ptyId, reason }) + } else { + sweep.push({ ptyId: entry.ptyId, incarnationId: entry.incarnationId as string }) + } + } + if (sweep.length > RELAY_PTY_SWEEP_MAX_PER_PASS) { + // Why refuse rather than truncate: at this size the disagreement is about ownership, not about + // a handful of leaked slots, and a truncated pass would work through the same list one connect + // at a time and destroy it all anyway. + return { + sweep: [], + skipped: [ + ...skipped, + ...sweep.map((target) => ({ + ptyId: target.ptyId, + reason: `refusing a ${sweep.length}-PTY sweep; over the ${RELAY_PTY_SWEEP_MAX_PER_PASS} per-pass ceiling` + })) + ] + } + } + return { sweep, skipped } +} From 7c6c8ef85e143ae4062475f13b418cc6ba45e42e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:17 -0700 Subject: [PATCH 112/398] fix(ssh): stop a late SFTP stream error crashing main, and keep the relay socket inside sun_path (#17862) * fix(ssh): stop SFTP stream errors crashing main and bound the relay socket path inside the protocol parser. Every transfer removed its listener on settle, so a STATUS reply that arrived late - the normal case behind a jump host that chroots its SFTP subsystem - threw synchronously out of Socket.emit('data') and killed the main process. Keep one durable listener per stream, and report a sandboxed SFTP namespace with an actionable message instead of a bare 'file does not exist'. 104 macOS) and bind failed with a bare 'listen EINVAL'. Fall back to a per-uid base whose length does not depend on $HOME, keeping the hashed socket name intact. * fix(ssh): validate the short socket dir before mutating it * fix(ssh): keep the SFTP session guarded, scope the relocated socket, narrow the chroot verdict Three review findings. The CLI-launcher install ran writeStringViaSftp in a loop over a bare conn.sftp(). That helper removes its own session 'error' listener at each settle, so between files and after the last one the emitter carried none -- and ssh2 raises a late STATUS reply synchronously out of Protocol.parse, which is the uncaught exception that kills main (#15479). The inline loop it replaced leaked one listener per file and covered this by accident. Extract writeStringsViaSftp, which owns the session latch, and share that latch with runSftpFallbackTransfer. SSH_FX_PERMISSION_DENIED is a mode/ownership refusal on a path the subsystem can see, not evidence of a chroot; sftp-namespace-resolution already treats only NO_SUCH_FILE as conclusive. Narrow the predicate to code 2 so a read-only home stops being reported as a bastion misconfiguration. The relocated socket had no version dimension. relaySocketNameForInstanceId hashes the target, not the build, and under $HOME the enclosing relay- dir supplied the rest -- so the short form made the path stable across updates. The next build would bind the path the previous relay still holds, the handshake would mismatch, and a relay holding live work would raise RelayEndpointHeldError with no way through. Add a hashed version segment under the short base, mirroring the relay-*/ shape so one pattern serves both, and teach the superseded sweep and force-stop about that base. The relocated tree now also gets reclaimed: nothing else walks it. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- ...ocket-path-limit-shell.integration.test.ts | 103 ++++++++ src/main/ssh/relay-socket-path-limit.test.ts | 250 ++++++++++++++++++ src/main/ssh/relay-socket-path-limit.ts | 128 +++++++++ src/main/ssh/sftp-stream-late-error.test.ts | 172 ++++++++++++ src/main/ssh/sftp-stream-late-error.ts | 108 ++++++++ src/main/ssh/sftp-upload.test.ts | 8 +- src/main/ssh/sftp-upload.ts | 37 +++ src/main/ssh/ssh-relay-deploy.ts | 66 ++++- src/main/ssh/ssh-relay-install-transfers.ts | 54 +++- src/main/ssh/ssh-relay-reset.ts | 7 +- src/main/ssh/ssh-relay-session.ts | 16 +- .../ssh/ssh-relay-superseded-endpoints.ts | 22 +- 12 files changed, 944 insertions(+), 27 deletions(-) create mode 100644 src/main/ssh/relay-socket-path-limit-shell.integration.test.ts create mode 100644 src/main/ssh/relay-socket-path-limit.test.ts create mode 100644 src/main/ssh/relay-socket-path-limit.ts create mode 100644 src/main/ssh/sftp-stream-late-error.test.ts create mode 100644 src/main/ssh/sftp-stream-late-error.ts diff --git a/src/main/ssh/relay-socket-path-limit-shell.integration.test.ts b/src/main/ssh/relay-socket-path-limit-shell.integration.test.ts new file mode 100644 index 00000000000..ce08556c3d7 --- /dev/null +++ b/src/main/ssh/relay-socket-path-limit-shell.integration.test.ts @@ -0,0 +1,103 @@ +import { execFile } from 'node:child_process' +import { chmod, mkdtemp, mkdir, rm, symlink, stat } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { promisify } from 'node:util' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + resolveShortRelaySocketDirCommand, + shortRelayVersionSegment +} from './relay-socket-path-limit' + +const run = promisify(execFile) + +// The generated script runs on the remote host's /bin/sh, so assert against a real shell rather +// than a string match: the hazard here is an ordering bug that only a filesystem can observe. +describe('short relay socket dir guard, against a real shell', () => { + let root: string + + const VERSION_SEGMENT = shortRelayVersionSegment('relay-0.1.0+test') + + // Retarget the generated script at a sandbox instead of the real /tmp path. + function scriptFor(dir: string): string { + return resolveShortRelaySocketDirCommand(VERSION_SEGMENT).replace( + /^dir=.*$/m, + `dir=${JSON.stringify(dir)}` + ) + } + + async function attempt(dir: string): Promise<{ ok: boolean; stdout: string }> { + try { + const { stdout } = await run('/bin/sh', ['-c', scriptFor(dir)]) + return { ok: true, stdout } + } catch { + return { ok: false, stdout: '' } + } + } + + beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-relay-dir-guard-')) + }) + + afterEach(async () => { + await rm(root, { recursive: true, force: true }) + }) + + it('creates the directory and its version segment when neither exists', async () => { + const dir = join(root, 'fresh') + const attempted = await attempt(dir) + expect(attempted.ok).toBe(true) + expect((await stat(dir)).mode & 0o777).toBe(0o700) + // The segment is what keeps a later build off the path this one binds. + expect((await stat(join(dir, VERSION_SEGMENT))).mode & 0o777).toBe(0o700) + expect(attempted.stdout.trim().endsWith(`${dir}/${VERSION_SEGMENT}`)).toBe(true) + }) + + it('refuses a planted symlink in the version segment without following it', async () => { + const dir = join(root, 'mine') + const victim = join(root, 'segment-victim') + await mkdir(dir) + await chmod(dir, 0o700) + await mkdir(victim) + await chmod(victim, 0o755) + await symlink(victim, join(dir, VERSION_SEGMENT)) + + const before = (await stat(victim)).mode & 0o777 + expect((await attempt(dir)).ok).toBe(false) + expect((await stat(victim)).mode & 0o777).toBe(before) + }) + + it('adopts a directory we already own at 0700, so reconnects keep working', async () => { + // Regression guard: `ls` decorates the mode with @ (xattrs), + (ACL) or . (SELinux), and an + // exact match refused a directory we own — which would have broken every reconnect. + const dir = join(root, 'mine') + await mkdir(dir) + await chmod(dir, 0o700) + expect((await attempt(dir)).ok).toBe(true) + expect((await attempt(dir)).ok).toBe(true) + }) + + it('refuses a planted symlink without changing what it points at', async () => { + const victim = join(root, 'victim') + const link = join(root, 'link') + await mkdir(victim) + await chmod(victim, 0o755) + await symlink(victim, link) + + const before = (await stat(victim)).mode & 0o777 + expect((await attempt(link)).ok).toBe(false) + // The point of the ordering: an unconditional chmod would have followed the link and + // rewritten the victim's mode before the owner check ever ran. + expect((await stat(victim)).mode & 0o777).toBe(before) + }) + + it('refuses an existing directory that is not 0700', async () => { + const dir = join(root, 'loose') + await mkdir(dir) + // Explicit chmod: mkdir's mode is masked by the process umask, so the fixture would not + // actually be world-writable and the test would not be testing what it claims. + await chmod(dir, 0o777) + expect((await attempt(dir)).ok).toBe(false) + expect((await stat(dir)).mode & 0o777).toBe(0o777) + }) +}) diff --git a/src/main/ssh/relay-socket-path-limit.test.ts b/src/main/ssh/relay-socket-path-limit.test.ts new file mode 100644 index 00000000000..7334ff9d914 --- /dev/null +++ b/src/main/ssh/relay-socket-path-limit.test.ts @@ -0,0 +1,250 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + app: { getAppPath: () => '/mock/app' } +})) + +vi.mock('fs', () => ({ + existsSync: vi.fn().mockReturnValue(true), + readFileSync: vi.fn().mockReturnValue('0.1.0+abcdef012345') +})) + +vi.mock('./relay-protocol', () => ({ + RELAY_VERSION: '0.1.0', + RELAY_REMOTE_DIR: '.orca-remote', + parseUnameToRelayPlatform: vi.fn(() => 'linux-x64'), + RELAY_SENTINEL: 'ORCA-RELAY v0.1.0 READY\n', + RELAY_SENTINEL_TIMEOUT_MS: 10_000 +})) + +vi.mock('./ssh-relay-deploy-helpers', () => ({ + uploadDirectory: vi.fn().mockResolvedValue(undefined), + waitForSentinel: vi.fn().mockResolvedValue({ + write: vi.fn(), + onData: vi.fn(), + onClose: vi.fn() + }), + isUnconfirmedSshCommandTermination: () => false, + execCommand: vi.fn().mockResolvedValue('') +})) + +vi.mock('./ssh-remote-node-resolution', () => ({ + resolveRemoteNodePath: vi.fn().mockResolvedValue('/usr/bin/node') +})) + +vi.mock('./ssh-relay-endpoint-credential', () => ({ + writeRelayEndpointCredential: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-relay-versioned-install', () => ({ + readLocalFullVersion: vi.fn().mockReturnValue('0.1.0+8d4e15ad63eb'), + computeRemoteRelayDir: (home: string, v: string) => `${home}/.orca-remote/relay-${v}`, + isRelayAlreadyInstalled: vi.fn().mockResolvedValue(true), + finalizeInstall: vi.fn().mockResolvedValue(undefined), + abandonInstall: vi.fn().mockResolvedValue(undefined), + gcOldRelayVersions: vi.fn().mockResolvedValue(undefined) +})) + +vi.mock('./ssh-relay-install-lock', () => ({ + acquireInstallLock: vi.fn().mockResolvedValue(undefined), + RELAY_INSTALL_LOCK_NAME: '.install-lock' +})) + +vi.mock('./ssh-relay-repair-lock', () => ({ + tryAcquireRelayRepairLock: vi.fn().mockResolvedValue('acquired') +})) + +vi.mock('./ssh-connection-utils', () => ({ + shellEscape: (s: string) => `'${s}'`, + createSshOperationAbortError: () => + Object.assign(new Error('SSH operation was cancelled'), { name: 'AbortError' }) +})) + +import { deployAndLaunchRelay } from './ssh-relay-deploy' +import { execCommand } from './ssh-relay-deploy-helpers' +import { forceStopRelayForTarget } from './ssh-relay-reset' +import { relaySocketNameForInstanceId } from './ssh-relay-instance-id' +import { + parseShortRelaySocketDir, + remoteSocketPathFitsLimit, + remoteUnixSocketPathByteLimit, + shortRelayVersionSegment, + SHORT_RELAY_SOCKET_DIR_PREFIX +} from './relay-socket-path-limit' +import { supersededRelayEndpointListCommand } from './ssh-relay-superseded-endpoints' +import { getRemoteHostPlatform } from './ssh-remote-platform' +import type { SshConnection } from './ssh-connection' + +const LINUX = getRemoteHostPlatform('linux-x64') +const DARWIN = getRemoteHostPlatform('darwin-arm64') +const WINDOWS = getRemoteHostPlatform('win32-x64') + +// The reporter's host: a managed-hosting container whose $HOME is 45 bytes (#10726). +const LONG_HOME = '/var/www/611f7cf9-f715-49e6-91d9-0ffac1d7c4c0' + +/** Matches the version this suite's mocked build reports. */ +const RELAY_VERSION_DIR_NAME = 'relay-0.1.0+8d4e15ad63eb' + +function makeMockConnection(): SshConnection { + return { + canRunConcurrentExecCommands: vi.fn().mockReturnValue(true), + exec: vi.fn().mockResolvedValue({ + on: vi.fn(), + stderr: { on: vi.fn() }, + stdin: {}, + stdout: { on: vi.fn() }, + close: vi.fn() + }), + writeFile: vi.fn().mockResolvedValue(undefined), + sftp: vi.fn().mockResolvedValue({ + mkdir: vi.fn((_p: string, cb: (err: Error | null) => void) => cb(null)), + createWriteStream: vi.fn().mockReturnValue({ + on: vi.fn((event: string, cb: () => void) => { + if (event === 'close') { + setTimeout(cb, 0) + } + }), + end: vi.fn() + }), + end: vi.fn() + }) + } as unknown as SshConnection +} + +function launchedSockPath(conn: SshConnection): string { + const launch = vi + .mocked(conn.exec) + .mock.calls.map(([command]) => command as string) + .find((command) => command.includes('--detached')) + return /--sock-path\s+'([^']+)'/.exec(launch ?? '')?.[1] ?? '' +} + +describe('remote unix socket path limit', () => { + it('uses the per-OS sun_path budget and ignores Windows named pipes', () => { + expect(remoteUnixSocketPathByteLimit(LINUX)).toBe(107) + expect(remoteUnixSocketPathByteLimit(DARWIN)).toBe(103) + expect(remoteUnixSocketPathByteLimit(WINDOWS)).toBeNull() + expect(remoteSocketPathFitsLimit(WINDOWS, `\\\\.\\pipe\\orca-relay-${'a'.repeat(400)}`)).toBe( + true + ) + }) + + it('measures bytes, not characters', () => { + // 1 + 52 two-byte characters = 105 bytes: fits Linux (107), not macOS (103). + const path = `/${'é'.repeat(52)}` + expect(path.length).toBe(53) + expect(remoteSocketPathFitsLimit(LINUX, path)).toBe(true) + expect(remoteSocketPathFitsLimit(DARWIN, path)).toBe(false) + }) + + it('accepts only the marker line as the short directory', () => { + const segment = shortRelayVersionSegment(RELAY_VERSION_DIR_NAME) + expect( + parseShortRelaySocketDir( + `Welcome to Ubuntu\nORCA-RELAY-SHORT-SOCKET-DIR /tmp/.orca-relay-1000/${segment}\n`, + segment + ) + ).toBe(`/tmp/.orca-relay-1000/${segment}`) + expect(parseShortRelaySocketDir('mkdir: permission denied\n', segment)).toBeNull() + expect( + parseShortRelaySocketDir(`ORCA-RELAY-SHORT-SOCKET-DIR /etc/${segment}\n`, segment) + ).toBeNull() + // A directory belonging to another build must not be adopted as this build's. + expect( + parseShortRelaySocketDir( + `ORCA-RELAY-SHORT-SOCKET-DIR /tmp/.orca-relay-1000/${shortRelayVersionSegment('relay-9.9.9+other')}\n`, + segment + ) + ).toBeNull() + }) +}) + +describe('relay launch with a long remote $HOME', () => { + beforeEach(() => { + vi.clearAllMocks() + }) + + it('keeps the launched socket path inside the remote sun_path limit', async () => { + const conn = makeMockConnection() + vi.mocked(execCommand) + .mockReset() + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ Linux x86_64') + .mockResolvedValueOnce(LONG_HOME) + .mockResolvedValueOnce('ORCA-NATIVE-DEPS-OK') + .mockResolvedValueOnce('') // launch namespace marker + .mockResolvedValueOnce( + `ORCA-RELAY-SHORT-SOCKET-DIR ${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/${shortRelayVersionSegment(RELAY_VERSION_DIR_NAME)}` + ) + .mockResolvedValueOnce('DEAD') + .mockResolvedValueOnce('READY') + .mockResolvedValue('') + + // A per-target relay instance id is what pushes the default path past the limit: + // 45-byte $HOME + `/.orca-remote/relay-0.1.0+8d4e15ad63eb` + `/relay-.sock` = 110 bytes. + const result = await deployAndLaunchRelay(conn, undefined, undefined, 'ssh-target-1') + + const sockPath = launchedSockPath(conn) + expect(sockPath).not.toBe('') + expect(Buffer.byteLength(sockPath, 'utf8')).toBeLessThanOrEqual( + remoteUnixSocketPathByteLimit(LINUX) as number + ) + expect(sockPath.startsWith(`${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/`)).toBe(true) + expect(result.sockPath).toBe(sockPath) + // The hashed socket name survives intact, so two targets cannot collide -- and the + // build's version segment sits above it, so the next Orca release binds a path of + // its own instead of the one this relay is still holding. + expect(sockPath).toBe( + `${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/${shortRelayVersionSegment(RELAY_VERSION_DIR_NAME)}/${relaySocketNameForInstanceId('ssh-target-1')}` + ) + expect(shortRelayVersionSegment('relay-0.1.0+next')).not.toBe( + shortRelayVersionSegment(RELAY_VERSION_DIR_NAME) + ) + }) + + it('sweeps superseded relays under the short base too, but never the live one', () => { + const currentShortSocketDir = `${SHORT_RELAY_SOCKET_DIR_PREFIX}1000/${shortRelayVersionSegment(RELAY_VERSION_DIR_NAME)}` + const script = supersededRelayEndpointListCommand({ + remoteHome: LONG_HOME, + currentRelayDir: `${LONG_HOME}/.orca-remote/${RELAY_VERSION_DIR_NAME}`, + sockName: relaySocketNameForInstanceId('ssh-target-1'), + currentShortSocketDir + }) + + // A relocated orphan lives outside $HOME, so the sweep that exists to make orphans + // visible has to look at the short base as well. + expect(script).toContain(`short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`) + expect(script).toContain('"$short_base"/relay-*/"$sock_name"') + expect(script).toContain(`short_current='${currentShortSocketDir}'`) + expect(script).toContain('[ -n "$short_current" ] && [ "$dir" = "$short_current" ] && continue') + }) + + it('leaves the socket in the versioned relay dir when it already fits', async () => { + const conn = makeMockConnection() + vi.mocked(execCommand) + .mockReset() + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ Linux x86_64') + .mockResolvedValueOnce('/home/user') + .mockResolvedValueOnce('ORCA-NATIVE-DEPS-OK') + .mockResolvedValueOnce('') + .mockResolvedValueOnce('DEAD') + .mockResolvedValueOnce('READY') + .mockResolvedValue('') + + await deployAndLaunchRelay(conn) + + expect(launchedSockPath(conn)).toBe( + '/home/user/.orca-remote/relay-0.1.0+8d4e15ad63eb/relay.sock' + ) + }) + + it('force-stop also looks for the socket under the short base', async () => { + const conn = makeMockConnection() + vi.mocked(execCommand).mockReset().mockResolvedValue('') + + await forceStopRelayForTarget(conn, 'ssh-1') + + const script = vi.mocked(execCommand).mock.calls[0]?.[1] as string + expect(script).toContain(`short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`) + expect(script).toContain('"$short_base"/relay-*/"$sock_name"') + }) +}) diff --git a/src/main/ssh/relay-socket-path-limit.ts b/src/main/ssh/relay-socket-path-limit.ts new file mode 100644 index 00000000000..417366e2bf6 --- /dev/null +++ b/src/main/ssh/relay-socket-path-limit.ts @@ -0,0 +1,128 @@ +/** + * Keeps the remote relay's Unix socket path inside `sockaddr_un.sun_path`. + * + * The default endpoint is `$HOME/.orca-remote/relay-/relay-.sock`, + * whose fixed suffix already costs ~66 bytes. A managed-hosting `$HOME` such as + * `/var/www/` pushes the whole path past the kernel cap and libuv reports only + * `listen EINVAL`, so the relay never starts (#10726). When that happens the socket + * moves to a fixed-length base whose length no longer depends on `$HOME`. + * + * Windows relays bind named pipes (`\\.\pipe\...`), which have no `sun_path` limit. + */ +import { createHash } from 'node:crypto' +import { isWindowsRemoteHost, type RemoteHostPlatform } from './ssh-remote-platform' + +/** + * `sizeof(sun_path)` per remote OS, including the terminating NUL: 108 on Linux, + * 104 on macOS/BSD. Compared against byte length, not character count — a non-ASCII + * `$HOME` costs more bytes than characters. + */ +const SUN_PATH_SIZE: Record<'linux' | 'darwin', number> = { linux: 108, darwin: 104 } + +export function remoteUnixSocketPathByteLimit(host: RemoteHostPlatform): number | null { + if (isWindowsRemoteHost(host)) { + return null + } + return SUN_PATH_SIZE[host.os === 'darwin' ? 'darwin' : 'linux'] - 1 +} + +export function remoteSocketPathFitsLimit(host: RemoteHostPlatform, sockPath: string): boolean { + const limit = remoteUnixSocketPathByteLimit(host) + return limit === null || Buffer.byteLength(sockPath, 'utf8') <= limit +} + +/** Fixed-length, per-uid base. `/tmp` is the only POSIX directory whose length is not user-dependent. */ +export const SHORT_RELAY_SOCKET_DIR_PREFIX = '/tmp/.orca-relay-' + +export function shortRelaySocketDirForUid(uid: string): string { + return `${SHORT_RELAY_SOCKET_DIR_PREFIX}${uid}` +} + +/** + * The version segment the relocated socket lives under, named to match the version + * directories in `$HOME/.orca-remote` so one sweep pattern covers both bases. + * + * Why it has to exist: `relaySocketNameForInstanceId` hashes the *target*, not the + * build, so the filename alone is version-independent. Under `$HOME` the enclosing + * `relay-` directory supplies that dimension; without it here, the next + * Orca build would bind the exact path the previous build's relay still holds. The + * daemon handshake compares build hashes exactly, so that meeting is a version + * mismatch — and if the incumbent holds live work, `resolveRelayEndpointBeforeRelaunch` + * raises `RelayEndpointHeldError` and the user cannot connect at all until the old + * relay is stopped. The version is hashed rather than spelled out because the whole + * point of this base is a bounded length. + */ +export function shortRelayVersionSegment(relayVersionDirName: string): string { + return `relay-${createHash('sha256').update(relayVersionDirName).digest('hex').slice(0, 12)}` +} + +/** + * The whole hashed socket name is kept — shortening happens by replacing the + * variable-length directory, never by truncating the hash, so two targets on one + * host can never land on the same socket. + */ +export function shortRelaySocketPath(shortVersionDir: string, sockName: string): string { + return `${shortVersionDir}/${sockName}` +} + +const SHORT_DIR_MARKER = 'ORCA-RELAY-SHORT-SOCKET-DIR' + +/** + * Create (or adopt) the per-uid short socket directory and its version segment, and + * print the segment's path. + * + * Validate before mutating, never the other way round: an unconditional `chmod` follows a + * symlink, so a path planted by another user would have its *target's* mode rewritten before + * the owner check could reject it. A fresh `mkdir` under `umask 077` already yields 0700 and + * proves we own it, so the only path that adopts an existing entry is the one that first + * proves — via `ls -ldn`, which reports the entry itself rather than what it points at — that + * it is a real directory, owned by this uid, already 0700. Nothing else is touched. + */ +export function resolveShortRelaySocketDirCommand(versionSegment: string): string { + return [ + 'uid=$(id -u) || exit 1', + `dir="${SHORT_RELAY_SOCKET_DIR_PREFIX}$uid"`, + 'umask 077', + ...adoptOwnedDirectoryCommand('$dir'), + // The version segment is validated the same way rather than trusted: `$dir` being + // 0700 and ours does not prove what an earlier run left inside it still is. + `ver="$dir/${versionSegment}"`, + ...adoptOwnedDirectoryCommand('$ver'), + `printf '%s %s\n' '${SHORT_DIR_MARKER}' "$ver"` + ].join('\n') +} + +function adoptOwnedDirectoryCommand(target: string): string[] { + return [ + `if mkdir "${target}" 2>/dev/null; then`, + ' :', + 'else', + // Why the sub(): ls decorates the mode with a trailing marker for extended attributes (@), + // ACLs (+) or an SELinux context (.), so an exact match would refuse a directory we own. + ` entry=$(ls -ldn "${target}" 2>/dev/null | awk 'NR==1{sub(/[.@+]$/, "", $1); print $1" "$3}')`, + ' case "$entry" in', + ' "drwx------ $uid") ;;', + ' *) exit 1 ;;', + ' esac', + 'fi' + ] +} + +/** Tolerates login-shell banner noise ahead of the marker line. */ +export function parseShortRelaySocketDir(output: string, versionSegment: string): string | null { + for (const line of output.split('\n')) { + const trimmed = line.trim() + if (!trimmed.startsWith(`${SHORT_DIR_MARKER} `)) { + continue + } + const dir = trimmed.slice(SHORT_DIR_MARKER.length + 1).trim() + if ( + dir.startsWith(`${SHORT_RELAY_SOCKET_DIR_PREFIX}`) && + dir.endsWith(`/${versionSegment}`) && + !/[\r\n]/.test(dir) + ) { + return dir + } + } + return null +} diff --git a/src/main/ssh/sftp-stream-late-error.test.ts b/src/main/ssh/sftp-stream-late-error.test.ts new file mode 100644 index 00000000000..4ed3bcf8b32 --- /dev/null +++ b/src/main/ssh/sftp-stream-late-error.test.ts @@ -0,0 +1,172 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import type { SFTPWrapper } from 'ssh2' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { uploadBuffer, uploadFile, writeStringViaSftp, writeStringsViaSftp } from './sftp-upload' +import { writeRelayFile } from './ssh-relay-install-transfers' +import type { SshConnection } from './ssh-connection' +import { getRemoteHostPlatform } from './ssh-remote-platform' + +/** The exact error ssh2 builds from a STATUS reply of SSH_FX_NO_SUCH_FILE. */ +function sftpNoSuchFileError(): Error { + return Object.assign(new Error('file does not exist'), { code: 2 }) +} + +let tempDir = '' +let localFile = '' + +beforeEach(async () => { + tempDir = await mkdtemp(join(tmpdir(), 'orca-sftp-late-')) + localFile = join(tempDir, 'relay.js') + await writeFile(localFile, 'console.log(1)\n') +}) + +afterEach(async () => { + await rm(tempDir, { recursive: true, force: true }) +}) + +function sftpDoubleReturning(stream: PassThrough): SFTPWrapper { + return Object.assign(new EventEmitter(), { + createWriteStream: () => stream + }) as unknown as SFTPWrapper +} + +describe('late SFTP stream errors', () => { + // ssh2 emits the OPEN failure from inside the protocol parser. If no listener is left, + // Node throws it synchronously up through Socket.emit('data') and the main process dies + // (#15479) — uncaught exceptions are re-thrown by installUncaughtPipeErrorGuard, unlike + // rejections, which are only logged. + it('does not throw when a write stream fails after uploadFile settles', async () => { + const stream = new PassThrough() + stream.resume() + + await uploadFile(sftpDoubleReturning(stream), localFile, '/home/user/.orca-remote/relay.js') + + expect(() => stream.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + it('does not throw when a write stream fails after writeStringViaSftp settles', async () => { + const stream = new PassThrough() + stream.resume() + + await writeStringViaSftp(sftpDoubleReturning(stream), '/home/user/.orca-remote/.version', 'v1') + + expect(() => stream.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + it('does not throw when a write stream fails after uploadBuffer settles', async () => { + const stream = new PassThrough() + stream.resume() + + await uploadBuffer(sftpDoubleReturning(stream), Buffer.from('x'), '/home/user/x') + + expect(() => stream.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + // The failure mode a per-file loop reintroduces: writeStringViaSftp removes its own + // session listener at each settle, so a session that ran N transfers ends up with zero + // listeners while it is still open and still able to deliver a STATUS reply. + it('does not throw when a session error arrives after a multi-file write settles', async () => { + const sftp = Object.assign(new EventEmitter(), { + createWriteStream: () => { + const stream = new PassThrough() + stream.resume() + return stream + }, + end: () => {} + }) as unknown as SFTPWrapper + + await writeStringsViaSftp({ sftp: () => Promise.resolve(sftp) }, [ + { path: '/home/user/.local/bin/orca', contents: '#!/bin/sh\n' }, + { path: '/home/user/.local/bin/orca.mjs', contents: 'export {}\n' } + ]) + + expect(() => sftp.emit('error', sftpNoSuchFileError())).not.toThrow() + }) + + it('still rejects a multi-file write with a session error raised during it', async () => { + const sftp = Object.assign(new EventEmitter(), { + createWriteStream: () => { + const stream = new PassThrough() + queueMicrotask(() => sftp.emit('error', sftpNoSuchFileError())) + return stream + }, + end: () => {} + }) as unknown as SFTPWrapper + + // The latch must sit behind the transfer's own prepended listener, or a real + // mid-transfer failure would be swallowed into a hang. + await expect( + writeStringsViaSftp({ sftp: () => Promise.resolve(sftp) }, [ + { path: '/home/user/.local/bin/orca', contents: '#!/bin/sh\n' } + ]) + ).rejects.toThrow('file does not exist') + }) + + it('still rejects with the SFTP error when it arrives during the transfer', async () => { + const stream = new PassThrough() + stream.resume() + const failing = Object.assign(new EventEmitter(), { + createWriteStream: () => { + queueMicrotask(() => stream.emit('error', sftpNoSuchFileError())) + return stream + } + }) as unknown as SFTPWrapper + + await expect(writeStringViaSftp(failing, '/home/user/x', 'v1')).rejects.toThrow( + 'file does not exist' + ) + }) +}) + +describe('sandboxed SFTP subsystem diagnosis', () => { + it('leaves a permission refusal as itself rather than blaming a chroot', async () => { + // SSH_FX_PERMISSION_DENIED is a mode/ownership refusal on a path the subsystem can + // see -- a read-only home, a root-owned parent, a quota. Rewriting it into "your + // bastion chroots SFTP" sends the user to fix ProxyJump for a chmod. + const conn = { + writeFile: () => Promise.reject(Object.assign(new Error('permission denied'), { code: 3 })) + } as unknown as SshConnection + + const failure: unknown = await writeRelayFile( + conn, + getRemoteHostPlatform('linux-x64'), + '/home/user/.orca-remote/relay-1/.version', + 'v1' + ).then( + () => null, + (err: unknown) => err + ) + + expect((failure as Error).message).toBe('permission denied') + expect(failure).not.toHaveProperty('sandboxedSftpNamespace') + }) + + it('replaces the bare SFTP status with an actionable relay-install message', async () => { + const conn = { + writeFile: () => Promise.reject(sftpNoSuchFileError()) + } as unknown as SshConnection + + await expect( + writeRelayFile( + conn, + getRemoteHostPlatform('linux-x64'), + '/home/user/.orca-remote/relay-1/.version', + 'v1' + ) + ).rejects.toThrow(/SFTP subsystem sees a different filesystem/) + }) + + it('leaves unrelated transfer failures untouched', async () => { + const conn = { + writeFile: () => Promise.reject(new Error('Connection lost')) + } as unknown as SshConnection + + await expect( + writeRelayFile(conn, getRemoteHostPlatform('linux-x64'), '/home/user/x', 'v1') + ).rejects.toThrow('Connection lost') + }) +}) diff --git a/src/main/ssh/sftp-stream-late-error.ts b/src/main/ssh/sftp-stream-late-error.ts new file mode 100644 index 00000000000..1630668f874 --- /dev/null +++ b/src/main/ssh/sftp-stream-late-error.ts @@ -0,0 +1,108 @@ +/** + * Why an SFTP stream needs an `'error'` listener that outlives its transfer. + * + * ssh2 answers an SFTP request by invoking the pending request's callback from inside + * the protocol parser, on the socket's `data` handler stack. For a write stream that + * callback is `WriteStream.open`'s, and it does a bare `this.emit('error', err)`. Node + * throws when `'error'` is emitted on an emitter with no listener, so once a transfer + * has settled and removed its listener, a late STATUS reply becomes a *synchronous + * throw* out of `Protocol.parse` -> `Socket.emit('data')`. + * + * That is an uncaught exception, not a rejection: `installUnhandledRejectionLogging` + * absorbs rejections, but `installUncaughtPipeErrorGuard` re-throws uncaught exceptions + * and the app dies (#15479). A jump host that sandboxes the SFTP subsystem into its own + * chroot makes a late `SSH_FX_NO_SUCH_FILE` the normal answer, so the listener has to + * outlive the transfer rather than the other way round. + * + * `runSftpFallbackTransfer` already does this for the session emitter; this is the same + * guarantee one level down, on the streams. + */ + +/** SSH_FX_* status code ssh2 copies onto the `Error` it builds from a STATUS reply. */ +const SSH_FX_NO_SUCH_FILE = 2 + +type ErrorEmitter = { + on(event: 'error', listener: (err: Error) => void): unknown +} + +type SftpSessionEmitter = ErrorEmitter & { + once(event: 'close', listener: () => void): unknown + removeListener(event: 'error', listener: (err: Error) => void): unknown +} + +export type SftpStreamErrorLatch = { + /** Call once the transfer has settled; any error after this point is the late one. */ + markTransferSettled(): void +} + +export function latchLateSftpStreamErrors( + stream: ErrorEmitter, + remotePath: string +): SftpStreamErrorLatch { + let settled = false + stream.on('error', (err: Error) => { + if (!settled) { + // The transfer's own listener owns this error and will reject with it. + return + } + console.warn( + `[sftp] Ignored late stream error for ${remotePath}: ${err instanceof Error ? err.message : String(err)}` + ) + }) + return { + markTransferSettled: () => { + settled = true + } + } +} + +/** + * Hold one `'error'` listener on the SFTP *session* for as long as the session lives. + * + * A transfer that attaches and removes its own session listener — `writeStringViaSftp` + * does, so a session error can reject the write in flight — leaves the emitter with zero + * listeners between transfers and after the last one. A late STATUS reply arriving in + * that window is the synchronous throw described above. Errors during a transfer still + * reach that transfer first: it prepends its listener ahead of this one. + * + * Attach this once, right after `conn.sftp()`, on every path that runs transfers over a + * session it owns. + */ +export function latchLateSftpSessionErrors(sftp: SftpSessionEmitter): void { + const swallowLateSftpError = (): void => {} + sftp.on('error', swallowLateSftpError) + sftp.once('close', () => sftp.removeListener('error', swallowLateSftpError)) +} + +/** + * A chrooted SFTP subsystem answers a path outside its namespace with + * `SSH_FX_NO_SUCH_FILE`, because the path genuinely does not exist in the view it + * serves. `SSH_FX_PERMISSION_DENIED` is not that: it is an ordinary mode/ownership + * refusal on a path the subsystem *can* see — a read-only home, a root-owned parent, + * a quota — and rewriting it into "your bastion chroots SFTP" would send the user to + * fix ProxyJump for a `chmod`. + */ +export function isSandboxedSftpNamespaceError(error: unknown): boolean { + return (error as { code?: unknown } | null)?.code === SSH_FX_NO_SUCH_FILE +} + +/** + * SFTP is not optional for a bundled-ssh2 relay install — `SshConnection.sftp()` is the + * only transfer route on that transport, and the exec-based `tar`/`cat` transfers are + * bound to the system-SSH transport, not selectable per operation. So a sandboxed SFTP + * subsystem is a clean failure with an actionable message, not a degraded mode. + */ +export function describeSandboxedSftpFailure(error: unknown, remotePath: string): Error { + const detail = error instanceof Error ? error.message : String(error) + return Object.assign( + new Error( + `Relay install could not reach ${remotePath} over SFTP (${detail}). ` + + 'The host answered the shell channel but its SFTP subsystem sees a different filesystem — ' + + 'typically a bastion or jump host that chroots SFTP to a transfer directory. ' + + 'Orca cannot install the relay through a sandboxed SFTP subsystem; connect to the target ' + + 'host directly (for example with ProxyJump) or allow SFTP access to the account home.', + { cause: error } + ), + { sandboxedSftpNamespace: true } + ) +} diff --git a/src/main/ssh/sftp-upload.test.ts b/src/main/ssh/sftp-upload.test.ts index db86fa17f51..d5cf25abe70 100644 --- a/src/main/ssh/sftp-upload.test.ts +++ b/src/main/ssh/sftp-upload.test.ts @@ -40,7 +40,9 @@ describe('sftp-upload', () => { }) const writeStream = vi.mocked(sftp.createWriteStream).mock.results[0]?.value as Writable expect(writeStream.listenerCount('close')).toBe(0) - expect(writeStream.listenerCount('error')).toBe(0) + // One durable 'error' listener stays for the stream's whole life: a STATUS reply that + // lands after the transfer settles must not throw into ssh2's parser (#15479). + expect(writeStream.listenerCount('error')).toBe(1) }) it('uses no-clobber writes for nested files during exclusive directory upload', async () => { @@ -59,7 +61,9 @@ describe('sftp-upload', () => { }) const writeStream = vi.mocked(sftp.createWriteStream).mock.results[0]?.value as Writable expect(writeStream.listenerCount('close')).toBe(0) - expect(writeStream.listenerCount('error')).toBe(0) + // One durable 'error' listener stays for the stream's whole life: a STATUS reply that + // lands after the transfer settles must not throw into ssh2's parser (#15479). + expect(writeStream.listenerCount('error')).toBe(1) }) it('uploads files from valid dot-dot-prefixed local directories', async () => { diff --git a/src/main/ssh/sftp-upload.ts b/src/main/ssh/sftp-upload.ts index 6344a7c1457..514df81ed1d 100644 --- a/src/main/ssh/sftp-upload.ts +++ b/src/main/ssh/sftp-upload.ts @@ -4,6 +4,11 @@ import { lstat, open, readdir, realpath } from 'node:fs/promises' import { isAbsolute, join as pathJoin, relative, sep } from 'node:path' import { finished } from 'node:stream/promises' import type { SFTPWrapper } from 'ssh2' +import { + latchLateSftpSessionErrors, + latchLateSftpStreamErrors, + type SftpStreamErrorLatch +} from './sftp-stream-late-error' export function mkdirSftp( sftp: SFTPWrapper, @@ -44,6 +49,7 @@ async function uploadFileAndJoinTeardown( let handleClose: Promise | undefined let readStream: ReadStream | undefined let writeStream: ReturnType | undefined + let writeStreamErrors: SftpStreamErrorLatch | undefined const closeHandle = (): Promise => { handleClose ??= handle.close() return handleClose @@ -67,6 +73,9 @@ async function uploadFileAndJoinTeardown( writeStream = sftp.createWriteStream(remotePath, { flags: options?.exclusive ? 'wx' : 'w' }) + // Why: the OPEN reply can land after this transfer settles; without a listener that + // outlives it, ssh2 throws it synchronously into the socket handler (#15479). + writeStreamErrors = latchLateSftpStreamErrors(writeStream, remotePath) readStream = handle.createReadStream({ autoClose: false }) const abortTransfer = (): void => { const reason = @@ -100,6 +109,7 @@ async function uploadFileAndJoinTeardown( options?.signal?.removeEventListener('abort', abortTransfer) } } finally { + writeStreamErrors?.markTransferSettled() readStream?.destroy() writeStream?.destroy() await closeHandle() @@ -117,8 +127,10 @@ export function uploadBuffer( const writeStream = sftp.createWriteStream(remotePath, { flags: options?.append ? 'a' : options?.exclusive ? 'wx' : 'w' }) + const lateErrors = latchLateSftpStreamErrors(writeStream, remotePath) const cleanupListeners = (): void => { + lateErrors.markTransferSettled() writeStream.off('close', onClose) writeStream.off('error', onError) } @@ -147,8 +159,10 @@ export function writeStringViaSftp( ): Promise { return new Promise((resolve, reject) => { const ws = sftp.createWriteStream(remotePath) + const lateErrors = latchLateSftpStreamErrors(ws, remotePath) let settled = false const cleanup = (): void => { + lateErrors.markTransferSettled() sftp.removeListener('error', onError) ws.removeListener('close', onClose) ws.removeListener('error', onError) @@ -177,6 +191,29 @@ export function writeStringViaSftp( }) } +/** + * Write several files over one SFTP session, ending it when they are all done. + * + * Owns the session's late-error latch, which is why a caller must not hand-roll this + * loop: `writeStringViaSftp` drops its own session listener at each settle, so between + * files and after the last one the emitter would carry none, and a late STATUS reply + * throws synchronously out of ssh2's parser into main (#15479). + */ +export async function writeStringsViaSftp( + conn: { sftp(): Promise }, + files: readonly { path: string; contents: string }[] +): Promise { + const sftp = await conn.sftp() + latchLateSftpSessionErrors(sftp) + try { + for (const file of files) { + await writeStringViaSftp(sftp, file.path, file.contents) + } + } finally { + sftp.end() + } +} + export async function uploadDirectory( sftp: SFTPWrapper, localDir: string, diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index fc65af35c0a..af7346f5d35 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -85,6 +85,14 @@ import { powerShellCommand, powerShellLiteral, powerShellNativeArg } from './ssh import { relaySocketNameForInstanceId } from './ssh-relay-instance-id' import { resolveRelayEndpointBeforeRelaunch } from './ssh-relay-endpoint-takeover' import { sweepSupersededRelayEndpoints } from './ssh-relay-superseded-endpoints' +import { + parseShortRelaySocketDir, + remoteSocketPathFitsLimit, + resolveShortRelaySocketDirCommand, + shortRelaySocketPath, + shortRelayVersionSegment, + SHORT_RELAY_SOCKET_DIR_PREFIX +} from './relay-socket-path-limit' import { isSshSessionLimitError } from './ssh-session-limit-error' import { isWindowsRelayPipePath, @@ -598,6 +606,13 @@ async function deployAndLaunchRelayAttempt( remoteHome, currentRelayDir: remoteRelayDir, sockName: relaySocketNameForInstanceId(relayInstanceId), + // Set only when this launch relocated past sun_path; the sweep must not reap + // the socket the transport it just handed back is talking to. + ...(launched.sockPath.startsWith(SHORT_RELAY_SOCKET_DIR_PREFIX) + ? { + currentShortSocketDir: launched.sockPath.slice(0, launched.sockPath.lastIndexOf('/')) + } + : {}), nodePath: launched.nodePath }) ) @@ -1636,9 +1651,13 @@ async function launchRelay( const escapedNode = shellEscape(nodePath) // Why: remoteRelayDir is shared across Orca targets for one account; hashing the target ID into the socket name stops cross-target attach. const sockName = relaySocketNameForInstanceId(relayInstanceId) - const sockFile = relayEndpointForHost(hostPlatform, remoteDir, sockName) - const endpointDir = relayHookEndpointDirForHost(hostPlatform, remoteDir, sockFile) + const defaultSockFile = relayEndpointForHost(hostPlatform, remoteDir, sockName) + const endpointDir = relayHookEndpointDirForHost(hostPlatform, remoteDir, defaultSockFile) const credentialFile = joinRemotePath(hostPlatform, remoteDir, `${sockName}.credential`) + // Why: a long remote $HOME pushes the default endpoint past sun_path and bind fails with a bare `listen EINVAL` (#10726). + const sockFile = remoteSocketPathFitsLimit(hostPlatform, defaultSockFile) + ? defaultSockFile + : await resolveShortPosixRelaySocketPath(conn, remoteDir, sockName, defaultSockFile, signal) if (isWindowsRemoteHost(hostPlatform)) { const activePipeMarkerPath = windowsActivePipeMarkerPath(hostPlatform, remoteDir, sockName) @@ -1720,7 +1739,10 @@ async function launchRelay( signal }) // Why: --log-file lets the relay rotate relay.log in-process; the shell redirect stays to capture pre-JS boot/crash output. - const launchCmd = `cd ${escapedDir} && chmod 600 ${shellEscape(credentialFile)} && nohup ${escapedNode} relay.js --detached --grace-time ${graceTime} --sock-path ${shellEscape(sockFile)} --credential-file ${shellEscape(credentialFile)} --log-file ${shellEscape(logFile)} > ${shellEscape(logFile)} 2>&1 ${shellEscape(logFile)} 2>&1 {}) launchChannel.on('error', () => {}) @@ -1781,6 +1803,44 @@ async function launchRelay( } } +/** + * Move the endpoint under a `$HOME`-independent base so its length is bounded. + * + * The hashed socket name is preserved in full: only the directory shrinks, so the + * short form stays deterministic per target and cannot collide with another target. + * The version directory's identity comes along as a hashed segment, so a later build + * still binds a path of its own rather than the one its predecessor is holding. + */ +async function resolveShortPosixRelaySocketPath( + conn: SshConnection, + remoteDir: string, + sockName: string, + defaultSockFile: string, + signal?: AbortSignal +): Promise { + const versionSegment = shortRelayVersionSegment(remoteDir.slice(remoteDir.lastIndexOf('/') + 1)) + const output = await execCommand(conn, resolveShortRelaySocketDirCommand(versionSegment), { + signal + }).catch((err: unknown) => { + if (isUnconfirmedSshCommandTermination(err)) { + throw err + } + signal?.throwIfAborted() + return '' + }) + const shortDir = parseShortRelaySocketDir(output, versionSegment) + if (!shortDir) { + throw new Error( + `Relay socket path ${defaultSockFile} exceeds the remote Unix socket limit and no short socket directory could be created on the host.` + ) + } + const shortSockFile = shortRelaySocketPath(shortDir, sockName) + console.warn( + `[ssh-relay] Socket path too long for sun_path; using ${shortSockFile} instead of ${defaultSockFile}` + ) + return shortSockFile +} + function waitForRelayPoll(delayMs: number, signal?: AbortSignal): Promise { return new Promise((resolve, reject) => { const onAbort = (): void => { diff --git a/src/main/ssh/ssh-relay-install-transfers.ts b/src/main/ssh/ssh-relay-install-transfers.ts index c5c99f99b70..6f8ede57984 100644 --- a/src/main/ssh/ssh-relay-install-transfers.ts +++ b/src/main/ssh/ssh-relay-install-transfers.ts @@ -13,6 +13,11 @@ import { type SftpNamespacePathMapping } from './sftp-namespace-resolution' import type { RemoteHostPlatform } from './ssh-remote-platform' +import { + describeSandboxedSftpFailure, + isSandboxedSftpNamespaceError, + latchLateSftpSessionErrors +} from './sftp-stream-late-error' export type RelayTransferOptions = { signal?: AbortSignal @@ -25,6 +30,18 @@ export async function uploadRelayDirectory( shellRemoteDir: string, hostPlatform: RemoteHostPlatform, options?: RelayTransferOptions +): Promise { + await withSandboxedSftpDiagnosis(shellRemoteDir, () => + uploadRelayDirectoryTransfer(conn, localRelayDir, shellRemoteDir, hostPlatform, options) + ) +} + +async function uploadRelayDirectoryTransfer( + conn: SshConnection, + localRelayDir: string, + shellRemoteDir: string, + hostPlatform: RemoteHostPlatform, + options?: RelayTransferOptions ): Promise { if (typeof conn.uploadDirectory === 'function') { await conn.uploadDirectory(localRelayDir, shellRemoteDir, { @@ -52,6 +69,18 @@ export async function writeRelayFile( shellRemotePath: string, contents: string, options?: RelayTransferOptions +): Promise { + await withSandboxedSftpDiagnosis(shellRemotePath, () => + writeRelayFileTransfer(conn, hostPlatform, shellRemotePath, contents, options) + ) +} + +async function writeRelayFileTransfer( + conn: SshConnection, + hostPlatform: RemoteHostPlatform, + shellRemotePath: string, + contents: string, + options?: RelayTransferOptions ): Promise { if (typeof conn.writeFile === 'function') { await conn.writeFile(shellRemotePath, contents, { @@ -77,7 +106,6 @@ async function runSftpFallbackTransfer( transfer: (sftp: SFTPWrapper) => Promise ): Promise { const sftp = await conn.sftp(options?.signal) - const swallowLateSftpError = (): void => {} let sftpEndRequested = false const endSftp = (): void => { if (!sftpEndRequested) { @@ -85,9 +113,7 @@ async function runSftpFallbackTransfer( sftp.end() } } - // A late session 'error' after settle would otherwise be unhandled and crash main. - sftp.on('error', swallowLateSftpError) - sftp.once('close', () => sftp.removeListener('error', swallowLateSftpError)) + latchLateSftpSessionErrors(sftp) try { await raceSftpFileTransferWithAbort( transfer(sftp), @@ -102,3 +128,23 @@ async function runSftpFallbackTransfer( endSftp() } } + +/** + * A jump host whose SFTP subsystem is chrooted answers a home path with + * SSH_FX_NO_SUCH_FILE even though the shell channel resolves it (#15479). SFTP is the + * only install route on the bundled-ssh2 transport, so say what the host did rather + * than surfacing a bare "file does not exist". + */ +async function withSandboxedSftpDiagnosis( + remotePath: string, + transfer: () => Promise +): Promise { + try { + return await transfer() + } catch (error) { + if (isSandboxedSftpNamespaceError(error)) { + throw describeSandboxedSftpFailure(error, remotePath) + } + throw error + } +} diff --git a/src/main/ssh/ssh-relay-reset.ts b/src/main/ssh/ssh-relay-reset.ts index 50101084478..3d585b525f2 100644 --- a/src/main/ssh/ssh-relay-reset.ts +++ b/src/main/ssh/ssh-relay-reset.ts @@ -2,6 +2,7 @@ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' import { execCommand } from './ssh-relay-deploy-helpers' import { relaySocketNameForInstanceId } from './ssh-relay-instance-id' +import { SHORT_RELAY_SOCKET_DIR_PREFIX } from './relay-socket-path-limit' export async function forceStopRelayForTarget( conn: SshConnection, @@ -12,8 +13,10 @@ export async function forceStopRelayForTarget( const script = [ `sock_name=${escapedSockName}`, 'base="${HOME}/.orca-remote"', - 'if [ -d "$base" ]; then', - ' for sock in "$base"/relay-*/"$sock_name" "$base"/"$sock_name"; do', + // Why: a long $HOME moves the socket to the sun_path-safe short base (#10726); reset must reach it there too. + `short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`, + 'if [ -d "$base" ] || [ -d "$short_base" ]; then', + ' for sock in "$base"/relay-*/"$sock_name" "$base"/"$sock_name" "$short_base"/relay-*/"$sock_name"; do', ' [ -S "$sock" ] || continue', ' pid=""', // Why: lsof ORs selectors by default; -a prevents reset from targeting diff --git a/src/main/ssh/ssh-relay-session.ts b/src/main/ssh/ssh-relay-session.ts index 04fe6020749..c303d33b428 100644 --- a/src/main/ssh/ssh-relay-session.ts +++ b/src/main/ssh/ssh-relay-session.ts @@ -5,6 +5,7 @@ import { randomUUID } from 'node:crypto' import type { BrowserWindow } from 'electron' import { deployAndLaunchRelay } from './ssh-relay-deploy' import { execCommand } from './ssh-relay-deploy-helpers' +import { writeStringsViaSftp } from './sftp-upload' import { isRelayVersionMismatchError } from './ssh-relay-version-mismatch-error' import { isRelayEndpointHeldError } from './ssh-relay-endpoint-incumbent' import { forgetRelayNodePtyRepairs, recoverRelayNodePtyForSpawn } from './ssh-relay-node-pty-repair' @@ -1414,20 +1415,7 @@ export class SshRelaySession { await conn.writeFile(file.path, file.contents, { hostPlatform }) } } else { - const sftp = await conn.sftp() - try { - for (const file of plan.files) { - await new Promise((resolve, reject) => { - const ws = sftp.createWriteStream(file.path) - sftp.once('error', reject) - ws.once('close', resolve) - ws.once('error', reject) - ws.end(file.contents) - }) - } - } finally { - sftp.end() - } + await writeStringsViaSftp(conn, plan.files) } for (const command of plan.postWriteCommands) { await execCommand(conn, command, { wrapCommand: !isWindowsRemoteHost(hostPlatform) }) diff --git a/src/main/ssh/ssh-relay-superseded-endpoints.ts b/src/main/ssh/ssh-relay-superseded-endpoints.ts index 1a9554f7b82..4b1ad5637ef 100644 --- a/src/main/ssh/ssh-relay-superseded-endpoints.ts +++ b/src/main/ssh/ssh-relay-superseded-endpoints.ts @@ -19,6 +19,7 @@ import type { SshConnection } from './ssh-connection' import { shellEscape } from './ssh-connection-utils' import { RELAY_REMOTE_DIR } from './relay-protocol' +import { SHORT_RELAY_SOCKET_DIR_PREFIX } from './relay-socket-path-limit' import { execCommand } from './ssh-relay-deploy-helpers' import { describeRelayEndpointIncumbent, @@ -53,6 +54,8 @@ export type SupersededRelaySweepOptions = { currentRelayDir: string /** Stable per-target socket filename, from `relaySocketNameForInstanceId`. */ sockName: string + /** Set only when this launch relocated its socket; that directory is never swept. */ + currentShortSocketDir?: string nodePath: string signal?: AbortSignal } @@ -63,15 +66,23 @@ export function supersededRelayEndpointListCommand(options: { remoteHome: string currentRelayDir: string sockName: string + currentShortSocketDir?: string }): string { return [ `base=${shellEscape(`${options.remoteHome}/${RELAY_REMOTE_DIR}`)}`, `sock_name=${shellEscape(options.sockName)}`, `current=${shellEscape(options.currentRelayDir)}`, - 'for sock in "$base"/relay-*/"$sock_name"; do', + // Why the second base: a host whose `$HOME` pushes the endpoint past `sun_path` binds + // under `/tmp/.orca-relay-/relay-/` instead (relay-socket-path-limit.ts). + // Those orphans are the same population this sweep exists to make visible, and the + // `$HOME` glob cannot see them. The uid is resolved on the host; the client never knows it. + `short_current=${shellEscape(options.currentShortSocketDir ?? '')}`, + `short_base="${SHORT_RELAY_SOCKET_DIR_PREFIX}$(id -u 2>/dev/null)"`, + 'for sock in "$base"/relay-*/"$sock_name" "$short_base"/relay-*/"$sock_name"; do', ' [ -S "$sock" ] || continue', ' dir=${sock%/*}', ' [ "$dir" = "$current" ] && continue', + ' [ -n "$short_current" ] && [ "$dir" = "$short_current" ] && continue', ' printf \'%s\\n\' "$sock"', 'done' ].join('\n') @@ -79,7 +90,14 @@ export function supersededRelayEndpointListCommand(options: { /** Remove a socket inode proven to have no holder, so version-dir GC can reclaim the tree. */ export function removeStaleRelayEndpointCommand(sockPath: string): string { - return `rm -f ${shellEscape(sockPath)}` + const remove = `rm -f ${shellEscape(sockPath)}` + if (!sockPath.startsWith(SHORT_RELAY_SOCKET_DIR_PREFIX)) { + return remove + } + // `gcOldRelayVersions` only walks `$HOME/.orca-remote`, so nothing else would ever + // reclaim a relocated version segment. `rmdir` fails while another target of the same + // build still has a socket there, which is exactly the condition for keeping it. + return `${remove}; rmdir ${shellEscape(sockPath.slice(0, sockPath.lastIndexOf('/')))} 2>/dev/null || true` } export function classifySupersededRelay( From 710405698472c3d9618c3114c7f15088be4d72cc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:21 -0700 Subject: [PATCH 113/398] fix(watcher): route relay watch-root capacity refusals off the fast ladder (#17950) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): stop two unrecoverable relay refusal loops A relay refusal that is a pure function of state the client cannot change was being retried forever, on two different paths. - pty.openClient: a superseded owner proof is refuted evidence, not a transient fault. The client kept re-presenting the identical proof, so every reconnect reproduced the same refusal until the relay was redeployed (#12895, #12931). It is now dropped exactly as a stale lease already is, and the claim re-asked without it. - fs.watch: the relay's watch-root capacity refusal was classified 'unavailable' and retried at 1 Hz per root for 60s, re-armed indefinitely. A folder workspace with more repos than the cap turns that into a permanent install storm scaled by the excess root count (#11196). It is now its own 'capacity' result that goes straight to the existing dormant backoff, mirroring what the local watcher path already does. * fix(watcher): route relay watch-root capacity refusals off the fast ladder A full watch-root cap is a decision, not a fault, so a 1 Hz reinstall per refused root only bills the relay the load that keeps the cap busy (#11196). Capacity refusals now go straight to the dormant backoff. The relay side no longer refuses on a slot it is about to hand back: an over-cap caused by roots still unsubscribing waits once on the teardowns settling — the release event, mirroring WatcherSupervisorCapacityWait — before it answers. A parked waiter is excluded from the accounting so it cannot take a slot from the root already reclaiming one. Drops the SSH owner-recovery half of this branch. Its premise — that a -32043 SUPERSEDED refusal is permanent — is false: the refusal fires only while the incumbent is 'active', and assertPtyConsumerOwnerRecovery explicitly admits the identical lower-generation proof once the incumbent flips to 'disconnected' (relay-pty-consumer-owner-displacement.test.ts proves it). The remedy could not work either: the proofless re-ask routes into refuseHeldPtyConsumerOwner, which is declared `: never` and, with sameClient true by construction, always throws. It would have traded one refusal loop for another, minus the checkpoints and minus the proof that resumes the claim once the relay reaps the incumbent. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ipc/filesystem-watcher-handlers.ts | 7 + .../ipc/filesystem-watcher-lifecycle-state.ts | 5 +- ...filesystem-watcher-remote-capacity.test.ts | 91 +++++++++++++ .../filesystem-watcher-remote-controller.ts | 3 +- .../ipc/filesystem-watcher-remote-dormant.ts | 4 +- .../ipc/filesystem-watcher-remote-install.ts | 5 + ...ilesystem-watcher-remote-provider-rearm.ts | 5 + .../ipc/filesystem-watcher-remote-removal.ts | 5 +- .../ipc/filesystem-watcher-remote-retry.ts | 6 + src/relay/fs-handler.test.ts | 114 +---------------- src/relay/relay-filesystem-watch-registry.ts | 18 ++- src/relay/relay-fs-test-dispatcher.ts | 120 ++++++++++++++++++ src/relay/relay-watch-root-capacity-gate.ts | 86 +++++++++++++ src/relay/relay-watch-root-capacity.test.ts | 107 ++++++++++++++++ src/relay/relay-watcher-root-capacity.ts | 20 ++- src/relay/relay-watcher-teardown-tracker.ts | 11 ++ src/shared/watch-root-capacity-refusal.ts | 9 ++ 17 files changed, 487 insertions(+), 129 deletions(-) create mode 100644 src/main/ipc/filesystem-watcher-remote-capacity.test.ts create mode 100644 src/relay/relay-fs-test-dispatcher.ts create mode 100644 src/relay/relay-watch-root-capacity-gate.ts create mode 100644 src/relay/relay-watch-root-capacity.test.ts create mode 100644 src/shared/watch-root-capacity-refusal.ts diff --git a/src/main/ipc/filesystem-watcher-handlers.ts b/src/main/ipc/filesystem-watcher-handlers.ts index 7a2b6230abd..bac74f63589 100644 --- a/src/main/ipc/filesystem-watcher-handlers.ts +++ b/src/main/ipc/filesystem-watcher-handlers.ts @@ -10,6 +10,7 @@ import { import { installRemoteWatcher, reinstallRemoteWatchersForConnection, + scheduleDormantRemoteWatcherRearm, scheduleRemoteWatcherRetry } from './filesystem-watcher-remote-controller' import { rememberDesiredRemoteWatcher } from './filesystem-watcher-remote-desired' @@ -41,6 +42,12 @@ export function registerFilesystemWatcherHandlers(): void { args.connectionId, args.worktreePath ) + if (result === 'capacity') { + // Why straight to the dormant backoff: the cap is full until some other root is released, + // which a 1 Hz reinstall cannot bring about — it only adds relay load per refused root. + scheduleDormantRemoteWatcherRearm(args.connectionId, args.worktreePath) + return + } if (result === 'unavailable') { if (!watcherLifecycleState.loggedUnavailableRemoteWatchers.has(key)) { watcherLifecycleState.loggedUnavailableRemoteWatchers.add(key) diff --git a/src/main/ipc/filesystem-watcher-lifecycle-state.ts b/src/main/ipc/filesystem-watcher-lifecycle-state.ts index 548d958f90c..7d6a120c0a7 100644 --- a/src/main/ipc/filesystem-watcher-lifecycle-state.ts +++ b/src/main/ipc/filesystem-watcher-lifecycle-state.ts @@ -36,7 +36,10 @@ export type RemoteWatcherState = { batch: RemoteWatcherEventBatch } -export type RemoteWatcherInstallResult = 'installed' | 'unavailable' | 'cancelled' +// Why 'capacity' is not 'unavailable': the relay refused because its watch-root cap is full, which is +// a decision, not a fault. The 1 Hz unavailable retry cannot change that answer, and a folder +// workspace whose repo count exceeds the cap turns it into a permanent per-root storm (#11196). +export type RemoteWatcherInstallResult = 'installed' | 'unavailable' | 'capacity' | 'cancelled' export type RemoteWatcherResyncState = { lastSentAt: number diff --git a/src/main/ipc/filesystem-watcher-remote-capacity.test.ts b/src/main/ipc/filesystem-watcher-remote-capacity.test.ts new file mode 100644 index 00000000000..97a8f82bf33 --- /dev/null +++ b/src/main/ipc/filesystem-watcher-remote-capacity.test.ts @@ -0,0 +1,91 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { handleMock, getSshFilesystemProviderMock } = vi.hoisted(() => ({ + handleMock: vi.fn(), + getSshFilesystemProviderMock: vi.fn() +})) + +vi.mock('electron', () => ({ + ipcMain: { handle: handleMock } +})) + +vi.mock('fs/promises', () => ({ stat: vi.fn() })) +vi.mock('@parcel/watcher', () => ({ subscribe: vi.fn() })) +vi.mock('./filesystem-watcher-wsl', () => ({ createWslWatcher: vi.fn() })) +vi.mock('../providers/ssh-filesystem-dispatch', () => ({ + getSshFilesystemProvider: getSshFilesystemProviderMock, + onSshFilesystemProviderRegistered: () => () => {} +})) + +import { WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE } from '../../shared/watch-root-capacity-refusal' +import { closeAllWatchers, registerFilesystemWatcherHandlers } from './filesystem-watcher' +import { watcherLifecycleState } from './filesystem-watcher-lifecycle-state' +import { getRemoteWatcherKey } from './filesystem-watcher-paths' + +type HandlerMap = Record unknown> + +describe('remote filesystem watcher capacity refusals', () => { + const handlers: HandlerMap = {} + + beforeEach(async () => { + handleMock.mockReset() + getSshFilesystemProviderMock.mockReset() + for (const key of Object.keys(handlers)) { + delete handlers[key] + } + handleMock.mockImplementation((channel, handler) => { + handlers[channel] = handler + }) + registerFilesystemWatcherHandlers() + await closeAllWatchers() + }) + + afterEach(async () => { + for (const dormant of watcherLifecycleState.dormantRemoteWatchers.values()) { + clearTimeout(dormant.timer) + } + watcherLifecycleState.dormantRemoteWatchers.clear() + await closeAllWatchers() + vi.useRealTimers() + }) + + // A folder workspace with more repos than the relay's watch-root cap leaves every excess root + // permanently refused; the 1 Hz unavailable ladder then bills the relay one install per root per + // second, which is the load that pinned it (#11196). + it('does not retry a relay watch-root capacity refusal on the fast ladder', async () => { + vi.useFakeTimers() + const watchMock = vi.fn(async () => { + throw new Error(WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE) + }) + getSshFilesystemProviderMock.mockReturnValue({ watch: watchMock }) + const sender = { isDestroyed: () => false, send: vi.fn(), once: vi.fn(), id: 1 } + const args = { worktreePath: '/home/me/repos/one', connectionId: 'conn-capacity' } + + await handlers['fs:watchWorktree']({ sender }, args) + const key = getRemoteWatcherKey(args.connectionId, args.worktreePath) + + expect(watchMock).toHaveBeenCalledTimes(1) + expect(watcherLifecycleState.pendingRemoteWatcherRetries.has(key)).toBe(false) + expect(watcherLifecycleState.dormantRemoteWatchers.has(key)).toBe(true) + + await vi.advanceTimersByTimeAsync(10_000) + expect(watchMock).toHaveBeenCalledTimes(1) + }) + + it('still retries an ordinary unavailable install on the fast ladder', async () => { + vi.useFakeTimers() + const watchMock = vi.fn(async () => { + throw new Error('Relay channel lost') + }) + getSshFilesystemProviderMock.mockReturnValue({ watch: watchMock }) + const sender = { isDestroyed: () => false, send: vi.fn(), once: vi.fn(), id: 2 } + const args = { worktreePath: '/home/me/repos/two', connectionId: 'conn-unavailable' } + + await handlers['fs:watchWorktree']({ sender }, args) + const key = getRemoteWatcherKey(args.connectionId, args.worktreePath) + + expect(watcherLifecycleState.pendingRemoteWatcherRetries.has(key)).toBe(true) + await vi.advanceTimersByTimeAsync(2_500) + expect(watchMock.mock.calls.length).toBeGreaterThan(1) + }) +}) diff --git a/src/main/ipc/filesystem-watcher-remote-controller.ts b/src/main/ipc/filesystem-watcher-remote-controller.ts index 75bc4aa75d5..1b6953cd292 100644 --- a/src/main/ipc/filesystem-watcher-remote-controller.ts +++ b/src/main/ipc/filesystem-watcher-remote-controller.ts @@ -63,7 +63,8 @@ export function reinstallRemoteWatchersForConnection(connectionId: string): void reinstallRemoteWatchersForConnectionCore(connectionId, { install: installRemoteWatcher, requestResync: requestRemoteWatcherResync, - scheduleRetry: scheduleRemoteWatcherRetry + scheduleRetry: scheduleRemoteWatcherRetry, + scheduleDormant: scheduleDormantRemoteWatcherRearm }) } diff --git a/src/main/ipc/filesystem-watcher-remote-dormant.ts b/src/main/ipc/filesystem-watcher-remote-dormant.ts index 741753b42e5..7c168114f51 100644 --- a/src/main/ipc/filesystem-watcher-remote-dormant.ts +++ b/src/main/ipc/filesystem-watcher-remote-dormant.ts @@ -97,8 +97,8 @@ async function rearmDormantRemoteWatcher( worktreePath, listeners.filter((_, index) => results[index] === 'installed') ) - // Why: 'cancelled' means shutdown or the last listener left, so only 'unavailable' stays dormant. - if (results.some((result) => result === 'unavailable')) { + // Why: 'cancelled' means shutdown or the last listener left, so only a refusal stays dormant. + if (results.some((result) => result === 'unavailable' || result === 'capacity')) { scheduleDormantRemoteWatcherRearmCore( connectionId, worktreePath, diff --git a/src/main/ipc/filesystem-watcher-remote-install.ts b/src/main/ipc/filesystem-watcher-remote-install.ts index c069c8dc16b..3ab5babbd91 100644 --- a/src/main/ipc/filesystem-watcher-remote-install.ts +++ b/src/main/ipc/filesystem-watcher-remote-install.ts @@ -1,5 +1,6 @@ import type { WebContents } from 'electron' import type { FsChangedPayload } from '../../shared/filesystem-entry-types' +import { isWatchRootCapacityRefusal } from '../../shared/watch-root-capacity-refusal' import { WATCH_BATCH_MAX_WAIT_MS, WATCH_BATCH_TRAILING_MS @@ -202,6 +203,10 @@ async function doInstallRemoteWatcher( if (cancelToken.cancelled || cancelToken.abortController.signal.aborted) { return 'cancelled' } + if (isWatchRootCapacityRefusal(err)) { + console.warn(`[filesystem-watcher] relay watch-root capacity reached for ${key}`) + return 'capacity' + } console.warn(`[filesystem-watcher] SSH watcher unavailable for ${key}:`, err) return 'unavailable' } finally { diff --git a/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts b/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts index 6b0ea28ef9b..3cc657c4086 100644 --- a/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts +++ b/src/main/ipc/filesystem-watcher-remote-provider-rearm.ts @@ -29,6 +29,7 @@ export function reinstallRemoteWatchersForConnectionCore( install: InstallRemoteWatcher requestResync: RequestRemoteWatcherResync scheduleRetry: ScheduleRemoteWatcherRetry + scheduleDormant: (connectionId: string, worktreePath: string) => void } ): void { if (watcherLifecycleState.remoteWatchersClosed) { @@ -90,6 +91,10 @@ export function reinstallRemoteWatchersForConnectionCore( desired.worktreePath, listeners.filter((_, index) => results[index] === 'installed') ) + if (results.some((result) => result === 'capacity')) { + dependencies.scheduleDormant(desired.connectionId, desired.worktreePath) + return + } if (results.some((result) => result === 'unavailable')) { for (const listener of listeners) { dependencies.scheduleRetry( diff --git a/src/main/ipc/filesystem-watcher-remote-removal.ts b/src/main/ipc/filesystem-watcher-remote-removal.ts index 61424dc1db9..0e9c741851a 100644 --- a/src/main/ipc/filesystem-watcher-remote-removal.ts +++ b/src/main/ipc/filesystem-watcher-remote-removal.ts @@ -9,6 +9,7 @@ import { } from './filesystem-watcher-listener-lifecycle' import { installRemoteWatcher, + scheduleDormantRemoteWatcherRearm, scheduleRemoteWatcherRetry } from './filesystem-watcher-remote-controller' @@ -75,7 +76,9 @@ export async function restoreRemoteWatcherAfterFailedRemoval( continue } const result = await installRemoteWatcher(sender, connectionId, worktreePath) - if (result === 'unavailable') { + if (result === 'capacity') { + scheduleDormantRemoteWatcherRearm(connectionId, worktreePath) + } else if (result === 'unavailable') { scheduleRemoteWatcherRetry(sender, connectionId, worktreePath) } sender.send('fs:changed', { diff --git a/src/main/ipc/filesystem-watcher-remote-retry.ts b/src/main/ipc/filesystem-watcher-remote-retry.ts index 0151ecf4419..d73f9ed4db4 100644 --- a/src/main/ipc/filesystem-watcher-remote-retry.ts +++ b/src/main/ipc/filesystem-watcher-remote-retry.ts @@ -100,6 +100,12 @@ export function scheduleRemoteWatcherRetryCore( listeners.filter((_, index) => results[index] === 'installed') ) } + // Why capacity leaves the fast window: the relay is refusing on a full watch-root cap, and a + // 1 Hz reinstall per refused root is exactly the load that keeps the cap busy (#11196). + if (results.some((result) => result === 'capacity')) { + dependencies.scheduleDormant(connectionId, worktreePath) + return + } // Why: don't re-arm on 'cancelled' (renderer stopped watching) — it would fire a stale overflow when the 60s window expires. if (results.some((result) => result === 'unavailable')) { for (const listener of listeners) { diff --git a/src/relay/fs-handler.test.ts b/src/relay/fs-handler.test.ts index 9dc4812ef98..fcea45c898c 100644 --- a/src/relay/fs-handler.test.ts +++ b/src/relay/fs-handler.test.ts @@ -8,6 +8,7 @@ import * as path from 'node:path' import { mkdtempSync, writeFileSync, mkdirSync, symlinkSync } from 'node:fs' import { tmpdir } from 'node:os' import { subscribeWithInProcessWatcher } from '../main/ipc/parcel-watcher-in-process-fallback' +import { createMockDispatcher } from './relay-fs-test-dispatcher' const { mockSubscribe } = vi.hoisted(() => ({ mockSubscribe: vi.fn() @@ -17,91 +18,6 @@ vi.mock('@parcel/watcher', () => ({ subscribe: mockSubscribe })) -function createMockDispatcher() { - const requestHandlers = new Map< - string, - ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => Promise - >() - const notificationHandlers = new Map< - string, - ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => void - >() - const detachListeners = new Set<(clientId: number) => void>() - const notifications: { method: string; params?: Record }[] = [] - - return { - onRequest: vi.fn( - ( - method: string, - handler: ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => Promise - ) => { - requestHandlers.set(method, handler) - } - ), - onNotification: vi.fn( - ( - method: string, - handler: ( - params: Record, - context?: { clientId: number; isStale: () => boolean } - ) => void - ) => { - notificationHandlers.set(method, handler) - } - ), - notify: vi.fn((method: string, params?: Record) => { - notifications.push({ method, params }) - }), - notifyClient: vi.fn(), - onClientDetached: vi.fn((listener: (clientId: number) => void) => { - detachListeners.add(listener) - return () => detachListeners.delete(listener) - }), - _requestHandlers: requestHandlers, - _notificationHandlers: notificationHandlers, - _notifications: notifications, - async callRequest( - method: string, - params: Record = {}, - context?: { clientId?: number; isStale: () => boolean } - ) { - const handler = requestHandlers.get(method) - if (!handler) { - throw new Error(`No handler for ${method}`) - } - return handler(params, { - clientId: context?.clientId ?? 1, - isStale: context?.isStale ?? (() => false) - }) - }, - callNotification( - method: string, - params: Record = {}, - context?: { clientId: number; isStale: () => boolean } - ) { - const handler = notificationHandlers.get(method) - if (!handler) { - throw new Error(`No handler for ${method}`) - } - handler(params, context ?? { clientId: 1, isStale: () => false }) - }, - detachClient(clientId: number) { - for (const listener of detachListeners) { - listener(clientId) - } - } - } -} - function statIdentity(stats: { dev?: number ino?: number @@ -837,34 +753,6 @@ describe('FsHandler', () => { await joined }) - it('blocks replacement watches behind physical unsubscribe and counts the pending slot', async () => { - let resolveUnsubscribe: () => void = () => {} - const unsubscribe = vi.fn( - () => - new Promise((resolve) => { - resolveUnsubscribe = resolve - }) - ) - mockSubscribe.mockResolvedValue({ unsubscribe }) - await dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) - dispatcher.callNotification('fs.unwatch', { rootPath: tmpDir }) - - const replacement = dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) - for (let index = 0; index < 19; index += 1) { - await dispatcher.callRequest('fs.watch', { - rootPath: path.join(tmpDir, `pending-cap-${index}`) - }) - } - await expect( - dispatcher.callRequest('fs.watch', { rootPath: path.join(tmpDir, 'over-pending-cap') }) - ).rejects.toThrow('Maximum number of file watchers reached') - expect(mockSubscribe).toHaveBeenCalledTimes(20) - - resolveUnsubscribe() - await replacement - expect(mockSubscribe).toHaveBeenCalledTimes(21) - }) - it('retains a failed native unsubscribe slot until acknowledged retry succeeds', async () => { const unsubscribe = vi .fn() diff --git a/src/relay/relay-filesystem-watch-registry.ts b/src/relay/relay-filesystem-watch-registry.ts index de583b2f50a..14f5d5b2ee8 100644 --- a/src/relay/relay-filesystem-watch-registry.ts +++ b/src/relay/relay-filesystem-watch-registry.ts @@ -16,7 +16,7 @@ import { type RelayWatcherTeardownState } from './relay-watcher-teardown-tracker' import { emitRelayWatcherTerminalFailure } from './relay-watcher-terminal-notifier' -import { assertRelayWatcherRootCapacity } from './relay-watcher-root-capacity' +import { RelayWatchRootCapacityGate } from './relay-watch-root-capacity-gate' import { normalizeRuntimePathForComparison } from '../shared/cross-platform-path' import { trackRelayWatcherSetup, @@ -33,6 +33,11 @@ const RELAY_WATCH_OPTIONS = buildParcelWatcherIgnoreOptions(WATCHER_IGNORE_DIRS) export class RelayFilesystemWatchRegistry { private readonly watches = new Map() private readonly pendingSetups = new Map() + private readonly capacityGate = new RelayWatchRootCapacityGate( + this.watches, + this.pendingSetups, + () => this.teardownTracker + ) private readonly teardownTracker: RelayWatcherTeardownTracker private readonly removalFence: RelayWatcherRemovalFence @@ -95,6 +100,10 @@ export class RelayFilesystemWatchRegistry { if (rootTeardown) { await rootTeardown } + const capacityRelease = this.capacityGate.release(rootKey, context?.signal) + if (capacityRelease) { + await capacityRelease + } const clientId = context?.clientId ?? 0 const isStale = context?.isStale ?? (() => false) const existing = this.watches.get(rootKey) @@ -107,12 +116,7 @@ export class RelayFilesystemWatchRegistry { return } - assertRelayWatcherRootCapacity( - this.watches.keys(), - this.pendingSetups.keys(), - this.teardownTracker.rootPaths(), - rootKey - ) + this.capacityGate.assert(rootKey) const state = createRelayWatcherState(rootKey, rootPath, clientId, isStale, watchId) this.watches.set(rootKey, state) diff --git a/src/relay/relay-fs-test-dispatcher.ts b/src/relay/relay-fs-test-dispatcher.ts new file mode 100644 index 00000000000..17d0e20bd19 --- /dev/null +++ b/src/relay/relay-fs-test-dispatcher.ts @@ -0,0 +1,120 @@ +import { vi, type Mock } from 'vitest' + +type MockRequestHandler = ( + params: Record, + context?: { clientId: number; isStale: () => boolean } +) => Promise +type MockNotificationHandler = ( + params: Record, + context?: { clientId: number; isStale: () => boolean } +) => void +type MockCallContext = { clientId?: number; isStale: () => boolean } + +// Explicit rather than inferred: vi.fn()'s inferred type is not nameable across project boundaries. +export type MockRelayFsDispatcher = { + onRequest: Mock + onNotification: Mock + notify: Mock + notifyClient: Mock + onClientDetached: Mock + _requestHandlers: Map + _notificationHandlers: Map + _notifications: { method: string; params?: Record }[] + callRequest: ( + method: string, + params?: Record, + context?: MockCallContext + ) => Promise + callNotification: ( + method: string, + params?: Record, + context?: { clientId: number; isStale: () => boolean } + ) => void + detachClient: (clientId: number) => void +} + +/** Records handlers and notifications so a test can drive FsHandler without a real transport. */ +export function createMockDispatcher(): MockRelayFsDispatcher { + const requestHandlers = new Map< + string, + ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => Promise + >() + const notificationHandlers = new Map< + string, + ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => void + >() + const detachListeners = new Set<(clientId: number) => void>() + const notifications: { method: string; params?: Record }[] = [] + + return { + onRequest: vi.fn( + ( + method: string, + handler: ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => Promise + ) => { + requestHandlers.set(method, handler) + } + ), + onNotification: vi.fn( + ( + method: string, + handler: ( + params: Record, + context?: { clientId: number; isStale: () => boolean } + ) => void + ) => { + notificationHandlers.set(method, handler) + } + ), + notify: vi.fn((method: string, params?: Record) => { + notifications.push({ method, params }) + }), + notifyClient: vi.fn(), + onClientDetached: vi.fn((listener: (clientId: number) => void) => { + detachListeners.add(listener) + return () => detachListeners.delete(listener) + }), + _requestHandlers: requestHandlers, + _notificationHandlers: notificationHandlers, + _notifications: notifications, + async callRequest( + method: string, + params: Record = {}, + context?: { clientId?: number; isStale: () => boolean } + ) { + const handler = requestHandlers.get(method) + if (!handler) { + throw new Error(`No handler for ${method}`) + } + return handler(params, { + clientId: context?.clientId ?? 1, + isStale: context?.isStale ?? (() => false) + }) + }, + callNotification( + method: string, + params: Record = {}, + context?: { clientId: number; isStale: () => boolean } + ) { + const handler = notificationHandlers.get(method) + if (!handler) { + throw new Error(`No handler for ${method}`) + } + handler(params, context ?? { clientId: 1, isStale: () => false }) + }, + detachClient(clientId: number) { + for (const listener of detachListeners) { + listener(clientId) + } + } + } +} diff --git a/src/relay/relay-watch-root-capacity-gate.ts b/src/relay/relay-watch-root-capacity-gate.ts new file mode 100644 index 00000000000..4e430332be2 --- /dev/null +++ b/src/relay/relay-watch-root-capacity-gate.ts @@ -0,0 +1,86 @@ +import { + assertRelayWatcherRootCapacity, + exceedsRelayWatcherRootCapacity +} from './relay-watcher-root-capacity' + +type RelayWatchRootTeardowns = { + rootPaths: () => string[] + /** Resolves when every teardown in flight has settled, or undefined when none is. */ + settlePending: () => Promise | undefined +} + +/** + * Decides whether a prospective watch root fits, and waits out an over-cap that only unsubscribing + * roots are causing. + * + * Why waiting beats refusing: a reconnect tears the old roots down as it installs the new ones, so + * the cap is briefly full of slots already promised back. The client answers a capacity refusal + * with a 60s-to-30min dormancy that no release event can shorten, so refusing on a transient + * overlap costs half an hour of blindness. Mirrors WatcherSupervisorCapacityWait. + */ +export class RelayWatchRootCapacityGate { + // Why tracked: a root parked on the wait has been granted nothing, so counting its setup entry + // would let it hold a slot away from the root already reclaiming one. + private readonly waiting = new Set() + + constructor( + private readonly activeRoots: ReadonlyMap, + private readonly setupRoots: ReadonlyMap, + // Thunk: the registry builds its teardown tracker after this field initializes. + private readonly teardowns: () => RelayWatchRootTeardowns + ) {} + + assert(rootKey: string): void { + assertRelayWatcherRootCapacity( + this.activeRoots.keys(), + this.claimedSetupRoots(rootKey), + this.teardowns().rootPaths(), + rootKey + ) + } + + /** + * The wait to hold before {@link assert}, or undefined when there is nothing to wait for. + * + * Undefined rather than a resolved promise so an install that already fits stays synchronous — + * a suspension here would let a concurrent watch of the same root join the setup, not the watch. + */ + release(rootKey: string, signal?: AbortSignal): Promise | undefined { + if ( + !exceedsRelayWatcherRootCapacity( + this.activeRoots.keys(), + this.claimedSetupRoots(rootKey), + this.teardowns().rootPaths(), + rootKey + ) + ) { + return undefined + } + const released = this.teardowns().settlePending() + if (!released) { + return undefined + } + this.waiting.add(rootKey) + // Once, and never past the caller: a genuinely full cap must still reach the refusal that sends + // the client dormant, and an unsubscribe that never settles must not park the request with it. + return (signal ? Promise.race([released, abortSignalSettled(signal)]) : released).finally( + () => { + this.waiting.delete(rootKey) + } + ) + } + + /** Setup roots that currently hold a slot — a parked capacity waiter holds none. */ + private claimedSetupRoots(rootKey: string): string[] { + return [...this.setupRoots.keys()].filter((key) => key === rootKey || !this.waiting.has(key)) + } +} + +/** Resolves (never rejects) when the request is abandoned, so a race can drop out of a wait. */ +function abortSignalSettled(signal: AbortSignal): Promise { + return signal.aborted + ? Promise.resolve() + : new Promise((resolve) => + signal.addEventListener('abort', () => resolve(), { once: true }) + ) +} diff --git a/src/relay/relay-watch-root-capacity.test.ts b/src/relay/relay-watch-root-capacity.test.ts new file mode 100644 index 00000000000..de5a5fc2b4b --- /dev/null +++ b/src/relay/relay-watch-root-capacity.test.ts @@ -0,0 +1,107 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import * as fs from 'node:fs/promises' +import * as path from 'node:path' +import { mkdtempSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { RelayContext } from './context' +import type { RelayDispatcher } from './dispatcher' +import { FsHandler } from './fs-handler' +import { subscribeWithInProcessWatcher } from '../main/ipc/parcel-watcher-in-process-fallback' +import { createMockDispatcher } from './relay-fs-test-dispatcher' + +const { mockSubscribe } = vi.hoisted(() => ({ + mockSubscribe: vi.fn() +})) + +vi.mock('@parcel/watcher', () => ({ + subscribe: mockSubscribe +})) + +describe('relay watch-root capacity', () => { + let dispatcher: ReturnType + let handler: FsHandler + let tmpDir: string + + beforeEach(() => { + mockSubscribe.mockReset() + mockSubscribe.mockResolvedValue({ unsubscribe: vi.fn() }) + tmpDir = mkdtempSync(path.join(tmpdir(), 'relay-fs-cap-')) + dispatcher = createMockDispatcher() + handler = new FsHandler(dispatcher as unknown as RelayDispatcher, new RelayContext(), { + dispose: vi.fn(), + forgetRoot: vi.fn(), + subscribe: subscribeWithInProcessWatcher + }) + }) + + afterEach(async () => { + handler.dispose() + await fs.rm(tmpDir, { recursive: true, force: true }) + }) + + it('blocks replacement watches behind physical unsubscribe and counts the pending slot', async () => { + let resolveUnsubscribe: () => void = () => {} + const unsubscribe = vi.fn( + () => + new Promise((resolve) => { + resolveUnsubscribe = resolve + }) + ) + mockSubscribe.mockResolvedValue({ unsubscribe }) + await dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) + dispatcher.callNotification('fs.unwatch', { rootPath: tmpDir }) + + const replacement = dispatcher.callRequest('fs.watch', { rootPath: tmpDir }) + for (let index = 0; index < 19; index += 1) { + await dispatcher.callRequest('fs.watch', { + rootPath: path.join(tmpDir, `pending-cap-${index}`) + }) + } + // The replacement claims the slot the teardown releases, so this cap is genuinely full: the + // request waits for the release event and is still refused once it has happened. + const overCap = dispatcher + .callRequest('fs.watch', { rootPath: path.join(tmpDir, 'over-pending-cap') }) + .then( + () => null, + (error: Error) => error + ) + expect(mockSubscribe).toHaveBeenCalledTimes(20) + + resolveUnsubscribe() + await replacement + expect(await overCap).toMatchObject({ message: 'Maximum number of file watchers reached' }) + expect(mockSubscribe).toHaveBeenCalledTimes(21) + }) + + it('waits out a teardown that frees a slot instead of refusing on it', async () => { + let resolveUnsubscribe: () => void = () => {} + mockSubscribe.mockResolvedValue({ + unsubscribe: vi.fn( + () => + new Promise((resolve) => { + resolveUnsubscribe = resolve + }) + ) + }) + for (let index = 0; index < 20; index += 1) { + await dispatcher.callRequest('fs.watch', { rootPath: path.join(tmpDir, `full-${index}`) }) + } + dispatcher.callNotification('fs.unwatch', { rootPath: path.join(tmpDir, 'full-0') }) + + // Why not a refusal: the slot is already promised back, and the client answers a capacity + // refusal with a 60s-to-30min dormancy that no release event can shorten. + let settled = false + const fresh = dispatcher + .callRequest('fs.watch', { rootPath: path.join(tmpDir, 'fresh') }) + .then(() => { + settled = true + }) + await Promise.resolve() + expect(settled).toBe(false) + expect(mockSubscribe).toHaveBeenCalledTimes(20) + + resolveUnsubscribe() + await fresh + expect(mockSubscribe).toHaveBeenCalledTimes(21) + }) +}) diff --git a/src/relay/relay-watcher-root-capacity.ts b/src/relay/relay-watcher-root-capacity.ts index 025462e7bd3..35178166f7c 100644 --- a/src/relay/relay-watcher-root-capacity.ts +++ b/src/relay/relay-watcher-root-capacity.ts @@ -1,14 +1,26 @@ +import { WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE } from '../shared/watch-root-capacity-refusal' + const MAX_RELAY_WATCH_ROOTS = 20 +// Why teardown roots count: a root still unsubscribing owns its native handles until it settles. +export function exceedsRelayWatcherRootCapacity( + activeRoots: Iterable, + pendingRoots: Iterable, + teardownRoots: Iterable, + prospectiveRoot: string +): boolean { + const physicalRoots = new Set([...activeRoots, ...pendingRoots, ...teardownRoots]) + physicalRoots.add(prospectiveRoot) + return physicalRoots.size > MAX_RELAY_WATCH_ROOTS +} + export function assertRelayWatcherRootCapacity( activeRoots: Iterable, pendingRoots: Iterable, teardownRoots: Iterable, prospectiveRoot: string ): void { - const physicalRoots = new Set([...activeRoots, ...pendingRoots, ...teardownRoots]) - physicalRoots.add(prospectiveRoot) - if (physicalRoots.size > MAX_RELAY_WATCH_ROOTS) { - throw new Error('Maximum number of file watchers reached') + if (exceedsRelayWatcherRootCapacity(activeRoots, pendingRoots, teardownRoots, prospectiveRoot)) { + throw new Error(WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE) } } diff --git a/src/relay/relay-watcher-teardown-tracker.ts b/src/relay/relay-watcher-teardown-tracker.ts index 91b1460054d..785e4dd16ea 100644 --- a/src/relay/relay-watcher-teardown-tracker.ts +++ b/src/relay/relay-watcher-teardown-tracker.ts @@ -98,6 +98,17 @@ export class RelayWatcherTeardownTracker { rootPaths(): string[] { return [...this.pending.keys(), ...this.failed.keys()] } + + /** + * The capacity-release event: resolves once every teardown in flight right now has settled. + * + * `undefined` when nothing is unsubscribing, which is the only honest answer to "could a slot + * still come back?" — a failed teardown keeps its handles and releases nothing. + */ + settlePending(): Promise | undefined { + const inFlight = [...this.pending.values()] + return inFlight.length === 0 ? undefined : Promise.allSettled(inFlight).then(() => undefined) + } } function callUnsubscribe(subscription: WatcherProcessSubscription): Promise { diff --git a/src/shared/watch-root-capacity-refusal.ts b/src/shared/watch-root-capacity-refusal.ts new file mode 100644 index 00000000000..234fecb4ef7 --- /dev/null +++ b/src/shared/watch-root-capacity-refusal.ts @@ -0,0 +1,9 @@ +// Why a shared string rather than an error code: the refusal crosses the relay wire as a JSON-RPC +// error message, and relays deploy independently of clients. Both sides must spell it the same way, +// and a client that does not recognise it simply falls back to the ordinary unavailable handling. +export const WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE = 'Maximum number of file watchers reached' + +export function isWatchRootCapacityRefusal(error: unknown): boolean { + const message = (error as { message?: unknown } | null | undefined)?.message + return typeof message === 'string' && message.includes(WATCH_ROOT_CAPACITY_REFUSAL_MESSAGE) +} From 07e1f953a522081f9ff3be95933f2834f5eb3d0e Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:25 -0700 Subject: [PATCH 114/398] fix(ssh): log an unanswered native-deps probe instead of launching silently (#18000) * fix(ssh): log an unanswered native-deps probe instead of launching silently The wrongful rebuild used to be the only visible symptom of a dropped exec channel; #17979 removed it, so a real transport failure now leaves no trace. Matches the install-path sibling, whose callers log the same class of failure. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ssh/ssh-relay-deploy.ts | 9 ++++++++- src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts | 5 +++++ 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index af7346f5d35..dda85057205 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -869,10 +869,17 @@ async function probeRequiredNativeDeps( return { status: 'unverifiable', missing: [] } } return { status: 'blocked', missing } - } catch { + } catch (error) { signal?.throwIfAborted() // Why: an unanswered probe says nothing about the deps; reporting MISSING here reset and // recompiled healthy relays, turning one dropped exec channel into a multi-minute reconnect. + // Why: the wrongful rebuild was the only visible symptom, so without this line a dropped exec + // channel leaves no trace at all. + console.warn( + `[ssh-relay] Native deps probe unanswered at ${remoteDir}; treating as unverifiable: ${ + error instanceof Error ? error.message : String(error) + }` + ) return { status: 'unverifiable', missing: [] } } } diff --git a/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts b/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts index 3939c954f65..cb8f8c39c7b 100644 --- a/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-probe-verdict.test.ts @@ -153,6 +153,11 @@ describe('native-deps repair probe verdicts', () => { expect(warnings().some((message) => message.includes('Repairing missing native deps'))).toBe( false ) + // Why: the wrongful rebuild used to be the only visible symptom of a dropped exec channel. + expect( + warnings().some((message) => message.includes('Native deps probe unanswered')), + 'an unanswered probe must still leave a trace' + ).toBe(true) expect(commands.some((command) => command.includes(NODE_PTY_RESET))).toBe(false) expect(commands.some((command) => command.includes(WATCHER_RESET))).toBe(false) expect(commands.some((command) => command.includes('npm install'))).toBe(false) From 7458c3918167479568b5b29460207e6f0432c746 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:33 -0700 Subject: [PATCH 115/398] fix(ssh): declare a wedged relay link lost, and stop reading silence as a verdict (#17817) * fix(ssh): declare a wedged relay link lost instead of suppressing the dead-link check * fix(ssh): make the Windows deps probe exit 0 on a real load failure, like its POSIX twin * fix(relay): reap a client that has stopped answering instead of holding its leases forever * test(relay): feed the primary before asserting the reaper exemption holds * fix(ssh): keep a lost link's verdict unverifiable instead of reporting absence * refactor(ssh): read the exec timeout from its typed code, not the message text * fix(relay): bound a client that clears the handshake and then never frames anything * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ai-vault/ssh-session-list.test.ts | 23 ++- src/main/ai-vault/ssh-session-list.ts | 4 +- ...annel-multiplexer-saturation-wedge.test.ts | 82 +++++++++ src/main/ssh/ssh-channel-multiplexer.test.ts | 20 ++- src/main/ssh/ssh-channel-multiplexer.ts | 27 ++- src/main/ssh/ssh-relay-deploy.ts | 2 +- src/main/ssh/ssh-relay-exec-command.ts | 19 +- .../ssh/ssh-relay-native-deps-install.test.ts | 25 +++ .../ssh/ssh-remote-platform-detection.test.ts | 27 +++ src/main/ssh/ssh-remote-platform-detection.ts | 5 +- .../ssh/ssh-request-outcome-verdict.test.ts | 34 ++++ .../commit-message-model-discovery.ts | 4 +- ...ge-text-generation-model-discovery.test.ts | 25 ++- ...e-text-generation-remote-execution.test.ts | 39 ++++- .../source-control-remote-generation.ts | 4 +- src/relay/dispatcher-client-lifecycle.ts | 72 +++++++- src/relay/dispatcher-contract.ts | 12 ++ src/relay/dispatcher-frame-codec.ts | 3 + .../dispatcher-silent-client-reaper.test.ts | 163 ++++++++++++++++++ 19 files changed, 560 insertions(+), 30 deletions(-) create mode 100644 src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts create mode 100644 src/main/ssh/ssh-request-outcome-verdict.test.ts create mode 100644 src/relay/dispatcher-silent-client-reaper.test.ts diff --git a/src/main/ai-vault/ssh-session-list.test.ts b/src/main/ai-vault/ssh-session-list.test.ts index 4d1abd483c3..52454dda966 100644 --- a/src/main/ai-vault/ssh-session-list.test.ts +++ b/src/main/ai-vault/ssh-session-list.test.ts @@ -1,6 +1,9 @@ import { beforeEach, describe, expect, it, vi } from 'vitest' import type { AiVaultListResult, AiVaultSession } from '../../shared/ai-vault-types' -import { SSH_MUX_REQUEST_TIMEOUT_CODE } from '../ssh/ssh-channel-multiplexer' +import { + createSshDisposalError, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from '../ssh/ssh-channel-multiplexer' const requestActiveSshAiVaultSessionList = vi.fn() const getActiveSshAiVaultHostInfo = vi.fn() @@ -144,6 +147,24 @@ describe('scanSshAiVaultSessions', () => { ]) }) + it('reports a host issue when the relay link was declared lost on a real scan budget', async () => { + // Declaring a wedged link lost trades SSH_MUX_REQUEST_TIMEOUT for CONNECTION_LOST on this leg. + // Both are unverifiable, so both must surface as a host issue rather than falling through to a + // crawl that would publish an authoritative-looking empty list + // (docs/reference/ssh-execution-boundary.md). + requestActiveSshAiVaultSessionList.mockRejectedValue(createSshDisposalError('connection_lost')) + + const result = await scanSshAiVaultSessions('dev-box', undefined, { + timeoutMs: 20_000, + relayTimeoutMs: 15_000 + }) + + expect(scanRemoteAiVaultSessions).not.toHaveBeenCalled() + expect(result.issues).toEqual([ + expect.objectContaining({ executionHostId: 'ssh:dev-box', kind: 'host' }) + ]) + }) + it('still falls back when the relay budget was too short for a fair attempt', async () => { requestActiveSshAiVaultSessionList.mockRejectedValue(relayTimeoutError()) scanRemoteAiVaultSessions.mockResolvedValue({ diff --git a/src/main/ai-vault/ssh-session-list.ts b/src/main/ai-vault/ssh-session-list.ts index 7a02883f812..79359873d69 100644 --- a/src/main/ai-vault/ssh-session-list.ts +++ b/src/main/ai-vault/ssh-session-list.ts @@ -10,7 +10,7 @@ import { SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE } from '../providers/ssh-filesystem-dispatch' import { getActiveSshAiVaultHostInfo, requestActiveSshAiVaultSessionList } from '../ipc/ssh' -import { isSshMuxRequestTimeoutError } from '../ssh/ssh-channel-multiplexer' +import { isSshRequestOutcomeUnverifiable } from '../ssh/ssh-channel-multiplexer' import { createAiVaultScanCancelledError } from './ai-vault-scan-cancellation' import { scanRemoteAiVaultSessions } from './remote-session-scanner' import { parseAiVaultListResult } from './session-list-result-validation' @@ -84,7 +84,7 @@ async function scanOneSshHost( throw error } if ( - isSshMuxRequestTimeoutError(error) && + isSshRequestOutcomeUnverifiable(error) && (relayTimeoutMs === undefined || relayTimeoutMs >= MEANINGFUL_RELAY_SCAN_ATTEMPT_MS) ) { return sshScanIssueResult(executionHostId, targetId, errorMessage(error)) diff --git a/src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts b/src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts new file mode 100644 index 00000000000..fd9034f27f9 --- /dev/null +++ b/src/main/ssh/ssh-channel-multiplexer-saturation-wedge.test.ts @@ -0,0 +1,82 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { encodeKeepAliveFrame, KEEPALIVE_SEND_MS, TIMEOUT_MS } from './relay-protocol' +import { SshChannelMultiplexer, type MultiplexerTransport } from './ssh-channel-multiplexer' + +type WedgedTransport = MultiplexerTransport & { writes: Buffer[]; feed: (chunk: Buffer) => void } + +/** + * A transport that accepts the first write, then reports backpressure forever: no drain, and no + * write settlement. This is a half-open TCP link — the socket buffer filled and the peer's FIN + * never arrived — which is what sleep/resume and a dropped NAT mapping produce in the field. + */ +function createWedgedTransport(): WedgedTransport { + const writes: Buffer[] = [] + let onData: (chunk: Buffer) => void = () => {} + return { + write: (data) => { + writes.push(data) + return false + }, + onData: (callback) => { + onData = callback + }, + onClose: () => {}, + onDrain: () => () => {}, + supportsWriteSettlement: true, + writes, + feed: (chunk) => onData(chunk) + } +} + +describe('SshChannelMultiplexer on a transport that saturates and never drains', () => { + let transport: WedgedTransport + let mux: SshChannelMultiplexer + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(0) + transport = createWedgedTransport() + mux = new SshChannelMultiplexer(transport) + }) + + afterEach(() => { + mux.dispose() + vi.restoreAllMocks() + vi.useRealTimers() + }) + + it('declares the link lost instead of suppressing the dead-link check forever', async () => { + // Drive well past every health window: keepalive interval, dead-link timeout, and the + // wake-gap grace that resets staleness after a suspend. + await vi.advanceTimersByTimeAsync(TIMEOUT_MS * 10 + KEEPALIVE_SEND_MS) + + expect(mux.isDisposed()).toBe(true) + }) + + it('fails a request parked behind saturation rather than leaving it pending forever', async () => { + const settled = vi.fn() + mux.request('pty.spawn', {}).then( + () => settled('resolved'), + () => settled('rejected') + ) + + await vi.advanceTimersByTimeAsync(TIMEOUT_MS * 10 + KEEPALIVE_SEND_MS) + + expect(settled).toHaveBeenCalledWith('rejected') + }) + + it('keeps a slow-but-alive peer connected while its own keepalives arrive', async () => { + // The regression guard for the fix above: backpressure on our uplink is not evidence of + // death, and the relay's own keepalive is what proves it. + let seq = 1 + const inbound = setInterval(() => { + transport.feed(encodeKeepAliveFrame(seq++, 0)) + }, KEEPALIVE_SEND_MS) + try { + await vi.advanceTimersByTimeAsync(TIMEOUT_MS * 10 + KEEPALIVE_SEND_MS) + expect(mux.isDisposed()).toBe(false) + } finally { + clearInterval(inbound) + } + }) +}) diff --git a/src/main/ssh/ssh-channel-multiplexer.test.ts b/src/main/ssh/ssh-channel-multiplexer.test.ts index eeca0ce5108..c1c2e960439 100644 --- a/src/main/ssh/ssh-channel-multiplexer.test.ts +++ b/src/main/ssh/ssh-channel-multiplexer.test.ts @@ -71,7 +71,6 @@ type MuxInternals = { disposeHandlers: unknown[] lastReceivedAt: number unackedTimestamps: Map - writerSaturated: boolean } function getMuxInternals(instance: SshChannelMultiplexer): MuxInternals { @@ -354,9 +353,10 @@ describe('SshChannelMultiplexer', () => { expect(mux.isDisposed()).toBe(true) }) - it('suppresses false death while locally saturated and rebases both clocks on drain', () => { + it('survives local saturation while the peer keeps talking, and rebases both clocks on drain', () => { mux.dispose() let drain = (): void => {} + let feed: (chunk: Buffer) => void = () => {} const written: Buffer[] = [] const saturatedTransport: MultiplexerTransport = { write: (data) => { @@ -367,21 +367,29 @@ describe('SshChannelMultiplexer', () => { onDrain: (callback) => { drain = callback }, - onData: vi.fn(), + onData: (callback) => { + feed = callback + }, onClose: vi.fn() } mux = new SshChannelMultiplexer(saturatedTransport) vi.advanceTimersByTime(5_000) - expect(getMuxInternals(mux).writerSaturated).toBe(true) - vi.advanceTimersByTime(25_000) + // The writer parked after its first frame: that is the saturation this test is about. + expect(written).toHaveLength(1) + // Why: backpressure on our uplink is not evidence of death. The relay's own keepalive is, + // and only that inbound traffic may keep the link alive — suppressing the check on + // saturation alone wedged a half-open link forever (see the saturation-wedge suite). + for (let tick = 0; tick < 5; tick++) { + feed(encodeKeepAliveFrame(0, 0)) + vi.advanceTimersByTime(5_000) + } expect(mux.isDisposed()).toBe(false) expect(written).toHaveLength(1) drain() const resumedAt = Date.now() const internals = getMuxInternals(mux) - expect(internals.writerSaturated).toBe(false) expect(internals.lastReceivedAt).toBe(resumedAt) expect(new Set(internals.unackedTimestamps.values())).toEqual(new Set([resumedAt])) diff --git a/src/main/ssh/ssh-channel-multiplexer.ts b/src/main/ssh/ssh-channel-multiplexer.ts index d87a7ad8f5b..a8443f86f88 100644 --- a/src/main/ssh/ssh-channel-multiplexer.ts +++ b/src/main/ssh/ssh-channel-multiplexer.ts @@ -74,11 +74,19 @@ function sshMuxRequestTimeoutError(method: string, timeoutMs: number): Error { }) } -export function isSshMuxRequestTimeoutError(error: unknown): boolean { - return ( - error instanceof Error && - (error as Error & { code?: unknown }).code === SSH_MUX_REQUEST_TIMEOUT_CODE - ) +/** + * True when a request may have run on the host despite failing here. + * + * A response deadline and a link declared lost are the same verdict: the frame reached the wire and + * the peer's answer did not come back, so the work is `unverifiable`, never absent. Declaring a + * wedged link lost at TIMEOUT_MS turned what used to surface as SSH_MUX_REQUEST_TIMEOUT into + * CONNECTION_LOST, so callers that phrase the verdict to a user must branch on this rather than on + * the timeout alone or they silently start reporting absence + * (docs/reference/ssh-execution-boundary.md). + */ +export function isSshRequestOutcomeUnverifiable(error: unknown): boolean { + const code = error instanceof Error ? (error as Error & { code?: unknown }).code : undefined + return code === SSH_MUX_REQUEST_TIMEOUT_CODE || code === 'CONNECTION_LOST' } export class SshChannelMultiplexer { @@ -102,7 +110,6 @@ export class SshChannelMultiplexer { private disposed = false private disposeReason: 'shutdown' | 'connection_lost' | null = null private decoderReadPaused = false - private writerSaturated = false // Track the oldest unacked outgoing message timestamp private unackedTimestamps = new Map() @@ -587,7 +594,12 @@ export class SshChannelMultiplexer { this.sendKeepAlive() - if (this.disposed || resumedAfterWake || this.decoderReadPaused || this.writerSaturated) { + // Why: a saturated writer used to suppress this check outright, which wedged a half-open + // link forever — no drain, so no frame ever left, and the writer's single-outstanding + // liveness guard silenced the one probe that could have noticed. The relay sends its own + // keepalive every KEEPALIVE_SEND_MS, so a slow-but-alive peer still refreshes + // lastReceivedAt; only a link that delivers nothing inbound is declared lost. + if (this.disposed || resumedAfterWake || this.decoderReadPaused) { return } @@ -650,7 +662,6 @@ export class SshChannelMultiplexer { } private handleWriterSaturationChange(saturated: boolean): void { - this.writerSaturated = saturated if (!saturated && !this.disposed) { this.rebaseHealthClocks(Date.now()) } diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index dda85057205..c5cb2dd552c 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -841,7 +841,7 @@ async function probeRequiredNativeDeps( hostPlatform, nodePath, remoteDir, - `try { & ${powerShellLiteral(nodePath)} -e ${powerShellNativeArg(probeJs)} } catch { 'MISSING' }` + `try { & ${powerShellLiteral(nodePath)} -e ${powerShellNativeArg(probeJs)}; if ($LASTEXITCODE -ne 0) { 'MISSING' } } catch { 'MISSING' }` ) : // Why: no `2>/dev/null` — it discarded the only line that says why node never reached the // script. stderr stays its own stream so it can't be mistaken for the verdict, mirroring diff --git a/src/main/ssh/ssh-relay-exec-command.ts b/src/main/ssh/ssh-relay-exec-command.ts index fb0a4fee012..bec41c8d43e 100644 --- a/src/main/ssh/ssh-relay-exec-command.ts +++ b/src/main/ssh/ssh-relay-exec-command.ts @@ -24,6 +24,16 @@ type SshCommandTerminationError = Error & { sshChannelCloseConfirmed: boolean } +// Why: callers must tell "the host answered no" from "the host never answered". Matching the +// message text is what let an unanswered probe be read as a definitive negative. +export const SSH_EXEC_TIMEOUT_CODE = 'SSH_EXEC_TIMEOUT' + +export function isSshExecTimeout(error: unknown): boolean { + return ( + error instanceof Error && (error as Partial<{ code: string }>).code === SSH_EXEC_TIMEOUT_CODE + ) +} + export function isUnconfirmedSshCommandTermination( error: unknown ): error is SshCommandTerminationError { @@ -164,8 +174,13 @@ export async function execCommand( } const timeout = setTimeout(() => { requestTermination( - new Error( - `Command "${redactRelayInstallMarkerTokens(command)}" timed out after ${timeoutMs / 1000}s` + Object.assign( + new Error( + `Command "${redactRelayInstallMarkerTokens(command)}" timed out after ${ + timeoutMs / 1000 + }s` + ), + { code: SSH_EXEC_TIMEOUT_CODE } ) ) }, timeoutMs) diff --git a/src/main/ssh/ssh-relay-native-deps-install.test.ts b/src/main/ssh/ssh-relay-native-deps-install.test.ts index 01625816170..edf19b1251b 100644 --- a/src/main/ssh/ssh-relay-native-deps-install.test.ts +++ b/src/main/ssh/ssh-relay-native-deps-install.test.ts @@ -394,6 +394,31 @@ describe('installNativeDeps (via deployAndLaunchRelay)', () => { } }) + it('does not rewrite node_modules when the health probe never answered', async () => { + // Why (#14830): a wedged `require("node-pty")` makes the probe time out. Reading that silence + // as "every native dep is missing" sent a healthy install through npm install + rebuild that + // could not help, and the retry loop burned the whole deploy budget at "Deploying relay…". + const conn = makeMockConnection(sftpCapture) + vi.mocked(isRelayAlreadyInstalled).mockResolvedValue(true) + feed([ + '__ORCA_REMOTE_PLATFORM__ Linux x86_64', + '/home/u', + { reject: 'Command "node -e ..." timed out after 30s' } // health probe never answered + ]) + + await deployAndLaunchRelay(conn).catch(() => {}) + + const execCalls = vi.mocked(execCommand).mock.calls.map(([, c]) => c) + expect(execCalls.some((c) => c.includes('npm install'))).toBe(false) + expect(execCalls.some((c) => c.includes('npm rebuild'))).toBe(false) + + const warnMessages = warnSpy.mock.calls.map((args) => String(args[0] ?? '')) + expect(warnMessages.some((m) => m.includes('Repairing missing native deps'))).toBe(false) + // Why no log assertion: the behavioural claim above is the real one. Asserting on warn text + // pinned wording that main's landed probe verdict does not use, and #18000 adds its own. + expect(execCalls.some((c) => c.includes("rm -rf 'node_modules/node-pty'"))).toBe(false) + }) + it('lets a probe SSH-channel failure bubble up rather than silently mapping to MISSING', async () => { const conn = makeMockConnection(sftpCapture) feed( diff --git a/src/main/ssh/ssh-remote-platform-detection.test.ts b/src/main/ssh/ssh-remote-platform-detection.test.ts index aa3fb3a0916..a69f01d04b4 100644 --- a/src/main/ssh/ssh-remote-platform-detection.test.ts +++ b/src/main/ssh/ssh-remote-platform-detection.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import type { SshConnection } from './ssh-connection' +import { SSH_EXEC_TIMEOUT_CODE } from './ssh-relay-exec-command' const execCommandMock = vi.hoisted(() => vi.fn()) @@ -226,6 +227,32 @@ describe('detectRemoteHostPlatform failure reporting', () => { expect((error as Error).cause).toMatchObject({ reason: 2 }) }) + it('does not call an unmappable uname unsupported when the PowerShell probe timed out', async () => { + // The timed-out channel is the better explanation, and it is identified by execCommand's typed + // code — matching "timed out after Ns" in the message let any look-alike claim the branch. + execCommandMock + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ CYGWIN_NT-10.0 x86_64\n') + .mockRejectedValueOnce( + Object.assign(new Error('Command "pwsh" timed out after 30s'), { + code: SSH_EXEC_TIMEOUT_CODE + }) + ) + + const error = await detectRemoteHostPlatform(conn).catch((err: unknown) => err) + + expect(error).toBeInstanceOf(Error) + expect((error as Error).message).not.toMatch(/unsupported/iu) + expect((error as Error).cause).toMatchObject({ code: SSH_EXEC_TIMEOUT_CODE }) + }) + + it('does not read a look-alike timeout message as a timed-out channel', async () => { + execCommandMock + .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ FreeBSD x86_64\n') + .mockRejectedValueOnce(new Error('the agent it launched timed out after 30s')) + + await expect(detectRemoteHostPlatform(conn)).resolves.toBeNull() + }) + it('falls through to PowerShell for a Cygwin uname it cannot map', async () => { execCommandMock .mockResolvedValueOnce('__ORCA_REMOTE_PLATFORM__ CYGWIN_NT-10.0 x86_64\n') diff --git a/src/main/ssh/ssh-remote-platform-detection.ts b/src/main/ssh/ssh-remote-platform-detection.ts index 5f830fb7449..6fd0f87c767 100644 --- a/src/main/ssh/ssh-remote-platform-detection.ts +++ b/src/main/ssh/ssh-remote-platform-detection.ts @@ -5,7 +5,7 @@ import { } from '../../shared/process-output-field-scanner' import { parseUnameToRelayPlatform, type RelayPlatform } from './relay-protocol' import { execCommand } from './ssh-relay-deploy-helpers' -import { isUnconfirmedSshCommandTermination } from './ssh-relay-exec-command' +import { isSshExecTimeout, isUnconfirmedSshCommandTermination } from './ssh-relay-exec-command' import { isSshSessionLimitError } from './ssh-session-limit-error' import { getRemoteHostPlatform, type RemoteHostPlatform } from './ssh-remote-platform' import { powerShellCommand } from './ssh-remote-powershell' @@ -14,7 +14,6 @@ const PLATFORM_PROBE_MARKER = '__ORCA_REMOTE_PLATFORM__' const MAX_UNAME_FIELD_CHARS = 64 const MAX_THROWN_OUTPUT_CHARS = 200 const MAX_LOGGED_OUTPUT_CHARS = 1000 -const EXEC_TIMEOUT_MESSAGE = /timed out after \d+s$/u type PlatformProbeOutcome = | { kind: 'detected'; platform: RelayPlatform } @@ -89,7 +88,7 @@ function isTransportShapedError(error: unknown): boolean { return ( isSshSessionLimitError(error) || isUnconfirmedSshCommandTermination(error) || - (error instanceof Error && EXEC_TIMEOUT_MESSAGE.test(error.message)) + isSshExecTimeout(error) ) } diff --git a/src/main/ssh/ssh-request-outcome-verdict.test.ts b/src/main/ssh/ssh-request-outcome-verdict.test.ts new file mode 100644 index 00000000000..e21368d2559 --- /dev/null +++ b/src/main/ssh/ssh-request-outcome-verdict.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'vitest' +import { + createSshDisposalError, + isSshRequestOutcomeUnverifiable, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from './ssh-channel-multiplexer' + +// docs/reference/ssh-execution-boundary.md: the vocabulary is live / unverifiable / exited, and +// loss of contact is never evidence of absence. Three call sites phrase this verdict to a user, so +// collapsing "unverifiable" into "could not be reached" is a user-visible lie. +describe('SSH request outcome verdict', () => { + it('treats a response deadline as unverifiable', () => { + const timedOut = Object.assign(new Error('Request "x" timed out after 30000ms'), { + code: SSH_MUX_REQUEST_TIMEOUT_CODE + }) + expect(isSshRequestOutcomeUnverifiable(timedOut)).toBe(true) + }) + + it('treats a link declared lost as unverifiable, not as absence', () => { + // The regression this exists for: declaring a wedged link lost at TIMEOUT_MS made those + // requests surface CONNECTION_LOST where they used to surface a timeout, silently downgrading + // the honest "may still be running on the remote host" to "could not be reached". + expect(isSshRequestOutcomeUnverifiable(createSshDisposalError('connection_lost'))).toBe(true) + }) + + it('does not claim unverifiable for a deliberate shutdown', () => { + expect(isSshRequestOutcomeUnverifiable(createSshDisposalError('shutdown'))).toBe(false) + }) + + it('does not claim unverifiable for an ordinary failure', () => { + expect(isSshRequestOutcomeUnverifiable(new Error('boom'))).toBe(false) + expect(isSshRequestOutcomeUnverifiable(undefined)).toBe(false) + }) +}) diff --git a/src/main/text-generation/commit-message-model-discovery.ts b/src/main/text-generation/commit-message-model-discovery.ts index 43873f01bf2..ec6be0f12bb 100644 --- a/src/main/text-generation/commit-message-model-discovery.ts +++ b/src/main/text-generation/commit-message-model-discovery.ts @@ -3,7 +3,7 @@ import type { CommitMessagePlan } from '../../shared/commit-message-plan' import { getAgentModelProbeSpec } from '../../shared/agent-model-probe-spec' import type { TuiAgent } from '../../shared/tui-agent' import { resolveCodexHomeProcessLockKeyForSpawnEnv } from '../codex-cli/codex-home-process-lock' -import { isSshMuxRequestTimeoutError } from '../ssh/ssh-channel-multiplexer' +import { isSshRequestOutcomeUnverifiable } from '../ssh/ssh-channel-multiplexer' import { WINDOWS_BATCH_UNSAFE_ARGUMENTS_ERROR } from '../win32-utils' import { finalizeModelDiscoveryOutput, @@ -206,7 +206,7 @@ export async function discoverModelsRemote(input: { console.error('[commit-message] Remote model discovery request failed:', error) return { success: false, - error: isSshMuxRequestTimeoutError(error) + error: isSshRequestOutcomeUnverifiable(error) ? `${spec.label} model discovery took longer than ${SOURCE_CONTROL_GENERATION_TIMEOUT_MS / 1000}s and may still be running on the remote host.` : `${spec.label} model discovery could not be reached on the remote PATH. Try again after the SSH connection recovers.` } diff --git a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts index 7ef438c06ae..4bc1a720f05 100644 --- a/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts +++ b/src/main/text-generation/commit-message-text-generation-model-discovery.test.ts @@ -1,7 +1,10 @@ import { spawn } from 'node:child_process' import type * as ChildProcess from 'node:child_process' import { beforeEach, describe, expect, it, vi } from 'vitest' -import { SSH_MUX_REQUEST_TIMEOUT_CODE } from '../ssh/ssh-channel-multiplexer' +import { + createSshDisposalError, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from '../ssh/ssh-channel-multiplexer' import { discoverCommitMessageModelsLocal, discoverCommitMessageModelsRemote @@ -475,6 +478,26 @@ describe('generateCommitMessageFromContext', () => { }) }) + it('keeps the unverifiable wording when the link is declared lost instead of timing out', async () => { + // Same regression as the exec leg: a wedged link now disposes the mux before the response + // deadline, so this branch sees CONNECTION_LOST. Reporting "could not be reached" for it + // asserts absence the client never observed (docs/reference/ssh-execution-boundary.md). + const result = await discoverCommitMessageModelsRemote( + 'cursor', + '/remote/repo', + async () => { + throw createSshDisposalError('connection_lost') + }, + 'npx cursor-agent' + ) + + expect(result).toEqual({ + success: false, + error: + 'Cursor model discovery took longer than 60s and may still be running on the remote host.' + }) + }) + it('reports remote model discovery spawn failures with remote install guidance', async () => { const result = await discoverCommitMessageModelsRemote('cursor', '/remote/repo', async () => ({ stdout: '', diff --git a/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts b/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts index 10442b12f0e..26708149d51 100644 --- a/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts +++ b/src/main/text-generation/commit-message-text-generation-remote-execution.test.ts @@ -1,5 +1,8 @@ import { describe, expect, it } from 'vitest' -import { SSH_MUX_REQUEST_TIMEOUT_CODE } from '../ssh/ssh-channel-multiplexer' +import { + createSshDisposalError, + SSH_MUX_REQUEST_TIMEOUT_CODE +} from '../ssh/ssh-channel-multiplexer' import { generateCommitMessageFromContext } from './commit-message-text-generation' describe('generateCommitMessageFromContext', () => { @@ -76,6 +79,40 @@ describe('generateCommitMessageFromContext', () => { }) }) + it('keeps the unverifiable wording when the link is declared lost instead of timing out', async () => { + // Declaring a wedged link lost disposes the mux before the 30s response deadline, so this leg + // now sees CONNECTION_LOST where it used to see SSH_MUX_REQUEST_TIMEOUT. Both mean the frame + // reached the wire and no answer came back, so both must keep "may still be running" — falling + // through to "could not be reached" asserts absence the client cannot observe + // (docs/reference/ssh-execution-boundary.md). + const result = await generateCommitMessageFromContext( + { + branch: 'main', + stagedSummary: 'M\tREADME.md', + stagedPatch: '+hello' + }, + { + agentId: 'custom', + model: '', + customAgentCommand: 'agent' + }, + { + kind: 'remote', + cwd: '/repo', + missingBinaryLocation: 'remote PATH', + execute: async () => { + throw createSshDisposalError('connection_lost') + } + } + ) + + expect(result).toEqual({ + success: false, + error: 'agent took longer than 60s to respond and may still be running on the remote host.', + canceled: undefined + }) + }) + it('sanitizes remote execution transport failures', async () => { const result = await generateCommitMessageFromContext( { diff --git a/src/main/text-generation/source-control-remote-generation.ts b/src/main/text-generation/source-control-remote-generation.ts index 5fe18d34718..ece1a6de1d8 100644 --- a/src/main/text-generation/source-control-remote-generation.ts +++ b/src/main/text-generation/source-control-remote-generation.ts @@ -1,5 +1,5 @@ import type { CommitMessagePlan } from '../../shared/commit-message-plan' -import { isSshMuxRequestTimeoutError } from '../ssh/ssh-channel-multiplexer' +import { isSshRequestOutcomeUnverifiable } from '../ssh/ssh-channel-multiplexer' import { WINDOWS_BATCH_UNSAFE_ARGUMENTS_ERROR } from '../win32-utils' import { finalizeFromAgentOutput, @@ -25,7 +25,7 @@ export async function runRemoteSourceControlPlan(input: { result = await target.execute(plan, target.cwd, SOURCE_CONTROL_GENERATION_TIMEOUT_MS, operation) } catch (error) { console.error('[commit-message] Remote generator request failed:', error) - if (isSshMuxRequestTimeoutError(error)) { + if (isSshRequestOutcomeUnverifiable(error)) { return { success: false, error: `${plan.label} took longer than ${SOURCE_CONTROL_GENERATION_TIMEOUT_MS / 1000}s to respond and may still be running on the remote host.` diff --git a/src/relay/dispatcher-client-lifecycle.ts b/src/relay/dispatcher-client-lifecycle.ts index 0aff8d977ac..672c51763e6 100644 --- a/src/relay/dispatcher-client-lifecycle.ts +++ b/src/relay/dispatcher-client-lifecycle.ts @@ -1,4 +1,4 @@ -import { FrameDecoder, KEEPALIVE_SEND_MS, encodeKeepAliveFrame } from './protocol' +import { FrameDecoder, KEEPALIVE_SEND_MS, TIMEOUT_MS, encodeKeepAliveFrame } from './protocol' import type { PtyConsumerCloseCause } from '../shared/pty-consumer-session-contract' import type { DispatcherClientWriter, @@ -12,6 +12,12 @@ import type { } from './dispatcher-contract' import { RelayDispatcherClientState } from './dispatcher-client-state' +// Why: a client that clears the endpoint handshake and then never frames anything is invisible to +// the silence window -- it has no lastReceivedAt to go stale -- so its socket, writer and client +// entry are held for the life of the relay. The bound is deliberately several windows wide: a real +// client frames immediately after the handshake, so only a peer that is already gone reaches it. +const SILENT_CONNECT_TIMEOUT_MS = TIMEOUT_MS * 6 + export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClientState { // Why: redirect outgoing frames to the reconnected socket without rebuilding the dispatcher + handler tree. // Why: a new multiplexer restarts at seq=1; reset state to avoid stalled acknowledgements. @@ -122,6 +128,9 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie bulkChain: Promise.resolve(), nextOutgoingSeq: 1, highestReceivedSeq: 0, + attachedAt: Date.now(), + lastReceivedAt: null, + keepaliveObserved: false, generation: 0, closed: false, droppedNotificationLog: null, @@ -144,6 +153,9 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie protected resetClient(client: RelayClient): void { client.nextOutgoingSeq = 1 client.highestReceivedSeq = 0 + client.attachedAt = Date.now() + client.lastReceivedAt = null + client.keepaliveObserved = false client.decoder.reset() client.generation++ client.closed = false @@ -163,14 +175,30 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie } protected startKeepalive(): void { + let lastTickAt = Date.now() this.keepaliveTimer = setInterval(() => { if (this.disposed) { return } + const now = Date.now() + // Why this threshold and not TIMEOUT_MS: a healthy client answers the PREVIOUS tick, so its + // lastReceivedAt is already up to KEEPALIVE_SEND_MS + RTT old. A tick gap beyond + // TIMEOUT_MS - KEEPALIVE_SEND_MS therefore pushes staleness past the window on its own, and a + // rebase armed at TIMEOUT_MS would not have fired -- reaping every client after a host + // suspend, a VM migration, or the relay's own event loop stalling. Mirrors the client's + // WAKE_GAP_MS guard (ssh-channel-multiplexer.ts). + const resumedAfterPause = now - lastTickAt >= TIMEOUT_MS - KEEPALIVE_SEND_MS + lastTickAt = now for (const client of this.clients.values()) { if (client.closed) { continue } + if (resumedAfterPause) { + client.attachedAt = now + if (client.lastReceivedAt !== null) { + client.lastReceivedAt = now + } + } client.writer.enqueue( 'liveness', () => { @@ -180,11 +208,53 @@ export abstract class RelayDispatcherClientLifecycle extends RelayDispatcherClie 13 ) } + this.reapSilentClients(now) }, KEEPALIVE_SEND_MS) // Why: unref so the keepalive interval doesn't pin the event loop and block process exit. this.keepaliveTimer.unref() } + /** + * Drop the transport of a client that has gone silent. The relay's writer parks forever on a + * half-open link and nothing else ever notices, so an abandoned viewer kept its owner lease and + * left every PTY it held paused — the shape behind the "SSH degrades until I cannot connect at + * all" reports. Reaping is a statement about the TRANSPORT only: the cause stays the cautious + * 'local' default because silence is not evidence the peer died, and the PTYs stay live for the + * replacement client to reclaim (docs/reference/ssh-execution-boundary.md). + */ + private reapSilentClients(now: number): void { + for (const client of Array.from(this.clients.values())) { + // Why the primary is exempt: closing it tears down the relay's own stdin/stdout, and nothing + // in production revives it -- setWrite() has no non-test caller. The leak this exists for is + // a socket client holding an owner lease, and the launch channel's own liveness is already + // owned by the client-side dead-link check. + if (client === this.primaryClient) { + continue + } + if (client.closed) { + continue + } + // Why a client that has never spoken gets its own, much wider bound: a relay is launched + // before its client finishes handshaking, and on a slow link that can exceed the silence + // window, so judging it there would break the connect it is still completing. Leaving it + // unbounded instead held its socket and client entry forever. + if (client.lastReceivedAt === null) { + if (now - client.attachedAt > SILENT_CONNECT_TIMEOUT_MS) { + this.closeClient(client, new Error('Relay client never spoke'), true) + } + continue + } + // Why keepaliveObserved gates this: not every client speaks the keepalive protocol. The + // remote `orca` CLI sends one `orca.cli` request and waits for a result budgeted in minutes + // (src/relay/remote-cli-timeout.ts), so judging it on inbound silence would kill + // `terminal wait`, `--wait` and `orchestration ask` after 20s. + if (!client.keepaliveObserved || now - client.lastReceivedAt <= TIMEOUT_MS) { + continue + } + this.closeClient(client, new Error('Relay client stopped answering'), true) + } + } + protected closeClient( client: RelayClient, error: Error, diff --git a/src/relay/dispatcher-contract.ts b/src/relay/dispatcher-contract.ts index 15d4578c9bf..23c62431d2c 100644 --- a/src/relay/dispatcher-contract.ts +++ b/src/relay/dispatcher-contract.ts @@ -51,6 +51,18 @@ export type RelayClient = { bulkChain: Promise nextOutgoingSeq: number highestReceivedSeq: number + // Why: a client that never frames anything has no staleness to measure, so the silence window + // cannot see it. Its attach time is the only clock it has. + attachedAt: number + // Why: the relay had no inbound-liveness signal at all, so a half-open client was never reaped + // and kept its owner lease and paused PTYs indefinitely. + lastReceivedAt: number | null + // Why silence is only held against a client that sends keepalives: not every client speaks that + // protocol. The remote `orca` CLI opens the socket, sends one `orca.cli` request and then waits + // for a result that is deliberately budgeted in minutes (src/relay/remote-cli-timeout.ts), so + // judging it on inbound silence would kill `terminal wait`, `--wait` and `orchestration ask` + // after 20s. Only a client that has proven it participates is eligible. + keepaliveObserved: boolean generation: number closed: boolean droppedNotificationLog: DroppedProducerNotificationLog | null diff --git a/src/relay/dispatcher-frame-codec.ts b/src/relay/dispatcher-frame-codec.ts index b25a0d608ee..5a25fa1100f 100644 --- a/src/relay/dispatcher-frame-codec.ts +++ b/src/relay/dispatcher-frame-codec.ts @@ -15,11 +15,14 @@ import { RelayDispatcherCapacitySignals } from './dispatcher-capacity-signals' export abstract class RelayDispatcherFrameCodec extends RelayDispatcherCapacitySignals { protected handleFrame(client: RelayClient, frame: DecodedFrame): void { + // Before the KeepAlive early return: a keepalive is the only proof a quiet client is still there. + client.lastReceivedAt = Date.now() if (frame.id > client.highestReceivedSeq) { client.highestReceivedSeq = frame.id } if (frame.type === MessageType.KeepAlive) { + client.keepaliveObserved = true return } diff --git a/src/relay/dispatcher-silent-client-reaper.test.ts b/src/relay/dispatcher-silent-client-reaper.test.ts new file mode 100644 index 00000000000..b26fdd6bf4d --- /dev/null +++ b/src/relay/dispatcher-silent-client-reaper.test.ts @@ -0,0 +1,163 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { RelayDispatcher } from './dispatcher' +import { encodeJsonRpcFrame, encodeKeepAliveFrame, KEEPALIVE_SEND_MS, TIMEOUT_MS } from './protocol' + +// The relay had no inbound-liveness signal at all: its writer parks forever on a half-open link, so +// an abandoned viewer kept its owner lease and left the PTYs it held paused until the process died. +describe('RelayDispatcher silent-client reaper', () => { + let dispatcher: RelayDispatcher + + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(0) + }) + + afterEach(() => { + dispatcher.dispose() + vi.useRealTimers() + }) + + it('detaches a client that spoke once and then stopped answering', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + dispatcher.feedClient(clientId, encodeKeepAliveFrame(1, 0)) + + vi.advanceTimersByTime(TIMEOUT_MS + KEEPALIVE_SEND_MS * 2) + + // 'local', not a peer close: silence is not evidence the peer died, and a consumer that read it + // as one would shorten the owner grace on a session that is still there. + expect(detachListener).toHaveBeenCalledWith(clientId, 'local') + }) + + it('keeps a quiet but answering client attached', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + // A client with nothing to say still answers the keepalive; that is the only proof required. + for (let tick = 0; tick < 10; tick += 1) { + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + dispatcher.feedClient(clientId, encodeKeepAliveFrame(tick + 1, 0)) + } + + // Asserted against this client specifically: the unattached primary sink has no peer answering + // it in this harness, so it is expected to be reaped and says nothing about the case under test. + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('does not reap every client on the first tick after the host slept', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + dispatcher.attachClient(() => true) + + // One tick fires far late because the process was paused, not because the peers went away. + vi.setSystemTime(10 * 60_000) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + expect(detachListener).not.toHaveBeenCalled() + }) + + it('does not judge a client that has not spoken yet by the silence window', () => { + // A relay is launched before its client finishes handshaking, and on a slow link that can + // outlast the window. Reaping there would break the connect the client is still completing. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + vi.advanceTimersByTime(TIMEOUT_MS * 5) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('still bounds a client that never speaks at all', () => { + // Otherwise it is invisible to the silence window forever -- no lastReceivedAt to go stale -- + // and its socket, writer and client entry are held for the life of the relay. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + vi.advanceTimersByTime(TIMEOUT_MS * 7) + + expect(detachListener).toHaveBeenCalledWith(clientId, 'local') + }) + + it("does not spend a mute client's connect budget while the host was suspended", () => { + // Same rebase the silence window gets: a paused process is not a peer that went away. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + vi.setSystemTime(60 * 60_000) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('never reaps a client that does not send keepalives at all', () => { + // The remote `orca` CLI opens the socket, sends one `orca.cli` request and then waits for a + // result budgeted in minutes (remote-cli-timeout.ts: 5min default, 10min for wait, 11min for + // orchestration ask). It has no keepalive timer, so judging it on inbound silence would abort + // `terminal wait`, `--wait` and `orchestration ask` after 20s. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + dispatcher.feedClient( + clientId, + encodeJsonRpcFrame({ jsonrpc: '2.0', id: 1, method: 'orca.cli', params: {} }, 1, 0) + ) + + vi.advanceTimersByTime(TIMEOUT_MS * 20) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('does not reap a healthy client when the relay itself stalls for most of the window', () => { + // The dead band this exists for: a healthy client answers the PREVIOUS tick, so its + // lastReceivedAt is already ~KEEPALIVE_SEND_MS old. A tick gap short of TIMEOUT_MS still pushes + // staleness past the window, so a rebase armed at TIMEOUT_MS would never fire and every client + // would be reaped after a host suspend, VM migration, or an event-loop stall. + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + const clientId = dispatcher.attachClient(() => true) + + // The client answers at t=5s, then the next tick at t=10s finds it already ~5s stale — normal. + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + dispatcher.feedClient(clientId, encodeKeepAliveFrame(1, 0)) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + // Now the relay stalls: the clock jumps but no tick runs, so the following tick lands 17s after + // the last one. Staleness is 22s (past the window) while the tick gap is under TIMEOUT_MS, so a + // rebase armed at TIMEOUT_MS would not fire and this healthy client would be reaped. + vi.setSystemTime(Date.now() + 12_000) + vi.advanceTimersByTime(KEEPALIVE_SEND_MS) + + expect(detachListener).not.toHaveBeenCalledWith(clientId, expect.anything()) + }) + + it('never reaps the primary client, whose sink cannot be revived', () => { + const detachListener = vi.fn() + dispatcher = new RelayDispatcher(() => true) + dispatcher.onClientDetached(detachListener) + + // Feed the primary first, or the test proves nothing: an unfed client is skipped by the + // never-spoken and no-keepalive guards, so it survives whether or not the exemption exists. + // A real primary answers keepalives, so the exemption is the only thing standing between it + // and the reaper. + dispatcher.feed(encodeKeepAliveFrame(1, 0)) + + vi.advanceTimersByTime(TIMEOUT_MS * 20) + + // Client id 1 is the primary sink; closing it would tear down the relay's own stdin/stdout and + // nothing in production calls setWrite() to bring it back. + expect(detachListener).not.toHaveBeenCalled() + }) +}) From 033a2a64e17f97d7a69bcaba7f1e21cf5fb82346 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:40 -0700 Subject: [PATCH 116/398] fix(remote-runtime): make every advertised recovery attempt reachable, and stop two recovery latches (#17822) * fix(remote-runtime): derive the recovery budget and stop faking a spent window #11305: RECOVERY_DELAYS_MS summed to 60,750ms against a hand-written REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS of 60,000ms, so the ladder's tail was unreachable. Derive the deadline from the schedule plus one RPC timeout per step so a half-open link can actually reach every backoff step, and pin the relation with a test that fails if the sum ever outgrows the budget. #12683: markDisconnected() is a UI latch, not proof the auto-recovery window ran out. Track deadline expiry on the recovery state and only let that license the same-handle reattach that bypasses require-replacement fencing. #12684: a recoverable connect() failure latched 'disconnected' with no armed retry, no parked retry and a Reconnect button that returned false. Schedule a bounded retry (which the deadline parks for online/resume) and let the button fire a parked retry. * fix(remote-runtime): stop a post-latch connect failure from re-arming the recovery window The last attempt's RPC budget expires at the same instant as the deadline, so a silently dropped link rejects after phase latched to 'disconnected'. begin() then started a fresh full-length window, so the budget never actually expired. Park the retry under the latched epoch instead, which keeps online/resume/Reconnect armed even when the deadline lands mid-attempt with nothing scheduled. Also fences the same-handle end-reuse window on its own 60s constant so the derived recovery budget no longer silently triples an unrelated stale-handle check. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- ...minalRemoteRuntimeReconnectBanner.test.tsx | 2 +- .../TerminalRemoteRuntimeReconnectBanner.tsx | 2 +- ...e-runtime-connect-failure-recovery.test.ts | 160 ++++++++++++++++++ ...ge-toast-flood-and-stuck-reconnect.test.ts | 5 +- ...mote-runtime-pty-deadline-reattach.test.ts | 67 +++++++- ...runtime-pty-latched-pane-retention.test.ts | 15 +- .../remote-runtime-pty-recovery-state.test.ts | 70 +++++++- .../remote-runtime-pty-recovery-state.ts | 47 ++++- ...-transport-create-outcome-recovery.test.ts | 11 +- ...ty-transport-stale-handle-recovery.test.ts | 5 +- ...ransport-sticky-replacement-policy.test.ts | 3 +- ...ime-pty-transport-stream-reconnect.test.ts | 3 +- ...-pty-transport-web-mirror-recovery.test.ts | 15 +- .../remote-runtime-pty-transport.ts | 57 ++++++- src/renderer/src/i18n/locales/es.json | 2 +- src/renderer/src/i18n/locales/ja.json | 2 +- src/renderer/src/i18n/locales/ko.json | 2 +- src/renderer/src/i18n/locales/zh.json | 2 +- 18 files changed, 426 insertions(+), 44 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts diff --git a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx index 70341d7a4f6..824a6a80744 100644 --- a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.test.tsx @@ -17,7 +17,7 @@ describe('TerminalRemoteRuntimeReconnectBanner', () => { render() expect(screen.getByText('Reconnecting to remote runtime')).toBeInTheDocument() - expect(screen.getByText(/retry for up to one minute/)).toBeInTheDocument() + expect(screen.getByText(/retrying automatically/)).toBeInTheDocument() expect(screen.queryByRole('button')).not.toBeInTheDocument() }) diff --git a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx index c629fa71a36..1cf0fee0903 100644 --- a/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalRemoteRuntimeReconnectBanner.tsx @@ -50,7 +50,7 @@ export function TerminalRemoteRuntimeReconnectBanner({ {retrying ? translate( 'auto.components.terminal.pane.TerminalRemoteRuntimeReconnectBanner.retryingBody', - 'Orca will retry for up to one minute. This terminal will resume if the connection returns.' + 'Orca is retrying automatically. This terminal will resume if the connection returns.' ) : translate( 'auto.components.terminal.pane.TerminalRemoteRuntimeReconnectBanner.disconnectedBody', diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts new file mode 100644 index 00000000000..6c12277f2c0 --- /dev/null +++ b/src/renderer/src/components/terminal-pane/remote-runtime-connect-failure-recovery.test.ts @@ -0,0 +1,160 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + createRemoteRuntimeTransportMocks, + type MultiplexSubscriptionCallbacks +} from './remote-runtime-pty-transport-test-harness' +import { + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS, + REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS +} from './remote-runtime-pty-recovery-state' + +let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null +let resolvedPaneHandle = 'terminal-1' + +const { runtimeCall, resetRemoteRuntimeTransport } = createRemoteRuntimeTransportMocks({ + getCallbacks: () => subscriptionCallbacks, + setCallbacks: (callbacks) => { + subscriptionCallbacks = callbacks + }, + getResolvedPaneHandle: () => resolvedPaneHandle, + setResolvedPaneHandle: (handle) => { + resolvedPaneHandle = handle + } +}) + +// #12684: connect() classified these failures as recoverable and then latched 'disconnected' with +// nothing armed — no backoff timer, no parked retry, and a Reconnect button that returned false. +describe('recoverable connect failures on a remote runtime pane', () => { + let resolvePaneCalls = 0 + + function installUnreachableRuntime(): void { + resolvePaneCalls = 0 + runtimeCall.mockImplementation(async (args: { method: string }) => { + if (args.method === 'terminal.resolvePane') { + resolvePaneCalls += 1 + } + throw Object.assign(new Error('Remote Orca runtime closed the connection.'), { + code: 'remote_runtime_unavailable' + }) + }) + } + + // Why: installUnreachableRuntime() rejects synchronously, so every failure lands during a backoff + // wait. A silently dropped link instead burns the whole RPC budget, so the rejection arrives while + // the attempt is still in flight — including after the auto-recovery deadline has already latched. + function installSilentlyDroppedRuntime(): void { + resolvePaneCalls = 0 + runtimeCall.mockImplementation(async (args: { method: string }) => { + if (args.method === 'terminal.resolvePane') { + resolvePaneCalls += 1 + } + await new Promise((resolve) => { + setTimeout(resolve, REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS) + }) + throw Object.assign(new Error('Remote Orca runtime closed the connection.'), { + code: 'remote_runtime_unavailable' + }) + }) + } + + beforeEach(() => { + resetRemoteRuntimeTransport() + }) + + it('keeps retrying a recoverable connect failure instead of latching immediately', async () => { + vi.useFakeTimers() + try { + installUnreachableRuntime() + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const onError = vi.fn() + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + + await transport.connect({ + url: '', + sessionId: 'remote:env-1@@', + callbacks: { onError } + }) + + // Loss of contact is unverifiable, not a dead terminal: automatic recovery must still be running. + expect(resolvePaneCalls).toBe(1) + expect(transport.getRecoveryState?.().phase).toBe('backoff') + expect(onError).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(1_000) + expect(resolvePaneCalls).toBeGreaterThan(1) + + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) + + it('leaves both revival paths armed once the recovery window is spent', async () => { + vi.useFakeTimers() + try { + installUnreachableRuntime() + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + // Why dynamic: resetRemoteRuntimeTransport() re-registers the module graph, and the retry + // registry only sees panes from the same instance the transport was loaded from. + const { retryAllRemoteRuntimePtyRecoveriesNow } = + await import('./remote-runtime-pty-recovery-state') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + + await transport.connect({ url: '', sessionId: 'remote:env-1@@', callbacks: {} }) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 1_000) + + expect(transport.getRecoveryState?.().phase).toBe('disconnected') + const callsAtCutoff = resolvePaneCalls + + // The cutoff stops self-initiated retries only; online/resume must still find a parked retry. + expect(retryAllRemoteRuntimePtyRecoveriesNow()).toBe(1) + await vi.advanceTimersByTimeAsync(1_000) + expect(resolvePaneCalls).toBeGreaterThan(callsAtCutoff) + + // ...and so must the Reconnect button, which returned false before #12684. + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 1_000) + expect(transport.getRecoveryState?.().phase).toBe('disconnected') + expect(transport.retryRecovery?.()).toBe(true) + + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) + it('keeps the window bounded when a silent drop fails after the deadline latched', async () => { + vi.useFakeTimers() + try { + installSilentlyDroppedRuntime() + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'tab-1', + leafId: 'pane:1' + }) + + void transport.connect({ url: '', sessionId: 'remote:env-1@@', callbacks: {} }) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS * 2) + + // The in-flight rejection must not begin a new epoch; that re-arms a full-length window forever. + expect(transport.getRecoveryState?.().phase).toBe('disconnected') + const callsAtCutoff = resolvePaneCalls + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS * 2) + expect(resolvePaneCalls).toBe(callsAtCutoff) + + // A deadline that lands mid-attempt parks nothing, so the latch must still stay revivable. + expect(transport.retryRecovery?.()).toBe(true) + + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts index b797a0f5877..9ac752b2b31 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-outage-toast-flood-and-stuck-reconnect.test.ts @@ -26,6 +26,7 @@ import { import { RuntimeRpcCallQueueOverloadError } from '../../../../shared/runtime-rpc-call-queue' import { withRemoteRuntimeTailscaleHint } from '../../../../shared/remote-runtime-tailscale-hint' import type { PtyTransportRecoveryState } from './pty-transport-types' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' const ELECTRON_IPC_PREFIX = "Error invoking remote method 'runtimeEnvironments:call': " @@ -409,7 +410,7 @@ describe('remote runtime outage: toast flood and stuck reconnect (issue3)', () = await vi.advanceTimersByTimeAsync(16_000) // Auto-recovery deadline latches the pane 'disconnected'. - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') // Connectivity restored; 'online'/system-resume trigger fires. @@ -417,7 +418,7 @@ describe('remote runtime outage: toast flood and stuck reconnect (issue3)', () = await vi.advanceTimersByTimeAsync(16_000) // Latch again, then the user clicks the Reconnect banner. - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) transport.retryRecovery?.() await vi.advanceTimersByTimeAsync(16_000) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts index 886516414ca..61758efb299 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-deadline-reattach.test.ts @@ -8,6 +8,7 @@ import { encodeTerminalStreamText } from '../../../../shared/terminal-stream-protocol' import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' describe('remote runtime pty reattach after the bounded recovery window', () => { const runtimeCall = vi.fn() @@ -198,7 +199,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => expect(handleEvents.getWebSessionTerminalHandleSubscriberCountForTests()).toBe(1) expect(transport.getRecoveryState?.().phase).not.toBe('disconnected') - await vi.advanceTimersByTimeAsync(50_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') // The cutoff must not tear down the accepted-snapshot listener; it is the only path back. expect(handleEvents.getWebSessionTerminalHandleSubscriberCountForTests()).toBe(1) @@ -227,7 +228,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => const { transport, onError } = await attachStalePane() const handleEvents = await import('../../runtime/web-session-terminal-handle-events') - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const listCallsAtCutoff = hostListCalls @@ -277,7 +278,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => const { transport, onError } = await attachStalePane() const handleEvents = await import('../../runtime/web-session-terminal-handle-events') - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') expect(handleEvents.getWebSessionTerminalHandleSubscriberCountForTests()).toBe(1) const listCallsAtCutoff = hostListCalls @@ -310,7 +311,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => const { retryAllRemoteRuntimePtyRecoveriesNow } = await import('./remote-runtime-pty-recovery-state') - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const listCallsAtCutoff = hostListCalls @@ -365,7 +366,7 @@ describe('remote runtime pty reattach after the bounded recovery window', () => callbacks: { onError: vi.fn() } }) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const callsBeforeRetry = runtimeCall.mock.calls.length @@ -378,4 +379,60 @@ describe('remote runtime pty reattach after the bounded recovery window', () => vi.useRealTimers() } }) + + // #12683: a fatal resubscribe latches the banner via markDisconnected(), but that latch is not + // evidence the auto-recovery window ran out, so it must not license reattaching a fenced handle. + it('does not reattach a fenced same handle when only a UI latch closed the window', async () => { + vi.useFakeTimers() + try { + const { createRemoteRuntimePtyTransport } = await import('./remote-runtime-pty-transport') + const handleEvents = await import('../../runtime/web-session-terminal-handle-events') + const transport = createRemoteRuntimePtyTransport('env-1', { + worktreeId: 'wt-1', + tabId: 'web-terminal-tab-1', + leafId: 'pane:1', + onPtyExit: vi.fn(), + onPtyRebind: vi.fn() + }) + transport.attach({ + existingPtyId: 'remote:env-1@@terminal-stale', + cols: 80, + rows: 24, + callbacks: { onError: vi.fn() } + }) + await vi.waitFor(() => expect(subscriptionSendBinary).toHaveBeenCalled()) + emitSnapshot(latestSubscribePayload().streamId, 'live before the drop') + + // The host keeps publishing the same handle, so no replacement can ever arrive. + runtimeCall.mockImplementation(async (args: { method: string }) => { + if (args.method !== 'session.tabs.list') { + return { ok: true, result: {} } + } + hostListCalls += 1 + return { ok: true, result: hostSnapshot('terminal-stale', hostListCalls + 1, 'epoch-1') } + }) + // The stream drops and the resubscribe fails fatally with a stale handle: markDisconnected() + // latches the banner, then stale routing fences the handle with require-replacement. + // Only that one attempt fails, so a later reattach would succeed and be observable. + runtimeSubscribe.mockImplementationOnce(async () => { + throw new Error('terminal_handle_stale') + }) + subscriptionCallbacks?.onClose?.() + await vi.advanceTimersByTimeAsync(16_000) + expect(transport.getRecoveryState?.().phase).not.toBe('idle') + + const subscribesBeforeRepublish = subscribedTerminalHandles().length + handleEvents.queueAcceptedWebSessionTerminalSnapshot( + hostSnapshot('terminal-stale', 9, 'epoch-2'), + 'env-1' + ) + await vi.advanceTimersByTimeAsync(1_000) + + // The recovery deadline never fired, so the fenced handle stays fenced. + expect(subscribedTerminalHandles()).toHaveLength(subscribesBeforeRepublish) + transport.destroy?.() + } finally { + vi.useRealTimers() + } + }) }) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts index 5892ee2c19d..3c1ec0421a9 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-latched-pane-retention.test.ts @@ -5,6 +5,7 @@ import { decodeTerminalStreamJson } from '../../../../shared/terminal-stream-protocol' import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' // Why: the recovery cutoff no longer tears down the retry registry entry or the accepted-snapshot // listener, so those two module-global collections are the only places a latched pane can accumulate. @@ -184,7 +185,7 @@ describe('remote runtime pty latched-pane retention', () => { for (let cycle = 0; cycle < 20; cycle += 1) { const transport = await attachStalePane(cycle) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') latched.push(await registries()) transport.destroy?.() @@ -208,7 +209,7 @@ describe('remote runtime pty latched-pane retention', () => { const settled: { subscribers: number; scheduled: number }[] = [] for (let cycle = 0; cycle < 20; cycle += 1) { const transport = await attachStalePane(cycle) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') transport.detach?.() await vi.advanceTimersByTimeAsync(1_000) @@ -227,7 +228,7 @@ describe('remote runtime pty latched-pane retention', () => { const transports: Awaited>[] = [] for (let pane = 0; pane < 8; pane += 1) { transports.push(await attachStalePane(pane)) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) } // Retention is per live pane, not per timeout: eight latched panes hold eight of each. expect(await registries()).toEqual({ subscribers: 8, scheduled: 8 }) @@ -247,7 +248,7 @@ describe('remote runtime pty latched-pane retention', () => { vi.useFakeTimers() try { const transport = await attachStalePane(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const baseline = await registries() @@ -277,7 +278,7 @@ describe('remote runtime pty latched-pane retention', () => { const { retryAllRemoteRuntimePtyRecoveriesNow } = await import('./remote-runtime-pty-recovery-state') const transport = await attachStalePane(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const baseline = await registries() @@ -295,7 +296,7 @@ describe('remote runtime pty latched-pane retention', () => { // A second trigger in the same window must find nothing to advance, so an online/resume // storm cannot stack fresh recovery epochs on one pane. expect(retryAllRemoteRuntimePtyRecoveriesNow()).toBe(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') observed.push({ ...(await registries()), timers: vi.getTimerCount(), revived }) } @@ -325,7 +326,7 @@ describe('remote runtime pty latched-pane retention', () => { try { const handleEvents = await import('../../runtime/web-session-terminal-handle-events') const transport = await attachStalePane(0) - await vi.advanceTimersByTimeAsync(66_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + 6_000) expect(transport.getRecoveryState?.().phase).toBe('disconnected') const baseline = await registries() diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts index d71af0b9afb..c245c184436 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.test.ts @@ -1,6 +1,8 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS, + REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS, + REMOTE_RUNTIME_RECOVERY_DELAYS_MS, RemoteRuntimePtyRecoveryState, retryAllRemoteRuntimePtyRecoveriesNow } from './remote-runtime-pty-recovery-state' @@ -63,7 +65,7 @@ describe('RemoteRuntimePtyRecoveryState', () => { const epoch = state.begin() state.schedule(epoch, retry) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(state.currentPhase).toBe('disconnected') expect(state.isActive).toBe(false) @@ -112,7 +114,7 @@ describe('RemoteRuntimePtyRecoveryState', () => { const state = new RemoteRuntimePtyRecoveryState() const firstEpoch = state.begin() - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const manualEpoch = state.begin() expect(manualEpoch).toBe(firstEpoch + 1) @@ -227,4 +229,68 @@ describe('RemoteRuntimePtyRecoveryState', () => { expect(retryAllRemoteRuntimePtyRecoveriesNow()).toBe(0) state.dispose() }) + + // #11305: the schedule and the deadline lived as two independent literals and drifted apart. + it('keeps the backoff schedule inside the auto-recovery budget it arms', () => { + const scheduleSumMs = REMOTE_RUNTIME_RECOVERY_DELAYS_MS.reduce( + (total, delayMs) => total + delayMs, + 0 + ) + + expect(scheduleSumMs).toBeLessThanOrEqual(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + // Every step also needs room for the attempt it leads into, or the tail is dead code + // whenever a half-open link makes each attempt burn its full RPC timeout. + expect( + scheduleSumMs + + REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length * REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS + ).toBeLessThanOrEqual(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + }) + + it('reaches every backoff step when each attempt burns a full RPC timeout', async () => { + vi.useFakeTimers() + const state = new RemoteRuntimePtyRecoveryState() + const attemptStartsMs: number[] = [] + const epoch = state.begin() + const startedAt = Date.now() + + const failSlowly = (currentEpoch: number): void => { + attemptStartsMs.push(Date.now() - startedAt) + // Silent-drop reconnects do not fail instantly; they time out. + setTimeout(() => { + if (state.isCurrent(currentEpoch)) { + state.schedule(currentEpoch, failSlowly) + } + }, REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS) + } + state.schedule(epoch, failSlowly) + + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + + expect(attemptStartsMs.length).toBeGreaterThanOrEqual(REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length) + expect(state.currentPhase).toBe('disconnected') + state.dispose() + }) + + // #12683: markDisconnected() is a UI latch, not proof the window ran out. + it('only reports the auto-recovery window spent when the deadline actually fired', async () => { + vi.useFakeTimers() + const state = new RemoteRuntimePtyRecoveryState() + const epoch = state.begin() + state.schedule(epoch, vi.fn()) + + state.markDisconnected() + expect(state.currentPhase).toBe('disconnected') + expect(state.autoRecoveryDeadlineExpired).toBe(false) + + const secondEpoch = state.begin() + state.schedule(secondEpoch, vi.fn()) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) + + expect(state.currentPhase).toBe('disconnected') + expect(state.autoRecoveryDeadlineExpired).toBe(true) + + state.markHealthy() + expect(state.autoRecoveryDeadlineExpired).toBe(false) + state.dispose() + }) }) diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts index fbdd50e443f..002e1320398 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-recovery-state.ts @@ -1,5 +1,17 @@ -const RECOVERY_DELAYS_MS = [250, 500, 1000, 2000, 4000, 8000, 15_000, 30_000] as const -export const REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS = 60_000 +export const REMOTE_RUNTIME_RECOVERY_DELAYS_MS = [ + 250, 500, 1000, 2000, 4000, 8000, 15_000, 30_000 +] as const + +// Why: mirrors DEFAULT_REMOTE_RUNTIME_TIMEOUT_MS in the main-process runtime router; a silently +// dropped link burns the whole RPC timeout on the attempt each backoff step leads into. +export const REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS = 15_000 + +// Why derived, not hand-tuned: a literal deadline drifted below the ladder it arms, making the last +// backoff steps unreachable dead code (#11305). Loss of contact is never evidence of exit, so the +// window must outlast the schedule it advertises rather than the schedule being trimmed to fit. +export const REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS = + REMOTE_RUNTIME_RECOVERY_DELAYS_MS.reduce((total, delayMs) => total + delayMs, 0) + + REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length * REMOTE_RUNTIME_RECOVERY_ATTEMPT_BUDGET_MS export type RemoteRuntimePtyRecoveryPhase = | 'idle' @@ -34,6 +46,9 @@ export class RemoteRuntimePtyRecoveryState { private deadlineTimer: ReturnType | null = null private pendingRetry: ((epoch: number) => void) | null = null private pendingEpoch: number | null = null + // Why: only the wall-clock deadline proves the auto-recovery window was actually spent; a UI latch + // via markDisconnected() must not forge that evidence (#12683). + private deadlineExpired = false constructor(private readonly onChange?: () => void) {} @@ -53,6 +68,10 @@ export class RemoteRuntimePtyRecoveryState { return this.attempt } + get autoRecoveryDeadlineExpired(): boolean { + return this.deadlineExpired + } + begin(): number { if (this.phase === 'disposed') { return this.epoch @@ -82,7 +101,10 @@ export class RemoteRuntimePtyRecoveryState { } this.clearRetryTimer() this.phase = 'backoff' - const delayMs = RECOVERY_DELAYS_MS[Math.min(this.attempt, RECOVERY_DELAYS_MS.length - 1)] + const delayMs = + REMOTE_RUNTIME_RECOVERY_DELAYS_MS[ + Math.min(this.attempt, REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length - 1) + ] this.attempt += 1 this.pendingRetry = retry this.pendingEpoch = epoch @@ -107,11 +129,22 @@ export class RemoteRuntimePtyRecoveryState { // Why: a wait that ends with no liveness evidence arms no timer, so park a retry or online/resume/reconnect find nothing to revive. parkRetryForExternalTrigger(epoch: number, retry: (epoch: number) => void): boolean { - if (!this.isCurrent(epoch) || this.pendingRetry !== null) { + return this.isCurrent(epoch) && this.parkRetry(retry) + } + + // Why: the deadline can latch while an attempt is still in flight, before schedule() parked anything, + // so the late failure has no live epoch to join and must not begin a new one — that would re-arm a + // full-length window and the budget would never actually expire. + parkRetryAfterDeadline(retry: (epoch: number) => void): boolean { + return this.phase === 'disconnected' && this.parkRetry(retry) + } + + private parkRetry(retry: (epoch: number) => void): boolean { + if (this.pendingRetry !== null) { return false } this.pendingRetry = retry - this.pendingEpoch = epoch + this.pendingEpoch = this.epoch scheduledRecoveries.add(this) return true } @@ -152,6 +185,7 @@ export class RemoteRuntimePtyRecoveryState { if (this.phase === 'disposed') { return } + this.deadlineExpired = false this.clearTimers() this.phase = 'idle' this.attempt = 0 @@ -175,6 +209,7 @@ export class RemoteRuntimePtyRecoveryState { if (this.phase === 'disposed') { return } + this.deadlineExpired = false this.epoch += 1 this.clearTimers() this.phase = 'idle' @@ -191,11 +226,13 @@ export class RemoteRuntimePtyRecoveryState { private armDeadline(epoch: number): void { this.clearDeadlineTimer() + this.deadlineExpired = false const timer = setTimeout(() => { if (this.deadlineTimer !== timer || !this.isCurrent(epoch)) { return } this.deadlineTimer = null + this.deadlineExpired = true // Why: the cutoff stops self-initiated retries but must keep the pane revivable by online/resume/reconnect. this.stopRetryTimer() this.phase = 'disconnected' diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts index b34ec4f3fea..59dc3612ed1 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-create-outcome-recovery.test.ts @@ -4,6 +4,7 @@ import { createRemoteRuntimeTransportMocks, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -81,7 +82,7 @@ describe('createRemoteRuntimePtyTransport', () => { let createCalls = 0 runtimeCall.mockImplementation(async (args: { method: string }) => { if (args.method === 'status.get') { - vi.setSystemTime(startedAt + 59_000) + vi.setSystemTime(startedAt + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS - 1_000) return { ok: true, result: { capabilities: [TERMINAL_CREATE_IDEMPOTENCY_RUNTIME_CAPABILITY] } @@ -163,7 +164,7 @@ describe('createRemoteRuntimePtyTransport', () => { transport.destroy?.() }) - it('stops unknown terminal-create recovery after one minute and remains manually retryable', async () => { + it('stops unknown terminal-create recovery at the cutoff and remains manually retryable', async () => { vi.useFakeTimers() try { let reachable = false @@ -205,7 +206,7 @@ describe('createRemoteRuntimePtyTransport', () => { onRecoveryStateChange: (state) => recoveryStates.push(state.phase) } }) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) await connect const callsAtCutoff = runtimeCall.mock.calls.length @@ -218,7 +219,7 @@ describe('createRemoteRuntimePtyTransport', () => { statusTimesOut = true expect(transport.retryRecovery?.()).toBe(true) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const callsAtManualCutoff = runtimeCall.mock.calls.length expect(transport.getRecoveryState?.().phase).toBe('disconnected') await vi.advanceTimersByTimeAsync(5 * 60_000) @@ -278,7 +279,7 @@ describe('createRemoteRuntimePtyTransport', () => { }) const connect = transport.connect({ url: '', callbacks: {} }) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) await connect expect(transport.getRecoveryState?.().phase).toBe('disconnected') diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts index 8817d5c23e4..53c1921fe74 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stale-handle-recovery.test.ts @@ -4,6 +4,7 @@ import { readyHostSessionInventoryResponse, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -72,7 +73,7 @@ describe('createRemoteRuntimePtyTransport', () => { expect(hostListCalls).toBe(callsAfterTwoWindows) expect(transport.getRecoveryState?.().phase).toBe('recovering') - await vi.advanceTimersByTimeAsync(9_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') expect(subscribedTerminalHandles()).toEqual(['terminal-stable']) transport.destroy?.() @@ -180,7 +181,7 @@ describe('createRemoteRuntimePtyTransport', () => { 'terminal-flapping' ]) expect(transport.isConnected()).toBe(false) - await vi.advanceTimersByTimeAsync(45_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') transport.destroy?.() } finally { diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts index 48a24a2c084..5a61a5587e6 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-sticky-replacement-policy.test.ts @@ -9,6 +9,7 @@ import { readyHostSessionInventoryResponse, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -160,7 +161,7 @@ describe('createRemoteRuntimePtyTransport', () => { expect(transport.isConnected()).toBe(false) expect(onPtyExit).not.toHaveBeenCalled() - await vi.advanceTimersByTimeAsync(44_001) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') expect(subscribedTerminalHandles()).toHaveLength(3) } finally { diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts index 4ef1075e3f7..e0b85414e3f 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-stream-reconnect.test.ts @@ -9,6 +9,7 @@ import { createRemoteRuntimeTransportMocks, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS } from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -534,7 +535,7 @@ describe('createRemoteRuntimePtyTransport', () => { partitioned = true callbacksByConnection[0].onClose?.() - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const disconnectedState = transport.getRecoveryState?.() const callsAtCutoff = runtimeSubscribe.mock.calls.length diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts index f8c07ea045d..784ed9d8768 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport-web-mirror-recovery.test.ts @@ -3,6 +3,10 @@ import { createRemoteRuntimeTransportMocks, type MultiplexSubscriptionCallbacks } from './remote-runtime-pty-transport-test-harness' +import { + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS, + REMOTE_RUNTIME_RECOVERY_DELAYS_MS +} from './remote-runtime-pty-recovery-state' let subscriptionCallbacks: MultiplexSubscriptionCallbacks = null let resolvedPaneHandle = 'terminal-1' @@ -140,10 +144,11 @@ describe('createRemoteRuntimePtyTransport', () => { existingPtyId: 'remote:env-1@@stale-client-handle', callbacks: {} }) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) const attemptsAtCutoff = runtimeSubscribe.mock.calls.length - expect(attemptsAtCutoff).toBe(8) + // Why: every backoff step must be reachable inside the window it arms (#11305). + expect(attemptsAtCutoff).toBeGreaterThanOrEqual(REMOTE_RUNTIME_RECOVERY_DELAYS_MS.length) expect(transport.getRecoveryState?.().phase).toBe('disconnected') await vi.advanceTimersByTimeAsync(5 * 60_000) @@ -235,7 +240,7 @@ describe('createRemoteRuntimePtyTransport', () => { await vi.advanceTimersByTimeAsync(250) expect(activateAttempts).toBe(2) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') rejectInFlight( @@ -292,7 +297,7 @@ describe('createRemoteRuntimePtyTransport', () => { await vi.advanceTimersByTimeAsync(250) expect(runtimeSubscribe).toHaveBeenCalledTimes(1) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') rejectSubscription( @@ -349,7 +354,7 @@ describe('createRemoteRuntimePtyTransport', () => { expect.objectContaining({ method: 'terminal.resolvePane' }) ) - await vi.advanceTimersByTimeAsync(60_000) + await vi.advanceTimersByTimeAsync(REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS) expect(transport.getRecoveryState?.().phase).toBe('disconnected') resolveMetadata({ diff --git a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts index 1e5782fe496..b1bd7336c86 100644 --- a/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts +++ b/src/renderer/src/components/terminal-pane/remote-runtime-pty-transport.ts @@ -90,6 +90,10 @@ const HOST_SESSION_POLL_MAX_MS = 1_000 const HOST_SESSION_ATTACH_TIMEOUT_MS = 15_000 const HOST_SESSION_INVENTORY_MAX_WINDOWS_PER_RECOVERY = 2 const HOST_SESSION_SAME_HANDLE_END_REUSE_LIMIT = 2 +// Why its own constant: this fences how long an end-then-reattach on the same handle still counts as +// one recovery, which is unrelated to how long auto-recovery keeps retrying. It read the recovery +// budget before that budget became a derived value, and must not drift with it. +const HOST_SESSION_SAME_HANDLE_END_REUSE_WINDOW_MS = 60_000 const MAX_SURFACED_TERMINAL_ERRORS = 8 const TERMINAL_CREATE_RETRY_DELAYS_MS = [250, 500, 1000, 2000, 4000, 8000, 15_000, 30_000] as const @@ -247,7 +251,9 @@ export function createRemoteRuntimePtyTransport( clearPublishedHandleWait() } if (recovery.currentPhase === 'disconnected') { - autoRecoveryWindowSpent = true + // Why: only the wall-clock deadline is evidence the window was spent; a UI latch from a fatal + // resubscribe must not license reattaching a fenced same handle (#12683). + autoRecoveryWindowSpent ||= recovery.autoRecoveryDeadlineExpired // Why: cached pixels may remain, but no stream from the exhausted epoch may keep delivering or accepting terminal traffic. subscriptionGeneration += 1 closeMultiplexedStream() @@ -326,7 +332,7 @@ export function createRemoteRuntimePtyTransport( if ( sameHandleEndReuseHandle !== targetHandle || sameHandleEndReuseAttachedAt === null || - Date.now() - sameHandleEndReuseAttachedAt >= REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS + Date.now() - sameHandleEndReuseAttachedAt >= HOST_SESSION_SAME_HANDLE_END_REUSE_WINDOW_MS ) { resetSameHandleEndReuse() return 'prefer-replacement' @@ -870,6 +876,43 @@ export function createRemoteRuntimePtyTransport( return false } + // Why: a recoverable connect failure is unverifiable contact loss, not a dead terminal, so retry + // whichever path can still reach the pane instead of latching with nothing armed (#12684). + function retryAfterRecoverableConnectFailure(nextEpoch: number): void { + if (destroyed || terminalEnded) { + return + } + if (connected && handle) { + scheduleResubscribeAfterTransportClose(getRecoveryReplacementPolicy(handle), nextEpoch) + return + } + replayLastTransportEntryPoint() + } + + // Why: schedule() both auto-retries inside the window and leaves the retry parked when the deadline + // latches, so online/resume and the Reconnect button always find something to fire. + function scheduleConnectRetryAfterRecoverableFailure(): void { + if (destroyed) { + return + } + // Why: an ambiguous create already owns a reconciliation-gated retry that only Reconnect may + // re-enter; auto-replaying here would just re-probe a runtime that cannot reconcile. + if (terminalCreateNeedsReconciliation || agentSessionRequiresHostAuthorityReplay) { + recovery.markDisconnected() + return + } + // Why: the last attempt's RPC budget expires at the same instant as the deadline, so a silent drop + // rejects after the latch. Beginning a new epoch there re-arms the whole window, so park instead. + if (recovery.currentPhase === 'disconnected') { + recovery.parkRetryAfterDeadline(retryAfterRecoverableConnectFailure) + return + } + const recoveryEpoch = recovery.isActive ? recovery.currentEpoch : recovery.begin() + if (!recovery.schedule(recoveryEpoch, retryAfterRecoverableConnectFailure)) { + recovery.markDisconnected() + } + } + async function attachHostSessionMirror( options: { cols?: number; rows?: number }, notifySpawn = true, @@ -1033,6 +1076,8 @@ export function createRemoteRuntimePtyTransport( kind === 'agent-session' ? agentSessionRequiresHostAuthorityReplay : terminalCreateNeedsReconciliation + // Why the same budget: this loop calls recovery.begin(), so a shorter local deadline would abandon + // the create while the recovery state still reports 'recovering' with nothing in flight. let recoveryDeadlineAt: number | null = recovery.isActive ? Date.now() + REMOTE_RUNTIME_AUTO_RECOVERY_TIMEOUT_MS : null @@ -2314,7 +2359,7 @@ export function createRemoteRuntimePtyTransport( } else if ( isRecoverableRemoteRuntimeConnectionError(toRemoteRuntimeClientErrorLike(error)) ) { - recovery.markDisconnected() + scheduleConnectRetryAfterRecoverableFailure() } else { recovery.cancel() emitRecoveryState() @@ -2616,6 +2661,12 @@ export function createRemoteRuntimePtyTransport( void transport.connect(lastConnectOptions) return true } + // Why: online/resume fires a parked retry; the button must not be weaker than an event (#12684). + if (!destroyed && !terminalEnded && recovery.currentPhase === 'disconnected') { + if (recovery.retryNow()) { + return true + } + } if ( destroyed || terminalEnded || diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index f4a4fd931de..0ac5d624ede 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -2881,7 +2881,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "Reconnecting to remote runtime", "disconnectedTitle": "Remote runtime disconnected", - "retryingBody": "Orca will retry for up to one minute. This terminal will resume if the connection returns.", + "retryingBody": "Orca is retrying automatically. This terminal will resume if the connection returns.", "disconnectedBody": "Automatic retries stopped. Reconnect to resume this terminal session.", "reconnectButton": "Reconnect" } diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index ee6e4440f79..d3af25ad697 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -2881,7 +2881,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "リモートランタイムに再接続中", "disconnectedTitle": "リモートランタイムが切断されました", - "retryingBody": "Orca は最大1分間再試行します。接続が復元されると、このターミナルが再開されます。", + "retryingBody": "Orca は自動的に再試行します。接続が復元されると、このターミナルが再開されます。", "disconnectedBody": "自動再試行が停止しました。このターミナルセッションを再開するには再接続してください。", "reconnectButton": "再接続" } diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 76ed151ca2c..47d9e48fd26 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -2886,7 +2886,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "원격 런타임에 다시 연결 중", "disconnectedTitle": "원격 런타임 연결 끊김", - "retryingBody": "Orca는 최대 1분간 재시도합니다. 연결이 복원되면 이 터미널이 재개됩니다.", + "retryingBody": "Orca가 자동으로 재시도합니다. 연결이 복원되면 이 터미널이 재개됩니다.", "disconnectedBody": "자동 재시도가 중지되었습니다. 이 터미널 세션을 재개하려면 다시 연결하세요.", "reconnectButton": "다시 연결" } diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 5c1324afa12..b097b789571 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -2896,7 +2896,7 @@ "TerminalRemoteRuntimeReconnectBanner": { "retryingTitle": "正在重新连接到远程运行时", "disconnectedTitle": "远程运行时已断开连接", - "retryingBody": "Orca 将重试最多一分钟。如果连接恢复,此终端将恢复。", + "retryingBody": "Orca 正在自动重试。如果连接恢复,此终端将恢复。", "disconnectedBody": "自动重试已停止。重新连接以恢复此终端会话。", "reconnectButton": "重新连接" } From 39330c5acab3529af627f6810f68575e3dc65a17 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:44 -0700 Subject: [PATCH 117/398] fix(relay): retire PTYs the host proves are gone, and stop two per-poll scan storms (#17832) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(relay): stop three CPU growth terms in a long-running remote session pty.resize gated only on `managed.disposed`, which is bookkeeping rather than liveness. A shell that exits without node-pty's `onExit` leaves an undisposed entry holding a closed master fd, and UnixTerminal.resize has no fd guard, so the ioctl threw `ioctl(2) failed, EBADF` into the dispatcher's generic parse-error catch. Nothing retired the entry, so it stayed advertised and kept activePtyCount above zero -- which is what stops a relay with an unlimited grace from reaching its idle-no-ptys exit (#12423). Probe liveness with the same helper attach/listProcesses use, retire a provably dead pid, and contain an ioctl failure over a live-or-unverifiable process. processHasChildren forked `pgrep -P` per pane per inspection poll, uncached. procps-ng opens six procfs files per process to resolve one ppid, so each call cost O(host process count). Answer from the TTL-cached `ps` table the same RPC already captured for the foreground lookup (#13537). The remote AI Vault scanner had no parse cache at all, so every forced rescan re-read and re-parsed the whole transcript corpus, including files untouched for a month. Give it the mtime+size keyed memo the local scanner has (#13753). * fix(pty): invalidate the descriptor when node-pty gives up the handle (#17930) Carried forward from PR #17930, which merged into this branch. Rebased onto current main; main's newer node-pty-fd-leak test is kept as-is. * fix(ai-vault): refresh codex titles on the remote parse-cache reuse path The remote cache keys on the transcript's (mtime, size, host), but codex titles live in $CODEX_HOME/session_index.jsonl and are written after the rollout — so a cache hit froze the fallback title forever. Mirrors the local scanner's existing reuse-path refresh via a shared core. * fix(relay): publish the exit a reap performs, and rescan for close decisions Two review findings on the CPU work. reapExitedPty told only the relay-internal exit listener, so a retirement left the client's pane mounted against a session the relay had already forgotten -- the next attach answered `PTY "" not found` with nothing before it to explain why. Pre-existing on three probe paths; resize made it user-triggered. Publish the same pending-exit the natural onExit path publishes, carrying -1 ("gone, status unrecoverable"), and skip it when onExit already reported the real code. processHasChildren now answers from a 500ms TTL-cached table. That is right for pty.inspectProcess, which every tracked pane polls, but pty.hasChildProcesses gates the window-close confirmation and workspace cleanup's idle evidence -- one destructive decision per answer, where a child started inside the window would be killed unasked. Give that RPC a fresh scan; pgrep used to. * fix(relay): publish a reap's exit only on proven-exited evidence The publication is a verdict the client acts on by retiring the pane, so it must not be reachable from the disposed-record sweep, which retires off our own bookkeeping rather than the host's process table. Only ESRCH earns it. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- config/patches/node-pty@1.1.0.patch | 78 ++++++- pnpm-lock.yaml | 6 +- .../remote-session-parse-cache.test.ts | 216 ++++++++++++++++++ .../ai-vault/remote-session-parse-cache.ts | 107 +++++++++ .../remote-session-scanner-codex-index.ts | 23 +- .../remote-session-scanner-sources.ts | 13 +- src/main/ai-vault/remote-session-scanner.ts | 43 +++- .../session-scanner-codex-cached-title.ts | 27 ++- .../ai-vault/session-scanner-parse-cache.ts | 2 + .../pty/node-pty-master-fd-retirement.test.ts | 194 ++++++++++++++++ .../pty-handler-resize-stale-pty.test.ts | 176 ++++++++++++++ src/relay/pty-handler-spawn-admission.test.ts | 15 ++ src/relay/pty-handler.ts | 117 +++++++++- src/relay/pty-shell-utils.test.ts | 85 +++++++ src/relay/pty-shell-utils.ts | 37 ++- 15 files changed, 1092 insertions(+), 47 deletions(-) create mode 100644 src/main/ai-vault/remote-session-parse-cache.test.ts create mode 100644 src/main/ai-vault/remote-session-parse-cache.ts create mode 100644 src/main/pty/node-pty-master-fd-retirement.test.ts create mode 100644 src/relay/pty-handler-resize-stale-pty.test.ts diff --git a/config/patches/node-pty@1.1.0.patch b/config/patches/node-pty@1.1.0.patch index 348ce6ef7ce..ff474f7d95e 100644 --- a/config/patches/node-pty@1.1.0.patch +++ b/config/patches/node-pty@1.1.0.patch @@ -138,8 +138,34 @@ index 8c4fca9022a6d6f015bca87f61625cde2278f428..0a01730616488119aa21ef441cf3c441 process.exit(0); //# sourceMappingURL=conpty_console_list_agent.js.map \ No newline at end of file +diff --git a/lib/terminal.js b/lib/terminal.js +index e2f9bc9131077b53ebc32d207207ad82804ff185..6c63bfaaf75128d88f9a2efece13476348780cfd 100644 +--- a/lib/terminal.js ++++ b/lib/terminal.js +@@ -172,6 +172,21 @@ var Terminal = /** @class */ (function () { + this.end = function () { }; + this._writable = false; + this._readable = false; ++ // Orca: libuv closes the master fd on EIO/EOF, and the kernel may hand ++ // that number straight to the next open(2). Retire it in the same block ++ // that gives up the handle so no later ioctl can address a reused fd. ++ // Inert on Windows, where `_fd` is written once and never read back. ++ // Upstream named this mechanism in microsoft/node-pty#220 ("fd number got ++ // reattached to something else"), closed 2025-12-19 as completed after ++ // only improving the error message; #827 is still open. Windows guards in ++ // windowsPtyAgent.ts, Unix does not. Orca tracking: #18109. ++ this._fd = -1; ++ // Orca: the write stream holds its own copy of that number, so retiring ++ // `_fd` alone leaves the queued and in-flight writes addressing it. ++ // Undefined on Windows and on `UnixTerminal.open()` handles. ++ if (this._writeStream) { ++ this._writeStream.dispose(); ++ } + }; + Terminal.prototype._parseEnv = function (env) { + var keys = Object.keys(env || {}); diff --git a/lib/unixTerminal.js b/lib/unixTerminal.js -index 1ec12f796a822c78fba9ad7f6448c3987e325c23..cec8b67aef02f8199e5606a0d257088bf1865877 100644 +index 1ec12f796a822c78fba9ad7f6448c3987e325c23..d838d795ecb9ea72e3bcc31113344947c006af7e 100644 --- a/lib/unixTerminal.js +++ b/lib/unixTerminal.js @@ -28,8 +28,12 @@ var native = utils_1.loadNativeModule('pty'); @@ -157,6 +183,56 @@ index 1ec12f796a822c78fba9ad7f6448c3987e325c23..cec8b67aef02f8199e5606a0d257088b var DEFAULT_FILE = 'sh'; var DEFAULT_NAME = 'xterm'; var DESTROY_SOCKET_TIMEOUT_MS = 200; +@@ -234,6 +238,11 @@ var UnixTerminal = /** @class */ (function (_super) { + * Gets the name of the process. + */ + get: function () { ++ // Orca: tcgetpgrp on a retired fd would name whatever process now ++ // owns that descriptor, so a closed master reports the spawn file. ++ if (this._fd < 0) { ++ return this._file; ++ } + if (process.platform === 'darwin') { + var title = pty.process(this._fd); + return (title !== 'kernel_task') ? title : this._file; +@@ -250,6 +259,11 @@ var UnixTerminal = /** @class */ (function (_super) { + if (cols <= 0 || rows <= 0 || isNaN(cols) || isNaN(rows) || cols === Infinity || rows === Infinity) { + throw new Error('resizing must be done using positive cols and rows'); + } ++ // Orca: a retired master is unreachable rather than EBADF-or-worse; cols ++ // and rows stay at the last size actually applied instead of a claim. ++ if (this._fd < 0) { ++ return; ++ } + pty.resize(this._fd, cols, rows); + this._cols = cols; + this._rows = rows; +@@ -287,8 +301,15 @@ var CustomWriteStream = /** @class */ (function () { + CustomWriteStream.prototype.dispose = function () { + clearImmediate(this._writeImmediate); + this._writeImmediate = undefined; ++ // Orca: retire this stream's own copy of the master fd and drop what has ++ // not shipped, so nothing queued here reaches a reused descriptor. ++ this._fd = -1; ++ this._writeQueue.length = 0; + }; + CustomWriteStream.prototype.write = function (data) { ++ if (this._fd < 0) { ++ return; ++ } + // Writes are put in a queue and processed asynchronously in order to handle + // backpressure from the kernel buffer. + var buffer = typeof data === 'string' +@@ -304,7 +325,8 @@ var CustomWriteStream = /** @class */ (function () { + CustomWriteStream.prototype._processWriteQueue = function () { + var _this = this; + this._writeImmediate = undefined; +- if (this._writeQueue.length === 0) { ++ // Orca: an in-flight fs.write can re-enter here after dispose(). ++ if (this._fd < 0 || this._writeQueue.length === 0) { + return; + } + var task = this._writeQueue[0]; diff --git a/src/conpty_console_list_agent.ts b/src/conpty_console_list_agent.ts index 181ccabbbe9c4948a9725fb1db907a68e9de01fc..67f31facf85562b67adbfbd04ce28ddd8eeb4a79 100644 --- a/src/conpty_console_list_agent.ts diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 1abec6c0590..eac56401344 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -115,7 +115,7 @@ patchedDependencies: '@xterm/addon-webgl@0.20.0-beta.299': 94687e89a0115e6e6aa102837f986debdc029c091527ee5eb4a4e17ceaf9473e '@xterm/xterm@6.1.0-beta.303': 98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d lint-staged@16.4.0: 7333b3837f80a7fbd045964db6d76ba4fc118e49134bdbabb00585b6b7b60673 - node-pty@1.1.0: 40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e + node-pty@1.1.0: e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa importers: @@ -156,7 +156,7 @@ importers: version: 3.3.1 node-pty: specifier: ^1.1.0 - version: 1.1.0(patch_hash=40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e) + version: 1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa) posthog-node: specifier: ^5.33.3 version: 5.33.3 @@ -12194,7 +12194,7 @@ snapshots: node-int64@0.4.0: {} - node-pty@1.1.0(patch_hash=40b6b6b814c89a8a29c995495ea86cff2cf6766f42124b5702aedf4ce0565c0e): + node-pty@1.1.0(patch_hash=e262847f57a1d4d3f2287a843822f7dcf3c9d8655892b07a69eba464e1317eaa): dependencies: node-addon-api: 7.1.1 diff --git a/src/main/ai-vault/remote-session-parse-cache.test.ts b/src/main/ai-vault/remote-session-parse-cache.test.ts new file mode 100644 index 00000000000..446563c0652 --- /dev/null +++ b/src/main/ai-vault/remote-session-parse-cache.test.ts @@ -0,0 +1,216 @@ +import { beforeEach, describe, expect, it } from 'vitest' +import type { FileReadResult } from '../providers/types' +import { getRemoteHostPlatform } from '../ssh/ssh-remote-platform' +import { resetRemoteSessionParseCacheForTests } from './remote-session-parse-cache' +import { scanRemoteAiVaultSessions } from './remote-session-scanner' +import { MemoryRemoteProvider, jsonLines } from './remote-session-scanner-test-fixtures' + +/** + * Counts whole-transcript reads, which is the cost #13753 is about. Codex's + * per-scan `session_index.jsonl` title lookup is one small file and is not part + * of the corpus term, so it is excluded rather than asserted on. + */ +class CountingRemoteProvider extends MemoryRemoteProvider { + readonly readFilePaths: string[] = [] + + override async readFile(filePath: string): Promise { + if (filePath.includes('/sessions/')) { + this.readFilePaths.push(filePath) + } + return await super.readFile(filePath) + } +} + +function transcript(sessionId: string, title: string, timestamp: string): string { + return jsonLines([ + { + timestamp, + type: 'session_meta', + payload: { id: sessionId, cwd: '/home/ada/repo' } + }, + { + timestamp, + type: 'response_item', + payload: { type: 'message', role: 'user', content: [{ type: 'text', text: title }] } + } + ]) +} + +function scan(provider: CountingRemoteProvider): ReturnType { + return scanRemoteAiVaultSessions({ + provider, + executionHostId: 'ssh:dev-box', + remoteHome: '/home/ada', + hostPlatform: getRemoteHostPlatform('linux-x64') + }) +} + +describe('remote AI Vault transcript re-reads', () => { + beforeEach(() => { + resetRemoteSessionParseCacheForTests() + }) + + it('does not re-read an unchanged corpus on the next scan', async () => { + const provider = new CountingRemoteProvider() + for (const day of ['07/07', '07/25', '08/10']) { + provider.addFile( + `/home/ada/.codex/sessions/2026/${day}/rollout-${day.replace('/', '')}.jsonl`, + transcript( + `session-${day.replace('/', '')}`, + `Work from ${day}`, + '2026-07-07T01:00:00.000Z' + ), + 1_000 + ) + } + + const first = await scan(provider) + expect(first.sessions).toHaveLength(3) + expect(provider.readFilePaths).toHaveLength(3) + + provider.readFilePaths.length = 0 + const second = await scan(provider) + + // Historical transcripts are immutable; a second pass must cost zero reads. + expect(provider.readFilePaths).toEqual([]) + expect(second.sessions.map((session) => session.title)).toEqual( + first.sessions.map((session) => session.title) + ) + }) + + it('re-reads a transcript that actually changed', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-live.jsonl' + provider.addFile( + path, + transcript('live-session', 'First prompt', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + + await scan(provider) + provider.readFilePaths.length = 0 + + provider.addFile( + path, + transcript('live-session', 'Second prompt', '2026-08-31T02:00:00.000Z'), + 2_000 + ) + const result = await scan(provider) + + expect(provider.readFilePaths).toEqual([path]) + expect(result.sessions[0]?.title).toBe('Second prompt') + }) + + it('re-reads when only the size changed under an unchanged mtime', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-grown.jsonl' + provider.addFile(path, transcript('grown-session', 'Short', '2026-08-31T01:00:00.000Z'), 1_000) + + await scan(provider) + provider.readFilePaths.length = 0 + + provider.addFile( + path, + transcript( + 'grown-session', + 'A much longer first prompt than before', + '2026-08-31T01:00:00.000Z' + ), + 1_000 + ) + const result = await scan(provider) + + expect(provider.readFilePaths).toEqual([path]) + expect(result.sessions[0]?.title).toBe('A much longer first prompt than before') + }) + + // Codex names threads in $CODEX_HOME/session_index.jsonl asynchronously, after + // the rollout's last append — so the transcript's mtime+size never changes to + // signal it. That file sits outside `sessions/`, hence outside the read count. + it('picks up a session_index title written after the transcript was cached', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-named-later.jsonl' + provider.addFile( + path, + transcript('named-later-session', 'First prompt', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + provider.addFile( + '/home/ada/.codex/session_index.jsonl', + jsonLines([{ id: 'some-other-session', thread_name: 'Unrelated thread' }]), + 1_000 + ) + + const first = await scan(provider) + expect(first.sessions[0]?.title).toBe('First prompt') + + provider.readFilePaths.length = 0 + provider.addFile( + '/home/ada/.codex/session_index.jsonl', + jsonLines([ + { id: 'some-other-session', thread_name: 'Unrelated thread' }, + { id: 'named-later-session', thread_name: 'Named by Codex after the fact' } + ]), + 2_000 + ) + const second = await scan(provider) + + expect(second.sessions[0]?.title).toBe('Named by Codex after the fact') + // The #13753 win is preserved: the index is read, the transcript is not. + expect(provider.readFilePaths).toEqual([]) + }) + + it('does not serve a cached parse to a different execution host', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-host.jsonl' + provider.addFile( + path, + transcript('host-session', 'Host scoped', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + + await scan(provider) + provider.readFilePaths.length = 0 + + const other = await scanRemoteAiVaultSessions({ + provider, + executionHostId: 'ssh:other-box', + remoteHome: '/home/ada', + hostPlatform: getRemoteHostPlatform('linux-x64') + }) + + expect(provider.readFilePaths).toEqual([path]) + expect(other.sessions[0]?.executionHostId).toBe('ssh:other-box') + }) + + it('does not cache a read that failed', async () => { + const provider = new CountingRemoteProvider() + const path = '/home/ada/.codex/sessions/2026/08/31/rollout-flaky.jsonl' + provider.addFile( + path, + transcript('flaky-session', 'Recovered', '2026-08-31T01:00:00.000Z'), + 1_000 + ) + + let failNextRead = true + const originalReadFile = provider.readFile.bind(provider) + provider.readFile = async (filePath: string): Promise => { + if (failNextRead && filePath === path) { + failNextRead = false + provider.readFilePaths.push(filePath) + throw new Error('EIO: transient relay read failure') + } + return await originalReadFile(filePath) + } + + const failed = await scan(provider) + expect(failed.sessions).toEqual([]) + expect(failed.issues).toHaveLength(1) + + provider.readFilePaths.length = 0 + const recovered = await scan(provider) + + expect(provider.readFilePaths).toEqual([path]) + expect(recovered.sessions[0]?.title).toBe('Recovered') + }) +}) diff --git a/src/main/ai-vault/remote-session-parse-cache.ts b/src/main/ai-vault/remote-session-parse-cache.ts new file mode 100644 index 00000000000..7d8b0ebe2c7 --- /dev/null +++ b/src/main/ai-vault/remote-session-parse-cache.ts @@ -0,0 +1,107 @@ +import type { AiVaultSession } from '../../shared/ai-vault-types' +import type { RemoteScannerContext, RemoteSessionCandidate } from './remote-session-scanner-types' + +// Matches the local scanner's cap. The relay sidecar is forked with +// --max-old-space-size=384, and a retained session row is a title, a preview +// window and counters — orders of magnitude smaller than the transcript it was +// parsed from, which is what the cache stops us re-reading. +const MAX_CACHE_ENTRIES = 4096 + +type RemoteSessionParseCacheEntry = { + mtimeMs: number + sizeBytes: number | null + hostKey: string + session: AiVaultSession | null +} + +// Module scope so it outlives one scan: the sidecar is retired only after 10 +// idle minutes, so it spans many passes of a 30s cadence. +const cache = new Map() + +export type RemoteSessionParseStats = { reused: number; parsed: number } + +export function createRemoteSessionParseStats(): RemoteSessionParseStats { + return { reused: 0, parsed: 0 } +} + +export function resetRemoteSessionParseCacheForTests(): void { + cache.clear() +} + +/** Identity of the host a parse result belongs to; a result is not portable across either field. */ +export function remoteSessionParseHostKey(context: RemoteScannerContext): string { + return `${context.executionHostId}\u0000${context.hostPlatform.relayPlatform}` +} + +function storeEntry(path: string, entry: RemoteSessionParseCacheEntry): void { + cache.delete(path) + cache.set(path, entry) + if (cache.size > MAX_CACHE_ENTRIES) { + const oldest = cache.keys().next() + if (!oldest.done) { + cache.delete(oldest.value) + } + } +} + +/** + * Parse a remote transcript, reusing the previous result when the file is + * provably unchanged. + * + * Why this exists: the remote scanner had no cache of any kind, so every pass + * re-read and re-parsed the whole corpus — up to 3000 whole-file reads, GBs of + * JSONL, including July transcripts that had not changed in a month — which is + * what pegged the relay host on the renderer's 30s forced-rescan cadence + * (#13753). The local scanner has had `parseAgentSessionFileCached` for exactly + * this reason; this is its remote counterpart. + * + * `(mtimeMs, sizeBytes)` is a sound validity key here because discovery already + * folds a source's `contentDependencyPath` stat into both fields + * (remote-session-scanner-discovery.ts), so a metadata-only transcript whose + * companion file changed still looks changed. Sources whose parse reads a file + * discovery does not stat — Codex looks its title up in `session_index.jsonl` — + * are not covered by that key and pass `refreshReusedSession` to re-derive the + * uncovered part without touching the transcript. + * + * Only a completed parse is stored. A read that threw stays uncached so a + * transient filesystem failure cannot pin a wrong answer for the corpus's life. + */ +export async function parseRemoteSessionFileCached(args: { + candidate: RemoteSessionCandidate + hostKey: string + parse: () => Promise + // Applied to a reused session only; must not re-read the transcript. + refreshReusedSession?: (session: AiVaultSession) => Promise + stats?: RemoteSessionParseStats +}): Promise { + const { file } = args.candidate + const entry = cache.get(file.path) + const unchanged = + entry !== undefined && + entry.hostKey === args.hostKey && + entry.mtimeMs === file.mtimeMs && + (entry.sizeBytes === null || file.sizeBytes === undefined || entry.sizeBytes === file.sizeBytes) + if (unchanged) { + if (args.stats) { + args.stats.reused++ + } + if (entry.session && args.refreshReusedSession) { + entry.session = await args.refreshReusedSession(entry.session) + } + // Refresh recency without re-parsing so the LRU evicts cold paths first. + storeEntry(file.path, entry) + return entry.session + } + + const session = await args.parse() + if (args.stats) { + args.stats.parsed++ + } + storeEntry(file.path, { + mtimeMs: file.mtimeMs, + sizeBytes: file.sizeBytes ?? null, + hostKey: args.hostKey, + session + }) + return session +} diff --git a/src/main/ai-vault/remote-session-scanner-codex-index.ts b/src/main/ai-vault/remote-session-scanner-codex-index.ts index 8f5ec3c1bbb..89951110cdf 100644 --- a/src/main/ai-vault/remote-session-scanner-codex-index.ts +++ b/src/main/ai-vault/remote-session-scanner-codex-index.ts @@ -3,10 +3,31 @@ import { joinRemotePath } from '../ssh/ssh-remote-platform' import { extractString, normalizeTitleText, parseJsonObject } from './session-scanner-values' import { remoteSessionContentLines } from './remote-session-content-lines' import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation' -import type { RemoteSessionFilesystemProvider } from './remote-session-scanner-types' +import type { + RemoteScannerContext, + RemoteSessionFilesystemProvider +} from './remote-session-scanner-types' const CODEX_SESSION_INDEX_FILE = 'session_index.jsonl' +// One index read per CODEX_HOME per scan (`context.titleCaches` is scan-scoped); +// used both by the transcript parse and by the parse cache's reuse path. +export function remoteCodexIndexedTitleReader( + codexHome: string, + context: RemoteScannerContext +): (sessionId: string) => Promise { + return async (sessionId) => + ( + await remoteCodexIndexTitles({ + provider: context.provider, + codexHome, + hostPlatform: context.hostPlatform, + titleCaches: context.titleCaches, + signal: context.signal + }) + ).get(sessionId) ?? null +} + export async function remoteCodexIndexTitles(args: { provider: RemoteSessionFilesystemProvider codexHome: string diff --git a/src/main/ai-vault/remote-session-scanner-sources.ts b/src/main/ai-vault/remote-session-scanner-sources.ts index 1fd0ca30a61..2c21a93db9a 100644 --- a/src/main/ai-vault/remote-session-scanner-sources.ts +++ b/src/main/ai-vault/remote-session-scanner-sources.ts @@ -16,7 +16,7 @@ import { partitionSubagentTranscriptPaths } from './session-scanner-subagent-tra import { partitionOmpSubagentTranscriptPaths } from './session-scanner-omp-subagent-transcripts' import type { FileWithMtime } from './session-scanner-types' import { normalizeAgentSessionsDir } from './session-scanner-values' -import { remoteCodexIndexTitles } from './remote-session-scanner-codex-index' +import { remoteCodexIndexedTitleReader } from './remote-session-scanner-codex-index' import { remoteClineSource } from './remote-session-scanner-cline-source' import type { RemoteParserOptions, @@ -216,16 +216,7 @@ function remoteCodexSources( executionHostId: context.executionHostId, executionHostPlatform: context.hostPlatform.os, signal: context.signal, - readIndexedTitle: async (sessionId) => - ( - await remoteCodexIndexTitles({ - provider: context.provider, - codexHome, - hostPlatform, - titleCaches: context.titleCaches, - signal: context.signal - }) - ).get(sessionId) ?? null + readIndexedTitle: remoteCodexIndexedTitleReader(codexHome, context) }) })) } diff --git a/src/main/ai-vault/remote-session-scanner.ts b/src/main/ai-vault/remote-session-scanner.ts index 568b42c27cd..7b586eb9196 100644 --- a/src/main/ai-vault/remote-session-scanner.ts +++ b/src/main/ai-vault/remote-session-scanner.ts @@ -12,6 +12,11 @@ import { dedupeCodexRolloutFileAliases, dedupeCodexSessionsBySessionId } from './codex-session-root-dedup' +import { + parseRemoteSessionFileCached, + remoteSessionParseHostKey +} from './remote-session-parse-cache' +import { remoteCodexIndexedTitleReader } from './remote-session-scanner-codex-index' import { discoverRemoteSourceCandidates } from './remote-session-scanner-discovery' import { remoteSessionSources } from './remote-session-scanner-sources' import type { @@ -25,6 +30,7 @@ import { errorMessage } from './session-scanner-values' import { mapRemoteScanBatches } from './remote-session-scan-batching' import { throwIfAiVaultScanCancelled } from './ai-vault-scan-cancellation' import { recordSessionScanIssue } from './session-scan-issues' +import { refreshCodexTitleFromIndex } from './session-scanner-codex-cached-title' import { limitRemoteScanFilesystemConcurrency } from './remote-session-scan-concurrency' import { aiVaultScanLimit } from '../../shared/ai-vault-session-depth' @@ -224,12 +230,21 @@ async function parseRemoteSessionCandidate( ): Promise { try { throwIfAiVaultScanCancelled(context.signal) - const read = await context.provider.readFile(candidate.file.path) - throwIfAiVaultScanCancelled(context.signal) - if (read.isBinary) { - return null - } - const session = await candidate.source.parse(candidate.file, read.content, context) + // The read is inside the cached parse: an unchanged transcript must not be + // pulled off the remote disk at all, which is the whole cost of #13753. + const session = await parseRemoteSessionFileCached({ + candidate, + hostKey: remoteSessionParseHostKey(context), + parse: async () => { + const read = await context.provider.readFile(candidate.file.path) + throwIfAiVaultScanCancelled(context.signal) + if (read.isBinary) { + return null + } + return await candidate.source.parse(candidate.file, read.content, context) + }, + refreshReusedSession: reusedCodexTitleRefresh(candidate, context) + }) throwIfAiVaultScanCancelled(context.signal) // Mirror the local rule: every session carries its sibling subagent // transcript count (row badge; recoverable signal at zero turns). The @@ -251,6 +266,22 @@ async function parseRemoteSessionCandidate( } } +// Codex thread names live in `/session_index.jsonl`, not the +// rollout, and are written after it — so a transcript-keyed cache hit would +// pin the fallback title forever. Local counterpart: +// session-scanner-parse-cache.ts's reuse path. +function reusedCodexTitleRefresh( + candidate: RemoteSessionCandidate, + context: RemoteScannerContext +): ((session: AiVaultSession) => Promise) | undefined { + const codexHome = candidate.source.agent === 'codex' ? candidate.source.codexHome : undefined + if (!codexHome) { + return undefined + } + const readIndexedTitle = remoteCodexIndexedTitleReader(codexHome, context) + return (session) => refreshCodexTitleFromIndex(session, readIndexedTitle) +} + function mergeRemoteSessions( cappedSessions: AiVaultSession[], scopeSessions: AiVaultSession[] diff --git a/src/main/ai-vault/session-scanner-codex-cached-title.ts b/src/main/ai-vault/session-scanner-codex-cached-title.ts index f356e30b890..ed4045bd1da 100644 --- a/src/main/ai-vault/session-scanner-codex-cached-title.ts +++ b/src/main/ai-vault/session-scanner-codex-cached-title.ts @@ -2,14 +2,29 @@ import type { AiVaultSession } from '../../shared/ai-vault-types' import type { SessionFileCandidate } from './session-scanner-types' import { readCodexSessionIndexTitle } from './session-scanner-codex-title-index' -export async function refreshCachedCodexTitle( +/** + * Codex names a thread in /session_index.jsonl asynchronously, + * after the rollout exists — often after the rollout's last append. A parse + * cache keyed on the transcript's own mtime/size therefore freezes the fallback + * title forever, so every reuse path re-derives it through here. + * + * Both caches share this: `session-scanner-parse-cache.ts` (local disk, via + * `refreshCachedCodexTitle`) and `remote-session-parse-cache.ts` (relay + * provider, whose reader lives in `remote-session-scanner-codex-index.ts`). + */ +export async function refreshCodexTitleFromIndex( + session: AiVaultSession, + readIndexedTitle: (sessionId: string) => Promise +): Promise { + const title = await readIndexedTitle(session.sessionId) + return title && title !== session.title ? { ...session, title } : session +} + +export function refreshCachedCodexTitle( candidate: SessionFileCandidate, session: AiVaultSession ): Promise { - const title = await readCodexSessionIndexTitle( - candidate.file.path, - candidate.codexHome, - session.sessionId + return refreshCodexTitleFromIndex(session, (sessionId) => + readCodexSessionIndexTitle(candidate.file.path, candidate.codexHome, sessionId) ) - return title && title !== session.title ? { ...session, title } : session } diff --git a/src/main/ai-vault/session-scanner-parse-cache.ts b/src/main/ai-vault/session-scanner-parse-cache.ts index 27fc5e0f950..139940ba005 100644 --- a/src/main/ai-vault/session-scanner-parse-cache.ts +++ b/src/main/ai-vault/session-scanner-parse-cache.ts @@ -209,6 +209,8 @@ export async function parseAgentSessionFileCached( entry.session = { ...entry.session, subagentTranscriptCount } } } + // Codex titles come from session_index.jsonl, which mtime+size can't see. + // Remote counterpart: remote-session-scanner.ts's reusedCodexTitleRefresh. if (entry.session && candidate.agent === 'codex') { entry.session = await refreshCachedCodexTitle(candidate, entry.session) } diff --git a/src/main/pty/node-pty-master-fd-retirement.test.ts b/src/main/pty/node-pty-master-fd-retirement.test.ts new file mode 100644 index 00000000000..a0b537de1ca --- /dev/null +++ b/src/main/pty/node-pty-master-fd-retirement.test.ts @@ -0,0 +1,194 @@ +import * as pty from 'node-pty' +import { describe, expect, it } from 'vitest' + +/** + * node-pty hands the master fd to libuv, which closes it on EIO/EOF, but upstream + * never invalidated `_fd`, and none of the three fd-addressed surfaces consulted + * anything: `resize()`, the `process` getter, and `CustomWriteStream`, which holds + * its own plain-number copy of the fd taken at spawn. Orca's patch retires all + * three in the same block that gives up the handle + * (config/patches/node-pty@1.1.0.patch). + * + * Scope: this narrows the window, it does not close it. libuv closes the fd + * synchronously inside `uv_close`, before the JS `'close'` that runs `_close()`, + * so callers still need their own liveness verdict for that tick — and a relay + * host installs node-pty from npm, where this patch is not applied at all. + */ + +const POSIX_SHELL = '/bin/sh' + +function spawnPty(command: string, cols = 80, rows = 24): pty.IPty { + return pty.spawn(POSIX_SHELL, ['-c', command], { + name: 'xterm-256color', + cols, + rows, + cwd: process.cwd(), + env: { ...process.env } + }) +} + +function masterFd(term: pty.IPty): number { + return (term as unknown as { fd: number }).fd +} + +type CustomWriteStream = { _fd: number; _writeQueue: unknown[]; write(data: string): void } + +/** + * `Terminal._close()` already shadows `terminal.write` with a no-op, so the stream + * itself is the surface that still reached the fd: a residual `_writeQueue` and an + * in-flight `fs.write` both re-enter it after the close. + */ +function writeStream(term: pty.IPty): CustomWriteStream { + return (term as unknown as { _writeStream: CustomWriteStream })._writeStream +} + +/** + * Run `command` to completion and let node-pty finish giving up the master. + * + * `destroy: false` exercises only the EIO/EOF read-error path, which reaches + * `_close()` without ever calling `destroy()` — the path the exit of a shell + * actually takes, and the one the write stream was previously never told about. + */ +async function retiredPty( + command = 'exit 0', + { destroy = true }: { destroy?: boolean } = {} +): Promise<{ term: pty.IPty; spawnFd: number }> { + const term = spawnPty(command) + const spawnFd = masterFd(term) + await new Promise((resolve) => { + term.onExit(() => resolve()) + }) + if (destroy) { + ;(term as unknown as { destroy?: () => void }).destroy?.() + } + await new Promise((resolve) => setTimeout(resolve, 400)) + return { term, spawnFd } +} + +// Windows never reaches this code: WindowsTerminal.resize goes through the conpty +// agent and reads no fd, so the sentinel is written and never consulted there. +const describeOnPosix = process.platform === 'win32' ? describe.skip : describe + +describeOnPosix('node-pty master fd retirement', () => { + it('invalidates the descriptor once it gives up the handle', async () => { + const { term, spawnFd } = await retiredPty() + + expect(spawnFd).toBeGreaterThanOrEqual(0) + expect(masterFd(term)).toBe(-1) + }, 15000) + + it('answers a resize past retirement without issuing the ioctl', async () => { + const { term } = await retiredPty() + + // Pre-patch this threw `ioctl(2) failed, EBADF` out of whatever called it. + expect(() => term.resize(200, 50)).not.toThrow() + // Geometry stays at the last size actually applied rather than claiming one + // that no descriptor ever received. + expect([term.cols, term.rows]).toEqual([80, 24]) + }, 15000) + + it('retires the write stream fd on _close(), not only on destroy()', async () => { + const { term } = await retiredPty('exit 0', { destroy: false }) + + // The stream copied the fd number at spawn, so `Terminal._fd = -1` alone + // leaves it addressing a descriptor the kernel may already have reissued. + expect(writeStream(term)._fd).toBe(-1) + + writeStream(term).write('x') + expect(writeStream(term)._writeQueue).toHaveLength(0) + }, 15000) + + it('names the spawn file rather than tcgetpgrp on a retired descriptor', async () => { + const { term } = await retiredPty() + + expect(term.process).toBe(POSIX_SHELL) + }, 15000) +}) + +// Linux frees the master synchronously enough that the very next pty is handed the +// same descriptor number every time, which makes the reuse hazard directly +// observable rather than a race to reproduce. +// +// Which is also the constraint on writing a case here: the kernel hands out the +// lowest free number, so a case that returns while its live pty is still open +// leaks that descriptor into the next case's premise as an off-by-one. Await the +// exit, never a fixed sleep. +const describeOnLinux = process.platform === 'linux' ? describe : describe.skip + +describeOnLinux('node-pty master fd reuse', () => { + it('cannot resize a live pty handed the retired descriptor number', async () => { + const { term: retired, spawnFd } = await retiredPty() + + const live = spawnPty('sleep 1; stty size') + let output = '' + live.onData((data) => { + output += data + }) + try { + // The premise of this test: the kernel really did reissue the number. If it + // stops holding, the assertion below would pass for the wrong reason. + expect(masterFd(live)).toBe(spawnFd) + + // Pre-patch this reached TIOCSWINSZ on `live`'s master and silently resized + // a terminal it has no relationship to — no error, nothing for a liveness + // probe of the retired pid to observe. + retired.resize(200, 50) + + await new Promise((resolve) => { + live.onExit(() => resolve()) + }) + expect(output.trim()).toBe('24 80') + } finally { + live.kill() + } + }, 15000) + + it('cannot write into a live pty handed the retired descriptor number', async () => { + const { term: retired, spawnFd } = await retiredPty('exit 0', { destroy: false }) + + const live = spawnPty('sleep 1') + let output = '' + live.onData((data) => { + output += data + }) + try { + expect(masterFd(live)).toBe(spawnFd) + + // Pre-patch this fs.write reached `live`'s master, and the line discipline + // echoed it straight back: a retired pane's bytes landing in an unrelated + // terminal. Nothing has to read them for the leak to be observable. + writeStream(retired).write('leak\r') + + // Await the exit rather than sleeping: a fixed wait leaves this descriptor + // open into the next case, whose `expect(masterFd(live)).toBe(spawnFd)` + // premise then sees the kernel hand out the lower number this pty was still + // holding. That is what broke `node 24 1/8` on Linux, where alone among the + // platforms these cases actually run. + await new Promise((resolve) => { + live.onExit(() => resolve()) + }) + expect(output).not.toContain('leak') + } finally { + live.kill() + } + }, 15000) + + it('does not name a live pty foreground process off the retired descriptor', async () => { + const { term: retired, spawnFd } = await retiredPty() + + // `exec` replaces the shell, so the foreground pgrp's cmdline is distinct + // from the file this pty was spawned with. + const live = spawnPty('exec sleep 5') + try { + expect(masterFd(live)).toBe(spawnFd) + // Let the shell finish exec'ing, or its own cmdline is still the fallback. + await new Promise((resolve) => setTimeout(resolve, 300)) + + // Pre-patch this read tcgetpgrp off `live`'s master and reported `sleep`, + // attributing an unrelated pane's process to a pty that had already exited. + expect(retired.process).toBe(POSIX_SHELL) + } finally { + live.kill() + } + }, 15000) +}) diff --git a/src/relay/pty-handler-resize-stale-pty.test.ts b/src/relay/pty-handler-resize-stale-pty.test.ts new file mode 100644 index 00000000000..0dcb6b56497 --- /dev/null +++ b/src/relay/pty-handler-resize-stale-pty.test.ts @@ -0,0 +1,176 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ + mockPtySpawn: vi.fn(), + mockCreateShellPromptReadinessProbe: vi.fn(), + mockPtyInstance: { + pid: process.pid, + onData: vi.fn(), + onExit: vi.fn(), + write: vi.fn(), + resize: vi.fn(), + kill: vi.fn(), + clear: vi.fn(), + pause: vi.fn(), + resume: vi.fn() + } +})) + +vi.mock('node-pty', () => ({ + spawn: mockPtySpawn +})) + +vi.mock('../main/pty/posix-pty-process-groups', () => ({ + forceKillPosixPtyProcessGroups: vi.fn((_pid: number, fallback: () => void) => fallback()) +})) + +vi.mock('../main/shell-prompt-readiness-probe', () => ({ + createShellPromptReadinessProbe: mockCreateShellPromptReadinessProbe +})) + +import type { PtyHandler } from './pty-handler' +import * as ptyShellUtils from './pty-shell-utils' +import { beginPtyHandlerTest, endPtyHandlerTest, testPtyId } from './pty-handler-test-harness' +import type { MockDispatcher } from './pty-handler-test-harness' + +const PTY_1 = testPtyId(1) +const STALE_PID = 424_242 + +/** + * node-pty's native `pty.resize` error when the ioctl reaches a closed master. + * + * Orca's node-pty patch retires `_fd` when it gives up the master, so a patched + * handle answers a late resize with a no-op. That leaves this handler two cases + * it still has to contain: the tick between libuv closing the fd and node-pty's + * own handler observing it, and a relay host, which installs node-pty from npm + * and has no such guard. + */ +function ebadfResize(): never { + throw new Error('ioctl(2) failed, EBADF') +} + +describe('PtyHandler.resize against a stale PTY handle', () => { + let dispatcher: MockDispatcher + let handler: PtyHandler + let originalPlatform: PropertyDescriptor | undefined + let resize: ReturnType + + beforeEach(async () => { + ;({ dispatcher, handler, originalPlatform } = beginPtyHandlerTest({ + mockPtySpawn, + mockPtyInstance, + mockCreateShellPromptReadinessProbe + })) + resize = vi.fn() + // A shell that exited without node-pty producing `onExit`: the record is + // still in the pool and undisposed, but the master behind it is gone. + mockPtySpawn.mockReturnValue({ ...mockPtyInstance, pid: STALE_PID, resize }) + await dispatcher.callRequest('pty.spawn', {}) + expect(handler.activePtyCount).toBe(1) + }) + + afterEach(async () => { + await endPtyHandlerTest(handler, originalPlatform) + }) + + it('retires an entry whose pid the host proves is gone, instead of issuing the ioctl', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(false) + resize.mockImplementation(ebadfResize) + + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + ).not.toThrow() + + expect(resize).not.toHaveBeenCalled() + // The record must leave the pool: while it stays, the relay keeps + // advertising a dead shell and `activePtyCount` never reaches zero, so a + // relay configured with an unlimited grace never reaches its idle exit. + expect(handler.activePtyCount).toBe(0) + }) + + it('contains an ioctl failure without re-classifying liveness, and keeps the record', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(true) + resize.mockImplementation(ebadfResize) + const stderr = vi.spyOn(process.stderr, 'write').mockReturnValue(true) + + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + ).not.toThrow() + // Repeats must stay contained too — this is the notification the client + // re-sends on every reconnect and every window resize. + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 90, rows: 30 }) + ).not.toThrow() + + // Loss of an fd observed the handle, not the host that owns the pid, so it is + // `unverifiable` and the claim is retained. Only the probe above retires. + expect(handler.activePtyCount).toBe(1) + expect(stderr.mock.calls.map(([line]) => String(line)).join('')).toContain( + 'ioctl(2) failed, EBADF' + ) + }) + + it('retires the entry when the pid goes absent between the pre-probe and the ioctl', () => { + // The race the catch-block re-probe exists for, and the one a constant + // liveness mock cannot express: alive when the pre-probe asks, gone by the + // time the ioctl fails. libuv closes the master synchronously inside + // `uv_close`, so this window opens before any JS guard can be set — and on + // a relay host, where node-pty comes from npm, there is no JS guard at all. + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValueOnce(true).mockReturnValueOnce(false) + resize.mockImplementation(ebadfResize) + const stderr = vi.spyOn(process.stderr, 'write').mockReturnValue(true) + + expect(() => + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + ).not.toThrow() + + expect(resize).toHaveBeenCalledTimes(1) + expect(handler.activePtyCount).toBe(0) + // Proven `exited` retires silently; only live-or-unverifiable is reported. + expect(stderr).not.toHaveBeenCalled() + }) + + it('publishes the exit so the client stops holding a pane on a retired session', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(false) + + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 120, rows: 40 }) + + // Retiring the record without this leaves the pane mounted against a + // session the relay has already forgotten: the next attach answers + // `PTY "" not found` and nothing before it explained why. + expect(dispatcher._notifications).toContainEqual({ + method: 'pty.exit', + params: { id: PTY_1, code: -1, incarnationId: expect.any(String) } + }) + }) + + it('does not republish an exit node-pty already reported', async () => { + // The listing sweep also reaps entries the natural `onExit` left behind; a + // second `pty.exit` would hand the client a duplicate carrying -1 in place + // of the real status. That sweep retires off `managed.disposed`, which is + // bookkeeping rather than liveness, so it never publishes a verdict at all. + const onExit = mockPtyInstance.onExit.mock.calls.at(-1)?.[0] as (e: { + exitCode: number + }) => void + onExit({ exitCode: 7 }) + await vi.runAllTimersAsync() + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(false) + + await dispatcher.callRequest('pty.listProcesses', {}) + + const exits = dispatcher._notifications.filter( + (notification) => notification.method === 'pty.exit' + ) + expect(exits).toHaveLength(1) + expect(exits[0]?.params).toMatchObject({ id: PTY_1, code: 7 }) + }) + + it('still resizes a live PTY, with the clamped geometry', () => { + vi.spyOn(ptyShellUtils, 'isProcessAlive').mockReturnValue(true) + + dispatcher.callNotification('pty.resize', { id: PTY_1, cols: 4_000, rows: 40 }) + + expect(resize).toHaveBeenCalledWith(500, 40) + expect(handler.activePtyCount).toBe(1) + }) +}) diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index 728f883d3b1..20b64d16b1c 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -87,6 +87,21 @@ describe('PtyHandler', () => { expect(notifMethods).not.toContain('pty.ackData') }) + it('rescans the process table for a close decision but not for a poll', async () => { + const hasChildren = vi.mocked(ptyShellUtils.processHasChildren) + const { id } = (await spawnPty({ cols: 80, rows: 24 })) as { id: string } + hasChildren.mockClear() + + await dispatcher.callRequest('pty.inspectProcess', { id }) + // The poll shares the TTL-cached table the foreground lookup already took. + expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid) + + await dispatcher.callRequest('pty.hasChildProcesses', { id }) + // This RPC only ever gates a destructive decision (window close, workspace + // cleanup), so it has to see a child started inside the 500ms window. + expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid, { fresh: true }) + }) + it('rejects strict process inspection for a missing relay PTY', async () => { await expect(dispatcher.callRequest('pty.inspectProcess', { id: 'missing' })).rejects.toThrow( 'terminal_gone' diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 25945203878..b7fe5bb160b 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -1990,8 +1990,7 @@ export class PtyHandler { } // Why: verify liveness because shells can exit without node-pty onExit. - if (managed.pty.pid && !isProcessAlive(managed.pty.pid)) { - this.reapExitedPty(managed) + if (this.reapPtyProvenExited(managed)) { throw new Error(`PTY "${id}" not found`) } @@ -2109,8 +2108,39 @@ export class PtyHandler { const cols = Math.max(1, Math.min(500, Math.floor(Number(params.cols) || 80))) const rows = Math.max(1, Math.min(500, Math.floor(Number(params.rows) || 24))) const managed = this.ptys.get(id) - if (managed && !managed.disposed) { + if (!managed || managed.disposed) { + return + } + // Why probe (same probe attach() and listProcesses() run): a shell that + // exited without node-pty's `onExit` leaves an undisposed entry behind, and + // while it stays the relay keeps advertising a dead shell and keeps holding + // `activePtyCount` above zero, which is what stops a relay with + // `relayGracePeriodSeconds: 0` from ever reaching its idle-no-ptys exit + // (#12423). This is retirement, not ioctl safety: only ESRCH from the host + // that owns the pid is evidence of `exited`. + if (this.reapPtyProvenExited(managed)) { + return + } + // The patched node-pty retires `_fd` in the same block that gives up the + // master (config/patches/node-pty@1.1.0.patch), which makes a resize past + // that point a no-op rather than a TIOCSWINSZ aimed at a reused descriptor. + // That covers only part of the window and does not cover this process at + // all: libuv closes the fd synchronously inside `uv_close`, before the JS + // `'close'` that runs `_close()`, and a relay host installs node-pty from + // npm, where the patch is not applied. So the catch below stays. + try { managed.pty.resize(cols, rows) + } catch (err) { + // A failed ioctl observed the handle, not the host's process table, so on + // its own it is `unverifiable`. Re-probe: a now-absent pid retires the + // entry, anything else keeps it and is contained here rather than + // escaping as a parse error on every later resize. + if (this.reapPtyProvenExited(managed)) { + return + } + process.stderr.write( + `[pty-handler] resize failed for PTY ${id} whose process is still live or unverifiable: ${err instanceof Error ? err.message : String(err)}\n` + ) } } @@ -2249,9 +2279,7 @@ export class PtyHandler { if (this.ptys.get(managed.id) !== managed || managed.disposed) { return } - const pid = managed.pty.pid - if (pid && !isProcessAlive(pid)) { - this.reapExitedPty(managed) + if (this.reapPtyProvenExited(managed)) { return } if (attemptsRemaining <= 0) { @@ -2278,12 +2306,22 @@ export class PtyHandler { managed.reapTimer = timer } - /** Retire every record for a PTY whose process is proven gone. Shared by the attach probe, the - * listing probe and the post-shutdown sweep so the three cannot drift on what "gone" retires. */ - private reapExitedPty(managed: ManagedPty): void { + /** + * Retire every record for a PTY whose process is proven gone. Shared by the attach probe, the + * listing probe and the post-shutdown sweep so the three cannot drift on what "gone" retires. + * + * `evidence` is not decoration: `exited` publishes a verdict to the client, and only ESRCH from + * the host that owns the pid earns it. The disposed-record sweep retires off our own + * bookkeeping, which says we tore the record down — not that the shell died — so it stays + * silent (docs/reference/ssh-execution-boundary.md). + */ + private reapExitedPty(managed: ManagedPty, evidence: 'exited' | 'record-torn-down'): void { managed.physicalExit?.markExited() this.releaseRelayIngress(managed) this.flushPtyOutput(managed.id) + if (evidence === 'exited') { + this.publishReapedExit(managed) + } this.notifyExitListener(managed) this.agentSessionOwners.release(managed.id) disposeManagedPty(managed) @@ -2291,6 +2329,53 @@ export class PtyHandler { this.clearPtyFlowState(managed.id) } + /** + * A reap is an exit the client has to hear about. `notifyExitListener` is + * relay-internal, so a retirement that stops there leaves the pane mounted + * against a session the relay has already forgotten — the next attach answers + * `PTY "" not found` and nothing before it said why. `resize` made that + * user-triggered. + * + * `-1` is this wire's "gone, status unrecoverable": the pid is proven absent + * (ESRCH from the host that owns it) but nothing waited on the shell, so no + * status exists. `ssh-relay-session` already publishes the same code for a + * dropped lease. Reached only from the proven-exited path — a client that acts + * on this retires the pane, so nothing weaker than ESRCH may reach it. + */ + private publishReapedExit(managed: ManagedPty): void { + // Why the guard: node-pty's own `onExit` already queued and published this + // pty's real exit code before reaching here, and the sweep also reaps + // entries that path left behind. + if (managed.exitListenerNotified || this.pendingExitByPty.has(managed.id)) { + return + } + this.pendingExitByPty.set(managed.id, { + id: managed.id, + code: -1, + incarnationId: managed.incarnationId + }) + this.publishPendingExit(managed.id) + } + + /** + * Retire this entry when the host proves its pid is gone; report whether it was. + * + * `managed.disposed` is bookkeeping, not liveness: it says we tore the record + * down, not that the shell died. A shell can exit without node-pty producing + * `onExit`, which leaves a non-disposed entry holding a handle whose master fd + * is already closed. Only `isProcessAlive` (ESRCH, from the host that owns the + * process) is positive evidence of absence; every other outcome is + * `unverifiable` and keeps its record and owner claim + * (docs/reference/ssh-execution-boundary.md). + */ + private reapPtyProvenExited(managed: ManagedPty): boolean { + if (!managed.pty.pid || isProcessAlive(managed.pty.pid)) { + return false + } + this.reapExitedPty(managed, 'exited') + return true + } + private async sendSignal(params: Record): Promise { const id = params.id as string const signal = params.signal as string @@ -2429,7 +2514,12 @@ export class PtyHandler { if (!managed || managed.disposed) { return false } - return await processHasChildren(managed.pty.pid) + // Fresh, not TTL-cached: this RPC exists to gate destructive decisions (the + // window-close confirmation, workspace cleanup's idle evidence), which act + // on the answer once. `pty.inspectProcess` below stays on the shared + // snapshot because it is the polled path, where a scan per pane per tick is + // the fork storm the cache removed. + return await processHasChildren(managed.pty.pid, { fresh: true }) } private async getForegroundProcess(params: Record): Promise { @@ -2494,8 +2584,11 @@ export class PtyHandler { } } for (const [entryIndex, [id, managed]] of managedEntries.entries()) { - if (managed.disposed || (managed.pty.pid && !isProcessAlive(managed.pty.pid))) { - this.reapExitedPty(managed) + if (managed.disposed) { + this.reapExitedPty(managed, 'record-torn-down') + continue + } + if (this.reapPtyProvenExited(managed)) { continue } // Reuse batched correlation; per-PTY tree scans recreate O(PTY × rows) work. diff --git a/src/relay/pty-shell-utils.test.ts b/src/relay/pty-shell-utils.test.ts index 6bfeab1a56a..0b4148a2a09 100644 --- a/src/relay/pty-shell-utils.test.ts +++ b/src/relay/pty-shell-utils.test.ts @@ -17,6 +17,7 @@ import { resetProcessTableSnapshotForTests } from '../shared/process-table-snaps import { getForegroundProcessName, isProcessAlive, + processHasChildren, resolveDefaultCwd, resolveWindowsDefaultShell } from './pty-shell-utils' @@ -525,3 +526,87 @@ describe('getForegroundProcessName', () => { await expect(getForegroundProcessName(100)).resolves.toBe('bash') }) }) + +describe('processHasChildren', () => { + // Why these assert on argv, not just the answer: the defect in #13537 was the + // cost of the answer. `pgrep -P` forks per pane per poll and opens six procfs + // files per host process to resolve one ppid, so the contract worth pinning is + // "no fork of its own, and share the foreground lookup's cached table". + const PS_TABLE = ['100 1 Ss bash', '101 100 S+ node /opt/codex', '200 1 Ss zsh'].join('\n') + + it('answers from the shared process table without forking pgrep', async () => { + await withProcessPlatform('linux', async () => { + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: PS_TABLE } + } + return new Error('unexpected command') + }) + + await expect(processHasChildren(100)).resolves.toBe(true) + await expect(processHasChildren(200)).resolves.toBe(false) + + expect(execFileMock.mock.calls.map((call) => call[0])).not.toContain('pgrep') + }) + }) + + it('shares one process-table capture across a burst of panes', async () => { + await withProcessPlatform('linux', async () => { + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: PS_TABLE } + } + return new Error('unexpected command') + }) + + const answers = await Promise.all([ + processHasChildren(100), + processHasChildren(100), + processHasChildren(200), + getForegroundProcessName(100, 'bash') + ]) + + expect(answers).toEqual([true, true, false, 'codex']) + expect(execFileMock).toHaveBeenCalledTimes(1) + }) + }) + + it('rescans for a close decision rather than serving a table from inside the TTL', async () => { + await withProcessPlatform('linux', async () => { + let table = PS_TABLE + mockExecFile((_command, args) => { + if (args[0] === '-axo') { + return { stdout: table } + } + return new Error('unexpected command') + }) + + // The poll answers from the cache, which is the whole point of the memo. + await expect(processHasChildren(200)).resolves.toBe(false) + table = [PS_TABLE, '201 200 S+ npm run build'].join('\n') + await expect(processHasChildren(200)).resolves.toBe(false) + expect(execFileMock).toHaveBeenCalledTimes(1) + + // A close or cleanup acts on the answer once and destructively, so a + // child started inside the 500ms window has to be visible to it. + await expect(processHasChildren(200, { fresh: true })).resolves.toBe(true) + expect(execFileMock).toHaveBeenCalledTimes(2) + }) + }) + + it('reports no children when the process table is unreadable', async () => { + await withProcessPlatform('linux', async () => { + mockExecFile(() => new Error('ps table unavailable')) + + await expect(processHasChildren(100)).resolves.toBe(false) + }) + }) + + it('spawns nothing on Windows, where the answer was always false', async () => { + await withProcessPlatform('win32', async () => { + await expect(processHasChildren(100)).resolves.toBe(false) + + expect(execFileMock).not.toHaveBeenCalled() + }) + }) +}) diff --git a/src/relay/pty-shell-utils.ts b/src/relay/pty-shell-utils.ts index e89ac60a7ed..474416d3a41 100644 --- a/src/relay/pty-shell-utils.ts +++ b/src/relay/pty-shell-utils.ts @@ -10,6 +10,7 @@ import { } from '../shared/agent-process-recognition' import { getFirstCommandToken } from '../shared/command-token-scanner' import { + getFreshProcessTableSnapshot, getProcessTableSnapshot, type ProcessTableIndex, type ProcessTableRow @@ -170,15 +171,37 @@ export async function resolveProcessCwd(pid: number, fallbackCwd: string): Promi } /** - * Check whether a process has child processes (via pgrep). + * Check whether a process has child processes. + * + * Why the shared snapshot and not `pgrep -P`: this answers one field of + * `pty.inspectProcess`, which every tracked pane polls on a 750ms/2000ms + * cadence, and the fork was neither cached nor coalesced. procps-ng opens six + * procfs files per process to resolve a ppid — including a `/proc//ctty` + * that never exists on Linux — so one call cost O(host process count) syscalls, + * ~4k opens per pgrep on a 690-process host, at up to 8 forks/sec (#13537). + * `getForegroundProcessName` in the same RPC already captured the TTL-cached + * `ps` table, whose index carries the parent/child map, so the answer is free. + * + * `fresh` opts out of that TTL. A poll can read a 500ms-old table because its + * next tick corrects it, but a close or cleanup decision acts on the answer + * once and destructively — a child that started inside the TTL would be killed + * with no confirmation. `pgrep` scanned per call, so anything that decides + * has to keep scanning per call. */ -export async function processHasChildren(pid: number): Promise { +export async function processHasChildren( + pid: number, + options?: { fresh?: boolean } +): Promise { + // Windows has no `ps`; the previous `pgrep` fork always failed here too, so + // this keeps the same answer without spawning anything to reach it. + if (process.platform === 'win32') { + return false + } try { - const { stdout } = await execFile('pgrep', ['-P', String(pid)], { - encoding: 'utf-8', - timeout: 3000 - }) - return stdout.trim().length > 0 + const rows = options?.fresh + ? await getFreshProcessTableSnapshot() + : await getProcessTableSnapshot() + return (getProcessTableIndex(rows).childrenByPpid.get(pid)?.length ?? 0) > 0 } catch { return false } From 3a8f806d6c71ac2e6f1e82b7d04ffd9384c662e0 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:48 -0700 Subject: [PATCH 118/398] fix(remote-terminal): keep the stream stall deadline armed on unacknowledged credit (#17871) * fix(remote-terminal): keep the stream stall deadline armed on unacknowledged credit A paired-runtime terminal could stall silently with a live socket, a live PTY and no transport error (#11265). Two compounding defects on the read side: - The stream watchdog re-armed its 30s delivery deadline from zero on every settled delivery, so sibling traffic postponed the verdict indefinitely, and it cleared the timer entirely once renderer parse credit hit zero. Re-arming required inbound output -- the exact thing an exhausted host ACK window stops -- so once an ACK went missing nothing could ever detect the stall. The deadline is now anchored to the oldest unsettled delivery and stays armed while delivered bytes remain unacknowledged to the host. - flushOutputAcknowledgement zeroed pendingAckBytes before knowing the ACK frame was accepted, permanently shrinking the host's send window. Unsent bytes are re-charged and the flush timer re-armed. Recovery still reports onTransportClose({recoverable:true}); no path claims the PTY exited. * fix(remote-terminal): stop rearming the ack flush for a stream the failed send dropped A failing ACK send tears the stream down inside sendFrame, so the re-charge path armed a 4ms timer on an unregistered stream whose watchdog was already disposed; every later send returned false on !ready and rescheduled again. Also adds the missing integration coverage for the real ack -> watchdog flow. * fix(i18n): restore the activity-options key the rebase dropped The branch's en.json predates #18245, which added both the translate() call and its key. Rebasing took the branch copy wholesale, silently dropping the key and failing verify:localization-catalog. * fix(i18n): union en.json with main so the rebase cannot drop keys --- ...remote-runtime-terminal-flow-controller.ts | 35 ++++++--- ...untime-terminal-parse-backpressure.test.ts | 50 +++++++++++++ ...te-runtime-terminal-stall-recovery.test.ts | 72 +++++++++++++++++++ .../remote-terminal-stream-watchdog.test.ts | 58 +++++++++++++++ .../remote-terminal-stream-watchdog.ts | 40 ++++++++--- 5 files changed, 237 insertions(+), 18 deletions(-) create mode 100644 src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts diff --git a/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts b/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts index 640ec25933e..0488dd323cf 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-flow-controller.ts @@ -77,27 +77,46 @@ export abstract class RemoteRuntimeTerminalFlowController extends RemoteRuntimeT stream: RemoteRuntimeMultiplexedTerminalState, bytes: number ): boolean { - if (this.streams.get(stream.streamId) !== stream) { + if (!this.isRegisteredStream(stream)) { return true } stream.pendingAckBytes += bytes if (stream.pendingAckBytes >= TERMINAL_MULTIPLEX_ACK_BATCH_BYTES) { return this.flushOutputAcknowledgement(stream) } - if (stream.ackFlushTimer === null) { - stream.ackFlushTimer = setTimeout(() => { - stream.ackFlushTimer = null - this.flushOutputAcknowledgement(stream) - }, TERMINAL_MULTIPLEX_ACK_FLUSH_MS) - } + this.scheduleOutputAcknowledgementFlush(stream) return true } + private scheduleOutputAcknowledgementFlush(stream: RemoteRuntimeMultiplexedTerminalState): void { + if (stream.ackFlushTimer !== null) { + return + } + stream.ackFlushTimer = setTimeout(() => { + stream.ackFlushTimer = null + this.flushOutputAcknowledgement(stream) + }, TERMINAL_MULTIPLEX_ACK_FLUSH_MS) + } + private flushOutputAcknowledgement(stream: RemoteRuntimeMultiplexedTerminalState): boolean { clearAckFlushTimer(stream) const bytes = stream.pendingAckBytes + if (bytes <= 0) { + return true + } stream.pendingAckBytes = 0 - return bytes <= 0 || this.acknowledgeOutput(stream, bytes) + if (this.acknowledgeOutput(stream, bytes)) { + // Why: only a frame the transport took reopens the host's send window. + stream.watchdog.recordOutputAcknowledged(bytes) + return true + } + // Why guarded: a failed send may have torn the stream down, and re-charging a dropped stream reschedules itself forever. + if (this.isRegisteredStream(stream)) { + // Why re-charged: dropping an unsent ack shrinks the host window for the stream's life, and the only retry trigger is the output that shrunken window blocks. + stream.pendingAckBytes += bytes + this.scheduleOutputAcknowledgementFlush(stream) + } + return false } getStreamsForE2e(): Iterable { diff --git a/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts b/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts index ba49c7cfa26..0a004a0c564 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-parse-backpressure.test.ts @@ -427,6 +427,56 @@ describe('remote terminal renderer backpressure', () => { expect(sentAckBytes()).toEqual([1]) }) + it('stops rearming the ack flush after a failed ACK tears the stream down', async () => { + vi.useFakeTimers() + try { + const { getRemoteRuntimeTerminalMultiplexer } = + await import('./remote-runtime-terminal-multiplexer') + const { takeCurrentTerminalDeliveryCredit } = + await import('../lib/pane-manager/terminal-delivery-credit') + const parseCredits: (() => void)[] = [] + const stream = await getRemoteRuntimeTerminalMultiplexer('windows-test').subscribeTerminal({ + terminal: 'term-wedge', + client: { id: 'mac-viewer', type: 'desktop' }, + callbacks: { + onData: () => { + const credit = takeCurrentTerminalDeliveryCredit() + if (credit) { + parseCredits.push(credit) + } + }, + onSnapshot: vi.fn() + } + }) + sendBinary.mockClear() + sendBinary.mockImplementation((bytes) => { + if (decodeTerminalStreamFrame(bytes)?.opcode === TerminalStreamOpcode.Ack) { + throw new Error('socket closed') + } + }) + callbacks?.onBinary( + encodeTerminalStreamFrame({ + opcode: TerminalStreamOpcode.Output, + streamId: stream.streamId, + seq: 1, + payload: encodeTerminalStreamText('x') + }) + ) + parseCredits[0]?.() + await vi.advanceTimersByTimeAsync(50) + + expect(unsubscribe).toHaveBeenCalledOnce() + expect(vi.getTimerCount()).toBe(0) + await vi.advanceTimersByTimeAsync(10_000) + expect(vi.getTimerCount()).toBe(0) + await vi.advanceTimersByTimeAsync(70_000) + expect(vi.getTimerCount()).toBe(0) + expect(sentAckBytes()).toEqual([1]) + } finally { + vi.useRealTimers() + } + }) + function sentAckBytes(): number[] { return sendBinary.mock.calls.flatMap(([bytes]) => { const frame = decodeTerminalStreamFrame(bytes) diff --git a/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts b/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts index 1e8b239d1ce..c6e64e4e410 100644 --- a/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts +++ b/src/renderer/src/runtime/remote-runtime-terminal-stall-recovery.test.ts @@ -11,6 +11,7 @@ import { REMOTE_TERMINAL_COMMAND_RESPONSE_TIMEOUT_MS, REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS } from './remote-terminal-stream-watchdog' +import { TERMINAL_MULTIPLEX_ACK_FLUSH_MS } from '../../../shared/terminal-multiplex-flow-control' describe('remote terminal stalled stream recovery', () => { const sendBinary = vi.fn() @@ -101,6 +102,77 @@ describe('remote terminal stalled stream recovery', () => { healthy.close() }) + it('keeps a stream alive once the transport takes the ack for its parsed output', async () => { + const { getRemoteRuntimeTerminalMultiplexer } = + await import('./remote-runtime-terminal-multiplexer') + const { takeCurrentTerminalDeliveryCredit } = + await import('../lib/pane-manager/terminal-delivery-credit') + const credits: (() => void)[] = [] + const onTransportClose = vi.fn() + const stream = await getRemoteRuntimeTerminalMultiplexer('windows-test').subscribeTerminal({ + terminal: 'term-acked', + client: { id: 'mac-viewer', type: 'desktop' }, + callbacks: { + onData: () => { + const credit = takeCurrentTerminalDeliveryCredit() + if (credit) { + credits.push(credit) + } + }, + onSnapshot: vi.fn(), + onTransportClose + } + }) + sendBinary.mockClear() + + emitOutput(stream.streamId, 'host output the renderer parses') + credits[0]?.() + await vi.advanceTimersByTimeAsync(TERMINAL_MULTIPLEX_ACK_FLUSH_MS) + expect(sentFrames(TerminalStreamOpcode.Ack)).toHaveLength(1) + + await vi.advanceTimersByTimeAsync(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS) + + expect(onTransportClose).not.toHaveBeenCalled() + expect(sentUnsubscribeStreamIds()).toEqual([]) + stream.close() + }) + + it('keeps the delivery deadline anchored while sibling frames keep settling', async () => { + const { getRemoteRuntimeTerminalMultiplexer } = + await import('./remote-runtime-terminal-multiplexer') + const { takeCurrentTerminalDeliveryCredit } = + await import('../lib/pane-manager/terminal-delivery-credit') + const credits: (() => void)[] = [] + const onTransportClose = vi.fn() + const stream = await getRemoteRuntimeTerminalMultiplexer('windows-test').subscribeTerminal({ + terminal: 'term-anchored', + client: { id: 'mac-viewer', type: 'desktop' }, + callbacks: { + onData: () => { + const credit = takeCurrentTerminalDeliveryCredit() + if (credit) { + credits.push(credit) + } + }, + onSnapshot: vi.fn(), + onTransportClose + } + }) + sendBinary.mockClear() + + emitOutput(stream.streamId, 'frame the renderer never parses') + await vi.advanceTimersByTimeAsync(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS - 5_000) + emitOutput(stream.streamId, 'sibling frame that settles') + credits[1]?.() + await vi.advanceTimersByTimeAsync(TERMINAL_MULTIPLEX_ACK_FLUSH_MS) + + expect(onTransportClose).not.toHaveBeenCalled() + await vi.advanceTimersByTimeAsync(5_000) + + expect(onTransportClose).toHaveBeenCalledWith({ recoverable: true }) + expect(sentUnsubscribeStreamIds()).toEqual([stream.streamId]) + }) + it('probes then restarts a stream when an entered command receives no frames', async () => { const { getRemoteRuntimeTerminalMultiplexer, REMOTE_TERMINAL_SNAPSHOT_REQUEST_TIMEOUT_MS } = await import('./remote-runtime-terminal-multiplexer') diff --git a/src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts b/src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts new file mode 100644 index 00000000000..66d65f9bcbc --- /dev/null +++ b/src/renderer/src/runtime/remote-terminal-stream-watchdog.test.ts @@ -0,0 +1,58 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS, + createRemoteTerminalStreamWatchdog +} from './remote-terminal-stream-watchdog' + +describe('remote terminal stream watchdog delivery deadline', () => { + beforeEach(() => { + vi.useFakeTimers() + }) + + afterEach(() => { + vi.useRealTimers() + }) + + it('anchors the deadline to the oldest unsettled delivery instead of the last settled sibling', () => { + const onStall = vi.fn() + const watchdog = createRemoteTerminalStreamWatchdog(onStall) + + watchdog.beginOutputDelivery(100) + for (let tick = 0; tick < 3; tick += 1) { + vi.advanceTimersByTime(9_000) + const settle = watchdog.beginOutputDelivery(10) + settle() + } + expect(onStall).not.toHaveBeenCalled() + + vi.advanceTimersByTime(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS - 27_000) + + expect(onStall).toHaveBeenCalledTimes(1) + expect(onStall.mock.calls[0]?.[0]).toMatchObject({ reason: 'delivery-credit-timeout' }) + }) + + it('stays armed while parsed bytes remain unacknowledged to the host', () => { + const onStall = vi.fn() + const watchdog = createRemoteTerminalStreamWatchdog(onStall) + + watchdog.beginOutputDelivery(100)() + vi.advanceTimersByTime(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS) + + expect(onStall).toHaveBeenCalledTimes(1) + expect(onStall.mock.calls[0]?.[0]).toMatchObject({ + outstandingDeliveryBytes: 0, + reason: 'delivery-credit-timeout' + }) + }) + + it('disarms once the acknowledgement reaches the transport', () => { + const onStall = vi.fn() + const watchdog = createRemoteTerminalStreamWatchdog(onStall) + + watchdog.beginOutputDelivery(100)() + watchdog.recordOutputAcknowledged(100) + vi.advanceTimersByTime(REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS * 2) + + expect(onStall).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts b/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts index 4cb6b081c5f..be9e1b9d404 100644 --- a/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts +++ b/src/renderer/src/runtime/remote-terminal-stream-watchdog.ts @@ -9,6 +9,8 @@ export type RemoteTerminalStreamStall = { export type RemoteTerminalStreamWatchdog = { beginOutputDelivery: (bytes: number) => () => void + /** Bytes whose ACK frame reached the transport, releasing the host's window. */ + recordOutputAcknowledged: (bytes: number) => void completeCommandResponseProbe: () => void recordCommandInput: (text: string) => void recordInbound: () => void @@ -21,6 +23,9 @@ export function createRemoteTerminalStreamWatchdog( let responseTimer: ReturnType | null = null let deliveryTimer: ReturnType | null = null let outstandingDeliveryBytes = 0 + // Why separate from parse credit: the host reopens its window on ACK frames, so bytes parsed but not yet ACKed are still the credit whose loss stops output. + let unacknowledgedBytes = 0 + let deliveryPendingSinceMs: number | null = null let lastInboundAtMs = Date.now() let commandResponseProbePending = false let disposed = false @@ -54,23 +59,29 @@ export function createRemoteTerminalStreamWatchdog( reason }) } - const armDeliveryTimer = (): void => { - clearDeliveryTimer() - if (outstandingDeliveryBytes <= 0 || disposed) { + // Why anchored, never restarted: a deadline re-armed by sibling settles is postponed forever, and one cleared at zero parse credit can only re-arm from inbound output — which is what the stall stops. + const syncDeliveryTimer = (): void => { + if (disposed || outstandingDeliveryBytes + unacknowledgedBytes <= 0) { + clearDeliveryTimer() + deliveryPendingSinceMs = null return } - deliveryTimer = setTimeout( - () => trip('delivery-credit-timeout'), - REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS + deliveryPendingSinceMs ??= Date.now() + if (deliveryTimer) { + return + } + const remainingMs = Math.max( + 0, + REMOTE_TERMINAL_DELIVERY_STALL_TIMEOUT_MS - (Date.now() - deliveryPendingSinceMs) ) + deliveryTimer = setTimeout(() => trip('delivery-credit-timeout'), remainingMs) } return { beginOutputDelivery(bytes) { outstandingDeliveryBytes += bytes - if (!deliveryTimer) { - armDeliveryTimer() - } + unacknowledgedBytes += bytes + syncDeliveryTimer() let settled = false return () => { if (settled || disposed) { @@ -78,9 +89,16 @@ export function createRemoteTerminalStreamWatchdog( } settled = true outstandingDeliveryBytes = Math.max(0, outstandingDeliveryBytes - bytes) - armDeliveryTimer() + syncDeliveryTimer() } }, + recordOutputAcknowledged(bytes) { + if (disposed) { + return + } + unacknowledgedBytes = Math.max(0, unacknowledgedBytes - bytes) + syncDeliveryTimer() + }, completeCommandResponseProbe() { commandResponseProbePending = false }, @@ -103,6 +121,8 @@ export function createRemoteTerminalStreamWatchdog( clearResponseTimer() clearDeliveryTimer() outstandingDeliveryBytes = 0 + unacknowledgedBytes = 0 + deliveryPendingSinceMs = null } } } From 278f9ee876bd0ddc9551032a3d823b04d45419fc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:53 -0700 Subject: [PATCH 119/398] fix(ssh): answer every MFA stage, stop dialling an unclaimed alias, and say where a clone failed (#17946) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): answer every MFA stage, not just the first ssh2 walks one flat auth-method list exactly once, so keyboard-interactive could only ever be offered a single time. A host running `AuthenticationMethods keyboard-interactive,keyboard-interactive` (or any ladder ending in a second challenge) partial-succeeds the first stage and then finds the list exhausted, which the user sees as "All configured authentication methods failed" — the reports in #8622 and #16820. Orca's own auth handler now runs for every target instead of only multi-key ones, and rebuilds its queue on each SSH_MSG_USERAUTH_FAILURE that carries partial success, narrowed to the methods the host still offers. Narrowing also stops keys being re-offered after the host has moved past publickey, which is what exhausts MaxAuthTries before the challenge is ever shown. Covered by a real ssh2 server fixture that stages partial success. * fix(git): say where a failing clone ran and why nothing could prompt Clones go through nonInteractiveGitEnv, so `ssh` runs with BatchMode=yes and an emptied SSH_ASKPASS. On a remote or paired-runtime clone that produces `fatal: Could not read from remote repository.` while the same `git clone` typed by hand on that box succeeds — the divergence in #14533. Nothing in the message said the clone ran on the other machine, under its keys, with the prompt deliberately disabled. getGitCloneFailureMessage now appends that fact, and names the two recognisable shapes: a publickey refusal (load the key into an agent there) and a host-key failure (record the key in that machine's known_hosts). Unrecognised SSH failures still get the where-it-ran note; non-SSH failures are untouched. One builder, so the SSH-target relay path and the runtime path both get it. * fix(ssh): stop dialling a bare alias no ssh_config block claims A wildcard `Host *` block supplies ProxyCommand/ProxyJump for every alias, so shouldUseSystemSshTransport picks the system transport for an alias whose own Host block was renamed or deleted, and buildSshArgs then dials that alias verbatim: no -l, no -p, no Hostname. Orca connects as the wildcard's user to the wildcard's host and discards the endpoint it stored (#11746). The signal #11746 assumed (hostBlockMatch, from the still-open #11707) does not exist, and `ssh -G` cannot supply it — it prints the merged config and answers for unknown aliases too. The config file is the only source of truth, so: - parseSshConfigAliasClaims retains raw Host patterns and flags Match blocks, which parseSshConfig discards because it mints importable targets. - sshConfigMayClaimAlias is sound in the negative direction only: an unreadable file, any Match block, or any non-catch-all pattern that might match all answer "claimed", so absence of evidence is never read as evidence of absence. Only a proven-unclaimed alias licenses an override. - buildSshArgs then states Hostname/Port/User, and only those: the wildcard is still the route, and -o Hostname does not change block selection, so the proxy keeps applying and %h expands to the host we mean. The verdict is injected rather than read inside buildSshArgs, so an arg builder does not answer differently per machine. Default is today's behaviour. Scoped to the system-SSH transport and the connection's own command/transport path. Port-forward processes and the ssh2 transport (#11707) are unchanged. * fix(ssh): read a negated Host group as uncertainty, and gate clone SSH guidance `Host * !prod` applies to every alias but `prod`, yet skipping both the catch-all and the `!` pattern answered "unclaimed" for `stage` — which licences overriding Hostname/Port/User against a block the user wrote. Any negation now makes the whole group uncertain; the function is only sound in the negative direction. Also require an ssh(1) diagnostic beside "could not read from remote repository" before appending the SSH clone note: git prints that same line for the HTTP remote helper, where advice about keys and agents is simply wrong. * fix(i18n): restore the activity-options key the rebase dropped * fix(i18n): union en.json with main so the rebase cannot drop keys --- src/main/ssh/ssh-config-alias-claim.test.ts | 80 ++++++ src/main/ssh/ssh-config-alias-claim.ts | 103 +++++++ src/main/ssh/ssh-config-parser.ts | 40 +++ src/main/ssh/ssh-connection.ts | 10 +- .../ssh-multi-factor-authentication.test.ts | 251 ++++++++++++++++++ .../ssh/ssh-multi-key-authentication.test.ts | 69 ++++- .../ssh/ssh-private-key-authentication.ts | 43 ++- src/main/ssh/ssh-system-fallback.test.ts | 46 ++++ src/main/ssh/system-ssh-args.ts | 41 +++ src/shared/git-clone-failure-message.test.ts | 55 ++++ src/shared/git-clone-failure-message.ts | 45 ++++ 11 files changed, 770 insertions(+), 13 deletions(-) create mode 100644 src/main/ssh/ssh-config-alias-claim.test.ts create mode 100644 src/main/ssh/ssh-config-alias-claim.ts create mode 100644 src/main/ssh/ssh-multi-factor-authentication.test.ts diff --git a/src/main/ssh/ssh-config-alias-claim.test.ts b/src/main/ssh/ssh-config-alias-claim.test.ts new file mode 100644 index 00000000000..ed6ec87971c --- /dev/null +++ b/src/main/ssh/ssh-config-alias-claim.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest' +import { parseSshConfigAliasClaims } from './ssh-config-parser' +import { sshConfigMayClaimAlias } from './ssh-config-alias-claim' + +function mayClaim(config: string, alias: string): boolean { + return sshConfigMayClaimAlias(alias, parseSshConfigAliasClaims(config)) +} + +describe('sshConfigMayClaimAlias', () => { + it('treats a wildcard-only config as proof that nothing claims the alias', () => { + const config = ` +Host * + ProxyCommand nc -X connect -x proxy:8080 %h %p + ForwardAgent yes +` + expect(mayClaim(config, 'prod')).toBe(false) + }) + + it('keeps a Host block that names the alias authoritative', () => { + const config = ` +Host * + ProxyCommand nc %h %p +Host prod + HostName prod.internal +` + expect(mayClaim(config, 'prod')).toBe(true) + }) + + it('keeps a glob that reaches the alias authoritative', () => { + // parseSshConfig drops these, which is why the claim check cannot reuse it. + const config = ` +Host prod-* + HostName prod.internal +` + expect(mayClaim(config, 'prod-web')).toBe(true) + expect(mayClaim(config, 'stage-web')).toBe(false) + }) + + it('matches single-character wildcards the way OpenSSH does', () => { + expect(mayClaim('Host prod?\n User ops\n', 'prod1')).toBe(true) + expect(mayClaim('Host prod?\n User ops\n', 'prod12')).toBe(false) + }) + + it('refuses to answer once any Match block is present', () => { + const config = ` +Host * + ProxyCommand nc %h %p +Match host prod + User ops +` + expect(mayClaim(config, 'prod')).toBe(true) + }) + + it('reads any negated group as uncertainty, because OpenSSH still applies its positives', () => { + // `Host * !prod` routes stage; answering false there would licence overriding a block the user + // wrote. The exempted alias is not worth a second matching rule to recover. + expect(mayClaim('Host * !prod\n ForwardAgent yes\n', 'stage')).toBe(true) + expect(mayClaim('Host * !prod\n ForwardAgent yes\n', 'prod')).toBe(true) + }) + + it('still proves absence when no group negates', () => { + expect(mayClaim('Host *\n ForwardAgent yes\nHost prod\n User ops\n', 'stage')).toBe(false) + }) + + it('reads an unreadable config as uncertainty, not absence', () => { + expect(sshConfigMayClaimAlias('prod', null)).toBe(true) + }) + + it('reads an empty alias as uncertainty', () => { + expect(sshConfigMayClaimAlias('', parseSshConfigAliasClaims('Host *\n'))).toBe(true) + }) +}) + +describe('parseSshConfigAliasClaims', () => { + it('retains the raw pattern groups and flags Match blocks', () => { + expect( + parseSshConfigAliasClaims('Host a b* # comment\n User x\nMatch final\n User y\n') + ).toEqual({ hostPatternGroups: [['a', 'b*']], hasMatchBlock: true }) + }) +}) diff --git a/src/main/ssh/ssh-config-alias-claim.ts b/src/main/ssh/ssh-config-alias-claim.ts new file mode 100644 index 00000000000..74a3ea0d44c --- /dev/null +++ b/src/main/ssh/ssh-config-alias-claim.ts @@ -0,0 +1,103 @@ +import { existsSync, statSync } from 'node:fs' +import { homedir } from 'node:os' +import { join } from 'node:path' +import { normalizeSshConfigAlias } from '../../shared/ssh-config-alias' +import { expandSshConfigIncludes } from './ssh-config-include-expander' +import { parseSshConfigAliasClaims, type SshConfigAliasClaims } from './ssh-config-parser' + +/** + * Whether anything in the user's ssh_config could claim this alias — i.e. whether a `Host` or + * `Match` block other than a bare catch-all applies to it. + * + * Sound in the negative direction only. `false` means the parsed config proves nothing claims the + * alias; every uncertainty (unreadable file, any `Match` block, any negated `Host` group, any + * pattern that might match) + * answers `true`, because "we could not tell" must never be read as "no block exists". Callers use + * `false` as licence to override what OpenSSH would resolve, so a wrong `false` breaks a config the + * user explicitly wrote, which is worse than the routing bug it exists to fix. + */ +export function sshConfigMayClaimAlias( + alias: string, + claims: SshConfigAliasClaims | null +): boolean { + const normalizedAlias = normalizeSshConfigAlias(alias) + if (!normalizedAlias || claims === null) { + return true + } + // A Match block's criteria (exec, originalhost, user, …) are not modelled here, and one that + // routes this alias is indistinguishable from one that does not. + if (claims.hasMatchBlock) { + return true + } + return claims.hostPatternGroups.some((patterns) => + // A negation makes the whole group uncertain: `Host * !prod` still routes every other alias, + // so skipping both the catch-all and the `!` would answer "unclaimed" for one that is claimed. + patterns.some((pattern) => pattern.startsWith('!')) + ? true + : patterns.some( + (pattern) => + !isCatchAllHostPattern(pattern) && matchesHostPattern(pattern, normalizedAlias) + ) + ) +} + +/** `Host *` — the block every alias matches, which is exactly the one that proves nothing. */ +function isCatchAllHostPattern(pattern: string): boolean { + return pattern.length > 0 && /^\*+$/.test(pattern) +} + +function matchesHostPattern(pattern: string, normalizedAlias: string): boolean { + let expression = '' + for (const character of normalizeSshConfigAlias(pattern)) { + if (character === '*') { + expression += '.*' + } else if (character === '?') { + expression += '.' + } else { + expression += character.replace(/[.+^${}()|[\]\\]/, '\\$&') + } + } + return new RegExp(`^${expression}$`).test(normalizedAlias) +} + +// Bounds how long an edit to an Included file can go unnoticed; buildSshArgs runs per remote +// command, so re-expanding Includes every time is not an option. +const CLAIM_CACHE_TTL_MS = 5_000 + +let cachedClaims: { key: string; readAt: number; claims: SshConfigAliasClaims } | null = null + +export function invalidateSshConfigAliasClaimCache(): void { + cachedClaims = null +} + +/** + * Parse of `~/.ssh/config` (Includes expanded), or null when it cannot be read. + * + * Null and empty are different answers here: an absent or unreadable file is the uncertainty case, + * while a readable file with no matching block is the proof {@link sshConfigMayClaimAlias} needs. + */ +export function loadUserSshConfigAliasClaims(): SshConfigAliasClaims | null { + const configPath = join(homedir(), '.ssh', 'config') + try { + if (!existsSync(configPath)) { + return null + } + // Why key on the root file only: an edited Include can go unnoticed, so the cache also expires. + const stats = statSync(configPath) + const key = `${stats.mtimeMs}:${stats.size}` + const now = Date.now() + if (cachedClaims?.key === key && now - cachedClaims.readAt < CLAIM_CACHE_TTL_MS) { + return cachedClaims.claims + } + const claims = parseSshConfigAliasClaims(expandSshConfigIncludes(configPath)) + cachedClaims = { key, readAt: now, claims } + return claims + } catch { + return null + } +} + +/** Convenience wrapper over the two above; used where the caller has no claims to inject. */ +export function mayUserSshConfigClaimAlias(alias: string): boolean { + return sshConfigMayClaimAlias(alias, loadUserSshConfigAliasClaims()) +} diff --git a/src/main/ssh/ssh-config-parser.ts b/src/main/ssh/ssh-config-parser.ts index 1ac6d8ed4bf..bdad3b383ab 100644 --- a/src/main/ssh/ssh-config-parser.ts +++ b/src/main/ssh/ssh-config-parser.ts @@ -145,6 +145,46 @@ function appendHosts(target: SshConfigHost[], entries: SshConfigHost[]): void { } } +/** + * Every `Host` pattern list in a config, plus whether any `Match` block is present. + * + * Deliberately raw where {@link parseSshConfig} is not: that one keeps only concrete aliases + * because it mints importable targets, so a `Host prod.*` or a `Match host prod` route is + * invisible to it. Answering "does anything in this file claim this alias?" needs those back. + */ +export type SshConfigAliasClaims = { + hostPatternGroups: string[][] + hasMatchBlock: boolean +} + +export function parseSshConfigAliasClaims(content: string): SshConfigAliasClaims { + const hostPatternGroups: string[][] = [] + let hasMatchBlock = false + + for (const rawLine of content.split('\n')) { + const line = rawLine.trim() + if (!line || line.startsWith('#')) { + continue + } + const directive = parseConfigDirective(line) + if (!directive) { + continue + } + if (directive.key === 'host') { + const patterns = splitHostPatterns(directive.rawValue) + if (patterns.length > 0) { + hostPatternGroups.push(patterns) + } + continue + } + if (directive.key === 'match') { + hasMatchBlock = true + } + } + + return { hostPatternGroups, hasMatchBlock } +} + function parseConfigDirective(line: string): { key: string; rawValue: string } | null { const match = line.match(/^([^=\s]+)(?:\s*=\s*|\s+)(.*)$/) if (!match) { diff --git a/src/main/ssh/ssh-connection.ts b/src/main/ssh/ssh-connection.ts index f51f23a3d71..98d383c5c92 100644 --- a/src/main/ssh/ssh-connection.ts +++ b/src/main/ssh/ssh-connection.ts @@ -74,6 +74,7 @@ import { isTransientReconnectError } from './ssh-reconnect-error-classification' import { SshReconnectLadder } from './ssh-reconnect-ladder' +import { mayUserSshConfigClaimAlias } from './ssh-config-alias-claim' import { getPassphrasePrivateKeyPath } from './ssh-private-key-authentication' import { requiresSystemSshForSecurityKey, @@ -108,7 +109,9 @@ type SshRemoteFileOptions = { /** Bounds the trust-source reads that run before the handshake, which nothing else times out. */ const HOST_KEY_SOURCE_READ_TIMEOUT_MS = 5_000 -const SSH_KEYBOARD_INTERACTIVE_MAX_ROUNDS = 4 +// Counts every INFO_REQUEST of the handshake, so it must cover each partial-success stage the auth +// queue will answer (MAX_PARTIAL_SUCCESS_STAGES) times the rounds a PAM stack spends per stage. +const SSH_KEYBOARD_INTERACTIVE_MAX_ROUNDS = 8 const SSH_KEYBOARD_INTERACTIVE_READY_TIMEOUT_MS = SSH_CREDENTIAL_TIMEOUT_MS + 5_000 const SSH_KEYBOARD_INTERACTIVE_MAX_PROMPTS = 8 const SSH_KEYBOARD_INTERACTIVE_TEXT_MAX = 4_096 @@ -1299,6 +1302,11 @@ export class SshConnection { if (this.systemSshResolvedConfig) { options.resolvedConfig = this.systemSshResolvedConfig } + // Why here and not inside buildSshArgs: the verdict reads ~/.ssh/config, and an arg builder + // that consults the filesystem answers differently on every machine, tests included. + if (this.target.configHost && !mayUserSshConfigClaimAlias(this.target.configHost)) { + options.aliasClaimedByConfig = false + } if (this.systemSshControlMasterDisabledForSession) { options.disableControlMaster = true } diff --git a/src/main/ssh/ssh-multi-factor-authentication.test.ts b/src/main/ssh/ssh-multi-factor-authentication.test.ts new file mode 100644 index 00000000000..275ea2e3247 --- /dev/null +++ b/src/main/ssh/ssh-multi-factor-authentication.test.ts @@ -0,0 +1,251 @@ +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + Client, + Server as Ssh2Server, + utils, + type AuthContext, + type Connection, + type KeyboardAuthContext, + type PasswordAuthContext +} from 'ssh2' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import type { SshTarget } from '../../shared/ssh-types' +import type { SshResolvedConfig } from './ssh-config-parser' +import { buildConnectConfig } from './ssh-connection-utils' + +// OpenSSH's default; a host that burns it disconnects before the MFA stage is reached. +const MAX_AUTH_TRIES = 6 +const PASSWORD = 'stage-one-password' +const PASSCODE = '123456' + +type AuthStage = 'password' | 'keyboard-interactive' + +type MfaServer = { + port: number + attempts: string[] + close: () => Promise +} + +/** An OpenSSH-style `AuthenticationMethods a,b` host: each stage partial-succeeds into the next. */ +async function startMultiFactorServer(stages: AuthStage[]): Promise { + const attempts: string[] = [] + const connections = new Set() + // Ed25519 keygen can produce an invalid 31-byte key; ECDSA points always start with 0x04. + const hostKey = utils.generateKeyPairSync('ecdsa', { bits: 256 }).private + const server = new Ssh2Server({ hostKeys: [hostKey] }, (connection) => { + connections.add(connection) + connection.on('error', () => {}) + connection.on('close', () => connections.delete(connection)) + let stage = 0 + let failures = 0 + const remaining = (): AuthStage[] => [stages[stage]!] + const fail = (context: AuthContext): void => { + failures += 1 + if (failures >= MAX_AUTH_TRIES) { + connection.end() + return + } + context.reject(remaining(), false) + } + connection.on('authentication', (context) => { + attempts.push(context.method) + if (context.method === 'none') { + context.reject(remaining(), false) + return + } + if (context.method !== stages[stage]) { + fail(context) + return + } + if (context.method === 'password') { + if ((context as PasswordAuthContext).password !== PASSWORD) { + fail(context) + return + } + stage += 1 + if (stage === stages.length) { + context.accept() + return + } + context.reject(remaining(), true) + return + } + const keyboard = context as KeyboardAuthContext + keyboard.prompt( + [{ prompt: 'Duo passcode:', echo: false }], + 'Duo two-factor login', + 'Approve the push or enter a passcode.', + (answers) => { + if (answers?.[0] !== PASSCODE) { + fail(context) + return + } + stage += 1 + if (stage === stages.length) { + context.accept() + return + } + context.reject(remaining(), true) + } + ) + }) + }) + await new Promise((resolve, reject) => { + server.once('error', reject) + server.listen(0, '127.0.0.1', () => { + server.removeListener('error', reject) + resolve() + }) + }) + const address = server.address() + if (!address || typeof address === 'string') { + throw new Error('MFA fixture did not bind a TCP port') + } + return { + port: address.port, + attempts, + close: async () => { + for (const connection of connections) { + connection.end() + } + await new Promise((resolve, reject) => { + server.close((error) => (error ? reject(error) : resolve())) + }) + } + } +} + +function makeTarget(port: number, overrides: Partial = {}): SshTarget { + return { + id: 'mfa-target', + label: 'hpc', + source: 'manual', + host: '127.0.0.1', + port, + username: 'fixture', + ...overrides + } +} + +function makeResolved(port: number, identityFile: string[]): SshResolvedConfig { + return { + hostname: '127.0.0.1', + port, + user: 'fixture', + identityFile, + identitiesOnly: true, + forwardAgent: false, + proxyUseFdpass: false, + controlMaster: 'no', + controlPersist: 'no', + userKnownHostsFiles: [], + globalKnownHostsFiles: [], + strictHostKeyChecking: 'ask', + hashKnownHosts: false, + updateHostKeys: 'no' + } +} + +/** Drives ssh2 the way SshConnection does: one credential per keyboard-interactive prompt. */ +function connectWithOrcaConfig( + target: SshTarget, + resolved: SshResolvedConfig | null, + password: string | undefined, + answers: string[] +): { ready: Promise; prompts: string[] } { + const prompts: string[] = [] + const config = buildConnectConfig(target, resolved, { + includeAgent: false, + includePrivateKey: true + }) + if (password != null) { + config.password = password + } + const ready = new Promise((resolve, reject) => { + const client = new Client() + let answerIndex = 0 + client.on('keyboard-interactive', (_name, _instructions, _lang, requested, finish) => { + for (const requestedPrompt of requested) { + prompts.push(requestedPrompt.prompt) + } + finish(requested.map(() => answers[answerIndex++] ?? '')) + }) + client.once('ready', () => { + client.end() + resolve() + }) + client.once('error', reject) + client.once('close', () => reject(new Error('SSH connection closed during authentication'))) + client.connect({ ...config, hostVerifier: () => true, readyTimeout: 10_000 }) + }) + return { ready, prompts } +} + +describe('multi-stage SSH authentication', () => { + let tempDir: string + let keyPaths: string[] + + beforeEach(() => { + tempDir = mkdtempSync(join(tmpdir(), 'orca-mfa-')) + keyPaths = ['id_a', 'id_b'].map((name) => { + const path = join(tempDir, name) + writeFileSync(path, utils.generateKeyPairSync('ecdsa', { bits: 256 }).private) + return path + }) + }) + + afterEach(() => { + rmSync(tempDir, { recursive: true, force: true }) + }) + + it('answers a keyboard-interactive stage that follows a password partial success', async () => { + const server = await startMultiFactorServer(['password', 'keyboard-interactive']) + try { + const { ready, prompts } = connectWithOrcaConfig(makeTarget(server.port), null, PASSWORD, [ + PASSCODE + ]) + + await expect(ready).resolves.toBeUndefined() + expect(prompts).toEqual(['Duo passcode:']) + } finally { + await server.close() + } + }) + + it('answers a second keyboard-interactive stage after the first partially succeeds', async () => { + const server = await startMultiFactorServer(['keyboard-interactive', 'keyboard-interactive']) + try { + const { ready, prompts } = connectWithOrcaConfig(makeTarget(server.port), null, undefined, [ + PASSCODE, + PASSCODE + ]) + + await expect(ready).resolves.toBeUndefined() + expect(prompts).toEqual(['Duo passcode:', 'Duo passcode:']) + } finally { + await server.close() + } + }) + + it('reaches the MFA stage without burning the host auth-try budget on rejected keys', async () => { + const server = await startMultiFactorServer(['password', 'keyboard-interactive']) + try { + const target = makeTarget(server.port, { source: 'ssh-config', configHost: 'hpc' }) + const { ready } = connectWithOrcaConfig( + target, + makeResolved(server.port, keyPaths), + PASSWORD, + [PASSCODE] + ) + + await expect(ready).resolves.toBeUndefined() + // After the password stage partially succeeds the host only offers keyboard-interactive; + // re-offering keys there is what exhausts MaxAuthTries on real MFA hosts. + expect(server.attempts.filter((method) => method === 'publickey')).toHaveLength(0) + } finally { + await server.close() + } + }) +}) diff --git a/src/main/ssh/ssh-multi-key-authentication.test.ts b/src/main/ssh/ssh-multi-key-authentication.test.ts index 1f17c7dae66..7cee28a6fda 100644 --- a/src/main/ssh/ssh-multi-key-authentication.test.ts +++ b/src/main/ssh/ssh-multi-key-authentication.test.ts @@ -77,6 +77,18 @@ function nextAuth( return result ?? false } +function partialSuccessAuth( + config: ConnectConfig, + authsLeft: AuthenticationType[] +): AuthenticationType | AnyAuthMethod | false { + let result: AuthenticationType | AnyAuthMethod | false | undefined + const handler = config.authHandler as AuthHandlerMiddleware + handler(authsLeft, true, (attempt) => { + result = attempt + }) + return result ?? false +} + describe('ordered SSH private-key authentication', () => { beforeEach(() => { vi.stubEnv('SSH_AUTH_SOCK', '') @@ -111,6 +123,15 @@ describe('ordered SSH private-key authentication', () => { expect(mockReadFileSync).not.toHaveBeenCalledWith('/keys/stale-imported') }) + it('still offers the ssh-agent when no readable key precedes it', () => { + vi.stubEnv('SSH_AUTH_SOCK', '/tmp/agent.sock') + const config = buildConnectConfig(makeTarget({ identityFile: undefined }), null) + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + expect(nextAuth(config, false)).toMatchObject({ type: 'agent', agent: '/tmp/agent.sock' }) + expect(nextAuth(config, false)).toBe('keyboard-interactive') + }) + it('keeps explicit manual keys and unresolved imported keys as singular overrides', () => { const manual = buildConnectConfig( makeTarget({ @@ -128,9 +149,53 @@ describe('ordered SSH private-key authentication', () => { }) expect(manual.privateKey).toEqual(Buffer.from('/keys/manual')) - expect(manual.authHandler).toBeUndefined() expect(unresolvedImport.privateKey).toEqual(Buffer.from('/keys/stale-imported')) - expect(unresolvedImport.authHandler).toBeUndefined() + // Single-key targets are still ordered by Orca's handler so an MFA host reaches + // keyboard-interactive once per stage rather than once per connection. + expect(nextAuth(manual, true)).toMatchObject({ type: 'none' }) + expect(nextAuth(manual, false)).toMatchObject({ + type: 'publickey', + key: Buffer.from('/keys/manual') + }) + expect(nextAuth(manual, false)).toBe('keyboard-interactive') + }) + + it('re-offers keyboard-interactive for each partial-success MFA stage', () => { + const config = buildConnectConfig(makeTarget(), makeResolved(), { + includeAgent: false, + includePrivateKey: true + }) + config.password = 'stage-one' + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + expect(nextAuth(config, false)).toMatchObject({ type: 'password' }) + // Partial success: the host accepted the password and now offers only the challenge. + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + }) + + it('stops re-offering methods the host no longer accepts after a partial success', () => { + const config = buildConnectConfig(makeTarget(), makeResolved(), { + includeAgent: false, + includePrivateKey: true + }) + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + expect(nextAuth(config, false)).toBe(false) + }) + + it('bounds the number of partial-success stages it will answer', () => { + const config = buildConnectConfig(makeTarget(), makeResolved(), { + includeAgent: false, + includePrivateKey: true + }) + + expect(nextAuth(config, true)).toMatchObject({ type: 'none' }) + for (let stage = 0; stage < 4; stage += 1) { + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe('keyboard-interactive') + } + expect(partialSuccessAuth(config, ['keyboard-interactive'])).toBe(false) }) it('offers every resolved key for a manually owned config-picker target', () => { diff --git a/src/main/ssh/ssh-private-key-authentication.ts b/src/main/ssh/ssh-private-key-authentication.ts index a01a58bf9b6..3daa7a8c844 100644 --- a/src/main/ssh/ssh-private-key-authentication.ts +++ b/src/main/ssh/ssh-private-key-authentication.ts @@ -3,6 +3,16 @@ import type { PrivateKeyFile } from './ssh-auth-resolution' const passphraseKeyPaths = new WeakMap() +// Bounds an `AuthenticationMethods a,b,c` ladder so a host that keeps replying +// "partial success" cannot keep the client prompting forever. +const MAX_PARTIAL_SUCCESS_STAGES = 4 + +function authMethodName(attempt: AuthenticationType | AnyAuthMethod): AuthenticationType { + const type = typeof attempt === 'string' ? attempt : attempt.type + // Agent identities are signed as publickey; a server's method list never names 'agent'. + return type === 'agent' ? 'publickey' : type +} + function buildAuthQueue( config: ConnectConfig, keys: PrivateKeyFile[] @@ -35,21 +45,34 @@ export function configurePrivateKeyAuthentication( passphraseKeyPath?: string ): void { const firstKey = keys[0] - if (!firstKey) { - return - } - config.privateKey = firstKey.contents - if (passphraseKeyPath) { - passphraseKeyPaths.set(config, passphraseKeyPath) - } - if (keys.length === 1) { - return + if (firstKey) { + config.privateKey = firstKey.contents + if (passphraseKeyPath) { + passphraseKeyPaths.set(config, passphraseKeyPath) + } } + // Why this replaces ssh2's own handler for every target, not just multi-key ones: ssh2 walks one + // flat method list exactly once, so keyboard-interactive can only ever be offered a single time. + // An MFA host running `AuthenticationMethods keyboard-interactive,keyboard-interactive` (or any + // ladder whose last stage is a second challenge) partial-succeeds the first stage and then finds + // the list exhausted — reported to the user as "All configured authentication methods failed". let queue: (AuthenticationType | AnyAuthMethod)[] = [] - config.authHandler = (authsLeft, _partialSuccess, next) => { + let partialSuccessStagesLeft = MAX_PARTIAL_SUCCESS_STAGES + config.authHandler = (authsLeft, partialSuccess, next) => { if (authsLeft == null) { queue = buildAuthQueue(config, keys) + partialSuccessStagesLeft = MAX_PARTIAL_SUCCESS_STAGES + } else if (partialSuccess && partialSuccessStagesLeft > 0) { + // A stage was accepted and the host now demands another method. Restart from a fresh queue + // narrowed to what it still offers: re-offering keys it has stopped accepting is what + // exhausts MaxAuthTries before the challenge is ever shown. + partialSuccessStagesLeft -= 1 + const offered = Array.isArray(authsLeft) ? authsLeft : [] + queue = buildAuthQueue(config, keys).filter((attempt) => { + const method = authMethodName(attempt) + return method !== 'none' && offered.includes(method) + }) } const attempt = queue.shift() next((attempt ?? false) as Parameters[0]) diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index 619efcd111c..aa59a1f2eb5 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -268,6 +268,52 @@ describe('spawnSystemSsh', () => { expect(args).not.toContain('ProxyCommand=ignored') }) + it('states the stored endpoint when no Host block claims the alias', () => { + // A wildcard `Host *` supplies the proxy for every alias, so an alias whose own block is gone + // still reads as config-backed and gets dialled bare - the #11746 P1. + const args = buildSshArgs( + createTarget({ + source: 'ssh-config', + configHost: 'prod', + host: '10.0.0.5', + port: 2222, + username: 'deploy' + }), + { aliasClaimedByConfig: false } + ) + + expect(args).toContain('Hostname=10.0.0.5') + expect(args.slice(args.indexOf('-p'))).toContain('2222') + expect(args.slice(args.indexOf('-l'))).toContain('deploy') + // The alias is still the destination so OpenSSH keeps applying the wildcard's proxy. + expect(args.at(-1)).toBe('prod') + expect(args).not.toContain('deploy@prod') + expect(args).not.toContain('-i') + expect(args).not.toContain('-J') + }) + + it('stays a no-op when the stored endpoint matches the unclaimed alias', () => { + const args = buildSshArgs( + createTarget({ source: 'ssh-config', configHost: 'prod', host: 'prod', username: '' }), + { aliasClaimedByConfig: false } + ) + + expect(args.some((arg) => arg.startsWith('Hostname='))).toBe(false) + expect(args).not.toContain('-p') + expect(args).not.toContain('-l') + expect(args.at(-1)).toBe('prod') + }) + + it('leaves a manual target alone even when nothing claims its alias', () => { + const args = buildSshArgs( + createTarget({ source: 'manual', configHost: 'prod', host: '10.0.0.5', port: 2222 }), + { aliasClaimedByConfig: false } + ) + + expect(args.some((arg) => arg.startsWith('Hostname='))).toBe(false) + expect(args).toContain('deploy@prod') + }) + it('passes an explicit main-owned OpenSSH config as one argument', () => { const args = buildSshArgs(createTarget({ configHost: 'isolated-host', source: 'ssh-config' }), { configFile: '/tmp/orca isolated/ssh_config' diff --git a/src/main/ssh/system-ssh-args.ts b/src/main/ssh/system-ssh-args.ts index 82e29d0a494..70fc4318ab6 100644 --- a/src/main/ssh/system-ssh-args.ts +++ b/src/main/ssh/system-ssh-args.ts @@ -8,6 +8,14 @@ export type SystemSshBuildArgsOptions = { suppressOrcaControlMaster?: boolean gssapiOnly?: boolean nonInteractive?: boolean + /** + * `false` only when the parsed ssh_config proves no `Host`/`Match` block claims `configHost`. + * + * Absent or `true` keeps the config alias fully authoritative, which is right whenever a block + * really does name it — and is the only safe default, since a caller that cannot answer must not + * be read as having answered "nothing claims it". See `sshConfigMayClaimAlias`. + */ + aliasClaimedByConfig?: boolean } export function buildSshArgs(target: SshTarget, options?: SystemSshBuildArgsOptions): string[] { @@ -51,6 +59,15 @@ export function buildSshArgs(target: SshTarget, options?: SystemSshBuildArgsOpti const useConfigHost = shouldUseOpenSshConfigHost(target) + // Why: a wildcard `Host *` block supplies ProxyCommand/ProxyJump for every alias, so an alias + // whose own Host block was renamed or deleted still looks config-backed and gets dialled bare — + // as the wildcard's user, at the wildcard's host, discarding the endpoint Orca stored. Keep the + // system transport (OpenSSH must still apply that proxy) but state the stored endpoint, and only + // where the config proves no block claims the alias. + if (useConfigHost && options?.aliasClaimedByConfig === false) { + appendUnclaimedAliasEndpoint(args, target) + } + if (!useConfigHost && target.port !== 22) { args.push('-p', String(target.port)) } @@ -122,6 +139,9 @@ export function getSystemSshBuildArgsFromOperationOptions( if (options?.nonInteractive === true) { buildArgsOptions.nonInteractive = true } + if (options?.aliasClaimedByConfig === false) { + buildArgsOptions.aliasClaimedByConfig = false + } return Object.keys(buildArgsOptions).length === 0 ? undefined : buildArgsOptions } @@ -172,6 +192,27 @@ function hasEnabledControlPath(value: string | undefined): boolean { return normalized != null && normalized !== '' && normalized !== 'none' } +/** + * Restore only Hostname/Port/User, and only where they diverge from the alias. + * + * Not `-i`/`-J`/ProxyCommand: the wildcard block is still the route to this network, and `-o + * Hostname=` does not change which blocks OpenSSH selects (matching uses the original destination), + * so the proxy keeps applying and `%h` now expands to the host we actually mean. + */ +function appendUnclaimedAliasEndpoint(args: string[], target: SshTarget): void { + const alias = target.configHost + const storedHost = target.host.trim() + if (storedHost && alias && storedHost !== alias) { + args.push('-o', `Hostname=${storedHost}`) + } + if (target.port && target.port !== 22) { + args.push('-p', String(target.port)) + } + if (target.username) { + args.push('-l', target.username) + } +} + function shouldUseOpenSshConfigHost(target: SshTarget): boolean { if (!target.configHost) { return false diff --git a/src/shared/git-clone-failure-message.test.ts b/src/shared/git-clone-failure-message.test.ts index 71ca28617e7..9b32b25d714 100644 --- a/src/shared/git-clone-failure-message.test.ts +++ b/src/shared/git-clone-failure-message.test.ts @@ -71,4 +71,59 @@ describe('getGitCloneFailureMessage', () => { ) expect(usedLineSplit).toBe(false) }) + + it('names the non-interactive remote clone when a key could not be offered', () => { + const stderr = + "Cloning into 'repo'...\n" + + 'git@github.com: Permission denied (publickey).\n' + + 'fatal: Could not read from remote repository.\n' + + const message = getGitCloneFailureMessage(stderr) + + expect(message).toContain('fatal: Could not read from remote repository.') + expect(message).toContain('BatchMode=yes') + expect(message).toContain('ssh-add') + }) + + it('points host key failures at the machine that runs the clone', () => { + const stderr = 'Host key verification failed.\nfatal: Could not read from remote repository.\n' + + const message = getGitCloneFailureMessage(stderr) + + expect(message).toContain('known_hosts') + expect(message).not.toContain('ssh-add') + }) + + it('still explains where an unrecognised SSH clone failure ran', () => { + const stderr = + 'kex_exchange_identification: read: Connection reset by peer\nfatal: Could not read from remote repository.\n' + + expect(getGitCloneFailureMessage(stderr)).toContain('BatchMode=yes') + }) + + it('leaves non-SSH clone failures untouched', () => { + expect( + getGitCloneFailureMessage("fatal: repository 'https://github.com/org/repo.git/' not found") + ).toBe("fatal: repository 'https://github.com/org/repo.git/' not found") + }) + + it('withholds SSH guidance when the failing transport was HTTPS', () => { + // Git reuses this line for the HTTP remote helper, so the string alone does not prove SSH. + const stderr = + 'remote: Invalid username or token.\n' + + "fatal: Authentication failed for 'https://github.com/org/repo.git/'\n" + + 'fatal: Could not read from remote repository.\n' + + expect(getGitCloneFailureMessage(stderr)).not.toContain('BatchMode=yes') + }) + + it('does not repeat the guidance when a relay message is re-parsed', () => { + const relayMessage = `Clone failed: ${getGitCloneFailureMessage( + 'git@github.com: Permission denied (publickey).\nfatal: Could not read from remote repository.\n' + )}` + + const reparsed = getGitCloneFailureMessage(relayMessage) + + expect(reparsed.match(/BatchMode=yes/g)).toHaveLength(1) + }) }) diff --git a/src/shared/git-clone-failure-message.ts b/src/shared/git-clone-failure-message.ts index 44223dad59f..e706d523e5c 100644 --- a/src/shared/git-clone-failure-message.ts +++ b/src/shared/git-clone-failure-message.ts @@ -1,8 +1,53 @@ import { stripCredentialsFromMessage } from './git-remote-error' +// Why: clones run under nonInteractiveGitEnv (GIT_TERMINAL_PROMPT=0, empty SSH_ASKPASS, +// `ssh -o BatchMode=yes`) so a background clone cannot hang on a prompt nobody sees. The cost is +// that git's own SSH errors read identically to an ordinary permission problem, and on a remote +// or paired-runtime clone the user is looking at their own working local `git clone` while Orca +// fails — with nothing in the message saying the clone ran somewhere else, without their agent. +const CLONE_HOST_NOTE = + 'The clone runs non-interactively (BatchMode=yes) on the machine that will hold the repository, using the SSH keys and agent on that machine rather than the ones on this computer.' +const CLONE_KEY_HINT = `${CLONE_HOST_NOTE} A passphrase-protected key cannot prompt there, so load it into an agent on that machine (ssh-add) and retry.` +const CLONE_HOST_KEY_HINT = `${CLONE_HOST_NOTE} It has not trusted this host key yet — connect once from a shell on that machine to record it in its known_hosts.` + +/** An ssh(1) diagnostic, i.e. a line only the SSH transport can have produced. */ +const SSH_TRANSPORT_DIAGNOSTIC = + /\bssh|permission denied \(|connection (?:closed|reset|refused|timed out) by/i + export function getGitCloneFailureMessage( stderr: string, options: { clonePath?: string | null } = {} +): string { + return appendCloneTransportGuidance( + getGitCloneFailureLine(stderr, options), + stripCredentialsFromMessage(stderr) + ) +} + +/** Guidance the raw git error omits: where the clone ran, and why nothing could prompt there. */ +function appendCloneTransportGuidance(message: string, scrubbedStderr: string): string { + // Re-entrant: remote-repo-clone re-parses a message the relay already built. + if (message.includes(CLONE_HOST_NOTE)) { + return message + } + if (/host key verification failed/i.test(scrubbedStderr)) { + return `${message} ${CLONE_HOST_KEY_HINT}` + } + if (/permission denied \(([^)]*publickey[^)]*)\)/i.test(scrubbedStderr)) { + return `${message} ${CLONE_KEY_HINT}` + } + // Every other SSH-transport failure still needs the one fact the reporter was missing — but only + // once something proves the transport was SSH: git prints this same line for the HTTP remote + // helper, where a note about keys and agents is simply wrong. + return /could not read from remote repository/i.test(scrubbedStderr) && + SSH_TRANSPORT_DIAGNOSTIC.test(scrubbedStderr) + ? `${message} ${CLONE_HOST_NOTE}` + : message +} + +function getGitCloneFailureLine( + stderr: string, + options: { clonePath?: string | null } = {} ): string { let fallbackLine: string | null = null From 074a2366bf1e50d8c6b6e45656ac6b7201fc36de Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:14:58 -0700 Subject: [PATCH 120/398] test(e2e): pass testInfo to startDockerSshRelayTarget in the freeze repro (#18257) The spec called startDockerSshRelayTarget() with no argument while the helper signature is (testInfo: TestInfo) and dereferences testInfo.workerIndex, so it threw before any Orca code ran and took the Docker SSH lane red on every PR. Fixes #16764 --- tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts index ee9f8c3845a..a7754a02507 100644 --- a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts +++ b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts @@ -55,11 +55,11 @@ test.describe('R2 Docker SSH bulk-open freeze', () => { test('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ orcaPage, registerPostElectronShutdownCleanup - }) => { + }, testInfo) => { test.setTimeout(420_000) let target: DockerSshRelayTarget | null = null try { - target = startDockerSshRelayTarget() + target = startDockerSshRelayTarget(testInfo) registerPostElectronShutdownCleanup(async () => { if (target) { cleanupDockerSshRelayTarget(target) From 7f8eb90ac3a7fa00102015f16f35dfbe923721cf Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:32:06 -0700 Subject: [PATCH 121/398] Align worktree host labels across desktop and mobile (#18237) * refactor: align worktree host labels across clients * fix(mobile): expose safe host display labels * fix(mobile): preserve legacy mixed-host labels --------- Co-authored-by: Merge Sim --- mobile/src/components/WorktreeListRow.test.ts | 41 ++++- mobile/src/components/WorktreeListRow.tsx | 37 +++- .../src/host-screen/use-host-repo-metadata.ts | 66 +++++++ .../host-screen/use-host-screen-controller.ts | 17 +- .../host-screen/use-host-screen-identity.ts | 6 + .../src/host-screen/use-host-screen-state.ts | 13 ++ mobile/src/worktree/workspace-list-types.ts | 4 + .../worktree-host-context-labels.test.ts | 164 ++++++++++++++++++ .../worktree/worktree-host-context-labels.ts | 92 ++++++++++ .../paired-settings.spec.ts | 6 + src/main/runtime/runtime-client-settings.ts | 16 +- src/main/runtime/runtime-store-contract.ts | 1 + .../worktree-list-groups-host-labels.test.ts | 47 +++++ .../worktree-list/grouping/host-labels.ts | 40 +++-- src/shared/worktree/host-context-labels.ts | 94 ++++++++++ 15 files changed, 618 insertions(+), 26 deletions(-) create mode 100644 mobile/src/worktree/worktree-host-context-labels.test.ts create mode 100644 mobile/src/worktree/worktree-host-context-labels.ts create mode 100644 src/shared/worktree/host-context-labels.ts diff --git a/mobile/src/components/WorktreeListRow.test.ts b/mobile/src/components/WorktreeListRow.test.ts index 01e0128cdc1..7b6d222d29e 100644 --- a/mobile/src/components/WorktreeListRow.test.ts +++ b/mobile/src/components/WorktreeListRow.test.ts @@ -30,7 +30,9 @@ vi.mock('lucide-react-native', () => ({ ChevronDown: 'ChevronDown', ChevronRight: 'ChevronRight', GitBranch: 'GitBranch', - GitPullRequest: 'GitPullRequest' + GitPullRequest: 'GitPullRequest', + Monitor: 'Monitor', + Server: 'Server' })) vi.mock('../platform/haptics', () => ({ triggerMediumImpact: vi.fn() })) @@ -225,4 +227,41 @@ describe('memoized worktree rows', () => { workingMode: 'monitoring' }) }) + + it('names the host with a glyph that matches the host kind', async () => { + const textNodes = (): string[] => + renderer!.root + .findAllByType('Text' as never) + .flatMap((node) => node.props.children) + .filter((child): child is string => typeof child === 'string') + + await act(async () => { + renderer = create( + createElement(ListRowHarness, { + item: { ...baseItem, hostId: 'ssh:ssh-1', hostContextLabel: 'openclaw' }, + now: 2_000 + }) + ) + }) + expect(textNodes()).toContain('openclaw') + expect(renderer!.root.findAllByType('Server' as never)).toHaveLength(1) + expect(renderer!.root.findAllByType('Monitor' as never)).toHaveLength(0) + + await act(async () => + renderer!.update( + createElement(ListRowHarness, { + item: { ...baseItem, hostContextLabel: 'Local Mac' }, + now: 2_000 + }) + ) + ) + expect(textNodes()).toContain('Local Mac') + expect(renderer!.root.findAllByType('Monitor' as never)).toHaveLength(1) + + await act(async () => + renderer!.update(createElement(ListRowHarness, { item: baseItem, now: 2_000 })) + ) + expect(textNodes()).not.toContain('Local Mac') + expect(renderer!.root.findAllByType('Monitor' as never)).toHaveLength(0) + }) }) diff --git a/mobile/src/components/WorktreeListRow.tsx b/mobile/src/components/WorktreeListRow.tsx index ba1c3865fc7..1de862e630c 100644 --- a/mobile/src/components/WorktreeListRow.tsx +++ b/mobile/src/components/WorktreeListRow.tsx @@ -1,6 +1,15 @@ import { memo } from 'react' -import { Bell, ChevronDown, ChevronRight, GitBranch, GitPullRequest } from 'lucide-react-native' +import { + Bell, + ChevronDown, + ChevronRight, + GitBranch, + GitPullRequest, + Monitor, + Server +} from 'lucide-react-native' import { Pressable, StyleSheet, Text, View } from 'react-native' +import { parseExecutionHostId, type ExecutionHostId } from '../../../src/shared/execution-host' import type { RepoIcon } from '../../../src/shared/repo-icon' import type { AgentWorkingMode } from '../../../src/shared/agent-status-types' import type { RuntimeWorktreeAgentRow } from '../../../src/shared/runtime-types' @@ -22,6 +31,11 @@ function displayBranch(branch: string): string { export type WorktreeListRowItem = { workspaceKind?: 'git' | 'folder-workspace' worktreeId: string + hostId?: ExecutionHostId + /** Present only when the list spans hosts; names the host this row runs on. */ + hostContextLabel?: string + /** Resolved host for the display label; present when legacy rows omit hostId. */ + hostContextHostId?: ExecutionHostId repo: string branch: string displayName: string @@ -150,6 +164,20 @@ function WorktreeListRowComponent({ Child )} + {item.hostContextLabel ? ( + + {/* Rows from hosts that predate hostId stamping are local: a remote row always carries one. */} + {(parseExecutionHostId(item.hostContextHostId ?? item.hostId)?.kind ?? 'local') === + 'local' ? ( + + ) : ( + + )} + + {item.hostContextLabel} + + + ) : null} {/* Repo glyph+name only when not already grouped under this repo; MobileRepoIcon falls back to a Folder (matching desktop's default) rather than a bare colored dot. */} @@ -306,6 +334,13 @@ const styles = StyleSheet.create({ fontSize: 10, color: colors.textMuted }, + hostBadge: { + flexShrink: 1, + maxWidth: 140 + }, + hostBadgeText: { + flexShrink: 1 + }, lineageToggle: { alignSelf: 'flex-start', flexDirection: 'row', diff --git a/mobile/src/host-screen/use-host-repo-metadata.ts b/mobile/src/host-screen/use-host-repo-metadata.ts index a5367a971ad..198efbaf3ea 100644 --- a/mobile/src/host-screen/use-host-repo-metadata.ts +++ b/mobile/src/host-screen/use-host-repo-metadata.ts @@ -1,13 +1,54 @@ import { useCallback } from 'react' +import { getRepoExecutionHostId } from '../../../src/shared/execution-host' import { setCachedRepos } from '../cache/repo-cache' import type { RpcClient } from '../transport/rpc-client' import type { ConnectionState, RpcSuccess } from '../transport/types' import type { RepoSummary } from '../worktree/host-worktree-rpc-types' import { repoColor } from '../worktree/repo-color' +import { + buildHostLabelById, + buildRepoHostIdByRepoId +} from '../worktree/worktree-host-context-labels' import type { HostScreenState } from './use-host-screen-state' const REPO_METADATA_REFRESH_MS = 60_000 +type SshTargetSummaryRow = { id: string; label: string } + +async function requestResult(client: RpcClient, method: string): Promise { + try { + const response = await client.sendRequest(method) + return response.ok ? (response as RpcSuccess).result : null + } catch { + // Best-effort: hosts that predate a method still list repos; labels degrade to host ids. + return null + } +} + +function readSshTargets(result: unknown): SshTargetSummaryRow[] { + const targets = (result as { targets?: unknown } | null)?.targets + if (!Array.isArray(targets)) { + return [] + } + return targets.filter( + (target): target is SshTargetSummaryRow => + typeof target === 'object' && + target !== null && + typeof (target as SshTargetSummaryRow).id === 'string' && + typeof (target as SshTargetSummaryRow).label === 'string' + ) +} + +function readHostPlatform(result: unknown): NodeJS.Platform | null { + const platform = (result as { platform?: unknown } | null)?.platform + return typeof platform === 'string' && platform ? (platform as NodeJS.Platform) : null +} + +function readHostSettingOverrides(result: unknown): unknown { + return (result as { settings?: { hostSettingOverrides?: unknown } } | null)?.settings + ?.hostSettingOverrides +} + export function useHostRepoMetadata(args: { client: RpcClient | null connState: ConnectionState @@ -20,7 +61,10 @@ export function useHostRepoMetadata(args: { fetchRepoMetadataInFlightRef, fetchRepoMetadataPendingRef, repoMetadataFetchedAtRef, + setHostLabelById, + setHostPlatform, setRepoColorsByName, + setRepoHostIdByRepoId, setRepoIconsByName, setRepoIdsByName } = state @@ -69,6 +113,28 @@ export function useHostRepoMetadata(args: { ) ) setRepoIdsByName(new Map(repoResult.repos.map((repo) => [repo.displayName, repo.id]))) + setRepoHostIdByRepoId(buildRepoHostIdByRepoId(repoResult.repos)) + // Why: rows only name their host when the list spans hosts, so a single-host + // catalog never pays for the label lookups. Counted over repos, not the id-keyed + // map: one repo id registered on two hosts is two hosts. + const hostIds = new Set(repoResult.repos.map((repo) => getRepoExecutionHostId(repo))) + if (hostIds.size > 1) { + const [sshTargets, hostSettings, hostPlatform] = await Promise.all([ + requestResult(requestClient, 'ssh.listTargetSummaries'), + requestResult(requestClient, 'settings.get'), + requestResult(requestClient, 'host.platform') + ]) + if (clientRef.current !== requestClient || hostId !== requestHostId) { + return + } + setHostLabelById( + buildHostLabelById({ + sshTargets: readSshTargets(sshTargets), + hostSettingOverrides: readHostSettingOverrides(hostSettings) + }) + ) + setHostPlatform(readHostPlatform(hostPlatform)) + } } while (fetchRepoMetadataPendingRef.current.has(requestClient)) } catch { // Repo metadata is decorative; the next refresh can retry. diff --git a/mobile/src/host-screen/use-host-screen-controller.ts b/mobile/src/host-screen/use-host-screen-controller.ts index 347fd36a72e..ad2bf9a5d77 100644 --- a/mobile/src/host-screen/use-host-screen-controller.ts +++ b/mobile/src/host-screen/use-host-screen-controller.ts @@ -14,6 +14,7 @@ import { useRelayRecoveryStatus } from '../transport/client-context-connection-metrics' import { applyWorktreeRowDisplayState } from '../worktree/worktree-host-row-identity' +import { applyWorktreeHostContextLabels } from '../worktree/worktree-host-context-labels' import { useWorkspaceSections } from '../worktree/use-workspace-sections' import { useHostRepoMetadata } from './use-host-repo-metadata' import { useHostScreenIdentity } from './use-host-screen-identity' @@ -95,17 +96,23 @@ export function useHostScreenController({ // Why: live `worktrees` is authoritative only while connected; under the amber // mount default, connecting/handshaking must keep the pre-reconnect list too. const base = connState === 'connected' ? state.worktrees : state.lastKnownWorktrees - return applyWorktreeRowDisplayState( - base, - state.sleptIds, - state.optimisticActiveWorktreeIdentity + return applyWorktreeHostContextLabels( + applyWorktreeRowDisplayState(base, state.sleptIds, state.optimisticActiveWorktreeIdentity), + { + repoHostIdByRepoId: state.repoHostIdByRepoId, + hostLabelById: state.hostLabelById, + hostPlatform: state.hostPlatform + } ) }, [ connState, state.worktrees, state.lastKnownWorktrees, state.sleptIds, - state.optimisticActiveWorktreeIdentity + state.optimisticActiveWorktreeIdentity, + state.repoHostIdByRepoId, + state.hostLabelById, + state.hostPlatform ]) const sectionsResult = useWorkspaceSections({ displayWorktrees, diff --git a/mobile/src/host-screen/use-host-screen-identity.ts b/mobile/src/host-screen/use-host-screen-identity.ts index c0b09f61714..0a45502090f 100644 --- a/mobile/src/host-screen/use-host-screen-identity.ts +++ b/mobile/src/host-screen/use-host-screen-identity.ts @@ -17,10 +17,13 @@ export function useHostScreenIdentity(args: { repoMetadataFetchedAtRef, setCatalogError, setError, + setHostLabelById, setHostName, + setHostPlatform, setLastKnownWorktrees, setPinnedIds, setRepoColorsByName, + setRepoHostIdByRepoId, setRepoIconsByName, setWorktrees, setWorktreesLoaded @@ -54,6 +57,9 @@ export function useHostScreenIdentity(args: { setError('') setRepoColorsByName(new Map()) setRepoIconsByName(new Map()) + setRepoHostIdByRepoId(new Map()) + setHostLabelById(new Map()) + setHostPlatform(null) repoMetadataFetchedAtRef.current = 0 // Why: useState initializer runs only on first mount, so re-seed the cache when Expo Router reuses this screen for a new hostId. const freshCache = hostId ? (getCachedWorktrees(hostId) as Worktree[] | null) : null diff --git a/mobile/src/host-screen/use-host-screen-state.ts b/mobile/src/host-screen/use-host-screen-state.ts index b27e15ffb91..ca6bd0d85e7 100644 --- a/mobile/src/host-screen/use-host-screen-state.ts +++ b/mobile/src/host-screen/use-host-screen-state.ts @@ -1,4 +1,5 @@ import { useRef, useState } from 'react' +import type { ExecutionHostId } from '../../../src/shared/execution-host' import type { RepoIcon } from '../../../src/shared/repo-icon' import type { WorkspaceStatusDefinition } from '../../../src/shared/worktree/types' import { getCachedWorktrees } from '../cache/worktree-cache' @@ -56,6 +57,12 @@ export function useHostScreenState(hostId: string | undefined, action: string | ) // displayName → repo id: filters key on repo id, but section headers/rows key on displayName, so bridge the two. const [repoIdsByName, setRepoIdsByName] = useState>(new Map()) + // Host-label inputs for rows: repo → host, SSH/override labels, and the host's own platform. + const [repoHostIdByRepoId, setRepoHostIdByRepoId] = useState>( + new Map() + ) + const [hostLabelById, setHostLabelById] = useState>(new Map()) + const [hostPlatform, setHostPlatform] = useState(null) const [showSortPicker, setShowSortPicker] = useState(false) const [showGroupPicker, setShowGroupPicker] = useState(false) const [showFilterModal, setShowFilterModal] = useState(false) @@ -93,13 +100,16 @@ export function useHostScreenState(hostId: string | undefined, action: string | fetchWorktreesInFlightRef, filters, groupMode, + hostLabelById, hostName, + hostPlatform, lastKnownWorktrees, newWorktreeModalRef, newWorktreeModalVisibleRef, optimisticActiveWorktreeIdentity, pinnedIds, repoColorsByName, + repoHostIdByRepoId, repoIconsByName, repoIdsByName, repoMetadataFetchedAtRef, @@ -113,11 +123,14 @@ export function useHostScreenState(hostId: string | undefined, action: string | setError, setFilters, setGroupMode, + setHostLabelById, setHostName, + setHostPlatform, setLastKnownWorktrees, setOptimisticActiveWorktreeIdentity, setPinnedIds, setRepoColorsByName, + setRepoHostIdByRepoId, setRepoIconsByName, setRepoIdsByName, setRouteActionState, diff --git a/mobile/src/worktree/workspace-list-types.ts b/mobile/src/worktree/workspace-list-types.ts index afa937f8baa..255e429e892 100644 --- a/mobile/src/worktree/workspace-list-types.ts +++ b/mobile/src/worktree/workspace-list-types.ts @@ -9,6 +9,10 @@ export type Worktree = { repoId: string hostId?: ExecutionHostId terminalPlatform?: NodeJS.Platform + /** Display-only; set when the list spans hosts, so rows say which host they run on. */ + hostContextLabel?: string + /** Resolved host for the display label; present when legacy rows omit hostId. */ + hostContextHostId?: ExecutionHostId repo: string branch: string displayName: string diff --git a/mobile/src/worktree/worktree-host-context-labels.test.ts b/mobile/src/worktree/worktree-host-context-labels.test.ts new file mode 100644 index 00000000000..a921ceebb90 --- /dev/null +++ b/mobile/src/worktree/worktree-host-context-labels.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it } from 'vitest' +import type { Worktree } from './workspace-list-types' +import { + applyWorktreeHostContextLabels, + buildHostLabelById, + buildRepoHostIdByRepoId, + getWorktreeHostContextLabels, + resolveWorktreeHostId +} from './worktree-host-context-labels' + +function worktree(overrides: Partial = {}): Worktree { + return { + workspaceKind: 'git', + worktreeId: 'repo-1::/home/me/orca', + repoId: 'repo-1', + repo: 'orca', + branch: 'main', + displayName: 'main', + path: '/home/me/orca', + liveTerminalCount: 0, + hasAttachedPty: false, + preview: '', + unread: false, + isPinned: false, + linkedPR: null, + ...overrides + } +} + +const sshHostId = 'ssh:ssh-1785104650217-eduhep' as const + +describe('buildHostLabelById', () => { + it('labels SSH targets by their registered label and lets a display override win', () => { + const labels = buildHostLabelById({ + sshTargets: [ + { id: 'ssh-1785104650217-eduhep', label: 'openclaw' }, + { id: 'ssh-blank', label: ' ' } + ], + hostSettingOverrides: { [sshHostId]: { displayLabel: 'openclaw (renamed)' } } + }) + expect(labels.get(sshHostId)).toBe('openclaw (renamed)') + expect(labels.has('ssh:ssh-blank')).toBe(false) + }) + + it('normalizes legacy raw SSH ids used by persisted display overrides', () => { + const labels = buildHostLabelById({ + sshTargets: [], + hostSettingOverrides: { 'ssh-1785104650217-eduhep': { displayLabel: 'openclaw' } } + }) + expect(labels.get(sshHostId)).toBe('openclaw') + }) + + it('accepts canonical SSH host ids from newer target-summary payloads', () => { + const labels = buildHostLabelById({ + sshTargets: [{ id: sshHostId, label: 'openclaw' }], + hostSettingOverrides: undefined + }) + expect(labels.get(sshHostId)).toBe('openclaw') + expect(labels.has('ssh:ssh:ssh-1785104650217-eduhep')).toBe(false) + }) + + it('tolerates a malformed settings payload', () => { + expect(buildHostLabelById({ sshTargets: [], hostSettingOverrides: 'nope' }).size).toBe(0) + expect(buildHostLabelById({ sshTargets: [], hostSettingOverrides: undefined }).size).toBe(0) + }) +}) + +describe('resolveWorktreeHostId', () => { + it('prefers the row host, then the repo host, then local', () => { + const repoHosts = buildRepoHostIdByRepoId([ + { id: 'repo-1', connectionId: 'ssh-1785104650217-eduhep' }, + { id: 'repo-2', executionHostId: 'runtime:env-1' }, + { id: 'repo-3' } + ]) + expect(resolveWorktreeHostId(worktree({ hostId: 'local', repoId: 'repo-1' }), repoHosts)).toBe( + 'local' + ) + expect(resolveWorktreeHostId(worktree({ repoId: 'repo-1' }), repoHosts)).toBe(sshHostId) + expect(resolveWorktreeHostId(worktree({ repoId: 'repo-2' }), repoHosts)).toBe('runtime:env-1') + expect(resolveWorktreeHostId(worktree({ repoId: 'repo-3' }), repoHosts)).toBe('local') + expect(resolveWorktreeHostId(worktree({ repoId: 'unknown' }), repoHosts)).toBe('local') + }) +}) + +describe('getWorktreeHostContextLabels', () => { + const sources = { + repoHostIdByRepoId: new Map(), + hostLabelById: new Map([[sshHostId, 'openclaw']]), + hostPlatform: 'darwin' as const + } + + it('returns nothing for a single-host list', () => { + const rows = [worktree({ hostId: 'local' }), worktree({ hostId: 'local', worktreeId: 'b' })] + expect(getWorktreeHostContextLabels(rows, sources)).toBeUndefined() + expect(applyWorktreeHostContextLabels(rows, sources)).toBe(rows) + }) + + it('names every row by host once the list spans hosts', () => { + const rows = [ + worktree({ hostId: 'local', worktreeId: 'a' }), + worktree({ hostId: sshHostId, worktreeId: 'b' }), + worktree({ hostId: 'ssh:unlabeled', worktreeId: 'c' }), + worktree({ hostId: 'runtime:env-1', worktreeId: 'd' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, sources) + expect(labeled.map((row) => row.hostContextLabel)).toEqual([ + 'Local Mac', + 'openclaw', + 'unlabeled', + 'env-1' + ]) + }) + + it('names the local host from the paired host platform, not the phone', () => { + const rows = [ + worktree({ hostId: 'local', worktreeId: 'a' }), + worktree({ hostId: sshHostId, worktreeId: 'b' }) + ] + const linux = applyWorktreeHostContextLabels(rows, { ...sources, hostPlatform: 'linux' }) + expect(linux[0].hostContextLabel).toBe('Local Linux') + const unknown = applyWorktreeHostContextLabels(rows, { ...sources, hostPlatform: null }) + expect(unknown[0].hostContextLabel).toBe('This computer') + }) + + it('keys labels by host-qualified identity so a shared id on two hosts gets two labels', () => { + const rows = [ + worktree({ hostId: 'local', worktreeId: 'same' }), + worktree({ hostId: sshHostId, worktreeId: 'same' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, sources) + expect(labeled.map((row) => row.hostContextLabel)).toEqual(['Local Mac', 'openclaw']) + }) + + it('falls back to the repo host for rows from hosts that predate hostId stamping', () => { + const rows = [ + worktree({ repoId: 'repo-local', worktreeId: 'a' }), + worktree({ repoId: 'repo-ssh', worktreeId: 'b' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, { + ...sources, + repoHostIdByRepoId: buildRepoHostIdByRepoId([ + { id: 'repo-local' }, + { id: 'repo-ssh', connectionId: 'ssh-1785104650217-eduhep' } + ]) + }) + expect(labeled.map((row) => row.hostContextLabel)).toEqual(['Local Mac', 'openclaw']) + expect(labeled.map((row) => row.hostContextHostId)).toEqual(['local', sshHostId]) + }) + + it('keeps labels distinct when legacy rows reuse an id across hosts', () => { + const rows = [ + worktree({ repoId: 'repo-local', worktreeId: 'same' }), + worktree({ repoId: 'repo-ssh', worktreeId: 'same' }) + ] + const labeled = applyWorktreeHostContextLabels(rows, { + ...sources, + repoHostIdByRepoId: buildRepoHostIdByRepoId([ + { id: 'repo-local' }, + { id: 'repo-ssh', connectionId: 'ssh-1785104650217-eduhep' } + ]) + }) + expect(labeled.map((row) => row.hostContextLabel)).toEqual(['Local Mac', 'openclaw']) + }) +}) diff --git a/mobile/src/worktree/worktree-host-context-labels.ts b/mobile/src/worktree/worktree-host-context-labels.ts new file mode 100644 index 00000000000..de33c62ca5d --- /dev/null +++ b/mobile/src/worktree/worktree-host-context-labels.ts @@ -0,0 +1,92 @@ +import { + LOCAL_EXECUTION_HOST_ID, + getRepoExecutionHostId, + normalizeExecutionHostId, + type ExecutionHostId +} from '../../../src/shared/execution-host' +import { getMixedHostContextLabels as getSharedMixedHostContextLabels } from '../../../src/shared/worktree/host-context-labels' +import { composeWorktreeHostIdentity } from '../../../src/shared/worktree/host-qualified-identity' +export { + buildHostLabelById, + getHostContextLabel +} from '../../../src/shared/worktree/host-context-labels' +import type { RepoSummary } from './host-worktree-rpc-types' +import type { Worktree } from './workspace-list-types' + +export type HostLabelSources = { + /** Host id per repo id from repo.list; rows from hosts that predate `hostId` fall back to it. */ + repoHostIdByRepoId: ReadonlyMap + /** User-facing labels for non-local hosts: SSH target labels, then per-host display overrides. */ + hostLabelById: ReadonlyMap + /** The paired host's own platform; the phone's platform must never name the desktop. */ + hostPlatform: NodeJS.Platform | null +} + +export function buildRepoHostIdByRepoId( + repos: readonly Pick[] +): Map { + return new Map(repos.map((repo) => [repo.id, getRepoExecutionHostId(repo)])) +} + +export function resolveWorktreeHostId( + worktree: Pick, + repoHostIdByRepoId: ReadonlyMap +): ExecutionHostId { + return ( + normalizeExecutionHostId(worktree.hostId) ?? + repoHostIdByRepoId.get(worktree.repoId) ?? + LOCAL_EXECUTION_HOST_ID + ) +} + +function getResolvedWorktreeRowIdentity( + worktree: Pick, + repoHostIdByRepoId: ReadonlyMap +): string { + return composeWorktreeHostIdentity( + resolveWorktreeHostId(worktree, repoHostIdByRepoId), + worktree.worktreeId + ) +} + +// Kept as a local adapter so existing mobile imports remain stable. + +/** + * Host label per row identity, only when the list spans more than one host — a single-host + * list gains nothing from a badge on every row. Mirrors the desktop sidebar's mixed-host rule. + */ +export function getWorktreeHostContextLabels( + worktrees: readonly Worktree[], + sources: HostLabelSources +): Map | undefined { + return getSharedMixedHostContextLabels(worktrees, { + getHostId: (worktree) => resolveWorktreeHostId(worktree, sources.repoHostIdByRepoId), + // Legacy hosts omit row.hostId; key by the resolved repo owner so duplicate + // worktree ids from different hosts do not overwrite each other's label. + getIdentity: (worktree) => getResolvedWorktreeRowIdentity(worktree, sources.repoHostIdByRepoId), + sources + }) +} + +export function applyWorktreeHostContextLabels( + worktrees: Worktree[], + sources: HostLabelSources +): Worktree[] { + const labels = getWorktreeHostContextLabels(worktrees, sources) + if (!labels) { + return worktrees + } + return worktrees.map((worktree) => { + const hostContextLabel = labels.get( + getResolvedWorktreeRowIdentity(worktree, sources.repoHostIdByRepoId) + ) + if (!hostContextLabel) { + return worktree + } + return { + ...worktree, + hostContextLabel, + hostContextHostId: resolveWorktreeHostId(worktree, sources.repoHostIdByRepoId) + } + }) +} diff --git a/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts b/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts index 85a6099a8b2..ac3fdd4154d 100644 --- a/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts +++ b/src/main/runtime/orca-runtime-tests/paired-settings.spec.ts @@ -23,6 +23,9 @@ describe('OrcaRuntimeService', () => { ...store, getSettings: () => ({ ...store.getSettings(), + hostSettingOverrides: { + 'ssh:target-1': { displayLabel: 'Build host', defaultWorktreeLocation: '/srv/worktrees' } + }, experimentalNewWorktreeCardStyle: true, compactWorktreeCards: true, minimaxGroupId: 'group-42', @@ -39,6 +42,9 @@ describe('OrcaRuntimeService', () => { minimaxUsageModels: 'general,abab6.5' }) expect(runtime.getClientSettings()).not.toHaveProperty('terminalQuickCommands') + expect(runtime.getClientSettings().hostSettingOverrides).toEqual({ + 'ssh:target-1': { displayLabel: 'Build host' } + }) expect(runtime.getClientTerminalQuickCommands()).toEqual(terminalQuickCommands) }) diff --git a/src/main/runtime/runtime-client-settings.ts b/src/main/runtime/runtime-client-settings.ts index 1fb815936e6..41a251ce649 100644 --- a/src/main/runtime/runtime-client-settings.ts +++ b/src/main/runtime/runtime-client-settings.ts @@ -10,6 +10,8 @@ import { } from '../../shared/terminal-quick-commands' import { haveSameDisabledTuiAgents } from '../../shared/tui-agent-selection' import type { GlobalSettings } from '../../shared/global-settings-types' +import { getHostDisplayLabelOverrides } from '../../shared/host-setting-overrides' +import type { ExecutionHostId } from '../../shared/execution-host' import type { TerminalQuickCommand } from '../../shared/terminal-quick-command-types' import { recordManagedHookInstallFailure } from '../agent-hooks/install-telemetry' import { applyAgentStatusHooksEnabled } from '../agent-hooks/managed-agent-hook-controls' @@ -37,6 +39,13 @@ export type RuntimeClientSettings = Pick< | 'artifactSharingEnabled' | 'worktreeVisibilityDefaults' | 'agentSkillSharingEnabled' +> & { + hostSettingOverrides: RuntimeHostDisplayLabelOverrides +} + +/** Safe paired projection: host labels only; filesystem defaults stay host-private. */ +export type RuntimeHostDisplayLabelOverrides = Partial< + Record > export type RuntimeClientSettingsUpdate = Pick< @@ -94,7 +103,12 @@ export class RuntimeClientSettingsController { prBotAuthorOverrides: settings.prBotAuthorOverrides ?? [], artifactSharingEnabled: isArtifactSharingEnabled(settings), worktreeVisibilityDefaults: settings.worktreeVisibilityDefaults ?? { external: 'hide' }, - agentSkillSharingEnabled: isAgentSkillSharingEnabled(settings) + agentSkillSharingEnabled: isAgentSkillSharingEnabled(settings), + hostSettingOverrides: Object.fromEntries( + [ + ...getHostDisplayLabelOverrides({ hostSettingOverrides: settings.hostSettingOverrides }) + ].map(([hostId, displayLabel]) => [hostId, { displayLabel }]) + ) as RuntimeHostDisplayLabelOverrides } } diff --git a/src/main/runtime/runtime-store-contract.ts b/src/main/runtime/runtime-store-contract.ts index aece6a8e7ad..649a8ac49b7 100644 --- a/src/main/runtime/runtime-store-contract.ts +++ b/src/main/runtime/runtime-store-contract.ts @@ -112,6 +112,7 @@ export type RuntimeStore = { terminalHiddenDeliveryGate?: GlobalSettings['terminalHiddenDeliveryGate'] terminalModelQueryAuthority?: GlobalSettings['terminalModelQueryAuthority'] worktreeVisibilityDefaults?: GlobalSettings['worktreeVisibilityDefaults'] + hostSettingOverrides?: GlobalSettings['hostSettingOverrides'] agentSkillSharingEnabled?: GlobalSettings['agentSkillSharingEnabled'] } // Why: narrow to `unknown` return so test mocks can return void without diff --git a/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts b/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts index a5f415c286a..e8cc793eeea 100644 --- a/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts +++ b/src/renderer/src/components/sidebar/worktree-list-groups-host-labels.test.ts @@ -155,6 +155,53 @@ describe('buildRows with pinned worktrees', () => { ]) }) + it('uses the registered SSH target label for openclaw rows', () => { + const sshRepo: Repo = { + ...remoteRepo, + id: 'repo-openclaw', + connectionId: 'openclaw', + executionHostId: 'ssh:openclaw' + } + const sshWorktree: Worktree = { + ...remoteWorktree, + id: 'wt-openclaw', + repoId: sshRepo.id + } + const rows = buildRows( + 'workspace-status', + [worktree, sshWorktree], + new Map([ + [repo.id, repo], + [sshRepo.id, sshRepo] + ]), + null, + new Set(), + undefined, + undefined, + undefined, + {}, + new Map([ + [worktree.id, worktree], + [sshWorktree.id, sshWorktree] + ]), + false, + undefined, + [], + new Set(), + new Map(), + new Map(), + [], + undefined, + [], + new Map([['ssh:openclaw', 'openclaw']]) + ) + + expect(rows.filter((row) => row.type === 'item')).toMatchObject([ + { worktree: { id: worktree.id }, hostContextLabel: LOCAL_HOST_LABEL }, + { worktree: { id: sshWorktree.id }, hostContextLabel: 'openclaw' } + ]) + }) + it('shows distinct Orca server names when status grouping mixes runtime hosts', () => { const firstRepo: Repo = { ...repo, diff --git a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts index e8265667cbb..d1c5c8be2d4 100644 --- a/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts +++ b/src/renderer/src/components/sidebar/worktree-list/grouping/host-labels.ts @@ -1,11 +1,14 @@ import type { Repo } from '../../../../../../shared/repo-types' import type { Worktree } from '../../../../../../shared/worktree/types' import { - getExecutionHostLabel, getRepoExecutionHostId, getWorktreeExecutionHostId } from '../../../../../../shared/execution-host' import type { ExecutionHostId } from '../../../../../../shared/execution-host' +import { + getHostContextLabel, + getMixedHostContextLabels as getSharedMixedHostContextLabels +} from '../../../../../../shared/worktree/host-context-labels' import { getWorktreeHostIdentity } from '../../../../../../shared/worktree/host-qualified-identity' import { getProjectGroupingForRepo, @@ -15,7 +18,7 @@ import { import { getFolderWorkspaceHostId } from '../../folder-workspace-host-id' import type { RenderableFolderWorkspace } from './folder-workspace-lanes' -function getRepoHostId(repoId: string, repoMap: Map): string | null { +function getRepoHostId(repoId: string, repoMap: Map): ExecutionHostId | null { const repo = repoMap.get(repoId) return repo ? getRepoExecutionHostId(repo) : null } @@ -28,14 +31,14 @@ function getRepoHostLabel( ): string | null { const setup = projectIndex?.setupByRepoId.get(repoId) if (setup) { - return hostLabelById?.get(setup.hostId) ?? getExecutionHostLabel(setup.hostId) + return getHostContextLabel(setup.hostId, { hostLabelById }) } const repo = repoMap.get(repoId) if (!repo) { return null } const hostId = getRepoExecutionHostId(repo) - return hostLabelById?.get(hostId) ?? getExecutionHostLabel(hostId) + return getHostContextLabel(hostId, { hostLabelById }) } export function getMixedHostContextLabels( @@ -45,16 +48,22 @@ export function getMixedHostContextLabels( hostLabelById: ReadonlyMap | undefined ): Map | undefined { const labelsByRepoId = new Map() - const uniqueLabels = new Set() + // Host identity, not the rendered label, determines whether rows are ambiguous: + // two hosts can intentionally share a user-facing label. + const uniqueHostIds = new Set() for (const repoId of group.repoIds) { const label = getRepoHostLabel(repoId, repoMap, projectIndex, hostLabelById) if (!label) { continue } labelsByRepoId.set(repoId, label) - uniqueLabels.add(label) + const setup = projectIndex?.setupByRepoId.get(repoId) + const hostId = setup?.hostId ?? getRepoHostId(repoId, repoMap) + if (hostId) { + uniqueHostIds.add(hostId) + } } - return uniqueLabels.size > 1 ? labelsByRepoId : undefined + return uniqueHostIds.size > 1 ? labelsByRepoId : undefined } /** @@ -128,17 +137,12 @@ export function getMixedWorktreeHostContextLabels( hostLabelById: ReadonlyMap | undefined, defaultHostId: ExecutionHostId ): Map | undefined { - const labelsByIdentity = new Map() - const uniqueHostIds = new Set() - for (const worktree of worktrees) { - const hostId = getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId) - uniqueHostIds.add(hostId) - labelsByIdentity.set( - getWorktreeHostIdentity(worktree), - hostLabelById?.get(hostId) ?? getExecutionHostLabel(hostId) - ) - } - return uniqueHostIds.size > 1 ? labelsByIdentity : undefined + return getSharedMixedHostContextLabels(worktrees, { + getHostId: (worktree) => + getWorktreeExecutionHostId(worktree, repoMap.get(worktree.repoId), defaultHostId), + getIdentity: getWorktreeHostIdentity, + sources: { hostLabelById } + }) } export function getHostWorktreeCounts( diff --git a/src/shared/worktree/host-context-labels.ts b/src/shared/worktree/host-context-labels.ts new file mode 100644 index 00000000000..4b9481c441b --- /dev/null +++ b/src/shared/worktree/host-context-labels.ts @@ -0,0 +1,94 @@ +import { + getExecutionHostLabel, + getLocalExecutionHostLabel, + normalizeExecutionHostId, + parseExecutionHostId, + toSshExecutionHostId, + type ExecutionHostId +} from '../execution-host' +import type { GlobalSettings } from '../global-settings-types' +import { getHostDisplayLabelOverrides } from '../host-setting-overrides' + +/** Inputs used by every client when spelling a host in a workspace row. */ +export type HostContextLabelSources = { + /** Explicit labels (SSH target names and per-host display overrides). */ + hostLabelById?: ReadonlyMap + /** The execution host's platform; clients must not use the device platform. */ + hostPlatform?: NodeJS.Platform | null +} + +/** Canonical user-facing host label used by desktop and mobile workspace rows. */ +export function getHostContextLabel( + hostId: ExecutionHostId, + sources: HostContextLabelSources = {} +): string { + const override = sources.hostLabelById?.get(hostId)?.trim() + if (override) { + return override + } + if (parseExecutionHostId(hostId)?.kind === 'local') { + // An explicit null means the paired host platform is unknown (mobile); an + // omitted platform means use the current process (desktop). + if (sources.hostPlatform === null) { + return 'This computer' + } + return sources.hostPlatform !== undefined + ? getLocalExecutionHostLabel(sources.hostPlatform) + : getExecutionHostLabel(hostId) + } + return getExecutionHostLabel(hostId) +} + +/** + * Build labels for SSH targets and apply persisted display-name overrides. + * Target summaries have appeared both as raw target ids and canonical `ssh:` ids + * across protocol versions, so accept either representation. + */ +export function buildHostLabelById(args: { + sshTargets: readonly { id: string; label: string }[] + hostSettingOverrides: unknown +}): Map { + const labels = new Map() + for (const target of args.sshTargets) { + const label = target.label.trim() + if (!target.id.trim() || !label) { + continue + } + const hostId = normalizeExecutionHostId(target.id) ?? toSshExecutionHostId(target.id) + if (hostId) { + labels.set(hostId, label) + } + } + const overrides = + args.hostSettingOverrides && typeof args.hostSettingOverrides === 'object' + ? getHostDisplayLabelOverrides({ + hostSettingOverrides: args.hostSettingOverrides as GlobalSettings['hostSettingOverrides'] + }) + : new Map() + for (const [hostId, label] of overrides) { + const normalized = normalizeExecutionHostId(hostId) ?? toSshExecutionHostId(hostId) + if (normalized) { + labels.set(normalized, label) + } + } + return labels +} + +/** Generic mixed-host projection shared by desktop grouping and mobile sections. */ +export function getMixedHostContextLabels( + items: readonly T[], + args: { + getHostId: (item: T) => ExecutionHostId + getIdentity: (item: T) => string + sources?: HostContextLabelSources + } +): Map | undefined { + const labelsByIdentity = new Map() + const hostIds = new Set() + for (const item of items) { + const hostId = args.getHostId(item) + hostIds.add(hostId) + labelsByIdentity.set(args.getIdentity(item), getHostContextLabel(hostId, args.sources)) + } + return hostIds.size > 1 ? labelsByIdentity : undefined +} From 510305e57434b65f69966645073552ed5985668b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:42:08 -0700 Subject: [PATCH 122/398] fix(relay): signal capacity loss instead of dropping, hanging, or truncating (#17870) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three failures with one shape: a payload past a fixed capacity was met with silence, with a wait that never ends, or with a prefix presented as a whole. **The workspace snapshot was silently dropped.** `workspace.changed` carries the tab/session list, and a snapshot past the producer frame capacity (12288 B on a Node <=21 remote) was dropped with only a relay stderr line, so the client kept a stale list forever. The relay now publishes per client and, for a client whose sink refused the frame, sends a compact `workspace.stale` marker on the control lane; the client re-reads through `workspace.get`, whose lane is budgeted in megabytes rather than in one producer frame. A new JSON-RPC notification rather than a new field on `workspace.changed`: `normalizeSnapshot(undefined, ns)` yields revision 0 and an empty session, so a Rule-1 field would make an old client replace its tab list with nothing — worse than the drop. An old client ignores the unknown method and is exactly where it is today. The marker retention/retry machinery is extracted from the `fs.changed` overflow path and shared by both. **The Windows upload hung, and the fix for it could truncate.** `#16432` was attributed to `[Console]::In.ReadToEnd()` materializing the base64 bundle. That is not what the reporter measured: he also measured `new IO.StreamReader([Console]::OpenStandardInput())` — an incremental reader — hanging at 1 MB. The limit is in the stdin the host hands PowerShell over a non-pty ssh exec, not in the string the script builds. - `uploadFileViaSystemSsh` — the user file-import path — was piping a whole file into one Windows stdin, unchunked and untimed. That is the path large files take; it now chunks into 32 KB writes and bounds each wait. - The Windows directory upload reuses that single-file path rather than repeating a weaker copy of chunk-read + write-buffer; the `ino`/`dev` TOCTOU verification comes with it. - A Windows write needing more than one exec lands on a `.orca-partial` staging path and is published by rename, so a failed chunk cannot leave a truncated artifact under the real name. `exclusive` is enforced once at the rename, not on the first chunk, where a retry met its own leftovers. - The mkdir batch reads stdin through the stream reader the reporter measured surviving 50 KB, not `[Console]::In`, which he measured wedging at that size. - `waitForChannelClose` takes an optional bound. A wedged PowerShell stays alive at idle CPU and never closes, so without one the promise is simply never settled and the caller waits forever with no error to show. **Quick Open showed a prefix as the whole workspace.** The mechanism "a full page means there is more" only works if the caller named the cap, and the failing UI named none — it hardcoded `truncated: false`. Quick Open now names `QUICK_OPEN_LISTING_MAX_RESULTS` on both the Electron IPC hop and the runtime-RPC hop (the field #17954 added to `files.listAll`), and reads a full page as truncation. The local hop honours the cap too, which it previously ignored. Rebase note on `fs.listFiles`: an earlier revision of this work also clamped the host unconditionally, and #17934 escalated an uncapped request to an explicit error. #17954 has since landed and made an oversized reply streamable, which removes the premise — the host no longer has to choose between a prefix and a refusal, so it returns the whole listing when no limit is named and only clamps a limit it was given. Keeping either would have regressed #17954 and hard-failed three in-tree callers that deliberately pass no options (`runtime-file-commands-search-runtime-files.ts:81`, `filesystem-read-handlers.ts:125`, `runtime-file-commands-constructor.ts:41`). --- .../filesystem/filesystem-search-handlers.ts | 8 +- .../ipc/remote-workspace-stale-resync.test.ts | 150 +++++++++ src/main/ipc/remote-workspace-stale-resync.ts | 62 ++++ src/main/ipc/remote-workspace.ts | 56 +++- src/main/ssh/ssh-system-fallback.test.ts | 41 ++- .../ssh/system-ssh-file-binary-transfer.ts | 158 +++++++++- src/main/ssh/system-ssh-file-transfer.ts | 154 +++++----- .../ssh/system-ssh-operation-lifecycle.ts | 28 +- .../ssh/system-ssh-windows-upload.test.ts | 288 ++++++++++++++++++ ...fs-handler-list-files-result-limit.test.ts | 116 +++++++ src/relay/fs-handler.ts | 6 +- src/relay/relay-client-resync-marker.ts | 168 ++++++++++ src/relay/relay-watcher-event-emitter.ts | 147 +-------- src/relay/workspace-session-handler.ts | 15 +- .../workspace-snapshot-publication.test.ts | 166 ++++++++++ src/relay/workspace-snapshot-publication.ts | 49 +++ .../quick-open-file-list.react.test.tsx | 31 ++ .../src/components/quick-open-file-list.ts | 10 +- .../src/runtime/runtime-file-search-client.ts | 11 +- src/shared/remote-workspace-types.ts | 13 + 20 files changed, 1404 insertions(+), 273 deletions(-) create mode 100644 src/main/ipc/remote-workspace-stale-resync.test.ts create mode 100644 src/main/ipc/remote-workspace-stale-resync.ts create mode 100644 src/main/ssh/system-ssh-windows-upload.test.ts create mode 100644 src/relay/fs-handler-list-files-result-limit.test.ts create mode 100644 src/relay/relay-client-resync-marker.ts create mode 100644 src/relay/workspace-snapshot-publication.test.ts create mode 100644 src/relay/workspace-snapshot-publication.ts diff --git a/src/main/ipc/filesystem/filesystem-search-handlers.ts b/src/main/ipc/filesystem/filesystem-search-handlers.ts index f6dd8d57284..77a26c1e58d 100644 --- a/src/main/ipc/filesystem/filesystem-search-handlers.ts +++ b/src/main/ipc/filesystem/filesystem-search-handlers.ts @@ -223,7 +223,13 @@ export function registerFilesystemSearchHandlers(context: FilesystemHandlerConte signal: controller?.signal }) } - return await listQuickOpenFiles(args.rootPath, store, args.excludePaths, controller?.signal) + return await listQuickOpenFiles( + args.rootPath, + store, + args.excludePaths, + controller?.signal, + args.maxResults + ) } finally { listFilesCancellations.finish(event, args.requestToken, controller) } diff --git a/src/main/ipc/remote-workspace-stale-resync.test.ts b/src/main/ipc/remote-workspace-stale-resync.test.ts new file mode 100644 index 00000000000..a75e6d81971 --- /dev/null +++ b/src/main/ipc/remote-workspace-stale-resync.test.ts @@ -0,0 +1,150 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Store } from '../persistence' +import { + REMOTE_WORKSPACE_STALE_NOTIFICATION, + type RemoteWorkspaceChangedEvent, + type RemoteWorkspaceSession +} from '../../shared/remote-workspace-types' + +const { getActiveMultiplexerMock, getSshConnectionStoreMock } = vi.hoisted(() => ({ + getActiveMultiplexerMock: vi.fn(), + getSshConnectionStoreMock: vi.fn() +})) + +vi.mock('electron', () => ({ + ipcMain: { handle: vi.fn(), removeHandler: vi.fn() } +})) + +vi.mock('./ssh', () => ({ + getActiveMultiplexer: getActiveMultiplexerMock, + getSshConnectionStore: getSshConnectionStoreMock +})) + +vi.mock('./remote-workspace-events', () => ({ + registerRemoteWorkspaceNotificationHandler: vi.fn(() => vi.fn()) +})) + +import { + _resetRemoteWorkspaceCachesForTests, + handleRemoteWorkspaceNotification, + registerRemoteWorkspaceHandlers +} from './remote-workspace' + +function session(activeTabId: string): RemoteWorkspaceSession { + return { + activeWorktreePath: '/remote/worktree', + activeTabId, + tabsByWorktreePath: { + '/remote/worktree': [{ id: activeTabId, worktreePath: '/remote/worktree' } as never] + }, + terminalLayoutsByTabId: {} + } +} + +describe('workspace.stale resync', () => { + const sent: RemoteWorkspaceChangedEvent[] = [] + const request = vi.fn() + const store = { getRepo: vi.fn(), getWorkspaceSession: vi.fn() } as unknown as Store + + beforeEach(() => { + sent.length = 0 + request.mockReset() + _resetRemoteWorkspaceCachesForTests() + getActiveMultiplexerMock.mockReset() + getActiveMultiplexerMock.mockImplementation(() => ({ request })) + getSshConnectionStoreMock.mockReset() + getSshConnectionStoreMock.mockImplementation(() => ({ + getTarget: (id: string) => ({ id, host: 'example.test', username: 'dev' }), + listTargets: () => [] + })) + const win = { + isDestroyed: () => false, + webContents: { + send: (_channel: string, event: RemoteWorkspaceChangedEvent) => sent.push(event) + } + } + registerRemoteWorkspaceHandlers(store, () => win as never) + }) + + it('re-reads the snapshot through workspace.get and publishes it to the renderer', async () => { + request.mockResolvedValue({ + namespace: 'target-1', + revision: 12, + updatedAt: 5, + schemaVersion: 1, + session: session('tab-from-other-device') + }) + + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + await vi.waitFor(() => expect(sent).toHaveLength(1)) + + expect(request).toHaveBeenCalledWith('workspace.get', { namespace: expect.any(String) }) + expect(sent[0].targetId).toBe('target-1') + expect(sent[0].snapshot.revision).toBe(12) + expect(sent[0].snapshot.session.activeTabId).toBe('tab-from-other-device') + // The marker names no author, so the renderer's own-echo filter must not discard the resync. + expect(sent[0].sourceClientId).toBeUndefined() + }) + + it('collapses a burst of markers into one extra read rather than one read per marker', async () => { + const released: ((value: unknown) => void)[] = [] + request.mockImplementation( + () => + new Promise((resolve) => { + released.push(resolve) + }) + ) + + for (let i = 0; i < 4; i++) { + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + } + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(1)) + + request.mockResolvedValue({ + namespace: 'target-1', + revision: 3, + updatedAt: 1, + schemaVersion: 1, + session: session('tab-a') + }) + released[0]?.({ + namespace: 'target-1', + revision: 2, + updatedAt: 1, + schemaVersion: 1, + session: session('tab-a') + }) + + // Exactly one follow-up read for the markers that landed mid-flight: never zero, never four. + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(2)) + await Promise.resolve() + expect(request).toHaveBeenCalledTimes(2) + }) + + it('stays silent when the re-read finds the session it already had', async () => { + request.mockResolvedValue({ + namespace: 'target-1', + revision: 4, + updatedAt: 1, + schemaVersion: 1, + session: session('tab-a') + }) + + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(1)) + await vi.waitFor(() => expect(sent).toHaveLength(1)) + + handleRemoteWorkspaceNotification('target-1', REMOTE_WORKSPACE_STALE_NOTIFICATION, { + namespace: 'target-1' + }) + await vi.waitFor(() => expect(request).toHaveBeenCalledTimes(2)) + await Promise.resolve() + expect(sent).toHaveLength(1) + }) +}) diff --git a/src/main/ipc/remote-workspace-stale-resync.ts b/src/main/ipc/remote-workspace-stale-resync.ts new file mode 100644 index 00000000000..38c96977e97 --- /dev/null +++ b/src/main/ipc/remote-workspace-stale-resync.ts @@ -0,0 +1,62 @@ +import type { RemoteWorkspaceObservedSnapshot } from '../../shared/remote-workspace-types' +import type { SshTarget } from '../../shared/ssh-types' +import { getRemoteSnapshot } from './remote-workspace-relay-sync' +import { getCachedRemoteWorkspaceSnapshot } from './remote-workspace-snapshot-cache' +import { remoteWorkspaceSessionMatchesSnapshot } from './remote-workspace-snapshot-normalization' + +type PendingResync = { promise: Promise; requeued: boolean } + +const pendingByTargetId = new Map() + +export function _resetRemoteWorkspaceStaleResyncForTests(): void { + pendingByTargetId.clear() +} + +export function isRemoteWorkspaceResyncInFlight(targetId: string): boolean { + return pendingByTargetId.has(targetId) +} + +/** + * The relay told us it could not deliver a snapshot, so pull it. `workspace.get` is a response, and + * responses are admitted against the megabyte-scale control/legacy-response budget rather than the + * single ~12KB producer frame that refused the broadcast — the payload was never too big for the + * link, only for that one lane. + */ +export function resyncStaleRemoteWorkspace( + target: SshTarget, + deliver: (snapshot: RemoteWorkspaceObservedSnapshot) => void, + onError: (error: unknown) => void = () => {} +): Promise { + const existing = pendingByTargetId.get(target.id) + if (existing) { + // Why: a burst of markers must collapse to one extra read, but never to zero — a marker that + // arrived while a read was already in flight may describe a revision that read did not see. + existing.requeued = true + return existing.promise + } + const pending: PendingResync = { requeued: false, promise: Promise.resolve() } + pending.promise = (async () => { + try { + do { + pending.requeued = false + const previous = getCachedRemoteWorkspaceSnapshot(target.id) + const snapshot = await getRemoteSnapshot(target) + if (!snapshot) { + return + } + // Suppress the echo: our own patch response already cached this session, and re-publishing it + // makes the renderer rehydrate a state it authored. + if (remoteWorkspaceSessionMatchesSnapshot(previous, snapshot.session)) { + continue + } + deliver(snapshot) + } while (pending.requeued) + } catch (error) { + onError(error) + } finally { + pendingByTargetId.delete(target.id) + } + })() + pendingByTargetId.set(target.id, pending) + return pending.promise +} diff --git a/src/main/ipc/remote-workspace.ts b/src/main/ipc/remote-workspace.ts index 6a6acec6adc..74339e8f694 100644 --- a/src/main/ipc/remote-workspace.ts +++ b/src/main/ipc/remote-workspace.ts @@ -2,10 +2,13 @@ import { ipcMain, type BrowserWindow } from 'electron' import type { Store } from '../persistence' import { getActiveMultiplexer, getSshConnectionStore } from './ssh' import { exportRemoteWorkspaceSession } from '../../shared/remote-workspace-session-projection' -import type { - RemoteWorkspaceChangedEvent, - RemoteWorkspaceObservedPatchResult, - RemoteWorkspaceSession +import { + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + REMOTE_WORKSPACE_STALE_NOTIFICATION, + type RemoteWorkspaceChangedEvent, + type RemoteWorkspaceObservedPatchResult, + type RemoteWorkspaceObservedSnapshot, + type RemoteWorkspaceSession } from '../../shared/remote-workspace-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' @@ -29,6 +32,10 @@ import { rememberRemoteWorkspaceSnapshot } from './remote-workspace-snapshot-cache' import { normalizeSnapshot } from './remote-workspace-snapshot-normalization' +import { + _resetRemoteWorkspaceStaleResyncForTests, + resyncStaleRemoteWorkspace +} from './remote-workspace-stale-resync' let mainWindowGetter: (() => BrowserWindow | null) | null = null let unregisterRemoteWorkspaceNotifications: (() => void) | null = null @@ -36,6 +43,7 @@ let unregisterRemoteWorkspaceNotifications: (() => void) | null = null export function _resetRemoteWorkspaceCachesForTests(): void { clearRemoteWorkspaceSnapshotCache() clearRemoteWorkspacePatchTails() + _resetRemoteWorkspaceStaleResyncForTests() } export function _getRemoteWorkspaceCacheSizesForTests(): { @@ -119,12 +127,40 @@ function exportSessionForTarget( }) } +function sendRemoteWorkspaceChanged( + targetId: string, + snapshot: RemoteWorkspaceObservedSnapshot, + sourceClientId: string | undefined +): void { + const event: RemoteWorkspaceChangedEvent = { + targetId, + snapshot, + ...(sourceClientId !== undefined ? { sourceClientId } : {}) + } + const win = mainWindowGetter?.() + if (win && !win.isDestroyed()) { + win.webContents.send('remoteWorkspace:changed', event) + } +} + export function handleRemoteWorkspaceNotification( targetId: string, method: string, params: Record ): void { - if (method !== 'workspace.changed') { + if (method === REMOTE_WORKSPACE_STALE_NOTIFICATION) { + const target = getSshConnectionStore()?.getTarget(targetId) + if (!target) { + return + } + // No sourceClientId on the resynced event: the marker names no author, and guessing one would + // let the renderer's own-echo filter discard another device's change. + void resyncStaleRemoteWorkspace(target, (snapshot) => + sendRemoteWorkspaceChanged(targetId, snapshot, undefined) + ) + return + } + if (method !== REMOTE_WORKSPACE_CHANGED_NOTIFICATION) { return } const target = getSshConnectionStore()?.getTarget(targetId) @@ -139,15 +175,7 @@ export function handleRemoteWorkspaceNotification( sourceClientId === CLIENT_ID ? rememberLocallyPatchedRemoteWorkspaceSnapshot(targetId, snapshot) : rememberRemoteWorkspaceSnapshot(targetId, snapshot) - const event: RemoteWorkspaceChangedEvent = { - targetId, - snapshot: observedSnapshot, - sourceClientId - } - const win = mainWindowGetter?.() - if (win && !win.isDestroyed()) { - win.webContents.send('remoteWorkspace:changed', event) - } + sendRemoteWorkspaceChanged(targetId, observedSnapshot, sourceClientId) } export function registerRemoteWorkspaceHandlers( diff --git a/src/main/ssh/ssh-system-fallback.test.ts b/src/main/ssh/ssh-system-fallback.test.ts index aa59a1f2eb5..c366e899bf6 100644 --- a/src/main/ssh/ssh-system-fallback.test.ts +++ b/src/main/ssh/ssh-system-fallback.test.ts @@ -105,6 +105,13 @@ type EventedProcess = EventEmitter & { killed: boolean } +// Windows writes read their source asynchronously before spawning, so a close emitted straight +// after the call can beat the listener. Emit it from the spawn instead. +function closeOnceSpawned(proc: EventedProcess): EventedProcess { + setImmediate(() => proc.emit('close', 0, null)) + return proc +} + function createEventedProcess(): EventedProcess { const proc = new EventEmitter() as EventedProcess proc.stdin = Object.assign(new EventEmitter(), { @@ -704,7 +711,7 @@ describe('spawnSystemSsh', () => { it('writes files to Windows system SSH targets with PowerShell stdin bytes', async () => { const proc = createEventedProcess() - spawnMock.mockReturnValue(proc) + spawnMock.mockImplementation(() => closeOnceSpawned(proc)) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -713,7 +720,6 @@ describe('spawnSystemSsh', () => { '0.1.0', { hostPlatform } ) - proc.emit('close', 0, null) await expect(promise).resolves.toBeUndefined() const args = spawnMock.mock.calls[0][1] as string[] @@ -725,7 +731,7 @@ describe('spawnSystemSsh', () => { it('writes binary buffers to Windows system SSH targets with CreateNew mode', async () => { const proc = createEventedProcess() - spawnMock.mockReturnValue(proc) + spawnMock.mockImplementation(() => closeOnceSpawned(proc)) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeBufferViaSystemSsh( @@ -734,7 +740,6 @@ describe('spawnSystemSsh', () => { Buffer.from('png'), { hostPlatform, exclusive: true } ) - proc.emit('close', 0, null) await expect(promise).resolves.toBeUndefined() const args = spawnMock.mock.calls[0][1] as string[] @@ -775,7 +780,7 @@ describe('spawnSystemSsh', () => { it('forces standalone SSH for Windows file writes when requested', async () => { const proc = createEventedProcess() - spawnMock.mockReturnValue(proc) + spawnMock.mockImplementation(() => closeOnceSpawned(proc)) const hostPlatform = getRemoteHostPlatform('win32-x64') const promise = writeFileViaSystemSsh( @@ -784,7 +789,6 @@ describe('spawnSystemSsh', () => { '0.1.0', { hostPlatform, disableControlMaster: true } ) - proc.emit('close', 0, null) await expect(promise).resolves.toBeUndefined() const args = spawnMock.mock.calls[0][1] as string[] @@ -793,7 +797,7 @@ describe('spawnSystemSsh', () => { expect(args[standaloneControlIdx + 1]).toBe('none') }) - it('uploads directories to Windows system SSH targets in one PowerShell batch', async () => { + it('uploads a Windows directory as a mkdir batch plus per-file writes, never one blob', async () => { const localDir = mkdtempSync(join(tmpdir(), 'orca-system-ssh-upload-')) writeFileSync(join(localDir, 'relay.js'), 'console.log("relay")') const spawned: EventedProcess[] = [] @@ -816,24 +820,17 @@ describe('spawnSystemSsh', () => { } const commands = spawnMock.mock.calls.map((call) => (call[1] as string[]).at(-1) ?? '') - expect(commands).toHaveLength(1) + // #16432: directories first (metadata only), then the file bytes on their own stdin. One batch + // meant base64-ing the whole bundle into a single PowerShell string, which the remote never read. + expect(commands).toHaveLength(2) expect(commands.every((command) => command.includes('powershell.exe'))).toBe(true) expect(commands.every((command) => !command.includes('/bin/sh'))).toBe(true) expect(commands.join('\n')).not.toContain('tar -xzf') - const payload = JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string) as { - kind: string - path: string - contentsBase64?: string - }[] - expect(payload).toEqual( - expect.arrayContaining([ - { kind: 'directory', path: 'C:/Users/me/.orca-remote/relay' }, - { - kind: 'file', - path: 'C:/Users/me/.orca-remote/relay/relay.js', - contentsBase64: Buffer.from('console.log("relay")').toString('base64') - } - ]) + expect(JSON.parse(spawned[0].stdin.end.mock.calls[0]?.[0] as string)).toEqual([ + 'C:/Users/me/.orca-remote/relay' + ]) + expect(Buffer.from(spawned[1].stdin.end.mock.calls[0]?.[0] as Buffer).toString('utf-8')).toBe( + 'console.log("relay")' ) }) diff --git a/src/main/ssh/system-ssh-file-binary-transfer.ts b/src/main/ssh/system-ssh-file-binary-transfer.ts index e86aa3c9ae0..b0c5b662ed1 100644 --- a/src/main/ssh/system-ssh-file-binary-transfer.ts +++ b/src/main/ssh/system-ssh-file-binary-transfer.ts @@ -74,7 +74,14 @@ export async function writeBufferViaSystemSsh( ): Promise { throwIfAborted(options?.signal) if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { - await writeBufferViaSystemSshWindows(target, remotePath, contents, options) + await writeWindowsBytesViaSystemSsh( + target, + remotePath, + contents.length, + (offset, maxBytes) => + Promise.resolve(contents.subarray(offset, Math.min(offset + maxBytes, contents.length))), + options + ) return } @@ -119,16 +126,28 @@ export async function uploadFileViaSystemSsh( } throwIfAborted(options?.signal) - const isWindows = options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform) + if (options?.hostPlatform && isWindowsRemoteHost(options.hostPlatform)) { + // #16432: a Windows host cannot take a whole file through one stdin, however the local side + // paces it — see WINDOWS_STDIN_WRITE_CHUNK_BYTES. This is the path that carries the large + // files, so it is the one that has to be chunked and bounded. + await writeWindowsBytesViaSystemSsh( + target, + remotePath, + openedStat.size, + async (offset, maxBytes) => { + const buffer = Buffer.allocUnsafe(Math.min(maxBytes, openedStat.size - offset)) + const { bytesRead } = await handle.read(buffer, 0, buffer.length, offset) + return buffer.subarray(0, bytesRead) + }, + options + ) + return + } + const channel = spawnSystemSshCommand( target, - isWindows - ? makeWindowsWriteFileCommand(remotePath, options) - : makePosixWriteFileCommand(remotePath, options), - { - wrapCommand: !isWindows, - ...getSystemSshBuildArgsFromOperationOptions(options) - } + makePosixWriteFileCommand(remotePath, options), + getSystemSshBuildArgsFromOperationOptions(options) ) const input = handle.createReadStream({ autoClose: false }) try { @@ -153,11 +172,79 @@ export async function uploadFileViaSystemSsh( } } -async function writeBufferViaSystemSshWindows( +/** + * #16432: Windows PowerShell 5.1 stops draining a redirected stdin over a non-pty ssh exec + * somewhere between 50KB and 1MB, depending on the host's `DefaultShell`, and it hangs rather than + * failing. The reporter measured that on both constructs he tried — `[Console]::In.ReadToEnd()` and + * `new IO.StreamReader([Console]::OpenStandardInput())`, the latter reading incrementally, which is + * why the limit cannot be attributed to materializing the payload. `Stream.CopyTo` reads the same + * `[Console]::OpenStandardInput()` object with the same incremental `Read` loop, so nothing in it + * escapes that limit either: no single write may exceed what one stdin is known to carry. + * + * 32KB is an order of magnitude under the low end of the measured range, and under 50KB, which the + * reporter measured succeeding against a stream reader on the worse of the two `DefaultShell` + * settings. + */ +export const WINDOWS_STDIN_WRITE_CHUNK_BYTES = 32 * 1024 + +/** No Windows stdin write should ever outlive this; a wedged PowerShell never closes on its own. */ +export const WINDOWS_STDIN_WRITE_TIMEOUT_MS = 60_000 + +/** Suffix for the path a multi-exec Windows write lands on before it is published by rename. */ +export const WINDOWS_STAGED_WRITE_SUFFIX = '.orca-partial' + +/** + * Splits one logical Windows write into stdin-sized execs. + * + * A write that needs more than one exec cannot land on the destination directly: a chunk failing + * mid-file would leave a truncated artifact under the real name with nothing marking it incomplete, + * and the retry would then meet its own leftovers — under `exclusive` the retry's `CreateNew` fails + * on them. Multi-exec creates therefore land on a staging path and are published by a rename, which + * is also where `exclusive` is enforced: once, at the destination, instead of smeared across the + * first chunk. A caller-requested append cannot be staged without reading the remote file back, so + * it keeps writing straight through, as its own protocol already implies. + */ +async function writeWindowsBytesViaSystemSsh( target: SshTarget, remotePath: string, - contents: Buffer, + totalBytes: number, + readChunk: (offset: number, maxBytes: number) => Promise, options: SystemSshWriteBufferOptions +): Promise { + throwIfAborted(options.signal) + const staged = !options.append && totalBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES + const writePath = staged ? `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` : remotePath + let offset = 0 + // An empty write still has to run: it is what creates (or truncates) the file. + do { + const chunk = await readChunk(offset, WINDOWS_STDIN_WRITE_CHUNK_BYTES) + if (chunk.length === 0 && offset < totalBytes) { + throw new Error(`Source ran short during upload of ${remotePath}`) + } + await writeWindowsChunkViaSystemSsh( + target, + writePath, + chunk, + { + ...options, + append: staged ? offset > 0 : options.append === true || offset > 0, + exclusive: staged ? false : options.exclusive === true && offset === 0 + }, + offset + ) + offset += chunk.length + } while (offset < totalBytes) + if (staged) { + await publishWindowsStagedWrite(target, writePath, remotePath, options) + } +} + +async function writeWindowsChunkViaSystemSsh( + target: SshTarget, + remotePath: string, + chunk: Buffer, + options: SystemSshWriteBufferOptions, + offset: number ): Promise { throwIfAborted(options.signal) const channel = spawnSystemSshCommand(target, makeWindowsWriteFileCommand(remotePath, options), { @@ -167,10 +254,37 @@ async function writeBufferViaSystemSshWindows( const closePromise = awaitWithSystemSshAbort( options.signal, () => channel.close(), - waitForChannelClose(channel, `write ${remotePath}`) + waitForChannelClose( + channel, + `write ${remotePath} at offset ${offset}`, + WINDOWS_STDIN_WRITE_TIMEOUT_MS + ) ) if (!options.signal?.aborted) { - channel.stdin.end(contents) + channel.stdin.end(chunk) + } + await closePromise +} + +async function publishWindowsStagedWrite( + target: SshTarget, + stagingPath: string, + remotePath: string, + options: SystemSshWriteBufferOptions +): Promise { + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand( + target, + makeWindowsPublishStagedFileCommand(stagingPath, remotePath, options.exclusive === true), + { wrapCommand: false, ...getSystemSshBuildArgsFromOperationOptions(options) } + ) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, `publish ${remotePath}`, WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end() } await closePromise } @@ -193,6 +307,24 @@ function makeWindowsWriteFileCommand( ) } +// `File::Move` throws when the destination exists, which is exactly the exclusive contract; the +// non-exclusive caller asked to replace, so it deletes first (a no-op on an absent path). +function makeWindowsPublishStagedFileCommand( + stagingPath: string, + remotePath: string, + exclusive: boolean +): string { + return powerShellCommand( + [ + '$ErrorActionPreference = "Stop"', + `$staging = ${powerShellLiteral(stagingPath)}`, + `$path = ${powerShellLiteral(remotePath)}`, + ...(exclusive ? [] : ['[System.IO.File]::Delete($path)']), + '[System.IO.File]::Move($staging, $path)' + ].join('; ') + ) +} + function makePosixWriteFileCommand( remotePath: string, options?: { append?: boolean; exclusive?: boolean } diff --git a/src/main/ssh/system-ssh-file-transfer.ts b/src/main/ssh/system-ssh-file-transfer.ts index 72cb4425d11..f728c0eab02 100644 --- a/src/main/ssh/system-ssh-file-transfer.ts +++ b/src/main/ssh/system-ssh-file-transfer.ts @@ -1,6 +1,5 @@ import { spawn } from 'node:child_process' -import { constants } from 'node:fs' -import { lstat, open, readdir } from 'node:fs/promises' +import { lstat, readdir } from 'node:fs/promises' import { join as pathJoin } from 'node:path' import { pipeline } from 'node:stream/promises' import type { SshTarget } from '../../shared/ssh-types' @@ -22,7 +21,12 @@ import { waitForProcess, type ProcessResult } from './system-ssh-operation-lifecycle' -import { writeBufferViaSystemSsh } from './system-ssh-file-binary-transfer' +import { + uploadFileViaSystemSsh, + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS, + writeBufferViaSystemSsh +} from './system-ssh-file-binary-transfer' type SystemSshOperationOptions = SystemSshBuildArgsOptions & { signal?: AbortSignal @@ -109,33 +113,36 @@ async function uploadDirectoryViaSystemSshWindows( if (!hostPlatform) { throw new Error('Windows system SSH upload requires a remote host platform') } - const entries = await collectWindowsUploadEntries( - localDir, - remoteDir, - hostPlatform, - options.signal - ) - await writeWindowsUploadPackageViaSystemSsh(target, entries, options) + const plan = await collectWindowsUploadPlan(localDir, remoteDir, hostPlatform, options.signal) + await createWindowsUploadDirectories(target, plan.directories, options) + for (const file of plan.files) { + throwIfAborted(options.signal) + // Reuses the single-file upload: it already opens O_NOFOLLOW, verifies the source did not + // change under it, and splits the bytes into stdin-sized writes staged under a partial name. + await uploadFileViaSystemSsh(target, file.localPath, file.remotePath, options) + } } -type WindowsUploadEntry = - | { - kind: 'directory' - path: string - } - | { - kind: 'file' - path: string - contentsBase64: string - } +type WindowsUploadPlan = { + directories: string[] + files: { localPath: string; remotePath: string }[] +} -async function collectWindowsUploadEntries( +/** + * #16432: this used to base64 every artifact into one JSON array and push the whole ~1.9MB string + * into one PowerShell stdin. Base64 inflates the payload 1.33x, and Windows PowerShell 5.1 cannot + * read a stdin that large over a non-pty ssh exec — it blocks forever instead of failing. Nothing + * about a directory upload requires one frame: the plan carries paths only, and the bytes go per + * file, in writes bounded by WINDOWS_STDIN_WRITE_CHUNK_BYTES. + */ +async function collectWindowsUploadPlan( localDir: string, remoteDir: string, hostPlatform: RemoteHostPlatform, - signal: AbortSignal | undefined -): Promise { - const entries: WindowsUploadEntry[] = [{ kind: 'directory', path: remoteDir }] + signal: AbortSignal | undefined, + plan: WindowsUploadPlan = { directories: [], files: [] } +): Promise { + plan.directories.push(remoteDir) const dirEntries = await readdir(localDir, { withFileTypes: true }) for (const entry of dirEntries) { throwIfAborted(signal) @@ -146,75 +153,68 @@ async function collectWindowsUploadEntries( continue } if (statResult.isDirectory()) { - entries.push( - ...(await collectWindowsUploadEntries(localPath, remotePath, hostPlatform, signal)) - ) + await collectWindowsUploadPlan(localPath, remotePath, hostPlatform, signal, plan) continue } - const buffer = await readLocalUploadFile(localPath, statResult) - entries.push({ kind: 'file', path: remotePath, contentsBase64: buffer.toString('base64') }) + plan.files.push({ localPath, remotePath }) } - return entries + return plan } -async function writeWindowsUploadPackageViaSystemSsh( +// Why the JSON envelope survives here: a path list is metadata, so this payload stays in the +// hundreds of bytes even for a deep tree. Batched anyway, so a pathological tree cannot walk back +// into the same stdin size that wedges PowerShell. +async function createWindowsUploadDirectories( target: SshTarget, - entries: WindowsUploadEntry[], + directories: readonly string[], options: SystemSshOperationOptions ): Promise { - throwIfAborted(options.signal) - const channel = spawnSystemSshCommand(target, makeWindowsUploadPackageCommand(), { - wrapCommand: false, - ...getSystemSshBuildArgsFromOperationOptions(options) - }) - const closePromise = awaitWithSystemSshAbort( - options.signal, - () => channel.close(), - waitForChannelClose(channel, 'windows relay upload') - ) - if (!options.signal?.aborted) { - channel.stdin.end(JSON.stringify(entries)) - } - await closePromise -} - -async function readLocalUploadFile( - localPath: string, - statResult: Awaited> -): Promise { - const handle = await open(localPath, constants.O_RDONLY | (constants.O_NOFOLLOW ?? 0)) - try { - const openedStat = await handle.stat() - if ( - !openedStat.isFile() || - openedStat.size !== statResult.size || - (statResult.ino !== 0 && openedStat.ino !== 0 && openedStat.ino !== statResult.ino) || - (statResult.dev !== 0 && openedStat.dev !== 0 && openedStat.dev !== statResult.dev) - ) { - throw new Error(`File changed during upload: ${localPath}`) + let batch: string[] = [] + let batchBytes = 0 + const flush = async (): Promise => { + if (batch.length === 0) { + return } - return await handle.readFile() - } finally { - await handle.close() + const payload = JSON.stringify(batch) + batch = [] + batchBytes = 0 + throwIfAborted(options.signal) + const channel = spawnSystemSshCommand(target, makeWindowsCreateDirectoriesCommand(), { + wrapCommand: false, + ...getSystemSshBuildArgsFromOperationOptions(options) + }) + const closePromise = awaitWithSystemSshAbort( + options.signal, + () => channel.close(), + waitForChannelClose(channel, 'windows relay upload mkdir', WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ) + if (!options.signal?.aborted) { + channel.stdin.end(payload) + } + await closePromise } + for (const directory of directories) { + const entryBytes = Buffer.byteLength(directory) + 4 + if (batch.length > 0 && batchBytes + entryBytes > WINDOWS_STDIN_WRITE_CHUNK_BYTES) { + await flush() + } + batch.push(directory) + batchBytes += entryBytes + } + await flush() } -function makeWindowsUploadPackageCommand(): string { +function makeWindowsCreateDirectoriesCommand(): string { return powerShellCommand( [ '$ErrorActionPreference = "Stop"', - '$json = [Console]::In.ReadToEnd()', + // The reporter measured this reader surviving 50KB where `[Console]::In` wedged at the same + // size (#16432); the batch above stays under that. + '$reader = New-Object System.IO.StreamReader([Console]::OpenStandardInput())', + 'try { $json = $reader.ReadToEnd() } finally { $reader.Dispose() }', 'if ([string]::IsNullOrWhiteSpace($json)) { return }', - '$items = $json | ConvertFrom-Json', - 'foreach ($item in @($items)) {', - ' $path = [string]$item.path', - ' if ($item.kind -eq "directory") {', - ' $null = [System.IO.Directory]::CreateDirectory($path)', - ' continue', - ' }', - ' $parent = [System.IO.Path]::GetDirectoryName($path)', - ' if ($parent) { $null = [System.IO.Directory]::CreateDirectory($parent) }', - ' [System.IO.File]::WriteAllBytes($path, [Convert]::FromBase64String([string]$item.contentsBase64))', + 'foreach ($path in @($json | ConvertFrom-Json)) {', + ' $null = [System.IO.Directory]::CreateDirectory([string]$path)', '}' ].join('; ') ) diff --git a/src/main/ssh/system-ssh-operation-lifecycle.ts b/src/main/ssh/system-ssh-operation-lifecycle.ts index c9c4424325b..5ebfb17ced1 100644 --- a/src/main/ssh/system-ssh-operation-lifecycle.ts +++ b/src/main/ssh/system-ssh-operation-lifecycle.ts @@ -3,13 +3,25 @@ import type { SystemSshCommandChannel } from './system-ssh-command' export type ProcessResult = { label: string; stderr: string } +/** + * `timeoutMs` bounds a remote consumer that never returns. Windows PowerShell 5.1 cannot drain a + * large redirected stdin over a non-pty ssh exec (#16432): the remote process stays alive at idle + * CPU, writes nothing, and never closes — so without a bound this promise is simply never settled + * and the caller waits forever with no error to show. + */ export function waitForChannelClose( channel: SystemSshCommandChannel, - label: string + label: string, + timeoutMs?: number ): Promise { return new Promise((resolve, reject) => { let stderr = '' + let timer: ReturnType | null = null const cleanup = (): void => { + if (timer) { + clearTimeout(timer) + timer = null + } channel.stderr.off('data', onStderrData) channel.off('error', onError) channel.off('close', onClose) @@ -18,6 +30,20 @@ export function waitForChannelClose( cleanup() fn(val as never) } + if (timeoutMs !== undefined) { + timer = setTimeout(() => { + // Settle before closing: the close we request would otherwise come back as a SIGTERM + // failure and mask the timeout, which is the only diagnosis a wedged remote gives. + settle( + reject, + new Error( + `${label} timed out after ${timeoutMs}ms with no response from the remote host: ${stderr.trim()}` + ) + ) + channel.close() + }, timeoutMs) + timer.unref?.() + } const onStderrData = (data: Buffer): void => { stderr += data.toString('utf-8') } diff --git a/src/main/ssh/system-ssh-windows-upload.test.ts b/src/main/ssh/system-ssh-windows-upload.test.ts new file mode 100644 index 00000000000..207c3e2df8e --- /dev/null +++ b/src/main/ssh/system-ssh-windows-upload.test.ts @@ -0,0 +1,288 @@ +/** + * #16432: the Windows relay upload pushed the whole bundle into one PowerShell stdin, which + * Windows PowerShell 5.1 cannot drain over a non-pty ssh exec — the remote blocks forever, and + * `waitForChannelClose()` had no timeout, so the UI sat at "Connecting…" with no error. Covered + * here: no write exceeds one stdin's worth on any Windows path (bundle upload *and* single-file + * upload, which is the one that carries large files), a partial write never lands under the real + * name, and a remote that never closes fails instead of hanging. + */ +import { EventEmitter } from 'node:events' +import { mkdirSync, mkdtempSync, writeFileSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { PassThrough, Writable } from 'node:stream' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type * as SystemSshOperationLifecycle from './system-ssh-operation-lifecycle' + +const { spawnSystemSshCommandMock, waitForChannelCloseSpy } = vi.hoisted(() => ({ + spawnSystemSshCommandMock: vi.fn(), + waitForChannelCloseSpy: vi.fn() +})) + +vi.mock('./system-ssh-command', () => ({ + spawnSystemSshCommand: spawnSystemSshCommandMock +})) + +// Delegates to the real implementation; the spy only records whether each wait was given a bound. +vi.mock('./system-ssh-operation-lifecycle', async (importActual) => { + const actual = (await importActual()) as typeof SystemSshOperationLifecycle + waitForChannelCloseSpy.mockImplementation(actual.waitForChannelClose) + return { ...actual, waitForChannelClose: waitForChannelCloseSpy } +}) + +import { uploadDirectoryViaSystemSsh } from './system-ssh-file-transfer' +import { + uploadFileViaSystemSsh, + WINDOWS_STAGED_WRITE_SUFFIX, + WINDOWS_STDIN_WRITE_CHUNK_BYTES, + WINDOWS_STDIN_WRITE_TIMEOUT_MS, + writeBufferViaSystemSsh +} from './system-ssh-file-binary-transfer' +import { waitForChannelClose } from './system-ssh-operation-lifecycle' +import { getRemoteHostPlatform } from './ssh-remote-platform' +import type { SshTarget } from '../../shared/ssh-types' + +type FakeChannel = EventEmitter & { + stdin: Writable + stderr: PassThrough + close: () => void + written: Buffer +} + +const target = { id: 'win-1', host: 'win.example', username: 'dev' } as unknown as SshTarget +const hostPlatform = getRemoteHostPlatform('win32-x64') +const remoteRoot = 'C:/Users/dev/.orca-remote' + +/** Recover the script from `powershell.exe ... -EncodedCommand `. */ +function decodePowerShellCommand(command: string): string { + const encoded = /-EncodedCommand (\S+)/.exec(command)?.[1] + return encoded === undefined ? command : Buffer.from(encoded, 'base64').toString('utf16le') +} + +function createFakeChannel(onEnd: (channel: FakeChannel) => void): FakeChannel { + const channel = new EventEmitter() as FakeChannel + channel.written = Buffer.alloc(0) + channel.stderr = new PassThrough() + channel.stdin = new Writable({ + write(chunk, _encoding, callback) { + channel.written = Buffer.concat([channel.written, Buffer.from(chunk)]) + callback() + }, + final(callback) { + callback() + onEnd(channel) + } + }) + channel.close = () => channel.emit('close', null, 'SIGTERM') + return channel +} + +type RecordedCommand = { script: string; stdin: Buffer } + +describe('Windows upload stdin framing', () => { + let localDir: string + const commands: RecordedCommand[] = [] + /** Index of the spawn that should report a non-zero exit, to model a chunk failing mid-file. */ + let failAtSpawn = -1 + + const fileWrites = (): RecordedCommand[] => + commands.filter((command) => command.script.includes('FileMode]::')) + const writtenPath = (command: RecordedCommand): string => + /\$path = '((?:[^']|'')*)'/.exec(command.script)?.[1].replace(/''/g, "'") ?? '' + const fileMode = (command: RecordedCommand): string | undefined => + /FileMode\]::(\w+)/.exec(command.script)?.[1] + + beforeEach(() => { + commands.length = 0 + failAtSpawn = -1 + waitForChannelCloseSpy.mockClear() + localDir = mkdtempSync(join(tmpdir(), 'orca-win-upload-')) + spawnSystemSshCommandMock.mockReset() + spawnSystemSshCommandMock.mockImplementation((_target: SshTarget, command: string) => { + const spawnIndex = spawnSystemSshCommandMock.mock.calls.length - 1 + return createFakeChannel((channel) => { + commands.push({ script: decodePowerShellCommand(command), stdin: channel.written }) + setImmediate(() => + spawnIndex === failAtSpawn + ? channel.emit('close', 1, null) + : channel.emit('close', 0, null) + ) + }) + }) + }) + + afterEach(async () => { + await rm(localDir, { recursive: true, force: true }) + }) + + it('never pushes a whole artifact bundle into one PowerShell stdin', async () => { + mkdirSync(join(localDir, 'node'), { recursive: true }) + // Comfortably past the ~50KB point at which the reporter measured PowerShell 5.1 wedging. + writeFileSync(join(localDir, 'node', 'relay.js'), Buffer.alloc(600 * 1024, 0x61)) + writeFileSync(join(localDir, 'index.js'), Buffer.alloc(300 * 1024, 0x62)) + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const largest = Math.max(...commands.map((command) => command.stdin.length)) + expect(largest).toBeLessThanOrEqual(WINDOWS_STDIN_WRITE_CHUNK_BYTES) + // The base64 + JSON envelope is gone entirely: nothing reads the bundle as one string. + expect(commands.some((command) => command.script.includes('FromBase64String'))).toBe(false) + // `[Console]::In` wedged at 50KB where the stream reader did not, so the mkdir batch — the one + // payload still read as a string — must use the reader the reporter measured surviving. + expect(commands.some((command) => command.script.includes('[Console]::In.ReadToEnd()'))).toBe( + false + ) + expect( + commands.filter((command) => command.script.includes('StreamReader([Console]::')) + ).toHaveLength(1) + }) + + it('bounds the single-file upload too, which is the path large files take', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3 + 11, 0x64) + const localPath = join(localDir, 'big.node') + writeFileSync(localPath, contents) + + await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/big.node`, { hostPlatform }) + + const writes = fileWrites() + expect(writes).toHaveLength(4) + expect(Math.max(...writes.map((write) => write.stdin.length))).toBe( + WINDOWS_STDIN_WRITE_CHUNK_BYTES + ) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + // A wedged PowerShell never closes on its own, so no wait on this path may be unbounded. + expect( + waitForChannelCloseSpy.mock.calls.every((call) => call[2] === WINDOWS_STDIN_WRITE_TIMEOUT_MS) + ).toBe(true) + }) + + it('writes every byte of every artifact across the chunked writes', async () => { + const contents = Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 2 + 17, 0x63) + writeFileSync(join(localDir, 'relay.js'), contents) + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + const writes = fileWrites() + expect(writes).toHaveLength(3) + expect(Buffer.concat(writes.map((write) => write.stdin)).equals(contents)).toBe(true) + // Only the first write creates the staging file; the rest must extend it or it is truncated. + expect(writes.map(fileMode)).toEqual(['Create', 'Append', 'Append']) + }) + + it('still creates an empty artifact on the host', async () => { + writeFileSync(join(localDir, 'empty.txt'), '') + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + expect(fileWrites().map(writtenPath)).toEqual([`${remoteRoot}/empty.txt`]) + expect(fileWrites()[0].stdin).toHaveLength(0) + expect(fileMode(fileWrites()[0])).toBe('Create') + }) + + it('lands a multi-chunk write on a staging path and publishes it by rename', async () => { + const remotePath = `${remoteRoot}/relay.js` + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + + await uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + + // Nothing touches the real name until every byte is on the host. + expect(fileWrites().map(writtenPath)).toEqual([ + `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}`, + `${remotePath}${WINDOWS_STAGED_WRITE_SUFFIX}` + ]) + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).toContain('[System.IO.File]::Delete($path)') + }) + + it('leaves no truncated file under the real name when a chunk fails mid-file', async () => { + writeFileSync(join(localDir, 'relay.js'), Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES * 3)) + // Spawns: 0 = mkdir batch, 1..3 = chunk writes. Fail the second chunk. + failAtSpawn = 2 + + await expect( + uploadDirectoryViaSystemSsh(target, localDir, remoteRoot, { hostPlatform }) + ).rejects.toThrow() + + expect(fileWrites().map(writtenPath)).not.toContain(`${remoteRoot}/relay.js`) + expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) + }) + + it('enforces exclusive once at the rename, so a retry is not blocked by its own leftovers', async () => { + const localPath = join(localDir, 'import.bin') + writeFileSync(localPath, Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1)) + + await uploadFileViaSystemSsh(target, localPath, `${remoteRoot}/import.bin`, { + hostPlatform, + exclusive: true + }) + + // CreateNew on chunk one would fail against a leftover staging file from a failed attempt; + // `File::Move` raising on an existing destination is what carries the exclusive contract. + expect(fileWrites().map(fileMode)).toEqual(['Create', 'Append']) + const publish = commands.at(-1)! + expect(publish.script).toContain('[System.IO.File]::Move($staging, $path)') + expect(publish.script).not.toContain('[System.IO.File]::Delete($path)') + }) + + it('keeps a single-chunk write on the destination, with the caller mode intact', async () => { + await writeBufferViaSystemSsh(target, `${remoteRoot}/version`, Buffer.from('1.2.3'), { + hostPlatform, + exclusive: true + }) + + expect(fileWrites()).toHaveLength(1) + expect(writtenPath(fileWrites()[0])).toBe(`${remoteRoot}/version`) + expect(fileMode(fileWrites()[0])).toBe('CreateNew') + expect(commands.some((command) => command.script.includes('::Move('))).toBe(false) + }) + + it('appends onto the destination rather than staging, since append cannot be staged', async () => { + const remotePath = `${remoteRoot}/log.bin` + await writeBufferViaSystemSsh( + target, + remotePath, + Buffer.alloc(WINDOWS_STDIN_WRITE_CHUNK_BYTES + 1), + { hostPlatform, append: true } + ) + + expect(fileWrites().map(writtenPath)).toEqual([remotePath, remotePath]) + expect(fileWrites().map(fileMode)).toEqual(['Append', 'Append']) + }) +}) + +describe('waitForChannelClose bounding', () => { + it('fails a remote that accepts stdin and never closes, instead of waiting forever', async () => { + vi.useFakeTimers() + try { + const channel = createFakeChannel(() => {}) + const settled = vi.fn() + const promise = waitForChannelClose(channel as never, 'windows relay upload', 1_000) + promise.then(settled, settled) + + await vi.advanceTimersByTimeAsync(999) + expect(settled).not.toHaveBeenCalled() + + await vi.advanceTimersByTimeAsync(2) + await expect(promise).rejects.toThrow(/timed out after 1000ms with no response/) + } finally { + vi.useRealTimers() + } + }) + + it('leaves an unbounded wait unbounded when no timeout is asked for', async () => { + vi.useFakeTimers() + try { + const channel = createFakeChannel(() => {}) + const settled = vi.fn() + // POSIX `cat` drains its stdin; only the Windows writes need the bound. + void waitForChannelClose(channel as never, 'posix write').then(settled, settled) + + await vi.advanceTimersByTimeAsync(60 * 60 * 1000) + expect(settled).not.toHaveBeenCalled() + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/src/relay/fs-handler-list-files-result-limit.test.ts b/src/relay/fs-handler-list-files-result-limit.test.ts new file mode 100644 index 00000000000..17876d76020 --- /dev/null +++ b/src/relay/fs-handler-list-files-result-limit.test.ts @@ -0,0 +1,116 @@ +/** + * #12547: the relay used to serialize an unbounded `fs.listFiles` reply into one response frame, + * which died as "Message too large" or over-capacity. #17954 fixed that by streaming the reply, so + * the size of a listing is no longer a correctness question and the host must NOT quietly impose a + * cap of its own — a caller that named no limit reads the array as the whole listing, and clients + * that predate `maxResults` on this call hardcode `truncated: false`, so a prefix would reach them + * as a complete tree with nothing on the wire to notice. The cap belongs to the caller; the host + * only clamps it to the ceiling the scan's retention budget assumes. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const { runListFilesScanMock } = vi.hoisted(() => ({ + runListFilesScanMock: vi.fn() +})) + +vi.mock('./fs-list-files-fallback-chain', () => ({ + runListFilesScan: runListFilesScanMock +})) + +vi.mock('@parcel/watcher', () => ({ subscribe: vi.fn() })) + +import { FsHandler } from './fs-handler' +import { RelayContext } from './context' +import type { RelayDispatcher } from './dispatcher' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../shared/quick-open-listing-limits' + +type ListFilesHandler = ( + params: Record, + context?: { clientId: number } +) => Promise + +function createHandler(): { listFiles: ListFilesHandler; dispose: () => void } { + const requestHandlers = new Map() + const dispatcher = { + onRequest: (method: string, handler: ListFilesHandler) => requestHandlers.set(method, handler), + onNotification: vi.fn(), + onClientDetached: vi.fn(), + notify: vi.fn(), + notifyBulk: vi.fn(), + publishProducerNotification: vi.fn(() => true), + activeClientIds: () => [], + producerEnvelopeBudget: () => Number.MAX_SAFE_INTEGER + } as unknown as RelayDispatcher + const handler = new FsHandler(dispatcher, new RelayContext(), { + dispose: vi.fn(), + forgetRoot: vi.fn(), + subscribe: vi.fn() + }) + return { listFiles: requestHandlers.get('fs.listFiles')!, dispose: () => handler.dispose() } +} + +describe('fs.listFiles result limit', () => { + let listFiles: ListFilesHandler + let dispose: () => void + + beforeEach(() => { + runListFilesScanMock.mockReset() + runListFilesScanMock.mockResolvedValue([]) + const created = createHandler() + listFiles = created.listFiles + dispose = created.dispose + return () => dispose() + }) + + function scanMaxResults(): unknown { + // runListFilesScan(rootPath, excludePathPrefixes, signal, maxResults, searchQuery) + return runListFilesScanMock.mock.calls[0][3] + } + + it('leaves a request that omitted maxResults unbounded rather than silently prefixing it', async () => { + await listFiles({ rootPath: '/remote/root' }, { clientId: 1 }) + + expect(scanMaxResults()).toBeUndefined() + }) + + it('ignores a malformed maxResults rather than treating it as a cap', async () => { + await listFiles({ rootPath: '/remote/root', maxResults: 'all' }, { clientId: 1 }) + + expect(scanMaxResults()).toBeUndefined() + }) + + it('answers an uncapped request in full, however large the tree is', async () => { + const files = Array.from( + { length: QUICK_OPEN_LISTING_MAX_RESULTS + 500 }, + (_, index) => `f${index}` + ) + runListFilesScanMock.mockResolvedValue(files) + + await expect(listFiles({ rootPath: '/remote/root' }, { clientId: 1 })).resolves.toEqual(files) + }) + + it('hands a client that named a cap the prefix it asked for', async () => { + runListFilesScanMock.mockResolvedValue( + Array.from({ length: QUICK_OPEN_LISTING_MAX_RESULTS }, (_, index) => `f${index}`) + ) + + const files = await listFiles( + { rootPath: '/remote/root', maxResults: QUICK_OPEN_LISTING_MAX_RESULTS }, + { clientId: 1 } + ) + + expect(files).toHaveLength(QUICK_OPEN_LISTING_MAX_RESULTS) + }) + + it('keeps a smaller client limit and clamps a larger one', async () => { + await listFiles({ rootPath: '/remote/root', maxResults: 33 }, { clientId: 1 }) + expect(scanMaxResults()).toBe(33) + + runListFilesScanMock.mockClear() + await listFiles( + { rootPath: '/remote/root', maxResults: QUICK_OPEN_LISTING_MAX_RESULTS * 10 }, + { clientId: 2 } + ) + expect(scanMaxResults()).toBe(QUICK_OPEN_LISTING_MAX_RESULTS) + }) +}) diff --git a/src/relay/fs-handler.ts b/src/relay/fs-handler.ts index ec11a806bb8..d20d0d073d6 100644 --- a/src/relay/fs-handler.ts +++ b/src/relay/fs-handler.ts @@ -25,6 +25,7 @@ import { writeRelayFile } from './fs-path-mutation-requests' import { buildExcludePathPrefixes } from '../shared/quick-open-filter' +import { resolveQuickOpenResultLimit } from '../shared/quick-open-listing-limits' import { maybeStreamRpcResponse, type GitResponseStreamRegistry } from './git-response-stream' import { readRelayFileContent, readRelayFileStreamMetadata } from './fs-handler-file-read' import { readRelayFileRange } from './fs-handler-file-range' @@ -217,11 +218,14 @@ export class FsHandler { context?: RequestContext ): Promise { const rootPath = expandTilde(params.rootPath as string) + // Why no host-side default: #17954 made an oversized reply streamable, so a caller that names no + // limit gets its whole listing instead of an unannounced prefix it would report as complete. + // A requested limit is still clamped to the shared ceiling the scan's retention budget assumes. const maxResults = typeof params.maxResults === 'number' && Number.isInteger(params.maxResults) && params.maxResults > 0 - ? Math.min(params.maxResults, 20_001) + ? resolveQuickOpenResultLimit(params.maxResults) : undefined const searchQuery = typeof params.searchQuery === 'string' && params.searchQuery.trim().length > 0 diff --git a/src/relay/relay-client-resync-marker.ts b/src/relay/relay-client-resync-marker.ts new file mode 100644 index 00000000000..4d6353ea59b --- /dev/null +++ b/src/relay/relay-client-resync-marker.ts @@ -0,0 +1,168 @@ +import type { RelayDispatcher } from './dispatcher' + +type RetainedMarker = { params: Record; estimatedBytes: number } + +type MarkerState = { + method: string + // Why: one outstanding marker per (client, key) keeps sustained backpressure bounded. + inFlight: Set + // Rejected markers, retained per client so they can be republished when the control lane frees up. + pending: Map> + capacityUnsubscribes: Map void> +} + +/** + * Publishes a small "your view is stale, re-read it" notification on the control lane after a + * producer-lane payload was refused. Shared by every producer whose oversized frame would otherwise + * desync a client silently; the marker is coalesced per key and retried on capacity, never dropped. + */ +export type RelayClientResyncMarkerPublisher = { + emit(clientId: number, markerKey: string, params: Record): void + forgetClient(clientId: number): void +} + +export function createRelayClientResyncMarkerPublisher( + dispatcher: RelayDispatcher, + method: string +): RelayClientResyncMarkerPublisher { + const state: MarkerState = { + method, + inFlight: new Set(), + pending: new Map(), + capacityUnsubscribes: new Map() + } + // In-flight keys need no sweep here: closing a client settles every queued and written frame first. + dispatcher.onClientDetached((clientId) => { + // Not every detach retires the id: invalidateClient() detaches the primary without removing it and + // setWrite() revives it, so dropping the markers here would desync the state the reconnect restores. + if (dispatcher.isClientAttached(clientId)) { + return + } + forgetClientMarkers(state, clientId) + }) + return { + emit(clientId, markerKey, params) { + const retained = state.pending.get(clientId)?.get(markerKey) + if (retained) { + // Latest generation wins: a retained marker that has not been sent yet must not replay a + // projection the producer has already moved past. + retained.params = params + retained.estimatedBytes = dispatcher.notificationFrameBytes(method, params) + return + } + // Per key, never per client alone: an outstanding marker for one subject must not suppress another's resync. + if (state.inFlight.has(markerId(clientId, markerKey))) { + return + } + publishMarker(dispatcher, state, clientId, markerKey, params) + }, + forgetClient(clientId) { + forgetClientMarkers(state, clientId) + } + } +} + +function markerId(clientId: number, markerKey: string): string { + return `${clientId} ${markerKey}` +} + +function forgetClientMarkers(state: MarkerState, clientId: number): void { + // Unsubscribe first so no re-entrant flush can observe a half-cleared client. + state.capacityUnsubscribes.get(clientId)?.() + state.capacityUnsubscribes.delete(clientId) + state.pending.delete(clientId) +} + +// Why: the control lane — on the producer lane the marker would hit the same full queue that just +// rejected the payload and be dropped, silently desyncing the client. +function publishMarker( + dispatcher: RelayDispatcher, + state: MarkerState, + clientId: number, + markerKey: string, + params: Record, + estimatedBytes?: number +): void { + const key = markerId(clientId, markerKey) + const frameBytes = estimatedBytes ?? dispatcher.notificationFrameBytes(state.method, params) + state.inFlight.add(key) + let settled = false + const accepted = dispatcher.tryNotifyClient( + clientId, + state.method, + params, + (result) => { + // Settles on write, drop, or client close, so the slot can never leak. + settled = true + state.inFlight.delete(key) + if (result.ok) { + return + } + // A frame the sink never wrote leaves the client just as desynced as a rejected one — setWrite + // fails every queued and in-flight frame this way. Retain unconditionally: a real detach clears + // it through onClientDetached, which fires after this settlement. + retainMarker(dispatcher, state, clientId, markerKey, params, frameBytes) + }, + { controlOverflow: 'reject' } + ) + if (accepted || settled) { + return + } + // Admission rejection has no settlement callback: retain the marker instead of desyncing the client. + state.inFlight.delete(key) + retainMarker(dispatcher, state, clientId, markerKey, params, frameBytes) +} + +function retainMarker( + dispatcher: RelayDispatcher, + state: MarkerState, + clientId: number, + markerKey: string, + params: Record, + estimatedBytes: number +): void { + if (!state.capacityUnsubscribes.has(clientId)) { + const unsubscribe = dispatcher.onClientCapacity(clientId, () => + flushPendingMarkers(dispatcher, state, clientId) + ) + if (!unsubscribe) { + // The client went away between admission and arming, so there is nothing left to resync. + return + } + state.capacityUnsubscribes.set(clientId, unsubscribe) + } + const retained = state.pending.get(clientId) + if (retained) { + retained.set(markerKey, { params, estimatedBytes }) + return + } + state.pending.set(clientId, new Map([[markerKey, { params, estimatedBytes }]])) +} + +function flushPendingMarkers( + dispatcher: RelayDispatcher, + state: MarkerState, + clientId: number +): void { + const retained = state.pending.get(clientId) + if (!retained) { + return + } + for (const [markerKey, marker] of Array.from(retained)) { + // Capacity fires on every lane; retry only frames the control queue can admit now. + if (!dispatcher.canAdmitControlFrame(clientId, marker.estimatedBytes)) { + continue + } + // Drop before republishing so a synchronous settlement cannot see the marker as still pending — + // and skip keys a re-entrant flush already took, which would otherwise send the marker twice. + if (!retained.delete(markerKey)) { + continue + } + publishMarker(dispatcher, state, clientId, markerKey, marker.params, marker.estimatedBytes) + } + // Identity check: a re-entrant flush may have retired this set and armed a fresh one to keep. + if (retained.size > 0 || state.pending.get(clientId) !== retained) { + return + } + forgetClientMarkers(state, clientId) +} diff --git a/src/relay/relay-watcher-event-emitter.ts b/src/relay/relay-watcher-event-emitter.ts index 739ae89b2f8..d42c00824bb 100644 --- a/src/relay/relay-watcher-event-emitter.ts +++ b/src/relay/relay-watcher-event-emitter.ts @@ -1,6 +1,10 @@ import type { WatcherProcessEvent } from '../main/ipc/parcel-watcher-process' import { resolveRuntimePath } from '../shared/cross-platform-path' import type { RelayDispatcher } from './dispatcher' +import { + createRelayClientResyncMarkerPublisher, + type RelayClientResyncMarkerPublisher +} from './relay-client-resync-marker' type MappedWatcherEvent = { kind: string @@ -13,15 +17,7 @@ type WatcherBatchSizing = { batchBytes: number } -type OverflowMarkerState = { - // Why: one outstanding marker per (client, root) keeps sustained backpressure bounded. - inFlight: Set - // Rejected markers, retained per client so they can be republished when the control lane frees up. - pending: Map> - capacityUnsubscribes: Map void> -} - -const overflowMarkerStates = new WeakMap() +const overflowMarkerPublishers = new WeakMap() export function emitRelayWatcherEvents( dispatcher: RelayDispatcher, @@ -164,147 +160,26 @@ function publishWatcherBatchToClient( } } -function overflowMarkerState(dispatcher: RelayDispatcher): OverflowMarkerState { - const existing = overflowMarkerStates.get(dispatcher) +function overflowMarkerPublisher(dispatcher: RelayDispatcher): RelayClientResyncMarkerPublisher { + const existing = overflowMarkerPublishers.get(dispatcher) if (existing) { return existing } - const state: OverflowMarkerState = { - inFlight: new Set(), - pending: new Map(), - capacityUnsubscribes: new Map() - } - overflowMarkerStates.set(dispatcher, state) - // In-flight keys need no sweep here: closing a client settles every queued and written frame first. - dispatcher.onClientDetached((clientId) => { - // Not every detach retires the id: invalidateClient() detaches the primary without removing it and - // setWrite() revives it, so dropping the markers here would desync the tree the reconnect restores. - if (dispatcher.isClientAttached(clientId)) { - return - } - forgetClientMarkers(state, clientId) - }) - return state -} - -function forgetClientMarkers(state: OverflowMarkerState, clientId: number): void { - // Unsubscribe first so no re-entrant flush can observe a half-cleared client. - state.capacityUnsubscribes.get(clientId)?.() - state.capacityUnsubscribes.delete(clientId) - state.pending.delete(clientId) + const publisher = createRelayClientResyncMarkerPublisher(dispatcher, 'fs.changed') + overflowMarkerPublishers.set(dispatcher, publisher) + return publisher } function overflowMarkerParams(rootPath: string): Record { return { events: [{ kind: 'overflow', absolutePath: rootPath }] } } -// Why: the control lane — on the producer lane the marker would hit the same full queue that just -// rejected the batch and be dropped, silently desyncing the remote file tree. function emitWatcherOverflowToClient( dispatcher: RelayDispatcher, clientId: number, rootPath: string ): void { - const state = overflowMarkerState(dispatcher) - // Per root, never per client alone: an outstanding marker for one tree must not suppress another's resync. - if ( - state.inFlight.has(`${clientId} ${rootPath}`) || - state.pending.get(clientId)?.has(rootPath) === true - ) { - return - } - publishOverflowMarker(dispatcher, state, clientId, rootPath) -} - -function publishOverflowMarker( - dispatcher: RelayDispatcher, - state: OverflowMarkerState, - clientId: number, - rootPath: string, - estimatedBytes?: number -): void { - const key = `${clientId} ${rootPath}` - const params = overflowMarkerParams(rootPath) - const frameBytes = estimatedBytes ?? dispatcher.notificationFrameBytes('fs.changed', params) - state.inFlight.add(key) - let settled = false - const accepted = dispatcher.tryNotifyClient( - clientId, - 'fs.changed', - params, - (result) => { - // Settles on write, drop, or client close, so the slot can never leak. - settled = true - state.inFlight.delete(key) - if (result.ok) { - return - } - // A frame the sink never wrote leaves the tree just as desynced as a rejected one — setWrite - // fails every queued and in-flight frame this way. Retain unconditionally: a real detach clears - // it through onClientDetached, which fires after this settlement. - retainOverflowMarker(dispatcher, state, clientId, rootPath, frameBytes) - }, - { controlOverflow: 'reject' } - ) - if (accepted || settled) { - return - } - // Admission rejection has no settlement callback: retain the marker instead of desyncing the tree. - state.inFlight.delete(key) - retainOverflowMarker(dispatcher, state, clientId, rootPath, frameBytes) -} - -function retainOverflowMarker( - dispatcher: RelayDispatcher, - state: OverflowMarkerState, - clientId: number, - rootPath: string, - estimatedBytes: number -): void { - if (!state.capacityUnsubscribes.has(clientId)) { - const unsubscribe = dispatcher.onClientCapacity(clientId, () => - flushPendingOverflowMarkers(dispatcher, state, clientId) - ) - if (!unsubscribe) { - // The client went away between admission and arming, so there is nothing left to resync. - return - } - state.capacityUnsubscribes.set(clientId, unsubscribe) - } - const roots = state.pending.get(clientId) - if (roots) { - roots.set(rootPath, estimatedBytes) - return - } - state.pending.set(clientId, new Map([[rootPath, estimatedBytes]])) -} - -function flushPendingOverflowMarkers( - dispatcher: RelayDispatcher, - state: OverflowMarkerState, - clientId: number -): void { - const roots = state.pending.get(clientId) - if (!roots) { - return - } - for (const [rootPath, estimatedBytes] of Array.from(roots)) { - // Capacity fires on every lane; retry only frames the control queue can admit now. - if (!dispatcher.canAdmitControlFrame(clientId, estimatedBytes)) { - continue - } - // Drop before republishing so a synchronous settlement cannot see the marker as still pending — - // and skip roots a re-entrant flush already took, which would otherwise send the marker twice. - if (!roots.delete(rootPath)) { - continue - } - publishOverflowMarker(dispatcher, state, clientId, rootPath, estimatedBytes) - } - // Identity check: a re-entrant flush may have retired this set and armed a fresh one to keep. - if (roots.size > 0 || state.pending.get(clientId) !== roots) { - return - } - forgetClientMarkers(state, clientId) + overflowMarkerPublisher(dispatcher).emit(clientId, rootPath, overflowMarkerParams(rootPath)) } export function emitRelayWatcherOverflow( diff --git a/src/relay/workspace-session-handler.ts b/src/relay/workspace-session-handler.ts index c87a8b6076f..f9b3ef2e507 100644 --- a/src/relay/workspace-session-handler.ts +++ b/src/relay/workspace-session-handler.ts @@ -2,6 +2,7 @@ import { existsSync, mkdirSync, readFileSync, renameSync, writeFileSync } from ' import { homedir } from 'node:os' import { dirname, join } from 'node:path' import type { RelayDispatcher } from './dispatcher' +import { publishWorkspaceSnapshotChange } from './workspace-snapshot-publication' type RemoteWorkspaceSnapshot = { namespace: string @@ -151,11 +152,15 @@ export class WorkspaceSessionHandler { session: patch.session as Record } this.write(snapshot) - this.dispatcher.notify('workspace.changed', { - namespace, - snapshot, - sourceClientId: typeof params.clientId === 'string' ? params.clientId : undefined - }) + publishWorkspaceSnapshotChange( + this.dispatcher, + { + namespace, + snapshot, + sourceClientId: typeof params.clientId === 'string' ? params.clientId : undefined + }, + namespace + ) return { ok: true, snapshot } } diff --git a/src/relay/workspace-snapshot-publication.test.ts b/src/relay/workspace-snapshot-publication.test.ts new file mode 100644 index 00000000000..804a13f452b --- /dev/null +++ b/src/relay/workspace-snapshot-publication.test.ts @@ -0,0 +1,166 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { RelayDispatcher } from './dispatcher' +import { relayWriterControlReserve } from './dispatcher-writer-admission' +import { encodeJsonRpcFrame, MessageType, type JsonRpcRequest } from './protocol' +import { WorkspaceSessionHandler } from './workspace-session-handler' +import { + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + REMOTE_WORKSPACE_STALE_NOTIFICATION +} from '../shared/remote-workspace-types' + +// The relay runs on the REMOTE host, so the sink default is that host's Node major. +// Node <= 21 defaults to a 16KB high-water mark, which is the 12288B capacity issue #15238 reports. +const NODE21_HWM = 16 * 1024 + +function decodeNotifications( + written: Buffer[] +): { method: string; params: Record }[] { + return written + .filter((buf) => buf[0] === MessageType.Regular) + .map((buf) => { + const len = buf.readUInt32BE(9) + return JSON.parse(buf.subarray(13, 13 + len).toString('utf-8')) as { + method?: string + params?: Record + } + }) + .filter( + (msg): msg is { method: string; params: Record } => + typeof msg.method === 'string' + ) + .map((msg) => ({ method: msg.method, params: msg.params ?? {} })) +} + +/** A session shaped like the report: several worktrees, each with a handful of tabs. */ +function oversizedSession(worktrees: number, tabsPerWorktree: number): Record { + const tabsByWorktreePath: Record = {} + const terminalLayoutsByTabId: Record = {} + for (let w = 0; w < worktrees; w++) { + const worktreePath = `/home/dev/orca/workspaces/project/feature-branch-${w}` + tabsByWorktreePath[worktreePath] = Array.from({ length: tabsPerWorktree }, (_, t) => ({ + id: `tab-${w}-${t}-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a`, + title: `claude — feature-branch-${w} — pane ${t}`, + worktreePath, + kind: 'terminal', + startupCommand: 'claude --dangerously-skip-permissions', + cwd: worktreePath + })) + for (let t = 0; t < tabsPerWorktree; t++) { + terminalLayoutsByTabId[`tab-${w}-${t}-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a`] = { + direction: 'row', + panes: [ + { id: `pane-${w}-${t}-a`, size: 50, remoteSessionId: `orca-remote-${w}-${t}-a` }, + { id: `pane-${w}-${t}-b`, size: 50, remoteSessionId: `orca-remote-${w}-${t}-b` } + ] + } + } + } + return { + activeWorktreePath: '/home/dev/orca/workspaces/project/feature-branch-0', + activeTabId: 'tab-0-0-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a', + tabsByWorktreePath, + terminalLayoutsByTabId + } +} + +describe('workspace snapshot publication over a bounded producer frame', () => { + let baseDir: string + let dispatcher: RelayDispatcher + let written: Buffer[] + + beforeEach(() => { + baseDir = mkdtempSync(join(tmpdir(), 'orca-workspace-publication-')) + written = [] + dispatcher = new RelayDispatcher( + (data) => { + written.push(Buffer.from(data)) + return true + }, + { + writableHighWaterMark: () => NODE21_HWM, + writableLength: () => 0, + supportsWriteCallback: false + } + ) + new WorkspaceSessionHandler(dispatcher, baseDir) + }) + + afterEach(() => { + dispatcher.dispose() + rmSync(baseDir, { recursive: true, force: true }) + }) + + async function patch(session: Record, id: number): Promise { + const req: JsonRpcRequest = { + jsonrpc: '2.0', + id, + method: 'workspace.patch', + params: { + namespace: 'ssh_host_project', + baseRevision: id - 1, + clientId: 'client-a', + patch: { kind: 'replace-session', session } + } + } + dispatcher.feed(encodeJsonRpcFrame(req, id, 0)) + await Promise.resolve() + await Promise.resolve() + } + + it('publishes the snapshot inline while it fits the producer frame', async () => { + await patch(oversizedSession(1, 1), 1) + + const methods = decodeNotifications(written).map((msg) => msg.method) + expect(methods).toContain(REMOTE_WORKSPACE_CHANGED_NOTIFICATION) + expect(methods).not.toContain(REMOTE_WORKSPACE_STALE_NOTIFICATION) + }) + + it('tells the client its view is stale instead of dropping an oversized snapshot', async () => { + const session = oversizedSession(3, 8) + // Pin the premise: this really is over the 12288B capacity the issue reports, so the assertion + // below measures the drop path and not a payload that happened to fit. + const capacity = NODE21_HWM - relayWriterControlReserve(NODE21_HWM) + expect(capacity).toBe(12288) + expect( + dispatcher.notificationFrameBytes(REMOTE_WORKSPACE_CHANGED_NOTIFICATION, { + namespace: 'ssh_host_project', + snapshot: { + namespace: 'ssh_host_project', + revision: 1, + updatedAt: 0, + schemaVersion: 1, + session + }, + sourceClientId: 'client-a' + }) + ).toBeGreaterThan(capacity) + + await patch(session, 1) + + const notifications = decodeNotifications(written) + expect(notifications.map((msg) => msg.method)).not.toContain( + REMOTE_WORKSPACE_CHANGED_NOTIFICATION + ) + const stale = notifications.filter((msg) => msg.method === REMOTE_WORKSPACE_STALE_NOTIFICATION) + expect(stale).toHaveLength(1) + expect(stale[0].params).toEqual({ namespace: 'ssh_host_project' }) + }) + + it('carries no revision or author, so a coalesced marker cannot replay a superseded generation', async () => { + const session = oversizedSession(3, 8) + await patch(session, 1) + written = [] + await patch({ ...session, activeTabId: 'tab-1-1-1f7a4c2e-9b0d-4e51-8a63-2c9f0d1e4b7a' }, 2) + + for (const marker of decodeNotifications(written).filter( + (msg) => msg.method === REMOTE_WORKSPACE_STALE_NOTIFICATION + )) { + expect(marker.params).not.toHaveProperty('revision') + expect(marker.params).not.toHaveProperty('sourceClientId') + expect(marker.params).not.toHaveProperty('snapshot') + } + }) +}) diff --git a/src/relay/workspace-snapshot-publication.ts b/src/relay/workspace-snapshot-publication.ts new file mode 100644 index 00000000000..3a621cb1172 --- /dev/null +++ b/src/relay/workspace-snapshot-publication.ts @@ -0,0 +1,49 @@ +import { + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + REMOTE_WORKSPACE_STALE_NOTIFICATION +} from '../shared/remote-workspace-types' +import type { RelayDispatcher } from './dispatcher' +import { + createRelayClientResyncMarkerPublisher, + type RelayClientResyncMarkerPublisher +} from './relay-client-resync-marker' + +const stalePublishers = new WeakMap() + +function stalePublisher(dispatcher: RelayDispatcher): RelayClientResyncMarkerPublisher { + const existing = stalePublishers.get(dispatcher) + if (existing) { + return existing + } + const publisher = createRelayClientResyncMarkerPublisher( + dispatcher, + REMOTE_WORKSPACE_STALE_NOTIFICATION + ) + stalePublishers.set(dispatcher, publisher) + return publisher +} + +/** + * Per client, because frame capacity is per client: one peer on a small sink must not cost the + * others their snapshot, and a peer that cannot take the snapshot still learns it is behind. + * The marker deliberately carries no revision or author — a coalesced marker would then replay a + * generation the producer has moved past, which is the same silent staleness this exists to remove. + */ +export function publishWorkspaceSnapshotChange( + dispatcher: RelayDispatcher, + params: Record, + namespace: string +): void { + for (const clientId of dispatcher.activeClientIds()) { + if ( + dispatcher.publishProducerNotification( + clientId, + REMOTE_WORKSPACE_CHANGED_NOTIFICATION, + params + ) + ) { + continue + } + stalePublisher(dispatcher).emit(clientId, namespace, { namespace }) + } +} diff --git a/src/renderer/src/components/quick-open-file-list.react.test.tsx b/src/renderer/src/components/quick-open-file-list.react.test.tsx index b59f08fae93..2673e90af76 100644 --- a/src/renderer/src/components/quick-open-file-list.react.test.tsx +++ b/src/renderer/src/components/quick-open-file-list.react.test.tsx @@ -7,6 +7,7 @@ import type { FolderWorkspace } from '../../../shared/folder-workspace-types' import type { ProjectGroup } from '../../../shared/project-group-types' import type { Worktree } from '../../../shared/worktree/types' import { folderWorkspaceKey } from '../../../shared/workspace-scope' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../../../shared/quick-open-listing-limits' import { useAppStore } from '@/store' import type { AppState } from '@/store/types' import { useRuntimeFileListForWorktree, type RuntimeFileListState } from './quick-open-file-list' @@ -201,10 +202,40 @@ describe('useRuntimeFileListForWorktree', () => { rootPath: '/srv/platform', excludePaths: undefined, requestToken: expect.any(String), + // #12547: the caller names the cap so a full page is readable as truncation. + maxResults: QUICK_OPEN_LISTING_MAX_RESULTS, signal: expect.any(AbortSignal) } ) expect(states.at(-1)?.files).toEqual(['packages/app/package.json']) + expect(states.at(-1)?.truncated).toBe(false) + }) + + // #12547: the host stops at the cap the caller names, so a full page is a prefix. Reporting + // truncated:false unconditionally is what left the user with a silent partial list. + it('reports a capped listing as truncated instead of as the whole workspace', async () => { + const states: RuntimeFileListState[] = [] + const workspaceKey = folderWorkspaceKey('folder-workspace-1') + listRuntimeFilesMock.mockResolvedValue( + Array.from({ length: QUICK_OPEN_LISTING_MAX_RESULTS }, (_, i) => `src/file-${i}.ts`) + ) + + useAppStore.setState({ + folderWorkspaces: [makeFolderWorkspace({ connectionId: 'ssh-1' })], + projectGroups: [makeProjectGroup({ connectionId: 'ssh-1' })], + repos: [], + worktreesByRepo: {} + } as Partial) + + await renderProbe({ + enabled: true, + onState: (state) => states.push(state), + worktreeId: workspaceKey + }) + await waitForListRuntimeFilesCall() + + expect(states.at(-1)?.files).toHaveLength(QUICK_OPEN_LISTING_MAX_RESULTS) + expect(states.at(-1)?.truncated).toBe(true) }) it('routes paired folder workspace queries to the owning runtime', async () => { diff --git a/src/renderer/src/components/quick-open-file-list.ts b/src/renderer/src/components/quick-open-file-list.ts index 04490ed7af6..7f605ce2e41 100644 --- a/src/renderer/src/components/quick-open-file-list.ts +++ b/src/renderer/src/components/quick-open-file-list.ts @@ -5,6 +5,7 @@ import type { Worktree } from '../../../shared/worktree/types' import { isWindowsAbsolutePathLike } from '../../../shared/cross-platform-path' import { createBrowserUuid } from '@/lib/browser-uuid' import { isQuickOpenRemoteQueryTooLarge } from '@/components/quick-open-search' +import { QUICK_OPEN_LISTING_MAX_RESULTS } from '../../../shared/quick-open-listing-limits' import { cancelRuntimeFileList, listRuntimeFiles, @@ -261,8 +262,15 @@ export function useRuntimeFileListForWorktree({ rootPath: worktreePath, excludePaths, requestToken, + maxResults: QUICK_OPEN_LISTING_MAX_RESULTS, signal: requestAbortController.signal - }).then((files) => ({ files, truncated: false })) + }).then((files) => ({ + // #12547: naming the cap is what makes a full page readable as "there is more". Reporting + // false unconditionally is what made the truncation silent — the host bounds the scan to + // the cap it is given, so a full page means there are more paths behind it. + files, + truncated: files.length >= QUICK_OPEN_LISTING_MAX_RESULTS + })) void request .then((result) => { diff --git a/src/renderer/src/runtime/runtime-file-search-client.ts b/src/renderer/src/runtime/runtime-file-search-client.ts index 31a91b2f150..8168b24bd54 100644 --- a/src/renderer/src/runtime/runtime-file-search-client.ts +++ b/src/renderer/src/runtime/runtime-file-search-client.ts @@ -48,6 +48,10 @@ export async function listRuntimeFiles( rootPath: string excludePaths?: string[] requestToken?: string + // Why: naming the cap is what makes a full page readable as "there is more". The host returns + // the whole listing when no limit is named, so a caller that never states one cannot tell a + // bound from a total. + maxResults?: number signal?: AbortSignal } ): Promise { @@ -57,7 +61,8 @@ export async function listRuntimeFiles( rootPath: args.rootPath, connectionId: context.connectionId, excludePaths: args.excludePaths, - requestToken: args.requestToken + requestToken: args.requestToken, + ...(args.maxResults === undefined ? {} : { maxResults: args.maxResults }) }) } return callRuntimeRpc( @@ -65,7 +70,9 @@ export async function listRuntimeFiles( 'files.listAll', { worktree: toRuntimeWorktreeSelector(context.worktreeId), - excludePaths: args.excludePaths + excludePaths: args.excludePaths, + // Optional on the host schema since #17954; an older host strips it and keeps its own default. + ...(args.maxResults === undefined ? {} : { maxResults: args.maxResults }) }, { timeoutMs: 15_000, ...(args.signal === undefined ? {} : { signal: args.signal }) } ) diff --git a/src/shared/remote-workspace-types.ts b/src/shared/remote-workspace-types.ts index f7beea81e8a..4d6eec6021e 100644 --- a/src/shared/remote-workspace-types.ts +++ b/src/shared/remote-workspace-types.ts @@ -59,6 +59,19 @@ export type RemoteWorkspaceObservedPatchResult = message?: string } +export const REMOTE_WORKSPACE_CHANGED_NOTIFICATION = 'workspace.changed' + +/** + * Sent instead of `workspace.changed` when the snapshot frame did not fit the client's producer + * frame capacity. Carries no session: the client re-reads through `workspace.get`, whose response + * lane is budgeted in megabytes rather than in one ~12KB producer frame. + * + * Wire contract: a relay that predates this never sends it, and a client that predates it drops it + * the same way it drops any unknown notification method — which is exactly the silent drop this + * replaces, so an un-negotiated pairing is never worse than before. + */ +export const REMOTE_WORKSPACE_STALE_NOTIFICATION = 'workspace.stale' + export type RemoteWorkspaceChangedEvent = { targetId: string snapshot: RemoteWorkspaceObservedSnapshot From 6cd477a2f14d5efb2890023d60b65fb495949d9a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 15:58:42 -0700 Subject: [PATCH 123/398] test(e2e): un-rot the SSH freeze repro and probe two failure modes nothing covered (#17940) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Test-only. No production code. ## The freeze repro was rotted in three ways, not one #16764 tracks four stale call sites. There were three separate problems: 1. **Stale call sites** — `execInTerminal` gained a `ptyId` and `splitActiveTerminalPane` gained a direction. (`startDockerSshRelayTarget`'s missing `testInfo` was the third; #18257 has since landed it on main.) 2. **It connected before session restore settled**, so the seeded tab never bound to a remote PTY and the terminal sat on "Connecting…" forever. 3. **It could never have passed, even once.** It waited for a one-shot `READY:` line through a 4000-char terminal window while its own 2 KB-every-8 ms flood buries that line within ~16 ms. Readiness is now keyed on the repeating `BG:` flood marker, which is strictly stronger — it proves the pane is streaming rather than merely started. It now runs end to end and prints a measurement instead of dying on a call site: ``` [freeze-repro R2] hiddenFloodMaxLagMs 2.1 bulkOpenMaxLagMs 41.5 interactionProbeMs 53.6 softFreeze false hardFreeze false ``` **It is still not CI-gateable, and the exclusion comment now says so.** The same spec on the same commit measured `bulkOpen 2575.6ms / interaction 3464.2ms` on a GitHub ubuntu runner against a 2500 ms soft budget — a ~60x spread on the number the budget reads, with the relay still streaming. That is the budget failing, not the product. The earlier draft of this comment claimed "repaired and passing", which was true only of the host it was measured on; gating this needs a host-relative oracle, not a bigger constant. ## New: a half-open link is judged, not wedged The fixture image has no `iptables` and the container has no `NET_ADMIN`, so `docker pause` is used instead — a harder case, because the container's TCP stack keeps ACKing: no FIN, no RST, and the socket looks perfectly healthy. Only an application-level probe can detect it. ``` [half-open] {"verdict":"reconnecting","verdictMs":25135,"budgetMs":90000} ``` Nothing in the suite covered the failure mode behind the "SSH hangs until I restart Orca" reports. ## New: resource accumulation measured on the remote host 6 terminals, then 5 reconnect cycles, counted on the container itself: ``` open: pts 1->6 (exactly 1/terminal), relay fds 25->30 (exactly 1/terminal) reconnect: pts flat at 6, relay procs flat at 1, node procs flat at 3 ``` `leakedMasterFdCount` is now **asserted**, not merely recorded. It counts PTY master fds held by non-relay processes: without `FD_CLOEXEC` a master is inherited by every later child, so terminal k adds k of them — the triangular signature measured as 15 across 5 terminals before the fix. #17914 patched the app and daemon and #17920 shipped the same patch to the relay host, and both are now on main, so the correct value is 0 and the probe holds it there: ``` baseline leakedMasterFdCount 0 6 terminals leakedMasterFdCount 0 (holders: only relay.js, n=6) reconnects leakedMasterFdCount 0 across all 5 cycles ``` Any growth here means the relay's node-pty rebuild did not take on that host, which is exactly what a remote-host probe exists to catch — and it is the half of #17914's claim that no unit test can reach. ## Routing Both new probes are claimed by `run-ssh-docker-e2e.mjs` (a Docker-gated spec no runner names self-skips everywhere and still reports green) **and** by the `ssh-terminal-source` route in `pr-e2e-source-routing.mjs`, so they run when the relay and SSH code they guard changes rather than only on a scheduled lane. --- config/scripts/pr-e2e-source-routing.mjs | 2 + config/scripts/run-ssh-docker-e2e.mjs | 36 +++- .../ssh-docker-bulk-open-freeze-repro.spec.ts | 60 +++++- tests/e2e/ssh-docker-half-open-link.spec.ts | 123 +++++++++++ .../ssh-docker-resource-accumulation.spec.ts | 204 ++++++++++++++++++ 5 files changed, 405 insertions(+), 20 deletions(-) create mode 100644 tests/e2e/ssh-docker-half-open-link.spec.ts create mode 100644 tests/e2e/ssh-docker-resource-accumulation.spec.ts diff --git a/config/scripts/pr-e2e-source-routing.mjs b/config/scripts/pr-e2e-source-routing.mjs index d81f4c040fe..78814b663cb 100644 --- a/config/scripts/pr-e2e-source-routing.mjs +++ b/config/scripts/pr-e2e-source-routing.mjs @@ -26,7 +26,9 @@ export const PR_E2E_SOURCE_ROUTES = [ specs: [ 'tests/e2e/pty-input-write-queue-ssh.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', + 'tests/e2e/ssh-docker-half-open-link.spec.ts', 'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts', + 'tests/e2e/ssh-docker-resource-accumulation.spec.ts', 'tests/e2e/ssh-docker-transport-drop-recovery.spec.ts', 'tests/e2e/ssh-port-forward-lifecycle.spec.ts', 'tests/e2e/ssh-reconnect-tab-destruction.spec.ts', diff --git a/config/scripts/run-ssh-docker-e2e.mjs b/config/scripts/run-ssh-docker-e2e.mjs index dddfa3e0148..9ab44b8457e 100644 --- a/config/scripts/run-ssh-docker-e2e.mjs +++ b/config/scripts/run-ssh-docker-e2e.mjs @@ -33,15 +33,31 @@ if (runtime.status !== 0) { // all. Recorded as a real gap, not as coverage living somewhere else. // ssh-codex-display-artifacts-repro.spec.ts — installs a real remote codex binary that CI // runners do not have (observed as `spawn codex ENOENT`). Runs in no CI lane at all. -// ssh-docker-bulk-open-freeze-repro.spec.ts — two reasons, both disqualifying: -// (a) it is a perf oracle, not a correctness one: SOFT_FREEZE_LAG_MS=2500 / -// HARD_FREEZE_LAG_MS=5000 measured by a renderer lag probe under a deliberate -// 5-pane output flood on a 420s budget. Same rule as ssh-docker-relay-perf above. -// (b) it is ROTTED: four call sites are out of date against terminal.ts's current -// helpers — execInTerminal gained a ptyId parameter and splitActiveTerminalPane -// gained a direction, so it cannot compile, let alone pass. Repairing it needs two -// semantic decisions (which ptyId to capture, which split direction) that change -// what the repro measures. Tracked in stablyai/orca#16764. +// ssh-docker-bulk-open-freeze-repro.spec.ts — un-rotted and now measurable, and marked +// `test.fixme` because its oracle cannot gate. Absent from this list AND skipped, so the +// two cannot drift: it is also reachable from the changed-specs lane whenever the spec +// itself is edited, and a wall-clock oracle that fails there is worth no more than one +// that fails here. +// The rot (#16764) is fixed: the stale call sites are repaired, it connects after session +// restore instead of before, and readiness keys on the repeating flood marker rather than +// a one-shot READY line the flood buries within ~16ms. It runs end to end and prints a +// measurement instead of dying on a call site. +// What it is NOT is portable. Three runs of the same measurement path: +// developer workstation: hiddenFlood 2.1ms bulkOpen 41.5ms interaction 53.6ms +// GitHub ubuntu runner A: hiddenFlood 1.5ms bulkOpen 2575.6ms interaction 3464.2ms +// GitHub ubuntu runner B: hiddenFlood 0.2ms bulkOpen 397.4ms interaction 3386.7ms +// bulkOpen swings 6.5x between two CI runs of the same code, so a fixed threshold on it is +// a coin flip; interaction sits stably ~64x over the workstation figure because it times a +// view remount, not the renderer freeze the issue reports, and only shares the budget +// constant because both are milliseconds. Every failure so far is the soft budget; hard +// has never tripped, and the relay was still streaming each time — the budget failed, not +// the product. Same rule as ssh-docker-relay-perf above. Gating needs a distribution +// first, then a host-relative oracle; a bigger constant, or a ratio picked from three +// samples, is the same arbitrary number in different clothes. +// COVERAGE GAP, recorded as such: 5 simultaneously flooding SSH panes exercise writer +// saturation, ACK/credit accounting and per-pane polling together, and nothing else covers +// that combination. Flip `test.fixme` back to `test` to run it. Tracked in +// stablyai/orca#16764. // // Why both projects: ssh-port-forward-lifecycle is @headful, which the headless project // grep-inverts away. @@ -71,8 +87,10 @@ const result = spawnSync( 'tests/e2e/ssh-ai-vault-session-history.spec.ts', 'tests/e2e/ssh-cold-activation-restore.spec.ts', 'tests/e2e/ssh-cold-hydration-gap-tab-seeding.spec.ts', + 'tests/e2e/ssh-docker-half-open-link.spec.ts', 'tests/e2e/ssh-docker-quick-open-large-listing.spec.ts', 'tests/e2e/ssh-docker-reconnect-pane-restore.spec.ts', + 'tests/e2e/ssh-docker-resource-accumulation.spec.ts', 'tests/e2e/ssh-docker-transport-drop-recovery.spec.ts', 'tests/e2e/ssh-external-image-preview.spec.ts', 'tests/e2e/ssh-lost-kill-tab-resurrection.spec.ts', diff --git a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts index a7754a02507..2de70c199d3 100644 --- a/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts +++ b/tests/e2e/ssh-docker-bulk-open-freeze-repro.spec.ts @@ -16,7 +16,7 @@ import { type DockerSshRelayTarget } from './helpers/docker-ssh-relay-target' import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' -import { waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { execInTerminal, focusLastTerminalPane, @@ -31,6 +31,7 @@ import { HARD_FREEZE_LAG_MS, SOFT_FREEZE_LAG_MS } from './helpers/remote-session const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' const REPORT_DIR = path.join(process.cwd(), 'test-results', 'freeze-repro') const SESSION_SPLITS = 5 +const FLOOD_READ_CHARS = 80_000 function shellQuote(value: string): string { return `'${value.replaceAll("'", "'\\''")}'` @@ -52,7 +53,32 @@ function continuousFloodCommand(runId: string, index: number): string { test.describe('R2 Docker SSH bulk-open freeze', () => { test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker SSH freeze repro') - test('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ + // Fixme: un-rotted and measurable, but its oracle is wall-clock and does not survive a change of + // host, so it cannot gate. Three runs of the same measurement path: + // + // host hiddenFlood bulkOpen interaction + // developer workstation 2.1ms 41.5ms 53.6ms + // GitHub ubuntu runner A 1.5ms 2575.6ms 3464.2ms + // GitHub ubuntu runner B 0.2ms 397.4ms 3386.7ms + // + // Two separate problems, and neither is the product. `bulkOpenMaxLagMs` swings 6.5x between two + // CI runs of the same code, so a fixed threshold on it is a coin flip; `interactionProbeMs` sits + // stably ~64x over the workstation figure, because it times two `setActiveView` round trips + // through a double rAF — a view remount cost, not the renderer freeze #16764 reports. It shares + // SOFT/HARD_FREEZE_LAG_MS with the lag probe only because both are milliseconds. `hardFreeze` + // has never tripped on any host; the failure is always the soft budget. + // + // Not converted to a ratio against a calibration run: with a 6.5x within-host swing on the very + // quantity that would be normalized, a threshold picked from three samples is the same arbitrary + // constant in dimensionless clothing. Gating needs a distribution first. + // + // Kept executable rather than deleted: flip `test.fixme` back to `test` to run it, which is how + // the numbers above were taken. Tracked in stablyai/orca#16764. + // + // The cost is real and is recorded in run-ssh-docker-e2e.mjs: 5 simultaneously flooding SSH panes + // exercise writer saturation, ACK/credit accounting and per-pane polling together, and nothing + // else covers that combination. It is a gap, not coverage living somewhere else. + test.fixme('bulk-open many flooding SSH terminals and measure renderer lag @freeze-repro', async ({ orcaPage, registerPostElectronShutdownCleanup }, testInfo) => { @@ -66,24 +92,36 @@ test.describe('R2 Docker SSH bulk-open freeze', () => { } }) + // Why: session restore must settle before the remote worktree is added, or the + // seeded terminal tab races tab hydration and never binds to the remote PTY. + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) await connectDockerSshRelayTarget(orcaPage, target, { remotePath: DOCKER_SSH_RELAY_REMOTE_REPO_PATH }) - await waitForSessionReady(orcaPage) - await waitForActiveWorktree(orcaPage) const runId = `${Date.now()}` // First terminal on the SSH worktree. - await waitForActiveTerminalManager(orcaPage) - await execInTerminal(orcaPage, continuousFloodCommand(runId, 0)) - await waitForTerminalOutput(orcaPage, `READY:SSH_BULK_${runId}_0`, 60_000) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + const firstPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, firstPtyId, continuousFloodCommand(runId, 0)) + // Why: the one-shot READY line is buried by the 2KB/8ms flood within ~16ms, so it is + // unobservable through the terminal read window. The repeating BG marker is the only + // stable readiness signal, and it also proves the pane is actually flooding. + await waitForTerminalOutput(orcaPage, `BG:SSH_BULK_${runId}_0:`, 60_000, FLOOD_READ_CHARS) for (let i = 1; i < SESSION_SPLITS; i += 1) { - await splitActiveTerminalPane(orcaPage) + await splitActiveTerminalPane(orcaPage, 'vertical') await focusLastTerminalPane(orcaPage) - await waitForActivePanePtyId(orcaPage, 30_000) - await execInTerminal(orcaPage, continuousFloodCommand(runId, i)) - await waitForTerminalOutput(orcaPage, `READY:SSH_BULK_${runId}_${i}`, 60_000) + const panePtyId = await waitForActivePanePtyId(orcaPage, 30_000) + await execInTerminal(orcaPage, panePtyId, continuousFloodCommand(runId, i)) + await waitForTerminalOutput( + orcaPage, + `BG:SSH_BULK_${runId}_${i}:`, + 60_000, + FLOOD_READ_CHARS + ) } // Leave the workspace view so panes go inactive while flooding. diff --git a/tests/e2e/ssh-docker-half-open-link.spec.ts b/tests/e2e/ssh-docker-half-open-link.spec.ts new file mode 100644 index 00000000000..c5b6f715dc9 --- /dev/null +++ b/tests/e2e/ssh-docker-half-open-link.spec.ts @@ -0,0 +1,123 @@ +/** + * Half-open SSH link probe. + * + * Freezes the remote host with `docker pause`. The container's TCP stack keeps + * ACKing, so the socket never sees a FIN or an RST — only the application stops + * answering. That is the wedge shape #17817 and #17838 are about: a link that + * looks perfectly healthy to TCP and can only be judged by an application probe. + * + * Requires: ORCA_E2E_SSH_DOCKER=1 and Docker available. + */ +import { execFileSync } from 'node:child_process' +import { expect, test } from './helpers/orca-app' +import { + cleanupDockerSshRelayTarget, + startDockerSshRelayTarget, + type DockerSshRelayTarget +} from './helpers/docker-ssh-relay-target' +import { connectDockerSshRelayTarget } from './helpers/docker-ssh-relay-connection' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { + execInTerminal, + waitForActivePanePtyId, + waitForActiveTerminalManager, + waitForTerminalOutput +} from './helpers/terminal' + +const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' +/** Generous: the point is that a verdict arrives at all, not its exact latency. */ +const LOST_VERDICT_BUDGET_MS = 90_000 + +function docker(args: string[]): void { + execFileSync('docker', args, { timeout: 30_000 }) +} + +async function readSshStatus( + page: Parameters[0], + targetId: string +): Promise { + return page.evaluate( + (id) => window.__store?.getState().sshConnectionStates.get(id)?.status ?? null, + targetId + ) +} + +test.describe('Docker SSH half-open link', () => { + test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker-backed SSH tests.') + test.skip(process.platform === 'win32', 'Uses docker pause against a Linux container.') + + test('declares a frozen host lost instead of wedging, and recovers @half-open', async ({ + orcaPage, + registerPostElectronShutdownCleanup + }, testInfo) => { + test.setTimeout(420_000) + let target: DockerSshRelayTarget | null = null + let paused = false + try { + target = startDockerSshRelayTarget(testInfo) + const captured = target + registerPostElectronShutdownCleanup(async () => { + cleanupDockerSshRelayTarget(captured) + }) + + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const remote = await connectDockerSshRelayTarget(orcaPage, target) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) + + const runId = String(Date.now()) + await execInTerminal(orcaPage, ptyId, `echo LIVE_${runId}`) + await waitForTerminalOutput(orcaPage, `LIVE_${runId}`, 60_000) + expect(await readSshStatus(orcaPage, remote.targetId)).toBe('connected') + + // Freeze the host: TCP keeps ACKing, the application stops answering. + docker(['pause', target.containerName]) + paused = true + const frozenAt = Date.now() + + let verdict: string | null = 'connected' + while (Date.now() - frozenAt < LOST_VERDICT_BUDGET_MS) { + verdict = await readSshStatus(orcaPage, remote.targetId) + if (verdict !== 'connected') { + break + } + await orcaPage.waitForTimeout(1_000) + } + const verdictMs = Date.now() - frozenAt + console.log( + `[half-open] ${JSON.stringify({ verdict, verdictMs, budgetMs: LOST_VERDICT_BUDGET_MS })}` + ) + + docker(['unpause', target.containerName]) + paused = false + + // Why this is the assertion: a wedged client sits on `connected` forever and + // never offers the user a reconnect. Any non-connected verdict is a pass. + expect( + verdict, + `client never left "connected" ${verdictMs}ms after the host was frozen` + ).not.toBe('connected') + + // The link must be usable again once the host thaws. + await expect + .poll(() => readSshStatus(orcaPage, remote.targetId), { timeout: 120_000 }) + .toBe('connected') + const recoveredPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, recoveredPtyId, `echo RECOVERED_${runId}`) + await waitForTerminalOutput(orcaPage, `RECOVERED_${runId}`, 90_000) + } finally { + if (target && paused) { + try { + docker(['unpause', target.containerName]) + } catch { + // The container may already be gone; cleanup below is authoritative. + } + } + if (target) { + cleanupDockerSshRelayTarget(target) + } + } + }) +}) diff --git a/tests/e2e/ssh-docker-resource-accumulation.spec.ts b/tests/e2e/ssh-docker-resource-accumulation.spec.ts new file mode 100644 index 00000000000..f028ef0d2ab --- /dev/null +++ b/tests/e2e/ssh-docker-resource-accumulation.spec.ts @@ -0,0 +1,204 @@ +/** + * Adversarial resource-accumulation probe for the SSH relay. + * + * Covers claims no unit test can reach, measured on the remote host itself: + * - #17914 / #17920: PTY master fds are close-on-exec, so /dev/pts and the + * relay's fd table must not grow per terminal beyond the terminals + * themselves. #17914 patches the app and terminal daemon; the relay installs + * node-pty from npm on the remote host, so #17920 ships the same patch as a + * relay asset and rebuilds there. Only a remote host can judge that second + * half, which is why leakedMasterFdCount is measured on the container. + * - #17817/#17821/#17831: repeated disconnect/reconnect must not accumulate + * relay processes, orphan PTYs, or fds. + * + * Requires: ORCA_E2E_SSH_DOCKER=1 and Docker available. + */ +import { expect, test } from './helpers/orca-app' +import { + cleanupDockerSshRelayTarget, + execDockerSshRelayTargetCommand, + startDockerSshRelayTarget, + type DockerSshRelayTarget +} from './helpers/docker-ssh-relay-target' +import { + connectDockerSshRelayTarget, + reconnectDockerSshRelayTarget +} from './helpers/docker-ssh-relay-connection' +import { readDockerSshRelayProcessSnapshots } from './helpers/docker-ssh-relay-processes' +import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' +import { + execInTerminal, + focusLastTerminalPane, + splitActiveTerminalPane, + waitForActivePanePtyId, + waitForActiveTerminalManager, + waitForTerminalOutput +} from './helpers/terminal' + +const RUN_DOCKER_SSH = process.env.ORCA_E2E_SSH_DOCKER === '1' +const TERMINAL_COUNT = 6 +const RECONNECT_CYCLES = 5 + +type RemoteResourceSample = { + ptsCount: number + relayFdCount: number + relayProcessCount: number + nodeProcessCount: number + /** + * PTY master fds held by processes other than the relay. A master fd without + * FD_CLOEXEC is inherited by every later-spawned child, so this grows ~N^2/2 + * across N terminals when the close-on-exec fix is absent (#17914). + */ + leakedMasterFdCount: number +} + +const COUNT_LEAKED_MASTER_FDS = [ + 'total=0', + 'for p in $(ls /proc | grep -E "^[0-9]+$"); do', + ' cmd=$(tr "\\0" " " < /proc/$p/cmdline 2>/dev/null || true)', + ' case "$cmd" in *relay.js*) continue;; esac', + ' n=$(ls -l /proc/$p/fd 2>/dev/null | grep -c "ptmx" || true)', + ' total=$((total+n))', + 'done', + 'echo $total' +].join('\n') + +const DESCRIBE_MASTER_FD_HOLDERS = [ + 'for p in $(ls /proc | grep -E "^[0-9]+$"); do', + ' cmd=$(tr "\\0" " " < /proc/$p/cmdline 2>/dev/null || true)', + ' n=$(ls -l /proc/$p/fd 2>/dev/null | grep -c "ptmx" || true)', + ' if [ "$n" != "0" ]; then echo "$p n=$n cmd=$cmd"; fi', + 'done' +].join('\n') + +function sampleRemoteResources(target: DockerSshRelayTarget): RemoteResourceSample { + const groups = readDockerSshRelayProcessSnapshots(target) + // Why: fd growth is only meaningful against the relay that owns the PTYs, so read + // the table of every relay group and sum, rather than assuming a single relay. + const relayFdCount = groups.reduce((total, group) => { + const raw = execDockerSshRelayTargetCommand( + target, + `ls /proc/${group.relayPid}/fd 2>/dev/null | wc -l` + ) + return total + Number(raw.trim() || '0') + }, 0) + const ptsCount = Number( + execDockerSshRelayTargetCommand(target, 'ls /dev/pts | grep -c "^[0-9]" || true').trim() || '0' + ) + const nodeProcessCount = Number( + execDockerSshRelayTargetCommand(target, 'pgrep -c node || true').trim() || '0' + ) + const leakedMasterFdCount = Number( + execDockerSshRelayTargetCommand(target, COUNT_LEAKED_MASTER_FDS).trim() || '0' + ) + return { + ptsCount, + relayFdCount, + relayProcessCount: groups.length, + nodeProcessCount, + leakedMasterFdCount + } +} + +test.describe('Docker SSH relay resource accumulation', () => { + test.skip(!RUN_DOCKER_SSH, 'Set ORCA_E2E_SSH_DOCKER=1 to run Docker-backed SSH tests.') + test.skip(process.platform === 'win32', 'Uses POSIX /proc and /dev/pts probes.') + + test('does not accumulate pts devices, relay fds, or relay processes @resource-accumulation', async ({ + orcaPage, + registerPostElectronShutdownCleanup + }, testInfo) => { + test.setTimeout(420_000) + let target: DockerSshRelayTarget | null = null + try { + target = startDockerSshRelayTarget(testInfo) + const captured = target + registerPostElectronShutdownCleanup(async () => { + cleanupDockerSshRelayTarget(captured) + }) + + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const remote = await connectDockerSshRelayTarget(orcaPage, target) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + + const runId = String(Date.now()) + const firstPtyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, firstPtyId, `echo PANE_READY_${runId}_0`) + await waitForTerminalOutput(orcaPage, `PANE_READY_${runId}_0`, 60_000) + + const baseline = sampleRemoteResources(target) + const samples: RemoteResourceSample[] = [] + + // Open N more terminals; each must cost a bounded, roughly constant amount. + for (let index = 1; index < TERMINAL_COUNT; index += 1) { + await splitActiveTerminalPane(orcaPage, 'vertical') + await focusLastTerminalPane(orcaPage) + const ptyId = await waitForActivePanePtyId(orcaPage, 60_000) + await execInTerminal(orcaPage, ptyId, `echo PANE_READY_${runId}_${index}`) + await waitForTerminalOutput(orcaPage, `PANE_READY_${runId}_${index}`, 60_000) + samples.push(sampleRemoteResources(target)) + } + + const withTerminals = samples.at(-1)! + const openedTerminals = TERMINAL_COUNT - 1 + const ptsGrowth = withTerminals.ptsCount - baseline.ptsCount + const fdGrowth = withTerminals.relayFdCount - baseline.relayFdCount + const fdPerTerminal = fdGrowth / openedTerminals + + console.log( + `[resource-accumulation] open ${JSON.stringify({ + baseline, + withTerminals, + openedTerminals, + ptsGrowth, + fdGrowth, + fdPerTerminal + })}` + ) + + console.log( + `[resource-accumulation] master-fd holders\n${execDockerSshRelayTargetCommand( + target, + DESCRIBE_MASTER_FD_HOLDERS + )}` + ) + + // Each remote terminal legitimately costs one pts device. + expect(ptsGrowth).toBeLessThanOrEqual(openedTerminals) + // Why: a master fd that leaks into every child would push this well past a + // small constant per terminal. Allow slack for the relay's own bookkeeping. + expect(fdPerTerminal).toBeLessThanOrEqual(4) + // Why an equality-shaped bound rather than slack: a master fd that is not close-on-exec is + // inherited by every later child, so terminal k adds k of them (1+2+3+4+5 = 15 was the + // observed pre-fix signature). With #17914's patch reaching the relay host through #17920 + // no non-relay process holds a master at all, so any growth here means the relay's node-pty + // rebuild did not take on this host — which is exactly what this probe exists to catch. + expect(withTerminals.leakedMasterFdCount).toBeLessThanOrEqual(baseline.leakedMasterFdCount) + expect(withTerminals.relayProcessCount).toBe(1) + + // Repeated reconnects must not accumulate anything on the host. + const reconnectSamples: RemoteResourceSample[] = [] + for (let cycle = 0; cycle < RECONNECT_CYCLES; cycle += 1) { + await reconnectDockerSshRelayTarget(orcaPage, remote.targetId) + reconnectSamples.push(sampleRemoteResources(target)) + } + console.log(`[resource-accumulation] reconnects ${JSON.stringify(reconnectSamples)}`) + + const first = reconnectSamples[0] + const last = reconnectSamples.at(-1)! + expect(last.relayProcessCount).toBe(1) + // Why: the interesting failure is monotonic growth across cycles, not the + // absolute count, so compare the last cycle against the first. + expect(last.ptsCount).toBeLessThanOrEqual(first.ptsCount) + expect(last.relayFdCount).toBeLessThanOrEqual(first.relayFdCount + 4) + expect(last.nodeProcessCount).toBeLessThanOrEqual(first.nodeProcessCount) + expect(last.leakedMasterFdCount).toBeLessThanOrEqual(first.leakedMasterFdCount) + } finally { + if (target) { + cleanupDockerSshRelayTarget(target) + } + } + }) +}) From b8b7a6be9d4089b8c101843f4430df5d6dbe5497 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:04:11 -0700 Subject: [PATCH 124/398] fix(activity): persist the agents unread filter and grouping (#18255) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(activity): persist the agents unread filter and grouping The Agents view's "Show unread threads only" toggle and Group-by select were plain component state in the sidebar and the Activity page, so both reset on every mount — including app restart — while their neighbours in the same toolbar (compact rows, show child agents) survived via the persisted UI store. Promote both to `agentsReadFilter` / `agentsGroupBy` persisted UI preferences, wired through the same seams as `agentsCompactMode`: shared type, default, strict client RPC schema, pairing-local field census, web read pin, store contract/actions, and hydration normalizers that reject unknown values. Both consumers now read the store, so the sidebar and the Activity page share one filter the way they already share compact mode. * refactor: centralize thread filter value domains Establish filter and groupby value domains as the single source of truth, with types derived from them to prevent drift between valid values and their normalizers. Extract common validation logic into a shared isMember helper to keep the two normalization functions in sync. * refactor: centralize thread filter value domains Consolidate filter value definitions in agents-view-thread-filters and use them in Zod schema validation to ensure consistent, persistent serialization of filter state. --- .../client-ui-pairing-local-fields.test.ts | 2 ++ .../runtime/rpc/methods/client-ui-schemas.ts | 6 +++++ .../activity/ActivityPrototypePage.tsx | 12 ++++------ .../activity/activity-thread-types.ts | 4 ++-- .../components/sidebar/SidebarAgentsList.tsx | 2 +- src/renderer/src/components/sidebar/index.tsx | 7 +++--- ...ui-hydration-workspace-preferences.test.ts | 21 +++++++++++++++++ .../ui/ui-slice-contract-preferences.ts | 6 +++++ .../slices/ui/ui-slice-hydration-actions.ts | 6 +++++ .../slices/ui/ui-slice-preference-actions.ts | 14 +++++++++++ .../web-preference-normalization.ts | 2 ++ .../src/web/web-preload-api-ui.test.ts | 4 ++++ src/shared/agents-view-thread-filters.ts | 23 +++++++++++++++++++ src/shared/constants.ts | 3 +++ src/shared/pairing-local-ui-fields.test.ts | 2 ++ src/shared/pairing-local-ui-fields.ts | 2 ++ src/shared/persisted-ui-state-types.ts | 6 +++++ src/shared/ui-chrome-types.ts | 4 ++++ 18 files changed, 113 insertions(+), 13 deletions(-) create mode 100644 src/shared/agents-view-thread-filters.ts diff --git a/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts b/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts index d6c08ad44cf..2e7ac3e93b1 100644 --- a/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts +++ b/src/main/runtime/rpc/methods/client-ui-pairing-local-fields.test.ts @@ -49,6 +49,8 @@ describe('client UI RPC pairing-local field seams', () => { agentsFilterRepoIds: ['repo-a'], agentsShowChildAgents: true, agentsCompactMode: false, + agentsReadFilter: 'unread', + agentsGroupBy: 'project', activityClearedAtByPaneKey: { 'tab-1:leaf-1': 123 }, manuallyUnreadTurnsByPaneKey: { 'tab-1:leaf-1': 321 } } diff --git a/src/main/runtime/rpc/methods/client-ui-schemas.ts b/src/main/runtime/rpc/methods/client-ui-schemas.ts index edb3f651c32..c772ba2fe67 100644 --- a/src/main/runtime/rpc/methods/client-ui-schemas.ts +++ b/src/main/runtime/rpc/methods/client-ui-schemas.ts @@ -3,6 +3,10 @@ import { isFeatureInteractionId, type FeatureInteractionId } from '../../../../shared/feature-interactions' +import { + ACTIVITY_GROUP_BY_VALUES, + THREAD_READ_FILTER_VALUES +} from '../../../../shared/agents-view-thread-filters' import { isFeatureTipId } from '../../../../shared/feature-tips' import { isReleaseChannel, type ReleaseChannel } from '../../../../shared/release-channel' import { @@ -126,6 +130,8 @@ const UiUpdateFields = z agentsFilterRepoIds: StringArray.optional(), agentsShowChildAgents: z.boolean().optional(), agentsCompactMode: z.boolean().optional(), + agentsReadFilter: z.enum(THREAD_READ_FILTER_VALUES).optional(), + agentsGroupBy: z.enum(ACTIVITY_GROUP_BY_VALUES).optional(), workspaceHostOrder: z.array(z.string()).optional(), automationHostFilter: z .union([ diff --git a/src/renderer/src/components/activity/ActivityPrototypePage.tsx b/src/renderer/src/components/activity/ActivityPrototypePage.tsx index 83581d4cf10..44c1f398122 100644 --- a/src/renderer/src/components/activity/ActivityPrototypePage.tsx +++ b/src/renderer/src/components/activity/ActivityPrototypePage.tsx @@ -21,22 +21,20 @@ import { useActivityTerminalLoadingLabel, useActivityTerminalPortalStatus } from './activity-terminal-portal-status' -import type { - ActivityGroupBy, - ActivityTerminalPortalSlotId, - ThreadReadFilter -} from './activity-thread-types' +import type { ActivityTerminalPortalSlotId } from './activity-thread-types' export * from './activity-prototype-page-exports' export default function ActivityPrototypePage(): React.JSX.Element { - const [readFilter, setReadFilter] = useState('all') - const [groupBy, setGroupBy] = useState('status') const [query, setQuery] = useState('') const activityFilterInputRef = useRef(null) // Why: bounds auto mark-read to one acknowledgement per selected thread turn. const autoAcknowledgedTurnRef = useRef(null) // Why store-backed: persisted preferences shared with the sidebar agents list. + const readFilter = useAppStore((s) => s.agentsReadFilter) + const setReadFilter = useAppStore((s) => s.setAgentsReadFilter) + const groupBy = useAppStore((s) => s.agentsGroupBy) + const setGroupBy = useAppStore((s) => s.setAgentsGroupBy) const compactMode = useAppStore((s) => s.agentsCompactMode) const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) diff --git a/src/renderer/src/components/activity/activity-thread-types.ts b/src/renderer/src/components/activity/activity-thread-types.ts index 7225a71b935..ed73642c8eb 100644 --- a/src/renderer/src/components/activity/activity-thread-types.ts +++ b/src/renderer/src/components/activity/activity-thread-types.ts @@ -9,8 +9,8 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { Worktree } from '../../../../shared/worktree/types' import type { ActivityPortalReadinessStatus } from './activity-portal-readiness-oscillation' -export type ThreadReadFilter = 'all' | 'unread' -export type ActivityGroupBy = 'none' | 'status' | 'project' | 'worktree' | 'agent' +export type { ActivityGroupBy, ThreadReadFilter } from '../../../../shared/ui-chrome-types' + export type ActivityEventState = Extract export type ActivityHookLiveAgentState = Extract< AgentStatusState, diff --git a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx index 5d82b56253a..ca5a8c5bce1 100644 --- a/src/renderer/src/components/sidebar/SidebarAgentsList.tsx +++ b/src/renderer/src/components/sidebar/SidebarAgentsList.tsx @@ -41,7 +41,7 @@ export default function SidebarAgentsList({ }: SidebarAgentsListProps): React.JSX.Element { // The search row is owned here and mounts conditionally, so subscribe this host to locale changes. useTranslation() - // Why store-backed: these are persisted preferences (agents* UI fields), unlike the momentary read filter/search. + // Why store-backed: these are persisted preferences (agents* UI fields), unlike the momentary search. const compactMode = useAppStore((s) => s.agentsCompactMode) const setCompactMode = useAppStore((s) => s.setAgentsCompactMode) const showChildAgents = useAppStore((s) => s.agentsShowChildAgents) diff --git a/src/renderer/src/components/sidebar/index.tsx b/src/renderer/src/components/sidebar/index.tsx index 90c31999bd3..8b837b5ef39 100644 --- a/src/renderer/src/components/sidebar/index.tsx +++ b/src/renderer/src/components/sidebar/index.tsx @@ -11,7 +11,6 @@ import WorkspaceKanbanDrawer from './WorkspaceKanbanDrawer' import type { VirtualizedScrollAnchor } from '@/hooks/useVirtualizedScrollAnchor' import { cn } from '@/lib/utils' import { FolderPlus, Loader2 } from 'lucide-react' -import type { ActivityGroupBy, ThreadReadFilter } from '@/components/activity/activity-thread-types' import { ActivityThreadCollapseContext } from '@/components/activity/activity-thread-collapse-context' import { useSidebarProjectDrop } from './useSidebarProjectDrop' import { useWorkspaceBoardPanel } from './useWorkspaceBoardPanel' @@ -59,8 +58,10 @@ function Sidebar({ const showAgentDashboard = settings?.experimentalAgentDashboardPopout === true const agentDashboardDrawerOpen = useAppStore((s) => s.agentDashboardDrawerOpen) const setAgentDashboardDrawerOpen = useAppStore((s) => s.setAgentDashboardDrawerOpen) - const [agentReadFilter, setAgentReadFilter] = React.useState('all') - const [agentGroupBy, setAgentGroupBy] = React.useState('status') + const agentReadFilter = useAppStore((s) => s.agentsReadFilter) + const setAgentReadFilter = useAppStore((s) => s.setAgentsReadFilter) + const agentGroupBy = useAppStore((s) => s.agentsGroupBy) + const setAgentGroupBy = useAppStore((s) => s.setAgentsGroupBy) const [agentQuery, setAgentQuery] = React.useState('') const [agentOptionsTarget, setAgentOptionsTarget] = React.useState(null) const agentsScrollTopRef = React.useRef(0) diff --git a/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts b/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts index 642297bb109..6d3a559c582 100644 --- a/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts +++ b/src/renderer/src/store/slices/ui-hydration-workspace-preferences.test.ts @@ -551,6 +551,27 @@ describe('createUISlice hydratePersistedUI', () => { expect(store.getState().agentsFilterRepoIds).toEqual([]) expect(store.getState().agentsShowChildAgents).toBe(false) expect(store.getState().agentsCompactMode).toBe(true) + expect(store.getState().agentsReadFilter).toBe('all') + expect(store.getState().agentsGroupBy).toBe('status') + }) + + it('restores the persisted agents read filter and grouping, rejecting unknown values', () => { + const store = createUIStore() + + store + .getState() + .hydratePersistedUI(makePersistedUI({ agentsReadFilter: 'unread', agentsGroupBy: 'project' })) + expect(store.getState().agentsReadFilter).toBe('unread') + expect(store.getState().agentsGroupBy).toBe('project') + + store.getState().hydratePersistedUI( + makePersistedUI({ + agentsReadFilter: 'bogus' as unknown as PersistedUIState['agentsReadFilter'], + agentsGroupBy: 'bogus' as unknown as PersistedUIState['agentsGroupBy'] + }) + ) + expect(store.getState().agentsReadFilter).toBe('all') + expect(store.getState().agentsGroupBy).toBe('status') }) it('sanitizes malformed agents repo filters before the repo catalog loads', () => { diff --git a/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts b/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts index 2e764e5cd79..1162f399cd5 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-contract-preferences.ts @@ -1,9 +1,11 @@ import type { PersistedUIState } from '../../../../../shared/persisted-ui-state-types' import type { + ActivityGroupBy, AgentActivityDisplayMode, ManualRepoOrderEntry, ProjectOrderBy, StatusBarItem, + ThreadReadFilter, WorktreeCardMode, WorktreeCardProperty, WorkspaceHostOrder, @@ -71,6 +73,10 @@ export type UISlicePreferences = { setAgentsShowChildAgents: (v: boolean) => void agentsCompactMode: boolean setAgentsCompactMode: (v: boolean) => void + agentsReadFilter: ThreadReadFilter + setAgentsReadFilter: (v: ThreadReadFilter) => void + agentsGroupBy: ActivityGroupBy + setAgentsGroupBy: (v: ActivityGroupBy) => void collapsedGroups: Set toggleCollapsedGroup: (key: string) => void worktreeCardProperties: WorktreeCardProperty[] diff --git a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts index f44774b3770..08adff8e2a3 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-hydration-actions.ts @@ -20,6 +20,10 @@ import { normalizeWorktreeCardProperties, normalizeAgentActivityDisplayMode } from '../../../../../shared/constants' +import { + normalizeActivityGroupBy, + normalizeThreadReadFilter +} from '../../../../../shared/agents-view-thread-filters' import { clampWorkspaceBoardColumnWidth, clampWorkspaceBoardOpacity, @@ -182,6 +186,8 @@ export function createUiHydrationActions(set: UISliceSet, _get: UISliceGet): Par ), agentsShowChildAgents: ui.agentsShowChildAgents === true, agentsCompactMode: ui.agentsCompactMode !== false, + agentsReadFilter: normalizeThreadReadFilter(ui.agentsReadFilter), + agentsGroupBy: normalizeActivityGroupBy(ui.agentsGroupBy), collapsedGroups: new Set(ui.collapsedGroups ?? []), uiZoomLevel: ui.uiZoomLevel ?? 0, editorFontZoomLevel: ui.editorFontZoomLevel ?? 0, diff --git a/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts b/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts index 971c64dfa52..9a9dc5947b1 100644 --- a/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts +++ b/src/renderer/src/store/slices/ui/ui-slice-preference-actions.ts @@ -1,4 +1,8 @@ import type { UISlice, UISliceGet, UISliceSet } from './ui-slice-contract' +import { + DEFAULT_AGENTS_GROUP_BY, + DEFAULT_AGENTS_READ_FILTER +} from '../../../../../shared/agents-view-thread-filters' import { DEFAULT_AGENT_ACTIVITY_DISPLAY_MODE, DEFAULT_SHOW_SLEEPING_WORKSPACES, @@ -170,6 +174,16 @@ export function createUiPreferenceActions(set: UISliceSet, get: UISliceGet): Par set({ agentsCompactMode: v }) window.api.ui.set({ agentsCompactMode: v }).catch(console.error) }, + agentsReadFilter: DEFAULT_AGENTS_READ_FILTER, + setAgentsReadFilter: (v) => { + set({ agentsReadFilter: v }) + window.api.ui.set({ agentsReadFilter: v }).catch(console.error) + }, + agentsGroupBy: DEFAULT_AGENTS_GROUP_BY, + setAgentsGroupBy: (v) => { + set({ agentsGroupBy: v }) + window.api.ui.set({ agentsGroupBy: v }).catch(console.error) + }, collapsedGroups: new Set(), toggleCollapsedGroup: (key) => diff --git a/src/renderer/src/web/preload-api/web-preference-normalization.ts b/src/renderer/src/web/preload-api/web-preference-normalization.ts index 9ae3ad00d4a..be7e02f029a 100644 --- a/src/renderer/src/web/preload-api/web-preference-normalization.ts +++ b/src/renderer/src/web/preload-api/web-preference-normalization.ts @@ -73,6 +73,8 @@ export function mergeHostWebUIState( agentsFilterRepoIds: local.agentsFilterRepoIds, agentsShowChildAgents: local.agentsShowChildAgents, agentsCompactMode: local.agentsCompactMode, + agentsReadFilter: local.agentsReadFilter, + agentsGroupBy: local.agentsGroupBy, activityClearedAtByPaneKey: local.activityClearedAtByPaneKey, manuallyUnreadTurnsByPaneKey: local.manuallyUnreadTurnsByPaneKey } satisfies Record & Partial diff --git a/src/renderer/src/web/web-preload-api-ui.test.ts b/src/renderer/src/web/web-preload-api-ui.test.ts index 94e9ba8e1eb..a01ffbad567 100644 --- a/src/renderer/src/web/web-preload-api-ui.test.ts +++ b/src/renderer/src/web/web-preload-api-ui.test.ts @@ -469,6 +469,8 @@ describe('web UI preload API', () => { agentsFilterRepoIds: ['repo-b'], agentsShowChildAgents: true, agentsCompactMode: false, + agentsReadFilter: 'unread', + agentsGroupBy: 'project', activityClearedAtByPaneKey: { 'tab-1:leaf-1': 123 }, manuallyUnreadTurnsByPaneKey: { 'tab-1:leaf-1': 321 } } @@ -481,6 +483,8 @@ describe('web UI preload API', () => { agentsFilterRepoIds: ['repo-a'], agentsShowChildAgents: false, agentsCompactMode: true, + agentsReadFilter: 'all', + agentsGroupBy: 'status', activityClearedAtByPaneKey: { 'tab-2:leaf-2': 456 }, manuallyUnreadTurnsByPaneKey: { 'tab-2:leaf-2': 654 } } diff --git a/src/shared/agents-view-thread-filters.ts b/src/shared/agents-view-thread-filters.ts new file mode 100644 index 00000000000..7f77015078b --- /dev/null +++ b/src/shared/agents-view-thread-filters.ts @@ -0,0 +1,23 @@ +/** The two filter value domains, in menu order. The types below, the client + * schema's `z.enum`s and the normalizers all derive from these, so a new value + * cannot drift out of any of them. */ +export const THREAD_READ_FILTER_VALUES = ['all', 'unread'] as const +export const ACTIVITY_GROUP_BY_VALUES = ['none', 'status', 'project', 'worktree', 'agent'] as const + +export type ThreadReadFilter = (typeof THREAD_READ_FILTER_VALUES)[number] +export type ActivityGroupBy = (typeof ACTIVITY_GROUP_BY_VALUES)[number] + +export const DEFAULT_AGENTS_READ_FILTER: ThreadReadFilter = 'all' +export const DEFAULT_AGENTS_GROUP_BY: ActivityGroupBy = 'status' + +export function normalizeThreadReadFilter(value: unknown): ThreadReadFilter { + return isMember(THREAD_READ_FILTER_VALUES, value) ? value : DEFAULT_AGENTS_READ_FILTER +} + +export function normalizeActivityGroupBy(value: unknown): ActivityGroupBy { + return isMember(ACTIVITY_GROUP_BY_VALUES, value) ? value : DEFAULT_AGENTS_GROUP_BY +} + +function isMember(catalog: readonly T[], value: unknown): value is T { + return typeof value === 'string' && (catalog as readonly string[]).includes(value) +} diff --git a/src/shared/constants.ts b/src/shared/constants.ts index 9d26b140dec..bb2f5940f0e 100644 --- a/src/shared/constants.ts +++ b/src/shared/constants.ts @@ -11,6 +11,7 @@ import { DEFAULT_STATUS_BAR_ITEMS } from './status-bar-defaults' import type { VoiceSettings } from './speech-types' import { cloneDefaultWorkspaceStatuses } from './workspace-statuses' import { DEFAULT_WORKTREE_CARD_PROPERTIES } from './worktree/card-properties' +import { DEFAULT_AGENTS_GROUP_BY, DEFAULT_AGENTS_READ_FILTER } from './agents-view-thread-filters' import { DEFAULT_USAGE_PERCENTAGE_DISPLAY } from './usage-percentage-display' import { DEFAULT_STATUS_BAR_USAGE_MODE } from './status-bar-usage-mode' import { buildDefaultSettings } from './default-global-settings' @@ -273,6 +274,8 @@ export function getDefaultUIState(): PersistedUIState { agentsFilterRepoIds: [], agentsShowChildAgents: false, agentsCompactMode: true, + agentsReadFilter: DEFAULT_AGENTS_READ_FILTER, + agentsGroupBy: DEFAULT_AGENTS_GROUP_BY, collapsedGroups: [], uiZoomLevel: 0, editorFontZoomLevel: 0, diff --git a/src/shared/pairing-local-ui-fields.test.ts b/src/shared/pairing-local-ui-fields.test.ts index ffc7baafe07..35bd5f33331 100644 --- a/src/shared/pairing-local-ui-fields.test.ts +++ b/src/shared/pairing-local-ui-fields.test.ts @@ -14,6 +14,8 @@ describe('pairing-local UI fields', () => { 'agentsFilterRepoIds', 'agentsShowChildAgents', 'agentsCompactMode', + 'agentsReadFilter', + 'agentsGroupBy', 'activityClearedAtByPaneKey', 'manuallyUnreadTurnsByPaneKey' ]) diff --git a/src/shared/pairing-local-ui-fields.ts b/src/shared/pairing-local-ui-fields.ts index d925c3094d8..f642bb2b821 100644 --- a/src/shared/pairing-local-ui-fields.ts +++ b/src/shared/pairing-local-ui-fields.ts @@ -17,6 +17,8 @@ export const PAIRING_LOCAL_UI_FIELDS = [ 'agentsFilterRepoIds', 'agentsShowChildAgents', 'agentsCompactMode', + 'agentsReadFilter', + 'agentsGroupBy', 'activityClearedAtByPaneKey', 'manuallyUnreadTurnsByPaneKey' ] as const satisfies readonly (keyof PersistedUIState)[] diff --git a/src/shared/persisted-ui-state-types.ts b/src/shared/persisted-ui-state-types.ts index 14e97bc34fc..b6d40480017 100644 --- a/src/shared/persisted-ui-state-types.ts +++ b/src/shared/persisted-ui-state-types.ts @@ -8,6 +8,7 @@ import type { StatusBarUsageMode } from './status-bar-usage-mode' import type { PersistedTrustedOrcaHooks } from './orca-yaml-hook-types' import type { CustomPet } from './pet-types' import type { + ActivityGroupBy, AgentActivityDisplayMode, ManualRepoOrderEntry, ProjectOrderBy, @@ -15,6 +16,7 @@ import type { RightSidebarTab, StatusBarItem, TaskResumeState, + ThreadReadFilter, TopLevelView, VisibleWorkspaceHostIds, WorkspaceHostOrder, @@ -81,6 +83,10 @@ export type PersistedUIState = { agentsShowChildAgents?: boolean /** Agents-view compact thread rows. Absent means on. */ agentsCompactMode?: boolean + /** Agents-view unread-only thread filter. Absent means 'all'. */ + agentsReadFilter?: ThreadReadFilter + /** Agents-view thread grouping. Absent means 'status'. */ + agentsGroupBy?: ActivityGroupBy collapsedGroups: string[] uiZoomLevel: number editorFontZoomLevel: number diff --git a/src/shared/ui-chrome-types.ts b/src/shared/ui-chrome-types.ts index 90a12cf5b02..fe6157d870b 100644 --- a/src/shared/ui-chrome-types.ts +++ b/src/shared/ui-chrome-types.ts @@ -49,6 +49,10 @@ export type WorktreeCardMode = 'Default' | 'Compact' export type AgentActivityDisplayMode = 'compact' | 'full' +// Re-exported so existing importers keep one home for UI chrome types; the +// value domain lives with the normalizers that police it. +export type { ActivityGroupBy, ThreadReadFilter } from './agents-view-thread-filters' + export type StatusBarItem = | 'claude' | 'codex' From 4b2d9b5aac16c87304068a4adfa5933d005ffbac Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:34:44 -0700 Subject: [PATCH 125/398] fix(path): stop seeded user bin dirs from outranking the inherited PATH (#18265) `patchPackagedProcessPath` prepends every seeded directory, so `~/bin` and `~/.local/bin` land ahead of the PATH a GUI-launched Electron inherited. That does more than make a tool findable, which is what seeding is for -- it re-ranks binaries the user already has, and those two directories are user-writable and can hold a wrapper for any system tool. On the #18234 reporter's box `~/.local/bin/gh` wraps `mise x gh -- gh`. Seeded ahead of /usr/bin we ran the wrapper where their own shell ran the real binary, and the wrapper's inner bare `gh` resolved back to itself. Measured in an Ubuntu 24.04 container: with their shell's ordering the chain exits in 22ms; with ours it never terminates and creates ~1,500 processes/second. Seed order now follows the rule the WSL twin already documents in posix-version-manager-bin-dirs.ts -- append, never prepend, because a login PATH that did resolve is authoritative. Version-manager shim dirs keep leading, since an nvm/mise/asdf user's runtime must still beat a system install; the generic user bin dirs move behind the inherited PATH. `getVersionManagerBinPaths` carries `~/bin` and `~/.local/bin` too (bun and pnpm install there), so they are filtered out of the leading group by name rather than by which list produced them. --- src/main/startup/configure-process.test.ts | 60 ++++++++++++++++++++++ src/main/startup/configure-process.ts | 47 ++++++++++++----- 2 files changed, 95 insertions(+), 12 deletions(-) diff --git a/src/main/startup/configure-process.test.ts b/src/main/startup/configure-process.test.ts index 4747f797cf9..ef7ec6a9b69 100644 --- a/src/main/startup/configure-process.test.ts +++ b/src/main/startup/configure-process.test.ts @@ -137,6 +137,66 @@ describe('patchPackagedProcessPath', () => { expect(segments).toContain('/usr/local/bin') }) + // Why this ordering is load-bearing (#18234): a seed exists so a GUI-launched + // Electron can *find* a tool, not to re-rank tools the user already has. + // `~/.local/bin` is user-writable and can hold a wrapper for any system tool. + // The reporter's `~/.local/bin/gh` wrapped `mise x gh -- gh`; seeded ahead of + // /usr/bin it ran instead of the real gh, and the wrapper's inner bare `gh` + // resolved back to itself. Measured in a container: with the login shell's + // ordering that chain exits in 22ms, with the seeded ordering it never + // terminates and creates ~1,300 processes/second. + it('never lets a seeded user dir overtake a system dir already on PATH', async () => { + const { app } = await import('electron') + const { patchPackagedProcessPath } = await import('./configure-process') + + setPlatform('linux') + Object.defineProperty(app, 'isPackaged', { configurable: true, value: true }) + process.env.HOME = '/home/tester' + process.env.PATH = '/usr/local/bin:/usr/bin:/bin' + + patchPackagedProcessPath() + + const segments = (process.env.PATH ?? '').split(':') + const localBin = segments.indexOf(join('/home/tester', '.local/bin')) + // Still reachable — that is what the seeding is for (#829). + expect(localBin).toBeGreaterThan(-1) + for (const systemDir of ['/usr/bin', '/bin', '/usr/local/bin']) { + expect(segments.indexOf(systemDir)).toBeLessThan(localBin) + } + expect(segments.indexOf(join('/home/tester', 'bin'))).toBeGreaterThan( + segments.indexOf('/usr/bin') + ) + }) + + it('keeps version-manager shims ahead of the inherited PATH', async () => { + const { app } = await import('electron') + const { patchPackagedProcessPath } = await import('./configure-process') + const { getVersionManagerBinPaths } = await import('../../shared/node-cli-command-resolution') + + setPlatform('linux') + Object.defineProperty(app, 'isPackaged', { configurable: true, value: true }) + process.env.HOME = '/home/tester' + process.env.PATH = '/usr/bin:/bin' + + patchPackagedProcessPath() + + const segments = (process.env.PATH ?? '').split(':') + const genericUserBinDirs = [join('/home/tester', 'bin'), join('/home/tester', '.local/bin')] + const seeded = getVersionManagerBinPaths({ platform: 'linux', homePath: '/home/tester' }) + const shimDirs = seeded.filter((dir) => !genericUserBinDirs.includes(dir)) + expect(shimDirs).not.toHaveLength(0) + // Why these keep leading: an nvm/mise/asdf user's runtime must beat a + // system install, which is the reason this seeding is ordered at all. + for (const dir of shimDirs) { + expect(segments.indexOf(dir)).toBeLessThan(segments.indexOf('/usr/bin')) + } + // Why these do not: the same list carries the generic user bin dirs, which + // hold whatever was last installed there rather than a managed toolchain. + for (const dir of genericUserBinDirs) { + expect(segments.indexOf(dir)).toBeGreaterThan(segments.indexOf('/usr/bin')) + } + }) + it('leaves PATH untouched when the app is not packaged', async () => { const { app } = await import('electron') const { patchPackagedProcessPath } = await import('./configure-process') diff --git a/src/main/startup/configure-process.ts b/src/main/startup/configure-process.ts index e49b7e04a03..6b167d7fe14 100644 --- a/src/main/startup/configure-process.ts +++ b/src/main/startup/configure-process.ts @@ -117,20 +117,37 @@ export function patchPackagedProcessPath(): void { } const home = process.env.HOME ?? '' - const extraPaths: string[] = [] + // Why two lists: a seed exists so a GUI-launched Electron can *find* a tool + // its minimal PATH omits. Putting one ahead of the inherited PATH does more + // than that — it re-ranks binaries the user already has, and `~/bin` and + // `~/.local/bin` are arbitrary user-writable directories that can shadow any + // system tool. On the #18234 reporter's box `~/.local/bin/gh` is a wrapper + // around `mise x gh -- gh`; hoisting it over /usr/bin/gh made us run the + // wrapper where their own shell ran the real binary, and the inner bare `gh` + // then resolved back to the wrapper. So: append these, and let a real + // ordering opinion come from the login shell via mergePathSegments. + const isGenericUserBinDir = (path: string): boolean => + process.platform !== 'win32' && + home !== '' && + (path === join(home, 'bin') || path === join(home, '.local/bin')) + const appendPaths: string[] = [] + // Why these still lead: version-manager shims must beat a system install or + // an nvm/mise/asdf user gets the wrong runtime, which is the whole reason + // this seeding is ordered rather than appended (see hydrate-shell-path.ts). + const prependPaths: string[] = [] if (process.platform !== 'win32') { - extraPaths.push('/opt/homebrew/bin', '/opt/homebrew/sbin', '/usr/local/bin', '/usr/local/sbin') + appendPaths.push('/opt/homebrew/bin', '/opt/homebrew/sbin', '/usr/local/bin', '/usr/local/sbin') if (process.platform === 'linux') { // Why: snap and Linuxbrew ship on Linux only, so seeding them elsewhere adds phantom PATH entries every spawn must stat. - extraPaths.push('/snap/bin', '/home/linuxbrew/.linuxbrew/bin') + appendPaths.push('/snap/bin', '/home/linuxbrew/.linuxbrew/bin') } - extraPaths.push('/nix/var/nix/profiles/default/bin') + appendPaths.push('/nix/var/nix/profiles/default/bin') if (home) { - extraPaths.push( + appendPaths.push( join(home, 'bin'), join(home, '.local/bin'), join(home, '.nix-profile/bin'), @@ -142,18 +159,24 @@ export function patchPackagedProcessPath(): void { } // Why: version-manager CLIs use env-node shebangs, so node must be on PATH or spawns fail (also seeds Windows user-local dirs). - extraPaths.push(...getVersionManagerBinPaths()) + // Why the filter: that list carries `~/bin` and `~/.local/bin` too, because + // bun/pnpm/npm --user also install there. Those two are generic user bin + // directories, not a version manager's own shim directory, so they hold + // whatever the user last dropped in them and must not outrank a system dir. + // The specific dirs (.volta/bin, .asdf/shims, mise shims, .bun/bin, …) keep + // leading, which is what the ordering was actually for. + prependPaths.push(...getVersionManagerBinPaths().filter((path) => !isGenericUserBinDir(path))) const pathKey = process.platform === 'win32' && process.env.Path !== undefined ? 'Path' : 'PATH' const currentPath = process.env[pathKey] ?? '' const pathDelimiter = getProcessPathDelimiter() - const existing = new Set(currentPath.split(pathDelimiter)) - const missing = extraPaths.filter((path) => !existing.has(path)) + const currentSegments = currentPath.split(pathDelimiter).filter(Boolean) + const existing = new Set(currentSegments) + const prepend = prependPaths.filter((path) => !existing.has(path)) + const append = appendPaths.filter((path) => !existing.has(path) && !prepend.includes(path)) - if (missing.length > 0) { - process.env[pathKey] = [...missing, ...currentPath.split(pathDelimiter).filter(Boolean)].join( - pathDelimiter - ) + if (prepend.length > 0 || append.length > 0) { + process.env[pathKey] = [...prepend, ...currentSegments, ...append].join(pathDelimiter) } } From c61ca56a9b0970a8ac68706db463f919e3acacdc Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:41:13 -0700 Subject: [PATCH 126/398] fix(ssh): resolve the worktree's execution host instead of guessing from one repo row (#17909) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(host-routing): resolve the execution host before reading a connection Three issues in one defect class: a resolver reads one spelling of one arbitrarily chosen row instead of resolving the worktree's execution host, so something local answers a question about a remote. returned that row's connectionId. With duplicate repo rows for one repo id it could pair a runtime owner with a client-owned SSH connection. It now resolves through the same ambiguity-aware index getRuntimeEnvironmentIdForWorktree uses, prefers the repo row for the host the worktree names, and derives the connection from the resolved host. Conflicting rows return `undefined` (this module's documented "cannot determine the host"), never `null`. `store.getRepo(worktree.repoId)?.connectionId ?? null`. `getRepo` is host-blind and the same repo id can exist on local, SSH and runtime hosts, so a remote worktree could spawn its PTY on the client with the remote cwd. resolveWorktreeLaunchHost picks the row for the worktree's host and reads the connection off that host; conflicting rows are unresolved, not local. session-partition owner maps that contradict each other. Both now compute through one shared function whose argument records the divergence. No behaviour change on either side: converging needs a read-both migration, since both partitions hold real data written by shipping builds. * fix(host-routing): keep nested SSH connections resolvable under a runtime host getRepoSshConnectionId read only the resolved execution host, so a repo row owned by a runtime that reaches a nested SSH target (connectionId: ssh-*, executionHostId: runtime:*) resolved to no connection — answering 'local' for a remote worktree, the same defect #17909 fixed in the other direction. * fix(host-routing): resolve both sides of the execution host through one rule The renderer resolver leaked between two different SSH hosts: a worktree on `ssh:m4air` whose only indexed repo row belonged to `openclaw` answered 'openclaw', because the host-scoped lookup missing fell through to an id-only one. Main's resolver, in the same change, answered 'm4air' — two resolvers, one right and one wrong, on identical input. Both sides now adapt one shared rule (`worktree-execution-host-resolution.ts`): the worktree's own host outranks every repo row, and a row on a different host is never evidence about this one. The renderer's WeakMap index becomes the memoizing adapter it always was; `resolveWorktreeLaunchHost` becomes main's mapping of unresolved onto its throw. Settles the rule the change previously answered two ways. `getRepoSshConnectionId` and `getSshTargetIdForExecutionHost` disagreed for a runtime host carrying a nested `connectionId`; they now compose, so the execution host is the single authority. On a `runtime:*` row that field is a paired HUB's private SSH target, spread through by `repoWithFetchedOwner` and unaddressable from this client — the project-first successor of the row nulls it for exactly that reason. That also fixes the `kind !== 'ssh'` fallback, which fired for `local`: a row declaring itself local handed out an SSH connection. --- ...ser-network-execution-host-for-worktree.ts | 14 +- .../runtime-workspace-session-controller.ts | 7 +- .../runtime/worktree-launch-host-repo.test.ts | 94 ++++++++++ src/main/runtime/worktree-launch-host-repo.ts | 43 +++++ .../src/lib/connection-context.test.ts | 147 +++++++++++++++ .../src/lib/connection-owner-resolution.ts | 28 ++- .../lib/workspace-session-host-persistence.ts | 6 +- src/shared/execution-host.test.ts | 37 ++++ src/shared/execution-host.ts | 38 ++++ .../workspace-session-partition-owner.test.ts | 29 +++ .../workspace-session-partition-owner.ts | 39 ++++ ...worktree-execution-host-resolution.test.ts | 167 ++++++++++++++++++ .../worktree-execution-host-resolution.ts | 109 ++++++++++++ 13 files changed, 749 insertions(+), 9 deletions(-) create mode 100644 src/main/runtime/worktree-launch-host-repo.test.ts create mode 100644 src/main/runtime/worktree-launch-host-repo.ts create mode 100644 src/shared/workspace-session-partition-owner.test.ts create mode 100644 src/shared/workspace-session-partition-owner.ts create mode 100644 src/shared/worktree-execution-host-resolution.test.ts create mode 100644 src/shared/worktree-execution-host-resolution.ts diff --git a/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts b/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts index 87411ad55d7..decc28aa45b 100644 --- a/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts +++ b/src/main/runtime/orca-runtime-resolve-browser-network-execution-host-for-worktree.ts @@ -10,6 +10,7 @@ import { import { resolveRuntimeBrowserNetworkExecutionHost } from './runtime-browser-network-execution-host' import { resolveLocalProjectRuntimeForWorktreeId } from '../local-project-runtime-resolution' import { getRegisteredSshState } from '../ssh/ssh-target-registry' +import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' import { folderWorkspaceKey, parseWorkspaceKey } from '../../shared/workspace-scope' import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { ResolvedWorktree } from './runtime-worktree-path-identity' @@ -124,7 +125,16 @@ export class OrcaRuntimeWithResolveBrowserNetworkExecutionHostForWorktree extend const parsed = parseWorkspaceKey(workspaceSelector) const worktreeSelector = parsed?.type === 'worktree' ? `id:${parsed.worktreeId}` : selector const worktree = await this.resolveWorktreeSelector(worktreeSelector) - const repo = this.store?.getRepo(worktree.repoId) ?? null + // Why: `getRepo(id)` is host-blind and the same repo id can exist on local, SSH and runtime + // hosts. Reading `connectionId` off an arbitrary row reports "local" for a remote worktree and + // spawns its PTY on the client with the remote cwd (#11163). Loss of a usable answer is + // `unresolved`, never `local`. + const resolution = resolveWorktreeLaunchHost(this.store?.getRepos() ?? [], worktree) + if (resolution.kind === 'ambiguous') { + throw new Error('worktree_execution_host_unresolved') + } + // Metadata only (display name, hook settings); the routing decision is `resolution.connectionId`. + const repo = resolution.repo ?? this.store?.getRepo(worktree.repoId) ?? null triggerTerminalSpawnPushTargetMaterialization( worktree.path, worktree.pushTarget, @@ -137,7 +147,7 @@ export class OrcaRuntimeWithResolveBrowserNetworkExecutionHostForWorktree extend scope: { id: worktree.id, path: worktree.path, - connectionId: repo?.connectionId ?? null, + connectionId: resolution.connectionId, repo, folderWorkspace: null }, diff --git a/src/main/runtime/runtime-workspace-session-controller.ts b/src/main/runtime/runtime-workspace-session-controller.ts index 3c252a06b60..168f76e256c 100644 --- a/src/main/runtime/runtime-workspace-session-controller.ts +++ b/src/main/runtime/runtime-workspace-session-controller.ts @@ -8,6 +8,7 @@ import { import type { FolderWorkspace } from '../../shared/folder-workspace-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' +import { workspaceSessionPartitionHostId } from '../../shared/workspace-session-partition-owner' import { parseWorkspaceKey } from '../../shared/workspace-scope' import type { RuntimeStore } from './runtime-store-contract' @@ -44,7 +45,11 @@ export class RuntimeWorkspaceSessionController { } const resolvedWorktreeId = scope?.type === 'worktree' ? scope.worktreeId : worktreeId const repo = store?.getRepo?.(getRepoIdFromWorktreeId(resolvedWorktreeId)) - return repo ? getRepoExecutionHostId(repo) : LOCAL_EXECUTION_HOST_ID + // Why: SSH worktrees keep their own `ssh:` partition here while the renderer writes + // them to 'local'; the shared owner map records that divergence (#12723). + return repo + ? workspaceSessionPartitionHostId(getRepoExecutionHostId(repo), 'host-partition') + : LOCAL_EXECUTION_HOST_ID } getHostId(worktreeId: string): ExecutionHostId { diff --git a/src/main/runtime/worktree-launch-host-repo.test.ts b/src/main/runtime/worktree-launch-host-repo.test.ts new file mode 100644 index 00000000000..8a81f2424d1 --- /dev/null +++ b/src/main/runtime/worktree-launch-host-repo.test.ts @@ -0,0 +1,94 @@ +import { describe, expect, it } from 'vitest' +import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' + +// Why (#11163): the terminal launch scope read +// `store.getRepo(worktree.repoId)?.connectionId ?? null` — one spelling of one arbitrarily chosen +// row instead of the worktree's execution host. A remote worktree then spawns its PTY on the +// client with the remote cwd (`DaemonProtocolError: Working directory "…" does not exist`). +describe('resolveWorktreeLaunchHost', () => { + const localRow = { id: 'shared', path: '/local/repo' } + const sshRow = { id: 'shared', path: '/remote/repo', connectionId: 'ssh-b' } + + it('reports ambiguous when duplicate repo rows disagree about the owning host', () => { + expect(resolveWorktreeLaunchHost([localRow, sshRow], { repoId: 'shared' })).toEqual({ + kind: 'ambiguous' + }) + }) + + it('resolves the row for the host the worktree names', () => { + expect( + resolveWorktreeLaunchHost([localRow, sshRow], { repoId: 'shared', hostId: 'ssh:ssh-b' }) + ).toEqual({ kind: 'resolved', repo: sshRow, connectionId: 'ssh-b' }) + expect( + resolveWorktreeLaunchHost([localRow, sshRow], { repoId: 'shared', hostId: 'local' }) + ).toEqual({ kind: 'resolved', repo: localRow, connectionId: null }) + }) + + // The settled rule: the execution host is authoritative, and a row on some *other* host is never + // evidence about this one — not for the connection, and not for the metadata row either. This is + // the question `getRepoSshConnectionId` and `getSshTargetIdForExecutionHost` once answered two + // ways; they now compose, and `execution-host.test.ts` pins the composition. + it('never hands a worktree a connection belonging to a different host', () => { + const clientOwnedRow = { id: 'r', path: '/p', connectionId: 'ssh-client' } + expect( + resolveWorktreeLaunchHost([clientOwnedRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', repo: null, connectionId: null }) + // Even the runtime host's *own* row contributes no PTY route: its nested target lives in that + // machine's namespace, so spawning against it here would dial the wrong box. The renderer + // reads the same resolution and does want that id — see execution-host.test.ts. + const nestedRow = { + id: 'r', + path: '/p', + connectionId: 'ssh-nested', + executionHostId: 'runtime:env-a' as const + } + expect( + resolveWorktreeLaunchHost([nestedRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', repo: nestedRow, connectionId: null }) + // Two SSH hosts, one shared repo id: the worktree's own host wins outright. + expect( + resolveWorktreeLaunchHost([{ id: 'r', path: '/p', connectionId: 'openclaw' }], { + repoId: 'r', + hostId: 'ssh:m4air' + }) + ).toEqual({ kind: 'resolved', repo: null, connectionId: 'm4air' }) + expect( + resolveWorktreeLaunchHost( + [ + { id: 'r', path: '/p', connectionId: 'openclaw' }, + { id: 'r', path: '/q', connectionId: 'm4air' } + ], + { repoId: 'r', hostId: 'ssh:m4air' } + ) + ).toEqual({ + kind: 'resolved', + repo: { id: 'r', path: '/q', connectionId: 'm4air' }, + connectionId: 'm4air' + }) + // A row declaring itself local hands out no SSH connection, whatever `connectionId` says. + expect( + resolveWorktreeLaunchHost([{ id: 'r', path: '/p', connectionId: 'openclaw' }], { + repoId: 'r', + hostId: 'local' + }) + ).toEqual({ kind: 'resolved', repo: null, connectionId: null }) + }) + + it('leaves a single unambiguous row alone', () => { + expect(resolveWorktreeLaunchHost([sshRow], { repoId: 'shared' })).toEqual({ + kind: 'resolved', + repo: sshRow, + connectionId: 'ssh-b' + }) + expect(resolveWorktreeLaunchHost([localRow], { repoId: 'shared' })).toEqual({ + kind: 'resolved', + repo: localRow, + connectionId: null + }) + expect(resolveWorktreeLaunchHost([], { repoId: 'shared' })).toEqual({ + kind: 'resolved', + repo: null, + connectionId: null + }) + }) +}) diff --git a/src/main/runtime/worktree-launch-host-repo.ts b/src/main/runtime/worktree-launch-host-repo.ts new file mode 100644 index 00000000000..fa2641655b3 --- /dev/null +++ b/src/main/runtime/worktree-launch-host-repo.ts @@ -0,0 +1,43 @@ +import { + createRepoRowExecutionHostLookup, + resolveWorktreeExecutionHost, + type ExecutionHostOwnerRow +} from '../../shared/worktree-execution-host-resolution' +import { getSshTargetIdForExecutionHost } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' + +export type LaunchHostRepo = Pick + +export type WorktreeLaunchHostResolution = + | { kind: 'resolved'; repo: T | null; connectionId: string | null } + | { kind: 'ambiguous' } + +/** + * Main-side adapter over the shared execution-host rule + * (`src/shared/worktree-execution-host-resolution.ts`), which the renderer's owner index answers + * with too. Two things are local to this side: + * + * - rival rows that disagree about the host are `ambiguous` and the launch scope throws, while an + * id nobody carries stays "no repo, no connection" — the launch path's long-standing behaviour + * for a worktree whose repo row has gone; + * - the connection comes off the *host*, not the resolved row. This is a client-dialable PTY + * route, so a `runtime:` host contributes nothing: its nested SSH target belongs to that + * machine's namespace and spawning against it here would dial the wrong box. The renderer wants + * the opposite answer from the same resolution, which is why the shared type carries both. + */ +export function resolveWorktreeLaunchHost( + repos: readonly T[], + worktree: { repoId: string; hostId?: string | null } +): WorktreeLaunchHostResolution { + const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) + if (resolution.kind === 'unresolved') { + return resolution.reason === 'ambiguous' + ? { kind: 'ambiguous' } + : { kind: 'resolved', repo: null, connectionId: null } + } + return { + kind: 'resolved', + repo: resolution.owner, + connectionId: getSshTargetIdForExecutionHost(resolution.hostId) + } +} diff --git a/src/renderer/src/lib/connection-context.test.ts b/src/renderer/src/lib/connection-context.test.ts index 2ef70ba95e2..fae6c74cdd8 100644 --- a/src/renderer/src/lib/connection-context.test.ts +++ b/src/renderer/src/lib/connection-context.test.ts @@ -27,6 +27,27 @@ function makeRepo(overrides: Partial & { id: string }): Repo { } } +function makeWorktree(overrides: Partial & { id: string; repoId: string }): Worktree { + return { + path: '/srv/repo', + head: 'abc123', + branch: 'refs/heads/main', + isBare: false, + isMainWorktree: false, + displayName: 'Workspace', + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 0, + ...overrides + } +} + describe('getConnectionId', () => { afterEach(() => { useAppStore.setState(initialState, true) @@ -525,6 +546,132 @@ describe('getConnectionIdFromState', () => { expect(getConnectionIdFromState(state, 'repo-ssh::/home/neil/repo-feature')).toBe('ssh-2') }) + it('refuses to resolve a connection when duplicate repo rows disagree about the owning host', () => { + // Why (#17799): a repo id carried by two rows — one runtime-owned, one holding a + // client-owned SSH connection — must not hand the client's connection to the runtime. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ id: 'repo-dup', executionHostId: 'runtime:env-a' }), + makeRepo({ id: 'repo-dup', connectionId: 'ssh-client' }) + ], + worktreesByRepo: {} + } + + expect(getConnectionIdFromState(state, 'repo-dup::/home/neil/repo-feature')).toBeUndefined() + }) + + it('still resolves duplicate repo rows that agree about the owning host', () => { + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ id: 'repo-dup', connectionId: 'ssh-same' }), + makeRepo({ id: 'repo-dup', connectionId: 'ssh-same', path: '/home/neil/other' }) + ], + worktreesByRepo: {} + } + + expect(getConnectionIdFromState(state, 'repo-dup::/home/neil/repo-feature')).toBe('ssh-same') + }) + + it('never hands a worktree the SSH connection of a different host', () => { + // Why (#11163): two SSH hosts, one shared repo id. The worktree names `ssh:m4air`; the only + // indexed row belongs to `openclaw`. An id-only fallback after the host lookup misses answers + // with the wrong host's connection — "Reconnect openclaw" on an m4air pane, and file reads + // routed to a machine that never held the path. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [makeRepo({ id: 'repo-shared', connectionId: 'openclaw' })], + worktreesByRepo: { + 'repo-shared': [ + makeWorktree({ + id: 'repo-shared::/srv/repo', + repoId: 'repo-shared', + hostId: 'ssh:m4air' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-shared::/srv/repo')).toBe('m4air') + }) + + it('never hands a runtime-hosted worktree a client-owned SSH connection', () => { + // The row is on `ssh:openclaw`, not on the runtime host, so it says nothing about this + // worktree. This is the cross-host case, not the nested-SSH one below. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [makeRepo({ id: 'repo-shared', connectionId: 'openclaw' })], + worktreesByRepo: { + 'repo-shared': [ + makeWorktree({ + id: 'repo-shared::/srv/repo', + repoId: 'repo-shared', + hostId: 'runtime:awin' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-shared::/srv/repo')).toBeNull() + }) + + it('keeps a runtime host nested SSH target, which decides local readability', () => { + // `repoWithFetchedOwner` stamps the runtime host and spreads the nested target through. The + // pane pairs it with the environment (`selectRuntimeAwareSshStatus`) for reconnect state, and + // `isNativeChatTranscriptLocalReadable` treats a null here as "this client can read it" — so + // dropping it would send a transcript read to the wrong machine. + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ + id: 'repo-runtime', + connectionId: 'ssh-nested', + executionHostId: 'runtime:env-a' + }) + ], + worktreesByRepo: { + 'repo-runtime': [ + makeWorktree({ + id: 'repo-runtime::/srv/repo', + repoId: 'repo-runtime', + hostId: 'runtime:env-a', + runtimeOwnerEnvironmentId: 'env-a' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-runtime::/srv/repo')).toBe('ssh-nested') + }) + + it('resolves the row on the SSH host the worktree names when both hosts carry the id', () => { + const state: ConnectionContextState = { + folderWorkspaces: [], + projectGroups: [], + repos: [ + makeRepo({ id: 'repo-shared', connectionId: 'openclaw' }), + makeRepo({ id: 'repo-shared', connectionId: 'm4air', path: '/srv/repo' }) + ], + worktreesByRepo: { + 'repo-shared': [ + makeWorktree({ + id: 'repo-shared::/srv/repo', + repoId: 'repo-shared', + hostId: 'ssh:m4air' + }) + ] + } + } + + expect(getConnectionIdFromState(state, 'repo-shared::/srv/repo')).toBe('m4air') + }) + it('indexes immutable worktree and repo snapshots once across repeated selector calls', () => { let worktreeIdReads = 0 let repoIdReads = 0 diff --git a/src/renderer/src/lib/connection-owner-resolution.ts b/src/renderer/src/lib/connection-owner-resolution.ts index 91169e939c6..f1b52607d46 100644 --- a/src/renderer/src/lib/connection-owner-resolution.ts +++ b/src/renderer/src/lib/connection-owner-resolution.ts @@ -1,5 +1,10 @@ import type { AppState } from '@/store/types' -import { getIndexedRepoMap, getIndexedWorktreeMap } from '@/store/worktree-repo-index' +import { + findIndexedRepoOwnerForHost, + resolveIndexedRepoOwner, + resolveIndexedWorktreeOwner +} from './worktree-runtime-owner-index' +import { resolveWorktreeExecutionHost } from '../../../shared/worktree-execution-host-resolution' import { FLOATING_TERMINAL_WORKTREE_ID } from '../../../shared/constants' import { getRepoIdFromWorktreeId } from '../../../shared/worktree/id' import { parseWorkspaceKey } from '../../../shared/workspace-scope' @@ -60,10 +65,25 @@ export function getConnectionIdFromState( } // Why: owner resolution runs from retained Zustand selectors, so unrelated // store writes must not flatten every worktree or scan every repository. - const worktree = getIndexedWorktreeMap(state.worktreesByRepo).get(worktreeId) + const worktreeResolution = resolveIndexedWorktreeOwner(state.worktreesByRepo, worktreeId) + if (worktreeResolution.kind === 'ambiguous') { + // Why (#17799): rows that disagree about the owner cannot name a connection. + // `undefined` is this module's documented "cannot determine the host" answer; + // collapsing it to `null` would authorize a local read of a remote path. + return undefined + } + const worktree = worktreeResolution.kind === 'resolved' ? worktreeResolution.owner : undefined const repoId = worktree?.repoId ?? getRepoIdFromWorktreeId(worktreeId) - const repo = getIndexedRepoMap(state.repos).get(repoId) - return repo ? (repo.connectionId ?? null) : undefined + // Why (#17799, #11163): one rule, shared with main's launch scope. The renderer's contribution is + // only the memoized index — unrelated store writes must not rescan every repository. + const resolution = resolveWorktreeExecutionHost( + { + byId: (id) => resolveIndexedRepoOwner(state.repos, id), + byHost: (id, hostId) => findIndexedRepoOwnerForHost(state.repos, id, hostId) + }, + { repoId, hostId: worktree?.hostId ?? null } + ) + return resolution.kind === 'resolved' ? resolution.connectionId : undefined } export function getConnectionIdForFileFromState( diff --git a/src/renderer/src/lib/workspace-session-host-persistence.ts b/src/renderer/src/lib/workspace-session-host-persistence.ts index 1ee386883d5..07dd0b2f775 100644 --- a/src/renderer/src/lib/workspace-session-host-persistence.ts +++ b/src/renderer/src/lib/workspace-session-host-persistence.ts @@ -9,6 +9,7 @@ import { parseExecutionHostId, type ExecutionHostId } from '../../../shared/execution-host' +import { workspaceSessionPartitionHostId } from '../../../shared/workspace-session-partition-owner' import { parseWorkspaceKey } from '../../../shared/workspace-scope' import { getRepoIdFromWorktreeId } from '../../../shared/worktree/id' import { @@ -187,8 +188,9 @@ export function buildHostSessionRouting(state: HostPersistenceState): HostSessio if (!repoHostId) { return LOCAL_EXECUTION_HOST_ID } - const parsed = parseExecutionHostId(repoHostId) - return parsed?.kind === 'runtime' ? parsed.id : LOCAL_EXECUTION_HOST_ID + // Why: SSH-owned worktrees stay in the 'local' partition here while the runtime writes them to + // `ssh:`; the shared owner map records that divergence (#12723). + return workspaceSessionPartitionHostId(repoHostId, 'local-partition') } return { hostIdByWorktreeId, claims } } diff --git a/src/shared/execution-host.test.ts b/src/shared/execution-host.test.ts index de895978f3b..9905fc5fe2a 100644 --- a/src/shared/execution-host.test.ts +++ b/src/shared/execution-host.test.ts @@ -4,7 +4,9 @@ import { LOCAL_EXECUTION_HOST_ID, getLocalExecutionHostLabel, getRepoExecutionHostId, + getRepoSshConnectionId, getSettingsFocusedExecutionHostId, + getSshTargetIdForExecutionHost, getWorktreeExecutionHostId, normalizeExecutionHostOrder, normalizeExecutionHostScope, @@ -113,6 +115,41 @@ describe('execution host identity', () => { expect(getWorktreeExecutionHostId({}, {}, 'runtime:focused-host')).toBe('runtime:focused-host') }) + // These two look interchangeable and are not: one answers "which SSH target holds this row's + // files", the other "which connection may this client dial". They agree except on a runtime + // host, where a nested target exists but is not dialable from here — so the pane that reads it + // needs one answer and the PTY route needs the other. + it('distinguishes the SSH target holding a row from the connection this client may dial', () => { + // Legacy spelling: `connectionId` alone *is* the host, so both answers agree. + expect(getRepoSshConnectionId({ connectionId: 'openclaw' })).toBe('openclaw') + expect(getSshTargetIdForExecutionHost('ssh:openclaw')).toBe('openclaw') + // Unified spelling, no legacy field. + expect(getRepoSshConnectionId({ executionHostId: 'ssh:m4air' })).toBe('m4air') + + // A row declaring itself local hands out no SSH connection, whatever the legacy field says: + // `local` has no SSH namespace to nest in, so the two spellings are contradicting each other. + expect( + getRepoSshConnectionId({ executionHostId: 'local', connectionId: 'openclaw' }) + ).toBeNull() + + // A runtime host does have its own namespace, and a nested target appears only in this field. + // Dropping it would make a nested-SSH workspace read as local — which is what decides whether + // this client tries to read the transcript itself. + expect( + getRepoSshConnectionId({ executionHostId: 'runtime:env-a', connectionId: 'ssh-nested' }) + ).toBe('ssh-nested') + // ...but that id is not dialable from this client alone, so the routing answer stays null. + expect(getSshTargetIdForExecutionHost('runtime:env-a')).toBeNull() + // A runtime host with no nested target is simply not on SSH. + expect(getRepoSshConnectionId({ executionHostId: 'runtime:env-a' })).toBeNull() + + // An ephemeral-VM target is an ordinary client-dialable target and stays an `ssh:` host. + expect(getRepoSshConnectionId({ connectionId: 'runtime-ssh-vm-1' })).toBe('runtime-ssh-vm-1') + expect(getRepoExecutionHostId({ connectionId: 'runtime-ssh-vm-1' })).toBe( + 'ssh:runtime-ssh-vm-1' + ) + }) + it('derives focused host compatibility from active runtime settings', () => { expect(getSettingsFocusedExecutionHostId(null)).toBe(LOCAL_EXECUTION_HOST_ID) expect(getSettingsFocusedExecutionHostId({ activeRuntimeEnvironmentId: 'runtime-1' })).toBe( diff --git a/src/shared/execution-host.ts b/src/shared/execution-host.ts index aacbc695cdc..a77d02b3882 100644 --- a/src/shared/execution-host.ts +++ b/src/shared/execution-host.ts @@ -166,6 +166,44 @@ export function getRepoExecutionHostId( return connectionId ? toSshExecutionHostId(connectionId) : LOCAL_EXECUTION_HOST_ID } +export function getSshTargetIdForExecutionHost( + executionHostId: string | null | undefined +): string | null { + const parsed = parseExecutionHostId(executionHostId) + return parsed?.kind === 'ssh' ? parsed.targetId : null +} + +// Why: SSH ownership has two spellings on a repo row — the legacy `connectionId` +// field and the unified `executionHostId`. Routing that reads the raw field answers +// "local" for a row that only carries `ssh:`, which runs a remote operation +// on the client. Resolve the host first, then read the connection off it. +// +// The two hosts that are not themselves SSH are not the same case: +// +// - `local` has no SSH namespace to nest in, so a surviving `connectionId` is a row +// contradicting itself — the shape main's `resolveRepoOwnershipEvidence` calls +// `contradictory`. Answering with it hands out an SSH connection for a row that declares +// itself local. +// - `runtime:` is a different machine with its own SSH targets, and a nested one appears +// only in this field (`repoWithFetchedOwner` spreads it through). It is not dialable on its +// own, but it is addressable as the pair (environmentId, targetId) — which is how the +// renderer reads it, recovering the environment from the worktree and looking the target up +// inside it (`selectRuntimeAwareSshStatus`). Dropping it makes a nested-SSH workspace read +// as local, which is what decides whether a transcript is read on this client. +// +// So this answers "which SSH target holds this row's files", not "which connection may this +// client dial". `getSshTargetIdForExecutionHost` answers the latter; callers routing a +// client-local PTY or Git provider want that one instead. +export function getRepoSshConnectionId( + repo: Pick +): string | null { + const host = parseExecutionHostId(getRepoExecutionHostId(repo)) + if (host?.kind === 'ssh') { + return host.targetId + } + return host?.kind === 'runtime' ? normalizeHostPart(repo.connectionId) : null +} + export function getWorktreeExecutionHostId( worktree: Pick, repo: Pick | undefined, diff --git a/src/shared/workspace-session-partition-owner.test.ts b/src/shared/workspace-session-partition-owner.test.ts new file mode 100644 index 00000000000..f4362471c66 --- /dev/null +++ b/src/shared/workspace-session-partition-owner.test.ts @@ -0,0 +1,29 @@ +import { describe, expect, it } from 'vitest' +import { workspaceSessionPartitionHostId } from './workspace-session-partition-owner' + +// Why (#12723): the renderer and the runtime used two independent owner maps for the same +// worktree's session state. They now share one function, so the divergence is a single argument +// and cannot drift further. Behaviour on both sides is unchanged. +describe('workspaceSessionPartitionHostId', () => { + it('keeps runtime worktrees in their own partition on both sides', () => { + expect(workspaceSessionPartitionHostId('runtime:env-a', 'local-partition')).toBe( + 'runtime:env-a' + ) + expect(workspaceSessionPartitionHostId('runtime:env-a', 'host-partition')).toBe('runtime:env-a') + }) + + it('keeps local worktrees local on both sides', () => { + expect(workspaceSessionPartitionHostId('local', 'local-partition')).toBe('local') + expect(workspaceSessionPartitionHostId('local', 'host-partition')).toBe('local') + }) + + it('records the SSH divergence as the only difference between the two models', () => { + expect(workspaceSessionPartitionHostId('ssh:devbox', 'local-partition')).toBe('local') + expect(workspaceSessionPartitionHostId('ssh:devbox', 'host-partition')).toBe('ssh:devbox') + }) + + it('falls back to the local partition for unparseable host ids', () => { + expect(workspaceSessionPartitionHostId(null, 'host-partition')).toBe('local') + expect(workspaceSessionPartitionHostId('nonsense', 'host-partition')).toBe('local') + }) +}) diff --git a/src/shared/workspace-session-partition-owner.ts b/src/shared/workspace-session-partition-owner.ts new file mode 100644 index 00000000000..83166fb76f3 --- /dev/null +++ b/src/shared/workspace-session-partition-owner.ts @@ -0,0 +1,39 @@ +import { + LOCAL_EXECUTION_HOST_ID, + parseExecutionHostId, + type ExecutionHostId +} from './execution-host' + +/** + * Where an SSH-owned worktree's durable session state lives. + * + * This is the single axis on which the renderer and the main-process runtime disagree today + * (stablyai/orca#12723). Both sides now compute their partition through this function so the + * divergence is one argument in one place instead of two independently drifting owner maps: + * + * - `local-partition` — the renderer's shipping model. SSH worktrees keep their session state in + * the `local` partition; partitioning them would double-own the data. + * - `host-partition` — the runtime's shipping model (#12671). Pane retirement, windowless PTY + * handoff and orchestration fences read-modify-write `ssh:`. + * + * Both partitions hold real data written by shipping builds, so neither side can simply adopt the + * other's answer: flipping a resolver orphans whichever store it stops reading. Converging needs a + * read-both transition (generalize `workspaceSessionPartitionIdsForHost`) and should converge on + * `host-partition`, since Orca Remote — SSH's successor — is already partitioned as `runtime:*`. + * Until then this function preserves today's behaviour exactly on both sides. + */ +export type WorkspaceSessionSshOwnership = 'local-partition' | 'host-partition' + +export function workspaceSessionPartitionHostId( + executionHostId: string | null | undefined, + sshOwnership: WorkspaceSessionSshOwnership +): ExecutionHostId { + const parsed = parseExecutionHostId(executionHostId) + if (parsed?.kind === 'runtime') { + return parsed.id + } + if (parsed?.kind === 'ssh') { + return sshOwnership === 'host-partition' ? parsed.id : LOCAL_EXECUTION_HOST_ID + } + return LOCAL_EXECUTION_HOST_ID +} diff --git a/src/shared/worktree-execution-host-resolution.test.ts b/src/shared/worktree-execution-host-resolution.test.ts new file mode 100644 index 00000000000..70ee6d304b6 --- /dev/null +++ b/src/shared/worktree-execution-host-resolution.test.ts @@ -0,0 +1,167 @@ +import { describe, expect, it } from 'vitest' +import { + createRepoRowExecutionHostLookup, + resolveWorktreeExecutionHost +} from './worktree-execution-host-resolution' + +// Why (#11163, #17799): main's terminal launch scope and the renderer's owner index both answer +// "which host does this worktree execute on". They used to answer it separately, and disagreed — +// main derived the host from the worktree while the renderer fell back to an id-only repo lookup, +// so a pane on one SSH host was routed to another. One rule now, exercised here directly. +const resolve = ( + repos: readonly { id: string; connectionId?: string; executionHostId?: string }[], + worktree: { repoId: string; hostId?: string | null } +): ReturnType => + resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos as never), worktree) as never + +describe('resolveWorktreeExecutionHost', () => { + describe('the worktree names its own host', () => { + it('routes to that host even when the only row belongs to a different SSH host', () => { + // The reproduced defect: `ssh:m4air` worktree, sole row on `openclaw`. + expect( + resolve([{ id: 'r', connectionId: 'openclaw' }], { repoId: 'r', hostId: 'ssh:m4air' }) + ).toEqual({ kind: 'resolved', hostId: 'ssh:m4air', connectionId: 'm4air', owner: null }) + }) + + it('answers before the repo row hydrates, because the host is not a guess', () => { + // Deliberate change from "unresolved": #6648 blocks destructive ops while the *host* is + // unknown. A worktree naming `ssh:m4air` is not that case — the repo row adds nothing the + // host id has not already settled, and refusing here stalls a remote pane on hydration. + expect(resolve([], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: null + }) + }) + + it('picks the row on that host when both SSH hosts carry the id', () => { + const openclaw = { id: 'r', connectionId: 'openclaw' } + const m4air = { id: 'r', connectionId: 'm4air' } + expect(resolve([openclaw, m4air], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: m4air + }) + expect(resolve([openclaw, m4air], { repoId: 'r', hostId: 'ssh:openclaw' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:openclaw', + connectionId: 'openclaw', + owner: openclaw + }) + }) + + it('matches a row that names the host in either spelling', () => { + const stamped = { id: 'r', executionHostId: 'ssh:m4air' } + expect(resolve([stamped], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: stamped + }) + }) + + it('takes no connection from a row on a different host, whatever this host is', () => { + // The row lives on `ssh:openclaw`; neither a local nor a runtime worktree may borrow it. + for (const hostId of ['local', 'runtime:env-a']) { + expect(resolve([{ id: 'r', connectionId: 'openclaw' }], { repoId: 'r', hostId })).toEqual({ + kind: 'resolved', + hostId, + connectionId: null, + owner: null + }) + } + }) + + it('reads a runtime host nested SSH target off the row on that same host', () => { + // Not a cross-host borrow: this row *is* the runtime host's row, and the nested target + // appears nowhere else. Nulling it makes the workspace read as local, which decides whether + // this client tries to read a transcript that lives on the nested host. + const nested = { id: 'r', connectionId: 'ssh-nested', executionHostId: 'runtime:env-a' } + expect(resolve([nested], { repoId: 'r', hostId: 'runtime:env-a' })).toEqual({ + kind: 'resolved', + hostId: 'runtime:env-a', + connectionId: 'ssh-nested', + owner: nested + }) + }) + + it('gives a local row no SSH connection even when it carries a stale one', () => { + const contradictory = { id: 'r', connectionId: 'openclaw', executionHostId: 'local' } + expect(resolve([contradictory], { repoId: 'r', hostId: 'local' })).toEqual({ + kind: 'resolved', + hostId: 'local', + connectionId: null, + owner: contradictory + }) + }) + }) + + describe('the worktree names no host', () => { + it('resolves from the sole row, in either spelling', () => { + const legacy = { id: 'r', connectionId: 'openclaw' } + expect(resolve([legacy], { repoId: 'r' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:openclaw', + connectionId: 'openclaw', + owner: legacy + }) + const stamped = { id: 'r', executionHostId: 'ssh:m4air' } + expect(resolve([stamped], { repoId: 'r' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + connectionId: 'm4air', + owner: stamped + }) + const local = { id: 'r' } + expect(resolve([local], { repoId: 'r' })).toEqual({ + kind: 'resolved', + hostId: 'local', + connectionId: null, + owner: local + }) + }) + + it('refuses when rival rows disagree about the host, including two SSH hosts', () => { + expect( + resolve( + [ + { id: 'r', connectionId: 'openclaw' }, + { id: 'r', connectionId: 'm4air' } + ], + { + repoId: 'r' + } + ) + ).toEqual({ kind: 'unresolved', reason: 'ambiguous' }) + expect( + resolve([{ id: 'r', connectionId: 'openclaw' }, { id: 'r' }], { repoId: 'r' }) + ).toEqual({ kind: 'unresolved', reason: 'ambiguous' }) + }) + + it('treats the two spellings of one host as agreement, not conflict', () => { + expect( + resolve( + [ + { id: 'r', connectionId: 'm4air' }, + { id: 'r', executionHostId: 'ssh:m4air' } + ], + { repoId: 'r' } + ) + ).toMatchObject({ kind: 'resolved', hostId: 'ssh:m4air', connectionId: 'm4air' }) + }) + + it('reports an unknown owner distinctly from a conflicting one', () => { + expect(resolve([], { repoId: 'r' })).toEqual({ kind: 'unresolved', reason: 'unknown' }) + }) + }) + + it('ignores an unparseable host id rather than treating it as a host', () => { + const row = { id: 'r', connectionId: 'openclaw' } + expect(resolve([row], { repoId: 'r', hostId: 'ssh:' })).toMatchObject({ + kind: 'resolved', + connectionId: 'openclaw' + }) + }) +}) diff --git a/src/shared/worktree-execution-host-resolution.ts b/src/shared/worktree-execution-host-resolution.ts new file mode 100644 index 00000000000..8abe5de72cc --- /dev/null +++ b/src/shared/worktree-execution-host-resolution.ts @@ -0,0 +1,109 @@ +/** + * One rule for "which host does this worktree execute on, and what connection routes there". + * + * Main and the renderer both have to answer it — the terminal launch scope picks a PTY route from + * it, the renderer picks a file-read route and the reconnect affordance from it — so the rule lives + * here instead of being re-derived per side. Two re-derivations already disagreed: main answered + * from the worktree's own host while the renderer fell back to an id-only repo lookup, so a pane on + * `ssh:m4air` was offered "Reconnect openclaw" and read its files off openclaw (#11163). + * + * `unresolved` is a distinct answer, never "local": the same repo id can exist on a local, an SSH + * and a runtime host at once, and loss of a usable answer must fail closed rather than authorize a + * client-side read of a remote path (#6648, #17799). + */ + +import type { Repo } from './repo-types' +import { + getRepoExecutionHostId, + getRepoSshConnectionId, + getSshTargetIdForExecutionHost, + normalizeExecutionHostId, + type ExecutionHostId +} from './execution-host' + +export type ExecutionHostOwnerRow = Pick + +export type ExecutionHostOwnerMatch = + | { kind: 'resolved'; owner: T } + | { kind: 'missing' } + | { kind: 'ambiguous' } + +/** + * How a caller finds repo rows. Main scans the store array; the renderer answers from a + * WeakMap-memoized index because owner resolution runs inside retained selectors. That is a + * performance difference, not a different rule. + */ +export type ExecutionHostOwnerLookup = { + /** The row for `repoId`, or `ambiguous` when rival rows disagree about the owning host. */ + byId: (repoId: string) => ExecutionHostOwnerMatch + /** The row for `repoId` on exactly `hostId`, or null when that host carries no row. */ + byHost: (repoId: string, hostId: ExecutionHostId) => T | null +} + +export type WorktreeExecutionHostResolution = + | { + kind: 'resolved' + hostId: ExecutionHostId + /** + * The SSH target whose filesystem holds this workspace — for a `runtime:` host, its nested + * target, addressable only as the pair with `hostId`. Callers deciding what *this client* + * may dial (a PTY route, a Git provider) must use `getSshTargetIdForExecutionHost(hostId)` + * instead; this field can name a host the client cannot reach on its own. + */ + connectionId: string | null + /** Display metadata only. The decisions are `hostId` / `connectionId`. */ + owner: T | null + } + | { kind: 'unresolved'; reason: 'ambiguous' | 'unknown' } + +export function resolveWorktreeExecutionHost( + lookup: ExecutionHostOwnerLookup, + worktree: { repoId: string; hostId?: string | null } +): WorktreeExecutionHostResolution { + const worktreeHostId = normalizeExecutionHostId(worktree.hostId) + if (worktreeHostId) { + // The worktree names its own host, which outranks every repo row. A row on a *different* host + // is not evidence about this one — falling back to it is the cross-host leak: one SSH host's + // pane routed to another. A row on *this* host still is evidence, and is the only place a + // runtime's nested SSH target appears. + const owner = lookup.byHost(worktree.repoId, worktreeHostId) + return { + kind: 'resolved', + hostId: worktreeHostId, + connectionId: + getSshTargetIdForExecutionHost(worktreeHostId) ?? + (owner ? getRepoSshConnectionId(owner) : null), + owner + } + } + const match = lookup.byId(worktree.repoId) + if (match.kind !== 'resolved') { + return { kind: 'unresolved', reason: match.kind === 'ambiguous' ? 'ambiguous' : 'unknown' } + } + return { + kind: 'resolved', + hostId: getRepoExecutionHostId(match.owner), + connectionId: getRepoSshConnectionId(match.owner), + owner: match.owner + } +} + +/** Array-backed lookup for callers holding the whole repo list (main's store). */ +export function createRepoRowExecutionHostLookup( + repos: readonly T[] +): ExecutionHostOwnerLookup { + const rowsFor = (repoId: string): T[] => repos.filter((repo) => repo.id === repoId) + return { + byId: (repoId) => { + const rows = rowsFor(repoId) + if (rows.length === 0) { + return { kind: 'missing' } + } + const hostIds = new Set(rows.map((repo) => getRepoExecutionHostId(repo))) + const owner = rows[0] + return hostIds.size > 1 || !owner ? { kind: 'ambiguous' } : { kind: 'resolved', owner } + }, + byHost: (repoId, hostId) => + rowsFor(repoId).find((repo) => getRepoExecutionHostId(repo) === hostId) ?? null + } +} From 7c94d12190ed26181e8defa69cfd9e8c80202e07 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:59:29 -0700 Subject: [PATCH 127/398] fix(ssh): route four host-blind seams through the resolved execution host (#17919) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(host-routing): resolve the execution host before reading a connection Three issues in one defect class: a resolver reads one spelling of one arbitrarily chosen row instead of resolving the worktree's execution host, so something local answers a question about a remote. returned that row's connectionId. With duplicate repo rows for one repo id it could pair a runtime owner with a client-owned SSH connection. It now resolves through the same ambiguity-aware index getRuntimeEnvironmentIdForWorktree uses, prefers the repo row for the host the worktree names, and derives the connection from the resolved host. Conflicting rows return `undefined` (this module's documented "cannot determine the host"), never `null`. `store.getRepo(worktree.repoId)?.connectionId ?? null`. `getRepo` is host-blind and the same repo id can exist on local, SSH and runtime hosts, so a remote worktree could spawn its PTY on the client with the remote cwd. resolveWorktreeLaunchHost picks the row for the worktree's host and reads the connection off that host; conflicting rows are unresolved, not local. session-partition owner maps that contradict each other. Both now compute through one shared function whose argument records the divergence. No behaviour change on either side: converging needs a read-both migration, since both partitions hold real data written by shipping builds. * fix(host-routing): keep nested SSH connections resolvable under a runtime host getRepoSshConnectionId read only the resolved execution host, so a repo row owned by a runtime that reaches a nested SSH target (connectionId: ssh-*, executionHostId: runtime:*) resolved to no connection — answering 'local' for a remote worktree, the same defect #17909 fixed in the other direction. * fix(host-routing): resolve both sides of the execution host through one rule The renderer resolver leaked between two different SSH hosts: a worktree on `ssh:m4air` whose only indexed repo row belonged to `openclaw` answered 'openclaw', because the host-scoped lookup missing fell through to an id-only one. Main's resolver, in the same change, answered 'm4air' — two resolvers, one right and one wrong, on identical input. Both sides now adapt one shared rule (`worktree-execution-host-resolution.ts`): the worktree's own host outranks every repo row, and a row on a different host is never evidence about this one. The renderer's WeakMap index becomes the memoizing adapter it always was; `resolveWorktreeLaunchHost` becomes main's mapping of unresolved onto its throw. Settles the rule the change previously answered two ways. `getRepoSshConnectionId` and `getSshTargetIdForExecutionHost` disagreed for a runtime host carrying a nested `connectionId`; they now compose, so the execution host is the single authority. On a `runtime:*` row that field is a paired HUB's private SSH target, spread through by `repoWithFetchedOwner` and unaddressable from this client — the project-first successor of the row nulls it for exactly that reason. That also fixes the `kind !== 'ssh'` fallback, which fired for `local`: a row declaring itself local handed out an SSH connection. * fix(ssh): resolve the execution host in the worktree scan and managed create The worktree scan and createManagedWorktree both picked remote-vs-local from repo.connectionId, so a row stamped only executionHostId: 'ssh:*' was scanned and created on the client against a remote path. The folder branch returns before the check, so its agent-trust write landed locally too. Refs #11163 * fix(ssh): stop over-rejecting and refusing SSH hosts the process owns runtimeRepoMatchesExecutionHost rejected an unstamped SSH repo against its own ssh:, so repo-add/clone dedupe could register a second row for a path the host already owns. assertHostIsSupported made the CLI/runtime RPC refuse --host ssh:* while the same process's IPC handler routed it correctly; setupExistingFolder now shares that registration. Clone still refuses, because nothing in this process clones onto an SSH host. Refs #11163 * test(ssh): retarget the SSH host-setup guard spec at the substitution it prevents setupProjectExistingFolder now registers the remote path through the same addRemoteRepoFromPath the desktop IPC uses, so it fails on the host's terms (connection not registered) rather than a categorical refusal. The local clone/probe side effects it exists to catch are still asserted absent. Refs #11163 * fix(cli): require an absolute path when setting a project up on an SSH host Routing --host ssh:* to the remote registration made relative paths newly reachable there, and they were resolved against the client cwd — registering a path that names the wrong machine. Refs #11163 * fix(repos): read the SSH registry directly so the runtime stays Node-bootable Routing runtime project setup through addRemoteRepoFromPath dragged ipc/ssh -- and its 25-module electron graph -- into the runtime bundle. ssh-target-registry already exists for exactly this; ipc/ssh only re-exports it. * fix(ssh): close the agent-launch and session-export host-blind twins Three sites left on the legacy spelling, all the same shape as the ones this branch already fixed: - `launchAgentTerminal` did `getRepo(worktree.repoId)` then wrote agent trust with that row's `connectionId`. Host-blind, so a repo id carried by two SSH hosts wrote a remote path into the *client's* Codex/Cursor/Copilot config and the agent on the host never saw the trust. Every sibling call site already passes the resolved `workspace.connectionId`; this was the last that did not. - `targetForWorktree` (workspace-session export) fell back to the same host-blind read, so a session could be published to a machine that never owned the worktree. Unresolvable ownership now exports to nobody. - `addRemoteRepoFromPath` minted `connectionId`-only rows while being the routing path this branch adds, so it kept creating rows in exactly the spelling the branch works around. It now stamps `toSshExecutionHostId(connectionId)` at creation; `reassignSshTargetId` already migrates both spellings, so target rename stays correct. Tests cover two *different* SSH hosts throughout — the case none of the earlier duplicate-row tests had, all of which were local-vs-ssh or runtime-vs-ssh. --- src/cli/handlers/project.ts | 10 +- src/cli/index-project-setup.test.ts | 31 +++ .../ipc/remote-workspace-patch-queue.test.ts | 5 +- src/main/ipc/remote-workspace.test.ts | 2 + src/main/ipc/remote-workspace.ts | 20 +- src/main/ipc/repos/remote-home-path.ts | 2 +- .../repos/remote-repo-registration.test.ts | 124 ++++++++++ .../ipc/repos/remote-repo-registration.ts | 12 +- .../agent-terminal-launch-trust-host.test.ts | 107 +++++++++ ...ged-worktree-create-execution-host.test.ts | 155 +++++++++++++ .../orca-runtime-create-managed-worktree.ts | 33 ++- ...-resolved-worktrees-for-explicit-target.ts | 14 +- .../orca-runtime-preserved-branch-cleanup.ts | 13 ++ ...orca-runtime-refresh-repo-worktree-scan.ts | 16 +- ...a-runtime-terminal-create-deduplication.ts | 13 +- .../repository-project-operations.spec.ts | 9 +- ...time-project-host-setup-controller.test.ts | 120 ++++++++++ .../runtime-project-host-setup-controller.ts | 42 +++- .../runtime-worktree-selection.test.ts | 55 +++++ .../runtime/runtime-worktree-selection.ts | 13 +- ...rktree-scan-execution-host-routing.test.ts | 214 ++++++++++++++++++ 21 files changed, 954 insertions(+), 56 deletions(-) create mode 100644 src/main/ipc/repos/remote-repo-registration.test.ts create mode 100644 src/main/runtime/agent-terminal-launch-trust-host.test.ts create mode 100644 src/main/runtime/managed-worktree-create-execution-host.test.ts create mode 100644 src/main/runtime/runtime-project-host-setup-controller.test.ts create mode 100644 src/main/runtime/runtime-worktree-selection.test.ts create mode 100644 src/main/runtime/worktree-scan-execution-host-routing.test.ts diff --git a/src/cli/handlers/project.ts b/src/cli/handlers/project.ts index 2b0a8d8ada2..a7bf9ee2abe 100644 --- a/src/cli/handlers/project.ts +++ b/src/cli/handlers/project.ts @@ -10,7 +10,7 @@ import type { ProjectHostSetupUpdateArgs, ProjectHostSetupUpdateResult } from '../../shared/project-types' -import type { ExecutionHostId } from '../../shared/execution-host' +import { getSshTargetIdForExecutionHost, type ExecutionHostId } from '../../shared/execution-host' import type { RepoKind } from '../../shared/repo-types' import type { CommandHandler, HandlerContext } from '../dispatch' import { @@ -111,10 +111,14 @@ export const PROJECT_HANDLERS: Record = { }, 'project setup-existing-folder': async ({ flags, client, cwd, json }) => { const rawPath = getRequiredStringFlag(flags, 'path') + const hostId = getRequiredHostId(flags) + // An SSH host's filesystem is not the CLI's, so resolving a relative path against the client + // cwd would register a path that names the wrong machine. + const pathIsOffClient = client.isRemote || getSshTargetIdForExecutionHost(hostId) !== null const args: ProjectHostSetupExistingFolderArgs = { projectId: getRequiredStringFlag(flags, 'project'), - hostId: getRequiredHostId(flags), - path: resolveRepoPathArgument(rawPath, cwd, client.isRemote, 'Remote project setup'), + hostId, + path: resolveRepoPathArgument(rawPath, cwd, pathIsOffClient, 'Remote project setup'), kind: getOptionalRepoKind(flags), displayName: getOptionalStringFlag(flags, 'display-name') } diff --git a/src/cli/index-project-setup.test.ts b/src/cli/index-project-setup.test.ts index 068c246ba7c..f104e45c39a 100644 --- a/src/cli/index-project-setup.test.ts +++ b/src/cli/index-project-setup.test.ts @@ -481,6 +481,37 @@ describe('orca cli worktree awareness', () => { process.exitCode = priorExitCode }) + it('rejects SSH project setup relative paths, which name the client filesystem', async () => { + // A local CLI reaching an `ssh:*` host is still off-client: resolving `./orca` against the + // CLI cwd would register a path that exists on the wrong machine. + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + const errSpy = vi.spyOn(console, 'error').mockImplementation(() => {}) + const priorExitCode = process.exitCode + + await main( + [ + 'project', + 'setup-existing-folder', + '--project', + 'github:stablyai/orca', + '--host', + 'ssh:openclaw', + '--path', + './orca', + '--json' + ], + '/tmp/repo' + ) + + expect(callMock).not.toHaveBeenCalled() + expect([...logSpy.mock.calls, ...errSpy.mock.calls].flat().join('\n')).toContain( + 'Remote project setup requires --path to be an absolute path on the remote server.' + ) + expect(process.exitCode).toBe(1) + + process.exitCode = priorExitCode + }) + it('rejects remote repo.add relative paths instead of resolving against client cwd', async () => { const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) const errSpy = vi.spyOn(console, 'error').mockImplementation(() => {}) diff --git a/src/main/ipc/remote-workspace-patch-queue.test.ts b/src/main/ipc/remote-workspace-patch-queue.test.ts index 30ff0af7761..6b5d342b3f6 100644 --- a/src/main/ipc/remote-workspace-patch-queue.test.ts +++ b/src/main/ipc/remote-workspace-patch-queue.test.ts @@ -77,8 +77,11 @@ describe('remoteWorkspace:setForConnectedTargets patch queue', () => { const handlers = new Map unknown>() const muxByTargetId = new Map }>() const getRepoMock = vi.fn() + // Ownership resolution reads the catalog, not one id-keyed row, so the fake has to project one. + const KNOWN_REPO_IDS = ['repo-target-1', 'repo-target-2', 'repo-reset', 'repo-newer'] const store = { - getRepo: getRepoMock + getRepo: getRepoMock, + getRepos: () => KNOWN_REPO_IDS.map((repoId) => getRepoMock(repoId)).filter(Boolean) } as unknown as Store const target: SshTarget = { diff --git a/src/main/ipc/remote-workspace.test.ts b/src/main/ipc/remote-workspace.test.ts index 434f8c7dec4..56eb4804a82 100644 --- a/src/main/ipc/remote-workspace.test.ts +++ b/src/main/ipc/remote-workspace.test.ts @@ -153,8 +153,10 @@ describe('remoteWorkspace:setForConnectedTargets', () => { const muxByTargetId = new Map }>() const getRepoMock = vi.fn() const getWorkspaceSessionMock = vi.fn() + // Ownership resolution reads the catalog, not one id-keyed row, so the fake has to project one. const store = { getRepo: getRepoMock, + getRepos: () => [getRepoMock('repo-target-1')].filter(Boolean), getWorkspaceSession: getWorkspaceSessionMock } as unknown as Store diff --git a/src/main/ipc/remote-workspace.ts b/src/main/ipc/remote-workspace.ts index 74339e8f694..f9479f15ce9 100644 --- a/src/main/ipc/remote-workspace.ts +++ b/src/main/ipc/remote-workspace.ts @@ -12,7 +12,10 @@ import { } from '../../shared/remote-workspace-types' import type { WorkspaceSessionState } from '../../shared/workspace-session-state-types' import { getRepoIdFromWorktreeId } from '../../shared/worktree/id' -import { parseExecutionHostId } from '../../shared/execution-host' +import { + createRepoRowExecutionHostLookup, + resolveWorktreeExecutionHost +} from '../../shared/worktree-execution-host-resolution' import { getRemoteWorkspaceNamespace } from './remote-workspace-namespace' import { registerRemoteWorkspaceNotificationHandler } from './remote-workspace-events' import { CLIENT_ID } from './remote-workspace-client-identity' @@ -108,12 +111,15 @@ function targetForWorktree( worktreeId: string, executionHostId?: string ): string | null { - const parsedHostId = parseExecutionHostId(executionHostId) - if (parsedHostId?.kind === 'ssh') { - return parsedHostId.targetId - } - const repoId = getRepoIdFromWorktreeId(worktreeId) - return store.getRepo(repoId)?.connectionId ?? null + // Why: this decides which SSH target a workspace session is exported to. The old fallback read + // `getRepo(id)?.connectionId`, which is host-blind — the same repo id can name rows on several + // hosts, so a session could be published to a machine that never owned the worktree (#11163). + // Unresolvable ownership exports to nobody rather than guessing. + const resolution = resolveWorktreeExecutionHost( + createRepoRowExecutionHostLookup(store.getRepos()), + { repoId: getRepoIdFromWorktreeId(worktreeId), hostId: executionHostId ?? null } + ) + return resolution.kind === 'resolved' ? resolution.connectionId : null } function exportSessionForTarget( diff --git a/src/main/ipc/repos/remote-home-path.ts b/src/main/ipc/repos/remote-home-path.ts index 2952764f872..602eed25684 100644 --- a/src/main/ipc/repos/remote-home-path.ts +++ b/src/main/ipc/repos/remote-home-path.ts @@ -1,4 +1,4 @@ -import { getActiveMultiplexer } from '../ssh' +import { getActiveMultiplexer } from '../../ssh/ssh-target-registry' export async function resolveRemoteHomePath(connectionId: string, path: string): Promise { if (path !== '~' && path !== '~/' && !path.startsWith('~/')) { diff --git a/src/main/ipc/repos/remote-repo-registration.test.ts b/src/main/ipc/repos/remote-repo-registration.test.ts new file mode 100644 index 00000000000..0a7069b1f63 --- /dev/null +++ b/src/main/ipc/repos/remote-repo-registration.test.ts @@ -0,0 +1,124 @@ +// Registration is now the runtime's SSH path too (`projectHostSetup.setupExistingFolder --host +// ssh:*`), so what it stamps decides what every downstream host resolver can read. It minted +// `connectionId`-only rows, leaving the unified spelling permanently empty, and deduped by raw +// `connectionId`, which cannot see a row stamped `executionHostId: 'ssh:*'`. +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../shared/repo-types' + +const getSshGitProviderMock = vi.hoisted(() => vi.fn()) +vi.mock('../../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock +})) + +vi.mock('../../repo-icon-autodetect', () => ({ + detectRepoIconAndUpstream: vi.fn(async () => ({})) +})) + +vi.mock('../../ssh/ssh-target-registry', () => ({ + getActiveMultiplexer: vi.fn(() => null) +})) + +vi.mock('./remote-home-path', () => ({ + resolveRemoteHomePath: vi.fn(async (_connectionId: string, path: string) => path) +})) + +import { addRemoteRepoFromPath } from './remote-repo-registration' + +function makeStore(repos: Repo[]) { + return { + getRepos: () => repos, + getSshTarget: () => undefined, + addRepo: (repo: Repo) => { + repos.push(repo) + } + } +} + +describe('addRemoteRepoFromPath', () => { + beforeEach(() => { + getSshGitProviderMock.mockReset() + getSshGitProviderMock.mockReturnValue({ + isGitRepoAsync: vi.fn(async () => ({ isRepo: true, rootPath: '/srv/app' })) + }) + }) + + it('stamps the unified execution-host spelling alongside the legacy connection id', async () => { + const repos: Repo[] = [] + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'm4air', + remotePath: '/srv/app' + }) + + expect('error' in result).toBe(false) + const repo = (result as { repo: Repo }).repo + expect(repo.connectionId).toBe('m4air') + expect(repo.executionHostId).toBe('ssh:m4air') + }) + + it('dedupes against a row that names the host in the unified spelling only', async () => { + const existing = { + id: 'existing', + path: '/srv/app', + displayName: 'app', + badgeColor: '#000', + addedAt: 0, + executionHostId: 'ssh:m4air' + } as Repo + const repos: Repo[] = [existing] + + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'm4air', + remotePath: '/srv/app' + }) + + expect(result).toEqual({ repo: existing, alreadyExisted: true }) + expect(repos).toHaveLength(1) + }) + + it('does not dedupe onto a row on a different SSH host at the same path', async () => { + // Two hosts can both hold /srv/app. Matching on path alone registers one host's repo as the + // other's — the mirror image of the id-only lookup this change removes. + const repos: Repo[] = [ + { + id: 'openclaw-row', + path: '/srv/app', + displayName: 'app', + badgeColor: '#000', + addedAt: 0, + connectionId: 'openclaw' + } as Repo + ] + + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'm4air', + remotePath: '/srv/app' + }) + + expect((result as { alreadyExisted: boolean }).alreadyExisted).toBe(false) + expect((result as { repo: Repo }).repo.executionHostId).toBe('ssh:m4air') + expect(repos).toHaveLength(2) + }) + + it('does not dedupe onto a local row that carries a stale connection id', async () => { + // The pullfrog case: a row declaring itself local must not answer as an SSH host. + const repos: Repo[] = [ + { + id: 'local-row', + path: '/srv/app', + displayName: 'app', + badgeColor: '#000', + addedAt: 0, + executionHostId: 'local', + connectionId: 'develop' + } as Repo + ] + + const result = await addRemoteRepoFromPath(makeStore(repos) as never, { + connectionId: 'develop', + remotePath: '/srv/app' + }) + + expect((result as { alreadyExisted: boolean }).alreadyExisted).toBe(false) + expect(repos).toHaveLength(2) + }) +}) diff --git a/src/main/ipc/repos/remote-repo-registration.ts b/src/main/ipc/repos/remote-repo-registration.ts index 3f3b94ff58b..35ab72056ae 100644 --- a/src/main/ipc/repos/remote-repo-registration.ts +++ b/src/main/ipc/repos/remote-repo-registration.ts @@ -3,9 +3,10 @@ import type { Store } from '../../persistence' import type { Repo } from '../../../shared/repo-types' import { DEFAULT_REPO_BADGE_COLOR } from '../../../shared/constants' import { normalizeRuntimePathForComparison } from '../../../shared/cross-platform-path' +import { getRepoSshConnectionId, toSshExecutionHostId } from '../../../shared/execution-host' import { getSshGitProvider } from '../../providers/ssh-git-dispatch' import { detectRepoIconAndUpstream } from '../../repo-icon-autodetect' -import { getActiveMultiplexer } from '../ssh' +import { getActiveMultiplexer } from '../../ssh/ssh-target-registry' import { resolveRemoteHomePath } from './remote-home-path' export async function addRemoteRepoFromPath( @@ -26,11 +27,13 @@ export async function addRemoteRepoFromPath( let repoKind: 'git' | 'folder' = args.kind ?? 'git' let resolvedPath = await resolveRemoteHomePath(args.connectionId, args.remotePath) + // Resolve the host: a row stamped only `executionHostId: 'ssh:*'` is the same registration, and + // missing it here registers a duplicate repo for a path the host already owns. const existing = store .getRepos() .find( (repo) => - repo.connectionId === args.connectionId && + getRepoSshConnectionId(repo) === args.connectionId && normalizeRuntimePathForComparison(repo.path) === normalizeRuntimePathForComparison(resolvedPath) ) @@ -61,7 +64,7 @@ export async function addRemoteRepoFromPath( .getRepos() .find( (repo) => - repo.connectionId === args.connectionId && + getRepoSshConnectionId(repo) === args.connectionId && normalizeRuntimePathForComparison(repo.path) === normalizeRuntimePathForComparison(resolvedPath) ) @@ -92,6 +95,9 @@ export async function addRemoteRepoFromPath( addedAt: Date.now(), kind: repoKind, connectionId: args.connectionId, + // Stamp the unified spelling at creation: this is now the runtime's SSH registration path too, + // and minting `connectionId`-only rows leaves every host-resolving reader on the legacy field. + executionHostId: toSshExecutionHostId(args.connectionId), ...(repoKind === 'git' ? { externalWorktreeVisibilityLegacy: false, diff --git a/src/main/runtime/agent-terminal-launch-trust-host.test.ts b/src/main/runtime/agent-terminal-launch-trust-host.test.ts new file mode 100644 index 00000000000..949b0c545fc --- /dev/null +++ b/src/main/runtime/agent-terminal-launch-trust-host.test.ts @@ -0,0 +1,107 @@ +// launchAgentTerminal read `store.getRepo(worktree.repoId)?.connectionId` for the trust write — +// host-blind, so the same repo id on two hosts wrote a remote path into the client's agent config +// and the agent on the host never saw the trust (#11163). Every sibling call site already passes +// the resolved `workspace.connectionId`; this was the last one that did not. +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const REMOTE_PATH = '/srv/app-feature' + +type RuntimeInternals = { + resolveWorktreeSelector: (selector: string) => Promise + buildStartupForAgent: (repo: unknown, agent: unknown, prompt: string) => unknown + markWorkspaceTrustedForAgent: ( + agent: unknown, + connectionId: string | null | undefined, + path: string + ) => Promise + createTerminal: (selector: string, opts: unknown) => Promise +} + +function makeRuntime(repos: readonly Record[], hostId?: string) { + const store = { + getSettings: () => ({ disabledTuiAgents: [], workspaceDir: '/tmp/workspaces' }), + getProjectHostSetups: () => [], + getRepos: () => repos, + getRepo: (id: string) => repos.find((repo) => repo.id === id) + } + const runtime = new OrcaRuntimeService(store as never) + const internals = runtime as unknown as RuntimeInternals + vi.spyOn(internals, 'resolveWorktreeSelector').mockResolvedValue({ + id: 'repo-shared::/srv/app-feature', + repoId: 'repo-shared', + path: REMOTE_PATH, + ...(hostId ? { hostId } : {}) + }) + vi.spyOn(internals, 'buildStartupForAgent').mockReturnValue({ + agent: 'codex', + startup: { command: 'codex', env: {}, startupCommandDelivery: 'none', telemetry: {} } + }) + const markTrusted = vi.fn(async () => {}) + vi.spyOn(internals, 'markWorkspaceTrustedForAgent').mockImplementation(markTrusted) + vi.spyOn(internals, 'createTerminal').mockResolvedValue({ id: 'pty-1' }) + return { runtime, markTrusted } +} + +describe('launchAgentTerminal trust write', () => { + beforeEach(() => { + vi.restoreAllMocks() + }) + + it('writes trust on the host the worktree names, not on a rival row', async () => { + // Two SSH hosts publish the same repo id; the worktree is on m4air. + const { runtime, markTrusted } = makeRuntime( + [ + { id: 'repo-shared', path: '/home/me/app', connectionId: 'openclaw' }, + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' } + ], + 'ssh:m4air' + ) + + await runtime.launchAgentTerminal('id:repo-shared::/srv/app-feature', { + agent: 'codex', + prompt: 'go' + } as never) + + expect(markTrusted).toHaveBeenCalledWith('codex', 'm4air', REMOTE_PATH) + }) + + it('writes trust locally for a local worktree even when a remote row shares the id', async () => { + const { runtime, markTrusted } = makeRuntime( + [ + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }, + { id: 'repo-shared', path: '/home/me/app' } + ], + 'local' + ) + + await runtime.launchAgentTerminal('id:repo-shared::/srv/app-feature', { + agent: 'codex', + prompt: 'go' + } as never) + + expect(markTrusted).toHaveBeenCalledWith('codex', null, REMOTE_PATH) + }) + + it('refuses rather than guessing when rival rows disagree and the worktree names no host', async () => { + const { runtime } = makeRuntime([ + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }, + { id: 'repo-shared', path: '/home/me/app' } + ]) + + await expect( + runtime.launchAgentTerminal('id:repo-shared::/srv/app-feature', { + agent: 'codex', + prompt: 'go' + } as never) + ).rejects.toThrow('worktree_execution_host_unresolved') + }) +}) diff --git a/src/main/runtime/managed-worktree-create-execution-host.test.ts b/src/main/runtime/managed-worktree-create-execution-host.test.ts new file mode 100644 index 00000000000..af1a73dc9cc --- /dev/null +++ b/src/main/runtime/managed-worktree-create-execution-host.test.ts @@ -0,0 +1,155 @@ +// createManagedWorktree used to pick remote-vs-local from the raw `connectionId` field, so a repo +// stamped only `executionHostId: 'ssh:*'` ran `git worktree add` on the client against a remote +// path — and the folder branch, which returns before that check, wrote agent trust locally for a +// remote workspace. Both are the #11163 shape: read the execution host, never one spelling of it. +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } +})) + +const createRuntimeFolderWorktreeMock = vi.hoisted(() => vi.fn()) +vi.mock('./runtime-folder-worktree-create', () => ({ + createRuntimeFolderWorktree: createRuntimeFolderWorktreeMock +})) + +const createRuntimeLocalManagedWorktreeMock = vi.hoisted(() => vi.fn()) +vi.mock('./runtime-local-worktree-create', () => ({ + createRuntimeLocalManagedWorktree: createRuntimeLocalManagedWorktreeMock +})) + +const trustMocks = vi.hoisted(() => ({ + local: vi.fn(async () => {}), + remote: vi.fn(async () => {}) +})) +vi.mock('./runtime-worktree-agent-startup', async (importOriginal) => ({ + ...(await importOriginal>()), + markLocalWorktreeTrusted: trustMocks.local, + markRemoteWorktreeTrusted: trustMocks.remote +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const TARGET_ID = 'remote-1' +const REMOTE_PATH = '/srv/app' + +type RuntimeInternals = { + resolveRepoSelector: (selector: string) => Promise + createManagedRemoteWorktree: (repo: unknown, args: unknown) => Promise + resolveLineageForWorktreeCreate: (input: unknown) => Promise + recordCreatedWorktreeLineage: (worktree: unknown, resolution: unknown) => unknown +} + +function makeRuntime(repo: Record): { + runtime: OrcaRuntimeService + createRemote: ReturnType +} { + const store = { + getSettings: () => ({ disabledTuiAgents: [], workspaceDir: '/tmp/workspaces' }), + getProjectHostSetups: () => [] + } + const runtime = new OrcaRuntimeService(store as never) + const internals = runtime as unknown as RuntimeInternals + vi.spyOn(internals, 'resolveRepoSelector').mockResolvedValue(repo) + vi.spyOn(internals, 'resolveLineageForWorktreeCreate').mockResolvedValue(null) + vi.spyOn(internals, 'recordCreatedWorktreeLineage').mockReturnValue({ + lineage: null, + workspaceLineage: null, + warnings: [] + }) + const createRemote = vi.fn().mockResolvedValue({ + worktree: { id: 'wt-1', path: '/srv/app-feature', branch: 'feature' } + }) + vi.spyOn(internals, 'createManagedRemoteWorktree').mockImplementation(createRemote) + return { runtime, createRemote } +} + +describe('createManagedWorktree execution-host routing', () => { + beforeEach(() => { + createRuntimeFolderWorktreeMock.mockReset() + createRuntimeFolderWorktreeMock.mockResolvedValue({ worktree: { id: 'folder-1' } }) + createRuntimeLocalManagedWorktreeMock.mockReset() + // Name the defect in the failure output: reaching this mock means a remote repo was routed + // into a client-side `git worktree add`. + createRuntimeLocalManagedWorktreeMock.mockRejectedValue( + new Error('local_worktree_create_ran_for_remote_repo') + ) + trustMocks.local.mockClear() + trustMocks.remote.mockClear() + }) + + it('creates on the SSH host for a repo stamped executionHostId only', async () => { + const { runtime, createRemote } = makeRuntime({ + id: 'repo-remote', + path: REMOTE_PATH, + kind: 'git', + executionHostId: `ssh:${TARGET_ID}` + }) + + await runtime.createManagedWorktree({ repoSelector: 'repo-remote', name: 'feature' } as never) + + // A local `git worktree add` against a remote path is the silent-substitution failure. + expect(createRuntimeLocalManagedWorktreeMock).not.toHaveBeenCalled() + expect(createRemote).toHaveBeenCalledWith( + expect.objectContaining({ id: 'repo-remote', connectionId: TARGET_ID }), + expect.anything() + ) + }) + + it('still creates on the SSH host for a legacy connectionId-only repo', async () => { + const { runtime, createRemote } = makeRuntime({ + id: 'repo-remote', + path: REMOTE_PATH, + kind: 'git', + connectionId: TARGET_ID + }) + + await runtime.createManagedWorktree({ repoSelector: 'repo-remote', name: 'feature' } as never) + + expect(createRuntimeLocalManagedWorktreeMock).not.toHaveBeenCalled() + expect(createRemote).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: TARGET_ID }), + expect.anything() + ) + }) + + it('marks a folder workspace trusted on its SSH host, not on the client', async () => { + const { runtime } = makeRuntime({ + id: 'repo-folder', + path: REMOTE_PATH, + kind: 'folder', + connectionId: TARGET_ID, + executionHostId: `ssh:${TARGET_ID}` + }) + + await runtime.createManagedWorktree({ repoSelector: 'repo-folder', name: 'notes' } as never) + + const deps = createRuntimeFolderWorktreeMock.mock.calls[0]?.[0]?.deps + await deps.markTrusted('codex', '/srv/app') + + expect(trustMocks.remote).toHaveBeenCalledWith('codex', TARGET_ID, '/srv/app') + expect(trustMocks.local).not.toHaveBeenCalled() + }) + + it('keeps a local folder workspace trusted on the client', async () => { + const { runtime } = makeRuntime({ + id: 'repo-folder-local', + path: '/Users/me/notes', + kind: 'folder' + }) + + await runtime.createManagedWorktree({ + repoSelector: 'repo-folder-local', + name: 'notes' + } as never) + + const deps = createRuntimeFolderWorktreeMock.mock.calls[0]?.[0]?.deps + await deps.markTrusted('codex', '/Users/me/notes') + + expect(trustMocks.local).toHaveBeenCalledWith('codex', '/Users/me/notes') + expect(trustMocks.remote).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/runtime/orca-runtime-create-managed-worktree.ts b/src/main/runtime/orca-runtime-create-managed-worktree.ts index a6f8ebfbd79..f9116f73405 100644 --- a/src/main/runtime/orca-runtime-create-managed-worktree.ts +++ b/src/main/runtime/orca-runtime-create-managed-worktree.ts @@ -4,6 +4,7 @@ import type { RuntimeManagedWorktreeCreateArgs } from './runtime-managed-worktre import type { CreateWorktreeResult } from '../../shared/worktree/create-types' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' import { isFolderRepo } from '../../shared/repo-kind' +import { getRepoSshConnectionId } from '../../shared/execution-host' import { createRuntimeFolderWorktree } from './runtime-folder-worktree-create' import { createRuntimeLocalManagedWorktree } from './runtime-local-worktree-create' import { prepareRuntimeLocalWorktreeSetup } from './runtime-local-worktree-setup' @@ -56,7 +57,13 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork draftStartup?.agent ?? (requestedAgentEnabled ? requestedAgent : undefined)) const effectiveDraftPaste = args.startupDraftPaste ?? draftStartup?.draftPaste + // Resolve the execution host once: SSH ownership has two spellings, and reading the raw + // `connectionId` field routes an `executionHostId: 'ssh:*'`-only repo down the local path, + // which runs `git worktree add` on the client against a remote path. + const sshConnectionId = getRepoSshConnectionId(repo) if (isFolderRepo(repo)) { + // A folder workspace is a registration, not a filesystem create, so it is host-agnostic — + // except for the agent trust write, which must land on the host that will run the agent. return createRuntimeFolderWorktree({ request: args, repo, @@ -68,7 +75,8 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork store: this.store, ptySpawnAvailable: Boolean(this.ptyController?.spawn), createTerminal: (selector, options) => this.createTerminal(selector, options), - markTrusted: (agent, path) => this.markLocalWorkspaceTrustedForAgent(agent, path), + markTrusted: (agent, path) => + this.markWorkspaceTrustedForAgent(agent, sshConnectionId, path), pasteDraft: (handle, draft) => this.pasteStartupDraftWhenReady(handle, draft), sendFollowup: (handle, followup) => this.sendStartupFollowupWhenReady(handle, followup), invalidateResolvedWorktrees: () => this.invalidateResolvedWorktreeCache(), @@ -89,15 +97,20 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork const lineageInput = args.lineage || args.comment ? { ...args.lineage, comment: args.comment } : undefined const lineageResolution = await this.resolveLineageForWorktreeCreate(lineageInput) - if (repo.connectionId) { - const result = await this.createManagedRemoteWorktree(repo, { - ...args, - activate: args.activate, - ...(effectiveStartup ? { startup: effectiveStartup } : {}), - ...(effectiveStartupFollowup ? { startupFollowup: effectiveStartupFollowup } : {}), - ...(effectiveCreatedWithAgent ? { createdWithAgent: effectiveCreatedWithAgent } : {}), - ...(effectiveDraftPaste ? { startupDraftPaste: effectiveDraftPaste } : {}) - }) + if (sshConnectionId) { + // Why normalize the row: the remote-create pipeline reads `repo.connectionId!` at every + // depth, so hand it the connection the resolved host actually names. + const result = await this.createManagedRemoteWorktree( + { ...repo, connectionId: sshConnectionId }, + { + ...args, + activate: args.activate, + ...(effectiveStartup ? { startup: effectiveStartup } : {}), + ...(effectiveStartupFollowup ? { startupFollowup: effectiveStartupFollowup } : {}), + ...(effectiveCreatedWithAgent ? { createdWithAgent: effectiveCreatedWithAgent } : {}), + ...(effectiveDraftPaste ? { startupDraftPaste: effectiveDraftPaste } : {}) + } + ) const recordedLineage = this.recordCreatedWorktreeLineage(result.worktree, lineageResolution) this.emitWorktreeLifecycle({ kind: 'created', diff --git a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts index f8632bb2643..1f0dbe591a3 100644 --- a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts +++ b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts @@ -19,7 +19,7 @@ import type { Repo } from '../../shared/repo-types' import type { ProjectExecutionRuntimeResolution } from '../../shared/project-execution-runtime' import type { RuntimeWorktreeScanResult } from './repo-worktree-resolution-scan' import { getSshGitProviderGeneration } from '../providers/ssh-git-dispatch' -import { getRepoExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, getRepoSshConnectionId } from '../../shared/execution-host' import type { RuntimeWorktreeScanCache } from './orca-runtime-core' import { resolveWorktreeScanCacheTtlMs } from './runtime-worktree-scan-cache' @@ -135,17 +135,21 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends repo: Repo, projectRuntimeByRepoId?: ReadonlyMap ): Promise { + // Resolve the execution host, not the raw field: an `executionHostId: 'ssh:*'` row with no + // `connectionId` would otherwise get a local project runtime and a `local:default` cache key, + // so its scan neither routes remotely nor re-runs when the SSH provider is replaced. + const sshConnectionId = getRepoSshConnectionId(repo) const projectRuntime = projectRuntimeByRepoId ? projectRuntimeByRepoId.get(repo.id) - : !repo.connectionId + : !sshConnectionId ? resolveLocalProjectRuntimeForRepo(this.requireStore(), repo) : undefined const runtimeKey = projectRuntime ? projectRuntime.status === 'resolved' ? projectRuntime.runtime.cacheKey : projectRuntime.repair.cacheKey - : repo.connectionId - ? `ssh:${repo.connectionId}:${getSshGitProviderGeneration(repo.connectionId)}` + : sshConnectionId + ? `ssh:${sshConnectionId}:${getSshGitProviderGeneration(sshConnectionId)}` : 'local:default' const now = Date.now() const scanScopeKey = `${repo.id}\0${getRepoExecutionHostId(repo)}` @@ -176,7 +180,7 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends return this.listRepoWorktreesForResolution(repo, projectRuntimeByRepoId) } if ( - (refresh.result.ok || !repo.connectionId) && + (refresh.result.ok || !sshConnectionId) && this.worktreeScanInFlight.get(scanScopeKey)?.promise === promise ) { const entry: RuntimeWorktreeScanCache = { diff --git a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts index 43853f0f9c0..810ffeca3be 100644 --- a/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts +++ b/src/main/runtime/orca-runtime-preserved-branch-cleanup.ts @@ -26,6 +26,8 @@ import { RuntimeAccountController } from './runtime-account-controller' import { RuntimeMobileSpeechCatalog } from './runtime-mobile-speech-catalog' import { RuntimeMobileDictationController } from './runtime-mobile-dictation-controller' import { RuntimeProjectHostSetupController } from './runtime-project-host-setup-controller' +import { addRemoteRepoFromPath } from '../ipc/repos/remote-repo-registration' +import type { Store } from '../persistence' import { RuntimeProjectGroupController } from './runtime-project-group-controller' import { RuntimeNestedRepoImport } from './runtime-nested-repo-import' import { RuntimeRepositoryRegistrationController } from './runtime-repository-registration-controller' @@ -194,6 +196,17 @@ export class OrcaRuntimeWithPreservedBranchCleanup extends OrcaRuntimeWithTermin listRepos: () => this.listRepos(), addRepo: (path, kind, hostId) => (this as RuntimeCommandSurfaceHost).addRepo(path, kind, hostId), + addRemoteRepo: async (remote) => { + // The same registration the desktop IPC handler uses, so both surfaces agree on SSH hosts. + const result = await addRemoteRepoFromPath(this.requireStore() as unknown as Store, remote) + if ('error' in result) { + throw new Error(result.error) + } + this.invalidateResolvedWorktreeCache() + this.invalidateWorktreeScanCacheForRepo(result.repo.id) + this.notifyReposChanged() + return result.repo + }, cloneRepo: (url, destination, hostId) => (this as RuntimeCommandSurfaceHost).cloneRepo(url, destination, hostId), invalidateResolvedWorktrees: () => this.invalidateResolvedWorktreeCache(), diff --git a/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts b/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts index 5e3282c5e53..1964b5fdd2f 100644 --- a/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts +++ b/src/main/runtime/orca-runtime-refresh-repo-worktree-scan.ts @@ -17,7 +17,7 @@ import { getSshGitProvider } from '../providers/ssh-git-dispatch' import type { GitWorktreeInfo } from '../../shared/worktree/types' import { listStoredWorktreeRowsForRepo } from './repo-worktree-row-resolution' import type { ResolvedWorktree } from './runtime-worktree-path-identity' -import { getRepoExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, getRepoSshConnectionId } from '../../shared/execution-host' export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget { /** @@ -31,8 +31,10 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK ): Promise { const scannedAt = Date.now() // SSH and WSL-routed repos run Git off-host, so a local admin-dir read cannot describe them. + // Resolve the execution host rather than reading `connectionId`: a row stamped only + // `executionHostId: 'ssh:*'` is just as off-host, and fingerprinting it stats client paths. const fingerprintCapable = - !repo.connectionId && + !getRepoSshConnectionId(repo) && // Why: a repo whose scan TTL already reaches the reconciliation interval can never reuse a // fingerprint, so reading one would be pure work. Agent-scratch roots are that case today. resolveWorktreeScanCacheTtlMs(repo) < WORKTREE_SCAN_ADMIN_RECONCILE_INTERVAL_MS && @@ -92,13 +94,17 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK repo: Repo, projectRuntime: ProjectExecutionRuntimeResolution | undefined ): Promise { - if (!repo.connectionId) { + // Why not `repo.connectionId`: SSH ownership has two spellings, and a repo carrying only + // `executionHostId: 'ssh:*'` would otherwise be scanned on the client against a remote path — + // `git worktree list` then reports nothing, so the remote worktrees never resolve at all. + const sshConnectionId = getRepoSshConnectionId(repo) + if (!sshConnectionId) { return await scanLocalRepoWorktreesForResolution( repo.path, getLocalProjectWorktreeGitOptionsForRuntime(repo, projectRuntime) ) } - const provider = getSshGitProvider(repo.connectionId) + const provider = getSshGitProvider(sshConnectionId) if (!provider) { return { ok: false, worktrees: this.listStoredWorktreesForResolution(repo) } } @@ -149,7 +155,7 @@ export class OrcaRuntimeWithRefreshRepoWorktreeScan extends OrcaRuntimeWithListK protected invalidateSshWorktreeScanCacheInternal(targetId: string): void { const repos = this.store?.getRepos() ?? [] - const affectedRepos = repos.filter((repo) => repo.connectionId === targetId) + const affectedRepos = repos.filter((repo) => getRepoSshConnectionId(repo) === targetId) const affectedScopeKeys = new Set( affectedRepos.map((repo) => `${repo.id}\0${getRepoExecutionHostId(repo)}`) ) diff --git a/src/main/runtime/orca-runtime-terminal-create-deduplication.ts b/src/main/runtime/orca-runtime-terminal-create-deduplication.ts index a1743a855ce..cdb24994fce 100644 --- a/src/main/runtime/orca-runtime-terminal-create-deduplication.ts +++ b/src/main/runtime/orca-runtime-terminal-create-deduplication.ts @@ -8,6 +8,7 @@ import { PTY_CONTROLLER_LIST_TIMEOUT_MS } from './orca-runtime-postlude' import { inferWorktreeIdFromPtyId } from './runtime-worktree-path-identity' import { getRegisteredSshState } from '../ssh/ssh-target-registry' import { LOCAL_EXECUTION_HOST_ID, toSshExecutionHostId } from '../../shared/execution-host' +import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' import type { TuiAgent } from '../../shared/tui-agent' export class OrcaRuntimeWithTerminalCreateDeduplication extends OrcaRuntimeWithCreateAgentSession { @@ -143,12 +144,20 @@ export class OrcaRuntimeWithTerminalCreateDeduplication extends OrcaRuntimeWithC opts: { agent: TuiAgent; prompt: string; title?: string } ): Promise { const worktree = await this.resolveWorktreeSelector(worktreeSelector) - const repo = this.store?.getRepo(worktree.repoId) + // Why: the trust write lands in an agent's config on the machine that runs it, keyed by the + // workspace path. `getRepo(id)` is host-blind, so reading `connectionId` off it wrote a remote + // path into the *client's* config — the agent on the host never sees the trust (#11163). + // Same shape as the folder-create trust write fixed alongside this; the agent-launch half. + const resolution = resolveWorktreeLaunchHost(this.store?.getRepos() ?? [], worktree) + if (resolution.kind === 'ambiguous') { + throw new Error('worktree_execution_host_unresolved') + } + const repo = resolution.repo ?? this.store?.getRepo(worktree.repoId) if (!repo) { throw new Error('Repository for the selected workspace is no longer available.') } const startup = this.buildStartupForAgent(repo, opts.agent, opts.prompt) - await this.markWorkspaceTrustedForAgent(opts.agent, repo.connectionId, worktree.path) + await this.markWorkspaceTrustedForAgent(opts.agent, resolution.connectionId, worktree.path) return await this.createTerminal(`id:${worktree.id}`, { command: startup.startup.command, env: startup.startup.env, diff --git a/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts b/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts index afe3dded381..c9879bf3d7c 100644 --- a/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts +++ b/src/main/runtime/orca-runtime-tests/repository-project-operations.spec.ts @@ -588,7 +588,7 @@ describe('OrcaRuntimeService', () => { } }) - it('refuses SSH hosts instead of setting the project up on the local machine', async () => { + it('never sets an SSH-hosted project up on the local machine', async () => { // Why: both inputs must be paths the pre-guard code would have accepted. An unwritable // destination fails at mkdir and a non-repo path fails at isGitRepo, which would leave the // side-effect assertions below unable to observe the local clone/probe they exist to catch. @@ -639,11 +639,14 @@ describe('OrcaRuntimeService', () => { // first so a regression reports the corruption rather than stopping at the first throw. expect(spawnSpy).not.toHaveBeenCalled() expect(repos).toHaveLength(0) + // Cloning onto an SSH host has no implementation here, so it still refuses outright. expect(cloneError).toMatchObject({ - message: expect.stringMatching(/SSH hosts are not supported/) + message: expect.stringMatching(/Cloning onto an SSH host is not supported/) }) + // Registering an existing remote path does have one, so this now fails on the host's own + // terms — the SSH connection is not registered — rather than on a categorical refusal. expect(existingFolderError).toMatchObject({ - message: expect.stringMatching(/SSH hosts are not supported/) + message: expect.stringMatching(/SSH connection "openclaw" not found or not connected/) }) } finally { spawnSpy.mockRestore() diff --git a/src/main/runtime/runtime-project-host-setup-controller.test.ts b/src/main/runtime/runtime-project-host-setup-controller.test.ts new file mode 100644 index 00000000000..bb3d928d406 --- /dev/null +++ b/src/main/runtime/runtime-project-host-setup-controller.test.ts @@ -0,0 +1,120 @@ +// The CLI/runtime RPC used to refuse `--host ssh:*` with "set the project up from the Orca desktop +// app" — while the desktop IPC handler in the *same process* routed it correctly through +// addRemoteRepoFromPath. Safe but wrong: the process refusing is the one that owns the connection. +import { describe, expect, it, vi } from 'vitest' +import { RuntimeProjectHostSetupController } from './runtime-project-host-setup-controller' +import { getProjectHostSetupForRepo } from '../../shared/project-host-setup-lookup' +import { projectHostSetupProjectionFromRepos } from '../../shared/project-host-setup-projection' +import type { Repo } from '../../shared/repo-types' + +const TARGET_ID = 'target-1' +const REMOTE_PATH = '/srv/app' + +const remoteRepo = { + id: 'repo-remote', + path: REMOTE_PATH, + displayName: 'app', + badgeColor: 'blue', + addedAt: 1, + kind: 'git', + connectionId: TARGET_ID +} as unknown as Repo + +function makeController(): { + controller: RuntimeProjectHostSetupController + addRepo: ReturnType + addRemoteRepo: ReturnType + cloneRepo: ReturnType + projectId: string +} { + const store = { + getProjects: () => projectHostSetupProjectionFromRepos([remoteRepo]).projects, + getProjectHostSetups: () => [], + updateRepo: (_id: string, updates: Record) => ({ ...remoteRepo, ...updates }) + } + const addRepo = vi.fn().mockResolvedValue(remoteRepo) + const addRemoteRepo = vi.fn().mockResolvedValue(remoteRepo) + const cloneRepo = vi.fn().mockResolvedValue(remoteRepo) + const controller = new RuntimeProjectHostSetupController({ + getStore: () => store as never, + listRepos: () => [remoteRepo], + addRepo, + addRemoteRepo, + cloneRepo, + invalidateResolvedWorktrees: vi.fn(), + invalidateWorktreeScan: vi.fn(), + notifyReposChanged: vi.fn() + }) + return { + controller, + addRepo, + addRemoteRepo, + cloneRepo, + projectId: getProjectHostSetupForRepo([], remoteRepo).projectId + } +} + +describe('RuntimeProjectHostSetupController host routing', () => { + it('registers an existing folder on an SSH host instead of refusing it (#11163)', async () => { + const { controller, addRepo, addRemoteRepo, projectId } = makeController() + + const result = await controller.setupExistingFolder({ + projectId, + hostId: `ssh:${TARGET_ID}`, + path: REMOTE_PATH, + kind: 'git' + }) + + expect(addRemoteRepo).toHaveBeenCalledWith({ + connectionId: TARGET_ID, + remotePath: REMOTE_PATH, + kind: 'git' + }) + // The local registration path validates the path against the client filesystem. + expect(addRepo).not.toHaveBeenCalled() + expect(result.repo.id).toBe(remoteRepo.id) + }) + + it('decodes a percent-encoded SSH target back to its connection id', async () => { + const { controller, addRemoteRepo, projectId } = makeController() + + await controller.setupExistingFolder({ + projectId, + hostId: 'ssh:my%20host', + path: REMOTE_PATH, + kind: 'folder' + }) + + expect(addRemoteRepo).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: 'my host', kind: 'folder' }) + ) + }) + + it('still uses the local registration for local and runtime hosts', async () => { + const { controller, addRepo, addRemoteRepo, projectId } = makeController() + + await controller.setupExistingFolder({ + projectId, + hostId: 'local', + path: REMOTE_PATH, + kind: 'git' + }) + + expect(addRepo).toHaveBeenCalledWith(REMOTE_PATH, 'git', 'local') + expect(addRemoteRepo).not.toHaveBeenCalled() + }) + + it('refuses to clone onto an SSH host, because nothing here clones remotely', async () => { + const { controller, cloneRepo, projectId } = makeController() + + await expect( + controller.setupClone({ + projectId, + hostId: `ssh:${TARGET_ID}`, + url: 'https://example.com/app.git', + destination: REMOTE_PATH + }) + ).rejects.toThrow(/Cloning onto an SSH host is not supported/) + expect(cloneRepo).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/runtime/runtime-project-host-setup-controller.ts b/src/main/runtime/runtime-project-host-setup-controller.ts index 28d8b297480..cdc3ab1c60f 100644 --- a/src/main/runtime/runtime-project-host-setup-controller.ts +++ b/src/main/runtime/runtime-project-host-setup-controller.ts @@ -13,7 +13,11 @@ import type { ProjectUpdateArgs } from '../../shared/project-types' import type { Repo } from '../../shared/repo-types' -import { parseExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' +import { + getSshTargetIdForExecutionHost, + parseExecutionHostId, + type ExecutionHostId +} from '../../shared/execution-host' import { getProjectIdForProviderIdentity } from '../../shared/project-host-setup-projection' import { getProjectHostSetupForRepo } from '../../shared/project-host-setup-lookup' import { invalidateAuthorizedRootsCache } from '../ipc/filesystem-auth' @@ -24,18 +28,29 @@ type RuntimeProjectHostSetupDependencies = { getStore: () => RuntimeStore | null listRepos: () => Repo[] addRepo: (path: string, kind: 'folder' | 'git', hostId: ExecutionHostId) => Promise + /** Register an existing path that lives on an SSH host; `addRepo` only reaches local/runtime hosts. */ + addRemoteRepo: (args: { + connectionId: string + remotePath: string + displayName?: string + kind: 'folder' | 'git' + }) => Promise cloneRepo: (url: string, destination: string, hostId: ExecutionHostId) => Promise invalidateResolvedWorktrees: () => void invalidateWorktreeScan: (repoId: string) => void notifyReposChanged: () => void } -function assertHostIsSupported(hostId: ExecutionHostId | null | undefined): void { +// Why clone alone still refuses: nothing in this process clones onto an SSH host. `cloneRepo` runs +// `git clone` on the client, so accepting `ssh:*` here would register the client's copy as the +// host's repo — a local answer to a remote question. Registering an existing remote path, by +// contrast, has a correct implementation this process already uses over IPC. +function assertCloneHostIsSupported(hostId: ExecutionHostId | null | undefined): void { if (parseExecutionHostId(hostId)?.kind !== 'ssh') { return } throw new Error( - 'SSH hosts are not supported by this operation. Set the project up from the Orca desktop app, which owns the SSH connection.' + 'Cloning onto an SSH host is not supported. Clone the repository on the host, then set the project up from that existing folder.' ) } @@ -82,18 +97,25 @@ export class RuntimeProjectHostSetupController { if (!this.deps.getStore()) { throw new Error('runtime_unavailable') } - assertHostIsSupported(args.hostId) + const kind = args.kind === 'folder' ? 'folder' : 'git' const knownRepoIds = new Set(this.deps.listRepos().map((repo) => repo.id)) - const repo = await this.deps.addRepo( - args.path, - args.kind === 'folder' ? 'folder' : 'git', - args.hostId - ) + // Why route rather than refuse: this process owns the SSH connection, and its own IPC handler + // already registers `ssh:*` hosts correctly. Refusing here only made the CLI and runtime RPC + // disagree with the desktop app about what the same process can do. + const sshTargetId = getSshTargetIdForExecutionHost(args.hostId) + const repo = sshTargetId + ? await this.deps.addRemoteRepo({ + connectionId: sshTargetId, + remotePath: args.path, + ...(args.displayName ? { displayName: args.displayName } : {}), + kind + }) + : await this.deps.addRepo(args.path, kind, args.hostId) return this.completeSetup(args, repo, !knownRepoIds.has(repo.id)) } async setupClone(args: ProjectHostSetupCloneArgs): Promise { - assertHostIsSupported(args.hostId) + assertCloneHostIsSupported(args.hostId) const knownRepoIds = new Set(this.deps.listRepos().map((repo) => repo.id)) const repo = await this.deps.cloneRepo(args.url, args.destination, args.hostId) return this.completeSetup( diff --git a/src/main/runtime/runtime-worktree-selection.test.ts b/src/main/runtime/runtime-worktree-selection.test.ts new file mode 100644 index 00000000000..0509d94a2b4 --- /dev/null +++ b/src/main/runtime/runtime-worktree-selection.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from 'vitest' +import { runtimeRepoMatchesExecutionHost } from './runtime-worktree-selection' + +describe('runtimeRepoMatchesExecutionHost', () => { + it('matches an unstamped SSH repo against its own host (#11163)', () => { + // The row spells its ownership as `connectionId`; the request spells it as `ssh:`. + // Rejecting it here makes repo-add/clone dedupe register a second row for the same path. + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'ssh:target-1')).toBe(true) + }) + + it('matches a stamped SSH repo against its own host', () => { + expect( + runtimeRepoMatchesExecutionHost( + { connectionId: 'target-1', executionHostId: 'ssh:target-1' }, + 'ssh:target-1' + ) + ).toBe(true) + }) + + it('rejects an unstamped SSH repo against a different SSH host', () => { + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'ssh:target-2')).toBe( + false + ) + }) + + it('rejects an unstamped SSH repo against local and runtime hosts', () => { + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'local')).toBe(false) + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' }, 'runtime:env-1')).toBe( + false + ) + }) + + it('keeps a host-less legacy repo adoptable by any host', () => { + expect(runtimeRepoMatchesExecutionHost({}, 'runtime:env-1')).toBe(true) + expect(runtimeRepoMatchesExecutionHost({}, 'local')).toBe(true) + expect(runtimeRepoMatchesExecutionHost({}, 'ssh:target-1')).toBe(true) + }) + + it('matches any repo when the caller names no host', () => { + expect(runtimeRepoMatchesExecutionHost({ connectionId: 'target-1' })).toBe(true) + expect(runtimeRepoMatchesExecutionHost({ executionHostId: 'runtime:env-1' }, null)).toBe(true) + }) + + it('keeps a stamped repo bound to the host it names', () => { + expect(runtimeRepoMatchesExecutionHost({ executionHostId: 'runtime:env-1' }, 'local')).toBe( + false + ) + expect( + runtimeRepoMatchesExecutionHost( + { executionHostId: 'local', connectionId: 'target-1' }, + 'local' + ) + ).toBe(true) + }) +}) diff --git a/src/main/runtime/runtime-worktree-selection.ts b/src/main/runtime/runtime-worktree-selection.ts index 5dd1c4fd9a4..7e3fc5be481 100644 --- a/src/main/runtime/runtime-worktree-selection.ts +++ b/src/main/runtime/runtime-worktree-selection.ts @@ -1,5 +1,5 @@ import type { Repo } from '../../shared/repo-types' -import type { ExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' import { splitWorktreeId } from '../../shared/worktree/id' import type { GitPushTarget } from '../../shared/worktree/types' @@ -38,8 +38,9 @@ export function getRuntimeWorktreeRemovalOptionsKey( } // Null executionHostId means host-unaware: path-only callers match any repo, and the first runtime -// host can adopt a legacy (unstamped) repo. But an unstamped repo with a connectionId is an SSH repo -// (resolves to ssh:), so it must not be adopted/matched by a runtime host at the same path. +// host can adopt a legacy (unstamped) repo. A repo that names a host in *either* spelling matches +// only that host — including its own ssh:, which an executionHostId-only comparison +// used to reject, so an unstamped SSH repo failed to dedupe against itself. export function runtimeRepoMatchesExecutionHost( repo: Pick, executionHostId?: ExecutionHostId | null @@ -47,10 +48,10 @@ export function runtimeRepoMatchesExecutionHost( if (executionHostId == null) { return true } - if (repo.executionHostId != null) { - return repo.executionHostId === executionHostId + if (repo.executionHostId == null && repo.connectionId == null) { + return true } - return repo.connectionId == null + return getRepoExecutionHostId(repo) === executionHostId } export function parseExactWorktreeIdSelector( diff --git a/src/main/runtime/worktree-scan-execution-host-routing.test.ts b/src/main/runtime/worktree-scan-execution-host-routing.test.ts new file mode 100644 index 00000000000..9e40789da82 --- /dev/null +++ b/src/main/runtime/worktree-scan-execution-host-routing.test.ts @@ -0,0 +1,214 @@ +// SSH ownership has two spellings on a repo row: the legacy `connectionId` field and the unified +// `executionHostId: 'ssh:*'`. This suite pins the scan and the terminal launch that follows it for +// the second spelling — the seam #17909 identified but could not test end to end (#11163). +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const electronMocks = vi.hoisted(() => { + const ipcMain = { + on: vi.fn(() => ipcMain), + removeListener: vi.fn(() => ipcMain), + emit: vi.fn(() => true) + } + return { + BrowserWindow: { fromId: vi.fn((): unknown => null) }, + webContents: { fromId: vi.fn((): unknown => null) }, + ipcMain, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } + } +}) +vi.mock('electron', () => electronMocks) + +const getSshGitProviderMock = vi.hoisted(() => vi.fn()) +vi.mock('../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock, + getSshGitProviderGeneration: vi.fn(() => 0), + SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE: 'unavailable', + requireSshGitProvider: (connectionId: string) => getSshGitProviderMock(connectionId) +})) + +const listWorktreesStrictMock = vi.hoisted(() => vi.fn()) +vi.mock('../git/worktree', async (importOriginal) => ({ + ...(await importOriginal>()), + listWorktreesStrict: listWorktreesStrictMock +})) + +vi.mock('./repo-worktree-admin-fingerprint', () => ({ + readRepoWorktreeAdminFingerprint: vi.fn(async () => null) +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const TARGET_ID = 'remote-1' +const REPO_ID = 'repo-remote' +const REPO_PATH = '/srv/app' +const WORKTREE_PATH = '/srv/app-feature' +const WORKTREE_ID = `${REPO_ID}::${WORKTREE_PATH}` +const MAIN_WORKTREE_ID = `${REPO_ID}::${REPO_PATH}` + +function makeMeta(overrides: Record = {}) { + return { + displayName: 'feature', + comment: '', + linkedIssue: null, + linkedPR: null, + linkedLinearIssue: null, + linkedGitLabMR: null, + linkedGitLabIssue: null, + isArchived: false, + isUnread: false, + isPinned: false, + sortOrder: 0, + lastActivityAt: 0, + ...overrides + } +} + +/** One repo row owned by an SSH host, stamped with `executionHostId` only — no `connectionId`. */ +function makeStore(repoOverrides: Record) { + const metaById: Record> = { + [WORKTREE_ID]: makeMeta({ + hostId: `ssh:${TARGET_ID}`, + instanceId: '11111111-1111-4111-8111-111111111111' + }), + [MAIN_WORKTREE_ID]: makeMeta({ + displayName: 'main', + hostId: `ssh:${TARGET_ID}`, + instanceId: '22222222-2222-4222-8222-222222222222' + }) + } + const repos = [ + { + id: REPO_ID, + path: REPO_PATH, + displayName: 'app', + badgeColor: 'blue', + addedAt: 1, + ...repoOverrides + } + ] + const store = { + getRepo: (id: string) => repos.find((repo) => repo.id === id), + getRepos: () => repos, + getAllWorktreeMeta: () => metaById, + getWorktreeMeta: (id: string) => metaById[id], + setWorktreeMeta: (id: string, meta: Record) => { + metaById[id] = { ...(metaById[id] ?? makeMeta()), ...meta } as never + return metaById[id] + }, + removeWorktreeMeta: () => {}, + getAllWorktreeLineage: () => ({}), + getAllWorkspaceLineage: () => ({}), + removeWorktreeLineage: vi.fn(), + removeWorkspaceLineage: vi.fn(), + getGitHubCache: () => undefined as never, + getSettings: () => ({ + workspaceDir: '/tmp/workspaces', + nestWorkspaces: false, + refreshLocalBaseRefOnWorktreeCreate: false, + branchPrefix: 'none', + branchPrefixCustom: '' + }), + getProjects: () => [] + } + return store +} + +type RuntimeInternals = { + listResolvedWorktrees: () => Promise<{ id: string; path: string; hostId?: string }[]> +} + +function makeRuntime(repoOverrides: Record): { + runtime: OrcaRuntimeService + list: () => Promise<{ id: string; path: string; hostId?: string }[]> +} { + const runtime = new OrcaRuntimeService(makeStore(repoOverrides) as never) + return { + runtime, + list: () => (runtime as unknown as RuntimeInternals).listResolvedWorktrees() + } +} + +describe('worktree scan execution-host routing', () => { + beforeEach(() => { + getSshGitProviderMock.mockReset() + listWorktreesStrictMock.mockReset() + listWorktreesStrictMock.mockResolvedValue([]) + }) + + it('scans an executionHostId-only SSH repo over its SSH provider, not on the client', async () => { + const listWorktrees = vi.fn(async () => [ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }, + { path: WORKTREE_PATH, head: 'def', branch: 'feature', isBare: false, isMainWorktree: false } + ]) + getSshGitProviderMock.mockReturnValue({ listWorktrees }) + const { list } = makeRuntime({ executionHostId: `ssh:${TARGET_ID}` }) + + const worktrees = await list() + + expect(getSshGitProviderMock).toHaveBeenCalledWith(TARGET_ID) + expect(listWorktrees).toHaveBeenCalledWith(REPO_PATH) + // A client-side `git worktree list` against a remote path is the silent-substitution failure. + expect(listWorktreesStrictMock).not.toHaveBeenCalled() + expect(worktrees.map((worktree) => worktree.path).sort()).toEqual([REPO_PATH, WORKTREE_PATH]) + expect(worktrees.every((worktree) => worktree.hostId === `ssh:${TARGET_ID}`)).toBe(true) + }) + + it('routes the PTY of an executionHostId-only SSH worktree to its host', async () => { + getSshGitProviderMock.mockReturnValue({ + listWorktrees: async () => [ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }, + { + path: WORKTREE_PATH, + head: 'def', + branch: 'feature', + isBare: false, + isMainWorktree: false + } + ] + }) + const { runtime } = makeRuntime({ executionHostId: `ssh:${TARGET_ID}` }) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-1' }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + } as never) + + await runtime.createTerminal(`id:${WORKTREE_ID}`) + + expect(spawn).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: TARGET_ID, cwd: WORKTREE_PATH }) + ) + }) + + it('still routes a legacy connectionId-only SSH repo the same way', async () => { + getSshGitProviderMock.mockReturnValue({ + listWorktrees: async () => [ + { path: REPO_PATH, head: 'abc', branch: 'main', isBare: false, isMainWorktree: true }, + { + path: WORKTREE_PATH, + head: 'def', + branch: 'feature', + isBare: false, + isMainWorktree: false + } + ] + }) + const { runtime } = makeRuntime({ connectionId: TARGET_ID }) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-1' }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + } as never) + + await runtime.createTerminal(`id:${WORKTREE_ID}`) + + expect(getSshGitProviderMock).toHaveBeenCalledWith(TARGET_ID) + expect(spawn).toHaveBeenCalledWith( + expect.objectContaining({ connectionId: TARGET_ID, cwd: WORKTREE_PATH }) + ) + }) +}) From 36e139ed19e172aba11351657a2ed5c0ed661186 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Wed, 2 Sep 2026 17:03:36 -0700 Subject: [PATCH 128/398] docs: update localized Android APK links to 0.0.47 Update localized README links to the latest verified mobile Android release. --- docs/readme/README.es.md | 4 ++-- docs/readme/README.fr.md | 4 ++-- docs/readme/README.ja.md | 4 ++-- docs/readme/README.ko.md | 4 ++-- docs/readme/README.pt.md | 4 ++-- docs/readme/README.zh-CN.md | 4 ++-- 6 files changed, 12 insertions(+), 12 deletions(-) diff --git a/docs/readme/README.es.md b/docs/readme/README.es.md index 638220150b5..f2247e0900d 100644 --- a/docs/readme/README.es.md +++ b/docs/readme/README.es.md @@ -36,7 +36,7 @@ Supervisa y dirige a tus agentes desde el teléfono — recibe una notificación cuando un agente termine y envía instrucciones de seguimiento desde cualquier lugar. -[App Store de iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [APK para Android](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) +[App Store de iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [APK para Android](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) @@ -227,7 +227,7 @@ yay -S stably-orca-bin Vincúlala con tu app de escritorio para supervisar y dirigir a tus agentes desde el teléfono. - **iOS:** [Descargar desde App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) -- **Android:** [Descargar el APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [Descargar el APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.fr.md b/docs/readme/README.fr.md index a64b1af58af..97c78d4e713 100644 --- a/docs/readme/README.fr.md +++ b/docs/readme/README.fr.md @@ -40,7 +40,7 @@ Surveillez et pilotez vos agents depuis votre téléphone — soyez notifié quand un agent termine, et envoyez des instructions de suivi où que vous soyez. -[App Store iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) +[App Store iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) @@ -235,7 +235,7 @@ yay -S stably-orca-bin Associez-la à l'app de bureau pour surveiller et piloter vos agents depuis votre téléphone. - **iOS :** [Télécharger sur l'App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) ou [rejoindre TestFlight](https://testflight.apple.com/join/YjeGMQBA) -- **Android :** [Télécharger l'APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android :** [Télécharger l'APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.ja.md b/docs/readme/README.ja.md index cc589be1cc5..cce2032a67c 100644 --- a/docs/readme/README.ja.md +++ b/docs/readme/README.ja.md @@ -36,7 +36,7 @@ スマートフォンからエージェントを監視・操作 — エージェントの完了を通知で受け取り、どこからでもフォローアップを送信できます。 -[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [ドキュメント →](https://www.onorca.dev/docs/mobile) +[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [ドキュメント →](https://www.onorca.dev/docs/mobile) @@ -227,7 +227,7 @@ yay -S stably-orca-bin デスクトップアプリとペアリングして、スマートフォンからエージェントを監視・操作できます。 - **iOS:** [App Store からダウンロード](https://apps.apple.com/us/app/orca-ide/id6766130217) -- **Android:** [APK をダウンロード](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [APK をダウンロード](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.ko.md b/docs/readme/README.ko.md index 01c8cd71ce1..4a75722ff8c 100644 --- a/docs/readme/README.ko.md +++ b/docs/readme/README.ko.md @@ -36,7 +36,7 @@ 휴대폰에서 에이전트를 모니터링하고 조종하세요 — 에이전트가 완료되면 알림을 받고 어디서든 후속 지시를 보낼 수 있습니다. -[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [문서 →](https://www.onorca.dev/docs/mobile) +[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [문서 →](https://www.onorca.dev/docs/mobile) @@ -230,7 +230,7 @@ yay -S stably-orca-bin 데스크톱 앱과 페어링해 휴대폰에서 에이전트를 모니터링하고 조종하세요. - **iOS:** [App Store에서 다운로드](https://apps.apple.com/us/app/orca-ide/id6766130217) 또는 [TestFlight 참여](https://testflight.apple.com/join/YjeGMQBA) -- **Android:** [APK 0.0.46 다운로드](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [설치 가이드](https://www.onorca.dev/docs/android-apk) +- **Android:** [APK 0.0.47 다운로드](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [설치 가이드](https://www.onorca.dev/docs/android-apk) --- diff --git a/docs/readme/README.pt.md b/docs/readme/README.pt.md index 970fcf6c8a0..86d998a4e5f 100644 --- a/docs/readme/README.pt.md +++ b/docs/readme/README.pt.md @@ -36,7 +36,7 @@ Monitore e conduza seus agentes pelo celular — receba uma notificação quando um agente terminar e envie instruções de acompanhamento de qualquer lugar. -[App Store para iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) +[App Store para iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [APK Android 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [Docs →](https://www.onorca.dev/docs/mobile) @@ -230,7 +230,7 @@ yay -S stably-orca-bin Conecte ao app desktop para monitorar e conduzir seus agentes pelo celular. - **iOS:** [Baixar na App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) ou [entrar no TestFlight](https://testflight.apple.com/join/YjeGMQBA) -- **Android:** [Baixar APK 0.0.46](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [Baixar APK 0.0.47](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- diff --git a/docs/readme/README.zh-CN.md b/docs/readme/README.zh-CN.md index d3ace8a6af4..d7bae3fba9e 100644 --- a/docs/readme/README.zh-CN.md +++ b/docs/readme/README.zh-CN.md @@ -36,7 +36,7 @@ 用手机监控并指挥你的智能体 — 智能体完成时收到通知,随时随地发送后续指令。 -[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) · [文档 →](https://www.onorca.dev/docs/mobile) +[iOS App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) · [Android APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) · [文档 →](https://www.onorca.dev/docs/mobile) @@ -227,7 +227,7 @@ yay -S stably-orca-bin 与桌面应用配对,用手机监控并指挥你的智能体。 - **iOS:** [从 App Store 下载](https://apps.apple.com/us/app/orca-ide/id6766130217) -- **Android:** [下载 APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.46/app-release.apk) +- **Android:** [下载 APK](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.47/app-release.apk) --- From d084a2a36aa6f044f51bfc0227346add1003d23c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 17:46:01 -0700 Subject: [PATCH 129/398] fix(ssh): decide remote-vs-local from the resolved execution host, not a raw field (#18294) `repoIsRemote` read `repo.connectionId` directly. That is one of four spellings of host ownership, so the predicate was wrong in both directions: a row carrying only `executionHostId: 'ssh:'` read as local and got the Linux-only `orca-ide` rename it cannot resolve through the relay shim, while a row that declares itself `local` with a stale `connectionId` read as remote and lost the rename it needs on a Linux desktop. The predicate now resolves the host first and asks "does an SSH target hold this row's files" via `getRepoSshConnectionId`. That keeps a `runtime:` host's nested SSH target remote (that machine reaches the files through its own relay shim) while a runtime with no nested target - a full Orca install - stays local, as do WSL and local. Its call sites did not all want that question: - The four launch-scope sites in main already hold the resolved PTY route on `TerminalWorkspaceLaunchScope.connectionId`. `scope.repo` is documented display metadata and can be a row from a different host than the worktree names, so they now read the route they will actually spawn on. A launch shape that disagrees with its own route is the bug, not a second predicate. - `launchAgentInNewTab` picked its repo row with a host-blind `store.repos.find`, so a worktree that names its own host could be shaped by another host's row. It now resolves through `getConnectionIdFromState`, the same rule the file already used for transcript readability. - `resolveAgentBackgroundLaunchHost` derived the route, the trust write and the launch shape from three reads of the raw field; one resolution now feeds all three. Also converts the raw `repo.connectionId` agent-detection probe eight lines above `buildWorktreeStartupForDraft`'s launch shape, which #17919 deferred precisely because converting it alone would have left that file internally inconsistent. Tests cover two distinct SSH hosts (a single-host fixture passes even when the answer comes off the wrong row, which is how the `ssh:m4air` -> openclaw leak survived review) and a `runtime:` host carrying a nested SSH target. --- ...ca-runtime-agent-session-operation.test.ts | 35 ++++ .../orca-runtime-create-agent-session.ts | 7 +- ...e-get-agent-session-execution-namespace.ts | 5 +- ...resolve-mobile-session-terminal-command.ts | 3 +- ...runtime-resolve-worktree-removal-target.ts | 5 +- .../runtime-worktree-agent-startup.test.ts | 111 +++++++++++- .../runtime/runtime-worktree-agent-startup.ts | 8 +- ...ent-background-session-launch-host.test.ts | 74 ++++++++ .../agent-background-session-launch-host.ts | 11 +- ...h-agent-in-new-tab-host-resolution.test.ts | 167 ++++++++++++++++++ .../src/lib/launch-agent-in-new-tab.ts | 17 +- src/shared/agent-launch-remote.test.ts | 33 ++++ src/shared/agent-launch-remote.ts | 31 +++- 13 files changed, 475 insertions(+), 32 deletions(-) create mode 100644 src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts create mode 100644 src/shared/agent-launch-remote.test.ts diff --git a/src/main/runtime/orca-runtime-agent-session-operation.test.ts b/src/main/runtime/orca-runtime-agent-session-operation.test.ts index ac64dfaa6b7..e99560c7b3a 100644 --- a/src/main/runtime/orca-runtime-agent-session-operation.test.ts +++ b/src/main/runtime/orca-runtime-agent-session-operation.test.ts @@ -149,6 +149,41 @@ describe('agent-session create operation ledger', () => { expect(createTerminal).toHaveBeenCalledOnce() }) + it('shapes the launch for the route it resolved, not a repo row on another host', async () => { + // `scope.repo` is display metadata and can be a row from a different host than the worktree + // names (#11163). Reading it made a locally-routed launch emit the SSH relay shim name. + const runtime = createRuntime({ + supportsAgentSessionClaims: () => true, + supportsAgentSessionCreateOperations: () => true + }) + const internal = runtime as unknown as { + resolveTerminalWorkspaceLaunchScope: ReturnType + } + internal.resolveTerminalWorkspaceLaunchScope.mockResolvedValue({ + id: 'worktree-1', + path: '/repo/worktree-1', + connectionId: null, + // The rival row names openclaw while the worktree resolved to no SSH route at all. + repo: { + id: 'repo-1', + connectionId: 'openclaw', + executionHostId: null, + path: '/srv/openclaw' + }, + folderWorkspace: null + }) + const createTerminal = vi.spyOn(runtime, 'createTerminal').mockResolvedValue(terminal()) + + await runtime.createAgentSession( + request(operationId(), { agent: 'claude-agent-teams', prompt: '' }) + ) + + expect(createTerminal).toHaveBeenCalledWith( + 'id:worktree-1', + expect.objectContaining({ command: expect.stringContaining('orca-ide claude-teams') }) + ) + }) + it('requests exact client legacy fallback before nested SSH side effects', async () => { const runtime = createRuntime() const internal = runtime as unknown as { diff --git a/src/main/runtime/orca-runtime-create-agent-session.ts b/src/main/runtime/orca-runtime-create-agent-session.ts index db2b71a0adc..2ac34f4a920 100644 --- a/src/main/runtime/orca-runtime-create-agent-session.ts +++ b/src/main/runtime/orca-runtime-create-agent-session.ts @@ -16,7 +16,6 @@ import { AGENT_SESSION_OPERATION_PER_CLIENT_LIMIT } from './orca-runtime-core' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { resolveTuiAgentLaunchArgs, @@ -152,9 +151,9 @@ export class OrcaRuntimeWithCreateAgentSession extends OrcaRuntimeWithGetAgentSe throw new Error('Selected agent is disabled. Choose an enabled agent before creating.') } const platform = this.getAgentLaunchPlatformForWorkspace(workspace) - const isRemote = workspace.repo - ? repoIsRemote(workspace.repo) - : Boolean(workspace.connectionId) + // Why: `workspace.repo` is display metadata and may be a row from another host; the launch + // shape must match the PTY route this scope already resolved. + const isRemote = Boolean(workspace.connectionId) const shell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts index 7172331a551..afbb70fc286 100644 --- a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts +++ b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts @@ -11,7 +11,6 @@ import type { } from '../../shared/agent-session-host-authority' import { canonicalizeAgentSessionIdentity } from './agent-session-claim-identity' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { buildAgentResumeStartupPlan } from '../../shared/tui-agent-startup' import { @@ -123,7 +122,9 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim throw new Error('Selected agent is disabled. Choose an enabled agent before resuming.') } const platform = this.getAgentLaunchPlatformForWorkspace(workspace) - const isRemote = workspace.repo ? repoIsRemote(workspace.repo) : Boolean(workspace.connectionId) + // Why: `workspace.repo` is display metadata and may be a row from another host; the launch + // shape must match the PTY route this scope already resolved. + const isRemote = Boolean(workspace.connectionId) const shell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts b/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts index ab63b16ee53..db30b7e2d11 100644 --- a/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts +++ b/src/main/runtime/orca-runtime-resolve-mobile-session-terminal-command.ts @@ -5,7 +5,6 @@ import type { WorktreeStartupLaunch } from '../../shared/worktree/launch-types' import type { TuiAgent } from '../../shared/tui-agent' import type { SleepingAgentLaunchConfig } from '../../shared/agent-session-resume' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { buildAgentStartupPlan } from '../../shared/tui-agent-startup' import { @@ -54,7 +53,7 @@ export class OrcaRuntimeWithResolveMobileSessionTerminalCommand extends OrcaRunt // Why: mobile may be iOS while the shell host is Windows/macOS/Linux or SSH Linux; quote for the host shell. const platform = this.getAgentLaunchPlatformForWorkspace(workspace) // Why: SSH runs the CLI through the relay shim (plain `orca`), so the Linux-only `orca-ide` rename must not apply. - const isRemote = workspace.repo ? repoIsRemote(workspace.repo) : repoIsRemote(workspace) + const isRemote = Boolean(workspace.connectionId) const queuedShell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts b/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts index 58fd6231fc7..0d934548332 100644 --- a/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts +++ b/src/main/runtime/orca-runtime-resolve-worktree-removal-target.ts @@ -14,7 +14,6 @@ import type { ForceDeleteWorktreeBranchResult } from '../../shared/worktree/crea import type { RuntimeTerminalRename } from '../../shared/runtime-types' import type { TerminalWorkspaceLaunchScope } from './runtime-legacy-worker-terminal-recovery-types' import type { TerminalCreateOptions } from './runtime-terminal-contracts' -import { repoIsRemote } from '../../shared/agent-launch-remote' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' import { resolveBareAgentLaunchCommand } from './runtime-agent-launch-resolution' @@ -175,7 +174,9 @@ export class OrcaRuntimeWithResolveWorktreeRemovalTarget extends OrcaRuntimeWith const settings = store.getSettings() const platform = this.getAgentLaunchPlatformForWorkspace(workspace) - const isRemote = workspace.repo ? repoIsRemote(workspace.repo) : Boolean(workspace.connectionId) + // Why: `workspace.repo` is display metadata and may be a row from another host; the launch + // shape must match the PTY route this scope already resolved. + const isRemote = Boolean(workspace.connectionId) const queuedShell = resolveLocalWindowsAgentStartupShell({ platform, isRemote, diff --git a/src/main/runtime/runtime-worktree-agent-startup.test.ts b/src/main/runtime/runtime-worktree-agent-startup.test.ts index e276595c4dd..87fcfd9dd58 100644 --- a/src/main/runtime/runtime-worktree-agent-startup.test.ts +++ b/src/main/runtime/runtime-worktree-agent-startup.test.ts @@ -1,14 +1,119 @@ import { describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../shared/repo-types' const mocks = vi.hoisted(() => ({ markCodexProjectTrusted: vi.fn(), markCopilotFolderTrusted: vi.fn(), - markCursorWorkspaceTrusted: vi.fn() + markCursorWorkspaceTrusted: vi.fn(), + detectRemoteAgents: vi.fn(), + detectInstalledAgentsWithShellPathHydration: vi.fn() })) -vi.mock('../agent-trust-presets', () => mocks) +vi.mock('../agent-trust-presets', () => ({ + markCodexProjectTrusted: mocks.markCodexProjectTrusted, + markCopilotFolderTrusted: mocks.markCopilotFolderTrusted, + markCursorWorkspaceTrusted: mocks.markCursorWorkspaceTrusted +})) -import { markLocalWorktreeTrusted } from './runtime-worktree-agent-startup' +vi.mock('../preflight/agent-detection', () => ({ + detectRemoteAgents: mocks.detectRemoteAgents, + detectInstalledAgentsWithShellPathHydration: mocks.detectInstalledAgentsWithShellPathHydration +})) + +import { + buildWorktreeStartupForAgent, + buildWorktreeStartupForDraft, + markLocalWorktreeTrusted +} from './runtime-worktree-agent-startup' + +function makeRepo(fields: Partial): Repo { + return { + id: 'repo-1', + name: 'repo', + path: '/srv/repo', + connectionId: null, + executionHostId: null, + ...fields + } as Repo +} + +const settings = { + agentCmdOverrides: {}, + agentDefaultArgs: {}, + agentDefaultEnv: {}, + disabledTuiAgents: [], + defaultTuiAgent: undefined, + terminalWindowsShell: null +} as never + +/** The launched CLI name is the whole decision: `orca` is the relay shim, `orca-ide` is local. */ +function launchCliNameFor(repo: Repo): string { + return buildWorktreeStartupForAgent({ + repo, + settings, + agent: 'claude-agent-teams', + getLaunchPlatform: () => 'linux', + toSessionOptions: () => undefined + }).startup.command.split(' ')[0]! +} + +describe('buildWorktreeStartupForAgent host resolution', () => { + // Why two hosts: one SSH fixture passes even when the launch shape is resolved off another + // host's row, which is the shape of the `ssh:m4air` -> openclaw leak. + it('drops the Linux-only rename for both spellings of SSH ownership on two hosts', () => { + expect(launchCliNameFor(makeRepo({ connectionId: 'm4air' }))).toBe('orca') + expect(launchCliNameFor(makeRepo({ executionHostId: 'ssh:openclaw' }))).toBe('orca') + }) + + it('keeps the Linux rename for a local row carrying a stale connection', () => { + expect(launchCliNameFor(makeRepo({ connectionId: 'm4air', executionHostId: 'local' }))).toBe( + 'orca-ide' + ) + }) + + it('drops the rename for a runtime host reaching a nested SSH target', () => { + expect( + launchCliNameFor(makeRepo({ connectionId: 'nested', executionHostId: 'runtime:vm-1' })) + ).toBe('orca') + }) + + it('keeps the rename for a runtime host with no nested SSH target', () => { + expect(launchCliNameFor(makeRepo({ executionHostId: 'runtime:vm-1' }))).toBe('orca-ide') + }) +}) + +describe('buildWorktreeStartupForDraft agent detection', () => { + it('probes the SSH host named only by executionHostId instead of this client', async () => { + mocks.detectRemoteAgents.mockResolvedValueOnce(['claude']) + mocks.detectInstalledAgentsWithShellPathHydration.mockResolvedValue([]) + + const result = await buildWorktreeStartupForDraft({ + repo: makeRepo({ executionHostId: 'ssh:openclaw' }), + settings, + draft: 'ship it', + getLaunchPlatform: () => 'linux' + }) + + expect(mocks.detectRemoteAgents).toHaveBeenCalledWith({ connectionId: 'openclaw' }) + expect(mocks.detectInstalledAgentsWithShellPathHydration).not.toHaveBeenCalled() + expect(result?.agent).toBe('claude') + }) + + it('probes this client for a local row carrying a stale connection', async () => { + mocks.detectRemoteAgents.mockClear() + mocks.detectInstalledAgentsWithShellPathHydration.mockResolvedValueOnce(['claude']) + + const result = await buildWorktreeStartupForDraft({ + repo: makeRepo({ connectionId: 'm4air', executionHostId: 'local' }), + settings, + draft: 'ship it', + getLaunchPlatform: () => 'linux' + }) + + expect(mocks.detectRemoteAgents).not.toHaveBeenCalled() + expect(result?.agent).toBe('claude') + }) +}) describe('markLocalWorktreeTrusted', () => { it('waits for the Codex trust write before resolving', async () => { diff --git a/src/main/runtime/runtime-worktree-agent-startup.ts b/src/main/runtime/runtime-worktree-agent-startup.ts index 7c662d633e2..8663771e998 100644 --- a/src/main/runtime/runtime-worktree-agent-startup.ts +++ b/src/main/runtime/runtime-worktree-agent-startup.ts @@ -3,6 +3,7 @@ import type { Repo } from '../../shared/repo-types' import type { TuiAgent } from '../../shared/tui-agent' import type { WorktreeStartupLaunch } from '../../shared/worktree/launch-types' import { repoIsRemote } from '../../shared/agent-launch-remote' +import { getRepoSshConnectionId } from '../../shared/execution-host' import { isTuiAgent, TUI_AGENT_CONFIG } from '../../shared/tui-agent-config' import { isTuiAgentEnabled, pickTuiAgent } from '../../shared/tui-agent-selection' import { @@ -55,10 +56,13 @@ export async function buildWorktreeStartupForDraft( : null if (!agent) { let detected: string[] = [] + // Why: detection has to run on the machine that will run the agent, and SSH ownership has two + // spellings — the raw field probes this client for an `executionHostId: 'ssh:*'`-only repo. + const sshConnectionId = getRepoSshConnectionId(repo) try { // Why: startup-draft fallback can run from sparse runtime launch envs too. - detected = repo.connectionId - ? await detectRemoteAgents({ connectionId: repo.connectionId }) + detected = sshConnectionId + ? await detectRemoteAgents({ connectionId: sshConnectionId }) : await detectInstalledAgentsWithShellPathHydration() } catch { detected = [] diff --git a/src/renderer/src/lib/agent-background-session-launch-host.test.ts b/src/renderer/src/lib/agent-background-session-launch-host.test.ts index 86fd8577fae..ef00efa5145 100644 --- a/src/renderer/src/lib/agent-background-session-launch-host.test.ts +++ b/src/renderer/src/lib/agent-background-session-launch-host.test.ts @@ -71,6 +71,80 @@ describe('resolveAgentBackgroundLaunchHost', () => { ).toThrow('unavailable or ambiguous') }) + // Why two hosts: a single-SSH fixture passes even when the route is read off another host's + // row, which is the shape of the `ssh:m4air` -> openclaw leak. + it('routes both spellings of SSH ownership to their own host', () => { + const legacy = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: 'm4air', + executionHostId: null, + path: '/srv/repo' + } as never + }) + const unified = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: null, + executionHostId: 'ssh:openclaw', + path: '/srv/repo' + } as never + }) + + expect(legacy).toMatchObject({ + connectionId: 'm4air', + isRemote: true, + expectedConnectionId: 'm4air' + }) + expect(unified).toMatchObject({ + connectionId: 'openclaw', + isRemote: true, + expectedConnectionId: 'openclaw' + }) + }) + + it('keeps a local row with a stale connection off the SSH route', () => { + const host = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: 'm4air', + executionHostId: 'local', + path: '/srv/repo' + } as never + }) + + expect(host).toMatchObject({ + connectionId: null, + isRemote: false, + expectedConnectionId: null + }) + }) + + it('keeps a runtime host reaching a nested SSH target remote', () => { + const host = resolveAgentBackgroundLaunchHost({ + store: makeFolderHostState({ connectionId: null, folderPath: '/project' }) as never, + worktreeId: 'repo-1::/srv/repo', + worktreePath: '/srv/repo', + repo: { + id: 'repo-1', + connectionId: 'nested', + executionHostId: 'runtime:vm-1', + path: '/srv/repo' + } as never + }) + + expect(host).toMatchObject({ connectionId: 'nested', isRemote: true }) + }) + it('uses Linux startup quoting for a local WSL folder', () => { const folderPath = '\\\\wsl.localhost\\Ubuntu\\home\\me\\project' const host = resolveAgentBackgroundLaunchHost({ diff --git a/src/renderer/src/lib/agent-background-session-launch-host.ts b/src/renderer/src/lib/agent-background-session-launch-host.ts index 7919f9f3af5..300dec7f6ff 100644 --- a/src/renderer/src/lib/agent-background-session-launch-host.ts +++ b/src/renderer/src/lib/agent-background-session-launch-host.ts @@ -6,6 +6,7 @@ import { getFolderWorkspaceConnectionId } from '@/lib/folder-workspace-connectio import { parseWorkspaceKey } from '../../../shared/workspace-scope' import { isWindowsAbsolutePathLike } from '../../../shared/cross-platform-path' import { repoIsRemote } from '../../../shared/agent-launch-remote' +import { getRepoSshConnectionId } from '../../../shared/execution-host' import { isWslUncPath } from '../../../shared/wsl-paths' type LaunchStore = ReturnType @@ -41,14 +42,18 @@ export function resolveAgentBackgroundLaunchHost(args: { }): AgentBackgroundLaunchHost { const { store, worktreeId, worktreePath, repo } = args if (repo) { + // Why: SSH ownership has two spellings, so the raw field spawns an `executionHostId: 'ssh:*'`-only + // repo on the client with a remote path. One resolution feeds the route, the trust write and the + // launch shape, which must not disagree about the host. + const sshConnectionId = getRepoSshConnectionId(repo) return { - connectionId: repo.connectionId ?? null, + connectionId: sshConnectionId, platform: getAgentLaunchPlatformForRepo( repo, - repo.connectionId ? undefined : getLocalProjectExecutionRuntimeContext(store, worktreeId) + sshConnectionId ? undefined : getLocalProjectExecutionRuntimeContext(store, worktreeId) ), isRemote: repoIsRemote(repo), - expectedConnectionId: repo.connectionId ?? null + expectedConnectionId: sshConnectionId } } const folderWorkspaceConnectionId = resolveFolderWorkspaceConnectionIdForLaunch(store, worktreeId) diff --git a/src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts b/src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts new file mode 100644 index 00000000000..bcb09c1b17c --- /dev/null +++ b/src/renderer/src/lib/launch-agent-in-new-tab-host-resolution.test.ts @@ -0,0 +1,167 @@ +// Execution-host coverage for launchAgentInNewTab, split from launch-agent-in-new-tab.test.ts to +// keep both files within the lines budget. + +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const mockCreateTab = vi.fn() +const mockQueueTabStartupCommand = vi.fn() + +type StoreRepo = { + id: string + connectionId: string | null + executionHostId?: string | null + path: string +} + +type StoreWorktree = { + id: string + repoId: string + projectId: string + hostId?: string | null + path: string + displayName: string +} + +const store = { + activeRepoId: 'repo-1', + activeWorktreeId: 'wt-1', + settings: { + agentCmdOverrides: {} as Record, + agentDefaultArgs: {} as Record, + agentDefaultEnv: {} as Record>, + activeRuntimeEnvironmentId: null as string | null + }, + projects: [{ id: 'repo-1', localWindowsRuntimePreference: { kind: 'inherit-global' as const } }], + repos: [] as StoreRepo[], + folderWorkspaces: [] as unknown[], + projectGroups: [] as unknown[], + sshConnectionStates: new Map(), + transientClearedAgentStatusConnectionIds: {} as Record, + worktreesByRepo: {} as Record, + allWorktrees: vi.fn(() => store.worktreesByRepo['repo-1'] ?? []), + tabsByWorktree: { 'wt-1': [{ id: 'tab-1' }] }, + openFiles: [] as { id: string; worktreeId: string }[], + browserTabsByWorktree: {} as Record, + tabBarOrderByWorktree: {} as Record, + terminalLayoutsByTabId: {} as Record< + string, + { activeLeafId: string | null; ptyIdsByLeafId?: Record } + >, + ptyIdsByTabId: {} as Record, + createTab: mockCreateTab, + closeTab: vi.fn(), + queueTabStartupCommand: mockQueueTabStartupCommand, + setActiveTabType: vi.fn(), + setTabBarOrder: vi.fn(), + setAgentStatus: vi.fn(), + seedNativeChatLaunchPrompt: vi.fn(), + seedNativeChatLaunchDraft: vi.fn(), + markNativeChatLaunchPromptFailed: vi.fn() +} + +vi.mock('@/store', () => ({ useAppStore: { getState: () => store } })) + +vi.mock('sonner', () => ({ toast: { message: vi.fn(), error: vi.fn() } })) + +vi.mock('@/components/tab-bar/reconcile-order', () => ({ + reconcileTabOrder: vi.fn( + (_stored, termIds: string[], editorIds: string[], browserIds: string[]) => [ + ...termIds, + ...editorIds, + ...browserIds + ] + ) +})) + +vi.mock('@/lib/agent-paste-draft', () => ({ pasteDraftWhenAgentReady: vi.fn() })) + +vi.mock('@/lib/telemetry', () => ({ + track: vi.fn(), + tuiAgentToAgentKind: (agent: string) => agent +})) + +vi.mock('@/runtime/web-runtime-session', () => ({ + createWebRuntimeSessionTerminal: vi.fn(), + isWebRuntimeSessionActive: vi.fn(() => false), + isWebTerminalSurfaceTabId: vi.fn(() => false) +})) + +function worktreeOn(hostId: string, path: string): StoreWorktree { + return { id: 'wt-1', repoId: 'repo-1', projectId: 'repo-1', hostId, path, displayName: 'main' } +} + +async function launchOnLinux(): Promise { + const { launchAgentInNewTab } = await import('./launch-agent-in-new-tab') + launchAgentInNewTab({ agent: 'claude-agent-teams', worktreeId: 'wt-1', launchPlatform: 'linux' }) +} + +function queuedCommand(): string { + return mockQueueTabStartupCommand.mock.calls[0]?.[1]?.command +} + +describe('launchAgentInNewTab execution host resolution', () => { + beforeEach(() => { + vi.clearAllMocks() + mockCreateTab.mockReturnValue({ id: 'tab-1' }) + store.settings = { + agentCmdOverrides: {}, + agentDefaultArgs: {}, + agentDefaultEnv: {}, + activeRuntimeEnvironmentId: null + } + store.tabsByWorktree = { 'wt-1': [{ id: 'tab-1' }] } + store.openFiles = [] + store.browserTabsByWorktree = {} + store.tabBarOrderByWorktree = {} + store.terminalLayoutsByTabId = {} + store.ptyIdsByTabId = {} + }) + + it('shapes the launch from the worktree host, not a rival repo row on another SSH host', async () => { + // `store.repos.find` is host-blind, so a worktree that names its own host could be shaped by + // an `ssh:openclaw` row it has nothing to do with (#11163). + store.repos = [ + { id: 'repo-1', connectionId: 'openclaw', path: '/srv/openclaw' }, + { id: 'repo-1', connectionId: null, executionHostId: 'local', path: '/repo' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('local', '/repo/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca-ide claude-teams '--dangerously-skip-permissions'") + }) + + it('keeps a worktree on one SSH host remote while a rival row names another', async () => { + store.repos = [ + { id: 'repo-1', connectionId: 'openclaw', path: '/srv/openclaw' }, + { id: 'repo-1', connectionId: null, executionHostId: 'ssh:m4air', path: '/srv/m4air' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('ssh:m4air', '/srv/m4air/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca claude-teams '--dangerously-skip-permissions'") + }) + + it('keeps a runtime host reaching a nested SSH target on the relay shim name', async () => { + store.repos = [ + { id: 'repo-1', connectionId: 'nested', executionHostId: 'runtime:vm-1', path: '/srv/vm' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('runtime:vm-1', '/srv/vm/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca claude-teams '--dangerously-skip-permissions'") + }) + + it('keeps a runtime host with no nested SSH target on the local CLI name', async () => { + store.repos = [ + { id: 'repo-1', connectionId: null, executionHostId: 'runtime:vm-1', path: '/srv/vm' } + ] + store.worktreesByRepo = { 'repo-1': [worktreeOn('runtime:vm-1', '/srv/vm/worktree')] } + + await launchOnLinux() + + expect(queuedCommand()).toBe("orca-ide claude-teams '--dangerously-skip-permissions'") + }) +}) diff --git a/src/renderer/src/lib/launch-agent-in-new-tab.ts b/src/renderer/src/lib/launch-agent-in-new-tab.ts index b6cbbb736d8..bf4eda09890 100644 --- a/src/renderer/src/lib/launch-agent-in-new-tab.ts +++ b/src/renderer/src/lib/launch-agent-in-new-tab.ts @@ -22,7 +22,6 @@ import { } from '../../../shared/tui-agent-launch-defaults' import { resolveLocalWindowsAgentStartupShell } from '../../../shared/windows-terminal-shell' import { TUI_AGENT_CONFIG } from '../../../shared/tui-agent-config' -import { repoIsRemote } from '../../../shared/agent-launch-remote' import { seedCommandCodeSubmittedPromptStatus } from '@/lib/command-code-prompt-status-seed' import type { TuiAgent } from '../../../shared/tui-agent' import type { LaunchSource } from '../../../shared/telemetry-events' @@ -96,16 +95,23 @@ export function launchAgentInNewTab(args: LaunchAgentInNewTabArgs): LaunchAgentI const store = useAppStore.getState() const worktree = store.allWorktrees?.().find((entry: { id: string }) => entry.id === worktreeId) const repo = worktree ? store.repos?.find((entry) => entry.id === worktree.repoId) : null + // Why: `store.repos.find` is host-blind and the same repo id can exist on local, SSH and runtime + // hosts, so the row it returns can belong to a different host than the worktree names (#11163). + // The shared resolver answers from the worktree's own host; `undefined` (rival rows disagree) is + // not evidence of a remote, and main rejects that launch anyway. + const worktreeSshConnectionId = getConnectionIdFromState(store, worktreeId) const resolvedLaunchPlatform = launchPlatform ?? (repo ? getAgentLaunchPlatformForRepo( repo, - repo.connectionId ? undefined : getLocalProjectExecutionRuntimeContext(store, worktreeId) + worktreeSshConnectionId + ? undefined + : getLocalProjectExecutionRuntimeContext(store, worktreeId) ) : CLIENT_PLATFORM) // Why: SSH remotes deploy the shim as plain `orca`, so skip the Linux-only `orca-ide` rename for remote launches. - const isRemote = repo ? repoIsRemote(repo) : false + const isRemote = Boolean(worktreeSshConnectionId) const queuedShell = resolveLocalWindowsAgentStartupShell({ platform: resolvedLaunchPlatform, isRemote, @@ -127,9 +133,8 @@ export function launchAgentInNewTab(args: LaunchAgentInNewTabArgs): LaunchAgentI agent, promptDelivery: viewModePromptDelivery, launchDraftText: trimmedPrompt, - nativeChatTranscriptIsLocalReadable: isNativeChatTranscriptLocalReadable( - getConnectionIdFromState(store, worktreeId) - ) + nativeChatTranscriptIsLocalReadable: + isNativeChatTranscriptLocalReadable(worktreeSshConnectionId) } const initialViewModeProps = initialAgentTabViewModeProps(store.settings, initialViewModeOptions) const startupPlanBase = { diff --git a/src/shared/agent-launch-remote.test.ts b/src/shared/agent-launch-remote.test.ts new file mode 100644 index 00000000000..4656dab39fc --- /dev/null +++ b/src/shared/agent-launch-remote.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { repoIsRemote } from './agent-launch-remote' + +describe('repoIsRemote', () => { + it('reads both spellings of SSH ownership on two different hosts', () => { + // Why two hosts: a single-host fixture passes even when the predicate answers from the wrong + // row, which is how the `ssh:m4air` -> openclaw leak survived review. + expect(repoIsRemote({ connectionId: 'm4air', executionHostId: null })).toBe(true) + expect(repoIsRemote({ connectionId: null, executionHostId: 'ssh:openclaw' })).toBe(true) + expect(repoIsRemote({ connectionId: 'm4air', executionHostId: 'ssh:m4air' })).toBe(true) + }) + + it('answers local for a row that declares itself local with a stale connection', () => { + expect(repoIsRemote({ connectionId: 'm4air', executionHostId: 'local' })).toBe(false) + }) + + it('keeps a runtime host with a nested SSH target remote', () => { + expect(repoIsRemote({ connectionId: 'nested-target', executionHostId: 'runtime:vm-1' })).toBe( + true + ) + }) + + it('keeps a runtime host with no nested SSH target local-shaped', () => { + // A runtime with no nested target is a full Orca install, not a relay shim, so it keeps the + // platform CLI name. + expect(repoIsRemote({ connectionId: null, executionHostId: 'runtime:vm-1' })).toBe(false) + }) + + it('keeps plain local and WSL rows local', () => { + expect(repoIsRemote({ connectionId: null, executionHostId: null })).toBe(false) + expect(repoIsRemote({ connectionId: null, executionHostId: 'local' })).toBe(false) + }) +}) diff --git a/src/shared/agent-launch-remote.ts b/src/shared/agent-launch-remote.ts index 08482815859..bec5aae49b9 100644 --- a/src/shared/agent-launch-remote.ts +++ b/src/shared/agent-launch-remote.ts @@ -1,11 +1,26 @@ +import type { Repo } from './repo-types' +import { getRepoSshConnectionId } from './execution-host' + /** - * Why: a repo reached over SSH runs the Orca CLI through the relay shim, which - * is always deployed as plain `orca` (Unix) / `orca.cmd` (Windows). The - * Linux-only `orca-ide` rename — which exists solely to avoid shadowing the - * GNOME Orca screen reader on a local desktop — must not be applied to those - * remotes, or `orca-ide claude-teams` lands on a PATH where it does not exist. - * `connectionId` is the SSH signal; WSL and local stay false. + * Why: a repo reached over SSH runs the Orca CLI through the relay shim, which is always deployed + * as plain `orca` (Unix) / `orca.cmd` (Windows). The Linux-only `orca-ide` rename — which exists + * solely to avoid shadowing the GNOME Orca screen reader on a local desktop — must not be applied + * to those remotes, or `orca-ide claude-teams` lands on a PATH where it does not exist. + * + * The question is "does an SSH target hold this row's files", not "what may this client dial", so + * it resolves the execution host instead of reading the raw `connectionId` field. SSH ownership has + * two spellings and the raw read is wrong in both directions: + * + * - a row carrying only `executionHostId: 'ssh:'` reads as local and gets the `orca-ide` + * rename it cannot resolve on the remote; + * - a row that declares itself `local` while a stale `connectionId` survives reads as remote and + * loses the rename it needs on a Linux desktop. + * + * `runtime:` keeps its nested SSH target (that machine reaches the files through its own relay + * shim), while a runtime host with no nested target is a full Orca install and stays false — as do + * WSL and local. Callers routing a client-local PTY want `getSshTargetIdForExecutionHost` instead; + * callers that already hold a resolved launch connection should read that, not re-derive here. */ -export function repoIsRemote(repo: { connectionId?: string | null }): boolean { - return Boolean(repo.connectionId) +export function repoIsRemote(repo: Pick): boolean { + return getRepoSshConnectionId(repo) !== null } From e827ce2ccb72c23df8a5d6e8ec3c35861b1b1adb Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 01:01:23 +0000 Subject: [PATCH 130/398] Update README downloads badge --- docs/assets/readme-downloads.svg | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index 0708d09d993..39fbcf45af1 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 36m + + downloads: 37m @@ -15,7 +15,7 @@ downloads downloads - 36m - 36m + 37m + 37m From 953df47fc440db62a21f15df9c3e4fd92e3a3f1d Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:04:48 -0700 Subject: [PATCH 131/398] feat(providers): dispatch git and filesystem providers by execution host (#18296) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `const c = repo.connectionId; c ? sshProvider(c) : local()` overloads `null` to mean both "resolved: local" and "could not resolve", so every path that cannot determine the host silently runs remote work on the client (#11163). It also cannot express a `runtime:` host at all. Add a host-keyed dispatch whose input is an `ExecutionHostId` — never null — with `local`, `ssh` and `runtime` as three symmetric entries, and which throws on an id that names no host instead of degrading to this machine. `ssh` carries `provider: null` for "remote, currently unreachable", which is now a different answer from "local" rather than the same one. `runtime:` is a distinct entry rather than a provider because main does not execute runtime hosts at all: they are forwarded over the environment transport, and a runtime row's `connectionId` names a target in the server's namespace. Dialing it from this client's SSH table would trade a silent-local bug for a silent-wrong-host one. First migrations, both to rows resolved via `getRepoExecutionHostId`: - repo-worktrees: an `executionHostId: 'ssh:*'`-only row no longer lists, root-matches, or strict-lists against a same-named local path. - workspace-space-repo-scan: same for the size scan, and `isRemote` no longer contradicts the `executionHostId` emitted beside it. --- .../execution-host-provider-dispatch.test.ts | 105 ++++++++++++ .../execution-host-provider-dispatch.ts | 155 ++++++++++++++++++ src/main/repo-worktrees.test.ts | 61 ++++++- src/main/repo-worktrees.ts | 40 +++-- src/main/workspace-space-repo-scan.ts | 88 +++++----- 5 files changed, 400 insertions(+), 49 deletions(-) create mode 100644 src/main/providers/execution-host-provider-dispatch.test.ts create mode 100644 src/main/providers/execution-host-provider-dispatch.ts diff --git a/src/main/providers/execution-host-provider-dispatch.test.ts b/src/main/providers/execution-host-provider-dispatch.test.ts new file mode 100644 index 00000000000..fd68deea787 --- /dev/null +++ b/src/main/providers/execution-host-provider-dispatch.test.ts @@ -0,0 +1,105 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { + ExecutionHostNotDispatchableError, + requireFilesystemProviderForHost, + requireGitProviderForHost, + resolveFilesystemRouteForHost, + resolveGitRouteForHost, + UnresolvableExecutionHostError +} from './execution-host-provider-dispatch' +import { registerSshGitProvider, unregisterSshGitProvider } from './ssh-git-dispatch' +import { + registerSshFilesystemProvider, + unregisterSshFilesystemProvider +} from './ssh-filesystem-dispatch' + +const connectionId = 'host-dispatch-target' +const gitProvider = { listWorktrees: async () => [] } as never +const filesystemProvider = { readDir: async () => [] } as never + +describe('execution host provider dispatch', () => { + afterEach(() => { + unregisterSshGitProvider(connectionId) + unregisterSshFilesystemProvider(connectionId) + }) + + it('routes `local` to the local entry rather than to a provider', () => { + expect(resolveGitRouteForHost('local')).toEqual({ kind: 'local', hostId: 'local' }) + expect(resolveFilesystemRouteForHost('local')).toEqual({ kind: 'local', hostId: 'local' }) + }) + + it('routes an ssh host to its registered provider', () => { + registerSshGitProvider(connectionId, gitProvider) + registerSshFilesystemProvider(connectionId, filesystemProvider) + + expect(resolveGitRouteForHost(`ssh:${connectionId}`)).toEqual({ + kind: 'ssh', + hostId: `ssh:${connectionId}`, + connectionId, + provider: gitProvider + }) + expect(resolveFilesystemRouteForHost(`ssh:${connectionId}`)).toEqual({ + kind: 'ssh', + hostId: `ssh:${connectionId}`, + connectionId, + provider: filesystemProvider + }) + expect(requireGitProviderForHost(`ssh:${connectionId}`)).toBe(gitProvider) + expect(requireFilesystemProviderForHost(`ssh:${connectionId}`)).toBe(filesystemProvider) + }) + + it('answers `unreachable`, not `local`, for an ssh host with no registered provider', () => { + const route = resolveGitRouteForHost(`ssh:${connectionId}`) + + // The distinction the old `connectionId ? ssh : local` shape could not spell. + expect(route.kind).toBe('ssh') + expect(route.kind === 'ssh' && route.provider).toBeNull() + expect(() => requireGitProviderForHost(`ssh:${connectionId}`)).toThrow( + /Remote connection dropped/ + ) + expect(() => requireFilesystemProviderForHost(`ssh:${connectionId}`)).toThrow( + /Remote connection dropped/ + ) + }) + + it('routes a runtime host to its own entry instead of collapsing it into local', () => { + expect(resolveGitRouteForHost('runtime:env-7')).toEqual({ + kind: 'runtime', + hostId: 'runtime:env-7', + environmentId: 'env-7' + }) + expect(resolveFilesystemRouteForHost('runtime:env-7')).toEqual({ + kind: 'runtime', + hostId: 'runtime:env-7', + environmentId: 'env-7' + }) + }) + + it('refuses to hand a runtime host to this process’s ssh table', () => { + // A runtime repo row carries the *server's* nested target id. Dialling it here would reach a + // same-named target in this client's namespace. + registerSshGitProvider(connectionId, gitProvider) + + expect(() => requireGitProviderForHost('runtime:env-7')).toThrow( + ExecutionHostNotDispatchableError + ) + expect(() => requireFilesystemProviderForHost('runtime:env-7')).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it('refuses to serve a local host from the remote-only accessor', () => { + expect(() => requireGitProviderForHost('local')).toThrow(ExecutionHostNotDispatchableError) + expect(() => requireFilesystemProviderForHost('local')).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it.each([null, undefined, '', 'nonsense', 'ssh:', 'runtime:', 'ssh:a|b'])( + 'throws instead of answering local for the unresolvable host %p', + (hostId) => { + expect(() => resolveGitRouteForHost(hostId)).toThrow(UnresolvableExecutionHostError) + expect(() => resolveFilesystemRouteForHost(hostId)).toThrow(UnresolvableExecutionHostError) + } + ) +}) diff --git a/src/main/providers/execution-host-provider-dispatch.ts b/src/main/providers/execution-host-provider-dispatch.ts new file mode 100644 index 00000000000..1034ceac140 --- /dev/null +++ b/src/main/providers/execution-host-provider-dispatch.ts @@ -0,0 +1,155 @@ +/** + * Host-keyed provider dispatch: one entry per execution host kind, with `local` among them. + * + * The incumbent spelling across main is `const c = repo.connectionId; c ? sshProvider(c) : local()`, + * where `null` means *both* "resolved: this is local" and "could not resolve". Every path that + * cannot determine the host therefore answers "local" and runs remote work on the client — the + * #11163 defect class, which has produced a reproduced cross-host leak (an `ssh:` worktree + * resolving to another target) and near-misses where a transcript that exists only on a remote host + * would have been read locally. The shape also cannot express a `runtime:` host at all. + * + * This module removes that spelling. Its input is an `ExecutionHostId`, which is never null, and an + * id that names no host throws instead of degrading. `getRepoExecutionHostId` / + * `getWorktreeExecutionHostId` / `resolveWorktreeExecutionHost` are the resolution layer that feeds + * it; the last one already answers `unresolved` as a distinct verdict rather than "local". + * + * Why a route union rather than a uniform `getGitProviderForHost(): IGitProvider`, which is the + * VS Code shape (`registerProvider(Schemas.file, …)` symmetric with `Schemas.vscodeRemote`, and + * `ENOPRO` when nothing matches). Two properties of this process, not style preferences: + * + * - `local` git and filesystem work is free functions taking per-worktree execution options + * (`wslDistro`, `sharedLinkPaths`, admission tier), not an `IGitProvider`. There is no local + * provider object to register, and a stateless one would silently drop WSL routing. + * - `runtime:` is not executed in this process *at all*. It is forwarded over the + * environment's transport (`runtimeEnvironments:call`) and the receiving server normalizes it to + * its own `local`. A repo row on a runtime host carries the server's *nested* SSH target in + * `connectionId`; that id is addressable only as the pair (environmentId, targetId). Handing it + * to this client's SSH table would dial a same-named target in the wrong namespace — turning a + * silent-local bug into a silent-wrong-host bug. `host-repo-catalog-snapshot` and + * `host-qualified-worktree-listing` already reject runtime hosts for the same reason. + * + * So the answer is Zed's shape — an enum on the owner (`Local { fs }` vs `Remote { … }`) — and the + * three kinds are symmetric variants of it. Callers switch exhaustively, so `runtime` can no longer + * collapse into `local` by omission. + * + * Note the deliberate second distinction inside the `ssh` variant: `provider: null` means "this host + * is remote and currently unreachable", which is not the same answer as "this host is local" and can + * no longer be spelled the same way. That mirrors the `live` / `unverifiable` / `exited` rule in + * docs/reference/ssh-execution-boundary.md — loss of contact is never evidence of locality. + */ + +import { + parseExecutionHostId, + type ExecutionHostId, + type LOCAL_EXECUTION_HOST_ID, + type ParsedExecutionHost +} from '../../shared/execution-host' +import { getSshGitProvider, SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from './ssh-git-dispatch' +import { + getSshFilesystemProvider, + SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE +} from './ssh-filesystem-dispatch' +import type { IFilesystemProvider, IGitProvider } from './types' + +/** An id that names no execution host. Never degrade to local — that is the whole defect class. */ +export class UnresolvableExecutionHostError extends Error { + constructor(readonly hostId: string | null | undefined) { + super( + `Cannot route work: ${JSON.stringify(hostId ?? null)} names no execution host. ` + + 'Refusing to fall back to this machine.' + ) + this.name = 'UnresolvableExecutionHostError' + } +} + +/** Asking this process for a host it does not execute is a routing mistake, not a fallback. */ +export class ExecutionHostNotDispatchableError extends Error { + constructor(readonly hostId: ExecutionHostId) { + super(`Execution host ${hostId} is not dispatched by this process.`) + this.name = 'ExecutionHostNotDispatchableError' + } +} + +type LocalRoute = { kind: 'local'; hostId: typeof LOCAL_EXECUTION_HOST_ID } +type RuntimeRoute = { kind: 'runtime'; hostId: `runtime:${string}`; environmentId: string } +type SshRoute = { + kind: 'ssh' + hostId: `ssh:${string}` + connectionId: string + /** `null` is "remote, currently unreachable" — never "local". */ + provider: TProvider | null +} + +export type ExecutionHostGitRoute = LocalRoute | RuntimeRoute | SshRoute +export type ExecutionHostFilesystemRoute = LocalRoute | RuntimeRoute | SshRoute + +// Takes an unvalidated string rather than `ExecutionHostId`: validating is the point, and host +// ids also arrive from persistence and IPC where the compiler cannot vouch for them. +function parseRoutableHost(hostId: string | null | undefined): ParsedExecutionHost { + const parsed = parseExecutionHostId(hostId) + if (!parsed) { + throw new UnresolvableExecutionHostError(hostId) + } + return parsed +} + +export function resolveGitRouteForHost(hostId: string | null | undefined): ExecutionHostGitRoute { + const parsed = parseRoutableHost(hostId) + switch (parsed.kind) { + case 'local': + return { kind: 'local', hostId: parsed.id } + case 'ssh': + return { + kind: 'ssh', + hostId: parsed.id, + connectionId: parsed.targetId, + provider: getSshGitProvider(parsed.targetId) ?? null + } + case 'runtime': + return { kind: 'runtime', hostId: parsed.id, environmentId: parsed.environmentId } + } +} + +export function resolveFilesystemRouteForHost( + hostId: string | null | undefined +): ExecutionHostFilesystemRoute { + const parsed = parseRoutableHost(hostId) + switch (parsed.kind) { + case 'local': + return { kind: 'local', hostId: parsed.id } + case 'ssh': + return { + kind: 'ssh', + hostId: parsed.id, + connectionId: parsed.targetId, + provider: getSshFilesystemProvider(parsed.targetId) ?? null + } + case 'runtime': + return { kind: 'runtime', hostId: parsed.id, environmentId: parsed.environmentId } + } +} + +/** For call sites that are structurally remote-only: local and runtime are both routing errors. */ +export function requireGitProviderForHost(hostId: string | null | undefined): IGitProvider { + const route = resolveGitRouteForHost(hostId) + if (route.kind !== 'ssh') { + throw new ExecutionHostNotDispatchableError(route.hostId) + } + if (!route.provider) { + throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) + } + return route.provider +} + +export function requireFilesystemProviderForHost( + hostId: string | null | undefined +): IFilesystemProvider { + const route = resolveFilesystemRouteForHost(hostId) + if (route.kind !== 'ssh') { + throw new ExecutionHostNotDispatchableError(route.hostId) + } + if (!route.provider) { + throw new Error(SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE) + } + return route.provider +} diff --git a/src/main/repo-worktrees.test.ts b/src/main/repo-worktrees.test.ts index d1c3935c1e1..a6b20c5445f 100644 --- a/src/main/repo-worktrees.test.ts +++ b/src/main/repo-worktrees.test.ts @@ -1,4 +1,4 @@ -import { beforeEach, describe, expect, it, vi } from 'vitest' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' const { listWorktreeGraphMock, listWorktreesMock, listWorktreesStrictMock } = vi.hoisted(() => ({ listWorktreeGraphMock: vi.fn(), @@ -19,6 +19,8 @@ import { listRepoWorktreeGraph, listRepoWorktrees } from './repo-worktrees' +import { registerSshGitProvider, unregisterSshGitProvider } from './providers/ssh-git-dispatch' +import { WorktreeCatalogUnavailableError } from '../shared/worktree/worktree-catalog-availability' describe('repo-worktrees', () => { beforeEach(() => { @@ -196,6 +198,63 @@ describe('repo-worktrees', () => { expect(listWorktreesStrictMock).not.toHaveBeenCalled() }) + // #11163: a row may spell its owner only as `executionHostId`. Reading `connectionId` answers + // "local" for it and runs the listing against a same-named path on this machine. + describe('rows that spell their owner only as executionHostId', () => { + const sshOnlyRepo = { + id: 'repo-1', + path: '/srv/repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0, + kind: 'git' as const, + executionHostId: 'ssh:host-a' as const + } + + afterEach(() => { + unregisterSshGitProvider('host-a') + unregisterSshGitProvider('nested-target') + }) + + it('never lists an ssh-owned row with local git', async () => { + await expect(listRepoWorktrees(sshOnlyRepo)).rejects.toThrow(WorktreeCatalogUnavailableError) + expect(listWorktreesMock).not.toHaveBeenCalled() + }) + + it('lists an ssh-owned row through its registered provider', async () => { + const listWorktrees = vi.fn().mockResolvedValue([{ path: '/srv/repo' }]) + registerSshGitProvider('host-a', { listWorktrees } as never) + + await expect(listRepoWorktrees(sshOnlyRepo)).resolves.toEqual([{ path: '/srv/repo' }]) + expect(listWorktrees).toHaveBeenCalledWith('/srv/repo') + expect(listWorktreesMock).not.toHaveBeenCalled() + }) + + it('keeps an ssh-owned root out of the local repo-root match', () => { + expect(isRepoRoot([sshOnlyRepo], '/srv/repo')).toBe(false) + }) + + it('rejects strict local listing for an ssh-owned row', async () => { + await expect(listLocalRepoWorktreesStrict(sshOnlyRepo)).rejects.toThrow('remote repository') + expect(listWorktreesStrictMock).not.toHaveBeenCalled() + }) + + it('refuses to answer a runtime-owned row from a same-named local target', async () => { + const listWorktrees = vi.fn().mockResolvedValue([{ path: '/wrong/host' }]) + registerSshGitProvider('nested-target', { listWorktrees } as never) + + await expect( + listRepoWorktrees({ + ...sshOnlyRepo, + executionHostId: 'runtime:env-7', + connectionId: 'nested-target' + }) + ).rejects.toThrow(WorktreeCatalogUnavailableError) + expect(listWorktrees).not.toHaveBeenCalled() + expect(listWorktreesMock).not.toHaveBeenCalled() + }) + }) + it('treats Windows repo root casing differences as the same local root', () => { const repos = [ { diff --git a/src/main/repo-worktrees.ts b/src/main/repo-worktrees.ts index b7c6e16f91a..228f303bfa2 100644 --- a/src/main/repo-worktrees.ts +++ b/src/main/repo-worktrees.ts @@ -2,7 +2,8 @@ import type { Repo } from '../shared/repo-types' import type { GitWorktreeInfo } from '../shared/worktree/types' import { listWorktreeGraph, listWorktrees, listWorktreesStrict } from './git/worktree' import { isFolderRepo } from '../shared/repo-kind' -import { getSshGitProvider } from './providers/ssh-git-dispatch' +import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' +import { resolveGitRouteForHost } from './providers/execution-host-provider-dispatch' import { areWorktreePathsEqual } from './ipc/worktree-logic' import { WorktreeCatalogUnavailableError } from '../shared/worktree/worktree-catalog-availability' @@ -16,8 +17,12 @@ function hasLocalRepoWorktreeListOptions(options: LocalRepoWorktreeListOptions | } export function isRepoRoot(repos: Repo[], resolvedTarget: string): boolean { + // Why: `!repo.connectionId` matched a remote path against a local one for a row that spells its + // owner only as `executionHostId: 'ssh:'`. Resolve the host instead of reading one field. return repos.some( - (repo) => !repo.connectionId && areWorktreePathsEqual(repo.path, resolvedTarget) + (repo) => + getRepoExecutionHostId(repo) === LOCAL_EXECUTION_HOST_ID && + areWorktreePathsEqual(repo.path, resolvedTarget) ) } @@ -41,17 +46,24 @@ export async function listRepoWorktrees( if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] } - if (repo.connectionId) { - const provider = getSshGitProvider(repo.connectionId) + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + if (route.kind === 'runtime') { + // A runtime row's `connectionId` names a target in the *server's* namespace, not one this + // client may dial. Reading it here would answer from a same-named local target. + throw new WorktreeCatalogUnavailableError( + `Worktree catalog unavailable for ${repo.path}: host ${route.hostId} is not reachable from this process.` + ) + } + if (route.kind === 'ssh') { // Why: runtime worktree resolution can run before SSH providers have reattached during startup. // Never fall back to local git against a server path, and never report the unreachable host as an // empty catalog (#14004) — callers treat a resolved listing as authoritative. - if (!provider) { + if (!route.provider) { throw new WorktreeCatalogUnavailableError( - `Worktree catalog unavailable for ${repo.path}: SSH connection "${repo.connectionId}" is not connected.` + `Worktree catalog unavailable for ${repo.path}: SSH connection "${route.connectionId}" is not connected.` ) } - return await provider.listWorktrees(repo.path) + return await route.provider.listWorktrees(repo.path) } return hasLocalRepoWorktreeListOptions(options) ? await listWorktrees(repo.path, options) @@ -72,9 +84,15 @@ export async function listRepoWorktreeGraph( if (isFolderRepo(repo)) { return [createFolderWorktree(repo)] } - if (repo.connectionId) { - const provider = getSshGitProvider(repo.connectionId) - return provider ? await provider.listWorktrees(repo.path) : [] + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + // An unreachable remote host answers `[]` here, unlike listRepoWorktrees above, which throws. + // Preserved as-is: this call site's callers treat the graph as best-effort. The inconsistency is + // real but is a separate behavior decision from resolving the host correctly. + if (route.kind === 'runtime') { + return [] + } + if (route.kind === 'ssh') { + return route.provider ? await route.provider.listWorktrees(repo.path) : [] } return hasLocalRepoWorktreeListOptions(options) ? await listWorktreeGraph(repo.path, options) @@ -85,7 +103,7 @@ export async function listLocalRepoWorktreesStrict( repo: Repo, options?: LocalRepoWorktreeListOptions ): Promise { - if (repo.connectionId) { + if (getRepoExecutionHostId(repo) !== LOCAL_EXECUTION_HOST_ID) { throw new Error('Cannot list worktrees for a remote repository') } if (isFolderRepo(repo)) { diff --git a/src/main/workspace-space-repo-scan.ts b/src/main/workspace-space-repo-scan.ts index 9a0bc316403..3bf3c6cf1f4 100644 --- a/src/main/workspace-space-repo-scan.ts +++ b/src/main/workspace-space-repo-scan.ts @@ -9,11 +9,13 @@ import type { WorkspaceSpaceWorktree } from '../shared/workspace-space-types' import { mapWithConcurrency } from '../shared/map-with-concurrency' -import { getRepoExecutionHostId } from '../shared/execution-host' +import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' import { readWorktreeMetaForHost } from './persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from './worktree-metadata-ownership' -import { getSshFilesystemProvider } from './providers/ssh-filesystem-dispatch' -import { getSshGitProvider } from './providers/ssh-git-dispatch' +import { + resolveFilesystemRouteForHost, + resolveGitRouteForHost +} from './providers/execution-host-provider-dispatch' import { createFolderWorktree, listRepoWorktrees } from './repo-worktrees' import { mergeWorktree } from './ipc/worktree-logic' import { getLocalProjectWorktreeGitOptions } from './project-runtime-git-options' @@ -89,16 +91,25 @@ async function listWorktreesForSpaceScan( if (isFolderRepo(repo)) { return { ok: true, worktrees: [createFolderWorktree(repo)] } } - if (repo.connectionId) { - const provider = getSshGitProvider(repo.connectionId) - if (!provider) { + // Why: the raw `connectionId` field answers "local" for a row that spells its owner only as + // `executionHostId: 'ssh:'`, which sizes a same-named path on this machine instead. + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + if (route.kind === 'runtime') { + return { + ok: false, + status: 'unavailable', + error: `Host ${route.hostId} is not reachable from this process.` + } + } + if (route.kind === 'ssh') { + if (!route.provider) { return { ok: false, status: 'unavailable', - error: `SSH connection "${repo.connectionId}" is not connected.` + error: `SSH connection "${route.connectionId}" is not connected.` } } - const worktrees = await provider.listWorktrees(repo.path, { signal }) + const worktrees = await route.provider.listWorktrees(repo.path, { signal }) throwIfWorkspaceSpaceScanAborted(signal) return { ok: true, worktrees } } @@ -175,7 +186,7 @@ export async function scanWorkspaceSpaceRepo(args: { executionHostId: getRepoExecutionHostId(repo), displayName: repo.displayName, path: repo.path, - isRemote: Boolean(repo.connectionId), + isRemote: getRepoExecutionHostId(repo) !== LOCAL_EXECUTION_HOST_ID, worktreeCount: 0, scannedWorktreeCount: 0, unavailableWorktreeCount: 1, @@ -193,7 +204,7 @@ export async function scanWorkspaceSpaceRepo(args: { { totalWorktreeCount: progress.totalWorktreeCount + worktrees.length }, options.onProgress ) - const remoteProvider = repo.connectionId ? getSshFilesystemProvider(repo.connectionId) : undefined + const filesystemRoute = resolveFilesystemRouteForHost(getRepoExecutionHostId(repo)) const rows = await mapWithConcurrency(worktrees, WORKTREE_SCAN_CONCURRENCY, async (worktree) => { throwIfWorkspaceSpaceScanAborted(options.signal) reportProgress( @@ -204,33 +215,36 @@ export async function scanWorkspaceSpaceRepo(args: { }, options.onProgress ) - const row = repo.connectionId - ? remoteProvider - ? await scanRemoteWorkspaceSpaceWorktree( - repo, - worktree, - scannedAt, - remoteProvider, - limiters.remoteFallbackTraversal, - options.signal + const row = + filesystemRoute.kind !== 'local' + ? filesystemRoute.kind === 'ssh' && filesystemRoute.provider + ? await scanRemoteWorkspaceSpaceWorktree( + repo, + worktree, + scannedAt, + filesystemRoute.provider, + limiters.remoteFallbackTraversal, + options.signal + ) + : createUnavailableWorkspaceSpaceRow( + repo, + worktree, + scannedAt, + 'unavailable', + filesystemRoute.kind === 'ssh' + ? `SSH filesystem for "${filesystemRoute.connectionId}" is not connected.` + : `Host ${filesystemRoute.hostId} is not reachable from this process.` + ) + : await limiters.localWorktree(() => + scanLocalWorkspaceSpaceWorktree( + repo, + worktree, + scannedAt, + args.readLocalDuDepthOne, + args.normalizeLocalDuPath, + options.signal + ) ) - : createUnavailableWorkspaceSpaceRow( - repo, - worktree, - scannedAt, - 'unavailable', - `SSH filesystem for "${repo.connectionId}" is not connected.` - ) - : await limiters.localWorktree(() => - scanLocalWorkspaceSpaceWorktree( - repo, - worktree, - scannedAt, - args.readLocalDuDepthOne, - args.normalizeLocalDuPath, - options.signal - ) - ) reportProgress( progress, { @@ -265,7 +279,7 @@ export async function scanWorkspaceSpaceRepo(args: { executionHostId: getRepoExecutionHostId(repo), displayName: repo.displayName, path: repo.path, - isRemote: Boolean(repo.connectionId), + isRemote: getRepoExecutionHostId(repo) !== LOCAL_EXECUTION_HOST_ID, worktreeCount: rows.length, ...summary, error: null From 9cda5a9dc01fedeb5aaa90ea68f572f1a9db3606 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 18:13:08 -0700 Subject: [PATCH 132/398] fix(worktrees): stop a resolved-worktree snapshot answering for repos it never saw (#18295) * fix(worktrees): stop a resolved-worktree snapshot answering for repos it never saw `listResolvedWorktrees` caches one fleet-wide snapshot for RESOLVED_WORKTREE_CACHE_TTL_MS (1s) and reuses it on time alone. Nothing invalidates it when a repo is registered, so for up to a second after a repo row lands, every caller reads a snapshot computed before that repo existed -- and reads the gap as a verdict. The visible failure is the SSH skill install. `resolveSkillSshTarget` resolves a workspace-scope destination through that snapshot, so installing into a worktree on a host connected moments earlier threw `skill-install-workspace-not-found`: the client asserting a remote workspace is absent on the strength of client-side bookkeeping that had never looked at the host. That is the shape `docs/reference/ssh-execution-boundary.md` rules out -- absence from a client-side set is not evidence about the execution host. It made `tests/e2e/ssh-skill-installation.spec.ts:108` fail 3 runs in 4 locally and deterministically in the Docker SSH lane, where connect-then-install lands inside the one-second window every time. The snapshot now carries the repo-registration revision it was computed under and is only reused while that revision still holds. The counter is the one `bumpLocalWorktreeScanGeneration` already advances on every repo add, removal and update, so the check is O(1) and cannot drift from the mutation sites. * fix(worktrees): key the snapshot on repo mutations only, not on generation reads Two things the headless-reattach lane surfaced. The revision I keyed the snapshot on was `generationSequence`, which `getLocalWorktreeScanGeneration` also advances when it mints a key for a repo id nothing has scanned yet. That is a read, not a mutation, so a read path could discard a snapshot that was still perfectly valid -- the mirror image of the staleness this fixes, and a way to make a lookup fail that would otherwise have succeeded. The counter now advances only where the scan generation is actually bumped: repo add, removal, update, and scan-cache invalidation. Separately, `pty-restore-record-seeding.test.ts` primed the cache by writing its private `resolved` field with a literal spelling out `worktrees`, `platformByRepoId` and `expiresAt`. That literal is a second copy of the cache's freshness contract, so adding a field to the real entry left the fake one failing the check: the primed snapshot was rejected, resolution fell through to a real scan, and the headless fixture -- which has no git -- got `selector_not_found`. It now primes through `getSnapshot` so the cache stamps its own entry and the two cannot drift again. The revision never moved during that test (0 before and after), so nothing was being invalidated; the fake entry simply never satisfied the contract. --- .../ipc/pty-restore-record-seeding.test.ts | 24 ++-- src/main/local-worktree-scan-generation.ts | 18 +++ ...-resolved-worktrees-for-explicit-target.ts | 7 +- .../worktree-scan-cache-ttl.spec.ts | 31 +++++ .../runtime-resolved-worktree-cache.test.ts | 112 ++++++++++++++++++ .../runtime-resolved-worktree-cache.ts | 31 ++++- 6 files changed, 207 insertions(+), 16 deletions(-) create mode 100644 src/main/runtime/runtime-resolved-worktree-cache.test.ts diff --git a/src/main/ipc/pty-restore-record-seeding.test.ts b/src/main/ipc/pty-restore-record-seeding.test.ts index 0434be17e67..0f1646e7812 100644 --- a/src/main/ipc/pty-restore-record-seeding.test.ts +++ b/src/main/ipc/pty-restore-record-seeding.test.ts @@ -12,6 +12,9 @@ import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { getDefaultWorkspaceSession } from '../../shared/constants' import { makePaneKey } from '../../shared/stable-pane-id' import { OrcaRuntimeService } from '../runtime/orca-runtime' +import type { RuntimeResolvedWorktreeCache } from '../runtime/runtime-resolved-worktree-cache' +import type { ResolvedWorktree } from '../runtime/runtime-worktree-path-identity' +import { getWorktreeScanMutationRevision } from '../local-worktree-scan-generation' import { registerPtyHandlers, clearProviderPtyState, @@ -359,15 +362,22 @@ describe('registerPtyHandlers', () => { } as never) // Why: selector resolution shells out to git for real repos; prime the // resolved-worktree cache so this headless fixture resolves offline. + // + // Why through getSnapshot and not a hand-written `resolved` entry: the cache decides freshness + // from fields it stamps itself, so a literal that mirrors them is a second copy of that + // contract and goes stale the moment a field is added. Let the cache stamp its own entry. const worktreeResolutionInternals = runtime as unknown as { - buildResolvedWorktreeFromId(id: string): unknown - resolvedWorktrees: object + buildResolvedWorktreeFromId(id: string): ResolvedWorktree + resolvedWorktrees: RuntimeResolvedWorktreeCache } - Reflect.set(worktreeResolutionInternals.resolvedWorktrees, 'resolved', { - worktrees: [worktreeResolutionInternals.buildResolvedWorktreeFromId(worktreeId)], - platformByRepoId: new Map([[repo.id, process.platform]]), - expiresAt: Date.now() + 60_000 - }) + await worktreeResolutionInternals.resolvedWorktrees.getSnapshot( + async () => ({ + worktrees: [worktreeResolutionInternals.buildResolvedWorktreeFromId(worktreeId)], + platformByRepoId: new Map([[repo.id, process.platform]]) + }), + 60_000, + getWorktreeScanMutationRevision() + ) setLocalPtyProvider({ spawn: vi.fn(async () => ({ id: ptyId, diff --git a/src/main/local-worktree-scan-generation.ts b/src/main/local-worktree-scan-generation.ts index c8cc3bd0ff9..a2c86afcc33 100644 --- a/src/main/local-worktree-scan-generation.ts +++ b/src/main/local-worktree-scan-generation.ts @@ -1,5 +1,6 @@ const generationByRepoId = new Map() let generationSequence = 0 +let mutationRevision = 0 export function getLocalWorktreeScanGeneration(repoId: string): number { const existing = generationByRepoId.get(repoId) @@ -13,6 +14,22 @@ export function getLocalWorktreeScanGeneration(repoId: string): number { export function bumpLocalWorktreeScanGeneration(repoId: string): void { generationByRepoId.set(repoId, ++generationSequence) + mutationRevision += 1 +} + +/** + * Advances on every event above that can change what a worktree scan would find — repo add, + * removal, update, and scan-cache invalidation — and on nothing else. A cache that must not answer + * for repos it never saw compares this in O(1) instead of walking the repo list. + * + * Why not `generationSequence`: that also advances when `getLocalWorktreeScanGeneration` mints a key + * for a repo id nothing has scanned yet, which is a read. Keying a snapshot on it would let a read + * path discard a snapshot that is still perfectly valid. + * + * Ordering-only: the value means nothing outside a same-process comparison. + */ +export function getWorktreeScanMutationRevision(): number { + return mutationRevision } export function isLocalWorktreeScanGenerationCurrent(repoId: string, generation: number): boolean { @@ -21,5 +38,6 @@ export function isLocalWorktreeScanGenerationCurrent(repoId: string, generation: export function resetLocalWorktreeScanGenerationsForTests(): void { generationSequence += 1 + mutationRevision += 1 generationByRepoId.clear() } diff --git a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts index 1f0dbe591a3..152ef547889 100644 --- a/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts +++ b/src/main/runtime/orca-runtime-list-known-resolved-worktrees-for-explicit-target.ts @@ -5,6 +5,7 @@ import { splitWorktreeIdForFilesystem } from '../../shared/worktree/id' import { isPathInsideOrEqual } from '../../shared/cross-platform-path' import type { ResolvedWorktreeSnapshot } from './runtime-resolved-worktree-cache' import { RESOLVED_WORKTREE_CACHE_TTL_MS } from './orca-runtime-postlude' +import { getWorktreeScanMutationRevision } from '../local-worktree-scan-generation' import { resolveLocalProjectRuntimeForRepo, resolveLocalProjectRuntimesForRepos @@ -65,8 +66,7 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends /** A warm fleet snapshot already answers any selector for free, so scoped scanning must yield to it. */ protected hasFreshResolvedWorktreeCache(): boolean { - const cached = this.resolvedWorktrees.peek() - return Boolean(cached && cached.expiresAt > Date.now()) + return this.resolvedWorktrees.isFresh(getWorktreeScanMutationRevision()) } protected async listResolvedWorktrees(): Promise { @@ -79,7 +79,8 @@ export class OrcaRuntimeWithListKnownResolvedWorktreesForExplicitTarget extends } return this.resolvedWorktrees.getSnapshot( () => this.computeResolvedWorktrees(), - RESOLVED_WORKTREE_CACHE_TTL_MS + RESOLVED_WORKTREE_CACHE_TTL_MS, + getWorktreeScanMutationRevision() ) } diff --git a/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts index 8dc8d7db3a2..f8ec37f7c2c 100644 --- a/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts +++ b/src/main/runtime/orca-runtime-tests/worktree-scan-cache-ttl.spec.ts @@ -5,6 +5,7 @@ import { resolveWorktreeScanCacheTtlMs } from '../orca-runtime-test-mocks.spec' import { store } from '../orca-runtime-test-fixtures.spec' +import { bumpLocalWorktreeScanGeneration } from '../../local-worktree-scan-generation' describe('resolveWorktreeScanCacheTtlMs', () => { const BASE_TTL_MS = 30_000 @@ -84,4 +85,34 @@ describe('resolveWorktreeScanCacheTtlMs', () => { vi.useRealTimers() } }) + + it('scans a repo registered after the last snapshot instead of answering from it', async () => { + // Why: the fleet snapshot only covers the repos that existed when it ran, so for a full TTL it + // reported a just-connected SSH host as having no worktrees at all — and callers that resolve a + // workspace through it turned that gap into `skill-install-workspace-not-found`. + vi.mocked(listWorktrees).mockClear() + const addedPath = '/tmp/repo-registered-later' + const repos = [ + { id: 'repo-1', path: '/tmp/repo', displayName: 'repo', badgeColor: 'blue', addedAt: 1 } + ] + const runtime = new OrcaRuntimeService({ ...store, getRepos: () => repos } as never) + const internals = runtime as unknown as { listResolvedWorktrees: () => Promise } + const scanCallsFor = (path: string): number => + vi.mocked(listWorktrees).mock.calls.filter((call) => call[0] === path).length + + await internals.listResolvedWorktrees() + expect(scanCallsFor(addedPath)).toBe(0) + + repos.push({ + id: 'repo-added', + path: addedPath, + displayName: 'added', + badgeColor: 'blue', + addedAt: 2 + }) + bumpLocalWorktreeScanGeneration('repo-added') + + await internals.listResolvedWorktrees() + expect(scanCallsFor(addedPath)).toBe(1) + }) }) diff --git a/src/main/runtime/runtime-resolved-worktree-cache.test.ts b/src/main/runtime/runtime-resolved-worktree-cache.test.ts new file mode 100644 index 00000000000..31cc4fa2b60 --- /dev/null +++ b/src/main/runtime/runtime-resolved-worktree-cache.test.ts @@ -0,0 +1,112 @@ +import { describe, expect, it } from 'vitest' +import { RuntimeResolvedWorktreeCache } from './runtime-resolved-worktree-cache' +import type { ResolvedWorktreeSnapshot } from './runtime-resolved-worktree-cache' +import { + bumpLocalWorktreeScanGeneration, + getLocalWorktreeScanGeneration, + getWorktreeScanMutationRevision +} from '../local-worktree-scan-generation' + +function snapshotOf(ids: string[]): ResolvedWorktreeSnapshot { + return { + worktrees: ids.map((id) => ({ id }) as ResolvedWorktreeSnapshot['worktrees'][number]), + platformByRepoId: new Map() + } +} + +describe('RuntimeResolvedWorktreeCache', () => { + it('reuses a snapshot inside the TTL while the repo inventory is unchanged', async () => { + const cache = new RuntimeResolvedWorktreeCache() + let computes = 0 + const compute = async (): Promise => { + computes += 1 + return snapshotOf(['repo-1::/a']) + } + + await cache.getSnapshot(compute, 60_000, 7) + const second = await cache.getSnapshot(compute, 60_000, 7) + + expect(computes).toBe(1) + expect(second.worktrees.map((worktree) => worktree.id)).toEqual(['repo-1::/a']) + }) + + it('recomputes when the repo inventory moved, even well inside the TTL', async () => { + // Why: this is the whole point. A snapshot taken before a repo was registered cannot testify + // that the repo's worktrees are absent — callers read the gap as "workspace not found". + const cache = new RuntimeResolvedWorktreeCache() + const results = [snapshotOf(['repo-1::/a']), snapshotOf(['repo-1::/a', 'repo-2::/b'])] + let computes = 0 + const compute = async (): Promise => results[computes++] + + await cache.getSnapshot(compute, 60_000, 7) + const afterRegistration = await cache.getSnapshot(compute, 60_000, 8) + + expect(computes).toBe(2) + expect(afterRegistration.worktrees.map((worktree) => worktree.id)).toEqual([ + 'repo-1::/a', + 'repo-2::/b' + ]) + }) + + it('does not join an in-flight compute that started under a stale inventory', async () => { + const cache = new RuntimeResolvedWorktreeCache() + const computed: number[] = [] + const compute = async (): Promise => { + computed.push(computed.length) + return snapshotOf([]) + } + + const first = cache.getSnapshot(compute, 60_000, 7) + const second = cache.getSnapshot(compute, 60_000, 8) + await Promise.all([first, second]) + + expect(computed).toHaveLength(2) + }) + + it('reports freshness against the inventory the snapshot was computed under', async () => { + const cache = new RuntimeResolvedWorktreeCache() + await cache.getSnapshot(async () => snapshotOf([]), 60_000, 7) + + expect(cache.isFresh(7)).toBe(true) + expect(cache.isFresh(8)).toBe(false) + cache.invalidateResolved() + expect(cache.isFresh(7)).toBe(false) + }) + + it('keeps a primed snapshot servable when nothing mutated', async () => { + // Why: the headless-reattach fixtures prime this cache once and then resolve a selector off it + // without any git available. Losing freshness for a reason other than a mutation strands them + // on a real scan, which is the failure this pairs with — a lookup that finds nothing because + // the snapshot was dropped, not because the worktree is gone. + const cache = new RuntimeResolvedWorktreeCache() + let computes = 0 + const prime = async (): Promise => { + computes += 1 + return snapshotOf(['repo-restore::/tmp/restore-records']) + } + await cache.getSnapshot(prime, 60_000, getWorktreeScanMutationRevision()) + + // A read that mints a scan generation for a repo nothing has scanned yet is not a mutation. + getLocalWorktreeScanGeneration(`repo-never-scanned-${Math.random()}`) + + expect(cache.isFresh(getWorktreeScanMutationRevision())).toBe(true) + const served = await cache.getSnapshot(prime, 60_000, getWorktreeScanMutationRevision()) + expect(computes).toBe(1) + expect(served.worktrees.map((worktree) => worktree.id)).toEqual([ + 'repo-restore::/tmp/restore-records' + ]) + }) +}) + +describe('getWorktreeScanMutationRevision', () => { + it('advances on a repo mutation and not on a first-seen generation read', () => { + const repoId = `repo-${Math.random()}` + const before = getWorktreeScanMutationRevision() + + getLocalWorktreeScanGeneration(repoId) + expect(getWorktreeScanMutationRevision()).toBe(before) + + bumpLocalWorktreeScanGeneration(repoId) + expect(getWorktreeScanMutationRevision()).toBe(before + 1) + }) +}) diff --git a/src/main/runtime/runtime-resolved-worktree-cache.ts b/src/main/runtime/runtime-resolved-worktree-cache.ts index b7ce735eefe..ef7532a6e91 100644 --- a/src/main/runtime/runtime-resolved-worktree-cache.ts +++ b/src/main/runtime/runtime-resolved-worktree-cache.ts @@ -5,9 +5,10 @@ export type ResolvedWorktreeSnapshot = { platformByRepoId: ReadonlyMap } -type ResolvedCache = ResolvedWorktreeSnapshot & { expiresAt: number } +type ResolvedCache = ResolvedWorktreeSnapshot & { expiresAt: number; inventoryRevision: number } type ResolvedInFlight = { generation: number + inventoryRevision: number promise: Promise } export class RuntimeResolvedWorktreeCache { @@ -19,25 +20,43 @@ export class RuntimeResolvedWorktreeCache { return this.resolved } + /** + * Why the revision and not the TTL alone: a snapshot only answers for the repos that were + * registered when it ran. A repo added afterwards — a remote host the user just connected — + * is missing from it for reasons that have nothing to do with what exists on that host, and + * callers read the gap as a verdict that the worktree does not exist. + */ + isFresh(inventoryRevision: number, now = Date.now()): boolean { + return Boolean( + this.resolved && + this.resolved.inventoryRevision === inventoryRevision && + this.resolved.expiresAt > now + ) + } + async getSnapshot( compute: () => Promise, - ttlMs: number + ttlMs: number, + inventoryRevision: number ): Promise { - if (this.resolved && this.resolved.expiresAt > Date.now()) { + if (this.resolved && this.isFresh(inventoryRevision)) { return this.resolved } const generation = this.resolvedGeneration - if (this.resolvedInFlight?.generation === generation) { + if ( + this.resolvedInFlight?.generation === generation && + this.resolvedInFlight.inventoryRevision === inventoryRevision + ) { return this.resolvedInFlight.promise } const promise = compute() - this.resolvedInFlight = { generation, promise } + this.resolvedInFlight = { generation, inventoryRevision, promise } try { const result = await promise if (generation === this.resolvedGeneration) { // Why stamped on completion, not entry: a compute that spent longer than the TTL would // otherwise publish an already-expired entry, so the next poll recomputes the same slow path. - this.resolved = { ...result, expiresAt: Date.now() + ttlMs } + this.resolved = { ...result, inventoryRevision, expiresAt: Date.now() + ttlMs } } return result } finally { From d5750648c289c3bcfc1a7a0c85b995b87b9b4391 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:26:07 -0700 Subject: [PATCH 133/398] fix(runtime): route runtime Git by resolved execution host, not repo connectionId (#18307) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `RuntimeGitTarget` carried `connectionId?: string` and no host id, so `undefined` spelled three different answers at once — "runtime: host", "unresolved", and "genuinely local". Its sole resolver read `store.getRepo(worktree.repoId)?.connectionId` and never looked at `worktree.hostId`, which outranks every repo row, so one arbitrarily chosen row decided the execution host for 36 downstream dispatches. The target now carries `executionHostId: ExecutionHostId` (never null, never optional), resolved through the shared rule that landed with #17909/#17919 and dispatched through the host-keyed routes from #18296. Dispatch sites call `requireRuntimeGitProvider`, where `null` means exactly one thing: the host is `local` and the command runs here as free functions. Four answers that used to collapse into one: - `ssh:x` with a rival row on `ssh:y` — routes to x. Previously the first row won, which is the reproduced cross-host leak. - `local` with a surviving `connectionId` — a row contradicting itself; no SSH connection is handed out. - `runtime:` — throws `ExecutionHostNotDispatchableError`. Its repo row's connection names a target in the *server's* namespace; dialling it here reaches a same-named target on this client. - rival rows disagreeing with no worktree host — `worktree_execution_host_unresolved`, matching the launch path rather than guessing a row. An unreachable SSH host still throws `SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE`; loss of contact is never evidence of locality (docs/reference/ssh-execution-boundary.md). `resolveWorktreeLaunchHost` keeps its exact signature and now delegates to `resolveWorktreeHostRouting`, the same resolution answering "which host is this on" rather than "what may this client dial" — the git target needs the first question because `local` and `runtime:` are two different non-SSH answers. No wire change: `RuntimeGitTarget` is main-process internal, and the SSH and local model-discovery host keys are byte-identical to before. `RuntimeFileTarget` has the same defect in ~30 filesystem dispatches and is deliberately left for a follow-up. --- docs/reference/ssh-execution-boundary.md | 2 +- .../execution-host-provider-dispatch.ts | 5 +- .../orca-runtime-git-branch-diff.test.ts | 2 +- .../orca-runtime-git-diff-budget.test.ts | 2 +- src/main/runtime/orca-runtime-git.test.ts | 42 ++-- ...runtime-persist-headless-terminal-title.ts | 24 +- .../methods/git-diff-transport-budget.test.ts | 2 +- src/main/runtime/runtime-file-command-host.ts | 4 +- ...ntime-git-branch-compare-admission.test.ts | 3 +- .../runtime-git-command-target.test.ts | 87 ++++++++ .../runtime/runtime-git-command-target.ts | 66 +++++- ...ime-git-conflict-operation-routing.test.ts | 1 + src/main/runtime/runtime-git-diff-commands.ts | 54 ++--- ...ntime-git-execution-host-ownership.test.ts | 2 +- .../runtime-git-generation-admission.test.ts | 3 +- .../runtime-git-generation-commands.ts | 106 ++++----- .../runtime/runtime-git-generation-context.ts | 34 +++ .../runtime/runtime-git-staging-commands.ts | 47 ++-- .../runtime-git-status-admission.test.ts | 5 +- .../runtime/runtime-git-status-commands.ts | 59 ++--- .../runtime/runtime-git-sync-commands.test.ts | 3 +- src/main/runtime/runtime-git-sync-commands.ts | 75 ++----- .../runtime-git-target-execution-host.test.ts | 205 ++++++++++++++++++ .../runtime/worktree-launch-host-repo.test.ts | 55 ++++- src/main/runtime/worktree-launch-host-repo.ts | 55 +++-- 25 files changed, 660 insertions(+), 283 deletions(-) create mode 100644 src/main/runtime/runtime-git-command-target.test.ts create mode 100644 src/main/runtime/runtime-git-target-execution-host.test.ts diff --git a/docs/reference/ssh-execution-boundary.md b/docs/reference/ssh-execution-boundary.md index 45b307411f6..87495960b76 100644 --- a/docs/reference/ssh-execution-boundary.md +++ b/docs/reference/ssh-execution-boundary.md @@ -13,7 +13,7 @@ Two consequences, both non-negotiable: The vocabulary is fixed: **`live` / `unverifiable` / `exited`**, taken from the incumbent `UnstoppedPtyVerdict`. Do not introduce synonyms, and never collapse `unverifiable` into either neighbour. `exited` requires positive evidence of absence from the host that owns the process; a transport failure can only ever produce `unverifiable`. -Rule 1 is stated at `src/main/source-control/repo-default-branch.ts:76-78`, `src/main/repo-worktrees.ts:45-48`, `OrcaRuntimeService.probeWorktreeDrift` in `src/main/runtime/orca-runtime.ts`, and `src/renderer/src/lib/connection-context.ts:22-24`. It is enforced throughout `src/main/runtime/orca-runtime-git.ts` by the guard that throws `SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE` whenever `target.connectionId` is set and no provider is registered — grep that constant for the current call sites rather than trusting a count. +Rule 1 is stated at `src/main/source-control/repo-default-branch.ts:76-78`, `src/main/repo-worktrees.ts:45-48`, `OrcaRuntimeService.probeWorktreeDrift` in `src/main/runtime/orca-runtime.ts`, and `src/renderer/src/lib/connection-context.ts:22-24`. It is enforced throughout `src/main/runtime/orca-runtime-git.ts` by `requireRuntimeGitProvider` in `src/main/runtime/runtime-git-command-target.ts`, which routes on the target's resolved `executionHostId` rather than on a repo row's `connectionId`: it throws `SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE` when an SSH host has no registered provider, throws `ExecutionHostNotDispatchableError` for a `runtime:` host this process does not execute, and returns `null` only for `local`. Grep those two names for the current call sites rather than trusting a count. `src/main/runtime/unstopped-pty-verification.ts:12-16` is the reference implementation of rule 2: it keeps `live` / `unverifiable` / `exited` as three distinct verdicts, and treats "we could not ask" as its own answer. diff --git a/src/main/providers/execution-host-provider-dispatch.ts b/src/main/providers/execution-host-provider-dispatch.ts index 1034ceac140..079905b9b37 100644 --- a/src/main/providers/execution-host-provider-dispatch.ts +++ b/src/main/providers/execution-host-provider-dispatch.ts @@ -45,6 +45,7 @@ import { type ParsedExecutionHost } from '../../shared/execution-host' import { getSshGitProvider, SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from './ssh-git-dispatch' +import type { SshGitProvider } from './ssh-git-provider' import { getSshFilesystemProvider, SSH_FILESYSTEM_PROVIDER_UNAVAILABLE_MESSAGE @@ -80,7 +81,9 @@ type SshRoute = { provider: TProvider | null } -export type ExecutionHostGitRoute = LocalRoute | RuntimeRoute | SshRoute +// The SSH table stores `SshGitProvider`; narrowing the route to `IGitProvider` would drop the +// remote-only methods (commit-message plans, push-target materialization) that callers need. +export type ExecutionHostGitRoute = LocalRoute | RuntimeRoute | SshRoute export type ExecutionHostFilesystemRoute = LocalRoute | RuntimeRoute | SshRoute // Takes an unvalidated string rather than `ExecutionHostId`: validating is the point, and host diff --git a/src/main/runtime/orca-runtime-git-branch-diff.test.ts b/src/main/runtime/orca-runtime-git-branch-diff.test.ts index aba155a6895..b27e09fefba 100644 --- a/src/main/runtime/orca-runtime-git-branch-diff.test.ts +++ b/src/main/runtime/orca-runtime-git-branch-diff.test.ts @@ -45,7 +45,7 @@ describe('RuntimeGitCommands branch diff', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/orca-runtime-git-diff-budget.test.ts b/src/main/runtime/orca-runtime-git-diff-budget.test.ts index b2e894d4bc6..3cc401b155c 100644 --- a/src/main/runtime/orca-runtime-git-diff-budget.test.ts +++ b/src/main/runtime/orca-runtime-git-diff-budget.test.ts @@ -57,7 +57,7 @@ function commands( return new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree, - ...(connectionId ? { connectionId } : {}), + executionHostId: connectionId ? (`ssh:${connectionId}` as const) : ('local' as const), ...(localGitOptions ? { localGitOptions } : {}) }), getRuntimeSettings: () => ({}) as GlobalSettings diff --git a/src/main/runtime/orca-runtime-git.test.ts b/src/main/runtime/orca-runtime-git.test.ts index 70fe676f98c..9b27200945e 100644 --- a/src/main/runtime/orca-runtime-git.test.ts +++ b/src/main/runtime/orca-runtime-git.test.ts @@ -86,9 +86,13 @@ function makeWorktree(path: string, linkedIssue: number | null = null): Resolved return worktree as unknown as ResolvedRuntimeGitWorktree } +function localTarget(worktreePath: string, linkedIssue: number | null = null) { + return { worktree: makeWorktree(worktreePath, linkedIssue), executionHostId: 'local' as const } +} + function makeCommands(worktreePath: string): RuntimeGitCommands { return new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({}) as GlobalSettings }) } @@ -125,6 +129,7 @@ describe('RuntimeGitCommands', () => { mocks.getStatus.mockResolvedValue({ entries: [], conflictOperation: 'none' }) const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree('/workspace/feature'), repo: { path: '/workspace/repo', symlinkPaths: ['node_modules'] } as never }), @@ -146,7 +151,7 @@ describe('RuntimeGitCommands', () => { resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), repo: { path: '/remote/repo', symlinkPaths: ['node_modules'] } as never, - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -162,6 +167,7 @@ describe('RuntimeGitCommands', () => { tempDirs.push(worktreePath) const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree(worktreePath), localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -186,7 +192,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -217,6 +223,7 @@ describe('RuntimeGitCommands', () => { it('prioritizes a local single-file discard without losing WSL routing', async () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree('/workspace/repo'), localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -237,7 +244,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -256,7 +263,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree('/remote/repo'), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -299,7 +306,7 @@ describe('RuntimeGitCommands', () => { message: 'docs: update readme' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({ commitMessageAi: { enabled: true, agentId: 'codex' }, @@ -352,6 +359,7 @@ describe('RuntimeGitCommands', () => { }) const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree: makeWorktree(worktreePath), localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -410,7 +418,7 @@ describe('RuntimeGitCommands', () => { message: 'feat: update readme' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({ sourceControlAi: { @@ -471,7 +479,7 @@ describe('RuntimeGitCommands', () => { } }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({ sourceControlAi: { @@ -542,7 +550,7 @@ describe('RuntimeGitCommands', () => { } }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -595,7 +603,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({ @@ -637,7 +645,7 @@ describe('RuntimeGitCommands', () => { mocks.getStagedCommitContext.mockResolvedValue(context) mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -663,7 +671,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 77), - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) @@ -687,7 +695,7 @@ describe('RuntimeGitCommands', () => { mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const getWorktreeLinkedIssue = vi.fn(() => 321) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, getWorktreeLinkedIssue }) @@ -713,7 +721,7 @@ describe('RuntimeGitCommands', () => { mocks.getStagedCommitContext.mockResolvedValue(context) mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, getWorktreeLinkedIssue: () => null }) @@ -734,7 +742,7 @@ describe('RuntimeGitCommands', () => { mocks.getStagedCommitContext.mockResolvedValue(context) mocks.generateCommitMessageFromContext.mockResolvedValue({ success: true, message: 'docs' }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, // Why: what the host reports when its store is not initialized yet. getWorktreeLinkedIssue: () => undefined @@ -765,7 +773,7 @@ describe('RuntimeGitCommands', () => { mocks.getPullRequestDraftContext.mockResolvedValue(context) mocks.generatePullRequestFieldsFromContext.mockResolvedValue({ success: true, fields: {} }) const commands = new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 123) }), + resolveRuntimeGitTarget: async () => localTarget(worktreePath, 123), getRuntimeSettings: () => ({}) as GlobalSettings, getWorktreeLinkedIssue: () => 321 }) @@ -827,7 +835,7 @@ describe('RuntimeGitCommands', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: makeWorktree(worktreePath, 55), - ...(connectionId ? { connectionId } : {}) + executionHostId: connectionId ? (`ssh:${connectionId}` as const) : ('local' as const) }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts b/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts index 3be605b117a..b9a24685be4 100644 --- a/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts +++ b/src/main/runtime/orca-runtime-persist-headless-terminal-title.ts @@ -8,7 +8,9 @@ import type { } from '../../shared/runtime-types' import type { ResolvedWorktree } from './runtime-worktree-path-identity' import type { Repo } from '../../shared/repo-types' +import { LOCAL_EXECUTION_HOST_ID, type ExecutionHostId } from '../../shared/execution-host' import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options' +import { resolveWorktreeHostRouting } from './worktree-launch-host-repo' export class OrcaRuntimeWithPersistHeadlessTerminalTitle extends OrcaRuntimeWithMoveHeadlessMobileSessionTab { // Persist a manual terminal rename so a headless rebuild keeps the title @@ -157,19 +159,31 @@ export class OrcaRuntimeWithPersistHeadlessTerminalTitle extends OrcaRuntimeWith return await this.notifier.saveMobileMarkdownTab(worktreeId, tabId, baseVersion, content) } + // Why: `getRepo(id)` is host-blind and never read `worktree.hostId`, which outranks every repo + // row. One id can name rows on local, SSH and runtime hosts at once, so an arbitrary row decided + // the execution host for ~36 downstream Git dispatches: a worktree on one SSH host routed to + // another, and a runtime host's *nested* target got dialled in this client's namespace (#11163). protected async resolveRuntimeGitTarget(worktreeSelector: string): Promise<{ worktree: ResolvedWorktree repo?: Repo - connectionId?: string + executionHostId: ExecutionHostId localGitOptions?: { wslDistro?: string } }> { const store = this.requireStore() const worktree = await this.resolveWorktreeSelector(worktreeSelector) - const repo = store.getRepo(worktree.repoId) - const connectionId = repo?.connectionId ?? undefined + const routing = resolveWorktreeHostRouting(store.getRepos(), worktree) + if (routing.kind === 'ambiguous') { + throw new Error('worktree_execution_host_unresolved') + } + const executionHostId = routing.kind === 'resolved' ? routing.hostId : LOCAL_EXECUTION_HOST_ID + // Metadata only (shared-link paths, source-control AI defaults); routing is `executionHostId`. + const repo = + (routing.kind === 'resolved' ? routing.repo : null) ?? store.getRepo(worktree.repoId) const localGitOptions = - repo && !connectionId ? getLocalProjectWorktreeGitOptions(store, repo) : {} - return { worktree, repo, connectionId, localGitOptions } + repo && executionHostId === LOCAL_EXECUTION_HOST_ID + ? getLocalProjectWorktreeGitOptions(store, repo) + : {} + return { worktree, repo, executionHostId, localGitOptions } } protected async resolveRuntimeFileTarget(worktreeSelector: string): Promise<{ diff --git a/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts b/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts index a4b22aaa891..72ae885c9b7 100644 --- a/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts +++ b/src/main/runtime/rpc/methods/git-diff-transport-budget.test.ts @@ -161,7 +161,7 @@ describe('remote git diff transport budget', () => { const commands = new RuntimeGitCommands({ resolveRuntimeGitTarget: async () => ({ worktree: { id: 'wt-1', path: '/remote/repo' } as unknown as ResolvedRuntimeGitWorktree, - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/runtime-file-command-host.ts b/src/main/runtime/runtime-file-command-host.ts index 74f71c9d9ab..b757ab29b14 100644 --- a/src/main/runtime/runtime-file-command-host.ts +++ b/src/main/runtime/runtime-file-command-host.ts @@ -42,9 +42,11 @@ export type RuntimeFileCommandHost = { pathText: string, absolutePath: string ): boolean | Promise + // `executionHostId`, not `connectionId`: a repo row's connection cannot tell `runtime:` from + // `local`, and this contract must not re-introduce that spelling. See runtime-git-command-target. resolveRuntimeGitTarget( selector: string - ): Promise<{ worktree: ResolvedRuntimeFileWorktree; connectionId?: string }> + ): Promise<{ worktree: ResolvedRuntimeFileWorktree; executionHostId: ExecutionHostId }> openFile( worktreeId: string, filePath: string, diff --git a/src/main/runtime/runtime-git-branch-compare-admission.test.ts b/src/main/runtime/runtime-git-branch-compare-admission.test.ts index f2b77c209de..f9d5ffa4cfd 100644 --- a/src/main/runtime/runtime-git-branch-compare-admission.test.ts +++ b/src/main/runtime/runtime-git-branch-compare-admission.test.ts @@ -36,6 +36,7 @@ function makeCommands(overrides: Partial = {}): RuntimeGitDiff head: 'a'.repeat(40) } }, + executionHostId: 'local', localGitOptions: { wslDistro: 'Ubuntu' }, ...overrides } as RuntimeGitTarget @@ -71,7 +72,7 @@ describe('RuntimeGitDiffCommands branch-compare admission', () => { it('forwards background admission to the SSH execution host', async () => { const getBranchCompare = vi.fn().mockResolvedValue({ summary: {}, entries: [] }) mocks.getSshGitProvider.mockReturnValue({ getBranchCompare }) - const commands = makeCommands({ connectionId: 'conn-1' }) + const commands = makeCommands({ executionHostId: 'ssh:conn-1' }) await commands.getRuntimeGitBranchCompare('id:wt-1', 'origin/main', 'background') diff --git a/src/main/runtime/runtime-git-command-target.test.ts b/src/main/runtime/runtime-git-command-target.test.ts new file mode 100644 index 00000000000..0edb9d0518a --- /dev/null +++ b/src/main/runtime/runtime-git-command-target.test.ts @@ -0,0 +1,87 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { ExecutionHostNotDispatchableError } from '../providers/execution-host-provider-dispatch' +import { registerSshGitProvider, unregisterSshGitProvider } from '../providers/ssh-git-dispatch' +import { + localGitOptionsForTarget, + requireRuntimeGitProvider, + runtimeGitRouteForTarget, + type RuntimeGitTarget +} from './runtime-git-command-target' + +const worktree = { + id: 'wt-1', + repoId: 'repo-1', + path: '/srv/app', + git: { path: '/srv/app', branch: 'main', isBare: false, isMainWorktree: false } +} as unknown as RuntimeGitTarget['worktree'] + +function target(overrides: Partial): RuntimeGitTarget { + return { worktree, executionHostId: 'local', ...overrides } +} + +describe('runtime Git target routing', () => { + const registered: string[] = [] + + function register(connectionId: string) { + const provider = { getStatus: async () => ({ entries: [] }) } + registerSshGitProvider(connectionId, provider as never) + registered.push(connectionId) + return provider + } + + afterEach(() => { + for (const connectionId of registered.splice(0)) { + unregisterSshGitProvider(connectionId) + } + }) + + it('routes each ssh host to its own provider', () => { + const m4air = register('m4air') + const openclaw = register('openclaw') + + expect(runtimeGitRouteForTarget(target({ executionHostId: 'ssh:m4air' }))).toEqual({ + kind: 'ssh', + connectionId: 'm4air', + provider: m4air + }) + expect(requireRuntimeGitProvider(target({ executionHostId: 'ssh:openclaw' }))).toBe(openclaw) + }) + + it('answers `local` with no provider, which is the only meaning `null` carries', () => { + expect(runtimeGitRouteForTarget(target({}))).toEqual({ kind: 'local' }) + expect(requireRuntimeGitProvider(target({}))).toBeNull() + }) + + // Loss of contact is never evidence of locality (docs/reference/ssh-execution-boundary.md). + it('keeps an unreachable ssh host remote instead of degrading it to local', () => { + const route = runtimeGitRouteForTarget(target({ executionHostId: 'ssh:gone' })) + + expect(route).toEqual({ kind: 'ssh', connectionId: 'gone', provider: null }) + expect(() => requireRuntimeGitProvider(target({ executionHostId: 'ssh:gone' }))).toThrow( + /Remote connection dropped/ + ) + }) + + it('refuses a runtime host even when a same-named target is registered here', () => { + register('nested-1') + + expect(() => runtimeGitRouteForTarget(target({ executionHostId: 'runtime:env-a' }))).toThrow( + ExecutionHostNotDispatchableError + ) + expect(() => requireRuntimeGitProvider(target({ executionHostId: 'runtime:env-a' }))).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it('keeps WSL routing on the local host and off every other one', () => { + const localGitOptions = { wslDistro: 'Ubuntu' } + + expect(localGitOptionsForTarget(target({ localGitOptions }))).toEqual(localGitOptions) + expect( + localGitOptionsForTarget(target({ executionHostId: 'ssh:m4air', localGitOptions })) + ).toEqual({}) + expect( + localGitOptionsForTarget(target({ executionHostId: 'runtime:env-a', localGitOptions })) + ).toEqual({}) + }) +}) diff --git a/src/main/runtime/runtime-git-command-target.ts b/src/main/runtime/runtime-git-command-target.ts index 47131ce7e63..48c4b5492e2 100644 --- a/src/main/runtime/runtime-git-command-target.ts +++ b/src/main/runtime/runtime-git-command-target.ts @@ -1,7 +1,14 @@ +import { LOCAL_EXECUTION_HOST_ID, type ExecutionHostId } from '../../shared/execution-host' import type { GlobalSettings } from '../../shared/global-settings-types' import type { Repo } from '../../shared/repo-types' import type { GitPushTarget, GitWorktreeInfo, Worktree } from '../../shared/worktree/types' import type { GitRuntimeOptions } from '../git/git-runtime-options' +import { + ExecutionHostNotDispatchableError, + resolveGitRouteForHost +} from '../providers/execution-host-provider-dispatch' +import { SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from '../providers/ssh-git-dispatch' +import type { SshGitProvider } from '../providers/ssh-git-provider' import type { CommitMessageAgentEnvironmentResolvers } from '../text-generation/commit-message-agent-environment' import type { PullRequestLinkedIssueMeta } from '../source-control/pull-request-linked-issue' import { normalizeRuntimeRelativePath } from './runtime-relative-paths' @@ -10,8 +17,21 @@ export type ResolvedRuntimeGitWorktree = Worktree & { git: GitWorktreeInfo } export type RuntimeGitTarget = { worktree: ResolvedRuntimeGitWorktree + /** + * Display and settings metadata only (shared-link paths, source-control AI defaults). It can be + * a same-id row from another host when the worktree's own host carries none, so it must not + * decide routing — `executionHostId` does. + */ repo?: Repo - connectionId?: string + /** + * The host this worktree's Git runs on. Never optional and never null: the field it replaced + * (`connectionId?: string`) spelled "runtime host", "unresolved" and "genuinely local" all as + * `undefined`, so every path that could not resolve answered "local" and ran remote work on the + * client (#11163). Unresolved now fails at resolution time instead of arriving here as a + * silently-local target. + */ + executionHostId: ExecutionHostId + /** Only consulted when `executionHostId` is `local`; see `localGitOptionsForTarget`. */ localGitOptions?: GitRuntimeOptions } @@ -29,8 +49,50 @@ export type RuntimeGitCommandHost = { persistMaterializedPushTarget?(worktreeId: string, pushTarget: GitPushTarget): void } +/** + * The two hosts this process can itself execute a runtime Git command on, narrowed from the shared + * host-keyed route in `src/main/providers/execution-host-provider-dispatch.ts`. + * + * `runtime:` is deliberately not a variant. Its Git is executed by that environment's own + * server, and the SSH target on its repo row is that server's *nested* one — addressable only as + * the pair (environmentId, targetId). Handing that id to this client's SSH table dials a + * same-named target in the wrong namespace, so it throws rather than routing. + */ +export type RuntimeGitRoute = + | { kind: 'local' } + /** `provider: null` is "remote and currently unreachable" — never "run it here". */ + | { kind: 'ssh'; connectionId: string; provider: SshGitProvider | null } + +export function runtimeGitRouteForTarget(target: RuntimeGitTarget): RuntimeGitRoute { + const route = resolveGitRouteForHost(target.executionHostId) + switch (route.kind) { + case 'local': + return { kind: 'local' } + case 'ssh': + return { kind: 'ssh', connectionId: route.connectionId, provider: route.provider } + case 'runtime': + throw new ExecutionHostNotDispatchableError(route.hostId) + } +} + +/** + * `null` means exactly one thing: the host is `local`, and this command runs here as free + * functions. An unreachable SSH host and a `runtime:` host both throw. + */ +export function requireRuntimeGitProvider(target: RuntimeGitTarget): SshGitProvider | null { + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'local') { + return null + } + if (!route.provider) { + throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) + } + return route.provider +} + export function localGitOptionsForTarget(target: RuntimeGitTarget): GitRuntimeOptions { - return target.connectionId ? {} : (target.localGitOptions ?? {}) + // WSL routing describes *this* machine; no remote host may inherit it. + return target.executionHostId === LOCAL_EXECUTION_HOST_ID ? (target.localGitOptions ?? {}) : {} } export function normalizeRuntimeGitRelativePath(filePath: string): string { diff --git a/src/main/runtime/runtime-git-conflict-operation-routing.test.ts b/src/main/runtime/runtime-git-conflict-operation-routing.test.ts index b9d42a0af70..0e6d8d22f5e 100644 --- a/src/main/runtime/runtime-git-conflict-operation-routing.test.ts +++ b/src/main/runtime/runtime-git-conflict-operation-routing.test.ts @@ -22,6 +22,7 @@ describe('getRuntimeGitConflictOperation', () => { const commands = new RuntimeGitStatusCommands({ resolveRuntimeGitTarget: async () => ({ worktree: { path: '/home/me/repo/feature' }, + executionHostId: 'local', localGitOptions: { wslDistro: 'Ubuntu' } }) } as never) diff --git a/src/main/runtime/runtime-git-diff-commands.ts b/src/main/runtime/runtime-git-diff-commands.ts index a6d94775ae5..ef4a4a2290f 100644 --- a/src/main/runtime/runtime-git-diff-commands.ts +++ b/src/main/runtime/runtime-git-diff-commands.ts @@ -14,14 +14,11 @@ import { } from '../git/status' import { awaitWindowsHostGitEnvironmentReady } from '../git/runner' import type { GitAdmissionTier } from '../git/command-runner/git-exec-options' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { normalizeRuntimeRelativePath } from './runtime-relative-paths' import { localGitOptionsForTarget, normalizeRuntimeGitRelativePath, + requireRuntimeGitProvider, type RuntimeGitCommandHost } from './runtime-git-command-target' @@ -38,11 +35,8 @@ export class RuntimeGitDiffCommands { ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return assertGitDiffWithinTransportBudget( await provider.getDiff(target.worktree.path, relativePath, staged, compareAgainstHead), maxContentBytes @@ -63,11 +57,8 @@ export class RuntimeGitDiffCommands { admissionTier: GitAdmissionTier = 'interactive' ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getBranchCompare(target.worktree.path, baseRef, { admissionTier }) } return getBranchCompare(target.worktree.path, baseRef, { @@ -81,11 +72,8 @@ export class RuntimeGitDiffCommands { commitId: string ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getCommitCompare(target.worktree.path, commitId) } return getCommitCompare(target.worktree.path, commitId, { @@ -104,11 +92,8 @@ export class RuntimeGitDiffCommands { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) const oldRelativePath = oldPath ? normalizeRuntimeGitRelativePath(oldPath) : undefined - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const results = await provider.getBranchDiff(target.worktree.path, compare.mergeBase, { includePatch: true, headOid: compare.headOid, @@ -152,11 +137,8 @@ export class RuntimeGitDiffCommands { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeRelativePath(args.filePath) const oldRelativePath = args.oldPath ? normalizeRuntimeRelativePath(args.oldPath) : undefined - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return assertGitDiffWithinTransportBudget( await provider.getCommitDiff(target.worktree.path, { commitOid: args.commitOid, @@ -192,11 +174,8 @@ export class RuntimeGitDiffCommands { ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const normalizedRelativePath = normalizeRuntimeGitRelativePath(relativePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getRemoteFileUrl(target.worktree.path, normalizedRelativePath, line) } await awaitWindowsHostGitEnvironmentReady({ cwd: target.worktree.path }) @@ -208,11 +187,8 @@ export class RuntimeGitDiffCommands { sha: string ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getRemoteCommitUrl(target.worktree.path, sha) } await awaitWindowsHostGitEnvironmentReady({ cwd: target.worktree.path }) diff --git a/src/main/runtime/runtime-git-execution-host-ownership.test.ts b/src/main/runtime/runtime-git-execution-host-ownership.test.ts index 8de68b2e9ca..9d42d003eb6 100644 --- a/src/main/runtime/runtime-git-execution-host-ownership.test.ts +++ b/src/main/runtime/runtime-git-execution-host-ownership.test.ts @@ -38,7 +38,7 @@ function remoteCommands(): RuntimeGitCommands { git: { path: '/remote/repo', branch: 'main', isBare: false, isMainWorktree: false } } as unknown as ResolvedRuntimeGitWorktree return new RuntimeGitCommands({ - resolveRuntimeGitTarget: async () => ({ worktree, connectionId: 'ssh-1' }), + resolveRuntimeGitTarget: async () => ({ worktree, executionHostId: 'ssh:ssh-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) } diff --git a/src/main/runtime/runtime-git-generation-admission.test.ts b/src/main/runtime/runtime-git-generation-admission.test.ts index 4d93779c7a8..6631930082d 100644 --- a/src/main/runtime/runtime-git-generation-admission.test.ts +++ b/src/main/runtime/runtime-git-generation-admission.test.ts @@ -60,6 +60,7 @@ function makeTarget(path: string, overrides: Partial = {}): Ru path, git: { path, branch: 'main', isBare: false, isMainWorktree: false, head: 'a'.repeat(40) } } as RuntimeGitTarget['worktree'], + executionHostId: 'local', ...overrides } } @@ -148,7 +149,7 @@ describe('RuntimeGitGenerationCommands admission', () => { }) const commands = makeCommands( makeTarget('/remote/repo', { - connectionId: 'conn-1', + executionHostId: 'ssh:conn-1', localGitOptions: { wslDistro: 'Ubuntu' } }) ) diff --git a/src/main/runtime/runtime-git-generation-commands.ts b/src/main/runtime/runtime-git-generation-commands.ts index c741525e0a0..bea10b2ef7d 100644 --- a/src/main/runtime/runtime-git-generation-commands.ts +++ b/src/main/runtime/runtime-git-generation-commands.ts @@ -3,12 +3,8 @@ import { getCommitMessageModelDiscoveryHostKey } from '../../shared/commit-messa import type { HostedReviewProvider } from '../../shared/hosted-review' import { withLinkedIssueDraftContext } from '../../shared/source-control-ai-action-variables' import type { TuiAgent } from '../../shared/tui-agent' -import { gitExecFileAsync } from '../git/runner' import { getStagedCommitContext } from '../git/status' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' +import { SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from '../providers/ssh-git-dispatch' import { loadPullRequestLinkedIssue } from '../source-control/pull-request-linked-issue' import { resolveHostedReviewBodyForGeneration } from '../source-control/pull-request-template' import { prepareLocalCommitMessageAgentEnv } from '../text-generation/commit-message-agent-environment' @@ -25,13 +21,18 @@ import { type GeneratePullRequestFieldsResult } from '../text-generation/commit-message-text-generation' import { getPullRequestDraftContext } from '../text-generation/pull-request-context' -import { localGitOptionsForTarget, type RuntimeGitCommandHost } from './runtime-git-command-target' +import { + localGitOptionsForTarget, + runtimeGitRouteForTarget, + type RuntimeGitCommandHost +} from './runtime-git-command-target' import { getRuntimeGitGenerationSettings, linkedIssueForTarget, linkedIssueMetaForTarget, localAgentRuntimeTargetForTarget, localTextGenerationTargetForTarget, + pullRequestDraftGitExec, type RuntimeCommitMessageSettingsOverride } from './runtime-git-generation-context' @@ -43,9 +44,10 @@ export class RuntimeGitGenerationCommands { settingsOverride?: RuntimeCommitMessageSettingsOverride ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) + const route = runtimeGitRouteForTarget(target) const discoveryHostKey = settingsOverride?.commitMessageDiscoveryHostKey ?? - getCommitMessageModelDiscoveryHostKey(target.connectionId ?? null) + getCommitMessageModelDiscoveryHostKey(route.kind === 'ssh' ? route.connectionId : null) const resolvedSettings = settingsOverride?.sourceControlAiResolvedParams ? { ok: true as const, params: settingsOverride.sourceControlAiResolvedParams } : resolveCommitMessageSettings( @@ -62,8 +64,8 @@ export class RuntimeGitGenerationCommands { return { success: false, error: resolvedSettings.error } } - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { + if (route.kind === 'ssh') { + const provider = route.provider if (!provider) { return { success: false, error: SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } } @@ -118,9 +120,11 @@ export class RuntimeGitGenerationCommands { async cancelRuntimeGenerateCommitMessage(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - await provider?.cancelGenerateCommitMessage(target.worktree.path, 'commit-message') + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'ssh') { + // Cancelling an unreachable host is a no-op, not a local cancel: the local registry is keyed + // by path and would abort an unrelated generation running here for the same path. + await route.provider?.cancelGenerateCommitMessage(target.worktree.path, 'commit-message') return { ok: true } } cancelGenerateCommitMessageLocal(target.worktree.path) @@ -140,9 +144,10 @@ export class RuntimeGitGenerationCommands { settingsOverride?: RuntimeCommitMessageSettingsOverride ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) + const route = runtimeGitRouteForTarget(target) const discoveryHostKey = settingsOverride?.commitMessageDiscoveryHostKey ?? - getCommitMessageModelDiscoveryHostKey(target.connectionId ?? null) + getCommitMessageModelDiscoveryHostKey(route.kind === 'ssh' ? route.connectionId : null) const resolvedSettings = settingsOverride?.sourceControlAiResolvedParams ? { ok: true as const, params: settingsOverride.sourceControlAiResolvedParams } : resolveCommitMessageSettings( @@ -159,8 +164,8 @@ export class RuntimeGitGenerationCommands { return { success: false, error: resolvedSettings.error } } - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId && !provider) { + const provider = route.kind === 'ssh' ? route.provider : null + if (route.kind === 'ssh' && !provider) { return { success: false, error: SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } } const issueMeta = linkedIssueMetaForTarget(this.host, target) @@ -168,56 +173,30 @@ export class RuntimeGitGenerationCommands { meta: issueMeta, provider: input.provider, repoPath: target.worktree.path, - connectionId: target.connectionId, - localGitOptions: target.connectionId - ? {} - : { - ...localGitOptionsForTarget(target), - admissionTier: 'interactive' - } + connectionId: route.kind === 'ssh' ? route.connectionId : undefined, + localGitOptions: + route.kind === 'ssh' + ? {} + : { + ...localGitOptionsForTarget(target), + admissionTier: 'interactive' + } }) let context: Awaited> try { const currentBody = await resolveHostedReviewBodyForGeneration({ body: input.body, repoPath: target.worktree.path, - connectionId: target.connectionId, + connectionId: route.kind === 'ssh' ? route.connectionId : undefined, provider: input.provider, useTemplate: input.useTemplate }) - context = target.connectionId - ? await getPullRequestDraftContext( - (argv, commandOptions) => { - const timeoutMs = commandOptions?.timeoutMs ?? commandOptions?.timeout - return timeoutMs === undefined - ? provider!.exec(argv, target.worktree.path) - : provider!.exec(argv, target.worktree.path, { timeoutMs }) - }, - { - base: input.base, - currentTitle: input.title, - currentBody, - currentDraft: input.draft - } - ) - : await getPullRequestDraftContext( - (argv, options) => - gitExecFileAsync(argv, { - cwd: target.worktree.path, - ...localGitOptionsForTarget(target), - ...(options?.maxBuffer === undefined ? {} : { maxBuffer: options.maxBuffer }), - ...(options?.timeoutMs === undefined && options?.timeout === undefined - ? {} - : { timeout: options?.timeoutMs ?? options?.timeout }), - admissionTier: 'interactive' - }), - { - base: input.base, - currentTitle: input.title, - currentBody, - currentDraft: input.draft - } - ) + context = await getPullRequestDraftContext(pullRequestDraftGitExec(target, route), { + base: input.base, + currentTitle: input.title, + currentBody, + currentDraft: input.draft + }) } catch (error) { return { success: false, @@ -234,7 +213,7 @@ export class RuntimeGitGenerationCommands { ...(linkedIssueDetails ? { linkedIssueDetails } : {}) } - if (target.connectionId) { + if (route.kind === 'ssh') { return generatePullRequestFieldsFromContext(context, resolvedSettings.params, { kind: 'remote', cwd: target.worktree.path, @@ -260,9 +239,9 @@ export class RuntimeGitGenerationCommands { async cancelRuntimeGeneratePullRequestFields(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - await provider?.cancelGenerateCommitMessage(target.worktree.path, 'pull-request-fields') + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'ssh') { + await route.provider?.cancelGenerateCommitMessage(target.worktree.path, 'pull-request-fields') return { ok: true } } cancelGeneratePullRequestFieldsLocal(target.worktree.path) @@ -279,10 +258,11 @@ export class RuntimeGitGenerationCommands { const agentCommandOverride = settingsOverride?.agentCmdOverrides?.[typedAgentId] ?? this.host.getRuntimeSettings().agentCmdOverrides?.[typedAgentId] - if (target.connectionId) { - const provider = getSshGitProvider(target.connectionId) + const route = runtimeGitRouteForTarget(target) + if (route.kind === 'ssh') { + const provider = route.provider if (!provider) { - return { success: false, error: `No git provider for connection "${target.connectionId}"` } + return { success: false, error: `No git provider for connection "${route.connectionId}"` } } return discoverCommitMessageModelsRemote( typedAgentId, diff --git a/src/main/runtime/runtime-git-generation-context.ts b/src/main/runtime/runtime-git-generation-context.ts index 70873155b56..141b5f18220 100644 --- a/src/main/runtime/runtime-git-generation-context.ts +++ b/src/main/runtime/runtime-git-generation-context.ts @@ -1,4 +1,6 @@ import type { GlobalSettings } from '../../shared/global-settings-types' +import { gitExecFileAsync } from '../git/runner' +import type { getPullRequestDraftContext } from '../text-generation/pull-request-context' import { mergeLegacyCommitMessageAiIntoSourceControlAi, type ResolvedSourceControlAiGenerationParams @@ -10,9 +12,41 @@ import type { PullRequestLinkedIssueMeta } from '../source-control/pull-request- import { localGitOptionsForTarget, type RuntimeGitCommandHost, + type RuntimeGitRoute, type RuntimeGitTarget } from './runtime-git-command-target' +type PullRequestDraftGitExec = Parameters[0] + +/** Runs the PR draft-context probes on whichever host `route` resolved to. */ +export function pullRequestDraftGitExec( + target: RuntimeGitTarget, + route: RuntimeGitRoute +): PullRequestDraftGitExec { + if (route.kind === 'ssh') { + const provider = route.provider + if (!provider) { + throw new Error('ssh_git_provider_unavailable') + } + return (argv, options) => { + const timeoutMs = options?.timeoutMs ?? options?.timeout + return timeoutMs === undefined + ? provider.exec(argv, target.worktree.path) + : provider.exec(argv, target.worktree.path, { timeoutMs }) + } + } + return (argv, options) => + gitExecFileAsync(argv, { + cwd: target.worktree.path, + ...localGitOptionsForTarget(target), + ...(options?.maxBuffer === undefined ? {} : { maxBuffer: options.maxBuffer }), + ...(options?.timeoutMs === undefined && options?.timeout === undefined + ? {} + : { timeout: options?.timeoutMs ?? options?.timeout }), + admissionTier: 'interactive' + }) +} + export type RuntimeCommitMessageSettingsOverride = Partial< Pick > & { diff --git a/src/main/runtime/runtime-git-staging-commands.ts b/src/main/runtime/runtime-git-staging-commands.ts index a856be7bf7b..97f791ac575 100644 --- a/src/main/runtime/runtime-git-staging-commands.ts +++ b/src/main/runtime/runtime-git-staging-commands.ts @@ -6,13 +6,10 @@ import { stageFile, unstageFile } from '../git/status' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { localGitOptionsForTarget, normalizeRuntimeGitRelativePath, + requireRuntimeGitProvider, type RuntimeGitCommandHost } from './runtime-git-command-target' @@ -22,11 +19,8 @@ export class RuntimeGitStagingCommands { async stageRuntimeGitPath(worktreeSelector: string, filePath: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.stageFile(target.worktree.path, relativePath) return { ok: true } } @@ -40,11 +34,8 @@ export class RuntimeGitStagingCommands { async unstageRuntimeGitPath(worktreeSelector: string, filePath: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.unstageFile(target.worktree.path, relativePath) return { ok: true } } @@ -61,11 +52,8 @@ export class RuntimeGitStagingCommands { ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePaths = filePaths.map((path) => normalizeRuntimeGitRelativePath(path)) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.bulkStageFiles(target.worktree.path, relativePaths) return { ok: true } } @@ -82,11 +70,8 @@ export class RuntimeGitStagingCommands { ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePaths = filePaths.map((path) => normalizeRuntimeGitRelativePath(path)) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.bulkUnstageFiles(target.worktree.path, relativePaths) return { ok: true } } @@ -103,11 +88,8 @@ export class RuntimeGitStagingCommands { ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePaths = filePaths.map((path) => normalizeRuntimeGitRelativePath(path)) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.bulkDiscardChanges(target.worktree.path, relativePaths) return { ok: true } } @@ -121,11 +103,8 @@ export class RuntimeGitStagingCommands { async discardRuntimeGitPath(worktreeSelector: string, filePath: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) const relativePath = normalizeRuntimeGitRelativePath(filePath) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.discardChanges(target.worktree.path, relativePath) return { ok: true } } diff --git a/src/main/runtime/runtime-git-status-admission.test.ts b/src/main/runtime/runtime-git-status-admission.test.ts index 1a0c5a0a47f..8f7525b74f6 100644 --- a/src/main/runtime/runtime-git-status-admission.test.ts +++ b/src/main/runtime/runtime-git-status-admission.test.ts @@ -30,7 +30,10 @@ describe('runtime git status admission', () => { return { entries: [], conflictOperation: 'none' } }) const commands = new RuntimeGitStatusCommands({ - resolveRuntimeGitTarget: async () => ({ worktree: { path: '/workspace/feature' } }) + resolveRuntimeGitTarget: async () => ({ + worktree: { path: '/workspace/feature' }, + executionHostId: 'local' + }) } as never) const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/runtime-git-status-commands.ts b/src/main/runtime/runtime-git-status-commands.ts index 28390cee7c9..bc6213a959f 100644 --- a/src/main/runtime/runtime-git-status-commands.ts +++ b/src/main/runtime/runtime-git-status-commands.ts @@ -14,12 +14,12 @@ import { getSubmoduleStatus as getGitSubmoduleStatus } from '../git/status' import type { GitProviderStatusOptions } from '../providers/types' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { getWorktreeSharedLinkPaths } from '../git/worktree-shared-directories' -import { localGitOptionsForTarget, type RuntimeGitCommandHost } from './runtime-git-command-target' +import { + localGitOptionsForTarget, + requireRuntimeGitProvider, + type RuntimeGitCommandHost +} from './runtime-git-command-target' export class RuntimeGitStatusCommands { constructor(private readonly host: RuntimeGitCommandHost) {} @@ -29,11 +29,8 @@ export class RuntimeGitStatusCommands { options?: GitProviderStatusOptions ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return options ? provider.getStatus(target.worktree.path, options) : provider.getStatus(target.worktree.path) @@ -56,11 +53,8 @@ export class RuntimeGitStatusCommands { area: GitStagingArea = 'unstaged' ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getSubmoduleStatus(target.worktree.path, submodulePath, area) } return getGitSubmoduleStatus(target.worktree.path, submodulePath, { @@ -75,11 +69,8 @@ export class RuntimeGitStatusCommands { relativePaths: string[] ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.checkIgnoredPaths(target.worktree.path, relativePaths) } return checkIgnoredPaths(target.worktree.path, relativePaths, { @@ -93,11 +84,8 @@ export class RuntimeGitStatusCommands { options: GitHistoryOptions = {} ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getHistory(target.worktree.path, options) } return getGitHistory(target.worktree.path, { @@ -109,11 +97,8 @@ export class RuntimeGitStatusCommands { async getRuntimeGitConflictOperation(worktreeSelector: string): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.detectConflictOperation(target.worktree.path) } return detectConflictOperation(target.worktree.path, localGitOptionsForTarget(target)) @@ -124,11 +109,8 @@ export class RuntimeGitStatusCommands { branch: string ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.checkoutBranch(target.worktree.path, branch) return { ok: true, branch } } @@ -141,11 +123,8 @@ export class RuntimeGitStatusCommands { async listRuntimeGitLocalBranches(worktreeSelector: string): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.listLocalBranches(target.worktree.path) } return listLocalBranches(target.worktree.path, localGitOptionsForTarget(target)) diff --git a/src/main/runtime/runtime-git-sync-commands.test.ts b/src/main/runtime/runtime-git-sync-commands.test.ts index f662c909cb6..b34aa171788 100644 --- a/src/main/runtime/runtime-git-sync-commands.test.ts +++ b/src/main/runtime/runtime-git-sync-commands.test.ts @@ -59,6 +59,7 @@ describe('RuntimeGitSyncCommands admission', () => { it('prioritizes local runtime git actions and preserves host routing', async () => { const commands = new RuntimeGitSyncCommands({ resolveRuntimeGitTarget: async () => ({ + executionHostId: 'local', worktree, localGitOptions: { wslDistro: 'Ubuntu' } }), @@ -107,7 +108,7 @@ describe('RuntimeGitSyncCommands admission', () => { const commands = new RuntimeGitSyncCommands({ resolveRuntimeGitTarget: async () => ({ worktree, - connectionId: 'conn-1' + executionHostId: 'ssh:conn-1' }), getRuntimeSettings: () => ({}) as GlobalSettings }) diff --git a/src/main/runtime/runtime-git-sync-commands.ts b/src/main/runtime/runtime-git-sync-commands.ts index f68b82aac79..542d459f743 100644 --- a/src/main/runtime/runtime-git-sync-commands.ts +++ b/src/main/runtime/runtime-git-sync-commands.ts @@ -5,16 +5,13 @@ import { gitSyncForkDefaultBranch } from '../git/fork-sync' import { gitFastForward, gitFetch, gitPull, gitPullRebaseFromBase, gitPush } from '../git/remote' import { abortMerge, abortRebase, commitChanges } from '../git/status' import { getUpstreamStatus } from '../git/upstream' -import { - getSshGitProvider, - SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE -} from '../providers/ssh-git-dispatch' import { materializeWorktreePushTargetRemote, materializeWorktreePushTargetRemoteSsh } from '../ipc/worktree-remote' import { localGitOptionsForTarget, + requireRuntimeGitProvider, type RuntimeGitCommandHost, type RuntimeGitTarget } from './runtime-git-command-target' @@ -37,11 +34,8 @@ export class RuntimeGitSyncCommands { async abortRuntimeGitMerge(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.abortMerge(target.worktree.path) return { ok: true } } @@ -54,11 +48,8 @@ export class RuntimeGitSyncCommands { async abortRuntimeGitRebase(worktreeSelector: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.abortRebase(target.worktree.path) return { ok: true } } @@ -74,11 +65,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.getUpstreamStatus(target.worktree.path, pushTarget) } return getUpstreamStatus(target.worktree.path, pushTarget, localGitOptionsForTarget(target)) @@ -89,11 +77,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -123,11 +108,8 @@ export class RuntimeGitSyncCommands { expectedUpstream: GitForkSyncExpectedUpstream ): Promise { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.syncForkDefaultBranch(target.worktree.path, expectedUpstream) } return gitSyncForkDefaultBranch(target.worktree.path, expectedUpstream, { @@ -141,11 +123,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -175,11 +154,8 @@ export class RuntimeGitSyncCommands { pushTarget?: GitPushTarget ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -206,11 +182,8 @@ export class RuntimeGitSyncCommands { async rebaseRuntimeGitFromBase(worktreeSelector: string, baseRef: string): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { await provider.rebaseFromBase(target.worktree.path, baseRef) return { ok: true } } @@ -228,11 +201,8 @@ export class RuntimeGitSyncCommands { forceWithLease?: boolean ): Promise<{ ok: true }> { const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { const materializedPushTarget = pushTarget ? await materializeWorktreePushTargetRemoteSsh(provider, target.worktree.path, pushTarget) : undefined @@ -268,11 +238,8 @@ export class RuntimeGitSyncCommands { throw new Error('Commit message is required') } const target = await this.host.resolveRuntimeGitTarget(worktreeSelector) - const provider = target.connectionId ? getSshGitProvider(target.connectionId) : null - if (target.connectionId) { - if (!provider) { - throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) - } + const provider = requireRuntimeGitProvider(target) + if (provider) { return provider.commit(target.worktree.path, message) } return commitChanges(target.worktree.path, message, { diff --git a/src/main/runtime/runtime-git-target-execution-host.test.ts b/src/main/runtime/runtime-git-target-execution-host.test.ts new file mode 100644 index 00000000000..e4d21c8b477 --- /dev/null +++ b/src/main/runtime/runtime-git-target-execution-host.test.ts @@ -0,0 +1,205 @@ +// `resolveRuntimeGitTarget` read `store.getRepo(worktree.repoId)?.connectionId` and never looked at +// `worktree.hostId`, so one arbitrarily chosen row decided the execution host for ~36 downstream +// Git dispatches. `undefined` there meant "runtime host", "unresolved" and "genuinely local" at +// once (#11163). These cases pin all four answers end to end, through the real SSH provider table. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } +})) + +import type * as GitStatusModule from '../git/status' + +const mocks = vi.hoisted(() => ({ getStatus: vi.fn() })) + +vi.mock('../git/status', async () => ({ + ...(await vi.importActual('../git/status')), + getStatus: mocks.getStatus +})) + +import { ExecutionHostNotDispatchableError } from '../providers/execution-host-provider-dispatch' +import { registerSshGitProvider, unregisterSshGitProvider } from '../providers/ssh-git-dispatch' +import { OrcaRuntimeService } from './orca-runtime' + +const REMOTE_PATH = '/srv/app-feature' +const WORKTREE_ID = 'repo-shared::/srv/app-feature' + +type RuntimeInternals = { + resolveWorktreeSelector: (selector: string) => Promise +} + +function makeRuntime(repos: readonly Record[], hostId?: string) { + const store = { + getSettings: () => ({ disabledTuiAgents: [], workspaceDir: '/tmp/workspaces' }), + getProjectHostSetups: () => [], + getProjects: () => [], + getRepos: () => repos, + getRepo: (id: string) => repos.find((repo) => repo.id === id) + } + const runtime = new OrcaRuntimeService(store as never) + vi.spyOn(runtime as unknown as RuntimeInternals, 'resolveWorktreeSelector').mockResolvedValue({ + id: WORKTREE_ID, + repoId: 'repo-shared', + path: REMOTE_PATH, + git: { path: REMOTE_PATH, branch: 'main', isBare: false, isMainWorktree: false }, + ...(hostId ? { hostId } : {}) + }) + return runtime +} + +function stubProvider() { + return { getStatus: vi.fn().mockResolvedValue({ entries: [] }) } +} + +describe('runtime Git target execution host', () => { + const registered: string[] = [] + + function register(connectionId: string) { + const provider = stubProvider() + registerSshGitProvider(connectionId, provider as never) + registered.push(connectionId) + return provider + } + + beforeEach(() => { + vi.restoreAllMocks() + mocks.getStatus.mockReset().mockResolvedValue({ entries: [] }) + }) + + afterEach(() => { + for (const connectionId of registered.splice(0)) { + unregisterSshGitProvider(connectionId) + } + }) + + // The case whose absence let the original cross-host leak through review: two SSH hosts, and the + // rival row is the one `getRepo` returns first. + it('serves an ssh worktree from the host it names, not from a rival row on another ssh host', async () => { + const openclaw = register('openclaw') + const m4air = register('m4air') + const runtime = makeRuntime( + [ + { id: 'repo-shared', path: '/home/me/app', connectionId: 'openclaw' }, + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' } + ], + 'ssh:m4air' + ) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).toHaveBeenCalledWith(REMOTE_PATH) + expect(openclaw.getStatus).not.toHaveBeenCalled() + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it("routes to the worktree's host even when the only repo row names a different ssh host", async () => { + const openclaw = register('openclaw') + const m4air = register('m4air') + const runtime = makeRuntime( + [{ id: 'repo-shared', path: '/home/me/app', connectionId: 'openclaw' }], + 'ssh:m4air' + ) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).toHaveBeenCalledWith(REMOTE_PATH) + expect(openclaw.getStatus).not.toHaveBeenCalled() + }) + + // `local` has no SSH namespace to nest in, so a surviving `connectionId` is a row contradicting + // itself. The old shape handed it out and dialled a remote host for a local workspace. + it('ignores a stale connection on a row that declares itself local', async () => { + const m4air = register('m4air') + const runtime = makeRuntime( + [ + { + id: 'repo-shared', + path: '/home/me/app', + executionHostId: 'local', + connectionId: 'm4air' + } + ], + 'local' + ) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).not.toHaveBeenCalled() + expect(mocks.getStatus).toHaveBeenCalled() + }) + + // A `runtime:` row's `connectionId` names a target in the *server's* namespace. Dialling it here + // reaches a same-named target on this client — a silent-wrong-host answer, worse than the + // silent-local one it replaced. + it('refuses a runtime host whose nested ssh target is also registered on this client', async () => { + const impostor = register('nested-1') + const runtime = makeRuntime( + [ + { + id: 'repo-shared', + path: '/srv/app', + executionHostId: 'runtime:env-a', + connectionId: 'nested-1' + } + ], + 'runtime:env-a' + ) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + ExecutionHostNotDispatchableError + ) + expect(impostor.getStatus).not.toHaveBeenCalled() + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it('refuses a runtime host with no nested ssh target rather than answering locally', async () => { + const runtime = makeRuntime( + [{ id: 'repo-shared', path: '/srv/app', executionHostId: 'runtime:env-a' }], + 'runtime:env-a' + ) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + ExecutionHostNotDispatchableError + ) + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it('refuses rather than guessing when rival rows disagree and the worktree names no host', async () => { + register('m4air') + const runtime = makeRuntime([ + { id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }, + { id: 'repo-shared', path: '/home/me/app' } + ]) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + 'worktree_execution_host_unresolved' + ) + expect(mocks.getStatus).not.toHaveBeenCalled() + }) + + it('still answers from the single row when the worktree names no host', async () => { + const m4air = register('m4air') + const runtime = makeRuntime([{ id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }]) + + await runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`) + + expect(m4air.getStatus).toHaveBeenCalledWith(REMOTE_PATH) + }) + + // Losing contact with a remote host is never evidence that its files are here + // (docs/reference/ssh-execution-boundary.md). + it('reports the dropped connection instead of reading a remote path locally', async () => { + const runtime = makeRuntime( + [{ id: 'repo-shared', path: '/srv/app', connectionId: 'm4air' }], + 'ssh:m4air' + ) + + await expect(runtime.getRuntimeGitStatus(`id:${WORKTREE_ID}`)).rejects.toThrow( + /Remote connection dropped/ + ) + expect(mocks.getStatus).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/runtime/worktree-launch-host-repo.test.ts b/src/main/runtime/worktree-launch-host-repo.test.ts index 8a81f2424d1..9525e74b0be 100644 --- a/src/main/runtime/worktree-launch-host-repo.test.ts +++ b/src/main/runtime/worktree-launch-host-repo.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { resolveWorktreeLaunchHost } from './worktree-launch-host-repo' +import { resolveWorktreeHostRouting, resolveWorktreeLaunchHost } from './worktree-launch-host-repo' // Why (#11163): the terminal launch scope read // `store.getRepo(worktree.repoId)?.connectionId ?? null` — one spelling of one arbitrarily chosen @@ -92,3 +92,56 @@ describe('resolveWorktreeLaunchHost', () => { }) }) }) + +// The same resolution answering "which host is this on" rather than "what may this client dial". +// The runtime Git target needs the first question, because `local` and `runtime:` are two different +// non-SSH answers and only one of them may run here. +describe('resolveWorktreeHostRouting', () => { + const runtimeRow = { + id: 'r', + path: '/p', + connectionId: 'ssh-nested', + executionHostId: 'runtime:env-a' as const + } + + it('keeps `runtime:` distinct from `local` where the launch answer collapses them', () => { + expect( + resolveWorktreeHostRouting([runtimeRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', hostId: 'runtime:env-a', repo: runtimeRow }) + // Both answer "no connection this client may dial"; only the routing view says which host. + expect( + resolveWorktreeLaunchHost([runtimeRow], { repoId: 'r', hostId: 'runtime:env-a' }) + ).toEqual({ kind: 'resolved', repo: runtimeRow, connectionId: null }) + }) + + it('answers the host the worktree names over a rival row on another ssh host', () => { + const rows = [ + { id: 'r', path: '/p', connectionId: 'openclaw' }, + { id: 'r', path: '/q', connectionId: 'm4air' } + ] + expect(resolveWorktreeHostRouting(rows, { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + repo: rows[1] + }) + // A row on some other host is not evidence about this one, so it contributes no metadata either. + expect(resolveWorktreeHostRouting([rows[0]], { repoId: 'r', hostId: 'ssh:m4air' })).toEqual({ + kind: 'resolved', + hostId: 'ssh:m4air', + repo: null + }) + }) + + it('separates "nobody carries this id" from "rival rows disagree"', () => { + expect(resolveWorktreeHostRouting([], { repoId: 'r' })).toEqual({ kind: 'unowned' }) + expect( + resolveWorktreeHostRouting( + [ + { id: 'r', path: '/p' }, + { id: 'r', path: '/q', connectionId: 'm4air' } + ], + { repoId: 'r' } + ) + ).toEqual({ kind: 'ambiguous' }) + }) +}) diff --git a/src/main/runtime/worktree-launch-host-repo.ts b/src/main/runtime/worktree-launch-host-repo.ts index fa2641655b3..4decb7acb56 100644 --- a/src/main/runtime/worktree-launch-host-repo.ts +++ b/src/main/runtime/worktree-launch-host-repo.ts @@ -3,7 +3,7 @@ import { resolveWorktreeExecutionHost, type ExecutionHostOwnerRow } from '../../shared/worktree-execution-host-resolution' -import { getSshTargetIdForExecutionHost } from '../../shared/execution-host' +import { getSshTargetIdForExecutionHost, type ExecutionHostId } from '../../shared/execution-host' import type { Repo } from '../../shared/repo-types' export type LaunchHostRepo = Pick @@ -12,32 +12,53 @@ export type WorktreeLaunchHostResolution = | { kind: 'resolved'; repo: T | null; connectionId: string | null } | { kind: 'ambiguous' } +export type WorktreeHostRouting = + /** `repo` is metadata; the host is the routing answer. */ + | { kind: 'resolved'; hostId: ExecutionHostId; repo: T | null } + /** No row carries this repo id and the worktree names no host — nothing ever named a host. */ + | { kind: 'unowned' } + /** Rival rows disagree about the host; guessing one is the cross-host leak. */ + | { kind: 'ambiguous' } + /** * Main-side adapter over the shared execution-host rule * (`src/shared/worktree-execution-host-resolution.ts`), which the renderer's owner index answers - * with too. Two things are local to this side: - * - * - rival rows that disagree about the host are `ambiguous` and the launch scope throws, while an - * id nobody carries stays "no repo, no connection" — the launch path's long-standing behaviour - * for a worktree whose repo row has gone; - * - the connection comes off the *host*, not the resolved row. This is a client-dialable PTY - * route, so a `runtime:` host contributes nothing: its nested SSH target belongs to that - * machine's namespace and spawning against it here would dial the wrong box. The renderer wants - * the opposite answer from the same resolution, which is why the shared type carries both. + * with too. What is local to this side is the disposal of the two `unresolved` reasons: rival rows + * that disagree about the host are `ambiguous` and callers throw, while an id nobody carries is + * `unowned` — the launch path's long-standing behaviour for a worktree whose repo row has gone. + */ +export function resolveWorktreeHostRouting( + repos: readonly T[], + worktree: { repoId: string; hostId?: string | null } +): WorktreeHostRouting { + const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) + if (resolution.kind === 'unresolved') { + return resolution.reason === 'ambiguous' ? { kind: 'ambiguous' } : { kind: 'unowned' } + } + return { kind: 'resolved', hostId: resolution.hostId, repo: resolution.owner } +} + +/** + * The same resolution, answering "what may this client dial" rather than "which host is this on". + * The connection comes off the *host*, not the resolved row: this is a client-dialable PTY route, + * so a `runtime:` host contributes nothing — its nested SSH target belongs to that machine's + * namespace and spawning against it here would dial the wrong box. The renderer wants the opposite + * answer from the same resolution, which is why the shared type carries both. */ export function resolveWorktreeLaunchHost( repos: readonly T[], worktree: { repoId: string; hostId?: string | null } ): WorktreeLaunchHostResolution { - const resolution = resolveWorktreeExecutionHost(createRepoRowExecutionHostLookup(repos), worktree) - if (resolution.kind === 'unresolved') { - return resolution.reason === 'ambiguous' - ? { kind: 'ambiguous' } - : { kind: 'resolved', repo: null, connectionId: null } + const routing = resolveWorktreeHostRouting(repos, worktree) + if (routing.kind === 'ambiguous') { + return { kind: 'ambiguous' } + } + if (routing.kind === 'unowned') { + return { kind: 'resolved', repo: null, connectionId: null } } return { kind: 'resolved', - repo: resolution.owner, - connectionId: getSshTargetIdForExecutionHost(resolution.hostId) + repo: routing.repo, + connectionId: getSshTargetIdForExecutionHost(routing.hostId) } } From 40d9927f014199ac7a96b669df14c2246a445814 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 19:52:19 -0700 Subject: [PATCH 134/398] fix(native-chat): show images in Codex structured chat (#18266) * fix(native-chat): render structured image refs from their runtime owner * fix(native-chat): keep transcript image keys stable * fix(native-chat): memoize the image runtime owner * fix(native-chat): keep image preview observation scoped * fix(native-chat): resolve runtime-only image owners * fix(native-chat): retain image preview cache leases --------- Co-authored-by: Merge Sim --- .../editor/local-image-src-cache.ts | 231 ++++++++++++++ .../editor/local-image-src-reader.ts | 23 ++ .../editor/rich-markdown-extensions.ts | 11 +- .../editor/rich-markdown-local-image.test.ts | 26 +- .../editor/useLocalImageSrc.test.ts | 116 +++++++ .../src/components/editor/useLocalImageSrc.ts | 291 ++++++------------ .../native-chat/NativeChatMessageList.tsx | 52 ++-- .../NativeChatStructuredSession.tsx | 3 + .../NativeChatTranscriptChrome.test.tsx | 209 +++++++++++++ .../NativeChatTranscriptChrome.tsx | 235 +++++++++++++- .../NativeChatTypingIndicatorRow.tsx | 21 ++ .../native-chat/native-chat-file-link.test.ts | 22 ++ .../native-chat/native-chat-file-link.ts | 13 +- .../native-chat-image-runtime-context.test.ts | 96 ++++++ .../native-chat-image-runtime-context.ts | 184 +++++++++++ 15 files changed, 1290 insertions(+), 243 deletions(-) create mode 100644 src/renderer/src/components/editor/local-image-src-cache.ts create mode 100644 src/renderer/src/components/editor/local-image-src-reader.ts create mode 100644 src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx create mode 100644 src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx create mode 100644 src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts create mode 100644 src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts diff --git a/src/renderer/src/components/editor/local-image-src-cache.ts b/src/renderer/src/components/editor/local-image-src-cache.ts new file mode 100644 index 00000000000..410ce1c4502 --- /dev/null +++ b/src/renderer/src/components/editor/local-image-src-cache.ts @@ -0,0 +1,231 @@ +import { + clearLocalImageCachePins, + isLocalImageCacheKeyPinned, + pinLocalImageCacheKey, + prunePinnedLocalImageCache, + unpinLocalImageCacheKey +} from './local-image-cache-pinning' + +const BLOB_URL_CACHE_MAX_SIZE = 100 +// Keep retained decoded image data bounded as well as entry count. +const BLOB_URL_CACHE_MAX_BYTES = 128 * 1024 * 1024 + +export const blobUrlCache = new Map() +const blobUrlCacheBytes = new Map() +export const inFlightBlobUrlLoads = new Map>() +// Incremented on release so a read resolving after its last consumer left +// cannot repopulate the cache. +const cacheKeyVersions = new Map() + +export function getLocalImageCacheKeyVersion(key: string): number { + return cacheKeyVersions.get(key) ?? 0 +} + +export function cleanupLocalImageCacheKeyVersion(key: string): void { + if ( + !blobUrlCache.has(key) && + !inFlightBlobUrlLoads.has(key) && + !isLocalImageCacheKeyPinned(key) + ) { + cacheKeyVersions.delete(key) + } +} + +let cacheGeneration = 0 +const cacheListeners = new Set<() => void>() +const pendingBlobUrlRevocations = new Set() +let pendingBlobUrlRevocationTimer: ReturnType | null = null + +function pruneImageCache(): void { + prunePinnedLocalImageCache(blobUrlCache, BLOB_URL_CACHE_MAX_SIZE, (url) => { + URL.revokeObjectURL(url) + }) + for (const key of blobUrlCacheBytes.keys()) { + if (!blobUrlCache.has(key)) { + blobUrlCacheBytes.delete(key) + cleanupLocalImageCacheKeyVersion(key) + } + } + let retainedBytes = 0 + for (const byteLength of blobUrlCacheBytes.values()) { + retainedBytes += byteLength + } + while (retainedBytes > BLOB_URL_CACHE_MAX_BYTES) { + const key = Array.from(blobUrlCache.keys()).find( + (candidate) => !isLocalImageCacheKeyPinned(candidate) + ) + if (key === undefined) { + return + } + const url = blobUrlCache.get(key) + blobUrlCache.delete(key) + retainedBytes -= blobUrlCacheBytes.get(key) ?? 0 + blobUrlCacheBytes.delete(key) + if (url) { + URL.revokeObjectURL(url) + } + cleanupLocalImageCacheKeyVersion(key) + } +} + +export function cacheLocalImageBlob( + key: string, + url: string, + byteLength: number, + expectedVersion?: number +): boolean { + if ( + (expectedVersion !== undefined && getLocalImageCacheKeyVersion(key) !== expectedVersion) || + byteLength > BLOB_URL_CACHE_MAX_BYTES + ) { + URL.revokeObjectURL(url) + cleanupLocalImageCacheKeyVersion(key) + return false + } + const previousUrl = blobUrlCache.get(key) + const previousBytes = blobUrlCacheBytes.get(key) ?? 0 + let retainedBytes = 0 + for (const bytes of blobUrlCacheBytes.values()) { + retainedBytes += bytes + } + let projectedEntries = blobUrlCache.size + (previousUrl === undefined ? 1 : 0) + let projectedBytes = retainedBytes - previousBytes + byteLength + + // Evict only unpinned entries. If visible leases consume the budget, fail + // closed so a late/non-visible decode cannot make retention unbounded. + while (projectedEntries > BLOB_URL_CACHE_MAX_SIZE || projectedBytes > BLOB_URL_CACHE_MAX_BYTES) { + const candidate = Array.from(blobUrlCache.keys()).find( + (candidateKey) => candidateKey !== key && !isLocalImageCacheKeyPinned(candidateKey) + ) + if (candidate === undefined) { + URL.revokeObjectURL(url) + cleanupLocalImageCacheKeyVersion(key) + return false + } + const candidateUrl = blobUrlCache.get(candidate) + const candidateBytes = blobUrlCacheBytes.get(candidate) ?? 0 + blobUrlCache.delete(candidate) + blobUrlCacheBytes.delete(candidate) + projectedEntries -= 1 + projectedBytes -= candidateBytes + if (candidateUrl) { + URL.revokeObjectURL(candidateUrl) + } + cleanupLocalImageCacheKeyVersion(candidate) + } + if (previousUrl !== undefined && previousUrl !== url) { + URL.revokeObjectURL(previousUrl) + } + blobUrlCacheBytes.delete(key) + blobUrlCacheBytes.set(key, byteLength) + blobUrlCache.set(key, url) + return true +} + +export function getLocalImageCacheGeneration(): number { + return cacheGeneration +} + +export function pinLocalImageCache(key: string): void { + pinLocalImageCacheKey(key) +} + +export function unpinLocalImageCache(key: string): void { + unpinLocalImageCacheKey(key) + pruneImageCache() + cleanupLocalImageCacheKeyVersion(key) +} + +export function subscribeToLocalImageCacheInvalidation(listener: () => void): () => void { + cacheListeners.add(listener) + return () => cacheListeners.delete(listener) +} + +function revokePendingBlobUrls(): void { + pendingBlobUrlRevocationTimer = null + for (const url of pendingBlobUrlRevocations) { + URL.revokeObjectURL(url) + } + pendingBlobUrlRevocations.clear() +} + +function scheduleBlobUrlRevocation(urls: string[]): void { + for (const url of urls) { + pendingBlobUrlRevocations.add(url) + } + if (pendingBlobUrlRevocationTimer !== null || pendingBlobUrlRevocations.size === 0) { + return + } + pendingBlobUrlRevocationTimer = setTimeout(revokePendingBlobUrls, 30_000) +} + +export function invalidateLocalImageCache(): void { + const staleUrls = Array.from(blobUrlCache.values()) + blobUrlCache.clear() + blobUrlCacheBytes.clear() + inFlightBlobUrlLoads.clear() + cacheKeyVersions.clear() + cacheGeneration += 1 + for (const listener of cacheListeners) { + listener() + } + if (staleUrls.length > 0) { + scheduleBlobUrlRevocation(staleUrls) + } +} + +export function releaseLocalImageBlob(key: string): void { + if (isLocalImageCacheKeyPinned(key)) { + return + } + cacheKeyVersions.set(key, getLocalImageCacheKeyVersion(key) + 1) + const inFlight = inFlightBlobUrlLoads.get(key) + if (inFlight) { + // A released lease must not be reused by a later visible lease: the + // released read may resolve null or be stale for the next owner. + inFlightBlobUrlLoads.delete(key) + } + const url = blobUrlCache.get(key) + if (url) { + blobUrlCache.delete(key) + blobUrlCacheBytes.delete(key) + URL.revokeObjectURL(url) + } + if (!inFlight) { + cleanupLocalImageCacheKeyVersion(key) + } +} + +export function resetLocalImageCacheState(): void { + if (pendingBlobUrlRevocationTimer !== null) { + clearTimeout(pendingBlobUrlRevocationTimer) + pendingBlobUrlRevocationTimer = null + } + revokePendingBlobUrls() + for (const url of blobUrlCache.values()) { + URL.revokeObjectURL(url) + } + blobUrlCache.clear() + blobUrlCacheBytes.clear() + clearLocalImageCachePins() + inFlightBlobUrlLoads.clear() + cacheKeyVersions.clear() + cacheGeneration = 0 + pendingBlobUrlRevocations.clear() + cacheListeners.clear() +} + +export function disposeLocalImageCacheState(): void { + if (typeof window !== 'undefined') { + window.removeEventListener('focus', invalidateLocalImageCache) + } + resetLocalImageCacheState() +} + +if (typeof window !== 'undefined') { + window.addEventListener('focus', invalidateLocalImageCache) +} + +if (import.meta !== undefined && import.meta.hot) { + import.meta.hot.dispose(disposeLocalImageCacheState) +} diff --git a/src/renderer/src/components/editor/local-image-src-reader.ts b/src/renderer/src/components/editor/local-image-src-reader.ts new file mode 100644 index 00000000000..5e845031bd0 --- /dev/null +++ b/src/renderer/src/components/editor/local-image-src-reader.ts @@ -0,0 +1,23 @@ +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { readRuntimeFilePreview } from '@/runtime/runtime-file-client' + +export function readLocalImagePreview( + absolutePath: string, + connectionId?: string | null, + runtimeContext?: Omit & { connectionId?: string | null } +) { + try { + if (!runtimeContext) { + return window.api.fs.readFile({ + filePath: absolutePath, + connectionId: connectionId ?? undefined + }) + } + return readRuntimeFilePreview( + { ...runtimeContext, connectionId: runtimeContext.connectionId ?? connectionId ?? undefined }, + absolutePath + ) + } catch (error) { + return Promise.reject(error) + } +} diff --git a/src/renderer/src/components/editor/rich-markdown-extensions.ts b/src/renderer/src/components/editor/rich-markdown-extensions.ts index 9876f2871cc..42904291a23 100644 --- a/src/renderer/src/components/editor/rich-markdown-extensions.ts +++ b/src/renderer/src/components/editor/rich-markdown-extensions.ts @@ -12,7 +12,11 @@ import { TableRow } from '@tiptap/extension-table-row' import { BlockMath, InlineMath } from '@tiptap/extension-mathematics' import { Markdown } from '@tiptap/markdown' import { createLowlight, common } from 'lowlight' -import { loadLocalImageSrc, onImageCacheInvalidated } from './useLocalImageSrc' +import { + acquireLocalImageSrcLease, + loadLocalImageSrc, + onImageCacheInvalidated +} from './useLocalImageSrc' import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' import { createRawMarkdownHtmlBlock, @@ -126,14 +130,18 @@ export function createRichMarkdownExtensions({ let currentSrc = node.attrs.src as string | undefined let currentContextVersion = getImageContextVersion(this.storage) + let releaseImageLease: (() => void) | undefined const loadImage = (src: string | undefined): void => { + releaseImageLease?.() + releaseImageLease = undefined const fp = this.storage.filePath as string const runtimeContext = this.storage.runtimeContext as | RuntimeFileOperationArgs | undefined const contextVersionAtLoad = getImageContextVersion(this.storage) if (src && fp) { + releaseImageLease = acquireLocalImageSrcLease(src, fp, undefined, runtimeContext) void loadLocalImageSrc(src, fp, undefined, runtimeContext).then((resolved) => { if (currentSrc !== src || currentContextVersion !== contextVersionAtLoad) { return @@ -187,6 +195,7 @@ export function createRichMarkdownExtensions({ return true }, destroy: () => { + releaseImageLease?.() if (reloadListeners instanceof Set) { reloadListeners.delete(reloadForContextChange) } diff --git a/src/renderer/src/components/editor/rich-markdown-local-image.test.ts b/src/renderer/src/components/editor/rich-markdown-local-image.test.ts index e343e918619..bcd4e64c6f1 100644 --- a/src/renderer/src/components/editor/rich-markdown-local-image.test.ts +++ b/src/renderer/src/components/editor/rich-markdown-local-image.test.ts @@ -4,7 +4,7 @@ import { Editor } from '@tiptap/core' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { createRichMarkdownExtensions } from './rich-markdown-extensions' import { createRichMarkdownEditorCodec } from './rich-markdown-source-transport' -import { resetLocalImageSrcStateForTests } from './useLocalImageSrc' +import { releaseLocalImageSrc, resetLocalImageSrcStateForTests } from './useLocalImageSrc' import { setRichMarkdownImageResolverContext } from './rich-markdown-image-context' async function flushPromises(): Promise { @@ -18,6 +18,7 @@ describe('rich markdown local images', () => { beforeEach(() => { resetLocalImageSrcStateForTests() vi.spyOn(URL, 'createObjectURL').mockReturnValue('blob:rich-local-image') + vi.spyOn(URL, 'revokeObjectURL').mockImplementation(() => undefined) globalThis.window.api = { ...globalThis.window.api, fs: { @@ -64,4 +65,27 @@ describe('rich markdown local images', () => { editor.destroy() } }) + + it('keeps a displayed image leased when another surface releases the same cache entry', async () => { + const host = document.createElement('div') + document.body.appendChild(host) + const editor = new Editor({ + element: host, + extensions: createRichMarkdownExtensions({ codec: createRichMarkdownEditorCodec() }), + content: '![](diagram.png)', + contentType: 'markdown' + }) + + try { + setRichMarkdownImageResolverContext(editor, { filePath: '/repo/docs/readme.md' }) + await flushPromises() + + releaseLocalImageSrc('diagram.png', '/repo/docs/readme.md') + + expect(URL.revokeObjectURL).not.toHaveBeenCalledWith('blob:rich-local-image') + expect(host.querySelector('img')?.src).toBe('blob:rich-local-image') + } finally { + editor.destroy() + } + }) }) diff --git a/src/renderer/src/components/editor/useLocalImageSrc.test.ts b/src/renderer/src/components/editor/useLocalImageSrc.test.ts index eaf5b031928..1c93693ff1c 100644 --- a/src/renderer/src/components/editor/useLocalImageSrc.test.ts +++ b/src/renderer/src/components/editor/useLocalImageSrc.test.ts @@ -7,9 +7,16 @@ import { getLocalImageCacheKey, invalidateLocalImageSrcCacheForTests, loadLocalImageSrc, + releaseLocalImageSrc, resetLocalImageSrcStateForTests, useLocalImageSrc } from './useLocalImageSrc' +import { + blobUrlCache, + cacheLocalImageBlob, + getLocalImageCacheKeyVersion, + pinLocalImageCache +} from './local-image-src-cache' type PreviewResult = { content: string @@ -128,6 +135,37 @@ describe('loadLocalImageSrc', () => { expect(URL.createObjectURL).toHaveBeenCalledTimes(1) }) + it('lets a mounted preview adopt an in-flight prewarm read', async () => { + const read = deferred() + const readFile = vi.fn().mockReturnValue(read.promise) + const renders: (string | undefined)[] = [] + vi.spyOn(URL, 'createObjectURL').mockReturnValue('blob:prewarmed') + setReadFile(readFile) + + const prewarm = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + const container = document.createElement('div') + const root: Root = createRoot(container) + await act(async () => { + root.render( + createElement(HookProbe, { + filePath: '/repo/docs/readme.md', + onRender: (displaySrc) => renders.push(displaySrc), + src: 'diagram.png' + }) + ) + }) + expect(readFile).toHaveBeenCalledTimes(1) + + await act(async () => { + read.resolve(binaryPreview()) + await flushPromises() + }) + + await expect(prewarm).resolves.toBe('blob:prewarmed') + expect(renders.at(-1)).toBe('blob:prewarmed') + root.unmount() + }) + it('does not revoke blob URLs still used by mounted previews during eviction', async () => { const readFile = vi.fn().mockResolvedValue(binaryPreview()) let nextUrl = 0 @@ -230,6 +268,84 @@ describe('loadLocalImageSrc', () => { expect(URL.revokeObjectURL).not.toHaveBeenCalledWith('blob:newer') }) + it('does not retain a read that resolves after its preview lease is released', async () => { + const read = deferred() + const readFile = vi.fn().mockReturnValue(read.promise) + vi.spyOn(URL, 'createObjectURL').mockReturnValue('blob:released') + setReadFile(readFile) + + const pending = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + releaseLocalImageSrc('diagram.png', '/repo/docs/readme.md') + read.resolve(binaryPreview()) + + await expect(pending).resolves.toBeNull() + expect(blobUrlCache.size).toBe(0) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:released') + }) + + it('starts a fresh read when a released lease becomes visible again', async () => { + const firstRead = deferred() + const secondRead = deferred() + const readFile = vi + .fn() + .mockReturnValueOnce(firstRead.promise) + .mockReturnValueOnce(secondRead.promise) + vi.spyOn(URL, 'createObjectURL') + .mockReturnValueOnce('blob:fresh') + .mockReturnValueOnce('blob:stale') + setReadFile(readFile) + + const stale = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + releaseLocalImageSrc('diagram.png', '/repo/docs/readme.md') + const fresh = loadLocalImageSrc('diagram.png', '/repo/docs/readme.md') + expect(readFile).toHaveBeenCalledTimes(2) + + secondRead.resolve(binaryPreview('AQ==')) + await expect(fresh).resolves.toBe('blob:fresh') + firstRead.resolve(binaryPreview()) + await expect(stale).resolves.toBeNull() + expect(blobUrlCache.size).toBe(1) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:stale') + }) + + it('cleans version metadata for released unique paths', () => { + for (let index = 0; index < 500; index += 1) { + const path = `/repo/docs/image-${index}.png` + releaseLocalImageSrc(path, '/repo/docs/readme.md') + expect(getLocalImageCacheKeyVersion(getLocalImageCacheKey(path, undefined, undefined))).toBe( + 0 + ) + } + }) + + it('fails closed when pinned previews already consume the entry or byte budget', () => { + for (let index = 0; index < 100; index += 1) { + const key = `pinned-${index}` + pinLocalImageCache(key) + expect(cacheLocalImageBlob(key, `blob:${index}`, 1)).toBe(true) + } + + expect(cacheLocalImageBlob('pinned-overflow', 'blob:overflow', 1)).toBe(false) + expect(blobUrlCache.size).toBe(100) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:overflow') + + expect( + cacheLocalImageBlob('large-overflow', 'blob:large-overflow', 128 * 1024 * 1024 + 1) + ).toBe(false) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:large-overflow') + }) + + it('does not exceed the decoded-byte budget when all retained entries are pinned', () => { + const retainedBytes = 80 * 1024 * 1024 + pinLocalImageCache('large-pinned-1') + pinLocalImageCache('large-pinned-2') + expect(cacheLocalImageBlob('large-pinned-1', 'blob:large-1', retainedBytes)).toBe(true) + expect(cacheLocalImageBlob('large-pinned-2', 'blob:large-2', retainedBytes)).toBe(false) + + expect(blobUrlCache.size).toBe(1) + expect(URL.revokeObjectURL).toHaveBeenCalledWith('blob:large-2') + }) + it('keeps runtime owners in separate image cache entries', async () => { const readFile = vi.fn().mockResolvedValue(binaryPreview()) vi.spyOn(URL, 'createObjectURL') diff --git a/src/renderer/src/components/editor/useLocalImageSrc.ts b/src/renderer/src/components/editor/useLocalImageSrc.ts index bd4cccd9e16..38339ee36ea 100644 --- a/src/renderer/src/components/editor/useLocalImageSrc.ts +++ b/src/renderer/src/components/editor/useLocalImageSrc.ts @@ -1,22 +1,21 @@ import { useEffect, useState } from 'react' import { resolveImageAbsolutePath } from './markdown-preview-links' import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' -import { readRuntimeFilePreview } from '@/runtime/runtime-file-client' +import { readLocalImagePreview } from './local-image-src-reader' import { - clearLocalImageCachePins, - pinLocalImageCacheKey, - prunePinnedLocalImageCache, - unpinLocalImageCacheKey -} from './local-image-cache-pinning' - -// Why: the renderer is served from http://localhost in dev mode, so file:// -// URLs in tags are blocked by cross-origin restrictions. Loading images -// via the existing fs.readFile IPC and converting to blob URLs bypasses this -// limitation and works identically in both dev and production modes. - -const BLOB_URL_CACHE_MAX_SIZE = 100 -const blobUrlCache = new Map() -const inFlightBlobUrlLoads = new Map>() + blobUrlCache, + cacheLocalImageBlob, + cleanupLocalImageCacheKeyVersion, + getLocalImageCacheGeneration, + getLocalImageCacheKeyVersion, + inFlightBlobUrlLoads, + invalidateLocalImageCache, + pinLocalImageCache, + releaseLocalImageBlob, + resetLocalImageCacheState, + subscribeToLocalImageCacheInvalidation, + unpinLocalImageCache +} from './local-image-src-cache' export function getLocalImageCacheKey( absolutePath: string, @@ -28,131 +27,32 @@ export function getLocalImageCacheKey( return [ runtimeEnvironmentId, runtimeContext?.connectionId ?? connectionId ?? 'local', + runtimeContext?.expectedExecutionHostId ?? 'unknown-host', + runtimeContext?.expectedSshTargetId ?? '', + runtimeContext?.expectedSshConnectionGeneration?.toString() ?? '', runtimeContext?.expectedExternalSshTargetId ?? '', runtimeContext?.worktreeId ?? 'unknown-worktree', + runtimeContext?.worktreePath ?? '', absolutePath ].join('\0') } -// Why: blob URLs hold references to in-memory Blob objects; without eviction -// the cache grows without bound and leaks memory. We evict the oldest entry -// (Map iteration order is insertion order) and revoke its blob URL so the -// browser can free the underlying data. -function cacheBlobUrl(key: string, url: string): void { - const previousUrl = blobUrlCache.get(key) - if (previousUrl !== undefined) { - blobUrlCache.delete(key) - if (previousUrl !== url) { - // Why: cache replacements must release the superseded Blob even when - // they come from rare stale state or future loader changes. - URL.revokeObjectURL(previousUrl) - } - } - blobUrlCache.set(key, url) - prunePinnedLocalImageCache(blobUrlCache, BLOB_URL_CACHE_MAX_SIZE, URL.revokeObjectURL) -} - -const cacheListeners = new Set<() => void>() -let cacheGeneration = 0 -const pendingBlobUrlRevocations = new Set() -let pendingBlobUrlRevocationTimer: ReturnType | null = null - -function base64ToBlobUrl(base64: string, mimeType: string): string { +function base64ToBlobUrl(base64: string, mimeType: string): { url: string; byteLength: number } { const binary = atob(base64.replace(/\s/g, '')) const bytes = new Uint8Array(binary.length) for (let i = 0; i < binary.length; i += 1) { bytes[i] = binary.charCodeAt(i) } - return URL.createObjectURL(new Blob([bytes], { type: mimeType })) -} - -function revokePendingBlobUrls(): void { - pendingBlobUrlRevocationTimer = null - for (const url of pendingBlobUrlRevocations) { - URL.revokeObjectURL(url) - } - pendingBlobUrlRevocations.clear() -} - -function scheduleBlobUrlRevocation(urls: string[]): void { - for (const url of urls) { - pendingBlobUrlRevocations.add(url) - } - if (pendingBlobUrlRevocationTimer !== null || pendingBlobUrlRevocations.size === 0) { - return - } - pendingBlobUrlRevocationTimer = setTimeout(revokePendingBlobUrls, 30_000) -} - -// Why: when the user switches back to the app after deleting or replacing -// image files externally, clearing the cache forces the preview to pick up -// the current filesystem state instead of showing stale in-memory blob URLs. -// Old blob URLs are revoked after a short delay so that elements still -// display the old data while the fresh IPC load completes, avoiding a visible -// flash. The 30-second window is generous enough for even slow IPC reads. -function invalidateImageCache(): void { - const staleUrls = Array.from(blobUrlCache.values()) - blobUrlCache.clear() - inFlightBlobUrlLoads.clear() - cacheGeneration += 1 - for (const listener of cacheListeners) { - listener() - } - // Why: defer revocation so the browser keeps the old blob data readable - // until replacement IPC loads complete, then free the underlying memory. - // 30 seconds is generous enough to cover slow machines or large images - // without risking a visible broken-image flash. - if (staleUrls.length > 0) { - scheduleBlobUrlRevocation(staleUrls) + return { + url: URL.createObjectURL(new Blob([bytes], { type: mimeType })), + byteLength: bytes.byteLength } } -function disposeImageCacheModuleState(): void { - if (typeof window !== 'undefined') { - window.removeEventListener('focus', invalidateImageCache) - } - if (pendingBlobUrlRevocationTimer !== null) { - clearTimeout(pendingBlobUrlRevocationTimer) - pendingBlobUrlRevocationTimer = null - } - revokePendingBlobUrls() - for (const url of blobUrlCache.values()) { - URL.revokeObjectURL(url) - } - blobUrlCache.clear() - clearLocalImageCachePins() - inFlightBlobUrlLoads.clear() - cacheListeners.clear() -} - -if (typeof window !== 'undefined') { - window.addEventListener('focus', invalidateImageCache) -} - -if (import.meta !== undefined && import.meta.hot) { - // Why: Vite can re-evaluate this module without a full renderer reload. - // Disposing the module-level listener and blob URLs prevents dev-session leaks. - import.meta.hot.dispose(disposeImageCacheModuleState) -} - -/** - * Subscribe to cache invalidation events (fired on window re-focus). - * Returns an unsubscribe function. - */ -export function onImageCacheInvalidated(listener: () => void): () => void { - cacheListeners.add(listener) - return () => { - cacheListeners.delete(listener) - } -} +export const onImageCacheInvalidated = subscribeToLocalImageCacheInvalidation function isExternalUrl(src: string): boolean { - return ( - src.startsWith('http://') || - src.startsWith('https://') || - src.startsWith('data:') || - src.startsWith('blob:') - ) + return /^(?:https?|data|blob):/i.test(src) } /** @@ -165,32 +65,22 @@ export function useLocalImageSrc( rawSrc: string | undefined, filePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null ): string | undefined { - const [generation, setGeneration] = useState(cacheGeneration) + const [generation, setGeneration] = useState(getLocalImageCacheGeneration()) useEffect(() => { - if (!rawSrc || isExternalUrl(rawSrc)) { - return - } - const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) - if (!absolutePath) { - return - } - const cacheKey = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) - pinLocalImageCacheKey(cacheKey) - return () => { - unpinLocalImageCacheKey(cacheKey) - prunePinnedLocalImageCache(blobUrlCache, BLOB_URL_CACHE_MAX_SIZE, URL.revokeObjectURL) - } + return acquireLocalImageSrcLease(rawSrc, filePath, connectionId, runtimeContext) }, [rawSrc, filePath, connectionId, runtimeContext]) useEffect(() => { - return onImageCacheInvalidated(() => setGeneration(cacheGeneration)) + return onImageCacheInvalidated(() => setGeneration(getLocalImageCacheGeneration())) }, []) const [displaySrc, setDisplaySrc] = useState(() => { - if (!rawSrc) { + if (!rawSrc || runtimeContext === null) { return undefined } if (isExternalUrl(rawSrc)) { @@ -207,7 +97,7 @@ export function useLocalImageSrc( }) useEffect(() => { - if (!rawSrc) { + if (!rawSrc || runtimeContext === null) { setDisplaySrc(undefined) return } @@ -236,7 +126,7 @@ export function useLocalImageSrc( if (cancelled) { return } - setDisplaySrc(cacheGeneration === effectGeneration && url ? url : undefined) + setDisplaySrc(getLocalImageCacheGeneration() === effectGeneration && url ? url : undefined) }) .catch(() => { if (!cancelled) { @@ -261,16 +151,16 @@ export async function loadLocalImageSrc( rawSrc: string, filePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null ): Promise { - if ( - rawSrc.startsWith('http://') || - rawSrc.startsWith('https://') || - rawSrc.startsWith('data:') || - rawSrc.startsWith('blob:') - ) { + if (isExternalUrl(rawSrc)) { return rawSrc } + if (runtimeContext === null) { + return null + } const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) if (!absolutePath) { @@ -289,8 +179,13 @@ export async function loadLocalImageSrc( export function loadLocalImageAbsolutePath( absolutePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null ): Promise { + if (runtimeContext === null) { + return Promise.resolve(null) + } const cacheKey = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) const cached = blobUrlCache.get(cacheKey) if (cached) { @@ -302,73 +197,79 @@ export function loadLocalImageAbsolutePath( return inFlight } - const readGeneration = cacheGeneration - const loadPromise = readImagePreview(absolutePath, connectionId, runtimeContext) + const readGeneration = getLocalImageCacheGeneration() + const readLeaseVersion = getLocalImageCacheKeyVersion(cacheKey) + const loadPromise = readLocalImagePreview(absolutePath, connectionId, runtimeContext) .then((result) => { - if (!result.isBinary || !result.content || cacheGeneration !== readGeneration) { - // Why: local image paths must stay behind IPC/runtime authorization; - // handing raw file: or relative paths back to Chromium can escape it. + if ( + !result.isBinary || + !result.content || + getLocalImageCacheGeneration() !== readGeneration + ) { return null } - const url = base64ToBlobUrl(result.content, result.mimeType ?? 'image/png') - if (cacheGeneration !== readGeneration) { + const { url, byteLength } = base64ToBlobUrl(result.content, result.mimeType ?? 'image/png') + if (getLocalImageCacheGeneration() !== readGeneration) { URL.revokeObjectURL(url) return null } - cacheBlobUrl(cacheKey, url) - return url + return cacheLocalImageBlob(cacheKey, url, byteLength, readLeaseVersion) ? url : null }) .catch(() => null) .finally(() => { if (inFlightBlobUrlLoads.get(cacheKey) === loadPromise) { inFlightBlobUrlLoads.delete(cacheKey) } + cleanupLocalImageCacheKeyVersion(cacheKey) }) inFlightBlobUrlLoads.set(cacheKey, loadPromise) return loadPromise } export function resetLocalImageSrcStateForTests(): void { - if (pendingBlobUrlRevocationTimer !== null) { - clearTimeout(pendingBlobUrlRevocationTimer) - pendingBlobUrlRevocationTimer = null - } - revokePendingBlobUrls() - for (const url of blobUrlCache.values()) { - URL.revokeObjectURL(url) - } - blobUrlCache.clear() - clearLocalImageCachePins() - inFlightBlobUrlLoads.clear() - cacheGeneration = 0 - pendingBlobUrlRevocations.clear() - cacheListeners.clear() + resetLocalImageCacheState() } export function invalidateLocalImageSrcCacheForTests(): void { - invalidateImageCache() + invalidateLocalImageCache() } -function readImagePreview( - absolutePath: string, +export function acquireLocalImageSrcLease( + rawSrc: string | undefined, + filePath: string, connectionId?: string | null, - runtimeContext?: Omit & { connectionId?: string | null } -) { - try { - if (!runtimeContext) { - return window.api.fs.readFile({ - filePath: absolutePath, - connectionId: connectionId ?? undefined - }) - } - return readRuntimeFilePreview( - { - ...runtimeContext, - connectionId: runtimeContext.connectionId ?? connectionId ?? undefined - }, - absolutePath - ) - } catch (error) { - return Promise.reject(error) + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null +): (() => void) | undefined { + if (!rawSrc || isExternalUrl(rawSrc) || runtimeContext === null) { + return undefined } + const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) + if (!absolutePath) { + return undefined + } + const key = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) + pinLocalImageCache(key) + return () => unpinLocalImageCache(key) +} + +/** Evict one no-longer-visible transcript preview immediately. */ +export function releaseLocalImageSrc( + rawSrc: string, + filePath: string, + connectionId?: string | null, + runtimeContext?: + | (Omit & { connectionId?: string | null }) + | null +): void { + if (!rawSrc || isExternalUrl(rawSrc) || runtimeContext === null) { + return + } + const absolutePath = resolveImageAbsolutePath(rawSrc, filePath) + if (!absolutePath) { + return + } + const key = getLocalImageCacheKey(absolutePath, connectionId, runtimeContext) + releaseLocalImageBlob(key) } diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx index 988db65f730..12b10e0f714 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.tsx @@ -16,34 +16,16 @@ import { shouldShowNativeChatTypingIndicator } from './native-chat-typing-indica import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' import { useNativeChatTurnStatus } from './use-native-chat-turn-status' import { nativeChatProseToMarkdown } from './native-chat-prose' +import { NativeChatTypingIndicatorRow } from './NativeChatTypingIndicatorRow' import { NativeChatAgentControls, NativeChatImageAttachments, ProviderFrameRow } from './NativeChatTranscriptChrome' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' export { ProviderFrameRow } from './NativeChatTranscriptChrome' -function TypingIndicatorRow(): React.JSX.Element { - return ( -
    -
    - {[0, 1, 2].map((i) => ( - - ))} -
    -
    - ) -} - function geometryOf(el: HTMLElement): ScrollGeometry { return { scrollTop: el.scrollTop, scrollHeight: el.scrollHeight, clientHeight: el.clientHeight } } @@ -62,7 +44,8 @@ function MessageRow({ allowFileUriLinks = false, deliveryFailed = false, activityExpandOverride, - structuredActivityUi = true + structuredActivityUi = true, + runtimeContext }: { message: NativeChatMessage expandSignal: boolean @@ -74,6 +57,7 @@ function MessageRow({ deliveryFailed?: boolean activityExpandOverride?: boolean structuredActivityUi?: boolean + runtimeContext?: RuntimeFileOperationArgs | null }): React.JSX.Element | null { const rowRef = useRef(null) const { prose, tools } = useMemo(() => splitNativeChatBlocks(message.blocks), [message.blocks]) @@ -113,7 +97,11 @@ function MessageRow({
    {markdown ? ( <> - + ) : ( - + )}
    {deliveryFailed ? ( @@ -152,7 +144,11 @@ function MessageRow({ isSystem && 'text-xs text-muted-foreground' )} > - + {markdown ? ( /** Turn timing/disclosure is available only on the structured Codex lane. */ showTurnStatus?: boolean + runtimeContext?: RuntimeFileOperationArgs | null }): React.JSX.Element { const scrollRef = useRef(null) const contentRef = useRef(null) @@ -231,7 +229,6 @@ export function NativeChatMessageList({ const stuckToBottomRef = useRef(stuckToBottom) stuckToBottomRef.current = stuckToBottom - const { hasMore, loadingEarlier, loadEarlier } = session // Keep hidden harness turns as fold boundaries, then strip them before render. @@ -401,6 +398,7 @@ export function NativeChatMessageList({ deliveryFailed={failedDeliveryMessageIds?.has(message.id) === true} structuredActivityUi={showTurnStatus} activityExpandOverride={turnKey ? expandedTurnIds.has(turnKey) : undefined} + runtimeContext={runtimeContext} /> {showTurnStatus && status && @@ -430,7 +428,7 @@ export function NativeChatMessageList({ workedSeconds={turnStatuses.active.workedSeconds} /> ) : null} - {!showTurnStatus && showTypingIndicator ? : null} + {!showTurnStatus && showTypingIndicator ? : null}
    {showJump ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index 7a02829a05a..f7d2fd62663 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -22,6 +22,7 @@ import { useStructuredAgentSession } from './use-structured-agent-session' import { translate } from '@/i18n/i18n' import { NativeChatOrchestrationPausedNotice } from './NativeChatOrchestrationPausedNotice' import { useNativeChatPasteBridge } from './use-native-chat-paste-bridge' +import { useNativeChatImageRuntimeContext } from './native-chat-image-runtime-context' function encodeQuestionAnswer(questionId: string, answer: string): string { return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` @@ -80,6 +81,7 @@ export function NativeChatStructuredSession(props: { const viewState = selectNativeChatViewState(session) const fontScale = useNativeChatFontScale(viewState.kind === 'ready') const fileLinkContext = useNativeChatFileLinkContext(props.tabId) + const imageRuntimeContext = useNativeChatImageRuntimeContext(props.tabId) const fileLinkClick = useNativeChatFileLinkClick(props.allowFileUriLinks ? fileLinkContext : null) const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null @@ -145,6 +147,7 @@ export function NativeChatStructuredSession(props: { showTurnStatus={props.agent === 'codex'} onLinkClick={fileLinkClick} allowFileUriLinks={fileLinkClick !== undefined} + runtimeContext={props.agent === 'codex' ? imageRuntimeContext : undefined} /> )}
    diff --git a/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx new file mode 100644 index 00000000000..3992bbce1f5 --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.test.tsx @@ -0,0 +1,209 @@ +// @vitest-environment happy-dom + +import { act, createElement } from 'react' +import { createRoot } from 'react-dom/client' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { + invalidateLocalImageSrcCacheForTests, + resetLocalImageSrcStateForTests +} from '@/components/editor/useLocalImageSrc' +import { NativeChatImageAttachments } from './NativeChatTranscriptChrome' + +globalThis.IS_REACT_ACT_ENVIRONMENT = true + +function runtimeContext(worktreeId: string): RuntimeFileOperationArgs { + return { + settings: { activeRuntimeEnvironmentId: null }, + worktreeId, + worktreePath: `/repo/${worktreeId}`, + expectedExecutionHostId: 'local' + } +} + +async function flushPromises(): Promise { + await Promise.resolve() + await Promise.resolve() +} + +beforeEach(() => { + resetLocalImageSrcStateForTests() + vi.stubGlobal('IntersectionObserver', undefined) + let urlSequence = 0 + vi.spyOn(URL, 'createObjectURL').mockImplementation(() => `blob:owner-${++urlSequence}`) + vi.spyOn(URL, 'revokeObjectURL').mockImplementation(() => undefined) + window.api = { + fs: { + readFile: vi.fn().mockResolvedValue({ + content: 'AA==', + isBinary: true, + mimeType: 'image/png' + }) + } + } as unknown as Window['api'] +}) + +afterEach(() => { + resetLocalImageSrcStateForTests() + vi.unstubAllGlobals() + vi.restoreAllMocks() +}) + +describe('NativeChatImageAttachments', () => { + it('pools visibility observation across image refs', async () => { + class FakeIntersectionObserver { + static instances: FakeIntersectionObserver[] = [] + readonly observe = vi.fn() + readonly unobserve = vi.fn() + readonly disconnect = vi.fn() + + constructor(_callback: IntersectionObserverCallback) { + FakeIntersectionObserver.instances.push(this) + } + } + vi.stubGlobal('IntersectionObserver', FakeIntersectionObserver) + + const container = document.createElement('div') + const root = createRoot(container) + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks: [ + { type: 'image-ref' as const, path: '/repo/one.png' }, + { type: 'image-ref' as const, path: '/repo/two.png' }, + { type: 'image-ref' as const, path: '/repo/three.png' } + ], + runtimeContext: runtimeContext('wt-1') + }) + ) + await flushPromises() + }) + + expect(FakeIntersectionObserver.instances).toHaveLength(1) + expect(FakeIntersectionObserver.instances[0]?.observe).toHaveBeenCalledTimes(3) + + root.unmount() + expect(FakeIntersectionObserver.instances[0]?.unobserve).toHaveBeenCalledTimes(3) + expect(FakeIntersectionObserver.instances[0]?.disconnect).toHaveBeenCalledOnce() + }) + + it('preserves same-image errors but retries when the runtime owner changes', async () => { + const container = document.createElement('div') + const root = createRoot(container) + const blocks = [{ type: 'image-ref' as const, path: '/repo/image.png' }] + const ownerOne = runtimeContext('wt-1') + + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks, + runtimeContext: ownerOne + }) + ) + await flushPromises() + }) + const firstOwnerSrc = container.querySelector('img')?.getAttribute('src') + expect(firstOwnerSrc).toBe('blob:owner-1') + + await act(async () => { + container.querySelector('img')?.dispatchEvent(new Event('error')) + }) + expect(container.querySelector('img')).toBeNull() + + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks, + runtimeContext: ownerOne + }) + ) + await flushPromises() + }) + expect(container.querySelector('img')).toBeNull() + + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks, + runtimeContext: runtimeContext('wt-2') + }) + ) + await flushPromises() + }) + expect(container.querySelector('img')?.getAttribute('src')).not.toBe(firstOwnerSrc) + expect(window.api.fs.readFile).toHaveBeenCalledTimes(2) + + root.unmount() + }) + + it('retries a failed thumbnail after the image cache refreshes', async () => { + const container = document.createElement('div') + const root = createRoot(container) + const props = { + blocks: [{ type: 'image-ref' as const, path: '/repo/image.png' }], + runtimeContext: runtimeContext('wt-1') + } + + await act(async () => { + root.render(createElement(NativeChatImageAttachments, props)) + await flushPromises() + }) + expect(container.querySelector('img')?.getAttribute('src')).toBe('blob:owner-1') + + await act(async () => { + container.querySelector('img')?.dispatchEvent(new Event('error')) + }) + expect(container.querySelector('img')).toBeNull() + + await act(async () => { + invalidateLocalImageSrcCacheForTests() + await flushPromises() + }) + + expect(container.querySelector('img')?.getAttribute('src')).toBe('blob:owner-2') + root.unmount() + }) + + it('keeps the observed element stable while a preview is materialized', async () => { + let callback: IntersectionObserverCallback | undefined + class FakeIntersectionObserver { + readonly observe = vi.fn() + readonly unobserve = vi.fn() + readonly disconnect = vi.fn() + + constructor(nextCallback: IntersectionObserverCallback) { + callback = nextCallback + } + } + vi.stubGlobal('IntersectionObserver', FakeIntersectionObserver) + + const container = document.createElement('div') + const root = createRoot(container) + await act(async () => { + root.render( + createElement(NativeChatImageAttachments, { + blocks: [{ type: 'image-ref' as const, path: '/repo/image.png' }], + runtimeContext: runtimeContext('wt-1') + }) + ) + await flushPromises() + }) + + const observedElement = container.firstElementChild + expect(observedElement).not.toBeNull() + if (!observedElement || !callback) { + throw new Error('image preview did not register visibility observation') + } + const notifyVisibility = callback + await act(async () => { + notifyVisibility( + [{ target: observedElement, isIntersecting: true } as IntersectionObserverEntry], + {} as IntersectionObserver + ) + await flushPromises() + }) + + expect(container.firstElementChild).toBe(observedElement) + root.unmount() + }) +}) diff --git a/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx index 206f0f8df5e..cc951002431 100644 --- a/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx +++ b/src/renderer/src/components/native-chat/NativeChatTranscriptChrome.tsx @@ -1,40 +1,241 @@ +import { useEffect, useRef, useState } from 'react' import { ArrowUp, Image as ImageIcon } from 'lucide-react' import { cn } from '@/lib/utils' import { translate } from '@/i18n/i18n' import { basename } from '@/lib/path' import type { NativeChatBlock } from '../../../../shared/native-chat-types' -import { isNativeChatPastedImagePath } from './native-chat-image-paste' import { NativeChatCopyButton } from './NativeChatCopyButton' import { nativeChatProviderFrameSummary } from '../../../../shared/native-chat-provider-frame-summary' +import { Dialog, DialogContent, DialogDescription, DialogTitle } from '@/components/ui/dialog' +import { + getLocalImageCacheKey, + useLocalImageSrc, + releaseLocalImageSrc +} from '@/components/editor/useLocalImageSrc' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { isNativeChatPastedImagePath } from './native-chat-image-paste' + +type VisibilityListener = (isVisible: boolean) => void + +const visibilityListeners = new Map() +let visibilityObserver: IntersectionObserver | null = null + +function observeTranscriptVisibility(element: Element, listener: VisibilityListener): () => void { + if (typeof IntersectionObserver === 'undefined') { + listener(true) + return () => {} + } + + visibilityObserver ??= new IntersectionObserver( + (entries) => { + for (const entry of entries) { + visibilityListeners.get(entry.target)?.(entry.isIntersecting) + } + }, + { rootMargin: '128px' } + ) + visibilityListeners.set(element, listener) + visibilityObserver.observe(element) + + return () => { + visibilityListeners.delete(element) + visibilityObserver?.unobserve(element) + if (visibilityListeners.size === 0) { + visibilityObserver?.disconnect() + visibilityObserver = null + } + } +} + +function renderableImageSource(source: string | undefined): boolean { + return Boolean(source && /^(?:https?|data|blob):/i.test(source)) +} + +function transcriptImageIdentity( + block: Extract, + runtimeContext: RuntimeFileOperationArgs | null | undefined +): string { + const source = block.url?.trim() || block.path + const filePath = block.path ?? source ?? '' + if (renderableImageSource(source)) { + return `external\0${source ?? ''}` + } + return `${source ?? ''}\0${filePath}\0${ + runtimeContext === null + ? 'unresolved' + : runtimeContext === undefined + ? 'pending' + : getLocalImageCacheKey(source ?? '', runtimeContext.connectionId, runtimeContext) + }` +} + +function TranscriptImagePreview({ + block, + runtimeContext +}: { + block: Extract + runtimeContext: RuntimeFileOperationArgs | null | undefined +}): React.JSX.Element { + const [open, setOpen] = useState(false) + const [near, setNear] = useState(false) + const [thumbnailErrorSrc, setThumbnailErrorSrc] = useState(null) + const [dialogErrorSrc, setDialogErrorSrc] = useState(null) + const ref = useRef(null) + const source = block.url?.trim() || block.path + const filePath = block.path ?? source ?? '' + const external = renderableImageSource(source) + const leaseActive = near || open + const localSrc = useLocalImageSrc( + leaseActive && !external && runtimeContext !== undefined ? source : undefined, + filePath, + runtimeContext?.connectionId, + runtimeContext + ) + const displaySrc = external && leaseActive ? source : localSrc + const label = + block.alt?.trim() || + (block.path && isNativeChatPastedImagePath(block.path) + ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') + : block.path + ? basename(block.path) + : 'Image') + const viewImageLabel = translate('components.native-chat.composer.viewAttachment', 'View image') + const fallback = ( +
    + + {label} +
    + ) + + useEffect(() => { + const element = ref.current + if (!element) { + return + } + return observeTranscriptVisibility(element, setNear) + }, []) + useEffect(() => { + const context = runtimeContext + if (!source || external || context === undefined || context === null) { + return + } + if (!leaseActive) { + releaseLocalImageSrc(source, filePath, context.connectionId, context) + } + return () => releaseLocalImageSrc(source, filePath, context.connectionId, context) + }, [external, filePath, leaseActive, runtimeContext, source]) + + const showPreview = + leaseActive && + Boolean(displaySrc) && + displaySrc !== thumbnailErrorSrc && + Boolean(source) && + (external || runtimeContext !== null) + + if (!showPreview) { + return
    {fallback}
    + } + return ( +
    + + + + {label} + + {translate('components.native-chat.composer.imagePreview', 'Full-size image preview')} + +
    + {displaySrc && displaySrc !== dialogErrorSrc ? ( + {label} setDialogErrorSrc(displaySrc)} + className="max-h-[75vh] max-w-full object-contain" + /> + ) : ( + fallback + )} +
    +
    +
    +
    + ) +} export function NativeChatImageAttachments({ - blocks + blocks, + runtimeContext, + enablePreview = runtimeContext !== undefined }: { blocks: NativeChatBlock[] + runtimeContext?: RuntimeFileOperationArgs | null + /** Keep legacy terminal chips unchanged until that lane opts into previews. */ + enablePreview?: boolean }): React.JSX.Element | null { const images = blocks.filter((block) => block.type === 'image-ref') if (images.length === 0) { return null } + const imageKeyCounts = new Map() + if (!enablePreview) { + return ( +
    + {images.map((image) => { + const label = image.alt ?? image.path ?? image.url ?? 'Image' + const imageKeyBase = `${label}-${image.url ?? ''}-${image.path ?? ''}` + const occurrence = imageKeyCounts.get(imageKeyBase) ?? 0 + imageKeyCounts.set(imageKeyBase, occurrence + 1) + const name = + image.path && isNativeChatPastedImagePath(image.path) + ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') + : image.path + ? basename(image.path) + : label + return ( +
    + + {name} +
    + ) + })} +
    + ) + } return (
    - {images.map((image, index) => { + {images.map((image) => { const label = image.alt ?? image.path ?? image.url ?? 'Image' - const name = - image.path && isNativeChatPastedImagePath(image.path) - ? translate('components.native-chat.composer.pastedImageLabel', 'Pasted image') - : image.path - ? basename(image.path) - : label + const imageKeyBase = `${label}-${image.url ?? ''}-${image.path ?? ''}` + const occurrence = imageKeyCounts.get(imageKeyBase) ?? 0 + imageKeyCounts.set(imageKeyBase, occurrence + 1) + const identity = transcriptImageIdentity(image, runtimeContext) return ( -
    - - {name} -
    + ) })}
    diff --git a/src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx b/src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx new file mode 100644 index 00000000000..59969c1dece --- /dev/null +++ b/src/renderer/src/components/native-chat/NativeChatTypingIndicatorRow.tsx @@ -0,0 +1,21 @@ +import { translate } from '@/i18n/i18n' + +export function NativeChatTypingIndicatorRow(): React.JSX.Element { + return ( +
    +
    + {[0, 1, 2].map((i) => ( + + ))} +
    +
    + ) +} diff --git a/src/renderer/src/components/native-chat/native-chat-file-link.test.ts b/src/renderer/src/components/native-chat/native-chat-file-link.test.ts index ab5d273ce54..eaf63279fe4 100644 --- a/src/renderer/src/components/native-chat/native-chat-file-link.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-file-link.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest' import type { Tab } from '../../../../shared/tab-types' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { AppState } from '@/store/types' +import { folderWorkspaceKey } from '../../../../shared/workspace-scope' import { resolveNativeChatFileLink, resolveNativeChatFileLinkContext, @@ -111,6 +112,27 @@ describe('resolveNativeChatFileLinkContext', () => { runtimeEnvironmentId: null }) }) + + it('resolves a folder workspace tab from its folder path when no projected worktree path exists', () => { + const folderId = 'folder-1' + const folderKey = folderWorkspaceKey(folderId) + const folderTab = terminalTab({ worktreeId: folderKey }) + expect( + resolveNativeChatFileLinkContext( + state({ + tabsByWorktree: { [folderKey]: [folderTab] }, + getKnownWorktreeById: () => undefined, + folderWorkspaces: [{ id: folderId, folderPath: '/workspace/platform' } as never], + worktreesByRepo: {} + }), + folderTab.id + ) + ).toEqual({ + worktreeId: folderKey, + worktreePath: '/workspace/platform', + runtimeEnvironmentId: null + }) + }) }) describe('resolveNativeChatFileLink', () => { diff --git a/src/renderer/src/components/native-chat/native-chat-file-link.ts b/src/renderer/src/components/native-chat/native-chat-file-link.ts index c2d77d066c5..431dc9a556e 100644 --- a/src/renderer/src/components/native-chat/native-chat-file-link.ts +++ b/src/renderer/src/components/native-chat/native-chat-file-link.ts @@ -6,6 +6,7 @@ import { } from '@/lib/explicit-file-link-target' import { getRuntimeEnvironmentIdForWorktree } from '@/lib/worktree-runtime-owner' import type { AppState } from '@/store/types' +import { parseWorkspaceKey } from '../../../../shared/workspace-scope' export type NativeChatFileLinkContext = { worktreeId: string @@ -86,13 +87,21 @@ export function resolveNativeChatFileLinkContext( const worktree = knownWorktree?.path ? knownWorktree : findWorktreeFallback(state.worktreesByRepo, worktreeId) - if (!worktree?.path) { + const workspaceScope = parseWorkspaceKey(worktreeId) + const worktreePath = + worktree?.path ?? + (workspaceScope?.type === 'folder' + ? (state.folderWorkspaces.find( + (workspace) => workspace.id === workspaceScope.folderWorkspaceId + )?.folderPath ?? null) + : null) + if (!worktreePath) { return null } return { worktreeId, - worktreePath: worktree.path, + worktreePath, runtimeEnvironmentId: getRuntimeEnvironmentIdForWorktree(state, worktreeId) } } diff --git a/src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts new file mode 100644 index 00000000000..cfda20998fa --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.test.ts @@ -0,0 +1,96 @@ +import { describe, expect, it } from 'vitest' +import { shallow } from 'zustand/shallow' +import type { AppState } from '@/store/types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import { + resolveNativeChatImageRuntimeContext, + selectNativeChatImageOwnerState +} from './native-chat-image-runtime-context' + +function state(): AppState { + const tab: TerminalTab = { + id: 'tab-1', + ptyId: null, + worktreeId: 'wt-1', + title: 'Terminal 1', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 0 + } + const worktree = { + id: 'wt-1', + repoId: 'repo', + path: '/repo/worktree', + hostId: 'local' + } + return { + activeWorkspaceExecutionHostId: 'local', + activeWorktreeId: 'wt-1', + detectedWorktreesByRepo: {}, + folderWorkspaces: [], + getKnownWorktreeById: () => worktree, + projectGroups: [], + removedRuntimeEnvironmentIds: new Set(), + repos: [{ id: 'repo', path: '/repo' }], + restoredRuntimeHostIdByWorkspaceSessionKey: {}, + runtimeEnvironmentCatalogHydrated: true, + runtimeEnvironments: [], + settings: { activeRuntimeEnvironmentId: null }, + sshConnectionStates: {}, + sshStateByEnvironment: {}, + tabsByWorktree: { 'wt-1': [tab] }, + unifiedTabsByWorktree: {}, + worktreesByRepo: { repo: [worktree] } + } as unknown as AppState +} + +describe('resolveNativeChatImageRuntimeContext', () => { + it('keeps unrelated store writes out of the image-owner selector', () => { + const storeState = state() + const first = selectNativeChatImageOwnerState(storeState) + const second = selectNativeChatImageOwnerState({ + ...storeState, + agentStatusByPaneKey: {} as AppState['agentStatusByPaneKey'] + }) + + expect(shallow(second, first)).toBe(true) + }) + + it('reuses derived settings when owner inputs are unchanged', () => { + const storeState = state() + const first = resolveNativeChatImageRuntimeContext(storeState, 'tab-1') + const second = resolveNativeChatImageRuntimeContext(storeState, 'tab-1') + + expect(first).not.toBeNull() + expect(second?.settings).toBe(first?.settings) + expect(shallow(second, first)).toBe(true) + }) + + it('derives a runtime host from an owner-only route during paired hydration', () => { + const storeState = state() + const ownerOnlyWorktree = { + id: 'wt-1', + repoId: 'repo', + path: '/repo/worktree', + runtimeOwnerEnvironmentId: 'owner-a' + } + const ownerState = { + ...storeState, + activeWorktreeId: null, + activeWorkspaceExecutionHostId: null, + getKnownWorktreeById: () => ownerOnlyWorktree, + worktreesByRepo: { repo: [ownerOnlyWorktree] }, + runtimeEnvironments: [{ id: 'owner-a' }] + } as unknown as AppState + + const context = resolveNativeChatImageRuntimeContext(ownerState, 'tab-1') + + expect(context).toMatchObject({ + worktreeId: 'wt-1', + worktreePath: '/repo/worktree', + expectedExecutionHostId: 'local', + settings: { activeRuntimeEnvironmentId: 'owner-a' } + }) + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts new file mode 100644 index 00000000000..600c5b63c1d --- /dev/null +++ b/src/renderer/src/components/native-chat/native-chat-image-runtime-context.ts @@ -0,0 +1,184 @@ +import { useAppStore } from '@/store' +import type { AppState } from '@/store/types' +import type { RuntimeFileOperationArgs } from '@/runtime/runtime-file-client' +import { + settingsForWorktreeOperationRoute, + resolveWorktreeOperationRouteResult +} from '@/lib/worktree-operation-route' +import { resolveNativeChatFileLinkContext } from './native-chat-file-link' +import { captureDirectSshMutationExpectation } from '@/lib/ssh-mutation-expectation' +import { + parseExecutionHostId, + toRuntimeExecutionHostId, + type ExecutionHostId +} from '../../../../shared/execution-host' +import { parseWorkspaceKey } from '../../../../shared/workspace-scope' +import { useMemo } from 'react' +import { useShallow } from 'zustand/react/shallow' + +/** The transcript must not read until ownership and its path are both known. */ +export type NativeChatImageRuntimeContext = RuntimeFileOperationArgs | null + +type OwnerState = Pick< + AppState, + | 'settings' + | 'repos' + | 'worktreesByRepo' + | 'detectedWorktreesByRepo' + | 'folderWorkspaces' + | 'projectGroups' + | 'runtimeEnvironments' + | 'runtimeEnvironmentCatalogHydrated' + | 'removedRuntimeEnvironmentIds' + | 'sshConnectionStates' + | 'sshStateByEnvironment' + | 'activeWorktreeId' + | 'activeWorkspaceExecutionHostId' + | 'restoredRuntimeHostIdByWorkspaceSessionKey' + | 'getKnownWorktreeById' + | 'tabsByWorktree' + | 'unifiedTabsByWorktree' +> + +// Keep the subscription limited to fields that can change image ownership. The +// derived context is computed during render, after Zustand has filtered updates. +export function selectNativeChatImageOwnerState(state: AppState): OwnerState { + return { + settings: state.settings, + repos: state.repos, + worktreesByRepo: state.worktreesByRepo, + detectedWorktreesByRepo: state.detectedWorktreesByRepo, + folderWorkspaces: state.folderWorkspaces, + projectGroups: state.projectGroups, + runtimeEnvironments: state.runtimeEnvironments, + runtimeEnvironmentCatalogHydrated: state.runtimeEnvironmentCatalogHydrated, + removedRuntimeEnvironmentIds: state.removedRuntimeEnvironmentIds, + sshConnectionStates: state.sshConnectionStates, + sshStateByEnvironment: state.sshStateByEnvironment, + activeWorktreeId: state.activeWorktreeId, + activeWorkspaceExecutionHostId: state.activeWorkspaceExecutionHostId, + restoredRuntimeHostIdByWorkspaceSessionKey: state.restoredRuntimeHostIdByWorkspaceSessionKey, + getKnownWorktreeById: state.getKnownWorktreeById, + tabsByWorktree: state.tabsByWorktree, + unifiedTabsByWorktree: state.unifiedTabsByWorktree + } +} + +// Route settings are cloned for the runtime operation contract. Reuse that +// clone while the store's source settings and selected runtime are unchanged so +// consumers do not treat an unrelated store update as a new image owner. +const settingsBySource = new WeakMap>() + +function stableSettingsForRoute( + settings: AppState['settings'], + runtimeEnvironmentId: string | null +): AppState['settings'] { + if (!settings) { + return settingsForWorktreeOperationRoute(settings, { + executionHostId: null, + runtimeEnvironmentId + }) + } + const source = settings as object + let byRuntime = settingsBySource.get(source) + if (!byRuntime) { + byRuntime = new Map() + settingsBySource.set(source, byRuntime) + } + const cacheKey = runtimeEnvironmentId ?? '' + const cached = byRuntime.get(cacheKey) + if (cached) { + return cached + } + const resolved = settingsForWorktreeOperationRoute(settings, { + executionHostId: null, + runtimeEnvironmentId + }) + byRuntime.set(cacheKey, resolved) + return resolved +} + +function resolvePath( + state: OwnerState, + worktreeId: string, + hostId: ExecutionHostId | null +): string | null { + const known = state.getKnownWorktreeById(worktreeId, hostId ?? undefined) + if (known?.path) { + return known.path + } + const workspace = parseWorkspaceKey(worktreeId) + if (workspace?.type === 'folder') { + return ( + state.folderWorkspaces.find((entry) => entry.id === workspace.folderWorkspaceId) + ?.folderPath ?? null + ) + } + for (const worktrees of Object.values(state.worktreesByRepo ?? {})) { + const match = worktrees.find( + (entry) => entry.id === worktreeId && (!hostId || entry.hostId === hostId) + ) + if (match?.path) { + return match.path + } + } + return null +} + +export function resolveNativeChatImageRuntimeContext( + state: OwnerState, + tabId: string +): NativeChatImageRuntimeContext { + const linkContext = resolveNativeChatFileLinkContext(state, tabId) + if (!linkContext) { + return null + } + const routeResolution = resolveWorktreeOperationRouteResult(state, linkContext.worktreeId) + if (routeResolution.kind !== 'resolved') { + return null + } + const route = routeResolution.route + const executionHostId = + route.executionHostId ?? + (route.runtimeEnvironmentId ? toRuntimeExecutionHostId(route.runtimeEnvironmentId) : null) + if (!executionHostId) { + return null + } + const worktreePath = resolvePath(state, linkContext.worktreeId, executionHostId) + if (!worktreePath) { + return null + } + const host = parseExecutionHostId(executionHostId) + if (!host) { + return null + } + const context: RuntimeFileOperationArgs = { + settings: stableSettingsForRoute(state.settings, route.runtimeEnvironmentId), + worktreeId: linkContext.worktreeId, + worktreePath, + expectedExecutionHostId: host.kind === 'ssh' ? host.id : 'local' + } + if (host.kind === 'ssh') { + try { + const expectation = captureDirectSshMutationExpectation( + state, + host.targetId, + route.runtimeEnvironmentId + ) + context.expectedSshTargetId = expectation.expectedSshTargetId + context.expectedSshConnectionGeneration = expectation.expectedSshConnectionGeneration + if (!route.runtimeEnvironmentId) { + context.connectionId = host.targetId + context.expectedExternalSshTargetId = host.targetId + } + } catch { + return null + } + } + return context +} + +export function useNativeChatImageRuntimeContext(tabId: string): NativeChatImageRuntimeContext { + const ownerState = useAppStore(useShallow(selectNativeChatImageOwnerState)) + return useMemo(() => resolveNativeChatImageRuntimeContext(ownerState, tabId), [ownerState, tabId]) +} From 21210aad3428a2ab52aba0e49e2c46a2922dfdc3 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Wed, 2 Sep 2026 20:23:11 -0700 Subject: [PATCH 135/398] fix(native-chat): make structured Codex launches race-resistant (#18251) * fix(native-chat): cancel close-racing structured launches * fix(native-chat): make structured launches observable and recoverable * fix(native-chat): reconcile merged session tab publications * refactor(native-chat): unify host snapshot versioning * refactor(native-chat): complete launches from host snapshots * fix(native-chat): replay unknown launches by intent * fix(native-chat): guard duplicate launches and bound sync recovery * test(native-chat): type owner fixture * test(native-chat): type owner fixture * fix(native-chat): back off structured session resubscribe * fix(native-chat): fence delayed local session snapshots * fix(native-chat): retry initial session sync safely * fix(native-chat): refresh before sync retry * test(native-chat): cover folder sync cursor cleanup * fix(native-chat): retry failed structured session subscriptions --------- Co-authored-by: Merge Sim --- ...ime-apply-mobile-session-tab-navigation.ts | 2 +- ...time-close-headless-mobile-terminal-tab.ts | 2 +- ...time-close-structured-agent-session-tab.ts | 8 +- ...e-runtime-owned-mobile-session-terminal.ts | 2 +- ...act-persisted-terminal-surface-identity.ts | 2 +- ...ile-session-tabs-from-workspace-session.ts | 2 +- ...untime-move-headless-mobile-session-tab.ts | 6 +- ...time-persist-headless-session-tab-props.ts | 4 +- ...me-persist-terminal-surface-retirements.ts | 2 +- ...lish-pty-backed-mobile-session-terminal.ts | 2 +- ...le-headless-mobile-session-browser-tabs.ts | 2 +- ...renderer-session-owned-mobile-terminals.ts | 2 +- ...tore-structured-agent-session-tabs-once.ts | 4 +- src/main/runtime/orca-runtime-runtime-id.ts | 15 ++ .../orca-runtime-stop-requested-pty-ids.ts | 2 +- ...mobile-snapshot-has-stale-preserved-tab.ts | 20 +- .../orca-runtime-sync-mobile-session-tabs.ts | 4 +- .../runtime/orca-runtime-sync-window-graph.ts | 2 +- .../mobile-session-tabs-part-06.spec.ts | 2 +- ...-touch-mobile-session-tabs-for-worktree.ts | 2 +- .../components/tab-bar/QuickLaunchButton.tsx | 22 +- .../TabBarCreateEntry.keyboard.test.tsx | 30 +++ .../components/tab-bar/TabBarCreateEntry.tsx | 27 +- .../tab-bar/TabBarCreateEntryRow.tsx | 13 +- ...abCloseCommands.structured-session.test.ts | 6 + .../tab-group/useTabGroupTabCloseCommands.ts | 5 + ...launch-agent-structured-chat-guard.test.ts | 29 ++- .../structured-agent-session-launch.test.ts | 162 ++++++++++-- .../lib/structured-agent-session-launch.ts | 239 +++++++++++++++--- ...local-structured-session-tabs-sync.test.ts | 211 +++++++++++++++- .../local-structured-session-tabs-sync.ts | 166 ++++++++++-- .../src/runtime/web-session-tabs-sync.test.ts | 34 +++ .../publisher-identity-fences.ts | 16 ++ .../tracking-decisions.ts | 22 +- 34 files changed, 931 insertions(+), 138 deletions(-) diff --git a/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts b/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts index 912a38b5abc..a2552ef12c3 100644 --- a/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts +++ b/src/main/runtime/orca-runtime-apply-mobile-session-tab-navigation.ts @@ -142,7 +142,7 @@ export class OrcaRuntimeWithApplyMobileSessionTabNavigation extends OrcaRuntimeW tabs } this.persistHeadlessTerminalActiveLeaf(worktreeId, activeTab) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } diff --git a/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts b/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts index 2786f8ac8a4..e0ac1bed9d0 100644 --- a/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts +++ b/src/main/runtime/orca-runtime-close-headless-mobile-terminal-tab.ts @@ -107,7 +107,7 @@ export class OrcaRuntimeWithCloseHeadlessMobileTerminalTab extends OrcaRuntimeWi : {}), tabs: nextTabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } diff --git a/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts b/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts index 57703367a52..9c6e5cda532 100644 --- a/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts +++ b/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts @@ -39,7 +39,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi })), tabs: nextTabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } @@ -49,7 +49,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi protected republishMobileSessionTabsSnapshot(worktreeId: string): void { const snapshot = this.mobileSessionTabsByWorktree.get(worktreeId) if (snapshot) { - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...snapshot, snapshotVersion: snapshot.snapshotVersion + 1 }) @@ -161,7 +161,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi })), tabs: nextTabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return true } @@ -203,7 +203,7 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi focusesHost, publicationEpoch: `headless:${Date.now().toString(36)}` }) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) // Why: browser group membership is otherwise live-only; persist it so a // later rebuild keeps the browser in its group instead of coalescing left. if (placedInTargetGroup && nextSnapshot.tabGroupLayout) { diff --git a/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts b/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts index ed9c9cec036..ee0b87693a8 100644 --- a/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts +++ b/src/main/runtime/orca-runtime-create-runtime-owned-mobile-session-terminal.ts @@ -139,7 +139,7 @@ export class OrcaRuntimeWithCreateRuntimeOwnedMobileSessionTerminal extends Orca ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, next) + this.storeMobileSessionSnapshot(worktreeId, next) const result = this.toMobileSessionTabsResult(next) const changeSequence = ++this.mobileSessionTabsChangeSequence for (const subscription of this.mobileSessionTabListeners) { diff --git a/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts b/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts index e71b0cb8010..6956644c748 100644 --- a/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts +++ b/src/main/runtime/orca-runtime-has-exact-persisted-terminal-surface-identity.ts @@ -65,7 +65,7 @@ export class OrcaRuntimeWithHasExactPersistedTerminalSurfaceIdentity extends Orc exactOnly: true }) if (retired) { - this.mobileSessionTabsByWorktree.set(candidate.worktreeId, retired.snapshot) + this.storeMobileSessionSnapshot(candidate.worktreeId, retired.snapshot) this.notifyMobileSessionTabsChanged(candidate.worktreeId) } } diff --git a/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts b/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts index 1a356c0cdb8..140093fc204 100644 --- a/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts +++ b/src/main/runtime/orca-runtime-hydrate-headless-mobile-session-tabs-from-workspace-session.ts @@ -236,7 +236,7 @@ export class OrcaRuntimeWithHydrateHeadlessMobileSessionTabsFromWorkspaceSession if (existing && headlessMobileSnapshotContentUnchanged(existing, nextSnapshot)) { continue } - this.mobileSessionTabsByWorktree.set(entryWorktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(entryWorktreeId, nextSnapshot) } return reconciledWorktreeIds } diff --git a/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts b/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts index 19a584fbe17..caaf892e65b 100644 --- a/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts +++ b/src/main/runtime/orca-runtime-move-headless-mobile-session-tab.ts @@ -72,7 +72,7 @@ export class OrcaRuntimeWithMoveHeadlessMobileSessionTab extends OrcaRuntimeWith if (nextGroups.length > 1 && snapshot.tabGroupLayout) { this.persistHeadlessTabGroups(worktreeId, nextGroups, snapshot.tabGroupLayout) } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return { moved: true } } @@ -111,7 +111,7 @@ export class OrcaRuntimeWithMoveHeadlessMobileSessionTab extends OrcaRuntimeWith tabGroupLayout: split.layout } this.persistHeadlessTabGroups(worktreeId, split.groups, split.layout) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return { moved: true } } @@ -147,7 +147,7 @@ export class OrcaRuntimeWithMoveHeadlessMobileSessionTab extends OrcaRuntimeWith tabGroupLayout: layout } this.persistHeadlessTabGroups(worktreeId, moved.groups, layout) - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) return { moved: true } } diff --git a/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts b/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts index e8eee89a1db..10489c6c701 100644 --- a/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts +++ b/src/main/runtime/orca-runtime-persist-headless-session-tab-props.ts @@ -95,7 +95,7 @@ export class OrcaRuntimeWithPersistHeadlessSessionTabProps extends OrcaRuntimeWi snapshotVersion: snapshot.snapshotVersion + 1, tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } @@ -181,7 +181,7 @@ export class OrcaRuntimeWithPersistHeadlessSessionTabProps extends OrcaRuntimeWi snapshotVersion: snapshot.snapshotVersion + 1, tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, nextSnapshot) + this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) } } diff --git a/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts b/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts index 1d275a91865..5c70a9e518a 100644 --- a/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts +++ b/src/main/runtime/orca-runtime-persist-terminal-surface-retirements.ts @@ -173,7 +173,7 @@ export class OrcaRuntimeWithPersistTerminalSurfaceRetirements extends OrcaRuntim : {}) }) if (retired) { - this.mobileSessionTabsByWorktree.set(worktreeId, retired.snapshot) + this.storeMobileSessionSnapshot(worktreeId, retired.snapshot) this.notifyMobileSessionTabsChanged(worktreeId) } } diff --git a/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts b/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts index b1ee888a65f..baad15289d1 100644 --- a/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts +++ b/src/main/runtime/orca-runtime-publish-pty-backed-mobile-session-terminal.ts @@ -142,7 +142,7 @@ export class OrcaRuntimeWithPublishPtyBackedMobileSessionTerminal extends OrcaRu ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.mobileSessionTabsByWorktree.set(worktreeId, next) + this.storeMobileSessionSnapshot(worktreeId, next) if (args.notify !== false) { this.notifyMobileSessionTabsChanged(worktreeId) } diff --git a/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts b/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts index a97746eaa30..6c4606ec5aa 100644 --- a/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts +++ b/src/main/runtime/orca-runtime-reconcile-headless-mobile-session-browser-tabs.ts @@ -47,7 +47,7 @@ export class OrcaRuntimeWithReconcileHeadlessMobileSessionBrowserTabs extends Or const active = activeStillPresent ? null : (nextTabs.find((tab) => tab.isActive) ?? nextTabs[0] ?? null) - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...existing, publicationEpoch: `headless-hydrated:${Date.now().toString(36)}`, snapshotVersion: existing.snapshotVersion + 1, diff --git a/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts b/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts index b5efd465589..bfab8f0ba9a 100644 --- a/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts +++ b/src/main/runtime/orca-runtime-restore-live-paired-renderer-session-owned-mobile-terminals.ts @@ -39,7 +39,7 @@ export class OrcaRuntimeWithRestoreLivePairedRendererSessionOwnedMobileTerminals continue } if (!existing) { - this.mobileSessionTabsByWorktree.set(targetWorktreeId, { + this.storeMobileSessionSnapshot(targetWorktreeId, { worktree: targetWorktreeId, publicationEpoch: `renderer-rescue:${Date.now().toString(36)}`, snapshotVersion: 0, diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index be255d723c1..36b5f65b12e 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -94,7 +94,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu ), tabs: existing.tabs.map((tab) => ({ ...tab, isActive: tab.id === id })) } - this.mobileSessionTabsByWorktree.set(input.workspaceId, snapshot) + this.storeMobileSessionSnapshot(input.workspaceId, snapshot) if (input.notify !== false) { this.emitMobileSessionTabsSnapshot(snapshot) } @@ -143,7 +143,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu ...(existing?.tabGroupLayout ? { tabGroupLayout: existing.tabGroupLayout } : {}), tabs } - this.mobileSessionTabsByWorktree.set(input.workspaceId, snapshot) + this.storeMobileSessionSnapshot(input.workspaceId, snapshot) if (input.notify !== false) { this.emitMobileSessionTabsSnapshot(snapshot) } diff --git a/src/main/runtime/orca-runtime-runtime-id.ts b/src/main/runtime/orca-runtime-runtime-id.ts index da55f2229b6..298cc2b7cb7 100644 --- a/src/main/runtime/orca-runtime-runtime-id.ts +++ b/src/main/runtime/orca-runtime-runtime-id.ts @@ -96,6 +96,21 @@ export class OrcaRuntimeWithRuntimeId { protected mobileSessionTabsByWorktree = new Map() + /** Single host writer for mobile session snapshots; versions are total-order stamps. */ + protected storeMobileSessionSnapshot( + worktreeId: string, + snapshot: RuntimeMobileSessionTabsSnapshot + ): RuntimeMobileSessionTabsSnapshot { + const existing = this.mobileSessionTabsByWorktree.get(worktreeId) + const snapshotVersion = existing + ? Math.max(snapshot.snapshotVersion, existing.snapshotVersion + 1) + : snapshot.snapshotVersion + const stamped = + snapshotVersion === snapshot.snapshotVersion ? snapshot : { ...snapshot, snapshotVersion } + this.mobileSessionTabsByWorktree.set(worktreeId, stamped) + return stamped + } + protected structuredAgentSessionTabRestorePromise: Promise | null = null protected structuredAgentSessionStartupRestorePromise: Promise | null = null diff --git a/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts b/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts index 233e739911b..278ccd28a74 100644 --- a/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts +++ b/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts @@ -57,7 +57,7 @@ export class OrcaRuntimeWithStopRequestedPtyIds extends OrcaRuntimeWithRuntimeId resolveOwner: (handle) => this.resolveNativeChatLaunchDraftOwner(handle), listMobileSnapshots: () => this.mobileSessionTabsByWorktree, setMobileSnapshot: (worktreeId, snapshot) => - this.mobileSessionTabsByWorktree.set(worktreeId, snapshot), + this.storeMobileSessionSnapshot(worktreeId, snapshot), scheduleMobileSnapshot: (worktreeId) => this.scheduleMobileSessionTabsChanged(worktreeId), notifyResolved: (tabId, resolution, event) => { this.notifier?.nativeChatLaunchDraftResolved?.(tabId, resolution) diff --git a/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts b/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts index 524018d889f..c32349352a1 100644 --- a/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts +++ b/src/main/runtime/orca-runtime-stored-mobile-snapshot-has-stale-preserved-tab.ts @@ -8,7 +8,6 @@ import type { import { getMobileSessionSnapshotTabIdentityKeys } from './mobile-session-tab-merge' import { getRuntimeBrowserPageRegistry } from './runtime-browser-page-registry' import { sameRuntimeBrowserPlacement } from '../../shared/runtime-browser-placement' -import { createHash } from 'node:crypto' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' export class OrcaRuntimeWithStoredMobileSnapshotHasStalePreservedTab extends OrcaRuntimeWithMergePreservedHeadlessMobileSessionTabs { @@ -115,23 +114,14 @@ export class OrcaRuntimeWithStoredMobileSnapshotHasStalePreservedTab extends Orc protected getMergedMobileSessionPublicationEpoch( snapshot: RuntimeMobileSessionTabsSnapshot, - preservedTabs: readonly RuntimeMobileSessionSnapshotTab[] + _preservedTabs: readonly RuntimeMobileSessionSnapshotTab[] ): string { // Why: preserved snapshots can merge repeatedly; strip the prior merge suffix first so the publication epoch stays idempotent. const normalizedPublicationEpoch = snapshot.publicationEpoch.split(':headless-merge:')[0] - const signature = createHash('sha1') - .update( - preservedTabs - .map((tab) => - tab.type === 'terminal' - ? `${tab.id}:${tab.parentTabId}:${tab.ptyId ?? ''}:${tab.leafId}` - : tab.id - ) - .join('|') - ) - .digest('hex') - .slice(0, 12) - return `${normalizedPublicationEpoch}:headless-merge:${signature}` + // The epoch identifies the publisher generation, not the merged content. + // Content changes are ordered by snapshotVersion, so encoding a merge hash + // here would make the identity oscillate and permanently fence later rows. + return normalizedPublicationEpoch } /** Serves a hydrating host renderer; the publisher counts this as a delivery, not a read. */ diff --git a/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts b/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts index 4d54a322f3e..51997577179 100644 --- a/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts +++ b/src/main/runtime/orca-runtime-sync-mobile-session-tabs.ts @@ -167,7 +167,7 @@ export class OrcaRuntimeWithSyncMobileSessionTabs extends OrcaRuntimeWithWriteOr const storedVersion = existing ? Math.max(nextSnapshot.snapshotVersion, existing.snapshotVersion + 1) : nextSnapshot.snapshotVersion - this.mobileSessionTabsByWorktree.set( + this.storeMobileSessionSnapshot( snapshot.worktree, storedVersion === nextSnapshot.snapshotVersion ? nextSnapshot @@ -195,7 +195,7 @@ export class OrcaRuntimeWithSyncMobileSessionTabs extends OrcaRuntimeWithWriteOr preserved.tabs.length === existing.tabs.length && preserved.tabs.every((tab, index) => tab === existing.tabs[index]) if (!preservedIsNoOp) { - this.mobileSessionTabsByWorktree.set(worktreeId, preserved) + this.storeMobileSessionSnapshot(worktreeId, preserved) } // Why: the stored entry is no longer the renderer's publication, so a // future renderer frame must be re-merged even if it reuses the pair. diff --git a/src/main/runtime/orca-runtime-sync-window-graph.ts b/src/main/runtime/orca-runtime-sync-window-graph.ts index f9c6ab358e3..7d44d4af2c1 100644 --- a/src/main/runtime/orca-runtime-sync-window-graph.ts +++ b/src/main/runtime/orca-runtime-sync-window-graph.ts @@ -238,7 +238,7 @@ export class OrcaRuntimeWithSyncWindowGraph extends OrcaRuntimeWithAttachWindow // the PTY touch path does) or the re-emitted payload — e.g. the // pending-handle → ready flip — is discarded and the client stays stale. // The accepted-renderer tracking is untouched: this is a main-local bump. - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...stored, snapshotVersion: stored.snapshotVersion + 1 }) diff --git a/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts b/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts index 27e052e1594..be1d4c8175c 100644 --- a/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts +++ b/src/main/runtime/orca-runtime-tests/mobile-session-tabs-part-06.spec.ts @@ -595,7 +595,7 @@ describe('OrcaRuntimeService', () => { const secondMerge = await runtime.listMobileSessionTabs(`id:${TEST_WORKTREE_ID}`) expect(secondMerge.publicationEpoch).toBe(firstMerge.publicationEpoch) - expect(secondMerge.publicationEpoch.match(/:headless-merge:/g) ?? []).toHaveLength(1) + expect(secondMerge.publicationEpoch).toBe('headless:stable-epoch') }) it('keeps the graph ready when a mobile snapshot references a removed folder workspace', () => { diff --git a/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts b/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts index 1481709ee66..b7ca09af705 100644 --- a/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts +++ b/src/main/runtime/orca-runtime-touch-mobile-session-tabs-for-worktree.ts @@ -21,7 +21,7 @@ export class OrcaRuntimeWithTouchMobileSessionTabsForWorktree extends OrcaRuntim if (!snapshot) { return } - this.mobileSessionTabsByWorktree.set(worktreeId, { + this.storeMobileSessionSnapshot(worktreeId, { ...snapshot, snapshotVersion: snapshot.snapshotVersion + 1 }) diff --git a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx index 6af2508048c..85d978e3bf6 100644 --- a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx +++ b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx @@ -1,5 +1,5 @@ import React, { useCallback } from 'react' -import { Settings as SettingsIcon } from 'lucide-react' +import { Loader2, Settings as SettingsIcon } from 'lucide-react' import { toast } from 'sonner' import { DropdownMenuItem, DropdownMenuShortcut } from '@/components/ui/dropdown-menu' import { getAgentCatalog, AgentIcon } from '@/lib/agent-catalog' @@ -15,6 +15,7 @@ import { filterEnabledTuiAgents } from '../../../../shared/tui-agent-selection' import { translate } from '@/i18n/i18n' +import { useStructuredCodexLaunchStatus } from '@/lib/structured-agent-session-launch' export type QuickLaunchAgentMenuItemsProps = { worktreeId: string @@ -116,6 +117,7 @@ function QuickLaunchAgentMenuItemsInner({ const openSettingsPage = useAppStore((s) => s.openSettingsPage) const openSettingsTarget = useAppStore((s) => s.openSettingsTarget) const newAgentShortcut = useOptionalShortcutLabel('tab.newAgent') + const structuredCodexLaunchStatus = useStructuredCodexLaunchStatus(worktreeId) const openAgentSettings = useCallback(() => { openSettingsTarget({ pane: 'agents', repoId: null }) @@ -197,21 +199,31 @@ function QuickLaunchAgentMenuItemsInner({ {agents.map((agent) => { const entry = getCatalogEntry(agent) const label = entry?.label ?? agent + const isStructuredCodexPending = + agent === 'codex' && structuredCodexLaunchStatus === 'pending' + const menuLabel = isStructuredCodexPending ? 'Starting Codex chat…' : label const showsDefaultAgentShortcut = newAgentShortcut !== null && defaultAgent !== 'blank' && agent === defaultAgent return ( runLaunch(agent)} className="gap-2 rounded-[7px] px-2 py-1.5 text-[12px] leading-5 font-medium" title={translate( 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', - 'Launch {{value0}} in a new terminal', - { value0: label } + isStructuredCodexPending + ? 'Starting Codex chat…' + : 'Launch {{value0}} in a new terminal', + isStructuredCodexPending ? undefined : { value0: label } )} > - - {label} + {isStructuredCodexPending ? ( +
    + + +
    + + +
    +
    +
    +

    Production · us-central1

    +

    Relay data plane

    +

    Private, aggregate operations view. No account, host, device, or pairing identifiers.

    +
    +
    + + Loading current state… +
    +
    + + + + +
    +
    +
    +
    +
    +
    + +
    +
    +

    Topology

    Request path

    + +
    +
    +
    + +

    Traffic

    Signals in selected window

    +
    + +
    + +
    +

    Control plane

    Services

    +
    +
    +
    + +
    + +
    +

    Delivery

    Recent workflows

    +
    +
    +
    +

    Planning

    Monthly run-rate

    +
    +
    +
    + + +
    +
    Orca Relay Operations · Server-side credentials · Read-only by default
    + + diff --git a/cloud/apps/relay-ops/public/styles.css b/cloud/apps/relay-ops/public/styles.css new file mode 100644 index 00000000000..17a974c3d9a --- /dev/null +++ b/cloud/apps/relay-ops/public/styles.css @@ -0,0 +1,165 @@ +:root { + --radius: 10px; + --background: #fff; + --foreground: #0a0a0a; + --card: #fff; + --primary: #171717; + --primary-foreground: #fafafa; + --secondary: #f5f5f5; + --muted: #f5f5f5; + --muted-foreground: #737373; + --accent: #f5f5f5; + --border: #e5e5e5; + --ring: #a1a1a1; + --destructive: #e40014; + --success: #15803d; + --warning: #895503; + --font-mono: 'SF Mono', SFMono-Regular, ui-monospace, Menlo, Consolas, monospace; +} +@media (prefers-color-scheme: dark) { + :root { + --background: #0a0a0a; + --foreground: #fafafa; + --card: #171717; + --primary: #e5e5e5; + --primary-foreground: #171717; + --secondary: #262626; + --muted: #262626; + --muted-foreground: #a1a1a1; + --accent: #262626; + --border: rgb(255 255 255 / 0.07); + --ring: #737373; + --destructive: #ff6568; + --success: #4ade80; + --warning: #fbbf24; + } +} +* { box-sizing: border-box; } +body { + margin: 0; + background: var(--background); + color: var(--foreground); + font-family: 'Geist', -apple-system, BlinkMacSystemFont, 'Segoe UI', sans-serif; + font-size: 14px; + letter-spacing: .01em; +} +button, select { font: inherit; } +button:focus-visible, select:focus-visible { outline: 2px solid var(--ring); outline-offset: 2px; } +.topbar { + height: 64px; + border-bottom: 1px solid var(--border); + padding: 0 max(24px, calc((100vw - 1440px) / 2)); + display: flex; + align-items: center; + justify-content: space-between; + position: sticky; + top: 0; + background: color-mix(in srgb, var(--background) 92%, transparent); + backdrop-filter: blur(16px); + z-index: 10; +} +.brand { display: flex; align-items: center; gap: 10px; } +.brand-mark { + width: 30px; height: 30px; border-radius: 9px; background: var(--primary); + color: var(--primary-foreground); display: grid; place-items: center; font-weight: 700; +} +.brand div { display: flex; align-items: baseline; gap: 7px; } +.brand span:last-child { color: var(--muted-foreground); font-size: 12px; } +.toolbar { display: flex; align-items: center; gap: 8px; } +.segmented { display: flex; padding: 3px; gap: 2px; border-radius: 8px; background: var(--secondary); } +.segmented button { border: 0; background: transparent; color: var(--muted-foreground); padding: 5px 10px; border-radius: 6px; cursor: pointer; } +.segmented button[aria-pressed="true"] { background: var(--card); color: var(--foreground); box-shadow: 0 1px 2px rgb(0 0 0 / .08); } +select, .button { height: 32px; border: 1px solid var(--border); border-radius: 8px; background: var(--card); color: var(--foreground); padding: 0 10px; } +.button { cursor: pointer; font-weight: 500; } +.button:hover { background: var(--accent); } +.button:disabled { opacity: .5; cursor: wait; } +main { max-width: 1440px; margin: 0 auto; padding: 42px 24px 72px; } +.page-heading { display: flex; justify-content: space-between; align-items: end; gap: 24px; margin-bottom: 28px; } +h1, h2, p { margin: 0; } +h1 { font-size: 28px; line-height: 1.2; letter-spacing: -.025em; margin-top: 5px; } +h2 { font-size: 16px; line-height: 1.25; letter-spacing: -.01em; margin-top: 3px; } +.lede { color: var(--muted-foreground); margin-top: 8px; max-width: 680px; } +.eyebrow { color: var(--muted-foreground); font-size: 11px; line-height: 1; font-weight: 600; text-transform: uppercase; letter-spacing: .05em; } +.freshness { color: var(--muted-foreground); display: flex; align-items: center; gap: 8px; font-size: 12px; white-space: nowrap; } +.status-dot { width: 7px; height: 7px; border-radius: 999px; background: var(--ring); display: inline-block; } +.status-dot.healthy { background: var(--success); } +.status-dot.unhealthy { background: var(--destructive); } +.summary-grid { display: grid; grid-template-columns: repeat(4, minmax(0, 1fr)); gap: 12px; margin-bottom: 12px; } +.metric-card, .panel { border: 1px solid var(--border); border-radius: 14px; background: var(--card); } +.metric-card { padding: 18px; min-height: 116px; } +.metric-card .value { font-size: 29px; font-weight: 600; letter-spacing: -.035em; margin-top: 16px; } +.metric-card .caption { color: var(--muted-foreground); font-size: 12px; margin-top: 3px; } +.skeleton { background: linear-gradient(90deg, var(--card), var(--muted), var(--card)); background-size: 200% 100%; animation: shimmer 1.6s infinite; } +@keyframes shimmer { to { background-position: -200% 0; } } +.panel { padding: 18px; } +.topology-panel { margin-bottom: 36px; } +.panel-heading, .section-heading { display: flex; align-items: center; justify-content: space-between; gap: 16px; margin-bottom: 18px; } +.section-heading { margin-top: 34px; } +.topology { display: grid; grid-template-columns: 1fr 48px 1fr 48px 2fr; align-items: stretch; gap: 8px; } +.topology-node { border: 1px solid var(--border); background: var(--background); border-radius: 10px; padding: 14px; min-width: 0; } +.topology-node strong { display: block; font-size: 13px; } +.topology-node span { display: block; color: var(--muted-foreground); font-size: 12px; margin-top: 4px; overflow: hidden; text-overflow: ellipsis; } +.topology-cells { display: grid; grid-template-columns: repeat(3, minmax(0, 1fr)); gap: 8px; } +.connector { display: grid; place-items: center; color: var(--muted-foreground); } +.connector::before { content: ''; width: 100%; height: 1px; background: var(--border); } +.chart-grid { display: grid; grid-template-columns: repeat(3, minmax(0, 1fr)); gap: 12px; margin-bottom: 36px; } +.chart-card { min-height: 220px; } +.chart-head { display: flex; justify-content: space-between; align-items: start; } +.chart-value { font-size: 22px; font-weight: 600; letter-spacing: -.03em; } +.chart-unit { color: var(--muted-foreground); font-size: 11px; } +.chart { width: 100%; height: 122px; margin-top: 18px; overflow: visible; } +.chart path.area { fill: color-mix(in srgb, var(--foreground) 5%, transparent); } +.chart path.line { fill: none; stroke: var(--foreground); stroke-width: 1.5; vector-effect: non-scaling-stroke; } +.chart line { stroke: var(--border); stroke-width: 1; } +.chart-empty { color: var(--muted-foreground); height: 120px; display: grid; place-items: center; font-size: 12px; } +.two-column { display: grid; grid-template-columns: 1.7fr 1fr; gap: 12px; margin-bottom: 12px; } +.three-column { display: grid; grid-template-columns: repeat(3, minmax(0, 1fr)); gap: 12px; margin-bottom: 12px; } +.two-column > *, .three-column > *, .chart-grid > * { min-width: 0; } +.table-wrap { overflow-x: auto; } +table { width: 100%; border-collapse: collapse; font-size: 12px; } +th { color: var(--muted-foreground); font-weight: 500; text-align: left; padding: 0 10px 10px; } +td { padding: 12px 10px; border-top: 1px solid var(--border); vertical-align: middle; } +td:first-child, th:first-child { padding-left: 0; } +td:last-child, th:last-child { padding-right: 0; } +.mono { font-family: var(--font-mono); font-size: 11px; } +.badge { display: inline-flex; align-items: center; border: 1px solid var(--border); border-radius: 999px; padding: 2px 7px; font-size: 11px; color: var(--muted-foreground); } +.badge.healthy { color: var(--success); border-color: color-mix(in srgb, var(--success) 25%, var(--border)); } +.badge.unhealthy { color: var(--destructive); border-color: color-mix(in srgb, var(--destructive) 25%, var(--border)); } +.service-list, .list { display: grid; } +.service-row, .list-row { padding: 11px 0; border-top: 1px solid var(--border); display: flex; align-items: center; justify-content: space-between; gap: 12px; min-width: 0; } +.service-row:first-child, .list-row:first-child { border-top: 0; padding-top: 0; } +.service-row:last-child, .list-row:last-child { padding-bottom: 0; } +.row-title { font-size: 12px; font-weight: 500; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +.row-caption, .meta, .muted { color: var(--muted-foreground); font-size: 11px; } +.panel-note { color: var(--muted-foreground); font-size: 11px; line-height: 1.45; margin: -8px 0 14px; } +.panel-link:hover { color: var(--foreground); text-decoration: underline; } +.row-caption { margin-top: 3px; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } +a { color: inherit; text-decoration: none; } +a:hover .row-title { text-decoration: underline; } +.cost-total { display: flex; align-items: baseline; gap: 7px; margin-bottom: 12px; } +.cost-total strong { font-size: 28px; letter-spacing: -.035em; } +.cost-line { display: flex; justify-content: space-between; gap: 10px; padding: 7px 0; border-top: 1px solid var(--border); font-size: 12px; } +.cost-note { color: var(--muted-foreground); font-size: 11px; line-height: 1.5; margin-top: 12px; } +.banner { border: 1px solid var(--border); border-radius: 10px; padding: 11px 13px; margin-bottom: 12px; font-size: 12px; } +.banner.error { color: var(--destructive); } +.banner.warning { color: var(--warning); } +.hidden { display: none !important; } +.power-actions { display: flex; align-items: center; gap: 8px; margin-top: 14px; flex-wrap: wrap; } +footer { border-top: 1px solid var(--border); padding: 24px; color: var(--muted-foreground); font-size: 11px; text-align: center; } +@media (max-width: 1050px) { + .summary-grid, .chart-grid { grid-template-columns: repeat(2, minmax(0, 1fr)); } + .three-column { grid-template-columns: 1fr; } + .topology { grid-template-columns: 1fr; } + .connector { height: 20px; } + .connector::before { height: 100%; width: 1px; } +} +@media (max-width: 760px) { + .topbar { height: auto; min-height: 64px; padding: 12px 16px; align-items: stretch; flex-direction: column; gap: 12px; } + .brand div { display: grid; gap: 0; } + .toolbar { flex-wrap: wrap; justify-content: flex-start; } + main { padding: 28px 16px 56px; } + .page-heading { align-items: flex-start; flex-direction: column; } + .summary-grid, .chart-grid, .two-column { grid-template-columns: 1fr; } + .topology-cells { grid-template-columns: 1fr; } + table { min-width: 600px; } +} diff --git a/cloud/apps/relay-ops/src/cost-model.test.ts b/cloud/apps/relay-ops/src/cost-model.test.ts new file mode 100644 index 00000000000..9ee6f6adf9d --- /dev/null +++ b/cloud/apps/relay-ops/src/cost-model.test.ts @@ -0,0 +1,85 @@ +import { describe, expect, it } from 'vitest' +import { buildCostModel } from './cost-model.js' +import { + RELAY_OPS_ENVIRONMENTS, + relayOpsCellsFromTerraform, + type RelayOpsCellConfig +} from './environment-config.js' +import type { ResourceInventory } from './resource-inventory.js' + +const regionalCellsSource = ` +relay_gce_cells = { + "staging-gce-c3" = { + hostname = "c3" + zone = "us-central1-a" + machine_type = "e2-standard-2" + capacity_requests = 4000 + } + "staging-gce-c4" = { + hostname = "c4" + region = "asia-east2" + zone = "asia-east2-a" + machine_type = "e2-standard-4" + capacity_requests = 6000 + database_pool_max = 10 + initially_enabled = false + } +} +` + +function inventory( + targetSize: number, + activationPolicy: string, + cells: RelayOpsCellConfig[] = RELAY_OPS_ENVIRONMENTS.staging.cells +): ResourceInventory { + return { + director: null, + auth: null, + sql: { state: targetSize ? 'RUNNABLE' : 'STOPPED', activationPolicy, tier: 'db-custom-1-3840', availabilityType: 'ZONAL', databaseVersion: 'POSTGRES_17' }, + certificate: null, + directorEndpoint: { health: null, ready: null, latencyMs: null }, + authEndpoint: { health: null, ready: null, latencyMs: null }, + cells: cells.map((cell) => ({ + ...cell, migName: `mig-${cell.hostname}`, targetSize, runningInstances: targetSize, + stable: true, template: 'template', imageDigest: null, + backendHealth: targetSize ? 'healthy' : 'empty', + endpoint: { health: null, ready: null, latencyMs: null } + })), + warnings: [] + } +} + +describe('buildCostModel', () => { + it('shows the sleeping staging floor without VM or SQL compute', () => { + const result = buildCostModel(RELAY_OPS_ENVIRONMENTS.staging, inventory(0, 'NEVER')) + expect(result.kind).toBe('planning-estimate') + expect(result.monthlyUsd).toBe(34) + expect(result.lines.find((line) => line.label === 'Relay cell VMs')?.monthlyUsd).toBe(0) + expect(result.actualBilling.available).toBe(false) + expect(result.caveats[0]).toContain('not the Cloud Billing invoice') + }) + + it('prices durable machine inventory and network floors by region', () => { + const cells = relayOpsCellsFromTerraform({ + environment: 'staging', + domain: 'relay-staging.onorca.dev', + source: regionalCellsSource + }) + const environment = { ...RELAY_OPS_ENVIRONMENTS.staging, cells } + const result = buildCostModel(environment, inventory(1, 'NEVER', cells)) + const machines = result.lines.find((line) => line.label === 'Relay cell VMs') + const network = result.lines.find((line) => line.label === 'Load balancer and network floor') + + expect(machines).toEqual({ + label: 'Relay cell VMs', + monthlyUsd: 185.79, + basis: '2 configured VM cells at regional machine rates × 730 hours' + }) + + expect(network).toEqual({ + label: 'Load balancer and network floor', + monthlyUsd: 29, + basis: 'shared HTTPS foundation plus NAT floor in 2 configured regions' + }) + }) +}) diff --git a/cloud/apps/relay-ops/src/cost-model.ts b/cloud/apps/relay-ops/src/cost-model.ts new file mode 100644 index 00000000000..39e062597a8 --- /dev/null +++ b/cloud/apps/relay-ops/src/cost-model.ts @@ -0,0 +1,115 @@ +import type { + RelayOpsEnvironment, + RelayOpsMachineType, + RelayOpsRegion +} from './environment-config.js' +import type { ResourceInventory } from './resource-inventory.js' + +export type CostLine = { + label: string + monthlyUsd: number + basis: string +} + +export type CostModel = { + kind: 'planning-estimate' + monthlyUsd: number + rangeUsd: [number, number] + actualBilling: { available: false; reason: string } + lines: CostLine[] + caveats: string[] +} + +const HOURS_PER_MONTH = 730 +const MACHINE_HOURLY_USD: Record< + RelayOpsRegion, + Record +> = { + 'us-central1': { + 'e2-standard-2': 0.06701142, + 'e2-standard-4': 0.13402284 + }, + 'asia-east2': { + 'e2-standard-2': 0.0938, + 'e2-standard-4': 0.1875 + } +} +const SHARED_NETWORK_FLOOR_USD = 19 +const REGIONAL_NAT_FLOOR_USD = 5 + +function round(value: number): number { + return Math.round(value * 100) / 100 +} + +export function buildCostModel( + environment: RelayOpsEnvironment, + resources: ResourceInventory +): CostModel { + const runningCells = resources.cells.filter((cell) => (cell.targetSize ?? 0) > 0) + const runningCellCount = runningCells.reduce((sum, cell) => sum + (cell.targetSize ?? 0), 0) + const compute = runningCells.reduce( + (sum, cell) => + sum + (cell.targetSize ?? 0) * MACHINE_HOURLY_USD[cell.region][cell.machineType], + 0 + ) * HOURS_PER_MONTH + const disks = runningCellCount * 30 * 0.1 + const sqlRunning = resources.sql?.activationPolicy === 'ALWAYS' + const sql = sqlRunning ? (environment.id === 'production' ? 105 : 52) : 0 + const cloudRunMinimums = + (resources.director?.minInstances ?? 0) + (resources.auth?.minInstances ?? 0) + const cloudRun = cloudRunMinimums * 10 + const configuredRegions = new Set( + (resources.cells.length > 0 ? resources.cells : environment.cells).map((cell) => cell.region) + ) + const networkFoundation = + SHARED_NETWORK_FLOOR_USD + configuredRegions.size * REGIONAL_NAT_FLOOR_USD + const observability = environment.id === 'production' ? 12 : 5 + const lines: CostLine[] = [ + { + label: 'Relay cell VMs', + monthlyUsd: round(compute), + basis: `${runningCellCount} configured VM cells at regional machine rates × 730 hours` + }, + { + label: 'Cell boot disks', + monthlyUsd: round(disks), + basis: `${runningCellCount} × 30 GB balanced persistent disk` + }, + { + label: 'Cloud SQL', + monthlyUsd: sql, + basis: sqlRunning ? `${resources.sql?.tier ?? 'configured tier'} active` : 'stopped' + }, + { + label: 'Cloud Run minimums', + monthlyUsd: cloudRun, + basis: `${cloudRunMinimums} configured minimum instances` + }, + { + label: 'Load balancer and network floor', + monthlyUsd: networkFoundation, + basis: `shared HTTPS foundation plus NAT floor in ${configuredRegions.size} configured region${configuredRegions.size === 1 ? '' : 's'}` + }, + { + label: 'Logs and monitoring allowance', + monthlyUsd: observability, + basis: 'planning allowance; varies with traffic and retention' + } + ] + const monthlyUsd = round(lines.reduce((sum, line) => sum + line.monthlyUsd, 0)) + return { + kind: 'planning-estimate', + monthlyUsd, + rangeUsd: [round(monthlyUsd * 0.85), round(monthlyUsd * 1.3)], + actualBilling: { + available: false, + reason: 'Cloud Billing export is not configured in either Relay project.' + }, + lines, + caveats: [ + 'This is a modeled run-rate, not the Cloud Billing invoice.', + 'Network egress, actual request traffic, credits, discounts, taxes, and free tiers are excluded.', + 'Use the GCP Billing report for authoritative spend and forecasts.' + ] + } +} diff --git a/cloud/apps/relay-ops/src/dashboard-snapshot.test.ts b/cloud/apps/relay-ops/src/dashboard-snapshot.test.ts new file mode 100644 index 00000000000..247b4b53ed0 --- /dev/null +++ b/cloud/apps/relay-ops/src/dashboard-snapshot.test.ts @@ -0,0 +1,62 @@ +import { describe, expect, it } from 'vitest' +import { DashboardSnapshotCache } from './dashboard-snapshot.js' +import type { DashboardSnapshot } from './dashboard-snapshot.js' +import type { GcloudClient } from './gcloud-client.js' + +const gcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } + +function snapshot(kind: 'good' | 'unavailable'): DashboardSnapshot { + const good = kind === 'good' + return { + generatedAt: '2026-07-15T12:00:00.000Z', + resources: { + director: good ? {} : null, + auth: good ? {} : null, + sql: good ? {} : null, + cells: [{ targetSize: good ? 1 : null }] + }, + monitoring: { + warnings: good + ? [] + : ['Cloud Monitoring credentials are unavailable. Run gcloud auth login.'] + }, + summary: { observedConnections: good ? 7 : 0 }, + warnings: good ? [] : ['Google Cloud credentials are unavailable. Run gcloud auth login.'], + stale: false, + staleReason: null + } as unknown as DashboardSnapshot +} + +describe('DashboardSnapshotCache', () => { + it('keeps the last good view when a later credential refresh fails', async () => { + let calls = 0 + const cache = new DashboardSnapshotCache(gcloud, 0, async () => { + calls += 1 + return snapshot(calls === 1 ? 'good' : 'unavailable') + }) + + const first = await cache.read('production', 30) + const second = await cache.read('production', 31) + + expect(first.stale).toBe(false) + expect(second.stale).toBe(true) + expect(second.summary.observedConnections).toBe(7) + expect(second.staleReason).toContain('credentials') + }) + + it('keeps the last good view when a later collector throws', async () => { + let calls = 0 + const cache = new DashboardSnapshotCache(gcloud, 0, async () => { + calls += 1 + if (calls > 1) throw new Error('sensitive collector context') + return snapshot('good') + }) + + await cache.read('production', 30) + const second = await cache.read('production', 31) + + expect(second.stale).toBe(true) + expect(second.summary.observedConnections).toBe(7) + expect(JSON.stringify(second)).not.toContain('sensitive collector context') + }) +}) diff --git a/cloud/apps/relay-ops/src/dashboard-snapshot.ts b/cloud/apps/relay-ops/src/dashboard-snapshot.ts new file mode 100644 index 00000000000..0d6f0c1151b --- /dev/null +++ b/cloud/apps/relay-ops/src/dashboard-snapshot.ts @@ -0,0 +1,167 @@ +import type { RelayOpsEnvironmentId } from './environment-config.js' +import { relayOpsEnvironment } from './environment-config.js' +import type { GcloudClient } from './gcloud-client.js' +import { readRelayWorkflowRuns } from './github-runs.js' +import { readMonitoringSnapshot } from './monitoring-snapshot.js' +import { readResourceInventory } from './resource-inventory.js' +import { buildCostModel } from './cost-model.js' + +export type DashboardSnapshot = Awaited> + +export async function buildDashboardSnapshot( + environmentId: RelayOpsEnvironmentId, + gcloud: GcloudClient, + options: { windowMinutes?: number; fetchImpl?: typeof fetch; now?: Date } = {} +) { + const generatedAt = (options.now ?? new Date()).toISOString() + const environment = relayOpsEnvironment(environmentId) + const [monitoringResult, resourceResult, workflowResult] = await Promise.allSettled([ + readMonitoringSnapshot(environment, gcloud, { + ...(options.now === undefined ? {} : { now: options.now }), + ...(options.windowMinutes === undefined ? {} : { windowMinutes: options.windowMinutes }), + ...(options.fetchImpl === undefined ? {} : { fetchImpl: options.fetchImpl }) + }), + readResourceInventory(environment, gcloud, options.fetchImpl), + readRelayWorkflowRuns() + ]) + if (monitoringResult.status === 'rejected' || resourceResult.status === 'rejected') { + const failed = [ + monitoringResult.status === 'rejected' ? 'Monitoring snapshot' : null, + resourceResult.status === 'rejected' ? 'Resource inventory' : null + ].filter(Boolean) + throw new Error(`${failed.join(' and ')} unavailable`) + } + const resources = resourceResult.value + const monitoring = monitoringResult.value + const warnings = [...resources.warnings, ...monitoring.warnings] + const poweredDigests = new Set( + resources.cells.filter((cell) => (cell.targetSize ?? 0) > 0).map((cell) => cell.imageDigest) + ) + if (poweredDigests.has(null)) warnings.push('A powered cell image digest is unavailable.') + if (poweredDigests.size > 1) warnings.push('Powered cells are not serving one immutable digest.') + const expectedCertificateDomain = `*.${new URL(environment.cells[0]!.origin).hostname + .split('.').slice(1).join('.')}` + if (resources.certificate && !resources.certificate.domains.includes(expectedCertificateDomain)) { + warnings.push('The Relay certificate domain does not match the configured cell domain.') + } + if (workflowResult.status === 'rejected') warnings.push('GitHub workflow history is unavailable.') + const observedConnections = monitoring.metrics.total_connections.latest ?? 0 + const observedControls = monitoring.metrics.controls.latest ?? 0 + const observedSplices = monitoring.metrics.splices.latest ?? 0 + const configuredCapacity = environment.cells + .filter((cell) => cell.configuredAdmission) + .reduce((sum, cell) => sum + cell.capacityRequests, 0) + const cellInventoryAvailable = resources.cells.every((cell) => cell.targetSize !== null) + const poweredCapacity = cellInventoryAvailable + ? resources.cells + .filter((cell) => (cell.targetSize ?? 0) > 0) + .reduce((sum, cell) => sum + cell.capacityRequests, 0) + : null + return { + schemaVersion: 1, + generatedAt, + environment: { + id: environment.id, + label: environment.label, + project: environment.project, + region: environment.region, + directorOrigin: environment.directorOrigin, + authOrigin: environment.authOrigin, + consoleLinks: { + project: `https://console.cloud.google.com/home/dashboard?project=${environment.project}`, + alerts: `https://console.cloud.google.com/monitoring/alerting?project=${environment.project}`, + compute: `https://console.cloud.google.com/compute/instanceGroups/list?project=${environment.project}` + } + }, + summary: { + observedConnections, + observedControls, + observedSplices, + configuredCapacity, + poweredCapacity, + activeCells: cellInventoryAvailable + ? resources.cells.filter( + (cell) => (cell.targetSize ?? 0) > 0 && cell.backendHealth === 'healthy' + ).length + : null, + totalCells: resources.cells.length + }, + resources, + monitoring, + workflows: workflowResult.status === 'fulfilled' ? workflowResult.value : [], + cost: buildCostModel(environment, resources), + warnings, + stale: false, + staleReason: null as string | null + } +} + +type SnapshotCacheEntry = { snapshot: DashboardSnapshot; expiresAt: number } +type SnapshotBuilder = ( + environment: RelayOpsEnvironmentId, + gcloud: GcloudClient, + options: { windowMinutes?: number } +) => Promise + +export class DashboardSnapshotCache { + private readonly entries = new Map() + private readonly pending = new Map>() + private readonly lastGood = new Map() + + constructor( + private readonly gcloud: GcloudClient, + private readonly ttlMs = 30_000, + private readonly builder: SnapshotBuilder = buildDashboardSnapshot + ) {} + + async read(environment: RelayOpsEnvironmentId, windowMinutes: number): Promise { + const key = `${environment}:${windowMinutes}` + const cached = this.entries.get(key) + if (cached && cached.expiresAt > Date.now()) return cached.snapshot + const existing = this.pending.get(key) + if (existing) return await existing + const request = this.builder(environment, this.gcloud, { windowMinutes }) + .then((snapshot) => { + const coreInventoryUnavailable = + snapshot.resources.director === null && + snapshot.resources.auth === null && + snapshot.resources.sql === null && + snapshot.resources.cells.every((cell) => cell.targetSize === null) + const monitoringCredentialsUnavailable = snapshot.monitoring.warnings.some( + (warning) => warning.includes('credentials are unavailable') + ) + const lastGood = this.lastGood.get(environment) + const result = (coreInventoryUnavailable || monitoringCredentialsUnavailable) && lastGood + ? { + ...lastGood, + stale: true, + staleReason: 'Local Google Cloud credentials are temporarily unavailable.', + warnings: [...new Set([...lastGood.warnings, ...snapshot.warnings])] + } + : snapshot + if (!coreInventoryUnavailable && !monitoringCredentialsUnavailable) { + this.lastGood.set(environment, snapshot) + } + this.entries.set(key, { snapshot: result, expiresAt: Date.now() + this.ttlMs }) + return result + }) + .catch((error: unknown) => { + const lastGood = this.lastGood.get(environment) + if (!lastGood) throw error + const stale = { + ...lastGood, + stale: true, + staleReason: 'The latest operations refresh failed; showing the last good snapshot.', + warnings: [...new Set([ + ...lastGood.warnings, + 'The latest operations refresh failed before a complete snapshot was available.' + ])] + } + this.entries.set(key, { snapshot: stale, expiresAt: Date.now() + this.ttlMs }) + return stale + }) + .finally(() => this.pending.delete(key)) + this.pending.set(key, request) + return await request + } +} diff --git a/cloud/apps/relay-ops/src/environment-config.test.ts b/cloud/apps/relay-ops/src/environment-config.test.ts new file mode 100644 index 00000000000..321167d7cc8 --- /dev/null +++ b/cloud/apps/relay-ops/src/environment-config.test.ts @@ -0,0 +1,148 @@ +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + RELAY_OPS_ENVIRONMENTS, + relayOpsCellsFromTerraform +} from './environment-config.js' + +const durableUsCell = ` + "production-gce-c26" = { + hostname = "c26" + zone = "us-central1-a" + machine_type = "e2-standard-4" + capacity_requests = 4000 + initially_enabled = false + } +` + +const durableAsiaCells = ` + "production-gce-c27" = { + hostname = "c27" + region = "asia-east2" + zone = "asia-east2-a" + machine_type = "e2-standard-4" + capacity_requests = 6000 + database_pool_max = 10 + initially_enabled = false + } + "production-gce-c28" = { + hostname = "c28" + region = "asia-east2" + zone = "asia-east2-b" + machine_type = "e2-standard-4" + capacity_requests = 6000 + database_pool_max = 10 + initially_enabled = false + } + "production-gce-c29" = { + hostname = "c29" + region = "asia-east2" + zone = "asia-east2-c" + machine_type = "e2-standard-4" + capacity_requests = 6000 + database_pool_max = 10 + initially_enabled = false + } +` + +const durableAsiaSource = ` +relay_gce_cells = {${durableUsCell}${durableAsiaCells}} +` + +const durableUsOnlySource = ` +relay_gce_cells = {${durableUsCell}} +` + +// Why: relay-ops sat at 18 cells for four days after C19-C22 shipped, which threw +// `selector membership must contain every configured cell exactly once` and blocked +// every production mutation. Reading Terraform here makes that drift fail the build. +function terraformCells(environment: 'production' | 'staging'): Array<{ + cellId: string + region: string + zone: string + machineType: string + capacityRequests: number + databasePoolMax: number +}> { + const tfvars = readFileSync( + fileURLToPath(new URL(`../../../infra/terraform/environments/${environment}.tfvars`, import.meta.url)), + 'utf8' + ) + const block = /relay_gce_cells\s*=\s*\{([\s\S]*)\n\}/.exec(tfvars)?.[1] ?? '' + return [...block.matchAll(/"([a-z]+-gce-c\d+)"\s*=\s*\{([\s\S]*?)\n {2}\}/g)] + .map((match) => ({ + cellId: match[1] ?? '', + region: /region\s*=\s*"([^"]+)"/.exec(match[2] ?? '')?.[1] ?? 'us-central1', + zone: /zone\s*=\s*"([^"]+)"/.exec(match[2] ?? '')?.[1] ?? '', + machineType: /machine_type\s*=\s*"([^"]+)"/.exec(match[2] ?? '')?.[1] ?? '', + capacityRequests: Number(/capacity_requests\s*=\s*(\d+)/.exec(match[2] ?? '')?.[1]), + databasePoolMax: Number(/database_pool_max\s*=\s*(\d+)/.exec(match[2] ?? '')?.[1] ?? 10) + })) + .sort((left, right) => cellOrdinal(left.cellId) - cellOrdinal(right.cellId)) +} + +const cellOrdinal = (cellId: string): number => Number(/c(\d+)$/.exec(cellId)?.[1] ?? 0) + +describe('relay operations environment config', () => { + it.each(['production', 'staging'] as const)( + 'matches the %s cells Terraform actually provisions', + (environment) => { + const expected = terraformCells(environment) + expect(expected.length).toBeGreaterThan(0) + expect( + RELAY_OPS_ENVIRONMENTS[environment].cells.map((cell) => ({ + cellId: cell.cellId, + region: cell.region, + zone: cell.zone, + machineType: cell.machineType, + capacityRequests: cell.capacityRequests, + databasePoolMax: cell.databasePoolMax + })) + ).toEqual(expected) + } + ) + + it('derives hostname and origin from the cell ordinal', () => { + const cells = RELAY_OPS_ENVIRONMENTS.production.cells + expect(cells.at(-1)).toMatchObject({ + hostname: `c${cells.length}`, + origin: `https://c${cells.length}.relay.onorca.dev` + }) + }) + + it('inventories Asia cells only when they exist in durable Terraform', () => { + const cells = relayOpsCellsFromTerraform({ + environment: 'production', + domain: 'relay.onorca.dev', + source: durableAsiaSource + }) + + expect(cells.slice(1)).toEqual([ + { + cellId: 'production-gce-c27', hostname: 'c27', origin: 'https://c27.relay.onorca.dev', + region: 'asia-east2', zone: 'asia-east2-a', machineType: 'e2-standard-4', + capacityRequests: 6000, databasePoolMax: 10, configuredAdmission: false + }, + { + cellId: 'production-gce-c28', hostname: 'c28', origin: 'https://c28.relay.onorca.dev', + region: 'asia-east2', zone: 'asia-east2-b', machineType: 'e2-standard-4', + capacityRequests: 6000, databasePoolMax: 10, configuredAdmission: false + }, + { + cellId: 'production-gce-c29', hostname: 'c29', origin: 'https://c29.relay.onorca.dev', + region: 'asia-east2', zone: 'asia-east2-c', machineType: 'e2-standard-4', + capacityRequests: 6000, databasePoolMax: 10, configuredAdmission: false + } + ]) + + const usOnlyCells = relayOpsCellsFromTerraform({ + environment: 'production', + domain: 'relay.onorca.dev', + source: durableUsOnlySource + }) + expect(usOnlyCells).toHaveLength(1) + expect(usOnlyCells.some((cell) => cell.region === 'asia-east2')).toBe(false) + expect(usOnlyCells.some((cell) => cell.cellId === 'production-gce-c27')).toBe(false) + }) +}) diff --git a/cloud/apps/relay-ops/src/environment-config.ts b/cloud/apps/relay-ops/src/environment-config.ts new file mode 100644 index 00000000000..da3de7765ed --- /dev/null +++ b/cloud/apps/relay-ops/src/environment-config.ts @@ -0,0 +1,125 @@ +import { readFileSync } from 'node:fs' +import { z } from 'zod' + +export type RelayOpsEnvironmentId = 'production' | 'staging' +export type RelayOpsRegion = 'us-central1' | 'asia-east2' +export type RelayOpsMachineType = 'e2-standard-2' | 'e2-standard-4' + +export type RelayOpsCellConfig = { + cellId: string + hostname: string + origin: string + region: RelayOpsRegion + zone: string + machineType: RelayOpsMachineType + capacityRequests: number + databasePoolMax: number + configuredAdmission: boolean +} + +export type RelayOpsEnvironment = { + id: RelayOpsEnvironmentId + label: string + project: string + region: string + directorOrigin: string + authOrigin: string + directorService: string + authService: string + sqlInstance: string + migPrefix: string + certificateName: string + cells: RelayOpsCellConfig[] +} + +const RegionSchema = z.enum(['us-central1', 'asia-east2']) +const MachineTypeSchema = z.enum(['e2-standard-2', 'e2-standard-4']) +const EnvironmentSchema = z.enum(['production', 'staging']) + +function cellOrdinal(cellId: string): number { + return Number(/c(\d+)$/.exec(cellId)?.[1] ?? 0) +} + +function required(body: string, pattern: RegExp, label: string): string { + const value = pattern.exec(body)?.[1] + if (!value) throw new Error(`Relay Ops could not read ${label} from durable Terraform config`) + return value +} + +export function relayOpsCellsFromTerraform(input: { + environment: RelayOpsEnvironmentId + domain: string + source: string +}): RelayOpsCellConfig[] { + const block = /relay_gce_cells\s*=\s*\{([\s\S]*)\n\}/.exec(input.source)?.[1] + if (!block) throw new Error('Relay Ops could not read durable Relay cells') + return [...block.matchAll(/"([a-z]+-gce-c\d+)"\s*=\s*\{([\s\S]*?)\n {2}\}/g)] + .map((match) => { + const cellId = match[1]! + const body = match[2]! + const hostname = required(body, /\bhostname\s*=\s*"([^"]+)"/, `${cellId} hostname`) + const configuredAdmission = /\binitially_enabled\s*=\s*(true|false)/.exec(body)?.[1] + return { + cellId, + hostname, + origin: `https://${hostname}.${input.domain}`, + region: RegionSchema.parse( + /\bregion\s*=\s*"([^"]+)"/.exec(body)?.[1] ?? 'us-central1' + ), + zone: required(body, /\bzone\s*=\s*"([^"]+)"/, `${cellId} zone`), + machineType: MachineTypeSchema.parse( + required(body, /\bmachine_type\s*=\s*"([^"]+)"/, `${cellId} machine type`) + ), + capacityRequests: Number( + required(body, /\bcapacity_requests\s*=\s*(\d+)/, `${cellId} capacity`) + ), + databasePoolMax: Number(/\bdatabase_pool_max\s*=\s*(\d+)/.exec(body)?.[1] ?? 10), + configuredAdmission: configuredAdmission === undefined || configuredAdmission === 'true' + } + }) + .sort((left, right) => cellOrdinal(left.cellId) - cellOrdinal(right.cellId)) +} + +function durableCells(environment: RelayOpsEnvironmentId, domain: string): RelayOpsCellConfig[] { + const source = readFileSync( + // Repository root; infra/terraform moves with this tree, so the relative depth holds. + new URL(`../../../infra/terraform/environments/${environment}.tfvars`, import.meta.url), + 'utf8' + ) + return relayOpsCellsFromTerraform({ environment, domain, source }) +} + +export const RELAY_OPS_ENVIRONMENTS: Record = { + production: { + id: 'production', + label: 'Production', + project: 'onorca-cloud', + region: 'us-central1', + directorOrigin: 'https://relay.onorca.dev', + authOrigin: 'https://login.onorca.dev', + directorService: 'orca-cloud-relay', + authService: 'orca-cloud-auth', + sqlInstance: 'orca-cloud-auth-db', + migPrefix: 'orca-cloud-relay-gce-', + certificateName: 'orca-cloud-relay-gce', + cells: durableCells('production', 'relay.onorca.dev') + }, + staging: { + id: 'staging', + label: 'Staging', + project: 'onorca-cloud-staging', + region: 'us-central1', + directorOrigin: 'https://relay-staging.onorca.dev', + authOrigin: 'https://auth-staging.onorca.dev', + directorService: 'orca-cloud-relay-staging', + authService: 'orca-cloud-auth-staging', + sqlInstance: 'orca-cloud-staging-auth-db', + migPrefix: 'orca-cloud-staging-relay-gce-', + certificateName: 'orca-cloud-staging-relay-gce', + cells: durableCells('staging', 'relay-staging.onorca.dev') + } +} + +export function relayOpsEnvironment(value: unknown): RelayOpsEnvironment { + return RELAY_OPS_ENVIRONMENTS[EnvironmentSchema.parse(value)] +} diff --git a/cloud/apps/relay-ops/src/gcloud-client.test.ts b/cloud/apps/relay-ops/src/gcloud-client.test.ts new file mode 100644 index 00000000000..3347c0cb367 --- /dev/null +++ b/cloud/apps/relay-ops/src/gcloud-client.test.ts @@ -0,0 +1,43 @@ +import { describe, expect, it } from 'vitest' +import { createGcloudClient } from './gcloud-client.js' + +describe('createGcloudClient', () => { + it('shares and caches one credential refresh across concurrent readers', async () => { + let calls = 0 + const token = 'a'.repeat(40) + const client = createGcloudClient(async () => { + calls += 1 + await Promise.resolve() + return token + }) + + const values = await Promise.all([ + client.accessToken(), + client.accessToken(), + client.accessToken() + ]) + + expect(values).toEqual([token, token, token]) + expect(await client.accessToken()).toBe(token) + expect(calls).toBe(1) + }) + + it('caches bounded identity tokens by audience', async () => { + const commands: string[][] = [] + const token = 'aaa.bbb.ccc' + const client = createGcloudClient(async (args) => { + commands.push(args) + return token + }) + await expect(client.identityToken?.('https://relay.example/admin')).resolves.toBe(token) + await expect(client.identityToken?.('https://relay.example/admin')).resolves.toBe(token) + expect(commands).toEqual([ + [ + 'auth', + 'print-identity-token', + '--audiences=https://relay.example/admin', + '--include-email' + ] + ]) + }) +}) diff --git a/cloud/apps/relay-ops/src/gcloud-client.ts b/cloud/apps/relay-ops/src/gcloud-client.ts new file mode 100644 index 00000000000..b565242ec0e --- /dev/null +++ b/cloud/apps/relay-ops/src/gcloud-client.ts @@ -0,0 +1,77 @@ +import { execFile } from 'node:child_process' +import { promisify } from 'node:util' + +const execFileAsync = promisify(execFile) +const TOKEN_PATTERN = /^[A-Za-z0-9._~+\/-]{32,8192}$/ + +export type GcloudClient = { + accessToken(): Promise + identityToken?(audience: string): Promise +} + +export class GcloudCommandError extends Error { + constructor(readonly operation: string) { + super(`${operation} is unavailable`) + } +} + +async function runGcloud(args: string[]): Promise { + try { + const result = await execFileAsync('gcloud', args, { + encoding: 'utf8', + timeout: 90_000, + maxBuffer: 8 * 1024 * 1024, + env: { + ...process.env, + CLOUDSDK_COMPONENT_MANAGER_DISABLE_UPDATE_CHECK: '1', + CLOUDSDK_CORE_DISABLE_PROMPTS: '1', + CLOUDSDK_CORE_DISABLE_USAGE_REPORTING: '1' + } + }) + return result.stdout.trim() + } catch { + // Gcloud stderr can echo command context; keep dashboard errors intentionally non-sensitive. + throw new GcloudCommandError(`gcloud ${args.slice(0, 3).join(' ')}`) + } +} + +export function createGcloudClient( + tokenCommand: (args: string[]) => Promise = runGcloud +): GcloudClient { + let cachedToken: { value: string; expiresAt: number } | null = null + const identityTokens = new Map() + let pendingToken: Promise | null = null + return { + async accessToken(): Promise { + if (cachedToken && cachedToken.expiresAt > Date.now()) return cachedToken.value + if (pendingToken) return await pendingToken + // One refresh avoids concurrent gcloud processes contending on the local credential store. + pendingToken = tokenCommand(['auth', 'print-access-token']) + .then((token) => { + if (!TOKEN_PATTERN.test(token)) throw new GcloudCommandError('gcloud access token') + cachedToken = { value: token, expiresAt: Date.now() + 5 * 60_000 } + return token + }) + .finally(() => { pendingToken = null }) + return await pendingToken + }, + async identityToken(audience: string): Promise { + const cached = identityTokens.get(audience) + if (cached && cached.expiresAt > Date.now()) return cached.value + const token = await tokenCommand([ + 'auth', + 'print-identity-token', + `--audiences=${audience}`, + '--include-email' + ]) + if ( + token.length > 8_192 || + !/^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/.test(token) + ) { + throw new GcloudCommandError('gcloud identity token') + } + identityTokens.set(audience, { value: token, expiresAt: Date.now() + 5 * 60_000 }) + return token + } + } +} diff --git a/cloud/apps/relay-ops/src/github-runs.ts b/cloud/apps/relay-ops/src/github-runs.ts new file mode 100644 index 00000000000..47ef8469fa9 --- /dev/null +++ b/cloud/apps/relay-ops/src/github-runs.ts @@ -0,0 +1,77 @@ +import { execFile } from 'node:child_process' +import { promisify } from 'node:util' +import { z } from 'zod' +import { relayRepositoryApiPath } from './relay-repository.js' + +const execFileAsync = promisify(execFile) + +const WorkflowRunSchema = z.object({ + id: z.number().int().positive(), + name: z.string(), + event: z.string(), + status: z.string(), + conclusion: z.string().nullable(), + head_sha: z.string(), + html_url: z.string().url(), + created_at: z.string(), + updated_at: z.string() +}) + +const WorkflowRunsSchema = z.object({ + workflow_runs: z.array(WorkflowRunSchema) +}) + +export type RelayWorkflowRun = { + id: number + name: string + status: string + conclusion: string | null + headSha: string + url: string + createdAt: string + updatedAt: string +} + +const RELAY_WORKFLOW_PATTERN = /(Relay|Auth|Power)/i + +export async function readRelayWorkflowRuns(): Promise { + let stdout: string + try { + const result = await execFileAsync( + 'gh', + [ + 'api', + '--method', + 'GET', + relayRepositoryApiPath('actions/runs'), + '-f', + 'per_page=100' + ], + { encoding: 'utf8', timeout: 30_000, maxBuffer: 4 * 1024 * 1024 } + ) + stdout = result.stdout + } catch { + throw new Error('GitHub workflow history is unavailable') + } + const runs = WorkflowRunsSchema.parse(JSON.parse(stdout) as unknown).workflow_runs + const counts = new Map() + return runs + .filter((run) => { + if (!RELAY_WORKFLOW_PATTERN.test(run.name)) return false + const count = counts.get(run.name) ?? 0 + if (count >= 2) return false + counts.set(run.name, count + 1) + return true + }) + .slice(0, 12) + .map((run) => ({ + id: run.id, + name: run.name, + status: run.status, + conclusion: run.conclusion, + headSha: run.head_sha.slice(0, 8), + url: run.html_url, + createdAt: run.created_at, + updatedAt: run.updated_at + })) +} diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts new file mode 100644 index 00000000000..18ee4495078 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.test.ts @@ -0,0 +1,373 @@ +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + livePreflightGcloud, + runIncidentLivePreflight +} from './incident-live-preflight-cli.js' +import type { IncidentSample } from './incident-monitor.js' +import type { AdmissionSelector } from './incident-selector.js' + +const directories: string[] = [] +const now = Date.parse('2026-07-28T12:00:00.000Z') +const selector = { + generation: 1, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: [], + general: [] + } +} + +function stateFile( + migrationPolicy: 'strict' | 'recover-forward' | 'capacity-transition' = 'strict', + overrides: Record = {} +): string { + const directory = mkdtempSync(join(tmpdir(), 'relay-live-preflight-')) + directories.push(directory) + const path = join(directory, 'state.json') + const expectedSelector = migrationPolicy === 'capacity-transition' + ? { + generation: 1, + membership: { + existingOnly: [], + migrationOnly: [], + general: ['production-gce-c1'] + } + } + : selector + writeFileSync(path, JSON.stringify({ + schemaVersion: 4, + environment: 'production', + expectedSelector, + migrationPolicy, + recoverySourceCellId: + migrationPolicy === 'recover-forward' ? 'production-gce-c1' : null, + capacityCellId: + migrationPolicy === 'capacity-transition' ? 'production-gce-c1' : null, + preDrainDryRun: true, + startedAt: new Date(now - 17 * 60_000).toISOString(), + windowStartedAt: new Date(now - 16 * 60_000).toISOString(), + durationMinutes: 15, + intervalMs: 60_000, + sampleCount: 16, + lastSampleAt: new Date(now - 60_007).toISOString(), + frozenAt: null, + completedAt: new Date(now - 60_000).toISOString(), + ...overrides + })) + return path +} + +function sample(): IncidentSample { + const observedAt = new Date(now).toISOString() + const signal = (value: number) => ({ value, observedAt }) + return { + collectedAt: observedAt, + selector, + expectedSelector: selector, + cells: [{ + cellId: 'production-gce-c1', + runtimeKnown: true, + powered: true, + expectedAdmissionState: 'existing-only' + }], + sources: { + 'active-probe': { + observedAt, + signals: { + 'director.health': signal(1), + 'director.ready': signal(1), + 'director.latency_ms': signal(1), + 'auth.health': signal(1), + 'auth.ready': signal(1), + 'auth.latency_ms': signal(1), + 'cell.production-gce-c1.health': signal(1), + 'cell.production-gce-c1.ready': signal(1), + 'cell.production-gce-c1.latency_ms': signal(1) + } + }, + 'cloud-monitoring': { + observedAt, + signals: { + 'cloud_sql.cpu': signal(0.1), + 'cloud_sql.memory': signal(0.1), + 'cloud_sql.backends': signal(1), + 'cloud_sql.lock_waits': signal(0), + 'cloud_sql.deadlocks': signal(0), + 'director.instances': signal(5), + 'director.cpu': signal(0.1), + 'director.memory': signal(0.1), + 'director.concurrency': signal(1), + 'director.errors': signal(0), + 'auth.errors': signal(0) + } + }, + 'relay-logs': { + observedAt, + signals: { + 'relay.pool_waiting': signal(0), + 'relay.pool_wait_ms': signal(0), + 'relay.postgres_retries': signal(0), + 'relay.postgres_retry_exhausted': signal(0), + 'cell.production-gce-c1.connections': signal(1), + 'cell.production-gce-c1.queued_bytes': signal(0) + } + }, + 'director-admin': { + observedAt, + signals: { + 'cell.production-gce-c1.admission_state': signal(0), + 'cell.production-gce-c1.heartbeat_fresh': signal(1), + 'cell.production-gce-c1.heartbeat_age_ms': signal(1), + 'cell.production-gce-c1.migration_blocked': signal(0), + 'cell.production-gce-c1.migration_target_inactive': signal(0) + } + } + } + } +} + +afterEach(() => { + for (const directory of directories.splice(0)) { + rmSync(directory, { recursive: true, force: true }) + } +}) + +describe('relay incident live preflight', () => { + it('accepts the package-manager argument separator', async () => { + await expect(runIncidentLivePreflight( + ['--', '--state-file', stateFile()], + { now: () => now, collect: async () => sample() } + )).resolves.toBeUndefined() + }) + + it('accepts one complete fresh green sample', async () => { + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: async () => sample() } + )).resolves.toBeUndefined() + }) + + it('rejects monitor evidence beyond the 25-minute lineage bound', async () => { + const path = stateFile('strict', { + startedAt: new Date(now - 26 * 60_000 - 1).toISOString() + }) + await expect(runIncidentLivePreflight( + ['--state-file', path], + { now: () => now, collect: async () => sample() } + )).rejects.toThrow('monitor evidence is incomplete or stale') + }) + + it('scales the evidence age bound by same-cap wave index', async () => { + const agedState = (ageMs: number) => stateFile('strict', { + startedAt: new Date(now - ageMs - 17 * 60_000).toISOString(), + windowStartedAt: new Date(now - ageMs - 16 * 60_000).toISOString(), + lastSampleAt: new Date(now - ageMs - 7).toISOString(), + completedAt: new Date(now - ageMs).toISOString() + }) + const deps = { now: () => now, collect: async () => sample() } + // One predecessor cell roll (~16 min) exceeds wave 0 but fits wave 1. + const oneRollOld = agedState(17 * 60_000) + await expect(runIncidentLivePreflight( + ['--state-file', oneRollOld], deps + )).rejects.toThrow('monitor evidence is incomplete or stale') + await expect(runIncidentLivePreflight( + ['--state-file', oneRollOld, '--wave-index', '0'], deps + )).rejects.toThrow('monitor evidence is incomplete or stale') + await expect(runIncidentLivePreflight( + ['--state-file', oneRollOld, '--wave-index', '1'], deps + )).resolves.toBeUndefined() + // Both edges of one predecessor job timeout: 5min + 75min exactly. + await expect(runIncidentLivePreflight( + ['--state-file', agedState(80 * 60_000), '--wave-index', '1'], deps + )).resolves.toBeUndefined() + await expect(runIncidentLivePreflight( + ['--state-file', agedState(80 * 60_000 + 1), '--wave-index', '1'], deps + )).rejects.toThrow('monitor evidence is incomplete or stale') + await expect(runIncidentLivePreflight( + ['--state-file', agedState(155 * 60_000), '--wave-index', '2'], deps + )).resolves.toBeUndefined() + await expect(runIncidentLivePreflight( + ['--state-file', agedState(155 * 60_000 + 1), '--wave-index', '2'], deps + )).rejects.toThrow('monitor evidence is incomplete or stale') + await expect(runIncidentLivePreflight( + ['--state-file', agedState(230 * 60_000), '--wave-index', '3'], deps + )).resolves.toBeUndefined() + await expect(runIncidentLivePreflight( + ['--state-file', agedState(230 * 60_000 + 1), '--wave-index', '3'], deps + )).rejects.toThrow('monitor evidence is incomplete or stale') + // The wave index is a strict single-use 0-3 argument. + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', '4'], deps + )).rejects.toThrow('usage:') + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', ''], deps + )).rejects.toThrow('usage:') + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--wave-index', '1', '--wave-index', '1'], + deps + )).rejects.toThrow('usage:') + }) + + it('expects the wave-adjusted live selector generation', async () => { + const agedPath = stateFile('strict', { + startedAt: new Date(now - 34 * 60_000).toISOString(), + windowStartedAt: new Date(now - 33 * 60_000).toISOString(), + lastSampleAt: new Date(now - 17 * 60_000 - 7).toISOString(), + completedAt: new Date(now - 17 * 60_000).toISOString() + }) + const liveAt = (generation: number) => + async (expectedSelector: AdmissionSelector) => ({ + ...sample(), + selector: { ...selector, generation }, + expectedSelector + }) + // One predecessor roll advanced the live selector by exactly 2. + await expect(runIncidentLivePreflight( + ['--state-file', agedPath, '--wave-index', '1'], + { now: () => now, collect: liveAt(selector.generation + 2) } + )).resolves.toBeUndefined() + // The sealed pre-roll generation must no longer satisfy wave 1. + await expect(runIncidentLivePreflight( + ['--state-file', agedPath, '--wave-index', '1'], + { now: () => now, collect: liveAt(selector.generation) } + )).rejects.toThrow('director-admin/selector_mismatch') + // Wave 0 still expects the sealed generation itself. + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: liveAt(selector.generation) } + )).resolves.toBeUndefined() + }) + + it('fails closed on a live threshold breach', async () => { + const unhealthy = sample() + unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: async () => unhealthy } + )).rejects.toThrow('cloud-monitoring/threshold_max') + }) + + it('enforces the signed migration policy', async () => { + const inactiveTarget = sample() + inactiveTarget.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_target_inactive' + ]!.value = 30 + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: async () => inactiveTarget } + )).rejects.toThrow('director-admin/threshold_max') + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('recover-forward')], + { now: () => now, collect: async () => inactiveTarget } + )).resolves.toBeUndefined() + inactiveTarget.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_blocked' + ]!.value = 1 + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('recover-forward')], + { now: () => now, collect: async () => inactiveTarget } + )).rejects.toThrow('director-admin/threshold_max') + }) + + it('binds capacity-transition evidence to its general cell', async () => { + const capacitySample = sample() + const capacitySelector = { + generation: 1, + membership: { + existingOnly: [], + migrationOnly: [], + general: ['production-gce-c1'] + } + } + capacitySample.selector = capacitySelector + capacitySample.expectedSelector = capacitySelector + capacitySample.cells[0]!.expectedAdmissionState = 'general' + capacitySample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.admission_state' + ]!.value = 2 + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('capacity-transition')], + { now: () => now, collect: async () => capacitySample } + )).resolves.toBeUndefined() + capacitySample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_target_inactive' + ]!.value = 1 + await expect(runIncidentLivePreflight( + ['--state-file', stateFile('capacity-transition')], + { now: () => now, collect: async () => capacitySample } + )).rejects.toThrow('director-admin/threshold_max') + }) + + it('rejects stale live evidence', async () => { + const stale = sample() + stale.sources['active-probe']!.observedAt = new Date(now - 60_001).toISOString() + await expect(runIncidentLivePreflight( + ['--state-file', stateFile()], + { now: () => now, collect: async () => stale } + )).rejects.toThrow('active-probe/source_stale') + }) + + it('retries freshness-only failures when explicitly requested', async () => { + const stale = sample() + stale.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = + new Date(now - 180_001).toISOString() + const missing = sample() + delete missing.sources['relay-logs'] + const collect = vi.fn() + .mockResolvedValueOnce(stale) + .mockResolvedValueOnce(missing) + .mockResolvedValueOnce(sample()) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--retry-freshness'], + { now: () => now, collect, wait } + )).resolves.toBeUndefined() + expect(collect).toHaveBeenCalledTimes(3) + expect(wait).toHaveBeenCalledTimes(2) + expect(wait).toHaveBeenNthCalledWith(1, 15_000) + expect(wait).toHaveBeenNthCalledWith(2, 15_000) + }) + + it('does not retry a threshold failure', async () => { + const unhealthy = sample() + unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.value = 0.9 + unhealthy.sources['cloud-monitoring']!.signals['cloud_sql.cpu']!.observedAt = + new Date(now - 180_001).toISOString() + const collect = vi.fn(async () => unhealthy) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--retry-freshness'], + { now: () => now, collect, wait } + )).rejects.toThrow('cloud-monitoring/threshold_max') + expect(collect).toHaveBeenCalledOnce() + expect(wait).not.toHaveBeenCalled() + }) + + it('fails closed after the bounded freshness retry window', async () => { + const stale = sample() + stale.sources['cloud-monitoring']!.observedAt = new Date(now - 180_001).toISOString() + const collect = vi.fn(async () => stale) + const wait = vi.fn(async () => undefined) + await expect(runIncidentLivePreflight( + ['--state-file', stateFile(), '--retry-freshness'], + { now: () => now, collect, wait } + )).rejects.toThrow('cloud-monitoring/source_stale') + expect(collect).toHaveBeenCalledTimes(5) + expect(wait).toHaveBeenCalledTimes(4) + }) + + it('uses the supplied admin token without minting through gcloud', async () => { + const identityToken = vi.fn(async () => 'minted.token.value') + const gcloud = livePreflightGcloud( + { accessToken: async () => 'access-token', identityToken }, + { ORCA_RELAY_ADMIN_ID_TOKEN: 'supplied.token.value' } + ) + await expect(gcloud.identityToken!('audience')).resolves.toBe( + 'supplied.token.value' + ) + expect(identityToken).not.toHaveBeenCalled() + }) +}) diff --git a/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts new file mode 100644 index 00000000000..e82627a3e80 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-live-preflight-cli.ts @@ -0,0 +1,195 @@ +import { readFile } from 'node:fs/promises' +import { resolve } from 'node:path' +import { pathToFileURL } from 'node:url' +import { z } from 'zod' +import { createGcloudClient } from './gcloud-client.js' +import { suppliedIdentityToken } from './incident-monitor-cli.js' +import { AdmissionSelectorSchema, type AdmissionSelector } from './incident-selector.js' +import { + evaluateIncidentSample, + preDrainDryRunPassed, + type IncidentSample +} from './incident-monitor.js' +import { createIncidentSampleCollector } from './incident-monitor-sources.js' + +const FRESHNESS_RETRY_ATTEMPTS = 5 +const FRESHNESS_RETRY_INTERVAL_MS = 15_000 +const MONITOR_EVIDENCE_MAX_AGE_MS = 5 * 60_000 +// Matches the same-cap cell job timeout-minutes; bounds each predecessor wave. +const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000 +const WAVE_INDEX_PATTERN = /^[0-3]$/ +const FRESHNESS_FAILURE_CODES = new Set([ + 'signal_missing', + 'signal_stale', + 'source_missing', + 'source_stale' +]) + +export function livePreflightGcloud( + gcloud: ReturnType, + environment: NodeJS.ProcessEnv = process.env +): ReturnType { + const token = suppliedIdentityToken(environment.ORCA_RELAY_ADMIN_ID_TOKEN) + return token ? { ...gcloud, identityToken: async () => token } : gcloud +} + +const PreflightStateSchema = z.object({ + schemaVersion: z.literal(4), + environment: z.literal('production'), + expectedSelector: AdmissionSelectorSchema, + migrationPolicy: z.enum(['strict', 'recover-forward', 'capacity-transition']), + recoverySourceCellId: z.string().nullable(), + capacityCellId: z.string().nullable(), + preDrainDryRun: z.literal(true), + startedAt: z.string(), + windowStartedAt: z.string(), + durationMinutes: z.literal(15), + intervalMs: z.literal(60_000), + sampleCount: z.number().int().min(16), + lastSampleAt: z.string(), + frozenAt: z.null(), + completedAt: z.string() +}).superRefine((state, context) => { + const validRecovery = + state.migrationPolicy === 'recover-forward' && + state.capacityCellId === null && + state.recoverySourceCellId !== null && + state.expectedSelector.membership.existingOnly.includes( + state.recoverySourceCellId + ) + const validStrict = + state.migrationPolicy === 'strict' && + state.recoverySourceCellId === null && + state.capacityCellId === null + const validCapacity = + state.migrationPolicy === 'capacity-transition' && + state.recoverySourceCellId === null && + state.capacityCellId !== null && + state.expectedSelector.membership.general.includes(state.capacityCellId) + if (!validRecovery && !validStrict && !validCapacity) { + context.addIssue({ + code: 'custom', + message: 'relay live preflight migration policy is invalid' + }) + } +}) + +export async function runIncidentLivePreflight( + argv: string[], + dependencies: { + now?: () => number + wait?: (ms: number) => Promise + collect?: (expectedSelector: AdmissionSelector) => Promise + gcloud?: ReturnType + environment?: NodeJS.ProcessEnv + } = {} +): Promise { + const args = argv[0] === '--' ? argv.slice(1) : argv + const freshnessRetryCount = args.filter((arg) => arg === '--retry-freshness').length + const rest = args.filter((arg) => arg !== '--retry-freshness') + const stateArgs: string[] = [] + let waveIndex = '0' + let waveIndexCount = 0 + for (let index = 0; index < rest.length; index += 1) { + if (rest[index] === '--wave-index') { + waveIndexCount += 1 + waveIndex = rest[index + 1] ?? '' + index += 1 + } else { + stateArgs.push(rest[index] as string) + } + } + if ( + freshnessRetryCount > 1 || + waveIndexCount > 1 || + !WAVE_INDEX_PATTERN.test(waveIndex) || + stateArgs.length !== 2 || + stateArgs[0] !== '--state-file' || + !stateArgs[1] + ) { + throw new Error( + 'usage: --state-file [--wave-index <0-3>] [--retry-freshness]' + ) + } + const state = PreflightStateSchema.parse( + JSON.parse(await readFile(resolve(stateArgs[1]), 'utf8')) + ) + const now = dependencies.now ?? Date.now + const completedAt = Date.parse(state.completedAt) + const windowStartedAt = Date.parse(state.windowStartedAt) + const lastSampleAt = Date.parse(state.lastSampleAt) + const evidenceAgeMs = now() - completedAt + // Later same-cap waves start after sequential predecessor cell rolls, so the + // freshness bound grows by one cell-job timeout per predecessor; the live + // samples collected below still hold every wave to current health. + const maxEvidenceAgeMs = + MONITOR_EVIDENCE_MAX_AGE_MS + Number(waveIndex) * WAVE_PREDECESSOR_TIMEOUT_MS + if ( + !preDrainDryRunPassed(state) || + !Number.isFinite(windowStartedAt) || + completedAt - windowStartedAt < 15 * 60_000 || + !Number.isFinite(lastSampleAt) || + lastSampleAt > completedAt || + completedAt - lastSampleAt > state.intervalMs || + !Number.isFinite(completedAt) || + evidenceAgeMs < 0 || + evidenceAgeMs > maxEvidenceAgeMs + ) { + throw new Error('relay live preflight monitor evidence is incomplete or stale') + } + const gcloud = livePreflightGcloud( + dependencies.gcloud ?? createGcloudClient(), + dependencies.environment + ) + // Each predecessor same-cap apply wave reversibly isolates and restores its + // cell, advancing the selector generation by exactly 2 with membership + // unchanged (rollback is single-cell, so it never reaches a later wave), so + // the live selector comparison must expect the wave-adjusted generation. + const collectOptions = { + environment: state.environment, + expectedSelector: { + ...state.expectedSelector, + generation: state.expectedSelector.generation + 2 * Number(waveIndex) + }, + ...(dependencies.now ? { now: dependencies.now } : {}) + } + const injected = dependencies.collect + const collect = injected + ? () => injected(collectOptions.expectedSelector) + : createIncidentSampleCollector(gcloud, collectOptions) + const wait = dependencies.wait ?? ((ms: number) => new Promise((resolveWait) => { + setTimeout(resolveWait, ms) + })) + const attempts = freshnessRetryCount === 1 ? FRESHNESS_RETRY_ATTEMPTS : 1 + for (let attempt = 1; attempt <= attempts; attempt++) { + const evaluation = evaluateIncidentSample( + await collect(), + now(), + state.migrationPolicy, + state.recoverySourceCellId, + state.capacityCellId + ) + if (evaluation.status === 'green') return + const freshnessOnly = evaluation.failures.every((failure) => + FRESHNESS_FAILURE_CODES.has(failure.code) + ) + if (!freshnessOnly || attempt === attempts) { + throw new Error( + `relay live preflight failed: ${evaluation.failures + .map((failure) => `${failure.source}/${failure.code}`) + .join(',')}` + ) + } + console.warn( + `relay live preflight awaiting fresh evidence (${attempt}/${attempts - 1})` + ) + await wait(FRESHNESS_RETRY_INTERVAL_MS) + } +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + runIncidentLivePreflight(process.argv.slice(2)).catch((error: unknown) => { + console.error(error instanceof Error ? error.message : 'relay live preflight failed') + process.exitCode = 1 + }) +} diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts new file mode 100644 index 00000000000..3e1f20a3cbb --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.test.ts @@ -0,0 +1,454 @@ +import { mkdtempSync, readFileSync, rmSync, statSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { + parseIncidentMonitorArguments, + runIncidentMonitorCli +} from './incident-monitor-cli.js' +import type { IncidentSample } from './incident-monitor.js' +import { RELAY_OPS_ENVIRONMENTS } from './environment-config.js' + +const directories: string[] = [] +const startedAt = Date.parse('2026-07-28T00:00:00.000Z') +const productionCells = RELAY_OPS_ENVIRONMENTS.production.cells.map((cell) => cell.cellId) +const selector = { + generation: 1, + membership: { + existingOnly: productionCells.slice(1), + migrationOnly: [], + general: [productionCells[0]!] + } +} +const selectorArguments = [ + '--expected-selector-generation', + '1', + '--expected-existing-only-cells', + productionCells.slice(1).join(','), + '--expected-migration-only-cells', + 'none', + '--expected-general-cells', + productionCells[0]! +] +const signal = (value: number, at: number) => ({ + value, + observedAt: new Date(at).toISOString() +}) + +function sample(at: number): IncidentSample { + const observedAt = new Date(at).toISOString() + const cellId = 'production-gce-c1' + return { + collectedAt: observedAt, + selector, + expectedSelector: selector, + cells: [{ + cellId, + runtimeKnown: true, + powered: true, + expectedAdmissionState: 'general' + }], + sources: { + 'active-probe': { + observedAt, + signals: { + 'director.health': signal(1, at), + 'director.ready': signal(1, at), + 'director.latency_ms': signal(1, at), + 'auth.health': signal(1, at), + 'auth.ready': signal(1, at), + 'auth.latency_ms': signal(1, at), + [`cell.${cellId}.health`]: signal(1, at), + [`cell.${cellId}.ready`]: signal(1, at), + [`cell.${cellId}.latency_ms`]: signal(1, at) + } + }, + 'cloud-monitoring': { + observedAt, + signals: { + 'cloud_sql.cpu': signal(0.1, at), + 'cloud_sql.memory': signal(0.1, at), + 'cloud_sql.backends': signal(1, at), + 'cloud_sql.lock_waits': signal(0, at), + 'cloud_sql.deadlocks': signal(0, at), + 'director.instances': signal(5, at), + 'director.cpu': signal(0.1, at), + 'director.memory': signal(0.1, at), + 'director.concurrency': signal(1, at), + 'director.errors': signal(0, at), + 'auth.errors': signal(0, at) + } + }, + 'relay-logs': { + observedAt, + signals: { + 'relay.pool_waiting': signal(0, at), + 'relay.pool_wait_ms': signal(0, at), + 'relay.postgres_retries': signal(0, at), + 'relay.postgres_retry_exhausted': signal(0, at), + [`cell.${cellId}.connections`]: signal(1, at), + [`cell.${cellId}.queued_bytes`]: signal(0, at) + } + }, + 'director-admin': { + observedAt, + signals: { + [`cell.${cellId}.admission_state`]: signal(2, at), + [`cell.${cellId}.heartbeat_fresh`]: signal(1, at), + [`cell.${cellId}.heartbeat_age_ms`]: signal(1, at), + [`cell.${cellId}.migration_blocked`]: signal(0, at), + [`cell.${cellId}.migration_target_inactive`]: signal(0, at) + } + } + } + } +} + +afterEach(() => { + for (const directory of directories.splice(0)) { + rmSync(directory, { recursive: true, force: true }) + } +}) + +describe('incident monitor CLI', () => { + it('requires an exact selector and rejects invalid membership', () => { + expect(() => parseIncidentMonitorArguments([])).toThrow( + '--expected-selector-generation' + ) + expect(() => + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + '--expected-selector-generation', + '1', + '--expected-existing-only-cells', + productionCells.slice(1).join(','), + '--expected-migration-only-cells', + 'none', + '--expected-general-cells', + 'production-gce-c99' + ]) + ).toThrow('every configured cell exactly once') + expect(() => + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--interval-seconds', + '61' + ]) + ).toThrow('between 1 and 60') + expect(() => + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--duration-minutes', + '14' + ]) + ).toThrow('between 15 and 90') + expect(() => + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + '--expected-selector-generation', + '0', + '--expected-existing-only-cells', + productionCells.slice(1).join(','), + '--expected-migration-only-cells', + productionCells[0]!, + '--expected-general-cells', + 'none' + ]) + ).toThrow('generation 0 cannot represent migration-only') + expect(() => + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--migration-policy', + 'recover-forward', + '--pre-drain-dry-run' + ]) + ).toThrow('requires an existing-only --recovery-source-cell-id') + expect(() => + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--migration-policy', + 'capacity-transition', + '--pre-drain-dry-run' + ]) + ).toThrow('requires a general --capacity-cell-id') + expect( + parseIncidentMonitorArguments([ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--migration-policy', + 'capacity-transition', + '--capacity-cell-id', + productionCells[0]!, + '--pre-drain-dry-run' + ]) + ).toMatchObject({ + migrationPolicy: 'capacity-transition', + capacityCellId: productionCells[0]!, + recoverySourceCellId: null + }) + }) + + it('writes private durable checkpoints for a green pre-drain dry run', async () => { + const directory = mkdtempSync(join(tmpdir(), 'relay-incident-cli-')) + directories.push(directory) + let now = startedAt + const output: string[] = [] + const code = await runIncidentMonitorCli( + [ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--pre-drain-dry-run', + '--output-directory', + directory + ], + { + cwd: directory, + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => sample(now), + writeOutput: (value) => output.push(value) + } + ) + expect(code).toBe(0) + const statePath = join(directory, 'incident-1.state.json') + const summaryPath = join(directory, 'incident-1.summaries.jsonl') + const markdownPath = join(directory, 'incident-1.summary.md') + expect(statSync(statePath).mode & 0o077).toBe(0) + expect(statSync(summaryPath).mode & 0o077).toBe(0) + expect(statSync(markdownPath).mode & 0o077).toBe(0) + const summaries = readFileSync(summaryPath, 'utf8') + .trim() + .split('\n') + .map((line) => JSON.parse(line)) + expect(summaries.map((entry) => entry.checkpointMinute)).toEqual([0, 5, 15]) + expect(output.join('')).not.toContain('token') + expect(readFileSync(markdownPath, 'utf8')).toContain( + '| 0 | 15 | green | 16 | none |' + ) + expect(JSON.parse(readFileSync(statePath, 'utf8'))).toMatchObject({ + completedAt: new Date(startedAt + 15 * 60_000).toISOString(), + frozenAt: null, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + sampleCount: 16 + }) + }) + + it('runs a recovery dry run without masking blocked migrations', async () => { + const directory = mkdtempSync(join(tmpdir(), 'relay-incident-cli-')) + directories.push(directory) + let now = startedAt + const recoverySelector = { + generation: 1, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: [], + general: productionCells.slice(1) + } + } + const recoverySample = (): IncidentSample => { + const current = sample(now) + current.selector = recoverySelector + current.expectedSelector = recoverySelector + current.cells[0]!.expectedAdmissionState = 'existing-only' + current.sources['director-admin']!.signals[ + 'cell.production-gce-c1.admission_state' + ] = signal(0, now) + current.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_target_inactive' + ] = signal(30, now) + return current + } + const args = [ + '--incident-id', + 'incident-1', + '--expected-selector-generation', + '1', + '--expected-existing-only-cells', + 'production-gce-c1', + '--expected-migration-only-cells', + 'none', + '--expected-general-cells', + productionCells.slice(1).join(','), + '--pre-drain-dry-run', + '--migration-policy', + 'recover-forward', + '--recovery-source-cell-id', + 'production-gce-c1', + '--output-directory', + directory + ] + const dependencies = { + cwd: directory, + now: () => now, + wait: async (ms: number) => { + now += ms + }, + collect: async () => recoverySample(), + writeOutput: () => {} + } + await expect(runIncidentMonitorCli(args, dependencies)).resolves.toBe(0) + expect( + JSON.parse(readFileSync(join(directory, 'incident-1.state.json'), 'utf8')) + ).toMatchObject({ + migrationPolicy: 'recover-forward', + recoverySourceCellId: 'production-gce-c1', + capacityCellId: null, + frozenAt: null + }) + }) + + it('fails a frozen pre-drain gate after its first sample', async () => { + const directory = mkdtempSync(join(tmpdir(), 'relay-incident-cli-')) + directories.push(directory) + let now = startedAt + let collections = 0 + let waits = 0 + const code = await runIncidentMonitorCli( + [ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--pre-drain-dry-run', + '--output-directory', + directory + ], + { + cwd: directory, + now: () => now, + wait: async (ms) => { + waits++ + now += ms + }, + collect: async () => { + collections++ + const unhealthy = sample(now) + unhealthy.sources['relay-logs']!.signals['relay.pool_waiting'] = + signal(801, now) + return unhealthy + }, + writeOutput: () => {} + } + ) + expect(code).toBe(2) + expect(collections).toBe(1) + expect(waits).toBe(0) + expect( + JSON.parse(readFileSync(join(directory, 'incident-1.state.json'), 'utf8')) + ).toMatchObject({ + completedAt: new Date(startedAt).toISOString(), + frozenAt: new Date(startedAt).toISOString(), + sampleCount: 1 + }) + }) + + it('requires --restart and preserves a latched freeze', async () => { + const directory = mkdtempSync(join(tmpdir(), 'relay-incident-cli-')) + directories.push(directory) + let now = startedAt + const args = [ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--duration-minutes', + '15', + '--output-directory', + directory + ] + await runIncidentMonitorCli(args, { + cwd: directory, + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + const unhealthy = sample(now) + unhealthy.sources['relay-logs']!.signals['relay.pool_waiting'] = signal(801, now) + return unhealthy + }, + writeOutput: () => {} + }) + await expect( + runIncidentMonitorCli(args, { + cwd: directory, + now: () => now, + collect: async () => sample(now), + writeOutput: () => {} + }) + ).rejects.toThrow('pass --restart') + const changedAdmission = [...args] + changedAdmission[changedAdmission.indexOf('--expected-selector-generation') + 1] = '2' + await expect( + runIncidentMonitorCli([...changedAdmission, '--restart'], { + cwd: directory, + now: () => now, + collect: async () => sample(now), + writeOutput: () => {} + }) + ).rejects.toThrow('do not match') + await expect( + runIncidentMonitorCli([...args, '--restart'], { + cwd: directory, + now: () => now, + collect: async () => sample(now), + writeOutput: () => {} + }) + ).resolves.toBe(2) + }) + + it('resumes a gracefully segmented monitor without resetting continuity', async () => { + const directory = mkdtempSync(join(tmpdir(), 'relay-incident-cli-')) + directories.push(directory) + let now = startedAt + const args = [ + '--incident-id', + 'incident-1', + ...selectorArguments, + '--duration-minutes', + '15', + '--output-directory', + directory + ] + const dependencies = { + cwd: directory, + now: () => now, + wait: async (ms: number) => { + now += ms + }, + collect: async () => sample(now), + writeOutput: () => {} + } + await expect( + runIncidentMonitorCli([...args, '--max-samples-this-run', '2'], dependencies) + ).resolves.toBe(0) + const statePath = join(directory, 'incident-1.state.json') + expect(JSON.parse(readFileSync(statePath, 'utf8'))).toMatchObject({ + completedAt: null, + sampleCount: 2, + lastSampleAt: new Date(startedAt + 60_000).toISOString() + }) + await expect( + runIncidentMonitorCli([...args, '--restart'], dependencies) + ).resolves.toBe(0) + expect(JSON.parse(readFileSync(statePath, 'utf8'))).toMatchObject({ + completedAt: new Date(startedAt + 15 * 60_000).toISOString(), + windowSequence: 0, + sampleCount: 17 + }) + }) +}) diff --git a/cloud/apps/relay-ops/src/incident-monitor-cli.ts b/cloud/apps/relay-ops/src/incident-monitor-cli.ts new file mode 100644 index 00000000000..e090be7ea58 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-monitor-cli.ts @@ -0,0 +1,493 @@ +import { randomUUID } from 'node:crypto' +import { + appendFile, + chmod, + mkdir, + open, + readFile, + rename, + stat, + writeFile +} from 'node:fs/promises' +import { dirname, resolve } from 'node:path' +import { pathToFileURL } from 'node:url' +import { z } from 'zod' +import { relayOpsEnvironment } from './environment-config.js' +import { createGcloudClient } from './gcloud-client.js' +import { + AdmissionSelectorSchema, + normalizeSelectorMembership, + type AdmissionSelector +} from './incident-selector.js' +import { + initialIncidentMonitorState, + preDrainDryRunPassed, + runIncidentMonitor, + type IncidentCheckpoint, + type IncidentSample, + type IncidentMonitorState +} from './incident-monitor.js' +import { createIncidentSampleCollector } from './incident-monitor-sources.js' + +const StateSchema = z.object({ + schemaVersion: z.literal(4), + incidentId: z.string(), + environment: z.enum(['production', 'staging']), + expectedSelector: AdmissionSelectorSchema, + preDrainDryRun: z.boolean(), + migrationPolicy: z.enum(['strict', 'recover-forward', 'capacity-transition']), + recoverySourceCellId: z.string().nullable(), + capacityCellId: z.string().nullable(), + startedAt: z.string(), + windowStartedAt: z.string().nullable(), + windowSequence: z.number().int().nonnegative(), + durationMinutes: z.number().int(), + intervalMs: z.number().int(), + nextCheckpointIndex: z.number().int().nonnegative(), + sampleCount: z.number().int().nonnegative(), + totalSampleCount: z.number().int().nonnegative(), + lastSampleAt: z.string().nullable(), + continuityEvents: z.array(z.object({ + recordedAt: z.string(), + windowSequence: z.number().int().nonnegative(), + failures: z.array(z.object({ + code: z.string(), + source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']), + signal: z.string().optional(), + observed: z.number().optional(), + threshold: z.number().optional() + })) + })), + frozenAt: z.string().nullable(), + failures: z.array(z.object({ + code: z.string(), + source: z.enum(['active-probe', 'cloud-monitoring', 'relay-logs', 'director-admin']), + signal: z.string().optional(), + observed: z.number().optional(), + threshold: z.number().optional() + })), + completedAt: z.string().nullable() +}) + +type CliOptions = { + environment: 'production' | 'staging' + incidentId: string + durationMinutes: number + intervalMs: number + expectedSelector: AdmissionSelector + stateFile: string + summaryFile: string + markdownFile: string + restart: boolean + preDrainDryRun: boolean + migrationPolicy: 'strict' | 'recover-forward' | 'capacity-transition' + recoverySourceCellId: string | null + capacityCellId: string | null + maxSamplesThisRun: number | null +} + +function parsePositiveInteger(value: string | undefined, name: string): number { + const parsed = Number(value) + if (!Number.isSafeInteger(parsed) || parsed <= 0) { + throw new Error(`${name} must be a positive integer`) + } + return parsed +} + +export function parseIncidentMonitorArguments(argv: string[], cwd = process.cwd()): CliOptions { + const flags = new Set(['restart', 'pre-drain-dry-run']) + const valueArguments = new Set([ + 'duration-minutes', + 'environment', + 'expected-selector-generation', + 'expected-existing-only-cells', + 'expected-migration-only-cells', + 'expected-general-cells', + 'incident-id', + 'interval-seconds', + 'max-samples-this-run', + 'migration-policy', + 'output-directory', + 'recovery-source-cell-id', + 'capacity-cell-id' + ]) + const values: Record = {} + const enabledFlags = new Set() + for (let index = 0; index < argv.length; index++) { + const argument = argv[index] + if (!argument?.startsWith('--')) throw new Error(`invalid argument ${argument ?? ''}`) + const name = argument.slice(2) + if (flags.has(name)) { + enabledFlags.add(name) + continue + } + if (!valueArguments.has(name)) throw new Error(`unknown argument --${name}`) + const value = argv[++index] + if (!value || value.startsWith('--')) throw new Error(`missing --${name} value`) + values[name] = value + } + const environment = z.enum(['production', 'staging']).parse( + values.environment ?? 'production' + ) + const incidentId = values['incident-id'] ?? randomUUID() + if (!/^[a-zA-Z0-9][a-zA-Z0-9._-]{7,127}$/.test(incidentId)) { + throw new Error('--incident-id must be 8-128 safe characters') + } + const selectorGeneration = Number(values['expected-selector-generation']) + if (!Number.isSafeInteger(selectorGeneration) || selectorGeneration < 0) { + throw new Error('--expected-selector-generation must be a nonnegative integer') + } + const configuredCells = new Set( + relayOpsEnvironment(environment).cells.map((cell) => cell.cellId) + ) + const cellList = (name: string): string[] => { + const value = values[name] + if (value === undefined) throw new Error(`--${name} is required; use none for an empty set`) + return value === 'none' ? [] : value.split(',').map((cellId) => cellId.trim()) + } + const expectedSelector = { + generation: selectorGeneration, + membership: normalizeSelectorMembership( + { + existingOnly: cellList('expected-existing-only-cells'), + migrationOnly: cellList('expected-migration-only-cells'), + general: cellList('expected-general-cells') + }, + configuredCells + ) + } + if (selectorGeneration === 0 && expectedSelector.membership.migrationOnly.length > 0) { + throw new Error('generation 0 cannot represent migration-only admission') + } + const preDrainDryRun = enabledFlags.has('pre-drain-dry-run') + const migrationPolicy = z.enum([ + 'strict', + 'recover-forward', + 'capacity-transition' + ]).parse( + values['migration-policy'] ?? 'strict' + ) + const recoverySourceCellId = + values['recovery-source-cell-id'] === undefined || + values['recovery-source-cell-id'] === 'none' + ? null + : values['recovery-source-cell-id'] + const capacityCellId = + values['capacity-cell-id'] === undefined || values['capacity-cell-id'] === 'none' + ? null + : values['capacity-cell-id'] + const durationMinutes = parsePositiveInteger( + values['duration-minutes'] ?? (preDrainDryRun ? '15' : '90'), + '--duration-minutes' + ) + const intervalMs = + parsePositiveInteger(values['interval-seconds'] ?? '60', '--interval-seconds') * 1_000 + if (durationMinutes < 15 || durationMinutes > 90) { + throw new Error('--duration-minutes must be between 15 and 90') + } + if (intervalMs > 60_000) { + throw new Error('--interval-seconds must be between 1 and 60') + } + if (preDrainDryRun && durationMinutes !== 15) { + throw new Error('--pre-drain-dry-run requires --duration-minutes 15') + } + if (migrationPolicy !== 'strict' && !preDrainDryRun) { + throw new Error(`--migration-policy ${migrationPolicy} requires --pre-drain-dry-run`) + } + if ( + migrationPolicy === 'recover-forward' && + ( + recoverySourceCellId === null || + !configuredCells.has(recoverySourceCellId) || + !expectedSelector.membership.existingOnly.includes(recoverySourceCellId) + ) + ) { + throw new Error( + '--migration-policy recover-forward requires an existing-only --recovery-source-cell-id' + ) + } + if (migrationPolicy !== 'recover-forward' && recoverySourceCellId !== null) { + throw new Error('--recovery-source-cell-id requires --migration-policy recover-forward') + } + if ( + migrationPolicy === 'capacity-transition' && + ( + capacityCellId === null || + !configuredCells.has(capacityCellId) || + !expectedSelector.membership.general.includes(capacityCellId) + ) + ) { + throw new Error( + '--migration-policy capacity-transition requires a general --capacity-cell-id' + ) + } + if (migrationPolicy !== 'capacity-transition' && capacityCellId !== null) { + throw new Error('--capacity-cell-id requires --migration-policy capacity-transition') + } + const directory = resolve(cwd, values['output-directory'] ?? '.relay-incidents') + const maxSamplesThisRun = values['max-samples-this-run'] + ? parsePositiveInteger(values['max-samples-this-run'], '--max-samples-this-run') + : null + return { + environment, + incidentId, + durationMinutes, + intervalMs, + expectedSelector, + stateFile: resolve(directory, `${incidentId}.state.json`), + summaryFile: resolve(directory, `${incidentId}.summaries.jsonl`), + markdownFile: resolve(directory, `${incidentId}.summary.md`), + restart: enabledFlags.has('restart'), + preDrainDryRun, + migrationPolicy, + recoverySourceCellId, + capacityCellId, + maxSamplesThisRun + } +} + +async function fileExists(path: string): Promise { + try { + await stat(path) + return true + } catch { + return false + } +} + +async function syncFile(path: string): Promise { + const handle = await open(path, 'r') + try { + await handle.sync() + } finally { + await handle.close() + } +} + +async function persistState(path: string, state: IncidentMonitorState): Promise { + await mkdir(dirname(path), { recursive: true, mode: 0o700 }) + await chmod(dirname(path), 0o700) + const temporaryPath = `${path}.tmp` + await writeFile(temporaryPath, `${JSON.stringify(state)}\n`, { mode: 0o600 }) + await chmod(temporaryPath, 0o600) + await syncFile(temporaryPath) + await rename(temporaryPath, path) + await syncFile(path) +} + +async function appendCheckpoint(path: string, checkpoint: IncidentCheckpoint): Promise { + await mkdir(dirname(path), { recursive: true, mode: 0o700 }) + await chmod(dirname(path), 0o700) + if (await fileExists(path)) { + const existing = (await readFile(path, 'utf8')) + .trim() + .split('\n') + .filter(Boolean) + .map((line) => JSON.parse(line) as IncidentCheckpoint) + if ( + existing.some((entry) => + entry.windowSequence === checkpoint.windowSequence && + entry.checkpointMinute === checkpoint.checkpointMinute + ) + ) return + } + await appendFile(path, `${JSON.stringify(checkpoint)}\n`, { mode: 0o600 }) + await chmod(path, 0o600) + await syncFile(path) +} + +function markdownFailure(checkpoint: IncidentCheckpoint): string { + if (checkpoint.failures.length === 0) return 'none' + return checkpoint.failures + .map((failure) => { + const signal = failure.signal ? `/${failure.signal}` : '' + return `${failure.source}/${failure.code}${signal}` + }) + .join(', ') +} + +async function writeMarkdownSummary( + path: string, + summaryPath: string, + incidentId: string, + environment: string +): Promise { + const checkpoints = (await readFile(summaryPath, 'utf8')) + .trim() + .split('\n') + .filter(Boolean) + .map((line) => JSON.parse(line) as IncidentCheckpoint) + const rows = checkpoints.map((checkpoint) => [ + `| ${checkpoint.windowSequence}`, + checkpoint.checkpointMinute, + checkpoint.status, + checkpoint.sampleCount, + `${markdownFailure(checkpoint)} |` + ].join(' | ')) + const markdown = [ + '# Relay incident monitor', + '', + `Incident: \`${incidentId}\``, + '', + `Environment: \`${environment}\``, + '', + `Expected selector: \`${JSON.stringify(checkpoints[0]?.expectedSelector ?? null)}\``, + '', + `Migration policy: \`${checkpoints[0]?.migrationPolicy ?? 'unknown'}\``, + '', + `Recovery source: \`${checkpoints[0]?.recoverySourceCellId ?? 'none'}\``, + '', + `Capacity cell: \`${checkpoints[0]?.capacityCellId ?? 'none'}\``, + '', + '| Window | Minute | Status | Samples | Failures |', + '| ---: | ---: | --- | ---: | --- |', + ...rows, + '' + ].join('\n') + const temporaryPath = `${path}.tmp` + await writeFile(temporaryPath, markdown, { mode: 0o600 }) + await chmod(temporaryPath, 0o600) + await syncFile(temporaryPath) + await rename(temporaryPath, path) + await syncFile(path) +} + +export function suppliedIdentityToken(value: string | undefined): string | null { + if (value === undefined) return null + if ( + value.length > 8_192 || + !/^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/.test(value) + ) { + throw new Error('ORCA_RELAY_ADMIN_ID_TOKEN is invalid') + } + return value +} + +async function readInitialState( + options: CliOptions, + now: () => number = Date.now +): Promise { + const exists = await fileExists(options.stateFile) + if (exists && !options.restart) { + throw new Error('incident state already exists; pass --restart to resume it') + } + if (!exists && options.restart) throw new Error('no incident state exists to restart') + if (!exists) { + return initialIncidentMonitorState({ + incidentId: options.incidentId, + environment: options.environment, + expectedSelector: options.expectedSelector, + preDrainDryRun: options.preDrainDryRun, + migrationPolicy: options.migrationPolicy, + recoverySourceCellId: options.recoverySourceCellId, + capacityCellId: options.capacityCellId, + startedAt: new Date(now()).toISOString(), + durationMinutes: options.durationMinutes, + intervalMs: options.intervalMs + }) + } + const state = StateSchema.parse( + JSON.parse(await readFile(options.stateFile, 'utf8')) + ) as IncidentMonitorState + if ( + state.incidentId !== options.incidentId || + state.environment !== options.environment || + JSON.stringify(state.expectedSelector) !== JSON.stringify(options.expectedSelector) || + state.preDrainDryRun !== options.preDrainDryRun || + state.migrationPolicy !== options.migrationPolicy || + state.recoverySourceCellId !== options.recoverySourceCellId || + state.capacityCellId !== options.capacityCellId || + state.durationMinutes !== options.durationMinutes || + state.intervalMs !== options.intervalMs + ) { + throw new Error('restart arguments do not match durable incident state') + } + return state +} + +export async function runIncidentMonitorCli( + argv: string[], + dependencies: { + cwd?: string + now?: () => number + wait?: (ms: number) => Promise + gcloud?: ReturnType + collect?: () => Promise + writeOutput?: (value: string) => void + environment?: NodeJS.ProcessEnv + } = {} +): Promise { + const options = parseIncidentMonitorArguments(argv, dependencies.cwd) + const state = await readInitialState(options, dependencies.now) + const baseGcloud = dependencies.gcloud ?? createGcloudClient() + const token = suppliedIdentityToken( + (dependencies.environment ?? process.env).ORCA_RELAY_ADMIN_ID_TOKEN + ) + await persistState(options.stateFile, state) + const gcloud = token + ? { ...baseGcloud, identityToken: async () => token } + : baseGcloud + const collect = + dependencies.collect ?? + createIncidentSampleCollector(gcloud, { + environment: options.environment, + expectedSelector: options.expectedSelector, + ...(dependencies.now ? { now: dependencies.now } : {}) + }) + let samplesThisRun = 0 + const segmentedCollect = async (): Promise => { + try { + return await collect() + } finally { + samplesThisRun++ + } + } + const wait = + dependencies.wait ?? + ((ms: number) => new Promise((resolvePromise) => setTimeout(resolvePromise, ms))) + const segmentComplete = Symbol('segment-complete') + const output = dependencies.writeOutput ?? ((value) => process.stdout.write(`${value}\n`)) + let result: IncidentMonitorState + try { + result = await runIncidentMonitor(state, { + now: dependencies.now ?? Date.now, + wait: async (ms) => { + if ( + options.maxSamplesThisRun !== null && + samplesThisRun >= options.maxSamplesThisRun + ) { + throw segmentComplete + } + await wait(ms) + }, + collect: segmentedCollect, + persist: async (nextState) => await persistState(options.stateFile, nextState), + checkpoint: async (checkpoint) => { + await appendCheckpoint(options.summaryFile, checkpoint) + await writeMarkdownSummary( + options.markdownFile, + options.summaryFile, + options.incidentId, + options.environment + ) + output(JSON.stringify(checkpoint)) + } + }) + } catch (error) { + if (error !== segmentComplete) throw error + return 0 + } + if (options.preDrainDryRun && !preDrainDryRunPassed(result)) return 2 + return result.frozenAt ? 2 : 0 +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + runIncidentMonitorCli(process.argv.slice(2)) + .then((code) => { + process.exitCode = code + }) + .catch((error: unknown) => { + console.error(error instanceof Error ? error.message : 'incident monitor failed') + process.exitCode = 1 + }) +} diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts new file mode 100644 index 00000000000..09b7b16fa45 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.test.ts @@ -0,0 +1,418 @@ +import { describe, expect, it } from 'vitest' +import { RELAY_OPS_ENVIRONMENTS } from './environment-config.js' +import type { GcloudClient } from './gcloud-client.js' +import { effectiveAdmissionState } from './incident-selector.js' +import { INCIDENT_MONITOR_THRESHOLDS } from './incident-monitor.js' +import { + directorSignals, + GOOGLE_METRICS, + readGoogleMetric, + readGoogleMetricWithEmptyRetry, + relayFiveMinuteDeltaSignal +} from './incident-monitor-sources.js' + +const now = Date.parse('2026-07-28T10:00:00.000Z') +const startAt = new Date(now - 5 * 60_000).toISOString() +const endAt = new Date(now).toISOString() +const productionCells = RELAY_OPS_ENVIRONMENTS.production.cells.map( + ({ cellId }) => cellId +) + +describe('incident monitor sources', () => { + it('uses legacy booleans only at selector generation zero', () => { + const membership = { + existingOnly: ['c1'], + migrationOnly: [], + general: ['c2'] + } + expect( + effectiveAdmissionState({ generation: 0, membership }, true, 'c1') + ).toBe('general') + expect( + effectiveAdmissionState({ generation: 1, membership }, true, 'c1') + ).toBe('existing-only') + }) + + it('sums DELTA metrics across the window and scopes every target exactly', async () => { + const filters: string[] = [] + const fetchImpl: typeof fetch = async (input) => { + const url = new URL(String(input)) + filters.push(url.searchParams.get('filter') ?? '') + return Response.json({ + timeSeries: [ + { + points: [ + { + interval: { endTime: new Date(now - 120_000).toISOString() }, + value: { int64Value: '2' } + }, + { + interval: { endTime: new Date(now - 60_000).toISOString() }, + value: { int64Value: '3' } + } + ] + } + ] + }) + } + const environment = RELAY_OPS_ENVIRONMENTS.production + const directorErrors = GOOGLE_METRICS.find( + (definition) => definition.signal === 'director.errors' + )! + const deadlocks = GOOGLE_METRICS.find( + (definition) => definition.signal === 'cloud_sql.deadlocks' + )! + await expect( + readGoogleMetric( + environment, + directorErrors, + 'secret-access-token', + startAt, + endAt, + fetchImpl + ) + ).resolves.toEqual({ + value: 5, + observedAt: new Date(now - 60_000).toISOString() + }) + await readGoogleMetric( + environment, + deadlocks, + 'secret-access-token', + startAt, + endAt, + fetchImpl + ) + expect(filters[0]).toContain( + 'resource.label."service_name"="orca-cloud-relay"' + ) + expect(filters[0]).toContain('metric.label."response_code"!="503"') + expect(filters[1]).toContain( + 'resource.label."database_id"="onorca-cloud:orca-cloud-auth-db"' + ) + }) + + it('zero-fills an expired sparse lock-wait point', async () => { + let pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + const fetchImpl: typeof fetch = async () => Response.json({ + timeSeries: [{ + points: [{ + interval: { endTime: new Date(pointAt).toISOString() }, + value: { int64Value: '1' } + }] + }] + }) + const definition = GOOGLE_METRICS.find( + ({ signal }) => signal === 'cloud_sql.lock_waits' + )! + await expect(readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + fetchImpl + )).resolves.toEqual({ + value: 1, + observedAt: new Date(pointAt).toISOString() + }) + await expect(readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + fetchImpl, + () => now + 3_000 + )).resolves.toEqual({ + value: 1, + observedAt: new Date(pointAt).toISOString() + }) + pointAt-- + await expect(readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + fetchImpl + )).resolves.toEqual({ value: 0, observedAt: endAt }) + }) + + it('freshens a sparse zero without masking a recent nonzero lock wait', async () => { + let value = 0 + const pointAt = now - INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + const readAt = now + 11_879 + const fetchImpl: typeof fetch = async () => Response.json({ + timeSeries: [{ + points: [{ + interval: { endTime: new Date(pointAt).toISOString() }, + value: { int64Value: String(value) } + }] + }] + }) + const definition = GOOGLE_METRICS.find( + ({ signal }) => signal === 'cloud_sql.lock_waits' + )! + + const sparseZero = await readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + fetchImpl, + () => readAt + ) + expect(sparseZero).toEqual({ + value: 0, + observedAt: new Date(readAt).toISOString() + }) + expect(readAt - pointAt).toBe(191_879) + expect(readAt - Date.parse(sparseZero!.observedAt)).toBe(0) + + value = 20 + await expect(readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + fetchImpl, + () => readAt + )).resolves.toEqual({ + value: 20, + observedAt: new Date(pointAt).toISOString() + }) + }) + + it('preserves a recent nonzero lock wait across staggered series', async () => { + const nonzeroAt = now - 179_000 + const zeroAt = now - 178_000 + const definition = GOOGLE_METRICS.find( + ({ signal }) => signal === 'cloud_sql.lock_waits' + )! + const metric = await readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + async () => Response.json({ + timeSeries: [ + { + points: [{ + interval: { endTime: new Date(nonzeroAt).toISOString() }, + value: { int64Value: '7' } + }] + }, + { + points: [{ + interval: { endTime: new Date(zeroAt).toISOString() }, + value: { int64Value: '0' } + }] + } + ] + }) + ) + + expect(metric).toEqual({ + value: 7, + observedAt: new Date(nonzeroAt).toISOString() + }) + + await expect(readGoogleMetric( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + async () => Response.json({ + timeSeries: [{ + points: [ + { + interval: { endTime: new Date(nonzeroAt).toISOString() }, + value: { int64Value: '7' } + }, + { + interval: { endTime: new Date(zeroAt).toISOString() }, + value: { int64Value: '0' } + } + ] + }] + }) + )).resolves.toEqual({ value: 0, observedAt: endAt }) + }) + + it('retries an empty required metric without weakening its value', async () => { + let calls = 0 + const waits: number[] = [] + const definition = GOOGLE_METRICS.find( + ({ signal }) => signal === 'cloud_sql.cpu' + )! + await expect(readGoogleMetricWithEmptyRetry( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + async () => Response.json(calls++ === 0 + ? { timeSeries: [] } + : { + timeSeries: [{ + points: [{ + interval: { endTime: new Date(now - 60_000).toISOString() }, + value: { doubleValue: 0.81 } + }] + }] + }), + () => now, + async (ms) => { waits.push(ms) } + )).resolves.toEqual({ + value: 0.81, + observedAt: new Date(now - 60_000).toISOString() + }) + expect(calls).toBe(2) + expect(waits).toEqual([2_000]) + }) + + it('still reports a required metric missing after bounded retries', async () => { + let calls = 0 + const waits: number[] = [] + const definition = GOOGLE_METRICS.find( + ({ signal }) => signal === 'director.concurrency' + )! + await expect(readGoogleMetricWithEmptyRetry( + RELAY_OPS_ENVIRONMENTS.production, + definition, + 'secret-access-token', + startAt, + endAt, + async () => { + calls++ + return Response.json({ timeSeries: [] }) + }, + () => now, + async (ms) => { waits.push(ms) } + )).resolves.toBeNull() + expect(calls).toBe(3) + expect(waits).toEqual([2_000, 2_000]) + }) + + it('timestamps sparse retry aggregates at query completion', () => { + expect(relayFiveMinuteDeltaSignal({ + available: true, + points: [ + { at: new Date(now - 22 * 60_000).toISOString(), value: 7 }, + { at: new Date(now - 4 * 60_000).toISOString(), value: 2 } + ] + }, endAt)).toEqual({ value: 2, observedAt: endAt }) + expect(relayFiveMinuteDeltaSignal({ + available: true, + points: [{ at: new Date(now - 22 * 60_000).toISOString(), value: 7 }] + }, endAt)).toEqual({ value: 0, observedAt: endAt }) + expect(relayFiveMinuteDeltaSignal({ + available: false, + points: [] + }, endAt)).toBeNull() + }) + + it('aggregates admin state without returning tokens or response identities', async () => { + const identityToken = 'secret.header.signature' + const sensitiveIdentity = 'user@example.test' + const gcloud: GcloudClient = { + accessToken: async () => 'unused', + identityToken: async () => identityToken + } + let activeRequests = 0 + let maximumActiveRequests = 0 + let requestCount = 0 + const fetchImpl: typeof fetch = async (_input, init) => { + requestCount++ + activeRequests++ + maximumActiveRequests = Math.max(maximumActiveRequests, activeRequests) + expect(new Headers(init?.headers).get('authorization')).toBe( + `Bearer ${identityToken}` + ) + const body = JSON.parse(String(init?.body)) as { + cellId?: string + sourceCellId?: string + targetCellId?: string + } + await Promise.resolve() + activeRequests-- + if (!body.cellId && !body.sourceCellId) { + return Response.json({ + selector: { + generation: 1, + membership: { + existingOnly: productionCells.slice(2), + migrationOnly: ['production-gce-c2'], + general: ['production-gce-c1'] + } + } + }) + } + if (body.cellId) { + return Response.json({ + status: { + // Selector-era monitoring must ignore this legacy compatibility bit. + enabled: body.cellId !== 'production-gce-c1', + connectionCapacity: { + hardCap: body.cellId === 'production-gce-c1' ? 1_000 : 600 + }, + runtime: { + lastHeartbeatAt: now - 1_000, + heartbeatFresh: true + }, + userId: sensitiveIdentity + } + }) + } + return Response.json({ + blocked: 0, + blockedExpiredUnregistered: + body.sourceCellId === 'production-gce-c1' && + body.targetCellId === 'production-gce-c2' + ? 1 + : 0, + registeredTargetInactive: 0, + userId: sensitiveIdentity + }) + } + const result = await directorSignals( + 'production', + { + generation: 1, + membership: { + existingOnly: productionCells.slice(2), + migrationOnly: ['production-gce-c2'], + general: ['production-gce-c1'] + } + }, + gcloud, + now, + fetchImpl + ) + expect(requestCount).toBe(productionCells.length * 2) + expect(maximumActiveRequests).toBe(1) + expect( + result.source.signals['cell.production-gce-c1.migration_blocked'] + ).toMatchObject({ value: 1 }) + expect(result.cells.find((cell) => cell.cellId === 'production-gce-c1')).toMatchObject({ + expectedAdmissionState: 'general' + }) + expect( + result.source.signals['cell.production-gce-c1.admission_state'] + ).toMatchObject({ value: 2 }) + expect( + result.source.signals['cell.production-gce-c1.connection_hard_cap'] + ).toMatchObject({ value: 1_000 }) + expect( + result.source.signals['cell.production-gce-c2.admission_state'] + ).toMatchObject({ value: 1 }) + const serialized = JSON.stringify(result) + expect(serialized).not.toContain(identityToken) + expect(serialized).not.toContain(sensitiveIdentity) + }) +}) diff --git a/cloud/apps/relay-ops/src/incident-monitor-sources.ts b/cloud/apps/relay-ops/src/incident-monitor-sources.ts new file mode 100644 index 00000000000..0b78c2c4f6b --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-monitor-sources.ts @@ -0,0 +1,646 @@ +import { z } from 'zod' +import { buildDashboardSnapshot } from './dashboard-snapshot.js' +import type { + RelayOpsEnvironment, + RelayOpsEnvironmentId +} from './environment-config.js' +import { relayOpsEnvironment } from './environment-config.js' +import type { GcloudClient } from './gcloud-client.js' +import type { RelayMetricSnapshot } from './monitoring-snapshot.js' +import { + AdmissionSelectorSchema, + effectiveAdmissionState, + normalizeSelectorMembership, + selectorCellState, + type AdmissionSelector, +} from './incident-selector.js' +import { + INCIDENT_MONITOR_THRESHOLDS, + type IncidentSample, + type IncidentSignal, + type IncidentSource +} from './incident-monitor.js' + +const NumericSchema = z.union([z.number(), z.string()]) + .transform(Number) + .pipe(z.number().finite()) +const MonitoringPointSchema = z.object({ + interval: z.object({ endTime: z.string() }), + value: z.object({ + doubleValue: NumericSchema.optional(), + int64Value: NumericSchema.optional(), + distributionValue: z.object({ + mean: NumericSchema.optional(), + range: z.object({ max: NumericSchema.optional() }).optional() + }).optional() + }) +}) +const MonitoringResponseSchema = z.object({ + timeSeries: z.array(z.object({ points: z.array(MonitoringPointSchema) })).default([]), + nextPageToken: z.string().optional() +}) +const CellStatusSchema = z.object({ + status: z.object({ + enabled: z.boolean(), + connectionCapacity: z + .object({ hardCap: z.number().int().positive() }) + .nullable(), + runtime: z.object({ + lastHeartbeatAt: z.number(), + heartbeatFresh: z.boolean() + }).nullable() + }) +}) +const SelectorStatusSchema = z.object({ + selector: AdmissionSelectorSchema +}) +const MigrationStatusSchema = z.object({ + blocked: z.number().int().nonnegative(), + registeredTargetInactive: z.number().int().nonnegative(), + blockedExpiredUnregistered: z.number().int().nonnegative() +}) + +export type GoogleMetricDefinition = { + signal: string + type: string + resourceFilter: string + aggregation: 'latest-max' | 'latest-sum' | 'window-sum' + emptyIsZero?: boolean + zeroAfterMs?: number +} + +export const GOOGLE_METRICS: GoogleMetricDefinition[] = [ + { + signal: 'cloud_sql.cpu', + type: 'cloudsql.googleapis.com/database/cpu/utilization', + resourceFilter: 'resource.type="cloudsql_database"', + aggregation: 'latest-max' + }, + { + signal: 'cloud_sql.memory', + type: 'cloudsql.googleapis.com/database/memory/utilization', + resourceFilter: 'resource.type="cloudsql_database"', + aggregation: 'latest-max' + }, + { + signal: 'cloud_sql.backends', + type: 'cloudsql.googleapis.com/database/postgresql/num_backends', + resourceFilter: 'resource.type="cloudsql_database"', + aggregation: 'latest-sum' + }, + { + signal: 'cloud_sql.lock_waits', + type: 'cloudsql.googleapis.com/database/postgresql/backends_in_wait', + resourceFilter: + 'resource.type="cloudsql_database" AND metric.label."wait_event_type"="Lock"', + aggregation: 'latest-max', + emptyIsZero: true, + zeroAfterMs: INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs + }, + { + signal: 'cloud_sql.deadlocks', + type: 'cloudsql.googleapis.com/database/postgresql/deadlock_count', + resourceFilter: 'resource.type="cloudsql_database"', + aggregation: 'window-sum', + emptyIsZero: true + }, + { + signal: 'director.instances', + type: 'run.googleapis.com/container/instance_count', + resourceFilter: 'resource.type="cloud_run_revision"', + aggregation: 'latest-sum' + }, + { + signal: 'director.cpu', + type: 'run.googleapis.com/container/cpu/utilizations', + resourceFilter: 'resource.type="cloud_run_revision"', + aggregation: 'latest-max' + }, + { + signal: 'director.memory', + type: 'run.googleapis.com/container/memory/utilizations', + resourceFilter: 'resource.type="cloud_run_revision"', + aggregation: 'latest-max' + }, + { + signal: 'director.concurrency', + type: 'run.googleapis.com/container/max_request_concurrencies', + resourceFilter: + 'resource.type="cloud_run_revision" AND metric.label."state"="active"', + aggregation: 'latest-max' + }, + { + signal: 'director.errors', + type: 'run.googleapis.com/request_count', + resourceFilter: + 'resource.type="cloud_run_revision" AND metric.label."response_code_class"="5xx" AND metric.label."response_code"!="503"', + aggregation: 'window-sum', + emptyIsZero: true + }, + { + signal: 'auth.errors', + type: 'run.googleapis.com/request_count', + resourceFilter: + 'resource.type="cloud_run_revision" AND metric.label."response_code_class"="5xx"', + aggregation: 'window-sum', + emptyIsZero: true + } +] + +function pointValue(point: z.infer): number { + return ( + point.value.doubleValue ?? + point.value.int64Value ?? + point.value.distributionValue?.range?.max ?? + point.value.distributionValue?.mean ?? + 0 + ) +} + +function serviceFilter(signal: string, directorService: string, authService: string): string { + if (signal.startsWith('director.')) { + return `resource.label."service_name"="${directorService}"` + } + if (signal.startsWith('auth.')) { + return `resource.label."service_name"="${authService}"` + } + return '' +} + +function targetFilter( + definition: GoogleMetricDefinition, + environment: RelayOpsEnvironment +): string { + if (definition.signal.startsWith('cloud_sql.')) { + return `resource.label."database_id"="${environment.project}:${environment.sqlInstance}"` + } + return serviceFilter( + definition.signal, + environment.directorService, + environment.authService + ) +} + +async function googleJson( + fetchImpl: typeof fetch, + token: string, + url: URL | string, + init: RequestInit = {} +): Promise { + const response = await fetchImpl(url, { + ...init, + headers: { + authorization: `Bearer ${token}`, + ...(init.body ? { 'content-type': 'application/json' } : {}) + }, + signal: AbortSignal.timeout(30_000) + }) + if (!response.ok) throw new Error(`Google telemetry returned ${response.status}`) + return await response.json() +} + +export async function readGoogleMetric( + environment: RelayOpsEnvironment, + definition: GoogleMetricDefinition, + token: string, + startAt: string, + endAt: string, + fetchImpl: typeof fetch, + now: () => number = () => Date.parse(endAt) +): Promise { + const url = new URL( + `https://monitoring.googleapis.com/v3/projects/${environment.project}/timeSeries` + ) + const filters = [ + `metric.type="${definition.type}"`, + definition.resourceFilter, + targetFilter(definition, environment) + ].filter(Boolean) + url.searchParams.set('filter', filters.join(' AND ')) + url.searchParams.set('interval.startTime', startAt) + url.searchParams.set('interval.endTime', endAt) + url.searchParams.set('view', 'FULL') + url.searchParams.set('pageSize', '1000') + const parsed = MonitoringResponseSchema.parse( + await googleJson(fetchImpl, token, url) + ) + if (parsed.nextPageToken) throw new Error('Google metric pagination is incomplete') + const queryEndMs = Date.parse(endAt) + const readAtMs = Math.max(queryEndMs, now()) + const readAt = new Date(readAtMs).toISOString() + const points = parsed.timeSeries.flatMap((series) => series.points) + if (points.length === 0) { + return definition.emptyIsZero ? { value: 0, observedAt: readAt } : null + } + const zeroAfterMs = definition.zeroAfterMs + if (definition.emptyIsZero && zeroAfterMs !== undefined) { + const latestSeriesPoints = parsed.timeSeries.flatMap((series) => { + const seriesNewestAt = Math.max( + ...series.points.map((point) => Date.parse(point.interval.endTime)) + ) + return series.points.filter( + (point) => Date.parse(point.interval.endTime) === seriesNewestAt + ) + }) + const futurePoints = latestSeriesPoints.filter( + (point) => Date.parse(point.interval.endTime) > queryEndMs + ) + if (futurePoints.length > 0) { + return { + value: Math.max(...futurePoints.map(pointValue)), + observedAt: new Date(Math.max( + ...futurePoints.map((point) => Date.parse(point.interval.endTime)) + )).toISOString() + } + } + const recentNonzero = latestSeriesPoints.filter((point) => { + const pointAt = Date.parse(point.interval.endTime) + return pointValue(point) > 0 && queryEndMs - pointAt <= zeroAfterMs + }) + if (recentNonzero.length === 0) return { value: 0, observedAt: readAt } + return { + value: Math.max(...recentNonzero.map(pointValue)), + observedAt: new Date(Math.min( + ...recentNonzero.map((point) => Date.parse(point.interval.endTime)) + )).toISOString() + } + } + const newestAt = Math.max(...points.map((point) => Date.parse(point.interval.endTime))) + const selected = definition.aggregation === 'window-sum' + ? points + : points.filter((point) => Date.parse(point.interval.endTime) === newestAt) + const values = selected.map(pointValue) + const value = + definition.aggregation !== 'latest-max' + ? values.reduce((total, entry) => total + entry, 0) + : Math.max(...values) + return { + value, + observedAt: new Date(newestAt).toISOString() + } +} + +export async function readGoogleMetricWithEmptyRetry( + environment: RelayOpsEnvironment, + definition: GoogleMetricDefinition, + token: string, + startAt: string, + endAt: string, + fetchImpl: typeof fetch, + now: () => number = () => Date.parse(endAt), + wait: (ms: number) => Promise = async (ms) => + await new Promise((resolve) => setTimeout(resolve, ms)) +): Promise { + for (let attempt = 0; attempt < 3; attempt++) { + const signal = await readGoogleMetric( + environment, + definition, + token, + startAt, + endAt, + fetchImpl, + now + ) + if (signal !== null || definition.emptyIsZero) return signal + if (attempt < 2) await wait(2_000) + } + return null +} + +function addSignal( + signals: Record, + name: string, + value: number | null, + observedAt: string | null +): void { + if (value === null || observedAt === null) return + signals[name] = { value, observedAt } +} + +function endpointSignals( + snapshot: Awaited>, + nowAt: string +): IncidentSource { + const signals: Record = {} + const addEndpoint = ( + prefix: string, + endpoint: { health: boolean | null; ready: boolean | null; latencyMs: number | null }, + unavailableIsZero = false + ) => { + addSignal( + signals, + `${prefix}.health`, + endpoint.health === null ? (unavailableIsZero ? 0 : null) : Number(endpoint.health), + nowAt + ) + addSignal( + signals, + `${prefix}.ready`, + endpoint.ready === null ? (unavailableIsZero ? 0 : null) : Number(endpoint.ready), + nowAt + ) + addSignal(signals, `${prefix}.latency_ms`, endpoint.latencyMs, nowAt) + } + addEndpoint('director', snapshot.resources.directorEndpoint) + addEndpoint('auth', snapshot.resources.authEndpoint) + for (const cell of snapshot.resources.cells) { + addEndpoint(`cell.${cell.cellId}`, cell.endpoint, true) + } + return { observedAt: nowAt, signals } +} + +export function relayFiveMinuteDeltaSignal( + metric: Pick, + endAt: string +): IncidentSignal | null { + if (!metric.available) return null + const endMs = Date.parse(endAt) + if (!Number.isFinite(endMs)) throw new Error('Relay telemetry end time is invalid') + return { + value: metric.points + .filter((point) => { + const pointMs = Date.parse(point.at) + return pointMs >= endMs - 300_000 && pointMs <= endMs + }) + .reduce((total, point) => total + point.value, 0), + observedAt: endAt + } +} + +function relaySignals( + snapshot: Awaited> +): IncidentSource { + const metrics = snapshot.monitoring.metrics + const signals: Record = {} + addSignal( + signals, + 'relay.pool_waiting', + metrics.db_waiters_max.latest, + metrics.db_waiters_max.latestAt + ) + addSignal( + signals, + 'relay.pool_wait_ms', + metrics.db_wait_ms_max.latest, + metrics.db_wait_ms_max.latestAt + ) + const retries = relayFiveMinuteDeltaSignal( + metrics.postgres_retries, + snapshot.monitoring.endAt + ) + if (retries) signals['relay.postgres_retries'] = retries + const retryExhausted = relayFiveMinuteDeltaSignal( + metrics.postgres_retry_exhausted, + snapshot.monitoring.endAt + ) + if (retryExhausted) signals['relay.postgres_retry_exhausted'] = retryExhausted + for (const cell of snapshot.resources.cells) { + addSignal( + signals, + `cell.${cell.cellId}.connections`, + metrics.total_connections.latestByCell[cell.cellId] ?? null, + metrics.total_connections.latestAt + ) + addSignal( + signals, + `cell.${cell.cellId}.queued_bytes`, + metrics.queued_bytes.latestByCell[cell.cellId] ?? null, + metrics.queued_bytes.latestAt + ) + } + const observedTimes = Object.values(signals).map((entry) => Date.parse(entry.observedAt)) + if (observedTimes.length === 0) throw new Error('Relay telemetry is unavailable') + const observedAt = new Date(Math.max(...observedTimes)).toISOString() + return { observedAt, signals } +} + +async function adminPost( + fetchImpl: typeof fetch, + origin: string, + token: string, + path: string, + body: unknown +): Promise { + const response = await fetchImpl(`${origin}${path}`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }) + if (!response.ok) throw new Error(`Relay admin telemetry returned ${response.status}`) + return await response.json() +} + +export async function directorSignals( + environmentId: RelayOpsEnvironmentId, + expectedSelector: AdmissionSelector, + gcloud: GcloudClient, + nowMs: number, + fetchImpl: typeof fetch +): Promise<{ + source: IncidentSource + selector: AdmissionSelector + cells: IncidentSample['cells'] +}> { + const environment = relayOpsEnvironment(environmentId) + if (!gcloud.identityToken) throw new Error('gcloud identity-token support is unavailable') + const token = await gcloud.identityToken(`${environment.directorOrigin}/v1/admin/drain`) + const configuredCellIds = new Set(environment.cells.map((cell) => cell.cellId)) + const rawSelector = SelectorStatusSchema.parse( + await adminPost( + fetchImpl, + environment.directorOrigin, + token, + '/v1/admin/admission-selector/status', + { v: 1 } + ) + ).selector + const selector = { + generation: rawSelector.generation, + membership: normalizeSelectorMembership(rawSelector.membership, configuredCellIds) + } + const statuses: Array<{ + cell: RelayOpsEnvironment['cells'][number] + status: z.infer['status'] + }> = [] + for (const cell of environment.cells) { + statuses.push({ + cell, + status: CellStatusSchema.parse( + await adminPost(fetchImpl, environment.directorOrigin, token, '/v1/admin/cell-status', { + v: 1, + cellId: cell.cellId + }) + ).status + }) + } + const migrationEntries: Array<{ + sourceCellId: string + migration: z.infer + }> = [] + const migrationTargets = new Set(expectedSelector.membership.migrationOnly) + for (const source of environment.cells) { + for (const target of environment.cells) { + if (source.cellId === target.cellId || !migrationTargets.has(target.cellId)) continue + migrationEntries.push({ + sourceCellId: source.cellId, + migration: MigrationStatusSchema.parse( + await adminPost( + fetchImpl, + environment.directorOrigin, + token, + '/v1/admin/evacuation-status', + { + v: 1, + sourceCellId: source.cellId, + targetCellId: target.cellId, + completeReady: false + } + ) + ) + }) + } + } + const migrationBySource = new Map() + for (const { sourceCellId, migration } of migrationEntries) { + const aggregate = migrationBySource.get(sourceCellId) ?? { blocked: 0, targetInactive: 0 } + aggregate.blocked += migration.blocked + migration.blockedExpiredUnregistered + aggregate.targetInactive += migration.registeredTargetInactive + migrationBySource.set(sourceCellId, aggregate) + } + const nowAt = new Date(nowMs).toISOString() + const signals: Record = {} + for (const { cell, status } of statuses) { + const prefix = `cell.${cell.cellId}` + const admissionState = effectiveAdmissionState( + selector, + status.enabled, + cell.cellId + ) + addSignal( + signals, + `${prefix}.admission_state`, + ['existing-only', 'migration-only', 'general'].indexOf(admissionState), + nowAt + ) + addSignal( + signals, + `${prefix}.connection_hard_cap`, + status.connectionCapacity?.hardCap ?? + INCIDENT_MONITOR_THRESHOLDS.cellConnections, + nowAt + ) + if (status.runtime) { + addSignal( + signals, + `${prefix}.heartbeat_fresh`, + Number(status.runtime.heartbeatFresh), + nowAt + ) + addSignal( + signals, + `${prefix}.heartbeat_age_ms`, + Math.max(0, nowMs - status.runtime.lastHeartbeatAt), + nowAt + ) + } + const migration = migrationBySource.get(cell.cellId) ?? { blocked: 0, targetInactive: 0 } + addSignal(signals, `${prefix}.migration_blocked`, migration.blocked, nowAt) + addSignal( + signals, + `${prefix}.migration_target_inactive`, + migration.targetInactive, + nowAt + ) + } + return { + source: { observedAt: nowAt, signals }, + selector, + cells: statuses.map(({ cell }) => ({ + cellId: cell.cellId, + runtimeKnown: true, + powered: true, + expectedAdmissionState: selectorCellState(expectedSelector, cell.cellId) + })) + } +} + +export type IncidentSampleCollectorOptions = { + environment: RelayOpsEnvironmentId + expectedSelector: AdmissionSelector + fetchImpl?: typeof fetch + now?: () => number +} + +export function createIncidentSampleCollector( + gcloud: GcloudClient, + options: IncidentSampleCollectorOptions +): () => Promise { + const fetchImpl = options.fetchImpl ?? fetch + const now = options.now ?? Date.now + return async () => { + const nowMs = now() + const nowAt = new Date(nowMs).toISOString() + const startAt = new Date(nowMs - 5 * 60_000).toISOString() + const environment = relayOpsEnvironment(options.environment) + const accessToken = gcloud.accessToken() + const cloudMetricEntries = accessToken.then(async (token) => await Promise.all( + GOOGLE_METRICS.map(async (definition) => [ + definition.signal, + await readGoogleMetricWithEmptyRetry( + environment, + definition, + token, + startAt, + nowAt, + fetchImpl, + now + ) + ] as const) + )) + const [snapshot, director, metricEntries] = await Promise.all([ + buildDashboardSnapshot(options.environment, gcloud, { + windowMinutes: 5, + now: new Date(nowMs), + fetchImpl + }), + directorSignals( + options.environment, + options.expectedSelector, + gcloud, + nowMs, + fetchImpl + ), + cloudMetricEntries + ]) + const cloudSignals = Object.fromEntries( + metricEntries.filter((entry): entry is [string, IncidentSignal] => entry[1] !== null) + ) + const relay = relaySignals(snapshot) + const poweredByCell = new Map( + snapshot.resources.cells.map((cell) => [ + cell.cellId, + { + runtimeKnown: cell.targetSize !== null, + powered: (cell.targetSize ?? 0) > 0 + } + ]) + ) + return { + collectedAt: nowAt, + selector: director.selector, + expectedSelector: options.expectedSelector, + sources: { + 'active-probe': endpointSignals(snapshot, nowAt), + 'cloud-monitoring': { observedAt: nowAt, signals: cloudSignals }, + 'relay-logs': relay, + 'director-admin': director.source + }, + cells: director.cells.map((cell) => ({ + ...cell, + runtimeKnown: poweredByCell.get(cell.cellId)?.runtimeKnown ?? false, + powered: poweredByCell.get(cell.cellId)?.powered ?? false + })) + } + } +} diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts new file mode 100644 index 00000000000..51153bb63e1 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -0,0 +1,786 @@ +import { describe, expect, it } from 'vitest' +import { + evaluateIncidentSample, + INCIDENT_CHECKPOINT_MINUTES, + INCIDENT_MONITOR_THRESHOLDS, + INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS, + initialIncidentMonitorState, + preDrainDryRunPassed, + runIncidentMonitor, + type IncidentSample +} from './incident-monitor.js' + +const startedAt = Date.parse('2026-07-28T00:00:00.000Z') +const selector = { + generation: 1, + membership: { + existingOnly: [], + migrationOnly: [], + general: ['production-gce-c1'] + } +} +const signal = (value: number, at = startedAt) => ({ + value, + observedAt: new Date(at).toISOString() +}) + +function healthySample(at = startedAt): IncidentSample { + const observedAt = new Date(at).toISOString() + return { + collectedAt: observedAt, + selector, + expectedSelector: selector, + cells: [{ + cellId: 'production-gce-c1', + runtimeKnown: true, + powered: true, + expectedAdmissionState: 'general' + }], + sources: { + 'active-probe': { + observedAt, + signals: { + 'director.health': signal(1, at), + 'director.ready': signal(1, at), + 'director.latency_ms': signal(100, at), + 'auth.health': signal(1, at), + 'auth.ready': signal(1, at), + 'auth.latency_ms': signal(100, at), + 'cell.production-gce-c1.health': signal(1, at), + 'cell.production-gce-c1.ready': signal(1, at), + 'cell.production-gce-c1.latency_ms': signal(100, at) + } + }, + 'cloud-monitoring': { + observedAt, + signals: { + 'cloud_sql.cpu': signal(0.2, at), + 'cloud_sql.memory': signal(0.3, at), + 'cloud_sql.backends': signal(12, at), + 'cloud_sql.lock_waits': signal(0, at), + 'cloud_sql.deadlocks': signal(0, at), + 'director.instances': signal(5, at), + 'director.cpu': signal(0.2, at), + 'director.memory': signal(0.3, at), + 'director.concurrency': signal(5, at), + 'director.errors': signal(0, at), + 'auth.errors': signal(0, at) + } + }, + 'relay-logs': { + observedAt, + signals: { + 'relay.pool_waiting': signal(0, at), + 'relay.pool_wait_ms': signal(1, at), + 'relay.postgres_retries': signal(0, at), + 'relay.postgres_retry_exhausted': signal(0, at), + 'cell.production-gce-c1.connections': signal(100, at), + 'cell.production-gce-c1.queued_bytes': signal(0, at) + } + }, + 'director-admin': { + observedAt, + signals: { + 'cell.production-gce-c1.admission_state': signal(2, at), + 'cell.production-gce-c1.heartbeat_fresh': signal(1, at), + 'cell.production-gce-c1.heartbeat_age_ms': signal(1_000, at), + 'cell.production-gce-c1.migration_blocked': signal(0, at), + 'cell.production-gce-c1.migration_target_inactive': signal(0, at) + } + } + } + } +} + +describe('incident monitor evaluator', () => { + it('accepts a complete fresh sample at every exact boundary', () => { + const sample = healthySample() + sample.sources['active-probe']!.signals['director.latency_ms'] = + signal(INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs) + sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = + signal(INCIDENT_MONITOR_THRESHOLDS.cloudSqlCpuUtilization) + sample.sources['relay-logs']!.signals['relay.pool_wait_ms'] = + signal(INCIDENT_MONITOR_THRESHOLDS.relayPoolWaitMs) + sample.sources['relay-logs']!.signals['relay.postgres_retries'] = + signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries) + sample.sources['cloud-monitoring']!.signals['cloud_sql.backends'] = + signal(INCIDENT_MONITOR_THRESHOLDS.cloudSqlBackends) + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + }) + + it('freezes when postgres retries exceed the recalibrated ceiling', () => { + const sample = healthySample() + sample.sources['relay-logs']!.signals['relay.postgres_retries'] = + signal(INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + 1) + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'freeze', + failures: [ + expect.objectContaining({ signal: 'relay.postgres_retries', threshold: 300 }) + ] + }) + }) + + it('allows missing auth readiness and legacy existing-only connections', () => { + const sample = healthySample() + const legacySelector = { + generation: 1, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: [], + general: [] + } + } + sample.selector = legacySelector + sample.expectedSelector = legacySelector + sample.cells[0]!.expectedAdmissionState = 'existing-only' + delete sample.sources['active-probe']!.signals['auth.ready'] + sample.sources['relay-logs']!.signals['cell.production-gce-c1.connections'] = + signal(900) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.admission_state' + ] = signal(0) + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + }) + + it('fails loudly on every missing or stale source', () => { + const missing = healthySample() + delete missing.sources['relay-logs'] + expect(evaluateIncidentSample(missing, startedAt).failures).toContainEqual({ + code: 'source_missing', + source: 'relay-logs' + }) + const stale = healthySample(startedAt - 180_001) + const failures = evaluateIncidentSample(stale, startedAt).failures + expect(failures.some((failure) => failure.source === 'cloud-monitoring')).toBe(true) + expect(failures.some((failure) => failure.source === 'active-probe')).toBe(true) + }) + + it('freezes on SQL, director, relay pool, heartbeat, and migration breaches', () => { + const sample = healthySample() + sample.sources['cloud-monitoring']!.signals['cloud_sql.cpu'] = signal(0.81) + sample.sources['cloud-monitoring']!.signals['director.instances'] = signal(7) + sample.sources['relay-logs']!.signals['relay.pool_waiting'] = signal(801) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.heartbeat_age_ms' + ] = signal(45_001) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_blocked' + ] = signal(1) + sample.sources['relay-logs']!.signals['cell.production-gce-c1.connections'] = + signal(501) + const evaluation = evaluateIncidentSample(sample, startedAt) + expect(evaluation.status).toBe('freeze') + expect(evaluation.failures.map((failure) => failure.signal)).toEqual( + expect.arrayContaining([ + 'cloud_sql.cpu', + 'director.instances', + 'relay.pool_waiting', + 'cell.production-gce-c1.connections', + 'cell.production-gce-c1.heartbeat_age_ms', + 'cell.production-gce-c1.migration_blocked' + ]) + ) + }) + + it('uses each cell reported physical connection cap', () => { + const sample = healthySample() + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.connection_hard_cap' + ] = signal(1_000) + sample.sources['relay-logs']!.signals['cell.production-gce-c1.connections'] = + signal(999) + + expect(evaluateIncidentSample(sample, startedAt).status).toBe('green') + sample.sources['relay-logs']!.signals['cell.production-gce-c1.connections'] = + signal(1_000) + expect(evaluateIncidentSample(sample, startedAt).status).toBe('freeze') + }) + + it('allows five active directors plus the warm rollback', () => { + const sample = healthySample() + sample.sources['cloud-monitoring']!.signals['director.instances'] = signal(6) + expect(evaluateIncidentSample(sample, startedAt).status).toBe('green') + }) + + it('allows bounded relay pool waiting below the latency ceiling', () => { + const sample = healthySample() + sample.sources['relay-logs']!.signals['relay.pool_waiting'] = signal(800) + sample.sources['relay-logs']!.signals['relay.pool_wait_ms'] = signal(2_500) + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + sample.sources['relay-logs']!.signals['relay.pool_wait_ms'] = signal(2_501) + expect(evaluateIncidentSample(sample, startedAt).failures).toContainEqual({ + code: 'threshold_max', + source: 'relay-logs', + signal: 'relay.pool_wait_ms', + observed: 2_501, + threshold: 2_500 + }) + }) + + it('bounds Cloud SQL backends above measured healthy peaks', () => { + const sample = healthySample() + sample.sources['cloud-monitoring']!.signals['cloud_sql.backends'] = signal(250) + expect(evaluateIncidentSample(sample, startedAt).status).toBe('green') + sample.sources['cloud-monitoring']!.signals['cloud_sql.backends'] = signal(251) + expect(evaluateIncidentSample(sample, startedAt).failures).toContainEqual({ + code: 'threshold_max', + source: 'cloud-monitoring', + signal: 'cloud_sql.backends', + observed: 251, + threshold: 250 + }) + }) + + it('bounds SQL lock waiters and keeps deadlocks zero-tolerance', () => { + const sample = healthySample() + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal(20) + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal(21) + expect(evaluateIncidentSample(sample, startedAt).failures).toContainEqual({ + code: 'threshold_max', + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits', + observed: 21, + threshold: 20 + }) + sample.sources['cloud-monitoring']!.signals['cloud_sql.lock_waits'] = signal(0) + sample.sources['cloud-monitoring']!.signals['cloud_sql.deadlocks'] = signal(1) + expect(evaluateIncidentSample(sample, startedAt).failures).toContainEqual({ + code: 'threshold_max', + source: 'cloud-monitoring', + signal: 'cloud_sql.deadlocks', + observed: 1, + threshold: 0 + }) + }) + + it('allows only registered target inactivity during forward recovery', () => { + const sample = healthySample() + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_target_inactive' + ] = signal(40) + expect(evaluateIncidentSample(sample, startedAt).failures).toContainEqual({ + code: 'threshold_max', + source: 'director-admin', + signal: 'cell.production-gce-c1.migration_target_inactive', + observed: 40, + threshold: 0 + }) + expect( + evaluateIncidentSample( + sample, + startedAt, + 'recover-forward', + 'production-gce-c1' + ) + ).toMatchObject({ + status: 'green', + failures: [] + }) + expect( + evaluateIncidentSample( + sample, + startedAt, + 'recover-forward', + 'production-gce-c2' + ).failures + ).toContainEqual(expect.objectContaining({ + signal: 'cell.production-gce-c1.migration_target_inactive' + })) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_blocked' + ] = signal(1) + expect( + evaluateIncidentSample( + sample, + startedAt, + 'recover-forward', + 'production-gce-c1' + ).failures + ).toContainEqual({ + code: 'threshold_max', + source: 'director-admin', + signal: 'cell.production-gce-c1.migration_blocked', + observed: 1, + threshold: 0 + }) + }) + + it('scopes registered target inactivity to the capacity cell', () => { + const sample = healthySample() + const scopedSelector = { + generation: 1, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: [], + general: ['production-gce-c2', 'production-gce-c3'] + } + } + sample.selector = scopedSelector + sample.expectedSelector = scopedSelector + sample.cells[0]!.expectedAdmissionState = 'existing-only' + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.admission_state' + ] = signal(0) + sample.cells.push({ + cellId: 'production-gce-c2', + runtimeKnown: true, + powered: true, + expectedAdmissionState: 'general' + }) + sample.cells.push({ + cellId: 'production-gce-c3', + runtimeKnown: true, + powered: true, + expectedAdmissionState: 'general' + }) + for (const sourceName of ['active-probe', 'relay-logs', 'director-admin'] as const) { + const signals = sample.sources[sourceName]!.signals + for (const [name, value] of Object.entries(signals)) { + if (name.includes('production-gce-c1')) { + for (const cellId of ['production-gce-c2', 'production-gce-c3']) { + signals[name.replace('production-gce-c1', cellId)] = value + } + } + } + } + for (const cellId of ['production-gce-c2', 'production-gce-c3']) { + sample.sources['director-admin']!.signals[ + `cell.${cellId}.admission_state` + ] = signal(2) + } + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.migration_target_inactive' + ] = signal(40) + expect( + evaluateIncidentSample( + sample, + startedAt, + 'capacity-transition', + null, + 'production-gce-c2' + ) + ).toMatchObject({ status: 'green', failures: [] }) + expect( + evaluateIncidentSample( + sample, + startedAt, + 'capacity-transition', + null, + 'production-gce-c3' + ) + ).toMatchObject({ status: 'green', failures: [] }) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c2.migration_target_inactive' + ] = signal(1) + expect( + evaluateIncidentSample( + sample, + startedAt, + 'capacity-transition', + null, + 'production-gce-c2' + ).failures + ).toContainEqual(expect.objectContaining({ + signal: 'cell.production-gce-c2.migration_target_inactive' + })) + }) + + it('freezes when expected admission has no powered runtime', () => { + const sample = healthySample() + sample.cells[0]!.powered = false + const evaluation = evaluateIncidentSample(sample, startedAt) + expect(evaluation.failures).toContainEqual({ + code: 'expected_admission_without_runtime', + source: 'director-admin', + signal: 'cell.production-gce-c1.powered', + observed: 0, + threshold: 1 + }) + }) + + it('ignores stale runtime signals for an expected offline existing-only cell', () => { + const sample = healthySample() + sample.cells[0] = { + ...sample.cells[0]!, + powered: false, + expectedAdmissionState: 'existing-only' + } + sample.expectedSelector = { + generation: sample.expectedSelector.generation, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: [], + general: [] + } + } + sample.selector = sample.expectedSelector + sample.sources['active-probe']!.signals['cell.production-gce-c1.health'] = signal(0) + sample.sources['active-probe']!.signals['cell.production-gce-c1.ready'] = signal(0) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.admission_state' + ] = signal(0) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.heartbeat_fresh' + ] = signal(0) + sample.sources['director-admin']!.signals[ + 'cell.production-gce-c1.heartbeat_age_ms' + ] = signal(9_000_000) + + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'green', + failures: [] + }) + }) + + it('freezes when cell power inventory is unavailable', () => { + const sample = healthySample() + sample.cells[0]!.runtimeKnown = false + expect(evaluateIncidentSample(sample, startedAt).failures).toContainEqual({ + code: 'runtime_power_unknown', + source: 'cloud-monitoring', + signal: 'cell.production-gce-c1.powered' + }) + }) + + it('freezes on selector generation or tri-state membership drift', () => { + const generation = healthySample() + generation.selector = { ...generation.selector, generation: 2 } + expect(evaluateIncidentSample(generation, startedAt).failures).toContainEqual( + expect.objectContaining({ code: 'selector_mismatch' }) + ) + + const membership = healthySample() + membership.selector = { + generation: 1, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: [], + general: [] + } + } + expect(evaluateIncidentSample(membership, startedAt).failures).toContainEqual( + expect.objectContaining({ code: 'selector_mismatch' }) + ) + }) +}) + +describe('incident monitor lifecycle', () => { + it('persists the exact 90-minute checkpoints while polling every minute', async () => { + let now = startedAt + const checkpoints: number[] = [] + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: false, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 90, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => healthySample(now), + persist: async () => {}, + checkpoint: async (summary) => { + checkpoints.push(summary.checkpointMinute) + } + }) + expect(checkpoints).toEqual([...INCIDENT_CHECKPOINT_MINUTES]) + expect(result.sampleCount).toBe(91) + expect(result.completedAt).not.toBeNull() + expect(result.frozenAt).toBeNull() + }) + + it('latches monitor freeze across a restart without rewriting its time', async () => { + let now = startedAt + let unhealthy = true + let persisted = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: false, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const stop = new Error('stop after first persistence') + await expect( + runIncidentMonitor(persisted, { + now: () => now, + wait: async () => { + throw stop + }, + collect: async () => { + const sample = healthySample(now) + if (unhealthy) { + sample.sources['relay-logs']!.signals['relay.pool_waiting'] = signal(31, now) + } + return sample + }, + persist: async (state) => { + persisted = structuredClone(state) + }, + checkpoint: async () => {} + }) + ).rejects.toThrow('stop after first persistence') + const frozenAt = persisted.frozenAt + unhealthy = false + now += 60_000 + const result = await runIncidentMonitor(persisted, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => healthySample(now), + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.frozenAt).toBe(frozenAt) + expect(preDrainDryRunPassed(result)).toBe(false) + }) + + it.each([15, 90])( + 'restarts a %i-minute continuous window after stale telemetry', + async (durationMinutes) => { + let now = startedAt + let staleInjected = false + const checkpoints: Array<[number, number]> = [] + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: durationMinutes === 15, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => { + if (!staleInjected && now === startedAt + 5 * 60_000) { + staleInjected = true + return healthySample(now - 180_001) + } + return healthySample(now) + }, + persist: async () => {}, + checkpoint: async (summary) => { + checkpoints.push([summary.windowSequence, summary.checkpointMinute]) + } + }) + expect(result.windowSequence).toBe(1) + expect(result.windowStartedAt).toBe( + new Date(startedAt + 6 * 60_000).toISOString() + ) + expect(result.completedAt).toBe( + new Date(startedAt + (durationMinutes + 6) * 60_000).toISOString() + ) + expect(result.sampleCount).toBe(durationMinutes + 1) + expect(result.continuityEvents).toHaveLength(1) + expect(result.continuityEvents[0]!.failures).toEqual( + expect.arrayContaining([ + expect.objectContaining({ code: 'source_stale' }) + ]) + ) + expect(checkpoints).toContainEqual([1, durationMinutes]) + expect(result.frozenAt).toBeNull() + } + ) + + it('resets at the next fresh sample after a runner gap', async () => { + let now = startedAt + 10 * 60_000 + const state = { + ...initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }), + lastSampleAt: new Date(startedAt).toISOString(), + sampleCount: 1, + totalSampleCount: 1 + } + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => healthySample(now), + persist: async () => {}, + checkpoint: async () => {} + }) + expect(result.windowSequence).toBe(1) + expect(result.windowStartedAt).toBe(new Date(startedAt + 10 * 60_000).toISOString()) + expect(result.continuityEvents[0]!.failures[0]!.code).toBe('monitor_gap') + expect(result.sampleCount).toBe(16) + }) + + it('fails a dry run after 25 total minutes of continuity resets', async () => { + let now = startedAt + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + now += ms + }, + collect: async () => + healthySample(now === startedAt + 10 * 60_000 ? now - 180_001 : now), + persist: async () => {}, + checkpoint: async () => {} + }) + + expect(result.completedAt).toBe( + new Date(startedAt + INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS).toISOString() + ) + expect(result.frozenAt).not.toBeNull() + expect(result.windowSequence).toBe(1) + expect(result.sampleCount).toBe(15) + expect(result.failures).toContainEqual({ + code: 'continuity_deadline_exceeded', + source: 'active-probe', + observed: INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS, + threshold: INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS + }) + expect(preDrainDryRunPassed(result)).toBe(false) + }) + + it('fails an overdue resumed dry run before collecting again', async () => { + const now = startedAt + INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS + 1 + let collections = 0 + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + const result = await runIncidentMonitor(state, { + now: () => now, + wait: async () => {}, + collect: async () => { + collections++ + return healthySample(now) + }, + persist: async () => {}, + checkpoint: async () => {} + }) + + expect(collections).toBe(0) + expect(result.failures).toContainEqual(expect.objectContaining({ + code: 'continuity_deadline_exceeded' + })) + }) + + it('requires a completed green 15-minute dry run', () => { + const state = { + ...initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }), + sampleCount: 16, + completedAt: new Date(startedAt + 15 * 60_000).toISOString() + } + expect(preDrainDryRunPassed(state)).toBe(true) + expect(preDrainDryRunPassed({ ...state, frozenAt: state.startedAt })).toBe(false) + expect(preDrainDryRunPassed({ + ...state, + completedAt: new Date(startedAt + INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS + 1).toISOString() + })).toBe(false) + }) + + it('keeps poll starts on the configured cadence after collection time', async () => { + let now = startedAt + const starts: number[] = [] + const waits: number[] = [] + const state = initialIncidentMonitorState({ + incidentId: 'incident-1', + environment: 'production', + expectedSelector: selector, + preDrainDryRun: true, + migrationPolicy: 'strict', + recoverySourceCellId: null, + capacityCellId: null, + startedAt: new Date(startedAt).toISOString(), + durationMinutes: 15, + intervalMs: 60_000 + }) + await runIncidentMonitor(state, { + now: () => now, + wait: async (ms) => { + waits.push(ms) + now += ms + }, + collect: async () => { + starts.push(now) + now += 15_000 + return healthySample(now) + }, + persist: async () => {}, + checkpoint: async () => {} + }) + expect(starts.slice(0, 3)).toEqual([ + startedAt, + startedAt + 60_000, + startedAt + 120_000 + ]) + expect(waits[0]).toBe(45_000) + }) +}) diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts new file mode 100644 index 00000000000..6073e351511 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -0,0 +1,800 @@ +import { + exactAdmissionSelector, + type AdmissionSelector, + type AdmissionState +} from './incident-selector.js' + +export const INCIDENT_MONITOR_THRESHOLDS = { + activeProbeMaxAgeMs: 60_000, + cloudDataMaxAgeMs: 180_000, + relayLogMaxAgeMs: 180_000, + heartbeatMaxAgeMs: 45_000, + endpointLatencyMs: 2_000, + cloudSqlCpuUtilization: 0.8, + cloudSqlMemoryUtilization: 0.9, + // Why: healthy latest-sum backends idle near 100 but spike to 216 in 1-minute + // bursts (~10 min/day exceeded the old bar of 160 on 2026-08-26, freezing a + // pre-drain gate on baseline noise). 250 clears measured healthy peaks while + // firing well before the verified 400-connection ceiling; pool-wait and + // exhausted-retry signals keep their strict thresholds. + cloudSqlBackends: 250, + // Bound the observed recovery load; deadlocks remain zero-tolerance. + cloudSqlLockWaits: 20, + cloudSqlDeadlocks: 0, + // Why: pool amplitude cannot discriminate the 2026-08-23 incident. Healthy + // fleet-wide bursts reach 43 waiters / 2.03s waits several times an hour, + // and a cell roll's reconnect surge peaks at 676 waiters, while the real + // incident peaked at 356 waiters and never crossed 2.5s (waits cap ~2s + // structurally). The old bars of 30/1000 froze pre-drain gates on baseline + // noise (~17% per 15-minute window). Incident-class contention is caught by + // the retry signals below at ~10x separation; these bars now fence only + // genuinely unbounded queueing, which grows past both. + relayPoolWaiting: 800, + relayPoolWaitMs: 2_500, + // Why: successful lock retries are the contention machinery working, not harm. + // Healthy 2026-08-26 baseline bursts to 234/5min (26% of windows crossed the old + // bar of 20, set unmeasured at the monitor's 2026-07-28 birth); the 2026-08-23 + // incident ran ~2,200-3,000/5min. 300 clears healthy bursts with ~10x incident + // margin; relayPostgresRetryExhausted below stays at zero tolerance, so any + // transaction that terminally fails still freezes the gate. + relayPostgresRetries: 300, + relayPostgresRetryExhausted: 0, + // Why: public admission is a per-instance semaphore, so fleet assignment capacity is + // concurrency x instances. A floor of 1 let the 2026-08-04 collapse from five instances + // to two pass unnoticed, which is the exact failure this monitor exists to catch. Keep in + // step with relay_min_instances in infra/terraform/environments/production.tfvars. + directorInstancesMin: 5, + // Five serving instances plus one warm scale-to-zero rollback during recovery. + directorInstancesMax: 6, + directorCpuUtilization: 0.8, + directorMemoryUtilization: 0.8, + directorConcurrency: 64, + directorErrors: 0, + authErrors: 0, + // Why: 800 exceeded the 600 hard cap, so this could never trigger on a capped cell. 500 is + // the ordinary admission limit a cell actually stops at (600 cap - 100 control-rebind reserve). + cellConnections: 500, + cellQueuedBytes: 48 * 1024 * 1024, + migrationBlocked: 0 +} as const + +export const INCIDENT_CHECKPOINT_MINUTES = [0, 5, 15, 30, 45, 60, 75, 90] as const +export const INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS = 25 * 60_000 + +export type IncidentSourceName = + | 'active-probe' + | 'cloud-monitoring' + | 'relay-logs' + | 'director-admin' + +export type IncidentMigrationPolicy = + | 'strict' + | 'recover-forward' + | 'capacity-transition' + +export type IncidentSignal = { + value: number + observedAt: string +} + +export type IncidentSource = { + observedAt: string + signals: Record +} + +export type IncidentCellExpectation = { + cellId: string + runtimeKnown: boolean + powered: boolean + expectedAdmissionState: AdmissionState +} + +export type IncidentSample = { + collectedAt: string + selector: AdmissionSelector + expectedSelector: AdmissionSelector + sources: Partial> + cells: IncidentCellExpectation[] +} + +export type IncidentFailure = { + code: string + source: IncidentSourceName + signal?: string + observed?: number + threshold?: number +} + +export type IncidentEvaluation = { + status: 'green' | 'freeze' + evaluatedAt: string + failures: IncidentFailure[] +} + +export type IncidentCheckpoint = { + schemaVersion: 4 + incidentId: string + environment: 'production' | 'staging' + expectedSelector: AdmissionSelector + preDrainDryRun: boolean + migrationPolicy: IncidentMigrationPolicy + recoverySourceCellId: string | null + capacityCellId: string | null + windowSequence: number + windowStartedAt: string + checkpointMinute: number + scheduledAt: string + recordedAt: string + status: 'green' | 'freeze' + frozenAt: string | null + sampleCount: number + failures: IncidentFailure[] + thresholds: typeof INCIDENT_MONITOR_THRESHOLDS +} + +export type IncidentMonitorState = { + schemaVersion: 4 + incidentId: string + environment: 'production' | 'staging' + expectedSelector: AdmissionSelector + preDrainDryRun: boolean + migrationPolicy: IncidentMigrationPolicy + recoverySourceCellId: string | null + capacityCellId: string | null + startedAt: string + windowStartedAt: string | null + windowSequence: number + durationMinutes: number + intervalMs: number + nextCheckpointIndex: number + sampleCount: number + totalSampleCount: number + lastSampleAt: string | null + continuityEvents: { + recordedAt: string + windowSequence: number + failures: IncidentFailure[] + }[] + frozenAt: string | null + failures: IncidentFailure[] + completedAt: string | null +} + +type NumericRule = { + source: IncidentSourceName + signal: string + comparison: 'max' | 'min' | 'equal' + threshold: number +} + +const NUMERIC_RULES: NumericRule[] = [ + { source: 'active-probe', signal: 'director.health', comparison: 'equal', threshold: 1 }, + { source: 'active-probe', signal: 'director.ready', comparison: 'equal', threshold: 1 }, + { + source: 'active-probe', + signal: 'director.latency_ms', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs + }, + { source: 'active-probe', signal: 'auth.health', comparison: 'equal', threshold: 1 }, + { + source: 'active-probe', + signal: 'auth.latency_ms', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs + }, + { + source: 'cloud-monitoring', + signal: 'cloud_sql.cpu', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.cloudSqlCpuUtilization + }, + { + source: 'cloud-monitoring', + signal: 'cloud_sql.memory', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.cloudSqlMemoryUtilization + }, + { + source: 'cloud-monitoring', + signal: 'cloud_sql.backends', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.cloudSqlBackends + }, + { + source: 'cloud-monitoring', + signal: 'director.instances', + comparison: 'min', + threshold: INCIDENT_MONITOR_THRESHOLDS.directorInstancesMin + }, + { + source: 'cloud-monitoring', + signal: 'director.instances', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.directorInstancesMax + }, + { + source: 'cloud-monitoring', + signal: 'director.cpu', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.directorCpuUtilization + }, + { + source: 'cloud-monitoring', + signal: 'director.memory', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.directorMemoryUtilization + }, + { + source: 'cloud-monitoring', + signal: 'director.concurrency', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.directorConcurrency + }, + { + source: 'cloud-monitoring', + signal: 'director.errors', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.directorErrors + }, + { + source: 'cloud-monitoring', + signal: 'auth.errors', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.authErrors + }, + { + source: 'cloud-monitoring', + signal: 'cloud_sql.lock_waits', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.cloudSqlLockWaits + }, + { + source: 'cloud-monitoring', + signal: 'cloud_sql.deadlocks', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.cloudSqlDeadlocks + }, + { + source: 'relay-logs', + signal: 'relay.pool_waiting', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.relayPoolWaiting + }, + { + source: 'relay-logs', + signal: 'relay.pool_wait_ms', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.relayPoolWaitMs + }, + { + source: 'relay-logs', + signal: 'relay.postgres_retries', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetries + }, + { + source: 'relay-logs', + signal: 'relay.postgres_retry_exhausted', + comparison: 'max', + threshold: INCIDENT_MONITOR_THRESHOLDS.relayPostgresRetryExhausted + } +] + +const SOURCE_MAX_AGE: Record = { + 'active-probe': INCIDENT_MONITOR_THRESHOLDS.activeProbeMaxAgeMs, + 'cloud-monitoring': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs, + 'relay-logs': INCIDENT_MONITOR_THRESHOLDS.relayLogMaxAgeMs, + 'director-admin': INCIDENT_MONITOR_THRESHOLDS.cloudDataMaxAgeMs +} + +function ageMs(timestamp: string, nowMs: number): number { + const parsed = Date.parse(timestamp) + return Number.isFinite(parsed) ? nowMs - parsed : Number.POSITIVE_INFINITY +} + +function addMissingSignal( + failures: IncidentFailure[], + source: IncidentSourceName, + signal: string +): void { + failures.push({ code: 'signal_missing', source, signal }) +} + +function checkRule( + failures: IncidentFailure[], + source: IncidentSourceName, + signals: Record, + rule: NumericRule +): void { + const signal = signals[rule.signal] + if (!signal) return addMissingSignal(failures, source, rule.signal) + const failed = + (rule.comparison === 'max' && signal.value > rule.threshold) || + (rule.comparison === 'min' && signal.value < rule.threshold) || + (rule.comparison === 'equal' && signal.value !== rule.threshold) + if (failed) { + failures.push({ + code: `threshold_${rule.comparison}`, + source, + signal: rule.signal, + observed: signal.value, + threshold: rule.threshold + }) + } +} + +function checkCell( + failures: IncidentFailure[], + sample: IncidentSample, + cell: IncidentCellExpectation, + migrationPolicy: IncidentMigrationPolicy, + recoverySourceCellId: string | null, + capacityCellId: string | null +): void { + const probe = sample.sources['active-probe']?.signals + const relay = sample.sources['relay-logs']?.signals + const admin = sample.sources['director-admin']?.signals + if (!cell.runtimeKnown) { + failures.push({ + code: 'runtime_power_unknown', + source: 'cloud-monitoring', + signal: `cell.${cell.cellId}.powered` + }) + } + if ( + cell.runtimeKnown && + cell.expectedAdmissionState !== 'existing-only' && + !cell.powered + ) { + failures.push({ + code: 'expected_admission_without_runtime', + source: 'director-admin', + signal: `cell.${cell.cellId}.powered`, + observed: 0, + threshold: 1 + }) + } + const checks = [ + ['active-probe', probe, `cell.${cell.cellId}.health`, cell.powered ? 1 : 0, 'equal'], + ['active-probe', probe, `cell.${cell.cellId}.ready`, cell.powered ? 1 : 0, 'equal'], + [ + 'active-probe', + probe, + `cell.${cell.cellId}.latency_ms`, + INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs, + 'max' + ], + [ + 'director-admin', + admin, + `cell.${cell.cellId}.admission_state`, + ['existing-only', 'migration-only', 'general'].indexOf( + cell.expectedAdmissionState + ), + 'equal' + ], + [ + 'director-admin', + admin, + `cell.${cell.cellId}.heartbeat_fresh`, + 1, + 'equal' + ], + [ + 'director-admin', + admin, + `cell.${cell.cellId}.heartbeat_age_ms`, + INCIDENT_MONITOR_THRESHOLDS.heartbeatMaxAgeMs, + 'max' + ], + [ + 'director-admin', + admin, + `cell.${cell.cellId}.migration_blocked`, + INCIDENT_MONITOR_THRESHOLDS.migrationBlocked, + 'max' + ], + [ + 'director-admin', + admin, + `cell.${cell.cellId}.migration_target_inactive`, + INCIDENT_MONITOR_THRESHOLDS.migrationBlocked, + 'max' + ], + [ + 'relay-logs', + relay, + `cell.${cell.cellId}.connections`, + (admin?.[`cell.${cell.cellId}.connection_hard_cap`]?.value ?? + INCIDENT_MONITOR_THRESHOLDS.cellConnections + 1) - 1, + 'max' + ], + [ + 'relay-logs', + relay, + `cell.${cell.cellId}.queued_bytes`, + INCIDENT_MONITOR_THRESHOLDS.cellQueuedBytes, + 'max' + ] + ] as const + for (const [source, signals, signalName, threshold, comparison] of checks) { + if ( + migrationPolicy === 'recover-forward' && + cell.cellId === recoverySourceCellId && + signalName.endsWith('.migration_target_inactive') + ) { + if (!signals?.[signalName]) addMissingSignal(failures, source, signalName) + continue + } + if ( + migrationPolicy === 'capacity-transition' && + capacityCellId !== null && + cell.cellId !== capacityCellId && + cell.expectedAdmissionState === 'existing-only' && + signalName.endsWith('.migration_target_inactive') + ) { + if (!signals?.[signalName]) addMissingSignal(failures, source, signalName) + continue + } + if ( + cell.expectedAdmissionState === 'existing-only' && + signalName.endsWith('.connections') + ) { + continue + } + if ( + !cell.powered && + [ + 'latency_ms', + 'heartbeat_fresh', + 'heartbeat_age_ms', + 'connections', + 'queued_bytes' + ].some((suffix) => signalName.endsWith(suffix)) + ) { + continue + } + if (!signals?.[signalName]) { + addMissingSignal(failures, source, signalName) + continue + } + const value = signals[signalName].value + const failed = comparison === 'equal' ? value !== threshold : value > threshold + if (failed) { + failures.push({ + code: `threshold_${comparison}`, + source, + signal: signalName, + observed: value, + threshold + }) + } + } +} + +export function evaluateIncidentSample( + sample: IncidentSample, + nowMs = Date.now(), + migrationPolicy: IncidentMigrationPolicy = 'strict', + recoverySourceCellId: string | null = null, + capacityCellId: string | null = null +): IncidentEvaluation { + const failures: IncidentFailure[] = [] + if (!exactAdmissionSelector(sample.selector, sample.expectedSelector)) { + failures.push({ + code: 'selector_mismatch', + source: 'director-admin', + signal: 'selector.generation', + observed: sample.selector.generation, + threshold: sample.expectedSelector.generation + }) + } + for (const [sourceName, maxAge] of Object.entries(SOURCE_MAX_AGE) as [ + IncidentSourceName, + number + ][]) { + const source = sample.sources[sourceName] + if (!source) { + failures.push({ code: 'source_missing', source: sourceName }) + continue + } + if (ageMs(source.observedAt, nowMs) < 0 || ageMs(source.observedAt, nowMs) > maxAge) { + failures.push({ + code: 'source_stale', + source: sourceName, + observed: ageMs(source.observedAt, nowMs), + threshold: maxAge + }) + } + for (const [signalName, signal] of Object.entries(source.signals)) { + if (ageMs(signal.observedAt, nowMs) < 0 || ageMs(signal.observedAt, nowMs) > maxAge) { + failures.push({ + code: 'signal_stale', + source: sourceName, + signal: signalName, + observed: ageMs(signal.observedAt, nowMs), + threshold: maxAge + }) + } + } + } + for (const rule of NUMERIC_RULES) { + const source = sample.sources[rule.source] + if (source) checkRule(failures, rule.source, source.signals, rule) + } + for (const cell of sample.cells) { + checkCell( + failures, + sample, + cell, + migrationPolicy, + recoverySourceCellId, + capacityCellId + ) + } + return { + status: failures.length === 0 ? 'green' : 'freeze', + evaluatedAt: new Date(nowMs).toISOString(), + failures + } +} + +export function initialIncidentMonitorState(input: { + incidentId: string + environment: 'production' | 'staging' + expectedSelector: AdmissionSelector + preDrainDryRun: boolean + migrationPolicy: IncidentMigrationPolicy + recoverySourceCellId: string | null + capacityCellId: string | null + startedAt: string + durationMinutes: number + intervalMs: number +}): IncidentMonitorState { + if (input.intervalMs < 1_000 || input.intervalMs > 60_000) { + throw new Error('incident monitor interval must be between 1 and 60 seconds') + } + if (input.durationMinutes < 15 || input.durationMinutes > 90) { + throw new Error('incident monitor duration must be between 15 and 90 minutes') + } + return { + schemaVersion: 4, + ...input, + windowStartedAt: input.startedAt, + windowSequence: 0, + nextCheckpointIndex: 0, + sampleCount: 0, + totalSampleCount: 0, + lastSampleAt: null, + continuityEvents: [], + frozenAt: null, + failures: [], + completedAt: null + } +} + +export type IncidentMonitorDependencies = { + now(): number + wait(ms: number): Promise + collect(): Promise + persist(state: IncidentMonitorState): Promise + checkpoint(summary: IncidentCheckpoint): Promise +} + +function checkpointMinutes(durationMinutes: number): number[] { + return INCIDENT_CHECKPOINT_MINUTES.filter((minute) => minute <= durationMinutes) +} + +const CONTINUITY_FAILURE_CODES = new Set([ + 'collector_failed', + 'monitor_gap', + 'signal_stale', + 'source_missing', + 'source_stale' +]) + +function resetContinuousWindow( + state: IncidentMonitorState, + recordedAt: string, + failures: IncidentFailure[] +): void { + if (state.windowStartedAt !== null) { + state.windowSequence++ + state.windowStartedAt = null + state.nextCheckpointIndex = 0 + state.sampleCount = 0 + state.completedAt = null + } + state.continuityEvents.push({ + recordedAt, + windowSequence: state.windowSequence, + failures + }) +} + +function completeContinuityDeadline( + state: IncidentMonitorState, + nowMs: number, + lineageStartMs: number +): void { + const recordedAt = new Date(nowMs).toISOString() + state.frozenAt ??= recordedAt + state.failures.push({ + code: 'continuity_deadline_exceeded', + source: 'active-probe', + observed: nowMs - lineageStartMs, + threshold: INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS + }) + state.completedAt = recordedAt +} + +export async function runIncidentMonitor( + initialState: IncidentMonitorState, + dependencies: IncidentMonitorDependencies +): Promise { + const state = structuredClone(initialState) + const lineageStartMs = Date.parse(state.startedAt) + if (!Number.isFinite(lineageStartMs)) { + throw new Error('incident monitor start time is invalid') + } + const checkpoints = checkpointMinutes(state.durationMinutes) + const lineageDeadlineMs = state.preDrainDryRun + ? lineageStartMs + INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS + : Number.POSITIVE_INFINITY + const resumedAt = dependencies.now() + const priorSampleMs = state.lastSampleAt + ? Date.parse(state.lastSampleAt) + : lineageStartMs + const gapThreshold = state.intervalMs + if (resumedAt - priorSampleMs > gapThreshold) { + resetContinuousWindow(state, new Date(resumedAt).toISOString(), [{ + code: 'monitor_gap', + source: 'active-probe', + observed: resumedAt - priorSampleMs, + threshold: gapThreshold + }]) + } + if (state.completedAt !== null) { + await dependencies.persist(state) + return state + } + while (state.completedAt === null) { + if (dependencies.now() > lineageDeadlineMs) { + completeContinuityDeadline(state, dependencies.now(), lineageStartMs) + await dependencies.persist(state) + break + } + const sampleStartedAt = dependencies.now() + let evaluation: IncidentEvaluation + try { + evaluation = evaluateIncidentSample( + await dependencies.collect(), + dependencies.now(), + state.migrationPolicy, + state.recoverySourceCellId, + state.capacityCellId + ) + } catch { + evaluation = { + status: 'freeze', + evaluatedAt: new Date(dependencies.now()).toISOString(), + failures: [{ + code: 'collector_failed', + source: 'cloud-monitoring' + }] + } + } + state.totalSampleCount++ + state.lastSampleAt = evaluation.evaluatedAt + const continuityFailures = evaluation.failures.filter((failure) => + CONTINUITY_FAILURE_CODES.has(failure.code) + ) + const thresholdFailures = evaluation.failures.filter((failure) => + !CONTINUITY_FAILURE_CODES.has(failure.code) + ) + if (continuityFailures.length > 0) { + resetContinuousWindow(state, evaluation.evaluatedAt, continuityFailures) + } else { + if (state.windowStartedAt === null) { + state.windowStartedAt = evaluation.evaluatedAt + } + state.sampleCount++ + } + if (thresholdFailures.length > 0) { + state.frozenAt ??= evaluation.evaluatedAt + state.failures = [...state.failures, ...thresholdFailures] + } + if (state.windowStartedAt === null) { + if (dependencies.now() >= lineageDeadlineMs) { + completeContinuityDeadline(state, dependencies.now(), lineageStartMs) + await dependencies.persist(state) + break + } + await dependencies.persist(state) + await dependencies.wait( + Math.max(0, Math.min(state.intervalMs, lineageDeadlineMs - dependencies.now())) + ) + continue + } + const startMs = Date.parse(state.windowStartedAt) + const endMs = startMs + state.durationMinutes * 60_000 + const elapsedMinutes = (dependencies.now() - startMs) / 60_000 + while ( + state.nextCheckpointIndex < checkpoints.length && + elapsedMinutes >= checkpoints[state.nextCheckpointIndex]! + ) { + const minute = checkpoints[state.nextCheckpointIndex]! + await dependencies.checkpoint({ + schemaVersion: 4, + incidentId: state.incidentId, + environment: state.environment, + expectedSelector: state.expectedSelector, + preDrainDryRun: state.preDrainDryRun, + migrationPolicy: state.migrationPolicy, + recoverySourceCellId: state.recoverySourceCellId, + capacityCellId: state.capacityCellId, + windowSequence: state.windowSequence, + windowStartedAt: state.windowStartedAt, + checkpointMinute: minute, + scheduledAt: new Date(startMs + minute * 60_000).toISOString(), + recordedAt: new Date(dependencies.now()).toISOString(), + status: state.frozenAt ? 'freeze' : 'green', + frozenAt: state.frozenAt, + sampleCount: state.sampleCount, + failures: state.failures, + thresholds: INCIDENT_MONITOR_THRESHOLDS + }) + state.nextCheckpointIndex++ + } + if (state.preDrainDryRun && state.frozenAt !== null) { + state.completedAt = new Date(dependencies.now()).toISOString() + await dependencies.persist(state) + break + } + if (dependencies.now() >= endMs) { + state.completedAt = new Date(dependencies.now()).toISOString() + await dependencies.persist(state) + break + } + if (dependencies.now() >= lineageDeadlineMs) { + completeContinuityDeadline(state, dependencies.now(), lineageStartMs) + await dependencies.persist(state) + break + } + await dependencies.persist(state) + await dependencies.wait( + Math.max( + 0, + Math.min(sampleStartedAt + state.intervalMs, endMs, lineageDeadlineMs) - dependencies.now() + ) + ) + } + return state +} + +export function preDrainDryRunPassed(state: Pick< + IncidentMonitorState, + | 'completedAt' + | 'durationMinutes' + | 'frozenAt' + | 'intervalMs' + | 'preDrainDryRun' + | 'sampleCount' + | 'startedAt' +>): boolean { + const minimumSamples = + Math.ceil((state.durationMinutes * 60_000) / state.intervalMs) + 1 + const lineageElapsedMs = state.completedAt === null + ? Number.POSITIVE_INFINITY + : Date.parse(state.completedAt) - Date.parse(state.startedAt) + return ( + state.preDrainDryRun && + state.durationMinutes === 15 && + state.completedAt !== null && + lineageElapsedMs >= 0 && + lineageElapsedMs <= INCIDENT_PRE_DRAIN_MAX_LINEAGE_MS && + state.frozenAt === null && + state.sampleCount >= minimumSamples + ) +} diff --git a/cloud/apps/relay-ops/src/incident-selector.ts b/cloud/apps/relay-ops/src/incident-selector.ts new file mode 100644 index 00000000000..7e7ddd91e59 --- /dev/null +++ b/cloud/apps/relay-ops/src/incident-selector.ts @@ -0,0 +1,77 @@ +import { z } from 'zod' + +export const AdmissionStateSchema = z.enum([ + 'existing-only', + 'migration-only', + 'general' +]) + +export type AdmissionState = z.infer + +export const SelectorMembershipSchema = z.object({ + existingOnly: z.array(z.string()), + migrationOnly: z.array(z.string()), + general: z.array(z.string()) +}) + +export type SelectorMembership = z.infer + +export const AdmissionSelectorSchema = z.object({ + generation: z.number().int().nonnegative(), + membership: SelectorMembershipSchema +}) + +export type AdmissionSelector = z.infer + +export function normalizeSelectorMembership( + membership: SelectorMembership, + configuredCellIds: ReadonlySet +): SelectorMembership { + const normalized = { + existingOnly: [...membership.existingOnly].sort(), + migrationOnly: [...membership.migrationOnly].sort(), + general: [...membership.general].sort() + } + const all = [ + ...normalized.existingOnly, + ...normalized.migrationOnly, + ...normalized.general + ] + if ( + all.length !== configuredCellIds.size || + new Set(all).size !== all.length || + all.some((cellId) => !configuredCellIds.has(cellId)) + ) { + throw new Error('selector membership must contain every configured cell exactly once') + } + return normalized +} + +export function selectorCellState( + selector: AdmissionSelector, + cellId: string +): AdmissionState { + if (selector.membership.existingOnly.includes(cellId)) return 'existing-only' + if (selector.membership.migrationOnly.includes(cellId)) return 'migration-only' + if (selector.membership.general.includes(cellId)) return 'general' + throw new Error(`selector does not contain ${cellId}`) +} + +export function effectiveAdmissionState( + selector: AdmissionSelector, + legacyEnabled: boolean, + cellId: string +): AdmissionState { + if (selector.generation === 0) return legacyEnabled ? 'general' : 'existing-only' + return selectorCellState(selector, cellId) +} + +export function exactAdmissionSelector( + actual: AdmissionSelector, + expected: AdmissionSelector +): boolean { + return ( + actual.generation === expected.generation && + JSON.stringify(actual.membership) === JSON.stringify(expected.membership) + ) +} diff --git a/cloud/apps/relay-ops/src/index.ts b/cloud/apps/relay-ops/src/index.ts new file mode 100644 index 00000000000..0544cfe7595 --- /dev/null +++ b/cloud/apps/relay-ops/src/index.ts @@ -0,0 +1,108 @@ +import { randomBytes, timingSafeEqual } from 'node:crypto' +import { readFile } from 'node:fs/promises' +import { resolve } from 'node:path' +import { serve } from '@hono/node-server' +import { Hono } from 'hono' +import { z } from 'zod' +import { DashboardSnapshotCache } from './dashboard-snapshot.js' +import { createGcloudClient } from './gcloud-client.js' +import { + dispatchStagingPowerWorkflow, + parseStagingPowerRequest +} from './staging-workflow.js' + +const QuerySchema = z.object({ + environment: z.enum(['production', 'staging']).default('production'), + window: z.coerce.number().int().min(30).max(1440).default(360) +}) +const port = z.coerce.number().int().min(1024).max(65_535).parse(process.env.PORT ?? 2455) +const controlsEnabled = process.env.RELAY_OPS_ENABLE_STAGING_CONTROLS === '1' +const csrfToken = randomBytes(32).toString('base64url') +const publicDirectory = resolve(import.meta.dirname, '../public') +const cache = new DashboardSnapshotCache(createGcloudClient()) +const app = new Hono() + +function safeEqual(left: string, right: string): boolean { + const leftBuffer = Buffer.from(left) + const rightBuffer = Buffer.from(right) + return leftBuffer.length === rightBuffer.length && timingSafeEqual(leftBuffer, rightBuffer) +} + +app.use('*', async (context, next) => { + await next() + context.header('Cache-Control', 'no-store') + context.header('Content-Security-Policy', [ + "default-src 'self'", + "script-src 'self'", + "style-src 'self'", + "font-src 'self'", + "connect-src 'self'", + "img-src 'self' data:", + "object-src 'none'", + "base-uri 'none'", + "frame-ancestors 'none'", + "form-action 'self'" + ].join('; ')) + context.header('Referrer-Policy', 'no-referrer') + context.header('X-Content-Type-Options', 'nosniff') + context.header('X-Frame-Options', 'DENY') +}) + +app.get('/health', (context) => context.json({ status: 'ok' })) +app.get('/api/config', (context) => context.json({ + stagingControlsEnabled: controlsEnabled, + csrfToken: controlsEnabled ? csrfToken : null +})) +app.get('/api/snapshot', async (context) => { + const query = QuerySchema.safeParse(context.req.query()) + if (!query.success) return context.json({ error: 'Invalid dashboard query' }, 400) + try { + return context.json(await cache.read(query.data.environment, query.data.window)) + } catch { + return context.json({ + error: 'Relay operations data is unavailable. Check local gcloud and gh authentication.' + }, 503) + } +}) +app.post('/api/staging/power', async (context) => { + if (!controlsEnabled) return context.json({ error: 'Staging controls are disabled' }, 403) + const origin = context.req.header('origin') + const expectedOrigin = `http://127.0.0.1:${port}` + if (origin !== expectedOrigin) return context.json({ error: 'Origin rejected' }, 403) + if (!safeEqual(context.req.header('x-csrf-token') ?? '', csrfToken)) { + return context.json({ error: 'Request token rejected' }, 403) + } + try { + const request = parseStagingPowerRequest(await context.req.json()) + await dispatchStagingPowerWorkflow(request) + return context.json({ accepted: true }) + } catch { + return context.json({ error: 'Invalid or failed staging workflow dispatch' }, 400) + } +}) + +const staticTypes: Record = { + '/app.js': 'text/javascript; charset=utf-8', + '/styles.css': 'text/css; charset=utf-8' +} +for (const [route, contentType] of Object.entries(staticTypes)) { + app.get(route, async (context) => { + const file = route.slice(1) + try { + const content = await readFile(resolve(publicDirectory, file)) + return context.body(content, 200, { 'Content-Type': contentType }) + } catch { + return context.notFound() + } + }) +} +app.get('/', async (context) => { + const html = await readFile(resolve(publicDirectory, 'index.html'), 'utf8') + return context.html(html) +}) + +serve({ fetch: app.fetch, hostname: '127.0.0.1', port }, () => { + // Loopback is intentional; operators may add authenticated Tailscale Serve separately. + console.log(`Orca Relay Operations: http://127.0.0.1:${port}`) + console.log(`Staging controls: ${controlsEnabled ? 'enabled through GitHub workflow' : 'read-only'}`) +}) diff --git a/cloud/apps/relay-ops/src/monitoring-snapshot.test.ts b/cloud/apps/relay-ops/src/monitoring-snapshot.test.ts new file mode 100644 index 00000000000..3bcd40c3c4b --- /dev/null +++ b/cloud/apps/relay-ops/src/monitoring-snapshot.test.ts @@ -0,0 +1,85 @@ +import { describe, expect, it } from 'vitest' +import { RELAY_METRICS, readMonitoringSnapshot } from './monitoring-snapshot.js' +import type { GcloudClient } from './gcloud-client.js' +import { RELAY_OPS_ENVIRONMENTS } from './environment-config.js' + +const gcloud: GcloudClient = { + accessToken: async () => 'a'.repeat(40) +} + +function distribution(at: string, mean: number | undefined, count = 1) { + return { + interval: { endTime: at }, + value: { distributionValue: { count, ...(mean === undefined ? {} : { mean }) } } + } +} + +describe('readMonitoringSnapshot', () => { + it('aggregates gauge series per minute and distribution deltas by count', async () => { + const fetchImpl: typeof fetch = async (input) => { + const url = new URL(String(input)) + if (url.pathname.endsWith('/alertPolicies')) { + return Response.json({ alertPolicies: [] }) + } + const filter = url.searchParams.get('filter') ?? '' + if (filter.includes('orca_relay_controls')) { + return Response.json({ timeSeries: [ + { + metric: { labels: { cell_id: 'production-gce-c1' } }, + resource: { type: 'gce_instance', labels: { instance_id: 'one' } }, + points: [ + distribution('2026-07-15T12:00:10Z', 1), + distribution('2026-07-15T12:00:50Z', 2) + ] + }, + { + metric: { labels: { cell_id: 'production-gce-c1' } }, + resource: { type: 'gce_instance', labels: { instance_id: 'two' } }, + points: [distribution('2026-07-15T12:00:20Z', 3)] + }, + { + metric: { labels: { cell_id: 'production-gce-c1' } }, + resource: { type: 'gce_instance', labels: { instance_id: 'stale' } }, + points: [distribution('2026-07-15T11:59:20Z', 100)] + } + ] }) + } + if (filter.includes('orca_relay_forwarded_bytes')) { + return Response.json({ timeSeries: [{ + metric: { labels: { cell_id: 'production-gce-c1' } }, + resource: { type: 'gce_instance', labels: { instance_id: 'one' } }, + points: [distribution('2026-07-15T12:00:20Z', 10, 4)] + }] }) + } + return Response.json({ timeSeries: [] }) + } + + const result = await readMonitoringSnapshot(RELAY_OPS_ENVIRONMENTS.production, gcloud, { + now: new Date('2026-07-15T12:01:00Z'), + windowMinutes: 30, + fetchImpl + }) + + expect(result.warnings).toEqual([]) + expect(result.metrics.controls.points).toEqual([ + { at: '2026-07-15T11:59:00.000Z', value: 100 }, + { at: '2026-07-15T12:00:00.000Z', value: 5 } + ]) + expect(result.metrics.controls.latestByCell).toEqual({ 'production-gce-c1': 5 }) + expect(result.metrics.forwarded_bytes.latest).toBe(40) + expect(result.metrics.postgres_retries.available).toBe(true) + expect(Object.keys(result.metrics)).toHaveLength(RELAY_METRICS.length) + }) + + it('degrades safely when credentials are unavailable', async () => { + const unavailable: GcloudClient = { + accessToken: async () => { throw new Error('sensitive context') } + } + const result = await readMonitoringSnapshot(RELAY_OPS_ENVIRONMENTS.production, unavailable) + expect(result.warnings).toEqual([ + 'Cloud Monitoring credentials are unavailable. Run gcloud auth login.' + ]) + expect(result.metrics.postgres_retries.available).toBe(false) + expect(JSON.stringify(result)).not.toContain('sensitive context') + }) +}) diff --git a/cloud/apps/relay-ops/src/monitoring-snapshot.ts b/cloud/apps/relay-ops/src/monitoring-snapshot.ts new file mode 100644 index 00000000000..9b5fe44bd4a --- /dev/null +++ b/cloud/apps/relay-ops/src/monitoring-snapshot.ts @@ -0,0 +1,379 @@ +import { z } from 'zod' +import type { RelayOpsEnvironment } from './environment-config.js' +import type { GcloudClient } from './gcloud-client.js' + +type MetricMode = 'gauge-sum' | 'delta-sum' | 'maximum' + +export type RelayMetricName = + | 'total_connections' + | 'controls' + | 'splices' + | 'pending_splices' + | 'queued_bytes' + | 'http_latency_ms' + | 'sql_latency_ms' + | 'heap_used_bytes' + | 'event_loop_ms_p99' + | 'forwarded_bytes' + | 'auth_successes' + | 'auth_failures' + | 'reconnects' + | 'sql_queries' + | 'sql_failures' + | 'assignment_5xx' + | 'postgres_retries' + | 'postgres_retry_exhausted' + | 'db_pool_total' + | 'db_pool_idle' + | 'db_pool_waiting' + | 'db_waiters_max' + | 'db_oldest_wait_ms' + | 'db_wait_ms_max' + +type MetricDefinition = { + name: RelayMetricName + label: string + unit: 'count' | 'bytes' | 'milliseconds' + mode: MetricMode +} + +export const RELAY_METRICS: MetricDefinition[] = [ + { name: 'total_connections', label: 'Connections', unit: 'count', mode: 'gauge-sum' }, + { name: 'controls', label: 'Desktop controls', unit: 'count', mode: 'gauge-sum' }, + { name: 'splices', label: 'Phone splices', unit: 'count', mode: 'gauge-sum' }, + { name: 'pending_splices', label: 'Pending splices', unit: 'count', mode: 'gauge-sum' }, + { name: 'queued_bytes', label: 'Queued bytes', unit: 'bytes', mode: 'maximum' }, + { name: 'http_latency_ms', label: 'HTTP latency', unit: 'milliseconds', mode: 'maximum' }, + { name: 'sql_latency_ms', label: 'SQL latency', unit: 'milliseconds', mode: 'maximum' }, + { name: 'heap_used_bytes', label: 'Heap used', unit: 'bytes', mode: 'maximum' }, + { + name: 'event_loop_ms_p99', + label: 'Event-loop p99', + unit: 'milliseconds', + mode: 'maximum' + }, + { name: 'forwarded_bytes', label: 'Forwarded bytes', unit: 'bytes', mode: 'delta-sum' }, + { name: 'auth_successes', label: 'Auth successes', unit: 'count', mode: 'delta-sum' }, + { name: 'auth_failures', label: 'Auth failures', unit: 'count', mode: 'delta-sum' }, + { name: 'reconnects', label: 'Reconnects', unit: 'count', mode: 'delta-sum' }, + { name: 'sql_queries', label: 'SQL queries', unit: 'count', mode: 'delta-sum' }, + { name: 'sql_failures', label: 'SQL failures', unit: 'count', mode: 'delta-sum' }, + { name: 'assignment_5xx', label: 'Assignment 5xx', unit: 'count', mode: 'delta-sum' }, + { + name: 'postgres_retries', + label: 'PostgreSQL retries', + unit: 'count', + mode: 'delta-sum' + }, + { + name: 'postgres_retry_exhausted', + label: 'PostgreSQL retry exhausted', + unit: 'count', + mode: 'delta-sum' + }, + { name: 'db_pool_total', label: 'Database pool total', unit: 'count', mode: 'gauge-sum' }, + { name: 'db_pool_idle', label: 'Database pool idle', unit: 'count', mode: 'gauge-sum' }, + { + name: 'db_pool_waiting', + label: 'Database pool waiting', + unit: 'count', + mode: 'gauge-sum' + }, + { + name: 'db_waiters_max', + label: 'Database waiters max', + unit: 'count', + mode: 'maximum' + }, + { + name: 'db_oldest_wait_ms', + label: 'Database oldest wait', + unit: 'milliseconds', + mode: 'maximum' + }, + { + name: 'db_wait_ms_max', + label: 'Database wait max', + unit: 'milliseconds', + mode: 'maximum' + } +] + +const NumericSchema = z.union([z.number(), z.string()]).transform((value) => Number(value)) +const DistributionSchema = z.object({ + count: NumericSchema.default(0), + mean: NumericSchema.optional() +}) +const PointSchema = z.object({ + interval: z.object({ endTime: z.string() }), + value: z.object({ + doubleValue: NumericSchema.optional(), + int64Value: NumericSchema.optional(), + distributionValue: DistributionSchema.optional() + }) +}) +const TimeSeriesSchema = z.object({ + metric: z.object({ labels: z.record(z.string()).default({}) }), + resource: z.object({ type: z.string(), labels: z.record(z.string()).default({}) }), + points: z.array(PointSchema).default([]) +}) +const TimeSeriesResponseSchema = z.object({ + timeSeries: z.array(TimeSeriesSchema).default([]) +}) + +export type MetricPoint = { at: string; value: number } + +export type RelayMetricSnapshot = MetricDefinition & { + available: boolean + points: MetricPoint[] + latest: number | null + latestAt: string | null + latestByCell: Record +} + +export type AlertPolicySnapshot = { + id: string + displayName: string + enabled: boolean + documentation: string | null +} + +export type MonitoringSnapshot = { + startAt: string + endAt: string + resolutionSeconds: number + metrics: Record + alertPolicies: AlertPolicySnapshot[] + warnings: string[] +} + +type ParsedPoint = { + atMs: number + bucketMs: number + value: number + sampleTotal: number + seriesKey: string + cellId: string +} + +function pointValue(point: z.infer): { value: number; sampleTotal: number } { + if (point.value.distributionValue) { + const count = point.value.distributionValue.count + const value = point.value.distributionValue.mean ?? 0 + return { value, sampleTotal: value * count } + } + const value = point.value.doubleValue ?? point.value.int64Value ?? 0 + return { value, sampleTotal: value } +} + +function parsePoints(series: z.infer[]): ParsedPoint[] { + return series.flatMap((entry) => { + const cellId = entry.metric.labels.cell_id ?? 'unknown' + const resourceId = + entry.resource.labels.instance_id ?? entry.resource.labels.revision_name ?? entry.resource.type + const seriesKey = `${cellId}:${resourceId}` + return entry.points.flatMap((point) => { + const atMs = Date.parse(point.interval.endTime) + if (!Number.isFinite(atMs)) return [] + const values = pointValue(point) + return [{ + atMs, + bucketMs: Math.floor(atMs / 60_000) * 60_000, + value: values.value, + sampleTotal: values.sampleTotal, + seriesKey, + cellId + }] + }) + }) +} + +function aggregatePoints(points: ParsedPoint[], mode: MetricMode): MetricPoint[] { + if (mode === 'gauge-sum') { + const buckets = new Map>() + for (const point of points) { + const bySeries = buckets.get(point.bucketMs) ?? new Map() + const previous = bySeries.get(point.seriesKey) + if (!previous || previous.atMs < point.atMs) bySeries.set(point.seriesKey, point) + buckets.set(point.bucketMs, bySeries) + } + return [...buckets.entries()] + .sort(([left], [right]) => left - right) + .map(([at, bySeries]) => ({ + at: new Date(at).toISOString(), + value: [...bySeries.values()].reduce((total, point) => total + point.value, 0) + })) + } + const buckets = new Map() + for (const point of points) { + const value = mode === 'delta-sum' ? point.sampleTotal : point.value + const previous = buckets.get(point.bucketMs) + buckets.set( + point.bucketMs, + mode === 'maximum' ? Math.max(previous ?? 0, value) : (previous ?? 0) + value + ) + } + return [...buckets.entries()] + .sort(([left], [right]) => left - right) + .map(([at, value]) => ({ at: new Date(at).toISOString(), value })) +} + +function latestByCell(points: ParsedPoint[], mode: MetricMode): Record { + const newestBucketByCell = new Map() + for (const point of points) { + newestBucketByCell.set( + point.cellId, + Math.max(newestBucketByCell.get(point.cellId) ?? 0, point.bucketMs) + ) + } + const newestBySeries = new Map() + for (const point of points) { + if (point.bucketMs !== newestBucketByCell.get(point.cellId)) continue + const previous = newestBySeries.get(point.seriesKey) + if (!previous || previous.atMs < point.atMs) newestBySeries.set(point.seriesKey, point) + } + const totals = new Map() + for (const point of newestBySeries.values()) { + const value = mode === 'delta-sum' ? point.sampleTotal : point.value + const previous = totals.get(point.cellId) + totals.set( + point.cellId, + mode === 'maximum' ? Math.max(previous ?? 0, value) : (previous ?? 0) + value + ) + } + return Object.fromEntries(totals) +} + +async function monitoringRequest( + fetchImpl: typeof fetch, + token: string, + url: URL +): Promise { + const response = await fetchImpl(url, { + headers: { authorization: `Bearer ${token}` }, + signal: AbortSignal.timeout(30_000) + }) + if (!response.ok) throw new Error(`Cloud Monitoring returned ${response.status}`) + return await response.json() +} + +async function readMetric( + environment: RelayOpsEnvironment, + definition: MetricDefinition, + token: string, + startAt: string, + endAt: string, + fetchImpl: typeof fetch +): Promise { + const url = new URL( + `https://monitoring.googleapis.com/v3/projects/${environment.project}/timeSeries` + ) + url.searchParams.set( + 'filter', + `metric.type="logging.googleapis.com/user/orca_relay_${definition.name}"` + ) + url.searchParams.set('interval.startTime', startAt) + url.searchParams.set('interval.endTime', endAt) + url.searchParams.set('view', 'FULL') + url.searchParams.set('pageSize', '1000') + const body = TimeSeriesResponseSchema.parse( + await monitoringRequest(fetchImpl, token, url) + ) + const parsed = parsePoints(body.timeSeries) + const points = aggregatePoints(parsed, definition.mode) + const latest = points.at(-1) ?? null + return { + ...definition, + available: true, + points, + latest: latest?.value ?? null, + latestAt: latest?.at ?? null, + latestByCell: latestByCell(parsed, definition.mode) + } +} + +async function readAlertPolicies( + environment: RelayOpsEnvironment, + token: string, + fetchImpl: typeof fetch +): Promise { + const url = new URL( + `https://monitoring.googleapis.com/v3/projects/${environment.project}/alertPolicies` + ) + url.searchParams.set('pageSize', '100') + const body = (await monitoringRequest(fetchImpl, token, url)) as { + alertPolicies?: Array<{ + name?: string + displayName?: string + enabled?: boolean + documentation?: { content?: string } + }> + } + return (body.alertPolicies ?? []) + .filter((policy) => policy.displayName?.startsWith('Orca Relay:')) + .map((policy) => ({ + id: policy.name ?? '', + displayName: policy.displayName ?? 'Orca Relay alert', + enabled: policy.enabled === true, + documentation: policy.documentation?.content ?? null + })) + .sort((left, right) => left.displayName.localeCompare(right.displayName)) +} + +function emptyMetric(definition: MetricDefinition): RelayMetricSnapshot { + return { + ...definition, + available: false, + points: [], + latest: null, + latestAt: null, + latestByCell: {} + } +} + +export async function readMonitoringSnapshot( + environment: RelayOpsEnvironment, + gcloud: GcloudClient, + options: { now?: Date; windowMinutes?: number; fetchImpl?: typeof fetch } = {} +): Promise { + const now = options.now ?? new Date() + const windowMinutes = Math.min(24 * 60, Math.max(30, options.windowMinutes ?? 360)) + const endAt = now.toISOString() + const startAt = new Date(now.getTime() - windowMinutes * 60_000).toISOString() + const fetchImpl = options.fetchImpl ?? fetch + const warnings: string[] = [] + let token: string + try { + token = await gcloud.accessToken() + } catch { + return { + startAt, + endAt, + resolutionSeconds: 60, + metrics: Object.fromEntries( + RELAY_METRICS.map((definition) => [definition.name, emptyMetric(definition)]) + ) as Record, + alertPolicies: [], + warnings: ['Cloud Monitoring credentials are unavailable. Run gcloud auth login.'] + } + } + const settled = await Promise.allSettled( + RELAY_METRICS.map((definition) => + readMetric(environment, definition, token, startAt, endAt, fetchImpl) + ) + ) + const metrics = {} as Record + settled.forEach((result, index) => { + const definition = RELAY_METRICS[index]! + if (result.status === 'fulfilled') metrics[definition.name] = result.value + else { + metrics[definition.name] = emptyMetric(definition) + warnings.push(`${definition.label} metric is unavailable.`) + } + }) + const alertPolicies = await readAlertPolicies(environment, token, fetchImpl).catch(() => { + warnings.push('Cloud Monitoring alert policies are unavailable.') + return [] + }) + return { startAt, endAt, resolutionSeconds: 60, metrics, alertPolicies, warnings } +} diff --git a/cloud/apps/relay-ops/src/relay-repository.test.ts b/cloud/apps/relay-ops/src/relay-repository.test.ts new file mode 100644 index 00000000000..e6a87c35e82 --- /dev/null +++ b/cloud/apps/relay-ops/src/relay-repository.test.ts @@ -0,0 +1,42 @@ +import { readdirSync, readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import { + RELAY_GITHUB_REPOSITORY, + RELAY_WORKFLOW_FILE_PREFIX, + relayRepositoryApiPath, + relayWorkflowFile +} from './relay-repository.js' + +const sourceDir = fileURLToPath(new URL('.', import.meta.url)) +const sources = readdirSync(sourceDir) + .filter((name) => name.endsWith('.ts') && name !== 'relay-repository.ts') + .map((name) => ({ name, text: readFileSync(`${sourceDir}${name}`, 'utf8') })) + +describe('relay repository identity', () => { + it('builds API paths and workflow filenames from the one repository name', () => { + expect(relayRepositoryApiPath('actions/runs')).toBe(`repos/${RELAY_GITHUB_REPOSITORY}/actions/runs`) + expect(relayWorkflowFile('power-relay-staging.yml')).toBe( + `${RELAY_WORKFLOW_FILE_PREFIX}power-relay-staging.yml` + ) + }) + + // Why: the public-repo copy renames the repository and prefixes every workflow file. Both have to + // be one edit, so no other module may restate either. + it('is the only module naming a GitHub repository', () => { + for (const { name, text } of sources) { + expect(text, `${name} restates a GitHub repository`).not.toMatch(/stablyai\//) + } + }) + + it('is the only module naming a workflow file', () => { + for (const { name, text } of sources) { + for (const match of text.matchAll(/'([^']*\.yml)'/g)) { + const file = match[1] ?? '' + expect(text, `${name} names ${file} outside relayWorkflowFile`).toMatch( + new RegExp(`relayWorkflowFile\\('${file.replaceAll('.', '\\.')}'\\)`) + ) + } + } + }) +}) diff --git a/cloud/apps/relay-ops/src/relay-repository.ts b/cloud/apps/relay-ops/src/relay-repository.ts new file mode 100644 index 00000000000..85158670adb --- /dev/null +++ b/cloud/apps/relay-ops/src/relay-repository.ts @@ -0,0 +1,14 @@ +// Single place naming the GitHub repository that holds the Relay workflows. When the Relay tree is +// copied to its public repository, only this file changes: the repository moves and every workflow +// file gains a prefix, while the workflow display names stay as they are. +export const RELAY_GITHUB_REPOSITORY = 'stablyai/orca-cloud' + +export const RELAY_WORKFLOW_FILE_PREFIX = '' + +export function relayWorkflowFile(name: string): string { + return `${RELAY_WORKFLOW_FILE_PREFIX}${name}` +} + +export function relayRepositoryApiPath(resource: string): string { + return `repos/${RELAY_GITHUB_REPOSITORY}/${resource}` +} diff --git a/cloud/apps/relay-ops/src/resource-inventory.test.ts b/cloud/apps/relay-ops/src/resource-inventory.test.ts new file mode 100644 index 00000000000..6d8b3070c98 --- /dev/null +++ b/cloud/apps/relay-ops/src/resource-inventory.test.ts @@ -0,0 +1,150 @@ +import { describe, expect, it } from 'vitest' +import { RELAY_OPS_ENVIRONMENTS } from './environment-config.js' +import type { GcloudClient } from './gcloud-client.js' +import { probeEndpointHealth, readResourceInventory } from './resource-inventory.js' + +const digest = `sha256:${'a'.repeat(64)}` +const runService = { + template: { + scaling: { minInstanceCount: 0, maxInstanceCount: 2 }, + containers: [{ image: 'registry/image:tag' }] + }, + conditions: [{ state: 'CONDITION_SUCCEEDED' }], + latestReadyRevision: 'projects/project/revisions/revision-one' +} + +describe('readResourceInventory', () => { + it('does not delay a healthy endpoint sample', async () => { + let calls = 0 + let waits = 0 + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async () => { + calls += 1 + return new Response(null, { status: 200 }) + }, + async () => { + waits += 1 + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(true) + expect(calls).toBe(2) + expect(waits).toBe(0) + }) + + it('retries one transient endpoint failure within the same sample', async () => { + const calls = new Map() + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async (input) => { + const path = new URL(String(input)).pathname + const call = (calls.get(path) ?? 0) + 1 + calls.set(path, call) + return new Response(null, { status: path === '/ready' && call === 1 ? 503 : 200 }) + }, + async (ms) => { + waits.push(ms) + } + ) + + expect(result.health).toBe(true) + expect(result.ready).toBe(true) + expect(calls).toEqual(new Map([['/health', 2], ['/ready', 2]])) + expect(waits).toEqual([11_000]) + }) + + it('fails closed when the endpoint retry is also unhealthy', async () => { + let calls = 0 + const waits: number[] = [] + const result = await probeEndpointHealth( + 'https://c9.relay.onorca.dev', + async () => { + calls += 1 + return new Response(null, { status: 503 }) + }, + async (ms) => { + waits.push(ms) + } + ) + + expect(result.health).toBe(false) + expect(result.ready).toBe(false) + expect(calls).toBe(4) + expect(waits).toEqual([11_000]) + }) + + it('uses aggregate REST inventory without probing sleeping staging endpoints', async () => { + const gcloud: GcloudClient = { accessToken: async () => 'a'.repeat(40) } + let publicProbeCalls = 0 + const fetchImpl: typeof fetch = async (input) => { + const url = new URL(String(input)) + if (url.hostname.endsWith('onorca.dev')) { + publicProbeCalls += 1 + return Response.json({ status: 'ok' }) + } + if (url.hostname === 'run.googleapis.com') return Response.json(runService) + if (url.hostname === 'sqladmin.googleapis.com') return Response.json({ + state: 'STOPPED', + databaseVersion: 'POSTGRES_17', + settings: { + activationPolicy: 'NEVER', + availabilityType: 'ZONAL', + tier: 'db-custom-1-3840' + } + }) + if (url.hostname === 'certificatemanager.googleapis.com') return Response.json({ + managed: { domains: ['*.relay-staging.onorca.dev'], state: 'ACTIVE' } + }) + if (url.pathname.includes('/instanceGroupManagers/')) { + const name = url.pathname.split('/').at(-1)! + return Response.json({ + name, + targetSize: 0, + size: '0', + instanceGroup: `projects/project/zones/zone/instanceGroups/${name}`, + instanceTemplate: `projects/project/global/instanceTemplates/template-${name}`, + status: { isStable: true } + }) + } + if (url.pathname.includes('/instanceTemplates/')) return Response.json({ + properties: { metadata: { items: [{ + key: 'startup-script', + value: `SECRET_TEXT\nORCA_RELAY_IMAGE_DIGEST=%s\\n' '${digest}'` + }] } } + }) + if (url.pathname.endsWith('/getHealth')) return Response.json([]) + throw new Error(`Unexpected request to ${url.hostname}${url.pathname}`) + } + + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.staging, + gcloud, + fetchImpl + ) + + expect(publicProbeCalls).toBe(0) + expect(result.cells.every((cell) => cell.targetSize === 0)).toBe(true) + expect(result.cells.every((cell) => cell.endpoint.health === null)).toBe(true) + expect(result.cells.every((cell) => cell.imageDigest === digest)).toBe(true) + expect(JSON.stringify(result)).not.toContain('SECRET_TEXT') + }) + + it('represents missing credentials as unknown inventory, never sleeping', async () => { + const gcloud: GcloudClient = { + accessToken: async () => { throw new Error('sensitive context') } + } + let fetchCalls = 0 + const result = await readResourceInventory( + RELAY_OPS_ENVIRONMENTS.production, + gcloud, + async () => { fetchCalls += 1; return Response.json({}) } + ) + expect(fetchCalls).toBe(0) + expect(result.cells.every((cell) => cell.targetSize === null)).toBe(true) + expect(result.cells.every((cell) => cell.backendHealth === 'unknown')).toBe(true) + expect(JSON.stringify(result)).not.toContain('sensitive context') + }) +}) diff --git a/cloud/apps/relay-ops/src/resource-inventory.ts b/cloud/apps/relay-ops/src/resource-inventory.ts new file mode 100644 index 00000000000..62ed3fd862b --- /dev/null +++ b/cloud/apps/relay-ops/src/resource-inventory.ts @@ -0,0 +1,366 @@ +import { z } from 'zod' +import type { RelayOpsEnvironment, RelayOpsCellConfig } from './environment-config.js' +import type { GcloudClient } from './gcloud-client.js' +import { INCIDENT_MONITOR_THRESHOLDS } from './incident-monitor.js' + +const RunServiceSchema = z.object({ + template: z.object({ + scaling: z.object({ + minInstanceCount: z.number().optional(), + maxInstanceCount: z.number().optional() + }).optional(), + containers: z.array(z.object({ image: z.string() })).min(1) + }), + conditions: z.array(z.object({ state: z.string() })).default([]), + latestReadyRevision: z.string().optional() +}) + +const SqlInstanceSchema = z.object({ + state: z.string(), + databaseVersion: z.string(), + settings: z.object({ + activationPolicy: z.string(), + availabilityType: z.string().optional(), + tier: z.string() + }) +}) + +const MigSchema = z.object({ + name: z.string(), + targetSize: z.number(), + size: z.union([z.string(), z.number()]).transform(Number).optional(), + instanceGroup: z.string(), + instanceTemplate: z.string(), + status: z.object({ isStable: z.boolean().default(false) }).default({ isStable: false }) +}) + +const TemplateSchema = z.object({ + properties: z.object({ + metadata: z.object({ + items: z.array(z.object({ key: z.string(), value: z.string().optional() })).default([]) + }).optional() + }) +}) + +const BackendHealthGroupSchema = z.object({ + healthStatus: z.array(z.object({ healthState: z.string() })).default([]) +}) +const BackendHealthSchema = z.union([ + BackendHealthGroupSchema, + z.array(z.object({ status: BackendHealthGroupSchema })) +]) + +const CertificateSchema = z.object({ + expireTime: z.string().optional(), + managed: z.object({ + domains: z.array(z.string()).default([]), + state: z.string() + }) +}) + +export type EndpointHealth = { + health: boolean | null + ready: boolean | null + latencyMs: number | null +} + +export type ServiceInventory = { + ready: boolean + revision: string | null + image: string + minInstances: number + maxInstances: number +} + +export type CellInventory = RelayOpsCellConfig & { + migName: string + targetSize: number | null + runningInstances: number | null + stable: boolean | null + template: string | null + imageDigest: string | null + backendHealth: 'healthy' | 'unhealthy' | 'empty' | 'unknown' + endpoint: EndpointHealth +} + +export type ResourceInventory = { + director: ServiceInventory | null + auth: ServiceInventory | null + sql: { + state: string + activationPolicy: string + tier: string + availabilityType: string + databaseVersion: string + } | null + certificate: { state: string; domains: string[]; expireTime: string | null } | null + directorEndpoint: EndpointHealth + authEndpoint: EndpointHealth + cells: CellInventory[] + warnings: string[] +} + +const unavailableEndpoint = (): EndpointHealth => ({ health: null, ready: null, latencyMs: null }) +const independentEndpointRetryDelayMs = 11_000 + +function finalSegment(value: string): string { + return value.split('/').at(-1) ?? value +} + +function parseService(value: unknown): ServiceInventory { + const service = RunServiceSchema.parse(value) + return { + ready: service.conditions.length > 0 && service.conditions.every( + (condition) => condition.state === 'CONDITION_SUCCEEDED' + ), + revision: service.latestReadyRevision ? finalSegment(service.latestReadyRevision) : null, + image: service.template.containers[0]!.image, + minInstances: service.template.scaling?.minInstanceCount ?? 0, + maxInstances: service.template.scaling?.maxInstanceCount ?? 0 + } +} + +async function googleRequest( + fetchImpl: typeof fetch, + token: string, + url: string, + init: RequestInit = {} +): Promise { + const response = await fetchImpl(url, { + ...init, + headers: { + authorization: `Bearer ${token}`, + ...(init.body ? { 'content-type': 'application/json' } : {}) + }, + signal: AbortSignal.timeout(30_000) + }) + if (!response.ok) throw new Error(`Google API returned ${response.status}`) + return await response.json() +} + +async function endpointProbe(origin: string, fetchImpl: typeof fetch): Promise { + const startedAt = performance.now() + const check = async (path: '/health' | '/ready'): Promise => { + try { + const response = await fetchImpl(`${origin}${path}`, { + redirect: 'error', + signal: AbortSignal.timeout(8_000) + }) + return response.ok + } catch { + return false + } + } + const [health, ready] = await Promise.all([check('/health'), check('/ready')]) + return { health, ready, latencyMs: Math.round(performance.now() - startedAt) } +} + +export async function probeEndpointHealth( + origin: string, + fetchImpl: typeof fetch, + wait: (ms: number) => Promise = async (ms) => + await new Promise((resolvePromise) => setTimeout(resolvePromise, ms)) +): Promise { + const first = await endpointProbe(origin, fetchImpl) + if ( + first.health && + first.ready && + first.latencyMs !== null && + first.latencyMs <= INCIDENT_MONITOR_THRESHOLDS.endpointLatencyMs + ) { + return first + } + // Outwait Relay's ten-second readiness cache before treating the retry as independent. + await wait(independentEndpointRetryDelayMs) + return await endpointProbe(origin, fetchImpl) +} + +function imageDigest(template: z.infer): string | null { + const startupScript = template.properties.metadata?.items.find( + (item) => item.key === 'startup-script' + )?.value + // Return only the immutable digest; startup metadata contains secret names and operational detail. + return startupScript?.match(/ORCA_RELAY_IMAGE_DIGEST=%s\\n' '(sha256:[a-f0-9]{64})'/)?.[1] ?? null +} + +function backendState(value: unknown): CellInventory['backendHealth'] { + const parsedHealth = BackendHealthSchema.parse(value) + const groups = Array.isArray(parsedHealth) + ? parsedHealth.map((group) => group.status) + : [parsedHealth] + const states = groups.flatMap((group) => group.healthStatus.map((status) => status.healthState)) + if (states.length === 0) return 'empty' + if (states.every((state) => state === 'HEALTHY')) return 'healthy' + return 'unhealthy' +} + +function unavailableCell(cell: RelayOpsCellConfig, migName: string): CellInventory { + return { + ...cell, + migName, + targetSize: null, + runningInstances: null, + stable: null, + template: null, + imageDigest: null, + backendHealth: 'unknown', + endpoint: unavailableEndpoint() + } +} + +async function readCell( + environment: RelayOpsEnvironment, + cell: RelayOpsCellConfig, + mig: z.infer | null, + token: string, + fetchImpl: typeof fetch +): Promise { + const migName = `${environment.migPrefix}${cell.hostname}` + if (!mig) return unavailableCell(cell, migName) + // An empty fixed-one MIG cannot serve and must never be woken by observation. + const endpoint = mig.targetSize > 0 + ? await probeEndpointHealth(cell.origin, fetchImpl) + : unavailableEndpoint() + const templateName = finalSegment(mig.instanceTemplate) + const [templateResult, healthResult] = await Promise.allSettled([ + googleRequest( + fetchImpl, + token, + `https://compute.googleapis.com/compute/v1/projects/${environment.project}/global/instanceTemplates/${templateName}` + ), + googleRequest( + fetchImpl, + token, + `https://compute.googleapis.com/compute/v1/projects/${environment.project}/global/backendServices/${migName}/getHealth`, + { method: 'POST', body: JSON.stringify({ group: mig.instanceGroup }) } + ) + ]) + return { + ...cell, + migName, + targetSize: mig.targetSize, + runningInstances: mig.size ?? (mig.status.isStable ? mig.targetSize : null), + stable: mig.status.isStable, + template: templateName, + imageDigest: templateResult.status === 'fulfilled' + ? imageDigest(TemplateSchema.parse(templateResult.value)) + : null, + backendHealth: healthResult.status === 'fulfilled' + ? backendState(healthResult.value) + : 'unknown', + endpoint + } +} + +function parsed( + result: PromiseSettledResult, + schema: S, + warning: string, + warnings: string[] +): z.infer | null { + if (result.status === 'rejected') { + warnings.push(warning) + return null + } + const parsedValue = schema.safeParse(result.value) + if (!parsedValue.success) { + warnings.push(warning) + return null + } + return parsedValue.data +} + +function unavailableInventory(environment: RelayOpsEnvironment, warning: string): ResourceInventory { + return { + director: null, + auth: null, + sql: null, + certificate: null, + directorEndpoint: unavailableEndpoint(), + authEndpoint: unavailableEndpoint(), + cells: environment.cells.map((cell) => + unavailableCell(cell, `${environment.migPrefix}${cell.hostname}`) + ), + warnings: [warning] + } +} + +export async function readResourceInventory( + environment: RelayOpsEnvironment, + gcloud: GcloudClient, + fetchImpl: typeof fetch = fetch +): Promise { + let token: string + try { + token = await gcloud.accessToken() + } catch { + return unavailableInventory( + environment, + 'Google Cloud credentials are unavailable. Run gcloud auth login.' + ) + } + const runUrl = (service: string) => + `https://run.googleapis.com/v2/projects/${environment.project}/locations/${environment.region}/services/${service}` + const migUrl = (cell: RelayOpsCellConfig) => + `https://compute.googleapis.com/compute/v1/projects/${environment.project}/zones/${cell.zone}/instanceGroupManagers/${environment.migPrefix}${cell.hostname}` + const settled = await Promise.allSettled([ + googleRequest(fetchImpl, token, runUrl(environment.directorService)), + googleRequest(fetchImpl, token, runUrl(environment.authService)), + googleRequest( + fetchImpl, + token, + `https://sqladmin.googleapis.com/sql/v1beta4/projects/${environment.project}/instances/${environment.sqlInstance}` + ), + googleRequest( + fetchImpl, + token, + `https://certificatemanager.googleapis.com/v1/projects/${environment.project}/locations/global/certificates/${environment.certificateName}` + ), + ...environment.cells.map((cell) => googleRequest(fetchImpl, token, migUrl(cell))) + ]) + const warnings: string[] = [] + const directorValue = parsed(settled[0]!, RunServiceSchema, 'Director service inventory is unavailable.', warnings) + const authValue = parsed(settled[1]!, RunServiceSchema, 'Auth service inventory is unavailable.', warnings) + const sqlValue = parsed(settled[2]!, SqlInstanceSchema, 'Cloud SQL inventory is unavailable.', warnings) + const certificateValue = parsed( + settled[3]!, CertificateSchema, 'TLS certificate inventory is unavailable.', warnings + ) + const migValues = environment.cells.map((cell, index) => parsed( + settled[index + 4]!, + MigSchema, + `${cell.hostname.toUpperCase()} MIG inventory is unavailable.`, + warnings + )) + const controlPlaneSleeping = + environment.id === 'staging' && sqlValue?.settings.activationPolicy === 'NEVER' + // Health probes would cold-start scale-to-zero Cloud Run services, so sleeping staging is inventory-only. + const [directorEndpoint, authEndpoint] = controlPlaneSleeping + ? [unavailableEndpoint(), unavailableEndpoint()] + : await Promise.all([ + probeEndpointHealth(environment.directorOrigin, fetchImpl), + probeEndpointHealth(environment.authOrigin, fetchImpl) + ]) + const cells = await Promise.all(environment.cells.map((cell, index) => + readCell(environment, cell, migValues[index] ?? null, token, fetchImpl) + )) + return { + director: directorValue ? parseService(directorValue) : null, + auth: authValue ? parseService(authValue) : null, + sql: sqlValue ? { + state: sqlValue.state, + activationPolicy: sqlValue.settings.activationPolicy, + tier: sqlValue.settings.tier, + availabilityType: sqlValue.settings.availabilityType ?? 'unknown', + databaseVersion: sqlValue.databaseVersion + } : null, + certificate: certificateValue ? { + state: certificateValue.managed.state, + domains: certificateValue.managed.domains, + expireTime: certificateValue.expireTime ?? null + } : null, + directorEndpoint, + authEndpoint, + cells, + warnings + } +} diff --git a/cloud/apps/relay-ops/src/staging-workflow.test.ts b/cloud/apps/relay-ops/src/staging-workflow.test.ts new file mode 100644 index 00000000000..881ffb2bf9a --- /dev/null +++ b/cloud/apps/relay-ops/src/staging-workflow.test.ts @@ -0,0 +1,19 @@ +import { describe, expect, it } from 'vitest' +import { parseStagingPowerRequest } from './staging-workflow.js' + +describe('parseStagingPowerRequest', () => { + it('accepts the exact reviewed confirmations', () => { + expect(parseStagingPowerRequest({ mode: 'status', confirmation: '' })).toEqual({ + mode: 'status', confirmation: '' + }) + expect(parseStagingPowerRequest({ mode: 'wake', confirmation: 'WAKE_STAGING' }).mode).toBe('wake') + expect(parseStagingPowerRequest({ mode: 'sleep', confirmation: 'SLEEP_STAGING' }).mode).toBe('sleep') + }) + + it('rejects missing, swapped, or additional fields', () => { + expect(() => parseStagingPowerRequest({ mode: 'wake', confirmation: '' })).toThrow() + expect(() => parseStagingPowerRequest({ mode: 'sleep', confirmation: 'WAKE_STAGING' })).toThrow() + expect(() => parseStagingPowerRequest({ mode: 'production', confirmation: '' })).toThrow() + expect(() => parseStagingPowerRequest({ mode: 'status', confirmation: '', project: 'other' })).toThrow() + }) +}) diff --git a/cloud/apps/relay-ops/src/staging-workflow.ts b/cloud/apps/relay-ops/src/staging-workflow.ts new file mode 100644 index 00000000000..749c6e865bb --- /dev/null +++ b/cloud/apps/relay-ops/src/staging-workflow.ts @@ -0,0 +1,36 @@ +import { execFile } from 'node:child_process' +import { promisify } from 'node:util' +import { z } from 'zod' +import { RELAY_GITHUB_REPOSITORY, relayWorkflowFile } from './relay-repository.js' + +const execFileAsync = promisify(execFile) +const DispatchSchema = z.discriminatedUnion('mode', [ + z.object({ mode: z.literal('status'), confirmation: z.literal('') }).strict(), + z.object({ mode: z.literal('wake'), confirmation: z.literal('WAKE_STAGING') }).strict(), + z.object({ mode: z.literal('sleep'), confirmation: z.literal('SLEEP_STAGING') }).strict() +]) + +export type StagingPowerRequest = z.infer + +export function parseStagingPowerRequest(value: unknown): StagingPowerRequest { + return DispatchSchema.parse(value) +} + +export async function dispatchStagingPowerWorkflow(request: StagingPowerRequest): Promise { + const args = [ + 'workflow', 'run', relayWorkflowFile('power-relay-staging.yml'), + '--repo', RELAY_GITHUB_REPOSITORY, + '-f', `mode=${request.mode}`, + '-f', 'wake-cells=configured' + ] + if (request.confirmation) args.push('-f', `confirmation=${request.confirmation}`) + try { + await execFileAsync('gh', args, { + encoding: 'utf8', + timeout: 30_000, + maxBuffer: 1024 * 1024 + }) + } catch { + throw new Error('Staging power workflow dispatch failed') + } +} diff --git a/cloud/apps/relay-ops/tsconfig.build.json b/cloud/apps/relay-ops/tsconfig.build.json new file mode 100644 index 00000000000..0a11719fef1 --- /dev/null +++ b/cloud/apps/relay-ops/tsconfig.build.json @@ -0,0 +1,4 @@ +{ + "extends": "./tsconfig.json", + "exclude": ["src/**/*.test.ts"] +} diff --git a/cloud/apps/relay-ops/tsconfig.json b/cloud/apps/relay-ops/tsconfig.json new file mode 100644 index 00000000000..d266baf8ad6 --- /dev/null +++ b/cloud/apps/relay-ops/tsconfig.json @@ -0,0 +1,11 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "src", + "outDir": "dist", + "types": ["node", "vitest"], + "exactOptionalPropertyTypes": true, + "verbatimModuleSyntax": true + }, + "include": ["src/**/*.ts"] +} diff --git a/cloud/apps/relay/Dockerfile b/cloud/apps/relay/Dockerfile new file mode 100644 index 00000000000..12516cbf749 --- /dev/null +++ b/cloud/apps/relay/Dockerfile @@ -0,0 +1,25 @@ +FROM node:24-alpine AS build +WORKDIR /app +RUN corepack enable +COPY package.json pnpm-lock.yaml pnpm-workspace.yaml tsconfig.base.json ./ +COPY packages/relay-contract/package.json packages/relay-contract/package.json +COPY apps/relay/package.json apps/relay/package.json +RUN pnpm install --frozen-lockfile +COPY packages/relay-contract packages/relay-contract +COPY apps/relay apps/relay +RUN pnpm --filter @orca-cloud/relay-contract build && pnpm --filter @orca-cloud/relay build + +FROM node:24-alpine AS runtime +ENV NODE_ENV=production +ENV PORT=8080 +WORKDIR /app +RUN corepack enable +COPY package.json pnpm-lock.yaml pnpm-workspace.yaml ./ +COPY packages/relay-contract/package.json packages/relay-contract/package.json +COPY apps/relay/package.json apps/relay/package.json +COPY --from=build /app/packages/relay-contract/dist packages/relay-contract/dist +COPY --from=build /app/apps/relay/dist apps/relay/dist +RUN pnpm install --prod --frozen-lockfile --filter @orca-cloud/relay... +USER node +EXPOSE 8080 +CMD ["node", "apps/relay/dist/index.js"] diff --git a/cloud/apps/relay/package.json b/cloud/apps/relay/package.json new file mode 100644 index 00000000000..4c2b2e4269c --- /dev/null +++ b/cloud/apps/relay/package.json @@ -0,0 +1,35 @@ +{ + "name": "@orca-cloud/relay", + "private": true, + "version": "0.0.0", + "type": "module", + "main": "dist/index.js", + "scripts": { + "build": "pnpm clean && tsc -p tsconfig.build.json", + "clean": "node -e \"require('fs').rmSync('dist', { recursive: true, force: true })\"", + "dev": "tsx watch src/index.ts", + "lint": "tsc -p tsconfig.json --noEmit", + "pretest": "pnpm --filter @orca-cloud/relay-contract build", + "start": "node dist/index.js", + "test": "vitest run", + "typecheck": "tsc -p tsconfig.json --noEmit" + }, + "dependencies": { + "@hono/node-server": "^1.19.14", + "@orca-cloud/relay-contract": "workspace:*", + "hono": "^4.12.27", + "jose": "^6.1.3", + "pg": "^8.22.0", + "tweetnacl": "^1.0.3", + "ws": "^8.18.3", + "zod": "^3.25.76" + }, + "devDependencies": { + "@types/node": "^24.10.0", + "@types/pg": "^8.20.0", + "@types/ws": "^8.18.1", + "tsx": "^4.21.0", + "typescript": "^5.9.3", + "vitest": "^4.0.8" + } +} diff --git a/cloud/apps/relay/src/admin-token-verifier.test.ts b/cloud/apps/relay/src/admin-token-verifier.test.ts new file mode 100644 index 00000000000..f0870dbb09c --- /dev/null +++ b/cloud/apps/relay/src/admin-token-verifier.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from 'vitest' +import { + RELAY_ASIA_PROOF_ADMIN_ROUTES, + RELAY_CAPACITY_ADMIN_ROUTES, + RELAY_FENCE_BROKER_ADMIN_ROUTES, + RELAY_FENCE_ADMIN_ROUTES, + RELAY_MONITOR_ADMIN_ROUTES, + relayAdminIdentityMayAccess +} from './admin-token-verifier.js' + +const mutationRoutes = [ + '/v1/admin/drain', + '/v1/admin/evacuate', + '/v1/admin/migration-complete', + '/v1/admin/migration-supersede-cell', + '/v1/admin/rebalance-dormant', + '/v1/admin/admission-selector/apply', + '/v1/admin/admission-selector/add-migration-cells', + '/v1/admin/cell-state', + '/v1/admin/cell-fence-adopt-legacy', + '/v1/admin/cell-fence-commit-legacy-adoption', + '/v1/admin/cell-fence-attest', + '/v1/admin/cell-fence-attempt-prepare', + '/v1/admin/cell-fence-attempt-start', + '/v1/admin/cell-fence-attempt-operation', + '/v1/admin/cell-fence-attempt-abort', + '/v1/admin/drain-attempt-prepare', + '/v1/admin/drain-attempt-send', + '/v1/admin/drain-attempt-receipt', + '/v1/admin/drain-attempt-recover-forward', + '/v1/admin/cell-config', + '/v1/admin/evacuate-cell', + '/v1/admin/cell-heartbeat', + '/v1/admin/regional-rehome-control', + '/v1/admin/regional-rehome-trust-probe' +] as const + +describe('Relay admin route authorization', () => { + it('allows the staging capacity identity only its transition routes', () => { + for (const route of RELAY_CAPACITY_ADMIN_ROUTES) { + expect(relayAdminIdentityMayAccess('capacity', route)).toBe(true) + } + for (const route of mutationRoutes) { + if ((RELAY_CAPACITY_ADMIN_ROUTES as readonly string[]).includes(route)) continue + expect(relayAdminIdentityMayAccess('capacity', route)).toBe(false) + } + expect(RELAY_CAPACITY_ADMIN_ROUTES).toContain('/v1/admin/cell-state') + expect(relayAdminIdentityMayAccess('capacity', '/v1/admin/evacuation-status')).toBe(false) + }) + + it('allows the Asia proof identity only its selector and status routes', () => { + for (const route of RELAY_ASIA_PROOF_ADMIN_ROUTES) { + expect(relayAdminIdentityMayAccess('asia-proof', route)).toBe(true) + } + expect(relayAdminIdentityMayAccess('asia-proof', '/v1/admin/drain')).toBe(false) + expect(relayAdminIdentityMayAccess('asia-proof', '/v1/admin/cell-state')).toBe(false) + expect(relayAdminIdentityMayAccess('asia-proof', '/v1/admin/admission-selector/apply')).toBe(false) + expect(relayAdminIdentityMayAccess('asia-proof', '/v1/admin/add-migration-cells')).toBe(false) + }) + + it('keeps the monitor identity on exact aggregate read routes', () => { + for (const route of RELAY_MONITOR_ADMIN_ROUTES) { + expect(relayAdminIdentityMayAccess('monitor', route)).toBe(true) + } + for (const route of [...mutationRoutes, ...RELAY_FENCE_ADMIN_ROUTES]) { + if ((RELAY_MONITOR_ADMIN_ROUTES as readonly string[]).includes(route)) continue + expect(relayAdminIdentityMayAccess('monitor', route)).toBe(false) + } + expect(RELAY_MONITOR_ADMIN_ROUTES).toContain('/v1/admin/evacuation-status') + }) + + it('allows only reviewed fence evidence mutations beyond aggregate reads', () => { + for (const route of RELAY_FENCE_ADMIN_ROUTES) { + expect(relayAdminIdentityMayAccess('fence', route)).toBe(true) + } + for (const route of mutationRoutes) { + if ((RELAY_FENCE_ADMIN_ROUTES as readonly string[]).includes(route)) continue + expect(relayAdminIdentityMayAccess('fence', route)).toBe(false) + } + }) + + it('allows the broker only exact fence inspection and mutation routes', () => { + for (const route of RELAY_FENCE_BROKER_ADMIN_ROUTES) { + expect(relayAdminIdentityMayAccess('fence-broker', route)).toBe(true) + } + for (const route of [...mutationRoutes, ...RELAY_FENCE_ADMIN_ROUTES]) { + if ((RELAY_FENCE_BROKER_ADMIN_ROUTES as readonly string[]).includes(route)) continue + expect(relayAdminIdentityMayAccess('fence-broker', route)).toBe(false) + } + }) + + it('rejects unknown routes for dedicated identities', () => { + for (const identity of ['capacity', 'asia-proof', 'monitor', 'fence', 'fence-broker'] as const) { + expect(relayAdminIdentityMayAccess(identity, '/v1/admin/future-mutation')).toBe(false) + expect(relayAdminIdentityMayAccess(identity, '/health')).toBe(false) + } + }) +}) diff --git a/cloud/apps/relay/src/admin-token-verifier.ts b/cloud/apps/relay/src/admin-token-verifier.ts new file mode 100644 index 00000000000..4b8ad26e695 --- /dev/null +++ b/cloud/apps/relay/src/admin-token-verifier.ts @@ -0,0 +1,202 @@ +import { createRemoteJWKSet, jwtVerify } from 'jose' +import type { RelayConfig } from './config.js' + +export const RELAY_MONITOR_ADMIN_ROUTES = [ + '/v1/admin/admission-selector/status', + '/v1/admin/cell-status', + '/v1/admin/evacuation-status', + '/v1/admin/regional-rehome-control', + '/v1/admin/runtime-status' +] as const + +export const RELAY_FENCE_ADMIN_ROUTES = [ + ...RELAY_MONITOR_ADMIN_ROUTES, + '/v1/admin/evacuation-capacity', + '/v1/admin/cell-fence-attempt-status' +] as const + +export const RELAY_CAPACITY_ADMIN_ROUTES = [ + '/v1/admin/admission-selector/status', + '/v1/admin/admission-selector/apply', + '/v1/admin/cell-state', + '/v1/admin/cell-status', + '/v1/admin/runtime-status', + '/v1/admin/drain' +] as const + +export const RELAY_ASIA_PROOF_ADMIN_ROUTES = [ + '/v1/admin/admission-selector/status', + '/v1/admin/admission-selector/apply-staging-asia-proof', + '/v1/admin/cell-status', + '/v1/admin/runtime-status' +] as const + +export const RELAY_FENCE_BROKER_ADMIN_ROUTES = [ + ...RELAY_FENCE_ADMIN_ROUTES, + '/v1/admin/cell-fence-adopt-legacy', + '/v1/admin/cell-fence-commit-legacy-adoption', + '/v1/admin/cell-fence-attest', + '/v1/admin/cell-fence-attempt-prepare', + '/v1/admin/cell-fence-attempt-start', + '/v1/admin/cell-fence-attempt-plan', + '/v1/admin/cell-fence-attempt-operation', + '/v1/admin/cell-fence-attempt-abort', + '/v1/admin/migration-supersede-cell' +] as const + +type RelayAdminIdentity = + | 'deploy' + | 'capacity' + | 'asia-proof' + | 'monitor' + | 'fence' + | 'fence-broker' + +function createGoogleServiceTokenVerifier(input: { + jwksUrl: string + audience: string + serviceAccount: string +}): (token: string) => Promise { + const jwks = createRemoteJWKSet(new URL(input.jwksUrl)) + return async (token) => { + try { + const { payload } = await jwtVerify(token, jwks, { + issuer: ['https://accounts.google.com', 'accounts.google.com'], + audience: input.audience, + algorithms: ['RS256'] + }) + return payload.email === input.serviceAccount && payload.email_verified === true + } catch { + return false + } + } +} + +function createGoogleServiceTokenIdentityVerifier(input: { + jwksUrl: string + audience: string + serviceAccounts: ReadonlyMap +}): (token: string) => Promise { + const jwks = createRemoteJWKSet(new URL(input.jwksUrl)) + return async (token) => { + try { + const { payload } = await jwtVerify(token, jwks, { + issuer: ['https://accounts.google.com', 'accounts.google.com'], + audience: input.audience, + algorithms: ['RS256'] + }) + if (payload.email_verified !== true || typeof payload.email !== 'string') return null + return input.serviceAccounts.get(payload.email) ?? null + } catch { + return null + } + } +} + +export function relayAdminIdentityMayAccess( + identity: RelayAdminIdentity, + route: string +): boolean { + if (identity === 'deploy') return route.startsWith('/v1/admin/') + if (identity === 'capacity') { + return (RELAY_CAPACITY_ADMIN_ROUTES as readonly string[]).includes(route) + } + if (identity === 'asia-proof') { + return (RELAY_ASIA_PROOF_ADMIN_ROUTES as readonly string[]).includes(route) + } + if (identity === 'monitor') { + return (RELAY_MONITOR_ADMIN_ROUTES as readonly string[]).includes(route) + } + if (identity === 'fence') { + return (RELAY_FENCE_ADMIN_ROUTES as readonly string[]).includes(route) + } + return (RELAY_FENCE_BROKER_ADMIN_ROUTES as readonly string[]).includes(route) +} + +export function createReadOnlyAdminTokenVerifier( + config: RelayConfig +): (token: string) => Promise { + const serviceAccounts = new Map() + if (config.monitorServiceAccount) serviceAccounts.set(config.monitorServiceAccount, 'monitor') + if (config.fenceServiceAccount) serviceAccounts.set(config.fenceServiceAccount, 'fence') + if (serviceAccounts.size === 0) return async () => false + const verifyIdentity = createGoogleServiceTokenIdentityVerifier({ + jwksUrl: config.adminJwksUrl, + audience: config.adminAudience, + serviceAccounts + }) + return async (token) => (await verifyIdentity(token)) !== null +} + +export function createAdminTokenVerifier( + config: RelayConfig +): (token: string, route?: string) => Promise { + const serviceAccounts = new Map([ + [config.deployServiceAccount, 'deploy'] + ]) + if (config.capacityServiceAccount) { + serviceAccounts.set(config.capacityServiceAccount, 'capacity') + } + if (config.asiaProofServiceAccount) { + serviceAccounts.set(config.asiaProofServiceAccount, 'asia-proof') + } + if (config.monitorServiceAccount) serviceAccounts.set(config.monitorServiceAccount, 'monitor') + if (config.fenceServiceAccount) serviceAccounts.set(config.fenceServiceAccount, 'fence') + if (config.fenceBrokerServiceAccount) { + serviceAccounts.set(config.fenceBrokerServiceAccount, 'fence-broker') + } + const verifyIdentity = createGoogleServiceTokenIdentityVerifier({ + jwksUrl: config.adminJwksUrl, + audience: config.adminAudience, + serviceAccounts + }) + return async (token, route) => { + const identity = await verifyIdentity(token) + return identity !== null && (!route || relayAdminIdentityMayAccess(identity, route)) + } +} + +export function createRegionalRehomeControlApplyTokenVerifier( + config: RelayConfig +): (token: string) => Promise { + return createGoogleServiceTokenVerifier({ + jwksUrl: config.adminJwksUrl, + audience: config.adminAudience, + serviceAccount: config.deployServiceAccount + }) +} + +export function createRuntimeTokenVerifier( + config: RelayConfig +): (token: string) => Promise { + if (!config.heartbeatAudience) return async () => false + return createGoogleServiceTokenVerifier({ + jwksUrl: config.adminJwksUrl, + audience: config.heartbeatAudience, + serviceAccount: config.runtimeServiceAccount + }) +} + +export function createRegionalRehomeTokenVerifier( + config: RelayConfig +): (token: string) => Promise { + if (!config.rehomeAudience || !config.rehomeDirectorServiceAccount) { + return async () => false + } + return createGoogleServiceTokenVerifier({ + jwksUrl: config.adminJwksUrl, + audience: config.rehomeAudience, + serviceAccount: config.rehomeDirectorServiceAccount + }) +} + +export function createRegionalRehomeRuntimeTokenVerifier( + config: RelayConfig +): (token: string) => Promise { + if (!config.rehomeAudience) return async () => false + return createGoogleServiceTokenVerifier({ + jwksUrl: config.adminJwksUrl, + audience: config.rehomeAudience, + serviceAccount: config.runtimeServiceAccount + }) +} diff --git a/cloud/apps/relay/src/app.ts b/cloud/apps/relay/src/app.ts new file mode 100644 index 00000000000..3df01d9e9ce --- /dev/null +++ b/cloud/apps/relay/src/app.ts @@ -0,0 +1,1850 @@ +import { + AssignmentRequestSchema, + isRelayCellConnectionHardCap, + RELAY_ADMISSION_BUDGETS, + RELAY_DEFAULT_REGION, + RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND, + RELAY_PROTOCOL_LIMITS, + RelayRegionSchema, + ResolveRequestSchema, + cellPlacementCeiling, + relayCellAdmissionBounds, + type RelayCellConnectionHardCap, + type RelayRegion +} from '@orca-cloud/relay-contract' +import { Hono, type Context } from 'hono' +import { SignJWT } from 'jose' +import { z } from 'zod' +import { + createAdminTokenVerifier, + createReadOnlyAdminTokenVerifier, + createRegionalRehomeControlApplyTokenVerifier, + createRegionalRehomeRuntimeTokenVerifier, + createRegionalRehomeTokenVerifier, + createRuntimeTokenVerifier +} from './admin-token-verifier.js' +import type { + CellFenceAttemptEvidence, + RelayAssignment, + RelayAssignmentStore +} from './assignment-store.js' +import { AssignmentRejectionLogWindow } from './assignment-rejection-log-window.js' +import { CELL_ADMISSION_STATES } from './cell-admission-selector.js' +import { RELAY_MAX_CELL_CAPACITY_REQUESTS, type RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { isRelayDatabaseTransientError } from './database.js' +import { googleMetadataIdentityToken } from './google-metadata-identity-token.js' +import { + RelayPublicAssignmentAdmission, + type AssignmentAdmissionRejection +} from './public-assignment-admission.js' +import { relayHostLogDigest } from './relay-host-log-digest.js' +import type { RelayRuntimeCounts } from './relay-observability.js' +import { + isRegionalRehomeTrustProbe, + probeRegionalRehomeTrust +} from './regional-rehome-trust-probe.js' +import { createRelayTokenVerifier, readBearer } from './relay-token-verifier.js' +import { stagingAsiaProofMembership } from './staging-asia-proof-admission.js' + +const RelayCellConnectionHardCapSchema = z.custom( + isRelayCellConnectionHardCap +) + +const ASSIGNMENT_REJECTION_LOG_WINDOW_MS = 10_000 +const REGION_CATALOG_CACHE_MS = 30_000 + +type AdmissionRejectionLogEntry = { + route: 'assign' | 'resolve' + lane: 'sticky' | 'placement' + hinted: boolean + relayHostId: string + reason: AssignmentAdmissionRejection +} + +export function createRelayApp( + config: RelayConfig, + operations: { + store: RelayCredentialStore + assignments: RelayAssignmentStore + drain: (graceMs: number) => void + drainHost?: (input: { + attemptId: string + userId: string + relayHostId: string + sourceAssignmentEpoch: number + graceMs: number + }) => 'accepted' | 'already-accepted' | 'host-not-connected' + regionalRehomeIdentityToken?: (audience: string) => Promise + regionalRehomeFetch?: typeof fetch + regionalRehomeTrustProbeHostExists?: (input: { + userId: string + relayHostId: string + }) => boolean + cellIncarnation?: string + isDraining?: () => boolean + runtimeCounts?: () => RelayRuntimeCounts + ready: () => Promise + recordAssignmentAdmission?: ( + outcome: 'sticky' | 'sticky-rejected' | 'placement' | 'placement-rejected' + ) => void + recordAssignmentRejectionReason?: ( + lane: 'sticky' | 'placement', + reason: AssignmentAdmissionRejection + ) => void + recordRegionRequest?: (region: RelayRegion | undefined) => void + recordRegionSelection?: (input: { + targetRegion: RelayRegion + selectedRegion?: RelayRegion + fallback: boolean + }) => void + } +): Hono { + const app = new Hono() + let regionCatalogCache: + | { expiresAt: number; value: Awaited> } + | undefined + let regionCatalogRefresh: + | Promise>> + | undefined + const regionCatalog = async () => { + const now = Date.now() + if (regionCatalogCache && regionCatalogCache.expiresAt > now) { + return regionCatalogCache.value + } + regionCatalogRefresh ??= operations.assignments.regionCatalog().then((value) => { + regionCatalogCache = { expiresAt: Date.now() + REGION_CATALOG_CACHE_MS, value } + return value + }) + try { + return await regionCatalogRefresh + } finally { + regionCatalogRefresh = undefined + } + } + const verifyRelayToken = createRelayTokenVerifier(config) + const verifyAdminToken = createAdminTokenVerifier(config) + const verifyReadOnlyAdminToken = createReadOnlyAdminTokenVerifier(config) + const verifyRegionalRehomeControlApplyToken = + createRegionalRehomeControlApplyTokenVerifier(config) + const verifyRuntimeToken = createRuntimeTokenVerifier(config) + const verifyRegionalRehomeToken = createRegionalRehomeTokenVerifier(config) + const verifyRegionalRehomeRuntimeToken = + createRegionalRehomeRuntimeTokenVerifier(config) + const regionalRehomeFetch = operations.regionalRehomeFetch ?? fetch + const regionalRehomeIdentityToken = + operations.regionalRehomeIdentityToken ?? + ((audience: string) => googleMetadataIdentityToken(audience, regionalRehomeFetch)) + const publicAssignmentAdmission = new RelayPublicAssignmentAdmission({ + maxConcurrent: config.publicAssignmentConcurrency, + maxQueued: config.publicAssignmentQueueMax, + waitMs: config.publicAssignmentWaitMs, + maxReservedConcurrent: config.publicResolveConcurrency, + reservedWaitMs: config.publicResolveWaitMs, + minIntervalMs: config.publicAssignmentRetryAfterSeconds * 1_000, + onRejected: (reason) => operations.recordAssignmentRejectionReason?.('placement', reason) + }) + const rejectPublicAssignment = (context: Context): Response => { + context.header('Retry-After', String(config.publicAssignmentRetryAfterSeconds)) + return context.json({ error: 'assignments_temporarily_unavailable' }, 503) + } + // Reconnecting hosts declare themselves and are verified against the durable + // assignment inside this bounded lane, so a placement backlog can never + // starve session recovery. Unhinted traffic never touches this lane. + const stickyRetryAfterSeconds = config.publicStickyRetryAfterSeconds ?? 2 + const stickyAssignmentAdmission = new RelayPublicAssignmentAdmission({ + maxConcurrent: config.publicStickyConcurrency ?? 1, + maxQueued: config.publicStickyQueueMax ?? 64, + waitMs: config.publicStickyWaitMs ?? 2_000, + minIntervalMs: stickyRetryAfterSeconds * 1_000, + onRejected: (reason) => operations.recordAssignmentRejectionReason?.('sticky', reason) + }) + const rejectStickyAssignment = (context: Context): Response => { + context.header('Retry-After', String(stickyRetryAfterSeconds)) + return context.json({ error: 'assignments_temporarily_unavailable' }, 503) + } + // Aggregate counters cannot separate a handful of pathological hosts from a broad + // population, so every admission rejection names its host and reason. Keyed on + // route:lane:reason rather than host, the log stays bounded under load. + const rejectionLogWindow = new AssignmentRejectionLogWindow({ + windowMs: ASSIGNMENT_REJECTION_LOG_WINDOW_MS, + onWindowClosed: ({ suppressed, sample }) => logAssignmentRejection({ ...sample, suppressed }) + }) + const logAdmissionRejection = (input: { + route: 'assign' | 'resolve' + lane: 'sticky' | 'placement' + hinted: boolean + relayHostId: string + reason: AssignmentAdmissionRejection | undefined + }): void => { + if (!input.reason) return + const entry: AdmissionRejectionLogEntry = { ...input, reason: input.reason } + if (!rejectionLogWindow.admit(`${entry.route}:${entry.lane}:${entry.reason}`, entry)) return + logAssignmentRejection(entry) + } + app.use('/v1/admin/*', async (context, next) => { + if ( + context.req.path === '/v1/admin/cell-heartbeat' || + context.req.path === '/v1/admin/cell-rehome-status' + ) { + return await next() + } + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyAdminToken(bearer))) return await next() + if (!(await verifyAdminToken(bearer, context.req.path))) { + return context.json({ error: 'invalid_token' }, 401) + } + return await next() + }) + + // Not /healthz: Google Front End reserves that path before the container. + app.get('/health', (context) => + context.json({ ok: true, connectionCapacityProtocol: 2 }) + ) + app.get('/ready', async (context) => + (await operations.ready()) + ? context.json({ ok: true }) + : context.json({ error: 'dependency_unavailable' }, 503) + ) + app.get('/v1/regions', async (context) => { + if (config.role === 'cell') return context.json({ error: 'director_only' }, 404) + return context.json({ v: 1, regions: await regionCatalog() }) + }) + app.post('/v1/assign', async (context) => { + if (config.role === 'cell') return context.json({ error: 'director_only' }, 404) + if (!config.publicAssignmentsEnabled) return rejectPublicAssignment(context) + const bearer = readBearer(context.req.header('authorization')) + if (!bearer) return context.json({ error: 'invalid_token' }, 401) + const claims = await verifyRelayToken(bearer) + if (!claims) return context.json({ error: 'invalid_token' }, 401) + if (Number(context.req.header('content-length') ?? 0) > 4 * 1024) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AssignmentRequestSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + if (body.data.relayHostId !== claims.relayHostId) { + return context.json({ error: 'host_identity_mismatch' }, 403) + } + const identity = { userId: claims.sub, relayHostId: claims.relayHostId } + const requestedRegion = body.data.preferredRegion + const targetRegion = + config.regionalPlacementEnabled !== false && requestedRegion + ? requestedRegion + : RELAY_DEFAULT_REGION + operations.recordRegionRequest?.(requestedRegion) + let admission: { release(): void } | null = null + let lane: 'sticky' | 'placement' = 'placement' + if (body.data.reconnect) { + let rejection: AssignmentAdmissionRejection | undefined + const fastLane = await stickyAssignmentAdmission.acquire(claims.relayHostId, (reason) => { + rejection = reason + }) + if (!fastLane) { + operations.recordAssignmentAdmission?.('sticky-rejected') + logAdmissionRejection({ + route: 'assign', + lane: 'sticky', + hinted: true, + relayHostId: claims.relayHostId, + reason: rejection + }) + return rejectStickyAssignment(context) + } + let verified = false + try { + verified = (await operations.assignments.resolve(identity)) !== null + } catch (error) { + fastLane.release() + operations.recordAssignmentAdmission?.('sticky-rejected') + if (isRelayDatabaseTransientError(error)) { + logAssignmentRejection({ + route: 'assign-verify', + lane: 'none', + hinted: true, + relayHostId: claims.relayHostId, + reason: operationError(error) + }) + return rejectStickyAssignment(context) + } + throw error + } + if (verified) { + lane = 'sticky' + admission = fastLane + operations.recordAssignmentAdmission?.('sticky') + } else { + // An unverified hint joins the placement lane with today's semantics. + fastLane.release() + } + } + if (!admission) { + let rejection: AssignmentAdmissionRejection | undefined + admission = await publicAssignmentAdmission.acquire(claims.relayHostId, (reason) => { + rejection = reason + }) + operations.recordAssignmentAdmission?.(admission ? 'placement' : 'placement-rejected') + if (!admission) { + logAdmissionRejection({ + route: 'assign', + lane: 'placement', + hinted: Boolean(body.data.reconnect), + relayHostId: claims.relayHostId, + reason: rejection + }) + return rejectPublicAssignment(context) + } + } + let assignment: RelayAssignment + try { + assignment = requestedRegion + ? await operations.assignments.assign(identity, requestedRegion, targetRegion) + : await operations.assignments.assign(identity) + } catch (error) { + if (isRelayAssignmentCapacityError(error) || isRelayDatabaseTransientError(error)) { + logAssignmentRejection({ + route: 'assign', + lane, + hinted: Boolean(body.data.reconnect), + relayHostId: claims.relayHostId, + reason: operationError(error) + }) + } + if (isRelayAssignmentCapacityError(error)) { + if (lane === 'placement') { + operations.recordRegionSelection?.({ targetRegion, fallback: false }) + } + return context.json({ error: operationError(error) }, 503) + } + if (isRelayDatabaseTransientError(error)) { + return lane === 'sticky' ? rejectStickyAssignment(context) : rejectPublicAssignment(context) + } + throw error + } finally { + admission.release() + } + operations.recordRegionSelection?.({ + targetRegion, + selectedRegion: assignment.region, + fallback: lane === 'placement' && assignment.region !== targetRegion + }) + // Grant-side counterpart of the rejection log: reconnect grants are rare + // enough to log and make "which cell is this host on" answerable. + if (lane === 'sticky') { + console.warn( + `[orca-relay] assignment granted lane=sticky host=${relayHostLogDigest(claims.relayHostId)}` + + ` cell=${assignment.cellId}` + ) + } + const lease = await new SignJWT({ + purpose: 'cell-assignment', + cellId: assignment.cellId, + cellUrl: assignment.cellUrl, + assignmentEpoch: assignment.assignmentEpoch, + relayHostId: claims.relayHostId + }) + .setProtectedHeader({ alg: 'HS256' }) + .setIssuer(config.publicUrl) + .setAudience('orca-relay-cell') + .setSubject(claims.sub) + .setIssuedAt() + .setExpirationTime('5m') + .sign(config.assignmentSigningKey) + return context.json({ + v: 1, + cellUrl: assignment.cellUrl, + assignmentEpoch: assignment.assignmentEpoch, + lease + }) + }) + app.post('/v1/resolve', async (context) => { + if (config.role === 'cell') return context.json({ error: 'director_only' }, 404) + if (!config.publicAssignmentsEnabled) return rejectPublicAssignment(context) + if (Number(context.req.header('content-length') ?? 0) > RELAY_PROTOCOL_LIMITS.maxHttpBodyBytes) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = ResolveRequestSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + let rejection: AssignmentAdmissionRejection | undefined + const admission = await publicAssignmentAdmission.acquireReserved( + body.data.relayHostId, + (reason) => { + rejection = reason + } + ) + if (!admission) { + logAdmissionRejection({ + route: 'resolve', + lane: 'placement', + hinted: false, + relayHostId: body.data.relayHostId, + reason: rejection + }) + return rejectPublicAssignment(context) + } + try { + const resolved = await operations.store.resolveResume( + body.data.relayHostId, + body.data.resumeToken + ) + if (!resolved) return context.json({ error: 'invalid_credential' }, 401) + const identity = { + userId: resolved.userId, + relayHostId: body.data.relayHostId + } + // This also migrates credentials created by the staging-only combined service. + const assignment = + (await operations.assignments.resolve(identity)) ?? + (await operations.assignments.assign(identity)) + return context.json({ + v: 1, + cellUrl: assignment.cellUrl, + assignmentEpoch: assignment.assignmentEpoch, + leaseExpiresAt: assignment.leaseExpiresAt + }) + } catch (error) { + if (isRelayAssignmentCapacityError(error) || isRelayDatabaseTransientError(error)) { + logAssignmentRejection({ + route: 'resolve', + lane: 'none', + hinted: false, + relayHostId: body.data.relayHostId, + reason: operationError(error) + }) + } + if (isRelayAssignmentCapacityError(error)) { + return context.json({ error: operationError(error) }, 503) + } + if (isRelayDatabaseTransientError(error)) return rejectPublicAssignment(context) + throw error + } finally { + admission.release() + } + }) + app.post('/v1/admin/drain', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + const body = z + .object({ v: z.literal(1), graceMs: z.number().int().nonnegative().max(60 * 60 * 1000) }) + .strict() + .safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + operations.drain(body.data.graceMs) + return context.json({ ok: true }) + }) + app.post('/v1/admin/host-drain', async (context) => { + if (config.role !== 'cell' || !operations.drainHost) { + return context.json({ error: 'cell_only' }, 404) + } + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyRegionalRehomeToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = RegionalHostDrainSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + if ( + body.data.sourceCellId !== config.cellId || + !operations.cellIncarnation || + body.data.sourceCellIncarnation !== operations.cellIncarnation + ) { + return context.json({ error: 'regional_rehome_source_generation_mismatch' }, 409) + } + try { + const trustProbe = isRegionalRehomeTrustProbe(body.data) + let sharedRuntimeIdentityRejected: true | undefined + if (trustProbe) { + const runtimeToken = await regionalRehomeIdentityToken(config.rehomeAudience!) + if ( + !(await verifyRegionalRehomeRuntimeToken(runtimeToken)) || + (await verifyRegionalRehomeToken(runtimeToken)) + ) { + throw new Error('regional_rehome_shared_runtime_rejection_not_proven') + } + if ( + !operations.regionalRehomeTrustProbeHostExists || + operations.regionalRehomeTrustProbeHostExists(body.data) + ) { + throw new Error('regional_rehome_trust_probe_host_not_absent') + } + sharedRuntimeIdentityRejected = true + } + const outcome = operations.drainHost(body.data) + return context.json({ + v: 1, + outcome, + ...(sharedRuntimeIdentityRejected ? { sharedRuntimeIdentityRejected } : {}) + }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/runtime-status', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = RuntimeStatusSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + return context.json({ + v: 1, + role: config.role, + cellId: config.cellId, + cellUrl: config.cellUrl, + region: config.region ?? RELAY_DEFAULT_REGION, + imageDigest: config.imageDigest ?? null, + draining: operations.isDraining?.() ?? false, + regionalRehomeProtocol: + config.rehomeAudience && config.rehomeDirectorServiceAccount ? 1 : 0, + connectionCapacity: + config.connectionHardCap === undefined + ? null + : { + hardCap: config.connectionHardCap, + controlRebindReserve: RELAY_ADMISSION_BUDGETS.reservedHostControls, + ordinaryConnectionLimit: relayCellAdmissionBounds(config.connectionHardCap) + .socketAdmissionCeiling, + unobservedBound: config.connectionUnobservedBound!, + normalAdmissionPause: cellPlacementCeiling( + config.connectionHardCap, + config.connectionUnobservedBound! + ) + }, + runtime: operations.runtimeCounts?.() ?? null + }) + }) + app.post('/v1/admin/cell-heartbeat', async (context) => { + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyRuntimeToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = CellHeartbeatSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + await operations.assignments.recordCellHeartbeat(body.data) + return context.json({ ok: true }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-rehome-status', async (context) => { + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyRuntimeToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = CellRegionalRehomeStatusSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + await operations.assignments.recordCellRegionalRehomeStatus(body.data) + return context.json({ ok: true }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/regional-rehome-control', async (context) => { + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyAdminToken(bearer, context.req.path))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = RegionalRehomeControlSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + if (body.data.action === 'inspect') { + const control = await operations.assignments.inspectRegionalRehomeControl() + return context.json({ v: 1, control }) + } + if (!(await verifyRegionalRehomeControlApplyToken(bearer))) { + return context.json({ error: 'insufficient_permission' }, 403) + } + try { + const control = await operations.assignments.applyRegionalRehomeControl(body.data) + return context.json({ v: 1, control }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/regional-rehome-trust-probe', async (context) => { + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + const bearer = readBearer(context.req.header('authorization')) + if (!bearer || !(await verifyAdminToken(bearer, context.req.path))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = RegionalRehomeTrustProbeSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + if (!config.rehomeAudience || !config.rehomeDirectorServiceAccount) { + return context.json({ error: 'regional_rehome_trust_not_configured' }, 409) + } + try { + const source = await operations.assignments.cellDeploymentStatus( + body.data.sourceCellId + ) + if ( + source.region !== RELAY_DEFAULT_REGION || + !source.runtime || + source.runtime.cellIncarnation !== body.data.sourceCellIncarnation || + !source.runtime.ready || + !source.runtime.heartbeatFresh || + source.runtime.regionalRehomeProtocol < 1 + ) { + throw new Error('regional_rehome_trust_probe_source_unavailable') + } + const result = await probeRegionalRehomeTrust({ + sourceCellUrl: source.cellUrl, + sourceCellId: body.data.sourceCellId, + sourceCellIncarnation: body.data.sourceCellIncarnation, + audience: config.rehomeAudience, + identityToken: regionalRehomeIdentityToken, + fetch: regionalRehomeFetch + }) + return context.json(result) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/evacuate', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminAssignmentMoveSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const migration = await operations.assignments.startEvacuation( + { userId: body.data.userId, relayHostId: body.data.relayHostId }, + body.data.targetCellId + ) + return context.json({ v: 1, migration }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/migration-complete', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminMigrationCompleteSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + await operations.assignments.completeEvacuation( + { userId: body.data.userId, relayHostId: body.data.relayHostId }, + body.data.assignmentEpoch + ) + return context.json({ ok: true }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/migration-supersede-cell', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminRegisteredCellMigrationSupersedeSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const superseded = await operations.assignments.supersedeRegisteredCellEvacuations( + body.data.sourceCellId, + body.data.currentTargetCellId, + body.data.replacementTargetCellId, + body.data.limit + ) + return context.json({ v: 1, superseded }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/rebalance-dormant', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminAssignmentMoveSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const assignment = await operations.assignments.rebalanceDormant( + { userId: body.data.userId, relayHostId: body.data.relayHostId }, + body.data.targetCellId + ) + return context.json({ v: 1, assignment }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/admission-selector/apply', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminAdmissionSelectorApplySchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const result = await operations.assignments.applyCellAdmissionSelector(body.data) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/admission-selector/apply-staging-asia-proof', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminStagingAsiaProofAdmissionApplySchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const current = await operations.assignments.inspectCellAdmissionSelector() + const result = await operations.assignments.applyCellAdmissionSelector({ + attemptId: body.data.attemptId, + expectedGeneration: body.data.expectedGeneration, + membership: stagingAsiaProofMembership(current.selector.membership, body.data.state) + }) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/admission-selector/status', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminAdmissionSelectorStatusSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const result = await operations.assignments.inspectCellAdmissionSelector( + body.data.attemptId + ) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/admission-selector/add-migration-cells', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminAdmissionSelectorAddMigrationCellsSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const result = await operations.assignments.addMigrationCells({ + attemptId: body.data.attemptId, + expectedGeneration: body.data.expectedGeneration, + cells: body.data.cells.map((cell) => ({ + id: cell.cellId, + url: cell.cellUrl, + capacityRequests: cell.capacityRequests, + region: cell.region, + connectionHardCap: cell.connectionHardCap, + connectionUnobservedBound: cell.connectionUnobservedBound + })) + }) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-state', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellStateSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + await operations.assignments.setCellAdmissionState( + body.data.cellId, + 'state' in body.data + ? body.data.state + : body.data.enabled + ? 'general' + : 'existing-only' + ) + return context.json({ ok: true }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-adopt-legacy', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceLegacyAdoptionSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const expiresAt = await operations.assignments.adoptLegacyCellFence( + body.data.cellId, + body.data.cellIncarnation + ) + return context.json({ v: 1, cellId: body.data.cellId, expiresAt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-commit-legacy-adoption', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceLegacyAdoptionCommitSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + await operations.assignments.commitLegacyCellFenceAdoption( + body.data.cellId, + body.data.cellIncarnation + ) + return context.json({ v: 1, cellId: body.data.cellId, committed: true }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attest', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceAttestSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const evidence = cellFenceAttemptEvidence(body.data) + const result = await operations.assignments.attestCellFenceAttempt( + evidence, + body.data.gceOperation + ) + return context.json({ + v: 1, + cellId: body.data.cellId, + expiresAt: result.expiresAt, + attempt: result.attempt + }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attempt-prepare', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceAttemptPrepareSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const evidence = cellFenceAttemptEvidence(body.data) + const attempt = await operations.assignments.prepareCellFenceAttempt(evidence) + return context.json({ v: 1, attempt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attempt-start', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceAttemptUpdateSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const evidence = cellFenceAttemptEvidence(body.data) + const result = await operations.assignments.startCellFenceApply( + evidence, + body.data.invocationId, + body.data.invocationRequestReason + ) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attempt-plan', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFencePlanSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const evidence = cellFenceAttemptEvidence(body.data) + const attempt = await operations.assignments.bindCellFencePlanGeneration( + evidence, + body.data.planObjectGeneration + ) + return context.json({ v: 1, attempt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attempt-operation', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceOperationSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const evidence = cellFenceAttemptEvidence(body.data) + const result = await operations.assignments.recordCellFenceOperation( + evidence, + body.data.invocationId, + body.data.invocationRequestReason, + body.data.gceOperation + ) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attempt-status', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceAttemptStatusSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const attempt = await operations.assignments.cellFenceAttempt(body.data.cellId) + return context.json({ v: 1, attempt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-fence-attempt-abort', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellFenceAttemptAbortSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const evidence = cellFenceAttemptEvidence(body.data) + const attempt = await operations.assignments.abortCellFenceAttempt(evidence) + return context.json({ v: 1, attempt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/drain-attempt-prepare', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminDrainAttemptPrepareSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const result = await operations.assignments.prepareCellDrainAttempt({ + attemptId: body.data.attemptId, + cellId: body.data.cellId, + cellIncarnation: body.data.cellIncarnation, + traceValue: body.data.traceValue, + plannedGraceMs: body.data.graceMs + }) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/drain-attempt-send', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminDrainAttemptMutationSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const attempt = await operations.assignments.beginCellDrainSend(body.data) + return context.json({ v: 1, attempt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/drain-attempt-receipt', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminDrainAttemptReceiptSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const attempt = await operations.assignments.recordCellDrainApplicationReceipt( + body.data + ) + return context.json({ v: 1, attempt }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/drain-attempt-recover-forward', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminDrainAttemptRecoverSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const result = await operations.assignments.prepareCellDrainRecovery(body.data) + return context.json({ v: 1, ...result }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/cell-config', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellConfigSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + await operations.assignments.configureCell( + { + id: body.data.cellId, + url: body.data.cellUrl, + capacityRequests: body.data.capacityRequests, + connectionHardCap: body.data.connectionHardCap, + connectionUnobservedBound: body.data.connectionUnobservedBound + }, + 'state' in body.data ? body.data.state : body.data.enabled + ) + return context.json({ ok: true }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/evacuate-cell', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellEvacuateSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const started = await operations.assignments.startActiveCellEvacuations( + body.data.sourceCellId, + body.data.targetCellId, + body.data.limit + ) + return context.json({ v: 1, started }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/evacuation-capacity', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellEvacuationCapacitySchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const capacity = await operations.assignments.cellEvacuationCapacity( + body.data.sourceCellId, + body.data.targetCellId + ) + return context.json({ v: 1, ...capacity }) + } catch (error) { + return context.json({ error: operationError(error) }, 409) + } + }) + app.post('/v1/admin/evacuation-status', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellEvacuationStatusSchema.safeParse( + await context.req.json().catch(() => null) + ) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + if (body.data.completeReady && (await verifyReadOnlyAdminToken(bearer))) { + return context.json({ error: 'insufficient_permission' }, 403) + } + const status = await operations.assignments.cellEvacuationStatus( + body.data.sourceCellId, + body.data.targetCellId, + body.data.completeReady + ) + return context.json({ v: 1, ...status }) + }) + app.post('/v1/admin/cell-status', async (context) => { + const bearer = readBearer(context.req.header('authorization')) + if (config.role !== 'director') return context.json({ error: 'director_only' }, 404) + if (!bearer || !(await verifyAdminToken(bearer))) { + return context.json({ error: 'invalid_token' }, 401) + } + if (requestTooLarge(context.req.header('content-length'))) { + return context.json({ error: 'request_too_large' }, 413) + } + const body = AdminCellStatusSchema.safeParse(await context.req.json().catch(() => null)) + if (!body.success) return context.json({ error: 'invalid_request' }, 400) + try { + const status = await operations.assignments.cellDeploymentStatus(body.data.cellId) + return context.json({ v: 1, status }) + } catch (error) { + return context.json({ error: operationError(error) }, 404) + } + }) + return app +} + +const AdminAssignmentMoveSchema = z + .object({ + v: z.literal(1), + userId: z.string().min(1).max(256), + relayHostId: z.string().regex(/^[A-Za-z0-9_-]{16}$/), + targetCellId: z.string().min(1).max(128) + }) + .strict() + +const RuntimeStatusSchema = z.object({ v: z.literal(1) }).strict() + +const RegionalRehomeSafetySchema = z + .object({ + observedAt: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + sqlFailures: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + reconnects: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + controlActivityRecoveryFailures: z + .number() + .int() + .nonnegative() + .max(Number.MAX_SAFE_INTEGER), + databasePoolWaiting: z.number().int().nonnegative().max(100), + databasePoolWaitersMax: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + databasePoolWaitMsMax: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + // Cells spread the full pool-pressure counts into the payload; rejecting + // the extra gauges 400'd every rehome-status heartbeat since 2026-08-15, + // so no cell ever recorded rehome protocol 1. + databasePoolTotal: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER).optional(), + databasePoolIdle: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER).optional(), + databasePoolOldestWaitMs: z + .number() + .int() + .nonnegative() + .max(Number.MAX_SAFE_INTEGER) + .optional() + }) + .strict() + +const CellHeartbeatSchema = z + .object({ + v: z.literal(1), + cellId: z.string().min(1).max(128), + cellUrl: z.string().url().max(2_048).refine(isCanonicalRelayOrigin), + region: RelayRegionSchema.default(RELAY_DEFAULT_REGION), + cellIncarnation: z.string().uuid(), + startedAt: z.number().int().positive().max(Number.MAX_SAFE_INTEGER), + ready: z.boolean(), + observedRequests: z.number().int().nonnegative().max(RELAY_MAX_CELL_CAPACITY_REQUESTS), + totalConnections: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CAPACITY_REQUESTS) + .optional(), + inFlightConnections: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CAPACITY_REQUESTS) + .optional(), + reservedConnectionUnits: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CAPACITY_REQUESTS) + .optional(), + enforcedConnectionUnits: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CAPACITY_REQUESTS) + .optional(), + connectionInclusionWatermark: z + .number() + .int() + .nonnegative() + .max(Number.MAX_SAFE_INTEGER) + .optional(), + connectionHardCap: RelayCellConnectionHardCapSchema.optional(), + connectionUnobservedBound: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND) + .optional() + }) + .strict() + .superRefine((value, context) => { + const values = [ + value.totalConnections, + value.inFlightConnections, + value.reservedConnectionUnits, + value.enforcedConnectionUnits, + value.connectionHardCap, + value.connectionUnobservedBound + ] + if ( + values.some((candidate) => candidate !== undefined) && + values.some((candidate) => candidate === undefined) + ) { + context.addIssue({ + code: 'custom', + message: 'connection telemetry must be complete' + }) + } + if ( + value.enforcedConnectionUnits !== undefined && + value.totalConnections !== undefined && + value.inFlightConnections !== undefined && + value.reservedConnectionUnits !== undefined && + value.enforcedConnectionUnits !== + value.totalConnections + + value.inFlightConnections + + value.reservedConnectionUnits + ) { + context.addIssue({ + code: 'custom', + message: 'connection telemetry sum does not match' + }) + } + if ( + value.connectionHardCap !== undefined && + value.connectionUnobservedBound! > + relayCellAdmissionBounds(value.connectionHardCap).maxUnobservedBound + ) { + context.addIssue({ + code: 'custom', + message: 'connection unobserved bound must leave ordinary admission capacity' + }) + } + }) + +const CellRegionalRehomeStatusSchema = z + .object({ + v: z.literal(1), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid(), + regionalRehomeProtocol: z.number().int().min(0).max(1), + safety: RegionalRehomeSafetySchema + }) + .strict() + +const RegionalRehomeControlSchema = z.discriminatedUnion('action', [ + z.object({ v: z.literal(1), action: z.literal('inspect') }).strict(), + z.object({ + v: z.literal(1), + action: z.literal('apply'), + expectedGeneration: z.number().int().nonnegative(), + enabled: z.boolean(), + notBefore: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + ratePerMinute: z.number().int().min(1).max(120), + preferenceMaxAgeMs: z + .number() + .int() + .min(60_000) + .max(30 * 24 * 60 * 60_000), + drainGraceMs: z.number().int().min(60_000).max(60 * 60_000), + confirmation: z.enum([ + 'ENABLE_REGIONAL_REHOMING', + 'DISABLE_REGIONAL_REHOMING' + ]) + }).strict() +]).superRefine((value, context) => { + if (value.action !== 'apply') return + const expected = value.enabled + ? 'ENABLE_REGIONAL_REHOMING' + : 'DISABLE_REGIONAL_REHOMING' + if (value.confirmation !== expected) { + context.addIssue({ code: 'custom', message: 'confirmation does not match state' }) + } +}) + +const RegionalRehomeTrustProbeSchema = z + .object({ + v: z.literal(1), + sourceCellId: z.string().min(1).max(128), + sourceCellIncarnation: z.string().uuid() + }) + .strict() + +const AdminMigrationCompleteSchema = z + .object({ + v: z.literal(1), + userId: z.string().min(1).max(256), + relayHostId: z.string().regex(/^[A-Za-z0-9_-]{16}$/), + assignmentEpoch: z.number().int().positive().max(Number.MAX_SAFE_INTEGER) + }) + .strict() + +const AdminRegisteredCellMigrationSupersedeSchema = z + .object({ + v: z.literal(1), + sourceCellId: z.string().min(1).max(128), + currentTargetCellId: z.string().min(1).max(128), + replacementTargetCellId: z.string().min(1).max(128), + limit: z.number().int().min(1).max(100), + confirmation: z.literal('SUPERSEDE_REGISTERED_CELL_MIGRATIONS') + }) + .strict() + .refine( + (value) => + new Set([ + value.sourceCellId, + value.currentTargetCellId, + value.replacementTargetCellId + ]).size === 3 + ) + +const CellIdSchema = z.string().min(1).max(128) +const CellAdmissionStateSchema = z.enum(CELL_ADMISSION_STATES) +const AdmissionSelectorAttemptIdSchema = z.string().regex(/^[A-Za-z0-9_-]{8,128}$/) +const AdmissionSelectorMembershipSchema = z + .object({ + existingOnly: z.array(CellIdSchema).max(256), + migrationOnly: z.array(CellIdSchema).max(256), + general: z.array(CellIdSchema).max(256) + }) + .strict() + +const AdminAdmissionSelectorApplySchema = z + .object({ + v: z.literal(1), + attemptId: AdmissionSelectorAttemptIdSchema, + expectedGeneration: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + expectedMembershipSha256: z.string().regex(/^[a-f0-9]{64}$/).optional(), + membership: AdmissionSelectorMembershipSchema + }) + .strict() + .refine( + (value) => value.expectedGeneration > 0 || value.expectedMembershipSha256 !== undefined + ) + +const AdminStagingAsiaProofAdmissionApplySchema = z + .object({ + v: z.literal(1), + attemptId: AdmissionSelectorAttemptIdSchema, + expectedGeneration: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + state: z.enum(['general', 'migration-only']) + }) + .strict() + +const AdminAdmissionSelectorStatusSchema = z + .object({ v: z.literal(1), attemptId: AdmissionSelectorAttemptIdSchema.optional() }) + .strict() + +const AdminAdmissionSelectorAddMigrationCellsSchema = z + .object({ + v: z.literal(1), + attemptId: AdmissionSelectorAttemptIdSchema, + expectedGeneration: z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER), + cells: z + .array( + z + .object({ + cellId: CellIdSchema, + cellUrl: z.string().url().max(2_048).refine(isCanonicalRelayOrigin), + capacityRequests: z + .number() + .int() + .positive() + .max(RELAY_MAX_CELL_CAPACITY_REQUESTS), + region: RelayRegionSchema.default(RELAY_DEFAULT_REGION), + connectionHardCap: RelayCellConnectionHardCapSchema, + connectionUnobservedBound: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND) + }) + .strict() + .superRefine((value, context) => { + if ( + value.connectionUnobservedBound > + relayCellAdmissionBounds(value.connectionHardCap).maxUnobservedBound + ) { + context.addIssue({ + code: 'custom', + path: ['connectionUnobservedBound'], + message: 'connection unobserved bound must leave ordinary admission capacity' + }) + } + }) + ) + .min(1) + .max(128) + }) + .strict() + .refine( + ({ cells }) => + new Set(cells.map(({ cellId }) => cellId)).size === cells.length && + new Set(cells.map(({ cellUrl }) => cellUrl)).size === cells.length + ) + +const AdminCellStateSchema = z.union([ + z.object({ v: z.literal(1), cellId: CellIdSchema, enabled: z.boolean() }).strict(), + z.object({ v: z.literal(1), cellId: CellIdSchema, state: CellAdmissionStateSchema }).strict() +]) + +const CellConfigShape = { + v: z.literal(1), + cellId: CellIdSchema, + cellUrl: z.string().url().max(2_048).refine(isCanonicalRelayOrigin), + capacityRequests: z.number().int().positive().max(RELAY_MAX_CELL_CAPACITY_REQUESTS), + connectionHardCap: RelayCellConnectionHardCapSchema.optional(), + connectionUnobservedBound: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND) + .optional() +} as const + +const AdminCellConfigSchema = z + .union([ + z + .object({ + ...CellConfigShape, + enabled: z.boolean() + }) + .strict(), + z + .object({ + ...CellConfigShape, + state: CellAdmissionStateSchema + }) + .strict() + ]) + .refine( + (value) => + (value.connectionHardCap === undefined) === + (value.connectionUnobservedBound === undefined), + { message: 'connection hard cap and unobserved bound must be configured together' } + ) + .refine( + (value) => + value.connectionHardCap === undefined || + value.connectionUnobservedBound! <= + relayCellAdmissionBounds(value.connectionHardCap).maxUnobservedBound, + { message: 'connection unobserved bound must leave ordinary admission capacity' } + ) + +const CellPairShape = { + v: z.literal(1), + sourceCellId: z.string().min(1).max(128), + targetCellId: z.string().min(1).max(128) +} as const + +const AdminCellEvacuateSchema = z + .object({ ...CellPairShape, limit: z.number().int().min(1).max(100) }) + .strict() + .refine((value) => value.sourceCellId !== value.targetCellId) + +const AdminCellEvacuationCapacitySchema = z + .object(CellPairShape) + .strict() + .refine((value) => value.sourceCellId !== value.targetCellId) + +const AdminCellEvacuationStatusSchema = z + .object({ ...CellPairShape, completeReady: z.boolean().default(false) }) + .strict() + .refine((value) => value.sourceCellId !== value.targetCellId) + +const TerraformStateLineageSchema = z + .string() + .regex(/^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/i) + +const CellFenceAttemptBaseShape = { + attemptId: z.string().uuid(), + environment: z.enum(['staging', 'production']), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid(), + migName: z.string().min(1).max(128), + instanceGroup: z.string().url().max(2048), + generationIdentity: z.string().url().max(2048), + fenceCommit: z.string().regex(/^[a-f0-9]{40}$/), + planSha256: z.string().regex(/^[a-f0-9]{64}$/), + planObjectName: z + .string() + .regex(/^terraform\/state\/relay-fence-plans\/(?:staging|production)\/[0-9a-f-]{36}\.tfplan$/), + varFileSha256: z.string().regex(/^[a-f0-9]{64}$/), + terraformStateLineage: TerraformStateLineageSchema, + terraformStateSerial: z.number().int().nonnegative().safe(), + terraformStateObjectGeneration: z.string().regex(/^[1-9][0-9]{0,30}$/), + terraformStateObjectSha256: z.string().regex(/^[a-f0-9]{64}$/), + requestReason: z + .string() + .regex(/^orca-relay-fence\/[0-9a-f]{8}-[0-9a-f-]{27}$/) +} as const +const CellFenceAttemptEvidenceShape = { + ...CellFenceAttemptBaseShape, + planObjectGeneration: z.string().regex(/^[1-9][0-9]{0,30}$/) +} as const + +const AdminCellFenceAttemptPrepareSchema = z + .object({ + v: z.literal(1), + ...CellFenceAttemptBaseShape, + confirmation: z.literal('PREPARE_TERRAFORM_CELL_FENCE') + }) + .strict() + +const AdminCellFencePlanSchema = z + .object({ + v: z.literal(1), + ...CellFenceAttemptEvidenceShape, + confirmation: z.literal('BIND_TERRAFORM_CELL_FENCE_PLAN') + }) + .strict() + +const AdminCellFenceAttemptUpdateSchema = z + .object({ + v: z.literal(1), + ...CellFenceAttemptEvidenceShape, + invocationId: z.string().uuid(), + invocationRequestReason: z.string().min(1).max(256), + confirmation: z.literal('START_TERRAFORM_CELL_FENCE') + }) + .strict() + +const AdminCellFenceOperationSchema = z + .object({ + v: z.literal(1), + ...CellFenceAttemptEvidenceShape, + invocationId: z.string().uuid(), + invocationRequestReason: z.string().min(1).max(256), + gceOperation: z.string().min(1).max(256), + confirmation: z.literal('RECORD_TERRAFORM_CELL_FENCE_OPERATION') + }) + .strict() + +const AdminCellFenceAttemptStatusSchema = z + .object({ v: z.literal(1), cellId: z.string().min(1).max(128) }) + .strict() + +const AdminCellFenceLegacyAdoptionSchema = z + .object({ + v: z.literal(1), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid(), + confirmation: z.literal('ADOPT_LEGACY_TERRAFORM_CELL_FENCE') + }) + .strict() + +const AdminCellFenceLegacyAdoptionCommitSchema = z + .object({ + v: z.literal(1), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid(), + confirmation: z.literal('COMMIT_LEGACY_TERRAFORM_CELL_FENCE_ADOPTION') + }) + .strict() + +const AdminCellFenceAttemptAbortSchema = z + .object({ + v: z.literal(1), + ...CellFenceAttemptBaseShape, + confirmation: z.literal('ABORT_UNSTARTED_TERRAFORM_CELL_FENCE') + }) + .strict() + +const AdminCellFenceAttestSchema = z + .object({ + v: z.literal(1), + ...CellFenceAttemptEvidenceShape, + gceOperation: z.string().min(1).max(256), + confirmation: z.literal('ATTEST_TERRAFORM_FENCED_CELL') + }) + .strict() + +const AdminDrainAttemptPrepareSchema = z + .object({ + v: z.literal(1), + attemptId: z.string().uuid(), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid(), + traceValue: z.string().uuid(), + graceMs: z.literal(120_000), + confirmation: z.literal('PREPARE_LEGACY_DRAIN') + }) + .strict() + +const AdminDrainAttemptMutationSchema = z + .object({ + v: z.literal(1), + attemptId: z.string().uuid(), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid() + }) + .strict() + +const AdminDrainAttemptReceiptSchema = AdminDrainAttemptMutationSchema.extend({ + traceValue: z.string().uuid(), + backendStatus: z.number().int().min(200).max(299), + backendInstance: z.string().min(1).max(256).optional() +}).strict() + +const AdminDrainAttemptRecoverSchema = z + .object({ + v: z.literal(1), + cellId: z.string().min(1).max(128), + cellIncarnation: z.string().uuid(), + confirmation: z.literal('RECOVER_LEGACY_DRAIN') + }) + .strict() + +const RegionalHostDrainSchema = z + .object({ + v: z.literal(1), + attemptId: z.string().uuid(), + userId: z.string().min(1).max(256), + relayHostId: z.string().regex(/^[A-Za-z0-9_-]{16}$/), + sourceCellId: z.string().min(1).max(128), + sourceCellIncarnation: z.string().uuid(), + sourceAssignmentEpoch: z.number().int().positive(), + graceMs: z.number().int().nonnegative().max(60 * 60 * 1000) + }) + .strict() + +const AdminCellStatusSchema = z + .object({ v: z.literal(1), cellId: z.string().min(1).max(128) }) + .strict() + +function cellFenceAttemptEvidence( + value: CellFenceAttemptEvidence +): CellFenceAttemptEvidence { + return { + attemptId: value.attemptId, + environment: value.environment, + cellId: value.cellId, + cellIncarnation: value.cellIncarnation, + migName: value.migName, + instanceGroup: value.instanceGroup, + generationIdentity: value.generationIdentity, + fenceCommit: value.fenceCommit, + planSha256: value.planSha256, + planObjectName: value.planObjectName, + planObjectGeneration: value.planObjectGeneration, + varFileSha256: value.varFileSha256, + terraformStateLineage: value.terraformStateLineage, + terraformStateSerial: value.terraformStateSerial, + terraformStateObjectGeneration: value.terraformStateObjectGeneration, + terraformStateObjectSha256: value.terraformStateObjectSha256, + requestReason: value.requestReason + } +} + +function operationError(error: unknown): string { + return error instanceof Error ? error.message : 'operation_failed' +} + +export { relayHostLogDigest } + +// `suppressed` is present only on a window-closing line, and counts the rejections +// that line stands for beyond the one already logged when the window opened. +function logAssignmentRejection(input: { + route: 'assign' | 'assign-verify' | 'resolve' + lane: 'sticky' | 'placement' | 'none' + hinted: boolean + relayHostId: string + reason: string + suppressed?: number +}): void { + console.warn( + `[orca-relay] assignment rejected route=${input.route} lane=${input.lane}` + + ` hinted=${input.hinted} reason=${input.reason}` + + ` host=${relayHostLogDigest(input.relayHostId)}` + + (input.suppressed === undefined ? '' : ` suppressed=${input.suppressed}`) + ) +} + +function isRelayAssignmentCapacityError(error: unknown): boolean { + return ( + error instanceof Error && + ['relay_capacity_exhausted', 'relay_connection_headroom_exhausted'].includes( + error.message + ) + ) +} + +function isCanonicalRelayOrigin(value: string): boolean { + const url = new URL(value) + const loopback = ['127.0.0.1', 'localhost', '::1', '[::1]'].includes(url.hostname) + return ( + url.origin === value && + url.pathname === '/' && + (url.protocol === 'https:' || (loopback && url.protocol === 'http:')) + ) +} + +function requestTooLarge(contentLength: string | undefined): boolean { + return Number(contentLength ?? 0) > RELAY_PROTOCOL_LIMITS.maxHttpBodyBytes +} diff --git a/cloud/apps/relay/src/assignment-cleanup-steps.test.ts b/cloud/apps/relay/src/assignment-cleanup-steps.test.ts new file mode 100644 index 00000000000..02c21f76ada --- /dev/null +++ b/cloud/apps/relay/src/assignment-cleanup-steps.test.ts @@ -0,0 +1,67 @@ +import { describe, expect, it, vi } from 'vitest' +import { + assignmentCleanupSteps, + runAssignmentCleanup, + type AssignmentCleanupStore +} from './assignment-cleanup-steps.js' + +function stubStore(overrides: Partial = {}) { + const calls: string[] = [] + const method = (name: string) => + vi.fn(async () => { + calls.push(name) + }) + const store: AssignmentCleanupStore = { + refreshRegionalRehomeLeases: method('refreshRegionalRehomeLeases'), + completeReadyEvacuations: method('completeReadyEvacuations'), + completeReadyRegionalRehomes: method('completeReadyRegionalRehomes'), + abortExpiredEvacuations: method('abortExpiredEvacuations'), + abortExpiredRegionalRehomes: method('abortExpiredRegionalRehomes'), + reapRegionalRehomeAttempts: method('reapRegionalRehomeAttempts'), + releaseExpiredActivityLeases: method('releaseExpiredActivityLeases'), + releaseExpiredActivity: method('releaseExpiredActivity'), + releaseExpiredRegionPreferences: method('releaseExpiredRegionPreferences'), + evacuateDeadCells: method('evacuateDeadCells'), + ...overrides + } + return { store, calls } +} + +describe('assignment cleanup steps', () => { + it('runs every later sweep when an early one fails, naming the step', async () => { + const { store, calls } = stubStore({ + completeReadyRegionalRehomes: vi.fn(async () => { + throw new Error('regional_rehome_assignment_mismatch') + }) + }) + const warn = vi.fn() + + await runAssignmentCleanup(store, warn) + + expect(calls).toEqual([ + 'refreshRegionalRehomeLeases', + 'completeReadyEvacuations', + 'abortExpiredEvacuations', + 'abortExpiredRegionalRehomes', + 'reapRegionalRehomeAttempts', + 'releaseExpiredActivityLeases', + 'releaseExpiredActivity', + 'releaseExpiredRegionPreferences', + 'evacuateDeadCells' + ]) + expect(warn).toHaveBeenCalledTimes(1) + expect(String(warn.mock.calls[0]![0])).toContain( + '[orca-relay] assignment cleanup failed: complete-ready-regional-rehomes' + ) + }) + + it('covers all ten sweeps exactly once per run', async () => { + const { store, calls } = stubStore() + + await runAssignmentCleanup(store) + + expect(calls).toHaveLength(10) + expect(new Set(calls).size).toBe(10) + expect(assignmentCleanupSteps(store)).toHaveLength(10) + }) +}) diff --git a/cloud/apps/relay/src/assignment-cleanup-steps.ts b/cloud/apps/relay/src/assignment-cleanup-steps.ts new file mode 100644 index 00000000000..0b6547d2ae4 --- /dev/null +++ b/cloud/apps/relay/src/assignment-cleanup-steps.ts @@ -0,0 +1,52 @@ +import { runRelayBackgroundOperation } from './relay-background-operation.js' + +// The ten periodic assignment sweeps the director runs every 30s. Each step +// re-derives its state from the database and is idempotent, so they carry no +// intra-tick ordering dependency — which is what makes per-step isolation +// sound: one failing sweep costs one tick of itself, never the other nine. +// (A single poisoned rehome row once silenced the whole chained form +// fleet-wide.) Sweep failures are logged, never fed into the rehome worker's +// dispatch-failure budget: a sweep exception is not a dispatch failure and +// must not durably disable regional rehoming. +export type AssignmentCleanupStore = { + refreshRegionalRehomeLeases(): Promise + completeReadyEvacuations(): Promise + completeReadyRegionalRehomes(): Promise + abortExpiredEvacuations(): Promise + abortExpiredRegionalRehomes(): Promise + reapRegionalRehomeAttempts(): Promise + releaseExpiredActivityLeases(): Promise + releaseExpiredActivity(): Promise + releaseExpiredRegionPreferences(): Promise + evacuateDeadCells(): Promise +} + +export function assignmentCleanupSteps( + assignments: AssignmentCleanupStore +): ReadonlyArray Promise]> { + return [ + ['refresh-regional-rehome-leases', () => assignments.refreshRegionalRehomeLeases()], + ['complete-ready-evacuations', () => assignments.completeReadyEvacuations()], + ['complete-ready-regional-rehomes', () => assignments.completeReadyRegionalRehomes()], + ['abort-expired-evacuations', () => assignments.abortExpiredEvacuations()], + ['abort-expired-regional-rehomes', () => assignments.abortExpiredRegionalRehomes()], + ['reap-regional-rehome-attempts', () => assignments.reapRegionalRehomeAttempts()], + ['release-expired-activity-leases', () => assignments.releaseExpiredActivityLeases()], + ['release-expired-activity', () => assignments.releaseExpiredActivity()], + ['release-expired-region-preferences', () => assignments.releaseExpiredRegionPreferences()], + ['evacuate-dead-cells', () => assignments.evacuateDeadCells()] + ] +} + +export async function runAssignmentCleanup( + assignments: AssignmentCleanupStore, + warn?: (message: string) => void +): Promise { + for (const [step, operation] of assignmentCleanupSteps(assignments)) { + await runRelayBackgroundOperation( + operation, + `[orca-relay] assignment cleanup failed: ${step}`, + warn + ) + } +} diff --git a/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts new file mode 100644 index 00000000000..6ac9521c3d6 --- /dev/null +++ b/cloud/apps/relay/src/assignment-connection-headroom-postgres.test.ts @@ -0,0 +1,342 @@ +import pg from 'pg' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { ASSIGNMENT_CONNECTION_HEADROOM_QUERY } from './assignment-connection-headroom-query.js' +import { RelayAssignmentStore } from './assignment-store.js' +import { + openRelayDatabase, + type RelayDatabase +} from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const headroomIndexName = 'relay_control_connection_reservation_headroom' +const cell = { + id: 'connection-headroom-postgres', + url: 'https://connection-headroom-postgres.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 +} + +describePostgres('PostgreSQL assignment connection headroom', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push( + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }) + ) + }) + + afterAll(async () => { + const database = databases[0] + if (database) { + await database.query( + `DELETE FROM relay_control_connection_reservations + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + await database.query( + `DELETE FROM relay_assignments + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + await database.query( + `DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, + [cell.id] + ) + await database.query( + `DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, + [cell.id] + ) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [cell.id]) + } + for (const connection of databases) await connection.close() + }) + + it('commits only one parallel assignment at the admission boundary', async () => { + const stores = databases.map( + (database) => + new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + ) + await databases[0]!.query( + `DELETE FROM relay_control_connection_reservations + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + await databases[0]!.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + await databases[0]!.query( + `DELETE FROM relay_assignments + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + await stores[0]!.reconcileCells([cell], false) + await stores[0]!.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + for (const suffix of ['1', '2']) { + await databases[0]!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + `connection-headroom-postgres-${suffix}`, + `headroomhost000${suffix}`, + cell.id, + 1, + 90_100, + 100, + 0, + 0, + 0, + 0, + 0, + 0 + ] + ) + } + + const results = await Promise.allSettled([ + stores[1]!.assign({ + userId: 'connection-headroom-postgres-1', + relayHostId: 'headroomhost0001' + }), + stores[2]!.assign({ + userId: 'connection-headroom-postgres-2', + relayHostId: 'headroomhost0002' + }) + ]) + expect(results.filter(({ status }) => status === 'fulfilled')).toHaveLength(1) + expect(results.filter(({ status }) => status === 'rejected')).toHaveLength(1) + + const assignments = await databases[0]!.query( + `SELECT COALESCE(SUM(reserved_controls), 0) AS count FROM relay_assignments + WHERE user_id LIKE 'connection-headroom-postgres-%'` + ) + const pending = await databases[0]!.query( + `SELECT COUNT(*) AS count FROM relay_assignment_activity_leases + WHERE user_id LIKE 'connection-headroom-postgres-%' + AND activity_kind = 'control' + AND activity_id LIKE 'control-pending:%'` + ) + const reservations = await databases[0]!.query( + `SELECT COUNT(*) AS count FROM relay_control_connection_reservations + WHERE user_id LIKE 'connection-headroom-postgres-%' + AND state <> 'released'` + ) + expect(Number(assignments[0]!.count)).toBe(1) + expect(Number(pending[0]!.count)).toBe(1) + expect(Number(reservations[0]!.count)).toBe(1) + }, 15_000) + + it('uses the composite headroom index when released history dominates', async () => { + await databases[0]!.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, created_at, timeout_at, updated_at) + SELECT 'headroom-plan-' || value, 'headroom-plan-' || value, + 'connection-headroom-postgres-plan', 'plan-host-' || value, + 1, ?, 'released', 1, 2, 1 + FROM generate_series(1, 50000) AS value`, + [cell.id] + ) + await databases[0]!.query(`ANALYZE relay_control_connection_reservations`) + + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + let reservationIndexPlan: Record | undefined + try { + const plan = await client.query( + `EXPLAIN (FORMAT JSON) ${ASSIGNMENT_CONNECTION_HEADROOM_QUERY}` + ) + reservationIndexPlan = findReservationHeadroomIndexPlan( + plan.rows[0]?.['QUERY PLAN'] + ) + } finally { + await client.end() + } + + expect(reservationIndexPlan).toBeDefined() + expect(String(reservationIndexPlan?.['Index Cond'])).toContain('cell_id') + expect(String(reservationIndexPlan?.['Index Cond'])).toContain('state = ANY') + expect(reservationIndexPlan?.['Filter']).toBeUndefined() + }) + + it('deduplicates expired debt after a fresh PostgreSQL snapshot', async () => { + const now = 200 + const store = new RelayAssignmentStore(databases[0]!, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + const identity = { + userId: 'connection-headroom-postgres-debt', + relayHostId: 'headroomdebt0001' + } + await databases[0]!.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES + (?, ?, ?, ?, 1, ?, 'late-arrival-debt', NULL, NULL, 100, 150, NULL, NULL, 150), + (?, ?, ?, ?, 1, ?, 'late-arrival-debt', NULL, NULL, 101, 150, NULL, NULL, 150)`, + [ + 'postgres-debt-1', + 'postgres-debt-1', + identity.userId, + identity.relayHostId, + cell.id, + 'postgres-debt-2', + 'postgres-debt-2', + identity.userId, + identity.relayHostId, + cell.id + ] + ) + + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionInclusionWatermark: 10, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + expect( + await databases[0]!.query( + `SELECT state + FROM relay_control_connection_reservations + WHERE user_id = ? + ORDER BY created_at`, + [identity.userId] + ) + ).toEqual([ + { state: 'late-arrival-debt' }, + { state: 'released' } + ]) + }) + + it('releases an aborted older epoch after a fresh PostgreSQL snapshot', async () => { + const now = 300 + const store = new RelayAssignmentStore(databases[0]!, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + const identity = { + userId: 'connection-headroom-postgres-aborted', + relayHostId: 'headroomabort001' + } + await databases[0]!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, 3, ?, ?, 0, 0, 0, 0, 0, 0)`, + [identity.userId, identity.relayHostId, cell.id, now, now] + ) + await databases[0]!.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, 1, 2, 0, 1, ?, ?, NULL, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 'source', + cell.id, + now - 2, + now - 3, + now - 1, + now - 4, + now - 1 + ] + ) + await databases[0]!.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES (?, ?, ?, ?, 2, ?, 'late-arrival-debt', NULL, NULL, ?, ?, NULL, NULL, ?)`, + [ + 'postgres-aborted-reservation', + 'postgres-aborted-reservation', + identity.userId, + identity.relayHostId, + cell.id, + now - 4, + now - 2, + now - 1 + ] + ) + + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionInclusionWatermark: 20, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + expect( + await databases[0]!.query( + `SELECT state FROM relay_control_connection_reservations + WHERE reservation_id = ?`, + ['postgres-aborted-reservation'] + ) + ).toEqual([{ state: 'released' }]) + }) +}) + +function findReservationHeadroomIndexPlan( + value: unknown +): Record | undefined { + if (value === null || typeof value !== 'object') return undefined + const record = value as Record + if (!Array.isArray(value) && record['Index Name'] === headroomIndexName) return record + for (const child of Object.values(value)) { + const match = findReservationHeadroomIndexPlan(child) + if (match) return match + } + return undefined +} diff --git a/cloud/apps/relay/src/assignment-connection-headroom-query.ts b/cloud/apps/relay/src/assignment-connection-headroom-query.ts new file mode 100644 index 00000000000..c0237aecf26 --- /dev/null +++ b/cloud/apps/relay/src/assignment-connection-headroom-query.ts @@ -0,0 +1,15 @@ +export const ASSIGNMENT_CONNECTION_HEADROOM_QUERY = + `SELECT limits.cell_id, limits.hard_cap, limits.unobserved_bound, + connection_snapshot.enforced_connection_units, + connection_snapshot.snapshot_at AS last_heartbeat_at, + connection_snapshot.cell_incarnation AS connection_incarnation, + current_runtime.cell_incarnation AS current_incarnation, + (SELECT COUNT(*) FROM relay_control_connection_reservations reservation + WHERE reservation.cell_id = limits.cell_id + AND reservation.state IN + ('reserved', 'late-arrival-debt', 'claimed')) AS outstanding_reservations + FROM relay_cell_connection_limits limits + LEFT JOIN relay_cell_connection_snapshots connection_snapshot + ON connection_snapshot.cell_id = limits.cell_id + LEFT JOIN relay_cell_runtime current_runtime + ON current_runtime.cell_id = limits.cell_id` diff --git a/cloud/apps/relay/src/assignment-connection-headroom.test.ts b/cloud/apps/relay/src/assignment-connection-headroom.test.ts new file mode 100644 index 00000000000..19f5d46aa33 --- /dev/null +++ b/cloud/apps/relay/src/assignment-connection-headroom.test.ts @@ -0,0 +1,1007 @@ +import { ASSIGNMENT_LIMITS } from '@orca-cloud/relay-contract' +import { afterEach, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayCellConfig } from './config.js' +import { + openInMemoryRelayDatabase, + type RelayDatabase +} from './database.js' + +const LIMITED_CELL: RelayCellConfig = { + id: 'limited', + url: 'https://limited.example.com', + capacityRequests: 1_000, + connectionHardCap: 600, + connectionUnobservedBound: 50 +} + +const databases: RelayDatabase[] = [] + +afterEach(async () => { + for (const database of databases.splice(0)) await database.close() +}) + +async function setup( + connectionUnits: number, + now: () => number = () => 100 +): Promise<{ database: RelayDatabase; store: RelayAssignmentStore }> { + const database = await openInMemoryRelayDatabase() + databases.push(database) + const store = new RelayAssignmentStore(database, now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells([LIMITED_CELL], true) + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: connectionUnits, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: connectionUnits, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + return { database, store } +} + +async function setupHeadroomReassignment(): Promise<{ + database: RelayDatabase + store: RelayAssignmentStore + source: RelayCellConfig + target: RelayCellConfig +}> { + const database = await openInMemoryRelayDatabase() + databases.push(database) + const store = new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + const source = { + ...LIMITED_CELL, + id: 'saturated-source', + url: 'https://saturated-source.example.com' + } + const target = { + ...LIMITED_CELL, + id: 'available-target', + url: 'https://available-target.example.com' + } + await store.reconcileCells([source, target], true) + await store.recordCellHeartbeat({ + cellId: source.id, + cellUrl: source.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 450, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 450, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + await store.recordCellHeartbeat({ + cellId: target.id, + cellUrl: target.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + return { database, store, source, target } +} + +describe('relay assignment connection headroom', () => { + it('requires exact, internally consistent telemetry from limited cells', async () => { + const database = await openInMemoryRelayDatabase() + databases.push(database) + const store = new RelayAssignmentStore(database, () => 100) + await store.reconcileCells([LIMITED_CELL], true) + const heartbeat = { + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0 + } + + await expect(store.recordCellHeartbeat(heartbeat)).rejects.toThrow( + 'cell_connection_telemetry_mismatch' + ) + await expect( + store.recordCellHeartbeat({ + ...heartbeat, + totalConnections: 550, + inFlightConnections: 2, + reservedConnectionUnits: 3, + enforcedConnectionUnits: 554, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + ).rejects.toThrow('cell_connection_telemetry_mismatch') + await expect( + store.recordCellHeartbeat({ + ...heartbeat, + totalConnections: 601, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 601, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + ).resolves.toBeUndefined() + }) + + it('reports the limited-cell capacity contract and telemetry components', async () => { + const { store } = await setup(0) + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 120, + inFlightConnections: 2, + reservedConnectionUnits: 3, + enforcedConnectionUnits: 125, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + expect((await store.cellDeploymentStatus(LIMITED_CELL.id)).connectionCapacity).toEqual({ + hardCap: 600, + controlRebindReserve: 100, + ordinaryConnectionLimit: 500, + unobservedBound: 50, + normalAdmissionPause: 450, + observedConnections: 120, + inFlightConnections: 2, + reservedConnectionUnits: 3, + enforcedConnectionUnits: 125, + pendingControlReservations: 0, + heartbeatFresh: true + }) + }) + + it('fails placement closed while a cell changes connection limits', async () => { + const { store } = await setup(449) + const expanded = { + ...LIMITED_CELL, + connectionHardCap: 1_000 as const + } + + await store.reconcileCells([expanded], true) + await expect( + store.assign({ userId: 'transition-user', relayHostId: 'host000000000099' }) + ).rejects.toThrow('relay_capacity_exhausted') + await expect(store.cellDeploymentStatus(expanded.id)).resolves.toMatchObject({ + connectionCapacity: { hardCap: 1_000, heartbeatFresh: false } + }) + + await store.recordCellHeartbeat({ + cellId: expanded.id, + cellUrl: expanded.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 200, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionHardCap: 1_000, + connectionUnobservedBound: 50 + }) + await expect(store.cellDeploymentStatus(expanded.id)).resolves.toMatchObject({ + connectionCapacity: { + hardCap: 1_000, + ordinaryConnectionLimit: 900, + normalAdmissionPause: 850, + heartbeatFresh: true + } + }) + }) + + it('fails placement closed while the admin path changes connection limits', async () => { + const { store } = await setup(449) + const expanded = { + ...LIMITED_CELL, + connectionHardCap: 1_000 as const + } + + await store.configureCell(expanded, 'general') + await expect( + store.assign({ userId: 'admin-transition', relayHostId: 'host000000000098' }) + ).rejects.toThrow('relay_capacity_exhausted') + await expect(store.cellDeploymentStatus(expanded.id)).resolves.toMatchObject({ + connectionCapacity: { hardCap: 1_000, heartbeatFresh: false } + }) + + await store.recordCellHeartbeat({ + cellId: expanded.id, + cellUrl: expanded.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 200, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionHardCap: 1_000, + connectionUnobservedBound: 50 + }) + await expect( + store.assign({ userId: 'admin-restored', relayHostId: 'host000000000097' }) + ).resolves.toMatchObject({ cellId: expanded.id }) + }) + + it('admits exactly one reservation at the admission boundary', async () => { + const { database, store } = await setup(449) + const results = await Promise.allSettled([ + store.assign({ userId: 'user-1', relayHostId: 'host000000000001' }), + store.assign({ userId: 'user-2', relayHostId: 'host000000000002' }) + ]) + + expect(results.filter(({ status }) => status === 'fulfilled')).toHaveLength(1) + expect( + results.filter(({ status }) => status === 'rejected').map((result) => + result.status === 'rejected' ? result.reason : null + ) + ).toEqual([expect.objectContaining({ message: 'relay_capacity_exhausted' })]) + expect( + await database.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE activity_kind = 'control' AND activity_id LIKE 'control-pending:%'` + ) + ).toHaveLength(1) + }) + + it('rejects at the boundary before mutating durable assignment state', async () => { + const { database, store } = await setup(450) + + await expect( + store.assign({ userId: 'user-1', relayHostId: 'host000000000001' }) + ).rejects.toThrow('relay_capacity_exhausted') + expect(await database.query(`SELECT * FROM relay_assignments`)).toEqual([]) + expect(await database.query(`SELECT * FROM relay_assignment_activity_leases`)).toEqual([]) + }) + + it('charges every active reservation state at the admission boundary', async () => { + const { database, store } = await setup(447) + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, created_at, timeout_at, updated_at) + VALUES + ('active-reserved', 'active-reserved', 'state-user', 'state-host-1', + 1, ?, 'reserved', 1, 2, 1), + ('active-debt', 'active-debt', 'state-user', 'state-host-2', + 1, ?, 'late-arrival-debt', 1, 2, 1), + ('active-claimed', 'active-claimed', 'state-user', 'state-host-3', + 1, ?, 'claimed', 1, 2, 1)`, + [LIMITED_CELL.id, LIMITED_CELL.id, LIMITED_CELL.id] + ) + + await expect( + store.assign({ userId: 'blocked-user', relayHostId: 'blockedhost00001' }) + ).rejects.toThrow('relay_capacity_exhausted') + expect(await database.query(`SELECT * FROM relay_assignment_activity_leases`)).toEqual([]) + }) + + it('does not charge released reservation history', async () => { + const { database, store } = await setup(449) + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, created_at, timeout_at, updated_at) + VALUES ('released-history', 'released-history', 'state-user', 'state-host', + 1, ?, 'released', 1, 2, 1)`, + [LIMITED_CELL.id] + ) + + await expect( + store.assign({ userId: 'admitted-user', relayHostId: 'admittedhost0001' }) + ).resolves.toMatchObject({ cellId: LIMITED_CELL.id }) + }) + + it('rejects sticky control restoration before mutating assignment state', async () => { + const { database, store } = await setup(450) + const identity = { userId: 'sticky-user', relayHostId: 'stickyhost000001' } + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + LIMITED_CELL.id, + 1, + 10_000, + 100, + 0, + 0, + 1, + 0, + 0, + 0 + ] + ) + + await expect(store.assign(identity)).rejects.toThrow( + 'relay_connection_headroom_exhausted' + ) + expect( + await database.query( + `SELECT cell_id, assignment_epoch, reserved_controls, reserved_invites + FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { + cell_id: LIMITED_CELL.id, + assignment_epoch: 1, + reserved_controls: 0, + reserved_invites: 1 + } + ]) + expect(await database.query(`SELECT * FROM relay_assignment_activity_leases`)).toEqual([]) + }) + + it('moves a zero-activity assignment off a general cell without connection headroom', async () => { + const { database, store, source, target } = await setupHeadroomReassignment() + const identity = { userId: 'movable-user', relayHostId: 'movablehost000001' } + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, 0, 0, 0, 0, 0, 0)`, + [identity.userId, identity.relayHostId, source.id, 7, 10_000, 100] + ) + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES (?, ?, ?, ?, ?, ?, 'late-arrival-debt', NULL, NULL, ?, ?, NULL, NULL, ?)`, + [ + 'superseded-reservation', + 'superseded-reservation', + identity.userId, + identity.relayHostId, + 7, + source.id, + 50, + 99, + 100 + ] + ) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: target.id, + assignmentEpoch: 8 + }) + expect( + await database.query( + `SELECT cell_id, assignment_epoch, reserved_controls + FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: target.id, assignment_epoch: 8, reserved_controls: 1 }]) + expect( + await database.query( + `SELECT cell_id, activity_kind FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: target.id, activity_kind: 'control' }]) + expect( + await database.query( + `SELECT cell_id, assignment_epoch, state + FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? + ORDER BY assignment_epoch`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { cell_id: source.id, assignment_epoch: 7, state: 'released' }, + { cell_id: target.id, assignment_epoch: 8, state: 'reserved' } + ]) + }) + + it('keeps non-control activity pinned when its cell has no connection headroom', async () => { + const { database, store, source } = await setupHeadroomReassignment() + const identity = { userId: 'pinned-user', relayHostId: 'pinnedhost0000001' } + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, 0, 0, 1, 0, 0, 0)`, + [identity.userId, identity.relayHostId, source.id, 7, 10_000, 100] + ) + + await expect(store.assign(identity)).rejects.toThrow( + 'relay_connection_headroom_exhausted' + ) + expect(await store.resolve(identity)).toMatchObject({ + cellId: source.id, + assignmentEpoch: 7 + }) + }) + + it.each(['existing-only', 'migration-only'] as const)( + 'keeps a zero-activity assignment pinned on a %s cell without connection headroom', + async (admission) => { + const { database, store, source } = await setupHeadroomReassignment() + const identity = { + userId: `pinned-${admission}-user`, + relayHostId: `pinned${admission.replace('-', '')}` + } + await store.setCellAdmissionState(source.id, admission) + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, 0, 0, 0, 0, 0, 0)`, + [identity.userId, identity.relayHostId, source.id, 7, 10_000, 100] + ) + + await expect(store.assign(identity)).rejects.toThrow( + 'relay_connection_headroom_exhausted' + ) + expect(await store.resolve(identity)).toMatchObject({ + cellId: source.id, + assignmentEpoch: 7 + }) + expect(await database.query(`SELECT * FROM relay_assignment_activity_leases`)).toEqual([]) + } + ) + + it('renames a pending reservation exactly once on control activation', async () => { + const { database, store } = await setup(449) + const identity = { userId: 'user-1', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + + await store.activateControl(identity, { + cellId: LIMITED_CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.activateControl(identity, { + cellId: LIMITED_CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + + expect( + await database.query( + `SELECT activity_id, request_units FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ activity_id: 'control:limited:1', request_units: 1 }]) + }) + + it('keeps expired pending reservations charged as late-arrival debt', async () => { + let now = 100 + const { database, store } = await setup(449, () => now) + await store.assign({ userId: 'user-1', relayHostId: 'host000000000001' }) + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + await store.releaseExpiredActivityLeases() + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + await expect( + store.assign({ userId: 'user-2', relayHostId: 'host000000000002' }) + ).rejects.toThrow('relay_capacity_exhausted') + expect( + await database.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE activity_kind = 'control' AND activity_id LIKE 'control-pending:%'` + ) + ).toHaveLength(0) + expect( + await database.query( + `SELECT state FROM relay_control_connection_reservations` + ) + ).toEqual([{ state: 'late-arrival-debt' }]) + }) + + it('reconciles late-arrival debt only after a covering absolute snapshot', async () => { + let now = 100 + const { database, store } = await setup(449, () => now) + const identity = { userId: 'late-user', relayHostId: 'latehost00000001' } + const assignment = await store.assign(identity) + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + await store.releaseExpiredActivityLeases() + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionInclusionWatermark: 4, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + await store.activateControl(identity, { + cellId: LIMITED_CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 5 + }) + expect( + await database.query( + `SELECT state, inclusion_watermark FROM relay_control_connection_reservations` + ) + ).toEqual([{ state: 'claimed', inclusion_watermark: 5 }]) + + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 450, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 450, + connectionInclusionWatermark: 5, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + expect( + await database.query( + `SELECT state, released_at FROM relay_control_connection_reservations` + ) + ).toEqual([{ state: 'released', released_at: now }]) + await expect( + store.assign({ userId: 'blocked-user', relayHostId: 'blockedhost00001' }) + ).rejects.toThrow('relay_capacity_exhausted') + }) + + it('rejects duplicate and out-of-order absolute snapshots', async () => { + const { database, store } = await setup(0) + const heartbeat = { + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 10, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 10, + connectionInclusionWatermark: 10, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } + await store.recordCellHeartbeat(heartbeat) + await expect(store.recordCellHeartbeat(heartbeat)).rejects.toThrow( + 'stale_connection_snapshot' + ) + await expect( + store.recordCellHeartbeat({ + ...heartbeat, + totalConnections: 9, + enforcedConnectionUnits: 9, + connectionInclusionWatermark: 9 + }) + ).rejects.toThrow('stale_connection_snapshot') + expect( + await database.query( + `SELECT inclusion_watermark, enforced_connection_units + FROM relay_cell_connection_snapshots` + ) + ).toEqual([{ inclusion_watermark: 10, enforced_connection_units: 10 }]) + }) + + it('keeps crash retries bound to one reservation claim', async () => { + let now = 100 + const { database, store } = await setup(448, () => now) + const identity = { userId: 'retry-user', relayHostId: 'retryhost0000001' } + const assignment = await store.assign(identity) + await new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }).assign(identity) + expect( + await database.query( + `SELECT state FROM relay_control_connection_reservations` + ) + ).toEqual([{ state: 'reserved' }]) + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + await store.releaseExpiredActivityLeases() + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 448, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 448, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + await store.assign(identity) + const restarted = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + + const activation = { + cellId: LIMITED_CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 5 + } + await restarted.activateControl(identity, activation) + await restarted.activateControl(identity, activation) + + expect( + await database.query( + `SELECT state, claim_activity_id + FROM relay_control_connection_reservations + ORDER BY created_at ASC, reservation_id ASC` + ) + ).toEqual([{ state: 'claimed', claim_activity_id: 'control:limited:1' }]) + }) + + it('releases only redundant expired debt after a fresh absolute snapshot', async () => { + let now = 100 + const { database, store } = await setup(449, () => now) + const identity = { userId: 'debt-user', relayHostId: 'debthost000000001' } + const assignment = await store.assign(identity) + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + await store.releaseExpiredActivityLeases() + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES (?, ?, ?, ?, ?, ?, 'late-arrival-debt', NULL, NULL, ?, ?, NULL, NULL, ?)`, + [ + 'duplicate-reservation', + 'duplicate-reservation', + identity.userId, + identity.relayHostId, + assignment.assignmentEpoch, + LIMITED_CELL.id, + 101, + now - 1, + now + ] + ) + + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 449, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 449, + connectionInclusionWatermark: 5, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + expect( + await database.query( + `SELECT state, COUNT(*) AS count + FROM relay_control_connection_reservations + GROUP BY state ORDER BY state` + ) + ).toEqual([ + { state: 'late-arrival-debt', count: 1 }, + { state: 'released', count: 1 } + ]) + }) + + it('releases expired debt for a snapshot-proven aborted older epoch', async () => { + const now = 1_000 + const { database, store } = await setup(0, () => now) + const identity = { userId: 'aborted-user', relayHostId: 'abortedhost00001' } + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, 0, 0, 0, 0, 0, 0)`, + [identity.userId, identity.relayHostId, LIMITED_CELL.id, 3, now, now] + ) + await database.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, 1, 2, 0, 1, ?, ?, NULL, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 'source', + LIMITED_CELL.id, + now - 2, + now - 3, + now - 1, + now - 4, + now - 1 + ] + ) + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES (?, ?, ?, ?, 2, ?, 'late-arrival-debt', NULL, NULL, ?, ?, NULL, NULL, ?)`, + [ + 'aborted-reservation', + 'aborted-reservation', + identity.userId, + identity.relayHostId, + LIMITED_CELL.id, + now - 4, + now - 2, + now - 1 + ] + ) + + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 5, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + + expect( + await database.query( + `SELECT state FROM relay_control_connection_reservations + WHERE reservation_id = ?`, + ['aborted-reservation'] + ) + ).toEqual([{ state: 'released' }]) + }) + + it('fails a limited cell closed on stale telemetry while leaving legacy cells unchanged', async () => { + let now = 100 + const { store } = await setup(0, () => now) + now += 45_001 + await expect( + store.assign({ userId: 'user-1', relayHostId: 'host000000000001' }) + ).rejects.toThrow('relay_capacity_exhausted') + + const legacyDatabase = await openInMemoryRelayDatabase() + databases.push(legacyDatabase) + const legacyStore = new RelayAssignmentStore(legacyDatabase, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + const legacy = { + id: 'legacy', + url: 'https://legacy.example.com', + capacityRequests: 10 + } + await legacyStore.reconcileCells([legacy], true) + await legacyStore.recordCellHeartbeat({ + cellId: legacy.id, + cellUrl: legacy.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: now, + ready: true, + observedRequests: 0 + }) + await expect( + legacyStore.assign({ userId: 'user-2', relayHostId: 'host000000000002' }) + ).resolves.toMatchObject({ cellId: 'legacy' }) + }) + + it('does not create connection debt for mixed-version uncapped cells', async () => { + let now = 100 + const legacy = { + id: 'legacy', + url: 'https://legacy.example.com', + capacityRequests: 10 + } + const { database, store } = await setup(0, () => now) + await store.reconcileCells([legacy], true) + await store.recordCellHeartbeat({ + cellId: legacy.id, + cellUrl: legacy.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 50, + ready: true, + observedRequests: 0 + }) + + await store.assign({ userId: 'legacy-user', relayHostId: 'legacyhost000001' }) + expect( + await database.query(`SELECT * FROM relay_control_connection_reservations`) + ).toEqual([]) + expect(await store.releaseExpiredActivityLeases()).toBe(0) + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + expect(await store.releaseExpiredActivityLeases()).toBe(1) + expect( + await database.query(`SELECT * FROM relay_control_connection_reservations`) + ).toEqual([]) + }) + + it('rejects dormant rebalance before changing the assignment epoch', async () => { + const now = ASSIGNMENT_LIMITS.dormantTtlMs + 100 + const database = await openInMemoryRelayDatabase() + databases.push(database) + const store = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + const source = { + id: 'source', + url: 'https://source.example.com', + capacityRequests: 10 + } + await store.reconcileCells([source, LIMITED_CELL], true) + await store.recordCellHeartbeat({ + cellId: source.id, + cellUrl: source.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: now - 50, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: now - 50, + ready: true, + observedRequests: 0, + totalConnections: 450, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 450, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + const identity = { userId: 'dormant-user', relayHostId: 'dormanthost00001' } + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + source.id, + 7, + 0, + 0, + 0, + 0, + 0, + 0, + 0, + 0 + ] + ) + + await expect(store.rebalanceDormant(identity, LIMITED_CELL.id)).rejects.toThrow( + 'relay_connection_headroom_exhausted' + ) + expect(await store.resolve(identity)).toMatchObject({ + cellId: source.id, + assignmentEpoch: 7 + }) + expect(await database.query(`SELECT * FROM relay_assignment_activity_leases`)).toEqual([]) + }) + + it('rejects an evacuation target at its pause before migration mutation', async () => { + const database = await openInMemoryRelayDatabase() + databases.push(database) + const store = new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + const source = { + id: 'source', + url: 'https://source.example.com', + capacityRequests: 10 + } + await store.reconcileCells( + [source, { ...LIMITED_CELL, initiallyEnabled: false }], + true + ) + await store.recordCellHeartbeat({ + cellId: source.id, + cellUrl: source.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 50, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 450, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 450, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + const identity = { userId: 'user-1', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(source.id) + await store.setCellEnabled(LIMITED_CELL.id, true) + + await expect( + store.startEvacuation(identity, LIMITED_CELL.id) + ).rejects.toThrow('relay_connection_headroom_exhausted') + expect(await database.query(`SELECT * FROM relay_assignment_migrations`)).toEqual([]) + expect(await store.resolve(identity)).toMatchObject({ + cellId: source.id, + assignmentEpoch: assignment.assignmentEpoch + }) + }) +}) diff --git a/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts new file mode 100644 index 00000000000..10193b78cc6 --- /dev/null +++ b/cloud/apps/relay/src/assignment-control-supersession-postgres.test.ts @@ -0,0 +1,130 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const cell = { + id: 'control-supersession-postgres', + url: 'https://control-supersession-postgres.example.com', + capacityRequests: 1_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 +} +const identity = { + userId: 'control-supersession-postgres-user', + relayHostId: 'supersedehost001' +} + +describePostgres('PostgreSQL control supersession', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push( + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }) + ) + }) + + afterAll(async () => { + const database = databases[0] + if (database) { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, + [identity.userId] + ) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + [identity.userId] + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [identity.userId]) + await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [cell.id]) + } + for (const connection of databases) await connection.close() + }) + + it('serializes parallel generations into one durable control', async () => { + const stores = databases.map((database) => new RelayAssignmentStore(database, () => 100)) + await stores[0]!.reconcileCells([cell]) + await stores[0]!.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark: 1, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + const assignment = await stores[0]!.assign(identity) + + await Promise.all([ + stores[0]!.activateControl(identity, { + cellId: cell.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 10 + }), + stores[1]!.activateControl(identity, { + cellId: cell.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 2, + connectionInclusionWatermark: 11 + }) + ]) + await databases[0]!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, 'control', ?, 1, 90100, 100), + (?, ?, ?, 'control', ?, 1, 90100, 100)`, + [ + identity.userId, + identity.relayHostId, + `control:${cell.id}:100`, + cell.id, + identity.userId, + identity.relayHostId, + `control:${cell.id}:101`, + cell.id + ] + ) + await stores[0]!.activateControl(identity, { + cellId: cell.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 102, + connectionInclusionWatermark: 12 + }) + + const controls = await databases[0]!.query( + `SELECT COUNT(*) AS count FROM relay_assignment_activity_leases + WHERE user_id = ? AND activity_kind = 'control'`, + [identity.userId] + ) + expect(Number(controls[0]!.count)).toBe(1) + const assignments = await databases[0]!.query( + `SELECT reserved_controls FROM relay_assignments WHERE user_id = ?`, + [identity.userId] + ) + expect(Number(assignments[0]!.reserved_controls)).toBe(1) + const cells = await databases[0]!.query( + `SELECT reserved_requests FROM relay_cells WHERE cell_id = ?`, + [cell.id] + ) + expect(Number(cells[0]!.reserved_requests)).toBe(1) + const claims = await databases[0]!.query( + `SELECT claim_activity_id FROM relay_control_connection_reservations + WHERE user_id = ? AND state = 'claimed'`, + [identity.userId] + ) + expect(claims).toEqual([{ claim_activity_id: `control:${cell.id}:102` }]) + }, 15_000) +}) diff --git a/cloud/apps/relay/src/assignment-deployment-status-postgres.test.ts b/cloud/apps/relay/src/assignment-deployment-status-postgres.test.ts new file mode 100644 index 00000000000..9e8e3b85bd9 --- /dev/null +++ b/cloud/apps/relay/src/assignment-deployment-status-postgres.test.ts @@ -0,0 +1,87 @@ +import pg from 'pg' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const schema = 'relay_deployment_status_test' +const cell = { + id: 'deployment-status-postgres', + url: 'https://deployment-status-postgres.example.com', + capacityRequests: 20 +} + +describePostgres('PostgreSQL deployment status', () => { + let database: RelayDatabase | undefined + + beforeAll(async () => { + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + await client.query(`CREATE SCHEMA ${schema}`) + } finally { + await client.end() + } + const url = new URL(databaseUrl!) + url.searchParams.set('options', `-c search_path=${schema}`) + database = await openRelayDatabase({ databaseUrl: url.toString(), dataDir: '' }) + }) + + afterAll(async () => { + await database?.close() + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + } finally { + await client.end() + } + }) + + it('classifies pending controls and restart blockers in one PostgreSQL query', async () => { + const store = new RelayAssignmentStore(database!, () => 100) + await store.reconcileCells([cell]) + await store.assign({ userId: 'status-user', relayHostId: 'a1b2c3d4e5f6' }) + + expect(await store.cellDeploymentStatus(cell.id)).toMatchObject({ + activityLeases: 1, + activityRequestUnits: 1, + reservedRequests: 1, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 0 + }) + + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES ('blocking-user', 'b1c2d3e4f5a6', 'invite:test', 'invite', ?, 1, 90100, 100), + ('malformed-user', 'c1d2e3f4a5b6', 'control-pending:bad', 'control', ?, 2, 90100, 100)`, + [cell.id, cell.id] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = reserved_requests + 3 WHERE cell_id = ?`, + [cell.id] + ) + + expect(await store.cellDeploymentStatus(cell.id)).toMatchObject({ + activityLeases: 3, + activityRequestUnits: 4, + reservedRequests: 4, + restartBlockingActivityLeases: 2, + restartBlockingActivityRequestUnits: 3, + restartBlockingReservedRequests: 3 + }) + + await database!.query( + `UPDATE relay_cells SET reserved_requests = 0 WHERE cell_id = ?`, + [cell.id] + ) + expect(await store.cellDeploymentStatus(cell.id)).toMatchObject({ + restartBlockingReservedRequests: -1 + }) + }) +}) diff --git a/cloud/apps/relay/src/assignment-identity-queue.test.ts b/cloud/apps/relay/src/assignment-identity-queue.test.ts new file mode 100644 index 00000000000..930f0b692df --- /dev/null +++ b/cloud/apps/relay/src/assignment-identity-queue.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it } from 'vitest' +import { AssignmentIdentityQueue } from './assignment-identity-queue.js' + +function deferred(): { + promise: Promise + resolve: () => void +} { + let resolve!: () => void + const promise = new Promise((complete) => { + resolve = complete + }) + return { promise, resolve } +} + +describe('AssignmentIdentityQueue', () => { + it('serializes operations for the same assignment', async () => { + const queue = new AssignmentIdentityQueue() + const firstStarted = deferred() + const firstRelease = deferred() + const started: string[] = [] + const identity = { userId: 'user-1', relayHostId: 'host-1' } + + const first = queue.run(identity, async () => { + started.push('first') + firstStarted.resolve() + await firstRelease.promise + }) + const second = queue.run(identity, async () => { + started.push('second') + }) + + await firstStarted.promise + expect(started).toEqual(['first']) + firstRelease.resolve() + await Promise.all([first, second]) + expect(started).toEqual(['first', 'second']) + }) + + it('allows different assignments to run concurrently', async () => { + const queue = new AssignmentIdentityQueue() + const release = deferred() + const started: string[] = [] + + const first = queue.run({ userId: 'user-1', relayHostId: 'host-1' }, async () => { + started.push('first') + await release.promise + }) + const second = queue.run({ userId: 'user-2', relayHostId: 'host-1' }, async () => { + started.push('second') + }) + + await second + expect(started).toEqual(['first', 'second']) + release.resolve() + await first + }) + + it('continues after a rejected operation', async () => { + const queue = new AssignmentIdentityQueue() + const identity = { userId: 'user-1', relayHostId: 'host-1' } + + const failed = queue.run(identity, async () => { + throw new Error('failed') + }) + const recovered = queue.run(identity, async () => 'recovered') + + await expect(failed).rejects.toThrow('failed') + await expect(recovered).resolves.toBe('recovered') + }) +}) diff --git a/cloud/apps/relay/src/assignment-identity-queue.ts b/cloud/apps/relay/src/assignment-identity-queue.ts new file mode 100644 index 00000000000..f39d9c0a03c --- /dev/null +++ b/cloud/apps/relay/src/assignment-identity-queue.ts @@ -0,0 +1,24 @@ +export type AssignmentIdentity = { + userId: string + relayHostId: string +} + +export class AssignmentIdentityQueue { + private readonly tails = new Map>() + + async run(identity: AssignmentIdentity, operation: () => Promise): Promise { + const key = JSON.stringify([identity.userId, identity.relayHostId]) + const previous = this.tails.get(key) ?? Promise.resolve() + const result = previous.catch(() => undefined).then(operation) + const tail = result.then( + () => undefined, + () => undefined + ) + this.tails.set(key, tail) + try { + return await result + } finally { + if (this.tails.get(key) === tail) this.tails.delete(key) + } + } +} diff --git a/cloud/apps/relay/src/assignment-inventory-snapshot.test.ts b/cloud/apps/relay/src/assignment-inventory-snapshot.test.ts new file mode 100644 index 00000000000..afc35f7e601 --- /dev/null +++ b/cloud/apps/relay/src/assignment-inventory-snapshot.test.ts @@ -0,0 +1,107 @@ +import { describe, expect, it } from 'vitest' +import { + formatAssignmentInventorySnapshot, + readAssignmentInventorySnapshot +} from './assignment-inventory-snapshot.js' +import { openInMemoryRelayDatabase } from './database.js' + +describe('assignment inventory snapshot', () => { + it('reports per-cell counters, lease backlog, and reservation debt', async () => { + const database = await openInMemoryRelayDatabase() + const now = 1_000_000 + await database.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES ('cell-a', 'https://a.example.test', 1, 4000, 3999, 5, ?, ?)`, + [now, now] + ) + await database.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES ('cell-a', 'general', ?)`, + [now] + ) + await database.query( + `INSERT INTO relay_cell_runtime + (cell_id, cell_url, cell_incarnation, started_at, ready, observed_requests, + last_heartbeat_at, updated_at) + VALUES ('cell-a', 'https://a.example.test', 'inc-1', ?, 1, 5, ?, ?)`, + [now - 60_000, now - 10_000, now] + ) + await database.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, request_units, + expires_at, updated_at) + VALUES + ('user-1', 'host-1', 'control:1', 'control', 'cell-a', 1, ?, ?), + ('user-1', 'host-2', 'control:1', 'control', 'cell-a', 3, ?, ?)`, + [now - 1, now, now + 90_000, now] + ) + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, assignment_epoch, + cell_id, state, created_at, timeout_at, updated_at) + VALUES + ('r1', 'k1', 'user-1', 'host-1', 1, 'cell-a', 'late-arrival-debt', ?, ?, ?), + ('r2', 'k2', 'user-1', 'host-2', 1, 'cell-a', 'reserved', ?, ?, ?), + ('r3', 'k3', 'user-1', 'host-3', 1, 'cell-a', 'claimed', ?, ?, ?), + ('r4', 'k4', 'user-1', 'host-4', 1, 'cell-a', 'released', ?, ?, ?)`, + [now, now, now, now, now, now, now, now, now, now, now, now] + ) + + const snapshot = await readAssignmentInventorySnapshot(database, now) + + expect(snapshot.cells).toEqual([ + { + cellId: 'cell-a', + region: 'us-central1', + admissionState: 'general', + enabled: true, + capacityRequests: 4000, + reservedRequests: 3999, + runtimeReady: true, + heartbeatAgeMs: 10_000 + } + ]) + expect(snapshot.activityLeases).toEqual({ total: 2, expired: 1, requestUnits: 4 }) + expect(snapshot.connectionReservations).toEqual({ outstanding: 3, lateArrivalDebt: 1 }) + expect(snapshot.regionalRehomes).toEqual({ + active: 0, + awaitingReceipt: 0, + targetRegistered: 0, + completedLast24Hours: 0, + abortedLast24Hours: 0, + oldestActiveAgeMs: null + }) + + const lines = formatAssignmentInventorySnapshot(snapshot) + expect(lines).toHaveLength(3) + expect(lines[0]).toContain('cellId=cell-a') + expect(lines[0]).toContain('reserved=3999') + expect(lines[1]).toContain('expiredLeases=1') + expect(lines[1]).toContain('lateArrivalDebt=1') + expect(lines[2]).toContain('active=0') + }) + + it('reports cells missing runtime and admission rows without failing', async () => { + const database = await openInMemoryRelayDatabase() + await database.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES ('cell-b', 'https://b.example.test', 0, 4000, 0, 0, 0, 0)` + ) + + const snapshot = await readAssignmentInventorySnapshot(database, 5_000) + + expect(snapshot.cells[0]).toMatchObject({ + cellId: 'cell-b', + admissionState: 'unset', + enabled: false, + runtimeReady: null, + heartbeatAgeMs: null + }) + expect(snapshot.activityLeases).toEqual({ total: 0, expired: 0, requestUnits: 0 }) + expect(snapshot.regionalRehomes.active).toBe(0) + }) +}) diff --git a/cloud/apps/relay/src/assignment-inventory-snapshot.ts b/cloud/apps/relay/src/assignment-inventory-snapshot.ts new file mode 100644 index 00000000000..0675bd49b94 --- /dev/null +++ b/cloud/apps/relay/src/assignment-inventory-snapshot.ts @@ -0,0 +1,170 @@ +import type { RelayDatabase, SqlRow } from './database.js' + +export type CellInventorySnapshotRow = { + cellId: string + region: string + admissionState: string + enabled: boolean + capacityRequests: number + reservedRequests: number + runtimeReady: boolean | null + heartbeatAgeMs: number | null +} + +export type AssignmentInventorySnapshot = { + cells: CellInventorySnapshotRow[] + activityLeases: { total: number; expired: number; requestUnits: number } + connectionReservations: { outstanding: number; lateArrivalDebt: number } + regionalRehomes: { + active: number + awaitingReceipt: number + targetRegistered: number + completedLast24Hours: number + abortedLast24Hours: number + oldestActiveAgeMs: number | null + } +} + +// Plain SELECTs only: this snapshot must never take the cell-inventory lock, +// or observability itself would add to the assign-path lock contention. +export async function readAssignmentInventorySnapshot( + database: RelayDatabase, + now: number +): Promise { + const cellRows = await database.query( + `SELECT cell.cell_id, cell.enabled, cell.capacity_requests, cell.reserved_requests, + admission.admission_state, region.region, + runtime.ready AS runtime_ready, runtime.last_heartbeat_at AS runtime_heartbeat_at + FROM relay_cells cell + LEFT JOIN relay_cell_admission admission ON admission.cell_id = cell.cell_id + LEFT JOIN relay_cell_regions region ON region.cell_id = cell.cell_id + LEFT JOIN relay_cell_runtime runtime ON runtime.cell_id = cell.cell_id + ORDER BY cell.cell_id ASC` + ) + const leaseRow = ( + await database.query( + `SELECT COUNT(*) AS total, + COALESCE(SUM(CASE WHEN expires_at <= ? THEN 1 ELSE 0 END), 0) AS expired, + COALESCE(SUM(request_units), 0) AS request_units + FROM relay_assignment_activity_leases`, + [now] + ) + )[0] + const reservationRow = ( + await database.query( + // relay_cells is append-only in production; joining it lets the composite + // index skip released history without hiding any production cell's debt. + `SELECT COUNT(reservation.reservation_id) AS outstanding, + COALESCE(SUM(CASE WHEN reservation.state = 'late-arrival-debt' + THEN 1 ELSE 0 END), 0) AS late_arrival_debt + FROM relay_cells cell + LEFT JOIN relay_control_connection_reservations reservation + ON reservation.cell_id = cell.cell_id + AND reservation.state IN ('reserved', 'late-arrival-debt', 'claimed')` + ) + )[0] + const regionalRehomeRow = ( + await database.query( + `SELECT + COALESCE(SUM(CASE WHEN attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + THEN 1 ELSE 0 END), 0) AS active, + COALESCE(SUM(CASE WHEN attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + AND attempt.drain_receipt_at IS NULL + THEN 1 ELSE 0 END), 0) AS awaiting_receipt, + COALESCE(SUM(CASE WHEN attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + AND migration.target_registered_at IS NOT NULL + THEN 1 ELSE 0 END), 0) AS target_registered, + COALESCE(SUM(CASE WHEN attempt.completed_at >= ? THEN 1 ELSE 0 END), 0) + AS completed_last_24_hours, + COALESCE(SUM(CASE WHEN attempt.aborted_at >= ? THEN 1 ELSE 0 END), 0) + AS aborted_last_24_hours, + MIN(CASE WHEN attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + THEN attempt.created_at END) AS oldest_active_at + FROM relay_region_rehome_attempts attempt + LEFT JOIN relay_assignment_migrations migration + ON migration.user_id = attempt.user_id + AND migration.relay_host_id = attempt.relay_host_id + AND migration.assignment_epoch = attempt.assignment_epoch`, + [now - 24 * 60 * 60_000, now - 24 * 60 * 60_000] + ) + )[0] + const oldestActiveAt = optionalInteger(regionalRehomeRow, 'oldest_active_at') + return { + cells: cellRows.map((row) => ({ + cellId: asText(row, 'cell_id'), + region: optionalText(row, 'region') ?? 'us-central1', + admissionState: optionalText(row, 'admission_state') ?? 'unset', + enabled: asInteger(row, 'enabled') === 1, + capacityRequests: asInteger(row, 'capacity_requests'), + reservedRequests: asInteger(row, 'reserved_requests'), + runtimeReady: row['runtime_ready'] == null ? null : asInteger(row, 'runtime_ready') === 1, + heartbeatAgeMs: + row['runtime_heartbeat_at'] == null ? null : now - asInteger(row, 'runtime_heartbeat_at') + })), + activityLeases: { + total: asInteger(leaseRow, 'total'), + expired: asInteger(leaseRow, 'expired'), + requestUnits: asInteger(leaseRow, 'request_units') + }, + connectionReservations: { + outstanding: asInteger(reservationRow, 'outstanding'), + lateArrivalDebt: asInteger(reservationRow, 'late_arrival_debt') + }, + regionalRehomes: { + active: asInteger(regionalRehomeRow, 'active'), + awaitingReceipt: asInteger(regionalRehomeRow, 'awaiting_receipt'), + targetRegistered: asInteger(regionalRehomeRow, 'target_registered'), + completedLast24Hours: asInteger(regionalRehomeRow, 'completed_last_24_hours'), + abortedLast24Hours: asInteger(regionalRehomeRow, 'aborted_last_24_hours'), + oldestActiveAgeMs: oldestActiveAt === null ? null : now - oldestActiveAt + } + } +} + +export function formatAssignmentInventorySnapshot( + snapshot: AssignmentInventorySnapshot +): string[] { + const lines = snapshot.cells.map( + (cell) => + `[orca-relay] cell inventory cellId=${cell.cellId}` + + ` region=${cell.region}` + + ` admission=${cell.admissionState} enabled=${cell.enabled}` + + ` capacity=${cell.capacityRequests} reserved=${cell.reservedRequests}` + + ` ready=${cell.runtimeReady ?? 'none'}` + + ` heartbeatAgeMs=${cell.heartbeatAgeMs ?? 'none'}` + ) + lines.push( + `[orca-relay] lease inventory leases=${snapshot.activityLeases.total}` + + ` expiredLeases=${snapshot.activityLeases.expired}` + + ` leaseRequestUnits=${snapshot.activityLeases.requestUnits}` + + ` outstandingReservations=${snapshot.connectionReservations.outstanding}` + + ` lateArrivalDebt=${snapshot.connectionReservations.lateArrivalDebt}` + ) + lines.push( + `[orca-relay] regional rehome inventory active=${snapshot.regionalRehomes.active}` + + ` awaitingReceipt=${snapshot.regionalRehomes.awaitingReceipt}` + + ` targetRegistered=${snapshot.regionalRehomes.targetRegistered}` + + ` completedLast24Hours=${snapshot.regionalRehomes.completedLast24Hours}` + + ` abortedLast24Hours=${snapshot.regionalRehomes.abortedLast24Hours}` + + ` oldestActiveAgeMs=${snapshot.regionalRehomes.oldestActiveAgeMs ?? 'none'}` + ) + return lines +} + +function asText(row: SqlRow | undefined, column: string): string { + return String(row?.[column] ?? '') +} + +function optionalText(row: SqlRow | undefined, column: string): string | null { + const value = row?.[column] + return value == null ? null : String(value) +} + +function asInteger(row: SqlRow | undefined, column: string): number { + return Number(row?.[column] ?? 0) +} + +function optionalInteger(row: SqlRow | undefined, column: string): number | null { + const value = row?.[column] + return value == null ? null : Number(value) +} diff --git a/cloud/apps/relay/src/assignment-rejection-log-window.test.ts b/cloud/apps/relay/src/assignment-rejection-log-window.test.ts new file mode 100644 index 00000000000..8ef078eff86 --- /dev/null +++ b/cloud/apps/relay/src/assignment-rejection-log-window.test.ts @@ -0,0 +1,100 @@ +import { describe, expect, it } from 'vitest' +import { AssignmentRejectionLogWindow } from './assignment-rejection-log-window.js' + +describe('assignment rejection log window', () => { + it('emits once per key per window and closes it with its own suppressed count', () => { + const timers = manualTimers() + const closed: { key: string; suppressed: number; sample: string }[] = [] + const window = new AssignmentRejectionLogWindow({ + windowMs: 10_000, + schedule: timers.schedule, + onWindowClosed: (input) => closed.push(input) + }) + const key = 'assign:placement:host-rate-limited' + + expect(window.admit(key, 'host-a')).toBe(true) + expect(window.admit(key, 'host-b')).toBe(false) + expect(window.admit(key, 'host-c')).toBe(false) + expect(closed).toEqual([]) + + timers.runPending() + expect(closed).toEqual([{ key, suppressed: 2, sample: 'host-c' }]) + + // The next window starts clean instead of inheriting the closed window's count. + expect(window.admit(key, 'host-d')).toBe(true) + timers.runPending() + expect(closed).toHaveLength(1) + }) + + it('reports the final count for a key that goes quiet', () => { + const timers = manualTimers() + const closed: { key: string; suppressed: number }[] = [] + const window = new AssignmentRejectionLogWindow({ + windowMs: 10_000, + schedule: timers.schedule, + onWindowClosed: ({ key, suppressed }) => closed.push({ key, suppressed }) + }) + + window.admit('assign:sticky:host-rate-limited', 'host-a') + window.admit('assign:sticky:host-rate-limited', 'host-a') + window.admit('assign:sticky:host-rate-limited', 'host-a') + // No further rejection ever arrives for this key. + timers.runPending() + + expect(closed).toEqual([{ key: 'assign:sticky:host-rate-limited', suppressed: 2 }]) + }) + + it('keeps distinct keys on independent windows', () => { + const timers = manualTimers() + const window = new AssignmentRejectionLogWindow({ + windowMs: 10_000, + schedule: timers.schedule, + onWindowClosed: () => undefined + }) + + expect(window.admit('assign:placement:host-rate-limited', 'host-a')).toBe(true) + expect(window.admit('assign:placement:queue-full', 'host-a')).toBe(true) + expect(window.admit('assign:sticky:host-rate-limited', 'host-a')).toBe(true) + expect(window.admit('assign:placement:queue-full', 'host-a')).toBe(false) + }) + + it('bounds the tracked keys and reports what an evicted window suppressed', () => { + const timers = manualTimers() + const closed: { key: string; suppressed: number }[] = [] + const window = new AssignmentRejectionLogWindow({ + windowMs: 10_000, + schedule: timers.schedule, + onWindowClosed: ({ key, suppressed }) => closed.push({ key, suppressed }) + }) + + window.admit('key-0', 'host-a') + window.admit('key-0', 'host-b') + for (let index = 1; index < 200; index++) window.admit(`key-${index}`, 'host-a') + + expect(closed[0]).toEqual({ key: 'key-0', suppressed: 1 }) + // Evicted keys emit again rather than staying silently suppressed. + expect(window.admit('key-0', 'host-a')).toBe(true) + expect(window.admit('key-199', 'host-a')).toBe(false) + }) +}) + +function manualTimers(): { + schedule: (callback: () => void) => () => void + runPending: () => void +} { + const pending = new Map void>() + let nextId = 0 + return { + schedule: (callback) => { + const id = nextId++ + pending.set(id, callback) + return () => pending.delete(id) + }, + runPending: () => { + for (const [id, callback] of [...pending]) { + pending.delete(id) + callback() + } + } + } +} diff --git a/cloud/apps/relay/src/assignment-rejection-log-window.ts b/cloud/apps/relay/src/assignment-rejection-log-window.ts new file mode 100644 index 00000000000..21ad4e72af9 --- /dev/null +++ b/cloud/apps/relay/src/assignment-rejection-log-window.ts @@ -0,0 +1,60 @@ +type CancelWindow = () => void + +// Admission rejections run at ~17/s per director instance, so the log summarizes +// instead of streaming: one line when a key's window opens, then a closing line +// carrying whatever that same window suppressed. The closing line is driven by the +// window's own timer, so a key that falls quiet still reports its final count +// instead of waiting for a rejection that may never arrive. +const MAX_TRACKED_KEYS = 64 + +type RejectionWindow = { + suppressed: number + sample: TSample + cancel: CancelWindow +} + +export class AssignmentRejectionLogWindow { + private readonly windows = new Map>() + + constructor( + private readonly options: { + windowMs: number + onWindowClosed: (input: { key: string; suppressed: number; sample: TSample }) => void + schedule?: (callback: () => void, delayMs: number) => CancelWindow + } + ) {} + + // Returns true when the caller should log this rejection immediately. + admit(key: string, sample: TSample): boolean { + const open = this.windows.get(key) + if (open) { + open.suppressed++ + // Keep the most recent host/reason so the closing line names a real rejection. + open.sample = sample + return false + } + const schedule = this.options.schedule ?? defaultSchedule + const window: RejectionWindow = { suppressed: 0, sample, cancel: () => undefined } + this.windows.set(key, window) + if (this.windows.size > MAX_TRACKED_KEYS) { + this.close(this.windows.keys().next().value!) + } + window.cancel = schedule(() => this.close(key), this.options.windowMs) + return true + } + + private close(key: string): void { + const window = this.windows.get(key) + if (!window) return + this.windows.delete(key) + window.cancel() + if (window.suppressed === 0) return + this.options.onWindowClosed({ key, suppressed: window.suppressed, sample: window.sample }) + } +} + +function defaultSchedule(callback: () => void, delayMs: number): CancelWindow { + const timer = setTimeout(callback, delayMs) + timer.unref?.() + return () => clearTimeout(timer) +} diff --git a/cloud/apps/relay/src/assignment-rejection-logging.test.ts b/cloud/apps/relay/src/assignment-rejection-logging.test.ts new file mode 100644 index 00000000000..b00d791fce9 --- /dev/null +++ b/cloud/apps/relay/src/assignment-rejection-logging.test.ts @@ -0,0 +1,244 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RelayAssignment } from './assignment-store.js' +import type { RelayConfig } from './config.js' + +const fakes = vi.hoisted(() => ({ + verifyRelayToken: vi.fn(async (token: string) => ({ + sub: 'user-1', + relayHostId: token + })) +})) + +vi.mock('./relay-token-verifier.js', () => ({ + createRelayTokenVerifier: () => fakes.verifyRelayToken, + readBearer: (value: string | undefined) => value?.replace(/^Bearer /, '') ?? null +})) + +import { createRelayApp, relayHostLogDigest } from './app.js' + +describe('assignment rejection logging', () => { + afterEach(() => { + vi.useRealTimers() + vi.restoreAllMocks() + }) + + it('logs the store reason and host digest when assign() rejects on capacity', async () => { + const host = 'hhhhhhhhhhhhhhhh' + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { + assign: vi.fn(async () => { + throw new Error('relay_capacity_exhausted') + }), + resolve: vi.fn(async () => assignment('cell-r', host)) + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const hinted = await app.request('/v1/assign', assignmentRequest(host, { reconnect: true })) + + expect(hinted.status).toBe(503) + expect(await hinted.json()).toEqual({ error: 'relay_capacity_exhausted' }) + const line = warn.mock.calls.map((call) => String(call[0])).find((entry) => + entry.includes('assignment rejected') + ) + expect(line).toContain('route=assign') + expect(line).toContain('lane=sticky') + expect(line).toContain('hinted=true') + expect(line).toContain('reason=relay_capacity_exhausted') + expect(line).toContain(`host=${relayHostLogDigest(host)}`) + expect(line).not.toContain(host) + }) + + it('logs an unhinted placement rejection without the raw host id', async () => { + const host = 'gggggggggggggggg' + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { + assign: vi.fn(async () => { + throw new Error('relay_connection_headroom_exhausted') + }), + resolve: vi.fn(async () => null) + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const response = await app.request('/v1/assign', assignmentRequest(host)) + + expect(response.status).toBe(503) + const line = warn.mock.calls.map((call) => String(call[0])).find((entry) => + entry.includes('assignment rejected') + ) + expect(line).toContain('route=assign') + expect(line).toContain('lane=placement') + expect(line).toContain('hinted=false') + expect(line).toContain('reason=relay_connection_headroom_exhausted') + expect(line).not.toContain(host) + }) + + it('names the throttled host once per window and closes it with the suppressed count', async () => { + vi.useFakeTimers() + const host = 'mmmmmmmmmmmmmmmm' + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { + assign: vi.fn(async () => assignment('cell-r', host)), + resolve: vi.fn(async () => null) + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + expect((await app.request('/v1/assign', assignmentRequest(host))).status).toBe(200) + expect((await app.request('/v1/assign', assignmentRequest(host))).status).toBe(503) + expect((await app.request('/v1/assign', assignmentRequest(host))).status).toBe(503) + + const throttled = (): string[] => + warn.mock.calls + .map((call) => String(call[0])) + .filter((entry) => entry.includes('reason=host-rate-limited')) + expect(throttled()).toHaveLength(1) + expect(throttled()[0]).toContain('route=assign') + expect(throttled()[0]).toContain('lane=placement') + expect(throttled()[0]).toContain(`host=${relayHostLogDigest(host)}`) + expect(throttled()[0]).not.toContain('suppressed=') + expect(throttled()[0]).not.toContain(host) + + await vi.advanceTimersByTimeAsync(10_000) + expect(throttled()).toHaveLength(2) + expect(throttled()[1]).toContain('suppressed=1') + expect(throttled()[1]).toContain(`host=${relayHostLogDigest(host)}`) + }) + + it('stays silent for successful assignments', async () => { + const host = 'ssssssssssssssss' + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { + assign: vi.fn(async () => assignment('cell-r', host)), + resolve: vi.fn(async () => null) + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + expect((await app.request('/v1/assign', assignmentRequest(host))).status).toBe(200) + expect( + warn.mock.calls.map((call) => String(call[0])).filter((entry) => + entry.includes('assignment rejected') + ) + ).toEqual([]) + }) +}) + +function assignmentRequest( + relayHostId: string, + extra: Record = {} +): RequestInit & { headers: Record } { + return { + method: 'POST', + headers: { + authorization: `Bearer ${relayHostId}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, relayHostId, ...extra }) + } +} + +function assignment(cellId: string, relayHostId: string): RelayAssignment { + return { + userId: 'user-1', + relayHostId, + cellId, + cellUrl: `https://${cellId}.relay.example.test`, + assignmentEpoch: 1, + leaseExpiresAt: Date.now() + 300_000 + } +} + +function config(overrides: Partial = {}): RelayConfig { + return { + port: 8080, + publicUrl: 'https://relay.example.test', + cellUrl: 'https://relay.example.test', + authIssuer: 'https://auth.example.test', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.test/jwks', + assignmentSigningKey: new TextEncoder().encode('assignment-key-with-at-least-32-bytes'), + role: 'director', + cellId: 'director', + cells: [], + adminAudience: 'https://relay.example.test/v1/admin/drain', + deployServiceAccount: 'deploy@example.test', + runtimeServiceAccount: 'runtime@example.test', + adminJwksUrl: 'https://auth.example.test/jwks', + databasePoolMax: 3, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './data', + ...overrides + } +} + +describe('assignment grant logging', () => { + afterEach(() => { + vi.restoreAllMocks() + }) + + it('logs sticky grants with cell id and host digest only', async () => { + const host = 'gggggggggggggggg' + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { + assign: vi.fn(async () => assignment('cell-r', host)), + resolve: vi.fn(async () => assignment('cell-r', host)) + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const response = await app.request('/v1/assign', assignmentRequest(host, { reconnect: true })) + + expect(response.status).toBe(200) + const line = warn.mock.calls.map((call) => String(call[0])).find((entry) => + entry.includes('assignment granted') + ) + expect(line).toContain('lane=sticky') + expect(line).toContain('cell=cell-r') + expect(line).toContain(`host=${relayHostLogDigest(host)}`) + expect(line).not.toContain(host) + }) + + it('does not log unhinted placement grants', async () => { + const host = 'pppppppppppppppp' + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { + assign: vi.fn(async () => assignment('cell-r', host)), + resolve: vi.fn(async () => null) + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + expect((await app.request('/v1/assign', assignmentRequest(host))).status).toBe(200) + expect( + warn.mock.calls.map((call) => String(call[0])).filter((entry) => + entry.includes('assignment granted') + ) + ).toEqual([]) + }) +}) diff --git a/cloud/apps/relay/src/assignment-store-lock-order.test.ts b/cloud/apps/relay/src/assignment-store-lock-order.test.ts new file mode 100644 index 00000000000..61baedfb1e2 --- /dev/null +++ b/cloud/apps/relay/src/assignment-store-lock-order.test.ts @@ -0,0 +1,486 @@ +import { describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayDatabase, RelayLockOptions, SqlRow } from './database.js' + +const identity = { userId: 'user-a', relayHostId: 'host000000000001' } +const activityId = 'splice:connection-1' + +class LockOrderDatabase implements RelayDatabase { + readonly lockedTables: string[] = [] + + constructor( + private readonly cleanupCandidate: boolean, + private readonly currentLeaseExpiresAt: number, + private readonly activityLeasePresent = true + ) {} + + async query(sql: string): Promise { + if (sql.includes('SELECT user_id, relay_host_id, activity_id')) { + return this.cleanupCandidate + ? [ + { + user_id: identity.userId, + relay_host_id: identity.relayHostId, + activity_id: activityId + } + ] + : [] + } + if ( + sql.includes('UPDATE relay_cells SET reserved_requests') && + sql.includes('RETURNING cell_id') + ) { + this.lockedTables.push('cell') + return [{ cell_id: 'cell-a' }] + } + return [{ changes: 1 }] + } + + async queryLocked(sql: string): Promise { + if (sql.includes('FROM relay_assignments ')) { + this.lockedTables.push('assignment') + return [{ cell_id: 'cell-a', assignment_epoch: 1 }] + } + if (sql.includes('FROM relay_assignment_activity_leases')) { + this.lockedTables.push('activity') + if (!this.activityLeasePresent) return [] + return [ + { + user_id: identity.userId, + relay_host_id: identity.relayHostId, + activity_id: activityId, + activity_kind: 'splice', + cell_id: 'cell-a', + request_units: 2, + expires_at: this.currentLeaseExpiresAt + } + ] + } + if (sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC') { + this.lockedTables.push('cell-inventory') + return [{ cell_id: 'cell-a', reserved_requests: 3, capacity_requests: 10 }] + } + if (sql.includes('FROM relay_cells')) { + this.lockedTables.push('cell') + return [{ reserved_requests: 3, capacity_requests: 10 }] + } + return [] + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +class ReassignmentLockOrderDatabase implements RelayDatabase { + readonly locks: string[] = [] + + async query(sql: string): Promise { + if (sql.includes('SELECT cell_id, region FROM relay_cell_regions')) { + return ['cell-a', 'cell-b'].map((cell_id) => ({ cell_id, region: 'us-central1' })) + } + if (sql.includes('SELECT region FROM relay_cell_regions')) { + return [{ region: 'us-central1' }] + } + if (sql.includes('SELECT * FROM relay_cells WHERE cell_id')) { + return [cellRow('cell-b', 1)] + } + if (sql.includes('JOIN relay_cell_runtime') && sql.includes('cell.cell_id = ?')) { + return [] + } + if (sql.includes('SELECT cell_id, observed_requests FROM relay_cell_runtime')) { + return [{ cell_id: 'cell-a', observed_requests: 0 }] + } + if (sql.includes('LEFT JOIN relay_cell_admission')) { + return [ + { cell_id: 'cell-a', admission_state: 'general' }, + { cell_id: 'cell-b', admission_state: 'general' } + ] + } + if (sql.includes('FROM relay_cell_connection_limits')) return [] + return [{ changes: 1 }] + } + + async queryLocked(sql: string, params: unknown[] = []): Promise { + if (sql.includes('FROM relay_assignments')) { + this.locks.push('assignment') + return [ + { + user_id: identity.userId, + relay_host_id: identity.relayHostId, + cell_id: 'cell-b', + assignment_epoch: 1, + lease_expires_at: 100, + last_activity_at: 100, + reserved_controls: 0, + reserved_splices: 0, + reserved_invites: 0, + pending_installs: 0, + pending_confirmations: 0, + migration_leases: 0 + } + ] + } + if (sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC') { + this.locks.push('cell-inventory') + return [cellRow('cell-a', 0), cellRow('cell-b', 1)] + } + if (sql.includes('SELECT * FROM relay_cells WHERE cell_id')) { + const cellId = String(params[0]) + this.locks.push(cellId) + return [cellRow(cellId, cellId === 'cell-b' ? 1 : 0)] + } + return [] + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +class AggregateCleanupDatabase implements RelayDatabase { + failIfUnavailable: boolean | null = null + + async query(): Promise { + return [{ changes: 1 }] + } + + async queryLocked( + sql: string, + _params: unknown[] = [], + options: RelayLockOptions = {} + ): Promise { + if (sql.includes('FROM relay_assignments WHERE lease_expires_at')) { + this.failIfUnavailable = options.failIfUnavailable ?? false + } + return [] + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +class HeartbeatLockDatabase implements RelayDatabase { + readonly locks: string[] = [] + readonly reservationCleanupTransactions: number[] = [] + legacyHeartbeatWritten = false + private transactionNumber = 0 + private activeTransaction = 0 + + constructor( + private readonly cleanupIncarnation = '11111111-1111-4111-8111-111111111111' + ) {} + + async query(sql: string, params: unknown[] = []): Promise { + if (sql.includes('SELECT region FROM relay_cell_regions')) { + return [{ cell_id: String(params[0]), region: 'us-central1' }] + } + if (sql.includes('SELECT * FROM relay_cells WHERE cell_id')) { + return [cellRow('cell-a', 0)] + } + if (sql.includes('UPDATE relay_cells SET observed_requests')) { + this.legacyHeartbeatWritten = true + } + if (sql.includes('UPDATE relay_control_connection_reservations')) { + this.reservationCleanupTransactions.push(this.activeTransaction) + } + return [{ changes: 1 }] + } + + async queryLocked(sql: string): Promise { + if (sql.includes('FROM relay_cells')) { + this.locks.push('cell') + return [cellRow('cell-a', 0)] + } + if (sql.includes('FROM relay_cell_runtime')) { + this.locks.push('runtime') + return this.activeTransaction === 2 + ? [{ cell_incarnation: this.cleanupIncarnation }] + : [] + } + if (sql.includes('FROM relay_cell_connection_limits')) { + this.locks.push('connection-limit') + return [{ hard_cap: 600, unobserved_bound: 99 }] + } + if (sql.includes('FROM relay_cell_connection_snapshots')) { + this.locks.push('snapshot') + return this.activeTransaction === 2 + ? [{ cell_incarnation: this.cleanupIncarnation, inclusion_watermark: 0 }] + : [] + } + return [] + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + this.activeTransaction = ++this.transactionNumber + const result = await operation(this) + this.activeTransaction = 0 + return result + } + + async close(): Promise {} +} + +class NewAssignmentLockDatabase implements RelayDatabase { + readonly inventoryLocks: string[] = [] + private generalLockFailed = false + + constructor( + private failGeneralOnce = false, + private readonly assignmentAppearsAfterFailure = false + ) {} + + async query(sql: string, params: unknown[] = []): Promise { + if (sql.includes('SELECT cell_id, region FROM relay_cell_regions')) { + return [ + { cell_id: 'cell-existing', region: 'us-central1' }, + { cell_id: 'cell-general', region: 'us-central1' } + ] + } + if (sql.includes('SELECT region FROM relay_cell_regions')) { + return [{ cell_id: String(params[0]), region: 'us-central1' }] + } + if (sql.includes('SELECT cell_id, observed_requests FROM relay_cell_runtime')) { + return [{ cell_id: 'cell-general', observed_requests: 0 }] + } + if (sql.includes('LEFT JOIN relay_cell_admission')) { + return [{ cell_id: 'cell-general', admission_state: 'general' }] + } + if (sql.includes('JOIN relay_cell_runtime') && sql.includes('cell.cell_id = ?')) { + return [{ cell_id: 'cell-existing' }] + } + if (sql.includes('FROM relay_cell_connection_limits')) { + return [ + { + cell_id: 'cell-general', + hard_cap: 600, + unobserved_bound: 99, + enforced_connection_units: 0, + outstanding_reservations: 0, + last_heartbeat_at: 100, + connection_incarnation: 'incarnation-a', + current_incarnation: 'incarnation-a' + } + ] + } + return [{ changes: 1 }] + } + + async queryLocked(sql: string): Promise { + if ( + sql.includes('FROM relay_assignments') && + this.assignmentAppearsAfterFailure && + this.generalLockFailed + ) { + return [ + { + user_id: identity.userId, + relay_host_id: identity.relayHostId, + cell_id: 'cell-existing', + assignment_epoch: 1, + lease_expires_at: 100, + last_activity_at: 100, + reserved_controls: 1, + reserved_splices: 0, + reserved_invites: 0, + pending_installs: 0, + pending_confirmations: 0, + migration_leases: 0 + } + ] + } + if (sql.includes('SELECT cell_id FROM relay_cell_admission')) { + this.inventoryLocks.push('general') + if (this.failGeneralOnce) { + this.failGeneralOnce = false + this.generalLockFailed = true + throw new Error('database_lock_unavailable') + } + return [cellRow('cell-general', 0)] + } + if (sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC') { + this.inventoryLocks.push('all') + return [cellRow('cell-existing', 0), cellRow('cell-general', 0)] + } + if (sql.includes('SELECT * FROM relay_cells WHERE cell_id')) { + return [cellRow('cell-general', 0)] + } + return [] + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +function cellRow(cellId: string, reservedRequests: number): SqlRow { + return { + cell_id: cellId, + cell_url: `https://${cellId}.example.com`, + enabled: 1, + capacity_requests: 10, + reserved_requests: reservedRequests, + observed_requests: 0 + } +} + +describe('RelayAssignmentStore activity lock order', () => { + it('updates the cell only after assignment activity during release', async () => { + const database = new LockOrderDatabase(false, 100) + const store = new RelayAssignmentStore(database, () => 100) + + await expect(store.releaseActivity(identity, activityId)).resolves.toBe(true) + expect(database.lockedTables).toEqual(['assignment', 'activity', 'cell']) + }) + + it('updates the cell only after inserting new activity', async () => { + const database = new LockOrderDatabase(false, 100, false) + const store = new RelayAssignmentStore(database, () => 100) + + await expect( + store.acquireActivity(identity, { activityId, kind: 'splice', cellId: 'cell-a' }) + ).resolves.toBeUndefined() + expect(database.lockedTables).toEqual(['assignment', 'activity', 'cell']) + }) + + it('rechecks an expired candidate after taking the assignment lock', async () => { + const database = new LockOrderDatabase(true, 101) + const store = new RelayAssignmentStore(database, () => 100) + + await expect(store.releaseExpiredActivityLeases()).resolves.toBe(0) + expect(database.lockedTables).toEqual(['assignment', 'activity']) + }) + + it('fails fast before aggregate cleanup waits on mixed-version assignment rows', async () => { + const database = new AggregateCleanupDatabase() + const store = new RelayAssignmentStore(database, () => 100) + + await expect(store.releaseExpiredActivity()).resolves.toBe(0) + expect(database.failIfUnavailable).toBe(true) + }) + + it('releases the placement capacity lock before heartbeat reservation cleanup', async () => { + const database = new HeartbeatLockDatabase() + const store = new RelayAssignmentStore(database, () => 100, { requireLiveCells: true }) + + await store.recordCellHeartbeat({ + cellId: 'cell-a', + cellUrl: 'https://cell-a.example.com', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionHardCap: 600, + connectionUnobservedBound: 99 + }) + + expect(database.locks).toEqual([ + 'cell', + 'runtime', + 'connection-limit', + 'snapshot', + 'runtime', + 'snapshot' + ]) + expect(database.legacyHeartbeatWritten).toBe(false) + expect(database.reservationCleanupTransactions).toEqual([2, 2, 2]) + }) + + it('does not let an old heartbeat clean replacement-incarnation reservations', async () => { + const database = new HeartbeatLockDatabase('22222222-2222-4222-8222-222222222222') + const store = new RelayAssignmentStore(database, () => 100, { requireLiveCells: true }) + + await store.recordCellHeartbeat({ + cellId: 'cell-a', + cellUrl: 'https://cell-a.example.com', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionHardCap: 600, + connectionUnobservedBound: 99 + }) + + expect(database.reservationCleanupTransactions).toEqual([]) + }) + + it('locks only general-admission inventory for a brand-new assignment', async () => { + const database = new NewAssignmentLockDatabase() + const store = new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45 + }) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-general', + assignmentEpoch: 1 + }) + expect(database.inventoryLocks).toEqual(['general']) + }) + + it('keeps a brand-new assignment retry scoped to general admission', async () => { + const database = new NewAssignmentLockDatabase(true) + const store = new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45 + }) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-general', + assignmentEpoch: 1 + }) + expect(database.inventoryLocks).toEqual(['general', 'general']) + }) + + it('restarts with full inventory if an assignment appears during a general retry', async () => { + const database = new NewAssignmentLockDatabase(true, true) + const store = new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45 + }) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-existing', + assignmentEpoch: 1 + }) + expect(database.inventoryLocks).toEqual(['general', 'general', 'all']) + }) + + it('probes the sticky cell before locking full inventory for dead-cell reassignment', async () => { + const database = new ReassignmentLockOrderDatabase() + const store = new RelayAssignmentStore(database, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45 + }) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-a', + assignmentEpoch: 2 + }) + expect(database.locks).toEqual([ + 'assignment', + 'cell-b', + 'assignment', + 'cell-inventory', + 'cell-b', + 'cell-a' + ]) + }) +}) diff --git a/cloud/apps/relay/src/assignment-store.test.ts b/cloud/apps/relay/src/assignment-store.test.ts new file mode 100644 index 00000000000..e9d27a83db4 --- /dev/null +++ b/cloud/apps/relay/src/assignment-store.test.ts @@ -0,0 +1,4471 @@ +import { ASSIGNMENT_LIMITS } from '@orca-cloud/relay-contract' +import { createHash } from 'node:crypto' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RelayCellConfig } from './config.js' +import { RelayAssignmentStore, STRANDED_MIGRATION_ABANDON_MS } from './assignment-store.js' +import { + encodeMembership, + type CellAdmissionMembership +} from './cell-admission-selector.js' +import { + openInMemoryRelayDatabase, + type RelayDatabase, + type RelayLockOptions, + type RelayTransactionOptions, + type SqlRow +} from './database.js' + +const CELLS: RelayCellConfig[] = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 2 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 2 } +] +const NO_EXPIRED_EVACUATION_DIAGNOSTICS = { + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 +} +const NO_REGISTERED_EVACUATION_DIAGNOSTICS = { + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0 +} +const NO_ACTIVE_MIGRATION_LEASE = { + oldestExpiresAt: null, + oldestRemainingMs: null +} +const FENCE_PLAN_BINDING = { + planObjectName: + 'terraform/state/relay-fence-plans/production/22222222-2222-4222-8222-222222222222.tfplan', + planObjectGeneration: '123456789', + varFileSha256: 'c'.repeat(64), + terraformStateLineage: '33333333-3333-4333-8333-333333333333', + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: 'd'.repeat(64), + requestReason: + 'orca-relay-fence/22222222-2222-4222-8222-222222222222' +} +const FENCE_INVOCATION_ID = '55555555-5555-4555-8555-555555555555' +const FENCE_INVOCATION_REASON = + `${FENCE_PLAN_BINDING.requestReason}/${FENCE_INVOCATION_ID}` + +async function applyGenerationZeroSelector( + store: RelayAssignmentStore, + input: { + attemptId: string + membership: CellAdmissionMembership + } +) { + const current = (await store.inspectCellAdmissionSelector()).selector.membership + return await store.applyCellAdmissionSelector({ + ...input, + expectedGeneration: 0, + expectedMembershipSha256: createHash('sha256') + .update(encodeMembership(current)) + .digest('hex') + }) +} + +function cellFenceEvidence(attemptId = '22222222-2222-4222-8222-222222222222') { + return { + attemptId, + environment: 'production' as const, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + ...FENCE_PLAN_BINDING + } +} + +class OneShotInventoryFailureDatabase implements RelayDatabase { + readonly retryLocks: string[] = [] + private armed = false + private failed = false + + constructor(private readonly delegate: RelayDatabase) {} + + arm(): void { + this.armed = true + } + + async query(sql: string, params: unknown[] = []): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params: unknown[] = [], + options: RelayLockOptions = {} + ): Promise { + return await this.lockedQuery(this.delegate, sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await this.delegate.transaction( + async (transaction) => + await operation({ + query: async (sql, params = []) => await transaction.query(sql, params), + queryLocked: async (sql, params = [], options = {}) => + await this.lockedQuery(transaction, sql, params, options), + transaction: async (nested) => await transaction.transaction(nested), + close: async () => {} + }) + ) + } + + async close(): Promise { + await this.delegate.close() + } + + private async lockedQuery( + database: RelayDatabase, + sql: string, + params: unknown[], + options: RelayLockOptions + ): Promise { + const inventory = sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC' + if (this.armed && inventory && options.failIfUnavailable) { + this.armed = false + this.failed = true + this.retryLocks.push('inventory-nowait-failed') + throw new Error('database_lock_unavailable') + } + if (this.failed && inventory) this.retryLocks.push('inventory-first') + if (this.failed && sql.includes('FROM relay_assignments')) { + this.retryLocks.push(options.failIfUnavailable ? 'assignment-nowait' : 'assignment') + } + return await database.queryLocked(sql, params, options) + } +} + +class RepeatedDrainAccountingFailureDatabase implements RelayDatabase { + refreshAttempts = 0 + private armed = false + private failuresRemaining = 0 + + constructor(private readonly delegate: RelayDatabase) {} + + arm(failures: number): void { + this.armed = true + this.failuresRemaining = failures + this.refreshAttempts = 0 + } + + async query(sql: string, params: unknown[] = []): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params: unknown[] = [], + options: RelayLockOptions = {} + ): Promise { + return await this.lockedQuery(this.delegate, sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await this.delegate.transaction( + async (transaction) => + await operation({ + query: async (sql, params = []) => await transaction.query(sql, params), + queryLocked: async (sql, params = [], options = {}) => + await this.lockedQuery(transaction, sql, params, options), + transaction: async (nested) => await transaction.transaction(nested), + close: async () => {} + }) + ) + } + + async close(): Promise { + await this.delegate.close() + } + + private async lockedQuery( + database: RelayDatabase, + sql: string, + params: unknown[], + options: RelayLockOptions + ): Promise { + const drainRefresh = + sql.includes('SELECT migration.*') && + sql.includes('WHERE migration.source_cell_id = ?') && + !sql.includes('source_cell_incarnation') + if (this.armed && drainRefresh) { + this.refreshAttempts++ + if (this.failuresRemaining > 0) { + this.failuresRemaining-- + throw new Error('migration_activity_accounting_mismatch') + } + } + return await database.queryLocked(sql, params, options) + } +} + +// Runs a side effect between the sticky lane and the placement lane, the window +// in which a host can acquire control after the sticky read called it dormant. +class SeedBetweenAssignmentLanesDatabase implements RelayDatabase { + private transactions = 0 + private armed = false + + constructor( + private readonly delegate: RelayDatabase, + private readonly seed: () => Promise + ) {} + + arm(): void { + this.armed = true + this.transactions = 0 + } + + async query(sql: string, params: unknown[] = []): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params: unknown[] = [], + options: RelayLockOptions = {} + ): Promise { + return await this.delegate.queryLocked(sql, params, options) + } + + async transaction( + operation: (transaction: RelayDatabase) => Promise, + options?: RelayTransactionOptions + ): Promise { + if (this.armed && ++this.transactions === 2) await this.seed() + return await this.delegate.transaction(operation, options) + } + + async close(): Promise { + await this.delegate.close() + } +} + +describe('RelayAssignmentStore', () => { + let database: RelayDatabase | undefined + + afterEach(async () => await database?.close()) + + async function setup(now: () => number, cells = CELLS): Promise { + database = await openInMemoryRelayDatabase() + const store = new RelayAssignmentStore(database, now) + await store.reconcileCells(cells) + return store + } + + async function setupWithHeartbeats( + now: () => number, + cells = CELLS + ): Promise { + database = await openInMemoryRelayDatabase() + const store = new RelayAssignmentStore(database, now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells(cells) + return store + } + + async function heartbeat( + store: RelayAssignmentStore, + cell = CELLS[0]!, + input: { + incarnation?: string + startedAt?: number + ready?: boolean + observedRequests?: number + } = {} + ): Promise { + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: input.incarnation ?? '11111111-1111-4111-8111-111111111111', + startedAt: input.startedAt ?? 50, + ready: input.ready ?? true, + observedRequests: input.observedRequests ?? 0, + ...(cell.connectionHardCap === undefined + ? {} + : { + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionHardCap: cell.connectionHardCap, + connectionUnobservedBound: cell.connectionUnobservedBound + }) + }) + } + + it('admits only cells with a fresh ready heartbeat', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + + await expect(store.assign(identity)).rejects.toThrow('relay_capacity_exhausted') + await heartbeat(store, CELLS[0]!, { ready: false }) + await expect(store.assign(identity)).rejects.toThrow('relay_capacity_exhausted') + await heartbeat(store) + expect((await store.assign(identity)).cellId).toBe('cell-a') + now += 45_001 + expect(await store.resolve(identity)).toBeNull() + }) + + it('fences stale incarnations and origin mismatches', async () => { + const store = await setupWithHeartbeats(() => 100, [CELLS[0]!]) + await heartbeat(store, CELLS[0]!, { startedAt: 50 }) + await expect( + heartbeat(store, { ...CELLS[0]!, url: 'https://other.example.com' }) + ).rejects.toThrow('cell_origin_mismatch') + await expect( + heartbeat(store, CELLS[0]!, { + incarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 50 + }) + ).rejects.toThrow('stale_cell_incarnation') + await heartbeat(store, CELLS[0]!, { + incarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 51 + }) + await expect(heartbeat(store, CELLS[0]!, { startedAt: 50 })).rejects.toThrow( + 'stale_cell_incarnation' + ) + }) + + it('records receipt-relative legacy drain states and pins post-send migrations', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await database!.query( + `UPDATE relay_assignments SET migration_leases = migration_leases + 1 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + const incarnation = '11111111-1111-4111-8111-111111111111' + const attemptId = '33333333-3333-4333-8333-333333333333' + const traceValue = '44444444-4444-4444-8444-444444444444' + + await expect( + store.prepareCellDrainAttempt({ + attemptId, + cellId: 'cell-a', + cellIncarnation: incarnation, + traceValue, + plannedGraceMs: 120_000 + }) + ).resolves.toMatchObject({ state: 'prepared', shouldSend: false }) + expect( + await database!.query(`SELECT * FROM relay_post_drain_migration_pins`) + ).toEqual([]) + await expect( + store.prepareCellDrainRecovery({ + attemptId, + cellId: 'cell-a', + cellIncarnation: incarnation + }) + ).resolves.toMatchObject({ + shouldSend: false, + preparedAttempt: { + attemptId, + state: 'prepared', + traceValue, + plannedGraceMs: 120_000 + } + }) + + await expect( + store.beginCellDrainSend({ + attemptId, + cellId: 'cell-a', + cellIncarnation: incarnation + }) + ).resolves.toMatchObject({ + state: 'send-may-have-started', + shouldSend: true, + sendPermitExpiresAt: 30_100 + }) + await expect( + database!.query( + `SELECT migration_leases FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).resolves.toEqual([{ migration_leases: 1 }]) + await expect( + store.beginCellDrainSend({ + attemptId, + cellId: 'cell-a', + cellIncarnation: incarnation + }) + ).resolves.toMatchObject({ + state: 'send-may-have-started', + shouldSend: false + }) + await expect( + store.prepareCellDrainRecovery({ + attemptId, + cellId: 'cell-a', + cellIncarnation: incarnation + }) + ).rejects.toThrow('drain_application_receipt_missing') + expect( + await database!.query( + `SELECT drain_attempt_id, source_cell_id, source_cell_incarnation, + target_cell_id, assignment_epoch + FROM relay_post_drain_migration_pins` + ) + ).toEqual([ + { + drain_attempt_id: attemptId, + source_cell_id: 'cell-a', + source_cell_incarnation: incarnation, + target_cell_id: 'cell-b', + assignment_epoch: migration.assignmentEpoch + } + ]) + + now = migration.expiresAt + 1 + expect(await store.abortExpiredEvacuations()).toBe(0) + expect(await store.releaseExpiredActivityLeases()).toBe(1) + expect( + await database!.query( + `SELECT activity_kind, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? ORDER BY activity_kind`, + [identity.userId] + ) + ).toEqual([ + { activity_kind: 'control', cell_id: 'cell-b' }, + { activity_kind: 'migration', cell_id: 'cell-b' } + ]) + + now = 200_000 + await expect( + store.recordCellDrainApplicationReceipt({ + attemptId, + cellId: 'cell-a', + cellIncarnation: incarnation, + traceValue, + backendStatus: 200 + }) + ).resolves.toMatchObject({ + state: 'application-receipt', + applicationReceiptAt: 200_000, + retryAfter: 350_000 + }) + await expect( + store.prepareCellDrainRecovery({ attemptId, cellId: 'cell-a', cellIncarnation: incarnation }) + ).rejects.toThrow('drain_recovery_too_early') + now = 350_000 + await expect( + store.prepareCellDrainRecovery({ attemptId, cellId: 'cell-a', cellIncarnation: incarnation }) + ).resolves.toEqual({ + shouldSend: true, + retryAfter: 350_000 + }) + await expect( + store.prepareCellDrainRecovery({ attemptId, cellId: 'cell-a', cellIncarnation: incarnation }) + ).resolves.toEqual({ + shouldSend: false, + retryAfter: 350_000 + }) + }) + + it('recovers a proven drain once after the source cell is replaced', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now) + const oldIncarnation = '11111111-1111-4111-8111-111111111111' + const newIncarnation = '22222222-2222-4222-8222-222222222222' + const newerIncarnation = '55555555-5555-4555-8555-555555555555' + const attempt = { + attemptId: '33333333-3333-4333-8333-333333333333', + cellId: 'cell-a', + cellIncarnation: oldIncarnation, + traceValue: '44444444-4444-4444-8444-444444444444', + plannedGraceMs: 120_000 + } + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + await store.prepareCellDrainAttempt(attempt) + await store.beginCellDrainSend(attempt) + now = 200 + await store.recordCellDrainApplicationReceipt({ + ...attempt, + backendStatus: 200 + }) + + now = 150_200 + await heartbeat(store, CELLS[0]!, { + incarnation: newIncarnation, + startedAt: 51 + }) + await expect( + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newIncarnation + }) + ).resolves.toEqual({ shouldSend: true, retryAfter: 150_200 }) + await expect( + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newIncarnation + }) + ).resolves.toEqual({ shouldSend: false, retryAfter: 150_200 }) + + now = 150_300 + await heartbeat(store, CELLS[0]!, { + incarnation: newerIncarnation, + startedAt: 52 + }) + const concurrentRecoveries = await Promise.all([ + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newerIncarnation + }), + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newerIncarnation + }) + ]) + expect(concurrentRecoveries.map((recovery) => recovery.shouldSend).sort()).toEqual([ + false, + true + ]) + await expect( + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newerIncarnation + }) + ).resolves.toEqual({ shouldSend: false, retryAfter: 150_200 }) + await expect( + database!.query( + `SELECT cell_incarnation FROM relay_cell_drain_recovery_attempts + WHERE drain_attempt_id = ? ORDER BY cell_incarnation`, + [attempt.attemptId] + ) + ).resolves.toEqual([ + { cell_incarnation: newIncarnation }, + { cell_incarnation: newerIncarnation } + ]) + }) + + it('rejects an invalid receipt and a prepared drain from an old incarnation', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now) + const oldIncarnation = '11111111-1111-4111-8111-111111111111' + const newIncarnation = '22222222-2222-4222-8222-222222222222' + const attempt = { + attemptId: '33333333-3333-4333-8333-333333333333', + cellId: 'cell-a', + cellIncarnation: oldIncarnation, + traceValue: '44444444-4444-4444-8444-444444444444', + plannedGraceMs: 120_000 + } + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + await store.prepareCellDrainAttempt(attempt) + now = 150_200 + await heartbeat(store, CELLS[0]!, { + incarnation: newIncarnation, + startedAt: 51 + }) + await expect( + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newIncarnation + }) + ).rejects.toThrow('drain_application_receipt_missing') + + await database!.query( + `UPDATE relay_cell_drain_attempt_states + SET state = 'application-receipt', application_receipt_at = ?, + receipt_cell_incarnation = ?, retry_after = ? + WHERE attempt_id = ?`, + [200, oldIncarnation, 150_200, attempt.attemptId] + ) + await expect( + store.prepareCellDrainRecovery({ + cellId: 'cell-a', + cellIncarnation: newIncarnation + }) + ).rejects.toThrow('drain_application_receipt_missing') + }) + + it('restores an expired registered migration lease before a prepared drain send', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + const attempt = { + attemptId: '55555555-5555-4555-8555-555555555555', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: '66666666-6666-4666-8666-666666666666', + plannedGraceMs: 120_000 + } + await store.prepareCellDrainAttempt(attempt) + + now = migration.expiresAt + 1 + expect(await store.releaseExpiredActivityLeases()).toBeGreaterThan(0) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + await expect( + store.beginCellDrainSend({ + attemptId: attempt.attemptId, + cellId: attempt.cellId, + cellIncarnation: attempt.cellIncarnation + }) + ).resolves.toMatchObject({ + state: 'send-may-have-started', + shouldSend: true + }) + await expect( + database!.query( + `SELECT activity_id, activity_kind, cell_id, request_units, expires_at + FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).resolves.toContainEqual({ + activity_id: `migration:${migration.assignmentEpoch}`, + activity_kind: 'migration', + cell_id: 'cell-b', + request_units: 1, + expires_at: now + ASSIGNMENT_LIMITS.migrationLeaseMs + }) + }) + + it('refreshes registered migration leases before prepared drain recovery', async () => { + let now = 100 + const cappedCells = CELLS.map((cell) => ({ + ...cell, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + })) + const store = await setupWithHeartbeats(() => now, cappedCells) + await heartbeat(store, cappedCells[0]!) + await heartbeat(store, cappedCells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + const attempt = { + attemptId: '99999999-9999-4999-8999-999999999999', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: 'aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa', + plannedGraceMs: 120_000 + } + await store.prepareCellDrainAttempt(attempt) + + now = migration.expiresAt - 500_000 + await heartbeat(store, cappedCells[0]!) + await heartbeat(store, cappedCells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + await expect(store.prepareCellDrainRecovery(attempt)).resolves.toMatchObject({ + shouldSend: false, + preparedAttempt: { attemptId: attempt.attemptId } + }) + const refreshedExpiresAt = now + ASSIGNMENT_LIMITS.migrationLeaseMs + await expect( + database!.query( + `SELECT expires_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, migration.assignmentEpoch] + ) + ).resolves.toEqual([{ expires_at: refreshedExpiresAt }]) + await expect( + database!.query( + `SELECT state, timeout_at FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, migration.assignmentEpoch] + ) + ).resolves.toEqual([{ state: 'reserved', timeout_at: refreshedExpiresAt }]) + }) + + it('bounds repeated accounting repair before prepared drain recovery', async () => { + const delegate = await openInMemoryRelayDatabase() + const retryDatabase = new RepeatedDrainAccountingFailureDatabase(delegate) + database = retryDatabase + const store = new RelayAssignmentStore(retryDatabase, () => 100, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells(CELLS) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + const attempt = { + attemptId: '99999999-9999-4999-8999-999999999999', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: 'aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa', + plannedGraceMs: 120_000 + } + await store.prepareCellDrainAttempt(attempt) + retryDatabase.arm(2) + + await expect(store.prepareCellDrainRecovery(attempt)).resolves.toMatchObject({ + shouldSend: false, + preparedAttempt: { attemptId: attempt.attemptId } + }) + expect(retryDatabase.refreshAttempts).toBe(3) + + retryDatabase.arm(4) + await expect(store.prepareCellDrainRecovery(attempt)).rejects.toThrow( + 'migration_activity_accounting_mismatch' + ) + expect(retryDatabase.refreshAttempts).toBe(4) + }) + + it('retires a pinned migration superseded by a newer authoritative assignment', async () => { + let now = 100 + const cells = [ + ...CELLS, + { + id: 'cell-c', + url: 'https://relay-c.example.com', + capacityRequests: 2 + } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + await heartbeat(store, cells[2]!, { + incarnation: '33333333-3333-4333-8333-333333333333' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + const attempt = { + attemptId: '99999999-9999-4999-8999-999999999999', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: 'aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa', + plannedGraceMs: 120_000 + } + await store.prepareCellDrainAttempt(attempt) + await store.beginCellDrainSend(attempt) + const receipt = await store.recordCellDrainApplicationReceipt({ + ...attempt, + backendStatus: 200 + }) + + await database!.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database!.query( + `UPDATE relay_assignments SET reserved_controls = 0, reserved_splices = 0, + reserved_invites = 0, pending_installs = 0, pending_confirmations = 0, + migration_leases = 0, lease_expires_at = 0, last_activity_at = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = 0 WHERE cell_id IN (?, ?)`, + ['cell-a', 'cell-b'] + ) + now = ASSIGNMENT_LIMITS.dormantTtlMs + 1 + await heartbeat(store, cells[2]!, { + incarnation: '33333333-3333-4333-8333-333333333333' + }) + const current = await store.rebalanceDormant(identity, 'cell-c') + await store.activateControl(identity, { + cellId: 'cell-c', + assignmentEpoch: current.assignmentEpoch, + generation: 1 + }) + + expect(receipt.retryAfter).toBeLessThanOrEqual(now) + await heartbeat(store, cells[0]!) + await expect(store.prepareCellDrainRecovery(attempt)).resolves.toMatchObject({ + shouldSend: true + }) + await expect( + database!.query( + `SELECT aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, migration.assignmentEpoch] + ) + ).resolves.toEqual([{ aborted_at: now }]) + await expect( + database!.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).resolves.toEqual([ + { cell_id: 'cell-c', assignment_epoch: current.assignmentEpoch } + ]) + await expect( + database!.query( + `SELECT activity_id, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).resolves.toEqual([{ activity_id: 'control:cell-c:1', cell_id: 'cell-c' }]) + }) + + it('refuses to restore a missing migration lease without durable target registration', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + const attempt = { + attemptId: '77777777-7777-4777-8777-777777777777', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: '88888888-8888-4888-8888-888888888888', + plannedGraceMs: 120_000 + } + await store.prepareCellDrainAttempt(attempt) + + now = migration.expiresAt + 1 + expect(await store.releaseExpiredActivityLeases()).toBeGreaterThan(0) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + await expect( + store.beginCellDrainSend({ + attemptId: attempt.attemptId, + cellId: attempt.cellId, + cellIncarnation: attempt.cellIncarnation + }) + ).rejects.toThrow('migration_activity_lease_shape_mismatch') + }) + + it('moves a dead assignment only after a completed exact-incarnation fence attempt', async () => { + let now = 100 + const cappedCells = CELLS.map((cell) => ({ + ...cell, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + })) + const store = await setupWithHeartbeats(() => now, cappedCells) + await heartbeat(store, cappedCells[0]!) + await heartbeat(store, cappedCells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.setCellEnabled('cell-a', false) + now += 45_001 + await heartbeat(store, cappedCells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + + await store.attestCellFence('cell-a', '11111111-1111-4111-8111-111111111111') + expect( + await database!.query( + `SELECT cell_id FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + ['cell-a'] + ) + ).toEqual([]) + expect(await store.evacuateDeadCells()).toBe(0) + expect( + await database!.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: 'cell-a' }]) + + now += 300_001 + await heartbeat(store, cappedCells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const evidence = { + attemptId: '22222222-2222-4222-8222-222222222222', + environment: 'production' as const, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + ...FENCE_PLAN_BINDING + } + await store.prepareCellFenceAttempt(evidence) + await store.bindCellFencePlanGeneration(evidence, evidence.planObjectGeneration) + await store.startCellFenceApply( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + await store.recordCellFenceOperation( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON, + 'operation-1' + ) + await store.attestCellFenceAttempt(evidence, 'operation-1') + + expect(await store.evacuateDeadCells()).toBe(1) + expect((await store.resolve(identity))?.cellId).toBe('cell-b') + }) + + it('preserves proven non-delivery and permits one fresh full-grace attempt', async () => { + const store = await setupWithHeartbeats(() => 100, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + const first = { + attemptId: '33333333-3333-4333-8333-333333333333', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: '44444444-4444-4444-8444-444444444444', + plannedGraceMs: 120_000 + } + await store.prepareCellDrainAttempt(first) + await store.beginCellDrainSend(first) + await expect(store.proveCellDrainNotDelivered(first)).resolves.toMatchObject({ + state: 'proven-not-delivered', + provenNotDeliveredAt: 100 + }) + await expect( + store.prepareCellDrainAttempt({ + ...first, + attemptId: '55555555-5555-4555-8555-555555555555', + traceValue: '66666666-6666-4666-8666-666666666666' + }) + ).resolves.toMatchObject({ state: 'prepared' }) + }) + + it('binds fence attestation to one durable Terraform attempt', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + const evidence = { + attemptId: '22222222-2222-4222-8222-222222222222', + environment: 'production' as const, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + ...FENCE_PLAN_BINDING + } + const prepared = await store.prepareCellFenceAttempt(evidence) + expect(prepared).toMatchObject({ createdAt: 100, expiresAt: 3_600_100 }) + expect(prepared.planObjectGeneration).toBeUndefined() + await expect( + store.bindCellFencePlanGeneration( + evidence, + evidence.planObjectGeneration + ) + ).resolves.toMatchObject({ + planObjectGeneration: evidence.planObjectGeneration + }) + await expect( + store.bindCellFencePlanGeneration( + evidence, + evidence.planObjectGeneration + ) + ).resolves.toMatchObject({ + planObjectGeneration: evidence.planObjectGeneration + }) + await expect( + store.bindCellFencePlanGeneration( + evidence, + '987654321' + ) + ).rejects.toThrow('cell_fence_plan_generation_mismatch') + for (const changed of [ + { attemptId: '33333333-3333-4333-8333-333333333333' }, + { environment: 'staging' as const }, + { cellId: 'cell-b' }, + { cellIncarnation: '33333333-3333-4333-8333-333333333333' }, + { migName: 'orca-relay-other' }, + { instanceGroup: `${evidence.instanceGroup}-other` }, + { generationIdentity: `${evidence.generationIdentity}-other` }, + { fenceCommit: 'c'.repeat(40) }, + { planSha256: 'c'.repeat(64) } + ]) { + await expect( + store.startCellFenceApply( + { ...evidence, ...changed }, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + ).rejects.toThrow() + } + await store.startCellFenceApply( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + await store.recordCellFenceOperation( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON, + 'operation-1' + ) + now += 45_001 + await expect( + store.attestCellFenceAttempt(evidence, 'operation-other') + ).rejects.toThrow('cell_fence_operation_not_attested') + const attested = await store.attestCellFenceAttempt(evidence, 'operation-1') + expect(attested).toMatchObject({ + expiresAt: now + 300_000, + attempt: { ...evidence, gceOperation: 'operation-1', completedAt: now } + }) + await expect(store.attestCellFenceAttempt(evidence, 'operation-1')).resolves.toEqual( + attested + ) + now += 300_001 + await expect( + store.attestCellFenceAttempt(evidence, 'operation-1') + ).resolves.toMatchObject({ + expiresAt: now + 300_000, + attempt: { completedAt: attested.attempt.completedAt } + }) + }) + + it('aborts only a Terraform fence whose apply never started', async () => { + const store = await setupWithHeartbeats(() => 100, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + const evidence = { + attemptId: '22222222-2222-4222-8222-222222222222', + environment: 'production' as const, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + ...FENCE_PLAN_BINDING + } + await store.prepareCellFenceAttempt(evidence) + await store.bindCellFencePlanGeneration( + evidence, + evidence.planObjectGeneration + ) + await expect(store.abortCellFenceAttempt(evidence)).resolves.toMatchObject({ + abortedAt: 100 + }) + await expect( + store.startCellFenceApply( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + ).rejects.toThrow( + 'cell_fence_attempt_aborted' + ) + }) + + it('adopts a stale disabled legacy fence only when no durable attempt exists', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + now += 45_001 + + await expect( + store.adoptLegacyCellFence('cell-a', '11111111-1111-4111-8111-111111111111') + ).resolves.toBe(now + 300_000) + await expect( + database!.query( + `SELECT cell_id, cell_incarnation FROM relay_cell_fences WHERE cell_id = ?`, + ['cell-a'] + ) + ).resolves.toEqual([ + { + cell_id: 'cell-a', + cell_incarnation: '11111111-1111-4111-8111-111111111111' + } + ]) + await expect( + database!.query( + `SELECT cell_id, cell_incarnation + FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + ['cell-a'] + ) + ).resolves.toEqual([]) + await store.commitLegacyCellFenceAdoption( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + await expect( + database!.query( + `SELECT cell_id, cell_incarnation + FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + ['cell-a'] + ) + ).resolves.toEqual([ + { + cell_id: 'cell-a', + cell_incarnation: '11111111-1111-4111-8111-111111111111' + } + ]) + + now += 1 + await expect( + store.adoptLegacyCellFence('cell-a', '11111111-1111-4111-8111-111111111111') + ).resolves.toBe(now + 300_000) + await store.commitLegacyCellFenceAdoption( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + await expect( + database!.query( + `SELECT attested_at, expires_at + FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + ['cell-a'] + ) + ).resolves.toEqual([{ attested_at: now, expires_at: now + 300_000 }]) + + await heartbeat(store) + await expect( + database!.query( + `SELECT cell_id FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + ['cell-a'] + ) + ).resolves.toEqual([]) + }) + + it.each(['active', 'completed', 'aborted'] as const)( + 'rejects legacy adoption after a %s durable attempt', + async (status) => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + const evidence = cellFenceEvidence() + await store.prepareCellFenceAttempt(evidence) + if (status === 'aborted') await store.abortCellFenceAttempt(evidence) + if (status === 'completed') { + await store.bindCellFencePlanGeneration(evidence, evidence.planObjectGeneration) + await store.startCellFenceApply( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + await store.recordCellFenceOperation( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON, + 'operation-1' + ) + } + now += 45_001 + if (status === 'completed') { + await store.attestCellFenceAttempt(evidence, 'operation-1') + } + + await expect( + store.adoptLegacyCellFence('cell-a', '11111111-1111-4111-8111-111111111111') + ).rejects.toThrow('legacy_cell_fence_attempt_exists') + } + ) + + it('serializes legacy adoption against durable attempt preparation', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + now += 45_001 + + const results = await Promise.allSettled([ + store.adoptLegacyCellFence('cell-a', '11111111-1111-4111-8111-111111111111'), + store.prepareCellFenceAttempt(cellFenceEvidence()) + ]) + expect(results.filter(({ status }) => status === 'fulfilled')).toHaveLength(1) + const attempts = await database!.query( + `SELECT COUNT(*) AS count FROM relay_cell_fence_attempts WHERE cell_id = ?`, + ['cell-a'] + ) + const fences = await database!.query( + `SELECT COUNT(*) AS count FROM relay_cell_fences WHERE cell_id = ?`, + ['cell-a'] + ) + const adoptions = await database!.query( + `SELECT COUNT(*) AS count FROM relay_cell_legacy_fence_adoptions + WHERE cell_id = ?`, + ['cell-a'] + ) + expect(Number(attempts[0]!.count) + Number(fences[0]!.count)).toBe(1) + expect(Number(adoptions[0]!.count)).toBe(0) + }) + + it('rejects an expired durable Terraform fence attempt', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + const evidence = { + attemptId: '22222222-2222-4222-8222-222222222222', + environment: 'production' as const, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + ...FENCE_PLAN_BINDING + } + await store.prepareCellFenceAttempt(evidence) + await store.bindCellFencePlanGeneration( + evidence, + evidence.planObjectGeneration + ) + now += 3_600_001 + await expect( + store.startCellFenceApply( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + ).rejects.toThrow( + 'cell_fence_attempt_expired' + ) + }) + + it('keeps a started Terraform fence attempt recoverable after its preparation TTL', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + await store.setCellEnabled('cell-a', false) + const evidence = { + attemptId: '22222222-2222-4222-8222-222222222222', + environment: 'production' as const, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + ...FENCE_PLAN_BINDING + } + await store.prepareCellFenceAttempt(evidence) + await store.bindCellFencePlanGeneration( + evidence, + evidence.planObjectGeneration + ) + await store.startCellFenceApply( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON + ) + now += 3_600_001 + await expect( + store.recordCellFenceOperation( + evidence, + FENCE_INVOCATION_ID, + FENCE_INVOCATION_REASON, + 'operation-1' + ) + ).resolves.toMatchObject({ + attempt: { gceOperation: 'operation-1' }, + invocation: { gceOperation: 'operation-1' } + }) + await expect( + store.prepareCellFenceAttempt({ + ...evidence, + attemptId: '44444444-4444-4444-8444-444444444444', + planObjectName: + 'terraform/state/relay-fence-plans/production/44444444-4444-4444-8444-444444444444.tfplan', + requestReason: + 'orca-relay-fence/44444444-4444-4444-8444-444444444444' + }) + ).rejects.toThrow('cell_fence_attempt_evidence_mismatch') + }) + + it('reports aggregate deployment state and heartbeat freshness without identities', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [CELLS[0]!]) + await heartbeat(store) + const identity = { userId: 'private-user', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + expect(await store.cellDeploymentStatus('cell-a')).toMatchObject({ + activityLeases: 1, + activityRequestUnits: 1, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 0 + }) + await store.activateControl(identity, { + cellId: 'cell-a', + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + + expect(await store.cellDeploymentStatus('cell-a')).toEqual({ + cellId: 'cell-a', + cellUrl: 'https://relay-a.example.com', + region: 'us-central1', + enabled: true, + admissionState: 'general', + capacityRequests: 2, + reservedRequests: 1, + assignments: 1, + activityLeases: 1, + activityRequestUnits: 1, + restartBlockingActivityLeases: 1, + restartBlockingActivityRequestUnits: 1, + restartBlockingReservedRequests: 1, + outgoingMigrations: 0, + incomingMigrations: 0, + connectionCapacity: null, + runtime: { + cellUrl: 'https://relay-a.example.com', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + lastHeartbeatAt: 100, + heartbeatFresh: true, + regionalRehomeProtocol: 0 + } + }) + now += 45_001 + expect((await store.cellDeploymentStatus('cell-a')).runtime?.heartbeatFresh).toBe(false) + await expect(store.cellDeploymentStatus('missing')).rejects.toThrow('cell_not_found') + }) + + it('keeps concurrent sticky assignment grants restart-safe', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'private-user', relayHostId: 'host000000000001' } + + await Promise.all(Array.from({ length: 20 }, async () => await store.assign(identity))) + + expect(await store.cellDeploymentStatus('cell-a')).toMatchObject({ + activityLeases: 1, + activityRequestUnits: 1, + reservedRequests: 1, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 0 + }) + }) + + it('classifies activated and non-control activity as restart-blocking', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'private-user', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + for (const kind of ['splice', 'invite', 'install', 'confirmation', 'migration'] as const) { + await store.acquireActivity(identity, { + activityId: `${kind}:test`, + kind, + cellId: assignment.cellId + }) + } + + expect(await store.cellDeploymentStatus('cell-a')).toMatchObject({ + activityLeases: 6, + activityRequestUnits: 7, + reservedRequests: 7, + restartBlockingActivityLeases: 6, + restartBlockingActivityRequestUnits: 7, + restartBlockingReservedRequests: 7 + }) + }) + + it('fails closed for malformed pending controls and unexplained reservations', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 } + ]) + await store.assign({ userId: 'private-user', relayHostId: 'host000000000001' }) + await database!.query( + `UPDATE relay_assignment_activity_leases SET request_units = 2 + WHERE cell_id = ?`, + ['cell-a'] + ) + expect(await store.cellDeploymentStatus('cell-a')).toMatchObject({ + restartBlockingActivityLeases: 1, + restartBlockingActivityRequestUnits: 2, + restartBlockingReservedRequests: 1 + }) + + await database!.query( + `UPDATE relay_assignment_activity_leases SET request_units = 1 WHERE cell_id = ?`, + ['cell-a'] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = 0 WHERE cell_id = ?`, + ['cell-a'] + ) + expect(await store.cellDeploymentStatus('cell-a')).toMatchObject({ + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: -1 + }) + + await database!.query( + `UPDATE relay_cells SET reserved_requests = 2 WHERE cell_id = ?`, + ['cell-a'] + ) + expect(await store.cellDeploymentStatus('cell-a')).toMatchObject({ + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 1 + }) + }) + + it('keeps a migration blocking when its target also has a pending control', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'private-user', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.startEvacuation(identity, 'cell-b') + + expect(await store.cellDeploymentStatus('cell-b')).toMatchObject({ + activityLeases: 2, + restartBlockingActivityLeases: 1, + restartBlockingActivityRequestUnits: 1, + restartBlockingReservedRequests: 1, + incomingMigrations: 1 + }) + }) + + it('starts declared candidates disabled without overwriting later operator state', async () => { + const candidate: RelayCellConfig = { + id: 'candidate', + url: 'https://candidate.example.com', + capacityRequests: 20, + initiallyEnabled: false + } + const store = await setup(() => 100, [candidate]) + expect((await store.cellDeploymentStatus('candidate')).enabled).toBe(false) + await store.setCellEnabled('candidate', true) + await store.reconcileCells([candidate]) + expect((await store.cellDeploymentStatus('candidate')).enabled).toBe(true) + }) + + it('moves a stranded host off an existing-only cell that stopped serving it', async () => { + // Why: C3's decommission created an existing-only cell that rejects + // attaches; the #194 pin then loops returning hosts forever while each + // grant refreshes their own activity (issue #225). C3 has no + // connection-limits row and expired leases are cleaned within a + // maintenance cycle, so the only durable evidence is the assignment row: + // a recent grant with no live real activity behind it. + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.setCellEnabled(first.cellId, false) + + // The pin holds while the last grant is young enough to still attach. + now += 10_000 + const pinned = await store.assign(identity) + expect(pinned.cellId).toBe(first.cellId) + + // The grant had ample time to attach and produced nothing live. + now += 61_000 + const moved = await store.assign(identity) + expect(moved.cellId).not.toBe(first.cellId) + expect(moved.assignmentEpoch).toBe(first.assignmentEpoch + 1) + + // The move is durable: no bounce back to the closed cell. + now += 1_000 + const settled = await store.assign(identity) + expect(settled.cellId).toBe(moved.cellId) + expect(settled.assignmentEpoch).toBe(moved.assignmentEpoch) + }) + + it('keeps a serving existing-only cell pinned for hosts with live activity', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.setCellEnabled(first.cellId, false) + // A live claimed control proves the cell still serves this host. + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, 'control', ?, 1, ?, ?)`, + [identity.userId, identity.relayHostId, 'control:live-1', first.cellId, now + 600_000, now] + ) + now += 61_000 + const pinned = await store.assign(identity) + expect(pinned.cellId).toBe(first.cellId) + }) + + it('keeps migration-only cells pinned inside the stranded window', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.setCellAdmissionState(first.cellId, 'migration-only') + now += 61_000 + const pinned = await store.assign(identity) + expect(pinned.cellId).toBe(first.cellId) + }) + + it('leaves quiet existing-only assignments to the normal dormancy rule', async () => { + // Why: outside the 15-minute stranded window there is no active retry + // loop to break; ordinary returns stay pinned until 24h dormancy. + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.setCellEnabled(first.cellId, false) + now += 20 * 60_000 + const pinned = await store.assign(identity) + expect(pinned.cellId).toBe(first.cellId) + }) + + it('reassigns an active assignment from an uncapped stale cell without a fence', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now) + await heartbeat(store, CELLS[0]!) + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.changeActivity(identity, 'invite', 1) + await database!.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, NULL, NULL, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 'cell-b', + 'cell-a', + -1, + 0, + 0, + 1, + 90_100, + now, + now + ] + ) + await database!.query( + `INSERT INTO relay_post_drain_migration_pins + (user_id, relay_host_id, assignment_epoch, drain_attempt_id, + source_cell_id, source_cell_incarnation, target_cell_id, + target_cell_incarnation, source_request_units, target_reserved_units, + pinned_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 0, + '33333333-3333-4333-8333-333333333333', + 'cell-b', + '22222222-2222-4222-8222-222222222222', + 'cell-a', + '11111111-1111-4111-8111-111111111111', + 0, + 1, + now + ] + ) + + now += 45_001 + await heartbeat(store, CELLS[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222' + }) + expect(await store.evacuateDeadCells()).toBe(1) + const replacement = await store.resolve(identity) + expect(replacement).toMatchObject({ + cellId: 'cell-b', + assignmentEpoch: first.assignmentEpoch + 1 + }) + expect( + await database!.query( + `SELECT activity_kind, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ activity_kind: 'control', cell_id: 'cell-b' }]) + }) + + it('does not let an unfenced capped cell starve eligible legacy recovery', async () => { + let now = 100 + const cells: RelayCellConfig[] = [ + { + id: 'cell-a', + url: 'https://relay-a.example.com', + capacityRequests: 10, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + for (const cell of cells) await heartbeat(store, cell) + await store.setCellEnabled('cell-b', false) + await store.setCellEnabled('cell-c', false) + const cappedIdentity = { userId: 'a-capped', relayHostId: 'host000000000001' } + expect(await store.assign(cappedIdentity)).toMatchObject({ cellId: 'cell-a' }) + + await store.setCellEnabled('cell-a', false) + await store.setCellEnabled('cell-b', true) + const legacyIdentity = { userId: 'z-legacy', relayHostId: 'host000000000002' } + expect(await store.assign(legacyIdentity)).toMatchObject({ cellId: 'cell-b' }) + await store.setCellEnabled('cell-c', true) + + now += 45_001 + await heartbeat(store, cells[2]!) + expect(await store.evacuateDeadCells(1)).toBe(1) + expect(await store.resolve(legacyIdentity)).toMatchObject({ cellId: 'cell-c' }) + expect( + await database!.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ?`, + [cappedIdentity.userId] + ) + ).toEqual([{ cell_id: 'cell-a' }]) + }) + + it('refuses evacuation into a cell without a fresh ready heartbeat', async () => { + const store = await setupWithHeartbeats(() => 100) + await heartbeat(store, CELLS[0]!) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + + await expect(store.startEvacuation(identity, 'cell-b')).rejects.toThrow( + 'target_cell_unavailable' + ) + }) + + it('keeps an active assignment sticky without increasing its epoch or reservation', async () => { + const store = await setup(() => 100) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const second = await store.assign(identity) + + expect(second).toEqual(first) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 0 }) + }) + + it('selects the least-loaded cell with a stable cell-id tie break', async () => { + const store = await setup(() => 100) + const first = await store.assign({ userId: 'user-a', relayHostId: 'host000000000001' }) + const second = await store.assign({ userId: 'user-b', relayHostId: 'host000000000002' }) + + expect([first.cellId, second.cellId]).toEqual(['cell-a', 'cell-b']) + }) + + it('admits exactly through the configured capacity boundary', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 2 } + ]) + await store.assign({ userId: 'user-a', relayHostId: 'host000000000001' }) + await store.assign({ userId: 'user-b', relayHostId: 'host000000000002' }) + + await expect( + store.assign({ userId: 'user-c', relayHostId: 'host000000000003' }) + ).rejects.toThrow('relay_capacity_exhausted') + expect(await cellReservations(database!)).toEqual({ 'cell-a': 2 }) + }) + + it('does not oversubscribe when assignments race', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 2 } + ]) + const results = await Promise.allSettled( + Array.from({ length: 8 }, (_, index) => + store.assign({ userId: `user-${index}`, relayHostId: `host${String(index).padStart(12, '0')}` }) + ) + ) + + expect(results.filter(({ status }) => status === 'fulfilled')).toHaveLength(2) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 2 }) + }) + + it('reassigns only after all activity expires and the dormant TTL elapses', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.changeActivity(identity, 'invite', 1) + await store.changeActivity(identity, 'control', -1) + + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + await store.releaseExpiredActivityLeases() + await store.releaseExpiredActivity() + await database!.query(`UPDATE relay_cells SET observed_requests = 2 WHERE cell_id = ?`, [ + first.cellId + ]) + now += ASSIGNMENT_LIMITS.dormantTtlMs + const reassigned = await store.assign(identity) + + expect(reassigned.cellId).not.toBe(first.cellId) + expect(reassigned.assignmentEpoch).toBe(first.assignmentEpoch + 1) + }) + + it('will not normally move an assignment while any durable activity remains', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.changeActivity(identity, 'migration', 1) + await store.changeActivity(identity, 'control', -1) + await database!.query(`UPDATE relay_cells SET observed_requests = 2 WHERE cell_id = ?`, [ + first.cellId + ]) + now += ASSIGNMENT_LIMITS.dormantTtlMs + 1 + + expect(await store.assign(identity)).toMatchObject({ + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch + }) + }) + + it('releases every expired activity reservation without going negative', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + for (const kind of ['splice', 'invite', 'install', 'confirmation', 'migration'] as const) { + await store.changeActivity(identity, kind, 1) + } + expect(await cellReservations(database!)).toEqual({ 'cell-a': 7 }) + + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + expect(await store.releaseExpiredActivityLeases()).toBe(1) + expect(await store.releaseExpiredActivity()).toBe(1) + expect(await store.releaseExpiredActivity()).toBe(0) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0 }) + }) + + it('isolates assignments by host and verifies both cell and epoch', async () => { + const store = await setup(() => 100) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + + await expect( + store.verifyCellAssignment({ ...identity, cellId: assignment.cellId, assignmentEpoch: 1 }) + ).resolves.toBe(true) + await expect( + store.verifyCellAssignment({ ...identity, cellId: 'cell-b', assignmentEpoch: 1 }) + ).resolves.toBe(false) + await expect( + store.verifyCellAssignment({ ...identity, cellId: assignment.cellId, assignmentEpoch: 2 }) + ).resolves.toBe(false) + await expect( + store.resolve({ userId: 'other', relayHostId: identity.relayHostId }) + ).resolves.toBeNull() + }) + + it('converts the director reservation into an idempotent active-control lease', async () => { + const store = await setup(() => 100) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + + const activityId = await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + + expect(activityId).toBe(`control:${assignment.cellId}:1`) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 0 }) + const leases = await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases` + ) + expect(leases).toEqual([{ activity_id: activityId }]) + }) + + it('transactionally supersedes older controls on only the same cell', async () => { + const cells = [ + { + id: 'cell-a', + url: 'https://relay-a.example.com', + capacityRequests: 10 + }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const store = await setup(() => 100, cells) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.setCellEnabled('cell-b', false) + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-b', true) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, 'control', 'cell-b', 1, 90100, 100), + (?, ?, ?, 'control', 'cell-b', 1, 90100, 100)`, + [ + identity.userId, + identity.relayHostId, + 'control:cell-b:2', + identity.userId, + identity.relayHostId, + 'control:cell-b:3' + ] + ) + const latest = await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 4 + }) + + expect( + await database!.query( + `SELECT activity_id, cell_id FROM relay_assignment_activity_leases + WHERE activity_kind = 'control' ORDER BY cell_id` + ) + ).toEqual([ + { activity_id: 'control:cell-a:1', cell_id: 'cell-a' }, + { activity_id: latest, cell_id: 'cell-b' } + ]) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 2 }) + expect( + await database!.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 2 }]) + }) + + it('re-reserves a sticky grant for a host holding no control lease', async () => { + const store = await setup(() => 100) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + const control = await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.releaseActivity(identity, control) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 0 }) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch + }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 0 }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases` + ) + ).toEqual([{ activity_id: 'control-pending:1' }]) + expect( + await database!.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 1 }]) + }) + + it('re-pins a host that took control after the sticky lane read it dormant', async () => { + const now = 100 + const delegate = await openInMemoryRelayDatabase() + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const seeded = new SeedBetweenAssignmentLanesDatabase(delegate, async () => { + await delegate.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, 'control:cell-a:1', 'control', 'cell-a', 1, ?, ?), + (?, ?, 'control-pending:1', 'control', 'cell-a', 1, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now, + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await delegate.query( + `UPDATE relay_assignments SET reserved_controls = 2 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await delegate.query( + `UPDATE relay_cells SET reserved_requests = 2 WHERE cell_id = 'cell-a'` + ) + }) + database = seeded + const store = new RelayAssignmentStore(seeded, () => now) + await store.reconcileCells(CELLS) + await delegate.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, 'cell-a', 1, ?, ?, 0, 0, 0, 0, 0, 0)`, + [identity.userId, identity.relayHostId, now, now - ASSIGNMENT_LIMITS.dormantTtlMs] + ) + seeded.arm() + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-a', + assignmentEpoch: 1 + }) + expect( + await delegate.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 2 }]) + expect(await cellReservations(delegate)).toEqual({ 'cell-a': 2, 'cell-b': 0 }) + expect( + await delegate.query( + `SELECT lease_expires_at, last_activity_at FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { lease_expires_at: now + ASSIGNMENT_LIMITS.activityLeaseMs, last_activity_at: now } + ]) + }) + + it('reserves control on the pinned cell when the only lease sits on another', async () => { + const now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await database!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, 'cell-b', 2, ?, ?, 1, 0, 0, 0, 0, 0)`, + [ + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, 'control:cell-a:1', 'control', 'cell-a', 1, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = 1 WHERE cell_id = 'cell-a'` + ) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-b', + assignmentEpoch: 2 + }) + expect( + await database!.query( + `SELECT activity_id, cell_id FROM relay_assignment_activity_leases + ORDER BY activity_id ASC` + ) + ).toEqual([ + { activity_id: 'control-pending:2', cell_id: 'cell-b' }, + { activity_id: 'control:cell-a:1', cell_id: 'cell-a' } + ]) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 1 }) + expect( + await database!.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 2 }]) + }) + + it('mints a pending control when the only lease is an older epoch pending', async () => { + const now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await database!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, 'cell-a', 3, ?, ?, 1, 0, 0, 0, 0, 0)`, + [ + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, 'control-pending:1', 'control', 'cell-a', 1, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = 1 WHERE cell_id = 'cell-a'` + ) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-a', + assignmentEpoch: 3 + }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + ORDER BY activity_id ASC` + ) + ).toEqual([{ activity_id: 'control-pending:1' }, { activity_id: 'control-pending:3' }]) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 2, 'cell-b': 0 }) + expect( + await database!.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 2 }]) + }) + + it('re-reserves a re-pinned grant for a host holding no control lease', async () => { + const now = 100 + const delegate = await openInMemoryRelayDatabase() + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const seeded = new SeedBetweenAssignmentLanesDatabase(delegate, async () => { + await delegate.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, 'splice:connection-1', 'splice', 'cell-a', 1, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await delegate.query( + `UPDATE relay_assignments SET reserved_splices = 1 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await delegate.query( + `UPDATE relay_cells SET reserved_requests = 1 WHERE cell_id = 'cell-a'` + ) + }) + database = seeded + const store = new RelayAssignmentStore(seeded, () => now) + await store.reconcileCells(CELLS) + await delegate.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, 'cell-a', 1, ?, ?, 0, 0, 0, 0, 0, 0)`, + [identity.userId, identity.relayHostId, now, now - ASSIGNMENT_LIMITS.dormantTtlMs] + ) + seeded.arm() + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: 'cell-a', + assignmentEpoch: 1 + }) + expect( + await delegate.query( + `SELECT reserved_controls, reserved_splices FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 1, reserved_splices: 1 }]) + expect( + await delegate.query( + `SELECT activity_id FROM relay_assignment_activity_leases + ORDER BY activity_id ASC` + ) + ).toEqual([{ activity_id: 'control-pending:1' }, { activity_id: 'splice:connection-1' }]) + expect(await cellReservations(delegate)).toEqual({ 'cell-a': 2, 'cell-b': 0 }) + }) + + it('extends the lease of a control-holding host on a sticky grant', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + + now = 40_000 + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: assignment.cellId + }) + expect( + await database!.query( + `SELECT lease_expires_at, last_activity_at FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { lease_expires_at: now + ASSIGNMENT_LIMITS.activityLeaseMs, last_activity_at: now } + ]) + }) + + // The old absolute write shortened a 15-minute migration lease to the 90s + // grant lease; only the touch's monotonic CASE keeps the longer one now. + it('never shortens a migration lease to the grant lease', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + const before = await database!.query( + `SELECT lease_expires_at FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + expect(before).toEqual([{ lease_expires_at: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs }]) + + now = 1000 + await expect(store.assign(identity)).resolves.toMatchObject({ cellId: 'cell-b' }) + expect( + await database!.query( + `SELECT lease_expires_at, last_activity_at FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { lease_expires_at: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs, last_activity_at: now } + ]) + }) + + it('moves an origin-scoped activity reservation without double counting', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.acquireActivity(identity, { + activityId: 'splice:connection-1', + kind: 'splice', + cellId: 'cell-a' + }) + await store.startEvacuation(identity, 'cell-b') + await store.acquireActivity(identity, { + activityId: 'splice:connection-1', + kind: 'splice', + cellId: 'cell-b' + }) + + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 6 }) + await expect(store.releaseActivity(identity, 'splice:connection-1')).resolves.toBe(true) + await expect(store.releaseActivity(identity, 'splice:connection-1')).resolves.toBe(false) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 4 }) + }) + + it('rejects a durable activity before it can exceed cell capacity', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 2 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + + await expect( + store.acquireActivity(identity, { + activityId: 'splice:connection-1', + kind: 'splice', + cellId: 'cell-a' + }) + ).rejects.toThrow('relay_capacity_exhausted') + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1 }) + }) + + it('moves active assignments target-first and completes only after source drain', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + + const migration = await store.startEvacuation(identity, 'cell-b') + expect(await store.startEvacuation(identity, 'cell-b')).toEqual(migration) + expect(migration).toMatchObject({ + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + previousEpoch: 1, + assignmentEpoch: 2 + }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 2 }) + + const targetControl = await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: 2, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { cellId: 'cell-b', assignmentEpoch: 2 }) + await expect(store.completeEvacuation(identity, 2)).rejects.toThrow( + 'migration_source_still_active' + ) + await store.releaseActivity(identity, sourceControl) + await store.completeEvacuation(identity, 2) + + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 1 }) + expect(await store.resolve(identity)).toMatchObject({ cellId: 'cell-b', assignmentEpoch: 2 }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases ORDER BY activity_id` + ) + ).toEqual([{ activity_id: targetControl }]) + await expect( + store.acquireActivity(identity, { + activityId: 'splice:late-source', + kind: 'splice', + cellId: 'cell-a' + }) + ).rejects.toThrow('activity_cell_not_authoritative') + }) + + it('automatically completes only after the source runtime is freshly quiescent', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await store.setCellEnabled('cell-a', false) + await heartbeat(store, cells[0]!, { observedRequests: 1 }) + + expect(await store.completeReadyEvacuations()).toBe(0) + await heartbeat(store, cells[0]!, { observedRequests: 0 }) + now = 150 + await heartbeat(store, cells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 125 + }) + expect(await store.completeReadyEvacuations()).toBe(0) + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 2 + }) + expect(await store.completeReadyEvacuations()).toBe(1) + }) + + it('completes a fenced migration only after the source heartbeat is stale', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!, { observedRequests: 1 }) + await heartbeat(store, cells[1]!) + await store.setCellEnabled('cell-b', false) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-b', true) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await store.setCellEnabled('cell-a', false) + + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toMatchObject({ + inProgress: 1, + blocked: 1 + }) + now += 45_001 + await heartbeat(store, cells[1]!) + await store.attestCellFence( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toMatchObject({ + inProgress: 0, + completed: 1 + }) + }) + + it('blocks aggregate fenced completion on assignment accounting mismatch', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!) + await store.setCellEnabled('cell-b', false) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-b', true) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await store.setCellEnabled('cell-a', false) + await database!.query( + `UPDATE relay_assignments SET reserved_splices = reserved_splices + 1 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + now += 45_001 + await heartbeat(store, cells[1]!) + await store.attestCellFence( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toMatchObject({ + inProgress: 1, + blocked: 1 + }) + }) + + it('blocks aggregate fenced completion on an unexpected third-cell lease', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + for (const cell of cells) await heartbeat(store, cell) + await store.setCellEnabled('cell-b', false) + await store.setCellEnabled('cell-c', false) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-b', true) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await store.setCellEnabled('cell-a', false) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 'splice:unexpected-cell', + 'splice', + 'cell-c', + 2, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now + ] + ) + await database!.query( + `UPDATE relay_assignments SET reserved_splices = reserved_splices + 1 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = reserved_requests + 2 WHERE cell_id = ?`, + ['cell-c'] + ) + now += 45_001 + await heartbeat(store, cells[1]!) + await store.attestCellFence( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toMatchObject({ + inProgress: 1, + blocked: 1 + }) + }) + + it('rolls back an unregistered expired evacuation with a strictly newer epoch', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + expect(await store.resolve(identity)).toMatchObject({ cellId: 'cell-a', assignmentEpoch: 3 }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 0 }) + }) + + // Why: completion needs an enabled target and expiry rollback needs an unregistered one, so a + // target disabled after it registered satisfies neither and the migration never leaves the table. + async function wedgeRegisteredMigration( + now: () => number, + cells: RelayCellConfig[], + disableTarget = true, + disableSource = true, + releaseSourceControl = true + ): Promise<{ + store: RelayAssignmentStore + identity: { userId: string; relayHostId: string } + }> { + const store = await setupWithHeartbeats(now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, 'cell-b') + const targetControl = await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + if (releaseSourceControl) await store.releaseActivity(identity, sourceControl) + if (disableSource) await store.setCellEnabled('cell-a', false) + // The desktop leaves the half-migrated target, then an operator disables that cell. + await store.releaseActivity(identity, targetControl) + if (disableTarget) await store.setCellEnabled('cell-b', false) + return { store, identity } + } + + async function insertReservedControlConnection( + identity: { userId: string; relayHostId: string }, + cellId: string, + assignmentEpoch: number, + now: number + ): Promise { + await database!.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, assignment_epoch, + cell_id, state, created_at, timeout_at, updated_at) + VALUES ('abandoned-reservation', 'abandoned-key', ?, ?, ?, ?, 'reserved', ?, ?, ?)`, + [identity.userId, identity.relayHostId, assignmentEpoch, cellId, now, now + 60_000, now] + ) + } + + it('reaps an expired migration whose registered target cell was disabled', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store, identity } = await wedgeRegisteredMigration(() => now, cells) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + // A disabled target can never satisfy completion, so waiting cannot help. + expect(await store.completeReadyEvacuations()).toBe(0) + expect(await store.abortExpiredEvacuations()).toBe(1) + // Rollback lands on the disabled source, so the host resolves to nothing and is placed afresh + // by evacuateDeadCells; the point is that the row is retired rather than reaped every tick. + expect(await store.resolve(identity)).toBeNull() + expect(await store.abortExpiredEvacuations()).toBe(0) + }) + + it('leaves a freshly disabled target for supersede-target to recover', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells) + + // Expired, but a disabled target is what supersede-target itself creates, so the operator + // window must stay open; aborting here would fail their run with migration_already_superseded. + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + expect(await store.abortExpiredEvacuations()).toBe(0) + + now += STRANDED_MIGRATION_ABANDON_MS + expect(await store.abortExpiredEvacuations()).toBe(1) + }) + + it('rolls an abandoned disabled target back while its source remains active', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store, identity } = await wedgeRegisteredMigration( + () => now, + cells, + true, + false, + false + ) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + expect( + await database!.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: 'cell-a', assignment_epoch: 3 }]) + expect( + await database!.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ completed_at: null, aborted_at: now }]) + }) + + it('starts the abandon window when an old migration target is disabled', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells, false, false) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + + await store.setCellEnabled('cell-b', false) + expect(await store.abortExpiredEvacuations()).toBe(0) + + now += STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + }) + + it('reaps an expired migration from a retired source when its target control is gone', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store, identity } = await wedgeRegisteredMigration(() => now, cells, false) + await applyGenerationZeroSelector(store, { + attemptId: 'retired_source_migration_target', + membership: { + existingOnly: ['cell-a'], + migrationOnly: ['cell-b'], + general: [] + } + }) + await insertReservedControlConnection(identity, 'cell-b', 2, now) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + expect( + await database!.query( + `SELECT assignment.cell_id, assignment.assignment_epoch, + migration.completed_at, migration.aborted_at + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + WHERE assignment.user_id = ? AND assignment.relay_host_id = ? + AND migration.assignment_epoch = ?`, + [identity.userId, identity.relayHostId, 2] + ) + ).toEqual([ + { + cell_id: 'cell-b', + assignment_epoch: 2, + completed_at: now, + aborted_at: null + } + ]) + expect( + await database!.query( + `SELECT reserved_controls, migration_leases FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 0, migration_leases: 0 }]) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([]) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 0 }) + expect( + await database!.query( + `SELECT state FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ state: 'released' }]) + }) + + it('reaps a retired-side migration with only a pending target control', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await applyGenerationZeroSelector(store, { + attemptId: 'retired_source_pending_target', + membership: { + existingOnly: ['cell-a'], + migrationOnly: ['cell-b'], + general: [] + } + }) + await insertReservedControlConnection(identity, 'cell-b', migration.assignmentEpoch, now) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + await database!.query( + `UPDATE relay_assignment_activity_leases SET expires_at = ? + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [now, identity.userId, identity.relayHostId, `control-pending:${migration.assignmentEpoch}`] + ) + expect(await store.abortExpiredEvacuations()).toBe(1) + expect( + await database!.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: 'cell-b', assignment_epoch: migration.assignmentEpoch }]) + expect( + await database!.query( + `SELECT reserved_controls, migration_leases FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ reserved_controls: 0, migration_leases: 0 }]) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([]) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 0 }) + expect( + await database!.query( + `SELECT state FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ state: 'released' }]) + }) + + it('preserves a fresh target grant created after abandoned migration selection', async () => { + let now = 100 + const cells: RelayCellConfig[] = [ + { + id: 'cell-a', + url: 'https://relay-a.example.com', + capacityRequests: 10, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }, + { + id: 'cell-b', + url: 'https://relay-b.example.com', + capacityRequests: 10, + connectionHardCap: 600, + connectionUnobservedBound: 50 + } + ] + const { store, identity } = await wedgeRegisteredMigration(() => now, cells, false) + await applyGenerationZeroSelector(store, { + attemptId: 'retired_source_fresh_target_grant', + membership: { + existingOnly: ['cell-a'], + migrationOnly: ['cell-b'], + general: [] + } + }) + + now += 45_001 + await heartbeat(store, cells[1]!) + await store.adoptLegacyCellFence( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + await store.commitLegacyCellFenceAdoption( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + await heartbeat(store, cells[1]!) + const query = database!.query.bind(database) + let grant: Awaited> | undefined + vi.spyOn(database!, 'query').mockImplementationOnce(async (sql, params) => { + const rows = await query(sql, params) + grant = await store.assign(identity) + return rows + }) + + expect(await store.abortExpiredEvacuations()).toBe(0) + expect(grant).toMatchObject({ cellId: 'cell-b', assignmentEpoch: 2 }) + expect( + await database!.query( + `SELECT activity_id, expires_at FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, 'control-pending:2'] + ) + ).toEqual([{ activity_id: 'control-pending:2', expires_at: grant!.leaseExpiresAt }]) + expect( + await database!.query( + `SELECT state FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND cell_id = ? AND state = 'reserved'`, + [identity.userId, identity.relayHostId, 2, 'cell-b'] + ) + ).toEqual([{ state: 'reserved' }]) + + await store.activateControl(identity, { + cellId: grant!.cellId, + assignmentEpoch: grant!.assignmentEpoch, + generation: 2 + }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control'`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ activity_id: 'control:cell-b:2' }]) + expect( + await database!.query( + `SELECT completed_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, 2] + ) + ).toEqual([{ completed_at: null }]) + }) + + it('keeps an abandoned target migration open while its source is still active', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store, identity } = await wedgeRegisteredMigration( + () => now, + cells, + false, + true, + false + ) + await applyGenerationZeroSelector(store, { + attemptId: 'retired_source_active', + membership: { + existingOnly: ['cell-a'], + migrationOnly: ['cell-b'], + general: [] + } + }) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.abortExpiredEvacuations()).toBe(0) + expect( + await database!.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ completed_at: null, aborted_at: null }]) + }) + + it('starts the abandon window when an old migration source is retired', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells, false, false) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + + await store.setCellEnabled('cell-a', false) + expect(await store.abortExpiredEvacuations()).toBe(0) + + now += STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + }) + + it('reaps after an old source migration lease is refreshed and expires', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells, false) + await applyGenerationZeroSelector(store, { + attemptId: 'retired_source_refreshed_lease', + membership: { + existingOnly: ['cell-a'], + migrationOnly: ['cell-b'], + general: [] + } + }) + now += STRANDED_MIGRATION_ABANDON_MS + ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + await database!.query( + `UPDATE relay_assignment_migrations SET expires_at = ?`, + [now - 1] + ) + + expect(await store.abortExpiredEvacuations()).toBe(1) + }) + + it('keeps a freshly retired target open despite an old retired source', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells, false) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + + await store.setCellEnabled('cell-b', false) + expect(await store.abortExpiredEvacuations()).toBe(0) + + now += STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + }) + + it('fails supersession closed when reaping wins after selection', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + const query = database!.query.bind(database) + vi.spyOn(database!, 'query').mockImplementationOnce(async (sql, params) => { + const rows = await query(sql, params) + expect(await store.abortExpiredEvacuations()).toBe(1) + return rows + }) + + await expect( + store.supersedeRegisteredCellEvacuations('cell-a', 'cell-b', 'cell-c', 100) + ).rejects.toThrow('migration_already_superseded') + }) + + it('fails supersession closed when reaping wins after preparation', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + await store.attestCellFence('cell-b', '11111111-1111-4111-8111-111111111111') + const supersede = store.supersedeRegisteredEvacuation.bind(store) + vi.spyOn(store, 'supersedeRegisteredEvacuation').mockImplementationOnce(async (...args) => { + expect(await store.abortExpiredEvacuations()).toBe(1) + return await supersede(...args) + }) + + await expect( + store.supersedeRegisteredCellEvacuations('cell-a', 'cell-b', 'cell-c', 100) + ).rejects.toThrow('migration_already_superseded') + }) + + it('fails supersession closed when reaping wins before an accounting retry', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const { store } = await wedgeRegisteredMigration(() => now, cells) + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[2]!) + await store.attestCellFence('cell-b', '11111111-1111-4111-8111-111111111111') + await database!.query(`UPDATE relay_cells SET reserved_requests = 1 WHERE cell_id = ?`, [ + 'cell-c' + ]) + const supersede = store.supersedeRegisteredEvacuation.bind(store) + let attempts = 0 + vi.spyOn(store, 'supersedeRegisteredEvacuation').mockImplementation(async (...args) => { + attempts++ + if (attempts === 2) expect(await store.abortExpiredEvacuations()).toBe(1) + return await supersede(...args) + }) + + await expect( + store.supersedeRegisteredCellEvacuations('cell-a', 'cell-b', 'cell-c', 100) + ).rejects.toThrow('migration_already_superseded') + expect(attempts).toBe(2) + }) + + it('reaps a pinned expired migration once its target cell is disabled', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ] + const { store, identity } = await wedgeRegisteredMigration(() => now, cells) + // Drains pin their migrations and nothing ever deletes a pin, so the pin outlives the drain. + await database!.query( + `INSERT INTO relay_post_drain_migration_pins + (user_id, relay_host_id, assignment_epoch, drain_attempt_id, source_cell_id, + source_cell_incarnation, target_cell_id, target_cell_incarnation, + source_request_units, target_reserved_units, pinned_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 2, + 'attempt-0000', + 'cell-a', + '11111111-1111-4111-8111-111111111111', + 'cell-b', + '11111111-1111-4111-8111-111111111111', + 1, + 1, + now + ] + ) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + expect(await store.completeReadyEvacuations()).toBe(0) + expect(await store.abortExpiredEvacuations()).toBe(1) + expect(await store.abortExpiredEvacuations()).toBe(0) + }) + + it('repairs an expired migration marker from its exact active target control', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: 2, + generation: 1 + }) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: 2, + generation: 1 + }) + expect(await store.abortExpiredEvacuations()).toBe(0) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toEqual({ + inProgress: 1, + oldestExpiresAt: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs, + oldestRemainingMs: -1, + targetRegistered: 1, + registeredSourceActive: 1, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 0, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + await store.releaseActivity(identity, sourceControl) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toEqual({ + inProgress: 0, + ...NO_ACTIVE_MIGRATION_LEASE, + targetRegistered: 0, + ...NO_REGISTERED_EVACUATION_DIAGNOSTICS, + completed: 1, + blocked: 0, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + expect(await store.resolve(identity)).toMatchObject({ cellId: 'cell-b', assignmentEpoch: 2 }) + }) + + it('keeps a registered migration pending while its proven target control is offline', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, 'cell-b') + const targetControl = await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 2 + }) + expect( + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + ).toBe(true) + expect(await store.completeReadyEvacuations()).toBe(0) + await store.releaseActivity(identity, sourceControl) + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + expect(await store.completeReadyEvacuations()).toBe(0) + await expect( + store.completeEvacuation(identity, migration.assignmentEpoch) + ).rejects.toThrow('migration_target_not_active') + await store.releaseActivity(identity, targetControl) + expect(await store.completeReadyEvacuations()).toBe(0) + + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toEqual({ + inProgress: 1, + oldestExpiresAt: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs, + oldestRemainingMs: + ASSIGNMENT_LIMITS.migrationLeaseMs - ASSIGNMENT_LIMITS.activityLeaseMs - 1, + targetRegistered: 1, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 1, + completed: 0, + blocked: 1, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 3 + }) + expect(await store.completeReadyEvacuations()).toBe(1) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toEqual({ + inProgress: 0, + ...NO_ACTIVE_MIGRATION_LEASE, + targetRegistered: 0, + ...NO_REGISTERED_EVACUATION_DIAGNOSTICS, + completed: 0, + blocked: 0, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 1 }) + }) + + it('completes a registered migration after its disabled source is proven dead', async () => { + let now = 100 + const store = await setupWithHeartbeats(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + await heartbeat(store, { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }) + await heartbeat(store, { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + now += 45_001 + await heartbeat( + store, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { startedAt: 50 } + ) + await store.attestCellFence( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + + const input = { + assignmentEpoch: migration.assignmentEpoch, + sourceCellId: 'cell-a', + targetCellId: 'cell-b' + } + await expect(store.completeEvacuationFromDeadSource(identity, input)).resolves.toEqual({ + changed: true, + assignmentEpoch: 2, + sourceCellId: 'cell-a', + targetCellId: 'cell-b' + }) + await expect(store.completeEvacuationFromDeadSource(identity, input)).resolves.toEqual({ + changed: false, + assignmentEpoch: 2, + sourceCellId: 'cell-a', + targetCellId: 'cell-b' + }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 1 }) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toMatchObject({ + inProgress: 0 + }) + }) + + it('retires an inactive registered migration after its source is proven dead', async () => { + let now = 100 + const cells = [ + { + id: 'cell-a', + url: 'https://relay-a.example.com', + capacityRequests: 10, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'cell-b', + url: 'https://relay-b.example.com', + capacityRequests: 10, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + now += 45_001 + await heartbeat(store, cells[1]!, { startedAt: 50 }) + await store.attestCellFence('cell-a', '11111111-1111-4111-8111-111111111111') + + await expect( + store.cellEvacuationStatus('cell-a', 'cell-b', true) + ).resolves.toMatchObject({ inProgress: 0, completed: 1, blocked: 0 }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 0 }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([]) + expect( + await database!.query( + `SELECT state FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? AND cell_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, 'cell-b', migration.assignmentEpoch] + ) + ).toEqual([{ state: 'released' }]) + expect( + await database!.query( + `SELECT cell_id, assignment_epoch, reserved_controls, migration_leases + FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { + cell_id: 'cell-b', + assignment_epoch: 2, + reserved_controls: 0, + migration_leases: 0 + } + ]) + }) + + it.each(['expired', 'previous-incarnation'] as const)( + 'retires a registered migration and its %s target control after the source dies', + async (targetControlState) => { + let now = 100 + const cells = [ + { + id: 'cell-a', + url: 'https://relay-a.example.com', + capacityRequests: 10, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + }, + { + id: 'cell-b', + url: 'https://relay-b.example.com', + capacityRequests: 10, + connectionHardCap: 600 as const, + connectionUnobservedBound: 50 + } + ] + const store = await setupWithHeartbeats(() => now, cells) + await heartbeat(store, cells[0]!) + await heartbeat(store, cells[1]!) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + now += + targetControlState === 'expired' + ? ASSIGNMENT_LIMITS.activityLeaseMs + 1 + : 45_001 + await heartbeat(store, cells[1]!, { + startedAt: targetControlState === 'previous-incarnation' ? now - 1 : 50, + incarnation: + targetControlState === 'previous-incarnation' + ? '22222222-2222-4222-8222-222222222222' + : '11111111-1111-4111-8111-111111111111' + }) + await store.attestCellFence( + 'cell-a', + '11111111-1111-4111-8111-111111111111' + ) + + await expect( + store.cellEvacuationStatus('cell-a', 'cell-b', false) + ).resolves.toMatchObject({ + inProgress: 1, + registeredCompletable: 0, + registeredTargetInactive: 1 + }) + await expect( + store.cellEvacuationStatus('cell-a', 'cell-b', true) + ).resolves.toMatchObject({ inProgress: 0, completed: 1, blocked: 0 }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 0 }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([]) + expect( + await database!.query( + `SELECT cell_id, assignment_epoch, reserved_controls, migration_leases + FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { + cell_id: 'cell-b', + assignment_epoch: migration.assignmentEpoch, + reserved_controls: 0, + migration_leases: 0 + } + ]) + } + ) + + it('refuses dead-source completion while the source heartbeat is fresh', async () => { + const store = await setupWithHeartbeats(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + await heartbeat(store, { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }) + await heartbeat(store, { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + + await expect( + store.attestCellFence('cell-a', '11111111-1111-4111-8111-111111111111') + ).rejects.toThrow('cell_fence_runtime_not_stale') + await expect( + store.completeEvacuationFromDeadSource(identity, { + assignmentEpoch: migration.assignmentEpoch, + sourceCellId: 'cell-a', + targetCellId: 'cell-b' + }) + ).rejects.toThrow('cell_fence_attestation_missing') + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 2 }) + }) + + it('supersedes a registered migration after its disabled target is proven unavailable', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + for (const cell of cells) await heartbeat(store, cell) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await database!.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES (?, ?, ?, ?, ?, ?, 'late-arrival-debt', NULL, NULL, ?, ?, NULL, NULL, ?)`, + [ + 'superseded-reservation', + 'superseded-reservation', + identity.userId, + identity.relayHostId, + migration.assignmentEpoch, + 'cell-b', + 100, + 100, + 100 + ] + ) + await store.setCellEnabled('cell-b', false) + now += 45_001 + await heartbeat(store, cells[0]!, { startedAt: 50 }) + await heartbeat(store, cells[2]!, { startedAt: 50 }) + await store.attestCellFence( + 'cell-b', + '11111111-1111-4111-8111-111111111111' + ) + const input = { + assignmentEpoch: migration.assignmentEpoch, + sourceCellId: 'cell-a', + currentTargetCellId: 'cell-b', + replacementTargetCellId: 'cell-c' + } + + const replacement = await store.supersedeRegisteredEvacuation(identity, input) + expect(replacement).toMatchObject({ + sourceCellId: 'cell-a', + targetCellId: 'cell-c', + previousEpoch: 2, + assignmentEpoch: 3 + }) + await expect(store.supersedeRegisteredEvacuation(identity, input)).resolves.toEqual(replacement) + expect(await store.resolve(identity)).toMatchObject({ cellId: 'cell-c', assignmentEpoch: 3 }) + expect( + await database!.query( + `SELECT state FROM relay_control_connection_reservations + WHERE reservation_id = ?`, + ['superseded-reservation'] + ) + ).toEqual([{ state: 'released' }]) + expect(await cellReservations(database!)).toEqual({ + 'cell-a': 1, + 'cell-b': 0, + 'cell-c': 2 + }) + expect( + await database!.query( + `SELECT activity_kind, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY cell_id, activity_id`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { activity_kind: 'control', cell_id: 'cell-a' }, + { activity_kind: 'control', cell_id: 'cell-c' }, + { activity_kind: 'migration', cell_id: 'cell-c' } + ]) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toMatchObject({ + inProgress: 0 + }) + expect(await store.cellEvacuationStatus('cell-a', 'cell-c', false)).toMatchObject({ + inProgress: 1 + }) + }) + + it('retries registered migration supersession with inventory locked first', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const delegate = await openInMemoryRelayDatabase() + const retryDatabase = new OneShotInventoryFailureDatabase(delegate) + database = retryDatabase + const store = new RelayAssignmentStore(retryDatabase, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells(cells) + for (const cell of cells) await heartbeat(store, cell) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.setCellEnabled('cell-b', false) + now += 45_001 + await heartbeat(store, cells[0]!, { startedAt: 50 }) + await heartbeat(store, cells[2]!, { startedAt: 50 }) + await store.attestCellFence( + 'cell-b', + '11111111-1111-4111-8111-111111111111' + ) + retryDatabase.arm() + + await expect( + store.supersedeRegisteredEvacuation(identity, { + assignmentEpoch: migration.assignmentEpoch, + sourceCellId: 'cell-a', + currentTargetCellId: 'cell-b', + replacementTargetCellId: 'cell-c' + }) + ).resolves.toMatchObject({ targetCellId: 'cell-c', assignmentEpoch: 3 }) + expect(retryDatabase.retryLocks).toEqual([ + 'inventory-nowait-failed', + 'inventory-first', + 'assignment-nowait' + ]) + }) + + it('reconciles durable cell accounting once before aggregate supersession', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + for (const cell of cells) await heartbeat(store, cell) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.setCellEnabled('cell-b', false) + now += 45_001 + await heartbeat(store, cells[0]!, { startedAt: 50 }) + await heartbeat(store, cells[2]!, { startedAt: 50 }) + await store.attestCellFence( + 'cell-b', + '11111111-1111-4111-8111-111111111111' + ) + await database!.query( + `UPDATE relay_cells + SET reserved_requests = CASE WHEN cell_id = ? THEN 1 ELSE 0 END + WHERE cell_id IN (?, ?)`, + ['cell-c', 'cell-a', 'cell-c'] + ) + + await expect( + store.supersedeRegisteredCellEvacuations('cell-a', 'cell-b', 'cell-c', 100) + ).resolves.toBe(1) + expect(await cellReservations(database!)).toEqual({ + 'cell-a': 1, + 'cell-b': 0, + 'cell-c': 2 + }) + expect( + await database!.query( + `SELECT assignment_epoch, target_cell_id, aborted_at + FROM relay_assignment_migrations + WHERE user_id = ? ORDER BY assignment_epoch`, + [identity.userId] + ) + ).toEqual([ + { assignment_epoch: 2, target_cell_id: 'cell-b', aborted_at: 45_101 }, + { assignment_epoch: 3, target_cell_id: 'cell-c', aborted_at: null } + ]) + }) + + it('preserves healthy third-cell activity during aggregate supersession', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 }, + { id: 'cell-d', url: 'https://relay-d.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + for (const cell of cells) await heartbeat(store, cell) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await database!.query( + `UPDATE relay_assignment_activity_leases SET cell_id = ? + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + ['cell-d', identity.userId, identity.relayHostId, sourceControl] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = ? WHERE cell_id = ?`, + [5, 'cell-c'] + ) + await store.setCellEnabled('cell-b', false) + now += 45_001 + await heartbeat(store, cells[0]!, { startedAt: 50 }) + await heartbeat(store, cells[2]!, { startedAt: 50 }) + await heartbeat(store, cells[3]!, { startedAt: 50 }) + await store.attestCellFence( + 'cell-b', + '11111111-1111-4111-8111-111111111111' + ) + + await expect( + store.supersedeRegisteredCellEvacuations('cell-a', 'cell-b', 'cell-c', 100) + ).resolves.toBe(1) + expect(await cellReservations(database!)).toEqual({ + 'cell-a': 0, + 'cell-b': 0, + 'cell-c': 1, + 'cell-d': 1 + }) + expect( + await database!.query( + `SELECT activity_kind, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY cell_id, activity_id`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { activity_kind: 'control', cell_id: 'cell-c' }, + { activity_kind: 'migration', cell_id: 'cell-c' }, + { activity_kind: 'control', cell_id: 'cell-d' } + ]) + }) + + it('refuses supersession while the registered target remains available', async () => { + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => 100, cells) + for (const cell of cells) await heartbeat(store, cell) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.setCellEnabled('cell-b', false) + + await expect( + store.attestCellFence('cell-b', '11111111-1111-4111-8111-111111111111') + ).rejects.toThrow('cell_fence_runtime_not_stale') + await expect( + store.supersedeRegisteredEvacuation(identity, { + assignmentEpoch: migration.assignmentEpoch, + sourceCellId: 'cell-a', + currentTargetCellId: 'cell-b', + replacementTargetCellId: 'cell-c' + }) + ).rejects.toThrow('cell_fence_attestation_missing') + expect( + await database!.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: 'cell-b', assignment_epoch: 2 }]) + }) + + it('serializes concurrent retries of registered migration supersession', async () => { + let now = 100 + const cells = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 10 } + ] + const store = await setupWithHeartbeats(() => now, cells) + for (const cell of cells) await heartbeat(store, cell) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled('cell-a', false) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.setCellEnabled('cell-b', false) + now += 45_001 + await heartbeat(store, cells[0]!, { startedAt: 50 }) + await heartbeat(store, cells[2]!, { startedAt: 50 }) + await store.attestCellFence( + 'cell-b', + '11111111-1111-4111-8111-111111111111' + ) + const input = { + assignmentEpoch: migration.assignmentEpoch, + sourceCellId: 'cell-a', + currentTargetCellId: 'cell-b', + replacementTargetCellId: 'cell-c' + } + + const results = await Promise.all([ + store.supersedeRegisteredEvacuation(identity, input), + store.supersedeRegisteredEvacuation(identity, input) + ]) + expect(results[0]).toEqual(results[1]) + expect( + await database!.query( + `SELECT assignment_epoch FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? ORDER BY assignment_epoch`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ assignment_epoch: 2 }, { assignment_epoch: 3 }]) + expect(await cellReservations(database!)).toEqual({ + 'cell-a': 1, + 'cell-b': 0, + 'cell-c': 2 + }) + }) + + it('classifies an expired migration blocked by a newer target assignment', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: 2, + generation: 1 + }) + await database!.query( + `UPDATE relay_assignments SET assignment_epoch = ? + WHERE user_id = ? AND relay_host_id = ?`, + [3, identity.userId, identity.relayHostId] + ) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toEqual({ + inProgress: 1, + oldestExpiresAt: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs, + oldestRemainingMs: -1, + targetRegistered: 0, + ...NO_REGISTERED_EVACUATION_DIAGNOSTICS, + completed: 0, + blocked: 0, + expiredUnregistered: 1, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 1, + blockedExpiredOnNewerTargetAssignment: 1 + }) + }) + + it('does not complete a registered migration after the assignment epoch advances', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await database!.query( + `UPDATE relay_assignments SET assignment_epoch = ? + WHERE user_id = ? AND relay_host_id = ?`, + [migration.assignmentEpoch + 1, identity.userId, identity.relayHostId] + ) + + expect(await store.completeReadyEvacuations()).toBe(0) + await expect( + store.completeEvacuation(identity, migration.assignmentEpoch) + ).rejects.toThrow('migration_assignment_mismatch') + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toMatchObject({ + inProgress: 1, + targetRegistered: 1 + }) + }) + + it('retires an expired migration without rewriting its newer target assignment', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: 2, + generation: 1 + }) + await database!.query( + `UPDATE relay_assignments SET assignment_epoch = ? + WHERE user_id = ? AND relay_host_id = ?`, + [3, identity.userId, identity.relayHostId] + ) + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + expect(await store.abortExpiredEvacuations()).toBe(1) + expect(await store.resolve(identity)).toMatchObject({ + cellId: 'cell-b', + assignmentEpoch: 3 + }) + expect( + await database!.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY activity_id`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { activity_id: 'control:cell-a:1' }, + { activity_id: 'control:cell-b:1' } + ]) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toMatchObject({ + inProgress: 0 + }) + }) + + it.each([ + { assignmentEpoch: 1, removeAssignment: false, expected: 'migration_assignment_mismatch' }, + { assignmentEpoch: 2, removeAssignment: true, expected: 'migration_assignment_missing' } + ])( + 'fails closed when expired migration assignment state is not superseding: $expected', + async ({ assignmentEpoch, removeAssignment, expected }) => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await store.startEvacuation(identity, 'cell-b') + if (removeAssignment) { + await database!.query( + `DELETE FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + } else { + await database!.query( + `UPDATE relay_assignments SET assignment_epoch = ? + WHERE user_id = ? AND relay_host_id = ?`, + [assignmentEpoch, identity.userId, identity.relayHostId] + ) + } + + now += ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + await expect(store.abortExpiredEvacuations()).rejects.toThrow(expected) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toMatchObject({ + inProgress: 1, + targetRegistered: 0 + }) + } + ) + + it('reconciles duplicated assignment counters and cell reservations after evacuation', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: 2, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: 2 + }) + await store.releaseActivity(identity, sourceControl) + await database!.query( + `UPDATE relay_assignments SET reserved_controls = 9, reserved_splices = 4, + reserved_invites = 3, pending_installs = 2, pending_confirmations = 2, + migration_leases = 7 WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = + CASE WHEN cell_id = ? THEN 11 ELSE 3 END`, + ['cell-a'] + ) + + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toEqual({ + inProgress: 0, + ...NO_ACTIVE_MIGRATION_LEASE, + targetRegistered: 0, + ...NO_REGISTERED_EVACUATION_DIAGNOSTICS, + completed: 1, + blocked: 0, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 1 }) + expect( + await database!.query( + `SELECT reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases + FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([ + { + reserved_controls: 1, + reserved_splices: 0, + reserved_invites: 0, + pending_installs: 0, + pending_confirmations: 0, + migration_leases: 0 + } + ]) + }) + + it('refuses reservation reconciliation when an activity lease has no assignment', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + ['orphan-user', 'orphanhost000001', 'control:orphan', 'control', 'cell-a', 1, 200, 100] + ) + + await expect(store.cellEvacuationStatus('cell-a', 'cell-b', true)).rejects.toThrow( + 'activity_lease_assignment_missing' + ) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 0 }) + }) + + it('refuses reservation reconciliation when an activity lease names no cell', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + await store.assign(identity) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + 'splice:missing-cell', + 'splice', + 'missing-cell', + 2, + 200, + 100 + ] + ) + + await expect(store.cellEvacuationStatus('cell-a', 'cell-b', true)).rejects.toThrow( + 'activity_lease_cell_missing' + ) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 1, 'cell-b': 0 }) + }) + + it('leaves accounting for cells outside the selected evacuation pair unchanged', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 20 } + ]) + await database!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + ['unrelated-user', 'unrelatedhost001', 'cell-c', 1, 200, 100, 9, 0, 0, 0, 0, 0] + ) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + 'unrelated-user', + 'unrelatedhost001', + 'control:unrelated', + 'control', + 'cell-c', + 1, + 200, + 100 + ] + ) + await database!.query( + `UPDATE relay_cells SET reserved_requests = 9 WHERE cell_id = ?`, + ['cell-c'] + ) + + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toMatchObject({ + inProgress: 0 + }) + expect(await cellReservations(database!)).toEqual({ + 'cell-a': 0, + 'cell-b': 0, + 'cell-c': 9 + }) + expect( + await database!.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + ['unrelated-user', 'unrelatedhost001'] + ) + ).toEqual([{ reserved_controls: 9 }]) + }) + + it('rebalances only a fully inactive assignment after the dormant TTL', async () => { + let now = 100 + const store = await setup(() => now, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const first = await store.assign(identity) + const control = await store.activateControl(identity, { + cellId: first.cellId, + assignmentEpoch: first.assignmentEpoch, + generation: 1 + }) + await expect(store.rebalanceDormant(identity, 'cell-b')).rejects.toThrow('assignment_active') + await store.releaseActivity(identity, control) + now += ASSIGNMENT_LIMITS.dormantTtlMs + 1 + + await expect(store.rebalanceDormant(identity, 'cell-b')).resolves.toMatchObject({ + cellId: 'cell-b', + assignmentEpoch: 2 + }) + expect(await cellReservations(database!)).toEqual({ 'cell-a': 0, 'cell-b': 1 }) + }) + + it('durably removes an unhealthy cell from new assignment without moving existing hosts', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + const existing = { userId: 'user-a', relayHostId: 'host000000000001' } + expect(await store.assign(existing)).toMatchObject({ cellId: 'cell-a' }) + await store.setCellEnabled('cell-a', false) + await store.reconcileCells([ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } + ]) + expect(await store.resolve(existing)).toMatchObject({ cellId: 'cell-a' }) + await expect( + store.assign({ userId: 'user-b', relayHostId: 'host000000000002' }) + ).resolves.toMatchObject({ cellId: 'cell-b' }) + await store.setCellEnabled('cell-a', true) + await expect(store.setCellEnabled('missing', false)).rejects.toThrow('cell_not_found') + }) + + it('bulk-migrates active controls without exposing assignment identities', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + await store.setCellEnabled('cell-b', false) + const identities = [1, 2, 3].map((index) => ({ + userId: `user-${index}`, + relayHostId: `host${String(index).padStart(12, '0')}` + })) + const sourceControls: string[] = [] + for (const identity of identities) { + const assignment = await store.assign(identity) + sourceControls.push( + await store.activateControl(identity, { + cellId: 'cell-a', + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + ) + } + await store.setCellEnabled('cell-b', true) + expect(await store.cellEvacuationCapacity('cell-a', 'cell-b')).toEqual({ + sourceAssignments: 3, + requiredTargetUnits: 6, + availableTargetUnits: 20 + }) + expect(await store.startActiveCellEvacuations('cell-a', 'cell-b', 2)).toBe(2) + expect(await store.startActiveCellEvacuations('cell-a', 'cell-b', 2)).toBe(1) + expect(await store.startActiveCellEvacuations('cell-a', 'cell-b', 2)).toBe(0) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toEqual({ + inProgress: 3, + oldestExpiresAt: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs, + oldestRemainingMs: ASSIGNMENT_LIMITS.migrationLeaseMs, + targetRegistered: 0, + ...NO_REGISTERED_EVACUATION_DIAGNOSTICS, + completed: 0, + blocked: 0, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + for (const identity of identities) { + const assignment = await store.resolve(identity) + await store.activateControl(identity, { + cellId: 'cell-b', + assignmentEpoch: assignment!.assignmentEpoch, + generation: 2 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: 'cell-b', + assignmentEpoch: assignment!.assignmentEpoch + }) + } + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toEqual({ + inProgress: 3, + oldestExpiresAt: 100 + ASSIGNMENT_LIMITS.migrationLeaseMs, + oldestRemainingMs: ASSIGNMENT_LIMITS.migrationLeaseMs, + targetRegistered: 3, + registeredSourceActive: 3, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 3, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + for (let index = 0; index < identities.length; index++) { + await store.releaseActivity(identities[index]!, sourceControls[index]!) + } + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', true)).toEqual({ + inProgress: 0, + ...NO_ACTIVE_MIGRATION_LEASE, + targetRegistered: 0, + ...NO_REGISTERED_EVACUATION_DIAGNOSTICS, + completed: 3, + blocked: 0, + ...NO_EXPIRED_EVACUATION_DIAGNOSTICS + }) + }) + + it.each(['cell-b', 'cell-c'])( + 'skips a batch row that concurrently moved from the selected source to %s', + async (movedCellId) => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 }, + { id: 'cell-c', url: 'https://relay-c.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: 'cell-a', + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + const startEvacuation = store.startEvacuation.bind(store) + vi.spyOn(store, 'startEvacuation').mockImplementationOnce(async (...args) => { + await database!.query( + `UPDATE relay_assignments SET cell_id = ? WHERE user_id = ? AND relay_host_id = ?`, + [movedCellId, identity.userId, identity.relayHostId] + ) + return await startEvacuation(...args) + }) + + expect(await store.startActiveCellEvacuations('cell-a', 'cell-b', 10)).toBe(0) + expect(await store.resolve(identity)).toMatchObject({ cellId: movedCellId }) + expect(await database!.query(`SELECT * FROM relay_assignment_migrations`)).toEqual([]) + } + ) + + it('does not count an existing migration after a stale batch row moves to its target', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: 'cell-a', + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + const startEvacuation = store.startEvacuation.bind(store) + vi.spyOn(store, 'startEvacuation').mockImplementationOnce(async (...args) => { + await startEvacuation(identity, 'cell-b') + return await startEvacuation(...args) + }) + + expect(await store.startActiveCellEvacuations('cell-a', 'cell-b', 10)).toBe(0) + expect(await store.resolve(identity)).toMatchObject({ cellId: 'cell-b' }) + expect(await store.cellEvacuationStatus('cell-a', 'cell-b', false)).toMatchObject({ + inProgress: 1 + }) + }) + + it('preserves operator-owned tagged URLs and admission state across reconciliation', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 } + ]) + await store.configureCell( + { id: 'cell-a', url: 'https://old---relay-a.example.com', capacityRequests: 20 }, + false + ) + await store.reconcileCells([ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 } + ]) + const rows = await database!.query( + `SELECT cell_url, enabled, capacity_requests FROM relay_cells WHERE cell_id = ?`, + ['cell-a'] + ) + expect(rows).toEqual([ + { + cell_url: 'https://old---relay-a.example.com', + enabled: 0, + capacity_requests: 20 + } + ]) + }) + + it('bulk-migrates activity even when its source control is temporarily absent', async () => { + const store = await setup(() => 100, [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 20 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 20 } + ]) + const identity = { userId: 'user-1', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + const control = await store.activateControl(identity, { + cellId: 'cell-a', + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.acquireActivity(identity, { + activityId: 'splice-without-control', + kind: 'splice', + cellId: 'cell-a' + }) + await store.releaseActivity(identity, control) + expect(await store.cellEvacuationCapacity('cell-a', 'cell-b')).toEqual({ + sourceAssignments: 1, + requiredTargetUnits: 3, + availableTargetUnits: 20 + }) + expect(await store.startActiveCellEvacuations('cell-a', 'cell-b', 10)).toBe(1) + expect(await store.resolve(identity)).toMatchObject({ cellId: 'cell-b', assignmentEpoch: 2 }) + }) +}) + +async function cellReservations(database: RelayDatabase): Promise> { + const rows = await database.query( + `SELECT cell_id, reserved_requests FROM relay_cells ORDER BY cell_id ASC` + ) + return Object.fromEntries(rows.map((row) => [String(row.cell_id), Number(row.reserved_requests)])) +} diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts new file mode 100644 index 00000000000..0b2b1ef72a9 --- /dev/null +++ b/cloud/apps/relay/src/assignment-store.ts @@ -0,0 +1,8347 @@ +import { randomUUID } from 'node:crypto' +import { performance } from 'node:perf_hooks' +import { + ASSIGNMENT_LIMITS, + mayNormallyReassign, + RELAY_ADMISSION_BUDGETS, + RELAY_DEFAULT_REGION, + RELAY_REGIONS, + RELAY_PROTOCOL_LIMITS, + type RelayRegion +} from '@orca-cloud/relay-contract' +import { + cellAdmissionState, + cellAdmissionStates, + ensureCellAdmission, + parseCellAdmissionState, + RelayCellAdmissionSelector, + setCellAdmissionBeforeBoundary, + stateFromEnabled, + synchronizeCellAdmissionBoundary, + type CellAdmissionMembership, + type CellAdmissionSelectorInspection, + type CellAdmissionState +} from './cell-admission-selector.js' +import { + RelayMigrationCellRegistrar, + type MigrationCellRegistration +} from './cell-admission-migration-registration.js' +import { + ASSIGNMENT_CONNECTION_HEADROOM_QUERY +} from './assignment-connection-headroom-query.js' +import { AssignmentIdentityQueue } from './assignment-identity-queue.js' +import type { RelayCellConfig } from './config.js' +import type { RelayDatabase, RelayTransactionOptions, SqlRow } from './database.js' +import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' +import { + combineRegionalRehomeSafety, + REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT, + REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT, + regionalRehomePoolPressure, + regionalRehomeSafetyFailure +} from './regional-rehome-safety.js' +import { + ABANDONED_REGISTERED_MIGRATION, + DURABLY_FENCED_MIGRATION_SOURCE, + REGISTERED_MIGRATION_ABANDON_MS +} from './registered-migration-abandonment.js' + +type AssignmentIdentity = { userId: string; relayHostId: string } +type CellHeartbeat = { + cellId: string + cellUrl: string + cellIncarnation: string + startedAt: number + ready: boolean + observedRequests: number + region?: RelayRegion + totalConnections?: number + inFlightConnections?: number + reservedConnectionUnits?: number + enforcedConnectionUnits?: number + connectionInclusionWatermark?: number + connectionHardCap?: number + connectionUnobservedBound?: number +} + +type CellRegionalRehomeStatus = { + cellId: string + cellIncarnation: string + regionalRehomeProtocol: number + safety: RegionalRehomeSafetySnapshot +} + +type RelayAssignmentStoreOptions = { + requireLiveCells?: boolean + heartbeatTtlMs?: number + recordControlRenewal?: (durationMs: number, outcome: ControlRenewalOutcome) => void +} + +export type ControlRenewalOutcome = + | 'renewed' + | 'assignment_not_found' + | 'activity_cell_not_authoritative' + | 'control_activity_not_found' + | 'control_activity_moved' + | 'database_error' + +const CONTROL_RENEWAL_OUTCOMES = new Set([ + 'renewed', + 'assignment_not_found', + 'activity_cell_not_authoritative', + 'control_activity_not_found', + 'control_activity_moved' +]) +export type RelayAssignment = AssignmentIdentity & { + cellId: string + cellUrl: string + assignmentEpoch: number + leaseExpiresAt: number + region?: RelayRegion +} + +export type RelayRegionCatalogEntry = { + region: RelayRegion + probeOrigins: string[] +} + +export type RelayAssignmentMigration = AssignmentIdentity & { + sourceCellId: string + targetCellId: string + previousEpoch: number + assignmentEpoch: number + expiresAt: number + targetRegisteredAt?: number +} + +export type RegionalRehomeAttempt = AssignmentIdentity & { + attemptId: string + preferredRegion: 'asia-east2' + sourceCellId: string + sourceCellUrl: string + sourceCellIncarnation: string + targetCellId: string + targetCellIncarnation: string + previousEpoch: number + assignmentEpoch: number + drainGraceMs: number + sendAttempts: number +} + +export type RegionalHostDrainOutcome = + | 'accepted' + | 'already-accepted' + | 'host-not-connected' + +export type RegionalRehomeFleetSafety = RegionalRehomeSafetySnapshot & { + requiredCells: number + missingCells: number + maxReconnects: number +} + +export type RegionalRehomeControl = { + generation: number + enabled: boolean + observationStartedAt: number + notBefore: number + ratePerMinute: number + preferenceMaxAgeMs: number + drainGraceMs: number +} + +type RegisteredEvacuationSupersessionInput = { + assignmentEpoch: number + sourceCellId: string + currentTargetCellId: string + replacementTargetCellId: string +} + +export type DeadSourceCompletionResult = { + changed: boolean + assignmentEpoch: number + sourceCellId: string + targetCellId: string +} + +export type CellEvacuationStatus = { + inProgress: number + oldestExpiresAt: number | null + oldestRemainingMs: number | null + targetRegistered: number + registeredSourceActive: number + registeredCompletable: number + registeredTargetInactive: number + completed: number + blocked: number + expiredUnregistered: number + repairableExpiredUnregistered: number + abortableExpiredUnregistered: number + blockedExpiredUnregistered: number + blockedExpiredOnNewerTargetAssignment: number +} + +export type CellEvacuationCapacity = { + sourceAssignments: number + requiredTargetUnits: number + availableTargetUnits: number +} + +export type CellDeploymentStatus = { + cellId: string + cellUrl: string + region: RelayRegion + enabled: boolean + admissionState: CellAdmissionState + capacityRequests: number + reservedRequests: number + assignments: number + activityLeases: number + activityRequestUnits: number + restartBlockingActivityLeases: number + restartBlockingActivityRequestUnits: number + restartBlockingReservedRequests: number + outgoingMigrations: number + incomingMigrations: number + connectionCapacity: null | { + hardCap: number + controlRebindReserve: number + ordinaryConnectionLimit: number + unobservedBound: number + normalAdmissionPause: number + observedConnections: number + inFlightConnections: number + reservedConnectionUnits: number + enforcedConnectionUnits: number + pendingControlReservations: number + heartbeatFresh: boolean + } + runtime: null | { + cellUrl: string + cellIncarnation: string + startedAt: number + ready: boolean + observedRequests: number + lastHeartbeatAt: number + heartbeatFresh: boolean + regionalRehomeProtocol: number + } +} + +export type CellFenceAttemptEvidence = { + attemptId: string + environment: 'staging' | 'production' + cellId: string + cellIncarnation: string + migName: string + instanceGroup: string + generationIdentity: string + fenceCommit: string + planSha256: string + planObjectName: string + planObjectGeneration?: string + varFileSha256: string + terraformStateLineage: string + terraformStateSerial: number + terraformStateObjectGeneration: string + terraformStateObjectSha256: string + requestReason: string +} + +export type CellFenceApplyInvocation = { + invocationId: string + requestReason: string + startedAt: number + gceOperation?: string +} + +export type CellFenceAttempt = CellFenceAttemptEvidence & { + applyInvocations?: CellFenceApplyInvocation[] + gceOperation?: string + createdAt: number + expiresAt: number + applyStartedAt?: number + completedAt?: number + abortedAt?: number +} + +export type CellDrainAttemptState = + | 'prepared' + | 'send-may-have-started' + | 'application-receipt' + | 'proven-not-delivered' + +export type CellDrainAttempt = { + attemptId: string + cellId: string + cellIncarnation: string + traceValue: string + plannedGraceMs: number + state: CellDrainAttemptState + preparedAt: number + sendMayHaveStartedAt?: number + sendPermitExpiresAt?: number + applicationReceiptAt?: number + backendSuccessStatus?: number + backendInstance?: string + receiptCellIncarnation?: string + retryAfter?: number + recoverForwardAttemptedAt?: number + provenNotDeliveredAt?: number +} + +export type AssignmentActivityKind = + | 'control' + | 'splice' + | 'invite' + | 'install' + | 'confirmation' + | 'migration' + +const ACTIVITY_COLUMN: Record = { + control: 'reserved_controls', + splice: 'reserved_splices', + invite: 'reserved_invites', + install: 'pending_installs', + confirmation: 'pending_confirmations', + migration: 'migration_leases' +} + +const ACTIVITY_REQUEST_UNITS: Record = { + control: 1, + splice: 2, + invite: 1, + install: 1, + confirmation: 1, + migration: 1 +} + +const ASSIGNMENT_LOCK_RETRY_DEADLINE_MS = 15_000 +// Why: stranded detection (issue #225) needs a grant old enough that a real +// attach would have registered (the 90s activity lease covers dial + +// activation), yet recent enough to prove an active retry loop rather than +// ordinary dormancy — which stays governed by the 24h rule. +const STRANDED_MIN_GRANT_AGE_MS = 60_000 +const STRANDED_RECENT_ACTIVITY_MS = 15 * 60_000 +const REGION_PREFERENCE_RETENTION_MS = 30 * 24 * 60 * 60_000 +const REGIONAL_REHOME_UNREGISTERED_REFRESH_MS = 5 * 60_000 +const REGIONAL_REHOME_MAX_REFRESH_MS = 24 * 60 * 60_000 +// Drain grace is enforced by session-scoped cell state that any control +// reconnect sheds, so receipted attempts can stall dual-homed past grace; +// redrains re-dispatch them with the elapsed (zero) grace until they detach. +export const REGIONAL_REHOME_REDRAIN_INTERVAL_MS = 60_000 +export const REGIONAL_REHOME_REDRAIN_SEND_LIMIT = 20 +export const REGIONAL_REHOME_QUARANTINE_FAILURES = 3 +export const REGIONAL_REHOME_QUARANTINE_MS = 15 * 60_000 +const REGIONAL_REHOME_QUARANTINE_EXCLUSION_LIMIT = 50 +const REGIONAL_REHOME_QUARANTINE_MEMORY_LIMIT = 1_000 +const REGIONAL_REHOME_OBSERVATION_MS = 24 * 60 * 60_000 +const ASSIGNMENT_LOCK_RETRY_MAX_DELAY_MS = 50 +type AssignmentInventoryScope = 'none' | 'general' | 'all' +type RetriedAssignmentInventoryScope = Exclude + +class AssignmentInventoryLockUnavailable extends Error { + constructor(readonly inventoryScope: RetriedAssignmentInventoryScope) { + super('database_lock_unavailable') + } +} + +class AssignmentInventoryScopeChanged extends Error { + constructor() { + super('assignment_inventory_scope_changed') + } +} + +// Debt holds connection headroom for a control that may still arrive shortly +// after its director-side timeout. Nothing legitimately arrives minutes late +// (attach deadline 10s, orphan grace 30s); unretired debt from hosts that +// never return otherwise starves connection headroom fleet-wide and turns +// every placement into relay_capacity_exhausted. +const LATE_ARRIVAL_DEBT_RETENTION_MS = 10 * 60 * 1_000 +const CELL_FENCE_TTL_MS = 5 * 60 * 1_000 +const CELL_FENCE_ATTEMPT_TTL_MS = 60 * 60 * 1_000 +const CELL_DRAIN_SEND_PERMIT_MS = 30_000 +const MAX_DRAIN_ACCOUNTING_REPAIR_ATTEMPTS = 3 +const CELL_FENCE_ATTEMPT_SELECT = ` + SELECT attempts.*, bindings.plan_object_name, bindings.plan_object_generation, + bindings.var_file_sha256, bindings.terraform_state_lineage, + bindings.terraform_state_serial, bindings.terraform_state_object_generation, + bindings.terraform_state_object_sha256, bindings.request_reason + FROM relay_cell_fence_attempts attempts + JOIN relay_cell_fence_plan_bindings bindings + ON bindings.attempt_id = attempts.attempt_id` +export const STRANDED_MIGRATION_ABANDON_MS = REGISTERED_MIGRATION_ABANDON_MS +const ABORTABLE_EXPIRED_MIGRATION = `( + ( + migration.target_registered_at IS NULL + AND NOT EXISTS ( + SELECT 1 FROM relay_post_drain_migration_pins pin + WHERE pin.user_id = migration.user_id + AND pin.relay_host_id = migration.relay_host_id + AND pin.assignment_epoch = migration.assignment_epoch + ) + ) + OR ${ABANDONED_REGISTERED_MIGRATION} +)` + +export class RelayAssignmentStore { + private readonly requireLiveCells: boolean + private readonly heartbeatTtlMs: number + // Poisoned attempts never complete or abort and stay the oldest rows, so + // unquarantined they eventually fill the sweeps' LIMIT page and starve + // every healthy candidate. Process-local on purpose: it resets on deploy + // and each director relearns within a few ticks. + private readonly regionalRehomeCandidateQuarantine = new Map< + string, + { failures: number; until: number } + >() + private readonly recordControlRenewal?: RelayAssignmentStoreOptions['recordControlRenewal'] + private readonly admissionSelector: RelayCellAdmissionSelector + private readonly migrationCellRegistrar: RelayMigrationCellRegistrar + private readonly activityQueue = new AssignmentIdentityQueue() + private assignmentTail: Promise = Promise.resolve() + private pendingRegionalRehomeDisableLog: Record | null = null + + constructor( + private readonly database: RelayDatabase, + private readonly now: () => number = Date.now, + options: RelayAssignmentStoreOptions = {} + ) { + this.requireLiveCells = options.requireLiveCells ?? false + this.heartbeatTtlMs = options.heartbeatTtlMs ?? 45_000 + this.recordControlRenewal = options.recordControlRenewal + this.admissionSelector = new RelayCellAdmissionSelector(database, now) + this.migrationCellRegistrar = new RelayMigrationCellRegistrar(database, now) + } + + async reconcileCells(cells: RelayCellConfig[], disableMissing = true): Promise { + await this.reconcileCellsWithOptions(cells, disableMissing) + } + + async reconcileCellsAtStartup(cells: RelayCellConfig[]): Promise { + await this.reconcileCellsWithOptions(cells, false, { reportRetries: false }) + } + + private async reconcileCellsWithOptions( + cells: RelayCellConfig[], + disableMissing: boolean, + transactionOptions?: RelayTransactionOptions + ): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT cell_id FROM relay_cells ORDER BY cell_id ASC`) + // Capacity rows share one global lock order with assignment transactions. + for (const cell of [...cells].sort((left, right) => left.id.localeCompare(right.id))) { + // Operator-owned enabled state and tagged URLs must survive revision restarts. + await transaction.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + capacity_requests = excluded.capacity_requests, + last_heartbeat_at = excluded.last_heartbeat_at, + updated_at = excluded.updated_at`, + [ + cell.id, + cell.url, + cell.initiallyEnabled === false ? 0 : 1, + cell.capacityRequests, + 0, + 0, + now, + now + ] + ) + await transaction.query( + `INSERT INTO relay_cell_regions (cell_id, region) VALUES (?, ?) + ON CONFLICT (cell_id) DO UPDATE SET region = excluded.region`, + [cell.id, cell.region ?? RELAY_DEFAULT_REGION] + ) + await ensureCellAdmission( + transaction, + cell.id, + stateFromEnabled(cell.initiallyEnabled !== false), + now + ) + if ( + cell.connectionHardCap !== undefined && + cell.connectionUnobservedBound !== undefined + ) { + const currentLimit = ( + await transaction.queryLocked( + `SELECT hard_cap, unobserved_bound FROM relay_cell_connection_limits + WHERE cell_id = ?`, + [cell.id] + ) + )[0] + const limitChanged = + currentLimit !== undefined && + (integer(currentLimit, 'hard_cap') !== cell.connectionHardCap || + integer(currentLimit, 'unobserved_bound') !== + cell.connectionUnobservedBound) + await transaction.query( + `INSERT INTO relay_cell_connection_limits + (cell_id, hard_cap, unobserved_bound, updated_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + hard_cap = excluded.hard_cap, + unobserved_bound = excluded.unobserved_bound, + updated_at = excluded.updated_at`, + [ + cell.id, + cell.connectionHardCap, + cell.connectionUnobservedBound, + now + ] + ) + if (limitChanged) { + await transaction.query( + `DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [cell.id] + ) + await transaction.query( + `DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, + [cell.id] + ) + } + } + } + const current = await transaction.query( + `SELECT cell_id, enabled FROM relay_cells ORDER BY cell_id ASC` + ) + for (const row of current) { + await ensureCellAdmission( + transaction, + text(row, 'cell_id'), + stateFromEnabled(integer(row, 'enabled') === 1), + now + ) + } + if (disableMissing) { + const ids = new Set(cells.map(({ id }) => id)) + for (const row of current) { + const id = text(row, 'cell_id') + if ( + !ids.has(id) && + (await cellAdmissionState(transaction, id)) !== 'existing-only' + ) { + await setCellAdmissionBeforeBoundary(transaction, id, 'existing-only', now) + } + } + } + await synchronizeCellAdmissionBoundary(transaction, now) + }, transactionOptions) + } + + async assign( + identity: AssignmentIdentity, + preferredRegion?: RelayRegion, + placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION + ): Promise { + const sticky = await this.assignStickyWithLockRetry(identity, preferredRegion) + if (sticky) return sticky + // Only placement needs the global inventory critical section; queueing those + // attempts locally avoids turning true placement bursts into NOWAIT storms. + return await this.serializeAssignment( + async () => await this.assignWithLockRetry(identity, preferredRegion, placementRegion) + ) + } + + private async assignStickyWithLockRetry( + identity: AssignmentIdentity, + preferredRegion?: RelayRegion + ): Promise { + return await this.withAssignmentLockRetry( + async (inventoryFirst) => + await this.assignStickyOnce(identity, inventoryFirst, preferredRegion) + ) + } + + private async assignWithLockRetry( + identity: AssignmentIdentity, + preferredRegion?: RelayRegion, + placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION + ): Promise { + const deadline = Date.now() + ASSIGNMENT_LOCK_RETRY_DEADLINE_MS + let inventoryScope: AssignmentInventoryScope = 'none' + while (true) { + try { + return await this.assignOnce(identity, inventoryScope, preferredRegion, placementRegion) + } catch (error) { + if (error instanceof AssignmentInventoryScopeChanged) { + inventoryScope = 'all' + continue + } + if ( + !(error instanceof AssignmentInventoryLockUnavailable) || + Date.now() >= deadline + ) { + throw error + } + inventoryScope = error.inventoryScope + } + await waitForAssignmentLockRetry(deadline) + } + } + + private async withAssignmentLockRetry( + operation: (inventoryFirst: boolean) => Promise + ): Promise { + const deadline = Date.now() + ASSIGNMENT_LOCK_RETRY_DEADLINE_MS + let inventoryFirst = false + while (true) { + try { + return await operation(inventoryFirst) + } catch (error) { + if (!isDatabaseLockUnavailable(error) || Date.now() >= deadline) throw error + inventoryFirst = true + } + // The retry joins the cell queue without holding later legacy-cycle locks. + await waitForAssignmentLockRetry(deadline) + } + } + + private async assignStickyOnce( + identity: AssignmentIdentity, + inventoryFirst: boolean, + preferredRegion?: RelayRegion + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const lockedCells = inventoryFirst + ? await this.lockCellInventory(transaction) + : undefined + const existing = await this.assignmentRow(transaction, identity, inventoryFirst) + if (!existing) return null + const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) + await this.recordRegionPreference(transaction, identity, preferredRegion, now) + if (mayNormallyReassign(activity(existing), now)) return null + // Why: a stranded host must fall through to placement — re-granting the + // pinned cell here is what refreshes its own activity and sustains the + // loop (issue #225). + if (await this.assignmentStrandedOnUnservedCell(transaction, identity, existing, now)) { + return null + } + + const currentCellId = text(existing, 'cell_id') + const hadControl = holdsControlLease( + activityLeases, + currentCellId, + integer(existing, 'assignment_epoch') + ) + const currentRow = lockedCells + ? lockedCells.find((row) => text(row, 'cell_id') === currentCellId) + : hadControl + ? ( + await transaction.query(`SELECT * FROM relay_cells WHERE cell_id = ?`, [ + currentCellId + ]) + )[0] + : ( + await transaction.queryLocked( + `SELECT * FROM relay_cells WHERE cell_id = ?`, + [currentCellId], + { failIfUnavailable: true } + ) + )[0] + if (!currentRow) throw new Error('assigned_cell_missing') + if ( + this.requireLiveCells && + !(await this.cellIsLive(transaction, currentCellId, now)) + ) { + return null + } + + if ( + !hadControl && + !(await this.cellHasConnectionHeadroom(transaction, currentCellId)) + ) { + if (requestUnits(existing) === 0) return null + throw new Error('relay_connection_headroom_exhausted') + } + + const leaseExpiresAt = now + ASSIGNMENT_LIMITS.activityLeaseMs + if (hadControl) { + await this.touchAssignment(transaction, identity, leaseExpiresAt, now) + } else { + const nextReservation = integer(currentRow, 'reserved_requests') + 1 + if (nextReservation > integer(currentRow, 'capacity_requests')) { + throw new Error('relay_capacity_exhausted') + } + await transaction.query( + `UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`, + [nextReservation, now, currentCellId] + ) + await this.adjustActivityCount(transaction, identity, 'control', 1, leaseExpiresAt, now) + await this.insertPendingControlLease( + transaction, + identity, + currentCellId, + integer(existing, 'assignment_epoch'), + now + ) + } + return this.result( + identity, + existing, + cell(currentRow, await this.cellRegion(transaction, currentCellId)), + leaseExpiresAt + ) + }) + } + + // Why: PR #194 pins assignments on existing-only cells so capacity pressure + // cannot scatter existing hosts — assuming those cells still serve them. A + // decommissioning cell (C3) broke that assumption: existing-only AND + // rejecting attaches, so pinned hosts loop forever while each grant + // refreshes their own last_activity_at, keeping dormancy-based + // reassignment permanently out of reach (issue #225). "Stranded" therefore + // requires proof the cell is not serving this host: a recent grant + // (last_activity_at inside the window) that had ample time to attach and + // still produced no live real activity. The evidence must come from the + // assignment row itself — reservation rows do not exist for cells without + // connection limits (C3), and expired pending leases are cleaned within a + // maintenance cycle, so neither reliably survives until the next grant. + private async assignmentStrandedOnUnservedCell( + transaction: RelayDatabase, + identity: AssignmentIdentity, + existing: SqlRow, + now: number + ): Promise { + const lastActivityAt = integer(existing, 'last_activity_at') + if ( + now < lastActivityAt + STRANDED_MIN_GRANT_AGE_MS || + now >= lastActivityAt + STRANDED_RECENT_ACTIVITY_MS + ) { + return false + } + const admissionRow = ( + await transaction.query( + `SELECT admission_state FROM relay_cell_admission WHERE cell_id = ?`, + [text(existing, 'cell_id')] + ) + )[0] + // Unknown or missing admission fails safe: the pin stays. + if (admissionRow?.['admission_state'] !== 'existing-only') { + return false + } + const liveLeases = ( + await transaction.query( + `SELECT COUNT(*) AS live FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND expires_at > ? + AND activity_id NOT LIKE 'control-pending:%'`, + [identity.userId, identity.relayHostId, now] + ) + )[0] + return integer(liveLeases!, 'live') === 0 + } + + private async serializeAssignment(operation: () => Promise): Promise { + const previous = this.assignmentTail + let release!: () => void + this.assignmentTail = new Promise((resolve) => (release = resolve)) + await previous + try { + return await operation() + } finally { + release() + } + } + + private async assignOnce( + identity: AssignmentIdentity, + inventoryScope: AssignmentInventoryScope, + preferredRegion?: RelayRegion, + placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION + ): Promise { + const now = this.now() + let retryScope: RetriedAssignmentInventoryScope = + inventoryScope === 'all' ? 'all' : 'general' + return await this.database.transaction(async (transaction) => { + let lockedCells = + inventoryScope === 'all' + ? await this.lockCellInventory(transaction) + : inventoryScope === 'general' + ? await this.lockGeneralCellInventory(transaction) + : undefined + const existing = await this.assignmentRow( + transaction, + identity, + inventoryScope !== 'none' + ) + retryScope = existing ? 'all' : 'general' + if (inventoryScope === 'general' && existing) { + throw new AssignmentInventoryScopeChanged() + } + const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) + await this.recordRegionPreference(transaction, identity, preferredRegion, now) + let forcedDeadReassignment = false + let connectionHeadroomReassignment = false + let strandedReassignment = false + if (existing && !mayNormallyReassign(activity(existing), now)) { + lockedCells ??= await this.lockCellInventory(transaction, true) + const admission = await cellAdmissionStates(transaction) + const currentRow = lockedCells.find( + (row) => text(row, 'cell_id') === text(existing, 'cell_id') + ) + if (!currentRow) throw new Error('assigned_cell_missing') + const current = cell( + currentRow, + await this.cellRegion(transaction, text(existing, 'cell_id')) + ) + // Why: an existing-only cell that rejects attaches (C3's decommission + // posture) must not re-pin the hosts it refuses; the dead-cell fence + // path below does not apply either — the cell is live, just unwilling. + strandedReassignment = await this.assignmentStrandedOnUnservedCell( + transaction, + identity, + existing, + now + ) + if ( + !strandedReassignment && + (!this.requireLiveCells || (await this.cellIsLive(transaction, current.cellId, now))) + ) { + const hadControl = holdsControlLease( + activityLeases, + current.cellId, + integer(existing, 'assignment_epoch') + ) + const hasConnectionHeadroom = + hadControl || + (await this.cellHasConnectionHeadroom(transaction, current.cellId)) + if (hasConnectionHeadroom) { + const leaseExpiresAt = now + ASSIGNMENT_LIMITS.activityLeaseMs + if (hadControl) { + await this.touchAssignment(transaction, identity, leaseExpiresAt, now) + } else { + await this.adjustCellReservation(transaction, current.cellId, 1) + await this.adjustActivityCount( + transaction, + identity, + 'control', + 1, + leaseExpiresAt, + now + ) + await this.insertPendingControlLease( + transaction, + identity, + current.cellId, + integer(existing, 'assignment_epoch'), + now + ) + } + return this.result(identity, existing, current, leaseExpiresAt) + } + if (requestUnits(existing) > 0) { + throw new Error('relay_connection_headroom_exhausted') + } + if (admission.get(current.cellId) !== 'general') { + throw new Error('relay_connection_headroom_exhausted') + } + connectionHeadroomReassignment = true + } + if (!strandedReassignment && !connectionHeadroomReassignment) { + if ( + (await this.deadCellRequiresCommittedFence( + transaction, + identity, + current.cellId, + integer(existing, 'assignment_epoch') + )) && + !(await this.cellHasCommittedFence(transaction, current.cellId, now)) + ) { + throw new Error('relay_capacity_exhausted') + } + forcedDeadReassignment = true + } + } + + lockedCells ??= existing + ? await this.lockCellInventory(transaction, true) + : await this.lockGeneralCellInventory(transaction, true) + const target = await this.leastLoadedCell( + transaction, + lockedCells, + placementRegion + ) + if (!target) throw new Error('relay_capacity_exhausted') + const previousUnits = existing ? requestUnits(existing) : 0 + if (existing) { + await this.adjustCellReservation(transaction, text(existing, 'cell_id'), -previousUnits) + if (forcedDeadReassignment || strandedReassignment) { + // A fenced/dead incarnation cannot own drainable work, and a + // stranded host's only leases are the unclaimed grant artifacts of + // its own loop. Removing them prevents late expiry from + // decrementing the replacement. + await transaction.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + } + } + await this.adjustCellReservation(transaction, target.cellId, 1) + const assignmentEpoch = existing ? integer(existing, 'assignment_epoch') + 1 : 1 + const leaseExpiresAt = now + ASSIGNMENT_LIMITS.activityLeaseMs + await transaction.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (user_id, relay_host_id) DO UPDATE SET + cell_id = excluded.cell_id, + assignment_epoch = excluded.assignment_epoch, + lease_expires_at = excluded.lease_expires_at, + last_activity_at = excluded.last_activity_at, + reserved_controls = excluded.reserved_controls, + reserved_splices = excluded.reserved_splices, + reserved_invites = excluded.reserved_invites, + pending_installs = excluded.pending_installs, + pending_confirmations = excluded.pending_confirmations, + migration_leases = excluded.migration_leases`, + [ + identity.userId, + identity.relayHostId, + target.cellId, + assignmentEpoch, + leaseExpiresAt, + now, + 1, + 0, + 0, + 0, + 0, + 0 + ] + ) + if (existing) { + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + text(existing, 'cell_id'), + integer(existing, 'assignment_epoch'), + now + ) + } + await this.insertPendingControlLease( + transaction, + identity, + target.cellId, + assignmentEpoch, + now + ) + return { ...identity, ...target, assignmentEpoch, leaseExpiresAt } + }).catch((error: unknown) => { + if (isDatabaseLockUnavailable(error)) { + throw new AssignmentInventoryLockUnavailable(retryScope) + } + throw error + }) + } + + async resolve(identity: AssignmentIdentity): Promise { + const rows = await this.database.query( + `SELECT assignment.*, cell.cell_url, region.region + FROM relay_assignments assignment + JOIN relay_cells cell ON cell.cell_id = assignment.cell_id + LEFT JOIN relay_cell_regions region ON region.cell_id = cell.cell_id + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + const row = rows[0] + if ( + row && + this.requireLiveCells && + !(await this.cellIsLive(this.database, text(row, 'cell_id'), this.now())) + ) { + return null + } + return row + ? { + ...identity, + cellId: text(row, 'cell_id'), + cellUrl: text(row, 'cell_url'), + region: optionalRelayRegion(row, 'region') ?? RELAY_DEFAULT_REGION, + assignmentEpoch: integer(row, 'assignment_epoch'), + leaseExpiresAt: integer(row, 'lease_expires_at') + } + : null + } + + async setCellEnabled(cellId: string, enabled: boolean): Promise { + await this.setCellAdmissionState(cellId, stateFromEnabled(enabled)) + } + + async setCellAdmissionState(cellId: string, state: CellAdmissionState): Promise { + await this.database.transaction( + async (transaction) => + await setCellAdmissionBeforeBoundary(transaction, cellId, state, this.now()) + ) + } + + async applyCellAdmissionSelector(input: { + attemptId: string + expectedGeneration: number + expectedMembershipSha256?: string + membership: CellAdmissionMembership + }): Promise<{ + changed: boolean + selector: { + generation: number + attemptId: string | null + membership: CellAdmissionMembership + } + }> { + return await this.admissionSelector.apply(input) + } + + async inspectCellAdmissionSelector( + attemptId?: string + ): Promise { + return await this.admissionSelector.inspect(attemptId) + } + + async addMigrationCells(input: { + attemptId: string + expectedGeneration: number + cells: MigrationCellRegistration[] + }): Promise<{ + changed: boolean + selector: { + generation: number + attemptId: string | null + membership: CellAdmissionMembership + } + }> { + return await this.migrationCellRegistrar.add(input) + } + + async recordCellHeartbeat(input: CellHeartbeat): Promise { + const now = this.now() + let inclusionWatermark: number | undefined + await this.database.transaction(async (transaction) => { + const configured = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [input.cellId]) + )[0] + if (!configured) throw new Error('cell_not_found') + if (text(configured, 'cell_url') !== input.cellUrl) throw new Error('cell_origin_mismatch') + if ( + (await this.cellRegion(transaction, input.cellId)) !== + (input.region ?? RELAY_DEFAULT_REGION) + ) { + throw new Error('cell_region_mismatch') + } + const current = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if ( + current && + ((text(current, 'cell_incarnation') === input.cellIncarnation && + input.startedAt !== integer(current, 'started_at')) || + (text(current, 'cell_incarnation') !== input.cellIncarnation && + input.startedAt <= integer(current, 'started_at'))) + ) { + throw new Error('stale_cell_incarnation') + } + const connectionLimit = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_connection_limits WHERE cell_id = ?`, + [input.cellId] + ) + )[0] + if (connectionLimit) { + if ( + input.totalConnections === undefined || + input.inFlightConnections === undefined || + input.reservedConnectionUnits === undefined || + input.enforcedConnectionUnits === undefined || + input.enforcedConnectionUnits !== + input.totalConnections + + input.inFlightConnections + + input.reservedConnectionUnits || + input.connectionHardCap !== integer(connectionLimit, 'hard_cap') || + input.connectionUnobservedBound !== + integer(connectionLimit, 'unobserved_bound') + ) { + throw new Error('cell_connection_telemetry_mismatch') + } + } else if ( + input.totalConnections !== undefined || + input.inFlightConnections !== undefined || + input.reservedConnectionUnits !== undefined || + input.enforcedConnectionUnits !== undefined || + input.connectionInclusionWatermark !== undefined || + input.connectionHardCap !== undefined || + input.connectionUnobservedBound !== undefined + ) { + throw new Error('cell_connection_limit_not_configured') + } + await transaction.query( + `INSERT INTO relay_cell_runtime + (cell_id, cell_url, cell_incarnation, started_at, ready, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_url = excluded.cell_url, + cell_incarnation = excluded.cell_incarnation, + started_at = excluded.started_at, + ready = excluded.ready, + observed_requests = excluded.observed_requests, + last_heartbeat_at = excluded.last_heartbeat_at, + updated_at = excluded.updated_at`, + [ + input.cellId, + input.cellUrl, + input.cellIncarnation, + input.startedAt, + input.ready ? 1 : 0, + input.observedRequests, + now, + now + ] + ) + // Live directors read relay_cell_runtime; keep heartbeats off the placement capacity lock. + if (!this.requireLiveCells) { + await transaction.query( + `UPDATE relay_cells SET observed_requests = ?, last_heartbeat_at = ?, updated_at = ? + WHERE cell_id = ?`, + [input.observedRequests, now, now, input.cellId] + ) + } + if (connectionLimit) { + const currentSnapshot = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [input.cellId] + ) + )[0] + const previousWatermark = + currentSnapshot && + text(currentSnapshot, 'cell_incarnation') === input.cellIncarnation + ? integer(currentSnapshot, 'inclusion_watermark') + : -1 + inclusionWatermark = input.connectionInclusionWatermark ?? previousWatermark + 1 + if ( + input.connectionInclusionWatermark !== undefined && + inclusionWatermark <= previousWatermark + ) { + throw new Error('stale_connection_snapshot') + } + await transaction.query( + `INSERT INTO relay_cell_connection_runtime + (cell_id, cell_incarnation, total_connections, in_flight_connections, + reserved_connection_units, enforced_connection_units, last_heartbeat_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + total_connections = excluded.total_connections, + in_flight_connections = excluded.in_flight_connections, + reserved_connection_units = excluded.reserved_connection_units, + enforced_connection_units = excluded.enforced_connection_units, + last_heartbeat_at = excluded.last_heartbeat_at, + updated_at = excluded.updated_at`, + [ + input.cellId, + input.cellIncarnation, + input.totalConnections, + input.inFlightConnections, + input.reservedConnectionUnits, + input.enforcedConnectionUnits, + now, + now + ] + ) + await transaction.query( + `INSERT INTO relay_cell_connection_snapshots + (cell_id, cell_incarnation, inclusion_watermark, total_connections, + in_flight_connections, reserved_connection_units, + enforced_connection_units, snapshot_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + inclusion_watermark = excluded.inclusion_watermark, + total_connections = excluded.total_connections, + in_flight_connections = excluded.in_flight_connections, + reserved_connection_units = excluded.reserved_connection_units, + enforced_connection_units = excluded.enforced_connection_units, + snapshot_at = excluded.snapshot_at`, + [ + input.cellId, + input.cellIncarnation, + inclusionWatermark, + input.totalConnections, + input.inFlightConnections, + input.reservedConnectionUnits, + input.enforcedConnectionUnits, + now + ] + ) + } + await transaction.query(`DELETE FROM relay_cell_fences WHERE cell_id = ?`, [input.cellId]) + await transaction.query(`DELETE FROM relay_cell_committed_fences WHERE cell_id = ?`, [ + input.cellId + ]) + await transaction.query( + `DELETE FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + [input.cellId] + ) + }) + if (inclusionWatermark !== undefined) { + await this.releaseHeartbeatConnectionReservationDebt( + input.cellId, + input.cellIncarnation, + inclusionWatermark, + now + ) + } + } + + async recordCellRegionalRehomeStatus(input: CellRegionalRehomeStatus): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + const runtime = ( + await transaction.queryLocked( + `SELECT cell_incarnation FROM relay_cell_runtime WHERE cell_id = ?`, + [input.cellId] + ) + )[0] + if (!runtime || text(runtime, 'cell_incarnation') !== input.cellIncarnation) { + throw new Error('stale_cell_incarnation') + } + await transaction.query( + `INSERT INTO relay_cell_capabilities + (cell_id, cell_incarnation, regional_rehome_protocol, last_heartbeat_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + regional_rehome_protocol = excluded.regional_rehome_protocol, + last_heartbeat_at = excluded.last_heartbeat_at`, + [input.cellId, input.cellIncarnation, input.regionalRehomeProtocol, now] + ) + const safety = input.safety + await transaction.query( + `INSERT INTO relay_cell_rehome_safety + (cell_id, cell_incarnation, observed_at, sql_failures, reconnects, + control_activity_recovery_failures, database_pool_waiting, + database_pool_waiters_max, database_pool_wait_ms_max) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + observed_at = excluded.observed_at, + sql_failures = excluded.sql_failures, + reconnects = excluded.reconnects, + control_activity_recovery_failures = + excluded.control_activity_recovery_failures, + database_pool_waiting = excluded.database_pool_waiting, + database_pool_waiters_max = excluded.database_pool_waiters_max, + database_pool_wait_ms_max = excluded.database_pool_wait_ms_max`, + [ + input.cellId, + input.cellIncarnation, + safety.observedAt, + safety.sqlFailures, + safety.reconnects, + safety.controlActivityRecoveryFailures, + safety.databasePoolWaiting, + safety.databasePoolWaitersMax, + safety.databasePoolWaitMsMax + ] + ) + }) + } + + private async releaseHeartbeatConnectionReservationDebt( + cellId: string, + cellIncarnation: string, + inclusionWatermark: number, + now: number + ): Promise { + await this.database.transaction(async (transaction) => { + const runtime = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, + [cellId] + ) + )[0] + const snapshot = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [cellId] + ) + )[0] + if ( + !runtime || + !snapshot || + text(runtime, 'cell_incarnation') !== cellIncarnation || + text(snapshot, 'cell_incarnation') !== cellIncarnation || + integer(snapshot, 'inclusion_watermark') !== inclusionWatermark + ) { + return + } + await transaction.query( + `UPDATE relay_control_connection_reservations + SET state = 'released', released_at = ?, updated_at = ? + WHERE cell_id = ? AND state = 'claimed' + AND inclusion_watermark IS NOT NULL + AND inclusion_watermark <= ?`, + [now, now, cellId, inclusionWatermark] + ) + await transaction.query( + `UPDATE relay_control_connection_reservations + SET state = 'released', released_at = ?, updated_at = ? + WHERE reservation_id IN ( + SELECT duplicate.reservation_id + FROM relay_control_connection_reservations duplicate + WHERE duplicate.cell_id = ? + AND duplicate.state = 'late-arrival-debt' + AND duplicate.claim_activity_id IS NULL + AND duplicate.timeout_at <= ? + AND EXISTS ( + SELECT 1 + FROM relay_control_connection_reservations retained + WHERE retained.user_id = duplicate.user_id + AND retained.relay_host_id = duplicate.relay_host_id + AND retained.assignment_epoch = duplicate.assignment_epoch + AND retained.cell_id = duplicate.cell_id + AND retained.state = 'late-arrival-debt' + AND retained.claim_activity_id IS NULL + AND retained.timeout_at <= ? + AND ( + retained.created_at < duplicate.created_at OR + ( + retained.created_at = duplicate.created_at AND + retained.reservation_id < duplicate.reservation_id + ) + ) + ) + )`, + [now, now, cellId, now, now] + ) + await transaction.query( + `UPDATE relay_control_connection_reservations + SET state = 'released', released_at = ?, updated_at = ? + WHERE cell_id = ? AND state = 'late-arrival-debt' + AND claim_activity_id IS NULL AND timeout_at <= ? + AND EXISTS ( + SELECT 1 FROM relay_assignments assignment + WHERE assignment.user_id = + relay_control_connection_reservations.user_id + AND assignment.relay_host_id = + relay_control_connection_reservations.relay_host_id + AND assignment.assignment_epoch > + relay_control_connection_reservations.assignment_epoch + ) + AND EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = + relay_control_connection_reservations.user_id + AND migration.relay_host_id = + relay_control_connection_reservations.relay_host_id + AND migration.assignment_epoch = + relay_control_connection_reservations.assignment_epoch + AND migration.target_cell_id = + relay_control_connection_reservations.cell_id + AND migration.aborted_at IS NOT NULL + )`, + [now, now, cellId, now] + ) + }) + } + + async attestCellFence(cellId: string, cellIncarnation: string): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellId]) + )[0] + if (!cell) throw new Error('cell_not_found') + if (integer(cell, 'enabled') !== 0) throw new Error('cell_fence_admission_enabled') + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + cellId + ]) + )[0] + if ( + !runtime || + text(runtime, 'cell_incarnation') !== cellIncarnation || + integer(runtime, 'last_heartbeat_at') > now - this.heartbeatTtlMs + ) { + throw new Error('cell_fence_runtime_not_stale') + } + const expiresAt = now + CELL_FENCE_TTL_MS + await transaction.query( + `INSERT INTO relay_cell_fences + (cell_id, cell_incarnation, attested_at, expires_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + attested_at = excluded.attested_at, + expires_at = excluded.expires_at`, + [cellId, cellIncarnation, now, expiresAt] + ) + return expiresAt + }) + } + + async adoptLegacyCellFence(cellId: string, cellIncarnation: string): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellId]) + )[0] + if (!cell) throw new Error('cell_not_found') + if (integer(cell, 'enabled') !== 0) throw new Error('cell_fence_admission_enabled') + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + cellId + ]) + )[0] + if ( + !runtime || + text(runtime, 'cell_incarnation') !== cellIncarnation || + integer(runtime, 'last_heartbeat_at') > now - this.heartbeatTtlMs + ) { + throw new Error('cell_fence_runtime_not_stale') + } + const attempt = ( + await transaction.queryLocked( + `SELECT attempt_id FROM relay_cell_fence_attempts + WHERE cell_id = ? ORDER BY created_at DESC LIMIT 1`, + [cellId] + ) + )[0] + if (attempt) throw new Error('legacy_cell_fence_attempt_exists') + const expiresAt = now + CELL_FENCE_TTL_MS + await transaction.query( + `INSERT INTO relay_cell_fences + (cell_id, cell_incarnation, attested_at, expires_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + attested_at = excluded.attested_at, + expires_at = excluded.expires_at`, + [cellId, cellIncarnation, now, expiresAt] + ) + return expiresAt + }) + } + + async commitLegacyCellFenceAdoption( + cellId: string, + cellIncarnation: string + ): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellId]) + )[0] + if (!cell) throw new Error('cell_not_found') + if (integer(cell, 'enabled') !== 0) throw new Error('cell_fence_admission_enabled') + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + cellId + ]) + )[0] + const fence = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_fences WHERE cell_id = ?`, [ + cellId + ]) + )[0] + const attempt = ( + await transaction.queryLocked( + `SELECT attempt_id FROM relay_cell_fence_attempts + WHERE cell_id = ? ORDER BY created_at DESC LIMIT 1`, + [cellId] + ) + )[0] + if (attempt) throw new Error('legacy_cell_fence_attempt_exists') + if ( + !runtime || + !fence || + text(runtime, 'cell_incarnation') !== cellIncarnation || + text(fence, 'cell_incarnation') !== cellIncarnation || + integer(runtime, 'last_heartbeat_at') > now - this.heartbeatTtlMs || + integer(fence, 'attested_at') < integer(runtime, 'last_heartbeat_at') || + integer(fence, 'expires_at') <= now + ) { + throw new Error('legacy_cell_fence_not_active') + } + await transaction.query( + `INSERT INTO relay_cell_legacy_fence_adoptions + (cell_id, cell_incarnation, attested_at, expires_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + attested_at = excluded.attested_at, + expires_at = excluded.expires_at`, + [cellId, cellIncarnation, now, integer(fence, 'expires_at')] + ) + }) + } + + async prepareCellFenceAttempt(input: CellFenceAttemptEvidence): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + let createdAt = now + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if (!cell) throw new Error('cell_not_found') + if (integer(cell, 'enabled') !== 0) throw new Error('cell_fence_admission_enabled') + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if (!runtime || text(runtime, 'cell_incarnation') !== input.cellIncarnation) { + throw new Error('cell_fence_runtime_mismatch') + } + const existing = ( + await transaction.queryLocked( + `${CELL_FENCE_ATTEMPT_SELECT} + WHERE attempts.cell_id = ? ORDER BY attempts.created_at DESC LIMIT 1`, + [input.cellId] + ) + )[0] + if (existing) { + const attempt = cellFenceAttempt(existing) + if (!attempt.completedAt && !attempt.abortedAt) { + assertCellFenceAttemptBase(existing, input) + return attempt + } + createdAt = Math.max(now, attempt.createdAt + 1) + } else { + const activeFence = ( + await transaction.queryLocked( + `SELECT cell_id FROM relay_cell_fences + WHERE cell_id = ? AND cell_incarnation = ? AND expires_at > ?`, + [input.cellId, input.cellIncarnation, now] + ) + )[0] + if (activeFence) throw new Error('cell_fence_already_attested') + } + const expiresAt = createdAt + CELL_FENCE_ATTEMPT_TTL_MS + await transaction.query( + `INSERT INTO relay_cell_fence_attempts + (attempt_id, environment, cell_id, cell_incarnation, mig_name, instance_group, + generation_identity, fence_commit, plan_sha256, gce_operation, created_at, + expires_at, apply_started_at, completed_at, aborted_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, ?, ?, NULL, NULL, NULL)`, + [ + input.attemptId, + input.environment, + input.cellId, + input.cellIncarnation, + input.migName, + input.instanceGroup, + input.generationIdentity, + input.fenceCommit, + input.planSha256, + createdAt, + expiresAt + ] + ) + await transaction.query( + `INSERT INTO relay_cell_fence_plan_bindings + (attempt_id, plan_object_name, plan_object_generation, var_file_sha256, + terraform_state_lineage, terraform_state_serial, + terraform_state_object_generation, terraform_state_object_sha256, + request_reason) + VALUES (?, ?, NULL, ?, ?, ?, ?, ?, ?)`, + [ + input.attemptId, + input.planObjectName, + input.varFileSha256, + input.terraformStateLineage, + input.terraformStateSerial, + input.terraformStateObjectGeneration, + input.terraformStateObjectSha256, + input.requestReason + ] + ) + const { planObjectGeneration: _ignored, ...prepared } = input + return { ...prepared, createdAt, expiresAt } + }) + } + + async bindCellFencePlanGeneration( + input: CellFenceAttemptEvidence, + planObjectGeneration: string + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const row = await lockedCellFenceAttempt(transaction, input.attemptId) + assertCellFenceAttemptBase(row, input) + if (integer(row, 'expires_at') <= now) throw new Error('cell_fence_attempt_expired') + if (row.apply_started_at !== null) throw new Error('cell_fence_apply_already_started') + if (row.completed_at !== null || row.aborted_at !== null) { + throw new Error('cell_fence_attempt_terminal') + } + const existing = optionalText(row, 'plan_object_generation') + if (existing && existing !== planObjectGeneration) { + throw new Error('cell_fence_plan_generation_mismatch') + } + if (!existing) { + await transaction.query( + `UPDATE relay_cell_fence_plan_bindings + SET plan_object_generation = ? WHERE attempt_id = ?`, + [planObjectGeneration, input.attemptId] + ) + row.plan_object_generation = planObjectGeneration + } + return cellFenceAttempt(row) + }) + } + + async startCellFenceApply( + input: CellFenceAttemptEvidence, + invocationId: string, + requestReason: string + ): Promise<{ attempt: CellFenceAttempt; invocation: CellFenceApplyInvocation }> { + const now = this.now() + return await this.database.transaction(async (transaction) => { + if (requestReason !== `${input.requestReason}/${invocationId}`) { + throw new Error('cell_fence_invocation_mismatch') + } + const row = await lockedCellFenceAttempt(transaction, input.attemptId) + assertActiveCellFenceAttempt(row, input, now) + const existing = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_fence_apply_invocations WHERE invocation_id = ?`, + [invocationId] + ) + )[0] + if (existing) { + if ( + text(existing, 'attempt_id') !== input.attemptId || + text(existing, 'request_reason') !== requestReason + ) { + throw new Error('cell_fence_invocation_mismatch') + } + return { + attempt: cellFenceAttempt(row), + invocation: cellFenceApplyInvocation(existing) + } + } + if (row.apply_started_at === null) { + await transaction.query( + `UPDATE relay_cell_fence_attempts SET apply_started_at = ? WHERE attempt_id = ?`, + [now, input.attemptId] + ) + row.apply_started_at = now + } + await transaction.query( + `INSERT INTO relay_cell_fence_apply_invocations + (invocation_id, attempt_id, request_reason, started_at, gce_operation) + VALUES (?, ?, ?, ?, NULL)`, + [invocationId, input.attemptId, requestReason, now] + ) + return { + attempt: cellFenceAttempt(row), + invocation: { invocationId, requestReason, startedAt: now } + } + }) + } + + async recordCellFenceOperation( + input: CellFenceAttemptEvidence, + invocationId: string, + requestReason: string, + gceOperation: string + ): Promise<{ attempt: CellFenceAttempt; invocation: CellFenceApplyInvocation }> { + const now = this.now() + return await this.database.transaction(async (transaction) => { + if (requestReason !== `${input.requestReason}/${invocationId}`) { + throw new Error('cell_fence_invocation_mismatch') + } + const row = await lockedCellFenceAttempt(transaction, input.attemptId) + assertActiveCellFenceAttempt(row, input, now) + if (row.apply_started_at === null) throw new Error('cell_fence_apply_not_started') + const invocation = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_fence_apply_invocations WHERE invocation_id = ?`, + [invocationId] + ) + )[0] + if ( + !invocation || + text(invocation, 'attempt_id') !== input.attemptId || + text(invocation, 'request_reason') !== requestReason + ) { + throw new Error('cell_fence_invocation_mismatch') + } + const existing = invocation.gce_operation + if (existing !== null && existing !== gceOperation) { + throw new Error('cell_fence_operation_mismatch') + } + await transaction.query( + `UPDATE relay_cell_fence_apply_invocations + SET gce_operation = ? WHERE invocation_id = ?`, + [gceOperation, invocationId] + ) + invocation.gce_operation = gceOperation + await transaction.query( + `UPDATE relay_cell_fence_attempts SET gce_operation = ? WHERE attempt_id = ?`, + [gceOperation, input.attemptId] + ) + row.gce_operation = gceOperation + return { + attempt: cellFenceAttempt(row), + invocation: cellFenceApplyInvocation(invocation) + } + }) + } + + async cellFenceAttempt(cellId: string): Promise { + const rows = await this.database.query( + `${CELL_FENCE_ATTEMPT_SELECT} + WHERE attempts.cell_id = ? ORDER BY attempts.created_at DESC LIMIT 1`, + [cellId] + ) + if (rows.length === 0) return null + const attempt = cellFenceAttempt(rows[0]!) + const invocations = await this.database.query( + `SELECT * FROM relay_cell_fence_apply_invocations + WHERE attempt_id = ? ORDER BY started_at, invocation_id`, + [attempt.attemptId] + ) + return { + ...attempt, + applyInvocations: invocations.map(cellFenceApplyInvocation) + } + } + + async abortCellFenceAttempt(input: CellFenceAttemptEvidence): Promise { + return await this.database.transaction(async (transaction) => { + const row = await lockedCellFenceAttempt(transaction, input.attemptId) + assertCellFenceAttemptBase(row, input) + if (row.completed_at !== null) throw new Error('cell_fence_attempt_completed') + if (row.aborted_at !== null) throw new Error('cell_fence_attempt_aborted') + if (row.apply_started_at !== null || row.gce_operation !== null) { + throw new Error('cell_fence_apply_may_have_started') + } + const abortedAt = this.now() + await transaction.query( + `UPDATE relay_cell_fence_attempts SET aborted_at = ? WHERE attempt_id = ?`, + [abortedAt, input.attemptId] + ) + row.aborted_at = abortedAt + return cellFenceAttempt(row) + }) + } + + async attestCellFenceAttempt( + input: CellFenceAttemptEvidence, + gceOperation: string + ): Promise<{ expiresAt: number; attempt: CellFenceAttempt }> { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const attemptRow = await lockedCellFenceAttempt(transaction, input.attemptId) + assertCellFenceAttemptEvidence(attemptRow, input) + if (attemptRow.completed_at !== null) { + if (attemptRow.gce_operation !== gceOperation) { + throw new Error('cell_fence_operation_not_attested') + } + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if (!cell) throw new Error('cell_not_found') + if (integer(cell, 'enabled') !== 0) throw new Error('cell_fence_admission_enabled') + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if ( + !runtime || + text(runtime, 'cell_incarnation') !== input.cellIncarnation || + integer(runtime, 'last_heartbeat_at') > now - this.heartbeatTtlMs + ) { + throw new Error('cell_fence_runtime_not_stale') + } + const expiresAt = now + CELL_FENCE_TTL_MS + await transaction.query( + `INSERT INTO relay_cell_fences + (cell_id, cell_incarnation, attested_at, expires_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + attested_at = excluded.attested_at, + expires_at = excluded.expires_at`, + [input.cellId, input.cellIncarnation, now, expiresAt] + ) + await this.recordCommittedCellFence( + transaction, + input.cellId, + input.cellIncarnation, + input.attemptId, + now, + expiresAt + ) + return { + expiresAt, + attempt: cellFenceAttempt(attemptRow) + } + } + assertActiveCellFenceAttempt(attemptRow, input, now) + if ( + attemptRow.apply_started_at === null || + attemptRow.gce_operation !== gceOperation + ) { + throw new Error('cell_fence_operation_not_attested') + } + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if (!cell) throw new Error('cell_not_found') + if (integer(cell, 'enabled') !== 0) throw new Error('cell_fence_admission_enabled') + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + input.cellId + ]) + )[0] + if ( + !runtime || + text(runtime, 'cell_incarnation') !== input.cellIncarnation || + integer(runtime, 'last_heartbeat_at') > now - this.heartbeatTtlMs + ) { + throw new Error('cell_fence_runtime_not_stale') + } + const expiresAt = now + CELL_FENCE_TTL_MS + await transaction.query( + `INSERT INTO relay_cell_fences + (cell_id, cell_incarnation, attested_at, expires_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_incarnation = excluded.cell_incarnation, + attested_at = excluded.attested_at, + expires_at = excluded.expires_at`, + [input.cellId, input.cellIncarnation, now, expiresAt] + ) + await this.recordCommittedCellFence( + transaction, + input.cellId, + input.cellIncarnation, + input.attemptId, + now, + expiresAt + ) + await transaction.query( + `UPDATE relay_cell_fence_attempts SET completed_at = ? WHERE attempt_id = ?`, + [now, input.attemptId] + ) + attemptRow.completed_at = now + return { expiresAt, attempt: cellFenceAttempt(attemptRow) } + }) + } + + async prepareCellDrainAttempt(input: { + attemptId: string + cellId: string + cellIncarnation: string + traceValue: string + plannedGraceMs: number + }): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.assertDrainCellGeneration( + transaction, + input.cellId, + input.cellIncarnation + ) + const existing = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_drain_attempt_states + WHERE cell_id = ? ORDER BY prepared_at DESC LIMIT 1`, + [input.cellId] + ) + )[0] + if (existing) { + if (text(existing, 'attempt_id') === input.attemptId) { + if ( + text(existing, 'cell_incarnation') !== input.cellIncarnation || + text(existing, 'trace_value') !== input.traceValue || + integer(existing, 'planned_grace_ms') !== input.plannedGraceMs + ) { + throw new Error('drain_attempt_generation_mismatch') + } + return { ...cellDrainAttempt(existing), shouldSend: false } + } + if ( + text(existing, 'state') !== 'proven-not-delivered' || + text(existing, 'cell_incarnation') !== input.cellIncarnation + ) { + throw new Error('drain_attempt_generation_mismatch') + } + } + await transaction.query( + `INSERT INTO relay_cell_drain_attempt_states + (attempt_id, cell_id, cell_incarnation, trace_value, planned_grace_ms, + state, prepared_at, send_may_have_started_at, send_permit_expires_at, + application_receipt_at, backend_success_status, backend_instance, + receipt_cell_incarnation, retry_after, recover_forward_attempted_at, + proven_not_delivered_at) + VALUES (?, ?, ?, ?, ?, 'prepared', ?, NULL, NULL, NULL, NULL, NULL, + NULL, NULL, NULL, NULL)`, + [ + input.attemptId, + input.cellId, + input.cellIncarnation, + input.traceValue, + input.plannedGraceMs, + now + ] + ) + return { + attemptId: input.attemptId, + cellId: input.cellId, + cellIncarnation: input.cellIncarnation, + traceValue: input.traceValue, + plannedGraceMs: input.plannedGraceMs, + state: 'prepared', + preparedAt: now, + shouldSend: false + } + }) + } + + async beginCellDrainSend(input: { + attemptId: string + cellId: string + cellIncarnation: string + }): Promise { + try { + return await this.beginCellDrainSendOnce(input) + } catch (error) { + if ( + !(error instanceof Error) || + error.message !== 'migration_activity_accounting_mismatch' + ) { + throw error + } + const activityCells = await this.database.query( + `SELECT DISTINCT lease.cell_id + FROM relay_assignment_activity_leases lease + WHERE EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = lease.user_id + AND migration.relay_host_id = lease.relay_host_id + AND migration.source_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ORDER BY lease.cell_id`, + [input.cellId] + ) + const cellIds = activityCells + .map((row) => text(row, 'cell_id')) + .filter((cellId) => cellId !== input.cellId) + for (const cellId of cellIds) { + await this.reconcileReservationAccounting(input.cellId, cellId) + } + await this.refreshDrainMigrationLeases(input) + return await this.beginCellDrainSendOnce(input) + } + } + + private async refreshDrainMigrationLeases(input: { + cellId: string + cellIncarnation: string + }): Promise { + await this.retireObsoleteDrainMigrations(input.cellId) + for (let repairAttempt = 0; ; repairAttempt++) { + try { + await this.refreshDrainMigrationLeasesOnce(input) + return + } catch (error) { + if ( + !(error instanceof Error) || + error.message !== 'migration_activity_accounting_mismatch' || + repairAttempt >= MAX_DRAIN_ACCOUNTING_REPAIR_ATTEMPTS + ) { + throw error + } + await this.reconcileDrainMigrationAccounting(input.cellId) + } + } + } + + private async reconcileDrainMigrationAccounting(sourceCellId: string): Promise { + const activityCells = await this.database.query( + `SELECT DISTINCT lease.cell_id + FROM relay_assignment_activity_leases lease + WHERE EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = lease.user_id + AND migration.relay_host_id = lease.relay_host_id + AND migration.source_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ORDER BY lease.cell_id`, + [sourceCellId] + ) + for (const cellId of activityCells + .map((row) => text(row, 'cell_id')) + .filter((cellId) => cellId !== sourceCellId)) { + await this.reconcileReservationAccounting(sourceCellId, cellId) + } + } + + private async retireObsoleteDrainMigrations(sourceCellId: string): Promise { + const candidates = await this.database.query( + `SELECT migration.user_id, migration.relay_host_id, + migration.assignment_epoch + FROM relay_assignment_migrations migration + JOIN relay_assignments assignment + ON assignment.user_id = migration.user_id + AND assignment.relay_host_id = migration.relay_host_id + WHERE migration.source_cell_id = ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + AND assignment.assignment_epoch > migration.assignment_epoch + ORDER BY migration.user_id, migration.relay_host_id`, + [sourceCellId] + ) + for (const candidate of candidates) { + await this.retireObsoleteDrainMigration(sourceCellId, candidate) + } + } + + private async retireObsoleteDrainMigration( + sourceCellId: string, + candidate: SqlRow + ): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + const identity = { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + } + const assignment = await this.assignmentRow(transaction, identity) + const assignmentEpoch = integer(candidate, 'assignment_epoch') + const migrationRow = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND source_cell_id = ? + AND completed_at IS NULL AND aborted_at IS NULL`, + [identity.userId, identity.relayHostId, assignmentEpoch, sourceCellId] + ) + )[0] + if (!migrationRow) return + if (!assignment || integer(assignment, 'assignment_epoch') <= assignmentEpoch) { + throw new Error('migration_assignment_mismatch') + } + const leases = await this.lockAssignmentActivities(transaction, identity) + const obsoleteActivityIds = new Set([ + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ]) + if (leases.some((lease) => obsoleteActivityIds.has(text(lease, 'activity_id')))) { + throw new Error('migration_activity_topology_mismatch') + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + text(migrationRow, 'target_cell_id'), + assignmentEpoch, + now + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + }) + } + + private async refreshDrainMigrationLeasesOnce(input: { + cellId: string + cellIncarnation: string + }): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + await this.assertDrainCellGeneration( + transaction, + input.cellId, + input.cellIncarnation + ) + const assignments = await transaction.queryLocked( + `SELECT assignment.* + FROM relay_assignments assignment + WHERE EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + AND migration.source_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ORDER BY assignment.user_id, assignment.relay_host_id`, + [input.cellId] + ) + const activityLeases = await transaction.queryLocked( + `SELECT lease.* + FROM relay_assignment_activity_leases lease + WHERE EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = lease.user_id + AND migration.relay_host_id = lease.relay_host_id + AND migration.source_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ORDER BY lease.user_id, lease.relay_host_id, lease.activity_id`, + [input.cellId] + ) + const migrations = await transaction.queryLocked( + `SELECT migration.* + FROM relay_assignment_migrations migration + WHERE migration.source_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ORDER BY migration.user_id, migration.relay_host_id`, + [input.cellId] + ) + const cells = await this.lockCellInventory(transaction) + for (const migrationRow of migrations) { + const identity = { + userId: text(migrationRow, 'user_id'), + relayHostId: text(migrationRow, 'relay_host_id') + } + const assignment = assignments.find( + (candidate) => + text(candidate, 'user_id') === identity.userId && + text(candidate, 'relay_host_id') === identity.relayHostId + ) + assertCurrentMigrationAssignment(assignment, migrationRow) + const leases = activityLeases.filter( + (lease) => + text(lease, 'user_id') === identity.userId && + text(lease, 'relay_host_id') === identity.relayHostId + ) + const migrationLeases = leases.filter( + (lease) => text(lease, 'activity_kind') === 'migration' + ) + const assignmentEpoch = integer(migrationRow, 'assignment_epoch') + const sourceRequestUnits = integer(migrationRow, 'source_request_units') + const targetCellId = text(migrationRow, 'target_cell_id') + const expiresAt = now + ASSIGNMENT_LIMITS.migrationLeaseMs + if (migrationLeases.length > 0) { + assertAssignmentActivityAccounting(assignment, leases, migrationRow) + await transaction.query( + `UPDATE relay_assignment_activity_leases SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? + AND activity_id IN (?, ?)`, + [ + expiresAt, + now, + identity.userId, + identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ] + ) + await transaction.query( + `UPDATE relay_assignments SET + lease_expires_at = CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [expiresAt, expiresAt, now, identity.userId, identity.relayHostId] + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [expiresAt, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + await this.refreshPendingControlReservation( + transaction, + identity, + targetCellId, + assignmentEpoch, + expiresAt, + now + ) + continue + } + if ( + optionalInteger(migrationRow, 'target_registered_at') === undefined || + integer(migrationRow, 'expires_at') > now || + sourceRequestUnits < 0 || + integer(migrationRow, 'target_reserved_units') !== sourceRequestUnits + 1 || + activityLeaseById(leases, migrationActivityId(assignmentEpoch)) || + !cells.some((cell) => text(cell, 'cell_id') === targetCellId) + ) { + throw new Error('migration_activity_lease_shape_mismatch') + } + await this.adjustCellReservation(transaction, targetCellId, sourceRequestUnits) + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, 'migration', ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + migrationActivityId(assignmentEpoch), + targetCellId, + sourceRequestUnits, + expiresAt, + now + ] + ) + await transaction.query( + `UPDATE relay_assignments SET migration_leases = migration_leases + 1, + lease_expires_at = CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [expiresAt, expiresAt, now, identity.userId, identity.relayHostId] + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [expiresAt, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + await this.refreshPendingControlReservation( + transaction, + identity, + targetCellId, + assignmentEpoch, + expiresAt, + now + ) + } + }) + } + + private async beginCellDrainSendOnce(input: { + attemptId: string + cellId: string + cellIncarnation: string + }): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.assertDrainCellGeneration( + transaction, + input.cellId, + input.cellIncarnation + ) + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_drain_attempt_states + WHERE attempt_id = ? AND cell_id = ?`, + [input.attemptId, input.cellId] + ) + )[0] + if (!row || text(row, 'cell_incarnation') !== input.cellIncarnation) { + throw new Error('drain_attempt_not_found') + } + if (text(row, 'state') !== 'prepared') { + return { ...cellDrainAttempt(row), shouldSend: false } + } + const assignments = await transaction.queryLocked( + `SELECT assignment.* + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + WHERE migration.source_cell_id = ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ORDER BY assignment.user_id, assignment.relay_host_id`, + [input.cellId] + ) + const activityLeases = await transaction.queryLocked( + `SELECT lease.* + FROM relay_assignment_activity_leases lease + WHERE EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = lease.user_id + AND migration.relay_host_id = lease.relay_host_id + AND migration.source_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ORDER BY lease.user_id, lease.relay_host_id, lease.activity_id`, + [input.cellId] + ) + const migrationIncarnations = await transaction.queryLocked( + `SELECT migration.*, + ( + SELECT incarnation.source_cell_incarnation + FROM relay_assignment_migration_incarnations incarnation + WHERE incarnation.user_id = migration.user_id + AND incarnation.relay_host_id = migration.relay_host_id + AND incarnation.assignment_epoch = migration.assignment_epoch + ) AS source_cell_incarnation, + ( + SELECT incarnation.target_cell_incarnation + FROM relay_assignment_migration_incarnations incarnation + WHERE incarnation.user_id = migration.user_id + AND incarnation.relay_host_id = migration.relay_host_id + AND incarnation.assignment_epoch = migration.assignment_epoch + ) AS target_cell_incarnation + FROM relay_assignment_migrations migration + WHERE migration.source_cell_id = ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ORDER BY migration.user_id, migration.relay_host_id`, + [input.cellId] + ) + if ( + migrationIncarnations.some( + (migration) => + optionalText(migration, 'source_cell_incarnation') !== input.cellIncarnation + ) + ) { + throw new Error('drain_migration_source_incarnation_mismatch') + } + for (const migrationRow of migrationIncarnations) { + const assignment = assignments.find( + (candidate) => + text(candidate, 'user_id') === text(migrationRow, 'user_id') && + text(candidate, 'relay_host_id') === text(migrationRow, 'relay_host_id') + ) + assertCurrentMigrationAssignment(assignment, migrationRow) + assertAssignmentActivityAccounting( + assignment, + activityLeases.filter( + (lease) => + text(lease, 'user_id') === text(migrationRow, 'user_id') && + text(lease, 'relay_host_id') === text(migrationRow, 'relay_host_id') + ), + migrationRow + ) + } + const sendPermitExpiresAt = now + CELL_DRAIN_SEND_PERMIT_MS + await transaction.query( + `UPDATE relay_cell_drain_attempt_states + SET state = 'send-may-have-started', send_may_have_started_at = ?, + send_permit_expires_at = ? + WHERE attempt_id = ?`, + [now, sendPermitExpiresAt, input.attemptId] + ) + await transaction.query( + `INSERT INTO relay_post_drain_migration_pins + (user_id, relay_host_id, assignment_epoch, drain_attempt_id, + source_cell_id, source_cell_incarnation, target_cell_id, + target_cell_incarnation, source_request_units, target_reserved_units, + pinned_at) + SELECT migration.user_id, migration.relay_host_id, + migration.assignment_epoch, ?, migration.source_cell_id, + incarnation.source_cell_incarnation, migration.target_cell_id, + incarnation.target_cell_incarnation, migration.source_request_units, + migration.target_reserved_units, ? + FROM relay_assignment_migrations migration + JOIN relay_assignment_migration_incarnations incarnation + ON incarnation.user_id = migration.user_id + AND incarnation.relay_host_id = migration.relay_host_id + AND incarnation.assignment_epoch = migration.assignment_epoch + WHERE migration.source_cell_id = ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ON CONFLICT (user_id, relay_host_id, assignment_epoch) DO NOTHING`, + [input.attemptId, now, input.cellId] + ) + row.state = 'send-may-have-started' + row.send_may_have_started_at = now + row.send_permit_expires_at = sendPermitExpiresAt + return { ...cellDrainAttempt(row), shouldSend: true } + }) + } + + async recordCellDrainApplicationReceipt(input: { + attemptId: string + cellId: string + cellIncarnation: string + traceValue: string + backendStatus: number + backendInstance?: string + }): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.assertDrainCellGeneration( + transaction, + input.cellId, + input.cellIncarnation + ) + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_drain_attempt_states + WHERE attempt_id = ? AND cell_id = ?`, + [input.attemptId, input.cellId] + ) + )[0] + if ( + !row || + text(row, 'cell_incarnation') !== input.cellIncarnation || + text(row, 'trace_value') !== input.traceValue + ) { + throw new Error('drain_attempt_not_found') + } + if (text(row, 'state') === 'application-receipt') return cellDrainAttempt(row) + if ( + text(row, 'state') !== 'send-may-have-started' || + input.backendStatus < 200 || + input.backendStatus >= 300 + ) { + throw new Error('drain_application_receipt_invalid') + } + const retryAfter = now + integer(row, 'planned_grace_ms') + 30_000 + await transaction.query( + `UPDATE relay_cell_drain_attempt_states + SET state = 'application-receipt', application_receipt_at = ?, + backend_success_status = ?, backend_instance = ?, + receipt_cell_incarnation = ?, retry_after = ? + WHERE attempt_id = ?`, + [ + now, + input.backendStatus, + input.backendInstance, + input.cellIncarnation, + retryAfter, + input.attemptId + ] + ) + row.state = 'application-receipt' + row.application_receipt_at = now + row.backend_success_status = input.backendStatus + row.backend_instance = input.backendInstance ?? null + row.receipt_cell_incarnation = input.cellIncarnation + row.retry_after = retryAfter + return cellDrainAttempt(row) + }) + } + + async proveCellDrainNotDelivered(input: { + attemptId: string + cellId: string + cellIncarnation: string + }): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_drain_attempt_states + WHERE attempt_id = ? AND cell_id = ?`, + [input.attemptId, input.cellId] + ) + )[0] + if (!row || text(row, 'cell_incarnation') !== input.cellIncarnation) { + throw new Error('drain_attempt_not_found') + } + if (text(row, 'state') === 'proven-not-delivered') return cellDrainAttempt(row) + if (text(row, 'state') !== 'send-may-have-started') { + throw new Error('drain_delivery_proof_invalid') + } + await transaction.query( + `UPDATE relay_cell_drain_attempt_states + SET state = 'proven-not-delivered', proven_not_delivered_at = ? + WHERE attempt_id = ?`, + [now, input.attemptId] + ) + row.state = 'proven-not-delivered' + row.proven_not_delivered_at = now + return cellDrainAttempt(row) + }) + } + + async prepareCellDrainRecovery(input: { + attemptId?: string + cellId: string + cellIncarnation: string + }): Promise<{ + shouldSend: boolean + retryAfter: number + preparedAttempt?: CellDrainAttempt + }> { + await this.refreshDrainMigrationLeases(input) + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.assertDrainCellGeneration( + transaction, + input.cellId, + input.cellIncarnation + ) + const attempt = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_drain_attempt_states + WHERE cell_id = ? + ORDER BY prepared_at DESC LIMIT 1`, + [input.cellId] + ) + )[0] + if (!attempt || (input.attemptId && text(attempt, 'attempt_id') !== input.attemptId)) { + throw new Error('drain_application_receipt_missing') + } + if (text(attempt, 'state') === 'prepared') { + if (text(attempt, 'cell_incarnation') !== input.cellIncarnation) { + throw new Error('drain_application_receipt_missing') + } + return { + shouldSend: false, + retryAfter: now, + preparedAttempt: cellDrainAttempt(attempt) + } + } + const applicationReceiptAt = optionalInteger(attempt, 'application_receipt_at') + const backendSuccessStatus = optionalInteger(attempt, 'backend_success_status') + const retryAfter = optionalInteger(attempt, 'retry_after') + if ( + text(attempt, 'state') !== 'application-receipt' || + optionalText(attempt, 'receipt_cell_incarnation') !== + text(attempt, 'cell_incarnation') || + applicationReceiptAt === undefined || + backendSuccessStatus === undefined || + backendSuccessStatus < 200 || + backendSuccessStatus >= 300 || + retryAfter === undefined + ) { + throw new Error('drain_application_receipt_missing') + } + if (now < retryAfter) throw new Error('drain_recovery_too_early') + const attemptId = text(attempt, 'attempt_id') + if (text(attempt, 'cell_incarnation') !== input.cellIncarnation) { + const priorRecovery = ( + await transaction.queryLocked( + `SELECT attempted_at FROM relay_cell_drain_recovery_attempts + WHERE drain_attempt_id = ? AND cell_incarnation = ?`, + [attemptId, input.cellIncarnation] + ) + )[0] + if (priorRecovery) return { shouldSend: false, retryAfter } + await transaction.query( + `INSERT INTO relay_cell_drain_recovery_attempts + (drain_attempt_id, cell_incarnation, attempted_at) VALUES (?, ?, ?)`, + [attemptId, input.cellIncarnation, now] + ) + return { shouldSend: true, retryAfter } + } + if (optionalInteger(attempt, 'recover_forward_attempted_at') !== undefined) { + return { shouldSend: false, retryAfter } + } + await transaction.query( + `UPDATE relay_cell_drain_attempt_states SET recover_forward_attempted_at = ? + WHERE attempt_id = ? AND recover_forward_attempted_at IS NULL`, + [now, attemptId] + ) + return { shouldSend: true, retryAfter } + }) + } + + async evacuateDeadCells(limit = 100): Promise { + if (!this.requireLiveCells) return 0 + const cutoff = this.now() - this.heartbeatTtlMs + const rows = await this.database.query( + `SELECT assignment.user_id, assignment.relay_host_id, assignment.cell_id + FROM relay_assignments assignment + JOIN relay_cells cell ON cell.cell_id = assignment.cell_id + LEFT JOIN relay_cell_committed_fences committed + ON committed.cell_id = assignment.cell_id + LEFT JOIN relay_cell_fence_attempts attempt + ON attempt.attempt_id = committed.attempt_id + LEFT JOIN relay_cell_fences fence ON fence.cell_id = assignment.cell_id + LEFT JOIN relay_cell_runtime runtime ON runtime.cell_id = cell.cell_id + WHERE (runtime.cell_id IS NULL OR runtime.ready != ? OR runtime.last_heartbeat_at <= ?) + AND ( + ( + cell.enabled = 1 + AND NOT EXISTS ( + SELECT 1 FROM relay_cell_connection_limits limits + WHERE limits.cell_id = assignment.cell_id + ) + AND NOT EXISTS ( + SELECT 1 FROM relay_post_drain_migration_pins pin + JOIN relay_assignment_migrations migration + ON migration.user_id = pin.user_id + AND migration.relay_host_id = pin.relay_host_id + AND migration.assignment_epoch = pin.assignment_epoch + WHERE pin.user_id = assignment.user_id + AND pin.relay_host_id = assignment.relay_host_id + AND pin.assignment_epoch = assignment.assignment_epoch + AND pin.target_cell_id = assignment.cell_id + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ) + OR ( + cell.enabled = 0 + AND ( + EXISTS ( + SELECT 1 FROM relay_cell_connection_limits limits + WHERE limits.cell_id = assignment.cell_id + ) + OR EXISTS ( + SELECT 1 FROM relay_post_drain_migration_pins pin + JOIN relay_assignment_migrations migration + ON migration.user_id = pin.user_id + AND migration.relay_host_id = pin.relay_host_id + AND migration.assignment_epoch = pin.assignment_epoch + WHERE pin.user_id = assignment.user_id + AND pin.relay_host_id = assignment.relay_host_id + AND pin.assignment_epoch = assignment.assignment_epoch + AND pin.target_cell_id = assignment.cell_id + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ) + AND attempt.completed_at IS NOT NULL + AND attempt.aborted_at IS NULL + AND committed.cell_incarnation = runtime.cell_incarnation + AND fence.cell_incarnation = committed.cell_incarnation + AND committed.attested_at >= runtime.last_heartbeat_at + AND committed.expires_at > ? + AND fence.expires_at > ? + ) + ) + ORDER BY assignment.user_id, assignment.relay_host_id LIMIT ?`, + [1, cutoff, this.now(), this.now(), limit] + ) + let moved = 0 + for (const row of rows) { + try { + const assignment = await this.assign({ + userId: text(row, 'user_id'), + relayHostId: text(row, 'relay_host_id') + }) + if (assignment.cellId !== text(row, 'cell_id')) moved++ + } catch (error) { + if (!(error instanceof Error && error.message === 'relay_capacity_exhausted')) throw error + } + } + return moved + } + + async configureCell( + cell: RelayCellConfig, + admission: boolean | CellAdmissionState + ): Promise { + const now = this.now() + const state = typeof admission === 'boolean' ? stateFromEnabled(admission) : admission + await this.database.transaction(async (transaction) => { + const cells = await transaction.queryLocked( + `SELECT * FROM relay_cells ORDER BY cell_id ASC` + ) + const current = cells.find((row) => text(row, 'cell_id') === cell.id) + if (current && integer(current, 'reserved_requests') > cell.capacityRequests) { + throw new Error('cell_capacity_below_reserved') + } + // Deploy automation owns tagged URLs; a restarted revision must not overwrite them. + await transaction.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + cell_url = excluded.cell_url, + enabled = excluded.enabled, + capacity_requests = excluded.capacity_requests, + updated_at = excluded.updated_at`, + [ + cell.id, + cell.url, + state === 'existing-only' ? 0 : 1, + cell.capacityRequests, + 0, + 0, + now, + now + ] + ) + await transaction.query( + `INSERT INTO relay_cell_regions (cell_id, region) VALUES (?, ?) + ON CONFLICT (cell_id) DO UPDATE SET region = excluded.region`, + [cell.id, cell.region ?? RELAY_DEFAULT_REGION] + ) + await ensureCellAdmission(transaction, cell.id, state, now) + await setCellAdmissionBeforeBoundary(transaction, cell.id, state, now) + if ( + cell.connectionHardCap !== undefined && + cell.connectionUnobservedBound !== undefined + ) { + const currentLimit = ( + await transaction.queryLocked( + `SELECT hard_cap, unobserved_bound FROM relay_cell_connection_limits + WHERE cell_id = ?`, + [cell.id] + ) + )[0] + const limitChanged = + currentLimit !== undefined && + (integer(currentLimit, 'hard_cap') !== cell.connectionHardCap || + integer(currentLimit, 'unobserved_bound') !== + cell.connectionUnobservedBound) + await transaction.query( + `INSERT INTO relay_cell_connection_limits + (cell_id, hard_cap, unobserved_bound, updated_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + hard_cap = excluded.hard_cap, + unobserved_bound = excluded.unobserved_bound, + updated_at = excluded.updated_at`, + [cell.id, cell.connectionHardCap, cell.connectionUnobservedBound, now] + ) + if (limitChanged) { + await transaction.query( + `DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, + [cell.id] + ) + await transaction.query( + `DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, + [cell.id] + ) + } + } + }) + } + + async startActiveCellEvacuations( + sourceCellId: string, + targetCellId: string, + limit: number + ): Promise { + const rows = await this.database.query( + `SELECT user_id, relay_host_id FROM relay_assignments + WHERE cell_id = ? AND + (reserved_controls > 0 OR reserved_splices > 0 OR reserved_invites > 0 OR + pending_installs > 0 OR pending_confirmations > 0 OR migration_leases > 0) + ORDER BY user_id, relay_host_id LIMIT ?`, + [sourceCellId, limit] + ) + let started = 0 + for (const row of rows) { + const migration = await this.startEvacuation( + { userId: text(row, 'user_id'), relayHostId: text(row, 'relay_host_id') }, + targetCellId, + sourceCellId + ) + if (migration) started++ + } + return started + } + + async cellEvacuationCapacity( + sourceCellId: string, + targetCellId: string + ): Promise { + const cells = await this.database.query( + `SELECT * FROM relay_cells WHERE cell_id IN (?, ?) ORDER BY cell_id`, + [sourceCellId, targetCellId] + ) + if (!cells.some((row) => text(row, 'cell_id') === sourceCellId)) { + throw new Error('source_cell_not_found') + } + const target = cells.find((row) => text(row, 'cell_id') === targetCellId) + if (!target) throw new Error('target_cell_not_found') + if (!(await this.cellIsLive(this.database, targetCellId, this.now()))) { + throw new Error('target_cell_unavailable') + } + const summary = ( + await this.database.query( + `SELECT COUNT(*) AS assignments, + COALESCE(SUM((SELECT COALESCE(SUM(lease.request_units), 0) + FROM relay_assignment_activity_leases lease + WHERE lease.user_id = assignment.user_id + AND lease.relay_host_id = assignment.relay_host_id)), 0) AS source_units + FROM relay_assignments assignment + WHERE assignment.cell_id = ? AND + (assignment.reserved_controls > 0 OR assignment.reserved_splices > 0 OR + assignment.reserved_invites > 0 OR assignment.pending_installs > 0 OR + assignment.pending_confirmations > 0 OR assignment.migration_leases > 0)`, + [sourceCellId] + ) + )[0]! + const sourceAssignments = integer(summary, 'assignments') + return { + sourceAssignments, + // Each migration reserves its future target control plus all source-owned units. + requiredTargetUnits: integer(summary, 'source_units') + sourceAssignments, + availableTargetUnits: + integer(target, 'capacity_requests') - integer(target, 'reserved_requests') + } + } + + async cellEvacuationStatus( + sourceCellId: string, + targetCellId: string, + completeReady: boolean + ): Promise { + let completed = 0 + let blocked = 0 + const sourceFenced = await this.cellHasActiveFence(sourceCellId) + if (completeReady) { + const candidates = await this.activeCellMigrations(sourceCellId, targetCellId) + for (const [index, row] of candidates.entries()) { + try { + const identity = { + userId: text(row, 'user_id'), + relayHostId: text(row, 'relay_host_id') + } + const assignmentEpoch = integer(row, 'assignment_epoch') + if (sourceFenced) { + await this.completeEvacuationFromDeadSource(identity, { + assignmentEpoch, + sourceCellId, + targetCellId + }) + } else { + await this.completeEvacuation(identity, assignmentEpoch) + } + completed++ + } catch (error) { + if (error instanceof Error && error.message === 'migration_cell_inventory_busy') { + blocked += candidates.length - index + break + } + if (isIncompleteMigration(error)) blocked++ + else if (!(error instanceof Error && error.message === 'migration_not_found')) throw error + } + } + } + const active = await this.activeCellMigrations(sourceCellId, targetCellId) + const oldestExpiresAt = + active.length === 0 ? null : Math.min(...active.map((row) => integer(row, 'expires_at'))) + if (completeReady && active.length === 0) { + await this.reconcileReservationAccounting(sourceCellId, targetCellId) + } + const registered = active.filter( + (row) => optionalInteger(row, 'target_registered_at') !== undefined + ) + const registeredSourceActive = registered.filter( + (row) => integer(row, 'source_activity_units') > 0 + ).length + const registeredCompletable = registered.filter( + (row) => + integer(row, 'source_activity_units') === 0 && + integer(row, 'target_control_current') === 1 + ).length + const registeredTargetInactive = registered.filter( + (row) => + integer(row, 'source_activity_units') === 0 && + integer(row, 'target_control_current') === 0 + ).length + const expiredUnregistered = active.filter( + (row) => + optionalInteger(row, 'target_registered_at') === undefined && + integer(row, 'expires_at') <= this.now() + ) + const repairableExpiredUnregistered = expiredUnregistered.filter( + (row) => migrationHasExactActiveTarget(row, targetCellId) + ).length + const abortableExpiredUnregistered = expiredUnregistered.filter( + (row) => integer(row, 'target_control_active') === 0 + ).length + const blockedExpiredUnregistered = + expiredUnregistered.length - + repairableExpiredUnregistered - + abortableExpiredUnregistered + return { + inProgress: active.length, + oldestExpiresAt, + oldestRemainingMs: oldestExpiresAt === null ? null : oldestExpiresAt - this.now(), + targetRegistered: registered.length, + registeredSourceActive, + registeredCompletable, + registeredTargetInactive, + completed, + blocked, + expiredUnregistered: expiredUnregistered.length, + repairableExpiredUnregistered, + abortableExpiredUnregistered, + blockedExpiredUnregistered, + blockedExpiredOnNewerTargetAssignment: expiredUnregistered.filter( + (row) => + integer(row, 'target_control_active') === 1 && + optionalText(row, 'current_cell_id') === targetCellId && + (optionalInteger(row, 'current_assignment_epoch') ?? 0) > + integer(row, 'assignment_epoch') + ).length + } + } + + async completeReadyEvacuations(limit = 100): Promise { + if (!Number.isSafeInteger(limit) || limit < 1 || limit > 100) { + throw new Error('invalid_evacuation_completion_limit') + } + const now = this.now() + const runtimeSafety = this.requireLiveCells + ? `AND EXISTS ( + SELECT 1 FROM relay_cells source_cell + WHERE source_cell.cell_id = migration.source_cell_id + AND source_cell.enabled = 0 + ) + AND EXISTS ( + SELECT 1 FROM relay_cells target_cell + WHERE target_cell.cell_id = migration.target_cell_id + AND target_cell.enabled = 1 + ) + AND EXISTS ( + SELECT 1 FROM relay_cell_runtime source_runtime + WHERE source_runtime.cell_id = migration.source_cell_id + AND source_runtime.ready = 1 + AND source_runtime.observed_requests = 0 + AND source_runtime.last_heartbeat_at > ? + ) + AND EXISTS ( + SELECT 1 FROM relay_cell_runtime target_runtime + WHERE target_runtime.cell_id = migration.target_cell_id + AND target_runtime.ready = 1 + AND target_runtime.last_heartbeat_at > ? + )` + : '' + const targetIncarnation = this.requireLiveCells + ? `AND target_control.updated_at >= ( + SELECT target_runtime.started_at FROM relay_cell_runtime target_runtime + WHERE target_runtime.cell_id = migration.target_cell_id + )` + : '' + // Selection avoids polling offline desktops; completeEvacuation rechecks + // current target activity and zero source ownership under row locks. + const candidates = await this.database.query( + `SELECT migration.user_id, migration.relay_host_id, migration.source_cell_id, + migration.target_cell_id, migration.assignment_epoch + FROM relay_assignment_migrations migration + WHERE migration.target_registered_at IS NOT NULL + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ${runtimeSafety} + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases source_lease + WHERE source_lease.user_id = migration.user_id + AND source_lease.relay_host_id = migration.relay_host_id + AND source_lease.cell_id = migration.source_cell_id + ) + AND EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases target_control + WHERE target_control.user_id = migration.user_id + AND target_control.relay_host_id = migration.relay_host_id + AND target_control.cell_id = migration.target_cell_id + AND target_control.activity_kind = 'control' + AND target_control.activity_id NOT LIKE 'control-pending:%' + AND target_control.expires_at > ? + ${targetIncarnation} + ) + ORDER BY migration.user_id, migration.relay_host_id + LIMIT ?`, + [ + ...(this.requireLiveCells + ? [now - this.heartbeatTtlMs, now - this.heartbeatTtlMs] + : []), + now, + limit + ] + ) + const pairs = new Map() + let completed = 0 + for (const row of candidates) { + const sourceCellId = text(row, 'source_cell_id') + const targetCellId = text(row, 'target_cell_id') + pairs.set(JSON.stringify([sourceCellId, targetCellId]), { sourceCellId, targetCellId }) + try { + await this.completeEvacuation( + { userId: text(row, 'user_id'), relayHostId: text(row, 'relay_host_id') }, + integer(row, 'assignment_epoch') + ) + completed++ + } catch (error) { + if (!isIncompleteMigration(error) && !isMissingMigration(error)) throw error + } + } + for (const { sourceCellId, targetCellId } of pairs.values()) { + if ((await this.activeCellMigrations(sourceCellId, targetCellId)).length === 0) { + await this.reconcileReservationAccounting(sourceCellId, targetCellId) + } + } + return completed + } + + async cellDeploymentStatus(cellId: string): Promise { + const cellRow = ( + await this.database.query( + `WITH activity AS ( + SELECT COUNT(*) AS activity_lease_count, + COALESCE(SUM(request_units), 0) AS activity_request_units, + COALESCE(SUM(CASE WHEN activity_kind = 'control' + AND activity_id LIKE 'control-pending:%' + AND request_units = 1 + THEN 1 ELSE 0 END), 0) AS pending_control_units + FROM relay_assignment_activity_leases WHERE cell_id = ? + ) + SELECT cell.*, runtime.cell_url AS runtime_cell_url, + admission.admission_state, + runtime.cell_incarnation AS runtime_cell_incarnation, + runtime.started_at AS runtime_started_at, runtime.ready AS runtime_ready, + runtime.observed_requests AS runtime_observed_requests, + runtime.last_heartbeat_at AS runtime_last_heartbeat_at, + COALESCE(capabilities.regional_rehome_protocol, 0) + AS runtime_regional_rehome_protocol, + activity.activity_lease_count, activity.activity_request_units, + activity.activity_lease_count - activity.pending_control_units + AS restart_blocking_activity_leases, + activity.activity_request_units - activity.pending_control_units + AS restart_blocking_activity_request_units, + cell.reserved_requests - activity.pending_control_units + AS restart_blocking_reserved_requests + FROM relay_cells cell + CROSS JOIN activity + LEFT JOIN relay_cell_admission admission ON admission.cell_id = cell.cell_id + LEFT JOIN relay_cell_runtime runtime ON runtime.cell_id = cell.cell_id + LEFT JOIN relay_cell_capabilities capabilities + ON capabilities.cell_id = runtime.cell_id + AND capabilities.cell_incarnation = runtime.cell_incarnation + WHERE cell.cell_id = ?`, + [cellId, cellId] + ) + )[0] + if (!cellRow) throw new Error('cell_not_found') + const assignmentRow = ( + await this.database.query( + `SELECT COUNT(*) AS assignment_count FROM relay_assignments WHERE cell_id = ?`, + [cellId] + ) + )[0]! + const migrationRow = ( + await this.database.query( + `SELECT + COALESCE(SUM(CASE WHEN source_cell_id = ? THEN 1 ELSE 0 END), 0) + AS outgoing_migrations, + COALESCE(SUM(CASE WHEN target_cell_id = ? THEN 1 ELSE 0 END), 0) + AS incoming_migrations + FROM relay_assignment_migrations + WHERE completed_at IS NULL AND aborted_at IS NULL + AND (source_cell_id = ? OR target_cell_id = ?)`, + [cellId, cellId, cellId, cellId] + ) + )[0]! + const connectionRow = ( + await this.database.query( + `SELECT limits.hard_cap, limits.unobserved_bound, + connection_runtime.total_connections, connection_runtime.in_flight_connections, + connection_runtime.reserved_connection_units, + connection_runtime.enforced_connection_units, + connection_runtime.last_heartbeat_at, + current_runtime.cell_incarnation AS current_incarnation, + connection_runtime.cell_incarnation AS connection_incarnation, + (SELECT COUNT(*) FROM relay_control_connection_reservations reservation + WHERE reservation.cell_id = limits.cell_id + AND reservation.state IN + ('reserved', 'late-arrival-debt', 'claimed')) AS outstanding_reservations + FROM relay_cell_connection_limits limits + LEFT JOIN relay_cell_connection_runtime connection_runtime + ON connection_runtime.cell_id = limits.cell_id + LEFT JOIN relay_cell_runtime current_runtime + ON current_runtime.cell_id = limits.cell_id + WHERE limits.cell_id = ?`, + [cellId] + ) + )[0] + const runtimeHeartbeat = optionalInteger(cellRow, 'runtime_last_heartbeat_at') + const connectionHeartbeat = connectionRow + ? optionalInteger(connectionRow, 'last_heartbeat_at') + : undefined + return { + cellId, + cellUrl: text(cellRow, 'cell_url'), + region: await this.cellRegion(this.database, cellId), + enabled: integer(cellRow, 'enabled') === 1, + admissionState: parseCellAdmissionState(text(cellRow, 'admission_state')), + capacityRequests: integer(cellRow, 'capacity_requests'), + reservedRequests: integer(cellRow, 'reserved_requests'), + assignments: integer(assignmentRow, 'assignment_count'), + activityLeases: integer(cellRow, 'activity_lease_count'), + activityRequestUnits: integer(cellRow, 'activity_request_units'), + restartBlockingActivityLeases: integer( + cellRow, + 'restart_blocking_activity_leases' + ), + restartBlockingActivityRequestUnits: integer( + cellRow, + 'restart_blocking_activity_request_units' + ), + restartBlockingReservedRequests: integer( + cellRow, + 'restart_blocking_reserved_requests' + ), + outgoingMigrations: integer(migrationRow, 'outgoing_migrations'), + incomingMigrations: integer(migrationRow, 'incoming_migrations'), + connectionCapacity: connectionRow + ? { + hardCap: integer(connectionRow, 'hard_cap'), + controlRebindReserve: RELAY_ADMISSION_BUDGETS.reservedHostControls, + ordinaryConnectionLimit: + integer(connectionRow, 'hard_cap') - + RELAY_ADMISSION_BUDGETS.reservedHostControls, + unobservedBound: integer(connectionRow, 'unobserved_bound'), + normalAdmissionPause: + integer(connectionRow, 'hard_cap') - + RELAY_ADMISSION_BUDGETS.reservedHostControls - + integer(connectionRow, 'unobserved_bound'), + observedConnections: + optionalInteger(connectionRow, 'total_connections') ?? 0, + inFlightConnections: + optionalInteger(connectionRow, 'in_flight_connections') ?? 0, + reservedConnectionUnits: + optionalInteger(connectionRow, 'reserved_connection_units') ?? 0, + enforcedConnectionUnits: + optionalInteger(connectionRow, 'enforced_connection_units') ?? 0, + pendingControlReservations: integer( + connectionRow, + 'outstanding_reservations' + ), + heartbeatFresh: + connectionHeartbeat !== undefined && + connectionHeartbeat > this.now() - this.heartbeatTtlMs && + optionalText(connectionRow, 'current_incarnation') === + optionalText(connectionRow, 'connection_incarnation') + } + : null, + runtime: + runtimeHeartbeat === undefined + ? null + : { + cellUrl: text(cellRow, 'runtime_cell_url'), + cellIncarnation: text(cellRow, 'runtime_cell_incarnation'), + startedAt: integer(cellRow, 'runtime_started_at'), + ready: integer(cellRow, 'runtime_ready') === 1, + observedRequests: integer(cellRow, 'runtime_observed_requests'), + lastHeartbeatAt: runtimeHeartbeat, + heartbeatFresh: runtimeHeartbeat > this.now() - this.heartbeatTtlMs, + regionalRehomeProtocol: integer( + cellRow, + 'runtime_regional_rehome_protocol' + ) + } + } + } + + async verifyCellAssignment(input: AssignmentIdentity & { + cellId: string + assignmentEpoch: number + }): Promise { + const rows = await this.database.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [input.userId, input.relayHostId] + ) + return Boolean( + rows[0] && + text(rows[0], 'cell_id') === input.cellId && + integer(rows[0], 'assignment_epoch') === input.assignmentEpoch + ) + } + + async changeActivity( + identity: AssignmentIdentity, + kind: AssignmentActivityKind, + delta: 1 | -1 + ): Promise { + await this.activityQueue.run(identity, async () => { + const column = ACTIVITY_COLUMN[kind] + const now = this.now() + await this.database.transaction(async (transaction) => { + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0] + if (!row) return + const before = integer(row, column) + const after = Math.max(0, before + delta) + await transaction.query( + `UPDATE relay_assignments SET ${column} = ?, lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [after, now + ASSIGNMENT_LIMITS.activityLeaseMs, now, identity.userId, identity.relayHostId] + ) + const requestDelta = ACTIVITY_REQUEST_UNITS[kind] * (after - before) + if (requestDelta !== 0) { + await this.lockCellInventory(transaction) + await this.adjustCellReservation(transaction, text(row, 'cell_id'), requestDelta) + } + }) + }) + } + + async acquireActivity( + identity: AssignmentIdentity, + input: { + activityId: string + kind: AssignmentActivityKind + cellId: string + expiresAt?: number + } + ): Promise { + validateActivityId(input.activityId) + await this.activityQueue.run(identity, async () => { + const now = this.now() + await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + if (!assignment) throw new Error('assignment_not_found') + const assignmentCellId = text(assignment, 'cell_id') + if (input.cellId !== assignmentCellId) { + // Origin controls may renew work only while the exact forward + // migration is active; completion must fence late source activity. + const migration = ( + await transaction.queryLocked( + `SELECT assignment_epoch FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? + AND source_cell_id = ? AND target_cell_id = ? + AND assignment_epoch = ? + AND completed_at IS NULL AND aborted_at IS NULL`, + [ + identity.userId, + identity.relayHostId, + input.cellId, + assignmentCellId, + integer(assignment, 'assignment_epoch') + ] + ) + )[0] + if (!migration) throw new Error('activity_cell_not_authoritative') + } + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + const existing = activityLeaseById(activityLeases, input.activityId) + const expiresAt = input.expiresAt ?? now + ASSIGNMENT_LIMITS.activityLeaseMs + if (!Number.isSafeInteger(expiresAt) || expiresAt <= now) { + throw new Error('invalid_activity_expiry') + } + if (existing && text(existing, 'activity_kind') === input.kind && text(existing, 'cell_id') === input.cellId) { + await transaction.query( + `UPDATE relay_assignment_activity_leases SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [expiresAt, now, identity.userId, identity.relayHostId, input.activityId] + ) + await this.touchAssignment(transaction, identity, expiresAt, now) + return + } + const units = ACTIVITY_REQUEST_UNITS[input.kind] + if (existing) { + await this.lockCellInventory(transaction) + await this.removeActivityLease(transaction, identity, existing, now) + await this.adjustCellReservation(transaction, input.cellId, units) + } + await this.adjustActivityCount(transaction, identity, input.kind, 1, expiresAt, now) + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + input.activityId, + input.kind, + input.cellId, + units, + expiresAt, + now + ] + ) + // Keep the contended cell row locked for only the final write and commit. + if (!existing) { + await this.adjustCellReservationAtomically(transaction, input.cellId, units) + } + }) + }) + } + + async renewControlActivity( + identity: AssignmentIdentity, + input: { activityId: string; cellId: string; expiresAt: number } + ): Promise { + validateActivityId(input.activityId) + const now = this.now() + const maximumExpiresAt = + now + + ASSIGNMENT_LIMITS.activityLeaseMs + + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2 + if ( + !Number.isSafeInteger(input.expiresAt) || + input.expiresAt <= now || + input.expiresAt > maximumExpiresAt + ) { + throw new Error('invalid_activity_expiry') + } + const startedAt = performance.now() + let outcome: ControlRenewalOutcome = 'database_error' + try { + outcome = + this.database.dialect === 'postgres' + ? await this.renewPostgresControlActivity(identity, input, now) + : await this.renewTransactionalControlActivity(identity, input, now) + if (outcome !== 'renewed') throw new Error(outcome) + } catch (error) { + const message = String((error as { message?: unknown }).message) + if (CONTROL_RENEWAL_OUTCOMES.has(message as ControlRenewalOutcome)) { + outcome = message as ControlRenewalOutcome + } + throw error + } finally { + this.recordControlRenewal?.(performance.now() - startedAt, outcome) + } + } + + private async renewPostgresControlActivity( + identity: AssignmentIdentity, + input: { activityId: string; cellId: string; expiresAt: number }, + now: number + ): Promise { + const row = ( + await this.database.query( + `WITH assignment_state AS MATERIALIZED ( + SELECT cell_id, assignment_epoch + FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ? + FOR UPDATE + ), migration_state AS MATERIALIZED ( + SELECT migration.assignment_epoch + FROM relay_assignment_migrations migration + JOIN assignment_state assignment + ON migration.target_cell_id = assignment.cell_id + AND migration.assignment_epoch = assignment.assignment_epoch + WHERE migration.user_id = ? AND migration.relay_host_id = ? + AND migration.source_cell_id = ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + FOR UPDATE OF migration + ), authorization_state AS MATERIALIZED ( + SELECT 1 AS authorized + FROM assignment_state assignment + WHERE assignment.cell_id = ? OR EXISTS (SELECT 1 FROM migration_state) + ), lease_state AS MATERIALIZED ( + SELECT lease.activity_kind, lease.cell_id + FROM relay_assignment_activity_leases lease + CROSS JOIN authorization_state + WHERE lease.user_id = ? AND lease.relay_host_id = ? AND lease.activity_id = ? + FOR UPDATE OF lease + ), renewed_lease AS ( + UPDATE relay_assignment_activity_leases lease + SET expires_at = GREATEST(lease.expires_at, ?), + updated_at = GREATEST(lease.updated_at, ?) + FROM lease_state state + WHERE lease.user_id = ? AND lease.relay_host_id = ? AND lease.activity_id = ? + AND state.activity_kind = 'control' AND state.cell_id = ? + RETURNING 1 + ), renewed_assignment AS ( + UPDATE relay_assignments assignment + SET lease_expires_at = GREATEST(assignment.lease_expires_at, ?), + last_activity_at = GREATEST(assignment.last_activity_at, ?) + WHERE assignment.user_id = ? AND assignment.relay_host_id = ? + AND EXISTS (SELECT 1 FROM renewed_lease) + RETURNING 1 + ) + SELECT CASE + WHEN NOT EXISTS (SELECT 1 FROM assignment_state) + THEN 'assignment_not_found' + WHEN NOT EXISTS (SELECT 1 FROM authorization_state) + THEN 'activity_cell_not_authoritative' + WHEN NOT EXISTS (SELECT 1 FROM lease_state) + THEN 'control_activity_not_found' + WHEN EXISTS ( + SELECT 1 FROM lease_state + WHERE activity_kind <> 'control' OR cell_id <> ? + ) THEN 'control_activity_moved' + WHEN EXISTS (SELECT 1 FROM renewed_assignment) THEN 'renewed' + ELSE 'control_activity_not_found' + END AS outcome`, + [ + identity.userId, + identity.relayHostId, + identity.userId, + identity.relayHostId, + input.cellId, + input.cellId, + identity.userId, + identity.relayHostId, + input.activityId, + input.expiresAt, + now, + identity.userId, + identity.relayHostId, + input.activityId, + input.cellId, + input.expiresAt, + now, + identity.userId, + identity.relayHostId, + input.cellId + ] + ) + )[0] + if (!row) throw new Error('missing_control_renewal_outcome') + const outcome = text(row, 'outcome') as ControlRenewalOutcome + if (!CONTROL_RENEWAL_OUTCOMES.has(outcome)) throw new Error('invalid_control_renewal_outcome') + return outcome + } + + private async renewTransactionalControlActivity( + identity: AssignmentIdentity, + input: { activityId: string; cellId: string; expiresAt: number }, + now: number + ): Promise { + // Bypass the process queue so a network-stalled activity call cannot suppress renewal. + await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + if (!assignment) throw new Error('assignment_not_found') + const assignmentCellId = text(assignment, 'cell_id') + if (input.cellId !== assignmentCellId) { + const migration = ( + await transaction.queryLocked( + `SELECT assignment_epoch FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? + AND source_cell_id = ? AND target_cell_id = ? + AND assignment_epoch = ? + AND completed_at IS NULL AND aborted_at IS NULL`, + [ + identity.userId, + identity.relayHostId, + input.cellId, + assignmentCellId, + integer(assignment, 'assignment_epoch') + ] + ) + )[0] + if (!migration) throw new Error('activity_cell_not_authoritative') + } + const lease = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, input.activityId] + ) + )[0] + if (!lease) throw new Error('control_activity_not_found') + if ( + text(lease, 'activity_kind') !== 'control' || + text(lease, 'cell_id') !== input.cellId + ) { + throw new Error('control_activity_moved') + } + await transaction.query( + `UPDATE relay_assignment_activity_leases SET + expires_at = CASE WHEN expires_at > ? THEN expires_at ELSE ? END, + updated_at = CASE WHEN updated_at > ? THEN updated_at ELSE ? END + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [ + input.expiresAt, + input.expiresAt, + now, + now, + identity.userId, + identity.relayHostId, + input.activityId + ] + ) + await transaction.query( + `UPDATE relay_assignments SET + lease_expires_at = CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = CASE WHEN last_activity_at > ? THEN last_activity_at ELSE ? END + WHERE user_id = ? AND relay_host_id = ?`, + [input.expiresAt, input.expiresAt, now, now, identity.userId, identity.relayHostId] + ) + }) + return 'renewed' + } + + async releaseActivity(identity: AssignmentIdentity, activityId: string): Promise { + validateActivityId(activityId) + return await this.activityQueue.run(identity, async () => { + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.assignmentRow(transaction, identity) + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + const existing = activityLeaseById(activityLeases, activityId) + if (!existing) return false + await transaction.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, activityId] + ) + await this.adjustActivityCount( + transaction, + identity, + activityKind(existing), + -1, + now, + now + ) + // Release paths can safely defer the cell-row lock until their final write. + await this.adjustCellReservationAtomically( + transaction, + text(existing, 'cell_id'), + -integer(existing, 'request_units') + ) + return true + }) + }) + } + + async activateControl( + identity: AssignmentIdentity, + input: { + cellId: string + assignmentEpoch: number + generation: number + connectionInclusionWatermark?: number + } + ): Promise { + const activityId = `control:${input.cellId}:${input.generation}` + validateActivityId(activityId) + return await this.activityQueue.run(identity, async () => { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + if ( + !assignment || + text(assignment, 'cell_id') !== input.cellId || + integer(assignment, 'assignment_epoch') !== input.assignmentEpoch + ) { + throw new Error('wrong_assignment') + } + const expiresAt = now + ASSIGNMENT_LIMITS.activityLeaseMs + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + const existing = activityLeaseById(activityLeases, activityId) + const pendingId = pendingControlActivityId(input.assignmentEpoch) + const pending = activityLeaseById(activityLeases, pendingId) + const retainedActivityId = + existing || (pending && text(pending, 'cell_id') === input.cellId) + ? existing + ? activityId + : pendingId + : activityId + await this.removeSupersededSameCellControls( + transaction, + identity, + activityLeases, + input.cellId, + retainedActivityId, + now + ) + if (existing) { + await transaction.query( + `UPDATE relay_assignment_activity_leases SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [expiresAt, now, identity.userId, identity.relayHostId, activityId] + ) + await this.touchAssignment(transaction, identity, expiresAt, now) + } else { + if (pending && text(pending, 'cell_id') === input.cellId) { + await transaction.query( + `UPDATE relay_assignment_activity_leases + SET activity_id = ?, expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [activityId, expiresAt, now, identity.userId, identity.relayHostId, pendingId] + ) + await this.touchAssignment(transaction, identity, expiresAt, now) + } else { + await this.lockCellInventory(transaction) + await this.adjustCellReservation(transaction, input.cellId, 1) + await this.adjustActivityCount(transaction, identity, 'control', 1, expiresAt, now) + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + activityId, + 'control', + input.cellId, + 1, + expiresAt, + now + ] + ) + } + } + await this.claimControlConnectionReservation( + transaction, + identity, + input.cellId, + input.assignmentEpoch, + activityId, + input.connectionInclusionWatermark, + now + ) + return activityId + }) + }) + } + + async startEvacuation( + identity: AssignmentIdentity, + targetCellId: string + ): Promise + async startEvacuation( + identity: AssignmentIdentity, + targetCellId: string, + expectedSourceCellId: string + ): Promise + async startEvacuation( + identity: AssignmentIdentity, + targetCellId: string, + expectedSourceCellId?: string + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + if (!assignment) throw new Error('assignment_not_found') + const sourceCellId = text(assignment, 'cell_id') + // Bulk selection is intentionally staleable; fence it before considering + // an existing migration that already moved the assignment to its target. + if (expectedSourceCellId && sourceCellId !== expectedSourceCellId) return null + const existing = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND completed_at IS NULL + AND aborted_at IS NULL AND expires_at > ? + ORDER BY assignment_epoch DESC LIMIT 1`, + [identity.userId, identity.relayHostId, now] + ) + )[0] + if (existing) { + if (text(existing, 'target_cell_id') !== targetCellId) { + throw new Error('migration_in_progress') + } + return migration(identity, existing) + } + if (sourceCellId === targetCellId) throw new Error('target_matches_source') + await this.lockAssignmentActivities(transaction, identity) + const cells = await this.lockCellInventory(transaction) + const target = cells.find((row) => text(row, 'cell_id') === targetCellId) + if (!target || integer(target, 'enabled') !== 1) throw new Error('target_cell_unavailable') + if (!(await this.cellIsLive(transaction, targetCellId, now))) { + throw new Error('target_cell_unavailable') + } + let sourceCellIncarnation: string | undefined + let targetCellIncarnation: string | undefined + if (this.requireLiveCells) { + const runtimes = await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime WHERE cell_id IN (?, ?) ORDER BY cell_id`, + [sourceCellId, targetCellId] + ) + const sourceRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === sourceCellId + ) + const targetRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === targetCellId + ) + if ( + !sourceRuntime || + integer(sourceRuntime, 'ready') !== 1 || + integer(sourceRuntime, 'last_heartbeat_at') <= now - this.heartbeatTtlMs + ) { + throw new Error('source_cell_unavailable') + } + if ( + !targetRuntime || + integer(targetRuntime, 'ready') !== 1 || + integer(targetRuntime, 'last_heartbeat_at') <= now - this.heartbeatTtlMs + ) { + throw new Error('target_cell_unavailable') + } + sourceCellIncarnation = text(sourceRuntime, 'cell_incarnation') + targetCellIncarnation = text(targetRuntime, 'cell_incarnation') + } + await this.assertCellConnectionHeadroom(transaction, targetCellId) + const sourceUnitsRow = ( + await transaction.query( + `SELECT COALESCE(SUM(request_units), 0) AS units + FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND cell_id = ?`, + [identity.userId, identity.relayHostId, sourceCellId] + ) + )[0] + const sourceRequestUnits = integer(sourceUnitsRow!, 'units') + const targetReservedUnits = sourceRequestUnits + 1 + await this.adjustCellReservation(transaction, targetCellId, targetReservedUnits) + const previousEpoch = integer(assignment, 'assignment_epoch') + const assignmentEpoch = previousEpoch + 1 + const expiresAt = now + ASSIGNMENT_LIMITS.migrationLeaseMs + await transaction.query( + `UPDATE relay_assignments SET cell_id = ?, assignment_epoch = ?, + reserved_controls = reserved_controls + 1, + migration_leases = migration_leases + 1, + lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + targetCellId, + assignmentEpoch, + expiresAt, + now, + identity.userId, + identity.relayHostId + ] + ) + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?), (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + 'control', + targetCellId, + 1, + expiresAt, + now, + identity.userId, + identity.relayHostId, + migrationActivityId(assignmentEpoch), + 'migration', + targetCellId, + sourceRequestUnits, + expiresAt, + now + ] + ) + await this.insertControlConnectionReservation( + transaction, + identity, + targetCellId, + assignmentEpoch, + expiresAt, + now + ) + await transaction.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, NULL, NULL, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + sourceCellId, + targetCellId, + previousEpoch, + assignmentEpoch, + sourceRequestUnits, + targetReservedUnits, + expiresAt, + now, + now + ] + ) + if (sourceCellIncarnation && targetCellIncarnation) { + await transaction.query( + `INSERT INTO relay_assignment_migration_incarnations + (user_id, relay_host_id, assignment_epoch, source_cell_incarnation, + target_cell_incarnation) + VALUES (?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + sourceCellIncarnation, + targetCellIncarnation + ] + ) + } + return { + ...identity, + sourceCellId, + targetCellId, + previousEpoch, + assignmentEpoch, + expiresAt + } + }) + } + + async markMigrationTargetRegistered( + identity: AssignmentIdentity, + input: { cellId: string; assignmentEpoch: number } + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const rows = await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND target_cell_id = ? AND completed_at IS NULL AND aborted_at IS NULL`, + [ + identity.userId, + identity.relayHostId, + input.assignmentEpoch, + input.cellId + ] + ) + if (!rows[0]) return false + await transaction.query( + `UPDATE relay_assignment_migrations + SET target_registered_at = COALESCE(target_registered_at, ?), updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + return true + }) + } + + async completeEvacuationFromDeadSource( + identity: AssignmentIdentity, + input: { + assignmentEpoch: number + sourceCellId: string + targetCellId: string + } + ): Promise { + try { + return await this.withAssignmentLockRetry( + async (inventoryFirst) => + await this.completeEvacuationFromDeadSourceOnce(identity, input, inventoryFirst) + ) + } catch (error) { + if (isDatabaseLockUnavailable(error)) { + throw new Error('migration_cell_inventory_busy') + } + throw error + } + } + + private async completeEvacuationFromDeadSourceOnce( + identity: AssignmentIdentity, + input: { + assignmentEpoch: number + sourceCellId: string + targetCellId: string + }, + inventoryFirst: boolean + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + let lockedCells: SqlRow[] | undefined + if (inventoryFirst) { + try { + lockedCells = await this.lockCellInventory(transaction) + } catch (error) { + if (isDatabaseLockTimeout(error)) { + throw new Error('database_lock_unavailable') + } + throw error + } + } + const assignment = await this.assignmentRow(transaction, identity, inventoryFirst) + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + if (!row) throw new Error('migration_not_found') + assertMigrationPair(row, input.sourceCellId, input.targetCellId) + if (optionalInteger(row, 'completed_at') !== undefined) { + return deadSourceCompletionResult(row, false) + } + if (optionalInteger(row, 'aborted_at') !== undefined) { + throw new Error('migration_already_aborted') + } + if (optionalInteger(row, 'target_registered_at') === undefined) { + throw new Error('migration_target_not_registered') + } + assertCurrentMigrationAssignment(assignment, row) + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + assertAssignmentActivityAccounting(assignment, activityLeases, row) + if ( + activityLeases.some( + (lease) => + ![input.sourceCellId, input.targetCellId].includes(text(lease, 'cell_id')) + ) + ) { + throw new Error('migration_activity_topology_mismatch') + } + if (activityUnitsForCell(activityLeases, input.sourceCellId) > 0) { + throw new Error('migration_source_still_active') + } + const cells = lockedCells ?? (await this.lockCellInventory(transaction, true)) + const source = cells.find((cell) => text(cell, 'cell_id') === input.sourceCellId) + const target = cells.find((cell) => text(cell, 'cell_id') === input.targetCellId) + if (!source || integer(source, 'enabled') !== 0) { + throw new Error('migration_source_admission_changed') + } + if (!target || integer(target, 'enabled') !== 1) { + throw new Error('migration_target_admission_changed') + } + await assertCellReservationAccounting(transaction, cells, [ + input.sourceCellId, + input.targetCellId + ]) + const runtimes = await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime WHERE cell_id IN (?, ?) ORDER BY cell_id`, + [input.sourceCellId, input.targetCellId] + ) + const sourceRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === input.sourceCellId + ) + const targetRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === input.targetCellId + ) + const freshAfter = now - this.heartbeatTtlMs + await this.requireCellFence(transaction, input.sourceCellId, sourceRuntime, now) + if (!sourceRuntime || integer(sourceRuntime, 'last_heartbeat_at') > freshAfter) { + throw new Error('migration_source_runtime_not_dead') + } + if ( + !targetRuntime || + integer(targetRuntime, 'ready') !== 1 || + integer(targetRuntime, 'last_heartbeat_at') <= freshAfter + ) { + throw new Error('migration_target_runtime_not_ready') + } + const targetIsActive = activityLeases.some( + (lease) => + text(lease, 'cell_id') === input.targetCellId && + text(lease, 'activity_kind') === 'control' && + !text(lease, 'activity_id').startsWith('control-pending:') && + integer(lease, 'expires_at') > now && + integer(lease, 'updated_at') >= integer(targetRuntime, 'started_at') + ) + if (!targetIsActive) { + const inactiveTargetControls = activityLeases.filter( + (lease) => + text(lease, 'cell_id') === input.targetCellId && + text(lease, 'activity_kind') === 'control' + ) + for (const lease of inactiveTargetControls) { + await this.removeActivityLease(transaction, identity, lease, now) + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + input.targetCellId, + input.assignmentEpoch, + now + ) + } + const migrationLease = activityLeaseById( + activityLeases, + migrationActivityId(input.assignmentEpoch) + ) + if (migrationLease) { + await this.removeActivityLease(transaction, identity, migrationLease, now) + } + await transaction.query( + `UPDATE relay_assignment_migrations SET completed_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + return deadSourceCompletionResult(row, true) + }) + } + + async supersedeRegisteredEvacuation( + identity: AssignmentIdentity, + input: RegisteredEvacuationSupersessionInput, + preservedActivityCellIds: readonly string[] = [] + ): Promise { + if (input.assignmentEpoch >= Number.MAX_SAFE_INTEGER) { + throw new Error('assignment_epoch_exhausted') + } + return await this.withAssignmentLockRetry( + async (inventoryFirst) => + await this.supersedeRegisteredEvacuationOnce( + identity, + input, + inventoryFirst, + preservedActivityCellIds + ) + ) + } + + private async supersedeRegisteredEvacuationOnce( + identity: AssignmentIdentity, + input: RegisteredEvacuationSupersessionInput, + inventoryFirst: boolean, + preservedActivityCellIds: readonly string[] + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const lockedCells = inventoryFirst + ? await this.lockCellInventory(transaction) + : undefined + const assignment = await this.assignmentRow(transaction, identity, inventoryFirst) + const existing = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + if (!existing) throw new Error('migration_not_found') + assertMigrationPair(existing, input.sourceCellId, input.currentTargetCellId) + if (optionalInteger(existing, 'completed_at') !== undefined) { + throw new Error('migration_already_completed') + } + if (optionalInteger(existing, 'aborted_at') !== undefined) { + const successor = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND previous_epoch = ? + AND source_cell_id = ? AND target_cell_id = ? + ORDER BY assignment_epoch DESC LIMIT 1`, + [ + identity.userId, + identity.relayHostId, + input.assignmentEpoch, + input.sourceCellId, + input.replacementTargetCellId + ] + ) + )[0] + if (!successor) throw new Error('migration_already_superseded') + return migration(identity, successor) + } + if (optionalInteger(existing, 'target_registered_at') === undefined) { + throw new Error('migration_target_not_registered') + } + const existingIncarnation = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migration_incarnations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + const existingPin = ( + await transaction.queryLocked( + `SELECT * FROM relay_post_drain_migration_pins + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + if (existingPin && !existingIncarnation) { + throw new Error('drain_migration_source_incarnation_mismatch') + } + assertCurrentMigrationAssignment(assignment, existing) + if (input.currentTargetCellId === input.replacementTargetCellId) { + throw new Error('target_matches_source') + } + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + assertAssignmentActivityAccounting(assignment, activityLeases, existing) + const additionalActivityCellIds = new Set( + activityLeases + .map((lease) => text(lease, 'cell_id')) + .filter( + (cellId) => + ![input.sourceCellId, input.currentTargetCellId].includes(cellId) + ) + ) + if ( + preservedActivityCellIds.includes(input.replacementTargetCellId) || + !matchesExactCellSet(additionalActivityCellIds, preservedActivityCellIds) + ) { + throw new Error('migration_activity_topology_mismatch') + } + const cells = lockedCells ?? (await this.lockCellInventory(transaction, true)) + const source = cells.find((cell) => text(cell, 'cell_id') === input.sourceCellId) + const currentTarget = cells.find( + (cell) => text(cell, 'cell_id') === input.currentTargetCellId + ) + const replacement = cells.find( + (cell) => text(cell, 'cell_id') === input.replacementTargetCellId + ) + if (!source || integer(source, 'enabled') !== 0) { + throw new Error('migration_source_admission_changed') + } + if (!currentTarget || integer(currentTarget, 'enabled') !== 0) { + throw new Error('migration_target_still_enabled') + } + if (!replacement || integer(replacement, 'enabled') !== 1) { + throw new Error('replacement_target_unavailable') + } + await assertCellReservationAccounting(transaction, cells, [ + input.sourceCellId, + input.currentTargetCellId, + input.replacementTargetCellId, + ...preservedActivityCellIds + ]) + const runtimeCellIds = [ + input.currentTargetCellId, + input.replacementTargetCellId, + ...preservedActivityCellIds + ] + const runtimes = await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime + WHERE cell_id IN (${runtimeCellIds.map(() => '?').join(', ')}) + ORDER BY cell_id`, + runtimeCellIds + ) + const currentRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === input.currentTargetCellId + ) + const replacementRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === input.replacementTargetCellId + ) + const freshAfter = now - this.heartbeatTtlMs + await this.requireCellFence( + transaction, + input.currentTargetCellId, + currentRuntime, + now + ) + if ( + currentRuntime && + integer(currentRuntime, 'ready') === 1 && + integer(currentRuntime, 'last_heartbeat_at') > freshAfter + ) { + throw new Error('migration_target_still_available') + } + if ( + !replacementRuntime || + integer(replacementRuntime, 'ready') !== 1 || + integer(replacementRuntime, 'last_heartbeat_at') <= freshAfter + ) { + throw new Error('replacement_target_unavailable') + } + if ( + preservedActivityCellIds.some((cellId) => { + const runtime = runtimes.find((row) => text(row, 'cell_id') === cellId) + return ( + !runtime || + integer(runtime, 'ready') !== 1 || + integer(runtime, 'last_heartbeat_at') <= freshAfter + ) + }) + ) { + throw new Error('migration_preserved_activity_cell_unavailable') + } + await this.assertCellConnectionHeadroom( + transaction, + input.replacementTargetCellId + ) + for (const lease of activityLeases.filter( + (candidate) => text(candidate, 'cell_id') === input.currentTargetCellId + )) { + await this.removeActivityLease(transaction, identity, lease, now) + } + const sourceRequestUnits = activityUnitsForCell(activityLeases, input.sourceCellId) + const targetReservedUnits = sourceRequestUnits + 1 + await this.adjustCellReservation( + transaction, + input.replacementTargetCellId, + targetReservedUnits + ) + const assignmentEpoch = input.assignmentEpoch + 1 + const expiresAt = now + ASSIGNMENT_LIMITS.migrationLeaseMs + await transaction.query( + `UPDATE relay_assignments SET cell_id = ?, assignment_epoch = ?, + reserved_controls = reserved_controls + 1, + migration_leases = migration_leases + 1, + lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + input.replacementTargetCellId, + assignmentEpoch, + expiresAt, + now, + identity.userId, + identity.relayHostId + ] + ) + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?), (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + 'control', + input.replacementTargetCellId, + 1, + expiresAt, + now, + identity.userId, + identity.relayHostId, + migrationActivityId(assignmentEpoch), + 'migration', + input.replacementTargetCellId, + sourceRequestUnits, + expiresAt, + now + ] + ) + await this.insertControlConnectionReservation( + transaction, + identity, + input.replacementTargetCellId, + assignmentEpoch, + expiresAt, + now + ) + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + input.currentTargetCellId, + input.assignmentEpoch, + now + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + await transaction.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, NULL, NULL, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + input.sourceCellId, + input.replacementTargetCellId, + input.assignmentEpoch, + assignmentEpoch, + sourceRequestUnits, + targetReservedUnits, + expiresAt, + now, + now + ] + ) + if (existingIncarnation) { + await transaction.query( + `INSERT INTO relay_assignment_migration_incarnations + (user_id, relay_host_id, assignment_epoch, source_cell_incarnation, + target_cell_incarnation) + VALUES (?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + text(existingIncarnation, 'source_cell_incarnation'), + text(replacementRuntime, 'cell_incarnation') + ] + ) + } + if (existingPin) { + await transaction.query( + `INSERT INTO relay_post_drain_migration_pins + (user_id, relay_host_id, assignment_epoch, drain_attempt_id, + source_cell_id, source_cell_incarnation, target_cell_id, + target_cell_incarnation, source_request_units, target_reserved_units, + pinned_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + text(existingPin, 'drain_attempt_id'), + input.sourceCellId, + text(existingPin, 'source_cell_incarnation'), + input.replacementTargetCellId, + text(replacementRuntime, 'cell_incarnation'), + sourceRequestUnits, + targetReservedUnits, + now + ] + ) + } + return { + ...identity, + sourceCellId: input.sourceCellId, + targetCellId: input.replacementTargetCellId, + previousEpoch: input.assignmentEpoch, + assignmentEpoch, + expiresAt + } + }) + } + + async supersedeRegisteredCellEvacuations( + sourceCellId: string, + currentTargetCellId: string, + replacementTargetCellId: string, + limit: number + ): Promise { + if (!Number.isSafeInteger(limit) || limit < 1 || limit > 100) { + throw new Error('invalid_evacuation_limit') + } + const rows = await this.database.query( + `SELECT user_id, relay_host_id, assignment_epoch + FROM relay_assignment_migrations + WHERE source_cell_id = ? AND target_cell_id = ? + AND target_registered_at IS NOT NULL + AND completed_at IS NULL AND aborted_at IS NULL + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_migrations earlier + WHERE earlier.user_id = relay_assignment_migrations.user_id + AND earlier.relay_host_id = relay_assignment_migrations.relay_host_id + AND earlier.source_cell_id = relay_assignment_migrations.source_cell_id + AND earlier.target_cell_id = relay_assignment_migrations.target_cell_id + AND earlier.target_registered_at IS NOT NULL + AND earlier.completed_at IS NULL AND earlier.aborted_at IS NULL + AND earlier.assignment_epoch < relay_assignment_migrations.assignment_epoch + ) + ORDER BY user_id, relay_host_id, assignment_epoch + LIMIT ?`, + [sourceCellId, currentTargetCellId, limit] + ) + let superseded = 0 + let reconciledAccounting = false + for (const row of rows) { + const identity = { + userId: text(row, 'user_id'), + relayHostId: text(row, 'relay_host_id') + } + const input = { + assignmentEpoch: integer(row, 'assignment_epoch'), + sourceCellId, + currentTargetCellId, + replacementTargetCellId + } + const prepared = await this.prepareRegisteredCellSupersession(identity, input) + if (prepared.retired) { + input.assignmentEpoch = integer(row, 'assignment_epoch') + await this.supersedeRegisteredEvacuation(identity, input) + superseded++ + continue + } + input.assignmentEpoch = prepared.assignmentEpoch + try { + await this.supersedeRegisteredEvacuation(identity, input) + } catch (error) { + if ( + error instanceof Error && + error.message === 'migration_activity_topology_mismatch' + ) { + const preservedActivityCellIds = await this.additionalActivityCellIds( + identity, + sourceCellId, + currentTargetCellId + ) + if ( + preservedActivityCellIds.length === 0 || + preservedActivityCellIds.includes(replacementTargetCellId) + ) { + throw error + } + await this.reconcileReservationAccounting(sourceCellId, currentTargetCellId) + await this.reconcileReservationAccounting(sourceCellId, replacementTargetCellId) + for (const cellId of preservedActivityCellIds) { + await this.reconcileReservationAccounting(sourceCellId, cellId) + } + await this.supersedeRegisteredEvacuation( + identity, + input, + preservedActivityCellIds + ) + superseded++ + continue + } + if ( + reconciledAccounting || + !(error instanceof Error) || + error.message !== 'migration_cell_reservation_accounting_mismatch' + ) { + throw error + } + await this.reconcileReservationAccounting(sourceCellId, currentTargetCellId) + await this.reconcileReservationAccounting(sourceCellId, replacementTargetCellId) + reconciledAccounting = true + await this.supersedeRegisteredEvacuation(identity, input) + } + superseded++ + } + return superseded + } + + private async prepareRegisteredCellSupersession( + identity: AssignmentIdentity, + input: RegisteredEvacuationSupersessionInput + ): Promise<{ assignmentEpoch: number; retired: boolean }> { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + const staleMigration = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + if (!assignment || !staleMigration) throw new Error('migration_assignment_mismatch') + assertMigrationPair( + staleMigration, + input.sourceCellId, + input.currentTargetCellId + ) + if ( + optionalInteger(staleMigration, 'completed_at') !== undefined || + optionalInteger(staleMigration, 'aborted_at') !== undefined + ) { + return { assignmentEpoch: integer(assignment, 'assignment_epoch'), retired: true } + } + if (optionalInteger(staleMigration, 'target_registered_at') === undefined) { + throw new Error('migration_not_registered_for_supersession') + } + const assignmentEpoch = integer(assignment, 'assignment_epoch') + if (assignmentEpoch < input.assignmentEpoch) { + throw new Error('migration_assignment_mismatch') + } + const assignmentCellId = text(assignment, 'cell_id') + if ( + ![input.currentTargetCellId, input.replacementTargetCellId].includes( + assignmentCellId + ) + ) { + throw new Error('migration_assignment_mismatch') + } + const leases = await this.lockAssignmentActivities(transaction, identity) + const staleIncarnation = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migration_incarnations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + const stalePin = ( + await transaction.queryLocked( + `SELECT * FROM relay_post_drain_migration_pins + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + )[0] + assertMigrationRecoveryMetadata(staleMigration, staleIncarnation, stalePin) + const failedRuntime = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, + [input.currentTargetCellId] + ) + )[0] + await this.requireCellFence( + transaction, + input.currentTargetCellId, + failedRuntime, + now + ) + if ( + assignmentEpoch > input.assignmentEpoch && + assignmentCellId === input.replacementTargetCellId + ) { + const obsoleteActivityIds = new Set([ + pendingControlActivityId(input.assignmentEpoch), + migrationActivityId(input.assignmentEpoch) + ]) + const obsoleteLeases = leases.filter((lease) => + obsoleteActivityIds.has(text(lease, 'activity_id')) + ) + assertAssignmentActivityCounts( + assignment, + leases, + leases.filter((lease) => activityKind(lease) === 'migration').length + ) + for (const lease of obsoleteLeases) { + const kind = activityKind(lease) + const expectedUnits = + kind === 'migration' + ? integer(staleMigration, 'source_request_units') + : ACTIVITY_REQUEST_UNITS.control + if ( + text(lease, 'cell_id') !== input.currentTargetCellId || + !['control', 'migration'].includes(kind) || + integer(lease, 'request_units') !== expectedUnits + ) { + throw new Error('migration_activity_topology_mismatch') + } + } + if (obsoleteLeases.length > 0) await this.lockCellInventory(transaction) + for (const lease of obsoleteLeases) { + await this.removeActivityLease(transaction, identity, lease, now) + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + input.currentTargetCellId, + input.assignmentEpoch, + now + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + return { assignmentEpoch, retired: true } + } + let migrationRow = staleMigration + if (assignmentEpoch > input.assignmentEpoch) { + const existingCurrent = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + if (existingCurrent) { + assertMigrationPair( + existingCurrent, + input.sourceCellId, + input.currentTargetCellId + ) + if ( + optionalInteger(existingCurrent, 'completed_at') !== undefined || + optionalInteger(existingCurrent, 'aborted_at') !== undefined || + optionalInteger(existingCurrent, 'target_registered_at') === undefined + ) { + throw new Error('migration_activity_topology_mismatch') + } + assertCurrentMigrationAssignment(assignment, existingCurrent) + const currentIncarnation = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migration_incarnations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + const currentPin = ( + await transaction.queryLocked( + `SELECT * FROM relay_post_drain_migration_pins + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + assertMigrationRecoveryMetadata( + existingCurrent, + currentIncarnation, + currentPin + ) + migrationRow = existingCurrent + } else { + if (leases.length !== 0) { + throw new Error('migration_activity_topology_mismatch') + } + assertAssignmentActivityCounts(assignment, leases, 0) + const currentIncarnations = await transaction.queryLocked( + `SELECT assignment_epoch FROM relay_assignment_migration_incarnations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + const currentPins = await transaction.queryLocked( + `SELECT assignment_epoch FROM relay_post_drain_migration_pins + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + if (currentIncarnations.length > 0 || currentPins.length > 0) { + throw new Error('migration_activity_topology_mismatch') + } + await transaction.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, 0, 1, ?, ?, NULL, NULL, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + input.sourceCellId, + input.currentTargetCellId, + input.assignmentEpoch, + assignmentEpoch, + now - 1, + now, + now, + now + ] + ) + migrationRow = { + ...staleMigration, + assignment_epoch: assignmentEpoch, + previous_epoch: input.assignmentEpoch, + source_request_units: 0, + target_reserved_units: 1, + expires_at: now - 1, + target_registered_at: now, + completed_at: null, + aborted_at: null + } + if (staleIncarnation) { + await transaction.query( + `INSERT INTO relay_assignment_migration_incarnations + (user_id, relay_host_id, assignment_epoch, + source_cell_incarnation, target_cell_incarnation) + VALUES (?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + text(staleIncarnation, 'source_cell_incarnation'), + text(staleIncarnation, 'target_cell_incarnation') + ] + ) + } + if (stalePin) { + await transaction.query( + `INSERT INTO relay_post_drain_migration_pins + (user_id, relay_host_id, assignment_epoch, drain_attempt_id, + source_cell_id, source_cell_incarnation, target_cell_id, + target_cell_incarnation, source_request_units, + target_reserved_units, pinned_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, 0, 1, ?)`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + text(stalePin, 'drain_attempt_id'), + input.sourceCellId, + text(stalePin, 'source_cell_incarnation'), + input.currentTargetCellId, + text(stalePin, 'target_cell_incarnation'), + now + ] + ) + } + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + input.currentTargetCellId, + input.assignmentEpoch, + now + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, input.assignmentEpoch] + ) + } + const migrationLeases = leases.filter( + (lease) => activityKind(lease) === 'migration' + ) + if (migrationLeases.length > 0) { + assertAssignmentActivityAccounting(assignment, leases, migrationRow) + return { assignmentEpoch, retired: false } + } + assertAssignmentActivityCounts(assignment, leases, 0) + const sourceRequestUnits = integer(migrationRow, 'source_request_units') + if ( + integer(migrationRow, 'expires_at') > now || + sourceRequestUnits < 0 || + integer(migrationRow, 'target_reserved_units') !== sourceRequestUnits + 1 + ) { + throw new Error('migration_activity_lease_shape_mismatch') + } + await this.lockCellInventory(transaction) + await this.adjustCellReservation( + transaction, + input.currentTargetCellId, + sourceRequestUnits + ) + const expiresAt = now + ASSIGNMENT_LIMITS.migrationLeaseMs + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, 'migration', ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + migrationActivityId(assignmentEpoch), + input.currentTargetCellId, + sourceRequestUnits, + expiresAt, + now + ] + ) + await transaction.query( + `UPDATE relay_assignments SET migration_leases = migration_leases + 1, + lease_expires_at = CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [expiresAt, expiresAt, now, identity.userId, identity.relayHostId] + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [expiresAt, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + return { assignmentEpoch, retired: false } + }) + } + + private async additionalActivityCellIds( + identity: AssignmentIdentity, + sourceCellId: string, + currentTargetCellId: string + ): Promise { + const rows = await this.database.query( + `SELECT DISTINCT cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? + AND cell_id NOT IN (?, ?) + ORDER BY cell_id`, + [identity.userId, identity.relayHostId, sourceCellId, currentTargetCellId] + ) + return rows.map((row) => text(row, 'cell_id')) + } + + async completeEvacuation( + identity: AssignmentIdentity, + assignmentEpoch: number + ): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND completed_at IS NULL AND aborted_at IS NULL`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + if (!row) throw new Error('migration_not_found') + if (optionalInteger(row, 'target_registered_at') === undefined) { + throw new Error('migration_target_not_registered') + } + const sourceCellId = text(row, 'source_cell_id') + const targetCellId = text(row, 'target_cell_id') + if ( + !assignment || + text(assignment, 'cell_id') !== targetCellId || + integer(assignment, 'assignment_epoch') !== assignmentEpoch + ) { + throw new Error('migration_assignment_mismatch') + } + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + const sourceUnits = activityUnitsForCell(activityLeases, sourceCellId) + if (sourceUnits > 0) throw new Error('migration_source_still_active') + let cellsLocked = false + let targetStartedAt = 0 + if (this.requireLiveCells) { + let cells: SqlRow[] + try { + cells = await this.lockCellInventory(transaction, true) + } catch (error) { + if (isDatabaseLockUnavailable(error)) { + // Mixed-version workers may still hold a cell-first lock; defer + // instead of waiting long enough to form their legacy lock cycle. + throw new Error('migration_cell_inventory_busy') + } + throw error + } + cellsLocked = true + const source = cells.find((cell) => text(cell, 'cell_id') === sourceCellId) + const target = cells.find((cell) => text(cell, 'cell_id') === targetCellId) + if (!source || integer(source, 'enabled') !== 0) { + throw new Error('migration_source_admission_changed') + } + if (!target || integer(target, 'enabled') !== 1) { + throw new Error('migration_target_admission_changed') + } + const runtimes = await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime WHERE cell_id IN (?, ?) ORDER BY cell_id`, + [sourceCellId, targetCellId] + ) + const sourceRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === sourceCellId + ) + const targetRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === targetCellId + ) + const freshAfter = now - this.heartbeatTtlMs + if ( + !sourceRuntime || + integer(sourceRuntime, 'ready') !== 1 || + integer(sourceRuntime, 'observed_requests') !== 0 || + integer(sourceRuntime, 'last_heartbeat_at') <= freshAfter + ) { + throw new Error('migration_source_runtime_not_quiescent') + } + if ( + !targetRuntime || + integer(targetRuntime, 'ready') !== 1 || + integer(targetRuntime, 'last_heartbeat_at') <= freshAfter + ) { + throw new Error('migration_target_runtime_not_ready') + } + targetStartedAt = integer(targetRuntime, 'started_at') + } + const targetIsActive = activityLeases.some( + (lease) => + text(lease, 'cell_id') === targetCellId && + text(lease, 'activity_kind') === 'control' && + !text(lease, 'activity_id').startsWith('control-pending:') && + integer(lease, 'expires_at') > now && + integer(lease, 'updated_at') >= targetStartedAt + ) + if (!targetIsActive) throw new Error('migration_target_not_active') + const lease = activityLeaseById(activityLeases, migrationActivityId(assignmentEpoch)) + if (lease && !cellsLocked) await this.lockCellInventory(transaction) + if (lease) await this.removeActivityLease(transaction, identity, lease, now) + await transaction.query( + `UPDATE relay_assignment_migrations SET completed_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + }) + } + + async rebalanceDormant( + identity: AssignmentIdentity, + targetCellId: string + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + if (!assignment) throw new Error('assignment_not_found') + if (!mayNormallyReassign(activity(assignment), now)) throw new Error('assignment_active') + const sourceCellId = text(assignment, 'cell_id') + if (sourceCellId === targetCellId) throw new Error('target_matches_source') + await this.lockAssignmentActivities(transaction, identity) + const cells = await this.lockCellInventory(transaction) + const admission = await cellAdmissionStates(transaction) + const targetRow = cells.find( + (row) => + text(row, 'cell_id') === targetCellId && + admission.get(targetCellId) === 'general' + ) + if (!targetRow) throw new Error('target_cell_unavailable') + if (!(await this.cellIsLive(transaction, targetCellId, now))) { + throw new Error('target_cell_unavailable') + } + await this.assertCellConnectionHeadroom(transaction, targetCellId) + const target = cell(targetRow, await this.cellRegion(transaction, targetCellId)) + await this.adjustCellReservation(transaction, targetCellId, 1) + const assignmentEpoch = integer(assignment, 'assignment_epoch') + 1 + const leaseExpiresAt = now + ASSIGNMENT_LIMITS.activityLeaseMs + await transaction.query( + `UPDATE relay_assignments SET cell_id = ?, assignment_epoch = ?, + reserved_controls = 1, lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + targetCellId, + assignmentEpoch, + leaseExpiresAt, + now, + identity.userId, + identity.relayHostId + ] + ) + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + sourceCellId, + assignmentEpoch - 1, + now + ) + await this.insertPendingControlLease( + transaction, + identity, + targetCellId, + assignmentEpoch, + now + ) + return { + ...identity, + ...target, + assignmentEpoch, + leaseExpiresAt + } + }) + } + + async inspectRegionalRehomeControl(): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.initializeRegionalRehomeControl(transaction, now) + const row = ( + await transaction.query( + `SELECT * FROM relay_region_rehome_control WHERE control_id = 'global'` + ) + )[0]! + return regionalRehomeControl(row) + }) + } + + async applyRegionalRehomeControl(input: { + expectedGeneration: number + enabled: boolean + notBefore: number + ratePerMinute: number + preferenceMaxAgeMs: number + drainGraceMs: number + }): Promise { + if (!Number.isSafeInteger(input.expectedGeneration) || input.expectedGeneration < 0) { + throw new Error('invalid_regional_rehome_generation') + } + if (!Number.isSafeInteger(input.notBefore) || input.notBefore < 0) { + throw new Error('invalid_regional_rehome_not_before') + } + if (!Number.isSafeInteger(input.ratePerMinute) || input.ratePerMinute < 1 || input.ratePerMinute > 120) { + throw new Error('invalid_regional_rehome_rate') + } + if ( + !Number.isSafeInteger(input.preferenceMaxAgeMs) || + input.preferenceMaxAgeMs < 60_000 || + input.preferenceMaxAgeMs > 30 * 24 * 60 * 60_000 + ) { + throw new Error('invalid_regional_rehome_preference_age') + } + if ( + !Number.isSafeInteger(input.drainGraceMs) || + input.drainGraceMs < 60_000 || + input.drainGraceMs > 60 * 60_000 + ) { + throw new Error('invalid_regional_rehome_drain_grace') + } + const now = this.now() + return await this.database.transaction(async (transaction) => { + await this.initializeRegionalRehomeControl(transaction, now) + const current = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_control WHERE control_id = 'global'` + ) + )[0]! + if (integer(current, 'generation') !== input.expectedGeneration) { + throw new Error('regional_rehome_generation_mismatch') + } + if ( + input.enabled && + input.notBefore < + integer(current, 'observation_started_at') + REGIONAL_REHOME_OBSERVATION_MS + ) { + throw new Error('regional_rehome_observation_window_incomplete') + } + await transaction.query( + `UPDATE relay_region_rehome_control + SET generation = generation + 1, enabled = ?, not_before = ?, + rate_per_minute = ?, preference_max_age_ms = ?, drain_grace_ms = ?, + updated_at = ? + WHERE control_id = 'global'`, + [ + input.enabled ? 1 : 0, + input.notBefore, + input.ratePerMinute, + input.preferenceMaxAgeMs, + input.drainGraceMs, + now + ] + ) + const updated = ( + await transaction.query( + `SELECT * FROM relay_region_rehome_control WHERE control_id = 'global'` + ) + )[0]! + return regionalRehomeControl(updated) + }) + } + + async disableRegionalRehomeControl(): Promise { + const now = this.now() + const result = await this.database.query( + `UPDATE relay_region_rehome_control + SET generation = generation + 1, enabled = 0, updated_at = ? + WHERE control_id = 'global' AND enabled = 1`, + [now] + ) + return integer(result[0]!, 'changes') === 1 + } + + private async initializeRegionalRehomeControl( + database: RelayDatabase, + now: number + ): Promise { + await database.query( + `INSERT INTO relay_region_rehome_control + (control_id, generation, enabled, observation_started_at, not_before, + rate_per_minute, preference_max_age_ms, drain_grace_ms, updated_at) + VALUES ('global', 0, 0, ?, 0, 10, ?, ?, ?) + ON CONFLICT (control_id) DO NOTHING`, + [now, 24 * 60 * 60_000, 60 * 60_000, now] + ) + } + + async regionalRehomeFleetSafety(): Promise { + return await this.readRegionalRehomeFleetSafety(this.database, this.now()) + } + + private async readRegionalRehomeFleetSafety( + database: RelayDatabase, + now: number + ): Promise { + const rows = await database.query( + `SELECT runtime.ready, runtime.last_heartbeat_at, + safety.cell_id AS safety_cell_id, safety.observed_at, + safety.sql_failures, safety.reconnects, + safety.control_activity_recovery_failures, + safety.database_pool_waiting, safety.database_pool_waiters_max, + safety.database_pool_wait_ms_max + FROM relay_cells cell + JOIN relay_cell_admission admission ON admission.cell_id = cell.cell_id + JOIN relay_cell_regions region ON region.cell_id = cell.cell_id + LEFT JOIN relay_cell_runtime runtime ON runtime.cell_id = cell.cell_id + LEFT JOIN relay_cell_capabilities capability ON capability.cell_id = cell.cell_id + LEFT JOIN relay_cell_rehome_safety safety + ON safety.cell_id = runtime.cell_id + AND safety.cell_incarnation = runtime.cell_incarnation + WHERE cell.enabled = 1 AND admission.admission_state = 'general' + AND ( + region.region = 'asia-east2' OR + (region.region = 'us-central1' AND capability.regional_rehome_protocol >= 1) + )` + ) + const valid = rows.filter( + (row) => + integer(row, 'ready') === 1 && + integer(row, 'last_heartbeat_at') > now - this.heartbeatTtlMs && + optionalText(row, 'safety_cell_id') !== undefined && + integer(row, 'observed_at') > now - 60_000 + ) + const missingCells = rows.length === 0 ? 1 : rows.length - valid.length + return { + requiredCells: rows.length, + missingCells, + observedAt: + missingCells > 0 + ? 0 + : Math.min(...valid.map((row) => integer(row, 'observed_at'))), + sqlFailures: valid.reduce((total, row) => total + integer(row, 'sql_failures'), 0), + reconnects: valid.reduce((total, row) => total + integer(row, 'reconnects'), 0), + maxReconnects: Math.max(0, ...valid.map((row) => integer(row, 'reconnects'))), + controlActivityRecoveryFailures: valid.reduce( + (total, row) => total + integer(row, 'control_activity_recovery_failures'), + 0 + ), + databasePoolWaiting: Math.max( + 0, + ...valid.map((row) => integer(row, 'database_pool_waiting')) + ), + databasePoolWaitersMax: Math.max( + 0, + ...valid.map((row) => integer(row, 'database_pool_waiters_max')) + ), + databasePoolWaitMsMax: Math.max( + 0, + ...valid.map((row) => integer(row, 'database_pool_wait_ms_max')) + ) + } + } + + async claimRegionalRehome( + processSafety?: RegionalRehomeSafetySnapshot + ): Promise { + const now = this.now() + // Directors poll every second; avoid taking the global worker-row lock while disabled. + const control = ( + await this.database.query( + `SELECT enabled, not_before + FROM relay_region_rehome_control + WHERE control_id = 'global'` + ) + )[0] + if (!control) { + await this.initializeRegionalRehomeControl(this.database, now) + return null + } + if (integer(control, 'enabled') !== 1 || integer(control, 'not_before') > now) { + return null + } + this.pendingRegionalRehomeDisableLog = null + const candidateSkips: RegionalRehomeCandidateSkip[] = [] + const claimResult = await this.database.transaction(async (transaction) => { + candidateSkips.length = 0 + await this.initializeRegionalRehomeControl(transaction, now) + const control = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_control WHERE control_id = 'global'` + ) + )[0]! + if (integer(control, 'enabled') !== 1 || integer(control, 'not_before') > now) { + return null + } + const intervalMs = Math.ceil(60_000 / integer(control, 'rate_per_minute')) + const preferenceCutoff = now - integer(control, 'preference_max_age_ms') + await transaction.query( + `INSERT INTO relay_region_rehome_worker_state + (worker_id, next_dispatch_at, paused_until, consecutive_failures, updated_at) + VALUES ('global', 0, 0, 0, ?) + ON CONFLICT (worker_id) DO NOTHING`, + [now] + ) + const worker = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` + ) + )[0]! + if ( + integer(worker, 'paused_until') > now || + integer(worker, 'next_dispatch_at') > now + ) { + return null + } + const effectiveProcessSafety = processSafety ?? cleanRegionalRehomeSafety(now) + const fleetSafety = await this.readRegionalRehomeFleetSafety(transaction, now) + if ( + !(await this.regionalRehomeSafetyAllowsClaim( + transaction, + worker, + effectiveProcessSafety, + fleetSafety, + now + )) + ) { + return null + } + const retry = ( + await transaction.queryLocked( + `SELECT attempt.*, source.cell_url AS source_cell_url + FROM relay_region_rehome_attempts attempt + JOIN relay_cells source ON source.cell_id = attempt.source_cell_id + JOIN relay_cell_runtime runtime ON runtime.cell_id = attempt.source_cell_id + JOIN relay_cell_capabilities capability + ON capability.cell_id = runtime.cell_id + AND capability.cell_incarnation = runtime.cell_incarnation + JOIN relay_assignment_migrations migration + ON migration.user_id = attempt.user_id + AND migration.relay_host_id = attempt.relay_host_id + AND migration.assignment_epoch = attempt.assignment_epoch + WHERE attempt.drain_receipt_at IS NULL + AND attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + AND attempt.send_attempts < 10 + AND (attempt.last_send_attempt_at IS NULL OR attempt.last_send_attempt_at <= ?) + AND runtime.cell_incarnation = attempt.source_cell_incarnation + AND runtime.ready = 1 AND runtime.last_heartbeat_at > ? + AND capability.regional_rehome_protocol >= 1 + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ORDER BY attempt.created_at, attempt.attempt_id + LIMIT 1`, + [now - 30_000, now - this.heartbeatTtlMs] + ) + )[0] + if (retry) { + const fleetSafety = await this.lockedRegionalRehomeFleetSafety(transaction, now) + if ( + !(await this.regionalRehomeSafetyAllowsClaim( + transaction, + worker, + effectiveProcessSafety, + fleetSafety, + now + )) + ) { + return null + } + await this.markRegionalRehomeDispatchClaimed( + transaction, + text(retry, 'attempt_id'), + now, + intervalMs + ) + retry.send_attempts = integer(retry, 'send_attempts') + 1 + return regionalRehomeAttempt(retry) + } + + // A drain receipt is not convergence: grace enforcement lives only in + // source-cell session state, and attempts have been observed stalled + // dual-homed well past grace with source leases still renewing. Such + // attempts are re-dispatched with the remaining (zero) grace so the + // source force-closes and the host re-resolves onto its registered + // target. + const redrain = ( + await transaction.queryLocked( + `SELECT attempt.*, source.cell_url AS source_cell_url + FROM relay_region_rehome_attempts attempt + JOIN relay_cells source ON source.cell_id = attempt.source_cell_id + JOIN relay_cell_runtime runtime ON runtime.cell_id = attempt.source_cell_id + JOIN relay_cell_capabilities capability + ON capability.cell_id = runtime.cell_id + AND capability.cell_incarnation = runtime.cell_incarnation + JOIN relay_assignment_migrations migration + ON migration.user_id = attempt.user_id + AND migration.relay_host_id = attempt.relay_host_id + AND migration.assignment_epoch = attempt.assignment_epoch + WHERE attempt.drain_receipt_at IS NOT NULL + AND attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + AND attempt.created_at + attempt.drain_grace_ms <= ? + AND attempt.send_attempts < ? + AND (attempt.last_send_attempt_at IS NULL OR attempt.last_send_attempt_at <= ?) + AND runtime.cell_incarnation = attempt.source_cell_incarnation + AND runtime.ready = 1 AND runtime.last_heartbeat_at > ? + AND capability.regional_rehome_protocol >= 1 + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + AND migration.target_registered_at IS NOT NULL + AND EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases source_lease + WHERE source_lease.user_id = attempt.user_id + AND source_lease.relay_host_id = attempt.relay_host_id + AND source_lease.cell_id = attempt.source_cell_id + ) + ORDER BY attempt.created_at, attempt.attempt_id + LIMIT 1`, + [ + now, + REGIONAL_REHOME_REDRAIN_SEND_LIMIT, + now - REGIONAL_REHOME_REDRAIN_INTERVAL_MS, + now - this.heartbeatTtlMs + ] + ) + )[0] + if (redrain) { + const fleetSafety = await this.lockedRegionalRehomeFleetSafety(transaction, now) + if ( + !(await this.regionalRehomeSafetyAllowsClaim( + transaction, + worker, + effectiveProcessSafety, + fleetSafety, + now + )) + ) { + return null + } + await this.markRegionalRehomeDispatchClaimed( + transaction, + text(redrain, 'attempt_id'), + now, + intervalMs + ) + redrain.send_attempts = integer(redrain, 'send_attempts') + 1 + redrain.drain_grace_ms = 0 + return regionalRehomeAttempt(redrain) + } + + const candidates = await transaction.query( + `SELECT preference.user_id, preference.relay_host_id, + preference.observed_at, assignment.cell_id AS source_cell_id, + assignment.assignment_epoch + FROM relay_assignment_region_preferences preference + JOIN relay_assignments assignment + ON assignment.user_id = preference.user_id + AND assignment.relay_host_id = preference.relay_host_id + JOIN relay_cell_regions region ON region.cell_id = assignment.cell_id + JOIN relay_cell_admission admission ON admission.cell_id = assignment.cell_id + JOIN relay_cell_runtime runtime ON runtime.cell_id = assignment.cell_id + JOIN relay_cell_capabilities capability + ON capability.cell_id = runtime.cell_id + AND capability.cell_incarnation = runtime.cell_incarnation + WHERE preference.preferred_region = 'asia-east2' + AND preference.observed_at >= ? + AND region.region = 'us-central1' + AND admission.admission_state = 'general' + AND runtime.ready = 1 AND runtime.last_heartbeat_at > ? + AND capability.regional_rehome_protocol >= 1 + AND EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases control + WHERE control.user_id = assignment.user_id + AND control.relay_host_id = assignment.relay_host_id + AND control.cell_id = assignment.cell_id + AND control.activity_kind = 'control' + AND control.activity_id NOT LIKE 'control-pending:%' + AND control.expires_at > ? + AND control.updated_at >= runtime.started_at + ) + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ) + ORDER BY preference.observed_at, preference.user_id, preference.relay_host_id + LIMIT 10`, + [preferenceCutoff, now - this.heartbeatTtlMs, now] + ) + for (const candidate of candidates) { + const claimed = await this.startRegionalRehomeCandidate(transaction, { + identity: { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + }, + sourceCellId: text(candidate, 'source_cell_id'), + assignmentEpoch: integer(candidate, 'assignment_epoch'), + preferenceCutoff, + drainGraceMs: integer(control, 'drain_grace_ms'), + processSafety: effectiveProcessSafety, + worker, + now, + skips: candidateSkips + }) + if (!claimed) continue + await this.markRegionalRehomeDispatchClaimed( + transaction, + claimed.attemptId, + now, + intervalMs + ) + return { ...claimed, sendAttempts: 1 } + } + if (candidates.length > 0) { + // Skipped candidates still cost all-rows FOR UPDATE inventory scans; + // charge the dispatch interval so skips are rate-limited like claims. + await this.markRegionalRehomeTickSkipped(transaction, now, intervalMs) + } + return null + }) + const pendingDisableLog = this.pendingRegionalRehomeDisableLog + this.pendingRegionalRehomeDisableLog = null + if (pendingDisableLog) console.warn(JSON.stringify(pendingDisableLog)) + if (claimResult === null && candidateSkips.length > 0) { + console.warn(JSON.stringify(aggregateRegionalRehomeCandidateSkips(candidateSkips))) + } + return claimResult + } + + private async startRegionalRehomeCandidate( + transaction: RelayDatabase, + input: { + identity: AssignmentIdentity + sourceCellId: string + assignmentEpoch: number + preferenceCutoff: number + drainGraceMs: number + processSafety: RegionalRehomeSafetySnapshot + worker: SqlRow + now: number + skips: RegionalRehomeCandidateSkip[] + } + ): Promise | null> { + const assignment = await this.assignmentRow(transaction, input.identity) + if ( + !assignment || + text(assignment, 'cell_id') !== input.sourceCellId || + integer(assignment, 'assignment_epoch') !== input.assignmentEpoch + ) { + input.skips.push({ reason: 'candidate_stale' }) + return null + } + const preference = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_region_preferences + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + )[0] + if ( + !preference || + text(preference, 'preferred_region') !== 'asia-east2' || + integer(preference, 'observed_at') < input.preferenceCutoff + ) { + input.skips.push({ reason: 'candidate_stale' }) + return null + } + const activeMigration = await transaction.queryLocked( + `SELECT assignment_epoch FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? + AND completed_at IS NULL AND aborted_at IS NULL`, + [input.identity.userId, input.identity.relayHostId] + ) + if (activeMigration.length > 0) { + input.skips.push({ reason: 'candidate_stale' }) + return null + } + const activityLeases = await this.lockAssignmentActivities(transaction, input.identity) + assertAssignmentActivityCounts(assignment, activityLeases, 0) + const cells = await this.lockCellInventory(transaction) + const admission = await cellAdmissionStates(transaction) + const regions = new Map( + (await transaction.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ + text(row, 'cell_id'), + relayRegion(row, 'region') + ]) + ) + const runtimes = await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime ORDER BY cell_id` + ) + const capabilities = await transaction.queryLocked( + `SELECT * FROM relay_cell_capabilities ORDER BY cell_id` + ) + const safetyRows = await transaction.queryLocked( + `SELECT * FROM relay_cell_rehome_safety ORDER BY cell_id` + ) + const source = cells.find((row) => text(row, 'cell_id') === input.sourceCellId) + const sourceRuntime = runtimes.find( + (row) => text(row, 'cell_id') === input.sourceCellId + ) + const sourceCapability = capabilities.find( + (row) => text(row, 'cell_id') === input.sourceCellId + ) + const sourceSafety = safetyRows.find( + (row) => text(row, 'cell_id') === input.sourceCellId + ) + const fleetSafety = regionalRehomeFleetSafetyFromInventory({ + cells, + admission, + regions, + runtimes, + capabilities, + safetyRows, + now: input.now, + heartbeatTtlMs: this.heartbeatTtlMs + }) + const safetyFailure = regionalRehomeFleetSafetyFailure( + input.processSafety, + fleetSafety, + input.now + ) + if (safetyFailure) { + await this.pauseRegionalRehomeForSafety( + transaction, + input.worker, + input.now, + safetyFailure, + fleetSafety + ) + return null + } + if ( + !source || + integer(source, 'enabled') !== 1 || + admission.get(input.sourceCellId) !== 'general' || + regions.get(input.sourceCellId) !== RELAY_DEFAULT_REGION || + !sourceRuntime || + integer(sourceRuntime, 'ready') !== 1 || + integer(sourceRuntime, 'last_heartbeat_at') <= input.now - this.heartbeatTtlMs || + !sourceCapability || + text(sourceCapability, 'cell_incarnation') !== + text(sourceRuntime, 'cell_incarnation') || + integer(sourceCapability, 'regional_rehome_protocol') < 1 + ) { + input.skips.push({ reason: 'source_ineligible', cellId: input.sourceCellId }) + return null + } + if (!regionalRehomeCellSafetyIsClean(sourceSafety, sourceRuntime, input.now)) { + input.skips.push(cellUncleanSkip('source_unclean', input.sourceCellId, sourceSafety)) + return null + } + const sourceControlActive = activityLeases.some( + (lease) => + text(lease, 'cell_id') === input.sourceCellId && + text(lease, 'activity_kind') === 'control' && + !text(lease, 'activity_id').startsWith('control-pending:') && + integer(lease, 'expires_at') > input.now && + integer(lease, 'updated_at') >= integer(sourceRuntime, 'started_at') + ) + if (!sourceControlActive) { + input.skips.push({ reason: 'source_control_inactive', cellId: input.sourceCellId }) + return null + } + const connectionHeadroom = await this.connectionHeadroomByCell(transaction) + const eligibleTargets = cells.filter((row) => { + const cellId = text(row, 'cell_id') + const runtime = runtimes.find((candidate) => text(candidate, 'cell_id') === cellId) + return ( + cellId !== input.sourceCellId && + integer(row, 'enabled') === 1 && + admission.get(cellId) === 'general' && + regions.get(cellId) === 'asia-east2' && + runtime !== undefined && + integer(runtime, 'ready') === 1 && + integer(runtime, 'last_heartbeat_at') > input.now - this.heartbeatTtlMs + ) + }) + const targetIsClean = (row: SqlRow): boolean => { + const cellId = text(row, 'cell_id') + return regionalRehomeCellSafetyIsClean( + safetyRows.find((safety) => text(safety, 'cell_id') === cellId), + runtimes.find((candidate) => text(candidate, 'cell_id') === cellId)!, + input.now + ) + } + const targetCandidates = eligibleTargets.filter( + (row) => targetIsClean(row) && connectionHeadroom.get(text(row, 'cell_id')) !== false + ) + if (targetCandidates.length === 0) { + const unclean = eligibleTargets.filter((row) => !targetIsClean(row)) + for (const row of unclean) { + const cellId = text(row, 'cell_id') + input.skips.push( + cellUncleanSkip( + 'target_unclean', + cellId, + safetyRows.find((safety) => text(safety, 'cell_id') === cellId) + ) + ) + } + if (unclean.length === 0) { + input.skips.push({ + reason: eligibleTargets.length === 0 ? 'no_eligible_target' : 'no_target_headroom' + }) + } + return null + } + targetCandidates.sort((left, right) => { + const leftRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === text(left, 'cell_id') + )! + const rightRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === text(right, 'cell_id') + )! + const leftLoad = + (integer(left, 'reserved_requests') + integer(leftRuntime, 'observed_requests')) / + integer(left, 'capacity_requests') + const rightLoad = + (integer(right, 'reserved_requests') + integer(rightRuntime, 'observed_requests')) / + integer(right, 'capacity_requests') + return leftLoad - rightLoad || + text(left, 'cell_id').localeCompare(text(right, 'cell_id')) + }) + const sourceRequestUnits = activityUnitsForCell(activityLeases, input.sourceCellId) + const targetReservedUnits = sourceRequestUnits + 1 + const target = targetCandidates.find( + (row) => + integer(row, 'reserved_requests') + targetReservedUnits <= + integer(row, 'capacity_requests') + ) + if (!target) { + input.skips.push({ reason: 'no_target_headroom' }) + return null + } + const targetCellId = text(target, 'cell_id') + const targetRuntime = runtimes.find( + (runtime) => text(runtime, 'cell_id') === targetCellId + )! + await this.adjustCellReservation(transaction, targetCellId, targetReservedUnits) + const previousEpoch = integer(assignment, 'assignment_epoch') + const assignmentEpoch = previousEpoch + 1 + const expiresAt = input.now + ASSIGNMENT_LIMITS.migrationLeaseMs + await transaction.query( + `UPDATE relay_assignments SET cell_id = ?, assignment_epoch = ?, + reserved_controls = reserved_controls + 1, + migration_leases = migration_leases + 1, + lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + targetCellId, + assignmentEpoch, + expiresAt, + input.now, + input.identity.userId, + input.identity.relayHostId + ] + ) + await transaction.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, 'control', ?, 1, ?, ?), + (?, ?, ?, 'migration', ?, ?, ?, ?)`, + [ + input.identity.userId, + input.identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + targetCellId, + expiresAt, + input.now, + input.identity.userId, + input.identity.relayHostId, + migrationActivityId(assignmentEpoch), + targetCellId, + sourceRequestUnits, + expiresAt, + input.now + ] + ) + await this.insertControlConnectionReservation( + transaction, + input.identity, + targetCellId, + assignmentEpoch, + expiresAt, + input.now + ) + await transaction.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, NULL, NULL, ?, ?)`, + [ + input.identity.userId, + input.identity.relayHostId, + input.sourceCellId, + targetCellId, + previousEpoch, + assignmentEpoch, + sourceRequestUnits, + targetReservedUnits, + expiresAt, + input.now, + input.now + ] + ) + await transaction.query( + `INSERT INTO relay_assignment_migration_incarnations + (user_id, relay_host_id, assignment_epoch, source_cell_incarnation, + target_cell_incarnation) + VALUES (?, ?, ?, ?, ?)`, + [ + input.identity.userId, + input.identity.relayHostId, + assignmentEpoch, + text(sourceRuntime, 'cell_incarnation'), + text(targetRuntime, 'cell_incarnation') + ] + ) + const attemptId = randomUUID() + await transaction.query( + `INSERT INTO relay_region_rehome_attempts + (attempt_id, user_id, relay_host_id, preferred_region, + source_cell_id, source_cell_incarnation, target_cell_id, + target_cell_incarnation, previous_epoch, assignment_epoch, + drain_grace_ms, send_attempts, last_send_attempt_at, + drain_receipt_at, drain_outcome, completed_at, aborted_at, + created_at, updated_at) + VALUES (?, ?, ?, 'asia-east2', ?, ?, ?, ?, ?, ?, ?, 0, NULL, + NULL, NULL, NULL, NULL, ?, ?)`, + [ + attemptId, + input.identity.userId, + input.identity.relayHostId, + input.sourceCellId, + text(sourceRuntime, 'cell_incarnation'), + targetCellId, + text(targetRuntime, 'cell_incarnation'), + previousEpoch, + assignmentEpoch, + input.drainGraceMs, + input.now, + input.now + ] + ) + return { + ...input.identity, + attemptId, + preferredRegion: 'asia-east2', + sourceCellId: input.sourceCellId, + sourceCellUrl: text(source, 'cell_url'), + sourceCellIncarnation: text(sourceRuntime, 'cell_incarnation'), + targetCellId, + targetCellIncarnation: text(targetRuntime, 'cell_incarnation'), + previousEpoch, + assignmentEpoch, + drainGraceMs: input.drainGraceMs + } + } + + private async lockedRegionalRehomeFleetSafety( + transaction: RelayDatabase, + now: number + ): Promise { + const cells = await this.lockCellInventory(transaction) + const admission = await cellAdmissionStates(transaction) + const regions = new Map( + (await transaction.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ + text(row, 'cell_id'), + relayRegion(row, 'region') + ]) + ) + const runtimes = await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime ORDER BY cell_id` + ) + const capabilities = await transaction.queryLocked( + `SELECT * FROM relay_cell_capabilities ORDER BY cell_id` + ) + const safetyRows = await transaction.queryLocked( + `SELECT * FROM relay_cell_rehome_safety ORDER BY cell_id` + ) + return regionalRehomeFleetSafetyFromInventory({ + cells, + admission, + regions, + runtimes, + capabilities, + safetyRows, + now, + heartbeatTtlMs: this.heartbeatTtlMs + }) + } + + private async regionalRehomeSafetyAllowsClaim( + transaction: RelayDatabase, + worker: SqlRow, + processSafety: RegionalRehomeSafetySnapshot, + fleetSafety: RegionalRehomeFleetSafety, + now: number + ): Promise { + const failure = regionalRehomeFleetSafetyFailure(processSafety, fleetSafety, now) + if (!failure) { + return true + } + await this.pauseRegionalRehomeForSafety(transaction, worker, now, failure, fleetSafety) + return false + } + + private async pauseRegionalRehomeForSafety( + transaction: RelayDatabase, + worker: SqlRow, + now: number, + reason: string, + fleetSafety: RegionalRehomeFleetSafety + ): Promise { + const disabled = await transaction.query( + `UPDATE relay_region_rehome_control + SET generation = generation + 1, enabled = 0, updated_at = ? + WHERE control_id = 'global' AND enabled = 1 + RETURNING generation`, + [now] + ) + // The durable disable is otherwise invisible: nothing else records why + // claims stopped and inspection only shows enabled=false. Logged after + // the transaction commits so a rollback cannot fabricate the record. + if (disabled.length > 0) { + this.pendingRegionalRehomeDisableLog = { + event: 'orca_relay_regional_rehome_safety_disabled', + reason, + controlGeneration: integer(disabled[0]!, 'generation'), + now, + requiredCells: fleetSafety.requiredCells, + missingCells: fleetSafety.missingCells, + observedAt: fleetSafety.observedAt, + sqlFailures: fleetSafety.sqlFailures, + reconnects: fleetSafety.reconnects, + maxReconnects: fleetSafety.maxReconnects, + controlActivityRecoveryFailures: fleetSafety.controlActivityRecoveryFailures, + databasePoolWaiting: fleetSafety.databasePoolWaiting, + databasePoolWaitersMax: fleetSafety.databasePoolWaitersMax, + databasePoolWaitMsMax: fleetSafety.databasePoolWaitMsMax + } + } + await this.incrementRegionalRehomeWorkerFailure(transaction, worker, now) + } + + private async markRegionalRehomeTickSkipped( + transaction: RelayDatabase, + now: number, + intervalMs: number + ): Promise { + await transaction.query( + `UPDATE relay_region_rehome_worker_state + SET next_dispatch_at = ?, updated_at = ? WHERE worker_id = 'global'`, + [now + intervalMs, now] + ) + } + + private async markRegionalRehomeDispatchClaimed( + transaction: RelayDatabase, + attemptId: string, + now: number, + intervalMs: number + ): Promise { + await transaction.query( + `UPDATE relay_region_rehome_attempts + SET send_attempts = send_attempts + 1, last_send_attempt_at = ?, updated_at = ? + WHERE attempt_id = ?`, + [now, now, attemptId] + ) + await transaction.query( + `UPDATE relay_region_rehome_worker_state + SET next_dispatch_at = ?, updated_at = ? WHERE worker_id = 'global'`, + [now + intervalMs, now] + ) + } + + async recordRegionalRehomeDrainReceipt( + attemptId: string, + outcome: RegionalHostDrainOutcome + ): Promise { + const now = this.now() + return await this.database.transaction(async (transaction) => { + const worker = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` + ) + )[0] + const attempt = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_attempts WHERE attempt_id = ?`, + [attemptId] + ) + )[0] + if (!attempt) throw new Error('regional_rehome_attempt_not_found') + // Any receipt proves the source cell answered: reset the failure budget + // even when a redrain repeats the stored outcome; otherwise a + // redrain-dominated stream lets scattered transient failures reach the + // durable three-failure disable. + if (worker) { + await transaction.query( + `UPDATE relay_region_rehome_worker_state + SET consecutive_failures = 0, paused_until = 0, updated_at = ? + WHERE worker_id = 'global'`, + [now] + ) + } + const existingOutcome = optionalText(attempt, 'drain_outcome') + if (existingOutcome === outcome) return false + // Redrains produce one receipt per dispatch; the latest outcome wins. + await transaction.query( + `UPDATE relay_region_rehome_attempts + SET drain_receipt_at = ?, drain_outcome = ?, updated_at = ? + WHERE attempt_id = ?`, + [now, outcome, now, attemptId] + ) + return true + }) + } + + async recordRegionalRehomeDispatchFailure(attemptId: string): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + const worker = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` + ) + )[0] + const attempt = ( + await transaction.queryLocked( + `SELECT attempt_id FROM relay_region_rehome_attempts WHERE attempt_id = ?`, + [attemptId] + ) + )[0] + if (!worker || !attempt) return + await this.incrementRegionalRehomeWorkerFailure(transaction, worker, now) + }) + } + + async recordRegionalRehomeWorkerFailure(): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + await transaction.query( + `INSERT INTO relay_region_rehome_worker_state + (worker_id, next_dispatch_at, paused_until, consecutive_failures, updated_at) + VALUES ('global', 0, 0, 0, ?) + ON CONFLICT (worker_id) DO NOTHING`, + [now] + ) + const worker = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_worker_state WHERE worker_id = 'global'` + ) + )[0]! + await this.incrementRegionalRehomeWorkerFailure(transaction, worker, now) + }) + } + + private async incrementRegionalRehomeWorkerFailure( + transaction: RelayDatabase, + worker: SqlRow, + now: number + ): Promise { + const failures = integer(worker, 'consecutive_failures') + 1 + await transaction.query( + `UPDATE relay_region_rehome_worker_state + SET consecutive_failures = ?, paused_until = ?, updated_at = ? + WHERE worker_id = 'global'`, + [failures, failures >= 3 ? now + 5 * 60_000 : 0, now] + ) + if (failures >= 3) { + await transaction.query( + `UPDATE relay_region_rehome_control + SET generation = generation + 1, enabled = 0, updated_at = ? + WHERE control_id = 'global' AND enabled = 1`, + [now] + ) + } + } + + async completeReadyRegionalRehomes(limit = 10): Promise { + const now = this.now() + const quarantined = this.quarantinedRegionalRehomeAttemptIds(now) + const exclusion = quarantined.length + ? ` AND attempt.attempt_id NOT IN (${quarantined.map(() => '?').join(', ')})` + : '' + const candidates = await this.database.query( + `SELECT attempt.attempt_id, attempt.user_id, attempt.relay_host_id, + attempt.assignment_epoch + FROM relay_region_rehome_attempts attempt + JOIN relay_assignment_migrations migration + ON migration.user_id = attempt.user_id + AND migration.relay_host_id = attempt.relay_host_id + AND migration.assignment_epoch = attempt.assignment_epoch + WHERE attempt.completed_at IS NULL AND attempt.aborted_at IS NULL + AND migration.target_registered_at IS NOT NULL + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases source_lease + WHERE source_lease.user_id = attempt.user_id + AND source_lease.relay_host_id = attempt.relay_host_id + AND source_lease.cell_id = attempt.source_cell_id + )${exclusion} + ORDER BY attempt.created_at, attempt.attempt_id + LIMIT ?`, + [...quarantined, limit] + ) + let completed = 0 + for (const candidate of candidates) { + // One poisoned row must not stall every later candidate: an invariant + // throw here blocked fleet completions head-of-line in production. + const attemptId = text(candidate, 'attempt_id') + try { + const changed = await this.completeRegionalRehomeCandidate( + { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + }, + integer(candidate, 'assignment_epoch'), + now + ) + if (changed) completed++ + this.regionalRehomeCandidateQuarantine.delete(attemptId) + } catch (error) { + this.recordRegionalRehomeCandidateFailure('complete', attemptId, now, error) + } + } + return completed + } + + // Repeated invariant failures quarantine the attempt out of the sweeps' + // LIMIT pages: poisoned rows are permanent and always the oldest, so + // without exclusion they eventually starve every healthy candidate. + private recordRegionalRehomeCandidateFailure( + operation: 'complete' | 'abort', + attemptId: string, + now: number, + error: unknown + ): void { + const entry = this.regionalRehomeCandidateQuarantine.get(attemptId) ?? { + failures: 0, + until: 0 + } + entry.failures++ + if (entry.failures >= REGIONAL_REHOME_QUARANTINE_FAILURES) { + entry.until = now + REGIONAL_REHOME_QUARANTINE_MS + } + this.regionalRehomeCandidateQuarantine.delete(attemptId) + this.regionalRehomeCandidateQuarantine.set(attemptId, entry) + if (this.regionalRehomeCandidateQuarantine.size > REGIONAL_REHOME_QUARANTINE_MEMORY_LIMIT) { + // Evict a non-quarantined entry first: mid-quarantine rows are excluded + // from the candidate pages and losing one returns it to the page with a + // reset counter. + let evict = this.regionalRehomeCandidateQuarantine.keys().next().value! + for (const [key, candidate] of this.regionalRehomeCandidateQuarantine) { + if (candidate.until <= now) { + evict = key + break + } + } + this.regionalRehomeCandidateQuarantine.delete(evict) + } + warnRegionalRehomeCandidateFailure(operation, attemptId, error) + } + + private quarantinedRegionalRehomeAttemptIds(now: number): string[] { + const excluded: string[] = [] + for (const [attemptId, entry] of this.regionalRehomeCandidateQuarantine) { + if (entry.until > now) excluded.push(attemptId) + if (excluded.length >= REGIONAL_REHOME_QUARANTINE_EXCLUSION_LIMIT) break + } + return excluded + } + + async refreshRegionalRehomeLeases(limit = 100): Promise { + const now = this.now() + const candidates = await this.database.query( + `SELECT user_id, relay_host_id, assignment_epoch + FROM relay_region_rehome_attempts + WHERE completed_at IS NULL AND aborted_at IS NULL + ORDER BY created_at, attempt_id LIMIT ?`, + [limit] + ) + let refreshed = 0 + for (const candidate of candidates) { + const identity = { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + } + const assignmentEpoch = integer(candidate, 'assignment_epoch') + const changed = await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + const attempt = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_attempts + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + const migration = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + if ( + !assignment || + !attempt || + !migration || + optionalInteger(attempt, 'completed_at') !== undefined || + optionalInteger(attempt, 'aborted_at') !== undefined || + optionalInteger(migration, 'completed_at') !== undefined || + optionalInteger(migration, 'aborted_at') !== undefined + ) { + return false + } + const attemptAgeMs = now - integer(attempt, 'created_at') + if (attemptAgeMs >= REGIONAL_REHOME_MAX_REFRESH_MS) { + return false + } + if ( + optionalInteger(migration, 'target_registered_at') === undefined && + attemptAgeMs >= REGIONAL_REHOME_UNREGISTERED_REFRESH_MS + ) { + await transaction.query( + `UPDATE relay_assignment_activity_leases + SET expires_at = CASE WHEN expires_at < ? THEN expires_at ELSE ? END, + updated_at = ? + WHERE user_id = ? AND relay_host_id = ? + AND activity_id IN (?, ?)`, + [ + now, + now, + now, + identity.userId, + identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ] + ) + await transaction.query( + `UPDATE relay_assignment_migrations + SET expires_at = CASE WHEN expires_at < ? THEN expires_at ELSE ? END, + updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + return false + } + const leases = await this.lockAssignmentActivities(transaction, identity) + const protectedIds = new Set([ + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ]) + const protectedLeases = leases.filter((lease) => + protectedIds.has(text(lease, 'activity_id')) + ) + if (protectedLeases.length === 0) return false + const expiresAt = now + ASSIGNMENT_LIMITS.migrationLeaseMs + await transaction.query( + `UPDATE relay_assignment_activity_leases + SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? + AND activity_id IN (?, ?)`, + [ + expiresAt, + now, + identity.userId, + identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ] + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [expiresAt, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + await transaction.query( + `UPDATE relay_assignments SET lease_expires_at = + CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [expiresAt, expiresAt, now, identity.userId, identity.relayHostId] + ) + return true + }) + if (changed) refreshed++ + } + return refreshed + } + + private async completeRegionalRehomeCandidate( + identity: AssignmentIdentity, + assignmentEpoch: number, + now: number + ): Promise { + return await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + const attempt = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_attempts + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + const migration = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + if ( + !attempt || + !migration || + optionalInteger(attempt, 'completed_at') !== undefined || + optionalInteger(attempt, 'aborted_at') !== undefined || + optionalInteger(migration, 'target_registered_at') === undefined || + optionalInteger(migration, 'completed_at') !== undefined || + optionalInteger(migration, 'aborted_at') !== undefined + ) { + return false + } + const sourceCellId = text(attempt, 'source_cell_id') + const targetCellId = text(attempt, 'target_cell_id') + if ( + !assignment || + text(assignment, 'cell_id') !== targetCellId || + integer(assignment, 'assignment_epoch') !== assignmentEpoch || + text(migration, 'source_cell_id') !== sourceCellId || + text(migration, 'target_cell_id') !== targetCellId + ) { + throw new Error('regional_rehome_assignment_mismatch') + } + const leases = await this.lockAssignmentActivities(transaction, identity) + if (activityUnitsForCell(leases, sourceCellId) !== 0) return false + await this.repairAssignmentActivityCounts( + transaction, + identity, + text(attempt, 'attempt_id'), + assignment, + leases, + migration + ) + const cells = await this.lockCellInventory(transaction) + const target = cells.find((cell) => text(cell, 'cell_id') === targetCellId) + const admission = await cellAdmissionStates(transaction) + if ( + !target || + integer(target, 'enabled') !== 1 || + !['general', 'migration-only'].includes(admission.get(targetCellId) ?? '') + ) { + return false + } + const targetRuntime = ( + await transaction.queryLocked( + `SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, + [targetCellId] + ) + )[0] + if ( + !targetRuntime || + integer(targetRuntime, 'ready') !== 1 || + integer(targetRuntime, 'last_heartbeat_at') <= now - this.heartbeatTtlMs + ) { + return false + } + const targetActive = leases.some( + (lease) => + text(lease, 'cell_id') === targetCellId && + text(lease, 'activity_kind') === 'control' && + !text(lease, 'activity_id').startsWith('control-pending:') && + integer(lease, 'expires_at') > now && + integer(lease, 'updated_at') >= integer(targetRuntime, 'started_at') + ) + if (!targetActive) return false + const migrationLease = activityLeaseById( + leases, + migrationActivityId(assignmentEpoch) + ) + if (migrationLease) { + await this.removeActivityLease(transaction, identity, migrationLease, now) + } + await transaction.query( + `UPDATE relay_assignment_migrations SET completed_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + await transaction.query( + `UPDATE relay_region_rehome_attempts SET completed_at = ?, updated_at = ? + WHERE attempt_id = ?`, + [now, now, text(attempt, 'attempt_id')] + ) + return true + }) + } + + // Why: counter skew left by a pre-fix sticky grant is reconstructible from the + // locked lease rows; lease shape and migration topology are not, so only a + // count mismatch is repaired and the re-assert still throws on anything else. + // reconcileReservationAccounting cannot stand in: it opens its own + // transaction, while this has to run inside the sweep's on rows it holds. + private async repairAssignmentActivityCounts( + transaction: RelayDatabase, + identity: AssignmentIdentity, + attemptId: string, + assignment: SqlRow, + leases: SqlRow[], + migrationRow: SqlRow + ): Promise { + try { + assertAssignmentActivityAccounting(assignment, leases, migrationRow) + return + } catch (error) { + if ( + !(error instanceof Error) || + error.message !== 'migration_activity_accounting_mismatch' + ) { + throw error + } + } + const counts = activityCounts(leases) + await transaction.query( + `UPDATE relay_assignments SET reserved_controls = ?, reserved_splices = ?, + reserved_invites = ?, pending_installs = ?, pending_confirmations = ?, + migration_leases = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + counts.control, + counts.splice, + counts.invite, + counts.install, + counts.confirmation, + counts.migration, + identity.userId, + identity.relayHostId + ] + ) + const repaired = await this.assignmentRow(transaction, identity) + if (!repaired) throw new Error('regional_rehome_assignment_mismatch') + assertAssignmentActivityAccounting(repaired, leases, migrationRow) + noteRegionalRehomeActivityCountsRepaired(attemptId) + } + + async reapRegionalRehomeAttempts(): Promise { + const now = this.now() + const completed = await this.database.query( + `UPDATE relay_region_rehome_attempts + SET completed_at = COALESCE(completed_at, ?), updated_at = ? + WHERE completed_at IS NULL AND aborted_at IS NULL + AND EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = relay_region_rehome_attempts.user_id + AND migration.relay_host_id = relay_region_rehome_attempts.relay_host_id + AND migration.assignment_epoch = relay_region_rehome_attempts.assignment_epoch + AND migration.completed_at IS NOT NULL + )`, + [now, now] + ) + const aborted = await this.database.query( + `UPDATE relay_region_rehome_attempts + SET aborted_at = COALESCE(aborted_at, ?), updated_at = ? + WHERE completed_at IS NULL AND aborted_at IS NULL + AND EXISTS ( + SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = relay_region_rehome_attempts.user_id + AND migration.relay_host_id = relay_region_rehome_attempts.relay_host_id + AND migration.assignment_epoch = relay_region_rehome_attempts.assignment_epoch + AND migration.aborted_at IS NOT NULL + )`, + [now, now] + ) + if (integer(aborted[0]!, 'changes') > 0) { + await this.disableRegionalRehomeControl() + } + return integer(completed[0]!, 'changes') + integer(aborted[0]!, 'changes') + } + + async abortExpiredRegionalRehomes(limit = 100): Promise { + const now = this.now() + const quarantined = this.quarantinedRegionalRehomeAttemptIds(now) + const exclusion = quarantined.length + ? ` AND attempt_id NOT IN (${quarantined.map(() => '?').join(', ')})` + : '' + const candidates = await this.database.query( + `SELECT attempt_id, user_id, relay_host_id, assignment_epoch + FROM relay_region_rehome_attempts + WHERE completed_at IS NULL AND aborted_at IS NULL + AND created_at <= ?${exclusion} + ORDER BY created_at, attempt_id LIMIT ?`, + [now - REGIONAL_REHOME_MAX_REFRESH_MS, ...quarantined, limit] + ) + let aborted = 0 + for (const candidate of candidates) { + const identity = { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + } + const assignmentEpoch = integer(candidate, 'assignment_epoch') + const attemptId = text(candidate, 'attempt_id') + // Isolated like the completion sweep: one poisoned row must not stall + // every later candidate. + let changed = false + try { + changed = await this.database.transaction(async (transaction) => { + const assignment = await this.assignmentRow(transaction, identity) + const attempt = ( + await transaction.queryLocked( + `SELECT * FROM relay_region_rehome_attempts + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + const migration = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + if ( + !assignment || + !attempt || + !migration || + optionalInteger(attempt, 'completed_at') !== undefined || + optionalInteger(attempt, 'aborted_at') !== undefined || + integer(attempt, 'created_at') > now - REGIONAL_REHOME_MAX_REFRESH_MS || + optionalInteger(migration, 'completed_at') !== undefined || + optionalInteger(migration, 'aborted_at') !== undefined + ) { + return false + } + const sourceCellId = text(attempt, 'source_cell_id') + const targetCellId = text(attempt, 'target_cell_id') + if ( + text(assignment, 'cell_id') !== targetCellId || + integer(assignment, 'assignment_epoch') !== assignmentEpoch + ) { + throw new Error('regional_rehome_assignment_mismatch') + } + const leases = await this.lockAssignmentActivities(transaction, identity) + if (activityUnitsForCell(leases, sourceCellId) > 0) return false + const targetActive = leases.some( + (lease) => + text(lease, 'cell_id') === targetCellId && + text(lease, 'activity_kind') === 'control' && + !text(lease, 'activity_id').startsWith('control-pending:') && + integer(lease, 'expires_at') > now + ) + if (targetActive) return false + const cells = await this.lockCellInventory(transaction) + const source = cells.find((cell) => text(cell, 'cell_id') === sourceCellId) + const admission = await cellAdmissionStates(transaction) + if ( + !source || + integer(source, 'enabled') !== 1 || + admission.get(sourceCellId) !== 'general' || + !(await this.cellIsLive(transaction, sourceCellId, now)) + ) { + return false + } + for (const activityId of [ + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ]) { + const lease = activityLeaseById(leases, activityId) + if (lease) await this.removeActivityLease(transaction, identity, lease, now) + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + targetCellId, + assignmentEpoch, + now + ) + await transaction.query( + `UPDATE relay_assignments SET cell_id = ?, assignment_epoch = ?, + lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + sourceCellId, + assignmentEpoch + 1, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now, + identity.userId, + identity.relayHostId + ] + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + await transaction.query( + `UPDATE relay_region_rehome_attempts SET aborted_at = ?, updated_at = ? + WHERE attempt_id = ?`, + [now, now, text(attempt, 'attempt_id')] + ) + await transaction.query( + `UPDATE relay_region_rehome_control + SET generation = generation + 1, enabled = 0, updated_at = ? + WHERE control_id = 'global' AND enabled = 1`, + [now] + ) + return true + }) + this.regionalRehomeCandidateQuarantine.delete(attemptId) + } catch (error) { + this.recordRegionalRehomeCandidateFailure('abort', attemptId, now, error) + } + if (changed) aborted++ + } + return aborted + } + + async abortExpiredEvacuations(): Promise { + const now = this.now() + const abandonedBefore = now - STRANDED_MIGRATION_ABANDON_MS + const candidates = await this.database.query( + `SELECT migration.user_id, migration.relay_host_id, migration.assignment_epoch + FROM relay_assignment_migrations migration + WHERE migration.expires_at <= ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + AND ${ABORTABLE_EXPIRED_MIGRATION} + ORDER BY user_id, relay_host_id, assignment_epoch`, + [now, now, abandonedBefore, abandonedBefore] + ) + let aborted = 0 + for (const candidate of candidates) { + const didAbort = await this.database.transaction(async (transaction) => { + const identity = { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + } + // Migration cleanup follows the same assignment-first order as evacuation. + const assignment = await this.assignmentRow(transaction, identity) + const assignmentEpoch = integer(candidate, 'assignment_epoch') + const regionalAttempt = ( + await transaction.queryLocked( + `SELECT attempt_id FROM relay_region_rehome_attempts + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND completed_at IS NULL AND aborted_at IS NULL`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + )[0] + const row = ( + await transaction.queryLocked( + `SELECT migration.* FROM relay_assignment_migrations migration + WHERE migration.user_id = ? AND migration.relay_host_id = ? + AND migration.assignment_epoch = ? + AND migration.expires_at <= ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + AND ${ABORTABLE_EXPIRED_MIGRATION}`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + now, + now, + abandonedBefore, + abandonedBefore + ] + ) + )[0] + if (!row) return false + const targetCellId = text(row, 'target_cell_id') + const activityLeases = await this.lockAssignmentActivities(transaction, identity) + if (!assignment) throw new Error('migration_assignment_missing') + const currentAssignmentEpoch = integer(assignment, 'assignment_epoch') + const assignmentEpochMatches = + text(assignment, 'cell_id') === targetCellId && + currentAssignmentEpoch === assignmentEpoch + const pendingTargetControl = activityLeaseById( + activityLeases, + pendingControlActivityId(assignmentEpoch) + ) + const targetGrantIsFresh = + assignmentEpochMatches && + pendingTargetControl !== undefined && + text(pendingTargetControl, 'cell_id') === targetCellId && + text(pendingTargetControl, 'activity_kind') === 'control' && + integer(pendingTargetControl, 'expires_at') > now + const targetIsActive = activityLeases.some( + (lease) => + text(lease, 'cell_id') === targetCellId && + text(lease, 'activity_kind') === 'control' && + text(lease, 'activity_id') !== pendingControlActivityId(assignmentEpoch) + ) + if (targetGrantIsFresh) return false + if (targetIsActive && assignmentEpochMatches) { + // A committed target control is stronger evidence than a failed follow-up + // write; repair the marker instead of rolling a live desktop backward. + await transaction.query( + `UPDATE relay_assignment_migrations + SET target_registered_at = COALESCE(target_registered_at, ?), updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + return false + } + if (!assignmentEpochMatches) { + if (currentAssignmentEpoch <= assignmentEpoch) { + throw new Error('migration_assignment_mismatch') + } + // A newer assignment is authoritative regardless of where it landed. + // Retire only this obsolete migration; never rewrite the newer epoch. + const obsoleteLeases = [ + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ] + .map((activityId) => activityLeaseById(activityLeases, activityId)) + .filter((lease): lease is SqlRow => lease !== undefined) + if (obsoleteLeases.length > 0) await this.lockCellInventory(transaction) + for (const lease of obsoleteLeases) { + await this.removeActivityLease(transaction, identity, lease, now) + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + targetCellId, + assignmentEpoch, + now + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + return true + } + const cells = await this.lockCellInventory(transaction) + const sourceCellId = text(row, 'source_cell_id') + const admissionRows = await transaction.query( + `SELECT cell_id, admission_state, updated_at FROM relay_cell_admission + WHERE cell_id IN (?, ?)`, + [sourceCellId, targetCellId] + ) + const sourceCell = cells.find((cell) => text(cell, 'cell_id') === sourceCellId) + const targetCell = cells.find((cell) => text(cell, 'cell_id') === targetCellId) + const sourceAdmission = admissionRows.find( + (admission) => text(admission, 'cell_id') === sourceCellId + ) + const targetAdmission = admissionRows.find( + (admission) => text(admission, 'cell_id') === targetCellId + ) + const registered = optionalInteger(row, 'target_registered_at') !== undefined + const sourceIsDurablyFenced = + registered && + ( + await transaction.query( + `SELECT 1 FROM relay_assignment_migrations migration + WHERE migration.user_id = ? AND migration.relay_host_id = ? + AND migration.assignment_epoch = ? + AND ${DURABLY_FENCED_MIGRATION_SOURCE}`, + [identity.userId, identity.relayHostId, assignmentEpoch] + ) + ).length === 1 + const retireOnTarget = + registered && + activityUnitsForCell(activityLeases, sourceCellId) === 0 && + sourceCell !== undefined && + integer(sourceCell, 'enabled') === 0 && + sourceAdmission !== undefined && + text(sourceAdmission, 'admission_state') === 'existing-only' && + (integer(sourceAdmission, 'updated_at') <= abandonedBefore || + sourceIsDurablyFenced) && + targetCell !== undefined && + integer(targetCell, 'enabled') === 1 && + targetAdmission !== undefined && + ['migration-only', 'general'].includes(text(targetAdmission, 'admission_state')) + const rollbackReason = + !registered || + (targetCell !== undefined && + integer(targetCell, 'enabled') === 0 && + targetAdmission !== undefined && + text(targetAdmission, 'admission_state') === 'existing-only' && + integer(targetAdmission, 'updated_at') <= abandonedBefore) + const regionalRollbackSourceAvailable = + !regionalAttempt || + (sourceCell !== undefined && + integer(sourceCell, 'enabled') === 1 && + sourceAdmission !== undefined && + text(sourceAdmission, 'admission_state') === 'general' && + (await this.cellIsLive(transaction, sourceCellId, now))) + const rollbackToSource = rollbackReason && regionalRollbackSourceAvailable + if (!retireOnTarget && !rollbackToSource) return false + for (const activityId of [ + pendingControlActivityId(assignmentEpoch), + migrationActivityId(assignmentEpoch) + ]) { + const lease = activityLeaseById(activityLeases, activityId) + if (lease) await this.removeActivityLease(transaction, identity, lease, now) + } + await this.releaseSupersededControlConnectionReservations( + transaction, + identity, + targetCellId, + assignmentEpoch, + now + ) + if (retireOnTarget) { + await transaction.query( + `UPDATE relay_assignment_migrations SET completed_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + return true + } + await transaction.query( + `UPDATE relay_assignments SET cell_id = ?, assignment_epoch = ?, + lease_expires_at = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + sourceCellId, + assignmentEpoch + 1, + now + ASSIGNMENT_LIMITS.activityLeaseMs, + now, + identity.userId, + identity.relayHostId + ] + ) + await transaction.query( + `UPDATE relay_assignment_migrations SET aborted_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, now, identity.userId, identity.relayHostId, assignmentEpoch] + ) + return true + }) + if (didAbort) aborted++ + } + return aborted + } + + async releaseExpiredActivityLeases(): Promise { + const now = this.now() + await this.database.query( + `UPDATE relay_control_connection_reservations + SET state = 'late-arrival-debt', updated_at = ? + WHERE state = 'reserved' AND timeout_at <= ?`, + [now, now] + ) + await this.database.query( + `UPDATE relay_control_connection_reservations + SET state = 'released', released_at = ?, updated_at = ? + WHERE state = 'late-arrival-debt' + AND claim_activity_id IS NULL + AND timeout_at <= ?`, + [now, now, now - LATE_ARRIVAL_DEBT_RETENTION_MS] + ) + const candidates = await this.database.query( + `SELECT user_id, relay_host_id, activity_id + FROM relay_assignment_activity_leases lease + WHERE expires_at <= ? + AND NOT EXISTS ( + SELECT 1 FROM relay_region_rehome_attempts rehome + WHERE rehome.user_id = lease.user_id + AND rehome.relay_host_id = lease.relay_host_id + AND rehome.completed_at IS NULL AND rehome.aborted_at IS NULL + AND lease.activity_id IN ( + 'control-pending:' || CAST(rehome.assignment_epoch AS TEXT), + 'migration:' || CAST(rehome.assignment_epoch AS TEXT) + ) + ) + AND NOT EXISTS ( + SELECT 1 FROM relay_post_drain_migration_pins pin + JOIN relay_assignment_migrations migration + ON migration.user_id = pin.user_id + AND migration.relay_host_id = pin.relay_host_id + AND migration.assignment_epoch = pin.assignment_epoch + WHERE pin.user_id = lease.user_id + AND pin.relay_host_id = lease.relay_host_id + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + AND lease.activity_id IN ( + 'control-pending:' || CAST(pin.assignment_epoch AS TEXT), + 'migration:' || CAST(pin.assignment_epoch AS TEXT) + ) + )`, + [now] + ) + let released = 0 + for (const candidate of candidates) { + let didRelease: boolean + try { + didRelease = await this.database.transaction(async (transaction) => { + const identity = { + userId: text(candidate, 'user_id'), + relayHostId: text(candidate, 'relay_host_id') + } + // Why: re-check under the canonical assignment-first lock so a concurrent + // renewal wins without cleanup reaping its refreshed lease. + // Several directors sweep the same expired rows. Skip a row another + // director is settling instead of waiting and retrying the transaction. + await this.assignmentRow(transaction, identity, true) + const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) + const lease = activityLeaseById(activityLeases, text(candidate, 'activity_id')) + if (!lease || integer(lease, 'expires_at') > now) return false + await this.lockCellInventory(transaction, true) + await this.removeActivityLease(transaction, identity, lease, now) + return true + }) + } catch (error) { + // Released cell images can still hold their legacy cell-first lock; + // expiry is durable, so a later maintenance sweep can safely retry it. + if (isDatabaseLockUnavailable(error)) continue + throw error + } + if (didRelease) released++ + } + return released + } + + async releaseExpiredActivity(): Promise { + const now = this.now() + try { + return await this.database.transaction(async (transaction) => { + const expired = await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE lease_expires_at <= ? AND + (reserved_controls > 0 OR reserved_splices > 0 OR reserved_invites > 0 OR + pending_installs > 0 OR pending_confirmations > 0 OR migration_leases > 0) + AND NOT EXISTS ( + SELECT 1 FROM relay_region_rehome_attempts rehome + WHERE rehome.user_id = relay_assignments.user_id + AND rehome.relay_host_id = relay_assignments.relay_host_id + AND rehome.completed_at IS NULL AND rehome.aborted_at IS NULL + ) + AND NOT EXISTS ( + SELECT 1 FROM relay_post_drain_migration_pins pin + JOIN relay_assignment_migrations migration + ON migration.user_id = pin.user_id + AND migration.relay_host_id = pin.relay_host_id + AND migration.assignment_epoch = pin.assignment_epoch + WHERE pin.user_id = relay_assignments.user_id + AND pin.relay_host_id = relay_assignments.relay_host_id + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) + ORDER BY user_id, relay_host_id`, + [now], + { failIfUnavailable: true } + ) + if (expired.length > 0) await this.lockCellInventory(transaction, true) + for (const row of expired) { + await this.adjustCellReservation(transaction, text(row, 'cell_id'), -requestUnits(row)) + await transaction.query( + `UPDATE relay_assignments SET reserved_controls = 0, reserved_splices = 0, + reserved_invites = 0, pending_installs = 0, pending_confirmations = 0, + migration_leases = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [text(row, 'user_id'), text(row, 'relay_host_id')] + ) + } + return expired.length + }) + } catch (error) { + // Aggregate expiry is reconstructible from durable leases; never wait + // long enough to form a mixed-version cell/assignment lock cycle. + if (isDatabaseLockUnavailable(error)) return 0 + throw error + } + } + + async releaseExpiredRegionPreferences(): Promise { + const result = await this.database.query( + `DELETE FROM relay_assignment_region_preferences WHERE observed_at < ?`, + [this.now() - REGION_PREFERENCE_RETENTION_MS] + ) + return integer(result[0]!, 'changes') + } + + private async reconcileReservationAccounting( + sourceCellId: string, + targetCellId: string + ): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + // Reconciliation takes the same assignment→activity→cell order as live + // mutations so correcting drift never races a credential or socket lease. + const assignments = await transaction.queryLocked( + `SELECT assignment.* FROM relay_assignments assignment + WHERE assignment.cell_id IN (?, ?) OR EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases scoped_lease + WHERE scoped_lease.user_id = assignment.user_id + AND scoped_lease.relay_host_id = assignment.relay_host_id + AND scoped_lease.cell_id IN (?, ?) + ) + ORDER BY assignment.user_id, assignment.relay_host_id`, + [sourceCellId, targetCellId, sourceCellId, targetCellId] + ) + const leases = await transaction.queryLocked( + `SELECT lease.* FROM relay_assignment_activity_leases lease + WHERE lease.cell_id IN (?, ?) OR EXISTS ( + SELECT 1 FROM relay_assignments assignment + WHERE assignment.user_id = lease.user_id + AND assignment.relay_host_id = lease.relay_host_id + AND ( + assignment.cell_id IN (?, ?) OR EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases scoped_lease + WHERE scoped_lease.user_id = lease.user_id + AND scoped_lease.relay_host_id = lease.relay_host_id + AND scoped_lease.cell_id IN (?, ?) + ) + ) + ) + ORDER BY lease.user_id, lease.relay_host_id, lease.activity_id`, + [ + sourceCellId, + targetCellId, + sourceCellId, + targetCellId, + sourceCellId, + targetCellId + ] + ) + const cells = await this.lockCellInventory(transaction) + const assignmentKeys = new Set( + assignments.map((row) => + assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id')) + ) + ) + const cellIds = new Set(cells.map((row) => text(row, 'cell_id'))) + const assignmentCounts = new Map< + string, + { counts: Record; leaseExpiresAt: number } + >() + const cellUnits = new Map() + + for (const lease of leases) { + const key = assignmentKey(text(lease, 'user_id'), text(lease, 'relay_host_id')) + if (!assignmentKeys.has(key)) throw new Error('activity_lease_assignment_missing') + const cellId = text(lease, 'cell_id') + if (!cellIds.has(cellId)) throw new Error('activity_lease_cell_missing') + const current = + assignmentCounts.get(key) ?? { + counts: emptyActivityCounts(), + leaseExpiresAt: 0 + } + const kind = activityKind(lease) + current.counts[kind]++ + current.leaseExpiresAt = Math.max(current.leaseExpiresAt, integer(lease, 'expires_at')) + assignmentCounts.set(key, current) + cellUnits.set(cellId, (cellUnits.get(cellId) ?? 0) + integer(lease, 'request_units')) + } + + for (const row of assignments) { + const current = assignmentCounts.get( + assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id')) + ) + const counts = current?.counts ?? emptyActivityCounts() + const differs = (Object.keys(ACTIVITY_COLUMN) as AssignmentActivityKind[]).some( + (kind) => integer(row, ACTIVITY_COLUMN[kind]) !== counts[kind] + ) + const expiryDiffers = + current !== undefined && integer(row, 'lease_expires_at') !== current.leaseExpiresAt + if (!differs && !expiryDiffers) continue + await transaction.query( + `UPDATE relay_assignments SET reserved_controls = ?, reserved_splices = ?, + reserved_invites = ?, pending_installs = ?, pending_confirmations = ?, + migration_leases = ?, lease_expires_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [ + counts.control, + counts.splice, + counts.invite, + counts.install, + counts.confirmation, + counts.migration, + current?.leaseExpiresAt ?? integer(row, 'lease_expires_at'), + text(row, 'user_id'), + text(row, 'relay_host_id') + ] + ) + } + + for (const row of cells.filter((cell) => + [sourceCellId, targetCellId].includes(text(cell, 'cell_id')) + )) { + const cellId = text(row, 'cell_id') + const expected = cellUnits.get(cellId) ?? 0 + if (expected > integer(row, 'capacity_requests')) { + throw new Error('relay_capacity_exhausted') + } + if (integer(row, 'reserved_requests') === expected) continue + await transaction.query( + `UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`, + [expected, now, cellId] + ) + } + }) + } + + private async lockCellInventory( + database: RelayDatabase, + failIfUnavailable = false + ): Promise { + // Every capacity-changing assignment takes the tiny cell inventory in one + // order; dynamically locking only the selected target allowed cross-cell cycles. + return await database.queryLocked( + `SELECT * FROM relay_cells ORDER BY cell_id ASC`, + [], + { failIfUnavailable } + ) + } + + private async lockGeneralCellInventory( + database: RelayDatabase, + failIfUnavailable = false + ): Promise { + return await database.queryLocked( + `SELECT * FROM relay_cells + WHERE cell_id IN ( + SELECT cell_id FROM relay_cell_admission WHERE admission_state = 'general' + ) + ORDER BY cell_id ASC`, + [], + { failIfUnavailable } + ) + } + + private async leastLoadedCell( + database: RelayDatabase, + lockedCells: SqlRow[] | undefined, + preferredRegion: RelayRegion + ): Promise { + const rows = lockedCells ?? (await this.lockCellInventory(database)) + const regions = new Map( + (await database.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ + text(row, 'cell_id'), + relayRegion(row, 'region') + ]) + ) + const admission = await cellAdmissionStates(database) + const runtimeLoad = this.requireLiveCells + ? new Map( + ( + await database.query( + `SELECT cell_id, observed_requests FROM relay_cell_runtime + WHERE ready = ? AND last_heartbeat_at > ?`, + [1, this.now() - this.heartbeatTtlMs] + ) + ).map((row) => [text(row, 'cell_id'), integer(row, 'observed_requests')]) + ) + : null + const connectionHeadroom = await this.connectionHeadroomByCell(database) + const candidates = rows.filter((row) => { + const cellId = text(row, 'cell_id') + return ( + admission.get(cellId) === 'general' && + integer(row, 'reserved_requests') < integer(row, 'capacity_requests') && + (!runtimeLoad || runtimeLoad.has(cellId)) && + connectionHeadroom.get(cellId) !== false + ) + }) + candidates.sort((left, right) => { + const leftLoad = + integer(left, 'reserved_requests') + + (runtimeLoad?.get(text(left, 'cell_id')) ?? integer(left, 'observed_requests')) + const rightLoad = + integer(right, 'reserved_requests') + + (runtimeLoad?.get(text(right, 'cell_id')) ?? integer(right, 'observed_requests')) + const loadDifference = + leftLoad / integer(left, 'capacity_requests') - + rightLoad / integer(right, 'capacity_requests') + return loadDifference || text(left, 'cell_id').localeCompare(text(right, 'cell_id')) + }) + const preferred = candidates.filter( + (candidate) => + (regions.get(text(candidate, 'cell_id')) ?? RELAY_DEFAULT_REGION) === preferredRegion + ) + const selected = preferred[0] ?? candidates[0] + return selected + ? cell(selected, regions.get(text(selected, 'cell_id')) ?? RELAY_DEFAULT_REGION) + : null + } + + async regionCatalog(): Promise { + const rows = await this.database.query( + `SELECT cell.cell_id, cell.cell_url, region.region + FROM relay_cells cell + LEFT JOIN relay_cell_regions region ON region.cell_id = cell.cell_id + JOIN relay_cell_admission admission ON admission.cell_id = cell.cell_id + JOIN relay_cell_runtime runtime ON runtime.cell_id = cell.cell_id + WHERE cell.enabled = 1 + AND admission.admission_state = 'general' + AND runtime.ready = ? AND runtime.last_heartbeat_at > ? + ORDER BY cell.cell_id ASC`, + [1, this.now() - this.heartbeatTtlMs] + ) + const origins = new Map() + for (const row of rows) { + const region = optionalRelayRegion(row, 'region') ?? RELAY_DEFAULT_REGION + const current = origins.get(region) ?? [] + if (current.length < 2) current.push(text(row, 'cell_url')) + origins.set(region, current) + } + return RELAY_REGIONS.flatMap((region) => { + const probeOrigins = origins.get(region) + return probeOrigins?.length ? [{ region, probeOrigins }] : [] + }) + } + + private async recordRegionPreference( + database: RelayDatabase, + identity: AssignmentIdentity, + preferredRegion: RelayRegion | undefined, + observedAt: number + ): Promise { + if (!preferredRegion) return + await database.query( + `INSERT INTO relay_assignment_region_preferences + (user_id, relay_host_id, preferred_region, observed_at) + VALUES (?, ?, ?, ?) + ON CONFLICT (user_id, relay_host_id) DO UPDATE SET + preferred_region = CASE + WHEN excluded.observed_at >= relay_assignment_region_preferences.observed_at + THEN excluded.preferred_region + ELSE relay_assignment_region_preferences.preferred_region + END, + observed_at = CASE + WHEN excluded.observed_at >= relay_assignment_region_preferences.observed_at + THEN excluded.observed_at + ELSE relay_assignment_region_preferences.observed_at + END`, + [identity.userId, identity.relayHostId, preferredRegion, observedAt] + ) + } + + private async cellRegion(database: RelayDatabase, cellId: string): Promise { + const row = ( + await database.query(`SELECT region FROM relay_cell_regions WHERE cell_id = ?`, [cellId]) + )[0] + return row ? relayRegion(row, 'region') : RELAY_DEFAULT_REGION + } + + private async connectionHeadroomByCell( + database: RelayDatabase + ): Promise> { + const now = this.now() + const rows = await database.query(ASSIGNMENT_CONNECTION_HEADROOM_QUERY) + return new Map( + rows.map((row) => { + const heartbeat = optionalInteger(row, 'last_heartbeat_at') + const hasFreshTelemetry = + heartbeat !== undefined && + heartbeat > now - this.heartbeatTtlMs && + optionalText(row, 'connection_incarnation') === + optionalText(row, 'current_incarnation') + const hasHeadroom = + hasFreshTelemetry && + integer(row, 'enforced_connection_units') + + integer(row, 'outstanding_reservations') + + integer(row, 'unobserved_bound') < + integer(row, 'hard_cap') - + RELAY_ADMISSION_BUDGETS.reservedHostControls + return [text(row, 'cell_id'), hasHeadroom] as const + }) + ) + } + + private async assertCellConnectionHeadroom( + database: RelayDatabase, + cellId: string + ): Promise { + if (!(await this.cellHasConnectionHeadroom(database, cellId))) { + throw new Error('relay_connection_headroom_exhausted') + } + } + + private async cellHasConnectionHeadroom( + database: RelayDatabase, + cellId: string + ): Promise { + return (await this.connectionHeadroomByCell(database)).get(cellId) !== false + } + + private async cellIsLive( + database: RelayDatabase, + cellId: string, + now: number + ): Promise { + if (!this.requireLiveCells) return true + const rows = await database.query( + `SELECT cell.cell_id FROM relay_cells cell + JOIN relay_cell_runtime runtime ON runtime.cell_id = cell.cell_id + WHERE cell.cell_id = ? AND runtime.ready = ? AND runtime.last_heartbeat_at > ?`, + [cellId, 1, now - this.heartbeatTtlMs] + ) + return rows.length === 1 + } + + private async cellHasActiveFence(cellId: string): Promise { + const rows = await this.database.query( + `SELECT fence.cell_id FROM relay_cell_fences fence + JOIN relay_cells cell ON cell.cell_id = fence.cell_id + JOIN relay_cell_runtime runtime ON runtime.cell_id = fence.cell_id + WHERE fence.cell_id = ? AND cell.enabled = 0 + AND fence.cell_incarnation = runtime.cell_incarnation + AND fence.attested_at >= runtime.last_heartbeat_at + AND fence.expires_at > ?`, + [cellId, this.now()] + ) + return rows.length === 1 + } + + private async recordCommittedCellFence( + transaction: RelayDatabase, + cellId: string, + cellIncarnation: string, + attemptId: string, + attestedAt: number, + expiresAt: number + ): Promise { + await transaction.query( + `INSERT INTO relay_cell_committed_fences + (cell_id, attempt_id, cell_incarnation, attested_at, expires_at) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT (cell_id) DO UPDATE SET + attempt_id = excluded.attempt_id, + cell_incarnation = excluded.cell_incarnation, + attested_at = excluded.attested_at, + expires_at = excluded.expires_at`, + [cellId, attemptId, cellIncarnation, attestedAt, expiresAt] + ) + } + + private async cellHasCommittedFence( + database: RelayDatabase, + cellId: string, + now: number + ): Promise { + const rows = await database.query( + `SELECT committed.cell_id + FROM relay_cell_committed_fences committed + JOIN relay_cell_fence_attempts attempt + ON attempt.attempt_id = committed.attempt_id + JOIN relay_cell_fences fence ON fence.cell_id = committed.cell_id + JOIN relay_cell_runtime runtime ON runtime.cell_id = committed.cell_id + JOIN relay_cells cell ON cell.cell_id = committed.cell_id + WHERE committed.cell_id = ? + AND attempt.completed_at IS NOT NULL + AND attempt.aborted_at IS NULL + AND cell.enabled = 0 + AND committed.cell_incarnation = runtime.cell_incarnation + AND fence.cell_incarnation = committed.cell_incarnation + AND committed.attested_at >= runtime.last_heartbeat_at + AND committed.expires_at > ? + AND fence.expires_at > ?`, + [cellId, now, now] + ) + return rows.length === 1 + } + + private async deadCellRequiresCommittedFence( + database: RelayDatabase, + identity: AssignmentIdentity, + cellId: string, + assignmentEpoch: number + ): Promise { + const row = ( + await database.query( + `SELECT + CASE WHEN EXISTS ( + SELECT 1 FROM relay_cell_connection_limits limits + WHERE limits.cell_id = ? + ) THEN 1 ELSE 0 END AS capped, + CASE WHEN EXISTS ( + SELECT 1 FROM relay_post_drain_migration_pins pin + JOIN relay_assignment_migrations migration + ON migration.user_id = pin.user_id + AND migration.relay_host_id = pin.relay_host_id + AND migration.assignment_epoch = pin.assignment_epoch + WHERE pin.user_id = ? AND pin.relay_host_id = ? + AND pin.assignment_epoch = ? + AND pin.target_cell_id = ? + AND migration.completed_at IS NULL + AND migration.aborted_at IS NULL + ) THEN 1 ELSE 0 END AS pinned`, + [cellId, identity.userId, identity.relayHostId, assignmentEpoch, cellId] + ) + )[0] + if (!row) return false + return integer(row, 'capped') === 1 || integer(row, 'pinned') === 1 + } + + private async assertDrainCellGeneration( + transaction: RelayDatabase, + cellId: string, + cellIncarnation: string + ): Promise { + const cell = ( + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellId]) + )[0] + const runtime = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_runtime WHERE cell_id = ?`, [ + cellId + ]) + )[0] + if (!cell || integer(cell, 'enabled') !== 0) { + throw new Error('drain_attempt_admission_enabled') + } + if (!runtime || text(runtime, 'cell_incarnation') !== cellIncarnation) { + throw new Error('drain_attempt_generation_mismatch') + } + } + + private async requireCellFence( + transaction: RelayDatabase, + cellId: string, + runtime: SqlRow | undefined, + now: number + ): Promise { + const fence = ( + await transaction.queryLocked(`SELECT * FROM relay_cell_fences WHERE cell_id = ?`, [ + cellId + ]) + )[0] + if ( + !fence || + !runtime || + text(fence, 'cell_incarnation') !== text(runtime, 'cell_incarnation') || + integer(fence, 'attested_at') < integer(runtime, 'last_heartbeat_at') || + integer(fence, 'expires_at') <= now + ) { + throw new Error('cell_fence_attestation_missing') + } + } + + private async activeCellMigrations(sourceCellId: string, targetCellId: string): Promise { + const targetRuntimeSafety = this.requireLiveCells + ? `AND EXISTS ( + SELECT 1 FROM relay_cell_runtime target_runtime + WHERE target_runtime.cell_id = migration.target_cell_id + AND lease.updated_at >= target_runtime.started_at + )` + : '' + return await this.database.query( + `SELECT migration.*, assignment.cell_id AS current_cell_id, + assignment.assignment_epoch AS current_assignment_epoch, + COALESCE(( + SELECT SUM(source_lease.request_units) + FROM relay_assignment_activity_leases source_lease + WHERE source_lease.user_id = migration.user_id + AND source_lease.relay_host_id = migration.relay_host_id + AND source_lease.cell_id = migration.source_cell_id + ), 0) AS source_activity_units, + CASE WHEN EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases lease + WHERE lease.user_id = migration.user_id + AND lease.relay_host_id = migration.relay_host_id + AND lease.cell_id = migration.target_cell_id + AND lease.activity_kind = 'control' + AND lease.activity_id NOT LIKE 'control-pending:%' + ) THEN 1 ELSE 0 END AS target_control_active, + CASE WHEN EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases lease + WHERE lease.user_id = migration.user_id + AND lease.relay_host_id = migration.relay_host_id + AND lease.cell_id = migration.target_cell_id + AND lease.activity_kind = 'control' + AND lease.activity_id NOT LIKE 'control-pending:%' + AND lease.expires_at > ? + ${targetRuntimeSafety} + ) THEN 1 ELSE 0 END AS target_control_current + FROM relay_assignment_migrations migration + LEFT JOIN relay_assignments assignment + ON assignment.user_id = migration.user_id + AND assignment.relay_host_id = migration.relay_host_id + WHERE migration.source_cell_id = ? AND migration.target_cell_id = ? + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + ORDER BY migration.user_id, migration.relay_host_id`, + [this.now(), sourceCellId, targetCellId] + ) + } + + private async insertPendingControlLease( + database: RelayDatabase, + identity: AssignmentIdentity, + cellId: string, + assignmentEpoch: number, + now: number + ): Promise { + const timeoutAt = now + ASSIGNMENT_LIMITS.activityLeaseMs + await database.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (user_id, relay_host_id, activity_id) DO UPDATE SET + expires_at = excluded.expires_at, updated_at = excluded.updated_at`, + [ + identity.userId, + identity.relayHostId, + pendingControlActivityId(assignmentEpoch), + 'control', + cellId, + 1, + timeoutAt, + now + ] + ) + await this.insertControlConnectionReservation( + database, + identity, + cellId, + assignmentEpoch, + timeoutAt, + now + ) + } + + private async insertControlConnectionReservation( + database: RelayDatabase, + identity: AssignmentIdentity, + cellId: string, + assignmentEpoch: number, + timeoutAt: number, + now: number + ): Promise { + const existing = await database.queryLocked( + `SELECT reservation_id + FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND cell_id = ? AND state IN ('reserved', 'late-arrival-debt') + AND claim_activity_id IS NULL + ORDER BY created_at ASC, reservation_id ASC + LIMIT 1`, + [identity.userId, identity.relayHostId, assignmentEpoch, cellId] + ) + if (existing.length > 0) return + const reservationId = randomUUID() + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + SELECT ?, ?, ?, ?, ?, ?, 'reserved', NULL, NULL, ?, ?, NULL, NULL, ? + FROM relay_cell_connection_limits WHERE cell_id = ?`, + [ + reservationId, + reservationId, + identity.userId, + identity.relayHostId, + assignmentEpoch, + cellId, + now, + timeoutAt, + now, + cellId + ] + ) + } + + private async refreshPendingControlReservation( + database: RelayDatabase, + identity: AssignmentIdentity, + cellId: string, + assignmentEpoch: number, + timeoutAt: number, + now: number + ): Promise { + await database.query( + `UPDATE relay_control_connection_reservations + SET state = 'reserved', timeout_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND cell_id = ? AND state IN ('reserved', 'late-arrival-debt') + AND claim_activity_id IS NULL`, + [ + timeoutAt, + now, + identity.userId, + identity.relayHostId, + assignmentEpoch, + cellId + ] + ) + } + + private async claimControlConnectionReservation( + database: RelayDatabase, + identity: AssignmentIdentity, + cellId: string, + assignmentEpoch: number, + activityId: string, + inclusionWatermark: number | undefined, + now: number + ): Promise { + await database.query( + `UPDATE relay_control_connection_reservations + SET claim_activity_id = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND cell_id = ? AND state = 'claimed' + AND claim_activity_id IS NOT NULL AND claim_activity_id <> ?`, + [ + activityId, + now, + identity.userId, + identity.relayHostId, + assignmentEpoch, + cellId, + activityId + ] + ) + // A released row's claim_activity_id is history, not a live claim: a control that + // reconnects under the same generation would otherwise leave its fresh reservation + // unclaimable, decaying through debt while the connection is already enforced. + const alreadyClaimed = await database.queryLocked( + `SELECT reservation_id FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND cell_id = ? AND claim_activity_id = ? AND state <> 'released' + ORDER BY created_at ASC, reservation_id ASC`, + [ + identity.userId, + identity.relayHostId, + assignmentEpoch, + cellId, + activityId + ] + ) + if (alreadyClaimed.length > 0) return + const reservation = ( + await database.queryLocked( + `SELECT * FROM relay_control_connection_reservations + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ? + AND cell_id = ? AND state IN ('reserved', 'late-arrival-debt') + AND claim_activity_id IS NULL + ORDER BY created_at ASC, reservation_id ASC`, + [identity.userId, identity.relayHostId, assignmentEpoch, cellId] + ) + )[0] + if (!reservation) return + await database.query( + `UPDATE relay_control_connection_reservations + SET state = 'claimed', inclusion_watermark = ?, claim_activity_id = ?, + claimed_at = ?, updated_at = ? + WHERE reservation_id = ?`, + [ + inclusionWatermark, + activityId, + now, + now, + text(reservation, 'reservation_id') + ] + ) + } + + private async releaseSupersededControlConnectionReservations( + database: RelayDatabase, + identity: AssignmentIdentity, + cellId: string, + assignmentEpoch: number, + now: number + ): Promise { + await database.query( + `UPDATE relay_control_connection_reservations + SET state = 'released', released_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND cell_id = ? + AND assignment_epoch = ? AND state <> 'released'`, + [ + now, + now, + identity.userId, + identity.relayHostId, + cellId, + assignmentEpoch + ] + ) + } + + private async assignmentRow( + database: RelayDatabase, + identity: AssignmentIdentity, + failIfUnavailable = false + ): Promise { + return ( + await database.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId], + { failIfUnavailable } + ) + )[0] + } + + private async lockAssignmentActivities( + database: RelayDatabase, + identity: AssignmentIdentity, + failIfUnavailable = false + ): Promise { + // Lease rows always precede the globally ordered capacity rows. + return await database.queryLocked( + `SELECT * FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY activity_id ASC`, + [identity.userId, identity.relayHostId], + { failIfUnavailable } + ) + } + + private async removeActivityLease( + database: RelayDatabase, + identity: AssignmentIdentity, + lease: SqlRow, + now: number + ): Promise { + const kind = activityKind(lease) + await this.adjustCellReservation( + database, + text(lease, 'cell_id'), + -integer(lease, 'request_units') + ) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, text(lease, 'activity_id')] + ) + await this.adjustActivityCount(database, identity, kind, -1, now, now) + } + + private async removeSupersededSameCellControls( + database: RelayDatabase, + identity: AssignmentIdentity, + leases: SqlRow[], + cellId: string, + retainedActivityId: string, + now: number + ): Promise { + const superseded = leases.filter( + (lease) => + activityKind(lease) === 'control' && + text(lease, 'cell_id') === cellId && + text(lease, 'activity_id') !== retainedActivityId + ) + if (superseded.length === 0) return + if ( + superseded.some( + (lease) => integer(lease, 'request_units') !== ACTIVITY_REQUEST_UNITS.control + ) + ) { + throw new Error('activity_lease_shape_mismatch') + } + const cells = await this.lockCellInventory(database) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control' + AND cell_id = ? AND activity_id <> ?`, + [identity.userId, identity.relayHostId, cellId, retainedActivityId] + ) + const remainingControls = + leases.filter((lease) => activityKind(lease) === 'control').length - superseded.length + await database.query( + `UPDATE relay_assignments SET reserved_controls = ?, last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [remainingControls, now, identity.userId, identity.relayHostId] + ) + const cellUnitsRow = ( + await database.query( + `SELECT COALESCE(SUM(request_units), 0) AS request_units + FROM relay_assignment_activity_leases WHERE cell_id = ?`, + [cellId] + ) + )[0]! + const cellRow = cells.find((cell) => text(cell, 'cell_id') === cellId) + const cellUnits = integer(cellUnitsRow, 'request_units') + if (!cellRow) throw new Error('assigned_cell_missing') + if (cellUnits > integer(cellRow, 'capacity_requests')) { + throw new Error('relay_capacity_exhausted') + } + await database.query( + `UPDATE relay_cells SET reserved_requests = ?, updated_at = ? WHERE cell_id = ?`, + [cellUnits, now, cellId] + ) + } + + private async adjustActivityCount( + database: RelayDatabase, + identity: AssignmentIdentity, + kind: AssignmentActivityKind, + delta: 1 | -1, + leaseExpiresAt: number, + now: number + ): Promise { + const column = ACTIVITY_COLUMN[kind] + await database.query( + `UPDATE relay_assignments SET ${column} = + CASE WHEN ${column} + ? < 0 THEN 0 ELSE ${column} + ? END, + lease_expires_at = + CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [delta, delta, leaseExpiresAt, leaseExpiresAt, now, identity.userId, identity.relayHostId] + ) + } + + private async touchAssignment( + database: RelayDatabase, + identity: AssignmentIdentity, + leaseExpiresAt: number, + now: number + ): Promise { + await database.query( + `UPDATE relay_assignments SET lease_expires_at = + CASE WHEN lease_expires_at > ? THEN lease_expires_at ELSE ? END, + last_activity_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [leaseExpiresAt, leaseExpiresAt, now, identity.userId, identity.relayHostId] + ) + } + + private async adjustCellReservation( + database: RelayDatabase, + cellId: string, + delta: number + ): Promise { + const row = (await database.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellId]))[0] + if (!row) throw new Error('assigned_cell_missing') + const next = integer(row, 'reserved_requests') + delta + if (next > integer(row, 'capacity_requests')) throw new Error('relay_capacity_exhausted') + await database.query( + `UPDATE relay_cells SET reserved_requests = + CASE WHEN reserved_requests + ? < 0 THEN 0 ELSE reserved_requests + ? END, + updated_at = ? WHERE cell_id = ?`, + [delta, delta, this.now(), cellId] + ) + } + + private async adjustCellReservationAtomically( + database: RelayDatabase, + cellId: string, + delta: number + ): Promise { + const rows = await database.query( + `UPDATE relay_cells SET reserved_requests = + CASE WHEN reserved_requests + ? < 0 THEN 0 ELSE reserved_requests + ? END, + updated_at = ? + WHERE cell_id = ? + AND (? <= 0 OR reserved_requests + ? <= capacity_requests) + RETURNING cell_id`, + [delta, delta, this.now(), cellId, delta, delta] + ) + if (rows.length > 0) return + const cell = ( + await database.query(`SELECT cell_id FROM relay_cells WHERE cell_id = ?`, [cellId]) + )[0] + if (!cell) throw new Error('assigned_cell_missing') + throw new Error('relay_capacity_exhausted') + } + + private result( + identity: AssignmentIdentity, + row: SqlRow, + cellRow: CellRow, + leaseExpiresAt: number + ): RelayAssignment { + return { + ...identity, + ...cellRow, + assignmentEpoch: integer(row, 'assignment_epoch'), + leaseExpiresAt + } + } +} + +type CellRow = { cellId: string; cellUrl: string; region: RelayRegion } + +function cell(row: SqlRow, region: RelayRegion): CellRow { + return { cellId: text(row, 'cell_id'), cellUrl: text(row, 'cell_url'), region } +} + +function relayRegion(row: SqlRow, field: string): RelayRegion { + const value = text(row, field) + if (!RELAY_REGIONS.includes(value as RelayRegion)) throw new Error(`invalid region field ${field}`) + return value as RelayRegion +} + +function optionalRelayRegion(row: SqlRow, field: string): RelayRegion | undefined { + const value = optionalText(row, field) + if (value === undefined) return undefined + if (!RELAY_REGIONS.includes(value as RelayRegion)) throw new Error(`invalid region field ${field}`) + return value as RelayRegion +} + +function activityLeaseById(rows: SqlRow[], activityId: string): SqlRow | undefined { + return rows.find((row) => text(row, 'activity_id') === activityId) +} + +// Why: mid-rehome a host legitimately holds the source control plus the +// target's pending control, so only the lease rows — never the counter a grant +// would overwrite — can say whether this cell is already reserved for it. +function holdsControlLease( + rows: SqlRow[], + cellId: string, + assignmentEpoch: number +): boolean { + const pendingId = pendingControlActivityId(assignmentEpoch) + return rows.some((row) => { + if (activityKind(row) !== 'control' || text(row, 'cell_id') !== cellId) return false + const activityId = text(row, 'activity_id') + return activityId === pendingId || !activityId.startsWith('control-pending:') + }) +} + +function activityCounts(rows: SqlRow[]): Record { + const counts = emptyActivityCounts() + for (const row of rows) counts[activityKind(row)]++ + return counts +} + +function activityUnitsForCell(rows: SqlRow[], cellId: string): number { + return rows.reduce( + (total, row) => + text(row, 'cell_id') === cellId ? total + integer(row, 'request_units') : total, + 0 + ) +} + +function activity(row: SqlRow) { + return { + relayHostId: text(row, 'relay_host_id'), + cellId: text(row, 'cell_id'), + assignmentEpoch: integer(row, 'assignment_epoch'), + leaseExpiresAt: integer(row, 'lease_expires_at'), + lastActivityAt: integer(row, 'last_activity_at'), + reservedControls: integer(row, 'reserved_controls'), + reservedSplices: integer(row, 'reserved_splices'), + reservedInvites: integer(row, 'reserved_invites'), + pendingInstalls: integer(row, 'pending_installs'), + pendingConfirmations: integer(row, 'pending_confirmations'), + migrationLeases: integer(row, 'migration_leases') + } +} + +function requestUnits(row: SqlRow): number { + return ( + integer(row, 'reserved_controls') + + 2 * integer(row, 'reserved_splices') + + integer(row, 'reserved_invites') + + integer(row, 'pending_installs') + + integer(row, 'pending_confirmations') + + integer(row, 'migration_leases') + ) +} + +function integer(row: SqlRow, field: string): number { + const value = Number(row[field]) + if (!Number.isSafeInteger(value)) throw new Error(`invalid_${field}`) + return value +} + +function text(row: SqlRow, field: string): string { + const value = row[field] + if (typeof value !== 'string') throw new Error(`invalid_${field}`) + return value +} + +function cellFenceAttempt(row: SqlRow): CellFenceAttempt { + const environment = text(row, 'environment') + if (!['staging', 'production'].includes(environment)) { + throw new Error('invalid_environment') + } + return { + attemptId: text(row, 'attempt_id'), + environment: environment as CellFenceAttempt['environment'], + cellId: text(row, 'cell_id'), + cellIncarnation: text(row, 'cell_incarnation'), + migName: text(row, 'mig_name'), + instanceGroup: text(row, 'instance_group'), + generationIdentity: text(row, 'generation_identity'), + fenceCommit: text(row, 'fence_commit'), + planSha256: text(row, 'plan_sha256'), + planObjectName: text(row, 'plan_object_name'), + planObjectGeneration: optionalText(row, 'plan_object_generation'), + varFileSha256: text(row, 'var_file_sha256'), + terraformStateLineage: text(row, 'terraform_state_lineage'), + terraformStateSerial: integer(row, 'terraform_state_serial'), + terraformStateObjectGeneration: text( + row, + 'terraform_state_object_generation' + ), + terraformStateObjectSha256: text(row, 'terraform_state_object_sha256'), + requestReason: text(row, 'request_reason'), + gceOperation: optionalText(row, 'gce_operation'), + createdAt: integer(row, 'created_at'), + expiresAt: integer(row, 'expires_at'), + applyStartedAt: optionalInteger(row, 'apply_started_at'), + completedAt: optionalInteger(row, 'completed_at'), + abortedAt: optionalInteger(row, 'aborted_at') + } +} + +function cellFenceApplyInvocation(row: SqlRow): CellFenceApplyInvocation { + return { + invocationId: text(row, 'invocation_id'), + requestReason: text(row, 'request_reason'), + startedAt: integer(row, 'started_at'), + gceOperation: optionalText(row, 'gce_operation') + } +} + +function assertCellFenceAttemptBase( + row: SqlRow, + input: CellFenceAttemptEvidence +): void { + const fields: [keyof CellFenceAttemptEvidence, string][] = [ + ['attemptId', 'attempt_id'], + ['environment', 'environment'], + ['cellId', 'cell_id'], + ['cellIncarnation', 'cell_incarnation'], + ['migName', 'mig_name'], + ['instanceGroup', 'instance_group'], + ['generationIdentity', 'generation_identity'], + ['fenceCommit', 'fence_commit'], + ['planSha256', 'plan_sha256'], + ['planObjectName', 'plan_object_name'], + ['varFileSha256', 'var_file_sha256'], + ['terraformStateLineage', 'terraform_state_lineage'], + ['terraformStateObjectGeneration', 'terraform_state_object_generation'], + ['terraformStateObjectSha256', 'terraform_state_object_sha256'], + ['requestReason', 'request_reason'] + ] + if ( + fields.some(([inputField, rowField]) => input[inputField] !== text(row, rowField)) || + input.terraformStateSerial !== integer(row, 'terraform_state_serial') + ) { + throw new Error('cell_fence_attempt_evidence_mismatch') + } +} + +function assertCellFenceAttemptEvidence( + row: SqlRow, + input: CellFenceAttemptEvidence +): void { + assertCellFenceAttemptBase(row, input) + if ( + !input.planObjectGeneration || + input.planObjectGeneration !== optionalText(row, 'plan_object_generation') + ) { + throw new Error('cell_fence_attempt_evidence_mismatch') + } +} + +function assertActiveCellFenceAttempt( + row: SqlRow, + input: CellFenceAttemptEvidence, + now: number +): void { + assertCellFenceAttemptEvidence(row, input) + if ( + integer(row, 'expires_at') <= now && + optionalInteger(row, 'apply_started_at') === undefined + ) { + throw new Error('cell_fence_attempt_expired') + } + if (row.completed_at !== null) throw new Error('cell_fence_attempt_completed') + if (row.aborted_at !== null) throw new Error('cell_fence_attempt_aborted') +} + +async function lockedCellFenceAttempt( + transaction: RelayDatabase, + attemptId: string +): Promise { + const row = ( + await transaction.queryLocked( + `${CELL_FENCE_ATTEMPT_SELECT} WHERE attempts.attempt_id = ?`, + [attemptId] + ) + )[0] + if (!row) throw new Error('cell_fence_attempt_not_found') + return row +} + +function pendingControlActivityId(assignmentEpoch: number): string { + return `control-pending:${assignmentEpoch}` +} + +function validateActivityId(activityId: string): void { + if (!activityId || activityId.length > 256) throw new Error('invalid_activity_id') +} + +function activityKind(row: SqlRow): AssignmentActivityKind { + const value = text(row, 'activity_kind') + if (!(value in ACTIVITY_REQUEST_UNITS)) throw new Error('invalid_activity_kind') + return value as AssignmentActivityKind +} + +function migrationActivityId(assignmentEpoch: number): string { + return `migration:${assignmentEpoch}` +} + +function isIncompleteMigration(error: unknown): boolean { + return ( + error instanceof Error && + [ + 'migration_target_not_registered', + 'migration_source_still_active', + 'migration_target_not_active', + 'migration_assignment_mismatch', + 'migration_source_admission_changed', + 'migration_target_admission_changed', + 'migration_source_runtime_not_quiescent', + 'migration_source_runtime_not_dead', + 'migration_target_runtime_not_ready', + 'migration_cell_inventory_busy', + 'migration_activity_accounting_mismatch', + 'migration_activity_topology_mismatch', + 'migration_activity_lease_shape_mismatch', + 'migration_cell_reservation_accounting_mismatch', + 'cell_fence_attestation_missing' + ].includes(error.message) + ) +} + +function isMissingMigration(error: unknown): boolean { + return error instanceof Error && error.message === 'migration_not_found' +} + +function isDatabaseLockUnavailable(error: unknown): boolean { + return error instanceof Error && error.message === 'database_lock_unavailable' +} + +function isDatabaseLockTimeout(error: unknown): boolean { + return String((error as { code?: unknown }).code) === '55P03' +} + +async function waitForAssignmentLockRetry(deadline: number): Promise { + const remainingMs = Math.max(0, deadline - Date.now()) + const maxDelayMs = Math.min(ASSIGNMENT_LOCK_RETRY_MAX_DELAY_MS, remainingMs) + const delayMs = Math.floor(Math.random() * (maxDelayMs + 1)) + await new Promise((resolve) => setTimeout(resolve, delayMs)) +} + +function migration(identity: AssignmentIdentity, row: SqlRow): RelayAssignmentMigration { + const targetRegisteredAt = optionalInteger(row, 'target_registered_at') + return { + ...identity, + sourceCellId: text(row, 'source_cell_id'), + targetCellId: text(row, 'target_cell_id'), + previousEpoch: integer(row, 'previous_epoch'), + assignmentEpoch: integer(row, 'assignment_epoch'), + expiresAt: integer(row, 'expires_at'), + ...(targetRegisteredAt === undefined ? {} : { targetRegisteredAt }) + } +} + +// Attempt ids are server-minted UUIDs and this codebase's invariant messages +// are snake_case slugs; anything else could carry secrets and logs redacted. +function warnRegionalRehomeCandidateFailure( + operation: 'complete' | 'abort', + attemptId: string, + error: unknown +): void { + const message = error instanceof Error ? error.message : '' + console.warn( + JSON.stringify({ + event: 'orca_relay_regional_rehome_candidate_failed', + operation, + attemptId, + reason: /^[a-z0-9_]{1,64}$/.test(message) ? message : 'redacted' + }) + ) +} + +// The attempt id is the only field: counts would say which host holds what. +function noteRegionalRehomeActivityCountsRepaired(attemptId: string): void { + console.warn( + JSON.stringify({ + event: 'orca_relay_regional_rehome_activity_counts_repaired', + attemptId + }) + ) +} + +function regionalRehomeAttempt(row: SqlRow): RegionalRehomeAttempt { + return { + attemptId: text(row, 'attempt_id'), + userId: text(row, 'user_id'), + relayHostId: text(row, 'relay_host_id'), + preferredRegion: 'asia-east2', + sourceCellId: text(row, 'source_cell_id'), + sourceCellUrl: text(row, 'source_cell_url'), + sourceCellIncarnation: text(row, 'source_cell_incarnation'), + targetCellId: text(row, 'target_cell_id'), + targetCellIncarnation: text(row, 'target_cell_incarnation'), + previousEpoch: integer(row, 'previous_epoch'), + assignmentEpoch: integer(row, 'assignment_epoch'), + drainGraceMs: integer(row, 'drain_grace_ms'), + sendAttempts: integer(row, 'send_attempts') + } +} + +function regionalRehomeControl(row: SqlRow): RegionalRehomeControl { + return { + generation: integer(row, 'generation'), + enabled: integer(row, 'enabled') === 1, + observationStartedAt: integer(row, 'observation_started_at'), + notBefore: integer(row, 'not_before'), + ratePerMinute: integer(row, 'rate_per_minute'), + preferenceMaxAgeMs: integer(row, 'preference_max_age_ms'), + drainGraceMs: integer(row, 'drain_grace_ms') + } +} + +function cleanRegionalRehomeSafety(now: number): RegionalRehomeSafetySnapshot { + return { + observedAt: now, + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } +} + +function regionalRehomeFleetSafetyFromInventory(input: { + cells: SqlRow[] + admission: ReadonlyMap + regions: ReadonlyMap + runtimes: SqlRow[] + capabilities: SqlRow[] + safetyRows: SqlRow[] + now: number + heartbeatTtlMs: number +}): RegionalRehomeFleetSafety { + const capabilities = new Map( + input.capabilities.map((row) => [text(row, 'cell_id'), row]) + ) + const runtimes = new Map(input.runtimes.map((row) => [text(row, 'cell_id'), row])) + const safetyRows = new Map( + input.safetyRows.map((row) => [text(row, 'cell_id'), row]) + ) + const required = input.cells.filter((row) => { + const cellId = text(row, 'cell_id') + const capability = capabilities.get(cellId) + return ( + integer(row, 'enabled') === 1 && + input.admission.get(cellId) === 'general' && + (input.regions.get(cellId) === 'asia-east2' || + (input.regions.get(cellId) === RELAY_DEFAULT_REGION && + capability !== undefined && + integer(capability, 'regional_rehome_protocol') >= 1)) + ) + }) + const valid = required.flatMap((row) => { + const cellId = text(row, 'cell_id') + const runtime = runtimes.get(cellId) + const safety = safetyRows.get(cellId) + if ( + !runtime || + integer(runtime, 'ready') !== 1 || + integer(runtime, 'last_heartbeat_at') <= input.now - input.heartbeatTtlMs || + !safety || + text(safety, 'cell_incarnation') !== text(runtime, 'cell_incarnation') || + integer(safety, 'observed_at') <= input.now - 60_000 + ) { + return [] + } + return [safety] + }) + const missingCells = required.length === 0 ? 1 : required.length - valid.length + return { + requiredCells: required.length, + missingCells, + observedAt: + missingCells > 0 + ? 0 + : Math.min(...valid.map((row) => integer(row, 'observed_at'))), + sqlFailures: valid.reduce((total, row) => total + integer(row, 'sql_failures'), 0), + reconnects: valid.reduce((total, row) => total + integer(row, 'reconnects'), 0), + maxReconnects: Math.max(0, ...valid.map((row) => integer(row, 'reconnects'))), + controlActivityRecoveryFailures: valid.reduce( + (total, row) => total + integer(row, 'control_activity_recovery_failures'), + 0 + ), + databasePoolWaiting: Math.max( + 0, + ...valid.map((row) => integer(row, 'database_pool_waiting')) + ), + databasePoolWaitersMax: Math.max( + 0, + ...valid.map((row) => integer(row, 'database_pool_waiters_max')) + ), + databasePoolWaitMsMax: Math.max( + 0, + ...valid.map((row) => integer(row, 'database_pool_wait_ms_max')) + ) + } +} + +function regionalRehomeFleetSafetyFailure( + processSafety: RegionalRehomeSafetySnapshot, + fleetSafety: RegionalRehomeFleetSafety, + now: number +): string | null { + if (fleetSafety.maxReconnects > REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT) { + return 'elevated_reconnects' + } + return regionalRehomeSafetyFailure( + combineRegionalRehomeSafety(processSafety, fleetSafety), + now, + fleetSafety.requiredCells + ) +} + +type RegionalRehomeCandidateSkip = { + reason: + | 'candidate_stale' + | 'source_ineligible' + | 'source_unclean' + | 'source_control_inactive' + | 'target_unclean' + | 'no_eligible_target' + | 'no_target_headroom' + cellId?: string + sqlFailures?: number + reconnects?: number +} + +function cellUncleanSkip( + reason: 'source_unclean' | 'target_unclean', + cellId: string, + safety: SqlRow | undefined +): RegionalRehomeCandidateSkip { + return { + reason, + cellId, + sqlFailures: safety === undefined ? undefined : integer(safety, 'sql_failures'), + reconnects: safety === undefined ? undefined : integer(safety, 'reconnects') + } +} + +// Candidate skips are otherwise invisible: they neither latch the control off +// nor produce attempts, so an operator cannot tell "skipping" from "idle". +// Cell ids and counters only — never free-form error text. +function aggregateRegionalRehomeCandidateSkips( + skips: readonly RegionalRehomeCandidateSkip[] +): Record { + // `candidates` counts skipped candidate iterations, not distinct cells: one + // unclean cell blocking six candidates reports candidates=6 on one cellId. + const aggregated = new Map() + for (const skip of skips) { + const key = `${skip.reason}:${skip.cellId ?? ''}` + const entry = aggregated.get(key) + if (entry) entry.candidates += 1 + else aggregated.set(key, { ...skip, candidates: 1 }) + } + return { + event: 'orca_relay_regional_rehome_candidates_skipped', + skips: [...aggregated.values()] + } +} + +function regionalRehomeCellSafetyIsClean( + safety: SqlRow | undefined, + runtime: SqlRow, + now: number +): boolean { + return ( + safety !== undefined && + text(safety, 'cell_incarnation') === text(runtime, 'cell_incarnation') && + integer(safety, 'observed_at') > now - 60_000 && + integer(safety, 'sql_failures') <= REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT && + integer(safety, 'reconnects') <= REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT && + integer(safety, 'control_activity_recovery_failures') === 0 && + !regionalRehomePoolPressure({ + databasePoolWaitersMax: integer(safety, 'database_pool_waiters_max'), + databasePoolWaitMsMax: integer(safety, 'database_pool_wait_ms_max') + }) + ) +} + +function assertMigrationPair(row: SqlRow, sourceCellId: string, targetCellId: string): void { + if ( + text(row, 'source_cell_id') !== sourceCellId || + text(row, 'target_cell_id') !== targetCellId + ) { + throw new Error('migration_pair_mismatch') + } +} + +function matchesExactCellSet( + actual: ReadonlySet, + expected: readonly string[] +): boolean { + const expectedSet = new Set(expected) + return ( + expectedSet.size === expected.length && + actual.size === expectedSet.size && + [...actual].every((cellId) => expectedSet.has(cellId)) + ) +} + +function assertCurrentMigrationAssignment( + assignment: SqlRow | undefined, + migrationRow: SqlRow +): asserts assignment is SqlRow { + if ( + !assignment || + text(assignment, 'cell_id') !== text(migrationRow, 'target_cell_id') || + integer(assignment, 'assignment_epoch') !== integer(migrationRow, 'assignment_epoch') + ) { + throw new Error('migration_assignment_mismatch') + } +} + +function assertMigrationRecoveryMetadata( + migrationRow: SqlRow, + incarnation: SqlRow | undefined, + pin: SqlRow | undefined +): void { + if (pin && !incarnation) { + throw new Error('drain_migration_source_incarnation_mismatch') + } + if ( + pin && + incarnation && + (text(pin, 'source_cell_id') !== text(migrationRow, 'source_cell_id') || + text(pin, 'target_cell_id') !== text(migrationRow, 'target_cell_id') || + text(pin, 'source_cell_incarnation') !== + text(incarnation, 'source_cell_incarnation') || + text(pin, 'target_cell_incarnation') !== + text(incarnation, 'target_cell_incarnation') || + integer(pin, 'source_request_units') !== + integer(migrationRow, 'source_request_units') || + integer(pin, 'target_reserved_units') !== + integer(migrationRow, 'target_reserved_units')) + ) { + throw new Error('migration_activity_topology_mismatch') + } +} + +function assertAssignmentActivityAccounting( + assignment: SqlRow, + leases: SqlRow[], + migrationRow: SqlRow +): void { + if ( + integer(migrationRow, 'target_reserved_units') !== + integer(migrationRow, 'source_request_units') + 1 + ) { + throw new Error('migration_activity_lease_shape_mismatch') + } + const counts = emptyActivityCounts() + for (const lease of leases) { + const kind = activityKind(lease) + const requestUnits = integer(lease, 'request_units') + if (kind === 'migration') { + if ( + text(lease, 'activity_id') !== + migrationActivityId(integer(migrationRow, 'assignment_epoch')) || + text(lease, 'cell_id') !== text(migrationRow, 'target_cell_id') || + requestUnits !== integer(migrationRow, 'source_request_units') + ) { + throw new Error('migration_activity_lease_shape_mismatch') + } + } else if (requestUnits !== ACTIVITY_REQUEST_UNITS[kind]) { + throw new Error('migration_activity_lease_shape_mismatch') + } + counts[kind]++ + } + assertAssignmentActivityCounts(assignment, leases, 1) +} + +function assertAssignmentActivityCounts( + assignment: SqlRow, + leases: SqlRow[], + expectedMigrations: number +): void { + const counts = activityCounts(leases) + if ( + (Object.keys(ACTIVITY_COLUMN) as AssignmentActivityKind[]).some( + (kind) => integer(assignment, ACTIVITY_COLUMN[kind]) !== counts[kind] + ) || + counts.migration !== expectedMigrations + ) { + throw new Error('migration_activity_accounting_mismatch') + } +} + +async function assertCellReservationAccounting( + transaction: RelayDatabase, + cells: SqlRow[], + cellIds: string[] +): Promise { + const uniqueCellIds = [...new Set(cellIds)] + const rows = await transaction.query( + `SELECT cell_id, COALESCE(SUM(request_units), 0) AS request_units + FROM relay_assignment_activity_leases + WHERE cell_id IN (${uniqueCellIds.map(() => '?').join(', ')}) + GROUP BY cell_id`, + uniqueCellIds + ) + const unitsByCell = new Map( + rows.map((row) => [text(row, 'cell_id'), integer(row, 'request_units')]) + ) + for (const cellId of uniqueCellIds) { + const cell = cells.find((row) => text(row, 'cell_id') === cellId) + if (!cell || integer(cell, 'reserved_requests') !== (unitsByCell.get(cellId) ?? 0)) { + throw new Error('migration_cell_reservation_accounting_mismatch') + } + } +} + +function deadSourceCompletionResult( + row: SqlRow, + changed: boolean +): DeadSourceCompletionResult { + return { + changed, + assignmentEpoch: integer(row, 'assignment_epoch'), + sourceCellId: text(row, 'source_cell_id'), + targetCellId: text(row, 'target_cell_id') + } +} + +function cellDrainAttempt(row: SqlRow): CellDrainAttempt { + const state = text(row, 'state') + if ( + ![ + 'prepared', + 'send-may-have-started', + 'application-receipt', + 'proven-not-delivered' + ].includes(state) + ) { + throw new Error('invalid_drain_attempt_state') + } + return { + attemptId: text(row, 'attempt_id'), + cellId: text(row, 'cell_id'), + cellIncarnation: text(row, 'cell_incarnation'), + traceValue: text(row, 'trace_value'), + plannedGraceMs: integer(row, 'planned_grace_ms'), + state: state as CellDrainAttemptState, + preparedAt: integer(row, 'prepared_at'), + sendMayHaveStartedAt: optionalInteger(row, 'send_may_have_started_at'), + sendPermitExpiresAt: optionalInteger(row, 'send_permit_expires_at'), + applicationReceiptAt: optionalInteger(row, 'application_receipt_at'), + backendSuccessStatus: optionalInteger(row, 'backend_success_status'), + backendInstance: optionalText(row, 'backend_instance'), + receiptCellIncarnation: optionalText(row, 'receipt_cell_incarnation'), + retryAfter: optionalInteger(row, 'retry_after'), + recoverForwardAttemptedAt: optionalInteger(row, 'recover_forward_attempted_at'), + provenNotDeliveredAt: optionalInteger(row, 'proven_not_delivered_at') + } +} + +function optionalInteger(row: SqlRow, field: string): number | undefined { + if (row[field] === null || row[field] === undefined) return undefined + return integer(row, field) +} + +function optionalText(row: SqlRow, field: string): string | undefined { + if (row[field] === null || row[field] === undefined) return undefined + return text(row, field) +} + +function migrationHasExactActiveTarget(row: SqlRow, targetCellId: string): boolean { + return ( + integer(row, 'target_control_active') === 1 && + optionalText(row, 'current_cell_id') === targetCellId && + optionalInteger(row, 'current_assignment_epoch') === integer(row, 'assignment_epoch') + ) +} + +function assignmentKey(userId: string, relayHostId: string): string { + return `${userId}\u0000${relayHostId}` +} + +function emptyActivityCounts(): Record { + return { + control: 0, + splice: 0, + invite: 0, + install: 0, + confirmation: 0, + migration: 0 + } +} diff --git a/cloud/apps/relay/src/cell-admission-migration-registration.ts b/cloud/apps/relay/src/cell-admission-migration-registration.ts new file mode 100644 index 00000000000..a6cd356a42b --- /dev/null +++ b/cloud/apps/relay/src/cell-admission-migration-registration.ts @@ -0,0 +1,346 @@ +import { + isRelayCellConnectionHardCap, + RELAY_DEFAULT_REGION, + RELAY_REGIONS, + relayCellAdmissionBounds, + type RelayCellConnectionHardCap, + type RelayRegion +} from '@orca-cloud/relay-contract' +import { + decodeMembership, + encodeMembership, + lockedSelector, + normalizeMembership, + requireSelectorMatchesAdmission, + type CellAdmissionMembership, + type CellAdmissionSelector +} from './cell-admission-selector.js' +import type { RelayDatabase, SqlRow } from './database.js' + +export type MigrationCellRegistration = { + id: string + url: string + capacityRequests: number + connectionHardCap: RelayCellConnectionHardCap + connectionUnobservedBound: number + region?: RelayRegion +} + +type NormalizedMigrationCellRegistration = MigrationCellRegistration & { region: RelayRegion } + +type AddMigrationCellsInput = { + attemptId: string + expectedGeneration: number + cells: MigrationCellRegistration[] +} + +const SELECTOR_ID = 'general' + +export class RelayMigrationCellRegistrar { + constructor( + private readonly database: RelayDatabase, + private readonly now: () => number + ) {} + + async add(input: AddMigrationCellsInput): Promise<{ + changed: boolean + selector: CellAdmissionSelector + }> { + validateAttempt(input) + const cells = normalizeMigrationCells(input.cells) + const encodedCells = encodeCells(cells) + return await this.database.transaction(async (transaction) => { + const inventory = await transaction.queryLocked( + `SELECT * FROM relay_cells ORDER BY cell_id ASC` + ) + const selector = await lockedSelector(transaction) + await requireSelectorMatchesAdmission(transaction, selector) + const intent = ( + await transaction.queryLocked( + `SELECT * FROM relay_admission_selector_intents WHERE attempt_id = ?`, + [input.attemptId] + ) + )[0] + const addition = ( + await transaction.queryLocked( + `SELECT * FROM relay_admission_selector_cell_additions WHERE attempt_id = ?`, + [input.attemptId] + ) + )[0] + if (Boolean(intent) !== Boolean(addition)) { + throw new Error('admission_selector_attempt_mismatch') + } + if (intent && addition) { + const membership = decodeMembership(text(intent, 'membership_json')) + if ( + integer(intent, 'expected_generation') !== input.expectedGeneration || + integer(intent, 'intended_generation') !== input.expectedGeneration + 1 || + text(addition, 'cells_json') !== encodedCells || + encodeMembership(membership) !== + encodeMembership( + membershipAfterAddition( + decodeMembership(text(intent, 'previous_membership_json')), + cells + ) + ) + ) { + throw new Error('admission_selector_attempt_mismatch') + } + if ( + selector.generation === input.expectedGeneration + 1 && + selector.attemptId === input.attemptId && + encodeMembership(selector.membership) === encodeMembership(membership) + ) { + await requireExactAddedCells(transaction, inventory, cells) + return { changed: false, selector } + } + throw new Error('admission_selector_generation_mismatch') + } + if (selector.generation < 1) { + throw new Error('admission_selector_boundary_inactive') + } + if (selector.generation !== input.expectedGeneration) { + throw new Error('admission_selector_generation_mismatch') + } + const inventoryIds = new Set(inventory.map((row) => text(row, 'cell_id'))) + const inventoryUrls = new Set(inventory.map((row) => text(row, 'cell_url'))) + if (cells.some((cell) => inventoryIds.has(cell.id) || inventoryUrls.has(cell.url))) { + throw new Error('admission_selector_cell_already_exists') + } + return await this.commit(transaction, input, cells, selector, encodedCells) + }) + } + + private async commit( + transaction: RelayDatabase, + input: AddMigrationCellsInput, + cells: NormalizedMigrationCellRegistration[], + selector: CellAdmissionSelector, + encodedCells: string + ): Promise<{ changed: true; selector: CellAdmissionSelector }> { + const membership = membershipAfterAddition(selector.membership, cells) + const encodedMembership = encodeMembership(membership) + const now = this.now() + await transaction.query( + `INSERT INTO relay_admission_selector_intents + (attempt_id, expected_generation, intended_generation, previous_membership_json, + membership_json, created_at, committed_at) + VALUES (?, ?, ?, ?, ?, ?, NULL)`, + [ + input.attemptId, + input.expectedGeneration, + input.expectedGeneration + 1, + encodeMembership(selector.membership), + encodedMembership, + now + ] + ) + await transaction.query( + `INSERT INTO relay_admission_selector_cell_additions (attempt_id, cells_json) + VALUES (?, ?)`, + [input.attemptId, encodedCells] + ) + for (const cell of cells) await insertCell(transaction, cell, now) + const intendedGeneration = input.expectedGeneration + 1 + const updated = await transaction.query( + `UPDATE relay_admission_selectors + SET generation = ?, attempt_id = ?, membership_json = ?, updated_at = ? + WHERE selector_id = ? AND generation = ?`, + [ + intendedGeneration, + input.attemptId, + encodedMembership, + now, + SELECTOR_ID, + input.expectedGeneration + ] + ) + if (integer(updated[0]!, 'changes') !== 1) { + throw new Error('admission_selector_generation_mismatch') + } + await transaction.query( + `UPDATE relay_admission_selector_intents SET committed_at = ? + WHERE attempt_id = ?`, + [now, input.attemptId] + ) + return { + changed: true, + selector: { + generation: intendedGeneration, + attemptId: input.attemptId, + membership + } + } + } +} + +function encodeCells(cells: NormalizedMigrationCellRegistration[]): string { + return JSON.stringify( + cells.map((cell) => + cell.region === RELAY_DEFAULT_REGION + ? { + id: cell.id, + url: cell.url, + capacityRequests: cell.capacityRequests, + connectionHardCap: cell.connectionHardCap, + connectionUnobservedBound: cell.connectionUnobservedBound + } + : cell + ) + ) +} + +async function insertCell( + transaction: RelayDatabase, + cell: NormalizedMigrationCellRegistration, + now: number +): Promise { + await transaction.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, 1, ?, 0, 0, ?, ?)`, + [cell.id, cell.url, cell.capacityRequests, now, now] + ) + await transaction.query( + `INSERT INTO relay_cell_regions (cell_id, region) VALUES (?, ?)`, + [cell.id, cell.region] + ) + await transaction.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES (?, 'migration-only', ?)`, + [cell.id, now] + ) + await transaction.query( + `INSERT INTO relay_cell_connection_limits + (cell_id, hard_cap, unobserved_bound, updated_at) + VALUES (?, ?, ?, ?)`, + [cell.id, cell.connectionHardCap, cell.connectionUnobservedBound, now] + ) +} + +function normalizeMigrationCells( + input: MigrationCellRegistration[] +): NormalizedMigrationCellRegistration[] { + if (input.length === 0 || input.length > 128) { + throw new Error('invalid_admission_selector_cells') + } + const cells = input + .map((cell) => ({ ...cell, region: cell.region ?? RELAY_DEFAULT_REGION })) + .sort((left, right) => left.id.localeCompare(right.id)) + if ( + new Set(cells.map(({ id }) => id)).size !== cells.length || + new Set(cells.map(({ url }) => url)).size !== cells.length + ) { + throw new Error('admission_selector_duplicate_cell') + } + for (const cell of cells) validateMigrationCell(cell) + return cells +} + +function validateMigrationCell(cell: NormalizedMigrationCellRegistration): void { + if ( + cell.id.length === 0 || + cell.id.length > 128 || + cell.url.length === 0 || + cell.url.length > 2_048 || + !Number.isSafeInteger(cell.capacityRequests) || + cell.capacityRequests <= 0 || + cell.capacityRequests > 100_000 || + !isRelayCellConnectionHardCap(cell.connectionHardCap) || + !Number.isSafeInteger(cell.connectionUnobservedBound) || + cell.connectionUnobservedBound < 0 || + cell.connectionUnobservedBound > + relayCellAdmissionBounds(cell.connectionHardCap).maxUnobservedBound || + !RELAY_REGIONS.includes(cell.region) + ) { + throw new Error('invalid_admission_selector_cell') + } +} + +function membershipAfterAddition( + membership: CellAdmissionMembership, + cells: NormalizedMigrationCellRegistration[] +): CellAdmissionMembership { + return normalizeMembership({ + existingOnly: membership.existingOnly, + migrationOnly: [...membership.migrationOnly, ...cells.map(({ id }) => id)], + general: membership.general + }) +} + +async function requireExactAddedCells( + database: RelayDatabase, + inventory: SqlRow[], + cells: NormalizedMigrationCellRegistration[] +): Promise { + const byId = new Map(inventory.map((row) => [text(row, 'cell_id'), row])) + for (const cell of cells) { + const row = byId.get(cell.id) + const admission = await admissionState(database, cell.id) + const region = await cellRegion(database, cell.id) + const limit = await connectionLimit(database, cell.id) + if ( + !row || + text(row, 'cell_url') !== cell.url || + integer(row, 'enabled') !== 1 || + integer(row, 'capacity_requests') !== cell.capacityRequests || + region !== cell.region || + admission !== 'migration-only' || + !limit || + integer(limit, 'hard_cap') !== cell.connectionHardCap || + integer(limit, 'unobserved_bound') !== cell.connectionUnobservedBound + ) { + throw new Error('admission_selector_attempt_mismatch') + } + } +} + +async function cellRegion(database: RelayDatabase, cellId: string): Promise { + return ( + await database.query(`SELECT region FROM relay_cell_regions WHERE cell_id = ?`, [cellId]) + )[0]?.region +} + +async function admissionState(database: RelayDatabase, cellId: string): Promise { + return ( + await database.query( + `SELECT admission_state FROM relay_cell_admission WHERE cell_id = ?`, + [cellId] + ) + )[0]?.admission_state +} + +async function connectionLimit( + database: RelayDatabase, + cellId: string +): Promise { + return ( + await database.query( + `SELECT hard_cap, unobserved_bound FROM relay_cell_connection_limits + WHERE cell_id = ?`, + [cellId] + ) + )[0] +} + +function validateAttempt(input: AddMigrationCellsInput): void { + if (!/^[A-Za-z0-9_-]{8,128}$/.test(input.attemptId)) { + throw new Error('invalid_admission_selector_attempt') + } + if (!Number.isSafeInteger(input.expectedGeneration) || input.expectedGeneration < 0) { + throw new Error('invalid_admission_selector_generation') + } +} + +function integer(row: SqlRow, field: string): number { + const value = Number(row[field]) + if (!Number.isSafeInteger(value)) throw new Error(`invalid integer field ${field}`) + return value +} + +function text(row: SqlRow, field: string): string { + const value = row[field] + if (typeof value !== 'string') throw new Error(`invalid text field ${field}`) + return value +} diff --git a/cloud/apps/relay/src/cell-admission-selector.test.ts b/cloud/apps/relay/src/cell-admission-selector.test.ts new file mode 100644 index 00000000000..41b84f91d79 --- /dev/null +++ b/cloud/apps/relay/src/cell-admission-selector.test.ts @@ -0,0 +1,580 @@ +import { ASSIGNMENT_LIMITS } from '@orca-cloud/relay-contract' +import { createHash } from 'node:crypto' +import { afterEach, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { encodeMembership } from './cell-admission-selector.js' +import { + openInMemoryRelayDatabase, + type RelayDatabase, + type RelayLockOptions, + type SqlRow +} from './database.js' + +const CELLS = [ + { id: 'c1', url: 'https://c1.example.com', capacityRequests: 20 }, + { id: 'c2', url: 'https://c2.example.com', capacityRequests: 20 }, + { id: 'c3', url: 'https://c3.example.com', capacityRequests: 20 } +] + +const INITIAL_MEMBERSHIP = { + existingOnly: ['c1'], + migrationOnly: ['c2'], + general: ['c3'] +} +const BASE_MEMBERSHIP = { + existingOnly: [], + migrationOnly: [], + general: ['c1', 'c2', 'c3'] +} + +class FailAfterIntentDatabase implements RelayDatabase { + private transactionsUntilFailure = Number.POSITIVE_INFINITY + + constructor(private readonly delegate: RelayDatabase) {} + + failSecondTransaction(): void { + this.transactionsUntilFailure = 2 + } + + async query(sql: string, params?: unknown[]): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + return await this.delegate.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + this.transactionsUntilFailure-- + if (this.transactionsUntilFailure === 0) { + this.transactionsUntilFailure = Number.POSITIVE_INFINITY + throw new Error('injected_commit_ambiguity') + } + return await this.delegate.transaction(operation) + } + + async close(): Promise { + await this.delegate.close() + } +} + +class AfterTransactionDatabase implements RelayDatabase { + private afterTransaction: (() => Promise) | undefined + + constructor(private readonly delegate: RelayDatabase) {} + + runAfterNextTransaction(operation: () => Promise): void { + this.afterTransaction = operation + } + + async query(sql: string, params?: unknown[]): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + return await this.delegate.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + const result = await this.delegate.transaction(operation) + const after = this.afterTransaction + this.afterTransaction = undefined + if (after) await after() + return result + } + + async close(): Promise { + await this.delegate.close() + } +} + +function membershipSha256(membership: typeof INITIAL_MEMBERSHIP): string { + return createHash('sha256').update(encodeMembership(membership)).digest('hex') +} + +const BASE_MEMBERSHIP_SHA256 = membershipSha256(BASE_MEMBERSHIP) + +describe('relay cell admission selector', () => { + let database: RelayDatabase | undefined + + afterEach(async () => await database?.close()) + + async function setup( + now: () => number, + requireLiveCells = false + ): Promise { + database = await openInMemoryRelayDatabase() + const store = new RelayAssignmentStore(database, now, { + requireLiveCells, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells(CELLS) + return store + } + + async function heartbeat(store: RelayAssignmentStore, cellIndex: number): Promise { + const cell = CELLS[cellIndex]! + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: `${cellIndex + 1}1111111-1111-4111-8111-111111111111`, + startedAt: 50, + ready: true, + observedRequests: 0 + }) + } + + it('preserves sticky assignments while reserving ordinary placement for general cells', async () => { + const store = await setup(() => 100) + const existing = { userId: 'existing', relayHostId: 'host000000000001' } + expect(await store.assign(existing)).toMatchObject({ cellId: 'c1' }) + + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_001', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + + expect(await store.assign(existing)).toMatchObject({ cellId: 'c1' }) + await expect( + store.assign({ userId: 'new', relayHostId: 'host000000000002' }) + ).resolves.toMatchObject({ cellId: 'c3' }) + await expect(store.startEvacuation(existing, 'c2')).resolves.toMatchObject({ + sourceCellId: 'c1', + targetCellId: 'c2' + }) + }) + + it('excludes existing-only and migration-only cells from dormant placement', async () => { + let now = 100 + const store = await setup(() => now) + const identity = { userId: 'dormant', relayHostId: 'host000000000001' } + const assignment = await store.assign(identity) + const control = await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.releaseActivity(identity, control) + now += ASSIGNMENT_LIMITS.dormantTtlMs + 1 + + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_002', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + + await expect(store.rebalanceDormant(identity, 'c2')).rejects.toThrow( + 'target_cell_unavailable' + ) + await expect(store.rebalanceDormant(identity, 'c3')).resolves.toMatchObject({ + cellId: 'c3' + }) + }) + + it('does not recover an unfenced dead existing-only assignment', async () => { + let now = 100 + const store = await setup(() => now, true) + await heartbeat(store, 0) + await heartbeat(store, 1) + await heartbeat(store, 2) + const identity = { userId: 'dead-source', relayHostId: 'host000000000001' } + expect(await store.assign(identity)).toMatchObject({ cellId: 'c1' }) + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_003', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + + now += 45_001 + await heartbeat(store, 1) + await heartbeat(store, 2) + expect(await store.evacuateDeadCells()).toBe(0) + expect(await store.resolve(identity)).toBeNull() + }) + + it('commits atomically and resolves an ambiguous response idempotently', async () => { + const store = await setup(() => 100) + const input = { + attemptId: 'cutover_004', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + } + + await expect(store.applyCellAdmissionSelector(input)).resolves.toMatchObject({ + changed: true, + selector: { generation: 1, membership: INITIAL_MEMBERSHIP } + }) + await expect(store.applyCellAdmissionSelector(input)).resolves.toMatchObject({ + changed: false, + selector: { generation: 1, membership: INITIAL_MEMBERSHIP } + }) + await expect(store.inspectCellAdmissionSelector(input.attemptId)).resolves.toMatchObject({ + intent: { state: 'committed' } + }) + expect(await store.cellDeploymentStatus('c1')).toMatchObject({ + enabled: false, + admissionState: 'existing-only' + }) + expect(await store.cellDeploymentStatus('c2')).toMatchObject({ + enabled: true, + admissionState: 'migration-only' + }) + }) + + it('rejects a generation-zero membership change between intent and commit', async () => { + const delegate = await openInMemoryRelayDatabase() + const hooked = new AfterTransactionDatabase(delegate) + database = hooked + const store = new RelayAssignmentStore(hooked, () => 100) + await store.reconcileCells(CELLS) + const inspected = (await store.inspectCellAdmissionSelector()).selector.membership + const changed = { + existingOnly: ['c2'], + migrationOnly: [], + general: ['c1', 'c3'] + } + hooked.runAfterNextTransaction(async () => { + await delegate.transaction(async (transaction) => { + await transaction.query( + `UPDATE relay_cells SET enabled = 0 WHERE cell_id = ?`, + ['c2'] + ) + await transaction.query( + `UPDATE relay_cell_admission SET admission_state = ? WHERE cell_id = ?`, + ['existing-only', 'c2'] + ) + await transaction.query( + `UPDATE relay_admission_selectors SET membership_json = ? + WHERE selector_id = ? AND generation = 0`, + [encodeMembership(changed), 'general'] + ) + }) + }) + + await expect(store.applyCellAdmissionSelector({ + attemptId: 'cutover_race_001', + expectedGeneration: 0, + expectedMembershipSha256: membershipSha256(inspected), + membership: inspected + })).rejects.toThrow('admission_selector_membership_mismatch') + await expect(store.applyCellAdmissionSelector({ + attemptId: 'cutover_race_001', + expectedGeneration: 0, + membership: inspected + })).rejects.toThrow('admission_selector_membership_fingerprint_required') + await expect(store.inspectCellAdmissionSelector()).resolves.toMatchObject({ + selector: { generation: 0, membership: changed } + }) + }) + + it('preserves admission age for cells unchanged by a selector CAS', async () => { + let now = 100 + const store = await setup(() => now) + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_age_001', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + + now = 200 + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_age_002', + expectedGeneration: 1, + membership: { + existingOnly: ['c1'], + migrationOnly: [], + general: ['c2', 'c3'] + } + }) + + expect( + await database!.query( + `SELECT cell_id, admission_state, updated_at + FROM relay_cell_admission ORDER BY cell_id` + ) + ).toEqual([ + { cell_id: 'c1', admission_state: 'existing-only', updated_at: 100 }, + { cell_id: 'c2', admission_state: 'general', updated_at: 200 }, + { cell_id: 'c3', admission_state: 'general', updated_at: 100 } + ]) + }) + + it('persists failed CAS intent without mutating current membership', async () => { + const store = await setup(() => 100) + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_005', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + + await expect( + store.applyCellAdmissionSelector({ + attemptId: 'stale_attempt', + expectedGeneration: 5, + membership: { + existingOnly: ['c1'], + migrationOnly: [], + general: ['c2', 'c3'] + } + }) + ).rejects.toThrow('admission_selector_generation_mismatch') + await expect(store.inspectCellAdmissionSelector('stale_attempt')).resolves.toMatchObject({ + selector: { generation: 1, membership: INITIAL_MEMBERSHIP }, + intent: { state: 'diverged' } + }) + }) + + it('rejects duplicate or incomplete selector membership before mutation', async () => { + const store = await setup(() => 100) + + await expect( + store.applyCellAdmissionSelector({ + attemptId: 'duplicate_001', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: { + existingOnly: ['c1', 'c1'], + migrationOnly: ['c2'], + general: ['c3'] + } + }) + ).rejects.toThrow('admission_selector_duplicate_cell') + await expect( + store.applyCellAdmissionSelector({ + attemptId: 'incomplete_001', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: { + existingOnly: ['c1'], + migrationOnly: ['c2'], + general: [] + } + }) + ).rejects.toThrow('admission_selector_incomplete_membership') + await expect(store.inspectCellAdmissionSelector()).resolves.toMatchObject({ + selector: { + generation: 0, + membership: { existingOnly: [], migrationOnly: [], general: ['c1', 'c2', 'c3'] } + } + }) + }) + + it('inspects an unchanged durable intent after an ambiguous apply failure', async () => { + const underlying = await openInMemoryRelayDatabase() + const failingDatabase = new FailAfterIntentDatabase(underlying) + database = failingDatabase + const store = new RelayAssignmentStore(failingDatabase, () => 100) + await store.reconcileCells(CELLS) + failingDatabase.failSecondTransaction() + + const input = { + attemptId: 'ambiguous_001', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + } + await expect(store.applyCellAdmissionSelector(input)).rejects.toThrow( + 'injected_commit_ambiguity' + ) + await expect(store.inspectCellAdmissionSelector(input.attemptId)).resolves.toMatchObject({ + selector: { generation: 0 }, + intent: { state: 'unchanged' } + }) + await expect(store.applyCellAdmissionSelector(input)).resolves.toMatchObject({ + changed: true, + selector: { generation: 1, membership: INITIAL_MEMBERSHIP } + }) + }) + + it('allows only one concurrent CAS and never re-enables legacy cells', async () => { + const store = await setup(() => 100) + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_006', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + await expect(store.setCellEnabled('c1', true)).rejects.toThrow( + 'admission_selector_boundary_active' + ) + + const nextMembership = { + existingOnly: ['c1'], + migrationOnly: [], + general: ['c2', 'c3'] + } + const results = await Promise.allSettled([ + store.applyCellAdmissionSelector({ + attemptId: 'promote_001', + expectedGeneration: 1, + membership: nextMembership + }), + store.applyCellAdmissionSelector({ + attemptId: 'promote_002', + expectedGeneration: 1, + membership: nextMembership + }) + ]) + expect(results.filter(({ status }) => status === 'fulfilled')).toHaveLength(1) + expect(results.filter(({ status }) => status === 'rejected')).toHaveLength(1) + + await expect( + store.applyCellAdmissionSelector({ + attemptId: 'reenable_001', + expectedGeneration: 2, + membership: { + existingOnly: [], + migrationOnly: [], + general: ['c1', 'c2', 'c3'] + } + }) + ).rejects.toThrow('admission_selector_legacy_reenable') + }) + + it('preserves the selector boundary across compatible reconciliation', async () => { + const store = await setup(() => 100) + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_007', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + + await expect(store.reconcileCells(CELLS, false)).resolves.toBeUndefined() + await expect(store.reconcileCells([CELLS[0]!, CELLS[2]!])).rejects.toThrow( + 'admission_selector_boundary_active' + ) + await expect(store.inspectCellAdmissionSelector()).resolves.toMatchObject({ + selector: { generation: 1, membership: INITIAL_MEMBERSHIP } + }) + }) + + it('atomically adds exact migration-only cells after the selector boundary', async () => { + const store = await setup(() => 100) + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_008', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + const input = { + attemptId: 'add_cells_001', + expectedGeneration: 1, + cells: [ + { + id: 'c5', + url: 'https://c5.example.com', + capacityRequests: 4_000, + connectionHardCap: 1_000 as const, + connectionUnobservedBound: 60 + }, + { + id: 'c4', + url: 'https://c4.example.com', + capacityRequests: 4_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 60 + } + ] + } + + await expect(store.addMigrationCells(input)).resolves.toMatchObject({ + changed: true, + selector: { + generation: 2, + attemptId: input.attemptId, + membership: { + existingOnly: ['c1'], + migrationOnly: ['c2', 'c4', 'c5'], + general: ['c3'] + } + } + }) + await expect(store.addMigrationCells(input)).resolves.toMatchObject({ + changed: false, + selector: { generation: 2 } + }) + await expect(store.inspectCellAdmissionSelector(input.attemptId)).resolves.toMatchObject({ + intent: { state: 'committed' } + }) + await expect(store.cellDeploymentStatus('c4')).resolves.toMatchObject({ + enabled: true, + admissionState: 'migration-only', + cellUrl: 'https://c4.example.com', + capacityRequests: 4_000, + connectionCapacity: { + hardCap: 600, + unobservedBound: 60 + } + }) + await expect(store.cellDeploymentStatus('c5')).resolves.toMatchObject({ + connectionCapacity: { + hardCap: 1_000, + ordinaryConnectionLimit: 900, + normalAdmissionPause: 840 + } + }) + }) + + it('rejects additive cells before cutover and exact-attempt config reuse', async () => { + const store = await setup(() => 100) + const cell = { + id: 'c4', + url: 'https://c4.example.com', + capacityRequests: 4_000, + connectionHardCap: 600 as const, + connectionUnobservedBound: 60 + } + await expect( + store.addMigrationCells({ + attemptId: 'add_cells_002', + expectedGeneration: 0, + cells: [cell] + }) + ).rejects.toThrow('admission_selector_boundary_inactive') + await store.applyCellAdmissionSelector({ + attemptId: 'cutover_009', + expectedGeneration: 0, + expectedMembershipSha256: BASE_MEMBERSHIP_SHA256, + membership: INITIAL_MEMBERSHIP + }) + await store.addMigrationCells({ + attemptId: 'add_cells_002', + expectedGeneration: 1, + cells: [cell] + }) + await expect( + store.addMigrationCells({ + attemptId: 'add_cells_002', + expectedGeneration: 1, + cells: [{ ...cell, capacityRequests: 3_999 }] + }) + ).rejects.toThrow('admission_selector_attempt_mismatch') + await expect( + store.applyCellAdmissionSelector({ + attemptId: 'add_cells_002', + expectedGeneration: 2, + membership: { + existingOnly: ['c1'], + migrationOnly: ['c2', 'c4'], + general: ['c3'] + } + }) + ).rejects.toThrow('admission_selector_attempt_mismatch') + }) +}) diff --git a/cloud/apps/relay/src/cell-admission-selector.ts b/cloud/apps/relay/src/cell-admission-selector.ts new file mode 100644 index 00000000000..92764e53e40 --- /dev/null +++ b/cloud/apps/relay/src/cell-admission-selector.ts @@ -0,0 +1,505 @@ +import { createHash } from 'node:crypto' +import type { RelayDatabase, SqlRow } from './database.js' + +export const CELL_ADMISSION_STATES = [ + 'existing-only', + 'migration-only', + 'general' +] as const + +export type CellAdmissionState = (typeof CELL_ADMISSION_STATES)[number] + +export type CellAdmissionMembership = { + existingOnly: string[] + migrationOnly: string[] + general: string[] +} + +export type CellAdmissionSelector = { + generation: number + attemptId: string | null + membership: CellAdmissionMembership +} + +export type CellAdmissionSelectorInspection = { + selector: CellAdmissionSelector + intent: null | { + attemptId: string + expectedGeneration: number + intendedGeneration: number + previousMembership: CellAdmissionMembership + membership: CellAdmissionMembership + state: 'unchanged' | 'committed' | 'diverged' + } +} + +type ApplySelectorInput = { + attemptId: string + expectedGeneration: number + expectedMembershipSha256?: string + membership: CellAdmissionMembership +} + +const SELECTOR_ID = 'general' + +export function stateFromEnabled(enabled: boolean): CellAdmissionState { + return enabled ? 'general' : 'existing-only' +} + +export function enabledForState(state: CellAdmissionState): number { + return state === 'existing-only' ? 0 : 1 +} + +export function parseCellAdmissionState(value: string): CellAdmissionState { + if (!CELL_ADMISSION_STATES.includes(value as CellAdmissionState)) { + throw new Error('invalid_cell_admission_state') + } + return value as CellAdmissionState +} + +export async function ensureCellAdmission( + transaction: RelayDatabase, + cellId: string, + fallback: CellAdmissionState, + now: number +): Promise { + await transaction.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES (?, ?, ?) ON CONFLICT (cell_id) DO NOTHING`, + [cellId, fallback, now] + ) +} + +export async function cellAdmissionState( + database: RelayDatabase, + cellId: string +): Promise { + const row = ( + await database.query( + `SELECT admission_state FROM relay_cell_admission WHERE cell_id = ?`, + [cellId] + ) + )[0] + if (!row) throw new Error('cell_admission_missing') + return admissionState(row) +} + +export async function cellAdmissionStates( + database: RelayDatabase +): Promise> { + const rows = await database.query( + `SELECT cell.cell_id, admission.admission_state + FROM relay_cells cell + LEFT JOIN relay_cell_admission admission ON admission.cell_id = cell.cell_id + ORDER BY cell.cell_id ASC` + ) + return new Map(rows.map((row) => [text(row, 'cell_id'), admissionState(row)])) +} + +export async function setCellAdmissionBeforeBoundary( + transaction: RelayDatabase, + cellId: string, + state: CellAdmissionState, + now: number +): Promise { + const cells = await transaction.queryLocked( + `SELECT cell_id FROM relay_cells ORDER BY cell_id ASC` + ) + if (!cells.some((row) => text(row, 'cell_id') === cellId)) { + throw new Error('cell_not_found') + } + if ((await lockedSelector(transaction)).generation > 0) { + throw new Error('admission_selector_boundary_active') + } + const updated = await transaction.query( + `UPDATE relay_cells + SET updated_at = CASE WHEN enabled <> ? THEN ? ELSE updated_at END, enabled = ? + WHERE cell_id = ?`, + [enabledForState(state), now, enabledForState(state), cellId] + ) + if (integer(updated[0]!, 'changes') !== 1) throw new Error('cell_not_found') + await ensureCellAdmission(transaction, cellId, state, now) + await transaction.query( + `UPDATE relay_cell_admission + SET updated_at = CASE WHEN admission_state <> ? THEN ? ELSE updated_at END, + admission_state = ? + WHERE cell_id = ?`, + [state, now, state, cellId] + ) + await synchronizeCellAdmissionBoundary(transaction, now) +} + +export async function synchronizeCellAdmissionBoundary( + transaction: RelayDatabase, + now: number +): Promise { + const selector = await lockedSelector(transaction) + if (selector.generation > 0) { + await requireSelectorMatchesAdmission(transaction, selector) + return + } + await transaction.query( + `UPDATE relay_admission_selectors SET membership_json = ?, updated_at = ? + WHERE selector_id = ? AND generation = 0`, + [encodeMembership(await currentMembership(transaction)), now, SELECTOR_ID] + ) +} + +export class RelayCellAdmissionSelector { + constructor( + private readonly database: RelayDatabase, + private readonly now: () => number + ) {} + + async apply(input: ApplySelectorInput): Promise<{ + changed: boolean + selector: CellAdmissionSelector + }> { + const membership = normalizeMembership(input.membership) + validateAttempt(input) + const encoded = encodeMembership(membership) + await this.persistIntent(input, membership, encoded) + return await this.database.transaction(async (transaction) => { + const cells = await transaction.queryLocked( + `SELECT cell_id FROM relay_cells ORDER BY cell_id ASC` + ) + requireExactMembership(cells, membership) + const selector = await lockedSelector(transaction) + if ( + selector.generation === input.expectedGeneration + 1 && + selector.attemptId === input.attemptId && + encodeMembership(selector.membership) === encoded + ) { + return { changed: false, selector } + } + if (selector.generation !== input.expectedGeneration) { + throw new Error('admission_selector_generation_mismatch') + } + requireExpectedMembership(selector, input.expectedMembershipSha256) + await requireSelectorMatchesAdmission(transaction, selector) + if (selector.generation > 0) { + const nextNonExisting = new Set([...membership.migrationOnly, ...membership.general]) + if (selector.membership.existingOnly.some((cellId) => nextNonExisting.has(cellId))) { + throw new Error('admission_selector_legacy_reenable') + } + } + const now = this.now() + for (const [state, cellIds] of membershipEntries(membership)) { + for (const cellId of cellIds) { + await transaction.query( + `UPDATE relay_cell_admission + SET updated_at = CASE WHEN admission_state <> ? THEN ? ELSE updated_at END, + admission_state = ? + WHERE cell_id = ?`, + [state, now, state, cellId] + ) + await transaction.query( + `UPDATE relay_cells + SET updated_at = CASE WHEN enabled <> ? THEN ? ELSE updated_at END, enabled = ? + WHERE cell_id = ?`, + [enabledForState(state), now, enabledForState(state), cellId] + ) + } + } + const intendedGeneration = input.expectedGeneration + 1 + const updated = await transaction.query( + `UPDATE relay_admission_selectors + SET generation = ?, attempt_id = ?, membership_json = ?, updated_at = ? + WHERE selector_id = ? AND generation = ?`, + [ + intendedGeneration, + input.attemptId, + encoded, + now, + SELECTOR_ID, + input.expectedGeneration + ] + ) + if (integer(updated[0]!, 'changes') !== 1) { + throw new Error('admission_selector_generation_mismatch') + } + await transaction.query( + `UPDATE relay_admission_selector_intents SET committed_at = ? + WHERE attempt_id = ?`, + [now, input.attemptId] + ) + return { + changed: true, + selector: { + generation: intendedGeneration, + attemptId: input.attemptId, + membership + } + } + }) + } + + async inspect(attemptId?: string): Promise { + return await this.database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT cell_id FROM relay_cells ORDER BY cell_id ASC`) + const selector = await lockedSelector(transaction) + await requireSelectorMatchesAdmission(transaction, selector) + if (!attemptId) return { selector, intent: null } + const row = ( + await transaction.queryLocked( + `SELECT * FROM relay_admission_selector_intents WHERE attempt_id = ?`, + [attemptId] + ) + )[0] + if (!row) return { selector, intent: null } + const membership = decodeMembership(text(row, 'membership_json')) + const previousMembership = decodeMembership(text(row, 'previous_membership_json')) + const intendedGeneration = integer(row, 'intended_generation') + const committed = + selector.generation === intendedGeneration && + selector.attemptId === attemptId && + encodeMembership(selector.membership) === encodeMembership(membership) + const unchanged = + selector.generation === integer(row, 'expected_generation') && + encodeMembership(selector.membership) === text(row, 'previous_membership_json') + return { + selector, + intent: { + attemptId, + expectedGeneration: integer(row, 'expected_generation'), + intendedGeneration, + previousMembership, + membership, + state: committed ? 'committed' : unchanged ? 'unchanged' : 'diverged' + } + } + }) + } + + private async persistIntent( + input: ApplySelectorInput, + membership: CellAdmissionMembership, + encoded: string + ): Promise { + await this.database.transaction(async (transaction) => { + const cells = await transaction.queryLocked( + `SELECT cell_id FROM relay_cells ORDER BY cell_id ASC` + ) + requireExactMembership(cells, membership) + const selector = await lockedSelector(transaction) + const existing = ( + await transaction.queryLocked( + `SELECT * FROM relay_admission_selector_intents WHERE attempt_id = ?`, + [input.attemptId] + ) + )[0] + if (existing) { + const addition = ( + await transaction.queryLocked( + `SELECT attempt_id FROM relay_admission_selector_cell_additions + WHERE attempt_id = ?`, + [input.attemptId] + ) + )[0] + if ( + addition || + integer(existing, 'expected_generation') !== input.expectedGeneration || + integer(existing, 'intended_generation') !== input.expectedGeneration + 1 || + text(existing, 'membership_json') !== encoded || + (input.expectedMembershipSha256 !== undefined && + membershipSha256(text(existing, 'previous_membership_json')) !== + input.expectedMembershipSha256) + ) { + throw new Error('admission_selector_attempt_mismatch') + } + return + } + requireExpectedMembership(selector, input.expectedMembershipSha256) + await transaction.query( + `INSERT INTO relay_admission_selector_intents + (attempt_id, expected_generation, intended_generation, previous_membership_json, + membership_json, created_at, committed_at) + VALUES (?, ?, ?, ?, ?, ?, NULL)`, + [ + input.attemptId, + input.expectedGeneration, + input.expectedGeneration + 1, + encodeMembership(selector.membership), + encoded, + this.now() + ] + ) + }) + } +} + +function requireExpectedMembership( + selector: CellAdmissionSelector, + expectedSha256: string | undefined +): void { + if (expectedSha256 === undefined) return + if ( + !/^[a-f0-9]{64}$/.test(expectedSha256) || + membershipSha256(encodeMembership(selector.membership)) !== expectedSha256 + ) { + throw new Error('admission_selector_membership_mismatch') + } +} + +function membershipSha256(encoded: string): string { + return createHash('sha256').update(encoded).digest('hex') +} + +export async function lockedSelector( + transaction: RelayDatabase +): Promise { + let row = ( + await transaction.queryLocked( + `SELECT * FROM relay_admission_selectors WHERE selector_id = ?`, + [SELECTOR_ID] + ) + )[0] + if (!row) { + const membership = await currentMembership(transaction) + await transaction.query( + `INSERT INTO relay_admission_selectors + (selector_id, generation, attempt_id, membership_json, updated_at) + VALUES (?, 0, NULL, ?, 0) ON CONFLICT (selector_id) DO NOTHING`, + [SELECTOR_ID, encodeMembership(membership)] + ) + row = ( + await transaction.queryLocked( + `SELECT * FROM relay_admission_selectors WHERE selector_id = ?`, + [SELECTOR_ID] + ) + )[0] + } + if (!row) throw new Error('admission_selector_missing') + return { + generation: integer(row, 'generation'), + attemptId: optionalText(row, 'attempt_id'), + membership: decodeMembership(text(row, 'membership_json')) + } +} + +async function currentMembership(database: RelayDatabase): Promise { + const rows = await database.query( + `SELECT cell.cell_id, admission.admission_state + FROM relay_cells cell + LEFT JOIN relay_cell_admission admission ON admission.cell_id = cell.cell_id + ORDER BY cell.cell_id ASC` + ) + const membership: CellAdmissionMembership = { + existingOnly: [], + migrationOnly: [], + general: [] + } + for (const row of rows) membership[keyForState(admissionState(row))].push(text(row, 'cell_id')) + return membership +} + +export async function requireSelectorMatchesAdmission( + database: RelayDatabase, + selector: CellAdmissionSelector +): Promise { + const current = await currentMembership(database) + if (encodeMembership(current) !== encodeMembership(selector.membership)) { + throw new Error('admission_selector_membership_drift') + } +} + +export function normalizeMembership( + input: CellAdmissionMembership +): CellAdmissionMembership { + const membership = { + existingOnly: [...input.existingOnly].sort(), + migrationOnly: [...input.migrationOnly].sort(), + general: [...input.general].sort() + } + const all = [...membership.existingOnly, ...membership.migrationOnly, ...membership.general] + if (new Set(all).size !== all.length) throw new Error('admission_selector_duplicate_cell') + return membership +} + +function requireExactMembership(rows: SqlRow[], membership: CellAdmissionMembership): void { + const expected = rows.map((row) => text(row, 'cell_id')).sort() + const actual = [ + ...membership.existingOnly, + ...membership.migrationOnly, + ...membership.general + ].sort() + if (JSON.stringify(actual) !== JSON.stringify(expected)) { + throw new Error('admission_selector_incomplete_membership') + } +} + +function membershipEntries( + membership: CellAdmissionMembership +): Array<[CellAdmissionState, string[]]> { + return [ + ['existing-only', membership.existingOnly], + ['migration-only', membership.migrationOnly], + ['general', membership.general] + ] +} + +export function encodeMembership(membership: CellAdmissionMembership): string { + return JSON.stringify(normalizeMembership(membership)) +} + +export function decodeMembership(value: string): CellAdmissionMembership { + const parsed = JSON.parse(value) as Partial + if ( + !Array.isArray(parsed.existingOnly) || + !Array.isArray(parsed.migrationOnly) || + !Array.isArray(parsed.general) || + [...parsed.existingOnly, ...parsed.migrationOnly, ...parsed.general].some( + (cellId) => typeof cellId !== 'string' + ) + ) { + throw new Error('admission_selector_invalid_membership') + } + return normalizeMembership({ + existingOnly: parsed.existingOnly as string[], + migrationOnly: parsed.migrationOnly as string[], + general: parsed.general as string[] + }) +} + +function admissionState(row: SqlRow): CellAdmissionState { + return parseCellAdmissionState(text(row, 'admission_state')) +} + +function keyForState(state: CellAdmissionState): keyof CellAdmissionMembership { + if (state === 'existing-only') return 'existingOnly' + if (state === 'migration-only') return 'migrationOnly' + return 'general' +} + +function validateAttempt(input: { + attemptId: string + expectedGeneration: number + expectedMembershipSha256?: string +}): void { + if (!/^[A-Za-z0-9_-]{8,128}$/.test(input.attemptId)) { + throw new Error('invalid_admission_selector_attempt') + } + if (!Number.isSafeInteger(input.expectedGeneration) || input.expectedGeneration < 0) { + throw new Error('invalid_admission_selector_generation') + } + if (input.expectedGeneration === 0 && input.expectedMembershipSha256 === undefined) { + throw new Error('admission_selector_membership_fingerprint_required') + } +} + +function optionalText(row: SqlRow, field: string): string | null { + if (row[field] === null || row[field] === undefined) return null + return text(row, field) +} + +function integer(row: SqlRow, field: string): number { + const value = Number(row[field]) + if (!Number.isSafeInteger(value)) throw new Error(`invalid integer field ${field}`) + return value +} + +function text(row: SqlRow, field: string): string { + const value = row[field] + if (typeof value !== 'string') throw new Error(`invalid text field ${field}`) + return value +} diff --git a/cloud/apps/relay/src/cell-admission-startup-postgres.test.ts b/cloud/apps/relay/src/cell-admission-startup-postgres.test.ts new file mode 100644 index 00000000000..6f93fecce6f --- /dev/null +++ b/cloud/apps/relay/src/cell-admission-startup-postgres.test.ts @@ -0,0 +1,83 @@ +import pg from 'pg' +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { reconcileCellAdmissionAtStartup } from './cell-admission-startup.js' +import type { RelayCellConfig } from './config.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const schema = 'relay_startup_retry_test' +const cell: RelayCellConfig = { + id: 'startup-retry-cell', + url: 'https://startup-retry.example.test', + capacityRequests: 4_000 +} + +afterEach(() => vi.restoreAllMocks()) + +describePostgres('PostgreSQL director startup reconciliation', () => { + const databases: RelayDatabase[] = [] + let scopedDatabaseUrl = '' + + beforeAll(async () => { + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + await client.query(`CREATE SCHEMA ${schema}`) + } finally { + await client.end() + } + const url = new URL(databaseUrl!) + url.searchParams.set('options', `-c search_path=${schema}`) + scopedDatabaseUrl = url.toString() + databases.push( + await openRelayDatabase({ databaseUrl: scopedDatabaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl: scopedDatabaseUrl, dataDir: '' }) + ) + }) + + afterAll(async () => { + for (const database of databases) await database.close() + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + } finally { + await client.end() + } + }) + + it('recovers after the cell inventory lock outlasts transaction retries', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const store = new RelayAssignmentStore(databases[0]!, () => 100) + await store.reconcileCells([cell], false) + let releaseLock: () => void = () => undefined + let reportLocked: () => void = () => undefined + const lockReleased = new Promise((resolve) => (releaseLock = resolve)) + const lockAcquired = new Promise((resolve) => (reportLocked = resolve)) + const holder = databases[1]!.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT cell_id FROM relay_cells ORDER BY cell_id ASC`) + reportLocked() + await lockReleased + }) + await lockAcquired + + const releaseTimer = setTimeout(releaseLock, 3_500) + try { + await reconcileCellAdmissionAtStartup({ role: 'director', cells: [cell] }, store) + } finally { + clearTimeout(releaseTimer) + releaseLock() + await holder + } + + expect(await databases[0]!.query(`SELECT cell_id FROM relay_cells`)).toEqual([ + { cell_id: cell.id } + ]) + const warnings = warn.mock.calls.flat().join('\n') + expect(warnings).toContain('orca_relay_startup_reconcile_recovered') + expect(warnings).not.toContain('orca_relay_postgres_transaction_exhausted') + }, 10_000) +}) diff --git a/cloud/apps/relay/src/cell-admission-startup.test.ts b/cloud/apps/relay/src/cell-admission-startup.test.ts new file mode 100644 index 00000000000..d08d9eefc16 --- /dev/null +++ b/cloud/apps/relay/src/cell-admission-startup.test.ts @@ -0,0 +1,119 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + reconcileCellAdmissionAtStartup, + roleOwnsAssignmentMaintenance +} from './cell-admission-startup.js' +import type { RelayCellConfig } from './config.js' + +const cells: RelayCellConfig[] = [ + { + id: 'candidate', + url: 'https://candidate.relay.example.com', + capacityRequests: 4_000, + initiallyEnabled: false + } +] + +afterEach(() => { + vi.useRealTimers() + vi.restoreAllMocks() +}) + +describe('cell admission startup authority', () => { + it('reserves global assignment maintenance for director-capable roles', () => { + expect(roleOwnsAssignmentMaintenance('cell')).toBe(false) + expect(roleOwnsAssignmentMaintenance('director')).toBe(true) + expect(roleOwnsAssignmentMaintenance('combined')).toBe(true) + }) + + it('does not let a cell process reconcile its own admission', async () => { + const reconcileCellsAtStartup = vi.fn() + await reconcileCellAdmissionAtStartup({ role: 'cell', cells }, { reconcileCellsAtStartup }) + expect(reconcileCellsAtStartup).not.toHaveBeenCalled() + }) + + it.each(['director', 'combined'] as const)( + 'lets the %s process reconcile configured admission without disabling missing cells', + async (role) => { + const reconcileCellsAtStartup = vi.fn() + await reconcileCellAdmissionAtStartup({ role, cells }, { reconcileCellsAtStartup }) + expect(reconcileCellsAtStartup).toHaveBeenCalledWith(cells) + } + ) + + it('waits out transient database pressure during director startup', async () => { + vi.useFakeTimers() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const lockUnavailable = Object.assign(new Error('lock unavailable'), { code: '55P03' }) + const reconcileCellsAtStartup = vi + .fn() + .mockRejectedValueOnce(lockUnavailable) + .mockRejectedValueOnce(lockUnavailable) + .mockResolvedValue(undefined) + + const startup = reconcileCellAdmissionAtStartup( + { role: 'director', cells }, + { reconcileCellsAtStartup } + ) + await vi.runAllTimersAsync() + await startup + + expect(reconcileCellsAtStartup).toHaveBeenCalledTimes(3) + expect(warn).toHaveBeenCalledWith( + expect.stringContaining('orca_relay_startup_reconcile_recovered') + ) + }) + + it('fails startup immediately for a permanent reconciliation error', async () => { + const failure = new Error('invalid cell configuration') + const reconcileCellsAtStartup = vi.fn().mockRejectedValue(failure) + + await expect( + reconcileCellAdmissionAtStartup({ role: 'director', cells }, { reconcileCellsAtStartup }) + ).rejects.toBe(failure) + expect(reconcileCellsAtStartup).toHaveBeenCalledTimes(1) + }) + + it('bounds transient startup retries', async () => { + vi.useFakeTimers() + vi.spyOn(Math, 'random').mockReturnValue(0) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const lockUnavailable = Object.assign(new Error('lock unavailable'), { code: '55P03' }) + const reconcileCellsAtStartup = vi.fn().mockRejectedValue(lockUnavailable) + + const startup = reconcileCellAdmissionAtStartup( + { role: 'director', cells }, + { reconcileCellsAtStartup } + ) + const rejection = expect(startup).rejects.toBe(lockUnavailable) + await vi.runAllTimersAsync() + await rejection + + expect(reconcileCellsAtStartup).toHaveBeenCalledTimes(20) + expect(warn).toHaveBeenCalledWith( + expect.stringContaining('orca_relay_startup_reconcile_exhausted') + ) + }) + + it('bounds retry wall time when each reconciliation is slow', async () => { + vi.useFakeTimers() + vi.setSystemTime(0) + vi.spyOn(Math, 'random').mockReturnValue(0) + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const lockUnavailable = Object.assign(new Error('lock unavailable'), { code: '55P03' }) + const reconcileCellsAtStartup = vi.fn().mockImplementation(async () => { + vi.setSystemTime(Date.now() + 10_000) + throw lockUnavailable + }) + + const startup = reconcileCellAdmissionAtStartup( + { role: 'director', cells }, + { reconcileCellsAtStartup } + ) + const rejection = expect(startup).rejects.toBe(lockUnavailable) + await vi.runAllTimersAsync() + await rejection + + expect(reconcileCellsAtStartup).toHaveBeenCalledTimes(5) + }) +}) diff --git a/cloud/apps/relay/src/cell-admission-startup.ts b/cloud/apps/relay/src/cell-admission-startup.ts new file mode 100644 index 00000000000..384b6aab304 --- /dev/null +++ b/cloud/apps/relay/src/cell-admission-startup.ts @@ -0,0 +1,56 @@ +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import { isRelayDatabaseTransientError } from './database.js' + +type CellAdmissionStartupConfig = Pick +type CellAdmissionStore = Pick + +const STARTUP_RECONCILE_ATTEMPTS = 20 +const STARTUP_RECONCILE_RETRY_WINDOW_MS = 45_000 +const STARTUP_RECONCILE_RETRY_BASE_MS = 250 +const STARTUP_RECONCILE_RETRY_JITTER_MS = 250 + +export function roleOwnsAssignmentMaintenance(role: RelayConfig['role']): boolean { + // Cell workers share the database but the director is the sole authority + // for global expiry, evacuation, and dead-cell maintenance. + return role !== 'cell' +} + +export async function reconcileCellAdmissionAtStartup( + config: CellAdmissionStartupConfig, + assignments: CellAdmissionStore +): Promise { + // Admission is operator/director state. A new worker must not enable itself + // before its distinct candidate has passed production preflight. + if (config.role === 'cell') return + const retryDeadline = Date.now() + STARTUP_RECONCILE_RETRY_WINDOW_MS + for (let attempt = 1; attempt <= STARTUP_RECONCILE_ATTEMPTS; attempt += 1) { + try { + await assignments.reconcileCellsAtStartup(config.cells) + if (attempt > 1) { + console.warn( + JSON.stringify({ event: 'orca_relay_startup_reconcile_recovered', attempts: attempt }) + ) + } + return + } catch (error) { + const remainingMs = retryDeadline - Date.now() + if ( + attempt === STARTUP_RECONCILE_ATTEMPTS || + remainingMs <= 0 || + !isRelayDatabaseTransientError(error) + ) { + if (isRelayDatabaseTransientError(error)) { + console.warn( + JSON.stringify({ event: 'orca_relay_startup_reconcile_exhausted', attempts: attempt }) + ) + } + throw error + } + const delayMs = + STARTUP_RECONCILE_RETRY_BASE_MS + + Math.floor(Math.random() * (STARTUP_RECONCILE_RETRY_JITTER_MS + 1)) + await new Promise((resolve) => setTimeout(resolve, Math.min(delayMs, remainingMs))) + } + } +} diff --git a/cloud/apps/relay/src/cell-connection-hard-cap-consistency.test.ts b/cloud/apps/relay/src/cell-connection-hard-cap-consistency.test.ts new file mode 100644 index 00000000000..56859c36132 --- /dev/null +++ b/cloud/apps/relay/src/cell-connection-hard-cap-consistency.test.ts @@ -0,0 +1,75 @@ +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { + isRelayCellConnectionHardCap, + RELAY_ADMISSION_BUDGETS, + RELAY_CELL_ADMISSION_BOUNDS, + RELAY_CELL_CONNECTION_HARD_CAP, + RELAY_CELL_CONNECTION_HARD_CAPS, + relayCellAdmissionBounds +} from '@orca-cloud/relay-contract' +import { describe, expect, it } from 'vitest' + +// Why: dev scripts run standalone in CI and Terraform cannot read TypeScript, so neither can +// import the constant. Both restate it instead. This asserts every restatement still agrees, so +// raising the cap fails loudly here rather than silently leaving a surface behind. +// Repository root, and every path read below stays inside the Relay tree, so this survives the +// move under cloud/ in the public repository. +const repositoryRoot = fileURLToPath(new URL('../../../', import.meta.url)) +const read = (path: string): string => readFileSync(`${repositoryRoot}${path}`, 'utf8') + +describe('cell connection hard cap stays consistent across surfaces that cannot import it', () => { + it('derives its own bounds from the cap', () => { + expect(RELAY_CELL_ADMISSION_BOUNDS.hardCap).toBe(RELAY_CELL_CONNECTION_HARD_CAP) + expect(RELAY_CELL_ADMISSION_BOUNDS.socketAdmissionCeiling).toBe( + RELAY_CELL_CONNECTION_HARD_CAP - RELAY_ADMISSION_BUDGETS.reservedHostControls + ) + expect(RELAY_CELL_ADMISSION_BOUNDS.maxUnobservedBound).toBe( + RELAY_CELL_ADMISSION_BOUNDS.socketAdmissionCeiling - 1 + ) + for (const hardCap of RELAY_CELL_CONNECTION_HARD_CAPS) { + const bounds = relayCellAdmissionBounds(hardCap) + expect(bounds.socketAdmissionCeiling).toBe( + hardCap - RELAY_ADMISSION_BUDGETS.reservedHostControls + ) + expect(bounds.maxUnobservedBound).toBe(bounds.socketAdmissionCeiling - 1) + } + }) + + it.each([ + ['dev/scripts/relay-recovery-wave-gate.mjs', /targetConnectionCap:\s*(\d+)/], + ['dev/scripts/deploy-relay-gce-multi-target.mjs', /DEFAULT_CONNECTION_CEILING\s*=\s*(\d+)/], + ['dev/scripts/deploy-relay-gce-multi-target.mjs', /CUTOVER_CONNECTION_HARD_CAP\s*=\s*(\d+)/] + ])('%s matches the contract cap', (path, pattern) => { + const match = read(path).match(pattern) + expect(match, `${path} no longer declares ${pattern}`).not.toBeNull() + expect(Number(match![1])).toBe(RELAY_CELL_CONNECTION_HARD_CAP) + }) + + it('every production cell declaring a cap uses a supported contract cap', () => { + const declared = [ + ...read('infra/terraform/environments/production.tfvars').matchAll( + /connection_hard_cap\s*=\s*(\d+)/g + ) + ].map((match) => Number(match[1])) + expect(declared.length).toBeGreaterThan(0) + expect(new Set(declared).has(RELAY_CELL_CONNECTION_HARD_CAP)).toBe(true) + expect(declared.every((hardCap) => isRelayCellConnectionHardCap(hardCap))).toBe(true) + }) + + it('the Terraform cell validation accepts exactly the supported contract caps', () => { + const match = read('infra/terraform/variables.tf').match( + /contains\(\[([^\]]+)\], cell\.connection_hard_cap\)/ + ) + expect(match, 'variables.tf no longer validates connection_hard_cap').not.toBeNull() + expect(match![1]!.split(',').map((value) => Number(value.trim()))).toEqual([ + ...RELAY_CELL_CONNECTION_HARD_CAPS + ]) + }) + + it('the Terraform unobserved bound leaves the contract rebind reserve', () => { + expect(read('infra/terraform/variables.tf')).toContain( + 'cell.connection_unobserved_bound < cell.connection_hard_cap - 100' + ) + }) +}) diff --git a/cloud/apps/relay/src/cell-fence-legacy-adoption-postgres.test.ts b/cloud/apps/relay/src/cell-fence-legacy-adoption-postgres.test.ts new file mode 100644 index 00000000000..9fc1b915481 --- /dev/null +++ b/cloud/apps/relay/src/cell-fence-legacy-adoption-postgres.test.ts @@ -0,0 +1,122 @@ +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const cell = { + id: 'legacy-fence-adoption-postgres', + url: 'https://legacy-fence-adoption-postgres.example.com', + capacityRequests: 100 +} +const incarnation = '11111111-1111-4111-8111-111111111111' +const attemptId = '22222222-2222-4222-8222-222222222222' + +describePostgres('PostgreSQL legacy fence adoption', () => { + const databases: RelayDatabase[] = [] + + beforeAll(async () => { + databases.push( + await openRelayDatabase({ databaseUrl, dataDir: '' }), + await openRelayDatabase({ databaseUrl, dataDir: '' }) + ) + await cleanup() + }) + + afterAll(async () => { + await cleanup() + for (const database of databases) await database.close() + }) + + async function cleanup(): Promise { + const database = databases[0] + if (!database) return + await database.query( + `DELETE FROM relay_cell_fence_plan_bindings WHERE attempt_id = ?`, + [attemptId] + ) + await database.query( + `DELETE FROM relay_cell_fence_apply_invocations WHERE attempt_id = ?`, + [attemptId] + ) + await database.query(`DELETE FROM relay_cell_committed_fences WHERE cell_id = ?`, [ + cell.id + ]) + await database.query( + `DELETE FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + [cell.id] + ) + await database.query(`DELETE FROM relay_cell_fence_attempts WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_fences WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_admission WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [cell.id]) + } + + it('allows either adoption or attempt preparation, never both', async () => { + let now = 100 + const stores = databases.map( + (database) => + new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + ) + await stores[0]!.reconcileCells([cell], false) + await stores[0]!.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: incarnation, + startedAt: 50, + ready: true, + observedRequests: 0 + }) + await stores[0]!.setCellEnabled(cell.id, false) + now += 45_001 + const evidence = { + attemptId, + environment: 'production' as const, + cellId: cell.id, + cellIncarnation: incarnation, + migName: 'orca-relay-c3', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c3', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c3-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + planObjectName: + 'terraform/state/relay-fence-plans/production/22222222-2222-4222-8222-222222222222.tfplan', + planObjectGeneration: '123456789', + varFileSha256: 'c'.repeat(64), + terraformStateLineage: '33333333-3333-4333-8333-333333333333', + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: 'd'.repeat(64), + requestReason: + 'orca-relay-fence/22222222-2222-4222-8222-222222222222' + } + + const results = await Promise.allSettled([ + stores[0]!.adoptLegacyCellFence(cell.id, incarnation), + stores[1]!.prepareCellFenceAttempt(evidence) + ]) + expect(results.filter(({ status }) => status === 'fulfilled')).toHaveLength(1) + expect(results.filter(({ status }) => status === 'rejected')).toHaveLength(1) + if (results[0]!.status === 'fulfilled') { + await stores[0]!.commitLegacyCellFenceAdoption(cell.id, incarnation) + } + const attempts = await databases[0]!.query( + `SELECT COUNT(*) AS count FROM relay_cell_fence_attempts WHERE cell_id = ?`, + [cell.id] + ) + const fences = await databases[0]!.query( + `SELECT COUNT(*) AS count FROM relay_cell_fences WHERE cell_id = ?`, + [cell.id] + ) + const adoptions = await databases[0]!.query( + `SELECT COUNT(*) AS count FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + [cell.id] + ) + expect(Number(attempts[0]!.count) + Number(fences[0]!.count)).toBe(1) + expect(Number(adoptions[0]!.count)).toBe(Number(fences[0]!.count)) + }) +}) diff --git a/cloud/apps/relay/src/cell-heartbeat-client.test.ts b/cloud/apps/relay/src/cell-heartbeat-client.test.ts new file mode 100644 index 00000000000..2aa708bed1b --- /dev/null +++ b/cloud/apps/relay/src/cell-heartbeat-client.test.ts @@ -0,0 +1,176 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RelayConfig } from './config.js' +import { startCellHeartbeat } from './cell-heartbeat-client.js' + +const CONFIG = { + role: 'cell', + cellId: 'cell-a', + cellUrl: 'https://relay-a.example.com', + directorUrl: 'https://relay.example.com', + heartbeatAudience: 'https://relay.example.com/v1/admin/cell-heartbeat' +} as RelayConfig + +describe('cell heartbeat client', () => { + afterEach(() => vi.restoreAllMocks()) + + it('sends an immediate authenticated heartbeat without putting credentials in the URL', async () => { + const requests: Array<{ url: string; init?: RequestInit }> = [] + const fetchImpl = vi.fn(async (url: string | URL | Request, init?: RequestInit) => { + requests.push({ url: String(url), init }) + return new Response('{}', { status: 200 }) + }) as typeof fetch + const client = startCellHeartbeat(CONFIG, { + ready: async () => true, + observedRequests: () => 7, + connectionCounts: () => ({ + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }), + fetch: fetchImpl, + identityToken: async () => 'secret-token', + now: () => 123, + incarnation: '11111111-1111-4111-8111-111111111111', + intervalMs: 60_000 + })! + await vi.waitFor(() => expect(requests).toHaveLength(1)) + client.stop() + + expect(requests[0]!.url).toBe('https://relay.example.com/v1/admin/cell-heartbeat') + expect(requests[0]!.url).not.toContain('secret-token') + expect(requests[0]!.init?.headers).toMatchObject({ authorization: 'Bearer secret-token' }) + expect(JSON.parse(String(requests[0]!.init?.body))).toEqual({ + v: 1, + cellId: 'cell-a', + cellUrl: 'https://relay-a.example.com', + region: 'us-central1', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 123, + ready: true, + observedRequests: 7 + }) + }) + + it('advertises per-host drain only when its dedicated trust boundary is configured', async () => { + const requests: RequestInit[] = [] + const client = startCellHeartbeat( + { + ...CONFIG, + rehomeAudience: 'https://relay.example.com/v1/admin/host-drain', + rehomeDirectorServiceAccount: 'relay-director@example.com' + }, + { + ready: async () => true, + observedRequests: () => 0, + connectionCounts: () => ({ + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }), + regionalRehomeSafety: () => ({ + observedAt: 120, + sqlFailures: 0, + reconnects: 2, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + }), + fetch: async (_input, init) => { + requests.push(init ?? {}) + return new Response('{}', { status: 200 }) + }, + identityToken: async () => 'secret-token', + now: () => 123, + incarnation: '11111111-1111-4111-8111-111111111111', + intervalMs: 60_000 + } + )! + await vi.waitFor(() => expect(requests).toHaveLength(2)) + client.stop() + + expect(JSON.parse(String(requests[1]!.body))).toMatchObject({ + regionalRehomeProtocol: 1, + safety: { + observedAt: 120, + sqlFailures: 0, + reconnects: 2, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } + }) + }) + + it('reports the complete enforced connection ledger for a limited cell', async () => { + const requests: RequestInit[] = [] + const client = startCellHeartbeat( + { + ...CONFIG, + connectionHardCap: 600, + connectionUnobservedBound: 60 + }, + { + ready: async () => true, + observedRequests: () => 9, + connectionCounts: () => ({ + totalConnections: 10, + inFlightConnections: 2, + reservedConnectionUnits: 3, + enforcedConnectionUnits: 15, + inclusionWatermark: 42 + }), + fetch: async (_input, init) => { + requests.push(init ?? {}) + return new Response('{}', { status: 200 }) + }, + identityToken: async () => 'secret-token', + now: () => 123, + incarnation: '11111111-1111-4111-8111-111111111111', + intervalMs: 60_000 + } + )! + await vi.waitFor(() => expect(requests).toHaveLength(1)) + client.stop() + + expect(JSON.parse(String(requests[0]!.body))).toMatchObject({ + totalConnections: 10, + inFlightConnections: 2, + reservedConnectionUnits: 3, + enforcedConnectionUnits: 15, + connectionInclusionWatermark: 42, + connectionHardCap: 600, + connectionUnobservedBound: 60 + }) + }) + + it('does not start outside an explicitly configured cell role', () => { + expect( + startCellHeartbeat({ ...CONFIG, role: 'director' }, { + ready: async () => true, + observedRequests: () => 0, + connectionCounts: () => ({ + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }) + }) + ).toBeNull() + expect( + startCellHeartbeat({ ...CONFIG, directorUrl: undefined }, { + ready: async () => true, + observedRequests: () => 0, + connectionCounts: () => ({ + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }) + }) + ).toBeNull() + }) +}) diff --git a/cloud/apps/relay/src/cell-heartbeat-client.ts b/cloud/apps/relay/src/cell-heartbeat-client.ts new file mode 100644 index 00000000000..3bbcd08ecd6 --- /dev/null +++ b/cloud/apps/relay/src/cell-heartbeat-client.ts @@ -0,0 +1,123 @@ +import { randomUUID } from 'node:crypto' +import type { RelayConfig } from './config.js' +import { googleMetadataIdentityToken } from './google-metadata-identity-token.js' +import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' + +type ConnectionCounts = { + totalConnections: number + inFlightConnections: number + reservedConnectionUnits: number + enforcedConnectionUnits: number + inclusionWatermark?: number +} + +type HeartbeatClientOptions = { + ready: () => Promise + observedRequests: () => number + connectionCounts: () => ConnectionCounts + regionalRehomeSafety?: () => RegionalRehomeSafetySnapshot + fetch?: typeof fetch + identityToken?: (audience: string) => Promise + now?: () => number + incarnation?: string + intervalMs?: number +} + +export type CellHeartbeatClient = { stop: () => void; send: () => Promise } + +export function startCellHeartbeat( + config: RelayConfig, + options: HeartbeatClientOptions +): CellHeartbeatClient | null { + if (config.role !== 'cell' || !config.directorUrl || !config.heartbeatAudience) return null + const fetchImpl = options.fetch ?? fetch + const tokenProvider = + options.identityToken ?? ((audience) => googleMetadataIdentityToken(audience, fetchImpl)) + const startedAt = (options.now ?? Date.now)() + const cellIncarnation = options.incarnation ?? randomUUID() + let stopped = false + let inFlight = false + + const send = async (): Promise => { + if (stopped || inFlight) return + inFlight = true + try { + const [token, ready] = await Promise.all([ + tokenProvider(config.heartbeatAudience!), + options.ready() + ]) + const connectionCounts = + config.connectionHardCap === undefined ? null : options.connectionCounts() + const response = await fetchImpl( + new URL('/v1/admin/cell-heartbeat', config.directorUrl).toString(), + { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + cellId: config.cellId, + cellUrl: config.cellUrl, + region: config.region ?? 'us-central1', + cellIncarnation, + startedAt, + ready, + observedRequests: options.observedRequests(), + ...(config.connectionHardCap === undefined + ? {} + : { + totalConnections: connectionCounts!.totalConnections, + inFlightConnections: connectionCounts!.inFlightConnections, + reservedConnectionUnits: connectionCounts!.reservedConnectionUnits, + enforcedConnectionUnits: connectionCounts!.enforcedConnectionUnits, + connectionInclusionWatermark: + connectionCounts!.inclusionWatermark, + connectionHardCap: config.connectionHardCap, + connectionUnobservedBound: config.connectionUnobservedBound + }) + }), + signal: AbortSignal.timeout(10_000) + } + ) + if (!response.ok) throw new Error(`director_heartbeat_${response.status}`) + if (options.regionalRehomeSafety) { + const statusResponse = await fetchImpl( + new URL('/v1/admin/cell-rehome-status', config.directorUrl).toString(), + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + v: 1, + cellId: config.cellId, + cellIncarnation, + regionalRehomeProtocol: + config.rehomeAudience && config.rehomeDirectorServiceAccount ? 1 : 0, + safety: options.regionalRehomeSafety() + }), + signal: AbortSignal.timeout(10_000) + } + ) + if (!statusResponse.ok && statusResponse.status !== 404) { + throw new Error(`director_rehome_status_${statusResponse.status}`) + } + } + } catch (error) { + // A heartbeat must fail closed without ever logging its bearer token. + console.warn('[orca-relay] cell heartbeat failed', error instanceof Error ? error.message : '') + } finally { + inFlight = false + } + } + const timer = setInterval(() => void send(), options.intervalMs ?? 15_000) + timer.unref() + void send() + return { + send, + stop: () => { + stopped = true + clearInterval(timer) + } + } +} diff --git a/cloud/apps/relay/src/config.test.ts b/cloud/apps/relay/src/config.test.ts new file mode 100644 index 00000000000..bb0522dcbd3 --- /dev/null +++ b/cloud/apps/relay/src/config.test.ts @@ -0,0 +1,264 @@ +import { describe, expect, it } from 'vitest' +import { + loadRelayConfig, + RELAY_CELL_CONNECTION_HARD_CAP, + RELAY_DATABASE_POOL_MAX, + RELAY_DIRECTOR_DATABASE_POOL_MAX, + RELAY_MAX_CELL_CAPACITY_REQUESTS, + RELAY_PUBLIC_RESOLVE_CONCURRENCY, + RELAY_PUBLIC_RESOLVE_WAIT_MS +} from './config.js' + +function cellEnvironment(capacity: number): NodeJS.ProcessEnv { + return { + ORCA_RELAY_PUBLIC_URL: 'https://c1.relay.example.com', + ORCA_RELAY_CELL_URL: 'https://c1.relay.example.com', + ORCA_RELAY_AUTH_ISSUER: 'https://auth.example.com', + ORCA_RELAY_JWKS_URL: 'https://auth.example.com/.well-known/jwks.json', + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: 'assignment-key-with-at-least-thirty-two-bytes', + ORCA_RELAY_ROLE: 'cell', + ORCA_RELAY_CELL_ID: 'gce-c1', + ORCA_RELAY_CELL_CAPACITY: String(capacity), + ORCA_RELAY_ADMIN_AUDIENCE: 'https://relay.example.com/v1/admin/drain', + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.iam.gserviceaccount.com' + } +} + +describe('GCE relay capacity configuration', () => { + it('requires distinct dedicated admin identities and accepts omitted values', () => { + const env = cellEnvironment(4_000) + expect(loadRelayConfig(env)).toMatchObject({ + capacityServiceAccount: undefined, + asiaProofServiceAccount: undefined, + monitorServiceAccount: undefined, + fenceServiceAccount: undefined + }) + env.ORCA_RELAY_MONITOR_SERVICE_ACCOUNT = '' + env.ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT = 'capacity@example.iam.gserviceaccount.com' + env.ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT = 'proof@example.iam.gserviceaccount.com' + env.ORCA_RELAY_FENCE_SERVICE_ACCOUNT = 'fence@example.iam.gserviceaccount.com' + expect(loadRelayConfig(env)).toMatchObject({ + capacityServiceAccount: 'capacity@example.iam.gserviceaccount.com', + asiaProofServiceAccount: 'proof@example.iam.gserviceaccount.com', + monitorServiceAccount: undefined, + fenceServiceAccount: 'fence@example.iam.gserviceaccount.com' + }) + env.ORCA_RELAY_MONITOR_SERVICE_ACCOUNT = env.ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT + expect(() => loadRelayConfig(env)).toThrow('relay admin service accounts must be distinct') + env.ORCA_RELAY_MONITOR_SERVICE_ACCOUNT = undefined + env.ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT = env.ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT + expect(() => loadRelayConfig(env)).toThrow('relay admin service accounts must be distinct') + }) + + it('requires a distinct paired identity and exact audience for regional host drain', () => { + const env = cellEnvironment(4_000) + env.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT = + 'relay-director@example.iam.gserviceaccount.com' + expect(() => loadRelayConfig(env)).toThrow( + 'relay rehome identity and audience must be configured together' + ) + env.ORCA_RELAY_REHOME_AUDIENCE = 'https://relay.example.com/v1/admin/host-drain' + expect(loadRelayConfig(env)).toMatchObject({ + rehomeDirectorServiceAccount: 'relay-director@example.iam.gserviceaccount.com', + rehomeAudience: 'https://relay.example.com/v1/admin/host-drain' + }) + env.ORCA_RELAY_REHOME_AUDIENCE = 'https://relay.example.com/v1/admin/drain' + expect(() => loadRelayConfig(env)).toThrow( + 'relay rehome audience must target the host drain route' + ) + env.ORCA_RELAY_REHOME_AUDIENCE = 'https://relay.example.com/v1/admin/host-drain' + env.ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT = + env.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT + expect(() => loadRelayConfig(env)).toThrow( + 'relay rehome director identity must differ from the cell runtime identity' + ) + }) + + it('accepts a measured capacity above the Cloud Run request ceiling', () => { + expect(loadRelayConfig(cellEnvironment(4_000)).cells[0]?.capacityRequests).toBe(4_000) + }) + + it('retains a finite process-wide capacity bound', () => { + expect(() => loadRelayConfig(cellEnvironment(RELAY_MAX_CELL_CAPACITY_REQUESTS + 1))).toThrow() + }) + + it('keeps legacy cells uncapped and requires a supported hard-cap pair', () => { + const env = cellEnvironment(4_000) + expect(loadRelayConfig(env)).toMatchObject({ + connectionHardCap: undefined, + connectionUnobservedBound: undefined + }) + + env.ORCA_RELAY_CELL_CONNECTION_HARD_CAP = String(RELAY_CELL_CONNECTION_HARD_CAP) + expect(() => loadRelayConfig(env)).toThrow( + 'connection hard cap and unobserved bound must be configured together' + ) + env.ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND = '60' + expect(loadRelayConfig(env)).toMatchObject({ + connectionHardCap: 600, + connectionUnobservedBound: 60 + }) + env.ORCA_RELAY_CELL_CONNECTION_HARD_CAP = '599' + expect(() => loadRelayConfig(env)).toThrow() + env.ORCA_RELAY_CELL_CONNECTION_HARD_CAP = '600' + env.ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND = '500' + expect(() => loadRelayConfig(env)).toThrow() + env.ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND = '600' + expect(() => loadRelayConfig(env)).toThrow() + env.ORCA_RELAY_CELL_CONNECTION_HARD_CAP = '1000' + env.ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND = '60' + expect(loadRelayConfig(env)).toMatchObject({ + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + }) + env.ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND = '900' + expect(() => loadRelayConfig(env)).toThrow() + }) + + it('accepts hard-cap metadata in director cell inventory', () => { + const env = cellEnvironment(4_000) + env.ORCA_RELAY_ROLE = 'director' + env.ORCA_RELAY_PUBLIC_URL = 'https://relay.example.com' + env.ORCA_RELAY_CELL_URL = 'https://relay.example.com' + env.ORCA_RELAY_CELLS_JSON = JSON.stringify([ + { + id: 'gce-c7', + url: 'https://c7.relay.example.com', + capacityRequests: 4_000, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + } + ]) + + expect(loadRelayConfig(env).cells[0]).toMatchObject({ + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + }) + }) + + it('accepts mixed 600- and 1000-cap director inventory', () => { + const env = cellEnvironment(4_000) + env.ORCA_RELAY_ROLE = 'director' + env.ORCA_RELAY_PUBLIC_URL = 'https://relay.example.com' + env.ORCA_RELAY_CELL_URL = 'https://relay.example.com' + env.ORCA_RELAY_CELLS_JSON = JSON.stringify([ + { + id: 'gce-c1', + url: 'https://c1.relay.example.com', + capacityRequests: 4_000, + connectionHardCap: 600, + connectionUnobservedBound: 60 + }, + { + id: 'gce-c2', + url: 'https://c2.relay.example.com', + capacityRequests: 4_000, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + } + ]) + + expect(loadRelayConfig(env).cells.map((cell) => cell.connectionHardCap)).toEqual([ + 600, 1_000 + ]) + }) + + it('accepts only a canonical served image digest', () => { + const env = cellEnvironment(4_000) + env.ORCA_RELAY_IMAGE_DIGEST = `sha256:${'a'.repeat(64)}` + expect(loadRelayConfig(env).imageDigest).toBe(env.ORCA_RELAY_IMAGE_DIGEST) + env.ORCA_RELAY_IMAGE_DIGEST = 'relay:latest' + expect(() => loadRelayConfig(env)).toThrow() + }) + + it('accepts a statically declared candidate that starts disabled', () => { + const env = cellEnvironment(4_000) + env.ORCA_RELAY_ROLE = 'director' + env.ORCA_RELAY_PUBLIC_URL = 'https://relay.example.com' + env.ORCA_RELAY_CELL_URL = 'https://relay.example.com' + env.ORCA_RELAY_CELLS_JSON = JSON.stringify([ + { + id: 'gce-candidate', + url: 'https://candidate.relay.example.com', + capacityRequests: 4_000, + region: 'us-central1', + initiallyEnabled: false + } + ]) + expect(loadRelayConfig(env).cells).toEqual([ + { + id: 'gce-candidate', + url: 'https://candidate.relay.example.com', + capacityRequests: 4_000, + region: 'us-central1', + initiallyEnabled: false + } + ]) + }) + + it('reserves a smaller PostgreSQL connection budget for directors', () => { + const cellEnv = cellEnvironment(4_000) + expect(loadRelayConfig(cellEnv).databasePoolMax).toBe(RELAY_DATABASE_POOL_MAX) + + const directorEnv: NodeJS.ProcessEnv = { ...cellEnv, ORCA_RELAY_ROLE: 'director' } + directorEnv.ORCA_RELAY_CELLS_JSON = JSON.stringify([ + { + id: 'gce-c1', + url: 'https://c1.relay.example.com', + capacityRequests: 4_000 + } + ]) + expect(loadRelayConfig(directorEnv).databasePoolMax).toBe(RELAY_DIRECTOR_DATABASE_POOL_MAX) + + directorEnv.ORCA_RELAY_DATABASE_POOL_MAX = '7' + expect(loadRelayConfig(directorEnv).databasePoolMax).toBe(7) + }) + + it('defaults to bounded public assignment admission and supports an emergency stop', () => { + const env = cellEnvironment(4_000) + expect(loadRelayConfig(env)).toMatchObject({ + publicAssignmentsEnabled: true, + regionalPlacementEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: RELAY_PUBLIC_RESOLVE_CONCURRENCY, + publicResolveWaitMs: RELAY_PUBLIC_RESOLVE_WAIT_MS, + publicAssignmentRetryAfterSeconds: 5 + }) + + env.ORCA_RELAY_PUBLIC_ASSIGNMENTS_ENABLED = 'false' + env.ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED = 'false' + env.ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY = '4' + env.ORCA_RELAY_PUBLIC_ASSIGNMENT_QUEUE_MAX = '256' + env.ORCA_RELAY_PUBLIC_ASSIGNMENT_WAIT_MS = '8000' + env.ORCA_RELAY_PUBLIC_ASSIGNMENT_RETRY_AFTER_SECONDS = '30' + expect(loadRelayConfig(env)).toMatchObject({ + publicAssignmentsEnabled: false, + regionalPlacementEnabled: false, + publicAssignmentConcurrency: 4, + publicAssignmentQueueMax: 256, + publicAssignmentWaitMs: 8_000, + publicResolveConcurrency: RELAY_PUBLIC_RESOLVE_CONCURRENCY, + publicResolveWaitMs: RELAY_PUBLIC_RESOLVE_WAIT_MS, + publicAssignmentRetryAfterSeconds: 30 + }) + }) + + it('fails closed when public request lanes exceed the director database pool', () => { + const env = cellEnvironment(4_000) + env.ORCA_RELAY_ROLE = 'director' + env.ORCA_RELAY_CELLS_JSON = JSON.stringify([ + { + id: 'gce-c1', + url: 'https://c1.relay.example.com', + capacityRequests: 4_000 + } + ]) + env.ORCA_RELAY_DATABASE_POOL_MAX = '2' + + expect(() => loadRelayConfig(env)).toThrow( + 'public relay admission must leave database pool headroom' + ) + }) +}) diff --git a/cloud/apps/relay/src/config.ts b/cloud/apps/relay/src/config.ts new file mode 100644 index 00000000000..2bf23444a71 --- /dev/null +++ b/cloud/apps/relay/src/config.ts @@ -0,0 +1,348 @@ +import { z } from 'zod' +import { + isRelayCellConnectionHardCap, + RELAY_CELL_CONNECTION_HARD_CAP, + RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND, + RELAY_DEFAULT_REGION, + RelayRegionSchema, + relayCellAdmissionBounds, + type RelayCellConnectionHardCap, + type RelayRegion +} from '@orca-cloud/relay-contract' + +export const RELAY_MAX_CELL_CAPACITY_REQUESTS = 100_000 +export const RELAY_DATABASE_POOL_MAX = 10 +export const RELAY_DIRECTOR_DATABASE_POOL_MAX = 3 +export const RELAY_PUBLIC_RESOLVE_CONCURRENCY = 1 +export const RELAY_PUBLIC_RESOLVE_WAIT_MS = 5_000 +export { RELAY_CELL_CONNECTION_HARD_CAP } + +const RelayCellConnectionHardCapSchema = z.custom( + isRelayCellConnectionHardCap +) + +const EnvironmentBooleanSchema = z + .enum(['true', 'false']) + .default('true') + .transform((value) => value === 'true') + +const OptionalServiceAccountSchema = z.preprocess( + (value) => (value === '' ? undefined : value), + z.string().email().optional() +) + +const EnvSchema = z.object({ + PORT: z.coerce.number().int().positive().default(8080), + ORCA_RELAY_PUBLIC_URL: z.string().url(), + ORCA_RELAY_CELL_URL: z.string().url(), + ORCA_RELAY_AUTH_ISSUER: z.string().url(), + ORCA_RELAY_AUTH_AUDIENCE: z.literal('orca-relay').default('orca-relay'), + ORCA_RELAY_JWKS_URL: z.string().url(), + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: z.string().min(32), + ORCA_RELAY_ROLE: z.enum(['combined', 'director', 'cell']).default('combined'), + ORCA_RELAY_CELL_ID: z.string().min(1).max(128).default('combined'), + ORCA_RELAY_REGION: RelayRegionSchema.default(RELAY_DEFAULT_REGION), + ORCA_RELAY_CELL_CAPACITY: z.coerce + .number() + .int() + .positive() + .max(RELAY_MAX_CELL_CAPACITY_REQUESTS) + .default(900), + ORCA_RELAY_CELL_CONNECTION_HARD_CAP: z.coerce + .number() + .int() + .pipe(RelayCellConnectionHardCapSchema) + .optional(), + ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND: z.coerce + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND) + .optional(), + ORCA_RELAY_CELLS_JSON: z.string().default('[]'), + ORCA_RELAY_ADMIN_AUDIENCE: z.string().url(), + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: z.string().email(), + ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT: OptionalServiceAccountSchema, + ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT: OptionalServiceAccountSchema, + ORCA_RELAY_MONITOR_SERVICE_ACCOUNT: OptionalServiceAccountSchema, + ORCA_RELAY_FENCE_SERVICE_ACCOUNT: OptionalServiceAccountSchema, + ORCA_RELAY_FENCE_BROKER_SERVICE_ACCOUNT: OptionalServiceAccountSchema, + ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT: OptionalServiceAccountSchema, + ORCA_RELAY_REHOME_AUDIENCE: z.preprocess( + (value) => (value === '' ? undefined : value), + z.string().url().optional() + ), + ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT: z.string().email().optional(), + ORCA_RELAY_DIRECTOR_URL: z.string().url().optional(), + ORCA_RELAY_HEARTBEAT_AUDIENCE: z.string().url().optional(), + ORCA_RELAY_IMAGE_DIGEST: z.string().regex(/^sha256:[a-f0-9]{64}$/).optional(), + ORCA_RELAY_ADMIN_JWKS_URL: z.string().url().default('https://www.googleapis.com/oauth2/v3/certs'), + ORCA_RELAY_DATABASE_POOL_MAX: z.coerce.number().int().positive().max(100).optional(), + ORCA_RELAY_PUBLIC_ASSIGNMENTS_ENABLED: EnvironmentBooleanSchema, + ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED: EnvironmentBooleanSchema, + ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY: z.coerce.number().int().positive().max(100).default(2), + ORCA_RELAY_PUBLIC_STICKY_CONCURRENCY: z.coerce.number().int().positive().max(100).default(1), + ORCA_RELAY_PUBLIC_STICKY_QUEUE_MAX: z.coerce.number().int().positive().max(4_096).default(64), + ORCA_RELAY_PUBLIC_STICKY_WAIT_MS: z.coerce.number().int().positive().max(30_000).default(2_000), + ORCA_RELAY_PUBLIC_STICKY_RETRY_AFTER_SECONDS: z.coerce + .number() + .int() + .positive() + .max(60) + .default(2), + ORCA_RELAY_PUBLIC_ASSIGNMENT_QUEUE_MAX: z.coerce + .number() + .int() + .positive() + .max(4_096) + .default(128), + ORCA_RELAY_PUBLIC_ASSIGNMENT_WAIT_MS: z.coerce + .number() + .int() + .positive() + .max(30_000) + .default(4_000), + ORCA_RELAY_PUBLIC_ASSIGNMENT_RETRY_AFTER_SECONDS: z.coerce + .number() + .int() + .positive() + .max(300) + .default(5), + DATABASE_URL: z.string().optional(), + ORCA_RELAY_DATA_DIR: z.string().default('./data/relay') +}) + +const RelayCellConfigSchema = z + .object({ + id: z.string().min(1).max(128), + url: z.string().url(), + capacityRequests: z.number().int().positive().max(RELAY_MAX_CELL_CAPACITY_REQUESTS), + region: RelayRegionSchema.default(RELAY_DEFAULT_REGION), + initiallyEnabled: z.boolean().optional(), + connectionHardCap: RelayCellConnectionHardCapSchema.optional(), + connectionUnobservedBound: z + .number() + .int() + .nonnegative() + .max(RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND) + .optional() + }) + .strict() + .superRefine((value, context) => { + if ( + (value.connectionHardCap === undefined) !== + (value.connectionUnobservedBound === undefined) + ) { + context.addIssue({ + code: 'custom', + message: 'connection hard cap and unobserved bound must be configured together' + }) + } else if ( + value.connectionHardCap !== undefined && + value.connectionUnobservedBound! > + relayCellAdmissionBounds(value.connectionHardCap).maxUnobservedBound + ) { + context.addIssue({ + code: 'custom', + path: ['connectionUnobservedBound'], + message: 'connection unobserved bound must leave ordinary admission capacity' + }) + } + }) + +export type RelayCellConfig = Omit, 'region'> & { + region?: RelayRegion +} + +export type RelayConfig = { + port: number + publicUrl: string + cellUrl: string + authIssuer: string + authAudience: 'orca-relay' + jwksUrl: string + assignmentSigningKey: Uint8Array + role: 'combined' | 'director' | 'cell' + cellId: string + region?: RelayRegion + cells: RelayCellConfig[] + adminAudience: string + deployServiceAccount: string + capacityServiceAccount?: string + asiaProofServiceAccount?: string + monitorServiceAccount?: string + fenceServiceAccount?: string + fenceBrokerServiceAccount?: string + rehomeDirectorServiceAccount?: string + rehomeAudience?: string + runtimeServiceAccount: string + directorUrl?: string + heartbeatAudience?: string + imageDigest?: string + connectionHardCap?: RelayCellConnectionHardCap + connectionUnobservedBound?: number + adminJwksUrl: string + databasePoolMax: number + publicAssignmentsEnabled: boolean + regionalPlacementEnabled?: boolean + publicAssignmentConcurrency: number + publicAssignmentQueueMax: number + publicAssignmentWaitMs: number + publicResolveConcurrency: number + publicResolveWaitMs: number + publicAssignmentRetryAfterSeconds: number + publicStickyConcurrency?: number + publicStickyQueueMax?: number + publicStickyWaitMs?: number + publicStickyRetryAfterSeconds?: number + databaseUrl?: string + dataDir: string +} + +function canonicalOrigin(value: string, name: string): string { + const url = new URL(value) + if (url.origin !== value || url.pathname !== '/') throw new Error(`${name} must be an origin`) + const loopback = ['127.0.0.1', 'localhost', '::1', '[::1]'].includes(url.hostname) + if (url.protocol !== 'https:' && !(loopback && url.protocol === 'http:')) { + throw new Error(`${name} must use HTTPS outside loopback development`) + } + return value +} + +export function loadRelayConfig(env: NodeJS.ProcessEnv = process.env): RelayConfig { + const parsed = EnvSchema.parse(env) + const adminServiceAccounts = [ + parsed.ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT, + parsed.ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT, + parsed.ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT, + parsed.ORCA_RELAY_MONITOR_SERVICE_ACCOUNT, + parsed.ORCA_RELAY_FENCE_SERVICE_ACCOUNT, + parsed.ORCA_RELAY_FENCE_BROKER_SERVICE_ACCOUNT, + parsed.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT + ].filter((value): value is string => value !== undefined) + if (new Set(adminServiceAccounts).size !== adminServiceAccounts.length) { + throw new Error('relay admin service accounts must be distinct') + } + if ( + (parsed.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT === undefined) !== + (parsed.ORCA_RELAY_REHOME_AUDIENCE === undefined) + ) { + throw new Error('relay rehome identity and audience must be configured together') + } + if ( + parsed.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT && + parsed.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT === + (parsed.ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT ?? parsed.ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT) + ) { + throw new Error('relay rehome director identity must differ from the cell runtime identity') + } + if ( + parsed.ORCA_RELAY_REHOME_AUDIENCE && + new URL(parsed.ORCA_RELAY_REHOME_AUDIENCE).pathname !== '/v1/admin/host-drain' + ) { + throw new Error('relay rehome audience must target the host drain route') + } + const directorUrl = parsed.ORCA_RELAY_DIRECTOR_URL + ? canonicalOrigin(parsed.ORCA_RELAY_DIRECTOR_URL, 'ORCA_RELAY_DIRECTOR_URL') + : undefined + const publicUrl = canonicalOrigin(parsed.ORCA_RELAY_PUBLIC_URL, 'ORCA_RELAY_PUBLIC_URL') + const configuredCells = z + .array(RelayCellConfigSchema) + .max(128) + .parse(JSON.parse(parsed.ORCA_RELAY_CELLS_JSON) as unknown) + .map((cell) => ({ ...cell, url: canonicalOrigin(cell.url, `cell ${cell.id}`) })) + const ownCell = { + id: parsed.ORCA_RELAY_CELL_ID, + region: parsed.ORCA_RELAY_REGION, + url: canonicalOrigin(parsed.ORCA_RELAY_CELL_URL, 'ORCA_RELAY_CELL_URL'), + capacityRequests: parsed.ORCA_RELAY_CELL_CAPACITY, + connectionHardCap: parsed.ORCA_RELAY_CELL_CONNECTION_HARD_CAP, + connectionUnobservedBound: parsed.ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND, + initiallyEnabled: true + } + if ( + (ownCell.connectionHardCap === undefined) !== + (ownCell.connectionUnobservedBound === undefined) + ) { + throw new Error('connection hard cap and unobserved bound must be configured together') + } + if ( + ownCell.connectionHardCap !== undefined && + ownCell.connectionUnobservedBound! > + relayCellAdmissionBounds(ownCell.connectionHardCap).maxUnobservedBound + ) { + throw new Error('connection unobserved bound must leave ordinary admission capacity') + } + const cells = parsed.ORCA_RELAY_ROLE === 'director' ? configuredCells : [ownCell] + if (cells.length === 0) throw new Error('director requires at least one configured cell') + if (new Set(cells.map(({ id }) => id)).size !== cells.length) { + throw new Error('relay cell ids must be unique') + } + const databasePoolMax = + parsed.ORCA_RELAY_DATABASE_POOL_MAX ?? + (parsed.ORCA_RELAY_ROLE === 'director' + ? RELAY_DIRECTOR_DATABASE_POOL_MAX + : RELAY_DATABASE_POOL_MAX) + if ( + parsed.ORCA_RELAY_ROLE !== 'cell' && + parsed.ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY >= databasePoolMax + ) { + throw new Error('public relay admission must leave database pool headroom') + } + if ( + parsed.ORCA_RELAY_ROLE !== 'cell' && + parsed.ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY + parsed.ORCA_RELAY_PUBLIC_STICKY_CONCURRENCY > + databasePoolMax + ) { + throw new Error('sticky and placement admission together must fit the database pool') + } + return { + port: parsed.PORT, + publicUrl, + cellUrl: ownCell.url, + authIssuer: canonicalOrigin(parsed.ORCA_RELAY_AUTH_ISSUER, 'ORCA_RELAY_AUTH_ISSUER'), + authAudience: parsed.ORCA_RELAY_AUTH_AUDIENCE, + jwksUrl: parsed.ORCA_RELAY_JWKS_URL, + assignmentSigningKey: new TextEncoder().encode(parsed.ORCA_RELAY_ASSIGNMENT_SIGNING_KEY), + role: parsed.ORCA_RELAY_ROLE, + cellId: ownCell.id, + region: ownCell.region, + cells, + adminAudience: parsed.ORCA_RELAY_ADMIN_AUDIENCE, + deployServiceAccount: parsed.ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT, + capacityServiceAccount: parsed.ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT, + asiaProofServiceAccount: parsed.ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT, + monitorServiceAccount: parsed.ORCA_RELAY_MONITOR_SERVICE_ACCOUNT, + fenceServiceAccount: parsed.ORCA_RELAY_FENCE_SERVICE_ACCOUNT, + fenceBrokerServiceAccount: parsed.ORCA_RELAY_FENCE_BROKER_SERVICE_ACCOUNT, + rehomeDirectorServiceAccount: parsed.ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT, + rehomeAudience: parsed.ORCA_RELAY_REHOME_AUDIENCE, + runtimeServiceAccount: + parsed.ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT ?? parsed.ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT, + directorUrl, + heartbeatAudience: + parsed.ORCA_RELAY_HEARTBEAT_AUDIENCE ?? + (directorUrl || parsed.ORCA_RELAY_ROLE === 'director' + ? new URL('/v1/admin/cell-heartbeat', directorUrl ?? publicUrl).toString() + : undefined), + imageDigest: parsed.ORCA_RELAY_IMAGE_DIGEST, + connectionHardCap: ownCell.connectionHardCap, + connectionUnobservedBound: ownCell.connectionUnobservedBound, + adminJwksUrl: parsed.ORCA_RELAY_ADMIN_JWKS_URL, + databasePoolMax, + publicAssignmentsEnabled: parsed.ORCA_RELAY_PUBLIC_ASSIGNMENTS_ENABLED, + regionalPlacementEnabled: parsed.ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED, + publicAssignmentConcurrency: parsed.ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY, + publicAssignmentQueueMax: parsed.ORCA_RELAY_PUBLIC_ASSIGNMENT_QUEUE_MAX, + publicAssignmentWaitMs: parsed.ORCA_RELAY_PUBLIC_ASSIGNMENT_WAIT_MS, + publicResolveConcurrency: RELAY_PUBLIC_RESOLVE_CONCURRENCY, + publicResolveWaitMs: RELAY_PUBLIC_RESOLVE_WAIT_MS, + publicAssignmentRetryAfterSeconds: parsed.ORCA_RELAY_PUBLIC_ASSIGNMENT_RETRY_AFTER_SECONDS, + publicStickyConcurrency: parsed.ORCA_RELAY_PUBLIC_STICKY_CONCURRENCY, + publicStickyQueueMax: parsed.ORCA_RELAY_PUBLIC_STICKY_QUEUE_MAX, + publicStickyWaitMs: parsed.ORCA_RELAY_PUBLIC_STICKY_WAIT_MS, + publicStickyRetryAfterSeconds: parsed.ORCA_RELAY_PUBLIC_STICKY_RETRY_AFTER_SECONDS, + databaseUrl: parsed.DATABASE_URL, + dataDir: parsed.ORCA_RELAY_DATA_DIR + } +} diff --git a/cloud/apps/relay/src/connection-reservation-debt-retention.test.ts b/cloud/apps/relay/src/connection-reservation-debt-retention.test.ts new file mode 100644 index 00000000000..fd50f9d6747 --- /dev/null +++ b/cloud/apps/relay/src/connection-reservation-debt-retention.test.ts @@ -0,0 +1,130 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayCellConfig } from './config.js' +import { openInMemoryRelayDatabase, type RelayDatabase } from './database.js' + +const DEBT_RETENTION_MS = 10 * 60 * 1_000 + +const LIMITED_CELL: RelayCellConfig = { + id: 'limited', + url: 'https://limited.example.com', + capacityRequests: 1_000, + connectionHardCap: 600, + connectionUnobservedBound: 50 +} + +const databases: RelayDatabase[] = [] + +afterEach(async () => { + for (const database of databases.splice(0)) await database.close() +}) + +async function setup(now: () => number): Promise<{ + database: RelayDatabase + store: RelayAssignmentStore +}> { + const database = await openInMemoryRelayDatabase() + databases.push(database) + const store = new RelayAssignmentStore(database, now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells([LIMITED_CELL], true) + await store.recordCellHeartbeat({ + cellId: LIMITED_CELL.id, + cellUrl: LIMITED_CELL.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionHardCap: 600, + connectionUnobservedBound: 50 + }) + return { database, store } +} + +async function insertDebt( + database: RelayDatabase, + reservationId: string, + relayHostId: string, + timeoutAt: number, + claimActivityId: string | null = null +): Promise { + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, assignment_epoch, + cell_id, state, claim_activity_id, created_at, timeout_at, updated_at) + VALUES (?, ?, 'user-1', ?, 1, ?, 'late-arrival-debt', ?, ?, ?, ?)`, + [ + reservationId, + reservationId, + relayHostId, + LIMITED_CELL.id, + claimActivityId, + timeoutAt - 10_000, + timeoutAt, + timeoutAt + ] + ) +} + +async function reservationStates(database: RelayDatabase): Promise> { + const rows = await database.query( + `SELECT reservation_id, state FROM relay_control_connection_reservations` + ) + return new Map(rows.map((row) => [String(row['reservation_id']), String(row['state'])])) +} + +describe('late-arrival debt retention', () => { + it('releases unclaimed debt past retention and keeps fresh or claimed debt', async () => { + const now = 100_000_000 + const { database, store } = await setup(() => now) + await insertDebt(database, 'stale-debt', 'host000000000001', now - DEBT_RETENTION_MS - 1) + await insertDebt(database, 'fresh-debt', 'host000000000002', now - 30_000) + await insertDebt( + database, + 'claimed-debt', + 'host000000000003', + now - DEBT_RETENTION_MS - 1, + 'control:1' + ) + + await store.releaseExpiredActivityLeases() + + expect(await reservationStates(database)).toEqual( + new Map([ + ['stale-debt', 'released'], + ['fresh-debt', 'late-arrival-debt'], + ['claimed-debt', 'late-arrival-debt'] + ]) + ) + }) + + it('restores placement once stale debt no longer consumes connection headroom', async () => { + const now = 100_000_000 + const { database, store } = await setup(() => now) + // Headroom needs enforced + outstanding + unobserved(50) < hardCap(600) - + // rebindReserve(100); 450 stale debt rows are exactly enough to block. + for (let index = 0; index < 450; index++) { + await insertDebt( + database, + `stale-${index}`, + `host${String(index).padStart(12, '0')}`, + now - DEBT_RETENTION_MS - 1 + ) + } + await expect( + store.assign({ userId: 'user-9', relayHostId: 'hostfffffffffff9' }) + ).rejects.toThrow('relay_capacity_exhausted') + + await store.releaseExpiredActivityLeases() + + await expect( + store.assign({ userId: 'user-9', relayHostId: 'hostfffffffffff9' }) + ).resolves.toMatchObject({ cellId: LIMITED_CELL.id }) + }) +}) diff --git a/cloud/apps/relay/src/control-lease-recovery-postgres.test.ts b/cloud/apps/relay/src/control-lease-recovery-postgres.test.ts new file mode 100644 index 00000000000..2fa01c27df8 --- /dev/null +++ b/cloud/apps/relay/src/control-lease-recovery-postgres.test.ts @@ -0,0 +1,223 @@ +import { EventEmitter } from 'node:events' +import { ASSIGNMENT_LIMITS, RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract' +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' +import { HostSessionRegistry, type HostSession } from './host-session-registry.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +const sourceCell = { + id: 'control-recovery-source', + url: 'https://control-recovery-source.example.com', + capacityRequests: 100 +} +const targetCell = { + id: 'control-recovery-target', + url: 'https://control-recovery-target.example.com', + capacityRequests: 100 +} +const userId = 'control-recovery-user' +const identities = ['controlrecovery1', 'controlrecovery2'].map((relayHostId) => ({ + userId, + relayHostId +})) + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) +} + +const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn() +} satisfies RelayRuntimeObserver + +type RegistryInternals = { + activate( + socket: WebSocket, + identity: RelayTokenClaims, + existing: HostSession | null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string + ): Promise + heartbeat(session: HostSession): void +} + +describePostgres('expired control lease after a database outage', () => { + let database: RelayDatabase + let now = 1_900_000_000_000 + + const removeFixtureRows = async (): Promise => { + await database.query(`DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, [ + userId + ]) + await database.query(`DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, [userId]) + await database.query(`DELETE FROM relay_assignment_migrations WHERE user_id = ?`, [userId]) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [userId]) + for (const cell of [sourceCell, targetCell]) { + await database.query(`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [cell.id]) + } + } + + const createRegistry = (store: RelayAssignmentStore): HostSessionRegistry => + new HostSessionRegistry( + { + role: 'cell', + cellId: sourceCell.id + } as RelayConfig, + vi.fn(), + {} as RelayCredentialStore, + store, + new ProcessQueuedByteBudget(), + observer, + () => now + ) + + const activate = async ( + registry: HostSessionRegistry, + store: RelayAssignmentStore, + index: number + ): Promise<{ activityId: string; session: HostSession; socket: FakeSocket }> => { + const identity = identities[index]! + const assignment = await store.assign(identity) + const socket = new FakeSocket() + const token = { + sub: identity.userId, + prof: 'control-recovery-profile', + relayHostId: identity.relayHostId, + purpose: 'host-control', + exp: 4_102_444_800 + } satisfies RelayTokenClaims + await (registry as unknown as RegistryInternals).activate( + socket as unknown as WebSocket, + token, + null, + 1, + false, + assignment.assignmentEpoch, + 'control-recovery-test' + ) + const session = registry.get(identity)! + clearInterval(session.heartbeatTimer!) + session.heartbeatTimer = null + return { activityId: session.controlActivityId!, session, socket } + } + + const heartbeat = (registry: HostSessionRegistry, session: HostSession): void => { + session.activityRenewalDueAt = now + session.lastPongAt = now + const internals = registry as unknown as RegistryInternals + internals.heartbeat(session) + } + + const leaseRows = async (relayHostId: string) => + await database.query( + `SELECT activity_id, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [userId, relayHostId] + ) + + const waitForRenewal = async ( + socket: FakeSocket, + relayHostId: string, + expectedRows: number + ): Promise => { + await expect + .poll( + async () => + socket.close.mock.calls.length > 0 || + (await leaseRows(relayHostId)).length === expectedRows + ) + .toBe(true) + } + + beforeAll(async () => { + database = await openRelayDatabase({ databaseUrl, dataDir: '' }) + }) + + beforeEach(async () => { + now = 1_900_000_000_000 + await removeFixtureRows() + }) + + afterEach(async () => { + await removeFixtureRows() + }) + + afterAll(async () => { + if (database) await database.close() + }) + + it('re-acquires a reaped lease while the assignment remains on this cell', async () => { + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([sourceCell]) + const registry = createRegistry(store) + const { activityId, session, socket } = await activate(registry, store, 0) + + now += ASSIGNMENT_LIMITS.activityLeaseMs + 1 + await store.releaseExpiredActivityLeases() + expect(await leaseRows(identities[0]!.relayHostId)).toHaveLength(0) + + heartbeat(registry, session) + await waitForRenewal(socket, identities[0]!.relayHostId, 1) + + expect(socket.close).not.toHaveBeenCalled() + expect(await leaseRows(identities[0]!.relayHostId)).toEqual([ + expect.objectContaining({ activity_id: activityId, cell_id: sourceCell.id }) + ]) + registry.drain(0) + }) + + it('closes without stealing back a lease that moved to another cell', async () => { + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([sourceCell, targetCell]) + const registry = createRegistry(store) + const { activityId, session, socket } = await activate(registry, store, 1) + const identity = identities[1]! + await database.query( + `UPDATE relay_assignment_activity_leases SET cell_id = ? + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [targetCell.id, identity.userId, identity.relayHostId, activityId] + ) + + await expect( + store.renewControlActivity(identity, { + activityId, + cellId: sourceCell.id, + expiresAt: now + ASSIGNMENT_LIMITS.activityLeaseMs + }) + ).rejects.toThrow('control_activity_moved') + + heartbeat(registry, session) + await expect.poll(() => socket.close.mock.calls.length).toBe(1) + + expect(socket.close).toHaveBeenCalledWith(RELAY_CLOSE_CODE.DRAINING, 'control activity moved') + expect(await leaseRows(identity.relayHostId)).toEqual([ + expect.objectContaining({ activity_id: activityId, cell_id: targetCell.id }) + ]) + registry.drain(0) + }) +}) diff --git a/cloud/apps/relay/src/control-renewal-postgres.test.ts b/cloud/apps/relay/src/control-renewal-postgres.test.ts new file mode 100644 index 00000000000..fb49e3af3a2 --- /dev/null +++ b/cloud/apps/relay/src/control-renewal-postgres.test.ts @@ -0,0 +1,349 @@ +import { ASSIGNMENT_LIMITS, RELAY_PROTOCOL_LIMITS } from '@orca-cloud/relay-contract' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { + openRelayDatabase, + type RelayDatabase, + type RelayLockOptions, + type SqlRow +} from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +const sourceCell = { + id: 'control-renewal-source', + url: 'https://control-renewal-source.example.com', + capacityRequests: 100 +} +const targetCell = { + id: 'control-renewal-target', + url: 'https://control-renewal-target.example.com', + capacityRequests: 100 +} +const userId = 'control-renewal-postgres-user' +const identities = Array.from({ length: 6 }, (_, index) => ({ + userId, + relayHostId: `controlrenewal${index + 1}` +})) + +function signal(): { promise: Promise; resolve: () => void } { + let resolve!: () => void + return { promise: new Promise((done) => (resolve = done)), resolve } +} + +class StallFirstTransactionDatabase implements RelayDatabase { + readonly stalled = signal() + readonly continue = signal() + private stallNext = true + + constructor(private readonly database: RelayDatabase) {} + + async query(sql: string, params?: unknown[]): Promise { + return await this.database.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + return await this.database.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + if (this.stallNext) { + this.stallNext = false + this.stalled.resolve() + await this.continue.promise + } + return await this.database.transaction(operation) + } + + async close(): Promise {} +} + +class StallFirstRenewalQueryDatabase implements RelayDatabase { + readonly dialect = 'postgres' as const + readonly stalled = signal() + readonly continue = signal() + private stallNext = true + + constructor(private readonly database: RelayDatabase) {} + + async query(sql: string, params?: unknown[]): Promise { + if (this.stallNext && sql.includes('WITH assignment_state AS MATERIALIZED')) { + this.stallNext = false + this.stalled.resolve() + await this.continue.promise + } + return await this.database.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + return await this.database.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await this.database.transaction(operation) + } + + async close(): Promise {} +} + +class RenewalQueryProbeDatabase implements RelayDatabase { + readonly dialect = 'postgres' as const + renewalQueries = 0 + transactions = 0 + + constructor(private readonly database: RelayDatabase) {} + + async query(sql: string, params?: unknown[]): Promise { + if (sql.includes('WITH assignment_state AS MATERIALIZED')) this.renewalQueries++ + return await this.database.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + return await this.database.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + this.transactions++ + return await this.database.transaction(operation) + } + + async close(): Promise {} +} + +describePostgres('PostgreSQL control renewal', () => { + let database: RelayDatabase + let now = 1_900_000_000_000 + const controlId = (cellId: string): string => `control:${cellId}:1` + + const removeFixtureRows = async (): Promise => { + await database.query(`DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, [ + userId + ]) + await database.query(`DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, [userId]) + await database.query(`DELETE FROM relay_assignment_migrations WHERE user_id = ?`, [userId]) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [userId]) + for (const cell of [sourceCell, targetCell]) { + await database.query(`DELETE FROM relay_cell_connection_snapshots WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_connection_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_connection_limits WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id = ?`, [cell.id]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [cell.id]) + } + } + + beforeAll(async () => { + database = await openRelayDatabase({ databaseUrl, dataDir: '' }) + await removeFixtureRows() + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([sourceCell]) + for (const identity of identities) { + const assignment = await store.assign(identity) + expect(assignment.cellId).toBe(sourceCell.id) + await store.activateControl(identity, { + cellId: sourceCell.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + } + await store.reconcileCells([sourceCell, targetCell]) + }) + + afterAll(async () => { + if (!database) return + await removeFixtureRows() + await database.close() + }) + + it('reaches PostgreSQL past an acquireActivity stalled in the identity queue', async () => { + const probe = new StallFirstTransactionDatabase(database) + const store = new RelayAssignmentStore(probe, () => now) + const identity = identities[0]! + const queued = store.acquireActivity(identity, { + activityId: 'splice:queue-blocker', + kind: 'splice', + cellId: sourceCell.id + }) + await probe.stalled.promise + const expiresAt = now + 105_000 + + try { + await Promise.race([ + store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt + }), + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error('renewal_waited_for_identity_queue')), 2_000) + ) + ]) + const row = ( + await database.query( + `SELECT expires_at FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, controlId(sourceCell.id)] + ) + )[0] + expect(Number(row!.expires_at)).toBe(expiresAt) + } finally { + probe.continue.resolve() + await queued + } + }) + + it('does not let a late older renewal rewind lease or activity timestamps', async () => { + const probe = new StallFirstRenewalQueryDatabase(database) + const store = new RelayAssignmentStore(probe, () => now) + const identity = identities[1]! + const earlierNow = now + const earlierExpiry = earlierNow + 105_000 + const earlier = store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: earlierExpiry + }) + await probe.stalled.promise + + now += 30_000 + const laterNow = now + const laterExpiry = laterNow + 105_000 + await store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: laterExpiry + }) + probe.continue.resolve() + await earlier + + const lease = ( + await database.query( + `SELECT expires_at, updated_at FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, controlId(sourceCell.id)] + ) + )[0] + const assignment = ( + await database.query( + `SELECT lease_expires_at, last_activity_at FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0] + expect(lease).toMatchObject({ + expires_at: String(laterExpiry), + updated_at: String(laterNow) + }) + expect(assignment).toMatchObject({ + lease_expires_at: String(laterExpiry), + last_activity_at: String(laterNow) + }) + }) + + it('allows the source only while its exact forward migration remains active', async () => { + const identity = identities[2]! + const store = new RelayAssignmentStore(database, () => now) + const migration = await store.startEvacuation(identity, targetCell.id) + const activeExpiry = now + 105_000 + + await expect( + store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: activeExpiry + }) + ).resolves.toBeUndefined() + + await database.query( + `UPDATE relay_assignment_migrations SET completed_at = ? + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [now, identity.userId, identity.relayHostId, migration.assignmentEpoch] + ) + now += 30_000 + await expect( + store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: now + 105_000 + }) + ).rejects.toThrow('activity_cell_not_authoritative') + + const lease = ( + await database.query( + `SELECT expires_at FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, controlId(sourceCell.id)] + ) + )[0] + expect(Number(lease!.expires_at)).toBe(activeExpiry) + }) + + it('does not resurrect a control released while its renewal was stalled', async () => { + const probe = new StallFirstRenewalQueryDatabase(database) + const renewalStore = new RelayAssignmentStore(probe, () => now) + const releaseStore = new RelayAssignmentStore(database, () => now) + const identity = identities[3]! + const renewal = renewalStore.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: now + 105_000 + }) + await probe.stalled.promise + + await expect(releaseStore.releaseActivity(identity, controlId(sourceCell.id))).resolves.toBe( + true + ) + probe.continue.resolve() + await expect(renewal).rejects.toThrow('control_activity_not_found') + const rows = await database.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, controlId(sourceCell.id)] + ) + expect(rows).toHaveLength(0) + }) + + it('rejects a renewal expiry beyond the maximum control lease horizon', async () => { + const identity = identities[4]! + const store = new RelayAssignmentStore(database, () => now) + const maximumExpiry = + now + + ASSIGNMENT_LIMITS.activityLeaseMs + + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2 + + await expect( + store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: maximumExpiry + 1 + }) + ).rejects.toThrow('invalid_activity_expiry') + }) + + it('uses one autocommitted PostgreSQL statement for a steady renewal', async () => { + const probe = new RenewalQueryProbeDatabase(database) + const store = new RelayAssignmentStore(probe, () => now) + const identity = identities[5]! + + await store.renewControlActivity(identity, { + activityId: controlId(sourceCell.id), + cellId: sourceCell.id, + expiresAt: now + 105_000 + }) + + expect(probe.renewalQueries).toBe(1) + expect(probe.transactions).toBe(0) + }) +}) diff --git a/cloud/apps/relay/src/control-reservation-reconnect-claim.test.ts b/cloud/apps/relay/src/control-reservation-reconnect-claim.test.ts new file mode 100644 index 00000000000..e6fb44e336c --- /dev/null +++ b/cloud/apps/relay/src/control-reservation-reconnect-claim.test.ts @@ -0,0 +1,201 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayCellConfig } from './config.js' +import { + openInMemoryRelayDatabase, + openRelayDatabase, + type RelayDatabase +} from './database.js' + +const KEY_PREFIX = 'reconnect-claim' +const USER_ID = `${KEY_PREFIX}-user-1` +const RELAY_HOST_ID = 'reconnectclaim01' +const CELL: RelayCellConfig = { + id: `${KEY_PREFIX}-cell`, + url: `https://${KEY_PREFIX}-cell.example.com`, + capacityRequests: 1_000, + connectionHardCap: 600, + connectionUnobservedBound: 50 +} +const CELL_INCARNATION = '11111111-1111-4111-8111-111111111111' +const IDENTITY = { userId: USER_ID, relayHostId: RELAY_HOST_ID } +const CONTROL_ACTIVITY_ID = `control:${CELL.id}:1` + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const backends: { name: string; open: () => Promise }[] = [ + { name: 'sqlite', open: openInMemoryRelayDatabase }, + ...(databaseUrl + ? [ + { + name: 'postgres', + open: () => openRelayDatabase({ databaseUrl, dataDir: '' }) + } + ] + : []) +] + +const databases: RelayDatabase[] = [] + +async function removeScopedRows(database: RelayDatabase): Promise { + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id LIKE '${KEY_PREFIX}-%'` + ) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id LIKE '${KEY_PREFIX}-%'` + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id LIKE '${KEY_PREFIX}-%'`) + for (const table of [ + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_connection_limits', + 'relay_cell_runtime', + 'relay_cells' + ]) { + await database.query(`DELETE FROM ${table} WHERE cell_id = ?`, [CELL.id]) + } +} + +afterEach(async () => { + for (const database of databases.splice(0)) { + await removeScopedRows(database) + await database.close() + } +}) + +async function setup( + open: () => Promise, + now: () => number +): Promise<{ database: RelayDatabase; store: RelayAssignmentStore }> { + const database = await open() + databases.push(database) + await removeScopedRows(database) + const store = new RelayAssignmentStore(database, now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells([CELL], false) + await heartbeat(store, 0) + return { database, store } +} + +async function heartbeat(store: RelayAssignmentStore, watermark: number): Promise { + await store.recordCellHeartbeat({ + cellId: CELL.id, + cellUrl: CELL.url, + cellIncarnation: CELL_INCARNATION, + startedAt: 50, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionHardCap: CELL.connectionHardCap, + connectionUnobservedBound: CELL.connectionUnobservedBound, + connectionInclusionWatermark: watermark + }) +} + +async function reservations( + database: RelayDatabase +): Promise<{ state: string; claimActivityId: string | null }[]> { + const rows = await database.query( + `SELECT state, claim_activity_id FROM relay_control_connection_reservations + WHERE user_id = ? ORDER BY created_at ASC, reservation_id ASC`, + [USER_ID] + ) + return rows.map((row) => ({ + state: String(row['state']), + claimActivityId: + row['claim_activity_id'] === null || row['claim_activity_id'] === undefined + ? null + : String(row['claim_activity_id']) + })) +} + +describe.each(backends)('control reservation claim on reconnect ($name)', ({ open }) => { + it('claims the fresh reservation when the same generation reconnects', async () => { + let now = 100_000_000 + const { database, store } = await setup(open, () => now) + + const assignment = await store.assign(IDENTITY) + await store.activateControl(IDENTITY, { + cellId: CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 1 + }) + // Cell telemetry now includes the control, so its reservation is released. + now += 1_000 + await heartbeat(store, 2) + expect(await reservations(database)).toEqual([ + { state: 'released', claimActivityId: CONTROL_ACTIVITY_ID } + ]) + + // Control drops and the host is re-granted the same cell and epoch. + now += 1_000 + await store.releaseActivity(IDENTITY, CONTROL_ACTIVITY_ID) + now += 1_000 + await store.assign(IDENTITY) + expect(await reservations(database)).toEqual([ + { state: 'released', claimActivityId: CONTROL_ACTIVITY_ID }, + { state: 'reserved', claimActivityId: null } + ]) + + now += 1_000 + await store.activateControl(IDENTITY, { + cellId: CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 3 + }) + + expect(await reservations(database)).toEqual([ + { state: 'released', claimActivityId: CONTROL_ACTIVITY_ID }, + { state: 'claimed', claimActivityId: CONTROL_ACTIVITY_ID } + ]) + }) + + it('still refuses to claim a second reservation while the first is outstanding', async () => { + let now = 100_000_000 + const { database, store } = await setup(open, () => now) + + const assignment = await store.assign(IDENTITY) + await store.activateControl(IDENTITY, { + cellId: CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 1 + }) + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, assignment_epoch, + cell_id, state, claim_activity_id, created_at, timeout_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, 'reserved', NULL, ?, ?, ?)`, + [ + `${KEY_PREFIX}-extra`, + `${KEY_PREFIX}-extra`, + USER_ID, + RELAY_HOST_ID, + assignment.assignmentEpoch, + CELL.id, + now + 1, + now + 90_000, + now + 1 + ] + ) + + now += 1_000 + await store.activateControl(IDENTITY, { + cellId: CELL.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1, + connectionInclusionWatermark: 2 + }) + + expect(await reservations(database)).toEqual([ + { state: 'claimed', claimActivityId: CONTROL_ACTIVITY_ID }, + { state: 'reserved', claimActivityId: null } + ]) + }) +}) diff --git a/cloud/apps/relay/src/credential-store.test.ts b/cloud/apps/relay/src/credential-store.test.ts new file mode 100644 index 00000000000..9f279b93791 --- /dev/null +++ b/cloud/apps/relay/src/credential-store.test.ts @@ -0,0 +1,301 @@ +import { describe, expect, it } from 'vitest' +import { + hashCredential, + RelayCredentialStore, + type CredentialReservation, + type RelayIdentity +} from './credential-store.js' +import { openInMemoryRelayDatabase, type RelayDatabase } from './database.js' + +const identity: RelayIdentity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } +const relayDeviceId = 'device-1' + +async function inviteBasis( + store: RelayCredentialStore, + generation = 1 +): Promise<{ reservation: CredentialReservation; basisConnId: string }> { + const invite = await store.createInvite(identity, relayDeviceId) + const reservation = await store.reserveCredential(identity.relayHostId, invite.inviteToken) + if (!reservation) throw new Error('invite did not reserve') + const basisConnId = `basis-${generation}` + await store.recordConnectionBasis({ + ...reservation, + basisConnId, + owningControlGeneration: generation, + deadline: reservation.leaseExpiresAt + }) + return { reservation, basisConnId } +} + +describe('relay credential store', () => { + it('issues invites with a skew margin under the client-side TTL ceiling', async () => { + const database = await openInMemoryRelayDatabase() + const store = new RelayCredentialStore(database, () => 1_000_000) + const invite = await store.createInvite(identity, relayDeviceId) + // Zero-tolerance released desktops reject expiry past local now + 10min; + // issuing 30s under the ceiling absorbs that much desktop clock lag. + expect(invite.expiresAt).toBe(1_000_000 + 10 * 60 * 1_000 - 30 * 1_000) + }) + + it('persists one invite reservation with bounded attempts and cooldown', async () => { + let now = 100 + const database = await openInMemoryRelayDatabase() + const store = new RelayCredentialStore(database, () => now) + const invite = await store.createInvite(identity, relayDeviceId) + const first = await store.reserveCredential(identity.relayHostId, invite.inviteToken, 'first') + expect(first).toMatchObject({ credentialKind: 'invite', reservationId: 'first' }) + expect(await store.reserveCredential(identity.relayHostId, invite.inviteToken, 'second')).toBeNull() + now = first!.leaseExpiresAt + expect(await store.reserveCredential(identity.relayHostId, invite.inviteToken, 'second')).toBeNull() + await database.query(`UPDATE relay_invites SET cooldown_until = ? WHERE token_hash = ?`, [ + now, + hashCredential(invite.inviteToken) + ]) + expect(await store.reserveCredential(identity.relayHostId, invite.inviteToken, 'second')).toMatchObject({ + reservationId: 'second' + }) + await database.close() + }) + + it('validates director moves without reservation and enforces account-global mint rate', async () => { + const database = await openInMemoryRelayDatabase() + const store = new RelayCredentialStore(database, () => 100) + const invite = await store.createInvite(identity, relayDeviceId) + expect(await store.validateInviteForMove(identity.relayHostId, invite.inviteToken)).toBe(true) + const rows = await database.query(`SELECT state, attempt_count FROM relay_invites WHERE token_hash = ?`, [ + hashCredential(invite.inviteToken) + ]) + expect(rows[0]).toMatchObject({ state: 'available', attempt_count: 0 }) + for (let index = 1; index < 30; index++) { + await store.createInvite(identity, `device-${index + 1}`) + } + await expect(store.createInvite(identity, 'device-over-limit')).rejects.toMatchObject({ + code: 'rate_limit_exceeded' + }) + expect(await database.query(`SELECT * FROM relay_audit_events`)).toHaveLength(30) + await database.close() + }) + + it('serializes relay-basis and authenticated-direct under one global result', async () => { + const database = await openInMemoryRelayDatabase() + const store = new RelayCredentialStore(database, () => 100) + const { basisConnId } = await inviteBasis(store) + await store.recordDirectAuthorization({ + ...identity, + relayDeviceId, + directAuthId: 'direct-1', + owningControlGeneration: 1, + deadline: 1_000 + }) + const base = { + ...identity, + relayDeviceId, + reqId: 'req-1', + newResumeTokenHash: hashCredential('pending-resume'), + owningControlGeneration: 1 + } + const [relay, direct] = await Promise.all([ + store.installCredential({ + ...base, + authorization: { mode: 'relay-basis' as const, basisConnId } + }), + store.installCredential({ + ...base, + authorization: { mode: 'authenticated-direct' as const, directAuthId: 'direct-1' } + }) + ]) + expect(relay).toEqual(direct) + expect(relay.currentVersion).toBe(1) + expect(await store.installStatus(base)).toEqual(relay) + const devices = await database.query(`SELECT * FROM relay_devices`) + expect(devices).toHaveLength(1) + await database.close() + }) + + it('rolls back token and invite effects when result persistence fails', async () => { + const database = await openInMemoryRelayDatabase() + const normalStore = new RelayCredentialStore(database, () => 100) + const { basisConnId, reservation } = await inviteBasis(normalStore) + const fault: RelayDatabase = { + query: (sql, params) => database.query(sql, params), + queryLocked: (sql, params) => database.queryLocked(sql, params), + close: () => database.close(), + transaction: async (operation) => + await database.transaction(async (transaction) => + await operation({ + query: async (sql, params) => { + if (sql.includes('INSERT INTO relay_install_results')) throw new Error('injected SQL failure') + return await transaction.query(sql, params) + }, + queryLocked: (sql, params) => transaction.queryLocked(sql, params), + transaction: (nested) => transaction.transaction(nested), + close: () => transaction.close() + }) + ) + } + const faultStore = new RelayCredentialStore(fault, () => 100) + const input = { + ...identity, + relayDeviceId, + reqId: 'req-fault', + newResumeTokenHash: hashCredential('pending-resume'), + owningControlGeneration: 1, + authorization: { mode: 'relay-basis' as const, basisConnId } + } + await expect(faultStore.installCredential(input)).rejects.toThrow('injected SQL failure') + expect(await normalStore.installStatus(input)).toBeNull() + expect(await database.query(`SELECT * FROM relay_devices`)).toEqual([]) + const invites = await database.query(`SELECT state FROM relay_invites WHERE token_hash = ?`, [ + reservation.tokenHash + ]) + expect(invites[0]?.state).toBe('reserved') + await database.close() + }) + + it('renews only a tuple-bound current credential and replays its committed result', async () => { + let now = 100 + const database = await openInMemoryRelayDatabase() + const store = new RelayCredentialStore(database, () => now) + const { basisConnId } = await inviteBasis(store) + const resumeToken = 'resume-token-1' + await store.installCredential({ + ...identity, + relayDeviceId, + reqId: 'install-1', + newResumeTokenHash: hashCredential(resumeToken), + owningControlGeneration: 1, + authorization: { mode: 'relay-basis', basisConnId } + }) + const reservation = await store.reserveCredential(identity.relayHostId, resumeToken) + if (!reservation) throw new Error('resume did not reserve') + expect(await store.resolveResume(identity.relayHostId, resumeToken)).toEqual({ + userId: identity.userId, + relayDeviceId + }) + await store.recordConnectionBasis({ + ...reservation, + basisConnId: 'resume-basis', + owningControlGeneration: 1, + deadline: 1_000 + }) + now = 200 + const confirmed = await store.confirmResume({ + ...identity, + reqId: 'confirm-1', + basisConnId: 'resume-basis', + owningControlGeneration: 1 + }) + expect(confirmed).toMatchObject({ renewed: true, acceptedAs: 'current' }) + await store.deactivateBasis('resume-basis') + expect( + await store.confirmResume({ + ...identity, + reqId: 'confirm-1', + basisConnId: 'resume-basis', + owningControlGeneration: 1 + }) + ).toEqual(confirmed) + await expect( + store.confirmResume({ + ...identity, + reqId: 'confirm-1', + basisConnId: 'other-basis', + owningControlGeneration: 1 + }) + ).rejects.toMatchObject({ code: 'confirmation_tuple_mismatch' }) + await database.close() + }) + + it('leaves a now-grace version unchanged and rejects late, expired, and revoked tuples', async () => { + let now = 100 + const database = await openInMemoryRelayDatabase() + const store = new RelayCredentialStore(database, () => now) + const { basisConnId } = await inviteBasis(store) + const firstToken = 'resume-token-1' + const first = await store.installCredential({ + ...identity, + relayDeviceId, + reqId: 'install-1', + newResumeTokenHash: hashCredential(firstToken), + owningControlGeneration: 1, + authorization: { mode: 'relay-basis', basisConnId } + }) + const firstReservation = await store.reserveCredential(identity.relayHostId, firstToken) + if (!firstReservation) throw new Error('first resume did not reserve') + await store.recordConnectionBasis({ + ...firstReservation, + basisConnId: 'grace-basis', + owningControlGeneration: 1, + deadline: 100_000_000 + }) + await store.recordDirectAuthorization({ + ...identity, + relayDeviceId, + directAuthId: 'rotate-direct', + owningControlGeneration: 1, + deadline: 1_000 + }) + now = 200 + await store.installCredential({ + ...identity, + relayDeviceId, + reqId: 'install-2', + newResumeTokenHash: hashCredential('resume-token-2'), + expectedCurrentHash: hashCredential(firstToken), + owningControlGeneration: 1, + authorization: { mode: 'authenticated-direct', directAuthId: 'rotate-direct' } + }) + const grace = await store.confirmResume({ + ...identity, + reqId: 'confirm-grace', + basisConnId: 'grace-basis', + owningControlGeneration: 1 + }) + expect(grace).toMatchObject({ acceptedAs: 'current', renewed: false, currentVersion: 2 }) + expect(grace.resumeExpiresAt).toBeGreaterThan(first.resumeExpiresAt) + + now += 24 * 60 * 60 * 1000 + 1 + await expect( + store.confirmResume({ + ...identity, + reqId: 'confirm-expired-grace', + basisConnId: 'grace-basis', + owningControlGeneration: 1 + }) + ).rejects.toMatchObject({ code: 'reject-expired' }) + + const currentReservation = await store.reserveCredential(identity.relayHostId, 'resume-token-2') + if (!currentReservation) throw new Error('current resume did not reserve') + await store.recordConnectionBasis({ + ...currentReservation, + basisConnId: 'revoked-basis', + owningControlGeneration: 1, + deadline: now + 1_000 + }) + await store.revoke(identity, relayDeviceId) + await expect( + store.confirmResume({ + ...identity, + reqId: 'confirm-revoked', + basisConnId: 'revoked-basis', + owningControlGeneration: 1 + }) + ).rejects.toMatchObject({ code: 'reject-revoked' }) + + await store.recordConnectionBasis({ + ...currentReservation, + basisConnId: 'late-basis', + owningControlGeneration: 1, + deadline: now - 1 + }) + await expect( + store.confirmResume({ + ...identity, + reqId: 'confirm-late', + basisConnId: 'late-basis', + owningControlGeneration: 1 + }) + ).rejects.toMatchObject({ code: 'confirmation_not_active' }) + await database.close() + }) +}) diff --git a/cloud/apps/relay/src/credential-store.ts b/cloud/apps/relay/src/credential-store.ts new file mode 100644 index 00000000000..a5697e8c699 --- /dev/null +++ b/cloud/apps/relay/src/credential-store.ts @@ -0,0 +1,745 @@ +import { createHash, randomBytes, randomUUID, timingSafeEqual } from 'node:crypto' +import { + decideResumeCommit, + RELAY_PROTOCOL_LIMITS, + type DeviceCredentialInstalled, + type DeviceResumeConfirmed +} from '@orca-cloud/relay-contract' +import type { RelayDatabase, SqlRow } from './database.js' + +const CREDENTIAL_GRACE_MS = 24 * 60 * 60 * 1000 +// Released desktops validate invite expiry against their own clock with zero +// tolerance at exactly inviteTtlMs; issuing under the ceiling keeps pairing +// working for clients whose clocks trail the cell by up to this margin. +const INVITE_ISSUE_SKEW_MARGIN_MS = 30 * 1000 + +export type RelayIdentity = { userId: string; relayHostId: string } +export type CredentialReservation = RelayIdentity & { + credentialKind: 'invite' | 'resume' + relayDeviceId: string + tokenHash: string + reservationId: string + leaseExpiresAt: number + acceptedCredentialVersion?: number + acceptedAs?: 'current' | 'grace' + resumeExpiresAt?: number + graceExpiresAt?: number +} + +export type InstallInput = RelayIdentity & { + relayDeviceId: string + reqId: string + newResumeTokenHash: string + expectedCurrentHash?: string + owningControlGeneration: number + authorization: + | { mode: 'relay-basis'; basisConnId: string } + | { mode: 'authenticated-direct'; directAuthId: string } +} + +export class RelayStoreError extends Error { + constructor(readonly code: string) { + super(code) + } +} + +export function hashCredential(token: string): string { + return createHash('sha256').update(token).digest('base64url') +} + +function equalHash(left: string, right: string): boolean { + const leftBytes = Buffer.from(left) + const rightBytes = Buffer.from(right) + return leftBytes.length === rightBytes.length && timingSafeEqual(leftBytes, rightBytes) +} + +function number(row: SqlRow, field: string): number { + const value = Number(row[field]) + if (!Number.isSafeInteger(value)) throw new RelayStoreError(`invalid_${field}`) + return value +} + +function optionalNumber(row: SqlRow, field: string): number | undefined { + return row[field] === null || row[field] === undefined ? undefined : number(row, field) +} + +function string(row: SqlRow, field: string): string { + const value = row[field] + if (typeof value !== 'string') throw new RelayStoreError(`invalid_${field}`) + return value +} + +export class RelayCredentialStore { + constructor( + private readonly database: RelayDatabase, + private readonly now: () => number = Date.now + ) {} + + async createInvite(identity: RelayIdentity, relayDeviceId: string): Promise<{ + inviteToken: string + expiresAt: number + maxAttempts: number + }> { + const inviteToken = randomBytes(32).toString('base64url') + const tokenHash = hashCredential(inviteToken) + const now = this.now() + const expiresAt = now + RELAY_PROTOCOL_LIMITS.inviteTtlMs - INVITE_ISSUE_SKEW_MARGIN_MS + await this.database.transaction(async (transaction) => { + await this.consumeRateWith(transaction, `account:${identity.userId}`, 'invite-mint', 30, 60_000, now) + await transaction.query( + `UPDATE relay_invites SET state = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ? + AND state IN (?, ?, ?)`, + [ + 'invalidated', now, identity.userId, identity.relayHostId, relayDeviceId, + 'available', 'reserved', 'cooldown' + ] + ) + await this.auditWith(transaction, { + type: 'invite-created', + ...identity, + relayDeviceId, + detail: { expiresAt } + }) + await transaction.query( + `INSERT INTO relay_invites + (user_id, relay_host_id, relay_device_id, token_hash, state, attempt_count, + max_attempts, expires_at, created_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, identity.relayHostId, relayDeviceId, tokenHash, 'available', 0, + RELAY_PROTOCOL_LIMITS.inviteMaxAttempts, expiresAt, now, now + ] + ) + }) + return { inviteToken, expiresAt, maxAttempts: RELAY_PROTOCOL_LIMITS.inviteMaxAttempts } + } + + async reserveCredential( + relayHostId: string, + token: string, + reservationId: string = randomUUID() + ): Promise { + const tokenHash = hashCredential(token) + const now = this.now() + return await this.database.transaction(async (transaction) => { + const inviteRows = await transaction.query( + `SELECT * FROM relay_invites WHERE relay_host_id = ? AND token_hash = ?`, + [relayHostId, tokenHash] + ) + const invite = inviteRows[0] + if (invite && equalHash(string(invite, 'token_hash'), tokenHash)) { + const expiresAt = number(invite, 'expires_at') + const attempts = number(invite, 'attempt_count') + const maxAttempts = number(invite, 'max_attempts') + let state = string(invite, 'state') + const reservationExpiresAt = optionalNumber(invite, 'reservation_expires_at') + if (expiresAt <= now) state = 'expired' + else if (state === 'reserved' && (reservationExpiresAt ?? 0) <= now) { + await transaction.query( + `UPDATE relay_invites SET state = ?, reservation_id = NULL, + reservation_expires_at = NULL, cooldown_until = ?, updated_at = ? + WHERE token_hash = ?`, + [ + 'cooldown', + now + RELAY_PROTOCOL_LIMITS.inviteAttemptCooldownMs, + now, + tokenHash + ] + ) + return null + } + const cooldownUntil = optionalNumber(invite, 'cooldown_until') ?? 0 + if ( + (state !== 'available' && !(state === 'cooldown' && cooldownUntil <= now)) || + attempts >= maxAttempts + ) { + return null + } + const leaseExpiresAt = Math.min( + expiresAt, + now + RELAY_PROTOCOL_LIMITS.inviteReservationLeaseMs + ) + await transaction.query( + `UPDATE relay_invites SET state = ?, attempt_count = ?, reservation_id = ?, + reservation_expires_at = ?, cooldown_until = NULL, updated_at = ? + WHERE token_hash = ?`, + ['reserved', attempts + 1, reservationId, leaseExpiresAt, now, tokenHash] + ) + return { + userId: string(invite, 'user_id'), + relayHostId, + relayDeviceId: string(invite, 'relay_device_id'), + credentialKind: 'invite', + tokenHash, + reservationId, + leaseExpiresAt + } + } + + const deviceRows = await transaction.query( + `SELECT * FROM relay_devices + WHERE relay_host_id = ? AND (current_hash = ? OR grace_hash = ?)`, + [relayHostId, tokenHash, tokenHash] + ) + const device = deviceRows[0] + if (!device || optionalNumber(device, 'revoked_at') !== undefined) return null + const currentHash = string(device, 'current_hash') + const currentExpiresAt = number(device, 'current_expires_at') + const graceHash = device.grace_hash + const graceExpiresAt = optionalNumber(device, 'grace_expires_at') + let acceptedAs: 'current' | 'grace' + let acceptedCredentialVersion: number + if (equalHash(currentHash, tokenHash) && currentExpiresAt > now) { + acceptedAs = 'current' + acceptedCredentialVersion = number(device, 'current_version') + } else if ( + typeof graceHash === 'string' && + equalHash(graceHash, tokenHash) && + (graceExpiresAt ?? 0) > now + ) { + acceptedAs = 'grace' + acceptedCredentialVersion = number(device, 'grace_version') + } else { + return null + } + return { + userId: string(device, 'user_id'), + relayHostId, + relayDeviceId: string(device, 'relay_device_id'), + credentialKind: 'resume', + tokenHash, + reservationId, + leaseExpiresAt: now + RELAY_PROTOCOL_LIMITS.hostAttachDeadlineMs, + acceptedCredentialVersion, + acceptedAs, + resumeExpiresAt: currentExpiresAt, + graceExpiresAt + } + }) + } + + async recordConnectionBasis(input: CredentialReservation & { + basisConnId: string + owningControlGeneration: number + deadline: number + }): Promise { + const now = this.now() + await this.database.query( + `INSERT INTO relay_connection_bases + (basis_conn_id, user_id, relay_host_id, relay_device_id, owning_control_generation, + credential_kind, invite_token_hash, accepted_credential_version, accepted_as, + deadline, active, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + input.basisConnId, input.userId, input.relayHostId, input.relayDeviceId, + input.owningControlGeneration, input.credentialKind, + input.credentialKind === 'invite' ? input.tokenHash : null, + input.acceptedCredentialVersion, input.acceptedAs, input.deadline, 1, now + ] + ) + } + + async failReservation(reservation: CredentialReservation): Promise { + if (reservation.credentialKind !== 'invite') return + const now = this.now() + await this.database.transaction(async (transaction) => { + const rows = await transaction.query( + `SELECT attempt_count, max_attempts FROM relay_invites + WHERE token_hash = ? AND reservation_id = ? AND state = ?`, + [reservation.tokenHash, reservation.reservationId, 'reserved'] + ) + const invite = rows[0] + if (!invite) return + const exhausted = number(invite, 'attempt_count') >= number(invite, 'max_attempts') + await transaction.query( + `UPDATE relay_invites SET state = ?, reservation_id = NULL, + reservation_expires_at = NULL, cooldown_until = ?, updated_at = ? + WHERE token_hash = ? AND reservation_id = ?`, + [ + exhausted ? 'invalidated' : 'cooldown', + exhausted ? null : now + RELAY_PROTOCOL_LIMITS.inviteAttemptCooldownMs, + now, + reservation.tokenHash, + reservation.reservationId + ] + ) + }) + } + + async recordDirectAuthorization(input: RelayIdentity & { + relayDeviceId: string + directAuthId: string + owningControlGeneration: number + deadline: number + }): Promise { + await this.database.query( + `INSERT INTO relay_direct_authorizations + (direct_auth_id, user_id, relay_host_id, relay_device_id, + owning_control_generation, deadline) + VALUES (?, ?, ?, ?, ?, ?)`, + [ + input.directAuthId, input.userId, input.relayHostId, input.relayDeviceId, + input.owningControlGeneration, input.deadline + ] + ) + } + + async installCredential(input: InstallInput): Promise { + let lastError: unknown + for (let attempt = 0; attempt <= 3; attempt++) { + try { + return await this.installCredentialOnce(input) + } catch (error) { + if (error instanceof RelayStoreError) throw error + const committed = await this.installStatus(input) + if (committed) return committed + const code = (error as { code?: unknown }).code + const retryable = ['23505', '40P01', '55P03', '40001'].includes(String(code)) + if (!retryable || attempt === 3) throw error + lastError = error + } + } + throw lastError + } + + private async installCredentialOnce(input: InstallInput): Promise { + return await this.database.transaction(async (transaction) => { + const existing = await this.installStatusWith(transaction, input) + if (existing) return existing + const now = this.now() + const basisInviteHash = await this.validateInstallAuthorization(transaction, input, now) + const devices = await transaction.query( + `SELECT * FROM relay_devices + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ?`, + [input.userId, input.relayHostId, input.relayDeviceId] + ) + const current = devices[0] + if (input.expectedCurrentHash) { + if (!current || !equalHash(string(current, 'current_hash'), input.expectedCurrentHash)) { + throw new RelayStoreError('current_hash_mismatch') + } + } + const currentVersion = current ? number(current, 'current_version') + 1 : 1 + const resumeExpiresAt = now + RELAY_PROTOCOL_LIMITS.resumeTtlMs + const graceExpiresAt = current ? now + CREDENTIAL_GRACE_MS : undefined + await transaction.query( + `INSERT INTO relay_devices + (user_id, relay_host_id, relay_device_id, current_hash, current_version, + current_expires_at, grace_hash, grace_version, grace_expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (user_id, relay_host_id, relay_device_id) DO UPDATE SET + current_hash = excluded.current_hash, + current_version = excluded.current_version, + current_expires_at = excluded.current_expires_at, + grace_hash = excluded.grace_hash, + grace_version = excluded.grace_version, + grace_expires_at = excluded.grace_expires_at, + revoked_at = NULL, + updated_at = excluded.updated_at`, + [ + input.userId, input.relayHostId, input.relayDeviceId, input.newResumeTokenHash, + currentVersion, resumeExpiresAt, current ? string(current, 'current_hash') : null, + current ? number(current, 'current_version') : null, graceExpiresAt, now + ] + ) + if (basisInviteHash) { + await transaction.query( + `UPDATE relay_invites SET state = ?, reservation_id = NULL, + reservation_expires_at = NULL, updated_at = ? WHERE token_hash = ?`, + ['consumed', now, basisInviteHash] + ) + } else { + await transaction.query( + `UPDATE relay_invites SET state = ?, reservation_id = NULL, + reservation_expires_at = NULL, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ? + AND state IN (?, ?, ?)`, + [ + 'invalidated', now, input.userId, input.relayHostId, input.relayDeviceId, + 'available', 'reserved', 'cooldown' + ] + ) + } + await transaction.query( + `UPDATE relay_direct_authorizations SET consumed_at = ? + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ? + AND consumed_at IS NULL`, + [now, input.userId, input.relayHostId, input.relayDeviceId] + ) + const result: DeviceCredentialInstalled = { + v: 1, + reqId: input.reqId, + authorizationMode: input.authorization.mode, + currentVersion, + resumeExpiresAt, + ...(graceExpiresAt === undefined ? {} : { graceExpiresAt }) + } + await transaction.query( + `INSERT INTO relay_install_results + (user_id, relay_host_id, relay_device_id, req_id, authorization_mode, + result_json, committed_at) VALUES (?, ?, ?, ?, ?, ?, ?)`, + [ + input.userId, input.relayHostId, input.relayDeviceId, input.reqId, + input.authorization.mode, JSON.stringify(result), now + ] + ) + await this.auditWith(transaction, { + type: 'credential-installed', + userId: input.userId, + relayHostId: input.relayHostId, + relayDeviceId: input.relayDeviceId, + detail: { reqId: input.reqId, authorizationMode: input.authorization.mode, currentVersion } + }) + return result + }) + } + + async installStatus(input: RelayIdentity & { + relayDeviceId: string + reqId: string + }): Promise { + return await this.installStatusWith(this.database, input) + } + + async confirmResume(input: RelayIdentity & { + reqId: string + basisConnId: string + owningControlGeneration: number + }): Promise { + return await this.database.transaction(async (transaction) => { + const prior = await transaction.query( + `SELECT basis_conn_id, result_json FROM relay_confirm_results + WHERE user_id = ? AND relay_host_id = ? AND req_id = ?`, + [input.userId, input.relayHostId, input.reqId] + ) + if (prior[0]) { + if (string(prior[0], 'basis_conn_id') !== input.basisConnId) { + throw new RelayStoreError('confirmation_tuple_mismatch') + } + return JSON.parse(string(prior[0], 'result_json')) as DeviceResumeConfirmed + } + const rows = await transaction.query( + `SELECT * FROM relay_connection_bases WHERE basis_conn_id = ?`, + [input.basisConnId] + ) + const basis = rows[0] + const now = this.now() + if ( + !basis || + string(basis, 'user_id') !== input.userId || + string(basis, 'relay_host_id') !== input.relayHostId || + string(basis, 'credential_kind') !== 'resume' || + number(basis, 'owning_control_generation') !== input.owningControlGeneration || + number(basis, 'active') !== 1 || + number(basis, 'deadline') < now + ) { + throw new RelayStoreError('confirmation_not_active') + } + const relayDeviceId = string(basis, 'relay_device_id') + await transaction.query( + `UPDATE relay_devices SET updated_at = updated_at + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ?`, + [input.userId, input.relayHostId, relayDeviceId] + ) + const devices = await transaction.query( + `SELECT * FROM relay_devices + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ?`, + [input.userId, input.relayHostId, relayDeviceId] + ) + const device = devices[0] + if (!device) throw new RelayStoreError('credential_not_found') + const acceptedVersion = number(basis, 'accepted_credential_version') + const decision = decideResumeCommit( + { + currentVersion: number(device, 'current_version'), + currentHash: string(device, 'current_hash'), + currentExpiresAt: number(device, 'current_expires_at'), + graceVersion: optionalNumber(device, 'grace_version'), + graceHash: typeof device.grace_hash === 'string' ? device.grace_hash : undefined, + graceExpiresAt: optionalNumber(device, 'grace_expires_at'), + revokedAt: optionalNumber(device, 'revoked_at') + }, + acceptedVersion, + now + ) + if (decision.startsWith('reject-')) throw new RelayStoreError(decision) + const renewed = decision === 'renew-current' + const resumeExpiresAt = renewed + ? now + RELAY_PROTOCOL_LIMITS.resumeTtlMs + : number(device, 'current_expires_at') + if (renewed) { + await transaction.query( + `UPDATE relay_devices SET current_expires_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ?`, + [resumeExpiresAt, now, input.userId, input.relayHostId, relayDeviceId] + ) + } + const result: DeviceResumeConfirmed = { + v: 1, + reqId: input.reqId, + currentVersion: number(device, 'current_version'), + acceptedAs: string(basis, 'accepted_as') as 'current' | 'grace', + renewed, + resumeExpiresAt, + ...(optionalNumber(device, 'grace_expires_at') === undefined + ? {} + : { graceExpiresAt: optionalNumber(device, 'grace_expires_at') }) + } + await transaction.query( + `INSERT INTO relay_confirm_results + (user_id, relay_host_id, req_id, basis_conn_id, tuple_json, result_json, committed_at) + VALUES (?, ?, ?, ?, ?, ?, ?)`, + [ + input.userId, input.relayHostId, input.reqId, input.basisConnId, + JSON.stringify({ + relayDeviceId, + acceptedCredentialVersion: acceptedVersion, + acceptedAs: result.acceptedAs, + confirmDeadline: number(basis, 'deadline'), + owningControlGeneration: input.owningControlGeneration + }), + JSON.stringify(result), + now + ] + ) + await this.auditWith(transaction, { + type: 'resume-confirmed', + userId: input.userId, + relayHostId: input.relayHostId, + relayDeviceId, + detail: { reqId: input.reqId, basisConnId: input.basisConnId, renewed } + }) + return result + }) + } + + async deactivateBasis(basisConnId: string): Promise { + await this.database.query(`UPDATE relay_connection_bases SET active = ? WHERE basis_conn_id = ?`, [ + 0, + basisConnId + ]) + } + + async revoke(identity: RelayIdentity, relayDeviceId: string): Promise { + await this.database.transaction(async (transaction) => { + const now = this.now() + await transaction.query( + `UPDATE relay_devices SET revoked_at = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ?`, + [now, now, identity.userId, identity.relayHostId, relayDeviceId] + ) + await transaction.query( + `UPDATE relay_invites SET state = ?, updated_at = ? + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ?`, + ['invalidated', now, identity.userId, identity.relayHostId, relayDeviceId] + ) + await this.auditWith(transaction, { + type: 'device-revoked', + ...identity, + relayDeviceId, + detail: {} + }) + }) + } + + async resolveResume( + relayHostId: string, + token: string + ): Promise<{ userId: string; relayDeviceId: string } | null> { + const tokenHash = hashCredential(token) + const rows = await this.database.query( + `SELECT * FROM relay_devices + WHERE relay_host_id = ? AND (current_hash = ? OR grace_hash = ?)`, + [relayHostId, tokenHash, tokenHash] + ) + const device = rows[0] + if (!device || optionalNumber(device, 'revoked_at') !== undefined) return null + const now = this.now() + const currentValid = + equalHash(string(device, 'current_hash'), tokenHash) && + number(device, 'current_expires_at') > now + const graceHash = device.grace_hash + const graceValid = + typeof graceHash === 'string' && + equalHash(graceHash, tokenHash) && + (optionalNumber(device, 'grace_expires_at') ?? 0) > now + return currentValid || graceValid + ? { userId: string(device, 'user_id'), relayDeviceId: string(device, 'relay_device_id') } + : null + } + + async validateInviteForMove(relayHostId: string, token: string): Promise { + return Boolean(await this.resolveInviteForMove(relayHostId, token)) + } + + async resolveInviteForMove( + relayHostId: string, + token: string + ): Promise<{ userId: string; relayDeviceId: string } | null> { + const tokenHash = hashCredential(token) + const rows = await this.database.query( + `SELECT user_id, relay_device_id, token_hash, state, attempt_count, max_attempts, expires_at + FROM relay_invites WHERE relay_host_id = ? AND token_hash = ?`, + [relayHostId, tokenHash] + ) + const invite = rows[0] + const valid = Boolean( + invite && + equalHash(string(invite, 'token_hash'), tokenHash) && + ['available', 'reserved', 'cooldown'].includes(string(invite, 'state')) && + number(invite, 'attempt_count') < number(invite, 'max_attempts') && + number(invite, 'expires_at') > this.now() + ) + return valid && invite + ? { userId: string(invite, 'user_id'), relayDeviceId: string(invite, 'relay_device_id') } + : null + } + + async consumeRate(input: { + scopeKey: string + kind: string + limit: number + windowMs: number + }): Promise { + await this.database.transaction(async (transaction) => { + await this.consumeRateWith( + transaction, + input.scopeKey, + input.kind, + input.limit, + input.windowMs, + this.now() + ) + }) + } + + async cleanup(): Promise { + const now = this.now() + await this.database.transaction(async (transaction) => { + await transaction.query( + `UPDATE relay_invites SET state = ?, reservation_id = NULL, + reservation_expires_at = NULL, updated_at = ? + WHERE expires_at <= ? AND state IN (?, ?, ?)`, + ['expired', now, now, 'available', 'reserved', 'cooldown'] + ) + await transaction.query( + `UPDATE relay_invites SET state = ?, reservation_id = NULL, + reservation_expires_at = NULL, cooldown_until = ?, updated_at = ? + WHERE state = ? AND reservation_expires_at <= ? AND expires_at > ?`, + ['cooldown', now + RELAY_PROTOCOL_LIMITS.inviteAttemptCooldownMs, now, 'reserved', now, now] + ) + await transaction.query( + `UPDATE relay_connection_bases SET active = ? WHERE active = ? AND deadline <= ?`, + [0, 1, now] + ) + await transaction.query( + `UPDATE relay_direct_authorizations SET consumed_at = ? + WHERE consumed_at IS NULL AND deadline <= ?`, + [now, now] + ) + await transaction.query( + `DELETE FROM relay_rate_windows WHERE window_started_at < ?`, + [now - 24 * 60 * 60 * 1000] + ) + }) + } + + private async installStatusWith( + database: RelayDatabase, + input: RelayIdentity & { relayDeviceId: string; reqId: string } + ): Promise { + const rows = await database.query( + `SELECT result_json FROM relay_install_results + WHERE user_id = ? AND relay_host_id = ? AND relay_device_id = ? AND req_id = ?`, + [input.userId, input.relayHostId, input.relayDeviceId, input.reqId] + ) + return rows[0] ? (JSON.parse(string(rows[0], 'result_json')) as DeviceCredentialInstalled) : null + } + + private async validateInstallAuthorization( + transaction: RelayDatabase, + input: InstallInput, + now: number + ): Promise { + if (input.authorization.mode === 'relay-basis') { + const rows = await transaction.query( + `SELECT * FROM relay_connection_bases WHERE basis_conn_id = ?`, + [input.authorization.basisConnId] + ) + const basis = rows[0] + if ( + !basis || + string(basis, 'user_id') !== input.userId || + string(basis, 'relay_host_id') !== input.relayHostId || + string(basis, 'relay_device_id') !== input.relayDeviceId || + string(basis, 'credential_kind') !== 'invite' || + number(basis, 'owning_control_generation') !== input.owningControlGeneration || + number(basis, 'active') !== 1 || + number(basis, 'deadline') < now + ) { + throw new RelayStoreError('invalid_relay_basis') + } + return string(basis, 'invite_token_hash') + } + const rows = await transaction.query( + `SELECT * FROM relay_direct_authorizations WHERE direct_auth_id = ?`, + [input.authorization.directAuthId] + ) + const direct = rows[0] + if ( + !direct || + string(direct, 'user_id') !== input.userId || + string(direct, 'relay_host_id') !== input.relayHostId || + string(direct, 'relay_device_id') !== input.relayDeviceId || + number(direct, 'owning_control_generation') !== input.owningControlGeneration || + number(direct, 'deadline') < now || + optionalNumber(direct, 'consumed_at') !== undefined + ) { + throw new RelayStoreError('invalid_direct_authorization') + } + await transaction.query( + `UPDATE relay_direct_authorizations SET consumed_at = ? WHERE direct_auth_id = ?`, + [now, input.authorization.directAuthId] + ) + return null + } + + private async consumeRateWith( + transaction: RelayDatabase, + scopeKey: string, + kind: string, + limit: number, + windowMs: number, + now: number + ): Promise { + const windowStartedAt = Math.floor(now / windowMs) * windowMs + const rows = await transaction.query( + `INSERT INTO relay_rate_windows + (scope_key, window_kind, window_started_at, count) VALUES (?, ?, ?, ?) + ON CONFLICT (scope_key, window_kind, window_started_at) DO UPDATE + SET count = relay_rate_windows.count + 1 RETURNING count`, + [scopeKey, kind, windowStartedAt, 1] + ) + if (number(rows[0]!, 'count') > limit) throw new RelayStoreError('rate_limit_exceeded') + } + + private async auditWith( + transaction: RelayDatabase, + event: RelayIdentity & { + type: string + relayDeviceId?: string + detail: Record + } + ): Promise { + await transaction.query( + `INSERT INTO relay_audit_events + (id, at, type, user_id, relay_host_id, relay_device_id, detail_json) + VALUES (?, ?, ?, ?, ?, ?, ?)`, + [ + randomUUID(), this.now(), event.type, event.userId, event.relayHostId, + event.relayDeviceId, JSON.stringify(event.detail) + ] + ) + } +} diff --git a/cloud/apps/relay/src/database-postgres-timeout.test.ts b/cloud/apps/relay/src/database-postgres-timeout.test.ts new file mode 100644 index 00000000000..a04a9eea042 --- /dev/null +++ b/cloud/apps/relay/src/database-postgres-timeout.test.ts @@ -0,0 +1,201 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const fakes = vi.hoisted(() => ({ + configs: [] as Array>, + query: vi.fn(async () => ({ rows: [], rowCount: 0 })), + release: vi.fn(), + end: vi.fn(async () => undefined) +})) + +vi.mock('pg', () => ({ + default: { + Pool: class { + totalCount = 1 + idleCount = 1 + waitingCount = 0 + end = fakes.end + on = vi.fn() + connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release })) + + constructor(config: Record) { + fakes.configs.push(config) + } + } + } +})) + +import { openRelayDatabase } from './database.js' +import { applyPostgresSchema } from './postgres-schema-startup.js' + +afterEach(() => { + vi.restoreAllMocks() +}) + +describe('PostgreSQL relay deadlines', () => { + beforeEach(() => { + fakes.configs.length = 0 + fakes.query.mockClear() + fakes.release.mockClear() + fakes.end.mockClear() + }) + + it('bounds pool acquisition, statements, locks, and abandoned transactions', async () => { + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused', + poolMax: 3, + applicationName: 'orca-relay/director/director' + }) + + expect(fakes.configs).toEqual([ + expect.objectContaining({ + max: 3, + application_name: 'orca-relay/director/director', + connectionTimeoutMillis: 2_000, + statement_timeout: 5_000, + lock_timeout: 1_000, + idle_in_transaction_session_timeout: 5_000 + }) + ]) + await database.close() + }) +}) + +describe('PostgreSQL schema startup', () => { + it('retries lock and statement timeouts with bounded backoff', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const query = vi + .fn<(statement: string) => Promise>() + .mockRejectedValueOnce(Object.assign(new Error('lock timeout'), { code: '55P03' })) + .mockRejectedValueOnce(Object.assign(new Error('statement timeout'), { code: '57014' })) + .mockResolvedValue(undefined) + const delays: number[] = [] + + await applyPostgresSchema(['CREATE TABLE test'], query, { + random: () => 0, + wait: async (delayMs) => { + delays.push(delayMs) + } + }) + + expect(query).toHaveBeenCalledTimes(3) + expect(delays).toEqual([125, 250]) + }) + + it('retries only the PostgreSQL concurrent type-creation collision', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const collision = Object.assign(new Error('duplicate type'), { + code: '23505', + constraint: 'pg_type_typname_nsp_index' + }) + const query = vi + .fn<(statement: string) => Promise>() + .mockRejectedValueOnce(collision) + .mockResolvedValue(undefined) + + await applyPostgresSchema(['CREATE TABLE IF NOT EXISTS test'], query, { + wait: async () => undefined + }) + + expect(query).toHaveBeenCalledTimes(2) + }) + + it('retries only the PostgreSQL concurrent index-creation collision', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const collision = Object.assign(new Error('duplicate index'), { + code: '23505', + constraint: 'pg_class_relname_nsp_index' + }) + const query = vi + .fn<(statement: string) => Promise>() + .mockRejectedValueOnce(collision) + .mockResolvedValue(undefined) + + await applyPostgresSchema(['CREATE INDEX IF NOT EXISTS test_index ON test(id)'], query, { + wait: async () => undefined + }) + + expect(query).toHaveBeenCalledTimes(2) + }) + + it.each([ + ['pg_type_typname_nsp_index', 'CREATE TABLE test'], + ['pg_class_relname_nsp_index', 'CREATE INDEX test_index ON test(id)'] + ])('does not retry %s for non-idempotent DDL', async (constraint, statement) => { + const error = Object.assign(new Error('duplicate catalog object'), { + code: '23505', + constraint + }) + const query = vi.fn<(statement: string) => Promise>().mockRejectedValue(error) + const pause = vi.fn(async () => undefined) + + await expect( + applyPostgresSchema([statement], query, { wait: pause }) + ).rejects.toBe(error) + + expect(pause).not.toHaveBeenCalled() + }) + + it('does not retry unrelated unique violations', async () => { + const error = Object.assign(new Error('duplicate row'), { + code: '23505', + constraint: 'application_key' + }) + const query = vi.fn<(statement: string) => Promise>().mockRejectedValue(error) + const pause = vi.fn(async () => undefined) + + await expect( + applyPostgresSchema(['CREATE TABLE test'], query, { wait: pause }) + ).rejects.toBe(error) + + expect(pause).not.toHaveBeenCalled() + }) + + it('fails immediately for non-timeout schema errors', async () => { + const error = Object.assign(new Error('permission denied'), { code: '42501' }) + const query = vi.fn<(statement: string) => Promise>().mockRejectedValue(error) + const pause = vi.fn(async () => undefined) + + await expect( + applyPostgresSchema(['CREATE TABLE test'], query, { wait: pause }) + ).rejects.toBe(error) + + expect(query).toHaveBeenCalledTimes(1) + expect(pause).not.toHaveBeenCalled() + }) + + it('stops retrying at the shared startup deadline', async () => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const error = Object.assign(new Error('lock timeout'), { code: '55P03' }) + const delays: number[] = [] + let now = 0 + const query = vi + .fn<(statement: string) => Promise>() + .mockImplementationOnce(async () => { + now = 200 + }) + .mockRejectedValue(error) + + await expect( + applyPostgresSchema(['CREATE TABLE first', 'CREATE TABLE second'], query, { + now: () => now, + random: () => 1, + retryDeadlineMs: 300, + wait: async (delayMs) => { + delays.push(delayMs) + now += delayMs + } + }) + ).rejects.toBe(error) + + expect(query).toHaveBeenCalledTimes(3) + expect(delays).toEqual([100]) + expect(console.warn).toHaveBeenLastCalledWith( + JSON.stringify({ + event: 'orca_relay_postgres_schema_retry_exhausted', + code: '55P03', + attempts: 2 + }) + ) + }) +}) diff --git a/cloud/apps/relay/src/database-transient-error.test.ts b/cloud/apps/relay/src/database-transient-error.test.ts new file mode 100644 index 00000000000..b29ff519328 --- /dev/null +++ b/cloud/apps/relay/src/database-transient-error.test.ts @@ -0,0 +1,18 @@ +import { describe, expect, it } from 'vitest' +import { isRelayDatabaseTransientError } from './database.js' + +describe('relay database transient errors', () => { + it.each(['40P01', '40001', '55P03', '57014', '53300', '57P03', '08001', '08006'])( + 'classifies PostgreSQL code %s as retryable overload', + (code) => { + expect(isRelayDatabaseTransientError({ code })).toBe(true) + } + ) + + it('classifies pool acquisition timeout without hiding programming failures', () => { + expect( + isRelayDatabaseTransientError(new Error('timeout exceeded when trying to connect')) + ).toBe(true) + expect(isRelayDatabaseTransientError(new TypeError('broken invariant'))).toBe(false) + }) +}) diff --git a/cloud/apps/relay/src/database.test.ts b/cloud/apps/relay/src/database.test.ts new file mode 100644 index 00000000000..32e50a7bc6a --- /dev/null +++ b/cloud/apps/relay/src/database.test.ts @@ -0,0 +1,154 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { openInMemoryRelayDatabase, openRelayDatabase } from './database.js' + +const temporaryDirectories: string[] = [] + +afterEach(() => { + for (const directory of temporaryDirectories.splice(0)) { + rmSync(directory, { recursive: true, force: true }) + } +}) + +describe('relay database', () => { + it('creates every durable relay state table', async () => { + const database = await openInMemoryRelayDatabase() + const rows = await database.query( + `SELECT name FROM sqlite_master WHERE type = 'table' AND name LIKE 'relay_%' ORDER BY name` + ) + expect(rows.map((row) => row.name)).toEqual([ + 'relay_admission_selector_cell_additions', + 'relay_admission_selector_intents', + 'relay_admission_selectors', + 'relay_assignment_activity_leases', + 'relay_assignment_migration_incarnations', + 'relay_assignment_migrations', + 'relay_assignment_region_preferences', + 'relay_assignments', + 'relay_audit_events', + 'relay_cell_admission', + 'relay_cell_capabilities', + 'relay_cell_committed_fences', + 'relay_cell_connection_limits', + 'relay_cell_connection_runtime', + 'relay_cell_connection_snapshots', + 'relay_cell_drain_attempt_states', + 'relay_cell_drain_attempts', + 'relay_cell_drain_recovery_attempts', + 'relay_cell_fence_apply_invocations', + 'relay_cell_fence_attempts', + 'relay_cell_fence_plan_bindings', + 'relay_cell_fences', + 'relay_cell_legacy_fence_adoptions', + 'relay_cell_regions', + 'relay_cell_rehome_safety', + 'relay_cell_runtime', + 'relay_cells', + 'relay_confirm_results', + 'relay_confirmable_splices', + 'relay_connection_bases', + 'relay_control_connection_reservations', + 'relay_devices', + 'relay_direct_authorizations', + 'relay_install_results', + 'relay_invites', + 'relay_migration_leases', + 'relay_post_drain_migration_pins', + 'relay_rate_windows', + 'relay_region_rehome_attempts', + 'relay_region_rehome_control', + 'relay_region_rehome_worker_state' + ]) + await database.close() + }) + + it('rolls back every effect and serializes concurrent transactions', async () => { + const database = await openInMemoryRelayDatabase() + await expect( + database.transaction(async (transaction) => { + await transaction.query( + `INSERT INTO relay_rate_windows + (scope_key, window_kind, window_started_at, count) VALUES (?, ?, ?, ?)`, + ['scope', 'invite', 1, 1] + ) + throw new Error('injected failure') + }) + ).rejects.toThrow('injected failure') + expect(await database.query(`SELECT * FROM relay_rate_windows`)).toEqual([]) + + await Promise.all([ + database.transaction(async (transaction) => { + await transaction.query( + `INSERT INTO relay_rate_windows + (scope_key, window_kind, window_started_at, count) VALUES (?, ?, ?, ?)`, + ['scope', 'invite', 1, 1] + ) + }), + database.transaction(async (transaction) => { + const rows = await transaction.query( + `SELECT count FROM relay_rate_windows + WHERE scope_key = ? AND window_kind = ? AND window_started_at = ?`, + ['scope', 'invite', 1] + ) + await transaction.query( + `UPDATE relay_rate_windows SET count = ? + WHERE scope_key = ? AND window_kind = ? AND window_started_at = ?`, + [Number(rows[0]?.count ?? 0) + 1, 'scope', 'invite', 1] + ) + }) + ]) + const rows = await database.query(`SELECT count FROM relay_rate_windows`) + expect(Number(rows[0]?.count)).toBe(2) + await database.close() + }) + + it('persists SQLite state across process-style reopen', async () => { + const dataDir = mkdtempSync(join(tmpdir(), 'orca-relay-db-')) + temporaryDirectories.push(dataDir) + const first = await openRelayDatabase({ dataDir }) + await first.query( + `INSERT INTO relay_rate_windows + (scope_key, window_kind, window_started_at, count) VALUES (?, ?, ?, ?)`, + ['scope', 'connect', 1, 7] + ) + await first.close() + const second = await openRelayDatabase({ dataDir }) + const rows = await second.query(`SELECT count FROM relay_rate_windows`) + expect(Number(rows[0]?.count)).toBe(7) + await second.close() + }) + + it('defaults cells created by an older schema user to the US on reopen', async () => { + const dataDir = mkdtempSync(join(tmpdir(), 'orca-relay-region-db-')) + temporaryDirectories.push(dataDir) + const first = await openRelayDatabase({ dataDir }) + await first.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, 1, 10, 0, 0, 1, 1)`, + ['legacy-cell', 'https://legacy.relay.example.com'] + ) + await first.close() + + const second = await openRelayDatabase({ dataDir }) + expect( + await second.query(`SELECT region FROM relay_cell_regions WHERE cell_id = ?`, [ + 'legacy-cell' + ]) + ).toEqual([{ region: 'us-central1' }]) + await second.close() + }) + + it('indexes region preference expiry by observation time', async () => { + const database = await openInMemoryRelayDatabase() + const rows = await database.query( + `SELECT sql FROM sqlite_master + WHERE type = 'index' AND name = 'relay_assignment_region_preferences_observed'` + ) + expect(rows[0]?.sql).toContain('(observed_at)') + await database.close() + }) +}) diff --git a/cloud/apps/relay/src/database.ts b/cloud/apps/relay/src/database.ts new file mode 100644 index 00000000000..f7208863f42 --- /dev/null +++ b/cloud/apps/relay/src/database.ts @@ -0,0 +1,935 @@ +import { mkdirSync } from 'node:fs' +import { join } from 'node:path' +import { DatabaseSync } from 'node:sqlite' +import pg from 'pg' +import { + emptyPostgresPoolPressureCounts, + PostgresPoolPressure, + type PostgresPoolPressureCounts +} from './postgres-pool-pressure.js' +import { applyPostgresSchema } from './postgres-schema-startup.js' + +export type SqlRow = Record +export type RelayLockOptions = { failIfUnavailable?: boolean } +export type RelayTransactionOptions = { reportRetries?: boolean } + +export interface RelayDatabase { + readonly dialect?: 'sqlite' | 'postgres' + query(sql: string, params?: unknown[]): Promise + queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise + transaction( + operation: (transaction: RelayDatabase) => Promise, + options?: RelayTransactionOptions + ): Promise + close(): Promise +} + +const SCHEMA = ` +CREATE TABLE IF NOT EXISTS relay_invites ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + relay_device_id TEXT NOT NULL, + token_hash TEXT NOT NULL UNIQUE, + state TEXT NOT NULL, + attempt_count BIGINT NOT NULL, + max_attempts BIGINT NOT NULL, + expires_at BIGINT NOT NULL, + reservation_id TEXT, + reservation_expires_at BIGINT, + cooldown_until BIGINT, + created_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, relay_device_id, token_hash) +); +CREATE INDEX IF NOT EXISTS relay_invites_device + ON relay_invites(user_id, relay_host_id, relay_device_id); + +CREATE TABLE IF NOT EXISTS relay_devices ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + relay_device_id TEXT NOT NULL, + current_hash TEXT NOT NULL, + current_version BIGINT NOT NULL, + current_expires_at BIGINT NOT NULL, + grace_hash TEXT, + grace_version BIGINT, + grace_expires_at BIGINT, + revoked_at BIGINT, + updated_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, relay_device_id) +); +CREATE INDEX IF NOT EXISTS relay_devices_current_hash ON relay_devices(relay_host_id, current_hash); +CREATE INDEX IF NOT EXISTS relay_devices_grace_hash ON relay_devices(relay_host_id, grace_hash); + +CREATE TABLE IF NOT EXISTS relay_install_results ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + relay_device_id TEXT NOT NULL, + req_id TEXT NOT NULL, + authorization_mode TEXT NOT NULL, + result_json TEXT NOT NULL, + committed_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, relay_device_id, req_id) +); + +CREATE TABLE IF NOT EXISTS relay_confirmable_splices ( + basis_conn_id TEXT PRIMARY KEY, + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + owning_control_generation BIGINT NOT NULL, + relay_device_id TEXT NOT NULL, + accepted_credential_version BIGINT NOT NULL, + accepted_as TEXT NOT NULL, + confirm_deadline BIGINT NOT NULL, + active BIGINT NOT NULL, + created_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_connection_bases ( + basis_conn_id TEXT PRIMARY KEY, + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + relay_device_id TEXT NOT NULL, + owning_control_generation BIGINT NOT NULL, + credential_kind TEXT NOT NULL, + invite_token_hash TEXT, + accepted_credential_version BIGINT, + accepted_as TEXT, + deadline BIGINT NOT NULL, + active BIGINT NOT NULL, + created_at BIGINT NOT NULL +); + +-- Why: the maintenance sweep matches (active, deadline) while inactive bases +-- accumulate unboundedly. Unindexed it seq-scans millions of rows every cycle +-- and holds the maintenance transaction open long enough to time out +-- assignment lock waits. +CREATE INDEX IF NOT EXISTS relay_connection_bases_active_deadline + ON relay_connection_bases(active, deadline); + +CREATE TABLE IF NOT EXISTS relay_direct_authorizations ( + direct_auth_id TEXT PRIMARY KEY, + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + relay_device_id TEXT NOT NULL, + owning_control_generation BIGINT NOT NULL, + deadline BIGINT NOT NULL, + consumed_at BIGINT +); + +CREATE TABLE IF NOT EXISTS relay_confirm_results ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + req_id TEXT NOT NULL, + basis_conn_id TEXT NOT NULL, + tuple_json TEXT NOT NULL, + result_json TEXT NOT NULL, + committed_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, req_id) +); + +CREATE TABLE IF NOT EXISTS relay_assignments ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + cell_id TEXT NOT NULL, + assignment_epoch BIGINT NOT NULL, + lease_expires_at BIGINT NOT NULL, + last_activity_at BIGINT NOT NULL, + reserved_controls BIGINT NOT NULL, + reserved_splices BIGINT NOT NULL, + reserved_invites BIGINT NOT NULL, + pending_installs BIGINT NOT NULL, + pending_confirmations BIGINT NOT NULL, + migration_leases BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id) +); + +CREATE TABLE IF NOT EXISTS relay_assignment_region_preferences ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + preferred_region TEXT NOT NULL + CHECK (preferred_region IN ('us-central1', 'asia-east2')), + observed_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id) +); +CREATE INDEX IF NOT EXISTS relay_assignment_region_preferences_observed + ON relay_assignment_region_preferences(observed_at); + +CREATE TABLE IF NOT EXISTS relay_region_rehome_worker_state ( + worker_id TEXT PRIMARY KEY, + next_dispatch_at BIGINT NOT NULL, + paused_until BIGINT NOT NULL, + consecutive_failures BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_region_rehome_control ( + control_id TEXT PRIMARY KEY, + generation BIGINT NOT NULL, + enabled BIGINT NOT NULL, + observation_started_at BIGINT NOT NULL, + not_before BIGINT NOT NULL, + rate_per_minute BIGINT NOT NULL, + preference_max_age_ms BIGINT NOT NULL, + drain_grace_ms BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_region_rehome_attempts ( + attempt_id TEXT PRIMARY KEY, + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + preferred_region TEXT NOT NULL CHECK (preferred_region = 'asia-east2'), + source_cell_id TEXT NOT NULL, + source_cell_incarnation TEXT NOT NULL, + target_cell_id TEXT NOT NULL, + target_cell_incarnation TEXT NOT NULL, + previous_epoch BIGINT NOT NULL, + assignment_epoch BIGINT NOT NULL, + drain_grace_ms BIGINT NOT NULL, + send_attempts BIGINT NOT NULL, + last_send_attempt_at BIGINT, + drain_receipt_at BIGINT, + drain_outcome TEXT CHECK ( + drain_outcome IN ('accepted', 'already-accepted', 'host-not-connected') + ), + completed_at BIGINT, + aborted_at BIGINT, + created_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL, + UNIQUE (user_id, relay_host_id, assignment_epoch) +); +CREATE INDEX IF NOT EXISTS relay_region_rehome_attempts_pending + ON relay_region_rehome_attempts(drain_receipt_at, last_send_attempt_at, completed_at, aborted_at); + +CREATE TABLE IF NOT EXISTS relay_cells ( + cell_id TEXT PRIMARY KEY, + cell_url TEXT NOT NULL UNIQUE, + enabled BIGINT NOT NULL, + capacity_requests BIGINT NOT NULL, + reserved_requests BIGINT NOT NULL, + observed_requests BIGINT NOT NULL, + last_heartbeat_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_cell_regions ( + cell_id TEXT PRIMARY KEY, + region TEXT NOT NULL CHECK (region IN ('us-central1', 'asia-east2')) +); + +CREATE TABLE IF NOT EXISTS relay_cell_admission ( + cell_id TEXT PRIMARY KEY, + admission_state TEXT NOT NULL + CHECK (admission_state IN ('existing-only', 'migration-only', 'general')), + updated_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_admission_selectors ( + selector_id TEXT PRIMARY KEY, + generation BIGINT NOT NULL, + attempt_id TEXT, + membership_json TEXT NOT NULL, + updated_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_admission_selector_intents ( + attempt_id TEXT PRIMARY KEY, + expected_generation BIGINT NOT NULL, + intended_generation BIGINT NOT NULL, + previous_membership_json TEXT NOT NULL, + membership_json TEXT NOT NULL, + created_at BIGINT NOT NULL, + committed_at BIGINT +); + +CREATE TABLE IF NOT EXISTS relay_admission_selector_cell_additions ( + attempt_id TEXT PRIMARY KEY, + cells_json TEXT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_cell_runtime ( + cell_id TEXT PRIMARY KEY, + cell_url TEXT NOT NULL, + cell_incarnation TEXT NOT NULL, + started_at BIGINT NOT NULL, + ready BIGINT NOT NULL, + observed_requests BIGINT NOT NULL, + last_heartbeat_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_cell_runtime_heartbeat + ON relay_cell_runtime(ready, last_heartbeat_at); + +CREATE TABLE IF NOT EXISTS relay_cell_capabilities ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + regional_rehome_protocol BIGINT NOT NULL, + last_heartbeat_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_cell_rehome_safety ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + observed_at BIGINT NOT NULL, + sql_failures BIGINT NOT NULL, + reconnects BIGINT NOT NULL, + control_activity_recovery_failures BIGINT NOT NULL, + database_pool_waiting BIGINT NOT NULL, + database_pool_waiters_max BIGINT NOT NULL, + database_pool_wait_ms_max BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_cell_connection_limits ( + cell_id TEXT PRIMARY KEY, + hard_cap BIGINT NOT NULL, + unobserved_bound BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); + +CREATE TABLE IF NOT EXISTS relay_cell_connection_runtime ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + total_connections BIGINT NOT NULL, + in_flight_connections BIGINT NOT NULL, + reserved_connection_units BIGINT NOT NULL, + enforced_connection_units BIGINT NOT NULL, + last_heartbeat_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_cell_connection_runtime_heartbeat + ON relay_cell_connection_runtime(last_heartbeat_at); + +CREATE TABLE IF NOT EXISTS relay_cell_connection_snapshots ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + inclusion_watermark BIGINT NOT NULL, + total_connections BIGINT NOT NULL, + in_flight_connections BIGINT NOT NULL, + reserved_connection_units BIGINT NOT NULL, + enforced_connection_units BIGINT NOT NULL, + snapshot_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_cell_connection_snapshot_freshness + ON relay_cell_connection_snapshots(snapshot_at); + +CREATE TABLE IF NOT EXISTS relay_cell_fences ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + attested_at BIGINT NOT NULL, + expires_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_cell_fences_expiry + ON relay_cell_fences(expires_at); + +CREATE TABLE IF NOT EXISTS relay_cell_committed_fences ( + cell_id TEXT PRIMARY KEY, + attempt_id TEXT NOT NULL UNIQUE, + cell_incarnation TEXT NOT NULL, + attested_at BIGINT NOT NULL, + expires_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_cell_committed_fences_expiry + ON relay_cell_committed_fences(expires_at); + +CREATE TABLE IF NOT EXISTS relay_cell_legacy_fence_adoptions ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + attested_at BIGINT NOT NULL, + expires_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_cell_legacy_fence_adoptions_expiry + ON relay_cell_legacy_fence_adoptions(expires_at); + +CREATE TABLE IF NOT EXISTS relay_cell_fence_attempts ( + attempt_id TEXT PRIMARY KEY, + environment TEXT NOT NULL, + cell_id TEXT NOT NULL, + cell_incarnation TEXT NOT NULL, + mig_name TEXT NOT NULL, + instance_group TEXT NOT NULL, + generation_identity TEXT NOT NULL, + fence_commit TEXT NOT NULL, + plan_sha256 TEXT NOT NULL, + gce_operation TEXT, + created_at BIGINT NOT NULL, + expires_at BIGINT NOT NULL, + apply_started_at BIGINT, + completed_at BIGINT, + aborted_at BIGINT +); +CREATE INDEX IF NOT EXISTS relay_cell_fence_attempts_expiry + ON relay_cell_fence_attempts(expires_at); +CREATE INDEX IF NOT EXISTS relay_cell_fence_attempts_cell + ON relay_cell_fence_attempts(cell_id, created_at); + +CREATE TABLE IF NOT EXISTS relay_cell_fence_plan_bindings ( + attempt_id TEXT PRIMARY KEY, + plan_object_name TEXT NOT NULL, + plan_object_generation TEXT, + var_file_sha256 TEXT NOT NULL, + terraform_state_lineage TEXT NOT NULL, + terraform_state_serial BIGINT NOT NULL, + terraform_state_object_generation TEXT NOT NULL, + terraform_state_object_sha256 TEXT NOT NULL, + request_reason TEXT NOT NULL, + FOREIGN KEY (attempt_id) REFERENCES relay_cell_fence_attempts(attempt_id) +); + +CREATE TABLE IF NOT EXISTS relay_cell_fence_apply_invocations ( + invocation_id TEXT PRIMARY KEY, + attempt_id TEXT NOT NULL, + request_reason TEXT NOT NULL UNIQUE, + started_at BIGINT NOT NULL, + gce_operation TEXT, + FOREIGN KEY (attempt_id) REFERENCES relay_cell_fence_attempts(attempt_id) +); +CREATE INDEX IF NOT EXISTS relay_cell_fence_apply_invocations_attempt + ON relay_cell_fence_apply_invocations(attempt_id, started_at); + +CREATE TABLE IF NOT EXISTS relay_cell_drain_attempts ( + cell_id TEXT PRIMARY KEY, + cell_incarnation TEXT NOT NULL, + planned_grace_ms BIGINT NOT NULL, + attempted_at BIGINT NOT NULL, + retry_after BIGINT NOT NULL, + recover_forward_attempted_at BIGINT +); + +CREATE TABLE IF NOT EXISTS relay_cell_drain_attempt_states ( + attempt_id TEXT PRIMARY KEY, + cell_id TEXT NOT NULL, + cell_incarnation TEXT NOT NULL, + trace_value TEXT NOT NULL UNIQUE, + planned_grace_ms BIGINT NOT NULL, + state TEXT NOT NULL CHECK ( + state IN ( + 'prepared', + 'send-may-have-started', + 'application-receipt', + 'proven-not-delivered' + ) + ), + prepared_at BIGINT NOT NULL, + send_may_have_started_at BIGINT, + send_permit_expires_at BIGINT, + application_receipt_at BIGINT, + backend_success_status BIGINT, + backend_instance TEXT, + receipt_cell_incarnation TEXT, + retry_after BIGINT, + recover_forward_attempted_at BIGINT, + proven_not_delivered_at BIGINT +); +CREATE INDEX IF NOT EXISTS relay_cell_drain_attempt_states_cell + ON relay_cell_drain_attempt_states(cell_id, prepared_at); + +CREATE TABLE IF NOT EXISTS relay_cell_drain_recovery_attempts ( + drain_attempt_id TEXT NOT NULL, + cell_incarnation TEXT NOT NULL, + attempted_at BIGINT NOT NULL, + PRIMARY KEY (drain_attempt_id, cell_incarnation), + FOREIGN KEY (drain_attempt_id) REFERENCES relay_cell_drain_attempt_states(attempt_id) +); + +CREATE TABLE IF NOT EXISTS relay_assignment_activity_leases ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + activity_id TEXT NOT NULL, + activity_kind TEXT NOT NULL, + cell_id TEXT NOT NULL, + request_units BIGINT NOT NULL, + expires_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, activity_id) +); +CREATE INDEX IF NOT EXISTS relay_assignment_activity_expiry + ON relay_assignment_activity_leases(expires_at); + +CREATE TABLE IF NOT EXISTS relay_control_connection_reservations ( + reservation_id TEXT PRIMARY KEY, + idempotency_key TEXT NOT NULL UNIQUE, + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + assignment_epoch BIGINT NOT NULL, + cell_id TEXT NOT NULL, + state TEXT NOT NULL + CHECK (state IN ('reserved', 'late-arrival-debt', 'claimed', 'released')), + inclusion_watermark BIGINT, + claim_activity_id TEXT, + created_at BIGINT NOT NULL, + timeout_at BIGINT NOT NULL, + claimed_at BIGINT, + released_at BIGINT, + updated_at BIGINT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_control_connection_reservation_headroom + ON relay_control_connection_reservations(cell_id, state); +CREATE INDEX IF NOT EXISTS relay_control_connection_reservation_assignment + ON relay_control_connection_reservations( + user_id, relay_host_id, assignment_epoch, cell_id, created_at + ); + +CREATE TABLE IF NOT EXISTS relay_rate_windows ( + scope_key TEXT NOT NULL, + window_kind TEXT NOT NULL, + window_started_at BIGINT NOT NULL, + count BIGINT NOT NULL, + PRIMARY KEY (scope_key, window_kind, window_started_at) +); + +CREATE TABLE IF NOT EXISTS relay_migration_leases ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + source_cell_id TEXT NOT NULL, + target_cell_id TEXT NOT NULL, + assignment_epoch BIGINT NOT NULL, + expires_at BIGINT NOT NULL, + completed_at BIGINT, + PRIMARY KEY (user_id, relay_host_id, assignment_epoch) +); + +CREATE TABLE IF NOT EXISTS relay_assignment_migrations ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + source_cell_id TEXT NOT NULL, + target_cell_id TEXT NOT NULL, + previous_epoch BIGINT NOT NULL, + assignment_epoch BIGINT NOT NULL, + source_request_units BIGINT NOT NULL, + target_reserved_units BIGINT NOT NULL, + expires_at BIGINT NOT NULL, + target_registered_at BIGINT, + completed_at BIGINT, + aborted_at BIGINT, + created_at BIGINT NOT NULL, + updated_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, assignment_epoch) +); +CREATE INDEX IF NOT EXISTS relay_assignment_migrations_active + ON relay_assignment_migrations(expires_at, completed_at, aborted_at); + +CREATE TABLE IF NOT EXISTS relay_assignment_migration_incarnations ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + assignment_epoch BIGINT NOT NULL, + source_cell_incarnation TEXT NOT NULL, + target_cell_incarnation TEXT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, assignment_epoch) +); + +CREATE TABLE IF NOT EXISTS relay_post_drain_migration_pins ( + user_id TEXT NOT NULL, + relay_host_id TEXT NOT NULL, + assignment_epoch BIGINT NOT NULL, + drain_attempt_id TEXT NOT NULL, + source_cell_id TEXT NOT NULL, + source_cell_incarnation TEXT NOT NULL, + target_cell_id TEXT NOT NULL, + target_cell_incarnation TEXT NOT NULL, + source_request_units BIGINT NOT NULL, + target_reserved_units BIGINT NOT NULL, + pinned_at BIGINT NOT NULL, + PRIMARY KEY (user_id, relay_host_id, assignment_epoch) +); +CREATE INDEX IF NOT EXISTS relay_post_drain_migration_pins_attempt + ON relay_post_drain_migration_pins(drain_attempt_id); + +CREATE TABLE IF NOT EXISTS relay_audit_events ( + id TEXT PRIMARY KEY, + at BIGINT NOT NULL, + type TEXT NOT NULL, + user_id TEXT, + relay_host_id TEXT, + relay_device_id TEXT, + detail_json TEXT NOT NULL +); +CREATE INDEX IF NOT EXISTS relay_audit_events_at ON relay_audit_events(at); +` + +function postgresSql(sql: string): string { + let index = 0 + return sql.replace(/\?/g, () => `$${++index}`) +} + +function returnsRows(sql: string): boolean { + return /^\s*(select|with)/i.test(sql) || /returning/i.test(sql) +} + +const POSTGRES_TRANSACTION_PHASES = [ + ['relay_region_rehome_', 'regional-rehome'], + ['relay_assignment_activity_leases', 'activity-lease'], + ['relay_assignment_migration', 'migration'], + ['relay_migration_leases', 'migration'], + ['relay_post_drain_migration_pins', 'migration'], + ['relay_cell_connection_runtime', 'cell-runtime'], + ['relay_cell_connection_snapshots', 'cell-runtime'], + ['relay_cell_runtime', 'cell-runtime'], + ['relay_cell_drain_', 'cell-operation'], + ['relay_cell_fence', 'cell-operation'], + ['relay_cell_committed_fences', 'cell-operation'], + ['relay_cell_legacy_fence_adoptions', 'cell-operation'], + ['relay_admission_selector', 'admission'], + ['relay_cell_admission', 'admission'], + ['relay_control_connection_reservations', 'connection'], + ['relay_confirmable_splices', 'connection'], + ['relay_connection_bases', 'connection'], + ['relay_direct_authorizations', 'connection'], + ['relay_confirm_results', 'connection'], + ['relay_assignment_region_preferences', 'assignment'], + ['relay_assignments', 'assignment'], + ['relay_cell_', 'cell-inventory'], + ['relay_cells', 'cell-inventory'], + ['relay_invites', 'credential'], + ['relay_devices', 'credential'], + ['relay_install_results', 'credential'], + ['relay_rate_windows', 'rate-limit'], + ['relay_audit_events', 'audit'] +] as const + +const postgresTransactionPhaseByError = new WeakMap() + +function postgresTransactionPhase(sql: string): string { + const normalized = sql.toLowerCase() + return POSTGRES_TRANSACTION_PHASES.find(([table]) => normalized.includes(table))?.[1] ?? 'other' +} + +function rememberPostgresTransactionPhase(error: unknown, sql: string): void { + if (typeof error === 'object' && error !== null) { + postgresTransactionPhaseByError.set(error, postgresTransactionPhase(sql)) + } +} + +function postgresTransactionErrorPhase(error: unknown): string { + return typeof error === 'object' && error !== null + ? (postgresTransactionPhaseByError.get(error) ?? 'transaction') + : 'transaction' +} + +class SqliteTransaction implements RelayDatabase { + readonly dialect = 'sqlite' as const + + constructor(protected readonly database: DatabaseSync) {} + + async query(sql: string, params: unknown[] = []): Promise { + const statement = this.database.prepare(sql) + const bound = params.map((value) => (value === undefined ? null : value)) as never[] + if (returnsRows(sql)) return statement.all(...bound) as SqlRow[] + const result = statement.run(...bound) + return [{ changes: Number(result.changes) }] + } + + async queryLocked( + sql: string, + params: unknown[] = [], + _options: RelayLockOptions = {} + ): Promise { + return await this.query(sql, params) + } + + async transaction( + operation: (transaction: RelayDatabase) => Promise, + _options: RelayTransactionOptions = {} + ): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +class SqliteDatabase extends SqliteTransaction { + private tail: Promise = Promise.resolve() + + override async query(sql: string, params: unknown[] = []): Promise { + await this.tail + return await super.query(sql, params) + } + + override async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + const previous = this.tail + let release!: () => void + this.tail = new Promise((resolve) => (release = resolve)) + await previous + this.database.exec('BEGIN IMMEDIATE') + try { + const result = await operation(new SqliteTransaction(this.database)) + this.database.exec('COMMIT') + return result + } catch (error) { + this.database.exec('ROLLBACK') + throw error + } finally { + release() + } + } + + override async close(): Promise { + await this.tail + this.database.close() + } +} + +class PostgresTransaction implements RelayDatabase { + readonly dialect = 'postgres' as const + + constructor(protected readonly client: pg.PoolClient) {} + + async query(sql: string, params: unknown[] = []): Promise { + try { + const result = await this.client.query(postgresSql(sql), params) + return returnsRows(sql) ? (result.rows as SqlRow[]) : [{ changes: result.rowCount ?? 0 }] + } catch (error) { + rememberPostgresTransactionPhase(error, sql) + throw error + } + } + + async queryLocked( + sql: string, + params: unknown[] = [], + options: RelayLockOptions = {} + ): Promise { + try { + return await this.query( + `${sql} FOR UPDATE${options.failIfUnavailable ? ' NOWAIT' : ''}`, + params + ) + } catch (error) { + if ( + options.failIfUnavailable && + String((error as { code?: unknown }).code) === '55P03' + ) { + throw new Error('database_lock_unavailable') + } + throw error + } + } + + async transaction( + operation: (transaction: RelayDatabase) => Promise, + _options: RelayTransactionOptions = {} + ): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +const POSTGRES_TRANSACTION_ATTEMPTS = 3 +const POSTGRES_RETRY_MAX_DELAY_MS = 25 +const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000 +const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 +const POSTGRES_LOCK_TIMEOUT_MS = 1_000 +const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000 + +function retryablePostgresTransactionError(error: unknown): boolean { + const code = String((error as { code?: unknown }).code) + return code === '40P01' || code === '40001' || code === '55P03' +} + +export function isRelayDatabaseTransientError(error: unknown): boolean { + const code = String((error as { code?: unknown }).code) + if (['40P01', '40001', '55P03', '57014', '53300', '57P03', '08001', '08006'].includes(code)) { + return true + } + return String((error as { message?: unknown }).message).includes( + 'timeout exceeded when trying to connect' + ) +} + +async function waitForPostgresRetry(random: () => number = Math.random): Promise { + const delayMs = Math.floor(random() * (POSTGRES_RETRY_MAX_DELAY_MS + 1)) + await new Promise((resolve) => setTimeout(resolve, delayMs)) +} + +class PostgresDatabase implements RelayDatabase { + readonly dialect = 'postgres' as const + private readonly pressure: PostgresPoolPressure + + constructor(private readonly pool: pg.Pool) { + this.pressure = new PostgresPoolPressure(pool) + } + + async query(sql: string, params: unknown[] = []): Promise { + const client = await this.pressure.connect() + try { + const result = await client.query(postgresSql(sql), params) + return returnsRows(sql) ? (result.rows as SqlRow[]) : [{ changes: result.rowCount ?? 0 }] + } finally { + client.release() + } + } + + async queryLocked( + sql: string, + params: unknown[] = [], + options: RelayLockOptions = {} + ): Promise { + try { + return await this.query( + `${sql} FOR UPDATE${options.failIfUnavailable ? ' NOWAIT' : ''}`, + params + ) + } catch (error) { + if ( + options.failIfUnavailable && + String((error as { code?: unknown }).code) === '55P03' + ) { + throw new Error('database_lock_unavailable') + } + throw error + } + } + + async transaction( + operation: (transaction: RelayDatabase) => Promise, + options: RelayTransactionOptions = {} + ): Promise { + for (let attempt = 1; attempt <= POSTGRES_TRANSACTION_ATTEMPTS; attempt++) { + const client = await this.pressure.connect() + try { + await client.query('BEGIN') + const result = await operation(new PostgresTransaction(client)) + await client.query('COMMIT') + return result + } catch (error) { + await client.query('ROLLBACK').catch(() => undefined) + if (!retryablePostgresTransactionError(error) || attempt === POSTGRES_TRANSACTION_ATTEMPTS) { + if (retryablePostgresTransactionError(error) && options.reportRetries !== false) { + console.warn( + JSON.stringify({ + event: 'orca_relay_postgres_transaction_exhausted', + code: String((error as { code?: unknown }).code), + attempts: attempt, + phase: postgresTransactionErrorPhase(error) + }) + ) + } + throw error + } + if (options.reportRetries !== false) { + console.warn( + JSON.stringify({ + event: 'orca_relay_postgres_transaction_retry', + code: String((error as { code?: unknown }).code), + attempt, + phase: postgresTransactionErrorPhase(error) + }) + ) + } + } finally { + client.release() + } + // A PostgreSQL transaction is unusable after an abort, so retry all work + // on a fresh pooled client with a small full-jitter delay. + await waitForPostgresRetry() + } + throw new Error('postgres_transaction_retry_exhausted') + } + + async close(): Promise { + await this.pool.end() + } + + consumePoolPressure(): PostgresPoolPressureCounts { + return this.pressure.consumeCounts() + } + + peekPoolPressure(): PostgresPoolPressureCounts { + return this.pressure.peekCounts() + } +} + +export function consumeRelayDatabasePoolPressure( + database: RelayDatabase +): PostgresPoolPressureCounts { + return database instanceof PostgresDatabase + ? database.consumePoolPressure() + : emptyPostgresPoolPressureCounts() +} + +export function readRelayDatabasePoolPressure( + database: RelayDatabase +): PostgresPoolPressureCounts { + return database instanceof PostgresDatabase + ? database.peekPoolPressure() + : emptyPostgresPoolPressureCounts() +} + +export function absorbPostgresIdleClientErrors(pool: Pick): void { + pool.on('error', () => { + // Why: node-postgres removes failed idle clients itself; leaving `error` + // unhandled would crash the cell and turn a SQL outage into autoheal churn. + console.warn('[orca-relay] idle PostgreSQL client failed') + }) +} + +async function applySchema(database: RelayDatabase): Promise { + for (const statement of SCHEMA.split(';')) { + if (statement.trim()) await database.query(statement) + } +} + +async function applySchemaWithPostgresRetries(database: RelayDatabase): Promise { + await applyPostgresSchema( + SCHEMA.split(';').filter((statement) => statement.trim()), + async (statement) => await database.query(statement) + ) +} + +async function backfillRelayCellRegions(database: RelayDatabase): Promise { + await database.query( + `INSERT INTO relay_cell_regions (cell_id, region) + SELECT cell_id, 'us-central1' FROM relay_cells WHERE true + ON CONFLICT (cell_id) DO NOTHING` + ) +} + +export async function openRelayDatabase(input: { + databaseUrl?: string + dataDir: string + poolMax?: number + applicationName?: string +}): Promise { + let database: RelayDatabase + if (input.databaseUrl) { + const pool = new pg.Pool({ + connectionString: input.databaseUrl, + max: input.poolMax ?? 10, + application_name: input.applicationName, + connectionTimeoutMillis: POSTGRES_CONNECTION_TIMEOUT_MS, + statement_timeout: POSTGRES_STATEMENT_TIMEOUT_MS, + lock_timeout: POSTGRES_LOCK_TIMEOUT_MS, + idle_in_transaction_session_timeout: POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS + }) + absorbPostgresIdleClientErrors(pool) + database = new PostgresDatabase(pool) + } else { + mkdirSync(input.dataDir, { recursive: true }) + const sqlite = new DatabaseSync(join(input.dataDir, 'orca-relay.sqlite')) + sqlite.exec('PRAGMA journal_mode = WAL; PRAGMA foreign_keys = ON;') + database = new SqliteDatabase(sqlite) + } + try { + if (input.databaseUrl) await applySchemaWithPostgresRetries(database) + else await applySchema(database) + await backfillRelayCellRegions(database) + return database + } catch (error) { + await database.close().catch(() => undefined) + throw error + } +} + +export async function openInMemoryRelayDatabase(): Promise { + const sqlite = new DatabaseSync(':memory:') + sqlite.exec('PRAGMA foreign_keys = ON;') + const database = new SqliteDatabase(sqlite) + await applySchema(database) + await backfillRelayCellRegions(database) + return database +} diff --git a/cloud/apps/relay/src/fault-injection-test-entry.ts b/cloud/apps/relay/src/fault-injection-test-entry.ts new file mode 100644 index 00000000000..3f2cc6b6246 --- /dev/null +++ b/cloud/apps/relay/src/fault-injection-test-entry.ts @@ -0,0 +1,58 @@ +import { loadRelayConfig } from './config.js' +import { reconcileCellAdmissionAtStartup } from './cell-admission-startup.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' +import { createRelayServer } from './relay-server.js' +import { readFileSync } from 'node:fs' + +if (process.env.NODE_ENV !== 'test') { + throw new Error('fault injection entry is test-only') +} + +const pattern = process.env.ORCA_RELAY_TEST_FAULT_SQL +const clockFile = process.env.ORCA_RELAY_TEST_CLOCK_FILE +if (!pattern && !clockFile) throw new Error('a test fault configuration is required') + +const config = loadRelayConfig() +const realDatabase = await openRelayDatabase({ + databaseUrl: config.databaseUrl, + dataDir: config.dataDir +}) +let faulted = false + +function wrap(transaction: RelayDatabase): RelayDatabase { + return { + query: async (sql, params) => { + if (pattern && !faulted && sql.includes(pattern)) { + faulted = true + throw new Error('injected SQL failure') + } + return await transaction.query(sql, params) + }, + queryLocked: async (sql, params, options) => + await transaction.queryLocked(sql, params, options), + transaction: async (operation) => + await transaction.transaction(async (nested) => await operation(wrap(nested))), + close: async () => await transaction.close() + } +} + +const database = wrap(realDatabase) +const now = clockFile + ? (): number => { + const offset = Number(readFileSync(clockFile, 'utf8')) + if (!Number.isFinite(offset)) throw new Error('invalid test clock offset') + return Date.now() + offset + } + : Date.now +const { server, sessions, assignments } = createRelayServer(config, database, { now }) +await reconcileCellAdmissionAtStartup(config, assignments) +server.listen(config.port, () => { + console.log(`[orca-relay] listening on ${config.publicUrl} (port ${config.port})`) +}) + +const shutdown = (): void => { + sessions.drain(0) + server.close(() => void realDatabase.close()) +} +process.once('SIGTERM', shutdown) +process.once('SIGINT', shutdown) diff --git a/cloud/apps/relay/src/google-metadata-identity-token.ts b/cloud/apps/relay/src/google-metadata-identity-token.ts new file mode 100644 index 00000000000..f51f6282a65 --- /dev/null +++ b/cloud/apps/relay/src/google-metadata-identity-token.ts @@ -0,0 +1,22 @@ +const METADATA_IDENTITY_ENDPOINT = + 'http://metadata.google.internal/computeMetadata/v1/instance/service-accounts/default/identity' +const METADATA_IDENTITY_TIMEOUT_MS = 5_000 + +export async function googleMetadataIdentityToken( + audience: string, + fetchImpl: typeof fetch = fetch +): Promise { + const endpoint = new URL(METADATA_IDENTITY_ENDPOINT) + endpoint.searchParams.set('audience', audience) + endpoint.searchParams.set('format', 'full') + const response = await fetchImpl(endpoint, { + headers: { 'Metadata-Flavor': 'Google' }, + signal: AbortSignal.timeout(METADATA_IDENTITY_TIMEOUT_MS) + }) + if (!response.ok) throw new Error(`metadata_identity_${response.status}`) + const token = (await response.text()).trim() + if (token.length === 0 || token.length > 16 * 1024) { + throw new Error('metadata_identity_invalid') + } + return token +} diff --git a/cloud/apps/relay/src/host-session-registry.test.ts b/cloud/apps/relay/src/host-session-registry.test.ts new file mode 100644 index 00000000000..f632d5d4358 --- /dev/null +++ b/cloud/apps/relay/src/host-session-registry.test.ts @@ -0,0 +1,1007 @@ +import { EventEmitter } from 'node:events' +import { + ASSIGNMENT_LIMITS, + CONTROL_CONTINUITY_LIMITS, + RELAY_CLOSE_CODE, + RELAY_PROTOCOL_LIMITS +} from '@orca-cloud/relay-contract' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import type { RelayCredentialStore } from './credential-store.js' +import { HostSessionRegistry, type HostSession } from './host-session-registry.js' +import { relayHostLogDigest } from './relay-host-log-digest.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import { + REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + REGIONAL_REHOME_TRUST_PROBE_USER_ID +} from './regional-rehome-trust-probe.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSING = 2 + readonly CLOSED = 3 + readyState = this.OPEN + readonly send = vi.fn() + readonly close = vi.fn((code?: number, reason?: string) => { + this.readyState = this.CLOSED + this.emit('close', code, Buffer.from(reason ?? '')) + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close') + }) +} + +type ActivateSession = ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: HostSession | null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion?: string +) => Promise + +const config = { + port: 8080, + publicUrl: 'https://relay-c3.example.com', + cellUrl: 'https://relay-c3.example.com', + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [ + { + id: 'production-gce-c3', + url: 'https://relay-c3.example.com', + capacityRequests: 4_000 + } + ], + adminAudience: 'https://relay-c3.example.com/v1/admin/drain', + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + adminJwksUrl: 'https://auth.example.com/admin-jwks', + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './test-data' +} satisfies RelayConfig + +const identity = { + sub: 'user-1', + prof: 'profile-1', + relayHostId: 'abcdefghijklmnop', + purpose: 'host-control', + exp: 4_102_444_800 +} satisfies RelayTokenClaims + +function deferred(): { + promise: Promise + resolve(value: T): void +} { + let resolve!: (value: T) => void + const promise = new Promise((complete) => { + resolve = complete + }) + return { promise, resolve } +} + +function createRegistry( + activateControl: RelayAssignmentStore['activateControl'], + store: Partial = {}, + verifyRelayToken: (token: string) => Promise = vi.fn() +): { + registry: HostSessionRegistry + activate: ActivateSession + acquireActivity: ReturnType + renewControlActivity: ReturnType + releaseActivity: ReturnType + observer: { + recordControlClose: ReturnType + recordSpliceClose: ReturnType + } +} { + const acquireActivity = vi.fn().mockResolvedValue(undefined) + const renewControlActivity = vi.fn().mockResolvedValue(undefined) + const releaseActivity = vi.fn().mockResolvedValue(true) + const assignments = { + activateControl, + markMigrationTargetRegistered: vi.fn().mockResolvedValue(undefined), + resolve: vi.fn().mockResolvedValue({ cellId: config.cellId }), + acquireActivity, + renewControlActivity, + releaseActivity + } as unknown as RelayAssignmentStore + const observer = { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql: vi.fn(), + recordControlClose: vi.fn(), + recordSpliceClose: vi.fn() + } satisfies RelayRuntimeObserver + const registry = new HostSessionRegistry( + config, + verifyRelayToken, + store as RelayCredentialStore, + assignments, + new ProcessQueuedByteBudget(), + observer + ) + // Mirrors the production signature exactly so a future positional shift fails to compile. + const bound = ( + registry as unknown as { + activate: ( + socket: WebSocket, + identity: RelayTokenClaims, + existing: HostSession | null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string, + connectionInclusionWatermark?: number + ) => Promise + } + ).activate.bind(registry) + const activate: ActivateSession = ( + socket, + identity, + existing, + generation, + rebind, + assignmentEpoch, + appVersion = '1.4.173' + ) => bound(socket, identity, existing, generation, rebind, assignmentEpoch, appVersion) + return { registry, activate, acquireActivity, renewControlActivity, releaseActivity, observer } +} + +describe('host session cleanup races', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('logs control closes with a host digest and counts them, never the raw id', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { activate, observer } = createRegistry(activateControl) + const socket = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + socket.emit('error', new RangeError('Max payload size exceeded')) + socket.close(1006, 'network reset') + + expect(observer.recordControlClose).toHaveBeenCalledWith(1006) + const line = warn.mock.calls + .map((call) => String(call[0])) + .find((entry) => entry.includes('control closed')) + expect(line).toContain(`host=${relayHostLogDigest(identity.relayHostId)}`) + expect(line).toContain('code=1006') + expect(line).toContain('Max payload size exceeded') + expect(line).not.toContain(identity.relayHostId) + } finally { + warn.mockRestore() + } + }) + + it('contains a dependency failure to one socket instead of the process', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + // The same rejection shape as a pg-pool connect timeout; unguarded, it + // became an unhandled rejection that crashed whole production cells. + const verifyRelayToken = vi.fn(async (): Promise => { + throw new Error('Connection terminated due to connection timeout') + }) + const { activate } = createRegistry(activateControl, {}, verifyRelayToken) + const socket = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + await activate(socket as unknown as WebSocket, identity, null, 1, false, 7) + socket.emit( + 'message', + Buffer.from(JSON.stringify({ type: 'auth-refresh', relayJwt: 'refreshed' })), + false + ) + await vi.waitFor(() => + expect(socket.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.LIMIT_EXCEEDED, + 'relay temporarily unavailable' + ) + ) + expect(warn).toHaveBeenCalledWith( + '[orca-relay] auth refresh failed: Connection terminated due to connection timeout' + ) + } finally { + warn.mockRestore() + } + }) + + it('attributes a control close to its client build without trusting the version string', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { activate } = createRegistry(activateControl) + const socket = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + // A client controls this string, so it must be bounded and stripped like any close reason. + await activate( + socket as unknown as WebSocket, + identity, + null, + 1, + false, + 1, + `1.4.173\n${'x'.repeat(200)}` + ) + socket.close(4408, 'replaced by a newer generation') + + const line = warn.mock.calls + .map((call) => String(call[0])) + .find((entry) => entry.includes('control closed')) + expect(line).toContain('app="1.4.173') + // The raw newline is escaped by JSON.stringify either way; only its escaped + // form proves the strip ran, so assert on that. + expect(line).not.toContain('\\n') + expect(line).toMatch(/app="[^"]{1,80}"/) + } finally { + warn.mockRestore() + } + }) + + it('reports the work a generation replacement destroyed, not the drained aftermath', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate } = createRegistry(activateControl, { + failReservation: vi.fn().mockResolvedValue(undefined) + }) + const firstSocket = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + await activate(firstSocket as unknown as WebSocket, identity, null, 1, false, 1) + const session = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + })! + // Teardown drains both maps before closing the socket, so a close handler that + // reads them live always reports zero regardless of what was actually killed. + session.activeSplices.set('conn-a', () => session.activeSplices.delete('conn-a')) + session.activeSplices.set('conn-b', () => session.activeSplices.delete('conn-b')) + const clientSocket = new FakeSocket() + session.pendingConns.set('conn-c', { + connId: 'conn-c', + connTicket: 'ticket', + reservation: { userId: identity.sub, relayHostId: identity.relayHostId }, + client: clientSocket as unknown as WebSocket, + attachTimer: setTimeout(() => {}, 60_000), + credentialActivityId: null + } as unknown as Parameters[1]) + + const secondSocket = new FakeSocket() + await activate(secondSocket as unknown as WebSocket, identity, session, 2, false, 1) + + const line = warn.mock.calls + .map((call) => String(call[0])) + .find((entry) => entry.includes('replaced by a newer generation')) + expect(line).toContain('splices=2') + expect(line).toContain('pending=1') + } finally { + warn.mockRestore() + } + }) + + it('keeps the first drain snapshot when a drain is retried', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate } = createRegistry(activateControl) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + // Captured before draining, because teardown removes the session from the map. + const session = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + })! + session.activeSplices.set('conn-a', () => session.activeSplices.delete('conn-a')) + + // POST /v1/admin/drain has no idempotency guard, and SIGTERM then SIGINT both + // reach drain(), so a second teardown can be scheduled for the same session. + registry.drain(0) + const scheduled = vi.getTimerCount() + registry.drain(0) + // Pin the premise: if drain ever gains an idempotency guard, the retry schedules no + // second teardown and the assertion below stops defending the write-once snapshot + // while still passing. Compare against the count before the retry rather than an + // absolute, since the session's heartbeat interval is also pending. + expect(vi.getTimerCount()).toBe(scheduled + 1) + vi.advanceTimersByTime(1) + + // Asserting registry state, not the log line: FakeSocket closes synchronously, so + // the line is already emitted before the second teardown runs and would pass either way. + expect(session.closingCounts).toEqual({ splices: 1, pending: 0 }) + }) + + it('drains only the incarnation-bound host and makes replay idempotent', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate } = createRegistry( + activateControl, + {}, + vi.fn(async () => identity) + ) + const firstSocket = new FakeSocket() + const secondSocket = new FakeSocket() + const secondIdentity = { + ...identity, + sub: 'user-2', + relayHostId: 'ponmlkjihgfedcba' + } + await activate(firstSocket as unknown as WebSocket, identity, null, 1, false, 7) + await activate(secondSocket as unknown as WebSocket, secondIdentity, null, 1, false, 3) + const trustProbe = { + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceAssignmentEpoch: 1, + graceMs: 0 + } + expect(registry.get(trustProbe)).toBeNull() + expect(registry.drainHost(trustProbe)).toBe('host-not-connected') + expect(registry.drainHost(trustProbe)).toBe('host-not-connected') + expect(firstSocket.send).not.toHaveBeenCalledWith(expect.stringContaining('"drain"')) + expect(secondSocket.send).not.toHaveBeenCalledWith(expect.stringContaining('"drain"')) + const request = { + attemptId: '11111111-1111-4111-8111-111111111111', + userId: identity.sub, + relayHostId: identity.relayHostId, + sourceAssignmentEpoch: 7, + graceMs: 30_000 + } + + expect(registry.drainHost(request)).toBe('accepted') + expect(firstSocket.send).toHaveBeenCalledWith( + JSON.stringify({ type: 'drain', graceMs: 30_000, recovery: 'resolve-director' }) + ) + expect(secondSocket.send).not.toHaveBeenCalledWith(expect.stringContaining('"drain"')) + firstSocket.emit( + 'message', + Buffer.from(JSON.stringify({ type: 'auth-refresh', relayJwt: 'refreshed' })), + false + ) + await vi.waitFor(() => expect(registry.get(request)?.state).toBe('drain-only')) + const timers = vi.getTimerCount() + expect(registry.drainHost(request)).toBe('already-accepted') + expect(vi.getTimerCount()).toBe(timers) + expect(() => + registry.drainHost({ + ...request, + attemptId: '22222222-2222-4222-8222-222222222222' + }) + ).toThrow('regional_rehome_attempt_conflict') + expect(() => + registry.drainHost({ ...request, sourceAssignmentEpoch: 8 }) + ).toThrow('regional_rehome_assignment_epoch_mismatch') + + const rebound = new FakeSocket() + await activate( + rebound as unknown as WebSocket, + identity, + registry.get(request), + 1, + true, + 7 + ) + expect(registry.get(request)?.state).toBe('drain-only') + expect(rebound.send).toHaveBeenCalledWith(expect.stringContaining('"type":"drain"')) + + await vi.advanceTimersByTimeAsync(30_000) + expect(registry.get(request)).toBeNull() + expect(registry.get({ userId: secondIdentity.sub, relayHostId: secondIdentity.relayHostId })) + .not.toBeNull() + expect(secondSocket.close).not.toHaveBeenCalled() + }) + + it('refreshes the logged client build when a control rebinds', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate } = createRegistry(activateControl) + const firstSocket = new FakeSocket() + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + await activate(firstSocket as unknown as WebSocket, identity, null, 1, false, 1, '1.4.100') + const session = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + })! + const rebindSocket = new FakeSocket() + await activate(rebindSocket as unknown as WebSocket, identity, session, 1, true, 1, '1.4.200') + + // The rebind closes the predecessor. That line is a churn line, so it must carry the + // build that socket ran, not the successor's — the refresh above lands before it closes. + const rebound = warn.mock.calls + .map((call) => String(call[0])) + .find((entry) => entry.includes('control rebound')) + expect(rebound).toContain('app="1.4.100"') + + rebindSocket.close(1006, 'network reset') + const line = warn.mock.calls + .map((call) => String(call[0])) + .find((entry) => entry.includes('code=1006')) + expect(line).toContain('app="1.4.200"') + } finally { + warn.mockRestore() + } + }) + + it('keeps every live control socket indexed when first activations overlap', async () => { + const firstControl = deferred() + const secondControl = deferred() + const activateControl = vi + .fn() + .mockReturnValueOnce(firstControl.promise) + .mockReturnValueOnce(secondControl.promise) + const { registry, activate } = createRegistry(activateControl) + const firstSocket = new FakeSocket() + const secondSocket = new FakeSocket() + + const first = activate(firstSocket as unknown as WebSocket, identity, null, 1, false, 1) + const second = activate(secondSocket as unknown as WebSocket, identity, null, 1, false, 1) + secondControl.resolve('control:production-gce-c3:1') + await Promise.resolve() + firstControl.resolve('control:production-gce-c3:1') + await Promise.all([first, second]) + + const liveSockets = [firstSocket, secondSocket].filter( + (socket) => socket.readyState === socket.OPEN + ) + const session = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(liveSockets).toHaveLength(1) + expect(session?.socket).toBe(liveSockets[0]) + }) + + it('does not publish a control that closes during activation', async () => { + const blocked = deferred() + const activateControl = vi + .fn() + .mockReturnValueOnce(blocked.promise) + const { registry, activate, releaseActivity } = createRegistry(activateControl) + const socket = new FakeSocket() + + const activation = activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + socket.close() + blocked.resolve('control:production-gce-c3:1') + await activation + + expect(registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })).toBeNull() + expect(releaseActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + 'control:production-gce-c3:1' + ) + }) + + it('does not rebind a control that closes during activation', async () => { + const blocked = deferred() + const activateControl = vi + .fn() + .mockResolvedValueOnce('control:production-gce-c3:1') + .mockReturnValueOnce(blocked.promise) + const { registry, activate, releaseActivity } = createRegistry(activateControl) + const originalSocket = new FakeSocket() + await activate(originalSocket as unknown as WebSocket, identity, null, 1, false, 1) + const original = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(original).not.toBeNull() + + const rebindSocket = new FakeSocket() + const rebinding = activate( + rebindSocket as unknown as WebSocket, + identity, + original, + 1, + true, + 1 + ) + rebindSocket.close() + blocked.resolve('control:production-gce-c3:1') + await rebinding + + const session = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(session).toBe(original) + expect(session?.socket).toBe(originalSocket) + expect(session?.state).toBe('active') + expect(releaseActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + 'control:production-gce-c3:1' + ) + }) + + it('rejects client lookup when the indexed control socket is not open', async () => { + const reservation = { + userId: identity.sub, + relayHostId: identity.relayHostId, + credentialKind: 'resume', + relayDeviceId: 'device-1', + leaseExpiresAt: Date.now() + 60_000 + } + const store = { + resolveResume: vi.fn().mockResolvedValue({ userId: identity.sub }), + reserveCredential: vi.fn().mockResolvedValue(reservation), + failReservation: vi.fn().mockResolvedValue(undefined) + } + const activateControl = vi + .fn() + .mockResolvedValueOnce('control:production-gce-c3:1') + const { registry, activate } = createRegistry(activateControl, store) + const controlSocket = new FakeSocket() + await activate(controlSocket as unknown as WebSocket, identity, null, 1, false, 1) + const session = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(session?.state).toBe('active') + // A dead socket that never delivered its close event: state stays active, + // so only the readyState guard can protect the lookup. + controlSocket.readyState = controlSocket.CLOSED + + const client = new FakeSocket() + await registry.acceptClient(client as unknown as WebSocket, identity.relayHostId, 'credential') + + expect(store.failReservation).toHaveBeenCalledWith(reservation) + expect(client.send).toHaveBeenCalledWith( + JSON.stringify({ type: 'relay-hello', ok: false, code: RELAY_CLOSE_CODE.HOST_OFFLINE }) + ) + expect(session?.pendingConns.size).toBe(0) + }) + + it('fails a control waiting behind a stalled activation without breaking serialization', async () => { + const stalled = deferred() + const activateControl = vi + .fn() + .mockReturnValueOnce(stalled.promise) + const { registry, activate } = createRegistry(activateControl) + const stalledSocket = new FakeSocket() + const first = activate(stalledSocket as unknown as WebSocket, identity, null, 1, false, 1) + const waitingSocket = new FakeSocket() + const second = activate(waitingSocket as unknown as WebSocket, identity, null, 1, false, 1) + + // Let the first activation reach the store (clearing its own queue timer) + // before the waiting control's deadline elapses. + await Promise.resolve() + await Promise.resolve() + vi.advanceTimersByTime(30_000) + expect(waitingSocket.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.LIMIT_EXCEEDED, + 'control activation queue stalled' + ) + // The waiting control never reached the store; serialization held. + expect(activateControl).toHaveBeenCalledOnce() + + stalled.resolve('control:production-gce-c3:1') + await Promise.all([first, second]) + const session = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId }) + expect(session?.socket).toBe(stalledSocket) + expect(session?.state).toBe('active') + }) + + it('rejects an activation that returns after drain begins', async () => { + const blocked = deferred() + const activateControl = vi + .fn() + .mockResolvedValueOnce('control:production-gce-c3:1') + .mockReturnValueOnce(blocked.promise) + const { registry, activate, renewControlActivity, releaseActivity } = + createRegistry(activateControl) + const originalSocket = new FakeSocket() + await activate(originalSocket as unknown as WebSocket, identity, null, 1, false, 1) + const original = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + }) + expect(original).not.toBeNull() + + const replacementSocket = new FakeSocket() + const replacement = activate( + replacementSocket as unknown as WebSocket, + identity, + original, + 2, + false, + 1 + ) + registry.drain(100) + blocked.resolve('control:production-gce-c3:2') + await replacement + vi.advanceTimersByTime(100) + vi.advanceTimersByTime(15_000) + + expect(replacementSocket.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.DRAINING, + 'relay draining' + ) + expect(registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })).toBeNull() + expect(renewControlActivity).not.toHaveBeenCalled() + expect(releaseActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + 'control:production-gce-c3:2' + ) + }) + + it('keeps a replacement mapped after stale orphan cleanup', async () => { + const activateControl = vi + .fn() + .mockResolvedValueOnce('control:production-gce-c3:1') + .mockResolvedValueOnce('control:production-gce-c3:2') + const { registry, activate } = createRegistry(activateControl) + const originalSocket = new FakeSocket() + await activate(originalSocket as unknown as WebSocket, identity, null, 1, false, 1) + const original = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + }) + expect(original).not.toBeNull() + originalSocket.close() + + const replacementSocket = new FakeSocket() + await activate( + replacementSocket as unknown as WebSocket, + identity, + original, + 2, + false, + 1 + ) + const replacement = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + }) + vi.advanceTimersByTime(CONTROL_CONTINUITY_LIMITS.orphanGraceMs) + + expect(replacement).not.toBeNull() + expect(registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })).toBe( + replacement + ) + expect(registry.runtimeCounts().controls).toBe(1) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('stops the heartbeat of an actively replaced generation', async () => { + const activateControl = vi + .fn() + .mockResolvedValueOnce('control:production-gce-c3:1') + .mockResolvedValueOnce('control:production-gce-c3:2') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + const originalSocket = new FakeSocket() + await activate(originalSocket as unknown as WebSocket, identity, null, 1, false, 1) + const original = registry.get({ + userId: identity.sub, + relayHostId: identity.relayHostId + }) + expect(original).not.toBeNull() + + await activate( + new FakeSocket() as unknown as WebSocket, + identity, + original, + 2, + false, + 1 + ) + vi.advanceTimersByTime(15_000) + + expect(renewControlActivity).toHaveBeenCalledOnce() + expect(renewControlActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + expect.objectContaining({ + activityId: 'control:production-gce-c3:2', + cellId: 'production-gce-c3' + }) + ) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('keeps 15s pings while halving steady-state control renewals', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + const socket = new FakeSocket() + const activatedAt = Date.now() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + for (let interval = 0; interval < 4; interval++) { + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + socket.emit('message', Buffer.from(JSON.stringify({ type: 'pong' })), false) + } + + const pings = socket.send.mock.calls.filter((call) => + String(call[0]).includes('"ping"') + ) + expect(pings).toHaveLength(4) + expect(renewControlActivity).toHaveBeenCalledTimes(2) + const firstExpiry = Number(renewControlActivity.mock.calls[0]![1].expiresAt) + const secondExpiry = Number(renewControlActivity.mock.calls[1]![1].expiresAt) + expect(firstExpiry).toBe( + activatedAt + + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs + + ASSIGNMENT_LIMITS.activityLeaseMs + + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs + ) + expect(secondExpiry - firstExpiry).toBe(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('retries a failed control renewal on the next ping', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + renewControlActivity.mockRejectedValueOnce(new Error('pool timeout')) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const socket = new FakeSocket() + try { + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + expect(renewControlActivity).toHaveBeenCalledOnce() + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + expect(renewControlActivity).toHaveBeenCalledTimes(2) + } finally { + warn.mockRestore() + registry.drain(0) + vi.advanceTimersByTime(0) + } + }) + + it('retries past a stalled control renewal without waiting for it', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + const stalled = deferred() + renewControlActivity.mockReturnValueOnce(stalled.promise) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2) + + expect(renewControlActivity).toHaveBeenCalledTimes(2) + stalled.resolve(undefined) + await vi.advanceTimersByTimeAsync(0) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('ignores a superseded renewal resolving after a fresher success', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + const stalled = deferred() + renewControlActivity.mockReturnValueOnce(stalled.promise) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2) + expect(renewControlActivity).toHaveBeenCalledTimes(2) + stalled.resolve(undefined) + await vi.advanceTimersByTimeAsync(0) + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + + expect(renewControlActivity).toHaveBeenCalledTimes(2) + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + + expect(renewControlActivity).toHaveBeenCalledTimes(3) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('re-acquires a missing control activity lease', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, acquireActivity, renewControlActivity } = + createRegistry(activateControl) + renewControlActivity.mockRejectedValueOnce(new Error('control_activity_not_found')) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + + expect(acquireActivity).toHaveBeenCalledWith( + { userId: identity.sub, relayHostId: identity.relayHostId }, + { + activityId: 'control:production-gce-c3:1', + kind: 'control', + cellId: config.cellId + } + ) + expect(socket.close).not.toHaveBeenCalled() + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('closes a control whose activity lease moved to another cell', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, acquireActivity, renewControlActivity } = + createRegistry(activateControl) + renewControlActivity.mockRejectedValueOnce(new Error('control_activity_moved')) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + + expect(acquireActivity).not.toHaveBeenCalled() + expect(socket.close).toHaveBeenCalledWith(RELAY_CLOSE_CODE.DRAINING, 'control activity moved') + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('closes when a missing control activity cannot be re-acquired on this cell', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, acquireActivity, renewControlActivity } = + createRegistry(activateControl) + renewControlActivity.mockRejectedValueOnce(new Error('control_activity_not_found')) + acquireActivity.mockRejectedValueOnce(new Error('activity_cell_not_authoritative')) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + + expect(socket.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.DRAINING, + 'control migration completed' + ) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('closes a control when renewal finds its cell is no longer authoritative', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + renewControlActivity.mockRejectedValueOnce(new Error('activity_cell_not_authoritative')) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + + expect(socket.close).toHaveBeenCalledWith( + RELAY_CLOSE_CODE.DRAINING, + 'control migration completed' + ) + registry.drain(0) + vi.advanceTimersByTime(0) + }) + + it('continues renewing throughout the drain grace period', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + registry.drain(60_000) + + for (let interval = 0; interval < 3; interval++) { + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + socket.emit('message', Buffer.from(JSON.stringify({ type: 'pong' })), false) + } + + expect(renewControlActivity).toHaveBeenCalledTimes(2) + expect(socket.readyState).toBe(socket.OPEN) + vi.advanceTimersByTime(15_000) + }) +}) + +describe('control renewal cadence across a rebind', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('keeps the halved cadence after a rebind lands under a stalled renewal', async () => { + const activateControl = vi + .fn() + .mockResolvedValue('control:production-gce-c3:1') + const { registry, activate, renewControlActivity } = createRegistry(activateControl) + const ping = RELAY_PROTOCOL_LIMITS.controlPingIntervalMs + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + + const beat = async (target: FakeSocket): Promise => { + await vi.advanceTimersByTimeAsync(ping) + target.emit('message', Buffer.from(JSON.stringify({ type: 'pong' })), false) + } + + // Age the session so its attempt counter is well above zero. + for (let tick = 0; tick < 5; tick++) await beat(socket) + expect(renewControlActivity).toHaveBeenCalledTimes(3) + + // The next renewal stalls and is still in flight when the control rebinds. + const stalled = deferred() + renewControlActivity.mockReturnValueOnce(stalled.promise) + for (let tick = 0; tick < 2; tick++) await beat(socket) + expect(renewControlActivity).toHaveBeenCalledTimes(4) + + const session = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })! + const rebindSocket = new FakeSocket() + await activate(rebindSocket as unknown as WebSocket, identity, session, 1, true, 1) + stalled.resolve(undefined) + await vi.advanceTimersByTimeAsync(0) + + const before = renewControlActivity.mock.calls.length + for (let tick = 0; tick < 4; tick++) await beat(rebindSocket) + expect(renewControlActivity.mock.calls.length - before).toBe(2) + + registry.drain(0) + vi.advanceTimersByTime(0) + }) +}) + +describe('control lease recovery after the session is gone', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => { + vi.clearAllTimers() + vi.useRealTimers() + }) + + it('does not re-acquire a lease for a session a newer generation already replaced', async () => { + const activateControl = vi + .fn() + .mockResolvedValueOnce('control:production-gce-c3:1') + .mockResolvedValueOnce('control:production-gce-c3:2') + const { registry, activate, acquireActivity, renewControlActivity } = + createRegistry(activateControl) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + try { + const socket = new FakeSocket() + await activate(socket as unknown as WebSocket, identity, null, 1, false, 1) + const session = registry.get({ userId: identity.sub, relayHostId: identity.relayHostId })! + + // The renewal is still in flight when a newer generation takes over. + let failRenewal!: (error: Error) => void + renewControlActivity.mockReturnValueOnce( + new Promise((_resolve, reject) => (failRenewal = reject)) + ) + await vi.advanceTimersByTimeAsync(RELAY_PROTOCOL_LIMITS.controlPingIntervalMs) + expect(renewControlActivity).toHaveBeenCalledOnce() + + const newer = new FakeSocket() + await activate(newer as unknown as WebSocket, identity, session, 2, false, 1) + + // Its release already removed the lease, so the renewal reports it missing. + failRenewal(new Error('control_activity_not_found')) + await vi.advanceTimersByTimeAsync(0) + + expect(acquireActivity).not.toHaveBeenCalled() + } finally { + warn.mockRestore() + registry.drain(0) + vi.advanceTimersByTime(0) + } + }) +}) diff --git a/cloud/apps/relay/src/host-session-registry.ts b/cloud/apps/relay/src/host-session-registry.ts new file mode 100644 index 00000000000..11b7d1de030 --- /dev/null +++ b/cloud/apps/relay/src/host-session-registry.ts @@ -0,0 +1,1226 @@ +import { createHash, createHmac, randomBytes, randomUUID, timingSafeEqual } from 'node:crypto' +import { + ASSIGNMENT_LIMITS, + AuthRefreshSchema, + buildHostChallengePlaintext, + buildHostProofMacInput, + buildHostProofTranscript, + CONTROL_CONTINUITY_LIMITS, + DeviceCredentialInstallSchema, + DeviceCredentialInstallStatusSchema, + DeviceResumeConfirmSchema, + DeviceRevokeSchema, + HostChallengeAckSchema, + HostHelloSchema, + InviteCreateSchema, + RELAY_PROTOCOL_LIMITS, + RELAY_CLOSE_CODE +} from '@orca-cloud/relay-contract' +import nacl from 'tweetnacl' +import type WebSocket from 'ws' +import type { RawData } from 'ws' +import type { RelayConfig } from './config.js' +import type { RelayAssignmentStore } from './assignment-store.js' +import { + RelayCredentialStore, + type CredentialReservation +} from './credential-store.js' +import { relayHostLogDigest } from './relay-host-log-digest.js' +import type { RelayTokenClaims } from './relay-token-verifier.js' +import type { RelayRuntimeObserver } from './relay-observability.js' +import type { PendingHostDataReservation } from './relay-connection-ledger.js' +import { closeRelayWebSocket } from './relay-websocket-close.js' +import { ProcessQueuedByteBudget, wireSplice } from './splice-forwarder.js' + +// Peer-supplied close reasons are logged; keep them printable and short. +function printableCloseReason(reason: Buffer | string): string { + return reason + .toString() + .replace(/[^\x20-\x7e]/g, '') + .slice(0, 80) +} + +type VerifyRelayToken = (token: string) => Promise +type HostState = 'proving' | 'active' | 'orphaned' | 'drain-only' | 'closed' + +const CONTROL_ACTIVITY_RENEWAL_INTERVAL_MS = RELAY_PROTOCOL_LIMITS.controlPingIntervalMs * 2 +// Preserve the existing 75s renewal runway after doubling the successful-call interval. +const CONTROL_ACTIVITY_LEASE_MS = + ASSIGNMENT_LIMITS.activityLeaseMs + + CONTROL_ACTIVITY_RENEWAL_INTERVAL_MS - + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs + +export type HostSession = { + identity: RelayTokenClaims + readonly relayHostId: string + readonly generation: number + readonly assignmentEpoch: number + readonly controlActivityId: string | null + readonly controlResumeSecret: string + // Why: reconnect churn is only actionable once it can be pinned to a client build. + // Refreshed on rebind so it describes the socket that closed, not the first one. + appVersion: string + state: HostState + socket: WebSocket | null + leaseExpiresAt: number + orphanTimer: ReturnType | null + heartbeatTimer: ReturnType | null + lastPongAt: number + activityRenewalDueAt: number + activityRenewalAttempt: number + activityRenewalCompletedAttempt: number + activeConnIds: Set + activeSplices: Map void> + pendingConns: Map + // Why: relay-initiated teardown drains these maps before closing the control + // socket, so the close handler would otherwise always report zero destroyed work. + closingCounts: { splices: number; pending: number } | null + regionalDrainAttemptId: string | null + regionalDrainTimer: ReturnType | null + regionalDrainExpiresAt: number | null +} + +export type RegionalHostDrainOutcome = + | 'accepted' + | 'already-accepted' + | 'host-not-connected' + +type PendingConnection = { + connId: string + connTicket: string + reservation: CredentialReservation + client: WebSocket + attachTimer: ReturnType + credentialActivityId: string | null + capacityReservation?: PendingHostDataReservation +} + +function decodeCanonicalBase64(value: string, bytes: number): Uint8Array | null { + if (!/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(value)) { + return null + } + const decoded = Buffer.from(value, 'base64') + return decoded.length === bytes && decoded.toString('base64') === value ? decoded : null +} + +function relayHostId(publicKey: Uint8Array): string { + return createHash('sha256').update(publicKey).digest('base64url').slice(0, 16) +} + +function payload(raw: RawData, expectedType: string): unknown { + if (typeof raw !== 'string' && !Buffer.isBuffer(raw)) return null + try { + const parsed = JSON.parse(raw.toString()) as Record + if (parsed.type !== expectedType) return null + const { type: _type, ...rest } = parsed + return rest + } catch { + return null + } +} + +function send(socket: WebSocket, type: string, message: object): void { + socket.send(JSON.stringify({ type, ...message })) +} + +// Hosts abandon connects after 15s; waiting much longer than that behind a +// stalled predecessor only accumulates doomed sockets. +const ACTIVATION_QUEUE_WAIT_MS = 30_000 + +export class HostSessionRegistry { + private readonly sessions = new Map() + private readonly activationQueues = new Map>() + private draining = false + + constructor( + private readonly config: RelayConfig, + private readonly verifyRelayToken: VerifyRelayToken, + private readonly store: RelayCredentialStore, + private readonly assignments: RelayAssignmentStore, + private readonly queuedByteBudget: ProcessQueuedByteBudget, + private readonly observer: RelayRuntimeObserver, + private readonly now: () => number = Date.now + ) {} + + async acceptClient( + socket: WebSocket, + hostId: string, + credential: string, + capacityReservation?: PendingHostDataReservation + ): Promise { + if (this.draining) { + capacityReservation?.release() + this.rejectClient(socket, RELAY_CLOSE_CODE.DRAINING) + return + } + if (this.config.role === 'cell') { + const outerIdentity = + (await this.store.resolveResume(hostId, credential)) ?? + (await this.store.resolveInviteForMove(hostId, credential)) + const assignment = outerIdentity + ? await this.assignments.resolve({ userId: outerIdentity.userId, relayHostId: hostId }) + : null + if (!assignment || assignment.cellId !== this.config.cellId) { + capacityReservation?.release() + this.observer.recordAuth(false) + this.rejectClient(socket, RELAY_CLOSE_CODE.WRONG_CELL) + return + } + } + const reservation = await this.store.reserveCredential(hostId, credential) + if (!reservation) { + capacityReservation?.release() + this.observer.recordAuth(false) + this.rejectClient(socket, RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL) + return + } + this.observer.recordAuth(true) + const session = this.sessions.get(this.key(reservation.userId, hostId)) + if ( + !session || + session.state !== 'active' || + !session.socket || + session.socket.readyState !== session.socket.OPEN + ) { + capacityReservation?.release() + await this.store.failReservation(reservation) + this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE) + return + } + if (session.activeConnIds.size + session.pendingConns.size >= 8) { + capacityReservation?.release() + await this.store.failReservation(reservation) + this.rejectClient(socket, RELAY_CLOSE_CODE.LIMIT_EXCEEDED) + return + } + const connId = randomUUID() + const connTicket = randomBytes(32).toString('base64url') + const identity = { userId: reservation.userId, relayHostId: hostId } + const credentialActivityId = + this.config.role === 'cell' + ? `${reservation.credentialKind === 'invite' ? 'invite' : 'confirmation'}:${connId}` + : null + if (credentialActivityId) { + try { + await this.assignments.acquireActivity(identity, { + activityId: credentialActivityId, + kind: reservation.credentialKind === 'invite' ? 'invite' : 'confirmation', + cellId: this.config.cellId + }) + } catch { + capacityReservation?.release() + await this.store.failReservation(reservation) + this.rejectClient(socket, RELAY_CLOSE_CODE.LIMIT_EXCEEDED) + return + } + } + const attachTimer = setTimeout(() => { + session.pendingConns.delete(connId) + capacityReservation?.release() + this.failReservationBestEffort(reservation) + if (credentialActivityId) this.releaseActivityBestEffort(identity, credentialActivityId) + this.rejectClient(socket, RELAY_CLOSE_CODE.HOST_OFFLINE) + }, RELAY_PROTOCOL_LIMITS.hostAttachDeadlineMs) + const pending: PendingConnection = { + connId, + connTicket, + reservation, + client: socket, + attachTimer, + credentialActivityId, + capacityReservation + } + capacityReservation?.bind(connId) + session.pendingConns.set(connId, pending) + send(session.socket, 'conn-open', { + connId, + connTicket, + kind: reservation.credentialKind, + relayDeviceId: reservation.relayDeviceId, + attachDeadlineMs: RELAY_PROTOCOL_LIMITS.hostAttachDeadlineMs + }) + socket.once('close', () => { + const current = session.pendingConns.get(connId) + if (current?.client === socket) { + clearTimeout(current.attachTimer) + session.pendingConns.delete(connId) + current.capacityReservation?.release() + this.failReservationBestEffort(current.reservation) + if (current.credentialActivityId) { + this.releaseActivityBestEffort(identity, current.credentialActivityId) + } + } + }) + } + + async acceptHostData( + socket: WebSocket, + connId: string, + connTicket: string, + generation: number + ): Promise { + const session = [...this.sessions.values()].find((candidate) => + candidate.pendingConns.has(connId) + ) + const pending = session?.pendingConns.get(connId) + if ( + !session || + !pending || + pending.connTicket !== connTicket || + session.generation !== generation || + (session.state !== 'active' && session.state !== 'drain-only') + ) { + this.observer.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid host data ticket') + return false + } + this.observer.recordAuth(true) + clearTimeout(pending.attachTimer) + session.pendingConns.delete(connId) + session.activeConnIds.add(connId) + const basisDeadline = + pending.reservation.credentialKind === 'resume' + ? this.now() + RELAY_PROTOCOL_LIMITS.resumeConfirmationDeadlineMs + : pending.reservation.leaseExpiresAt + const identity = { + userId: pending.reservation.userId, + relayHostId: pending.reservation.relayHostId + } + const spliceActivityId = this.config.role === 'cell' ? `splice:${connId}` : null + try { + if (spliceActivityId) { + await this.assignments.acquireActivity(identity, { + activityId: spliceActivityId, + kind: 'splice', + cellId: this.config.cellId + }) + } + await this.store.recordConnectionBasis({ + ...pending.reservation, + basisConnId: connId, + owningControlGeneration: session.generation, + deadline: basisDeadline + }) + } catch { + session.activeConnIds.delete(connId) + await this.store.failReservation(pending.reservation) + if (spliceActivityId) this.releaseActivityBestEffort(identity, spliceActivityId) + if (pending.credentialActivityId) { + this.releaseActivityBestEffort(identity, pending.credentialActivityId) + } + this.rejectClient(pending.client, RELAY_CLOSE_CODE.LIMIT_EXCEEDED) + socket.close(RELAY_CLOSE_CODE.LIMIT_EXCEEDED, 'basis persistence failed') + return false + } + const close = wireSplice({ + client: pending.client, + host: socket, + budget: this.queuedByteBudget, + onClose: () => { + session.activeConnIds.delete(connId) + session.activeSplices.delete(connId) + this.deactivateBasisBestEffort(connId) + if (spliceActivityId) this.releaseActivityBestEffort(identity, spliceActivityId) + if (pending.credentialActivityId) { + this.releaseActivityBestEffort(identity, pending.credentialActivityId) + } + }, + onForwardedBytes: (bytes) => this.observer.recordForwardedBytes(bytes), + onClosed: (closeInfo) => { + this.observer.recordSpliceClose?.(closeInfo.trigger) + // Only abnormal closes are logged; routine peer disconnects would be + // one line per phone backgrounding. + if ( + closeInfo.code === RELAY_CLOSE_CODE.LIMIT_EXCEEDED || + closeInfo.trigger.includes('error') || + closeInfo.trigger.includes('oversize') + ) { + console.warn( + `[orca-relay] splice closed host=${relayHostLogDigest(session.relayHostId)}` + + ` trigger=${closeInfo.trigger} code=${closeInfo.code}` + + ` reason=${JSON.stringify(closeInfo.reason)}` + ) + } + } + }) + session.activeSplices.set(connId, close) + if (pending.client.readyState !== pending.client.OPEN || socket.readyState !== socket.OPEN) { + close() + return false + } + send(pending.client, 'relay-hello', { + ok: true, + credentialKind: pending.reservation.credentialKind, + leaseExpiresAt: pending.reservation.leaseExpiresAt, + ...(pending.reservation.credentialKind === 'resume' + ? { + acceptedCredentialVersion: pending.reservation.acceptedCredentialVersion, + acceptedAs: pending.reservation.acceptedAs, + resumeExpiresAt: pending.reservation.resumeExpiresAt, + ...(pending.reservation.graceExpiresAt === undefined + ? {} + : { graceExpiresAt: pending.reservation.graceExpiresAt }) + } + : {}) + }) + return true + } + + acceptControl( + socket: WebSocket, + identity: RelayTokenClaims, + connectionInclusionWatermark?: number + ): void { + if (this.draining) { + socket.close(RELAY_CLOSE_CODE.DRAINING, 'relay draining') + return + } + let firstFrameTimer: ReturnType | null = setTimeout(() => { + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'host hello timeout') + }, 2_000) + socket.once('message', (raw, isBinary) => { + if (firstFrameTimer) clearTimeout(firstFrameTimer) + firstFrameTimer = null + if (isBinary) { + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'host hello must be text') + return + } + this.guardSessionTask( + () => + this.beginProof( + socket, + identity, + payload(raw, 'host-hello'), + connectionInclusionWatermark + ), + socket, + 'host hello proof' + ) + }) + } + + // A dependency failure (e.g. a database connect timeout) must cost one + // handshake or command, not the process: an unhandled rejection here has + // crashed whole cells and wiped their in-memory draining flag. 4429 is the + // endpoint-scoped close in the contract, so the client retries this cell. + private guardSessionTask( + task: () => Promise, + socket: WebSocket | null, + context: string + ): void { + void Promise.resolve() + .then(task) + .catch((error: unknown) => { + const message = (error instanceof Error ? error.message : 'unknown') + // Untruncated, unlike peer-supplied close reasons: this is the + // primary diagnostic for the next rejection class. + .replace(/[^\x20-\x7e]/g, '') + console.warn(`[orca-relay] ${context} failed: ${message}`) + socket?.close(RELAY_CLOSE_CODE.LIMIT_EXCEEDED, 'relay temporarily unavailable') + }) + // Terminal: a throw in the handler above must not itself crash the process. + .catch(() => {}) + } + + get(identity: RelayIdentityKey): HostSession | null { + return this.sessions.get(this.key(identity.userId, identity.relayHostId)) ?? null + } + + hasActiveControl(identity: RelayIdentityKey): boolean { + const session = this.get(identity) + return ( + session !== null && + session.state === 'active' && + session.socket !== null && + session.socket.readyState === session.socket.OPEN + ) + } + + runtimeCounts(): { controls: number; splices: number; pendingSplices: number } { + let controls = 0 + let splices = 0 + let pendingSplices = 0 + for (const session of this.sessions.values()) { + const socket = session.socket + if ( + socket !== null && + socket.readyState === socket.OPEN && + (session.state === 'active' || session.state === 'drain-only') + ) { + controls++ + } + splices += session.activeSplices.size + pendingSplices += session.pendingConns.size + } + return { controls, splices, pendingSplices } + } + + drain(graceMs: number): void { + this.draining = true + for (const session of this.sessions.values()) { + if (session.state === 'closed') continue + session.state = 'drain-only' + if (session.socket) send(session.socket, 'drain', { graceMs, recovery: 'resolve-director' }) + setTimeout(() => this.closeDrainedSession(session), graceMs) + } + } + + drainHost(input: { + attemptId: string + userId: string + relayHostId: string + sourceAssignmentEpoch: number + graceMs: number + }): RegionalHostDrainOutcome { + const session = this.get(input) + if (!session || session.state === 'closed') return 'host-not-connected' + if (session.assignmentEpoch !== input.sourceAssignmentEpoch) { + throw new Error('regional_rehome_assignment_epoch_mismatch') + } + if (session.regionalDrainAttemptId) { + if (session.regionalDrainAttemptId !== input.attemptId) { + throw new Error('regional_rehome_attempt_conflict') + } + this.reassertRegionalDrain(session) + return 'already-accepted' + } + session.regionalDrainAttemptId = input.attemptId + session.regionalDrainExpiresAt = this.now() + input.graceMs + this.reassertRegionalDrain(session) + session.regionalDrainTimer = setTimeout( + () => this.closeDrainedSession(session), + input.graceMs + ) + return 'accepted' + } + + isDraining(): boolean { + return this.draining + } + + private async beginProof( + socket: WebSocket, + identity: RelayTokenClaims, + candidate: unknown, + connectionInclusionWatermark?: number + ): Promise { + const hello = HostHelloSchema.safeParse(candidate) + const hostPublicKey = hello.success + ? decodeCanonicalBase64(hello.data.hostPublicKeyB64, 32) + : null + if (!hello.success) { + this.observer.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid host hello') + return + } + if (!hostPublicKey) { + this.observer.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid host public key') + return + } + if ( + hello.data.relayHostId !== identity.relayHostId || + relayHostId(hostPublicKey) !== identity.relayHostId + ) { + this.observer.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'host key binding mismatch') + return + } + // Combined is staging-only compatibility; stamped cells require the durable director epoch. + const assignmentValid = + this.config.role === 'combined' + ? hello.data.assignmentEpoch === 1 + : await this.assignments.verifyCellAssignment({ + userId: identity.sub, + relayHostId: identity.relayHostId, + cellId: this.config.cellId, + assignmentEpoch: hello.data.assignmentEpoch + }) + if (!assignmentValid) { + this.observer.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.WRONG_CELL, 'wrong assignment epoch') + return + } + + const key = this.key(identity.sub, identity.relayHostId) + const existing = this.sessions.get(key) + const rebind = Boolean( + existing && + hello.data.controlResumeSecret && + hello.data.controlResumeSecret === existing.controlResumeSecret && + (existing.state === 'orphaned' || existing.state === 'active') + ) + const generation = rebind ? existing!.generation : (existing?.generation ?? 0) + 1 + const ephemeral = nacl.box.keyPair() + const challengeNonce = randomBytes(nacl.box.nonceLength) + const challengeSecret = randomBytes(32) + const challengeId = randomUUID() + const issuedAt = this.now() + const expiresAt = issuedAt + 10_000 + const transcript = buildHostProofTranscript({ + relayOrigin: this.config.publicUrl, + relayEphemeralPublicKey: ephemeral.publicKey, + challengeNonce, + challengeId, + issuedAt, + expiresAt, + userId: identity.sub, + profileId: identity.prof, + organizationId: identity.org ?? '', + relayHostId: identity.relayHostId, + hostPublicKey, + assignmentEpoch: hello.data.assignmentEpoch, + previousGeneration: hello.data.previousGeneration, + resumeRequested: rebind + }) + const plaintext = buildHostChallengePlaintext(transcript, challengeSecret) + const ciphertext = nacl.box(plaintext, challengeNonce, hostPublicKey, ephemeral.secretKey) + const expectedProof = createHmac('sha256', challengeSecret) + .update(buildHostProofMacInput(transcript)) + .digest() + send(socket, 'host-challenge', { + challengeId, + relayEphemeralPublicKeyB64: Buffer.from(ephemeral.publicKey).toString('base64'), + nonceB64: Buffer.from(challengeNonce).toString('base64'), + ciphertextB64: Buffer.from(ciphertext).toString('base64'), + expiresAt + }) + const proofTimer = setTimeout(() => { + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'host proof timeout') + }, 10_000) + socket.once('message', (raw, isBinary) => { + clearTimeout(proofTimer) + const ack = isBinary ? null : HostChallengeAckSchema.safeParse(payload(raw, 'host-challenge-ack')) + const proof = ack?.success ? decodeCanonicalBase64(ack.data.proofB64, 32) : null + if ( + !ack?.success || + ack.data.challengeId !== challengeId || + !proof || + this.now() > expiresAt || + !timingSafeEqual(proof, expectedProof) + ) { + this.observer.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid host proof') + return + } + this.observer.recordAuth(true) + this.guardSessionTask( + () => + this.activate( + socket, + identity, + existing ?? null, + generation, + rebind, + hello.data.assignmentEpoch, + hello.data.appVersion, + connectionInclusionWatermark + ), + socket, + 'host activation' + ) + }) + } + + private activate( + socket: WebSocket, + identity: RelayTokenClaims, + existing: HostSession | null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string, + connectionInclusionWatermark?: number + ): Promise { + const key = this.key(identity.sub, identity.relayHostId) + const previous = this.activationQueues.get(key) ?? Promise.resolve() + // The timeout only fails this waiting socket; the queue entry still chains + // behind the stalled predecessor so activations never run concurrently. + let queueWaitExpired = false + const queueWaitTimer = setTimeout(() => { + queueWaitExpired = true + socket.close(RELAY_CLOSE_CODE.LIMIT_EXCEEDED, 'control activation queue stalled') + }, ACTIVATION_QUEUE_WAIT_MS) + queueWaitTimer.unref?.() + const activation = previous.catch(() => undefined).then(async () => { + clearTimeout(queueWaitTimer) + if (queueWaitExpired) return + if ((this.sessions.get(key) ?? null) !== existing) { + socket.close(RELAY_CLOSE_CODE.PEER_DROPPED, 'control activation superseded') + return + } + await this.activateCurrent( + socket, + identity, + existing, + generation, + rebind, + assignmentEpoch, + appVersion, + connectionInclusionWatermark + ) + }) + this.activationQueues.set(key, activation) + const cleanup = (): void => { + if (this.activationQueues.get(key) === activation) this.activationQueues.delete(key) + } + void activation.then(cleanup, cleanup) + return activation + } + + private async activateCurrent( + socket: WebSocket, + identity: RelayTokenClaims, + existing: HostSession | null, + generation: number, + rebind: boolean, + assignmentEpoch: number, + appVersion: string, + connectionInclusionWatermark?: number + ): Promise { + let controlActivityId: string | null = null + if (this.config.role === 'cell') { + try { + controlActivityId = await this.assignments.activateControl( + { userId: identity.sub, relayHostId: identity.relayHostId }, + { + cellId: this.config.cellId, + assignmentEpoch, + generation, + connectionInclusionWatermark + } + ) + await this.assignments.markMigrationTargetRegistered( + { userId: identity.sub, relayHostId: identity.relayHostId }, + { cellId: this.config.cellId, assignmentEpoch } + ) + } catch { + // A failure after activateControl succeeded must not strand the + // acquired control activity until lease expiry. + if (controlActivityId) { + this.releaseActivityBestEffort( + { userId: identity.sub, relayHostId: identity.relayHostId }, + controlActivityId + ) + } + socket.close(RELAY_CLOSE_CODE.WRONG_CELL, 'assignment changed during host proof') + return + } + } + if (this.draining || socket.readyState !== socket.OPEN) { + if (controlActivityId) { + this.releaseActivityBestEffort( + { userId: identity.sub, relayHostId: identity.relayHostId }, + controlActivityId + ) + } + if (socket.readyState === socket.OPEN) { + socket.close(RELAY_CLOSE_CODE.DRAINING, 'relay draining') + } + return + } + if (existing) this.observer.recordReconnect() + if (rebind && existing) { + const previousSocket = existing.socket + if (existing.orphanTimer) clearTimeout(existing.orphanTimer) + existing.orphanTimer = null + existing.socket = socket + existing.state = existing.regionalDrainAttemptId ? 'drain-only' : 'active' + existing.appVersion = appVersion + existing.leaseExpiresAt = this.now() + 55 * 60 * 1000 + existing.lastPongAt = this.now() + existing.activityRenewalDueAt = + this.now() + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs + this.wireActiveControl(existing) + this.sendHelloAck(existing) + if (existing.regionalDrainAttemptId) this.reassertRegionalDrain(existing) + previousSocket?.close(RELAY_CLOSE_CODE.PEER_DROPPED, 'control rebound') + return + } + if (existing) { + existing.state = 'closed' + if (existing.heartbeatTimer) clearInterval(existing.heartbeatTimer) + if (existing.orphanTimer) clearTimeout(existing.orphanTimer) + if (existing.regionalDrainTimer) clearTimeout(existing.regionalDrainTimer) + existing.heartbeatTimer = null + existing.orphanTimer = null + existing.regionalDrainTimer = null + existing.closingCounts ??= { + splices: existing.activeSplices.size, + pending: existing.pendingConns.size + } + for (const close of existing.activeSplices.values()) close() + for (const pending of existing.pendingConns.values()) { + clearTimeout(pending.attachTimer) + pending.capacityReservation?.release() + this.failReservationBestEffort(pending.reservation) + if (pending.credentialActivityId) { + this.releaseActivityBestEffort( + { + userId: pending.reservation.userId, + relayHostId: pending.reservation.relayHostId + }, + pending.credentialActivityId + ) + } + this.rejectClient(pending.client, RELAY_CLOSE_CODE.PEER_DROPPED) + } + existing.socket?.close(RELAY_CLOSE_CODE.PEER_DROPPED, 'replaced by a newer generation') + this.releaseControlActivity(existing) + } + const session: HostSession = { + identity, + relayHostId: identity.relayHostId, + generation, + assignmentEpoch, + controlActivityId, + controlResumeSecret: randomBytes(32).toString('base64url'), + appVersion, + state: 'active', + socket, + leaseExpiresAt: this.now() + 55 * 60 * 1000, + orphanTimer: null, + heartbeatTimer: null, + lastPongAt: this.now(), + activityRenewalDueAt: this.now() + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs, + activityRenewalAttempt: 0, + activityRenewalCompletedAttempt: 0, + activeConnIds: new Set(), + activeSplices: new Map(), + pendingConns: new Map(), + closingCounts: null, + regionalDrainAttemptId: null, + regionalDrainTimer: null, + regionalDrainExpiresAt: null + } + this.sessions.set(this.key(identity.sub, identity.relayHostId), session) + this.wireActiveControl(session) + this.sendHelloAck(session) + } + + private wireActiveControl(session: HostSession): void { + const socket = session.socket! + const wiredAt = this.now() + // Why: pin the build to THIS socket. A rebind refreshes session.appVersion and + // only then closes the predecessor, whose close event always lands after that + // write, so reading it at log time would stamp the successor's build. + const appVersion = session.appVersion + let socketError: string | null = null + // Why: an unhandled ws 'error' (e.g. an oversize control frame) would + // otherwise throw process-wide; the message also explains the close below. + socket.on('error', (error) => { + socketError ??= error.message + }) + socket.once('close', (code, reason) => { + this.observer.recordControlClose?.(code) + // One line per control close makes reconnect churners attributable by + // host digest without exposing the raw relay host id. + console.warn( + `[orca-relay] control closed host=${relayHostLogDigest(session.relayHostId)}` + + ` gen=${session.generation} state=${session.state} ageMs=${this.now() - wiredAt}` + + ` app=${JSON.stringify(printableCloseReason(appVersion))}` + + ` splices=${session.closingCounts?.splices ?? session.activeSplices.size}` + + ` pending=${session.closingCounts?.pending ?? session.pendingConns.size}` + + ` code=${code} reason=${JSON.stringify(printableCloseReason(reason))}` + + (socketError === null ? '' : ` error=${JSON.stringify(printableCloseReason(socketError))}`) + ) + }) + socket.on('message', (raw, isBinary) => { + if (isBinary || (session.state !== 'active' && session.state !== 'drain-only')) return + try { + const parsed = JSON.parse(raw.toString()) as Record + if (parsed.type === 'pong') { + session.lastPongAt = this.now() + return + } + if (parsed.type === 'auth-refresh') { + // Close the socket the message arrived on: after a rebind, + // session.socket already points at the successor. + this.guardSessionTask(() => this.acceptRefresh(session, raw), socket, 'auth refresh') + return + } + this.guardSessionTask( + () => this.acceptControlCommand(session, parsed.type, raw), + socket, + 'control command' + ) + } catch { + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid control JSON') + } + }) + socket.once('close', () => { + if (session.socket !== socket || session.state === 'closed') return + session.socket = null + session.state = session.regionalDrainAttemptId ? 'drain-only' : 'orphaned' + if (session.heartbeatTimer) clearInterval(session.heartbeatTimer) + session.heartbeatTimer = null + session.orphanTimer = setTimeout(() => { + session.state = 'closed' + for (const close of session.activeSplices.values()) close() + const key = this.key(session.identity.sub, session.relayHostId) + if (this.sessions.get(key) === session) this.sessions.delete(key) + this.releaseControlActivity(session) + }, CONTROL_CONTINUITY_LIMITS.orphanGraceMs) + }) + if (session.heartbeatTimer) clearInterval(session.heartbeatTimer) + session.heartbeatTimer = setInterval( + () => this.heartbeat(session), + RELAY_PROTOCOL_LIMITS.controlPingIntervalMs + ) + } + + private async acceptRefresh(session: HostSession, raw: RawData): Promise { + const parsed = AuthRefreshSchema.safeParse(payload(raw, 'auth-refresh')) + if (!parsed.success) return + const refreshed = await this.verifyRelayToken(parsed.data.relayJwt) + const sameIdentity = + refreshed && + refreshed.sub === session.identity.sub && + refreshed.prof === session.identity.prof && + refreshed.org === session.identity.org && + refreshed.relayHostId === session.identity.relayHostId + if (!sameIdentity) { + session.socket?.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'refresh identity changed') + return + } + session.identity = refreshed + if (!session.regionalDrainAttemptId) session.state = 'active' + } + + private heartbeat(session: HostSession): void { + const key = this.key(session.identity.sub, session.relayHostId) + if (this.sessions.get(key) !== session || session.state === 'closed') { + if (session.heartbeatTimer) clearInterval(session.heartbeatTimer) + session.heartbeatTimer = null + return + } + const now = this.now() + if (!session.socket) return + const controlActivityId = session.controlActivityId + if (controlActivityId && now >= session.activityRenewalDueAt) { + const attempt = ++session.activityRenewalAttempt + const startedAt = now + void this.assignments + .renewControlActivity( + { userId: session.identity.sub, relayHostId: session.relayHostId }, + { + activityId: controlActivityId, + cellId: this.config.cellId, + expiresAt: startedAt + CONTROL_ACTIVITY_LEASE_MS + } + ) + .then(() => { + if (attempt <= session.activityRenewalCompletedAttempt) return + session.activityRenewalCompletedAttempt = attempt + session.activityRenewalDueAt = startedAt + CONTROL_ACTIVITY_RENEWAL_INTERVAL_MS + }) + .catch(async (error: unknown) => { + if (error instanceof Error && error.message === 'activity_cell_not_authoritative') { + // Completion fences a late drain-only heartbeat after all source work is gone. + session.socket?.close(RELAY_CLOSE_CODE.DRAINING, 'control migration completed') + return + } + if (error instanceof Error && error.message === 'control_activity_not_found') { + if ( + this.sessions.get(key) !== session || + session.state === 'closed' || + !session.socket + ) { + return + } + try { + await this.assignments.acquireActivity( + { userId: session.identity.sub, relayHostId: session.relayHostId }, + { + activityId: controlActivityId, + kind: 'control', + cellId: this.config.cellId + } + ) + this.observer.recordControlActivityRecovery?.(true) + } catch (acquireError: unknown) { + this.observer.recordControlActivityRecovery?.(false) + if ( + acquireError instanceof Error && + acquireError.message === 'activity_cell_not_authoritative' + ) { + session.socket?.close(RELAY_CLOSE_CODE.DRAINING, 'control migration completed') + return + } + console.warn('[orca-relay] control activity recovery failed') + } + return + } + if (error instanceof Error && error.message === 'control_activity_moved') { + session.socket?.close(RELAY_CLOSE_CODE.DRAINING, 'control activity moved') + return + } + console.warn('[orca-relay] control activity renewal failed') + }) + // Terminal handler: a throw inside the async catch above (e.g. a + // future await) must not become a process-killing rejection. + .catch(() => { + console.warn('[orca-relay] control activity renewal handling failed') + }) + } + if (now - session.lastPongAt > 75_000) { + session.socket.close(RELAY_CLOSE_CODE.PEER_DROPPED, 'control silence timeout') + return + } + const expiresAt = session.identity.exp * 1000 + if (now > expiresAt + CONTROL_CONTINUITY_LIMITS.expiredAuthExistingSpliceGraceMs) { + session.socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'relay authorization expired') + return + } + if (now > expiresAt) session.state = 'drain-only' + if (now > session.leaseExpiresAt) { + send(session.socket, 'drain', { graceMs: 0, recovery: 'resolve-director' }) + session.socket.close(RELAY_CLOSE_CODE.DRAINING, 'control lease expired') + return + } + send(session.socket, 'ping', { t: now }) + } + + private sendHelloAck(session: HostSession): void { + if (!session.socket) return + send(session.socket, 'host-hello-ack', { + v: 1, + generation: session.generation, + controlResumeSecret: session.controlResumeSecret, + leaseExpiresAt: session.leaseExpiresAt, + activeConnIds: [...session.activeConnIds], + pendingConns: [...session.pendingConns.values()].map((pending) => ({ + connId: pending.connId, + connTicket: pending.connTicket + })) + }) + } + + private closeDrainedSession(session: HostSession): void { + if (session.state === 'closed') return + if (session.heartbeatTimer) clearInterval(session.heartbeatTimer) + if (session.orphanTimer) clearTimeout(session.orphanTimer) + if (session.regionalDrainTimer) clearTimeout(session.regionalDrainTimer) + session.heartbeatTimer = null + session.orphanTimer = null + session.regionalDrainTimer = null + session.regionalDrainExpiresAt = null + session.closingCounts ??= { + splices: session.activeSplices.size, + pending: session.pendingConns.size + } + for (const close of session.activeSplices.values()) { + close(RELAY_CLOSE_CODE.DRAINING, 'relay draining') + } + for (const pending of session.pendingConns.values()) { + clearTimeout(pending.attachTimer) + pending.capacityReservation?.release() + this.failReservationBestEffort(pending.reservation) + if (pending.credentialActivityId) { + this.releaseActivityBestEffort( + { + userId: pending.reservation.userId, + relayHostId: pending.reservation.relayHostId + }, + pending.credentialActivityId + ) + } + this.rejectClient(pending.client, RELAY_CLOSE_CODE.DRAINING) + } + session.pendingConns.clear() + session.state = 'closed' + if (session.socket) { + closeRelayWebSocket( + session.socket, + RELAY_CLOSE_CODE.DRAINING, + 'resolve configured director' + ) + } + const key = this.key(session.identity.sub, session.relayHostId) + if (this.sessions.get(key) === session) this.sessions.delete(key) + this.releaseControlActivity(session) + } + + private reassertRegionalDrain(session: HostSession): void { + session.state = 'drain-only' + if (!session.socket) return + send(session.socket, 'drain', { + graceMs: Math.max(0, (session.regionalDrainExpiresAt ?? this.now()) - this.now()), + recovery: 'resolve-director' + }) + } + + private key(userId: string, hostId: string): string { + return `${userId}\0${hostId}` + } + + private async acceptControlCommand( + session: HostSession, + type: unknown, + raw: RawData + ): Promise { + if (typeof type !== 'string' || !session.socket) return + try { + const identity = { userId: session.identity.sub, relayHostId: session.relayHostId } + if (type === 'invite-create') { + if (session.state !== 'active') throw new Error('authorization_expired') + const request = InviteCreateSchema.parse(payload(raw, type)) + const activityId = `invite-offer:${request.reqId}` + if (this.config.role === 'cell') { + await this.assignments.acquireActivity(identity, { + activityId, + kind: 'invite', + cellId: this.config.cellId + }) + } + let invite + try { + invite = await this.store.createInvite(identity, request.relayDeviceId) + if (this.config.role === 'cell') { + await this.assignments.acquireActivity(identity, { + activityId, + kind: 'invite', + cellId: this.config.cellId, + expiresAt: invite.expiresAt + }) + } + } catch (error) { + if (this.config.role === 'cell') { + await this.assignments.releaseActivity(identity, activityId) + } + throw error + } + send(session.socket, 'invite-created', { reqId: request.reqId, ...invite }) + return + } + if (type === 'device-credential-install') { + const request = DeviceCredentialInstallSchema.parse(payload(raw, type)) + if ( + session.state !== 'active' && + request.authorization.mode === 'authenticated-direct' + ) { + throw new Error('authorization_expired') + } + const installActivityId = `install:${request.reqId}` + if (this.config.role === 'cell') { + await this.assignments.acquireActivity(identity, { + activityId: installActivityId, + kind: 'install', + cellId: this.config.cellId + }) + } + let result + try { + if (request.authorization.mode === 'authenticated-direct') { + await this.store.recordDirectAuthorization({ + ...identity, + relayDeviceId: request.relayDeviceId, + directAuthId: request.authorization.directAuthId, + owningControlGeneration: session.generation, + deadline: this.now() + RELAY_PROTOCOL_LIMITS.resumeConfirmationDeadlineMs + }) + } + result = await this.store.installCredential({ + ...identity, + ...request, + owningControlGeneration: session.generation + }) + } finally { + if (this.config.role === 'cell') { + await this.assignments.releaseActivity(identity, installActivityId) + } + } + if (request.authorization.mode === 'relay-basis') { + await this.assignments.releaseActivity( + identity, + `invite:${request.authorization.basisConnId}` + ) + } + send(session.socket, 'device-credential-installed', result) + return + } + if (type === 'device-credential-install-status') { + const request = DeviceCredentialInstallStatusSchema.parse(payload(raw, type)) + const result = await this.store.installStatus({ ...identity, ...request }) + send(session.socket, 'device-credential-install-status-result', { + v: 1, + reqId: request.reqId, + state: result ? 'committed' : 'not-found', + ...(result ? { result } : {}) + }) + return + } + if (type === 'device-resume-confirm') { + const request = DeviceResumeConfirmSchema.parse(payload(raw, type)) + const result = await this.store.confirmResume({ + ...identity, + ...request, + owningControlGeneration: session.generation + }) + await this.assignments.releaseActivity(identity, `confirmation:${request.basisConnId}`) + send(session.socket, 'device-resume-confirmed', result) + return + } + if (type === 'device-revoke') { + const request = DeviceRevokeSchema.parse(payload(raw, type)) + await this.store.revoke(identity, request.relayDeviceId) + send(session.socket, 'device-revoked', { reqId: request.reqId }) + return + } + this.sendControlError(session, undefined, 'unknown_control_message') + } catch (error) { + const reqId = (() => { + const candidate = payload(raw, type) + return typeof candidate === 'object' && candidate && 'reqId' in candidate + ? String(candidate.reqId) + : undefined + })() + this.sendControlError( + session, + reqId, + error instanceof Error ? error.message : 'control_operation_failed' + ) + } + } + + private sendControlError(session: HostSession, reqId: string | undefined, code: string): void { + if (session.socket) send(session.socket, 'control-error', { ...(reqId ? { reqId } : {}), code }) + } + + private rejectClient(socket: WebSocket, code: number): void { + send(socket, 'relay-hello', { ok: false, code }) + closeRelayWebSocket(socket, code, 'relay connection rejected') + } + + private releaseControlActivity(session: HostSession): void { + if (!session.controlActivityId) return + this.releaseActivityBestEffort( + { userId: session.identity.sub, relayHostId: session.relayHostId }, + session.controlActivityId + ) + } + + private releaseActivityBestEffort(identity: RelayIdentityKey, activityId: string): void { + void this.assignments.releaseActivity(identity, activityId).catch(() => { + // Why: expiry cleanup is the durable fallback; a transient SQL failure while + // closing a socket must not become an unhandled rejection that kills the cell. + console.warn('[orca-relay] activity release deferred to lease cleanup') + }) + } + + private failReservationBestEffort(reservation: CredentialReservation): void { + // Why: reservation deadlines and basis cleanup are durable recovery paths; + // transient SQL errors during socket callbacks must stay process-contained. + void this.store.failReservation(reservation).catch(() => { + console.warn('[orca-relay] reservation release deferred to credential cleanup') + }) + } + + private deactivateBasisBestEffort(connId: string): void { + void this.store.deactivateBasis(connId).catch(() => { + console.warn('[orca-relay] basis deactivation deferred to credential cleanup') + }) + } +} + +export type RelayIdentityKey = { userId: string; relayHostId: string } diff --git a/cloud/apps/relay/src/index.ts b/cloud/apps/relay/src/index.ts new file mode 100644 index 00000000000..635f3ae9b38 --- /dev/null +++ b/cloud/apps/relay/src/index.ts @@ -0,0 +1,131 @@ +import { + formatAssignmentInventorySnapshot, + readAssignmentInventorySnapshot +} from './assignment-inventory-snapshot.js' +import { RelayAssignmentStore } from './assignment-store.js' +import { loadRelayConfig } from './config.js' +import { startCellHeartbeat } from './cell-heartbeat-client.js' +import { + reconcileCellAdmissionAtStartup, + roleOwnsAssignmentMaintenance +} from './cell-admission-startup.js' +import { + consumeRelayDatabasePoolPressure, + openRelayDatabase, + readRelayDatabasePoolPressure +} from './database.js' +import { runAssignmentCleanup } from './assignment-cleanup-steps.js' +import { runRelayBackgroundOperation } from './relay-background-operation.js' +import { observedRelayRequests } from './relay-observability.js' +import { startRegionalRehomeWorker } from './regional-rehome-worker.js' +import { createRelayServer } from './relay-server.js' +import { + formatRegisteredMigrationInventory, + readRegisteredMigrationInventory +} from './registered-migration-inventory.js' + +const config = loadRelayConfig() +const database = await openRelayDatabase({ + databaseUrl: config.databaseUrl, + dataDir: config.dataDir, + poolMax: config.databasePoolMax, + applicationName: `orca-relay/${config.role}/${config.cellId}` +}) +await reconcileCellAdmissionAtStartup(config, new RelayAssignmentStore(database)) +const { + server, + sessions, + store, + assignments, + observability, + runtimeCounts, + connectionSnapshot, + ready, + cellIncarnation +} = createRelayServer(config, database) +const cleanupTimer = setInterval( + () => + void runRelayBackgroundOperation( + () => store.cleanup(), + '[orca-relay] credential cleanup failed' + ), + 30_000 +) +const assignmentCleanupTimer = roleOwnsAssignmentMaintenance(config.role) + ? setInterval(() => { + void runAssignmentCleanup(assignments) + }, 30_000) + : null +const inventorySnapshotTimer = roleOwnsAssignmentMaintenance(config.role) + ? setInterval(() => { + void runRelayBackgroundOperation(async () => { + const snapshot = await readAssignmentInventorySnapshot(database, Date.now()) + for (const line of formatAssignmentInventorySnapshot(snapshot)) console.warn(line) + }, '[orca-relay] inventory snapshot failed') + }, 60_000) + : null +const migrationInventoryTimer = roleOwnsAssignmentMaintenance(config.role) + ? setInterval(() => { + void runRelayBackgroundOperation(async () => { + const inventory = await readRegisteredMigrationInventory(database, Date.now()) + for (const line of formatRegisteredMigrationInventory(inventory)) console.warn(line) + }, '[orca-relay] migration inventory failed') + }, 5 * 60_000) + : null +cleanupTimer.unref() +assignmentCleanupTimer?.unref() +inventorySnapshotTimer?.unref() +migrationInventoryTimer?.unref() +observability.start(() => ({ + ...runtimeCounts(), + ...consumeRelayDatabasePoolPressure(database) +})) +const regionalRehomeWorker = startRegionalRehomeWorker(config, assignments, { + safetySnapshot: () => ({ + ...observability.regionalRehomeRuntimeSafety(), + ...readRelayDatabasePoolPressure(database) + }) +}) +const heartbeat = startCellHeartbeat(config, { + ready, + incarnation: cellIncarnation, + observedRequests: () => observedRelayRequests(runtimeCounts()), + connectionCounts: () => { + const snapshot = connectionSnapshot() + const counts = runtimeCounts() + return { + totalConnections: snapshot?.physicalConnections ?? counts.totalConnections, + inFlightConnections: + snapshot?.inFlightConnections ?? counts.inFlightConnections ?? 0, + reservedConnectionUnits: + snapshot?.reservedConnectionUnits ?? counts.reservedConnectionUnits ?? 0, + enforcedConnectionUnits: + snapshot?.enforcedConnectionUnits ?? + counts.enforcedConnectionUnits ?? + counts.totalConnections, + inclusionWatermark: snapshot?.inclusionWatermark + } + }, + regionalRehomeSafety: () => ({ + ...observability.regionalRehomeRuntimeSafety(), + ...readRelayDatabasePoolPressure(database) + }) +}) + +server.listen(config.port, () => { + console.log(`[orca-relay] listening on ${config.publicUrl} (port ${config.port})`) +}) + +const shutdown = (): void => { + clearInterval(cleanupTimer) + if (assignmentCleanupTimer) clearInterval(assignmentCleanupTimer) + if (inventorySnapshotTimer) clearInterval(inventorySnapshotTimer) + if (migrationInventoryTimer) clearInterval(migrationInventoryTimer) + observability.stop() + heartbeat?.stop() + regionalRehomeWorker?.stop() + sessions.drain(0) + server.close(() => void database.close()) +} +process.once('SIGTERM', shutdown) +process.once('SIGINT', shutdown) diff --git a/cloud/apps/relay/src/migration-recovery-postgres.test.ts b/cloud/apps/relay/src/migration-recovery-postgres.test.ts new file mode 100644 index 00000000000..aee41278d49 --- /dev/null +++ b/cloud/apps/relay/src/migration-recovery-postgres.test.ts @@ -0,0 +1,1376 @@ +import { ASSIGNMENT_LIMITS } from '@orca-cloud/relay-contract' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { + RelayAssignmentStore, + STRANDED_MIGRATION_ABANDON_MS, + type RelayAssignmentMigration +} from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' +import { readRegisteredMigrationInventory } from './registered-migration-inventory.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +type RecoveryFixture = { + store: RelayAssignmentStore + identity: { userId: string; relayHostId: string } + migration: RelayAssignmentMigration + sourceControlId: string + targetControlId: string + cellIncarnation: string + cells: { + source: { id: string; url: string; capacityRequests: number } + failed: { id: string; url: string; capacityRequests: number } + replacement: { id: string; url: string; capacityRequests: number } + } + advancePastHeartbeat: () => void + advance: (milliseconds: number) => void + heartbeat: (cell: { id: string; url: string }) => Promise +} + +describePostgres('PostgreSQL migration recovery', () => { + let database: RelayDatabase + let sequence = 0 + + beforeAll(async () => { + database = await openRelayDatabase({ databaseUrl, dataDir: '' }) + }) + + afterAll(async () => { + await database.query( + `DELETE FROM relay_control_connection_reservations + WHERE user_id LIKE 'recovery-user-%'` + ) + await database.query( + `DELETE FROM relay_post_drain_migration_pins WHERE user_id LIKE 'recovery-user-%'` + ) + await database.query( + `DELETE FROM relay_assignment_migration_incarnations WHERE user_id LIKE 'recovery-user-%'` + ) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id LIKE 'recovery-user-%'` + ) + await database.query( + `DELETE FROM relay_assignment_migrations WHERE user_id LIKE 'recovery-user-%'` + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id LIKE 'recovery-user-%'`) + await database.query( + `DELETE FROM relay_cell_fence_apply_invocations WHERE attempt_id IN ( + SELECT attempt_id FROM relay_cell_fence_attempts + WHERE cell_id LIKE 'recovery-cell-%' + )` + ) + await database.query( + `DELETE FROM relay_cell_committed_fences WHERE cell_id LIKE 'recovery-cell-%'` + ) + await database.query( + `DELETE FROM relay_cell_legacy_fence_adoptions + WHERE cell_id LIKE 'recovery-cell-%'` + ) + await database.query( + `DELETE FROM relay_cell_fence_plan_bindings WHERE attempt_id IN ( + SELECT attempt_id FROM relay_cell_fence_attempts + WHERE cell_id LIKE 'recovery-cell-%' + )` + ) + await database.query( + `DELETE FROM relay_cell_fence_attempts WHERE cell_id LIKE 'recovery-cell-%'` + ) + await database.query(`DELETE FROM relay_cell_fences WHERE cell_id LIKE 'recovery-cell-%'`) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id LIKE 'recovery-cell-%'`) + await database.query(`DELETE FROM relay_cell_admission WHERE cell_id LIKE 'recovery-cell-%'`) + await database.query(`DELETE FROM relay_cells WHERE cell_id LIKE 'recovery-cell-%'`) + await database.close() + }) + + async function fixture(replacementCapacity = 20): Promise { + sequence++ + let now = 100 + const suffix = String(sequence) + const cellIncarnation = `11111111-1111-4111-8111-${suffix.padStart(12, '0')}` + const cells = { + source: { + id: `recovery-cell-${suffix}-source`, + url: `https://recovery-${suffix}-source.example.com`, + capacityRequests: 20 + }, + failed: { + id: `recovery-cell-${suffix}-failed`, + url: `https://recovery-${suffix}-failed.example.com`, + capacityRequests: 20 + }, + replacement: { + id: `recovery-cell-${suffix}-replacement`, + url: `https://recovery-${suffix}-replacement.example.com`, + capacityRequests: replacementCapacity + } + } + const store = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells(Object.values(cells), false) + const heartbeat = async (cell: { id: string; url: string }): Promise => { + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation, + startedAt: 50, + ready: true, + observedRequests: 0 + }) + } + for (const cell of Object.values(cells)) await heartbeat(cell) + const identity = { + userId: `recovery-user-${suffix}`, + relayHostId: `recoveryhost${suffix.padStart(4, '0')}` + } + const sourceControlId = `control:${cells.source.id}:1` + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + cells.source.id, + 1, + 90_100, + now, + 1, + 0, + 0, + 0, + 0, + 0 + ] + ) + await database.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?)`, + [ + identity.userId, + identity.relayHostId, + sourceControlId, + 'control', + cells.source.id, + 1, + 90_100, + now + ] + ) + await database.query( + `UPDATE relay_cells SET reserved_requests = 1 WHERE cell_id = ?`, + [cells.source.id] + ) + await store.setCellEnabled(cells.source.id, false) + const migration = await store.startEvacuation(identity, cells.failed.id) + const targetControlId = await store.activateControl(identity, { + cellId: cells.failed.id, + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: cells.failed.id, + assignmentEpoch: migration.assignmentEpoch + }) + return { + store, + identity, + migration, + sourceControlId, + targetControlId, + cellIncarnation, + cells, + advancePastHeartbeat: () => { + now += 45_001 + }, + advance: (milliseconds) => { + now += milliseconds + }, + heartbeat + } + } + + async function completeSourceFence(input: RecoveryFixture): Promise { + const attemptId = input.cellIncarnation + const invocationId = input.cellIncarnation + const requestReason = `relay-recovery-test/${attemptId}` + const evidence = { + attemptId, + environment: 'production' as const, + cellId: input.cells.source.id, + cellIncarnation: input.cellIncarnation, + migName: input.cells.source.id, + instanceGroup: `https://compute.example/instanceGroups/${input.cells.source.id}`, + generationIdentity: `https://compute.example/instanceTemplates/${input.cells.source.id}`, + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + planObjectName: `relay-fence-plans/${attemptId}.tfplan`, + planObjectGeneration: '1', + varFileSha256: 'c'.repeat(64), + terraformStateLineage: input.cellIncarnation, + terraformStateSerial: 1, + terraformStateObjectGeneration: '1', + terraformStateObjectSha256: 'd'.repeat(64), + requestReason + } + await input.store.prepareCellFenceAttempt(evidence) + await input.store.bindCellFencePlanGeneration(evidence, evidence.planObjectGeneration) + await input.store.startCellFenceApply( + evidence, + invocationId, + `${requestReason}/${invocationId}` + ) + await input.store.recordCellFenceOperation( + evidence, + invocationId, + `${requestReason}/${invocationId}`, + `operation-${input.cells.source.id}` + ) + await input.store.attestCellFenceAttempt( + evidence, + `operation-${input.cells.source.id}` + ) + } + + it('reaps an abandoned registered migration onto its healthy target', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advance( + ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + ) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + expect( + await database.query( + `SELECT assignment.cell_id, assignment.assignment_epoch, + migration.completed_at, migration.aborted_at + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + AND migration.assignment_epoch = assignment.assignment_epoch + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([ + { + cell_id: input.cells.failed.id, + assignment_epoch: '2', + completed_at: String( + 100 + ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + ), + aborted_at: null + } + ]) + }) + + it('reaps immediately after the retired source has a durable completed fence', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await completeSourceFence(input) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + expect( + await database.query( + `SELECT assignment.cell_id, assignment.assignment_epoch, + migration.completed_at, migration.aborted_at + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + AND migration.assignment_epoch = assignment.assignment_epoch + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([ + { + cell_id: input.cells.failed.id, + assignment_epoch: '2', + completed_at: String(100 + 45_001 + ASSIGNMENT_LIMITS.migrationLeaseMs + 1), + aborted_at: null + } + ]) + }) + + it('reaps immediately after the retired source has an adopted legacy fence', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await input.store.adoptLegacyCellFence( + input.cells.source.id, + input.cellIncarnation + ) + await input.store.commitLegacyCellFenceAdoption( + input.cells.source.id, + input.cellIncarnation + ) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + + expect( + ( + await readRegisteredMigrationInventory( + database, + 100 + 45_001 + ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + ) + ).abandoned + ).toBe(1) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + expect( + await database.query( + `SELECT assignment.cell_id, assignment.assignment_epoch, + migration.completed_at, migration.aborted_at + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + AND migration.assignment_epoch = assignment.assignment_epoch + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([ + { + cell_id: input.cells.failed.id, + assignment_epoch: '2', + completed_at: String(100 + 45_001 + ASSIGNMENT_LIMITS.migrationLeaseMs + 1), + aborted_at: null + } + ]) + }) + + it('does not reap after temporary legacy adoption without durable commit', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await input.store.adoptLegacyCellFence( + input.cells.source.id, + input.cellIncarnation + ) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + + expect( + ( + await readRegisteredMigrationInventory( + database, + 100 + 45_001 + ASSIGNMENT_LIMITS.migrationLeaseMs + 1 + ) + ).abandoned + ).toBe(0) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(0) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('invalidates an adopted legacy fence when the source heartbeats', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await input.store.adoptLegacyCellFence( + input.cells.source.id, + input.cellIncarnation + ) + await input.store.commitLegacyCellFenceAdoption( + input.cells.source.id, + input.cellIncarnation + ) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + await input.heartbeat(input.cells.source) + + expect( + await database.query( + `SELECT cell_id FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + [input.cells.source.id] + ) + ).toEqual([]) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(0) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('invalidates an adopted legacy fence when the source restarts', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await input.store.adoptLegacyCellFence( + input.cells.source.id, + input.cellIncarnation + ) + await input.store.commitLegacyCellFenceAdoption( + input.cells.source.id, + input.cellIncarnation + ) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + await input.store.recordCellHeartbeat({ + cellId: input.cells.source.id, + cellUrl: input.cells.source.url, + cellIncarnation: '99999999-9999-4999-8999-999999999999', + startedAt: 51, + ready: true, + observedRequests: 0 + }) + + expect( + await database.query( + `SELECT cell_id FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + [input.cells.source.id] + ) + ).toEqual([]) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(0) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('serializes legacy adoption with a returning source heartbeat', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await input.store.adoptLegacyCellFence( + input.cells.source.id, + input.cellIncarnation + ) + + const [commit, heartbeatResult] = await Promise.allSettled([ + input.store.commitLegacyCellFenceAdoption( + input.cells.source.id, + input.cellIncarnation + ), + input.heartbeat(input.cells.source) + ]) + expect(heartbeatResult.status).toBe('fulfilled') + expect(['fulfilled', 'rejected']).toContain(commit.status) + expect( + await database.query( + `SELECT cell_id FROM relay_cell_legacy_fence_adoptions WHERE cell_id = ?`, + [input.cells.source.id] + ) + ).toEqual([]) + input.advance( + ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + ) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('restores the 24-hour guard when the fenced source heartbeats again', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await completeSourceFence(input) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + await input.heartbeat(input.cells.source) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(0) + expect( + await database.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([{ completed_at: null, aborted_at: null }]) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('restores the 24-hour guard when the fenced source restarts', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await completeSourceFence(input) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + await input.store.recordCellHeartbeat({ + cellId: input.cells.source.id, + cellUrl: input.cells.source.url, + cellIncarnation: '99999999-9999-4999-8999-999999999999', + startedAt: 51, + ready: true, + observedRequests: 0 + }) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(0) + expect( + await database.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([{ completed_at: null, aborted_at: null }]) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('serializes fenced cleanup with a returning source heartbeat', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advancePastHeartbeat() + await completeSourceFence(input) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + + const [cleanup, heartbeatResult] = await Promise.allSettled([ + input.store.abortExpiredEvacuations(), + input.heartbeat(input.cells.source) + ]) + expect(cleanup.status).toBe('fulfilled') + expect(heartbeatResult.status).toBe('fulfilled') + if (cleanup.status !== 'fulfilled') throw cleanup.reason + if (heartbeatResult.status !== 'fulfilled') throw heartbeatResult.reason + expect([0, 1]).toContain(cleanup.value) + const migrations = await database.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + if (cleanup.value === 1) { + expect(migrations[0]?.completed_at).not.toBeNull() + } else { + expect(migrations).toEqual([{ completed_at: null, aborted_at: null }]) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + } + }) + + it('keeps a freshly retired source protected without a durable fence', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + input.advance(ASSIGNMENT_LIMITS.migrationLeaseMs + 1) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(0) + expect( + await database.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([{ completed_at: null, aborted_at: null }]) + input.advance(STRANDED_MIGRATION_ABANDON_MS) + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + }) + + it('rolls an abandoned disabled target back to its source', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advance( + ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + ) + + await expect(input.store.abortExpiredEvacuations()).resolves.toBe(1) + expect( + await database.query( + `SELECT assignment.cell_id, assignment.assignment_epoch, + migration.completed_at, migration.aborted_at + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + WHERE assignment.user_id = ? AND assignment.relay_host_id = ? + ORDER BY migration.assignment_epoch`, + [input.identity.userId, input.identity.relayHostId] + ) + ).toEqual([ + { + cell_id: input.cells.source.id, + assignment_epoch: '3', + completed_at: null, + aborted_at: String( + 100 + ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + ) + } + ]) + }) + + it('serializes cleanup against supersession without reporting a false target', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + await input.store.releaseActivity(input.identity, input.targetControlId) + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advance( + ASSIGNMENT_LIMITS.migrationLeaseMs + STRANDED_MIGRATION_ABANDON_MS + 1 + ) + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence(input.cells.failed.id, input.cellIncarnation) + + const [cleanup, supersession] = await Promise.allSettled([ + input.store.abortExpiredEvacuations(), + input.store.supersedeRegisteredCellEvacuations( + input.cells.source.id, + input.cells.failed.id, + input.cells.replacement.id, + 100 + ) + ]) + expect(cleanup.status).toBe('fulfilled') + if (cleanup.status !== 'fulfilled') throw cleanup.reason + if (supersession.status === 'fulfilled') { + expect([0, 1]).toContain(supersession.value) + const assignment = await database.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ?`, + [input.identity.userId] + ) + expect(assignment).toEqual([{ + cell_id: + supersession.value === 1 + ? input.cells.replacement.id + : input.cells.source.id + }]) + if (supersession.value === 0) expect(cleanup.value).toBeGreaterThanOrEqual(1) + } else { + expect(cleanup.value).toBeGreaterThanOrEqual(1) + expect(supersession.reason).toMatchObject({ message: 'migration_already_superseded' }) + expect( + await database.query( + `SELECT cell_id FROM relay_assignments WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ cell_id: input.cells.source.id }]) + } + }) + + it('completes a registered migration from a dead source idempotently', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.failed) + await input.store.attestCellFence( + input.cells.source.id, + input.cellIncarnation + ) + const completion = { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + targetCellId: input.cells.failed.id + } + + await expect( + input.store.cellEvacuationStatus( + input.cells.source.id, + input.cells.failed.id, + true + ) + ).resolves.toMatchObject({ inProgress: 0, completed: 1 }) + await expect( + input.store.completeEvacuationFromDeadSource(input.identity, completion) + ).resolves.toMatchObject({ changed: false }) + expect(await reservations(input.cells)).toEqual({ source: 0, failed: 1, replacement: 0 }) + }) + + it('fails dead-source completion without a stale source heartbeat', async () => { + const input = await fixture() + await input.store.releaseActivity(input.identity, input.sourceControlId) + + await expect( + input.store.completeEvacuationFromDeadSource(input.identity, { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + targetCellId: input.cells.failed.id + }) + ).rejects.toThrow('cell_fence_attestation_missing') + expect(await reservations(input.cells)).toEqual({ source: 0, failed: 2, replacement: 0 }) + }) + + it('supersedes a registered migration with exact accounting and idempotency', async () => { + const input = await fixture() + await database.query( + `INSERT INTO relay_control_connection_reservations + (reservation_id, idempotency_key, user_id, relay_host_id, + assignment_epoch, cell_id, state, inclusion_watermark, + claim_activity_id, created_at, timeout_at, claimed_at, released_at, + updated_at) + VALUES (?, ?, ?, ?, ?, ?, 'late-arrival-debt', NULL, NULL, 100, 100, NULL, NULL, 100)`, + [ + `superseded-${input.identity.userId}`, + `superseded-${input.identity.userId}`, + input.identity.userId, + input.identity.relayHostId, + input.migration.assignmentEpoch, + input.cells.failed.id + ] + ) + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence( + input.cells.failed.id, + input.cellIncarnation + ) + const supersession = { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + } + + const first = await input.store.supersedeRegisteredEvacuation( + input.identity, + supersession + ) + const retry = await input.store.supersedeRegisteredEvacuation( + input.identity, + supersession + ) + expect(retry).toEqual(first) + expect(first).toMatchObject({ previousEpoch: 2, assignmentEpoch: 3 }) + expect( + await database.query( + `SELECT state FROM relay_control_connection_reservations + WHERE reservation_id = ?`, + [`superseded-${input.identity.userId}`] + ) + ).toEqual([{ state: 'released' }]) + expect(await reservations(input.cells)).toEqual({ source: 1, failed: 0, replacement: 2 }) + expect( + await database.query( + `SELECT assignment_epoch, completed_at, aborted_at + FROM relay_assignment_migrations WHERE user_id = ? + ORDER BY assignment_epoch`, + [input.identity.userId] + ) + ).toEqual([ + { assignment_epoch: '2', completed_at: null, aborted_at: '45101' }, + { assignment_epoch: '3', completed_at: null, aborted_at: null } + ]) + }) + + it('reconciles durable cell accounting before aggregate supersession', async () => { + const input = await fixture() + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence( + input.cells.failed.id, + input.cellIncarnation + ) + await database.query( + `UPDATE relay_cells + SET reserved_requests = CASE WHEN cell_id = ? THEN 1 ELSE 0 END + WHERE cell_id IN (?, ?)`, + [input.cells.replacement.id, input.cells.source.id, input.cells.replacement.id] + ) + + await expect( + input.store.supersedeRegisteredCellEvacuations( + input.cells.source.id, + input.cells.failed.id, + input.cells.replacement.id, + 100 + ) + ).resolves.toBe(1) + expect(await reservations(input.cells)).toEqual({ + source: 1, + failed: 0, + replacement: 2 + }) + }) + + it('rebuilds an expired registered migration lease before supersession', async () => { + const input = await fixture() + await pinMigration(input, input.migration.assignmentEpoch) + await clearActivities(input) + await database.query( + `UPDATE relay_assignment_migrations SET expires_at = 0 + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [input.identity.userId, input.identity.relayHostId, input.migration.assignmentEpoch] + ) + await fenceFailedCell(input) + + await expect(supersedeAggregate(input)).resolves.toBe(1) + expect(await reservations(input.cells)).toEqual({ + source: 0, + failed: 0, + replacement: 1 + }) + expect( + await database.query( + `SELECT assignment_epoch, source_request_units, target_reserved_units, aborted_at + FROM relay_assignment_migrations WHERE user_id = ? ORDER BY assignment_epoch`, + [input.identity.userId] + ) + ).toEqual([ + { + assignment_epoch: '2', + source_request_units: '1', + target_reserved_units: '2', + aborted_at: '45101' + }, + { + assignment_epoch: '3', + source_request_units: '0', + target_reserved_units: '1', + aborted_at: null + } + ]) + }) + + it('retires an obsolete row without discarding replacement activity', async () => { + const input = await fixture() + await pinMigration(input, input.migration.assignmentEpoch) + await fenceFailedCell(input) + await input.store.supersedeRegisteredEvacuation(input.identity, { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + }) + await database.query( + `UPDATE relay_assignment_migrations SET aborted_at = NULL + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [input.identity.userId, input.identity.relayHostId, input.migration.assignmentEpoch] + ) + const activityBefore = await assignmentActivities(input) + + await expect(supersedeAggregate(input)).resolves.toBe(1) + expect(await assignmentActivities(input)).toEqual(activityBefore) + expect(await reservations(input.cells)).toEqual({ + source: 1, + failed: 0, + replacement: 2 + }) + expect( + await database.query( + `SELECT assignment_epoch, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? ORDER BY assignment_epoch`, + [input.identity.userId] + ) + ).toEqual([ + { assignment_epoch: '2', aborted_at: '45101' }, + { assignment_epoch: '3', aborted_at: null } + ]) + }) + + it('anchors a newer dormant failed-cell epoch and preserves drain evidence', async () => { + const input = await fixture() + const drainAttemptId = await pinMigration(input, input.migration.assignmentEpoch) + await clearActivities(input) + await database.query( + `UPDATE relay_assignments SET assignment_epoch = 4 + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + await database.query( + `UPDATE relay_assignment_migrations SET expires_at = 0 + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [input.identity.userId, input.identity.relayHostId, input.migration.assignmentEpoch] + ) + await fenceFailedCell(input) + + await expect(supersedeAggregate(input)).resolves.toBe(1) + expect(await reservations(input.cells)).toEqual({ + source: 0, + failed: 0, + replacement: 1 + }) + expect( + await database.query( + `SELECT assignment_epoch, source_request_units, target_reserved_units, aborted_at + FROM relay_assignment_migrations WHERE user_id = ? ORDER BY assignment_epoch`, + [input.identity.userId] + ) + ).toEqual([ + { + assignment_epoch: '2', + source_request_units: '1', + target_reserved_units: '2', + aborted_at: '45101' + }, + { + assignment_epoch: '4', + source_request_units: '0', + target_reserved_units: '1', + aborted_at: '45101' + }, + { + assignment_epoch: '5', + source_request_units: '0', + target_reserved_units: '1', + aborted_at: null + } + ]) + expect( + await database.query( + `SELECT assignment_epoch, drain_attempt_id, source_request_units, + target_reserved_units + FROM relay_post_drain_migration_pins + WHERE user_id = ? ORDER BY assignment_epoch`, + [input.identity.userId] + ) + ).toEqual([ + { + assignment_epoch: '2', + drain_attempt_id: drainAttemptId, + source_request_units: '1', + target_reserved_units: '2' + }, + { + assignment_epoch: '4', + drain_attempt_id: drainAttemptId, + source_request_units: '0', + target_reserved_units: '1' + }, + { + assignment_epoch: '5', + drain_attempt_id: drainAttemptId, + source_request_units: '0', + target_reserved_units: '1' + } + ]) + }) + + it('uses an existing current-epoch migration once when retiring an older row', async () => { + const input = await fixture() + await pinMigration(input, input.migration.assignmentEpoch) + await clearActivities(input) + await database.query( + `UPDATE relay_assignments SET assignment_epoch = 4 + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + await database.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, + previous_epoch, assignment_epoch, source_request_units, + target_reserved_units, expires_at, target_registered_at, + completed_at, aborted_at, created_at, updated_at) + VALUES (?, ?, ?, ?, 2, 4, 0, 1, 0, 100, NULL, NULL, 100, 100)`, + [ + input.identity.userId, + input.identity.relayHostId, + input.cells.source.id, + input.cells.failed.id + ] + ) + await database.query( + `INSERT INTO relay_assignment_migration_incarnations + (user_id, relay_host_id, assignment_epoch, source_cell_incarnation, + target_cell_incarnation) + VALUES (?, ?, 4, ?, ?)`, + [ + input.identity.userId, + input.identity.relayHostId, + input.cellIncarnation, + input.cellIncarnation + ] + ) + await pinMigration(input, 4) + await fenceFailedCell(input) + + await expect(supersedeAggregate(input)).resolves.toBe(1) + expect( + await database.query( + `SELECT assignment_epoch, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? ORDER BY assignment_epoch`, + [input.identity.userId] + ) + ).toEqual([ + { assignment_epoch: '2', aborted_at: '45101' }, + { assignment_epoch: '4', aborted_at: '45101' }, + { assignment_epoch: '5', aborted_at: null } + ]) + expect(await reservations(input.cells)).toEqual({ + source: 0, + failed: 0, + replacement: 1 + }) + }) + + it('rejects a newer failed-cell epoch with ambiguous activity', async () => { + const input = await fixture() + await database.query( + `UPDATE relay_assignments SET assignment_epoch = 4 + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + await fenceFailedCell(input) + + await expect(supersedeAggregate(input)).rejects.toThrow( + 'migration_activity_topology_mismatch' + ) + expect( + await database.query( + `SELECT assignment_epoch, aborted_at FROM relay_assignment_migrations + WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ assignment_epoch: '2', aborted_at: null }]) + expect(await reservations(input.cells)).toEqual({ + source: 1, + failed: 2, + replacement: 0 + }) + }) + + it('leaves lease repair retryable when supersession later fails', async () => { + const input = await fixture(1) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'migration'`, + [input.identity.userId, input.identity.relayHostId] + ) + await database.query( + `UPDATE relay_assignments SET migration_leases = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + await database.query( + `UPDATE relay_cells SET reserved_requests = reserved_requests - 1 + WHERE cell_id = ?`, + [input.cells.failed.id] + ) + await database.query( + `UPDATE relay_assignment_migrations SET expires_at = 0 + WHERE user_id = ? AND relay_host_id = ? AND assignment_epoch = ?`, + [input.identity.userId, input.identity.relayHostId, input.migration.assignmentEpoch] + ) + await fenceFailedCell(input) + + await expect(supersedeAggregate(input)).rejects.toThrow( + 'relay_capacity_exhausted' + ) + expect( + await database.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND activity_kind = 'migration'`, + [input.identity.userId] + ) + ).toEqual([{ activity_id: 'migration:2' }]) + expect( + await database.query( + `SELECT aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND assignment_epoch = 2`, + [input.identity.userId] + ) + ).toEqual([{ aborted_at: null }]) + + await database.query( + `UPDATE relay_cells SET capacity_requests = 2 WHERE cell_id = ?`, + [input.cells.replacement.id] + ) + await expect(supersedeAggregate(input)).resolves.toBe(1) + expect(await reservations(input.cells)).toEqual({ + source: 1, + failed: 0, + replacement: 2 + }) + }) + + it('serializes concurrent supersession retries without duplicating capacity', async () => { + const input = await fixture() + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence( + input.cells.failed.id, + input.cellIncarnation + ) + const supersession = { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + } + + const results = await Promise.all([ + input.store.supersedeRegisteredEvacuation(input.identity, supersession), + input.store.supersedeRegisteredEvacuation(input.identity, supersession) + ]) + expect(results[0]).toEqual(results[1]) + expect(await reservations(input.cells)).toEqual({ source: 1, failed: 0, replacement: 2 }) + expect( + await database.query( + `SELECT COUNT(*) AS count FROM relay_assignment_migrations WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ count: '2' }]) + }) + + it('fails a competing aggregate supersession instead of reporting the wrong target', async () => { + const input = await fixture() + const alternate = { + id: `recovery-cell-${sequence}-alternate`, + url: `https://recovery-${sequence}-alternate.example.com`, + capacityRequests: 20 + } + await input.store.reconcileCells([...Object.values(input.cells), alternate], false) + await input.heartbeat(alternate) + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.heartbeat(alternate) + await input.store.attestCellFence(input.cells.failed.id, input.cellIncarnation) + + const results = await Promise.allSettled([ + input.store.supersedeRegisteredCellEvacuations( + input.cells.source.id, + input.cells.failed.id, + input.cells.replacement.id, + 100 + ), + input.store.supersedeRegisteredCellEvacuations( + input.cells.source.id, + input.cells.failed.id, + alternate.id, + 100 + ) + ]) + const fulfilled = results.filter( + (result): result is PromiseFulfilledResult => result.status === 'fulfilled' + ) + const rejected = results.filter( + (result): result is PromiseRejectedResult => result.status === 'rejected' + ) + expect(fulfilled).toHaveLength(1) + expect(fulfilled[0]!.value).toBe(1) + expect(rejected).toHaveLength(1) + expect(rejected[0]!.reason).toMatchObject({ message: 'migration_already_superseded' }) + const successor = await database.query( + `SELECT assignment.cell_id, migration.target_cell_id + FROM relay_assignments assignment + JOIN relay_assignment_migrations migration + ON migration.user_id = assignment.user_id + AND migration.relay_host_id = assignment.relay_host_id + AND migration.assignment_epoch = assignment.assignment_epoch + WHERE assignment.user_id = ? AND migration.previous_epoch = ?`, + [input.identity.userId, input.migration.assignmentEpoch] + ) + expect(successor).toHaveLength(1) + expect(successor[0]!.cell_id).toBe(successor[0]!.target_cell_id) + expect([input.cells.replacement.id, alternate.id]).toContain( + successor[0]!.target_cell_id + ) + }) + + it('rolls back every supersession change when replacement capacity is insufficient', async () => { + const input = await fixture(1) + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence( + input.cells.failed.id, + input.cellIncarnation + ) + + await expect( + input.store.supersedeRegisteredEvacuation(input.identity, { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + }) + ).rejects.toThrow('relay_capacity_exhausted') + expect(await reservations(input.cells)).toEqual({ source: 1, failed: 2, replacement: 0 }) + expect( + await database.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ cell_id: input.cells.failed.id, assignment_epoch: '2' }]) + expect( + await database.query( + `SELECT assignment_epoch, aborted_at FROM relay_assignment_migrations + WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ assignment_epoch: '2', aborted_at: null }]) + }) + + it('rolls back supersession when migration request-unit shape drifted', async () => { + const input = await fixture() + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence(input.cells.failed.id, input.cellIncarnation) + await database.query( + `UPDATE relay_assignment_activity_leases + SET request_units = request_units + 1 + WHERE user_id = ? AND activity_kind = 'migration'`, + [input.identity.userId] + ) + await database.query( + `UPDATE relay_cells SET reserved_requests = reserved_requests + 1 WHERE cell_id = ?`, + [input.cells.failed.id] + ) + + await expect( + input.store.supersedeRegisteredEvacuation(input.identity, { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + }) + ).rejects.toThrow('migration_activity_lease_shape_mismatch') + expect(await reservations(input.cells)).toEqual({ source: 1, failed: 3, replacement: 0 }) + expect( + await database.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ cell_id: input.cells.failed.id, assignment_epoch: '2' }]) + }) + + it('rolls back supersession when locked cell reservation accounting drifted', async () => { + const input = await fixture() + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence(input.cells.failed.id, input.cellIncarnation) + await database.query( + `UPDATE relay_cells SET reserved_requests = 1 WHERE cell_id = ?`, + [input.cells.replacement.id] + ) + + await expect( + input.store.supersedeRegisteredEvacuation(input.identity, { + assignmentEpoch: input.migration.assignmentEpoch, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + }) + ).rejects.toThrow('migration_cell_reservation_accounting_mismatch') + expect(await reservations(input.cells)).toEqual({ source: 1, failed: 2, replacement: 1 }) + expect( + await database.query( + `SELECT assignment_epoch, aborted_at FROM relay_assignment_migrations + WHERE user_id = ?`, + [input.identity.userId] + ) + ).toEqual([{ assignment_epoch: '2', aborted_at: null }]) + }) + + it('rejects supersession before incrementing the maximum safe epoch', async () => { + const input = await fixture() + await expect( + input.store.supersedeRegisteredEvacuation(input.identity, { + assignmentEpoch: Number.MAX_SAFE_INTEGER, + sourceCellId: input.cells.source.id, + currentTargetCellId: input.cells.failed.id, + replacementTargetCellId: input.cells.replacement.id + }) + ).rejects.toThrow('assignment_epoch_exhausted') + }) + + async function fenceFailedCell(input: RecoveryFixture): Promise { + await input.store.setCellEnabled(input.cells.failed.id, false) + input.advancePastHeartbeat() + await input.heartbeat(input.cells.source) + await input.heartbeat(input.cells.replacement) + await input.store.attestCellFence( + input.cells.failed.id, + input.cellIncarnation + ) + } + + async function supersedeAggregate(input: RecoveryFixture): Promise { + return await input.store.supersedeRegisteredCellEvacuations( + input.cells.source.id, + input.cells.failed.id, + input.cells.replacement.id, + 100 + ) + } + + async function clearActivities(input: RecoveryFixture): Promise { + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + await database.query( + `UPDATE relay_assignments SET reserved_controls = 0, reserved_splices = 0, + reserved_invites = 0, pending_installs = 0, pending_confirmations = 0, + migration_leases = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [input.identity.userId, input.identity.relayHostId] + ) + await database.query( + `UPDATE relay_cells SET reserved_requests = 0 + WHERE cell_id IN (?, ?, ?)`, + [input.cells.source.id, input.cells.failed.id, input.cells.replacement.id] + ) + } + + async function pinMigration( + input: RecoveryFixture, + assignmentEpoch: number + ): Promise { + const drainAttemptId = `${input.identity.userId}-drain-${assignmentEpoch}` + await database.query( + `INSERT INTO relay_post_drain_migration_pins + (user_id, relay_host_id, assignment_epoch, drain_attempt_id, + source_cell_id, source_cell_incarnation, target_cell_id, + target_cell_incarnation, source_request_units, target_reserved_units, + pinned_at) + SELECT migration.user_id, migration.relay_host_id, + migration.assignment_epoch, ?, migration.source_cell_id, + incarnation.source_cell_incarnation, migration.target_cell_id, + incarnation.target_cell_incarnation, migration.source_request_units, + migration.target_reserved_units, 100 + FROM relay_assignment_migrations migration + JOIN relay_assignment_migration_incarnations incarnation + ON incarnation.user_id = migration.user_id + AND incarnation.relay_host_id = migration.relay_host_id + AND incarnation.assignment_epoch = migration.assignment_epoch + WHERE migration.user_id = ? AND migration.relay_host_id = ? + AND migration.assignment_epoch = ?`, + [ + drainAttemptId, + input.identity.userId, + input.identity.relayHostId, + assignmentEpoch + ] + ) + return drainAttemptId + } + + async function assignmentActivities(input: RecoveryFixture): Promise { + return await database.query( + `SELECT activity_id, activity_kind, cell_id, request_units + FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY activity_id`, + [input.identity.userId, input.identity.relayHostId] + ) + } + + async function reservations(cells: RecoveryFixture['cells']): Promise> { + const rows = await database.query( + `SELECT cell_id, reserved_requests FROM relay_cells + WHERE cell_id IN (?, ?, ?) ORDER BY cell_id`, + [cells.source.id, cells.failed.id, cells.replacement.id] + ) + return Object.fromEntries( + rows.map((row) => { + const cellId = String(row.cell_id) + const name = cellId.endsWith('-source') + ? 'source' + : cellId.endsWith('-failed') + ? 'failed' + : 'replacement' + return [name, Number(row.reserved_requests)] + }) + ) + } +}) diff --git a/cloud/apps/relay/src/observed-relay-database.ts b/cloud/apps/relay/src/observed-relay-database.ts new file mode 100644 index 00000000000..4720cae6c98 --- /dev/null +++ b/cloud/apps/relay/src/observed-relay-database.ts @@ -0,0 +1,46 @@ +import type { + RelayDatabase, + RelayLockOptions, + RelayTransactionOptions, + SqlRow +} from './database.js' +import { timedRelayOperation, type RelayRuntimeObserver } from './relay-observability.js' + +export function observeRelayDatabase( + database: RelayDatabase, + observer: RelayRuntimeObserver +): RelayDatabase { + const query = (sql: string, params?: unknown[]): Promise => + timedRelayOperation( + () => database.query(sql, params), + (durationMs, success) => observer.recordSql(durationMs, success) + ) + const queryLocked = ( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise => + timedRelayOperation( + () => database.queryLocked(sql, params, options), + (durationMs, success) => observer.recordSql(durationMs, success), + (error) => + // NOWAIT contention is an intentional sweep deferral, not a SQL-health failure. + options?.failIfUnavailable === true && + error instanceof Error && + error.message === 'database_lock_unavailable' + ) + return { + dialect: database.dialect, + query, + queryLocked, + transaction: async ( + operation: (transaction: RelayDatabase) => Promise, + options?: RelayTransactionOptions + ): Promise => + await database.transaction( + async (transaction) => await operation(observeRelayDatabase(transaction, observer)), + options + ), + close: async () => await database.close() + } +} diff --git a/cloud/apps/relay/src/postgres-drain-send-locking.test.ts b/cloud/apps/relay/src/postgres-drain-send-locking.test.ts new file mode 100644 index 00000000000..024bf46db5e --- /dev/null +++ b/cloud/apps/relay/src/postgres-drain-send-locking.test.ts @@ -0,0 +1,215 @@ +import { afterAll, afterEach, beforeAll, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const source = { + id: 'drain-send-lock-source', + url: 'https://drain-send-lock-source.example.com', + capacityRequests: 100 +} +const target = { + id: 'drain-send-lock-target', + url: 'https://drain-send-lock-target.example.com', + capacityRequests: 100 +} +const identity = { + userId: 'drain-send-lock-user', + relayHostId: 'drainsendlock01' +} +const sourceIncarnation = '11111111-1111-4111-8111-111111111111' +const targetIncarnation = '22222222-2222-4222-8222-222222222222' +const attemptId = '33333333-3333-4333-8333-333333333333' + +describePostgres('PostgreSQL drain-send locking', () => { + let database: RelayDatabase + + beforeAll(async () => { + database = await openRelayDatabase({ databaseUrl, dataDir: '' }) + }) + + afterAll(async () => await database.close()) + + afterEach(async () => { + await database.query(`DELETE FROM relay_post_drain_migration_pins WHERE user_id = ?`, [ + identity.userId + ]) + await database.query(`DELETE FROM relay_cell_drain_attempt_states WHERE attempt_id = ?`, [ + attemptId + ]) + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, + [identity.userId] + ) + await database.query(`DELETE FROM relay_migration_leases WHERE user_id = ?`, [ + identity.userId + ]) + await database.query(`DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, [ + identity.userId + ]) + await database.query( + `DELETE FROM relay_assignment_migration_incarnations WHERE user_id = ?`, + [identity.userId] + ) + await database.query(`DELETE FROM relay_assignment_migrations WHERE user_id = ?`, [ + identity.userId + ]) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [identity.userId]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id IN (?, ?)`, [ + source.id, + target.id + ]) + await database.query(`DELETE FROM relay_cell_admission WHERE cell_id IN (?, ?)`, [ + source.id, + target.id + ]) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + source.id, + target.id + ]) + }) + + it('locks active migrations without locking the nullable incarnation lookup', async () => { + const now = 1_700_000_000_000 + const store = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells([source, target]) + await store.recordCellHeartbeat({ + cellId: source.id, + cellUrl: source.url, + cellIncarnation: sourceIncarnation, + startedAt: now - 1, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: target.id, + cellUrl: target.url, + cellIncarnation: targetIncarnation, + startedAt: now - 1, + ready: true, + observedRequests: 0 + }) + await store.assign(identity) + await store.setCellEnabled(source.id, false) + await store.startEvacuation(identity, target.id) + await store.prepareCellDrainAttempt({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation, + traceValue: '44444444-4444-4444-8444-444444444444', + plannedGraceMs: 120_000 + }) + await expect( + store.prepareCellDrainRecovery({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation + }) + ).resolves.toMatchObject({ + shouldSend: false, + preparedAttempt: { attemptId, state: 'prepared' } + }) + + await expect( + store.beginCellDrainSend({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation + }) + ).resolves.toMatchObject({ state: 'send-may-have-started', shouldSend: true }) + await expect( + store.beginCellDrainSend({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation + }) + ).resolves.toMatchObject({ state: 'send-may-have-started', shouldSend: false }) + await expect( + store.prepareCellDrainRecovery({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation + }) + ).rejects.toThrow('drain_application_receipt_missing') + await expect( + database.query(`SELECT drain_attempt_id FROM relay_post_drain_migration_pins`) + ).resolves.toEqual([{ drain_attempt_id: attemptId }]) + }) + + it('restores an expired registered migration lease with PostgreSQL locks', async () => { + let now = 1_700_000_000_000 + const startedAt = now - 1 + const store = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells([source, target]) + await store.recordCellHeartbeat({ + cellId: source.id, + cellUrl: source.url, + cellIncarnation: sourceIncarnation, + startedAt, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: target.id, + cellUrl: target.url, + cellIncarnation: targetIncarnation, + startedAt, + ready: true, + observedRequests: 0 + }) + await store.assign(identity) + await store.setCellEnabled(source.id, false) + const migration = await store.startEvacuation(identity, target.id) + await store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: migration.assignmentEpoch + }) + await store.prepareCellDrainAttempt({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation, + traceValue: '44444444-4444-4444-8444-444444444444', + plannedGraceMs: 120_000 + }) + + now = migration.expiresAt + 1 + expect(await store.releaseExpiredActivityLeases()).toBeGreaterThan(0) + await store.recordCellHeartbeat({ + cellId: source.id, + cellUrl: source.url, + cellIncarnation: sourceIncarnation, + startedAt, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: target.id, + cellUrl: target.url, + cellIncarnation: targetIncarnation, + startedAt, + ready: true, + observedRequests: 0 + }) + await expect( + store.beginCellDrainSend({ + attemptId, + cellId: source.id, + cellIncarnation: sourceIncarnation + }) + ).resolves.toMatchObject({ state: 'send-may-have-started', shouldSend: true }) + await expect( + database.query( + `SELECT activity_kind, cell_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).resolves.toContainEqual({ activity_kind: 'migration', cell_id: target.id }) + }) +}) diff --git a/cloud/apps/relay/src/postgres-idle-client-error.test.ts b/cloud/apps/relay/src/postgres-idle-client-error.test.ts new file mode 100644 index 00000000000..50a7bb48ab5 --- /dev/null +++ b/cloud/apps/relay/src/postgres-idle-client-error.test.ts @@ -0,0 +1,22 @@ +import { describe, expect, it, vi } from 'vitest' +import { absorbPostgresIdleClientErrors } from './database.js' + +describe('PostgreSQL idle-client failure handling', () => { + it('absorbs the pool error without logging connection details', () => { + let listener: ((error: Error) => void) | undefined + const pool = { + on: vi.fn((_event: string, value: (error: Error) => void) => { + listener = value + return pool + }) + } + const warning = vi.spyOn(console, 'warn').mockImplementation(() => {}) + + absorbPostgresIdleClientErrors(pool as never) + expect(() => listener?.(new Error('postgres://user:secret@database'))).not.toThrow() + expect(warning).toHaveBeenCalledWith('[orca-relay] idle PostgreSQL client failed') + expect(JSON.stringify(warning.mock.calls)).not.toContain('secret') + + warning.mockRestore() + }) +}) diff --git a/cloud/apps/relay/src/postgres-maintenance-sweep-plans.test.ts b/cloud/apps/relay/src/postgres-maintenance-sweep-plans.test.ts new file mode 100644 index 00000000000..a8d0e1e3bf9 --- /dev/null +++ b/cloud/apps/relay/src/postgres-maintenance-sweep-plans.test.ts @@ -0,0 +1,61 @@ +import pg from 'pg' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +// Inactive bases outlive their sweep and are never pruned, so the table only +// grows; production reached ~2.24M rows of which 3 were active. Enough rows +// here that a sequential scan is the cheaper plan without the index. +const INACTIVE_ROWS = 20_000 +const OWNED_PREFIX = 'sweep-plan-' + +describePostgres('PostgreSQL maintenance sweep plans', () => { + let database: RelayDatabase + let client: pg.Client + + beforeAll(async () => { + database = await openRelayDatabase({ databaseUrl, dataDir: '' }) + // Only ever touch this suite's own rows: the database is shared with the + // other PostgreSQL suites running in parallel. + await database.query( + `DELETE FROM relay_connection_bases WHERE basis_conn_id LIKE ?`, + [`${OWNED_PREFIX}%`] + ) + await database.query( + `INSERT INTO relay_connection_bases + (basis_conn_id, user_id, relay_host_id, relay_device_id, + owning_control_generation, credential_kind, deadline, active, created_at) + SELECT '${OWNED_PREFIX}' || generation, 'sweep-plan-user', 'sweepplan01', + 'sweep-plan-device', 1, 'invite', 1000, 0, 1000 + FROM generate_series(1, ${INACTIVE_ROWS}) AS generation` + ) + await database.query(`ANALYZE relay_connection_bases`) + client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + }) + + afterAll(async () => { + await client.end() + await database.query( + `DELETE FROM relay_connection_bases WHERE basis_conn_id LIKE ?`, + [`${OWNED_PREFIX}%`] + ) + await database.close() + }) + + // Why: a seq scan here held the maintenance transaction open long enough to + // time out assignment lock waits fleet-wide (2026-08-05 incident). + it('matches expired active bases by index instead of scanning the table', async () => { + const result = await client.query( + `EXPLAIN UPDATE relay_connection_bases SET active = $1 + WHERE active = $2 AND deadline <= $3`, + [0, 1, 2000] + ) + const plan = result.rows.map((row) => String(row['QUERY PLAN'])).join('\n') + + expect(plan).not.toMatch(/Seq Scan on relay_connection_bases/) + expect(plan).toMatch(/relay_connection_bases_active_deadline/) + }) +}) diff --git a/cloud/apps/relay/src/postgres-pool-pressure.test.ts b/cloud/apps/relay/src/postgres-pool-pressure.test.ts new file mode 100644 index 00000000000..2e020bf43fb --- /dev/null +++ b/cloud/apps/relay/src/postgres-pool-pressure.test.ts @@ -0,0 +1,48 @@ +import { describe, expect, it, vi } from 'vitest' +import { PostgresPoolPressure } from './postgres-pool-pressure.js' + +describe('PostgreSQL pool pressure', () => { + it('reports current waiters and interval high-water marks', async () => { + let now = 1_000 + let resolveConnection!: (client: unknown) => void + const connection = new Promise((resolve) => { + resolveConnection = resolve + }) + const pool = { + totalCount: 3, + idleCount: 0, + waitingCount: 0, + connect: vi.fn(() => { + pool.waitingCount++ + return connection + }) + } + const pressure = new PostgresPoolPressure(pool as never, () => now) + const pending = pressure.connect() + now = 1_750 + + expect(pressure.consumeCounts()).toMatchObject({ + databasePoolTotal: 3, + databasePoolIdle: 0, + databasePoolWaiting: 1, + databasePoolWaitersMax: 1, + databasePoolOldestWaitMs: 750, + databasePoolWaitMsMax: 750 + }) + + now = 2_250 + pool.waitingCount-- + resolveConnection({ query: vi.fn(), release: vi.fn() }) + await pending + expect(pressure.consumeCounts()).toMatchObject({ + databasePoolWaiting: 0, + databasePoolWaitersMax: 1, + databasePoolOldestWaitMs: 0, + databasePoolWaitMsMax: 1_250 + }) + expect(pressure.consumeCounts()).toMatchObject({ + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + }) + }) +}) diff --git a/cloud/apps/relay/src/postgres-pool-pressure.ts b/cloud/apps/relay/src/postgres-pool-pressure.ts new file mode 100644 index 00000000000..e18bf2bdd42 --- /dev/null +++ b/cloud/apps/relay/src/postgres-pool-pressure.ts @@ -0,0 +1,93 @@ +import type pg from 'pg' + +export type PostgresPoolPressureCounts = { + databasePoolTotal: number + databasePoolIdle: number + databasePoolWaiting: number + databasePoolWaitersMax: number + databasePoolOldestWaitMs: number + databasePoolWaitMsMax: number +} + +const emptyCounts = (): PostgresPoolPressureCounts => ({ + databasePoolTotal: 0, + databasePoolIdle: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolOldestWaitMs: 0, + databasePoolWaitMsMax: 0 +}) + +export class PostgresPoolPressure { + private readonly waiters = new Map() + private waitersMax = 0 + private waitMsMax = 0 + private lastConsumed = emptyCounts() + + constructor( + private readonly pool: pg.Pool, + private readonly now: () => number = Date.now + ) {} + + async connect(): Promise { + const waitingBefore = this.pool.waitingCount + const connection = this.pool.connect() + if (this.pool.waitingCount <= waitingBefore) return await connection + + const waiter = Symbol() + const startedAt = this.now() + this.waiters.set(waiter, startedAt) + this.waitersMax = Math.max(this.waitersMax, this.waiters.size) + try { + return await connection + } finally { + this.waitMsMax = Math.max(this.waitMsMax, this.now() - startedAt) + this.waiters.delete(waiter) + } + } + + consumeCounts(): PostgresPoolPressureCounts { + const counts = this.readCounts() + this.lastConsumed = counts + this.waitersMax = this.waiters.size + this.waitMsMax = counts.databasePoolOldestWaitMs + return counts + } + + peekCounts(): PostgresPoolPressureCounts { + const current = this.readCounts() + return { + ...current, + databasePoolWaitersMax: Math.max( + current.databasePoolWaitersMax, + this.lastConsumed.databasePoolWaitersMax + ), + databasePoolOldestWaitMs: Math.max( + current.databasePoolOldestWaitMs, + this.lastConsumed.databasePoolOldestWaitMs + ), + databasePoolWaitMsMax: Math.max( + current.databasePoolWaitMsMax, + this.lastConsumed.databasePoolWaitMsMax + ) + } + } + + private readCounts(): PostgresPoolPressureCounts { + const now = this.now() + const oldestWaitMs = + this.waiters.size === 0 ? 0 : Math.max(0, now - Math.min(...this.waiters.values())) + return { + databasePoolTotal: this.pool.totalCount, + databasePoolIdle: this.pool.idleCount, + databasePoolWaiting: this.waiters.size, + databasePoolWaitersMax: Math.max(this.waitersMax, this.waiters.size), + databasePoolOldestWaitMs: oldestWaitMs, + databasePoolWaitMsMax: Math.max(this.waitMsMax, oldestWaitMs) + } + } +} + +export function emptyPostgresPoolPressureCounts(): PostgresPoolPressureCounts { + return emptyCounts() +} diff --git a/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts b/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts new file mode 100644 index 00000000000..5208f137ae8 --- /dev/null +++ b/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts @@ -0,0 +1,55 @@ +import pg from 'pg' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { openRelayDatabase, type RelayDatabase } from './database.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const schema = 'relay_schema_concurrency_test' + +describePostgres('PostgreSQL schema concurrency', () => { + let scopedUrl = '' + + beforeAll(async () => { + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + await client.query(`CREATE SCHEMA ${schema}`) + } finally { + await client.end() + } + const url = new URL(databaseUrl!) + url.searchParams.set('options', `-c search_path=${schema}`) + scopedUrl = url.toString() + }) + + afterAll(async () => { + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + } finally { + await client.end() + } + }) + + it('opens five directors when one new table is absent', async () => { + const initial = await openRelayDatabase({ databaseUrl: scopedUrl, dataDir: '' }) + await initial.query(`DROP TABLE relay_cell_legacy_fence_adoptions`) + await initial.close() + + const results = await Promise.allSettled( + Array.from({ length: 5 }, async (): Promise => + await openRelayDatabase({ databaseUrl: scopedUrl, dataDir: '' }) + ) + ) + const databases = results.flatMap((result) => + result.status === 'fulfilled' ? [result.value] : [] + ) + try { + expect(results.every((result) => result.status === 'fulfilled')).toBe(true) + } finally { + await Promise.all(databases.map(async (database) => await database.close())) + } + }) +}) diff --git a/cloud/apps/relay/src/postgres-schema-startup.ts b/cloud/apps/relay/src/postgres-schema-startup.ts new file mode 100644 index 00000000000..5e6260ad2fb --- /dev/null +++ b/cloud/apps/relay/src/postgres-schema-startup.ts @@ -0,0 +1,83 @@ +const RETRYABLE_SCHEMA_CODES = new Set(['55P03', '57014']) +const DEFAULT_RETRY_DEADLINE_MS = 30_000 +const RETRY_BASE_DELAY_MS = 250 +const RETRY_MAX_DELAY_MS = 2_000 + +type SchemaStartupOptions = { + now?: () => number + random?: () => number + retryDeadlineMs?: number + wait?: (delayMs: number) => Promise +} + +function retryDelayMs(attempt: number, random: () => number): number { + const ceiling = Math.min( + RETRY_BASE_DELAY_MS * 2 ** (attempt - 1), + RETRY_MAX_DELAY_MS + ) + return Math.ceil(ceiling * (0.5 + random() * 0.5)) +} + +function wait(delayMs: number): Promise { + return new Promise((resolve) => setTimeout(resolve, delayMs)) +} + +function retryableSchemaError(error: unknown, statement: string): boolean { + const value = error as { code?: unknown; constraint?: unknown } + return ( + RETRYABLE_SCHEMA_CODES.has(String(value.code)) || + (value.code === '23505' && + ((value.constraint === 'pg_type_typname_nsp_index' && + /^\s*CREATE\s+TABLE\s+IF\s+NOT\s+EXISTS\b/i.test(statement)) || + (value.constraint === 'pg_class_relname_nsp_index' && + /^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+IF\s+NOT\s+EXISTS\b/i.test(statement)))) + ) +} + +export async function applyPostgresSchema( + statements: string[], + query: (statement: string) => Promise, + options: SchemaStartupOptions = {} +): Promise { + const now = options.now ?? Date.now + const random = options.random ?? Math.random + const pause = options.wait ?? wait + const deadlineAt = now() + (options.retryDeadlineMs ?? DEFAULT_RETRY_DEADLINE_MS) + + for (const statement of statements) { + let attempt = 1 + while (true) { + try { + await query(statement) + break + } catch (error) { + const code = String((error as { code?: unknown }).code) + const remainingMs = deadlineAt - now() + const retryable = retryableSchemaError(error, statement) + if (!retryable || remainingMs <= 0) { + if (retryable) { + console.warn( + JSON.stringify({ + event: 'orca_relay_postgres_schema_retry_exhausted', + code, + attempts: attempt + }) + ) + } + throw error + } + const delayMs = Math.min(remainingMs, retryDelayMs(attempt, random)) + console.warn( + JSON.stringify({ + event: 'orca_relay_postgres_schema_retry', + code, + attempt, + delayMs + }) + ) + await pause(delayMs) + attempt += 1 + } + } + } +} diff --git a/cloud/apps/relay/src/postgres-transaction-recovery.test.ts b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts new file mode 100644 index 00000000000..a49c07d2e7d --- /dev/null +++ b/cloud/apps/relay/src/postgres-transaction-recovery.test.ts @@ -0,0 +1,1446 @@ +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { + openRelayDatabase, + type RelayDatabase, + type RelayLockOptions, + type SqlRow +} from './database.js' + +function postgresTestDatabaseUrl(value: string | undefined): string | undefined { + if (!value) return undefined + const url = new URL(value) + // Detect the intended deadlock before the one-second runtime lock deadline. + url.searchParams.set('options', '-c deadlock_timeout=100ms') + return url.toString() +} + +const databaseUrl = postgresTestDatabaseUrl(process.env.ORCA_RELAY_TEST_POSTGRES_URL) +const describePostgres = databaseUrl ? describe : describe.skip + +type QueryLockHook = (phase: 'before' | 'after', sql: string) => Promise + +class TransactionProbeDatabase implements RelayDatabase { + attempts = 0 + + constructor( + private readonly database: RelayDatabase, + private readonly hook: QueryLockHook, + private readonly probeQueries = false + ) {} + + async query(sql: string, params?: unknown[]): Promise { + return await this.database.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + return await this.database.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await this.database.transaction(async (transaction) => { + this.attempts++ + return await operation( + new QueryLockProbeTransaction(transaction, this.hook, this.probeQueries) + ) + }) + } + + async close(): Promise {} +} + +class QueryLockProbeTransaction implements RelayDatabase { + constructor( + private readonly transactionDatabase: RelayDatabase, + private readonly hook: QueryLockHook, + private readonly probeQueries: boolean + ) {} + + async query(sql: string, params?: unknown[]): Promise { + if (this.probeQueries) await this.hook('before', sql) + const rows = await this.transactionDatabase.query(sql, params) + if (this.probeQueries) await this.hook('after', sql) + return rows + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + await this.hook('before', sql) + const rows = await this.transactionDatabase.queryLocked(sql, params, options) + await this.hook('after', sql) + return rows + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await operation(this) + } + + async close(): Promise {} +} + +function signal(): { promise: Promise; resolve: () => void } { + let resolve!: () => void + return { promise: new Promise((done) => (resolve = done)), resolve } +} + +describePostgres('PostgreSQL transaction recovery', () => { + let database: RelayDatabase + + beforeAll(async () => { + database = await openRelayDatabase({ databaseUrl, dataDir: '' }) + }) + + afterAll(async () => { + await database.close() + }) + + afterEach(async () => { + for (const identity of [ + { userId: 'released-order-user', relayHostId: 'releasedorder001' }, + { userId: 'normalized-order-user', relayHostId: 'normalizedorder1' }, + { userId: 'atomic-final-user-a', relayHostId: 'atomicfinalhosta' }, + { userId: 'atomic-final-user-b', relayHostId: 'atomicfinalhostb' }, + { userId: 'lease-assignment-contention-user', relayHostId: 'leaseassignment1' } + ]) { + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database.query( + `DELETE FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + } + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + 'released-order-cell', + 'normalized-order-cell' + ]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, ['atomic-final-cell']) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [ + 'lease-assignment-contention-cell' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id LIKE 'reconcile-user-%'` + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id LIKE 'reconcile-user-%'`) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + 'reconcile-cell-a', + 'reconcile-cell-b' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + ['completion-race-user'] + ) + await database.query(`DELETE FROM relay_assignment_migrations WHERE user_id = ?`, [ + 'completion-race-user' + ]) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [ + 'completion-race-user' + ]) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + 'completion-race-source', + 'completion-race-target' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + ['completion-contention-user'] + ) + await database.query(`DELETE FROM relay_assignment_migrations WHERE user_id = ?`, [ + 'completion-contention-user' + ]) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [ + 'completion-contention-user' + ]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id IN (?, ?)`, [ + 'completion-contention-source', + 'completion-contention-target' + ]) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + 'completion-contention-source', + 'completion-contention-target' + ]) + await database.query(`DELETE FROM relay_cell_fences WHERE cell_id = ?`, [ + 'dead-source-contention-source' + ]) + await database.query( + `DELETE FROM relay_control_connection_reservations WHERE user_id = ?`, + ['dead-source-contention-user'] + ) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + ['dead-source-contention-user'] + ) + await database.query( + `DELETE FROM relay_assignment_migration_incarnations WHERE user_id = ?`, + ['dead-source-contention-user'] + ) + await database.query(`DELETE FROM relay_assignment_migrations WHERE user_id = ?`, [ + 'dead-source-contention-user' + ]) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [ + 'dead-source-contention-user' + ]) + await database.query(`DELETE FROM relay_cell_runtime WHERE cell_id IN (?, ?)`, [ + 'dead-source-contention-source', + 'dead-source-contention-target' + ]) + await database.query(`DELETE FROM relay_cell_admission WHERE cell_id IN (?, ?)`, [ + 'dead-source-contention-source', + 'dead-source-contention-target' + ]) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + 'dead-source-contention-source', + 'dead-source-contention-target' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id IN (?, ?)`, + ['lease-contention-user', 'aggregate-contention-user'] + ) + await database.query( + `DELETE FROM relay_assignments + WHERE user_id IN (?, ?)`, + ['lease-contention-user', 'aggregate-contention-user'] + ) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?)`, [ + 'lease-contention-cell', + 'aggregate-contention-cell' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + ['sustained-lock-user'] + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ?`, [ + 'sustained-lock-user' + ]) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [ + 'sustained-lock-cell' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ?`, + ['sticky-cell-contention-user'] + ) + await database.query( + `DELETE FROM relay_assignments WHERE user_id = ?`, + ['sticky-cell-contention-user'] + ) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [ + 'sticky-cell-contention-cell' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id LIKE 'postgres-user-%'` + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id LIKE 'postgres-user-%'`) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?, ?)`, [ + 'cell-a', + 'cell-b', + 'cell-c' + ]) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id LIKE 'sticky-fast-user-%'` + ) + await database.query( + `DELETE FROM relay_assignments WHERE user_id LIKE 'sticky-fast-user-%'` + ) + await database.query(`DELETE FROM relay_cells WHERE cell_id = ?`, [ + 'sticky-fast-cell' + ]) + }) + + it('retries an entire deadlock victim transaction on a fresh attempt', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const keys = ['deadlock-a', 'deadlock-b'] + for (const key of keys) { + await database.query( + `INSERT INTO relay_rate_windows + (scope_key, window_kind, window_started_at, count) VALUES (?, ?, ?, ?) + ON CONFLICT (scope_key, window_kind, window_started_at) DO UPDATE SET count = ?`, + [key, 'transaction-recovery', 1, 0, 0] + ) + } + + let firstAttemptArrivals = 0 + let releaseFirstAttempts!: () => void + const firstAttemptsReady = new Promise((resolve) => { + releaseFirstAttempts = resolve + }) + const attempts = [0, 0] + const mutateWithOppositeLockOrder = async ( + operationIndex: number, + firstKey: string, + secondKey: string + ): Promise => { + await database.transaction(async (transaction) => { + attempts[operationIndex] = (attempts[operationIndex] ?? 0) + 1 + await transaction.queryLocked( + `SELECT * FROM relay_rate_windows + WHERE scope_key = ? AND window_kind = ? AND window_started_at = ?`, + [firstKey, 'transaction-recovery', 1] + ) + if (attempts[operationIndex] === 1) { + firstAttemptArrivals++ + if (firstAttemptArrivals === 2) releaseFirstAttempts() + await firstAttemptsReady + } + await transaction.queryLocked( + `SELECT * FROM relay_rate_windows + WHERE scope_key = ? AND window_kind = ? AND window_started_at = ?`, + [secondKey, 'transaction-recovery', 1] + ) + await transaction.query( + `UPDATE relay_rate_windows SET count = count + 1 + WHERE scope_key = ? AND window_kind = ? AND window_started_at = ?`, + [firstKey, 'transaction-recovery', 1] + ) + }) + } + + await Promise.all([ + mutateWithOppositeLockOrder(0, keys[0]!, keys[1]!), + mutateWithOppositeLockOrder(1, keys[1]!, keys[0]!) + ]) + + expect([...attempts].sort()).toEqual([1, 2]) + const rows = await database.query( + `SELECT scope_key, count FROM relay_rate_windows + WHERE window_kind = ? ORDER BY scope_key`, + ['transaction-recovery'] + ) + expect(rows).toEqual([ + { scope_key: keys[0], count: '1' }, + { scope_key: keys[1], count: '1' } + ]) + expect(warn).toHaveBeenCalledWith( + expect.stringContaining('"event":"orca_relay_postgres_transaction_retry"') + ) + expect(warn).toHaveBeenCalledWith(expect.stringContaining('"phase":"rate-limit"')) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('"event":"orca_relay_postgres_transaction_exhausted"') + ) + warn.mockRestore() + }, 10_000) + + it('defers assignment around a released-cell activity transaction', async () => { + const now = 1_100_000_000_000 + const identity = { userId: 'released-order-user', relayHostId: 'releasedorder001' } + const cellId = 'released-order-cell' + const activityId = 'splice:released-order' + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, [ + identity.userId, + identity.relayHostId + ]) + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([ + { id: cellId, url: 'https://released-order.example.com', capacityRequests: 100 } + ]) + const assignment = await seedStore.assign(identity) + await seedStore.acquireActivity(identity, { activityId, kind: 'splice', cellId }) + + const assignmentLocked = signal() + const legacyCellLocked = signal() + let gateFirstAssignmentAttempt = true + const directorDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + gateFirstAssignmentAttempt && + phase === 'after' && + sql.includes('FROM relay_assignments') + ) { + gateFirstAssignmentAttempt = false + assignmentLocked.resolve() + await legacyCellLocked.promise + } + }) + const directorStore = new RelayAssignmentStore(directorDatabase, () => now) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + const directorAssign = directorStore.assign(identity) + await assignmentLocked.promise + let legacyAttempts = 0 + const legacyRelease = database.transaction(async (transaction) => { + legacyAttempts++ + const lease = ( + await transaction.queryLocked( + `SELECT * FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, activityId] + ) + )[0] + if (!lease) return + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cellId]) + legacyCellLocked.resolve() + await transaction.query( + `UPDATE relay_cells SET reserved_requests = reserved_requests - 2 WHERE cell_id = ?`, + [cellId] + ) + await transaction.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_id = ?`, + [identity.userId, identity.relayHostId, activityId] + ) + await transaction.query( + `UPDATE relay_assignments SET reserved_splices = reserved_splices - 1 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + }) + + await Promise.all([directorAssign, legacyRelease]) + + expect(directorDatabase.attempts).toBe(2) + expect(legacyAttempts).toBe(1) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_retry') + ) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_exhausted') + ) + await expect(seedStore.resolve(identity)).resolves.toMatchObject({ + cellId, + assignmentEpoch: assignment.assignmentEpoch + }) + const state = await database.query( + `SELECT assignment.reserved_controls, assignment.reserved_splices, + cell.reserved_requests + FROM relay_assignments assignment + JOIN relay_cells cell ON cell.cell_id = assignment.cell_id + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + expect(state).toEqual([ + { reserved_controls: '1', reserved_splices: '0', reserved_requests: '1' } + ]) + warn.mockRestore() + }, 15_000) + + it('renews an existing control without waiting on a legacy cell lock', async () => { + const now = 1_150_000_000_000 + const identity = { userId: 'sustained-lock-user', relayHostId: 'sustainedlock01' } + const cell = { + id: 'sustained-lock-cell', + url: 'https://sustained-lock.example.com', + capacityRequests: 100 + } + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([cell]) + await seedStore.assign(identity) + + const legacyCellLocked = signal() + const releaseLegacyCell = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id]) + legacyCellLocked.resolve() + await releaseLegacyCell.promise + }) + await legacyCellLocked.promise + + let inventoryAttempts = 0 + const assignmentDatabase = new TransactionProbeDatabase( + database, + async (phase, sql) => { + if (phase !== 'before' || !sql.includes('FROM relay_cells ORDER BY')) return + inventoryAttempts++ + } + ) + const assignmentStore = new RelayAssignmentStore(assignmentDatabase, () => now) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + await expect(assignmentStore.assign(identity)).resolves.toMatchObject({ + cellId: cell.id + }) + releaseLegacyCell.resolve() + await expect(legacyTransaction).resolves.toBeUndefined() + expect(inventoryAttempts).toBe(0) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_retry') + ) + warn.mockRestore() + }, 15_000) + + it('retries a new sticky control without forming a legacy cell-first deadlock', async () => { + const now = 1_175_000_000_000 + const identity = { + userId: 'sticky-cell-contention-user', + relayHostId: 'stickycellwait1' + } + const cell = { + id: 'sticky-cell-contention-cell', + url: 'https://sticky-cell-contention.example.com', + capacityRequests: 100 + } + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([cell]) + await seedStore.assign(identity) + await seedStore.changeActivity(identity, 'migration', 1) + // Drops the grant's pending lease too: a sticky control the rows do not + // show is what sends the retry through the NOWAIT cell lock. + await seedStore.releaseActivity(identity, 'control-pending:1') + + const legacyCellLocked = signal() + const directorAssignmentLocked = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id]) + legacyCellLocked.resolve() + await directorAssignmentLocked.promise + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + }) + await legacyCellLocked.promise + + let signalFirstAttempt = true + const directorLockOrder: string[] = [] + const assignmentDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if (phase === 'before') { + if (sql.includes('FROM relay_assignments WHERE user_id = ?')) { + directorLockOrder.push('assignment') + } else if (sql.includes('FROM relay_assignment_activity_leases')) { + directorLockOrder.push('activity') + } else if (sql.includes('FROM relay_cells ORDER BY')) { + directorLockOrder.push('cell-inventory') + } else if (sql.includes('FROM relay_cells WHERE cell_id = ?')) { + directorLockOrder.push('cell') + } + } + if ( + signalFirstAttempt && + phase === 'after' && + sql.includes('FROM relay_assignments WHERE user_id = ?') + ) { + signalFirstAttempt = false + directorAssignmentLocked.resolve() + } + }) + const store = new RelayAssignmentStore(assignmentDatabase, () => now) + + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: cell.id, + assignmentEpoch: 1 + }) + await expect(legacyTransaction).resolves.toBeUndefined() + expect(assignmentDatabase.attempts).toBe(2) + expect(directorLockOrder).toEqual([ + 'assignment', + 'activity', + 'cell', + 'cell-inventory', + 'assignment', + 'activity' + ]) + const state = await database.query( + `SELECT assignment.reserved_controls, assignment.migration_leases, + cell.reserved_requests + FROM relay_assignments assignment + JOIN relay_cells cell ON cell.cell_id = assignment.cell_id + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + expect(state).toEqual([ + { reserved_controls: '1', migration_leases: '1', reserved_requests: '2' } + ]) + }, 15_000) + + it('serializes normalized assignment and activity release without a retry', async () => { + const now = 1_200_000_000_000 + const identity = { userId: 'normalized-order-user', relayHostId: 'normalizedorder1' } + const cellId = 'normalized-order-cell' + const activityId = 'splice:normalized-order' + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, [ + identity.userId, + identity.relayHostId + ]) + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([ + { id: cellId, url: 'https://normalized-order.example.com', capacityRequests: 100 } + ]) + await seedStore.assign(identity) + await seedStore.acquireActivity(identity, { activityId, kind: 'splice', cellId }) + + const assignmentLocked = signal() + const releaseReachedAssignment = signal() + const continueAssignment = signal() + let gateFirstAssignmentAttempt = true + const assignDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + gateFirstAssignmentAttempt && + phase === 'after' && + sql.includes('FROM relay_assignments') + ) { + gateFirstAssignmentAttempt = false + assignmentLocked.resolve() + await continueAssignment.promise + } + }) + const releaseDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if (phase === 'before' && sql.includes('FROM relay_assignments')) { + releaseReachedAssignment.resolve() + } + }) + const assignStore = new RelayAssignmentStore(assignDatabase, () => now) + const releaseStore = new RelayAssignmentStore(releaseDatabase, () => now) + + const assign = assignStore.assign(identity) + await assignmentLocked.promise + const release = releaseStore.releaseActivity(identity, activityId) + await releaseReachedAssignment.promise + continueAssignment.resolve() + await Promise.all([assign, release]) + + expect(assignDatabase.attempts).toBe(1) + expect(releaseDatabase.attempts).toBe(1) + const state = await database.query( + `SELECT assignment.reserved_controls, assignment.reserved_splices, + cell.reserved_requests + FROM relay_assignments assignment + JOIN relay_cells cell ON cell.cell_id = assignment.cell_id + WHERE assignment.user_id = ? AND assignment.relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + expect(state).toEqual([ + { reserved_controls: '1', reserved_splices: '0', reserved_requests: '1' } + ]) + }, 15_000) + + it('takes the shared cell lock only for the final atomic activity write', async () => { + const now = 1_250_000_000_000 + const cell = { + id: 'atomic-final-cell', + url: 'https://atomic-final.example.com', + capacityRequests: 4 + } + const identities = [ + { userId: 'atomic-final-user-a', relayHostId: 'atomicfinalhosta' }, + { userId: 'atomic-final-user-b', relayHostId: 'atomicfinalhostb' } + ] + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([cell]) + await Promise.all(identities.map(async (identity) => await seedStore.assign(identity))) + + const firstAtCellWrite = signal() + const secondAtCellWrite = signal() + const releaseFirst = signal() + const releaseSecond = signal() + const isAtomicCellWrite = (sql: string): boolean => + sql.includes('UPDATE relay_cells SET reserved_requests') && + sql.includes('RETURNING cell_id') + const firstDatabase = new TransactionProbeDatabase( + database, + async (phase, sql) => { + if (phase !== 'before' || !isAtomicCellWrite(sql)) return + firstAtCellWrite.resolve() + await releaseFirst.promise + }, + true + ) + const secondDatabase = new TransactionProbeDatabase( + database, + async (phase, sql) => { + if (phase !== 'before' || !isAtomicCellWrite(sql)) return + secondAtCellWrite.resolve() + await releaseSecond.promise + }, + true + ) + const first = new RelayAssignmentStore(firstDatabase, () => now).acquireActivity( + identities[0]!, + { activityId: 'splice:atomic-final-a', kind: 'splice', cellId: cell.id } + ) + await firstAtCellWrite.promise + const second = new RelayAssignmentStore(secondDatabase, () => now).acquireActivity( + identities[1]!, + { activityId: 'splice:atomic-final-b', kind: 'splice', cellId: cell.id } + ) + let reachTimeout: ReturnType | undefined + const secondReachedCellWrite = await Promise.race([ + secondAtCellWrite.promise.then(() => true), + new Promise((resolve) => { + reachTimeout = setTimeout(() => resolve(false), 2_000) + }) + ]) + if (reachTimeout) clearTimeout(reachTimeout) + if (!secondReachedCellWrite) { + releaseFirst.resolve() + releaseSecond.resolve() + await Promise.allSettled([first, second]) + } + expect(secondReachedCellWrite).toBe(true) + + releaseFirst.resolve() + await expect(first).resolves.toBeUndefined() + releaseSecond.resolve() + await expect(second).rejects.toThrow('relay_capacity_exhausted') + await expect( + database.query( + `SELECT cell.reserved_requests, COUNT(lease.activity_id) AS leases, + COALESCE(SUM(lease.request_units), 0) AS lease_units + FROM relay_cells cell + LEFT JOIN relay_assignment_activity_leases lease ON lease.cell_id = cell.cell_id + WHERE cell.cell_id = ? GROUP BY cell.cell_id, cell.reserved_requests`, + [cell.id] + ) + ).resolves.toEqual([{ reserved_requests: '4', leases: '3', lease_units: '4' }]) + await expect( + database.query( + `SELECT user_id, reserved_splices FROM relay_assignments + WHERE user_id IN (?, ?) ORDER BY user_id`, + [identities[0]!.userId, identities[1]!.userId] + ) + ).resolves.toEqual([ + { user_id: identities[0]!.userId, reserved_splices: '1' }, + { user_id: identities[1]!.userId, reserved_splices: '0' } + ]) + }, 15_000) + + it('reconciles drift under the assignment-first lock order while activity waits', async () => { + const now = 1_300_000_000_000 + const cells = [ + { id: 'reconcile-cell-a', url: 'https://reconcile-a.example.com', capacityRequests: 100 }, + { id: 'reconcile-cell-b', url: 'https://reconcile-b.example.com', capacityRequests: 100 } + ] + const identities = Array.from({ length: 12 }, (_, index) => ({ + userId: `reconcile-user-${index}`, + relayHostId: `reconcilehost${String(index).padStart(4, '0')}` + })) + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells(cells) + for (const identity of identities) await seedStore.assign(identity) + await database.query( + `UPDATE relay_assignments SET reserved_controls = 7, reserved_splices = 5 + WHERE user_id LIKE 'reconcile-user-%'` + ) + await database.query( + `UPDATE relay_cells SET reserved_requests = 42 + WHERE cell_id IN (?, ?)`, + ['reconcile-cell-a', 'reconcile-cell-b'] + ) + + const assignmentsLocked = signal() + const activityReachedAssignment = signal() + const continueReconciliation = signal() + let gateReconciliation = true + const reconcileDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + gateReconciliation && + phase === 'after' && + sql.includes('SELECT assignment.* FROM relay_assignments assignment') + ) { + gateReconciliation = false + assignmentsLocked.resolve() + await continueReconciliation.promise + } + }) + const activityDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if (phase === 'before' && sql.includes('FROM relay_assignments WHERE user_id')) { + activityReachedAssignment.resolve() + } + }) + const reconcileStore = new RelayAssignmentStore(reconcileDatabase, () => now) + const activityStore = new RelayAssignmentStore(activityDatabase, () => now) + + const reconciliation = reconcileStore.cellEvacuationStatus( + 'reconcile-cell-a', + 'reconcile-cell-b', + true + ) + await assignmentsLocked.promise + const activity = activityStore.acquireActivity(identities[0]!, { + activityId: 'splice:reconcile', + kind: 'splice', + cellId: 'reconcile-cell-a' + }) + await activityReachedAssignment.promise + continueReconciliation.resolve() + await Promise.all([reconciliation, activity]) + + expect(reconcileDatabase.attempts).toBe(1) + expect(activityDatabase.attempts).toBe(1) + const reservations = await database.query( + `SELECT cell.cell_id, cell.reserved_requests, + COALESCE(SUM(lease.request_units), 0) AS lease_units + FROM relay_cells cell + LEFT JOIN relay_assignment_activity_leases lease ON lease.cell_id = cell.cell_id + WHERE cell.cell_id IN (?, ?) + GROUP BY cell.cell_id, cell.reserved_requests ORDER BY cell.cell_id`, + ['reconcile-cell-a', 'reconcile-cell-b'] + ) + expect(reservations).toEqual([ + { cell_id: 'reconcile-cell-a', reserved_requests: '8', lease_units: '8' }, + { cell_id: 'reconcile-cell-b', reserved_requests: '6', lease_units: '6' } + ]) + }, 15_000) + + it('fences a source activity queued behind evacuation completion', async () => { + const now = 1_400_000_000_000 + const identity = { + userId: 'completion-race-user', + relayHostId: 'completionrace01' + } + const sourceCellId = 'completion-race-source' + const targetCellId = 'completion-race-target' + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([ + { + id: sourceCellId, + url: 'https://completion-source.example.com', + capacityRequests: 100 + }, + { + id: targetCellId, + url: 'https://completion-target.example.com', + capacityRequests: 100 + } + ]) + const assignment = await seedStore.assign(identity) + const sourceControl = await seedStore.activateControl(identity, { + cellId: sourceCellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + const migration = await seedStore.startEvacuation(identity, targetCellId) + await seedStore.activateControl(identity, { + cellId: targetCellId, + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await seedStore.markMigrationTargetRegistered(identity, { + cellId: targetCellId, + assignmentEpoch: migration.assignmentEpoch + }) + await seedStore.releaseActivity(identity, sourceControl) + + const assignmentLocked = signal() + const activityReachedAssignment = signal() + const continueCompletion = signal() + let gateCompletion = true + const completionDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + gateCompletion && + phase === 'after' && + sql.includes('FROM relay_assignments WHERE user_id') + ) { + gateCompletion = false + assignmentLocked.resolve() + await continueCompletion.promise + } + }) + const activityDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if (phase === 'before' && sql.includes('FROM relay_assignments WHERE user_id')) { + activityReachedAssignment.resolve() + } + }) + const completionStore = new RelayAssignmentStore(completionDatabase, () => now) + const activityStore = new RelayAssignmentStore(activityDatabase, () => now) + + const completion = completionStore.completeReadyEvacuations() + await assignmentLocked.promise + const activity = activityStore.acquireActivity(identity, { + activityId: 'install:queued-source-work', + kind: 'install', + cellId: sourceCellId + }) + const activityRejected = expect(activity).rejects.toThrow( + 'activity_cell_not_authoritative' + ) + await activityReachedAssignment.promise + continueCompletion.resolve() + + await expect(completion).resolves.toBe(1) + await activityRejected + await expect( + seedStore.cellEvacuationStatus(sourceCellId, targetCellId, false) + ).resolves.toMatchObject({ inProgress: 0 }) + }, 15_000) + + it('defers completion instead of deadlocking with a legacy cell-first lock', async () => { + const now = 1_500_000_000_000 + const identity = { + userId: 'completion-contention-user', + relayHostId: 'completionwait01' + } + const sourceCell = { + id: 'completion-contention-source', + url: 'https://completion-contention-source.example.com', + capacityRequests: 100 + } + const targetCell = { + id: 'completion-contention-target', + url: 'https://completion-contention-target.example.com', + capacityRequests: 100 + } + const store = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true + }) + await store.reconcileCells([sourceCell, targetCell]) + await store.recordCellHeartbeat({ + cellId: sourceCell.id, + cellUrl: sourceCell.url, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + startedAt: now - 100, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: targetCell.id, + cellUrl: targetCell.url, + cellIncarnation: '22222222-2222-4222-8222-222222222222', + startedAt: now - 100, + ready: true, + observedRequests: 0 + }) + const assignment = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: sourceCell.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + const migration = await store.startEvacuation(identity, targetCell.id) + await store.activateControl(identity, { + cellId: targetCell.id, + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: targetCell.id, + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + await store.setCellEnabled(sourceCell.id, false) + + const legacyCellLocked = signal() + const completionAssignmentLocked = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [ + sourceCell.id + ]) + legacyCellLocked.resolve() + await completionAssignmentLocked.promise + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + }) + await legacyCellLocked.promise + + let signalCompletion = true + const completionDatabase = new TransactionProbeDatabase( + database, + async (phase, sql) => { + if ( + signalCompletion && + phase === 'after' && + sql.includes('FROM relay_assignments WHERE user_id') + ) { + signalCompletion = false + completionAssignmentLocked.resolve() + } + } + ) + const completionStore = new RelayAssignmentStore(completionDatabase, () => now, { + requireLiveCells: true + }) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + await expect(completionStore.completeReadyEvacuations()).resolves.toBe(0) + await expect(legacyTransaction).resolves.toBeUndefined() + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_retry') + ) + await expect(store.completeReadyEvacuations()).resolves.toBe(1) + }, 15_000) + + it('normalizes persistent dead-source inventory contention', async () => { + let now = 1_550_000_000_000 + const identity = { + userId: 'dead-source-contention-user', + relayHostId: 'deadsourcewait01' + } + const secondIdentity = { + userId: identity.userId, + relayHostId: 'deadsourcewait02' + } + const sourceCell = { + id: 'dead-source-contention-source', + url: 'https://dead-source-contention-source.example.com', + capacityRequests: 100 + } + const targetCell = { + id: 'dead-source-contention-target', + url: 'https://dead-source-contention-target.example.com', + capacityRequests: 100 + } + const sourceIncarnation = '11111111-1111-4111-8111-111111111111' + const targetIncarnation = '22222222-2222-4222-8222-222222222222' + const store = new RelayAssignmentStore(database, () => now, { + requireLiveCells: true + }) + await store.reconcileCells([sourceCell, targetCell]) + await store.setCellEnabled(targetCell.id, false) + await store.recordCellHeartbeat({ + cellId: sourceCell.id, + cellUrl: sourceCell.url, + cellIncarnation: sourceIncarnation, + startedAt: now - 100, + ready: true, + observedRequests: 0 + }) + await store.recordCellHeartbeat({ + cellId: targetCell.id, + cellUrl: targetCell.url, + cellIncarnation: targetIncarnation, + startedAt: now - 100, + ready: true, + observedRequests: 0 + }) + const assignment = await store.assign(identity) + const sourceControl = await store.activateControl(identity, { + cellId: sourceCell.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.setCellEnabled(targetCell.id, true) + const migration = await store.startEvacuation(identity, targetCell.id) + await store.activateControl(identity, { + cellId: targetCell.id, + assignmentEpoch: migration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(identity, { + cellId: targetCell.id, + assignmentEpoch: migration.assignmentEpoch + }) + await store.releaseActivity(identity, sourceControl) + const secondAssignment = await store.assign(secondIdentity) + expect(secondAssignment.cellId).toBe(sourceCell.id) + const secondSourceControl = await store.activateControl(secondIdentity, { + cellId: sourceCell.id, + assignmentEpoch: secondAssignment.assignmentEpoch, + generation: 1 + }) + const secondMigration = await store.startEvacuation(secondIdentity, targetCell.id) + await store.activateControl(secondIdentity, { + cellId: targetCell.id, + assignmentEpoch: secondMigration.assignmentEpoch, + generation: 1 + }) + await store.markMigrationTargetRegistered(secondIdentity, { + cellId: targetCell.id, + assignmentEpoch: secondMigration.assignmentEpoch + }) + await store.releaseActivity(secondIdentity, secondSourceControl) + await store.setCellEnabled(sourceCell.id, false) + now += 45_001 + await store.recordCellHeartbeat({ + cellId: targetCell.id, + cellUrl: targetCell.url, + cellIncarnation: targetIncarnation, + startedAt: now - 45_101, + ready: true, + observedRequests: 0 + }) + await store.attestCellFence(sourceCell.id, sourceIncarnation) + + const legacyCellLocked = signal() + const completionAssignmentLocked = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [ + sourceCell.id + ]) + legacyCellLocked.resolve() + await completionAssignmentLocked.promise + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + }) + await legacyCellLocked.promise + + let signalCompletion = true + const fallbackInventoryLocked = signal() + const freshAssignmentLocked = signal() + const persistentInventoryLocked = signal() + let releasePersistentInventory = false + const completionDatabase = new TransactionProbeDatabase( + database, + async (phase, sql) => { + if ( + signalCompletion && + phase === 'after' && + sql.includes('FROM relay_assignments WHERE user_id') + ) { + signalCompletion = false + completionAssignmentLocked.resolve() + } else if ( + phase === 'after' && + sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC' + ) { + fallbackInventoryLocked.resolve() + await freshAssignmentLocked.promise + } + } + ) + const completionStore = new RelayAssignmentStore(completionDatabase, () => now, { + requireLiveCells: true + }) + const freshAssignmentTransaction = database.transaction(async (transaction) => { + await fallbackInventoryLocked.promise + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + freshAssignmentLocked.resolve() + await transaction.queryLocked(`SELECT * FROM relay_cells ORDER BY cell_id ASC`) + persistentInventoryLocked.resolve() + while (!releasePersistentInventory) { + await new Promise((resolve) => setTimeout(resolve, 250)) + await transaction.query(`SELECT 1`) + } + }) + + const completion = completionStore.cellEvacuationStatus( + sourceCell.id, + targetCell.id, + true + ) + await persistentInventoryLocked.promise + await expect( + completion + ).resolves.toMatchObject({ inProgress: 2, completed: 0, blocked: 2 }) + await expect(legacyTransaction).resolves.toBeUndefined() + releasePersistentInventory = true + await expect(freshAssignmentTransaction).resolves.toBeUndefined() + await expect( + store.cellEvacuationStatus(sourceCell.id, targetCell.id, true) + ).resolves.toMatchObject({ inProgress: 0, completed: 2, blocked: 0 }) + }, 25_000) + + it('defers expired lease cleanup around a legacy cell-first lock', async () => { + const now = 1_600_000_000_000 + const identity = { + userId: 'lease-contention-user', + relayHostId: 'leasecontention1' + } + const cell = { + id: 'lease-contention-cell', + url: 'https://lease-contention-cell.example.com', + capacityRequests: 100 + } + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([cell]) + await store.assign(identity) + await database.query( + `UPDATE relay_assignment_activity_leases SET expires_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [now - 1, identity.userId, identity.relayHostId] + ) + + const legacyCellLocked = signal() + const cleanupAssignmentLocked = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id]) + legacyCellLocked.resolve() + await cleanupAssignmentLocked.promise + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + }) + await legacyCellLocked.promise + + let signalCleanup = true + const cleanupDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + signalCleanup && + phase === 'after' && + sql.includes('FROM relay_assignments WHERE user_id') + ) { + signalCleanup = false + cleanupAssignmentLocked.resolve() + } + }) + const cleanupStore = new RelayAssignmentStore(cleanupDatabase, () => now) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + await expect(cleanupStore.releaseExpiredActivityLeases()).resolves.toBe(0) + await expect(legacyTransaction).resolves.toBeUndefined() + expect(cleanupDatabase.attempts).toBe(1) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_retry') + ) + await expect(store.releaseExpiredActivityLeases()).resolves.toBe(1) + }, 15_000) + + it('skips an expired lease when another director owns its assignment lock', async () => { + const now = 1_650_000_000_000 + const identity = { + userId: 'lease-assignment-contention-user', + relayHostId: 'leaseassignment1' + } + const cell = { + id: 'lease-assignment-contention-cell', + url: 'https://lease-assignment-contention-cell.example.com', + capacityRequests: 100 + } + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([cell]) + await store.assign(identity) + await database.query( + `UPDATE relay_assignment_activity_leases SET expires_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [now - 1, identity.userId, identity.relayHostId] + ) + + const assignmentLocked = signal() + const releaseAssignment = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + assignmentLocked.resolve() + await releaseAssignment.promise + }) + await assignmentLocked.promise + + const cleanupDatabase = new TransactionProbeDatabase(database, async () => undefined) + const cleanupStore = new RelayAssignmentStore(cleanupDatabase, () => now) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + await expect(cleanupStore.releaseExpiredActivityLeases()).resolves.toBe(0) + expect(cleanupDatabase.attempts).toBe(1) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_retry') + ) + warn.mockRestore() + + releaseAssignment.resolve() + await expect(legacyTransaction).resolves.toBeUndefined() + await expect(store.releaseExpiredActivityLeases()).resolves.toBe(1) + }, 15_000) + + it('defers aggregate expiry cleanup around a legacy cell-first lock', async () => { + const now = 1_700_000_000_000 + const identity = { + userId: 'aggregate-contention-user', + relayHostId: 'aggregatewait01' + } + const cell = { + id: 'aggregate-contention-cell', + url: 'https://aggregate-contention-cell.example.com', + capacityRequests: 100 + } + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([cell]) + await store.assign(identity) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database.query( + `UPDATE relay_assignments SET lease_expires_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [now - 1, identity.userId, identity.relayHostId] + ) + + const legacyCellLocked = signal() + const cleanupAssignmentsLocked = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked(`SELECT * FROM relay_cells WHERE cell_id = ?`, [cell.id]) + legacyCellLocked.resolve() + await cleanupAssignmentsLocked.promise + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + }) + await legacyCellLocked.promise + + let signalCleanup = true + const cleanupDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + signalCleanup && + phase === 'after' && + sql.includes('FROM relay_assignments WHERE lease_expires_at') + ) { + signalCleanup = false + cleanupAssignmentsLocked.resolve() + } + }) + const cleanupStore = new RelayAssignmentStore(cleanupDatabase, () => now) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + await expect(cleanupStore.releaseExpiredActivity()).resolves.toBe(0) + await expect(legacyTransaction).resolves.toBeUndefined() + expect(cleanupDatabase.attempts).toBe(1) + expect(warn).not.toHaveBeenCalledWith( + expect.stringContaining('orca_relay_postgres_transaction_retry') + ) + await expect(store.releaseExpiredActivity()).resolves.toBe(1) + }, 15_000) + + it('defers aggregate expiry before waiting on a legacy assignment row', async () => { + const now = 1_700_000_000_000 + const identity = { + userId: 'aggregate-contention-user', + relayHostId: 'aggregatewait01' + } + const cell = { + id: 'aggregate-contention-cell', + url: 'https://aggregate-contention-cell.example.com', + capacityRequests: 100 + } + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells([cell]) + await store.assign(identity) + await database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await database.query( + `UPDATE relay_assignments SET lease_expires_at = ? + WHERE user_id = ? AND relay_host_id = ?`, + [now - 1, identity.userId, identity.relayHostId] + ) + + const legacyAssignmentLocked = signal() + const releaseLegacyAssignment = signal() + const legacyTransaction = database.transaction(async (transaction) => { + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + legacyAssignmentLocked.resolve() + await releaseLegacyAssignment.promise + }) + await legacyAssignmentLocked.promise + + const timedOut = Symbol('timed-out') + const cleanup = store.releaseExpiredActivity() + const result = await Promise.race([ + cleanup, + new Promise((resolve) => { + setTimeout(() => resolve(timedOut), 750) + }) + ]) + releaseLegacyAssignment.resolve() + await legacyTransaction + if (result === timedOut) await cleanup + + expect(result).toBe(0) + await expect(store.releaseExpiredActivity()).resolves.toBe(1) + }, 15_000) + + it('keeps concurrent dormant reassignments capacity-consistent', async () => { + const now = 1_000_000_000_000 + const serializedDatabase = new TransactionProbeDatabase(database, async () => undefined) + const store = new RelayAssignmentStore(serializedDatabase, () => now) + await database.query( + `DELETE FROM relay_assignment_activity_leases WHERE user_id LIKE 'postgres-user-%'` + ) + await database.query(`DELETE FROM relay_assignments WHERE user_id LIKE 'postgres-user-%'`) + await database.query(`DELETE FROM relay_cells WHERE cell_id IN (?, ?, ?)`, [ + 'cell-a', + 'cell-b', + 'cell-c' + ]) + await store.reconcileCells([ + { id: 'cell-a', url: 'https://cell-a.example.com', capacityRequests: 100 }, + { id: 'cell-b', url: 'https://cell-b.example.com', capacityRequests: 100 }, + { id: 'cell-c', url: 'https://cell-c.example.com', capacityRequests: 100 } + ]) + const identities = Array.from({ length: 30 }, (_, index) => ({ + userId: `postgres-user-${index}`, + relayHostId: `postgreshost${String(index).padStart(5, '0')}` + })) + await Promise.all(identities.map(async (identity) => await store.assign(identity))) + + await database.query(`DELETE FROM relay_assignment_activity_leases`) + await database.query( + `UPDATE relay_assignments SET lease_expires_at = ?, last_activity_at = ?, + reserved_controls = 0, reserved_splices = 0, reserved_invites = 0, + pending_installs = 0, pending_confirmations = 0, migration_leases = 0`, + [0, 0] + ) + await database.query(`UPDATE relay_cells SET reserved_requests = 0`) + + await Promise.all(identities.map(async (identity) => await store.assign(identity))) + + const reservations = await database.query( + `SELECT cell_id, reserved_requests FROM relay_cells ORDER BY cell_id` + ) + expect(reservations).toEqual([ + { cell_id: 'cell-a', reserved_requests: '10' }, + { cell_id: 'cell-b', reserved_requests: '10' }, + { cell_id: 'cell-c', reserved_requests: '10' } + ]) + expect(serializedDatabase.attempts).toBe(121) + }, 10_000) + + it('renews independent sticky assignments concurrently without the placement queue', async () => { + const now = 1_800_000_000_000 + const identities = Array.from({ length: 6 }, (_, index) => ({ + userId: `sticky-fast-user-${index}`, + relayHostId: `stickyfast${String(index).padStart(6, '0')}` + })) + const cell = { + id: 'sticky-fast-cell', + url: 'https://sticky-fast.example.com', + capacityRequests: 100 + } + const seedStore = new RelayAssignmentStore(database, () => now) + await seedStore.reconcileCells([cell]) + for (const identity of identities) await seedStore.assign(identity) + + const allRenewalsReachedDatabase = signal() + const releaseRenewals = signal() + let renewalArrivals = 0 + const probedDatabase = new TransactionProbeDatabase(database, async (phase, sql) => { + if ( + phase !== 'after' || + !sql.includes('FROM relay_assignments WHERE user_id = ?') + ) { + return + } + renewalArrivals++ + if (renewalArrivals === identities.length) allRenewalsReachedDatabase.resolve() + await releaseRenewals.promise + }) + const store = new RelayAssignmentStore(probedDatabase, () => now) + const renewals = identities.map(async (identity) => await store.assign(identity)) + let reachedBeforeTimeout = false + try { + reachedBeforeTimeout = await Promise.race([ + allRenewalsReachedDatabase.promise.then(() => true), + new Promise((resolve) => setTimeout(() => resolve(false), 1_000)) + ]) + } finally { + releaseRenewals.resolve() + await Promise.all(renewals) + } + + expect(reachedBeforeTimeout).toBe(true) + expect(renewalArrivals).toBe(identities.length) + expect(probedDatabase.attempts).toBe(identities.length) + const state = await database.query( + `SELECT COUNT(*) AS assignments, SUM(reserved_controls) AS controls + FROM relay_assignments WHERE user_id LIKE 'sticky-fast-user-%'` + ) + expect(state).toEqual([{ assignments: '6', controls: '6' }]) + }, 10_000) +}) diff --git a/cloud/apps/relay/src/public-assignment-admission.test.ts b/cloud/apps/relay/src/public-assignment-admission.test.ts new file mode 100644 index 00000000000..dbc9a7278be --- /dev/null +++ b/cloud/apps/relay/src/public-assignment-admission.test.ts @@ -0,0 +1,334 @@ +import { describe, expect, it, vi } from 'vitest' +import { RelayPublicAssignmentAdmission } from './public-assignment-admission.js' + +describe('public assignment admission', () => { + it('bounds global concurrency and repeated work for one relay host', async () => { + let now = 0 + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 2, + minIntervalMs: 5_000, + now: () => now + }) + const first = await admission.acquire('host-a') + expect(first).not.toBeNull() + await expect(admission.acquire('host-a')).resolves.toBeNull() + const second = await admission.acquire('host-b') + expect(second).not.toBeNull() + await expect(admission.acquire('host-c')).resolves.toBeNull() + + first?.release() + await expect(admission.acquire('host-a')).resolves.toBeNull() + now = 5_000 + const recovered = await admission.acquire('host-a') + expect(recovered).not.toBeNull() + recovered?.release() + second?.release() + }) + + it('releases capacity once when callers settle more than once', async () => { + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + minIntervalMs: 0, + now: vi.fn(() => 0) + }) + const lease = await admission.acquire('host-a') + lease?.release() + lease?.release() + await expect(admission.acquire('host-b')).resolves.not.toBeNull() + }) + + it('grants ordinary waiters in FIFO order without raising concurrency', async () => { + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 2, + waitMs: 1_000, + minIntervalMs: 0, + schedule: () => () => undefined + }) + const first = await admission.acquire('host-a') + const secondPromise = admission.acquire('host-b') + const thirdPromise = admission.acquire('host-c') + + await expect(admission.acquire('host-d')).resolves.toBeNull() + first?.release() + const second = await secondPromise + expect(second).not.toBeNull() + let thirdSettled = false + void thirdPromise.then(() => { thirdSettled = true }) + await Promise.resolve() + expect(thirdSettled).toBe(false) + second?.release() + const third = await thirdPromise + expect(third).not.toBeNull() + third?.release() + }) + + it('fails an ordinary wait closed when its bounded wait expires', async () => { + let expire!: () => void + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 1, + waitMs: 1_000, + minIntervalMs: 0, + schedule: (callback) => { + expire = callback + return () => undefined + } + }) + const first = await admission.acquire('host-a') + const waiting = admission.acquire('host-b') + + expire() + await expect(waiting).resolves.toBeNull() + first?.release() + }) + + it('gives a bounded reserved waiter the next public slot', async () => { + let expire!: () => void + let cancelled = false + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 2, + maxReservedConcurrent: 1, + reservedWaitMs: 1_000, + minIntervalMs: 0, + schedule: (callback) => { + expire = callback + return () => { + cancelled = true + } + } + }) + const first = await admission.acquire('host-a') + const second = await admission.acquire('host-b') + const reservedPromise = admission.acquireReserved('host-c') + + await expect(admission.acquire('host-d')).resolves.toBeNull() + await expect(admission.acquireReserved('host-e')).resolves.toBeNull() + first?.release() + const reserved = await reservedPromise + + expect(reserved).not.toBeNull() + expect(cancelled).toBe(true) + await expect(admission.acquire('host-d')).resolves.toBeNull() + reserved?.release() + await expect(admission.acquire('host-d')).resolves.not.toBeNull() + second?.release() + expire() + }) + + it('lets reserved recovery displace the same host from the ordinary queue', async () => { + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 1, + waitMs: 1_000, + maxReservedConcurrent: 1, + reservedWaitMs: 1_000, + minIntervalMs: 0, + schedule: () => () => undefined + }) + const active = await admission.acquire('host-a') + const ordinary = admission.acquire('host-b') + const reserved = admission.acquireReserved('host-b') + + await expect(ordinary).resolves.toBeNull() + active?.release() + const recovered = await reserved + expect(recovered).not.toBeNull() + recovered?.release() + }) + + it('fails a reserved wait closed when its bounded wait expires', async () => { + let expire!: () => void + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 2, + maxReservedConcurrent: 1, + reservedWaitMs: 1_000, + minIntervalMs: 0, + schedule: (callback) => { + expire = callback + return () => undefined + } + }) + const first = await admission.acquire('host-a') + const second = await admission.acquire('host-b') + const reserved = admission.acquireReserved('host-c') + + expire() + await expect(reserved).resolves.toBeNull() + first?.release() + await expect(admission.acquire('host-d')).resolves.not.toBeNull() + second?.release() + }) + + it('serializes same-host assignment and reserved recovery without losing priority', async () => { + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 2, + maxReservedConcurrent: 1, + reservedWaitMs: 1_000, + minIntervalMs: 0, + schedule: () => () => undefined + }) + const assignment = await admission.acquire('host-a') + const reservedPromise = admission.acquireReserved('host-a') + + await Promise.resolve() + await expect(admission.acquire('host-b')).resolves.toBeNull() + assignment?.release() + const reserved = await reservedPromise + + expect(reserved).not.toBeNull() + await expect(admission.acquire('host-a')).resolves.toBeNull() + reserved?.release() + await expect(admission.acquire('host-b')).resolves.not.toBeNull() + }) + + it('separates a self-throttled host from a saturated director', async () => { + let now = 0 + const reasons: string[] = [] + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 0, + minIntervalMs: 5_000, + now: () => now, + onRejected: (reason) => reasons.push(reason) + }) + const held = await admission.acquire('host-a') + + // Same host while its own attempt is still running. + await expect(admission.acquire('host-a')).resolves.toBeNull() + // A different host with the single slot taken and no queue. + await expect(admission.acquire('host-b')).resolves.toBeNull() + held?.release() + // Same host again, now inside its retry penalty rather than in flight. + await expect(admission.acquire('host-a')).resolves.toBeNull() + + expect(reasons).toEqual(['host-in-flight', 'queue-full', 'host-rate-limited']) + + now = 5_000 + const recovered = await admission.acquire('host-a') + expect(recovered).not.toBeNull() + expect(reasons).toHaveLength(3) + recovered?.release() + }) + + it('reports a timed-out wait separately from a full queue', async () => { + let expire!: () => void + const reasons: string[] = [] + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 1, + waitMs: 1_000, + minIntervalMs: 0, + schedule: (callback) => { + expire = callback + return () => undefined + }, + onRejected: (reason) => reasons.push(reason) + }) + const first = await admission.acquire('host-a') + const waiting = admission.acquire('host-b') + + expire() + await expect(waiting).resolves.toBeNull() + expect(reasons).toEqual(['wait-timeout']) + first?.release() + }) + + it('reports a displaced queue entry as superseded, not as a timeout', async () => { + const reasons: string[] = [] + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 1, + waitMs: 1_000, + maxReservedConcurrent: 1, + reservedWaitMs: 1_000, + minIntervalMs: 0, + schedule: () => () => undefined, + onRejected: (reason) => reasons.push(reason) + }) + const active = await admission.acquire('host-a') + const ordinary = admission.acquire('host-b') + const reserved = admission.acquireReserved('host-b') + + await expect(ordinary).resolves.toBeNull() + expect(reasons).toEqual(['superseded']) + active?.release() + const recovered = await reserved + expect(recovered).not.toBeNull() + recovered?.release() + }) + + it('distinguishes a busy reserved lane from a reserved host already in flight', async () => { + const reasons: string[] = [] + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 4, + // Two slots so the lane still has room when the same host asks twice. + maxReservedConcurrent: 2, + reservedWaitMs: 1_000, + minIntervalMs: 0, + schedule: () => () => undefined, + onRejected: (reason) => reasons.push(reason) + }) + const reserved = await admission.acquireReserved('host-a') + expect(reserved).not.toBeNull() + + await expect(admission.acquireReserved('host-a')).resolves.toBeNull() + expect(reasons).toEqual(['host-in-flight']) + + const second = await admission.acquireReserved('host-b') + expect(second).not.toBeNull() + await expect(admission.acquireReserved('host-c')).resolves.toBeNull() + + expect(reasons).toEqual(['host-in-flight', 'reserved-unavailable']) + reserved?.release() + second?.release() + }) + + it('tells each caller its own rejection reason without crossing requests', async () => { + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 1, + maxQueued: 1, + waitMs: 1_000, + maxReservedConcurrent: 1, + reservedWaitMs: 1_000, + minIntervalMs: 5_000, + now: () => 0, + schedule: () => () => undefined + }) + const queued: string[] = [] + const displaced: string[] = [] + const throttled: string[] = [] + + const active = await admission.acquire('host-a') + // 'host-b' waits, then reserved recovery for the same host displaces it. + const ordinary = admission.acquire('host-b', (reason) => displaced.push(reason)) + const reserved = admission.acquireReserved('host-b', (reason) => queued.push(reason)) + await expect(ordinary).resolves.toBeNull() + active?.release() + const recovered = await reserved + recovered?.release() + await admission.acquire('host-a', (reason) => throttled.push(reason)) + + expect(displaced).toEqual(['superseded']) + expect(queued).toEqual([]) + expect(throttled).toEqual(['host-rate-limited']) + }) + + it('stays silent on every granted acquisition', async () => { + const reasons: string[] = [] + const admission = new RelayPublicAssignmentAdmission({ + maxConcurrent: 2, + maxReservedConcurrent: 1, + minIntervalMs: 0, + onRejected: (reason) => reasons.push(reason) + }) + const first = await admission.acquire('host-a') + const second = await admission.acquireReserved('host-b') + + expect(first).not.toBeNull() + expect(second).not.toBeNull() + expect(reasons).toEqual([]) + first?.release() + second?.release() + }) +}) diff --git a/cloud/apps/relay/src/public-assignment-admission.ts b/cloud/apps/relay/src/public-assignment-admission.ts new file mode 100644 index 00000000000..bd6d7f1a488 --- /dev/null +++ b/cloud/apps/relay/src/public-assignment-admission.ts @@ -0,0 +1,239 @@ +type AssignmentAdmissionLease = { release(): void } +type CancelWait = () => void +// Per-call sink: acquire() keeps returning null so callers stay unchanged, and the +// reason rides out of band to whoever made this particular request. +type RejectionSink = (reason: AssignmentAdmissionRejection) => void +type PendingAssignment = { + relayHostId: string + resolve: (lease: AssignmentAdmissionLease | null) => void + cancelWait: CancelWait + notifyRejected?: RejectionSink +} + +// Every rejection here becomes an identical 503, so without the reason a busy +// director and a self-throttling host are indistinguishable in production. +export type AssignmentAdmissionRejection = + | 'host-in-flight' + | 'host-rate-limited' + | 'queue-full' + | 'wait-timeout' + | 'superseded' + | 'reserved-unavailable' + +const MAX_TRACKED_HOSTS = 4_096 + +export class RelayPublicAssignmentAdmission { + private active = 0 + private activeReserved = 0 + private readonly activeAssignmentHosts = new Set() + private readonly activeReservedHosts = new Set() + private readonly queuedAssignmentHosts = new Set() + private readonly lastAttemptByHost = new Map() + private readonly lastReservedAttemptByHost = new Map() + private readonly pendingAssignments: PendingAssignment[] = [] + private pendingReserved: PendingAssignment | undefined + + constructor( + private readonly options: { + maxConcurrent: number + maxQueued?: number + waitMs?: number + maxReservedConcurrent?: number + reservedWaitMs?: number + minIntervalMs: number + now?: () => number + schedule?: (callback: () => void, delayMs: number) => CancelWait + onRejected?: (reason: AssignmentAdmissionRejection) => void + } + ) {} + + async acquire( + relayHostId: string, + notifyRejected?: RejectionSink + ): Promise { + const now = (this.options.now ?? Date.now)() + const lastAttempt = this.lastAttemptByHost.get(relayHostId) + if ( + this.activeAssignmentHosts.has(relayHostId) || + this.activeReservedHosts.has(relayHostId) || + this.queuedAssignmentHosts.has(relayHostId) || + this.pendingReserved?.relayHostId === relayHostId + ) { + return this.reject('host-in-flight', notifyRejected) + } + if (lastAttempt !== undefined && now - lastAttempt < this.options.minIntervalMs) { + return this.reject('host-rate-limited', notifyRejected) + } + if ( + this.active < this.options.maxConcurrent && + this.pendingReserved === undefined && + this.pendingAssignments.length === 0 + ) { + this.recordAttempt(this.lastAttemptByHost, relayHostId, now) + return this.createLease(relayHostId, false) + } + if (this.pendingAssignments.length >= (this.options.maxQueued ?? 0)) { + return this.reject('queue-full', notifyRejected) + } + + return await new Promise((resolve) => { + const schedule = this.options.schedule ?? defaultSchedule + let cancelWait: CancelWait = () => undefined + const pending: PendingAssignment = { + relayHostId, + resolve, + cancelWait: () => cancelWait(), + notifyRejected + } + this.pendingAssignments.push(pending) + this.queuedAssignmentHosts.add(relayHostId) + cancelWait = schedule(() => { + const index = this.pendingAssignments.indexOf(pending) + if (index === -1) return + this.pendingAssignments.splice(index, 1) + this.queuedAssignmentHosts.delete(relayHostId) + resolve(this.reject('wait-timeout', notifyRejected)) + }, this.options.waitMs ?? 1_000) + }) + } + + async acquireReserved( + relayHostId: string, + notifyRejected?: RejectionSink + ): Promise { + const maxReservedConcurrent = this.options.maxReservedConcurrent ?? 0 + const now = (this.options.now ?? Date.now)() + const lastAttempt = this.lastReservedAttemptByHost.get(relayHostId) + if (maxReservedConcurrent === 0 || this.activeReserved >= maxReservedConcurrent) { + return this.reject('reserved-unavailable', notifyRejected) + } + if (this.activeReservedHosts.has(relayHostId)) { + return this.reject('host-in-flight', notifyRejected) + } + if (this.pendingReserved !== undefined) { + return this.reject('reserved-unavailable', notifyRejected) + } + if (lastAttempt !== undefined && now - lastAttempt < this.options.minIntervalMs) { + return this.reject('host-rate-limited', notifyRejected) + } + this.cancelQueuedAssignment(relayHostId) + if ( + this.active < this.options.maxConcurrent && + !this.activeAssignmentHosts.has(relayHostId) + ) { + this.recordAttempt(this.lastReservedAttemptByHost, relayHostId, now) + return this.createLease(relayHostId, true) + } + + return await new Promise((resolve) => { + const schedule = this.options.schedule ?? defaultSchedule + let cancelWait: CancelWait = () => undefined + this.pendingReserved = { + relayHostId, + resolve, + cancelWait: () => cancelWait(), + notifyRejected + } + cancelWait = schedule(() => { + if (this.pendingReserved?.relayHostId !== relayHostId) return + this.pendingReserved = undefined + resolve(this.reject('wait-timeout', notifyRejected)) + this.grantPendingAssignments() + }, this.options.reservedWaitMs ?? 1_000) + }) + } + + private createLease(relayHostId: string, reserved: boolean): AssignmentAdmissionLease { + this.active++ + if (reserved) { + this.activeReserved++ + this.activeReservedHosts.add(relayHostId) + } else { + this.activeAssignmentHosts.add(relayHostId) + } + let released = false + return { + release: () => { + if (released) return + released = true + this.active = Math.max(0, this.active - 1) + if (reserved) { + this.activeReserved = Math.max(0, this.activeReserved - 1) + this.activeReservedHosts.delete(relayHostId) + } else { + this.activeAssignmentHosts.delete(relayHostId) + } + this.grantPendingReserved() + this.grantPendingAssignments() + } + } + } + + private grantPendingReserved(): void { + const pending = this.pendingReserved + if ( + !pending || + this.active >= this.options.maxConcurrent || + this.activeAssignmentHosts.has(pending.relayHostId) + ) { + return + } + this.pendingReserved = undefined + pending.cancelWait() + this.recordAttempt( + this.lastReservedAttemptByHost, + pending.relayHostId, + (this.options.now ?? Date.now)() + ) + pending.resolve(this.createLease(pending.relayHostId, true)) + } + + private grantPendingAssignments(): void { + while ( + this.pendingReserved === undefined && + this.active < this.options.maxConcurrent && + this.pendingAssignments.length > 0 + ) { + const pending = this.pendingAssignments.shift()! + this.queuedAssignmentHosts.delete(pending.relayHostId) + pending.cancelWait() + this.recordAttempt( + this.lastAttemptByHost, + pending.relayHostId, + (this.options.now ?? Date.now)() + ) + pending.resolve(this.createLease(pending.relayHostId, false)) + } + } + + private cancelQueuedAssignment(relayHostId: string): void { + const index = this.pendingAssignments.findIndex( + (pending) => pending.relayHostId === relayHostId + ) + if (index === -1) return + const [pending] = this.pendingAssignments.splice(index, 1) + this.queuedAssignmentHosts.delete(relayHostId) + pending?.cancelWait() + // The sink rides on the pending record: this rejects a different caller's request. + pending?.resolve(this.reject('superseded', pending.notifyRejected)) + } + + private reject(reason: AssignmentAdmissionRejection, notifyRejected?: RejectionSink): null { + this.options.onRejected?.(reason) + notifyRejected?.(reason) + return null + } + + private recordAttempt(attempts: Map, relayHostId: string, now: number): void { + attempts.delete(relayHostId) + attempts.set(relayHostId, now) + if (attempts.size > MAX_TRACKED_HOSTS) { + attempts.delete(attempts.keys().next().value!) + } + } +} + +function defaultSchedule(callback: () => void, delayMs: number): CancelWait { + const timer = setTimeout(callback, delayMs) + return () => clearTimeout(timer) +} diff --git a/cloud/apps/relay/src/public-assignment-circuit-breaker.test.ts b/cloud/apps/relay/src/public-assignment-circuit-breaker.test.ts new file mode 100644 index 00000000000..94be0536497 --- /dev/null +++ b/cloud/apps/relay/src/public-assignment-circuit-breaker.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it, vi } from 'vitest' +import { createRelayApp } from './app.js' +import type { RelayConfig } from './config.js' + +describe('public assignment circuit breaker', () => { + it('rejects assign and resolve without invoking relay state', async () => { + const assign = vi.fn() + const resolveResume = vi.fn() + const app = createRelayApp(config(), { + store: { resolveResume } as never, + assignments: { assign } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + for (const path of ['/v1/assign', '/v1/resolve']) { + const response = await app.request(path, { method: 'POST' }) + expect(response.status).toBe(503) + expect(response.headers.get('retry-after')).toBe('5') + expect(await response.json()).toEqual({ error: 'assignments_temporarily_unavailable' }) + } + expect(assign).not.toHaveBeenCalled() + expect(resolveResume).not.toHaveBeenCalled() + expect((await app.request('/health')).status).toBe(200) + expect((await app.request('/v1/admin/drain', { method: 'POST' })).status).toBe(401) + }) +}) + +function config(): RelayConfig { + return { + port: 8080, + publicUrl: 'https://relay.example.test', + cellUrl: 'https://relay.example.test', + authIssuer: 'https://auth.example.test', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.test/jwks', + assignmentSigningKey: new TextEncoder().encode('assignment-key-with-at-least-32-bytes'), + role: 'director', + cellId: 'director', + cells: [], + adminAudience: 'https://relay.example.test/v1/admin/drain', + deployServiceAccount: 'deploy@example.test', + runtimeServiceAccount: 'runtime@example.test', + adminJwksUrl: 'https://auth.example.test/jwks', + databasePoolMax: 3, + publicAssignmentsEnabled: false, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './data' + } +} diff --git a/cloud/apps/relay/src/public-assignment-overload.test.ts b/cloud/apps/relay/src/public-assignment-overload.test.ts new file mode 100644 index 00000000000..2d0426893c4 --- /dev/null +++ b/cloud/apps/relay/src/public-assignment-overload.test.ts @@ -0,0 +1,236 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RelayAssignment } from './assignment-store.js' +import type { RelayConfig } from './config.js' + +const fakes = vi.hoisted(() => ({ + verifyRelayToken: vi.fn(async (token: string) => ({ + sub: 'user-1', + relayHostId: token + })) +})) + +vi.mock('./relay-token-verifier.js', () => ({ + createRelayTokenVerifier: () => fakes.verifyRelayToken, + readBearer: (value: string | undefined) => value?.replace(/^Bearer /, '') ?? null +})) + +import { createRelayApp } from './app.js' + +describe('public assignment overload', () => { + it('rejects duplicate and excess work before it reaches assignment state', async () => { + const hostA = 'aaaaaaaaaaaaaaaa' + const hostB = 'bbbbbbbbbbbbbbbb' + const hostC = 'cccccccccccccccc' + const pending = new Map>>() + const assign = vi.fn(async ({ relayHostId }: { relayHostId: string }) => { + const operation = deferred() + pending.set(relayHostId, operation) + return await operation.promise + }) + const app = createRelayApp(config({ publicAssignmentQueueMax: 0 }), { + store: {} as never, + assignments: { assign } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const first = app.request('/v1/assign', assignmentRequest(hostA)) + await vi.waitFor(() => expect(pending.has(hostA)).toBe(true)) + const duplicate = await app.request('/v1/assign', assignmentRequest(hostA)) + expect(duplicate.status).toBe(503) + + const second = app.request('/v1/assign', assignmentRequest(hostB)) + await vi.waitFor(() => expect(pending.has(hostB)).toBe(true)) + const excess = await app.request('/v1/assign', assignmentRequest(hostC)) + expect(excess.status).toBe(503) + expect(excess.headers.get('retry-after')).toBe('5') + expect(assign).toHaveBeenCalledTimes(2) + + pending.get(hostA)?.resolve(assignment('cell-a', hostA)) + pending.get(hostB)?.resolve(assignment('cell-b', hostB)) + expect((await first).status).toBe(200) + expect((await second).status).toBe(200) + expect((await app.request('/v1/assign', assignmentRequest(hostA))).status).toBe(503) + expect(assign).toHaveBeenCalledTimes(2) + }) + + it('fairly drains a bounded incident-scale burst without raising database concurrency', async () => { + const pending = new Map>>() + const admittedHosts: string[] = [] + let active = 0 + let highWater = 0 + const assign = vi.fn(async ({ relayHostId }: { relayHostId: string }) => { + admittedHosts.push(relayHostId) + active++ + highWater = Math.max(highWater, active) + const operation = deferred() + pending.set(relayHostId, operation) + try { + return await operation.promise + } finally { + active-- + pending.delete(relayHostId) + } + }) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { assign } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + const hosts = Array.from( + { length: 430 }, + (_, index) => `host-${String(index).padStart(11, '0')}` + ) + + const firstWave = hosts.map((host) => + app.request('/v1/assign', assignmentRequest(host)) + ) + await vi.waitFor(() => expect(assign).toHaveBeenCalledTimes(2)) + const retryWave = await Promise.all( + hosts.map((host) => app.request('/v1/assign', assignmentRequest(host))) + ) + + expect(retryWave.every((response) => response.status === 503)).toBe(true) + expect(assign).toHaveBeenCalledTimes(2) + while (assign.mock.calls.length < 130) { + const activeOperations = [...pending] + const expectedCalls = assign.mock.calls.length + activeOperations.length + for (const [host, operation] of activeOperations) { + operation.resolve(assignment(`cell-${host}`, host)) + } + await vi.waitFor(() => expect(assign).toHaveBeenCalledTimes(expectedCalls)) + } + for (const [host, operation] of pending) { + operation.resolve(assignment(`cell-${host}`, host)) + } + const firstResponses = await Promise.all(firstWave) + expect(firstResponses.filter((response) => response.status === 200)).toHaveLength(130) + expect(firstResponses.filter((response) => response.status === 503)).toHaveLength(300) + expect(admittedHosts).toEqual(hosts.slice(0, 130)) + expect(highWater).toBe(2) + }) + + it('reserves one bounded resolve lane while assignment work is saturated', async () => { + const pendingAssignments = new Map>>() + const pendingResolve = deferred<{ userId: string; relayDeviceId: string } | null>() + const assign = vi.fn(async ({ relayHostId }: { relayHostId: string }) => { + const operation = deferred() + pendingAssignments.set(relayHostId, operation) + return await operation.promise + }) + const resolveResume = vi.fn(async () => await pendingResolve.promise) + const resolve = vi.fn(async ({ relayHostId }: { relayHostId: string }) => + assignment('target-cell', relayHostId) + ) + const app = createRelayApp(config({ publicAssignmentQueueMax: 0 }), { + store: { resolveResume } as never, + assignments: { assign, resolve } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const assignmentA = app.request('/v1/assign', assignmentRequest('aaaaaaaaaaaaaaaa')) + const assignmentB = app.request('/v1/assign', assignmentRequest('bbbbbbbbbbbbbbbb')) + await vi.waitFor(() => expect(assign).toHaveBeenCalledTimes(2)) + + const recovery = app.request('/v1/resolve', resolveRequest('cccccccccccccccc', 1)) + await Promise.resolve() + expect(resolveResume).not.toHaveBeenCalled() + const excessResolve = await app.request( + '/v1/resolve', + resolveRequest('dddddddddddddddd', 2) + ) + const excessAssign = await app.request('/v1/assign', assignmentRequest('eeeeeeeeeeeeeeee')) + + expect(excessResolve.status).toBe(503) + expect(excessAssign.status).toBe(503) + expect(assign).toHaveBeenCalledTimes(2) + expect(resolveResume).not.toHaveBeenCalled() + + pendingAssignments + .get('aaaaaaaaaaaaaaaa') + ?.resolve(assignment('cell-a', 'aaaaaaaaaaaaaaaa')) + expect((await assignmentA).status).toBe(200) + await vi.waitFor(() => expect(resolveResume).toHaveBeenCalledTimes(1)) + pendingResolve.resolve({ userId: 'user-1', relayDeviceId: 'device-1' }) + expect((await recovery).status).toBe(200) + expect(resolve).toHaveBeenCalledTimes(1) + + for (const [host, operation] of pendingAssignments) { + operation.resolve(assignment(`cell-${host}`, host)) + } + expect((await assignmentB).status).toBe(200) + }) +}) + +function assignmentRequest(relayHostId: string): RequestInit { + return { + method: 'POST', + headers: { + authorization: `Bearer ${relayHostId}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, relayHostId }) + } +} + +function resolveRequest(relayHostId: string, fill: number): RequestInit { + return { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + relayHostId, + resumeToken: Buffer.alloc(32, fill).toString('base64url') + }) + } +} + +function assignment(cellId: string, relayHostId: string): RelayAssignment { + return { + userId: 'user-1', + relayHostId, + cellId, + cellUrl: `https://${cellId}.relay.example.test`, + assignmentEpoch: 1, + leaseExpiresAt: Date.now() + 300_000 + } +} + +function deferred() { + let resolve!: (value: T) => void + const promise = new Promise((resolvePromise) => { + resolve = resolvePromise + }) + return { promise, resolve } +} + +function config(overrides: Partial = {}): RelayConfig { + return { + port: 8080, + publicUrl: 'https://relay.example.test', + cellUrl: 'https://relay.example.test', + authIssuer: 'https://auth.example.test', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.test/jwks', + assignmentSigningKey: new TextEncoder().encode('assignment-key-with-at-least-32-bytes'), + role: 'director', + cellId: 'director', + cells: [], + adminAudience: 'https://relay.example.test/v1/admin/drain', + deployServiceAccount: 'deploy@example.test', + runtimeServiceAccount: 'runtime@example.test', + adminJwksUrl: 'https://auth.example.test/jwks', + databasePoolMax: 3, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './data', + ...overrides + } +} diff --git a/cloud/apps/relay/src/public-assignment-reconnect-lane.test.ts b/cloud/apps/relay/src/public-assignment-reconnect-lane.test.ts new file mode 100644 index 00000000000..145d2c737cb --- /dev/null +++ b/cloud/apps/relay/src/public-assignment-reconnect-lane.test.ts @@ -0,0 +1,202 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RelayAssignment } from './assignment-store.js' +import type { RelayConfig } from './config.js' + +const fakes = vi.hoisted(() => ({ + verifyRelayToken: vi.fn(async (token: string) => ({ + sub: 'user-1', + relayHostId: token + })) +})) + +vi.mock('./relay-token-verifier.js', () => ({ + createRelayTokenVerifier: () => fakes.verifyRelayToken, + readBearer: (value: string | undefined) => value?.replace(/^Bearer /, '') ?? null +})) + +import { createRelayApp } from './app.js' + +describe('public assignment reconnect lane', () => { + it('admits a verified reconnect past a saturated placement queue', async () => { + const reconnecting = 'rrrrrrrrrrrrrrrr' + const newcomerA = 'aaaaaaaaaaaaaaaa' + const newcomerB = 'bbbbbbbbbbbbbbbb' + const blocked = 'cccccccccccccccc' + const pending = new Map>>() + const assign = vi.fn(async ({ relayHostId }: { relayHostId: string }) => { + if (relayHostId === reconnecting) return assignment('cell-r', reconnecting) + const operation = deferred() + pending.set(relayHostId, operation) + return await operation.promise + }) + const resolve = vi.fn(async ({ relayHostId }: { relayHostId: string }) => + relayHostId === reconnecting ? assignment('cell-r', reconnecting) : null + ) + const outcomes: string[] = [] + const app = createRelayApp(config({ publicAssignmentQueueMax: 0 }), { + store: {} as never, + assignments: { assign, resolve } as never, + drain: vi.fn(), + ready: vi.fn(async () => true), + recordAssignmentAdmission: (outcome) => outcomes.push(outcome) + }) + + // Saturate the placement lane with two in-flight newcomers, zero queue. + const first = app.request('/v1/assign', assignmentRequest(newcomerA)) + await vi.waitFor(() => expect(pending.has(newcomerA)).toBe(true)) + const second = app.request('/v1/assign', assignmentRequest(newcomerB)) + await vi.waitFor(() => expect(pending.has(newcomerB)).toBe(true)) + expect((await app.request('/v1/assign', assignmentRequest(blocked))).status).toBe(503) + + const reconnected = await app.request( + '/v1/assign', + assignmentRequest(reconnecting, { reconnect: true }) + ) + + expect(reconnected.status).toBe(200) + expect(resolve).toHaveBeenCalledWith({ userId: 'user-1', relayHostId: reconnecting }) + expect(assign).toHaveBeenCalledWith({ userId: 'user-1', relayHostId: reconnecting }) + expect(outcomes).toContain('sticky') + + pending.get(newcomerA)?.resolve(assignment('cell-a', newcomerA)) + pending.get(newcomerB)?.resolve(assignment('cell-b', newcomerB)) + expect((await first).status).toBe(200) + expect((await second).status).toBe(200) + }) + + it('rate-limits repeat fast-lane attempts with the short retry-after', async () => { + const host = 'rrrrrrrrrrrrrrrr' + const assign = vi.fn(async () => assignment('cell-r', host)) + const resolve = vi.fn(async () => assignment('cell-r', host)) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { assign, resolve } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + expect( + (await app.request('/v1/assign', assignmentRequest(host, { reconnect: true }))).status + ).toBe(200) + const repeat = await app.request('/v1/assign', assignmentRequest(host, { reconnect: true })) + + expect(repeat.status).toBe(503) + expect(repeat.headers.get('retry-after')).toBe('2') + expect(resolve).toHaveBeenCalledTimes(1) + expect(assign).toHaveBeenCalledTimes(1) + }) + + it('sends an unverified reconnect hint through the placement lane unchanged', async () => { + const host = 'nnnnnnnnnnnnnnnn' + const assign = vi.fn(async () => assignment('cell-n', host)) + const resolve = vi.fn(async () => null) + const outcomes: string[] = [] + const app = createRelayApp(config(), { + store: {} as never, + assignments: { assign, resolve } as never, + drain: vi.fn(), + ready: vi.fn(async () => true), + recordAssignmentAdmission: (outcome) => outcomes.push(outcome) + }) + + const response = await app.request('/v1/assign', assignmentRequest(host, { reconnect: true })) + + expect(response.status).toBe(200) + expect(assign).toHaveBeenCalledTimes(1) + expect(outcomes).toEqual(['placement']) + }) + + it('rejects on the fast lane when the verification probe fails transiently', async () => { + const host = 'tttttttttttttttt' + const assign = vi.fn() + const resolve = vi.fn(async () => { + throw Object.assign(new Error('deadlock detected'), { code: '40P01' }) + }) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { assign, resolve } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const response = await app.request('/v1/assign', assignmentRequest(host, { reconnect: true })) + + expect(response.status).toBe(503) + expect(response.headers.get('retry-after')).toBe('2') + expect(assign).not.toHaveBeenCalled() + }) + + it('never probes for unhinted requests', async () => { + const host = 'uuuuuuuuuuuuuuuu' + const assign = vi.fn(async () => assignment('cell-u', host)) + const resolve = vi.fn() + const app = createRelayApp(config(), { + store: {} as never, + assignments: { assign, resolve } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + expect((await app.request('/v1/assign', assignmentRequest(host))).status).toBe(200) + expect(resolve).not.toHaveBeenCalled() + }) +}) + +function assignmentRequest(relayHostId: string, extra: { reconnect?: boolean } = {}): RequestInit { + return { + method: 'POST', + headers: { + authorization: `Bearer ${relayHostId}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, relayHostId, ...extra }) + } +} + +function assignment(cellId: string, relayHostId: string): RelayAssignment { + return { + userId: 'user-1', + relayHostId, + cellId, + cellUrl: `https://${cellId}.relay.example.test`, + assignmentEpoch: 1, + leaseExpiresAt: Date.now() + 300_000 + } +} + +function deferred() { + let resolve!: (value: T) => void + const promise = new Promise((resolvePromise) => { + resolve = resolvePromise + }) + return { promise, resolve } +} + +function config(overrides: Partial = {}): RelayConfig { + return { + port: 8080, + publicUrl: 'https://relay.example.test', + cellUrl: 'https://relay.example.test', + authIssuer: 'https://auth.example.test', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.test/jwks', + assignmentSigningKey: new TextEncoder().encode('assignment-key-with-at-least-32-bytes'), + role: 'director', + cellId: 'director', + cells: [], + adminAudience: 'https://relay.example.test/v1/admin/drain', + deployServiceAccount: 'deploy@example.test', + runtimeServiceAccount: 'runtime@example.test', + adminJwksUrl: 'https://auth.example.test/jwks', + databasePoolMax: 3, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './data', + ...overrides + } +} diff --git a/cloud/apps/relay/src/regional-host-drain-app.test.ts b/cloud/apps/relay/src/regional-host-drain-app.test.ts new file mode 100644 index 00000000000..1cd34902520 --- /dev/null +++ b/cloud/apps/relay/src/regional-host-drain-app.test.ts @@ -0,0 +1,536 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RelayConfig } from './config.js' + +vi.mock('./admin-token-verifier.js', () => ({ + createAdminTokenVerifier: () => async (token: string, route?: string) => + token === 'deploy-token' || + (token === 'monitor-token' && + (!route || route === '/v1/admin/regional-rehome-control')), + createReadOnlyAdminTokenVerifier: () => async () => false, + createRegionalRehomeControlApplyTokenVerifier: () => async (token: string) => + token === 'deploy-token', + createRegionalRehomeRuntimeTokenVerifier: () => async (token: string) => + token === 'runtime-token', + createRegionalRehomeTokenVerifier: () => async (token: string) => token === 'rehome-token', + createRuntimeTokenVerifier: () => async (token: string) => token === 'runtime-token' +})) + +vi.mock('./relay-token-verifier.js', () => ({ + createRelayTokenVerifier: () => async () => null, + readBearer: (value: string | undefined) => value?.replace(/^Bearer /, '') ?? null +})) + +import { createRelayApp } from './app.js' +import { RelayObservability } from './relay-observability.js' +import { emptyPostgresPoolPressureCounts } from './postgres-pool-pressure.js' +import { + REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + REGIONAL_REHOME_TRUST_PROBE_USER_ID +} from './regional-rehome-trust-probe.js' + +const cellIncarnation = '11111111-1111-4111-8111-111111111111' +const request = { + v: 1, + attemptId: '22222222-2222-4222-8222-222222222222', + userId: 'user-1', + relayHostId: 'abcdefghijklmnop', + sourceCellId: 'production-gce-c7', + sourceCellIncarnation: cellIncarnation, + sourceAssignmentEpoch: 7, + graceMs: 60_000 +} + +describe('regional host drain endpoint', () => { + it('accepts only the dedicated identity and exact cell generation', async () => { + const drainHost = vi.fn(() => 'accepted' as const) + const app = createRelayApp(config(), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost, + cellIncarnation, + ready: vi.fn(async () => true) + }) + + const accepted = await post(app, 'rehome-token', request) + expect(accepted.status).toBe(200) + expect(await accepted.json()).toEqual({ v: 1, outcome: 'accepted' }) + expect(drainHost).toHaveBeenCalledWith(request) + + expect((await post(app, 'deploy-token', request)).status).toBe(401) + expect( + (await post(app, 'rehome-token', { + ...request, + sourceCellIncarnation: '33333333-3333-4333-8333-333333333333' + })).status + ).toBe(409) + expect(drainHost).toHaveBeenCalledOnce() + }) + + it('rejects malformed identities before touching the session registry', async () => { + const drainHost = vi.fn(() => 'accepted' as const) + const app = createRelayApp(config(), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost, + cellIncarnation, + ready: vi.fn(async () => true) + }) + + expect( + (await post(app, 'rehome-token', { ...request, relayHostId: 'raw-host-id' })).status + ).toBe(400) + expect(drainHost).not.toHaveBeenCalled() + }) + + it('is unavailable on the director', async () => { + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost: vi.fn(() => 'accepted' as const), + cellIncarnation, + ready: vi.fn(async () => true) + }) + expect((await post(app, 'rehome-token', request)).status).toBe(404) + }) + + it('proves the shared runtime identity is rejected without touching a session', async () => { + const drainHost = vi.fn(() => 'host-not-connected' as const) + const app = createRelayApp(config(), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost, + regionalRehomeTrustProbeHostExists: vi.fn(() => false), + regionalRehomeIdentityToken: vi.fn(async () => 'runtime-token'), + cellIncarnation, + ready: vi.fn(async () => true) + }) + const probe = { + ...request, + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceAssignmentEpoch: 1, + graceMs: 0 + } + + expect(REGIONAL_REHOME_TRUST_PROBE_HOST_ID).toHaveLength(16) + for (let call = 0; call < 2; call++) { + const response = await post(app, 'rehome-token', probe) + expect(response.status).toBe(200) + expect(await response.json()).toEqual({ + v: 1, + outcome: 'host-not-connected', + sharedRuntimeIdentityRejected: true + }) + } + expect(drainHost).toHaveBeenCalledTimes(2) + }) + + it('fails closed if the synthetic host is not provably absent', async () => { + const drainHost = vi.fn(() => 'host-not-connected' as const) + const app = createRelayApp(config(), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost, + regionalRehomeTrustProbeHostExists: vi.fn(() => true), + regionalRehomeIdentityToken: vi.fn(async () => 'runtime-token'), + cellIncarnation, + ready: vi.fn(async () => true) + }) + const response = await post(app, 'rehome-token', { + ...request, + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceAssignmentEpoch: 1, + graceMs: 0 + }) + + expect(response.status).toBe(409) + expect(drainHost).not.toHaveBeenCalled() + }) + + it('rechecks synthetic host absence after asynchronous identity proof', async () => { + let hostExists = false + const drainHost = vi.fn(() => 'host-not-connected' as const) + const app = createRelayApp(config(), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost, + regionalRehomeTrustProbeHostExists: vi.fn(() => hostExists), + regionalRehomeIdentityToken: vi.fn(async () => { + hostExists = true + return 'runtime-token' + }), + cellIncarnation, + ready: vi.fn(async () => true) + }) + const response = await post(app, 'rehome-token', { + ...request, + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceAssignmentEpoch: 1, + graceMs: 0 + }) + + expect(response.status).toBe(409) + expect(drainHost).not.toHaveBeenCalled() + }) + + it('fails closed when the cell cannot prove its runtime identity rejection', async () => { + const drainHost = vi.fn(() => 'host-not-connected' as const) + const app = createRelayApp(config(), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + drainHost, + regionalRehomeTrustProbeHostExists: vi.fn(() => false), + regionalRehomeIdentityToken: vi.fn(async () => 'rehome-token'), + cellIncarnation, + ready: vi.fn(async () => true) + }) + const response = await post(app, 'rehome-token', { + ...request, + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceAssignmentEpoch: 1, + graceMs: 0 + }) + + expect(response.status).toBe(409) + expect(drainHost).not.toHaveBeenCalled() + }) +}) + +describe('regional rehome director controls', () => { + it('records cell capability separately from the legacy heartbeat contract', async () => { + const recordCellRegionalRehomeStatus = vi.fn().mockResolvedValue(undefined) + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: { recordCellRegionalRehomeStatus } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + const body = { + v: 1, + cellId: 'production-gce-c7', + cellIncarnation, + regionalRehomeProtocol: 1, + safety: { + observedAt: 100, + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } + } + const response = await postPath( + app, + '/v1/admin/cell-rehome-status', + 'runtime-token', + body + ) + + expect(response.status).toBe(200) + expect(recordCellRegionalRehomeStatus).toHaveBeenCalledWith(body) + }) + + it('accepts the exact safety payload the cell heartbeat composes', async () => { + // Why: index.ts spreads the FULL pool-pressure counts into safety; a + // strict schema missing any produced field 400s every heartbeat (the + // 2026-08-15..26 outage that left all cells at rehome protocol 0). + const recordCellRegionalRehomeStatus = vi.fn().mockResolvedValue(undefined) + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: { recordCellRegionalRehomeStatus } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + const observability = new RelayObservability( + { role: 'cell', cellId: 'production-gce-c7', region: 'us-central1' }, + () => {} + ) + observability.stop() + const body = { + v: 1, + cellId: 'production-gce-c7', + cellIncarnation, + regionalRehomeProtocol: 1, + safety: { + ...observability.regionalRehomeRuntimeSafety(), + ...emptyPostgresPoolPressureCounts() + } + } + const response = await postPath( + app, + '/v1/admin/cell-rehome-status', + 'runtime-token', + body + ) + + expect(response.status).toBe(200) + expect(recordCellRegionalRehomeStatus).toHaveBeenCalledWith(body) + }) + + it('applies the durable switch only with matching confirmation', async () => { + const inspectRegionalRehomeControl = vi.fn().mockResolvedValue({ + generation: 0, + enabled: false + }) + const applyRegionalRehomeControl = vi.fn(async (input) => ({ + ...input, + generation: input.expectedGeneration + 1 + })) + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: { + inspectRegionalRehomeControl, + applyRegionalRehomeControl + } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + expect((await postPath( + app, + '/v1/admin/regional-rehome-control', + 'deploy-token', + { v: 1, action: 'inspect' } + )).status).toBe(200) + const apply = { + v: 1, + action: 'apply', + expectedGeneration: 0, + enabled: true, + notBefore: 100, + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60_000, + confirmation: 'ENABLE_REGIONAL_REHOMING' + } + expect((await postPath( + app, + '/v1/admin/regional-rehome-control', + 'deploy-token', + apply + )).status).toBe(200) + expect(applyRegionalRehomeControl).toHaveBeenCalledOnce() + expect((await postPath( + app, + '/v1/admin/regional-rehome-control', + 'monitor-token', + { v: 1, action: 'inspect' } + )).status).toBe(200) + expect((await postPath( + app, + '/v1/admin/regional-rehome-control', + 'monitor-token', + apply + )).status).toBe(403) + expect((await postPath( + app, + '/v1/admin/regional-rehome-control', + 'deploy-token', + { ...apply, confirmation: 'DISABLE_REGIONAL_REHOMING' } + )).status).toBe(400) + }) + + it('probes dedicated trust twice and returns only aggregate proof', async () => { + const requests: Array<{ url: string; init?: RequestInit }> = [] + const cellDeploymentStatus = vi.fn().mockResolvedValue({ + cellId: 'production-gce-c7', + cellUrl: 'https://c7.relay.example.test', + region: 'us-central1', + runtime: { + cellIncarnation, + ready: true, + heartbeatFresh: true, + regionalRehomeProtocol: 1 + } + }) + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: { cellDeploymentStatus } as never, + drain: vi.fn(), + regionalRehomeIdentityToken: vi.fn(async () => 'rehome-token'), + regionalRehomeFetch: (async (url, init) => { + requests.push({ url: String(url), init }) + return Response.json({ + v: 1, + outcome: 'host-not-connected', + sharedRuntimeIdentityRejected: true + }) + }) as typeof fetch, + ready: vi.fn(async () => true) + }) + const response = await postPath( + app, + '/v1/admin/regional-rehome-trust-probe', + 'deploy-token', + { v: 1, sourceCellId: 'production-gce-c7', sourceCellIncarnation: cellIncarnation } + ) + + expect(response.status).toBe(200) + const responseBody = await response.json() + expect(responseBody).toEqual({ + v: 1, + dedicatedIdentity: { + accepted: true, + firstOutcome: 'host-not-connected', + secondOutcome: 'host-not-connected', + idempotent: true + }, + sharedRuntimeIdentityRejected: true, + proven: true + }) + expect(requests).toHaveLength(2) + expect(requests.map(({ url }) => url)).toEqual([ + 'https://c7.relay.example.test/v1/admin/host-drain', + 'https://c7.relay.example.test/v1/admin/host-drain' + ]) + const bodies = requests.map(({ init }) => JSON.parse(String(init?.body))) + expect(bodies[0]).toEqual(bodies[1]) + expect(bodies[0]).toMatchObject({ + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceAssignmentEpoch: 1, + graceMs: 0 + }) + expect(requests.every(({ init }) => init?.signal instanceof AbortSignal)).toBe(true) + expect(JSON.stringify(responseBody)).not.toContain('rehome-token') + }) + + it('restricts trust probes to deploy authorization and strict input', async () => { + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: {} as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + const body = { + v: 1, + sourceCellId: 'production-gce-c7', + sourceCellIncarnation: cellIncarnation + } + expect((await postPath( + app, + '/v1/admin/regional-rehome-trust-probe', + 'monitor-token', + body + )).status).toBe(401) + expect((await postPath( + app, + '/v1/admin/regional-rehome-trust-probe', + 'deploy-token', + { ...body, unexpected: true } + )).status).toBe(400) + }) + + it('fails closed when the source rejects the dedicated identity', async () => { + const cellDeploymentStatus = vi.fn().mockResolvedValue({ + cellUrl: 'https://c7.relay.example.test', + region: 'us-central1', + runtime: { + cellIncarnation, + ready: true, + heartbeatFresh: true, + regionalRehomeProtocol: 1 + } + }) + const sourceFetch = vi.fn().mockResolvedValue( + Response.json({ error: 'invalid_token' }, { status: 401 }) + ) + const app = createRelayApp(config({ role: 'director', cellId: 'director' }), { + store: {} as never, + assignments: { cellDeploymentStatus } as never, + drain: vi.fn(), + regionalRehomeIdentityToken: vi.fn(async () => 'rehome-token'), + regionalRehomeFetch: sourceFetch, + ready: vi.fn(async () => true) + }) + const response = await postPath( + app, + '/v1/admin/regional-rehome-trust-probe', + 'deploy-token', + { v: 1, sourceCellId: 'production-gce-c7', sourceCellIncarnation: cellIncarnation } + ) + + expect(response.status).toBe(409) + expect(sourceFetch).toHaveBeenCalledOnce() + expect(JSON.stringify(await response.json())).not.toContain('rehome-token') + }) +}) + +async function post( + app: ReturnType, + token: string, + body: unknown +): Promise { + return await app.request('/v1/admin/host-drain', { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body) + }) +} + +async function postPath( + app: ReturnType, + path: string, + token: string, + body: unknown +): Promise { + return await app.request(path, { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body) + }) +} + +function config(overrides: Partial = {}): RelayConfig { + return { + port: 8080, + publicUrl: 'https://c7.relay.example.test', + cellUrl: 'https://c7.relay.example.test', + region: 'us-central1', + authIssuer: 'https://auth.example.test', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.test/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c7', + cells: [], + adminAudience: 'https://relay.example.test/v1/admin/drain', + deployServiceAccount: 'deploy@example.test', + rehomeDirectorServiceAccount: 'relay-director@example.test', + rehomeAudience: 'https://relay.example.test/v1/admin/host-drain', + runtimeServiceAccount: 'relay-cell@example.test', + adminJwksUrl: 'https://auth.example.test/jwks', + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './data', + ...overrides + } +} diff --git a/cloud/apps/relay/src/regional-rehome-postgres.test.ts b/cloud/apps/relay/src/regional-rehome-postgres.test.ts new file mode 100644 index 00000000000..d36e26ecd68 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-postgres.test.ts @@ -0,0 +1,552 @@ +import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' +import { + REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT, + REGIONAL_REHOME_SQL_FAILURES_LIMIT, + REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT +} from './regional-rehome-safety.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip + +describePostgres('PostgreSQL regional rehoming', () => { + let primary: RelayDatabase + let secondary: RelayDatabase + let sequence = 0 + + beforeAll(async () => { + primary = await openRelayDatabase({ databaseUrl, dataDir: '' }) + secondary = await openRelayDatabase({ databaseUrl, dataDir: '' }) + }) + + beforeEach(async () => await cleanup()) + + afterAll(async () => { + await cleanup() + await secondary.close() + await primary.close() + }) + + async function cleanup(): Promise { + await primary.query( + `DELETE FROM relay_region_rehome_attempts WHERE user_id LIKE 'pg-rehome-user-%'` + ) + await primary.query(`DELETE FROM relay_region_rehome_worker_state`) + await primary.query(`DELETE FROM relay_region_rehome_control`) + await primary.query( + `DELETE FROM relay_control_connection_reservations + WHERE user_id LIKE 'pg-rehome-user-%'` + ) + await primary.query( + `DELETE FROM relay_assignment_migration_incarnations + WHERE user_id LIKE 'pg-rehome-user-%'` + ) + await primary.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id LIKE 'pg-rehome-user-%'` + ) + await primary.query( + `DELETE FROM relay_assignment_migrations WHERE user_id LIKE 'pg-rehome-user-%'` + ) + await primary.query( + `DELETE FROM relay_assignment_region_preferences + WHERE user_id LIKE 'pg-rehome-user-%'` + ) + await primary.query(`DELETE FROM relay_assignments WHERE user_id LIKE 'pg-rehome-user-%'`) + for (const table of [ + 'relay_cell_rehome_safety', + 'relay_cell_capabilities', + 'relay_cell_connection_snapshots', + 'relay_cell_connection_runtime', + 'relay_cell_runtime', + 'relay_cell_connection_limits', + 'relay_cell_admission', + 'relay_cell_regions', + 'relay_cells' + ]) { + await primary.query(`DELETE FROM ${table} WHERE cell_id LIKE 'pg-rehome-cell-%'`) + } + } + + it('claims through ambient per-cell sql retry noise', async () => { + const context = await fixture() + await primary.query( + `UPDATE relay_cell_rehome_safety + SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT} + WHERE cell_id IN (?, ?)`, + [context.source.id, context.target.id] + ) + + expect(await context.store.claimRegionalRehome()).not.toBeNull() + }) + + it('skips an unclean cell without latching the control off', async () => { + const context = await fixture() + await primary.query( + `UPDATE relay_cell_rehome_safety + SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT + 1} + WHERE cell_id = ?`, + [context.target.id] + ) + + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 1, + enabled: true + }) + expect(await primary.query( + `SELECT next_dispatch_at FROM relay_region_rehome_worker_state` + )).toEqual([{ next_dispatch_at: String(context.now() + 6_000) }]) + }) + + it('lets only one director claim a host', async () => { + const context = await fixture() + const claims = await Promise.all([ + context.store.claimRegionalRehome(), + context.competingStore.claimRegionalRehome() + ]) + + expect(claims.filter(Boolean)).toHaveLength(1) + expect(await primary.query( + `SELECT COUNT(*) AS count FROM relay_region_rehome_attempts + WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ count: '1' }]) + expect(await primary.query( + `SELECT COUNT(*) AS count FROM relay_assignment_migrations + WHERE user_id = ? AND completed_at IS NULL AND aborted_at IS NULL`, + [context.identity.userId] + )).toEqual([{ count: '1' }]) + }) + + it('increments the disable generation once across competing directors', async () => { + const context = await fixture() + const disabled = await Promise.all([ + context.store.disableRegionalRehomeControl(), + context.competingStore.disableRegionalRehomeControl() + ]) + + expect(disabled.sort()).toEqual([false, true]) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + }) + + it('records one receipt across competing directors', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + const receipts = await Promise.all([ + context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted'), + context.competingStore.recordRegionalRehomeDrainReceipt( + attempt!.attemptId, + 'accepted' + ) + ]) + + expect(receipts.sort()).toEqual([false, true]) + expect(await primary.query( + `SELECT drain_outcome FROM relay_region_rehome_attempts WHERE attempt_id = ?`, + [attempt!.attemptId] + )).toEqual([{ drain_outcome: 'accepted' }]) + }) + + it('rechecks a preference changed while the assignment row is locked', async () => { + const context = await fixture() + let unlock!: () => void + let locked!: () => void + const lockedPromise = new Promise((resolve) => (locked = resolve)) + const unlockPromise = new Promise((resolve) => (unlock = resolve)) + const held = secondary.transaction(async (transaction) => { + await transaction.queryLocked( + `SELECT * FROM relay_assignments WHERE user_id = ? AND relay_host_id = ?`, + [context.identity.userId, context.identity.relayHostId] + ) + locked() + await unlockPromise + }) + await lockedPromise + const claim = context.store.claimRegionalRehome() + await primary.query( + `UPDATE relay_assignment_region_preferences SET preferred_region = 'us-central1', + observed_at = ? WHERE user_id = ? AND relay_host_id = ?`, + [context.now(), context.identity.userId, context.identity.relayHostId] + ) + unlock() + await held + + await expect(claim).resolves.toBeNull() + expect(await primary.query( + `SELECT COUNT(*) AS count FROM relay_assignment_migrations WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ count: '0' }]) + }) + + it('rechecks fleet safety under locks before mutating a candidate', async () => { + const context = await fixture() + let unlock!: () => void + let locked!: () => void + const lockedPromise = new Promise((resolve) => (locked = resolve)) + const unlockPromise = new Promise((resolve) => (unlock = resolve)) + const held = secondary.transaction(async (transaction) => { + await transaction.queryLocked( + `SELECT * FROM relay_cell_rehome_safety WHERE cell_id = ?`, + [context.target.id] + ) + await transaction.query( + `UPDATE relay_cell_rehome_safety SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_LIMIT + 1} WHERE cell_id = ?`, + [context.target.id] + ) + locked() + await unlockPromise + }) + await lockedPromise + const claim = context.store.claimRegionalRehome() + unlock() + await held + + await expect(claim).resolves.toBeNull() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + expect(await primary.query( + `SELECT COUNT(*) AS count FROM relay_assignment_migrations WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ count: '0' }]) + }) + + it('pauses when one required cell exceeds the reconnect limit', async () => { + const context = await fixture() + await primary.query( + `UPDATE relay_cell_rehome_safety SET reconnects = ? WHERE cell_id = ?`, + [REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT + 1, context.source.id] + ) + + await expect(context.store.claimRegionalRehome()).resolves.toBeNull() + await expect(context.store.inspectRegionalRehomeControl()).resolves.toMatchObject({ + generation: 2, + enabled: false + }) + expect(await primary.query( + `SELECT COUNT(*) AS count FROM relay_assignment_migrations WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ count: '0' }]) + }) + + it('does not retry a drain against a replacement source incarnation', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + context.advance(31_000) + await heartbeat( + context.store, + context.source, + '33333333-3333-4333-8333-333333333333', + 1, + context.now() + ) + + await expect(context.competingStore.claimRegionalRehome()).resolves.toBeNull() + expect(await primary.query( + `SELECT send_attempts FROM relay_region_rehome_attempts WHERE attempt_id = ?`, + [attempt!.attemptId] + )).toEqual([{ send_attempts: '1' }]) + }) + + it('makes concurrent completion and expiry cleanup idempotent', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + const targetControl = await context.store.activateControl(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(context.identity, context.sourceControl) + context.advance(24 * 60 * 60_000) + await heartbeat( + context.store, + context.source, + '11111111-1111-4111-8111-111111111111', + 1, + 900_000, + 2 + ) + await heartbeat( + context.store, + context.target, + '22222222-2222-4222-8222-222222222222', + 0, + 900_000, + 2 + ) + await context.store.renewControlActivity(context.identity, { + activityId: targetControl, + cellId: context.target.id, + expiresAt: context.now() + 90_000 + }) + + const outcomes = await Promise.all([ + context.store.completeReadyRegionalRehomes(), + context.competingStore.abortExpiredRegionalRehomes() + ]) + expect(outcomes).toEqual(expect.arrayContaining([0, 1])) + expect(await primary.query( + `SELECT completed_at IS NOT NULL AS completed, aborted_at IS NOT NULL AS aborted + FROM relay_assignment_migrations WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ completed: true, aborted: false }]) + }) + + it('will not complete against a replacement target incarnation', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(context.identity, context.sourceControl) + context.advance(1) + await heartbeat( + context.store, + context.target, + '44444444-4444-4444-8444-444444444444', + 0, + context.now() + ) + + await expect(context.store.completeReadyRegionalRehomes()).resolves.toBe(0) + expect(await primary.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ completed_at: null, aborted_at: null }]) + }) + + it('does not roll an unregistered target back to a stale regional source', async () => { + const context = await fixture() + await context.store.claimRegionalRehome() + context.advance(6 * 60_000) + await heartbeat( + context.store, + context.target, + '22222222-2222-4222-8222-222222222222', + 0, + 900_000, + 2 + ) + + await expect(context.store.refreshRegionalRehomeLeases()).resolves.toBe(0) + await expect(context.store.abortExpiredEvacuations()).resolves.toBe(0) + expect(await primary.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ cell_id: context.target.id, assignment_epoch: '2' }]) + }) + + it('completes after the drained host re-resolves through the director', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + // The drain recovery lands while both controls are still live. + await context.store.assign(context.identity, 'asia-east2') + expect(await controlAccounting(context.identity)).toEqual({ + reservedControls: 2, + controlLeases: 2 + }) + + await context.store.activateControl(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(context.identity, context.sourceControl) + + await expect(context.store.completeReadyRegionalRehomes()).resolves.toBe(1) + expect(await primary.query( + `SELECT completed_at IS NOT NULL AS completed FROM relay_assignment_migrations + WHERE user_id = ?`, + [context.identity.userId] + )).toEqual([{ completed: true }]) + expect(await controlAccounting(context.identity)).toEqual({ + reservedControls: 1, + controlLeases: 1 + }) + }) + + it('repairs a skewed control counter before completing the rehome', async () => { + const context = await fixture() + const attempt = await context.store.claimRegionalRehome() + await context.store.activateControl(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(context.identity, { + cellId: context.target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(context.identity, context.sourceControl) + // Damage already written by a pre-fix sticky grant. + await primary.query( + `UPDATE relay_assignments SET reserved_controls = 0 WHERE user_id = ?`, + [context.identity.userId] + ) + + await expect(context.store.completeReadyRegionalRehomes()).resolves.toBe(1) + expect(await controlAccounting(context.identity)).toEqual({ + reservedControls: 1, + controlLeases: 1 + }) + }) + + async function controlAccounting(identity: { + userId: string + relayHostId: string + }): Promise<{ reservedControls: number; controlLeases: number }> { + const assignment = ( + await primary.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0]! + const leases = await primary.query( + `SELECT COUNT(*) AS controls FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control'`, + [identity.userId, identity.relayHostId] + ) + return { + reservedControls: Number(assignment.reserved_controls), + controlLeases: Number(leases[0]!.controls) + } + } + + async function fixture() { + sequence++ + let now = 1_000_000 + const suffix = String(sequence) + const source = cell(suffix, 'source', 'us-central1') + const target = cell(suffix, 'target', 'asia-east2') + const store = new RelayAssignmentStore(primary, () => now, storeOptions) + const competingStore = new RelayAssignmentStore(secondary, () => now, storeOptions) + await store.inspectRegionalRehomeControl() + now += 24 * 60 * 60_000 + await store.applyRegionalRehomeControl({ + expectedGeneration: 0, + enabled: true, + notBefore: now, + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60_000 + }) + await store.reconcileCells([source, target]) + await heartbeat( + store, + source, + '11111111-1111-4111-8111-111111111111', + 1, + 900_000 + ) + await heartbeat( + store, + target, + '22222222-2222-4222-8222-222222222222', + 0, + 900_000 + ) + const identity = { + userId: `pg-rehome-user-${suffix}`, + relayHostId: `rehomehost${suffix.padStart(6, '0')}` + } + const assignment = await store.assign(identity, undefined, 'us-central1') + const sourceControl = await store.activateControl(identity, { + cellId: source.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.assign(identity, 'asia-east2') + return { + store, + competingStore, + identity, + source, + target, + sourceControl, + now: () => now, + advance: (milliseconds: number) => { + now += milliseconds + } + } + } +}) + +const storeOptions = { + requireLiveCells: true, + heartbeatTtlMs: 45_000 +} + +function cell(suffix: string, role: string, region: 'us-central1' | 'asia-east2') { + return { + id: `pg-rehome-cell-${suffix}-${role}`, + url: `https://pg-rehome-${suffix}-${role}.example.test`, + region, + capacityRequests: 100, + connectionHardCap: 1_000 as const, + connectionUnobservedBound: 60 + } +} + +async function heartbeat( + store: RelayAssignmentStore, + cellConfig: ReturnType, + cellIncarnation: string, + regionalRehomeProtocol: number, + startedAt: number, + connectionInclusionWatermark = 1 +): Promise { + await store.recordCellHeartbeat({ + cellId: cellConfig.id, + cellUrl: cellConfig.url, + region: cellConfig.region, + cellIncarnation, + startedAt, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + }) + await store.recordCellRegionalRehomeStatus({ + cellId: cellConfig.id, + cellIncarnation, + regionalRehomeProtocol, + safety: { + observedAt: 1_000_000 + 24 * 60 * 60_000, + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } + }) +} diff --git a/cloud/apps/relay/src/regional-rehome-safety.test.ts b/cloud/apps/relay/src/regional-rehome-safety.test.ts new file mode 100644 index 00000000000..5c7d10d37a0 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-safety.test.ts @@ -0,0 +1,109 @@ +import { describe, expect, it } from 'vitest' +import { + REGIONAL_REHOME_POOL_WAIT_MS_MAX_LIMIT, + REGIONAL_REHOME_POOL_WAITERS_MAX_LIMIT, + REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT, + REGIONAL_REHOME_SQL_FAILURES_LIMIT, + regionalRehomeSafetyFailure +} from './regional-rehome-safety.js' + +const NOW = 1_787_900_000_000 + +function safety(overrides: Partial[0]> = {}) { + return { + observedAt: NOW - 1_000, + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0, + ...overrides + } +} + +describe('regionalRehomeSafetyFailure', () => { + it('passes the measured healthy-fleet baseline of pool micro-waits', () => { + // Every production cell idles at 1-2 peak waiters resolved in ~1ms; a + // zero-tolerance bar here disables the worker on its first tick. + expect( + regionalRehomeSafetyFailure( + safety({ databasePoolWaitersMax: 2, databasePoolWaitMsMax: 1 }), + NOW, + 19 + ) + ).toBeNull() + }) + + it('passes routine client reconnect churn', () => { + expect( + regionalRehomeSafetyFailure(safety({ reconnects: 19 * 80 }), NOW, 19) + ).toBeNull() + }) + + it('still fails closed on each pool pressure bound', () => { + for (const overrides of [ + { databasePoolWaitersMax: REGIONAL_REHOME_POOL_WAITERS_MAX_LIMIT + 1 }, + { databasePoolWaitMsMax: REGIONAL_REHOME_POOL_WAIT_MS_MAX_LIMIT + 1 } + ]) { + expect(regionalRehomeSafetyFailure(safety(overrides), NOW, 19)).toBe( + 'database_pool_pressure' + ) + } + }) + + it('still fails closed on a reconnect storm', () => { + expect( + regionalRehomeSafetyFailure( + safety({ reconnects: 19 * REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT + 1 }), + NOW, + 19 + ) + ).toBe('elevated_reconnects') + }) + + it('passes ambient 55P03 retry noise', () => { + // Fleet-wide retried lock timeouts peaked at 82 counted failures per + // minute over Aug 25-28; the combined snapshot can span two windows. + expect(regionalRehomeSafetyFailure(safety({ sqlFailures: 164 }), NOW, 19)).toBeNull() + }) + + it('still fails closed on a sql failure storm', () => { + expect( + regionalRehomeSafetyFailure( + safety({ sqlFailures: REGIONAL_REHOME_SQL_FAILURES_LIMIT + 1 }), + NOW, + 19 + ) + ).toBe('sql_failures') + // Literal storm magnitude (measured 2026-08-28) pins the bar itself: the + // limit must sit below real storm scale, not merely exist. + expect(regionalRehomeSafetyFailure(safety({ sqlFailures: 395 }), NOW, 19)).toBe( + 'sql_failures' + ) + // The bar is exclusive: exactly at the limit still passes. + expect( + regionalRehomeSafetyFailure( + safety({ sqlFailures: REGIONAL_REHOME_SQL_FAILURES_LIMIT }), + NOW, + 19 + ) + ).toBeNull() + }) + + it('keeps zero-tolerance for real failure signals', () => { + expect( + regionalRehomeSafetyFailure( + safety({ controlActivityRecoveryFailures: 1 }), + NOW, + 19 + ) + ).toBe('control_recovery_failures') + expect(regionalRehomeSafetyFailure(safety({ observedAt: 0 }), NOW, 19)).toBe( + 'monitoring_stale' + ) + expect( + regionalRehomeSafetyFailure(safety({ observedAt: NOW - 61_000 }), NOW, 19) + ).toBe('monitoring_stale') + }) +}) diff --git a/cloud/apps/relay/src/regional-rehome-safety.ts b/cloud/apps/relay/src/regional-rehome-safety.ts new file mode 100644 index 00000000000..4da51f71b0d --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-safety.ts @@ -0,0 +1,86 @@ +import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' + +// Limits sit well above the healthy-fleet baseline measured in production on +// 2026-08-28 (peak 2 waiters / 1ms pool waits on every cell; up to ~80 +// reconnects per published two-window row on the busiest cell). Sustained +// pool saturation still trips: waiters-max 16 is 8x baseline yet far under a +// backed-up pool, and 250ms peak wait is 1/10 of the incident-monitor alert. +// Instantaneous databasePoolWaiting is not checked separately: it is bounded +// by databasePoolWaitersMax within every published window. +export const REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT = 250 +export const REGIONAL_REHOME_POOL_WAITERS_MAX_LIMIT = 16 +export const REGIONAL_REHOME_POOL_WAIT_MS_MAX_LIMIT = 250 + +// Counted SQL failures are dominated by relay_cells 55P03 lock-timeout +// retries the transaction wrapper heals in place (Aug 25-28: ~1000 retried +// attempts per day vs ~30 exhausted, those clustered in one storm), so a +// zero bar disabled the worker on ambient noise just like the original pool +// bars. Fleet-wide ambient noise peaked at 82 counted failures per minute +// over four days; genuine database distress produced 395-457. The combined +// snapshot spans up to two 30s windows per process (pathological ambient +// alignment ~164), so 250 stays clear of noise while storms still trip. +// Terminal outages also trip the pool bars and the worker's own +// dispatch-failure budget; the sql bar only needs to catch storms. +export const REGIONAL_REHOME_SQL_FAILURES_LIMIT = 250 +// Per-cell candidate cleanliness is a soft skip, not a durable latch; the +// worst ambient per-cell publish carried ~24 counted failures (two windows +// of 12). The fleet bar deliberately dominates: cells that are individually +// clean can sum past 250, and a fleet-wide sum at that scale is a storm. +export const REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT = 40 + +export function regionalRehomePoolPressure(safety: { + databasePoolWaitersMax: number + databasePoolWaitMsMax: number +}): boolean { + return ( + safety.databasePoolWaitersMax > REGIONAL_REHOME_POOL_WAITERS_MAX_LIMIT || + safety.databasePoolWaitMsMax > REGIONAL_REHOME_POOL_WAIT_MS_MAX_LIMIT + ) +} + +export function regionalRehomeSafetyFailure( + safety: RegionalRehomeSafetySnapshot, + now: number, + requiredCells: number +): string | null { + if (safety.observedAt === 0 || now - safety.observedAt > 60_000) { + return 'monitoring_stale' + } + if (safety.sqlFailures > REGIONAL_REHOME_SQL_FAILURES_LIMIT) return 'sql_failures' + if (regionalRehomePoolPressure(safety)) { + return 'database_pool_pressure' + } + if (safety.controlActivityRecoveryFailures > 0) { + return 'control_recovery_failures' + } + const reconnectLimit = + Math.max(1, requiredCells) * REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT + if (safety.reconnects > reconnectLimit) return 'elevated_reconnects' + return null +} + +export function combineRegionalRehomeSafety( + processSafety: RegionalRehomeSafetySnapshot, + fleetSafety: RegionalRehomeSafetySnapshot +): RegionalRehomeSafetySnapshot { + return { + observedAt: Math.min(processSafety.observedAt, fleetSafety.observedAt), + sqlFailures: processSafety.sqlFailures + fleetSafety.sqlFailures, + reconnects: fleetSafety.reconnects, + controlActivityRecoveryFailures: + processSafety.controlActivityRecoveryFailures + + fleetSafety.controlActivityRecoveryFailures, + databasePoolWaiting: Math.max( + processSafety.databasePoolWaiting, + fleetSafety.databasePoolWaiting + ), + databasePoolWaitersMax: Math.max( + processSafety.databasePoolWaitersMax, + fleetSafety.databasePoolWaitersMax + ), + databasePoolWaitMsMax: Math.max( + processSafety.databasePoolWaitMsMax, + fleetSafety.databasePoolWaitMsMax + ) + } +} diff --git a/cloud/apps/relay/src/regional-rehome-store.test.ts b/cloud/apps/relay/src/regional-rehome-store.test.ts new file mode 100644 index 00000000000..43f293b1131 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-store.test.ts @@ -0,0 +1,1399 @@ +import { describe, expect, it } from 'vitest' +import { + RelayAssignmentStore, + REGIONAL_REHOME_QUARANTINE_FAILURES, + REGIONAL_REHOME_QUARANTINE_MS, + REGIONAL_REHOME_REDRAIN_SEND_LIMIT +} from './assignment-store.js' +import { openInMemoryRelayDatabase, type RelayDatabase, type SqlRow } from './database.js' +import { + REGIONAL_REHOME_SQL_FAILURES_LIMIT, + REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT +} from './regional-rehome-safety.js' +import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' + +const source = { + id: 'us-c1', + url: 'https://us-c1.relay.example.test', + region: 'us-central1' as const, + capacityRequests: 100, + connectionHardCap: 1_000 as const, + connectionUnobservedBound: 60 +} +const target = { + id: 'asia-c1', + url: 'https://asia-c1.relay.example.test', + region: 'asia-east2' as const, + capacityRequests: 100, + connectionHardCap: 1_000 as const, + connectionUnobservedBound: 60 +} +const sourceIncarnation = '11111111-1111-4111-8111-111111111111' +const targetIncarnation = '22222222-2222-4222-8222-222222222222' + +describe('regional rehome assignment state', () => { + it('does not open a transaction while the worker is disabled', async () => { + const delegate = await openInMemoryRelayDatabase() + const database = new TransactionCountingDatabase(delegate) + const store = new RelayAssignmentStore(database, () => 1_000_000) + await store.inspectRegionalRehomeControl() + database.transactionCalls = 0 + + await expect(store.claimRegionalRehome()).resolves.toBeNull() + expect(database.transactionCalls).toBe(0) + await database.close() + }) + + it('initializes a missing control row without opening a transaction', async () => { + const delegate = await openInMemoryRelayDatabase() + const database = new TransactionCountingDatabase(delegate) + const store = new RelayAssignmentStore(database, () => 1_000_000) + + await expect(store.claimRegionalRehome()).resolves.toBeNull() + expect(database.transactionCalls).toBe(0) + await expect(store.inspectRegionalRehomeControl()).resolves.toMatchObject({ + generation: 0, + enabled: false + }) + await database.close() + }) + + it('uses a generation-bound durable kill switch', async () => { + const context = await setup() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 1, + enabled: true, + ratePerMinute: 10 + }) + expect(await context.store.disableRegionalRehomeControl()).toBe(true) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + await expect(context.store.applyRegionalRehomeControl({ + expectedGeneration: 1, + enabled: true, + notBefore: context.now(), + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60_000 + })).rejects.toThrow('regional_rehome_generation_mismatch') + await expect(context.store.applyRegionalRehomeControl({ + expectedGeneration: 2, + enabled: true, + notBefore: context.now(), + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60_000 + })).resolves.toMatchObject({ generation: 3, enabled: true }) + await context.database.close() + }) + + it('moves one live preferred host and completes without disabling its source cell', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const neighbor = { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' } + const sourceControl = await activatePreferredSource(context, identity) + await activateSource(context, neighbor) + + const attempt = await context.store.claimRegionalRehome() + expect(attempt).toMatchObject({ + userId: identity.userId, + relayHostId: identity.relayHostId, + sourceCellId: source.id, + sourceCellIncarnation: sourceIncarnation, + targetCellId: target.id, + targetCellIncarnation: targetIncarnation, + previousEpoch: 1, + assignmentEpoch: 2, + sendAttempts: 1 + }) + expect(await context.store.resolve(neighbor)).toMatchObject({ cellId: source.id }) + expect(await context.store.completeReadyRegionalRehomes()).toBe(0) + expect( + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + ).toBe(true) + expect( + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + ).toBe(false) + + const targetControl = await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + expect(await context.store.completeReadyRegionalRehomes()).toBe(0) + await context.store.releaseActivity(identity, sourceControl) + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + + expect(await context.store.resolve(identity)).toMatchObject({ + cellId: target.id, + assignmentEpoch: 2 + }) + expect(await context.store.resolve(neighbor)).toMatchObject({ cellId: source.id }) + expect(await context.database.query( + `SELECT completed_at, aborted_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + )).toEqual([{ completed_at: context.now(), aborted_at: null }]) + expect(await context.database.query( + `SELECT completed_at, aborted_at FROM relay_region_rehome_attempts` + )).toEqual([{ completed_at: context.now(), aborted_at: null }]) + expect(targetControl).toMatch(/^control:/) + await context.database.close() + }) + + it('completes from durable activity when the drain response was lost', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + expect(await context.database.query( + `SELECT drain_receipt_at, completed_at, aborted_at + FROM relay_region_rehome_attempts` + )).toEqual([{ drain_receipt_at: null, completed_at: context.now(), aborted_at: null }]) + await context.database.close() + }) + + it('requires a fresh preference and an advertised source capability', async () => { + const context = await setup({ sourceProtocol: 0 }) + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.database.query(`SELECT * FROM relay_assignment_migrations`)).toEqual([]) + await context.database.close() + }) + + it('fails fleet safety closed until source and target telemetry is fresh', async () => { + const context = await setup() + await context.database.query( + `DELETE FROM relay_cell_rehome_safety WHERE cell_id = ?`, + [target.id] + ) + expect(await context.store.regionalRehomeFleetSafety()).toMatchObject({ + requiredCells: 2, + missingCells: 1, + observedAt: 0 + }) + await heartbeat(context.store, source, sourceIncarnation, 1, 2, { + observedAt: context.now(), + sqlFailures: 0, + reconnects: 2, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + }) + await heartbeat(context.store, target, targetIncarnation, 0, 2, { + observedAt: context.now(), + sqlFailures: 1, + reconnects: 3, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + }) + expect(await context.store.regionalRehomeFleetSafety()).toMatchObject({ + requiredCells: 2, + missingCells: 0, + observedAt: context.now(), + sqlFailures: 1, + reconnects: 5 + }) + await context.database.close() + }) + + it('claims through the measured healthy baseline of pool micro-waits and churn', async () => { + const context = await setup() + const baseline = { + observedAt: context.now(), + sqlFailures: 0, + reconnects: 42, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 2, + databasePoolWaitersMax: 2, + databasePoolWaitMsMax: 1 + } + await heartbeat(context.store, source, sourceIncarnation, 1, 2, baseline) + await heartbeat(context.store, target, targetIncarnation, 0, 2, baseline) + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + + expect(await context.store.claimRegionalRehome()).toMatchObject({ + sourceCellId: source.id, + targetCellId: target.id + }) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + enabled: true + }) + await context.database.close() + }) + + it('latches off on a per-cell reconnect storm even when the fleet sum is low', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + await context.database.query( + `UPDATE relay_cell_rehome_safety SET reconnects = 251 WHERE cell_id = ?`, + [source.id] + ) + + const warnings = collectDisableWarnings() + try { + expect(await context.store.claimRegionalRehome()).toBeNull() + } finally { + warnings.restore() + } + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + expect(warnings.entries).toMatchObject([ + { reason: 'elevated_reconnects', maxReconnects: 251, controlGeneration: 2 } + ]) + await context.database.close() + }) + + it('latches off on sustained pool pressure and logs the disable exactly once', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + await context.database.query( + `UPDATE relay_cell_rehome_safety SET database_pool_waiters_max = 17 WHERE cell_id = ?`, + [target.id] + ) + + const warnings = collectDisableWarnings() + try { + expect(await context.store.claimRegionalRehome()).toBeNull() + // Already disabled: the next tick returns before the gate and stays silent. + expect(await context.store.claimRegionalRehome()).toBeNull() + } finally { + warnings.restore() + } + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + expect(warnings.entries).toMatchObject([ + { reason: 'database_pool_pressure', databasePoolWaitersMax: 17 } + ]) + await context.database.close() + }) + + it('claims through ambient per-cell sql retry noise', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + await context.database.query( + `UPDATE relay_cell_rehome_safety SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT}` + ) + + expect(await context.store.claimRegionalRehome()).not.toBeNull() + await context.database.close() + }) + + it('skips an unclean cell without latching the control off', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + await activatePreferredSource(context, { + userId: 'user-2', + relayHostId: 'ponmlkjihgfedcba' + }) + // Above the per-cell cleanliness bar but below the fleet storm bar: the + // candidate is skipped this tick while the worker stays enabled. + await context.database.query( + `UPDATE relay_cell_rehome_safety SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT + 1} WHERE cell_id = ?`, + [target.id] + ) + + const warnings = collectEventWarnings('orca_relay_regional_rehome_candidates_skipped') + try { + expect(await context.store.claimRegionalRehome()).toBeNull() + } finally { + warnings.restore() + } + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 1, + enabled: true + }) + // The skip is visible and named, and both candidates blocked by the one + // unclean cell accumulate into a single entry. + expect(warnings.entries).toMatchObject([ + { + skips: [ + { + reason: 'target_unclean', + cellId: target.id, + sqlFailures: REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT + 1, + candidates: 2 + } + ] + } + ]) + // A skipped tick is charged the dispatch interval: candidate scans stay + // rate-limited even when nothing claims. + expect( + await context.database.query( + `SELECT next_dispatch_at FROM relay_region_rehome_worker_state` + ) + ).toEqual([{ next_dispatch_at: context.now() + 6_000 }]) + expect(await context.database.query(`SELECT * FROM relay_assignment_migrations`)).toEqual([]) + await context.database.close() + }) + + it('does not throttle or log an idle tick with no candidates', async () => { + const context = await setup() + + const warnings = collectEventWarnings('orca_relay_regional_rehome_candidates_skipped') + try { + expect(await context.store.claimRegionalRehome()).toBeNull() + } finally { + warnings.restore() + } + expect(warnings.entries).toEqual([]) + expect( + await context.database.query( + `SELECT next_dispatch_at FROM relay_region_rehome_worker_state` + ) + ).toEqual([{ next_dispatch_at: 0 }]) + await context.database.close() + }) + + it('atomically latches durable control off when candidate safety changes', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + await context.database.query( + `UPDATE relay_cell_rehome_safety SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_LIMIT + 1} WHERE cell_id = ?`, + [target.id] + ) + + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + expect(await context.database.query(`SELECT * FROM relay_assignment_migrations`)).toEqual([]) + await context.database.close() + }) + + it('rechecks locked fleet safety before retrying a drain dispatch', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + expect(await context.store.claimRegionalRehome()).not.toBeNull() + context.advance(31_000) + await context.database.query( + `UPDATE relay_cell_rehome_safety SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_LIMIT + 1} WHERE cell_id = ?`, + [target.id] + ) + + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + await context.database.close() + }) + + it('latches off after three dispatch failures and resumes only through CAS', async () => { + const context = await setup() + await activatePreferredSource(context, { + userId: 'user-1', + relayHostId: 'abcdefghijklmnop' + }) + const first = await context.store.claimRegionalRehome() + for (let index = 0; index < 3; index++) { + await context.store.recordRegionalRehomeDispatchFailure(first!.attemptId) + } + context.advance(5 * 60_000 - 1) + expect(await context.store.claimRegionalRehome()).toBeNull() + context.advance(1) + await heartbeat(context.store, source, sourceIncarnation, 1, 2, { + observedAt: context.now(), + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + }) + await heartbeat(context.store, target, targetIncarnation, 0, 2, { + observedAt: context.now(), + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + }) + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 2, + enabled: false + }) + await context.store.applyRegionalRehomeControl({ + expectedGeneration: 2, + enabled: true, + notBefore: context.now(), + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60_000 + }) + const retry = await context.store.claimRegionalRehome() + expect(retry).toMatchObject({ attemptId: first!.attemptId, sendAttempts: 2 }) + expect(await context.database.query( + `SELECT COUNT(*) AS count FROM relay_assignment_migrations` + )).toEqual([{ count: 1 }]) + await context.database.close() + }) + + it('refreshes only the migration leases while source splices drain', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + await context.store.acquireActivity(identity, { + activityId: 'splice:source', + kind: 'splice', + cellId: source.id + }) + await context.store.claimRegionalRehome() + const before = await context.database.query( + `SELECT activity_id, expires_at FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY activity_id`, + [identity.userId, identity.relayHostId] + ) + context.advance(60_000) + expect(await context.store.refreshRegionalRehomeLeases()).toBe(1) + const after = await context.database.query( + `SELECT activity_id, expires_at FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? ORDER BY activity_id`, + [identity.userId, identity.relayHostId] + ) + const beforeById = new Map(before.map((row) => [row.activity_id, row.expires_at])) + const afterById = new Map(after.map((row) => [row.activity_id, row.expires_at])) + expect(Number(afterById.get('migration:2'))).toBeGreaterThan( + Number(beforeById.get('migration:2')) + ) + expect(Number(afterById.get('control-pending:2'))).toBeGreaterThan( + Number(beforeById.get('control-pending:2')) + ) + expect(afterById.get('splice:source')).toBe(beforeById.get('splice:source')) + await context.database.close() + }) + + it('stops refreshing an unregistered target and lets normal rollback retire it', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + await context.store.claimRegionalRehome() + context.advance(6 * 60_000) + expect(await context.store.refreshRegionalRehomeLeases()).toBe(0) + await heartbeat(context.store, source, sourceIncarnation, 1, 2) + expect(await context.store.abortExpiredEvacuations()).toBe(1) + expect(await context.store.reapRegionalRehomeAttempts()).toBe(1) + expect(await context.store.resolve(identity)).toMatchObject({ + cellId: source.id, + assignmentEpoch: 3 + }) + expect(await context.database.query( + `SELECT completed_at, aborted_at FROM relay_region_rehome_attempts` + )).toEqual([{ completed_at: null, aborted_at: context.now() }]) + await context.database.close() + }) + + it('does not roll an unregistered target back to a stale regional source', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + await context.store.claimRegionalRehome() + context.advance(6 * 60_000) + await heartbeat(context.store, target, targetIncarnation, 0, 2) + + expect(await context.store.refreshRegionalRehomeLeases()).toBe(0) + expect(await context.store.abortExpiredEvacuations()).toBe(0) + expect( + await context.database.query( + `SELECT cell_id, assignment_epoch FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ cell_id: target.id, assignment_epoch: 2 }]) + await context.database.close() + }) + + it('rolls back an inactive registered target only after the 24-hour bound', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt( + attempt!.attemptId, + 'accepted' + ) + const targetControl = await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.store.releaseActivity(identity, sourceControl) + await context.store.releaseActivity(identity, targetControl) + context.advance(24 * 60 * 60_000) + await heartbeat(context.store, source, sourceIncarnation, 1, 2) + expect(await context.store.abortExpiredRegionalRehomes()).toBe(1) + expect(await context.store.resolve(identity)).toMatchObject({ + cellId: source.id, + assignmentEpoch: 3 + }) + await context.database.close() + }) + + it('redrains a receipted dual-homed attempt once its grace elapses', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + + // Before grace elapses a receipted attempt is not re-dispatched. + context.advance(30 * 60_000) + await freshHeartbeats(context) + expect(await context.store.claimRegionalRehome()).toBeNull() + + context.advance(30 * 60_000 + 1) + await freshHeartbeats(context) + const redrain = await context.store.claimRegionalRehome() + expect(redrain).toMatchObject({ + attemptId: attempt!.attemptId, + drainGraceMs: 0, + sendAttempts: 2 + }) + // The per-dispatch receipt replaces the original without a mismatch. + await expect( + context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'host-not-connected') + ).resolves.toBe(true) + + // Redrains are spaced: nothing new inside the redrain interval. + context.advance(30_000) + await freshHeartbeats(context) + expect(await context.store.claimRegionalRehome()).toBeNull() + context.advance(30_001) + await freshHeartbeats(context) + expect(await context.store.claimRegionalRehome()).toMatchObject({ + attemptId: attempt!.attemptId, + drainGraceMs: 0, + sendAttempts: 3 + }) + + // Once the host actually leaves the source, completion wins over redrain. + await context.store.releaseActivity(identity, sourceControl) + context.advance(60_001) + await freshHeartbeats(context) + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 2 + }) + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + await context.database.close() + }) + + it('resets the failure budget on a repeated redrain receipt outcome', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + + context.advance(60 * 60_000 + 1) + await freshHeartbeats(context) + expect(await context.store.claimRegionalRehome()).toMatchObject({ + attemptId: attempt!.attemptId, + drainGraceMs: 0 + }) + // The repeated outcome still proves the source answered. + expect( + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + ).toBe(false) + await context.store.recordRegionalRehomeDispatchFailure(attempt!.attemptId) + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + generation: 1, + enabled: true + }) + await context.database.close() + }) + + it('does not redrain before the target registers or when the fleet is unsafe', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + + // Past grace but the target never registered: force-closing the source + // would disconnect the host with nowhere proven to land. + context.advance(60 * 60_000 + 1) + await freshHeartbeats(context) + expect(await context.store.claimRegionalRehome()).toBeNull() + + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.database.query( + `UPDATE relay_cell_rehome_safety SET sql_failures = ${REGIONAL_REHOME_SQL_FAILURES_LIMIT + 1} WHERE cell_id = ?`, + [target.id] + ) + expect(await context.store.claimRegionalRehome()).toBeNull() + expect(await context.store.inspectRegionalRehomeControl()).toMatchObject({ + enabled: false + }) + await context.database.close() + }) + + it('completes healthy candidates past a poisoned attempt and logs it', async () => { + const context = await setup() + const poisoned = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const healthy = { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' } + const poisonedSource = await activatePreferredSource(context, poisoned) + const healthySource = await activatePreferredSource(context, healthy) + const first = await context.store.claimRegionalRehome() + context.advance(6_000) + const second = await context.store.claimRegionalRehome() + expect(first!.userId).toBe(poisoned.userId) + expect(second!.userId).toBe(healthy.userId) + for (const [identity, attempt, sourceControl] of [ + [poisoned, first, poisonedSource], + [healthy, second, healthySource] + ] as const) { + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + } + // The production poison shape: the assignment moved past the attempt. + await context.database.query( + `UPDATE relay_assignments SET assignment_epoch = assignment_epoch + 5 + WHERE user_id = ?`, + [poisoned.userId] + ) + const warnings = collectCandidateFailureWarnings() + try { + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + } finally { + warnings.restore() + } + expect(warnings.entries).toEqual([ + { + event: 'orca_relay_regional_rehome_candidate_failed', + operation: 'complete', + attemptId: first!.attemptId, + reason: 'regional_rehome_assignment_mismatch' + } + ]) + expect(await context.database.query( + `SELECT completed_at FROM relay_region_rehome_attempts WHERE attempt_id = ?`, + [second!.attemptId] + )).toEqual([{ completed_at: context.now() }]) + await context.database.close() + }) + + it('quarantines a repeatedly failing candidate and redacts free-form errors', async () => { + const context = await setup() + const poisoned = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const source1 = await activatePreferredSource(context, poisoned) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(poisoned, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(poisoned, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(poisoned, source1) + await context.database.query( + `UPDATE relay_assignments SET assignment_epoch = assignment_epoch + 5 + WHERE user_id = ?`, + [poisoned.userId] + ) + const warnings = collectCandidateFailureWarnings() + try { + for (let round = 0; round < REGIONAL_REHOME_QUARANTINE_FAILURES; round++) { + expect(await context.store.completeReadyRegionalRehomes()).toBe(0) + } + expect(warnings.entries).toHaveLength(REGIONAL_REHOME_QUARANTINE_FAILURES) + // Quarantined: the poisoned row leaves the candidate page entirely. + expect(await context.store.completeReadyRegionalRehomes()).toBe(0) + expect(warnings.entries).toHaveLength(REGIONAL_REHOME_QUARANTINE_FAILURES) + // After the quarantine window it is retried (and fails) once more. + context.advance(REGIONAL_REHOME_QUARANTINE_MS + 1) + await context.store.completeReadyRegionalRehomes() + expect(warnings.entries).toHaveLength(REGIONAL_REHOME_QUARANTINE_FAILURES + 1) + // A free-form error (never a slug) reaches the log only as 'redacted'. + expect( + warnings.entries.every( + (entry) => entry.reason === 'regional_rehome_assignment_mismatch' + ) + ).toBe(true) + context.advance(REGIONAL_REHOME_QUARANTINE_MS + 1) + const database = context.database + const original = database.transaction.bind(database) + database.transaction = () => { + throw new Error('postgresql://secret@database.invalid/relay') + } + try { + await context.store.completeReadyRegionalRehomes() + } finally { + database.transaction = original + } + const last = warnings.entries.at(-1)! + expect(last.reason).toBe('redacted') + expect(JSON.stringify(last)).not.toContain('secret') + } finally { + warnings.restore() + } + await context.database.close() + }) + + it('aborts healthy expired candidates past a poisoned attempt', async () => { + const context = await setup() + const poisoned = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const healthy = { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' } + const poisonedSource = await activatePreferredSource(context, poisoned) + const healthySource = await activatePreferredSource(context, healthy) + const first = await context.store.claimRegionalRehome() + context.advance(6_000) + const second = await context.store.claimRegionalRehome() + for (const [identity, attempt, sourceControl] of [ + [poisoned, first, poisonedSource], + [healthy, second, healthySource] + ] as const) { + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + const targetControl = await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + await context.store.releaseActivity(identity, targetControl) + } + await context.database.query( + `UPDATE relay_assignments SET assignment_epoch = assignment_epoch + 5 + WHERE user_id = ?`, + [poisoned.userId] + ) + context.advance(24 * 60 * 60_000) + await freshHeartbeats(context) + const warnings = collectCandidateFailureWarnings() + try { + expect(await context.store.abortExpiredRegionalRehomes()).toBe(1) + } finally { + warnings.restore() + } + expect(warnings.entries).toEqual([ + { + event: 'orca_relay_regional_rehome_candidate_failed', + operation: 'abort', + attemptId: first!.attemptId, + reason: 'regional_rehome_assignment_mismatch' + } + ]) + expect(await context.database.query( + `SELECT aborted_at FROM relay_region_rehome_attempts WHERE attempt_id = ?`, + [second!.attemptId] + )).toEqual([{ aborted_at: context.now() }]) + await context.database.close() + }) + + it('keeps dual-control accounting when the drained host re-resolves mid-rehome', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + expect(await controlAccounting(context, identity)).toEqual({ + reservedControls: 2, + controlLeases: 2 + }) + + // The drained host re-resolves through the director while both the source + // control and the target's pending control are still live. + await context.store.assign(identity, 'asia-east2') + expect(await controlAccounting(context, identity)).toEqual({ + reservedControls: 2, + controlLeases: 2 + }) + + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + expect(await controlAccounting(context, identity)).toEqual({ + reservedControls: 1, + controlLeases: 1 + }) + + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + expect(await context.database.query( + `SELECT completed_at FROM relay_assignment_migrations + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + )).toEqual([{ completed_at: context.now() }]) + expect(await cellReservations(context)).toEqual({ [source.id]: 0, [target.id]: 1 }) + await context.database.close() + }) + + it('grants a host whose counter was skewed without duplicating its control', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.releaseActivity(identity, sourceControl) + // Damage already written by a pre-fix sticky grant. + await context.database.query( + `UPDATE relay_assignments SET reserved_controls = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + + await context.store.assign(identity, 'asia-east2') + expect(await context.database.query( + `SELECT activity_id FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control'`, + [identity.userId, identity.relayHostId] + )).toEqual([{ activity_id: `control:${target.id}:1` }]) + expect(await cellReservations(context)).toEqual({ [source.id]: 0, [target.id]: 2 }) + await context.database.close() + }) + + it('repairs a skewed control counter before completing the rehome', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + // Damage already written by a pre-fix sticky grant. + await context.database.query( + `UPDATE relay_assignments SET reserved_controls = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + expect(await controlAccounting(context, identity)).toEqual({ + reservedControls: 1, + controlLeases: 1 + }) + expect(await context.database.query( + `SELECT migration_leases FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + )).toEqual([{ migration_leases: 0 }]) + await context.database.close() + }) + + it('reports a repaired counter only for the candidate it repaired', async () => { + const context = await setup() + const skewed = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const clean = { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' } + const skewedSource = await activatePreferredSource(context, skewed) + const cleanSource = await activatePreferredSource(context, clean) + const first = await context.store.claimRegionalRehome() + context.advance(6_000) + const second = await context.store.claimRegionalRehome() + for (const [identity, attempt, sourceControl] of [ + [skewed, first, skewedSource], + [clean, second, cleanSource] + ] as const) { + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + } + // Damage already written by a pre-fix sticky grant, on one host only. + await context.database.query( + `UPDATE relay_assignments SET reserved_controls = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [skewed.userId, skewed.relayHostId] + ) + + const warnings = collectCandidateFailureWarnings([ + 'orca_relay_regional_rehome_activity_counts_repaired' + ]) + try { + expect(await context.store.completeReadyRegionalRehomes()).toBe(2) + } finally { + warnings.restore() + } + expect(warnings.entries).toEqual([ + { + event: 'orca_relay_regional_rehome_activity_counts_repaired', + attemptId: first!.attemptId + } + ]) + await context.database.close() + }) + + it('never repairs past a migration lease whose shape is wrong', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + await context.database.query( + `UPDATE relay_assignments SET reserved_controls = 0 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + await context.database.query( + `UPDATE relay_assignment_migrations SET target_reserved_units = 9 + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + + const warnings = collectCandidateFailureWarnings() + try { + expect(await context.store.completeReadyRegionalRehomes()).toBe(0) + } finally { + warnings.restore() + } + expect(warnings.entries).toEqual([ + { + event: 'orca_relay_regional_rehome_candidate_failed', + operation: 'complete', + attemptId: attempt!.attemptId, + reason: 'migration_activity_lease_shape_mismatch' + } + ]) + expect(await controlAccounting(context, identity)).toEqual({ + reservedControls: 0, + controlLeases: 1 + }) + await context.database.close() + }) + + it('stays silent when a repair cannot make the counts whole', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: attempt!.assignmentEpoch + }) + await context.store.releaseActivity(identity, sourceControl) + // A vanished migration lease is repairable arithmetic on the first assert + // but still wrong on the re-assert: no repaired event may leak out. + await context.database.query( + `DELETE FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'migration'`, + [identity.userId, identity.relayHostId] + ) + + const warnings = collectCandidateFailureWarnings([ + 'orca_relay_regional_rehome_candidate_failed', + 'orca_relay_regional_rehome_activity_counts_repaired' + ]) + try { + expect(await context.store.completeReadyRegionalRehomes()).toBe(0) + } finally { + warnings.restore() + } + expect(warnings.entries).toEqual([ + { + event: 'orca_relay_regional_rehome_candidate_failed', + operation: 'complete', + attemptId: attempt!.attemptId, + reason: 'migration_activity_accounting_mismatch' + } + ]) + await context.database.close() + }) + + it('caps redrain dispatches at the send limit', async () => { + const context = await setup() + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.database.query( + `UPDATE relay_region_rehome_attempts SET send_attempts = ? WHERE attempt_id = ?`, + [REGIONAL_REHOME_REDRAIN_SEND_LIMIT, attempt!.attemptId] + ) + context.advance(60 * 60_000 + 1) + await freshHeartbeats(context) + expect(await context.store.claimRegionalRehome()).toBeNull() + await context.database.close() + }) +}) + +class TransactionCountingDatabase implements RelayDatabase { + transactionCalls = 0 + + constructor(private readonly delegate: RelayDatabase) {} + + query(sql: string, params?: unknown[]): Promise { + return this.delegate.query(sql, params) + } + + queryLocked( + sql: string, + params?: unknown[], + options?: { failIfUnavailable?: boolean } + ): Promise { + return this.delegate.queryLocked(sql, params, options) + } + + transaction( + operation: (transaction: RelayDatabase) => Promise, + options?: { reportRetries?: boolean } + ): Promise { + this.transactionCalls += 1 + return this.delegate.transaction(operation, options) + } + + close(): Promise { + return this.delegate.close() + } +} + +type Context = Awaited> + +function collectDisableWarnings() { + const entries: Record[] = [] + const original = console.warn + console.warn = (line: unknown, ...rest: unknown[]) => { + try { + const parsed = JSON.parse(line as string) as Record + if (parsed.event === 'orca_relay_regional_rehome_safety_disabled') { + entries.push(parsed) + return + } + } catch { + // fall through to the real console for non-JSON lines + } + original(line, ...rest) + } + return { + entries, + restore: () => { + console.warn = original + } + } +} + +async function setup(options: { sourceProtocol?: number } = {}) { + let clock = 1_000_000 + const database = await openInMemoryRelayDatabase() + const store = new RelayAssignmentStore(database, () => clock, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.inspectRegionalRehomeControl() + clock += 24 * 60 * 60_000 + await store.applyRegionalRehomeControl({ + expectedGeneration: 0, + enabled: true, + notBefore: clock, + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60 * 60_000 + }) + await store.reconcileCells([source, target]) + await heartbeat(store, source, sourceIncarnation, options.sourceProtocol ?? 1) + await heartbeat(store, target, targetIncarnation, 0) + return { + database, + store, + now: () => clock, + advance: (milliseconds: number) => { + clock += milliseconds + } + } +} + +function collectEventWarnings(event: string) { + const entries: Record[] = [] + const original = console.warn + console.warn = (line: unknown, ...rest: unknown[]) => { + try { + const parsed = JSON.parse(line as string) as Record + if (parsed.event === event) { + entries.push(parsed) + return + } + } catch { + // fall through to the real console for non-JSON lines + } + original(line, ...rest) + } + return { + entries, + restore: () => { + console.warn = original + } + } +} + +function collectCandidateFailureWarnings( + events: string[] = ['orca_relay_regional_rehome_candidate_failed'] +) { + const entries: Record[] = [] + const original = console.warn + console.warn = (line: unknown, ...rest: unknown[]) => { + try { + const parsed = JSON.parse(line as string) as Record + if (events.includes(parsed.event as string)) { + entries.push(parsed) + return + } + } catch { + // fall through to the real console for non-JSON lines + } + original(line, ...rest) + } + return { + entries, + restore: () => { + console.warn = original + } + } +} + +async function controlAccounting( + context: Context, + identity: { userId: string; relayHostId: string } +): Promise<{ reservedControls: number; controlLeases: number }> { + const assignment = ( + await context.database.query( + `SELECT reserved_controls FROM relay_assignments + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + )[0]! + const leases = await context.database.query( + `SELECT COUNT(*) AS controls FROM relay_assignment_activity_leases + WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control'`, + [identity.userId, identity.relayHostId] + ) + return { + reservedControls: Number(assignment.reserved_controls), + controlLeases: Number(leases[0]!.controls) + } +} + +async function cellReservations(context: Context): Promise> { + const rows = await context.database.query( + `SELECT cell_id, reserved_requests FROM relay_cells ORDER BY cell_id` + ) + return Object.fromEntries( + rows.map((row) => [String(row.cell_id), Number(row.reserved_requests)]) + ) +} + +async function freshHeartbeats(context: Context): Promise { + const safety = { + observedAt: context.now(), + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } + // The clock doubles as a strictly-increasing connection inclusion watermark. + await heartbeat(context.store, source, sourceIncarnation, 1, context.now(), safety) + await heartbeat(context.store, target, targetIncarnation, 0, context.now(), safety) +} + +async function activatePreferredSource( + context: Context, + identity: { userId: string; relayHostId: string } +): Promise { + const assignment = await context.store.assign(identity, undefined, 'us-central1') + const control = await context.store.activateControl(identity, { + cellId: source.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await context.store.assign(identity, 'asia-east2') + return control +} + +async function activateSource( + context: Context, + identity: { userId: string; relayHostId: string } +): Promise { + const assignment = await context.store.assign(identity, undefined, 'us-central1') + return await context.store.activateControl(identity, { + cellId: source.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) +} + +async function heartbeat( + store: RelayAssignmentStore, + cell: typeof source | typeof target, + cellIncarnation: string, + regionalRehomeProtocol: number, + connectionInclusionWatermark = 1, + regionalRehomeSafety?: RegionalRehomeSafetySnapshot +): Promise { + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + region: cell.region, + cellIncarnation, + startedAt: 900_000, + ready: true, + observedRequests: 0, + totalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + connectionInclusionWatermark, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + }) + await store.recordCellRegionalRehomeStatus({ + cellId: cell.id, + cellIncarnation, + regionalRehomeProtocol, + safety: regionalRehomeSafety ?? { + observedAt: 1_000_000 + 24 * 60 * 60_000, + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } + }) +} diff --git a/cloud/apps/relay/src/regional-rehome-target-selection.test.ts b/cloud/apps/relay/src/regional-rehome-target-selection.test.ts new file mode 100644 index 00000000000..e2190168735 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-target-selection.test.ts @@ -0,0 +1,163 @@ +import { describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import { openInMemoryRelayDatabase } from './database.js' +import { REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT } from './regional-rehome-safety.js' + +const cell = (id: string, region: 'us-central1' | 'asia-east2') => ({ + id, + url: `https://${id}.relay.example.test`, + region, + capacityRequests: 100, + connectionHardCap: 1_000 as const, + connectionUnobservedBound: 60 +}) +const source = cell('us-c1', 'us-central1') +const noHeadroom = cell('asia-a', 'asia-east2') +const unclean = cell('asia-b', 'asia-east2') +const highLoad = cell('asia-c', 'asia-east2') +const lowLoad = cell('asia-d', 'asia-east2') + +const incarnation = (n: number) => + `${String(n).repeat(8)}-${String(n).repeat(4)}-4${String(n).repeat(3)}` + + `-8${String(n).repeat(3)}-${String(n).repeat(12)}` + +async function setup() { + let clock = 1_000_000 + const database = await openInMemoryRelayDatabase() + const store = new RelayAssignmentStore(database, () => clock, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.inspectRegionalRehomeControl() + clock += 24 * 60 * 60_000 + await store.applyRegionalRehomeControl({ + expectedGeneration: 0, + enabled: true, + notBefore: clock, + ratePerMinute: 10, + preferenceMaxAgeMs: 24 * 60 * 60_000, + drainGraceMs: 60 * 60_000 + }) + await store.reconcileCells([source, noHeadroom, unclean, highLoad, lowLoad]) + const beat = async ( + config: typeof source, + n: number, + protocol: number, + state: { observedRequests: number; enforcedConnections: number; sqlFailures: number } + ) => { + await store.recordCellHeartbeat({ + cellId: config.id, + cellUrl: config.url, + region: config.region, + cellIncarnation: incarnation(n), + startedAt: 900_000, + ready: true, + observedRequests: state.observedRequests, + totalConnections: state.enforcedConnections, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: state.enforcedConnections, + connectionInclusionWatermark: clock, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + }) + await store.recordCellRegionalRehomeStatus({ + cellId: config.id, + cellIncarnation: incarnation(n), + regionalRehomeProtocol: protocol, + safety: { + observedAt: clock, + sqlFailures: state.sqlFailures, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolWaitMsMax: 0 + } + }) + } + const activatePreferredSource = async () => { + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const assignment = await store.assign(identity, undefined, 'us-central1') + await store.activateControl(identity, { + cellId: source.id, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.assign(identity, 'asia-east2') + } + return { database, store, beat, activatePreferredSource } +} + +const UNCLEAN = REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT + 1 + +describe('regional rehome target selection', () => { + it('never selects a target without connection headroom, even at lowest load', async () => { + const context = await setup() + await context.beat(source, 1, 1, { + observedRequests: 0, + enforcedConnections: 0, + sqlFailures: 0 + }) + // Lowest load but the connection hard cap is exhausted. + await context.beat(noHeadroom, 2, 0, { + observedRequests: 0, + enforcedConnections: 999, + sqlFailures: 0 + }) + await context.beat(unclean, 3, 0, { + observedRequests: 0, + enforcedConnections: 0, + sqlFailures: UNCLEAN + }) + await context.beat(highLoad, 4, 0, { + observedRequests: 50, + enforcedConnections: 0, + sqlFailures: 0 + }) + await context.beat(lowLoad, 5, 0, { + observedRequests: 10, + enforcedConnections: 0, + sqlFailures: 0 + }) + await context.activatePreferredSource() + + const attempt = await context.store.claimRegionalRehome() + expect(attempt?.targetCellId).toBe(lowLoad.id) + await context.database.close() + }) + + it('falls to the next clean target when the load winner goes unclean', async () => { + const context = await setup() + await context.beat(source, 1, 1, { + observedRequests: 0, + enforcedConnections: 0, + sqlFailures: 0 + }) + await context.beat(noHeadroom, 2, 0, { + observedRequests: 0, + enforcedConnections: 999, + sqlFailures: 0 + }) + await context.beat(unclean, 3, 0, { + observedRequests: 0, + enforcedConnections: 0, + sqlFailures: UNCLEAN + }) + await context.beat(highLoad, 4, 0, { + observedRequests: 50, + enforcedConnections: 0, + sqlFailures: 0 + }) + await context.beat(lowLoad, 5, 0, { + observedRequests: 10, + enforcedConnections: 0, + sqlFailures: UNCLEAN + }) + await context.activatePreferredSource() + + const attempt = await context.store.claimRegionalRehome() + expect(attempt?.targetCellId).toBe(highLoad.id) + await context.database.close() + }) +}) diff --git a/cloud/apps/relay/src/regional-rehome-trust-probe.ts b/cloud/apps/relay/src/regional-rehome-trust-probe.ts new file mode 100644 index 00000000000..57ae2468886 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-trust-probe.ts @@ -0,0 +1,107 @@ +import { z } from 'zod' + +export const REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID = + '00000000-0000-4000-8000-000000000001' +export const REGIONAL_REHOME_TRUST_PROBE_USER_ID = 'regional-rehome-trust-probe' +export const REGIONAL_REHOME_TRUST_PROBE_HOST_ID = 'trustprobe000001' + +const SOURCE_PROBE_TIMEOUT_MS = 10_000 + +const SourceProbeResponseSchema = z + .object({ + v: z.literal(1), + outcome: z.enum(['accepted', 'already-accepted', 'host-not-connected']), + sharedRuntimeIdentityRejected: z.literal(true) + }) + .strict() + +type SourceProbeOutcome = z.infer['outcome'] + +export type RegionalRehomeTrustProbeResult = { + v: 1 + dedicatedIdentity: { + accepted: boolean + firstOutcome: SourceProbeOutcome + secondOutcome: SourceProbeOutcome + idempotent: boolean + } + sharedRuntimeIdentityRejected: boolean + proven: boolean +} + +export async function probeRegionalRehomeTrust(input: { + sourceCellUrl: string + sourceCellId: string + sourceCellIncarnation: string + audience: string + identityToken: (audience: string) => Promise + fetch: typeof fetch +}): Promise { + const token = await input.identityToken(input.audience) + const body = JSON.stringify({ + v: 1, + attemptId: REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID, + userId: REGIONAL_REHOME_TRUST_PROBE_USER_ID, + relayHostId: REGIONAL_REHOME_TRUST_PROBE_HOST_ID, + sourceCellId: input.sourceCellId, + sourceCellIncarnation: input.sourceCellIncarnation, + sourceAssignmentEpoch: 1, + graceMs: 0 + }) + const call = async (): Promise> => { + const response = await input.fetch( + new URL('/v1/admin/host-drain', input.sourceCellUrl), + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body, + signal: AbortSignal.timeout(SOURCE_PROBE_TIMEOUT_MS) + } + ) + if (!response.ok) throw new Error(`regional_rehome_trust_probe_source_${response.status}`) + const parsed = SourceProbeResponseSchema.safeParse(await response.json()) + if (!parsed.success) throw new Error('regional_rehome_trust_probe_source_invalid_response') + return parsed.data + } + const first = await call() + const second = await call() + const idempotent = first.outcome === second.outcome + const sharedRuntimeIdentityRejected = + first.sharedRuntimeIdentityRejected && second.sharedRuntimeIdentityRejected + const proven = + first.outcome === 'host-not-connected' && + second.outcome === 'host-not-connected' && + idempotent && + sharedRuntimeIdentityRejected + if (!proven) throw new Error('regional_rehome_trust_probe_not_proven') + return { + v: 1, + dedicatedIdentity: { + accepted: true, + firstOutcome: first.outcome, + secondOutcome: second.outcome, + idempotent + }, + sharedRuntimeIdentityRejected, + proven + } +} + +export function isRegionalRehomeTrustProbe(input: { + attemptId: string + userId: string + relayHostId: string + sourceAssignmentEpoch: number + graceMs: number +}): boolean { + return ( + input.attemptId === REGIONAL_REHOME_TRUST_PROBE_ATTEMPT_ID && + input.userId === REGIONAL_REHOME_TRUST_PROBE_USER_ID && + input.relayHostId === REGIONAL_REHOME_TRUST_PROBE_HOST_ID && + input.sourceAssignmentEpoch === 1 && + input.graceMs === 0 + ) +} diff --git a/cloud/apps/relay/src/regional-rehome-worker.test.ts b/cloud/apps/relay/src/regional-rehome-worker.test.ts new file mode 100644 index 00000000000..af905bb9ab2 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-worker.test.ts @@ -0,0 +1,227 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import { + combineRegionalRehomeSafety, + REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT, + regionalRehomeSafetyFailure +} from './regional-rehome-safety.js' +import { startRegionalRehomeWorker } from './regional-rehome-worker.js' + +describe('regional rehome worker', () => { + afterEach(() => vi.restoreAllMocks()) + + it('sends an incarnation- and source-epoch-bound drain without exposing identity', async () => { + let now = 0 + const attempt = { + attemptId: '11111111-1111-4111-8111-111111111111', + userId: 'private-user', + relayHostId: 'abcdefghijklmnop', + preferredRegion: 'asia-east2', + sourceCellId: 'production-gce-c7', + sourceCellUrl: 'https://c7.relay.example.test', + sourceCellIncarnation: '22222222-2222-4222-8222-222222222222', + targetCellId: 'production-gce-c27', + targetCellIncarnation: '33333333-3333-4333-8333-333333333333', + previousEpoch: 7, + assignmentEpoch: 8, + drainGraceMs: 60_000, + sendAttempts: 1 + } + const claimRegionalRehome = vi.fn().mockResolvedValueOnce(null).mockResolvedValue(attempt) + const recordRegionalRehomeDrainReceipt = vi.fn().mockResolvedValue(true) + const recordRegionalRehomeWorkerFailure = vi.fn().mockResolvedValue(undefined) + const assignments = { + claimRegionalRehome, + recordRegionalRehomeDrainReceipt, + recordRegionalRehomeWorkerFailure + } as unknown as RelayAssignmentStore + const requests: Array<{ url: string; init?: RequestInit }> = [] + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const worker = startRegionalRehomeWorker(config(), assignments, { + now: () => now, + safetySnapshot: () => safety(now), + intervalMs: 60_000, + identityToken: async (audience) => { + expect(audience).toBe('https://relay.example.test/v1/admin/host-drain') + return 'secret-token' + }, + fetch: (async (url, init) => { + requests.push({ url: String(url), init }) + return Response.json({ v: 1, outcome: 'accepted' }) + }) as typeof fetch + })! + await settleWorker() + now = 1_000 + await worker.run() + worker.stop() + + expect(requests).toHaveLength(1) + expect(requests[0]!.url).toBe('https://c7.relay.example.test/v1/admin/host-drain') + expect(requests[0]!.url).not.toContain('secret-token') + expect(requests[0]!.init?.headers).toMatchObject({ + authorization: 'Bearer secret-token' + }) + expect(JSON.parse(String(requests[0]!.init?.body))).toEqual({ + v: 1, + attemptId: '11111111-1111-4111-8111-111111111111', + userId: 'private-user', + relayHostId: 'abcdefghijklmnop', + sourceCellId: 'production-gce-c7', + sourceCellIncarnation: '22222222-2222-4222-8222-222222222222', + sourceAssignmentEpoch: 7, + graceMs: 60_000 + }) + expect(recordRegionalRehomeDrainReceipt).toHaveBeenCalledWith( + '11111111-1111-4111-8111-111111111111', + 'accepted' + ) + const logs = warn.mock.calls.map((call) => String(call[0])).join('\n') + expect(logs).not.toContain('private-user') + expect(logs).not.toContain('abcdefghijklmnop') + }) + + it('fails closed before the observation gate and records bounded dispatch failures', async () => { + let now = 0 + const attempt = { + attemptId: '11111111-1111-4111-8111-111111111111', + userId: 'private-user', + relayHostId: 'abcdefghijklmnop', + sourceCellId: 'source', + sourceCellUrl: 'https://source.example.test', + sourceCellIncarnation: '22222222-2222-4222-8222-222222222222', + targetCellId: 'target', + previousEpoch: 1, + assignmentEpoch: 2, + drainGraceMs: 60_000, + sendAttempts: 1 + } + const claimRegionalRehome = vi.fn().mockResolvedValueOnce(null).mockResolvedValue(attempt) + const assignments = { + claimRegionalRehome, + recordRegionalRehomeDispatchFailure: vi.fn().mockResolvedValue(undefined), + recordRegionalRehomeWorkerFailure: vi.fn().mockResolvedValue(undefined) + } as unknown as RelayAssignmentStore + const worker = startRegionalRehomeWorker(config(), assignments, { + now: () => now, + safetySnapshot: () => safety(now), + intervalMs: 60_000, + identityToken: async () => { + throw new Error('token unavailable') + } + })! + await settleWorker() + claimRegionalRehome.mockClear() + now = 100 + await worker.run() + worker.stop() + expect(assignments.recordRegionalRehomeDispatchFailure).toHaveBeenCalledWith( + '11111111-1111-4111-8111-111111111111' + ) + expect(assignments.recordRegionalRehomeWorkerFailure).not.toHaveBeenCalled() + }) + + it('passes unsafe process telemetry to the durable claim gate', async () => { + let now = 0 + let sqlFailures = 0 + const claimRegionalRehome = vi.fn().mockResolvedValue(null) + const assignments = { + claimRegionalRehome + } as unknown as RelayAssignmentStore + const worker = startRegionalRehomeWorker(config(), assignments, { + now: () => now, + safetySnapshot: () => ({ ...safety(now), sqlFailures }), + intervalMs: 60_000 + })! + await settleWorker() + claimRegionalRehome.mockClear() + now = 100 + sqlFailures = 1 + await worker.run() + worker.stop() + + expect(claimRegionalRehome).toHaveBeenCalledWith( + expect.objectContaining({ observedAt: 100, sqlFailures: 1 }) + ) + }) + + it('starts inert on directors so durable control can enable without a restart', async () => { + let now = 0 + const claimRegionalRehome = vi.fn().mockResolvedValue(null) + const assignments = { + claimRegionalRehome + } as unknown as RelayAssignmentStore + const worker = startRegionalRehomeWorker( + config(), + assignments, + { + now: () => now, + safetySnapshot: () => safety(now), + intervalMs: 60_000 + } + ) + expect(worker).not.toBeNull() + await settleWorker() + claimRegionalRehome.mockClear() + now = 100 + await worker!.run() + worker!.stop() + expect(claimRegionalRehome).toHaveBeenCalledOnce() + + expect( + startRegionalRehomeWorker( + config({ role: 'cell' }), + {} as RelayAssignmentStore, + { safetySnapshot: () => safety(1) } + ) + ).toBeNull() + }) + + it('treats the reconnect threshold as per-cell and excludes the director', () => { + const cells = 2 + const limit = cells * REGIONAL_REHOME_RECONNECTS_PER_CELL_LIMIT + const processSafety = { ...safety(100), reconnects: limit * 10 } + const fleetSafety = { ...safety(100), reconnects: limit } + expect(regionalRehomeSafetyFailure( + combineRegionalRehomeSafety(processSafety, fleetSafety), + 100, + cells + )).toBeNull() + expect(regionalRehomeSafetyFailure( + combineRegionalRehomeSafety(processSafety, { ...fleetSafety, reconnects: limit + 1 }), + 100, + cells + )).toBe('elevated_reconnects') + }) +}) + +function config(overrides: Partial = {}): RelayConfig { + return { + role: 'director', + rehomeAudience: 'https://relay.example.test/v1/admin/host-drain', + rehomeDirectorServiceAccount: 'relay-director@example.test', + ...overrides + } as RelayConfig +} + +function safety(observedAt: number) { + return { + requiredCells: 2, + missingCells: 0, + observedAt, + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0, + databasePoolTotal: 3, + databasePoolIdle: 3, + databasePoolWaiting: 0, + databasePoolWaitersMax: 0, + databasePoolOldestWaitMs: 0, + databasePoolWaitMsMax: 0 + } +} + +async function settleWorker(): Promise { + await Promise.resolve() + await Promise.resolve() +} diff --git a/cloud/apps/relay/src/regional-rehome-worker.ts b/cloud/apps/relay/src/regional-rehome-worker.ts new file mode 100644 index 00000000000..97c63a61025 --- /dev/null +++ b/cloud/apps/relay/src/regional-rehome-worker.ts @@ -0,0 +1,122 @@ +import { z } from 'zod' +import type { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import { googleMetadataIdentityToken } from './google-metadata-identity-token.js' +import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' + +type RegionalRehomeWorkerOptions = { + fetch?: typeof fetch + identityToken?: (audience: string) => Promise + now?: () => number + intervalMs?: number + requestTimeoutMs?: number + safetySnapshot?: () => RegionalRehomeSafetySnapshot +} + +export type RegionalRehomeWorker = { + run: () => Promise + stop: () => void +} + +const RegionalHostDrainResponseSchema = z + .object({ + v: z.literal(1), + outcome: z.enum(['accepted', 'already-accepted', 'host-not-connected']) + }) + .strict() + +export function startRegionalRehomeWorker( + config: RelayConfig, + assignments: RelayAssignmentStore, + options: RegionalRehomeWorkerOptions = {} +): RegionalRehomeWorker | null { + if ( + config.role !== 'director' || + !config.rehomeAudience || + !config.rehomeDirectorServiceAccount || + !options.safetySnapshot + ) { + return null + } + const audience = config.rehomeAudience + const safetySnapshot = options.safetySnapshot + const now = options.now ?? Date.now + const fetchImpl = options.fetch ?? fetch + const tokenProvider = + options.identityToken ?? + ((audience: string) => googleMetadataIdentityToken(audience, fetchImpl)) + let stopped = false + let inFlight = false + const run = async (): Promise => { + if (stopped || inFlight) return + inFlight = true + let attemptId: string | null = null + try { + const processSafety = safetySnapshot() + const attempt = await assignments.claimRegionalRehome(processSafety) + if (!attempt) return + attemptId = attempt.attemptId + const token = await tokenProvider(audience) + const response = await fetchImpl( + new URL('/v1/admin/host-drain', attempt.sourceCellUrl), + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + v: 1, + attemptId: attempt.attemptId, + userId: attempt.userId, + relayHostId: attempt.relayHostId, + sourceCellId: attempt.sourceCellId, + sourceCellIncarnation: attempt.sourceCellIncarnation, + sourceAssignmentEpoch: attempt.previousEpoch, + graceMs: attempt.drainGraceMs + }), + signal: AbortSignal.timeout(options.requestTimeoutMs ?? 10_000) + } + ) + if (!response.ok) throw new Error(`regional_rehome_source_${response.status}`) + const body = RegionalHostDrainResponseSchema.safeParse(await response.json()) + if (!body.success) throw new Error('regional_rehome_source_invalid_response') + await assignments.recordRegionalRehomeDrainReceipt( + attempt.attemptId, + body.data.outcome + ) + console.warn( + JSON.stringify({ + event: 'orca_relay_regional_rehome_dispatched', + sourceCellId: attempt.sourceCellId, + targetCellId: attempt.targetCellId, + outcome: body.data.outcome, + sendAttempts: attempt.sendAttempts + }) + ) + } catch (error) { + await (attemptId + ? assignments.recordRegionalRehomeDispatchFailure(attemptId) + : assignments.recordRegionalRehomeWorkerFailure() + ).catch(() => undefined) + console.warn( + JSON.stringify({ + event: 'orca_relay_regional_rehome_dispatch_failed', + reason: error instanceof Error ? error.message : 'unknown' + }) + ) + } finally { + inFlight = false + } + } + const timer = setInterval(() => void run(), options.intervalMs ?? 1_000) + timer.unref() + void run() + return { + run, + stop: () => { + stopped = true + clearInterval(timer) + } + } +} diff --git a/cloud/apps/relay/src/registered-migration-abandonment.ts b/cloud/apps/relay/src/registered-migration-abandonment.ts new file mode 100644 index 00000000000..9127f935f2a --- /dev/null +++ b/cloud/apps/relay/src/registered-migration-abandonment.ts @@ -0,0 +1,81 @@ +export const REGISTERED_MIGRATION_ABANDON_MS = 24 * 60 * 60 * 1_000 + +export const DURABLY_FENCED_MIGRATION_SOURCE = `( + EXISTS ( + SELECT 1 FROM relay_cell_committed_fences committed_source_fence + JOIN relay_cell_fence_attempts source_fence_attempt + ON source_fence_attempt.attempt_id = committed_source_fence.attempt_id + JOIN relay_cell_runtime source_runtime + ON source_runtime.cell_id = committed_source_fence.cell_id + WHERE committed_source_fence.cell_id = migration.source_cell_id + AND source_fence_attempt.cell_id = committed_source_fence.cell_id + AND source_fence_attempt.cell_incarnation = committed_source_fence.cell_incarnation + AND source_fence_attempt.completed_at IS NOT NULL + AND source_fence_attempt.aborted_at IS NULL + AND source_runtime.cell_incarnation = committed_source_fence.cell_incarnation + AND committed_source_fence.attested_at >= source_runtime.last_heartbeat_at + ) + OR EXISTS ( + SELECT 1 FROM relay_cell_legacy_fence_adoptions adopted_source_fence + JOIN relay_cell_runtime source_runtime + ON source_runtime.cell_id = adopted_source_fence.cell_id + WHERE adopted_source_fence.cell_id = migration.source_cell_id + AND source_runtime.cell_incarnation = adopted_source_fence.cell_incarnation + AND adopted_source_fence.attested_at >= source_runtime.last_heartbeat_at + ) +)` + +// Ordinary recovery waits on admission age; a completed fence proves the source cannot return. +export const ABANDONED_REGISTERED_MIGRATION = `( + migration.target_registered_at IS NOT NULL + AND + migration.expires_at <= ? + AND + ( + EXISTS ( + SELECT 1 FROM relay_cells target_cell + JOIN relay_cell_admission target_admission + ON target_admission.cell_id = target_cell.cell_id + WHERE target_cell.cell_id = migration.target_cell_id + AND target_cell.enabled = 0 + AND target_admission.admission_state = 'existing-only' + AND target_admission.updated_at <= ? + ) + OR ( + EXISTS ( + SELECT 1 FROM relay_cells source_cell + JOIN relay_cell_admission source_admission + ON source_admission.cell_id = source_cell.cell_id + WHERE source_cell.cell_id = migration.source_cell_id + AND source_cell.enabled = 0 + AND source_admission.admission_state = 'existing-only' + AND ( + source_admission.updated_at <= ? + OR ${DURABLY_FENCED_MIGRATION_SOURCE} + ) + ) + AND EXISTS ( + SELECT 1 FROM relay_cells target_cell + JOIN relay_cell_admission target_admission + ON target_admission.cell_id = target_cell.cell_id + WHERE target_cell.cell_id = migration.target_cell_id + AND target_cell.enabled = 1 + AND target_admission.admission_state IN ('migration-only', 'general') + ) + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases source_activity + WHERE source_activity.user_id = migration.user_id + AND source_activity.relay_host_id = migration.relay_host_id + AND source_activity.cell_id = migration.source_cell_id + ) + ) + ) + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases target_control + WHERE target_control.user_id = migration.user_id + AND target_control.relay_host_id = migration.relay_host_id + AND target_control.cell_id = migration.target_cell_id + AND target_control.activity_kind = 'control' + AND target_control.activity_id NOT LIKE 'control-pending:%' + ) +)` diff --git a/cloud/apps/relay/src/registered-migration-inventory-postgres.test.ts b/cloud/apps/relay/src/registered-migration-inventory-postgres.test.ts new file mode 100644 index 00000000000..ad318b2ad3a --- /dev/null +++ b/cloud/apps/relay/src/registered-migration-inventory-postgres.test.ts @@ -0,0 +1,139 @@ +import pg from 'pg' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import { openRelayDatabase, type RelayDatabase } from './database.js' +import { REGISTERED_MIGRATION_ABANDON_MS } from './registered-migration-abandonment.js' +import { readRegisteredMigrationInventory } from './registered-migration-inventory.js' + +const databaseUrl = process.env.ORCA_RELAY_TEST_POSTGRES_URL +const describePostgres = databaseUrl ? describe : describe.skip +const schema = 'relay_migration_inventory_test' + +describePostgres('PostgreSQL registered migration inventory', () => { + let database: RelayDatabase | undefined + + beforeAll(async () => { + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + await client.query(`CREATE SCHEMA ${schema}`) + } finally { + await client.end() + } + const url = new URL(databaseUrl!) + url.searchParams.set('options', `-c search_path=${schema}`) + database = await openRelayDatabase({ databaseUrl: url.toString(), dataDir: '' }) + }) + + afterAll(async () => { + await database?.close() + const client = new pg.Client({ connectionString: databaseUrl }) + await client.connect() + try { + await client.query(`DROP SCHEMA IF EXISTS ${schema} CASCADE`) + } finally { + await client.end() + } + }) + + it('counts a disabled-target migration even while its source is active', async () => { + const now = REGISTERED_MIGRATION_ABANDON_MS + 10_000 + await database!.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES ('cell-a', 'https://a.example.test', 1, 4000, 1, 0, 0, 1), + ('cell-b', 'https://b.example.test', 0, 4000, 0, 0, 0, 1)` + ) + await database!.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES ('cell-a', 'general', 1), ('cell-b', 'existing-only', 1)` + ) + await database!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES ('user-1', '111111111111', 'cell-b', 2, ?, ?, 0, 0, 0, 0, 0, 0)`, + [now, now] + ) + await database!.query( + `INSERT INTO relay_assignment_activity_leases + (user_id, relay_host_id, activity_id, activity_kind, cell_id, + request_units, expires_at, updated_at) + VALUES ('user-1', '111111111111', 'control:source:1', 'control', 'cell-a', 1, ?, ?)`, + [now, now] + ) + await database!.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, previous_epoch, + assignment_epoch, source_request_units, target_reserved_units, expires_at, + target_registered_at, created_at, updated_at) + VALUES ('user-1', '111111111111', 'cell-a', 'cell-b', 1, 2, 1, 2, 2, 2, 1, 2)` + ) + + expect(await readRegisteredMigrationInventory(database!, now)).toEqual({ + open: 1, + inactive: 1, + abandoned: 1, + pairs: [ + { + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + open: 1, + inactive: 1, + abandoned: 1 + } + ] + }) + + await database!.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES ('cell-c', 'https://c.example.test', 1, 4000, 0, 0, 0, 1)` + ) + await database!.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES ('cell-c', 'migration-only', 1)` + ) + await database!.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES ('user-2', '222222222222', 'cell-c', 2, ?, ?, 0, 0, 0, 0, 0, 0)`, + [now, now] + ) + await database!.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, previous_epoch, + assignment_epoch, source_request_units, target_reserved_units, expires_at, + target_registered_at, created_at, updated_at) + VALUES ('user-2', '222222222222', 'cell-a', 'cell-c', 1, 2, 1, 2, ?, NULL, 1, 2)`, + [now + 60_000] + ) + + expect(await readRegisteredMigrationInventory(database!, now)).toEqual({ + open: 2, + inactive: 1, + abandoned: 1, + pairs: [ + { + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + open: 1, + inactive: 1, + abandoned: 1 + }, + { + sourceCellId: 'cell-a', + targetCellId: 'cell-c', + open: 1, + inactive: 0, + abandoned: 0 + } + ] + }) + }) +}) diff --git a/cloud/apps/relay/src/registered-migration-inventory.test.ts b/cloud/apps/relay/src/registered-migration-inventory.test.ts new file mode 100644 index 00000000000..41aa26ac732 --- /dev/null +++ b/cloud/apps/relay/src/registered-migration-inventory.test.ts @@ -0,0 +1,117 @@ +import { describe, expect, it } from 'vitest' +import { openInMemoryRelayDatabase, type RelayDatabase } from './database.js' +import { REGISTERED_MIGRATION_ABANDON_MS } from './registered-migration-abandonment.js' +import { + formatRegisteredMigrationInventory, + readRegisteredMigrationInventory +} from './registered-migration-inventory.js' + +describe('registered migration inventory', () => { + it('does not restart abandonment when an old source migration lease is refreshed', async () => { + const database = await openInMemoryRelayDatabase() + const now = REGISTERED_MIGRATION_ABANDON_MS + 10_000 + await insertCell(database, 'cell-a', false, 'existing-only') + await insertCell(database, 'cell-b', true, 'migration-only') + await insertMigration(database, now - 1) + + const fresh = await readRegisteredMigrationInventory(database, now) + + expect(fresh).toEqual({ + open: 1, + inactive: 1, + abandoned: 1, + pairs: [ + { + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + open: 1, + inactive: 1, + abandoned: 1 + } + ] + }) + expect(formatRegisteredMigrationInventory(fresh)).toEqual([ + '[orca-relay] migration inventory open=1 expiredRegisteredInactive=1 abandonedRegistered=1', + '[orca-relay] migration pair sourceCellId=cell-a targetCellId=cell-b open=1 expiredRegisteredInactive=1 abandoned=1' + ]) + expect( + ( + await readRegisteredMigrationInventory( + database, + now + 1 + ) + ).abandoned + ).toBe(1) + }) + + it('includes unregistered open migrations without logging a host identity', async () => { + const database = await openInMemoryRelayDatabase() + const now = 10_000 + await insertCell(database, 'cell-a', false, 'existing-only') + await insertCell(database, 'cell-b', true, 'migration-only') + await insertMigration(database, now + 60_000, null) + + const inventory = await readRegisteredMigrationInventory(database, now) + + expect(inventory).toEqual({ + open: 1, + inactive: 0, + abandoned: 0, + pairs: [ + { + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + open: 1, + inactive: 0, + abandoned: 0 + } + ] + }) + expect(formatRegisteredMigrationInventory(inventory).join('\n')).not.toContain( + '111111111111' + ) + }) +}) + +async function insertCell( + database: RelayDatabase, + cellId: string, + enabled: boolean, + admissionState: string +): Promise { + await database.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, ?, 4000, 0, 0, 0, 1)`, + [cellId, `https://${cellId}.example.test`, enabled ? 1 : 0] + ) + await database.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES (?, ?, 1)`, + [cellId, admissionState] + ) +} + +async function insertMigration( + database: RelayDatabase, + expiresAt: number, + targetRegisteredAt: number | null = 2 +): Promise { + await database.query( + `INSERT INTO relay_assignments + (user_id, relay_host_id, cell_id, assignment_epoch, lease_expires_at, + last_activity_at, reserved_controls, reserved_splices, reserved_invites, + pending_installs, pending_confirmations, migration_leases) + VALUES ('user-1', '111111111111', 'cell-b', 2, ?, ?, 0, 0, 0, 0, 0, 0)`, + [expiresAt, expiresAt] + ) + await database.query( + `INSERT INTO relay_assignment_migrations + (user_id, relay_host_id, source_cell_id, target_cell_id, previous_epoch, + assignment_epoch, source_request_units, target_reserved_units, expires_at, + target_registered_at, created_at, updated_at) + VALUES ('user-1', '111111111111', 'cell-a', 'cell-b', 1, 2, 1, 2, ?, ?, 1, 2)`, + [expiresAt, targetRegisteredAt] + ) +} diff --git a/cloud/apps/relay/src/registered-migration-inventory.ts b/cloud/apps/relay/src/registered-migration-inventory.ts new file mode 100644 index 00000000000..3d47ad31732 --- /dev/null +++ b/cloud/apps/relay/src/registered-migration-inventory.ts @@ -0,0 +1,104 @@ +import type { RelayDatabase, SqlRow } from './database.js' +import { + ABANDONED_REGISTERED_MIGRATION, + REGISTERED_MIGRATION_ABANDON_MS +} from './registered-migration-abandonment.js' + +export type RegisteredMigrationInventory = { + open: number + inactive: number + abandoned: number + pairs: Array<{ + sourceCellId: string + targetCellId: string + open: number + inactive: number + abandoned: number + }> +} + +export async function readRegisteredMigrationInventory( + database: RelayDatabase, + now: number +): Promise { + const abandonedBefore = now - REGISTERED_MIGRATION_ABANDON_MS + const openRows = await database.query( + `SELECT migration.source_cell_id, migration.target_cell_id, COUNT(*) AS open + FROM relay_assignment_migrations migration + WHERE migration.completed_at IS NULL AND migration.aborted_at IS NULL + GROUP BY migration.source_cell_id, migration.target_cell_id + ORDER BY migration.source_cell_id, migration.target_cell_id` + ) + const inactiveRows = await database.query( + `SELECT migration.source_cell_id, migration.target_cell_id, + COUNT(*) AS inactive, + COALESCE(SUM(CASE WHEN ${ABANDONED_REGISTERED_MIGRATION} + THEN 1 ELSE 0 END), 0) AS abandoned + FROM relay_assignment_migrations migration + WHERE migration.target_registered_at IS NOT NULL + AND migration.completed_at IS NULL AND migration.aborted_at IS NULL + AND migration.expires_at <= ? + AND NOT EXISTS ( + SELECT 1 FROM relay_assignment_activity_leases target_control + WHERE target_control.user_id = migration.user_id + AND target_control.relay_host_id = migration.relay_host_id + AND target_control.cell_id = migration.target_cell_id + AND target_control.activity_kind = 'control' + AND target_control.activity_id NOT LIKE 'control-pending:%' + ) + GROUP BY migration.source_cell_id, migration.target_cell_id + ORDER BY migration.source_cell_id, migration.target_cell_id`, + [now, abandonedBefore, abandonedBefore, now] + ) + const inactiveByPair = new Map( + inactiveRows.map((row) => [ + JSON.stringify([text(row, 'source_cell_id'), text(row, 'target_cell_id')]), + row + ]) + ) + const pairs = openRows.map((row) => { + const sourceCellId = text(row, 'source_cell_id') + const targetCellId = text(row, 'target_cell_id') + const inactive = inactiveByPair.get(JSON.stringify([sourceCellId, targetCellId])) + return { + sourceCellId, + targetCellId, + open: integer(row, 'open'), + inactive: integer(inactive, 'inactive'), + abandoned: integer(inactive, 'abandoned') + } + }) + return { + open: pairs.reduce((total, pair) => total + pair.open, 0), + inactive: pairs.reduce((total, pair) => total + pair.inactive, 0), + abandoned: pairs.reduce((total, pair) => total + pair.abandoned, 0), + pairs + } +} + +export function formatRegisteredMigrationInventory( + inventory: RegisteredMigrationInventory +): string[] { + const lines = [ + `[orca-relay] migration inventory open=${inventory.open}` + + ` expiredRegisteredInactive=${inventory.inactive}` + + ` abandonedRegistered=${inventory.abandoned}` + ] + for (const pair of inventory.pairs) { + lines.push( + `[orca-relay] migration pair sourceCellId=${pair.sourceCellId}` + + ` targetCellId=${pair.targetCellId} open=${pair.open}` + + ` expiredRegisteredInactive=${pair.inactive}` + + ` abandoned=${pair.abandoned}` + ) + } + return lines +} + +function text(row: SqlRow | undefined, column: string): string { + return String(row?.[column] ?? '') +} + +function integer(row: SqlRow | undefined, column: string): number { + return Number(row?.[column] ?? 0) +} diff --git a/cloud/apps/relay/src/relay-background-operation.test.ts b/cloud/apps/relay/src/relay-background-operation.test.ts new file mode 100644 index 00000000000..2c85317ba67 --- /dev/null +++ b/cloud/apps/relay/src/relay-background-operation.test.ts @@ -0,0 +1,44 @@ +import { describe, expect, it, vi } from 'vitest' +import { runRelayBackgroundOperation } from './relay-background-operation.js' + +describe('relay background operations', () => { + it('redacts free-form failure messages that could carry secrets', async () => { + const warn = vi.fn() + + await expect( + runRelayBackgroundOperation( + () => Promise.reject(new Error('postgresql://secret@database.invalid/relay')), + '[orca-relay] credential cleanup failed', + warn + ) + ).resolves.toBeUndefined() + + expect(warn).toHaveBeenCalledWith('[orca-relay] credential cleanup failed: Error: redacted') + expect(String(warn.mock.calls[0])).not.toContain('secret') + }) + + it('logs invariant slugs and SQLSTATE codes verbatim', async () => { + const warn = vi.fn() + const locked = Object.assign(new Error('regional_rehome_assignment_mismatch'), { + code: '55P03' + }) + + await runRelayBackgroundOperation( + () => Promise.reject(locked), + '[orca-relay] assignment cleanup failed', + warn + ) + + expect(warn).toHaveBeenCalledWith( + '[orca-relay] assignment cleanup failed: Error: regional_rehome_assignment_mismatch code=55P03' + ) + }) + + it('does not warn after successful maintenance work', async () => { + const warn = vi.fn() + + await runRelayBackgroundOperation(() => Promise.resolve(), 'unused', warn) + + expect(warn).not.toHaveBeenCalled() + }) +}) diff --git a/cloud/apps/relay/src/relay-background-operation.ts b/cloud/apps/relay/src/relay-background-operation.ts new file mode 100644 index 00000000000..be3943b7b1a --- /dev/null +++ b/cloud/apps/relay/src/relay-background-operation.ts @@ -0,0 +1,28 @@ +type Warn = (message: string) => void + +// A swallowed error hides which sweep step is failing (a silently dead +// cleanup chain stalled production rehoming for hours), but raw messages can +// embed secrets such as connection strings. Only two provably inert shapes +// are logged: this codebase's snake_case invariant slugs, and five-character +// SQLSTATE codes; everything else stays redacted. +function describeFailure(error: unknown): string { + const name = error instanceof Error ? error.name : typeof error + const message = error instanceof Error ? error.message : '' + const slug = /^[a-z0-9_]{1,64}$/.test(message) ? message : 'redacted' + const code = (error as { code?: unknown } | null)?.code + const sqlState = typeof code === 'string' && /^[0-9A-Z]{5}$/.test(code) ? ` code=${code}` : '' + return `${name}: ${slug}${sqlState}` +} + +export async function runRelayBackgroundOperation( + operation: () => Promise, + failureMessage: string, + warn: Warn = console.warn +): Promise { + try { + await operation() + } catch (error) { + // Dependency outages must fail readiness without crashing liveness. + warn(`${failureMessage}: ${describeFailure(error)}`) + } +} diff --git a/cloud/apps/relay/src/relay-connection-hard-cap.blackbox.test.ts b/cloud/apps/relay/src/relay-connection-hard-cap.blackbox.test.ts new file mode 100644 index 00000000000..230711e27d8 --- /dev/null +++ b/cloud/apps/relay/src/relay-connection-hard-cap.blackbox.test.ts @@ -0,0 +1,301 @@ +import { createHash, createHmac } from 'node:crypto' +import { generateKeyPair, exportJWK, SignJWT } from 'jose' +import { mkdtempSync, rmSync } from 'node:fs' +import { createServer, type Server } from 'node:http' +import { createServer as createNetServer } from 'node:net' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { + buildHostProofMacInput, + HOST_CHALLENGE_PLAINTEXT_DOMAIN +} from '@orca-cloud/relay-contract' +import nacl from 'tweetnacl' +import { afterEach, describe, expect, it, vi } from 'vitest' +import WebSocket from 'ws' +import type { RawData } from 'ws' +import type { RelayConfig } from './config.js' +import { openRelayDatabase, type RelayDatabase } from './database.js' +import { createRelayServer } from './relay-server.js' + +async function unusedPort(): Promise { + const server = createNetServer() + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const address = server.address() + if (!address || typeof address === 'string') throw new Error('missing test port') + await new Promise((resolve) => server.close(() => resolve())) + return address.port +} + +function openSocket(url: string, headers: Record = {}): Promise { + return new Promise((resolve, reject) => { + const socket = new WebSocket(url, { headers }) + socket.once('open', () => resolve(socket)) + socket.once('error', reject) + }) +} + +function rejectedUpgrade( + url: string, + headers: Record = {} +): Promise { + return new Promise((resolve, reject) => { + const socket = new WebSocket(url, { headers }) + socket.once('open', () => reject(new Error(`upgrade unexpectedly opened: ${url}`))) + socket.once('unexpected-response', (_request, response) => { + response.resume() + resolve(response.statusCode) + }) + socket.once('error', () => undefined) + }) +} + +function nextMessage(socket: WebSocket): Promise> { + return new Promise((resolve, reject) => { + socket.once('message', (data: RawData) => { + try { + resolve(JSON.parse(data.toString()) as Record) + } catch (error) { + reject(error) + } + }) + socket.once('error', reject) + }) +} + +async function proveControl( + socket: WebSocket, + hostId: string, + keyPair: nacl.BoxKeyPair, + rebind?: { secret: string; generation: number } +): Promise> { + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hostId, + assignmentEpoch: 1, + hostPublicKeyB64: Buffer.from(keyPair.publicKey).toString('base64'), + appVersion: 'test', + ...(rebind + ? { + controlResumeSecret: rebind.secret, + previousGeneration: rebind.generation + } + : {}) + }) + ) + const challenge = await nextMessage(socket) + const plaintext = nacl.box.open( + Buffer.from(String(challenge.ciphertextB64), 'base64'), + Buffer.from(String(challenge.nonceB64), 'base64'), + Buffer.from(String(challenge.relayEphemeralPublicKeyB64), 'base64'), + keyPair.secretKey + ) + if (!plaintext) throw new Error('host challenge did not decrypt') + const domainLength = new TextEncoder().encode( + `${HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0` + ).length + const transcriptLength = new DataView( + plaintext.buffer, + plaintext.byteOffset + domainLength, + 4 + ).getUint32(0, false) + const transcriptStart = domainLength + 4 + const transcript = plaintext.slice(transcriptStart, transcriptStart + transcriptLength) + const secret = plaintext.slice(transcriptStart + transcriptLength) + socket.send( + JSON.stringify({ + type: 'host-challenge-ack', + challengeId: challenge.challengeId, + proofB64: createHmac('sha256', secret) + .update(buildHostProofMacInput(transcript)) + .digest('base64') + }) + ) + return await nextMessage(socket) +} + +describe('relay connection hard cap', () => { + const cleanup: Array<() => Promise | void> = [] + + afterEach(async () => { + for (const close of cleanup.splice(0).reverse()) await close() + }) + + it('preserves control headroom and never disturbs established sockets', async () => { + const keys = await generateKeyPair('ES256') + const publicJwk = await exportJWK(keys.publicKey) + const adminKeys = await generateKeyPair('RS256') + const adminPublicJwk = await exportJWK(adminKeys.publicKey) + const jwks = createServer((_request, response) => { + response.setHeader('content-type', 'application/json') + response.end( + JSON.stringify({ + keys: [ + { ...publicJwk, kid: 'test-key', alg: 'ES256', use: 'sig' }, + { ...adminPublicJwk, kid: 'admin-key', alg: 'RS256', use: 'sig' } + ] + }) + ) + }) + await new Promise((resolve) => jwks.listen(0, '127.0.0.1', resolve)) + cleanup.push(() => new Promise((resolve) => jwks.close(() => resolve()))) + const jwksAddress = jwks.address() + if (!jwksAddress || typeof jwksAddress === 'string') throw new Error('missing JWKS address') + const issuer = `http://127.0.0.1:${jwksAddress.port}` + const port = await unusedPort() + const relayUrl = `http://127.0.0.1:${port}` + const dataDir = mkdtempSync(join(tmpdir(), 'orca-relay-hard-cap-')) + cleanup.push(() => rmSync(dataDir, { recursive: true, force: true })) + const database: RelayDatabase = await openRelayDatabase({ dataDir }) + cleanup.push(() => database.close()) + const config = { + port, + publicUrl: relayUrl, + cellUrl: relayUrl, + authIssuer: issuer, + authAudience: 'orca-relay', + jwksUrl: `${issuer}/jwks`, + assignmentSigningKey: new TextEncoder().encode('test-assignment-key-with-at-least-32-bytes'), + role: 'combined', + cellId: 'combined', + cells: [ + { + id: 'combined', + url: relayUrl, + capacityRequests: 900, + connectionHardCap: 600, + connectionUnobservedBound: 60 + } + ], + adminAudience: `${relayUrl}/admin`, + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + connectionHardCap: 600, + connectionUnobservedBound: 60, + adminJwksUrl: `${issuer}/jwks`, + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir + } satisfies RelayConfig + const relay = createRelayServer(config, database, { + connectionLedgerLimits: { hardCap: 5, controlReserve: 1 } + }) + relay.server.listen(port, '127.0.0.1') + await new Promise((resolve) => relay.server.once('listening', resolve)) + cleanup.push(() => new Promise((resolve) => relay.server.close(() => resolve()))) + const wsUrl = relayUrl.replace('http:', 'ws:') + const adminToken = await new SignJWT({ + email: 'deploy@example.com', + email_verified: true + }) + .setProtectedHeader({ alg: 'RS256', kid: 'admin-key' }) + .setIssuer('https://accounts.google.com') + .setAudience(`${relayUrl}/admin`) + .setSubject('deploy-subject') + .setIssuedAt() + .setExpirationTime('5m') + .sign(adminKeys.privateKey) + const statusResponse = await fetch(`${relayUrl}/v1/admin/runtime-status`, { + method: 'POST', + headers: { + authorization: `Bearer ${adminToken}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1 }) + }) + expect(statusResponse.status).toBe(200) + expect(await statusResponse.json()).toMatchObject({ + connectionCapacity: { + hardCap: 600, + controlRebindReserve: 100, + ordinaryConnectionLimit: 500, + unobservedBound: 60, + normalAdmissionPause: 440 + } + }) + + const unmatchedData = await Promise.all( + ['unmatched-1', 'unmatched-2', 'unmatched-3'].map((connectionId) => + openSocket(`${wsUrl}/v1/host/data/${connectionId}`) + ) + ) + cleanup.push(() => unmatchedData.forEach((socket) => socket.terminate())) + expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(3) + expect(await rejectedUpgrade(`${wsUrl}/v1/connect/abcdefghijklmnop`)).toBe(503) + expect(unmatchedData.every((socket) => socket.readyState === WebSocket.OPEN)).toBe(true) + + const hostKeyPair = nacl.box.keyPair() + const hostId = createHash('sha256') + .update(hostKeyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const token = await new SignJWT({ + prof: 'profile-1', + org: 'org-1', + purpose: 'host-control', + relayHostId: hostId + }) + .setProtectedHeader({ alg: 'ES256', kid: 'test-key' }) + .setIssuer(issuer) + .setAudience('orca-relay') + .setSubject('user-1') + .setIssuedAt() + .setExpirationTime('5m') + .sign(keys.privateKey) + const controlHeaders = { authorization: `Bearer ${token}` } + const control = await openSocket(`${wsUrl}/v1/host/control`, controlHeaders) + cleanup.push(() => control.terminate()) + const controlAck = await proveControl(control, hostId, hostKeyPair) + expect(controlAck).toMatchObject({ type: 'host-hello-ack', generation: 1 }) + expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(4) + expect(await rejectedUpgrade(`${wsUrl}/v1/host/data/another`)).toBe(503) + const unrelatedToken = await new SignJWT({ + prof: 'profile-1', + purpose: 'host-control', + relayHostId: 'ponmlkjihgfedcba' + }) + .setProtectedHeader({ alg: 'ES256', kid: 'test-key' }) + .setIssuer(issuer) + .setAudience('orca-relay') + .setSubject('user-2') + .setIssuedAt() + .setExpirationTime('5m') + .sign(keys.privateKey) + expect( + await rejectedUpgrade(`${wsUrl}/v1/host/control`, { + authorization: `Bearer ${unrelatedToken}` + }) + ).toBe(503) + const failedBorrower = await openSocket(`${wsUrl}/v1/host/control`, controlHeaders) + cleanup.push(() => failedBorrower.terminate()) + expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(5) + const failedBorrowerClosed = new Promise((resolve) => + failedBorrower.once('close', (code) => resolve(code)) + ) + failedBorrower.send(JSON.stringify({ type: 'host-hello' })) + expect(await failedBorrowerClosed).toBe(4401) + await vi.waitFor(() => expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(4)) + expect(control.readyState).toBe(WebSocket.OPEN) + const replacementControl = await openSocket(`${wsUrl}/v1/host/control`, controlHeaders) + cleanup.push(() => replacementControl.terminate()) + expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(5) + const controlClosed = new Promise((resolve) => + control.once('close', (code) => resolve(code)) + ) + const replacementAck = await proveControl(replacementControl, hostId, hostKeyPair, { + secret: String(controlAck.controlResumeSecret), + generation: 1 + }) + expect(replacementAck).toMatchObject({ type: 'host-hello-ack', generation: 1 }) + expect(await controlClosed).toBe(4408) + await vi.waitFor(() => expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(4)) + expect(relay.runtimeCounts().enforcedConnectionUnits).toBe(4) + }) +}) diff --git a/cloud/apps/relay/src/relay-connection-ledger.test.ts b/cloud/apps/relay/src/relay-connection-ledger.test.ts new file mode 100644 index 00000000000..f807c34d20d --- /dev/null +++ b/cloud/apps/relay/src/relay-connection-ledger.test.ts @@ -0,0 +1,172 @@ +import { EventEmitter } from 'node:events' +import { describe, expect, it } from 'vitest' +import type WebSocket from 'ws' +import { RelayConnectionLedger } from './relay-connection-ledger.js' + +function socket(): { webSocket: WebSocket; close: () => void } { + const emitter = new EventEmitter() + return { + webSocket: emitter as WebSocket, + close: () => emitter.emit('close') + } +} + +describe('RelayConnectionLedger', () => { + it('orders connection admissions before covering absolute snapshots', () => { + const ledger = new RelayConnectionLedger(6, 2) + const upgrade = ledger.tryReserveControl(false)! + const beforePromotion = ledger.snapshot() + const controlSocket = socket() + upgrade.promote(controlSocket.webSocket) + const afterPromotion = ledger.snapshot() + + expect(beforePromotion.inclusionWatermark).toBeGreaterThanOrEqual( + upgrade.inclusionWatermark + ) + expect(afterPromotion.inclusionWatermark).toBeGreaterThan( + beforePromotion.inclusionWatermark + ) + expect(afterPromotion).toMatchObject({ + physicalConnections: 1, + inFlightConnections: 0, + enforcedConnectionUnits: 1 + }) + }) + + it.each([600, 1_000, 3_000])( + 'enforces the %i-unit boundary with 100 control units reserved', + (hardCap) => { + const ledger = new RelayConnectionLedger(hardCap, 100) + const sockets: Array<{ webSocket: WebSocket; close: () => void }> = [] + for (let index = 0; index < (hardCap - 100) / 2; index++) { + const admission = ledger.tryReservePhone() + expect(admission).not.toBeNull() + const phoneSocket = socket() + admission!.upgrade.promote(phoneSocket.webSocket) + admission!.hostData.bind(`connection-${index}`) + sockets.push(phoneSocket) + } + expect(ledger.tryReservePhone()).toBeNull() + expect(ledger.tryReserveControl(false)).toBeNull() + for (let index = 0; index < 100; index++) { + const upgrade = ledger.tryReserveControl(true) + expect(upgrade).not.toBeNull() + const controlSocket = socket() + upgrade!.promote(controlSocket.webSocket) + sockets.push(controlSocket) + } + + expect(ledger.counts().enforcedConnectionUnits).toBe(hardCap) + expect(ledger.tryReserveControl(true)).toBeNull() + sockets.at(-1)!.close() + const replacement = ledger.tryReserveControl(true) + expect(replacement).not.toBeNull() + replacement!.release() + } + ) + + it('reserves both phone legs below the control reserve', () => { + const ledger = new RelayConnectionLedger(6, 2) + const first = ledger.tryReservePhone() + const second = ledger.tryReservePhone() + + expect(first).not.toBeNull() + expect(second).not.toBeNull() + expect(ledger.tryReservePhone()).toBeNull() + expect(ledger.counts()).toEqual({ + physicalConnections: 0, + inFlightConnections: 2, + reservedConnectionUnits: 2, + enforcedConnectionUnits: 4 + }) + expect(ledger.tryReserveControl(false)).toBeNull() + expect(ledger.tryReserveControl(true)).not.toBeNull() + expect(ledger.tryReserveControl(true)).not.toBeNull() + expect(ledger.tryReserveControl(true)).toBeNull() + }) + + it('transfers a pending host-data unit without increasing enforced capacity', () => { + const ledger = new RelayConnectionLedger(6, 2) + const phone = ledger.tryReservePhone()! + const phoneSocket = socket() + phone.upgrade.promote(phoneSocket.webSocket) + phone.hostData.bind('connection-1') + + const data = ledger.tryReserveHostData('connection-1')! + expect(ledger.counts().enforcedConnectionUnits).toBe(2) + const dataSocket = socket() + data.promote(dataSocket.webSocket) + expect(ledger.counts().enforcedConnectionUnits).toBe(2) + expect(data.commitHostData()).toBe(true) + + dataSocket.close() + dataSocket.close() + expect(ledger.counts()).toEqual({ + physicalConnections: 1, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 1 + }) + }) + + it('restores a claimed reservation after rejected host-data authentication', () => { + const ledger = new RelayConnectionLedger(6, 2) + const phone = ledger.tryReservePhone()! + const phoneSocket = socket() + phone.upgrade.promote(phoneSocket.webSocket) + phone.hostData.bind('connection-1') + + const rejected = ledger.tryReserveHostData('connection-1')! + const rejectedSocket = socket() + rejected.promote(rejectedSocket.webSocket) + rejectedSocket.close() + + expect(ledger.counts()).toEqual({ + physicalConnections: 1, + inFlightConnections: 0, + reservedConnectionUnits: 1, + enforcedConnectionUnits: 2 + }) + expect(ledger.tryReserveHostData('connection-1')).not.toBeNull() + }) + + it('does not restore a claimed reservation after the phone closes', () => { + const ledger = new RelayConnectionLedger(6, 2) + const phone = ledger.tryReservePhone()! + const phoneSocket = socket() + phone.upgrade.promote(phoneSocket.webSocket) + phone.hostData.bind('connection-1') + const data = ledger.tryReserveHostData('connection-1')! + const dataSocket = socket() + data.promote(dataSocket.webSocket) + + phone.hostData.release() + phoneSocket.close() + dataSocket.close() + + expect(ledger.counts()).toEqual({ + physicalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }) + expect(ledger.tryReserveHostData('connection-1')).not.toBeNull() + }) + + it('releases failed in-flight upgrades exactly once', () => { + const ledger = new RelayConnectionLedger(6, 2) + const phone = ledger.tryReservePhone()! + + phone.upgrade.release() + phone.upgrade.release() + phone.hostData.release() + phone.hostData.release() + + expect(ledger.counts()).toEqual({ + physicalConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }) + }) +}) diff --git a/cloud/apps/relay/src/relay-connection-ledger.ts b/cloud/apps/relay/src/relay-connection-ledger.ts new file mode 100644 index 00000000000..9c678695fa8 --- /dev/null +++ b/cloud/apps/relay/src/relay-connection-ledger.ts @@ -0,0 +1,229 @@ +import type WebSocket from 'ws' + +export type RelayConnectionLedgerCounts = { + physicalConnections: number + inFlightConnections: number + reservedConnectionUnits: number + enforcedConnectionUnits: number +} + +export type RelayConnectionLedgerSnapshot = RelayConnectionLedgerCounts & { + inclusionWatermark: number +} + +export type PendingHostDataReservation = { + bind: (connectionId: string) => void + release: () => void +} + +type ReservationState = 'reserved' | 'claimed' | 'consumed' | 'released' +type UpgradeState = 'in-flight' | 'physical' | 'released' + +class HostDataReservation implements PendingHostDataReservation { + private state: ReservationState = 'reserved' + private connectionId: string | null = null + + constructor(private readonly ledger: RelayConnectionLedger) {} + + bind(connectionId: string): void { + if (this.state !== 'reserved' || this.connectionId !== null) { + throw new Error('host_data_reservation_already_bound') + } + this.connectionId = connectionId + this.ledger.bindHostData(connectionId, this) + } + + release(): void { + if (this.state === 'reserved') { + this.ledger.releaseReserved(this) + } + if (this.state !== 'consumed') { + this.state = 'released' + } + } + + claim(): void { + if (this.state !== 'reserved') throw new Error('host_data_reservation_unavailable') + this.state = 'claimed' + } + + consume(): boolean { + if (this.state !== 'claimed') return false + this.state = 'consumed' + return true + } + + restore(): void { + if (this.state !== 'claimed') return + this.state = this.ledger.restoreReserved(this) ? 'reserved' : 'released' + } + + matches(connectionId: string): boolean { + return this.connectionId === connectionId + } + + get boundConnectionId(): string | null { + return this.connectionId + } +} + +export class RelayConnectionUpgrade { + private state: UpgradeState = 'in-flight' + + constructor( + private readonly ledger: RelayConnectionLedger, + readonly inclusionWatermark: number, + private readonly hostDataReservation: HostDataReservation | null = null + ) {} + + promote(socket: WebSocket): void { + if (this.state !== 'in-flight') throw new Error('connection_upgrade_not_in_flight') + this.state = 'physical' + this.ledger.promoteUpgrade() + socket.once('close', () => this.release()) + } + + commitHostData(): boolean { + return this.hostDataReservation?.consume() ?? false + } + + release(): void { + if (this.state === 'released') return + if (this.state === 'in-flight') this.ledger.releaseUpgrade() + else this.ledger.releasePhysical() + this.state = 'released' + this.hostDataReservation?.restore() + } +} + +export type PhoneConnectionAdmission = { + upgrade: RelayConnectionUpgrade + hostData: PendingHostDataReservation +} + +export class RelayConnectionLedger { + private physicalConnections = 0 + private inFlightConnections = 0 + private reservedConnectionUnits = 0 + private inclusionWatermark = 0 + private readonly hostDataByConnectionId = new Map() + + constructor( + private readonly hardCap: number, + private readonly controlReserve: number + ) { + if (hardCap <= 0 || controlReserve < 0 || controlReserve >= hardCap) { + throw new Error('invalid_connection_ledger_capacity') + } + } + + tryReserveControl(rebind: boolean): RelayConnectionUpgrade | null { + return this.reserveUpgrade(rebind ? this.hardCap : this.normalAdmissionLimit) + } + + tryReservePhone(): PhoneConnectionAdmission | null { + if (this.enforcedConnectionUnits + 2 > this.normalAdmissionLimit) return null + const reservation = new HostDataReservation(this) + const inclusionWatermark = this.advanceWatermark() + this.inFlightConnections++ + this.reservedConnectionUnits++ + return { + upgrade: new RelayConnectionUpgrade(this, inclusionWatermark), + hostData: reservation + } + } + + tryReserveHostData(connectionId: string): RelayConnectionUpgrade | null { + const reservation = this.hostDataByConnectionId.get(connectionId) + if (reservation?.matches(connectionId)) { + this.hostDataByConnectionId.delete(connectionId) + reservation.claim() + const inclusionWatermark = this.advanceWatermark() + this.reservedConnectionUnits-- + this.inFlightConnections++ + return new RelayConnectionUpgrade(this, inclusionWatermark, reservation) + } + return this.reserveUpgrade(this.normalAdmissionLimit) + } + + counts(): RelayConnectionLedgerCounts { + return { + physicalConnections: this.physicalConnections, + inFlightConnections: this.inFlightConnections, + reservedConnectionUnits: this.reservedConnectionUnits, + enforcedConnectionUnits: this.enforcedConnectionUnits + } + } + + snapshot(): RelayConnectionLedgerSnapshot { + return { + ...this.counts(), + inclusionWatermark: this.advanceWatermark() + } + } + + bindHostData(connectionId: string, reservation: HostDataReservation): void { + if (this.hostDataByConnectionId.has(connectionId)) { + throw new Error('host_data_reservation_conflict') + } + this.hostDataByConnectionId.set(connectionId, reservation) + } + + releaseReserved(reservation: HostDataReservation): void { + const connectionId = reservation.boundConnectionId + if (connectionId && this.hostDataByConnectionId.get(connectionId) === reservation) { + this.hostDataByConnectionId.delete(connectionId) + } + this.advanceWatermark() + this.reservedConnectionUnits-- + } + + restoreReserved(reservation: HostDataReservation): boolean { + const connectionId = reservation.boundConnectionId + if (!connectionId || this.hostDataByConnectionId.has(connectionId)) { + return false + } + this.advanceWatermark() + this.reservedConnectionUnits++ + this.hostDataByConnectionId.set(connectionId, reservation) + return true + } + + promoteUpgrade(): void { + this.advanceWatermark() + this.inFlightConnections-- + this.physicalConnections++ + } + + releaseUpgrade(): void { + this.advanceWatermark() + this.inFlightConnections-- + } + + releasePhysical(): void { + this.advanceWatermark() + this.physicalConnections-- + } + + private reserveUpgrade(limit: number): RelayConnectionUpgrade | null { + if (this.enforcedConnectionUnits + 1 > limit) return null + const inclusionWatermark = this.advanceWatermark() + this.inFlightConnections++ + return new RelayConnectionUpgrade(this, inclusionWatermark) + } + + private advanceWatermark(): number { + this.inclusionWatermark++ + return this.inclusionWatermark + } + + private get normalAdmissionLimit(): number { + return this.hardCap - this.controlReserve + } + + private get enforcedConnectionUnits(): number { + return ( + this.physicalConnections + this.inFlightConnections + this.reservedConnectionUnits + ) + } +} diff --git a/cloud/apps/relay/src/relay-first-frame-failure.blackbox.test.ts b/cloud/apps/relay/src/relay-first-frame-failure.blackbox.test.ts new file mode 100644 index 00000000000..57e47a065d3 --- /dev/null +++ b/cloud/apps/relay/src/relay-first-frame-failure.blackbox.test.ts @@ -0,0 +1,124 @@ +import { createServer as createNetServer } from 'node:net' +import { RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract' +import { afterEach, describe, expect, it, vi } from 'vitest' +import WebSocket from 'ws' +import type { RelayConfig } from './config.js' +import type { RelayDatabase } from './database.js' +import { createRelayServer } from './relay-server.js' + +async function unusedPort(): Promise { + const server = createNetServer() + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const address = server.address() + if (!address || typeof address === 'string') throw new Error('missing test port') + await new Promise((resolve) => server.close(() => resolve())) + return address.port +} + +function openSocket(url: string): Promise { + return new Promise((resolve, reject) => { + const socket = new WebSocket(url) + socket.once('open', () => resolve(socket)) + socket.once('error', reject) + }) +} + +describe('relay first-frame failures', () => { + const cleanup: Array<() => Promise | void> = [] + + afterEach(async () => { + for (const close of cleanup.splice(0).reverse()) await close() + vi.restoreAllMocks() + }) + + it('contains phone authentication failures and releases connection capacity', async () => { + const port = await unusedPort() + const relayUrl = `http://127.0.0.1:${port}` + const poolTimeout = new Error('injected pool connection timeout') + const database: RelayDatabase = { + query: vi.fn(async () => { + throw poolTimeout + }), + queryLocked: vi.fn(async () => { + throw poolTimeout + }), + transaction: vi.fn(async (operation) => await operation(database)), + close: vi.fn(async () => undefined) + } + const config = { + port, + publicUrl: relayUrl, + cellUrl: relayUrl, + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [{ id: 'production-gce-c3', url: relayUrl, capacityRequests: 4_000 }], + adminAudience: `${relayUrl}/admin`, + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + connectionHardCap: 600, + connectionUnobservedBound: 60, + adminJwksUrl: 'https://auth.example.com/admin-jwks', + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './test-data' + } satisfies RelayConfig + const relay = createRelayServer(config, database, { + connectionLedgerLimits: { hardCap: 5, controlReserve: 1 } + }) + relay.server.listen(port, '127.0.0.1') + await new Promise((resolve) => relay.server.once('listening', resolve)) + cleanup.push(() => new Promise((resolve) => relay.server.close(() => resolve()))) + vi.spyOn(console, 'log').mockImplementation(() => undefined) + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const unhandled: unknown[] = [] + const onUnhandled = (reason: unknown): void => { + unhandled.push(reason) + } + process.on('unhandledRejection', onUnhandled) + cleanup.push(() => { + process.off('unhandledRejection', onUnhandled) + }) + + const socket = await openSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop` + ) + cleanup.push(() => socket.terminate()) + expect(relay.connectionSnapshot()).toMatchObject({ + physicalConnections: 1, + reservedConnectionUnits: 1, + enforcedConnectionUnits: 2 + }) + const closed = new Promise((resolve) => + socket.once('close', (code) => resolve(code)) + ) + socket.send( + JSON.stringify({ + type: 'relay-auth', + v: 1, + mode: 'connect', + credential: Buffer.alloc(32, 1).toString('base64url') + }) + ) + + expect(await closed).toBe(RELAY_CLOSE_CODE.LIMIT_EXCEEDED) + await vi.waitFor(() => + expect(relay.connectionSnapshot()).toMatchObject({ + physicalConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0 + }) + ) + await new Promise((resolve) => setImmediate(resolve)) + expect(unhandled).toEqual([]) + }) +}) diff --git a/cloud/apps/relay/src/relay-host-log-digest.ts b/cloud/apps/relay/src/relay-host-log-digest.ts new file mode 100644 index 00000000000..b232170a6d4 --- /dev/null +++ b/cloud/apps/relay/src/relay-host-log-digest.ts @@ -0,0 +1,7 @@ +import { createHash } from 'node:crypto' + +// Store-level 503s were previously silent; the digest keeps per-host log +// correlation possible without emitting the raw relay host id. +export function relayHostLogDigest(relayHostId: string): string { + return createHash('sha256').update(relayHostId).digest('hex').slice(0, 12) +} diff --git a/cloud/apps/relay/src/relay-host-proof-failure.blackbox.test.ts b/cloud/apps/relay/src/relay-host-proof-failure.blackbox.test.ts new file mode 100644 index 00000000000..924523685e3 --- /dev/null +++ b/cloud/apps/relay/src/relay-host-proof-failure.blackbox.test.ts @@ -0,0 +1,153 @@ +import { createHash } from 'node:crypto' +import { createServer, type Server } from 'node:http' +import { createServer as createNetServer } from 'node:net' +import { exportJWK, generateKeyPair, SignJWT } from 'jose' +import { RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract' +import nacl from 'tweetnacl' +import { afterEach, describe, expect, it, vi } from 'vitest' +import WebSocket from 'ws' +import type { RelayConfig } from './config.js' +import type { RelayDatabase } from './database.js' +import { createRelayServer } from './relay-server.js' + +async function unusedPort(): Promise { + const server = createNetServer() + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const address = server.address() + if (!address || typeof address === 'string') throw new Error('missing test port') + await new Promise((resolve) => server.close(() => resolve())) + return address.port +} + +describe('relay host proof failures', () => { + const cleanup: Array<() => Promise | void> = [] + + afterEach(async () => { + for (const close of cleanup.splice(0).reverse()) await close() + vi.restoreAllMocks() + }) + + // The production crash signature: a pg-pool connect timeout rejecting out of + // verifyCellAssignment inside beginProof killed whole cells as an unhandled + // rejection. The guard must contain it to this one handshake. + it('contains a database timeout during host hello to one socket', async () => { + const keys = await generateKeyPair('ES256') + const publicJwk = await exportJWK(keys.publicKey) + const jwksServer: Server = createServer((_request, response) => { + response.setHeader('content-type', 'application/json') + response.end( + JSON.stringify({ keys: [{ ...publicJwk, kid: 'test-key', alg: 'ES256', use: 'sig' }] }) + ) + }) + await new Promise((resolve) => jwksServer.listen(0, '127.0.0.1', resolve)) + cleanup.push(() => new Promise((resolve) => jwksServer.close(() => resolve()))) + const jwksAddress = jwksServer.address() + if (!jwksAddress || typeof jwksAddress === 'string') throw new Error('missing JWKS address') + const issuer = `http://127.0.0.1:${jwksAddress.port}` + + const port = await unusedPort() + const relayUrl = `http://127.0.0.1:${port}` + const poolTimeout = new Error('Connection terminated due to connection timeout') + const database: RelayDatabase = { + query: vi.fn(async () => { + throw poolTimeout + }), + queryLocked: vi.fn(async () => { + throw poolTimeout + }), + transaction: vi.fn(async (operation) => await operation(database)), + close: vi.fn(async () => undefined) + } + const config = { + port, + publicUrl: relayUrl, + cellUrl: relayUrl, + authIssuer: issuer, + authAudience: 'orca-relay', + jwksUrl: issuer, + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [{ id: 'production-gce-c3', url: relayUrl, capacityRequests: 4_000 }], + adminAudience: `${relayUrl}/admin`, + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + connectionHardCap: 600, + connectionUnobservedBound: 60, + adminJwksUrl: `${issuer}/admin-jwks`, + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './test-data' + } satisfies RelayConfig + const relay = createRelayServer(config, database) + relay.server.listen(port, '127.0.0.1') + await new Promise((resolve) => relay.server.once('listening', resolve)) + cleanup.push(() => new Promise((resolve) => relay.server.close(() => resolve()))) + vi.spyOn(console, 'log').mockImplementation(() => undefined) + const warn = vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const unhandled: unknown[] = [] + const onUnhandled = (reason: unknown): void => { + unhandled.push(reason) + } + process.on('unhandledRejection', onUnhandled) + cleanup.push(() => { + process.off('unhandledRejection', onUnhandled) + }) + + const keyPair = nacl.box.keyPair() + const hostId = createHash('sha256') + .update(keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const token = await new SignJWT({ + prof: 'profile-1', + org: 'org-1', + purpose: 'host-control', + relayHostId: hostId + }) + .setProtectedHeader({ alg: 'ES256', kid: 'test-key' }) + .setIssuer(issuer) + .setAudience('orca-relay') + .setSubject('user-1') + .setIssuedAt() + .setExpirationTime('5m') + .sign(keys.privateKey) + const socket = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/host/control`, { + headers: { authorization: `Bearer ${token}` }, + perMessageDeflate: false + }) + cleanup.push(() => socket.terminate()) + await new Promise((resolve, reject) => { + socket.once('open', resolve) + socket.once('error', reject) + }) + const closed = new Promise((resolve) => + socket.once('close', (code) => resolve(code)) + ) + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hostId, + assignmentEpoch: 1, + hostPublicKeyB64: Buffer.from(keyPair.publicKey).toString('base64'), + appVersion: 'test' + }) + ) + + expect(await closed).toBe(RELAY_CLOSE_CODE.LIMIT_EXCEEDED) + expect( + warn.mock.calls.map((call) => String(call[0])) + ).toContain( + '[orca-relay] host hello proof failed: Connection terminated due to connection timeout' + ) + await new Promise((resolve) => setImmediate(resolve)) + expect(unhandled).toEqual([]) + }) +}) diff --git a/cloud/apps/relay/src/relay-observability.test.ts b/cloud/apps/relay/src/relay-observability.test.ts new file mode 100644 index 00000000000..2b9ceb0b72a --- /dev/null +++ b/cloud/apps/relay/src/relay-observability.test.ts @@ -0,0 +1,246 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RelayDatabase } from './database.js' +import { observeRelayDatabase } from './observed-relay-database.js' +import { + observedRelayRequests, + RelayObservability, + type RelayProcessCounts +} from './relay-observability.js' + +const counts: RelayProcessCounts = { + totalConnections: 9, + preAuthConnections: 1, + controls: 2, + splices: 3, + pendingSplices: 1, + queuedBytes: 4096, + databasePoolTotal: 3, + databasePoolIdle: 0, + databasePoolWaiting: 2, + databasePoolWaitersMax: 3, + databasePoolOldestWaitMs: 750, + databasePoolWaitMsMax: 1_250 +} + +describe('relay observability', () => { + it('emits safe readiness dependency outcomes', () => { + const entries: Array> = [] + const observability = new RelayObservability( + { role: 'cell', cellId: 'production-gce-c28', region: 'asia-east2' }, + (entry) => entries.push(entry) + ) + + observability.recordReadiness({ + ready: false, + failure: 'sql_failed', + jwksLatencyMs: 12, + sqlLatencyMs: 2_001, + totalLatencyMs: 2_013 + }) + + expect(entries).toEqual([ + { + severity: 'WARNING', + message: 'Orca Relay readiness check', + event: 'orca_relay_readiness_check', + metricVersion: 1, + role: 'cell', + cellId: 'production-gce-c28', + region: 'asia-east2', + ready: false, + failure: 'sql_failed', + jwksLatencyMs: 12, + sqlLatencyMs: 2_001, + totalLatencyMs: 2_013 + } + ]) + }) + + it('excludes sockets stuck in closing state from observed relay work', () => { + expect(observedRelayRequests(counts)).toBe(7) + }) + + it('keeps rejection reasons separate per lane and resets them each flush', () => { + const entries: Array> = [] + const observability = new RelayObservability( + { role: 'director', cellId: 'director', region: 'us-central1' }, + (entry) => entries.push(entry) + ) + observability.recordAssignmentAdmission('placement-rejected') + observability.recordAssignmentRejectionReason('placement', 'host-rate-limited') + observability.recordAssignmentRejectionReason('placement', 'host-rate-limited') + observability.recordAssignmentRejectionReason('placement', 'queue-full') + observability.recordAssignmentRejectionReason('sticky', 'wait-timeout') + observability.flush(counts) + observability.flush(counts) + + expect(entries[0]).toMatchObject({ + placementAssignmentRejectionsDelta: 1, + placementRejectionsByReasonDelta: { 'host-rate-limited': 2, 'queue-full': 1 }, + stickyRejectionsByReasonDelta: { 'wait-timeout': 1 } + }) + expect(entries[1]).toMatchObject({ + placementRejectionsByReasonDelta: {}, + stickyRejectionsByReasonDelta: {} + }) + }) + + it('aggregates coarse region requests, selections, fallbacks, and outages', () => { + const entries: Array> = [] + const observability = new RelayObservability( + { role: 'director', cellId: 'director', region: 'us-central1' }, + (entry) => entries.push(entry) + ) + observability.recordRegionRequest('asia-east2') + observability.recordRegionRequest(undefined) + observability.recordRegionSelection({ + targetRegion: 'asia-east2', + selectedRegion: 'us-central1', + fallback: true + }) + observability.recordRegionSelection({ targetRegion: 'asia-east2', fallback: false }) + observability.flush(counts) + observability.flush(counts) + + expect(entries[0]).toMatchObject({ + requestedRegionsDelta: { 'asia-east2': 1, unhinted: 1 }, + selectedRegionsDelta: { 'us-central1': 1 }, + regionFallbacksDelta: { 'asia-east2': 1 }, + unavailableRegionsDelta: { 'asia-east2': 1 } + }) + expect(entries[1]).toMatchObject({ + requestedRegionsDelta: {}, + selectedRegionsDelta: {}, + regionFallbacksDelta: {}, + unavailableRegionsDelta: {} + }) + }) + + it('emits bounded aggregate runtime signals without identities or credentials', () => { + const entries: Array> = [] + const observability = new RelayObservability( + { role: 'cell', cellId: 'staging-c1', region: 'asia-east2' }, + (entry) => entries.push(entry) + ) + observability.recordAuth(true) + observability.recordAuth(false) + observability.recordForwardedBytes(123) + observability.recordHttp(45.6789) + observability.recordReconnect() + observability.recordSql(12.3456, true) + observability.recordSql(4, false) + observability.recordControlRenewal(2, 'renewed') + observability.recordControlRenewal(8, 'control_activity_not_found') + observability.recordControlRenewal(4, 'renewed') + observability.recordControlActivityRecovery(true) + observability.recordControlActivityRecovery(false) + observability.flush(counts) + observability.flush(counts) + + expect(entries[0]).toMatchObject({ + event: 'orca_relay_runtime_metrics', + metricVersion: 2, + role: 'cell', + cellId: 'staging-c1', + region: 'asia-east2', + ...counts, + forwardedBytesDelta: 123, + authSuccessesDelta: 1, + authFailuresDelta: 1, + reconnectsDelta: 1, + sqlQueriesDelta: 2, + sqlFailuresDelta: 1, + sqlLatencyMsMax: 12.346, + controlRenewalsByOutcomeDelta: { renewed: 2, control_activity_not_found: 1 }, + controlRenewalsDelta: 3, + controlRenewalSuccessesDelta: 2, + controlRenewalLeaseMissesDelta: 1, + controlRenewalLatencyMsP50: 4, + controlRenewalLatencyMsP95: 8, + controlRenewalLatencyMsMax: 8, + controlActivityRecoveriesDelta: 1, + controlActivityRecoveryFailuresDelta: 1, + httpLatencyMsMax: 45.679 + }) + expect(entries[1]).toMatchObject({ + forwardedBytesDelta: 0, + authSuccessesDelta: 0, + authFailuresDelta: 0, + reconnectsDelta: 0, + sqlQueriesDelta: 0, + sqlFailuresDelta: 0, + sqlLatencyMsMax: 0, + controlRenewalsByOutcomeDelta: {}, + controlRenewalsDelta: 0, + controlRenewalSuccessesDelta: 0, + controlRenewalLeaseMissesDelta: 0, + controlRenewalLatencyMsP50: 0, + controlRenewalLatencyMsP95: 0, + controlRenewalLatencyMsMax: 0, + controlActivityRecoveriesDelta: 0, + controlActivityRecoveryFailuresDelta: 0, + httpLatencyMsMax: 0 + }) + expect(JSON.stringify(entries)).not.toMatch(/token|credential|userId|relayHostId/) + }) + + it('aggregates control and splice closes as bounded per-reason deltas', () => { + const entries: Array> = [] + const observability = new RelayObservability( + { role: 'cell', cellId: 'staging-c1', region: 'us-central1' }, + (entry) => entries.push(entry) + ) + observability.recordControlClose(1006) + observability.recordControlClose(1006) + observability.recordControlClose(4402) + observability.recordSpliceClose('host-oversize-frame') + observability.recordSpliceClose('queue-limit') + observability.flush(counts) + observability.flush(counts) + + expect(entries[0]).toMatchObject({ + controlClosesByCodeDelta: { 1006: 2, 4402: 1 }, + spliceClosesByTriggerDelta: { 'host-oversize-frame': 1, 'queue-limit': 1 } + }) + expect(entries[1]).toMatchObject({ + controlClosesByCodeDelta: {}, + spliceClosesByTriggerDelta: {} + }) + }) + + it('observes successful and failed database calls including transactions', async () => { + const recordSql = vi.fn() + const underlying: RelayDatabase = { + query: vi.fn(async () => [{ ok: true }]), + queryLocked: vi.fn(async (sql, _params, options) => { + throw new Error( + options?.failIfUnavailable && sql === 'SELECT 3' + ? 'database_lock_unavailable' + : 'database unavailable' + ) + }), + transaction: async (operation) => await operation(underlying), + close: vi.fn(async () => {}) + } + const database = observeRelayDatabase(underlying, { + recordAuth: vi.fn(), + recordForwardedBytes: vi.fn(), + recordHttp: vi.fn(), + recordReconnect: vi.fn(), + recordSql + }) + + await database.query('SELECT 1') + await expect(database.transaction(async (tx) => await tx.queryLocked('SELECT 2'))).rejects.toThrow( + 'database unavailable' + ) + await expect( + database.queryLocked('SELECT 3', [], { failIfUnavailable: true }) + ).rejects.toThrow('database_lock_unavailable') + await expect( + database.queryLocked('SELECT 4', [], { failIfUnavailable: true }) + ).rejects.toThrow('database unavailable') + expect(recordSql).toHaveBeenCalledTimes(4) + expect(recordSql.mock.calls.map((call) => call[1])).toEqual([true, false, true, false]) + }) +}) diff --git a/cloud/apps/relay/src/relay-observability.ts b/cloud/apps/relay/src/relay-observability.ts new file mode 100644 index 00000000000..6125ede8d1a --- /dev/null +++ b/cloud/apps/relay/src/relay-observability.ts @@ -0,0 +1,336 @@ +import { monitorEventLoopDelay, performance } from 'node:perf_hooks' +import type { RelayRegion } from '@orca-cloud/relay-contract' +import type { ControlRenewalOutcome } from './assignment-store.js' +import type { PostgresPoolPressureCounts } from './postgres-pool-pressure.js' +import type { RelayReadinessObservation } from './relay-readiness.js' + +export type RelayRuntimeCounts = { + totalConnections: number + inFlightConnections?: number + reservedConnectionUnits?: number + enforcedConnectionUnits?: number + preAuthConnections: number + controls: number + splices: number + pendingSplices: number + queuedBytes: number +} + +export function observedRelayRequests(counts: RelayRuntimeCounts): number { + return counts.preAuthConnections + counts.controls + counts.splices + counts.pendingSplices +} + +export type RelayProcessCounts = RelayRuntimeCounts & PostgresPoolPressureCounts + +export type RegionalRehomeRuntimeSafety = { + observedAt: number + sqlFailures: number + reconnects: number + controlActivityRecoveryFailures: number +} + +export type RegionalRehomeSafetySnapshot = RegionalRehomeRuntimeSafety & { + databasePoolWaiting: number + databasePoolWaitersMax: number + databasePoolWaitMsMax: number +} + +export type AssignmentAdmissionOutcome = + | 'sticky' + | 'sticky-rejected' + | 'placement' + | 'placement-rejected' + +export type AssignmentAdmissionLane = 'sticky' | 'placement' + +export interface RelayRuntimeObserver { + recordAuth(success: boolean): void + recordForwardedBytes(bytes: number): void + recordHttp(durationMs: number): void + recordReconnect(): void + recordSql(durationMs: number, success: boolean): void + recordControlRenewal?(durationMs: number, outcome: ControlRenewalOutcome): void + recordControlActivityRecovery?(success: boolean): void + recordAssignmentAdmission?(outcome: AssignmentAdmissionOutcome): void + recordAssignmentRejectionReason?(lane: AssignmentAdmissionLane, reason: string): void + recordRegionRequest?(region: RelayRegion | undefined): void + recordRegionSelection?(input: { + targetRegion: RelayRegion + selectedRegion?: RelayRegion + fallback: boolean + }): void + recordControlClose?(code: number): void + recordSpliceClose?(trigger: string): void +} + +type RelayMetricDeltas = { + forwardedBytes: number + authSuccesses: number + authFailures: number + reconnects: number + sqlQueries: number + sqlFailures: number + sqlLatencyMsMax: number + httpLatencyMsMax: number + stickyAssignments: number + stickyAssignmentRejections: number + placementAssignments: number + placementAssignmentRejections: number + stickyRejectionsByReason: Record + placementRejectionsByReason: Record + requestedRegions: Record + selectedRegions: Record + regionFallbacks: Record + unavailableRegions: Record + controlClosesByCode: Record + spliceClosesByTrigger: Record + controlRenewalLatenciesMs: number[] + controlRenewalsByOutcome: Record + controlActivityRecoveries: number + controlActivityRecoveryFailures: number +} + +type MetricWriter = (entry: Record) => void + +const emptyDeltas = (): RelayMetricDeltas => ({ + forwardedBytes: 0, + authSuccesses: 0, + authFailures: 0, + reconnects: 0, + sqlQueries: 0, + sqlFailures: 0, + sqlLatencyMsMax: 0, + httpLatencyMsMax: 0, + stickyAssignments: 0, + stickyAssignmentRejections: 0, + placementAssignments: 0, + placementAssignmentRejections: 0, + stickyRejectionsByReason: {}, + placementRejectionsByReason: {}, + requestedRegions: {}, + selectedRegions: {}, + regionFallbacks: {}, + unavailableRegions: {}, + controlClosesByCode: {}, + spliceClosesByTrigger: {}, + controlRenewalLatenciesMs: [], + controlRenewalsByOutcome: {}, + controlActivityRecoveries: 0, + controlActivityRecoveryFailures: 0 +}) + +function percentile(values: number[], percentileRank: number): number { + if (values.length === 0) return 0 + const sorted = [...values].sort((left, right) => left - right) + return sorted[Math.ceil(percentileRank * sorted.length) - 1] ?? 0 +} + +export class RelayObservability implements RelayRuntimeObserver { + private readonly eventLoop = monitorEventLoopDelay({ resolution: 20 }) + private deltas = emptyDeltas() + private timer: ReturnType | null = null + private lastFlushAt = 0 + private lastFlushedSafety = { + sqlFailures: 0, + reconnects: 0, + controlActivityRecoveryFailures: 0 + } + + constructor( + private readonly identity: { role: string; cellId: string; region: RelayRegion }, + private readonly write: MetricWriter = (entry) => console.log(JSON.stringify(entry)) + ) {} + + recordAuth(success: boolean): void { + if (success) this.deltas.authSuccesses++ + else this.deltas.authFailures++ + } + + recordForwardedBytes(bytes: number): void { + this.deltas.forwardedBytes += bytes + } + + recordHttp(durationMs: number): void { + this.deltas.httpLatencyMsMax = Math.max(this.deltas.httpLatencyMsMax, durationMs) + } + + recordReconnect(): void { + this.deltas.reconnects++ + } + + recordAssignmentAdmission(outcome: AssignmentAdmissionOutcome): void { + if (outcome === 'sticky') this.deltas.stickyAssignments++ + else if (outcome === 'sticky-rejected') this.deltas.stickyAssignmentRejections++ + else if (outcome === 'placement') this.deltas.placementAssignments++ + else this.deltas.placementAssignmentRejections++ + } + + recordAssignmentRejectionReason(lane: AssignmentAdmissionLane, reason: string): void { + const counts = + lane === 'sticky' + ? this.deltas.stickyRejectionsByReason + : this.deltas.placementRejectionsByReason + counts[reason] = (counts[reason] ?? 0) + 1 + } + + recordRegionRequest(region: RelayRegion | undefined): void { + increment(this.deltas.requestedRegions, region ?? 'unhinted') + } + + recordRegionSelection(input: { + targetRegion: RelayRegion + selectedRegion?: RelayRegion + fallback: boolean + }): void { + if (input.selectedRegion) increment(this.deltas.selectedRegions, input.selectedRegion) + else increment(this.deltas.unavailableRegions, input.targetRegion) + if (input.fallback) increment(this.deltas.regionFallbacks, input.targetRegion) + } + + recordSql(durationMs: number, success: boolean): void { + this.deltas.sqlQueries++ + if (!success) this.deltas.sqlFailures++ + this.deltas.sqlLatencyMsMax = Math.max(this.deltas.sqlLatencyMsMax, durationMs) + } + + recordControlRenewal(durationMs: number, outcome: ControlRenewalOutcome): void { + this.deltas.controlRenewalLatenciesMs.push(durationMs) + this.deltas.controlRenewalsByOutcome[outcome] = + (this.deltas.controlRenewalsByOutcome[outcome] ?? 0) + 1 + } + + recordControlActivityRecovery(success: boolean): void { + if (success) this.deltas.controlActivityRecoveries++ + else this.deltas.controlActivityRecoveryFailures++ + } + + recordReadiness(observation: RelayReadinessObservation): void { + this.write({ + severity: observation.ready ? 'INFO' : 'WARNING', + message: 'Orca Relay readiness check', + event: 'orca_relay_readiness_check', + metricVersion: 1, + ...this.identity, + ...observation + }) + } + + recordControlClose(code: number): void { + const key = String(code) + this.deltas.controlClosesByCode[key] = (this.deltas.controlClosesByCode[key] ?? 0) + 1 + } + + recordSpliceClose(trigger: string): void { + this.deltas.spliceClosesByTrigger[trigger] = + (this.deltas.spliceClosesByTrigger[trigger] ?? 0) + 1 + } + + start(readCounts: () => RelayProcessCounts, intervalMs = 30_000): void { + if (this.timer) return + this.eventLoop.enable() + this.timer = setInterval(() => this.flush(readCounts()), intervalMs) + this.timer.unref() + } + + stop(): void { + if (this.timer) clearInterval(this.timer) + this.timer = null + this.eventLoop.disable() + } + + regionalRehomeRuntimeSafety(): RegionalRehomeRuntimeSafety { + return { + observedAt: this.lastFlushAt, + sqlFailures: this.lastFlushedSafety.sqlFailures + this.deltas.sqlFailures, + reconnects: this.lastFlushedSafety.reconnects + this.deltas.reconnects, + controlActivityRecoveryFailures: + this.lastFlushedSafety.controlActivityRecoveryFailures + + this.deltas.controlActivityRecoveryFailures + } + } + + flush(counts: RelayProcessCounts): void { + this.lastFlushAt = Date.now() + const deltas = this.deltas + this.lastFlushedSafety = { + sqlFailures: deltas.sqlFailures, + reconnects: deltas.reconnects, + controlActivityRecoveryFailures: deltas.controlActivityRecoveryFailures + } + this.deltas = emptyDeltas() + const memory = process.memoryUsage() + const p99 = this.eventLoop.count === 0 ? 0 : this.eventLoop.percentile(99) / 1_000_000 + this.eventLoop.reset() + this.write({ + severity: 'INFO', + message: 'Orca Relay runtime metrics', + event: 'orca_relay_runtime_metrics', + metricVersion: 2, + role: this.identity.role, + cellId: this.identity.cellId, + region: this.identity.region, + ...counts, + forwardedBytesDelta: deltas.forwardedBytes, + authSuccessesDelta: deltas.authSuccesses, + authFailuresDelta: deltas.authFailures, + reconnectsDelta: deltas.reconnects, + stickyAssignmentsDelta: deltas.stickyAssignments, + stickyAssignmentRejectionsDelta: deltas.stickyAssignmentRejections, + placementAssignmentsDelta: deltas.placementAssignments, + placementAssignmentRejectionsDelta: deltas.placementAssignmentRejections, + stickyRejectionsByReasonDelta: deltas.stickyRejectionsByReason, + placementRejectionsByReasonDelta: deltas.placementRejectionsByReason, + requestedRegionsDelta: deltas.requestedRegions, + selectedRegionsDelta: deltas.selectedRegions, + regionFallbacksDelta: deltas.regionFallbacks, + unavailableRegionsDelta: deltas.unavailableRegions, + controlClosesByCodeDelta: deltas.controlClosesByCode, + spliceClosesByTriggerDelta: deltas.spliceClosesByTrigger, + sqlQueriesDelta: deltas.sqlQueries, + sqlFailuresDelta: deltas.sqlFailures, + sqlLatencyMsMax: Number(deltas.sqlLatencyMsMax.toFixed(3)), + controlRenewalsByOutcomeDelta: deltas.controlRenewalsByOutcome, + controlRenewalsDelta: deltas.controlRenewalLatenciesMs.length, + controlRenewalSuccessesDelta: deltas.controlRenewalsByOutcome.renewed ?? 0, + controlRenewalLeaseMissesDelta: + deltas.controlRenewalsByOutcome.control_activity_not_found ?? 0, + controlActivityRecoveriesDelta: deltas.controlActivityRecoveries, + controlActivityRecoveryFailuresDelta: deltas.controlActivityRecoveryFailures, + controlRenewalLatencyMsP50: Number( + percentile(deltas.controlRenewalLatenciesMs, 0.5).toFixed(3) + ), + controlRenewalLatencyMsP95: Number( + percentile(deltas.controlRenewalLatenciesMs, 0.95).toFixed(3) + ), + controlRenewalLatencyMsMax: Number( + Math.max(0, ...deltas.controlRenewalLatenciesMs).toFixed(3) + ), + httpLatencyMsMax: Number(deltas.httpLatencyMsMax.toFixed(3)), + heapUsedBytes: memory.heapUsed, + heapTotalBytes: memory.heapTotal, + eventLoopDelayMsP99: Number(p99.toFixed(3)) + }) + } +} + +function increment(counts: Record, key: string): void { + counts[key] = (counts[key] ?? 0) + 1 +} + +export function timedRelayOperation( + operation: () => Promise, + observe: (durationMs: number, success: boolean) => void, + isExpectedError: (error: unknown) => boolean = () => false +): Promise { + const startedAt = performance.now() + return operation().then( + (result) => { + observe(performance.now() - startedAt, true) + return result + }, + (error: unknown) => { + observe(performance.now() - startedAt, isExpectedError(error)) + throw error + } + ) +} diff --git a/cloud/apps/relay/src/relay-readiness.test.ts b/cloud/apps/relay/src/relay-readiness.test.ts new file mode 100644 index 00000000000..a85d9a46aa7 --- /dev/null +++ b/cloud/apps/relay/src/relay-readiness.test.ts @@ -0,0 +1,109 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RelayDatabase } from './database.js' +import { + createRelayReadiness, + type RelayReadinessObservation +} from './relay-readiness.js' + +function database(query: () => Promise[]>): RelayDatabase { + return { + query, + queryLocked: query, + transaction: async (operation) => await operation(database(query)), + close: async () => {} + } +} + +describe('relay readiness', () => { + it('fails readiness while liveness remains independent of SQL and JWKS', async () => { + const jwksFailure = createRelayReadiness(database(async () => [{ ready: 1 }]), 'https://jwks', { + fetch: vi.fn(async () => new Response('', { status: 503 })) as typeof fetch, + cacheMs: 0 + }) + const sqlFailure = createRelayReadiness( + database(async () => { + throw new Error('sql down') + }), + 'https://jwks', + { + fetch: vi.fn(async () => new Response('{}', { status: 200 })) as typeof fetch, + cacheMs: 0 + } + ) + + expect(await jwksFailure()).toBe(false) + expect(await sqlFailure()).toBe(false) + }) + + it.each([ + { + name: 'JWKS HTTP failure', + fetch: vi.fn(async () => new Response('', { status: 503 })) as typeof fetch, + query: vi.fn(async () => [{ ready: 1 }]), + failure: 'jwks_http_failed' + }, + { + name: 'JWKS timeout', + fetch: vi.fn(async () => { + throw new DOMException('redacted', 'TimeoutError') + }) as typeof fetch, + query: vi.fn(async () => [{ ready: 1 }]), + failure: 'jwks_timed_out' + }, + { + name: 'JWKS fetch failure', + fetch: vi.fn(async () => { + throw new Error('redacted') + }) as typeof fetch, + query: vi.fn(async () => [{ ready: 1 }]), + failure: 'jwks_fetch_failed' + }, + { + name: 'SQL failure', + fetch: vi.fn(async () => new Response('{}', { status: 200 })) as typeof fetch, + query: vi.fn(async () => { + throw new Error('redacted') + }), + failure: 'sql_failed' + } + ])('reports a safe reason for $name', async ({ fetch, query, failure }) => { + const observations: RelayReadinessObservation[] = [] + const ready = createRelayReadiness(database(query), 'https://jwks', { + fetch, + cacheMs: 0, + observe: (observation) => observations.push(observation) + }) + + expect(await ready()).toBe(false) + expect(observations).toEqual([ + expect.objectContaining({ ready: false, failure }) + ]) + expect(JSON.stringify(observations)).not.toContain('redacted') + if (failure.startsWith('jwks_')) expect(query).not.toHaveBeenCalled() + }) + + it('reports the initial success but not healthy repeats or cached reads', async () => { + const observations: RelayReadinessObservation[] = [] + let now = 100 + const ready = createRelayReadiness(database(async () => [{ ready: 1 }]), 'https://jwks', { + fetch: vi.fn(async () => new Response('{}', { status: 200 })) as typeof fetch, + cacheMs: 10_000, + now: () => now, + observe: (observation) => observations.push(observation) + }) + + expect(await ready()).toBe(true) + now += 1_000 + expect(await ready()).toBe(true) + now += 10_000 + expect(await ready()).toBe(true) + expect(observations).toEqual([ + { + ready: true, + jwksLatencyMs: 0, + sqlLatencyMs: 0, + totalLatencyMs: 0 + } + ]) + }) +}) diff --git a/cloud/apps/relay/src/relay-readiness.ts b/cloud/apps/relay/src/relay-readiness.ts new file mode 100644 index 00000000000..0b7303367f1 --- /dev/null +++ b/cloud/apps/relay/src/relay-readiness.ts @@ -0,0 +1,89 @@ +import type { RelayDatabase } from './database.js' + +export type RelayReadinessFailure = + | 'jwks_fetch_failed' + | 'jwks_http_failed' + | 'jwks_timed_out' + | 'sql_failed' + +export type RelayReadinessObservation = { + ready: boolean + failure?: RelayReadinessFailure + jwksLatencyMs: number + sqlLatencyMs: number + totalLatencyMs: number +} + +type RelayReadinessOptions = { + fetch?: typeof fetch + timeoutMs?: number + cacheMs?: number + now?: () => number + observe?: (observation: RelayReadinessObservation) => void +} + +function fetchFailure(error: unknown): RelayReadinessFailure { + return error instanceof Error && error.name === 'TimeoutError' + ? 'jwks_timed_out' + : 'jwks_fetch_failed' +} + +export function createRelayReadiness( + database: RelayDatabase, + jwksUrl: string, + options: RelayReadinessOptions = {} +): () => Promise { + const fetchImpl = options.fetch ?? fetch + const timeoutMs = options.timeoutMs ?? 2_000 + const cacheMs = options.cacheMs ?? 10_000 + const now = options.now ?? Date.now + let cachedAt = Number.NEGATIVE_INFINITY + let cached = false + let lastObservedReady: boolean | undefined + + return async () => { + if (now() - cachedAt < cacheMs) return cached + const startedAt = now() + let jwksCompletedAt = startedAt + let sqlStartedAt = startedAt + let failure: RelayReadinessFailure | undefined + try { + let response: Response + try { + response = await fetchImpl(jwksUrl, { signal: AbortSignal.timeout(timeoutMs) }) + } catch (error) { + failure = fetchFailure(error) + throw error + } finally { + jwksCompletedAt = now() + } + if (!response.ok) { + failure = 'jwks_http_failed' + throw new Error(failure) + } + sqlStartedAt = now() + try { + await database.query('SELECT 1 AS ready') + } catch (error) { + failure = 'sql_failed' + throw error + } + } catch { + // The load balancer only needs the boolean; the safe reason is emitted below. + } + const completedAt = now() + cached = failure === undefined + cachedAt = completedAt + if (!cached || cached !== lastObservedReady) { + options.observe?.({ + ready: cached, + ...(failure ? { failure } : {}), + jwksLatencyMs: Math.max(0, jwksCompletedAt - startedAt), + sqlLatencyMs: failure?.startsWith('jwks_') ? 0 : Math.max(0, completedAt - sqlStartedAt), + totalLatencyMs: Math.max(0, completedAt - startedAt) + }) + } + lastObservedReady = cached + return cached + } +} diff --git a/cloud/apps/relay/src/relay-region-app.test.ts b/cloud/apps/relay/src/relay-region-app.test.ts new file mode 100644 index 00000000000..30cf3bf3e26 --- /dev/null +++ b/cloud/apps/relay/src/relay-region-app.test.ts @@ -0,0 +1,155 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RelayConfig } from './config.js' + +const fakes = vi.hoisted(() => ({ + verifyRelayToken: vi.fn(async (token: string) => ({ sub: 'user-1', relayHostId: token })) +})) + +vi.mock('./relay-token-verifier.js', () => ({ + createRelayTokenVerifier: () => fakes.verifyRelayToken, + readBearer: (value: string | undefined) => value?.replace(/^Bearer /, '') ?? null +})) + +import { createRelayApp } from './app.js' + +describe('Relay region API', () => { + it('passes a valid preference to placement and records coarse outcomes', async () => { + const assign = vi.fn(async () => ({ + userId: 'user-1', + relayHostId: 'asiahost00000001', + cellId: 'asia-c1', + cellUrl: 'https://asia-c1.relay.example.test', + region: 'asia-east2' as const, + assignmentEpoch: 1, + leaseExpiresAt: Date.now() + 300_000 + })) + const recordRegionRequest = vi.fn() + const recordRegionSelection = vi.fn() + const app = createRelayApp(config(), { + store: {} as never, + assignments: { assign, resolve: vi.fn(async () => null) } as never, + drain: vi.fn(), + ready: vi.fn(async () => true), + recordRegionRequest, + recordRegionSelection + }) + + const response = await app.request( + '/v1/assign', + assignmentRequest('asiahost00000001', { preferredRegion: 'asia-east2' }) + ) + + expect(response.status).toBe(200) + expect(assign).toHaveBeenCalledWith( + { userId: 'user-1', relayHostId: 'asiahost00000001' }, + 'asia-east2', + 'asia-east2' + ) + expect(recordRegionRequest).toHaveBeenCalledWith('asia-east2') + expect(recordRegionSelection).toHaveBeenCalledWith({ + targetRegion: 'asia-east2', + selectedRegion: 'asia-east2', + fallback: false + }) + }) + + it('keeps the preference for observation while the kill switch places US-first', async () => { + const assign = vi.fn(async () => ({ + userId: 'user-1', + relayHostId: 'killhost00000001', + cellId: 'us-c1', + cellUrl: 'https://us-c1.relay.example.test', + region: 'us-central1' as const, + assignmentEpoch: 1, + leaseExpiresAt: Date.now() + 300_000 + })) + const app = createRelayApp(config({ regionalPlacementEnabled: false }), { + store: {} as never, + assignments: { assign, resolve: vi.fn(async () => null) } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const response = await app.request( + '/v1/assign', + assignmentRequest('killhost00000001', { preferredRegion: 'asia-east2' }) + ) + + expect(response.status).toBe(200) + expect(assign).toHaveBeenCalledWith( + { userId: 'user-1', relayHostId: 'killhost00000001' }, + 'asia-east2', + 'us-central1' + ) + }) + + it('exposes only the store-provided healthy catalog from directors', async () => { + const regionCatalog = vi.fn(async () => [ + { region: 'us-central1' as const, probeOrigins: ['https://us.relay.example.test'] } + ]) + const app = createRelayApp(config(), { + store: {} as never, + assignments: { regionCatalog } as never, + drain: vi.fn(), + ready: vi.fn(async () => true) + }) + + const response = await app.request('/v1/regions') + + expect(response.status).toBe(200) + expect(await response.json()).toEqual({ + v: 1, + regions: [ + { region: 'us-central1', probeOrigins: ['https://us.relay.example.test'] } + ] + }) + + const burst = await Promise.all( + Array.from({ length: 50 }, () => app.request('/v1/regions')) + ) + expect(burst.every(({ status }) => status === 200)).toBe(true) + expect(regionCatalog).toHaveBeenCalledOnce() + }) +}) + +function assignmentRequest(relayHostId: string, extra: Record): RequestInit { + return { + method: 'POST', + headers: { + authorization: `Bearer ${relayHostId}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, relayHostId, ...extra }) + } +} + +function config(overrides: Partial = {}): RelayConfig { + return { + port: 8080, + publicUrl: 'https://relay.example.test', + cellUrl: 'https://relay.example.test', + region: 'us-central1', + authIssuer: 'https://auth.example.test', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.test/jwks', + assignmentSigningKey: new TextEncoder().encode('assignment-key-with-at-least-32-bytes'), + role: 'director', + cellId: 'director', + cells: [], + adminAudience: 'https://relay.example.test/v1/admin/drain', + deployServiceAccount: 'deploy@example.test', + runtimeServiceAccount: 'runtime@example.test', + adminJwksUrl: 'https://auth.example.test/jwks', + databasePoolMax: 3, + publicAssignmentsEnabled: true, + regionalPlacementEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './data', + ...overrides + } +} diff --git a/cloud/apps/relay/src/relay-region-placement.test.ts b/cloud/apps/relay/src/relay-region-placement.test.ts new file mode 100644 index 00000000000..76cf5479a41 --- /dev/null +++ b/cloud/apps/relay/src/relay-region-placement.test.ts @@ -0,0 +1,196 @@ +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayCellConfig } from './config.js' +import { openInMemoryRelayDatabase, type RelayDatabase } from './database.js' + +const CELLS: RelayCellConfig[] = [ + { + id: 'us-c1', + url: 'https://us-c1.relay.example.com', + region: 'us-central1', + capacityRequests: 100 + }, + { + id: 'asia-c1', + url: 'https://asia-c1.relay.example.com', + region: 'asia-east2', + capacityRequests: 100 + }, + { + id: 'asia-c2', + url: 'https://asia-c2.relay.example.com', + region: 'asia-east2', + capacityRequests: 100 + }, + { + id: 'asia-c3', + url: 'https://asia-c3.relay.example.com', + region: 'asia-east2', + capacityRequests: 100 + } +] + +describe('Relay regional placement', () => { + let database: RelayDatabase + let now: number + let store: RelayAssignmentStore + + beforeEach(async () => { + database = await openInMemoryRelayDatabase() + now = 1_000 + store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells(CELLS) + }) + + afterEach(async () => await database.close()) + + it('prefers the requested region and keeps unhinted placement US-first', async () => { + await expect( + store.assign({ userId: 'asia-user', relayHostId: 'asiahost00000001' }, 'asia-east2') + ).resolves.toMatchObject({ cellId: 'asia-c1', region: 'asia-east2' }) + await expect( + store.assign({ userId: 'us-user', relayHostId: 'ushost0000000001' }) + ).resolves.toMatchObject({ cellId: 'us-c1', region: 'us-central1' }) + }) + + it('falls back globally only when the target region has no safe general cell', async () => { + await store.configureCell(CELLS[1]!, 'migration-only') + await store.configureCell(CELLS[2]!, 'migration-only') + await store.configureCell(CELLS[3]!, 'migration-only') + + await expect( + store.assign({ userId: 'fallback-user', relayHostId: 'fallbackhost0001' }, 'asia-east2') + ).resolves.toMatchObject({ cellId: 'us-c1', region: 'us-central1' }) + }) + + it('preserves sticky assignments while updating only explicit preferences', async () => { + const identity = { userId: 'sticky-user', relayHostId: 'stickyhost000001' } + await expect(store.assign(identity)).resolves.toMatchObject({ cellId: 'us-c1' }) + now = 2_000 + await expect(store.assign(identity, 'asia-east2')).resolves.toMatchObject({ cellId: 'us-c1' }) + now = 3_000 + await store.assign(identity) + + expect( + await database.query( + `SELECT preferred_region, observed_at FROM relay_assignment_region_preferences + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).toEqual([{ preferred_region: 'asia-east2', observed_at: 2_000 }]) + }) + + it('uses the explicit placement region as the server-side kill switch', async () => { + await expect( + store.assign( + { userId: 'kill-user', relayHostId: 'killhost00000001' }, + 'asia-east2', + 'us-central1' + ) + ).resolves.toMatchObject({ cellId: 'us-c1', region: 'us-central1' }) + expect(await database.query(`SELECT preferred_region FROM relay_assignment_region_preferences`)) + .toEqual([{ preferred_region: 'asia-east2' }]) + }) + + it('returns at most two deterministic healthy general probe origins per region', async () => { + for (const [index, cell] of CELLS.entries()) { + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + region: cell.region, + cellIncarnation: `00000000-0000-4000-8000-${String(index + 1).padStart(12, '0')}`, + startedAt: 1, + ready: true, + observedRequests: 0 + }) + } + await store.configureCell(CELLS[2]!, 'migration-only') + + await expect(store.regionCatalog()).resolves.toEqual([ + { region: 'us-central1', probeOrigins: ['https://us-c1.relay.example.com'] }, + { + region: 'asia-east2', + probeOrigins: [ + 'https://asia-c1.relay.example.com', + 'https://asia-c3.relay.example.com' + ] + } + ]) + + await database.query(`UPDATE relay_cells SET enabled = 0 WHERE cell_id = ?`, ['asia-c3']) + await expect(store.regionCatalog()).resolves.toEqual([ + { region: 'us-central1', probeOrigins: ['https://us-c1.relay.example.com'] }, + { region: 'asia-east2', probeOrigins: ['https://asia-c1.relay.example.com'] } + ]) + }) + + it('rejects a heartbeat whose explicit region conflicts with registration', async () => { + await expect( + store.recordCellHeartbeat({ + cellId: 'asia-c1', + cellUrl: 'https://asia-c1.relay.example.com', + region: 'us-central1', + cellIncarnation: '00000000-0000-4000-8000-000000000001', + startedAt: 1, + ready: true, + observedRequests: 0 + }) + ).rejects.toThrow('cell_region_mismatch') + }) + + it('defaults cells inserted by an old process after startup to US', async () => { + const oldCell = { + id: 'old-us-cell', + url: 'https://old-us-cell.relay.example.com', + capacityRequests: 100 + } + await database.query( + `INSERT INTO relay_cells + (cell_id, cell_url, enabled, capacity_requests, reserved_requests, + observed_requests, last_heartbeat_at, updated_at) + VALUES (?, ?, 1, 100, 0, 0, ?, ?)`, + [oldCell.id, oldCell.url, now, now] + ) + await database.query( + `INSERT INTO relay_cell_admission (cell_id, admission_state, updated_at) + VALUES (?, 'general', ?)`, + [oldCell.id, now] + ) + await Promise.all(CELLS.map((cell) => store.configureCell(cell, 'migration-only'))) + + const identity = { userId: 'old-user', relayHostId: 'oldhost000000001' } + await expect(store.assign(identity)).resolves.toMatchObject({ + cellId: oldCell.id, + region: 'us-central1' + }) + await expect(store.resolve(identity)).resolves.toMatchObject({ + cellId: oldCell.id, + region: 'us-central1' + }) + await expect( + store.recordCellHeartbeat({ + cellId: oldCell.id, + cellUrl: oldCell.url, + cellIncarnation: '00000000-0000-4000-8000-000000000099', + startedAt: 1, + ready: true, + observedRequests: 0 + }) + ).resolves.toBeUndefined() + }) + + it('expires identity-linked region preferences after 30 days', async () => { + const identity = { userId: 'expiry-user', relayHostId: 'expiryhost000001' } + await store.assign(identity, 'asia-east2') + now += 30 * 24 * 60 * 60_000 + 1 + + await expect(store.releaseExpiredRegionPreferences()).resolves.toBe(1) + await expect( + database.query( + `SELECT preferred_region FROM relay_assignment_region_preferences + WHERE user_id = ? AND relay_host_id = ?`, + [identity.userId, identity.relayHostId] + ) + ).resolves.toEqual([]) + }) +}) diff --git a/cloud/apps/relay/src/relay-server.ts b/cloud/apps/relay/src/relay-server.ts new file mode 100644 index 00000000000..a14240cfa6a --- /dev/null +++ b/cloud/apps/relay/src/relay-server.ts @@ -0,0 +1,533 @@ +import { createAdaptorServer } from '@hono/node-server' +import { + hasAdmissionCapacity, + HostDataAuthSchema, + RELAY_ADMISSION_BUDGETS, + RELAY_CLOSE_CODE, + RELAY_DEFAULT_REGION, + RELAY_PROTOCOL_LIMITS, + RelayAuthSchema +} from '@orca-cloud/relay-contract' +import type { IncomingMessage } from 'node:http' +import { randomUUID } from 'node:crypto' +import { performance } from 'node:perf_hooks' +import { WebSocketServer } from 'ws' +import type WebSocket from 'ws' +import type { RawData } from 'ws' +import { createRelayApp } from './app.js' +import { RelayAssignmentStore } from './assignment-store.js' +import type { RelayConfig } from './config.js' +import { RelayCredentialStore } from './credential-store.js' +import type { RelayDatabase } from './database.js' +import { HostSessionRegistry } from './host-session-registry.js' +import { observeRelayDatabase } from './observed-relay-database.js' +import { RelayObservability } from './relay-observability.js' +import { + RelayConnectionLedger, + type RelayConnectionUpgrade +} from './relay-connection-ledger.js' +import { createRelayReadiness } from './relay-readiness.js' +import { createRelayTokenVerifier, readBearer } from './relay-token-verifier.js' +import { closeRelayWebSocket } from './relay-websocket-close.js' +import { ProcessQueuedByteBudget } from './splice-forwarder.js' + +function rejectUpgrade(socket: NodeJS.WritableStream, status: number, message: string): void { + socket.write(`HTTP/1.1 ${status} ${message}\r\nConnection: close\r\nContent-Length: 0\r\n\r\n`) + if ('destroy' in socket && typeof socket.destroy === 'function') socket.destroy() +} + +function noDelay(socket: WebSocket): void { + const transport = (socket as WebSocket & { _socket?: { setNoDelay: (enabled: boolean) => void } }) + ._socket + transport?.setNoDelay(true) +} + +// Why: a ws receiver error (oversize or malformed frame) with no 'error' +// listener throws process-wide; ws itself already closes the socket after +// emitting it, so logging is all that is left to do. +function guardSocketErrors(socket: WebSocket, kind: string): void { + socket.on('error', (error) => { + console.warn(`[orca-relay] ${kind} socket error: ${error.message}`) + }) +} + +function admissionSource(request: IncomingMessage): string { + const forwarded = request.headers['x-forwarded-for'] + const chain = (Array.isArray(forwarded) ? forwarded.join(',') : forwarded ?? '') + .split(',') + .map((entry) => entry.trim()) + .filter(Boolean) + // Google Front End appends client and load-balancer addresses after any + // caller-supplied values, so only the penultimate hop is trustworthy. + return chain.length >= 2 ? chain.at(-2)! : (request.socket.remoteAddress ?? 'unknown') +} + +function firstPayload(raw: RawData, expectedType: string): unknown { + try { + const parsed = JSON.parse(raw.toString()) as Record + if (parsed.type !== expectedType) return null + const { type: _type, ...rest } = parsed + return rest + } catch { + return null + } +} + +export function createRelayServer( + config: RelayConfig, + database: RelayDatabase, + options: { + now?: () => number + connectionLedgerLimits?: { hardCap: number; controlReserve: number } + cellIncarnation?: string + } = {} +) { + const cellIncarnation = options.cellIncarnation ?? randomUUID() + const observability = new RelayObservability({ + role: config.role, + cellId: config.cellId, + region: config.region ?? RELAY_DEFAULT_REGION + }) + const observedDatabase = observeRelayDatabase(database, observability) + const controls = new WebSocketServer({ + noServer: true, + clientTracking: false, + perMessageDeflate: false, + maxPayload: 1024 * 1024 + }) + const verifyRelayToken = createRelayTokenVerifier(config) + const store = new RelayCredentialStore(observedDatabase, options.now) + const assignments = new RelayAssignmentStore(observedDatabase, options.now, { + requireLiveCells: config.role === 'director', + recordControlRenewal: (durationMs, outcome) => + observability.recordControlRenewal?.(durationMs, outcome) + }) + const ready = createRelayReadiness(observedDatabase, config.jwksUrl, { + observe: (observation) => observability.recordReadiness(observation) + }) + const queuedBytes = new ProcessQueuedByteBudget() + const sessions = new HostSessionRegistry( + config, + verifyRelayToken, + store, + assignments, + queuedBytes, + observability, + options.now + ) + const app = createRelayApp(config, { + store, + assignments, + drain: (graceMs) => sessions.drain(graceMs), + drainHost: (input) => sessions.drainHost(input), + regionalRehomeTrustProbeHostExists: (input) => sessions.get(input) !== null, + cellIncarnation, + isDraining: () => sessions.isDraining(), + runtimeCounts: () => runtimeCounts(), + ready, + recordAssignmentAdmission: (outcome) => observability.recordAssignmentAdmission?.(outcome), + recordAssignmentRejectionReason: (lane, reason) => + observability.recordAssignmentRejectionReason?.(lane, reason), + recordRegionRequest: (region) => observability.recordRegionRequest?.(region), + recordRegionSelection: (input) => observability.recordRegionSelection?.(input) + }) + const observedFetch: typeof app.fetch = async (...args) => { + const startedAt = performance.now() + try { + return await app.fetch(...args) + } finally { + observability.recordHttp(performance.now() - startedAt) + } + } + const server = createAdaptorServer({ fetch: observedFetch }) + const clients = new WebSocketServer({ + noServer: true, + clientTracking: false, + perMessageDeflate: false, + maxPayload: RELAY_PROTOCOL_LIMITS.maxFrameBytes + }) + const dataSockets = new WebSocketServer({ + noServer: true, + clientTracking: false, + perMessageDeflate: false, + maxPayload: RELAY_PROTOCOL_LIMITS.maxFrameBytes + }) + let preAuthConnections = 0 + let totalConnections = 0 + const configuredConnectionLimits = + config.connectionHardCap === undefined + ? null + : (options.connectionLedgerLimits ?? { + hardCap: config.connectionHardCap, + controlReserve: RELAY_ADMISSION_BUDGETS.reservedHostControls + }) + const connectionLedger = + configuredConnectionLimits === null + ? null + : new RelayConnectionLedger( + configuredConnectionLimits.hardCap, + configuredConnectionLimits.controlReserve + ) + const preAuthBySource = new Map() + const preAuthAttemptsBySource = new Map() + + const admit = (source: string): boolean => { + const now = Date.now() + const currentWindow = preAuthAttemptsBySource.get(source) + const attempts = + !currentWindow || now - currentWindow.windowStartedAt >= 60_000 + ? { windowStartedAt: now, count: 0 } + : currentWindow + const sourceCount = preAuthBySource.get(source) ?? 0 + if ( + !hasAdmissionCapacity({ + totalRequests: + connectionLedger?.counts().physicalConnections ?? totalConnections, + preAuthConnections, + sourcePreAuthConnections: sourceCount, + totalRequestCeiling: + configuredConnectionLimits === null + ? undefined + : configuredConnectionLimits.hardCap - configuredConnectionLimits.controlReserve + }) || + attempts.count >= RELAY_ADMISSION_BUDGETS.maxPreAuthAttemptsPerSourcePerMinute + ) { + return false + } + attempts.count++ + preAuthAttemptsBySource.delete(source) + preAuthAttemptsBySource.set(source, attempts) + // A bounded LRU keeps source churn from becoming its own memory attack. + if (preAuthAttemptsBySource.size > 4_096) { + preAuthAttemptsBySource.delete(preAuthAttemptsBySource.keys().next().value!) + } + preAuthConnections++ + preAuthBySource.set(source, sourceCount + 1) + return true + } + const authenticated = (source: string): void => { + preAuthConnections = Math.max(0, preAuthConnections - 1) + const next = (preAuthBySource.get(source) ?? 1) - 1 + if (next <= 0) preAuthBySource.delete(source) + else preAuthBySource.set(source, next) + } + + const trackConnection = ( + socket: WebSocket, + upgrade: RelayConnectionUpgrade | null + ): void => { + if (upgrade) { + upgrade.promote(socket) + return + } + totalConnections++ + socket.once('close', () => { + totalConnections = Math.max(0, totalConnections - 1) + }) + } + + const awaitFirstFrame = ( + socket: WebSocket, + source: string, + callback: (raw: RawData) => Promise + ): void => { + let finished = false + const timer = setTimeout(() => { + if (finished) return + finished = true + authenticated(source) + observability.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'first frame timeout') + }, RELAY_PROTOCOL_LIMITS.firstFrameDeadlineMs) + socket.once('message', (raw, binary) => { + if (finished) return + finished = true + clearTimeout(timer) + authenticated(source) + if (binary) { + observability.recordAuth(false) + socket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'first frame must be text') + return + } + void callback(raw).catch((error: unknown) => { + console.warn( + '[orca-relay] first frame handler failed', + error instanceof Error ? error.message : '' + ) + closeRelayWebSocket( + socket, + RELAY_CLOSE_CODE.LIMIT_EXCEEDED, + 'relay temporarily unavailable' + ) + }) + }) + socket.once('close', () => { + if (!finished) { + finished = true + clearTimeout(timer) + authenticated(source) + } + }) + } + + server.on('upgrade', (request, socket, head) => { + const url = new URL(request.url ?? '/', config.publicUrl) + const source = admissionSource(request) + if (url.search) { + rejectUpgrade(socket, 400, 'Bad Request') + return + } + if (url.pathname.startsWith('/v1/connect/')) { + const hostId = decodeURIComponent(url.pathname.slice('/v1/connect/'.length)) + if (!/^[A-Za-z0-9_-]{16}$/.test(hostId)) { + rejectUpgrade(socket, 429, 'Too Many Requests') + return + } + const phoneAdmission = connectionLedger?.tryReservePhone() ?? null + if (connectionLedger && !phoneAdmission) { + rejectUpgrade(socket, 503, 'Service Unavailable') + return + } + if (!admit(source)) { + phoneAdmission?.upgrade.release() + phoneAdmission?.hostData.release() + rejectUpgrade(socket, 429, 'Too Many Requests') + return + } + const releasePhoneUpgrade = (): void => { + authenticated(source) + phoneAdmission?.upgrade.release() + phoneAdmission?.hostData.release() + } + socket.once('close', releasePhoneUpgrade) + try { + clients.handleUpgrade(request, socket, head, (webSocket) => { + socket.off('close', releasePhoneUpgrade) + trackConnection(webSocket, phoneAdmission?.upgrade ?? null) + guardSocketErrors(webSocket, 'client') + if (phoneAdmission) { + webSocket.once('close', () => phoneAdmission.hostData.release()) + } + noDelay(webSocket) + awaitFirstFrame(webSocket, source, async (raw) => { + const auth = RelayAuthSchema.safeParse(firstPayload(raw, 'relay-auth')) + if (!auth.success) { + phoneAdmission?.hostData.release() + observability.recordAuth(false) + webSocket.send( + JSON.stringify({ type: 'relay-hello', ok: false, code: RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL }) + ) + webSocket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid relay auth') + return + } + if (config.role === 'director') { + const invite = await store.resolveInviteForMove(hostId, auth.data.credential) + const identity = invite ? { userId: invite.userId, relayHostId: hostId } : null + // Released combined-service invites gain their first durable cell assignment here. + const assignment = identity + ? (await assignments.resolve(identity)) ?? (await assignments.assign(identity)) + : null + if (!invite || !assignment) { + phoneAdmission?.hostData.release() + observability.recordAuth(false) + webSocket.send( + JSON.stringify({ + type: 'relay-hello', + ok: false, + code: RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL + }) + ) + webSocket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid invite') + return + } + phoneAdmission?.hostData.release() + observability.recordAuth(true) + webSocket.send( + JSON.stringify({ + type: 'relay-moved', + v: 1, + cellUrl: assignment.cellUrl, + assignmentEpoch: assignment.assignmentEpoch + }) + ) + webSocket.close(RELAY_CLOSE_CODE.DRAINING, 'connect to assigned cell') + return + } + await sessions.acceptClient( + webSocket, + hostId, + auth.data.credential, + phoneAdmission?.hostData + ) + }) + }) + } catch { + socket.off('close', releasePhoneUpgrade) + releasePhoneUpgrade() + socket.destroy() + } + return + } + if (url.pathname.startsWith('/v1/host/data/')) { + if (config.role === 'director') { + rejectUpgrade(socket, 404, 'Not Found') + return + } + const connId = decodeURIComponent(url.pathname.slice('/v1/host/data/'.length)) + if (!connId || connId.length > 128) { + rejectUpgrade(socket, 429, 'Too Many Requests') + return + } + const dataUpgrade = connectionLedger?.tryReserveHostData(connId) ?? null + if (connectionLedger && !dataUpgrade) { + rejectUpgrade(socket, 503, 'Service Unavailable') + return + } + if (!admit(source)) { + dataUpgrade?.release() + rejectUpgrade(socket, 429, 'Too Many Requests') + return + } + const releaseDataUpgrade = (): void => { + authenticated(source) + dataUpgrade?.release() + } + socket.once('close', releaseDataUpgrade) + try { + dataSockets.handleUpgrade(request, socket, head, (webSocket) => { + socket.off('close', releaseDataUpgrade) + trackConnection(webSocket, dataUpgrade) + guardSocketErrors(webSocket, 'host-data') + noDelay(webSocket) + awaitFirstFrame(webSocket, source, async (raw) => { + const auth = HostDataAuthSchema.safeParse(firstPayload(raw, 'host-data-auth')) + if (!auth.success) { + observability.recordAuth(false) + webSocket.close(RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL, 'invalid host data auth') + return + } + const accepted = await sessions.acceptHostData( + webSocket, + connId, + auth.data.connTicket, + auth.data.generation + ) + if (accepted) dataUpgrade?.commitHostData() + }) + }) + } catch { + socket.off('close', releaseDataUpgrade) + releaseDataUpgrade() + socket.destroy() + } + return + } + if (url.pathname !== '/v1/host/control') { + rejectUpgrade(socket, 404, 'Not Found') + return + } + if (config.role === 'director') { + rejectUpgrade(socket, 404, 'Not Found') + return + } + const bearer = readBearer(request.headers.authorization) + if (!bearer) { + observability.recordAuth(false) + rejectUpgrade(socket, 401, 'Unauthorized') + return + } + void verifyRelayToken(bearer).then((identity) => { + if (socket.destroyed) return + if (!identity) { + observability.recordAuth(false) + rejectUpgrade(socket, 401, 'Unauthorized') + return + } + const isRebind = sessions.hasActiveControl({ + userId: identity.sub, + relayHostId: identity.relayHostId + }) + const controlUpgrade = connectionLedger?.tryReserveControl(isRebind) ?? null + if ( + (connectionLedger && !controlUpgrade) || + (!connectionLedger && totalConnections >= RELAY_ADMISSION_BUDGETS.cloudRunConcurrency) + ) { + rejectUpgrade( + socket, + connectionLedger ? 503 : 429, + connectionLedger ? 'Service Unavailable' : 'Too Many Requests' + ) + return + } + // Register the release before anything else can throw: the outer catch + // destroys the socket, so a close-registered release cannot leak the + // reserved connection unit. + const releaseControlUpgrade = (): void => controlUpgrade?.release() + socket.once('close', releaseControlUpgrade) + observability.recordAuth(true) + try { + controls.handleUpgrade(request, socket, head, (webSocket) => { + socket.off('close', releaseControlUpgrade) + trackConnection(webSocket, controlUpgrade) + guardSocketErrors(webSocket, 'control') + noDelay(webSocket) + sessions.acceptControl( + webSocket, + identity, + controlUpgrade?.inclusionWatermark + ) + }) + } catch { + socket.off('close', releaseControlUpgrade) + releaseControlUpgrade() + socket.destroy() + } + }).catch((error: unknown) => { + // A throw in the upgrade handling above must cost this socket, not the process. + console.warn( + `[orca-relay] control upgrade failed: ${error instanceof Error ? error.message : 'unknown'}` + ) + socket.destroy() + }) + }) + + server.on('close', () => { + controls.close() + clients.close() + dataSockets.close() + }) + const runtimeCounts = () => { + const ledgerCounts = connectionLedger?.counts() + return { + totalConnections: ledgerCounts?.physicalConnections ?? totalConnections, + preAuthConnections, + ...sessions.runtimeCounts(), + queuedBytes: queuedBytes.current(), + ...(ledgerCounts + ? { + inFlightConnections: ledgerCounts.inFlightConnections, + reservedConnectionUnits: ledgerCounts.reservedConnectionUnits, + enforcedConnectionUnits: ledgerCounts.enforcedConnectionUnits + } + : {}) + } + } + const connectionSnapshot = () => connectionLedger?.snapshot() + return { + server, + sessions, + store, + assignments, + queuedBytes, + observability, + runtimeCounts, + connectionSnapshot, + ready, + cellIncarnation + } +} + +export function closeWithDrain(socket: WebSocket, graceMs: number): void { + socket.send(JSON.stringify({ type: 'drain', graceMs, recovery: 'resolve-director' })) + socket.close(RELAY_CLOSE_CODE.DRAINING, 'resolve configured director') +} diff --git a/cloud/apps/relay/src/relay-token-verifier.ts b/cloud/apps/relay/src/relay-token-verifier.ts new file mode 100644 index 00000000000..d3dc5300a44 --- /dev/null +++ b/cloud/apps/relay/src/relay-token-verifier.ts @@ -0,0 +1,35 @@ +import { createRemoteJWKSet, jwtVerify } from 'jose' +import { z } from 'zod' +import type { RelayConfig } from './config.js' + +const ClaimsSchema = z.object({ + sub: z.string().min(1), + prof: z.string().min(1), + org: z.string().min(1).optional(), + relayHostId: z.string().regex(/^[A-Za-z0-9_-]{16}$/), + purpose: z.literal('host-control'), + exp: z.number().int().positive() +}) + +export type RelayTokenClaims = z.infer + +export function createRelayTokenVerifier(config: RelayConfig): (token: string) => Promise { + const jwks = createRemoteJWKSet(new URL(config.jwksUrl)) + return async (token) => { + try { + const verified = await jwtVerify(token, jwks, { + issuer: config.authIssuer, + audience: config.authAudience, + algorithms: ['ES256'] + }) + return ClaimsSchema.parse(verified.payload) + } catch { + return null + } + } +} + +export function readBearer(value: string | undefined): string | null { + const match = /^Bearer ([^\s]+)$/.exec(value ?? '') + return match?.[1] ?? null +} diff --git a/cloud/apps/relay/src/relay-websocket-close.test.ts b/cloud/apps/relay/src/relay-websocket-close.test.ts new file mode 100644 index 00000000000..222da3f1a7e --- /dev/null +++ b/cloud/apps/relay/src/relay-websocket-close.test.ts @@ -0,0 +1,44 @@ +import { EventEmitter } from 'node:events' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import { closeRelayWebSocket } from './relay-websocket-close.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readonly CLOSING = 2 + readonly CLOSED = 3 + readyState = this.OPEN + readonly close = vi.fn(() => { + this.readyState = this.CLOSING + }) + readonly terminate = vi.fn(() => { + this.readyState = this.CLOSED + this.emit('close') + }) +} + +describe('relay WebSocket close', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => vi.useRealTimers()) + + it('terminates a peer that does not complete the close handshake', () => { + const socket = new FakeSocket() + closeRelayWebSocket(socket as unknown as WebSocket, 4408, 'relay draining') + + expect(socket.close).toHaveBeenCalledWith(4408, 'relay draining') + vi.advanceTimersByTime(999) + expect(socket.terminate).not.toHaveBeenCalled() + vi.advanceTimersByTime(1) + expect(socket.terminate).toHaveBeenCalledOnce() + }) + + it('cancels forced termination after the peer closes', () => { + const socket = new FakeSocket() + closeRelayWebSocket(socket as unknown as WebSocket, 4408, 'relay draining') + socket.readyState = socket.CLOSED + socket.emit('close') + vi.runAllTimers() + + expect(socket.terminate).not.toHaveBeenCalled() + }) +}) diff --git a/cloud/apps/relay/src/relay-websocket-close.ts b/cloud/apps/relay/src/relay-websocket-close.ts new file mode 100644 index 00000000000..d4a7f61a3ee --- /dev/null +++ b/cloud/apps/relay/src/relay-websocket-close.ts @@ -0,0 +1,22 @@ +import type WebSocket from 'ws' + +const RELAY_WEBSOCKET_FORCE_CLOSE_MS = 1_000 +const forceCloseTimers = new WeakMap>() + +export function closeRelayWebSocket(socket: WebSocket, code: number, reason: string): void { + if (socket.readyState === socket.CLOSED) return + if (!forceCloseTimers.has(socket)) { + const timer = setTimeout(() => { + forceCloseTimers.delete(socket) + if (socket.readyState !== socket.CLOSED) socket.terminate() + }, RELAY_WEBSOCKET_FORCE_CLOSE_MS) + timer.unref() + forceCloseTimers.set(socket, timer) + socket.once('close', () => { + const pending = forceCloseTimers.get(socket) + if (pending) clearTimeout(pending) + forceCloseTimers.delete(socket) + }) + } + if (socket.readyState === socket.OPEN) socket.close(code, reason) +} diff --git a/cloud/apps/relay/src/relay.blackbox.test.ts b/cloud/apps/relay/src/relay.blackbox.test.ts new file mode 100644 index 00000000000..38134213e76 --- /dev/null +++ b/cloud/apps/relay/src/relay.blackbox.test.ts @@ -0,0 +1,2622 @@ +import { spawn, type ChildProcess } from 'node:child_process' +import { createHash, createHmac } from 'node:crypto' +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { createServer, type Server } from 'node:http' +import { createServer as createNetServer } from 'node:net' +import { dirname, resolve } from 'node:path' +import { tmpdir } from 'node:os' +import { fileURLToPath } from 'node:url' +import { exportJWK, generateKeyPair, jwtVerify, SignJWT } from 'jose' +import { + buildHostProofMacInput, + HOST_CHALLENGE_PLAINTEXT_DOMAIN +} from '@orca-cloud/relay-contract' +import nacl from 'tweetnacl' +import { afterAll, beforeAll, describe, expect, it } from 'vitest' +import WebSocket from 'ws' +import type { RawData } from 'ws' +import { RelayAssignmentStore } from './assignment-store.js' +import { + encodeMembership, + type CellAdmissionMembership +} from './cell-admission-selector.js' +import { openRelayDatabase } from './database.js' + +const appDirectory = resolve(dirname(fileURLToPath(import.meta.url)), '..') +const assignmentKey = 'test-assignment-key-with-at-least-32-bytes' +let relayProcess: ChildProcess +let jwksServer: Server +let relayUrl: string +let issuer: string +let privateKey: Awaited>['privateKey'] +let adminPrivateKey: Awaited>['privateKey'] +let relayDataDirectory: string +let adminAudience: string +let forwardedSourceSequence = 0 + +function forwardedHeaders(): Record { + forwardedSourceSequence++ + return { + 'x-forwarded-for': `spoofed, 192.0.2.${forwardedSourceSequence}, 35.191.0.1` + } +} + +async function unusedPort(): Promise { + const server = createNetServer() + await new Promise((resolveListen) => server.listen(0, '127.0.0.1', resolveListen)) + const address = server.address() + if (!address || typeof address === 'string') throw new Error('missing test port') + await new Promise((resolveClose) => server.close(() => resolveClose())) + return address.port +} + +async function waitForRelay(child: ChildProcess): Promise { + await new Promise((resolveReady, reject) => { + let stderr = '' + const timeout = setTimeout(() => reject(new Error('relay did not start')), 10_000) + child.stdout?.on('data', (chunk: Buffer) => { + if (chunk.toString().includes('[orca-relay] listening')) { + clearTimeout(timeout) + resolveReady() + } + }) + child.stderr?.on('data', (chunk: Buffer) => (stderr += chunk.toString())) + child.once('exit', (code) => + reject(new Error(`relay exited before ready: ${code}\n${stderr}`)) + ) + }) +} + +async function relayToken(audience: string, relayHostId = 'abcdefghijklmnop'): Promise { + return await new SignJWT({ + prof: 'profile-1', + org: 'org-1', + purpose: 'host-control', + relayHostId + }) + .setProtectedHeader({ alg: 'ES256', kid: 'test-key' }) + .setIssuer(issuer) + .setAudience(audience) + .setSubject('user-1') + .setIssuedAt() + .setExpirationTime('5m') + .sign(privateKey) +} + +async function adminToken(): Promise { + return await googleServiceToken(adminAudience) +} + +async function googleServiceToken(audience: string, email = 'deploy@example.com'): Promise { + return await new SignJWT({ email, email_verified: true }) + .setProtectedHeader({ alg: 'RS256', kid: 'admin-key' }) + .setIssuer('https://accounts.google.com') + .setAudience(audience) + .setSubject('deploy-subject') + .setIssuedAt() + .setExpirationTime('5m') + .sign(adminPrivateKey) +} + +async function postCellHeartbeat( + directorUrl: string, + cell: { id: string; url: string }, + overrides: { incarnation?: string; startedAt?: number; ready?: boolean } = {} +): Promise { + const audience = `${directorUrl}/v1/admin/cell-heartbeat` + return await fetch(audience, { + method: 'POST', + headers: { + authorization: `Bearer ${await googleServiceToken(audience)}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + v: 1, + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: overrides.incarnation ?? '11111111-1111-4111-8111-111111111111', + startedAt: overrides.startedAt ?? Date.now(), + ready: overrides.ready ?? true, + observedRequests: 0 + }) + }) +} + +function spawnTopologyRelay(input: { + url: string + dataDirectory: string + role: 'director' | 'cell' + cellId: string + cells?: Array<{ id: string; url: string; capacityRequests: number }> +}): ChildProcess { + return spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: new URL(input.url).port, + ORCA_RELAY_PUBLIC_URL: input.url, + ORCA_RELAY_CELL_URL: input.url, + ORCA_RELAY_CELL_ID: input.cellId, + ORCA_RELAY_CELL_CAPACITY: '10', + ORCA_RELAY_CELLS_JSON: JSON.stringify(input.cells ?? []), + ORCA_RELAY_ROLE: input.role, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: input.dataDirectory, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_MONITOR_SERVICE_ACCOUNT: 'monitor@example.com', + ORCA_RELAY_FENCE_SERVICE_ACCOUNT: 'fence@example.com', + ORCA_RELAY_FENCE_BROKER_SERVICE_ACCOUNT: 'broker@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) +} + +function nextMessage(socket: WebSocket): Promise> { + return new Promise((resolveMessage, reject) => { + const cleanup = (): void => { + socket.off('message', onMessage) + socket.off('error', onError) + socket.off('close', onClose) + } + const onMessage = (data: RawData): void => { + cleanup() + try { + resolveMessage(JSON.parse(data.toString()) as Record) + } catch (error) { + reject(error) + } + } + const onError = (error: Error): void => { + cleanup() + reject(error) + } + const onClose = (code: number, reason: Buffer): void => { + cleanup() + reject(new Error(`socket closed before message: ${code} ${reason.toString()}`)) + } + socket.once('message', onMessage) + socket.once('error', onError) + socket.once('close', onClose) + }) +} + +function nextRawMessage(socket: WebSocket): Promise<{ data: Buffer; binary: boolean }> { + return new Promise((resolveMessage, reject) => { + socket.once('message', (data, binary) => + resolveMessage({ data: Buffer.from(data as ArrayBuffer), binary }) + ) + socket.once('error', reject) + }) +} + +function collectMessages(socket: WebSocket, count: number): Promise[]> { + return new Promise((resolveMessages, reject) => { + const messages: Record[] = [] + const onMessage = (data: RawData): void => { + try { + messages.push(JSON.parse(data.toString()) as Record) + if (messages.length === count) { + socket.off('message', onMessage) + resolveMessages(messages) + } + } catch (error) { + reject(error) + } + } + socket.on('message', onMessage) + socket.once('error', reject) + }) +} + +async function installDirectCredential(input: { + host: WebSocket + relayDeviceId: string + reqId: string + resumeToken: string +}): Promise> { + const response = nextMessage(input.host) + input.host.send( + JSON.stringify({ + type: 'device-credential-install', + v: 1, + reqId: input.reqId, + relayDeviceId: input.relayDeviceId, + newResumeTokenHash: createHash('sha256').update(input.resumeToken).digest('base64url'), + authorization: { mode: 'authenticated-direct', directAuthId: `direct-${input.reqId}` } + }) + ) + return await response +} + +async function attachPhone(input: { + host: WebSocket + hostAck: Record + hostId: string + credential: string +}): Promise<{ + phone: WebSocket + data: WebSocket + connId: string + hello: Record +}> { + const headers = forwardedHeaders() + const phone = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${input.hostId}`, { + headers + }) + await new Promise((resolveOpen, reject) => { + phone.once('open', resolveOpen) + phone.once('error', reject) + }) + const connOpenPromise = nextMessage(input.host) + phone.send(JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: input.credential })) + const connOpen = await connOpenPromise + expect(connOpen.type).toBe('conn-open') + const connId = String(connOpen.connId) + const data = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/host/data/${connId}`, { + headers + }) + await new Promise((resolveOpen, reject) => { + data.once('open', resolveOpen) + data.once('error', reject) + }) + const helloPromise = nextMessage(phone) + data.send( + JSON.stringify({ + type: 'host-data-auth', + v: 1, + connTicket: connOpen.connTicket, + generation: input.hostAck.generation + }) + ) + const hello = await helloPromise + expect(hello).toMatchObject({ type: 'relay-hello', ok: true }) + return { phone, data, connId, hello } +} + +async function openHostControl(input?: { + controlResumeSecret?: string + previousGeneration?: number + keyPair?: nacl.BoxKeyPair + assignmentEpoch?: number +}): Promise<{ socket: WebSocket; ack: Record; keyPair: nacl.BoxKeyPair }> { + const keyPair = input?.keyPair ?? nacl.box.keyPair() + const hostId = createHash('sha256').update(keyPair.publicKey).digest('base64url').slice(0, 16) + const socket = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/host/control`, { + headers: { authorization: `Bearer ${await relayToken('orca-relay', hostId)}` }, + perMessageDeflate: false + }) + await new Promise((resolveOpen, reject) => { + socket.once('open', resolveOpen) + socket.once('error', reject) + }) + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hostId, + assignmentEpoch: input?.assignmentEpoch ?? 1, + hostPublicKeyB64: Buffer.from(keyPair.publicKey).toString('base64'), + appVersion: 'test', + ...(input?.controlResumeSecret + ? { controlResumeSecret: input.controlResumeSecret } + : {}), + ...(input?.previousGeneration === undefined + ? {} + : { previousGeneration: input.previousGeneration }) + }) + ) + const challenge = await nextMessage(socket) + expect(challenge.type).toBe('host-challenge') + const plaintext = nacl.box.open( + Buffer.from(String(challenge.ciphertextB64), 'base64'), + Buffer.from(String(challenge.nonceB64), 'base64'), + Buffer.from(String(challenge.relayEphemeralPublicKeyB64), 'base64'), + keyPair.secretKey + ) + if (!plaintext) throw new Error('host challenge did not decrypt') + const domain = new TextEncoder().encode(`${HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`) + expect(plaintext.slice(0, domain.length)).toEqual(domain) + const transcriptLength = new DataView( + plaintext.buffer, + plaintext.byteOffset + domain.length, + 4 + ).getUint32(0, false) + const transcriptStart = domain.length + 4 + const transcript = plaintext.slice(transcriptStart, transcriptStart + transcriptLength) + const secret = plaintext.slice(transcriptStart + transcriptLength) + const proofB64 = createHmac('sha256', secret) + .update(buildHostProofMacInput(transcript)) + .digest('base64') + socket.send( + JSON.stringify({ + type: 'host-challenge-ack', + challengeId: challenge.challengeId, + proofB64 + }) + ) + return { socket, ack: await nextMessage(socket), keyPair } +} + +beforeAll(async () => { + const keys = await generateKeyPair('ES256') + privateKey = keys.privateKey + const publicJwk = await exportJWK(keys.publicKey) + const adminKeys = await generateKeyPair('RS256') + adminPrivateKey = adminKeys.privateKey + const adminPublicJwk = await exportJWK(adminKeys.publicKey) + jwksServer = createServer((_request, response) => { + response.setHeader('content-type', 'application/json') + response.end( + JSON.stringify({ + keys: [ + { ...publicJwk, kid: 'test-key', alg: 'ES256', use: 'sig' }, + { ...adminPublicJwk, kid: 'admin-key', alg: 'RS256', use: 'sig' } + ] + }) + ) + }) + await new Promise((resolveListen) => jwksServer.listen(0, '127.0.0.1', resolveListen)) + const jwksAddress = jwksServer.address() + if (!jwksAddress || typeof jwksAddress === 'string') throw new Error('missing JWKS address') + issuer = `http://127.0.0.1:${jwksAddress.port}` + const relayPort = await unusedPort() + relayUrl = `http://127.0.0.1:${relayPort}` + adminAudience = `${relayUrl}/v1/admin/drain` + relayDataDirectory = mkdtempSync(resolve(tmpdir(), 'orca-relay-blackbox-')) + relayProcess = spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: String(relayPort), + ORCA_RELAY_PUBLIC_URL: relayUrl, + ORCA_RELAY_CELL_URL: relayUrl, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: relayDataDirectory, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_MONITOR_SERVICE_ACCOUNT: 'monitor@example.com', + ORCA_RELAY_FENCE_SERVICE_ACCOUNT: 'fence@example.com', + ORCA_RELAY_IMAGE_DIGEST: `sha256:${'a'.repeat(64)}`, + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + await waitForRelay(relayProcess) +}) + +afterAll(async () => { + relayProcess?.kill('SIGTERM') + await new Promise((resolveClose) => jwksServer?.close(() => resolveClose())) + rmSync(relayDataDirectory, { recursive: true, force: true }) +}) + +describe('served relay URL', () => { + it('exposes only /health, never /healthz', async () => { + expect(await (await fetch(`${relayUrl}/health`)).json()).toEqual({ + ok: true, + connectionCapacityProtocol: 2 + }) + expect(await (await fetch(`${relayUrl}/ready`)).json()).toEqual({ ok: true }) + expect((await fetch(`${relayUrl}/healthz`)).status).toBe(404) + }) + + it('keeps liveness healthy when dependency readiness fails', async () => { + const port = await unusedPort() + const url = `http://127.0.0.1:${port}` + const dataDirectory = mkdtempSync(resolve(tmpdir(), 'orca-relay-unready-')) + const processUnderTest = spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: String(port), + ORCA_RELAY_PUBLIC_URL: url, + ORCA_RELAY_CELL_URL: url, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: 'http://127.0.0.1:1/jwks', + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: dataDirectory, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_IMAGE_DIGEST: `sha256:${'a'.repeat(64)}`, + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + try { + await waitForRelay(processUnderTest) + expect((await fetch(`${url}/health`)).status).toBe(200) + expect((await fetch(`${url}/ready`)).status).toBe(503) + } finally { + processUnderTest.kill('SIGTERM') + await new Promise((resolveExit) => processUnderTest.once('exit', () => resolveExit())) + rmSync(dataDirectory, { recursive: true, force: true }) + } + }) + + it('rejects missing and broad-audience bearer tokens', async () => { + const body = JSON.stringify({ v: 1, relayHostId: 'abcdefghijklmnop' }) + expect((await fetch(`${relayUrl}/v1/assign`, { method: 'POST', body })).status).toBe(401) + expect( + ( + await fetch(`${relayUrl}/v1/assign`, { + method: 'POST', + headers: { authorization: `Bearer ${await relayToken('orca-cloud')}` }, + body + }) + ).status + ).toBe(401) + }) + + it('bounds slow first-frame admission and rejects URL credentials before upgrade', async () => { + const slow: WebSocket[] = [] + for (let index = 0; index < 4; index++) { + const socket = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop` + ) + await new Promise((resolveOpen, reject) => { + socket.once('open', resolveOpen) + socket.once('error', reject) + }) + slow.push(socket) + } + const limited = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop` + ) + const limitedStatus = await new Promise((resolveStatus) => { + limited.once('unexpected-response', (_request, response) => + resolveStatus(response.statusCode ?? 0) + ) + limited.once('error', () => {}) + }) + expect(limitedStatus).toBe(429) + const isolated = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop`, + { headers: { 'x-forwarded-for': 'spoofed, 198.51.100.9, 35.191.0.1' } } + ) + await new Promise((resolveOpen, reject) => { + isolated.once('open', resolveOpen) + isolated.once('error', reject) + }) + isolated.close() + for (const socket of slow) socket.close() + + const leaked = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop?credential=secret` + ) + const leakedStatus = await new Promise((resolveStatus) => { + leaked.once('unexpected-response', (_request, response) => + resolveStatus(response.statusCode ?? 0) + ) + leaked.once('error', () => {}) + }) + expect(leakedStatus).toBe(400) + }) + + it('enforces process-wide slow-auth and per-source rate budgets', async () => { + const slow: WebSocket[] = [] + for (let index = 0; index < 45; index++) { + const socket = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop`, + { headers: { 'x-forwarded-for': `spoofed, 198.51.100.${index + 1}, 35.191.0.1` } } + ) + await new Promise((resolveOpen, reject) => { + socket.once('open', resolveOpen) + socket.once('error', reject) + }) + slow.push(socket) + } + const globallyLimited = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop`, + { headers: { 'x-forwarded-for': 'spoofed, 203.0.113.1, 35.191.0.1' } } + ) + const globalStatus = await new Promise((resolveStatus) => { + globallyLimited.once('unexpected-response', (_request, response) => + resolveStatus(response.statusCode ?? 0) + ) + globallyLimited.once('error', () => {}) + }) + expect(globalStatus).toBe(429) + await Promise.all( + slow.map( + (socket) => + new Promise((resolveClose) => { + socket.once('close', () => resolveClose()) + socket.close() + }) + ) + ) + + for (let index = 0; index < 30; index++) { + const socket = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop`, + { headers: { 'x-forwarded-for': 'spoofed, 203.0.113.2, 35.191.0.1' } } + ) + await new Promise((resolveOpen, reject) => { + socket.once('open', resolveOpen) + socket.once('error', reject) + }) + const closed = new Promise((resolveClose) => socket.once('close', () => resolveClose())) + socket.send('{}') + await closed + } + const rateLimited = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/abcdefghijklmnop`, + { headers: { 'x-forwarded-for': 'spoofed, 203.0.113.2, 35.191.0.1' } } + ) + const rateStatus = await new Promise((resolveStatus) => { + rateLimited.once('unexpected-response', (_request, response) => + resolveStatus(response.statusCode ?? 0) + ) + rateLimited.once('error', () => {}) + }) + expect(rateStatus).toBe(429) + }) + + it('returns a signed combined-service assignment for the bound host', async () => { + const response = await fetch(`${relayUrl}/v1/assign`, { + method: 'POST', + headers: { + authorization: `Bearer ${await relayToken('orca-relay')}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, relayHostId: 'abcdefghijklmnop' }) + }) + expect(response.status).toBe(200) + const assignment = (await response.json()) as { + v: number + cellUrl: string + assignmentEpoch: number + lease: string + } + expect(assignment).toMatchObject({ v: 1, cellUrl: relayUrl, assignmentEpoch: 1 }) + const verified = await jwtVerify(assignment.lease, new TextEncoder().encode(assignmentKey), { + issuer: relayUrl, + audience: 'orca-relay-cell', + algorithms: ['HS256'] + }) + expect(verified.payload).toMatchObject({ relayHostId: 'abcdefghijklmnop' }) + }) + + it('requires canonical host key possession and supports same-generation rebind', async () => { + const first = await openHostControl() + expect(first.ack).toMatchObject({ type: 'host-hello-ack', v: 1, generation: 1 }) + const firstClose = new Promise((resolveClose) => + first.socket.once('close', (code) => resolveClose(code)) + ) + const rebound = await openHostControl({ + keyPair: first.keyPair, + controlResumeSecret: String(first.ack.controlResumeSecret), + previousGeneration: 1 + }) + expect(rebound.ack).toMatchObject({ type: 'host-hello-ack', v: 1, generation: 1 }) + expect(await firstClose).toBe(4408) + rebound.socket.close() + }) + + it('rejects a scoped token presented by a different host key', async () => { + const claimedKey = nacl.box.keyPair() + const wrongKey = nacl.box.keyPair() + const hostId = createHash('sha256').update(claimedKey.publicKey).digest('base64url').slice(0, 16) + const socket = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/host/control`, { + headers: { authorization: `Bearer ${await relayToken('orca-relay', hostId)}` } + }) + await new Promise((resolveOpen) => socket.once('open', resolveOpen)) + const closed = new Promise((resolveClose) => + socket.once('close', (code) => resolveClose(code)) + ) + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hostId, + assignmentEpoch: 1, + hostPublicKeyB64: Buffer.from(wrongKey.publicKey).toString('base64'), + appVersion: 'test' + }) + ) + expect(await closed).toBe(4401) + }) + + it('returns typed 4409 without a cell URL for a stale assignment epoch', async () => { + const keyPair = nacl.box.keyPair() + const hostId = createHash('sha256').update(keyPair.publicKey).digest('base64url').slice(0, 16) + const socket = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/host/control`, { + headers: { authorization: `Bearer ${await relayToken('orca-relay', hostId)}` } + }) + await new Promise((resolveOpen, reject) => { + socket.once('open', resolveOpen) + socket.once('error', reject) + }) + const closed = new Promise<{ code: number; reason: string }>((resolveClose) => + socket.once('close', (code, reason) => + resolveClose({ code, reason: reason.toString() }) + ) + ) + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hostId, + assignmentEpoch: 2, + hostPublicKeyB64: Buffer.from(keyPair.publicKey).toString('base64'), + appVersion: 'test' + }) + ) + const result = await closed + expect(result.code).toBe(4409) + expect(result.reason).not.toContain('http') + }) + + it('keeps a pending attach usable after a bad ticket and rejects ticket replay', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const inviteResponse = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'ticket-invite', relayDeviceId: 'ticket-device' }) + ) + const invite = await inviteResponse + const phone = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen, reject) => { + phone.once('open', resolveOpen) + phone.once('error', reject) + }) + const connectionPromise = nextMessage(host.socket) + phone.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: invite.inviteToken }) + ) + const connection = await connectionPromise + const dataUrl = `${relayUrl.replace('http:', 'ws:')}/v1/host/data/${connection.connId}` + + const badData = new WebSocket(dataUrl, { headers: forwardedHeaders() }) + await new Promise((resolveOpen, reject) => { + badData.once('open', resolveOpen) + badData.once('error', reject) + }) + const badClosed = new Promise((resolveClose) => + badData.once('close', (code) => resolveClose(code)) + ) + badData.send( + JSON.stringify({ + type: 'host-data-auth', + v: 1, + connTicket: Buffer.alloc(32, 1).toString('base64url'), + generation: host.ack.generation + }) + ) + expect(await badClosed).toBe(4401) + + const data = new WebSocket(dataUrl, { headers: forwardedHeaders() }) + await new Promise((resolveOpen, reject) => { + data.once('open', resolveOpen) + data.once('error', reject) + }) + const hello = nextMessage(phone) + data.send( + JSON.stringify({ + type: 'host-data-auth', + v: 1, + connTicket: connection.connTicket, + generation: host.ack.generation + }) + ) + expect(await hello).toMatchObject({ type: 'relay-hello', ok: true }) + + const replay = new WebSocket(dataUrl, { headers: forwardedHeaders() }) + await new Promise((resolveOpen, reject) => { + replay.once('open', resolveOpen) + replay.once('error', reject) + }) + const replayClosed = new Promise((resolveClose) => + replay.once('close', (code) => resolveClose(code)) + ) + replay.send( + JSON.stringify({ + type: 'host-data-auth', + v: 1, + connTicket: connection.connTicket, + generation: host.ack.generation + }) + ) + expect(await replayClosed).toBe(4401) + phone.close() + data.close() + host.socket.close() + }) + + it('attaches before success, preserves text/binary opcodes, installs, and confirms resume', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const invitePromise = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'invite-1', relayDeviceId: 'device-1' }) + ) + const invite = await invitePromise + expect(invite.type).toBe('invite-created') + const first = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: String(invite.inviteToken) + }) + + const hostText = nextRawMessage(first.data) + first.phone.send('phone-text') + expect(await hostText).toEqual({ data: Buffer.from('phone-text'), binary: false }) + const phoneBinary = nextRawMessage(first.phone) + first.data.send(Buffer.from([1, 2, 3]), { binary: true }) + expect(await phoneBinary).toEqual({ data: Buffer.from([1, 2, 3]), binary: true }) + + const resumeToken = Buffer.alloc(32, 9).toString('base64url') + const installedPromise = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-credential-install', + v: 1, + reqId: 'install-1', + relayDeviceId: 'device-1', + newResumeTokenHash: createHash('sha256').update(resumeToken).digest('base64url'), + authorization: { mode: 'relay-basis', basisConnId: first.connId } + }) + ) + const installed = await installedPromise + expect(installed).toMatchObject({ + type: 'device-credential-installed', + authorizationMode: 'relay-basis', + currentVersion: 1 + }) + const resolved = await fetch(`${relayUrl}/v1/resolve`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, relayHostId: hostId, resumeToken }) + }) + expect(resolved.status).toBe(200) + expect(await resolved.json()).toMatchObject({ v: 1, cellUrl: relayUrl, assignmentEpoch: 1 }) + first.phone.close() + + const resumed = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: resumeToken + }) + const confirmedPromise = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'confirm-1', + basisConnId: resumed.connId + }) + ) + expect(await confirmedPromise).toMatchObject({ + type: 'device-resume-confirmed', + renewed: true, + acceptedAs: 'current' + }) + resumed.phone.close() + resumed.data.close() + host.socket.close() + }) + + it('does not renew outer-only or injected confirmations and reports offline/peer loss', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const resumeToken = Buffer.alloc(32, 11).toString('base64url') + expect( + await installDirectCredential({ + host: host.socket, + relayDeviceId: 'outer-device', + reqId: 'outer-install', + resumeToken + }) + ).toMatchObject({ type: 'device-credential-installed', currentVersion: 1 }) + + const first = await attachPhone({ host: host.socket, hostAck: host.ack, hostId, credential: resumeToken }) + const originalExpiry = first.hello.resumeExpiresAt + first.phone.close() + first.data.close() + await new Promise((resolveWait) => setTimeout(resolveWait, 25)) + const closedBasis = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'closed-basis-confirm', + basisConnId: first.connId + }) + ) + expect(await closedBasis).toMatchObject({ + type: 'control-error', + reqId: 'closed-basis-confirm', + code: 'confirmation_not_active' + }) + const second = await attachPhone({ host: host.socket, hostAck: host.ack, hostId, credential: resumeToken }) + expect(second.hello.resumeExpiresAt).toBe(originalExpiry) + const injectedResult = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'injected-confirm', + basisConnId: second.connId, + relayDeviceId: 'injected', + acceptedCredentialVersion: 99 + }) + ) + expect(await injectedResult).toMatchObject({ type: 'control-error', reqId: 'injected-confirm' }) + const phoneClosed = new Promise((resolveClose) => + second.phone.once('close', (code) => resolveClose(code)) + ) + second.data.close() + expect(await phoneClosed).toBe(4408) + + await new Promise((resolveClose) => { + host.socket.once('close', () => resolveClose()) + host.socket.close() + }) + const offline = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen) => offline.once('open', resolveOpen)) + const offlineHello = nextMessage(offline) + offline.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: resumeToken }) + ) + expect(await offlineHello).toEqual({ type: 'relay-hello', ok: false, code: 4404 }) + }) + + it('rejects the first post-E2EE confirmation after its server-owned deadline', async () => { + const originalUrl = relayUrl + const clockPort = await unusedPort() + const clockUrl = `http://127.0.0.1:${clockPort}` + const clockData = mkdtempSync(resolve(tmpdir(), 'orca-relay-clock-')) + const clockFile = resolve(clockData, 'offset-ms') + writeFileSync(clockFile, '0') + const clockProcess = spawn( + process.execPath, + ['--import', 'tsx', 'src/fault-injection-test-entry.ts'], + { + cwd: appDirectory, + env: { + ...process.env, + NODE_ENV: 'test', + PORT: String(clockPort), + ORCA_RELAY_PUBLIC_URL: clockUrl, + ORCA_RELAY_CELL_URL: clockUrl, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: clockData, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_TEST_CLOCK_FILE: clockFile + }, + stdio: ['ignore', 'pipe', 'pipe'] + } + ) + relayUrl = clockUrl + try { + await waitForRelay(clockProcess) + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const resumeToken = Buffer.alloc(32, 25).toString('base64url') + await installDirectCredential({ + host: host.socket, + relayDeviceId: 'late-confirm-device', + reqId: 'late-confirm-install', + resumeToken + }) + const splice = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: resumeToken + }) + writeFileSync(clockFile, '31000') + const rejected = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'late-first-confirm', + basisConnId: splice.connId + }) + ) + expect(await rejected).toMatchObject({ + type: 'control-error', + reqId: 'late-first-confirm', + code: 'confirmation_not_active' + }) + splice.phone.close() + splice.data.close() + host.socket.close() + } finally { + clockProcess.kill('SIGKILL') + await new Promise((resolveExit) => clockProcess.once('exit', () => resolveExit())) + relayUrl = originalUrl + rmSync(clockData, { recursive: true, force: true }) + } + }) + + it('fences a competing generation and rejects its old immutable basis', async () => { + const firstHost = await openHostControl() + const hostId = createHash('sha256') + .update(firstHost.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const resumeToken = Buffer.alloc(32, 12).toString('base64url') + await installDirectCredential({ + host: firstHost.socket, + relayDeviceId: 'fenced-device', + reqId: 'fenced-install', + resumeToken + }) + const splice = await attachPhone({ + host: firstHost.socket, + hostAck: firstHost.ack, + hostId, + credential: resumeToken + }) + const replacement = await openHostControl({ keyPair: firstHost.keyPair, previousGeneration: 1 }) + expect(replacement.ack.generation).toBe(2) + const rejected = nextMessage(replacement.socket) + replacement.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'wrong-generation-confirm', + basisConnId: splice.connId + }) + ) + expect(await rejected).toMatchObject({ + type: 'control-error', + reqId: 'wrong-generation-confirm', + code: 'confirmation_not_active' + }) + replacement.socket.close() + }) + + it('serializes late direct and invite authorization modes into one served result', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const inviteResponse = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'race-invite', relayDeviceId: 'race-device' }) + ) + const invite = await inviteResponse + const splice = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: String(invite.inviteToken) + }) + const responses = collectMessages(host.socket, 2) + const base = { + type: 'device-credential-install', + v: 1, + reqId: 'race-install', + relayDeviceId: 'race-device', + newResumeTokenHash: createHash('sha256').update('race-resume').digest('base64url') + } + host.socket.send( + JSON.stringify({ + ...base, + authorization: { mode: 'relay-basis', basisConnId: splice.connId } + }) + ) + host.socket.send( + JSON.stringify({ + ...base, + authorization: { mode: 'authenticated-direct', directAuthId: 'late-direct' } + }) + ) + const installed = await responses + expect(installed).toHaveLength(2) + expect(installed[0]).toEqual(installed[1]) + expect(installed[0]).toMatchObject({ type: 'device-credential-installed', currentVersion: 1 }) + splice.phone.close() + splice.data.close() + host.socket.close() + }) + + it('reconciles a direct commit after its response is ignored and rejects bad authorization', async () => { + const firstHost = await openHostControl() + const hostId = createHash('sha256') + .update(firstHost.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const inviteResponse = nextMessage(firstHost.socket) + firstHost.socket.send( + JSON.stringify({ + type: 'invite-create', + reqId: 'lost-response-invite', + relayDeviceId: 'lost-response-device' + }) + ) + const invite = await inviteResponse + const resumeToken = Buffer.alloc(32, 26).toString('base64url') + firstHost.socket.send( + JSON.stringify({ + type: 'device-credential-install', + v: 1, + reqId: 'lost-response-install', + relayDeviceId: 'lost-response-device', + newResumeTokenHash: createHash('sha256').update(resumeToken).digest('base64url'), + authorization: { mode: 'authenticated-direct', directAuthId: 'lost-response-direct' } + }) + ) + // The coordinator deliberately ignores the acknowledgement and recovers + // only from the durable status after its control transport disappears. + await new Promise((resolveWait) => setTimeout(resolveWait, 25)) + firstHost.socket.terminate() + const rebound = await openHostControl({ + keyPair: firstHost.keyPair, + controlResumeSecret: String(firstHost.ack.controlResumeSecret), + previousGeneration: Number(firstHost.ack.generation) + }) + const status = nextMessage(rebound.socket) + rebound.socket.send( + JSON.stringify({ + type: 'device-credential-install-status', + v: 1, + reqId: 'lost-response-install', + relayDeviceId: 'lost-response-device' + }) + ) + expect(await status).toMatchObject({ + type: 'device-credential-install-status-result', + state: 'committed', + result: { authorizationMode: 'authenticated-direct', currentVersion: 1 } + }) + + const invalidatedInvite = new WebSocket( + `${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, + { headers: forwardedHeaders() } + ) + await new Promise((resolveOpen, reject) => { + invalidatedInvite.once('open', resolveOpen) + invalidatedInvite.once('error', reject) + }) + const rejectedInvite = nextMessage(invalidatedInvite) + invalidatedInvite.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: invite.inviteToken }) + ) + expect(await rejectedInvite).toEqual({ type: 'relay-hello', ok: false, code: 4401 }) + + const unauthorized = nextMessage(rebound.socket) + rebound.socket.send( + JSON.stringify({ + type: 'device-credential-install', + v: 1, + reqId: 'unauthorized-install', + relayDeviceId: 'unauthorized-device', + newResumeTokenHash: createHash('sha256').update('unauthorized').digest('base64url'), + authorization: { mode: 'relay-basis', basisConnId: 'unknown-basis' } + }) + ) + expect(await unauthorized).toMatchObject({ + type: 'control-error', + reqId: 'unauthorized-install', + code: 'invalid_relay_basis' + }) + const missing = nextMessage(rebound.socket) + rebound.socket.send( + JSON.stringify({ + type: 'device-credential-install-status', + v: 1, + reqId: 'unauthorized-install', + relayDeviceId: 'unauthorized-device' + }) + ) + expect(await missing).toMatchObject({ + type: 'device-credential-install-status-result', + state: 'not-found' + }) + rebound.socket.close() + }) + + it('serializes confirmation against direct rotation, relay rotation, and revoke', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const firstToken = Buffer.alloc(32, 21).toString('base64url') + await installDirectCredential({ + host: host.socket, + relayDeviceId: 'confirm-race-device', + reqId: 'confirm-race-initial', + resumeToken: firstToken + }) + + const firstResume = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: firstToken + }) + const secondToken = Buffer.alloc(32, 22).toString('base64url') + const directResponses = collectMessages(host.socket, 2) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'confirm-before-direct', + basisConnId: firstResume.connId + }) + ) + host.socket.send( + JSON.stringify({ + type: 'device-credential-install', + v: 1, + reqId: 'direct-after-confirm', + relayDeviceId: 'confirm-race-device', + newResumeTokenHash: createHash('sha256').update(secondToken).digest('base64url'), + expectedCurrentHash: createHash('sha256').update(firstToken).digest('base64url'), + authorization: { mode: 'authenticated-direct', directAuthId: 'direct-after-confirm' } + }) + ) + const directResults = await directResponses + expect(directResults.find((result) => result.type === 'device-resume-confirmed')).toMatchObject({ + reqId: 'confirm-before-direct', + currentVersion: 1, + renewed: true + }) + expect(directResults.find((result) => result.type === 'device-credential-installed')).toMatchObject({ + reqId: 'direct-after-confirm', + currentVersion: 2 + }) + const replayPromise = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'confirm-before-direct', + basisConnId: firstResume.connId + }) + ) + expect(await replayPromise).toEqual( + directResults.find((result) => result.type === 'device-resume-confirmed') + ) + + const secondResume = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: secondToken + }) + const inviteResponse = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'invite-create', + reqId: 'relay-rotation-invite', + relayDeviceId: 'confirm-race-device' + }) + ) + const invite = await inviteResponse + const inviteSplice = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: String(invite.inviteToken) + }) + const thirdToken = Buffer.alloc(32, 23).toString('base64url') + const relayResponses = collectMessages(host.socket, 2) + host.socket.send( + JSON.stringify({ + type: 'device-credential-install', + v: 1, + reqId: 'relay-before-confirm', + relayDeviceId: 'confirm-race-device', + newResumeTokenHash: createHash('sha256').update(thirdToken).digest('base64url'), + expectedCurrentHash: createHash('sha256').update(secondToken).digest('base64url'), + authorization: { mode: 'relay-basis', basisConnId: inviteSplice.connId } + }) + ) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'confirm-after-relay', + basisConnId: secondResume.connId + }) + ) + const relayResults = await relayResponses + expect(relayResults.find((result) => result.type === 'device-credential-installed')).toMatchObject({ + reqId: 'relay-before-confirm', + currentVersion: 3 + }) + expect(relayResults.find((result) => result.type === 'device-resume-confirmed')).toMatchObject({ + reqId: 'confirm-after-relay', + currentVersion: 3, + renewed: false + }) + + const retiredResume = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: secondToken + }) + const fourthToken = Buffer.alloc(32, 24).toString('base64url') + expect( + await installDirectCredential({ + host: host.socket, + relayDeviceId: 'confirm-race-device', + reqId: 'retire-before-confirm', + resumeToken: fourthToken + }) + ).toMatchObject({ type: 'device-credential-installed', currentVersion: 4 }) + const retiredResult = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'confirm-retired', + basisConnId: retiredResume.connId + }) + ) + expect(await retiredResult).toMatchObject({ + type: 'control-error', + reqId: 'confirm-retired', + code: 'reject-retired' + }) + + const currentResume = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: fourthToken + }) + const revokeResponses = collectMessages(host.socket, 2) + host.socket.send( + JSON.stringify({ + type: 'device-revoke', + reqId: 'revoke-before-confirm', + relayDeviceId: 'confirm-race-device' + }) + ) + host.socket.send( + JSON.stringify({ + type: 'device-resume-confirm', + v: 1, + reqId: 'confirm-after-revoke', + basisConnId: currentResume.connId + }) + ) + const revokeResults = await revokeResponses + expect(revokeResults.find((result) => result.type === 'device-revoked')).toMatchObject({ + reqId: 'revoke-before-confirm' + }) + expect(revokeResults.find((result) => result.type === 'control-error')).toMatchObject({ + reqId: 'confirm-after-revoke', + code: 'reject-revoked' + }) + + for (const splice of [firstResume, secondResume, inviteSplice, retiredResume, currentResume]) { + splice.phone.close() + splice.data.close() + } + host.socket.close() + }) + + it('enforces the eight-splice host limit with typed 4429 recovery', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const resumeToken = Buffer.alloc(32, 13).toString('base64url') + await installDirectCredential({ + host: host.socket, + relayDeviceId: 'limit-device', + reqId: 'limit-install', + resumeToken + }) + const splices = [] + for (let index = 0; index < 8; index++) { + splices.push( + await attachPhone({ host: host.socket, hostAck: host.ack, hostId, credential: resumeToken }) + ) + } + const ninth = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen) => ninth.once('open', resolveOpen)) + const rejected = nextMessage(ninth) + ninth.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: resumeToken }) + ) + expect(await rejected).toEqual({ type: 'relay-hello', ok: false, code: 4429 }) + for (const splice of splices) { + splice.phone.close() + splice.data.close() + } + host.socket.close() + }) + + it('rolls an aborted invite reservation into bounded cooldown before retry', async () => { + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const inviteResponse = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'cooldown-invite', relayDeviceId: 'cooldown-device' }) + ) + const invite = await inviteResponse + const first = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen) => first.once('open', resolveOpen)) + const firstOpen = nextMessage(host.socket) + first.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: invite.inviteToken }) + ) + await firstOpen + first.close() + await new Promise((resolveWait) => setTimeout(resolveWait, 50)) + + const cooldown = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen) => cooldown.once('open', resolveOpen)) + const cooldownHello = nextMessage(cooldown) + cooldown.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: invite.inviteToken }) + ) + expect(await cooldownHello).toEqual({ type: 'relay-hello', ok: false, code: 4401 }) + + await new Promise((resolveWait) => setTimeout(resolveWait, 2_050)) + const retry = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen) => retry.once('open', resolveOpen)) + const retriedOpen = nextMessage(host.socket) + retry.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: invite.inviteToken }) + ) + expect(await retriedOpen).toMatchObject({ type: 'conn-open', kind: 'invite' }) + retry.close() + host.socket.close() + }) + + it('keeps install status and every effect not-found after injected SQL failure', async () => { + const originalUrl = relayUrl + const faultPort = await unusedPort() + const faultUrl = `http://127.0.0.1:${faultPort}` + const faultData = mkdtempSync(resolve(tmpdir(), 'orca-relay-fault-')) + const faultProcess = spawn( + process.execPath, + ['--import', 'tsx', 'src/fault-injection-test-entry.ts'], + { + cwd: appDirectory, + env: { + ...process.env, + NODE_ENV: 'test', + PORT: String(faultPort), + ORCA_RELAY_PUBLIC_URL: faultUrl, + ORCA_RELAY_CELL_URL: faultUrl, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: faultData, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_TEST_FAULT_SQL: 'INSERT INTO relay_install_results' + }, + stdio: ['ignore', 'pipe', 'pipe'] + } + ) + relayUrl = faultUrl + try { + await waitForRelay(faultProcess) + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const inviteResponse = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'fault-invite', relayDeviceId: 'fault-device' }) + ) + const invite = await inviteResponse + const splice = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: String(invite.inviteToken) + }) + const install = { + type: 'device-credential-install', + v: 1, + reqId: 'fault-install', + relayDeviceId: 'fault-device', + newResumeTokenHash: createHash('sha256').update('fault-resume').digest('base64url'), + authorization: { mode: 'relay-basis', basisConnId: splice.connId } + } + const failed = nextMessage(host.socket) + host.socket.send(JSON.stringify(install)) + expect(await failed).toMatchObject({ + type: 'control-error', + reqId: 'fault-install', + code: 'injected SQL failure' + }) + const status = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-credential-install-status', + v: 1, + reqId: 'fault-install', + relayDeviceId: 'fault-device' + }) + ) + expect(await status).toMatchObject({ + type: 'device-credential-install-status-result', + state: 'not-found' + }) + const retried = nextMessage(host.socket) + host.socket.send(JSON.stringify(install)) + expect(await retried).toMatchObject({ + type: 'device-credential-installed', + currentVersion: 1 + }) + splice.phone.close() + splice.data.close() + host.socket.close() + } finally { + faultProcess.kill('SIGKILL') + await new Promise((resolveExit) => faultProcess.once('exit', () => resolveExit())) + relayUrl = originalUrl + rmSync(faultData, { recursive: true, force: true }) + } + }) + + it('falls back to an invite only after a failed direct attempt is authoritatively not-found', async () => { + const originalUrl = relayUrl + const faultPort = await unusedPort() + const faultUrl = `http://127.0.0.1:${faultPort}` + const faultData = mkdtempSync(resolve(tmpdir(), 'orca-relay-direct-fault-')) + const faultProcess = spawn( + process.execPath, + ['--import', 'tsx', 'src/fault-injection-test-entry.ts'], + { + cwd: appDirectory, + env: { + ...process.env, + NODE_ENV: 'test', + PORT: String(faultPort), + ORCA_RELAY_PUBLIC_URL: faultUrl, + ORCA_RELAY_CELL_URL: faultUrl, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: faultData, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_TEST_FAULT_SQL: 'INSERT INTO relay_direct_authorizations' + }, + stdio: ['ignore', 'pipe', 'pipe'] + } + ) + relayUrl = faultUrl + try { + await waitForRelay(faultProcess) + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const inviteResponse = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'invite-create', + reqId: 'fallback-invite', + relayDeviceId: 'fallback-device' + }) + ) + const invite = await inviteResponse + const splice = await attachPhone({ + host: host.socket, + hostAck: host.ack, + hostId, + credential: String(invite.inviteToken) + }) + const installBase = { + type: 'device-credential-install', + v: 1, + reqId: 'fallback-install', + relayDeviceId: 'fallback-device', + newResumeTokenHash: createHash('sha256').update('fallback-resume').digest('base64url') + } + const directFailure = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + ...installBase, + authorization: { mode: 'authenticated-direct', directAuthId: 'failed-direct' } + }) + ) + expect(await directFailure).toMatchObject({ + type: 'control-error', + reqId: 'fallback-install', + code: 'injected SQL failure' + }) + const status = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + type: 'device-credential-install-status', + v: 1, + reqId: 'fallback-install', + relayDeviceId: 'fallback-device' + }) + ) + expect(await status).toMatchObject({ + type: 'device-credential-install-status-result', + state: 'not-found' + }) + const fallback = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ + ...installBase, + authorization: { mode: 'relay-basis', basisConnId: splice.connId } + }) + ) + expect(await fallback).toMatchObject({ + type: 'device-credential-installed', + authorizationMode: 'relay-basis', + currentVersion: 1 + }) + splice.phone.close() + splice.data.close() + host.socket.close() + } finally { + faultProcess.kill('SIGKILL') + await new Promise((resolveExit) => faultProcess.once('exit', () => resolveExit())) + relayUrl = originalUrl + rmSync(faultData, { recursive: true, force: true }) + } + }) + + it('keeps director HTTP routes off cells and enforces the durable cell epoch', async () => { + const originalUrl = relayUrl + const cellPort = await unusedPort() + const cellUrl = `http://127.0.0.1:${cellPort}` + const cellData = mkdtempSync(resolve(tmpdir(), 'orca-relay-cell-')) + const keyPair = nacl.box.keyPair() + const hostId = createHash('sha256') + .update(keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const database = await openRelayDatabase({ dataDir: cellData }) + const assignments = new RelayAssignmentStore(database, () => 100) + await assignments.reconcileCells([{ id: 'cell-a', url: cellUrl, capacityRequests: 10 }]) + await assignments.assign({ userId: 'user-1', relayHostId: hostId }) + await database.close() + const cellProcess = spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: String(cellPort), + ORCA_RELAY_PUBLIC_URL: cellUrl, + ORCA_RELAY_CELL_URL: cellUrl, + ORCA_RELAY_CELL_ID: 'cell-a', + ORCA_RELAY_CELL_CAPACITY: '10', + ORCA_RELAY_ROLE: 'cell', + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: cellData, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + relayUrl = cellUrl + try { + await waitForRelay(cellProcess) + const token = await relayToken('orca-relay', hostId) + const assignmentResponse = await fetch(`${cellUrl}/v1/assign`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, relayHostId: hostId }) + }) + expect(assignmentResponse.status).toBe(404) + const cellStatusResponse = await fetch(`${cellUrl}/v1/admin/cell-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: 'cell-a' }) + }) + expect(cellStatusResponse.status).toBe(404) + + const stale = new WebSocket(`${cellUrl.replace('http:', 'ws:')}/v1/host/control`, { + headers: { authorization: `Bearer ${token}` } + }) + await new Promise((resolveOpen) => stale.once('open', resolveOpen)) + const staleClosed = new Promise((resolveClose) => + stale.once('close', (code) => resolveClose(code)) + ) + stale.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: hostId, + assignmentEpoch: 2, + hostPublicKeyB64: Buffer.from(keyPair.publicKey).toString('base64'), + appVersion: 'test' + }) + ) + expect(await staleClosed).toBe(4409) + + const assigned = await openHostControl({ keyPair }) + expect(assigned.ack).toMatchObject({ type: 'host-hello-ack', generation: 1 }) + const invitePromise = nextMessage(assigned.socket) + assigned.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'cell-invite', relayDeviceId: 'cell-device' }) + ) + const invite = await invitePromise + const splice = await attachPhone({ + host: assigned.socket, + hostAck: assigned.ack, + hostId, + credential: String(invite.inviteToken) + }) + const observedDatabase = await openRelayDatabase({ dataDir: cellData }) + const activities = await observedDatabase.query( + `SELECT activity_kind, request_units FROM relay_assignment_activity_leases + ORDER BY activity_kind, activity_id` + ) + await observedDatabase.close() + expect(activities).toEqual([ + { activity_kind: 'control', request_units: 1 }, + { activity_kind: 'invite', request_units: 1 }, + { activity_kind: 'invite', request_units: 1 }, + { activity_kind: 'splice', request_units: 2 } + ]) + splice.phone.close() + splice.data.close() + assigned.socket.close() + } finally { + cellProcess.kill('SIGTERM') + await new Promise((resolveExit) => cellProcess.once('exit', () => resolveExit())) + relayUrl = originalUrl + rmSync(cellData, { recursive: true, force: true }) + } + }) + + it('runs target-first dual-cell evacuation before releasing the source control', async () => { + const originalUrl = relayUrl + const dataDirectory = mkdtempSync(resolve(tmpdir(), 'orca-relay-topology-')) + const directorUrl = `http://127.0.0.1:${await unusedPort()}` + const cellAUrl = `http://127.0.0.1:${await unusedPort()}` + const cellBUrl = `http://127.0.0.1:${await unusedPort()}` + const cells = [ + { id: 'cell-a', url: cellAUrl, capacityRequests: 10 }, + { id: 'cell-b', url: cellBUrl, capacityRequests: 10 } + ] + const keyPair = nacl.box.keyPair() + const hostId = createHash('sha256') + .update(keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const seedDatabase = await openRelayDatabase({ dataDir: dataDirectory }) + const seedAssignments = new RelayAssignmentStore(seedDatabase) + await seedAssignments.reconcileCells(cells) + await seedAssignments.assign({ userId: 'user-1', relayHostId: hostId }) + await seedDatabase.close() + + const cellA = spawnTopologyRelay({ + url: cellAUrl, + dataDirectory, + role: 'cell', + cellId: 'cell-a' + }) + let director: ChildProcess | undefined + let cellB: ChildProcess | undefined + try { + await waitForRelay(cellA) + director = spawnTopologyRelay({ + url: directorUrl, + dataDirectory, + role: 'director', + cellId: 'director', + cells + }) + await waitForRelay(director) + expect( + await fetch(`${directorUrl}/v1/admin/cell-heartbeat`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: '{}' + }) + ).toMatchObject({ status: 401 }) + const heartbeatAudience = `${directorUrl}/v1/admin/cell-heartbeat` + expect( + await fetch(heartbeatAudience, { + method: 'POST', + headers: { + authorization: `Bearer ${await googleServiceToken(`${directorUrl}/wrong`)}`, + 'content-type': 'application/json' + }, + body: '{}' + }) + ).toMatchObject({ status: 401 }) + expect( + await fetch(heartbeatAudience, { + method: 'POST', + headers: { + authorization: `Bearer ${await googleServiceToken(heartbeatAudience, 'wrong@example.com')}`, + 'content-type': 'application/json' + }, + body: '{}' + }) + ).toMatchObject({ status: 401 }) + expect(await postCellHeartbeat(directorUrl, cells[0]!, { startedAt: 1_000 })).toMatchObject({ + status: 200 + }) + expect( + await postCellHeartbeat(directorUrl, cells[0]!, { + incarnation: '33333333-3333-4333-8333-333333333333', + startedAt: 999 + }) + ).toMatchObject({ status: 409 }) + expect( + await postCellHeartbeat(directorUrl, cells[1]!, { + incarnation: '22222222-2222-4222-8222-222222222222', + startedAt: 2_000 + }) + ).toMatchObject({ status: 200 }) + relayUrl = cellAUrl + const sourceHost = await openHostControl({ keyPair, assignmentEpoch: 1 }) + const token = await adminToken() + const recoveryRequests = [ + { + path: 'cell-fence-attempt-prepare', + body: { + v: 1, + attemptId: '44444444-4444-4444-8444-444444444444', + environment: 'production', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: + 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + planObjectName: + 'terraform/state/relay-fence-plans/production/44444444-4444-4444-8444-444444444444.tfplan', + varFileSha256: 'c'.repeat(64), + terraformStateLineage: 'c739dab4-e6e1-e627-02a9-504b3dda1a2c', + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: 'd'.repeat(64), + requestReason: + 'orca-relay-fence/44444444-4444-4444-8444-444444444444' + }, + confirmation: 'PREPARE_TERRAFORM_CELL_FENCE', + expected: { status: 409, body: { error: 'cell_fence_admission_enabled' } } + }, + { + path: 'cell-fence-attempt-abort', + body: { + v: 1, + attemptId: '44444444-4444-4444-8444-444444444444', + environment: 'production', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: 'orca-relay-c1', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: + 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenceCommit: 'a'.repeat(40), + planSha256: 'b'.repeat(64), + planObjectName: + 'terraform/state/relay-fence-plans/production/44444444-4444-4444-8444-444444444444.tfplan', + varFileSha256: 'c'.repeat(64), + terraformStateLineage: 'c739dab4-e6e1-e627-02a9-504b3dda1a2c', + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: 'd'.repeat(64), + requestReason: + 'orca-relay-fence/44444444-4444-4444-8444-444444444444' + }, + confirmation: 'ABORT_UNSTARTED_TERRAFORM_CELL_FENCE', + expected: { status: 409, body: { error: 'cell_fence_attempt_not_found' } } + }, + { + path: 'migration-supersede-cell', + body: { + v: 1, + sourceCellId: 'cell-a', + currentTargetCellId: 'cell-b', + replacementTargetCellId: 'cell-c', + limit: 100 + }, + confirmation: 'SUPERSEDE_REGISTERED_CELL_MIGRATIONS', + expected: { status: 200, body: { v: 1, superseded: 0 } } + }, + { + path: 'drain-attempt-prepare', + body: { + v: 1, + attemptId: '55555555-5555-4555-8555-555555555555', + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: '66666666-6666-4666-8666-666666666666', + graceMs: 120_000 + }, + confirmation: 'PREPARE_LEGACY_DRAIN', + expected: { status: 409, body: { error: 'drain_attempt_admission_enabled' } } + }, + { + path: 'drain-attempt-recover-forward', + body: { + v: 1, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111' + }, + confirmation: 'RECOVER_LEGACY_DRAIN', + expected: { status: 409, body: { error: 'drain_attempt_admission_enabled' } } + } + ] + for (const request of recoveryRequests) { + const url = `${directorUrl}/v1/admin/${request.path}` + expect( + await fetch(url, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(request.body) + }) + ).toMatchObject({ status: 400 }) + const confirmed = await fetch(url, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ ...request.body, confirmation: request.confirmation }) + }) + expect(confirmed.status).toBe(request.expected.status) + expect(await confirmed.json()).toEqual(request.expected.body) + } + const statusUrl = `${directorUrl}/v1/admin/cell-status` + expect( + await fetch(statusUrl, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: 'cell-a' }) + }) + ).toMatchObject({ status: 401 }) + expect( + await fetch(statusUrl, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: 'cell-a', userId: 'must-not-be-accepted' }) + }) + ).toMatchObject({ status: 400 }) + const sourceStatusResponse = await fetch(statusUrl, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: 'cell-a' }) + }) + expect(sourceStatusResponse.status).toBe(200) + const sourceStatusText = await sourceStatusResponse.text() + expect(sourceStatusText).not.toContain('user-1') + expect(sourceStatusText).not.toContain(hostId) + expect(JSON.parse(sourceStatusText)).toMatchObject({ + v: 1, + status: { + cellId: 'cell-a', + cellUrl: cellAUrl, + enabled: true, + assignments: 1, + activityLeases: 1, + activityRequestUnits: 1, + restartBlockingActivityLeases: 1, + restartBlockingActivityRequestUnits: 1, + restartBlockingReservedRequests: 1, + runtime: { cellUrl: cellAUrl, ready: true, heartbeatFresh: true } + } + }) + const cellState = async (cellId: string, enabled: boolean): Promise => + await fetch(`${directorUrl}/v1/admin/cell-state`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId, enabled }) + }) + const configuredTarget = await fetch(`${directorUrl}/v1/admin/cell-config`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + cellId: 'cell-b', + cellUrl: cellBUrl, + capacityRequests: 10, + state: 'existing-only' + }) + }) + expect(configuredTarget.status).toBe(200) + expect(await configuredTarget.json()).toEqual({ ok: true }) + const incompleteLimit = await fetch(`${directorUrl}/v1/admin/cell-config`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + cellId: 'cell-b', + cellUrl: cellBUrl, + capacityRequests: 10, + connectionHardCap: 600, + enabled: false + }) + }) + expect(incompleteLimit.status).toBe(400) + const capacity = await fetch(`${directorUrl}/v1/admin/evacuation-capacity`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, sourceCellId: 'cell-a', targetCellId: 'cell-b' }) + }) + expect(capacity.status).toBe(200) + expect(await capacity.json()).toEqual({ + v: 1, + sourceAssignments: 1, + requiredTargetUnits: 2, + availableTargetUnits: 10 + }) + const disabledTarget = await fetch(`${directorUrl}/v1/admin/evacuate`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + userId: 'user-1', + relayHostId: hostId, + targetCellId: 'cell-b' + }) + }) + expect(disabledTarget.status).toBe(409) + expect(await disabledTarget.json()).toEqual({ error: 'target_cell_unavailable' }) + expect((await cellState('cell-b', true)).status).toBe(200) + const migrationResponse = await fetch(`${directorUrl}/v1/admin/evacuate-cell`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + limit: 10 + }) + }) + expect(migrationResponse.status).toBe(200) + expect(await migrationResponse.json()).toEqual({ v: 1, started: 1 }) + const pendingStatus = await fetch(`${directorUrl}/v1/admin/evacuation-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + sourceCellId: 'cell-a', + targetCellId: 'cell-b' + }) + }) + expect(pendingStatus.status).toBe(200) + expect(await pendingStatus.json()).toMatchObject({ + v: 1, + inProgress: 1, + targetRegistered: 0, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + }) + // Completion must prove the source has stopped admitting new work, even in + // the served admin workflow that performs its own topology preflight. + expect((await cellState('cell-a', false)).status).toBe(200) + + cellB = spawnTopologyRelay({ + url: cellBUrl, + dataDirectory, + role: 'cell', + cellId: 'cell-b' + }) + await waitForRelay(cellB) + relayUrl = cellBUrl + const targetHost = await openHostControl({ keyPair, assignmentEpoch: 2 }) + const completionBody = JSON.stringify({ + v: 1, + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + completeReady: true + }) + const premature = await fetch(`${directorUrl}/v1/admin/evacuation-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: completionBody + }) + expect(premature.status).toBe(200) + expect(await premature.json()).toMatchObject({ + v: 1, + inProgress: 1, + targetRegistered: 1, + registeredSourceActive: 1, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 1, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + }) + + const sourceClosed = new Promise((resolveClose) => + sourceHost.socket.once('close', (code) => resolveClose(code)) + ) + const drain = await fetch(`${cellAUrl}/v1/admin/drain`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, graceMs: 0 }) + }) + expect(drain.status).toBe(200) + expect(await sourceClosed).toBe(4503) + + let completed: Response | undefined + for (let attempt = 0; attempt < 20; attempt++) { + completed = await fetch(`${directorUrl}/v1/admin/evacuation-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: completionBody + }) + if ( + completed.status === 200 && + ((await completed.clone().json()) as { inProgress?: number }).inProgress === 0 + ) { + break + } + await new Promise((resolveWait) => setTimeout(resolveWait, 10)) + } + expect(completed?.status).toBe(200) + expect(await completed?.json()).toMatchObject({ + v: 1, + inProgress: 0, + targetRegistered: 0, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 1, + blocked: 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + }) + const observedDatabase = await openRelayDatabase({ dataDir: dataDirectory }) + const assignment = await new RelayAssignmentStore(observedDatabase).resolve({ + userId: 'user-1', + relayHostId: hostId + }) + const reservations = await observedDatabase.query( + `SELECT cell_id, reserved_requests FROM relay_cells ORDER BY cell_id` + ) + await observedDatabase.close() + expect(assignment).toMatchObject({ cellId: 'cell-b', assignmentEpoch: 2 }) + expect(reservations).toEqual([ + { cell_id: 'cell-a', reserved_requests: 0 }, + { cell_id: 'cell-b', reserved_requests: 1 } + ]) + const selectorStatusBefore = await fetch( + `${directorUrl}/v1/admin/admission-selector/status`, + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1 }) + } + ) + expect(selectorStatusBefore.status).toBe(200) + const selectorBefore = (await selectorStatusBefore.json()) as { + selector: { membership: CellAdmissionMembership } + } + const selectorInput = { + v: 1, + attemptId: 'blackbox_cutover', + expectedGeneration: 0, + expectedMembershipSha256: createHash('sha256') + .update(encodeMembership(selectorBefore.selector.membership)) + .digest('hex'), + membership: { + existingOnly: ['cell-a'], + migrationOnly: [], + general: ['cell-b'] + } + } + const unsafeSelectorInput = { ...selectorInput, expectedMembershipSha256: undefined } + const unsafeSelectorResponse = await fetch( + `${directorUrl}/v1/admin/admission-selector/apply`, + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(unsafeSelectorInput) + } + ) + expect(unsafeSelectorResponse.status).toBe(400) + const selectorResponse = await fetch( + `${directorUrl}/v1/admin/admission-selector/apply`, + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(selectorInput) + } + ) + expect(selectorResponse.status).toBe(200) + expect(await selectorResponse.json()).toMatchObject({ + v: 1, + changed: true, + selector: { generation: 1, membership: selectorInput.membership } + }) + const selectorStatus = await fetch( + `${directorUrl}/v1/admin/admission-selector/status`, + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, attemptId: selectorInput.attemptId }) + } + ) + expect(selectorStatus.status).toBe(200) + expect(await selectorStatus.json()).toMatchObject({ + v: 1, + selector: { generation: 1, membership: selectorInput.membership }, + intent: { attemptId: selectorInput.attemptId, state: 'committed' } + }) + const addCellsInput = { + v: 1, + attemptId: 'blackbox_add_cells', + expectedGeneration: 1, + cells: [ + { + cellId: 'cell-c', + cellUrl: 'https://relay-c.example.com', + capacityRequests: 20, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + } + ] + } + const addCellsResponse = await fetch( + `${directorUrl}/v1/admin/admission-selector/add-migration-cells`, + { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(addCellsInput) + } + ) + expect(addCellsResponse.status).toBe(200) + expect(await addCellsResponse.json()).toMatchObject({ + v: 1, + changed: true, + selector: { + generation: 2, + membership: { + existingOnly: ['cell-a'], + migrationOnly: ['cell-c'], + general: ['cell-b'] + } + } + }) + expect((await cellState('cell-a', true)).status).toBe(409) + targetHost.socket.close() + } finally { + for (const child of [cellA, cellB, director]) { + if (child && child.exitCode === null) child.kill('SIGKILL') + } + relayUrl = originalUrl + rmSync(dataDirectory, { recursive: true, force: true }) + } + }) + + it('recovers an unexpired invite through a restarted configured director without reservation', async () => { + const combinedUrl = relayUrl + const host = await openHostControl() + const hostId = createHash('sha256') + .update(host.keyPair.publicKey) + .digest('base64url') + .slice(0, 16) + const invitePromise = nextMessage(host.socket) + host.socket.send( + JSON.stringify({ type: 'invite-create', reqId: 'move-invite', relayDeviceId: 'move-device' }) + ) + const invite = await invitePromise + host.socket.close() + + relayProcess.kill('SIGKILL') + await new Promise((resolveExit) => relayProcess.once('exit', () => resolveExit())) + const directorPort = await unusedPort() + relayUrl = `http://127.0.0.1:${directorPort}` + relayProcess = spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: String(directorPort), + ORCA_RELAY_PUBLIC_URL: relayUrl, + ORCA_RELAY_CELL_URL: 'https://relay-c2.onorca.dev', + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: relayDataDirectory, + ORCA_RELAY_ROLE: 'director', + ORCA_RELAY_CELLS_JSON: JSON.stringify([ + { + id: 'cell-c2', + url: 'https://relay-c2.onorca.dev', + capacityRequests: 900 + } + ]), + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + await waitForRelay(relayProcess) + expect( + await postCellHeartbeat(relayUrl, { + id: 'cell-c2', + url: 'https://relay-c2.onorca.dev' + }) + ).toMatchObject({ status: 200 }) + + const phone = new WebSocket(`${relayUrl.replace('http:', 'ws:')}/v1/connect/${hostId}`, { + headers: forwardedHeaders() + }) + await new Promise((resolveOpen) => phone.once('open', resolveOpen)) + const movedPromise = nextMessage(phone) + const closed = new Promise((resolveClose) => + phone.once('close', (code) => resolveClose(code)) + ) + phone.send( + JSON.stringify({ + type: 'relay-auth', + v: 1, + mode: 'connect', + credential: invite.inviteToken + }) + ) + expect(await movedPromise).toEqual({ + type: 'relay-moved', + v: 1, + cellUrl: 'https://relay-c2.onorca.dev', + assignmentEpoch: 1 + }) + expect(await closed).toBe(4503) + + relayProcess.kill('SIGTERM') + await new Promise((resolveExit) => relayProcess.once('exit', () => resolveExit())) + relayUrl = combinedUrl + relayProcess = spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: new URL(combinedUrl).port, + ORCA_RELAY_PUBLIC_URL: combinedUrl, + ORCA_RELAY_CELL_URL: combinedUrl, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: relayDataDirectory, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT: 'capacity@example.com', + ORCA_RELAY_MONITOR_SERVICE_ACCOUNT: 'monitor@example.com', + ORCA_RELAY_FENCE_SERVICE_ACCOUNT: 'fence@example.com', + ORCA_RELAY_IMAGE_DIGEST: `sha256:${'a'.repeat(64)}`, + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + await waitForRelay(relayProcess) + }) + + it('uses configured-director recovery and typed 4503 on unplanned SIGTERM', async () => { + const originalUrl = relayUrl + const signalPort = await unusedPort() + const signalUrl = `http://127.0.0.1:${signalPort}` + const signalData = mkdtempSync(resolve(tmpdir(), 'orca-relay-sigterm-')) + const signalProcess = spawn(process.execPath, ['--import', 'tsx', 'src/index.ts'], { + cwd: appDirectory, + env: { + ...process.env, + PORT: String(signalPort), + ORCA_RELAY_PUBLIC_URL: signalUrl, + ORCA_RELAY_CELL_URL: signalUrl, + ORCA_RELAY_AUTH_ISSUER: issuer, + ORCA_RELAY_JWKS_URL: `${issuer}/jwks`, + ORCA_RELAY_ASSIGNMENT_SIGNING_KEY: assignmentKey, + ORCA_RELAY_DATA_DIR: signalData, + ORCA_RELAY_ADMIN_AUDIENCE: adminAudience, + ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT: 'deploy@example.com', + ORCA_RELAY_ADMIN_JWKS_URL: `${issuer}/jwks` + }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + relayUrl = signalUrl + try { + await waitForRelay(signalProcess) + const host = await openHostControl() + const drain = nextMessage(host.socket) + const closed = new Promise((resolveClose) => + host.socket.once('close', (code) => resolveClose(code)) + ) + const exited = new Promise((resolveExit) => + signalProcess.once('exit', () => resolveExit()) + ) + signalProcess.kill('SIGTERM') + expect(await drain).toEqual({ + type: 'drain', + graceMs: 0, + recovery: 'resolve-director' + }) + expect(await closed).toBe(4503) + await exited + } finally { + if (signalProcess.exitCode === null) signalProcess.kill('SIGKILL') + relayUrl = originalUrl + rmSync(signalData, { recursive: true, force: true }) + } + }) + + it('authenticates admin drain with exact Google audience and deploy identity', async () => { + const host = await openHostControl() + const drainMessage = nextMessage(host.socket) + const closed = new Promise((resolveClose) => + host.socket.once('close', (code) => resolveClose(code)) + ) + const token = await new SignJWT({ email: 'deploy@example.com', email_verified: true }) + .setProtectedHeader({ alg: 'RS256', kid: 'admin-key' }) + .setIssuer('https://accounts.google.com') + .setAudience(adminAudience) + .setSubject('deploy-subject') + .setIssuedAt() + .setExpirationTime('5m') + .sign(adminPrivateKey) + const response = await fetch(`${relayUrl}/v1/admin/drain`, { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, graceMs: 0 }) + }) + expect(response.status).toBe(200) + expect(await drainMessage).toEqual({ + type: 'drain', + graceMs: 0, + recovery: 'resolve-director' + }) + expect(await closed).toBe(4503) + }) + + it('keeps dedicated capacity, monitor, and fence identities on exact routes', async () => { + const capacity = await googleServiceToken(adminAudience, 'capacity@example.com') + const monitor = await googleServiceToken(adminAudience, 'monitor@example.com') + const fence = await googleServiceToken(adminAudience, 'fence@example.com') + const broker = await googleServiceToken(adminAudience, 'broker@example.com') + const post = async (path: string, token: string, body: unknown): Promise => + await fetch(`${relayUrl}${path}`, { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body) + }) + + expect((await post('/v1/admin/runtime-status', capacity, { v: 1 })).status).toBe(200) + expect( + (await post('/v1/admin/evacuation-status', capacity, { + v: 1, + sourceCellId: 'source', + targetCellId: 'target', + completeReady: false + })).status + ).toBe(401) + expect( + (await post('/v1/admin/cell-fence-attempt-status', capacity, { + v: 1, + cellId: 'production-gce-c1' + })).status + ).toBe(401) + + expect((await post('/v1/admin/runtime-status', monitor, { v: 1 })).status).toBe(200) + expect((await post('/v1/admin/drain', monitor, { v: 1, graceMs: 0 })).status).toBe(401) + expect( + (await post('/v1/admin/cell-fence-attempt-status', monitor, { + v: 1, + cellId: 'production-gce-c1' + })).status + ).toBe(401) + + expect((await post('/v1/admin/runtime-status', fence, { v: 1 })).status).toBe(200) + expect((await post('/v1/admin/drain', fence, { v: 1, graceMs: 0 })).status).toBe(401) + expect( + (await post('/v1/admin/cell-fence-attempt-status', fence, { + v: 1, + cellId: 'production-gce-c1' + })).status + ).toBe(404) + expect( + (await post('/v1/admin/cell-fence-attest', fence, { + v: 1 + })).status + ).toBe(401) + + const directorPort = await unusedPort() + const directorUrl = `http://127.0.0.1:${directorPort}` + const directorData = mkdtempSync(resolve(tmpdir(), 'orca-relay-monitor-auth-')) + const director = spawnTopologyRelay({ + url: directorUrl, + dataDirectory: directorData, + role: 'director', + cellId: 'director', + cells: [{ id: 'cell-a', url: 'https://cell-a.example.com', capacityRequests: 10 }] + }) + try { + await waitForRelay(director) + const adoption = { + v: 1, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + confirmation: 'ADOPT_LEGACY_TERRAFORM_CELL_FENCE' + } + const postDirector = async (token: string): Promise => + await fetch(`${directorUrl}/v1/admin/cell-fence-adopt-legacy`, { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(adoption) + }) + expect( + (await postDirector(fence)).status + ).toBe(401) + expect( + (await postDirector(broker)).status + ).toBe(409) + const commitAdoption = { + v: 1, + cellId: 'cell-a', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + confirmation: 'COMMIT_LEGACY_TERRAFORM_CELL_FENCE_ADOPTION' + } + const postCommit = async (token: string): Promise => + await fetch(`${directorUrl}/v1/admin/cell-fence-commit-legacy-adoption`, { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(commitAdoption) + }) + expect((await postCommit(fence)).status).toBe(401) + expect((await postCommit(broker)).status).toBe(409) + const response = await fetch(`${directorUrl}/v1/admin/evacuation-status`, { + method: 'POST', + headers: { + authorization: `Bearer ${monitor}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + v: 1, + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + completeReady: true + }) + }) + expect(response.status).toBe(403) + const fenceResponse = await fetch(`${directorUrl}/v1/admin/evacuation-status`, { + method: 'POST', + headers: { + authorization: `Bearer ${fence}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + v: 1, + sourceCellId: 'cell-a', + targetCellId: 'cell-b', + completeReady: true + }) + }) + expect(fenceResponse.status).toBe(403) + } finally { + if (director.exitCode === null) director.kill('SIGKILL') + rmSync(directorData, { recursive: true, force: true }) + } + }) + + it('reports only authenticated aggregate runtime identity and the served digest', async () => { + const token = await adminToken() + expect( + await fetch(`${relayUrl}/v1/admin/runtime-status`, { + method: 'POST', + headers: { 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1 }) + }) + ).toMatchObject({ status: 401 }) + expect( + await fetch(`${relayUrl}/v1/admin/runtime-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, hostId: 'must-not-be-accepted' }) + }) + ).toMatchObject({ status: 400 }) + const response = await fetch(`${relayUrl}/v1/admin/runtime-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1 }) + }) + expect(response.status).toBe(200) + expect(await response.json()).toEqual({ + v: 1, + role: 'combined', + cellId: 'combined', + cellUrl: relayUrl, + region: 'us-central1', + imageDigest: `sha256:${'a'.repeat(64)}`, + draining: true, + regionalRehomeProtocol: 0, + connectionCapacity: null, + runtime: { + totalConnections: 0, + preAuthConnections: 0, + controls: 0, + splices: 0, + pendingSplices: 0, + queuedBytes: 0 + } + }) + }) +}) diff --git a/cloud/apps/relay/src/splice-forwarder.test.ts b/cloud/apps/relay/src/splice-forwarder.test.ts new file mode 100644 index 00000000000..3aa5e75fc97 --- /dev/null +++ b/cloud/apps/relay/src/splice-forwarder.test.ts @@ -0,0 +1,135 @@ +import { EventEmitter } from 'node:events' +import { describe, expect, it, vi } from 'vitest' +import type WebSocket from 'ws' +import { + ProcessQueuedByteBudget, + wireSplice, + type SpliceCloseInfo +} from './splice-forwarder.js' + +class FakeSocket extends EventEmitter { + readonly OPEN = 1 + readyState = 1 + bufferedAmount = 0 + sent: Array<{ data: unknown; binary: boolean }> = [] + closes: Array<{ code: number; reason: string }> = [] + _socket = { pause: vi.fn(), resume: vi.fn(), setNoDelay: vi.fn() } + + send(data: unknown, options: { binary: boolean }): void { + this.sent.push({ data, binary: options.binary }) + } + + close(code: number, reason: string): void { + this.readyState = 3 + this.closes.push({ code, reason }) + this.emit('close', code, Buffer.from(reason)) + } +} + +describe('splice forwarder', () => { + it('preserves text/binary opcodes and propagates peer close', () => { + const client = new FakeSocket() + const host = new FakeSocket() + const onClose = vi.fn() + const onForwardedBytes = vi.fn() + wireSplice({ + client: client as unknown as WebSocket, + host: host as unknown as WebSocket, + budget: new ProcessQueuedByteBudget(), + onClose, + onForwardedBytes + }) + client.emit('message', Buffer.from('text'), false) + client.emit('message', Buffer.from([1, 2]), true) + expect(host.sent).toEqual([ + { data: Buffer.from('text'), binary: false }, + { data: Buffer.from([1, 2]), binary: true } + ]) + expect(onForwardedBytes.mock.calls).toEqual([[4], [2]]) + client.emit('close', 1000, Buffer.alloc(0)) + expect(host.closes[0]?.code).toBe(4408) + expect(onClose).toHaveBeenCalledOnce() + }) + + it('hard-closes a wedged splice and releases its global queued-byte reservation', () => { + const client = new FakeSocket() + const host = new FakeSocket() + host.bufferedAmount = 1024 * 1024 + const budget = new ProcessQueuedByteBudget() + wireSplice({ + client: client as unknown as WebSocket, + host: host as unknown as WebSocket, + budget, + onClose: vi.fn() + }) + client.emit('message', Buffer.alloc(8 * 1024 * 1024), true) + expect(budget.current()).toBe(8 * 1024 * 1024) + client.emit('message', Buffer.alloc(512 * 1024), true) + expect(client.closes[0]?.code).toBe(4429) + expect(host.closes[0]?.code).toBe(4429) + expect(budget.current()).toBe(0) + }) + + it('reports the close trigger so limit and oversize kills are attributable', () => { + const wire = (): { client: FakeSocket; host: FakeSocket; closes: SpliceCloseInfo[] } => { + const client = new FakeSocket() + const host = new FakeSocket() + const closes: SpliceCloseInfo[] = [] + wireSplice({ + client: client as unknown as WebSocket, + host: host as unknown as WebSocket, + budget: new ProcessQueuedByteBudget(), + onClose: vi.fn(), + onClosed: (closeInfo) => closes.push(closeInfo) + }) + return { client, host, closes } + } + + const queueLimit = wire() + queueLimit.host.bufferedAmount = 1024 * 1024 + queueLimit.client.emit('message', Buffer.alloc(8 * 1024 * 1024), true) + queueLimit.client.emit('message', Buffer.alloc(512 * 1024), true) + expect(queueLimit.closes).toEqual([ + { code: 4429, reason: 'relay queue limit exceeded', trigger: 'queue-limit' } + ]) + + // The ws receiver error for a frame above maxPayload names the payload cap. + const oversize = wire() + oversize.host.emit('error', new RangeError('Max payload size exceeded')) + expect(oversize.closes[0]?.trigger).toBe('host-oversize-frame') + + const peerClose = wire() + peerClose.client.emit('close', 1001, Buffer.alloc(0)) + expect(peerClose.closes[0]?.trigger).toBe('client-closed') + expect(peerClose.closes).toHaveLength(1) + }) + + it('queues one full catalog-sized frame for a backpressured peer without closing', async () => { + // Why: the desktop's worktree catalog response exceeds 1MiB on large + // workspaces; a single maxFrameBytes frame must survive backpressure. + const client = new FakeSocket() + const host = new FakeSocket() + host.bufferedAmount = 1024 * 1024 + const budget = new ProcessQueuedByteBudget() + const onClose = vi.fn() + wireSplice({ + client: client as unknown as WebSocket, + host: host as unknown as WebSocket, + budget, + onClose + }) + client.emit('message', Buffer.alloc(8 * 1024 * 1024), true) + expect(client.closes).toEqual([]) + expect(host.closes).toEqual([]) + expect(budget.current()).toBe(8 * 1024 * 1024) + + // Peer drains; the queued frame flushes and the reservation releases. + host.bufferedAmount = 0 + await new Promise((resolve) => setTimeout(resolve, 60)) + expect(host.sent.some((frame) => (frame.data as Buffer).byteLength === 8 * 1024 * 1024)).toBe( + true + ) + expect(budget.current()).toBe(0) + expect(onClose).not.toHaveBeenCalled() + }) +}) diff --git a/cloud/apps/relay/src/splice-forwarder.ts b/cloud/apps/relay/src/splice-forwarder.ts new file mode 100644 index 00000000000..b0a31b49830 --- /dev/null +++ b/cloud/apps/relay/src/splice-forwarder.ts @@ -0,0 +1,158 @@ +import { RELAY_ADMISSION_BUDGETS, RELAY_CLOSE_CODE } from '@orca-cloud/relay-contract' +import type WebSocket from 'ws' +import type { RawData } from 'ws' +import { closeRelayWebSocket } from './relay-websocket-close.js' + +type QueuedFrame = { data: RawData; binary: boolean; bytes: number } + +export class ProcessQueuedByteBudget { + private queued = 0 + + reserve(bytes: number): boolean { + if (this.queued + bytes > RELAY_ADMISSION_BUDGETS.maxProcessQueuedBytes) return false + this.queued += bytes + return true + } + + release(bytes: number): void { + this.queued = Math.max(0, this.queued - bytes) + } + + current(): number { + return this.queued + } +} + +function frameBytes(data: RawData): number { + if (typeof data === 'string') return Buffer.byteLength(data) + if (Array.isArray(data)) return data.reduce((total, part) => total + part.byteLength, 0) + return data.byteLength +} + +function transport(socket: WebSocket): { pause(): void; resume(): void; setNoDelay(value: boolean): void } | null { + return ( + socket as WebSocket & { + _socket?: { pause(): void; resume(): void; setNoDelay(value: boolean): void } + } + )._socket ?? null +} + +export type SpliceCloseInfo = { code: number; reason: string; trigger: string } + +// The ws receiver kills a connection whose frame exceeds maxPayload with this +// message; surfacing it separately is what makes catalog-growth kills visible. +function errorTrigger(side: 'client' | 'host', error: Error): string { + return /max payload/i.test(error.message) ? `${side}-oversize-frame` : `${side}-error` +} + +export function wireSplice(input: { + client: WebSocket + host: WebSocket + budget: ProcessQueuedByteBudget + onClose: () => void + onForwardedBytes?: (bytes: number) => void + onClosed?: (closeInfo: SpliceCloseInfo) => void +}): (code?: number, reason?: string) => void { + let closed = false + const timers = new Set>() + const cleanups: Array<() => void> = [] + + const close = ( + code: number = RELAY_CLOSE_CODE.PEER_DROPPED, + reason = 'peer connection dropped', + trigger = 'external' + ): void => { + if (closed) return + closed = true + for (const timer of timers) clearTimeout(timer) + timers.clear() + for (const cleanup of cleanups) cleanup() + closeRelayWebSocket(input.client, code, reason) + closeRelayWebSocket(input.host, code, reason) + input.onClosed?.({ code, reason, trigger }) + input.onClose() + } + + const direction = (source: WebSocket, target: WebSocket): void => { + const queue: QueuedFrame[] = [] + let queuedBytes = 0 + let wedgedSince: number | null = null + cleanups.push(() => { + input.budget.release(queuedBytes) + queuedBytes = 0 + queue.length = 0 + transport(source)?.resume() + }) + + const flush = (): void => { + if (closed) return + if (target.readyState !== target.OPEN || source.readyState !== source.OPEN) { + close(undefined, undefined, 'peer-gone') + return + } + while ( + queue.length > 0 && + target.bufferedAmount <= RELAY_ADMISSION_BUDGETS.spliceLowWaterBytes + ) { + const frame = queue.shift()! + queuedBytes -= frame.bytes + input.budget.release(frame.bytes) + target.send(frame.data, { binary: frame.binary }) + } + if (queue.length === 0) { + wedgedSince = null + transport(source)?.resume() + return + } + if ( + wedgedSince !== null && + Date.now() - wedgedSince >= RELAY_ADMISSION_BUDGETS.spliceWedgedTimeoutMs + ) { + close(RELAY_CLOSE_CODE.LIMIT_EXCEEDED, 'wedged relay link', 'wedged') + return + } + const timer = setTimeout(() => { + timers.delete(timer) + flush() + }, 25) + timers.add(timer) + } + + source.on('message', (data, binary) => { + if (closed) return + const bytes = frameBytes(data) + if ( + queue.length === 0 && + target.readyState === target.OPEN && + target.bufferedAmount <= RELAY_ADMISSION_BUDGETS.spliceHighWaterBytes + ) { + target.send(data, { binary }) + input.onForwardedBytes?.(bytes) + return + } + if ( + queuedBytes + bytes > RELAY_ADMISSION_BUDGETS.spliceHardQueuedBytes || + !input.budget.reserve(bytes) + ) { + close(RELAY_CLOSE_CODE.LIMIT_EXCEEDED, 'relay queue limit exceeded', 'queue-limit') + return + } + queue.push({ data, binary, bytes }) + queuedBytes += bytes + input.onForwardedBytes?.(bytes) + wedgedSince ??= Date.now() + transport(source)?.pause() + if (queue.length === 1) flush() + }) + } + + transport(input.client)?.setNoDelay(true) + transport(input.host)?.setNoDelay(true) + direction(input.client, input.host) + direction(input.host, input.client) + input.client.once('close', () => close(undefined, undefined, 'client-closed')) + input.host.once('close', () => close(undefined, undefined, 'host-closed')) + input.client.once('error', (error) => close(undefined, undefined, errorTrigger('client', error))) + input.host.once('error', (error) => close(undefined, undefined, errorTrigger('host', error))) + return close +} diff --git a/cloud/apps/relay/src/staging-asia-proof-admission.test.ts b/cloud/apps/relay/src/staging-asia-proof-admission.test.ts new file mode 100644 index 00000000000..3bbee3ab9a3 --- /dev/null +++ b/cloud/apps/relay/src/staging-asia-proof-admission.test.ts @@ -0,0 +1,37 @@ +import { describe, expect, it } from 'vitest' +import { stagingAsiaProofMembership } from './staging-asia-proof-admission.js' + +describe('staging Asia proof admission', () => { + it('promotes only C4 and preserves every other cell', () => { + expect(stagingAsiaProofMembership({ + existingOnly: ['staging-gce-c1'], + migrationOnly: ['staging-gce-c4'], + general: ['staging-gce-c2'] + }, 'general')).toEqual({ + existingOnly: ['staging-gce-c1'], + migrationOnly: [], + general: ['staging-gce-c2', 'staging-gce-c4'] + }) + }) + + it('rolls back only C4', () => { + expect(stagingAsiaProofMembership({ + existingOnly: ['staging-gce-c1'], + migrationOnly: [], + general: ['staging-gce-c2', 'staging-gce-c4'] + }, 'migration-only')).toEqual({ + existingOnly: ['staging-gce-c1'], + migrationOnly: ['staging-gce-c4'], + general: ['staging-gce-c2'] + }) + }) + + it('rejects absent or irreversible C4 state', () => { + expect(() => stagingAsiaProofMembership({ + existingOnly: [], migrationOnly: [], general: ['staging-gce-c2'] + }, 'general')).toThrow('staging_asia_proof_cell_not_transitionable') + expect(() => stagingAsiaProofMembership({ + existingOnly: ['staging-gce-c4'], migrationOnly: [], general: [] + }, 'migration-only')).toThrow('staging_asia_proof_cell_not_transitionable') + }) +}) diff --git a/cloud/apps/relay/src/staging-asia-proof-admission.ts b/cloud/apps/relay/src/staging-asia-proof-admission.ts new file mode 100644 index 00000000000..a45a338768d --- /dev/null +++ b/cloud/apps/relay/src/staging-asia-proof-admission.ts @@ -0,0 +1,30 @@ +import type { CellAdmissionMembership } from './cell-admission-selector.js' + +export const STAGING_ASIA_PROOF_CELL_ID = 'staging-gce-c4' + +export type StagingAsiaProofAdmissionState = 'general' | 'migration-only' + +export function stagingAsiaProofMembership( + current: CellAdmissionMembership, + state: StagingAsiaProofAdmissionState +): CellAdmissionMembership { + const currentStates = [ + current.existingOnly.includes(STAGING_ASIA_PROOF_CELL_ID), + current.migrationOnly.includes(STAGING_ASIA_PROOF_CELL_ID), + current.general.includes(STAGING_ASIA_PROOF_CELL_ID) + ] + if (currentStates.filter(Boolean).length !== 1 || currentStates[0]) { + throw new Error('staging_asia_proof_cell_not_transitionable') + } + return { + existingOnly: [...current.existingOnly], + migrationOnly: current.migrationOnly + .filter((cellId) => cellId !== STAGING_ASIA_PROOF_CELL_ID) + .concat(state === 'migration-only' ? [STAGING_ASIA_PROOF_CELL_ID] : []) + .sort(), + general: current.general + .filter((cellId) => cellId !== STAGING_ASIA_PROOF_CELL_ID) + .concat(state === 'general' ? [STAGING_ASIA_PROOF_CELL_ID] : []) + .sort() + } +} diff --git a/cloud/apps/relay/tsconfig.build.json b/cloud/apps/relay/tsconfig.build.json new file mode 100644 index 00000000000..489ddfd34d6 --- /dev/null +++ b/cloud/apps/relay/tsconfig.build.json @@ -0,0 +1,10 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "declaration": true, + "noEmit": false, + "outDir": "dist", + "rootDir": "src" + }, + "exclude": ["src/**/*.test.ts"] +} diff --git a/cloud/apps/relay/tsconfig.json b/cloud/apps/relay/tsconfig.json new file mode 100644 index 00000000000..a552e34dbe9 --- /dev/null +++ b/cloud/apps/relay/tsconfig.json @@ -0,0 +1,5 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "noEmit": true }, + "include": ["src/**/*.ts"] +} diff --git a/cloud/apps/relay/vitest.config.ts b/cloud/apps/relay/vitest.config.ts new file mode 100644 index 00000000000..f56bbd7be3a --- /dev/null +++ b/cloud/apps/relay/vitest.config.ts @@ -0,0 +1,44 @@ +import { readdirSync, readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { defaultExclude, defineConfig } from 'vitest/config' + +// Every test file that opens ORCA_RELAY_TEST_POSTGRES_URL shares one CI +// database, and those tests take session-level locks with a 1s lock_timeout. +// Running them alongside anything else collides into lock timeouts, capacity +// exhaustion, and afterAll hangs. Keep them in their own serialized project so +// only they give up file parallelism; the rest of the suite opens SQLite data +// directories and stays fully parallel. +const sourceDirectory = fileURLToPath(new URL('src', import.meta.url)) +const sharedPostgresTests = readdirSync(sourceDirectory) + .filter((entry) => entry.endsWith('.test.ts')) + .filter((entry) => + readFileSync(`${sourceDirectory}/${entry}`, 'utf8').includes( + 'ORCA_RELAY_TEST_POSTGRES_URL' + ) + ) + .map((entry) => `src/${entry}`) + +const timeouts = { testTimeout: 15_000, hookTimeout: 15_000 } + +export default defineConfig({ + test: { + projects: [ + { + test: { + ...timeouts, + name: 'relay', + include: ['src/**/*.test.ts'], + exclude: [...defaultExclude, ...sharedPostgresTests] + } + }, + { + test: { + ...timeouts, + name: 'relay-postgres', + include: sharedPostgresTests, + fileParallelism: false + } + } + ] + } +}) diff --git a/cloud/dev/contracts/production-cloud-sql-app-consumers.json b/cloud/dev/contracts/production-cloud-sql-app-consumers.json new file mode 100644 index 00000000000..1cf4ba600dc --- /dev/null +++ b/cloud/dev/contracts/production-cloud-sql-app-consumers.json @@ -0,0 +1,15 @@ +{ + "comment": "Production Cloud SQL consumers owned by the private orca-cloud application tree (auth and API services). The relay ships without them, so the values the connection budget needs are published here; the private repository binds every field back to its source in its own CI.", + "authInstances": 2, + "authPoolMax": 10, + "apiInstances": 10, + "apiPoolMax": 5, + "maxConnections": 400, + "sources": { + "authInstances": "private apps tfvars: auth service max instances", + "authPoolMax": "private auth service: pg.Pool max", + "apiInstances": "private apps tfvars: API service max instances", + "apiPoolMax": "private API service: pg.Pool max", + "maxConnections": "Cloud SQL tier default; no max_connections flag is set" + } +} diff --git a/cloud/dev/fixtures/terraform-root-partition/families.json b/cloud/dev/fixtures/terraform-root-partition/families.json new file mode 100644 index 00000000000..9664c1eb299 --- /dev/null +++ b/cloud/dev/fixtures/terraform-root-partition/families.json @@ -0,0 +1,267 @@ +{ + "comment": "Terraform root ownership for every resource family declared under infra/terraform (relay), infra/terraform-foundation, and infra/terraform-apps. env_conditional families live in relay for production and in apps for staging (complementary counts). never_in_relay_state lists foundation/apps families that were introduced with their new root and so must NOT appear in infra/terraform/relay-root-carve-removed.tf. Generated by dev/scripts/terraform-root-partition.mjs --write; the test asserts this file matches the declarations.", + "foundation": [ + "google_artifact_registry_repository.api", + "google_iam_workload_identity_pool.github", + "google_project_service.required", + "google_project_service.sqladmin", + "google_service_account.runtime", + "google_sql_database_instance.auth", + "google_storage_bucket_iam_member.cloud_sql_rollout_lease", + "google_storage_bucket_iam_member.cloud_sql_rollout_lease_bucket_reader" + ], + "apps": [ + "cloudflare_record.artifact_onorca", + "cloudflare_record.artifact_usercontent", + "cloudflare_record.auth", + "google_artifact_registry_repository_iam_member.github_production_app_deploy_writer", + "google_cloud_run_domain_mapping.artifacts", + "google_cloud_run_domain_mapping.auth", + "google_cloud_run_v2_job.skill_storage_monitor", + "google_cloud_run_v2_job_iam_member.github_production_app_skill_monitor_developer", + "google_cloud_run_v2_job_iam_member.skill_storage_scheduler_invoker", + "google_cloud_run_v2_service.api", + "google_cloud_run_v2_service.auth", + "google_cloud_run_v2_service_iam_member.github_production_app_api_developer", + "google_cloud_run_v2_service_iam_member.github_production_app_auth_developer", + "google_cloud_scheduler_job.skill_storage_monitor", + "google_iam_workload_identity_pool_provider.github_production_app_deploy", + "google_logging_metric.skill_api", + "google_logging_metric.skill_archive_rejection", + "google_logging_metric.skill_database", + "google_logging_metric.skill_digest_mismatch", + "google_logging_metric.skill_finalize_saturation", + "google_logging_metric.skill_route_latency", + "google_logging_metric.skill_signing_failure", + "google_logging_metric.skill_storage_inventory", + "google_logging_metric.skill_storage_overdue_quarantine", + "google_logging_project_exclusion.skill_share_bearer_request_urls", + "google_monitoring_alert_policy.skill_api_5xx", + "google_monitoring_alert_policy.skill_cloud_run_resource_pressure", + "google_monitoring_alert_policy.skill_database_migration_failure", + "google_monitoring_alert_policy.skill_digest_mismatch", + "google_monitoring_alert_policy.skill_finalize_failures", + "google_monitoring_alert_policy.skill_route_latency", + "google_monitoring_alert_policy.skill_signing_failure", + "google_monitoring_alert_policy.skill_storage_lifecycle_failure", + "google_monitoring_dashboard.skill_sharing", + "google_project_iam_custom_role.skill_storage_inventory", + "google_project_iam_member.auth_runtime_cloudsql_client", + "google_project_iam_member.github_artifact_writer", + "google_project_iam_member.github_cloud_run_developer", + "google_project_iam_member.runtime_skills_cloudsql_client", + "google_secret_manager_secret.auth", + "google_secret_manager_secret.auth_database_url", + "google_secret_manager_secret.skills_database_url", + "google_secret_manager_secret_iam_member.auth_database_url_accessor", + "google_secret_manager_secret_iam_member.auth_runtime_accessor", + "google_secret_manager_secret_iam_member.runtime_skills_database_url_accessor", + "google_secret_manager_secret_version.auth_database_url", + "google_secret_manager_secret_version.skills_database_url", + "google_service_account.auth_runtime", + "google_service_account.github_production_app_deploy", + "google_service_account.skill_storage_monitor", + "google_service_account.skill_storage_scheduler", + "google_service_account_iam_member.github_auth_runtime_service_account_user", + "google_service_account_iam_member.github_production_app_auth_runtime_user", + "google_service_account_iam_member.github_production_app_deploy_workload_identity_user", + "google_service_account_iam_member.github_production_app_runtime_user", + "google_service_account_iam_member.github_production_app_skill_monitor_user", + "google_service_account_iam_member.github_runtime_service_account_user", + "google_service_account_iam_member.github_skill_storage_monitor_user", + "google_service_account_iam_member.runtime_skill_package_signer", + "google_sql_database.auth", + "google_sql_database.skills", + "google_sql_user.auth", + "google_sql_user.skills", + "google_storage_bucket.artifacts", + "google_storage_bucket.skill_packages", + "google_storage_bucket_iam_member.runtime_artifact_object_admin", + "google_storage_bucket_iam_member.runtime_skill_package_object_user", + "google_storage_bucket_iam_member.skill_storage_monitor_inventory", + "random_password.auth_database", + "random_password.skills_database", + "time_sleep.skill_metric_descriptor_propagation" + ], + "relay": [ + "google_artifact_registry_repository_iam_member.github_production_relay_staging_mirror_writer", + "google_artifact_registry_repository_iam_member.github_production_relay_writer", + "google_artifact_registry_repository_iam_member.github_relay_asia_topology_artifact_reader", + "google_artifact_registry_repository_iam_member.github_staging_relay_deploy_artifact_reader", + "google_certificate_manager_certificate.relay_gce", + "google_certificate_manager_certificate_map.relay_gce", + "google_certificate_manager_certificate_map_entry.relay_gce", + "google_certificate_manager_dns_authorization.relay_gce", + "google_cloud_run_domain_mapping.relay", + "google_cloud_run_domain_mapping.relay_cell", + "google_cloud_run_v2_service.relay", + "google_cloud_run_v2_service.relay_cell", + "google_cloud_run_v2_service.relay_fence_broker", + "google_cloud_run_v2_service_iam_member.github_production_relay_director_developer", + "google_cloud_run_v2_service_iam_member.github_production_relay_fence_broker_developer", + "google_cloud_run_v2_service_iam_member.github_staging_relay_capacity_developer", + "google_cloud_run_v2_service_iam_member.github_staging_relay_deploy_auth_developer", + "google_cloud_run_v2_service_iam_member.github_staging_relay_deploy_director_developer", + "google_cloud_run_v2_service_iam_member.relay_fence_broker_deploy_invoker", + "google_cloud_run_v2_service_iam_member.relay_fence_broker_invoker", + "google_compute_backend_service.relay_gce_cell", + "google_compute_firewall.relay_gce_iap_ssh", + "google_compute_firewall.relay_gce_load_balancer", + "google_compute_global_address.relay_gce", + "google_compute_global_forwarding_rule.relay_gce", + "google_compute_health_check.relay_gce_liveness", + "google_compute_health_check.relay_gce_readiness", + "google_compute_instance_group_manager.relay_gce_cell", + "google_compute_instance_template.relay_gce_cell", + "google_compute_network.relay_gce", + "google_compute_router.relay_gce", + "google_compute_router.relay_gce_additional", + "google_compute_router_nat.relay_gce", + "google_compute_router_nat.relay_gce_additional", + "google_compute_subnetwork.relay_gce", + "google_compute_subnetwork.relay_gce_additional", + "google_compute_target_https_proxy.relay_gce", + "google_compute_url_map.relay_gce", + "google_iam_workload_identity_pool_provider.github_fence", + "google_iam_workload_identity_pool_provider.github_monitor", + "google_iam_workload_identity_pool_provider.github_production_relay_capacity", + "google_iam_workload_identity_pool_provider.github_relay_asia_proof", + "google_iam_workload_identity_pool_provider.github_relay_asia_topology", + "google_iam_workload_identity_pool_provider.github_staging_relay_capacity", + "google_iam_workload_identity_pool_provider.github_staging_relay_deploy", + "google_logging_metric.relay_incident", + "google_logging_metric.relay_snapshot", + "google_monitoring_alert_policy.relay_assignment_5xx", + "google_monitoring_alert_policy.relay_assignment_edge_429", + "google_monitoring_alert_policy.relay_cloud_sql_backends", + "google_monitoring_alert_policy.relay_custom", + "google_monitoring_alert_policy.relay_gce_connection_headroom", + "google_monitoring_alert_policy.relay_postgres_retry_exhausted", + "google_project_iam_custom_role.github_production_relay_capacity_mutation", + "google_project_iam_custom_role.github_relay_asia_topology_mutation", + "google_project_iam_custom_role.github_relay_asia_topology_read", + "google_project_iam_custom_role.github_relay_asia_topology_state_list", + "google_project_iam_custom_role.github_staging_relay_capacity_mutation", + "google_project_iam_custom_role.github_staging_relay_power", + "google_project_iam_custom_role.relay_fence_broker_mutation", + "google_project_iam_member.github_fence_cloudsql_viewer", + "google_project_iam_member.github_fence_compute_viewer", + "google_project_iam_member.github_fence_logging_viewer", + "google_project_iam_member.github_fence_monitoring_viewer", + "google_project_iam_member.github_monitor_compute_viewer", + "google_project_iam_member.github_production_relay_capacity_artifact_reader", + "google_project_iam_member.github_production_relay_capacity_mutation", + "google_project_iam_member.github_production_relay_capacity_viewer", + "google_project_iam_member.github_relay_asia_proof_logging_viewer", + "google_project_iam_member.github_relay_asia_proof_monitoring_viewer", + "google_project_iam_member.github_relay_asia_topology_mutation", + "google_project_iam_member.github_relay_asia_topology_read", + "google_project_iam_member.github_relay_monitor_cloudsql_viewer", + "google_project_iam_member.github_relay_monitor_logging_viewer", + "google_project_iam_member.github_relay_monitor_monitoring_viewer", + "google_project_iam_member.github_staging_relay_capacity_artifact_reader", + "google_project_iam_member.github_staging_relay_capacity_mutation", + "google_project_iam_member.github_staging_relay_capacity_viewer", + "google_project_iam_member.github_staging_relay_deploy_compute_viewer", + "google_project_iam_member.github_staging_relay_power", + "google_project_iam_member.relay_director_runtime_cloudsql_client", + "google_project_iam_member.relay_fence_broker_artifact_reader", + "google_project_iam_member.relay_fence_broker_compute_viewer", + "google_project_iam_member.relay_fence_broker_logging_viewer", + "google_project_iam_member.relay_fence_broker_mutation", + "google_project_iam_member.relay_runtime_artifact_reader", + "google_project_iam_member.relay_runtime_cloudsql_client", + "google_project_iam_member.relay_runtime_log_writer", + "google_secret_manager_secret.relay_assignment_signing_key", + "google_secret_manager_secret.relay_database_url", + "google_secret_manager_secret.relay_regional_placement_enabled", + "google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor", + "google_secret_manager_secret_iam_member.relay_assignment_signing_key_director_accessor", + "google_secret_manager_secret_iam_member.relay_database_url_accessor", + "google_secret_manager_secret_iam_member.relay_database_url_director_accessor", + "google_secret_manager_secret_iam_member.relay_regional_placement_deploy_accessor", + "google_secret_manager_secret_iam_member.relay_regional_placement_deploy_adder", + "google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer", + "google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor", + "google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor", + "google_secret_manager_secret_version.relay_assignment_signing_key", + "google_secret_manager_secret_version.relay_database_url", + "google_secret_manager_secret_version.relay_regional_placement_enabled", + "google_service_account.github_fence", + "google_service_account.github_monitor", + "google_service_account.github_production_relay_capacity", + "google_service_account.github_relay_asia_proof", + "google_service_account.github_relay_asia_topology", + "google_service_account.github_staging_relay_capacity", + "google_service_account.github_staging_relay_deploy", + "google_service_account.relay_director_runtime", + "google_service_account.relay_fence_broker", + "google_service_account.relay_runtime", + "google_service_account_iam_member.github_accepted_repository_workload_identity_user", + "google_service_account_iam_member.github_fence_workload_identity_user", + "google_service_account_iam_member.github_monitor_workload_identity_user", + "google_service_account_iam_member.github_production_relay_capacity_runtime_user", + "google_service_account_iam_member.github_production_relay_capacity_workload_identity_user", + "google_service_account_iam_member.github_relay_asia_proof_workload_identity_user", + "google_service_account_iam_member.github_relay_asia_topology_runtime_user", + "google_service_account_iam_member.github_relay_asia_topology_workload_identity_user", + "google_service_account_iam_member.github_relay_director_runtime_service_account_user", + "google_service_account_iam_member.github_relay_fence_broker_service_account_user", + "google_service_account_iam_member.github_relay_runtime_service_account_user", + "google_service_account_iam_member.github_staging_relay_capacity_runtime_user", + "google_service_account_iam_member.github_staging_relay_capacity_workload_identity_user", + "google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user", + "google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user", + "google_service_account_iam_member.relay_fence_broker_requester_token_creator", + "google_sql_database.relay", + "google_sql_user.relay", + "google_storage_bucket_iam_member.github_production_relay_capacity_state", + "google_storage_bucket_iam_member.github_relay_asia_topology_state", + "google_storage_bucket_iam_member.github_relay_asia_topology_state_list", + "google_storage_bucket_iam_member.github_staging_relay_capacity_state", + "google_storage_bucket_iam_member.github_staging_relay_deploy_state", + "google_storage_bucket_iam_member.github_staging_relay_deploy_state_list", + "google_storage_bucket_iam_member.relay_fence_broker_bucket_reader", + "google_storage_bucket_iam_member.relay_fence_broker_state_objects", + "random_password.relay_assignment_signing_key", + "random_password.relay_database" + ], + "env_conditional": { + "google_iam_workload_identity_pool_provider.github": { + "production": "relay", + "staging": "apps" + }, + "google_project_iam_member.github_cloudsql_viewer": { + "production": "relay", + "staging": "apps" + }, + "google_project_iam_member.github_compute_viewer": { + "production": "relay", + "staging": "apps" + }, + "google_project_iam_member.github_logging_viewer": { + "production": "relay", + "staging": "apps" + }, + "google_project_iam_member.github_monitoring_viewer": { + "production": "relay", + "staging": "apps" + }, + "google_service_account.github_deploy": { + "production": "relay", + "staging": "apps" + }, + "google_service_account_iam_member.github_workload_identity_user": { + "production": "relay", + "staging": "apps" + }, + "google_storage_bucket_iam_member.github_terraform_state_reader": { + "production": "relay", + "staging": "apps" + } + }, + "state_orphans": { + "staging": [], + "production": [] + } +} diff --git a/cloud/dev/scripts/capture-terraform-plan-baseline.mjs b/cloud/dev/scripts/capture-terraform-plan-baseline.mjs new file mode 100644 index 00000000000..734d67536c2 --- /dev/null +++ b/cloud/dev/scripts/capture-terraform-plan-baseline.mjs @@ -0,0 +1,83 @@ +#!/usr/bin/env node +import { execFileSync } from 'node:child_process' +import { mkdirSync, readFileSync, writeFileSync } from 'node:fs' +import { join } from 'node:path' + +// Why: the relay root can never plan zero-diff (cell templates roll one at a time by design), +// so the split is gated on plan EQUIVALENCE: the normalized change set for a root's addresses +// must be identical before and after a state move. This captures that set deterministically. +// Read-only: -lock=false, -refresh=false, no apply. Forgets from `removed` blocks are excluded +// because the baseline has none. + +const usage = + 'usage: capture-terraform-plan-baseline.mjs --root --env --out [--tag ]' + +export function normalizePlan(planJson) { + const changes = (planJson.resource_changes ?? []) + .filter((entry) => !entry.change.actions.includes('forget')) + .map((entry) => ({ + address: entry.address, + actions: entry.change.actions, + before: entry.change.before ?? null, + after: entry.change.after ?? null, + after_unknown: entry.change.after_unknown ?? null + })) + .sort((left, right) => (left.address < right.address ? -1 : left.address > right.address ? 1 : 0)) + return changes +} + +export function summarize(changes) { + const counts = { create: 0, update: 0, delete: 0, replace: 0, 'no-op': 0, read: 0 } + for (const change of changes) { + const key = change.actions.join('-') + if (key === 'create') counts.create += 1 + else if (key === 'update') counts.update += 1 + else if (key === 'delete') counts.delete += 1 + else if (key === 'delete-create' || key === 'create-delete') counts.replace += 1 + else if (key === 'read') counts.read += 1 + else counts['no-op'] += 1 + } + return counts +} + +function argument(flag) { + const index = process.argv.indexOf(flag) + return index >= 0 ? process.argv[index + 1] : undefined +} + +if (process.argv[1] && import.meta.url.endsWith(process.argv[1].split('/').pop())) { + const root = argument('--root') + const environment = argument('--env') + const out = argument('--out') + const tag = argument('--tag') ?? `${environment}-${root.replaceAll('/', '_')}` + if (!root || !['staging', 'production'].includes(environment) || !out) { + process.stderr.write(`${usage}\n`) + process.exit(2) + } + // The Cloudflare override only applies to the root that still declares the records; the relay + // root dropped them in the carve and errors on a -var for an undeclared variable. + const declaresArtifactDns = readFileSync(join(root, 'variables.tf'), 'utf8').includes( + 'variable "manage_artifact_dns"' + ) + mkdirSync(out, { recursive: true }) + const planFile = join(out, `${tag}.tfplan`) + execFileSync( + 'terraform', + [ + `-chdir=${root}`, 'plan', '-input=false', '-lock=false', '-refresh=false', '-no-color', + `-var-file=environments/${environment}.tfvars`, + ...(declaresArtifactDns ? ['-var', 'manage_artifact_dns=false'] : []), + `-out=${planFile}` + ], + { stdio: ['ignore', 'inherit', 'inherit'] } + ) + const json = JSON.parse( + execFileSync('terraform', [`-chdir=${root}`, 'show', '-json', planFile], { + encoding: 'utf8', + maxBuffer: 256 * 1024 * 1024 + }) + ) + const normalized = normalizePlan(json) + writeFileSync(join(out, `${tag}.norm.json`), `${JSON.stringify(normalized, null, 1)}\n`) + process.stdout.write(`${tag}: ${JSON.stringify(summarize(normalized))}\n`) +} diff --git a/cloud/dev/scripts/capture-terraform-plan-baseline.test.mjs b/cloud/dev/scripts/capture-terraform-plan-baseline.test.mjs new file mode 100644 index 00000000000..b003f3db773 --- /dev/null +++ b/cloud/dev/scripts/capture-terraform-plan-baseline.test.mjs @@ -0,0 +1,15 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { normalizePlan, summarize } from './capture-terraform-plan-baseline.mjs' + +test('normalizes, sorts, and drops forgets so a removed block cannot skew equivalence', () => { + const changes = normalizePlan({ + resource_changes: [ + { address: 'b.two', change: { actions: ['update'], before: { x: 1 }, after: { x: 2 } } }, + { address: 'a.one', change: { actions: ['forget'], before: {}, after: null } }, + { address: 'c.three', change: { actions: ['delete', 'create'], before: {}, after: {}, after_unknown: { id: true } } } + ] + }) + assert.deepEqual(changes.map((change) => change.address), ['b.two', 'c.three']) + assert.deepEqual(summarize(changes), { create: 0, update: 1, delete: 0, replace: 1, 'no-op': 0, read: 0 }) +}) diff --git a/cloud/dev/scripts/classify-relay-production-capacity-director.mjs b/cloud/dev/scripts/classify-relay-production-capacity-director.mjs new file mode 100644 index 00000000000..e24bac7b6e3 --- /dev/null +++ b/cloud/dev/scripts/classify-relay-production-capacity-director.mjs @@ -0,0 +1,109 @@ +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' +import { isDeepStrictEqual } from 'node:util' + +function parseArguments(argv) { + if (argv.length !== 2 || argv[0] !== '--capacity-service-account' || !argv[1]) { + throw new Error('missing --capacity-service-account') + } + return argv[1] +} + +export function classifyProductionCapacityDirector(state, capacityServiceAccount) { + const currentIdentity = state.currentCapacityServiceAccount + if (currentIdentity !== null && currentIdentity !== capacityServiceAccount) { + throw new Error('director has an unexpected capacity identity') + } + const { + baseCells, + currentCells, + capacityCellIds, + targetCellId, + targetHardCap + } = state + if ( + !Array.isArray(baseCells) || + !Array.isArray(currentCells) || + !Array.isArray(capacityCellIds) || + ![600, 1000].includes(targetHardCap) || + new Set(capacityCellIds).size !== capacityCellIds.length || + !capacityCellIds.includes(targetCellId) || + baseCells.length !== currentCells.length || + new Set(baseCells.map((cell) => cell?.id)).size !== baseCells.length || + new Set(currentCells.map((cell) => cell?.id)).size !== currentCells.length + ) { + throw new Error('director topology transition input is invalid') + } + const capacityCells = new Set(capacityCellIds) + const normalizedCurrent = currentCells.map((current, index) => { + const base = baseCells[index] + if ( + typeof current?.id !== 'string' || + current.id !== base?.id + ) { + throw new Error('director topology cell identity is invalid') + } + if (!capacityCells.has(current.id)) { + if (!isDeepStrictEqual(current, base)) { + throw new Error('director topology changed outside the capacity rollout') + } + return current + } + if ( + base.connectionHardCap !== 1000 || + base.connectionUnobservedBound !== 60 || + ![600, 1000].includes(current.connectionHardCap) || + current.connectionUnobservedBound !== 60 + ) { + throw new Error('director capacity rollout state is invalid') + } + return { + ...current, + connectionHardCap: base.connectionHardCap, + connectionUnobservedBound: base.connectionUnobservedBound + } + }) + if ( + !isDeepStrictEqual(normalizedCurrent, baseCells) || + capacityCellIds.some((cellId) => !baseCells.some((cell) => cell.id === cellId)) + ) { + throw new Error('director topology is outside the reviewed capacity envelope') + } + const withTargetCap = (hardCap) => currentCells.map((cell) => + cell.id === targetCellId + ? { ...cell, connectionHardCap: hardCap, connectionUnobservedBound: 60 } + : cell + ) + const desiredCells = withTargetCap(targetHardCap) + const predecessorCells = withTargetCap(targetHardCap === 600 ? 1000 : 600) + const topologyPhase = isDeepStrictEqual(currentCells, desiredCells) + ? 'desired' + : isDeepStrictEqual(currentCells, predecessorCells) + ? 'predecessor' + : null + if (!topologyPhase) throw new Error('director topology is not a reviewed transition state') + return { + topologyPhase, + directorReady: + topologyPhase === 'desired' && currentIdentity === capacityServiceAccount, + desiredCells + } +} + +export function main(argv = process.argv.slice(2)) { + const capacityServiceAccount = parseArguments(argv) + const state = JSON.parse(readFileSync(0, 'utf8')) + process.stdout.write(`${JSON.stringify({ + event: 'relay_production_capacity_director_classified', + ...classifyProductionCapacityDirector(state, capacityServiceAccount) + })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + try { + main() + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + } +} diff --git a/cloud/dev/scripts/classify-relay-production-capacity-director.test.mjs b/cloud/dev/scripts/classify-relay-production-capacity-director.test.mjs new file mode 100644 index 00000000000..13b188ac042 --- /dev/null +++ b/cloud/dev/scripts/classify-relay-production-capacity-director.test.mjs @@ -0,0 +1,93 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + classifyProductionCapacityDirector +} from './classify-relay-production-capacity-director.mjs' + +const capacityServiceAccount = + 'orca-cloud-gha-relay-cap@onorca-cloud.iam.gserviceaccount.com' +const capacityCellIds = ['production-gce-c25', 'production-gce-c26'] +const baseCells = [ + { id: 'production-gce-c17', connectionHardCap: 600, connectionUnobservedBound: 60 }, + { id: 'production-gce-c25', connectionHardCap: 1000, connectionUnobservedBound: 60 }, + { id: 'production-gce-c26', connectionHardCap: 1000, connectionUnobservedBound: 60 } +] +const mixedCells = [ + baseCells[0], + { ...baseCells[1], connectionHardCap: 600 }, + baseCells[2] +] + +function state(overrides = {}) { + return { + baseCells, + currentCells: mixedCells, + capacityCellIds, + targetCellId: 'production-gce-c25', + targetHardCap: 1000, + currentCapacityServiceAccount: capacityServiceAccount, + ...overrides + } +} + +test('classifies one target while preserving completed rollout cells', () => { + assert.deepEqual( + classifyProductionCapacityDirector(state(), capacityServiceAccount), + { + topologyPhase: 'predecessor', + directorReady: false, + desiredCells: baseCells + } + ) +}) + +test('skips deployment only for exact topology and identity', () => { + assert.deepEqual( + classifyProductionCapacityDirector(state({ currentCells: baseCells }), capacityServiceAccount), + { topologyPhase: 'desired', directorReady: true, desiredCells: baseCells } + ) + assert.deepEqual( + classifyProductionCapacityDirector(state({ + currentCells: baseCells, + targetHardCap: 600 + }), capacityServiceAccount), + { + topologyPhase: 'predecessor', + directorReady: false, + desiredCells: mixedCells + } + ) +}) + +test('rejects an unknown topology or capacity identity', () => { + assert.throws( + () => classifyProductionCapacityDirector(state({ + currentCells: [baseCells[0], { ...baseCells[1], connectionHardCap: 700 }, baseCells[2]] + }), capacityServiceAccount), + /rollout state is invalid/ + ) + assert.throws( + () => classifyProductionCapacityDirector(state({ + currentCapacityServiceAccount: 'unexpected@onorca-cloud.iam.gserviceaccount.com' + }), capacityServiceAccount), + /unexpected capacity identity/ + ) + assert.throws( + () => classifyProductionCapacityDirector(state({ + currentCells: [{ ...baseCells[0], connectionHardCap: 1000 }, mixedCells[1], mixedCells[2]] + }), capacityServiceAccount), + /outside the capacity rollout/ + ) + assert.throws( + () => classifyProductionCapacityDirector(state({ + currentCells: [baseCells[0], { ...mixedCells[1], url: 'https://wrong.invalid' }, baseCells[2]] + }), capacityServiceAccount), + /outside the reviewed capacity envelope/ + ) + assert.throws( + () => classifyProductionCapacityDirector(state({ + targetCellId: 'production-gce-c17' + }), capacityServiceAccount), + /transition input is invalid/ + ) +}) diff --git a/cloud/dev/scripts/classify-relay-staging-bootstrap.mjs b/cloud/dev/scripts/classify-relay-staging-bootstrap.mjs new file mode 100644 index 00000000000..fd50b332c57 --- /dev/null +++ b/cloud/dev/scripts/classify-relay-staging-bootstrap.mjs @@ -0,0 +1,47 @@ +import { pathToFileURL } from 'node:url' + +export function classifyStagingBootstrap({ c2Kind, c2Admission, c3Kind, c3Admission }) { + for (const kind of [c2Kind, c3Kind]) { + if (!['legacy', 'modern'].includes(kind)) throw new Error('bootstrap runtime kind is invalid') + } + for (const admission of [c2Admission, c3Admission]) { + if (!['general', 'migration-only'].includes(admission)) { + throw new Error('bootstrap admission is not recoverable') + } + } + if (c2Kind === 'modern' && c3Kind === 'modern') return 'complete' + if (c2Kind === 'modern') return 'roll-c3' + if (c3Kind === 'modern') return 'roll-c2' + if (c2Admission === 'general') return 'normalize-and-roll-both' + if (c3Admission === 'general') return 'resume-c2-then-c3' + throw new Error('bootstrap has no general fallback') +} + +function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + return { + c2Kind: values['c2-kind'], + c2Admission: values['c2-admission'], + c3Kind: values['c3-kind'], + c3Admission: values['c3-admission'] + } +} + +export function main(argv = process.argv.slice(2)) { + process.stdout.write(`${classifyStagingBootstrap(parseArguments(argv))}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + try { + main() + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + } +} diff --git a/cloud/dev/scripts/classify-relay-staging-bootstrap.test.mjs b/cloud/dev/scripts/classify-relay-staging-bootstrap.test.mjs new file mode 100644 index 00000000000..66caa4fde4c --- /dev/null +++ b/cloud/dev/scripts/classify-relay-staging-bootstrap.test.mjs @@ -0,0 +1,60 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { classifyStagingBootstrap } from './classify-relay-staging-bootstrap.mjs' + +const state = (c2Kind, c2Admission, c3Kind, c3Admission) => ({ + c2Kind, + c2Admission, + c3Kind, + c3Admission +}) + +test('classifies the fresh legacy bootstrap', () => { + assert.equal( + classifyStagingBootstrap(state('legacy', 'general', 'legacy', 'general')), + 'normalize-and-roll-both' + ) +}) + +test('retries normalization after a failed C3 restart', () => { + assert.equal( + classifyStagingBootstrap(state('legacy', 'general', 'legacy', 'migration-only')), + 'normalize-and-roll-both' + ) +}) + +test('resumes after C2 isolation', () => { + assert.equal( + classifyStagingBootstrap(state('legacy', 'migration-only', 'legacy', 'general')), + 'resume-c2-then-c3' + ) +}) + +test('resumes after C2 apply or a partial C3 transition', () => { + for (const c2Admission of ['general', 'migration-only']) { + for (const c3Admission of ['general', 'migration-only']) { + assert.equal( + classifyStagingBootstrap(state('modern', c2Admission, 'legacy', c3Admission)), + 'roll-c3' + ) + } + } +}) + +test('repairs an unexpected modern C3 before rolling legacy C2', () => { + assert.equal( + classifyStagingBootstrap(state('legacy', 'migration-only', 'modern', 'general')), + 'roll-c2' + ) +}) + +test('accepts already complete modern cells and rejects no-fallback legacy state', () => { + assert.equal( + classifyStagingBootstrap(state('modern', 'migration-only', 'modern', 'general')), + 'complete' + ) + assert.throws( + () => classifyStagingBootstrap(state('legacy', 'migration-only', 'legacy', 'migration-only')), + /no general fallback/ + ) +}) diff --git a/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs b/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs new file mode 100644 index 00000000000..76193746f2c --- /dev/null +++ b/cloud/dev/scripts/cloud-sql-rollout-lock-census.mjs @@ -0,0 +1,305 @@ +// Derives, from workflow and script content, which workflows roll out against the shared Cloud SQL +// instance. Hand lists go stale silently; everything here is read back off disk. +import { readFileSync, readdirSync } from 'node:fs' +import { + RELAY_WORKFLOW_DIRECTORY, + RELAY_WORKFLOW_FILE_PREFIX, + relayWorkflowFile +} from './relay-repository.mjs' + +export const WORKFLOW_ROOT = RELAY_WORKFLOW_DIRECTORY +export const SCRIPT_ROOT = new URL('./', import.meta.url) + +export const LEASE_ACTION = './.github/actions/cloud-sql-rollout-lease' +export const PRODUCTION_LEASE = { + bucket: 'onorca-cloud-terraform-state', + object: 'terraform/state/cloud-sql-rollout/production.lock' +} +export const STAGING_LEASE = { + bucket: 'onorca-cloud-staging-terraform-state', + object: 'terraform/state/cloud-sql-rollout/staging.lock' +} + +export const PRODUCTION_GROUP = 'production-cloud-sql-rollout' +export const STAGING_GROUP = 'relay-staging-mutation' +export const SELECTABLE_GROUP = + "${{ inputs.environment == 'production' && 'production-cloud-sql-rollout' || 'relay-staging-mutation' }}" + +const selectable = (production, staging) => + `\${{ inputs.environment == 'production' && '${production}' || '${staging}' }}` + +export const SELECTABLE_LEASE = { + bucket: selectable(PRODUCTION_LEASE.bucket, STAGING_LEASE.bucket), + object: selectable(PRODUCTION_LEASE.object, STAGING_LEASE.object) +} + +export const LOCK_GROUPS = new Set([PRODUCTION_GROUP, STAGING_GROUP, SELECTABLE_GROUP]) + +export function readWorkflow(file) { + return readFileSync(new URL(file, WORKFLOW_ROOT), 'utf8') +} + +export function workflowFiles() { + return readdirSync(WORKFLOW_ROOT) + .filter((name) => name.endsWith('.yml') && name.startsWith(RELAY_WORKFLOW_FILE_PREFIX)) + .sort() +} + +// --- YAML-shaped readers (line based; the workflows are hand-written and uniformly indented) --- + +function indentOf(line) { + return line.length - line.trimStart().length +} + +function blockAfter(lines, index) { + const base = indentOf(lines[index]) + const body = [] + for (let i = index + 1; i < lines.length; i += 1) { + if (lines[i].trim() === '') { + body.push(lines[i]) + continue + } + if (indentOf(lines[i]) <= base) break + body.push(lines[i]) + } + return body +} + +export function concurrencyBlocks(text) { + const lines = text.split('\n') + const blocks = [] + lines.forEach((line, index) => { + if (line.trim() !== 'concurrency:') return + const body = blockAfter(lines, index) + blocks.push({ + group: body.find((l) => l.trim().startsWith('group:'))?.trim().slice('group:'.length).trim(), + cancelInProgress: body + .find((l) => l.trim().startsWith('cancel-in-progress:')) + ?.trim() + .slice('cancel-in-progress:'.length) + .trim() + }) + }) + return blocks +} + +export function jobs(text) { + const lines = text.split('\n') + const start = lines.findIndex((line) => line === 'jobs:') + if (start === -1) return [] + const found = [] + for (let i = start + 1; i < lines.length; i += 1) { + const match = /^ {2}([A-Za-z0-9_-]+):\s*$/.exec(lines[i]) + if (!match) continue + found.push({ id: match[1], start: i, body: blockAfter(lines, i) }) + } + return found.map((job) => ({ ...job, text: job.body.join('\n') })) +} + +function scalarField(jobText, key) { + const lines = jobText.split('\n') + const index = lines.findIndex((line) => /^ {4}[A-Za-z-]+:/.test(line) && line.trim().startsWith(`${key}:`)) + if (index === -1) return undefined + const inline = lines[index].trim().slice(`${key}:`.length).trim() + if (inline !== '' && inline !== '>-' && inline !== '|') return inline + return blockAfter(lines, index).join(' ').replace(/\s+/g, ' ').trim() +} + +export function jobNeeds(jobText) { + const raw = scalarField(jobText, 'needs') + if (!raw) return [] + return raw + .replace(/^\[|\]$/g, '') + .split(/[,\n]|\s+-\s+/) + .map((entry) => entry.replace(/^-/, '').trim()) + .filter(Boolean) +} + +export function jobIf(jobText) { + return scalarField(jobText, 'if') ?? '' +} + +export function leaseSteps(text) { + const lines = text.split('\n') + const steps = [] + lines.forEach((line, index) => { + if (line.trim() !== `- uses: ${LEASE_ACTION}`) return + const body = blockAfter(lines, index) + const read = (key) => + body.find((l) => l.trim().startsWith(`${key}:`))?.trim().slice(`${key}:`.length).trim() + steps.push({ + line: index + 1, + bucket: read('bucket'), + object: read('object'), + release: read('release') + }) + }) + return steps +} + +export function leaseStepsByJob(file) { + const text = readWorkflow(file) + const steps = leaseSteps(text) + return jobs(text).map((job) => ({ + id: job.id, + steps: steps.filter((step) => step.line > job.start + 1 && step.line <= job.start + 1 + job.body.length) + })) +} + +// --- trigger and reusable-call graph --- + +export function triggers(text) { + const lines = text.split('\n') + const index = lines.findIndex((line) => line === 'on:') + if (index === -1) return [] + return blockAfter(lines, index) + .map((line) => /^ {2}([a-z_]+):/.exec(line)?.[1]) + .filter(Boolean) +} + +export function isEntrypoint(text) { + return triggers(text).some((trigger) => trigger !== 'workflow_call') +} + +export function reusableCalls(text) { + const counts = new Map() + for (const match of text.matchAll(/uses: \.\/\.github\/workflows\/([A-Za-z0-9._-]+\.yml)/g)) { + counts.set(match[1], (counts.get(match[1]) ?? 0) + 1) + } + return counts +} + +export function entrypointsFor(file, seen = new Set()) { + if (seen.has(file)) return new Set() + seen.add(file) + if (isEntrypoint(readWorkflow(file))) return new Set([file]) + const reached = new Set() + for (const candidate of workflowFiles()) { + if (candidate === file) continue + if (!reusableCalls(readWorkflow(candidate)).has(file)) continue + for (const entry of entrypointsFor(candidate, seen)) reached.add(entry) + } + return reached +} + +// --- what counts as a Cloud SQL connection-budget rollout --- + +const COMMAND_PREFIX = /^(?:-\s+)?(?:run:\s*)?(?:[a-z_]+\s*=\s*"?\$\(\s*)?(?:if\s+|then\s+|else\s+|&&\s+|\|\|\s+|!\s+)*/ + +function commandLines(text) { + return text.split('\n').map((line) => line.trim().replace(COMMAND_PREFIX, '')) +} + +export function appliesTerraform(text) { + return commandLines(text).some((line) => /^terraform\b.*\bapply\b/.test(line)) +} + +export function runsCloudRunMutation(text) { + return commandLines(text).some((line) => + /^gcloud run (?:deploy\b|services (?:update|replace)\b|jobs (?:update|deploy)\b)/.test(line) + ) +} + +// Scripts that mint a Cloud Run revision, plus every script that re-exports one of them. +export function revisionMintingScripts() { + const self = new URL(import.meta.url).pathname.split('/').pop() + const names = readdirSync(SCRIPT_ROOT).filter( + (name) => name.endsWith('.mjs') && !name.endsWith('.test.mjs') && name !== self + ) + const source = new Map( + names.map((name) => [name, readFileSync(new URL(name, SCRIPT_ROOT), 'utf8')]) + ) + const minting = new Set( + names.filter((name) => { + const text = source.get(name) + return ( + text.includes("'--no-traffic'") || + /'run',\s*'services',\s*'update'/.test(text) || + /'run',\s*'deploy'/.test(text) + ) + }) + ) + for (let changed = true; changed; ) { + changed = false + for (const name of names) { + if (minting.has(name)) continue + const imports = [...source.get(name).matchAll(/from '\.\/([A-Za-z0-9._-]+\.mjs)'/g)].map( + (match) => match[1] + ) + if (!imports.some((imported) => minting.has(imported))) continue + minting.add(name) + changed = true + } + } + return minting +} + +export function mutatesSharedInstance(text, minters = revisionMintingScripts()) { + if (appliesTerraform(text)) return 'terraform apply against the reviewed relay cell templates' + if (runsCloudRunMutation(text)) return 'gcloud mints or replaces a Cloud Run revision' + for (const script of minters) { + if (text.includes(`dev/scripts/${script}`)) return `runs ${script}, which mints a Cloud Run revision` + } + return undefined +} + +// --- the declared contract --- + +const production = (extra = {}) => ({ env: 'production', group: PRODUCTION_GROUP, ...extra }) +const staging = (extra = {}) => ({ env: 'staging', group: STAGING_GROUP, ...extra }) +const eitherEnvironment = () => ({ env: 'selectable', group: SELECTABLE_GROUP }) + +// Keys are workflow filenames, which the public copy prefixes; the prefix lives in one place. +const named = (entries) => + Object.fromEntries( + entries.map(([file, entry]) => [ + relayWorkflowFile(file), + entry.leaseFiles + ? { ...entry, leaseFiles: entry.leaseFiles.map((member) => relayWorkflowFile(member)) } + : entry + ]) + ) + +export const LEASED_WORKFLOWS = named([ + ['deploy-relay-fence-broker.yml', production()], + ['deploy-relay-production.yml', production()], + ['deploy-relay-production-director.yml', production()], + ['deploy-relay-production-multi-target.yml', production()], + [ + 'deploy-relay-production-capacity.yml', + production({ + leaseFiles: ['deploy-relay-production-capacity-job.yml'], + reentrant: true + }) + ], + [ + 'deploy-relay-production-same-cap.yml', + production({ + leaseFiles: ['deploy-relay-production-same-cap-job.yml'], + reentrant: true + }) + ], + [ + 'operate-relay-production-rehome.yml', + production({ leaseFiles: ['operate-relay-production-rehome-job.yml'] }) + ], + ['deploy-relay-asia-topology.yml', eitherEnvironment()], + ['operate-relay-asia-admission.yml', eitherEnvironment()], + ['deploy-relay-staging.yml', staging()], + ['deploy-relay-staging-gce-candidate.yml', staging()], + ['bootstrap-relay-staging-capacity.yml', staging()], + ['power-relay-staging.yml', staging()], + ['prove-relay-asia-staging.yml', staging()], + [ + 'prove-relay-staging-capacity.yml', + staging({ exclusiveBy: "inputs.mode == 'refresh-asia-c4-image'" }) + ], + ['recover-relay-staging-c4-image.yml', staging()] +]) + +export const NOT_A_CLOUD_SQL_CANDIDATE = named([ + [ + 'monitor-relay-production.yml', + 'Read-only. Its identity holds monitoring, logging, Cloud SQL and compute viewer roles only, and it runs `gcloud sql instances describe`, never a mutation. It consumes no connection budget, so the durable lease would only let monitoring block a rollout and a rollout block monitoring.' + ] +]) diff --git a/cloud/dev/scripts/deploy-relay-blue-green.mjs b/cloud/dev/scripts/deploy-relay-blue-green.mjs new file mode 100644 index 00000000000..88e4f8ccc60 --- /dev/null +++ b/cloud/dev/scripts/deploy-relay-blue-green.mjs @@ -0,0 +1,1132 @@ +import { spawnSync } from 'node:child_process' +import { createHash } from 'node:crypto' +import { pathToFileURL } from 'node:url' +import { isDeepStrictEqual } from 'node:util' + +const POLL_INTERVAL_MS = 5_000 +const MIGRATION_TIMEOUT_MS = 14 * 60 * 1000 +const CONNECTION_CAPACITY_PROTOCOL = 2 +export const DIRECTOR_REGIONAL_PLACEMENT_SECRET = + 'orca-cloud-relay-regional-placement-enabled' +export const DIRECTOR_REGIONAL_PLACEMENT_ENV = + 'ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED' +export const DIRECTOR_REHOME_IDENTITY_ENV = + 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT' +export const DIRECTOR_REHOME_AUDIENCE_ENV = 'ORCA_RELAY_REHOME_AUDIENCE' +export const SELECTOR_ROLLBACK_TAG = 'selector-rollback' +export const SELECTOR_REVISION_MARKER = '3' +export const DIRECTOR_ADMISSION_ENVIRONMENT = Object.freeze({ + ORCA_RELAY_DATABASE_POOL_MAX: '3', + ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY: '2', + ORCA_RELAY_PUBLIC_ASSIGNMENT_QUEUE_MAX: '128', + ORCA_RELAY_PUBLIC_ASSIGNMENT_RETRY_AFTER_SECONDS: '5', + ORCA_RELAY_PUBLIC_ASSIGNMENT_WAIT_MS: '4000' +}) +const DIRECTOR_STARTUP_PROBE = + 'tcpSocket.port=8080,timeoutSeconds=120,periodSeconds=120,failureThreshold=1' + +export function taggedRevisionOrigin(serviceOrigin, tag) { + const url = new URL(serviceOrigin) + if ( + url.protocol !== 'https:' || + url.pathname !== '/' || + url.search || + url.hash || + !url.hostname.endsWith('.run.app') + ) { + throw new Error('Cloud Run service origin is not canonical') + } + if (!/^[a-z][a-z0-9-]{0,62}$/.test(tag)) throw new Error('invalid Cloud Run tag') + return `https://${tag}---${url.hostname}` +} + +export function activeRevision(service) { + const active = (service.status?.traffic ?? []).filter((entry) => Number(entry.percent ?? 0) > 0) + if (active.length !== 1 || Number(active[0].percent) !== 100 || !active[0].revisionName) { + throw new Error('relay service must have exactly one revision receiving 100% traffic') + } + return active[0].revisionName +} + +export function trafficTags(service) { + return (service.status?.traffic ?? []) + .map((entry) => entry.tag) + .filter((tag) => typeof tag === 'string') +} + +export function taggedTraffic(service, tag) { + const traffic = (service.status?.traffic ?? []).find((entry) => entry.tag === tag) + if (!traffic?.url || !traffic.revisionName) throw new Error(`Cloud Run tag ${tag} is not ready`) + return { origin: traffic.url, revision: traffic.revisionName } +} + +export function revisionEnvironment(revision) { + const entries = revision.spec?.containers?.[0]?.env ?? [] + return Object.fromEntries( + entries + .filter((entry) => entry.name && 'value' in entry) + .map((entry) => [entry.name, entry.value]) + ) +} + +export function revisionSecretEnvironment(revision) { + const entries = revision.spec?.containers?.[0]?.env ?? [] + return Object.fromEntries( + entries + .map((entry) => { + const reference = entry.valueSource?.secretKeyRef ?? entry.valueFrom?.secretKeyRef + return [entry.name, reference && { + secret: reference.secret ?? reference.name, + version: reference.version ?? reference.key + }] + }) + .filter(([name, reference]) => name && reference) + ) +} + +export function revisionMinimumInstances(revision) { + return Number(revision.metadata?.annotations?.['autoscaling.knative.dev/minScale'] ?? 0) +} + +export function revisionMaximumInstances(revision) { + const value = Number(revision.metadata?.annotations?.['autoscaling.knative.dev/maxScale']) + if (!Number.isSafeInteger(value) || value < 1) { + throw new Error('serving revision has no bounded maximum instance count') + } + return value +} + +function hasExpectedEnvironment(environment, expected) { + return Object.entries(expected).every(([key, value]) => environment[key] === value) +} + +function projectServiceAccount(config, argument) { + const value = config[argument] + if (value === undefined) return undefined + const suffix = `@${config.project}.iam.gserviceaccount.com` + const account = value.endsWith(suffix) ? value.slice(0, -suffix.length) : '' + if (!/^[a-z][a-z0-9-]{4,28}[a-z0-9]$/.test(account)) { + throw new Error(`--${argument} must belong to the selected project`) + } + return value +} + +function directorCellsJson(value, { allowMissingRegion = false } = {}) { + if (value === undefined) return undefined + if (value.length > 100_000 || /[\r\n]/.test(value)) { + throw new Error('--director-cells-json is invalid') + } + const cells = JSON.parse(value) + if (!Array.isArray(cells) || cells.length < 1 || cells.length > 100) { + throw new Error('--director-cells-json must contain 1..100 cells') + } + const ids = new Set() + for (const cell of cells) { + const keys = Object.keys(cell ?? {}).sort() + const expectedKeys = [ + 'capacityRequests', + 'id', + 'initiallyEnabled', + ...(allowMissingRegion && cell?.region === undefined ? [] : ['region']), + 'url', + ...(cell?.connectionHardCap === undefined + ? [] + : ['connectionHardCap', 'connectionUnobservedBound']) + ].sort() + let origin + try { + origin = new URL(cell?.url) + } catch { + throw new Error('--director-cells-json contains an invalid cell URL') + } + const cap = cell?.connectionHardCap + const bound = cell?.connectionUnobservedBound + if ( + JSON.stringify(keys) !== JSON.stringify(expectedKeys) || + !/^[a-z][a-z0-9-]{0,39}$/.test(cell.id ?? '') || + ids.has(cell.id) || + origin.protocol !== 'https:' || + origin.origin !== cell.url || + !Number.isSafeInteger(cell.capacityRequests) || + cell.capacityRequests < 1 || + !['us-central1', 'asia-east2'].includes( + cell.region ?? (allowMissingRegion ? 'us-central1' : undefined) + ) || + typeof cell.initiallyEnabled !== 'boolean' || + (cap !== undefined && + (![600, 1_000, 3_000].includes(cap) || + !Number.isSafeInteger(bound) || + bound < 0 || + bound >= cap - 100)) + ) { + throw new Error('--director-cells-json contains an invalid cell') + } + ids.add(cell.id) + } + return JSON.stringify( + cells.map((cell) => ({ + id: cell.id, + url: cell.url, + capacityRequests: cell.capacityRequests, + region: cell.region ?? 'us-central1', + initiallyEnabled: cell.initiallyEnabled, + ...(cell.connectionHardCap === undefined + ? {} + : { + connectionHardCap: cell.connectionHardCap, + connectionUnobservedBound: cell.connectionUnobservedBound + }) + })) + ) +} + +export function directorTopologyChange(currentValue, desiredValue, cellId) { + const current = JSON.parse(directorCellsJson(currentValue, { allowMissingRegion: true })) + const desired = JSON.parse(directorCellsJson(desiredValue)) + if ( + current.length !== desired.length || + desired.some(({ id }, index) => id !== current[index]?.id) + ) { + throw new Error('director topology changes the Relay cell set or order') + } + let changed = false + for (let index = 0; index < desired.length; index += 1) { + const before = structuredClone(current[index]) + const after = structuredClone(desired[index]) + if (after.id === cellId) { + delete before.connectionHardCap + delete before.connectionUnobservedBound + delete after.connectionHardCap + delete after.connectionUnobservedBound + changed = !isDeepStrictEqual(current[index], desired[index]) + } + if (!isDeepStrictEqual(before, after)) { + throw new Error('director topology changes fields outside the reviewed capacity pair') + } + } + if (!desired.some(({ id }) => id === cellId)) { + throw new Error('director topology omits the reviewed capacity cell') + } + return { changed, value: JSON.stringify(desired) } +} + +export function directorCellSetAddition(currentValue, desiredValue) { + const current = JSON.parse(directorCellsJson(currentValue, { allowMissingRegion: true })) + const desired = JSON.parse(directorCellsJson(desiredValue)) + if (desired.length < current.length) { + throw new Error('director topology addition cannot remove cells') + } + const desiredById = new Map(desired.map((cell) => [cell.id, cell])) + if (current.some((cell) => !isDeepStrictEqual(desiredById.get(cell.id), cell))) { + throw new Error('director topology addition changes an existing cell') + } + const currentIds = new Set(current.map((cell) => cell.id)) + const additions = desired.filter((cell) => !currentIds.has(cell.id)) + if (additions.some((cell) => cell.initiallyEnabled !== false)) { + throw new Error('director topology additions must start disabled') + } + return { changed: additions.length > 0, value: JSON.stringify(desired) } +} + +export function directorDeploymentEnvironment(config) { + const imageDigest = config.image?.match(/@(sha256:[a-f0-9]{64})$/)?.[1] + if (config.image !== undefined && imageDigest === undefined) { + throw new Error('--image must use an immutable digest for director deployments') + } + const environment = { + ...DIRECTOR_ADMISSION_ENVIRONMENT, + ORCA_RELAY_ADMISSION_SELECTOR_VERSION: SELECTOR_REVISION_MARKER, + ...(imageDigest === undefined ? {} : { ORCA_RELAY_IMAGE_DIGEST: imageDigest }) + } + const serviceAccount = projectServiceAccount(config, 'capacity-service-account') + const asiaProofServiceAccount = projectServiceAccount(config, 'asia-proof-service-account') + const rehomeDirectorServiceAccount = projectServiceAccount( + config, + 'rehome-director-service-account' + ) + const cellsJson = directorCellsJson(config['director-cells-json']) + if (serviceAccount !== undefined) { + environment.ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT = serviceAccount + } + if (asiaProofServiceAccount !== undefined) { + environment.ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT = asiaProofServiceAccount + } + if (rehomeDirectorServiceAccount !== undefined) { + environment[DIRECTOR_REHOME_IDENTITY_ENV] = rehomeDirectorServiceAccount + environment[DIRECTOR_REHOME_AUDIENCE_ENV] = config['rehome-audience'] + } + if (cellsJson !== undefined) environment.ORCA_RELAY_CELLS_JSON = cellsJson + return environment +} + +export function environmentUpdateValue(environment) { + const entries = Object.entries(environment) + if (entries.every(([key, value]) => !key.includes(',') && !value.includes(','))) { + return entries.map(([key, value]) => `${key}=${value}`).join(',') + } + const delimiter = ['~', '|', '@', '%', ';'].find((candidate) => + entries.every(([key, value]) => !key.includes(candidate) && !value.includes(candidate)) + ) + if (!delimiter) throw new Error('candidate environment has no safe gcloud delimiter') + return `^${delimiter}^${entries.map(([key, value]) => `${key}=${value}`).join(delimiter)}` +} + +export function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error(`invalid argument ${key ?? ''}`) + values[key.slice(2)] = value + } + const required = ['project', 'region', 'service', 'image', 'role', 'release-id'] + for (const key of required) if (!values[key]) throw new Error(`missing --${key}`) + if (!['director', 'cell'].includes(values.role)) throw new Error('--role must be director or cell') + if (values['min-instances'] !== undefined && !/^(0|[1-9][0-9]*)$/.test(values['min-instances'])) { + throw new Error('--min-instances must be a nonnegative integer') + } + if (values['max-instances'] !== undefined && !/^[1-9][0-9]*$/.test(values['max-instances'])) { + throw new Error('--max-instances must be a positive integer') + } + if (values.role === 'cell') { + for (const key of ['director-origin', 'admin-audience']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!['enabled', 'disabled'].includes(values['final-admission'] ?? 'enabled')) { + throw new Error('--final-admission must be enabled or disabled') + } + if ( + values['capacity-service-account'] !== undefined || + values['director-cells-json'] !== undefined || + values['runtime-service-account'] !== undefined || + values['rehome-director-service-account'] !== undefined || + values['rehome-audience'] !== undefined || + values['expected-rehome-generation'] !== undefined || + values['rehome-control-origin'] !== undefined + ) { + throw new Error('director configuration arguments require --role director') + } + } + if ( + values.role === 'director' && + values['capacity-cell-id'] !== undefined && + values['director-cells-json'] === undefined + ) { + throw new Error('--capacity-cell-id requires --director-cells-json') + } + if (values['regional-placement-enabled'] !== undefined) { + throw new Error('regional placement changes use the audited runtime-setting step') + } + if ( + values['regional-placement-secret-version'] !== undefined && + !/^[1-9][0-9]*$/.test(values['regional-placement-secret-version']) + ) { + throw new Error('--regional-placement-secret-version must be a positive integer') + } + if (!['true', 'false'].includes(values['prune-revisions'] ?? 'false')) { + throw new Error('--prune-revisions must be true or false') + } + if (!['true', 'false'].includes(values['bootstrap-runtime-identity'] ?? 'false')) { + throw new Error('--bootstrap-runtime-identity must be true or false') + } + if (values.role === 'director') { + directorDeploymentEnvironment(values) + projectServiceAccount(values, 'runtime-service-account') + projectServiceAccount(values, 'predecessor-runtime-service-account') + if ( + values['bootstrap-runtime-identity'] === 'true' && + (!values['runtime-service-account'] || + !values['predecessor-runtime-service-account'] || + !values['predecessor-image-digest']) + ) { + throw new Error('runtime identity bootstrap requires exact predecessor digest and identities') + } + if ( + values['predecessor-image-digest'] !== undefined && + !/^sha256:[a-f0-9]{64}$/.test(values['predecessor-image-digest']) + ) { + throw new Error('--predecessor-image-digest must be an immutable digest') + } + const rehomePair = [ + 'rehome-director-service-account', + 'rehome-audience' + ].map((key) => values[key] !== undefined) + if (rehomePair[0] !== rehomePair[1]) { + throw new Error('rehome identity and audience must be configured together') + } + if (values['rehome-audience'] !== undefined) { + const audience = new URL(values['rehome-audience']) + if ( + audience.protocol !== 'https:' || + audience.pathname !== '/v1/admin/host-drain' || + audience.search || + audience.hash + ) { + throw new Error('--rehome-audience must be an exact host-drain HTTPS URL') + } + } + const controlArguments = [ + 'expected-rehome-generation', + 'rehome-control-origin', + 'admin-audience' + ].map((key) => values[key] !== undefined) + if (controlArguments.some(Boolean) && !controlArguments.every(Boolean)) { + throw new Error('durable rehome verification arguments must be configured together') + } + if ( + values['expected-rehome-generation'] !== undefined && + !/^(0|[1-9][0-9]*)$/.test(values['expected-rehome-generation']) + ) { + throw new Error('--expected-rehome-generation must be a nonnegative integer') + } + if (values['rehome-control-origin'] !== undefined) { + const origin = new URL(values['rehome-control-origin']) + if (origin.protocol !== 'https:' || origin.origin !== values['rehome-control-origin']) { + throw new Error('--rehome-control-origin must be an HTTPS origin') + } + } + } + return values +} + +function commandJson(args) { + const result = spawnSync('gcloud', args, { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }) + if (result.status !== 0) { + throw new Error(`gcloud ${args.slice(0, 4).join(' ')} failed: ${result.stderr.trim()}`) + } + return JSON.parse(result.stdout) +} + +function commandText(args, { sensitive = false } = {}) { + const result = spawnSync('gcloud', args, { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }) + if (result.status !== 0) { + const detail = sensitive ? 'credential command failed' : result.stderr.trim() + throw new Error(`gcloud ${args.slice(0, 4).join(' ')} failed: ${detail}`) + } + return result.stdout.trim() +} + +export function suppliedAdminIdentityToken(environment = process.env) { + const token = environment.ORCA_RELAY_ADMIN_ID_TOKEN + if (token === undefined) return null + // The workflow supplies a masked Google ID token because external-account gcloud cannot mint one directly. + if (token.length > 8_192 || !/^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/.test(token)) { + throw new Error('invalid supplied admin identity token') + } + return token +} + +function adminIdentityToken(config) { + return ( + suppliedAdminIdentityToken() ?? + commandText(['auth', 'print-identity-token', `--audiences=${config['admin-audience']}`], { + sensitive: true + }) + ) +} + +function serviceArguments(config) { + return ['--project', config.project, '--region', config.region] +} + +export function directorStartupProbeArguments(role) { + return role === 'director' ? ['--startup-probe', DIRECTOR_STARTUP_PROBE] : [] +} + +function describeService(config) { + return commandJson([ + 'run', + 'services', + 'describe', + config.service, + ...serviceArguments(config), + '--format=json' + ]) +} + +function describeRevision(config, revision) { + return commandJson([ + 'run', + 'revisions', + 'describe', + revision, + ...serviceArguments(config), + '--format=json' + ]) +} + +function listRevisions(config) { + return commandJson([ + 'run', + 'revisions', + 'list', + '--service', + config.service, + ...serviceArguments(config), + '--format=json' + ]) +} + +function deleteRevision(config, revision) { + commandText([ + 'run', + 'revisions', + 'delete', + revision, + ...serviceArguments(config), + '--quiet' + ]) +} + +function updateTraffic(config, args) { + commandText([ + 'run', + 'services', + 'update-traffic', + config.service, + ...serviceArguments(config), + ...args, + '--quiet' + ]) +} + +function removeDirectorTrafficTags(config, operations, retained = new Set()) { + const tags = trafficTags(operations.describeService(config)).filter( + (tag) => !retained.has(tag) + ) + if (tags.length === 0) return + try { + operations.updateTraffic(config, [`--remove-tags=${tags.join(',')}`]) + } catch (error) { + const remaining = new Set(trafficTags(operations.describeService(config))) + if (tags.some((tag) => remaining.has(tag))) throw error + } +} + +function pruneDirectorRevisions(config, operations) { + const service = operations.describeService(config) + const retained = new Set([ + activeRevision(service), + taggedTraffic(service, SELECTOR_ROLLBACK_TAG).revision + ]) + for (const revision of operations.listRevisions(config)) { + const name = revision.metadata?.name + if (!name) throw new Error('director revision list contains an unnamed revision') + if (!retained.has(name)) operations.deleteRevision(config, name) + } + const remaining = operations + .listRevisions(config) + .map((revision) => revision.metadata?.name) + if ( + remaining.length !== retained.size || + remaining.some((name) => !name || !retained.has(name)) + ) { + throw new Error('old director revisions remain after deployment') + } +} + +function deployCandidate( + config, + tag, + env = {}, + image = config.image, + minInstances = config['min-instances'], + maxInstances, + regionalPlacementVersion +) { + const args = [ + 'run', + 'services', + 'update', + config.service, + ...serviceArguments(config), + '--image', + image, + '--tag', + tag, + '--no-traffic' + ] + if (maxInstances !== undefined) args.push('--max', String(maxInstances)) + if (config.role === 'director') { + args.push( + '--update-secrets', + `${DIRECTOR_REGIONAL_PLACEMENT_ENV}=${DIRECTOR_REGIONAL_PLACEMENT_SECRET}:${regionalPlacementVersion}` + ) + if (config['runtime-service-account'] !== undefined) { + args.push('--service-account', config['runtime-service-account']) + } + } + args.push(...directorStartupProbeArguments(config.role)) + if (minInstances !== undefined) { + args.push('--min-instances', String(minInstances)) + } + const entries = Object.entries(env) + if (entries.length > 0) { + args.push('--update-env-vars', environmentUpdateValue(env)) + } + args.push('--quiet') + commandText(args) +} + +function revisionShape(revision, mutableEnvironment, allowServiceAccountChange = false) { + const spec = structuredClone(revision.spec ?? {}) + if (allowServiceAccountChange) delete spec.serviceAccountName + for (const container of spec.containers ?? []) { + delete container.image + delete container.startupProbe + container.env = (container.env ?? []).filter( + ({ name }) => !(name in mutableEnvironment) + ) + } + const ignoredAnnotations = new Set([ + 'autoscaling.knative.dev/minScale', + 'run.googleapis.com/client-name', + 'run.googleapis.com/client-version', + 'run.googleapis.com/operation-id', + 'serving.knative.dev/creator' + ]) + const annotations = Object.fromEntries( + Object.entries(revision.metadata?.annotations ?? {}).filter( + ([name]) => !ignoredAnnotations.has(name) + ) + ) + return { spec, annotations } +} + +function assertPreservedRevisionShape( + serving, + candidate, + mutableEnvironment, + allowServiceAccountChange = false +) { + if (!isDeepStrictEqual( + revisionShape(serving, mutableEnvironment, allowServiceAccountChange), + revisionShape(candidate, mutableEnvironment, allowServiceAccountChange) + )) { + throw new Error('director candidate changed unrelated revision shape') + } +} + +function releaseLabel(releaseId) { + const normalized = releaseId.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-|-$/g, '') + if (!normalized) throw new Error('release id has no usable characters') + return normalized.slice(0, 32).replace(/-$/g, '') +} + +export function cloudRunTrafficTag(service, prefix, releaseId) { + if (!/^[a-z][a-z0-9-]*$/.test(service)) throw new Error('invalid Cloud Run service name') + if (!/^[a-z][a-z0-9-]*$/.test(prefix)) throw new Error('invalid Cloud Run tag prefix') + const digest = createHash('sha256').update(releaseLabel(releaseId)).digest('hex').slice(0, 9) + const tag = `${prefix}-${digest}` + // Cloud Run imposes this combined bound in addition to the standalone tag bound. + if (service.length + tag.length > 46) throw new Error('Cloud Run service leaves no safe tag space') + return tag +} + +function cellIdentifier(sourceCellId, releaseId) { + const suffix = releaseLabel(releaseId) + const available = 128 - suffix.length - 2 + return `${sourceCellId.slice(0, available)}--${suffix}` +} + +async function waitForHealth(origin, connectionCapacityProtocol) { + const response = await fetch(`${origin}/health`, { signal: AbortSignal.timeout(15_000) }) + const body = await response.json() + if ( + !response.ok || + body.ok !== true || + (connectionCapacityProtocol !== undefined && + body.connectionCapacityProtocol !== connectionCapacityProtocol) + ) { + throw new Error(`candidate health failed at ${origin}`) + } +} + +export async function waitForEvacuationCapacity( + adminPost, + sourceCellId, + targetCellId, + { pollIntervalMs = POLL_INTERVAL_MS, timeoutMs = MIGRATION_TIMEOUT_MS } = {} +) { + const deadline = Date.now() + timeoutMs + while (Date.now() < deadline) { + try { + return await adminPost('/v1/admin/evacuation-capacity', { + v: 1, + sourceCellId, + targetCellId + }) + } catch (error) { + if (!(error instanceof Error) || !error.message.endsWith(': target_cell_unavailable')) throw error + await new Promise((resolve) => setTimeout(resolve, pollIntervalMs)) + } + } + throw new Error('timed out waiting for candidate cell readiness') +} + +async function adminClient(config) { + const token = adminIdentityToken(config) + return async (path, body) => { + const response = await fetch(`${config['director-origin']}${path}`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }) + const result = await response.json().catch(() => ({ error: `http_${response.status}` })) + if (!response.ok) throw new Error(`${path} failed: ${result.error ?? response.status}`) + return result + } +} + +export async function assertRegionalRehomeDisabled( + config, + origin, + fetchImpl = fetch +) { + const response = await fetchImpl(`${origin}/v1/admin/regional-rehome-control`, { + method: 'POST', + headers: { + authorization: `Bearer ${adminIdentityToken(config)}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, action: 'inspect' }), + signal: AbortSignal.timeout(30_000) + }) + const result = await response.json().catch(() => ({ error: `http_${response.status}` })) + if (!response.ok) { + throw new Error(`regional rehome inspection failed: ${result.error ?? response.status}`) + } + const generation = Number(config['expected-rehome-generation']) + if ( + result.v !== 1 || + result.control?.enabled !== false || + result.control?.generation !== generation + ) { + throw new Error('regional rehome control is not durably disabled at the expected generation') + } + return result.control +} + +function assertDirectorRevisionIdentity(revision, config) { + const expectedRuntimeServiceAccount = config['runtime-service-account'] + if ( + expectedRuntimeServiceAccount !== undefined && + revision.spec?.serviceAccountName !== expectedRuntimeServiceAccount + ) { + throw new Error('director revision uses an unexpected runtime service account') + } +} + +export async function deployDirector(config, tag, overrides = {}) { + const operations = { + deployCandidate, + describeService, + describeRevision, + listRevisions, + deleteRevision, + updateTraffic, + waitForHealth, + assertRegionalRehomeDisabled, + ...overrides + } + // Why: gcloud does not carry minScale onto a new revision, and the candidate below takes + // 100% of traffic. Inheriting the serving floor keeps one owner for the number instead of + // restating it here; public admission is per-instance, so losing it shrinks fleet capacity. + const initialService = operations.describeService(config) + const servingRevision = operations.describeRevision(config, activeRevision(initialService)) + const bootstrapRuntimeIdentity = config['bootstrap-runtime-identity'] === 'true' + if (bootstrapRuntimeIdentity) { + if ( + servingRevision.spec?.serviceAccountName !== + config['predecessor-runtime-service-account'] + ) { + throw new Error('director predecessor runtime service account does not match') + } + if ( + config['predecessor-runtime-service-account'] === config['runtime-service-account'] + ) { + throw new Error('director runtime identity bootstrap has already completed') + } + const predecessorDigest = servingRevision.spec?.containers?.[0]?.image?.split('@').at(-1) + if (predecessorDigest !== config['predecessor-image-digest']) { + throw new Error('director predecessor image digest does not match') + } + } else { + assertDirectorRevisionIdentity(servingRevision, config) + } + const servingMinimumInstances = revisionMinimumInstances(servingRevision) + const servingMaximumInstances = revisionMaximumInstances(servingRevision) + const requiredMaximumInstances = config['max-instances'] === undefined + ? servingMaximumInstances + : Number(config['max-instances']) + if (servingMaximumInstances !== requiredMaximumInstances) { + throw new Error( + `serving revision holds ${servingMaximumInstances} maximum instances, expected ${requiredMaximumInstances}` + ) + } + const servingSecrets = revisionSecretEnvironment(servingRevision) + const servingRegionalPlacementVersion = + servingSecrets[DIRECTOR_REGIONAL_PLACEMENT_ENV]?.version + const targetRegionalPlacementVersion = + config['regional-placement-secret-version'] ?? servingRegionalPlacementVersion + if (!/^[1-9][0-9]*$/.test(targetRegionalPlacementVersion ?? '')) { + throw new Error('director deployment requires an exact regional placement secret version') + } + const rollbackRegionalPlacementVersion = + /^[1-9][0-9]*$/.test(servingRegionalPlacementVersion ?? '') + ? servingRegionalPlacementVersion + : targetRegionalPlacementVersion + const requiredMinimumInstances = + config['min-instances'] === undefined + ? servingMinimumInstances + : Number(config['min-instances']) + const requiredCapacityProtocol = + config['prune-revisions'] === 'true' ? CONNECTION_CAPACITY_PROTOCOL : undefined + const currentEnvironment = revisionEnvironment(servingRevision) + const deploymentEnvironment = directorDeploymentEnvironment(config) + const mutableEnvironment = { + ...deploymentEnvironment, + [DIRECTOR_REGIONAL_PLACEMENT_ENV]: '' + } + const topology = config['director-cells-json'] === undefined + ? { changed: false } + : config['capacity-cell-id'] === undefined + ? directorCellSetAddition( + currentEnvironment.ORCA_RELAY_CELLS_JSON, + deploymentEnvironment.ORCA_RELAY_CELLS_JSON + ) + : directorTopologyChange( + currentEnvironment.ORCA_RELAY_CELLS_JSON, + deploymentEnvironment.ORCA_RELAY_CELLS_JSON, + config['capacity-cell-id'] + ) + const verifyRehomeDisabled = async (origin) => { + if (config['expected-rehome-generation'] === undefined) return + await operations.assertRegionalRehomeDisabled(config, origin) + } + if (!bootstrapRuntimeIdentity) { + await verifyRehomeDisabled(config['rehome-control-origin']) + } + removeDirectorTrafficTags(config, operations) + let deployed + let promoted = false + try { + operations.deployCandidate( + config, + SELECTOR_ROLLBACK_TAG, + deploymentEnvironment, + config.image, + 0, + requiredMaximumInstances, + rollbackRegionalPlacementVersion + ) + const rollback = taggedTraffic( + operations.describeService(config), + SELECTOR_ROLLBACK_TAG + ) + const rollbackRevision = operations.describeRevision(config, rollback.revision) + assertDirectorRevisionIdentity(rollbackRevision, config) + assertPreservedRevisionShape( + servingRevision, + rollbackRevision, + mutableEnvironment, + bootstrapRuntimeIdentity + ) + const rollbackEnvironment = revisionEnvironment(rollbackRevision) + const rollbackSecrets = revisionSecretEnvironment(rollbackRevision) + if ( + rollbackEnvironment.ORCA_RELAY_ROLE !== 'director' || + rollbackEnvironment.ORCA_RELAY_ADMISSION_SELECTOR_VERSION !== + SELECTOR_REVISION_MARKER || + !hasExpectedEnvironment(rollbackEnvironment, deploymentEnvironment) || + rollbackSecrets[DIRECTOR_REGIONAL_PLACEMENT_ENV]?.secret !== + DIRECTOR_REGIONAL_PLACEMENT_SECRET || + rollbackSecrets[DIRECTOR_REGIONAL_PLACEMENT_ENV]?.version !== + rollbackRegionalPlacementVersion || + revisionMinimumInstances(rollbackRevision) !== 0 + ) { + throw new Error('rollback revision is not selector-compatible') + } + await operations.waitForHealth(rollback.origin, requiredCapacityProtocol) + await verifyRehomeDisabled(rollback.origin) + operations.deployCandidate( + config, + tag, + deploymentEnvironment, + config.image, + requiredMinimumInstances, + requiredMaximumInstances, + targetRegionalPlacementVersion + ) + const candidate = taggedTraffic(operations.describeService(config), tag) + const candidateRevision = operations.describeRevision(config, candidate.revision) + assertDirectorRevisionIdentity(candidateRevision, config) + assertPreservedRevisionShape( + servingRevision, + candidateRevision, + mutableEnvironment, + bootstrapRuntimeIdentity + ) + const environment = revisionEnvironment(candidateRevision) + const secrets = revisionSecretEnvironment(candidateRevision) + if ( + environment.ORCA_RELAY_ROLE !== 'director' || + environment.ORCA_RELAY_ADMISSION_SELECTOR_VERSION !== SELECTOR_REVISION_MARKER || + !hasExpectedEnvironment(environment, deploymentEnvironment) || + secrets[DIRECTOR_REGIONAL_PLACEMENT_ENV]?.secret !== + DIRECTOR_REGIONAL_PLACEMENT_SECRET || + secrets[DIRECTOR_REGIONAL_PLACEMENT_ENV]?.version !== + targetRegionalPlacementVersion + ) { + throw new Error('stable relay service is not selector-compatible') + } + // Fail before the traffic move, so a candidate that lost the floor never serves. + const candidateMinimumInstances = revisionMinimumInstances(candidateRevision) + if (candidateMinimumInstances !== requiredMinimumInstances) { + throw new Error( + `candidate holds ${candidateMinimumInstances} minimum instances, expected ${requiredMinimumInstances}` + ) + } + await operations.waitForHealth(candidate.origin, requiredCapacityProtocol) + await verifyRehomeDisabled(candidate.origin) + operations.updateTraffic(config, [`--to-tags=${tag}=100`]) + promoted = true + removeDirectorTrafficTags(config, operations, new Set([SELECTOR_ROLLBACK_TAG])) + deployed = { + event: 'director_deployed', + revision: candidate.revision, + rollbackRevision: rollback.revision, + topologyChanged: topology.changed + } + } catch (error) { + const recoveryErrors = [error] + if (promoted) { + try { + operations.updateTraffic(config, [`--to-tags=${SELECTOR_ROLLBACK_TAG}=100`]) + } catch (rollbackError) { + recoveryErrors.push(rollbackError) + } + } + try { + removeDirectorTrafficTags( + config, + operations, + promoted ? new Set([SELECTOR_ROLLBACK_TAG]) : new Set() + ) + } catch (cleanupError) { + recoveryErrors.push(cleanupError) + } + if (recoveryErrors.length > 1) { + const details = recoveryErrors + .map((failure) => failure instanceof Error ? failure.message : String(failure)) + .join('; ') + throw new AggregateError(recoveryErrors, `director deploy recovery failed: ${details}`) + } + throw error + } + if (config['prune-revisions'] === 'true') pruneDirectorRevisions(config, operations) + process.stdout.write(`${JSON.stringify(deployed)}\n`) +} + +async function waitForTargetRegistration(adminPost, sourceCellId, targetCellId) { + const deadline = Date.now() + MIGRATION_TIMEOUT_MS + while (Date.now() < deadline) { + const status = await adminPost('/v1/admin/evacuation-status', { + v: 1, + sourceCellId, + targetCellId, + completeReady: false + }) + process.stdout.write( + `${JSON.stringify({ event: 'migration_registration', ...status })}\n` + ) + if (status.inProgress === status.targetRegistered) return + await new Promise((resolve) => setTimeout(resolve, POLL_INTERVAL_MS)) + } + throw new Error('timed out waiting for target control registrations') +} + +async function waitForCompletion(adminPost, sourceCellId, targetCellId) { + const deadline = Date.now() + MIGRATION_TIMEOUT_MS + while (Date.now() < deadline) { + const status = await adminPost('/v1/admin/evacuation-status', { + v: 1, + sourceCellId, + targetCellId, + completeReady: true + }) + process.stdout.write(`${JSON.stringify({ event: 'migration_completion', ...status })}\n`) + if (status.inProgress === 0) return + await new Promise((resolve) => setTimeout(resolve, POLL_INTERVAL_MS)) + } + throw new Error('timed out waiting for source activity to drain') +} + +async function startAllEvacuations(adminPost, sourceCellId, targetCellId) { + for (;;) { + const result = await adminPost('/v1/admin/evacuate-cell', { + v: 1, + sourceCellId, + targetCellId, + limit: 100 + }) + process.stdout.write(`${JSON.stringify({ event: 'migration_batch', started: result.started })}\n`) + if (result.started === 0) return + } +} + +async function deployCell(config, tag, oldTag, drainTag) { + const initialService = describeService(config) + const currentRevision = activeRevision(initialService) + const currentRevisionState = describeRevision(config, currentRevision) + const currentEnv = revisionEnvironment(currentRevisionState) + const currentImage = currentRevisionState.spec?.containers?.[0]?.image + const sourceCellId = currentEnv.ORCA_RELAY_CELL_ID + const sourceOrigin = currentEnv.ORCA_RELAY_CELL_URL + const capacityRequests = Number(currentEnv.ORCA_RELAY_CELL_CAPACITY) + if ( + currentEnv.ORCA_RELAY_ROLE !== 'cell' || + !sourceCellId || + !sourceOrigin || + !currentImage || + !Number.isInteger(capacityRequests) || + capacityRequests <= 0 + ) { + throw new Error('active cell revision has invalid role, identity, image, or capacity') + } + updateTraffic(config, [`--update-tags=${drainTag}=${currentRevision}`]) + const drainRevision = taggedTraffic(describeService(config), drainTag) + const previousOrigin = taggedRevisionOrigin(initialService.status.url, oldTag) + const candidateOrigin = taggedRevisionOrigin(initialService.status.url, tag) + const targetCellId = cellIdentifier(sourceCellId, config['release-id']) + const adminPost = await adminClient(config) + + await adminPost('/v1/admin/cell-config', { + v: 1, + cellId: targetCellId, + cellUrl: candidateOrigin, + capacityRequests, + enabled: false + }) + deployCandidate(config, tag, { + ORCA_RELAY_CELL_ID: targetCellId, + ORCA_RELAY_CELL_URL: candidateOrigin, + ORCA_RELAY_PUBLIC_URL: candidateOrigin + }) + const candidate = taggedTraffic(describeService(config), tag) + if (candidate.origin !== candidateOrigin) { + throw new Error('queried candidate tag URL mismatches its configured origin') + } + await waitForHealth(candidate.origin) + deployCandidate( + config, + oldTag, + { + ORCA_RELAY_CELL_ID: sourceCellId, + ORCA_RELAY_CELL_URL: previousOrigin, + ORCA_RELAY_PUBLIC_URL: previousOrigin + }, + currentImage + ) + const previous = taggedTraffic(describeService(config), oldTag) + if (previous.origin !== previousOrigin) { + throw new Error('queried previous tag URL mismatches its configured origin') + } + await waitForHealth(previous.origin) + + // HTTP health precedes the authenticated heartbeat that makes a migration target eligible. + await waitForEvacuationCapacity(adminPost, sourceCellId, targetCellId) + await adminPost('/v1/admin/cell-state', { v: 1, cellId: sourceCellId, enabled: false }) + let sourceReconfigured = false + try { + const capacity = await adminPost('/v1/admin/evacuation-capacity', { + v: 1, + sourceCellId, + targetCellId + }) + process.stdout.write(`${JSON.stringify({ event: 'migration_capacity', ...capacity })}\n`) + if (capacity.requiredTargetUnits > capacity.availableTargetUnits) { + throw new Error('candidate lacks durable reservation headroom for target-first migration') + } + // The keeper uses the old image and cell identity without competing with live controls. + await adminPost('/v1/admin/cell-config', { + v: 1, + cellId: sourceCellId, + cellUrl: previous.origin, + capacityRequests, + enabled: false + }) + sourceReconfigured = true + await adminPost('/v1/admin/cell-config', { + v: 1, + cellId: targetCellId, + cellUrl: candidate.origin, + capacityRequests, + enabled: true + }) + } catch (error) { + if (sourceReconfigured) { + await adminPost('/v1/admin/cell-config', { + v: 1, + cellId: sourceCellId, + cellUrl: sourceOrigin, + capacityRequests, + enabled: true + }) + } else { + await adminPost('/v1/admin/cell-state', { v: 1, cellId: sourceCellId, enabled: true }) + } + throw error + } + await startAllEvacuations(adminPost, sourceCellId, targetCellId) + + await fetch(`${drainRevision.origin}/v1/admin/drain`, { + method: 'POST', + headers: { + authorization: `Bearer ${adminIdentityToken(config)}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, graceMs: 120_000 }), + signal: AbortSignal.timeout(30_000) + }).then(async (response) => { + if (!response.ok) throw new Error(`old revision drain failed: ${response.status}`) + }) + + await waitForTargetRegistration(adminPost, sourceCellId, targetCellId) + updateTraffic(config, [`--to-tags=${tag}=100`]) + await waitForCompletion(adminPost, sourceCellId, targetCellId) + const finalEnabled = (config['final-admission'] ?? 'enabled') === 'enabled' + if (!finalEnabled) { + // GCE-backed staging keeps stamped cells runnable for regression without assigning normal traffic. + await adminPost('/v1/admin/cell-state', { v: 1, cellId: targetCellId, enabled: false }) + } + process.stdout.write( + `${JSON.stringify({ + event: 'cell_deployed', + service: config.service, + sourceCellId, + targetCellId, + enabled: finalEnabled, + drainRevision: drainRevision.revision, + previousRevision: previous.revision, + candidateRevision: candidate.revision + })}\n` + ) +} + +export async function main(argv = process.argv.slice(2)) { + const config = parseArguments(argv) + const tag = cloudRunTrafficTag(config.service, 'candidate', config['release-id']) + const oldTag = cloudRunTrafficTag(config.service, 'previous', config['release-id']) + const drainTag = cloudRunTrafficTag(config.service, 'drain', config['release-id']) + if (config.role === 'director') await deployDirector(config, tag) + else await deployCell(config, tag, oldTag, drainTag) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/deploy-relay-blue-green.test.mjs b/cloud/dev/scripts/deploy-relay-blue-green.test.mjs new file mode 100644 index 00000000000..6e56676098b --- /dev/null +++ b/cloud/dev/scripts/deploy-relay-blue-green.test.mjs @@ -0,0 +1,814 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { fileURLToPath } from 'node:url' +import { + activeRevision, + cloudRunTrafficTag, + DIRECTOR_ADMISSION_ENVIRONMENT, + DIRECTOR_REGIONAL_PLACEMENT_ENV, + DIRECTOR_REGIONAL_PLACEMENT_SECRET, + DIRECTOR_REHOME_AUDIENCE_ENV, + DIRECTOR_REHOME_IDENTITY_ENV, + assertRegionalRehomeDisabled, + deployDirector, + directorDeploymentEnvironment, + directorCellSetAddition, + directorStartupProbeArguments, + directorTopologyChange, + environmentUpdateValue, + parseArguments, + revisionEnvironment, + revisionSecretEnvironment, + suppliedAdminIdentityToken, + taggedRevisionOrigin, + taggedTraffic, + trafficTags, + waitForEvacuationCapacity +} from './deploy-relay-blue-green.mjs' + +// Terraform declares these values; audited blue/green stamps the same values without targeting +// the drifted service, so this contract must fail before either side can silently diverge. +function terraformDirectorEnvironment(names) { + const read = (name) => + readFileSync(fileURLToPath(new URL(`../../infra/terraform/${name}`, import.meta.url)), 'utf8') + const relay = read('relay.tf') + const variables = read('variables.tf') + const production = read('environments/production.tfvars') + return Object.fromEntries( + names.map((name) => { + const block = new RegExp(`name\\s*=\\s*"${name}"\\s*\\n\\s*value\\s*=\\s*([^\\n]+)`).exec(relay) + assert.ok(block, `${name} is not set by relay.tf`) + const variable = /var\.([a-z_]+)/.exec(block[1]) + assert.ok(variable, `${name} is not sourced from a Terraform variable`) + // An environment override wins over the variable default, as Terraform resolves it. + const override = new RegExp(`^${variable[1]}\\s*=\\s*(\\S+)`, 'm').exec(production) + const fallback = new RegExp( + `variable\\s+"${variable[1]}"[\\s\\S]*?default\\s*=\\s*(\\S+)` + ).exec(variables) + assert.ok(override || fallback, `${variable[1]} has neither an override nor a default`) + return [name, String((override ?? fallback)[1]).replace(/"/g, '')] + }) + ) +} + +test('director admission environment matches what Terraform deploys', () => { + const names = Object.keys(DIRECTOR_ADMISSION_ENVIRONMENT) + assert.deepEqual(terraformDirectorEnvironment(names), { ...DIRECTOR_ADMISSION_ENVIRONMENT }) + + const relay = readFileSync( + fileURLToPath(new URL('../../infra/terraform/relay.tf', import.meta.url)), + 'utf8' + ) + assert.match( + relay, + /name = "ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED"[\s\S]*?secret\s+= google_secret_manager_secret\.relay_regional_placement_enabled\.secret_id[\s\S]*?version = data\.external\.relay_serving_regional_placement_version\.result\.version/ + ) + assert.match(relay, /data "external" "relay_serving_regional_placement_version"/) + assert.match(relay, /read-relay-serving-regional-placement-version\.mjs/) + assert.doesNotMatch(relay, /template\[0\]\.containers\[0\]\.env/) +}) + +test('validates the final stamped-cell admission state', () => { + const required = [ + '--project', + 'project', + '--region', + 'region', + '--service', + 'service', + '--image', + 'image', + '--role', + 'cell', + '--release-id', + 'release', + '--director-origin', + 'https://relay.example.com', + '--admin-audience', + 'https://relay.example.com/v1/admin/drain' + ] + + assert.equal(parseArguments(required)['final-admission'], undefined) + assert.equal( + parseArguments([...required, '--final-admission', 'disabled'])['final-admission'], + 'disabled' + ) + assert.throws(() => parseArguments([...required, '--final-admission', 'sometimes'])) + assert.equal(parseArguments([...required, '--min-instances', '0'])['min-instances'], '0') + assert.throws(() => parseArguments([...required, '--min-instances', '-1'])) + assert.throws(() => + parseArguments([...required, '--capacity-service-account', 'relay@example.com']) + ) +}) + +test('validates optional director capacity configuration', () => { + const cells = [ + { + id: 'staging-gce-c3', + url: 'https://c3.relay-staging.onorca.dev', + capacityRequests: 4_000, + initiallyEnabled: false, + region: 'us-central1', + connectionHardCap: 600, + connectionUnobservedBound: 60 + } + ] + const config = { + project: 'onorca-cloud-staging', + 'capacity-service-account': + 'orca-cloud-staging-gha-cap@onorca-cloud-staging.iam.gserviceaccount.com', + 'director-cells-json': JSON.stringify(cells) + } + assert.deepEqual(directorDeploymentEnvironment(config), { + ...DIRECTOR_ADMISSION_ENVIRONMENT, + ORCA_RELAY_ADMISSION_SELECTOR_VERSION: '3', + ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT: + 'orca-cloud-staging-gha-cap@onorca-cloud-staging.iam.gserviceaccount.com', + ORCA_RELAY_CELLS_JSON: JSON.stringify([ + { + id: cells[0].id, + url: cells[0].url, + capacityRequests: cells[0].capacityRequests, + region: cells[0].region, + initiallyEnabled: cells[0].initiallyEnabled, + connectionHardCap: cells[0].connectionHardCap, + connectionUnobservedBound: cells[0].connectionUnobservedBound + } + ]) + }) + assert.throws( + () => + directorDeploymentEnvironment({ + ...config, + 'capacity-service-account': 'foreign@other-project.iam.gserviceaccount.com' + }), + /selected project/ + ) + assert.throws( + () => + directorDeploymentEnvironment({ + ...config, + 'director-cells-json': JSON.stringify([{ ...cells[0], unexpected: true }]) + }), + /invalid cell/ + ) + assert.match( + environmentUpdateValue(directorDeploymentEnvironment(config)), + /^\^~\^ORCA_RELAY_DATABASE_POOL_MAX=/ + ) + assert.equal(environmentUpdateValue({ FIRST: 'one', SECOND: 'two' }), 'FIRST=one,SECOND=two') + assert.deepEqual( + directorTopologyChange( + JSON.stringify([ + { + capacityRequests: 4_000, + connectionHardCap: 600, + connectionUnobservedBound: 60, + id: 'staging-gce-c3', + initiallyEnabled: false, + region: 'us-central1', + url: 'https://c3.relay-staging.onorca.dev' + } + ]), + JSON.stringify([{ ...cells[0], connectionHardCap: 1_000 }]), + 'staging-gce-c3' + ), + { + changed: true, + value: directorDeploymentEnvironment({ + 'director-cells-json': JSON.stringify([{ ...cells[0], connectionHardCap: 1_000 }]) + }).ORCA_RELAY_CELLS_JSON + } + ) + assert.throws( + () => + directorTopologyChange( + JSON.stringify(cells), + JSON.stringify([{ ...cells[0], url: 'https://wrong.relay-staging.onorca.dev' }]), + 'staging-gce-c3' + ), + /outside the reviewed capacity pair/ + ) +}) + +test('validates exact director runtime and regional rehome identities', () => { + const base = [ + '--project', + 'onorca-cloud', + '--region', + 'us-central1', + '--service', + 'orca-cloud-relay', + '--image', + `relay@sha256:${'a'.repeat(64)}`, + '--role', + 'director', + '--release-id', + 'rehome', + '--runtime-service-account', + 'relay-director@onorca-cloud.iam.gserviceaccount.com', + '--rehome-director-service-account', + 'relay-director@onorca-cloud.iam.gserviceaccount.com', + '--rehome-audience', + 'https://relay.onorca.dev/v1/admin/host-drain', + '--expected-rehome-generation', + '7', + '--rehome-control-origin', + 'https://relay.onorca.dev', + '--admin-audience', + 'https://relay.onorca.dev/v1/admin/drain' + ] + const config = parseArguments(base) + assert.deepEqual(directorDeploymentEnvironment(config), { + ...DIRECTOR_ADMISSION_ENVIRONMENT, + ORCA_RELAY_ADMISSION_SELECTOR_VERSION: '3', + ORCA_RELAY_IMAGE_DIGEST: `sha256:${'a'.repeat(64)}`, + [DIRECTOR_REHOME_IDENTITY_ENV]: + 'relay-director@onorca-cloud.iam.gserviceaccount.com', + [DIRECTOR_REHOME_AUDIENCE_ENV]: + 'https://relay.onorca.dev/v1/admin/host-drain' + }) + const missingAudience = [...base] + missingAudience.splice(missingAudience.indexOf('--rehome-audience'), 2) + assert.throws(() => parseArguments(missingAudience), /configured together/) + const invalidOrigin = [...base] + invalidOrigin[invalidOrigin.indexOf('--rehome-control-origin') + 1] = + 'http://relay.onorca.dev' + assert.throws( + () => parseArguments(invalidOrigin), + /HTTPS origin/ + ) + const mutableImage = [...base] + mutableImage[mutableImage.indexOf('--image') + 1] = 'relay:latest' + assert.throws(() => parseArguments(mutableImage), /immutable digest/) +}) + +test('requires durable regional rehome control to be disabled at the exact generation', async () => { + const config = { + 'admin-audience': 'https://relay.onorca.dev/v1/admin/drain', + 'expected-rehome-generation': '7' + } + const environment = process.env.ORCA_RELAY_ADMIN_ID_TOKEN + process.env.ORCA_RELAY_ADMIN_ID_TOKEN = 'aaa.bbb.ccc' + try { + const control = await assertRegionalRehomeDisabled( + config, + 'https://candidate.example.test', + async (url, init) => { + assert.equal(url, 'https://candidate.example.test/v1/admin/regional-rehome-control') + assert.equal(init.headers.authorization, 'Bearer aaa.bbb.ccc') + return new Response(JSON.stringify({ + v: 1, + control: { generation: 7, enabled: false } + })) + } + ) + assert.equal(control.generation, 7) + await assert.rejects( + assertRegionalRehomeDisabled(config, 'https://candidate.example.test', async () => + new Response(JSON.stringify({ + v: 1, + control: { generation: 8, enabled: false } + })) + ), + /expected generation/ + ) + await assert.rejects( + assertRegionalRehomeDisabled(config, 'https://candidate.example.test', async () => + new Response(JSON.stringify({ + v: 1, + control: { generation: 7, enabled: true } + })) + ), + /durably disabled/ + ) + } finally { + if (environment === undefined) delete process.env.ORCA_RELAY_ADMIN_ID_TOKEN + else process.env.ORCA_RELAY_ADMIN_ID_TOKEN = environment + } +}) + +test('rejects literal regional placement changes outside the runtime-setting step', () => { + const base = { + project: 'onorca-cloud', + region: 'us-central1', + service: 'orca-cloud-relay', + image: `us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:${'a'.repeat(64)}`, + role: 'director', + 'release-id': 'regional-kill-switch' + } + const args = Object.entries(base).flatMap(([key, value]) => [`--${key}`, value]) + + assert.doesNotThrow(() => parseArguments(args)) + assert.throws( + () => parseArguments([...args, '--regional-placement-enabled', 'false']), + /audited runtime-setting step/ + ) +}) + +test('appends disabled Asia cells without changing the existing director topology', () => { + const current = [ + { + id: 'production-gce-c26', + url: 'https://c26.relay.onorca.dev', + capacityRequests: 4_000, + initiallyEnabled: true, + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + } + ] + const asia = { + id: 'production-gce-c27', + url: 'https://c27.relay.onorca.dev', + region: 'asia-east2', + capacityRequests: 6_000, + initiallyEnabled: false, + connectionHardCap: 3_000, + connectionUnobservedBound: 60 + } + + assert.deepEqual( + directorCellSetAddition( + JSON.stringify(current), + JSON.stringify([asia, { ...current[0], region: 'us-central1' }]) + ), + { + changed: true, + value: directorDeploymentEnvironment({ + 'director-cells-json': JSON.stringify([asia, { ...current[0], region: 'us-central1' }]) + }).ORCA_RELAY_CELLS_JSON + } + ) + const exact = JSON.stringify([{ ...current[0], region: 'us-central1' }, asia]) + assert.deepEqual(directorCellSetAddition(exact, exact), { + changed: false, + value: directorDeploymentEnvironment({ 'director-cells-json': exact }) + .ORCA_RELAY_CELLS_JSON + }) + assert.throws( + () => directorCellSetAddition(JSON.stringify(current), JSON.stringify([{ ...current[0], region: 'us-central1', capacityRequests: 6_000 }, asia])), + /changes an existing cell/ + ) + assert.throws( + () => directorCellSetAddition(JSON.stringify(current), JSON.stringify([{ ...current[0], region: 'us-central1' }, { ...asia, initiallyEnabled: true }])), + /must start disabled/ + ) +}) + +test('pins a director startup probe above the bounded reconciliation window', () => { + assert.deepEqual(directorStartupProbeArguments('director'), [ + '--startup-probe', + 'tcpSocket.port=8080,timeoutSeconds=120,periodSeconds=120,failureThreshold=1' + ]) + assert.deepEqual(directorStartupProbeArguments('cell'), []) +}) + +test('bounds traffic tags by the Cloud Run service-plus-tag contract', () => { + const service = 'orca-cloud-relay-staging-c1' + const candidate = cloudRunTrafficTag(service, 'candidate', '29247170608-1-19cc312a') + assert.match(candidate, /^candidate-[a-f0-9]{9}$/) + assert.equal(service.length + candidate.length, 46) + assert.equal(candidate, cloudRunTrafficTag(service, 'candidate', '29247170608-1-19cc312a')) + assert.notEqual(candidate, cloudRunTrafficTag(service, 'candidate', '29247170608-2-19cc312a')) + assert.throws(() => cloudRunTrafficTag(`${service}-too-long`, 'candidate', 'release')) +}) + +test('derives and validates a Cloud Run tagged revision origin', () => { + assert.equal( + taggedRevisionOrigin( + 'https://orca-cloud-relay-staging-c1-gjzz5mc7ka-uc.a.run.app', + 'candidate-123' + ), + 'https://candidate-123---orca-cloud-relay-staging-c1-gjzz5mc7ka-uc.a.run.app' + ) + assert.throws(() => taggedRevisionOrigin('https://relay-staging.onorca.dev', 'candidate-123')) + assert.throws(() => + taggedRevisionOrigin( + 'https://orca-cloud-relay-staging-c1-gjzz5mc7ka-uc.a.run.app', + '123-invalid' + ) + ) +}) + +test('requires exactly one active revision and reads queried tag metadata', () => { + const service = { + status: { + traffic: [ + { percent: 100, revisionName: 'relay-00001-old' }, + { + percent: 0, + revisionName: 'relay-00002-new', + tag: 'candidate-123', + url: 'https://candidate-123---relay-hash-uc.a.run.app' + } + ] + } + } + assert.equal(activeRevision(service), 'relay-00001-old') + assert.deepEqual(taggedTraffic(service, 'candidate-123'), { + origin: 'https://candidate-123---relay-hash-uc.a.run.app', + revision: 'relay-00002-new' + }) + assert.throws(() => + activeRevision({ status: { traffic: [{ percent: 50 }, { percent: 50 }] } }) + ) + assert.deepEqual(trafficTags(service), ['candidate-123']) +}) + +function directorHarness({ + deployFailure, + deleteFailure, + cleanupReportsFailure = false, + cleanupFailsBeforeRemoval = false, + servingMinimum = 1, + servingMaximum = 5, + // Reproduces gcloud dropping minScale from a newly created revision. + dropRequestedMinimum = false, + servingServiceAccount, + servingImageDigest = `sha256:${'f'.repeat(64)}` +} = {}) { + const state = { + activeRevision: 'relay-00001-old', + tags: new Map([['candidate-old', 'relay-00000-stale']]), + revisions: new Map([ + ['relay-00000-stale', { env: { ORCA_RELAY_ROLE: 'director' }, minimum: 1, maximum: 5 }], + [ + 'relay-00001-old', + { + env: { + ORCA_RELAY_ROLE: 'director' + }, + secrets: { + [DIRECTOR_REGIONAL_PLACEMENT_ENV]: { + secret: DIRECTOR_REGIONAL_PLACEMENT_SECRET, + version: '1' + } + }, + minimum: servingMinimum, + maximum: servingMaximum, + serviceAccount: servingServiceAccount, + image: `relay@${servingImageDigest}` + } + ] + ]), + nextRevision: 2 + } + const removed = [] + const healthProtocols = [] + let pendingCleanupFailure = cleanupFailsBeforeRemoval + const operations = { + describeService: () => ({ + status: { + traffic: [ + { percent: 100, revisionName: state.activeRevision }, + ...[...state.tags].map(([tag, revisionName]) => ({ + tag, + revisionName, + url: `https://${tag}---relay-hash-uc.a.run.app` + })) + ] + } + }), + describeRevision: (_config, revision) => { + const value = state.revisions.get(revision) + return { + metadata: { + annotations: { + 'autoscaling.knative.dev/minScale': String(value?.minimum ?? 0), + 'autoscaling.knative.dev/maxScale': String(value?.maximum ?? 5) + } + }, + spec: { + serviceAccountName: value?.serviceAccount, + containers: [ + { + image: value?.image, + env: [ + ...Object.entries(value?.env ?? {}).map(([name, value]) => ({ name, value })), + ...Object.entries(value?.secrets ?? {}).map(([name, secretKeyRef]) => ({ + name, + valueSource: { secretKeyRef } + })) + ] + } + ] + } + } + }, + deployCandidate: (config, tag, env, image, minimum, maximum, regionalVersion) => { + const revision = `relay-${String(state.nextRevision++).padStart(5, '0')}-new` + state.revisions.set(revision, { + env: { ORCA_RELAY_ROLE: 'director', ...env }, + secrets: { + [DIRECTOR_REGIONAL_PLACEMENT_ENV]: { + secret: DIRECTOR_REGIONAL_PLACEMENT_SECRET, + version: regionalVersion + } + }, + minimum: dropRequestedMinimum ? 0 : Number(minimum ?? 1), + maximum, + serviceAccount: + config['runtime-service-account'] ?? + state.revisions.get(state.activeRevision)?.serviceAccount, + image: image ?? state.revisions.get(state.activeRevision)?.image + }) + state.tags.set(tag, revision) + if (deployFailure) throw deployFailure + }, + listRevisions: () => + [...state.revisions].map(([name]) => ({ metadata: { name } })), + deleteRevision: (_config, revision) => { + if (deleteFailure) throw deleteFailure + state.revisions.delete(revision) + }, + updateTraffic: (_config, args) => { + const remove = args.find((argument) => argument.startsWith('--remove-tags=')) + if (remove) { + const tags = remove.slice('--remove-tags='.length).split(',') + removed.push(tags) + if (pendingCleanupFailure && tags.includes('candidate-new')) { + pendingCleanupFailure = false + throw new Error('failed to remove promoted candidate tag') + } + for (const tag of tags) state.tags.delete(tag) + if (cleanupReportsFailure) throw new Error('gcloud reported failed latest revision') + return + } + const promote = args.find((argument) => argument.startsWith('--to-tags=')) + assert.ok(promote) + const tag = promote.slice('--to-tags='.length).split('=')[0] + state.activeRevision = state.tags.get(tag) + }, + waitForHealth: async (_origin, connectionCapacityProtocol) => { + healthProtocols.push(connectionCapacityProtocol) + } + } + return { state, removed, healthProtocols, operations } +} + +test('director deploy removes stale and promoted Cloud Run tags', async () => { + const harness = directorHarness() + const config = { + project: 'onorca-cloud-staging', + 'capacity-service-account': + 'orca-cloud-staging-gha-cap@onorca-cloud-staging.iam.gserviceaccount.com' + } + await deployDirector(config, 'candidate-new', harness.operations) + assert.deepEqual(harness.removed, [['candidate-old'], ['candidate-new']]) + assert.equal(harness.state.activeRevision, 'relay-00003-new') + assert.deepEqual([...harness.state.tags.keys()], ['selector-rollback']) + assert.deepEqual([...harness.state.revisions.keys()], [ + 'relay-00000-stale', + 'relay-00001-old', + 'relay-00002-new', + 'relay-00003-new' + ]) + for (const revision of ['relay-00002-new', 'relay-00003-new']) { + assert.deepEqual( + Object.fromEntries( + Object.entries(harness.state.revisions.get(revision).env).filter(([key]) => + key in DIRECTOR_ADMISSION_ENVIRONMENT + ) + ), + DIRECTOR_ADMISSION_ENVIRONMENT + ) + assert.equal( + harness.state.revisions.get(revision).env.ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT, + config['capacity-service-account'] + ) + } + assert.deepEqual(harness.healthProtocols, [undefined, undefined]) +}) + +test('director deploy stamps the durable regional placement secret reference', async () => { + const harness = directorHarness() + await deployDirector({}, 'candidate-new', harness.operations) + for (const revision of ['relay-00002-new', 'relay-00003-new']) { + assert.deepEqual( + harness.state.revisions.get(revision).secrets[DIRECTOR_REGIONAL_PLACEMENT_ENV], + { secret: DIRECTOR_REGIONAL_PLACEMENT_SECRET, version: '1' } + ) + } +}) + +test('bootstraps both rollback and candidate onto the distinct director identity', async () => { + const predecessor = 'relay-runtime@onorca-cloud.iam.gserviceaccount.com' + const director = 'relay-director@onorca-cloud.iam.gserviceaccount.com' + const harness = directorHarness({ servingServiceAccount: predecessor }) + const inspected = [] + harness.operations.assertRegionalRehomeDisabled = async (_config, origin) => { + inspected.push(origin) + } + await deployDirector({ + project: 'onorca-cloud', + 'runtime-service-account': director, + 'predecessor-runtime-service-account': predecessor, + 'predecessor-image-digest': `sha256:${'f'.repeat(64)}`, + 'bootstrap-runtime-identity': 'true', + 'expected-rehome-generation': '0', + 'rehome-control-origin': 'https://relay.onorca.dev' + }, 'candidate-new', harness.operations) + assert.equal( + harness.state.revisions.get(harness.state.activeRevision).serviceAccount, + director + ) + assert.equal( + harness.state.revisions.get(harness.state.tags.get('selector-rollback')).serviceAccount, + director + ) + assert.deepEqual(inspected, [ + 'https://selector-rollback---relay-hash-uc.a.run.app', + 'https://candidate-new---relay-hash-uc.a.run.app' + ]) +}) + +test('steady-state director deploy rejects the predecessor identity', async () => { + const harness = directorHarness({ + servingServiceAccount: 'relay-runtime@onorca-cloud.iam.gserviceaccount.com' + }) + await assert.rejects(deployDirector({ + 'runtime-service-account': 'relay-director@onorca-cloud.iam.gserviceaccount.com' + }, 'candidate-new', harness.operations), /unexpected runtime service account/) +}) + +test('director deploy prunes old revisions when requested', async () => { + const harness = directorHarness() + await deployDirector({ 'prune-revisions': 'true' }, 'candidate-new', harness.operations) + assert.deepEqual([...harness.state.revisions.keys()], [ + 'relay-00002-new', + 'relay-00003-new' + ]) + assert.deepEqual(harness.healthProtocols, [2, 2]) +}) + +test('a prune failure preserves the active and rollback traffic pair', async () => { + const harness = directorHarness({ deleteFailure: new Error('injected delete failure') }) + await assert.rejects( + deployDirector({ 'prune-revisions': 'true' }, 'candidate-new', harness.operations), + /injected delete failure/ + ) + assert.equal(harness.state.activeRevision, 'relay-00003-new') + assert.deepEqual([...harness.state.tags.keys()], ['selector-rollback']) +}) + +test('director deploy removes a candidate tag left by a failed update', async () => { + const failure = new Error('candidate failed to become ready') + const harness = directorHarness({ deployFailure: failure }) + await assert.rejects(deployDirector({}, 'candidate-new', harness.operations), failure) + assert.deepEqual(harness.removed, [['candidate-old'], ['selector-rollback']]) + assert.equal(harness.state.tags.size, 0) +}) + +test('director candidate inherits the serving warm-instance floor', async () => { + const harness = directorHarness({ servingMinimum: 5 }) + await deployDirector({}, 'candidate-new', harness.operations) + const serving = harness.state.revisions.get(harness.state.activeRevision) + assert.equal(serving.minimum, 5) + // The standby rollback revision must stay cold. + const rollback = harness.state.revisions.get(harness.state.tags.get('selector-rollback')) + assert.equal(rollback.minimum, 0) +}) + +test('director deploy rejects a serving maximum above the checked budget', async () => { + const harness = directorHarness({ servingMaximum: 6 }) + await assert.rejects( + deployDirector({ 'max-instances': '5' }, 'candidate-new', harness.operations), + /holds 6 maximum instances, expected 5/ + ) +}) + +test('an explicit --min-instances still overrides the serving floor', async () => { + const harness = directorHarness({ servingMinimum: 5 }) + await deployDirector({ 'min-instances': '0' }, 'candidate-new', harness.operations) + assert.equal(harness.state.revisions.get(harness.state.activeRevision).minimum, 0) +}) + +test('director deploy refuses to move traffic onto a candidate that lost the floor', async () => { + const harness = directorHarness({ servingMinimum: 5, dropRequestedMinimum: true }) + await assert.rejects( + deployDirector({}, 'candidate-new', harness.operations), + /candidate holds 0 minimum instances, expected 5/ + ) + // Traffic never moved, so the original revision still serves. + assert.equal(harness.state.activeRevision, 'relay-00001-old') +}) + +test('director deploy rejects unrelated revision-shape drift', async () => { + const harness = directorHarness() + const describeRevision = harness.operations.describeRevision + harness.operations.describeRevision = (config, revision) => { + const described = describeRevision(config, revision) + described.spec.containerConcurrency = revision === 'relay-00001-old' ? 80 : 1_000 + return described + } + await assert.rejects( + deployDirector({}, 'candidate-new', harness.operations), + /unrelated revision shape/ + ) + assert.equal(harness.state.activeRevision, 'relay-00001-old') +}) + +test('director cleanup verifies success when gcloud reports a stale revision failure', async () => { + const harness = directorHarness({ cleanupReportsFailure: true }) + await deployDirector({}, 'candidate-new', harness.operations) + assert.deepEqual(harness.removed, [['candidate-old'], ['candidate-new']]) + assert.deepEqual([...harness.state.tags.keys()], ['selector-rollback']) +}) + +test('director deploy restores rollback traffic after promoted-tag cleanup fails', async () => { + const harness = directorHarness({ cleanupFailsBeforeRemoval: true }) + await assert.rejects( + deployDirector({}, 'candidate-new', harness.operations), + /failed to remove promoted candidate tag/ + ) + assert.equal( + harness.state.activeRevision, + harness.state.tags.get('selector-rollback') + ) + assert.deepEqual([...harness.state.tags.keys()], ['selector-rollback']) +}) + +test('reads only literal revision environment values', () => { + assert.deepEqual( + revisionEnvironment({ + spec: { + containers: [ + { + env: [ + { name: 'ORCA_RELAY_CELL_ID', value: 'staging-c1' }, + { name: 'ORCA_RELAY_CELL_CAPACITY', value: '900' }, + { name: 'DATABASE_URL', valueFrom: { secretKeyRef: { name: 'database' } } } + ] + } + ] + } + }), + { ORCA_RELAY_CELL_ID: 'staging-c1', ORCA_RELAY_CELL_CAPACITY: '900' } + ) +}) + +test('reads only Secret Manager revision environment references', () => { + assert.deepEqual( + revisionSecretEnvironment({ + spec: { + containers: [{ env: [ + { name: 'LITERAL', value: 'true' }, + { + name: DIRECTOR_REGIONAL_PLACEMENT_ENV, + valueSource: { secretKeyRef: { + secret: DIRECTOR_REGIONAL_PLACEMENT_SECRET, + version: 'latest' + } } + }, + { + name: 'GCP_SECRET_SHAPE', + valueFrom: { secretKeyRef: { name: 'gcp-secret', key: '2' } } + } + ] }] + } + }), + { + [DIRECTOR_REGIONAL_PLACEMENT_ENV]: { + secret: DIRECTOR_REGIONAL_PLACEMENT_SECRET, + version: 'latest' + }, + GCP_SECRET_SHAPE: { + secret: 'gcp-secret', + version: '2' + } + } + ) +}) + +test('accepts only a bounded JWT-shaped supplied admin identity token', () => { + assert.equal(suppliedAdminIdentityToken({}), null) + assert.equal(suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }), 'aaa.bbb.ccc') + assert.throws(() => suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: '' })) + assert.throws(() => suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: 'not-a-jwt' })) + assert.throws(() => + suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: `aaa.${'b'.repeat(8_190)}.ccc` }) + ) +}) + +test('waits for authenticated target readiness without hiding other capacity errors', async () => { + let attempts = 0 + const capacity = await waitForEvacuationCapacity( + async () => { + attempts += 1 + if (attempts < 3) throw new Error('/v1/admin/evacuation-capacity failed: target_cell_unavailable') + return { requiredTargetUnits: 2, availableTargetUnits: 4_000 } + }, + 'source', + 'target', + { pollIntervalMs: 1, timeoutMs: 100 } + ) + assert.equal(attempts, 3) + assert.equal(capacity.availableTargetUnits, 4_000) + await assert.rejects( + waitForEvacuationCapacity(async () => { + throw new Error('/v1/admin/evacuation-capacity failed: forbidden') + }, 'source', 'target'), + /forbidden/ + ) +}) diff --git a/cloud/dev/scripts/deploy-relay-gce-candidate.mjs b/cloud/dev/scripts/deploy-relay-gce-candidate.mjs new file mode 100644 index 00000000000..6a0928041c9 --- /dev/null +++ b/cloud/dev/scripts/deploy-relay-gce-candidate.mjs @@ -0,0 +1,871 @@ +import { readFileSync } from 'node:fs' +import { spawnSync } from 'node:child_process' +import { pathToFileURL } from 'node:url' +import { + inspectAdmissionSelector, + selectorCellState, + transitionAdmissionSelector +} from './relay-admission-selector.mjs' + +const DEFAULT_POLL_INTERVAL_MS = 5_000 +const DEFAULT_TIMEOUT_MS = 14 * 60 * 1_000 +const ADMIN_RETRY_ATTEMPTS = 3 +const ADMIN_RETRY_BASE_MS = 250 +const CONNECTION_CONTROL_REBIND_RESERVE = 100 +const SUPPORTED_CONNECTION_HARD_CAPS = new Set([600, 1_000, 3_000]) +const RETRYABLE_ADMIN_PATHS = new Set([ + '/v1/admin/runtime-status', + '/v1/admin/cell-status', + '/v1/admin/evacuation-capacity', + '/v1/admin/evacuation-status' +]) + +function canonicalOrigin(value, name) { + const url = new URL(value) + if (url.protocol !== 'https:' || url.origin !== value || url.pathname !== '/') { + throw new Error(`${name} must be a canonical HTTPS origin`) + } + return value +} + +function adminAudience(value) { + const url = new URL(value) + if ( + url.protocol !== 'https:' || + url.pathname !== '/v1/admin/drain' || + url.search || + url.hash || + url.toString() !== value + ) { + throw new Error('--admin-audience must be the canonical HTTPS director drain URL') + } + return value +} + +function positiveInteger(value, name, maximum = Number.MAX_SAFE_INTEGER) { + const parsed = Number(value) + if (!Number.isInteger(parsed) || parsed <= 0 || parsed > maximum) { + throw new Error(`${name} must be a positive integer`) + } + return parsed +} + +function nonnegativeInteger(value, name) { + const parsed = Number(value) + if (!Number.isInteger(parsed) || parsed < 0) { + throw new Error(`${name} must be a nonnegative integer`) + } + return parsed +} + +export function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error(`invalid argument ${key ?? ''}`) + values[key.slice(2)] = value + } + for (const key of [ + 'project', + 'director-origin', + 'admin-audience', + 'topology-file', + 'source-cell-id', + 'target-cell-id', + 'runtime-service-account', + 'mode' + ]) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if ( + ![ + 'audit', + 'preflight', + 'recover-forward', + 'continue-evacuation', + 'disable-cell', + 'execute', + 'reset-empty-candidate', + 'enable-empty-cell' + ].includes(values.mode) + ) { + throw new Error( + '--mode must be audit, preflight, recover-forward, continue-evacuation, disable-cell, execute, reset-empty-candidate, or enable-empty-cell' + ) + } + return { + project: values.project, + directorOrigin: canonicalOrigin(values['director-origin'], '--director-origin'), + adminAudience: adminAudience(values['admin-audience']), + topologyFile: values['topology-file'], + sourceCellId: values['source-cell-id'], + targetCellId: values['target-cell-id'], + runtimeServiceAccount: values['runtime-service-account'], + mode: values.mode, + batchSize: positiveInteger(values['batch-size'] ?? 100, '--batch-size', 100), + drainGraceMs: positiveInteger( + values['drain-grace-ms'] ?? 120_000, + '--drain-grace-ms', + 60 * 60 * 1_000 + ), + pollIntervalMs: positiveInteger( + values['poll-interval-ms'] ?? DEFAULT_POLL_INTERVAL_MS, + '--poll-interval-ms', + 60_000 + ), + timeoutMs: positiveInteger( + values['timeout-ms'] ?? DEFAULT_TIMEOUT_MS, + '--timeout-ms', + 60 * 60 * 1_000 + ) + } +} + +export function deployment(value, cellId) { + if (!value || typeof value !== 'object') throw new Error(`missing topology for ${cellId}`) + const expected = { + cellId, + origin: canonicalOrigin(value.origin, `${cellId} origin`), + region: String(value.region ?? 'us-central1'), + zone: String(value.zone ?? ''), + migName: String(value.mig_name ?? ''), + instanceGroup: String(value.instance_group ?? ''), + backendName: String(value.backend_name ?? ''), + backendId: String(value.backend_id ?? ''), + urlMapName: String(value.url_map_name ?? ''), + generationIdentity: String(value.generation_identity ?? ''), + image: String(value.image ?? ''), + imageDigest: String(value.image ?? '').split('@')[1] ?? '', + capacityRequests: positiveInteger(value.capacity_requests, `${cellId} capacity`), + databasePoolMax: positiveInteger( + value.database_pool_max ?? 10, + `${cellId} database pool maximum`, + 100 + ), + connectionHardCap: + value.connection_hard_cap === null || value.connection_hard_cap === undefined + ? undefined + : positiveInteger(value.connection_hard_cap, `${cellId} connection hard cap`), + connectionUnobservedBound: + value.connection_unobserved_bound === null || + value.connection_unobserved_bound === undefined + ? undefined + : nonnegativeInteger( + value.connection_unobserved_bound, + `${cellId} unobserved connection bound` + ), + initiallyEnabled: value.initially_enabled, + fenced: value.fenced, + desiredTargetSize: value.desired_target_size + } + if (!/^[a-z0-9-]+$/.test(expected.zone)) throw new Error(`${cellId} has an invalid zone`) + if (!['us-central1', 'asia-east2'].includes(expected.region) || !expected.zone.startsWith(`${expected.region}-`)) { + throw new Error(`${cellId} has an invalid region`) + } + for (const [name, resource] of [ + ['MIG', expected.migName], + ['instance group', expected.instanceGroup], + ['backend', expected.backendName], + ['backend ID', expected.backendId] + ]) { + if (!resource) throw new Error(`${cellId} has no ${name}`) + } + if (!/^[a-z0-9.-]+\/[a-z0-9._/-]+@sha256:[a-f0-9]{64}$/.test(expected.image)) { + throw new Error(`${cellId} image is not digest-pinned`) + } + if (typeof expected.initiallyEnabled !== 'boolean') { + throw new Error(`${cellId} has no initial admission state`) + } + if ( + (expected.connectionHardCap === undefined) !== + (expected.connectionUnobservedBound === undefined) || + (expected.connectionHardCap !== undefined && + (!SUPPORTED_CONNECTION_HARD_CAPS.has(expected.connectionHardCap) || + expected.connectionUnobservedBound >= + expected.connectionHardCap - CONNECTION_CONTROL_REBIND_RESERVE)) + ) { + throw new Error(`${cellId} has invalid connection capacity`) + } + return expected +} + +export function assertDeploymentConnectionCapacity(expected, runtime, director) { + if (expected.connectionHardCap === undefined) { + if (runtime !== null || director !== null) { + throw new Error(`${expected.cellId} connection capacity differs from Terraform`) + } + return + } + const hardCap = expected.connectionHardCap + const unobservedBound = expected.connectionUnobservedBound + const ordinaryConnectionLimit = hardCap - CONNECTION_CONTROL_REBIND_RESERVE + const normalAdmissionPause = ordinaryConnectionLimit - unobservedBound + const matches = (capacity) => + capacity?.hardCap === hardCap && + capacity.controlRebindReserve === CONNECTION_CONTROL_REBIND_RESERVE && + capacity.ordinaryConnectionLimit === ordinaryConnectionLimit && + capacity.unobservedBound === unobservedBound && + capacity.normalAdmissionPause === normalAdmissionPause + if (!matches(runtime) || !matches(director) || director.heartbeatFresh !== true) { + throw new Error(`${expected.cellId} connection capacity differs from Terraform`) + } +} + +export function selectDeployments(topology, sourceCellId, targetCellId) { + if (sourceCellId === targetCellId) throw new Error('source and target cell IDs must differ') + const source = deployment(topology[sourceCellId], sourceCellId) + const target = deployment(topology[targetCellId], targetCellId) + for (const key of ['origin', 'migName', 'instanceGroup', 'backendName', 'backendId']) { + if (source[key] === target[key]) throw new Error(`source and target ${key} overlap`) + } + if (target.initiallyEnabled) throw new Error('candidate must be declared initially disabled') + return { source, target } +} + +export function validateMig(mig, instances, expected) { + if (Number(mig.targetSize) !== 1) throw new Error(`${expected.cellId} MIG is not fixed-one`) + const policy = mig.updatePolicy ?? {} + if ( + policy.replacementMethod !== 'RECREATE' || + Number(policy.maxSurge?.fixed ?? policy.maxSurge) !== 0 || + Number(policy.maxUnavailable?.fixed ?? policy.maxUnavailable) !== 1 + ) { + throw new Error(`${expected.cellId} MIG replacement policy is unsafe`) + } + const serving = instances.filter( + (entry) => entry.instanceStatus === 'RUNNING' && entry.currentAction === 'NONE' + ) + if (instances.length !== 1 || serving.length !== 1) { + throw new Error(`${expected.cellId} MIG must have one running endpoint`) + } + return serving[0].instance.split('/').at(-1) +} + +export function validateInstance(instance, expected, runtimeServiceAccount) { + const publicConfigs = (instance.networkInterfaces ?? []).flatMap( + (network) => network.accessConfigs ?? [] + ) + if (publicConfigs.length !== 0) throw new Error(`${expected.cellId} instance has a public IP`) + const serviceAccounts = (instance.serviceAccounts ?? []).map((entry) => entry.email) + if (serviceAccounts.length !== 1 || serviceAccounts[0] !== runtimeServiceAccount) { + throw new Error(`${expected.cellId} runtime service account mismatch`) + } +} + +export function validateBackend(backend, expected) { + if ( + backend.protocol !== 'HTTP' || + Number(backend.timeoutSec) !== 86_400 || + (backend.backends ?? []).length !== 1 || + backend.backends[0].group !== expected.instanceGroup + ) { + throw new Error(`${expected.cellId} backend topology mismatch`) + } +} + +export function defaultCommandJson(args) { + const result = spawnSync('gcloud', args, { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }) + if (result.status !== 0) { + throw new Error(`gcloud ${args.slice(0, 4).join(' ')} failed: ${result.stderr.trim()}`) + } + return JSON.parse(result.stdout) +} + +export function suppliedAdminIdentityToken(environment = process.env) { + const token = environment.ORCA_RELAY_ADMIN_ID_TOKEN + if (token === undefined) return null + return validatedIdentityToken(token, 'admin') +} + +export function suppliedFenceMutationIdentityToken(environment = process.env) { + const token = environment.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN + if (token === undefined) return null + return validatedIdentityToken(token, 'fence mutation') +} + +function validatedIdentityToken(token, label) { + // WIF supplies a masked Google ID token because external-account gcloud cannot mint one directly. + if (token.length > 8_192 || !/^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/.test(token)) { + throw new Error(`invalid supplied ${label} identity token`) + } + return token +} + +export function defaultIdentityToken(audience) { + const supplied = suppliedAdminIdentityToken() + if (supplied !== null) return supplied + const result = spawnSync( + 'gcloud', + ['auth', 'print-identity-token', `--audiences=${audience}`], + { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] } + ) + if (result.status !== 0) throw new Error('gcloud identity-token command failed') + return result.stdout.trim() +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({ error: `http_${response.status}` })) + if (!response.ok) throw new Error(`${label} failed: ${body.error ?? response.status}`) + return body +} + +export function createAdminPost(config, deps, token) { + return async (origin, path, body) => { + const requestToken = typeof token === 'function' ? token(path) : token + for (let attempt = 1; attempt <= ADMIN_RETRY_ATTEMPTS; attempt++) { + let response + try { + response = await deps.fetch(`${origin}${path}`, { + method: 'POST', + headers: { + authorization: `Bearer ${requestToken}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }) + } catch (error) { + if (!RETRYABLE_ADMIN_PATHS.has(path) || attempt === ADMIN_RETRY_ATTEMPTS) throw error + deps.emit({ event: 'candidate_admin_retry', path, attempt, reason: 'transport' }) + await deps.wait(deps.random() * ADMIN_RETRY_BASE_MS * 2 ** (attempt - 1)) + continue + } + if ( + RETRYABLE_ADMIN_PATHS.has(path) && + ([502, 503, 504].includes(response.status) || + (path === '/v1/admin/evacuation-status' && response.status === 500)) && + attempt < ADMIN_RETRY_ATTEMPTS + ) { + // These endpoints are read-only or transactionally idempotent, so a + // lost response may be retried without widening deployment authority. + deps.emit({ + event: 'candidate_admin_retry', + path, + attempt, + reason: `http_${response.status}` + }) + await response.arrayBuffer().catch(() => undefined) + await deps.wait(deps.random() * ADMIN_RETRY_BASE_MS * 2 ** (attempt - 1)) + continue + } + return await responseJson(response, path) + } + throw new Error(`${path} retry attempts exhausted`) + } +} + +async function checkHttp(deps, origin, path) { + const response = await deps.fetch(`${origin}${path}`, { signal: AbortSignal.timeout(15_000) }) + const body = await response.json().catch(() => ({})) + if (!response.ok || body.ok !== true) throw new Error(`${origin}${path} is unavailable`) +} + +export async function inspectCell(config, deps, adminPost, expected) { + const common = ['--project', config.project, '--zone', expected.zone, '--format=json'] + const mig = deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + expected.migName, + ...common + ]) + const instances = deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'list-instances', + expected.migName, + ...common + ]) + const instanceName = validateMig(mig, instances, expected) + const instance = deps.commandJson([ + 'compute', + 'instances', + 'describe', + instanceName, + ...common + ]) + validateInstance(instance, expected, config.runtimeServiceAccount) + const backend = deps.commandJson([ + 'compute', + 'backend-services', + 'describe', + expected.backendName, + '--global', + '--project', + config.project, + '--format=json' + ]) + validateBackend(backend, expected) + await checkHttp(deps, expected.origin, '/health') + await checkHttp(deps, expected.origin, '/ready') + const runtime = await adminPost(expected.origin, '/v1/admin/runtime-status', { v: 1 }) + if ( + runtime.role !== 'cell' || + runtime.cellId !== expected.cellId || + runtime.cellUrl !== expected.origin || + (runtime.region ?? 'us-central1') !== expected.region || + runtime.imageDigest !== expected.imageDigest + ) { + throw new Error(`${expected.cellId} served runtime does not match Terraform topology`) + } + const status = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: expected.cellId + }) + if ( + status.status?.cellUrl !== expected.origin || + (status.status?.region ?? 'us-central1') !== expected.region || + status.status?.runtime?.cellUrl !== expected.origin || + status.status?.runtime?.ready !== true || + status.status?.runtime?.heartbeatFresh !== true + ) { + throw new Error(`${expected.cellId} has no fresh ready authenticated heartbeat`) + } + assertDeploymentConnectionCapacity( + expected, + runtime.connectionCapacity ?? null, + status.status.connectionCapacity ?? null + ) + return { + ...status.status, + draining: runtime.draining === true, + process: runtime.runtime ?? null, + runtimeConnectionCapacity: runtime.connectionCapacity ?? null + } +} + +async function waitForMigration(config, deps, adminPost, completeReady) { + const deadline = deps.now() + config.timeoutMs + while (deps.now() < deadline) { + const status = await adminPost(config.directorOrigin, '/v1/admin/evacuation-status', { + v: 1, + sourceCellId: config.sourceCellId, + targetCellId: config.targetCellId, + completeReady + }) + deps.emit({ event: completeReady ? 'migration_completion' : 'migration_registration', ...status }) + if (completeReady ? status.inProgress === 0 : status.inProgress === status.targetRegistered) { + return status + } + if ( + completeReady && + status.targetRegistered === status.inProgress && + status.registeredSourceActive === 0 && + status.registeredCompletable === 0 && + status.registeredTargetInactive === status.inProgress + ) { + // CI waiting cannot revive an offline desktop; keep its proven migration + // pending until that target control reconnects. + return status + } + await deps.wait(config.pollIntervalMs) + } + throw new Error('timed out waiting for candidate migration') +} + +export async function setCellState(config, adminPost, cellId, enabled) { + const post = async (path, body) => await adminPost(config.directorOrigin, path, body) + const inspected = await inspectAdmissionSelector(post) + if (inspected.selector.generation > 0) { + await transitionAdmissionSelector(post, { + [cellId]: enabled ? 'general' : 'existing-only' + }) + return + } + await post('/v1/admin/cell-state', { v: 1, cellId, enabled }) +} + +function assertNoDurableActivity(status, operation) { + const activity = [ + status.assignments, + status.activityLeases, + status.reservedRequests, + status.outgoingMigrations, + status.incomingMigrations + ] + if (activity.some((value) => Number(value) !== 0)) { + throw new Error(`${operation} requires zero durable activity`) + } +} + +async function recoverCandidateFailure( + config, + deps, + adminPost, + source, + target, + allowEmptyAdmissionRollback, + selectorActive +) { + const status = await adminPost(config.directorOrigin, '/v1/admin/evacuation-status', { + v: 1, + sourceCellId: source.cellId, + targetCellId: target.cellId, + completeReady: false + }).catch(() => null) + if (!selectorActive && allowEmptyAdmissionRollback && status?.inProgress === 0) { + await setCellState(config, adminPost, source.cellId, true).catch(() => undefined) + await setCellState(config, adminPost, target.cellId, false).catch(() => undefined) + return + } + if (!selectorActive && status && status.inProgress > 0 && status.targetRegistered === 0) { + await setCellState(config, adminPost, source.cellId, true).catch(() => undefined) + deps.emit({ + event: 'candidate_rollback_waiting_for_lease_expiry', + sourceCellId: source.cellId, + targetCellId: target.cellId, + inProgress: status.inProgress + }) + return + } + deps.emit({ + event: 'candidate_forward_recovery_required', + sourceCellId: source.cellId, + targetCellId: target.cellId, + targetRegistered: status?.targetRegistered ?? null + }) +} + +export async function drainSource( + config, + deps, + token, + source, + graceMs = config.drainGraceMs, + traceValue +) { + const response = await deps.fetch(`${source.origin}/v1/admin/drain`, { + method: 'POST', + headers: { + authorization: `Bearer ${token}`, + 'content-type': 'application/json', + ...(traceValue ? { 'x-orca-drain-trace': traceValue } : {}) + }, + body: JSON.stringify({ v: 1, graceMs }), + signal: AbortSignal.timeout(30_000) + }) + await responseJson(response, 'source drain') + return { + backendStatus: response.status, + backendInstance: response.headers.get('x-orca-backend-instance') ?? undefined + } +} + +async function verifyCandidateCompletion( + config, + adminPost, + source, + target, + event, + eventName = 'candidate_complete' +) { + const finalSource = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: source.cellId + }) + const finalTarget = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: target.cellId + }) + if ( + finalSource.status.activityLeases !== 0 || + finalSource.status.reservedRequests !== 0 || + finalSource.status.outgoingMigrations !== 0 || + finalSource.status.runtime?.observedRequests !== 0 || + finalTarget.status.incomingMigrations !== 0 || + finalTarget.status.reservedRequests !== finalTarget.status.activityRequestUnits + ) { + throw new Error('aggregate post-migration counts are not reconciled') + } + event({ + event: eventName, + sourceCellId: source.cellId, + targetCellId: target.cellId, + dormantSourceAssignments: finalSource.status.assignments, + targetAssignments: finalTarget.status.assignments, + targetActivityLeases: finalTarget.status.activityLeases, + targetReservedRequests: finalTarget.status.reservedRequests + }) +} + +export async function runCandidateDeployment(config, overrides = {}) { + const deps = { + commandJson: overrides.commandJson ?? defaultCommandJson, + identityToken: overrides.identityToken ?? defaultIdentityToken, + fetch: overrides.fetch ?? fetch, + emit: + overrides.emit ?? + ((event) => process.stdout.write(`${JSON.stringify(event)}\n`)), + now: overrides.now ?? Date.now, + wait: overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))), + random: overrides.random ?? Math.random + } + const topology = JSON.parse(readFileSync(config.topologyFile, 'utf8')) + const { source, target } = selectDeployments( + topology, + config.sourceCellId, + config.targetCellId + ) + const token = deps.identityToken(config.adminAudience) + const adminPost = createAdminPost(config, deps, token) + const selectorPost = async (path, body) => + await adminPost(config.directorOrigin, path, body) + const selectorInspection = await inspectAdmissionSelector(selectorPost) + const selectorActive = selectorInspection.selector.generation > 0 + const sourceStatus = await inspectCell(config, deps, adminPost, source) + const targetStatus = await inspectCell(config, deps, adminPost, target) + const sourceAdmission = selectorActive + ? selectorCellState(selectorInspection.selector, source.cellId) + : sourceStatus.enabled + ? 'general' + : 'existing-only' + const targetAdmission = selectorActive + ? selectorCellState(selectorInspection.selector, target.cellId) + : targetStatus.enabled + ? 'general' + : 'existing-only' + if (config.mode === 'audit') { + const migration = await adminPost(config.directorOrigin, '/v1/admin/evacuation-status', { + v: 1, + sourceCellId: source.cellId, + targetCellId: target.cellId, + completeReady: false + }) + // Forward recovery needs durable aggregate evidence without exposing assignment identities. + deps.emit({ + event: 'candidate_audit', + source: aggregateCellStatus(sourceStatus), + target: aggregateCellStatus(targetStatus), + migration + }) + return + } + if (config.mode === 'recover-forward') { + if ( + selectorActive + ? sourceAdmission !== 'existing-only' || targetAdmission !== 'migration-only' + : sourceStatus.enabled || !targetStatus.enabled + ) { + throw new Error( + selectorActive + ? 'forward recovery requires existing-only source and migration-only target' + : 'forward recovery requires disabled source and enabled target' + ) + } + await drainSource(config, deps, token, source) + await waitForMigration(config, deps, adminPost, false) + const completion = await waitForMigration(config, deps, adminPost, true) + if (completion.inProgress > 0) { + deps.emit({ + event: 'candidate_forward_pending', + sourceCellId: source.cellId, + targetCellId: target.cellId, + inProgress: completion.inProgress, + registeredSourceActive: completion.registeredSourceActive, + registeredCompletable: completion.registeredCompletable, + registeredTargetInactive: completion.registeredTargetInactive + }) + throw new Error('forward recovery remains pending for inactive target controls') + } + await verifyCandidateCompletion( + config, + adminPost, + source, + target, + deps.emit, + 'candidate_forward_recovered' + ) + return + } + if (config.mode === 'disable-cell') { + if (targetStatus.enabled) await setCellState(config, adminPost, target.cellId, false) + // Disabling new admission preserves origin-owned sessions and durable recovery work. + deps.emit({ + event: 'cell_admission_disabled', + targetCellId: target.cellId, + changed: targetStatus.enabled, + assignments: targetStatus.assignments, + activityLeases: targetStatus.activityLeases, + reservedRequests: targetStatus.reservedRequests, + outgoingMigrations: targetStatus.outgoingMigrations, + incomingMigrations: targetStatus.incomingMigrations + }) + return + } + // Repair is safe only before a candidate owns assignments or origin-scoped work. + if (config.mode === 'reset-empty-candidate') { + assertNoDurableActivity(targetStatus, 'candidate admission reset') + if (targetStatus.enabled) await setCellState(config, adminPost, target.cellId, false) + deps.emit({ + event: 'candidate_admission_reset', + targetCellId: target.cellId, + changed: targetStatus.enabled + }) + return + } + if (config.mode === 'enable-empty-cell') { + assertNoDurableActivity(targetStatus, 'cell admission enable') + if (selectorActive ? targetAdmission === 'general' : targetStatus.enabled) { + deps.emit({ event: 'cell_admission_enabled', targetCellId: target.cellId, changed: false }) + return + } + } + const continuingEvacuation = config.mode === 'continue-evacuation' + if (continuingEvacuation) { + if ( + selectorActive + ? targetAdmission !== 'migration-only' + : !targetStatus.enabled + ) { + throw new Error( + selectorActive + ? 'continued evacuation requires migration-only target' + : 'continued evacuation requires enabled target' + ) + } + } else if (!selectorActive && targetStatus.enabled) { + throw new Error('candidate cell is already enabled') + } else if ( + selectorActive && + !['migration-only', 'existing-only'].includes(targetAdmission) + ) { + throw new Error('candidate cell must not be generally admitted') + } + const capacity = await adminPost(config.directorOrigin, '/v1/admin/evacuation-capacity', { + v: 1, + sourceCellId: source.cellId, + targetCellId: target.cellId + }) + if (capacity.requiredTargetUnits > capacity.availableTargetUnits) { + throw new Error('candidate lacks survivor request-unit headroom') + } + deps.emit({ + event: 'candidate_preflight', + mode: config.mode, + sourceCellId: source.cellId, + targetCellId: target.cellId, + sourceOrigin: source.origin, + targetOrigin: target.origin, + sourceMig: source.migName, + targetMig: target.migName, + sourceBackend: source.backendName, + targetBackend: target.backendName, + sourceDigest: source.imageDigest, + targetDigest: target.imageDigest, + sourceAssignments: capacity.sourceAssignments, + requiredTargetUnits: capacity.requiredTargetUnits, + availableTargetUnits: capacity.availableTargetUnits + }) + if (config.mode === 'preflight') return + if (config.mode === 'enable-empty-cell') { + await setCellState(config, adminPost, target.cellId, true) + deps.emit({ event: 'cell_admission_enabled', targetCellId: target.cellId, changed: true }) + return + } + // Fresh execution starts from source-only admission; continuation preserves its target. + if ( + !continuingEvacuation && + (selectorActive ? sourceAdmission !== 'existing-only' : !sourceStatus.enabled) + ) { + throw new Error( + selectorActive ? 'source cell is not existing-only' : 'source cell is not enabled' + ) + } + if (selectorActive && targetAdmission !== 'migration-only') { + throw new Error('target cell is not migration-only') + } + + let migrationsStarted = 0 + try { + if (!selectorActive) { + if (sourceStatus.enabled) await setCellState(config, adminPost, source.cellId, false) + if (!targetStatus.enabled) await setCellState(config, adminPost, target.cellId, true) + } + for (;;) { + const result = await adminPost(config.directorOrigin, '/v1/admin/evacuate-cell', { + v: 1, + sourceCellId: source.cellId, + targetCellId: target.cellId, + limit: config.batchSize + }) + migrationsStarted += result.started + deps.emit({ event: 'migration_batch', started: result.started, totalStarted: migrationsStarted }) + if (result.started === 0) break + } + } catch (error) { + await recoverCandidateFailure( + config, + deps, + adminPost, + source, + target, + !continuingEvacuation, + selectorActive + ) + throw error + } + + try { + if (selectorActive) { + const currentSelector = await inspectAdmissionSelector(selectorPost) + if ( + currentSelector.selector.generation !== selectorInspection.selector.generation || + JSON.stringify(currentSelector.selector.membership) !== + JSON.stringify(selectorInspection.selector.membership) + ) { + throw new Error('admission selector changed before drain') + } + } + await drainSource(config, deps, token, source) + await waitForMigration(config, deps, adminPost, false) + const completion = await waitForMigration(config, deps, adminPost, true) + if (completion.inProgress > 0) { + throw new Error('candidate migration remains pending for inactive target controls') + } + } catch (error) { + // A completion response can be lost after its transaction commits. Never + // reverse admission here merely because no in-progress row remains. + await recoverCandidateFailure( + config, + deps, + adminPost, + source, + target, + false, + selectorActive + ) + throw error + } + + await verifyCandidateCompletion(config, adminPost, source, target, deps.emit) +} + +export function aggregateCellStatus(status) { + return { + cellId: status.cellId, + enabled: status.enabled, + assignments: status.assignments, + activityLeases: status.activityLeases, + activityRequestUnits: status.activityRequestUnits, + reservedRequests: status.reservedRequests, + outgoingMigrations: status.outgoingMigrations, + incomingMigrations: status.incomingMigrations, + runtimeReady: status.runtime?.ready ?? false, + heartbeatFresh: status.runtime?.heartbeatFresh ?? false, + observedRequests: status.runtime?.observedRequests ?? null + } +} + +export async function main(argv = process.argv.slice(2)) { + await runCandidateDeployment(parseArguments(argv)) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/deploy-relay-gce-candidate.test.mjs b/cloud/dev/scripts/deploy-relay-gce-candidate.test.mjs new file mode 100644 index 00000000000..3eccd1cf191 --- /dev/null +++ b/cloud/dev/scripts/deploy-relay-gce-candidate.test.mjs @@ -0,0 +1,889 @@ +import assert from 'node:assert/strict' +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { test } from 'node:test' +import { + assertDeploymentConnectionCapacity, + createAdminPost, + deployment, + parseArguments, + runCandidateDeployment, + selectDeployments, + suppliedAdminIdentityToken, + validateBackend, + validateInstance, + validateMig +} from './deploy-relay-gce-candidate.mjs' + +const runtimeServiceAccount = 'orca-relay@example.iam.gserviceaccount.com' +const digestA = `sha256:${'a'.repeat(64)}` +const digestB = `sha256:${'b'.repeat(64)}` + +test('retries evacuation-status HTTP 500 without widening other admin retries', async () => { + const events = [] + let calls = 0 + const adminPost = createAdminPost({}, { + fetch: async () => { + calls++ + return calls === 1 + ? new Response(JSON.stringify({ error: 'transient' }), { status: 500 }) + : new Response(JSON.stringify({ ok: true }), { status: 200 }) + }, + emit: (event) => events.push(event), + wait: async () => {}, + random: () => 0 + }, 'token') + assert.deepEqual( + await adminPost('https://relay.example', '/v1/admin/evacuation-status', {}), + { ok: true } + ) + assert.equal(calls, 2) + assert.deepEqual(events, [{ + event: 'candidate_admin_retry', + path: '/v1/admin/evacuation-status', + attempt: 1, + reason: 'http_500' + }]) + + calls = 0 + await assert.rejects( + createAdminPost({}, { + fetch: async () => { + calls++ + return new Response(JSON.stringify({ error: 'persistent' }), { status: 500 }) + }, + emit: () => {}, + wait: async () => {}, + random: () => 0 + }, 'token')('https://relay.example', '/v1/admin/runtime-status', {}), + /runtime-status failed: persistent/ + ) + assert.equal(calls, 1) +}) + +test('verifies a 1,000-cap cell against both runtime and director telemetry', () => { + const expected = deployment( + { + ...topology().target, + connection_hard_cap: 1_000, + connection_unobserved_bound: 60 + }, + 'target' + ) + const capacity = { + hardCap: 1_000, + controlRebindReserve: 100, + ordinaryConnectionLimit: 900, + unobservedBound: 60, + normalAdmissionPause: 840 + } + assert.doesNotThrow(() => + assertDeploymentConnectionCapacity(expected, capacity, { + ...capacity, + heartbeatFresh: true + }) + ) + assert.throws( + () => + assertDeploymentConnectionCapacity(expected, capacity, { + ...capacity, + hardCap: 600, + heartbeatFresh: true + }), + /differs from Terraform/ + ) +}) + +test('verifies a 3,000-cap regional cell with a 2,840 placement boundary', () => { + const expected = deployment( + { + ...topology().target, + connection_hard_cap: 3_000, + connection_unobserved_bound: 60 + }, + 'target' + ) + const capacity = { + hardCap: 3_000, + controlRebindReserve: 100, + ordinaryConnectionLimit: 2_900, + unobservedBound: 60, + normalAdmissionPause: 2_840 + } + + assert.doesNotThrow(() => + assertDeploymentConnectionCapacity(expected, capacity, { + ...capacity, + heartbeatFresh: true + }) + ) +}) + +function topology() { + return { + source: { + origin: 'https://c1.relay.example.com', + zone: 'us-central1-b', + mig_name: 'relay-c1', + instance_group: 'https://compute.example/instanceGroups/relay-c1', + backend_name: 'relay-c1', + backend_id: 'https://compute.example/backendServices/relay-c1', + image: `us-central1-docker.pkg.dev/project/repo/relay@${digestA}`, + capacity_requests: 4_000, + initially_enabled: true + }, + target: { + origin: 'https://c2.relay.example.com', + zone: 'us-central1-c', + mig_name: 'relay-c2', + instance_group: 'https://compute.example/instanceGroups/relay-c2', + backend_name: 'relay-c2', + backend_id: 'https://compute.example/backendServices/relay-c2', + image: `us-central1-docker.pkg.dev/project/repo/relay@${digestB}`, + capacity_requests: 4_000, + initially_enabled: false + } + } +} + +function withTopology(operation) { + const directory = mkdtempSync(join(tmpdir(), 'relay-gce-candidate-')) + const file = join(directory, 'topology.json') + writeFileSync(file, JSON.stringify(topology())) + return Promise.resolve(operation(file)).finally(() => rmSync(directory, { recursive: true })) +} + +function config(topologyFile, mode = 'preflight') { + return { + project: 'test-project', + directorOrigin: 'https://relay.example.com', + adminAudience: 'https://relay.example.com/v1/admin/drain', + topologyFile, + sourceCellId: 'source', + targetCellId: 'target', + runtimeServiceAccount, + mode, + batchSize: 100, + drainGraceMs: 120_000, + pollIntervalMs: 1, + timeoutMs: 1_000 + } +} + +function response(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' } + }) +} + +function migrationResponse(body) { + return response({ + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0, + ...body + }) +} + +function fakeCommand(args) { + const name = args[args.indexOf('describe') + 1] ?? args[args.indexOf('list-instances') + 1] + if (args.includes('list-instances')) { + return [ + { + instance: `https://compute.example/instances/${name}-vm`, + instanceStatus: 'RUNNING', + currentAction: 'NONE' + } + ] + } + if (args.includes('instance-groups')) { + return { + targetSize: 1, + updatePolicy: { + replacementMethod: 'RECREATE', + maxSurge: { fixed: 0 }, + maxUnavailable: { fixed: 1 } + } + } + } + if (args.includes('instances')) { + return { + networkInterfaces: [{ networkIP: '10.42.0.2' }], + serviceAccounts: [{ email: runtimeServiceAccount }] + } + } + const cell = name === 'relay-c1' ? topology().source : topology().target + return { protocol: 'HTTP', timeoutSec: 86_400, backends: [{ group: cell.instance_group }] } +} + +function harness({ + failTargetEnable = false, + failDrain = false, + failAfterRegistration = false, + failCompletionResponse = false, + sourceEnabled = true, + targetEnabled = false, + targetAssignments = 0, + migrationInProgress = 0, + migrationTargetRegistered = 0, + migrationTargetInactive = 0, + transientCompletionFailures = 0, + dormantSourceAssignments = 0, + selectorGeneration = 0 +} = {}) { + const state = { + source: { + enabled: selectorGeneration > 0 ? false : sourceEnabled, + assignments: 2, + activityLeases: 2 + }, + target: { + enabled: selectorGeneration > 0 ? true : targetEnabled, + assignments: targetAssignments, + activityLeases: targetAssignments + } + } + const events = [] + const stateChanges = [] + let batch = 0 + let migrationCompleted = false + let remainingTransientCompletionFailures = transientCompletionFailures + const fetch = async (url, options = {}) => { + const parsed = new URL(url) + if (parsed.pathname === '/health' || parsed.pathname === '/ready') return response({ ok: true }) + const body = JSON.parse(options.body ?? '{}') + if (parsed.pathname === '/v1/admin/runtime-status') { + const target = parsed.origin.includes('c2.') + return response({ + v: 1, + role: 'cell', + cellId: target ? 'target' : 'source', + cellUrl: parsed.origin, + imageDigest: target ? digestB : digestA + }) + } + if (parsed.pathname === '/v1/admin/admission-selector/status') { + return response({ + v: 1, + selector: { + generation: selectorGeneration, + attemptId: null, + membership: { + existingOnly: + selectorGeneration > 0 + ? ['source'] + : Object.keys(state).filter((cellId) => !state[cellId].enabled), + migrationOnly: selectorGeneration > 0 ? ['target'] : [], + general: + selectorGeneration > 0 + ? [] + : Object.keys(state).filter((cellId) => state[cellId].enabled) + } + }, + intent: null + }) + } + if (parsed.pathname === '/v1/admin/cell-status') { + const cell = state[body.cellId] + return response({ + v: 1, + status: { + cellId: body.cellId, + cellUrl: topology()[body.cellId].origin, + enabled: cell.enabled, + admissionState: + selectorGeneration > 0 + ? body.cellId === 'source' + ? 'existing-only' + : 'migration-only' + : cell.enabled + ? 'general' + : 'existing-only', + assignments: cell.assignments, + activityLeases: cell.activityLeases, + activityRequestUnits: cell.activityLeases, + reservedRequests: cell.activityLeases, + outgoingMigrations: 0, + incomingMigrations: 0, + runtime: { + cellUrl: topology()[body.cellId].origin, + ready: true, + heartbeatFresh: true, + observedRequests: cell.activityLeases + } + } + }) + } + if (parsed.pathname === '/v1/admin/evacuation-capacity') { + return response({ sourceAssignments: 2, requiredTargetUnits: 4, availableTargetUnits: 4_000 }) + } + if (parsed.pathname === '/v1/admin/cell-state') { + stateChanges.push([body.cellId, body.enabled]) + if (failTargetEnable && body.cellId === 'target' && body.enabled) { + return response({ error: 'injected_enable_failure' }, 409) + } + state[body.cellId].enabled = body.enabled + return response({ ok: true }) + } + if (parsed.pathname === '/v1/admin/evacuate-cell') { + if (failAfterRegistration && batch > 0) { + return response({ error: 'injected_batch_failure' }, 503) + } + const started = batch++ === 0 ? 2 : 0 + return response({ v: 1, started }) + } + if (parsed.pathname === '/v1/admin/evacuation-status') { + if (failAfterRegistration) { + return migrationResponse({ + v: 1, + inProgress: 2, + targetRegistered: 1, + registeredSourceActive: 1, + completed: 0, + blocked: 1 + }) + } + if (failDrain) { + return migrationResponse({ + v: 1, + inProgress: 2, + targetRegistered: 0, + completed: 0, + blocked: 0 + }) + } + if (body.completeReady) { + state.source.assignments = dormantSourceAssignments + state.source.activityLeases = 0 + state.target.assignments = 2 + state.target.activityLeases = 2 + if (remainingTransientCompletionFailures > 0) { + remainingTransientCompletionFailures-- + throw new TypeError('injected transient fetch failure') + } + if (migrationTargetInactive > 0) { + return migrationResponse({ + v: 1, + inProgress: migrationTargetInactive, + targetRegistered: migrationTargetInactive, + registeredTargetInactive: migrationTargetInactive, + completed: 0, + blocked: migrationTargetInactive + }) + } + migrationCompleted = true + if (failCompletionResponse) { + return response({ error: 'injected_completion_response_failure' }, 503) + } + return migrationResponse({ + v: 1, + inProgress: 0, + targetRegistered: 0, + completed: 2, + blocked: 0 + }) + } + if (migrationCompleted) { + return migrationResponse({ + v: 1, + inProgress: 0, + targetRegistered: 0, + completed: 0, + blocked: 0 + }) + } + if (batch === 0) { + return migrationResponse({ + v: 1, + inProgress: migrationInProgress, + targetRegistered: migrationTargetRegistered, + completed: 0, + blocked: 0 + }) + } + return migrationResponse({ + v: 1, + inProgress: 2, + targetRegistered: 2, + registeredCompletable: 2, + completed: 0, + blocked: 0 + }) + } + if (parsed.pathname === '/v1/admin/drain') { + return failDrain ? response({ error: 'injected_drain_failure' }, 503) : response({ ok: true }) + } + return response({ error: 'unexpected_request' }, 500) + } + return { + overrides: { + commandJson: fakeCommand, + identityToken: () => 'secret-token-never-emitted', + fetch, + emit: (event) => events.push(event), + wait: async () => undefined, + random: () => 0 + }, + events, + stateChanges, + state + } +} + +test('requires explicit dry-run/execute inputs and a distinct disabled candidate', () => { + assert.throws(() => parseArguments([]), /missing --project/) + assert.throws( + () => + parseArguments([ + '--project', + 'project', + '--director-origin', + 'https://relay.example.com', + '--admin-audience', + 'https://relay.example.com/not-drain', + '--topology-file', + 'topology.json', + '--source-cell-id', + 'source', + '--target-cell-id', + 'target', + '--runtime-service-account', + runtimeServiceAccount, + '--mode', + 'preflight' + ]), + /director drain URL/ + ) + assert.throws(() => selectDeployments(topology(), 'source', 'source'), /must differ/) + const overlapping = topology() + overlapping.target.backend_id = overlapping.source.backend_id + assert.throws(() => selectDeployments(overlapping, 'source', 'target'), /backendId overlap/) + const enabled = topology() + enabled.target.initially_enabled = true + assert.throws(() => selectDeployments(enabled, 'source', 'target'), /initially disabled/) +}) + +test('accepts only a bounded JWT-shaped supplied admin identity token', () => { + assert.equal(suppliedAdminIdentityToken({}), null) + assert.equal(suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }), 'aaa.bbb.ccc') + assert.throws(() => suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: '' })) + assert.throws(() => suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: 'not-a-jwt' })) + assert.throws(() => + suppliedAdminIdentityToken({ ORCA_RELAY_ADMIN_ID_TOKEN: `aaa.${'b'.repeat(8_190)}.ccc` }) + ) +}) + +test('rejects unsafe fixed-one topology, public IPs, and backend overlap', () => { + const expected = selectDeployments(topology(), 'source', 'target').target + assert.throws(() => validateMig({ targetSize: 2 }, [], expected), /fixed-one/) + assert.throws( + () => + validateInstance( + { + networkInterfaces: [{ accessConfigs: [{ natIP: '203.0.113.1' }] }], + serviceAccounts: [{ email: runtimeServiceAccount }] + }, + expected, + runtimeServiceAccount + ), + /public IP/ + ) + assert.throws( + () => validateBackend({ protocol: 'HTTP', timeoutSec: 86_400, backends: [] }, expected), + /topology mismatch/ + ) +}) + +test('preflights exact served digests and survivor headroom without mutating admission', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness() + await runCandidateDeployment(config(file), overrides) + assert.deepEqual(stateChanges, []) + assert.equal(events[0].event, 'candidate_preflight') + assert.equal(events[0].targetDigest, digestB) + assert.equal(JSON.stringify(events).includes('secret-token'), false) + }) +}) + +test('audits a partially committed migration without changing admission or completing rows', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + sourceEnabled: false, + targetEnabled: true, + targetAssignments: 1, + migrationInProgress: 2, + migrationTargetRegistered: 2 + }) + await runCandidateDeployment(config(file, 'audit'), overrides) + assert.deepEqual(stateChanges, []) + assert.deepEqual(events, [ + { + event: 'candidate_audit', + source: { + cellId: 'source', + enabled: false, + assignments: 2, + activityLeases: 2, + activityRequestUnits: 2, + reservedRequests: 2, + outgoingMigrations: 0, + incomingMigrations: 0, + runtimeReady: true, + heartbeatFresh: true, + observedRequests: 2 + }, + target: { + cellId: 'target', + enabled: true, + assignments: 1, + activityLeases: 1, + activityRequestUnits: 1, + reservedRequests: 1, + outgoingMigrations: 0, + incomingMigrations: 0, + runtimeReady: true, + heartbeatFresh: true, + observedRequests: 1 + }, + migration: { + v: 1, + inProgress: 2, + targetRegistered: 2, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + } + } + ]) + }) +}) + +test('preflights with source admission disabled but still refuses execution', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ sourceEnabled: false }) + await runCandidateDeployment(config(file), overrides) + assert.deepEqual(stateChanges, []) + assert.equal(events.at(-1).event, 'candidate_preflight') + await assert.rejects( + runCandidateDeployment(config(file, 'execute'), overrides), + /source cell is not enabled/ + ) + assert.deepEqual(stateChanges, []) + }) +}) + +test('explicitly resets only an empty declared candidate to disabled admission', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + sourceEnabled: false, + targetEnabled: true + }) + await runCandidateDeployment(config(file, 'reset-empty-candidate'), overrides) + assert.deepEqual(stateChanges, [['target', false]]) + assert.deepEqual(events.at(-1), { + event: 'candidate_admission_reset', + targetCellId: 'target', + changed: true + }) + }) +}) + +test('refuses to reset candidate admission while it owns durable activity', async () => { + await withTopology(async (file) => { + const { overrides, stateChanges } = harness({ targetEnabled: true, targetAssignments: 1 }) + await assert.rejects( + runCandidateDeployment(config(file, 'reset-empty-candidate'), overrides), + /requires zero durable activity/ + ) + assert.deepEqual(stateChanges, []) + }) +}) + +test('disables only new admission while preserving durable candidate activity', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + targetEnabled: true, + targetAssignments: 1 + }) + await runCandidateDeployment(config(file, 'disable-cell'), overrides) + assert.deepEqual(stateChanges, [['target', false]]) + assert.deepEqual(events.at(-1), { + event: 'cell_admission_disabled', + targetCellId: 'target', + changed: true, + assignments: 1, + activityLeases: 1, + reservedRequests: 1, + outgoingMigrations: 0, + incomingMigrations: 0 + }) + }) +}) + +test('explicitly enables only an empty preflighted cell for admission', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ sourceEnabled: false }) + await runCandidateDeployment(config(file, 'enable-empty-cell'), overrides) + assert.deepEqual(stateChanges, [['target', true]]) + assert.equal(events.at(-2).event, 'candidate_preflight') + assert.deepEqual(events.at(-1), { + event: 'cell_admission_enabled', + targetCellId: 'target', + changed: true + }) + }) +}) + +test('refuses to enable cell admission while it owns durable activity', async () => { + await withTopology(async (file) => { + const { overrides, stateChanges } = harness({ targetAssignments: 1 }) + await assert.rejects( + runCandidateDeployment(config(file, 'enable-empty-cell'), overrides), + /requires zero durable activity/ + ) + assert.deepEqual(stateChanges, []) + }) +}) + +test('executes target-first evacuation and verifies aggregate drained counts', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness() + await runCandidateDeployment(config(file, 'execute'), overrides) + assert.deepEqual(stateChanges.slice(0, 2), [ + ['source', false], + ['target', true] + ]) + assert.equal(events.at(-1).event, 'candidate_complete') + assert.equal(events.at(-1).targetAssignments, 2) + }) +}) + +test('executes within selector membership without legacy admission writes', async () => { + await withTopology(async (file) => { + const testHarness = harness({ selectorGeneration: 1 }) + await runCandidateDeployment(config(file, 'execute'), testHarness.overrides) + assert.deepEqual(testHarness.stateChanges, []) + assert.equal(testHarness.state.source.enabled, false) + assert.equal(testHarness.state.target.enabled, true) + }) +}) + +test('never restores legacy general admission after selector-era failure', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorGeneration: 1, + failAfterRegistration: true + }) + await assert.rejects( + runCandidateDeployment(config(file, 'execute'), testHarness.overrides), + /injected_batch_failure/ + ) + assert.deepEqual(testHarness.stateChanges, []) + assert.equal(testHarness.state.source.enabled, false) + }) +}) + +test('continues a partial evacuation without resetting target admission', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + sourceEnabled: true, + targetEnabled: true, + targetAssignments: 1, + migrationInProgress: 1, + migrationTargetRegistered: 1 + }) + await runCandidateDeployment(config(file, 'continue-evacuation'), overrides) + assert.deepEqual(stateChanges, [['source', false]]) + assert.equal(events.at(-1).event, 'candidate_complete') + }) +}) + +test('refuses continued evacuation unless target admission is already enabled', async () => { + await withTopology(async (file) => { + const { overrides, stateChanges } = harness() + await assert.rejects( + runCandidateDeployment(config(file, 'continue-evacuation'), overrides), + /requires enabled target/ + ) + assert.deepEqual(stateChanges, []) + }) +}) + +test('preserves partial target admission when continued batching fails', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + failAfterRegistration: true, + sourceEnabled: true, + targetEnabled: true, + targetAssignments: 1, + migrationInProgress: 1, + migrationTargetRegistered: 1 + }) + await assert.rejects( + runCandidateDeployment(config(file, 'continue-evacuation'), overrides), + /injected_batch_failure/ + ) + assert.deepEqual(stateChanges, [['source', false]]) + assert.equal(events.at(-1).event, 'candidate_forward_recovery_required') + }) +}) + +test('resumes only a committed forward migration and preserves dormant source assignments', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + sourceEnabled: false, + targetEnabled: true, + migrationInProgress: 2, + migrationTargetRegistered: 2, + dormantSourceAssignments: 7 + }) + await runCandidateDeployment(config(file, 'recover-forward'), overrides) + assert.deepEqual(stateChanges, []) + assert.deepEqual(events.at(-1), { + event: 'candidate_forward_recovered', + sourceCellId: 'source', + targetCellId: 'target', + dormantSourceAssignments: 7, + targetAssignments: 2, + targetActivityLeases: 2, + targetReservedRequests: 2 + }) + }) +}) + +test('retries a transient idempotent completion request without reversing admission', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + sourceEnabled: false, + targetEnabled: true, + migrationInProgress: 2, + migrationTargetRegistered: 2, + transientCompletionFailures: 1 + }) + await runCandidateDeployment(config(file, 'recover-forward'), overrides) + assert.deepEqual(stateChanges, []) + assert.deepEqual( + events.filter(({ event }) => event === 'candidate_admin_retry'), + [ + { + event: 'candidate_admin_retry', + path: '/v1/admin/evacuation-status', + attempt: 1, + reason: 'transport' + } + ] + ) + assert.equal(events.at(-1).event, 'candidate_forward_recovered') + }) +}) + +test('stops forward recovery promptly when only registered offline targets remain', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ + sourceEnabled: false, + targetEnabled: true, + migrationInProgress: 2, + migrationTargetRegistered: 2, + migrationTargetInactive: 2 + }) + await assert.rejects( + runCandidateDeployment(config(file, 'recover-forward'), overrides), + /pending for inactive target controls/ + ) + assert.deepEqual(stateChanges, []) + assert.deepEqual(events.at(-1), { + event: 'candidate_forward_pending', + sourceCellId: 'source', + targetCellId: 'target', + inProgress: 2, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 2 + }) + }) +}) + +test('refuses forward recovery unless source and target admission match committed direction', async () => { + await withTopology(async (file) => { + const { overrides, stateChanges } = harness() + await assert.rejects( + runCandidateDeployment(config(file, 'recover-forward'), overrides), + /requires disabled source and enabled target/ + ) + assert.deepEqual(stateChanges, []) + }) +}) + +test('re-enables an intact source when candidate admission fails before migration', async () => { + await withTopology(async (file) => { + const { overrides, stateChanges } = harness({ failTargetEnable: true }) + await assert.rejects( + runCandidateDeployment(config(file, 'execute'), overrides), + /injected_enable_failure/ + ) + assert.deepEqual(stateChanges, [ + ['source', false], + ['target', true], + ['source', true], + ['target', false] + ]) + }) +}) + +test('re-enables the source and waits for lease rollback when no target registered', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ failDrain: true }) + await assert.rejects( + runCandidateDeployment(config(file, 'execute'), overrides), + /injected_drain_failure/ + ) + assert.deepEqual(stateChanges.slice(-1), [['source', true]]) + assert.equal(events.at(-1).event, 'candidate_rollback_waiting_for_lease_expiry') + }) +}) + +test('preserves both routes for forward recovery after a target registration', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ failAfterRegistration: true }) + await assert.rejects( + runCandidateDeployment(config(file, 'execute'), overrides), + /injected_batch_failure/ + ) + assert.deepEqual(stateChanges, [ + ['source', false], + ['target', true] + ]) + assert.equal(events.at(-1).event, 'candidate_forward_recovery_required') + assert.equal(events.at(-1).targetRegistered, 1) + }) +}) + +test('does not reverse admission after completion commits but its response is lost', async () => { + await withTopology(async (file) => { + const { overrides, events, stateChanges } = harness({ failCompletionResponse: true }) + await assert.rejects( + runCandidateDeployment(config(file, 'execute'), overrides), + /injected_completion_response_failure/ + ) + assert.deepEqual(stateChanges, [ + ['source', false], + ['target', true] + ]) + assert.equal(events.at(-1).event, 'candidate_forward_recovery_required') + assert.equal(events.at(-1).targetRegistered, 0) + }) +}) diff --git a/cloud/dev/scripts/deploy-relay-gce-multi-target.mjs b/cloud/dev/scripts/deploy-relay-gce-multi-target.mjs new file mode 100644 index 00000000000..e36570947ad --- /dev/null +++ b/cloud/dev/scripts/deploy-relay-gce-multi-target.mjs @@ -0,0 +1,3135 @@ +import { readFileSync } from 'node:fs' +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { resolve4 } from 'node:dns/promises' +import { pathToFileURL } from 'node:url' +import { + aggregateCellStatus, + assertDeploymentConnectionCapacity, + createAdminPost, + defaultCommandJson, + defaultIdentityToken, + deployment, + drainSource, + inspectCell, + setCellState, + suppliedFenceMutationIdentityToken, + validateBackend, + validateInstance, + validateMig +} from './deploy-relay-gce-candidate.mjs' +import { + abortSupersededTerraformFenceBeforeUpload, + abortTerraformFenceBeforeApply, + adoptLegacyTerraformFence, + assertTerraformFenceSet, + assertTerraformFenceZeroDiff, + deleteTerraformFencePlan, + downloadTerraformFencePlan, + inspectCompletedTerraformFenceProgress, + inspectTerraformFenceProgress, + assertTerraformFenceStateFenced, + readTerraformStateObjectBinding, + recoverSupersededCompletedTerraformFence, + resolveTerraformFencePlanGeneration, + resumeTerraformFence, + runTerraformFenceApply, + uploadTerraformFencePlan +} from './relay-gce-terraform-fence.mjs' +import { + addExactMigrationCells, + applyExactAdmissionSelector, + inspectAdmissionSelector, + membershipWithStates, + selectorCellState +} from './relay-admission-selector.mjs' + +const DEFAULT_BATCH_SIZE = 100 +const DEFAULT_CONNECTION_CEILING = 600 +const DEFAULT_MINIMUM_LEASE_MS = 10 * 60 * 1_000 +const DEFAULT_POLL_MS = 5_000 +const DEFAULT_TIMEOUT_MS = 14 * 60 * 1_000 +const CUTOVER_CONNECTION_HARD_CAP = 600 +const CUTOVER_CONTROL_REBIND_RESERVE = 100 +const MAX_PRE_AUTH_CONNECTIONS = 45 +const SELECTOR_ROLLBACK_TAG = 'selector-rollback' +const SELECTOR_REVISION_MARKER = '3' +const FENCE_BROKER_MUTATION_ROUTES = new Set([ + '/v1/admin/cell-fence-adopt-legacy', + '/v1/admin/cell-fence-commit-legacy-adoption', + '/v1/admin/cell-fence-attest', + '/v1/admin/cell-fence-attempt-prepare', + '/v1/admin/cell-fence-attempt-start', + '/v1/admin/cell-fence-attempt-plan', + '/v1/admin/cell-fence-attempt-operation', + '/v1/admin/cell-fence-attempt-abort', + '/v1/admin/migration-supersede-cell' +]) + +function positiveInteger(value, name, maximum = Number.MAX_SAFE_INTEGER) { + const parsed = Number(value) + if (!Number.isInteger(parsed) || parsed <= 0 || parsed > maximum) { + throw new Error(`${name} must be a positive integer`) + } + return parsed +} + +function nonnegativeInteger(value, name, maximum = Number.MAX_SAFE_INTEGER) { + const parsed = Number(value) + if (!Number.isSafeInteger(parsed) || parsed < 0 || parsed > maximum) { + throw new Error(`${name} must be a nonnegative integer`) + } + return parsed +} + +function canonicalOrigin(value, name) { + const url = new URL(value) + if (url.protocol !== 'https:' || url.origin !== value || url.pathname !== '/') { + throw new Error(`${name} must be a canonical HTTPS origin`) + } + return value +} + +function targetIds(value, minimum) { + const ids = [...new Set(value.split(',').map((item) => item.trim()).filter(Boolean))].sort() + if (ids.length < minimum || ids.some((id) => !/^[a-z][a-z0-9-]{0,127}$/.test(id))) { + throw new Error( + `--target-cell-ids must contain at least ${minimum} distinct cell ID${minimum === 1 ? '' : 's'}` + ) + } + return ids +} + +function optionalCellIds(value, name) { + const ids = [...new Set(String(value ?? '').split(',').map((item) => item.trim()).filter(Boolean))] + .sort() + if (ids.some((id) => !/^[a-z][a-z0-9-]{0,127}$/.test(id))) { + throw new Error(`${name} contains an invalid cell ID`) + } + return ids +} + +export function parseMultiTargetArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error(`invalid argument ${key ?? ''}`) + values[key.slice(2)] = value + } + for (const key of [ + 'project', + 'director-origin', + 'admin-audience', + 'topology-file', + 'source-cell-id', + 'target-cell-ids', + 'runtime-service-account', + 'mode' + ]) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if ( + ![ + 'audit', + 'preflight', + 'execute', + 'cutover-admission', + 'add-migration-cells', + 'promote-general-cell', + 'retire-migration-cell', + 'recover-forward', + 'fence-source', + 'abort-fence-source', + 'supersede-target' + ].includes(values.mode) + ) { + throw new Error( + '--mode must be audit, preflight, execute, cutover-admission, add-migration-cells, promote-general-cell, retire-migration-cell, recover-forward, fence-source, abort-fence-source, or supersede-target' + ) + } + const singleTargetMode = [ + 'add-migration-cells', + 'promote-general-cell', + 'retire-migration-cell' + ].includes(values.mode) + const parsedTargetIds = targetIds(values['target-cell-ids'], singleTargetMode ? 1 : 2) + const generalCellIds = optionalCellIds(values['general-cell-ids'], '--general-cell-ids') + const failedTargetCellId = values['failed-target-cell-id'] + const replacementTargetCellId = values['replacement-target-cell-id'] + if ( + ['promote-general-cell', 'retire-migration-cell'].includes(values.mode) && + parsedTargetIds.length !== 1 + ) { + throw new Error(`${values.mode} requires exactly one target cell ID`) + } + if (values.mode === 'supersede-target') { + if (!failedTargetCellId || !replacementTargetCellId) { + throw new Error('supersede-target requires failed and replacement target cell IDs') + } + if ( + failedTargetCellId === replacementTargetCellId || + !parsedTargetIds.includes(failedTargetCellId) || + !parsedTargetIds.includes(replacementTargetCellId) || + parsedTargetIds.length !== 2 + ) { + throw new Error('supersede-target target set must exactly match failed and replacement') + } + } + const selectorMode = [ + 'cutover-admission', + 'add-migration-cells', + 'promote-general-cell', + 'retire-migration-cell' + ].includes(values.mode) + if (selectorMode) { + for (const key of ['director-region', 'director-service', 'director-min-instances']) { + if (!values[key]) throw new Error(`${values.mode} requires --${key}`) + } + if (values.mode === 'cutover-admission' && generalCellIds.length === 0) { + throw new Error('cutover-admission requires --general-cell-ids') + } + if ( + values['selector-attempt-id'] && + !/^[A-Za-z0-9_-]{8,128}$/.test(values['selector-attempt-id']) + ) { + throw new Error('--selector-attempt-id is invalid') + } + if ( + ['add-migration-cells', 'promote-general-cell', 'retire-migration-cell'].includes( + values.mode + ) && + !values['selector-attempt-id'] + ) { + throw new Error(`${values.mode} requires --selector-attempt-id`) + } + } + const capacityBoundModes = [ + 'cutover-admission', + 'add-migration-cells', + 'recover-forward', + 'fence-source' + ] + if ( + capacityBoundModes.includes(values.mode) && + !values['unobserved-connection-bound'] + ) { + throw new Error(`${values.mode} requires --unobserved-connection-bound`) + } + const adminAudience = new URL(values['admin-audience']) + if ( + adminAudience.protocol !== 'https:' || + adminAudience.pathname !== '/v1/admin/drain' || + adminAudience.search || + adminAudience.hash + ) { + throw new Error('--admin-audience must be the director drain URL') + } + if (!['staging', 'production'].includes(values.environment ?? 'production')) { + throw new Error('--environment must be staging or production') + } + if ( + ['fence-source', 'abort-fence-source', 'supersede-target'].includes(values.mode) && + !/^[a-f0-9]{40}$/.test(values['fence-commit'] ?? '') + ) { + throw new Error('Terraform fence modes require the exact --fence-commit') + } + const completedFenceFields = { + attemptId: values['completed-fence-attempt-id'], + fenceCommit: values['completed-fence-commit'], + gceOperation: values['completed-fence-operation'], + terraformStateSerial: values['completed-fence-state-serial'], + planObjectGeneration: values['completed-fence-plan-generation'], + terraformStateObjectGeneration: values['completed-fence-state-generation'], + terraformStateObjectSha256: values['completed-fence-state-sha256'], + principalEmail: values['fence-broker-service-account'] + } + const completedFenceValues = Object.values(completedFenceFields) + const completedFenceRecovery = + completedFenceValues.every((value) => value === undefined) + ? undefined + : completedFenceFields + if ( + completedFenceRecovery && + (values.mode !== 'supersede-target' || + completedFenceValues.some((value) => value === undefined) || + !/^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/i.test( + completedFenceRecovery.attemptId + ) || + !/^[a-f0-9]{40}$/.test(completedFenceRecovery.fenceCommit) || + completedFenceRecovery.fenceCommit === values['fence-commit'] || + !/^[A-Za-z0-9._-]{1,256}$/.test(completedFenceRecovery.gceOperation) || + !/^[1-9][0-9]{0,30}$/.test(completedFenceRecovery.planObjectGeneration) || + !/^[1-9][0-9]{0,30}$/.test( + completedFenceRecovery.terraformStateObjectGeneration + ) || + !/^[a-f0-9]{64}$/.test( + completedFenceRecovery.terraformStateObjectSha256 + ) || + !/^[^@\s]+@[^@\s]+\.gserviceaccount\.com$/.test( + completedFenceRecovery.principalEmail + )) + ) { + throw new Error('completed Terraform fence recovery inputs are invalid') + } + const minimumLeaseRemainingMs = positiveInteger( + values['minimum-lease-remaining-ms'] ?? DEFAULT_MINIMUM_LEASE_MS, + '--minimum-lease-remaining-ms', + 60 * 60 * 1_000 + ) + if (minimumLeaseRemainingMs < DEFAULT_MINIMUM_LEASE_MS) { + throw new Error('--minimum-lease-remaining-ms cannot be below 600000') + } + return { + project: values.project, + directorOrigin: canonicalOrigin(values['director-origin'], '--director-origin'), + adminAudience: adminAudience.toString(), + topologyFile: values['topology-file'], + sourceCellId: values['source-cell-id'], + targetCellIds: parsedTargetIds, + generalCellIds, + directorRegion: values['director-region'], + directorService: values['director-service'], + directorMinimumInstances: selectorMode + ? positiveInteger(values['director-min-instances'], '--director-min-instances', 1_000) + : undefined, + selectorAttemptId: values['selector-attempt-id'], + unobservedConnectionBound: + capacityBoundModes.includes(values.mode) + ? nonnegativeInteger( + values['unobserved-connection-bound'], + '--unobserved-connection-bound', + CUTOVER_CONNECTION_HARD_CAP - CUTOVER_CONTROL_REBIND_RESERVE - 1 + ) + : undefined, + failedTargetCellId, + replacementTargetCellId, + completedFenceRecovery: completedFenceRecovery + ? { + ...completedFenceRecovery, + terraformStateSerial: nonnegativeInteger( + completedFenceRecovery.terraformStateSerial, + '--completed-fence-state-serial' + ) + } + : undefined, + runtimeServiceAccount: values['runtime-service-account'], + environment: values.environment ?? 'production', + fenceCommit: values['fence-commit'], + terraformDir: values['terraform-dir'] ?? 'infra/terraform', + terraformVarFile: + values['terraform-var-file'] ?? + `environments/${values.environment ?? 'production'}.tfvars`, + mode: values.mode, + batchSize: positiveInteger(values['batch-size'] ?? DEFAULT_BATCH_SIZE, '--batch-size', 100), + connectionCeiling: positiveInteger( + values['connection-ceiling'] ?? DEFAULT_CONNECTION_CEILING, + '--connection-ceiling', + 100_000 + ), + minimumLeaseRemainingMs, + drainGraceMs: positiveInteger( + values['drain-grace-ms'] ?? 120_000, + '--drain-grace-ms', + 60 * 60 * 1_000 + ), + pollIntervalMs: positiveInteger( + values['poll-interval-ms'] ?? DEFAULT_POLL_MS, + '--poll-interval-ms', + 60_000 + ), + timeoutMs: positiveInteger( + values['timeout-ms'] ?? DEFAULT_TIMEOUT_MS, + '--timeout-ms', + 60 * 60 * 1_000 + ) + } +} + +export function selectMultiTargetDeployments(topology, sourceCellId, targetCellIds) { + const legacyCapacityTopology = Object.values(topology).every( + (cell) => + cell && + typeof cell === 'object' && + !Object.hasOwn(cell, 'connection_hard_cap') && + !Object.hasOwn(cell, 'connection_unobserved_bound') + ) + const fromTopology = (cellId) => ({ + ...deployment(topology[cellId], cellId), + legacyCapacityTopology + }) + const source = fromTopology(sourceCellId) + const targets = targetCellIds.map(fromTopology) + const resources = new Map() + for (const cell of [source, ...targets]) { + for (const key of ['origin', 'migName', 'instanceGroup', 'backendName', 'backendId']) { + const resourceKey = `${key}:${cell[key]}` + const previous = resources.get(resourceKey) + if (previous) throw new Error(`${cell.cellId} ${key} overlaps ${previous}`) + resources.set(resourceKey, cell.cellId) + } + } + if (targets.some((target) => target.initiallyEnabled)) { + throw new Error('every target must be declared initially disabled') + } + return { source, targets } +} + +function revisionEnvironment(revision) { + return Object.fromEntries( + (revision.spec?.containers?.[0]?.env ?? []) + .filter((entry) => entry.name && 'value' in entry) + .map((entry) => [entry.name, entry.value]) + ) +} + +function revisionMinimum(revision) { + return Number(revision.metadata?.annotations?.['autoscaling.knative.dev/minScale'] ?? 0) +} + +function directorInventory(environment, revisionName) { + let cells + try { + cells = JSON.parse(environment.ORCA_RELAY_CELLS_JSON) + } catch { + throw new Error(`${revisionName} has an invalid director inventory`) + } + const ids = Array.isArray(cells) ? cells.map((cell) => cell?.id) : [] + if ( + ids.length === 0 || + ids.some((id) => typeof id !== 'string' || id.length === 0) || + new Set(ids).size !== ids.length + ) { + throw new Error(`${revisionName} has an invalid director inventory`) + } + return JSON.stringify(cells) +} + +export function verifySelectorCompatibleDirector(config, deps) { + const common = [ + '--project', + config.project, + '--region', + config.directorRegion, + '--format=json' + ] + const service = deps.commandJson([ + 'run', + 'services', + 'describe', + config.directorService, + ...common + ]) + const active = (service.status?.traffic ?? []).filter( + (entry) => Number(entry.percent ?? 0) > 0 + ) + const rollback = (service.status?.traffic ?? []).find( + (entry) => entry.tag === SELECTOR_ROLLBACK_TAG + ) + if ( + active.length !== 1 || + Number(active[0].percent) !== 100 || + !active[0].revisionName || + !rollback?.revisionName || + Number(rollback.percent ?? 0) !== 0 || + rollback.revisionName === active[0].revisionName + ) { + throw new Error('director lacks an isolated compatible rollback revision') + } + const revisions = deps.commandJson([ + 'run', + 'revisions', + 'list', + '--service', + config.directorService, + ...common + ]) + const allowed = new Set([active[0].revisionName, rollback.revisionName]) + const names = revisions.map((revision) => revision.metadata?.name).filter(Boolean) + if ( + revisions.length !== 2 || + names.length !== 2 || + names.some((name) => !allowed.has(name)) + ) { + throw new Error('old or pre-selector director revisions still exist') + } + let compatibleImage + let compatibleInventory + for (const revisionName of allowed) { + const revision = deps.commandJson([ + 'run', + 'revisions', + 'describe', + revisionName, + ...common + ]) + const environment = revisionEnvironment(revision) + if ( + environment.ORCA_RELAY_ROLE !== 'director' || + environment.ORCA_RELAY_ADMISSION_SELECTOR_VERSION !== SELECTOR_REVISION_MARKER + ) { + throw new Error(`${revisionName} is not selector-compatible`) + } + const inventory = directorInventory(environment, revisionName) + if (compatibleInventory && inventory !== compatibleInventory) { + throw new Error('active and rollback director inventories do not match') + } + compatibleInventory = inventory + const image = revision.spec?.containers?.[0]?.image + if (!image || (compatibleImage && image !== compatibleImage)) { + throw new Error('active and rollback director images do not match') + } + compatibleImage = image + if (revisionName === rollback.revisionName && revisionMinimum(revision) !== 0) { + throw new Error('selector rollback revision is not scale-to-zero') + } + if ( + revisionName === active[0].revisionName && + !(revisionMinimum(revision) >= config.directorMinimumInstances) + ) { + throw new Error('active selector revision is below the required floor') + } + } + return { + activeRevision: active[0].revisionName, + rollbackRevision: rollback.revisionName + } +} + +function verifyActiveSelectorDirector(config, deps, cellId) { + const common = [ + '--project', + config.project, + '--region', + config.directorRegion, + '--format=json' + ] + const service = deps.commandJson([ + 'run', + 'services', + 'describe', + config.directorService, + ...common + ]) + const active = (service.status?.traffic ?? []).filter( + (entry) => Number(entry.percent ?? 0) > 0 + ) + if ( + active.length !== 1 || + Number(active[0].percent) !== 100 || + !active[0].revisionName + ) { + throw new Error('director lacks one active selector revision') + } + const revision = deps.commandJson([ + 'run', + 'revisions', + 'describe', + active[0].revisionName, + ...common + ]) + const environment = revisionEnvironment(revision) + const inventory = JSON.parse(directorInventory(environment, active[0].revisionName)) + if ( + environment.ORCA_RELAY_ROLE !== 'director' || + environment.ORCA_RELAY_ADMISSION_SELECTOR_VERSION !== SELECTOR_REVISION_MARKER || + !(revisionMinimum(revision) >= config.directorMinimumInstances) || + !inventory.some((cell) => cell.id === cellId) + ) { + throw new Error('active director is not compatible with the promoted cell') + } + return { activeRevision: active[0].revisionName } +} + +export function pruneIncompatibleDirectorRevisions(config, deps) { + const common = [ + '--project', + config.project, + '--region', + config.directorRegion, + '--format=json' + ] + const service = deps.commandJson([ + 'run', + 'services', + 'describe', + config.directorService, + ...common + ]) + const traffic = service.status?.traffic ?? [] + const active = traffic.filter((entry) => Number(entry.percent ?? 0) > 0) + const rollback = traffic.find((entry) => entry.tag === SELECTOR_ROLLBACK_TAG) + const unexpectedTags = traffic.filter( + (entry) => entry.tag && entry.tag !== SELECTOR_ROLLBACK_TAG + ) + if ( + active.length !== 1 || + Number(active[0].percent) !== 100 || + !active[0].revisionName || + !rollback?.revisionName || + Number(rollback.percent ?? 0) !== 0 || + rollback.revisionName === active[0].revisionName || + unexpectedTags.length > 0 + ) { + throw new Error('director traffic is not ready for selector cutover') + } + const activeRevision = deps.commandJson([ + 'run', + 'revisions', + 'describe', + active[0].revisionName, + ...common + ]) + const rollbackRevision = deps.commandJson([ + 'run', + 'revisions', + 'describe', + rollback.revisionName, + ...common + ]) + const activeEnvironment = revisionEnvironment(activeRevision) + const rollbackEnvironment = revisionEnvironment(rollbackRevision) + const activeInventory = directorInventory(activeEnvironment, active[0].revisionName) + const rollbackInventory = directorInventory(rollbackEnvironment, rollback.revisionName) + if ( + activeEnvironment.ORCA_RELAY_ROLE !== 'director' || + rollbackEnvironment.ORCA_RELAY_ROLE !== 'director' || + activeEnvironment.ORCA_RELAY_ADMISSION_SELECTOR_VERSION !== + SELECTOR_REVISION_MARKER || + rollbackEnvironment.ORCA_RELAY_ADMISSION_SELECTOR_VERSION !== + SELECTOR_REVISION_MARKER || + !activeRevision.spec?.containers?.[0]?.image || + activeRevision.spec?.containers?.[0]?.image !== + rollbackRevision.spec?.containers?.[0]?.image || + activeInventory !== rollbackInventory || + revisionMinimum(rollbackRevision) !== 0 || + !(revisionMinimum(activeRevision) >= config.directorMinimumInstances) + ) { + throw new Error('director compatibility pair failed before revision pruning') + } + const retained = new Set([active[0].revisionName, rollback.revisionName]) + const revisions = deps.commandJson([ + 'run', + 'revisions', + 'list', + '--service', + config.directorService, + ...common + ]) + for (const revision of revisions) { + const revisionName = revision.metadata?.name + if (!revisionName) throw new Error('director revision list contains an unnamed revision') + if (retained.has(revisionName)) continue + deps.command([ + 'run', + 'revisions', + 'delete', + revisionName, + '--project', + config.project, + '--region', + config.directorRegion, + '--quiet' + ]) + } +} + +export function cutoverMembership(topology, config) { + const all = Object.keys(topology).sort() + const migration = new Set(config.targetCellIds) + const general = new Set(config.generalCellIds) + if ([...migration].some((cellId) => general.has(cellId))) { + throw new Error('general and migration-only cell sets overlap') + } + if (migration.has(config.sourceCellId) || general.has(config.sourceCellId)) { + throw new Error('cutover source must remain existing-only') + } + for (const cellId of [...migration, ...general]) { + if (!all.includes(cellId)) throw new Error(`selector cell ${cellId} is absent from topology`) + } + return { + existingOnly: all.filter((cellId) => !migration.has(cellId) && !general.has(cellId)), + migrationOnly: [...migration].sort(), + general: [...general].sort() + } +} + +function assertDistinctCutoverResources(cells) { + const resources = new Map() + for (const cell of cells) { + for (const key of ['origin', 'migName', 'instanceGroup', 'backendName', 'backendId']) { + const resourceKey = `${key}:${cell[key]}` + const previous = resources.get(resourceKey) + if (previous) throw new Error(`${cell.cellId} ${key} overlaps ${previous}`) + resources.set(resourceKey, cell.cellId) + } + } +} + +export function assertCutoverCellReady(cellId, status, unobservedConnectionBound) { + const capacity = status.runtimeConnectionCapacity + const directorCapacity = status.connectionCapacity + const process = status.process + const expectedPause = + CUTOVER_CONNECTION_HARD_CAP - + CUTOVER_CONTROL_REBIND_RESERVE - + unobservedConnectionBound + if ( + !capacity || + capacity.hardCap !== CUTOVER_CONNECTION_HARD_CAP || + capacity.controlRebindReserve !== CUTOVER_CONTROL_REBIND_RESERVE || + capacity.ordinaryConnectionLimit !== + CUTOVER_CONNECTION_HARD_CAP - CUTOVER_CONTROL_REBIND_RESERVE || + capacity.unobservedBound !== unobservedConnectionBound || + capacity.normalAdmissionPause !== expectedPause || + !directorCapacity || + directorCapacity.hardCap !== capacity.hardCap || + directorCapacity.controlRebindReserve !== capacity.controlRebindReserve || + directorCapacity.ordinaryConnectionLimit !== capacity.ordinaryConnectionLimit || + directorCapacity.unobservedBound !== capacity.unobservedBound || + directorCapacity.normalAdmissionPause !== capacity.normalAdmissionPause || + directorCapacity.heartbeatFresh !== true || + expectedPause <= 0 + ) { + throw new Error(`${cellId} does not expose the reviewed connection-capacity policy`) + } + if ( + !process || + !Number.isSafeInteger(process.enforcedConnectionUnits) || + process.enforcedConnectionUnits < 0 || + !Number.isSafeInteger(process.preAuthConnections) || + process.preAuthConnections < 0 || + !Number.isSafeInteger(directorCapacity.pendingControlReservations) || + directorCapacity.pendingControlReservations < 0 + ) { + throw new Error(`${cellId} has incomplete connection-capacity evidence`) + } + if (process.preAuthConnections >= MAX_PRE_AUTH_CONNECTIONS) { + throw new Error(`${cellId} has insufficient pre-auth connection headroom`) + } + const committedUnits = + process.enforcedConnectionUnits + directorCapacity.pendingControlReservations + if (!Number.isSafeInteger(committedUnits) || committedUnits >= expectedPause) { + throw new Error(`${cellId} has insufficient normal-admission connection headroom`) + } + if (status.draining) throw new Error(`${cellId} is draining before selector cutover`) +} + +async function inspectCutoverCells(topology, config, deps, adminPost, membership) { + const selectedIds = [...membership.migrationOnly, ...membership.general] + const selected = selectedIds.map((cellId) => deployment(topology[cellId], cellId)) + assertDistinctCutoverResources([ + deployment(topology[config.sourceCellId], config.sourceCellId), + ...selected + ]) + for (const cell of selected) { + const status = await inspectCell(config, deps, adminPost, cell) + assertCutoverCellReady(cell.cellId, status, config.unobservedConnectionBound) + } +} + +function projection(total, quota, assignments) { + return assignments === 0 ? 0 : Math.ceil((total * quota) / assignments) +} + +function connectionProjection(current, sourceConnections, sourceAssignments, quota) { + const unboundConnections = Math.max(0, sourceConnections - sourceAssignments) + return current + quota + unboundConnections +} + +function targetConnectionCeiling(config, target) { + return Math.min( + config.connectionCeiling, + target.connectionHardCap ?? CUTOVER_CONNECTION_HARD_CAP + ) +} + +function connectionReservationHeadroom(status, cellId) { + const capacity = status.connectionCapacity + const values = [ + capacity?.hardCap, + capacity?.controlRebindReserve, + capacity?.ordinaryConnectionLimit, + capacity?.unobservedBound, + capacity?.normalAdmissionPause, + capacity?.enforcedConnectionUnits, + capacity?.pendingControlReservations + ] + if ( + capacity?.heartbeatFresh !== true || + values.some((value) => !Number.isSafeInteger(value) || value < 0) || + capacity.ordinaryConnectionLimit !== capacity.hardCap - capacity.controlRebindReserve || + capacity.normalAdmissionPause !== + capacity.ordinaryConnectionLimit - capacity.unobservedBound + ) { + throw new Error(`${cellId} has inconsistent connection-reservation capacity`) + } + return Math.max( + 0, + capacity.normalAdmissionPause - + capacity.enforcedConnectionUnits - + capacity.pendingControlReservations + ) +} + +export function allocateTargetQuotas({ + sourceAssignments, + sourceConnections, + requiredTargetUnits, + targets, + connectionCeiling +}) { + const quotas = new Map(targets.map((target) => [target.cellId, 0])) + for (let assigned = 0; assigned < sourceAssignments; assigned++) { + const candidates = targets + .map((target) => { + const quota = quotas.get(target.cellId) + 1 + const projectedConnections = connectionProjection( + target.currentConnections, + sourceConnections, + sourceAssignments, + quota + ) + const projectedUnits = projection(requiredTargetUnits, quota, sourceAssignments) + return { target, quota, projectedConnections, projectedUnits } + }) + .filter( + ({ target, quota, projectedConnections, projectedUnits }) => + projectedConnections < (target.connectionCeiling ?? connectionCeiling) && + quota <= target.availableConnectionReservations && + projectedUnits <= target.availableTargetUnits + ) + .sort( + (left, right) => + left.projectedConnections - right.projectedConnections || + left.target.cellId.localeCompare(right.target.cellId) + ) + const selected = candidates[0] + if (!selected) throw new Error('multi-target connection or request-unit headroom exhausted') + quotas.set(selected.target.cellId, selected.quota) + } + return targets.map((target) => ({ + ...target, + quota: quotas.get(target.cellId), + projectedConnections: connectionProjection( + target.currentConnections, + sourceConnections, + sourceAssignments, + quotas.get(target.cellId) + ), + projectedUnits: projection( + requiredTargetUnits, + quotas.get(target.cellId), + sourceAssignments + ) + })) +} + +function defaultCommand(args) { + const result = spawnSync('gcloud', args, { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }) + if (result.status !== 0) { + throw new Error(`gcloud ${args.slice(0, 5).join(' ')} failed: ${result.stderr.trim()}`) + } +} + +function defaultCommandResult(args) { + const result = spawnSync('gcloud', args, { + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'] + }) + return { status: result.status, stdout: result.stdout, stderr: result.stderr } +} + +function validatedProcessCounts(counts, cellId) { + for (const field of [ + 'totalConnections', + 'preAuthConnections', + 'controls', + 'splices', + 'pendingSplices', + 'queuedBytes' + ]) { + if (!Number.isSafeInteger(counts?.[field]) || counts[field] < 0) { + throw new Error(`${cellId} runtime metrics have invalid ${field}`) + } + } + return counts +} + +function processCounts(config, deps, status, cellId) { + if (status.process) return validatedProcessCounts(status.process, cellId) + const filter = [ + 'resource.type="gce_instance"', + 'jsonPayload.event="orca_relay_runtime_metrics"', + `jsonPayload.cellId="${cellId}"` + ].join(' AND ') + const entries = deps.commandJson([ + 'logging', + 'read', + filter, + '--project', + config.project, + '--freshness=5m', + '--limit=1', + '--order=desc', + '--format=json' + ]) + const entry = entries[0] + if (!entry?.jsonPayload) throw new Error(`${cellId} has no fresh runtime metrics`) + const timestamp = Date.parse(entry.timestamp) + if (!Number.isFinite(timestamp) || timestamp < deps.now() - 90_000) { + throw new Error(`${cellId} runtime metrics are stale`) + } + return validatedProcessCounts(entry.jsonPayload, cellId) +} + +function runtimeConnections(counts, cellId) { + const count = counts?.totalConnections + if (!Number.isSafeInteger(count) || count < 0) { + throw new Error(`${cellId} runtime does not expose totalConnections`) + } + return count +} + +function runtimeIncarnation(status, cellId) { + const incarnation = status.runtime?.cellIncarnation + if (typeof incarnation !== 'string' || incarnation.length === 0) { + throw new Error(`${cellId} has no exact runtime incarnation`) + } + return incarnation +} + +async function pairStatus(config, adminPost, targetCellId, completeReady) { + return await adminPost(config.directorOrigin, '/v1/admin/evacuation-status', { + v: 1, + sourceCellId: config.sourceCellId, + targetCellId, + completeReady + }) +} + +async function allPairStatuses(config, adminPost, completeReady) { + const statuses = [] + for (const targetCellId of config.targetCellIds) { + statuses.push({ + targetCellId, + status: await pairStatus(config, adminPost, targetCellId, completeReady) + }) + } + return statuses +} + +function statusTotals(statuses) { + return statuses.reduce( + (totals, { status }) => ({ + inProgress: totals.inProgress + status.inProgress, + targetRegistered: totals.targetRegistered + status.targetRegistered, + registeredSourceActive: totals.registeredSourceActive + status.registeredSourceActive, + registeredCompletable: totals.registeredCompletable + status.registeredCompletable, + registeredTargetInactive: + totals.registeredTargetInactive + status.registeredTargetInactive, + completed: totals.completed + status.completed, + blocked: totals.blocked + status.blocked, + expiredUnregistered: totals.expiredUnregistered + status.expiredUnregistered, + repairableExpiredUnregistered: + totals.repairableExpiredUnregistered + status.repairableExpiredUnregistered, + abortableExpiredUnregistered: + totals.abortableExpiredUnregistered + status.abortableExpiredUnregistered, + blockedExpiredUnregistered: + totals.blockedExpiredUnregistered + status.blockedExpiredUnregistered, + blockedExpiredOnNewerTargetAssignment: + totals.blockedExpiredOnNewerTargetAssignment + + status.blockedExpiredOnNewerTargetAssignment + }), + { + inProgress: 0, + targetRegistered: 0, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + } + ) +} + +function hasDurableTargetOwnership(totals) { + return ( + totals.inProgress === totals.targetRegistered && + totals.targetRegistered === + totals.registeredSourceActive + + totals.registeredCompletable + + totals.registeredTargetInactive && + totals.registeredSourceActive === 0 && + totals.expiredUnregistered === 0 && + totals.repairableExpiredUnregistered === 0 && + totals.abortableExpiredUnregistered === 0 && + totals.blockedExpiredUnregistered === 0 && + totals.blockedExpiredOnNewerTargetAssignment === 0 + ) +} + +function boundedUnregisteredMigrations(totals, unobservedConnectionBound) { + const unregistered = totals.inProgress - totals.targetRegistered + return ( + Number.isSafeInteger(unobservedConnectionBound) && + unregistered > 0 && + unregistered <= unobservedConnectionBound && + totals.targetRegistered === + totals.registeredSourceActive + + totals.registeredCompletable + + totals.registeredTargetInactive && + totals.registeredSourceActive === 0 && + totals.blocked === 0 && + totals.expiredUnregistered === 0 && + totals.repairableExpiredUnregistered === 0 && + totals.abortableExpiredUnregistered === 0 && + totals.blockedExpiredUnregistered === 0 && + totals.blockedExpiredOnNewerTargetAssignment === 0 + ) +} + +function isSettledOrOffline(totals) { + return ( + hasDurableTargetOwnership(totals) && + totals.registeredCompletable === 0 && + totals.blocked === totals.registeredTargetInactive + ) +} + +function assertLeaseGate(statuses, minimumLeaseRemainingMs) { + const remaining = statuses + .map(({ status }) => status.oldestRemainingMs) + .filter((value) => value !== null) + if (remaining.length === 0 || Math.min(...remaining) < minimumLeaseRemainingMs) { + throw new Error('oldest migration lease has insufficient time remaining') + } +} + +async function targetRuntime(config, adminPost, target) { + if (config.mode !== 'fence-source') { + const runtime = await adminPost(target.origin, '/v1/admin/runtime-status', { v: 1 }) + const count = runtime.runtime?.totalConnections + if (!Number.isSafeInteger(count) || count < 0) { + throw new Error(`${target.cellId} runtime does not expose totalConnections`) + } + if (count >= targetConnectionCeiling(config, target)) { + throw new Error(`${target.cellId} reached the connection ceiling`) + } + return count + } + const result = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: target.cellId + }) + const status = result.status + if ( + status?.cellId !== target.cellId || + status.cellUrl !== target.origin || + status.runtime?.cellUrl !== target.origin || + status.runtime?.ready !== true || + status.runtime?.heartbeatFresh !== true + ) { + throw new Error(`${target.cellId} has no fresh matching director runtime snapshot`) + } + const values = [ + status.runtime.observedRequests, + status.connectionCapacity?.observedConnections, + status.connectionCapacity?.enforcedConnectionUnits + ] + if (values.some((value) => !Number.isSafeInteger(value) || value < 0)) { + throw new Error(`${target.cellId} has incomplete director runtime counts`) + } + const count = Math.max(...values) + connectionReservationHeadroom(status, target.cellId) + if (count >= Math.min(config.connectionCeiling, status.connectionCapacity.hardCap)) { + throw new Error(`${target.cellId} reached the connection ceiling`) + } + return count +} + +async function checkPublicCellEndpoint(deps, cell, path) { + const response = await deps.fetch(`${cell.origin}${path}`, { + signal: AbortSignal.timeout(15_000) + }) + const body = await response.json().catch(() => ({})) + if (!response.ok || body.ok !== true) { + throw new Error(`${cell.cellId} ${path} is unavailable`) + } +} + +async function inspectGeneralPromotionTarget(config, deps, adminPost, target) { + await checkPublicCellEndpoint(deps, target, '/health') + await checkPublicCellEndpoint(deps, target, '/ready') + const result = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: target.cellId + }) + const status = result.status + if ( + status?.cellId !== target.cellId || + status.cellUrl !== target.origin || + status.runtime?.cellUrl !== target.origin || + status.runtime?.ready !== true || + status.runtime?.heartbeatFresh !== true + ) { + throw new Error(`${target.cellId} has no fresh matching director runtime snapshot`) + } + connectionReservationHeadroom(status, target.cellId) + const currentConnections = Math.max( + status.runtime.observedRequests, + status.connectionCapacity.observedConnections, + status.connectionCapacity.enforcedConnectionUnits + ) + if ( + !Number.isSafeInteger(currentConnections) || + currentConnections >= targetConnectionCeiling(config, target) + ) { + throw new Error(`${target.cellId} reached the connection ceiling`) + } +} + +function validateReviewedInstanceTemplate(template, expected, capacityPredecessor) { + const startupScript = (template.properties?.metadata?.items ?? []) + .find((item) => item.key === 'startup-script')?.value + const configuredDigest = startupScript?.match( + /ORCA_RELAY_IMAGE_DIGEST=%s\\n' '(sha256:[a-f0-9]{64})'/ + )?.[1] + const configuredImages = [ + ...String(startupScript ?? '').matchAll( + /'(?:[a-z0-9.-]+\/)+[a-z0-9._/-]+@(sha256:[a-f0-9]{64})'/g + ) + ].map((match) => match[1]) + if ( + template.selfLink !== expected.generationIdentity || + configuredDigest !== expected.imageDigest || + !configuredImages.includes(expected.imageDigest) + ) { + throw new Error(`${expected.cellId} instance template does not pin the reviewed image`) + } + const hardCaps = [ + ...String(startupScript ?? '').matchAll( + /^ printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '([0-9]+)'$/gm + ) + ].map((match) => Number(match[1])) + const unobservedBounds = [ + ...String(startupScript ?? '').matchAll( + /^ printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '([0-9]+)'$/gm + ) + ].map((match) => Number(match[1])) + const hardCap = hardCaps[0] + const unobservedBound = unobservedBounds[0] + const exactCapacityPredecessor = + capacityPredecessor !== undefined && + hardCaps.length === 1 && + unobservedBounds.length === 1 && + hardCap === capacityPredecessor.hardCap && + unobservedBound === capacityPredecessor.unobservedBound + if (expected.connectionHardCap === undefined) { + if (!exactCapacityPredecessor && (hardCaps.length !== 0 || unobservedBounds.length !== 0)) { + throw new Error(`${expected.cellId} instance template capacity differs from Terraform`) + } + return exactCapacityPredecessor + ? { + ...expected, + connectionHardCap: hardCap, + connectionUnobservedBound: unobservedBound + } + : expected + } + const isReviewedPredecessor = + exactCapacityPredecessor && expected.connectionHardCap === 1_000 + if ( + hardCaps.length !== 1 || + unobservedBounds.length !== 1 || + !Number.isSafeInteger(hardCap) || + !Number.isSafeInteger(unobservedBound) || + (hardCap !== expected.connectionHardCap && !isReviewedPredecessor) || + unobservedBound !== expected.connectionUnobservedBound + ) { + throw new Error(`${expected.cellId} instance template capacity is outside reviewed rollout`) + } + return { + ...expected, + connectionHardCap: hardCap, + connectionUnobservedBound: unobservedBound + } +} + +async function inspectDirectorObservedCell(config, deps, adminPost, expected) { + const common = ['--project', config.project, '--zone', expected.zone, '--format=json'] + const mig = deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + expected.migName, + ...common + ]) + const instances = deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'list-instances', + expected.migName, + ...common + ]) + const instanceName = validateMig(mig, instances, expected) + if ( + mig.instanceTemplate !== expected.generationIdentity || + instances[0]?.version?.instanceTemplate !== expected.generationIdentity + ) { + throw new Error(`${expected.cellId} MIG does not serve the reviewed generation`) + } + const instance = deps.commandJson([ + 'compute', + 'instances', + 'describe', + instanceName, + ...common + ]) + validateInstance(instance, expected, config.runtimeServiceAccount) + const templateName = new URL(expected.generationIdentity).pathname.split('/').at(-1) + const template = deps.commandJson([ + 'compute', + 'instance-templates', + 'describe', + templateName, + '--project', + config.project, + '--format=json' + ]) + const deployed = validateReviewedInstanceTemplate( + template, + expected, + config.mode === 'fence-source' && + (expected.connectionHardCap !== undefined || expected.legacyCapacityTopology) + ? { + hardCap: config.connectionCeiling, + unobservedBound: config.unobservedConnectionBound + } + : undefined + ) + const backend = deps.commandJson([ + 'compute', + 'backend-services', + 'describe', + expected.backendName, + '--global', + '--project', + config.project, + '--format=json' + ]) + validateBackend(backend, expected) + await assertCellRoute(config, deps, expected) + await checkPublicCellEndpoint(deps, expected, '/health') + await checkPublicCellEndpoint(deps, expected, '/ready') + const result = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: expected.cellId + }) + const status = result.status + if ( + status?.cellId !== expected.cellId || + status.cellUrl !== expected.origin || + status.runtime?.cellUrl !== expected.origin || + status.runtime?.ready !== true || + status.runtime?.heartbeatFresh !== true + ) { + throw new Error(`${expected.cellId} has no fresh matching director runtime snapshot`) + } + connectionReservationHeadroom(status, expected.cellId) + assertDeploymentConnectionCapacity( + deployed, + status.connectionCapacity ?? null, + status.connectionCapacity ?? null + ) + const connectionValues = [ + status.runtime.observedRequests, + status.connectionCapacity.observedConnections, + status.connectionCapacity.enforcedConnectionUnits + ] + if (connectionValues.some((value) => !Number.isSafeInteger(value) || value < 0)) { + throw new Error(`${expected.cellId} has incomplete director runtime counts`) + } + const currentConnections = Math.max(...connectionValues) + if (currentConnections >= targetConnectionCeiling(config, deployed)) { + throw new Error(`${expected.cellId} reached the connection ceiling`) + } + return { + ...status, + process: { totalConnections: currentConnections } + } +} + +async function waitForMultiStatus( + config, + deps, + adminPost, + targets, + completeReady, + allowBoundedUnregistered = false, + requireZero = false +) { + const deadline = deps.now() + config.timeoutMs + while (deps.now() < deadline) { + const statuses = await allPairStatuses(config, adminPost, completeReady) + for (const target of targets) await targetRuntime(config, adminPost, target) + const totals = statusTotals(statuses) + deps.emit({ + event: completeReady ? 'multi_migration_completion' : 'multi_migration_registration', + ...totals + }) + const boundedUnregistered = + allowBoundedUnregistered && + boundedUnregisteredMigrations(totals, config.unobservedConnectionBound) + const boundedSettlement = boundedUnregistered && totals.registeredCompletable === 0 + if ( + completeReady + ? (requireZero + ? totals.inProgress === 0 + : isSettledOrOffline(totals) || boundedSettlement) + : totals.inProgress === totals.targetRegistered || boundedUnregistered + ) { + if (boundedUnregistered) { + deps.emit({ + event: completeReady + ? 'multi_migration_bounded_offline_complete' + : 'multi_migration_bounded_offline_registered', + unobservedConnectionBound: config.unobservedConnectionBound, + unregistered: totals.inProgress - totals.targetRegistered, + ...totals + }) + } + return statuses + } + await deps.wait(config.pollIntervalMs) + } + throw new Error( + completeReady + ? 'timed out waiting for multi-target completion' + : 'timed out waiting for multi-target registration' + ) +} + +async function waitForRecoveredSourceZero( + config, + deps, + adminPost, + source, + expectedIncarnation +) { + const deadline = deps.now() + config.timeoutMs + while (deps.now() < deadline) { + const status = await inspectCell(config, deps, adminPost, source) + const counts = processCounts(config, deps, status, source.cellId) + deps.emit({ + event: 'source_recovery_runtime_settlement', + draining: status.draining, + activityLeases: status.activityLeases, + reservedRequests: status.reservedRequests, + controls: counts.controls, + splices: counts.splices, + pendingSplices: counts.pendingSplices + }) + if ( + runtimeIncarnation(status, source.cellId) !== expectedIncarnation + ) { + throw new Error('source incarnation changed during recovery settlement') + } + if ( + status.draining && + status.activityLeases === 0 && + status.reservedRequests === 0 && + counts.controls === 0 && + counts.splices === 0 && + counts.pendingSplices === 0 + ) { + return + } + await deps.wait(config.pollIntervalMs) + } + throw new Error('timed out waiting for recovered source runtime to reach zero') +} + +async function rollbackBeforeDrain(config, deps, adminPost, targets, selectorActive) { + let statuses + try { + statuses = await allPairStatuses(config, adminPost, false) + } catch (error) { + deps.emit({ + event: 'multi_forward_recovery_required', + targetRegistered: null, + reason: 'registration_status_unavailable' + }) + throw new Error('cannot prove zero target registrations; preserving forward recovery', { + cause: error + }) + } + const totals = statusTotals(statuses) + if (totals.targetRegistered > 0) { + deps.emit({ event: 'multi_forward_recovery_required', ...totals }) + return + } + if (selectorActive) { + deps.emit({ event: 'multi_rollback_preserved_selector', ...totals }) + return + } + await setCellState(config, adminPost, config.sourceCellId, true).catch(() => undefined) + for (const target of targets) { + await setCellState(config, adminPost, target.cellId, false).catch(() => undefined) + } + deps.emit({ event: 'multi_rollback_waiting_for_lease_expiry', ...totals }) +} + +async function publishMigrations(config, deps, adminPost, plannedTargets) { + for (const target of plannedTargets) { + let remaining = target.quota + while (remaining > 0) { + const limit = Math.min(config.batchSize, remaining) + let result + try { + result = await adminPost(config.directorOrigin, '/v1/admin/evacuate-cell', { + v: 1, + sourceCellId: config.sourceCellId, + targetCellId: target.cellId, + limit + }) + } catch (error) { + if ( + config.mode !== 'recover-forward' || + !(error instanceof Error) || + error.message !== + '/v1/admin/evacuate-cell failed: relay_connection_headroom_exhausted' + ) { + throw error + } + deps.emit({ + event: 'multi_recovery_target_headroom_paused', + targetCellId: target.cellId, + remaining + }) + break + } + if ( + !Number.isSafeInteger(result.started) || + result.started < 0 || + result.started > limit + ) { + throw new Error(`${target.cellId} migration quota could not be filled deterministically`) + } + if (result.started === 0 && config.mode === 'recover-forward') { + deps.emit({ + event: 'multi_recovery_quota_depleted', + targetCellId: target.cellId, + remaining + }) + break + } + if (result.started === 0) { + throw new Error(`${target.cellId} migration quota could not be filled deterministically`) + } + remaining -= result.started + deps.emit({ + event: 'multi_migration_batch', + targetCellId: target.cellId, + started: result.started, + remaining + }) + } + } +} + +function sourceMig(config, deps, source) { + const common = ['--project', config.project, '--zone', source.zone, '--format=json'] + return { + mig: deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + source.migName, + ...common + ]), + instances: deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'list-instances', + source.migName, + ...common + ]) + } +} + +function validateFencedMig(mig, source) { + const policy = mig.updatePolicy ?? {} + if ( + Number(mig.targetSize) !== 0 || + mig.instanceTemplate !== source.generationIdentity || + policy.replacementMethod !== 'RECREATE' || + Number(policy.maxSurge?.fixed ?? policy.maxSurge) !== 0 || + Number(policy.maxUnavailable?.fixed ?? policy.maxUnavailable) !== 1 + ) { + throw new Error(`${source.cellId} MIG fence topology is unsafe`) + } +} + +function canonicalBackendServiceId(value) { + if (typeof value !== 'string') return null + const prefix = 'https://www.googleapis.com/compute/v1/' + const resource = value.startsWith(prefix) ? value.slice(prefix.length) : value + return /^projects\/[a-z][a-z0-9-]{4,29}\/global\/backendServices\/[A-Za-z0-9_-]{1,63}$/.test( + resource + ) + ? resource + : null +} + +export function sameBackendServiceResource(left, right) { + if (left === right) return true + const leftId = canonicalBackendServiceId(left) + return leftId !== null && leftId === canonicalBackendServiceId(right) +} + +function validateRetainedRoute(urlMap, source) { + const hostname = new URL(source.origin).hostname + const hostRule = (urlMap.hostRules ?? []).find((rule) => + (rule.hosts ?? []).includes(hostname) + ) + const matcher = (urlMap.pathMatchers ?? []).find( + (candidate) => candidate.name === hostRule?.pathMatcher + ) + if ( + !source.urlMapName || + urlMap.name !== source.urlMapName || + !sameBackendServiceResource(matcher?.defaultService, source.backendId) || + (matcher.pathRules?.length ?? 0) !== 0 || + (matcher.routeRules?.length ?? 0) !== 0 || + matcher.defaultRouteAction !== undefined || + matcher.defaultUrlRedirect !== undefined || + matcher.headerAction !== undefined + ) { + throw new Error(`${source.cellId} retained route topology mismatch`) + } +} + +async function assertCellRoute(config, deps, cell) { + const urlMap = deps.commandJson([ + 'compute', + 'url-maps', + 'describe', + cell.urlMapName, + '--global', + '--project', + config.project, + '--format=json' + ]) + validateRetainedRoute(urlMap, cell) + const proxy = deps.commandJson([ + 'compute', + 'target-https-proxies', + 'describe', + cell.urlMapName, + '--global', + '--project', + config.project, + '--format=json' + ]) + const forwardingRule = deps.commandJson([ + 'compute', + 'forwarding-rules', + 'describe', + cell.urlMapName, + '--global', + '--project', + config.project, + '--format=json' + ]) + const address = deps.commandJson([ + 'compute', + 'addresses', + 'describe', + cell.urlMapName, + '--global', + '--project', + config.project, + '--format=json' + ]) + const resolved = await deps.resolve4(new URL(cell.origin).hostname) + if ( + proxy.name !== cell.urlMapName || + proxy.urlMap !== urlMap.selfLink || + forwardingRule.name !== cell.urlMapName || + forwardingRule.target !== proxy.selfLink || + forwardingRule.IPAddress !== address.address || + forwardingRule.portRange !== '443-443' || + forwardingRule.loadBalancingScheme !== 'EXTERNAL_MANAGED' || + !Array.isArray(resolved) || + resolved.length === 0 || + resolved.some((value) => value !== address.address) + ) { + throw new Error(`${cell.cellId} live frontend topology mismatch`) + } +} + +async function inspectFencedSource(config, deps, adminPost, source, mig) { + validateFencedMig(mig, source) + const backend = deps.commandJson([ + 'compute', + 'backend-services', + 'describe', + source.backendName, + '--global', + '--project', + config.project, + '--format=json' + ]) + validateBackend(backend, source) + await assertCellRoute(config, deps, source) + const result = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: source.cellId + }) + const status = result.status + if ( + status?.cellUrl !== source.origin || + (status.runtime !== null && status.runtime?.cellUrl !== source.origin) + ) { + throw new Error(`${source.cellId} fenced runtime does not match Terraform topology`) + } + return { ...status, draining: true, process: null } +} + +async function inspectFenceCandidate(config, deps, adminPost, cell, mig) { + const targetSize = Number(mig.targetSize) + const policy = mig.updatePolicy ?? {} + if ( + ![0, 1].includes(targetSize) || + policy.replacementMethod !== 'RECREATE' || + Number(policy.maxSurge?.fixed ?? policy.maxSurge) !== 0 || + Number(policy.maxUnavailable?.fixed ?? policy.maxUnavailable) !== 1 + ) { + throw new Error(`${cell.cellId} MIG fence topology is unsafe`) + } + const backend = deps.commandJson([ + 'compute', + 'backend-services', + 'describe', + cell.backendName, + '--global', + '--project', + config.project, + '--format=json' + ]) + validateBackend(backend, cell) + const result = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: cell.cellId + }) + if ( + result.status?.cellUrl !== cell.origin || + result.status.runtime?.cellUrl !== cell.origin + ) { + throw new Error(`${cell.cellId} retained topology does not match Terraform`) + } + return result.status +} + +const CAPACITY_SNAPSHOT_ATTEMPTS = 3 +const CAPACITY_SNAPSHOT_RETRY_MS = 250 +const RECOVERY_CATCH_UP_PASSES = 5 +const RECOVERY_TARGET_OWNERSHIP_TIMEOUT_MS = 2 * 60 * 1_000 + +function conservativeRecoveryCapacity(rounds, targets, requireZeroProof) { + const capacities = rounds.flat() + for (const capacity of capacities) { + if ( + !Number.isSafeInteger(capacity.sourceAssignments) || + capacity.sourceAssignments < 0 || + !Number.isSafeInteger(capacity.requiredTargetUnits) || + capacity.requiredTargetUnits < capacity.sourceAssignments || + !Number.isSafeInteger(capacity.availableTargetUnits) || + capacity.availableTargetUnits < 0 + ) { + throw new Error('target capacity snapshot is internally inconsistent') + } + } + const observedSourceAssignments = capacities.map( + (capacity) => capacity.sourceAssignments + ) + const sourceAssignments = requireZeroProof + ? Math.max(...observedSourceAssignments) + : Math.min(...observedSourceAssignments) + const requiredTargetUnits = + sourceAssignments === 0 + ? 0 + : requireZeroProof + ? Math.max(...capacities.map((capacity) => capacity.requiredTargetUnits)) + : Math.max( + ...capacities.map((capacity) => + Math.ceil( + sourceAssignments * + (capacity.requiredTargetUnits / capacity.sourceAssignments) + ) + ) + ) + return { + sourceAssignments, + requiredTargetUnits, + availableTargetUnits: new Map(targets.map((target, index) => [ + target.cellId, + Math.min(...rounds.map((round) => round[index].availableTargetUnits)) + ])) + } +} + +async function readTargetCapacitySnapshot( + config, + deps, + adminPost, + source, + targets, + requireRecoveryZeroProof = false +) { + for (let attempt = 0; attempt < CAPACITY_SNAPSHOT_ATTEMPTS; attempt++) { + const rounds = [] + for (let round = 0; round < 2; round++) { + const capacities = [] + for (const target of targets) { + capacities.push(await adminPost( + config.directorOrigin, + '/v1/admin/evacuation-capacity', + { v: 1, sourceCellId: source.cellId, targetCellId: target.cellId } + )) + } + rounds.push(capacities) + } + if (config.mode === 'recover-forward') { + return conservativeRecoveryCapacity(rounds, targets, requireRecoveryZeroProof) + } + const capacities = rounds.flat() + const baseline = capacities[0] + if (capacities.every((capacity) => + capacity.sourceAssignments === baseline.sourceAssignments && + capacity.requiredTargetUnits === baseline.requiredTargetUnits + )) { + return { + sourceAssignments: baseline.sourceAssignments, + requiredTargetUnits: baseline.requiredTargetUnits, + availableTargetUnits: new Map(targets.map((target, index) => [ + target.cellId, + Math.min(...rounds.map((round) => round[index].availableTargetUnits)) + ])) + } + } + if (attempt + 1 < CAPACITY_SNAPSHOT_ATTEMPTS) { + await deps.wait(CAPACITY_SNAPSHOT_RETRY_MS) + } + } + throw new Error('target capacity snapshots disagree') +} + +async function preflight( + config, + deps, + adminPost, + source, + targets, + sourceFence = null, + selector = null, + coveredSourceConnections = null +) { + const sourceStatus = sourceFence + ? await inspectFencedSource(config, deps, adminPost, source, sourceFence.mig) + : config.mode === 'fence-source' + ? { + ...await inspectDirectorObservedCell(config, deps, adminPost, source), + process: null + } + : await inspectCell(config, deps, adminPost, source) + const sourceProcess = sourceFence + ? null + : processCounts(config, deps, sourceStatus, source.cellId) + const targetStatuses = [] + for (const target of targets) { + const status = config.mode === 'fence-source' + ? { + ...await inspectDirectorObservedCell(config, deps, adminPost, target), + process: null + } + : await inspectCell(config, deps, adminPost, target) + const targetProcess = processCounts(config, deps, status, target.cellId) + const targetAdmission = selector + ? selectorCellState(selector, target.cellId) + : status.enabled + ? 'general' + : 'existing-only' + if ( + ['preflight', 'execute'].includes(config.mode) && + (selector ? targetAdmission === 'general' : status.enabled) + ) { + throw new Error(`${target.cellId} must not start in general admission`) + } + targetStatuses.push({ + ...target, + status, + process: targetProcess, + currentConnections: runtimeConnections(targetProcess, target.cellId), + availableConnectionReservations: connectionReservationHeadroom( + status, + target.cellId + ), + connectionCeiling: Math.min( + config.connectionCeiling, + status.connectionCapacity.hardCap + ) + }) + } + if (config.mode === 'audit' || config.mode === 'recover-forward') { + const statuses = await allPairStatuses(config, adminPost, false) + deps.emit({ + event: config.mode === 'audit' ? 'multi_target_audit' : 'multi_forward_recovery_preflight', + ...statusTotals(statuses) + }) + if (config.mode === 'audit') { + return { + sourceProcess, + sourceStatus, + plannedTargets: targetStatuses, + sourceAlreadyFenced: Boolean(sourceFence) + } + } + } + const capacity = await readTargetCapacitySnapshot( + config, + deps, + adminPost, + source, + targets + ) + const { sourceAssignments, requiredTargetUnits } = capacity + const observedSourceConnections = sourceFence + ? 0 + : runtimeConnections(sourceProcess, source.cellId) + const sourceConnections = + coveredSourceConnections === null + ? observedSourceConnections + : sourceAssignments + + Math.max(0, observedSourceConnections - coveredSourceConnections) + for (const target of targetStatuses) { + target.availableTargetUnits = capacity.availableTargetUnits.get(target.cellId) + } + deps.emit({ + event: 'multi_target_capacity_snapshot', + sourceConnections, + observedSourceConnections, + sourceAssignments, + requiredTargetUnits, + targets: targetStatuses.map((target) => ({ + cellId: target.cellId, + currentConnections: target.currentConnections, + availableConnectionReservations: target.availableConnectionReservations, + availableTargetUnits: target.availableTargetUnits + })) + }) + if ( + config.mode === 'execute' && + (selector + ? selectorCellState(selector, source.cellId) !== 'existing-only' + : !sourceStatus.enabled) + ) { + throw new Error(selector ? 'source cell is not existing-only' : 'source cell is not enabled') + } + const plannedTargets = allocateTargetQuotas({ + sourceAssignments, + sourceConnections, + requiredTargetUnits, + targets: targetStatuses, + connectionCeiling: config.connectionCeiling + }) + deps.emit({ + event: 'multi_target_preflight', + source: aggregateCellStatus(sourceStatus), + sourceConnections, + observedSourceConnections, + sourceAssignments, + targets: plannedTargets.map((target) => ({ + cellId: target.cellId, + quota: target.quota, + currentConnections: target.currentConnections, + projectedConnections: target.projectedConnections, + projectedUnits: target.projectedUnits + })) + }) + return { sourceProcess, sourceStatus, plannedTargets, sourceAlreadyFenced: Boolean(sourceFence) } +} + +async function assertRecoveryPreDrain( + config, + deps, + adminPost, + source, + targets, + selectorPost, + expectedSelector +) { + const statuses = await allPairStatuses(config, adminPost, false) + const totals = statusTotals(statuses) + const capacity = await readTargetCapacitySnapshot( + config, + deps, + adminPost, + source, + targets, + true + ) + if (capacity.sourceAssignments !== 0 || capacity.requiredTargetUnits !== 0) { + deps.emit({ + event: 'multi_forward_recovery_catch_up', + sourceAssignments: capacity.sourceAssignments, + requiredTargetUnits: capacity.requiredTargetUnits + }) + return false + } + for (const target of targets) await targetRuntime(config, adminPost, target) + if (expectedSelector) { + const current = await inspectAdmissionSelector(selectorPost) + if ( + current.selector.generation !== expectedSelector.generation || + JSON.stringify(current.selector.membership) !== + JSON.stringify(expectedSelector.membership) + ) { + throw new Error('admission selector changed before recovery drain') + } + } + deps.emit({ + event: 'multi_forward_recovery_ready_to_drain', + ...totals + }) + return true +} + +async function waitForRecoveryTargetOwnership(config, deps, adminPost, targets) { + const deadline = + deps.now() + Math.min(config.timeoutMs, RECOVERY_TARGET_OWNERSHIP_TIMEOUT_MS) + while (deps.now() < deadline) { + const statuses = await allPairStatuses(config, adminPost, false) + for (const target of targets) await targetRuntime(config, adminPost, target) + const totals = statusTotals(statuses) + deps.emit({ event: 'multi_recovery_target_ownership', ...totals }) + assertLeaseGate(statuses, config.minimumLeaseRemainingMs) + if (hasDurableTargetOwnership(totals)) return + if (boundedUnregisteredMigrations(totals, config.unobservedConnectionBound)) { + const unregistered = totals.inProgress - totals.targetRegistered + deps.emit({ + event: + totals.targetRegistered === 0 + ? 'multi_recovery_bounded_unregistered' + : 'multi_recovery_bounded_mixed_registration', + unobservedConnectionBound: config.unobservedConnectionBound, + unregistered, + ...totals + }) + return + } + await deps.wait(config.pollIntervalMs) + } + throw new Error('timed out waiting for recovery target ownership') +} + +async function runEvacuation( + config, + deps, + adminPost, + token, + source, + targets, + plannedTargets, + sourceStatus, + selectorActive, + selectorPost, + expectedSelector +) { + let drainAttempted = false + try { + if (!selectorActive) { + await setCellState(config, adminPost, source.cellId, false) + for (const target of targets) await setCellState(config, adminPost, target.cellId, true) + } + await publishMigrations(config, deps, adminPost, plannedTargets) + const statuses = await allPairStatuses(config, adminPost, false) + assertLeaseGate(statuses, config.minimumLeaseRemainingMs) + for (const target of targets) await targetRuntime(config, adminPost, target) + if (selectorActive) { + const current = await inspectAdmissionSelector(selectorPost) + if ( + current.selector.generation !== expectedSelector.generation || + JSON.stringify(current.selector.membership) !== + JSON.stringify(expectedSelector.membership) + ) { + throw new Error('admission selector changed before drain') + } + } + const attemptId = randomUUID() + const traceValue = randomUUID() + const prepared = await adminPost( + config.directorOrigin, + '/v1/admin/drain-attempt-prepare', + { + v: 1, + attemptId, + cellId: source.cellId, + cellIncarnation: runtimeIncarnation(sourceStatus, source.cellId), + traceValue, + graceMs: 120_000, + confirmation: 'PREPARE_LEGACY_DRAIN' + } + ) + if (prepared.state !== 'prepared') { + throw new Error('planned drain already recorded; use recover-forward') + } + drainAttempted = true + const sending = await adminPost( + config.directorOrigin, + '/v1/admin/drain-attempt-send', + { + v: 1, + attemptId, + cellId: source.cellId, + cellIncarnation: runtimeIncarnation(sourceStatus, source.cellId) + } + ) + if ( + sending.attempt?.state !== 'send-may-have-started' || + sending.attempt.shouldSend !== true || + !Number.isSafeInteger(sending.attempt.sendPermitExpiresAt) || + deps.now() >= sending.attempt.sendPermitExpiresAt + ) { + throw new Error('drain send permit unavailable') + } + const receipt = await drainSource(config, deps, token, source, 120_000, traceValue) + await adminPost(config.directorOrigin, '/v1/admin/drain-attempt-receipt', { + v: 1, + attemptId, + cellId: source.cellId, + cellIncarnation: runtimeIncarnation(sourceStatus, source.cellId), + traceValue, + ...receipt + }) + deps.emit({ event: 'source_drain_accepted', sourceCellId: source.cellId }) + await waitForMultiStatus(config, deps, adminPost, targets, false) + await waitForMultiStatus(config, deps, adminPost, targets, true) + deps.emit({ event: 'multi_target_complete', sourceCellId: source.cellId }) + } catch (error) { + if (drainAttempted) { + const statuses = await allPairStatuses(config, adminPost, false).catch(() => []) + deps.emit({ event: 'multi_forward_recovery_required', ...statusTotals(statuses) }) + } else { + await rollbackBeforeDrain(config, deps, adminPost, targets, selectorActive) + } + throw error + } +} + +async function waitForFence(config, deps, adminPost, source) { + const deadline = deps.now() + config.timeoutMs + while (deps.now() < deadline) { + const mig = deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + source.migName, + '--project', + config.project, + '--zone', + source.zone, + '--format=json' + ]) + const instances = deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'list-instances', + source.migName, + '--project', + config.project, + '--zone', + source.zone, + '--format=json' + ]) + const status = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: source.cellId + }) + if ( + Number(mig.targetSize) === 0 && + instances.length === 0 && + status.status?.cellUrl === source.origin && + status.status.enabled === false && + !status.status.runtime?.heartbeatFresh + ) { + const incarnation = status.status.runtime?.cellIncarnation + if (typeof incarnation !== 'string' || incarnation.length === 0) { + throw new Error('fenced source has no exact runtime incarnation') + } + return incarnation + } + await deps.wait(config.pollIntervalMs) + } + throw new Error('timed out waiting for durable source fence') +} + +function terraformFenceConfig(config, cell, cellIncarnation) { + return { + project: config.project, + environment: config.environment, + terraformDir: config.terraformDir, + varFile: config.terraformVarFile, + lockTimeout: '5m', + fenceCommit: config.fenceCommit, + cellIncarnation, + cell + } +} + +function fenceAttemptBody(attempt) { + return { + v: 1, + attemptId: attempt.attemptId, + environment: attempt.environment, + cellId: attempt.cellId, + cellIncarnation: attempt.cellIncarnation, + migName: attempt.migName, + instanceGroup: attempt.instanceGroup, + generationIdentity: attempt.generationIdentity, + fenceCommit: attempt.fenceCommit, + planSha256: attempt.planSha256, + planObjectName: attempt.planObjectName, + planObjectGeneration: attempt.planObjectGeneration, + varFileSha256: attempt.varFileSha256, + terraformStateLineage: attempt.terraformStateLineage, + terraformStateSerial: attempt.terraformStateSerial, + terraformStateObjectGeneration: attempt.terraformStateObjectGeneration, + terraformStateObjectSha256: attempt.terraformStateObjectSha256, + requestReason: attempt.requestReason, + ...(attempt.gceOperation ? { gceOperation: attempt.gceOperation } : {}) + } +} + +async function runTerraformManagedFence( + config, + deps, + adminPost, + cell, + cellIncarnation, + alreadyFenced, + preApplyGuard, + postApplyGuard +) { + const fenceConfig = terraformFenceConfig(config, cell, cellIncarnation) + const inspectProgress = async (_expected, attempt) => + await inspectTerraformFenceProgress( + fenceConfig, + { + terraform: deps.terraform, + gcloudJson: deps.commandJson + }, + attempt + ) + const attest = async (attempt) => { + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attest', { + ...fenceAttemptBody(attempt), + confirmation: 'ATTEST_TERRAFORM_FENCED_CELL' + }) + } + const attemptResult = await adminPost( + config.directorOrigin, + '/v1/admin/cell-fence-attempt-status', + { v: 1, cellId: cell.cellId } + ) + let existingAttempt = attemptResult.attempt + const planStore = { + uploadPlan: async (planPath, attempt) => + await uploadTerraformFencePlan( + fenceConfig, + { command: deps.command, commandJson: deps.commandJson }, + planPath, + attempt + ), + downloadPlan: async (attempt, planPath) => + await downloadTerraformFencePlan( + fenceConfig, + { command: deps.command }, + attempt, + planPath + ), + deletePlan: async (attempt) => + await deleteTerraformFencePlan( + fenceConfig, + { commandResult: deps.commandResult }, + attempt + ), + stateObjectBinding: async (statePath) => + await readTerraformStateObjectBinding( + fenceConfig, + { command: deps.command, commandJson: deps.commandJson }, + statePath + ) + } + if ( + existingAttempt && + existingAttempt.fenceCommit !== config.fenceCommit && + config.completedFenceRecovery + ) { + await deps.terraformFenceRecoverCompleted( + fenceConfig, + { + terraform: deps.terraform, + assertCommittedFenceSet: async () => + assertTerraformFenceSet(fenceConfig, { terraform: deps.terraform }), + loadAttempt: async () => existingAttempt, + resolvePlan: async (attempt) => + await resolveTerraformFencePlanGeneration( + fenceConfig, + { commandResult: deps.commandResult }, + attempt + ), + stateObjectBinding: planStore.stateObjectBinding, + downloadPlan: planStore.downloadPlan, + deletePlan: planStore.deletePlan, + inspectCompletedProgress: async (_expected, attempt, recovery) => + await inspectCompletedTerraformFenceProgress( + fenceConfig, + { + terraform: deps.terraform, + gcloudJson: deps.commandJson + }, + attempt, + recovery + ), + markOperation: async (attempt, invocation) => + await adminPost( + config.directorOrigin, + '/v1/admin/cell-fence-attempt-operation', + { + ...fenceAttemptBody(attempt), + invocationId: invocation.invocationId, + invocationRequestReason: invocation.requestReason, + confirmation: 'RECORD_TERRAFORM_CELL_FENCE_OPERATION' + } + ), + assertZeroDiff: async () => + assertTerraformFenceZeroDiff(fenceConfig, { + terraform: deps.terraform + }), + postApplyGuard, + attest, + emit: deps.emit + }, + config.completedFenceRecovery + ) + return + } + if ( + existingAttempt && + !existingAttempt.abortedAt && + !existingAttempt.completedAt && + existingAttempt.fenceCommit !== config.fenceCommit + ) { + await deps.terraformFenceSupersede(fenceConfig, { + assertCommittedFenceSet: async () => + assertTerraformFenceSet(fenceConfig, { terraform: deps.terraform }), + loadAttempt: async () => existingAttempt, + resolvePlan: async (attempt) => + await resolveTerraformFencePlanGeneration( + fenceConfig, + { commandResult: deps.commandResult }, + attempt + ), + inspectProgress, + abortAttempt: async (attempt) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-abort', { + ...fenceAttemptBody(attempt), + confirmation: 'ABORT_UNSTARTED_TERRAFORM_CELL_FENCE' + }), + emit: deps.emit + }) + existingAttempt = null + } + if ( + existingAttempt && + !existingAttempt.abortedAt && + (alreadyFenced || !existingAttempt.completedAt) + ) { + await deps.terraformFenceResume(fenceConfig, { + terraform: deps.terraform, + assertCommittedFenceSet: async () => + assertTerraformFenceSet(fenceConfig, { terraform: deps.terraform }), + loadAttempt: async () => existingAttempt, + resolvePlan: async (attempt) => + await resolveTerraformFencePlanGeneration( + fenceConfig, + { commandResult: deps.commandResult }, + attempt + ), + bindPlan: async (attempt) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-plan', { + ...fenceAttemptBody(attempt), + confirmation: 'BIND_TERRAFORM_CELL_FENCE_PLAN' + }), + inspectProgress, + assertZeroDiff: async () => + assertTerraformFenceZeroDiff(fenceConfig, { terraform: deps.terraform }), + assertStateFenced: async () => + assertTerraformFenceStateFenced(fenceConfig, { terraform: deps.terraform }), + preApplyGuard, + postApplyGuard, + markOperation: async (attempt, invocation) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-operation', { + ...fenceAttemptBody(attempt), + invocationId: invocation.invocationId, + invocationRequestReason: invocation.requestReason, + confirmation: 'RECORD_TERRAFORM_CELL_FENCE_OPERATION' + }), + markApplyStarted: async (attempt, invocation) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-start', { + ...fenceAttemptBody(attempt), + invocationId: invocation.invocationId, + invocationRequestReason: invocation.requestReason, + confirmation: 'START_TERRAFORM_CELL_FENCE' + }), + attest, + ...planStore, + emit: deps.emit + }) + return + } + if (alreadyFenced) { + await deps.terraformFenceAdopt(fenceConfig, { + loadAttempt: async () => + ( + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-status', { + v: 1, + cellId: cell.cellId + }) + ).attempt, + assertCommittedFenceSet: async () => + assertTerraformFenceSet(fenceConfig, { terraform: deps.terraform }), + assertStateFenced: async () => + assertTerraformFenceStateFenced(fenceConfig, { terraform: deps.terraform }), + preApplyGuard, + postApplyGuard, + attest: async (incarnation) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-adopt-legacy', { + v: 1, + cellId: cell.cellId, + cellIncarnation: incarnation, + confirmation: 'ADOPT_LEGACY_TERRAFORM_CELL_FENCE' + }), + commitAdoption: async (incarnation) => + await adminPost( + config.directorOrigin, + '/v1/admin/cell-fence-commit-legacy-adoption', + { + v: 1, + cellId: cell.cellId, + cellIncarnation: incarnation, + confirmation: 'COMMIT_LEGACY_TERRAFORM_CELL_FENCE_ADOPTION' + } + ), + emit: deps.emit + }) + return + } + if (existingAttempt && !existingAttempt.abortedAt && !existingAttempt.completedAt) { + throw new Error('prepared Terraform fence attempt must be aborted before replacement') + } + await deps.terraformFenceApply(fenceConfig, { + terraform: deps.terraform, + assertCommittedFenceSet: async () => + assertTerraformFenceSet(fenceConfig, { terraform: deps.terraform }), + inspectProgress, + assertZeroDiff: async () => + assertTerraformFenceZeroDiff(fenceConfig, { terraform: deps.terraform }), + preApplyGuard, + postApplyGuard, + ...planStore, + prepareAttempt: async (attempt) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-prepare', { + ...fenceAttemptBody(attempt), + confirmation: 'PREPARE_TERRAFORM_CELL_FENCE' + }), + bindPlan: async (attempt) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-plan', { + ...fenceAttemptBody(attempt), + confirmation: 'BIND_TERRAFORM_CELL_FENCE_PLAN' + }), + markApplyStarted: async (attempt, invocation) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-start', { + ...fenceAttemptBody(attempt), + invocationId: invocation.invocationId, + invocationRequestReason: invocation.requestReason, + confirmation: 'START_TERRAFORM_CELL_FENCE' + }), + markOperation: async (attempt, invocation) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-operation', { + ...fenceAttemptBody(attempt), + invocationId: invocation.invocationId, + invocationRequestReason: invocation.requestReason, + confirmation: 'RECORD_TERRAFORM_CELL_FENCE_OPERATION' + }), + attest, + emit: deps.emit + }) +} + +async function abortTerraformManagedFence(config, deps, adminPost, cell) { + const result = await adminPost( + config.directorOrigin, + '/v1/admin/cell-fence-attempt-status', + { v: 1, cellId: cell.cellId } + ) + const attempt = result.attempt + const fenceConfig = terraformFenceConfig(config, cell, attempt?.cellIncarnation) + await deps.terraformFenceAbort(fenceConfig, { + terraform: deps.terraform, + assertCommittedFenceSet: async () => + assertTerraformFenceSet(fenceConfig, { terraform: deps.terraform }), + loadAttempt: async () => attempt, + resolvePlan: async (value) => + await resolveTerraformFencePlanGeneration( + fenceConfig, + { commandResult: deps.commandResult }, + value + ), + bindPlan: async (value) => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-plan', { + ...fenceAttemptBody(value), + confirmation: 'BIND_TERRAFORM_CELL_FENCE_PLAN' + }), + inspectProgress: async () => + await inspectTerraformFenceProgress( + fenceConfig, + { + terraform: deps.terraform, + gcloudJson: deps.commandJson + }, + attempt + ), + abortAttempt: async () => + await adminPost(config.directorOrigin, '/v1/admin/cell-fence-attempt-abort', { + ...fenceAttemptBody(attempt), + confirmation: 'ABORT_UNSTARTED_TERRAFORM_CELL_FENCE' + }), + deletePlan: async (value) => + await deleteTerraformFencePlan( + fenceConfig, + { commandResult: deps.commandResult }, + value + ), + emit: deps.emit + }) +} + +async function fenceSource( + config, + deps, + adminPost, + source, + targets, + sourceProcess, + sourceStatus, + sourceAlreadyFenced +) { + if (sourceStatus.enabled) throw new Error('source fencing requires disabled admission') + const counts = sourceProcess + if ( + (!sourceAlreadyFenced && + (!counts || + sourceStatus.runtime?.observedRequests !== 0 || + counts.controls !== 0 || + counts.splices !== 0 || + counts.pendingSplices !== 0)) || + sourceStatus.activityLeases !== 0 || + sourceStatus.reservedRequests !== 0 + ) { + throw new Error('source fencing requires zero source-owned work') + } + const statuses = await allPairStatuses(config, adminPost, false) + const totals = statusTotals(statuses) + if ( + totals.inProgress !== sourceStatus.outgoingMigrations || + !hasDurableTargetOwnership(totals) + ) { + throw new Error('source fencing requires full migration coverage and durable target ownership') + } + for (const target of targets) await targetRuntime(config, adminPost, target) + const cellIncarnation = runtimeIncarnation(sourceStatus, source.cellId) + const postApplyGuard = async (expectedIncarnation) => { + const actualIncarnation = await waitForFence(config, deps, adminPost, source) + if (actualIncarnation !== expectedIncarnation) { + throw new Error('fenced source incarnation changed') + } + const finalSourceMig = sourceMig(config, deps, source) + if (finalSourceMig.instances.length !== 0) { + throw new Error('fenced source still has an instance') + } + await inspectFencedSource(config, deps, adminPost, source, finalSourceMig.mig) + } + await runTerraformManagedFence( + config, + deps, + adminPost, + source, + cellIncarnation, + sourceAlreadyFenced, + async () => { + const latestStatus = sourceAlreadyFenced + ? await inspectFencedSource( + config, + deps, + adminPost, + source, + sourceMig(config, deps, source).mig + ) + : { + ...await inspectDirectorObservedCell(config, deps, adminPost, source), + process: null + } + const latestCounts = sourceAlreadyFenced + ? null + : processCounts(config, deps, latestStatus, source.cellId) + if ( + latestStatus.enabled || + runtimeIncarnation(latestStatus, source.cellId) !== cellIncarnation || + (!sourceAlreadyFenced && + (latestStatus.runtime?.observedRequests !== 0 || + latestCounts.controls !== 0 || + latestCounts.splices !== 0 || + latestCounts.pendingSplices !== 0)) || + latestStatus.activityLeases !== 0 || + latestStatus.reservedRequests !== 0 + ) { + throw new Error('source fencing guards changed before Terraform apply') + } + const latestStatuses = await allPairStatuses(config, adminPost, false) + const latestTotals = statusTotals(latestStatuses) + if ( + latestTotals.inProgress !== latestStatus.outgoingMigrations || + !hasDurableTargetOwnership(latestTotals) + ) { + throw new Error('source migration coverage or guards changed before Terraform apply') + } + for (const target of targets) { + await inspectDirectorObservedCell(config, deps, adminPost, target) + } + }, + postApplyGuard + ) + await waitForMultiStatus(config, deps, adminPost, targets, true, false, true) + deps.emit({ event: 'source_fenced', sourceCellId: source.cellId, targetSize: 0 }) +} + +async function runTargetSupersession( + config, + deps, + adminPost, + source, + targets, + selector +) { + const failed = targets.find((target) => target.cellId === config.failedTargetCellId) + const replacement = targets.find( + (target) => target.cellId === config.replacementTargetCellId + ) + if (!failed || !replacement) throw new Error('supersession topology is incomplete') + const sourceStatus = await adminPost(config.directorOrigin, '/v1/admin/cell-status', { + v: 1, + cellId: source.cellId + }) + if ( + sourceStatus.status?.cellUrl !== source.origin || + (selector + ? selectorCellState(selector, source.cellId) !== 'existing-only' + : sourceStatus.status.enabled) + ) { + throw new Error('supersession requires retained disabled source topology') + } + const failedMig = sourceMig(config, deps, failed) + const failedStatus = await inspectFenceCandidate( + config, + deps, + adminPost, + failed, + failedMig.mig + ) + if ( + selector + ? selectorCellState(selector, failed.cellId) !== 'existing-only' + : failedStatus.enabled + ) { + throw new Error( + selector + ? 'failed target admission must be existing-only' + : 'failed target admission must be disabled' + ) + } + const replacementStatus = await inspectDirectorObservedCell( + config, + deps, + adminPost, + replacement + ) + const migrationStatus = await pairStatus(config, adminPost, failed.cellId, false) + if ( + !Number.isSafeInteger(migrationStatus.targetRegistered) || + migrationStatus.targetRegistered < 1 + ) { + throw new Error('failed target has no registered migrations to supersede') + } + const failedConnections = failedStatus.runtime?.observedRequests + if (!Number.isSafeInteger(failedConnections) || failedConnections < 0) { + throw new Error('failed target has no exact runtime connection snapshot') + } + const replacementConnections = runtimeConnections( + replacementStatus.process, + replacement.cellId + ) + const projectedConnections = + replacementConnections + + Math.max(failedConnections, migrationStatus.targetRegistered) + if (projectedConnections >= targetConnectionCeiling(config, replacement)) { + throw new Error('replacement target lacks conservative connection headroom') + } + const cellIncarnation = runtimeIncarnation(failedStatus, failed.cellId) + await runTerraformManagedFence( + config, + deps, + adminPost, + failed, + cellIncarnation, + Number(failedMig.mig.targetSize) === 0, + async () => { + const latestMig = sourceMig(config, deps, failed) + const latestStatus = await inspectFenceCandidate( + config, + deps, + adminPost, + failed, + latestMig.mig + ) + if ( + selector + ? selectorCellState(selector, failed.cellId) !== 'existing-only' + : latestStatus.enabled + ) { + throw new Error('failed target admission changed before apply') + } + if (runtimeIncarnation(latestStatus, failed.cellId) !== cellIncarnation) { + throw new Error('failed target incarnation changed before apply') + } + const latestMigration = await pairStatus(config, adminPost, failed.cellId, false) + if (latestMigration.targetRegistered < 1) { + throw new Error('failed target migrations changed before apply') + } + await inspectDirectorObservedCell(config, deps, adminPost, replacement) + }, + async (expectedIncarnation) => { + const actualIncarnation = await waitForFence(config, deps, adminPost, failed) + if (actualIncarnation !== expectedIncarnation) { + throw new Error('failed target incarnation changed') + } + const finalFailedMig = sourceMig(config, deps, failed) + if (finalFailedMig.instances.length !== 0) { + throw new Error('failed target fence still has an instance') + } + await inspectFencedSource(config, deps, adminPost, failed, finalFailedMig.mig) + } + ) + if ( + selector && + selectorCellState(selector, replacement.cellId) !== 'migration-only' + ) { + throw new Error('replacement target admission must be migration-only') + } + if (!selector && !replacementStatus.enabled) { + await setCellState(config, adminPost, replacement.cellId, true) + } + let superseded = 0 + while (true) { + const result = await adminPost( + config.directorOrigin, + '/v1/admin/migration-supersede-cell', + { + v: 1, + sourceCellId: source.cellId, + currentTargetCellId: failed.cellId, + replacementTargetCellId: replacement.cellId, + limit: config.batchSize, + confirmation: 'SUPERSEDE_REGISTERED_CELL_MIGRATIONS' + } + ) + if ( + !Number.isSafeInteger(result.superseded) || + result.superseded < 0 || + result.superseded > config.batchSize + ) { + throw new Error('invalid registered supersession result') + } + superseded += result.superseded + if (result.superseded === 0) break + } + const remaining = await pairStatus(config, adminPost, failed.cellId, false) + if (remaining.targetRegistered !== 0) { + throw new Error('registered target supersession did not reconcile') + } + if (superseded !== migrationStatus.targetRegistered) { + throw new Error('registered target supersession count changed') + } + deps.emit({ + event: 'registered_target_superseded', + sourceCellId: source.cellId, + failedTargetCellId: failed.cellId, + replacementTargetCellId: replacement.cellId, + superseded, + remainingUnregistered: remaining.inProgress + }) +} + +export async function runMultiTargetDeployment(config, overrides = {}) { + const deps = { + commandJson: overrides.commandJson ?? defaultCommandJson, + command: overrides.command ?? defaultCommand, + commandResult: overrides.commandResult ?? defaultCommandResult, + terraform: overrides.terraform, + terraformFenceApply: overrides.terraformFenceApply ?? runTerraformFenceApply, + terraformFenceAdopt: + overrides.terraformFenceAdopt ?? adoptLegacyTerraformFence, + terraformFenceResume: overrides.terraformFenceResume ?? resumeTerraformFence, + terraformFenceRecoverCompleted: + overrides.terraformFenceRecoverCompleted ?? + recoverSupersededCompletedTerraformFence, + terraformFenceAbort: overrides.terraformFenceAbort ?? abortTerraformFenceBeforeApply, + terraformFenceSupersede: + overrides.terraformFenceSupersede ?? abortSupersededTerraformFenceBeforeUpload, + identityToken: overrides.identityToken ?? defaultIdentityToken, + mutationIdentityToken: + overrides.mutationIdentityToken ?? + (() => suppliedFenceMutationIdentityToken()), + fetch: overrides.fetch ?? fetch, + emit: overrides.emit ?? ((event) => process.stdout.write(`${JSON.stringify(event)}\n`)), + now: overrides.now ?? Date.now, + wait: overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))), + random: overrides.random ?? Math.random, + resolve4: overrides.resolve4 ?? resolve4 + } + const topology = JSON.parse(readFileSync(config.topologyFile, 'utf8')) + const { source, targets } = selectMultiTargetDeployments( + topology, + config.sourceCellId, + config.targetCellIds + ) + const token = deps.identityToken(config.adminAudience) + const mutationToken = + ['fence-source', 'abort-fence-source', 'supersede-target'].includes(config.mode) + ? deps.mutationIdentityToken(config.adminAudience) + : null + if ( + ['fence-source', 'abort-fence-source', 'supersede-target'].includes(config.mode) && + mutationToken === null + ) { + throw new Error('Terraform fence mode requires a broker mutation identity token') + } + const adminPost = createAdminPost( + config, + deps, + (path) => (FENCE_BROKER_MUTATION_ROUTES.has(path) ? mutationToken : token) + ) + if (config.mode === 'abort-fence-source') { + await abortTerraformManagedFence(config, deps, adminPost, source) + return + } + const selectorPost = async (path, body) => + await adminPost(config.directorOrigin, path, body) + const selectorInspection = await inspectAdmissionSelector(selectorPost) + const selectorActive = selectorInspection.selector.generation > 0 + if (config.mode === 'cutover-admission') { + const membership = cutoverMembership(topology, config) + if ( + selectorActive && + JSON.stringify(selectorInspection.selector.membership) !== JSON.stringify(membership) + ) { + throw new Error('selector boundary is already active with different membership') + } + await inspectCutoverCells(topology, config, deps, adminPost, membership) + pruneIncompatibleDirectorRevisions(config, deps) + const director = verifySelectorCompatibleDirector(config, deps) + await inspectCutoverCells(topology, config, deps, adminPost, membership) + const result = await applyExactAdmissionSelector(selectorPost, membership, { + requireBoundary: false, + attemptId: config.selectorAttemptId, + expectedCurrentSelector: selectorInspection.selector + }) + deps.emit({ + event: 'admission_selector_cutover', + generation: result.selector.generation, + membership: result.selector.membership, + ...director + }) + return + } + if (config.mode === 'add-migration-cells') { + if (!selectorActive) throw new Error('admission selector boundary is not active') + pruneIncompatibleDirectorRevisions(config, deps) + const director = verifySelectorCompatibleDirector(config, deps) + if ( + targets.some( + (target) => + target.connectionHardCap === undefined || + target.connectionUnobservedBound === undefined + ) + ) { + throw new Error('migration cells require reviewed connection capacity') + } + const result = await addExactMigrationCells( + selectorPost, + { + attemptId: config.selectorAttemptId, + cells: targets.map((target) => ({ + cellId: target.cellId, + cellUrl: target.origin, + region: target.region, + capacityRequests: target.capacityRequests, + connectionHardCap: target.connectionHardCap, + connectionUnobservedBound: target.connectionUnobservedBound + })) + }, + { expectedCurrentSelector: selectorInspection.selector } + ) + deps.emit({ + event: 'migration_cells_added', + generation: result.selector.generation, + membership: result.selector.membership, + cellIds: targets.map(({ cellId }) => cellId), + ...director + }) + return + } + if (config.mode === 'promote-general-cell') { + if (!selectorActive) throw new Error('admission selector boundary is not active') + const [promoted] = targets + if ( + !promoted || + selectorCellState(selectorInspection.selector, promoted.cellId) !== 'migration-only' + ) { + throw new Error('promoted cell admission must be migration-only') + } + const director = verifyActiveSelectorDirector(config, deps, promoted.cellId) + await inspectGeneralPromotionTarget(config, deps, adminPost, promoted) + const result = await applyExactAdmissionSelector( + selectorPost, + membershipWithStates(selectorInspection.selector, { + [promoted.cellId]: 'general' + }), + { + attemptId: config.selectorAttemptId, + expectedCurrentSelector: selectorInspection.selector + } + ) + deps.emit({ + event: 'migration_cell_promoted_general', + generation: result.selector.generation, + membership: result.selector.membership, + cellId: promoted.cellId, + ...director + }) + return + } + if (config.mode === 'retire-migration-cell') { + if (!selectorActive) throw new Error('admission selector boundary is not active') + const [retiring] = targets + if ( + !retiring || + selectorCellState(selectorInspection.selector, retiring.cellId) !== 'migration-only' + ) { + throw new Error('retired cell admission must be migration-only') + } + pruneIncompatibleDirectorRevisions(config, deps) + const director = verifySelectorCompatibleDirector(config, deps) + const result = await applyExactAdmissionSelector( + selectorPost, + membershipWithStates(selectorInspection.selector, { + [retiring.cellId]: 'existing-only' + }), + { + attemptId: config.selectorAttemptId, + expectedCurrentSelector: selectorInspection.selector + } + ) + deps.emit({ + event: 'migration_cell_retired', + generation: result.selector.generation, + membership: result.selector.membership, + cellId: retiring.cellId, + ...director + }) + return + } + if (config.mode === 'supersede-target') { + await runTargetSupersession( + config, + deps, + adminPost, + source, + targets, + selectorActive ? selectorInspection.selector : null + ) + return + } + const sourceFence = config.mode === 'fence-source' ? sourceMig(config, deps, source) : null + if ( + sourceFence && + ![0, 1].includes(Number(sourceFence.mig.targetSize)) + ) { + throw new Error('source MIG must be fixed-one or already fenced') + } + const fencedResume = sourceFence && Number(sourceFence.mig.targetSize) === 0 ? sourceFence : null + const { sourceProcess, sourceStatus, plannedTargets, sourceAlreadyFenced } = await preflight( + config, + deps, + adminPost, + source, + targets, + fencedResume, + selectorActive ? selectorInspection.selector : null + ) + if (config.mode === 'audit' || config.mode === 'preflight') return + if (config.mode === 'fence-source') { + await fenceSource( + config, + deps, + adminPost, + source, + targets, + sourceProcess, + sourceStatus, + sourceAlreadyFenced + ) + return + } + if (config.mode === 'recover-forward') { + const invalidSelectorAdmission = + selectorActive && + (selectorCellState(selectorInspection.selector, source.cellId) !== 'existing-only' || + plannedTargets.some( + (target) => + selectorCellState(selectorInspection.selector, target.cellId) !== + 'migration-only' + )) + if ( + invalidSelectorAdmission || + (!selectorActive && + (sourceStatus.enabled || plannedTargets.some((target) => !target.status.enabled))) + ) { + throw new Error( + selectorActive + ? 'forward recovery requires existing-only source and migration-only targets' + : 'forward recovery requires disabled source and enabled targets' + ) + } + let coveredSourceConnections = runtimeConnections(sourceProcess, source.cellId) + let recoveryTargets = plannedTargets + for (let pass = 1; pass <= RECOVERY_CATCH_UP_PASSES; pass++) { + await publishMigrations(config, deps, adminPost, recoveryTargets) + if ( + await assertRecoveryPreDrain( + config, + deps, + adminPost, + source, + targets, + selectorPost, + selectorActive ? selectorInspection.selector : null + ) + ) { + break + } + if (pass === RECOVERY_CATCH_UP_PASSES) { + throw new Error('source assignments did not quiesce within bounded recovery catch-up') + } + const catchUp = await preflight( + config, + deps, + adminPost, + source, + targets, + null, + selectorActive ? selectorInspection.selector : null, + coveredSourceConnections + ) + coveredSourceConnections = Math.max( + coveredSourceConnections, + runtimeConnections(catchUp.sourceProcess, source.cellId) + ) + recoveryTargets = catchUp.plannedTargets + } + if (!sourceStatus.draining) { + const activeSourceTransports = + sourceProcess.controls + sourceProcess.splices + sourceProcess.pendingSplices + if (activeSourceTransports > 0) { + const recovery = await adminPost( + config.directorOrigin, + '/v1/admin/drain-attempt-recover-forward', + { + v: 1, + cellId: source.cellId, + cellIncarnation: runtimeIncarnation(sourceStatus, source.cellId), + confirmation: 'RECOVER_LEGACY_DRAIN' + } + ) + await waitForRecoveryTargetOwnership(config, deps, adminPost, targets) + if (recovery.preparedAttempt) { + const attempt = recovery.preparedAttempt + if ( + attempt.state !== 'prepared' || + attempt.cellId !== source.cellId || + attempt.cellIncarnation !== runtimeIncarnation(sourceStatus, source.cellId) || + attempt.plannedGraceMs !== 120_000 || + typeof attempt.attemptId !== 'string' || + typeof attempt.traceValue !== 'string' + ) { + throw new Error('prepared drain recovery state is invalid') + } + const sending = await adminPost( + config.directorOrigin, + '/v1/admin/drain-attempt-send', + { + v: 1, + attemptId: attempt.attemptId, + cellId: source.cellId, + cellIncarnation: attempt.cellIncarnation + } + ) + if ( + sending.attempt?.state !== 'send-may-have-started' || + sending.attempt.shouldSend !== true || + !Number.isSafeInteger(sending.attempt.sendPermitExpiresAt) || + deps.now() >= sending.attempt.sendPermitExpiresAt + ) { + throw new Error('prepared drain recovery send permit unavailable') + } + const receipt = await drainSource( + config, + deps, + token, + source, + attempt.plannedGraceMs, + attempt.traceValue + ) + await adminPost(config.directorOrigin, '/v1/admin/drain-attempt-receipt', { + v: 1, + attemptId: attempt.attemptId, + cellId: source.cellId, + cellIncarnation: attempt.cellIncarnation, + traceValue: attempt.traceValue, + ...receipt + }) + deps.emit({ + event: 'source_prepared_drain_recovered', + sourceCellId: source.cellId + }) + } else if (recovery.shouldSend === true) { + await drainSource(config, deps, token, source, 0) + deps.emit({ event: 'source_recovery_drain_accepted', sourceCellId: source.cellId }) + } else { + const latestSource = await inspectCell(config, deps, adminPost, source) + const expectedIncarnation = runtimeIncarnation(sourceStatus, source.cellId) + if (runtimeIncarnation(latestSource, source.cellId) !== expectedIncarnation) { + throw new Error('source incarnation changed before recovery drain reissue') + } + if (latestSource.draining) { + deps.emit({ + event: 'source_recovery_drain_already_applied', + sourceCellId: source.cellId + }) + } else { + const latestStatuses = await allPairStatuses(config, adminPost, false) + const latestTotals = statusTotals(latestStatuses) + assertLeaseGate(latestStatuses, config.minimumLeaseRemainingMs) + if ( + !hasDurableTargetOwnership(latestTotals) && + !boundedUnregisteredMigrations( + latestTotals, + config.unobservedConnectionBound + ) + ) { + throw new Error('recovery target ownership changed before drain reissue') + } + await drainSource(config, deps, token, source, 0) + deps.emit({ + event: 'source_recovery_drain_reissued_after_non_delivery', + sourceCellId: source.cellId, + ...latestTotals + }) + } + } + } else { + deps.emit({ + event: 'source_recovery_drain_not_needed', + sourceCellId: source.cellId + }) + } + } + await waitForMultiStatus(config, deps, adminPost, targets, false, true) + await waitForRecoveredSourceZero( + config, + deps, + adminPost, + source, + runtimeIncarnation(sourceStatus, source.cellId) + ) + await waitForMultiStatus(config, deps, adminPost, targets, true, true) + deps.emit({ event: 'multi_target_complete', sourceCellId: source.cellId }) + return + } + if ( + selectorActive && + targets.some( + (target) => + selectorCellState(selectorInspection.selector, target.cellId) !== 'migration-only' + ) + ) { + throw new Error('evacuation targets must be migration-only') + } + await runEvacuation( + config, + deps, + adminPost, + token, + source, + targets, + plannedTargets, + sourceStatus, + selectorActive, + selectorPost, + selectorInspection.selector + ) +} + +export async function main(argv = process.argv.slice(2)) { + await runMultiTargetDeployment(parseMultiTargetArguments(argv)) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/deploy-relay-gce-multi-target.test.mjs b/cloud/dev/scripts/deploy-relay-gce-multi-target.test.mjs new file mode 100644 index 00000000000..1ad4e240d56 --- /dev/null +++ b/cloud/dev/scripts/deploy-relay-gce-multi-target.test.mjs @@ -0,0 +1,3320 @@ +import assert from 'node:assert/strict' +import { mkdtempSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { test } from 'node:test' +import { + allocateTargetQuotas, + assertCutoverCellReady, + cutoverMembership, + parseMultiTargetArguments, + pruneIncompatibleDirectorRevisions, + runMultiTargetDeployment, + sameBackendServiceResource, + selectMultiTargetDeployments, + verifySelectorCompatibleDirector +} from './deploy-relay-gce-multi-target.mjs' + +const runtimeServiceAccount = 'orca-relay@example.iam.gserviceaccount.com' +const digest = (value) => `sha256:${value.repeat(64)}` + +test('matches only exact canonical backend-service resource forms', () => { + const resource = 'projects/onorca-cloud/global/backendServices/orca-cloud-relay-gce-c11' + assert.equal( + sameBackendServiceResource( + `https://www.googleapis.com/compute/v1/${resource}`, + resource + ), + true + ) + assert.equal( + sameBackendServiceResource( + `https://www.googleapis.com/compute/v1/${resource}`, + resource.replace('c11', 'c12') + ), + false + ) + assert.equal( + sameBackendServiceResource( + `https://compute.example/compute/v1/${resource}`, + resource + ), + false + ) +}) + +test('requires only compatible active and rollback director revisions', () => { + let rollbackInventory = ['c1'] + const revision = (minimum = 1, inventory = ['c1']) => ({ + metadata: { + annotations: { 'autoscaling.knative.dev/minScale': String(minimum) } + }, + spec: { + containers: [ + { + image: 'registry.example/relay@sha256:abc', + env: [ + { name: 'ORCA_RELAY_ROLE', value: 'director' }, + { name: 'ORCA_RELAY_ADMISSION_SELECTOR_VERSION', value: '3' }, + { + name: 'ORCA_RELAY_CELLS_JSON', + value: JSON.stringify(inventory.map((id) => ({ id }))) + } + ] + } + ] + } + }) + let names = ['active', 'rollback'] + const deps = { + command: (args) => { + const revision = args[3] + names = names.filter((name) => name !== revision) + }, + commandJson: (args) => { + if (args.includes('services')) { + return { + status: { + traffic: [ + { percent: 100, revisionName: 'active' }, + { tag: 'selector-rollback', revisionName: 'rollback' } + ] + } + } + } + if (args.includes('list')) { + return names.map((name) => ({ metadata: { name } })) + } + return args.includes('rollback') + ? revision(0, rollbackInventory) + : revision(1) + } + } + const config = { + project: 'project', + directorRegion: 'region', + directorService: 'service', + directorMinimumInstances: 1 + } + assert.deepEqual(verifySelectorCompatibleDirector(config, deps), { + activeRevision: 'active', + rollbackRevision: 'rollback' + }) + rollbackInventory = ['c1', 'c2'] + assert.throws( + () => verifySelectorCompatibleDirector(config, deps), + /director inventories do not match/ + ) + assert.throws( + () => pruneIncompatibleDirectorRevisions(config, deps), + /director compatibility pair failed/ + ) + rollbackInventory = ['c1'] + names = ['active', 'rollback', 'legacy'] + assert.throws( + () => verifySelectorCompatibleDirector(config, deps), + /old or pre-selector director revisions/ + ) + pruneIncompatibleDirectorRevisions(config, deps) + assert.deepEqual(names, ['active', 'rollback']) + assert.deepEqual(verifySelectorCompatibleDirector(config, deps), { + activeRevision: 'active', + rollbackRevision: 'rollback' + }) + names = ['active', 'rollback', null] + assert.throws( + () => verifySelectorCompatibleDirector(config, deps), + /old or pre-selector director revisions/ + ) + assert.throws( + () => pruneIncompatibleDirectorRevisions(config, deps), + /unnamed revision/ + ) +}) + +test('requires the active director to meet the configured floor', () => { + let activeMinimum = 5 + let rollbackMinimum = 0 + const revision = (minimum) => ({ + metadata: { + annotations: { 'autoscaling.knative.dev/minScale': String(minimum) } + }, + spec: { + containers: [ + { + image: 'registry.example/relay@sha256:abc', + env: [ + { name: 'ORCA_RELAY_ROLE', value: 'director' }, + { name: 'ORCA_RELAY_ADMISSION_SELECTOR_VERSION', value: '3' }, + { + name: 'ORCA_RELAY_CELLS_JSON', + value: JSON.stringify([{ id: 'c1' }]) + } + ] + } + ] + } + }) + const deps = { + command: () => {}, + commandJson: (args) => { + if (args.includes('services')) { + return { + status: { + traffic: [ + { percent: 100, revisionName: 'active' }, + { tag: 'selector-rollback', revisionName: 'rollback' } + ] + } + } + } + if (args.includes('list')) { + return ['active', 'rollback'].map((name) => ({ metadata: { name } })) + } + return args.includes('rollback') ? revision(rollbackMinimum) : revision(activeMinimum) + } + } + const config = { + project: 'project', + directorRegion: 'region', + directorService: 'service', + directorMinimumInstances: 5 + } + + assert.deepEqual(verifySelectorCompatibleDirector(config, deps), { + activeRevision: 'active', + rollbackRevision: 'rollback' + }) + pruneIncompatibleDirectorRevisions(config, deps) + + activeMinimum = 4 + assert.throws( + () => verifySelectorCompatibleDirector(config, deps), + /active selector revision is below the required floor/ + ) + assert.throws( + () => pruneIncompatibleDirectorRevisions(config, deps), + /director compatibility pair failed/ + ) + + // Every comparison against NaN is false, so negating the minimum check rejects it. + activeMinimum = 'warm' + assert.throws( + () => verifySelectorCompatibleDirector(config, deps), + /active selector revision is below the required floor/ + ) + assert.throws( + () => pruneIncompatibleDirectorRevisions(config, deps), + /director compatibility pair failed/ + ) + + activeMinimum = 5 + rollbackMinimum = 1 + assert.throws( + () => verifySelectorCompatibleDirector(config, deps), + /selector rollback revision is not scale-to-zero/ + ) + assert.throws( + () => pruneIncompatibleDirectorRevisions(config, deps), + /director compatibility pair failed/ + ) +}) + +function topology() { + const cell = (id, hostname, initiallyEnabled) => ({ + origin: `https://${hostname}.relay.example.com`, + zone: `us-central1-${hostname}`, + mig_name: `relay-${hostname}`, + instance_group: `https://compute.example/instanceGroups/relay-${hostname}`, + backend_name: `relay-${hostname}`, + backend_id: `https://compute.example/backendServices/relay-${hostname}`, + url_map_name: 'orca-relay', + generation_identity: `https://compute.example/instanceTemplates/relay-${hostname}-abc`, + image: `us-central1-docker.pkg.dev/project/repo/relay@${digest(id)}`, + capacity_requests: 4_000, + connection_hard_cap: 600, + connection_unobserved_bound: 40, + initially_enabled: initiallyEnabled, + fenced: false, + desired_target_size: 1, + target_size: 1 + }) + return { + source: cell('a', 'a', true), + target1: cell('b', 'b', false), + target2: cell('c', 'c', false), + general: cell('d', 'd', true) + } +} + +function withTopology(operation, value = topology()) { + const directory = mkdtempSync(join(tmpdir(), 'relay-gce-multi-')) + const file = join(directory, 'topology.json') + writeFileSync(file, JSON.stringify(value)) + return Promise.resolve(operation(file)).finally(() => rmSync(directory, { recursive: true })) +} + +function config(topologyFile, mode = 'preflight') { + const selectorMutation = [ + 'cutover-admission', + 'add-migration-cells', + 'promote-general-cell', + 'retire-migration-cell' + ].includes(mode) + return { + project: 'test-project', + directorOrigin: 'https://relay.example.com', + adminAudience: 'https://relay.example.com/v1/admin/drain', + topologyFile, + sourceCellId: 'source', + targetCellIds: ['promote-general-cell', 'retire-migration-cell'].includes(mode) + ? ['target1'] + : ['target1', 'target2'], + generalCellIds: mode === 'cutover-admission' ? ['general'] : [], + directorRegion: selectorMutation ? 'us-central1' : undefined, + directorService: selectorMutation ? 'relay-director' : undefined, + directorMinimumInstances: selectorMutation ? 1 : undefined, + selectorAttemptId: + mode === 'add-migration-cells' + ? 'add_cells_test' + : mode === 'promote-general-cell' + ? 'promote_cell_test' + : mode === 'retire-migration-cell' + ? 'retire_cell_test' + : undefined, + unobservedConnectionBound: + selectorMutation || mode === 'fence-source' ? 40 : undefined, + failedTargetCellId: mode === 'supersede-target' ? 'target1' : undefined, + replacementTargetCellId: mode === 'supersede-target' ? 'target2' : undefined, + runtimeServiceAccount, + environment: 'production', + fenceCommit: 'a'.repeat(40), + terraformDir: 'infra/terraform', + terraformVarFile: 'environments/production.tfvars', + mode, + batchSize: 2, + connectionCeiling: mode === 'fence-source' ? 600 : 6, + minimumLeaseRemainingMs: 600_000, + drainGraceMs: 120_000, + pollIntervalMs: 1, + timeoutMs: 1_000 + } +} + +function response(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' } + }) +} + +function harness({ + leaseRemainingMs = 900_000, + refreshedLeaseRemainingMs = leaseRemainingMs, + failAfterDrain = false, + failEvacuationStatus = false, + fence = false, + allowPreFenceCompletion = false, + loseResizeResponse = false, + resumeNeedsPreApply = false, + loseDrainSendResponse = false, + loseDrainBeforeAccept = false, + preparedDrainAttempt = false, + recoverableDrain = false, + recoveryAlreadyAttempted = false, + preexistingRegisteredMigrations = 0, + supersede = false, + cleanupBeforeSupersession = false, + offlineTargetMigrations = 0, + offlineTargetMigrationsByTarget = {}, + unregisteredTargetMigrations = 0, + registeredSourceActive = 0, + recoveryRegistrationDelayReads = 0, + failedTargetEnabled = false, + replacementHeartbeatFresh = true, + replacementRuntimeCellUrl, + legacySource = false, + legacyMetricAgeMs = 0, + selectorMembership = null, + capacitySourceAssignments = [], + capacityRequiredTargetUnits = [], + headroomFailureTarget = null, + sourceAssignments = fence ? 0 : 4, + sourceRequiredTargetUnits = fence ? 0 : 8, + sourceObservedRequests = null, + sourceOutgoingMigrations = null, + existingFenceAttempt = false, + alreadyFencedSource = false, + supersededFenceAttempt = false, + completedFenceAttempt = false, + activeDirectorMinimum = 1, + misroutedCell = null, + routeOverrideCell = null, + frontendMisbound = false, + templateHardCap = 600, + templateUnobservedBound = 40, + directorHardCap = templateHardCap, + directorUnobservedBound = templateUnobservedBound, + directorCapacityOverrides = {}, + liveMigTemplateOverrides = {}, + topologyValue = topology() +} = {}) { + const directorCapacity = (cellId) => { + const hardCap = directorCapacityOverrides[cellId]?.hardCap ?? directorHardCap + const unobservedBound = + directorCapacityOverrides[cellId]?.unobservedBound ?? directorUnobservedBound + return { + hardCap, + controlRebindReserve: 100, + ordinaryConnectionLimit: hardCap - 100, + unobservedBound, + normalAdmissionPause: hardCap - 100 - unobservedBound + } + } + const cells = { + source: { + enabled: !fence && !supersede, + assignments: 4, + activityLeases: fence ? 0 : 4, + totalConnections: fence ? 1 : 5, + controls: fence ? 0 : 4, + observedRequests: sourceObservedRequests ?? (fence ? 0 : 4), + heartbeatFresh: !alreadyFencedSource, + draining: fence + }, + target1: { + enabled: failedTargetEnabled || (fence && !supersede), + assignments: fence ? 2 : 0, + activityLeases: fence ? 2 : 0, + totalConnections: fence ? 2 : 0, + controls: fence ? 2 : 0, + heartbeatFresh: !supersede, + draining: false + }, + target2: { + enabled: fence && !supersede, + assignments: fence ? 2 : 0, + activityLeases: fence ? 2 : 0, + totalConnections: fence ? 2 : 0, + controls: fence ? 2 : 0, + heartbeatFresh: replacementHeartbeatFresh, + draining: false + }, + general: { + enabled: true, + assignments: 0, + activityLeases: 0, + totalConnections: 0, + controls: 0, + heartbeatFresh: true, + draining: false + } + } + for (const cellId of Object.keys(topologyValue)) { + cells[cellId] ??= { + enabled: fence && !supersede, + assignments: fence ? 2 : 0, + activityLeases: fence ? 2 : 0, + totalConnections: fence ? 2 : 0, + controls: fence ? 2 : 0, + heartbeatFresh: true, + draining: false + } + } + const events = [] + const stateChanges = [] + const batches = [] + const publishedMigrations = new Map() + const runtimeInspections = new Map() + let drained = fence + let remainingSourceAssignments = sourceAssignments + let remainingRequiredTargetUnits = sourceRequiredTargetUnits + const migSizes = Object.fromEntries( + Object.keys(topologyValue).map((cellId) => [ + cellId, + cellId === 'source' && alreadyFencedSource + ? 0 + : cellId === 'target1' && completedFenceAttempt + ? 0 + : 1 + ]) + ) + const instanceCounts = { ...migSizes } + let supersessionRemaining = supersede ? 2 : 0 + let remainingOfflineTargetMigrations = offlineTargetMigrations + const initialOfflineTargetMigrationsByTarget = new Map( + Object.entries(offlineTargetMigrationsByTarget) + ) + const remainingOfflineTargetMigrationsByTarget = new Map( + initialOfflineTargetMigrationsByTarget + ) + const targetMigrationTotal = (cellId) => + initialOfflineTargetMigrationsByTarget.has(cellId) + ? initialOfflineTargetMigrationsByTarget.get(cellId) + : 2 + const remainingOfflineTargetMigrationCount = (cellId) => + remainingOfflineTargetMigrationsByTarget.has(cellId) + ? remainingOfflineTargetMigrationsByTarget.get(cellId) + : remainingOfflineTargetMigrations + const clearOfflineTargetMigrations = (cellId) => { + if (remainingOfflineTargetMigrationsByTarget.has(cellId)) { + remainingOfflineTargetMigrationsByTarget.set(cellId, 0) + return + } + remainingOfflineTargetMigrations = 0 + } + let drainReceiptRecorded = recoverableDrain + let drainSendStarted = false + let recoveryDrainPrepared = recoveryAlreadyAttempted + let recoveryLeasesRefreshed = false + const recoveryRegistrationReads = new Map() + let headroomFailureInjected = false + let currentLeaseRemainingMs = leaseRemainingMs + const drainGraces = [] + const timeline = [] + let fencedCompletions = 0 + let resizeAttempts = 0 + let capacityReads = 0 + let addedCells = [] + let fenceAttested = false + let fenceAttempt = existingFenceAttempt || supersededFenceAttempt || completedFenceAttempt + ? { + attemptId: '44444444-4444-4444-8444-444444444444', + environment: 'production', + cellId: 'source', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + migName: topologyValue.source.mig_name, + instanceGroup: topologyValue.source.instance_group, + generationIdentity: topologyValue.source.generation_identity, + fenceCommit: (supersededFenceAttempt || completedFenceAttempt ? 'b' : 'a').repeat(40), + planSha256: 'd'.repeat(64), + planObjectName: + 'terraform/state/relay-fence-plans/production/44444444-4444-4444-8444-444444444444.tfplan', + ...(supersededFenceAttempt && !completedFenceAttempt + ? {} + : { planObjectGeneration: '123456789' }), + varFileSha256: 'e'.repeat(64), + terraformStateLineage: '55555555-5555-4555-8555-555555555555', + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: 'f'.repeat(64), + requestReason: + 'orca-relay-fence/44444444-4444-4444-8444-444444444444', + createdAt: Date.now(), + expiresAt: Date.now() + 3_600_000, + ...(completedFenceAttempt + ? { + applyStartedAt: 101, + applyInvocations: [ + { + invocationId: '66666666-6666-4666-8666-666666666666', + requestReason: + 'orca-relay-fence/44444444-4444-4444-8444-444444444444/66666666-6666-4666-8666-666666666666', + startedAt: 101 + } + ] + } + : {}) + } + : null + let selector = selectorMembership + ? { generation: 1, attemptId: 'existing-selector', membership: selectorMembership } + : null + const selectorIntents = new Map() + const cellForOrigin = (origin) => { + const entry = Object.entries(topologyValue).find(([, value]) => value.origin === origin) + return entry?.[0] + } + const cellForMig = (name) => { + const entry = Object.entries(topologyValue).find(([, value]) => value.mig_name === name) + return entry?.[0] + } + const commandJson = (args) => { + if (args[0] === 'run') { + if (args.includes('services')) { + return { + status: { + traffic: [ + { percent: 100, revisionName: 'active' }, + { percent: 0, tag: 'selector-rollback', revisionName: 'rollback' } + ] + } + } + } + if (args.includes('list')) { + return ['active', 'rollback'].map((name) => ({ metadata: { name } })) + } + const revisionName = args[3] + return { + metadata: { + annotations: { + 'autoscaling.knative.dev/minScale': + revisionName === 'rollback' ? '0' : String(activeDirectorMinimum) + } + }, + spec: { + containers: [ + { + image: 'registry.example/relay@sha256:abc', + env: [ + { name: 'ORCA_RELAY_ROLE', value: 'director' }, + { name: 'ORCA_RELAY_ADMISSION_SELECTOR_VERSION', value: '3' }, + { + name: 'ORCA_RELAY_CELLS_JSON', + value: JSON.stringify( + Object.keys(topologyValue).map((id) => ({ id })) + ) + } + ] + } + ] + } + } + } + if (args.includes('instance-templates')) { + const templateName = args[args.indexOf('describe') + 1] + const expected = Object.values(topologyValue).find((cell) => + cell.generation_identity.endsWith(`/instanceTemplates/${templateName}`) + ) + return { + selfLink: expected.generation_identity, + properties: { + metadata: { + items: [ + { + key: 'startup-script', + value: [ + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${expected.image.split('@')[1]}'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${templateHardCap}'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '${templateUnobservedBound}'`, + `docker pull '${expected.image}'`, + `docker run '${expected.image}'` + ].join('\n') + } + ] + } + } + } + } + if (args[0] === 'logging') { + return [ + { + timestamp: new Date(Date.now() - legacyMetricAgeMs).toISOString(), + jsonPayload: { + totalConnections: cells.source.totalConnections, + preAuthConnections: 0, + controls: cells.source.controls, + splices: 0, + pendingSplices: 0, + queuedBytes: 0 + } + } + ] + } + const describeIndex = args.indexOf('describe') + const listIndex = args.indexOf('list-instances') + const name = args[describeIndex >= 0 ? describeIndex + 1 : listIndex + 1] + const migCell = cellForMig(name) + if (args.includes('list-instances')) { + return instanceCounts[migCell] === 0 + ? [] + : [ + { + instance: `https://compute.example/instances/${name}-vm`, + instanceStatus: 'RUNNING', + currentAction: 'NONE', + version: { + name: 'primary', + instanceTemplate: topologyValue[migCell].generation_identity + } + } + ] + } + if (args.includes('instance-groups')) { + return { + targetSize: migSizes[migCell], + instanceTemplate: + liveMigTemplateOverrides[migCell] ?? topologyValue[migCell].generation_identity, + updatePolicy: { + replacementMethod: 'RECREATE', + maxSurge: { fixed: 0 }, + maxUnavailable: { fixed: 1 } + } + } + } + if (args.includes('instances')) { + return { + networkInterfaces: [{ networkIP: '10.42.0.2' }], + serviceAccounts: [{ email: runtimeServiceAccount }] + } + } + if (args.includes('target-https-proxies')) { + return { + name: 'orca-relay', + selfLink: + 'https://www.googleapis.com/compute/v1/projects/test-project/global/targetHttpsProxies/orca-relay', + urlMap: frontendMisbound + ? 'https://www.googleapis.com/compute/v1/projects/test-project/global/urlMaps/other' + : 'https://www.googleapis.com/compute/v1/projects/test-project/global/urlMaps/orca-relay' + } + } + if (args.includes('forwarding-rules')) { + return { + name: 'orca-relay', + target: + 'https://www.googleapis.com/compute/v1/projects/test-project/global/targetHttpsProxies/orca-relay', + IPAddress: '203.0.113.10', + portRange: '443-443', + loadBalancingScheme: 'EXTERNAL_MANAGED' + } + } + if (args.includes('addresses')) { + return { name: 'orca-relay', address: '203.0.113.10' } + } + if (args.includes('url-maps')) { + return { + name: 'orca-relay', + selfLink: + 'https://www.googleapis.com/compute/v1/projects/test-project/global/urlMaps/orca-relay', + hostRules: Object.entries(topologyValue).map(([cellId, cell]) => ({ + hosts: [new URL(cell.origin).hostname], + pathMatcher: cellId + })), + pathMatchers: Object.entries(topologyValue).map(([cellId, cell]) => ({ + name: cellId, + defaultService: + cellId === misroutedCell ? topologyValue.general.backend_id : cell.backend_id, + ...(cellId === routeOverrideCell + ? { + pathRules: [ + { paths: ['/v1/*'], service: topologyValue.general.backend_id } + ] + } + : {}) + })) + } + } + const id = Object.entries(topologyValue).find(([, value]) => value.backend_name === name)?.[0] + return { + protocol: 'HTTP', + timeoutSec: 86_400, + backends: [{ group: topologyValue[id].instance_group }] + } + } + const fetch = async (url, options = {}) => { + const parsed = new URL(url) + if (parsed.pathname === '/health' || parsed.pathname === '/ready') return response({ ok: true }) + const body = JSON.parse(options.body ?? '{}') + if (parsed.pathname === '/v1/admin/runtime-status') { + const id = cellForOrigin(parsed.origin) + const cell = cells[id] + const capacity = directorCapacity(id) + runtimeInspections.set(id, (runtimeInspections.get(id) ?? 0) + 1) + return response({ + v: 1, + role: 'cell', + cellId: id, + cellUrl: topologyValue[id].origin, + imageDigest: topologyValue[id].image.split('@')[1], + draining: cell.draining, + connectionCapacity: { + ...capacity + }, + runtime: + legacySource && id === 'source' + ? null + : { + totalConnections: cell.totalConnections, + preAuthConnections: 0, + enforcedConnectionUnits: cell.totalConnections, + controls: cell.controls, + splices: 0, + pendingSplices: 0, + queuedBytes: 0 + } + }) + } + if (parsed.pathname === '/v1/admin/cell-status') { + const cell = cells[body.cellId] + const capacity = directorCapacity(body.cellId) + return response({ + v: 1, + status: { + cellId: body.cellId, + cellUrl: topologyValue[body.cellId].origin, + enabled: cell.enabled, + assignments: cell.assignments, + activityLeases: cell.activityLeases, + activityRequestUnits: cell.activityLeases, + reservedRequests: cell.activityLeases, + connectionCapacity: { + ...capacity, + observedConnections: cell.totalConnections, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: cell.totalConnections, + pendingControlReservations: 0, + heartbeatFresh: cell.heartbeatFresh + }, + outgoingMigrations: + sourceOutgoingMigrations ?? + (fence + ? Object.keys(topologyValue) + .filter((cellId) => cellId !== 'source' && cellId !== 'general') + .reduce((total, cellId) => total + targetMigrationTotal(cellId), 0) + : 0), + incomingMigrations: 0, + runtime: { + cellUrl: + body.cellId === 'target2' && replacementRuntimeCellUrl + ? replacementRuntimeCellUrl + : topologyValue[body.cellId].origin, + cellIncarnation: '11111111-1111-4111-8111-111111111111', + ready: true, + heartbeatFresh: cell.heartbeatFresh, + observedRequests: cell.observedRequests ?? cell.controls + } + } + }) + } + if (parsed.pathname === '/v1/admin/admission-selector/status') { + selector ??= { + generation: 0, + attemptId: null, + membership: { + existingOnly: Object.keys(cells).filter((cellId) => !cells[cellId].enabled), + migrationOnly: [], + general: Object.keys(cells).filter((cellId) => cells[cellId].enabled) + } + } + return response({ + v: 1, + selector, + intent: body.attemptId ? selectorIntents.get(body.attemptId) ?? null : null + }) + } + if (parsed.pathname === '/v1/admin/admission-selector/apply') { + selector = { + generation: body.expectedGeneration + 1, + attemptId: body.attemptId, + membership: body.membership + } + selectorIntents.set(body.attemptId, { + attemptId: body.attemptId, + expectedGeneration: body.expectedGeneration, + intendedGeneration: selector.generation, + membership: body.membership, + state: 'committed' + }) + return response({ v: 1, changed: true, selector }) + } + if (parsed.pathname === '/v1/admin/admission-selector/add-migration-cells') { + addedCells = body.cells + selector = { + generation: body.expectedGeneration + 1, + attemptId: body.attemptId, + membership: { + ...selector.membership, + migrationOnly: [ + ...selector.membership.migrationOnly, + ...body.cells.map(({ cellId }) => cellId) + ].sort() + } + } + selectorIntents.set(body.attemptId, { + attemptId: body.attemptId, + expectedGeneration: body.expectedGeneration, + intendedGeneration: selector.generation, + membership: selector.membership, + state: 'committed' + }) + return response({ v: 1, changed: true, selector }) + } + if (parsed.pathname === '/v1/admin/evacuation-capacity') { + const read = capacityReads++ + return response({ + v: 1, + sourceAssignments: capacitySourceAssignments[read] ?? remainingSourceAssignments, + requiredTargetUnits: + capacityRequiredTargetUnits[read] ?? remainingRequiredTargetUnits, + availableTargetUnits: 4_000 + }) + } + if (parsed.pathname === '/v1/admin/cell-state') { + cells[body.cellId].enabled = body.enabled + stateChanges.push([body.cellId, body.enabled]) + return response({ ok: true }) + } + if (parsed.pathname === '/v1/admin/drain-attempt-prepare') { + return response({ v: 1, state: 'prepared' }) + } + if (parsed.pathname === '/v1/admin/drain-attempt-send') { + const shouldSend = !drainSendStarted + drainSendStarted = true + if (loseDrainSendResponse) throw new Error('injected_send_transition_response_loss') + return response({ + v: 1, + attempt: { + state: 'send-may-have-started', + shouldSend, + sendPermitExpiresAt: Date.now() + 30_000 + } + }) + } + if (parsed.pathname === '/v1/admin/drain-attempt-receipt') { + drainReceiptRecorded = true + return response({ v: 1, attempt: { state: 'application-receipt' } }) + } + if (parsed.pathname === '/v1/admin/drain-attempt-recover-forward') { + recoveryLeasesRefreshed = true + currentLeaseRemainingMs = refreshedLeaseRemainingMs + if (!drainReceiptRecorded) { + if (preparedDrainAttempt && !drainSendStarted) { + return response({ + v: 1, + shouldSend: false, + retryAfter: Date.now(), + preparedAttempt: { + attemptId: '33333333-3333-4333-8333-333333333333', + cellId: 'source', + cellIncarnation: '11111111-1111-4111-8111-111111111111', + traceValue: '44444444-4444-4444-8444-444444444444', + plannedGraceMs: 120_000, + state: 'prepared' + } + }) + } + return response({ error: 'drain_application_receipt_missing' }, 409) + } + const shouldSend = !recoveryDrainPrepared + recoveryDrainPrepared = true + return response({ v: 1, shouldSend, retryAfter: Date.now() }) + } + if (parsed.pathname === '/v1/admin/evacuate-cell') { + if (body.targetCellId === headroomFailureTarget && !headroomFailureInjected) { + headroomFailureInjected = true + const partiallyStarted = Math.min(1, remainingSourceAssignments) + publishedMigrations.set( + body.targetCellId, + (publishedMigrations.get(body.targetCellId) ?? 0) + partiallyStarted + ) + remainingSourceAssignments -= partiallyStarted + return response({ error: 'relay_connection_headroom_exhausted' }, 409) + } + const started = Math.min(body.limit, remainingSourceAssignments) + batches.push([body.targetCellId, started]) + publishedMigrations.set( + body.targetCellId, + (publishedMigrations.get(body.targetCellId) ?? 0) + started + ) + remainingSourceAssignments -= started + if (remainingSourceAssignments === 0) remainingRequiredTargetUnits = 0 + return response({ v: 1, started }) + } + if (parsed.pathname === '/v1/admin/cell-fence-attest') { + fenceAttested = true + return response({ v: 1, cellId: body.cellId, expiresAt: Date.now() + 300_000 }) + } + if (parsed.pathname === '/v1/admin/cell-fence-adopt-legacy') { + fenceAttested = true + return response({ v: 1, cellId: body.cellId, expiresAt: Date.now() + 300_000 }) + } + if (parsed.pathname === '/v1/admin/cell-fence-commit-legacy-adoption') { + return response({ v: 1, cellId: body.cellId, committed: true }) + } + if (parsed.pathname === '/v1/admin/cell-fence-attempt-prepare') { + fenceAttempt = body + return response({ v: 1, attempt: body }) + } + if (parsed.pathname === '/v1/admin/cell-fence-attempt-start') { + fenceAttempt = { ...body, applyStartedAt: Date.now() } + return response({ + v: 1, + attempt: fenceAttempt, + invocation: { + invocationId: body.invocationId, + requestReason: body.invocationRequestReason, + startedAt: fenceAttempt.applyStartedAt + } + }) + } + if (parsed.pathname === '/v1/admin/cell-fence-attempt-operation') { + fenceAttempt = body + return response({ + v: 1, + attempt: body, + invocation: { + invocationId: body.invocationId, + requestReason: body.invocationRequestReason, + startedAt: Date.now(), + gceOperation: body.gceOperation + } + }) + } + if (parsed.pathname === '/v1/admin/cell-fence-attempt-status') { + return response({ v: 1, attempt: fenceAttempt }) + } + if (parsed.pathname === '/v1/admin/cell-fence-attempt-abort') { + fenceAttempt = { ...body, abortedAt: Date.now() } + return response({ v: 1, attempt: fenceAttempt }) + } + if (parsed.pathname === '/v1/admin/evacuation-status') { + if (failEvacuationStatus) { + return response({ error: 'injected_status_failure' }, 500) + } + const completed = + body.completeReady && (!fence || fenceAttested || allowPreFenceCompletion) + if (completed && fenceAttested) clearOfflineTargetMigrations(body.targetCellId) + const migrationTotal = targetMigrationTotal(body.targetCellId) + const remainingOfflineTargetMigrationsForCell = + remainingOfflineTargetMigrationCount(body.targetCellId) + const supersededTarget = supersede && body.targetCellId === 'target1' + const recoveryRegistrationRead = + recoveryRegistrationReads.get(body.targetCellId) ?? 0 + if (recoveryLeasesRefreshed) { + recoveryRegistrationReads.set(body.targetCellId, recoveryRegistrationRead + 1) + } + const recoveryRegistrationSettled = + recoveryRegistrationRead >= recoveryRegistrationDelayReads + const remainingMigrations = supersededTarget + ? supersessionRemaining + : completed + ? remainingOfflineTargetMigrationsForCell + unregisteredTargetMigrations + : migrationTotal + const registeredMigrations = supersededTarget + ? supersessionRemaining + : drained + ? remainingMigrations - unregisteredTargetMigrations + : Math.min( + remainingMigrations, + preexistingRegisteredMigrations + + (recoveryLeasesRefreshed && recoveryRegistrationSettled + ? publishedMigrations.get(body.targetCellId) ?? 0 + : 0) + ) + const completableMigrations = + drained && !completed + ? migrationTotal - + remainingOfflineTargetMigrationsForCell - + unregisteredTargetMigrations - + registeredSourceActive + : recoveryLeasesRefreshed + ? Math.max( + 0, + registeredMigrations - + remainingOfflineTargetMigrationsForCell - + registeredSourceActive + ) + : 0 + if (completed) { + if (failAfterDrain) return response({ error: 'injected_completion_failure' }, 500) + if (fenceAttested) fencedCompletions++ + cells.source.activityLeases = 0 + for (const targetCellId of Object.keys(cells).filter((id) => id !== 'source')) { + cells[targetCellId].activityLeases = 2 + } + } + timeline.push({ + kind: 'evacuation_status', + targetCellId: body.targetCellId, + completeReady: body.completeReady, + inProgress: remainingMigrations, + registeredTargetInactive: remainingOfflineTargetMigrationsForCell + }) + return response({ + v: 1, + inProgress: remainingMigrations, + oldestExpiresAt: + remainingMigrations === 0 ? null : Date.now() + currentLeaseRemainingMs, + oldestRemainingMs: + remainingMigrations === 0 ? null : currentLeaseRemainingMs, + targetRegistered: registeredMigrations, + registeredSourceActive, + registeredCompletable: completableMigrations, + registeredTargetInactive: remainingOfflineTargetMigrationsForCell, + completed: + completed + ? migrationTotal - + remainingOfflineTargetMigrationsForCell - + unregisteredTargetMigrations + : 0, + blocked: completed ? remainingOfflineTargetMigrationsForCell : 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + }) + } + if (parsed.pathname === '/v1/admin/migration-supersede-cell') { + const superseded = cleanupBeforeSupersession ? 0 : supersessionRemaining + supersessionRemaining = 0 + return response({ v: 1, superseded }) + } + if (parsed.pathname === '/v1/admin/drain') { + drainGraces.push(body.graceMs) + if (loseDrainBeforeAccept && body.graceMs === 120_000) { + throw new Error('injected_drain_response_loss') + } + drained = true + cells.source.draining = true + cells.source.activityLeases = 0 + cells.source.controls = 0 + for (const targetCellId of Object.keys(cells).filter((id) => id !== 'source')) { + cells[targetCellId].totalConnections = 2 + } + return response({ ok: true }) + } + return response({ error: 'unexpected_request' }, 500) + } + return { + overrides: { + commandJson, + terraform: (args) => { + if (args.includes('console')) return 'true\n' + if (args.includes('plan')) return '' + if (args.includes('show') && args.length > 3) { + return JSON.stringify({ resource_changes: [] }) + } + if (args.includes('show')) { + return JSON.stringify({ + values: { + root_module: { + resources: [ + { + address: + 'google_compute_instance_group_manager.relay_gce_cell["source"]', + values: { + name: topologyValue.source.mig_name, + zone: topologyValue.source.zone, + instance_group: topologyValue.source.instance_group, + target_size: migSizes.source, + version: [ + { instance_template: topologyValue.source.generation_identity } + ] + } + } + ] + } + } + }) + } + throw new Error(`unexpected terraform command: ${args.join(' ')}`) + }, + command: (args) => { + throw new Error(`unexpected direct gcloud mutation: ${args.join(' ')}`) + }, + terraformFenceApply: async (fenceConfig, callbacks) => { + const attempt = { + attemptId: '44444444-4444-4444-8444-444444444444', + environment: fenceConfig.environment, + cellId: fenceConfig.cell.cellId, + cellIncarnation: fenceConfig.cellIncarnation, + migName: fenceConfig.cell.migName, + instanceGroup: fenceConfig.cell.instanceGroup, + generationIdentity: fenceConfig.cell.generationIdentity, + fenceCommit: fenceConfig.fenceCommit, + planSha256: 'd'.repeat(64), + planObjectName: + `terraform/state/relay-fence-plans/${fenceConfig.environment}/44444444-4444-4444-8444-444444444444.tfplan`, + planObjectGeneration: '123456789', + varFileSha256: 'e'.repeat(64), + terraformStateLineage: '55555555-5555-4555-8555-555555555555', + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: 'f'.repeat(64), + requestReason: + 'orca-relay-fence/44444444-4444-4444-8444-444444444444' + } + await callbacks.prepareAttempt(attempt) + await callbacks.preApplyGuard() + const invocation = { + invocationId: '66666666-6666-4666-8666-666666666666', + requestReason: + `${attempt.requestReason}/66666666-6666-4666-8666-666666666666` + } + await callbacks.markApplyStarted(attempt, invocation) + const cellId = fenceConfig.cell.cellId + migSizes[cellId] = 0 + instanceCounts[cellId] = 0 + cells[cellId].heartbeatFresh = false + resizeAttempts++ + if (loseResizeResponse && resizeAttempts === 1) { + throw new Error('injected_resize_response_loss') + } + const completed = { ...attempt, gceOperation: 'operation-1' } + await callbacks.markOperation(completed, invocation) + await callbacks.postApplyGuard(attempt.cellIncarnation) + await callbacks.attest(completed) + }, + terraformFenceAdopt: async (fenceConfig, callbacks) => { + assert.equal(await callbacks.loadAttempt(), null) + await callbacks.assertCommittedFenceSet() + await callbacks.preApplyGuard() + await callbacks.assertStateFenced() + await callbacks.postApplyGuard(fenceConfig.cellIncarnation) + assert.equal(await callbacks.loadAttempt(), null) + await callbacks.attest(fenceConfig.cellIncarnation) + await callbacks.postApplyGuard(fenceConfig.cellIncarnation) + await callbacks.commitAdoption(fenceConfig.cellIncarnation) + callbacks.emit({ + event: 'terraform_cell_fence_legacy_adopted', + cellId: fenceConfig.cell.cellId + }) + }, + terraformFenceResume: async (fenceConfig, callbacks) => { + const attempt = await callbacks.loadAttempt() + if (resumeNeedsPreApply) await callbacks.preApplyGuard() + const completed = { + ...attempt, + cellIncarnation: fenceConfig.cellIncarnation, + gceOperation: 'operation-1' + } + await callbacks.postApplyGuard(completed.cellIncarnation) + await callbacks.attest(completed) + }, + terraformFenceRecoverCompleted: async ( + fenceConfig, + callbacks, + recovery + ) => { + const attempt = await callbacks.loadAttempt() + callbacks.emit({ + event: 'terraform_cell_fence_completed_attempt_recovered', + cellId: fenceConfig.cell.cellId, + attemptId: attempt.attemptId, + gceOperation: recovery.gceOperation + }) + }, + terraformFenceAbort: async (fenceConfig, callbacks) => { + await callbacks.abortAttempt() + callbacks.emit({ + event: 'terraform_fence_aborted_before_apply', + cellId: fenceConfig.cell.cellId + }) + }, + terraformFenceSupersede: async (fenceConfig, callbacks) => { + const attempt = await callbacks.loadAttempt() + await callbacks.abortAttempt(attempt) + callbacks.emit({ + event: 'terraform_fence_superseded_before_upload', + cellId: fenceConfig.cell.cellId, + previousFenceCommit: attempt.fenceCommit, + fenceCommit: fenceConfig.fenceCommit + }) + }, + identityToken: () => 'aaa.bbb.ccc', + mutationIdentityToken: () => 'ddd.eee.fff', + resolve4: async () => ['203.0.113.10'], + fetch, + emit: (event) => { + events.push(event) + timeline.push({ kind: 'event', event }) + }, + wait: async () => undefined, + random: () => 0 + }, + batches, + events, + stateChanges, + drainGraces, + timeline, + fencedCompletions: () => fencedCompletions, + capacityReads: () => capacityReads, + selector: () => selector, + addedCells: () => addedCells, + runtimeInspections + } +} + +test('parses deterministic target sets with single-cell selector exceptions', () => { + const common = [ + '--project', + 'project', + '--director-origin', + 'https://relay.example.com', + '--admin-audience', + 'https://relay.example.com/v1/admin/drain', + '--topology-file', + 'topology.json', + '--source-cell-id', + 'source', + '--runtime-service-account', + runtimeServiceAccount, + '--mode', + 'preflight' + ] + assert.deepEqual( + parseMultiTargetArguments([...common, '--target-cell-ids', 'target2,target1']).targetCellIds, + ['target1', 'target2'] + ) + assert.throws( + () => parseMultiTargetArguments([...common, '--target-cell-ids', 'target1']), + /at least 2/ + ) + const cutover = [ + ...common.slice(0, -2), + '--mode', + 'cutover-admission', + '--target-cell-ids', + 'target1,target2', + '--general-cell-ids', + 'general', + '--director-region', + 'us-central1', + '--director-service', + 'relay-director', + '--director-min-instances', + '5' + ] + assert.throws( + () => parseMultiTargetArguments(cutover), + /unobserved-connection-bound/ + ) + assert.equal( + parseMultiTargetArguments([ + ...cutover, + '--unobserved-connection-bound', + '40' + ]).unobservedConnectionBound, + 40 + ) + const recovery = [ + ...common.slice(0, -2), + '--mode', + 'recover-forward', + '--target-cell-ids', + 'target1,target2' + ] + assert.throws( + () => parseMultiTargetArguments(recovery), + /unobserved-connection-bound/ + ) + assert.equal( + parseMultiTargetArguments([ + ...recovery, + '--unobserved-connection-bound', + '60' + ]).unobservedConnectionBound, + 60 + ) + const fenceSource = [ + ...common.slice(0, -2), + '--mode', + 'fence-source', + '--target-cell-ids', + 'target1,target2', + '--fence-commit', + 'a'.repeat(40) + ] + assert.throws( + () => parseMultiTargetArguments(fenceSource), + /unobserved-connection-bound/ + ) + assert.equal( + parseMultiTargetArguments([ + ...fenceSource, + '--unobserved-connection-bound', + '60' + ]).unobservedConnectionBound, + 60 + ) + const addCells = [ + ...common.slice(0, -2), + '--mode', + 'add-migration-cells', + '--target-cell-ids', + 'target1,target2', + '--director-region', + 'us-central1', + '--director-service', + 'relay-director', + '--director-min-instances', + '5', + '--unobserved-connection-bound', + '40' + ] + assert.throws( + () => parseMultiTargetArguments(addCells), + /selector-attempt-id/ + ) + assert.equal( + parseMultiTargetArguments([ + ...addCells, + '--target-cell-ids', + 'target1', + '--selector-attempt-id', + 'add_cells_parse' + ]).targetCellIds.length, + 1 + ) + const retireCell = [ + ...common.slice(0, -2), + '--mode', + 'retire-migration-cell', + '--target-cell-ids', + 'target1', + '--director-region', + 'us-central1', + '--director-service', + 'relay-director', + '--director-min-instances', + '5', + '--selector-attempt-id', + 'retire_cell_parse' + ] + assert.equal( + parseMultiTargetArguments(retireCell).mode, + 'retire-migration-cell' + ) + assert.throws( + () => + parseMultiTargetArguments([ + ...retireCell, + '--target-cell-ids', + 'target1,target2' + ]), + /exactly one/ + ) + const promoteCell = [ + ...common.slice(0, -2), + '--mode', + 'promote-general-cell', + '--target-cell-ids', + 'target1', + '--director-region', + 'us-central1', + '--director-service', + 'relay-director', + '--director-min-instances', + '5', + '--selector-attempt-id', + 'promote_cell_parse' + ] + assert.equal( + parseMultiTargetArguments(promoteCell).mode, + 'promote-general-cell' + ) + assert.throws( + () => + parseMultiTargetArguments([ + ...promoteCell, + '--target-cell-ids', + 'target1,target2' + ]), + /exactly one/ + ) +}) + +test('requires a complete exact completed-fence recovery pin set', () => { + const args = [ + '--project', + 'project', + '--director-origin', + 'https://relay.example.com', + '--admin-audience', + 'https://relay.example.com/v1/admin/drain', + '--topology-file', + 'topology.json', + '--source-cell-id', + 'source', + '--target-cell-ids', + 'target1,target2', + '--runtime-service-account', + runtimeServiceAccount, + '--mode', + 'supersede-target', + '--failed-target-cell-id', + 'target1', + '--replacement-target-cell-id', + 'target2', + '--fence-commit', + 'a'.repeat(40), + '--completed-fence-attempt-id', + '44444444-4444-4444-8444-444444444444', + '--completed-fence-commit', + 'b'.repeat(40), + '--completed-fence-operation', + 'operation-1', + '--completed-fence-state-serial', + '61', + '--completed-fence-plan-generation', + '123', + '--completed-fence-state-generation', + '456', + '--completed-fence-state-sha256', + 'c'.repeat(64), + '--fence-broker-service-account', + 'fence-broker@example.gserviceaccount.com' + ] + assert.equal( + parseMultiTargetArguments(args).completedFenceRecovery + .terraformStateSerial, + 61 + ) + assert.throws( + () => parseMultiTargetArguments(args.slice(0, -2)), + /recovery inputs are invalid/ + ) +}) + +test('requires the cutover source to remain existing-only', () => { + assert.throws( + () => + cutoverMembership(topology(), { + sourceCellId: 'source', + targetCellIds: ['target1'], + generalCellIds: ['source', 'target2'] + }), + /source must remain existing-only/ + ) +}) + +test('requires exact cutover connection-capacity evidence and live headroom', () => { + const status = { + draining: false, + connectionCapacity: { + hardCap: 600, + controlRebindReserve: 100, + ordinaryConnectionLimit: 500, + unobservedBound: 40, + normalAdmissionPause: 460, + pendingControlReservations: 20, + heartbeatFresh: true + }, + runtimeConnectionCapacity: { + hardCap: 600, + controlRebindReserve: 100, + ordinaryConnectionLimit: 500, + unobservedBound: 40, + normalAdmissionPause: 460 + }, + process: { + enforcedConnectionUnits: 400, + preAuthConnections: 3 + } + } + assert.doesNotThrow(() => assertCutoverCellReady('target1', status, 40)) + assert.throws( + () => + assertCutoverCellReady( + 'target1', + { + ...status, + runtimeConnectionCapacity: { + ...status.runtimeConnectionCapacity, + unobservedBound: 39 + } + }, + 40 + ), + /reviewed connection-capacity policy/ + ) + assert.throws( + () => + assertCutoverCellReady( + 'target1', + { + ...status, + connectionCapacity: { + ...status.connectionCapacity, + pendingControlReservations: 60 + } + }, + 40 + ), + /normal-admission connection headroom/ + ) + assert.throws( + () => + assertCutoverCellReady( + 'target1', + { + ...status, + process: { ...status.process, preAuthConnections: 45 } + }, + 40 + ), + /pre-auth connection headroom/ + ) +}) + +test('cuts over only after every proposed cell passes exact readiness evidence', async () => { + await withTopology(async (file) => { + const testHarness = harness() + await runMultiTargetDeployment( + config(file, 'cutover-admission'), + testHarness.overrides + ) + assert.deepEqual(testHarness.selector(), { + generation: 1, + attemptId: testHarness.selector().attemptId, + membership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + assert.equal(testHarness.events.at(-1).event, 'admission_selector_cutover') + assert.deepEqual(Object.fromEntries(testHarness.runtimeInspections), { + target1: 2, + target2: 2, + general: 2 + }) + }) +}) + +test('rejects a different active selector before readiness checks or revision mutation', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source', 'target2'], + migrationOnly: ['target1'], + general: ['general'] + } + }) + await assert.rejects( + runMultiTargetDeployment( + config(file, 'cutover-admission'), + testHarness.overrides + ), + /already active with different membership/ + ) + assert.equal(testHarness.runtimeInspections.size, 0) + }) +}) + +test('registers new Terraform targets as one selector generation', async () => { + await withTopology(async (file) => { + const reviewed = topology() + reviewed.target1.connection_hard_cap = 1_000 + reviewed.target1.connection_unobserved_bound = 60 + writeFileSync(file, JSON.stringify(reviewed)) + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source'], + migrationOnly: [], + general: ['general'] + }, + activeDirectorMinimum: 5 + }) + await runMultiTargetDeployment( + { ...config(file, 'add-migration-cells'), directorMinimumInstances: 5 }, + testHarness.overrides + ) + assert.deepEqual(testHarness.selector(), { + generation: 2, + attemptId: 'add_cells_test', + membership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + assert.equal(testHarness.events.at(-1).event, 'migration_cells_added') + assert.deepEqual(testHarness.addedCells(), [ + { + cellId: 'target1', + cellUrl: topology().target1.origin, + capacityRequests: 4_000, + region: 'us-central1', + connectionHardCap: 1_000, + connectionUnobservedBound: 60 + }, + { + cellId: 'target2', + cellUrl: topology().target2.origin, + capacityRequests: 4_000, + region: 'us-central1', + connectionHardCap: 600, + connectionUnobservedBound: 40 + } + ]) + assert.equal(testHarness.runtimeInspections.size, 0) + }) +}) + +test('promotes exactly one healthy migration cell to general admission', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'promote-general-cell'), + testHarness.overrides + ) + assert.deepEqual(testHarness.selector(), { + generation: 2, + attemptId: 'promote_cell_test', + membership: { + existingOnly: ['source'], + migrationOnly: ['target2'], + general: ['general', 'target1'] + } + }) + assert.equal(testHarness.events.at(-1).event, 'migration_cell_promoted_general') + }) +}) + +test('rejects promotion below the configured director floor', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await assert.rejects( + runMultiTargetDeployment( + { ...config(file, 'promote-general-cell'), directorMinimumInstances: 2 }, + testHarness.overrides + ), + /active director is not compatible/ + ) + assert.equal(testHarness.events.length, 0) + }) +}) + +test('rejects general promotion unless the exact cell is migration-only', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target2'], + general: ['general', 'target1'] + } + }) + await assert.rejects( + runMultiTargetDeployment( + config(file, 'promote-general-cell'), + testHarness.overrides + ), + /must be migration-only/ + ) + assert.equal(testHarness.events.length, 0) + }) +}) + +test('retires exactly one migration cell without changing other membership', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'retire-migration-cell'), + testHarness.overrides + ) + assert.deepEqual(testHarness.selector(), { + generation: 2, + attemptId: 'retire_cell_test', + membership: { + existingOnly: ['source', 'target1'], + migrationOnly: ['target2'], + general: ['general'] + } + }) + assert.equal(testHarness.events.at(-1).event, 'migration_cell_retired') + assert.equal(testHarness.runtimeInspections.size, 0) + }) +}) + +test('rejects retirement unless the exact cell is migration-only', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target2'], + general: ['general', 'target1'] + } + }) + await assert.rejects( + runMultiTargetDeployment( + config(file, 'retire-migration-cell'), + testHarness.overrides + ), + /must be migration-only/ + ) + assert.equal(testHarness.events.length, 0) + }) +}) + +test('rejects overlapping generations and allocates below the connection ceiling', () => { + const selected = selectMultiTargetDeployments(topology(), 'source', ['target1', 'target2']) + assert.equal(selected.targets.length, 2) + const planned = allocateTargetQuotas({ + sourceAssignments: 4, + sourceConnections: 5, + requiredTargetUnits: 8, + connectionCeiling: 6, + targets: [ + { + cellId: 'target1', + currentConnections: 0, + availableConnectionReservations: 10, + availableTargetUnits: 10 + }, + { + cellId: 'target2', + currentConnections: 0, + availableConnectionReservations: 10, + availableTargetUnits: 10 + } + ] + }) + assert.deepEqual( + planned.map(({ cellId, quota, projectedConnections }) => ({ + cellId, + quota, + projectedConnections + })), + [ + { cellId: 'target1', quota: 2, projectedConnections: 3 }, + { cellId: 'target2', quota: 2, projectedConnections: 3 } + ] + ) + assert.throws(() => + allocateTargetQuotas({ + sourceAssignments: 4, + sourceConnections: 20, + requiredTargetUnits: 8, + connectionCeiling: 4, + targets: [ + { + cellId: 'target1', + currentConnections: 0, + availableConnectionReservations: 10, + availableTargetUnits: 10 + }, + { + cellId: 'target2', + currentConnections: 0, + availableConnectionReservations: 10, + availableTargetUnits: 10 + } + ] + }) + ) + assert.deepEqual( + allocateTargetQuotas({ + sourceAssignments: 1, + sourceConnections: 700, + requiredTargetUnits: 1, + connectionCeiling: 1_000, + targets: [ + { + cellId: 'target-600', + currentConnections: 0, + connectionCeiling: 600, + availableConnectionReservations: 1, + availableTargetUnits: 1 + }, + { + cellId: 'target-1000', + currentConnections: 0, + connectionCeiling: 1_000, + availableConnectionReservations: 1, + availableTargetUnits: 1 + } + ] + }).map(({ cellId, projectedConnections }) => ({ cellId, projectedConnections })), + [ + { cellId: 'target-600', projectedConnections: 699 }, + { cellId: 'target-1000', projectedConnections: 700 } + ] + ) +}) + +test('assumes every unbound source connection can land on one target', () => { + const targets = [ + { + cellId: 'target1', + currentConnections: 0, + availableConnectionReservations: 10, + availableTargetUnits: 10 + }, + { + cellId: 'target2', + currentConnections: 0, + availableConnectionReservations: 10, + availableTargetUnits: 10 + } + ] + const planned = allocateTargetQuotas({ + sourceAssignments: 4, + sourceConnections: 9, + requiredTargetUnits: 8, + connectionCeiling: 8, + targets + }) + assert.deepEqual( + planned.map(({ quota, projectedConnections }) => ({ quota, projectedConnections })), + [ + { quota: 2, projectedConnections: 7 }, + { quota: 2, projectedConnections: 7 } + ] + ) + assert.throws(() => + allocateTargetQuotas({ + sourceAssignments: 4, + sourceConnections: 9, + requiredTargetUnits: 8, + connectionCeiling: 7, + targets + }) + ) +}) + +test('serializes deterministic target quotas and completes after drain acceptance', async () => { + await withTopology(async (file) => { + const testHarness = harness() + await runMultiTargetDeployment(config(file, 'execute'), testHarness.overrides) + assert.deepEqual(testHarness.batches, [ + ['target1', 2], + ['target2', 2] + ]) + assert.deepEqual(testHarness.stateChanges.slice(0, 3), [ + ['source', false], + ['target1', true], + ['target2', true] + ]) + assert.equal(testHarness.events.some((event) => event.event === 'source_drain_accepted'), true) + assert.equal(testHarness.events.at(-1).event, 'multi_target_complete') + }) +}) + +test('uses fresh aggregate logs for a legacy source without runtime counts', async () => { + await withTopology(async (file) => { + const testHarness = harness({ legacySource: true }) + await runMultiTargetDeployment(config(file), testHarness.overrides) + assert.equal(testHarness.events.at(-1).sourceConnections, 5) + }) +}) + +test('rejects stale legacy source telemetry', async () => { + await withTopology(async (file) => { + const testHarness = harness({ legacySource: true, legacyMetricAgeMs: 90_001 }) + await assert.rejects( + runMultiTargetDeployment(config(file), testHarness.overrides), + /runtime metrics are stale/ + ) + }) +}) + +test('refuses a drain with an expiring lease and restores pre-drain admission', async () => { + await withTopology(async (file) => { + const testHarness = harness({ leaseRemainingMs: 599_999 }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'execute'), testHarness.overrides), + /insufficient time/ + ) + assert.deepEqual(testHarness.stateChanges.slice(-3), [ + ['source', true], + ['target1', false], + ['target2', false] + ]) + assert.equal( + testHarness.events.some((event) => event.event === 'source_drain_accepted'), + false + ) + }) +}) + +test('preserves forward recovery when target registration status is unavailable', async () => { + await withTopology(async (file) => { + const testHarness = harness({ failEvacuationStatus: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'execute'), testHarness.overrides), + /cannot prove zero target registrations/ + ) + assert.equal( + testHarness.stateChanges.some(([cellId, enabled]) => cellId === 'source' && enabled), + false + ) + assert.equal( + testHarness.stateChanges.some( + ([cellId, enabled]) => cellId.startsWith('target') && !enabled + ), + false + ) + assert.deepEqual(testHarness.events.at(-1), { + event: 'multi_forward_recovery_required', + targetRegistered: null, + reason: 'registration_status_unavailable' + }) + }) +}) + +test('audits partial migration state without requiring new migration headroom', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + capacitySourceAssignments: [1_000], + capacityRequiredTargetUnits: [2_000] + }) + await runMultiTargetDeployment(config(file, 'audit'), testHarness.overrides) + assert.equal(testHarness.capacityReads(), 0) + assert.deepEqual(testHarness.events.at(-1), { + event: 'multi_target_audit', + inProgress: 4, + targetRegistered: 0, + registeredSourceActive: 0, + registeredCompletable: 0, + registeredTargetInactive: 0, + completed: 0, + blocked: 0, + expiredUnregistered: 0, + repairableExpiredUnregistered: 0, + abortableExpiredUnregistered: 0, + blockedExpiredUnregistered: 0, + blockedExpiredOnNewerTargetAssignment: 0 + }) + }) +}) + +test('never restores source admission after drain acceptance', async () => { + await withTopology(async (file) => { + const testHarness = harness({ failAfterDrain: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'execute'), testHarness.overrides), + /injected_completion_failure/ + ) + assert.equal(testHarness.events.some((event) => event.event === 'source_drain_accepted'), true) + assert.equal( + testHarness.stateChanges.some(([cellId, enabled]) => cellId === 'source' && enabled), + false + ) + assert.equal(testHarness.events.at(-1).event, 'multi_forward_recovery_required') + }) +}) + +test('never restores source admission after the send transition becomes ambiguous', async () => { + await withTopology(async (file) => { + const testHarness = harness({ loseDrainSendResponse: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'execute'), testHarness.overrides), + /injected_send_transition_response_loss/ + ) + assert.equal( + testHarness.stateChanges.some(([cellId, enabled]) => cellId === 'source' && enabled), + false + ) + assert.equal(testHarness.events.at(-1).event, 'multi_forward_recovery_required') + }) +}) + +test('freezes forward recovery when a legacy drain response is ambiguous', async () => { + await withTopology(async (file) => { + const testHarness = harness({ loseDrainBeforeAccept: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'execute'), testHarness.overrides), + /injected_drain_response_loss/ + ) + assert.equal( + testHarness.stateChanges.some(([cellId, enabled]) => cellId === 'source' && enabled), + false + ) + await assert.rejects( + runMultiTargetDeployment(config(file, 'recover-forward'), testHarness.overrides), + /drain_application_receipt_missing/ + ) + assert.deepEqual(testHarness.drainGraces, [120_000]) + }) +}) + +test('resumes an exact prepared drain without allocating new migrations', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + preparedDrainAttempt: true, + preexistingRegisteredMigrations: 2, + sourceAssignments: 0, + sourceRequiredTargetUnits: 0, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, [120_000]) + assert.deepEqual(testHarness.batches, []) + assert.equal(testHarness.capacityReads(), 8) + assert.equal( + testHarness.events.some( + (event) => event.event === 'source_prepared_drain_recovered' + ), + true + ) + assert.deepEqual(testHarness.events.at(-1), { + event: 'multi_target_complete', + sourceCellId: 'source' + }) + }) +}) + +test('registers only remaining assignments before recovering a replacement drain', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + preexistingRegisteredMigrations: 1, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.deepEqual(testHarness.batches, [ + ['target1', 2], + ['target2', 2] + ]) + assert.equal(testHarness.capacityReads(), 8) + assert.deepEqual(testHarness.drainGraces, [0]) + const eventNames = testHarness.events.map((event) => event.event) + assert.equal( + testHarness.events.find( + (event) => event.event === 'multi_forward_recovery_preflight' + ).targetRegistered, + 2 + ) + assert.ok( + eventNames.lastIndexOf('multi_migration_batch') < + eventNames.indexOf('multi_forward_recovery_ready_to_drain') + ) + assert.ok( + eventNames.indexOf('multi_forward_recovery_ready_to_drain') < + eventNames.indexOf('source_recovery_drain_accepted') + ) + }) +}) + +test('refreshes registered migration leases before the recovery drain gate', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + preexistingRegisteredMigrations: 2, + sourceAssignments: 0, + sourceRequiredTargetUnits: 0, + leaseRemainingMs: 599_999, + refreshedLeaseRemainingMs: 900_000, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('waits for catch-up migrations to gain durable target ownership', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + recoveryRegistrationDelayReads: 2, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + const ownershipEvents = testHarness.events.filter( + (event) => event.event === 'multi_recovery_target_ownership' + ) + assert.equal(ownershipEvents.length, 3) + assert.equal(ownershipEvents[0].inProgress > ownershipEvents[0].targetRegistered, true) + assert.equal(ownershipEvents.at(-1).inProgress, ownershipEvents.at(-1).targetRegistered) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('times out before drain when catch-up migrations remain unregistered', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + recoveryRegistrationDelayReads: Number.POSITIVE_INFINITY, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + let now = 0 + testHarness.overrides.now = () => now + testHarness.overrides.wait = async (ms) => { + now += ms + } + await assert.rejects( + runMultiTargetDeployment( + { ...config(file, 'recover-forward'), timeoutMs: 3 }, + testHarness.overrides + ), + /timed out waiting for recovery target ownership/ + ) + assert.deepEqual(testHarness.drainGraces, []) + }) +}) + +test('drains a fully unregistered recovery within the reviewed bound', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + recoveryRegistrationDelayReads: Number.POSITIVE_INFINITY, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + { + ...config(file, 'recover-forward'), + unobservedConnectionBound: 4 + }, + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, [0]) + assert.equal( + testHarness.events.some( + (event) => event.event === 'multi_recovery_bounded_unregistered' + ), + true + ) + }) +}) + +test('drains mixed target registrations within the unobserved bound', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + preexistingRegisteredMigrations: 1, + recoveryRegistrationDelayReads: Number.POSITIVE_INFINITY, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + { + ...config(file, 'recover-forward'), + unobservedConnectionBound: 2 + }, + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, [0]) + const event = testHarness.events.find( + (entry) => entry.event === 'multi_recovery_bounded_mixed_registration' + ) + assert.equal(event.unregistered, 2) + assert.equal(event.targetRegistered, 2) + }) +}) + +test('reissues a proven non-delivered recovery drain and settles bounded offline clients', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + recoveryAlreadyAttempted: true, + preexistingRegisteredMigrations: 1, + unregisteredTargetMigrations: 1, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + { + ...config(file, 'recover-forward'), + unobservedConnectionBound: 2 + }, + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, [0]) + assert.equal( + testHarness.events.some( + (event) => event.event === 'source_recovery_drain_reissued_after_non_delivery' + ), + true + ) + assert.equal( + testHarness.events.some( + (event) => event.event === 'multi_migration_bounded_offline_complete' + ), + true + ) + }) +}) + +test('rejects mixed target registrations above the unobserved bound', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + preexistingRegisteredMigrations: 1, + recoveryRegistrationDelayReads: Number.POSITIVE_INFINITY, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + let now = 0 + testHarness.overrides.now = () => now + testHarness.overrides.wait = async (ms) => { + now += ms + } + await assert.rejects( + runMultiTargetDeployment( + { + ...config(file, 'recover-forward'), + timeoutMs: 3, + unobservedConnectionBound: 1 + }, + testHarness.overrides + ), + /timed out waiting for recovery target ownership/ + ) + assert.deepEqual(testHarness.drainGraces, []) + }) +}) + +test('times out before drain while a target registration remains source-active', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + preexistingRegisteredMigrations: 1, + registeredSourceActive: 1, + recoveryRegistrationDelayReads: Number.POSITIVE_INFINITY, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + let now = 0 + testHarness.overrides.now = () => now + testHarness.overrides.wait = async (ms) => { + now += ms + } + await assert.rejects( + runMultiTargetDeployment( + { + ...config(file, 'recover-forward'), + timeoutMs: 3, + unobservedConnectionBound: 2 + }, + testHarness.overrides + ), + /timed out waiting for recovery target ownership/ + ) + assert.deepEqual(testHarness.drainGraces, []) + }) +}) + +test('allows recovery drain with durable offline target registrations', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + offlineTargetMigrations: 1, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('catches up source assignments that arrive during recovery publication', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + sourceAssignments: 5, + sourceRequiredTargetUnits: 10, + capacitySourceAssignments: [4, 4, 4, 4, 1, 1, 1, 1], + capacityRequiredTargetUnits: [8, 8, 8, 8, 2, 2, 2, 2], + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.equal( + testHarness.batches.reduce((total, [, limit]) => total + limit, 0), + 5 + ) + assert.equal( + testHarness.events.some( + (event) => event.event === 'multi_forward_recovery_catch_up' + ), + true + ) + assert.deepEqual( + testHarness.events + .filter((event) => event.event === 'multi_target_preflight') + .map((event) => ({ + sourceConnections: event.sourceConnections, + observedSourceConnections: event.observedSourceConnections + })), + [ + { sourceConnections: 5, observedSourceConnections: 5 }, + { sourceConnections: 1, observedSourceConnections: 5 } + ] + ) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('replans recover-forward after transactional target headroom rejection', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + headroomFailureTarget: 'target1', + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.equal( + testHarness.events.some( + (event) => + event.event === 'multi_recovery_target_headroom_paused' && + event.targetCellId === 'target1' + ), + true + ) + assert.equal( + testHarness.events.some( + (event) => event.event === 'multi_forward_recovery_catch_up' + ), + true + ) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('uses conservative changing capacity samples during recovery', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + sourceAssignments: 5, + sourceRequiredTargetUnits: 10, + capacitySourceAssignments: [5, 4, 5, 4], + capacityRequiredTargetUnits: [10, 8, 10, 8], + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.equal( + testHarness.batches.reduce((total, [, limit]) => total + limit, 0), + 5 + ) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('requires every final recovery sample to prove zero source assignments', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + preexistingRegisteredMigrations: 2, + sourceAssignments: 1, + sourceRequiredTargetUnits: 2, + capacitySourceAssignments: [1, 1, 1, 1, 0, 0, 0, 1], + capacityRequiredTargetUnits: [2, 2, 2, 2, 0, 0, 0, 2], + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.equal( + testHarness.events.some( + (event) => event.event === 'multi_forward_recovery_catch_up' + ), + true + ) + assert.deepEqual(testHarness.drainGraces, [0]) + }) +}) + +test('stops after bounded recovery catch-up cannot quiesce the source', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + recoverableDrain: true, + sourceAssignments: 10, + sourceRequiredTargetUnits: 20, + capacitySourceAssignments: Array.from({ length: 40 }, () => 1), + capacityRequiredTargetUnits: Array.from({ length: 40 }, () => 2), + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await assert.rejects( + runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ), + /did not quiesce within bounded recovery catch-up/ + ) + assert.equal( + testHarness.batches.reduce((total, [, limit]) => total + limit, 0), + 5 + ) + assert.deepEqual(testHarness.drainGraces, []) + }) +}) + +test('settles a drained source while registered target users remain offline', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + allowPreFenceCompletion: true, + offlineTargetMigrations: 1, + selectorMembership: { + existingOnly: ['source'], + migrationOnly: ['target1', 'target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'recover-forward'), + testHarness.overrides + ) + assert.deepEqual(testHarness.drainGraces, []) + assert.deepEqual(testHarness.events.at(-1), { + event: 'multi_target_complete', + sourceCellId: 'source' + }) + }) +}) + +test('fences a quiescent disabled source and completes through stale-source checks', async () => { + await withTopology(async (file) => { + const testHarness = harness({ fence: true }) + const authorizationByOrigin = new Map() + const fetch = testHarness.overrides.fetch + testHarness.overrides.fetch = async (url, options) => { + const parsed = new URL(url) + if (parsed.pathname.startsWith('/v1/admin/')) { + authorizationByOrigin.set( + parsed.origin, + new Set([ + ...(authorizationByOrigin.get(parsed.origin) ?? []), + options?.headers?.authorization + ]) + ) + } + return await fetch(url, options) + } + await runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + assert.deepEqual( + authorizationByOrigin.get('https://relay.example.com'), + new Set(['Bearer aaa.bbb.ccc', 'Bearer ddd.eee.fff']) + ) + assert.equal(authorizationByOrigin.has('https://a.relay.example.com'), false) + assert.equal(authorizationByOrigin.has('https://b.relay.example.com'), false) + assert.equal(authorizationByOrigin.has('https://c.relay.example.com'), false) + assert.deepEqual(testHarness.events.at(-1), { + event: 'source_fenced', + sourceCellId: 'source', + targetSize: 0 + }) + }) +}) + +test('adopts an already-fenced source only through the legacy no-op path', async () => { + await withTopology(async (file) => { + const testHarness = harness({ fence: true, alreadyFencedSource: true }) + await runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + assert.deepEqual( + testHarness.events.find( + ({ event }) => event === 'terraform_cell_fence_legacy_adopted' + ), + { event: 'terraform_cell_fence_legacy_adopted', cellId: 'source' } + ) + assert.deepEqual(testHarness.events.at(-1), { + event: 'source_fenced', + sourceCellId: 'source', + targetSize: 0 + }) + }) +}) + +test('refuses to attest a fence when selected targets omit an outgoing migration', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + alreadyFencedSource: true, + sourceOutgoingMigrations: 5 + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /full migration coverage/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }) +}) + +test('clears the complete 156-row legacy backlog before reporting the source fenced', async () => { + const rollout = topology() + const additionalTarget = (id, hostname) => ({ + ...rollout.target2, + origin: `https://${hostname}.relay.example.com`, + zone: `us-central1-${hostname}`, + mig_name: `relay-${hostname}`, + instance_group: `https://compute.example/instanceGroups/relay-${hostname}`, + backend_name: `relay-${hostname}`, + backend_id: `https://compute.example/backendServices/relay-${hostname}`, + generation_identity: `https://compute.example/instanceTemplates/relay-${hostname}-abc`, + image: `us-central1-docker.pkg.dev/project/repo/relay@${digest(id)}` + }) + rollout.target3 = additionalTarget('e', 'e') + rollout.target4 = additionalTarget('f', 'f') + rollout.target5 = additionalTarget('1', 'g') + rollout.target6 = additionalTarget('2', 'h') + const migrationCounts = { + target1: 58, + target2: 63, + target3: 24, + target4: 10, + target5: 1, + target6: 0 + } + + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + alreadyFencedSource: true, + offlineTargetMigrationsByTarget: migrationCounts, + topologyValue: rollout + }) + const fenceConfig = config(file, 'fence-source') + fenceConfig.targetCellIds = Object.keys(migrationCounts) + + await runMultiTargetDeployment(fenceConfig, testHarness.overrides) + + const sourceFencedIndex = testHarness.timeline.findIndex( + ({ kind, event }) => kind === 'event' && event.event === 'source_fenced' + ) + assert.notEqual(sourceFencedIndex, -1) + let observedInitialTotal = 0 + for (const [targetCellId, initialCount] of Object.entries(migrationCounts)) { + const initialIndex = testHarness.timeline.findIndex( + (entry) => + entry.kind === 'evacuation_status' && + entry.targetCellId === targetCellId && + entry.inProgress === initialCount && + entry.registeredTargetInactive === initialCount + ) + const clearedIndex = testHarness.timeline.findIndex( + (entry) => + entry.kind === 'evacuation_status' && + entry.targetCellId === targetCellId && + entry.inProgress === 0 + ) + assert.notEqual(initialIndex, -1) + observedInitialTotal += testHarness.timeline[initialIndex].inProgress + if (initialCount > 0) assert.ok(clearedIndex > initialIndex) + else assert.ok(clearedIndex >= initialIndex) + assert.ok(clearedIndex < sourceFencedIndex) + } + assert.equal(observedInitialTotal, 156) + assert.equal(testHarness.fencedCompletions(), 6) + }, rollout) +}) + +test('refuses legacy adoption when the live MIG template changed', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + alreadyFencedSource: true, + liveMigTemplateOverrides: { + source: `${topology().source.generation_identity}-other` + } + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /source MIG fence topology is unsafe/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }) +}) + +test('fences with reviewed 600 templates during the declared 1000 rollout', async () => { + const rollout = topology() + for (const [cellId, cell] of Object.entries(rollout)) { + cell.connection_unobserved_bound = 60 + if (['target1', 'target2'].includes(cellId)) cell.connection_hard_cap = 1_000 + } + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + templateUnobservedBound: 60, + directorUnobservedBound: 60 + }) + const fenceConfig = config(file, 'fence-source') + fenceConfig.unobservedConnectionBound = 60 + await runMultiTargetDeployment(fenceConfig, testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + }, rollout) +}) + +test('fences exact configured capacity when the saved Terraform output is legacy', async () => { + const rollout = topology() + for (const cell of Object.values(rollout)) { + delete cell.connection_hard_cap + delete cell.connection_unobserved_bound + } + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + templateUnobservedBound: 60, + directorUnobservedBound: 60 + }) + const fenceConfig = config(file, 'fence-source') + fenceConfig.unobservedConnectionBound = 60 + await runMultiTargetDeployment(fenceConfig, testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + }, rollout) +}) + +test('refuses legacy Terraform output when live capacity differs from broker config', async () => { + const rollout = topology() + for (const cell of Object.values(rollout)) { + delete cell.connection_hard_cap + delete cell.connection_unobserved_bound + } + await withTopology(async (file) => { + const testHarness = harness({ fence: true }) + const fenceConfig = config(file, 'fence-source') + fenceConfig.unobservedConnectionBound = 60 + await assert.rejects( + runMultiTargetDeployment(fenceConfig, testHarness.overrides), + /instance template capacity differs from Terraform/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }, rollout) +}) + +test('refuses explicit null capacity as a legacy Terraform output', async () => { + const rollout = topology() + for (const cell of Object.values(rollout)) { + cell.connection_hard_cap = null + cell.connection_unobserved_bound = null + } + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + templateUnobservedBound: 60, + directorUnobservedBound: 60 + }) + const fenceConfig = config(file, 'fence-source') + fenceConfig.unobservedConnectionBound = 60 + await assert.rejects( + runMultiTargetDeployment(fenceConfig, testHarness.overrides), + /instance template capacity differs from Terraform/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }, rollout) +}) + +test('refuses a mixed legacy and declared capacity topology', async () => { + const rollout = topology() + for (const cell of Object.values(rollout)) { + delete cell.connection_hard_cap + delete cell.connection_unobserved_bound + } + rollout.general.connection_hard_cap = null + rollout.general.connection_unobserved_bound = null + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + templateUnobservedBound: 60, + directorUnobservedBound: 60 + }) + const fenceConfig = config(file, 'fence-source') + fenceConfig.unobservedConnectionBound = 60 + await assert.rejects( + runMultiTargetDeployment(fenceConfig, testHarness.overrides), + /instance template capacity differs from Terraform/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }, rollout) +}) + +test('fences after the active template reaches the declared 1000 capacity', async () => { + const rollout = topology() + for (const cellId of Object.keys(rollout)) { + rollout[cellId].connection_hard_cap = 1_000 + } + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + templateHardCap: 1_000, + directorHardCap: 1_000 + }) + await runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + }, rollout) +}) + +test('refuses a rollout predecessor whose template has the wrong bound', async () => { + const rollout = topology() + for (const cellId of ['target1', 'target2']) { + rollout[cellId].connection_hard_cap = 1_000 + rollout[cellId].connection_unobserved_bound = 60 + } + await withTopology(async (file) => { + const testHarness = harness({ fence: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /instance template capacity is outside reviewed rollout/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }, rollout) +}) + +test('refuses a rollout predecessor when the director reports another cap', async () => { + const rollout = topology() + for (const cellId of ['target1', 'target2']) { + rollout[cellId].connection_hard_cap = 1_000 + } + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + directorCapacityOverrides: { + target1: { hardCap: 1_000, unobservedBound: 40 } + } + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /connection capacity differs from Terraform/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }, rollout) +}) + +test('preserves direct target runtime reads outside fence mode', async () => { + await withTopology(async (file) => { + const testHarness = harness({}) + const fetch = testHarness.overrides.fetch + const authorizationByOrigin = new Map() + testHarness.overrides.fetch = async (url, options) => { + const parsed = new URL(url) + if (parsed.pathname.startsWith('/v1/admin/')) { + authorizationByOrigin.set( + parsed.origin, + new Set([ + ...(authorizationByOrigin.get(parsed.origin) ?? []), + options?.headers?.authorization + ]) + ) + } + return await fetch(url, options) + } + + await runMultiTargetDeployment(config(file, 'preflight'), testHarness.overrides) + assert.deepEqual( + authorizationByOrigin.get('https://relay.example.com'), + new Set(['Bearer aaa.bbb.ccc']) + ) + assert.deepEqual( + authorizationByOrigin.get('https://b.relay.example.com'), + new Set(['Bearer aaa.bbb.ccc']) + ) + assert.deepEqual( + authorizationByOrigin.get('https://c.relay.example.com'), + new Set(['Bearer aaa.bbb.ccc']) + ) + }) +}) + +test('refuses to fence when a cell hostname routes to the wrong backend', async () => { + await withTopology(async (file) => { + const testHarness = harness({ fence: true, misroutedCell: 'target1' }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /retained route topology mismatch/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }) +}) + +test('refuses to fence when a cell route overrides the reviewed backend', async () => { + await withTopology(async (file) => { + const testHarness = harness({ fence: true, routeOverrideCell: 'target1' }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /retained route topology mismatch/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }) +}) + +test('refuses to fence when the live HTTPS frontend uses another URL map', async () => { + await withTopology(async (file) => { + const testHarness = harness({ fence: true, frontendMisbound: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /live frontend topology mismatch/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }) +}) + +test('refuses to fence when the fresh source heartbeat reports active work', async () => { + await withTopology(async (file) => { + const testHarness = harness({ fence: true, sourceObservedRequests: 1 }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /source fencing requires zero source-owned work/ + ) + assert.equal(testHarness.fencedCompletions(), 0) + }) +}) + +test('fences a quiescent source while registered target users remain offline', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + alreadyFencedSource: true, + offlineTargetMigrations: 1 + }) + await runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + assert.deepEqual(testHarness.events.at(-1), { + event: 'source_fenced', + sourceCellId: 'source', + targetSize: 0 + }) + }) +}) + +test('refuses to fence source-active or unregistered migrations', async () => { + for (const migrationState of [ + { registeredSourceActive: 1 }, + { unregisteredTargetMigrations: 1 } + ]) { + await withTopology(async (file) => { + const testHarness = harness({ fence: true, ...migrationState }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /full migration coverage and durable target ownership/ + ) + }) + } +}) + +test('resumes fenced completion after losing the resize response', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + fence: true, + loseResizeResponse: true, + resumeNeedsPreApply: true + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides), + /injected_resize_response_loss/ + ) + await runMultiTargetDeployment(config(file, 'fence-source'), testHarness.overrides) + assert.equal(testHarness.fencedCompletions(), 2) + assert.deepEqual(testHarness.events.at(-1), { + event: 'source_fenced', + sourceCellId: 'source', + targetSize: 0 + }) + }) +}) + +test('records a proven pre-apply Terraform fence abort', async () => { + await withTopology(async (file) => { + const testHarness = harness({ existingFenceAttempt: true }) + await runMultiTargetDeployment( + config(file, 'abort-fence-source'), + testHarness.overrides + ) + assert.deepEqual(testHarness.events.at(-1), { + event: 'terraform_fence_aborted_before_apply', + cellId: 'source' + }) + }) +}) + +test('fences a failed registered target before aggregate supersession', async () => { + await withTopology(async (file) => { + const testHarness = harness({ supersede: true }) + await runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ) + assert.equal( + testHarness.stateChanges.some( + ([cellId, enabled]) => cellId === 'target2' && enabled + ), + true + ) + assert.deepEqual(testHarness.events.at(-1), { + event: 'registered_target_superseded', + sourceCellId: 'source', + failedTargetCellId: 'target1', + replacementTargetCellId: 'target2', + superseded: 2, + remainingUnregistered: 0 + }) + assert.equal(testHarness.runtimeInspections.get('target2') ?? 0, 0) + }) +}) + +test('fails closed when cleanup wins before aggregate supersession selects rows', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + cleanupBeforeSupersession: true + }) + + await assert.rejects( + runMultiTargetDeployment(config(file, 'supersede-target'), testHarness.overrides), + /registered target supersession count changed/ + ) + assert.deepEqual(testHarness.events, []) + }) +}) + +test('does not accept a capacity predecessor for target supersession', async () => { + const rollout = topology() + rollout.target2.connection_hard_cap = 1_000 + await withTopology(async (file) => { + const testHarness = harness({ supersede: true }) + await assert.rejects( + runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ), + /instance template capacity is outside reviewed rollout/ + ) + }, rollout) +}) + +test('does not accept capacity-bearing templates from legacy output for supersession', async () => { + const rollout = topology() + for (const cell of Object.values(rollout)) { + delete cell.connection_hard_cap + delete cell.connection_unobserved_bound + } + await withTopology(async (file) => { + const testHarness = harness({ supersede: true }) + await assert.rejects( + runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ), + /instance template capacity differs from Terraform/ + ) + }, rollout) +}) + +test('replaces a superseded unuploaded fence attempt before target fencing', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + supersededFenceAttempt: true + }) + await runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ) + assert.deepEqual(testHarness.events[0], { + event: 'terraform_fence_superseded_before_upload', + cellId: 'target1', + previousFenceCommit: 'b'.repeat(40), + fenceCommit: 'a'.repeat(40) + }) + assert.equal(testHarness.events.at(-1).event, 'registered_target_superseded') + }) +}) + +test('recovers a pinned completed older fence before registered supersession', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + completedFenceAttempt: true + }) + const deploymentConfig = config(file, 'supersede-target') + deploymentConfig.completedFenceRecovery = { + attemptId: '44444444-4444-4444-8444-444444444444', + fenceCommit: 'b'.repeat(40), + gceOperation: 'operation-1', + terraformStateSerial: 7, + planObjectGeneration: '123456789', + terraformStateObjectGeneration: '222222222', + terraformStateObjectSha256: 'c'.repeat(64), + principalEmail: 'fence-broker@example.gserviceaccount.com' + } + + await runMultiTargetDeployment(deploymentConfig, testHarness.overrides) + + assert.equal( + testHarness.events[0].event, + 'terraform_cell_fence_completed_attempt_recovered' + ) + assert.equal(testHarness.events.at(-1).event, 'registered_target_superseded') + }) +}) + +test('fails supersession on stale replacement heartbeat without calling its admin API', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + replacementHeartbeatFresh: false + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'supersede-target'), testHarness.overrides), + /no fresh matching director runtime snapshot/ + ) + assert.equal(testHarness.runtimeInspections.get('target2') ?? 0, 0) + assert.deepEqual(testHarness.events, []) + }) +}) + +test('fails supersession on mismatched director runtime without calling cell admin', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + replacementRuntimeCellUrl: 'https://wrong.relay.example.com' + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'supersede-target'), testHarness.overrides), + /no fresh matching director runtime snapshot/ + ) + assert.equal(testHarness.runtimeInspections.get('target2') ?? 0, 0) + assert.deepEqual(testHarness.events, []) + }) +}) + +test('reads the broker mutation token without treating the audience as the environment', async () => { + await withTopology(async (file) => { + const testHarness = harness({ supersede: true }) + delete testHarness.overrides.mutationIdentityToken + const previous = process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN + process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN = 'ddd.eee.fff' + try { + await runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ) + } finally { + if (previous === undefined) { + delete process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN + } else { + process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN = previous + } + } + assert.equal(testHarness.events.at(-1).event, 'registered_target_superseded') + }) +}) + +test('fails closed when the broker child has no mutation token', async () => { + await withTopology(async (file) => { + const testHarness = harness({ supersede: true }) + delete testHarness.overrides.mutationIdentityToken + const previous = process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN + delete process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN + try { + await assert.rejects( + runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ), + /requires a broker mutation identity token/ + ) + } finally { + if (previous !== undefined) { + process.env.ORCA_RELAY_FENCE_MUTATION_ID_TOKEN = previous + } + } + assert.equal(testHarness.events.length, 0) + }) +}) + +test('uses Terraform fencing for selector-era target supersession', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + selectorMembership: { + existingOnly: ['source', 'target1'], + migrationOnly: ['target2'], + general: ['general'] + } + }) + await runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ) + assert.deepEqual(testHarness.stateChanges, []) + assert.equal(testHarness.events.at(-1).event, 'registered_target_superseded') + }) +}) + +test('resumes failed-target supersession after losing the resize response', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + supersede: true, + loseResizeResponse: true + }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'supersede-target'), testHarness.overrides), + /injected_resize_response_loss/ + ) + await runMultiTargetDeployment( + config(file, 'supersede-target'), + testHarness.overrides + ) + assert.equal(testHarness.events.at(-1).event, 'registered_target_superseded') + }) +}) + +test('refuses to fence a failed target while its admission remains enabled', async () => { + await withTopology(async (file) => { + const testHarness = harness({ supersede: true, failedTargetEnabled: true }) + await assert.rejects( + runMultiTargetDeployment(config(file, 'supersede-target'), testHarness.overrides), + /failed target admission must be disabled/ + ) + assert.equal(testHarness.events.length, 0) + }) +}) + +test('retries until two complete target-capacity rounds agree', async () => { + await withTopology(async (file) => { + const testHarness = harness({ capacitySourceAssignments: [4, 3] }) + await runMultiTargetDeployment(config(file), testHarness.overrides) + assert.equal(testHarness.capacityReads(), 8) + assert.equal(testHarness.events.at(-1).sourceAssignments, 4) + }) +}) + +test('fails closed after bounded inconsistent target-capacity rounds', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + capacitySourceAssignments: Array.from( + { length: 12 }, + (_, index) => index % 2 === 0 ? 4 : 3 + ) + }) + await assert.rejects( + runMultiTargetDeployment(config(file), testHarness.overrides), + /target capacity snapshots disagree/ + ) + assert.equal(testHarness.capacityReads(), 12) + assert.deepEqual(testHarness.stateChanges, []) + }) +}) + +test('emits aggregate capacity inputs before rejecting insufficient headroom', async () => { + await withTopology(async (file) => { + const testHarness = harness({ + capacityRequiredTargetUnits: Array.from({ length: 4 }, () => 20_000) + }) + await assert.rejects( + runMultiTargetDeployment(config(file), testHarness.overrides), + /multi-target connection or request-unit headroom exhausted/ + ) + const snapshot = testHarness.events.at(-1) + assert.equal(snapshot.event, 'multi_target_capacity_snapshot') + assert.equal(snapshot.sourceConnections, 5) + assert.equal(snapshot.sourceAssignments, 4) + assert.equal(snapshot.requiredTargetUnits, 20_000) + assert.deepEqual(snapshot.targets, [ + { + cellId: 'target1', + currentConnections: 0, + availableConnectionReservations: 460, + availableTargetUnits: 4_000 + }, + { + cellId: 'target2', + currentConnections: 0, + availableConnectionReservations: 460, + availableTargetUnits: 4_000 + } + ]) + }) +}) + +test('rejects quotas above enforced target reservation headroom', () => { + assert.throws( + () => + allocateTargetQuotas({ + sourceAssignments: 3, + sourceConnections: 3, + requiredTargetUnits: 3, + connectionCeiling: 600, + targets: [ + { + cellId: 'target1', + currentConnections: 0, + availableConnectionReservations: 1, + availableTargetUnits: 10 + }, + { + cellId: 'target2', + currentConnections: 0, + availableConnectionReservations: 1, + availableTargetUnits: 10 + } + ] + }), + /multi-target connection or request-unit headroom exhausted/ + ) +}) diff --git a/cloud/dev/scripts/github-smoke-token.mjs b/cloud/dev/scripts/github-smoke-token.mjs new file mode 100644 index 00000000000..f0bd67c1024 --- /dev/null +++ b/cloud/dev/scripts/github-smoke-token.mjs @@ -0,0 +1,133 @@ +const REQUEST_TIMEOUT_MS = 15_000 +const JWT_PATTERN = /^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/ + +export async function requestGitHubSmokeTokens( + authOrigin, + fetchImpl = fetch, + environment = process.env, + options = {} +) { + const origin = canonicalHttpsOrigin(authOrigin) + const requestUrl = environment.ACTIONS_ID_TOKEN_REQUEST_URL + const requestToken = environment.ACTIONS_ID_TOKEN_REQUEST_TOKEN + if (!requestUrl || !requestToken) throw new Error('GitHub OIDC request context is unavailable') + const audience = `${origin}/v1/internal/github-smoke-token` + const oidcUrl = new URL(requestUrl) + oidcUrl.searchParams.set('audience', audience) + const oidcResponse = await fetchImpl(oidcUrl, { + headers: { authorization: `Bearer ${requestToken}` }, + signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS) + }) + const oidc = await readJson(oidcResponse, 'GitHub OIDC request') + if (!oidcResponse.ok || !validJwt(oidc.value)) { + throw new Error(`GitHub OIDC request failed with ${oidcResponse.status}`) + } + const exchangeResponse = await fetchImpl(audience, { + method: 'POST', + headers: { + authorization: `Bearer ${oidc.value}`, + ...(options.relayAsiaLoad ? { 'content-type': 'application/json' } : {}) + }, + ...(options.relayAsiaLoad + ? { body: JSON.stringify({ relayAsiaLoad: parseRelayAsiaLoadOptions(options.relayAsiaLoad) }) } + : {}), + signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS) + }) + const exchange = await readJson(exchangeResponse, 'Orca smoke identity exchange') + if (!exchangeResponse.ok) { + throw new Error(`Orca smoke identity exchange failed with ${exchangeResponse.status}`) + } + return { + ...parseAccessTokens(exchange.accessTokens), + ...(options.relayAsiaLoad + ? { + relayAsiaLoadPrincipals: parseRelayAsiaLoadPrincipals( + exchange.relayAsiaLoadPrincipals, + options.relayAsiaLoad.principalCount + ) + } + : {}) + } +} + +function parseRelayAsiaLoadOptions(value) { + if ( + !value || typeof value !== 'object' || + !Number.isSafeInteger(value.shardIndex) || value.shardIndex < 0 || value.shardIndex > 3 || + !Number.isSafeInteger(value.principalCount) || value.principalCount < 1 || + value.principalCount > 32 + ) throw new Error('Relay Asia load principal request is invalid') + return { v: 1, shardIndex: value.shardIndex, principalCount: value.principalCount } +} + +function canonicalHttpsOrigin(value) { + const url = new URL(value) + if (url.protocol !== 'https:' || url.pathname !== '/' || url.search || url.hash) { + throw new Error('auth origin must be canonical HTTPS') + } + return url.origin +} + +function validJwt(value) { + return typeof value === 'string' && value.length <= 8192 && JWT_PATTERN.test(value) +} + +function parseAccessTokens(value) { + if (!value || typeof value !== 'object' || Array.isArray(value)) { + throw new Error('Orca smoke identity response is malformed') + } + const expected = ['outsider', 'owner', 'recipient'] + if (Object.keys(value).sort().join(',') !== expected.join(',')) { + throw new Error('Orca smoke identity response principals are invalid') + } + return Object.fromEntries( + expected.map((name) => { + const principal = value[name] + if ( + !principal || + typeof principal !== 'object' || + typeof principal.userId !== 'string' || + !/^[A-Za-z0-9_-]{1,128}$/.test(principal.userId) || + typeof principal.accessToken !== 'string' || + !validJwt(principal.accessToken) || + typeof principal.expiresAt !== 'number' || + principal.expiresAt <= Date.now() || + principal.expiresAt > Date.now() + 610_000 + ) { + throw new Error('Orca smoke identity response principal is malformed') + } + return [name, principal] + }) + ) +} + +function parseRelayAsiaLoadPrincipals(value, expectedCount) { + if (!Array.isArray(value) || value.length !== expectedCount) { + throw new Error('Relay Asia load principal response is invalid') + } + const userIds = new Set() + return value.map((principal, principalIndex) => { + if ( + !principal || typeof principal !== 'object' || + principal.principalIndex !== principalIndex || + typeof principal.userId !== 'string' || + !/^usr_relay_asia_load_[A-Za-z0-9_-]{32}$/.test(principal.userId) || + typeof principal.profileId !== 'string' || + principal.profileId !== principal.userId.replace(/^usr_/, 'prof_') || + userIds.has(principal.userId) || + typeof principal.accessToken !== 'string' || !validJwt(principal.accessToken) || + typeof principal.expiresAt !== 'number' || principal.expiresAt <= Date.now() || + principal.expiresAt > Date.now() + 610_000 + ) throw new Error('Relay Asia load principal response is malformed') + userIds.add(principal.userId) + return principal + }) +} + +async function readJson(response, label) { + try { + return await response.json() + } catch { + throw new Error(`${label} did not return JSON`) + } +} diff --git a/cloud/dev/scripts/github-smoke-token.test.mjs b/cloud/dev/scripts/github-smoke-token.test.mjs new file mode 100644 index 00000000000..9ffb0b864d8 --- /dev/null +++ b/cloud/dev/scripts/github-smoke-token.test.mjs @@ -0,0 +1,111 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { requestGitHubSmokeTokens } from './github-smoke-token.mjs' + +const jwt = (value) => `${value}.${value}.${value}` + +test('exchanges the runner OIDC token without returning request credentials', async () => { + const requests = [] + const fetchImpl = async (url, init) => { + requests.push({ url: String(url), init }) + if (requests.length === 1) return Response.json({ value: jwt('github') }) + return Response.json({ + accessTokens: Object.fromEntries( + ['owner', 'recipient', 'outsider'].map((name) => [ + name, + { userId: `usr_${name}`, accessToken: jwt(name), expiresAt: Date.now() + 600_000 } + ]) + ) + }) + } + const result = await requestGitHubSmokeTokens( + 'https://auth-staging.onorca.dev', + fetchImpl, + { + ACTIONS_ID_TOKEN_REQUEST_URL: 'https://actions.example.test/token?api-version=1', + ACTIONS_ID_TOKEN_REQUEST_TOKEN: 'runner-request-token' + } + ) + assert.equal(result.owner.userId, 'usr_owner') + assert.match(requests[0].url, /audience=https%3A%2F%2Fauth-staging\.onorca\.dev/) + assert.equal(requests[0].init.headers.authorization, 'Bearer runner-request-token') + assert.equal(requests[1].init.headers.authorization, `Bearer ${jwt('github')}`) +}) + +test('fails with bounded errors and never includes credentials', async () => { + await assert.rejects( + requestGitHubSmokeTokens( + 'https://auth-staging.onorca.dev', + async () => new Response('denied', { status: 403 }), + { + ACTIONS_ID_TOKEN_REQUEST_URL: 'https://actions.example.test/token', + ACTIONS_ID_TOKEN_REQUEST_TOKEN: 'private-request-token' + } + ), + (error) => { + assert.doesNotMatch(String(error), /private-request-token|denied/) + return true + } + ) +}) + +test('requests and validates an exact Relay Asia principal batch', async () => { + const requests = [] + const principals = Array.from({ length: 32 }, (_, principalIndex) => { + const suffix = String(principalIndex).padStart(32, 'a') + return { + principalIndex, + userId: `usr_relay_asia_load_${suffix}`, + profileId: `prof_relay_asia_load_${suffix}`, + accessToken: jwt(`load${principalIndex}`), + expiresAt: Date.now() + 600_000 + } + }) + const result = await requestGitHubSmokeTokens( + 'https://auth-staging.onorca.dev', + async (url, init) => { + requests.push({ url: String(url), init }) + return requests.length === 1 + ? Response.json({ value: jwt('github') }) + : Response.json({ + accessTokens: Object.fromEntries(['owner', 'recipient', 'outsider'].map((name) => [ + name, + { userId: `usr_${name}`, accessToken: jwt(name), expiresAt: Date.now() + 600_000 } + ])), + relayAsiaLoadPrincipals: principals + }) + }, + { + ACTIONS_ID_TOKEN_REQUEST_URL: 'https://actions.example.test/token', + ACTIONS_ID_TOKEN_REQUEST_TOKEN: 'runner-request-token' + }, + { relayAsiaLoad: { shardIndex: 3, principalCount: 32 } } + ) + assert.equal(result.relayAsiaLoadPrincipals.length, 32) + assert.deepEqual(JSON.parse(requests[1].init.body), { + relayAsiaLoad: { v: 1, shardIndex: 3, principalCount: 32 } + }) + assert.equal(requests[1].init.headers['content-type'], 'application/json') +}) + +test('rejects malformed or duplicate Relay Asia principal batches', async () => { + const environment = { + ACTIONS_ID_TOKEN_REQUEST_URL: 'https://actions.example.test/token', + ACTIONS_ID_TOKEN_REQUEST_TOKEN: 'runner-request-token' + } + let request = 0 + await assert.rejects(requestGitHubSmokeTokens( + 'https://auth-staging.onorca.dev', + async () => ++request === 1 + ? Response.json({ value: jwt('github') }) + : Response.json({ + accessTokens: Object.fromEntries(['owner', 'recipient', 'outsider'].map((name) => [ + name, + { userId: `usr_${name}`, accessToken: jwt(name), expiresAt: Date.now() + 600_000 } + ])), + relayAsiaLoadPrincipals: [] + }), + environment, + { relayAsiaLoad: { shardIndex: 0, principalCount: 32 } } + ), /principal response is invalid/) +}) diff --git a/cloud/dev/scripts/infra.mjs b/cloud/dev/scripts/infra.mjs new file mode 100644 index 00000000000..6ab85df08b7 --- /dev/null +++ b/cloud/dev/scripts/infra.mjs @@ -0,0 +1,146 @@ +import { execFileSync } from 'node:child_process' +import { existsSync } from 'node:fs' +import { homedir } from 'node:os' +import { join } from 'node:path' +import { assertStagingRelayAwake } from './staging-relay-apply-guard.mjs' + +// The relay root keeps its historical path so every existing caller — 9 workflows, the fence +// broker, and the infra:* scripts — is unchanged when --root is omitted. The foundation and apps +// roots stay in the private repository with the services they own. +const ROOT_DIRECTORIES = { + relay: join('infra', 'terraform') +} + +const command = process.argv[2] +const environment = readEnvironment(process.argv.slice(3)) +const root = readRoot(process.argv.slice(3)) +const tool = process.env.IAC_TOOL || findTool() + +if (!command || !['init', 'plan', 'apply'].includes(command)) { + exitWithUsage() +} + +if (!environment) { + exitWithUsage('Missing --env staging|production') +} + +if (!root) { + exitWithUsage(`Unknown --root; expected one of ${Object.keys(ROOT_DIRECTORIES).join('|')}`) +} + +const rootDirectory = ROOT_DIRECTORIES[root] +const terraformDir = join(process.cwd(), rootDirectory) +const backendConfig = join(terraformDir, 'backend', `${environment}.hcl`) +const varFile = join(terraformDir, 'environments', `${environment}.tfvars`) + +if (!existsSync(backendConfig)) { + throw new Error(`Backend config not found: ${backendConfig}`) +} + +if (!existsSync(varFile)) { + throw new Error(`Variable file not found: ${varFile}`) +} + +const chdir = `-chdir=${rootDirectory}` + +if (command === 'init') { + run([chdir, 'init', `-backend-config=backend/${environment}.hcl`]) +} else if (command === 'plan') { + run([ + chdir, + 'plan', + `-var-file=environments/${environment}.tfvars`, + `-out=${environment}.tfplan` + ]) +} else { + // A normal staging apply must not implicitly wake or partially mutate a sleeping data plane. + // Only the relay root can touch that data plane; the guard would refuse app work for no reason. + if (environment === 'staging' && root === 'relay') assertStagingRelayAwake() + run([chdir, 'apply', `${environment}.tfplan`]) +} + +function readEnvironment(args) { + const envIndex = args.indexOf('--env') + if (envIndex >= 0) { + return args[envIndex + 1] + } + + return process.env.ORCA_CLOUD_ENV +} + +function readRoot(args) { + const rootIndex = args.indexOf('--root') + const requested = rootIndex >= 0 ? args[rootIndex + 1] : 'relay' + return requested in ROOT_DIRECTORIES ? requested : undefined +} + +function findTool() { + for (const candidate of ['tofu', 'terraform']) { + try { + execFileSync(candidate, ['version'], { stdio: 'ignore' }) + return candidate + } catch { + // Try the next compatible IaC binary. + } + } + + throw new Error('Install Terraform or OpenTofu, or set IAC_TOOL.') +} + +function run(args) { + execFileSync(tool, args, { env: terraformEnv(), stdio: 'inherit' }) +} + +function terraformEnv() { + if ( + process.env.GOOGLE_APPLICATION_CREDENTIALS || + process.env.GOOGLE_CREDENTIALS || + process.env.GOOGLE_OAUTH_ACCESS_TOKEN + ) { + return process.env + } + + const token = readGcloudAccessToken() + if (!token) { + return process.env + } + + // Local convenience: Terraform uses ADC, while engineers often only have + // gcloud CLI auth. CI should use Workload Identity instead. + return { ...process.env, GOOGLE_OAUTH_ACCESS_TOKEN: token } +} + +function readGcloudAccessToken() { + for (const candidate of gcloudCandidates()) { + try { + return execFileSync(candidate, ['auth', 'print-access-token'], { + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'ignore'] + }).trim() + } catch { + // Try the next common gcloud location. + } + } + + return null +} + +function gcloudCandidates() { + return [ + process.env.GCLOUD_PATH, + 'gcloud', + join(homedir(), 'Downloads', 'google-cloud-sdk', 'bin', 'gcloud'), + join(homedir(), 'google-cloud-sdk', 'bin', 'gcloud') + ].filter(Boolean) +} + +function exitWithUsage(message) { + if (message) { + console.error(message) + } + + console.error( + 'Usage: pnpm infra: --env staging|production [--root relay]' + ) + process.exit(1) +} diff --git a/cloud/dev/scripts/infra.test.mjs b/cloud/dev/scripts/infra.test.mjs new file mode 100644 index 00000000000..267bcf540ac --- /dev/null +++ b/cloud/dev/scripts/infra.test.mjs @@ -0,0 +1,77 @@ +import assert from 'node:assert/strict' +import { execFileSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { fileURLToPath } from 'node:url' + +const repository = fileURLToPath(new URL('../../', import.meta.url)) +const script = 'dev/scripts/infra.mjs' + +// IAC_TOOL=echo prints the argv the real binary would have received, so the root a flag selects +// is observable without running Terraform. +function invoke(args) { + return execFileSync('node', [script, ...args], { + cwd: repository, + encoding: 'utf8', + env: { ...process.env, IAC_TOOL: 'echo' } + }).trim() +} + +function rejects(args) { + try { + execFileSync('node', [script, ...args], { + cwd: repository, + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'], + env: { ...process.env, IAC_TOOL: 'echo' } + }) + } catch (error) { + return error.stderr + } + throw new Error(`expected ${args.join(' ')} to exit non-zero`) +} + +// Why: 9 relay workflows, the fence broker, and the three infra:* package scripts all invoke this +// without --root. If the default ever moves off infra/terraform they break silently at the plan. +test('omitting --root keeps every existing caller on the relay root', () => { + for (const environment of ['staging', 'production']) { + assert.equal( + invoke(['init', '--env', environment]), + `-chdir=infra/terraform init -backend-config=backend/${environment}.hcl` + ) + assert.equal( + invoke(['plan', '--env', environment]), + `-chdir=infra/terraform plan -var-file=environments/${environment}.tfvars -out=${environment}.tfplan` + ) + } +}) + +// Only the relay root ships here; the foundation and apps roots stay in the private repository. +test('each root name selects exactly its own directory', () => { + const directories = { relay: 'infra/terraform' } + for (const [root, directory] of Object.entries(directories)) { + for (const environment of ['staging', 'production']) { + assert.equal( + invoke(['init', '--env', environment, '--root', root]), + `-chdir=${directory} init -backend-config=backend/${environment}.hcl` + ) + } + } +}) + +test('an unknown root fails closed rather than falling back to the relay root', () => { + const stderr = rejects(['plan', '--env', 'staging', '--root', 'releay']) + assert.match(stderr, /Unknown --root/) + assert.doesNotMatch(stderr, /infra\/terraform /) +}) + +test('a missing environment still fails before any root is resolved', () => { + assert.match(rejects(['plan']), /Missing --env/) +}) + +// Why: the guard refuses a staging apply while the relay data plane is asleep. Applying it to the +// app or foundation roots would block work that never touches that data plane. +test('the sleeping staging relay guard is scoped to the relay root', () => { + const source = readFileSync(new URL('./infra.mjs', import.meta.url), 'utf8') + assert.match(source, /environment === 'staging' && root === 'relay'/) +}) diff --git a/cloud/dev/scripts/load-relay-controls.mjs b/cloud/dev/scripts/load-relay-controls.mjs new file mode 100644 index 00000000000..32cd1858dd6 --- /dev/null +++ b/cloud/dev/scripts/load-relay-controls.mjs @@ -0,0 +1,570 @@ +import { createHash, createPrivateKey, createPublicKey } from 'node:crypto' +import { readFileSync } from 'node:fs' +import { monitorEventLoopDelay } from 'node:perf_hooks' +import { setTimeout as delay } from 'node:timers/promises' +import { RelayLoadControlPeer } from './relay-load-control-peer.mjs' +import { requestGitHubSmokeTokens } from './github-smoke-token.mjs' +import { relayLoadFailureReason } from './relay-load-connection-failure.mjs' +import { + assertRelayLoadDirectorCapacityToken, + waitForRelayLoadDirectorCapacity, + waitForRelayLoadRequestUnits +} from './relay-load-director-capacity-gate.mjs' +import { waitForRelayLoadPhaseBarrier } from './relay-load-phase-barrier.mjs' +import { + proveRelayLoadPlacementBoundary, + proveRelayLoadRegionalFallback +} from './relay-load-placement-boundary.mjs' +import { + proveRelayLoadRebindBoundary, + waitForRelayLoadRebindGate +} from './relay-load-rebind-boundary.mjs' +import { proveRelayLoadRegionBehavior } from './relay-load-region-behavior.mjs' +import { + openRelayLoadInviteOffers, + proveRelayLoadRequestUnitBoundary +} from './relay-load-request-unit-boundary.mjs' +import { + assertRelayLoadRampAccepted, + relayLoadRunHasDisallowedFailures, + runRelayLoadWithShutdown +} from './relay-load-run-lifecycle.mjs' +import { createRelayLoadReaderEvidence } from './relay-load-reader-evidence.mjs' +import { + parseRelayLoadArguments, + relayLoadPrincipalIndex, + relayLoadReaderEvidenceError, + relayLoadSpliceIndexes, + relayLoadSpliceProfile, + relayLoadSpliceStartDelayMs +} from './relay-load-profile.mjs' + +function signingKey(path) { + if (!path) return {} + const key = createPrivateKey(readFileSync(path, 'utf8')) + const signingKeyId = createHash('sha256') + .update(createPublicKey(key).export({ type: 'spki', format: 'der' })) + .digest('base64url') + .slice(0, 16) + return { signingKey: key, signingKeyId } +} + +function report(state, final = false) { + const elapsedSeconds = Math.max(1, (Date.now() - state.startedAt) / 1000) + const memory = process.memoryUsage() + const cpu = process.cpuUsage(state.generatorBaselineCpu) + const rssMiB = memory.rss / 1_048_576 + state.generatorPeakRssMiB = Math.max(state.generatorPeakRssMiB, rssMiB) + const readerQueueEvidence = state.readerEvidence?.snapshot() ?? [] + const output = { + event: final ? 'relay_load_complete' : 'relay_load_progress', + controls: state.controls, + shardCount: state.shardCount, + shardIndex: state.shardIndex, + configuredRampSeconds: state.rampMs / 1000, + configuredSteadySeconds: state.durationMs / 1000, + configuredSpliceHoldSeconds: state.spliceHoldMs / 1000, + requiredLeaseHorizons: state.requiredLeaseHorizons, + configuredSplices: state.splices, + configuredSlowReaderSplices: state.slowReaderSplices, + configuredWedgedReaderSplices: state.wedgedReaderSplices, + active: state.active.size, + peakActive: state.peakActive, + steadyMinimumActive: state.steadyMinimumActive, + connected: state.connected, + connectionFailures: state.connectionFailures, + rampConnectionFailures: state.rampConnectionFailures, + steadyConnectionFailures: state.steadyConnectionFailures, + transitionConnectionFailures: state.transitionConnectionFailures, + connectionFailuresByReason: state.connectionFailuresByReason, + closes: state.closes, + unexpectedCloses: state.unexpectedCloses, + unexpectedClosesByCode: state.unexpectedClosesByCode, + drains: state.drains, + pings: state.pings, + pingRate: Number((state.pings / elapsedSeconds).toFixed(2)), + tokens: state.tokens, + tokenRate: Number((state.tokens / elapsedSeconds).toFixed(2)), + refreshes: state.refreshes, + refreshErrors: state.refreshErrors, + protocolErrors: state.protocolErrors, + socketErrors: state.socketErrors, + rebindProbesOpened: state.rebindProbesOpened, + rebindOverflowReason: state.rebindOverflowReason, + placementOverflowReason: state.placementOverflowReason, + regionalFallbacksProved: state.regionalFallbacksProved, + oldClientUsFirstProved: state.oldClientUsFirstProved, + stickyAssignmentProved: state.stickyAssignmentProved, + requestUnitInvitesOpened: state.requestUnitInvitesOpened, + requestUnitPrincipalCount: state.requestUnitPrincipalCount, + relayAsiaLoadPrincipalCount: state.relayAsiaLoadPrincipalCount, + requestUnitOverflowReason: state.requestUnitOverflowReason, + requestUnitCleanupProved: state.requestUnitCleanupProved, + phaseBarrierPassed: state.phaseBarrierPassed, + activeSplices: state.activeSplices, + peakActiveSplices: state.peakActiveSplices, + completedSplices: state.completedSplices, + failedSplices: state.failedSplices, + slowReaderSplicesCompleted: state.slowReaderSplicesCompleted, + wedgedReaderSplicesClosed: state.wedgedReaderSplicesClosed, + readerQueueEvidence, + readerQueuedBytesPeak: Math.max( + 0, + ...readerQueueEvidence.map(({ increaseBytes }) => increaseBytes) + ), + readerClosesByCode: state.readerClosesByCode, + controlHeadroom: Math.max(0, state.controls - state.active.size), + generatorRssMiB: Number(rssMiB.toFixed(1)), + generatorPeakRssMiB: Number(state.generatorPeakRssMiB.toFixed(1)), + generatorRssGrowthMiB: Number( + Math.max(0, state.generatorPeakRssMiB - state.generatorBaselineRssMiB).toFixed(1) + ), + generatorHeapUsedMiB: Number((memory.heapUsed / 1_048_576).toFixed(1)), + generatorCpuPercent: Number( + (((cpu.user + cpu.system) / 1_000_000 / elapsedSeconds) * 100).toFixed(1) + ), + generatorEventLoopP99Ms: Number((state.eventLoopDelay.percentile(99) / 1_000_000).toFixed(2)), + shutdownEvidence: final ? state.shutdownEvidence : undefined, + elapsedSeconds: Number(elapsedSeconds.toFixed(1)) + } + console.log(JSON.stringify(output)) + return output +} + +const config = parseRelayLoadArguments(process.argv.slice(2)) +let accessToken = process.env.ORCA_RELAY_LOAD_ACCESS_TOKEN +let accessTokenProviderForIndex +const adminToken = process.env.ORCA_RELAY_ADMIN_ID_TOKEN +if ( + config.placementOverflowProbes > 0 || config.regionalFallbackProbes > 0 || + config.slowReaderSplices + config.wedgedReaderSplices > 0 || + config.requestUnitOverflowProbes > 0 || config.requestUnitCleanupTimeoutMs > 0 +) { + assertRelayLoadDirectorCapacityToken({ + directorOrigin: config.directorOrigin, + adminToken + }, Date.now, + config.rampMs + config.durationMs + config.wedgedReaderHoldMs + + (config.phaseBarrierDir ? 2 * config.phaseBarrierTimeoutMs : 0) + + config.spliceRampMs + config.requestUnitCleanupTimeoutMs + 120_000) +} +const key = signingKey(config.signingKeyFile) +if (!accessToken && !key.signingKey && process.env.ACTIONS_ID_TOKEN_REQUEST_URL && + process.env.ACTIONS_ID_TOKEN_REQUEST_TOKEN) { + let tokens + let refresh + const loadOptions = config.relayAsiaLoadPrincipalCount > 0 + ? { + relayAsiaLoad: { + shardIndex: config.shardIndex, + principalCount: config.relayAsiaLoadPrincipalCount + } + } + : undefined + const smokeTokens = async () => { + const expiresAt = config.relayAsiaLoadPrincipalCount > 0 + ? tokens?.relayAsiaLoadPrincipals?.[0]?.expiresAt + : tokens?.owner?.expiresAt + if (expiresAt > Date.now() + 60_000) return tokens + refresh ??= requestGitHubSmokeTokens( + config.authOrigin, + fetch, + process.env, + loadOptions + ) + try { + tokens = await refresh + return tokens + } finally { + refresh = undefined + } + } + accessTokenProviderForIndex = (index) => async () => { + const current = await smokeTokens() + return config.relayAsiaLoadPrincipalCount > 0 + ? current.relayAsiaLoadPrincipals[ + relayLoadPrincipalIndex( + index, + config.shardCount, + current.relayAsiaLoadPrincipals.length + ) + ].accessToken + : current.owner.accessToken + } + await smokeTokens() +} +if (!accessToken && !accessTokenProviderForIndex && !key.signingKey) { + throw new Error('provide GitHub OIDC, ORCA_RELAY_LOAD_ACCESS_TOKEN, or --signing-key-file') +} +const eventLoopDelay = monitorEventLoopDelay({ resolution: 20 }) +eventLoopDelay.enable() +const generatorBaselineRssMiB = process.memoryUsage().rss / 1_048_576 +const generatorBaselineCpu = process.cpuUsage() +const state = { + ...config, + startedAt: Date.now(), + active: new Set(), + peakActive: 0, + steadyMinimumActive: null, + steadyStarted: false, + connected: 0, + connectionFailures: 0, + rampConnectionFailures: 0, + steadyConnectionFailures: 0, + transitionConnectionFailures: 0, + connectionFailuresByReason: {}, + closes: 0, + unexpectedCloses: 0, + unexpectedClosesByCode: {}, + drains: 0, + pings: 0, + tokens: 0, + refreshes: 0, + refreshErrors: 0, + protocolErrors: 0, + socketErrors: 0, + rebindProbesOpened: 0, + rebindOverflowReason: null, + placementOverflowReason: null, + regionalFallbacksProved: 0, + oldClientUsFirstProved: 0, + stickyAssignmentProved: 0, + requestUnitInvitesOpened: 0, + requestUnitOverflowReason: null, + requestUnitCleanupProved: 0, + phaseBarrierPassed: false, + activeSplices: 0, + peakActiveSplices: 0, + completedSplices: 0, + failedSplices: 0, + slowReaderSplicesCompleted: 0, + wedgedReaderSplicesClosed: 0, + readerEvidence: null, + readerClosesByCode: {}, + generatorBaselineRssMiB, + generatorBaselineCpu, + generatorPeakRssMiB: generatorBaselineRssMiB, + peerShutdowns: 0, + shutdownEvidence: null, + eventLoopDelay, + stopping: false, + transitionWindow: false +} +const peers = new Map() +const reconnectTimers = new Set() + +async function readRuntimeQueuedBytes(origin) { + const response = await fetch(`${origin}/v1/admin/runtime-status`, { + method: 'POST', + headers: { authorization: `Bearer ${adminToken}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1 }), + signal: AbortSignal.timeout(5_000) + }) + if (response.status === 401 || response.status === 403) { + throw new Error('reader evidence identity was rejected') + } + if (!response.ok) throw new Error(`reader runtime status returned ${response.status}`) + const status = await response.json() + const queuedBytes = status?.runtime?.queuedBytes + if (!Number.isSafeInteger(queuedBytes) || queuedBytes < 0) { + throw new Error('reader runtime queued bytes are invalid') + } + return queuedBytes +} + +async function observeReaderPressure(input) { + if (!state.readerEvidence) throw new Error('reader evidence baseline is unavailable') + await state.readerEvidence.observe(input) + state.generatorPeakRssMiB = Math.max( + state.generatorPeakRssMiB, + process.memoryUsage().rss / 1_048_576 + ) +} + +function recordSteadyMinimum() { + if (!state.steadyStarted || state.stopping) return + state.steadyMinimumActive = Math.min(state.steadyMinimumActive, state.active.size) +} + +function scheduleReconnect(peer) { + if (state.stopping) return + const timeout = setTimeout(() => { + reconnectTimers.delete(timeout) + void connect(peer) + }, Math.floor(Math.random() * (config.reconnectMaxMs + 1))) + reconnectTimers.add(timeout) +} + +function observe(type, detail) { + if (type === 'connected') { + state.active.add(detail.index) + state.connected++ + state.peakActive = Math.max(state.peakActive, state.active.size) + recordSteadyMinimum() + } else if (type === 'closed') { + state.active.delete(detail.index) + state.closes++ + if (!detail.stopped && !detail.expectedDrain) { + state.unexpectedCloses++ + const code = String(detail.code) + state.unexpectedClosesByCode[code] = (state.unexpectedClosesByCode[code] ?? 0) + 1 + } + if (!detail.stopped) scheduleReconnect(peers.get(detail.index)) + recordSteadyMinimum() + } else if (type === 'drain') state.drains++ + else if (type === 'ping') state.pings++ + else if (type === 'token') state.tokens++ + else if (type === 'refresh') state.refreshes++ + else if (type === 'refreshError') state.refreshErrors++ + else if (type === 'protocolError') state.protocolErrors++ + else if (type === 'socketError') state.socketErrors++ + else if (type === 'spliceOpened') { + state.activeSplices++ + state.peakActiveSplices = Math.max(state.peakActiveSplices, state.activeSplices) + } else if (type === 'spliceCompleted') { + state.completedSplices++ + if (detail.readerMode === 'slow') state.slowReaderSplicesCompleted++ + } else if (type === 'spliceWedged') { + state.wedgedReaderSplicesClosed++ + const code = String(detail.code) + state.readerClosesByCode[code] = (state.readerClosesByCode[code] ?? 0) + 1 + } else if (type === 'spliceClosed') state.activeSplices-- + else if (type === 'spliceFailed') state.failedSplices++ + else if (type === 'shutdown') state.peerShutdowns++ +} + +async function connect(peer) { + try { + await peer.connect() + } catch (error) { + state.connectionFailures++ + if (state.steadyStarted) state.steadyConnectionFailures++ + else if (state.transitionWindow) state.transitionConnectionFailures++ + else state.rampConnectionFailures++ + const reason = relayLoadFailureReason(error) + state.connectionFailuresByReason[reason] = + (state.connectionFailuresByReason[reason] ?? 0) + 1 + scheduleReconnect(peer) + } +} + +const peerOptions = (index, overrides = {}) => ({ + ...config, + ...key, + accessToken, + ...(accessTokenProviderForIndex + ? { accessTokenProvider: accessTokenProviderForIndex(index) } + : {}), + seed: 0x4f524341 ^ config.shardIndex, + ...overrides +}) +if (config.regionBehaviorProbes > 0) { + const proofIndex = config.controls * config.shardCount + 10_000 + const regionProof = await proveRelayLoadRegionBehavior({ + oldClientPeer: new RelayLoadControlPeer( + proofIndex, + peerOptions(proofIndex, { preferredRegion: undefined }), + () => undefined + ), + stickyPeer: new RelayLoadControlPeer( + proofIndex + 1, + peerOptions(proofIndex + 1, { preferredRegion: 'asia-east2' }), + () => undefined + ), + asiaOrigin: config.capacityCellOrigin + }) + state.oldClientUsFirstProved = regionProof.oldClientUsFirst ? 1 : 0 + state.stickyAssignmentProved = regionProof.stickyAssignmentPreserved ? 1 : 0 +} +const initialConnections = [] +for (let localIndex = 0; localIndex < config.controls; localIndex++) { + const globalIndex = localIndex * config.shardCount + config.shardIndex + const peer = new RelayLoadControlPeer(globalIndex, peerOptions(globalIndex), observe) + peers.set(globalIndex, peer) + const rampOffset = + config.controls === 1 ? 0 : Math.floor((localIndex / (config.controls - 1)) * config.rampMs) + const offset = config.rampStartDelayMs + rampOffset + initialConnections.push(delay(offset).then(() => connect(peer))) +} +const progressTimer = setInterval(() => report(state), 10_000) +progressTimer.unref() +await runRelayLoadWithShutdown(async () => { + await Promise.all(initialConnections) + assertRelayLoadRampAccepted(state.rampConnectionFailures, config.maxRampConnectionFailures) + if ( + config.rebindProbes > 0 || config.placementOverflowProbes > 0 || + config.regionalFallbackProbes > 0 + ) { + state.transitionWindow = config.rebindDelayMs > 0 + await waitForRelayLoadRebindGate({ + delay, + delayMs: config.rebindDelayMs, + activeCount: () => state.active.size, + requiredCount: config.controls + }) + state.transitionWindow = false + } + if (config.placementOverflowProbes > 0 || config.regionalFallbackProbes > 0) { + const closesBeforeBoundary = state.closes + await waitForRelayLoadDirectorCapacity({ + directorOrigin: config.directorOrigin, + adminToken, + cellId: config.capacityCellId, + hardCap: config.capacityHardCap, + unobservedBound: config.capacityUnobservedBound, + requiredConnections: config.aggregateControls + }) + if (state.active.size !== config.controls || state.closes !== closesBeforeBoundary) { + throw new Error('ordinary controls changed during the director capacity gate') + } + const overflowIndex = config.controls * config.shardCount + config.shardIndex + if (config.placementOverflowProbes > 0) { + state.placementOverflowReason = await proveRelayLoadPlacementBoundary({ + peer: new RelayLoadControlPeer(overflowIndex, peerOptions(overflowIndex), observe), + failureReason: relayLoadFailureReason + }) + } + if (config.regionalFallbackProbes > 0) { + await proveRelayLoadRegionalFallback({ + peer: new RelayLoadControlPeer(overflowIndex, peerOptions(overflowIndex), () => undefined), + blockedOrigin: config.capacityCellOrigin + }) + state.regionalFallbacksProved = 1 + } + if (state.active.size !== config.controls || state.closes !== closesBeforeBoundary) { + throw new Error('ordinary controls changed during the placement boundary probe') + } + await waitForRelayLoadDirectorCapacity({ + directorOrigin: config.directorOrigin, + adminToken, + cellId: config.capacityCellId, + hardCap: config.capacityHardCap, + unobservedBound: config.capacityUnobservedBound, + requiredConnections: config.aggregateControls, + requiredSamples: 1 + }) + if (state.active.size !== config.controls || state.closes !== closesBeforeBoundary) { + throw new Error('ordinary controls changed before post-probe capacity verification') + } + } + const rebindResult = await proveRelayLoadRebindBoundary({ + peers: [...state.active].map((index) => peers.get(index)), + probeCount: config.rebindProbes, + holdMs: config.rebindHoldMs, + delay, + failureReason: relayLoadFailureReason, + requireOverflow: config.requireRebindOverflow + }) + state.rebindProbesOpened = rebindResult.opened + state.rebindOverflowReason = rebindResult.overflowReason + if (config.requestUnitInvites > 0) { + state.requestUnitInvitesOpened = await openRelayLoadInviteOffers({ + peers: [...state.active].sort((left, right) => left - right).map((index) => peers.get(index)), + count: config.requestUnitInvites, + ratePerSecond: config.requestUnitInvitesPerSecond + }) + } + if (config.requestUnitOverflowProbes > 0) { + await waitForRelayLoadRequestUnits({ + directorOrigin: config.directorOrigin, + adminToken, + cellId: config.capacityCellId, + capacityRequests: config.requestUnitCapacity, + expectedRequestUnits: config.requestUnitCapacity, + expectedActivityLeases: config.requestUnitCapacity + }) + state.requestUnitOverflowReason = await proveRelayLoadRequestUnitBoundary( + peers.get([...state.active][0]) + ) + } + if (config.phaseBarrierDir) { + await waitForRelayLoadPhaseBarrier({ + directory: config.phaseBarrierDir, + shardCount: config.shardCount, + shardIndex: config.shardIndex, + timeoutMs: config.phaseBarrierTimeoutMs + }) + state.phaseBarrierPassed = true + } + state.steadyStarted = true + state.steadyMinimumActive = state.active.size + const spliceIndexes = relayLoadSpliceIndexes(config) + const readerOrigins = spliceIndexes.flatMap((index, spliceIndex) => + relayLoadSpliceProfile(config, spliceIndex).readerMode === 'normal' + ? [] + : [peers.get(index).lastAssignment.cellUrl] + ) + state.readerEvidence = await createRelayLoadReaderEvidence(readerOrigins, { + readQueuedBytes: readRuntimeQueuedBytes, + delay + }) + if (config.phaseBarrierDir) { + await waitForRelayLoadPhaseBarrier({ + directory: `${config.phaseBarrierDir}-splices`, + shardCount: config.shardCount, + shardIndex: config.shardIndex, + timeoutMs: config.phaseBarrierTimeoutMs + }) + } + const splicePromises = spliceIndexes.map((index, spliceIndex) => + delay(relayLoadSpliceStartDelayMs(config, spliceIndex)).then(() => + peers.get(index).openSplice({ + payloadBytes: config.splicePayloadBytes, + ...relayLoadSpliceProfile(config, spliceIndex), + observeReaderPressure, + holdMs: config.spliceHoldMs + }) + ) + ) + await Promise.all([...splicePromises, delay(config.durationMs)]) +}, async () => { + state.stopping = true + clearInterval(progressTimer) + for (const timeout of reconnectTimers) clearTimeout(timeout) + reconnectTimers.clear() + await Promise.all([...peers.values()].map((peer) => peer.shutdown())) + eventLoopDelay.disable() + state.shutdownEvidence = { + peerShutdowns: state.peerShutdowns, + activeControls: state.active.size, + activeSplices: state.activeSplices, + reconnectTimers: reconnectTimers.size + } +}) +if (config.requestUnitCleanupTimeoutMs > 0) { + await waitForRelayLoadRequestUnits({ + directorOrigin: config.directorOrigin, + adminToken, + cellId: config.capacityCellId, + capacityRequests: config.requestUnitCapacity, + expectedRequestUnits: 0, + expectedActivityLeases: 0, + timeoutMs: config.requestUnitCleanupTimeoutMs + }) + state.requestUnitCleanupProved = 1 +} +const result = report(state, true) +const minimumPeak = config.allowPartial ? 1 : Math.ceil(config.controls * 0.95) +if (result.peakActive < minimumPeak) { + throw new Error(`peak active controls ${result.peakActive} below required ${minimumPeak}`) +} +if (result.steadyMinimumActive < minimumPeak) { + throw new Error( + `steady minimum active controls ${result.steadyMinimumActive} below required ${minimumPeak}` + ) +} +if (relayLoadRunHasDisallowedFailures(result, config)) { + throw new Error('relay load run observed connection, protocol, refresh, or socket errors') +} +const readerEvidenceError = relayLoadReaderEvidenceError(result, config) +if (readerEvidenceError) throw new Error(readerEvidenceError) +if ( + result.failedSplices > 0 || + result.completedSplices + result.wedgedReaderSplicesClosed !== config.splices || + result.shutdownEvidence.peerShutdowns !== config.controls || + result.shutdownEvidence.activeControls !== 0 || + result.shutdownEvidence.activeSplices !== 0 || + result.shutdownEvidence.reconnectTimers !== 0 +) { + throw new Error('relay load run did not complete splices or shut down cleanly') +} diff --git a/cloud/dev/scripts/operate-relay-asia-admission.mjs b/cloud/dev/scripts/operate-relay-asia-admission.mjs new file mode 100644 index 00000000000..45322bbe620 --- /dev/null +++ b/cloud/dev/scripts/operate-relay-asia-admission.mjs @@ -0,0 +1,440 @@ +import { createHash } from 'node:crypto' +import { fileURLToPath } from 'node:url' +import { + addExactMigrationCells, + applyExactAdmissionSelector, + inspectAdmissionSelector, + membershipWithStates, + selectorCellState +} from './relay-admission-selector.mjs' + +const SHAPES = { + staging: { + directorOrigin: 'https://relay-staging.onorca.dev', + domain: 'relay-staging.onorca.dev', + allCells: ['staging-gce-c4'] + }, + production: { + directorOrigin: 'https://relay.onorca.dev', + domain: 'relay.onorca.dev', + allCells: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'] + } +} + +function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['environment', 'mode', 'cell-ids', 'image-digest']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!/^sha256:[a-f0-9]{64}$/.test(values['image-digest'])) { + throw new Error('--image-digest is invalid') + } + if (![ + 'inspect', 'initialize', 'verify', 'registered', 'register', + 'promote', 'recover-promotion', 'rollback' + ].includes(values.mode)) { + throw new Error('--mode is invalid') + } + const expectedGeneration = values.mode === 'inspect' + ? undefined + : Number(values['expected-generation']) + if ( + values.mode !== 'inspect' && + (!Number.isSafeInteger(expectedGeneration) || expectedGeneration < 0) + ) { + throw new Error('--expected-generation is invalid') + } + const shape = SHAPES[values.environment] + if (!shape) throw new Error('--environment is invalid') + const cells = values['cell-ids'].split(',').map((value) => value.trim()).filter(Boolean) + const distinct = new Set(cells) + if (distinct.size !== cells.length || cells.some((cell) => !shape.allCells.includes(cell))) { + throw new Error('--cell-ids are invalid') + } + const exact = (expected) => JSON.stringify([...cells].sort()) === JSON.stringify([...expected].sort()) + if ( + (['inspect', 'initialize', 'register', 'registered', 'verify'].includes(values.mode) && + !exact(shape.allCells)) || + (['promote', 'recover-promotion'].includes(values.mode) && values.environment === 'production' && + !exact(['production-gce-c27']) && !exact(['production-gce-c28', 'production-gce-c29'])) || + (['promote', 'recover-promotion'].includes(values.mode) && values.environment === 'staging' && !exact(shape.allCells)) || + (values.mode === 'rollback' && cells.length === 0) + ) throw new Error('--cell-ids do not match the reviewed admission wave') + const attemptId = values['attempt-id'] + if (!['inspect', 'verify', 'registered'].includes(values.mode) && + !/^[A-Za-z0-9_-]{8,128}$/.test(attemptId ?? '')) { + throw new Error('--attempt-id is invalid') + } + return { + environment: values.environment, + mode: values.mode, + cells, + expectedGeneration, + expectedMembershipSha256: values['expected-membership-sha256'], + imageDigest: values['image-digest'], + attemptId, + token: process.env.ORCA_RELAY_ADMIN_ID_TOKEN ?? '' + } +} + +function hostname(cellId) { + return cellId.split('-').at(-1) +} + +function cellOrigin(shape, cellId) { + return `https://${hostname(cellId)}.${shape.domain}` +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({})) + if (!response.ok) throw new Error(`${label} returned ${response.status}`) + return body +} + +function defaultPost(fetchImpl, token) { + return async (url, body) => await responseJson(await fetchImpl(url, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }), new URL(url).pathname) +} + +async function verifyRuntime(fetchImpl, post, shape, cellId, imageDigest, requireDirector) { + const origin = cellOrigin(shape, cellId) + const [health, ready, runtime] = await Promise.all([ + fetchImpl(`${origin}/health`, { redirect: 'error', signal: AbortSignal.timeout(8_000) }), + fetchImpl(`${origin}/ready`, { redirect: 'error', signal: AbortSignal.timeout(8_000) }), + post(`${origin}/v1/admin/runtime-status`, { v: 1 }) + ]) + if (!health.ok || !ready.ok) throw new Error(`${cellId} is not ready`) + if ( + runtime.cellId !== cellId || + runtime.cellUrl !== origin || + runtime.region !== 'asia-east2' || + runtime.imageDigest !== imageDigest || + runtime.draining !== false || + runtime.connectionCapacity?.hardCap !== 3_000 || + runtime.connectionCapacity?.unobservedBound !== 60 + ) throw new Error(`${cellId} runtime does not match the reviewed Asia shape`) + if (requireDirector) { + const result = await post(`${shape.directorOrigin}/v1/admin/cell-status`, { v: 1, cellId }) + if ( + result.status?.cellUrl !== origin || + result.status?.runtime?.heartbeatFresh !== true || + result.status?.runtime?.ready !== true + ) throw new Error(`${cellId} has no fresh ready director heartbeat`) + } +} + +function membershipStates(selector, cells) { + return Object.fromEntries(cells.map((cellId) => [cellId, selectorCellState(selector, cellId)])) +} + +function inspectedMembershipStates(selector, cells) { + const known = new Set([ + ...selector.membership.existingOnly, + ...selector.membership.migrationOnly, + ...selector.membership.general + ]) + return Object.fromEntries(cells.map((cellId) => [ + cellId, + known.has(cellId) ? selectorCellState(selector, cellId) : 'absent' + ])) +} + +function sameMembership(left, right) { + return JSON.stringify(left) === JSON.stringify(right) +} + +function membershipSha256(membership) { + return createHash('sha256').update(JSON.stringify(membership)).digest('hex') +} + +async function initializeAdmissionBoundary(post, selectorPost, shape, config, current) { + if (config.expectedGeneration !== 0) { + throw new Error('admission boundary initialization requires generation 0') + } + if ( + !/^[a-f0-9]{64}$/.test(config.expectedMembershipSha256 ?? '') || + membershipSha256(current.selector.membership) !== config.expectedMembershipSha256 + ) { + throw new Error('admission membership changed before boundary initialization') + } + const targetStates = inspectedMembershipStates(current.selector, config.cells) + if (Object.values(targetStates).some((state) => state !== 'absent')) { + throw new Error('Asia cell exists before admission boundary initialization') + } + const intendedMembership = current.intent?.previousMembership ?? current.selector.membership + const exactCommitted = (inspection) => + inspection.intent?.state === 'committed' && + inspection.intent.expectedGeneration === 0 && + inspection.selector.generation === 1 && + inspection.selector.attemptId === config.attemptId && + sameMembership(inspection.intent.previousMembership, intendedMembership) && + sameMembership(inspection.intent.membership, intendedMembership) && + sameMembership(inspection.selector.membership, intendedMembership) + const exactUnchanged = (inspection) => + inspection.intent?.state === 'unchanged' && + inspection.intent.expectedGeneration === 0 && + inspection.selector.generation === 0 && + sameMembership(inspection.intent.previousMembership, intendedMembership) && + sameMembership(inspection.intent.membership, intendedMembership) && + sameMembership(inspection.selector.membership, intendedMembership) + if (exactCommitted(current)) { + return { + mode: config.mode, + generation: current.selector.generation, + states: inspectedMembershipStates(current.selector, config.cells), + recovered: true + } + } + if (current.intent && !exactUnchanged(current)) { + throw new Error('admission boundary initialization attempt diverged') + } + const request = { + v: 1, + attemptId: config.attemptId, + expectedGeneration: 0, + expectedMembershipSha256: config.expectedMembershipSha256, + membership: intendedMembership + } + let applyError + for (let attempt = 0; attempt < 2; attempt++) { + try { + await post(`${shape.directorOrigin}/v1/admin/admission-selector/apply`, request) + } catch (error) { + applyError = error + } + const verified = await inspectAdmissionSelector(selectorPost, config.attemptId) + if (exactCommitted(verified)) { + return { + mode: config.mode, + generation: verified.selector.generation, + states: targetStates, + recovered: current.intent !== null || applyError !== undefined || attempt > 0 + } + } + if (!exactUnchanged(verified)) { + throw new Error('admission boundary initialization did not commit exactly', { + cause: applyError + }) + } + } + throw new Error('admission boundary initialization remained unchanged after retry', { + cause: applyError + }) +} + +export async function operateRelayAsiaAdmission(config, dependencies = {}) { + const shape = SHAPES[config.environment] + const fetchImpl = dependencies.fetch ?? fetch + const post = dependencies.post ?? defaultPost(fetchImpl, config.token) + const selectorPost = (path, body) => { + if (config.environment !== 'staging' || path !== '/v1/admin/admission-selector/apply') { + return post(`${shape.directorOrigin}${path}`, body) + } + const state = selectorCellState({ membership: body.membership }, 'staging-gce-c4') + if (!['general', 'migration-only'].includes(state)) { + throw new Error('staging proof can only transition C4 between reviewed states') + } + return post(`${shape.directorOrigin}/v1/admin/admission-selector/apply-staging-asia-proof`, { + v: 1, + attemptId: body.attemptId, + expectedGeneration: body.expectedGeneration, + state + }) + } + const current = await inspectAdmissionSelector( + selectorPost, + ['inspect', 'verify', 'registered'].includes(config.mode) ? undefined : config.attemptId + ) + if (config.mode === 'inspect') { + return { + mode: config.mode, + generation: current.selector.generation, + membership: current.selector.membership, + membershipSha256: membershipSha256(current.selector.membership), + states: inspectedMembershipStates(current.selector, config.cells) + } + } + if ( + !current.intent && + current.selector.generation !== config.expectedGeneration + ) { + throw new Error('admission selector generation changed') + } + if (current.intent && current.intent.expectedGeneration !== config.expectedGeneration) { + throw new Error('admission attempt generation does not match') + } + if (config.mode === 'initialize') { + return await initializeAdmissionBoundary(post, selectorPost, shape, config, current) + } + if (config.mode === 'recover-promotion') { + if (config.cells.every( + (cellId) => selectorCellState(current.selector, cellId) === 'migration-only' + )) { + return { + mode: config.mode, + promoted: false, + generation: current.selector.generation, + states: membershipStates(current.selector, config.cells) + } + } + if (!current.intent) { + if ( + current.selector.generation !== config.expectedGeneration + ) throw new Error('promotion state changed without the reviewed attempt') + return { + mode: config.mode, + promoted: false, + generation: current.selector.generation, + states: membershipStates(current.selector, config.cells) + } + } + const expectedMembership = membershipWithStates( + { membership: current.intent.previousMembership }, + Object.fromEntries(config.cells.map((cellId) => [cellId, 'general'])) + ) + if ( + current.intent.state !== 'committed' || + JSON.stringify(current.intent.membership) !== JSON.stringify(expectedMembership) || + config.cells.some((cellId) => selectorCellState(current.selector, cellId) !== 'general') + ) throw new Error('promotion attempt is not the current general state') + return { + mode: config.mode, + promoted: true, + generation: current.selector.generation, + states: membershipStates(current.selector, config.cells) + } + } + if (config.mode === 'rollback') { + for (const cellId of config.cells) { + if (!['general', 'migration-only'].includes(selectorCellState(current.selector, cellId))) { + throw new Error(`${cellId} cannot roll back to migration-only`) + } + } + } else if (config.mode === 'register') { + const known = new Set([ + ...current.selector.membership.existingOnly, + ...current.selector.membership.migrationOnly, + ...current.selector.membership.general + ]) + if (!current.intent && config.cells.some((cellId) => known.has(cellId))) { + throw new Error('Asia cell is already registered') + } + await Promise.all(config.cells.map((cellId) => + verifyRuntime(fetchImpl, post, shape, cellId, config.imageDigest, false) + )) + } else if (config.mode !== 'registered') { + await Promise.all(config.cells.map((cellId) => + verifyRuntime(fetchImpl, post, shape, cellId, config.imageDigest, true) + )) + } + if (config.mode === 'registered') { + if (config.cells.some( + (cellId) => selectorCellState(current.selector, cellId) !== 'migration-only' + )) throw new Error('Asia cells are not registered migration-only') + await Promise.all(config.cells.map((cellId) => + verifyRuntime(fetchImpl, post, shape, cellId, config.imageDigest, false) + )) + } + if (['verify', 'registered'].includes(config.mode)) { + return { + mode: config.mode, + generation: current.selector.generation, + states: membershipStates(current.selector, config.cells) + } + } + if (config.mode === 'register') { + if (current.intent) { + const expectedCells = new Set(config.cells) + const addedCells = current.intent.membership.migrationOnly.filter( + (cellId) => !current.intent.previousMembership.migrationOnly.includes(cellId) + ) + if ( + current.intent.state !== 'committed' || + addedCells.length !== expectedCells.size || + addedCells.some((cellId) => !expectedCells.has(cellId)) || + current.selector.generation !== config.expectedGeneration + 1 || + JSON.stringify(current.selector.membership) !== JSON.stringify(current.intent.membership) + ) { + throw new Error('admission attempt does not match the requested Asia registration') + } + return { + mode: config.mode, + generation: current.selector.generation, + states: membershipStates(current.selector, config.cells), + recovered: true + } + } + const result = await addExactMigrationCells( + selectorPost, + { + attemptId: config.attemptId, + cells: config.cells.map((cellId) => ({ + cellId, + cellUrl: cellOrigin(shape, cellId), + region: 'asia-east2', + capacityRequests: 6_000, + connectionHardCap: 3_000, + connectionUnobservedBound: 60 + })) + }, + { expectedCurrentSelector: current.selector } + ) + return { mode: config.mode, generation: result.selector.generation, states: membershipStates(result.selector, config.cells) } + } + const desiredState = config.mode === 'promote' ? 'general' : 'migration-only' + if (current.intent) { + const expectedMembership = membershipWithStates( + { membership: current.intent.previousMembership }, + Object.fromEntries(config.cells.map((cellId) => [cellId, desiredState])) + ) + if ( + current.intent.state !== 'committed' || + JSON.stringify(current.intent.membership) !== JSON.stringify(expectedMembership) || + current.selector.generation !== config.expectedGeneration + 1 || + JSON.stringify(current.selector.membership) !== JSON.stringify(current.intent.membership) + ) { + throw new Error('admission attempt does not match the requested Asia transition') + } + return { + mode: config.mode, + generation: current.selector.generation, + states: membershipStates(current.selector, config.cells), + recovered: true + } + } + if (config.mode === 'promote' && config.cells.some( + (cellId) => selectorCellState(current.selector, cellId) !== 'migration-only' + )) throw new Error('Asia promotion requires migration-only cells') + if ( + config.mode === 'promote' && + config.environment === 'production' && + config.cells.includes('production-gce-c28') && + selectorCellState(current.selector, 'production-gce-c27') !== 'general' + ) { + throw new Error('Asia expansion requires the C27 canary to be general') + } + const result = await applyExactAdmissionSelector( + selectorPost, + membershipWithStates(current.selector, Object.fromEntries( + config.cells.map((cellId) => [cellId, desiredState]) + )), + { attemptId: config.attemptId, expectedCurrentSelector: current.selector } + ) + return { mode: config.mode, generation: result.selector.generation, states: membershipStates(result.selector, config.cells) } +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const config = parseArguments(process.argv.slice(2)) + if (!config.token) throw new Error('ORCA_RELAY_ADMIN_ID_TOKEN is required') + console.log(JSON.stringify(await operateRelayAsiaAdmission(config))) +} diff --git a/cloud/dev/scripts/operate-relay-asia-admission.test.mjs b/cloud/dev/scripts/operate-relay-asia-admission.test.mjs new file mode 100644 index 00000000000..7f23bb1f8e6 --- /dev/null +++ b/cloud/dev/scripts/operate-relay-asia-admission.test.mjs @@ -0,0 +1,512 @@ +import assert from 'node:assert/strict' +import { createHash } from 'node:crypto' +import { test } from 'node:test' +import { operateRelayAsiaAdmission } from './operate-relay-asia-admission.mjs' + +const digest = `sha256:${'a'.repeat(64)}` +const membershipDigest = (membership) => + createHash('sha256').update(JSON.stringify(membership)).digest('hex') + +function harness(initialSelector) { + const initialMembership = structuredClone(initialSelector.membership) + let selector = structuredClone(initialSelector) + const intents = new Map() + const requests = [] + let fetches = 0 + let failAfterIntent = false + const post = async (url, body) => { + const parsed = new URL(url) + requests.push({ path: parsed.pathname, body }) + if (parsed.pathname === '/v1/admin/runtime-status') { + const cell = parsed.hostname.split('.')[0] + return { + cellId: `production-gce-${cell}`, + cellUrl: parsed.origin, + region: 'asia-east2', + imageDigest: digest, + draining: false, + connectionCapacity: { hardCap: 3_000, unobservedBound: 60 } + } + } + if (parsed.pathname === '/v1/admin/cell-status') { + return { + status: { + cellUrl: `https://${body.cellId.split('-').at(-1)}.relay.onorca.dev`, + runtime: { heartbeatFresh: true, ready: true } + } + } + } + if (parsed.pathname.endsWith('/status')) { + return { selector, intent: body.attemptId ? intents.get(body.attemptId) ?? null : null } + } + if (parsed.pathname.endsWith('/add-migration-cells')) { + selector = { + generation: selector.generation + 1, + attemptId: body.attemptId, + membership: { + ...selector.membership, + migrationOnly: [...selector.membership.migrationOnly, ...body.cells.map((cell) => cell.cellId)].sort() + } + } + } else if (parsed.pathname.endsWith('/apply-staging-asia-proof')) { + const membership = structuredClone(selector.membership) + membership.migrationOnly = membership.migrationOnly.filter((cell) => cell !== 'staging-gce-c4') + membership.general = membership.general.filter((cell) => cell !== 'staging-gce-c4') + membership[body.state === 'general' ? 'general' : 'migrationOnly'].push('staging-gce-c4') + selector = { generation: selector.generation + 1, attemptId: body.attemptId, membership } + } else if (parsed.pathname.endsWith('/apply')) { + if ( + body.expectedMembershipSha256 && + body.expectedMembershipSha256 !== membershipDigest(selector.membership) + ) throw new Error('admission_selector_membership_mismatch') + if (failAfterIntent) { + failAfterIntent = false + intents.set(body.attemptId, { + state: 'unchanged', + expectedGeneration: body.expectedGeneration, + previousMembership: structuredClone(selector.membership), + membership: structuredClone(body.membership) + }) + throw new Error('failure after intent persistence') + } + selector = { generation: selector.generation + 1, attemptId: body.attemptId, membership: body.membership } + } else throw new Error(`unexpected ${parsed.pathname}`) + intents.set(body.attemptId, { + state: 'committed', expectedGeneration: body.expectedGeneration, + previousMembership: initialSelector.membership, + membership: selector.membership + }) + return { changed: true, selector } + } + const fetch = async () => { + fetches++ + return new Response(null, { status: 200 }) + } + const commitWithoutResponse = async (path, body) => { + await post(`https://relay.onorca.dev${path}`, body) + throw new Error('response lost after commit') + } + return { + post, fetch, requests, commitWithoutResponse, + failNextApplyAfterIntent: () => (failAfterIntent = true), + apply: async (attemptId, membership) => await post( + 'https://relay.onorca.dev/v1/admin/admission-selector/apply', + { attemptId, expectedGeneration: selector.generation, membership } + ), + fetchCount: () => fetches, selector: () => selector + } +} + +const baseSelector = { + generation: 7, + membership: { existingOnly: [], migrationOnly: [], general: ['production-gce-c26'] } +} + +test('inspects generation zero without requiring target registration or making a mutation', async () => { + const subject = harness({ + generation: 0, + membership: { + existingOnly: ['staging-gce-c3'], + migrationOnly: [], + general: ['staging-gce-c1', 'staging-gce-c2'] + } + }) + const result = await operateRelayAsiaAdmission({ + environment: 'staging', mode: 'inspect', cells: ['staging-gce-c4'], + imageDigest: digest, token: 'not-logged' + }, subject) + assert.equal(result.generation, 0) + assert.equal(result.states['staging-gce-c4'], 'absent') + assert.deepEqual(result.membership, subject.selector().membership) + assert.equal(result.membershipSha256, membershipDigest(subject.selector().membership)) + assert.deepEqual(subject.requests.map(({ path }) => path), [ + '/v1/admin/admission-selector/status' + ]) +}) + +test('initializes generation zero without changing membership', async () => { + const membership = { + existingOnly: ['staging-gce-c3'], + migrationOnly: [], + general: ['staging-gce-c1', 'staging-gce-c2'] + } + const subject = harness({ generation: 0, membership }) + const result = await operateRelayAsiaAdmission({ + environment: 'staging', mode: 'initialize', cells: ['staging-gce-c4'], + expectedGeneration: 0, imageDigest: digest, + expectedMembershipSha256: membershipDigest(membership), + attemptId: 'asia_boundary_0', token: 'not-logged' + }, subject) + const request = subject.requests.find(({ path }) => path.endsWith('/apply')) + assert.deepEqual(request.body, { + v: 1, + attemptId: 'asia_boundary_0', + expectedGeneration: 0, + expectedMembershipSha256: membershipDigest(membership), + membership + }) + assert.equal(result.generation, 1) + assert.equal(result.states['staging-gce-c4'], 'absent') + assert.deepEqual(subject.selector().membership, membership) +}) + +test('retries the same fingerprint-bound initialization after intent persistence', async () => { + const membership = { + existingOnly: ['staging-gce-c3'], + migrationOnly: [], + general: ['staging-gce-c1', 'staging-gce-c2'] + } + const subject = harness({ generation: 0, membership }) + subject.failNextApplyAfterIntent() + const result = await operateRelayAsiaAdmission({ + environment: 'staging', mode: 'initialize', cells: ['staging-gce-c4'], + expectedGeneration: 0, imageDigest: digest, + expectedMembershipSha256: membershipDigest(membership), + attemptId: 'asia_boundary_intent_retry', token: 'not-logged' + }, subject) + const applies = subject.requests.filter(({ path }) => path.endsWith('/apply')) + assert.equal(applies.length, 2) + assert.deepEqual(applies[1].body, applies[0].body) + assert.equal(result.recovered, true) + assert.equal(result.generation, 1) + assert.deepEqual(subject.selector().membership, membership) +}) + +test('recovers a committed generation-zero initialization', async () => { + const membership = { + existingOnly: ['staging-gce-c3'], + migrationOnly: [], + general: ['staging-gce-c1', 'staging-gce-c2'] + } + const subject = harness({ generation: 0, membership }) + const config = { + environment: 'staging', mode: 'initialize', cells: ['staging-gce-c4'], + expectedGeneration: 0, imageDigest: digest, + expectedMembershipSha256: membershipDigest(membership), + attemptId: 'asia_boundary_retry', token: 'not-logged' + } + await assert.rejects(subject.commitWithoutResponse( + '/v1/admin/admission-selector/apply', + { + v: 1, + attemptId: config.attemptId, + expectedGeneration: 0, + expectedMembershipSha256: membershipDigest(membership), + membership + } + ), /response lost after commit/) + const recovered = await operateRelayAsiaAdmission(config, subject) + assert.equal(recovered.recovered, true) + assert.equal(recovered.generation, 1) + assert.deepEqual(subject.selector().membership, membership) +}) + +test('rejects generation-zero membership drift after inspect', async () => { + const inspected = { + existingOnly: ['staging-gce-c3'], + migrationOnly: [], + general: ['staging-gce-c1', 'staging-gce-c2'] + } + const changed = { + existingOnly: ['staging-gce-c2', 'staging-gce-c3'], + migrationOnly: [], + general: ['staging-gce-c1'] + } + const subject = harness({ generation: 0, membership: changed }) + await assert.rejects(operateRelayAsiaAdmission({ + environment: 'staging', mode: 'initialize', cells: ['staging-gce-c4'], + expectedGeneration: 0, imageDigest: digest, + expectedMembershipSha256: membershipDigest(inspected), + attemptId: 'asia_boundary_drift', token: 'not-logged' + }, subject), /membership changed/) + assert.equal(subject.requests.some(({ path }) => path.endsWith('/apply')), false) +}) + +test('registers all three Asia cells atomically with region and exact limits', async () => { + const subject = harness(baseSelector) + const result = await operateRelayAsiaAdmission({ + environment: 'production', mode: 'register', + cells: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'], + expectedGeneration: 7, imageDigest: digest, attemptId: 'asia_register_7', token: 'not-logged' + }, subject) + const request = subject.requests.find(({ path }) => path.endsWith('/add-migration-cells')) + assert.equal(request.body.cells.length, 3) + assert.ok(request.body.cells.every((cell) => + cell.region === 'asia-east2' && cell.capacityRequests === 6_000 && + cell.connectionHardCap === 3_000 && cell.connectionUnobservedBound === 60 + )) + assert.equal(result.generation, 8) + assert.deepEqual(new Set(Object.values(result.states)), new Set(['migration-only'])) +}) + +test('promotes the canary only after runtime and director-heartbeat checks', async () => { + const subject = harness({ + generation: 8, + membership: { existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] } + }) + const result = await operateRelayAsiaAdmission({ + environment: 'production', mode: 'promote', cells: ['production-gce-c27'], + expectedGeneration: 8, imageDigest: digest, attemptId: 'asia_promote_8', token: 'not-logged' + }, subject) + assert.equal(subject.fetchCount(), 2) + assert.equal(result.states['production-gce-c27'], 'general') +}) + +test('checks registered migration-only cells before director configuration without requiring heartbeat', async () => { + const subject = harness({ + generation: 8, + membership: { + existingOnly: [], + migrationOnly: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'], + general: ['production-gce-c26'] + } + }) + const result = await operateRelayAsiaAdmission({ + environment: 'production', mode: 'registered', + cells: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'], + expectedGeneration: 8, imageDigest: digest, token: 'not-logged' + }, subject) + assert.equal(subject.fetchCount(), 6) + assert.equal(subject.requests.filter(({ path }) => path === '/v1/admin/cell-status').length, 0) + assert.deepEqual(new Set(Object.values(result.states)), new Set(['migration-only'])) +}) + +test('rolls back admission without requiring an unhealthy runtime to answer', async () => { + const subject = harness({ + generation: 9, + membership: { existingOnly: [], migrationOnly: [], general: ['production-gce-c26', 'production-gce-c27'] } + }) + const result = await operateRelayAsiaAdmission({ + environment: 'production', mode: 'rollback', cells: ['production-gce-c27'], + expectedGeneration: 9, imageDigest: digest, attemptId: 'asia_rollback_9', token: 'not-logged' + }, subject) + assert.equal(subject.fetchCount(), 0) + assert.equal(result.states['production-gce-c27'], 'migration-only') +}) + +test('uses the server-enforced C4-only route for staging proof transitions', async () => { + const subject = harness({ + generation: 3, + membership: { + existingOnly: ['staging-gce-c1'], + migrationOnly: [], + general: ['staging-gce-c2', 'staging-gce-c4'] + } + }) + const result = await operateRelayAsiaAdmission({ + environment: 'staging', mode: 'rollback', cells: ['staging-gce-c4'], + expectedGeneration: 3, imageDigest: digest, attemptId: 'asia_staging_rollback', + token: 'not-logged' + }, subject) + const request = subject.requests.find( + ({ path }) => path.endsWith('/apply-staging-asia-proof') + ) + assert.deepEqual(request.body, { + v: 1, + attemptId: 'asia_staging_rollback', + expectedGeneration: 3, + state: 'migration-only' + }) + assert.deepEqual(subject.selector().membership, { + existingOnly: ['staging-gce-c1'], + migrationOnly: ['staging-gce-c4'], + general: ['staging-gce-c2'] + }) + assert.equal(result.states['staging-gce-c4'], 'migration-only') +}) + +test('fails closed when the exact selector generation moved', async () => { + const subject = harness(baseSelector) + await assert.rejects(operateRelayAsiaAdmission({ + environment: 'production', mode: 'register', + cells: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'], + expectedGeneration: 6, imageDigest: digest, attemptId: 'asia_register_6', token: 'not-logged' + }, subject), /generation changed/) + assert.equal(subject.fetchCount(), 0) +}) + +test('recovers a committed registration when the workflow retries the original generation', async () => { + const subject = harness(baseSelector) + const config = { + environment: 'production', mode: 'register', + cells: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'], + expectedGeneration: 7, imageDigest: digest, attemptId: 'asia_register_retry', + token: 'not-logged' + } + await assert.rejects(subject.commitWithoutResponse( + '/v1/admin/admission-selector/add-migration-cells', + { + v: 1, + attemptId: config.attemptId, + expectedGeneration: config.expectedGeneration, + cells: config.cells.map((cellId) => ({ cellId })) + } + ), /response lost after commit/) + const recovered = await operateRelayAsiaAdmission(config, subject) + assert.equal(recovered.recovered, true) + assert.equal(recovered.generation, 8) +}) + +test('recovers a committed promotion when the workflow retries the original generation', async () => { + const subject = harness({ + generation: 8, + membership: { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + } + }) + const config = { + environment: 'production', mode: 'promote', cells: ['production-gce-c27'], + expectedGeneration: 8, imageDigest: digest, attemptId: 'asia_promote_retry', + token: 'not-logged' + } + await assert.rejects(subject.commitWithoutResponse( + '/v1/admin/admission-selector/apply', + { + v: 1, + attemptId: config.attemptId, + expectedGeneration: config.expectedGeneration, + membership: { + existingOnly: [], migrationOnly: [], + general: ['production-gce-c26', 'production-gce-c27'] + } + } + ), /response lost after commit/) + const recovered = await operateRelayAsiaAdmission(config, subject) + assert.equal(recovered.recovered, true) + assert.equal(recovered.generation, 9) + assert.equal(recovered.states['production-gce-c27'], 'general') +}) + +test('inspects an ambiguous promotion without creating a new transition', async () => { + const untouched = harness({ + generation: 8, + membership: { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + } + }) + const config = { + environment: 'production', mode: 'recover-promotion', cells: ['production-gce-c27'], + expectedGeneration: 8, imageDigest: digest, attemptId: 'asia_recover_promote', + token: 'not-logged' + } + const absent = await operateRelayAsiaAdmission(config, untouched) + assert.equal(absent.promoted, false) + assert.equal(untouched.requests.some(({ path }) => path.endsWith('/apply')), false) + + const committed = harness({ + generation: 8, + membership: { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + } + }) + await assert.rejects(committed.commitWithoutResponse('/v1/admin/admission-selector/apply', { + v: 1, + attemptId: config.attemptId, + expectedGeneration: 8, + membership: { + existingOnly: [], migrationOnly: [], general: ['production-gce-c26', 'production-gce-c27'] + } + }), /response lost after commit/) + const recovered = await operateRelayAsiaAdmission(config, committed) + assert.equal(recovered.promoted, true) + assert.equal(recovered.generation, 9) +}) + +test('treats an already rolled-back promotion as recovered', async () => { + const subject = harness({ + generation: 8, + membership: { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + } + }) + const config = { + environment: 'production', mode: 'recover-promotion', cells: ['production-gce-c27'], + expectedGeneration: 8, imageDigest: digest, attemptId: 'asia_recover_after_rollback', + token: 'not-logged' + } + await assert.rejects(subject.commitWithoutResponse('/v1/admin/admission-selector/apply', { + v: 1, attemptId: config.attemptId, expectedGeneration: 8, + membership: { + existingOnly: [], migrationOnly: [], general: ['production-gce-c26', 'production-gce-c27'] + } + }), /response lost after commit/) + await subject.apply('later_rollback', { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + }) + const recovered = await operateRelayAsiaAdmission(config, subject) + assert.equal(recovered.promoted, false) + assert.equal(recovered.generation, 10) +}) + +test('recovers a committed rollback when the workflow retries the original generation', async () => { + const subject = harness({ + generation: 9, + membership: { + existingOnly: [], migrationOnly: [], general: ['production-gce-c26', 'production-gce-c27'] + } + }) + const config = { + environment: 'production', mode: 'rollback', cells: ['production-gce-c27'], + expectedGeneration: 9, imageDigest: digest, attemptId: 'asia_rollback_retry', + token: 'not-logged' + } + await assert.rejects(subject.commitWithoutResponse( + '/v1/admin/admission-selector/apply', + { + v: 1, + attemptId: config.attemptId, + expectedGeneration: config.expectedGeneration, + membership: { + existingOnly: [], migrationOnly: ['production-gce-c27'], + general: ['production-gce-c26'] + } + } + ), /response lost after commit/) + const recovered = await operateRelayAsiaAdmission(config, subject) + assert.equal(recovered.recovered, true) + assert.equal(recovered.generation, 10) + assert.equal(recovered.states['production-gce-c27'], 'migration-only') +}) + +test('rejects a committed transition retry after a later selector change', async () => { + const subject = harness({ + generation: 8, + membership: { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + } + }) + const config = { + environment: 'production', mode: 'promote', cells: ['production-gce-c27'], + expectedGeneration: 8, imageDigest: digest, attemptId: 'asia_stale_promote', + token: 'not-logged' + } + await assert.rejects(subject.commitWithoutResponse('/v1/admin/admission-selector/apply', { + v: 1, attemptId: config.attemptId, expectedGeneration: 8, + membership: { + existingOnly: [], migrationOnly: [], general: ['production-gce-c26', 'production-gce-c27'] + } + }), /response lost after commit/) + await subject.apply('later_rollback', { + existingOnly: [], migrationOnly: ['production-gce-c27'], general: ['production-gce-c26'] + }) + await assert.rejects( + operateRelayAsiaAdmission(config, subject), + /does not match the requested Asia transition/ + ) +}) + +test('requires the C27 canary before promoting C28 and C29', async () => { + const subject = harness({ + generation: 8, + membership: { + existingOnly: [], + migrationOnly: ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'], + general: ['production-gce-c26'] + } + }) + await assert.rejects(operateRelayAsiaAdmission({ + environment: 'production', mode: 'promote', + cells: ['production-gce-c28', 'production-gce-c29'], expectedGeneration: 8, + imageDigest: digest, attemptId: 'asia_wave_before_canary', token: 'not-logged' + }, subject), /C27 canary/) +}) diff --git a/cloud/dev/scripts/operate-relay-regional-rehome.mjs b/cloud/dev/scripts/operate-relay-regional-rehome.mjs new file mode 100644 index 00000000000..2ccce39926d --- /dev/null +++ b/cloud/dev/scripts/operate-relay-regional-rehome.mjs @@ -0,0 +1,312 @@ +import { pathToFileURL } from 'node:url' +import { inspectAdmissionSelector } from './relay-admission-selector.mjs' + +const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' +const MODES = new Set(['inspect', 'enable', 'pause', 'disable', 'recover-enable']) + +function canonicalCells(value) { + if (value === 'none') return [] + const cells = value.split(',').map((cell) => cell.trim()).filter(Boolean).sort() + if ( + cells.length === 0 || + new Set(cells).size !== cells.length || + cells.some((cell) => !/^production-gce-c(?:[1-9]|[12][0-9])$/.test(cell)) + ) throw new Error('selector membership is invalid') + return cells +} + +function integer(value, name, { minimum = 0, maximum = Number.MAX_SAFE_INTEGER } = {}) { + const parsed = Number(value) + if (!Number.isSafeInteger(parsed) || parsed < minimum || parsed > maximum) { + throw new Error(`${name} is invalid`) + } + return parsed +} + +export function parseRegionalRehomeArguments(argv, environment = process.env) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['mode', 'director-origin', 'expected-control-generation']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!MODES.has(values.mode)) throw new Error('--mode is invalid') + if (values['director-origin'] !== DIRECTOR_ORIGIN) { + throw new Error('--director-origin must be the production Relay origin') + } + const recovery = values.mode === 'recover-enable' + const mutation = values.mode !== 'inspect' && !recovery + const mutationKeys = [ + 'not-before', + 'rate-per-minute', + 'preference-max-age-ms', + 'drain-grace-ms', + 'confirmation' + ] + if (mutation && mutationKeys.some((key) => values[key] === undefined)) { + throw new Error('mutations require the complete durable control shape') + } + if (!mutation && !recovery && mutationKeys.some((key) => values[key] !== undefined)) { + throw new Error('inspect cannot carry mutation arguments') + } + const selectorKeys = [ + 'expected-selector-generation', + 'expected-existing-only-cells', + 'expected-migration-only-cells', + 'expected-general-cells' + ] + if (!recovery && selectorKeys.some((key) => values[key] === undefined)) { + throw new Error('operation requires exact selector state') + } + if (recovery && selectorKeys.some((key) => values[key] !== undefined)) { + throw new Error('enable recovery cannot depend on selector diagnostics') + } + const expectedConfirmation = { + enable: 'ENABLE_REGIONAL_REHOMING', + pause: 'PAUSE_REGIONAL_REHOMING', + disable: 'DISABLE_REGIONAL_REHOMING' + }[values.mode] + if (mutation && values.confirmation !== expectedConfirmation) { + throw new Error('confirmation does not match the requested control action') + } + if (recovery && values.confirmation !== 'RECOVER_FAILED_REGIONAL_REHOME_ENABLE') { + throw new Error('confirmation does not authorize failed-enable recovery') + } + if ( + recovery && + mutationKeys + .filter((key) => key !== 'confirmation') + .some((key) => values[key] !== undefined) + ) throw new Error('enable recovery cannot carry durable control shape arguments') + const token = environment.ORCA_RELAY_ADMIN_ID_TOKEN + if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') + return { + mode: values.mode, + directorOrigin: DIRECTOR_ORIGIN, + ...(!recovery + ? { + expectedSelectorGeneration: integer( + values['expected-selector-generation'], + '--expected-selector-generation' + ), + expectedMembership: { + existingOnly: canonicalCells(values['expected-existing-only-cells']), + migrationOnly: canonicalCells(values['expected-migration-only-cells']), + general: canonicalCells(values['expected-general-cells']) + } + } + : {}), + expectedControlGeneration: integer( + values['expected-control-generation'], + '--expected-control-generation' + ), + ...(mutation + ? { + notBefore: integer(values['not-before'], '--not-before'), + ratePerMinute: integer(values['rate-per-minute'], '--rate-per-minute', { + minimum: 1, + maximum: 120 + }), + preferenceMaxAgeMs: integer( + values['preference-max-age-ms'], + '--preference-max-age-ms', + { minimum: 60_000, maximum: 30 * 24 * 60 * 60_000 } + ), + drainGraceMs: integer(values['drain-grace-ms'], '--drain-grace-ms', { + minimum: 60_000, + maximum: 60 * 60_000 + }) + } + : {}), + token + } +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({})) + if (!response.ok) throw new Error(`${label} returned ${response.status}: ${body.error ?? 'unknown'}`) + return body +} + +function exactMembership(actual, expected) { + return ['existingOnly', 'migrationOnly', 'general'].every( + (key) => JSON.stringify(actual[key]) === JSON.stringify(expected[key]) + ) +} + +function assertControl(control, expected) { + if ( + (expected.generation !== undefined && control?.generation !== expected.generation) || + typeof control.enabled !== 'boolean' || + !Number.isSafeInteger(control.observationStartedAt) || + !Number.isSafeInteger(control.notBefore) || + !Number.isSafeInteger(control.ratePerMinute) || + !Number.isSafeInteger(control.preferenceMaxAgeMs) || + !Number.isSafeInteger(control.drainGraceMs) + ) throw new Error('director returned an invalid regional rehome control') + if (expected.enabled !== undefined && control.enabled !== expected.enabled) { + throw new Error('regional rehome enabled state does not match') + } + return control +} + +async function verifiedDisabledControl(post, generation) { + return assertControl((await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'inspect' + })).control, { generation, enabled: false }) +} + +async function applyDisabledControl(post, before) { + return assertControl((await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'apply', + expectedGeneration: before.generation, + enabled: false, + notBefore: before.notBefore, + ratePerMinute: before.ratePerMinute, + preferenceMaxAgeMs: before.preferenceMaxAgeMs, + drainGraceMs: before.drainGraceMs, + confirmation: 'DISABLE_REGIONAL_REHOMING' + })).control, { generation: before.generation + 1, enabled: false }) +} + +async function resolveAmbiguousDisable(post, before, firstError) { + const observed = assertControl((await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'inspect' + })).control, {}) + if (observed.generation === before.generation + 1 && !observed.enabled) { + return observed + } + if (observed.generation !== before.generation || !observed.enabled) { + throw new AggregateError( + [firstError], + 'failed-enable recovery reached an unexpected control generation' + ) + } + try { + return await applyDisabledControl(post, before) + } catch (retryError) { + try { + return await verifiedDisabledControl(post, before.generation + 1) + } catch (readbackError) { + throw new AggregateError( + [firstError, retryError, readbackError], + 'failed-enable recovery exhausted two bounded CAS attempts' + ) + } + } +} + +export async function recoverRegionalRehomeEnable(config, post) { + const before = assertControl((await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'inspect' + })).control, {}) + if ( + before.generation < config.expectedControlGeneration || + (before.generation === config.expectedControlGeneration && before.enabled) + ) throw new Error('durable control cannot belong to the failed enable attempt') + if (!before.enabled) { + const verified = await verifiedDisabledControl(post, before.generation) + return { mode: config.mode, recovered: false, control: verified } + } + let applied + try { + applied = await applyDisabledControl(post, before) + } catch (error) { + applied = await resolveAmbiguousDisable(post, before, error) + } + const verified = await verifiedDisabledControl(post, applied.generation) + return { mode: config.mode, recovered: true, control: verified } +} + +export async function operateRegionalRehome(config, dependencies = {}) { + const fetchImpl = dependencies.fetch ?? fetch + const post = dependencies.post ?? (async (path, body) => await responseJson( + await fetchImpl(`${config.directorOrigin}${path}`, { + method: 'POST', + headers: { + authorization: `Bearer ${config.token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }), + path + )) + if (config.mode === 'recover-enable') { + return await recoverRegionalRehomeEnable(config, post) + } + const selector = (await inspectAdmissionSelector(post)).selector + if ( + selector.generation !== config.expectedSelectorGeneration || + !exactMembership(selector.membership, config.expectedMembership) + ) throw new Error('admission selector does not match the reviewed generation and membership') + + const inspected = await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'inspect' + }) + const before = assertControl(inspected.control, { + generation: config.expectedControlGeneration + }) + if (config.mode === 'inspect') return { mode: config.mode, selector, control: before } + if (config.mode === 'enable' && before.enabled) { + throw new Error('regional rehome is already enabled; inspect before changing its rate') + } + if (config.mode === 'pause' && !before.enabled) { + throw new Error('regional rehome is already paused') + } + const enabled = config.mode === 'enable' + const applied = await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'apply', + expectedGeneration: config.expectedControlGeneration, + enabled, + notBefore: config.notBefore, + ratePerMinute: config.ratePerMinute, + preferenceMaxAgeMs: config.preferenceMaxAgeMs, + drainGraceMs: config.drainGraceMs, + confirmation: enabled + ? 'ENABLE_REGIONAL_REHOMING' + : 'DISABLE_REGIONAL_REHOMING' + }) + const after = assertControl(applied.control, { + generation: config.expectedControlGeneration + 1, + enabled + }) + const verified = assertControl((await post('/v1/admin/regional-rehome-control', { + v: 1, + action: 'inspect' + })).control, { + generation: after.generation, + enabled + }) + return { mode: config.mode, selector, control: verified } +} + +export async function main( + argv = process.argv.slice(2), + environment = process.env, + dependencies = {}, + write = (value) => process.stdout.write(value) +) { + const result = await operateRegionalRehome( + parseRegionalRehomeArguments(argv, environment), + dependencies + ) + write(`${JSON.stringify({ event: 'relay_regional_rehome_control', ...result })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs b/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs new file mode 100644 index 00000000000..8ffe38dfe09 --- /dev/null +++ b/cloud/dev/scripts/operate-relay-regional-rehome.test.mjs @@ -0,0 +1,265 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + main, + operateRegionalRehome, + parseRegionalRehomeArguments, + recoverRegionalRehomeEnable +} from './operate-relay-regional-rehome.mjs' + +const membership = { + existingOnly: ['production-gce-c1'], + migrationOnly: ['production-gce-c2'], + general: ['production-gce-c7', 'production-gce-c27'] +} + +function argumentsFor(mode, confirmation) { + return [ + '--mode', mode, + '--director-origin', 'https://relay.onorca.dev', + '--expected-selector-generation', '11', + '--expected-existing-only-cells', membership.existingOnly.join(','), + '--expected-migration-only-cells', membership.migrationOnly.join(','), + '--expected-general-cells', membership.general.join(','), + '--expected-control-generation', '4', + ...(mode === 'inspect' ? [] : [ + '--not-before', '2000000000000', + '--rate-per-minute', '10', + '--preference-max-age-ms', '86400000', + '--drain-grace-ms', '60000', + '--confirmation', confirmation + ]) + ] +} + +function control(generation, enabled) { + return { + generation, + enabled, + observationStartedAt: 1, + notBefore: 2_000_000_000_000, + ratePerMinute: 10, + preferenceMaxAgeMs: 86_400_000, + drainGraceMs: 60_000 + } +} + +test('parses exact selector and typed control confirmation', () => { + const parsed = parseRegionalRehomeArguments( + argumentsFor('enable', 'ENABLE_REGIONAL_REHOMING'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + assert.equal(parsed.expectedSelectorGeneration, 11) + assert.equal(parsed.expectedControlGeneration, 4) + assert.equal(parsed.ratePerMinute, 10) + assert.throws( + () => parseRegionalRehomeArguments( + argumentsFor('pause', 'DISABLE_REGIONAL_REHOMING'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ), + /confirmation/ + ) + assert.throws( + () => parseRegionalRehomeArguments( + argumentsFor('inspect').concat('--rate-per-minute', '10'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ), + /inspect cannot/ + ) +}) + +test('binds enable to exact selector and durable control generations', async () => { + const requests = [] + const controls = [ + { generation: 4, enabled: false }, + { generation: 5, enabled: true }, + { generation: 5, enabled: true } + ].map((control) => ({ + observationStartedAt: 1, + notBefore: 0, + ratePerMinute: 10, + preferenceMaxAgeMs: 86_400_000, + drainGraceMs: 60_000, + ...control + })) + const config = parseRegionalRehomeArguments( + argumentsFor('enable', 'ENABLE_REGIONAL_REHOMING'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + const result = await operateRegionalRehome(config, { + post: async (path, body) => { + requests.push({ path, body }) + if (path === '/v1/admin/admission-selector/status') { + return { selector: { generation: 11, membership } } + } + return { v: 1, control: controls.shift() } + } + }) + assert.equal(result.control.generation, 5) + assert.deepEqual(requests[2].body, { + v: 1, + action: 'apply', + expectedGeneration: 4, + enabled: true, + notBefore: 2_000_000_000_000, + ratePerMinute: 10, + preferenceMaxAgeMs: 86_400_000, + drainGraceMs: 60_000, + confirmation: 'ENABLE_REGIONAL_REHOMING' + }) +}) + +test('fails closed on selector drift before reading or mutating control', async () => { + let calls = 0 + const config = parseRegionalRehomeArguments( + argumentsFor('disable', 'DISABLE_REGIONAL_REHOMING'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + await assert.rejects( + operateRegionalRehome(config, { + post: async () => { + calls += 1 + return { selector: { generation: 12, membership } } + } + }), + /selector/ + ) + assert.equal(calls, 1) +}) + +test('failed-enable recovery CAS-disables an advanced enabled generation', async () => { + const requests = [] + let current = control(7, true) + const result = await recoverRegionalRehomeEnable({ + mode: 'recover-enable', + expectedControlGeneration: 4 + }, async (_path, body) => { + requests.push(body) + if (body.action === 'inspect') return { control: current } + assert.equal(body.expectedGeneration, 7) + current = control(8, false) + throw new Error('enable recovery response was lost') + }) + assert.equal(result.recovered, true) + assert.deepEqual(result.control, control(8, false)) + assert.deepEqual(requests[1], { + v: 1, + action: 'apply', + expectedGeneration: 7, + enabled: false, + notBefore: 2_000_000_000_000, + ratePerMinute: 10, + preferenceMaxAgeMs: 86_400_000, + drainGraceMs: 60_000, + confirmation: 'DISABLE_REGIONAL_REHOMING' + }) +}) + +test('failed-enable recovery retries once when the first CAS never commits', async () => { + let current = control(7, true) + const applyRequests = [] + const result = await recoverRegionalRehomeEnable({ + mode: 'recover-enable', + expectedControlGeneration: 4 + }, async (_path, body) => { + if (body.action === 'inspect') return { control: current } + applyRequests.push(body) + if (applyRequests.length === 1) { + throw new Error('disable request was lost before commit') + } + current = control(8, false) + throw new Error('retry response was lost after commit') + }) + assert.equal(applyRequests.length, 2) + assert.deepEqual(applyRequests[1], applyRequests[0]) + assert.equal(result.recovered, true) + assert.deepEqual(result.control, control(8, false)) +}) + +test('failed-enable recovery stops after two uncommitted CAS attempts', async () => { + let applyCalls = 0 + await assert.rejects( + recoverRegionalRehomeEnable({ + mode: 'recover-enable', + expectedControlGeneration: 4 + }, async (_path, body) => { + if (body.action === 'inspect') return { control: control(7, true) } + applyCalls += 1 + throw new Error(`disable attempt ${applyCalls} was lost before commit`) + }), + /exhausted two bounded CAS attempts/ + ) + assert.equal(applyCalls, 2) +}) + +test('failed-enable recovery is a verified no-op before enable and after cleanup', async () => { + for (const current of [control(4, false), control(8, false)]) { + const requests = [] + const result = await recoverRegionalRehomeEnable({ + mode: 'recover-enable', + expectedControlGeneration: 4 + }, async (_path, body) => { + requests.push(body) + return { control: current } + }) + assert.equal(result.recovered, false) + assert.equal(result.control.enabled, false) + assert.deepEqual(requests.map(({ action }) => action), ['inspect', 'inspect']) + } +}) + +test('failed-enable recovery rejects an unchanged pre-existing enabled state', async () => { + await assert.rejects( + recoverRegionalRehomeEnable({ + mode: 'recover-enable', + expectedControlGeneration: 4 + }, async () => ({ control: control(4, true) })), + /cannot belong to the failed enable attempt/ + ) +}) + +test('parses recovery without depending on selector diagnostics', () => { + const recoveryArguments = [ + '--mode', 'recover-enable', + '--director-origin', 'https://relay.onorca.dev', + '--expected-control-generation', '4', + '--confirmation', 'RECOVER_FAILED_REGIONAL_REHOME_ENABLE' + ] + const parsed = parseRegionalRehomeArguments( + recoveryArguments, + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ) + assert.equal(parsed.expectedControlGeneration, 4) + assert.equal(parsed.expectedMembership, undefined) + assert.throws( + () => parseRegionalRehomeArguments( + recoveryArguments.concat('--not-before', '2000000000000'), + { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' } + ), + /cannot carry durable control shape/ + ) +}) + +test('main executes recovery mode and emits verified disabled control', async () => { + let current = control(5, true) + let output = '' + await main([ + '--mode', 'recover-enable', + '--director-origin', 'https://relay.onorca.dev', + '--expected-control-generation', '4', + '--confirmation', 'RECOVER_FAILED_REGIONAL_REHOME_ENABLE' + ], { ORCA_RELAY_ADMIN_ID_TOKEN: 'token' }, { + post: async (_path, body) => { + if (body.action === 'apply') current = control(6, false) + return { control: current } + } + }, (value) => { + output += value + }) + assert.deepEqual(JSON.parse(output), { + event: 'relay_regional_rehome_control', + mode: 'recover-enable', + recovered: true, + control: control(6, false) + }) +}) diff --git a/cloud/dev/scripts/power-staging-relay.mjs b/cloud/dev/scripts/power-staging-relay.mjs new file mode 100644 index 00000000000..55704294192 --- /dev/null +++ b/cloud/dev/scripts/power-staging-relay.mjs @@ -0,0 +1,487 @@ +import { readFileSync } from 'node:fs' +import { spawnSync } from 'node:child_process' +import { pathToFileURL } from 'node:url' +import { + inspectAdmissionSelector, + selectorCellState +} from './relay-admission-selector.mjs' + +const PROJECT = 'onorca-cloud-staging' +const REGION = 'us-central1' +const DIRECTOR_ORIGIN = 'https://relay-staging.onorca.dev' +const ADMIN_AUDIENCE = `${DIRECTOR_ORIGIN}/v1/admin/drain` +const SQL_INSTANCE = 'orca-cloud-staging-auth-db' +const CLOUD_RUN_SERVICES = [ + { name: 'orca-cloud-relay-staging', healthOrigin: DIRECTOR_ORIGIN }, + { name: 'orca-cloud-auth-staging', healthOrigin: 'https://auth-staging.onorca.dev' } +] +const POLL_INTERVAL_MS = 5_000 +const WAKE_TIMEOUT_MS = 12 * 60 * 1_000 + +export function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error(`invalid argument ${key ?? ''}`) + const name = key.slice(2) + if (!['mode', 'wake-cells', 'topology-file'].includes(name)) { + throw new Error(`unsupported argument --${name}`) + } + values[name] = value + } + if (!['status', 'sleep', 'wake'].includes(values.mode)) { + throw new Error('--mode must be status, sleep, or wake') + } + if (!['configured', 'all'].includes(values['wake-cells'])) { + throw new Error('--wake-cells must be configured or all') + } + if (!values['topology-file']) throw new Error('missing --topology-file') + return { + mode: values.mode, + wakeCells: values['wake-cells'], + topologyFile: values['topology-file'] + } +} + +function canonicalStagingCell(cellId, value) { + if (!/^staging-gce-[a-z0-9-]+$/.test(cellId)) throw new Error(`unsafe staging cell ID ${cellId}`) + if (!value || typeof value !== 'object') throw new Error(`missing topology for ${cellId}`) + const cell = { + cellId, + migName: String(value.mig_name ?? ''), + zone: String(value.zone ?? ''), + origin: String(value.origin ?? ''), + initiallyEnabled: value.initially_enabled + } + if (!/^orca-cloud-staging-relay-gce-[a-z0-9-]+$/.test(cell.migName)) { + throw new Error(`${cellId} has an unsafe MIG name`) + } + if (!/^(?:us-central1|asia-east2)-[a-z]$/.test(cell.zone)) { + throw new Error(`${cellId} has an unsafe zone`) + } + const origin = new URL(cell.origin) + if ( + origin.protocol !== 'https:' || + origin.origin !== cell.origin || + !origin.hostname.endsWith('.relay-staging.onorca.dev') + ) { + throw new Error(`${cellId} has an unsafe origin`) + } + if (typeof cell.initiallyEnabled !== 'boolean') { + throw new Error(`${cellId} has no initial admission state`) + } + return cell +} + +export function readStagingTopology(file) { + const parsed = JSON.parse(readFileSync(file, 'utf8')) + if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed)) { + throw new Error('staging topology must be an object') + } + const cells = Object.entries(parsed) + .map(([cellId, value]) => canonicalStagingCell(cellId, value)) + .sort((left, right) => left.cellId.localeCompare(right.cellId)) + if (cells.length < 2 || cells.length > 10) { + throw new Error('staging topology must contain 2..10 cells') + } + if (cells.filter((cell) => cell.initiallyEnabled).length < 2) { + throw new Error('staging topology must retain two configured admission cells') + } + return cells +} + +function defaultCommand(args, json) { + const result = spawnSync('gcloud', args, { + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'] + }) + if (result.status !== 0) { + throw new Error(`gcloud ${args.slice(0, 5).join(' ')} failed: ${result.stderr.trim()}`) + } + return json ? JSON.parse(result.stdout) : result.stdout.trim() +} + +function suppliedAdminToken(environment = process.env) { + const token = environment.ORCA_RELAY_ADMIN_ID_TOKEN + if (!token || token.length > 8_192 || !/^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/.test(token)) { + throw new Error('workflow did not supply a valid masked staging admin token') + } + return token +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({ error: `http_${response.status}` })) + if (!response.ok) throw new Error(`${label} failed: ${body.error ?? response.status}`) + return body +} + +function createAdminPost(deps) { + const token = deps.adminToken() + return async (origin, path, body) => + await responseJson( + await deps.fetch(`${origin}${path}`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }), + path + ) +} + +async function waitUntil(deps, label, operation, timeoutMs = WAKE_TIMEOUT_MS) { + const deadline = deps.now() + timeoutMs + let lastError + while (deps.now() < deadline) { + try { + const result = await operation() + if (result) return result + } catch (error) { + lastError = error + } + await deps.wait(POLL_INTERVAL_MS) + } + const detail = lastError instanceof Error ? `: ${lastError.message}` : '' + throw new Error(`timed out waiting for ${label}${detail}`) +} + +async function checkHealth(deps, origin, path) { + const response = await deps.fetch(`${origin}${path}`, { signal: AbortSignal.timeout(15_000) }) + const body = await response.json().catch(() => ({})) + return response.ok && body.ok === true +} + +function sqlActivationPolicy(instance) { + return String(instance.settings?.activationPolicy ?? '') +} + +function describeSql(deps) { + return deps.commandJson([ + 'sql', + 'instances', + 'describe', + SQL_INSTANCE, + '--project', + PROJECT, + '--format=json' + ]) +} + +async function ensureSqlPolicy(deps, policy) { + if (sqlActivationPolicy(describeSql(deps)) === policy) return false + deps.command([ + 'sql', + 'instances', + 'patch', + SQL_INSTANCE, + '--project', + PROJECT, + `--activation-policy=${policy}`, + '--quiet' + ]) + await waitUntil(deps, `Cloud SQL activation policy ${policy}`, () => + sqlActivationPolicy(describeSql(deps)) === policy + ) + return true +} + +function describeMig(deps, cell) { + return deps.commandJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + cell.migName, + '--project', + PROJECT, + '--zone', + cell.zone, + '--format=json' + ]) +} + +async function setMigSize(deps, cell, size) { + const before = describeMig(deps, cell) + if (Number(before.targetSize) !== size) { + deps.command([ + 'compute', + 'instance-groups', + 'managed', + 'resize', + cell.migName, + '--project', + PROJECT, + '--zone', + cell.zone, + `--size=${size}`, + '--quiet' + ]) + } + await waitUntil(deps, `${cell.cellId} size ${size}`, () => { + const current = describeMig(deps, cell) + return Number(current.targetSize) === size && current.status?.isStable === true + }) +} + +function activeRevisionName(service) { + const active = (service.status?.traffic ?? []).filter((entry) => Number(entry.percent ?? 0) > 0) + if (active.length !== 1 || Number(active[0].percent) !== 100 || !active[0].revisionName) { + throw new Error('Cloud Run service must have exactly one active revision') + } + return active[0].revisionName +} + +function describeRunService(deps, name) { + return deps.commandJson([ + 'run', + 'services', + 'describe', + name, + '--project', + PROJECT, + '--region', + REGION, + '--format=json' + ]) +} + +function describeRunRevision(deps, name) { + return deps.commandJson([ + 'run', + 'revisions', + 'describe', + name, + '--project', + PROJECT, + '--region', + REGION, + '--format=json' + ]) +} + +function revisionMinimum(revision) { + return Number(revision.metadata?.annotations?.['autoscaling.knative.dev/minScale'] ?? 0) +} + +async function ensureCloudRunScaleToZero(deps, service) { + const before = describeRunService(deps, service.name) + const activeRevision = describeRunRevision(deps, activeRevisionName(before)) + if (revisionMinimum(activeRevision) === 0) return false + + const latestName = before.status?.latestReadyRevisionName + const latest = latestName ? describeRunRevision(deps, latestName) : null + if (!latest || revisionMinimum(latest) !== 0) { + deps.command([ + 'run', + 'services', + 'update', + service.name, + '--project', + PROJECT, + '--region', + REGION, + '--min-instances=0', + '--quiet' + ]) + } + deps.command([ + 'run', + 'services', + 'update-traffic', + service.name, + '--project', + PROJECT, + '--region', + REGION, + '--to-latest', + '--quiet' + ]) + await waitUntil(deps, `${service.name} scale-to-zero revision`, async () => { + const current = describeRunService(deps, service.name) + const revision = describeRunRevision(deps, activeRevisionName(current)) + return revisionMinimum(revision) === 0 && (await checkHealth(deps, service.healthOrigin, '/health')) + }) + return true +} + +async function cellStatus(adminPost, cell) { + const response = await adminPost(DIRECTOR_ORIGIN, '/v1/admin/cell-status', { + v: 1, + cellId: cell.cellId + }) + if (!response.status || response.status.cellId !== cell.cellId) { + throw new Error(`${cell.cellId} returned an invalid status`) + } + return response.status +} + +function assertQuiescent(status) { + const active = { + activityLeases: status.activityLeases, + activityRequestUnits: status.activityRequestUnits, + outgoingMigrations: status.outgoingMigrations, + incomingMigrations: status.incomingMigrations, + observedRequests: status.runtime?.observedRequests ?? 0 + } + if (Object.values(active).some((value) => Number(value) !== 0)) { + throw new Error(`${status.cellId} still has active Relay work: ${JSON.stringify(active)}`) + } +} + +async function setCellState(adminPost, cell, enabled) { + await adminPost(DIRECTOR_ORIGIN, '/v1/admin/cell-state', { + v: 1, + cellId: cell.cellId, + enabled + }) +} + +async function stagingStatus(deps, cells) { + return { + event: 'staging_relay_power_status', + project: PROJECT, + sqlActivationPolicy: sqlActivationPolicy(describeSql(deps)), + cells: cells.map((cell) => ({ + cellId: cell.cellId, + initiallyEnabled: cell.initiallyEnabled, + targetSize: Number(describeMig(deps, cell).targetSize) + })), + cloudRun: CLOUD_RUN_SERVICES.map((service) => { + const described = describeRunService(deps, service.name) + const revision = describeRunRevision(deps, activeRevisionName(described)) + return { service: service.name, activeRevisionMinimum: revisionMinimum(revision) } + }) + } +} + +async function sleepStaging(deps, cells) { + const sqlPolicy = sqlActivationPolicy(describeSql(deps)) + if (sqlPolicy === 'NEVER') { + const runningCells = cells.filter((cell) => Number(describeMig(deps, cell).targetSize) !== 0) + if (runningCells.length > 0) { + // SQL-off plus running workers is an unknown partial state; never kill those workers blindly. + throw new Error( + `staging is partially asleep with running cells: ${runningCells.map((cell) => cell.cellId).join(', ')}` + ) + } + deps.emit({ event: 'staging_relay_sleep_reconciled', alreadyAsleep: true }) + return + } + + const adminPost = createAdminPost(deps) + const initial = await Promise.all(cells.map((cell) => cellStatus(adminPost, cell))) + for (const status of initial) assertQuiescent(status) + const selector = await inspectAdmissionSelector( + async (path, body) => await adminPost(DIRECTOR_ORIGIN, path, body) + ) + if (selector.selector.generation > 0) { + throw new Error( + 'staging sleep cannot reverse the monotonic admission selector; keep staging awake' + ) + } + const previouslyEnabled = new Set(initial.filter((status) => status.enabled).map((status) => status.cellId)) + + for (const cell of cells) await setCellState(adminPost, cell, false) + await deps.wait(15_000) + try { + const disabled = await Promise.all(cells.map((cell) => cellStatus(adminPost, cell))) + for (const status of disabled) { + if (status.enabled) throw new Error(`${status.cellId} admission did not disable`) + assertQuiescent(status) + } + for (const service of CLOUD_RUN_SERVICES) await ensureCloudRunScaleToZero(deps, service) + const final = await Promise.all(cells.map((cell) => cellStatus(adminPost, cell))) + for (const status of final) assertQuiescent(status) + } catch (error) { + for (const cell of cells.filter((candidate) => previouslyEnabled.has(candidate.cellId))) { + await setCellState(adminPost, cell, true).catch(() => undefined) + } + throw error + } + + await Promise.all(cells.map((cell) => setMigSize(deps, cell, 0))) + await ensureSqlPolicy(deps, 'NEVER') + deps.emit({ event: 'staging_relay_slept', stoppedCells: cells.map((cell) => cell.cellId) }) +} + +async function wakeStaging(deps, cells, wakeCells) { + await ensureSqlPolicy(deps, 'ALWAYS') + await waitUntil(deps, 'staging director health', () => checkHealth(deps, DIRECTOR_ORIGIN, '/health')) + for (const service of CLOUD_RUN_SERVICES) await ensureCloudRunScaleToZero(deps, service) + + const adminPost = createAdminPost(deps) + const selector = await inspectAdmissionSelector( + async (path, body) => await adminPost(DIRECTOR_ORIGIN, path, body) + ) + const selectorActive = selector.selector.generation > 0 + // Existing-only cells may still own live or dormant assignments, so a + // selector-era wake restores the complete retained topology. + const selected = selectorActive + ? cells + : cells.filter((cell) => wakeCells === 'all' || cell.initiallyEnabled) + await Promise.all( + cells.map((cell) => setMigSize(deps, cell, selected.includes(cell) ? 1 : 0)) + ) + for (const cell of selected) { + await waitUntil(deps, `${cell.cellId} health`, () => checkHealth(deps, cell.origin, '/health')) + await waitUntil(deps, `${cell.cellId} readiness`, () => checkHealth(deps, cell.origin, '/ready')) + } + + for (const cell of cells) { + if (selectorActive) { + const status = await waitUntil(deps, `${cell.cellId} authenticated heartbeat`, async () => { + const current = await cellStatus(adminPost, cell) + return current.runtime?.heartbeatFresh && current.runtime.ready ? current : null + }) + if (status.admissionState !== selectorCellState(selector.selector, cell.cellId)) { + throw new Error(`${cell.cellId} admission does not match selector`) + } + continue + } + if (!selected.includes(cell)) { + await setCellState(adminPost, cell, false) + continue + } + const status = await waitUntil(deps, `${cell.cellId} authenticated heartbeat`, async () => { + const current = await cellStatus(adminPost, cell) + return current.runtime?.heartbeatFresh && current.runtime.ready ? current : null + }) + if (cell.initiallyEnabled && !status.enabled) await setCellState(adminPost, cell, true) + if (!cell.initiallyEnabled && status.enabled) await setCellState(adminPost, cell, false) + } + deps.emit({ + event: 'staging_relay_woke', + runningCells: selected.map((cell) => cell.cellId), + admissionCells: selectorActive + ? selector.selector.membership.general + : selected.filter((cell) => cell.initiallyEnabled).map((cell) => cell.cellId) + }) +} + +export async function runStagingRelayPower(config, overrides = {}) { + const deps = { + command: overrides.command ?? ((args) => defaultCommand(args, false)), + commandJson: overrides.commandJson ?? ((args) => defaultCommand(args, true)), + fetch: overrides.fetch ?? fetch, + adminToken: overrides.adminToken ?? suppliedAdminToken, + emit: overrides.emit ?? ((event) => process.stdout.write(`${JSON.stringify(event)}\n`)), + now: overrides.now ?? Date.now, + wait: overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))) + } + const cells = readStagingTopology(config.topologyFile) + if (config.mode === 'status') deps.emit(await stagingStatus(deps, cells)) + else if (config.mode === 'sleep') await sleepStaging(deps, cells) + else await wakeStaging(deps, cells, config.wakeCells) +} + +export async function main(argv = process.argv.slice(2)) { + await runStagingRelayPower(parseArguments(argv)) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/power-staging-relay.test.mjs b/cloud/dev/scripts/power-staging-relay.test.mjs new file mode 100644 index 00000000000..21c7794d48e --- /dev/null +++ b/cloud/dev/scripts/power-staging-relay.test.mjs @@ -0,0 +1,359 @@ +import assert from 'node:assert/strict' +import { mkdtempSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { test } from 'node:test' +import { + parseArguments, + readStagingTopology, + runStagingRelayPower +} from './power-staging-relay.mjs' + +function topologyFile(overrides = {}) { + const directory = mkdtempSync(join(tmpdir(), 'staging-relay-power-')) + const topology = { + 'staging-gce-c1': { + mig_name: 'orca-cloud-staging-relay-gce-c1', + zone: 'us-central1-b', + origin: 'https://c1.relay-staging.onorca.dev', + initially_enabled: true + }, + 'staging-gce-c2': { + mig_name: 'orca-cloud-staging-relay-gce-c2', + zone: 'us-central1-c', + origin: 'https://c2.relay-staging.onorca.dev', + initially_enabled: true + }, + 'staging-gce-c3': { + mig_name: 'orca-cloud-staging-relay-gce-c3', + zone: 'us-central1-a', + origin: 'https://c3.relay-staging.onorca.dev', + initially_enabled: false + }, + 'staging-gce-c4': { + mig_name: 'orca-cloud-staging-relay-gce-c4', + zone: 'asia-east2-a', + origin: 'https://c4.relay-staging.onorca.dev', + initially_enabled: false + }, + ...overrides + } + const file = join(directory, 'topology.json') + writeFileSync(file, JSON.stringify(topology)) + return file +} + +function argumentConfig(file, mode, wakeCells = 'configured') { + return parseArguments([ + '--mode', + mode, + '--wake-cells', + wakeCells, + '--topology-file', + file + ]) +} + +function response(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' } + }) +} + +function harness({ + sqlPolicy = 'ALWAYS', + migSize = 1, + observedRequests = 0, + selectorGeneration = 0 +} = {}) { + const cells = new Map( + ['staging-gce-c1', 'staging-gce-c2', 'staging-gce-c3', 'staging-gce-c4'].map((cellId, index) => [ + cellId, + { + enabled: index < 2, + targetSize: migSize, + observedRequests, + initiallyEnabled: index < 2 + } + ]) + ) + const revisions = new Map([ + ['orca-cloud-relay-staging', { active: 'relay-00001', latest: 'relay-00001', min: 1 }], + ['orca-cloud-auth-staging', { active: 'auth-00001', latest: 'auth-00001', min: 1 }] + ]) + const revisionMinimums = new Map([ + ['relay-00001', 1], + ['auth-00001', 1] + ]) + const commands = [] + const events = [] + let activationPolicy = sqlPolicy + let clock = 0 + + function cellForMig(name) { + return [...cells.entries()].find(([, value], index) => { + const suffix = `c${index + 1}` + return name.endsWith(suffix) && value + }) + } + + function commandJson(args) { + if (args[0] === 'sql') return { settings: { activationPolicy } } + if (args[0] === 'compute') { + const entry = cellForMig(args[4]) + return { targetSize: entry[1].targetSize, status: { isStable: true } } + } + if (args[0] === 'run' && args[1] === 'services') { + const state = revisions.get(args[3]) + return { + status: { + latestReadyRevisionName: state.latest, + traffic: [{ percent: 100, revisionName: state.active }] + } + } + } + if (args[0] === 'run' && args[1] === 'revisions') { + return { + metadata: { + annotations: { + 'autoscaling.knative.dev/minScale': String(revisionMinimums.get(args[3]) ?? 0) + } + } + } + } + throw new Error(`unexpected JSON command ${args.join(' ')}`) + } + + function command(args) { + commands.push(args) + if (args[0] === 'sql') { + activationPolicy = args.find((arg) => arg.startsWith('--activation-policy='))?.split('=')[1] + return + } + if (args[0] === 'compute') { + const entry = cellForMig(args[4]) + entry[1].targetSize = Number(args.find((arg) => arg.startsWith('--size='))?.split('=')[1]) + return + } + if (args[0] === 'run' && args[1] === 'services' && args[2] === 'update') { + const state = revisions.get(args[3]) + state.latest = `${args[3]}-power` + revisionMinimums.set(state.latest, 0) + return + } + if (args[0] === 'run' && args[1] === 'services' && args[2] === 'update-traffic') { + const state = revisions.get(args[3]) + state.active = state.latest + return + } + throw new Error(`unexpected command ${args.join(' ')}`) + } + + async function fetchImpl(url, options = {}) { + const parsed = new URL(url) + if (!options.method) return response({ ok: true }) + const body = JSON.parse(options.body) + if (parsed.pathname === '/v1/admin/cell-status') { + const state = cells.get(body.cellId) + const index = [...cells.keys()].indexOf(body.cellId) + return response({ + v: 1, + status: { + cellId: body.cellId, + enabled: state.enabled, + admissionState: + selectorGeneration > 0 + ? index === 0 + ? 'existing-only' + : index === 1 + ? 'migration-only' + : index === 2 + ? 'general' + : 'migration-only' + : state.enabled + ? 'general' + : 'existing-only', + assignments: 0, + reservedRequests: 0, + activityLeases: 0, + activityRequestUnits: 0, + outgoingMigrations: 0, + incomingMigrations: 0, + runtime: { + ready: state.targetSize === 1, + heartbeatFresh: state.targetSize === 1, + observedRequests: state.observedRequests + } + } + }) + } + if (parsed.pathname === '/v1/admin/admission-selector/status') { + return response({ + v: 1, + selector: { + generation: selectorGeneration, + attemptId: null, + membership: { + existingOnly: + selectorGeneration > 0 + ? ['staging-gce-c1'] + : [...cells] + .filter(([, state]) => !state.enabled) + .map(([cellId]) => cellId), + migrationOnly: + selectorGeneration > 0 ? ['staging-gce-c2', 'staging-gce-c4'] : [], + general: + selectorGeneration > 0 + ? ['staging-gce-c3'] + : [...cells] + .filter(([, state]) => state.enabled) + .map(([cellId]) => cellId) + } + }, + intent: null + }) + } + if (parsed.pathname === '/v1/admin/cell-state') { + cells.get(body.cellId).enabled = body.enabled + return response({ ok: true }) + } + throw new Error(`unexpected fetch ${parsed.pathname}`) + } + + return { + cells, + commands, + events, + deps: { + command, + commandJson, + fetch: fetchImpl, + adminToken: () => 'header.payload.signature', + emit: (event) => events.push(event), + now: () => clock, + wait: async (ms) => { + clock += ms + } + }, + sqlPolicy: () => activationPolicy + } +} + +test('accepts only explicit staging power arguments and topology', () => { + const file = topologyFile() + assert.equal(argumentConfig(file, 'status').mode, 'status') + assert.equal(readStagingTopology(file).at(-1).zone, 'asia-east2-a') + assert.throws(() => argumentConfig(file, 'destroy')) + assert.throws(() => parseArguments(['--mode', 'sleep', '--wake-cells', 'configured'])) + + const unsafe = topologyFile({ + 'staging-gce-c1': { + mig_name: 'orca-cloud-relay-gce-c1', + zone: 'us-central1-a', + origin: 'https://c1.relay.onorca.dev', + initially_enabled: true + } + }) + assert.throws(() => readStagingTopology(unsafe), /unsafe/) + + const unreviewedRegion = topologyFile({ + 'staging-gce-c4': { + mig_name: 'orca-cloud-staging-relay-gce-c4', + zone: 'europe-west1-b', + origin: 'https://c4.relay-staging.onorca.dev', + initially_enabled: false + } + }) + assert.throws(() => readStagingTopology(unreviewedRegion), /unsafe zone/) +}) + +test('refuses sleep before changing admission when a cell has active requests', async () => { + const testHarness = harness({ observedRequests: 1 }) + await assert.rejects( + runStagingRelayPower(argumentConfig(topologyFile(), 'sleep'), testHarness.deps), + /still has active Relay work/ + ) + assert.equal(testHarness.commands.length, 0) + assert.equal(testHarness.cells.get('staging-gce-c1').enabled, true) +}) + +test('sleeps only after disabling admission and proving zero active work', async () => { + const testHarness = harness() + await runStagingRelayPower(argumentConfig(topologyFile(), 'sleep'), testHarness.deps) + + assert.equal(testHarness.sqlPolicy(), 'NEVER') + assert.deepEqual([...testHarness.cells.values()].map((cell) => cell.targetSize), [0, 0, 0, 0]) + assert.deepEqual([...testHarness.cells.values()].map((cell) => cell.enabled), [false, false, false, false]) + assert.equal(testHarness.events.at(-1).event, 'staging_relay_slept') + assert.equal( + testHarness.commands.filter((args) => args[0] === 'run' && args[2] === 'update').length, + 2 + ) +}) + +test('refuses staging sleep after the monotonic selector boundary', async () => { + const testHarness = harness({ selectorGeneration: 1 }) + await assert.rejects( + runStagingRelayPower(argumentConfig(topologyFile(), 'sleep'), testHarness.deps), + /cannot reverse the monotonic admission selector/ + ) + assert.deepEqual([...testHarness.cells.values()].map((cell) => cell.targetSize), [1, 1, 1, 1]) +}) + +test('refuses to terminate workers from an unknown partially asleep state', async () => { + const testHarness = harness({ sqlPolicy: 'NEVER', migSize: 1 }) + await assert.rejects( + runStagingRelayPower(argumentConfig(topologyFile(), 'sleep'), testHarness.deps), + /partially asleep with running cells/ + ) + assert.equal(testHarness.commands.length, 0) + assert.deepEqual([...testHarness.cells.values()].map((cell) => cell.targetSize), [1, 1, 1, 1]) +}) + +test('wakes SQL and configured cells while leaving the candidate off and disabled', async () => { + const testHarness = harness({ sqlPolicy: 'NEVER', migSize: 0 }) + for (const cell of testHarness.cells.values()) cell.enabled = false + for (const state of testHarness.cells.values()) state.observedRequests = 0 + await runStagingRelayPower(argumentConfig(topologyFile(), 'wake'), testHarness.deps) + + assert.equal(testHarness.sqlPolicy(), 'ALWAYS') + assert.deepEqual([...testHarness.cells.values()].map((cell) => cell.targetSize), [1, 1, 0, 0]) + assert.deepEqual([...testHarness.cells.values()].map((cell) => cell.enabled), [true, true, false, false]) + assert.deepEqual(testHarness.events.at(-1), { + event: 'staging_relay_woke', + runningCells: ['staging-gce-c1', 'staging-gce-c2'], + admissionCells: ['staging-gce-c1', 'staging-gce-c2'] + }) +}) + +test('wakes every retained cell without rewriting selector-era admission', async () => { + const testHarness = harness({ + sqlPolicy: 'NEVER', + migSize: 0, + selectorGeneration: 1 + }) + const states = [...testHarness.cells.values()] + states[0].enabled = false + states[1].enabled = true + states[2].enabled = true + states[3].enabled = true + await runStagingRelayPower(argumentConfig(topologyFile(), 'wake'), testHarness.deps) + + assert.deepEqual(states.map((cell) => cell.targetSize), [1, 1, 1, 1]) + assert.deepEqual(states.map((cell) => cell.enabled), [false, true, true, true]) + assert.deepEqual(testHarness.events.at(-1), { + event: 'staging_relay_woke', + runningCells: ['staging-gce-c1', 'staging-gce-c2', 'staging-gce-c3', 'staging-gce-c4'], + admissionCells: ['staging-gce-c3'] + }) +}) + +test('status is read-only and reports the current billable floor controls', async () => { + const testHarness = harness() + await runStagingRelayPower(argumentConfig(topologyFile(), 'status'), testHarness.deps) + assert.equal(testHarness.commands.length, 0) + assert.equal(testHarness.events[0].project, 'onorca-cloud-staging') + assert.equal(testHarness.events[0].sqlActivationPolicy, 'ALWAYS') + assert.equal(testHarness.events[0].cells.length, 4) +}) diff --git a/cloud/dev/scripts/prepare-relay-asia-director-cells.mjs b/cloud/dev/scripts/prepare-relay-asia-director-cells.mjs new file mode 100644 index 00000000000..cd39a83c549 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-asia-director-cells.mjs @@ -0,0 +1,78 @@ +import { readFileSync, writeFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +function argumentsFrom(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['current-json', 'topology-json', 'output', 'cell-ids', 'image-digest']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + return values +} + +export function prepareRelayAsiaDirectorCells({ currentCells, topology, cellIds, imageDigest }) { + if (!Array.isArray(currentCells) || !topology || Array.isArray(topology)) { + throw new Error('director inputs are invalid') + } + const additions = cellIds.split(',').map((value) => value.trim()).filter(Boolean) + if ( + additions.length === 0 || + new Set(additions).size !== additions.length + ) throw new Error('director Asia additions are invalid') + const currentIds = new Set(currentCells.map((cell) => cell.id)) + if (currentIds.size !== currentCells.length) throw new Error('current director cells contain duplicates') + const normalizedCurrent = currentCells.map((cell) => ({ + ...cell, + region: cell.region ?? 'us-central1' + })) + const desiredCells = additions.map((cellId) => { + const cell = topology[cellId] + if ( + !cell || + cell.region !== 'asia-east2' || + cell.capacity_requests !== 6_000 || + cell.database_pool_max !== 10 || + cell.connection_hard_cap !== 3_000 || + cell.connection_unobserved_bound !== 60 || + cell.initially_enabled !== false || + cell.image?.split('@')[1] !== imageDigest + ) throw new Error(`${cellId} state output does not match the reviewed Asia shape`) + return { + id: cellId, + url: cell.origin, + capacityRequests: cell.capacity_requests, + region: cell.region, + initiallyEnabled: false, + connectionHardCap: cell.connection_hard_cap, + connectionUnobservedBound: cell.connection_unobserved_bound + } + }) + const desiredById = new Map(desiredCells.map((cell) => [cell.id, cell])) + for (const current of normalizedCurrent) { + const desired = desiredById.get(current.id) + if (!desired) continue + for (const [key, value] of Object.entries(desired)) { + if (current[key] !== value) { + throw new Error(`${current.id} director configuration differs from the reviewed Asia shape`) + } + } + } + const configured = new Set(normalizedCurrent.map((cell) => cell.id)) + return [...normalizedCurrent, ...desiredCells.filter((cell) => !configured.has(cell.id))] +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const values = argumentsFrom(process.argv.slice(2)) + const result = prepareRelayAsiaDirectorCells({ + currentCells: JSON.parse(readFileSync(values['current-json'], 'utf8')), + topology: JSON.parse(readFileSync(values['topology-json'], 'utf8')), + cellIds: values['cell-ids'], + imageDigest: values['image-digest'] + }) + writeFileSync(values.output, `${JSON.stringify(result)}\n`, { mode: 0o600 }) +} diff --git a/cloud/dev/scripts/prepare-relay-asia-director-cells.test.mjs b/cloud/dev/scripts/prepare-relay-asia-director-cells.test.mjs new file mode 100644 index 00000000000..9a223725006 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-asia-director-cells.test.mjs @@ -0,0 +1,60 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { prepareRelayAsiaDirectorCells } from './prepare-relay-asia-director-cells.mjs' + +const digest = `sha256:${'a'.repeat(64)}` +const topologyCell = (ordinal, zone) => ({ + origin: `https://c${ordinal}.relay.onorca.dev`, region: 'asia-east2', zone, + capacity_requests: 6_000, database_pool_max: 10, + connection_hard_cap: 3_000, connection_unobserved_bound: 60, + initially_enabled: false, + image: `us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@${digest}` +}) + +test('preserves current order, defaults predecessor regions, and appends exact Asia cells', () => { + const current = [{ + id: 'production-gce-c1', url: 'https://c1.relay.onorca.dev', + capacityRequests: 4_000, initiallyEnabled: false + }] + const result = prepareRelayAsiaDirectorCells({ + currentCells: current, + topology: { + 'production-gce-c27': topologyCell(27, 'asia-east2-a'), + 'production-gce-c28': topologyCell(28, 'asia-east2-b'), + 'production-gce-c29': topologyCell(29, 'asia-east2-c') + }, + cellIds: 'production-gce-c27,production-gce-c28,production-gce-c29', + imageDigest: digest + }) + assert.equal(result[0].region, 'us-central1') + assert.deepEqual(result.slice(1).map(({ id }) => id), [ + 'production-gce-c27', 'production-gce-c28', 'production-gce-c29' + ]) + assert.ok(result.slice(1).every((cell) => + cell.region === 'asia-east2' && cell.initiallyEnabled === false && + cell.connectionHardCap === 3_000 + )) +}) + +test('is idempotent for an exact existing Asia cell and rejects director drift', () => { + const topology = { 'production-gce-c27': topologyCell(27, 'asia-east2-a') } + const current = prepareRelayAsiaDirectorCells({ + currentCells: [], topology, cellIds: 'production-gce-c27', imageDigest: digest + }) + assert.deepEqual(prepareRelayAsiaDirectorCells({ + currentCells: current, topology, cellIds: 'production-gce-c27', imageDigest: digest + }), current) + assert.throws(() => prepareRelayAsiaDirectorCells({ + currentCells: [{ ...current[0], capacityRequests: 5_999 }], + topology, cellIds: 'production-gce-c27', imageDigest: digest + }), /director configuration differs/) +}) + +test('rejects a mismatching topology state output', () => { + const wrong = topologyCell(27, 'asia-east2-a') + wrong.database_pool_max = 20 + assert.throws(() => prepareRelayAsiaDirectorCells({ + currentCells: [], topology: { 'production-gce-c27': wrong }, + cellIds: 'production-gce-c27', imageDigest: digest + }), /does not match/) +}) diff --git a/cloud/dev/scripts/prepare-relay-asia-topology-input.mjs b/cloud/dev/scripts/prepare-relay-asia-topology-input.mjs new file mode 100644 index 00000000000..db76a83ede0 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-asia-topology-input.mjs @@ -0,0 +1,113 @@ +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +const BOOT_IMAGE = 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21' +const SHAPES = { + staging: { project: 'onorca-cloud-staging', cells: { 'staging-gce-c4': 'asia-east2-a' } }, + production: { + project: 'onorca-cloud', + cells: { + 'production-gce-c27': 'asia-east2-a', + 'production-gce-c28': 'asia-east2-b', + 'production-gce-c29': 'asia-east2-c' + } + } +} + +function argumentsFrom(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['existing-json', 'environment', 'cell-ids', 'image']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + return values +} + +function canonical(value) { + if (Array.isArray(value)) return value.map(canonical) + if (value && typeof value === 'object') { + return Object.fromEntries(Object.keys(value).sort().map((key) => [key, canonical(value[key])])) + } + return value +} + +export function prepareRelayAsiaTopologyInput({ + existingCells, + existingAdditionalRegions, + environment, + cellIds, + image +}) { + const shape = SHAPES[environment] + if (!shape) throw new Error('invalid environment') + const requested = cellIds.split(',').map((value) => value.trim()).filter(Boolean).sort() + const expected = Object.keys(shape.cells).sort() + if (new Set(requested).size !== requested.length || JSON.stringify(requested) !== JSON.stringify(expected)) { + throw new Error('cell IDs do not match the reviewed Asia topology') + } + const prefix = `us-central1-docker.pkg.dev/${shape.project}/orca-cloud/relay@sha256:` + if (!image.startsWith(prefix) || !/sha256:[a-f0-9]{64}$/.test(image)) { + throw new Error('image is not the environment Relay image pinned by digest') + } + if (!existingCells || Array.isArray(existingCells) || typeof existingCells !== 'object') { + throw new Error('existing Relay cells must be an object') + } + if ( + !existingAdditionalRegions || + Array.isArray(existingAdditionalRegions) || + typeof existingAdditionalRegions !== 'object' || + JSON.stringify(canonical(existingAdditionalRegions)) !== + JSON.stringify(canonical({ 'asia-east2': '10.42.1.0/24' })) + ) { + throw new Error('Asia subnet must be committed before topology planning') + } + const additions = Object.fromEntries(expected.map((cellId) => { + const hostname = cellId.split('-').at(-1) + return [cellId, { + hostname, + region: 'asia-east2', + zone: shape.cells[cellId], + machine_type: 'e2-standard-4', + boot_disk_gb: 30, + boot_image: BOOT_IMAGE, + capacity_requests: 6_000, + database_pool_max: 10, + image, + initially_enabled: false, + connection_hard_cap: 3_000, + connection_unobserved_bound: 60 + }] + })) + for (const cellId of expected) { + if (!existingCells[cellId]) { + throw new Error('Asia cells must be committed before topology planning') + } + if ( + JSON.stringify(canonical(existingCells[cellId])) !== + JSON.stringify(canonical(additions[cellId])) + ) { + throw new Error('committed Asia cell differs from the reviewed topology') + } + } + return { + relay_gce_additional_region_subnetwork_cidrs: existingAdditionalRegions, + relay_gce_cells: existingCells + } +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const values = argumentsFrom(process.argv.slice(2)) + const existing = JSON.parse(readFileSync(values['existing-json'], 'utf8')) + prepareRelayAsiaTopologyInput({ + existingCells: existing.relay_gce_cells, + existingAdditionalRegions: existing.relay_gce_additional_region_subnetwork_cidrs, + environment: values.environment, + cellIds: values['cell-ids'], + image: values.image + }) +} diff --git a/cloud/dev/scripts/prepare-relay-asia-topology-input.test.mjs b/cloud/dev/scripts/prepare-relay-asia-topology-input.test.mjs new file mode 100644 index 00000000000..03622ffbf56 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-asia-topology-input.test.mjs @@ -0,0 +1,87 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { prepareRelayAsiaTopologyInput } from './prepare-relay-asia-topology-input.mjs' + +const image = `us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:${'a'.repeat(64)}` + +const additionalRegions = { 'asia-east2': '10.42.1.0/24' } + +const productionCells = () => Object.fromEntries([ + [27, 'asia-east2-a'], + [28, 'asia-east2-b'], + [29, 'asia-east2-c'] +].map(([ordinal, zone]) => [`production-gce-c${ordinal}`, { + hostname: `c${ordinal}`, region: 'asia-east2', zone, + machine_type: 'e2-standard-4', boot_disk_gb: 30, + boot_image: 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21', + capacity_requests: 6_000, database_pool_max: 10, image, initially_enabled: false, + connection_hard_cap: 3_000, connection_unobserved_bound: 60 + }])) + +test('accepts the exact production topology only after it is durably committed', () => { + const existing = { + 'production-gce-c26': { hostname: 'c26', image: 'existing' }, + ...productionCells() + } + const result = prepareRelayAsiaTopologyInput({ existingCells: existing, + existingAdditionalRegions: additionalRegions, environment: 'production', + cellIds: 'production-gce-c27,production-gce-c28,production-gce-c29', image }) + assert.equal(result.relay_gce_cells, existing) + assert.equal(result.relay_gce_additional_region_subnetwork_cidrs, additionalRegions) + assert.deepEqual(existing['production-gce-c27'], { + hostname: 'c27', region: 'asia-east2', zone: 'asia-east2-a', + machine_type: 'e2-standard-4', boot_disk_gb: 30, + boot_image: 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21', + capacity_requests: 6_000, database_pool_max: 10, image, initially_enabled: false, + connection_hard_cap: 3_000, connection_unobserved_bound: 60 + }) +}) + +test('accepts the one exact committed staging Asia cell', () => { + const stagingImage = image.replace('onorca-cloud/', 'onorca-cloud-staging/') + const stagingCell = { + hostname: 'c4', region: 'asia-east2', zone: 'asia-east2-a', + machine_type: 'e2-standard-4', boot_disk_gb: 30, + boot_image: 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21', + capacity_requests: 6_000, database_pool_max: 10, image: stagingImage, + initially_enabled: false, connection_hard_cap: 3_000, + connection_unobserved_bound: 60 + } + const result = prepareRelayAsiaTopologyInput({ + existingCells: { 'staging-gce-c3': { hostname: 'c3' }, 'staging-gce-c4': stagingCell }, + existingAdditionalRegions: additionalRegions, + environment: 'staging', + cellIds: 'staging-gce-c4', + image: stagingImage + }) + assert.equal(result.relay_gce_cells['staging-gce-c4'].zone, 'asia-east2-a') + assert.equal(result.relay_gce_cells['staging-gce-c4'].image, stagingImage) +}) + +test('rejects an uncommitted subnet or cell, partial wave, wrong image, and drift', () => { + assert.throws(() => prepareRelayAsiaTopologyInput({ + existingCells: productionCells(), existingAdditionalRegions: additionalRegions, + environment: 'production', cellIds: 'production-gce-c27', image + }), /cell IDs/) + assert.throws(() => prepareRelayAsiaTopologyInput({ + existingCells: productionCells(), existingAdditionalRegions: additionalRegions, + environment: 'production', + cellIds: 'production-gce-c27,production-gce-c28,production-gce-c29', + image: image.replace('onorca-cloud/', 'other-project/') + }), /environment Relay image/) + assert.throws(() => prepareRelayAsiaTopologyInput({ + existingCells: productionCells(), existingAdditionalRegions: {}, environment: 'production', + cellIds: 'production-gce-c27,production-gce-c28,production-gce-c29', image + }), /subnet must be committed/) + const missing = productionCells() + delete missing['production-gce-c29'] + assert.throws(() => prepareRelayAsiaTopologyInput({ + existingCells: missing, existingAdditionalRegions: additionalRegions, environment: 'production', + cellIds: 'production-gce-c27,production-gce-c28,production-gce-c29', image + }), /cells must be committed/) + assert.throws(() => prepareRelayAsiaTopologyInput({ + existingCells: { ...productionCells(), 'production-gce-c27': {} }, + existingAdditionalRegions: additionalRegions, environment: 'production', + cellIds: 'production-gce-c27,production-gce-c28,production-gce-c29', image + }), /differs from the reviewed topology/) +}) diff --git a/cloud/dev/scripts/prepare-relay-capacity-canary.mjs b/cloud/dev/scripts/prepare-relay-capacity-canary.mjs new file mode 100644 index 00000000000..7bb7574d302 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-capacity-canary.mjs @@ -0,0 +1,183 @@ +import { pathToFileURL } from 'node:url' +import { + applyExactAdmissionSelector, + inspectAdmissionSelector, + membershipWithStates, + selectorCellState +} from './relay-admission-selector.mjs' + +function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['director-origin', 'cell-id', 'mode']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + const origin = new URL(values['director-origin']) + if (origin.protocol !== 'https:' || origin.origin !== values['director-origin']) { + throw new Error('--director-origin must be a canonical HTTPS origin') + } + if (!['isolate', 'activate', 'restore-fallback', 'restore'].includes(values.mode)) { + throw new Error('--mode must be isolate, activate, restore-fallback, or restore') + } + const cellOrigin = values['cell-origin'] ? new URL(values['cell-origin']) : null + if ( + values.mode === 'isolate' && + (!cellOrigin || cellOrigin.protocol !== 'https:' || cellOrigin.origin !== values['cell-origin']) + ) { + throw new Error('--cell-origin must be a canonical HTTPS origin for isolate mode') + } + const restoreGeneralCellIds = values['general-cell-ids']?.split(',').filter(Boolean) ?? [] + if ( + ['restore-fallback', 'restore'].includes(values.mode) && + restoreGeneralCellIds.length === 0 + ) { + throw new Error('--general-cell-ids is required for restore modes') + } + if (values.mode === 'restore' && !restoreGeneralCellIds.includes(values['cell-id'])) { + throw new Error('--general-cell-ids must include the canary for restore mode') + } + if (values.mode === 'restore-fallback' && restoreGeneralCellIds.includes(values['cell-id'])) { + throw new Error('--general-cell-ids cannot include the canary for fallback restore') + } + if (new Set(restoreGeneralCellIds).size !== restoreGeneralCellIds.length) { + throw new Error('--general-cell-ids must be distinct') + } + return { + directorOrigin: origin.origin, + cellOrigin: cellOrigin?.origin, + cellId: values['cell-id'], + mode: values.mode, + restoreGeneralCellIds + } +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({})) + if (!response.ok) throw new Error(`${label} returned ${response.status}`) + return body +} + +function sameMembership(left, right) { + return JSON.stringify(left) === JSON.stringify(right) +} + +async function legacyCellState(post, cellId) { + const result = await post('/v1/admin/cell-status', { v: 1, cellId }) + if (result.status?.cellId !== cellId) throw new Error('legacy admission status is invalid') + const state = result.status.admissionState + if (!['existing-only', 'migration-only', 'general'].includes(state)) { + throw new Error('legacy admission status is invalid') + } + return state +} + +async function applyLegacyStates(post, before, states, order) { + const expected = membershipWithStates(before.selector, states) + let changed = false + for (const cellId of order) { + const desired = states[cellId] + const current = await legacyCellState(post, cellId) + if (current === 'existing-only' && desired !== current) { + throw new Error(`legacy admission cannot re-enable existing-only cell ${cellId}`) + } + if (current === desired) continue + let cause + try { + await post('/v1/admin/cell-state', { v: 1, cellId, state: desired }) + } catch (error) { + cause = error + } + if ((await legacyCellState(post, cellId)) !== desired) { + const detail = cause instanceof Error ? `: ${cause.message}` : '' + throw new Error(`legacy admission did not commit ${cellId} exactly${detail}`, { cause }) + } + changed = true + } + const verified = await inspectAdmissionSelector(post) + if (verified.selector.generation !== 0 || + !sameMembership(verified.selector.membership, expected)) { + throw new Error('legacy admission membership changed unexpectedly') + } + return { changed, selector: verified.selector } +} + +export async function prepareCapacityCanary(config, overrides = {}) { + const fetchImpl = overrides.fetch ?? fetch + const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN + if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') + const postAt = async (origin, path, body) => + await responseJson( + await fetchImpl(`${origin}${path}`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }), + path + ) + const post = async (path, body) => await postAt(config.directorOrigin, path, body) + const before = await inspectAdmissionSelector(post) + const state = selectorCellState(before.selector, config.cellId) + if (state === 'existing-only' && config.mode !== 'restore-fallback') { + throw new Error('capacity canary cannot restore existing-only admission') + } + const states = + config.mode === 'isolate' + ? { [config.cellId]: 'migration-only' } + : config.mode === 'activate' + ? Object.fromEntries([ + ...before.selector.membership.general.map((cellId) => [ + cellId, + cellId === config.cellId ? 'general' : 'migration-only' + ]), + [config.cellId, 'general'] + ]) + : Object.fromEntries([ + ...config.restoreGeneralCellIds.map((cellId) => [cellId, 'general']), + ...(config.mode === 'restore-fallback' + ? [[config.cellId, state === 'existing-only' ? 'existing-only' : 'migration-only']] + : []) + ]) + const membership = membershipWithStates(before.selector, states) + const legacyOrder = + config.mode === 'activate' + ? [config.cellId, ...before.selector.membership.general.filter((id) => id !== config.cellId)] + : config.mode === 'restore-fallback' + ? [...config.restoreGeneralCellIds, config.cellId] + : Object.keys(states) + const result = before.selector.generation === 0 + ? await applyLegacyStates(post, before, states, legacyOrder) + : sameMembership(membership, before.selector.membership) + ? { changed: false, selector: before.selector } + : await applyExactAdmissionSelector(post, membership, { + expectedCurrentSelector: before.selector + }) + if (config.mode === 'isolate') { + await postAt(config.cellOrigin, '/v1/admin/drain', { v: 1, graceMs: 0 }) + } + return { + changed: result.changed, + generation: result.selector.generation, + ...(config.mode === 'isolate' ? { drained: true } : {}) + } +} + +export async function main(argv = process.argv.slice(2)) { + const config = parseArguments(argv) + const result = await prepareCapacityCanary(config) + process.stdout.write( + `${JSON.stringify({ event: 'relay_capacity_canary_admission', cellId: config.cellId, mode: config.mode, ...result })}\n` + ) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/prepare-relay-capacity-canary.test.mjs b/cloud/dev/scripts/prepare-relay-capacity-canary.test.mjs new file mode 100644 index 00000000000..d8e752aa500 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-capacity-canary.test.mjs @@ -0,0 +1,300 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { relayWorkflowUrl } from './relay-repository.mjs' +import { prepareCapacityCanary } from './prepare-relay-capacity-canary.mjs' + +function harness(initialState, options = {}) { + const { + generation = 4, + ambiguousCellState = false, + rejectCellState = false, + fallbackState = 'general', + extraGeneralCellIds = [] + } = options + let selector = { + generation, + attemptId: 'initial', + membership: { + existingOnly: [ + 'staging-gce-c1', + ...(initialState === 'existing-only' ? ['staging-gce-c3'] : []) + ], + migrationOnly: [ + ...(fallbackState === 'migration-only' ? ['staging-gce-c2'] : []), + ...(initialState === 'migration-only' ? ['staging-gce-c3'] : []) + ], + general: [ + ...extraGeneralCellIds, + ...(fallbackState === 'general' ? ['staging-gce-c2'] : []), + ...(initialState === 'general' ? ['staging-gce-c3'] : []) + ].sort() + } + } + let intent = null + let applies = 0 + let drains = 0 + const cellStateChanges = [] + const fetch = async (url, options) => { + const path = new URL(url).pathname + const body = JSON.parse(options.body) + if (path === '/v1/admin/drain') { + assert.deepEqual(body, { v: 1, graceMs: 0 }) + drains++ + return Response.json({ ok: true }) + } + if (path === '/v1/admin/cell-status') { + const state = selector.membership.existingOnly.includes(body.cellId) + ? 'existing-only' + : selector.membership.migrationOnly.includes(body.cellId) + ? 'migration-only' + : 'general' + return Response.json({ status: { cellId: body.cellId, admissionState: state } }) + } + if (path === '/v1/admin/cell-state') { + assert.equal(selector.generation, 0) + if (rejectCellState) { + return Response.json({ error: 'invalid_token' }, { status: 401 }) + } + const keys = { + 'existing-only': 'existingOnly', + 'migration-only': 'migrationOnly', + general: 'general' + } + for (const cells of Object.values(selector.membership)) { + const index = cells.indexOf(body.cellId) + if (index !== -1) cells.splice(index, 1) + } + selector.membership[keys[body.state]].push(body.cellId) + for (const cells of Object.values(selector.membership)) cells.sort() + cellStateChanges.push({ cellId: body.cellId, state: body.state }) + if (ambiguousCellState) throw new Error('response lost') + return Response.json({ ok: true }) + } + if (path.endsWith('/status')) return Response.json({ selector, intent }) + applies++ + selector = { + generation: body.expectedGeneration + 1, + attemptId: body.attemptId, + membership: body.membership + } + intent = { + attemptId: body.attemptId, + expectedGeneration: body.expectedGeneration, + intendedGeneration: selector.generation, + membership: selector.membership, + state: 'committed' + } + return Response.json({ changed: true, selector }) + } + return { + fetch, + selector: () => selector, + applies: () => applies, + drains: () => drains, + cellStateChanges + } +} + +const config = { + directorOrigin: 'https://relay.example.com', + cellOrigin: 'https://c3.relay.example.com', + cellId: 'staging-gce-c3', + mode: 'isolate', + restoreGeneralCellIds: [] +} + +test('isolates a general canary as migration-only', async () => { + const testHarness = harness('general') + assert.deepEqual( + await prepareCapacityCanary(config, { fetch: testHarness.fetch, token: 'masked' }), + { changed: true, generation: 5, drained: true } + ) + assert.deepEqual(testHarness.selector().membership.migrationOnly, [config.cellId]) + assert.equal(testHarness.applies(), 1) + assert.equal(testHarness.drains(), 1) +}) + +test('activates the canary as the only general cell', async () => { + const testHarness = harness('migration-only') + assert.deepEqual( + await prepareCapacityCanary( + { ...config, mode: 'activate' }, + { fetch: testHarness.fetch, token: 'masked' } + ), + { changed: true, generation: 5 } + ) + assert.deepEqual(testHarness.selector().membership, { + existingOnly: ['staging-gce-c1'], + migrationOnly: ['staging-gce-c2'], + general: [config.cellId] + }) + assert.equal(testHarness.applies(), 1) +}) + +test('restores the reviewed staging general membership', async () => { + const testHarness = harness('migration-only') + assert.deepEqual( + await prepareCapacityCanary( + { + ...config, + mode: 'restore', + restoreGeneralCellIds: ['staging-gce-c2', config.cellId] + }, + { fetch: testHarness.fetch, token: 'masked' } + ), + { changed: true, generation: 5 } + ) + assert.deepEqual(testHarness.selector().membership, { + existingOnly: ['staging-gce-c1'], + migrationOnly: [], + general: ['staging-gce-c2', config.cellId] + }) +}) + +test('restores the fallback without promoting a possibly drained canary', async () => { + const testHarness = harness('general') + await prepareCapacityCanary( + { + ...config, + mode: 'restore-fallback', + restoreGeneralCellIds: ['staging-gce-c2'] + }, + { fetch: testHarness.fetch, token: 'masked' } + ) + assert.deepEqual(testHarness.selector().membership, { + existingOnly: ['staging-gce-c1'], + migrationOnly: [config.cellId], + general: ['staging-gce-c2'] + }) +}) + +test('restores the fallback while preserving an irreversible canary', async () => { + const testHarness = harness('existing-only') + await prepareCapacityCanary( + { + ...config, + mode: 'restore-fallback', + restoreGeneralCellIds: ['staging-gce-c2'] + }, + { fetch: testHarness.fetch, token: 'masked' } + ) + assert.deepEqual(testHarness.selector().membership, { + existingOnly: ['staging-gce-c1', config.cellId], + migrationOnly: [], + general: ['staging-gce-c2'] + }) +}) + +test('uses exact legacy admission writes before the selector boundary', async () => { + const testHarness = harness('general', { generation: 0 }) + assert.deepEqual( + await prepareCapacityCanary(config, { fetch: testHarness.fetch, token: 'masked' }), + { changed: true, generation: 0, drained: true } + ) + assert.deepEqual(testHarness.cellStateChanges, [ + { cellId: config.cellId, state: 'migration-only' } + ]) + assert.equal(testHarness.drains(), 1) +}) + +test('promotes a legacy canary before demoting its fallback', async () => { + const testHarness = harness('migration-only', { generation: 0 }) + await prepareCapacityCanary( + { ...config, mode: 'activate' }, + { fetch: testHarness.fetch, token: 'masked' } + ) + assert.deepEqual(testHarness.cellStateChanges, [ + { cellId: config.cellId, state: 'general' }, + { cellId: 'staging-gce-c2', state: 'migration-only' } + ]) +}) + +test('makes the canary sole general with the live legacy membership shape', async () => { + const extraGeneralCellIds = ['combined', 'staging-c1', 'staging-c2'] + const testHarness = harness('migration-only', { generation: 0, extraGeneralCellIds }) + const activate = { ...config, mode: 'activate' } + await prepareCapacityCanary(activate, { fetch: testHarness.fetch, token: 'masked' }) + assert.deepEqual(testHarness.cellStateChanges, [ + { cellId: config.cellId, state: 'general' }, + { cellId: 'combined', state: 'migration-only' }, + { cellId: 'staging-c1', state: 'migration-only' }, + { cellId: 'staging-c2', state: 'migration-only' }, + { cellId: 'staging-gce-c2', state: 'migration-only' } + ]) +}) + +test('restores a legacy fallback before demoting the target', async () => { + const testHarness = harness('general', { + generation: 0, + fallbackState: 'migration-only' + }) + await prepareCapacityCanary( + { + ...config, + mode: 'restore-fallback', + restoreGeneralCellIds: ['staging-gce-c2'] + }, + { fetch: testHarness.fetch, token: 'masked' } + ) + assert.deepEqual(testHarness.cellStateChanges, [ + { cellId: 'staging-gce-c2', state: 'general' }, + { cellId: config.cellId, state: 'migration-only' } + ]) +}) + +test('keeps an already restored legacy fallback unchanged', async () => { + const testHarness = harness('migration-only', { generation: 0 }) + assert.deepEqual( + await prepareCapacityCanary( + { + ...config, + mode: 'restore-fallback', + restoreGeneralCellIds: ['staging-gce-c2'] + }, + { fetch: testHarness.fetch, token: 'masked' } + ), + { changed: false, generation: 0 } + ) + assert.deepEqual(testHarness.cellStateChanges, []) + assert.deepEqual(testHarness.selector().membership, { + existingOnly: ['staging-gce-c1'], + migrationOnly: [config.cellId], + general: ['staging-gce-c2'] + }) +}) + +test('recovers an ambiguous legacy admission response by exact readback', async () => { + const testHarness = harness('general', { generation: 0, ambiguousCellState: true }) + await assert.doesNotReject( + prepareCapacityCanary(config, { fetch: testHarness.fetch, token: 'masked' }) + ) + assert.equal(testHarness.drains(), 1) +}) + +test('reports a rejected legacy admission write without draining', async () => { + const testHarness = harness('general', { generation: 0, rejectCellState: true }) + const operation = prepareCapacityCanary(config, { fetch: testHarness.fetch, token: 'masked' }) + await assert.rejects(operation, /cell-state returned 401/) + assert.equal(testHarness.drains(), 0) +}) + +test('the staging workflow supplies every required capacity transition argument', () => { + const workflow = readFileSync( + relayWorkflowUrl('prove-relay-staging-capacity.yml'), + 'utf8' + ) + const verifyCalls = workflow.match( + /node dev\/scripts\/verify-relay-capacity-transition\.mjs[\s\S]*?(?=\n\s*\n|\n\s*- name:)/g + ) + assert.ok(verifyCalls?.length >= 5) + for (const call of verifyCalls) { + for (const flag of ['--cell-origin', '--heartbeat', '--admission', '--draining', '--activity']) { + assert.match(call, new RegExp(flag)) + } + } + const isolate = workflow.match( + /node dev\/scripts\/prepare-relay-capacity-canary\.mjs[\s\S]*?--mode isolate/ + )?.[0] + assert.match(isolate, /--cell-origin/) +}) diff --git a/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs b/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs new file mode 100644 index 00000000000..5791c9f20e6 --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-production-capacity-canary.mjs @@ -0,0 +1,116 @@ +import { pathToFileURL } from 'node:url' +import { + applyExactAdmissionSelector, + inspectAdmissionSelector, + membershipWithStates, + selectorCellState +} from './relay-admission-selector.mjs' + +const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' +export const PRODUCTION_CAPACITY_CELL_IDS = [ + 'production-gce-c7', + 'production-gce-c8', + 'production-gce-c9', + 'production-gce-c10', + 'production-gce-c13', + 'production-gce-c14', + 'production-gce-c15', + 'production-gce-c16', + 'production-gce-c19', + 'production-gce-c20', + 'production-gce-c21', + 'production-gce-c22', + 'production-gce-c23', + 'production-gce-c24', + 'production-gce-c25', + 'production-gce-c26' +] + +function cellOrigin(cellId) { + return `https://${cellId.slice('production-gce-'.length)}.relay.onorca.dev` +} + +export function parseProductionCapacityCellArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + if (!['isolate', 'drain', 'activate'].includes(values.mode)) { + throw new Error('--mode must be isolate, drain, or activate') + } + const cellId = values['cell-id'] + if (!PRODUCTION_CAPACITY_CELL_IDS.includes(cellId)) { + throw new Error('production capacity target is not approved') + } + const expectedCellOrigin = cellOrigin(cellId) + if ( + values['director-origin'] !== DIRECTOR_ORIGIN || + values['cell-origin'] !== expectedCellOrigin + ) { + throw new Error('production capacity target origin is not exact') + } + return { + directorOrigin: DIRECTOR_ORIGIN, + cellOrigin: expectedCellOrigin, + cellId, + mode: values.mode + } +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({})) + if (!response.ok) throw new Error(`${label} returned ${response.status}`) + return body +} + +export async function prepareProductionCapacityCell(config, overrides = {}) { + const fetchImpl = overrides.fetch ?? fetch + const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN + if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') + const postAt = async (origin, path, body) => + await responseJson( + await fetchImpl(`${origin}${path}`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify(body), + signal: AbortSignal.timeout(30_000) + }), + path + ) + const post = async (path, body) => await postAt(config.directorOrigin, path, body) + if (config.mode === 'drain') { + await postAt(config.cellOrigin, '/v1/admin/drain', { v: 1, graceMs: 0 }) + return { changed: false, drained: true } + } + const before = await inspectAdmissionSelector(post) + const state = selectorCellState(before.selector, config.cellId) + if (state === 'existing-only') throw new Error('production capacity target is irreversible') + const desiredState = config.mode === 'isolate' ? 'migration-only' : 'general' + const membership = membershipWithStates(before.selector, { [config.cellId]: desiredState }) + const result = await applyExactAdmissionSelector(post, membership, { + expectedCurrentSelector: before.selector + }) + return { + changed: result.changed, + generation: result.selector.generation, + admissionState: desiredState + } +} + +export async function main(argv = process.argv.slice(2)) { + const config = parseProductionCapacityCellArguments(argv) + const result = await prepareProductionCapacityCell(config) + process.stdout.write( + `${JSON.stringify({ event: 'relay_production_capacity_canary', cellId: config.cellId, mode: config.mode, ...result })}\n` + ) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs b/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs new file mode 100644 index 00000000000..c5d0a9db3bc --- /dev/null +++ b/cloud/dev/scripts/prepare-relay-production-capacity-canary.test.mjs @@ -0,0 +1,173 @@ +import assert from 'node:assert/strict' +import { describe, it } from 'node:test' +import { + parseProductionCapacityCellArguments, + prepareProductionCapacityCell, + PRODUCTION_CAPACITY_CELL_IDS +} from './prepare-relay-production-capacity-canary.mjs' + +const config = { + directorOrigin: 'https://relay.onorca.dev', + cellOrigin: 'https://c26.relay.onorca.dev', + cellId: 'production-gce-c26' +} + +const membership = { + existingOnly: ['production-gce-c1'], + migrationOnly: ['production-gce-c17'], + general: ['production-gce-c25', 'production-gce-c26'] +} + +function response(body, status = 200) { + return new Response(JSON.stringify(body), { + status, + headers: { 'content-type': 'application/json' } + }) +} + +function canaryFetch() { + let selector = { generation: 20, attemptId: null, membership } + const calls = [] + const fetch = async (url, init) => { + const path = new URL(url).pathname + const body = JSON.parse(init.body) + calls.push({ path, body }) + if (path === '/v1/admin/admission-selector/status') { + return response({ + v: 1, + selector, + intent: body.attemptId + ? { + attemptId: body.attemptId, + state: 'committed', + expectedGeneration: selector.generation - 1, + intendedGeneration: selector.generation, + membership: selector.membership + } + : null + }) + } + if (path === '/v1/admin/admission-selector/apply') { + selector = { + generation: selector.generation + 1, + attemptId: body.attemptId, + membership: body.membership + } + return response({ v: 1, changed: true, selector }) + } + if (path === '/v1/admin/drain') return response({ v: 1, draining: true }) + throw new Error(`unexpected ${path}`) + } + return { calls, fetch, selector: () => selector } +} + +describe('production Relay capacity cell admission', () => { + it('allows only the serving rollout cells', () => { + assert.deepEqual(PRODUCTION_CAPACITY_CELL_IDS, [ + 'production-gce-c7', + 'production-gce-c8', + 'production-gce-c9', + 'production-gce-c10', + 'production-gce-c13', + 'production-gce-c14', + 'production-gce-c15', + 'production-gce-c16', + 'production-gce-c19', + 'production-gce-c20', + 'production-gce-c21', + 'production-gce-c22', + 'production-gce-c23', + 'production-gce-c24', + 'production-gce-c25', + 'production-gce-c26' + ]) + assert.deepEqual(parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c7.relay.onorca.dev', + '--cell-id', 'production-gce-c7', + '--mode', 'isolate' + ]), { + directorOrigin: 'https://relay.onorca.dev', + cellOrigin: 'https://c7.relay.onorca.dev', + cellId: 'production-gce-c7', + mode: 'isolate' + }) + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c17.relay.onorca.dev', + '--cell-id', 'production-gce-c17', + '--mode', 'isolate' + ]), /not approved/) + assert.throws(() => parseProductionCapacityCellArguments([ + '--director-origin', 'https://relay.onorca.dev', + '--cell-origin', 'https://c8.relay.onorca.dev', + '--cell-id', 'production-gce-c7', + '--mode', 'isolate' + ]), /origin is not exact/) + }) + + it('isolates only the selected cell without depending on its runtime', async () => { + const fake = canaryFetch() + const result = await prepareProductionCapacityCell( + { ...config, mode: 'isolate' }, + { fetch: fake.fetch, token: 'token' } + ) + assert.equal(result.admissionState, 'migration-only') + assert.deepEqual(fake.selector().membership, { + existingOnly: ['production-gce-c1'], + migrationOnly: ['production-gce-c17', 'production-gce-c26'], + general: ['production-gce-c25'] + }) + assert.doesNotMatch(fake.calls.map(({ path }) => path).join(','), /\/v1\/admin\/drain/) + }) + + it('drains the selected cell independently after durable isolation', async () => { + const fake = canaryFetch() + const result = await prepareProductionCapacityCell( + { ...config, mode: 'drain' }, + { fetch: fake.fetch, token: 'token' } + ) + assert.deepEqual(result, { changed: false, drained: true }) + assert.deepEqual(fake.calls, [{ + path: '/v1/admin/drain', + body: { v: 1, graceMs: 0 } + }]) + }) + + it('restores only the selected cell to general admission', async () => { + const fake = canaryFetch() + await prepareProductionCapacityCell( + { ...config, mode: 'isolate' }, + { fetch: fake.fetch, token: 'token' } + ) + const result = await prepareProductionCapacityCell( + { ...config, mode: 'activate' }, + { fetch: fake.fetch, token: 'token' } + ) + assert.equal(result.admissionState, 'general') + assert.deepEqual(fake.selector().membership, membership) + }) + + it('refuses an irreversible existing-only target', async () => { + const fetch = async () => response({ + v: 1, + selector: { + generation: 20, + attemptId: null, + membership: { + existingOnly: ['production-gce-c26'], + migrationOnly: ['production-gce-c17'], + general: ['production-gce-c25'] + } + }, + intent: null + }) + await assert.rejects( + prepareProductionCapacityCell( + { ...config, mode: 'isolate' }, + { fetch, token: 'token' } + ), + /irreversible/ + ) + }) +}) diff --git a/cloud/dev/scripts/probe-relay-legacy-admission.mjs b/cloud/dev/scripts/probe-relay-legacy-admission.mjs new file mode 100644 index 00000000000..3529c7d70ac --- /dev/null +++ b/cloud/dev/scripts/probe-relay-legacy-admission.mjs @@ -0,0 +1,73 @@ +import { randomBytes } from 'node:crypto' +import { pathToFileURL } from 'node:url' + +const WRONG_CELL = 4409 +const DRAINING = 4503 + +function once(socket, event, listener) { + if (typeof socket.once === 'function') { + socket.once(event, listener) + return + } + if (typeof socket.addEventListener !== 'function') { + throw new Error('WebSocket event API is unavailable') + } + socket.addEventListener(event, (value) => { + if (event === 'close') listener(value.code) + else if (event === 'error') listener(value.error ?? new Error(value.message)) + else listener() + }, { once: true }) +} + +export function parseLegacyAdmissionProbeArguments(argv) { + if (argv.length !== 2 || argv[0] !== '--cell-origin') throw new Error('invalid arguments') + const origin = new URL(argv[1]) + if (origin.protocol !== 'https:' || origin.origin !== argv[1]) { + throw new Error('--cell-origin must be a canonical HTTPS origin') + } + return { cellOrigin: origin.origin } +} + +export async function probeLegacyAdmission(config, overrides = {}) { + const Socket = overrides.WebSocket ?? globalThis.WebSocket + if (typeof Socket !== 'function') throw new Error('WebSocket is unavailable') + const random = overrides.randomBytes ?? randomBytes + const timeoutMs = overrides.timeoutMs ?? 15_000 + const hostId = random(12).toString('base64url') + const credential = random(32).toString('base64url') + const url = `${config.cellOrigin.replace('https://', 'wss://')}/v1/connect/${hostId}` + await new Promise((resolve, reject) => { + const socket = new Socket(url) + const timer = setTimeout(() => { + if (typeof socket.terminate === 'function') socket.terminate() + else socket.close() + reject(new Error('legacy admission probe timed out')) + }, timeoutMs) + once(socket, 'open', () => { + socket.send(JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential })) + }) + once(socket, 'close', (code) => { + clearTimeout(timer) + if (code === WRONG_CELL) resolve() + else if (code === DRAINING) reject(new Error('legacy cell is draining')) + else reject(new Error(`legacy admission probe closed with ${code}`)) + }) + once(socket, 'error', (error) => { + clearTimeout(timer) + reject(new Error(`legacy admission probe failed: ${error.message}`)) + }) + }) + return { accepting: true } +} + +export async function main(argv = process.argv.slice(2)) { + const result = await probeLegacyAdmission(parseLegacyAdmissionProbeArguments(argv)) + process.stdout.write(`${JSON.stringify({ event: 'relay_legacy_admission_verified', ...result })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/probe-relay-legacy-admission.test.mjs b/cloud/dev/scripts/probe-relay-legacy-admission.test.mjs new file mode 100644 index 00000000000..7c0d078b4f0 --- /dev/null +++ b/cloud/dev/scripts/probe-relay-legacy-admission.test.mjs @@ -0,0 +1,89 @@ +import assert from 'node:assert/strict' +import { EventEmitter } from 'node:events' +import { test } from 'node:test' +import { + parseLegacyAdmissionProbeArguments, + probeLegacyAdmission +} from './probe-relay-legacy-admission.mjs' + +function socketClosingWith(code, observed) { + return class extends EventEmitter { + constructor(url) { + super() + observed.url = url + queueMicrotask(() => this.emit('open')) + } + + send(payload) { + observed.payload = JSON.parse(payload) + queueMicrotask(() => this.emit('close', code)) + } + + terminate() {} + } +} + +function nativeSocketClosingWith(code) { + return class extends EventTarget { + constructor() { + super() + queueMicrotask(() => this.dispatchEvent(new Event('open'))) + } + + send() { + const event = new Event('close') + Object.defineProperty(event, 'code', { value: code }) + queueMicrotask(() => this.dispatchEvent(event)) + } + + close() {} + } +} + +const config = { cellOrigin: 'https://c2.relay.example.com' } +const random = (length) => Buffer.alloc(length, length) + +test('accepts only a canonical cell origin', () => { + assert.deepEqual( + parseLegacyAdmissionProbeArguments(['--cell-origin', config.cellOrigin]), + config + ) + assert.throws( + () => parseLegacyAdmissionProbeArguments(['--cell-origin', `${config.cellOrigin}/path`]), + /canonical/ + ) +}) + +test('proves admission with a synthetic invalid credential and exposes no identifier', async () => { + const observed = {} + assert.deepEqual( + await probeLegacyAdmission(config, { + WebSocket: socketClosingWith(4409, observed), + randomBytes: random + }), + { accepting: true } + ) + assert.match(observed.url, /^wss:\/\/c2\.relay\.example\.com\/v1\/connect\/[A-Za-z0-9_-]{16}$/) + assert.deepEqual(Object.keys(observed.payload).sort(), ['credential', 'mode', 'type', 'v']) +}) + +test('uses the dependency-free Node WebSocket event API', async () => { + await assert.doesNotReject( + probeLegacyAdmission(config, { + WebSocket: nativeSocketClosingWith(4409), + randomBytes: random + }) + ) +}) + +test('rejects the legacy draining close and any unknown outcome', async () => { + for (const [code, message] of [[4503, /draining/], [4401, /closed with 4401/]]) { + await assert.rejects( + probeLegacyAdmission(config, { + WebSocket: socketClosingWith(code, {}), + randomBytes: random + }), + message + ) + } +}) diff --git a/cloud/dev/scripts/probe-relay-rehome-trust.mjs b/cloud/dev/scripts/probe-relay-rehome-trust.mjs new file mode 100644 index 00000000000..7500d8bd14c --- /dev/null +++ b/cloud/dev/scripts/probe-relay-rehome-trust.mjs @@ -0,0 +1,80 @@ +import { pathToFileURL } from 'node:url' + +const PRODUCTION_CELL = /^production-gce-c(?:7|8|9|10|13|14|15|16|19|20|21|22|23|24|25|26)$/ +const DIRECTOR_ORIGIN = 'https://relay.onorca.dev' + +export function parseRehomeTrustProbeArguments(argv, environment = process.env) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['director-origin', 'cell-id', 'cell-incarnation']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (values['director-origin'] !== DIRECTOR_ORIGIN) { + throw new Error('--director-origin must be the production Relay origin') + } + if (!PRODUCTION_CELL.test(values['cell-id'])) throw new Error('--cell-id is not approved') + if (!/^[0-9a-f]{8}-[0-9a-f]{4}-4[0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}$/i.test( + values['cell-incarnation'] + )) throw new Error('--cell-incarnation is invalid') + const token = environment.ORCA_RELAY_ADMIN_ID_TOKEN + if (!token || token.length > 8_192 || !/^[^.]+\.[^.]+\.[^.]+$/.test(token)) { + throw new Error('admin identity token is unavailable') + } + return { + directorOrigin: DIRECTOR_ORIGIN, + cellId: values['cell-id'], + cellIncarnation: values['cell-incarnation'], + token + } +} + +export async function probeRehomeTrust(config, dependencies = {}) { + const fetchImpl = dependencies.fetch ?? fetch + const response = await fetchImpl( + `${config.directorOrigin}/v1/admin/regional-rehome-trust-probe`, + { + method: 'POST', + headers: { + authorization: `Bearer ${config.token}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + v: 1, + sourceCellId: config.cellId, + sourceCellIncarnation: config.cellIncarnation + }), + signal: AbortSignal.timeout(30_000) + } + ) + const body = await response.json().catch(() => ({})) + if (!response.ok) { + throw new Error(`application-mediated rehome trust probe returned ${response.status}`) + } + if ( + body.v !== 1 || + body.dedicatedIdentity?.firstOutcome !== 'host-not-connected' || + body.dedicatedIdentity?.secondOutcome !== 'host-not-connected' || + body.dedicatedIdentity?.accepted !== true || + body.dedicatedIdentity?.idempotent !== true || + body.sharedRuntimeIdentityRejected !== true || + body.proven !== true + ) throw new Error('application-mediated rehome trust proof is incomplete') + return body +} + +export async function main(argv = process.argv.slice(2)) { + const result = await probeRehomeTrust(parseRehomeTrustProbeArguments(argv)) + process.stdout.write(`${JSON.stringify({ event: 'relay_rehome_trust_verified', ...result })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs b/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs new file mode 100644 index 00000000000..509e9d53c7d --- /dev/null +++ b/cloud/dev/scripts/probe-relay-rehome-trust.test.mjs @@ -0,0 +1,70 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + parseRehomeTrustProbeArguments, + probeRehomeTrust +} from './probe-relay-rehome-trust.mjs' + +const argv = [ + '--director-origin', 'https://relay.onorca.dev', + '--cell-id', 'production-gce-c7', + '--cell-incarnation', '11111111-1111-4111-8111-111111111111' +] +const environment = { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' } + +test('binds the application-mediated probe to an exact approved cell incarnation', () => { + assert.equal(parseRehomeTrustProbeArguments(argv, environment).cellId, 'production-gce-c7') + assert.throws(() => parseRehomeTrustProbeArguments( + argv.with(1, 'https://other.example.test'), + environment + )) + assert.throws(() => parseRehomeTrustProbeArguments(argv, { + ORCA_RELAY_ADMIN_ID_TOKEN: 'not-a-token' + })) +}) + +test('requires complete aggregate application-mediated trust proof', async () => { + const config = parseRehomeTrustProbeArguments(argv, environment) + const result = await probeRehomeTrust(config, { + fetch: async (url, init) => { + assert.equal(url, 'https://relay.onorca.dev/v1/admin/regional-rehome-trust-probe') + assert.deepEqual(JSON.parse(init.body), { + v: 1, + sourceCellId: 'production-gce-c7', + sourceCellIncarnation: '11111111-1111-4111-8111-111111111111' + }) + return Response.json({ + v: 1, + dedicatedIdentity: { + firstOutcome: 'host-not-connected', + secondOutcome: 'host-not-connected', + accepted: true, + idempotent: true + }, + sharedRuntimeIdentityRejected: true, + proven: true + }) + } + }) + assert.equal(result.proven, true) +}) + +test('rejects partial or mismatched proof', async () => { + const config = parseRehomeTrustProbeArguments(argv, environment) + await assert.rejects( + probeRehomeTrust(config, { + fetch: async () => Response.json({ + v: 1, + dedicatedIdentity: { + firstOutcome: 'host-not-connected', + secondOutcome: 'host-not-connected', + accepted: true, + idempotent: true + }, + sharedRuntimeIdentityRejected: false, + proven: false + }) + }), + /incomplete/ + ) +}) diff --git a/cloud/dev/scripts/production-cell-image-digest-consistency.test.mjs b/cloud/dev/scripts/production-cell-image-digest-consistency.test.mjs new file mode 100644 index 00000000000..612d48abdcd --- /dev/null +++ b/cloud/dev/scripts/production-cell-image-digest-consistency.test.mjs @@ -0,0 +1,85 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { PRODUCTION_CAPACITY_CELL_IDS } from './prepare-relay-production-capacity-canary.mjs' +import { readRelayWorkflow } from './relay-repository.mjs' + +const production = source('infra/terraform/environments/production.tfvars') +const dispatchWorkflow = readRelayWorkflow('deploy-relay-production-capacity.yml') +const jobWorkflow = readRelayWorkflow('deploy-relay-production-capacity-job.yml') + +const RELAY_REPOSITORY = 'us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay' + +function source(path) { + return readFileSync(new URL(`../../${path}`, import.meta.url), 'utf8') +} + +// Slice each "" = { ... } entry out of relay_gce_cells. +function productionCells() { + const block = production.slice(production.indexOf('relay_gce_cells = {')) + const cells = new Map() + for (const match of block.matchAll(/"(production-gce-c\d+)" = \{([\s\S]*?)\n {2}\}/g)) { + cells.set(match[1], match[2]) + } + assert.ok(cells.size > 0, 'relay_gce_cells parsed empty') + return cells +} + +function hardCap(body) { + const match = body.match(/connection_hard_cap\s*=\s*(\d+)/) + return match ? Number(match[1]) : undefined +} + +function imageDigest(body) { + const match = body.match(/^\s*image\s*=\s*"([^"]+)"/m) + assert.ok(match, 'cell entry has no image') + const [repository, digest] = match[1].split('@') + assert.equal(repository, RELAY_REPOSITORY) + assert.match(digest, /^sha256:[0-9a-f]{64}$/) + return digest +} + +function workflowPin(workflow, name) { + const match = workflow.match(new RegExp(`${name}: (sha256:[0-9a-f]{64})`)) + assert.ok(match, `${name} is missing or not a full digest`) + return match[1] +} + +test('every production cell pins a full relay image digest', () => { + for (const [cellId, body] of productionCells()) { + assert.match(imageDigest(body), /^sha256:[0-9a-f]{64}$/, `${cellId} image digest`) + } +}) + +test('the 1,000-cap cells are exactly the canonical capacity set', () => { + const thousandCap = [...productionCells()] + .filter(([, body]) => hardCap(body) === 1000) + .map(([cellId]) => cellId) + assert.deepEqual([...thousandCap].sort(), [...PRODUCTION_CAPACITY_CELL_IDS].sort()) +}) + +test('the 1,000-cap cells all serve one image digest', () => { + const digests = new Map() + for (const [cellId, body] of productionCells()) { + if (hardCap(body) !== 1000) continue + const digest = imageDigest(body) + if (!digests.has(digest)) digests.set(digest, []) + digests.get(digest).push(cellId) + } + assert.equal( + digests.size, + 1, + `1,000-cap cells split across digests: ${JSON.stringify([...digests])}` + ) + assert.equal([...digests.values()][0].length, PRODUCTION_CAPACITY_CELL_IDS.length) +}) + +// COMPATIBLE_CELL_IMAGE_DIGEST is one half of a reviewed (director, cell) skew pair, not a +// claim about what the fleet serves; it is re-derived by hand for each capacity wave. So it +// is deliberately NOT tied to the tfvars digest — only to its twin in the dispatch workflow. +test('both capacity workflows declare the same reviewed image pins', () => { + for (const name of ['PREDECESSOR_IMAGE_DIGEST', 'COMPATIBLE_CELL_IMAGE_DIGEST']) { + assert.equal(workflowPin(dispatchWorkflow, name), workflowPin(jobWorkflow, name), name) + } + workflowPin(jobWorkflow, 'COMPATIBLE_DIRECTOR_IMAGE_DIGEST') +}) diff --git a/cloud/dev/scripts/production-cloud-sql-rollout-lock.test.mjs b/cloud/dev/scripts/production-cloud-sql-rollout-lock.test.mjs new file mode 100644 index 00000000000..d79d4f91046 --- /dev/null +++ b/cloud/dev/scripts/production-cloud-sql-rollout-lock.test.mjs @@ -0,0 +1,210 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { + LEASED_WORKFLOWS, + LOCK_GROUPS, + NOT_A_CLOUD_SQL_CANDIDATE, + PRODUCTION_LEASE, + SELECTABLE_LEASE, + STAGING_LEASE, + concurrencyBlocks, + entrypointsFor, + jobIf, + jobNeeds, + jobs, + leaseSteps, + leaseStepsByJob, + mutatesSharedInstance, + readWorkflow, + reusableCalls, + revisionMintingScripts, + workflowFiles +} from './cloud-sql-rollout-lock-census.mjs' +import { relayWorkflowFile } from './relay-repository.mjs' + +const expectedLease = { production: PRODUCTION_LEASE, staging: STAGING_LEASE, selectable: SELECTABLE_LEASE } +const leasedFiles = Object.keys(LEASED_WORKFLOWS) + +function contractFiles(file) { + return [file, ...(LEASED_WORKFLOWS[file].leaseFiles ?? [])] +} + +test('every locked workflow declares exactly one lock group that never cancels', () => { + for (const file of leasedFiles) { + const blocks = concurrencyBlocks(readWorkflow(file)) + assert.equal(blocks.length, 1, `${file} must declare exactly one concurrency block`) + assert.equal(blocks[0].group, LEASED_WORKFLOWS[file].group, file) + assert.equal(blocks[0].cancelInProgress, 'false', file) + } +}) + +test('every locked workflow takes the lease for its own environment', () => { + for (const file of leasedFiles) { + const lease = expectedLease[LEASED_WORKFLOWS[file].env] + const steps = contractFiles(file).flatMap((member) => leaseSteps(readWorkflow(member))) + assert.ok(steps.length > 0, `${file} must use the Cloud SQL rollout lease action`) + for (const step of steps) { + assert.equal(step.bucket, lease.bucket, file) + assert.equal(step.object, lease.object, file) + } + } +}) + +test('the lease step runs after the credential that authorizes it', () => { + for (const file of leasedFiles) { + for (const member of contractFiles(file)) { + const text = readWorkflow(member) + const lines = text.split('\n') + for (const step of leaseSteps(text)) { + const before = lines.slice(0, step.line - 1) + const gcloud = before.lastIndexOf(' - uses: google-github-actions/setup-gcloud@v2') + assert.notEqual(gcloud, -1, `${member}: lease step at line ${step.line} has no setup-gcloud before it`) + assert.ok( + before.lastIndexOf(' - uses: actions/checkout@v4') !== -1, + `${member}: lease step at line ${step.line} runs before the local action is checked out` + ) + } + } + } +}) + +test('multi-wave workflows hold one lease per run and free it exactly once', () => { + for (const file of leasedFiles) { + const entry = LEASED_WORKFLOWS[file] + const waveFiles = entry.leaseFiles ?? [] + const callCount = waveFiles.reduce( + (total, member) => total + (reusableCalls(readWorkflow(file)).get(member) ?? 0), + 0 + ) + if (!entry.reentrant) { + assert.ok(callCount <= 1, `${file} calls a leased reusable job ${callCount} times; it needs a release job`) + for (const member of contractFiles(file)) { + for (const step of leaseSteps(readWorkflow(member))) { + assert.equal(step.release, undefined, `${member} must leave release at its default`) + } + } + continue + } + + assert.ok(callCount > 1, `${file} no longer calls its reusable job more than once`) + const steps = contractFiles(file).flatMap((member) => leaseSteps(readWorkflow(member))) + const released = steps.filter((step) => step.release === "'true'") + assert.equal(released.length, 1, `${file} must free the run lease exactly once`) + for (const step of steps) { + if (step === released[0]) continue + assert.equal(step.release, "'false'", `${file} wave jobs must hold the lease`) + } + + const callerJobs = jobs(readWorkflow(file)) + const releaseJob = leaseStepsByJob(file).find((job) => + job.steps.some((step) => step.release === "'true'") + ) + assert.ok(releaseJob, `${file} must free the lease from its own job`) + const guard = jobIf(callerJobs.find((job) => job.id === releaseJob.id).text) + assert.match(guard, /always\(\)/, `${file}: the release job must run on failure and cancellation`) + + const holders = callerJobs + .filter( + (job) => + job.id !== releaseJob.id && + (waveFiles.some((member) => job.text.includes(`uses: ./.github/workflows/${member}`)) || + leaseStepsByJob(file).find((entry) => entry.id === job.id)?.steps.length > 0) + ) + .map((job) => job.id) + const needs = jobNeeds(callerJobs.find((job) => job.id === releaseJob.id).text) + for (const holder of holders) { + assert.ok(needs.includes(holder), `${file}: the release job must need ${holder}`) + } + } +}) + +test('workflows with two lease-holding jobs can never run them together', () => { + for (const file of leasedFiles) { + const entry = LEASED_WORKFLOWS[file] + if (entry.reentrant) continue + const holding = leaseStepsByJob(file).filter((job) => job.steps.length > 0) + if (holding.length <= 1) continue + assert.ok(entry.exclusiveBy, `${file} has ${holding.length} lease-holding jobs and no exclusivity guard`) + const guards = holding.map((job) => jobIf(jobs(readWorkflow(file)).find((j) => j.id === job.id).text)) + assert.equal( + guards.filter((guard) => guard.includes(entry.exclusiveBy)).length, + 1, + `${file}: exactly one job may run when ${entry.exclusiveBy}` + ) + assert.equal( + guards.filter((guard) => guard.includes(entry.exclusiveBy.replace('==', '!='))).length, + guards.length - 1, + `${file}: every other lease-holding job must be excluded when ${entry.exclusiveBy}` + ) + } +}) + +test('census: no workflow rolls out against the shared instance outside the lease', () => { + const minters = revisionMintingScripts() + assert.ok(minters.size > 0, 'the revision-minting script scan found nothing and is vacuous') + const flagged = new Map() + for (const file of workflowFiles()) { + const reason = mutatesSharedInstance(readWorkflow(file), minters) + if (reason) flagged.set(file, reason) + } + assert.ok(flagged.size > 0, 'the rollout census found no candidates and is vacuous') + + for (const [file, reason] of flagged) { + for (const entrypoint of entrypointsFor(file)) { + assert.ok( + entrypoint in LEASED_WORKFLOWS || entrypoint in NOT_A_CLOUD_SQL_CANDIDATE, + `${entrypoint} reaches ${file} (${reason}) but is neither leased nor recorded as a non-candidate` + ) + } + } + + for (const file of workflowFiles()) { + const groups = concurrencyBlocks(readWorkflow(file)).map((block) => block.group) + if (!groups.some((group) => LOCK_GROUPS.has(group))) continue + assert.ok( + file in LEASED_WORKFLOWS || file in NOT_A_CLOUD_SQL_CANDIDATE, + `${file} sits in a Cloud SQL lock group but is neither leased nor recorded as a non-candidate` + ) + } + + for (const [file, reason] of Object.entries(NOT_A_CLOUD_SQL_CANDIDATE)) { + assert.ok(typeof reason === 'string' && reason.length > 40, `${file} needs a real reason`) + const groups = concurrencyBlocks(readWorkflow(file)).map((block) => block.group) + assert.ok( + flagged.has(file) || groups.some((group) => LOCK_GROUPS.has(group)), + `${file} is recorded as a non-candidate but nothing would have flagged it` + ) + } + + for (const file of leasedFiles) { + assert.doesNotThrow(() => readWorkflow(file), `${file} is leased but does not exist`) + assert.ok(!(file in NOT_A_CLOUD_SQL_CANDIDATE), `${file} cannot be both leased and a non-candidate`) + } +}) + +// The API and auth deploy scripts share this contract but stay in the private repository. +const serviceCapScripts = ['dev/scripts/deploy-relay-blue-green.mjs'] + +test('budgets tagged Cloud Run candidates outside the service-wide instance cap', () => { + for (const file of serviceCapScripts) { + const script = readFileSync(new URL(`../../${file}`, import.meta.url), 'utf8') + assert.match(script, /'--no-traffic'/, file) + assert.match(script, /'--max'/, file) + } + const budget = readFileSync( + new URL('../../dev/scripts/relay-cloud-sql-connection-budget.mjs', import.meta.url), + 'utf8' + ) + assert.match(budget, /directly addressable tagged revisions outside service-level caps/) + assert.match( + budget, + /apiCandidate: retainedDirectorRollback \+ inputs\.apiInstances \* inputs\.apiPoolMax/ + ) + const director = readWorkflow(relayWorkflowFile('deploy-relay-production-director.yml')) + const capacity = readWorkflow(relayWorkflowFile('deploy-relay-production-capacity-job.yml')) + const asia = readWorkflow(relayWorkflowFile('operate-relay-asia-admission.yml')) + assert.match(director, /--max-instances "\$\{DIRECTOR_MAX_INSTANCES\}"/) + assert.match(capacity, /--max-instances 5/) + assert.match(asia, /--max-instances "\$\{DIRECTOR_MAX_INSTANCES\}"/) +}) diff --git a/cloud/dev/scripts/read-relay-production-capacity-identity.mjs b/cloud/dev/scripts/read-relay-production-capacity-identity.mjs new file mode 100644 index 00000000000..764b2732cfb --- /dev/null +++ b/cloud/dev/scripts/read-relay-production-capacity-identity.mjs @@ -0,0 +1,31 @@ +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' + +const CAPACITY_IDENTITY_NAME = 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT' + +export function readProductionCapacityIdentity(revision) { + const env = revision?.spec?.containers?.[0]?.env + if (!Array.isArray(env)) throw new Error('director revision environment is missing') + const matches = env.filter((entry) => entry?.name === CAPACITY_IDENTITY_NAME) + if (matches.length === 0) return null + if (matches.length !== 1) throw new Error('duplicate capacity identity') + const value = matches[0]?.value + if (typeof value !== 'string' || value.length === 0) { + throw new Error('capacity identity is not a literal string') + } + return value +} + +export function main() { + const revision = JSON.parse(readFileSync(0, 'utf8')) + process.stdout.write(`${JSON.stringify(readProductionCapacityIdentity(revision))}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + try { + main() + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + } +} diff --git a/cloud/dev/scripts/read-relay-production-capacity-identity.test.mjs b/cloud/dev/scripts/read-relay-production-capacity-identity.test.mjs new file mode 100644 index 00000000000..c75ddcbaf77 --- /dev/null +++ b/cloud/dev/scripts/read-relay-production-capacity-identity.test.mjs @@ -0,0 +1,44 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { readProductionCapacityIdentity } from './read-relay-production-capacity-identity.mjs' + +function revision(env) { + return { spec: { containers: [{ env }] } } +} + +test('reads absent, exact, and foreign literal capacity identities', () => { + assert.equal(readProductionCapacityIdentity(revision([])), null) + assert.equal( + readProductionCapacityIdentity(revision([ + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', value: 'capacity@example.test' } + ])), + 'capacity@example.test' + ) + assert.equal( + readProductionCapacityIdentity(revision([ + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', value: 'foreign@example.test' } + ])), + 'foreign@example.test' + ) +}) + +test('rejects malformed or duplicate capacity identity entries', () => { + for (const entry of [ + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', value: null }, + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', value: '' }, + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', valueSource: { secretKeyRef: {} } } + ]) { + assert.throws( + () => readProductionCapacityIdentity(revision([entry])), + /not a literal string/ + ) + } + assert.throws( + () => readProductionCapacityIdentity(revision([ + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', value: 'one@example.test' }, + { name: 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT', value: 'two@example.test' } + ])), + /duplicate capacity identity/ + ) + assert.throws(() => readProductionCapacityIdentity({}), /environment is missing/) +}) diff --git a/cloud/dev/scripts/read-relay-serving-regional-placement-version.mjs b/cloud/dev/scripts/read-relay-serving-regional-placement-version.mjs new file mode 100644 index 00000000000..80d40277e5a --- /dev/null +++ b/cloud/dev/scripts/read-relay-serving-regional-placement-version.mjs @@ -0,0 +1,106 @@ +import { spawnSync } from 'node:child_process' +import { fileURLToPath } from 'node:url' + +const SECRET = 'orca-cloud-relay-regional-placement-enabled' + +function validate(input) { + for (const key of ['project', 'region', 'service', 'bootstrap_version']) { + if (typeof input?.[key] !== 'string' || !input[key] || /[\r\n]/.test(input[key])) { + throw new Error(`invalid ${key}`) + } + } + if (!/^[a-z][a-z0-9-]{0,62}$/.test(input.service)) throw new Error('invalid service') + if (!/^[1-9][0-9]*$/.test(input.bootstrap_version)) { + throw new Error('invalid bootstrap_version') + } +} + +export function classifyRelayServiceDescribeFailure(args, stderr) { + const serviceDescribe = args[0] === 'run' && args[1] === 'services' && args[2] === 'describe' + if (serviceDescribe && (stderr.includes('NOT_FOUND') || /Cannot find service \[[^\]\r\n]+\]/.test(stderr))) { + return 'NOT_FOUND' + } + return 'GCLOUD_FAILED' +} + +function defaultRun(args) { + const result = spawnSync('gcloud', args, { + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'], + maxBuffer: 10 * 1024 * 1024 + }) + if (result.status !== 0) { + const error = new Error('gcloud read failed') + error.code = classifyRelayServiceDescribeFailure(args, result.stderr) + throw error + } + return JSON.parse(result.stdout) +} + +function gcloudArguments(kind, input, revision) { + return [ + 'run', kind, 'describe', revision ?? input.service, + '--project', input.project, + '--region', input.region, + '--format=json' + ] +} + +export function readRelayServingRegionalPlacementVersion(input, dependencies = {}) { + validate(input) + const run = dependencies.run ?? defaultRun + let service + try { + service = run(gcloudArguments('services', input)) + } catch (error) { + if (error?.code === 'NOT_FOUND') return { version: input.bootstrap_version } + throw error + } + const serving = (service.status?.traffic ?? []).filter( + (entry) => Number(entry.percent ?? 0) > 0 + ) + if ( + serving.length !== 1 || + Number(serving[0].percent) !== 100 || + typeof serving[0].revisionName !== 'string' + ) { + throw new Error('Relay director must have exactly one revision serving 100% traffic') + } + const revision = run(gcloudArguments('revisions', input, serving[0].revisionName)) + const references = (revision.spec?.containers ?? []).flatMap((container) => + (container.env ?? []).filter( + (environment) => environment.name === 'ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED' + ) + ) + if (references.length === 0) return { version: input.bootstrap_version } + const reference = normalizeSecretReference(references[0]) + if ( + references.length !== 1 || + reference?.secret !== SECRET || + !/^[1-9][0-9]*$/.test(reference?.version ?? '') + ) { + throw new Error('serving regional placement secret reference is invalid') + } + return { version: reference.version } +} + +// Why: the v2 API reports `valueSource.secretKeyRef.{secret,version}`, but +// `gcloud run revisions describe --format=json` emits the Knative v1 shape +// `valueFrom.secretKeyRef.{name,key}`, where `name` may be a full resource path. +// A `key` of "latest" is deliberately left invalid: the director's serving +// version must be a pinned integer for this data source to mean anything. +export function normalizeSecretReference(environment) { + const v2 = environment?.valueSource?.secretKeyRef + if (v2) return { secret: v2.secret, version: v2.version } + const v1 = environment?.valueFrom?.secretKeyRef + if (!v1) return undefined + const secret = typeof v1.name === 'string' ? v1.name.replace(/^projects\/[^/]+\/secrets\//, '') : undefined + return { secret, version: v1.key } +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const chunks = [] + for await (const chunk of process.stdin) chunks.push(chunk) + const input = JSON.parse(Buffer.concat(chunks).toString('utf8')) + process.stdout.write(`${JSON.stringify(readRelayServingRegionalPlacementVersion(input))}\n`) +} diff --git a/cloud/dev/scripts/read-relay-serving-regional-placement-version.test.mjs b/cloud/dev/scripts/read-relay-serving-regional-placement-version.test.mjs new file mode 100644 index 00000000000..8043fc23e94 --- /dev/null +++ b/cloud/dev/scripts/read-relay-serving-regional-placement-version.test.mjs @@ -0,0 +1,138 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + classifyRelayServiceDescribeFailure, + readRelayServingRegionalPlacementVersion +} from './read-relay-serving-regional-placement-version.mjs' + +const input = { + project: 'onorca-cloud', + region: 'us-central1', + service: 'orca-cloud-relay', + bootstrap_version: '7' +} + +function revision(version = '11') { + return { + spec: { + containers: [{ + env: [{ + name: 'ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED', + valueSource: { + secretKeyRef: { + secret: 'orca-cloud-relay-regional-placement-enabled', + version + } + } + }] + }] + } + } +} + +// Why: `gcloud run revisions describe --format=json` emits the Knative v1 shape, where the +// secret lives in `name` and the version in `key`, and `name` may be the full resource path. +function v1Revision(name, key) { + return { + spec: { + containers: [{ + env: [{ + name: 'ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED', + valueFrom: { secretKeyRef: { name, key } } + }] + }] + } + } +} + +function serving() { + return { status: { traffic: [{ revisionName: 'relay-serving', percent: 100 }] } } +} + +test('reads the exact version from the sole traffic-serving revision', () => { + const calls = [] + const result = readRelayServingRegionalPlacementVersion(input, { + run: (args) => { + calls.push(args) + return calls.length === 1 + ? { + status: { + traffic: [ + { revisionName: 'relay-failed-latest', tag: 'candidate' }, + { revisionName: 'relay-serving', percent: 100 } + ] + } + } + : revision() + } + }) + + assert.deepEqual(result, { version: '11' }) + assert.equal(calls[1][3], 'relay-serving') +}) + +test('reads the gcloud v1 secret reference shape by bare id and by full resource path', () => { + for (const name of [ + 'orca-cloud-relay-regional-placement-enabled', + 'projects/120364513935/secrets/orca-cloud-relay-regional-placement-enabled' + ]) { + assert.deepEqual(readRelayServingRegionalPlacementVersion(input, { + run: (args) => args[1] === 'services' ? serving() : v1Revision(name, '1') + }), { version: '1' }) + } +}) + +test('rejects a v1 reference that names another secret or a floating version', () => { + assert.throws(() => readRelayServingRegionalPlacementVersion(input, { + run: (args) => args[1] === 'services' + ? serving() + : v1Revision('projects/120364513935/secrets/some-other-secret', '1') + }), /secret reference is invalid/) + assert.throws(() => readRelayServingRegionalPlacementVersion(input, { + run: (args) => args[1] === 'services' + ? serving() + : v1Revision('orca-cloud-relay-regional-placement-enabled', 'latest') + }), /secret reference is invalid/) +}) + +test('falls back only when the service or setting is absent', () => { + const notFound = new Error('not found') + notFound.code = 'NOT_FOUND' + assert.deepEqual(readRelayServingRegionalPlacementVersion(input, { + run: () => { throw notFound } + }), { version: '7' }) + assert.deepEqual(readRelayServingRegionalPlacementVersion(input, { + run: (args) => args[1] === 'services' + ? { status: { traffic: [{ revisionName: 'relay-serving', percent: 100 }] } } + : { spec: { containers: [{ env: [] }] } } + }), { version: '7' }) +}) + +test('classifies real absent-service stderr without weakening revision failures', () => { + const serviceArgs = ['run', 'services', 'describe', 'missing-service'] + const stderr = 'ERROR: (gcloud.run.services.describe) Cannot find service [missing-service]' + assert.equal(classifyRelayServiceDescribeFailure(serviceArgs, stderr), 'NOT_FOUND') + assert.equal( + classifyRelayServiceDescribeFailure(['run', 'revisions', 'describe', 'missing-revision'], stderr), + 'GCLOUD_FAILED' + ) + assert.equal(classifyRelayServiceDescribeFailure(serviceArgs, 'PERMISSION_DENIED'), 'GCLOUD_FAILED') +}) + +test('rejects ambiguous traffic, malformed references, and read failures', () => { + assert.throws(() => readRelayServingRegionalPlacementVersion(input, { + run: () => ({ + status: { traffic: [{ revisionName: 'a', percent: 50 }, { revisionName: 'b', percent: 50 }] } + }) + }), /exactly one revision/) + assert.throws(() => readRelayServingRegionalPlacementVersion(input, { + run: (args) => args[1] === 'services' + ? { status: { traffic: [{ revisionName: 'relay-serving', percent: 100 }] } } + : revision('latest') + }), /secret reference is invalid/) + const denied = new Error('denied') + denied.code = 'GCLOUD_FAILED' + assert.throws(() => readRelayServingRegionalPlacementVersion(input, { + run: () => { throw denied } + }), denied) +}) diff --git a/cloud/dev/scripts/relay-admission-selector.mjs b/cloud/dev/scripts/relay-admission-selector.mjs new file mode 100644 index 00000000000..f41528c7e7f --- /dev/null +++ b/cloud/dev/scripts/relay-admission-selector.mjs @@ -0,0 +1,277 @@ +import { createHash } from 'node:crypto' + +const STATES = ['existing-only', 'migration-only', 'general'] + +function normalizeMembership(input) { + const membership = { + existingOnly: [...input.existingOnly].sort(), + migrationOnly: [...input.migrationOnly].sort(), + general: [...input.general].sort() + } + const all = [...membership.existingOnly, ...membership.migrationOnly, ...membership.general] + if (new Set(all).size !== all.length) throw new Error('selector membership contains duplicates') + return membership +} + +function encodedMembership(membership) { + return JSON.stringify(normalizeMembership(membership)) +} + +function membershipSha256(membership) { + return createHash('sha256').update(encodedMembership(membership)).digest('hex') +} + +function normalizeMigrationCells(input) { + const cells = [...input] + .map((cell) => ({ + cellId: cell.cellId, + cellUrl: cell.cellUrl, + capacityRequests: cell.capacityRequests, + ...(cell.region ? { region: cell.region } : {}), + connectionHardCap: cell.connectionHardCap, + connectionUnobservedBound: cell.connectionUnobservedBound + })) + .sort((left, right) => left.cellId.localeCompare(right.cellId)) + if ( + cells.length === 0 || + new Set(cells.map(({ cellId }) => cellId)).size !== cells.length || + new Set(cells.map(({ cellUrl }) => cellUrl)).size !== cells.length + ) { + throw new Error('migration cell registration must contain distinct cells') + } + return cells +} + +function membershipWithMigrationCells(membership, cells) { + const known = new Set([ + ...membership.existingOnly, + ...membership.migrationOnly, + ...membership.general + ]) + if (cells.some(({ cellId }) => known.has(cellId))) { + throw new Error('migration cell registration contains an existing selector cell') + } + return normalizeMembership({ + existingOnly: membership.existingOnly, + migrationOnly: [...membership.migrationOnly, ...cells.map(({ cellId }) => cellId)], + general: membership.general + }) +} + +function assertSelector(value) { + if ( + !value || + !Number.isSafeInteger(value.generation) || + value.generation < 0 || + !value.membership + ) { + throw new Error('director returned an invalid admission selector') + } + return { + generation: value.generation, + attemptId: value.attemptId ?? null, + membership: normalizeMembership(value.membership) + } +} + +export function selectorAttemptId(expectedGeneration, membership) { + const digest = createHash('sha256') + .update(`${expectedGeneration}:${encodedMembership(membership)}`) + .digest('hex') + .slice(0, 24) + return `selector_${expectedGeneration}_${digest}` +} + +export function membershipWithStates(selector, states) { + const byCell = new Map() + for (const [state, key] of [ + ['existing-only', 'existingOnly'], + ['migration-only', 'migrationOnly'], + ['general', 'general'] + ]) { + for (const cellId of selector.membership[key]) byCell.set(cellId, state) + } + for (const [cellId, state] of Object.entries(states)) { + if (!byCell.has(cellId)) throw new Error(`selector does not contain ${cellId}`) + if (!STATES.includes(state)) throw new Error(`invalid admission state for ${cellId}`) + if (byCell.get(cellId) === 'existing-only' && state !== 'existing-only') { + throw new Error(`selector cannot re-enable existing-only cell ${cellId}`) + } + byCell.set(cellId, state) + } + return normalizeMembership({ + existingOnly: [...byCell].filter(([, state]) => state === 'existing-only').map(([id]) => id), + migrationOnly: [...byCell].filter(([, state]) => state === 'migration-only').map(([id]) => id), + general: [...byCell].filter(([, state]) => state === 'general').map(([id]) => id) + }) +} + +export async function inspectAdmissionSelector(post, attemptId) { + const result = await post('/v1/admin/admission-selector/status', { + v: 1, + ...(attemptId ? { attemptId } : {}) + }) + return { + selector: assertSelector(result.selector), + intent: result.intent + ? { + ...result.intent, + previousMembership: result.intent.previousMembership + ? normalizeMembership(result.intent.previousMembership) + : undefined, + membership: normalizeMembership(result.intent.membership) + } + : null + } +} + +function exactSelector(actual, expected) { + return ( + actual.generation === expected.generation && + encodedMembership(actual.membership) === encodedMembership(expected.membership) + ) +} + +export async function applyExactAdmissionSelector(post, membership, options = {}) { + const before = await inspectAdmissionSelector(post) + const desired = normalizeMembership(membership) + if ( + options.expectedCurrentSelector && + !exactSelector(before.selector, options.expectedCurrentSelector) + ) { + throw new Error('admission selector changed before exact apply') + } + if (options.requireBoundary !== false && before.selector.generation < 1) { + throw new Error('admission selector boundary is not active') + } + if (encodedMembership(before.selector.membership) === encodedMembership(desired)) { + return { changed: false, selector: before.selector } + } + const attemptId = + options.attemptId ?? selectorAttemptId(before.selector.generation, desired) + const expected = { + generation: before.selector.generation + 1, + membership: desired + } + let result + try { + result = await post('/v1/admin/admission-selector/apply', { + v: 1, + attemptId, + expectedGeneration: before.selector.generation, + ...(before.selector.generation === 0 + ? { expectedMembershipSha256: membershipSha256(before.selector.membership) } + : {}), + membership: desired + }) + } catch (error) { + const inspected = await inspectAdmissionSelector(post, attemptId) + if ( + inspected.intent?.state === 'committed' && + exactSelector(inspected.selector, expected) + ) { + return { changed: true, selector: inspected.selector, recovered: true } + } + if ( + inspected.intent?.state === 'unchanged' && + exactSelector(inspected.selector, before.selector) + ) { + throw new Error('admission selector apply remained unchanged after an ambiguous response', { + cause: error + }) + } + throw new Error('admission selector apply diverged after an ambiguous response', { + cause: error + }) + } + const applied = assertSelector(result.selector) + if (!exactSelector(applied, expected)) { + throw new Error('admission selector apply returned unexpected membership') + } + const verified = await inspectAdmissionSelector(post, attemptId) + if ( + verified.intent?.state !== 'committed' || + !exactSelector(verified.selector, expected) + ) { + throw new Error('admission selector commit could not be verified') + } + return { changed: result.changed === true, selector: verified.selector } +} + +export async function addExactMigrationCells(post, input, options = {}) { + const cells = normalizeMigrationCells(input.cells) + const attemptId = input.attemptId + if (!/^[A-Za-z0-9_-]{8,128}$/.test(attemptId ?? '')) { + throw new Error('migration cell registration requires an exact attempt ID') + } + const before = await inspectAdmissionSelector(post, attemptId) + let expectedGeneration + let expectedMembership + if (before.intent) { + expectedGeneration = before.intent.expectedGeneration + expectedMembership = normalizeMembership(before.intent.membership) + } else { + if (before.selector.generation < 1) { + throw new Error('admission selector boundary is not active') + } + if ( + options.expectedCurrentSelector && + !exactSelector(before.selector, options.expectedCurrentSelector) + ) { + throw new Error('admission selector changed before cell registration') + } + expectedGeneration = before.selector.generation + expectedMembership = membershipWithMigrationCells(before.selector.membership, cells) + } + const expected = { + generation: expectedGeneration + 1, + membership: expectedMembership + } + let result + try { + result = await post('/v1/admin/admission-selector/add-migration-cells', { + v: 1, + attemptId, + expectedGeneration, + cells + }) + } catch (error) { + const inspected = await inspectAdmissionSelector(post, attemptId) + if ( + !before.intent && + inspected.intent?.state === 'committed' && + exactSelector(inspected.selector, expected) + ) { + return { changed: true, selector: inspected.selector, recovered: true } + } + throw new Error('migration cell registration did not commit exactly', { cause: error }) + } + const applied = assertSelector(result.selector) + if (!exactSelector(applied, expected)) { + throw new Error('migration cell registration returned unexpected membership') + } + const verified = await inspectAdmissionSelector(post, attemptId) + if (verified.intent?.state !== 'committed' || !exactSelector(verified.selector, expected)) { + throw new Error('migration cell registration commit could not be verified') + } + return { changed: result.changed === true, selector: verified.selector } +} + +export async function transitionAdmissionSelector(post, states, options = {}) { + const current = await inspectAdmissionSelector(post) + if (current.selector.generation < 1) { + throw new Error('admission selector boundary is not active') + } + return await applyExactAdmissionSelector( + post, + membershipWithStates(current.selector, states), + options + ) +} + +export function selectorCellState(selector, cellId) { + if (selector.membership.existingOnly.includes(cellId)) return 'existing-only' + if (selector.membership.migrationOnly.includes(cellId)) return 'migration-only' + if (selector.membership.general.includes(cellId)) return 'general' + throw new Error(`selector does not contain ${cellId}`) +} diff --git a/cloud/dev/scripts/relay-admission-selector.test.mjs b/cloud/dev/scripts/relay-admission-selector.test.mjs new file mode 100644 index 00000000000..b45df5ac34b --- /dev/null +++ b/cloud/dev/scripts/relay-admission-selector.test.mjs @@ -0,0 +1,232 @@ +import assert from 'node:assert/strict' +import { createHash } from 'node:crypto' +import { test } from 'node:test' +import { + addExactMigrationCells, + applyExactAdmissionSelector, + membershipWithStates, + selectorAttemptId, + transitionAdmissionSelector +} from './relay-admission-selector.mjs' + +const initialMembership = { + existingOnly: ['legacy'], + migrationOnly: ['target'], + general: ['general'] +} + +function selectorHarness({ ambiguous = null, generation = 1 } = {}) { + let selector = { generation, attemptId: 'initial', membership: initialMembership } + const intents = new Map() + const requests = [] + let applies = 0 + const post = async (path, body) => { + if (path.endsWith('/status')) { + return { + selector, + intent: body.attemptId ? intents.get(body.attemptId) ?? null : null + } + } + applies++ + requests.push(body) + const before = selector + const committed = { + generation: body.expectedGeneration + 1, + attemptId: body.attemptId, + membership: body.membership + } + intents.set(body.attemptId, { + attemptId: body.attemptId, + expectedGeneration: body.expectedGeneration, + intendedGeneration: committed.generation, + previousMembership: before.membership, + membership: body.membership, + state: ambiguous === 'unchanged' ? 'unchanged' : 'committed' + }) + if (ambiguous !== 'unchanged') selector = committed + if (ambiguous) throw new Error('lost selector response') + return { changed: true, selector } + } + return { post, selector: () => selector, applies: () => applies, requests } +} + +test('derives deterministic attempts and applies exact selector transitions', async () => { + const harness = selectorHarness() + const desired = membershipWithStates(harness.selector(), { target: 'general' }) + assert.equal( + selectorAttemptId(1, desired), + selectorAttemptId(1, { + existingOnly: ['legacy'], + migrationOnly: [], + general: ['target', 'general'] + }) + ) + const result = await transitionAdmissionSelector(harness.post, { target: 'general' }) + assert.equal(result.selector.generation, 2) + assert.deepEqual(result.selector.membership.general, ['general', 'target']) +}) + +test('accepts only an exact committed result after an ambiguous response', async () => { + const harness = selectorHarness({ ambiguous: 'committed' }) + const result = await applyExactAdmissionSelector(harness.post, { + existingOnly: ['legacy', 'target'], + migrationOnly: [], + general: ['general'] + }) + assert.equal(result.recovered, true) + assert.equal(harness.applies(), 1) + assert.equal(result.selector.generation, 2) +}) + +test('binds a generation-zero cutover to the inspected membership', async () => { + const harness = selectorHarness({ generation: 0 }) + await applyExactAdmissionSelector( + harness.post, + { + existingOnly: ['legacy', 'target'], + migrationOnly: [], + general: ['general'] + }, + { requireBoundary: false } + ) + assert.equal( + harness.requests[0].expectedMembershipSha256, + createHash('sha256').update(JSON.stringify(initialMembership)).digest('hex') + ) +}) + +test('stops on an unchanged ambiguous result without replaying', async () => { + const harness = selectorHarness({ ambiguous: 'unchanged' }) + await assert.rejects( + applyExactAdmissionSelector(harness.post, { + existingOnly: ['legacy', 'target'], + migrationOnly: [], + general: ['general'] + }), + /remained unchanged/ + ) + assert.equal(harness.applies(), 1) + assert.equal(harness.selector().generation, 1) +}) + +test('never restores an existing-only cell', () => { + assert.throws( + () => membershipWithStates({ membership: initialMembership }, { legacy: 'general' }), + /cannot re-enable/ + ) +}) + +test('refuses an exact apply after the inspected selector changes', async () => { + const harness = selectorHarness() + await assert.rejects( + applyExactAdmissionSelector( + harness.post, + { + existingOnly: ['legacy', 'target'], + migrationOnly: [], + general: ['general'] + }, + { + expectedCurrentSelector: { + generation: 0, + membership: { + existingOnly: ['target'], + migrationOnly: [], + general: ['general', 'legacy'] + } + } + } + ), + /changed before exact apply/ + ) + assert.equal(harness.applies(), 0) +}) + +test('adds exact migration cells and recovers a committed response loss', async () => { + let selector = { generation: 1, attemptId: 'initial', membership: initialMembership } + const intents = new Map() + let additions = 0 + const post = async (path, body) => { + if (path.endsWith('/status')) { + return { + selector, + intent: body.attemptId ? intents.get(body.attemptId) ?? null : null + } + } + additions++ + selector = { + generation: body.expectedGeneration + 1, + attemptId: body.attemptId, + membership: { + ...selector.membership, + migrationOnly: [ + ...selector.membership.migrationOnly, + ...body.cells.map(({ cellId }) => cellId) + ].sort() + } + } + intents.set(body.attemptId, { + attemptId: body.attemptId, + expectedGeneration: body.expectedGeneration, + intendedGeneration: selector.generation, + membership: selector.membership, + state: 'committed' + }) + throw new Error('lost cell registration response') + } + const result = await addExactMigrationCells(post, { + attemptId: 'add_cells_exact', + cells: [ + { + cellId: 'target-2', + cellUrl: 'https://target-2.example.com', + capacityRequests: 4_000, + connectionHardCap: 600, + connectionUnobservedBound: 60 + } + ] + }) + assert.equal(result.recovered, true) + assert.equal(additions, 1) + assert.deepEqual(result.selector.membership.migrationOnly, ['target', 'target-2']) +}) + +test('does not recover an attempt owned by another selector operation', async () => { + const selector = { + generation: 2, + attemptId: 'selector_collision', + membership: initialMembership + } + const post = async (path, body) => { + if (path.endsWith('/status')) { + return { + selector, + intent: body.attemptId + ? { + attemptId: body.attemptId, + expectedGeneration: 1, + intendedGeneration: 2, + membership: initialMembership, + state: 'committed' + } + : null + } + } + throw new Error('admission_selector_attempt_mismatch') + } + await assert.rejects( + addExactMigrationCells(post, { + attemptId: 'selector_collision', + cells: [ + { + cellId: 'target-2', + cellUrl: 'https://target-2.example.com', + capacityRequests: 4_000, + connectionHardCap: 600, + connectionUnobservedBound: 60 + } + ] + }), + /did not commit exactly/ + ) +}) diff --git a/cloud/dev/scripts/relay-asia-admission-workflow.test.mjs b/cloud/dev/scripts/relay-asia-admission-workflow.test.mjs new file mode 100644 index 00000000000..21d712601d0 --- /dev/null +++ b/cloud/dev/scripts/relay-asia-admission-workflow.test.mjs @@ -0,0 +1,323 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { relayWorkflowUrl } from './relay-repository.mjs' + +const workflow = readFileSync( + relayWorkflowUrl('operate-relay-asia-admission.yml'), + 'utf8' +) +const iam = readFileSync(new URL('../../infra/terraform/relay-github-actions.tf', import.meta.url), 'utf8') +const stagingProof = readFileSync( + relayWorkflowUrl('prove-relay-asia-staging.yml'), + 'utf8' +) +const directorWorkflow = readFileSync( + relayWorkflowUrl('deploy-relay-production-director.yml'), + 'utf8' +) +const terraformReadme = readFileSync( + new URL('../../infra/terraform/README.md', import.meta.url), + 'utf8' +) +const proofIam = readFileSync( + new URL('../../infra/terraform/relay-asia-proof-iam.tf', import.meta.url), + 'utf8' +) +const relayTerraform = readFileSync( + new URL('../../infra/terraform/relay.tf', import.meta.url), + 'utf8' +) +const rolloutEvidence = readFileSync( + new URL('./relay-asia-rollout-evidence.mjs', import.meta.url), + 'utf8' +) +const admissionBudgets = readFileSync( + new URL('../../packages/relay-contract/src/admission-budgets.ts', import.meta.url), + 'utf8' +) + +test('offers the exact audited admission modes under the shared deployment lock', () => { + for (const mode of [ + 'inspect', 'initialize', 'verify', 'register', 'configure', 'promote', 'rollback' + ]) { + assert.match(workflow, new RegExp(`\\b${mode}\\b`)) + } + assert.match(workflow, /production-cloud-sql-rollout/) + assert.match(workflow, /relay-staging-mutation/) + assert.match(workflow, /selector-generation/) + assert.match(workflow, /selector-attempt-id/) +}) + +test('requires exact confirmations and uses the existing admin identity', () => { + assert.match(workflow, /INITIALIZE_ADMISSION_SELECTOR/) + assert.match(workflow, /REGISTER_ASIA_MIGRATION_ONLY/) + assert.match(workflow, /PROMOTE_ASIA_GENERAL/) + assert.match(workflow, /ROLLBACK_ASIA_MIGRATION_ONLY/) + assert.match(workflow, /CONFIGURE_ASIA_DIRECTOR/) + assert.match(workflow, /GCP_RELAY_DEPLOY_SERVICE_ACCOUNT/) + assert.match(workflow, /id_token_audience: \$\{\{ env\.DIRECTOR_ORIGIN \}\}\/v1\/admin\/drain/) + assert.match(iam, /"operate-relay-asia-admission\.yml"/) +}) + +test('discovers generation read-only and explicitly initializes only generation zero', () => { + assert.match(workflow, /leave empty only for inspect/) + assert.match(workflow, /test -z "\$\{EXPECTED_SELECTOR_GENERATION\}"/) + assert.match(workflow, /test "\$\{EXPECTED_SELECTOR_GENERATION\}" = 0/) + assert.match(workflow, /\^\(0\|\[1-9\]\[0-9\]\*\)\$/) + assert.match(workflow, /selector-membership-sha256/) + assert.match(workflow, /\^\[a-f0-9\]\{64\}\$/) + assert.match(workflow, /director-image-digest/) + assert.match(workflow, /\.spec\.containers\[0\]\.image == \$image/) +}) + +test('uploads one sanitized machine-readable admission result', () => { + assert.match(workflow, /sanitize-relay-asia-admission-result\.mjs/) + const upload = /- name: Upload sanitized admission result\n([\s\S]*?)(?=\n - name:)/ + .exec(workflow)?.[1] + assert.ok(upload) + assert.match( + upload, + /if: \$\{\{ inputs\.mode != 'configure' && steps\.admission-operation\.outcome == 'success' \}\}/ + ) + assert.match(upload, /uses: actions\/upload-artifact@v4/) + assert.match( + upload, + /relay-asia-admission-result-\$\{\{ github\.run_id \}\}-\$\{\{ github\.run_attempt \}\}/ + ) + assert.match(upload, /path: \$\{\{ runner\.temp \}\}\/relay-asia-admission-result\/result\.json/) + assert.match(upload, /if-no-files-found: error/) + assert.match(upload, /retention-days: 7/) + assert.ok( + workflow.indexOf('Upload sanitized admission result') > + workflow.indexOf('Upload immutable C27 canary evidence') + ) +}) + +test('binds selector operations and director configuration to reviewed implementations', () => { + assert.match(workflow, /operate-relay-asia-admission\.mjs/) + assert.match(workflow, /prepare-relay-asia-director-cells\.mjs/) + assert.match(workflow, /deploy-relay-blue-green\.mjs/) + assert.match(workflow, /--prune-revisions false/) + assert.doesNotMatch(workflow, /gcloud secrets versions add/) + assert.match(workflow, /orca-cloud-relay-regional-placement-enabled/) + assert.match(workflow, /\.valueSource\.secretKeyRef/) + assert.match(workflow, /jq -er --arg secret "\$\{REGIONAL_PLACEMENT_SECRET\}"/) + assert.doesNotMatch(workflow, /jq -e --arg secret "\$\{REGIONAL_PLACEMENT_SECRET\}"/) + assert.doesNotMatch(workflow, /--regional-placement-enabled/) + assert.doesNotMatch(workflow, /"\$\{\{ inputs\./) + assert.doesNotMatch(workflow, /dns/i) +}) + +test('requires immutable staged evidence and a timed C27 canary before expansion', () => { + assert.match(workflow, /actions: read/) + assert.match(workflow, /actions\/download-artifact@v4/) + assert.match(workflow, /relay-asia-staging-\$\{EVIDENCE_RUN_ID\}-\$\{EVIDENCE_RUN_ATTEMPT\}/) + assert.match(workflow, /evidence_kind=staging/) + assert.match(workflow, /load-relay-controls\.mjs/) + assert.match(workflow, /--controls 1/) + assert.match(workflow, /--splices 1/) + assert.match(workflow, /--splice-hold-seconds 60/) + assert.match(workflow, /--relay-asia-load-principals 1/) + assert.match(workflow, /--duration-seconds 300/) + assert.match(workflow, /--required-lease-horizons 2/) + assert.match(workflow, /pnpm\/action-setup@v4/) + assert.match(workflow, /Install exact C27 canary dependencies/) + assert.match(workflow, /pnpm install --frozen-lockfile/) + assert.match(workflow, /pnpm --filter @orca-cloud\/relay-contract build/) + assert.ok( + workflow.indexOf('Build the C27 canary Relay contract') < + workflow.indexOf('Run a real five-minute C27 control and splice canary') + ) + assert.match(workflow, /--load-report "\$\{RUNNER_TEMP\}\/relay-asia-c27-load\.json"/) + assert.match(workflow, /states\["production-gce-c28"\].*= migration-only/) + assert.match(workflow, /states\["production-gce-c29"\].*= migration-only/) + assert.match(workflow, /relay-asia-c27-canary-\$\{\{ github\.run_id \}\}-\$\{\{ github\.run_attempt \}\}/) + assert.match(workflow, /id: c27-evidence-upload/) + assert.match(workflow, /Return an unproven C27 canary to migration-only/) + assert.match(workflow, /steps\.c27-evidence-upload\.outcome != 'success'/) + assert.match(workflow, /--mode recover-promotion[\s\S]*?--attempt-id "\$\{SELECTOR_ATTEMPT_ID\}"/) + assert.match(workflow, /--attempt-id "\$\{SELECTOR_ATTEMPT_ID\}-rollback"/) + assert.match(workflow, /evidence_kind=c27/) + assert.match(workflow, /orca_relay_runtime_metrics/) + assert.match(workflow, /relay-asia-rollout-evidence\.mjs create-c27/) + assert.match(workflow, /retention-days: 7/) + assert.match(workflow, /Require the exact director image before promotion/) + assert.match(workflow, /DIRECTOR_ORIGIN.*\/v1\/admin\/runtime-status/) + assert.match(workflow, /\.imageDigest.*IMAGE_DIGEST/) + const provenance = /- name: Verify evidence provenance and rollout binding before authentication\n([\s\S]*?)(?=\n - id: auth)/ + .exec(workflow)?.[1] + assert.ok(provenance) + assert.match(provenance, /\.head_sha \| select\(type == "string" and test\("\^\[a-f0-9\]\{40\}\$"\)\)/) + assert.match(provenance, /--commit-sha "\$\{evidence_commit_sha\}"/) + assert.doesNotMatch(provenance, /--commit-sha "\$\{GITHUB_SHA\}"/) +}) + +test('creates staging evidence only after the bounded launch-path load and rollback', () => { + assert.match(stagingProof, /runs-on: \[self-hosted, linux, x64, relay-asia-east2-load\]/) + assert.doesNotMatch(stagingProof, /group: relay-asia-east2-load/) + assert.match(stagingProof, /pnpm\/action-setup@v4/) + assert.match(stagingProof, /pnpm install --frozen-lockfile/) + assert.match(stagingProof, /pnpm --filter @orca-cloud\/relay-contract build/) + assert.match(stagingProof, /run_phase launch 5 5/) + assert.doesNotMatch(stagingProof, /run_phase control|run_phase mixed/) + assert.match(stagingProof, /--aggregate-controls "\$\(\(controls \* 4\)\)"/) + assert.match(stagingProof, /--aggregate-splices "\$\(\(splices \* 4\)\)"/) + assert.match(stagingProof, /--required-lease-horizons 2/) + assert.match(stagingProof, /--splice-ramp-seconds 120/) + assert.match(stagingProof, /--max-generator-rss-growth-mib 512/) + assert.match(stagingProof, /--relay-asia-load-principals 32/) + assert.match(stagingProof, /ulimit -n/) + assert.match(stagingProof, /--region-behavior-probes 1/) + assert.match(stagingProof, /--capacity-cell-origin https:\/\/c4\.relay-staging\.onorca\.dev/) + assert.match(stagingProof, /--rebind-probes 2/) + assert.match(stagingProof, /--skip-rebind-overflow-check/) + assert.doesNotMatch(stagingProof, /--request-unit-invites|--regional-fallback-probes/) + assert.match(stagingProof, /--aggregate-reader-splices.*echo 5/) + assert.match(stagingProof, /--aggregate-reader-bytes.*echo 12582912/) + assert.match(stagingProof, /--phase-barrier-dir "\$\{proof_dir\}\/\$\{phase\}-barrier"/) + assert.match(stagingProof, /--duration-seconds 210/) + assert.match(stagingProof, /trap stop_shards EXIT/) + assert.match(stagingProof, /if ! wait "\$\{pid\}"; then failed=1; break; fi/) + assert.match(stagingProof, /connectionFailuresByReason/) + assert.match(stagingProof, /--launch-report "\$\{proof_dir\}\/launch\.json"/) + assert.match(stagingProof, /id-token: write/) + assert.match(stagingProof, /STAGING_GCP_RELAY_ASIA_PROOF_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(stagingProof, /STAGING_GCP_RELAY_ASIA_PROOF_SERVICE_ACCOUNT/) + assert.doesNotMatch(stagingProof, /STAGING_GCP_DEPLOY_SERVICE_ACCOUNT/) + assert.doesNotMatch(stagingProof, /STAGING_RELAY_LOAD_ACCESS_TOKEN/) + assert.doesNotMatch(stagingProof, /secrets versions access|signing-key-file/) + assert.match(stagingProof, /relay-asia-rollout-evidence\.mjs create-staging/) + assert.match(stagingProof, /Require the exact staging director image before promotion/) + assert.match(stagingProof, /DIRECTOR_ORIGIN.*\/v1\/admin\/runtime-status/) + assert.match(stagingProof, /\.imageDigest.*IMAGE_DIGEST/) + assert.match(stagingProof, /Return staging C4 to migration-only/) + assert.match(stagingProof, /steps\.promote\.outcome != 'skipped'/) + assert.match(stagingProof, /--mode recover-promotion[\s\S]*?--attempt-id "\$\{PROMOTE_ATTEMPT_ID\}"/) + assert.match(stagingProof, /--mode rollback[\s\S]*?--expected-generation "\$\{promoted_generation\}"/) + assert.match(stagingProof, /if: \$\{\{ success\(\) \}\}/) + assert.match( + stagingProof, + /recover:\n if: \$\{\{ always\(\) && github\.ref == 'refs\/heads\/main' \}\}/ + ) + assert.match(stagingProof, /needs: prove/) + assert.match(stagingProof, /Recover staging C4 with a fresh identity/) + assert.equal((stagingProof.match(/google-github-actions\/auth@v2/g) ?? []).length, 2) + assert.equal((stagingProof.match(/--mode recover-promotion/g) ?? []).length, 2) + assert.equal((stagingProof.match(/--mode rollback/g) ?? []).length, 2) +}) + +test('keeps the private runner below its 64-port Cloud NAT allocation', () => { + const profile = /run_phase launch (\d+) (\d+)/.exec(stagingProof) + const controlsPerShard = Number(profile?.[1]) + const splicesPerShard = Number(profile?.[2]) + const rebindProbes = Number(/--rebind-probes (\d+)/.exec(stagingProof)?.[1]) + const runtimeStatusSockets = 1 + assert.ok( + controlsPerShard * 4 + splicesPerShard * 4 * 2 + rebindProbes + runtimeStatusSockets < 64 + ) +}) + +test('paces one-source staging upgrades below the Relay anti-abuse ceiling', () => { + const splicesPerShard = Number(/run_phase launch \d+ (\d+)/.exec(stagingProof)?.[1]) + const rebindProbes = Number(/--rebind-probes (\d+)/.exec(stagingProof)?.[1]) + const spliceRampMs = Number(/--splice-ramp-seconds (\d+)/.exec(stagingProof)?.[1]) * 1000 + const ceiling = Number( + /maxPreAuthAttemptsPerSourcePerMinute: (\d+)/.exec(admissionBudgets)?.[1] + ) + const totalSplices = splicesPerShard * 4 + const attempts = Array.from({ length: totalSplices }, (_, ordinal) => + Math.floor(ordinal * spliceRampMs / (totalSplices - 1)) + ).flatMap((startedAt) => [startedAt, startedAt]) + attempts.push(...Array.from({ length: 4 + rebindProbes }, () => 0)) + const busiestMinute = Math.max(...attempts.map((startedAt) => + attempts.filter((attempt) => attempt >= startedAt && attempt < startedAt + 60_000).length + )) + assert.ok(busiestMinute < ceiling) +}) + +test('reserves rollback time beyond the complete bounded staging proof envelope', () => { + const timeoutMinutes = Number(/timeout-minutes: (\d+)/.exec(stagingProof)?.[1]) + assert.equal(timeoutMinutes, 75) + const spliceRampSeconds = Number(/--splice-ramp-seconds (\d+)/.exec(stagingProof)?.[1]) + const launchSeconds = 180 + spliceRampSeconds + 210 + 60 + const setupEvidenceAndRollbackSeconds = 10 * 60 + const envelopeMinutes = Math.ceil((launchSeconds + setupEvidenceAndRollbackSeconds) / 60) + assert.ok(timeoutMinutes - envelopeMinutes >= 30) + assert.match(stagingProof, /--ramp-seconds 180/) + assert.match(stagingProof, /--duration-seconds 210/) +}) + +test('binds the staging proof to one least-privilege Google identity', () => { + assert.match( + proofIam, + /github_relay_asia_proof_workflow_file = "prove-relay-asia-staging\.yml"/ + ) + assert.match( + proofIam, + /assertion\.workflow_ref == '\$\{prefix\}\$\{local\.github_relay_asia_proof_workflow_file\}@refs\/heads\/main'/ + ) + assert.match(proofIam, /assertion\.environment == 'staging'/) + assert.match(proofIam, /assertion\.event_name == 'workflow_dispatch'/) + assert.match(proofIam, /roles\/logging\.viewer/) + assert.match(proofIam, /roles\/monitoring\.viewer/) + assert.match(rolloutEvidence, /readCloudSqlBackends/) + assert.match(rolloutEvidence, /cloudSql: await readCloudSqlBackends/) + assert.doesNotMatch(proofIam, /compute\.|cloudsql\.|secretmanager\.|roles\/editor|roles\/run\./) +}) + +test('keeps the production US-first switch in durable Secret Manager state', () => { + assert.match(directorWorkflow, /options: \[preserve, enable, disable\]/) + assert.match(directorWorkflow, /default: preserve/) + assert.match(directorWorkflow, /gcloud secrets versions add/) + assert.match(directorWorkflow, /preserve\) desired="\$\{current\}"/) + assert.match(directorWorkflow, /--regional-placement-secret-version "\$\{target_version\}"/) + assert.match(directorWorkflow, /test "\$\{CEILING\}" = "\$\{DIRECTOR_MAX_INSTANCES\}"/) + assert.match(directorWorkflow, /orca-cloud-relay-regional-placement-enabled/) + assert.match(directorWorkflow, /\.valueSource\.secretKeyRef \/\/ \.valueFrom\.secretKeyRef/) + assert.match(directorWorkflow, /\.version \/\/ \.key/) + assert.match(directorWorkflow, /\.secret \/\/ \.name/) + assert.match(workflow, /\.valueSource\.secretKeyRef \/\/ \.valueFrom\.secretKeyRef/) + assert.doesNotMatch(directorWorkflow, /--regional-placement-enabled/) + assert.doesNotMatch(workflow, /inputs\.regional-placement-enabled/) +}) + +test('prunes incompatible production revisions only when explicitly confirmed', () => { + assert.match( + directorWorkflow, + /prune-incompatible-revisions:[\s\S]*?default: false[\s\S]*?type: boolean/ + ) + assert.match(directorWorkflow, /PRUNE_INCOMPATIBLE_RELAY_DIRECTOR_REVISIONS/) + assert.match( + directorWorkflow, + /test "\$\{REGIONAL_PLACEMENT_MODE\}" = preserve[\s\S]*?test "\$\{CONFIRMATION\}" = PRUNE_INCOMPATIBLE_RELAY_DIRECTOR_REVISIONS/ + ) + assert.match( + directorWorkflow, + /--prune-revisions "\$\{PRUNE_INCOMPATIBLE_REVISIONS\}"/ + ) +}) + +test('documents the exact regional-placement secret bootstrap before director rollout', () => { + for (const address of [ + 'google_secret_manager_secret.relay_regional_placement_enabled', + 'google_secret_manager_secret_version.relay_regional_placement_enabled', + 'google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor', + 'google_secret_manager_secret_iam_member.relay_regional_placement_deploy_accessor[0]', + 'google_secret_manager_secret_iam_member.relay_regional_placement_deploy_adder[0]', + 'google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer[0]' + ]) { + assert.match(terraformReadme, new RegExp(address.replaceAll(/[.[\]]/g, '\\$&'))) + } + assert.match(terraformReadme, /Before the first director deployment/) + assert.match(terraformReadme, /Pass the exact environment tfvars/) + // The Cloudflare records left with the apps root; a -var for a variable this root no longer + // declares is a hard error, so no relay procedure may still tell an operator to pass it. + assert.doesNotMatch(terraformReadme, /manage_artifact_dns/) + assert.match(terraformReadme, /exactly these six additions/) + assert.match(terraformReadme, /version metadata/) + assert.match( + relayTerraform, + /resource "google_secret_manager_secret_iam_member" "relay_regional_placement_deploy_viewer"[\s\S]*?role\s+= "roles\/secretmanager\.viewer"/ + ) +}) diff --git a/cloud/dev/scripts/relay-asia-cloud-sql-metrics.mjs b/cloud/dev/scripts/relay-asia-cloud-sql-metrics.mjs new file mode 100644 index 00000000000..489402b0399 --- /dev/null +++ b/cloud/dev/scripts/relay-asia-cloud-sql-metrics.mjs @@ -0,0 +1,33 @@ +import { spawnSync } from 'node:child_process' + +function accessToken() { + const result = spawnSync('gcloud', ['auth', 'print-access-token'], { + encoding: 'utf8', timeout: 30_000 + }) + const token = result.stdout.trim() + if (result.status !== 0 || token.length < 20) { + throw new Error('Google access token is unavailable') + } + return token +} + +export async function readCloudSqlBackends(environment, startedAt, endedAt) { + const production = environment === 'production' + if (!production && environment !== 'staging') throw new Error('Cloud SQL environment is invalid') + const project = production ? 'onorca-cloud' : 'onorca-cloud-staging' + const instance = production ? 'orca-cloud-auth-db' : 'orca-cloud-staging-auth-db' + const url = new URL(`https://monitoring.googleapis.com/v3/projects/${project}/timeSeries`) + url.searchParams.set('filter', `metric.type = "cloudsql.googleapis.com/database/postgresql/num_backends" AND resource.labels.database_id = "${project}:${instance}"`) + url.searchParams.set('interval.startTime', startedAt) + url.searchParams.set('interval.endTime', endedAt) + url.searchParams.set('aggregation.alignmentPeriod', '60s') + url.searchParams.set('aggregation.perSeriesAligner', 'ALIGN_MAX') + url.searchParams.set('aggregation.crossSeriesReducer', 'REDUCE_MAX') + url.searchParams.set('view', 'FULL') + const response = await fetch(url, { + headers: { authorization: `Bearer ${accessToken()}` }, + redirect: 'error', signal: AbortSignal.timeout(30_000) + }) + if (!response.ok) throw new Error(`Cloud SQL metric query returned ${response.status}`) + return await response.json() +} diff --git a/cloud/dev/scripts/relay-asia-rollout-evidence.mjs b/cloud/dev/scripts/relay-asia-rollout-evidence.mjs new file mode 100644 index 00000000000..e833a0ae16b --- /dev/null +++ b/cloud/dev/scripts/relay-asia-rollout-evidence.mjs @@ -0,0 +1,544 @@ +import { readFileSync, writeFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' +import { readCloudSqlBackends } from './relay-asia-cloud-sql-metrics.mjs' +import { RELAY_GITHUB_REPOSITORY, relayWorkflowPath } from './relay-repository.mjs' + +const REPOSITORY = RELAY_GITHUB_REPOSITORY +const ADMISSION_WORKFLOW = relayWorkflowPath('operate-relay-asia-admission.yml') +const STAGING_WORKFLOW = relayWorkflowPath('prove-relay-asia-staging.yml') +const STAGING_CELL = 'staging-gce-c4' +const C27 = 'production-gce-c27' +const DIGEST_PATTERN = /^sha256:[a-f0-9]{64}$/ +const SHA_PATTERN = /^[a-f0-9]{40}$/ +const MAX_LOG_EDGE_GAP_MS = 120_000 +const MAX_LOG_SAMPLE_GAP_MS = 120_000 +const CLOUD_SQL_LIMIT = 320 +const C27_CANARY_MINIMUM_MS = 5 * 60_000 +const GENERATOR_CPU_PERCENT_LIMIT = 80 +const GENERATOR_EVENT_LOOP_P99_MS_LIMIT = 100 +const GENERATOR_RSS_GROWTH_MIB_LIMIT = 512 +const DATABASE_POOL_TRANSIENT_WAITERS_MAX = 4 +const DATABASE_POOL_TRANSIENT_WAIT_MS_MAX = 50 + +function object(value, label) { + if (!value || typeof value !== 'object' || Array.isArray(value)) { + throw new Error(`${label} is invalid`) + } + return value +} + +function positiveInteger(value, label) { + const number = Number(value) + if (!Number.isSafeInteger(number) || number < 1) throw new Error(`${label} is invalid`) + return number +} + +function instant(value, label) { + const date = new Date(value) + if (!Number.isFinite(date.valueOf())) throw new Error(`${label} is invalid`) + return date +} + +function exactCells(actual, expected) { + return Array.isArray(actual) && + JSON.stringify([...actual].sort()) === JSON.stringify([...expected].sort()) +} + +function source(input, environment, workflow) { + if (input.repository !== REPOSITORY) throw new Error('repository is invalid') + if (!SHA_PATTERN.test(input.commitSha)) throw new Error('commit SHA is invalid') + return { + repository: input.repository, + workflow, + environment, + runId: positiveInteger(input.runId, 'run ID'), + runAttempt: positiveInteger(input.runAttempt, 'run attempt'), + commitSha: input.commitSha + } +} + +function baseEvidence(input, environment, cells, workflow = ADMISSION_WORKFLOW) { + if (!DIGEST_PATTERN.test(input.imageDigest)) throw new Error('image digest is invalid') + return { + version: 1, + source: source(input, environment, workflow), + imageDigest: input.imageDigest, + topology: { + cellIds: cells, + selectorGeneration: positiveInteger(input.selectorGeneration, 'selector generation') + } + } +} + +export function buildStagingEvidence(input) { + const start = instant(input.startedAt, 'staging proof start') + const end = instant(input.endedAt, 'staging proof end') + const launch = loadReports(input.launchReport, 'launch load report') + const expectedLoad = { + controls: 20, splices: 20, slowReaders: 4, wedgedReaders: 1, + minimumSeconds: 210 + } + assertLoadReports(launch, expectedLoad) + const metrics = runtimeMetrics(input.logs, start, end, STAGING_CELL) + assertPassingRuntimeMetrics(metrics, 'staging proof', 1) + const minimumConcurrentSplices = expectedLoad.splices - expectedLoad.wedgedReaders + if ( + metrics.targetControlsMax < expectedLoad.controls || + metrics.targetSplicesMax < minimumConcurrentSplices + ) { + throw new Error('staging launch load did not reach C4 at the reviewed levels') + } + metrics.cloudSqlBackendsMax = cloudSqlMaximum(input.cloudSql, start, end) + if (metrics.cloudSqlBackendsMax >= CLOUD_SQL_LIMIT) { + throw new Error(`Cloud SQL backends must remain below ${CLOUD_SQL_LIMIT}`) + } + return { + ...baseEvidence(input, 'staging', [STAGING_CELL], STAGING_WORKFLOW), + kind: 'staging-asia-readiness', + window: { startedAt: start.toISOString(), endedAt: end.toISOString() }, + load: { launch: loadSummary(launch) }, + metrics + } +} + +function loadReports(value, label) { + if (!Array.isArray(value) || value.length < 2) throw new Error(`${label} must be sharded`) + return value.map((report) => object(report, label)) +} + +function total(reports, key) { + return reports.reduce((sum, report) => sum + number(report[key], key), 0) +} + +function loadSummary(reports) { + return { + shards: reports.length, + controls: total(reports, 'controls'), + peakActive: total(reports, 'peakActive'), + steadyMinimumActive: total(reports, 'steadyMinimumActive'), + peakActiveSplices: total(reports, 'peakActiveSplices'), + completedSplices: total(reports, 'completedSplices'), + slowReaderSplicesCompleted: total(reports, 'slowReaderSplicesCompleted'), + wedgedReaderSplicesClosed: total(reports, 'wedgedReaderSplicesClosed'), + regionalFallbacksProved: total(reports, 'regionalFallbacksProved'), + oldClientUsFirstProved: total(reports, 'oldClientUsFirstProved'), + stickyAssignmentProved: total(reports, 'stickyAssignmentProved'), + requestUnitInvitesOpened: total(reports, 'requestUnitInvitesOpened'), + requestUnitPrincipalCounts: reports.map((report) => report.requestUnitPrincipalCount), + requestUnitOverflowReasons: + reports.map((report) => report.requestUnitOverflowReason).filter(Boolean), + requestUnitCleanupProved: total(reports, 'requestUnitCleanupProved'), + phaseBarrierPassed: reports.every((report) => report.phaseBarrierPassed === true), + rebindProbesOpened: total(reports, 'rebindProbesOpened'), + rebindOverflowReasons: reports.map((report) => report.rebindOverflowReason).filter(Boolean), + readerQueueEvidence: reports.flatMap((report) => report.readerQueueEvidence), + readerQueuedBytesPeak: Math.max(...reports.map((report) => report.readerQueuedBytesPeak)), + generatorCpuPercentMax: Math.max(...reports.map((report) => report.generatorCpuPercent)), + generatorEventLoopP99MsMax: Math.max( + ...reports.map((report) => report.generatorEventLoopP99Ms) + ), + generatorRssGrowthMiBMax: Math.max(...reports.map((report) => report.generatorRssGrowthMiB)), + configuredSteadySeconds: Math.min(...reports.map((report) => report.configuredSteadySeconds)) + } +} + +function assertLoadReports(reports, expected) { + const shardCount = reports.length + if ( + reports.some((report, index) => + report.event !== 'relay_load_complete' || + report.shardCount !== shardCount || report.shardIndex !== index || + report.relayAsiaLoadPrincipalCount !== 32 || + number(report.configuredSteadySeconds, 'staging steady seconds') < expected.minimumSeconds || + number(report.configuredSpliceHoldSeconds, 'staging splice hold seconds') < + expected.minimumSeconds || + report.requiredLeaseHorizons !== 2 || report.phaseBarrierPassed !== true + ) || + total(reports, 'controls') !== expected.controls + ) { + throw new Error('staging load profile does not match') + } + if ( + total(reports, 'peakActive') !== expected.controls || + total(reports, 'steadyMinimumActive') !== expected.controls + ) { + throw new Error('staging load did not sustain the required controls') + } + for (const key of [ + 'rampConnectionFailures', 'steadyConnectionFailures', 'transitionConnectionFailures', + 'unexpectedCloses', 'protocolErrors', 'refreshErrors', 'socketErrors', 'failedSplices' + ]) { + if (total(reports, key) !== 0) throw new Error(`staging load ${key} must be zero`) + } + const peakActiveSplices = total(reports, 'peakActiveSplices') + if ( + total(reports, 'configuredSplices') !== expected.splices || + total(reports, 'configuredSlowReaderSplices') !== expected.slowReaders || + total(reports, 'configuredWedgedReaderSplices') !== expected.wedgedReaders || + peakActiveSplices < expected.splices - expected.wedgedReaders || + peakActiveSplices > expected.splices || + total(reports, 'completedSplices') !== expected.splices - expected.wedgedReaders || + total(reports, 'slowReaderSplicesCompleted') !== expected.slowReaders || + total(reports, 'wedgedReaderSplicesClosed') !== expected.wedgedReaders + ) throw new Error('staging mixed load evidence does not match') + if ( + total(reports, 'regionalFallbacksProved') !== 0 || + total(reports, 'oldClientUsFirstProved') !== 1 || + total(reports, 'stickyAssignmentProved') !== 1 || + total(reports, 'requestUnitInvitesOpened') !== 0 || + reports.some((report) => report.requestUnitPrincipalCount !== 0) || + total(reports, 'requestUnitCleanupProved') !== 0 || + reports.some((report) => report.requestUnitOverflowReason !== null) || + total(reports, 'rebindProbesOpened') !== 2 || + reports.some((report) => report.rebindOverflowReason !== null) + ) throw new Error('staging launch-path evidence does not match') + const readerReports = reports.filter( + (report) => report.configuredSlowReaderSplices + report.configuredWedgedReaderSplices > 0 + ) + if (expected.slowReaders > 0 && readerReports.length !== 1) { + throw new Error('staging reader pressure must have one causal owner') + } + for (const report of reports) { + const queue = report.readerQueueEvidence + if (!Array.isArray(queue)) throw new Error('staging reader queue evidence is invalid') + if (expected.slowReaders === 0 && queue.length !== 0) { + throw new Error('control load unexpectedly contains reader queue evidence') + } + const ownsReaderPressure = readerReports.includes(report) + if (expected.slowReaders > 0 && ownsReaderPressure && ( + queue.length !== 1 || + queue[0]?.origin !== 'https://c4.relay-staging.onorca.dev' || + number(queue[0]?.baselineBytes, 'reader queue baseline') > + number(queue[0]?.peakBytes, 'reader queue peak') || + number(queue[0]?.increaseBytes, 'reader queue increase') <= 0 || + queue[0].peakBytes - queue[0].baselineBytes !== queue[0].increaseBytes || + report.readerQueuedBytesPeak !== queue[0].increaseBytes + )) throw new Error('staging reader queue evidence is not causal') + if (expected.slowReaders > 0 && !ownsReaderPressure && ( + queue.length !== 0 || report.readerQueuedBytesPeak !== 0 + )) throw new Error('non-owner shard contains reader queue evidence') + } + for (const report of reports) { + if ( + number(report.generatorCpuPercent, 'generator CPU') >= GENERATOR_CPU_PERCENT_LIMIT || + number(report.generatorEventLoopP99Ms, 'generator event loop') >= + GENERATOR_EVENT_LOOP_P99_MS_LIMIT || + number(report.generatorRssGrowthMiB, 'generator RSS growth') >= + GENERATOR_RSS_GROWTH_MIB_LIMIT + ) throw new Error('staging load generator has insufficient headroom') + const shutdown = object(report.shutdownEvidence, 'load shutdown evidence') + if ( + shutdown.peerShutdowns !== report.controls || shutdown.activeControls !== 0 || + shutdown.activeSplices !== 0 || shutdown.reconnectTimers !== 0 + ) throw new Error('staging load cleanup is incomplete') + } +} + +function number(value, label) { + if (typeof value !== 'number' || !Number.isFinite(value) || value < 0) { + throw new Error(`${label} is invalid`) + } + return value +} + +function sumMap(value, label) { + const entries = Object.entries(object(value ?? {}, label)) + return entries.reduce((total, [key, count]) => { + if (!key) throw new Error(`${label} is invalid`) + return total + number(count, label) + }, 0) +} + +function assertCoverage(timestamps, start, end, label) { + if (timestamps.length === 0) throw new Error(`${label} has no samples`) + const ordered = timestamps.map((value) => instant(value, `${label} timestamp`).valueOf()) + .sort((left, right) => left - right) + if ( + ordered[0] > start.valueOf() + MAX_LOG_EDGE_GAP_MS || + ordered.at(-1) < end.valueOf() - MAX_LOG_EDGE_GAP_MS + ) throw new Error(`${label} does not cover the canary window`) + if (ordered.some((timestamp, index) => index > 0 && timestamp - ordered[index - 1] > MAX_LOG_SAMPLE_GAP_MS)) { + throw new Error(`${label} has a sampling gap`) + } +} + +function pointValue(point) { + const value = point?.value + if (typeof value?.doubleValue === 'number') return value.doubleValue + if (typeof value?.int64Value === 'string') return Number(value.int64Value) + return NaN +} + +function cloudSqlMaximum(response, start, end) { + const points = (object(response, 'Cloud SQL response').timeSeries ?? []) + .flatMap((series) => series.points ?? []) + assertCoverage(points.map((point) => point.interval?.endTime), start, end, 'Cloud SQL metrics') + const values = points.map(pointValue) + if (values.some((value) => !Number.isFinite(value) || value < 0)) { + throw new Error('Cloud SQL backend metric is invalid') + } + return Math.max(...values) +} + +export function buildC27CanaryEvidence(input) { + const start = instant(input.startedAt, 'canary start') + const end = instant(input.endedAt, 'canary end') + if (end.valueOf() - start.valueOf() < C27_CANARY_MINIMUM_MS) { + throw new Error('C27 canary window is shorter than 5 minutes') + } + const load = object(input.loadReport, 'C27 load report') + assertC27CanaryLoad(load) + const metrics = runtimeMetrics(input.logs, start, end, C27) + assertPassingRuntimeMetrics(metrics, 'C27 canary') + metrics.cloudSqlBackendsMax = cloudSqlMaximum(input.cloudSql, start, end) + assertPassingCanary(metrics) + return { + ...baseEvidence(input, 'production', [C27]), + kind: 'production-c27-canary', + window: { startedAt: start.toISOString(), endedAt: end.toISOString() }, + load, + metrics + } +} + +function assertC27CanaryLoad(report) { + if ( + report.event !== 'relay_load_complete' || + report.controls !== 1 || report.shardCount !== 1 || report.shardIndex !== 0 || + report.relayAsiaLoadPrincipalCount !== 1 || + number(report.configuredSteadySeconds, 'canary steady seconds') < 300 || + number(report.configuredSpliceHoldSeconds, 'canary splice hold seconds') < 60 || + number(report.requiredLeaseHorizons, 'canary lease horizons') < 2 || + report.peakActive !== 1 || report.steadyMinimumActive !== 1 || + report.configuredSplices !== 1 || report.peakActiveSplices !== 1 || + report.completedSplices !== 1 || report.failedSplices !== 0 + ) throw new Error('C27 control and splice canary did not match') + for (const key of [ + 'connectionFailures', 'unexpectedCloses', 'protocolErrors', + 'refreshErrors', 'socketErrors' + ]) { + if (number(report[key], key) !== 0) throw new Error(`C27 canary ${key} must be zero`) + } + const shutdown = object(report.shutdownEvidence, 'C27 load shutdown evidence') + if ( + shutdown.peerShutdowns !== 1 || shutdown.activeControls !== 0 || + shutdown.activeSplices !== 0 || shutdown.reconnectTimers !== 0 + ) throw new Error('C27 load cleanup is incomplete') +} + +function runtimeMetrics(logs, start, end, targetCellId) { + const entries = logs.map((entry) => object(entry, 'runtime metric entry')) + const directorEntries = entries.filter((entry) => entry.jsonPayload?.role === 'director') + const cellEntries = entries.filter((entry) => + entry.jsonPayload?.role === 'cell' && + entry.jsonPayload?.cellId === targetCellId && + entry.jsonPayload?.region === 'asia-east2' + ) + assertCoverage(directorEntries.map((entry) => entry.timestamp), start, end, 'director metrics') + assertCoverage(cellEntries.map((entry) => entry.timestamp), start, end, `${targetCellId} metrics`) + const identifiedDirectors = Map.groupBy( + directorEntries.filter((entry) => entry.resource?.labels?.instance_id), + (entry) => entry.resource.labels.instance_id + ) + for (const [instanceId, instanceEntries] of identifiedDirectors) { + assertCoverage(instanceEntries.map((entry) => entry.timestamp), start, end, `director ${instanceId}`) + } + const directorPayloads = directorEntries.map((entry) => entry.jsonPayload) + const cellPayloads = cellEntries.map((entry) => entry.jsonPayload) + const payloads = [...directorPayloads, ...cellPayloads] + return { + asiaSelections: directorPayloads.reduce( + (total, payload) => total + number(payload.selectedRegionsDelta?.['asia-east2'] ?? 0, 'Asia selections'), 0 + ), + regionFallbacks: directorPayloads.reduce( + (total, payload) => total + sumMap(payload.regionFallbacksDelta, 'region fallbacks'), 0 + ), + usRegionFallbacks: directorPayloads.reduce( + (total, payload) => total + number( + payload.regionFallbacksDelta?.['us-central1'] ?? 0, + 'US region fallbacks' + ), 0 + ), + unavailableRegions: directorPayloads.reduce( + (total, payload) => total + sumMap(payload.unavailableRegionsDelta, 'unavailable regions'), 0 + ), + relaySqlFailures: payloads.reduce( + (total, payload) => total + number(payload.sqlFailuresDelta, 'Relay SQL failures'), 0 + ), + databasePoolWaitingMax: Math.max(...payloads.map( + (payload) => number(payload.databasePoolWaiting, 'database pool waiting') + )), + databasePoolWaitersMax: Math.max(...payloads.map( + (payload) => number(payload.databasePoolWaitersMax, 'database pool waiters') + )), + databasePoolWaitMsMax: Math.max(...payloads.map( + (payload) => number(payload.databasePoolWaitMsMax, 'database pool wait time') + )), + targetControlsMax: Math.max(...cellPayloads.map( + (payload) => number(payload.controls, 'target controls') + )), + targetSplicesMax: Math.max(...cellPayloads.map( + (payload) => number(payload.splices, 'target splices') + )) + } +} + +function assertPassingRuntimeMetrics(metrics, label, expectedRegionFallbacks = 0) { + if (number(metrics.asiaSelections, 'Asia selections') < 1) { + throw new Error(`${label} observed no Asia selections`) + } + if ( + number(metrics.regionFallbacks, 'regionFallbacks') !== expectedRegionFallbacks || + number(metrics.usRegionFallbacks, 'usRegionFallbacks') !== expectedRegionFallbacks + ) { + throw new Error(`${label} regionFallbacks did not match the intentional probes`) + } + for (const key of [ + 'unavailableRegions', 'relaySqlFailures', 'databasePoolWaitingMax' + ]) { + if (number(metrics[key], key) !== 0) throw new Error(`${label} ${key} must be zero`) + } + if ( + number(metrics.databasePoolWaitersMax, 'databasePoolWaitersMax') > + DATABASE_POOL_TRANSIENT_WAITERS_MAX || + number(metrics.databasePoolWaitMsMax, 'databasePoolWaitMsMax') > + DATABASE_POOL_TRANSIENT_WAIT_MS_MAX + ) throw new Error(`${label} transient database pool pressure exceeded its bound`) +} + +function assertPassingCanary(metrics) { + if ( + number(metrics.targetControlsMax, 'C27 controls') < 1 || + number(metrics.targetSplicesMax, 'C27 splices') < 1 + ) throw new Error('C27 canary traffic did not reach C27') + if (number(metrics.cloudSqlBackendsMax, 'Cloud SQL backends') >= CLOUD_SQL_LIMIT) { + throw new Error(`Cloud SQL backends must remain below ${CLOUD_SQL_LIMIT}`) + } +} + +function assertProvenance(evidence, run, expected) { + const evidenceSource = object(evidence.source, 'evidence source') + if ( + run.id !== evidenceSource.runId || + run.run_attempt !== evidenceSource.runAttempt || + run.conclusion !== 'success' || + run.event !== 'workflow_dispatch' || + run.head_branch !== 'main' || + run.head_sha !== evidenceSource.commitSha || + run.repository?.full_name !== evidenceSource.repository || + run.path?.split('@')[0] !== expected.workflow || + evidenceSource.workflow !== expected.workflow || + evidenceSource.repository !== REPOSITORY || + expected.repository !== REPOSITORY || + evidenceSource.environment !== expected.environment || + evidenceSource.commitSha !== expected.commitSha + ) throw new Error('evidence workflow provenance does not match') +} + +export function verifyRolloutEvidence(evidence, run, expected) { + object(evidence, 'evidence') + object(run, 'workflow run') + if (evidence.version !== 1 || evidence.kind !== expected.kind) { + throw new Error('evidence kind is invalid') + } + assertProvenance(evidence, run, expected) + if (evidence.imageDigest !== expected.imageDigest) throw new Error('evidence image digest does not match') + if (!exactCells(evidence.topology?.cellIds, expected.cellIds)) { + throw new Error('evidence topology does not match') + } + positiveInteger(evidence.topology?.selectorGeneration, 'evidence selector generation') + if ( + expected.selectorGeneration !== undefined && + evidence.topology.selectorGeneration !== expected.selectorGeneration + ) throw new Error('evidence selector generation does not match') + const now = instant(expected.now, 'verification time') + const proofTime = instant(evidence.window?.endedAt, 'evidence time') + const maxAgeMs = expected.kind === 'staging-asia-readiness' ? 24 * 60 * 60_000 : 6 * 60 * 60_000 + if (proofTime > now || now.valueOf() - proofTime.valueOf() > maxAgeMs) { + throw new Error('rollout evidence is stale') + } + if (expected.kind === 'production-c27-canary') { + const start = instant(evidence.window?.startedAt, 'canary start') + if (proofTime.valueOf() - start.valueOf() < C27_CANARY_MINIMUM_MS) { + throw new Error('C27 canary window is shorter than 5 minutes') + } + assertC27CanaryLoad(object(evidence.load, 'canary load')) + assertPassingCanary(object(evidence.metrics, 'canary metrics')) + } + return evidence +} + +function argumentsMap(argv) { + const values = new Map() + for (let index = 1; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined || values.has(key.slice(2))) { + throw new Error('invalid rollout evidence arguments') + } + values.set(key.slice(2), value) + } + return { command: argv[0], values } +} + +function required(values, key) { + const value = values.get(key) + if (!value) throw new Error(`missing --${key}`) + return value +} + +function commonInput(values) { + return { + repository: required(values, 'repository'), runId: required(values, 'run-id'), + runAttempt: required(values, 'run-attempt'), commitSha: required(values, 'commit-sha'), + imageDigest: required(values, 'image-digest'), + selectorGeneration: required(values, 'selector-generation') + } +} + +async function main(argv) { + const { command, values } = argumentsMap(argv) + const output = required(values, 'output') + if (command === 'create-staging') { + writeFileSync(output, `${JSON.stringify(buildStagingEvidence({ + ...commonInput(values), startedAt: required(values, 'started-at'), + endedAt: required(values, 'ended-at'), + launchReport: JSON.parse(readFileSync(required(values, 'launch-report'), 'utf8')), + logs: JSON.parse(readFileSync(required(values, 'logs-json'), 'utf8')), + cloudSql: await readCloudSqlBackends( + 'staging', required(values, 'started-at'), required(values, 'ended-at') + ) + }), null, 2)}\n`) + return + } + if (command === 'create-c27') { + const startedAt = required(values, 'started-at') + const endedAt = required(values, 'ended-at') + writeFileSync(output, `${JSON.stringify(buildC27CanaryEvidence({ + ...commonInput(values), startedAt, endedAt, + loadReport: JSON.parse(readFileSync(required(values, 'load-report'), 'utf8')), + logs: JSON.parse(readFileSync(required(values, 'logs-json'), 'utf8')), + cloudSql: await readCloudSqlBackends('production', startedAt, endedAt) + }), null, 2)}\n`) + return + } + if (!['verify-staging', 'verify-c27'].includes(command)) throw new Error('invalid evidence command') + const kind = command === 'verify-staging' ? 'staging-asia-readiness' : 'production-c27-canary' + verifyRolloutEvidence( + JSON.parse(readFileSync(required(values, 'evidence'), 'utf8')), + JSON.parse(readFileSync(required(values, 'run-json'), 'utf8')), + { + kind, repository: REPOSITORY, + workflow: kind === 'staging-asia-readiness' ? STAGING_WORKFLOW : ADMISSION_WORKFLOW, + environment: kind === 'staging-asia-readiness' ? 'staging' : 'production', + commitSha: required(values, 'commit-sha'), imageDigest: required(values, 'image-digest'), + cellIds: [kind === 'staging-asia-readiness' ? STAGING_CELL : C27], + selectorGeneration: values.has('selector-generation') + ? positiveInteger(values.get('selector-generation'), 'selector generation') : undefined, + now: required(values, 'now') + } + ) + writeFileSync(output, 'verified\n') +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) await main(process.argv.slice(2)) diff --git a/cloud/dev/scripts/relay-asia-rollout-evidence.test.mjs b/cloud/dev/scripts/relay-asia-rollout-evidence.test.mjs new file mode 100644 index 00000000000..5d7b316605c --- /dev/null +++ b/cloud/dev/scripts/relay-asia-rollout-evidence.test.mjs @@ -0,0 +1,357 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { RELAY_GITHUB_REPOSITORY, relayWorkflowPath } from './relay-repository.mjs' +import { + buildC27CanaryEvidence, + buildStagingEvidence, + verifyRolloutEvidence +} from './relay-asia-rollout-evidence.mjs' + +const digest = `sha256:${'a'.repeat(64)}` +const commitSha = 'b'.repeat(40) +const repository = RELAY_GITHUB_REPOSITORY +const start = new Date('2026-08-13T12:00:00.000Z') +const end = new Date(start.valueOf() + 15 * 60_000) +const canaryEnd = new Date(start.valueOf() + 5 * 60_000) + +function sourceInput(overrides = {}) { + return { + repository, runId: 123, runAttempt: 2, commitSha, imageDigest: digest, + selectorGeneration: 9, ...overrides + } +} + +function metricLog(timestamp, role, cellId, overrides = {}) { + return { + timestamp: timestamp.toISOString(), + resource: { labels: role === 'director' ? { instance_id: 'director-1' } : {} }, + jsonPayload: { + role, cellId, region: role === 'cell' ? 'asia-east2' : 'us-central1', + controls: role === 'cell' ? 2_840 : 0, + splices: role === 'cell' ? 120 : 0, + selectedRegionsDelta: role === 'director' ? { 'asia-east2': 1 } : {}, + regionFallbacksDelta: {}, unavailableRegionsDelta: {}, sqlFailuresDelta: 0, + databasePoolWaiting: 0, databasePoolWaitersMax: 0, databasePoolWaitMsMax: 0, + ...overrides + } + } +} + +function completeLogs(cellId = 'production-gce-c27', windowEnd = end) { + const samples = (windowEnd.valueOf() - start.valueOf()) / 60_000 + 1 + return Array.from({ length: samples }, (_, minute) => { + const timestamp = new Date(start.valueOf() + minute * 60_000) + return [ + metricLog(timestamp, 'director', 'production-director'), + metricLog(timestamp, 'cell', cellId) + ] + }).flat() +} + +function cloudSql(max = 319, windowEnd = end) { + const samples = (windowEnd.valueOf() - start.valueOf()) / 60_000 + return { + timeSeries: [{ + points: Array.from({ length: samples }, (_, minute) => ({ + interval: { endTime: new Date(start.valueOf() + (minute + 1) * 60_000).toISOString() }, + value: { int64Value: String(minute === samples - 1 ? max : 300) } + })) + }] + } +} + +function loadReport({ + controls, shardIndex, splices = 0, slow = 0, wedged = 0, + launch = false +}) { + return { + event: 'relay_load_complete', controls, shardCount: 4, shardIndex, + configuredRampSeconds: 180, configuredSteadySeconds: 210, requiredLeaseHorizons: 2, + configuredSpliceHoldSeconds: 210, + configuredSplices: splices, configuredSlowReaderSplices: slow, + configuredWedgedReaderSplices: wedged, + peakActive: controls, steadyMinimumActive: controls, + connectionFailures: 0, rampConnectionFailures: 0, steadyConnectionFailures: 0, + transitionConnectionFailures: 0, unexpectedCloses: 0, protocolErrors: 0, + refreshErrors: 0, socketErrors: 0, failedSplices: 0, + regionalFallbacksProved: 0, + oldClientUsFirstProved: launch ? 1 : 0, + stickyAssignmentProved: launch ? 1 : 0, + requestUnitInvitesOpened: 0, + requestUnitPrincipalCount: 0, + relayAsiaLoadPrincipalCount: 32, + requestUnitOverflowReason: null, + requestUnitCleanupProved: 0, + phaseBarrierPassed: true, + rebindProbesOpened: launch ? 2 : 0, + rebindOverflowReason: null, + peakActiveSplices: splices, completedSplices: splices - wedged, + slowReaderSplicesCompleted: slow, wedgedReaderSplicesClosed: wedged, + readerQueuedBytesPeak: slow > 0 ? 1_024 : 0, + generatorCpuPercent: 25, generatorEventLoopP99Ms: 20, generatorRssGrowthMiB: 10, + readerQueueEvidence: slow > 0 ? [{ + origin: 'https://c4.relay-staging.onorca.dev', + baselineBytes: 128, + peakBytes: 1_152, + increaseBytes: 1_024 + }] : [], + shutdownEvidence: { + peerShutdowns: controls, activeControls: 0, activeSplices: 0, reconnectTimers: 0 + } + } +} + +function stagingInput(overrides = {}) { + const logs = completeLogs('staging-gce-c4') + logs[0].jsonPayload.regionFallbacksDelta = { 'us-central1': 1 } + return { + ...sourceInput({ selectorGeneration: 4 }), + startedAt: start.toISOString(), endedAt: end.toISOString(), + launchReport: Array.from({ length: 4 }, (_, shardIndex) => loadReport({ + controls: 5, shardIndex, splices: 5, launch: shardIndex === 0, + slow: shardIndex === 0 ? 4 : 0, wedged: shardIndex === 0 ? 1 : 0 + })), + logs, cloudSql: cloudSql(), ...overrides + } +} + +function canaryInput(overrides = {}) { + const load = loadReport({ controls: 1, shardIndex: 0, splices: 1 }) + Object.assign(load, { + shardCount: 1, + configuredSteadySeconds: 300, + configuredSpliceHoldSeconds: 60, + relayAsiaLoadPrincipalCount: 1 + }) + return { + ...sourceInput(), startedAt: start.toISOString(), endedAt: canaryEnd.toISOString(), + loadReport: load, + logs: completeLogs('production-gce-c27', canaryEnd), cloudSql: cloudSql(319, canaryEnd), + ...overrides + } +} + +function workflowRun(evidence, overrides = {}) { + return { + id: evidence.source.runId, run_attempt: evidence.source.runAttempt, + conclusion: 'success', event: 'workflow_dispatch', head_branch: 'main', + head_sha: evidence.source.commitSha, + repository: { full_name: evidence.source.repository }, path: evidence.source.workflow, + ...overrides + } +} + +function verifyExpected(kind, overrides = {}) { + const staging = kind === 'staging-asia-readiness' + return { + kind, repository, + workflow: staging + ? relayWorkflowPath('prove-relay-asia-staging.yml') + : relayWorkflowPath('operate-relay-asia-admission.yml'), + environment: staging ? 'staging' : 'production', commitSha, imageDigest: digest, + cellIds: [staging ? 'staging-gce-c4' : 'production-gce-c27'], + now: new Date(end.valueOf() + 60_000).toISOString(), ...overrides + } +} + +test('accepts the exact sharded staging load, telemetry, and provenance proof', () => { + const evidence = buildStagingEvidence(stagingInput()) + assert.equal(evidence.load.launch.controls, 20) + assert.equal(evidence.load.launch.peakActiveSplices, 20) + assert.equal(evidence.load.launch.readerQueueEvidence.length, 1) + assert.equal(verifyRolloutEvidence( + evidence, workflowRun(evidence), verifyExpected('staging-asia-readiness') + ), evidence) +}) + +test('ignores unrelated legacy cell metrics outside the Asia proof', () => { + const input = stagingInput() + const legacy = metricLog(start, 'cell', 'staging-gce-c1', { + region: undefined, + sqlFailuresDelta: 1 + }) + delete legacy.jsonPayload.databasePoolWaiting + delete legacy.jsonPayload.databasePoolWaitersMax + delete legacy.jsonPayload.databasePoolWaitMsMax + input.logs.push(legacy) + + assert.equal(buildStagingEvidence(input).metrics.relaySqlFailures, 0) +}) + +test('accepts bounded transient pool waits without a sampled queue', () => { + const input = stagingInput() + input.logs[0].jsonPayload.databasePoolWaitersMax = 4 + input.logs[0].jsonPayload.databasePoolWaitMsMax = 50 + + const metrics = buildStagingEvidence(input).metrics + assert.equal(metrics.databasePoolWaitingMax, 0) + assert.equal(metrics.databasePoolWaitersMax, 4) + assert.equal(metrics.databasePoolWaitMsMax, 50) +}) + +test('requires exactly the intentional sticky-assignment fallback in staging', () => { + const missing = stagingInput() + missing.logs[0].jsonPayload.regionFallbacksDelta = {} + assert.throws(() => buildStagingEvidence(missing), /intentional probes/) + + const extra = stagingInput() + extra.logs[2].jsonPayload.regionFallbacksDelta = { 'asia-east2': 1 } + assert.throws(() => buildStagingEvidence(extra), /intentional probes/) + + const substituted = stagingInput() + substituted.logs[0].jsonPayload.regionFallbacksDelta = { 'asia-east2': 1 } + assert.throws(() => buildStagingEvidence(substituted), /intentional probes/) +}) + +test('accepts the wedged splice outside the sustained non-wedged peak', () => { + const input = stagingInput() + input.launchReport[0].peakActiveSplices = 4 + input.logs.filter((entry) => entry.jsonPayload.role === 'cell') + .forEach((entry) => { entry.jsonPayload.splices = 19 }) + const evidence = buildStagingEvidence(input) + assert.equal(evidence.load.launch.peakActiveSplices, 19) + + input.launchReport[0].peakActiveSplices = 3 + assert.throws(() => buildStagingEvidence(input), /mixed load evidence/) +}) + +test('rejects staging evidence with incomplete load or cleanup', () => { + const input = stagingInput() + input.launchReport[0].steadyMinimumActive-- + assert.throws(() => buildStagingEvidence(input), /required controls/) + const cleanup = stagingInput() + cleanup.launchReport[0].shutdownEvidence.activeControls = 1 + assert.throws(() => buildStagingEvidence(cleanup), /cleanup/) + const nonCausal = stagingInput() + nonCausal.launchReport[0].readerQueueEvidence[0].increaseBytes = 0 + assert.throws(() => buildStagingEvidence(nonCausal), /not causal/) + const multipleOwners = stagingInput() + multipleOwners.launchReport[0].configuredSlowReaderSplices-- + multipleOwners.launchReport[0].slowReaderSplicesCompleted-- + multipleOwners.launchReport[1] = loadReport({ + controls: 5, shardIndex: 1, splices: 5, slow: 1 + }) + assert.throws(() => buildStagingEvidence(multipleOwners), /one causal owner/) + const overloaded = stagingInput() + overloaded.launchReport[0].generatorCpuPercent = 80 + assert.throws(() => buildStagingEvidence(overloaded), /insufficient headroom/) + const missingLaunchProof = stagingInput() + missingLaunchProof.launchReport[0].rebindProbesOpened = 0 + assert.throws(() => buildStagingEvidence(missingLaunchProof), /launch-path/) + const offTarget = stagingInput() + offTarget.logs.filter((entry) => entry.jsonPayload.role === 'cell') + .forEach((entry) => { entry.jsonPayload.controls = 19 }) + assert.throws(() => buildStagingEvidence(offTarget), /did not reach C4/) +}) + +test('rejects staging evidence with a shortened splice hold', () => { + const input = stagingInput() + input.launchReport[0].configuredSpliceHoldSeconds = 60 + assert.throws(() => buildStagingEvidence(input), /load profile does not match/) +}) + +for (const [label, value] of [['missing', undefined], ['malformed', 'invalid']]) { + test(`rejects staging evidence with a ${label} splice hold`, () => { + const input = stagingInput() + input.launchReport[0].configuredSpliceHoldSeconds = value + assert.throws(() => buildStagingEvidence(input), /staging splice hold seconds is invalid/) + }) +} + +for (const [label, mutate, message] of [ + ['phase barrier', (input) => { input.launchReport[0].phaseBarrierPassed = false }, /profile/], + ['old-client routing', (input) => { input.launchReport[0].oldClientUsFirstProved = 0 }, /launch-path/], + ['sticky routing', (input) => { input.launchReport[0].stickyAssignmentProved = 0 }, /launch-path/] +]) { + test(`rejects staging evidence without ${label} proof`, () => { + const input = stagingInput() + mutate(input) + assert.throws(() => buildStagingEvidence(input), message) + }) +} + +test('rejects staging evidence from a non-canonical repository', () => { + assert.throws(() => buildStagingEvidence(stagingInput({ repository: 'fork/orca-cloud' })), /repository/) +}) + +test('rejects mismatched staging provenance, digest, topology, or age', () => { + const evidence = buildStagingEvidence(stagingInput()) + const expected = verifyExpected('staging-asia-readiness') + assert.throws(() => verifyRolloutEvidence(evidence, workflowRun(evidence, { + head_sha: 'c'.repeat(40) + }), expected), /provenance/) + assert.throws(() => verifyRolloutEvidence(evidence, workflowRun(evidence), { + ...expected, commitSha: 'c'.repeat(40) + }), /provenance/) + assert.throws(() => verifyRolloutEvidence(evidence, workflowRun(evidence), { + ...expected, imageDigest: `sha256:${'c'.repeat(64)}` + }), /image digest/) + assert.throws(() => verifyRolloutEvidence({ + ...evidence, topology: { ...evidence.topology, cellIds: ['staging-gce-c3'] } + }, workflowRun(evidence), expected), /topology/) + assert.throws(() => verifyRolloutEvidence(evidence, workflowRun(evidence), { + ...expected, now: new Date(end.valueOf() + 25 * 60 * 60_000).toISOString() + }), /stale/) +}) + +test('builds and verifies a passing continuous 5-minute C27 canary', () => { + const evidence = buildC27CanaryEvidence(canaryInput()) + assert.equal(evidence.load.completedSplices, 1) + assert.equal(evidence.metrics.asiaSelections, 6) + assert.equal(evidence.metrics.cloudSqlBackendsMax, 319) + assert.equal(verifyRolloutEvidence( + evidence, workflowRun(evidence), + verifyExpected('production-c27-canary', { selectorGeneration: 9 }) + ), evidence) +}) + +test('rejects short, sparse, or unrelated-cell-only C27 coverage', () => { + assert.throws(() => buildC27CanaryEvidence(canaryInput({ + endedAt: new Date(canaryEnd.valueOf() - 1).toISOString() + })), /shorter than 5 minutes/) + const sparse = completeLogs('production-gce-c27', canaryEnd).filter((entry) => + entry.timestamp === start.toISOString() || entry.timestamp === canaryEnd.toISOString() + ) + assert.throws(() => buildC27CanaryEvidence(canaryInput({ logs: sparse })), /sampling gap/) + assert.throws(() => buildC27CanaryEvidence(canaryInput({ + logs: completeLogs('production-gce-c26', canaryEnd) + })), /production-gce-c27 metrics has no samples/) +}) + +test('rejects a C27 canary without a real control and splice', () => { + const input = canaryInput() + input.loadReport.completedSplices = 0 + assert.throws(() => buildC27CanaryEvidence(input), /did not match/) + const shortHold = canaryInput() + shortHold.loadReport.configuredSpliceHoldSeconds = 59 + assert.throws(() => buildC27CanaryEvidence(shortHold), /did not match/) +}) + +for (const [label, mutation, message] of [ + ['Asia selections', (input) => input.logs.forEach((entry) => { entry.jsonPayload.selectedRegionsDelta = {} }), /no Asia selections/], + ['region fallbacks', (input) => { input.logs[0].jsonPayload.regionFallbacksDelta = { 'asia-east2': 1 } }, /regionFallbacks/], + ['unavailable regions', (input) => { input.logs[0].jsonPayload.unavailableRegionsDelta = { 'asia-east2': 1 } }, /unavailableRegions/], + ['Relay SQL failures', (input) => { input.logs[0].jsonPayload.sqlFailuresDelta = 1 }, /relaySqlFailures/], + ['pool waiting', (input) => { input.logs[0].jsonPayload.databasePoolWaiting = 1 }, /databasePoolWaitingMax/], + ['pool waiters', (input) => { input.logs[0].jsonPayload.databasePoolWaitersMax = 5 }, /transient database pool pressure/], + ['pool wait time', (input) => { input.logs[0].jsonPayload.databasePoolWaitMsMax = 51 }, /transient database pool pressure/], + ['C27 controls', (input) => input.logs.filter((entry) => entry.jsonPayload.role === 'cell') + .forEach((entry) => { entry.jsonPayload.controls = 0 }), /did not reach C27/], + ['C27 splices', (input) => input.logs.filter((entry) => entry.jsonPayload.role === 'cell') + .forEach((entry) => { entry.jsonPayload.splices = 0 }), /did not reach C27/], + ['Cloud SQL headroom', (input) => { input.cloudSql = cloudSql(320, canaryEnd) }, /below 320/] +]) { + test(`rejects C27 evidence with ${label}`, () => { + const input = canaryInput() + mutation(input) + assert.throws(() => buildC27CanaryEvidence(input), message) + }) +} + +test('rejects C27 evidence from a different selector generation', () => { + const evidence = buildC27CanaryEvidence(canaryInput()) + assert.throws(() => verifyRolloutEvidence( + evidence, workflowRun(evidence), + verifyExpected('production-c27-canary', { selectorGeneration: 10 }) + ), /selector generation/) +}) diff --git a/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs b/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs new file mode 100644 index 00000000000..965a3ce142c --- /dev/null +++ b/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs @@ -0,0 +1,124 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { relayWorkflowUrl } from './relay-repository.mjs' + +const workflow = readFileSync( + relayWorkflowUrl('deploy-relay-asia-topology.yml'), + 'utf8' +) +const iam = readFileSync( + new URL('../../infra/terraform/relay-asia-topology-iam.tf', import.meta.url), + 'utf8' +) +const cells = readFileSync( + new URL('../../infra/terraform/relay-gce-cells.tf', import.meta.url), + 'utf8' +) +const variables = readFileSync( + new URL('../../infra/terraform/variables.tf', import.meta.url), + 'utf8' +) + +test('uses only its exact workflow-bound topology identity', () => { + assert.match(workflow, /production-cloud-sql-rollout/) + assert.match(workflow, /relay-staging-mutation/) + assert.match(workflow, /RELAY_ASIA_TOPOLOGY_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(workflow, /RELAY_ASIA_TOPOLOGY_SERVICE_ACCOUNT/) + assert.doesNotMatch(workflow, /GCP_DEPLOY_SERVICE_ACCOUNT/) + assert.match( + iam, + /assertion\.workflow_ref == '\$\{prefix\}\$\{local\.github_relay_asia_topology_workflow_file\}@refs\/heads\/main'/ + ) + assert.match(iam, /assertion\.ref == 'refs\/heads\/main'/) + assert.match(iam, /assertion\.event_name == 'workflow_dispatch'/) + assert.match(iam, /assertion\.environment == '\$\{var\.environment\}'/) +}) + +test('plans only additive Asia topology and applies the saved plan', () => { + assert.equal((workflow.match(/manage_artifact_dns=false/g) ?? []).length, 2) + for (const target of [ + 'relay_gce_additional', + 'google_compute_instance_template.relay_gce_cell', + 'google_compute_instance_group_manager.relay_gce_cell', + 'google_compute_backend_service.relay_gce_cell', + 'google_compute_url_map.relay_gce' + ]) assert.match(workflow, new RegExp(target.replaceAll('.', '\\.'))) + assert.match(workflow, /apply -input=false -auto-approve "\$\{\{ steps\.plan\.outputs\.plan \}\}"/) + assert.match(workflow, /prepare-relay-asia-topology-input\.mjs/) + assert.doesNotMatch(workflow, /steps\.variables\.outputs\.file/) + assert.match(workflow, /\.variables\.relay_gce_cells\.value/) + assert.match( + workflow, + /\.variables\.relay_gce_additional_region_subnetwork_cidrs\.value/ + ) + assert.doesNotMatch(workflow, /terraform -chdir=infra\/terraform console/) + assert.equal((workflow.match(/-var-file="\$\{TF_VARS\}"/g) ?? []).length, 2) + assert.doesNotMatch(workflow, /terraform[^\n]*apply[^\n]*-target/) + assert.doesNotMatch(workflow, /google_(?:sql|cloudflare|dns|certificate_manager)/) +}) + +test('validates before apply and proves convergence afterward', () => { + assert.equal((workflow.match(/validate-relay-asia-topology-plan\.mjs/g) ?? []).length, 2) + assert.match(workflow, /APPLY_RELAY_ASIA_TOPOLOGY/) + assert.match(workflow, /test "\$\(jq -er '\.changes'/) + assert.match(workflow, /Register the exact new cells atomically as migration-only/) +}) + +test('checks the connection budget and production live ceiling before planning', () => { + assert.match(workflow, /relay-cloud-sql-connection-budget\.mjs/) + assert.match(workflow, /gcloud sql instances describe "\$\{CLOUD_SQL_INSTANCE\}"/) + assert.match(workflow, /select\(\.name == "max_connections"\)/) + assert.match(workflow, /VERIFIED_DEFAULT_MAX_CONNECTIONS_TIER: db-custom-4-15360/) + assert.match(workflow, /VERIFIED_DEFAULT_MAX_CONNECTIONS_DATABASE_VERSION: POSTGRES_17/) + assert.match(workflow, /live_source=verified-shape-default/) + assert.match(workflow, /test "\$\(jq -er '\.settings\.tier'/) + assert.match(workflow, /test "\$\(jq -er '\.databaseVersion'/) + assert.match(workflow, /test "\$\{live_max\}" = "\$\{checked_max\}"/) + assert.ok( + workflow.indexOf('relay-cloud-sql-connection-budget.mjs') < + workflow.indexOf('terraform -chdir=infra/terraform plan') + ) +}) + +test('binds computed Asia references to the matching Terraform cell resources', () => { + assert.match(cells, /instance_template = google_compute_instance_template\.relay_gce_cell\[each\.key\]\.self_link/) + assert.match(cells, /group\s+= google_compute_instance_group_manager\.relay_gce_cell\[each\.key\]\.instance_group/) + assert.match(cells, /default_service = google_compute_backend_service\.relay_gce_cell\[cell\.key\]\.id/) + assert.match(cells, /subnetwork = local\.relay_gce_subnetworks\[each\.value\.region\]/) +}) + +test('keeps cross-variable region constraints in Terraform 1.5 check blocks', () => { + assert.doesNotMatch(variables, /region != var\.region/) + assert.doesNotMatch(variables, /cell\.region == var\.region/) + assert.match(cells, /check "relay_gce_fixed_one_topology"[\s\S]*?region != var\.region/) + assert.match(cells, /cell\.region == var\.region[\s\S]*?configured subnetwork/) +}) + +test('the custom role cannot delete topology or mutate SQL and DNS', () => { + assert.doesNotMatch(iam, /compute\.[A-Za-z]+\.delete/) + assert.doesNotMatch( + iam, + /roles\/viewer|cloudsql\.instances\.(?:update|delete)|dns\.|certificatemanager|cloudflare/i + ) + assert.match(iam, /resource "google_project_iam_custom_role" "github_relay_asia_topology_read"/) + assert.match(iam, /"cloudsql\.instances\.get"/) + assert.match(iam, /"run\.revisions\.get"/) + assert.match(iam, /"run\.services\.get"/) + assert.match(iam, /"serviceusage\.services\.list"/) + assert.match(iam, /"compute\.networks\.updatePolicy"/) + assert.match(iam, /"compute\.healthChecks\.useReadOnly"/) + assert.match(iam, /"compute\.instanceGroups\.create"/) + assert.match(iam, /"compute\.instances\.use"/) + assert.match(iam, /roles\/storage\.objectAdmin/) + assert.match(iam, /default\.tfstate/) + assert.match(iam, /default\.tflock/) + assert.match( + iam, + /resource "google_project_iam_custom_role" "github_relay_asia_topology_state_list"[\s\S]*?permissions = \["storage\.objects\.list"\]/ + ) + assert.match( + iam, + /resource "google_storage_bucket_iam_member" "github_relay_asia_topology_state_list"[\s\S]*?role\s+= google_project_iam_custom_role\.github_relay_asia_topology_state_list\[0\]\.id/ + ) +}) diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs new file mode 100644 index 00000000000..79036918f23 --- /dev/null +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.mjs @@ -0,0 +1,148 @@ +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +const DEFAULT_PATHS = { + productionTfvars: new URL('../../infra/terraform/environments/production.tfvars', import.meta.url), + terraformVariables: new URL('../../infra/terraform/variables.tf', import.meta.url), + relayConfig: new URL('../../apps/relay/src/config.ts', import.meta.url) +} + +// Auth and API live outside the Relay tree, so their consumption is published as a contract +// rather than parsed from their source. production-cloud-sql-app-consumers.test.mjs binds it back. +const APP_CONSUMERS_CONTRACT = new URL( + '../contracts/production-cloud-sql-app-consumers.json', + import.meta.url +) + +const APP_CONSUMER_FIELDS = ['authInstances', 'authPoolMax', 'apiInstances', 'apiPoolMax', 'maxConnections'] + +export function readProductionCloudSqlAppConsumers(contract) { + const parsed = contract ?? JSON.parse(readFileSync(APP_CONSUMERS_CONTRACT, 'utf8')) + for (const field of APP_CONSUMER_FIELDS) { + const value = parsed[field] + if (!Number.isSafeInteger(value) || value <= 0) { + throw new Error(`could not read ${field} from the app consumer contract`) + } + } + return parsed +} + +function requiredInteger(source, pattern, label) { + const value = Number(source.match(pattern)?.[1]) + if (!Number.isSafeInteger(value) || value < 0) throw new Error(`could not read ${label}`) + return value +} + +function productionCells(source, defaultPoolMax) { + const fencedMatch = source.match(/relay_gce_fenced_cells\s*=\s*\[([^\]]*)\]/) + if (!fencedMatch) throw new Error('could not read fenced Relay cells') + const fenced = new Set([...fencedMatch[1].matchAll(/"([^"]+)"/g)].map((match) => match[1])) + const cells = [...source.matchAll(/"(production-gce-[^"]+)"\s*=\s*\{([\s\S]*?)\n\s*\}/g)].map( + ([, id, body]) => ({ + id, + fenced: fenced.has(id), + region: body.match(/\bregion\s*=\s*"([^"]+)"/)?.[1] ?? 'us-central1', + poolMax: body.match(/\bdatabase_pool_max\s*=\s*(\d+)/) + ? Number(body.match(/\bdatabase_pool_max\s*=\s*(\d+)/)[1]) + : defaultPoolMax + }) + ) + if (cells.length === 0) throw new Error('could not read production Relay cells') + return cells +} + +export function calculateRelayCloudSqlConnectionBudget(inputs) { + const consumers = { + cells: inputs.cellPoolTotal + inputs.asiaCellCount * inputs.asiaPoolMax, + directors: inputs.directorInstances * inputs.directorPoolMax, + auth: inputs.authInstances * inputs.authPoolMax, + api: inputs.apiInstances * inputs.apiPoolMax + } + const configuredMaximum = Object.values(consumers).reduce((total, value) => total + value, 0) + const retainedDirectorRollback = inputs.directorInstances * inputs.directorPoolMax + const candidateOverlap = { + relayDirectorCandidate: retainedDirectorRollback * 2, + apiCandidate: retainedDirectorRollback + inputs.apiInstances * inputs.apiPoolMax, + authCandidate: retainedDirectorRollback + inputs.authInstances * inputs.authPoolMax, + relayCells: retainedDirectorRollback + } + const rolloutOverlap = Math.max(...Object.values(candidateOverlap)) + const operatingMaximum = configuredMaximum + rolloutOverlap + inputs.maintenanceAdminAllowance + const usableCeiling = inputs.maxConnections - inputs.explicitReserve + const budgetedTotal = operatingMaximum + inputs.explicitReserve + return { + maxConnections: inputs.maxConnections, + consumers, + asia: { cells: inputs.asiaCellCount, poolMax: inputs.asiaPoolMax }, + configuredMaximum, + rolloutOverlap: { + ...candidateOverlap, + retainedDirectorRollback, + maximum: rolloutOverlap, + reason: 'serialized rollouts include directly addressable tagged revisions outside service-level caps' + }, + maintenanceAdminAllowance: inputs.maintenanceAdminAllowance, + maintenanceAdminAllowanceReason: 'covers bounded work outside configured services', + explicitReserve: inputs.explicitReserve, + explicitReserveReason: 'remains unavailable to configured services and planned rollouts', + usableCeiling, + operatingMaximum, + remainingWithinUsableCeiling: usableCeiling - operatingMaximum, + budgetedTotal, + unallocated: inputs.maxConnections - budgetedTotal, + withinBudget: operatingMaximum <= usableCeiling && budgetedTotal < inputs.maxConnections + } +} + +export function readRelayCloudSqlConnectionBudget({ + sources, + appConsumers, + proposedAsiaCellCount = 3, + asiaPoolMax = 10, + maxConnections, + maintenanceAdminAllowance = 5, + explicitReserve = 10 +} = {}) { + const read = (name) => sources?.[name] ?? readFileSync(DEFAULT_PATHS[name], 'utf8') + const apps = readProductionCloudSqlAppConsumers(appConsumers) + const productionTfvars = read('productionTfvars') + const terraformVariables = read('terraformVariables') + const relayConfig = read('relayConfig') + const cells = productionCells( + productionTfvars, + requiredInteger(relayConfig, /RELAY_DATABASE_POOL_MAX\s*=\s*(\d+)/, 'Relay pool maximum') + ) + const poweredCells = cells.filter(({ fenced }) => !fenced) + const configuredAsiaCells = poweredCells.filter(({ region }) => region === 'asia-east2') + const nonAsiaCells = poweredCells.filter(({ region }) => region !== 'asia-east2') + const cellPoolTotal = nonAsiaCells.reduce((total, cell) => total + cell.poolMax, 0) + const asiaCellCount = configuredAsiaCells.length || proposedAsiaCellCount + const configuredAsiaPoolMax = configuredAsiaCells[0]?.poolMax ?? asiaPoolMax + if (configuredAsiaCells.some(({ poolMax }) => poolMax !== configuredAsiaPoolMax)) { + throw new Error('Asia Relay cells must use one checked pool maximum') + } + return calculateRelayCloudSqlConnectionBudget({ + cellPoolTotal, + asiaCellCount, + asiaPoolMax: configuredAsiaPoolMax, + directorInstances: requiredInteger(productionTfvars, /relay_max_instances\s*=\s*(\d+)/, 'director instances'), + directorPoolMax: requiredInteger( + terraformVariables, + /variable\s+"relay_director_database_pool_max"[\s\S]*?default\s*=\s*(\d+)/, + 'director pool maximum' + ), + authInstances: apps.authInstances, + authPoolMax: apps.authPoolMax, + apiInstances: apps.apiInstances, + apiPoolMax: apps.apiPoolMax, + maxConnections: maxConnections ?? apps.maxConnections, + maintenanceAdminAllowance, + explicitReserve + }) +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const report = readRelayCloudSqlConnectionBudget() + console.log(JSON.stringify({ event: 'relay_cloud_sql_connection_budget', ...report }, null, 2)) + if (!report.withinBudget) process.exitCode = 1 +} diff --git a/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs new file mode 100644 index 00000000000..a26d24c274d --- /dev/null +++ b/cloud/dev/scripts/relay-cloud-sql-connection-budget.test.mjs @@ -0,0 +1,110 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { + calculateRelayCloudSqlConnectionBudget, + readRelayCloudSqlConnectionBudget +} from './relay-cloud-sql-connection-budget.mjs' + +test('production plus three Asia pools preserves allowance and reserve below the ceiling', () => { + const report = readRelayCloudSqlConnectionBudget() + + assert.deepEqual(report.consumers, { cells: 230, directors: 15, auth: 20, api: 50 }) + assert.deepEqual(report.asia, { cells: 3, poolMax: 10 }) + assert.equal(report.configuredMaximum, 315) + assert.equal(report.rolloutOverlap.relayDirectorCandidate, 30) + assert.equal(report.rolloutOverlap.apiCandidate, 65) + assert.equal(report.rolloutOverlap.authCandidate, 35) + assert.equal(report.rolloutOverlap.relayCells, 15) + assert.equal(report.rolloutOverlap.retainedDirectorRollback, 15) + assert.equal(report.rolloutOverlap.maximum, 65) + assert.equal(report.maintenanceAdminAllowance, 5) + assert.equal(report.explicitReserve, 10) + assert.equal(report.usableCeiling, 390) + assert.equal(report.operatingMaximum, 385) + assert.equal(report.remainingWithinUsableCeiling, 5) + assert.equal(report.budgetedTotal, 395) + assert.equal(report.unallocated, 5) + assert.equal(report.withinBudget, true) +}) + +test('fails closed when pool growth consumes the explicit reserve', () => { + const report = calculateRelayCloudSqlConnectionBudget({ + cellPoolTotal: 200, + asiaCellCount: 3, + asiaPoolMax: 20, + directorInstances: 5, + directorPoolMax: 3, + authInstances: 2, + authPoolMax: 10, + apiInstances: 20, + apiPoolMax: 5, + maxConnections: 400, + maintenanceAdminAllowance: 5, + explicitReserve: 10 + }) + + assert.equal(report.operatingMaximum, 515) + assert.equal(report.withinBudget, false) +}) + +test('excludes fenced cell pools and reads per-cell pool overrides', () => { + const report = readRelayCloudSqlConnectionBudget({ + proposedAsiaCellCount: 1, + appConsumers: { authInstances: 1, authPoolMax: 10, apiInstances: 1, apiPoolMax: 5, maxConnections: 100 }, + sources: { + productionTfvars: ` + relay_max_instances = 1 + relay_gce_fenced_cells = ["production-gce-c1"] + relay_gce_cells = { + "production-gce-c1" = { database_pool_max = 99 + } + "production-gce-c2" = { database_pool_max = 4 + } + } + `, + terraformVariables: 'variable "relay_director_database_pool_max" { default = 3 }', + relayConfig: 'export const RELAY_DATABASE_POOL_MAX = 10' + }, + maxConnections: 100, + maintenanceAdminAllowance: 1, + explicitReserve: 1 + }) + + assert.equal(report.consumers.cells, 14) + assert.equal(report.operatingMaximum, 46) + assert.equal(report.budgetedTotal, 47) +}) + +test('requires strict headroom below the physical ceiling', () => { + const report = calculateRelayCloudSqlConnectionBudget({ + cellPoolTotal: 20, + asiaCellCount: 0, + asiaPoolMax: 10, + directorInstances: 1, + directorPoolMax: 3, + authInstances: 1, + authPoolMax: 10, + apiInstances: 1, + apiPoolMax: 5, + maxConnections: 50, + maintenanceAdminAllowance: 9, + explicitReserve: 3 + }) + + assert.equal(report.budgetedTotal, 63) + assert.equal(report.withinBudget, false) +}) + +test('pages Relay channels when Cloud SQL backends consume headroom', () => { + const terraform = readFileSync( + new URL('../../infra/terraform/relay-observability.tf', import.meta.url), + 'utf8' + ) + const policy = terraform.match( + /resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" \{([\s\S]*?)\n\}/ + )?.[1] + + assert.ok(policy) + assert.match(policy, /notification_channels\s*=\s*var\.relay_alert_notification_channels/) +}) diff --git a/cloud/dev/scripts/relay-gce-terraform-fence.mjs b/cloud/dev/scripts/relay-gce-terraform-fence.mjs new file mode 100644 index 00000000000..82b9dcbff6c --- /dev/null +++ b/cloud/dev/scripts/relay-gce-terraform-fence.mjs @@ -0,0 +1,1387 @@ +import { execFileSync } from 'node:child_process' +import { createHash, randomUUID } from 'node:crypto' +import { + chmodSync, + mkdtempSync, + readFileSync, + rmSync, + statSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' + +const MIG_ADDRESS_PREFIX = 'google_compute_instance_group_manager.relay_gce_cell' +const PLAN_OBJECT_PREFIX = 'terraform/state/relay-fence-plans' +const TERRAFORM_STATE_LINEAGE = /^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/i +const GENERATED_MIG_VERSION_NAME = + /^0\/[0-9]{4}-[0-9]{2}-[0-9]{2} [0-9]{2}:[0-9]{2}:[0-9]{2}(?:\.[0-9]+)?\+00:00$/ + +function changedActions(change) { + return change.change.actions.filter((action) => action !== 'no-op' && action !== 'read') +} + +export function validateTerraformFencePlan(plan, expected) { + const changes = (plan.resource_changes ?? []).filter( + (change) => changedActions(change).length > 0 + ) + if (changes.length !== 1) throw new Error('fence plan must contain exactly one mutation') + const [change] = changes + if (change.address !== `${MIG_ADDRESS_PREFIX}["${expected.cellId}"]`) { + throw new Error('fence plan mutates an unexpected resource') + } + if ( + JSON.stringify(change.change.actions) !== JSON.stringify(['update']) || + Number(change.change.before?.target_size) !== 1 || + Number(change.change.after?.target_size) !== 0 + ) { + throw new Error('fence plan must update only the requested MIG from one to zero') + } + for (const field of ['name', 'zone', 'instance_group']) { + if ( + change.change.before?.[field] !== change.change.after?.[field] || + change.change.after?.[field] !== expected[field] + ) { + throw new Error(`fence plan changed or mismatched MIG ${field}`) + } + } + const beforeGeneration = change.change.before?.version?.[0]?.instance_template + const afterGeneration = change.change.after?.version?.[0]?.instance_template + if ( + beforeGeneration !== afterGeneration || + afterGeneration !== expected.generationIdentity + ) { + throw new Error('fence plan changed or mismatched the MIG generation') + } + return change +} + +export function validateTerraformFenceCompletionPlan(plan, expected) { + const changes = (plan.resource_changes ?? []).filter( + (change) => changedActions(change).length > 0 + ) + if (changes.length === 0) return + if (changes.length !== 1) { + throw new Error('completed fence plan contains an unexpected mutation') + } + const [change] = changes + const before = change.change.before + const after = change.change.after + const beforeVersion = before?.version?.[0] + const afterVersion = after?.version?.[0] + if ( + change.address !== `${MIG_ADDRESS_PREFIX}["${expected.cellId}"]` || + JSON.stringify(change.change.actions) !== JSON.stringify(['update']) || + Number(before?.target_size) !== 0 || + Number(after?.target_size) !== 0 || + before?.name !== expected.name || + after?.name !== expected.name || + before?.zone !== expected.zone || + after?.zone !== expected.zone || + before?.instance_group !== expected.instance_group || + after?.instance_group !== expected.instance_group || + before?.version?.length !== 1 || + after?.version?.length !== 1 || + beforeVersion?.instance_template !== expected.generationIdentity || + afterVersion?.instance_template !== expected.generationIdentity || + !GENERATED_MIG_VERSION_NAME.test(beforeVersion?.name ?? '') || + afterVersion?.name !== 'primary' + ) { + throw new Error('completed fence plan is not a safe provider normalization') + } + const normalizedBefore = structuredClone(before) + normalizedBefore.version[0].name = afterVersion.name + if (JSON.stringify(normalizedBefore) !== JSON.stringify(after)) { + throw new Error('completed fence plan changes more than the provider version label') + } + return change +} + +export function terraformFenceState(state, expected) { + const resources = state.values?.root_module?.resources ?? [] + const matches = resources.filter( + (resource) => resource.address === `${MIG_ADDRESS_PREFIX}["${expected.cellId}"]` + ) + if (matches.length !== 1) throw new Error('Terraform state has no unique requested MIG') + const values = matches[0].values + if ( + values.name !== expected.name || + values.zone !== expected.zone || + values.instance_group !== expected.instance_group || + values.version?.[0]?.instance_template !== expected.generationIdentity + ) { + throw new Error('Terraform state MIG identity does not match the reviewed topology') + } + const targetSize = Number(values.target_size) + if (![0, 1].includes(targetSize)) throw new Error('Terraform state MIG size is unsafe') + return targetSize +} + +export function classifyTerraformFenceProgress({ + stateTargetSize, + liveTargetSize, + instanceCount, + operationStatus, + operationError = false, + operationAuditBound = false +}) { + if (operationError) throw new Error('Terraform fence GCE operation failed') + if (operationStatus === 'DONE' && !operationAuditBound) { + throw new Error('Terraform fence GCE operation lacks exact audit binding') + } + if ( + stateTargetSize === 0 && + liveTargetSize === 0 && + instanceCount === 0 && + operationStatus === 'DONE' + ) { + return 'complete' + } + if ( + [0, 1].includes(stateTargetSize) && + [0, 1].includes(liveTargetSize) && + ['PENDING', 'RUNNING'].includes(operationStatus) + ) { + return 'in-progress' + } + if ( + stateTargetSize === 1 && + liveTargetSize === 0 && + instanceCount === 0 && + operationStatus === 'DONE' + ) { + return 'reconcile-state' + } + if ( + stateTargetSize === 1 && + liveTargetSize === 1 && + instanceCount === 1 && + operationStatus === 'ABSENT' + ) { + return 'not-started' + } + throw new Error('Terraform fence progress is ambiguous or unsafe') +} + +function sha256(path, readFile) { + return createHash('sha256').update(readFile(path)).digest('hex') +} + +function terraformJson(deps, args) { + return JSON.parse(deps.terraform(args, { encoding: 'utf8' })) +} + +export function terraformProcessStdio(options = {}) { + if (!options.encoding) return 'inherit' + return [options.input === undefined ? 'ignore' : 'pipe', 'pipe', 'pipe'] +} + +function defaultTerraform(args, options = {}) { + return execFileSync(process.env.IAC_TOOL || 'terraform', args, { + ...options, + stdio: terraformProcessStdio(options) + }) +} + +function defaultGit(args, options = {}) { + return execFileSync('git', args, { + ...options, + stdio: options.encoding ? ['ignore', 'pipe', 'pipe'] : 'inherit' + }) +} + +function privatePlanDirectory(deps) { + const previousMask = process.umask(0o077) + try { + const directory = deps.mkdtemp(join(deps.tmpdir(), 'orca-relay-fence-')) + deps.chmod(directory, 0o700) + return directory + } finally { + process.umask(previousMask) + } +} + +function exactMigExpected(cell) { + return { + cellId: cell.cellId, + name: cell.migName, + zone: cell.zone, + instance_group: cell.instanceGroup, + generationIdentity: cell.generationIdentity + } +} + +function assertFenceIdentity(cell) { + if (!cell.generationIdentity) throw new Error('requested cell has no generation identity') +} + +function assertAttemptMatches(config, attempt, requirePlanGeneration = true) { + for (const [field, expected] of Object.entries({ + environment: config.environment, + cellId: config.cell.cellId, + cellIncarnation: config.cellIncarnation, + migName: config.cell.migName, + instanceGroup: config.cell.instanceGroup, + generationIdentity: config.cell.generationIdentity, + fenceCommit: config.fenceCommit + })) { + if (attempt[field] !== expected) throw new Error(`fence attempt ${field} mismatch`) + } + if (!/^[a-f0-9]{64}$/.test(attempt.planSha256 ?? '')) { + throw new Error('fence attempt has no valid saved-plan digest') + } + if ( + attempt.planObjectName !== + `${PLAN_OBJECT_PREFIX}/${config.environment}/${attempt.attemptId}.tfplan` + ) { + throw new Error('fence attempt saved-plan object mismatch') + } + if ( + requirePlanGeneration && + !/^[1-9][0-9]{0,30}$/.test(attempt.planObjectGeneration ?? '') + ) { + throw new Error('fence attempt has no valid saved-plan generation') + } + if (!/^[a-f0-9]{64}$/.test(attempt.varFileSha256 ?? '')) { + throw new Error('fence attempt has no valid variable-file digest') + } + if ( + !TERRAFORM_STATE_LINEAGE.test(attempt.terraformStateLineage ?? '') || + !Number.isSafeInteger(attempt.terraformStateSerial) || + attempt.terraformStateSerial < 0 + ) { + throw new Error('fence attempt has no valid Terraform state identity') + } + if ( + !/^[1-9][0-9]{0,30}$/.test( + attempt.terraformStateObjectGeneration ?? '' + ) || + !/^[a-f0-9]{64}$/.test(attempt.terraformStateObjectSha256 ?? '') + ) { + throw new Error('fence attempt has no valid Terraform state object binding') + } + if (attempt.requestReason !== `orca-relay-fence/${attempt.attemptId}`) { + throw new Error('fence attempt request-reason mismatch') + } +} + +export function terraformFenceStateIdentity(config, deps = {}) { + const terraform = deps.terraform ?? defaultTerraform + const state = terraformJson({ terraform }, [ + `-chdir=${config.terraformDir}`, + 'state', + 'pull' + ]) + if ( + !TERRAFORM_STATE_LINEAGE.test(state.lineage ?? '') || + !Number.isSafeInteger(state.serial) || + state.serial < 0 + ) { + throw new Error('Terraform state has no valid lineage or serial') + } + return { lineage: state.lineage, serial: state.serial } +} + +export function assertReviewedFenceCheckout(config, deps = {}) { + const environment = deps.environment ?? process.env + const git = deps.git ?? defaultGit + const readFile = deps.readFile ?? readFileSync + const varPath = join(config.terraformDir, config.varFile) + const imageCommit = environment.ORCA_RELAY_FENCE_IMAGE_COMMIT + if (imageCommit !== undefined) { + if (!/^[a-f0-9]{40}$/.test(imageCommit) || imageCommit !== config.fenceCommit) { + throw new Error('fence commit does not match immutable broker image') + } + return sha256(varPath, readFile) + } + const head = git(['rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() + if (head !== config.fenceCommit) throw new Error('fence commit does not match checked-out HEAD') + const status = git(['status', '--porcelain=v1', '--untracked-files=no'], { + encoding: 'utf8' + }).trim() + if (status) throw new Error('Terraform fence requires a clean checkout') + const terraformStatus = git( + ['status', '--porcelain=v1', '--untracked-files=all', '--', config.terraformDir], + { encoding: 'utf8' } + ).trim() + if (terraformStatus) throw new Error('Terraform fence directory contains unreviewed files') + git(['ls-files', '--error-unmatch', '--', varPath], { encoding: 'utf8' }) + return sha256(varPath, readFile) +} + +function assertAttemptCheckoutBinding(config, attempt, deps) { + const digest = assertReviewedFenceCheckout(config, deps) + if (digest !== attempt.varFileSha256) { + throw new Error('reviewed Terraform variable-file digest changed') + } +} + +function assertReplayStateIdentity(progress, attempt) { + if ( + progress.stateLineage !== attempt.terraformStateLineage || + progress.stateSerial !== attempt.terraformStateSerial + ) { + throw new Error('Terraform state lineage or serial changed before saved-plan replay') + } +} + +function assertStateObjectBinding(binding, attempt) { + if ( + binding.generation !== attempt.terraformStateObjectGeneration || + binding.sha256 !== attempt.terraformStateObjectSha256 || + binding.lineage !== attempt.terraformStateLineage || + binding.serial !== attempt.terraformStateSerial + ) { + throw new Error('Terraform pre-state object generation or digest changed') + } +} + +function completedStateBranch(progress, attempt) { + if (progress.stateLineage !== attempt.terraformStateLineage) { + throw new Error('completed Terraform fence state identity is unexpected') + } + if (progress.stateSerial === attempt.terraformStateSerial) return 'replay' + if (progress.stateSerial === attempt.terraformStateSerial + 1) return 'complete' + throw new Error('completed Terraform fence state identity is unexpected') +} + +async function persistInvocationOperations(attempt, progress, markOperation) { + let updated = attempt + for (const observed of progress.invocationOperations ?? []) { + const recorded = (updated.applyInvocations ?? []).find( + (value) => value.invocationId === observed.invocationId + ) + if (!observed.gceOperation || recorded?.gceOperation) continue + const marked = await markOperation( + { ...updated, gceOperation: observed.gceOperation }, + observed + ) + updated = { + ...marked.attempt, + applyInvocations: (updated.applyInvocations ?? []).map((value) => + value.invocationId === marked.invocation.invocationId + ? marked.invocation + : value + ) + } + } + return updated +} + +export function assertTerraformFenceZeroDiff(config, deps = {}) { + const terraform = deps.terraform ?? defaultTerraform + const directory = privatePlanDirectory({ + mkdtemp: deps.mkdtemp ?? mkdtempSync, + chmod: deps.chmod ?? chmodSync, + tmpdir: deps.tmpdir ?? tmpdir + }) + const planPath = join(directory, 'completion.tfplan') + try { + terraform( + [ + `-chdir=${config.terraformDir}`, + 'plan', + '-input=false', + '-refresh=false', + `-lock-timeout=${config.lockTimeout}`, + `-var-file=${config.varFile}`, + `-target=${MIG_ADDRESS_PREFIX}["${config.cell.cellId}"]`, + `-out=${planPath}`, + '-no-color' + ], + { encoding: 'utf8' } + ) + const plan = terraformJson({ terraform }, [ + `-chdir=${config.terraformDir}`, + 'show', + '-json', + planPath + ]) + validateTerraformFenceCompletionPlan(plan, exactMigExpected(config.cell)) + } catch { + throw new Error('completed Terraform fence has an unsafe reviewed diff') + } finally { + ;(deps.remove ?? rmSync)(directory, { recursive: true, force: true }) + } +} + +export function assertTerraformFenceStateFenced(config, deps = {}) { + const terraform = deps.terraform ?? defaultTerraform + const state = terraformJson({ terraform }, [ + `-chdir=${config.terraformDir}`, + 'show', + '-json' + ]) + if (terraformFenceState(state, exactMigExpected(config.cell)) !== 0) { + throw new Error('Terraform state does not record the requested cell fence') + } +} + +export async function adoptLegacyTerraformFence(config, deps) { + for (const dependency of [ + 'loadAttempt', + 'assertCommittedFenceSet', + 'assertStateFenced', + 'preApplyGuard', + 'postApplyGuard', + 'attest', + 'commitAdoption' + ]) { + if (typeof deps[dependency] !== 'function') throw new Error(`missing ${dependency} dependency`) + } + assertFenceIdentity(config.cell) + assertReviewedFenceCheckout(config, deps) + if (await deps.loadAttempt()) { + throw new Error('legacy Terraform fence adoption requires no durable attempt') + } + await deps.assertCommittedFenceSet() + await deps.preApplyGuard() + await deps.assertStateFenced() + await deps.postApplyGuard(config.cellIncarnation) + if (await deps.loadAttempt()) { + throw new Error('durable fence attempt appeared during legacy adoption') + } + await deps.attest(config.cellIncarnation) + await deps.postApplyGuard(config.cellIncarnation) + await deps.commitAdoption(config.cellIncarnation) + deps.emit?.({ + event: 'terraform_cell_fence_legacy_adopted', + cellId: config.cell.cellId + }) +} + +function planObjectUri(config, attempt) { + return `gs://${config.project}-terraform-state/${attempt.planObjectName}` +} + +export async function readTerraformStateObjectBinding( + config, + deps, + statePath +) { + const uri = `gs://${config.project}-terraform-state/terraform/state/default.tfstate` + const metadata = deps.commandJson([ + 'storage', + 'objects', + 'describe', + uri, + '--format=json' + ]) + const generation = String(metadata.generation ?? '') + if (!/^[1-9][0-9]{0,30}$/.test(generation)) { + throw new Error('Terraform state object has no valid generation') + } + deps.command([ + 'storage', + 'cp', + `${uri}#${generation}`, + statePath, + `--if-generation-match=${generation}`, + '--quiet' + ]) + ;(deps.chmod ?? chmodSync)(statePath, 0o600) + const contents = (deps.readFile ?? readFileSync)(statePath) + const state = JSON.parse(contents.toString()) + if ( + !TERRAFORM_STATE_LINEAGE.test(state.lineage ?? '') || + !Number.isSafeInteger(state.serial) || + state.serial < 0 + ) { + throw new Error('Terraform state object identity is invalid') + } + return { + generation, + sha256: createHash('sha256').update(contents).digest('hex'), + lineage: state.lineage, + serial: state.serial + } +} + +export async function uploadTerraformFencePlan(config, deps, planPath, attempt) { + const uri = planObjectUri(config, attempt) + deps.command([ + 'storage', + 'cp', + planPath, + uri, + '--if-generation-match=0', + '--quiet' + ]) + const metadata = deps.commandJson([ + 'storage', + 'objects', + 'describe', + uri, + '--format=json' + ]) + const generation = String(metadata.generation ?? '') + if (!/^[1-9][0-9]{0,30}$/.test(generation)) { + throw new Error('uploaded fence plan has no valid object generation') + } + return { generation } +} + +export async function resolveTerraformFencePlanGeneration(config, deps, attempt) { + const result = deps.commandResult([ + 'storage', + 'objects', + 'describe', + planObjectUri(config, attempt), + '--format=value(generation)' + ]) + if (result.status !== 0) { + if (/(?:404|not found|no urls matched)/i.test(result.stderr ?? '')) { + return { generation: null } + } + throw new Error('durable fence plan object could not be inspected') + } + const generation = String(result.stdout ?? '').trim() + if (!/^[1-9][0-9]{0,30}$/.test(generation)) { + throw new Error('durable fence plan has no valid object generation') + } + return { generation } +} + +export async function downloadTerraformFencePlan(config, deps, attempt, planPath) { + const uri = `${planObjectUri(config, attempt)}#${attempt.planObjectGeneration}` + deps.command([ + 'storage', + 'cp', + uri, + planPath, + `--if-generation-match=${attempt.planObjectGeneration}`, + '--quiet' + ]) + ;(deps.chmod ?? chmodSync)(planPath, 0o600) +} + +export async function deleteTerraformFencePlan(config, deps, attempt) { + const result = deps.commandResult([ + 'storage', + 'rm', + `${planObjectUri(config, attempt)}#${attempt.planObjectGeneration}`, + `--if-generation-match=${attempt.planObjectGeneration}`, + '--quiet' + ]) + if (result.status === 0) return + if (/(?:404|not found|no urls matched)/i.test(result.stderr ?? '')) return + throw new Error('exact Terraform fence plan generation could not be deleted') +} + +export async function runTerraformFenceApply(config, overrides = {}) { + const terraform = overrides.terraform ?? defaultTerraform + const deps = { + terraform, + git: overrides.git ?? defaultGit, + mkdtemp: overrides.mkdtemp ?? mkdtempSync, + chmod: overrides.chmod ?? chmodSync, + tmpdir: overrides.tmpdir ?? tmpdir, + readFile: overrides.readFile ?? readFileSync, + remove: overrides.remove ?? rmSync, + stat: overrides.stat ?? statSync, + randomUUID: overrides.randomUUID ?? randomUUID, + inspectProgress: overrides.inspectProgress, + prepareAttempt: overrides.prepareAttempt, + bindPlan: overrides.bindPlan, + markApplyStarted: overrides.markApplyStarted, + markOperation: overrides.markOperation, + attest: overrides.attest, + uploadPlan: overrides.uploadPlan, + deletePlan: overrides.deletePlan, + stateObjectBinding: overrides.stateObjectBinding, + assertZeroDiff: + overrides.assertZeroDiff ?? + (async () => assertTerraformFenceZeroDiff(config, { terraform })), + preApplyGuard: overrides.preApplyGuard, + postApplyGuard: overrides.postApplyGuard, + assertCommittedFenceSet: + overrides.assertCommittedFenceSet ?? + (() => assertTerraformFenceSet(config, { terraform })), + emit: overrides.emit ?? (() => {}) + } + for (const dependency of [ + 'inspectProgress', + 'prepareAttempt', + 'bindPlan', + 'markApplyStarted', + 'markOperation', + 'attest', + 'uploadPlan', + 'deletePlan', + 'stateObjectBinding', + 'assertZeroDiff', + 'preApplyGuard', + 'postApplyGuard' + ]) { + if (typeof deps[dependency] !== 'function') throw new Error(`missing ${dependency} dependency`) + } + assertFenceIdentity(config.cell) + const varFileSha256 = assertReviewedFenceCheckout(config, deps) + await deps.assertCommittedFenceSet() + const expected = exactMigExpected(config.cell) + const directory = privatePlanDirectory(deps) + const planPath = join(directory, 'fence.tfplan') + try { + const initialProgress = await deps.inspectProgress(expected, null) + if (classifyTerraformFenceProgress(initialProgress) !== 'not-started') { + throw new Error('Terraform fence is not in the exact initial live state') + } + const stateBinding = await deps.stateObjectBinding( + join(directory, 'pre-state.tfstate') + ) + deps.terraform([ + `-chdir=${config.terraformDir}`, + 'plan', + '-input=false', + '-refresh=false', + `-lock-timeout=${config.lockTimeout}`, + `-var-file=${config.varFile}`, + `-target=${MIG_ADDRESS_PREFIX}["${config.cell.cellId}"]`, + `-out=${planPath}` + ]) + deps.chmod(planPath, 0o600) + const postPlanStateBinding = await deps.stateObjectBinding( + join(directory, 'post-plan-state.tfstate') + ) + if ( + JSON.stringify(postPlanStateBinding) !== JSON.stringify(stateBinding) + ) { + throw new Error('Terraform state object changed while creating the fence plan') + } + if ((deps.stat(planPath).mode & 0o077) !== 0) { + throw new Error('saved fence plan permissions are not private') + } + const plan = terraformJson(deps, [ + `-chdir=${config.terraformDir}`, + 'show', + '-json', + planPath + ]) + validateTerraformFencePlan(plan, expected) + const planSha256 = sha256(planPath, deps.readFile) + const attemptId = deps.randomUUID() + const planObjectName = + `${PLAN_OBJECT_PREFIX}/${config.environment}/${attemptId}.tfplan` + const attempt = { + attemptId, + environment: config.environment, + cellId: config.cell.cellId, + cellIncarnation: config.cellIncarnation, + migName: config.cell.migName, + instanceGroup: config.cell.instanceGroup, + generationIdentity: config.cell.generationIdentity, + fenceCommit: config.fenceCommit, + planSha256, + planObjectName, + varFileSha256, + terraformStateLineage: stateBinding.lineage, + terraformStateSerial: stateBinding.serial, + terraformStateObjectGeneration: stateBinding.generation, + terraformStateObjectSha256: stateBinding.sha256, + requestReason: `orca-relay-fence/${attemptId}` + } + const prepared = await deps.prepareAttempt(attempt) + let durableAttempt = prepared?.attempt ?? prepared + if ( + !durableAttempt || + !Number.isSafeInteger(durableAttempt.createdAt) || + !Number.isSafeInteger(durableAttempt.expiresAt) + ) { + throw new Error('durable fence attempt has no creation or expiry time') + } + assertAttemptMatches(config, durableAttempt, false) + const uploaded = await deps.uploadPlan(planPath, durableAttempt) + const bound = await deps.bindPlan({ + ...durableAttempt, + planObjectGeneration: uploaded.generation + }) + durableAttempt = bound?.attempt ?? bound + assertAttemptMatches(config, durableAttempt) + assertAttemptCheckoutBinding(config, durableAttempt, deps) + await deps.preApplyGuard() + validateTerraformFencePlan( + terraformJson(deps, [ + `-chdir=${config.terraformDir}`, + 'show', + '-json', + planPath + ]), + expected + ) + if (sha256(planPath, deps.readFile) !== planSha256) { + throw new Error('saved fence plan digest changed') + } + assertStateObjectBinding( + await deps.stateObjectBinding(join(directory, 'pre-apply-state.tfstate')), + durableAttempt + ) + const invocationId = deps.randomUUID() + const invocationRequestReason = + `${durableAttempt.requestReason}/${invocationId}` + const startedResult = await deps.markApplyStarted(durableAttempt, { + invocationId, + requestReason: invocationRequestReason + }) + const startedAttempt = { + ...startedResult.attempt, + applyInvocations: [ + ...(durableAttempt.applyInvocations ?? []), + startedResult.invocation + ] + } + if (!Number.isSafeInteger(startedAttempt?.applyStartedAt)) { + throw new Error('durable fence attempt has no apply-start time') + } + let applyError + try { + deps.terraform( + [ + `-chdir=${config.terraformDir}`, + 'apply', + '-input=false', + `-lock-timeout=${config.lockTimeout}`, + planPath + ], + { + env: { + ...process.env, + GOOGLE_REQUEST_REASON: invocationRequestReason + } + } + ) + } catch (error) { + applyError = error + } + const progress = await deps.inspectProgress(expected, startedAttempt) + const classification = classifyTerraformFenceProgress(progress) + const observedAttempt = await persistInvocationOperations( + startedAttempt, + progress, + deps.markOperation + ) + if (classification !== 'complete') { + const reason = applyError ? 'apply response failed' : 'apply did not converge' + throw new Error(`${reason}; recover-forward required`) + } + if (!progress.gceOperation) throw new Error('completed fence has no GCE operation evidence') + if (completedStateBranch(progress, startedAttempt) !== 'complete') { + throw new Error('Terraform state did not persist the completed fence') + } + await deps.assertZeroDiff() + await deps.postApplyGuard(startedAttempt.cellIncarnation) + const completedAttempt = { + ...observedAttempt, + gceOperation: progress.gceOperation + } + await deps.attest(completedAttempt) + await deps.deletePlan(completedAttempt) + deps.emit({ + event: 'terraform_cell_fenced', + cellId: config.cell.cellId, + attemptId: startedAttempt.attemptId, + planSha256 + }) + return durableAttempt + } finally { + deps.remove(directory, { recursive: true, force: true }) + } +} + +export async function inspectTerraformFenceProgress(config, deps, attempt) { + const terraform = deps.terraform ?? defaultTerraform + const expected = exactMigExpected(config.cell) + const stateIdentity = terraformFenceStateIdentity(config, { terraform }) + const state = terraformJson({ terraform }, [ + `-chdir=${config.terraformDir}`, + 'show', + '-json' + ]) + const stateTargetSize = terraformFenceState(state, expected) + const live = deps.gcloudJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + config.cell.migName, + '--project', + config.project, + '--zone', + config.cell.zone, + '--format=json' + ]) + const instances = deps.gcloudJson([ + 'compute', + 'instance-groups', + 'managed', + 'list-instances', + config.cell.migName, + '--project', + config.project, + '--zone', + config.cell.zone, + '--format=json' + ]) + const operations = deps.gcloudJson([ + 'compute', + 'operations', + 'list', + '--project', + config.project, + `--filter=zone:(${config.cell.zone}) AND targetLink:${config.cell.migName}`, + '--sort-by=~insertTime', + '--limit=20', + '--format=json' + ]) + const operationCandidates = operations.filter((operation) => { + const insertedAt = Date.parse(operation.insertTime) + return ( + typeof operation.name === 'string' && + operation.targetLink === + `https://www.googleapis.com/compute/v1/projects/${config.project}/zones/${config.cell.zone}/instanceGroupManagers/${config.cell.migName}` && + operation.operationType === 'compute.instanceGroupManagers.resize' && + Number.isSafeInteger(attempt?.applyStartedAt) && + Number.isFinite(insertedAt) && + insertedAt >= attempt.applyStartedAt + ) + }) + const invocations = attempt?.applyInvocations ?? [] + if (attempt?.applyStartedAt && invocations.length === 0) { + throw new Error('Terraform fence apply has no durable invocation ledger') + } + const auditEntries = invocations.length > 0 + ? deps.gcloudJson([ + 'logging', + 'read', + `protoPayload.requestMetadata.requestAttributes.reason:"${attempt.requestReason}/" AND protoPayload.resourceName="projects/${config.project}/zones/${config.cell.zone}/instanceGroupManagers/${config.cell.migName}"`, + '--project', + config.project, + '--limit=20', + '--format=json' + ]) + : [] + const invocationOperations = invocations.map((invocation) => { + const matchingAudits = (Array.isArray(auditEntries) ? auditEntries : []).filter((entry) => { + const payload = entry.protoPayload ?? {} + return ( + payload.requestMetadata?.requestAttributes?.reason === invocation.requestReason && + payload.resourceName === + `projects/${config.project}/zones/${config.cell.zone}/instanceGroupManagers/${config.cell.migName}` && + String(payload.methodName ?? '').endsWith('instanceGroupManagers.resize') && + Number(payload.request?.size ?? payload.request?.targetSize) === 0 + ) + }) + if (matchingAudits.length > 1) { + throw new Error('Terraform fence invocation has ambiguous audit operations') + } + const responseName = String( + matchingAudits[0]?.protoPayload?.response?.name ?? '' + ) + const operationName = responseName.includes('/operations/') + ? responseName.slice(responseName.lastIndexOf('/') + 1) + : responseName + const expectedName = invocation.gceOperation ?? operationName + const operation = expectedName + ? operationCandidates.find((candidate) => candidate.name === expectedName) + : undefined + if ( + (invocation.gceOperation && operationName && invocation.gceOperation !== operationName) || + (expectedName && !operation) + ) { + throw new Error('Terraform fence invocation operation mismatch') + } + return { + ...invocation, + gceOperation: operation?.name, + operationStatus: operation?.status ?? 'ABSENT', + operationError: Boolean(operation?.error), + auditBound: Boolean(matchingAudits.length === 1 && operation) + } + }) + const boundOperations = invocationOperations.filter( + (invocation) => invocation.gceOperation + ) + const finalOperation = boundOperations.at(-1) + const operationStatus = invocationOperations.some((invocation) => + ['PENDING', 'RUNNING'].includes(invocation.operationStatus) + ) + ? 'RUNNING' + : (finalOperation?.operationStatus ?? 'ABSENT') + return { + stateTargetSize, + stateLineage: stateIdentity.lineage, + stateSerial: stateIdentity.serial, + liveTargetSize: Number(live.targetSize), + instanceCount: instances.length, + operationStatus, + operationError: invocationOperations.some( + (invocation) => invocation.operationError + ), + operationAuditBound: + boundOperations.length > 0 && + boundOperations.every((invocation) => invocation.auditBound), + gceOperation: finalOperation?.gceOperation, + invocationOperations + } +} + +function responseOperationName(entry) { + const name = String(entry?.protoPayload?.response?.name ?? '') + return name.includes('/operations/') + ? name.slice(name.lastIndexOf('/') + 1) + : name +} + +export async function inspectCompletedTerraformFenceProgress( + config, + deps, + attempt, + recovery +) { + if (!Number.isSafeInteger(attempt?.applyStartedAt)) { + throw new Error('completed fence recovery has no durable apply start') + } + const terraform = deps.terraform ?? defaultTerraform + const expected = exactMigExpected(config.cell) + const stateIdentity = terraformFenceStateIdentity(config, { terraform }) + const state = terraformJson({ terraform }, [ + `-chdir=${config.terraformDir}`, + 'show', + '-json' + ]) + const stateTargetSize = terraformFenceState(state, expected) + const live = deps.gcloudJson([ + 'compute', + 'instance-groups', + 'managed', + 'describe', + config.cell.migName, + '--project', + config.project, + '--zone', + config.cell.zone, + '--format=json' + ]) + const instances = deps.gcloudJson([ + 'compute', + 'instance-groups', + 'managed', + 'list-instances', + config.cell.migName, + '--project', + config.project, + '--zone', + config.cell.zone, + '--format=json' + ]) + const targetLink = + `https://www.googleapis.com/compute/v1/projects/${config.project}/zones/${config.cell.zone}/instanceGroupManagers/${config.cell.migName}` + const operations = deps.gcloudJson([ + 'compute', + 'operations', + 'list', + '--project', + config.project, + `--filter=zone:(${config.cell.zone}) AND targetLink:${config.cell.migName}`, + '--sort-by=~insertTime', + '--limit=20', + '--format=json' + ]) + const operationCandidates = operations.filter((operation) => { + const insertedAt = Date.parse(operation.insertTime) + return ( + typeof operation.name === 'string' && + operation.targetLink === targetLink && + operation.operationType === 'compute.instanceGroupManagers.resize' && + Number.isFinite(insertedAt) && + insertedAt >= attempt.applyStartedAt + ) + }) + if ( + operationCandidates.length !== 1 || + operationCandidates[0].name !== recovery.gceOperation + ) { + throw new Error('completed fence recovery has no unique Compute operation') + } + const resourceName = + `projects/${config.project}/zones/${config.cell.zone}/instanceGroupManagers/${config.cell.migName}` + const auditEntries = deps.gcloudJson([ + 'logging', + 'read', + `protoPayload.authenticationInfo.principalEmail="${recovery.principalEmail}" AND protoPayload.resourceName="${resourceName}" AND protoPayload.methodName:"instanceGroupManagers.resize" AND timestamp>="${new Date(attempt.applyStartedAt).toISOString()}"`, + '--project', + config.project, + '--limit=20', + '--format=json' + ]) + const matchingAudits = (Array.isArray(auditEntries) ? auditEntries : []).filter( + (entry) => { + const payload = entry.protoPayload ?? {} + const timestamp = Date.parse(entry.timestamp) + return ( + payload.authenticationInfo?.principalEmail === recovery.principalEmail && + payload.resourceName === resourceName && + String(payload.methodName ?? '').endsWith('instanceGroupManagers.resize') && + Number(payload.request?.size ?? payload.request?.targetSize) === 0 && + Number.isFinite(timestamp) && + timestamp >= attempt.applyStartedAt && + responseOperationName(entry) === recovery.gceOperation + ) + } + ) + if (matchingAudits.length !== 1) { + throw new Error('completed fence recovery has no unique Audit Log operation') + } + const [operation] = operationCandidates + return { + stateTargetSize, + stateLineage: stateIdentity.lineage, + stateSerial: stateIdentity.serial, + liveTargetSize: Number(live.targetSize), + instanceCount: instances.length, + liveStable: live.status?.isStable === true, + operationStatus: operation.status, + operationError: Boolean(operation.error), + gceOperation: operation.name + } +} + +export function assertTerraformFenceSet(config, deps = {}) { + const terraform = deps.terraform ?? defaultTerraform + const result = terraform( + [ + `-chdir=${config.terraformDir}`, + 'console', + `-var-file=${config.varFile}` + ], + { + encoding: 'utf8', + input: + `contains(var.relay_gce_fenced_cells, ${JSON.stringify(config.cell.cellId)}) && ` + + `try(local.relay_gce_cell_target_sizes[${JSON.stringify(config.cell.cellId)}], -1) == 0\n` + } + ) + if (result.trim() !== 'true') { + throw new Error('requested cell is not in the committed Terraform fence set') + } +} + +export async function recoverSupersededCompletedTerraformFence( + config, + deps, + recovery +) { + assertFenceIdentity(config.cell) + const varFileSha256 = assertReviewedFenceCheckout(config, deps) + await deps.assertCommittedFenceSet() + const attempt = await deps.loadAttempt(config.cell.cellId) + if ( + !attempt || + attempt.fenceCommit === config.fenceCommit || + attempt.attemptId !== recovery.attemptId || + attempt.fenceCommit !== recovery.fenceCommit || + attempt.terraformStateSerial !== recovery.terraformStateSerial || + attempt.planObjectGeneration !== recovery.planObjectGeneration + ) { + throw new Error('completed fence recovery does not match the pinned attempt') + } + assertAttemptMatches({ ...config, fenceCommit: attempt.fenceCommit }, attempt) + if ( + varFileSha256 !== attempt.varFileSha256 || + !Number.isSafeInteger(attempt.applyStartedAt) || + attempt.abortedAt || + (attempt.gceOperation && attempt.gceOperation !== recovery.gceOperation) + ) { + throw new Error('completed fence recovery attempt is not adoptable') + } + const invocations = attempt.applyInvocations ?? [] + if ( + invocations.length !== 1 || + (invocations[0].gceOperation && + invocations[0].gceOperation !== recovery.gceOperation) || + invocations[0].startedAt < attempt.applyStartedAt + ) { + throw new Error('completed fence recovery invocation ledger is unsafe') + } + const resolved = await deps.resolvePlan(attempt) + const planExists = resolved.generation === attempt.planObjectGeneration + if (!planExists && !(attempt.completedAt && resolved.generation === null)) { + throw new Error('completed fence recovery saved-plan generation changed') + } + const directory = privatePlanDirectory({ + mkdtemp: deps.mkdtemp ?? mkdtempSync, + chmod: deps.chmod ?? chmodSync, + tmpdir: deps.tmpdir ?? tmpdir + }) + try { + const stateBinding = await deps.stateObjectBinding( + join(directory, 'completed-state.tfstate') + ) + if ( + stateBinding.generation !== recovery.terraformStateObjectGeneration || + stateBinding.sha256 !== recovery.terraformStateObjectSha256 || + stateBinding.lineage !== attempt.terraformStateLineage || + stateBinding.serial !== attempt.terraformStateSerial + 1 + ) { + throw new Error('completed fence recovery state object changed') + } + if (planExists) { + const planPath = join(directory, 'fence.tfplan') + await deps.downloadPlan(attempt, planPath) + if ((deps.stat ?? statSync)(planPath).mode & 0o077) { + throw new Error('downloaded saved fence plan permissions are not private') + } + if (sha256(planPath, deps.readFile ?? readFileSync) !== attempt.planSha256) { + throw new Error('completed fence recovery saved-plan digest changed') + } + validateTerraformFencePlan( + terraformJson( + { terraform: deps.terraform ?? defaultTerraform }, + [`-chdir=${config.terraformDir}`, 'show', '-json', planPath] + ), + exactMigExpected(config.cell) + ) + } + const progress = await deps.inspectCompletedProgress( + exactMigExpected(config.cell), + attempt, + recovery + ) + if ( + progress.stateLineage !== attempt.terraformStateLineage || + progress.stateSerial !== attempt.terraformStateSerial + 1 || + progress.stateTargetSize !== 0 || + progress.liveTargetSize !== 0 || + progress.instanceCount !== 0 || + progress.liveStable !== true || + progress.operationStatus !== 'DONE' || + progress.operationError || + progress.gceOperation !== recovery.gceOperation + ) { + throw new Error('completed fence recovery production evidence is unsafe') + } + const invocation = { + ...invocations[0], + gceOperation: recovery.gceOperation + } + const marked = attempt.gceOperation + ? { attempt, invocation } + : await deps.markOperation( + { ...attempt, gceOperation: recovery.gceOperation }, + invocation + ) + await deps.assertZeroDiff() + await deps.postApplyGuard(attempt.cellIncarnation) + await deps.attest({ + ...marked.attempt, + applyInvocations: [marked.invocation], + gceOperation: recovery.gceOperation + }) + if (planExists) await deps.deletePlan(attempt) + deps.emit?.({ + event: 'terraform_cell_fence_completed_attempt_recovered', + cellId: config.cell.cellId, + attemptId: attempt.attemptId, + gceOperation: recovery.gceOperation + }) + } finally { + ;(deps.remove ?? rmSync)(directory, { recursive: true, force: true }) + } +} + +export async function resumeTerraformFence(config, deps) { + assertFenceIdentity(config.cell) + assertReviewedFenceCheckout(config, deps) + await deps.assertCommittedFenceSet() + let attempt = await deps.loadAttempt(config.cell.cellId) + assertAttemptMatches(config, attempt, false) + if (!attempt.planObjectGeneration) { + const resolved = await deps.resolvePlan(attempt) + if (!resolved.generation) throw new Error('durable fence plan object is missing') + const bound = await deps.bindPlan({ + ...attempt, + planObjectGeneration: resolved.generation + }) + attempt = bound?.attempt ?? bound + } + assertAttemptMatches(config, attempt) + assertAttemptCheckoutBinding(config, attempt, deps) + let progress = await deps.inspectProgress(exactMigExpected(config.cell), attempt) + let classification = classifyTerraformFenceProgress(progress) + if (classification === 'complete') { + if (completedStateBranch(progress, attempt) === 'replay') { + classification = 'reconcile-state' + } else { + await deps.assertZeroDiff() + } + } + if (classification !== 'complete') { + if (classification === 'in-progress' && Number.isSafeInteger(attempt.applyStartedAt)) { + throw new Error('Terraform fence operation is still running; recover-forward required') + } + attempt = await persistInvocationOperations( + attempt, + progress, + deps.markOperation + ) + assertReplayStateIdentity(progress, attempt) + await deps.preApplyGuard() + const directory = privatePlanDirectory({ + mkdtemp: deps.mkdtemp ?? mkdtempSync, + chmod: deps.chmod ?? chmodSync, + tmpdir: deps.tmpdir ?? tmpdir + }) + const planPath = join(directory, 'fence.tfplan') + try { + assertStateObjectBinding( + await deps.stateObjectBinding( + join(directory, 'replay-pre-state.tfstate') + ), + attempt + ) + await deps.downloadPlan(attempt, planPath) + const stat = (deps.stat ?? statSync)(planPath) + if ((stat.mode & 0o077) !== 0) { + throw new Error('downloaded saved fence plan permissions are not private') + } + if (sha256(planPath, deps.readFile ?? readFileSync) !== attempt.planSha256) { + throw new Error('downloaded saved fence plan digest mismatch') + } + validateTerraformFencePlan( + terraformJson( + { terraform: deps.terraform ?? defaultTerraform }, + [`-chdir=${config.terraformDir}`, 'show', '-json', planPath] + ), + exactMigExpected(config.cell) + ) + const invocationId = (deps.randomUUID ?? randomUUID)() + const invocationRequestReason = `${attempt.requestReason}/${invocationId}` + const started = await deps.markApplyStarted(attempt, { + invocationId, + requestReason: invocationRequestReason + }) + attempt = { + ...started.attempt, + applyInvocations: [ + ...(attempt.applyInvocations ?? []), + started.invocation + ] + } + let applyError + try { + ;(deps.terraform ?? defaultTerraform)( + [ + `-chdir=${config.terraformDir}`, + 'apply', + '-input=false', + `-lock-timeout=${config.lockTimeout}`, + planPath + ], + { + env: { + ...process.env, + GOOGLE_REQUEST_REASON: invocationRequestReason + } + } + ) + } catch (error) { + applyError = error + } + progress = await deps.inspectProgress(exactMigExpected(config.cell), attempt) + classification = classifyTerraformFenceProgress(progress) + attempt = await persistInvocationOperations( + attempt, + progress, + deps.markOperation + ) + if (classification !== 'complete') { + const reason = applyError ? 'saved-plan replay response failed' : 'saved-plan replay did not converge' + throw new Error(`${reason}; recover-forward required`) + } + if (completedStateBranch(progress, attempt) !== 'complete') { + throw new Error('Terraform state did not persist the replayed fence') + } + await deps.assertZeroDiff() + } finally { + ;(deps.remove ?? rmSync)(directory, { recursive: true, force: true }) + } + } + const gceOperation = progress.gceOperation ?? attempt.gceOperation + if (!gceOperation) throw new Error('completed fence has no GCE operation evidence') + await deps.postApplyGuard(attempt.cellIncarnation) + await deps.attest({ ...attempt, gceOperation }) + await deps.deletePlan(attempt) + deps.emit?.({ + event: 'terraform_cell_fence_resumed', + cellId: config.cell.cellId, + attemptId: attempt.attemptId + }) +} + +export async function abortTerraformFenceBeforeApply(config, deps) { + assertFenceIdentity(config.cell) + assertReviewedFenceCheckout(config, deps) + await deps.assertCommittedFenceSet() + let attempt = await deps.loadAttempt(config.cell.cellId) + assertAttemptMatches(config, attempt, false) + if (!attempt.planObjectGeneration) { + const resolved = await deps.resolvePlan(attempt) + if (resolved.generation) { + const bound = await deps.bindPlan({ + ...attempt, + planObjectGeneration: resolved.generation + }) + attempt = bound?.attempt ?? bound + } + } + assertAttemptMatches(config, attempt, Boolean(attempt.planObjectGeneration)) + assertAttemptCheckoutBinding(config, attempt, deps) + const progress = await deps.inspectProgress(exactMigExpected(config.cell), attempt) + if (classifyTerraformFenceProgress(progress) !== 'not-started') { + throw new Error('cannot abort after Terraform fence apply may have started') + } + if (attempt.applyStartedAt) throw new Error('cannot abort after Terraform fence apply was marked') + await deps.abortAttempt(attempt) + if (attempt.planObjectGeneration) await deps.deletePlan(attempt) + deps.emit?.({ event: 'terraform_fence_aborted_before_apply', cellId: config.cell.cellId }) +} + +export async function abortSupersededTerraformFenceBeforeUpload(config, deps) { + assertFenceIdentity(config.cell) + const varFileSha256 = assertReviewedFenceCheckout(config, deps) + await deps.assertCommittedFenceSet() + const attempt = await deps.loadAttempt(config.cell.cellId) + if ( + !attempt || + !/^[a-f0-9]{40}$/.test(attempt.fenceCommit ?? '') || + attempt.fenceCommit === config.fenceCommit + ) { + throw new Error('prepared Terraform fence attempt is not from a superseded commit') + } + assertAttemptMatches({ ...config, fenceCommit: attempt.fenceCommit }, attempt, false) + if (attempt.varFileSha256 !== varFileSha256) { + throw new Error('superseded Terraform fence variable-file digest changed') + } + if ( + attempt.planObjectGeneration || + attempt.applyStartedAt || + attempt.completedAt || + attempt.gceOperation || + (attempt.applyInvocations?.length ?? 0) > 0 + ) { + throw new Error('cannot supersede a Terraform fence attempt after plan upload') + } + const resolved = await deps.resolvePlan(attempt) + if (resolved?.generation !== null) { + throw new Error('cannot supersede a Terraform fence attempt with a saved plan') + } + const progress = await deps.inspectProgress(exactMigExpected(config.cell), attempt) + if (classifyTerraformFenceProgress(progress) !== 'not-started') { + throw new Error('cannot supersede after Terraform fence apply may have started') + } + await deps.abortAttempt(attempt) + deps.emit?.({ + event: 'terraform_fence_superseded_before_upload', + cellId: config.cell.cellId, + previousFenceCommit: attempt.fenceCommit, + fenceCommit: config.fenceCommit + }) +} diff --git a/cloud/dev/scripts/relay-gce-terraform-fence.test.mjs b/cloud/dev/scripts/relay-gce-terraform-fence.test.mjs new file mode 100644 index 00000000000..306e28c887b --- /dev/null +++ b/cloud/dev/scripts/relay-gce-terraform-fence.test.mjs @@ -0,0 +1,1522 @@ +import assert from 'node:assert/strict' +import { createHash } from 'node:crypto' +import { + chmodSync, + existsSync, + mkdtempSync, + readFileSync, + rmSync, + writeFileSync +} from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { test } from 'node:test' +import { + abortSupersededTerraformFenceBeforeUpload, + abortTerraformFenceBeforeApply, + adoptLegacyTerraformFence, + assertReviewedFenceCheckout, + assertTerraformFenceSet, + assertTerraformFenceStateFenced, + assertTerraformFenceZeroDiff, + classifyTerraformFenceProgress, + deleteTerraformFencePlan, + inspectCompletedTerraformFenceProgress, + inspectTerraformFenceProgress, + recoverSupersededCompletedTerraformFence, + runTerraformFenceApply, + resumeTerraformFence, + terraformFenceState, + terraformProcessStdio, + validateTerraformFenceCompletionPlan, + validateTerraformFencePlan +} from './relay-gce-terraform-fence.mjs' + +const cell = { + cellId: 'production-gce-c1', + migName: 'orca-relay-c1', + zone: 'us-central1-a', + instanceGroup: 'https://compute.example/instanceGroups/orca-relay-c1', + generationIdentity: 'https://compute.example/instanceTemplates/orca-relay-c1-abc', + fenced: true, + desiredTargetSize: 0 +} + +const expected = { + cellId: cell.cellId, + name: cell.migName, + zone: cell.zone, + instance_group: cell.instanceGroup, + generationIdentity: cell.generationIdentity +} +const stateLineage = 'c739dab4-e6e1-e627-02a9-504b3dda1a2c' + +test('adopts a legacy fence only after repeated no-op and live guards', async () => { + const events = [] + const calls = [] + await adoptLegacyTerraformFence( + { + environment: 'production', + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars', + fenceCommit: 'a'.repeat(40), + cellIncarnation: '11111111-1111-4111-8111-111111111111', + cell + }, + { + environment: { ORCA_RELAY_FENCE_IMAGE_COMMIT: 'a'.repeat(40) }, + readFile: () => Buffer.from('reviewed variables'), + loadAttempt: async () => { + calls.push('attempt') + return null + }, + assertCommittedFenceSet: async () => calls.push('fence-set'), + preApplyGuard: async () => calls.push('pre'), + assertStateFenced: async () => calls.push('state-fenced'), + postApplyGuard: async () => calls.push('post'), + attest: async () => calls.push('attest'), + commitAdoption: async () => calls.push('commit'), + emit: (event) => events.push(event) + } + ) + assert.deepEqual(calls, [ + 'attempt', + 'fence-set', + 'pre', + 'state-fenced', + 'post', + 'attempt', + 'attest', + 'post', + 'commit' + ]) + assert.deepEqual(events, [ + { event: 'terraform_cell_fence_legacy_adopted', cellId: cell.cellId } + ]) +}) + +test('refuses legacy adoption when a durable attempt appears during proof', async () => { + let reads = 0 + let attested = false + await assert.rejects( + adoptLegacyTerraformFence( + { + environment: 'production', + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars', + fenceCommit: 'a'.repeat(40), + cellIncarnation: '11111111-1111-4111-8111-111111111111', + cell + }, + { + environment: { ORCA_RELAY_FENCE_IMAGE_COMMIT: 'a'.repeat(40) }, + readFile: () => Buffer.from('reviewed variables'), + loadAttempt: async () => (++reads === 1 ? null : { attemptId: 'new' }), + assertCommittedFenceSet: async () => {}, + preApplyGuard: async () => {}, + assertStateFenced: async () => {}, + postApplyGuard: async () => {}, + attest: async () => { + attested = true + }, + commitAdoption: async () => {} + } + ), + /durable fence attempt appeared/ + ) + assert.equal(attested, false) +}) + +test('does not commit legacy adoption when the final guard fails', async () => { + let postGuards = 0 + let committed = false + await assert.rejects( + adoptLegacyTerraformFence( + { + environment: 'production', + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars', + fenceCommit: 'a'.repeat(40), + cellIncarnation: '11111111-1111-4111-8111-111111111111', + cell + }, + { + environment: { ORCA_RELAY_FENCE_IMAGE_COMMIT: 'a'.repeat(40) }, + readFile: () => Buffer.from('reviewed variables'), + loadAttempt: async () => null, + assertCommittedFenceSet: async () => {}, + preApplyGuard: async () => {}, + assertStateFenced: async () => {}, + postApplyGuard: async () => { + postGuards++ + if (postGuards === 2) throw new Error('final guard failed') + }, + attest: async () => {}, + commitAdoption: async () => { + committed = true + } + } + ), + /final guard failed/ + ) + assert.equal(committed, false) +}) + +test('pipes reviewed Terraform console expressions to stdin', () => { + assert.deepEqual( + terraformProcessStdio({ encoding: 'utf8', input: 'contains(...)\n' }), + ['pipe', 'pipe', 'pipe'] + ) + assert.deepEqual(terraformProcessStdio({ encoding: 'utf8' }), [ + 'ignore', + 'pipe', + 'pipe' + ]) + assert.equal(terraformProcessStdio(), 'inherit') +}) + +test('checks the exact committed fence cell through Terraform console', () => { + let invocation + assertTerraformFenceSet( + { + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars', + cell + }, + { + terraform: (args, options) => { + invocation = { args, options } + return 'true\n' + } + } + ) + assert.deepEqual(invocation.args, [ + '-chdir=infra/terraform', + 'console', + '-var-file=environments/production.tfvars' + ]) + assert.equal( + invocation.options.input, + 'contains(var.relay_gce_fenced_cells, "production-gce-c1") && ' + + 'try(local.relay_gce_cell_target_sizes["production-gce-c1"], -1) == 0\n' + ) +}) + +test('binds a gitless broker checkout to its immutable image commit', () => { + const config = { + fenceCommit: 'a'.repeat(40), + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars' + } + const contents = Buffer.from('reviewed production variables') + const digest = assertReviewedFenceCheckout(config, { + environment: { ORCA_RELAY_FENCE_IMAGE_COMMIT: config.fenceCommit }, + readFile: () => contents, + git: () => { + throw new Error('git must not run inside the immutable broker image') + } + }) + assert.equal(digest, createHash('sha256').update(contents).digest('hex')) +}) + +test('rejects a broker image built for a different fence commit', () => { + assert.throws( + () => + assertReviewedFenceCheckout( + { + fenceCommit: 'a'.repeat(40), + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars' + }, + { + environment: { ORCA_RELAY_FENCE_IMAGE_COMMIT: 'b'.repeat(40) }, + readFile: () => Buffer.from('reviewed production variables') + } + ), + /immutable broker image/ + ) +}) + +function plan(actions = ['update'], address = undefined) { + return { + resource_changes: [ + { + address: + address ?? + `google_compute_instance_group_manager.relay_gce_cell["${cell.cellId}"]`, + change: { + actions, + before: { + name: cell.migName, + zone: cell.zone, + instance_group: cell.instanceGroup, + target_size: 1, + version: [{ instance_template: cell.generationIdentity }] + }, + after: { + name: cell.migName, + zone: cell.zone, + instance_group: cell.instanceGroup, + target_size: 0, + version: [{ instance_template: cell.generationIdentity }] + } + } + } + ] + } +} + +function state(targetSize = 0) { + return { + values: { + root_module: { + resources: [ + { + address: + `google_compute_instance_group_manager.relay_gce_cell["${cell.cellId}"]`, + values: { + name: cell.migName, + zone: cell.zone, + instance_group: cell.instanceGroup, + target_size: targetSize, + version: [{ instance_template: cell.generationIdentity }] + } + } + ] + } + } + } +} + +test('accepts only one exact in-place MIG resize from one to zero', () => { + assert.equal(validateTerraformFencePlan(plan(), expected).change.after.target_size, 0) + for (const actions of [['create'], ['delete'], ['delete', 'create']]) { + assert.throws(() => validateTerraformFencePlan(plan(actions), expected)) + } + assert.throws(() => + validateTerraformFencePlan( + plan(['update'], 'google_compute_backend_service.relay_gce_cell["production-gce-c1"]'), + expected + ) + ) + const unrelated = plan() + unrelated.resource_changes.push({ + address: 'google_compute_url_map.relay_gce[0]', + change: { actions: ['update'], before: {}, after: {} } + }) + assert.throws(() => validateTerraformFencePlan(unrelated, expected)) +}) + +function completionPlan() { + const result = plan() + const before = result.resource_changes[0].change.before + const after = result.resource_changes[0].change.after + before.target_size = 0 + before.version[0].name = '0/2026-07-31 03:52:56.639922+00:00' + after.version[0].name = 'primary' + return result +} + +test('accepts only the empty MIG provider version-label normalization', () => { + assert.equal(validateTerraformFenceCompletionPlan({ resource_changes: [] }, expected), undefined) + assert.equal( + validateTerraformFenceCompletionPlan(completionPlan(), expected).change.after.version[0] + .name, + 'primary' + ) + + const resized = completionPlan() + resized.resource_changes[0].change.after.target_size = 1 + assert.throws(() => validateTerraformFenceCompletionPlan(resized, expected)) + + const replaced = completionPlan() + replaced.resource_changes[0].change.after.version[0].instance_template += '-other' + assert.throws(() => validateTerraformFenceCompletionPlan(replaced, expected)) + + const extraChange = completionPlan() + extraChange.resource_changes[0].change.after.update_policy = { type: 'PROACTIVE' } + assert.throws(() => validateTerraformFenceCompletionPlan(extraChange, expected)) + + const arbitraryLabel = completionPlan() + arbitraryLabel.resource_changes[0].change.before.version[0].name = 'other' + assert.throws(() => validateTerraformFenceCompletionPlan(arbitraryLabel, expected)) +}) + +test('binds Terraform state to the exact MIG generation', () => { + assert.equal(terraformFenceState(state(0), expected), 0) + const replaced = state(0) + replaced.values.root_module.resources[0].values.version[0].instance_template += '-other' + assert.throws(() => terraformFenceState(replaced, expected)) +}) + +test('requires Terraform state to record the exact completed fence', () => { + let invocation + assertTerraformFenceStateFenced(applyConfig(), { + terraform: (args) => { + invocation = args + return JSON.stringify(state(0)) + } + }) + assert.deepEqual(invocation, ['-chdir=infra/terraform', 'show', '-json']) + + assert.throws( + () => + assertTerraformFenceStateFenced(applyConfig(), { + terraform: () => JSON.stringify(state(1)) + }), + /does not record the requested cell fence/ + ) + const replaced = state(0) + replaced.values.root_module.resources[0].values.version[0].instance_template += '-other' + assert.throws(() => + assertTerraformFenceStateFenced(applyConfig(), { + terraform: () => JSON.stringify(replaced) + }) + ) +}) + +test('builds and verifies fence plans without unrelated live refreshes', () => { + let zeroDiffArgs + assertTerraformFenceZeroDiff(applyConfig(), { + terraform: (args) => { + if (args.includes('plan')) zeroDiffArgs = args + if (args.includes('show')) return JSON.stringify({ resource_changes: [] }) + } + }) + assert.equal(zeroDiffArgs.includes('-refresh=false'), true) + assert.equal( + zeroDiffArgs.includes( + '-target=google_compute_instance_group_manager.relay_gce_cell["production-gce-c1"]' + ), + true + ) +}) + +test('classifies complete, in-progress, and conclusively not-started fences', () => { + assert.equal( + classifyTerraformFenceProgress({ + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationAuditBound: true + }), + 'complete' + ) + assert.equal( + classifyTerraformFenceProgress({ + stateTargetSize: 1, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'RUNNING' + }), + 'in-progress' + ) + assert.equal( + classifyTerraformFenceProgress({ + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT' + }), + 'not-started' + ) + assert.equal( + classifyTerraformFenceProgress({ + stateTargetSize: 1, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationAuditBound: true + }), + 'reconcile-state' + ) + assert.throws( + () => + classifyTerraformFenceProgress({ + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationAuditBound: false + }), + /audit binding/ + ) +}) + +function applyHarness({ + loseApplyResponse = false, + guardError = null, + postGuardError = null, + tamperPlan = false, + progress +} = {}) { + const root = mkdtempSync(join(tmpdir(), 'relay-fence-test-')) + const calls = [] + const applyEnvironments = [] + const events = [] + let planPath + let progressReads = 0 + const terraform = (args, options = {}) => { + calls.push(args) + if (args.includes('state') && args.includes('pull')) { + return JSON.stringify({ lineage: stateLineage, serial: 7 }) + } + if (args.includes('plan')) { + planPath = args.find((arg) => arg.startsWith('-out=')).slice(5) + writeFileSync(planPath, 'private saved plan', { mode: 0o644 }) + chmodSync(planPath, 0o644) + return + } + if (args.includes('show')) return JSON.stringify(plan()) + if (args.includes('apply')) { + applyEnvironments.push(options.env) + if (loseApplyResponse) throw new Error('lost response') + } + } + const evidence = [] + return { + root, + calls, + events, + evidence, + applyEnvironments, + planPath: () => planPath, + overrides: { + terraform, + git: (args) => { + if (args.includes('rev-parse')) return `${applyConfig().fenceCommit}\n` + return '' + }, + assertCommittedFenceSet: async () => {}, + tmpdir: () => root, + randomUUID: () => '11111111-1111-4111-8111-111111111111', + inspectProgress: async () => { + if (progressReads++ === 0) { + return { + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT', + operationError: false, + operationAuditBound: false, + stateLineage, + stateSerial: 7, + invocationOperations: [] + } + } + return progress ?? { + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 8, + gceOperation: 'operation-1', + invocationOperations: [ + { + invocationId: '11111111-1111-4111-8111-111111111111', + requestReason: + 'orca-relay-fence/11111111-1111-4111-8111-111111111111/11111111-1111-4111-8111-111111111111', + startedAt: 101, + gceOperation: 'operation-1', + operationStatus: 'DONE', + operationError: false, + auditBound: true + } + ] + } + }, + uploadPlan: async () => ({ generation: '123456789' }), + stateObjectBinding: async () => ({ + generation: '987654321', + sha256: createHash('sha256').update('pre-state object').digest('hex'), + lineage: stateLineage, + serial: 7 + }), + bindPlan: async (value) => { + evidence.push(['bind', value]) + return { attempt: value } + }, + deletePlan: async (value) => evidence.push(['delete', value]), + assertZeroDiff: async () => {}, + prepareAttempt: async (value) => { + evidence.push(['prepare', value]) + return { attempt: { ...value, createdAt: 100, expiresAt: 3_600_100 } } + }, + markApplyStarted: async (value, invocation) => { + const started = { ...value, applyStartedAt: 101 } + const durableInvocation = { ...invocation, startedAt: 101 } + evidence.push(['started', started]) + return { attempt: started, invocation: durableInvocation } + }, + markOperation: async (value, invocation) => { + const durableInvocation = { + ...invocation, + gceOperation: value.gceOperation + } + evidence.push(['operation', value]) + return { attempt: value, invocation: durableInvocation } + }, + attest: async (value) => evidence.push(['attest', value]), + preApplyGuard: async () => { + if (guardError) throw guardError + if (tamperPlan) writeFileSync(planPath, 'tampered plan', { mode: 0o600 }) + }, + postApplyGuard: async () => { + if (postGuardError) throw postGuardError + }, + emit: (event) => events.push(event) + }, + cleanup: () => rmSync(root, { recursive: true, force: true }) + } +} + +function applyConfig() { + return { + project: 'project', + environment: 'production', + terraformDir: 'infra/terraform', + varFile: 'environments/production.tfvars', + lockTimeout: '5m', + fenceCommit: 'a'.repeat(40), + cellIncarnation: '22222222-2222-4222-8222-222222222222', + cell + } +} + +function durableAttempt(config = applyConfig(), overrides = {}) { + const attemptId = '11111111-1111-4111-8111-111111111111' + const varFile = join(config.terraformDir, config.varFile) + const result = { + attemptId, + environment: config.environment, + cellId: cell.cellId, + cellIncarnation: config.cellIncarnation, + migName: cell.migName, + instanceGroup: cell.instanceGroup, + generationIdentity: cell.generationIdentity, + fenceCommit: config.fenceCommit, + planSha256: createHash('sha256').update('private saved plan').digest('hex'), + planObjectName: `terraform/state/relay-fence-plans/${config.environment}/${attemptId}.tfplan`, + planObjectGeneration: '123456789', + varFileSha256: createHash('sha256').update(readFileSync(varFile)).digest('hex'), + terraformStateLineage: stateLineage, + terraformStateSerial: 7, + terraformStateObjectGeneration: '987654321', + terraformStateObjectSha256: createHash('sha256') + .update('pre-state object') + .digest('hex'), + requestReason: `orca-relay-fence/${attemptId}`, + createdAt: 100, + expiresAt: 3_600_100, + ...overrides + } + if (result.applyStartedAt && result.applyInvocations === undefined) { + const invocationId = '66666666-6666-4666-8666-666666666666' + result.applyInvocations = [ + { + invocationId, + requestReason: `${result.requestReason}/${invocationId}`, + startedAt: result.applyStartedAt, + gceOperation: result.gceOperation + } + ] + } + return result +} + +test('applies and attests the exact private saved plan', async () => { + const harness = applyHarness() + try { + await runTerraformFenceApply(applyConfig(), harness.overrides) + assert.deepEqual( + harness.evidence.map(([event]) => event), + ['prepare', 'bind', 'started', 'operation', 'attest', 'delete'] + ) + assert.equal(harness.calls.filter((args) => args.includes('apply')).length, 1) + const planArgs = harness.calls.find((args) => args.includes('plan')) + assert.equal(planArgs.includes('-refresh=false'), true) + assert.equal( + harness.applyEnvironments[0].GOOGLE_REQUEST_REASON, + 'orca-relay-fence/11111111-1111-4111-8111-111111111111/11111111-1111-4111-8111-111111111111' + ) + assert.equal(harness.events[0].event, 'terraform_cell_fenced') + assert.equal(existsSync(harness.planPath()), false) + } finally { + harness.cleanup() + } +}) + +test('rejects a malformed Terraform lineage before plan upload', async () => { + const harness = applyHarness() + const prepareAttempt = harness.overrides.prepareAttempt + harness.overrides.prepareAttempt = async (value) => { + const prepared = await prepareAttempt(value) + return { + attempt: { + ...prepared.attempt, + terraformStateLineage: 'not-a-terraform-lineage' + } + } + } + try { + await assert.rejects( + runTerraformFenceApply(applyConfig(), harness.overrides), + /valid Terraform state identity/ + ) + assert.deepEqual( + harness.evidence.map(([event]) => event), + ['prepare'] + ) + assert.equal(harness.calls.filter((args) => args.includes('apply')).length, 0) + } finally { + harness.cleanup() + } +}) + +test('recovers a successful fence after losing the apply response', async () => { + const harness = applyHarness({ loseApplyResponse: true }) + try { + await runTerraformFenceApply(applyConfig(), harness.overrides) + assert.equal(harness.evidence.some(([event]) => event === 'attest'), true) + } finally { + harness.cleanup() + } +}) + +test('does not start apply after a final pre-apply guard failure', async () => { + const harness = applyHarness({ guardError: new Error('guard failed') }) + try { + await assert.rejects( + runTerraformFenceApply(applyConfig(), harness.overrides), + /guard failed/ + ) + assert.equal(harness.calls.some((args) => args.includes('apply')), false) + assert.deepEqual(harness.evidence.map(([event]) => event), ['prepare', 'bind']) + } finally { + harness.cleanup() + } +}) + +test('rejects a saved plan whose digest changes before apply', async () => { + const harness = applyHarness({ tamperPlan: true }) + try { + await assert.rejects( + runTerraformFenceApply(applyConfig(), harness.overrides), + /digest changed/ + ) + assert.equal(harness.calls.some((args) => args.includes('apply')), false) + } finally { + harness.cleanup() + } +}) + +test('requires recover-forward when an apply remains in progress', async () => { + const harness = applyHarness({ + loseApplyResponse: true, + progress: { + stateTargetSize: 1, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'RUNNING', + operationError: false, + stateLineage, + stateSerial: 7, + gceOperation: 'operation-1' + } + }) + try { + await assert.rejects( + runTerraformFenceApply(applyConfig(), harness.overrides), + /recover-forward required/ + ) + assert.notEqual(harness.evidence.at(-1)[0], 'attest') + } finally { + harness.cleanup() + } +}) + +test('does not attest before retained topology and heartbeat guards pass', async () => { + const harness = applyHarness({ postGuardError: new Error('topology mismatch') }) + try { + await assert.rejects( + runTerraformFenceApply(applyConfig(), harness.overrides), + /topology mismatch/ + ) + assert.notEqual(harness.evidence.at(-1)[0], 'attest') + } finally { + harness.cleanup() + } +}) + +test('aborts only when state and live GCE prove apply never began', async () => { + let aborted = false + let deleted = false + const config = applyConfig() + const attempt = durableAttempt(config) + const common = { + git: (args) => (args.includes('rev-parse') ? `${config.fenceCommit}\n` : ''), + loadAttempt: async () => attempt, + deletePlan: async () => { + deleted = true + } + } + await abortTerraformFenceBeforeApply(config, { + ...common, + assertCommittedFenceSet: async () => {}, + inspectProgress: async () => ({ + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT', + stateLineage, + stateSerial: 7 + }), + abortAttempt: async () => { + aborted = true + } + }) + assert.equal(aborted, true) + assert.equal(deleted, true) + await assert.rejects( + abortTerraformFenceBeforeApply(config, { + ...common, + assertCommittedFenceSet: async () => {}, + inspectProgress: async () => ({ + stateTargetSize: 1, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'RUNNING', + stateLineage, + stateSerial: 7 + }), + abortAttempt: async () => {} + }), + /cannot abort/ + ) +}) + +test('supersedes only an older unuploaded fence attempt proven not started', async () => { + const config = applyConfig() + const attempt = durableAttempt(config, { + fenceCommit: 'b'.repeat(40), + planObjectGeneration: undefined + }) + const events = [] + let aborted = false + const common = { + git: (args) => (args.includes('rev-parse') ? `${config.fenceCommit}\n` : ''), + assertCommittedFenceSet: async () => {}, + loadAttempt: async () => attempt, + resolvePlan: async () => ({ generation: null }), + inspectProgress: async () => ({ + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT', + stateLineage, + stateSerial: 7 + }), + abortAttempt: async () => { + aborted = true + }, + emit: (event) => events.push(event) + } + await abortSupersededTerraformFenceBeforeUpload(config, common) + assert.equal(aborted, true) + assert.deepEqual(events, [ + { + event: 'terraform_fence_superseded_before_upload', + cellId: cell.cellId, + previousFenceCommit: 'b'.repeat(40), + fenceCommit: config.fenceCommit + } + ]) + + await assert.rejects( + abortSupersededTerraformFenceBeforeUpload(config, { + ...common, + loadAttempt: async () => ({ ...attempt, planObjectGeneration: '123456789' }) + }), + /after plan upload/ + ) + await assert.rejects( + abortSupersededTerraformFenceBeforeUpload(config, { + ...common, + resolvePlan: async () => ({ generation: '123456789' }) + }), + /with a saved plan/ + ) +}) + +test('resumes and attests when state and live GCE are already zero', async () => { + let attested + const config = applyConfig() + const attempt = durableAttempt(config, { + applyStartedAt: 101, + gceOperation: 'operation-1', + completedAt: 120 + }) + let deleted = false + await resumeTerraformFence(config, { + git: (args) => (args.includes('rev-parse') ? `${config.fenceCommit}\n` : ''), + assertCommittedFenceSet: async () => {}, + loadAttempt: async () => attempt, + inspectProgress: async () => ({ + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 8, + gceOperation: 'operation-1' + }), + preApplyGuard: async () => {}, + postApplyGuard: async () => {}, + assertZeroDiff: async () => {}, + deletePlan: async () => { + deleted = true + }, + attest: async (value) => { + attested = value + } + }) + assert.equal(attested.attemptId, attempt.attemptId) + assert.equal(deleted, true) +}) + +function replayHarness({ + initialProgress, + finalProgress, + applyError = null, + zeroDiffError = null, + downloadedPlan = 'private saved plan', + attemptOverrides = {} +} = {}) { + const config = applyConfig() + const root = mkdtempSync(join(tmpdir(), 'relay-fence-replay-test-')) + const attempt = durableAttempt(config, { + applyStartedAt: 101, + ...attemptOverrides + }) + const progress = [ + initialProgress ?? { + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT', + operationError: false, + stateLineage, + stateSerial: 7 + }, + finalProgress ?? { + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 8, + gceOperation: 'operation-1' + } + ] + let progressIndex = 0 + let applies = 0 + let downloads = 0 + let deleted = false + let attested = false + let zeroDiffChecks = 0 + return { + config, + attempt, + applies: () => applies, + downloads: () => downloads, + deleted: () => deleted, + attested: () => attested, + zeroDiffChecks: () => zeroDiffChecks, + deps: { + git: (args) => (args.includes('rev-parse') ? `${config.fenceCommit}\n` : ''), + tmpdir: () => root, + assertCommittedFenceSet: async () => {}, + loadAttempt: async () => attempt, + resolvePlan: async () => ({ generation: '123456789' }), + bindPlan: async (value) => ({ attempt: value }), + inspectProgress: async (_expected, currentAttempt) => { + const value = progress[Math.min(progressIndex++, progress.length - 1)] + if (!value.gceOperation || value.invocationOperations) return value + const invocation = currentAttempt.applyInvocations?.at(-1) + return { + ...value, + invocationOperations: invocation + ? [ + { + ...invocation, + gceOperation: value.gceOperation, + operationStatus: value.operationStatus, + operationError: value.operationError, + auditBound: value.operationAuditBound + } + ] + : [] + } + }, + preApplyGuard: async () => {}, + postApplyGuard: async () => {}, + stateObjectBinding: async () => ({ + generation: attempt.terraformStateObjectGeneration, + sha256: attempt.terraformStateObjectSha256, + lineage: stateLineage, + serial: 7 + }), + assertZeroDiff: async () => { + zeroDiffChecks++ + if (zeroDiffError) throw zeroDiffError + }, + downloadPlan: async (value, path) => { + assert.equal( + value.planObjectGeneration, + attempt.planObjectGeneration ?? '123456789' + ) + downloads++ + writeFileSync(path, downloadedPlan, { mode: 0o600 }) + }, + terraform: (args) => { + if (args.includes('show')) return JSON.stringify(plan()) + if (args.includes('apply')) { + applies++ + if (applyError) throw applyError + } + }, + markOperation: async (value, invocation) => ({ + attempt: value, + invocation: { ...invocation, gceOperation: value.gceOperation } + }), + markApplyStarted: async (value, invocation) => ({ + attempt: { ...value, applyStartedAt: value.applyStartedAt ?? 101 }, + invocation: { ...invocation, startedAt: 101 } + }), + attest: async () => { + attested = true + }, + deletePlan: async () => { + deleted = true + } + }, + cleanup: () => rmSync(root, { recursive: true, force: true }) + } +} + +test('replays the exact durable plan after crashing immediately after apply-start', async () => { + const harness = replayHarness() + try { + await resumeTerraformFence(harness.config, harness.deps) + assert.equal(harness.downloads(), 1) + assert.equal(harness.applies(), 1) + assert.equal(harness.attested(), true) + assert.equal(harness.deleted(), true) + } finally { + harness.cleanup() + } +}) + +test('recovers an uploaded plan whose generation was not bound before runner loss', async () => { + const harness = replayHarness({ + attemptOverrides: { + planObjectGeneration: undefined, + applyStartedAt: undefined + } + }) + try { + await resumeTerraformFence(harness.config, harness.deps) + assert.equal(harness.downloads(), 1) + assert.equal(harness.applies(), 1) + assert.equal(harness.attested(), true) + } finally { + harness.cleanup() + } +}) + +test('replays the exact durable plan to reconcile live zero with stale state', async () => { + const harness = replayHarness({ + initialProgress: { + stateTargetSize: 1, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 7, + gceOperation: 'operation-1' + } + }) + try { + await resumeTerraformFence(harness.config, harness.deps) + assert.equal(harness.applies(), 1) + assert.equal(harness.attested(), true) + } finally { + harness.cleanup() + } +}) + +test('does not replay a stale saved plan after the first apply already persisted state', async () => { + const harness = replayHarness({ + initialProgress: { + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 8, + gceOperation: 'operation-1' + } + }) + try { + await resumeTerraformFence(harness.config, harness.deps) + assert.equal(harness.downloads(), 0) + assert.equal(harness.applies(), 0) + assert.equal(harness.deleted(), true) + assert.equal(harness.zeroDiffChecks(), 1) + } finally { + harness.cleanup() + } +}) + +test('replays when the recorded Terraform serial is unchanged even if refresh sees zero', async () => { + const harness = replayHarness({ + initialProgress: { + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 7, + gceOperation: 'operation-1' + } + }) + try { + await resumeTerraformFence(harness.config, harness.deps) + assert.equal(harness.applies(), 1) + assert.equal(harness.zeroDiffChecks(), 1) + } finally { + harness.cleanup() + } +}) + +test('freezes a serial-plus-one completion unless the reviewed targeted plan is zero diff', async () => { + const harness = replayHarness({ + initialProgress: { + stateTargetSize: 0, + liveTargetSize: 0, + instanceCount: 0, + operationStatus: 'DONE', + operationError: false, + operationAuditBound: true, + stateLineage, + stateSerial: 8, + gceOperation: 'operation-1' + }, + zeroDiffError: new Error('not zero diff') + }) + try { + await assert.rejects( + resumeTerraformFence(harness.config, harness.deps), + /not zero diff/ + ) + assert.equal(harness.attested(), false) + assert.equal(harness.deleted(), false) + } finally { + harness.cleanup() + } +}) + +test('rejects saved-plan object generation and hash mismatches', async () => { + const generation = replayHarness({ + attemptOverrides: { planObjectGeneration: '0' } + }) + try { + await assert.rejects( + resumeTerraformFence(generation.config, generation.deps), + /saved-plan generation/ + ) + } finally { + generation.cleanup() + } + const digest = replayHarness({ downloadedPlan: 'tampered plan' }) + try { + await assert.rejects( + resumeTerraformFence(digest.config, digest.deps), + /saved fence plan digest mismatch/ + ) + assert.equal(digest.applies(), 0) + assert.equal(digest.deleted(), false) + } finally { + digest.cleanup() + } +}) + +test('rejects Terraform state lineage or serial drift before replay', async () => { + const harness = replayHarness({ + initialProgress: { + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT', + operationError: false, + stateLineage, + stateSerial: 8 + } + }) + try { + await assert.rejects( + resumeTerraformFence(harness.config, harness.deps), + /lineage or serial changed/ + ) + assert.equal(harness.downloads(), 0) + } finally { + harness.cleanup() + } +}) + +test('keeps the durable plan when replay cannot acquire the Terraform lock', async () => { + const notStarted = { + stateTargetSize: 1, + liveTargetSize: 1, + instanceCount: 1, + operationStatus: 'ABSENT', + operationError: false, + stateLineage, + stateSerial: 7 + } + const harness = replayHarness({ + initialProgress: notStarted, + finalProgress: notStarted, + applyError: new Error('state lock unavailable') + }) + try { + await assert.rejects( + resumeTerraformFence(harness.config, harness.deps), + /recover-forward required/ + ) + assert.equal(harness.deleted(), false) + assert.equal(harness.attested(), false) + } finally { + harness.cleanup() + } +}) + +test('deletes the exact object generation permanently and accepts confirmed absence', async () => { + const config = applyConfig() + const attempt = durableAttempt(config) + let args + await deleteTerraformFencePlan( + config, + { + commandResult: (value) => { + args = value + return { status: 0, stderr: '' } + } + }, + attempt + ) + assert.equal( + args[2], + `gs://project-terraform-state/${attempt.planObjectName}#${attempt.planObjectGeneration}` + ) + await assert.doesNotReject( + deleteTerraformFencePlan( + config, + { commandResult: () => ({ status: 1, stderr: '404 not found' }) }, + attempt + ) + ) + await assert.rejects( + deleteTerraformFencePlan( + config, + { commandResult: () => ({ status: 1, stderr: 'permission denied' }) }, + attempt + ), + /could not be deleted/ + ) +}) + +test('binds a DONE resize operation to the exact post-start audit request reason', async () => { + const config = applyConfig() + const attempt = durableAttempt(config, { applyStartedAt: 101 }) + const targetLink = + `https://www.googleapis.com/compute/v1/projects/${config.project}/zones/${cell.zone}/instanceGroupManagers/${cell.migName}` + const terraform = (args) => + args.includes('state') + ? JSON.stringify({ lineage: stateLineage, serial: 8 }) + : JSON.stringify(state(0)) + const gcloudJson = (args) => { + if (args.includes('list-instances')) return [] + if (args.includes('managed')) return { targetSize: 0 } + if (args[0] === 'compute') { + return [ + { + name: 'operation-1', + insertTime: new Date(102).toISOString(), + targetLink, + operationType: 'compute.instanceGroupManagers.resize', + status: 'DONE' + } + ] + } + return [ + { + protoPayload: { + requestMetadata: { + requestAttributes: { + reason: attempt.applyInvocations[0].requestReason + } + }, + resourceName: + `projects/${config.project}/zones/${cell.zone}/instanceGroupManagers/${cell.migName}`, + methodName: 'v1.compute.instanceGroupManagers.resize', + request: { size: 0 }, + response: { name: 'operation-1' } + } + } + ] + } + const progress = await inspectTerraformFenceProgress( + config, + { terraform, gcloudJson }, + attempt + ) + assert.equal(progress.operationAuditBound, true) + assert.equal(classifyTerraformFenceProgress(progress), 'complete') + const unbound = await inspectTerraformFenceProgress( + config, + { + terraform, + gcloudJson: (args) => (args[0] === 'logging' ? [] : gcloudJson(args)) + }, + attempt + ) + assert.throws( + () => classifyTerraformFenceProgress(unbound), + /ambiguous or unsafe/ + ) +}) + +test('inspects an older completed fence through exact principal and operation evidence', async () => { + const config = applyConfig() + const attempt = durableAttempt(config, { applyStartedAt: 101 }) + const gceOperation = 'operation-1' + const principalEmail = 'fence-broker@example.gserviceaccount.com' + const targetLink = + `https://www.googleapis.com/compute/v1/projects/${config.project}/zones/${cell.zone}/instanceGroupManagers/${cell.migName}` + const resourceName = + `projects/${config.project}/zones/${cell.zone}/instanceGroupManagers/${cell.migName}` + const terraform = (args) => + args.includes('state') + ? JSON.stringify({ lineage: stateLineage, serial: 8 }) + : JSON.stringify(state(0)) + const progress = await inspectCompletedTerraformFenceProgress( + config, + { + terraform, + gcloudJson: (args) => { + if (args.includes('list-instances')) return [] + if (args.includes('managed')) { + return { targetSize: 0, status: { isStable: true } } + } + if (args[0] === 'compute') { + return [ + { + name: gceOperation, + insertTime: new Date(102).toISOString(), + targetLink, + operationType: 'compute.instanceGroupManagers.resize', + status: 'DONE' + } + ] + } + return [ + { + timestamp: new Date(103).toISOString(), + protoPayload: { + authenticationInfo: { principalEmail }, + resourceName, + methodName: 'v1.compute.instanceGroupManagers.resize', + request: { size: '0' }, + response: { name: gceOperation } + } + } + ] + } + }, + attempt, + { gceOperation, principalEmail } + ) + assert.deepEqual(progress, { + stateTargetSize: 0, + stateLineage, + stateSerial: 8, + liveTargetSize: 0, + instanceCount: 0, + liveStable: true, + operationStatus: 'DONE', + operationError: false, + gceOperation + }) +}) + +test('adopts only the pinned completed older attempt without replaying Terraform', async () => { + const config = applyConfig() + const root = mkdtempSync(join(tmpdir(), 'relay-fence-completed-recovery-test-')) + const attempt = durableAttempt(config, { + fenceCommit: 'b'.repeat(40), + applyStartedAt: 101 + }) + const recovery = { + attemptId: attempt.attemptId, + fenceCommit: attempt.fenceCommit, + gceOperation: 'operation-1', + terraformStateSerial: 7, + planObjectGeneration: attempt.planObjectGeneration, + terraformStateObjectGeneration: '222222222', + terraformStateObjectSha256: 'c'.repeat(64), + principalEmail: 'fence-broker@example.gserviceaccount.com' + } + const events = [] + let applies = 0 + try { + await recoverSupersededCompletedTerraformFence( + config, + { + git: (args) => (args.includes('rev-parse') ? `${config.fenceCommit}\n` : ''), + tmpdir: () => root, + assertCommittedFenceSet: async () => {}, + loadAttempt: async () => attempt, + resolvePlan: async () => ({ generation: attempt.planObjectGeneration }), + stateObjectBinding: async () => ({ + generation: recovery.terraformStateObjectGeneration, + sha256: recovery.terraformStateObjectSha256, + lineage: stateLineage, + serial: 8 + }), + downloadPlan: async (_value, path) => + writeFileSync(path, 'private saved plan', { mode: 0o600 }), + terraform: (args) => { + if (args.includes('apply')) applies++ + if (args.includes('show')) return JSON.stringify(plan()) + }, + inspectCompletedProgress: async () => ({ + stateTargetSize: 0, + stateLineage, + stateSerial: 8, + liveTargetSize: 0, + instanceCount: 0, + liveStable: true, + operationStatus: 'DONE', + operationError: false, + gceOperation: recovery.gceOperation + }), + markOperation: async (value, invocation) => { + events.push('operation') + return { attempt: value, invocation } + }, + assertZeroDiff: async () => events.push('zero-diff'), + postApplyGuard: async () => events.push('post-apply'), + attest: async () => events.push('attest'), + deletePlan: async () => events.push('delete'), + emit: () => events.push('emit') + }, + recovery + ) + assert.equal(applies, 0) + assert.deepEqual(events, [ + 'operation', + 'zero-diff', + 'post-apply', + 'attest', + 'delete', + 'emit' + ]) + } finally { + rmSync(root, { recursive: true, force: true }) + } +}) + +test('continues after an adopted fence was attested and its plan was deleted', async () => { + const config = applyConfig() + const root = mkdtempSync(join(tmpdir(), 'relay-fence-completed-retry-test-')) + const gceOperation = 'operation-1' + const attempt = durableAttempt(config, { + fenceCommit: 'b'.repeat(40), + applyStartedAt: 101, + completedAt: 200, + gceOperation + }) + const recovery = { + attemptId: attempt.attemptId, + fenceCommit: attempt.fenceCommit, + gceOperation, + terraformStateSerial: 7, + planObjectGeneration: attempt.planObjectGeneration, + terraformStateObjectGeneration: '222222222', + terraformStateObjectSha256: 'c'.repeat(64), + principalEmail: 'fence-broker@example.gserviceaccount.com' + } + let attested = false + try { + await recoverSupersededCompletedTerraformFence( + config, + { + git: (args) => (args.includes('rev-parse') ? `${config.fenceCommit}\n` : ''), + tmpdir: () => root, + assertCommittedFenceSet: async () => {}, + loadAttempt: async () => attempt, + resolvePlan: async () => ({ generation: null }), + stateObjectBinding: async () => ({ + generation: recovery.terraformStateObjectGeneration, + sha256: recovery.terraformStateObjectSha256, + lineage: stateLineage, + serial: 8 + }), + inspectCompletedProgress: async () => ({ + stateTargetSize: 0, + stateLineage, + stateSerial: 8, + liveTargetSize: 0, + instanceCount: 0, + liveStable: true, + operationStatus: 'DONE', + operationError: false, + gceOperation + }), + markOperation: async () => { + throw new Error('must not rebind') + }, + assertZeroDiff: async () => {}, + postApplyGuard: async () => {}, + attest: async () => { + attested = true + }, + downloadPlan: async () => { + throw new Error('must not download a deleted plan') + }, + deletePlan: async () => { + throw new Error('must not delete an absent plan') + } + }, + recovery + ) + assert.equal(attested, true) + } finally { + rmSync(root, { recursive: true, force: true }) + } +}) diff --git a/cloud/dev/scripts/relay-load-connection-failure.mjs b/cloud/dev/scripts/relay-load-connection-failure.mjs new file mode 100644 index 00000000000..0238740b5a2 --- /dev/null +++ b/cloud/dev/scripts/relay-load-connection-failure.mjs @@ -0,0 +1,37 @@ +export function relayLoadFailureReason(error) { + const message = error instanceof Error ? error.message : String(error) + const tokenExchange = /^relay token exchange failed: ([1-5][0-9]{2})$/.exec(message) + if (tokenExchange) return `token_http_${tokenExchange[1]}` + const assignment = + /^relay assignment failed: ([1-5][0-9]{2})(?: (relay_capacity_exhausted|relay_connection_headroom_exhausted))?$/.exec( + message + ) + if (assignment?.[1] === '503' && assignment[2]) return 'assignment_capacity_exhausted' + if (assignment) return `assignment_http_${assignment[1]}` + const closed = /^control closed: ([0-9]{4})\b/.exec(message) + if (closed) return `control_close_${closed[1]}` + if (message === 'control open timeout') return 'control_open_timeout' + if (message === 'control response timeout') return 'control_response_timeout' + if (message === 'relay token exchange timeout') return 'token_timeout' + if (message === 'relay assignment timeout') return 'assignment_timeout' + if (message === 'WebSocket was closed before the connection was established') { + return 'socket_closed_before_open' + } + if (message === 'relay token exchange omitted token') return 'token_response_invalid' + if (message === 'relay assignment response invalid') return 'assignment_response_invalid' + if (message === 'expected host challenge') return 'host_challenge_invalid' + if (message === 'host proof challenge did not decrypt') return 'host_challenge_decrypt_failed' + if (message === 'expected host hello acknowledgement') return 'host_ack_invalid' + const socketResponse = /^Unexpected server response: ([1-5][0-9]{2})\b/.exec(message) + if (socketResponse) return `socket_http_${socketResponse[1]}` + if (/\b(?:ECONNREFUSED|ECONNRESET|EHOSTUNREACH|ETIMEDOUT)\b/.test(message)) { + return 'socket_transport' + } + return 'unknown' +} + +export function discardFailedLoadSocket(socket) { + if (!socket) return + socket.on('error', () => undefined) + socket.terminate() +} diff --git a/cloud/dev/scripts/relay-load-connection-failure.test.mjs b/cloud/dev/scripts/relay-load-connection-failure.test.mjs new file mode 100644 index 00000000000..f583f7b6659 --- /dev/null +++ b/cloud/dev/scripts/relay-load-connection-failure.test.mjs @@ -0,0 +1,57 @@ +import assert from 'node:assert/strict' +import { EventEmitter } from 'node:events' +import test from 'node:test' +import { + discardFailedLoadSocket, + relayLoadFailureReason +} from './relay-load-connection-failure.mjs' + +test('classifies only bounded aggregate connection failure reasons', () => { + assert.equal(relayLoadFailureReason(new Error('relay token exchange failed: 503')), 'token_http_503') + assert.equal(relayLoadFailureReason(new Error('relay assignment failed: 503')), 'assignment_http_503') + assert.equal( + relayLoadFailureReason( + new Error('relay assignment failed: 503 relay_connection_headroom_exhausted') + ), + 'assignment_capacity_exhausted' + ) + for (const status of [400, 429, 500]) { + assert.equal( + relayLoadFailureReason( + new Error(`${`relay assignment failed: ${status}`} relay_capacity_exhausted`) + ), + `assignment_http_${status}` + ) + } + assert.equal(relayLoadFailureReason(new Error('control closed: 4404 wrong cell')), 'control_close_4404') + assert.equal(relayLoadFailureReason(new Error('control open timeout')), 'control_open_timeout') + assert.equal(relayLoadFailureReason(new Error('relay token exchange timeout')), 'token_timeout') + assert.equal(relayLoadFailureReason(new Error('relay assignment timeout')), 'assignment_timeout') + assert.equal( + relayLoadFailureReason(new Error('relay token exchange omitted token')), + 'token_response_invalid' + ) + assert.equal( + relayLoadFailureReason(new Error('relay assignment response invalid')), + 'assignment_response_invalid' + ) + assert.equal( + relayLoadFailureReason(new Error('Unexpected server response: 503 Service Unavailable')), + 'socket_http_503' + ) + assert.equal(relayLoadFailureReason(new Error('connect ECONNRESET 127.0.0.1')), 'socket_transport') + assert.equal(relayLoadFailureReason(new Error('expected host challenge')), 'host_challenge_invalid') + assert.equal( + relayLoadFailureReason(new Error('host proof challenge did not decrypt')), + 'host_challenge_decrypt_failed' + ) + assert.equal(relayLoadFailureReason(new Error('expected host hello acknowledgement')), 'host_ack_invalid') + assert.equal(relayLoadFailureReason(new Error('host-sensitive detail')), 'unknown') +}) + +test('absorbs the setup error emitted while discarding a failed socket', () => { + const socket = new EventEmitter() + socket.terminate = () => socket.emit('error', new Error('closed before open')) + + assert.doesNotThrow(() => discardFailedLoadSocket(socket)) +}) diff --git a/cloud/dev/scripts/relay-load-control-peer.mjs b/cloud/dev/scripts/relay-load-control-peer.mjs new file mode 100644 index 00000000000..bb954a5d150 --- /dev/null +++ b/cloud/dev/scripts/relay-load-control-peer.mjs @@ -0,0 +1,891 @@ +import { createHash, createHmac } from 'node:crypto' +import { createRequire } from 'node:module' +import { controlPhase } from './relay-load-model.mjs' +import { discardFailedLoadSocket } from './relay-load-connection-failure.mjs' + +const requireFromRelay = createRequire(new URL('../../apps/relay/package.json', import.meta.url)) +const nacl = requireFromRelay('tweetnacl') +const WebSocket = requireFromRelay('ws') +const { SignJWT } = await import(requireFromRelay.resolve('jose')) +const { buildHostProofMacInput, HOST_CHALLENGE_PLAINTEXT_DOMAIN } = await import( + requireFromRelay.resolve('@orca-cloud/relay-contract') +) + +const CAPACITY_ASSIGNMENT_ERRORS = [ + 'relay_capacity_exhausted', + 'relay_connection_headroom_exhausted' +] + +function waitForOpen(socket, timeoutMs = 10_000) { + return new Promise((resolve, reject) => { + const timer = setTimeout(() => finish(new Error('control open timeout')), timeoutMs) + const finish = (error) => { + clearTimeout(timer) + socket.off('open', onOpen) + socket.off('close', onClose) + socket.off('error', onError) + if (error) reject(error) + else resolve() + } + const onOpen = () => finish() + const onClose = (code, reason) => finish(new Error(`control closed: ${code} ${reason}`)) + const onError = (error) => finish(error) + socket.once('open', onOpen) + socket.once('close', onClose) + socket.once('error', onError) + }) +} + +function nextJson(socket, timeoutMs = 10_000) { + return new Promise((resolve, reject) => { + const timer = setTimeout(() => finish(new Error('control response timeout')), timeoutMs) + const finish = (error, value) => { + clearTimeout(timer) + socket.off('message', onMessage) + socket.off('close', onClose) + socket.off('error', onError) + if (error) reject(error) + else resolve(value) + } + const onMessage = (data) => { + try { + finish(undefined, JSON.parse(data.toString())) + } catch (error) { + finish(error) + } + } + const onClose = (code, reason) => finish(new Error(`control closed: ${code} ${reason}`)) + const onError = (error) => finish(error) + socket.once('message', onMessage) + socket.once('close', onClose) + socket.once('error', onError) + }) +} + +function nextFrame(socket, timeoutMs = 10_000) { + return new Promise((resolve, reject) => { + const timer = setTimeout(() => finish(new Error('relay frame timeout')), timeoutMs) + const finish = (error, value) => { + clearTimeout(timer) + socket.off('message', onMessage) + socket.off('close', onClose) + socket.off('error', onError) + if (error) reject(error) + else resolve(value) + } + const onMessage = (data, binary) => finish(undefined, { bytes: Buffer.from(data), binary }) + const onClose = (code, reason) => finish(new Error(`splice closed: ${code} ${reason}`)) + const onError = (error) => finish(error) + socket.once('message', onMessage) + socket.once('close', onClose) + socket.once('error', onError) + }) +} + +function receiveBinaryStream(socket, expectedBytes, timeoutMs) { + return new Promise((resolve, reject) => { + let receivedBytes = 0 + const hash = createHash('sha256') + const timer = setTimeout(() => finish(new Error('relay stream timeout')), timeoutMs) + const finish = (error, value) => { + clearTimeout(timer) + socket.off('message', onMessage) + socket.off('close', onClose) + socket.off('error', onError) + if (error) reject(error) + else resolve(value) + } + const onMessage = (data, binary) => { + if (!binary) return finish(new Error('relay changed stream opcode')) + const bytes = Buffer.from(data) + receivedBytes += bytes.byteLength + hash.update(bytes) + if (receivedBytes > expectedBytes) return finish(new Error('relay expanded reader stream')) + if (receivedBytes === expectedBytes) { + finish(undefined, { bytes: receivedBytes, digest: hash.digest('hex') }) + } + } + const onClose = (code, reason) => finish(new Error(`splice closed: ${code} ${reason}`)) + const onError = (error) => finish(error) + socket.on('message', onMessage) + socket.once('close', onClose) + socket.once('error', onError) + }) +} + +function closeInfo(socket, timeoutMs) { + return new Promise((resolve, reject) => { + const timer = setTimeout(() => finish(new Error('reader close timeout')), timeoutMs) + const finish = (error, value) => { + clearTimeout(timer) + socket.off('close', onClose) + socket.off('error', onError) + if (error) reject(error) + else resolve(value) + } + const onClose = (code, reason) => finish(undefined, { code, reason: reason.toString() }) + const onError = (error) => finish(error) + socket.once('close', onClose) + socket.once('error', onError) + }) +} + +export function relayLoadWedgedCloseAccepted(closeCodes) { + return closeCodes.length === 2 && closeCodes[1] === 4429 && + (closeCodes[0] === 4429 || closeCodes[0] === 1006) +} + +function waitForClose(socket, timeoutMs = 10_000) { + if (!socket || socket.readyState === socket.CLOSED) return Promise.resolve() + return new Promise((resolve, reject) => { + const timer = setTimeout(() => { + socket.off('close', onClose) + discardFailedLoadSocket(socket) + reject(new Error('control close timeout')) + }, timeoutMs) + const onClose = () => { + clearTimeout(timer) + resolve() + } + socket.once('close', onClose) + }) +} + +async function cancelResponse(response) { + try { + await response.body?.cancel() + } catch { + // Preserve the bounded failure classification. + } +} + +function proofForChallenge(challenge, hostSecretKey) { + const plaintext = nacl.box.open( + Buffer.from(challenge.ciphertextB64, 'base64'), + Buffer.from(challenge.nonceB64, 'base64'), + Buffer.from(challenge.relayEphemeralPublicKeyB64, 'base64'), + hostSecretKey + ) + if (!plaintext) throw new Error('host proof challenge did not decrypt') + const domain = new TextEncoder().encode(`${HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`) + const transcriptLength = new DataView( + plaintext.buffer, + plaintext.byteOffset + domain.length, + 4 + ).getUint32(0, false) + const transcriptStart = domain.length + 4 + const transcript = plaintext.slice(transcriptStart, transcriptStart + transcriptLength) + const secret = plaintext.slice(transcriptStart + transcriptLength) + return createHmac('sha256', secret).update(buildHostProofMacInput(transcript)).digest('base64') +} + +export class RelayLoadControlPeer { + constructor(index, options, observe) { + if (options.directorOrigin && options.targetOrigin) { + throw new Error('provide either directorOrigin or targetOrigin, not both') + } + this.index = index + this.options = options + this.observe = observe + this.keys = nacl.box.keyPair() + this.relayHostId = createHash('sha256') + .update(this.keys.publicKey) + .digest('base64url') + .slice(0, 16) + this.phase = controlPhase(index, options.seed) + this.socket = null + this.generation = undefined + this.controlResumeSecret = undefined + this.lastAssignment = undefined + this.refreshTimer = null + this.stopped = false + this.connecting = false + this.inFlight = new Set() + this.shutdownPromise = null + this.abortController = new AbortController() + this.drainExpected = false + this.controlWaiters = new Set() + this.spliceSockets = new Set() + this.spliceSequence = 0 + } + + connect() { + if ( + this.stopped || + this.connecting || + (this.socket !== null && this.socket.readyState === this.socket.OPEN) + ) { + return Promise.resolve() + } + this.connecting = true + const operation = this.connectOnce() + this.inFlight.add(operation) + const finish = () => { + this.connecting = false + this.inFlight.delete(operation) + } + operation.then(finish, finish) + return operation + } + + assignedCellUrl() { + return this.lastAssignment?.cellUrl + } + + async connectOnce() { + let socket = null + try { + const relayToken = await this.relayToken() + if (this.stopped) return + const assignment = await this.assignment(relayToken) + if (this.stopped) return + this.lastAssignment = assignment + socket = this.createSocket(assignment, relayToken) + this.socket = socket + this.drainExpected = false + await waitForOpen(socket) + if (this.stopped || this.socket !== socket) return + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: this.relayHostId, + assignmentEpoch: assignment.assignmentEpoch, + hostPublicKeyB64: Buffer.from(this.keys.publicKey).toString('base64'), + appVersion: 'relay-load', + ...(this.generation === undefined ? {} : { previousGeneration: this.generation }), + ...(this.controlResumeSecret === undefined + ? {} + : { controlResumeSecret: this.controlResumeSecret }) + }) + ) + const challenge = await nextJson(socket) + if (this.stopped || this.socket !== socket) return + if (challenge.type !== 'host-challenge') throw new Error('expected host challenge') + socket.send( + JSON.stringify({ + type: 'host-challenge-ack', + challengeId: challenge.challengeId, + proofB64: proofForChallenge(challenge, this.keys.secretKey) + }) + ) + const ack = await nextJson(socket) + if (this.stopped || this.socket !== socket) return + if (ack.type !== 'host-hello-ack') throw new Error('expected host hello acknowledgement') + this.generation = ack.generation + this.controlResumeSecret = ack.controlResumeSecret + socket.on('message', (data) => this.onMessage(socket, data)) + socket.once('close', (code) => this.onClose(socket, code)) + socket.once('error', (error) => this.observe('socketError', { index: this.index, error })) + this.observe('connected', { index: this.index }) + this.scheduleRefresh(this.phase.refreshOffsetMs) + } catch (error) { + discardFailedLoadSocket(socket) + if (this.socket === socket) this.socket = null + if (this.stopped) return + throw error + } + } + + createSocket(assignment, relayToken) { + return new WebSocket(`${assignment.cellUrl.replace(/^http/, 'ws')}/v1/host/control`, { + headers: { authorization: `Bearer ${relayToken}` }, + perMessageDeflate: false + }) + } + + shutdown() { + if (this.shutdownPromise) return this.shutdownPromise + this.stopped = true + this.abortController.abort() + if (this.refreshTimer) clearTimeout(this.refreshTimer) + this.refreshTimer = null + this.shutdownPromise = this.shutdownOnce() + return this.shutdownPromise + } + + async shutdownOnce() { + const socket = this.socket + const closed = waitForClose(socket) + for (const spliceSocket of this.spliceSockets) { + if ( + spliceSocket.readyState !== spliceSocket.CLOSED && + spliceSocket.readyState !== spliceSocket.CLOSING + ) { + spliceSocket.close(1000, 'load complete') + } + } + this.rejectControlWaiters(new Error('control stopped')) + if (socket && socket.readyState !== socket.CLOSED && socket.readyState !== socket.CLOSING) { + socket.close(1000, 'load complete') + } + const settled = async () => { + while (this.inFlight.size > 0) { + await Promise.allSettled([...this.inFlight]) + } + } + await Promise.all([closed, settled()]) + this.observe('shutdown', { + index: this.index, + activeControls: this.socket?.readyState === this.socket?.OPEN ? 1 : 0, + activeSpliceSockets: this.spliceSockets.size, + inFlightOperations: this.inFlight.size, + refreshTimerActive: this.refreshTimer !== null + }) + } + + openSplice(options = {}) { + if (this.stopped) return Promise.reject(new Error('control stopped')) + const operation = this.openSpliceOnce(options) + this.inFlight.add(operation) + const finish = () => this.inFlight.delete(operation) + operation.then(finish, finish) + return operation + } + + openInviteOffer() { + if (this.stopped) return Promise.reject(new Error('control stopped')) + const operation = this.openInviteOfferOnce() + this.inFlight.add(operation) + const finish = () => this.inFlight.delete(operation) + operation.then(finish, finish) + return operation + } + + async openInviteOfferOnce() { + if (!this.socket || this.socket.readyState !== this.socket.OPEN) { + throw new Error('active control required for invite offer') + } + const sequence = this.spliceSequence++ + const reqId = `load-offer-${this.index}-${sequence}` + const response = this.waitForControlMessage( + (message) => + message.reqId === reqId && + (message.type === 'invite-created' || message.type === 'control-error') + ) + this.socket.send(JSON.stringify({ + type: 'invite-create', + reqId, + relayDeviceId: `load-offer-device-${this.index}-${sequence}` + })) + const result = await response + if (result.type === 'control-error') throw new Error(`invite offer failed: ${result.code}`) + if ( + typeof result.inviteToken !== 'string' || + !Number.isSafeInteger(result.expiresAt) || + result.expiresAt <= Date.now() + ) throw new Error('relay invite offer response invalid') + } + + async openSpliceOnce({ + payloadBytes = 64, + readerMode = 'normal', + readerHoldMs = 0, + streamBytes = payloadBytes, + frameBytes = payloadBytes, + observeReaderPressure = async () => undefined, + readerDelay = async (ms) => await new Promise((resolve) => setTimeout(resolve, ms)), + slowReaderHoldMs = 0, + holdMs = 0 + } = {}) { + if ( + !this.socket || + this.socket.readyState !== this.socket.OPEN || + this.generation === undefined || + this.lastAssignment === undefined + ) { + throw new Error('active control required for splice') + } + if (!Number.isSafeInteger(payloadBytes) || payloadBytes < 1) { + throw new Error('splice payload bytes must be positive') + } + const sequence = this.spliceSequence++ + const reqId = `load-invite-${this.index}-${sequence}` + const relayDeviceId = `load-device-${this.index}-${sequence}` + let phone + let data + let opened = false + try { + const invitePromise = this.waitForControlMessage( + (message) => message.type === 'invite-created' && message.reqId === reqId + ) + this.socket.send(JSON.stringify({ type: 'invite-create', reqId, relayDeviceId })) + const invite = await invitePromise + if (typeof invite.inviteToken !== 'string') throw new Error('relay invite response invalid') + + phone = this.createClientSocket(this.lastAssignment) + this.trackSpliceSocket(phone) + await waitForOpen(phone) + const connectionPromise = this.waitForControlMessage( + (message) => message.type === 'conn-open' && message.relayDeviceId === relayDeviceId + ) + phone.send( + JSON.stringify({ type: 'relay-auth', v: 1, mode: 'connect', credential: invite.inviteToken }) + ) + const connection = await connectionPromise + if (typeof connection.connId !== 'string' || typeof connection.connTicket !== 'string') { + throw new Error('relay connection response invalid') + } + + data = this.createHostDataSocket(this.lastAssignment, connection.connId) + this.trackSpliceSocket(data) + await waitForOpen(data) + const phoneHello = nextJson(phone) + data.send( + JSON.stringify({ + type: 'host-data-auth', + v: 1, + connTicket: connection.connTicket, + generation: this.generation + }) + ) + if ((await phoneHello).ok !== true) throw new Error('relay rejected load splice') + + if (slowReaderHoldMs > 0 && readerMode === 'normal') { + readerMode = 'slow' + readerHoldMs = slowReaderHoldMs + streamBytes = payloadBytes + frameBytes = payloadBytes + } + if (!['normal', 'slow', 'wedged'].includes(readerMode)) { + throw new Error('reader mode is invalid') + } + const pausedSocket = readerMode === 'normal' ? undefined : phone._socket + if (readerMode !== 'normal' && !pausedSocket) throw new Error('reader transport unavailable') + if (readerMode === 'wedged') { + const closes = [closeInfo(phone, readerHoldMs + 10_000), closeInfo(data, readerHoldMs + 10_000)] + pausedSocket.pause() + const readerPausedAt = Date.now() + const readerPressure = observeReaderPressure({ + cellOrigin: this.lastAssignment.cellUrl, + readerMode, + streamBytes + }) + const [sent] = await Promise.all([ + this.sendReaderStream(data, sequence, streamBytes, frameBytes, readerDelay), + readerPressure + ]) + await readerDelay(Math.max(0, readerHoldMs - (Date.now() - readerPausedAt))) + pausedSocket.resume() + const closeEvidence = await Promise.all(closes) + const closeCodes = closeEvidence.map(({ code }) => code) + if (!relayLoadWedgedCloseAccepted(closeCodes)) { + throw new Error(`wedged reader close codes: ${closeCodes.join(',')}`) + } + opened = true + this.observe('spliceOpened', { index: this.index, readerMode }) + this.observe('spliceWedged', { index: this.index, code: 4429, streamBytes: sent.bytes }) + return + } + + const expectedStreamBytes = readerMode === 'slow' ? streamBytes : payloadBytes + const expectedFrameBytes = readerMode === 'slow' ? frameBytes : payloadBytes + const phoneStream = receiveBinaryStream( + phone, + expectedStreamBytes, + readerMode === 'slow' ? readerHoldMs + 10_000 : 10_000 + ) + const readerPausedAt = pausedSocket ? Date.now() : 0 + if (pausedSocket) pausedSocket.pause() + const readerPressure = readerMode === 'slow' + ? observeReaderPressure({ + cellOrigin: this.lastAssignment.cellUrl, + readerMode, + streamBytes: expectedStreamBytes + }) + : Promise.resolve() + const [sent] = await Promise.all([ + this.sendReaderStream( + data, + sequence, + expectedStreamBytes, + expectedFrameBytes, + readerDelay + ), + readerPressure + ]) + if (readerMode === 'slow') { + await readerDelay(Math.max(0, readerHoldMs - (Date.now() - readerPausedAt))) + pausedSocket.resume() + } + const receivedByPhone = await phoneStream + if (receivedByPhone.bytes !== sent.bytes || receivedByPhone.digest !== sent.digest) { + throw new Error('relay changed host-to-client splice payload') + } + + const textPayload = `orca-relay-load:${this.index}:${sequence}:${payloadBytes}` + const dataFrame = nextFrame(data) + phone.send(textPayload) + const receivedByHost = await dataFrame + if (receivedByHost.binary || receivedByHost.bytes.toString() !== textPayload) { + throw new Error('relay changed client-to-host splice payload') + } + opened = true + this.observe('spliceOpened', { index: this.index, readerMode }) + if (holdMs > 0 && !(await this.waitForSpliceHold(holdMs, [phone, data]))) return + this.observe('spliceCompleted', { index: this.index, readerMode }) + } catch (error) { + if (!this.stopped) this.observe('spliceFailed', { index: this.index, error }) + throw error + } finally { + await Promise.all([this.closeSpliceSocket(phone), this.closeSpliceSocket(data)]) + if (opened) this.observe('spliceClosed', { index: this.index }) + } + } + + waitForSpliceHold(holdMs, sockets) { + if (this.stopped) return Promise.resolve(false) + return new Promise((resolve, reject) => { + const finish = (error, completed = false) => { + clearTimeout(timer) + this.abortController.signal.removeEventListener('abort', onAbort) + for (const socket of sockets) { + socket.off('close', onClose) + socket.off('error', onError) + } + if (error) reject(error) + else resolve(completed) + } + const onAbort = () => finish(undefined, false) + const onClose = (code, reason) => finish(new Error(`splice closed: ${code} ${reason}`)) + const onError = (error) => finish(error) + const timer = setTimeout(() => finish(undefined, true), holdMs) + this.abortController.signal.addEventListener('abort', onAbort, { once: true }) + for (const socket of sockets) { + socket.once('close', onClose) + socket.once('error', onError) + } + }) + } + + createClientSocket(assignment) { + return new WebSocket( + `${assignment.cellUrl.replace(/^http/, 'ws')}/v1/connect/${this.relayHostId}`, + { perMessageDeflate: false } + ) + } + + createHostDataSocket(assignment, connId) { + return new WebSocket( + `${assignment.cellUrl.replace(/^http/, 'ws')}/v1/host/data/${connId}`, + { perMessageDeflate: false } + ) + } + + splicePayload(sequence, payloadBytes) { + const seed = createHash('sha256') + .update(`orca-relay-load:${this.index}:${sequence}`) + .digest() + return Buffer.allocUnsafe(payloadBytes).map((_, index) => seed[index % seed.length]) + } + + async sendReaderStream(socket, sequence, streamBytes, frameBytes, delay) { + const hash = createHash('sha256') + let sentBytes = 0 + let frameIndex = 0 + while (sentBytes < streamBytes) { + const bytes = Math.min(frameBytes, streamBytes - sentBytes) + const payload = this.splicePayload(sequence + frameIndex, bytes) + await this.sendReaderFrame(socket, payload) + hash.update(payload) + sentBytes += bytes + frameIndex++ + while (socket.bufferedAmount > frameBytes) await delay(10) + } + return { bytes: sentBytes, digest: hash.digest('hex') } + } + + sendReaderFrame(socket, payload) { + if (socket.send.length < 2) { + socket.send(payload) + return Promise.resolve() + } + return new Promise((resolve, reject) => { + const timer = setTimeout(() => reject(new Error('reader send timeout')), 10_000) + socket.send(payload, (error) => { + clearTimeout(timer) + if (error) reject(error) + else resolve() + }) + }) + } + + trackSpliceSocket(socket) { + this.spliceSockets.add(socket) + socket.once('close', () => this.spliceSockets.delete(socket)) + } + + async closeSpliceSocket(socket) { + if (!socket) return + const closed = waitForClose(socket).catch(() => undefined) + if (socket.readyState !== socket.CLOSED && socket.readyState !== socket.CLOSING) { + socket.close(1000, 'splice complete') + } + await closed + this.spliceSockets.delete(socket) + } + + async openRebindProbe() { + if ( + !this.socket || + this.socket.readyState !== this.socket.OPEN || + this.generation === undefined || + this.controlResumeSecret === undefined || + this.lastAssignment === undefined + ) { + throw new Error('active control required for rebind probe') + } + const relayToken = await this.relayToken() + const socket = new WebSocket( + `${this.lastAssignment.cellUrl.replace(/^http/, 'ws')}/v1/host/control`, + { + headers: { authorization: `Bearer ${relayToken}` }, + perMessageDeflate: false + } + ) + try { + await waitForOpen(socket) + socket.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId: this.relayHostId, + assignmentEpoch: this.lastAssignment.assignmentEpoch, + hostPublicKeyB64: Buffer.from(this.keys.publicKey).toString('base64'), + appVersion: 'relay-load-rebind-proof', + previousGeneration: this.generation, + controlResumeSecret: this.controlResumeSecret + }) + ) + const challenge = await nextJson(socket) + if (challenge.type !== 'host-challenge') throw new Error('expected host challenge') + socket.on('error', () => undefined) + const closed = new Promise((resolve) => socket.once('close', resolve)) + return { + close: async () => { + const closeCompleted = waitForClose(socket) + if (socket.readyState !== socket.CLOSED && socket.readyState !== socket.CLOSING) { + socket.close(1000, 'rebind boundary proved') + } + await closeCompleted + }, + closed, + isOpen: () => socket.readyState === socket.OPEN + } + } catch (error) { + discardFailedLoadSocket(socket) + await waitForClose(socket).catch(() => undefined) + throw error + } + } + + async relayToken() { + const accessToken = this.options.accessTokenProvider + ? await this.options.accessTokenProvider() + : this.options.accessToken + if (accessToken) { + const body = await this.requestJson( + `${this.options.authOrigin}/v1/desktop/auth/relay-token`, + { + method: 'POST', + headers: { + authorization: `Bearer ${accessToken}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + relayHostId: this.relayHostId, + hostPublicKeyB64: Buffer.from(this.keys.publicKey).toString('base64') + }) + }, + 'relay token exchange timeout', + (status) => `relay token exchange failed: ${status}` + ) + if (typeof body.relayToken !== 'string') throw new Error('relay token exchange omitted token') + if (!this.stopped) this.observe('token', { index: this.index }) + return body.relayToken + } + const token = await new SignJWT({ + prof: `load-profile-${this.index}`, + org: 'relay-load', + purpose: 'host-control', + relayHostId: this.relayHostId + }) + .setProtectedHeader({ alg: 'ES256', kid: this.options.signingKeyId }) + .setIssuer(this.options.authOrigin) + .setAudience('orca-relay') + .setSubject(`load-user-${this.index}`) + .setIssuedAt() + .setExpirationTime('5m') + .sign(this.options.signingKey) + if (!this.stopped) this.observe('token', { index: this.index }) + return token + } + + async requestAssignment(preferredRegion) { + return await this.assignment(await this.relayToken(), preferredRegion) + } + + async assignment(relayToken, preferredRegion = this.options.preferredRegion) { + if (!this.options.directorOrigin) { + if (!this.options.targetOrigin) throw new Error('relay target origin missing') + return { cellUrl: this.options.targetOrigin, assignmentEpoch: 1 } + } + const body = await this.requestJson( + `${this.options.directorOrigin}/v1/assign`, + { + method: 'POST', + headers: { authorization: `Bearer ${relayToken}`, 'content-type': 'application/json' }, + body: JSON.stringify({ + v: 1, + relayHostId: this.relayHostId, + ...(preferredRegion ? { preferredRegion } : {}) + }) + }, + 'relay assignment timeout', + (status, errorCode) => + `relay assignment failed: ${status}${errorCode ? ` ${errorCode}` : ''}`, + CAPACITY_ASSIGNMENT_ERRORS + ) + if ( + typeof body.cellUrl !== 'string' || + !Number.isSafeInteger(body.assignmentEpoch) || + body.assignmentEpoch < 1 + ) { + throw new Error('relay assignment response invalid') + } + return body + } + + async requestJson(url, init, timeoutMessage, httpErrorMessage, allowedErrorCodes = []) { + const controller = new AbortController() + const onShutdown = () => controller.abort() + if (this.abortController.signal.aborted) controller.abort() + else this.abortController.signal.addEventListener('abort', onShutdown, { once: true }) + let timedOut = false + const timer = setTimeout(() => { + timedOut = true + controller.abort() + }, this.options.requestTimeoutMs ?? 10_000) + try { + const response = await fetch(url, { ...init, signal: controller.signal }) + if (!response.ok) { + let bodyConsumed = false + let errorCode + if (allowedErrorCodes.length > 0) { + try { + const body = await response.json() + bodyConsumed = true + if (allowedErrorCodes.includes(body?.error)) errorCode = body.error + } catch { + // Preserve the bounded status-only classification. + } + } + if (!bodyConsumed) await cancelResponse(response) + throw new Error(httpErrorMessage(response.status, errorCode)) + } + return await response.json() + } catch (error) { + if (timedOut) throw new Error(timeoutMessage, { cause: error }) + throw error + } finally { + clearTimeout(timer) + this.abortController.signal.removeEventListener('abort', onShutdown) + } + } + + onMessage(socket, data) { + let message + try { + message = JSON.parse(data.toString()) + } catch { + this.observe('protocolError', { index: this.index }) + return + } + for (const waiter of this.controlWaiters) { + if (waiter.matches(message)) { + this.controlWaiters.delete(waiter) + clearTimeout(waiter.timer) + waiter.resolve(message) + return + } + } + if (message.type === 'ping') { + socket.send(JSON.stringify({ type: 'pong', t: message.t })) + this.observe('ping', { index: this.index }) + } else if (message.type === 'drain') { + this.drainExpected = true + this.observe('drain', { index: this.index }) + } + } + + onClose(socket, code) { + if (this.socket !== socket) return + this.socket = null + if (this.refreshTimer) clearTimeout(this.refreshTimer) + this.refreshTimer = null + this.rejectControlWaiters(new Error(`control closed: ${code}`)) + this.observe('closed', { + index: this.index, + code, + stopped: this.stopped, + expectedDrain: this.drainExpected + }) + this.drainExpected = false + } + + waitForControlMessage(matches, timeoutMs = 10_000) { + if (this.stopped) return Promise.reject(new Error('control stopped')) + return new Promise((resolve, reject) => { + const waiter = { + matches, + resolve, + reject, + timer: setTimeout(() => { + this.controlWaiters.delete(waiter) + reject(new Error('control response timeout')) + }, timeoutMs) + } + this.controlWaiters.add(waiter) + }) + } + + rejectControlWaiters(error) { + for (const waiter of this.controlWaiters) { + clearTimeout(waiter.timer) + waiter.reject(error) + } + this.controlWaiters.clear() + } + + scheduleRefresh(delayMs) { + if (this.stopped) return + this.refreshTimer = setTimeout(() => { + void this.refresh().then( + () => this.scheduleRefresh(this.phase.refreshIntervalMs), + () => this.scheduleRefresh(this.phase.refreshIntervalMs) + ) + }, delayMs) + } + + refresh() { + if (this.stopped) return Promise.resolve() + const operation = this.refreshOnce() + this.inFlight.add(operation) + const finish = () => this.inFlight.delete(operation) + operation.then(finish, finish) + return operation + } + + async refreshOnce() { + const socket = this.socket + if (this.stopped || !socket || socket.readyState !== socket.OPEN) return + try { + const relayJwt = await this.relayToken() + if (this.stopped || this.socket !== socket || socket.readyState !== socket.OPEN) return + socket.send(JSON.stringify({ type: 'auth-refresh', relayJwt })) + this.observe('refresh', { index: this.index }) + } catch (error) { + if (!this.stopped) this.observe('refreshError', { index: this.index, error }) + } + } +} diff --git a/cloud/dev/scripts/relay-load-control-peer.test.mjs b/cloud/dev/scripts/relay-load-control-peer.test.mjs new file mode 100644 index 00000000000..95ebd89e1ef --- /dev/null +++ b/cloud/dev/scripts/relay-load-control-peer.test.mjs @@ -0,0 +1,903 @@ +import assert from 'node:assert/strict' +import { generateKeyPairSync } from 'node:crypto' +import { EventEmitter } from 'node:events' +import { createRequire } from 'node:module' +import test from 'node:test' +import { + RelayLoadControlPeer, + relayLoadWedgedCloseAccepted +} from './relay-load-control-peer.mjs' + +const requireFromRelay = createRequire(new URL('../../apps/relay/package.json', import.meta.url)) +const nacl = requireFromRelay('tweetnacl') +const { buildHostChallengePlaintext } = await import( + requireFromRelay.resolve('@orca-cloud/relay-contract') +) + +function deferred() { + let resolve + let reject + const promise = new Promise((resolvePromise, rejectPromise) => { + resolve = resolvePromise + reject = rejectPromise + }) + return { promise, resolve, reject } +} + +function peerOptions(overrides = {}) { + const { privateKey } = generateKeyPairSync('ec', { namedCurve: 'P-256' }) + return { + authOrigin: 'https://auth.test', + directorOrigin: 'https://director.test', + reconnectMaxMs: 0, + seed: 1, + signingKey: privateKey, + signingKeyId: 'test-key', + ...overrides + } +} + +function response(body) { + return { ok: true, status: 200, json: async () => body } +} + +function fakeOpenSocket() { + const socket = new EventEmitter() + socket.OPEN = 1 + socket.CLOSING = 2 + socket.CLOSED = 3 + socket.readyState = socket.OPEN + socket.sent = [] + socket.send = (message) => socket.sent.push(message) + socket.close = (code = 1000, reason = '') => { + socket.readyState = socket.CLOSING + queueMicrotask(() => { + socket.readyState = socket.CLOSED + socket.emit('close', code, Buffer.from(reason)) + }) + } + socket.terminate = socket.close + return socket +} + +function fakeHandshakeSocket() { + const socket = fakeOpenSocket() + socket.CONNECTING = 0 + socket.readyState = socket.CONNECTING + socket.open = () => { + socket.readyState = socket.OPEN + socket.emit('open') + } + socket.message = (message) => socket.emit('message', Buffer.from(JSON.stringify(message))) + return socket +} + +function openOnNextTurn(socket) { + queueMicrotask(() => socket.open()) + return socket +} + +function validChallenge(peer) { + const relayKeys = nacl.box.keyPair() + const nonce = nacl.randomBytes(nacl.box.nonceLength) + const plaintext = buildHostChallengePlaintext( + new Uint8Array([1, 2, 3]), + nacl.randomBytes(32) + ) + const ciphertext = nacl.box(plaintext, nonce, peer.keys.publicKey, relayKeys.secretKey) + return { + type: 'host-challenge', + challengeId: 'test-challenge', + ciphertextB64: Buffer.from(ciphertext).toString('base64'), + nonceB64: Buffer.from(nonce).toString('base64'), + relayEphemeralPublicKeyB64: Buffer.from(relayKeys.publicKey).toString('base64') + } +} + +test('shutdown waits for pending assignment and prevents a late connection', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + const assignment = deferred() + const assignmentStarted = deferred() + global.fetch = () => { + assignmentStarted.resolve() + return assignment.promise + } + const observations = [] + const peer = new RelayLoadControlPeer(0, peerOptions(), (type) => observations.push(type)) + + const connecting = peer.connect() + await assignmentStarted.promise + let shutdownFinished = false + const shutdown = peer.shutdown().then(() => { + shutdownFinished = true + }) + await new Promise((resolve) => setImmediate(resolve)) + assert.equal(shutdownFinished, false) + + assignment.resolve(response({ cellUrl: 'https://cell.test', assignmentEpoch: 1 })) + await Promise.all([connecting, shutdown]) + assert.equal(peer.socket, null) + assert.equal(observations.includes('connected'), false) +}) + +test('shutdown after socket open prevents a late handshake', async () => { + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ directorOrigin: undefined, targetOrigin: 'https://cell.test' }), + (type) => observations.push(type) + ) + const socket = fakeHandshakeSocket() + const socketCreated = deferred() + peer.createSocket = () => { + socketCreated.resolve() + return socket + } + + const connecting = peer.connect() + await socketCreated.promise + socket.open() + await Promise.all([connecting, peer.shutdown()]) + + assert.deepEqual(socket.sent, []) + assert.equal(observations.includes('connected'), false) +}) + +test('shutdown after challenge prevents a late acknowledgement', async () => { + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ directorOrigin: undefined, targetOrigin: 'https://cell.test' }), + (type) => observations.push(type) + ) + const socket = fakeHandshakeSocket() + const socketCreated = deferred() + const helloSent = deferred() + socket.send = (message) => { + socket.sent.push(message) + helloSent.resolve() + } + peer.createSocket = () => { + socketCreated.resolve() + return socket + } + + const connecting = peer.connect() + await socketCreated.promise + socket.open() + await helloSent.promise + socket.message({ type: 'host-challenge' }) + await Promise.all([connecting, peer.shutdown()]) + + assert.equal(socket.sent.length, 1) + assert.equal(observations.includes('connected'), false) +}) + +test('shutdown after host acknowledgement prevents a late connected observation', async () => { + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ directorOrigin: undefined, targetOrigin: 'https://cell.test' }), + (type) => observations.push(type) + ) + const socket = fakeHandshakeSocket() + const socketCreated = deferred() + const helloSent = deferred() + const proofSent = deferred() + socket.send = (message) => { + socket.sent.push(message) + if (socket.sent.length === 1) helloSent.resolve() + else proofSent.resolve() + } + peer.createSocket = () => { + socketCreated.resolve() + return socket + } + + const connecting = peer.connect() + await socketCreated.promise + socket.open() + await helloSent.promise + socket.message(validChallenge(peer)) + await proofSent.promise + socket.message({ type: 'host-hello-ack', generation: 1, controlResumeSecret: 'test-secret' }) + await Promise.all([connecting, peer.shutdown()]) + + assert.equal(socket.sent.length, 2) + assert.equal(observations.includes('connected'), false) +}) + +test('shutdown prevents a pending refresh from sending or reporting success', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + const token = deferred() + const tokenStarted = deferred() + global.fetch = () => { + tokenStarted.resolve() + return token.promise + } + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ accessToken: 'test-access-token', directorOrigin: undefined }), + (type) => observations.push(type) + ) + const socket = fakeOpenSocket() + peer.socket = socket + + const refreshing = peer.refresh() + await tokenStarted.promise + const shutdown = peer.shutdown() + token.resolve(response({ relayToken: 'test-relay-token' })) + await Promise.all([refreshing, shutdown]) + + assert.deepEqual(socket.sent, []) + assert.equal(observations.includes('refresh'), false) + assert.equal(observations.includes('refreshError'), false) +}) + +test('shutdown suppresses a late refresh error', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + const token = deferred() + const tokenStarted = deferred() + global.fetch = () => { + tokenStarted.resolve() + return token.promise + } + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ accessToken: 'test-access-token', directorOrigin: undefined }), + (type) => observations.push(type) + ) + peer.socket = fakeOpenSocket() + + const refreshing = peer.refresh() + await tokenStarted.promise + const shutdown = peer.shutdown() + token.reject(new Error('late token failure')) + await Promise.all([refreshing, shutdown]) + + assert.equal(observations.includes('refreshError'), false) +}) + +test('shutdown aborts an HTTP request that would otherwise remain pending', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + const requestStarted = deferred() + let aborted = false + global.fetch = (_url, { signal }) => + new Promise((_resolve, reject) => { + requestStarted.resolve() + signal.addEventListener( + 'abort', + () => { + aborted = true + reject(new Error('request aborted')) + }, + { once: true } + ) + }) + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ accessToken: 'test-access-token', directorOrigin: undefined }), + (type) => observations.push(type) + ) + + const connecting = peer.connect() + await requestStarted.promise + await Promise.all([connecting, peer.shutdown()]) + + assert.equal(aborted, true) + assert.equal(observations.includes('connected'), false) +}) + +test('bounds a pending HTTP request with a classified timeout', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + global.fetch = (_url, { signal }) => + new Promise((_resolve, reject) => { + signal.addEventListener('abort', () => reject(new Error('request aborted')), { once: true }) + }) + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ + accessToken: 'test-access-token', + directorOrigin: undefined, + requestTimeoutMs: 1 + }), + () => undefined + ) + + await assert.rejects(peer.connect(), /relay token exchange timeout/) + await peer.shutdown() +}) + +test('bounds a stalled token response body', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + global.fetch = async (_url, { signal }) => ({ + ok: true, + status: 200, + json: () => + new Promise((_resolve, reject) => { + signal.addEventListener('abort', () => reject(new Error('body aborted')), { once: true }) + }) + }) + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ + accessToken: 'test-access-token', + directorOrigin: undefined, + requestTimeoutMs: 1 + }), + () => undefined + ) + + await assert.rejects(peer.connect(), /relay token exchange timeout/) + await peer.shutdown() +}) + +test('bounds a stalled assignment response body', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + global.fetch = async (_url, { signal }) => ({ + ok: true, + status: 200, + json: () => + new Promise((_resolve, reject) => { + signal.addEventListener('abort', () => reject(new Error('body aborted')), { once: true }) + }) + }) + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ requestTimeoutMs: 1 }), + () => undefined + ) + + await assert.rejects(peer.connect(), /relay assignment timeout/) + await peer.shutdown() +}) + +test('shutdown aborts a stalled successful response body', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + const bodyStarted = deferred() + let bodyAborted = false + global.fetch = async (_url, { signal }) => ({ + ok: true, + status: 200, + json: () => + new Promise((_resolve, reject) => { + bodyStarted.resolve() + signal.addEventListener( + 'abort', + () => { + bodyAborted = true + reject(new Error('body aborted')) + }, + { once: true } + ) + }) + }) + const observations = [] + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ accessToken: 'test-access-token', directorOrigin: undefined }), + (type) => observations.push(type) + ) + + const connecting = peer.connect() + await bodyStarted.promise + await Promise.all([connecting, peer.shutdown()]) + + assert.equal(bodyAborted, true) + assert.equal(observations.includes('connected'), false) +}) + +test('cancels a rejected HTTP response body', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + let canceled = false + global.fetch = async () => ({ + ok: false, + status: 503, + body: { + cancel: async () => { + canceled = true + } + } + }) + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ accessToken: 'test-access-token', directorOrigin: undefined }), + () => undefined + ) + + await assert.rejects(peer.connect(), /relay token exchange failed: 503/) + await peer.shutdown() + assert.equal(canceled, true) +}) + +test('preserves only an exact capacity assignment rejection reason', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + global.fetch = async () => ({ + ok: false, + status: 503, + json: async () => ({ error: 'relay_connection_headroom_exhausted' }) + }) + const peer = new RelayLoadControlPeer(0, peerOptions(), () => undefined) + + await assert.rejects( + peer.connect(), + /relay assignment failed: 503 relay_connection_headroom_exhausted/ + ) + await peer.shutdown() +}) + +test('sends preferred region and preserves the genuine director epoch', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + let assignmentRequest + global.fetch = async (_url, init) => { + assignmentRequest = JSON.parse(init.body) + return response({ cellUrl: 'https://asia-cell.test', assignmentEpoch: 47 }) + } + const peer = new RelayLoadControlPeer( + 0, + peerOptions({ preferredRegion: 'asia-east2' }), + () => undefined + ) + + const assignment = await peer.assignment('relay-token') + + assert.deepEqual(assignmentRequest, { + v: 1, + relayHostId: peer.relayHostId, + preferredRegion: 'asia-east2' + }) + assert.equal(assignment.assignmentEpoch, 47) + await peer.shutdown() +}) + +test('omits preferred region and rejects a fabricated director epoch', async (context) => { + const originalFetch = global.fetch + context.after(() => { + global.fetch = originalFetch + }) + let assignmentRequest + global.fetch = async (_url, init) => { + assignmentRequest = JSON.parse(init.body) + return response({ cellUrl: 'https://cell.test', assignmentEpoch: 0 }) + } + const peer = new RelayLoadControlPeer(0, peerOptions(), () => undefined) + + await assert.rejects(peer.assignment('relay-token'), /assignment response invalid/) + assert.equal('preferredRegion' in assignmentRequest, false) + await peer.shutdown() +}) + +test('rejects ambiguous direct and director assignment modes', () => { + assert.throws( + () => + new RelayLoadControlPeer( + 0, + peerOptions({ targetOrigin: 'https://cell.test' }), + () => undefined + ), + /either directorOrigin or targetOrigin/ + ) +}) + +test('opens a genuine splice and verifies payloads in both directions', async () => { + const observations = [] + const peer = new RelayLoadControlPeer(3, peerOptions(), (type, detail) => { + observations.push({ type, detail }) + }) + const control = fakeOpenSocket() + const phone = fakeHandshakeSocket() + const data = fakeHandshakeSocket() + let dataPayloadBytes = 0 + let observationStarted = false + let sentBeforeObservation = false + let paused = 0 + let resumed = 0 + phone._socket = { + pause: () => paused++, + resume: () => resumed++ + } + control.send = (raw) => { + const message = JSON.parse(raw) + if (message.type === 'invite-create') { + queueMicrotask(() => + control.emit( + 'message', + Buffer.from( + JSON.stringify({ + type: 'invite-created', + reqId: message.reqId, + inviteToken: 'invite-token' + }) + ) + ) + ) + } + } + phone.send = (raw) => { + if (typeof raw === 'string' && raw.startsWith('{') && JSON.parse(raw).type === 'relay-auth') { + queueMicrotask(() => + control.emit( + 'message', + Buffer.from( + JSON.stringify({ + type: 'conn-open', + relayDeviceId: 'load-device-3-0', + connId: 'connection-1', + connTicket: 'connection-ticket' + }) + ) + ) + ) + return + } + queueMicrotask(() => data.emit('message', Buffer.from(raw), false)) + } + data.send = (raw) => { + if (typeof raw === 'string') { + queueMicrotask(() => phone.emit('message', Buffer.from(JSON.stringify({ ok: true })), false)) + return + } + const dataPayload = Buffer.from(raw) + if (!observationStarted) sentBeforeObservation = true + dataPayloadBytes += dataPayload.byteLength + queueMicrotask(() => phone.emit('message', dataPayload, true)) + } + peer.socket = control + peer.generation = 9 + peer.lastAssignment = { cellUrl: 'https://cell.test', assignmentEpoch: 47 } + control.on('message', (raw) => peer.onMessage(control, raw)) + peer.createClientSocket = () => openOnNextTurn(phone) + peer.createHostDataSocket = () => openOnNextTurn(data) + + await peer.openSplice({ + readerMode: 'slow', + readerHoldMs: 1, + streamBytes: 300 * 1024, + frameBytes: 64 * 1024, + observeReaderPressure: async () => { observationStarted = true } + }) + + assert.equal(dataPayloadBytes, 300 * 1024) + assert.equal(sentBeforeObservation, false) + assert.equal(paused, 1) + assert.equal(resumed, 1) + assert.equal(observations.filter(({ type }) => type === 'spliceCompleted').length, 1) + assert.equal(observations.some(({ type }) => type === 'spliceFailed'), false) + await peer.shutdown() + assert.deepEqual(observations.at(-1), { + type: 'shutdown', + detail: { + index: 3, + activeControls: 0, + activeSpliceSockets: 0, + inFlightOperations: 0, + refreshTimerActive: false + } + }) +}) + +test('opens invitation leases and preserves exact capacity errors', async () => { + const peer = new RelayLoadControlPeer(4, peerOptions(), () => undefined) + const control = fakeOpenSocket() + peer.socket = control + control.on('message', (raw) => peer.onMessage(control, raw)) + let calls = 0 + control.send = (raw) => { + const request = JSON.parse(raw) + calls++ + queueMicrotask(() => control.emit('message', Buffer.from(JSON.stringify( + calls === 1 + ? { + type: 'invite-created', reqId: request.reqId, + inviteToken: 'invite-token', expiresAt: Date.now() + 60_000 + } + : { type: 'control-error', reqId: request.reqId, code: 'relay_capacity_exhausted' } + )))) + } + + await peer.openInviteOffer() + await assert.rejects(peer.openInviteOffer(), /invite offer failed: relay_capacity_exhausted/) + await peer.shutdown() +}) + +test('can request the same assignment with an explicit replacement preference', async (context) => { + const originalFetch = global.fetch + context.after(() => { global.fetch = originalFetch }) + const requests = [] + global.fetch = async (url, init) => { + if (String(url).endsWith('/v1/assign')) { + requests.push(JSON.parse(init.body)) + return response({ cellUrl: 'https://asia-cell.test', assignmentEpoch: 3 }) + } + return response({ relayToken: 'relay-token' }) + } + const peer = new RelayLoadControlPeer( + 5, + peerOptions({ accessToken: 'access-token', preferredRegion: 'asia-east2' }), + () => undefined + ) + + assert.equal((await peer.requestAssignment('us-central1')).cellUrl, 'https://asia-cell.test') + assert.equal(requests[0].preferredRegion, 'us-central1') + await peer.shutdown() +}) + +test('uses a refreshable workflow access-token provider', async (context) => { + const originalFetch = global.fetch + context.after(() => { global.fetch = originalFetch }) + let providerCalls = 0 + let authorization + global.fetch = async (url, init) => { + if (String(url).endsWith('/v1/desktop/auth/relay-token')) { + authorization = init.headers.authorization + return response({ relayToken: 'relay-token' }) + } + return response({ cellUrl: 'https://cell.test', assignmentEpoch: 1 }) + } + const peer = new RelayLoadControlPeer(6, peerOptions({ + accessTokenProvider: async () => { + providerCalls++ + return 'refreshed-access-token' + } + }), () => undefined) + + await peer.requestAssignment('asia-east2') + assert.equal(providerCalls, 1) + assert.equal(authorization, 'Bearer refreshed-access-token') + await peer.shutdown() +}) + +test('accepts a forced close only when the responsive splice leg receives 4429', async () => { + const observations = [] + const peer = new RelayLoadControlPeer(7, peerOptions(), (type, detail) => { + observations.push({ type, detail }) + }) + const control = fakeOpenSocket() + const phone = fakeHandshakeSocket() + const data = fakeHandshakeSocket() + let streamStarted + const streamStartedPromise = new Promise((resolve) => { streamStarted = resolve }) + phone._socket = { pause: () => undefined, resume: () => undefined } + control.send = (raw) => { + const message = JSON.parse(raw) + queueMicrotask(() => control.emit('message', Buffer.from(JSON.stringify({ + type: 'invite-created', reqId: message.reqId, inviteToken: 'invite-token' + })))) + } + phone.send = (raw) => { + if (typeof raw === 'string' && raw.startsWith('{')) { + queueMicrotask(() => control.emit('message', Buffer.from(JSON.stringify({ + type: 'conn-open', relayDeviceId: 'load-device-7-0', + connId: 'connection-7', connTicket: 'connection-ticket' + })))) + } + } + data.send = (raw) => { + if (typeof raw === 'string') { + queueMicrotask(() => phone.emit('message', Buffer.from(JSON.stringify({ ok: true })), false)) + } else { + streamStarted() + } + } + peer.socket = control + peer.generation = 11 + peer.lastAssignment = { cellUrl: 'https://cell.test', assignmentEpoch: 9 } + control.on('message', (raw) => peer.onMessage(control, raw)) + peer.createClientSocket = () => openOnNextTurn(phone) + peer.createHostDataSocket = () => openOnNextTurn(data) + + await peer.openSplice({ + readerMode: 'wedged', + readerHoldMs: 10_001, + streamBytes: 300 * 1024, + frameBytes: 64 * 1024, + observeReaderPressure: async () => { + await streamStartedPromise + phone.close(1006) + data.close(4429, 'wedged relay link') + }, + readerDelay: async () => undefined + }) + + assert.equal(observations.filter(({ type }) => type === 'spliceWedged').length, 1) + assert.equal(observations.find(({ type }) => type === 'spliceWedged').detail.code, 4429) + await peer.shutdown() +}) + +test('requires one 4429 and rejects unrelated wedged close codes', () => { + assert.equal(relayLoadWedgedCloseAccepted([4429, 4429]), true) + assert.equal(relayLoadWedgedCloseAccepted([1006, 4429]), true) + assert.equal(relayLoadWedgedCloseAccepted([4429, 1006]), false) + assert.equal(relayLoadWedgedCloseAccepted([1006, 1006]), false) + assert.equal(relayLoadWedgedCloseAccepted([1000, 4429]), false) + assert.equal(relayLoadWedgedCloseAccepted([4429]), false) + assert.equal(relayLoadWedgedCloseAccepted([1006, 4429, 4429]), false) +}) + +test('streams reader frames with bounded send-side backpressure', async () => { + const peer = new RelayLoadControlPeer(9, peerOptions(), () => undefined) + let inFlight = 0 + let peakInFlight = 0 + let sentBytes = 0 + const socket = { + bufferedAmount: 0, + send(payload, callback) { + inFlight++ + peakInFlight = Math.max(peakInFlight, inFlight) + sentBytes += payload.byteLength + this.bufferedAmount = payload.byteLength + queueMicrotask(() => { + this.bufferedAmount = 0 + inFlight-- + callback() + }) + } + } + + const result = await peer.sendReaderStream( + socket, + 0, + 1024 * 1024, + 64 * 1024, + async () => undefined + ) + + assert.equal(result.bytes, 1024 * 1024) + assert.equal(sentBytes, 1024 * 1024) + assert.equal(peakInFlight, 1) + await peer.shutdown() +}) + +test('reports a bidirectional splice payload mismatch', async () => { + const observations = [] + const peer = new RelayLoadControlPeer(2, peerOptions(), (type) => observations.push(type)) + const control = fakeOpenSocket() + const phone = fakeHandshakeSocket() + const data = fakeHandshakeSocket() + control.send = (raw) => { + const message = JSON.parse(raw) + queueMicrotask(() => + control.emit( + 'message', + Buffer.from( + JSON.stringify({ type: 'invite-created', reqId: message.reqId, inviteToken: 'token' }) + ) + ) + ) + } + phone.send = (raw) => { + if (typeof raw === 'string' && raw.startsWith('{') && JSON.parse(raw).type === 'relay-auth') { + queueMicrotask(() => + control.emit( + 'message', + Buffer.from( + JSON.stringify({ + type: 'conn-open', + relayDeviceId: 'load-device-2-0', + connId: 'connection-2', + connTicket: 'ticket' + }) + ) + ) + ) + } + } + data.send = (raw) => { + if (typeof raw === 'string') { + queueMicrotask(() => phone.emit('message', Buffer.from(JSON.stringify({ ok: true })), false)) + } else { + queueMicrotask(() => phone.emit('message', Buffer.alloc(Buffer.from(raw).byteLength), true)) + } + } + peer.socket = control + peer.generation = 4 + peer.lastAssignment = { cellUrl: 'https://cell.test', assignmentEpoch: 2 } + control.on('message', (raw) => peer.onMessage(control, raw)) + peer.createClientSocket = () => openOnNextTurn(phone) + peer.createHostDataSocket = () => openOnNextTurn(data) + + await assert.rejects(peer.openSplice(), /changed host-to-client splice payload/) + assert.equal(observations.includes('spliceFailed'), true) + await peer.shutdown() +}) + +test('shutdown closes both splice legs and waits for the in-flight splice', async () => { + const spliceOpened = deferred() + const observations = [] + const peer = new RelayLoadControlPeer(5, peerOptions(), (type, detail) => { + observations.push({ type, detail }) + if (type === 'spliceOpened') spliceOpened.resolve() + }) + const control = fakeOpenSocket() + const phone = fakeHandshakeSocket() + const data = fakeHandshakeSocket() + control.send = (raw) => { + const message = JSON.parse(raw) + queueMicrotask(() => + control.emit( + 'message', + Buffer.from( + JSON.stringify({ type: 'invite-created', reqId: message.reqId, inviteToken: 'token' }) + ) + ) + ) + } + phone.send = (raw) => { + if (typeof raw === 'string' && raw.startsWith('{')) { + queueMicrotask(() => + control.emit( + 'message', + Buffer.from( + JSON.stringify({ + type: 'conn-open', + relayDeviceId: 'load-device-5-0', + connId: 'connection-5', + connTicket: 'ticket' + }) + ) + ) + ) + } else { + queueMicrotask(() => data.emit('message', Buffer.from(raw), false)) + } + } + data.send = (raw) => { + if (typeof raw === 'string') { + queueMicrotask(() => phone.emit('message', Buffer.from(JSON.stringify({ ok: true })), false)) + } else { + queueMicrotask(() => phone.emit('message', Buffer.from(raw), true)) + } + } + peer.socket = control + peer.generation = 8 + peer.lastAssignment = { cellUrl: 'https://cell.test', assignmentEpoch: 3 } + control.on('message', (raw) => peer.onMessage(control, raw)) + peer.createClientSocket = () => openOnNextTurn(phone) + peer.createHostDataSocket = () => openOnNextTurn(data) + + const splice = peer.openSplice({ holdMs: 60_000 }) + await spliceOpened.promise + await Promise.all([splice, peer.shutdown()]) + + assert.equal(phone.readyState, phone.CLOSED) + assert.equal(data.readyState, data.CLOSED) + assert.equal(peer.inFlight.size, 0) + assert.equal(observations.at(-1).type, 'shutdown') + assert.equal(observations.at(-1).detail.activeSpliceSockets, 0) +}) diff --git a/cloud/dev/scripts/relay-load-director-capacity-gate.mjs b/cloud/dev/scripts/relay-load-director-capacity-gate.mjs new file mode 100644 index 00000000000..a1701d1a17e --- /dev/null +++ b/cloud/dev/scripts/relay-load-director-capacity-gate.mjs @@ -0,0 +1,147 @@ +import { setTimeout as delayDefault } from 'node:timers/promises' + +function integer(value) { + return Number.isSafeInteger(value) && value >= 0 ? value : undefined +} + +export function assertRelayLoadDirectorCapacityToken(config, now = Date.now, timeoutMs = 0) { + if (!config.adminToken || config.adminToken.length > 8_192) { + throw new Error('director capacity identity token is unavailable') + } + const origin = new URL(config.directorOrigin) + if (origin.protocol !== 'https:' || origin.origin !== config.directorOrigin) { + throw new Error('director capacity origin must be canonical HTTPS') + } + let claims + try { + const parts = config.adminToken.split('.') + if (parts.length !== 3) throw new Error('invalid token shape') + claims = JSON.parse(Buffer.from(parts[1], 'base64url').toString('utf8')) + } catch { + throw new Error('director capacity identity token is invalid') + } + const expectedAudience = new URL('/v1/admin/drain', origin).toString() + const audiences = Array.isArray(claims.aud) ? claims.aud : [claims.aud] + const expiresAt = integer(claims.exp) + if ( + !audiences.includes(expectedAudience) || + typeof claims.email !== 'string' || + claims.email.length === 0 || + claims.email_verified !== true || + expiresAt === undefined || + expiresAt * 1_000 <= now() + timeoutMs + ) { + throw new Error('director capacity identity token is not bound to this proof') + } +} + +function matchingHeartbeat(status, config) { + const capacity = status?.connectionCapacity + const runtime = status?.runtime + const heartbeatAt = integer(runtime?.lastHeartbeatAt) + const matches = + status?.cellId === config.cellId && + status?.admissionState === 'general' && + runtime?.ready === true && + runtime?.heartbeatFresh === true && + capacity?.heartbeatFresh === true && + integer(capacity?.hardCap) === config.hardCap && + integer(capacity?.unobservedBound) === config.unobservedBound && + integer(capacity?.normalAdmissionPause) === config.requiredConnections && + integer(capacity?.observedConnections) === config.requiredConnections && + integer(capacity?.enforcedConnectionUnits) === config.requiredConnections && + integer(capacity?.inFlightConnections) === 0 && + integer(capacity?.reservedConnectionUnits) === 0 && + integer(capacity?.pendingControlReservations) === 0 && + heartbeatAt !== undefined + return matches ? heartbeatAt : undefined +} + +async function cellStatus(fetchImpl, config) { + const response = await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, { + method: 'POST', + headers: { + authorization: `Bearer ${config.adminToken}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ v: 1, cellId: config.cellId }), + signal: AbortSignal.timeout(30_000) + }) + if (response.status === 401 || response.status === 403) { + throw new Error('director capacity identity was rejected') + } + if (!response.ok) { + await response.arrayBuffer().catch(() => undefined) + return undefined + } + const result = await response.json().catch(() => undefined) + if (!result?.status) throw new Error('director capacity status is invalid') + return result.status +} + +export async function waitForRelayLoadDirectorCapacity(config, overrides = {}) { + const fetchImpl = overrides.fetch ?? fetch + const delay = overrides.delay ?? delayDefault + const now = overrides.now ?? Date.now + const timeoutMs = overrides.timeoutMs ?? 120_000 + const pollMs = overrides.pollMs ?? 1_000 + assertRelayLoadDirectorCapacityToken(config, now, timeoutMs) + const deadline = now() + timeoutMs + let baselineHeartbeatAt = config.baselineHeartbeatAt + let previousHeartbeatAt + let matchingSamples = 0 + const requiredSamples = config.requiredSamples ?? 2 + for (;;) { + const status = await cellStatus(fetchImpl, config) + const currentHeartbeatAt = integer(status?.runtime?.lastHeartbeatAt) + if (baselineHeartbeatAt === undefined && currentHeartbeatAt !== undefined) { + baselineHeartbeatAt = currentHeartbeatAt + } + const heartbeatAt = status ? matchingHeartbeat(status, config) : undefined + if (heartbeatAt !== undefined && heartbeatAt > baselineHeartbeatAt) { + if (previousHeartbeatAt === undefined || heartbeatAt > previousHeartbeatAt) { + previousHeartbeatAt = heartbeatAt + matchingSamples++ + if (matchingSamples === requiredSamples) return { heartbeatAt } + } + } else { + previousHeartbeatAt = undefined + matchingSamples = 0 + } + if (now() >= deadline) throw new Error('director capacity did not converge after recovery') + await delay(pollMs) + } +} + +function matchingRequestUnits(status, config) { + return ( + status?.cellId === config.cellId && + status?.admissionState === 'general' && + status?.capacityRequests === config.capacityRequests && + status?.reservedRequests === config.expectedRequestUnits && + status?.activityRequestUnits === config.expectedRequestUnits && + status?.activityLeases === config.expectedActivityLeases && + status?.runtime?.observedRequests === config.expectedRequestUnits && + status?.runtime?.ready === true && + status?.runtime?.heartbeatFresh === true + ) +} + +export async function waitForRelayLoadRequestUnits(config, overrides = {}) { + const fetchImpl = overrides.fetch ?? fetch + const delay = overrides.delay ?? delayDefault + const now = overrides.now ?? Date.now + const timeoutMs = overrides.timeoutMs ?? config.timeoutMs ?? 120_000 + const pollMs = overrides.pollMs ?? 1_000 + const requiredSamples = overrides.requiredSamples ?? 2 + assertRelayLoadDirectorCapacityToken(config, now, timeoutMs) + const deadline = now() + timeoutMs + let matches = 0 + for (;;) { + const status = await cellStatus(fetchImpl, config) + matches = matchingRequestUnits(status, config) ? matches + 1 : 0 + if (matches === requiredSamples) return + if (now() >= deadline) throw new Error('Relay request-unit accounting did not converge') + await delay(pollMs) + } +} diff --git a/cloud/dev/scripts/relay-load-director-capacity-gate.test.mjs b/cloud/dev/scripts/relay-load-director-capacity-gate.test.mjs new file mode 100644 index 00000000000..fe0a2813b80 --- /dev/null +++ b/cloud/dev/scripts/relay-load-director-capacity-gate.test.mjs @@ -0,0 +1,260 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + assertRelayLoadDirectorCapacityToken, + waitForRelayLoadDirectorCapacity, + waitForRelayLoadRequestUnits +} from './relay-load-director-capacity-gate.mjs' + +function token(claims) { + const encode = (value) => Buffer.from(JSON.stringify(value)).toString('base64url') + return `${encode({ alg: 'none' })}.${encode(claims)}.signature` +} + +const config = { + directorOrigin: 'https://relay-staging.example.com', + adminToken: token({ + aud: 'https://relay-staging.example.com/v1/admin/drain', + email: 'capacity@example.com', + email_verified: true, + exp: 4_000_000_000 + }), + cellId: 'staging-gce-c3', + hardCap: 1_000, + unobservedBound: 60, + requiredConnections: 840 +} + +function status(lastHeartbeatAt, overrides = {}) { + return { + cellId: config.cellId, + admissionState: 'general', + runtime: { ready: true, heartbeatFresh: true, lastHeartbeatAt, observedRequests: 0 }, + connectionCapacity: { + hardCap: 1_000, + unobservedBound: 60, + normalAdmissionPause: 840, + observedConnections: 840, + enforcedConnectionUnits: 840, + inFlightConnections: 0, + reservedConnectionUnits: 0, + pendingControlReservations: 0, + heartbeatFresh: true, + ...overrides + } + } +} + +function response(value, responseStatus = 200) { + return { + ok: responseStatus >= 200 && responseStatus < 300, + status: responseStatus, + json: async () => value, + arrayBuffer: async () => new ArrayBuffer(0) + } +} + +test('requires two advancing exact director capacity heartbeats', async () => { + const heartbeats = [101, 116, 131] + let calls = 0 + const result = await waitForRelayLoadDirectorCapacity(config, { + fetch: async () => response({ status: status(heartbeats[calls++]) }), + delay: async () => undefined + }) + assert.equal(calls, 3) + assert.deepEqual(result, { heartbeatAt: 131 }) +}) + +test('resets after a newer heartbeat undercounts recovered controls', async () => { + const samples = [ + status(101), + status(116), + status(131, { observedConnections: 839 }), + status(146), + status(161) + ] + let calls = 0 + await waitForRelayLoadDirectorCapacity(config, { + fetch: async () => response({ status: samples[calls++] }), + delay: async () => undefined + }) + assert.equal(calls, 5) +}) + +test('fails closed when exact advancing telemetry never converges', async () => { + let elapsed = 0 + await assert.rejects( + waitForRelayLoadDirectorCapacity(config, { + fetch: async () => response({ status: status(101) }), + delay: async (milliseconds) => { + elapsed += milliseconds + }, + now: () => elapsed, + timeoutMs: 2_000, + pollMs: 1_000 + }), + /did not converge/ + ) +}) + +test('rejects an unauthorized capacity identity without retrying', async () => { + let calls = 0 + await assert.rejects( + waitForRelayLoadDirectorCapacity(config, { + fetch: async () => { + calls++ + return response({}, 401) + }, + delay: async () => undefined + }), + /identity was rejected/ + ) + assert.equal(calls, 1) +}) + +test('binds the admin token to the canonical director audience', async () => { + await assert.rejects( + waitForRelayLoadDirectorCapacity( + { + ...config, + directorOrigin: 'https://relay-staging.example.com/path' + }, + { fetch: async () => response({ status: status(101) }), now: () => 0 } + ), + /canonical HTTPS/ + ) + await assert.rejects( + waitForRelayLoadDirectorCapacity( + { + ...config, + adminToken: token({ + aud: 'https://other.example.com/v1/admin/drain', + email: 'capacity@example.com', + email_verified: true, + exp: 4_000_000_000 + }) + }, + { fetch: async () => response({ status: status(101) }), now: () => 0 } + ), + /not bound/ + ) +}) + +test('rejects a missing or malformed admin token during startup preflight', () => { + assert.throws( + () => assertRelayLoadDirectorCapacityToken({ ...config, adminToken: undefined }, () => 0), + /unavailable/ + ) + assert.throws( + () => assertRelayLoadDirectorCapacityToken({ ...config, adminToken: 'not-a-jwt' }, () => 0), + /invalid/ + ) + assert.throws( + () => + assertRelayLoadDirectorCapacityToken( + { + ...config, + adminToken: token({ + aud: 'https://relay-staging.example.com/v1/admin/drain', + exp: 4_000_000_000 + }) + }, + () => 0 + ), + /not bound/ + ) +}) + +test('supports one newer exact post-probe heartbeat', async () => { + const heartbeats = [146, 161] + let calls = 0 + const result = await waitForRelayLoadDirectorCapacity( + { ...config, requiredSamples: 1 }, + { + fetch: async () => response({ status: status(heartbeats[calls++]) }), + delay: async () => undefined, + now: () => 0 + } + ) + assert.equal(calls, 2) + assert.deepEqual(result, { heartbeatAt: 161 }) +}) + +test('requires consecutive exact request-unit accounting samples', async () => { + const requestConfig = { + ...config, + capacityRequests: 6_000, + expectedRequestUnits: 6_000, + expectedActivityLeases: 6_000 + } + const samples = [5_999, 6_000, 6_000] + let calls = 0 + await waitForRelayLoadRequestUnits(requestConfig, { + fetch: async () => response({ + status: { + ...status(100), + runtime: { + ...status(100).runtime, + observedRequests: samples[calls] + }, + capacityRequests: 6_000, + reservedRequests: samples[calls], + activityRequestUnits: samples[calls], + activityLeases: samples[calls++] + } + }), + delay: async () => undefined, + now: () => 0 + }) + assert.equal(calls, 3) +}) + +test('requires the cell runtime to observe every request unit', async () => { + let elapsed = 0 + await assert.rejects(waitForRelayLoadRequestUnits({ + ...config, + capacityRequests: 6_000, + expectedRequestUnits: 6_000, + expectedActivityLeases: 6_000, + timeoutMs: 1_000 + }, { + fetch: async () => response({ + status: { + ...status(100), + runtime: { ...status(100).runtime, observedRequests: 5_999 }, + capacityRequests: 6_000, + reservedRequests: 6_000, + activityRequestUnits: 6_000, + activityLeases: 6_000 + } + }), + delay: async (milliseconds) => { elapsed += milliseconds }, + now: () => elapsed, + pollMs: 1_000 + }), /did not converge/) +}) + +test('fails closed when request-unit accounting does not clean up', async () => { + let elapsed = 0 + await assert.rejects(waitForRelayLoadRequestUnits({ + ...config, + capacityRequests: 6_000, + expectedRequestUnits: 0, + expectedActivityLeases: 0, + timeoutMs: 2_000 + }, { + fetch: async () => response({ + status: { + ...status(100), + runtime: { ...status(100).runtime, observedRequests: 1 }, + capacityRequests: 6_000, + reservedRequests: 1, + activityRequestUnits: 1, + activityLeases: 1 + } + }), + delay: async (milliseconds) => { elapsed += milliseconds }, + now: () => elapsed, + pollMs: 1_000 + }), /did not converge/) +}) diff --git a/cloud/dev/scripts/relay-load-model.mjs b/cloud/dev/scripts/relay-load-model.mjs new file mode 100644 index 00000000000..77de00422c6 --- /dev/null +++ b/cloud/dev/scripts/relay-load-model.mjs @@ -0,0 +1,76 @@ +const HEARTBEAT_INTERVAL_MS = 15_000 +const REFRESH_MIN_MS = 180_000 +const REFRESH_MAX_MS = 240_000 + +function mix32(value) { + let mixed = value >>> 0 + mixed = Math.imul(mixed ^ (mixed >>> 16), 0x21f0aaad) + mixed = Math.imul(mixed ^ (mixed >>> 15), 0x735a2d97) + return (mixed ^ (mixed >>> 15)) >>> 0 +} + +function fraction(seed, index, stream) { + return mix32(seed ^ Math.imul(index + 1, 0x9e3779b1) ^ stream) / 0x1_0000_0000 +} + +export function controlPhase(controlIndex, seed = 0x4f524341) { + const refreshIntervalMs = Math.round( + REFRESH_MIN_MS + fraction(seed, controlIndex, 2) * (REFRESH_MAX_MS - REFRESH_MIN_MS) + ) + return { + heartbeatOffsetMs: Math.floor(fraction(seed, controlIndex, 1) * HEARTBEAT_INTERVAL_MS), + refreshIntervalMs, + refreshOffsetMs: Math.floor(fraction(seed, controlIndex, 3) * refreshIntervalMs), + reconnectJitterMs: Math.floor(fraction(seed, controlIndex, 4) * 30_000) + } +} + +export function modeledRelayLoad(controlCount, durationMs = 15 * 60_000, seed) { + if (!Number.isInteger(controlCount) || controlCount < 1) throw new Error('controlCount must be positive') + const bins = Array.from({ length: Math.ceil(durationMs / 1000) }, () => ({ pings: 0, refreshes: 0 })) + for (let controlIndex = 0; controlIndex < controlCount; controlIndex++) { + const phase = controlPhase(controlIndex, seed) + for (let at = phase.heartbeatOffsetMs; at < durationMs; at += HEARTBEAT_INTERVAL_MS) { + bins[Math.floor(at / 1000)].pings++ + } + for (let at = phase.refreshOffsetMs; at < durationMs; at += phase.refreshIntervalMs) { + bins[Math.floor(at / 1000)].refreshes++ + } + } + const totals = bins.reduce( + (result, bin) => ({ + pings: result.pings + bin.pings, + refreshes: result.refreshes + bin.refreshes + }), + { pings: 0, refreshes: 0 } + ) + const durationSeconds = durationMs / 1000 + return { + controlCount, + durationMs, + expectedPingRate: controlCount / (HEARTBEAT_INTERVAL_MS / 1000), + expectedRefreshRate: controlCount / ((REFRESH_MIN_MS + REFRESH_MAX_MS) / 2 / 1000), + observedPingRate: totals.pings / durationSeconds, + observedRefreshRate: totals.refreshes / durationSeconds, + maxPingBurst: Math.max(...bins.map(({ pings }) => pings)), + maxRefreshBurst: Math.max(...bins.map(({ refreshes }) => refreshes)) + } +} + +export function assertSpreadModel(model) { + const pingTolerance = model.expectedPingRate * 0.03 + 1 + const refreshTolerance = model.expectedRefreshRate * 0.08 + 1 + if (Math.abs(model.observedPingRate - model.expectedPingRate) > pingTolerance) { + throw new Error('modeled heartbeat rate diverged from the 15-second contract') + } + if (Math.abs(model.observedRefreshRate - model.expectedRefreshRate) > refreshTolerance) { + throw new Error('modeled token refresh rate diverged from the 180-240 second contract') + } + if (model.maxPingBurst > model.expectedPingRate * 1.3 + 5) { + throw new Error('heartbeat phase spreading produced a reconnect cliff') + } + // One-second bins have Poisson-sized tails even with uniform phase spreading. + if (model.maxRefreshBurst > model.expectedRefreshRate * 1.75 + 5) { + throw new Error('refresh phase spreading produced an auth herd') + } +} diff --git a/cloud/dev/scripts/relay-load-model.test.mjs b/cloud/dev/scripts/relay-load-model.test.mjs new file mode 100644 index 00000000000..e38548efe3b --- /dev/null +++ b/cloud/dev/scripts/relay-load-model.test.mjs @@ -0,0 +1,25 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { assertSpreadModel, controlPhase, modeledRelayLoad } from './relay-load-model.mjs' + +test('phases are deterministic, bounded, and separated by stream', () => { + assert.deepEqual(controlPhase(42), controlPhase(42)) + assert.notDeepEqual(controlPhase(42), controlPhase(43)) + const phase = controlPhase(42) + assert.ok(phase.heartbeatOffsetMs >= 0 && phase.heartbeatOffsetMs < 15_000) + assert.ok(phase.refreshIntervalMs >= 180_000 && phase.refreshIntervalMs <= 240_000) + assert.ok(phase.refreshOffsetMs >= 0 && phase.refreshOffsetMs < phase.refreshIntervalMs) + assert.ok(phase.reconnectJitterMs >= 0 && phase.reconnectJitterMs < 30_000) +}) + +for (const [controls, expectedPings, expectedRefreshes] of [ + [4_000, 267, 19], + [10_000, 667, 48] +]) { + test(`${controls} modeled controls spread heartbeat and token refresh load`, () => { + const model = modeledRelayLoad(controls) + assert.equal(Math.round(model.expectedPingRate), expectedPings) + assert.equal(Math.round(model.expectedRefreshRate), expectedRefreshes) + assert.doesNotThrow(() => assertSpreadModel(model)) + }) +} diff --git a/cloud/dev/scripts/relay-load-phase-barrier.mjs b/cloud/dev/scripts/relay-load-phase-barrier.mjs new file mode 100644 index 00000000000..03e0e800074 --- /dev/null +++ b/cloud/dev/scripts/relay-load-phase-barrier.mjs @@ -0,0 +1,37 @@ +import { access, mkdir, open } from 'node:fs/promises' +import { join } from 'node:path' +import { setTimeout as delayDefault } from 'node:timers/promises' + +export async function waitForRelayLoadPhaseBarrier(config, overrides = {}) { + const delay = overrides.delay ?? delayDefault + const now = overrides.now ?? Date.now + const timeoutMs = overrides.timeoutMs ?? config.timeoutMs + if ( + typeof config.directory !== 'string' || config.directory.length === 0 || + !Number.isSafeInteger(config.shardCount) || config.shardCount < 2 || + !Number.isSafeInteger(config.shardIndex) || config.shardIndex < 0 || + config.shardIndex >= config.shardCount || + !Number.isSafeInteger(timeoutMs) || timeoutMs < 1 + ) throw new Error('invalid Relay load phase barrier') + + await mkdir(config.directory, { recursive: true }) + const marker = join(config.directory, `${config.shardIndex}.ready`) + const handle = await open(marker, 'wx') + await handle.close() + const deadline = now() + timeoutMs + for (;;) { + const ready = await Promise.all( + Array.from({ length: config.shardCount }, async (_, index) => { + try { + await access(join(config.directory, `${index}.ready`)) + return true + } catch { + return false + } + }) + ) + if (ready.every(Boolean)) return + if (now() >= deadline) throw new Error('Relay load phase barrier timed out') + await delay(100) + } +} diff --git a/cloud/dev/scripts/relay-load-phase-barrier.test.mjs b/cloud/dev/scripts/relay-load-phase-barrier.test.mjs new file mode 100644 index 00000000000..2fc16d3650a --- /dev/null +++ b/cloud/dev/scripts/relay-load-phase-barrier.test.mjs @@ -0,0 +1,65 @@ +import assert from 'node:assert/strict' +import { mkdtemp, readFile, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import test from 'node:test' +import { waitForRelayLoadPhaseBarrier } from './relay-load-phase-barrier.mjs' + +const loadHarness = await readFile(new URL('./load-relay-controls.mjs', import.meta.url), 'utf8') + +test('releases every shard only after all readiness markers exist', async () => { + const directory = await mkdtemp(join(tmpdir(), 'relay-load-barrier-')) + try { + let firstResolved = false + const first = waitForRelayLoadPhaseBarrier({ + directory, shardCount: 2, shardIndex: 0, timeoutMs: 1_000 + }).then(() => { firstResolved = true }) + await new Promise((resolve) => setTimeout(resolve, 20)) + assert.equal(firstResolved, false) + await Promise.all([ + first, + waitForRelayLoadPhaseBarrier({ + directory, shardCount: 2, shardIndex: 1, timeoutMs: 1_000 + }) + ]) + assert.equal(firstResolved, true) + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) + +test('fails closed on a duplicate shard or incomplete barrier', async () => { + const directory = await mkdtemp(join(tmpdir(), 'relay-load-barrier-')) + try { + const nowValues = [0, 2] + await assert.rejects( + waitForRelayLoadPhaseBarrier( + { directory, shardCount: 2, shardIndex: 0, timeoutMs: 1 }, + { now: () => nowValues.shift() ?? 2, delay: async () => undefined } + ), + /timed out/ + ) + await assert.rejects( + waitForRelayLoadPhaseBarrier({ + directory, shardCount: 2, shardIndex: 0, timeoutMs: 1 + }), + /EEXIST/ + ) + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) + +test('synchronizes splice ramps after every shard finishes reader baselines', () => { + assert.match( + loadHarness, + /createRelayLoadReaderEvidence[\s\S]*?phaseBarrierDir\}-splices[\s\S]*?splicePromises/ + ) +}) + +test('budgets both shard barriers and the splice ramp in token lifetime', () => { + assert.match( + loadHarness, + /phaseBarrierDir \? 2 \* config\.phaseBarrierTimeoutMs : 0[\s\S]*?config\.spliceRampMs/ + ) +}) diff --git a/cloud/dev/scripts/relay-load-placement-boundary.mjs b/cloud/dev/scripts/relay-load-placement-boundary.mjs new file mode 100644 index 00000000000..84f19b1dc80 --- /dev/null +++ b/cloud/dev/scripts/relay-load-placement-boundary.mjs @@ -0,0 +1,29 @@ +export async function proveRelayLoadPlacementBoundary({ peer, failureReason }) { + let connected = false + try { + await peer.connect() + connected = true + } catch (error) { + const reason = failureReason(error) + if (reason !== 'assignment_capacity_exhausted') { + throw new Error(`placement overflow was not rejected: ${reason}`) + } + return reason + } finally { + await peer.shutdown() + } + if (connected) throw new Error('placement overflow unexpectedly connected') +} + +export async function proveRelayLoadRegionalFallback({ peer, blockedOrigin }) { + try { + await peer.connect() + const assignedOrigin = peer.assignedCellUrl() + if (!assignedOrigin || assignedOrigin === blockedOrigin) { + throw new Error('regional fallback did not leave the full preferred cell') + } + return true + } finally { + await peer.shutdown() + } +} diff --git a/cloud/dev/scripts/relay-load-placement-boundary.test.mjs b/cloud/dev/scripts/relay-load-placement-boundary.test.mjs new file mode 100644 index 00000000000..89dd338816f --- /dev/null +++ b/cloud/dev/scripts/relay-load-placement-boundary.test.mjs @@ -0,0 +1,53 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + proveRelayLoadPlacementBoundary, + proveRelayLoadRegionalFallback +} from './relay-load-placement-boundary.mjs' + +function peer(connect) { + return { connect, shutdown: async () => undefined } +} + +test('requires the next fresh placement to receive HTTP 503', async () => { + assert.equal( + await proveRelayLoadPlacementBoundary({ + peer: peer(async () => { + throw new Error('relay assignment failed: 503 relay_connection_headroom_exhausted') + }), + failureReason: (error) => + error.message === 'relay assignment failed: 503 relay_connection_headroom_exhausted' + ? 'assignment_capacity_exhausted' + : 'unknown' + }), + 'assignment_capacity_exhausted' + ) + await assert.rejects( + proveRelayLoadPlacementBoundary({ + peer: peer(async () => undefined), + failureReason: () => 'unknown' + }), + /placement overflow unexpectedly connected/ + ) +}) + +test('requires a preferred-region fallback to leave the full cell', async () => { + let shutdowns = 0 + assert.equal(await proveRelayLoadRegionalFallback({ + peer: { + connect: async () => undefined, + assignedCellUrl: () => 'https://c3.relay-staging.onorca.dev', + shutdown: async () => { shutdowns++ } + }, + blockedOrigin: 'https://c4.relay-staging.onorca.dev' + }), true) + assert.equal(shutdowns, 1) + await assert.rejects(proveRelayLoadRegionalFallback({ + peer: { + connect: async () => undefined, + assignedCellUrl: () => 'https://c4.relay-staging.onorca.dev', + shutdown: async () => undefined + }, + blockedOrigin: 'https://c4.relay-staging.onorca.dev' + }), /did not leave/) +}) diff --git a/cloud/dev/scripts/relay-load-profile.mjs b/cloud/dev/scripts/relay-load-profile.mjs new file mode 100644 index 00000000000..0a0552fc029 --- /dev/null +++ b/cloud/dev/scripts/relay-load-profile.mjs @@ -0,0 +1,404 @@ +export const RELAY_CONTROL_LEASE_HORIZON_SECONDS = 105 +export const RELAY_LOAD_SPLICE_HIGH_WATER_BYTES = 256 * 1024 +export const RELAY_LOAD_SPLICE_WEDGED_TIMEOUT_MS = 10_000 +export const RELAY_LOAD_MAX_AGGREGATE_READER_SPLICES = 16 +export const RELAY_LOAD_MAX_AGGREGATE_READER_BYTES = 64 * 1024 * 1024 + +const DEFAULT_SLOW_READER_STREAM_BYTES = 1024 * 1024 +const DEFAULT_WEDGED_READER_STREAM_BYTES = 8 * 1024 * 1024 +const DEFAULT_READER_FRAME_BYTES = 64 * 1024 + +export function parseRelayLoadArguments(argv) { + const values = new Map() + const flags = new Set() + for (let index = 0; index < argv.length; index++) { + const argument = argv[index] + if (argument === '--') continue + if ( + [ + '--allow-partial', + '--allow-planned-transition-retries', + '--skip-rebind-overflow-check' + ].includes(argument) + ) { + flags.add(argument) + continue + } + if (!argument.startsWith('--') || index + 1 >= argv.length) { + throw new Error(`invalid argument: ${argument}`) + } + values.set(argument, argv[++index]) + } + const integer = (name, fallback) => { + const parsed = Number(values.get(name) ?? fallback) + if (!Number.isSafeInteger(parsed) || parsed < 0) { + throw new Error(`${name} must be a nonnegative integer`) + } + return parsed + } + const targetOrigin = values.get('--target-origin')?.replace(/\/$/, '') + const directorOrigin = values.get('--director-origin')?.replace(/\/$/, '') + const authOrigin = values.get('--auth-origin')?.replace(/\/$/, '') + if ((!targetOrigin && !directorOrigin) || (targetOrigin && directorOrigin) || !authOrigin) { + throw new Error('provide --auth-origin and exactly one target or director origin') + } + const controls = integer('--controls', 100) + const maxRampConnectionFailures = integer('--max-ramp-connection-failures', 0) + const maxUnexpectedCloses = integer('--max-unexpected-closes', 0) + const rebindProbes = integer('--rebind-probes', 0) + const placementOverflowProbes = integer('--placement-overflow-probes', 0) + const regionalFallbackProbes = integer('--regional-fallback-probes', 0) + const regionBehaviorProbes = integer('--region-behavior-probes', 0) + const requestUnitInvites = integer('--request-unit-invites', 0) + const requestUnitInvitesPerSecond = integer('--request-unit-invites-per-second', 0) + const requestUnitPrincipalCount = integer('--request-unit-principals', 0) + const relayAsiaLoadPrincipalCount = integer('--relay-asia-load-principals', 0) + const requestUnitOverflowProbes = integer('--request-unit-overflow-probes', 0) + const requestUnitCapacity = values.has('--request-unit-capacity') + ? integer('--request-unit-capacity', 0) + : undefined + const requestUnitCleanupTimeoutMs = integer( + '--request-unit-cleanup-timeout-seconds', + 0 + ) * 1000 + const rebindHoldMs = integer('--rebind-hold-ms', 4_000) + const rebindDelayMs = integer('--rebind-delay-seconds', 0) * 1000 + const capacityCellId = values.get('--capacity-cell-id') + const capacityCellOrigin = values.get('--capacity-cell-origin')?.replace(/\/$/, '') + const capacityHardCap = values.has('--capacity-hard-cap') + ? integer('--capacity-hard-cap', 0) + : undefined + const capacityUnobservedBound = values.has('--capacity-unobserved-bound') + ? integer('--capacity-unobserved-bound', 0) + : undefined + const shardCount = integer('--shard-count', 1) + const shardIndex = integer('--shard-index', 0) + const durationMs = integer('--duration-seconds', 900) * 1000 + const spliceHoldMs = integer( + '--splice-hold-seconds', + values.get('--duration-seconds') ?? 900 + ) * 1000 + const splices = integer('--splices', 0) + const spliceRampMs = integer('--splice-ramp-seconds', 0) * 1000 + const slowReaderSplices = integer('--slow-reader-splices', 0) + const wedgedReaderSplices = integer('--wedged-reader-splices', 0) + const splicePayloadBytes = integer('--splice-payload-bytes', 1_024) + const slowReaderStreamBytes = integer( + '--slow-reader-stream-bytes', + DEFAULT_SLOW_READER_STREAM_BYTES + ) + const wedgedReaderStreamBytes = integer( + '--wedged-reader-stream-bytes', + DEFAULT_WEDGED_READER_STREAM_BYTES + ) + const readerFrameBytes = integer('--reader-frame-bytes', DEFAULT_READER_FRAME_BYTES) + const slowReaderHoldMs = integer('--slow-reader-hold-ms', 2_000) + const wedgedReaderHoldMs = integer('--wedged-reader-hold-ms', 12_000) + const maxGeneratorRssGrowthMiB = integer('--max-generator-rss-growth-mib', 512) + const requiredLeaseHorizons = integer('--required-lease-horizons', 0) + const aggregateControls = values.has('--aggregate-controls') + ? integer('--aggregate-controls', 0) + : controls + const aggregateSplices = values.has('--aggregate-splices') + ? integer('--aggregate-splices', 0) + : splices + const aggregateRequestUnitInvites = values.has('--aggregate-request-unit-invites') + ? integer('--aggregate-request-unit-invites', 0) + : requestUnitInvites + const phaseBarrierDir = values.get('--phase-barrier-dir') + const phaseBarrierTimeoutMs = integer('--phase-barrier-timeout-seconds', 180) * 1000 + if (controls < 1 || controls > 10_000) throw new Error('--controls must be between 1 and 10000') + if (rebindProbes > controls) throw new Error('--rebind-probes cannot exceed --controls') + if (splices > controls) throw new Error('--splices cannot exceed --controls') + if (shardCount > 1 && capacityHardCap !== undefined) { + if (!values.has('--aggregate-controls') || !values.has('--aggregate-splices')) { + throw new Error('capacity-bound sharding requires explicit aggregate controls and splices') + } + if (aggregateControls !== controls * shardCount || aggregateSplices !== splices * shardCount) { + throw new Error('aggregate controls and splices must match every equal-sized shard') + } + } else if (aggregateControls !== controls || aggregateSplices !== splices) { + throw new Error('aggregate controls and splices require matching sharded local counts') + } + if ( + capacityHardCap !== undefined && + aggregateControls + 2 * aggregateSplices > capacityHardCap - 100 + ) { + throw new Error('controls plus splice connection units exceed ordinary cell admission') + } + if (slowReaderSplices + wedgedReaderSplices > splices) { + throw new Error('reader splice counts cannot exceed --splices') + } + if (splices > 0 && (spliceHoldMs < 1_000 || spliceHoldMs > durationMs)) { + throw new Error('--splice-hold-seconds must be between 1 and the steady duration') + } + if (slowReaderSplices + wedgedReaderSplices > 0 && !directorOrigin) { + throw new Error('reader evidence requires --director-origin') + } + if (splicePayloadBytes < 1 || splicePayloadBytes > 1_048_576) { + throw new Error('--splice-payload-bytes must be between 1 and 1048576') + } + if ( + slowReaderSplices > 0 && + slowReaderStreamBytes <= RELAY_LOAD_SPLICE_HIGH_WATER_BYTES + ) { + throw new Error('--slow-reader-stream-bytes must exceed the 256 KiB splice high-water mark') + } + if ( + wedgedReaderSplices > 0 && + wedgedReaderStreamBytes <= RELAY_LOAD_SPLICE_HIGH_WATER_BYTES + ) { + throw new Error('--wedged-reader-stream-bytes must exceed the 256 KiB splice high-water mark') + } + const localReaderSplices = slowReaderSplices + wedgedReaderSplices + const localReaderBytes = slowReaderSplices * slowReaderStreamBytes + + wedgedReaderSplices * wedgedReaderStreamBytes + const aggregateReaderSplices = values.has('--aggregate-reader-splices') + ? integer('--aggregate-reader-splices', 0) + : localReaderSplices * shardCount + const aggregateReaderBytes = values.has('--aggregate-reader-bytes') + ? integer('--aggregate-reader-bytes', 0) + : localReaderBytes * shardCount + if ( + aggregateReaderSplices < localReaderSplices || + aggregateReaderBytes < localReaderBytes || + (shardCount === 1 && + (aggregateReaderSplices !== localReaderSplices || aggregateReaderBytes !== localReaderBytes)) + ) { + throw new Error('aggregate reader bounds do not cover the local shard') + } + if (aggregateReaderSplices > RELAY_LOAD_MAX_AGGREGATE_READER_SPLICES) { + throw new Error('aggregate reader splice count exceeds the reviewed bound') + } + if (aggregateReaderBytes > RELAY_LOAD_MAX_AGGREGATE_READER_BYTES) { + throw new Error('aggregate reader stream bytes exceed the reviewed bound') + } + if (readerFrameBytes < 1 || readerFrameBytes > 1_048_576) { + throw new Error('--reader-frame-bytes must be between 1 and 1048576') + } + if (slowReaderSplices > 0 && slowReaderHoldMs >= RELAY_LOAD_SPLICE_WEDGED_TIMEOUT_MS) { + throw new Error('--slow-reader-hold-ms must stay below the wedged timeout') + } + if (wedgedReaderSplices > 0 && wedgedReaderHoldMs <= RELAY_LOAD_SPLICE_WEDGED_TIMEOUT_MS) { + throw new Error('--wedged-reader-hold-ms must exceed the wedged timeout') + } + if (wedgedReaderHoldMs > 30_000) { + throw new Error('--wedged-reader-hold-ms cannot exceed 30000') + } + if (maxGeneratorRssGrowthMiB < 1) { + throw new Error('--max-generator-rss-growth-mib must be positive') + } + if (placementOverflowProbes > 1) { + throw new Error('--placement-overflow-probes must be zero or one') + } + if (regionalFallbackProbes > 1) { + throw new Error('--regional-fallback-probes must be zero or one') + } + if (regionBehaviorProbes > 1 || requestUnitOverflowProbes > 1) { + throw new Error('regional behavior and request-unit overflow probes must be zero or one') + } + if (placementOverflowProbes > 0 && shardCount > 1) { + throw new Error('placement overflow proof requires one coordinated generator') + } + if ( + placementOverflowProbes > 0 && + (!directorOrigin || + !capacityCellId || + capacityHardCap === undefined || + capacityUnobservedBound === undefined) + ) { + throw new Error('placement overflow requires exact capacity cell, hard cap, and bound') + } + if (regionalFallbackProbes > 0 && ( + !directorOrigin || !capacityCellId || !capacityCellOrigin || + capacityHardCap === undefined || capacityUnobservedBound === undefined || + (shardCount > 1 && shardIndex !== 0) + )) throw new Error('regional fallback requires the coordinating capacity shard') + if (regionBehaviorProbes > 0 && ( + !directorOrigin || !capacityCellOrigin || preferredRegionValue(values) !== 'asia-east2' || + (shardCount > 1 && shardIndex !== 0) + )) throw new Error('regional behavior proof requires the coordinating Asia shard') + if (phaseBarrierDir && shardCount < 2) { + throw new Error('phase barrier requires multiple shards') + } + if (phaseBarrierTimeoutMs < 1_000) { + throw new Error('phase barrier timeout must be at least one second') + } + if (requestUnitInvites > 0) { + if ( + !directorOrigin || !capacityCellId || requestUnitCapacity === undefined || + requestUnitCapacity < 1 || requestUnitInvitesPerSecond < 1 || + requestUnitInvitesPerSecond > 20 || requestUnitPrincipalCount < 1 || + requestUnitPrincipalCount > 32 || requestUnitInvites > requestUnitPrincipalCount * 30 || + aggregateSplices !== 0 || !phaseBarrierDir || + aggregateRequestUnitInvites !== requestUnitInvites * shardCount || + aggregateControls + aggregateRequestUnitInvites !== requestUnitCapacity + ) throw new Error('request-unit proof does not reach the exact reviewed capacity') + } else if ( + requestUnitCapacity !== undefined || requestUnitInvitesPerSecond !== 0 || + requestUnitPrincipalCount !== 0 || + requestUnitOverflowProbes > 0 || + requestUnitCleanupTimeoutMs > 0 || aggregateRequestUnitInvites !== 0 + ) throw new Error('request-unit proof options require invite offers') + if ( + requestUnitOverflowProbes > 0 && + (shardIndex !== 0 || requestUnitCleanupTimeoutMs < 600_000) + ) throw new Error('request-unit overflow requires the cleanup-owning coordinator') + if (requestUnitCleanupTimeoutMs > 0 && requestUnitOverflowProbes !== 1) { + throw new Error('request-unit cleanup requires the overflow proof') + } + if ( + relayAsiaLoadPrincipalCount > 32 || + (relayAsiaLoadPrincipalCount > 0 && + (!directorOrigin || preferredRegionValue(values) !== 'asia-east2')) + ) throw new Error('Relay Asia load principals require a regional director proof') + if (capacityCellOrigin) { + const origin = new URL(capacityCellOrigin) + if (origin.protocol !== 'https:' || origin.origin !== capacityCellOrigin) { + throw new Error('--capacity-cell-origin must be canonical HTTPS') + } + } + if (shardCount < 1 || shardIndex >= shardCount) throw new Error('invalid shard index/count') + const minimumDurationMs = requiredLeaseHorizons * RELAY_CONTROL_LEASE_HORIZON_SECONDS * 1000 + if (durationMs < minimumDurationMs) { + throw new Error(`--duration-seconds must cover ${requiredLeaseHorizons} lease horizons`) + } + return { + targetOrigin, + directorOrigin, + authOrigin, + preferredRegion: preferredRegionValue(values), + controls, + maxRampConnectionFailures, + maxUnexpectedCloses, + rebindProbes, + placementOverflowProbes, + regionalFallbackProbes, + regionBehaviorProbes, + requestUnitInvites, + requestUnitInvitesPerSecond, + requestUnitPrincipalCount, + relayAsiaLoadPrincipalCount, + requestUnitOverflowProbes, + requestUnitCapacity, + requestUnitCleanupTimeoutMs, + rebindHoldMs, + rebindDelayMs, + capacityCellId, + capacityCellOrigin, + capacityHardCap, + capacityUnobservedBound, + durationMs, + spliceHoldMs, + rampMs: integer('--ramp-seconds', 60) * 1000, + rampStartDelayMs: integer('--ramp-start-delay-ms', 0), + reconnectMaxMs: integer('--reconnect-max-seconds', 30) * 1000, + shardCount, + shardIndex, + aggregateControls, + aggregateSplices, + aggregateRequestUnitInvites, + phaseBarrierDir, + phaseBarrierTimeoutMs, + aggregateReaderSplices, + aggregateReaderBytes, + signingKeyFile: values.get('--signing-key-file'), + splices, + spliceRampMs, + splicePayloadBytes, + slowReaderSplices, + wedgedReaderSplices, + slowReaderStreamBytes, + wedgedReaderStreamBytes, + readerFrameBytes, + slowReaderHoldMs, + wedgedReaderHoldMs, + maxGeneratorRssGrowthMiB, + requiredLeaseHorizons, + allowPartial: flags.has('--allow-partial'), + allowPlannedTransitionRetries: flags.has('--allow-planned-transition-retries'), + requireRebindOverflow: !flags.has('--skip-rebind-overflow-check') + } +} + +function preferredRegionValue(values) { + return values.get('--preferred-region') +} + +export function relayLoadSpliceIndexes(config) { + return Array.from( + { length: config.splices }, + (_, localIndex) => localIndex * config.shardCount + config.shardIndex + ) +} + +export function relayLoadSpliceStartDelayMs(config, localIndex) { + if (!Number.isSafeInteger(localIndex) || localIndex < 0 || localIndex >= config.splices) { + throw new Error('invalid local splice index') + } + const totalSplices = config.splices * config.shardCount + if (totalSplices <= 1) return 0 + const globalOrdinal = localIndex * config.shardCount + config.shardIndex + return Math.floor(globalOrdinal * config.spliceRampMs / (totalSplices - 1)) +} + +export function relayLoadPrincipalIndex(peerIndex, shardCount, principalCount) { + if ( + !Number.isSafeInteger(peerIndex) || peerIndex < 0 || + !Number.isSafeInteger(shardCount) || shardCount < 1 || + !Number.isSafeInteger(principalCount) || principalCount < 1 + ) throw new Error('invalid Relay load principal mapping') + return Math.floor(peerIndex / shardCount) % principalCount +} + +export function relayLoadSpliceProfile(config, spliceIndex) { + if (spliceIndex < config.wedgedReaderSplices) { + return { + readerMode: 'wedged', + readerHoldMs: config.wedgedReaderHoldMs, + streamBytes: config.wedgedReaderStreamBytes, + frameBytes: config.readerFrameBytes + } + } + if (spliceIndex < config.wedgedReaderSplices + config.slowReaderSplices) { + return { + readerMode: 'slow', + readerHoldMs: config.slowReaderHoldMs, + streamBytes: config.slowReaderStreamBytes, + frameBytes: config.readerFrameBytes + } + } + return { + readerMode: 'normal', + readerHoldMs: 0, + streamBytes: config.splicePayloadBytes, + frameBytes: config.splicePayloadBytes + } +} + +export function relayLoadReaderEvidenceError(result, config) { + const readerSplices = config.slowReaderSplices + config.wedgedReaderSplices + if (readerSplices > 0 && result.generatorRssGrowthMiB > config.maxGeneratorRssGrowthMiB) { + return 'load generator exceeded its RSS growth budget' + } + if (result.slowReaderSplicesCompleted !== config.slowReaderSplices) { + return 'slow-reader streams did not all complete' + } + if (result.wedgedReaderSplicesClosed !== config.wedgedReaderSplices) { + return 'wedged-reader streams did not all close at the relay limit' + } + if ( + readerSplices > 0 && + (result.readerQueueEvidence.length === 0 || + result.readerQueueEvidence.some(({ increaseBytes }) => increaseBytes < 1)) + ) { + return 'reader streams produced no causal Relay queued-byte evidence' + } + if ( + config.wedgedReaderSplices > 0 && + result.readerClosesByCode['4429'] !== config.wedgedReaderSplices + ) { + return 'wedged-reader streams did not close with 4429' + } + return undefined +} diff --git a/cloud/dev/scripts/relay-load-profile.test.mjs b/cloud/dev/scripts/relay-load-profile.test.mjs new file mode 100644 index 00000000000..40724f9c34e --- /dev/null +++ b/cloud/dev/scripts/relay-load-profile.test.mjs @@ -0,0 +1,388 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + parseRelayLoadArguments, + relayLoadPrincipalIndex, + relayLoadReaderEvidenceError, + relayLoadSpliceIndexes, + relayLoadSpliceProfile, + relayLoadSpliceStartDelayMs +} from './relay-load-profile.mjs' + +const required = ['--auth-origin', 'https://auth.test', '--director-origin', 'https://relay.test'] + +test('accepts the 2840-control two-lease-horizon Asia proof profile', () => { + const config = parseRelayLoadArguments([ + ...required, + '--controls', + '2840', + '--duration-seconds', + '210', + '--required-lease-horizons', + '2', + '--preferred-region', + 'asia-east2', + '--relay-asia-load-principals', + '32', + '--capacity-hard-cap', + '3000' + ]) + + assert.equal(config.controls, 2840) + assert.equal(config.durationMs, 210_000) + assert.equal(config.requiredLeaseHorizons, 2) + assert.equal(config.preferredRegion, 'asia-east2') + assert.equal(config.relayAsiaLoadPrincipalCount, 32) + assert.equal(config.splices, 0) +}) + +test('binds synthetic Relay principals to a bounded Asia director proof', () => { + assert.throws(() => parseRelayLoadArguments([ + ...required, '--relay-asia-load-principals', '1' + ]), /require a regional director proof/) + assert.throws(() => parseRelayLoadArguments([ + ...required, '--preferred-region', 'asia-east2', + '--relay-asia-load-principals', '33' + ]), /require a regional director proof/) +}) + +test('rejects a mixed profile beyond the ordinary 2900-unit boundary', () => { + assert.throws( + () => parseRelayLoadArguments([ + ...required, + '--controls', '2840', + '--splices', '31', + '--capacity-hard-cap', '3000' + ]), + /exceed ordinary cell admission/ + ) + expectMixedProfile(parseRelayLoadArguments([ + ...required, + '--controls', '2600', + '--splices', '120', + '--capacity-hard-cap', '3000' + ])) +}) + +test('requires reviewed aggregate totals for capacity-bound shards', () => { + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--controls', '2840', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0' + ]), + /requires explicit aggregate controls and splices/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--controls', '2840', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2840', '--aggregate-splices', '0' + ]), + /must match every equal-sized shard/ + ) + const config = parseRelayLoadArguments([ + ...required, '--controls', '710', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2840', '--aggregate-splices', '0' + ]) + assert.equal(config.aggregateControls, 2840) + assert.equal(config.aggregateSplices, 0) +}) + +test('allows one coordinating shard to prove regional fallback', () => { + const config = parseRelayLoadArguments([ + ...required, '--controls', '710', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2840', '--aggregate-splices', '0', + '--regional-fallback-probes', '1', '--capacity-cell-id', 'staging-gce-c4', + '--capacity-cell-origin', 'https://c4.relay-staging.onorca.dev', + '--capacity-unobserved-bound', '60', '--rebind-probes', '160' + ]) + assert.equal(config.regionalFallbackProbes, 1) + assert.equal(config.capacityCellOrigin, 'https://c4.relay-staging.onorca.dev') + assert.throws(() => parseRelayLoadArguments([ + ...required, '--controls', '710', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '1', + '--aggregate-controls', '2840', '--aggregate-splices', '0', + '--regional-fallback-probes', '1', '--capacity-cell-id', 'staging-gce-c4', + '--capacity-cell-origin', 'https://c4.relay-staging.onorca.dev', + '--capacity-unobserved-bound', '60' + ]), /coordinating capacity shard/) +}) + +test('accepts the exact sharded request-unit and region behavior proof', () => { + const config = parseRelayLoadArguments([ + ...required, '--preferred-region', 'asia-east2', + '--controls', '710', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2840', '--aggregate-splices', '0', + '--capacity-cell-id', 'staging-gce-c4', + '--capacity-cell-origin', 'https://c4.relay-staging.onorca.dev', + '--request-unit-invites', '790', '--request-unit-invites-per-second', '2', + '--request-unit-principals', '32', + '--aggregate-request-unit-invites', '3160', + '--request-unit-capacity', '6000', '--request-unit-overflow-probes', '1', + '--request-unit-cleanup-timeout-seconds', '630', + '--region-behavior-probes', '1', '--phase-barrier-dir', '/tmp/load-barrier' + ]) + assert.equal(config.aggregateRequestUnitInvites, 3_160) + assert.equal(config.requestUnitCapacity, 6_000) + assert.equal(config.requestUnitPrincipalCount, 32) + assert.equal(config.requestUnitCleanupTimeoutMs, 630_000) + assert.equal(config.regionBehaviorProbes, 1) + assert.equal(config.phaseBarrierDir, '/tmp/load-barrier') + + assert.throws(() => parseRelayLoadArguments([ + ...required, '--controls', '710', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2840', '--aggregate-splices', '0', + '--capacity-cell-id', 'staging-gce-c4', '--request-unit-invites', '789', + '--request-unit-invites-per-second', '2', + '--request-unit-principals', '32', + '--aggregate-request-unit-invites', '3156', '--request-unit-capacity', '6000', + '--phase-barrier-dir', '/tmp/load-barrier' + ]), /does not reach the exact reviewed capacity/) + + assert.throws(() => parseRelayLoadArguments([ + ...required, '--controls', '710', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2840', '--aggregate-splices', '0', + '--capacity-cell-id', 'staging-gce-c4', '--request-unit-invites', '790', + '--request-unit-invites-per-second', '2', '--request-unit-principals', '26', + '--aggregate-request-unit-invites', '3160', '--request-unit-capacity', '6000', + '--phase-barrier-dir', '/tmp/load-barrier' + ]), /does not reach the exact reviewed capacity/) +}) + +function expectMixedProfile(config) { + assert.equal(config.controls + 2 * config.splices, 2840) +} + +test('rejects a run shorter than its required lease horizons', () => { + assert.throws( + () => + parseRelayLoadArguments([ + ...required, + '--duration-seconds', + '209', + '--required-lease-horizons', + '2' + ]), + /must cover 2 lease horizons/ + ) +}) + +test('bounds an explicit splice hold within the steady window', () => { + const config = parseRelayLoadArguments([ + ...required, + '--controls', '1', + '--splices', '1', + '--duration-seconds', '300', + '--splice-hold-seconds', '60' + ]) + assert.equal(config.durationMs, 300_000) + assert.equal(config.spliceHoldMs, 60_000) + assert.throws(() => parseRelayLoadArguments([ + ...required, + '--controls', '1', + '--splices', '1', + '--duration-seconds', '300', + '--splice-hold-seconds', '301' + ]), /between 1 and the steady duration/) +}) + +test('maps splice ownership deterministically within a shard', () => { + const config = parseRelayLoadArguments([ + ...required, + '--controls', + '4', + '--splices', + '3', + '--shard-count', + '4', + '--shard-index', + '2' + ]) + + assert.deepEqual(relayLoadSpliceIndexes(config), [2, 6, 10]) +}) + +test('staggered shards form one deterministic splice ramp', () => { + const config = parseRelayLoadArguments([ + ...required, + '--controls', '4', + '--splices', '3', + '--splice-ramp-seconds', '11', + '--shard-count', '4', + '--shard-index', '2' + ]) + + assert.equal(config.spliceRampMs, 11_000) + assert.deepEqual( + [0, 1, 2].map((index) => relayLoadSpliceStartDelayMs(config, index)), + [2_000, 6_000, 10_000] + ) + assert.throws(() => relayLoadSpliceStartDelayMs(config, 3), /invalid local splice index/) +}) + +test('distributes each shard invite wave below the account rate limit', () => { + const identities = Array.from( + { length: 710 }, + (_, localIndex) => relayLoadPrincipalIndex(localIndex * 4 + 2, 4, 32) + ) + const offers = Array.from({ length: 790 }, (_, index) => identities[index % identities.length]) + const counts = new Map() + for (const principal of offers) counts.set(principal, (counts.get(principal) ?? 0) + 1) + assert.equal(counts.size, 32) + assert.equal(Math.max(...counts.values()), 26) +}) + +test('requires exactly one assignment mode', () => { + assert.throws( + () => + parseRelayLoadArguments([ + ...required, + '--target-origin', + 'https://cell.test' + ]), + /exactly one target or director origin/ + ) +}) + +test('requires separate recoverable and wedged reader profiles', () => { + const config = parseRelayLoadArguments([ + ...required, + '--controls', '4', + '--splices', '3', + '--slow-reader-splices', '1', + '--wedged-reader-splices', '1', + '--slow-reader-stream-bytes', '524288', + '--wedged-reader-stream-bytes', '1048576', + '--slow-reader-hold-ms', '9000', + '--wedged-reader-hold-ms', '11000' + ]) + + assert.deepEqual(relayLoadSpliceProfile(config, 0), { + readerMode: 'wedged', readerHoldMs: 11_000, streamBytes: 1_048_576, frameBytes: 65_536 + }) + assert.deepEqual(relayLoadSpliceProfile(config, 1), { + readerMode: 'slow', readerHoldMs: 9_000, streamBytes: 524_288, frameBytes: 65_536 + }) + assert.equal(relayLoadSpliceProfile(config, 2).readerMode, 'normal') +}) + +test('bounds reader stream, timeout, and splice load per shard', () => { + assert.throws( + () => parseRelayLoadArguments([...required, '--controls', '2', '--splices', '3']), + /splices cannot exceed/ + ) + assert.throws( + () => + parseRelayLoadArguments([ + ...required, + '--splices', + '1', + '--slow-reader-splices', + '2' + ]), + /reader splice counts cannot exceed/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--splices', '1', '--slow-reader-splices', '1', + '--slow-reader-stream-bytes', '262144' + ]), + /must exceed the 256 KiB/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--splices', '1', '--wedged-reader-splices', '1', + '--wedged-reader-stream-bytes', '262144' + ]), + /must exceed the 256 KiB/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--splices', '1', '--slow-reader-splices', '1', + '--slow-reader-hold-ms', '10000' + ]), + /must stay below the wedged timeout/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--splices', '1', '--wedged-reader-splices', '1', + '--wedged-reader-hold-ms', '10000' + ]), + /must exceed the wedged timeout/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--controls', '17', '--splices', '17', '--slow-reader-splices', '17' + ]), + /reader splice count exceeds/ + ) + assert.throws( + () => parseRelayLoadArguments([ + ...required, '--controls', '9', '--splices', '9', '--wedged-reader-splices', '9' + ]), + /reader stream bytes exceed/ + ) +}) + +test('supports one bounded reader-owning shard without multiplying its pressure', () => { + const owner = parseRelayLoadArguments([ + ...required, + '--controls', '650', '--splices', '30', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '0', + '--aggregate-controls', '2600', '--aggregate-splices', '120', + '--slow-reader-splices', '10', '--wedged-reader-splices', '1', + '--aggregate-reader-splices', '11', '--aggregate-reader-bytes', '18874368' + ]) + const peer = parseRelayLoadArguments([ + ...required, + '--controls', '650', '--splices', '30', '--capacity-hard-cap', '3000', + '--shard-count', '4', '--shard-index', '1', + '--aggregate-controls', '2600', '--aggregate-splices', '120', + '--aggregate-reader-splices', '11', '--aggregate-reader-bytes', '18874368' + ]) + + assert.equal(owner.aggregateReaderSplices, 11) + assert.equal(owner.aggregateReaderBytes, 18 * 1024 * 1024) + assert.equal(peer.slowReaderSplices + peer.wedgedReaderSplices, 0) + assert.equal(peer.aggregateReaderSplices, 11) +}) + +test('requires queue, memory, and expected close evidence', () => { + const config = parseRelayLoadArguments([ + ...required, '--splices', '2', '--slow-reader-splices', '1', + '--wedged-reader-splices', '1', '--max-generator-rss-growth-mib', '100' + ]) + const passing = { + generatorRssGrowthMiB: 50, + slowReaderSplicesCompleted: 1, + wedgedReaderSplicesClosed: 1, + readerQueueEvidence: [ + { origin: 'https://cell.test', baselineBytes: 8, peakBytes: 65_544, increaseBytes: 65_536 } + ], + readerClosesByCode: { '4429': 1 } + } + + assert.equal(relayLoadReaderEvidenceError(passing, config), undefined) + assert.match( + relayLoadReaderEvidenceError({ + ...passing, + readerQueueEvidence: [ + { origin: 'https://cell.test', baselineBytes: 8, peakBytes: 8, increaseBytes: 0 } + ] + }, config), + /no causal Relay queued-byte evidence/ + ) + assert.match( + relayLoadReaderEvidenceError({ ...passing, generatorRssGrowthMiB: 101 }, config), + /RSS growth budget/ + ) + assert.match( + relayLoadReaderEvidenceError({ ...passing, readerClosesByCode: { '4429': 0 } }, config), + /did not close with 4429/ + ) +}) diff --git a/cloud/dev/scripts/relay-load-reader-evidence.mjs b/cloud/dev/scripts/relay-load-reader-evidence.mjs new file mode 100644 index 00000000000..f20b5029587 --- /dev/null +++ b/cloud/dev/scripts/relay-load-reader-evidence.mjs @@ -0,0 +1,51 @@ +const DEFAULT_TIMEOUT_MS = 8_000 +const DEFAULT_POLL_MS = 100 + +export async function createRelayLoadReaderEvidence(origins, dependencies) { + const distinctOrigins = [...new Set(origins)].sort() + const baselines = new Map(await Promise.all(distinctOrigins.map(async (origin) => [ + origin, + await dependencies.readQueuedBytes(origin) + ]))) + const peaks = new Map(baselines) + const pending = new Map() + const now = dependencies.now ?? Date.now + const delay = dependencies.delay + const timeoutMs = dependencies.timeoutMs ?? DEFAULT_TIMEOUT_MS + const pollMs = dependencies.pollMs ?? DEFAULT_POLL_MS + + const observe = async ({ cellOrigin }) => { + if (!baselines.has(cellOrigin)) throw new Error('reader origin lacks a run baseline') + if (peaks.get(cellOrigin) > baselines.get(cellOrigin)) return + const current = pending.get(cellOrigin) + if (current) return await current + const proof = (async () => { + const baseline = baselines.get(cellOrigin) + const deadline = now() + timeoutMs + for (;;) { + const queuedBytes = await dependencies.readQueuedBytes(cellOrigin) + peaks.set(cellOrigin, Math.max(peaks.get(cellOrigin), queuedBytes)) + if (queuedBytes > baseline) return + if (now() >= deadline) { + throw new Error('reader stream produced no causal Relay queued-byte increase') + } + await delay(pollMs) + } + })() + pending.set(cellOrigin, proof) + try { + await proof + } finally { + pending.delete(cellOrigin) + } + } + + const snapshot = () => distinctOrigins.map((origin) => ({ + origin, + baselineBytes: baselines.get(origin), + peakBytes: peaks.get(origin), + increaseBytes: peaks.get(origin) - baselines.get(origin) + })) + + return { observe, snapshot } +} diff --git a/cloud/dev/scripts/relay-load-reader-evidence.test.mjs b/cloud/dev/scripts/relay-load-reader-evidence.test.mjs new file mode 100644 index 00000000000..ce417f5bc42 --- /dev/null +++ b/cloud/dev/scripts/relay-load-reader-evidence.test.mjs @@ -0,0 +1,76 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { createRelayLoadReaderEvidence } from './relay-load-reader-evidence.mjs' + +test('requires a queue increase above the pre-injection baseline for every origin', async () => { + const samples = new Map([ + ['https://a.test', [7, 7, 11]], + ['https://b.test', [0, 3]] + ]) + let now = 0 + const evidence = await createRelayLoadReaderEvidence([...samples.keys()], { + readQueuedBytes: async (origin) => samples.get(origin).shift(), + delay: async (ms) => { now += ms }, + now: () => now + }) + + await Promise.all([ + evidence.observe({ cellOrigin: 'https://a.test' }), + evidence.observe({ cellOrigin: 'https://b.test' }) + ]) + + assert.deepEqual(evidence.snapshot(), [ + { origin: 'https://a.test', baselineBytes: 7, peakBytes: 11, increaseBytes: 4 }, + { origin: 'https://b.test', baselineBytes: 0, peakBytes: 3, increaseBytes: 3 } + ]) +}) + +test('shares one causal proof across concurrent readers on the same cell', async () => { + const samples = [4, 4, 9] + let reads = 0 + let now = 0 + const evidence = await createRelayLoadReaderEvidence(['https://cell.test'], { + readQueuedBytes: async () => { reads++; return samples.shift() }, + delay: async (ms) => { now += ms }, + now: () => now + }) + + await Promise.all([ + evidence.observe({ cellOrigin: 'https://cell.test' }), + evidence.observe({ cellOrigin: 'https://cell.test' }) + ]) + + assert.equal(reads, 3) + assert.equal(evidence.snapshot()[0].increaseBytes, 5) +}) + +test('reuses a completed causal proof for later readers on the same cell', async () => { + const samples = [4, 9] + let reads = 0 + const evidence = await createRelayLoadReaderEvidence(['https://cell.test'], { + readQueuedBytes: async () => { reads++; return samples.shift() }, + delay: async () => undefined + }) + + await evidence.observe({ cellOrigin: 'https://cell.test' }) + await evidence.observe({ cellOrigin: 'https://cell.test' }) + + assert.equal(reads, 2) + assert.equal(evidence.snapshot()[0].increaseBytes, 5) +}) + +test('rejects a pre-existing nonzero queue that never increases', async () => { + let now = 0 + const evidence = await createRelayLoadReaderEvidence(['https://cell.test'], { + readQueuedBytes: async () => 9, + delay: async (ms) => { now += ms }, + now: () => now, + timeoutMs: 200, + pollMs: 100 + }) + + await assert.rejects( + evidence.observe({ cellOrigin: 'https://cell.test' }), + /no causal Relay queued-byte increase/ + ) +}) diff --git a/cloud/dev/scripts/relay-load-rebind-boundary.mjs b/cloud/dev/scripts/relay-load-rebind-boundary.mjs new file mode 100644 index 00000000000..2eb72c7d6ba --- /dev/null +++ b/cloud/dev/scripts/relay-load-rebind-boundary.mjs @@ -0,0 +1,59 @@ +export async function waitForRelayLoadRebindGate({ + delay, + delayMs, + activeCount, + requiredCount +}) { + await delay(delayMs) + const active = activeCount() + if (active !== requiredCount) { + throw new Error(`rebind boundary requires ${requiredCount} active controls, found ${active}`) + } +} + +export async function proveRelayLoadRebindBoundary({ + peers, + probeCount, + holdMs, + delay, + failureReason, + requireOverflow = true +}) { + if (probeCount === 0) return { opened: 0, overflowReason: null } + if (peers.length < probeCount) throw new Error('insufficient active controls for rebind proof') + + const probes = [] + try { + const opened = await Promise.allSettled( + peers.slice(0, probeCount).map((peer) => peer.openRebindProbe()) + ) + for (const result of opened) { + if (result.status === 'fulfilled') probes.push(result.value) + } + const rejected = opened.find((result) => result.status === 'rejected') + if (rejected) throw rejected.reason + + let overflowReason = null + if (requireOverflow) { + try { + const overflow = await peers[0].openRebindProbe() + await overflow.close() + } catch (error) { + overflowReason = failureReason(error) + } + if (overflowReason !== 'socket_http_503') { + throw new Error(`rebind overflow was not rejected at the hard cap: ${overflowReason}`) + } + } + const closedIndex = await Promise.race([ + delay(holdMs).then(() => -1), + ...probes.map((probe, index) => probe.closed.then(() => index)) + ]) + if (closedIndex >= 0 || probes.some((probe) => !probe.isOpen())) { + throw new Error('rebind probe closed before the hold completed') + } + return { opened: probes.length, overflowReason } + } finally { + await Promise.all(probes.map((probe) => probe.close())) + } +} diff --git a/cloud/dev/scripts/relay-load-rebind-boundary.test.mjs b/cloud/dev/scripts/relay-load-rebind-boundary.test.mjs new file mode 100644 index 00000000000..50ad2a48383 --- /dev/null +++ b/cloud/dev/scripts/relay-load-rebind-boundary.test.mjs @@ -0,0 +1,164 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + proveRelayLoadRebindBoundary, + waitForRelayLoadRebindGate +} from './relay-load-rebind-boundary.mjs' + +function peer(open) { + return { openRebindProbe: open } +} + +function probe(onClose = () => undefined) { + let open = true + let resolveClosed + const closed = new Promise((resolve) => { + resolveClosed = resolve + }) + return { + close: () => { + if (!open) return + open = false + onClose() + resolveClosed() + }, + closed, + isOpen: () => open + } +} + +test('holds the requested rebind overlap and requires a hard-cap rejection', async () => { + let openCalls = 0 + let closes = 0 + let heldFor = null + const peers = [ + peer(async () => { + openCalls++ + if (openCalls === 3) throw new Error('Unexpected server response: 503') + return probe(() => closes++) + }), + peer(async () => { + openCalls++ + return probe(() => closes++) + }) + ] + const result = await proveRelayLoadRebindBoundary({ + peers, + probeCount: 2, + holdMs: 4_000, + delay: async (milliseconds) => { + heldFor = milliseconds + }, + failureReason: (error) => + error.message.includes('503') ? 'socket_http_503' : 'unknown' + }) + + assert.deepEqual(result, { opened: 2, overflowReason: 'socket_http_503' }) + assert.equal(heldFor, 4_000) + assert.equal(closes, 2) +}) + +test('closes successful probes when a boundary probe fails', async () => { + let closes = 0 + const peers = [ + peer(async () => probe(() => closes++)), + peer(async () => { + throw new Error('probe failed') + }) + ] + + await assert.rejects( + proveRelayLoadRebindBoundary({ + peers, + probeCount: 2, + holdMs: 0, + delay: async () => undefined, + failureReason: () => 'unknown' + }), + /probe failed/ + ) + assert.equal(closes, 1) +}) + +test('waits for every replacement socket to finish closing', async () => { + let finishClose + const closeFinished = new Promise((resolve) => { + finishClose = resolve + }) + const closingProbe = probe() + closingProbe.close = () => closeFinished + let completed = false + const boundary = proveRelayLoadRebindBoundary({ + peers: [peer(async () => closingProbe)], + probeCount: 1, + holdMs: 0, + delay: async () => undefined, + failureReason: () => 'unknown', + requireOverflow: false + }).then(() => { + completed = true + }) + + await new Promise((resolve) => setImmediate(resolve)) + assert.equal(completed, false) + finishClose() + await boundary + assert.equal(completed, true) +}) + +test('can prove reserved replacement headroom below the physical cap', async () => { + let closes = 0 + const result = await proveRelayLoadRebindBoundary({ + peers: [peer(async () => probe(() => closes++))], + probeCount: 1, + holdMs: 0, + delay: async () => undefined, + failureReason: () => 'unknown', + requireOverflow: false + }) + assert.deepEqual(result, { opened: 1, overflowReason: null }) + assert.equal(closes, 1) +}) + +test('fails when a replacement closes before the hold completes', async () => { + let heldProbe + await assert.rejects( + proveRelayLoadRebindBoundary({ + peers: [ + peer(async () => { + heldProbe = probe() + return heldProbe + }) + ], + probeCount: 1, + holdMs: 4_000, + delay: async () => { + heldProbe.close() + }, + failureReason: () => 'unknown', + requireOverflow: false + }), + /closed before the hold completed/ + ) +}) + +test('delays the boundary until every ordinary control has recovered', async () => { + let active = 899 + await assert.rejects( + waitForRelayLoadRebindGate({ + delay: async () => undefined, + delayMs: 0, + activeCount: () => active, + requiredCount: 900 + }), + /requires 900 active controls/ + ) + await waitForRelayLoadRebindGate({ + delay: async () => { + active = 900 + }, + delayMs: 1, + activeCount: () => active, + requiredCount: 900 + }) +}) diff --git a/cloud/dev/scripts/relay-load-region-behavior.mjs b/cloud/dev/scripts/relay-load-region-behavior.mjs new file mode 100644 index 00000000000..b0333b32abe --- /dev/null +++ b/cloud/dev/scripts/relay-load-region-behavior.mjs @@ -0,0 +1,35 @@ +const ASSIGNMENT_RETRY_DELAY_MS = 5_100 + +const waitPastAssignmentRateLimit = (schedule) => + new Promise((resolve) => schedule(resolve, ASSIGNMENT_RETRY_DELAY_MS)) + +export async function proveRelayLoadRegionBehavior({ + oldClientPeer, + stickyPeer, + asiaOrigin, + scheduleAssignmentRetry = setTimeout +}) { + try { + await oldClientPeer.connect() + if (!oldClientPeer.assignedCellUrl() || oldClientPeer.assignedCellUrl() === asiaOrigin) { + throw new Error('unhinted client did not use the US-first path') + } + } finally { + await oldClientPeer.shutdown() + } + + try { + await stickyPeer.connect() + if (stickyPeer.assignedCellUrl() !== asiaOrigin) { + throw new Error('preferred Asia client did not reach the Asia cell') + } + await waitPastAssignmentRateLimit(scheduleAssignmentRetry) + const reassigned = await stickyPeer.requestAssignment('us-central1') + if (reassigned.cellUrl !== asiaOrigin) { + throw new Error('valid sticky assignment moved after preference changed') + } + } finally { + await stickyPeer.shutdown() + } + return { oldClientUsFirst: true, stickyAssignmentPreserved: true } +} diff --git a/cloud/dev/scripts/relay-load-region-behavior.test.mjs b/cloud/dev/scripts/relay-load-region-behavior.test.mjs new file mode 100644 index 00000000000..5882e3021a1 --- /dev/null +++ b/cloud/dev/scripts/relay-load-region-behavior.test.mjs @@ -0,0 +1,47 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { proveRelayLoadRegionBehavior } from './relay-load-region-behavior.mjs' + +function peer(origin, reassigned = origin) { + let shutdowns = 0 + return { + connect: async () => undefined, + assignedCellUrl: () => origin, + requestAssignment: async () => ({ cellUrl: reassigned }), + shutdown: async () => { shutdowns++ }, + shutdowns: () => shutdowns + } +} + +test('proves unhinted US-first placement and sticky Asia preservation', async () => { + const oldClientPeer = peer('https://c3.relay-staging.onorca.dev') + const stickyPeer = peer('https://c4.relay-staging.onorca.dev') + let retryDelayMs = 0 + assert.deepEqual(await proveRelayLoadRegionBehavior({ + oldClientPeer, + stickyPeer, + asiaOrigin: 'https://c4.relay-staging.onorca.dev', + scheduleAssignmentRetry: (resolve, delayMs) => { + retryDelayMs = delayMs + resolve() + } + }), { oldClientUsFirst: true, stickyAssignmentPreserved: true }) + assert.equal(retryDelayMs, 5_100) + assert.equal(oldClientPeer.shutdowns(), 1) + assert.equal(stickyPeer.shutdowns(), 1) +}) + +test('rejects Asia placement for an unhinted client or a moved sticky assignment', async () => { + await assert.rejects(proveRelayLoadRegionBehavior({ + oldClientPeer: peer('https://c4.relay-staging.onorca.dev'), + stickyPeer: peer('https://c4.relay-staging.onorca.dev'), + asiaOrigin: 'https://c4.relay-staging.onorca.dev', + scheduleAssignmentRetry: (resolve) => resolve() + }), /US-first/) + await assert.rejects(proveRelayLoadRegionBehavior({ + oldClientPeer: peer('https://c3.relay-staging.onorca.dev'), + stickyPeer: peer('https://c4.relay-staging.onorca.dev', 'https://c3.relay-staging.onorca.dev'), + asiaOrigin: 'https://c4.relay-staging.onorca.dev', + scheduleAssignmentRetry: (resolve) => resolve() + }), /sticky assignment moved/) +}) diff --git a/cloud/dev/scripts/relay-load-request-unit-boundary.mjs b/cloud/dev/scripts/relay-load-request-unit-boundary.mjs new file mode 100644 index 00000000000..a5dbb59d96b --- /dev/null +++ b/cloud/dev/scripts/relay-load-request-unit-boundary.mjs @@ -0,0 +1,43 @@ +import { setTimeout as delayDefault } from 'node:timers/promises' + +export async function openRelayLoadInviteOffers({ + peers, + count, + ratePerSecond, + concurrency = 8, + delay = delayDefault, + now = Date.now +}) { + if ( + !Array.isArray(peers) || peers.length === 0 || + !Number.isSafeInteger(count) || count < 0 || + !Number.isSafeInteger(ratePerSecond) || ratePerSecond < 1 || ratePerSecond > 20 || + !Number.isSafeInteger(concurrency) || concurrency < 1 + ) throw new Error('invalid Relay invite-offer load') + let next = 0 + let nextStartAt = now() + const workers = Array.from({ length: Math.min(concurrency, count) }, async () => { + for (;;) { + const index = next++ + if (index >= count) return + const scheduledAt = Math.max(nextStartAt, now()) + nextStartAt = scheduledAt + 1_000 / ratePerSecond + await delay(Math.max(0, scheduledAt - now())) + await peers[index % peers.length].openInviteOffer() + } + }) + await Promise.all(workers) + return count +} + +export async function proveRelayLoadRequestUnitBoundary(peer) { + try { + await peer.openInviteOffer() + } catch (error) { + if (error instanceof Error && error.message === 'invite offer failed: relay_capacity_exhausted') { + return 'relay_capacity_exhausted' + } + throw new Error('request-unit overflow was not rejected safely', { cause: error }) + } + throw new Error('request-unit overflow unexpectedly succeeded') +} diff --git a/cloud/dev/scripts/relay-load-request-unit-boundary.test.mjs b/cloud/dev/scripts/relay-load-request-unit-boundary.test.mjs new file mode 100644 index 00000000000..739e055c93a --- /dev/null +++ b/cloud/dev/scripts/relay-load-request-unit-boundary.test.mjs @@ -0,0 +1,52 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + openRelayLoadInviteOffers, + proveRelayLoadRequestUnitBoundary +} from './relay-load-request-unit-boundary.mjs' + +test('distributes the exact invite count with bounded concurrency', async () => { + let active = 0 + let peak = 0 + const delays = [] + const calls = [0, 0, 0] + const peers = calls.map((_, index) => ({ + async openInviteOffer() { + calls[index]++ + active++ + peak = Math.max(peak, active) + await Promise.resolve() + active-- + } + })) + assert.equal(await openRelayLoadInviteOffers({ + peers, + count: 8, + ratePerSecond: 2, + concurrency: 2, + delay: async (milliseconds) => { delays.push(milliseconds) }, + now: () => 0 + }), 8) + assert.deepEqual(calls, [3, 3, 2]) + assert.ok(peak <= 2) + assert.equal(delays.length, 8) + assert.ok(Math.max(...delays) >= 3_500) +}) + +test('accepts only the exact request-unit exhaustion error', async () => { + assert.equal(await proveRelayLoadRequestUnitBoundary({ + openInviteOffer: async () => { + throw new Error('invite offer failed: relay_capacity_exhausted') + } + }), 'relay_capacity_exhausted') + await assert.rejects( + proveRelayLoadRequestUnitBoundary({ + openInviteOffer: async () => { throw new Error('control response timeout') } + }), + /not rejected safely/ + ) + await assert.rejects( + proveRelayLoadRequestUnitBoundary({ openInviteOffer: async () => undefined }), + /unexpectedly succeeded/ + ) +}) diff --git a/cloud/dev/scripts/relay-load-run-lifecycle.mjs b/cloud/dev/scripts/relay-load-run-lifecycle.mjs new file mode 100644 index 00000000000..dffd76a813b --- /dev/null +++ b/cloud/dev/scripts/relay-load-run-lifecycle.mjs @@ -0,0 +1,25 @@ +export function assertRelayLoadRampAccepted(rampConnectionFailures, maximum) { + if (rampConnectionFailures > maximum) { + throw new Error('relay load ramp exceeded the allowed connection failures') + } +} + +export async function runRelayLoadWithShutdown(operation, shutdown) { + try { + return await operation() + } finally { + await shutdown() + } +} + +export function relayLoadRunHasDisallowedFailures(result, config) { + return ( + result.rampConnectionFailures > config.maxRampConnectionFailures || + (!config.allowPlannedTransitionRetries && result.transitionConnectionFailures > 0) || + result.steadyConnectionFailures > 0 || + result.unexpectedCloses > config.maxUnexpectedCloses || + result.protocolErrors > 0 || + result.refreshErrors > 0 || + result.socketErrors > 0 + ) +} diff --git a/cloud/dev/scripts/relay-load-run-lifecycle.test.mjs b/cloud/dev/scripts/relay-load-run-lifecycle.test.mjs new file mode 100644 index 00000000000..5aec8e08ab9 --- /dev/null +++ b/cloud/dev/scripts/relay-load-run-lifecycle.test.mjs @@ -0,0 +1,68 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + assertRelayLoadRampAccepted, + relayLoadRunHasDisallowedFailures, + runRelayLoadWithShutdown +} from './relay-load-run-lifecycle.mjs' + +test('always shuts down peers when a load phase fails', async () => { + const events = [] + await assert.rejects( + runRelayLoadWithShutdown( + async () => { + events.push('run') + throw new Error('boundary failed') + }, + async () => events.push('shutdown') + ), + /boundary failed/ + ) + assert.deepEqual(events, ['run', 'shutdown']) +}) + +test('fails immediately when the strict ramp budget is exceeded', () => { + assert.doesNotThrow(() => assertRelayLoadRampAccepted(0, 0)) + assert.throws(() => assertRelayLoadRampAccepted(1, 0), /ramp exceeded/) +}) + +test('rejects connection failures during the transition window', () => { + const result = { + rampConnectionFailures: 0, + transitionConnectionFailures: 1, + steadyConnectionFailures: 0, + unexpectedCloses: 0, + protocolErrors: 0, + refreshErrors: 0, + socketErrors: 0 + } + assert.equal( + relayLoadRunHasDisallowedFailures(result, { + maxRampConnectionFailures: 0, + maxUnexpectedCloses: 0 + }), + true + ) +}) + +test('allows only explicitly planned transition retries', () => { + const result = { + rampConnectionFailures: 0, + transitionConnectionFailures: 1, + steadyConnectionFailures: 0, + unexpectedCloses: 0, + protocolErrors: 0, + refreshErrors: 0, + socketErrors: 0 + } + const config = { + allowPlannedTransitionRetries: true, + maxRampConnectionFailures: 0, + maxUnexpectedCloses: 0 + } + assert.equal(relayLoadRunHasDisallowedFailures(result, config), false) + assert.equal( + relayLoadRunHasDisallowedFailures({ ...result, steadyConnectionFailures: 1 }, config), + true + ) +}) diff --git a/cloud/dev/scripts/relay-monitor-evidence.mjs b/cloud/dev/scripts/relay-monitor-evidence.mjs new file mode 100644 index 00000000000..7f387663f60 --- /dev/null +++ b/cloud/dev/scripts/relay-monitor-evidence.mjs @@ -0,0 +1,355 @@ +import { createHash } from 'node:crypto' +import { chmod, readFile, readdir, stat, writeFile } from 'node:fs/promises' +import { basename, join, resolve } from 'node:path' +import { pathToFileURL } from 'node:url' + +const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]{1,127}$/ +const SHA = /^[a-f0-9]{40}$/ +const JWT = /^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+$/ +const EVIDENCE_MAX_AGE_MS = 5 * 60_000 +// Matches the same-cap cell job timeout-minutes; bounds each predecessor wave. +const WAVE_PREDECESSOR_TIMEOUT_MS = 75 * 60_000 +const WAVE_INDEX = /^[0-3]$/ +const EVIDENCE_SAMPLE_INTERVAL_MS = 60_000 +const EVIDENCE_MAX_LINEAGE_MS = 25 * 60_000 +const MIGRATION_POLICIES = new Set([ + 'strict', + 'recover-forward', + 'capacity-transition' +]) +const MUTATION_MODES = new Set([ + 'capacity-transition', + 'continue-evacuation', + 'disable-cell', + 'enable-empty-cell', + 'execute', + 'fence-source', + 'recover-forward', + 'reset-empty-candidate' +]) + +function argumentsByName(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const name = argv[index] + const value = argv[index + 1] + if (!name?.startsWith('--') || !value || value.startsWith('--')) { + throw new Error('relay monitor evidence arguments are invalid') + } + values[name.slice(2)] = value + } + return values +} + +async function sha256(path) { + return createHash('sha256').update(await readFile(path)).digest('hex') +} + +async function regularFile(path) { + try { + return (await stat(path)).isFile() + } catch { + return false + } +} + +function provenance(values) { + const runAttempt = Number(values['run-attempt']) + if ( + !SAFE_ID.test(values['incident-id'] ?? '') || + !SAFE_ID.test(values['run-id'] ?? '') || + !Number.isSafeInteger(runAttempt) || + runAttempt < 1 || + !SHA.test(values['commit-sha'] ?? '') || + !['dry-run', 'monitor'].includes(values.mode) + ) { + throw new Error('relay monitor evidence provenance is invalid') + } + return { + incidentId: values['incident-id'], + runId: values['run-id'], + runAttempt, + commitSha: values['commit-sha'], + mode: values.mode + } +} + +export async function createEvidenceManifest(argv) { + const values = argumentsByName(argv) + const directory = resolve(values.directory ?? '') + const expected = provenance(values) + const candidates = [ + `${expected.incidentId}.state.json`, + `${expected.incidentId}.summaries.jsonl`, + `${expected.incidentId}.summary.md` + ] + const files = {} + for (const name of candidates) { + const path = join(directory, name) + if (await regularFile(path)) files[name] = await sha256(path) + } + if (!files[`${expected.incidentId}.state.json`]) { + throw new Error('relay monitor durable state is missing') + } + const manifest = { + schemaVersion: 1, + ...expected, + files + } + const path = join(directory, 'evidence-manifest.json') + await writeFile(path, `${JSON.stringify(manifest)}\n`, { mode: 0o600 }) + await chmod(path, 0o600) + return manifest +} + +async function readAndVerifyManifest(directory, expected) { + const manifest = JSON.parse( + await readFile(join(directory, 'evidence-manifest.json'), 'utf8') + ) + if ( + manifest.schemaVersion !== 1 || + manifest.incidentId !== expected.incidentId || + manifest.runId !== expected.runId || + manifest.runAttempt !== expected.runAttempt || + manifest.commitSha !== expected.commitSha || + manifest.mode !== expected.mode + ) { + throw new Error('relay monitor evidence provenance does not match') + } + const names = Object.keys(manifest.files ?? {}) + if (!names.includes(`${expected.incidentId}.state.json`)) { + throw new Error('relay monitor evidence has no durable state') + } + for (const name of names) { + if (basename(name) !== name || !/^[A-Za-z0-9._-]+$/.test(name)) { + throw new Error('relay monitor evidence file name is invalid') + } + if (await sha256(join(directory, name)) !== manifest.files[name]) { + throw new Error('relay monitor evidence hash does not match') + } + } + const allowed = new Set([...names, 'evidence-manifest.json']) + const unexpected = (await readdir(directory)).filter((name) => !allowed.has(name)) + if (unexpected.length > 0) throw new Error('relay monitor evidence has unexpected files') + return manifest +} + +function validMigrationPolicyState(state) { + return ( + ( + state.migrationPolicy === 'strict' && + state.recoverySourceCellId === null && + state.capacityCellId === null + ) || + ( + state.migrationPolicy === 'recover-forward' && + state.capacityCellId === null && + typeof state.recoverySourceCellId === 'string' && + state.expectedSelector?.membership?.existingOnly?.includes( + state.recoverySourceCellId + ) + ) || + ( + state.migrationPolicy === 'capacity-transition' && + state.recoverySourceCellId === null && + typeof state.capacityCellId === 'string' && + state.expectedSelector?.membership?.general?.includes(state.capacityCellId) + ) + ) +} + +export async function verifyRestoredEvidence(argv) { + const values = argumentsByName(argv) + const directory = resolve(values.directory ?? '') + const expected = provenance(values) + await readAndVerifyManifest(directory, expected) + const state = JSON.parse( + await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8') + ) + if ( + state.schemaVersion !== 4 || + state.incidentId !== expected.incidentId || + state.environment !== 'production' || + state.preDrainDryRun !== (expected.mode === 'dry-run') || + !MIGRATION_POLICIES.has(state.migrationPolicy) || + !validMigrationPolicyState(state) + ) { + throw new Error('relay monitor restored state does not match provenance') + } + return state +} + +function validCompletedDryRunState(state, expected, nowMs, maxAgeMs) { + const completedAt = Date.parse(state.completedAt) + const startedAt = Date.parse(state.startedAt) + const windowStartedAt = Date.parse(state.windowStartedAt) + const lastSampleAt = Date.parse(state.lastSampleAt) + const age = nowMs - completedAt + return ( + state.schemaVersion === 4 && + state.incidentId === expected.incidentId && + state.environment === 'production' && + state.preDrainDryRun === true && + validMigrationPolicyState(state) && + state.durationMinutes === 15 && + state.intervalMs === EVIDENCE_SAMPLE_INTERVAL_MS && + state.sampleCount >= 16 && + state.frozenAt === null && + Number.isFinite(startedAt) && + completedAt - startedAt >= 0 && + completedAt - startedAt <= EVIDENCE_MAX_LINEAGE_MS && + Number.isFinite(windowStartedAt) && + completedAt - windowStartedAt >= 15 * 60_000 && + Number.isFinite(lastSampleAt) && + lastSampleAt <= completedAt && + completedAt - lastSampleAt <= state.intervalMs && + Number.isFinite(completedAt) && + age >= 0 && + age <= maxAgeMs + ) +} + +export async function verifyDryRunAuthority(argv, now = Date.now) { + const values = argumentsByName(argv) + const directory = resolve(values.directory ?? '') + const expected = provenance(values) + if (expected.mode !== 'dry-run') throw new Error('relay mutation requires dry-run evidence') + const manifest = await readAndVerifyManifest(directory, expected) + const state = JSON.parse( + await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8') + ) + const requiredMigrationPolicy = values['required-migration-policy'] + // Later same-cap waves start after sequential predecessor cell rolls, so the + // freshness bound grows by one cell-job timeout per predecessor; single-use + // consumption, needs-chaining, and each wave's live preflight recheck keep + // holding the mutation to current health. + const waveIndex = values['wave-index'] ?? '0' + if (!WAVE_INDEX.test(waveIndex)) { + throw new Error('relay monitor wave index is invalid') + } + const maxAgeMs = + EVIDENCE_MAX_AGE_MS + Number(waveIndex) * WAVE_PREDECESSOR_TIMEOUT_MS + if ( + !MIGRATION_POLICIES.has(requiredMigrationPolicy) || + state.migrationPolicy !== requiredMigrationPolicy || + !validCompletedDryRunState(state, expected, now(), maxAgeMs) + ) { + throw new Error('relay monitor dry-run authority is incomplete or stale') + } + return { manifest, state } +} + +function exactSelector(actual, expected) { + const membership = (selector) => { + if ( + !selector?.membership || + !['existingOnly', 'migrationOnly', 'general'].every((key) => + Array.isArray(selector.membership[key]) + ) + ) return null + const normalized = Object.fromEntries( + ['existingOnly', 'migrationOnly', 'general'].map((key) => [ + key, + [...selector.membership[key]].sort() + ]) + ) + const all = Object.values(normalized).flat() + return new Set(all).size === all.length ? normalized : null + } + const actualMembership = membership(actual) + const expectedMembership = membership(expected) + return Boolean( + actualMembership && + expectedMembership && + actual?.generation === expected?.generation && + JSON.stringify(actualMembership) === JSON.stringify(expectedMembership) + ) +} + +export async function verifyMutationEvidence( + argv, + environment = process.env, + fetchImpl = fetch, + now = Date.now +) { + const values = argumentsByName(argv) + const directory = resolve(values.directory ?? '') + const expected = provenance(values) + if (expected.mode !== 'dry-run') throw new Error('mutation requires dry-run evidence') + const manifest = await readAndVerifyManifest(directory, expected) + const state = JSON.parse( + await readFile(join(directory, `${expected.incidentId}.state.json`), 'utf8') + ) + const mutationMode = values['mutation-mode'] + if (!MUTATION_MODES.has(mutationMode)) { + throw new Error('relay monitor mutation mode is invalid') + } + const scopedRecoverySourceCellId = + values['scoped-recovery-source-cell-id'] + const recoveryMutation = ['fence-source', 'recover-forward'].includes(mutationMode) + const scopedRecoveryMutation = + ['execute', 'recover-forward'].includes(mutationMode) && + Boolean(scopedRecoverySourceCellId) + if (scopedRecoverySourceCellId && !scopedRecoveryMutation) { + throw new Error('relay monitor scoped recovery evidence is invalid') + } + const requiredMigrationPolicy = mutationMode === 'capacity-transition' + ? 'capacity-transition' + : recoveryMutation || scopedRecoveryMutation ? 'recover-forward' : 'strict' + if (state.migrationPolicy !== requiredMigrationPolicy) { + throw new Error('relay monitor migration policy does not match mutation') + } + const expectedRecoverySourceCellId = scopedRecoveryMutation + ? scopedRecoverySourceCellId + : values['source-cell-id'] + if ( + (recoveryMutation || scopedRecoveryMutation) && + state.recoverySourceCellId !== expectedRecoverySourceCellId + ) { + throw new Error('relay monitor recovery source does not match mutation') + } + if ( + mutationMode === 'capacity-transition' && + state.capacityCellId !== values['source-cell-id'] + ) { + throw new Error('relay monitor capacity cell does not match mutation') + } + if ( + !validCompletedDryRunState(state, expected, now(), EVIDENCE_MAX_AGE_MS) + ) { + throw new Error('relay monitor dry-run evidence is incomplete or stale') + } + const token = environment.ORCA_RELAY_ADMIN_ID_TOKEN + const origin = values['director-origin'] + if (!token || !JWT.test(token) || !origin?.startsWith('https://')) { + throw new Error('relay monitor live selector verification is unavailable') + } + const response = await fetchImpl(`${origin}/v1/admin/admission-selector/status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1 }), + signal: AbortSignal.timeout(30_000) + }) + if (!response.ok) throw new Error('relay monitor live selector verification failed') + const current = (await response.json()).selector + if (!exactSelector(current, state.expectedSelector)) { + throw new Error('relay admission selector changed after the dry run') + } + return { manifest, state } +} + +async function main() { + const [command, ...argv] = process.argv.slice(2) + if (command === 'create') await createEvidenceManifest(argv) + else if (command === 'verify-restore') await verifyRestoredEvidence(argv) + else if (command === 'verify-authority') await verifyDryRunAuthority(argv) + else if (command === 'verify-mutation') await verifyMutationEvidence(argv) + else throw new Error('relay monitor evidence command is invalid') +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + console.error(error instanceof Error ? error.message : 'relay monitor evidence failed') + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/relay-monitor-evidence.test.mjs b/cloud/dev/scripts/relay-monitor-evidence.test.mjs new file mode 100644 index 00000000000..45116761119 --- /dev/null +++ b/cloud/dev/scripts/relay-monitor-evidence.test.mjs @@ -0,0 +1,515 @@ +import assert from 'node:assert/strict' +import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import test from 'node:test' +import { relayWorkflowPath, relayWorkflowUrl } from './relay-repository.mjs' +import { + createEvidenceManifest, + verifyDryRunAuthority, + verifyMutationEvidence, + verifyRestoredEvidence +} from './relay-monitor-evidence.mjs' + +const now = Date.parse('2026-07-28T12:00:00.000Z') +const provenance = [ + '--incident-id', + 'relay-123', + '--run-id', + '123', + '--run-attempt', + '1', + '--commit-sha', + 'a'.repeat(40), + '--mode', + 'dry-run' +] +const selector = { + generation: 2, + membership: { + existingOnly: ['c1'], + migrationOnly: ['c2'], + general: ['c3'] + } +} + +async function evidenceDirectory(migrationPolicy = 'strict') { + const directory = await mkdtemp(join(tmpdir(), 'relay-monitor-evidence-')) + const state = { + schemaVersion: 4, + incidentId: 'relay-123', + environment: 'production', + preDrainDryRun: true, + migrationPolicy, + recoverySourceCellId: migrationPolicy === 'recover-forward' ? 'c1' : null, + capacityCellId: migrationPolicy === 'capacity-transition' ? 'c3' : null, + startedAt: new Date(now - 17 * 60_000).toISOString(), + durationMinutes: 15, + intervalMs: 60_000, + sampleCount: 16, + windowStartedAt: new Date(now - 16 * 60_000).toISOString(), + lastSampleAt: new Date(now - 60_007).toISOString(), + completedAt: new Date(now - 60_000).toISOString(), + frozenAt: null, + expectedSelector: selector + } + await writeFile( + join(directory, 'relay-123.state.json'), + `${JSON.stringify(state)}\n` + ) + return directory +} + +test('creates and verifies exact restart provenance and hashes', async () => { + const directory = await evidenceDirectory() + try { + await createEvidenceManifest(['--directory', directory, ...provenance]) + await assert.doesNotReject( + verifyRestoredEvidence(['--directory', directory, ...provenance]) + ) + await writeFile(join(directory, 'relay-123.state.json'), '{}\n') + await assert.rejects( + verifyRestoredEvidence(['--directory', directory, ...provenance]), + /hash does not match/ + ) + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) + +test('requires fresh green evidence and rechecks the live selector', async () => { + const directory = await evidenceDirectory() + try { + await createEvidenceManifest(['--directory', directory, ...provenance]) + const verifyAuthority = () => verifyDryRunAuthority( + [ + '--directory', + directory, + ...provenance, + '--required-migration-policy', + 'strict' + ], + () => now + ) + await assert.doesNotReject(verifyAuthority()) + const fetchImpl = async (_input, init) => { + assert.equal( + new Headers(init.headers).get('authorization'), + 'Bearer aaa.bbb.ccc' + ) + return Response.json({ selector }) + } + await assert.doesNotReject( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenance, + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ) + ) + const statePath = join(directory, 'relay-123.state.json') + const state = JSON.parse(await readFile(statePath, 'utf8')) + state.completedAt = new Date(now - 300_001).toISOString() + await writeFile(statePath, `${JSON.stringify(state)}\n`) + await createEvidenceManifest(['--directory', directory, ...provenance]) + await assert.rejects( + verifyAuthority(), + /authority is incomplete or stale/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenance, + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ), + /incomplete or stale/ + ) + state.completedAt = new Date(now - 60_000).toISOString() + state.lastSampleAt = new Date(now - 120_001).toISOString() + await writeFile(statePath, `${JSON.stringify(state)}\n`) + await createEvidenceManifest(['--directory', directory, ...provenance]) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenance, + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ), + /incomplete or stale/ + ) + state.lastSampleAt = new Date(now - 60_007).toISOString() + state.startedAt = new Date(now - 26 * 60_000 - 1).toISOString() + await writeFile(statePath, `${JSON.stringify(state)}\n`) + await createEvidenceManifest(['--directory', directory, ...provenance]) + await assert.rejects( + verifyAuthority(), + /authority is incomplete or stale/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + directory, + ...provenance, + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ), + /incomplete or stale/ + ) + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) + +test('later same-cap waves accept evidence aged by predecessor cell rolls', async () => { + const directory = await evidenceDirectory() + try { + const statePath = join(directory, 'relay-123.state.json') + const state = JSON.parse(await readFile(statePath, 'utf8')) + const authorityAt = (...waveArgs) => verifyDryRunAuthority( + [ + '--directory', + directory, + ...provenance, + '--required-migration-policy', + 'strict', + ...waveArgs.flatMap((waveIndex) => ['--wave-index', waveIndex]) + ], + () => now + ) + const ageState = async (ageMs) => { + state.completedAt = new Date(now - ageMs).toISOString() + state.lastSampleAt = new Date(now - ageMs - 7).toISOString() + state.startedAt = new Date(now - ageMs - 17 * 60_000).toISOString() + state.windowStartedAt = new Date(now - ageMs - 16 * 60_000).toISOString() + await writeFile(statePath, `${JSON.stringify(state)}\n`) + await createEvidenceManifest(['--directory', directory, ...provenance]) + } + // The wave-0 bound in isolation: exactly 5 minutes, flag or no flag. + await ageState(5 * 60_000) + await assert.doesNotReject(authorityAt()) + await assert.doesNotReject(authorityAt('0')) + await ageState(5 * 60_000 + 1) + await assert.rejects(authorityAt(), /authority is incomplete or stale/) + await assert.rejects(authorityAt('0'), /authority is incomplete or stale/) + // One predecessor cell roll (~16 min) exceeds wave 0 but fits wave 1. + await ageState(17 * 60_000) + await assert.rejects(authorityAt('0'), /authority is incomplete or stale/) + await assert.doesNotReject(authorityAt('1')) + await assert.rejects(authorityAt('4'), /wave index is invalid/) + await assert.rejects(authorityAt('x'), /wave index is invalid/) + // Both edges of one predecessor job timeout: 5min + 75min exactly. + await ageState(80 * 60_000) + await assert.doesNotReject(authorityAt('1')) + await ageState(80 * 60_000 + 1) + await assert.rejects(authorityAt('1'), /authority is incomplete or stale/) + await assert.doesNotReject(authorityAt('2')) + // Wave 2 and wave 3 edges: 5min + 2 * 75min and 5min + 3 * 75min exactly. + await ageState(155 * 60_000) + await assert.doesNotReject(authorityAt('2')) + await ageState(155 * 60_000 + 1) + await assert.rejects(authorityAt('2'), /authority is incomplete or stale/) + await ageState(230 * 60_000) + await assert.doesNotReject(authorityAt('3')) + await ageState(230 * 60_000 + 1) + await assert.rejects(authorityAt('3'), /authority is incomplete or stale/) + } finally { + await rm(directory, { recursive: true, force: true }) + } +}) + +test('binds migration policies to their exact mutations', async () => { + const strictDirectory = await evidenceDirectory() + const recoveryDirectory = await evidenceDirectory('recover-forward') + const capacityDirectory = await evidenceDirectory('capacity-transition') + const fetchImpl = async () => Response.json({ selector }) + try { + await createEvidenceManifest(['--directory', strictDirectory, ...provenance]) + await createEvidenceManifest(['--directory', recoveryDirectory, ...provenance]) + await createEvidenceManifest(['--directory', capacityDirectory, ...provenance]) + await assert.rejects( + verifyDryRunAuthority( + [ + '--directory', + recoveryDirectory, + ...provenance, + '--required-migration-policy', + 'strict' + ], + () => now + ), + /authority is incomplete or stale/ + ) + const verify = (directory, mutationMode, sourceCellId = 'c1') => verifyMutationEvidence( + [ + '--directory', + directory, + ...provenance, + '--mutation-mode', + mutationMode, + '--source-cell-id', + sourceCellId, + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ) + await assert.doesNotReject(verify(strictDirectory, 'execute')) + await assert.doesNotReject( + verify(capacityDirectory, 'capacity-transition', 'c3') + ) + await assert.doesNotReject(verify(recoveryDirectory, 'recover-forward')) + await assert.doesNotReject(verify(recoveryDirectory, 'fence-source')) + await assert.doesNotReject( + verifyMutationEvidence( + [ + '--directory', + recoveryDirectory, + ...provenance, + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c12', + '--scoped-recovery-source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ) + ) + await assert.doesNotReject( + verifyMutationEvidence( + [ + '--directory', + recoveryDirectory, + ...provenance, + '--mutation-mode', + 'recover-forward', + '--source-cell-id', + 'c12', + '--scoped-recovery-source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ) + ) + await assert.rejects( + verify(recoveryDirectory, 'execute'), + /migration policy does not match/ + ) + await assert.rejects( + verify(strictDirectory, 'recover-forward'), + /migration policy does not match/ + ) + await assert.rejects( + verify(strictDirectory, 'capacity-transition', 'c3'), + /migration policy does not match/ + ) + await assert.rejects( + verify(capacityDirectory, 'capacity-transition', 'c1'), + /capacity cell does not match/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + recoveryDirectory, + ...provenance, + '--mutation-mode', + 'recover-forward', + '--source-cell-id', + 'c9', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ), + /recovery source does not match/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + recoveryDirectory, + ...provenance, + '--mutation-mode', + 'fence-source', + '--source-cell-id', + 'c1', + '--scoped-recovery-source-cell-id', + 'c1', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ), + /scoped recovery evidence is invalid/ + ) + await assert.rejects( + verifyMutationEvidence( + [ + '--directory', + recoveryDirectory, + ...provenance, + '--mutation-mode', + 'execute', + '--source-cell-id', + 'c12', + '--scoped-recovery-source-cell-id', + 'c9', + '--director-origin', + 'https://relay.example' + ], + { ORCA_RELAY_ADMIN_ID_TOKEN: 'aaa.bbb.ccc' }, + fetchImpl, + () => now + ), + /recovery source does not match/ + ) + } finally { + await rm(strictDirectory, { recursive: true, force: true }) + await rm(recoveryDirectory, { recursive: true, force: true }) + await rm(capacityDirectory, { recursive: true, force: true }) + } +}) + +test('workflow reruns restore the prior attempt into one stable incident', async () => { + const workflow = await readFile( + relayWorkflowUrl('monitor-relay-production-job.yml'), + 'utf8' + ) + assert.match(workflow, /INCIDENT_ID: relay-\$\{\{ github\.run_id \}\}-\$\{\{ inputs\.mode \}\}/) + assert.doesNotMatch(workflow, /INCIDENT_ID:.*run_attempt/) + assert.match(workflow, /actions\/download-artifact@v4/) + assert.match(workflow, /verify-restore/) + assert.match(workflow, /RESTART_FLAG=--restart/) + assert.equal(workflow.match(/--capacity-cell-id/g)?.length, 3) + const dispatchWorkflow = await readFile( + relayWorkflowUrl('monitor-relay-production.yml'), + 'utf8' + ) + assert.match(dispatchWorkflow, /- capacity-transition/) + assert.match(dispatchWorkflow, /capacity-cell-id: \$\{\{ inputs\.capacity-cell-id \}\}/) +}) + +test('same-cap and rehome mutations require complete strict dry-run authority', async () => { + for (const name of [ + 'deploy-relay-production-same-cap-job.yml', + 'operate-relay-production-rehome-job.yml' + ]) { + const workflow = await readFile( + relayWorkflowUrl(name), + 'utf8' + ) + assert.match(workflow, /relay-monitor-evidence\.mjs verify-authority/) + assert.match(workflow, /--required-migration-policy strict/) + assert.doesNotMatch(workflow, /relay-monitor-evidence\.mjs verify-restore/) + } +}) + +test('production mutation workflows consume and live-recheck dry-run evidence', async () => { + for (const name of [ + 'deploy-relay-production.yml', + 'deploy-relay-production-multi-target.yml' + ]) { + const workflow = await readFile( + relayWorkflowUrl(name), + 'utf8' + ) + assert.match(workflow, /actions\/download-artifact@v4/) + assert.match(workflow, /verify-mutation/) + assert.match(workflow, /--mutation-mode "\$\{DEPLOY_MODE\}"/) + assert.match(workflow, /--source-cell-id "\$\{SOURCE_CELL_ID\}"/) + assert.match(workflow, /incident:relay-preflight/) + assert.match(workflow, /Reject previously consumed dry-run evidence/) + assert.match(workflow, /actions\/upload-artifact@v4/) + assert.match(workflow, /relay-monitor-consumed-/) + assert.match(workflow, /ORCA_RELAY_ADMIN_ID_TOKEN/) + assert.match(workflow, /github\.ref == 'refs\/heads\/main'/) + assert.ok( + workflow.indexOf('pnpm install --frozen-lockfile') < + workflow.indexOf('id: google-auth') + ) + } +}) + +test('monitor and mutation workflows share the production Cloud SQL rollout lock', async () => { + for (const name of [ + 'monitor-relay-production.yml', + 'deploy-relay-production.yml', + 'deploy-relay-production-multi-target.yml', + 'deploy-relay-production-capacity.yml' + ]) { + const workflow = await readFile( + relayWorkflowUrl(name), + 'utf8' + ) + assert.match(workflow, /group: production-cloud-sql-rollout/) + } +}) + +test('monitor uses a reusable job so exact job_workflow_ref is present', async () => { + const wrapper = await readFile( + relayWorkflowUrl('monitor-relay-production.yml'), + 'utf8' + ) + assert.ok(wrapper.includes(`uses: ./${relayWorkflowPath('monitor-relay-production-job.yml')}`)) + const job = await readFile( + relayWorkflowUrl('monitor-relay-production-job.yml'), + 'utf8' + ) + assert.match(job, /workflow_call:/) + assert.match(job, /environment: production/) +}) diff --git a/cloud/dev/scripts/relay-production-capacity-wave.mjs b/cloud/dev/scripts/relay-production-capacity-wave.mjs new file mode 100644 index 00000000000..e76c264e5ed --- /dev/null +++ b/cloud/dev/scripts/relay-production-capacity-wave.mjs @@ -0,0 +1,172 @@ +import { readFile, writeFile } from 'node:fs/promises' +import { pathToFileURL } from 'node:url' +import { PRODUCTION_CAPACITY_CELL_IDS } from './prepare-relay-production-capacity-canary.mjs' + +const APPROVED_CELLS = new Set(PRODUCTION_CAPACITY_CELL_IDS) + +function argumentsByName(argv, allowed) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const argument = argv[index] + const value = argv[index + 1] + if (!argument?.startsWith('--') || !value || value.startsWith('--')) { + throw new Error('capacity wave arguments are invalid') + } + const name = argument.slice(2) + if (!allowed.has(name) || name in values) { + throw new Error(`capacity wave argument --${name} is invalid`) + } + values[name] = value + } + return values +} + +export function parseCapacityWave(waveCellIds, confirmation) { + const cells = waveCellIds.split(',') + if ( + cells.length < 2 || + cells.length > 4 || + cells.some((cell) => !APPROVED_CELLS.has(cell)) || + new Set(cells).size !== cells.length || + cells.join(',') !== waveCellIds + ) { + throw new Error('capacity wave must contain two to four unique approved cells') + } + if (confirmation !== `RAISE_SELECTED_WAVE_TO_1000 ${waveCellIds}`) { + throw new Error('capacity wave confirmation does not match the selected cells') + } + return cells +} + +function exactMembership(membership) { + const keys = ['existingOnly', 'migrationOnly', 'general'] + if (!keys.every((key) => Array.isArray(membership?.[key]))) return false + const cells = keys.flatMap((key) => membership[key]) + return cells.every((cell) => typeof cell === 'string') && new Set(cells).size === cells.length +} + +export function capacityWavePreflightState(state, waveCellIds, waveIndex, targetCellId) { + const cells = parseCapacityWave( + waveCellIds, + `RAISE_SELECTED_WAVE_TO_1000 ${waveCellIds}` + ) + const index = Number(waveIndex) + const membership = state.expectedSelector?.membership + if ( + !Number.isSafeInteger(index) || + index < 0 || + index >= cells.length || + targetCellId !== cells[index] || + state.schemaVersion !== 4 || + state.environment !== 'production' || + state.preDrainDryRun !== true || + state.migrationPolicy !== 'capacity-transition' || + state.recoverySourceCellId !== null || + state.capacityCellId !== cells[0] || + !Number.isSafeInteger(state.expectedSelector?.generation) || + !exactMembership(membership) || + !cells.every((cell) => membership.general.includes(cell)) || + state.sampleCount < 16 || + state.frozenAt !== null || + typeof state.completedAt !== 'string' + ) { + throw new Error('capacity wave evidence does not match this step') + } + return { + schemaVersion: 4, + environment: 'production', + expectedSelector: { + generation: state.expectedSelector.generation + index * 2, + membership + }, + migrationPolicy: 'capacity-transition', + recoverySourceCellId: null, + capacityCellId: targetCellId + } +} + +export function capacityWaveResumePreflightState(state, waveCellIds, targetCellId) { + const cells = parseCapacityWave( + waveCellIds, + `RAISE_SELECTED_WAVE_TO_1000 ${waveCellIds}` + ) + const index = cells.indexOf(targetCellId) + const base = capacityWavePreflightState( + state, + waveCellIds, + String(index), + targetCellId + ) + const membership = base.expectedSelector.membership + const general = membership.general.filter((cell) => cell !== targetCellId) + const capacityCellId = general.includes(state.capacityCellId) + ? state.capacityCellId + : cells.find((cell) => general.includes(cell)) + if (!capacityCellId) throw new Error('capacity wave resume has no general evidence cell') + return { + ...base, + expectedSelector: { + generation: base.expectedSelector.generation + 1, + membership: { + existingOnly: membership.existingOnly, + migrationOnly: [...membership.migrationOnly, targetCellId].sort(), + general + } + }, + capacityCellId + } +} + +async function main(argv) { + const [command, ...arguments_] = argv + if (command === 'validate') { + const values = argumentsByName( + arguments_, + new Set(['wave-cell-ids', 'confirmation']) + ) + process.stdout.write(`${JSON.stringify(parseCapacityWave( + values['wave-cell-ids'] ?? '', + values.confirmation ?? '' + ))}\n`) + return + } + if (command === 'build-preflight' || command === 'build-resume-preflight') { + const values = argumentsByName( + arguments_, + new Set([ + 'state-file', + 'wave-cell-ids', + ...(command === 'build-preflight' ? ['wave-index'] : []), + 'target-cell-id', + 'output-file' + ]) + ) + const state = JSON.parse(await readFile(values['state-file'] ?? '', 'utf8')) + const preflight = command === 'build-preflight' + ? capacityWavePreflightState( + state, + values['wave-cell-ids'] ?? '', + values['wave-index'] ?? '', + values['target-cell-id'] ?? '' + ) + : capacityWaveResumePreflightState( + state, + values['wave-cell-ids'] ?? '', + values['target-cell-id'] ?? '' + ) + await writeFile( + values['output-file'] ?? '', + `${JSON.stringify(preflight)}\n`, + { mode: 0o600 } + ) + return + } + throw new Error('capacity wave command is invalid') +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + main(process.argv.slice(2)).catch((error) => { + console.error(error instanceof Error ? error.message : 'capacity wave failed') + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/relay-production-capacity-wave.test.mjs b/cloud/dev/scripts/relay-production-capacity-wave.test.mjs new file mode 100644 index 00000000000..2cb8ebfa538 --- /dev/null +++ b/cloud/dev/scripts/relay-production-capacity-wave.test.mjs @@ -0,0 +1,129 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + capacityWavePreflightState, + capacityWaveResumePreflightState, + parseCapacityWave +} from './relay-production-capacity-wave.mjs' + +const wave = [ + 'production-gce-c22', + 'production-gce-c21', + 'production-gce-c20', + 'production-gce-c19' +] + +function evidence(overrides = {}) { + return { + schemaVersion: 4, + environment: 'production', + preDrainDryRun: true, + migrationPolicy: 'capacity-transition', + recoverySourceCellId: null, + capacityCellId: wave[0], + expectedSelector: { + generation: 39, + membership: { + existingOnly: ['production-gce-c1'], + migrationOnly: ['production-gce-c17', 'production-gce-c18'], + general: [...wave, 'production-gce-c23'] + } + }, + sampleCount: 16, + frozenAt: null, + completedAt: '2026-08-11T21:00:00.000Z', + ...overrides + } +} + +test('accepts only exact confirmed waves of two to four approved cells', () => { + for (const cells of [wave.slice(0, 2), wave]) { + const value = cells.join(',') + assert.deepEqual( + parseCapacityWave(value, `RAISE_SELECTED_WAVE_TO_1000 ${value}`), + cells + ) + } + for (const [cells, confirmation] of [ + [[wave[0]], `RAISE_SELECTED_WAVE_TO_1000 ${wave[0]}`], + [[...wave, 'production-gce-c16'], `RAISE_SELECTED_WAVE_TO_1000 ${wave.join(',')}`], + [[wave[0], wave[0]], `RAISE_SELECTED_WAVE_TO_1000 ${wave[0]},${wave[0]}`], + [[wave[0], 'production-gce-c17'], `RAISE_SELECTED_WAVE_TO_1000 ${wave[0]},production-gce-c17`], + [[wave[0], ` ${wave[1]}`], `RAISE_SELECTED_WAVE_TO_1000 ${wave[0]}, ${wave[1]}`], + [wave, 'RAISE_SELECTED_WAVE_TO_1000 production-gce-c22'] + ]) { + assert.throws(() => parseCapacityWave(cells.join(','), confirmation)) + } +}) + +test('derives each continuation preflight from the sealed selector generation', () => { + for (const [index, cell] of wave.entries()) { + const state = capacityWavePreflightState( + evidence(), + wave.join(','), + String(index), + cell + ) + assert.equal(state.expectedSelector.generation, 39 + index * 2) + assert.equal(state.capacityCellId, cell) + assert.deepEqual(state.expectedSelector.membership, evidence().expectedSelector.membership) + } +}) + +test('derives an exact isolated-cell resume state from sealed wave evidence', () => { + const state = capacityWaveResumePreflightState( + evidence(), + wave.join(','), + wave[3] + ) + assert.equal(state.expectedSelector.generation, 46) + assert.equal(state.capacityCellId, wave[0]) + assert.deepEqual(state.expectedSelector.membership, { + existingOnly: ['production-gce-c1'], + migrationOnly: ['production-gce-c17', 'production-gce-c18', wave[3]].sort(), + general: [wave[0], wave[1], wave[2], 'production-gce-c23'] + }) +}) + +test('resume rejects a target outside the exact sealed wave', () => { + assert.throws( + () => capacityWaveResumePreflightState( + evidence(), + wave.join(','), + 'production-gce-c16' + ), + /does not match/ + ) +}) + +test('resume rebinds first-cell evidence to another general wave cell', () => { + const state = capacityWaveResumePreflightState( + evidence(), + wave.join(','), + wave[0] + ) + assert.equal(state.capacityCellId, wave[1]) + assert.equal(state.expectedSelector.generation, 40) + assert.ok(state.expectedSelector.membership.migrationOnly.includes(wave[0])) + assert.ok(state.expectedSelector.membership.general.includes(wave[1])) +}) + +test('rejects reordered, incomplete, frozen, or mismatched wave evidence', () => { + const calls = [ + () => capacityWavePreflightState(evidence(), wave.join(','), '1', wave[0]), + () => capacityWavePreflightState(evidence({ capacityCellId: wave[1] }), wave.join(','), '0', wave[0]), + () => capacityWavePreflightState(evidence({ sampleCount: 15 }), wave.join(','), '0', wave[0]), + () => capacityWavePreflightState(evidence({ frozenAt: '2026-08-11T20:59:00.000Z' }), wave.join(','), '0', wave[0]), + () => capacityWavePreflightState(evidence({ completedAt: null }), wave.join(','), '0', wave[0]), + () => capacityWavePreflightState(evidence({ + expectedSelector: { + ...evidence().expectedSelector, + membership: { + ...evidence().expectedSelector.membership, + general: wave.slice(1) + } + } + }), wave.join(','), '0', wave[0]) + ] + for (const call of calls) assert.throws(call, /does not match/) +}) diff --git a/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs b/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs new file mode 100644 index 00000000000..1c9e3aee5f6 --- /dev/null +++ b/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs @@ -0,0 +1,440 @@ +import assert from 'node:assert/strict' +import { spawnSync } from 'node:child_process' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { readRelayWorkflow, relayWorkflowPath } from './relay-repository.mjs' +import { PRODUCTION_CAPACITY_CELL_IDS } from './prepare-relay-production-capacity-canary.mjs' + +const dispatchWorkflow = readRelayWorkflow('deploy-relay-production-capacity.yml') +const workflow = readRelayWorkflow('deploy-relay-production-capacity-job.yml') +const terraform = source('infra/terraform/relay-github-actions.tf') +const production = source('infra/terraform/environments/production.tfvars') +const capacityCells = PRODUCTION_CAPACITY_CELL_IDS + +function source(path) { + return readFileSync(new URL(`../../${path}`, import.meta.url), 'utf8') +} + +function resource(type, name) { + const start = terraform.indexOf(`resource "${type}" "${name}"`) + assert.notEqual(start, -1, `${type}.${name} is missing`) + const next = terraform.indexOf('\nresource "', start + 1) + return terraform.slice(start, next === -1 ? undefined : next) +} + +function ordered(...markers) { + let previous = -1 + for (const marker of markers) { + const current = workflow.indexOf(marker) + assert.ok(current > previous, `${marker} is missing or out of order`) + previous = current + } +} + +function mutationConfirmation(mode, targetCellId, confirmation) { + const stepStart = workflow.indexOf(' - name: Require exact mutation confirmation') + const runMarker = ' run: |\n' + const runStart = workflow.indexOf(runMarker, stepStart) + runMarker.length + const runEnd = workflow.indexOf('\n - name:', runStart) + const script = workflow.slice(runStart, runEnd).replace(/^ {10}/gm, '') + return spawnSync('bash', ['-euo', 'pipefail', '-c', script], { + env: { + ...process.env, + DEPLOY_MODE: mode, + TARGET_CELL_ID: targetCellId, + CONFIRMATION: confirmation + } + }).status +} + +function imageCompatibility(activeImage, desiredImage) { + const start = workflow.indexOf(' ACTIVE_IMAGE_DIGEST="${ACTIVE_IMAGE##*@}"') + const end = workflow.indexOf('\n CURRENT_CELLS_JSON=', start) + const script = workflow.slice(start, end).replace(/^ {10}/gm, '') + return spawnSync('bash', ['-euo', 'pipefail', '-c', script], { + env: { + ...process.env, + ACTIVE_IMAGE: activeImage, + DESIRED_IMAGE: desiredImage, + DESIRED_IMAGE_DIGEST: desiredImage.split('@').at(-1), + COMPATIBLE_DIRECTOR_IMAGE_DIGEST: 'sha256:01b7fc3e6dce66180034f268a2dc92c05458706c5b3a0dc4450dcdd6161f6e73', + COMPATIBLE_CELL_IMAGE_DIGEST: 'sha256:c77ec7aef565009fdb645b0989806859bfa40a7aa14e4a57ab55ac92fee6c34f' + } + }).status +} + +test('production capacity mutation is restricted to the exact serving rollout set', () => { + assert.match(workflow, /TARGET_CELL_ID: \$\{\{ inputs\.target-cell-id \}\}/) + const targetInput = dispatchWorkflow.slice( + dispatchWorkflow.indexOf(' target-cell-id:'), + dispatchWorkflow.indexOf(' wave-cell-ids:') + ) + assert.deepEqual( + [...targetInput.matchAll(/^\s+- (production-gce-c\d+)$/gm)].map((match) => match[1]), + capacityCells + ) + assert.match(workflow, new RegExp(`CAPACITY_CELL_IDS: ${capacityCells.join(',')}`)) + for (const cellId of capacityCells) { + assert.match(dispatchWorkflow, new RegExp(`^\\s+- ${cellId}$`, 'm')) + } + assert.match(workflow, /CELL_ORIGIN="https:\/\/\$\{TARGET_HOSTNAME\}\.relay\.onorca\.dev"/) + assert.match(workflow, /echo "TARGET_HOSTNAME=\$\{TARGET_HOSTNAME\}"/) + assert.match(workflow, /\} >> "\$\{GITHUB_ENV\}"/) + assert.match(workflow, /RAISE_SELECTED_CELL_TO_1000/) + assert.match(workflow, /ROLL_BACK_SELECTED_CELL_TO_600/) + assert.match(workflow, /TARGET_HARD_CAP=600[\s\S]*?else[\s\S]*?TARGET_HARD_CAP=1000/) +}) + +test('rollback confirmation is bound to the exact selected cell', () => { + assert.equal( + mutationConfirmation('apply', 'production-gce-c25', 'RAISE_SELECTED_CELL_TO_1000'), + 0 + ) + assert.equal( + mutationConfirmation( + 'rollback', + 'production-gce-c25', + 'ROLL_BACK_SELECTED_CELL_TO_600 production-gce-c25' + ), + 0 + ) + assert.notEqual( + mutationConfirmation('rollback', 'production-gce-c25', 'ROLL_BACK_SELECTED_CELL_TO_600'), + 0 + ) + assert.notEqual( + mutationConfirmation( + 'rollback', + 'production-gce-c25', + 'ROLL_BACK_SELECTED_CELL_TO_600 production-gce-c26' + ), + 0 + ) +}) + +test('production configuration selects only serving cells for 1,000 and the compatible image', () => { + const cell = (cellId) => production.slice( + production.indexOf(`"${cellId}"`), + production.indexOf('\n }', production.indexOf(`"${cellId}"`)) + ) + for (const cellId of capacityCells) { + assert.match(cell(cellId), /connection_hard_cap\s+= 1000/) + assert.match( + cell(cellId), + /sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563/ + ) + } + for (const cellId of ['production-gce-c17', 'production-gce-c18']) { + assert.match(cell(cellId), /connection_hard_cap\s+= 600/) + assert.doesNotMatch(cell(cellId), /connection_hard_cap\s+= 1000/) + assert.match(cell(cellId), /sha256:0e83408b0dc08531f1e8182019dc151afc38d63ddde4ad5cc01e40247ef3681d/) + } + assert.equal(production.match(/connection_hard_cap\s+= 1000/g)?.length, capacityCells.length) + // Asia cells legitimately share this digest, so scope the uniqueness check to the capacity set. + assert.equal( + new Set(capacityCells.map((cellId) => cell(cellId).match(/sha256:[0-9a-f]{64}/)[0])).size, + 1 + ) + assert.match( + workflow, + /COMPATIBLE_DIRECTOR_IMAGE_DIGEST: sha256:01b7fc3e6dce66180034f268a2dc92c05458706c5b3a0dc4450dcdd6161f6e73/ + ) + assert.match( + workflow, + /test "\$\{ACTIVE_IMAGE_DIGEST\}" = "\$\{COMPATIBLE_DIRECTOR_IMAGE_DIGEST\}"/ + ) + assert.match( + workflow, + /test "\$\{DESIRED_IMAGE_DIGEST\}" = "\$\{COMPATIBLE_CELL_IMAGE_DIGEST\}"/ + ) + assert.match(workflow, /\.\[\$cell\]\.connection_hard_cap = \$cap/) + assert.match(workflow, /baseCells:\$baseCells/) + assert.match(workflow, /capacityCellIds:\(\$capacityCellIds \| split\(","\)\)/) + assert.match( + workflow, + /Verify current selected-cell capacity[\s\S]*?TOPOLOGY_PHASE.*predecessor[\s\S]*?CURRENT_CAP=600[\s\S]*?DESIRED_IMAGE_DIGEST.*PREDECESSOR_IMAGE_DIGEST/ + ) +}) + +test('director and cell image compatibility is an exact reviewed pair', () => { + const repository = 'us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@' + const director = `${repository}sha256:01b7fc3e6dce66180034f268a2dc92c05458706c5b3a0dc4450dcdd6161f6e73` + const cell = `${repository}sha256:c77ec7aef565009fdb645b0989806859bfa40a7aa14e4a57ab55ac92fee6c34f` + const other = `${repository}sha256:${'a'.repeat(64)}` + assert.equal(imageCompatibility(cell, cell), 0) + assert.equal(imageCompatibility(director, cell), 0) + assert.notEqual(imageCompatibility(director, other), 0) + assert.notEqual(imageCompatibility(other, cell), 0) +}) + +// The matching check on google_project_service.required lives with the foundation root, which +// stays in the private repository. + +test('apply consumes fresh evidence before arming mutation cleanup', () => { + for (const marker of [ + 'Require fresh dry-run evidence reference', + 'Verify dry-run artifact before cloud authentication', + 'Reject previously consumed dry-run evidence', + 'Verify fresh dry-run evidence against the live selector', + 'Recheck every live safety signal', + 'Publish the consumed-evidence marker' + ]) { + assert.match(workflow, new RegExp(marker)) + } + assert.match(workflow, /--mutation-mode capacity-transition/) + assert.match(workflow, /--source-cell-id "\$\{TARGET_CELL_ID\}"/) + ordered( + 'Publish the consumed-evidence marker', + 'Arm fail-closed mutation cleanup', + 'Reversibly isolate only the selected cell', + 'Deploy only the reviewed director topology', + 'Plan and apply only the empty selected cell', + 'Restore only the selected cell to general admission', + 'Verify the live general selected cell' + ) +}) + +test('wave apply consumes one proof and runs fail-closed cells sequentially', () => { + assert.match(dispatchWorkflow, /- wave-apply/) + assert.match(dispatchWorkflow, /group: production-cloud-sql-rollout/) + assert.match(dispatchWorkflow, /Validate the exact wave request/) + assert.match(dispatchWorkflow, /Verify wave evidence against the live selector/) + assert.match(dispatchWorkflow, /Require exact 600\/60 predecessor wave cells/) + assert.match( + dispatchWorkflow, + /COMPATIBLE_CELL_IMAGE_DIGEST: sha256:c77ec7aef565009fdb645b0989806859bfa40a7aa14e4a57ab55ac92fee6c34f/ + ) + assert.match( + dispatchWorkflow, + /Require exact 600\/60 predecessor wave cells[\s\S]*?--expected-image-digests \\\n\s+"\$\{PREDECESSOR_IMAGE_DIGEST\},\$\{COMPATIBLE_CELL_IMAGE_DIGEST\}"/ + ) + assert.match(dispatchWorkflow, /Publish the consumed-evidence marker/) + assert.match( + dispatchWorkflow, + /OUTPUT_DIRECTORY: \$\{\{ github\.workspace \}\}\/relay-monitor-evidence/ + ) + assert.match( + dispatchWorkflow, + /path: \$\{\{ github\.workspace \}\}\/relay-monitor-evidence/ + ) + assert.doesNotMatch(dispatchWorkflow, /strategy:/) + for (const [index, dependency] of [ + [1, 'wave_gate'], + [2, 'wave_cell_1'], + [3, 'wave_cell_2'], + [4, 'wave_cell_3'] + ]) { + const start = dispatchWorkflow.indexOf(` wave_cell_${index}:`) + const end = dispatchWorkflow.indexOf(`\n wave_cell_${index + 1}:`, start) + const job = dispatchWorkflow.slice(start, end === -1 ? undefined : end) + assert.match(job, new RegExp(`needs: (?:\\[wave_gate, )?${dependency}`)) + assert.match(job, /evidence-mode: continuation/) + assert.match(job, new RegExp(`wave-index: '${index - 1}'`)) + } + assert.match(workflow, /Download this workflow's wave authority/) + assert.match(workflow, /run-id: \$\{\{ github\.run_id \}\}/) + assert.match(workflow, /relay-production-capacity-wave\.mjs build-preflight/) + assert.match(workflow, /Require the exact wave predecessor topology/) + assert.match(workflow, /test "\$\{TOPOLOGY_PHASE\}" = predecessor/) + assert.match(workflow, /Recheck exact wave state and every live safety signal/) + assert.match(workflow, /if test "\$\{WAVE_INDEX\}" != 0; then RETRY_ARGS=\(--retry-freshness\); fi/) + assert.equal(workflow.match(/--retry-freshness/g)?.length, 2) + ordered( + 'Recheck exact wave state and every live safety signal', + 'Arm fail-closed mutation cleanup', + 'Reversibly isolate only the selected cell', + 'Verify the live general selected cell' + ) +}) + +test('Terraform mutation targets only the selected cell and has fail-closed recovery', () => { + assert.match( + workflow, + /google_compute_instance_template\.relay_gce_cell\[\\"\$\{TARGET_CELL_ID\}\\"\]/ + ) + assert.match( + workflow, + /google_compute_instance_group_manager\.relay_gce_cell\[\\"\$\{TARGET_CELL_ID\}\\"\]/ + ) + assert.doesNotMatch(workflow, /relay_gce_cell\["production-gce-c26"\]/) + assert.doesNotMatch(workflow, /target=google_cloud_run_v2_service\.relay/) + assert.match(workflow, /validate-relay-capacity-plan\.mjs/) + assert.match(workflow, /--mode bootstrap-cell/) + assert.match(workflow, /--capacity-service-account "\$\{CAPACITY_SERVICE_ACCOUNT\}"/) + assert.match(workflow, /failure\(\) && inputs\.mode != 'verify'/) + assert.match(workflow, /test "\$\{MUTATION_STARTED:-false\}" = true \|\| exit 0/) + assert.match(workflow, /--mode isolate/) + assert.doesNotMatch(workflow, /rolling-action restart/) + assert.equal(workflow.match(/manage_artifact_dns=false/g)?.length, 5) + assert.match(workflow, /OFFLINE_ROLLBACK=true/) + assert.match(workflow, /--runtime unavailable/) + assert.equal(workflow.match(/--expected-image-digests/g)?.length, 7) + assert.match(workflow, /PREDECESSOR_IMAGE_DIGEST: sha256:0e83408b/) + assert.match(workflow, /classify-relay-production-capacity-director\.mjs/) + assert.match(workflow, /CURRENT_CAPACITY_SERVICE_ACCOUNT_JSON/) + assert.match(workflow, /if test "\$\{DIRECTOR_READY\}" = true; then exit 0; fi/) + assert.match( + workflow, + /Keep the selected cell isolated after a failed mutation[\s\S]*?--mode isolate[\s\S]*?--mode drain/ + ) + const cleanup = workflow.slice(workflow.indexOf('id: cleanup-auth')) + assert.match(cleanup, /Keep the selected cell isolated after a failed mutation/) + assert.match(cleanup, /steps\.cleanup-auth\.outputs\.id_token/) + assert.doesNotMatch(cleanup, /steps\.deploy-auth\.outputs\.id_token/) + ordered( + 'Reversibly isolate only the selected cell', + 'Drain the selected cell or prove an offline rollback', + 'id: restart-auth-one', + 'Require restart-safe selected-cell activity', + 'id: restart-auth-two', + 'Require extended restart-safe selected-cell activity', + 'Deploy only the reviewed director topology', + 'id: director-transition-auth', + 'Require fail-closed director transition', + 'id: capacity-auth', + 'Plan and apply only the empty selected cell', + 'id: capacity-transition-auth', + 'Verify fresh exact selected-cell heartbeat before admission' + ) + const restartGate = workflow.slice( + workflow.indexOf('id: restart-auth-one'), + workflow.indexOf('Deploy only the reviewed director topology') + ) + assert.equal(restartGate.match(/--timeout-ms 450000/g)?.length, 2) + assert.match(restartGate, /steps\.restart-auth-one\.outputs\.id_token/) + assert.match(restartGate, /steps\.restart-auth-two\.outputs\.id_token/) + assert.match(restartGate, /capacity transition verification timed out:/) + assert.match(workflow, /timeout-minutes: 75/) +}) + +test('wave resume is bound to the failed run and exact isolated selector state', () => { + assert.match(dispatchWorkflow, /- wave-resume/) + assert.match(dispatchWorkflow, /evidence-mode: resume/) + assert.match(dispatchWorkflow, /source-wave-run-id: \$\{\{ inputs\.source-wave-run-id \}\}/) + assert.match(workflow, /Download the failed wave authority for resume/) + assert.match(workflow, /test "\$\{SOURCE_SHA\}" = "\$\{MONITOR_SHA\}"/) + assert.match(workflow, /test "\$\{SOURCE_ATTEMPT\}" = "\$\{EXPECTED_SOURCE_ATTEMPT\}"/) + for (const boundary of [ + '31554591366:31555510376:production-gce-c16,production-gce-c15,production-gce-c14,production-gce-c13:production-gce-c13', + '31562760783:31563664692:production-gce-c10,production-gce-c9,production-gce-c8,production-gce-c7:production-gce-c10', + '31571019947:31572080665:production-gce-c9,production-gce-c8,production-gce-c7:production-gce-c8', + 'a917e8e1fc1a2654e8cb81ba39b57733ec56be9c', + '6082e9ca89a918ca51f0c87db003f5e8805b64b7', + 'e59958130c9d9b7a6cd805df2678d08997842c7c', + 'EXPECTED_SOURCE_ATTEMPT=2' + ]) { + assert.match(workflow, new RegExp(boundary)) + } + assert.match(workflow, /\*\) exit 1 ;;/) + assert.match(workflow, /test "\$\{MONITOR_SHA\}" = "\$\{EXPECTED_SHA\}"/) + assert.match(workflow, /\.head_branch == "main"/) + assert.match(workflow, /\.head_repository\.full_name == env\.GITHUB_REPOSITORY/) + assert.ok(workflow.includes(`.path == "${relayWorkflowPath('monitor-relay-production.yml')}"`)) + assert.ok(workflow.includes(`.path == "${relayWorkflowPath('deploy-relay-production-capacity.yml')}"`)) + assert.match(workflow, /build-resume-preflight/) + assert.match(workflow, /RESUME_SELECTED_CELL_TO_1000 \$\{TARGET_CELL_ID\}/) + assert.match( + workflow, + /Recheck exact isolated resume state[\s\S]*?--hard-cap 600[\s\S]*?--admission migration-only[\s\S]*?--draining required/ + ) +}) + +test('GCE capacity identity is exact-workflow and narrowly permissioned', () => { + const provider = resource( + 'google_iam_workload_identity_pool_provider', + 'github_production_relay_capacity' + ) + assert.match(provider, /concat\(local\.relay_github_leading_repository_claims, \[/) + for (const boundary of [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'production'", + 'local.relay_github_workflow_conditions["github_production_relay_capacity"]' + ]) { + assert.match(provider, new RegExp(boundary.replaceAll(/[.*+?^${}()|[\]\\]/g, '\\$&'))) + } + // The workflow pair itself is pinned in the clause the provider renders, once per accepted + // repository, and each repository supplies its own workflow-ref head. + for (const boundary of [ + "assertion.workflow_ref == '${prefix}${local.github_production_relay_capacity_workflow_file}@refs/heads/main'", + "assertion.job_workflow_ref == '${prefix}${local.github_production_relay_capacity_job_workflow_file}@refs/heads/main'" + ]) { + assert.match(terraform, new RegExp(boundary.replaceAll(/[.*+?^${}()|[\]\\]/g, '\\$&'))) + } + const role = resource( + 'google_project_iam_custom_role', + 'github_production_relay_capacity_mutation' + ) + assert.match(role, /compute\.instanceGroupManagers\.update/) + assert.match(role, /compute\.instanceTemplates\.create/) + assert.doesNotMatch( + role, + /compute\.(?:disks\.delete|instances\.(?:delete|start|stop|update))|cloudsql|secretmanager/ + ) + const state = resource( + 'google_storage_bucket_iam_member', + 'github_production_relay_capacity_state' + ) + assert.match(state, /objects\/terraform\/state\/default\.tfstate/) + assert.match(state, /objects\/terraform\/state\/default\.tflock/) +}) + +test('deploy and capacity identities are used in their intended phases', () => { + const jobStart = workflow.indexOf(' capacity:') + const stepsStart = workflow.indexOf(' steps:', jobStart) + const jobHeader = workflow.slice(jobStart, stepsStart) + assert.deepEqual(jobHeader.match(/^\s+if:.*$/gm), [ + " if: ${{ github.ref == 'refs/heads/main' }}" + ]) + const configurationStart = workflow.indexOf('Require production workflow configuration') + const configurationEnd = workflow.indexOf('- uses: actions/checkout@v4', configurationStart) + assert.ok(configurationStart >= 0) + assert.ok(configurationEnd > configurationStart) + const configurationStep = workflow.slice(configurationStart, configurationEnd) + for (const name of [ + 'GCP_REGION', + 'DEPLOY_WORKLOAD_IDENTITY_PROVIDER', + 'DEPLOY_SERVICE_ACCOUNT', + 'CAPACITY_WORKLOAD_IDENTITY_PROVIDER', + 'CAPACITY_SERVICE_ACCOUNT' + ]) { + assert.match(configurationStep, new RegExp(`test -n "\\$\\{${name}\\}"`)) + } + ordered('Require production workflow configuration', 'id: deploy-auth') + ordered('id: deploy-auth', 'Reversibly isolate only the selected cell', 'id: capacity-auth') + assert.match(workflow, /steps\.deploy-auth\.outputs\.id_token/) + assert.doesNotMatch(workflow, /steps\.capacity-auth\.outputs\.id_token/) + assert.equal( + workflow.match(/steps\.capacity-transition-auth\.outputs\.id_token/g)?.length, + 3 + ) + assert.match(workflow, /PRODUCTION_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(workflow, /PRODUCTION_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT/) + assert.match(workflow, /read-relay-production-capacity-identity\.mjs/) + assert.match(workflow, /\(\.directorReady \| type\) == "boolean"/) + assert.match(workflow, /\(\.directorReady \| tostring\)/) +}) + +test('director readiness extraction preserves only JSON booleans', { + skip: spawnSync('jq', ['--version']).status !== 0 +}, () => { + const filter = `if (.directorReady | type) == "boolean" then + (.directorReady | tostring) + else error("invalid directorReady classification") end` + const extract = (input) => spawnSync('jq', ['-er', filter], { + encoding: 'utf8', + input: JSON.stringify(input) + }) + for (const value of [true, false]) { + const result = extract({ directorReady: value }) + assert.equal(result.status, 0) + assert.equal(result.stdout.trim(), String(value)) + } + for (const input of [ + { directorReady: 'true' }, + { directorReady: 'false' }, + { directorReady: null }, + {} + ]) { + assert.notEqual(extract(input).status, 0) + } +}) diff --git a/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs b/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs new file mode 100644 index 00000000000..7e8ea2a05c1 --- /dev/null +++ b/cloud/dev/scripts/relay-production-identity-boundaries.test.mjs @@ -0,0 +1,169 @@ +import assert from 'node:assert/strict' +import { readFile } from 'node:fs/promises' +import test from 'node:test' +import { readRelayWorkflow, relayWorkflowFile } from './relay-repository.mjs' +import { readWorkflow, workflowFiles } from './cloud-sql-rollout-lock-census.mjs' + +async function source(path) { + return await readFile(new URL(`../../${path}`, import.meta.url), 'utf8') +} + +// Why: the shared deploy identity is relay-owned in production and moves with the relay +// extraction, so it needs a name the public repo can carry without touching the app pair. The +// generic names are retired; a workflow that still reads them would silently resolve to nothing. +test('no workflow names the retired generic production deploy identity', async () => { + const files = workflowFiles() + assert.ok(files.length > 20) + const relayReaders = [] + for (const file of files) { + const workflow = readWorkflow(file) + assert.doesNotMatch(workflow, /PRODUCTION_GCP_WORKLOAD_IDENTITY_PROVIDER\b/, file) + assert.doesNotMatch(workflow, /PRODUCTION_GCP_DEPLOY_SERVICE_ACCOUNT\b/, file) + if (/PRODUCTION_GCP_RELAY_DEPLOY_/.test(workflow)) relayReaders.push(file) + } + assert.deepEqual(relayReaders.sort(), [ + 'deploy-relay-fence-broker.yml', + 'deploy-relay-production-capacity-job.yml', + 'deploy-relay-production-capacity.yml', + 'deploy-relay-production-director.yml', + 'deploy-relay-production-multi-target.yml', + 'deploy-relay-production-same-cap-job.yml', + 'deploy-relay-production-same-cap.yml', + 'deploy-relay-production.yml', + 'operate-relay-asia-admission.yml', + 'operate-relay-production-rehome-job.yml', + 'publish-relay-production.yml' + ].map((name) => relayWorkflowFile(name)).sort()) +}) + +test('monitor workflow has no shared deploy identity fallback', async () => { + const workflow = readRelayWorkflow('monitor-relay-production-job.yml') + assert.match(workflow, /PRODUCTION_GCP_RELAY_MONITOR_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(workflow, /PRODUCTION_GCP_RELAY_MONITOR_SERVICE_ACCOUNT/) + assert.doesNotMatch(workflow, /PRODUCTION_GCP_DEPLOY_SERVICE_ACCOUNT/) + assert.doesNotMatch(workflow, /PRODUCTION_GCP_WORKLOAD_IDENTITY_PROVIDER/) +}) + +test('relay fencing uses the dedicated requester and private broker', async () => { + const workflow = readRelayWorkflow('deploy-relay-production-multi-target.yml') + assert.match(workflow, /Reject direct-runner Terraform fence aborts/) + assert.match(workflow, /inputs\.mode == 'fence-source'/) + assert.match(workflow, /inputs\.mode == 'abort-fence-source'/) + assert.match(workflow, /inputs\.mode == 'supersede-target'/) + assert.match(workflow, /PRODUCTION_GCP_RELAY_FENCE_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(workflow, /PRODUCTION_GCP_RELAY_FENCE_SERVICE_ACCOUNT/) + assert.match(workflow, /PRODUCTION_GCP_RELAY_FENCE_BROKER_URI/) + assert.match(workflow, /Invoke private target-supersession broker/) + assert.match(workflow, /Invoke private source-fence broker/) + assert.match(workflow, /Require exact broker cell contract/) + assert.match(workflow, /Require exact source-fence broker contract/) + assert.match(workflow, /Require private fence-broker environment/) + assert.match( + workflow, + /DEPLOY_MODE\}" = "execute" \|\|\s+"\$\{DEPLOY_MODE\}" = "recover-forward"\) &&\s+"\$\{SOURCE_CELL_ID\}" = "production-gce-c12"/ + ) + assert.match( + workflow, + /--scoped-recovery-source-cell-id\s+production-gce-c3/ + ) + assert.match( + workflow, + /test "\$\{FAILED_TARGET_CELL_ID\}" = "production-gce-c12"/ + ) + assert.match( + workflow, + /test "\$\{REPLACEMENT_TARGET_CELL_ID\}" = "production-gce-c13"/ + ) + assert.match( + workflow, + /test "\$\{TARGET_CELL_IDS\}" = "production-gce-c12,production-gce-c13"/ + ) + assert.match( + workflow, + /test "\$\{TARGET_CELL_IDS\}" = "production-gce-c7,production-gce-c8,production-gce-c10,production-gce-c13,production-gce-c17,production-gce-c18"/ + ) + const jobGate = workflow.slice( + workflow.indexOf('jobs:'), + workflow.indexOf('runs-on:') + ) + assert.doesNotMatch(jobGate, /PRODUCTION_GCP_RELAY_FENCE_/) + const brokerStep = workflow.slice( + workflow.indexOf('- name: Invoke private target-supersession broker'), + workflow.indexOf('- name: Preflight or run multi-target evacuation') + ) + assert.match(brokerStep, /steps\.google-fence-broker-auth\.outputs\.id_token/) + assert.doesNotMatch(brokerStep, /PRODUCTION_GCP_DEPLOY_SERVICE_ACCOUNT/) + const sourceFenceStep = workflow.slice( + workflow.indexOf('- name: Invoke private source-fence broker'), + workflow.indexOf('- name: Preflight or run multi-target evacuation') + ) + assert.match(sourceFenceStep, /steps\.google-fence-broker-auth\.outputs\.id_token/) + assert.match(sourceFenceStep, /\/v1\/fence-source/) + assert.doesNotMatch(sourceFenceStep, /PRODUCTION_GCP_DEPLOY_SERVICE_ACCOUNT/) +}) + +test('Terraform binds dedicated identities to exact OIDC and resource boundaries', async () => { + const terraform = await source('infra/terraform/relay-github-actions.tf') + for (const claim of ['job_workflow_ref', 'workflow_ref', 'ref', 'environment']) { + assert.match(terraform, new RegExp(`assertion\\.${claim}`)) + } + assert.match(terraform, /github_monitor_workflow_file/) + assert.match(terraform, /github_fence_workflow_file/) + assert.match(terraform, /github_production_relay_capacity_job_workflow_file/) + assert.match(terraform, /google_service_account" "github_monitor"/) + assert.match(terraform, /google_service_account" "github_fence"/) + assert.match(terraform, /google_service_account\.github_monitor\[0\]\.member/) + assert.match(terraform, /service_account_id = google_service_account\.github_fence\[0\]\.name/) + assert.match(terraform, /attribute\.relay_ops_identity\/monitor/) + assert.match(terraform, /attribute\.relay_ops_identity\/fence/) + assert.doesNotMatch(terraform, /github_relay_fence_operator/) + assert.doesNotMatch(terraform, /github_terraform_fence_state_writer/) + const broker = await source('infra/terraform/relay-fence-broker.tf') + assert.match(broker, /max_instance_request_concurrency = 1/) + assert.match(broker, /max_instance_count = 1/) + assert.match(broker, /roles\/run\.invoker/) + assert.match(broker, /google_service_account\.github_fence\[0\]\.member/) + assert.doesNotMatch(broker, /allUsers/) + const brokerDeploy = readRelayWorkflow('deploy-relay-fence-broker.yml') + assert.match(brokerDeploy, /sha-\$\{GITHUB_SHA\}/) + assert.match(brokerDeploy, /gcloud run services update/) + assert.match(brokerDeploy, /\.status\.traffic/) + assert.doesNotMatch(brokerDeploy, /latestReadyRevisionName/) + assert.doesNotMatch(brokerDeploy, /--set-env-vars/) +}) + +test('Terraform exposes the audited production environment values', async () => { + const outputs = await source('infra/terraform/outputs.tf') + for (const output of [ + 'github_relay_monitor_workload_identity_provider', + 'github_relay_monitor_service_account', + 'github_relay_fence_workload_identity_provider', + 'github_relay_fence_service_account' + ]) { + assert.match(outputs, new RegExp(`output "${output}"`)) + } +}) + +test('production mutations pass the minted admin token to live preflight', async () => { + const workflow = readRelayWorkflow('deploy-relay-production.yml') + const recheck = workflow.slice( + workflow.indexOf('- name: Recheck all live safety signals'), + workflow.indexOf('- name: Create single-use dry-run marker') + ) + assert.match( + recheck, + /ORCA_RELAY_ADMIN_ID_TOKEN: \$\{\{ steps\.google-auth\.outputs\.id_token \}\}/ + ) + const multiTarget = readRelayWorkflow('deploy-relay-production-multi-target.yml') + const multiTargetRecheck = multiTarget.slice( + multiTarget.indexOf('- name: Recheck all live safety signals'), + multiTarget.indexOf('- name: Create single-use dry-run marker') + ) + assert.match(multiTargetRecheck, /steps\.google-auth\.outputs\.id_token/) + assert.match(multiTargetRecheck, /inputs\.mode != 'supersede-target'/) +}) + +test('fence broker pins the production-proven Terraform planner', async () => { + const dockerfile = await source('apps/relay-fence-broker/Dockerfile') + assert.match(dockerfile, /FROM hashicorp\/terraform:1\.15\.8 AS terraform/) +}) diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.mjs new file mode 100644 index 00000000000..e391c4c4381 --- /dev/null +++ b/cloud/dev/scripts/relay-production-same-cap-wave.mjs @@ -0,0 +1,161 @@ +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' + +export const SAME_CAP_CELLS = [ + 'production-gce-c7', 'production-gce-c8', 'production-gce-c9', 'production-gce-c10', + 'production-gce-c13', 'production-gce-c14', 'production-gce-c15', 'production-gce-c16', + 'production-gce-c19', 'production-gce-c20', 'production-gce-c21', 'production-gce-c22', + 'production-gce-c23', 'production-gce-c24', 'production-gce-c25', 'production-gce-c26', + 'production-gce-c27', 'production-gce-c28', 'production-gce-c29' +] + +function digest(value, name) { + if (!/^sha256:[a-f0-9]{64}$/.test(value ?? '')) throw new Error(`${name} is invalid`) + return value +} + +function cells(value) { + const parsed = value.split(',').map((cell) => cell.trim()).filter(Boolean) + if ( + parsed.length < 1 || + parsed.length > 4 || + new Set(parsed).size !== parsed.length || + parsed.some((cell) => !SAME_CAP_CELLS.includes(cell)) + ) throw new Error('same-cap wave cells are invalid') + return parsed +} + +export function validateSameCapWave(input) { + if (!['verify', 'canary-apply', 'batch-apply', 'rollback'].includes(input.mode)) { + throw new Error('same-cap wave mode is invalid') + } + const selected = cells(input.cellIds) + const targetDigest = digest(input.targetDigest, 'target digest') + const rollbackDigest = digest(input.rollbackDigest, 'rollback digest') + if (targetDigest === rollbackDigest) throw new Error('target and rollback digests must differ') + if (input.mode === 'canary-apply' && selected.length !== 1) { + throw new Error('canary mode requires exactly one cell') + } + if (input.mode === 'batch-apply' && (selected.length < 2 || selected.length > 4)) { + throw new Error('batch mode requires two to four cells') + } + // Later waves expect the selector to advance by exactly 2 per predecessor, + // which a resumed rollback cell (isolate skipped, +1) violates. + if (input.mode === 'rollback' && selected.length !== 1) { + throw new Error('rollback mode requires exactly one cell') + } + const mutation = input.mode !== 'verify' + const expectedConfirmation = input.mode === 'rollback' + ? `ROLL_BACK_RELAY_SAME_CAP ${rollbackDigest} ${selected.join(',')}` + : `ROLL_RELAY_SAME_CAP ${targetDigest} ${selected.join(',')}` + if (mutation && input.confirmation !== expectedConfirmation) { + throw new Error('same-cap confirmation does not match the exact digest and cells') + } + if (!mutation && input.confirmation) throw new Error('verify does not accept confirmation') + if (input.mode === 'batch-apply' && !/^[1-9][0-9]*$/.test(input.canaryRunId ?? '')) { + throw new Error('batch mode requires a canary run ID') + } + if (input.mode !== 'batch-apply' && input.canaryRunId) { + throw new Error('only batch mode accepts a canary run ID') + } + return { cells: selected, targetDigest, rollbackDigest } +} + +export function canaryAuthority(input) { + const wave = validateSameCapWave({ ...input, mode: 'canary-apply', canaryRunId: '' }) + if (!/^[0-9a-f]{40}$/.test(input.commitSha ?? '')) throw new Error('commit SHA is invalid') + if (!/^[1-9][0-9]*$/.test(input.runId ?? '')) throw new Error('run ID is invalid') + const selectorGeneration = Number(input.selectorGeneration) + const rehomeGeneration = Number(input.rehomeGeneration) + if (!Number.isSafeInteger(selectorGeneration) || selectorGeneration < 0) { + throw new Error('selector generation is invalid') + } + if (!Number.isSafeInteger(rehomeGeneration) || rehomeGeneration < 0) { + throw new Error('rehome generation is invalid') + } + return { + v: 1, + commitSha: input.commitSha, + runId: input.runId, + cellId: wave.cells[0], + targetDigest: wave.targetDigest, + rollbackDigest: wave.rollbackDigest, + selectorGeneration: selectorGeneration + 2, + rehomeGeneration + } +} + +export function verifyCanaryAuthority(authority, expected) { + if ( + authority?.v !== 1 || + authority.commitSha !== expected.commitSha || + authority.runId !== expected.runId || + authority.targetDigest !== expected.targetDigest || + authority.rollbackDigest !== expected.rollbackDigest || + authority.selectorGeneration !== Number(expected.selectorGeneration) || + authority.rehomeGeneration !== Number(expected.rehomeGeneration) || + !SAME_CAP_CELLS.includes(authority.cellId) + ) throw new Error('canary authority does not match this batch') + return authority +} + +function values(argv) { + const result = {} + for (let index = 0; index < argv.length; index += 2) { + if (!argv[index]?.startsWith('--') || argv[index + 1] === undefined) { + throw new Error('invalid arguments') + } + result[argv[index].slice(2)] = argv[index + 1] + } + return result +} + +export function main(argv = process.argv.slice(2)) { + const command = argv.shift() + const input = values(argv) + if (command === 'validate') { + const wave = validateSameCapWave({ + mode: input.mode, + cellIds: input['cell-ids'], + targetDigest: input['target-digest'], + rollbackDigest: input['rollback-digest'], + confirmation: input.confirmation, + canaryRunId: input['canary-run-id'] + }) + process.stdout.write(`${JSON.stringify(wave.cells)}\n`) + return + } + if (command === 'create-canary') { + process.stdout.write(`${JSON.stringify(canaryAuthority({ + mode: 'canary-apply', + cellIds: input['cell-id'], + targetDigest: input['target-digest'], + rollbackDigest: input['rollback-digest'], + confirmation: input.confirmation, + commitSha: input['commit-sha'], + runId: input['run-id'], + selectorGeneration: input['selector-generation'], + rehomeGeneration: input['rehome-generation'] + }))}\n`) + return + } + if (command === 'verify-canary') { + verifyCanaryAuthority(JSON.parse(readFileSync(input.file, 'utf8')), { + commitSha: input['commit-sha'], + runId: input['run-id'], + targetDigest: input['target-digest'], + rollbackDigest: input['rollback-digest'], + selectorGeneration: input['selector-generation'], + rehomeGeneration: input['rehome-generation'] + }) + return + } + throw new Error('unknown same-cap wave command') +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + try { main() } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + } +} diff --git a/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs new file mode 100644 index 00000000000..0b45ae85a99 --- /dev/null +++ b/cloud/dev/scripts/relay-production-same-cap-wave.test.mjs @@ -0,0 +1,106 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + canaryAuthority, + validateSameCapWave, + verifyCanaryAuthority +} from './relay-production-same-cap-wave.mjs' + +const targetDigest = `sha256:${'a'.repeat(64)}` +const rollbackDigest = `sha256:${'b'.repeat(64)}` + +test('requires one canary or a bounded reviewed batch', () => { + assert.deepEqual(validateSameCapWave({ + mode: 'canary-apply', + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7` + }).cells, ['production-gce-c7']) + assert.throws(() => validateSameCapWave({ + mode: 'canary-apply', + cellIds: 'production-gce-c7,production-gce-c8', + targetDigest, + rollbackDigest, + confirmation: 'wrong' + }), /canary/) + assert.deepEqual(validateSameCapWave({ + mode: 'batch-apply', + cellIds: 'production-gce-c8,production-gce-c9', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c8,production-gce-c9`, + canaryRunId: '42' + }).cells, ['production-gce-c8', 'production-gce-c9']) + assert.deepEqual(validateSameCapWave({ + mode: 'canary-apply', + cellIds: 'production-gce-c28', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c28` + }).cells, ['production-gce-c28']) + assert.throws(() => validateSameCapWave({ + mode: 'canary-apply', + cellIds: 'production-gce-c30', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c30` + }), /cells/) +}) + +test('binds rollback confirmation to the exact digest and ordered cells', () => { + assert.throws(() => validateSameCapWave({ + mode: 'rollback', + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_BACK_RELAY_SAME_CAP ${targetDigest} production-gce-c7` + }), /confirmation/) +}) + +test('rollback rolls exactly one cell so later waves stay unreachable', () => { + const cellIds = 'production-gce-c7,production-gce-c8' + assert.throws(() => validateSameCapWave({ + mode: 'rollback', + cellIds, + targetDigest, + rollbackDigest, + confirmation: `ROLL_BACK_RELAY_SAME_CAP ${rollbackDigest} ${cellIds}` + }), /rollback mode requires exactly one cell/) + assert.deepEqual(validateSameCapWave({ + mode: 'rollback', + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_BACK_RELAY_SAME_CAP ${rollbackDigest} production-gce-c7` + }).cells, ['production-gce-c7']) +}) + +test('seals and verifies canary authority for later batches', () => { + const authority = canaryAuthority({ + cellIds: 'production-gce-c7', + targetDigest, + rollbackDigest, + confirmation: `ROLL_RELAY_SAME_CAP ${targetDigest} production-gce-c7`, + commitSha: 'c'.repeat(40), + runId: '42', + selectorGeneration: '11', + rehomeGeneration: '4' + }) + assert.equal(verifyCanaryAuthority(authority, { + commitSha: 'c'.repeat(40), + runId: '42', + targetDigest, + rollbackDigest, + selectorGeneration: '13', + rehomeGeneration: '4' + }).cellId, 'production-gce-c7') + assert.throws(() => verifyCanaryAuthority(authority, { + commitSha: 'd'.repeat(40), + runId: '42', + targetDigest, + rollbackDigest, + selectorGeneration: '11', + rehomeGeneration: '4' + }), /does not match/) +}) diff --git a/cloud/dev/scripts/relay-public-workflow-contract.test.mjs b/cloud/dev/scripts/relay-public-workflow-contract.test.mjs new file mode 100644 index 00000000000..56393d07bd1 --- /dev/null +++ b/cloud/dev/scripts/relay-public-workflow-contract.test.mjs @@ -0,0 +1,82 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { + isEntrypoint, + jobIf, + jobNeeds, + jobs, + readWorkflow, + workflowFiles +} from './cloud-sql-rollout-lock-census.mjs' +import { relayWorkflowFile } from './relay-repository.mjs' + +// Why: this repository publishes the relay's operate surface next to the desktop app. Three +// invariants make that safe, and each of them is one careless edit away from being lost. +const OPERATIONS_GATE = "vars.ORCA_CLOUD_OPERATIONS_ENABLED == 'true'" + +// Cloud Verify is the only cloud workflow that must run on every pull request. +const UNGATED = relayWorkflowFile('verify.yml') + +const relayWorkflows = () => workflowFiles().filter((file) => file !== UNGATED) + +test('the copy carries every relay workflow', () => { + assert.equal(relayWorkflows().length, 24) +}) + +// Why: workflow_run chains match by display name, not filename. Renaming a file is safe; renaming +// one of these silently breaks the recovery chain with no failing run to notice. +test('the recovery chain keeps the display names it is matched by', () => { + const names = Object.fromEntries( + ['prove-relay-staging-capacity.yml', 'recover-relay-staging-c4-image.yml', 'requeue-relay-staging-c4-recovery.yml'].map( + (name) => [name, /^name: (.+)$/m.exec(readWorkflow(relayWorkflowFile(name)))?.[1]] + ) + ) + assert.deepEqual(names, { + 'prove-relay-staging-capacity.yml': 'Prove Relay Staging Capacity', + 'recover-relay-staging-c4-image.yml': 'Recover Relay Staging C4 Image', + 'requeue-relay-staging-c4-recovery.yml': 'Requeue Relay Staging C4 Recovery' + }) + const recover = readWorkflow(relayWorkflowFile('recover-relay-staging-c4-image.yml')) + const requeue = readWorkflow(relayWorkflowFile('requeue-relay-staging-c4-recovery.yml')) + assert.ok(recover.includes(`workflows: [${names['prove-relay-staging-capacity.yml']}]`)) + assert.ok(requeue.includes(`workflows: [${names['recover-relay-staging-c4-image.yml']}]`)) +}) + +// Why: this repository holds none of the GCP credentials these workflows would need. Every one +// authenticates through Workload Identity read from a variable, so any repository secret other +// than the automatic token would be a credential the owner has to store here. +test('no cloud workflow reads a repository secret', () => { + for (const file of workflowFiles()) { + for (const [, name] of readWorkflow(file).matchAll(/secrets\.([A-Za-z_][A-Za-z0-9_]*)/g)) { + assert.equal(name, 'GITHUB_TOKEN', `${file} reads secrets.${name}`) + } + } +}) + +// Why: the operations gate is what makes the whole surface inert until the owner enables it. A +// job that can start without a gated dependency would run the moment someone dispatches it. +test('every job that can start on its own is gated on the operations variable', () => { + const reachable = [] + for (const file of relayWorkflows()) { + const text = readWorkflow(file) + if (!isEntrypoint(text)) continue + for (const job of jobs(text)) { + if (jobNeeds(job.text).length > 0) continue + reachable.push(`${file}:${job.id}`) + assert.ok(jobIf(job.text).includes(OPERATIONS_GATE), `${file}:${job.id} is not gated`) + } + } + assert.ok(reachable.length >= 20, `only ${reachable.length} root jobs were checked`) +}) + +// Why: reusable jobs inherit the caller's gate. Gating them again would be dead configuration +// that reads as protection, and every caller is already checked above. +test('reusable workflows carry no gate of their own', () => { + for (const file of relayWorkflows()) { + const text = readWorkflow(file) + if (isEntrypoint(text)) continue + for (const job of jobs(text)) { + assert.ok(!jobIf(job.text).includes(OPERATIONS_GATE), `${file}:${job.id} regates a reusable job`) + } + } +}) diff --git a/cloud/dev/scripts/relay-recovery-wave-gate.mjs b/cloud/dev/scripts/relay-recovery-wave-gate.mjs new file mode 100644 index 00000000000..08d53d3680f --- /dev/null +++ b/cloud/dev/scripts/relay-recovery-wave-gate.mjs @@ -0,0 +1,517 @@ +const EXPECTED_KEYS = { + report: ['schemaVersion', 'environment', 'load', 'outcomes'], + environment: [ + 'projectId', + 'directorOrigin', + 'databaseVcpu', + 'databasePoolMax', + 'publicConcurrentMax', + 'resolvePrioritySlots', + 'directorMinInstances', + 'directorMaxInstances', + 'cloudRunConcurrency', + 'rolloutOldPublicConcurrentMax', + 'rolloutOldResolvePrioritySlots', + 'rolloutNewPublicConcurrentMax', + 'rolloutNewResolvePrioritySlots' + ], + load: [ + 'drainingDesktops', + 'backgroundRequestsPerMinute', + 'backgroundAssignmentRequestsPerMinute', + 'backgroundAssignment503PerMinute', + 'targetConnectionCap', + 'targetCells' + ], + targetCell: ['cellId', 'peakConnections', 'recoveredControls'], + outcomes: [ + 'migrationExpirations', + 'migrationAborts', + 'transactionRetryExhaustions', + 'keyProvenTargetRegistrations', + 'oldestMigrationLeaseRemainingAtDrainMs', + 'targetRegistrationDurationMs', + 'assignmentSuccessesPerMinuteBaseline', + 'assignmentSuccessesPerMinuteMinimum', + 'eligibleResolveRequests', + 'resolve2xx', + 'resolveOverload', + 'readinessChecks', + 'readinessFailures', + 'maintenanceOperations', + 'maintenanceFailures', + 'directorPeakInstances', + 'rolloutOverlapPeakInstances', + 'rolloutOverlapPeakPublicOperations', + 'rolloutOverlapEligibleResolveRequests', + 'rolloutOverlapResolve2xx', + 'rolloutOverlapResolveOverload', + 'rolloutOverlapReadinessFailures', + 'rolloutOverlapPoolWaitP95Ms', + 'rolloutOverlapDatabaseCpuPercentMax', + 'poolWaitP95Ms', + 'poolWaitMaxMs', + 'databaseCpuPercentP95', + 'databaseCpuPercentMax', + 'recoveryDurationMs' + ] +} + +const LIMITS = { + drainingDesktopsMin: 760, + drainingDesktopsMax: 840, + backgroundRequestsPerMinuteMin: 10_450, + backgroundRequestsPerMinuteMax: 11_550, + backgroundAssignment503PerMinuteMin: 8_500, + backgroundAssignment503PerMinuteMax: 10_500, + targetCellCount: 2, + targetConnectionCap: 600, + databaseVcpu: 2, + databasePoolMax: 3, + publicConcurrentMax: 2, + resolvePrioritySlots: 1, + directorMinInstances: 1, + directorMaxInstances: 2, + directorPeakInstances: 2, + rolloutOverlapPeakInstances: 4, + rolloutOverlapPeakPublicOperations: 8, + cloudRunConcurrency: 80, + rolloutOldPublicConcurrentMax: 2, + rolloutOldResolvePrioritySlots: 0, + rolloutNewPublicConcurrentMax: 2, + rolloutNewResolvePrioritySlots: 1, + assignmentThroughputRetentionMin: 0.9, + resolveSuccessRateMin: 0.95, + resolveOverloadRateMaxExclusive: 0.01, + poolWaitP95MsMaxExclusive: 500, + poolWaitMaxMsMaxExclusive: 5_000, + databaseCpuPercentP95MaxExclusive: 70, + databaseCpuPercentMaxMaxExclusive: 85, + oldestMigrationLeaseRemainingAtDrainMsMin: 10 * 60_000, + targetRegistrationDurationMsMax: 5 * 60_000, + recoveryDurationMsMax: 14 * 60_000 +} + +export function evaluateRecoveryWaveReport(input) { + const report = parseReport(input) + const recoveredControls = report.load.targetCells.reduce( + (total, cell) => total + cell.recoveredControls, + 0 + ) + const peakTargetConnections = Math.max( + ...report.load.targetCells.map((cell) => cell.peakConnections) + ) + const assignmentThroughputRetention = ratio( + report.outcomes.assignmentSuccessesPerMinuteMinimum, + report.outcomes.assignmentSuccessesPerMinuteBaseline + ) + const resolveSuccessRate = ratio( + report.outcomes.resolve2xx, + report.outcomes.eligibleResolveRequests + ) + const resolveOverloadRate = ratio( + report.outcomes.resolveOverload, + report.outcomes.eligibleResolveRequests + ) + const rolloutOverlapResolveSuccessRate = ratio( + report.outcomes.rolloutOverlapResolve2xx, + report.outcomes.rolloutOverlapEligibleResolveRequests + ) + const rolloutOverlapResolveOverloadRate = ratio( + report.outcomes.rolloutOverlapResolveOverload, + report.outcomes.rolloutOverlapEligibleResolveRequests + ) + const thresholds = [ + equal('database_vcpu', report.environment.databaseVcpu, LIMITS.databaseVcpu), + equal('database_pool_max', report.environment.databasePoolMax, LIMITS.databasePoolMax), + equal( + 'public_concurrent_max', + report.environment.publicConcurrentMax, + LIMITS.publicConcurrentMax + ), + equal( + 'resolve_priority_slots', + report.environment.resolvePrioritySlots, + LIMITS.resolvePrioritySlots + ), + equal( + 'director_min_instances', + report.environment.directorMinInstances, + LIMITS.directorMinInstances + ), + equal( + 'director_max_instances', + report.environment.directorMaxInstances, + LIMITS.directorMaxInstances + ), + equal( + 'cloud_run_concurrency', + report.environment.cloudRunConcurrency, + LIMITS.cloudRunConcurrency + ), + equal( + 'rollout_old_public_concurrent_max', + report.environment.rolloutOldPublicConcurrentMax, + LIMITS.rolloutOldPublicConcurrentMax + ), + equal( + 'rollout_old_resolve_priority_slots', + report.environment.rolloutOldResolvePrioritySlots, + LIMITS.rolloutOldResolvePrioritySlots + ), + equal( + 'rollout_new_public_concurrent_max', + report.environment.rolloutNewPublicConcurrentMax, + LIMITS.rolloutNewPublicConcurrentMax + ), + equal( + 'rollout_new_resolve_priority_slots', + report.environment.rolloutNewResolvePrioritySlots, + LIMITS.rolloutNewResolvePrioritySlots + ), + between( + 'draining_desktops', + report.load.drainingDesktops, + LIMITS.drainingDesktopsMin, + LIMITS.drainingDesktopsMax + ), + between( + 'background_requests_per_minute', + report.load.backgroundRequestsPerMinute, + LIMITS.backgroundRequestsPerMinuteMin, + LIMITS.backgroundRequestsPerMinuteMax + ), + between( + 'background_assignment_requests_per_minute', + report.load.backgroundAssignmentRequestsPerMinute, + LIMITS.backgroundRequestsPerMinuteMin, + LIMITS.backgroundRequestsPerMinuteMax + ), + between( + 'background_assignment_503_per_minute', + report.load.backgroundAssignment503PerMinute, + LIMITS.backgroundAssignment503PerMinuteMin, + LIMITS.backgroundAssignment503PerMinuteMax + ), + atMost( + 'background_assignment_requests_within_total', + report.load.backgroundAssignmentRequestsPerMinute, + report.load.backgroundRequestsPerMinute + ), + atMost( + 'background_assignment_503_within_assignments', + report.load.backgroundAssignment503PerMinute, + report.load.backgroundAssignmentRequestsPerMinute + ), + equal('target_cell_count', report.load.targetCells.length, LIMITS.targetCellCount), + equal( + 'target_connection_cap', + report.load.targetConnectionCap, + LIMITS.targetConnectionCap + ), + atMost( + 'peak_target_connections', + peakTargetConnections, + report.load.targetConnectionCap + ), + equal('recovered_controls', recoveredControls, report.load.drainingDesktops), + equal('migration_expirations', report.outcomes.migrationExpirations, 0), + equal('migration_aborts', report.outcomes.migrationAborts, 0), + equal('transaction_retry_exhaustions', report.outcomes.transactionRetryExhaustions, 0), + equal( + 'key_proven_target_registrations', + report.outcomes.keyProvenTargetRegistrations, + report.load.drainingDesktops + ), + atLeast( + 'oldest_migration_lease_remaining_at_drain_ms', + report.outcomes.oldestMigrationLeaseRemainingAtDrainMs, + LIMITS.oldestMigrationLeaseRemainingAtDrainMsMin + ), + atMost( + 'target_registration_duration_ms', + report.outcomes.targetRegistrationDurationMs, + LIMITS.targetRegistrationDurationMsMax + ), + atLeast( + 'assignment_throughput_retention', + assignmentThroughputRetention, + LIMITS.assignmentThroughputRetentionMin + ), + atLeast('resolve_success_rate', resolveSuccessRate, LIMITS.resolveSuccessRateMin), + lessThan( + 'resolve_overload_rate', + resolveOverloadRate, + LIMITS.resolveOverloadRateMaxExclusive + ), + atLeast('eligible_resolve_requests', report.outcomes.eligibleResolveRequests, 100), + equal('readiness_failures', report.outcomes.readinessFailures, 0), + atLeast('readiness_checks', report.outcomes.readinessChecks, 1), + equal('maintenance_failures', report.outcomes.maintenanceFailures, 0), + atLeast('maintenance_operations', report.outcomes.maintenanceOperations, 1), + equal( + 'director_peak_instances', + report.outcomes.directorPeakInstances, + LIMITS.directorPeakInstances + ), + equal( + 'rollout_overlap_peak_instances', + report.outcomes.rolloutOverlapPeakInstances, + LIMITS.rolloutOverlapPeakInstances + ), + equal( + 'rollout_overlap_peak_public_operations', + report.outcomes.rolloutOverlapPeakPublicOperations, + LIMITS.rolloutOverlapPeakPublicOperations + ), + atLeast( + 'rollout_overlap_eligible_resolve_requests', + report.outcomes.rolloutOverlapEligibleResolveRequests, + 100 + ), + atLeast( + 'rollout_overlap_resolve_success_rate', + rolloutOverlapResolveSuccessRate, + LIMITS.resolveSuccessRateMin + ), + lessThan( + 'rollout_overlap_resolve_overload_rate', + rolloutOverlapResolveOverloadRate, + LIMITS.resolveOverloadRateMaxExclusive + ), + equal( + 'rollout_overlap_readiness_failures', + report.outcomes.rolloutOverlapReadinessFailures, + 0 + ), + lessThan( + 'rollout_overlap_pool_wait_p95_ms', + report.outcomes.rolloutOverlapPoolWaitP95Ms, + LIMITS.poolWaitP95MsMaxExclusive + ), + lessThan( + 'rollout_overlap_database_cpu_percent_max', + report.outcomes.rolloutOverlapDatabaseCpuPercentMax, + LIMITS.databaseCpuPercentMaxMaxExclusive + ), + lessThan( + 'pool_wait_p95_ms', + report.outcomes.poolWaitP95Ms, + LIMITS.poolWaitP95MsMaxExclusive + ), + lessThan( + 'pool_wait_max_ms', + report.outcomes.poolWaitMaxMs, + LIMITS.poolWaitMaxMsMaxExclusive + ), + lessThan( + 'database_cpu_percent_p95', + report.outcomes.databaseCpuPercentP95, + LIMITS.databaseCpuPercentP95MaxExclusive + ), + lessThan( + 'database_cpu_percent_max', + report.outcomes.databaseCpuPercentMax, + LIMITS.databaseCpuPercentMaxMaxExclusive + ), + atMost( + 'recovery_duration_ms', + report.outcomes.recoveryDurationMs, + LIMITS.recoveryDurationMsMax + ) + ] + return { + schemaVersion: 1, + status: thresholds.every(({ pass }) => pass) ? 'PASS' : 'FAIL', + environment: { + projectId: report.environment.projectId, + directorOrigin: report.environment.directorOrigin + }, + metrics: { + recoveredControls, + peakTargetConnections, + assignmentThroughputRetention, + resolveSuccessRate, + resolveOverloadRate, + rolloutOverlapResolveSuccessRate, + rolloutOverlapResolveOverloadRate + }, + thresholds + } +} + +function parseReport(input) { + const report = strictObject(input, EXPECTED_KEYS.report, 'report') + if (report.schemaVersion !== 1) throw new Error('unsupported report schemaVersion') + const environment = strictObject( + report.environment, + EXPECTED_KEYS.environment, + 'environment' + ) + assertSafeEnvironment(environment) + const load = strictObject(report.load, EXPECTED_KEYS.load, 'load') + if (!Array.isArray(load.targetCells)) throw new Error('load.targetCells must be an array') + const targetCells = load.targetCells.map((value, index) => { + const cell = strictObject(value, EXPECTED_KEYS.targetCell, `load.targetCells[${index}]`) + if (!/^[a-z0-9-]{1,128}$/.test(cell.cellId)) throw new Error('target cellId is invalid') + return { + cellId: cell.cellId, + peakConnections: nonnegativeNumber(cell.peakConnections, 'peakConnections'), + recoveredControls: nonnegativeNumber(cell.recoveredControls, 'recoveredControls') + } + }) + if (new Set(targetCells.map(({ cellId }) => cellId)).size !== targetCells.length) { + throw new Error('target cell IDs must be unique') + } + const outcomes = strictObject(report.outcomes, EXPECTED_KEYS.outcomes, 'outcomes') + return { + schemaVersion: 1, + environment: { + projectId: environment.projectId, + directorOrigin: environment.directorOrigin, + databaseVcpu: positiveNumber(environment.databaseVcpu, 'databaseVcpu'), + databasePoolMax: positiveNumber(environment.databasePoolMax, 'databasePoolMax'), + publicConcurrentMax: positiveNumber( + environment.publicConcurrentMax, + 'publicConcurrentMax' + ), + resolvePrioritySlots: positiveNumber( + environment.resolvePrioritySlots, + 'resolvePrioritySlots' + ), + directorMinInstances: positiveNumber( + environment.directorMinInstances, + 'directorMinInstances' + ), + directorMaxInstances: positiveNumber( + environment.directorMaxInstances, + 'directorMaxInstances' + ), + cloudRunConcurrency: positiveNumber( + environment.cloudRunConcurrency, + 'cloudRunConcurrency' + ), + rolloutOldPublicConcurrentMax: positiveNumber( + environment.rolloutOldPublicConcurrentMax, + 'rolloutOldPublicConcurrentMax' + ), + rolloutOldResolvePrioritySlots: nonnegativeNumber( + environment.rolloutOldResolvePrioritySlots, + 'rolloutOldResolvePrioritySlots' + ), + rolloutNewPublicConcurrentMax: positiveNumber( + environment.rolloutNewPublicConcurrentMax, + 'rolloutNewPublicConcurrentMax' + ), + rolloutNewResolvePrioritySlots: positiveNumber( + environment.rolloutNewResolvePrioritySlots, + 'rolloutNewResolvePrioritySlots' + ) + }, + load: { + drainingDesktops: positiveNumber(load.drainingDesktops, 'drainingDesktops'), + backgroundRequestsPerMinute: positiveNumber( + load.backgroundRequestsPerMinute, + 'backgroundRequestsPerMinute' + ), + backgroundAssignmentRequestsPerMinute: positiveNumber( + load.backgroundAssignmentRequestsPerMinute, + 'backgroundAssignmentRequestsPerMinute' + ), + backgroundAssignment503PerMinute: nonnegativeNumber( + load.backgroundAssignment503PerMinute, + 'backgroundAssignment503PerMinute' + ), + targetConnectionCap: positiveNumber(load.targetConnectionCap, 'targetConnectionCap'), + targetCells + }, + outcomes: Object.fromEntries( + EXPECTED_KEYS.outcomes.map((key) => [key, nonnegativeNumber(outcomes[key], `outcomes.${key}`)]) + ) + } +} + +function assertSafeEnvironment(environment) { + if (typeof environment.projectId !== 'string') throw new Error('projectId must be a string') + if ( + environment.projectId !== 'local' && + !environment.projectId.endsWith('-staging') && + !environment.projectId.endsWith('-test') + ) { + throw new Error('recovery-wave reports must come from an isolated non-production project') + } + if (typeof environment.directorOrigin !== 'string') { + throw new Error('directorOrigin must be a string') + } + const origin = new URL(environment.directorOrigin) + const loopback = ['localhost', '127.0.0.1', '::1', '[::1]'].includes(origin.hostname) + const isolatedHost = + loopback || origin.hostname.endsWith('.test') || origin.hostname.includes('staging') + if ( + origin.origin !== environment.directorOrigin || + origin.pathname !== '/' || + (!loopback && origin.protocol !== 'https:') || + !isolatedHost + ) { + throw new Error('directorOrigin must identify a canonical isolated non-production origin') + } +} + +function strictObject(value, keys, name) { + if (!value || typeof value !== 'object' || Array.isArray(value)) { + throw new Error(`${name} must be an object`) + } + const actual = Object.keys(value).sort() + const expected = [...keys].sort() + if (actual.length !== expected.length || actual.some((key, index) => key !== expected[index])) { + throw new Error(`${name} has unexpected or missing fields`) + } + return value +} + +function positiveNumber(value, name) { + const number = nonnegativeNumber(value, name) + if (number <= 0) throw new Error(`${name} must be positive`) + return number +} + +function nonnegativeNumber(value, name) { + if (typeof value !== 'number' || !Number.isFinite(value) || value < 0) { + throw new Error(`${name} must be a finite nonnegative number`) + } + return value +} + +function ratio(numerator, denominator) { + return denominator === 0 ? 0 : numerator / denominator +} + +function equal(name, observed, limit) { + return threshold(name, observed, '==', limit, observed === limit) +} + +function atLeast(name, observed, limit) { + return threshold(name, observed, '>=', limit, observed >= limit) +} + +function atMost(name, observed, limit) { + return threshold(name, observed, '<=', limit, observed <= limit) +} + +function lessThan(name, observed, limit) { + return threshold(name, observed, '<', limit, observed < limit) +} + +function between(name, observed, minimum, maximum) { + return threshold( + name, + observed, + 'between_inclusive', + [minimum, maximum], + observed >= minimum && observed <= maximum + ) +} + +function threshold(name, observed, operator, limit, pass) { + return { name, observed, operator, limit, pass } +} diff --git a/cloud/dev/scripts/relay-recovery-wave-gate.test.mjs b/cloud/dev/scripts/relay-recovery-wave-gate.test.mjs new file mode 100644 index 00000000000..4ac1de83e72 --- /dev/null +++ b/cloud/dev/scripts/relay-recovery-wave-gate.test.mjs @@ -0,0 +1,214 @@ +import assert from 'node:assert/strict' +import test from 'node:test' +import { evaluateRecoveryWaveReport } from './relay-recovery-wave-gate.mjs' + +test('passes a complete isolated production-shaped recovery report', () => { + const result = evaluateRecoveryWaveReport(passingReport()) + + assert.equal(result.status, 'PASS') + assert.equal(result.metrics.recoveredControls, 800) + assert.equal(result.metrics.peakTargetConnections, 425) + assert.equal(result.metrics.resolveSuccessRate, 0.99) + assert.equal(result.thresholds.every(({ pass }) => pass), true) + assert.equal( + result.thresholds.find(({ name }) => name === 'recovery_duration_ms')?.limit, + 840_000 + ) +}) + +for (const [name, mutate, failedThreshold] of [ + [ + 'target connection ceiling', + (report) => { + report.load.targetCells[0].peakConnections = 601 + }, + 'peak_target_connections' + ], + [ + 'migration expiration', + (report) => { + report.outcomes.migrationExpirations = 1 + }, + 'migration_expirations' + ], + [ + 'migration abort', + (report) => { + report.outcomes.migrationAborts = 1 + }, + 'migration_aborts' + ], + [ + 'full-wave registration deadline', + (report) => { + report.outcomes.targetRegistrationDurationMs = 300_001 + }, + 'target_registration_duration_ms' + ], + [ + 'legacy assignment background shape', + (report) => { + report.load.backgroundAssignment503PerMinute = 100 + }, + 'background_assignment_503_per_minute' + ], + [ + 'production director instance topology', + (report) => { + report.outcomes.directorPeakInstances = 1 + }, + 'director_peak_instances' + ], + [ + 'old/new rollout overlap topology', + (report) => { + report.outcomes.rolloutOverlapPeakInstances = 2 + }, + 'rollout_overlap_peak_instances' + ], + [ + 'old revision shared admission mode', + (report) => { + report.environment.rolloutOldResolvePrioritySlots = 1 + }, + 'rollout_old_resolve_priority_slots' + ], + [ + 'old/new rollout overlap resolve availability', + (report) => { + report.outcomes.rolloutOverlapResolveOverload = 2 + }, + 'rollout_overlap_resolve_overload_rate' + ], + [ + 'resolve availability', + (report) => { + report.outcomes.resolve2xx = 940 + report.outcomes.resolveOverload = 20 + }, + 'resolve_success_rate' + ], + [ + 'pool wait', + (report) => { + report.outcomes.poolWaitP95Ms = 500 + }, + 'pool_wait_p95_ms' + ], + [ + 'database CPU', + (report) => { + report.outcomes.databaseCpuPercentMax = 85 + }, + 'database_cpu_percent_max' + ], + [ + 'recovery deadline', + (report) => { + report.outcomes.recoveryDurationMs = 840_001 + }, + 'recovery_duration_ms' + ], + [ + 'non-public database maintenance', + (report) => { + report.outcomes.maintenanceFailures = 1 + }, + 'maintenance_failures' + ] +]) { + test(`fails closed on ${name}`, () => { + const report = passingReport() + mutate(report) + const result = evaluateRecoveryWaveReport(report) + + assert.equal(result.status, 'FAIL') + assert.equal( + result.thresholds.find(({ name: thresholdName }) => thresholdName === failedThreshold)?.pass, + false + ) + }) +} + +test('rejects production provenance and unexpected report fields', () => { + const productionProject = passingReport() + productionProject.environment.projectId = 'onorca-cloud' + assert.throws( + () => evaluateRecoveryWaveReport(productionProject), + /isolated non-production project/ + ) + + const productionOrigin = passingReport() + productionOrigin.environment.directorOrigin = 'https://relay.onorca.dev' + assert.throws( + () => evaluateRecoveryWaveReport(productionOrigin), + /isolated non-production origin/ + ) + + const extraField = passingReport() + extraField.environment.accessToken = 'must-not-be-accepted' + assert.throws(() => evaluateRecoveryWaveReport(extraField), /unexpected or missing fields/) +}) + +function passingReport() { + return { + schemaVersion: 1, + environment: { + projectId: 'onorca-cloud-staging', + directorOrigin: 'https://relay-staging.onorca.dev', + databaseVcpu: 2, + databasePoolMax: 3, + publicConcurrentMax: 2, + resolvePrioritySlots: 1, + directorMinInstances: 1, + directorMaxInstances: 2, + cloudRunConcurrency: 80, + rolloutOldPublicConcurrentMax: 2, + rolloutOldResolvePrioritySlots: 0, + rolloutNewPublicConcurrentMax: 2, + rolloutNewResolvePrioritySlots: 1 + }, + load: { + drainingDesktops: 800, + backgroundRequestsPerMinute: 11_000, + backgroundAssignmentRequestsPerMinute: 10_950, + backgroundAssignment503PerMinute: 9_500, + targetConnectionCap: 600, + targetCells: [ + { cellId: 'target-a', peakConnections: 425, recoveredControls: 400 }, + { cellId: 'target-b', peakConnections: 419, recoveredControls: 400 } + ] + }, + outcomes: { + migrationExpirations: 0, + migrationAborts: 0, + transactionRetryExhaustions: 0, + keyProvenTargetRegistrations: 800, + oldestMigrationLeaseRemainingAtDrainMs: 660_000, + targetRegistrationDurationMs: 240_000, + assignmentSuccessesPerMinuteBaseline: 1_400, + assignmentSuccessesPerMinuteMinimum: 1_330, + eligibleResolveRequests: 1_000, + resolve2xx: 990, + resolveOverload: 5, + readinessChecks: 180, + readinessFailures: 0, + maintenanceOperations: 30, + maintenanceFailures: 0, + directorPeakInstances: 2, + rolloutOverlapPeakInstances: 4, + rolloutOverlapPeakPublicOperations: 8, + rolloutOverlapEligibleResolveRequests: 100, + rolloutOverlapResolve2xx: 99, + rolloutOverlapResolveOverload: 0, + rolloutOverlapReadinessFailures: 0, + rolloutOverlapPoolWaitP95Ms: 180, + rolloutOverlapDatabaseCpuPercentMax: 78, + poolWaitP95Ms: 120, + poolWaitMaxMs: 900, + databaseCpuPercentP95: 55, + databaseCpuPercentMax: 72, + recoveryDurationMs: 360_000 + } + } +} diff --git a/cloud/dev/scripts/relay-region-observation-evidence.mjs b/cloud/dev/scripts/relay-region-observation-evidence.mjs new file mode 100644 index 00000000000..1c23a9c9b00 --- /dev/null +++ b/cloud/dev/scripts/relay-region-observation-evidence.mjs @@ -0,0 +1,152 @@ +import { createHash } from 'node:crypto' +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' + +const WINDOW_MS = 24 * 60 * 60_000 +const BUCKET_MS = 60 * 60_000 +const METRICS = [ + 'requestedRegionsDelta', + 'selectedRegionsDelta', + 'regionFallbacksDelta', + 'unavailableRegionsDelta' +] +const REGION_KEYS = new Set(['asia-east2', 'us-central1', 'unhinted']) + +function integer(value, name) { + const parsed = Number(value) + if (!Number.isSafeInteger(parsed) || parsed < 0) throw new Error(`${name} is invalid`) + return parsed +} + +function digest(value, name) { + if (!/^sha256:[a-f0-9]{64}$/.test(value ?? '')) throw new Error(`${name} is invalid`) + return value +} + +function metric(value, name) { + if (!value || typeof value !== 'object' || Array.isArray(value)) { + throw new Error(`${name} is invalid`) + } + return Object.fromEntries(Object.entries(value).map(([key, count]) => { + if (!REGION_KEYS.has(key)) throw new Error(`${name} has an unknown aggregate key`) + return [key, integer(count, `${name}.${key}`)] + })) +} + +function sumMetric(total, value) { + for (const [key, count] of Object.entries(value)) total[key] = (total[key] ?? 0) + count +} + +function evidenceSha256(evidence) { + return createHash('sha256').update(JSON.stringify(evidence)).digest('hex') +} + +export function createRegionObservationEvidence(entries, bindings, now = Date.now()) { + if (!Array.isArray(entries)) throw new Error('runtime metrics response must be an array') + if (!/^[0-9a-f]{40}$/.test(bindings.commitSha ?? '')) throw new Error('commit SHA is invalid') + const directorImageDigest = digest(bindings.directorImageDigest, 'director digest') + const selectorGeneration = integer(bindings.selectorGeneration, 'selector generation') + const controlGeneration = integer(bindings.controlGeneration, 'control generation') + const start = now - WINDOW_MS + const buckets = Array.from({ length: 24 }, () => 0) + const totals = Object.fromEntries(METRICS.map((name) => [name, {}])) + let samples = 0 + for (const entry of entries) { + const timestamp = Date.parse(entry?.timestamp ?? '') + const payload = entry?.jsonPayload + if ( + !Number.isFinite(timestamp) || + timestamp < start || + timestamp > now + 60_000 || + payload?.event !== 'orca_relay_runtime_metrics' || + payload.role !== 'director' + ) continue + const bucket = Math.min(23, Math.floor((timestamp - start) / BUCKET_MS)) + buckets[bucket] += 1 + samples += 1 + for (const name of METRICS) sumMetric(totals[name], metric(payload[name], name)) + } + if (buckets.some((count) => count === 0)) { + throw new Error('24-hour region evidence has a missing hourly bucket') + } + if ( + !Number.isSafeInteger(totals.requestedRegionsDelta['asia-east2']) || + totals.requestedRegionsDelta['asia-east2'] < 1 || + !Number.isSafeInteger(totals.selectedRegionsDelta['asia-east2']) || + totals.selectedRegionsDelta['asia-east2'] < 1 + ) throw new Error('24-hour region evidence has no Asia request and selection activity') + const evidence = { + v: 1, + commitSha: bindings.commitSha, + directorImageDigest, + selectorGeneration, + controlGeneration, + windowStartedAt: start, + windowEndedAt: now, + hourlySampleCounts: buckets, + samples, + totals + } + return { evidence, sha256: evidenceSha256(evidence) } +} + +export function verifyRegionObservationEvidence(sealed, bindings) { + if ( + sealed?.sha256 !== evidenceSha256(sealed?.evidence) || + sealed.evidence?.commitSha !== bindings.commitSha || + sealed.evidence?.directorImageDigest !== bindings.directorImageDigest || + sealed.evidence?.selectorGeneration !== Number(bindings.selectorGeneration) || + sealed.evidence?.controlGeneration !== Number(bindings.controlGeneration) || + !Array.isArray(sealed.evidence?.hourlySampleCounts) || + sealed.evidence.hourlySampleCounts.length !== 24 || + sealed.evidence.hourlySampleCounts.some((count) => integer(count, 'bucket') < 1) || + integer(sealed.evidence?.totals?.requestedRegionsDelta?.['asia-east2'], 'Asia requests') < 1 || + integer(sealed.evidence?.totals?.selectedRegionsDelta?.['asia-east2'], 'Asia selections') < 1 + ) throw new Error('sealed 24-hour region evidence does not match enable authority') + return sealed.evidence +} + +function values(argv) { + const result = {} + for (let index = 0; index < argv.length; index += 2) { + if (!argv[index]?.startsWith('--') || argv[index + 1] === undefined) { + throw new Error('invalid arguments') + } + result[argv[index].slice(2)] = argv[index + 1] + } + return result +} + +async function stdinJson(input) { + const chunks = [] + for await (const chunk of input) chunks.push(chunk) + return JSON.parse(Buffer.concat(chunks).toString('utf8')) +} + +export async function main(argv = process.argv.slice(2), input = process.stdin) { + const command = argv.shift() + const args = values(argv) + const bindings = { + commitSha: args['commit-sha'], + directorImageDigest: args['director-image-digest'], + selectorGeneration: args['selector-generation'], + controlGeneration: args['control-generation'] + } + if (command === 'create') { + const sealed = createRegionObservationEvidence(await stdinJson(input), bindings) + process.stdout.write(`${JSON.stringify(sealed)}\n`) + return + } + if (command === 'verify') { + verifyRegionObservationEvidence(JSON.parse(readFileSync(args.file, 'utf8')), bindings) + return + } + throw new Error('unknown region observation evidence command') +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/relay-region-observation-evidence.test.mjs b/cloud/dev/scripts/relay-region-observation-evidence.test.mjs new file mode 100644 index 00000000000..3426c5da5d6 --- /dev/null +++ b/cloud/dev/scripts/relay-region-observation-evidence.test.mjs @@ -0,0 +1,53 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + createRegionObservationEvidence, + verifyRegionObservationEvidence +} from './relay-region-observation-evidence.mjs' + +const now = Date.parse('2026-08-14T12:00:00Z') +const bindings = { + commitSha: 'a'.repeat(40), + directorImageDigest: `sha256:${'b'.repeat(64)}`, + selectorGeneration: 11, + controlGeneration: 4 +} + +function entries() { + return Array.from({ length: 24 }, (_, index) => ({ + timestamp: new Date(now - (index * 60 + 30) * 60_000).toISOString(), + jsonPayload: { + event: 'orca_relay_runtime_metrics', + role: 'director', + requestedRegionsDelta: { 'asia-east2': index === 0 ? 2 : 0 }, + selectedRegionsDelta: { 'asia-east2': index === 0 ? 1 : 0 }, + regionFallbacksDelta: { 'asia-east2': index === 0 ? 1 : 0 }, + unavailableRegionsDelta: {} + } + })) +} + +test('seals all 24 hourly aggregate region buckets', () => { + const sealed = createRegionObservationEvidence(entries(), bindings, now) + assert.equal(sealed.evidence.hourlySampleCounts.length, 24) + assert.equal(sealed.evidence.totals.requestedRegionsDelta['asia-east2'], 2) + assert.equal(verifyRegionObservationEvidence(sealed, bindings).samples, 24) +}) + +test('rejects missing coverage, missing Asia activity, and changed bindings', () => { + assert.throws(() => createRegionObservationEvidence(entries().slice(1), bindings, now), /missing hourly/) + const noAsia = entries().map((entry) => ({ + ...entry, + jsonPayload: { + ...entry.jsonPayload, + requestedRegionsDelta: {}, + selectedRegionsDelta: {} + } + })) + assert.throws(() => createRegionObservationEvidence(noAsia, bindings, now), /no Asia/) + const sealed = createRegionObservationEvidence(entries(), bindings, now) + assert.throws(() => verifyRegionObservationEvidence(sealed, { + ...bindings, + commitSha: 'c'.repeat(40) + }), /does not match/) +}) diff --git a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs new file mode 100644 index 00000000000..5eab757255a --- /dev/null +++ b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs @@ -0,0 +1,193 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { fileURLToPath } from 'node:url' +import { relayWorkflowUrl } from './relay-repository.mjs' + +function workflow(name) { + return readFileSync( + fileURLToPath(relayWorkflowUrl(name)), + 'utf8' + ) +} + +test('same-cap wrapper is reusable, canary-bound, and sequential', () => { + const wrapper = workflow('deploy-relay-production-same-cap.yml') + const job = workflow('deploy-relay-production-same-cap-job.yml') + assert.match(wrapper, /options: \[verify, canary-apply, batch-apply, rollback\]/) + assert.match(wrapper, /relay-same-cap-canary-\$\{\{ inputs\.canary-run-id \}\}/) + assert.match(wrapper, /needs: \[gate, cell_1\]/) + assert.match(wrapper, /needs: \[gate, cell_2\]/) + assert.match(wrapper, /needs: \[gate, cell_3\]/) + assert.match(job, /on:\n workflow_call:/) + assert.match(job, /c27\|c28\|c29/) + assert.match(job, /EXPECTED_HARD_CAP=3000/) + assert.match(job, /EXPECTED_REGION=asia-east2/) + assert.match(job, /--hard-cap "\$\{EXPECTED_HARD_CAP\}"/) + assert.match(job, /--regional-rehome-protocol "\$\{DESIRED_REHOME_PROTOCOL\}"/) + assert.match(job, /--argjson protocol "\$\{PREDECESSOR_REHOME_PROTOCOL\}"/) + assert.match(job, /runtime predecessor mismatch fields=/) + // A rollback interrupted between apply and restore must be resumable. + assert.match(job, /ROLLBACK_RESUME=true/) + assert.match(job, /test "\$\{LIVE_IMAGE_DIGEST\}" = "\$\{DESIRED_IMAGE_DIGEST\}"/) + // Resume must skip BOTH the drain (no restart will clear the flag) and the + // apply (state already converged), and prove convergence instead. + assert.match( + job, + /Reversibly isolate and drain only the selected cell\n if: \$\{\{ inputs\.mode != 'verify' && env\.ROLLBACK_RESUME != 'true' \}\}/ + ) + assert.match( + job, + /Apply only the selected same-cap template and MIG\n if: \$\{\{ inputs\.mode != 'verify' && env\.ROLLBACK_RESUME != 'true' \}\}/ + ) + assert.match( + job, + /Require converged Terraform state and a stable MIG on resume\n if: \$\{\{ inputs\.mode != 'verify' && env\.ROLLBACK_RESUME == 'true' \}\}/ + ) + assert.match(job, /resume found unconverged resources/) + // A canary or batch cell that failed before its template apply also + // resumes here with template drift from repo changes since its last roll; + // only a plan the reviewed validator approves for the image the cell + // already serves may pass, and resume still applies nothing. + assert.match(job, /requiring reviewed rollback-image drift/) + assert.match( + job, + /--image "\$\{DESIRED_IMAGE\}" \\\n {16}--rollback-image "\$\{DESIRED_IMAGE\}"/ + ) + // The relaxation is only safe if the reviewed validator actually runs on + // the NON-converged branch, in same-cap-cell mode, with the trust config + // the validator requires, restricted to the template-and-MIG change pair. + assert.match( + job, + /if ! terraform -chdir=infra\/terraform show -json[\s\S]{0,220}\| length == 0' >\/dev\/null\n then\n/ + ) + assert.match( + job, + /requiring reviewed rollback-image drift'\n[\s\S]{0,400}?\n {16}--mode same-cap-cell --cell-id "\$\{TARGET_CELL_ID\}" \\\n/ + ) + assert.match( + job, + /Require converged Terraform state and a stable MIG on resume[\s\S]{0,200}DIRECTOR_RUNTIME_SERVICE_ACCOUNT: \$\{\{ vars\.PRODUCTION_GCP_RELAY_DIRECTOR_RUNTIME_SERVICE_ACCOUNT \}\}/ + ) + assert.match( + job, + /--rollback-image "\$\{DESIRED_IMAGE\}" \\\n {16}--rehome-director-service-account "\$\{DIRECTOR_RUNTIME_SERVICE_ACCOUNT\}"/ + ) + assert.match(job, /host-drain \\\n {14}\| jq -e '\.changes == 2' >\/dev\/null/) + assert.match(job, /resume requires the isolated migration-only cell/) + assert.match(job, /test "\$\{TARGET_INCARNATION\}" = "\$\{SOURCE_INCARNATION\}"/) + assert.match(job, /\(.regionalRehomeProtocol \/\/ 0\) == \$protocol/) + assert.match(job, /\(\.draining == false or \$drainingOk\)/) + // Selector expectations must follow the mutations' returned generations, + // not fixed offsets: isolate is a no-op on a cell a failed canary already + // isolated, and the restore inspect must expect post-restore membership. + assert.match(job, /SELECTOR_GENERATION_AFTER_ISOLATE=\$\{EFFECTIVE_SELECTOR_GENERATION\}/) + assert.match(job, /SELECTOR_GENERATION_AFTER_ISOLATE=\$\{ISOLATE_GENERATION\}/) + assert.match(job, /--expected-selector-generation "\$\{SELECTOR_GENERATION_AFTER_ISOLATE\}"/) + assert.match(job, /--expected-selector-generation "\$\{SELECTOR_GENERATION_AFTER_ACTIVATE\}"/) + assert.match(job, /--expected-migration-only-cells "\$\{RESTORED_MIGRATION_CELLS\}"/) + assert.match(job, /--expected-general-cells "\$\{RESTORED_GENERAL_CELLS\}"/) + assert.match(job, /FAILSAFE_GENERATION/) + // Later batch waves start after ~16-min predecessor rolls, so BOTH evidence + // age checks must scale by wave or cell_2+ can never pass; the bound's + // per-wave step is the cell job timeout, so the two must move together. + assert.match(job, /--required-migration-policy strict \\\n --wave-index "\$\{WAVE_INDEX\}"/) + assert.match(job, /dry-run\.state\.json" \\\n --wave-index "\$\{WAVE_INDEX\}" "\$\{RETRY_ARGS\[@\]\}"/) + assert.match(job, /timeout-minutes: 75/) + // Both age gates step by the cell job timeout above; the constant is + // duplicated across the two languages, so pin each copy to it. + for (const source of [ + '../../dev/scripts/relay-monitor-evidence.mjs', + '../../apps/relay-ops/src/incident-live-preflight-cli.ts' + ]) { + const body = readFileSync(fileURLToPath(new URL(source, import.meta.url)), 'utf8') + assert.match(body, /WAVE_PREDECESSOR_TIMEOUT_MS = 75 \* 60_000/) + assert.match(body, /\^\[0-3\]\$/) + } + // Aged-evidence replay via job re-runs is fenced: mutations are + // single-dispatch, so a failed cell needs a fresh gate and monitor run. + assert.match(job, /test "\$\{GITHUB_RUN_ATTEMPT\}" = 1/) + for (const index of [0, 1, 2, 3]) { + assert.match(wrapper, new RegExp(`wave-index: '${index}'`)) + } + assert.doesNotMatch(job, /EFFECTIVE_SELECTOR_GENERATION \+ 1\)/) + assert.doesNotMatch(job, /EFFECTIVE_SELECTOR_GENERATION \+ 2\)/) + assert.match(job, /\$region == "us-central1" and \$protocol == 0 and [.]region == null/) + assert.match(job, /[.]regionalRehomeProtocol \/\/ 0/) + assert.match(job, /runtime predecessor normalized legacy fields=/) + assert.match(job, /probe-relay-rehome-trust[.]mjs/) + assert.doesNotMatch(job, /service_account: \$\{\{ vars\.PRODUCTION_GCP_RELAY_(?:DIRECTOR_)?RUNTIME_SERVICE_ACCOUNT/) + assert.doesNotMatch(job, /roles\/iam\.serviceAccountTokenCreator/) +}) + +// Why: the same-cap caller defines release_lease itself, and a caller-defined job presents the +// caller as job_workflow_ref, so the pair must admit the caller alongside its reusable job. +test('shared deploy WIF admits the exact same-cap reusable workflow pair and the caller itself', () => { + const terraform = readFileSync( + fileURLToPath(new URL('../../infra/terraform/relay-github-actions.tf', import.meta.url)), + 'utf8' + ) + const providerStart = terraform.indexOf( + 'resource "google_iam_workload_identity_pool_provider" "github"' + ) + const providerEnd = terraform.indexOf('\nresource "', providerStart + 1) + const sharedProvider = terraform.slice(providerStart, providerEnd) + assert.ok(providerStart >= 0 && providerEnd > providerStart) + assert.match(sharedProvider, /local\.relay_github_workflow_conditions\["github"\]/) + // The pairing itself now lives in the clause the provider renders, once per accepted repository. + assert.match( + terraform, + /assertion\.workflow_ref == '\$\{prefix\}\$\{local\.github_production_relay_same_cap_workflow_file\}@refs\/heads\/main' && \(assertion\.job_workflow_ref == '\$\{prefix\}\$\{local\.github_production_relay_same_cap_job_workflow_file\}@refs\/heads\/main' \|\| assertion\.job_workflow_ref == '\$\{prefix\}\$\{local\.github_production_relay_same_cap_workflow_file\}@refs\/heads\/main'\)/ + ) +}) + +test('pause and disable precede optional installation and cloud diagnostics', () => { + const job = workflow('operate-relay-production-rehome-job.yml') + const emergency = job.indexOf('Apply emergency durable pause or disable before diagnostics') + const install = job.indexOf('pnpm install --frozen-lockfile') + const revision = job.indexOf('Verify exact serving and rollback director identities') + assert.ok(emergency > 0) + assert.ok(emergency < install) + assert.ok(emergency < revision) + assert.match(job, /inputs\.mode == 'pause' \|\| inputs\.mode == 'disable'/) + assert.match(job, /Seal 24-hour aggregate region observation evidence/) + assert.match(job, /--freshness=25h --limit=30000/) + assert.match(job, /relay-region-observation-\$\{\{ github\.run_id \}\}-\$\{\{ github\.run_attempt \}\}/) + assert.match(job, /test "\$\{RATE_PER_MINUTE\}" = 10/) +}) + +test('a failed enable independently restores and verifies durable disabled state', () => { + const job = workflow('operate-relay-production-rehome-job.yml') + const enable = job.indexOf('Apply exact durable regional rehome enable') + const evidence = job.indexOf('Read fresh aggregate completion and abort evidence') + const summary = job.indexOf('Publish aggregate control evidence') + const recovery = job.indexOf('Fail closed after an unsuccessful enable run') + assert.ok(enable > 0 && enable < evidence && evidence < summary && summary < recovery) + const recoveryStep = job.slice(recovery) + assert.match( + recoveryStep, + /failure\(\) && inputs\.mode == 'enable' && steps\.google-auth\.outcome == 'success'/ + ) + assert.match(recoveryStep, /--mode recover-enable/) + assert.match(recoveryStep, /--expected-control-generation "\$\{EXPECTED_CONTROL_GENERATION\}"/) + assert.match(recoveryStep, /RECOVER_FAILED_REGIONAL_REHOME_ENABLE/) + assert.match(recoveryStep, /\.control\.enabled == false/) + assert.doesNotMatch(recoveryStep, /gcloud|pnpm/) +}) + +test('director rollout has a strict one-time identity bootstrap', () => { + const workflowBody = workflow('deploy-relay-production-director.yml') + const script = readFileSync( + fileURLToPath(new URL('./deploy-relay-blue-green.mjs', import.meta.url)), + 'utf8' + ) + assert.match(workflowBody, /BOOTSTRAP_RELAY_DIRECTOR_REHOME_IDENTITY/) + assert.match(workflowBody, /--predecessor-runtime-service-account/) + assert.match(workflowBody, /--expected-rehome-generation/) + assert.match(script, /args\.push\('--service-account', config\['runtime-service-account'\]\)/) + assert.match(script, /director predecessor runtime service account does not match/) + const candidateProof = script.indexOf('await verifyRehomeDisabled(candidate.origin)') + const trafficMove = script.indexOf('operations.updateTraffic(config, [`--to-tags=') + assert.ok(candidateProof > 0 && candidateProof < trafficMove) + assert.equal(script.indexOf('verifyRehomeDisabled', trafficMove), -1) +}) diff --git a/cloud/dev/scripts/relay-rehome-aggregate-evidence.mjs b/cloud/dev/scripts/relay-rehome-aggregate-evidence.mjs new file mode 100644 index 00000000000..82bc2fb522a --- /dev/null +++ b/cloud/dev/scripts/relay-rehome-aggregate-evidence.mjs @@ -0,0 +1,65 @@ +import { pathToFileURL } from 'node:url' + +const INVENTORY = /^\[orca-relay\] regional rehome inventory active=(\d+) awaitingReceipt=(\d+) targetRegistered=(\d+) completedLast24Hours=(\d+) abortedLast24Hours=(\d+) oldestActiveAgeMs=(none|\d+)$/ + +function count(value, name) { + const parsed = Number(value) + if (!Number.isSafeInteger(parsed) || parsed < 0) throw new Error(`${name} is invalid`) + return parsed +} + +export function parseRegionalRehomeInventory(entries, options = {}) { + if (!Array.isArray(entries)) throw new Error('logging response must be an array') + const parsed = entries.flatMap((entry) => { + const match = INVENTORY.exec(entry?.textPayload ?? '') + const timestamp = Date.parse(entry?.timestamp ?? '') + if (!match || !Number.isFinite(timestamp)) return [] + return [{ + timestamp, + active: count(match[1], 'active'), + awaitingReceipt: count(match[2], 'awaiting receipt'), + targetRegistered: count(match[3], 'target registered'), + completedLast24Hours: count(match[4], 'completed'), + abortedLast24Hours: count(match[5], 'aborted'), + oldestActiveAgeMs: match[6] === 'none' ? null : count(match[6], 'oldest active age') + }] + }).sort((left, right) => right.timestamp - left.timestamp) + if (parsed.length === 0) throw new Error('no aggregate regional rehome inventory evidence') + const latest = parsed[0] + const now = options.now ?? Date.now() + const maxAgeMs = options.maxAgeMs ?? 15 * 60_000 + if (latest.timestamp > now + 60_000 || latest.timestamp < now - maxAgeMs) { + throw new Error('aggregate regional rehome inventory evidence is stale') + } + return latest +} + +function argumentsMap(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + if (!argv[index]?.startsWith('--') || argv[index + 1] === undefined) { + throw new Error('invalid arguments') + } + values[argv[index].slice(2)] = argv[index + 1] + } + return values +} + +export async function main(argv = process.argv.slice(2), input = process.stdin) { + const values = argumentsMap(argv) + const maxAgeMs = count(values['max-age-ms'] ?? 900_000, '--max-age-ms') + const chunks = [] + for await (const chunk of input) chunks.push(chunk) + const evidence = parseRegionalRehomeInventory( + JSON.parse(Buffer.concat(chunks).toString('utf8')), + { maxAgeMs } + ) + process.stdout.write(`${JSON.stringify({ event: 'relay_rehome_aggregate_evidence', ...evidence })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/relay-rehome-aggregate-evidence.test.mjs b/cloud/dev/scripts/relay-rehome-aggregate-evidence.test.mjs new file mode 100644 index 00000000000..2ce2627fb93 --- /dev/null +++ b/cloud/dev/scripts/relay-rehome-aggregate-evidence.test.mjs @@ -0,0 +1,38 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { parseRegionalRehomeInventory } from './relay-rehome-aggregate-evidence.mjs' + +const now = Date.parse('2026-08-14T12:00:00Z') + +test('selects the newest fresh aggregate-only regional rehome inventory', () => { + const result = parseRegionalRehomeInventory([ + { + timestamp: '2026-08-14T11:58:00Z', + textPayload: '[orca-relay] regional rehome inventory active=2 awaitingReceipt=1 targetRegistered=1 completedLast24Hours=9 abortedLast24Hours=0 oldestActiveAgeMs=30000' + }, + { + timestamp: '2026-08-14T11:50:00Z', + textPayload: '[orca-relay] regional rehome inventory active=1 awaitingReceipt=0 targetRegistered=1 completedLast24Hours=8 abortedLast24Hours=0 oldestActiveAgeMs=none' + } + ], { now, maxAgeMs: 5 * 60_000 }) + assert.deepEqual(result, { + timestamp: Date.parse('2026-08-14T11:58:00Z'), + active: 2, + awaitingReceipt: 1, + targetRegistered: 1, + completedLast24Hours: 9, + abortedLast24Hours: 0, + oldestActiveAgeMs: 30_000 + }) +}) + +test('rejects stale, malformed, and identity-bearing lookalikes', () => { + assert.throws(() => parseRegionalRehomeInventory([{ + timestamp: '2026-08-14T11:00:00Z', + textPayload: '[orca-relay] regional rehome inventory active=0 awaitingReceipt=0 targetRegistered=0 completedLast24Hours=0 abortedLast24Hours=0 oldestActiveAgeMs=none' + }], { now, maxAgeMs: 5 * 60_000 }), /stale/) + assert.throws(() => parseRegionalRehomeInventory([{ + timestamp: '2026-08-14T11:59:00Z', + textPayload: '[orca-relay] regional rehome inventory active=0 hostId=secret' + }], { now }), /no aggregate/) +}) diff --git a/cloud/dev/scripts/relay-repository.mjs b/cloud/dev/scripts/relay-repository.mjs new file mode 100644 index 00000000000..bf41bed8012 --- /dev/null +++ b/cloud/dev/scripts/relay-repository.mjs @@ -0,0 +1,29 @@ +import { readFileSync } from 'node:fs' + +// Single place naming the repository the Relay workflows live in and where their files sit. The +// public-repo copy moves this tree under cloud/, prefixes every workflow filename, and changes the +// owning repository, so only this module changes: nothing else may restate any of the three. +export const RELAY_GITHUB_REPOSITORY = 'stablyai/orca' + +export const RELAY_WORKFLOW_FILE_PREFIX = 'cloud-' + +// Where .github/workflows sits relative to this file. Workflows stay at the repository root while +// this tree moves under cloud/, so the depth changes at the copy even though the layout does not. +export const RELAY_WORKFLOW_DIRECTORY = new URL('../../../.github/workflows/', import.meta.url) + +export function relayWorkflowFile(name) { + return `${RELAY_WORKFLOW_FILE_PREFIX}${name}` +} + +// Repository-relative path, the shape GitHub reports in workflow_ref and evidence payloads. +export function relayWorkflowPath(name) { + return `.github/workflows/${relayWorkflowFile(name)}` +} + +export function relayWorkflowUrl(name) { + return new URL(relayWorkflowFile(name), RELAY_WORKFLOW_DIRECTORY) +} + +export function readRelayWorkflow(name) { + return readFileSync(relayWorkflowUrl(name), 'utf8') +} diff --git a/cloud/dev/scripts/relay-repository.test.mjs b/cloud/dev/scripts/relay-repository.test.mjs new file mode 100644 index 00000000000..56383db4a1e --- /dev/null +++ b/cloud/dev/scripts/relay-repository.test.mjs @@ -0,0 +1,36 @@ +import assert from 'node:assert/strict' +import { readdirSync, readFileSync } from 'node:fs' +import test from 'node:test' +import { fileURLToPath } from 'node:url' +import { + RELAY_GITHUB_REPOSITORY, + RELAY_WORKFLOW_FILE_PREFIX, + readRelayWorkflow, + relayWorkflowFile, + relayWorkflowPath, + relayWorkflowUrl +} from './relay-repository.mjs' + +const directory = fileURLToPath(new URL('.', import.meta.url)) +// The Relay copy takes the scripts named for it. Everything else stays with the applications. +const relayScripts = readdirSync(directory) + .filter((name) => name.includes('relay') && name.endsWith('.mjs')) + .filter((name) => !name.startsWith('relay-repository.')) + +test('workflow identity is derived, never restated', () => { + assert.equal(relayWorkflowFile('deploy-relay-staging.yml'), `${RELAY_WORKFLOW_FILE_PREFIX}deploy-relay-staging.yml`) + assert.equal(relayWorkflowPath('deploy-relay-staging.yml'), `.github/workflows/${relayWorkflowFile('deploy-relay-staging.yml')}`) + assert.ok(relayWorkflowUrl('deploy-relay-staging.yml').pathname.endsWith(relayWorkflowPath('deploy-relay-staging.yml'))) + assert.match(readRelayWorkflow('deploy-relay-staging.yml'), /^name:/m) + assert.match(RELAY_GITHUB_REPOSITORY, /^[\w.-]+\/[\w.-]+$/) +}) + +// Why: the public-repo copy changes the owning repository, the workflow filenames, and the depth +// this tree sits at. Each has to be one edit here, so no Relay script may restate any of them. +test('no Relay script restates the repository or the workflow directory', () => { + for (const name of relayScripts) { + const text = readFileSync(`${directory}${name}`, 'utf8') + assert.doesNotMatch(text, /stablyai\//, `${name} restates the GitHub repository`) + assert.doesNotMatch(text, /\.github\/workflows/, `${name} restates the workflow directory`) + } +}) diff --git a/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs b/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs new file mode 100644 index 00000000000..1f0e6fce3ad --- /dev/null +++ b/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs @@ -0,0 +1,159 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import { test } from 'node:test' +import { relayWorkflowFile, relayWorkflowUrl } from './relay-repository.mjs' + +const workflow = readFileSync( + relayWorkflowUrl('prove-relay-staging-capacity.yml'), + 'utf8' +) +const recoveryWorkflow = readFileSync( + relayWorkflowUrl('recover-relay-staging-c4-image.yml'), + 'utf8' +) +const requeueWorkflow = readFileSync( + relayWorkflowUrl('requeue-relay-staging-c4-recovery.yml'), + 'utf8' +) +const githubActions = readFileSync( + new URL('../../infra/terraform/relay-github-actions.tf', import.meta.url), + 'utf8' +) +const cells = readFileSync( + new URL('../../infra/terraform/relay-gce-cells.tf', import.meta.url), + 'utf8' +) +const relay = readFileSync(new URL('../../infra/terraform/relay.tf', import.meta.url), 'utf8') +const stagingTfvars = readFileSync( + new URL('../../infra/terraform/environments/staging.tfvars', import.meta.url), + 'utf8' +) +const productionTfvars = readFileSync( + new URL('../../infra/terraform/environments/production.tfvars', import.meta.url), + 'utf8' +) + +const launchDigest = '5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563' + +// Scoped to the Asia cells by name: the production capacity cells now serve this digest too, +// so a file-wide count no longer isolates Asia. +const asiaCells = ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'] + +function productionCell(cellId) { + const start = productionTfvars.indexOf(`"${cellId}"`) + assert.notEqual(start, -1, `${cellId} is missing`) + return productionTfvars.slice(start, productionTfvars.indexOf('\n }', start)) +} + +test('pins staging C4 and all production Asia cells to the same launch image', () => { + assert.equal(stagingTfvars.match(new RegExp(launchDigest, 'g'))?.length, 1) + for (const cellId of asiaCells) { + assert.match(productionCell(cellId), new RegExp(`relay@sha256:${launchDigest}"`), cellId) + } + assert.match(recoveryWorkflow, new RegExp(`TARGET_IMAGE_DIGEST: sha256:${launchDigest}`)) +}) + +test('refreshes only empty staging C4 through the trusted capacity identity', () => { + const refresh = workflow.slice(workflow.indexOf(' refresh-asia-c4-image:')) + assert.match(refresh, /terraform_version: 1\.15\.8/) + assert.doesNotMatch(refresh, /terraform_version: 1\.5\.7/) + assert.match(refresh, /github\.ref == 'refs\/heads\/main'/) + assert.match(refresh, /REFRESH_STAGING_ASIA_C4_IMAGE/) + assert.match(refresh, /APPROVED_PREDECESSOR_IMAGE_DIGEST/) + assert.match(refresh, /--activity quiescent/) + assert.match(refresh, /--admission migration-only/) + assert.match(refresh, /fence_digests="\$\{PREDECESSOR_IMAGE_DIGEST\}"/) + assert.match(refresh, /--expected-image-digests "\$\{TARGET_IMAGE_DIGEST\}"/) + assert.match(refresh, /--mode same-cap-image/) + assert.match(refresh, /test "\$\(jq -r '\.changes'/) + assert.match(refresh, /REFRESH_PHASE=\$\{refresh_phase\}/) + assert.match(refresh, /MUTATION_STARTED=false/) + assert.match(refresh, /MIG_STABLE_AT_MS=/) + assert.match(refresh, /PLAN_CHANGES=/) + assert.match(refresh, /obsolete-template-delete/) + assert.match( + refresh, + /\*:manager-convergence\|\*:replacement-with-obsolete-template\) refresh_phase=converging/ + ) + assert.match(refresh, /--mode isolate/) + assert.match(refresh, /--mode verify/) + assert.match(refresh, /\.status\.runtime\.ready/) + assert.match(refresh, /\.status\.runtime\.startedAt/) + assert.match(refresh, /\.status\.runtime\.lastHeartbeatAt/) + assert.match(refresh, /--runtime unavailable/) + const plan = refresh.indexOf('Save, validate, and classify the exact C4 plan') + const currentState = refresh.indexOf('Verify the exact selector and current C4 state') + const isolate = refresh.indexOf('--mode isolate') + const apply = refresh.indexOf('terraform -chdir=infra/terraform apply -auto-approve') + assert.ok(plan < currentState && currentState < isolate && isolate < apply) + assert.match(refresh, /Require an empty targeted Terraform readback/) +}) + +test('recovers a failed or cancelled C4 refresh from an independent workflow', () => { + assert.match(recoveryWorkflow, /terraform_version: 1\.15\.8/) + assert.doesNotMatch(recoveryWorkflow, /terraform_version: 1\.5\.7/) + assert.match(recoveryWorkflow, /workflow_run:/) + assert.match(recoveryWorkflow, /workflows: \[Prove Relay Staging Capacity\]/) + assert.match(recoveryWorkflow, /RECOVER_STAGING_ASIA_C4_IMAGE/) + assert.match(recoveryWorkflow, /outputs:\n\s+recover: \$\{\{ steps\.trigger\.outputs\.recover \}\}/) + assert.match(recoveryWorkflow, /if test "\$\{count\}" = 0; then\n\s+echo "recover=false"/) + assert.match(recoveryWorkflow, /needs: gate\n\s+if: \$\{\{ needs\.gate\.outputs\.recover == 'true' \}\}/) + assert.match(recoveryWorkflow, /concurrency:\n\s+group: relay-staging-mutation/) + assert.match(recoveryWorkflow, /group: relay-staging-mutation/) + assert.match(recoveryWorkflow, /\.name == "refresh-asia-c4-image"/) + assert.match(recoveryWorkflow, /PREDECESSOR_IMAGE_DIGEST: sha256:ce16d13/) + assert.match(recoveryWorkflow, /TARGET_IMAGE_DIGEST: sha256:5aedbca5/) + assert.match(recoveryWorkflow, /id: preflight-auth/) + assert.match(recoveryWorkflow, /id: verify-auth/) + assert.match(recoveryWorkflow, /\.status\.runtime\.ready/) + assert.match(recoveryWorkflow, /--runtime unavailable/) + assert.match(recoveryWorkflow, /current_digest.*\^sha256:\[a-f0-9\]\{64\}\$/) + assert.match(recoveryWorkflow, /expected_digests="\$\{expected_digests\},\$\{current_digest\}"/) + assert.match(recoveryWorkflow, /--timeout-ms 240000/) + assert.doesNotMatch(recoveryWorkflow, /--timeout-ms 900000/) + assert.match(recoveryWorkflow, /-var manage_artifact_dns=false -lock-timeout=5m/) + assert.match(recoveryWorkflow, /--mode same-cap-image/) + assert.match(recoveryWorkflow, /test "\$\(jq -r '\.changes'/) + assert.equal(recoveryWorkflow.match(/\*:replacement-with-obsolete-template/g)?.length, 2) + assert.match( + recoveryWorkflow, + /\^\(replacement\|replacement-with-obsolete-template\|manager-convergence\)\$/ + ) + const preflightAuth = recoveryWorkflow.indexOf('id: preflight-auth') + const preflight = recoveryWorkflow.indexOf('Inspect the exact C4 recovery state') + const recoveryPlan = recoveryWorkflow.indexOf('Classify both exact recovery end states') + const apply = recoveryWorkflow.indexOf('Apply and stabilize the saved predecessor plan') + const restore = recoveryWorkflow.indexOf('Restart only when the plan did not replace C4') + const verifyAuth = recoveryWorkflow.indexOf('id: verify-auth') + const verify = recoveryWorkflow.indexOf('Verify the recovered image and unchanged isolation') + assert.ok(preflightAuth < preflight && preflight < recoveryPlan && recoveryPlan < apply) + assert.ok(apply < restore) + assert.ok(restore < verifyAuth && verifyAuth < verify) + assert.match(githubActions, /"recover-relay-staging-c4-image\.yml"/) +}) + +test('requeues a protected C4 recovery cancelled while pending', () => { + assert.match(requeueWorkflow, /workflows: \[Recover Relay Staging C4 Image\]/) + assert.match(requeueWorkflow, /conclusion == 'cancelled'/) + assert.match(requeueWorkflow, /permissions:\n\s+actions: write\n\s+contents: read/) + assert.match(requeueWorkflow, /group: relay-staging-c4-recovery-requeue/) + assert.match(requeueWorkflow, /\.name == "recover" and\n\s+\.conclusion == "cancelled"/) + assert.match(requeueWorkflow, /\.started_at == null/) + assert.match(requeueWorkflow, /\.name == "gate" and \.conclusion == "success"/) + assert.match(requeueWorkflow, /\.name == "recover" and \.status != "completed"/) + assert.match(requeueWorkflow, /actions\/runs\/\$\{run_id\}\/jobs\?filter=latest/) + assert.match(requeueWorkflow, /if test "\$\{active\}" != 0; then exit 0; fi/) + assert.ok(requeueWorkflow.includes(`gh workflow run ${relayWorkflowFile('recover-relay-staging-c4-image.yml')}`)) + assert.match(requeueWorkflow, /-f confirmation=RECOVER_STAGING_ASIA_C4_IMAGE/) + assert.doesNotMatch(requeueWorkflow, /id-token: write/) + assert.doesNotMatch(requeueWorkflow, /relay-staging-mutation/) +}) + +test('keeps cell-only plans independent from service-account description drift', () => { + assert.match(cells, /runtime_service_account\s+= local\.relay_runtime_service_account_email/) + assert.match( + cells, + /rehome_director_service_account\s+= local\.relay_director_runtime_service_account_email/ + ) + assert.match(relay, /var\.environment == "staging" \? "Orca Relay"/) +}) diff --git a/cloud/dev/scripts/relay-staging-capacity-identity.test.mjs b/cloud/dev/scripts/relay-staging-capacity-identity.test.mjs new file mode 100644 index 00000000000..31a38d08124 --- /dev/null +++ b/cloud/dev/scripts/relay-staging-capacity-identity.test.mjs @@ -0,0 +1,269 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { relayWorkflowUrl } from './relay-repository.mjs' + +const terraform = readFileSync( + new URL('../../infra/terraform/relay-github-actions.tf', import.meta.url), + 'utf8' +) +const outputs = readFileSync(new URL('../../infra/terraform/outputs.tf', import.meta.url), 'utf8') +const workflow = readFileSync( + relayWorkflowUrl('prove-relay-staging-capacity.yml'), + 'utf8' +) +const deployWorkflow = readFileSync( + relayWorkflowUrl('deploy-relay-staging.yml'), + 'utf8' +) +const publishWorkflow = readFileSync( + relayWorkflowUrl('publish-relay-production.yml'), + 'utf8' +) +const bootstrapWorkflow = readFileSync( + relayWorkflowUrl('bootstrap-relay-staging-capacity.yml'), + 'utf8' +) + +function resource(type, name) { + const start = terraform.indexOf(`resource "${type}" "${name}"`) + assert.notEqual(start, -1, `${type}.${name} is missing`) + const next = terraform.indexOf('\nresource "', start + 1) + return terraform.slice(start, next === -1 ? undefined : next) +} + +function terraformStringList(block, attribute) { + const match = block.match(new RegExp(`${attribute}\\s*=\\s*\\[([\\s\\S]*?)\\]`)) + assert.ok(match, `${attribute} is missing`) + return [...match[1].matchAll(/"([^"]+)"/g)].map((entry) => entry[1]) +} + +test('capacity workflow uses only its exact staging identity', () => { + assert.match(workflow, /STAGING_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(workflow, /STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT/) + assert.doesNotMatch(workflow, /vars\.STAGING_GCP_WORKLOAD_IDENTITY_PROVIDER/) + assert.doesNotMatch(workflow, /vars\.STAGING_GCP_DEPLOY_SERVICE_ACCOUNT/) + + const provider = resource( + 'google_iam_workload_identity_pool_provider', + 'github_staging_relay_capacity' + ) + // The three repository claims are pinned once in relay-shared.tf; every provider concatenates + // that list rather than restating the repository on its own. + assert.match(provider, /concat\(local\.relay_github_leading_repository_claims, \[/) + for (const boundary of [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'staging'" + ]) { + assert.match(provider, new RegExp(boundary.replaceAll(/[.*+?^${}()|[\]\\]/g, '\\$&'))) + } + assert.match(provider, /local\.relay_github_workflow_conditions\["github_staging_relay_capacity"\]/) + assert.deepEqual( + terraformStringList(terraform, 'github_staging_relay_capacity_workflow_files'), + [ + 'bootstrap-relay-staging-capacity.yml', + 'prove-relay-staging-capacity.yml', + 'recover-relay-staging-c4-image.yml' + ] + ) + assert.match( + terraform, + /for workflow_file in local\.github_staging_relay_capacity_workflow_files : "assertion\.workflow_ref == '\$\{prefix\}\$\{workflow_file\}@refs\/heads\/main'"/ + ) + assert.doesNotMatch(terraform, /github_staging_relay_capacity_workflow_file\s*=/) +}) + +test('job gates do not read environment variables before the environment is attached', () => { + for (const source of [workflow, deployWorkflow, bootstrapWorkflow]) { + const jobGate = source.match(/^\s{4}if:.*$/m)?.[0] ?? '' + assert.doesNotMatch(jobGate, /STAGING_GCP_RELAY_CAPACITY_/) + } +}) + +test('capacity identity has bounded mutation and state permissions', () => { + const role = resource( + 'google_project_iam_custom_role', + 'github_staging_relay_capacity_mutation' + ) + assert.deepEqual(terraformStringList(role, 'permissions'), [ + 'compute.disks.create', + 'compute.healthChecks.use', + 'compute.images.useReadOnly', + 'compute.instanceGroupManagers.get', + 'compute.instanceGroupManagers.update', + 'compute.instances.create', + 'compute.instances.setLabels', + 'compute.instances.setMetadata', + 'compute.instances.setTags', + 'compute.instanceTemplates.create', + 'compute.instanceTemplates.delete', + 'compute.instanceTemplates.get', + 'compute.instanceTemplates.useReadOnly', + 'compute.networks.use', + 'compute.subnetworks.use', + 'compute.zoneOperations.get' + ]) + assert.doesNotMatch( + role, + /compute\.(?:disks\.delete|instances\.(?:delete|start|stop|update))|cloudsql|secretmanager/ + ) + + for (const source of [workflow, bootstrapWorkflow]) { + assert.match(source, /instance-groups managed recreate-instances/) + assert.match(source, /--instances/) + assert.doesNotMatch(source, /rolling-action restart/) + } + + const state = resource( + 'google_storage_bucket_iam_member', + 'github_staging_relay_capacity_state' + ) + assert.match(state, /roles\/storage\.objectAdmin/) + assert.match(state, /objects\/terraform\/state\/default\.tfstate/) + assert.match(state, /objects\/terraform\/state\/default\.tflock/) + assert.doesNotMatch(state, /resource\.name\.startsWith/) + + const runtime = resource( + 'google_service_account_iam_member', + 'github_staging_relay_capacity_runtime_user' + ) + assert.match(runtime, /google_service_account\.relay_runtime\.name/) + assert.match(runtime, /roles\/iam\.serviceAccountUser/) + + const cloudRun = resource( + 'google_cloud_run_v2_service_iam_member', + 'github_staging_relay_capacity_developer' + ) + assert.match(cloudRun, /name\s*=\s*var\.relay_cloud_run_service_name/) + assert.doesNotMatch(cloudRun, /google_cloud_run_v2_service\.relay/) + + const relay = readFileSync(new URL('../../infra/terraform/relay.tf', import.meta.url), 'utf8') + const startup = readFileSync( + new URL('../../infra/terraform/relay-gce-startup.sh.tftpl', import.meta.url), + 'utf8' + ) + assert.match(relay, /ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT/) + assert.match(startup, /ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT/) +}) + +test('capacity identity exposes only its provider and service account', () => { + assert.match(outputs, /output "github_staging_relay_capacity_workload_identity_provider"/) + assert.match(outputs, /output "github_staging_relay_capacity_service_account"/) +}) + +test('director capacity configuration stays on the audited blue-green path', () => { + assert.match(deployWorkflow, /STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT/) + assert.match(deployWorkflow, /--capacity-service-account "\$\{CAPACITY_SERVICE_ACCOUNT\}"/) + assert.match(deployWorkflow, /expected-image-digest/) + assert.match(deployWorkflow, /var\.relay_gce_cells\["staging-gce-c4"\]\.image/) + assert.match(deployWorkflow, /init -reconfigure \\\n\s+-backend-config=backend\/staging\.hcl/) + assert.ok( + deployWorkflow.indexOf('id: google-auth') < + deployWorkflow.indexOf('Bind the request to the checked-in staging C4 image') + ) + assert.match(deployWorkflow, /artifacts docker images describe "\$\{IMAGE\}"/) + assert.doesNotMatch(deployWorkflow, /docker (?:build|push)/) + assert.match(workflow, /--director-cells-json "\$\{DESIRED_CELLS_JSON\}"/) + assert.doesNotMatch(workflow, /target=google_cloud_run_v2_service\.relay/) + assert.doesNotMatch(workflow, /--mode director/) +}) + +test('mirrors the exact production manifest through the production deploy identity', () => { + assert.match(publishWorkflow, /options: \[publish, mirror-staging\]/) + assert.match(publishWorkflow, /MIRROR_RELAY_PRODUCTION_IMAGE_TO_STAGING/) + assert.match(publishWorkflow, /docker pull "\$\{source_image\}"/) + assert.match(publishWorkflow, /docker tag "\$\{source_image\}" "\$\{target_tag\}"/) + assert.match(publishWorkflow, /test "\$\{source_digest\}" = "\$\{MIRROR_DIGEST\}"/) + assert.match(publishWorkflow, /test "\$\{target_digest\}" = "\$\{MIRROR_DIGEST\}"/) + const mirrorWriter = resource( + 'google_artifact_registry_repository_iam_member', + 'github_production_relay_staging_mirror_writer' + ) + assert.match(mirrorWriter, /var\.environment == "staging"/) + assert.match(mirrorWriter, /roles\/artifactregistry\.writer/) + assert.match( + mirrorWriter, + /serviceAccount:orca-cloud-gha-deploy@onorca-cloud\.iam\.gserviceaccount\.com/ + ) +}) + +test('cells bootstrap one at a time with bounded deploy and capacity identities', () => { + assert.match(bootstrapWorkflow, /STAGING_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(bootstrapWorkflow, /STAGING_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT/) + assert.match(bootstrapWorkflow, /STAGING_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER/) + assert.match(bootstrapWorkflow, /STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT/) + assert.match(bootstrapWorkflow, /--mode bootstrap-cell/) + assert.match(bootstrapWorkflow, /--cell-id "\$\{fallback_cell_id\}"/) + assert.match(bootstrapWorkflow, /google_compute_instance_template\.relay_gce_cell/) + assert.match(bootstrapWorkflow, /google_compute_instance_group_manager\.relay_gce_cell/) + assert.match(bootstrapWorkflow, /--mode restore-fallback/) + assert.doesNotMatch(bootstrapWorkflow, /target=google_cloud_run_v2_service\.relay/) + assert.ok( + bootstrapWorkflow.indexOf('id: deploy-auth') < bootstrapWorkflow.indexOf('id: capacity-auth') + ) + assert.match( + bootstrapWorkflow, + /restore_fallback\(\) \{[\s\S]*?verify_fallback[\s\S]*?--mode restore-fallback/ + ) + const rollCell = bootstrapWorkflow.slice(bootstrapWorkflow.indexOf('roll_cell()')) + assert.ok( + rollCell.indexOf('verify_fallback\n trap restore_fallback EXIT') < + rollCell.indexOf('--mode isolate') + ) + assert.match( + bootstrapWorkflow, + /staging-gce-c2 general[\s\S]*?trap restore_legacy_c3_fallback EXIT[\s\S]*?--mode isolate/ + ) + const normalize = bootstrapWorkflow.slice( + bootstrapWorkflow.indexOf('normalize_legacy_c3() {'), + bootstrapWorkflow.indexOf('\n roll_cell()', bootstrapWorkflow.indexOf('normalize_legacy_c3() {')) + ) + const trapInstalled = normalize.indexOf('trap restore_legacy_c3_fallback EXIT') + const isolated = normalize.indexOf('--mode isolate', trapInstalled) + const recreated = normalize.indexOf('recreate-instances', isolated) + const restored = normalize.indexOf('--mode restore', recreated) + const c3Verified = normalize.indexOf('staging-gce-c3 general', restored) + const restoreDisabled = normalize.indexOf('legacy_c3_isolated=false', c3Verified) + const trapCleared = normalize.indexOf('trap - EXIT', restoreDisabled) + assert.ok( + trapInstalled < isolated && + isolated < recreated && + recreated < restored && + restored < c3Verified && + c3Verified < restoreDisabled && + restoreDisabled < trapCleared + ) + assert.equal(normalize.indexOf('trap - EXIT', trapInstalled), trapCleared) + assert.match(bootstrapWorkflow, /--heartbeat either/) + assert.doesNotMatch(bootstrapWorkflow, /heartbeat=stale/) + assert.match( + rollCell, + /"\$\{desired_cap\}" "\$\{desired_bound\}" absent-or-stale[\s\S]*?deploy-relay-blue-green\.mjs[\s\S]*?"\$\{desired_cap\}" "\$\{desired_bound\}"/ + ) +}) + +test('workflows read desired topology from configuration and gate exact predecessors', () => { + assert.match(workflow, /<<< 'local\.relay_director_cells_json' \| jq -r '\.'/) + assert.match(bootstrapWorkflow, /<<< 'local\.relay_director_cells_json' \| jq -r '\.'/) + assert.match(workflow, /1000\/0\)[\s\S]*?PREDECESSOR_C3_CAP=600/) + assert.match(workflow, /1000\/60\)[\s\S]*?PREDECESSOR_C3_BOUND=0/) + assert.match(workflow, /600\/60\)[\s\S]*?PREDECESSOR_C3_CAP=1000/) + assert.match(workflow, /Unsupported staging capacity transition/) +}) + +test('capacity apply resumes after director or cell success and preserves the no-op restart proof', () => { + for (const phase of ['predecessor', 'director-ready', 'cell-ready', 'cell-active']) { + assert.match(workflow, new RegExp(`TRANSITION_PHASE=${phase}`)) + } + assert.match(workflow, /--argjson expected "\$\{PREDECESSOR_CELLS_JSON\}"/) + assert.match(workflow, /--argjson expected "\$\{DESIRED_CELLS_JSON\}"/) + assert.match( + workflow, + /test "\$\{TRANSITION_PHASE\}" = cell-ready; then[\s\S]*?test "\$\{CELL_PLAN_CHANGES\}" = 0/ + ) + assert.match( + workflow, + /test "\$\{TRANSITION_PHASE\}" = cell-active; then[\s\S]*?recreate_fixed_one_instance/ + ) + assert.match(workflow, /--admission migration-only/) +}) diff --git a/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs b/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs new file mode 100644 index 00000000000..fece62f9ea2 --- /dev/null +++ b/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs @@ -0,0 +1,228 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import { readWorkflow, workflowFiles } from './cloud-sql-rollout-lock-census.mjs' +import { + RELAY_WORKFLOW_FILE_PREFIX, + relayWorkflowFile, + relayWorkflowPath +} from './relay-repository.mjs' + +const identity = readFileSync( + new URL('../../infra/terraform/relay-staging-deploy-iam.tf', import.meta.url), + 'utf8' +) +const shared = readFileSync( + new URL('../../infra/terraform/relay-shared.tf', import.meta.url), + 'utf8' +) +const variables = readFileSync( + new URL('../../infra/terraform/variables.tf', import.meta.url), + 'utf8' +) +const outputs = readFileSync(new URL('../../infra/terraform/outputs.tf', import.meta.url), 'utf8') +const stagingTfvars = readFileSync( + new URL('../../infra/terraform/environments/staging.tfvars', import.meta.url), + 'utf8' +) + +// The exact five staging Relay workflows the relay-owned deploy identity serves. +const DEPLOY_WORKFLOWS = [ + 'bootstrap-relay-staging-capacity.yml', + 'deploy-relay-staging-gce-candidate.yml', + 'deploy-relay-staging.yml', + 'operate-relay-asia-admission.yml', + 'power-relay-staging.yml' +] + +const GENERIC_STAGING_PAIR = + /vars\.STAGING_GCP_(?:WORKLOAD_IDENTITY_PROVIDER|DEPLOY_SERVICE_ACCOUNT)\b/ + +const workflowNames = workflowFiles +// DEPLOY_WORKFLOWS holds the names Terraform pins; the files on disk carry the copy's prefix. +const workflow = (name) => readWorkflow(relayWorkflowFile(name)) + +function block(type, name) { + const start = identity.indexOf(`resource "${type}" "${name}"`) + assert.notEqual(start, -1, `${type}.${name} is missing`) + const next = identity.indexOf('\nresource "', start + 1) + return identity.slice(start, next === -1 ? undefined : next) +} + +function declaredFamilies() { + return [...identity.matchAll(/^resource "([a-z0-9_]+)" "([a-z0-9_]+)"/gm)] + .map((match) => `${match[1]}.${match[2]}`) + .sort() +} + +function providerWorkflowFiles() { + const match = identity.match( + /github_staging_relay_deploy_workflow_files\s*=\s*\[([\s\S]*?)\n {2}\]/ + ) + assert.ok(match, 'github_staging_relay_deploy_workflow_files is missing') + return [...match[1].matchAll(/"([^"]+)"/g)].map((entry) => entry[1]) +} + +function variableDefault(name) { + const match = variables.match( + new RegExp(`variable "${name}" \\{[\\s\\S]*?default\\s*=\\s*"([^"]*)"`) + ) + assert.ok(match, `variable ${name} has no default`) + return match[1] +} + +test('no Relay workflow authenticates as the shared staging deploy identity', () => { + for (const name of workflowNames()) { + if (!name.includes('relay')) continue + assert.doesNotMatch(readWorkflow(name), GENERIC_STAGING_PAIR, name) + } +}) + +test('the five staging Relay workflows name the relay deploy pair', () => { + for (const name of DEPLOY_WORKFLOWS) { + const source = workflow(name) + assert.match(source, /vars\.STAGING_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER\b/, name) + assert.match(source, /vars\.STAGING_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT\b/, name) + } +}) + +// Why: the Asia workflow serves both environments from one job. Repointing its staging arm must +// not move production off the relay-owned shared account. +test('the Asia admission production arm keeps the production deploy pair', () => { + const source = workflow('operate-relay-asia-admission.yml') + assert.match(source, /vars\.PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER\b/) + assert.match(source, /vars\.PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT\b/) +}) + +test('the provider allowlists exactly those five workflow refs', () => { + const files = providerWorkflowFiles() + assert.deepEqual([...files].sort(), [...DEPLOY_WORKFLOWS].sort()) + for (const file of files) { + assert.match(file, /^[a-z0-9-]+\.yml$/) + } + // Each accepted repository turns that file list into its own exact refs. + assert.match( + identity, + /for workflow_file in local\.github_staging_relay_deploy_workflow_files : "assertion\.workflow_ref == '\$\{prefix\}\$\{workflow_file\}@refs\/heads\/main'"/ + ) + + const provider = block( + 'google_iam_workload_identity_pool_provider', + 'github_staging_relay_deploy' + ) + assert.match(provider, /workload_identity_pool_provider_id\s*=\s*"github-relay-deploy"/) + assert.match(provider, /concat\(local\.relay_github_leading_repository_claims/) + assert.match(provider, /assertion\.ref == 'refs\/heads\/main'/) + assert.match(provider, /assertion\.environment == 'staging'/) + assert.match(provider, /local\.relay_github_workflow_conditions\["github_staging_relay_deploy"\]/) + // A prefix match would turn the allowlist into a namespace grant with no Terraform diff. + assert.doesNotMatch(provider, /startsWith|endsWith/) +}) + +// Why: the documented attribute_condition limit is 4096 characters and the expression grows with +// every workflow added. Render it the way Terraform does and keep the headroom visible. +test('the rendered attribute condition stays inside the provider limit', () => { + const repository = `${variableDefault('github_owner')}/${variableDefault('github_repo')}` + const claims = [ + `assertion.repository == '${repository}'`, + `assertion.repository_id == '${variableDefault('github_repo_id')}'`, + `assertion.repository_owner_id == '${variableDefault('github_owner_id')}'` + ] + const workflowRefs = providerWorkflowFiles().map( + (file) => `${repository}/${relayWorkflowPath(file)}@refs/heads/main` + ) + const rendered = [ + ...claims, + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'staging'", + `(${workflowRefs.map((ref) => `assertion.workflow_ref == '${ref}'`).join(' || ')})` + ].join(' && ') + assert.ok(rendered.length < 4096, `rendered condition is ${rendered.length} characters`) + // 797 is the private repository's rendered length. This copy prefixes every workflow filename, + // which is the only difference, so the pin still moves the moment a workflow is added or dropped. + assert.equal(rendered.length, 797 + workflowRefs.length * RELAY_WORKFLOW_FILE_PREFIX.length) +}) + +// Why: the census is the point. A binding added here without a workflow step behind it, or one +// silently dropped, changes what the staging Relay credential can reach. +test('the staging deploy identity declares exactly its enumerated grants', () => { + assert.deepEqual(declaredFamilies(), [ + 'google_artifact_registry_repository_iam_member.github_staging_relay_deploy_artifact_reader', + 'google_cloud_run_v2_service_iam_member.github_staging_relay_deploy_auth_developer', + 'google_cloud_run_v2_service_iam_member.github_staging_relay_deploy_director_developer', + 'google_iam_workload_identity_pool_provider.github_staging_relay_deploy', + 'google_project_iam_member.github_staging_relay_deploy_compute_viewer', + 'google_service_account.github_staging_relay_deploy', + 'google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user', + 'google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user', + 'google_storage_bucket_iam_member.github_staging_relay_deploy_state', + 'google_storage_bucket_iam_member.github_staging_relay_deploy_state_list' + ]) + // Each grant carries a comment naming the workflow step that needs it; the account, its + // provider, and the pool binding are the identity itself and are covered by the file header. + const identityFamilies = new Set([ + 'google_service_account.github_staging_relay_deploy', + 'google_iam_workload_identity_pool_provider.github_staging_relay_deploy', + 'google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user' + ]) + for (const family of declaredFamilies()) { + if (identityFamilies.has(family)) continue + const [type, name] = family.split('.') + const preceding = identity.slice(0, identity.indexOf(`resource "${type}" "${name}"`)).trimEnd() + assert.match(preceding.slice(preceding.lastIndexOf('\n') + 1), /^#/, `${family} has no justifying comment`) + } + assert.match(identity, /var\.environment == "staging"/) + + const state = block('google_storage_bucket_iam_member', 'github_staging_relay_deploy_state') + assert.match(state, /roles\/storage\.objectViewer/) + assert.match(state, /objects\/terraform\/state\/default\.tfstate/) + assert.match(state, /objects\/terraform\/state\/default\.tflock/) + assert.doesNotMatch(state, /resource\.name\.startsWith/) + assert.doesNotMatch(state, /objectAdmin/) + + const director = block( + 'google_cloud_run_v2_service_iam_member', + 'github_staging_relay_deploy_director_developer' + ) + assert.match(director, /name\s*=\s*var\.relay_cloud_run_service_name/) + + // Project-wide run.developer or artifactregistry.writer would let the staging Relay credential + // deploy the API service or push images; the shared account holds both today. + const projectRoles = declaredFamilies() + .filter((family) => family.startsWith('google_project_iam_member.')) + .map((family) => block('google_project_iam_member', family.split('.')[1]).match(/role\s*=\s*"([^"]+)"/)[1]) + assert.deepEqual(projectRoles, ['roles/compute.viewer']) + assert.doesNotMatch(identity, /roles\/artifactregistry\.writer/) + assert.doesNotMatch(identity, /"roles\/(?:owner|editor|viewer)"/) +}) + +// Why: the auth-plane grants are guarded on a variable, so an unset tfvars entry would drop them +// silently and Power Relay Staging would fail only on the sleep path. +test('staging pins the shared auth service the power workflow scales', () => { + assert.match(stagingTfvars, /relay_staging_power_auth_service_name\s*=\s*"orca-cloud-auth-staging"/) + assert.match(variables, /variable "relay_staging_power_auth_service_name"/) + for (const name of [ + 'github_staging_relay_deploy_auth_developer', + 'github_staging_relay_deploy_auth_runtime_user' + ]) { + const type = name.endsWith('runtime_user') + ? 'google_service_account_iam_member' + : 'google_cloud_run_v2_service_iam_member' + assert.match(block(type, name), /var\.relay_staging_power_auth_service_name != ""/) + } +}) + +// Why: flipping this local is what moves the staging cells' startup metadata and the director's +// ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT onto the new account. Production must keep the shared one. +test('the deploy account email is environment-conditional', () => { + assert.match( + shared, + /relay_github_deploy_service_account_email = \(\s*var\.environment == "production"\s*\? "\$\{var\.name_prefix\}-gha-deploy@\$\{var\.project_id\}\.iam\.gserviceaccount\.com"\s*: "\$\{var\.name_prefix\}-gha-relay@\$\{var\.project_id\}\.iam\.gserviceaccount\.com"\s*\)/ + ) + assert.match(identity, /account_id\s*=\s*"\$\{var\.name_prefix\}-gha-relay"/) +}) + +test('the identity is exposed through its own outputs', () => { + assert.match(outputs, /output "github_staging_relay_deploy_workload_identity_provider"/) + assert.match(outputs, /output "github_staging_relay_deploy_service_account"/) +}) diff --git a/cloud/dev/scripts/render-workload-identity-conditions.mjs b/cloud/dev/scripts/render-workload-identity-conditions.mjs new file mode 100644 index 00000000000..9a7c2e5cc09 --- /dev/null +++ b/cloud/dev/scripts/render-workload-identity-conditions.mjs @@ -0,0 +1,535 @@ +// Renders every Workload Identity provider `attribute_condition` exactly as +// Terraform would, so contract tests can pin the resulting strings without a +// plan. Understands only the HCL subset those expressions use. +// +// Each root is loaded on its own: the relay and apps roots both declare a provider named +// `github` while the staging copy waits on its state surgery, and only separate scopes can +// show that the two render the same string. +// +// Only the relay root ships in this repository. The apps root is still declared so this stays a +// straight copy of the private original, and is skipped when its directory is absent. +import { existsSync } from 'node:fs' +import { readFile } from 'node:fs/promises' + +const TERRAFORM_ROOTS = { + relay: { + directory: 'infra/terraform', + sources: [ + 'infra/terraform/relay-shared.tf', + 'infra/terraform/relay-github-workflow-trust.tf', + 'infra/terraform/relay-github-actions.tf', + 'infra/terraform/relay-staging-deploy-iam.tf', + 'infra/terraform/relay-asia-topology-iam.tf', + 'infra/terraform/relay-asia-proof-iam.tf' + ] + }, + apps: { + directory: 'infra/terraform-apps', + sources: ['infra/terraform-apps/github-actions.tf'] + } +} + +export function hasTerraformRoot(root) { + const directory = TERRAFORM_ROOTS[root]?.directory + return directory !== undefined && existsSync(repoFile(directory)) +} + +export const TERRAFORM_ROOT_NAMES = Object.keys(TERRAFORM_ROOTS).filter(hasTerraformRoot) + +const PROVIDER_RESOURCE = 'google_iam_workload_identity_pool_provider' + +function repoFile(path) { + return new URL(`../../${path}`, import.meta.url) +} + +function skipTrivia(src, index) { + let i = index + for (;;) { + while (i < src.length && /\s/.test(src[i])) i += 1 + if (src[i] === '#') { + while (i < src.length && src[i] !== '\n') i += 1 + continue + } + return i + } +} + +// Returns the index just past the closing quote of the string starting at `i`. +function endOfString(src, i) { + let cursor = i + 1 + while (src[cursor] !== '"') { + if (src[cursor] === '\\') { + cursor += 2 + continue + } + if (src[cursor] === '$' && src[cursor + 1] === '{') { + cursor = endOfInterpolation(src, cursor + 2).next + continue + } + cursor += 1 + } + return cursor + 1 +} + +function endOfInterpolation(src, i) { + let depth = 1 + let cursor = i + while (depth > 0) { + const char = src[cursor] + if (char === undefined) throw new Error('unterminated interpolation') + if (char === '"') { + cursor = endOfString(src, cursor) + continue + } + if (char === '{') depth += 1 + else if (char === '}') { + depth -= 1 + if (depth === 0) break + } + cursor += 1 + } + return { text: src.slice(i, cursor), next: cursor + 1 } +} + +// Stands in for a loop variable when the collection is empty: the body still has to be parsed +// once to find where it ends, and any attribute of the probe is another probe. +const PROBE = new Proxy( + {}, + { + get: (target, key) => (key === Symbol.toPrimitive ? () => '' : PROBE) + } +) + +function readMember(value, key) { + if (value === PROBE) return PROBE + if (value === null || value === undefined) throw new Error(`cannot read ${String(key)} of ${value}`) + if (Array.isArray(value)) { + if (typeof key !== 'number') throw new Error(`list index must be a number, got ${String(key)}`) + if (!Number.isInteger(key) || key < 0 || key >= value.length) { + throw new Error(`list index ${key} is out of range`) + } + return value[key] + } + if (typeof value !== 'object') throw new Error(`cannot index ${typeof value}`) + if (!Object.hasOwn(value, key)) throw new Error(`unknown attribute ${String(key)}`) + return value[key] +} + +// [key, value] pairs the way HCL iterates: list index and element, or object key and value. +function collectionEntries(collection) { + if (Array.isArray(collection)) return collection.map((item, index) => [index, item]) + if (collection && typeof collection === 'object') return Object.entries(collection) + throw new Error(`cannot iterate ${typeof collection}`) +} + +class ExpressionParser { + constructor(source, scope) { + this.source = source + this.scope = scope + this.index = 0 + } + + parse() { + const value = this.parseTernary() + this.index = skipTrivia(this.source, this.index) + if (this.index !== this.source.length) { + throw new Error(`trailing expression text: ${this.source.slice(this.index)}`) + } + return value + } + + peek(token) { + this.index = skipTrivia(this.source, this.index) + return this.source.startsWith(token, this.index) + } + + eat(token) { + if (!this.peek(token)) return false + this.index += token.length + return true + } + + expect(token) { + if (!this.eat(token)) { + throw new Error(`expected ${token} at ${this.source.slice(this.index, this.index + 40)}`) + } + } + + parseTernary() { + const condition = this.parseOr() + if (!this.eat('?')) return condition + const consequent = this.parseTernary() + this.expect(':') + const alternate = this.parseTernary() + return condition ? consequent : alternate + } + + parseOr() { + let left = this.parseAnd() + while (this.eat('||')) left = Boolean(this.parseAnd()) || Boolean(left) + return left + } + + parseAnd() { + let left = this.parseEquality() + while (this.eat('&&')) left = Boolean(this.parseEquality()) && Boolean(left) + return left + } + + parseEquality() { + let left = this.parseUnary() + for (;;) { + if (this.eat('==')) left = left === this.parseUnary() + else if (this.eat('!=')) left = left !== this.parseUnary() + else return left + } + } + + parseUnary() { + return this.parsePostfix(this.parsePrimary()) + } + + parsePrimary() { + if (this.eat('(')) { + const value = this.parseTernary() + this.expect(')') + return value + } + if (this.peek('"')) return this.parseString() + if (this.peek('[')) return this.parseList() + if (this.peek('{')) return this.parseObject() + const number = /^[0-9]+/.exec(this.source.slice(this.index)) + if (number) { + this.index += number[0].length + return Number(number[0]) + } + return this.parseIdentifier() + } + + parsePostfix(value) { + let current = value + for (;;) { + if (this.eat('.')) { + current = readMember(current, this.readWord()) + continue + } + if (this.peek('[')) { + this.index += 1 + const key = this.parseTernary() + this.expect(']') + current = readMember(current, key) + continue + } + return current + } + } + + // `for a in x : body` / `for a, b in x : body`, shared by list and object comprehensions. + parseComprehension(readBody) { + const names = [this.readWord()] + if (this.eat(',')) names.push(this.readWord()) + this.expect('in') + const collection = this.parseUnary() + this.expect(':') + const bodyStart = skipTrivia(this.source, this.index) + const entries = collectionEntries(collection) + const bodyParser = ([key, item]) => { + const bindings = { ...this.scope.bindings } + if (names.length === 1) bindings[names[0]] = Array.isArray(collection) ? item : key + else { + bindings[names[0]] = key + bindings[names[1]] = item + } + const parser = new ExpressionParser(this.source, { ...this.scope, bindings }) + parser.index = bodyStart + return parser + } + // Parse once with a probe binding to find where the body ends, because an empty + // collection would never parse it. + const probe = bodyParser(entries[0] ?? [PROBE, PROBE]) + readBody(probe) + this.index = probe.index + return entries.map((entry) => readBody(bodyParser(entry))) + } + + parseObject() { + this.expect('{') + if (this.eat('for')) { + const pairs = this.parseComprehension((parser) => { + const key = parser.parseTernary() + parser.expect('=>') + return [key, parser.parseTernary()] + }) + this.expect('}') + return Object.fromEntries(pairs) + } + const object = {} + if (this.eat('}')) return object + for (;;) { + const key = this.peek('"') ? this.parseString() : this.readWord() + this.expect('=') + object[key] = this.parseTernary() + this.eat(',') + if (this.eat('}')) return object + } + } + + parseString() { + this.index = skipTrivia(this.source, this.index) + const src = this.source + let cursor = this.index + 1 + let rendered = '' + while (src[cursor] !== '"') { + if (src[cursor] === '\\') { + rendered += src[cursor + 1] + cursor += 2 + continue + } + if (src[cursor] === '$' && src[cursor + 1] === '{') { + const { text, next } = endOfInterpolation(src, cursor + 2) + rendered += String(evaluate(text, this.scope)) + cursor = next + continue + } + rendered += src[cursor] + cursor += 1 + } + this.index = cursor + 1 + return rendered + } + + parseList() { + this.expect('[') + if (this.eat('for')) { + const items = this.parseComprehension((parser) => parser.parseTernary()) + this.expect(']') + return items + } + const items = [] + if (this.eat(']')) return items + for (;;) { + items.push(this.parseTernary()) + if (this.eat(',')) { + if (this.eat(']')) return items + continue + } + this.expect(']') + return items + } + } + + readWord() { + this.index = skipTrivia(this.source, this.index) + const match = /^[A-Za-z_][A-Za-z0-9_]*/.exec(this.source.slice(this.index)) + if (!match) throw new Error(`expected identifier at ${this.source.slice(this.index, this.index + 40)}`) + this.index += match[0].length + return match[0] + } + + parseIdentifier() { + const word = this.readWord() + if (word === 'join') { + this.expect('(') + const separator = this.parseTernary() + this.expect(',') + const parts = this.parseTernary() + this.eat(',') + this.expect(')') + return parts.join(separator) + } + if (word === 'concat') { + this.expect('(') + const lists = [] + for (;;) { + lists.push(this.parseTernary()) + if (this.eat(',')) { + if (this.eat(')')) break + continue + } + this.expect(')') + break + } + return lists.flat() + } + if (word === 'length') { + this.expect('(') + const value = this.parseTernary() + this.eat(',') + this.expect(')') + return collectionEntries(value).length + } + if (word === 'local') { + this.expect('.') + return resolveLocal(this.readWord(), this.scope) + } + if (word === 'var') { + this.expect('.') + const name = this.readWord() + if (!(name in this.scope.variables)) throw new Error(`unknown variable ${name}`) + return this.scope.variables[name] + } + if (word in this.scope.bindings) return this.scope.bindings[word] + if (word === 'true') return true + if (word === 'false') return false + throw new Error(`unsupported identifier ${word}`) + } +} + +function evaluate(source, scope) { + return new ExpressionParser(source, scope).parse() +} + +function resolveLocal(name, scope) { + if (scope.resolved.has(name)) return scope.resolved.get(name) + if (!scope.locals.has(name)) throw new Error(`unknown local ${name}`) + if (scope.resolving.has(name)) throw new Error(`local cycle at ${name}`) + scope.resolving.add(name) + const value = evaluate(scope.locals.get(name), { ...scope, bindings: {} }) + scope.resolving.delete(name) + scope.resolved.set(name, value) + return value +} + +function collectLocals(source, locals) { + const blockPattern = /^locals \{$/gm + let match + while ((match = blockPattern.exec(source)) !== null) { + const end = source.indexOf('\n}\n', match.index) + const body = source.slice(match.index + match[0].length, end) + let name = null + let buffer = [] + const flush = () => { + if (name) locals.set(name, buffer.join('\n')) + } + for (const line of body.split('\n')) { + const assignment = /^ {2}([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$/.exec(line) + if (assignment) { + flush() + name = assignment[1] + buffer = [assignment[2]] + continue + } + if (name && line.trim() !== '' && !line.trim().startsWith('#')) buffer.push(line) + } + flush() + } +} + +function collectProviderFields(source, fields) { + const pattern = new RegExp(`resource "${PROVIDER_RESOURCE}" "([A-Za-z_0-9]+)" \\{`, 'g') + let match + while ((match = pattern.exec(source)) !== null) { + const end = source.indexOf('\n}\n', match.index) + const body = source.slice(match.index, end) + const conditionStart = body.indexOf(' attribute_condition = ') + const conditionEnd = body.indexOf('\n\n oidc {', conditionStart) + const countStart = body.indexOf(' count = ') + fields.set(match[1], { + count: body.slice(countStart + ' count = '.length, body.indexOf('\n', countStart)), + condition: body.slice(conditionStart + ' attribute_condition = '.length, conditionEnd) + }) + } +} + +// Values that are not a plain quoted string (a list of objects, say) are read with the +// expression parser; anything it cannot evaluate is left undefined, exactly as before. +function parseValueAt(source, index) { + const parser = new ExpressionParser(source, { + locals: new Map(), + variables: {}, + bindings: {}, + resolved: new Map(), + resolving: new Set() + }) + parser.index = index + return parser.parseTernary() +} + +function collectVariableDefaults(source, variables) { + const pattern = /variable "([A-Za-z_0-9]+)" \{([\s\S]*?)\n\}/g + let match + while ((match = pattern.exec(source)) !== null) { + const body = match[2] + const fallback = /\n\s*default\s*=\s*"([^"]*)"/.exec(body) + if (fallback) { + variables[match[1]] = fallback[1] + continue + } + const assignment = /\n\s*default\s*=\s*/.exec(body) + if (!assignment) continue + const start = match.index + match[0].indexOf(body) + assignment.index + assignment[0].length + try { + variables[match[1]] = parseValueAt(source, start) + } catch { + // A default this evaluator does not understand is not one any condition reads. + } + } +} + +function collectTfvars(source, variables) { + let offset = 0 + for (const line of source.split('\n')) { + const quoted = /^([A-Za-z_][A-Za-z0-9_]*)\s*=\s*"([^"]*)"\s*$/.exec(line) + if (quoted) { + variables[quoted[1]] = quoted[2] + offset += line.length + 1 + continue + } + const structured = /^([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(?=[[{])/.exec(line) + if (structured) { + try { + variables[structured[1]] = parseValueAt(source, offset + structured[0].length) + } catch { + // Same as above: unreadable here means unread by every condition. + } + } + offset += line.length + 1 + } +} + +async function loadScope(root, environment) { + const { directory, sources } = TERRAFORM_ROOTS[root] ?? {} + if (!sources) throw new Error(`unknown terraform root ${root}`) + const locals = new Map() + const providers = new Map() + for (const path of sources) { + const source = await readFile(repoFile(path), 'utf8') + collectLocals(source, locals) + collectProviderFields(source, providers) + } + const variables = {} + collectVariableDefaults(await readFile(repoFile(`${directory}/variables.tf`), 'utf8'), variables) + collectTfvars( + await readFile(repoFile(`${directory}/environments/${environment}.tfvars`), 'utf8'), + variables + ) + return { + providers, + scope: { locals, variables, bindings: {}, resolved: new Map(), resolving: new Set() } + } +} + +// Rendered `attribute_condition` per provider that the given root creates in the environment. +export async function renderRootAttributeConditions(root, environment) { + const { providers, scope } = await loadScope(root, environment) + const rendered = {} + for (const [name, fields] of providers) { + if (evaluate(fields.count, scope) === 0) continue + rendered[name] = evaluate(fields.condition, scope) + } + return rendered +} + +// Every root's rendered conditions, keyed by root and then by provider. +export async function renderAttributeConditions(environment) { + const rendered = {} + for (const root of TERRAFORM_ROOT_NAMES) { + rendered[root] = await renderRootAttributeConditions(root, environment) + } + return rendered +} + +export async function readTerraformLocal(name, environment, root = 'relay') { + const { scope } = await loadScope(root, environment) + return resolveLocal(name, scope) +} diff --git a/cloud/dev/scripts/run-relay-load-model.mjs b/cloud/dev/scripts/run-relay-load-model.mjs new file mode 100644 index 00000000000..6915b8eb8c2 --- /dev/null +++ b/cloud/dev/scripts/run-relay-load-model.mjs @@ -0,0 +1,8 @@ +import { assertSpreadModel, modeledRelayLoad } from './relay-load-model.mjs' + +const counts = process.argv.slice(2).length > 0 ? process.argv.slice(2).map(Number) : [4_000, 10_000] +for (const count of counts) { + const model = modeledRelayLoad(count) + assertSpreadModel(model) + console.log(JSON.stringify(model)) +} diff --git a/cloud/dev/scripts/run-relay-recovery-wave-gate.mjs b/cloud/dev/scripts/run-relay-recovery-wave-gate.mjs new file mode 100644 index 00000000000..fcc58d3474c --- /dev/null +++ b/cloud/dev/scripts/run-relay-recovery-wave-gate.mjs @@ -0,0 +1,19 @@ +import { readFileSync, statSync } from 'node:fs' +import { evaluateRecoveryWaveReport } from './relay-recovery-wave-gate.mjs' + +const args = process.argv.slice(2) +if (args.length !== 2 || args[0] !== '--report') { + process.stderr.write('usage: pnpm load:relay:recovery-gate -- --report \n') + process.exitCode = 1 +} else { + try { + if (statSync(args[1]).size > 1024 * 1024) throw new Error('report exceeds 1 MiB') + const report = JSON.parse(readFileSync(args[1], 'utf8')) + const result = evaluateRecoveryWaveReport(report) + process.stdout.write(`${JSON.stringify(result)}\n`) + if (result.status !== 'PASS') process.exitCode = 1 + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + } +} diff --git a/cloud/dev/scripts/sanitize-relay-asia-admission-result.mjs b/cloud/dev/scripts/sanitize-relay-asia-admission-result.mjs new file mode 100644 index 00000000000..1ac07841514 --- /dev/null +++ b/cloud/dev/scripts/sanitize-relay-asia-admission-result.mjs @@ -0,0 +1,98 @@ +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +const MODES = new Set([ + 'inspect', 'initialize', 'verify', 'registered', 'register', + 'promote', 'recover-promotion', 'rollback' +]) +const STATES = new Set(['absent', 'existing-only', 'migration-only', 'general']) + +function validCellId(value) { + return typeof value === 'string' && value.length >= 1 && value.length <= 128 +} + +function cellIds(value, label) { + if ( + !Array.isArray(value) || value.length > 256 || + value.some((cellId) => !validCellId(cellId)) + ) { + throw new Error(`${label} is invalid`) + } + return [...value] +} + +function membership(value) { + if (value === undefined || value === null) return null + if (typeof value !== 'object' || Array.isArray(value)) { + throw new Error('membership is invalid') + } + return { + existingOnly: cellIds(value.existingOnly, 'existing-only membership'), + migrationOnly: cellIds(value.migrationOnly, 'migration-only membership'), + general: cellIds(value.general, 'general membership') + } +} + +function states(value) { + if (typeof value !== 'object' || Array.isArray(value)) throw new Error('states are invalid') + const entries = Object.entries(value) + if ( + entries.length === 0 || entries.length > 256 || + entries.some(([cellId, state]) => !validCellId(cellId) || !STATES.has(state)) + ) { + throw new Error('states are invalid') + } + return Object.fromEntries(entries) +} + +function optionalBoolean(value, label) { + if (value === undefined || value === null) return null + if (typeof value !== 'boolean') throw new Error(`${label} is invalid`) + return value +} + +export function sanitizeRelayAsiaAdmissionResult(input) { + if (!input || typeof input !== 'object' || Array.isArray(input)) { + throw new Error('admission result must be an object') + } + if (!MODES.has(input.mode)) throw new Error('mode is invalid') + if (!Number.isSafeInteger(input.generation) || input.generation < 0) { + throw new Error('generation is invalid') + } + if ( + input.membershipSha256 !== undefined && input.membershipSha256 !== null && + !/^[a-f0-9]{64}$/.test(input.membershipSha256) + ) throw new Error('membership SHA-256 is invalid') + const sanitizedMembership = membership(input.membership) + if ( + input.mode === 'inspect' && + (sanitizedMembership === null || !/^[a-f0-9]{64}$/.test(input.membershipSha256 ?? '')) + ) throw new Error('inspect membership evidence is incomplete') + const recovered = optionalBoolean(input.recovered, 'recovered') + const promoted = optionalBoolean(input.promoted, 'promoted') + if (input.mode === 'recover-promotion' && promoted === null) { + throw new Error('promotion recovery evidence is incomplete') + } + return { + v: 1, + mode: input.mode, + generation: input.generation, + states: states(input.states), + membership: sanitizedMembership, + membershipSha256: input.membershipSha256 ?? null, + recovered, + promoted + } +} + +function main() { + try { + const input = JSON.parse(readFileSync(0, 'utf8')) + process.stdout.write(`${JSON.stringify(sanitizeRelayAsiaAdmissionResult(input))}\n`) + } catch { + console.error('invalid Relay Asia admission result') + process.exitCode = 1 + } +} + +if (process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1]) main() diff --git a/cloud/dev/scripts/sanitize-relay-asia-admission-result.test.mjs b/cloud/dev/scripts/sanitize-relay-asia-admission-result.test.mjs new file mode 100644 index 00000000000..2f1cf64a8cb --- /dev/null +++ b/cloud/dev/scripts/sanitize-relay-asia-admission-result.test.mjs @@ -0,0 +1,120 @@ +import assert from 'node:assert/strict' +import { spawnSync } from 'node:child_process' +import { fileURLToPath } from 'node:url' +import { test } from 'node:test' +import { sanitizeRelayAsiaAdmissionResult } from './sanitize-relay-asia-admission-result.mjs' + +const script = fileURLToPath(new URL('./sanitize-relay-asia-admission-result.mjs', import.meta.url)) +const expectedKeys = [ + 'v', 'mode', 'generation', 'states', 'membership', 'membershipSha256', 'recovered', 'promoted' +] + +test('preserves explicit false and emits only the admission evidence allowlist', () => { + const result = sanitizeRelayAsiaAdmissionResult({ + mode: 'verify', generation: 7, states: { 'staging-gce-c4': 'migration-only' }, + membership: { + existingOnly: [], migrationOnly: ['staging-gce-c4'], general: [], token: 'nested-secret' + }, + membershipSha256: 'a'.repeat(64), + recovered: false, promoted: false, + token: 'secret', credential: 'secret', userId: 'user', relayHostId: 'host', arbitrary: true + }) + + assert.deepEqual(Object.keys(result), expectedKeys) + assert.equal(result.recovered, false) + assert.equal(result.promoted, false) + assert.equal(JSON.stringify(result).includes('secret'), false) + assert.deepEqual(Object.keys(result.membership), ['existingOnly', 'migrationOnly', 'general']) + assert.equal('token' in result, false) + assert.equal('credential' in result, false) + assert.equal('userId' in result, false) + assert.equal('relayHostId' in result, false) + assert.equal('arbitrary' in result, false) +}) + +test('uses null for missing optional admission evidence', () => { + assert.deepEqual(sanitizeRelayAsiaAdmissionResult({ + mode: 'verify', generation: 0, states: { 'staging-gce-c4': 'absent' } + }), { + v: 1, + mode: 'verify', + generation: 0, + states: { 'staging-gce-c4': 'absent' }, + membership: null, + membershipSha256: null, + recovered: null, + promoted: null + }) +}) + +test('preserves valid historical cell IDs accepted by the director schema', () => { + const legacyCellId = 'legacy staging cell' + const result = sanitizeRelayAsiaAdmissionResult({ + mode: 'inspect', + generation: 0, + states: { [legacyCellId]: 'general' }, + membership: { existingOnly: [], migrationOnly: [], general: [legacyCellId] }, + membershipSha256: 'a'.repeat(64) + }) + + assert.deepEqual(result.states, { [legacyCellId]: 'general' }) + assert.deepEqual(result.membership.general, [legacyCellId]) +}) + +test('sanitizes stdin through the command-line entry point', () => { + const run = spawnSync(process.execPath, [script], { + input: JSON.stringify({ + mode: 'verify', generation: 0, states: { 'staging-gce-c4': 'absent' }, + recovered: false, token: 'secret' + }), + encoding: 'utf8' + }) + + assert.equal(run.status, 0, run.stderr) + const result = JSON.parse(run.stdout) + assert.deepEqual(Object.keys(result), expectedKeys) + assert.equal(result.recovered, false) + assert.equal(JSON.stringify(result).includes('secret'), false) +}) + +test('requires the evidence needed by every sequenced operation mode', () => { + const states = { 'staging-gce-c4': 'migration-only' } + const membership = { + existingOnly: [], migrationOnly: ['staging-gce-c4'], general: [] + } + const validByMode = { + inspect: { membership, membershipSha256: 'a'.repeat(64) }, + initialize: {}, + verify: {}, + registered: {}, + register: {}, + promote: {}, + 'recover-promotion': { promoted: false }, + rollback: {} + } + + for (const [mode, evidence] of Object.entries(validByMode)) { + assert.doesNotThrow(() => sanitizeRelayAsiaAdmissionResult({ + mode, generation: 1, states, ...evidence + })) + assert.throws(() => sanitizeRelayAsiaAdmissionResult({ mode, generation: 1, ...evidence })) + } + assert.throws(() => sanitizeRelayAsiaAdmissionResult({ + mode: 'inspect', generation: 1, states, membership + })) + assert.throws(() => sanitizeRelayAsiaAdmissionResult({ + mode: 'inspect', generation: 1, states, membershipSha256: 'a'.repeat(64) + })) + assert.throws(() => sanitizeRelayAsiaAdmissionResult({ + mode: 'recover-promotion', generation: 1, states + })) +}) + +test('rejects invalid stdin without echoing it', () => { + const run = spawnSync(process.execPath, [script], { input: '{secret', encoding: 'utf8' }) + + assert.equal(run.status, 1) + assert.equal(run.stdout, '') + assert.equal(run.stderr, 'invalid Relay Asia admission result\n') + assert.equal(run.stderr.includes('{secret'), false) +}) diff --git a/cloud/dev/scripts/smoke-relay.mjs b/cloud/dev/scripts/smoke-relay.mjs new file mode 100644 index 00000000000..71689680a94 --- /dev/null +++ b/cloud/dev/scripts/smoke-relay.mjs @@ -0,0 +1,211 @@ +import { + createHash, + createHmac, + createPrivateKey, + createPublicKey, + randomUUID +} from 'node:crypto' +import { readFileSync } from 'node:fs' +import { createRequire } from 'node:module' + +const requireFromRelay = createRequire(new URL('../../apps/relay/package.json', import.meta.url)) + +const relayUrl = process.argv[2]?.replace(/\/$/, '') +if (!relayUrl) throw new Error('usage: smoke-relay.mjs ') + +const health = await fetch(`${relayUrl}/health`) +if (!health.ok || (await health.json()).ok !== true) { + throw new Error(`relay health failed: ${health.status}`) +} + +const accessToken = process.env.ORCA_RELAY_SMOKE_ACCESS_TOKEN +const authUrl = process.env.ORCA_RELAY_SMOKE_AUTH_URL?.replace(/\/$/, '') +const signingKeyFile = process.env.ORCA_RELAY_SMOKE_SIGNING_KEY_FILE +if (!authUrl || (!accessToken && !signingKeyFile)) { + console.log('relay health smoke passed; provide auth URL plus an access token or operator signing-key file for a splice round-trip') + process.exit(0) +} + +const nacl = requireFromRelay('tweetnacl') +const WebSocket = requireFromRelay('ws') +const { buildHostProofMacInput, HOST_CHALLENGE_PLAINTEXT_DOMAIN } = await import( + requireFromRelay.resolve('@orca-cloud/relay-contract') +) + +function nextMessage(socket) { + return new Promise((resolve, reject) => { + const onMessage = (data) => { + cleanup() + try { + resolve(JSON.parse(data.toString())) + } catch (error) { + reject(error) + } + } + const onError = (error) => { + cleanup() + reject(error) + } + const onClose = (code, reason) => { + cleanup() + reject(new Error(`socket closed before message: ${code} ${reason.toString()}`)) + } + const cleanup = () => { + socket.off('message', onMessage) + socket.off('error', onError) + socket.off('close', onClose) + } + socket.once('message', onMessage) + socket.once('error', onError) + socket.once('close', onClose) + }) +} + +function opened(socket) { + return new Promise((resolve, reject) => { + socket.once('open', resolve) + socket.once('error', reject) + }) +} + +const hostKeys = nacl.box.keyPair() +const relayHostId = createHash('sha256') + .update(hostKeys.publicKey) + .digest('base64url') + .slice(0, 16) +let relayToken +if (accessToken) { + const tokenResponse = await fetch(`${authUrl}/v1/desktop/auth/relay-token`, { + method: 'POST', + headers: { + authorization: `Bearer ${accessToken}`, + 'content-type': 'application/json' + }, + body: JSON.stringify({ + relayHostId, + hostPublicKeyB64: Buffer.from(hostKeys.publicKey).toString('base64') + }) + }) + if (!tokenResponse.ok) throw new Error(`relay-token exchange failed: ${tokenResponse.status}`) + ;({ relayToken } = await tokenResponse.json()) +} else { + // This operator-only path proves the deployed data plane before desktop UI + // exists; possession of the auth signing key remains the security boundary. + const { SignJWT } = await import(requireFromRelay.resolve('jose')) + const privateKey = createPrivateKey(readFileSync(signingKeyFile, 'utf8')) + const keyId = createHash('sha256') + .update(createPublicKey(privateKey).export({ type: 'spki', format: 'der' })) + .digest('base64url') + .slice(0, 16) + relayToken = await new SignJWT({ + prof: 'staging-smoke-profile', + org: 'staging-smoke-org', + purpose: 'host-control', + relayHostId + }) + .setProtectedHeader({ alg: 'ES256', kid: keyId }) + .setIssuer(authUrl) + .setAudience('orca-relay') + .setSubject('staging-smoke-user') + .setIssuedAt() + .setExpirationTime('5m') + .sign(privateKey) +} +if (!relayToken) throw new Error('relay token was not produced') + +const wsOrigin = relayUrl.replace(/^http/, 'ws') +const control = new WebSocket(`${wsOrigin}/v1/host/control`, { + headers: { authorization: `Bearer ${relayToken}` }, + perMessageDeflate: false +}) +await opened(control) +control.send( + JSON.stringify({ + type: 'host-hello', + v: 1, + relayHostId, + assignmentEpoch: 1, + hostPublicKeyB64: Buffer.from(hostKeys.publicKey).toString('base64'), + appVersion: 'relay-smoke' + }) +) +const challenge = await nextMessage(control) +const plaintext = nacl.box.open( + Buffer.from(challenge.ciphertextB64, 'base64'), + Buffer.from(challenge.nonceB64, 'base64'), + Buffer.from(challenge.relayEphemeralPublicKeyB64, 'base64'), + hostKeys.secretKey +) +if (!plaintext) throw new Error('host proof challenge did not decrypt') +const domain = new TextEncoder().encode(`${HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`) +const transcriptLength = new DataView( + plaintext.buffer, + plaintext.byteOffset + domain.length, + 4 +).getUint32(0, false) +const transcriptStart = domain.length + 4 +const transcript = plaintext.slice(transcriptStart, transcriptStart + transcriptLength) +const secret = plaintext.slice(transcriptStart + transcriptLength) +const proofB64 = createHmac('sha256', secret) + .update(buildHostProofMacInput(transcript)) + .digest('base64') +control.send(JSON.stringify({ + type: 'host-challenge-ack', + challengeId: challenge.challengeId, + proofB64 +})) +const hostAck = await nextMessage(control) + +const relayDeviceId = `smoke-${randomUUID()}` +const invitePromise = nextMessage(control) +control.send(JSON.stringify({ type: 'invite-create', reqId: randomUUID(), relayDeviceId })) +const invite = await invitePromise + +const phone = new WebSocket(`${wsOrigin}/v1/connect/${relayHostId}`, { perMessageDeflate: false }) +await opened(phone) +const connectionPromise = nextMessage(control) +phone.send(JSON.stringify({ + type: 'relay-auth', + v: 1, + mode: 'connect', + credential: invite.inviteToken +})) +const connection = await connectionPromise +const data = new WebSocket(`${wsOrigin}/v1/host/data/${connection.connId}`, { + perMessageDeflate: false +}) +await opened(data) +const phoneHelloPromise = nextMessage(phone) +data.send(JSON.stringify({ + type: 'host-data-auth', + v: 1, + connTicket: connection.connTicket, + generation: hostAck.generation +})) +const phoneHello = await phoneHelloPromise +if (phoneHello.ok !== true) throw new Error('relay rejected smoke splice') + +const phoneEcho = new Promise((resolve, reject) => { + phone.once('message', (bytes, binary) => resolve({ bytes: Buffer.from(bytes), binary })) + phone.once('error', reject) +}) +data.send(Buffer.from([0x4f, 0x52, 0x43, 0x41])) +const echoed = await phoneEcho +if (!echoed.binary || !echoed.bytes.equals(Buffer.from([0x4f, 0x52, 0x43, 0x41]))) { + throw new Error('relay splice changed binary payload or opcode') +} + +const dataEcho = new Promise((resolve, reject) => { + data.once('message', (bytes, binary) => resolve({ bytes: Buffer.from(bytes), binary })) + data.once('error', reject) +}) +phone.send('orca-relay-smoke') +const returned = await dataEcho +if (returned.binary || returned.bytes.toString() !== 'orca-relay-smoke') { + throw new Error('relay splice changed text payload or opcode') +} + +phone.close() +data.close() +control.close() +console.log('relay authenticated splice smoke passed') diff --git a/cloud/dev/scripts/staging-relay-apply-guard.mjs b/cloud/dev/scripts/staging-relay-apply-guard.mjs new file mode 100644 index 00000000000..7595a585769 --- /dev/null +++ b/cloud/dev/scripts/staging-relay-apply-guard.mjs @@ -0,0 +1,51 @@ +import { execFileSync } from 'node:child_process' + +const PROJECT = 'onorca-cloud-staging' +const SQL_INSTANCE = 'orca-cloud-staging-auth-db' +const MIG_PREFIX = 'orca-cloud-staging-relay-gce-' + +function defaultGcloud(args) { + return execFileSync('gcloud', args, { + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'pipe'] + }).trim() +} + +export function stagingRelayPowerState(gcloud = defaultGcloud) { + const sqlActivationPolicy = gcloud([ + 'sql', + 'instances', + 'describe', + SQL_INSTANCE, + '--project', + PROJECT, + '--format=value(settings.activationPolicy)' + ]) + const groups = JSON.parse( + gcloud([ + 'compute', + 'instance-groups', + 'managed', + 'list', + '--project', + PROJECT, + `--filter=name~'^${MIG_PREFIX}'`, + '--format=json(name,targetSize)' + ]) + ) + return { sqlActivationPolicy, groups } +} + +export function assertStagingRelayAwake(gcloud = defaultGcloud) { + const state = stagingRelayPowerState(gcloud) + const sleepingGroups = state.groups.filter((group) => Number(group.targetSize) !== 1) + if ( + state.sqlActivationPolicy !== 'ALWAYS' || + state.groups.length < 2 || + sleepingGroups.length > 0 + ) { + throw new Error( + 'Staging Relay is asleep or partially awake. Run the Power Relay Staging wake workflow before any staging Terraform apply.' + ) + } +} diff --git a/cloud/dev/scripts/staging-relay-apply-guard.test.mjs b/cloud/dev/scripts/staging-relay-apply-guard.test.mjs new file mode 100644 index 00000000000..8035ffaf3a4 --- /dev/null +++ b/cloud/dev/scripts/staging-relay-apply-guard.test.mjs @@ -0,0 +1,30 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + assertStagingRelayAwake, + stagingRelayPowerState +} from './staging-relay-apply-guard.mjs' + +function gcloud(policy, targetSizes) { + return (args) => + args[0] === 'sql' + ? policy + : JSON.stringify( + targetSizes.map((targetSize, index) => ({ + name: `orca-cloud-staging-relay-gce-c${index + 1}`, + targetSize + })) + ) +} + +test('reads and accepts a fully awake staging topology', () => { + const command = gcloud('ALWAYS', [1, 1, 1]) + assert.equal(stagingRelayPowerState(command).groups.length, 3) + assert.doesNotThrow(() => assertStagingRelayAwake(command)) +}) + +test('refuses Terraform apply while SQL or any staging cell is asleep', () => { + assert.throws(() => assertStagingRelayAwake(gcloud('NEVER', [0, 0, 0])), /wake workflow/) + assert.throws(() => assertStagingRelayAwake(gcloud('ALWAYS', [1, 0, 0])), /partially awake/) + assert.throws(() => assertStagingRelayAwake(gcloud('ALWAYS', [])), /partially awake/) +}) diff --git a/cloud/dev/scripts/terraform-root-partition.mjs b/cloud/dev/scripts/terraform-root-partition.mjs new file mode 100644 index 00000000000..07ada7f624c --- /dev/null +++ b/cloud/dev/scripts/terraform-root-partition.mjs @@ -0,0 +1,131 @@ +#!/usr/bin/env node +import { readdirSync, readFileSync, writeFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +// Why: the relay Terraform root is being carved into foundation / relay / apps. Until every +// family is assigned to exactly one root per environment, a state move can silently orphan or +// double-manage a resource. This file is the single authority; the test pins it to the .tf files. + +export const ROOTS = ['foundation', 'apps', 'relay'] +export const ENVIRONMENTS = ['production', 'staging'] + +// Each root is its own Terraform directory; a family is declared in exactly the roots that own +// it somewhere, so the directory is the authority the fixture is checked against. Only the relay +// directory ships here, so only relay declarations can be read back; the partition still names +// all three roots because it is the authority for the whole carve. +export const ROOT_DIRECTORIES = { + relay: 'infra/terraform/' +} + +export const DECLARING_ROOTS = Object.keys(ROOT_DIRECTORIES) + +export const fixturePath = fileURLToPath( + new URL('../fixtures/terraform-root-partition/families.json', import.meta.url) +) + +function rootDirectory(root) { + const relative = ROOT_DIRECTORIES[root] + if (!relative) throw new Error(`unknown root ${root}`) + return fileURLToPath(new URL(`../../${relative}`, import.meta.url)) +} + +export function declaredFamilies(root = 'relay') { + const directory = rootDirectory(root) + const families = new Map() + for (const entry of readdirSync(directory).filter((name) => name.endsWith('.tf')).sort()) { + const text = readFileSync(`${directory}${entry}`, 'utf8') + for (const match of text.matchAll(/^resource "([a-z0-9_]+)" "([a-z0-9_]+)"/gm)) { + const address = `${match[1]}.${match[2]}` + if (families.has(address)) throw new Error(`duplicate declaration ${address}`) + families.set(address, entry) + } + } + return families +} + +// A family may be declared in two roots at once (the environment-conditional ten), so ownership +// is per (root, environment) while declaration is per root. +export function declaredRootFamilies(root) { + return new Set(declaredFamilies(root).keys()) +} + +export function ownedRootFamilies(partition, root) { + const families = new Set(partition[root]) + for (const [family, owners] of Object.entries(partition.env_conditional)) { + if (Object.values(owners).includes(root)) families.add(family) + } + return families +} + +export function readPartition(path = fixturePath) { + return JSON.parse(readFileSync(path, 'utf8')) +} + +// Family address for a state entry: strips [index] / ["key"] and ignores data sources. +export function familyOf(stateAddress) { + return stateAddress.replace(/\[.*$/, '') +} + +export function rootFor(partition, family, environment) { + if (!ENVIRONMENTS.includes(environment)) throw new Error(`unknown environment ${environment}`) + const conditional = partition.env_conditional[family] + if (conditional) return conditional[environment] + for (const root of ROOTS) if (partition[root].includes(family)) return root + return undefined +} + +export function expectedRootFamilies(partition, root, environment) { + const families = new Set(partition[root]) + for (const [family, owners] of Object.entries(partition.env_conditional)) { + if (owners[environment] === root) families.add(family) + } + return families +} + +// Compares `terraform state list` output for one root against the partition. +export function auditStateList(partition, root, environment, stateList) { + const expected = expectedRootFamilies(partition, root, environment) + const orphans = new Set(partition.state_orphans[environment] ?? []) + const unexpected = [] + const seen = new Set() + for (const line of stateList.split('\n').map((entry) => entry.trim()).filter(Boolean)) { + if (orphans.has(line)) continue + if (line.startsWith('data.')) continue + const family = familyOf(line) + seen.add(family) + if (!expected.has(family)) unexpected.push(line) + } + return { unexpected, seen: [...seen].sort() } +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const [command, root, environment, stateFile] = process.argv.slice(2) + const partition = readPartition() + if (command === 'audit') { + const { unexpected } = auditStateList(partition, root, environment, readFileSync(stateFile, 'utf8')) + if (unexpected.length > 0) { + process.stderr.write(`entries in ${root}/${environment} state outside its partition:\n`) + for (const line of unexpected) process.stderr.write(` ${line}\n`) + process.exitCode = 1 + } else { + process.stdout.write(`${root}/${environment}: state matches partition\n`) + } + } else if (command === 'list') { + for (const family of [...expectedRootFamilies(partition, root, environment)].sort()) { + process.stdout.write(`${family}\n`) + } + } else if (command === 'write-families') { + // Refreshes only the declared-family lists; ownership edits stay manual. + for (const name of DECLARING_ROOTS) { + for (const family of declaredRootFamilies(name)) { + if (rootFor(partition, family, 'production') === undefined) { + throw new Error(`unassigned family ${family}; add it to the fixture first`) + } + } + } + writeFileSync(fixturePath, `${JSON.stringify(partition, null, 2)}\n`) + } else { + process.stderr.write('usage: terraform-root-partition.mjs audit|list [state-list-file]\n') + process.exitCode = 2 + } +} diff --git a/cloud/dev/scripts/terraform-root-partition.test.mjs b/cloud/dev/scripts/terraform-root-partition.test.mjs new file mode 100644 index 00000000000..b92cb92366d --- /dev/null +++ b/cloud/dev/scripts/terraform-root-partition.test.mjs @@ -0,0 +1,117 @@ +import assert from 'node:assert/strict' +import { existsSync, readdirSync, readFileSync } from 'node:fs' +import { test } from 'node:test' +import { + DECLARING_ROOTS, + ENVIRONMENTS, + ROOTS, + auditStateList, + declaredFamilies, + declaredRootFamilies, + expectedRootFamilies, + ownedRootFamilies, + readPartition, + rootFor +} from './terraform-root-partition.mjs' + +const partition = readPartition() +const declared = new Map(DECLARING_ROOTS.flatMap((root) => [...declaredFamilies(root)].map(([family, file]) => [family, `${root}/${file}`]))) + +test('every declared resource family is assigned to exactly one root per environment', () => { + for (const family of declared.keys()) { + for (const environment of ENVIRONMENTS) { + const owners = ROOTS.filter((root) => expectedRootFamilies(partition, root, environment).has(family)) + assert.deepEqual(owners, [rootFor(partition, family, environment)], `${family} in ${environment}`) + } + } +}) + +// Only the relay directory ships here, so only the families the partition assigns to a declaring +// root can be checked back against a .tf file. The full listing is still checked for duplicates. +test('the partition names no family that is not declared', () => { + const listed = [ + ...ROOTS.flatMap((root) => partition[root]), + ...Object.keys(partition.env_conditional) + ] + for (const root of DECLARING_ROOTS) { + for (const family of ownedRootFamilies(partition, root)) { + assert.ok(declared.has(family), `${family} is not declared`) + } + } + assert.equal(new Set(listed).size, listed.length, 'a family is listed twice') +}) + +test('environment-conditional families are owned by different roots per environment', () => { + for (const [family, owners] of Object.entries(partition.env_conditional)) { + assert.deepEqual(Object.keys(owners).sort(), [...ENVIRONMENTS].sort(), family) + assert.notEqual(owners.production, owners.staging, `${family} is not really conditional`) + } +}) + +// Why: this is the census that survives the carve. Filename prefixes stopped meaning anything +// once each root became its own directory, so ownership is checked against the directory that +// declares the family. A root declares exactly what it owns in at least one environment; the +// environment-conditional ten are therefore declared twice, once per complementary count. +test('each root declares exactly the families it owns in some environment', () => { + for (const root of DECLARING_ROOTS) { + assert.deepEqual( + [...declaredRootFamilies(root)].sort(), + [...ownedRootFamilies(partition, root)].sort(), + root + ) + } +}) + +// Why: the removed blocks were a guard for the config-first window between the carve and the two +// state surgeries. Both are done; a removed block that resurfaces would silently turn a stray apply +// from "destroy" into "forget" and hide a real ownership mistake. +test('the carved families are gone from the relay root and nothing is guarded by a removed block', () => { + const relay = declaredRootFamilies('relay') + for (const family of [...partition.foundation, ...partition.apps]) { + assert.ok(!relay.has(family), `${family} is still declared in the relay root`) + } + assert.ok(!existsSync(new URL('../../infra/terraform/relay-root-carve-removed.tf', import.meta.url))) + for (const file of readdirSync(new URL('../../infra/terraform/', import.meta.url))) { + if (!file.endsWith('.tf')) continue + const source = readFileSync(new URL(`../../infra/terraform/${file}`, import.meta.url), 'utf8') + assert.doesNotMatch(source, /^removed \{/m, `${file} declares a removed block`) + } +}) + +// Why: every binding on the shared deploy account follows that account (relay in production, +// apps in staging). Letting one drift back into foundation recreates the dependency cycle the +// split exists to remove: foundation would need the account while relay needs the pool. +test('foundation owns only what both other roots must be able to bootstrap against', () => { + assert.deepEqual(partition.foundation, [ + 'google_artifact_registry_repository.api', + 'google_iam_workload_identity_pool.github', + 'google_project_service.required', + 'google_project_service.sqladmin', + 'google_service_account.runtime', + 'google_sql_database_instance.auth', + 'google_storage_bucket_iam_member.cloud_sql_rollout_lease', + 'google_storage_bucket_iam_member.cloud_sql_rollout_lease_bucket_reader' + ]) +}) + +// Why: the two staging orphans were cleared by the runbook; the allowance is empty so a stray +// entry is reported instead of silently tolerated again. +test('audit reports entries outside the partition and no longer tolerates the staging orphans', () => { + const stateList = [ + 'google_project_service.required["run.googleapis.com"]', + 'google_secret_manager_secret_iam_member.runtime_artifact_write_secret_accessor[0]', + 'data.google_compute_image.relay_gce_cos[0]', + 'google_cloud_run_v2_service.relay' + ].join('\n') + assert.deepEqual(partition.state_orphans, { staging: [], production: [] }) + const foundation = auditStateList(partition, 'foundation', 'staging', stateList) + assert.deepEqual(foundation.unexpected, [ + 'google_secret_manager_secret_iam_member.runtime_artifact_write_secret_accessor[0]', + 'google_cloud_run_v2_service.relay' + ]) + const relay = auditStateList(partition, 'relay', 'staging', stateList) + assert.deepEqual(relay.unexpected, [ + 'google_project_service.required["run.googleapis.com"]', + 'google_secret_manager_secret_iam_member.runtime_artifact_write_secret_accessor[0]' + ]) +}) diff --git a/cloud/dev/scripts/validate-relay-asia-topology-plan.mjs b/cloud/dev/scripts/validate-relay-asia-topology-plan.mjs new file mode 100644 index 00000000000..953a214156e --- /dev/null +++ b/cloud/dev/scripts/validate-relay-asia-topology-plan.mjs @@ -0,0 +1,295 @@ +import { readFileSync } from 'node:fs' +import { fileURLToPath } from 'node:url' + +const REGION = 'asia-east2' +const CELL_SHAPES = { + production: { + domain: 'relay.onorca.dev', + project: 'onorca-cloud', + cells: { + 'production-gce-c27': 'asia-east2-a', + 'production-gce-c28': 'asia-east2-b', + 'production-gce-c29': 'asia-east2-c' + } + }, + staging: { + domain: 'relay-staging.onorca.dev', + project: 'onorca-cloud-staging', + cells: { 'staging-gce-c4': 'asia-east2-a' } + } +} + +function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['plan-json', 'environment', 'cell-ids', 'region', 'image']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!(values.environment in CELL_SHAPES)) throw new Error('--environment is invalid') + const cells = values['cell-ids'].split(',').map((value) => value.trim()).filter(Boolean) + const expectedCells = Object.keys(CELL_SHAPES[values.environment].cells) + if (new Set(cells).size !== cells.length || JSON.stringify(cells.sort()) !== JSON.stringify(expectedCells.sort())) { + throw new Error('--cell-ids must be the exact reviewed Asia topology set') + } + if (values.region !== REGION) throw new Error('--region must be asia-east2') + const expectedImagePrefix = `us-central1-docker.pkg.dev/${CELL_SHAPES[values.environment].project}/orca-cloud/relay@sha256:` + if (!values.image.startsWith(expectedImagePrefix) || !/sha256:[a-f0-9]{64}$/.test(values.image)) { + throw new Error('--image must be the environment Relay image pinned by digest') + } + return { planJson: values['plan-json'], environment: values.environment, cells, image: values.image } +} + +function address(resource, key) { + return `${resource}[${JSON.stringify(key)}]` +} + +function actions(change) { + return change.change?.actions ?? [] +} + +function sameActions(change, expected) { + return JSON.stringify(actions(change)) === JSON.stringify(expected) +} + +function startupValue(script, name) { + return new RegExp(`printf '${name}=%s\\\\n' '([^']+)'`).exec(script)?.[1] +} + +function relayGceName(environment) { + return environment === 'production' ? 'orca-cloud-relay-gce' : 'orca-cloud-staging-relay-gce' +} + +function unknownOrMatches(value, predicate) { + return value === undefined || value === null || predicate(String(value)) +} + +function requireCellTemplate(change, config, cellId) { + const after = change.change.after + const script = after?.metadata_startup_script ?? '' + if ( + after?.machine_type !== 'e2-standard-4' || + after?.labels?.['orca-relay-cell'] !== cellId || + after?.labels?.['orca-relay-region'] !== REGION || + !unknownOrMatches( + after?.network_interface?.[0]?.subnetwork, + (value) => value.includes(`/regions/${REGION}/subnetworks/`) + ) || + (after?.network_interface?.[0]?.access_config?.length ?? 0) !== 0 || + startupValue(script, 'ORCA_RELAY_REGION') !== REGION || + startupValue(script, 'ORCA_RELAY_CELL_CAPACITY') !== '6000' || + startupValue(script, 'ORCA_RELAY_DATABASE_POOL_MAX') !== '10' || + startupValue(script, 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP') !== '3000' || + startupValue(script, 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND') !== '60' || + startupValue(script, 'ORCA_RELAY_IMAGE_DIGEST') !== config.image.split('@')[1] || + !script.includes(`docker pull '${config.image}'`) || + !script.trimEnd().includes(`'${config.image}'`) + ) throw new Error(`${change.address} does not have the reviewed Asia cell shape`) +} + +function requireCellManager(change, config, cellId) { + const after = change.change.after + const hostname = cellId.split('-').at(-1) + const version = after?.version?.[0] + if ( + after?.zone !== CELL_SHAPES[config.environment].cells[cellId] || + after?.target_size !== 1 || + after?.version?.length !== 1 || + version?.name !== 'primary' || + !unknownOrMatches(version?.instance_template, (value) => + value.includes( + `/global/instanceTemplates/${relayGceName(config.environment)}-${hostname}-` + )) || + after?.update_policy?.[0]?.replacement_method !== 'RECREATE' || + after?.update_policy?.[0]?.max_surge_fixed !== 0 || + after?.update_policy?.[0]?.max_unavailable_fixed !== 1 + ) throw new Error(`${change.address} does not have the reviewed fixed-one Asia MIG shape`) +} + +function requireCellBackend(change, config, cellId) { + const after = change.change.after + const backend = after?.backend?.[0] + const zone = CELL_SHAPES[config.environment].cells[cellId] + const hostname = cellId.split('-').at(-1) + const name = `${relayGceName(config.environment)}-${hostname}` + if ( + after?.timeout_sec !== 86_400 || + after?.connection_draining_timeout_sec !== 300 || + after?.load_balancing_scheme !== 'EXTERNAL_MANAGED' || + after?.protocol !== 'HTTP' || + after?.port_name !== 'relay' || + after?.session_affinity !== 'NONE' || + after?.health_checks?.length !== 1 || + !unknownOrMatches(after.health_checks[0], (value) => + value.endsWith(`/global/healthChecks/${relayGceName(config.environment)}-ready`)) || + after?.backend?.length !== 1 || + backend?.balancing_mode !== 'UTILIZATION' || + backend?.max_utilization !== 0.8 || + backend?.capacity_scaler !== 1 || + !unknownOrMatches(backend?.group, (value) => + value.endsWith(`/zones/${zone}/instanceGroups/${name}`)) + ) throw new Error(`${change.address} does not have the reviewed Asia backend shape`) +} + +function requireNetworkResource(change, config) { + const after = change.change.after + const networkSuffix = `/global/networks/${relayGceName(config.environment)}` + if (after?.region !== REGION) throw new Error(`${change.address} is outside asia-east2`) + if ( + change.address.startsWith('google_compute_subnetwork.') && + (after.ip_cidr_range !== '10.42.1.0/24' || + after.private_ip_google_access !== true || + after.stack_type !== 'IPV4_ONLY' || + !unknownOrMatches(after.network, (value) => value.endsWith(networkSuffix))) + ) throw new Error(`${change.address} does not have the reviewed Asia subnet shape`) + if ( + change.address.startsWith('google_compute_router.') && + !unknownOrMatches(after.network, (value) => value.endsWith(networkSuffix)) + ) throw new Error(`${change.address} does not have the reviewed Asia router shape`) + if ( + change.address.startsWith('google_compute_router_nat.') && + (after.nat_ip_allocate_option !== 'AUTO_ONLY' || + after.source_subnetwork_ip_ranges_to_nat !== 'LIST_OF_SUBNETWORKS' || + after.subnetwork?.length !== 1 || + !unknownOrMatches(after.subnetwork[0]?.name, (value) => + value.endsWith( + `/regions/${REGION}/subnetworks/${relayGceName(config.environment)}-${REGION}` + )) || + JSON.stringify(after.subnetwork[0]?.source_ip_ranges_to_nat) !== + JSON.stringify(['ALL_IP_RANGES'])) + ) throw new Error(`${change.address} does not have the reviewed Asia NAT shape`) +} + +function canonical(value) { + if (!Array.isArray(value)) return JSON.stringify(value ?? []) + return JSON.stringify([...value].sort((left, right) => JSON.stringify(left).localeCompare(JSON.stringify(right)))) +} + +function normalizeDescription(value) { + return { ...value, description: value.description ?? '' } +} + +function normalizeMatcher(value) { + const apiPrefix = 'https://www.googleapis.com/compute/v1/' + const defaultService = value.default_service + return { + ...normalizeDescription(value), + default_service: typeof defaultService === 'string' && defaultService.startsWith(apiPrefix) + ? defaultService.slice(apiPrefix.length) + : defaultService + } +} + +function requireUrlMap(change, config) { + const before = change.change.before ?? {} + const after = change.change.after ?? {} + const permitted = new Set(['host_rule', 'path_matcher', 'fingerprint']) + const changed = new Set([...Object.keys(before), ...Object.keys(after)].filter( + (key) => JSON.stringify(before[key]) !== JSON.stringify(after[key]) + )) + if ([...changed].some((key) => !permitted.has(key))) { + throw new Error('shared URL map changes outside host routing') + } + const newHosts = new Set() + const newMatchers = new Set() + for (const cellId of config.cells) { + const hostname = cellId.split('-').at(-1) + const host = `${hostname}.${CELL_SHAPES[config.environment].domain}` + const hostRules = after.host_rule?.filter( + (rule) => + rule.hosts?.length === 1 && + rule.hosts[0] === host && + rule.path_matcher === `cell-${hostname}` + ) ?? [] + if (hostRules.length !== 1) { + throw new Error(`shared URL map has no exact host for ${cellId}`) + } + const matchers = after.path_matcher?.filter( + (matcher) => + matcher.name === `cell-${hostname}` && + unknownOrMatches(matcher.default_service, (value) => + value.endsWith( + `/global/backendServices/${relayGceName(config.environment)}-${hostname}` + )) + ) ?? [] + if (matchers.length !== 1) { + throw new Error(`shared URL map has no exact backend route for ${cellId}`) + } + newHosts.add(host) + newMatchers.add(`cell-${hostname}`) + } + const preservedHostRules = (after.host_rule ?? []).filter( + (rule) => !(rule.hosts?.length === 1 && newHosts.has(rule.hosts[0])) + ) + const preservedMatchers = (after.path_matcher ?? []).filter( + (matcher) => !newMatchers.has(matcher.name) + ) + if ( + sameActions(change, ['update']) && + canonical(preservedHostRules.map(normalizeDescription)) !== + canonical((before.host_rule ?? []).map(normalizeDescription)) || + sameActions(change, ['update']) && + canonical(preservedMatchers.map(normalizeMatcher)) !== + canonical((before.path_matcher ?? []).map(normalizeMatcher)) + ) { + throw new Error('shared URL map does not preserve every existing exact route') + } +} + +export function validateRelayAsiaTopologyPlan(plan, config) { + if (!Array.isArray(plan.resource_changes)) throw new Error('Terraform plan has no resource changes') + const required = new Map([ + [address('google_compute_subnetwork.relay_gce_additional', REGION), [['create'], ['no-op']]], + [address('google_compute_router.relay_gce_additional', REGION), [['create'], ['no-op']]], + [address('google_compute_router_nat.relay_gce_additional', REGION), [['create'], ['no-op']]], + ['google_compute_url_map.relay_gce[0]', [['update'], ['no-op']]] + ]) + for (const cellId of config.cells) { + required.set(address('google_compute_instance_template.relay_gce_cell', cellId), [['create'], ['no-op']]) + required.set(address('google_compute_instance_group_manager.relay_gce_cell', cellId), [['create'], ['no-op']]) + required.set(address('google_compute_backend_service.relay_gce_cell', cellId), [['create'], ['no-op']]) + } + const byAddress = new Map(plan.resource_changes.map((change) => [change.address, change])) + for (const [resourceAddress, allowedActions] of required) { + const change = byAddress.get(resourceAddress) + if (!change || !allowedActions.some((expected) => sameActions(change, expected))) { + throw new Error(`${resourceAddress} is absent or has an unreviewed topology action`) + } + const cellId = config.cells.find((candidate) => resourceAddress.endsWith(`[${JSON.stringify(candidate)}]`)) + if (resourceAddress.startsWith('google_compute_instance_template.') && cellId) { + requireCellTemplate(change, config, cellId) + } else if (resourceAddress.startsWith('google_compute_instance_group_manager.') && cellId) { + requireCellManager(change, config, cellId) + } else if (resourceAddress.startsWith('google_compute_backend_service.') && cellId) { + requireCellBackend(change, config, cellId) + } else if ( + resourceAddress.startsWith('google_compute_subnetwork.') || + resourceAddress.startsWith('google_compute_router.') || + resourceAddress.startsWith('google_compute_router_nat.') + ) { + requireNetworkResource(change, config) + } else if (resourceAddress === 'google_compute_url_map.relay_gce[0]') { + requireUrlMap(change, config) + } + } + const changes = plan.resource_changes.filter((change) => !actions(change).every( + (action) => action === 'no-op' || action === 'read' + )) + for (const change of changes) { + const allowedActions = required.get(change.address) + if (!allowedActions || !allowedActions.some((expected) => sameActions(change, expected))) { + throw new Error(`${change.address} has an unreviewed topology action`) + } + } + return { environment: config.environment, cells: config.cells, changes: changes.length } +} + +if (process.argv[1] === fileURLToPath(import.meta.url)) { + const config = parseArguments(process.argv.slice(2)) + const plan = JSON.parse(readFileSync(config.planJson, 'utf8')) + console.log(JSON.stringify(validateRelayAsiaTopologyPlan(plan, config))) +} diff --git a/cloud/dev/scripts/validate-relay-asia-topology-plan.test.mjs b/cloud/dev/scripts/validate-relay-asia-topology-plan.test.mjs new file mode 100644 index 00000000000..154030489f5 --- /dev/null +++ b/cloud/dev/scripts/validate-relay-asia-topology-plan.test.mjs @@ -0,0 +1,223 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { validateRelayAsiaTopologyPlan } from './validate-relay-asia-topology-plan.mjs' + +const image = `us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:${'a'.repeat(64)}` +const config = { environment: 'staging', cells: ['staging-gce-c4'], image } +const create = (address, after = {}) => ({ address, change: { actions: ['create'], after } }) +const script = [ + `printf 'ORCA_RELAY_REGION=%s\\n' 'asia-east2'`, + `printf 'ORCA_RELAY_CELL_CAPACITY=%s\\n' '6000'`, + `printf 'ORCA_RELAY_DATABASE_POOL_MAX=%s\\n' '10'`, + `printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '3000'`, + `printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`, + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${image.split('@')[1]}'`, + `docker pull '${image}'`, + `'${image}'` +].join('\n') +const resources = [ + create('google_compute_subnetwork.relay_gce_additional["asia-east2"]', { + region: 'asia-east2', ip_cidr_range: '10.42.1.0/24', private_ip_google_access: true, + stack_type: 'IPV4_ONLY', + network: 'projects/p/global/networks/orca-cloud-staging-relay-gce' + }), + create('google_compute_router.relay_gce_additional["asia-east2"]', { + region: 'asia-east2', network: 'projects/p/global/networks/orca-cloud-staging-relay-gce' + }), + create('google_compute_router_nat.relay_gce_additional["asia-east2"]', { + region: 'asia-east2', nat_ip_allocate_option: 'AUTO_ONLY', + source_subnetwork_ip_ranges_to_nat: 'LIST_OF_SUBNETWORKS', + subnetwork: [{ + name: 'projects/p/regions/asia-east2/subnetworks/orca-cloud-staging-relay-gce-asia-east2', + source_ip_ranges_to_nat: ['ALL_IP_RANGES'] + }] + }), + create('google_compute_instance_template.relay_gce_cell["staging-gce-c4"]', { + machine_type: 'e2-standard-4', + labels: { 'orca-relay-cell': 'staging-gce-c4', 'orca-relay-region': 'asia-east2' }, + network_interface: [{ + subnetwork: 'projects/p/regions/asia-east2/subnetworks/relay', access_config: [] + }], + metadata_startup_script: script + }), + create('google_compute_instance_group_manager.relay_gce_cell["staging-gce-c4"]', { + zone: 'asia-east2-a', target_size: 1, + version: [{ + name: 'primary', + instance_template: 'projects/p/global/instanceTemplates/orca-cloud-staging-relay-gce-c4-abc' + }], + update_policy: [{ replacement_method: 'RECREATE', max_surge_fixed: 0, max_unavailable_fixed: 1 }] + }), + create('google_compute_backend_service.relay_gce_cell["staging-gce-c4"]', { + timeout_sec: 86_400, connection_draining_timeout_sec: 300, + load_balancing_scheme: 'EXTERNAL_MANAGED', protocol: 'HTTP', port_name: 'relay', + session_affinity: 'NONE', + health_checks: ['projects/p/global/healthChecks/orca-cloud-staging-relay-gce-ready'], + backend: [{ + balancing_mode: 'UTILIZATION', max_utilization: 0.8, capacity_scaler: 1, + group: 'projects/p/zones/asia-east2-a/instanceGroups/orca-cloud-staging-relay-gce-c4' + }] + }), + { + address: 'google_compute_url_map.relay_gce[0]', + change: { + actions: ['update'], + before: { host_rule: [], path_matcher: [], fingerprint: 'old' }, + after: { + host_rule: [{ + hosts: ['c4.relay-staging.onorca.dev'], path_matcher: 'cell-c4' + }], + path_matcher: [{ + name: 'cell-c4', + default_service: 'projects/p/global/backendServices/orca-cloud-staging-relay-gce-c4' + }], + fingerprint: null + } + } + } +] + +test('accepts the exact additive staging Asia topology', () => { + assert.deepEqual(validateRelayAsiaTopologyPlan({ resource_changes: resources }, config), { + environment: 'staging', cells: ['staging-gce-c4'], changes: 7 + }) +}) + +test('accepts an idempotent empty plan', () => { + const noChanges = structuredClone(resources).map((resource) => ({ + ...resource, + change: { + ...resource.change, + actions: ['no-op'], + before: structuredClone(resource.change.after) + } + })) + assert.equal(validateRelayAsiaTopologyPlan({ resource_changes: noChanges }, config).changes, 0) +}) + +test('rejects a plan that omits any required topology resource', () => { + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: resources.slice(1) }, config), + /absent or has an unreviewed topology action/ + ) +}) + +test('rejects any US or unrelated mutation', () => { + const plan = structuredClone(resources) + plan.push(create('google_compute_subnetwork.relay_gce[0]')) + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /unreviewed topology action/ + ) +}) + +test('rejects delete and replacement actions', () => { + for (const invalidActions of [['delete'], ['create', 'delete']]) { + const plan = structuredClone(resources) + plan[0].change.actions = invalidActions + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /unreviewed topology action/ + ) + } +}) + +test('rejects a cell with different limits or image', () => { + const plan = structuredClone(resources) + plan[3].change.after.metadata_startup_script = script.replace("'3000'", "'5000'") + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /reviewed Asia cell shape/ + ) +}) + +test('rejects shared URL-map changes outside exact host routing', () => { + const plan = structuredClone(resources) + plan[6].change.after.default_service = 'unreviewed' + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /outside host routing/ + ) +}) + +test('rejects removal of an existing exact route', () => { + const plan = structuredClone(resources) + plan[6].change.before.host_rule = [{ hosts: ['c1.relay-staging.onorca.dev'] }] + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /preserve every existing exact route/ + ) +}) + +test('accepts provider normalization of preserved route descriptions', () => { + const plan = structuredClone(resources) + const matcher = { + name: 'cell-c1', + description: '', + default_service: + 'https://www.googleapis.com/compute/v1/projects/p/global/backendServices/orca-cloud-staging-relay-gce-c1' + } + plan[6].change.before.host_rule = [{ + description: '', hosts: ['c1.relay-staging.onorca.dev'], path_matcher: 'cell-c1' + }] + plan[6].change.before.path_matcher = [matcher] + plan[6].change.after.host_rule.unshift({ + description: null, hosts: ['c1.relay-staging.onorca.dev'], path_matcher: 'cell-c1' + }) + plan[6].change.after.path_matcher.unshift({ + ...matcher, + description: null, + default_service: 'projects/p/global/backendServices/orca-cloud-staging-relay-gce-c1' + }) + assert.equal(validateRelayAsiaTopologyPlan({ resource_changes: plan }, config).changes, 7) +}) + +test('rejects a changed preserved route backend', () => { + const plan = structuredClone(resources) + plan[6].change.before.path_matcher = [{ + name: 'cell-c1', + default_service: + 'https://www.googleapis.com/compute/v1/projects/p/global/backendServices/orca-cloud-staging-relay-gce-c1' + }] + plan[6].change.after.path_matcher.unshift({ + name: 'cell-c1', + default_service: 'projects/p/global/backendServices/orca-cloud-staging-relay-gce-c2' + }) + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /preserve every existing exact route/ + ) +}) + +test('rejects a different Asia subnet range', () => { + const plan = structuredClone(resources) + plan[0].change.after.ip_cidr_range = '10.99.0.0/24' + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: plan }, config), + /reviewed Asia subnet shape/ + ) +}) + +test('rejects incomplete NAT, backend, and URL routing shapes', () => { + const nat = structuredClone(resources) + nat[2].change.after.subnetwork[0].source_ip_ranges_to_nat = [] + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: nat }, config), + /reviewed Asia NAT shape/ + ) + + const backend = structuredClone(resources) + backend[5].change.after.backend[0].group = 'projects/p/zones/asia-east2-a/instanceGroups/wrong' + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: backend }, config), + /reviewed Asia backend shape/ + ) + + const route = structuredClone(resources) + route[6].change.after.path_matcher[0].default_service = + 'projects/p/global/backendServices/wrong' + assert.throws( + () => validateRelayAsiaTopologyPlan({ resource_changes: route }, config), + /no exact backend route/ + ) +}) diff --git a/cloud/dev/scripts/validate-relay-capacity-plan.mjs b/cloud/dev/scripts/validate-relay-capacity-plan.mjs new file mode 100644 index 00000000000..7307295d206 --- /dev/null +++ b/cloud/dev/scripts/validate-relay-capacity-plan.mjs @@ -0,0 +1,519 @@ +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' + +const SERVICE_ACCOUNT_EMAIL = + /^[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com$/ + +function parseArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of ['mode', 'cell-id', 'hard-cap', 'unobserved-bound']) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!['cell', 'bootstrap-cell', 'same-cap-cell', 'same-cap-image'].includes(values.mode)) { + throw new Error('--mode must be cell, bootstrap-cell, same-cap-cell, or same-cap-image') + } + const integer = (key) => { + const value = Number(values[key]) + if (!Number.isSafeInteger(value) || value < 0) throw new Error(`--${key} is invalid`) + return value + } + if (!values.image) throw new Error('missing --image') + if (values.mode === 'bootstrap-cell' && !values['capacity-service-account']) { + throw new Error('missing --capacity-service-account') + } + if ( + values.mode === 'same-cap-cell' && + (!values['rollback-image'] || + !values['rehome-director-service-account'] || + !values['rehome-audience']) + ) throw new Error('same-cap validation requires rollback image and rehome trust config') + if (values.mode === 'same-cap-image' && !values['rollback-image']) { + throw new Error('same-cap image validation requires a rollback image') + } + if ( + values['capacity-service-account'] !== undefined && + !SERVICE_ACCOUNT_EMAIL.test(values['capacity-service-account']) + ) { + throw new Error('--capacity-service-account is invalid') + } + return { + mode: values.mode, + cellId: values['cell-id'], + hardCap: integer('hard-cap'), + unobservedBound: integer('unobserved-bound'), + image: values.image, + capacityServiceAccount: values['capacity-service-account'], + rollbackImage: values['rollback-image'], + rehomeDirectorServiceAccount: values['rehome-director-service-account'], + rehomeAudience: values['rehome-audience'] + } +} + +function mutations(plan) { + if (!Array.isArray(plan.resource_changes)) throw new Error('Terraform plan has no resource changes') + return plan.resource_changes.filter(({ change }) => { + const actions = change?.actions + return Array.isArray(actions) && !actions.every((action) => ['no-op', 'read'].includes(action)) + }) +} + +function sameActions(change, expected) { + return JSON.stringify(change.change?.actions) === JSON.stringify(expected) +} + +function changedPaths(before, after, path = []) { + if (Object.is(before, after)) return [] + const beforeObject = before !== null && typeof before === 'object' + const afterObject = after !== null && typeof after === 'object' + if (!beforeObject || !afterObject || Array.isArray(before) !== Array.isArray(after)) { + return [path.join('.')] + } + const keys = new Set([...Object.keys(before), ...Object.keys(after)]) + return [...keys].flatMap((key) => changedPaths(before[key], after[key], [...path, key])) +} + +function unknownPaths(value, path = []) { + if (value === true) return [path.join('.')] + if (value === null || typeof value !== 'object') return [] + return Object.entries(value).flatMap(([key, nested]) => unknownPaths(nested, [...path, key])) +} + +function valueAtPath(value, path) { + return path.split('.').reduce((current, key) => current?.[key], value) +} + +function providerDefaultPaths(change, paths) { + const empty = (value) => + value === '' || + value === 0 || + (Array.isArray(value) && value.length === 0) || + (value !== null && + typeof value === 'object' && + !Array.isArray(value) && + Object.keys(value).length === 0) + return paths.filter( + (path) => + empty(valueAtPath(change.change.before, path)) && + valueAtPath(change.change.after, path) === null + ) +} + +function canonicalResourcePaths(change, paths) { + const canonical = (value) => + typeof value === 'string' + ? value.replace('https://www.googleapis.com/compute/v1/', '') + : value + return paths.filter( + (path) => + canonical(valueAtPath(change.change.before, path)) === + canonical(valueAtPath(change.change.after, path)) + ) +} + +function bootstrapRestartNormalizationPaths(change, mode) { + if (!['bootstrap-cell', 'same-cap-cell', 'same-cap-image'].includes(mode)) return [] + const policyMatches = + valueAtPath(change.change.before, 'update_policy.0.minimal_action') === 'RESTART' && + valueAtPath(change.change.after, 'update_policy.0.minimal_action') === 'REPLACE' + const priorVersion = valueAtPath(change.change.before, 'version.0.name') + const versionMatches = + typeof priorVersion === 'string' && + /^0\/\d{4}-\d{2}-\d{2} \d{2}:\d{2}:\d{2}\.\d{6}\+00:00$/.test(priorVersion) && + valueAtPath(change.change.after, 'version.0.name') === 'primary' + return policyMatches && versionMatches + ? ['update_policy.0.minimal_action', 'version.0.name'] + : [] +} + +function requireOnlyPaths(change, allowed, required = [], allowedUnknown = new Set()) { + const paths = changedPaths(change.change.before, change.change.after) + const unknown = unknownPaths(change.change.after_unknown) + const unexpected = paths.filter((path) => !allowed.has(path)) + const unexpectedUnknown = unknown.filter((path) => !allowedUnknown.has(path)) + if ( + unexpected.length > 0 || + unexpectedUnknown.length > 0 || + required.some((path) => !paths.includes(path)) + ) { + throw new Error(`${change.address} changes outside the reviewed capacity fields`) + } +} + +function relayImage(script) { + const lines = script.split('\n') + const starts = lines.flatMap((line, index) => + line === 'docker run --detach \\' ? [index] : []) + const commands = starts.map((start) => { + const end = lines.findIndex((line, index) => index > start && !line.endsWith(' \\')) + return end < 0 ? [] : lines.slice(start, end + 1) + }) + const relayCommands = commands.filter((command) => + command.filter((line) => line === ' --name orca-relay \\').length === 1) + if (relayCommands.length !== 1) return null + const command = relayCommands[0] + const image = /^ '([^'\n]+@sha256:[a-f0-9]{64})'$/.exec(command.at(-1))?.[1] + const digests = [...command.join('\n').matchAll(/'([^'\n]+@sha256:[a-f0-9]{64})'/g)] + return image && digests.length === 1 ? image : null +} + +function normalizedStartupScript( + script, + stripCapacityIdentity = false, + stripRehomeConfig = false, + preserveCapacity = false +) { + const image = relayImage(script) + if (!image) throw new Error('cell plan startup script has no Relay image') + const digest = image.split('@')[1] + const capacityAssignment = + /^ printf 'ORCA_RELAY_CELL_CONNECTION_(?:HARD_CAP|UNOBSERVED_BOUND)=%s\\n' '[0-9]+'$/ + const capacityIdentity = + /^ printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com'$/ + const rehomeConfig = + /^ printf 'ORCA_RELAY_REHOME_(?:DIRECTOR_SERVICE_ACCOUNT|AUDIENCE)=%s\\n' '[^'\n]+'$/ + return script + .split('\n') + .filter( + (line) => + (preserveCapacity || !capacityAssignment.test(line)) && + (!stripCapacityIdentity || !capacityIdentity.test(line)) && + (!stripRehomeConfig || !rehomeConfig.test(line)) + ) + .join('\n') + .replaceAll(image, '') + .replaceAll(digest, '') +} + +function hasExactSingleAssignment(lines, pattern, expected) { + const assignments = lines.filter((line) => pattern.test(line)) + return assignments.length === 1 && assignments[0] === expected +} + +function requireDesiredStartupScript(script, config) { + const lines = typeof script === 'string' ? script.split('\n') : [] + const expected = [ + [ + /^ printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '[0-9]+'$/, + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${config.hardCap}'` + ], + [ + /^ printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '[0-9]+'$/, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '${config.unobservedBound}'` + ] + ] + if (config.mode === 'bootstrap-cell') { + expected.push([ + /^ printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '[a-z][a-z0-9-]{4,28}[a-z0-9]@[a-z0-9-]+\.iam\.gserviceaccount\.com'$/, + ` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '${config.capacityServiceAccount}'` + ]) + } + if (config.mode === 'same-cap-cell') { + expected.push( + [ + /^ printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '[^'\n]+'$/, + ` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${config.rehomeDirectorServiceAccount}'` + ], + [ + /^ printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '[^'\n]+'$/, + ` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${config.rehomeAudience}'` + ] + ) + } + if ( + typeof script !== 'string' || + relayImage(script) !== config.image || + expected.some(([pattern, line]) => !hasExactSingleAssignment(lines, pattern, line)) + ) { + throw new Error('cell plan does not contain the reviewed image and capacity') + } +} + +function plannedResources(module) { + if (!module) return [] + return [ + ...(module.resources ?? []), + ...(module.child_modules ?? []).flatMap(plannedResources) + ] +} + +function canonicalResource(value) { + return typeof value === 'string' + ? value.replace('https://www.googleapis.com/compute/v1/', '') + : value +} + +function requireDesiredPlannedCell(plan, config) { + const resources = plannedResources(plan.planned_values?.root_module) + const templateAddress = `google_compute_instance_template.relay_gce_cell[${JSON.stringify(config.cellId)}]` + const managerAddress = `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(config.cellId)}]` + const template = resources.find(({ address }) => address === templateAddress)?.values + const manager = resources.find(({ address }) => address === managerAddress)?.values + if (!template || !manager) throw new Error('convergence plan has no exact planned C26 state') + requireDesiredStartupScript(template.metadata_startup_script, config) + const templateReference = canonicalResource(template.self_link ?? template.id) + const managerReference = canonicalResource(manager.version?.[0]?.instance_template) + if (!templateReference || templateReference !== managerReference) { + throw new Error('convergence plan does not bind the MIG to the reviewed template') + } +} + +function validateManagerUpdate(manager, config) { + if (!sameActions(manager, ['update'])) { + throw new Error('cell plan has unexpected MIG actions') + } + const managerComputed = new Set([ + 'fingerprint', + 'operation', + 'status', + 'version.0.instance_template' + ]) + const managerUnknown = unknownPaths(manager.change.after_unknown) + requireOnlyPaths( + manager, + new Set([ + 'version.0.instance_template', + ...bootstrapRestartNormalizationPaths(manager, config.mode), + ...managerUnknown.filter((path) => managerComputed.has(path)) + ]), + ['version.0.instance_template'], + managerComputed + ) +} + +function requireReplacementTemplateDependency(plan, template, manager) { + const configuredManager = plan.configuration?.root_module?.resources?.find( + ({ address }) => address === 'google_compute_instance_group_manager.relay_gce_cell' + ) + const expectedVersionExpression = [{ + instance_template: { + references: ['google_compute_instance_template.relay_gce_cell', 'each.key'] + }, + name: { constant_value: 'primary' } + }] + if ( + JSON.stringify(configuredManager?.expressions?.version) !== + JSON.stringify(expectedVersionExpression) || + template.change.after?.self_link != null || + template.change.after_unknown?.self_link !== true || + manager.change.after?.version?.[0]?.instance_template != null || + manager.change.after_unknown?.version?.[0]?.instance_template !== true + ) { + throw new Error('cell plan does not bind the MIG to the reviewed template dependency') + } +} + +function cellPlan(plan, changes, config) { + const templateAddress = `google_compute_instance_template.relay_gce_cell[${JSON.stringify(config.cellId)}]` + const managerAddress = `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(config.cellId)}]` + const template = changes.find( + ({ address, deposed }) => address === templateAddress && deposed === undefined + ) + const manager = changes.find(({ address }) => address === managerAddress) + const obsoleteTemplates = changes.filter( + ({ address, deposed }) => address === templateAddress && typeof deposed === 'string' + ) + const allowsObsoleteTemplates = + config.mode === 'same-cap-image' && + obsoleteTemplates.length > 0 && + obsoleteTemplates.every((change) => sameActions(change, ['delete'])) + if ( + !template || + !manager || + changes.length !== 2 + obsoleteTemplates.length || + (obsoleteTemplates.length > 0 && !allowsObsoleteTemplates) + ) { + throw new Error('cell plan must change only the exact instance template and MIG') + } + if (!sameActions(template, ['create', 'delete']) || !sameActions(manager, ['update'])) { + throw new Error('cell plan has unexpected replacement actions') + } + const templateComputed = new Set([ + 'confidential_instance_config', + 'creation_timestamp', + 'disk.0.architecture', + 'disk.0.interface', + 'disk.0.mode', + 'disk.0.provisioned_iops', + 'disk.0.provisioned_throughput', + 'disk.0.type', + 'id', + 'metadata_fingerprint', + 'name', + 'network_interface.0.internal_ipv6_prefix_length', + 'network_interface.0.ipv6_access_type', + 'network_interface.0.ipv6_address', + 'network_interface.0.name', + 'network_interface.0.network', + 'network_interface.0.stack_type', + 'network_interface.0.subnetwork_project', + 'numeric_id', + 'region', + 'self_link', + 'self_link_unique', + 'tags_fingerprint' + ]) + const templateDefaults = [ + 'description', + 'disk.0.disk_name', + 'disk.0.guest_os_features', + 'disk.0.labels', + 'disk.0.resource_manager_tags', + 'disk.0.resource_policies', + 'disk.0.source', + 'disk.0.source_snapshot', + 'instance_description', + 'key_revocation_action_type', + 'min_cpu_platform', + 'network_interface.0.network_ip', + 'network_interface.0.nic_type', + 'network_interface.0.queue_count', + 'scheduling.0.availability_domain', + 'scheduling.0.instance_termination_action', + 'scheduling.0.min_node_cpus', + 'scheduling.0.termination_time' + ] + const templateCanonicalResources = [ + 'disk.0.source_image', + 'network_interface.0.subnetwork' + ] + const templateUnknown = unknownPaths(template.change.after_unknown) + requireOnlyPaths( + template, + new Set([ + 'metadata_startup_script', + ...templateUnknown.filter((path) => templateComputed.has(path)), + ...providerDefaultPaths(template, templateDefaults), + ...canonicalResourcePaths(template, templateCanonicalResources) + ]), + ['metadata_startup_script'], + templateComputed + ) + validateManagerUpdate(manager, config) + requireReplacementTemplateDependency(plan, template, manager) + const beforeScript = template.change.before?.metadata_startup_script + const script = template.change.after?.metadata_startup_script + requireDesiredStartupScript(script, config) + const sameCap = ['same-cap-cell', 'same-cap-image'].includes(config.mode) + if ( + typeof beforeScript !== 'string' || + (sameCap && relayImage(beforeScript) !== config.rollbackImage) || + normalizedStartupScript( + beforeScript, + config.mode === 'bootstrap-cell', + config.mode === 'same-cap-cell', + sameCap + ) !== normalizedStartupScript( + script, + config.mode === 'bootstrap-cell', + config.mode === 'same-cap-cell', + sameCap + ) + ) { + throw new Error('cell plan does not contain the reviewed image and capacity') + } +} + +function convergenceCellPlan(plan, changes, config) { + const templateAddress = `google_compute_instance_template.relay_gce_cell[${JSON.stringify(config.cellId)}]` + const managerAddress = `google_compute_instance_group_manager.relay_gce_cell[${JSON.stringify(config.cellId)}]` + const manager = changes.find(({ address }) => address === managerAddress) + const obsoleteTemplates = changes.filter( + ({ address, deposed }) => address === templateAddress && typeof deposed === 'string' + ) + if ( + (!manager && obsoleteTemplates.length === 0) || + (obsoleteTemplates.length > 1 && config.mode !== 'same-cap-image') || + changes.length !== (manager ? 1 : 0) + obsoleteTemplates.length + ) { + throw new Error('cell convergence plan changes outside the exact template and MIG') + } + if (manager) validateManagerUpdate(manager, config) + if (obsoleteTemplates.some((change) => !sameActions(change, ['delete']))) { + throw new Error('cell convergence plan has unexpected obsolete-template actions') + } + requireDesiredPlannedCell(plan, config) +} + +export function validateCapacityPlan(plan, config) { + if (!['cell', 'bootstrap-cell', 'same-cap-cell', 'same-cap-image'].includes(config.mode)) { + throw new Error('capacity Terraform plans may change only a cell') + } + if ( + config.mode === 'bootstrap-cell' && + !SERVICE_ACCOUNT_EMAIL.test(config.capacityServiceAccount ?? '') + ) { + throw new Error('capacity Terraform plan has an invalid service account') + } + if ( + config.mode === 'same-cap-cell' && + (!SERVICE_ACCOUNT_EMAIL.test(config.rehomeDirectorServiceAccount ?? '') || + !/^https:\/\/[^/]+\/v1\/admin\/host-drain$/.test(config.rehomeAudience ?? '') || + !/^.+@sha256:[a-f0-9]{64}$/.test(config.rollbackImage ?? '')) + ) throw new Error('same-cap Terraform plan has invalid rehome trust config') + if ( + config.mode === 'same-cap-image' && + !/^.+@sha256:[a-f0-9]{64}$/.test(config.rollbackImage ?? '') + ) throw new Error('same-cap image Terraform plan has an invalid rollback image') + const changes = mutations(plan) + if (changes.length === 0) { + return { + mode: config.mode, + changes: 0, + ...(config.mode === 'same-cap-image' ? { changeKind: 'none' } : {}) + } + } + const replacement = changes.some( + ({ address, deposed, change }) => + address === `google_compute_instance_template.relay_gce_cell[${JSON.stringify(config.cellId)}]` && + deposed === undefined && + JSON.stringify(change?.actions) === JSON.stringify(['create', 'delete']) + ) + if (replacement) cellPlan(plan, changes, config) + else convergenceCellPlan(plan, changes, config) + const obsoleteTemplateOnly = changes.every( + ({ address, deposed, change }) => + address === `google_compute_instance_template.relay_gce_cell[${JSON.stringify(config.cellId)}]` && + typeof deposed === 'string' && + JSON.stringify(change?.actions) === JSON.stringify(['delete']) + ) + return { + mode: config.mode, + changes: changes.length, + ...(config.mode === 'same-cap-image' + ? { + changeKind: replacement + ? changes.some( + ({ address, deposed }) => + address === `google_compute_instance_template.relay_gce_cell[${JSON.stringify(config.cellId)}]` && + typeof deposed === 'string' + ) + ? 'replacement-with-obsolete-template' + : 'replacement' + : obsoleteTemplateOnly + ? 'obsolete-template-delete' + : 'manager-convergence' + } + : {}) + } +} + +export function main(argv = process.argv.slice(2)) { + const config = parseArguments(argv) + const plan = JSON.parse(readFileSync(0, 'utf8')) + process.stdout.write(`${JSON.stringify({ event: 'relay_capacity_plan_verified', ...validateCapacityPlan(plan, config) })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + try { + main() + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + } +} diff --git a/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs b/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs new file mode 100644 index 00000000000..207285dc570 --- /dev/null +++ b/cloud/dev/scripts/validate-relay-capacity-plan.test.mjs @@ -0,0 +1,646 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { validateCapacityPlan as validateCapacityPlanRaw } from './validate-relay-capacity-plan.mjs' + +const config = { + cellId: 'staging-gce-c3', + hardCap: 1_000, + unobservedBound: 60 +} + +function replacementConfiguration(references = [ + 'google_compute_instance_template.relay_gce_cell', + 'each.key' +]) { + return { + root_module: { + resources: [{ + address: 'google_compute_instance_group_manager.relay_gce_cell', + expressions: { + version: [{ + instance_template: { references }, + name: { constant_value: 'primary' } + }] + } + }] + } + } +} + +function validateCapacityPlan(plan, planConfig) { + const replacement = plan.resource_changes.some(({ change }) => + JSON.stringify(change?.actions) === JSON.stringify(['create', 'delete'])) + return validateCapacityPlanRaw( + replacement && !plan.configuration + ? { ...plan, configuration: replacementConfiguration() } + : plan, + planConfig + ) +} + +test('accepts only the exact canary template replacement and MIG update', () => { + const image = `us-docker.pkg.dev/project/relay/image@sha256:${'a'.repeat(64)}` + const startupScript = (cap, bound, selectedImage, extra = '') => + [ + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${cap}'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '${bound}'`, + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${selectedImage.split('@')[1]}'`, + `docker pull '${selectedImage}'`, + extra, + 'docker run --detach \\', + ' --name cloud-sql-proxy \\', + ` 'us-docker.pkg.dev/project/proxy@sha256:${'c'.repeat(64)}'`, + 'docker run --detach \\', + ' --name orca-relay \\', + ` '${selectedImage}'` + ].join('\n') + const script = startupScript(1_000, 60, image) + const template = { + address: 'google_compute_instance_template.relay_gce_cell["staging-gce-c3"]', + change: { + actions: ['create', 'delete'], + before: { + metadata_startup_script: startupScript( + 600, + 60, + `us-docker.pkg.dev/project/relay/image@sha256:${'b'.repeat(64)}` + ) + }, + after: { metadata_startup_script: script, self_link: null }, + after_unknown: { self_link: true } + } + } + const manager = { + address: 'google_compute_instance_group_manager.relay_gce_cell["staging-gce-c3"]', + change: { + actions: ['update'], + before: { target_size: 1, version: [{ instance_template: 'old' }] }, + after: { target_size: 1, version: [{ instance_template: null }] }, + after_unknown: { version: [{ instance_template: true }] } + } + } + const cellConfig = { ...config, mode: 'cell', image } + assert.deepEqual(validateCapacityPlan({ resource_changes: [] }, cellConfig), { + mode: 'cell', + changes: 0 + }) + assert.deepEqual(validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), { + mode: 'cell', + changes: 2 + }) + const wrongTemplateManager = structuredClone(manager) + wrongTemplateManager.change.after.version[0].instance_template = + 'projects/project/global/instanceTemplates/unreviewed' + assert.throws( + () => validateCapacityPlan( + { resource_changes: [template, wrongTemplateManager] }, + cellConfig + ), + /does not bind the MIG/ + ) + assert.throws( + () => validateCapacityPlanRaw( + { + resource_changes: [template, manager], + configuration: replacementConfiguration([ + 'google_compute_instance_template.relay_gce_cell', + 'var.unreviewed_key' + ]) + }, + cellConfig + ), + /reviewed template dependency/ + ) + assert.throws( + () => validateCapacityPlanRaw( + { resource_changes: [template, manager] }, + cellConfig + ), + /reviewed template dependency/ + ) + const capacityServiceAccount = + 'orca-cloud-staging-gha-cap@onorca-cloud-staging.iam.gserviceaccount.com' + const bootstrapTemplate = structuredClone(template) + const bootstrapManager = structuredClone(manager) + bootstrapTemplate.change.after.metadata_startup_script = [ + ` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' '${capacityServiceAccount}'`, + script + ].join('\n') + const bootstrapConfig = { + ...cellConfig, + mode: 'bootstrap-cell', + capacityServiceAccount + } + const plannedValues = (startupScript) => ({ + root_module: { + resources: [ + { + address: bootstrapTemplate.address, + values: { + metadata_startup_script: startupScript, + self_link: 'projects/project/global/instanceTemplates/c26-reviewed' + } + }, + { + address: bootstrapManager.address, + values: { + version: [{ + instance_template: + 'https://www.googleapis.com/compute/v1/projects/project/global/instanceTemplates/c26-reviewed' + }] + } + } + ] + } + }) + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, bootstrapManager] }, + bootstrapConfig + ), + { mode: 'bootstrap-cell', changes: 2 } + ) + const managerOnly = structuredClone(bootstrapManager) + assert.deepEqual( + validateCapacityPlan( + { + resource_changes: [managerOnly], + planned_values: plannedValues( + bootstrapTemplate.change.after.metadata_startup_script + ) + }, + bootstrapConfig + ), + { mode: 'bootstrap-cell', changes: 1 } + ) + const decoyPlannedValues = plannedValues( + bootstrapTemplate.change.after.metadata_startup_script.replaceAll( + image, + `us-docker.pkg.dev/project/relay/image@sha256:${'b'.repeat(64)}` + ) + `\n# decoy '${image}'` + ) + assert.throws( + () => validateCapacityPlan( + { resource_changes: [managerOnly], planned_values: decoyPlannedValues }, + bootstrapConfig + ), + /reviewed image and capacity/ + ) + const obsoleteTemplate = structuredClone(bootstrapTemplate) + obsoleteTemplate.deposed = 'retired-template' + obsoleteTemplate.change.actions = ['delete'] + obsoleteTemplate.change.after = null + assert.deepEqual( + validateCapacityPlan( + { + resource_changes: [managerOnly, obsoleteTemplate], + planned_values: plannedValues( + bootstrapTemplate.change.after.metadata_startup_script + ) + }, + bootstrapConfig + ), + { mode: 'bootstrap-cell', changes: 2 } + ) + assert.deepEqual( + validateCapacityPlan( + { + resource_changes: [obsoleteTemplate], + planned_values: plannedValues( + bootstrapTemplate.change.after.metadata_startup_script + ) + }, + bootstrapConfig + ), + { mode: 'bootstrap-cell', changes: 1 } + ) + const wrongManagerReference = plannedValues( + bootstrapTemplate.change.after.metadata_startup_script + ) + wrongManagerReference.root_module.resources[1].values.version[0].instance_template = + 'projects/project/global/instanceTemplates/not-reviewed' + assert.throws( + () => validateCapacityPlan( + { resource_changes: [managerOnly], planned_values: wrongManagerReference }, + bootstrapConfig + ), + /does not bind the MIG/ + ) + assert.throws( + () => validateCapacityPlan( + { resource_changes: [managerOnly] }, + bootstrapConfig + ), + /no exact planned C26 state/ + ) + const restartedBootstrapManager = structuredClone(bootstrapManager) + restartedBootstrapManager.change.before.update_policy = [{ minimal_action: 'RESTART' }] + restartedBootstrapManager.change.after.update_policy = [{ minimal_action: 'REPLACE' }] + restartedBootstrapManager.change.before.version[0].name = + '0/2026-08-10 23:30:14.196895+00:00' + restartedBootstrapManager.change.after.version[0].name = 'primary' + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, restartedBootstrapManager] }, + bootstrapConfig + ), + { mode: 'bootstrap-cell', changes: 2 } + ) + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [template, restartedBootstrapManager] }, + cellConfig + ), + /outside the reviewed capacity fields/ + ) + const unrecognizedRestartManager = structuredClone(restartedBootstrapManager) + unrecognizedRestartManager.change.before.version[0].name = 'operator-version' + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, unrecognizedRestartManager] }, + bootstrapConfig + ), + /outside the reviewed capacity fields/ + ) + const unrecognizedRestartPolicy = structuredClone(restartedBootstrapManager) + unrecognizedRestartPolicy.change.before.update_policy[0].minimal_action = 'REFRESH' + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, unrecognizedRestartPolicy] }, + bootstrapConfig + ), + /outside the reviewed capacity fields/ + ) + const unrecognizedPrimaryVersion = structuredClone(restartedBootstrapManager) + unrecognizedPrimaryVersion.change.after.version[0].name = 'other' + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, unrecognizedPrimaryVersion] }, + bootstrapConfig + ), + /outside the reviewed capacity fields/ + ) + const unrecognizedRestoredPolicy = structuredClone(restartedBootstrapManager) + unrecognizedRestoredPolicy.change.after.update_policy[0].minimal_action = 'REFRESH' + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, unrecognizedRestoredPolicy] }, + bootstrapConfig + ), + /outside the reviewed capacity fields/ + ) + const missingRestartPolicy = structuredClone(restartedBootstrapManager) + delete missingRestartPolicy.change.before.update_policy + delete missingRestartPolicy.change.after.update_policy + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, missingRestartPolicy] }, + bootstrapConfig + ), + /outside the reviewed capacity fields/ + ) + const missingRestartVersion = structuredClone(restartedBootstrapManager) + missingRestartVersion.change.before.version[0].name = 'primary' + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, missingRestartVersion] }, + bootstrapConfig + ), + /outside the reviewed capacity fields/ + ) + const duplicateHardCapTemplate = structuredClone(template) + duplicateHardCapTemplate.change.after.metadata_startup_script = [ + script, + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '600'` + ].join('\n') + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [duplicateHardCapTemplate, structuredClone(manager)] }, + cellConfig + ), + /reviewed image and capacity/ + ) + const duplicateIdentityTemplate = structuredClone(bootstrapTemplate) + duplicateIdentityTemplate.change.after.metadata_startup_script = [ + duplicateIdentityTemplate.change.after.metadata_startup_script, + ` printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\\n' 'other-capacity@onorca-cloud-staging.iam.gserviceaccount.com'` + ].join('\n') + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [duplicateIdentityTemplate, structuredClone(bootstrapManager)] }, + bootstrapConfig + ), + /reviewed image and capacity/ + ) + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, bootstrapManager] }, + { ...bootstrapConfig, capacityServiceAccount: 'invalid' } + ), + /invalid service account/ + ) + assert.throws( + () => + validateCapacityPlan( + { resource_changes: [bootstrapTemplate, bootstrapManager] }, + cellConfig + ), + /reviewed image and capacity/ + ) + manager.change.after.target_size = 0 + assert.throws( + () => validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), + /outside the reviewed capacity fields/ + ) + manager.change.after.target_size = 1 + template.change.after.metadata_startup_script = startupScript(1_000, 60, image, 'curl bad') + assert.throws( + () => validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), + /reviewed image and capacity/ + ) + template.change.after.metadata_startup_script = startupScript( + 1_000, + 60, + image, + 'curl bad # ORCA_RELAY_CELL_CONNECTION_HARD_CAP=' + ) + assert.throws( + () => validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), + /reviewed image and capacity/ + ) + template.change.after.metadata_startup_script = script + template.change.after_unknown = { + id: true, + self_link: true, + disk: [{ architecture: true }] + } + manager.change.after_unknown = { + fingerprint: true, + version: [{ instance_template: true }] + } + assert.deepEqual(validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), { + mode: 'cell', + changes: 2 + }) + template.change.before.description = '' + template.change.after.description = null + template.change.before.disk = [{ + architecture: '', + source_image: 'projects/cos-cloud/global/images/cos-stable-1' + }] + template.change.after.disk = [{ + source_image: 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-1' + }] + template.change.after_unknown.disk = [{ architecture: true }] + assert.deepEqual(validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), { + mode: 'cell', + changes: 2 + }) + template.change.after.disk[0].source_image = + 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/different' + assert.throws( + () => validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), + /outside the reviewed capacity fields/ + ) + template.change.after.disk[0].source_image = + 'https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-1' + manager.change.after_unknown = { target_size: true } + assert.throws( + () => validateCapacityPlan({ resource_changes: [template, manager] }, cellConfig), + /outside the reviewed capacity fields/ + ) +}) + +test('same-cap mode preserves 1000/60 while adding only the reviewed trust config', () => { + const rollbackImage = `us-docker.pkg.dev/project/relay/image@sha256:${'d'.repeat(64)}` + const image = `us-docker.pkg.dev/project/relay/image@sha256:${'e'.repeat(64)}` + const directorIdentity = 'relay-director@project.iam.gserviceaccount.com' + const audience = 'https://relay.example.com/v1/admin/host-drain' + const startup = ({ selectedImage, cap = 1_000, trust = false }) => [ + ` printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\\n' '${cap}'`, + ` printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\\n' '60'`, + ...(trust ? [ + ` printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\\n' '${directorIdentity}'`, + ` printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${audience}'` + ] : []), + `printf 'ORCA_RELAY_IMAGE_DIGEST=%s\\n' '${selectedImage.split('@')[1]}'`, + `docker pull '${selectedImage}'`, + 'docker run --detach \\', + ' --name orca-relay \\', + ` '${selectedImage}'` + ].join('\n') + const template = { + address: 'google_compute_instance_template.relay_gce_cell["staging-gce-c3"]', + change: { + actions: ['create', 'delete'], + before: { metadata_startup_script: startup({ selectedImage: rollbackImage }) }, + after: { + metadata_startup_script: startup({ selectedImage: image, trust: true }), + self_link: null + }, + after_unknown: { self_link: true } + } + } + const manager = { + address: 'google_compute_instance_group_manager.relay_gce_cell["staging-gce-c3"]', + change: { + actions: ['update'], + before: { target_size: 1, version: [{ instance_template: 'old' }] }, + after: { target_size: 1, version: [{ instance_template: null }] }, + after_unknown: { version: [{ instance_template: true }] } + } + } + const sameCapConfig = { + ...config, + mode: 'same-cap-cell', + image, + rollbackImage, + rehomeDirectorServiceAccount: directorIdentity, + rehomeAudience: audience + } + assert.deepEqual( + validateCapacityPlan({ resource_changes: [template, manager] }, sameCapConfig), + { mode: 'same-cap-cell', changes: 2 } + ) + // A pre-template-apply rollback resume validates drift for the image the + // cell already serves: the template leaves and re-enters the rollback image. + const resumeTemplate = structuredClone(template) + resumeTemplate.change.after.metadata_startup_script = startup({ + selectedImage: rollbackImage, + trust: true + }) + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [resumeTemplate, manager] }, + { ...sameCapConfig, image: rollbackImage } + ), + { mode: 'same-cap-cell', changes: 2 } + ) + const asiaTemplate = structuredClone(template) + const asiaManager = structuredClone(manager) + asiaTemplate.address = + 'google_compute_instance_template.relay_gce_cell["production-gce-c28"]' + asiaManager.address = + 'google_compute_instance_group_manager.relay_gce_cell["production-gce-c28"]' + asiaTemplate.change.before.metadata_startup_script = startup({ + selectedImage: rollbackImage, + cap: 3_000 + }) + asiaTemplate.change.after.metadata_startup_script = startup({ + selectedImage: image, + cap: 3_000, + trust: true + }) + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [asiaTemplate, asiaManager] }, + { ...sameCapConfig, cellId: 'production-gce-c28', hardCap: 3_000 } + ), + { mode: 'same-cap-cell', changes: 2 } + ) + const changedCap = structuredClone(template) + changedCap.change.before.metadata_startup_script = startup({ + selectedImage: rollbackImage, + cap: 600 + }) + assert.throws( + () => validateCapacityPlan({ resource_changes: [changedCap, manager] }, sameCapConfig), + /reviewed image and capacity/ + ) + assert.throws( + () => validateCapacityPlan( + { resource_changes: [template, manager] }, + { ...sameCapConfig, rollbackImage: image } + ), + /reviewed image and capacity/ + ) + const wrongTrust = structuredClone(template) + wrongTrust.change.after.metadata_startup_script = startup({ + selectedImage: image, + trust: true + }).replace(directorIdentity, 'other-director@project.iam.gserviceaccount.com') + assert.throws( + () => validateCapacityPlan({ resource_changes: [wrongTrust, manager] }, sameCapConfig), + /reviewed image and capacity/ + ) + + const imageOnly = structuredClone(template) + imageOnly.change.after.metadata_startup_script = startup({ selectedImage: image }) + const imageOnlyConfig = { + ...config, + mode: 'same-cap-image', + image, + rollbackImage + } + assert.deepEqual( + validateCapacityPlan({ resource_changes: [imageOnly, manager] }, imageOnlyConfig), + { mode: 'same-cap-image', changes: 2, changeKind: 'replacement' } + ) + assert.throws( + () => validateCapacityPlan( + { resource_changes: [imageOnly, manager] }, + { ...imageOnlyConfig, rollbackImage: image } + ), + /reviewed image and capacity/ + ) + const changedTrust = structuredClone(imageOnly) + changedTrust.change.after.metadata_startup_script += + `\n printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\\n' '${audience}'` + assert.throws( + () => validateCapacityPlan({ resource_changes: [changedTrust, manager] }, imageOnlyConfig), + /reviewed image and capacity/ + ) + const plannedValues = { + root_module: { + resources: [ + { + address: imageOnly.address, + values: { + metadata_startup_script: imageOnly.change.after.metadata_startup_script, + self_link: 'projects/project/global/instanceTemplates/target' + } + }, + { + address: manager.address, + values: { + version: [{ instance_template: 'projects/project/global/instanceTemplates/target' }] + } + } + ] + } + } + const obsoleteTemplate = structuredClone(imageOnly) + obsoleteTemplate.deposed = 'obsolete' + obsoleteTemplate.change.actions = ['delete'] + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [obsoleteTemplate], planned_values: plannedValues }, + imageOnlyConfig + ), + { mode: 'same-cap-image', changes: 1, changeKind: 'obsolete-template-delete' } + ) + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [imageOnly, manager, obsoleteTemplate] }, + imageOnlyConfig + ), + { mode: 'same-cap-image', changes: 3, changeKind: 'replacement-with-obsolete-template' } + ) + const anotherObsoleteTemplate = { + ...structuredClone(obsoleteTemplate), + deposed: 'another-obsolete' + } + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [imageOnly, manager, obsoleteTemplate, anotherObsoleteTemplate] }, + imageOnlyConfig + ), + { mode: 'same-cap-image', changes: 4, changeKind: 'replacement-with-obsolete-template' } + ) + assert.deepEqual( + validateCapacityPlan( + { + resource_changes: [obsoleteTemplate, anotherObsoleteTemplate], + planned_values: plannedValues + }, + imageOnlyConfig + ), + { mode: 'same-cap-image', changes: 2, changeKind: 'obsolete-template-delete' } + ) + assert.deepEqual( + validateCapacityPlan( + { + resource_changes: [manager, obsoleteTemplate, anotherObsoleteTemplate], + planned_values: plannedValues + }, + imageOnlyConfig + ), + { mode: 'same-cap-image', changes: 3, changeKind: 'manager-convergence' } + ) + const invalidObsoleteTemplate = structuredClone(anotherObsoleteTemplate) + invalidObsoleteTemplate.change.actions = ['update'] + assert.throws( + () => validateCapacityPlan( + { resource_changes: [imageOnly, manager, obsoleteTemplate, invalidObsoleteTemplate] }, + imageOnlyConfig + ), + /change only the exact instance template and MIG/ + ) + assert.deepEqual( + validateCapacityPlan( + { resource_changes: [manager], planned_values: plannedValues }, + imageOnlyConfig + ), + { mode: 'same-cap-image', changes: 1, changeKind: 'manager-convergence' } + ) +}) diff --git a/cloud/dev/scripts/verify-relay-capacity-transition.mjs b/cloud/dev/scripts/verify-relay-capacity-transition.mjs new file mode 100644 index 00000000000..e5ebe77d45f --- /dev/null +++ b/cloud/dev/scripts/verify-relay-capacity-transition.mjs @@ -0,0 +1,473 @@ +import { pathToFileURL } from 'node:url' + +const CAPACITY_PROTOCOL = 2 + +function integer(value, name) { + const parsed = Number(value) + if (!Number.isSafeInteger(parsed) || parsed < 0) throw new Error(`${name} is invalid`) + return parsed +} + +function signedInteger(value, name) { + if (typeof value !== 'number' || !Number.isSafeInteger(value)) { + throw new Error(`${name} is invalid`) + } + return value +} + +export function parseCapacityTransitionArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of [ + 'director-origin', + 'cell-origin', + 'cell-id', + 'heartbeat', + 'admission', + 'draining', + 'activity' + ]) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!['fresh', 'stale', 'either'].includes(values.heartbeat)) { + throw new Error('--heartbeat must be fresh, stale, or either') + } + if ( + !['general', 'migration-only', 'general-or-migration-only', 'non-general', 'either'].includes( + values.admission + ) + ) { + throw new Error( + '--admission must be general, migration-only, general-or-migration-only, non-general, or either' + ) + } + if (!['required', 'forbidden', 'either'].includes(values.draining)) { + throw new Error('--draining must be required, forbidden, or either') + } + if (!['quiescent', 'restart-safe', 'allowed'].includes(values.activity)) { + throw new Error('--activity must be quiescent, restart-safe, or allowed') + } + const runtime = values.runtime ?? 'required' + if (!['required', 'unavailable'].includes(runtime)) { + throw new Error('--runtime must be required or unavailable') + } + if ( + values.activity === 'restart-safe' && + runtime === 'required' && + (values.admission !== 'migration-only' || values.draining !== 'required') + ) { + throw new Error('restart-safe activity requires migration-only admission and draining') + } + if ( + runtime === 'unavailable' && + (values.heartbeat !== 'stale' || + values.admission !== 'migration-only' || + values.draining !== 'either' || + values.activity !== 'restart-safe') + ) { + throw new Error('unavailable runtime requires stale migration-only durable state') + } + const origin = new URL(values['director-origin']) + const cellOrigin = new URL(values['cell-origin']) + if ( + origin.protocol !== 'https:' || + origin.origin !== values['director-origin'] || + cellOrigin.protocol !== 'https:' || + cellOrigin.origin !== values['cell-origin'] + ) { + throw new Error('origins must be canonical HTTPS origins') + } + const hardCap = values['hard-cap'] === undefined + ? undefined + : integer(values['hard-cap'], '--hard-cap') + const unobservedBound = values['unobserved-bound'] === undefined + ? undefined + : integer(values['unobserved-bound'], '--unobserved-bound') + if ((hardCap === undefined) !== (unobservedBound === undefined)) { + throw new Error('capacity expectations must be paired') + } + if (runtime === 'unavailable' && hardCap !== undefined) { + throw new Error('unavailable runtime cannot prove live capacity') + } + const expectedImageDigests = values['expected-image-digests']?.split(',') ?? [] + if ( + new Set(expectedImageDigests).size !== expectedImageDigests.length || + expectedImageDigests.some((digest) => !/^sha256:[a-f0-9]{64}$/.test(digest)) + ) { + throw new Error('--expected-image-digests is invalid') + } + if (runtime === 'unavailable' && expectedImageDigests.length > 0) { + throw new Error('unavailable runtime cannot prove a live image') + } + const regionalRehomeProtocol = values['regional-rehome-protocol'] === undefined + ? undefined + : integer(values['regional-rehome-protocol'], '--regional-rehome-protocol') + if (regionalRehomeProtocol !== undefined && ![0, 1].includes(regionalRehomeProtocol)) { + throw new Error('--regional-rehome-protocol must be 0 or 1') + } + if (runtime === 'unavailable' && regionalRehomeProtocol !== undefined) { + throw new Error('unavailable runtime cannot prove the regional rehome protocol') + } + return { + directorOrigin: origin.origin, + cellOrigin: cellOrigin.origin, + cellId: values['cell-id'], + heartbeat: values.heartbeat, + admission: values.admission, + draining: values.draining, + activity: values.activity, + runtime, + expectedImageDigests, + ...(regionalRehomeProtocol === undefined ? {} : { regionalRehomeProtocol }), + hardCap, + unobservedBound, + timeoutMs: integer(values['timeout-ms'] ?? 180_000, '--timeout-ms') + } +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({})) + if (!response.ok) throw new Error(`${label} returned ${response.status}`) + return body +} + +async function cellRuntime(fetchImpl, config, token) { + let response + try { + response = await fetchImpl(`${config.cellOrigin}/v1/admin/runtime-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1 }), + signal: AbortSignal.timeout(30_000) + }) + } catch (error) { + if (config.runtime === 'unavailable') return null + throw error + } + if ([502, 503, 504].includes(response.status)) { + await response.arrayBuffer().catch(() => undefined) + return null + } + return await responseJson(response, 'cell runtime status') +} + +function offlineRollbackMatches(status, config) { + if (status.admissionState !== 'migration-only') { + throw new Error('capacity transition admission does not match the required state') + } + const durableCounts = [ + status.activityLeases, + status.activityRequestUnits, + status.reservedRequests, + status.restartBlockingActivityLeases, + status.restartBlockingActivityRequestUnits, + status.outgoingMigrations, + status.incomingMigrations, + status.connectionCapacity?.pendingControlReservations + ] + const restartBlockingReservedRequests = signedInteger( + status.restartBlockingReservedRequests, + 'restart-blocking reserved requests' + ) + // Only a positive remainder is unexplained; the other gates reject real work. + return heartbeatMatches(status, config.heartbeat) && + durableCounts.every((value) => integer(value, 'durable activity count') === 0) && + restartBlockingReservedRequests <= 0 +} + +function directorActivityMatches(status, config, restartBlockingReservedRequests) { + const durable = [ + status.activityLeases, + status.reservedRequests, + status.outgoingMigrations, + status.incomingMigrations + ] + // Reconnect reservations survive replacement; draining prevents activation on this process. + const transient = [ + status.connectionCapacity?.observedConnections, + status.connectionCapacity?.inFlightConnections, + status.connectionCapacity?.reservedConnectionUnits, + status.connectionCapacity?.enforcedConnectionUnits, + status.connectionCapacity?.pendingControlReservations + ] + const restartSafe = config.activity !== 'restart-safe' || (() => { + // Only a positive remainder is unexplained; the other gates reject real work. + return integer( + status.restartBlockingActivityLeases, + 'restart-blocking activity leases' + ) === 0 && + integer( + status.restartBlockingActivityRequestUnits, + 'restart-blocking activity request units' + ) === 0 && + restartBlockingReservedRequests <= 0 && + integer(status.outgoingMigrations, 'outgoing migrations') === 0 && + integer(status.incomingMigrations, 'incoming migrations') === 0 + })() + const quiescent = [...durable, ...transient] + .filter((value) => value !== undefined) + .every((value) => integer(value, 'activity count') === 0) + if ( + (config.admission === 'general' && status.admissionState !== 'general') || + (config.admission === 'migration-only' && status.admissionState !== 'migration-only') || + // A failed same-cap canary leaves its cell migration-only; the documented + // rollback recovery must accept that state alongside a completed general roll. + (config.admission === 'general-or-migration-only' && + !['general', 'migration-only'].includes(status.admissionState)) || + (config.admission === 'non-general' && + !['existing-only', 'migration-only'].includes(status.admissionState)) + ) { + throw new Error('capacity transition admission does not match the required state') + } + return config.activity === 'allowed' || + (config.activity === 'restart-safe' ? restartSafe : quiescent) +} + +function capacityMatches(status, config) { + if (config.hardCap === undefined) return true + const capacity = status.connectionCapacity + return ( + capacity?.hardCap === config.hardCap && + capacity.unobservedBound === config.unobservedBound && + capacity.controlRebindReserve === 100 && + capacity.ordinaryConnectionLimit === config.hardCap - 100 && + capacity.normalAdmissionPause === config.hardCap - 100 - config.unobservedBound + ) +} + +function heartbeatMatches(status, expectation) { + if (expectation === 'either') return true + const fresh = status.connectionCapacity?.heartbeatFresh ?? status.runtime?.heartbeatFresh + return fresh === (expectation === 'fresh') +} + +function aggregateCount(value) { + const parsed = Number(value) + return Number.isSafeInteger(parsed) && parsed >= 0 ? parsed : null +} + +function signedAggregateCount(value) { + return typeof value === 'number' && Number.isSafeInteger(value) ? value : null +} + +function capacityObservation(capacity) { + if (capacity === null || capacity === undefined) return null + return { + hardCap: aggregateCount(capacity.hardCap), + controlRebindReserve: aggregateCount(capacity.controlRebindReserve), + ordinaryConnectionLimit: aggregateCount(capacity.ordinaryConnectionLimit), + unobservedBound: aggregateCount(capacity.unobservedBound), + normalAdmissionPause: aggregateCount(capacity.normalAdmissionPause), + observedConnections: aggregateCount(capacity.observedConnections), + inFlightConnections: aggregateCount(capacity.inFlightConnections), + reservedConnectionUnits: aggregateCount(capacity.reservedConnectionUnits), + enforcedConnectionUnits: aggregateCount(capacity.enforcedConnectionUnits), + pendingControlReservations: aggregateCount(capacity.pendingControlReservations), + heartbeatFresh: typeof capacity.heartbeatFresh === 'boolean' + ? capacity.heartbeatFresh + : null + } +} + +function transitionObservation(runtime, status) { + return { + runtimeAvailable: runtime !== null, + admissionState: ['general', 'migration-only', 'existing-only'].includes(status.admissionState) + ? status.admissionState + : null, + draining: typeof runtime?.draining === 'boolean' ? runtime.draining : null, + runtime: runtime === null + ? null + : { + totalConnections: aggregateCount(runtime.runtime?.totalConnections), + preAuthConnections: aggregateCount(runtime.runtime?.preAuthConnections), + inFlightConnections: aggregateCount(runtime.runtime?.inFlightConnections), + reservedConnectionUnits: aggregateCount(runtime.runtime?.reservedConnectionUnits), + enforcedConnectionUnits: aggregateCount(runtime.runtime?.enforcedConnectionUnits), + controls: aggregateCount(runtime.runtime?.controls), + splices: aggregateCount(runtime.runtime?.splices), + pendingSplices: aggregateCount(runtime.runtime?.pendingSplices), + queuedBytes: aggregateCount(runtime.runtime?.queuedBytes) + }, + director: { + activityLeases: aggregateCount(status.activityLeases), + activityRequestUnits: aggregateCount(status.activityRequestUnits), + reservedRequests: aggregateCount(status.reservedRequests), + restartBlockingActivityLeases: aggregateCount(status.restartBlockingActivityLeases), + restartBlockingActivityRequestUnits: + aggregateCount(status.restartBlockingActivityRequestUnits), + restartBlockingReservedRequests: + signedAggregateCount(status.restartBlockingReservedRequests), + outgoingMigrations: aggregateCount(status.outgoingMigrations), + incomingMigrations: aggregateCount(status.incomingMigrations) + }, + runtimeCapacity: capacityObservation(runtime?.connectionCapacity), + directorCapacity: capacityObservation(status.connectionCapacity), + runtimeHeartbeatFresh: typeof status.runtime?.heartbeatFresh === 'boolean' + ? status.runtime.heartbeatFresh + : null + } +} + +function runtimeQuiescent(runtime, config) { + if ( + runtime.role !== 'cell' || + runtime.cellId !== config.cellId || + runtime.cellUrl !== config.cellOrigin || + // Legacy pre-rehome images omit the field; the exact digest binds absence to protocol 0. + (config.regionalRehomeProtocol !== undefined && + (runtime.regionalRehomeProtocol ?? 0) !== config.regionalRehomeProtocol) || + (config.expectedImageDigests?.length > 0 && + !config.expectedImageDigests.includes(runtime.imageDigest)) + ) { + throw new Error('capacity transition runtime does not match the cell') + } + if ( + (config.draining === 'required' && runtime.draining !== true) || + (config.draining === 'forbidden' && runtime.draining === true) + ) { + return false + } + const counts = [runtime.runtime?.totalConnections, runtime.runtime?.preAuthConnections] + if (counts.some((value) => value === undefined)) { + throw new Error('capacity transition runtime is incomplete') + } + if (runtime.connectionCapacity !== null && runtime.connectionCapacity !== undefined) { + if (runtime.runtime?.enforcedConnectionUnits === undefined) { + throw new Error('capacity transition runtime is incomplete') + } + counts.push(runtime.runtime.enforcedConnectionUnits) + } + const quiescent = counts.every((value) => integer(value, 'runtime connection count') === 0) + const restartSafe = config.activity !== 'restart-safe' || + [ + runtime.runtime?.preAuthConnections, + runtime.runtime?.inFlightConnections, + runtime.runtime?.reservedConnectionUnits, + runtime.runtime?.controls, + runtime.runtime?.splices, + runtime.runtime?.pendingSplices, + runtime.runtime?.queuedBytes + ].every((value) => integer(value, 'live runtime count') === 0) + if ( + config.heartbeat === 'fresh' && + !capacityMatches({ connectionCapacity: runtime.connectionCapacity }, config) + ) { + return false + } + return config.activity === 'allowed' || + (config.activity === 'restart-safe' ? restartSafe : quiescent) +} + +export async function verifyCapacityTransition(config, overrides = {}) { + if ( + config.activity === 'restart-safe' && + config.runtime !== 'unavailable' && + (config.admission !== 'migration-only' || config.draining !== 'required') + ) { + throw new Error('restart-safe activity requires migration-only admission and draining') + } + const fetchImpl = overrides.fetch ?? fetch + const wait = overrides.wait ?? ((ms) => new Promise((resolve) => setTimeout(resolve, ms))) + const now = overrides.now ?? Date.now + const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN + if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') + const health = await responseJson( + await fetchImpl(`${config.directorOrigin}/health`, { + signal: AbortSignal.timeout(15_000) + }), + 'director health' + ) + if (health.ok !== true || health.connectionCapacityProtocol !== CAPACITY_PROTOCOL) { + throw new Error('director is not capacity-protocol compatible') + } + const deadline = now() + config.timeoutMs + let restartSafeSamples = 0 + let lastObservation = { runtimeAvailable: false } + for (;;) { + const runtime = await cellRuntime(fetchImpl, config, token) + lastObservation = { runtimeAvailable: runtime !== null } + if ((runtime === null) === (config.runtime === 'unavailable')) { + const result = await responseJson( + await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, { + method: 'POST', + headers: { authorization: `Bearer ${token}`, 'content-type': 'application/json' }, + body: JSON.stringify({ v: 1, cellId: config.cellId }), + signal: AbortSignal.timeout(30_000) + }), + 'cell status' + ) + const status = result.status + if ( + status?.cellId !== config.cellId || + status.cellUrl !== config.cellOrigin || + status.runtime?.cellUrl !== config.cellOrigin + ) { + throw new Error('capacity transition director status does not match the cell') + } + lastObservation = transitionObservation(runtime, status) + const restartBlockingReservedRequests = + config.activity === 'restart-safe' + ? signedInteger( + status.restartBlockingReservedRequests, + 'restart-blocking reserved requests' + ) + : null + const matches = runtime === null + ? offlineRollbackMatches(status, config) + : runtimeQuiescent(runtime, config) && + directorActivityMatches(status, config, restartBlockingReservedRequests) && + capacityMatches(status, config) && + heartbeatMatches(status, config.heartbeat) + if (matches && (config.activity !== 'restart-safe' || restartSafeSamples === 1)) { + return { + cellId: status.cellId, + admissionState: status.admissionState, + assignments: integer(status.assignments, 'assignments'), + hardCap: runtime === null ? null : status.connectionCapacity?.hardCap ?? null, + unobservedBound: + runtime === null ? null : status.connectionCapacity?.unobservedBound ?? null, + heartbeatFresh: + status.connectionCapacity?.heartbeatFresh ?? status.runtime?.heartbeatFresh ?? false, + imageDigest: runtime?.imageDigest ?? null, + ...(config.activity === 'restart-safe' + ? { restartBlockingReservedRequests } + : {}) + } + } + restartSafeSamples = matches ? 1 : 0 + } else { + restartSafeSamples = 0 + } + if (config.activity === 'restart-safe') { + lastObservation = { + ...lastObservation, + restartSafeSamples, + requiredRestartSafeSamples: 2 + } + } + if (now() >= deadline) { + throw new Error( + `capacity transition verification timed out: ${JSON.stringify(lastObservation)}` + ) + } + await wait(5_000) + } +} + +export async function main(argv = process.argv.slice(2)) { + const result = await verifyCapacityTransition(parseCapacityTransitionArguments(argv)) + process.stdout.write(`${JSON.stringify({ event: 'relay_capacity_transition_verified', ...result })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs b/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs new file mode 100644 index 00000000000..fb865257c4a --- /dev/null +++ b/cloud/dev/scripts/verify-relay-capacity-transition.test.mjs @@ -0,0 +1,1096 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { + parseCapacityTransitionArguments, + verifyCapacityTransition +} from './verify-relay-capacity-transition.mjs' + +const imageDigest = `sha256:${'a'.repeat(64)}` +const config = { + directorOrigin: 'https://relay.example.com', + cellOrigin: 'https://c3.relay.example.com', + cellId: 'staging-gce-c3', + heartbeat: 'fresh', + admission: 'non-general', + draining: 'required', + activity: 'quiescent', + expectedImageDigests: [imageDigest], + hardCap: 1_000, + unobservedBound: 60, + timeoutMs: 1 +} + +function response(body) { + return Response.json(body) +} + +function harness({ + heartbeatFresh = true, + active = 0, + preAuthConnections = 0, + inFlightConnections = 0, + reservedConnectionUnits = 0, + enforcedConnectionUnits = active, + activityLeases = active, + activityRequestUnits = activityLeases, + reservedRequests = activityRequestUnits, + restartBlockingActivityLeases = activityLeases, + restartBlockingActivityRequestUnits = activityRequestUnits, + restartBlockingReservedRequests = reservedRequests, + outgoingMigrations = 0, + incomingMigrations = 0, + controls = 0, + splices = 0, + pendingSplices = 0, + queuedBytes = 0, + pendingControlReservations = 0, + runtimeHardCap = 1_000, + directorHardCap = 1_000, + runtimeImageDigest = imageDigest, + protocol = 2, + regionalRehomeProtocol = 1, + legacy = false, + admission = config.admission, + draining = config.draining +} = {}) { + return async (url) => { + const parsed = new URL(url) + if (parsed.pathname === '/health') { + return response({ ok: true, connectionCapacityProtocol: protocol }) + } + if (parsed.pathname === '/v1/admin/runtime-status') { + return response({ + role: 'cell', + cellId: config.cellId, + cellUrl: config.cellOrigin, + imageDigest: runtimeImageDigest, + ...(regionalRehomeProtocol === null ? {} : { regionalRehomeProtocol }), + draining: draining === 'required', + connectionCapacity: legacy + ? null + : { + hardCap: runtimeHardCap, + unobservedBound: 60, + controlRebindReserve: 100, + ordinaryConnectionLimit: runtimeHardCap - 100, + normalAdmissionPause: runtimeHardCap - 160 + }, + runtime: { + totalConnections: active, + preAuthConnections, + controls, + splices, + pendingSplices, + queuedBytes, + ...(legacy ? {} : { + inFlightConnections, + reservedConnectionUnits, + enforcedConnectionUnits + }) + } + }) + } + return response({ + status: { + cellId: config.cellId, + cellUrl: config.cellOrigin, + admissionState: admission === 'non-general' ? 'migration-only' : admission, + runtime: { heartbeatFresh, cellUrl: config.cellOrigin }, + assignments: 900, + activityLeases, + activityRequestUnits, + restartBlockingActivityLeases, + restartBlockingActivityRequestUnits, + restartBlockingReservedRequests, + reservedRequests, + outgoingMigrations, + incomingMigrations, + connectionCapacity: { + hardCap: directorHardCap, + unobservedBound: 60, + controlRebindReserve: 100, + ordinaryConnectionLimit: directorHardCap - 100, + normalAdmissionPause: directorHardCap - 160, + observedConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + pendingControlReservations, + heartbeatFresh + } + } + }) + } +} + +test('parses a paired reviewed capacity', () => { + assert.deepEqual( + parseCapacityTransitionArguments([ + '--director-origin', 'https://relay.example.com', + '--cell-origin', 'https://c3.relay.example.com', + '--cell-id', 'staging-gce-c3', + '--heartbeat', 'stale', + '--admission', 'migration-only', + '--draining', 'required', + '--activity', 'restart-safe', + '--expected-image-digests', imageDigest, + '--hard-cap', '1000', + '--unobserved-bound', '60' + ]), + { + ...config, + heartbeat: 'stale', + admission: 'migration-only', + activity: 'restart-safe', + runtime: 'required', + expectedImageDigests: [imageDigest], + timeoutMs: 180_000 + } + ) +}) + +test('accepts either exact predecessor image without weakening the live state checks', async () => { + const predecessor = `sha256:${'b'.repeat(64)}` + const compatible = `sha256:${'c'.repeat(64)}` + const predecessorConfig = parseCapacityTransitionArguments([ + '--director-origin', 'https://relay.example.com', + '--cell-origin', 'https://c3.relay.example.com', + '--cell-id', 'staging-gce-c3', + '--heartbeat', 'fresh', + '--admission', 'general', + '--draining', 'forbidden', + '--activity', 'allowed', + '--expected-image-digests', `${predecessor},${compatible}`, + '--hard-cap', '600', + '--unobserved-bound', '60' + ]) + assert.deepEqual(predecessorConfig.expectedImageDigests, [predecessor, compatible]) + assert.deepEqual( + { + heartbeat: predecessorConfig.heartbeat, + admission: predecessorConfig.admission, + draining: predecessorConfig.draining, + hardCap: predecessorConfig.hardCap, + unobservedBound: predecessorConfig.unobservedBound + }, + { + heartbeat: 'fresh', + admission: 'general', + draining: 'forbidden', + hardCap: 600, + unobservedBound: 60 + } + ) + const liveState = { + runtimeHardCap: 600, + directorHardCap: 600, + runtimeImageDigest: compatible, + admission: 'general', + draining: 'forbidden' + } + assert.equal( + (await verifyCapacityTransition(predecessorConfig, { + fetch: harness(liveState), + token: 'masked-token' + })).imageDigest, + compatible + ) + await assert.rejects( + verifyCapacityTransition(predecessorConfig, { + fetch: harness({ + ...liveState, + runtimeImageDigest: `sha256:${'d'.repeat(64)}` + }), + token: 'masked-token' + }), + /runtime does not match the cell/ + ) +}) + +test('requires the exact regional rehome protocol when requested', async () => { + const rehomeConfig = { + ...config, + regionalRehomeProtocol: 1 + } + await verifyCapacityTransition(rehomeConfig, { + token: 'token', + fetch: harness({ regionalRehomeProtocol: 1 }) + }) + await assert.rejects( + verifyCapacityTransition(rehomeConfig, { + token: 'token', + fetch: harness({ regionalRehomeProtocol: 0 }), + now: () => 1, + wait: async () => undefined + }), + /runtime does not match/ + ) + assert.equal( + parseCapacityTransitionArguments([ + '--director-origin', 'https://relay.example.com', + '--cell-origin', 'https://c3.relay.example.com', + '--cell-id', 'staging-gce-c3', + '--heartbeat', 'fresh', + '--admission', 'general', + '--draining', 'forbidden', + '--activity', 'allowed', + '--regional-rehome-protocol', '1' + ]).regionalRehomeProtocol, + 1 + ) +}) + +test('binds an absent rehome protocol to 0 for legacy pre-rehome images', async () => { + // A rolled-back cell runs an image that omits the field entirely; the + // documented contract binds absence to protocol 0. + await verifyCapacityTransition( + { ...config, regionalRehomeProtocol: 0 }, + { token: 'token', fetch: harness({ regionalRehomeProtocol: null }) } + ) + await verifyCapacityTransition( + { ...config, regionalRehomeProtocol: 0 }, + { token: 'token', fetch: harness({ regionalRehomeProtocol: 0 }) } + ) + await assert.rejects( + verifyCapacityTransition( + { ...config, regionalRehomeProtocol: 1 }, + { + token: 'token', + fetch: harness({ regionalRehomeProtocol: null }), + now: () => 1, + wait: async () => undefined + } + ), + /runtime does not match/ + ) +}) + +test('restart-safe mode requires the exact isolated drain state', () => { + assert.throws( + () => parseCapacityTransitionArguments([ + '--director-origin', 'https://relay.example.com', + '--cell-origin', 'https://c3.relay.example.com', + '--cell-id', 'staging-gce-c3', + '--heartbeat', 'either', + '--admission', 'general', + '--draining', 'required', + '--activity', 'restart-safe' + ]), + /requires migration-only admission and draining/ + ) +}) + +test('offline rollback requires two stale zero-durable-activity samples', async () => { + const offlineConfig = { + ...config, + heartbeat: 'stale', + admission: 'migration-only', + draining: 'either', + activity: 'restart-safe', + runtime: 'unavailable', + expectedImageDigests: [], + hardCap: undefined, + unobservedBound: undefined, + timeoutMs: 10_000 + } + const base = harness({ + heartbeatFresh: false, + admission: 'migration-only', + draining: 'forbidden', + activityLeases: 0, + activityRequestUnits: 0, + reservedRequests: 0, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: -1 + }) + let runtimeReads = 0 + let directorReads = 0 + let now = 0 + const result = await verifyCapacityTransition(offlineConfig, { + fetch: async (url, options) => { + const pathname = new URL(url).pathname + if (pathname === '/v1/admin/runtime-status') { + runtimeReads += 1 + return Response.json({ error: 'backend_unavailable' }, { status: 503 }) + } + if (pathname === '/v1/admin/cell-status') directorReads += 1 + return await base(url, options) + }, + token: 'masked-token', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + }) + assert.equal(result.cellId, config.cellId) + assert.equal(result.hardCap, null) + assert.equal(result.restartBlockingReservedRequests, -1) + assert.equal(runtimeReads, 2) + assert.equal(directorReads, 2) +}) + +test('offline rollback rejects a reachable cell or durable work', async () => { + const offlineConfig = { + ...config, + heartbeat: 'stale', + admission: 'migration-only', + draining: 'either', + activity: 'restart-safe', + runtime: 'unavailable', + expectedImageDigests: [], + hardCap: undefined, + unobservedBound: undefined, + timeoutMs: 0 + } + await assert.rejects( + verifyCapacityTransition(offlineConfig, { + fetch: harness({ heartbeatFresh: false, admission: 'migration-only' }), + token: 'masked-token' + }), + /timed out/ + ) + const active = harness({ + heartbeatFresh: false, + admission: 'migration-only', + activityLeases: 1 + }) + await assert.rejects( + verifyCapacityTransition(offlineConfig, { + fetch: async (url, options) => + new URL(url).pathname === '/v1/admin/runtime-status' + ? Response.json({ error: 'backend_unavailable' }, { status: 503 }) + : await active(url, options), + token: 'masked-token' + }), + /timed out/ + ) + await assert.rejects( + verifyCapacityTransition(offlineConfig, { + fetch: async (url, options) => { + if (new URL(url).pathname === '/v1/admin/runtime-status') { + return Response.json({ error: 'backend_unavailable' }, { status: 503 }) + } + const result = await active(url, options) + if (new URL(url).pathname !== '/v1/admin/cell-status') return result + const body = await result.json() + body.status.cellId = 'production-gce-other' + return response(body) + }, + token: 'masked-token' + }), + /does not match the cell/ + ) +}) + +test('offline rollback arguments cannot claim live capacity', () => { + assert.throws( + () => parseCapacityTransitionArguments([ + '--director-origin', 'https://relay.example.com', + '--cell-origin', 'https://c3.relay.example.com', + '--cell-id', 'staging-gce-c3', + '--heartbeat', 'stale', + '--admission', 'migration-only', + '--draining', 'either', + '--activity', 'restart-safe', + '--runtime', 'unavailable', + '--hard-cap', '600', + '--unobserved-bound', '60' + ]), + /cannot prove live capacity/ + ) +}) + +test('accepts a quiescent matching cell without exposing assignments', async () => { + assert.deepEqual( + await verifyCapacityTransition(config, { + fetch: harness(), + token: 'masked-token' + }), + { + cellId: config.cellId, + admissionState: 'migration-only', + assignments: 900, + hardCap: 1_000, + unobservedBound: 60, + heartbeatFresh: true, + imageDigest + } + ) +}) + +test('migration-only admission rejects the irreversible existing-only state', async () => { + const migrationConfig = { ...config, admission: 'migration-only' } + await verifyCapacityTransition(migrationConfig, { + fetch: harness({ admission: 'migration-only' }), + token: 'masked-token' + }) + await assert.rejects( + verifyCapacityTransition(migrationConfig, { + fetch: harness({ admission: 'existing-only' }), + token: 'masked-token' + }), + /admission does not match/ + ) +}) + +test('requires the exact runtime image and both cell origins', async () => { + await assert.rejects( + verifyCapacityTransition(config, { + fetch: harness({ runtimeImageDigest: `sha256:${'b'.repeat(64)}` }), + token: 'masked-token' + }), + /runtime does not match the cell/ + ) + const base = harness() + for (const location of ['runtime', 'director']) { + await assert.rejects( + verifyCapacityTransition(config, { + fetch: async (url, options) => { + const result = await base(url, options) + const path = new URL(url).pathname + if ( + (location === 'runtime' && path !== '/v1/admin/runtime-status') || + (location === 'director' && path !== '/v1/admin/cell-status') + ) return result + const body = await result.json() + if (location === 'runtime') body.cellUrl = 'https://other.example.com' + else body.status.cellUrl = 'https://other.example.com' + return response(body) + }, + token: 'masked-token' + }), + /does not match the cell/ + ) + } +}) + +test('accepts a quiescent legacy runtime before its first cap transition', async () => { + assert.equal( + ( + await verifyCapacityTransition( + { ...config, hardCap: undefined, unobservedBound: undefined, heartbeat: 'either' }, + { fetch: harness({ legacy: true }), token: 'masked-token' } + ) + ).cellId, + config.cellId + ) +}) + +test('uses the legacy runtime heartbeat when capacity telemetry is not registered', async () => { + const result = await verifyCapacityTransition( + { + ...config, + hardCap: undefined, + unobservedBound: undefined, + heartbeat: 'fresh', + draining: 'forbidden', + activity: 'allowed' + }, + { fetch: harness({ legacy: true, draining: 'forbidden' }), token: 'masked-token' } + ) + assert.equal(result.heartbeatFresh, true) +}) + +test('rejects old directors and active cells', async () => { + await assert.rejects( + verifyCapacityTransition(config, { fetch: harness({ protocol: 1 }), token: 'masked' }), + /not capacity-protocol compatible/ + ) + let now = 0 + await assert.rejects( + verifyCapacityTransition(config, { + fetch: harness({ active: 1 }), + token: 'masked', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + }), + /timed out/ + ) +}) + +test('accepts active controls only when the transition explicitly allows them', async () => { + const activeConfig = { + ...config, + admission: 'general', + draining: 'forbidden', + activity: 'allowed' + } + const result = await verifyCapacityTransition(activeConfig, { + fetch: harness({ active: 900, admission: 'general', draining: 'forbidden' }), + token: 'masked' + }) + assert.equal(result.admissionState, 'general') +}) + +test('accepts only rejected reconnect traffic at the restart gate', async () => { + const restartConfig = { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 10_000 + } + const reconnecting = harness({ + active: 10, + activityLeases: 839, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 0, + pendingControlReservations: 839 + }) + let now = 0 + let runtimeReads = 0 + const fetch = async (url, options) => { + if (new URL(url).pathname === '/v1/admin/runtime-status') runtimeReads += 1 + return await reconnecting(url, options) + } + assert.equal((await verifyCapacityTransition(restartConfig, { + fetch, + token: 'masked-token', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + })).cellId, config.cellId) + assert.equal(runtimeReads, 2) + await assert.rejects( + verifyCapacityTransition({ ...config, timeoutMs: 0 }, { + fetch: reconnecting, + token: 'masked-token' + }), + /timed out/ + ) +}) + +test('restart-safe settling resets after data-plane admission appears', async () => { + const restartConfig = { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 20_000 + } + const safe = harness({ active: 1, activityLeases: 0 }) + const unsafe = harness({ active: 1, activityLeases: 0, preAuthConnections: 1 }) + let runtimeReads = 0 + let now = 0 + const result = await verifyCapacityTransition(restartConfig, { + fetch: async (url, options) => { + if (new URL(url).pathname !== '/v1/admin/runtime-status') { + return await safe(url, options) + } + runtimeReads += 1 + return await (runtimeReads === 2 ? unsafe : safe)(url, options) + }, + token: 'masked-token', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + }) + assert.equal(result.cellId, config.cellId) + assert.equal(runtimeReads, 4) +}) + +test('restart-safe settling resets after director activity appears', async () => { + const restartConfig = { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 20_000 + } + const safe = harness({ + activityLeases: 839, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 0 + }) + const unsafe = harness({ + activityLeases: 840, + restartBlockingActivityLeases: 1, + restartBlockingActivityRequestUnits: 1, + restartBlockingReservedRequests: 1 + }) + let directorReads = 0 + let runtimeReads = 0 + let now = 0 + const result = await verifyCapacityTransition(restartConfig, { + fetch: async (url, options) => { + if (new URL(url).pathname === '/v1/admin/runtime-status') runtimeReads += 1 + if (new URL(url).pathname !== '/v1/admin/cell-status') { + return await safe(url, options) + } + directorReads += 1 + return await (directorReads === 2 ? unsafe : safe)(url, options) + }, + token: 'masked-token', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + }) + assert.equal(result.cellId, config.cellId) + assert.equal(runtimeReads, 4) +}) + +test('restart gate rejects malformed restart aggregates', async (t) => { + const restartConfig = { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 0 + } + for (const field of [ + 'restartBlockingActivityLeases', + 'restartBlockingActivityRequestUnits' + ]) { + for (const value of [undefined, -1]) { + await t.test(`${field} ${value === undefined ? 'missing' : 'negative'}`, async () => { + const base = harness({ [field]: value }) + const fetch = value === undefined + ? async (url, options) => { + const result = await base(url, options) + if (new URL(url).pathname !== '/v1/admin/cell-status') return result + const body = await result.json() + delete body.status[field] + return response(body) + } + : base + await assert.rejects( + verifyCapacityTransition(restartConfig, { + fetch, + token: 'masked-token' + }), + /is invalid/ + ) + }) + } + } + for (const value of [undefined, null, '0', false]) { + await t.test(`restartBlockingReservedRequests ${String(value)}`, async () => { + const base = harness({ + preAuthConnections: 1, + restartBlockingActivityLeases: 1, + restartBlockingReservedRequests: value + }) + await assert.rejects( + verifyCapacityTransition(restartConfig, { + fetch: async (url, options) => { + const result = await base(url, options) + if ( + value !== undefined || + new URL(url).pathname !== '/v1/admin/cell-status' + ) return result + const body = await result.json() + delete body.status.restartBlockingReservedRequests + return response(body) + }, + token: 'masked-token' + }), + /is invalid/ + ) + }) + } +}) + +test('offline rollback validates restart reservation accounting before other blockers', async (t) => { + const offlineConfig = { + ...config, + heartbeat: 'stale', + admission: 'migration-only', + draining: 'either', + activity: 'restart-safe', + runtime: 'unavailable', + expectedImageDigests: [], + hardCap: undefined, + unobservedBound: undefined, + timeoutMs: 0 + } + for (const value of [undefined, null, '0', false]) { + await t.test(String(value), async () => { + const base = harness({ + heartbeatFresh: false, + activityLeases: 1, + restartBlockingReservedRequests: value + }) + await assert.rejects( + verifyCapacityTransition(offlineConfig, { + fetch: async (url, options) => { + const pathname = new URL(url).pathname + if (pathname === '/v1/admin/runtime-status') { + return Response.json({ error: 'backend_unavailable' }, { status: 503 }) + } + const result = await base(url, options) + if (value !== undefined || pathname !== '/v1/admin/cell-status') return result + const body = await result.json() + delete body.status.restartBlockingReservedRequests + return response(body) + }, + token: 'masked-token' + }), + /is invalid/ + ) + }) + } +}) + +test('negative restart reservation accounting is safe and remains observable', async () => { + const result = await verifyCapacityTransition( + { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 10_000 + }, + { + fetch: harness({ restartBlockingReservedRequests: -1 }), + token: 'masked-token', + now: () => 0, + wait: async () => undefined + } + ) + assert.equal(result.restartBlockingReservedRequests, -1) +}) + +test('negative restart reservation accounting cannot mask real work', async () => { + await assert.rejects( + verifyCapacityTransition( + { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 0 + }, + { + fetch: harness({ + restartBlockingActivityLeases: 1, + restartBlockingReservedRequests: -1 + }), + token: 'masked-token' + } + ), + /"restartBlockingReservedRequests":-1/ + ) +}) + +test('restart gate rejects live or durable cell work', async (t) => { + const restartConfig = { + ...config, + admission: 'migration-only', + activity: 'restart-safe', + timeoutMs: 0 + } + const unsafeStates = [ + ['control', { active: 1, activityLeases: 0, controls: 1 }], + ['splice', { active: 1, activityLeases: 0, splices: 1 }], + ['pending splice', { active: 1, activityLeases: 0, pendingSplices: 1 }], + ['queued data', { active: 1, activityLeases: 0, queuedBytes: 1 }], + ['pre-auth connection', { active: 1, activityLeases: 0, preAuthConnections: 1 }], + ['in-flight connection', { + activityLeases: 0, + inFlightConnections: 1, + enforcedConnectionUnits: 1 + }], + ['reserved data unit', { + active: 1, + activityLeases: 0, + reservedConnectionUnits: 1, + enforcedConnectionUnits: 2 + }], + ['activity lease', { activityLeases: 1 }], + ['misaccounted pending control lease', { + activityLeases: 1, + activityRequestUnits: 2, + reservedRequests: 2, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 1, + restartBlockingReservedRequests: 1 + }], + ['cell reservation', { reservedRequests: 1 }], + ['outgoing migration', { outgoingMigrations: 1 }], + ['incoming migration', { incomingMigrations: 1 }] + ] + for (const [name, state] of unsafeStates) { + await t.test(name, async () => { + await assert.rejects( + verifyCapacityTransition(restartConfig, { + fetch: harness(state), + token: 'masked-token' + }), + /timed out/ + ) + }) + } +}) + +test('restart timeout reports only aggregate blockers', async () => { + const base = harness({ + controls: 2, + restartBlockingActivityLeases: 3, + restartBlockingActivityRequestUnits: 4, + restartBlockingReservedRequests: 5, + incomingMigrations: 6 + }) + await assert.rejects( + verifyCapacityTransition( + { + ...config, + admission: 'migration-only', + draining: 'required', + activity: 'restart-safe', + timeoutMs: 0 + }, + { fetch: base, token: 'masked-token' } + ), + (error) => { + assert.match(error.message, /"controls":2/) + assert.match(error.message, /"restartBlockingActivityLeases":3/) + assert.match(error.message, /"incomingMigrations":6/) + assert.doesNotMatch(error.message, /user|host|secret|token/i) + return true + } + ) +}) + +test('restart timeout reports one of two safe settling samples', async () => { + const base = harness() + await assert.rejects( + verifyCapacityTransition( + { + ...config, + admission: 'migration-only', + draining: 'required', + activity: 'restart-safe', + timeoutMs: 0 + }, + { fetch: base, token: 'masked-token' } + ), + (error) => { + assert.match(error.message, /"restartSafeSamples":1/) + assert.match(error.message, /"requiredRestartSafeSamples":2/) + return true + } + ) +}) + +test('quiescent timeout reports connection and capacity aggregates', async () => { + const base = harness({ + active: 2, + enforcedConnectionUnits: 3, + activityLeases: 0, + activityRequestUnits: 0, + reservedRequests: 0, + restartBlockingActivityLeases: 0, + restartBlockingActivityRequestUnits: 0, + restartBlockingReservedRequests: 0, + pendingControlReservations: 4, + heartbeatFresh: false + }) + await assert.rejects( + verifyCapacityTransition({ ...config, timeoutMs: 0 }, { fetch: base, token: 'masked-token' }), + (error) => { + assert.match(error.message, /"totalConnections":2/) + assert.match(error.message, /"enforcedConnectionUnits":3/) + assert.match(error.message, /"pendingControlReservations":4/) + assert.match(error.message, /"heartbeatFresh":false/) + return true + } + ) +}) + +test('timeout distinguishes runtime and director capacity views', async () => { + const base = harness({ runtimeHardCap: 999 }) + await assert.rejects( + verifyCapacityTransition({ ...config, timeoutMs: 0 }, { fetch: base, token: 'masked-token' }), + (error) => { + assert.match(error.message, /"runtimeCapacity":\{"hardCap":999/) + assert.match(error.message, /"directorCapacity":\{"hardCap":1000/) + return true + } + ) +}) + +test('offline timeout reports every durable activity aggregate', async () => { + const base = harness({ + activityLeases: 1, + activityRequestUnits: 2, + reservedRequests: 3, + pendingControlReservations: 4, + heartbeatFresh: false + }) + await assert.rejects( + verifyCapacityTransition( + { + ...config, + admission: 'migration-only', + draining: 'either', + activity: 'restart-safe', + heartbeat: 'stale', + runtime: 'unavailable', + hardCap: undefined, + unobservedBound: undefined, + timeoutMs: 0 + }, + { + fetch: async (url, options) => + new URL(url).pathname === '/v1/admin/runtime-status' + ? Response.json({ error: 'backend_unavailable' }, { status: 503 }) + : await base(url, options), + token: 'masked-token' + } + ), + (error) => { + assert.match(error.message, /"activityLeases":1/) + assert.match(error.message, /"activityRequestUnits":2/) + assert.match(error.message, /"reservedRequests":3/) + assert.match(error.message, /"pendingControlReservations":4/) + return true + } + ) +}) + +test('waits for a drained runtime to become quiescent', async () => { + let runtimeReads = 0 + const base = harness() + const fetch = async (url, options) => { + if (new URL(url).pathname === '/v1/admin/runtime-status' && runtimeReads++ === 0) { + return Response.json({ + role: 'cell', + cellId: config.cellId, + cellUrl: config.cellOrigin, + imageDigest, + draining: true, + connectionCapacity: { + hardCap: 1_000, + unobservedBound: 60, + controlRebindReserve: 100, + ordinaryConnectionLimit: 900, + normalAdmissionPause: 840 + }, + runtime: { totalConnections: 1, preAuthConnections: 0, enforcedConnectionUnits: 1 } + }) + } + return await base(url, options) + } + let now = 0 + const result = await verifyCapacityTransition( + { ...config, timeoutMs: 10_000 }, + { + fetch, + token: 'masked', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + } + ) + assert.equal(result.cellId, config.cellId) + assert.equal(runtimeReads, 2) +}) + +test('waits for transient cell routing after a replacement becomes stable', async () => { + for (const status of [502, 503, 504]) { + let runtimeReads = 0 + let unavailableResponse + const base = harness() + const fetch = async (url, options) => { + if (new URL(url).pathname === '/v1/admin/runtime-status' && runtimeReads++ === 0) { + unavailableResponse = Response.json({ error: 'backend_unavailable' }, { status }) + return unavailableResponse + } + return await base(url, options) + } + let now = 0 + const result = await verifyCapacityTransition( + { ...config, timeoutMs: 10_000 }, + { + fetch, + token: 'masked', + now: () => now, + wait: async (milliseconds) => { + now += milliseconds + } + } + ) + assert.equal(result.cellId, config.cellId) + assert.equal(runtimeReads, 2) + assert.equal(unavailableResponse.bodyUsed, true) + } +}) + +test('persistent cell unavailability fails closed at the deadline', async () => { + let runtimeReads = 0 + let directorStatusReads = 0 + let waits = 0 + let now = 0 + const base = harness() + await assert.rejects( + verifyCapacityTransition( + { ...config, timeoutMs: 10_000 }, + { + fetch: async (url, options) => { + const pathname = new URL(url).pathname + if (pathname === '/v1/admin/runtime-status') { + runtimeReads += 1 + return Response.json({ error: 'backend_unavailable' }, { status: 503 }) + } + if (pathname === '/v1/admin/cell-status') directorStatusReads += 1 + return await base(url, options) + }, + token: 'masked', + now: () => now, + wait: async (milliseconds) => { + waits += 1 + now += milliseconds + } + } + ), + /capacity transition verification timed out: \{"runtimeAvailable":false\}/ + ) + assert.equal(runtimeReads, 3) + assert.equal(directorStatusReads, 0) + assert.equal(waits, 2) +}) + +test('general-or-migration-only admits both recovery states and nothing else', async () => { + for (const [admission, accepted] of [ + ['general', true], + ['migration-only', true], + ['existing-only', false] + ]) { + const attempt = verifyCapacityTransition( + { + ...config, + admission: 'general-or-migration-only', + draining: 'either', + activity: 'allowed' + }, + { fetch: harness({ admission, draining: 'either' }), token: 'masked' } + ) + if (accepted) { + await attempt + } else { + await assert.rejects(attempt, /admission does not match/) + } + } +}) + +test('does not retry a rejected cell admin token', async () => { + let waits = 0 + const base = harness() + await assert.rejects( + verifyCapacityTransition(config, { + fetch: async (url, options) => + new URL(url).pathname === '/v1/admin/runtime-status' + ? Response.json({ error: 'invalid_token' }, { status: 401 }) + : await base(url, options), + token: 'masked', + wait: async () => { + waits += 1 + } + }), + /cell runtime status returned 401/ + ) + assert.equal(waits, 0) +}) diff --git a/cloud/dev/scripts/verify-relay-legacy-bootstrap.mjs b/cloud/dev/scripts/verify-relay-legacy-bootstrap.mjs new file mode 100644 index 00000000000..1c51ac28f36 --- /dev/null +++ b/cloud/dev/scripts/verify-relay-legacy-bootstrap.mjs @@ -0,0 +1,292 @@ +import { createHash } from 'node:crypto' +import { readFileSync } from 'node:fs' +import { pathToFileURL } from 'node:url' + +const CAPACITY_PROTOCOL = 2 +const LEGACY_RUNTIME_KEYS = ['cellId', 'cellUrl', 'imageDigest', 'role', 'v'] +const METRIC_COUNTS = [ + 'totalConnections', + 'preAuthConnections', + 'controls', + 'splices', + 'pendingSplices', + 'queuedBytes' +] + +function integer(value, name) { + if (!Number.isSafeInteger(value) || value < 0) throw new Error(`${name} is invalid`) + return value +} + +export function parseLegacyBootstrapArguments(argv) { + const values = {} + for (let index = 0; index < argv.length; index += 2) { + const key = argv[index] + const value = argv[index + 1] + if (!key?.startsWith('--') || value === undefined) throw new Error('invalid arguments') + values[key.slice(2)] = value + } + for (const key of [ + 'director-origin', + 'cell-origin', + 'cell-id', + 'admission', + 'expected-image-digest', + 'metrics-after', + 'metrics-file' + ]) { + if (!values[key]) throw new Error(`missing --${key}`) + } + if (!['general', 'migration-only'].includes(values.admission)) { + throw new Error('--admission must be general or migration-only') + } + const directorOrigin = new URL(values['director-origin']) + const cellOrigin = new URL(values['cell-origin']) + if ( + directorOrigin.protocol !== 'https:' || + directorOrigin.origin !== values['director-origin'] || + cellOrigin.protocol !== 'https:' || + cellOrigin.origin !== values['cell-origin'] + ) { + throw new Error('origins must be canonical HTTPS origins') + } + if (!/^sha256:[a-f0-9]{64}$/.test(values['expected-image-digest'])) { + throw new Error('--expected-image-digest is invalid') + } + const metricsAfter = Date.parse(values['metrics-after']) + if (!Number.isFinite(metricsAfter)) throw new Error('--metrics-after is invalid') + const runtimeStartedAfter = values['runtime-started-after'] === undefined + ? undefined + : Date.parse(values['runtime-started-after']) + if (runtimeStartedAfter !== undefined && !Number.isFinite(runtimeStartedAfter)) { + throw new Error('--runtime-started-after is invalid') + } + const previousIncarnationDigest = values['previous-incarnation-digest'] + if ( + previousIncarnationDigest !== undefined && + !/^[a-f0-9]{64}$/.test(previousIncarnationDigest) + ) { + throw new Error('--previous-incarnation-digest is invalid') + } + const hardCap = values['hard-cap'] === undefined + ? undefined + : integer(Number(values['hard-cap']), '--hard-cap') + const unobservedBound = values['unobserved-bound'] === undefined + ? undefined + : integer(Number(values['unobserved-bound']), '--unobserved-bound') + if ((hardCap === undefined) !== (unobservedBound === undefined)) { + throw new Error('capacity expectations must be paired') + } + const capacityState = values['capacity-state'] ?? (hardCap === undefined ? 'absent' : 'stale') + if (!['absent', 'stale', 'absent-or-stale'].includes(capacityState)) { + throw new Error('--capacity-state is invalid') + } + if ((capacityState === 'absent') !== (hardCap === undefined)) { + throw new Error('capacity state and expectations do not match') + } + return { + directorOrigin: directorOrigin.origin, + cellOrigin: cellOrigin.origin, + cellId: values['cell-id'], + admission: values.admission, + expectedImageDigest: values['expected-image-digest'], + metricsAfter, + metricsFile: values['metrics-file'], + runtimeStartedAfter, + previousIncarnationDigest, + hardCap, + unobservedBound, + capacityState + } +} + +async function responseJson(response, label) { + const body = await response.json().catch(() => ({})) + if (!response.ok) throw new Error(`${label} returned ${response.status}`) + return body +} + +function requireLegacyRuntime(runtime, config) { + if (JSON.stringify(Object.keys(runtime).sort()) !== JSON.stringify(LEGACY_RUNTIME_KEYS)) { + throw new Error('cell runtime does not match the reviewed legacy contract') + } + if ( + runtime.v !== 1 || + runtime.role !== 'cell' || + runtime.cellId !== config.cellId || + runtime.cellUrl !== config.cellOrigin || + runtime.imageDigest !== config.expectedImageDigest + ) { + throw new Error('legacy cell runtime identity does not match') + } +} + +function requireDirectorQuiescence(status, config, now) { + const capacity = status.connectionCapacity + const staleCapacityTransition = config.capacityState !== 'absent' && capacity !== null + if ( + status.cellId !== config.cellId || + status.cellUrl !== config.cellOrigin || + status.enabled !== true || + status.admissionState !== config.admission || + status.runtime?.ready !== true || + (status.runtime.heartbeatFresh !== true && + !(staleCapacityTransition && status.runtime.heartbeatFresh === false)) || + status.runtime.cellUrl !== config.cellOrigin + ) { + throw new Error('legacy cell is not fresh, ready, and in the required admission state') + } + if (config.capacityState === 'absent' && capacity !== null) { + throw new Error('legacy cell has unexpected connection capacity') + } + if ( + config.capacityState !== 'absent' && + !(config.capacityState === 'absent-or-stale' && capacity === null) && + (capacity?.hardCap !== config.hardCap || + capacity.unobservedBound !== config.unobservedBound || + capacity.controlRebindReserve !== 100 || + capacity.ordinaryConnectionLimit !== config.hardCap - 100 || + capacity.normalAdmissionPause !== config.hardCap - 100 - config.unobservedBound || + capacity.heartbeatFresh !== false) + ) { + throw new Error('legacy cell connection capacity does not match') + } + integer(status.runtime.startedAt, 'runtime started at') + integer(status.runtime.lastHeartbeatAt, 'runtime heartbeat at') + if (!/^[0-9a-f-]{36}$/.test(status.runtime.cellIncarnation)) { + throw new Error('runtime cell incarnation is invalid') + } + const incarnationDigest = createHash('sha256') + .update(status.runtime.cellIncarnation) + .digest('hex') + if (incarnationDigest === config.previousIncarnationDigest) { + throw new Error('legacy fallback heartbeat incarnation did not change') + } + if ( + config.runtimeStartedAfter !== undefined && + (integer(status.runtime.startedAt, 'runtime started at') < config.runtimeStartedAfter || + status.runtime.startedAt > now + 30_000) + ) { + throw new Error('legacy fallback heartbeat predates the replacement') + } + const activity = [ + status.reservedRequests, + status.activityLeases, + status.activityRequestUnits, + status.outgoingMigrations, + status.incomingMigrations, + status.runtime.observedRequests + ] + if (capacity !== null) { + activity.push( + capacity.observedConnections, + capacity.inFlightConnections, + capacity.reservedConnectionUnits, + capacity.enforcedConnectionUnits, + capacity.pendingControlReservations + ) + } + if (activity.some((value) => integer(value, 'director activity count') !== 0)) { + throw new Error('legacy fallback has durable activity') + } + integer(status.assignments, 'assignments') + return incarnationDigest +} + +function requireFreshZeroMetrics(metrics, config, now) { + if (!Array.isArray(metrics)) throw new Error('legacy runtime metrics are invalid') + const samples = metrics.filter((entry) => { + const timestamp = Date.parse(entry?.timestamp) + return Number.isFinite(timestamp) && timestamp >= config.metricsAfter && timestamp <= now + 30_000 + }) + const timestamps = new Set(samples.map((entry) => entry.timestamp)) + if (samples.length < 2 || timestamps.size < 2) { + throw new Error('legacy runtime metrics need two post-boundary samples') + } + const latest = Math.max(...samples.map((entry) => Date.parse(entry.timestamp))) + if (latest < now - 90_000) throw new Error('legacy runtime metrics are stale') + for (const sample of samples) { + if (sample.cellId !== config.cellId || sample.metricVersion !== 1) { + throw new Error('legacy runtime metrics do not match the cell') + } + if (METRIC_COUNTS.some((field) => integer(sample[field], field) !== 0)) { + throw new Error('legacy runtime metrics are not quiescent') + } + } + return samples.length +} + +export async function verifyLegacyBootstrap(config, overrides = {}) { + const fetchImpl = overrides.fetch ?? fetch + const token = overrides.token ?? process.env.ORCA_RELAY_ADMIN_ID_TOKEN + const now = overrides.now?.() ?? Date.now() + const metrics = overrides.metrics ?? JSON.parse(readFileSync(config.metricsFile, 'utf8')) + if (!token || token.length > 8_192) throw new Error('admin identity token is unavailable') + const publicChecks = await Promise.all([ + responseJson( + await fetchImpl(`${config.directorOrigin}/health`, { + signal: AbortSignal.timeout(15_000) + }), + 'director health' + ), + responseJson( + await fetchImpl(`${config.cellOrigin}/health`, { signal: AbortSignal.timeout(15_000) }), + 'cell health' + ), + responseJson( + await fetchImpl(`${config.cellOrigin}/ready`, { signal: AbortSignal.timeout(15_000) }), + 'cell readiness' + ) + ]) + if ( + publicChecks[0].ok !== true || + publicChecks[0].connectionCapacityProtocol !== CAPACITY_PROTOCOL || + publicChecks[1].ok !== true || + publicChecks[2].ok !== true + ) { + throw new Error('legacy fallback public checks failed') + } + const headers = { authorization: `Bearer ${token}`, 'content-type': 'application/json' } + const [runtime, result] = await Promise.all([ + responseJson( + await fetchImpl(`${config.cellOrigin}/v1/admin/runtime-status`, { + method: 'POST', + headers, + body: JSON.stringify({ v: 1 }), + signal: AbortSignal.timeout(30_000) + }), + 'cell runtime status' + ), + responseJson( + await fetchImpl(`${config.directorOrigin}/v1/admin/cell-status`, { + method: 'POST', + headers, + body: JSON.stringify({ v: 1, cellId: config.cellId }), + signal: AbortSignal.timeout(30_000) + }), + 'cell status' + ) + ]) + requireLegacyRuntime(runtime, config) + const incarnationDigest = requireDirectorQuiescence(result.status, config, now) + const metricSamples = requireFreshZeroMetrics(metrics, config, now) + return { + cellId: config.cellId, + admissionState: result.status.admissionState, + assignments: integer(result.status.assignments, 'assignments'), + metricSamples, + incarnationDigest + } +} + +export async function main(argv = process.argv.slice(2)) { + const result = await verifyLegacyBootstrap(parseLegacyBootstrapArguments(argv)) + process.stdout.write(`${JSON.stringify({ event: 'relay_legacy_bootstrap_verified', ...result })}\n`) +} + +if (import.meta.url === pathToFileURL(process.argv[1]).href) { + main().catch((error) => { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`) + process.exitCode = 1 + }) +} diff --git a/cloud/dev/scripts/verify-relay-legacy-bootstrap.test.mjs b/cloud/dev/scripts/verify-relay-legacy-bootstrap.test.mjs new file mode 100644 index 00000000000..0cb3cb5a25a --- /dev/null +++ b/cloud/dev/scripts/verify-relay-legacy-bootstrap.test.mjs @@ -0,0 +1,395 @@ +import assert from 'node:assert/strict' +import { test } from 'node:test' +import { relayWorkflowUrl } from './relay-repository.mjs' +import { + parseLegacyBootstrapArguments, + verifyLegacyBootstrap +} from './verify-relay-legacy-bootstrap.mjs' + +const now = Date.parse('2026-08-10T22:10:00Z') +const digest = `sha256:${'a'.repeat(64)}` +const config = { + directorOrigin: 'https://relay.example.com', + cellOrigin: 'https://c3.relay.example.com', + cellId: 'staging-gce-c3', + admission: 'general', + expectedImageDigest: digest, + metricsAfter: now - 120_000, + metricsFile: '/unused', + runtimeStartedAfter: undefined, + previousIncarnationDigest: undefined, + hardCap: undefined, + unobservedBound: undefined, + capacityState: 'absent' +} + +const metrics = [30, 60].map((secondsAgo) => ({ + timestamp: new Date(now - secondsAgo * 1_000).toISOString(), + cellId: config.cellId, + metricVersion: 1, + totalConnections: 0, + preAuthConnections: 0, + controls: 0, + splices: 0, + pendingSplices: 0, + queuedBytes: 0 +})) + +function response(body, status = 200) { + return Response.json(body, { status }) +} + +function harness({ + runtime = {}, + status = {}, + ready = true, + cell = config, + runtimeHeartbeatFresh = true +} = {}) { + return async (url) => { + const path = new URL(url).pathname + if (path === '/health') { + return response( + url.startsWith(config.directorOrigin) + ? { ok: true, connectionCapacityProtocol: 2 } + : { ok: true } + ) + } + if (path === '/ready') return ready ? response({ ok: true }) : response({ error: 'no' }, 503) + if (path === '/v1/admin/runtime-status') { + return response({ + v: 1, + role: 'cell', + cellId: cell.cellId, + cellUrl: cell.cellOrigin, + imageDigest: digest, + ...runtime + }) + } + return response({ + status: { + cellId: cell.cellId, + cellUrl: cell.cellOrigin, + enabled: true, + admissionState: 'general', + assignments: 12, + reservedRequests: 0, + activityLeases: 0, + activityRequestUnits: 0, + outgoingMigrations: 0, + incomingMigrations: 0, + connectionCapacity: null, + runtime: { + cellUrl: cell.cellOrigin, + cellIncarnation: '00000000-0000-4000-8000-000000000001', + startedAt: now - 90_000, + ready: true, + observedRequests: 0, + lastHeartbeatAt: now - 1_000, + heartbeatFresh: runtimeHeartbeatFresh + }, + ...status + } + }) + } +} + +test('parses the exact legacy bootstrap evidence boundary', () => { + assert.deepEqual( + parseLegacyBootstrapArguments([ + '--director-origin', config.directorOrigin, + '--cell-origin', config.cellOrigin, + '--cell-id', config.cellId, + '--admission', 'general', + '--expected-image-digest', digest, + '--metrics-after', new Date(config.metricsAfter).toISOString(), + '--metrics-file', '/tmp/metrics.json' + ]), + { ...config, metricsFile: '/tmp/metrics.json' } + ) +}) + +test('accepts the exact old runtime shape with two fresh zero samples', async () => { + const result = await verifyLegacyBootstrap(config, { + fetch: harness(), + metrics, + now: () => now, + token: 'masked-token' + }) + assert.deepEqual( + { ...result, incarnationDigest: undefined }, + { + cellId: config.cellId, + admissionState: 'general', + assignments: 12, + metricSamples: 2, + incarnationDigest: undefined + } + ) + assert.match(result.incarnationDigest, /^[a-f0-9]{64}$/) +}) + +test('rejects a new or unknown runtime shape on the legacy-only path', async () => { + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness({ runtime: { draining: false } }), + metrics, + now: () => now, + token: 'masked-token' + }), + /reviewed legacy contract/ + ) +}) + +test('rejects active durable state and active runtime metrics', async () => { + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness({ status: { activityLeases: 1 } }), + metrics, + now: () => now, + token: 'masked-token' + }), + /durable activity/ + ) + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness(), + metrics: metrics.map((entry, index) => ({ + ...entry, + controls: index === 0 ? 1 : 0 + })), + now: () => now, + token: 'masked-token' + }), + /not quiescent/ + ) + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness(), + metrics: metrics.map((entry) => ({ ...entry, pendingSplices: null })), + now: () => now, + token: 'masked-token' + }), + /pendingSplices is invalid/ + ) +}) + +test('rejects stale, pre-boundary, or single runtime samples', async () => { + for (const invalidMetrics of [ + metrics.map((entry) => ({ ...entry, timestamp: new Date(now - 180_000).toISOString() })), + [metrics[0]], + metrics.map((entry) => ({ ...entry, timestamp: new Date(now - 100_000).toISOString() })) + ]) { + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness(), + metrics: invalidMetrics, + now: () => now, + token: 'masked-token' + }), + /samples|stale/ + ) + } +}) + +test('rejects the wrong legacy image and an unready cell', async () => { + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness({ runtime: { imageDigest: `sha256:${'b'.repeat(64)}` } }), + metrics, + now: () => now, + token: 'masked-token' + }), + /identity does not match/ + ) + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness({ ready: false }), + metrics, + now: () => now, + token: 'masked-token' + }), + /readiness returned 503/ + ) +}) + +test('binds the director heartbeat to the replacement boundary', async () => { + const replacementConfig = { ...config, runtimeStartedAfter: now - 60_000 } + await assert.rejects( + verifyLegacyBootstrap(replacementConfig, { + fetch: harness(), + metrics, + now: () => now, + token: 'masked-token' + }), + /heartbeat predates/ + ) + await verifyLegacyBootstrap(replacementConfig, { + fetch: harness({ status: { runtime: { + cellUrl: config.cellOrigin, + cellIncarnation: '00000000-0000-4000-8000-000000000002', + startedAt: now - 30_000, + ready: true, + observedRequests: 0, + lastHeartbeatAt: now - 1_000, + heartbeatFresh: true + } } }), + metrics, + now: () => now, + token: 'masked-token' + }) +}) + +test('accepts a drained migration-only legacy target', async () => { + const migrationConfig = { ...config, admission: 'migration-only' } + const result = await verifyLegacyBootstrap(migrationConfig, { + fetch: harness({ status: { admissionState: 'migration-only' } }), + metrics, + now: () => now, + token: 'masked-token' + }) + assert.equal(result.admissionState, 'migration-only') +}) + +test('accepts exact stale capacity during the director-first transition', async () => { + const capacity = { + hardCap: 600, + unobservedBound: 60, + controlRebindReserve: 100, + ordinaryConnectionLimit: 500, + normalAdmissionPause: 440, + observedConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + pendingControlReservations: 0, + heartbeatFresh: false + } + const transition = { + ...config, + admission: 'migration-only', + hardCap: 600, + unobservedBound: 60, + capacityState: 'stale' + } + await verifyLegacyBootstrap(transition, { + fetch: harness({ status: { admissionState: 'migration-only', connectionCapacity: capacity } }), + metrics, + now: () => now, + token: 'masked-token' + }) + await assert.rejects( + verifyLegacyBootstrap(transition, { + fetch: harness({ status: { + admissionState: 'migration-only', + connectionCapacity: { ...capacity, heartbeatFresh: true } + } }), + metrics, + now: () => now, + token: 'masked-token' + }), + /capacity does not match/ + ) +}) + +test('resumes either legacy cell after the director update', async () => { + const capacity = { + hardCap: 600, + unobservedBound: 60, + controlRebindReserve: 100, + ordinaryConnectionLimit: 500, + normalAdmissionPause: 440, + observedConnections: 0, + inFlightConnections: 0, + reservedConnectionUnits: 0, + enforcedConnectionUnits: 0, + pendingControlReservations: 0, + heartbeatFresh: false + } + for (const number of [2, 3]) { + const cell = { + ...config, + cellId: `staging-gce-c${number}`, + cellOrigin: `https://c${number}.relay.example.com`, + admission: 'migration-only', + hardCap: 600, + unobservedBound: 60, + capacityState: 'absent-or-stale' + } + const cellMetrics = metrics.map((entry) => ({ ...entry, cellId: cell.cellId })) + for (const connectionCapacity of [null, capacity]) { + await verifyLegacyBootstrap(cell, { + fetch: harness({ + cell, + runtimeHeartbeatFresh: connectionCapacity === null, + status: { admissionState: 'migration-only', connectionCapacity } + }), + metrics: cellMetrics, + now: () => now, + token: 'masked-token' + }) + } + } +}) + +test('requires a fresh director heartbeat before capacity is configured', async () => { + await assert.rejects( + verifyLegacyBootstrap(config, { + fetch: harness({ runtimeHeartbeatFresh: false }), + metrics, + now: () => now, + token: 'masked-token' + }), + /fresh, ready/ + ) +}) + +test('requires a new director heartbeat incarnation after restart', async () => { + const before = await verifyLegacyBootstrap(config, { + fetch: harness(), + metrics, + now: () => now, + token: 'masked-token' + }) + await assert.rejects( + verifyLegacyBootstrap( + { ...config, previousIncarnationDigest: before.incarnationDigest }, + { fetch: harness(), metrics, now: () => now, token: 'masked-token' } + ), + /incarnation did not change/ + ) +}) + +test('rejects same-instance samples written before restart completion', async () => { + await assert.rejects( + verifyLegacyBootstrap( + { ...config, metricsAfter: now - 20_000 }, + { fetch: harness(), metrics, now: () => now, token: 'masked-token' } + ), + /post-boundary samples/ + ) +}) + +test('the bootstrap workflow binds zero metrics to a replacement C3 instance', async () => { + const { readFile } = await import('node:fs/promises') + const workflow = await readFile( + relayWorkflowUrl('bootstrap-relay-staging-capacity.yml'), + 'utf8' + ) + assert.match(workflow, /resource\.labels\.instance_id=.*\$\{instance_id\}/) + assert.match(workflow, /timestamp>=.*\$\{after\}/) + assert.doesNotMatch(workflow, /legacy_c3_instance_id.*!=/) + const c2Proof = workflow.indexOf('staging-gce-c2 general "${legacy_pre_boundary}"') + const isolate = workflow.indexOf('--mode isolate', c2Proof) + const drainedProof = workflow.indexOf('staging-gce-c3 migration-only', isolate) + const recreate = workflow.indexOf('recreate-instances', drainedProof) + const replacementProof = workflow.indexOf('"${legacy_c3_old_incarnation}"', recreate) + const stable = workflow.indexOf('wait-until', recreate) + const metricsBoundary = workflow.indexOf('legacy_c3_metrics_boundary=', stable) + assert.ok(c2Proof < isolate && isolate < drainedProof && drainedProof < recreate) + assert.ok(recreate < stable && stable < metricsBoundary && metricsBoundary < replacementProof) + assert.match(workflow, /recreate-instances[\s\S]*?--instances "\$\{legacy_c3_instance\}"/) + assert.match(workflow, /incarnation_args=\(--previous-incarnation-digest/) + assert.match(workflow, /trap restore_legacy_c3_fallback EXIT/) + assert.match(workflow, /--mode restore-fallback[\s\S]*--general-cell-ids staging-gce-c2/) +}) diff --git a/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs b/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs new file mode 100644 index 00000000000..f25e442d8c0 --- /dev/null +++ b/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs @@ -0,0 +1,157 @@ +import assert from 'node:assert/strict' +import test from 'node:test' + +import { + hasTerraformRoot, + renderAttributeConditions +} from './render-workload-identity-conditions.mjs' + +// GCP rejects an attribute_condition longer than this. +const ATTRIBUTE_CONDITION_LIMIT = 4096 + +const EXPECTED_CONDITIONS = { + staging: { + relay: { + github_staging_relay_capacity: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/prove-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/recover-relay-staging-c4-image.yml@refs/heads/main')) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-prove-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-recover-relay-staging-c4-image.yml@refs/heads/main')))", + github_staging_relay_deploy: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-staging-gce-candidate.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/power-relay-staging.yml@refs/heads/main')) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-staging-gce-candidate.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-power-relay-staging.yml@refs/heads/main')))", + github_relay_asia_topology: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && assertion.event_name == 'workflow_dispatch' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-asia-topology.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-asia-topology.yml@refs/heads/main'))", + github_relay_asia_proof: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && assertion.event_name == 'workflow_dispatch' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/prove-relay-asia-staging.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-prove-relay-asia-staging.yml@refs/heads/main'))", + }, + // The relay root creates this provider only in production, so staging has exactly one + // definition and it lives here. + apps: { + github: + "assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-auth-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/load-skill-finalization-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/recover-skill-object-staging.yml@refs/heads/main')", + }, + }, + production: { + relay: { + github: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-fence-broker.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-director.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-multi-target.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/publish-relay-production.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-production-rehome.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-production-rehome-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap.yml@refs/heads/main' && (assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap-job.yml@refs/heads/main' || assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap.yml@refs/heads/main')))) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-fence-broker.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-director.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-publish-relay-production.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && (assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main' || assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main')))))", + github_monitor: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/monitor-relay-production.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/monitor-relay-production-job.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-monitor-relay-production.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-monitor-relay-production-job.yml@refs/heads/main'))", + github_fence: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-multi-target.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-multi-target.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main'))", + github_production_relay_capacity: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-capacity.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-capacity-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap-job.yml@refs/heads/main'))) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main'))))", + github_relay_asia_topology: + "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && assertion.event_name == 'workflow_dispatch' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-asia-topology.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-asia-topology.yml@refs/heads/main'))", + }, + apps: { + github_production_app_deploy: + "assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-auth-production.yml@refs/heads/main')", + }, + }, +} + +// Every repository the relay root accepts while the public extraction runs, with the workflow-ref +// head each one contributes. The apps root is not part of the dual accept. +const ACCEPTED_REPOSITORIES = [ + { + claims: + "assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420'", + workflowHead: 'stablyai/orca-cloud/.github/workflows/' + }, + { + claims: + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420'", + workflowHead: 'stablyai/orca/.github/workflows/cloud-' + } +] + +// [root, provider, condition] for every provider the environment creates, across all roots. +async function flatten(environment) { + const rendered = await renderAttributeConditions(environment) + return Object.entries(rendered).flatMap(([root, providers]) => + Object.entries(providers).map(([provider, condition]) => [root, provider, condition]) + ) +} + +// Only roots whose directory ships can be rendered; the apps root stays in the private +// repository, so its expectations sit above unused until that directory is present. +const expectedRoots = (environment) => + Object.fromEntries( + Object.entries(EXPECTED_CONDITIONS[environment]).filter(([root]) => hasTerraformRoot(root)) + ) + +for (const environment of Object.keys(EXPECTED_CONDITIONS)) { + const roots = expectedRoots(environment) + test(`${environment} renders the exact reviewed attribute conditions`, async () => { + const rendered = await renderAttributeConditions(environment) + assert.deepEqual(Object.keys(rendered).sort(), Object.keys(roots).sort()) + for (const [root, providers] of Object.entries(roots)) { + assert.deepEqual(Object.keys(rendered[root]).sort(), Object.keys(providers).sort(), root) + for (const [provider, condition] of Object.entries(providers)) { + assert.equal(rendered[root][provider], condition, `${environment} ${root} ${provider}`) + } + } + }) + + test(`${environment} attribute conditions stay under the GCP length limit`, async () => { + for (const [root, provider, condition] of await flatten(environment)) { + assert.ok( + condition.length < ATTRIBUTE_CONDITION_LIMIT, + `${environment} ${root} ${provider} is ${condition.length} chars` + ) + } + }) + + test(`${environment} pins repository, branch, and environment on every provider`, async () => { + for (const [root, provider, condition] of await flatten(environment)) { + for (const pin of [ + "assertion.repository == 'stablyai/orca-cloud'", + "assertion.repository_id == '1273841466'", + "assertion.repository_owner_id == '127256420'", + "assertion.ref == 'refs/heads/main'", + `assertion.environment == '${environment}'` + ]) { + assert.ok(condition.includes(pin), `${environment} ${root} ${provider} is missing ${pin}`) + } + assert.ok( + condition.includes('assertion.workflow_ref ==') || + condition.includes('assertion.job_workflow_ref =='), + `${environment} ${root} ${provider} names no workflow` + ) + } + }) + + // A prefix or suffix match would turn each allowlist into a namespace grant. + test(`${environment} attribute conditions compare workflows only by equality`, async () => { + for (const [root, provider, condition] of await flatten(environment)) { + assert.doesNotMatch( + condition, + /startsWith|endsWith|matches|in \[/, + `${environment} ${root} ${provider}` + ) + } + }) +} + +// Why: the dual accept is only safe if each OR arm carries its own repository claims. An arm that +// inherited them, or a workflow ref that named the other repository, would let one repository's +// workflows run under the other's proof. +for (const environment of Object.keys(EXPECTED_CONDITIONS)) { + test(`${environment} admits both repositories through every relay provider`, async () => { + const rendered = await renderAttributeConditions(environment) + for (const [provider, condition] of Object.entries(rendered.relay)) { + assert.ok( + condition.startsWith("assertion.ref == 'refs/heads/main' && "), + `${provider} does not lead with the repository-independent claims` + ) + const refs = [...condition.matchAll(/(?:job_)?workflow_ref == '([^']+)'/g)].map( + (match) => match[1] + ) + const perRepository = ACCEPTED_REPOSITORIES.map((repository) => { + assert.ok(condition.includes(`(${repository.claims} && `), `${provider} misses an arm`) + return refs.filter((ref) => ref.startsWith(repository.workflowHead)).length + }) + assert.equal(refs.length, perRepository[0] + perRepository[1], `${provider} names a stray ref`) + assert.equal(perRepository[0], perRepository[1], `${provider} arms are not the same size`) + assert.ok(perRepository[0] > 0, `${provider} names no workflow`) + } + }) +} diff --git a/cloud/docs/orca-relay-capacity-testing.md b/cloud/docs/orca-relay-capacity-testing.md new file mode 100644 index 00000000000..73c0e5c6e5f --- /dev/null +++ b/cloud/docs/orca-relay-capacity-testing.md @@ -0,0 +1,259 @@ +# Orca Relay capacity testing + +This harness covers two different launch gates. The deterministic model proves that phase spreading produces the required aggregate heartbeat and auth-refresh rates without a synchronized cliff. The control harness opens real WebSockets, completes the host-key challenge, answers heartbeats, refreshes authorization, and reconnects with full jitter. + +Neither mode sends phone payloads or terminal content. Reports contain aggregate counts only. Access tokens and signing keys must be supplied through the documented secret paths and are never printed by the harness. + +## Deterministic 4k/10k model + +Run: + +```sh +pnpm load:relay:model +``` + +The default profiles are 4,000 and 10,000 standing controls over 15 modeled minutes. The command fails unless: + +- 15-second heartbeats produce approximately 267 and 667 pings per second; +- uniformly distributed 180–240-second token refreshes produce approximately 19 and 48 exchanges per second; +- one-second heartbeat and refresh bins remain inside the reviewed burst bounds. + +This is a schedule gate, not evidence that a cell can hold those connections. + +## Real control load + +Use a relay-scoped token source in one of two ways: + +1. Set `ORCA_RELAY_LOAD_ACCESS_TOKEN` and pass `--auth-origin`. Every control and refresh then uses the real auth-plane relay-token exchange. +2. Pass `--signing-key-file` containing the environment's auth signing key. This operator-only mode isolates relay capacity from auth capacity. Prefer memory-backed process substitution directly with `node`; `pnpm` may close that file descriptor. + +Never put either credential in command-line arguments, URLs, shell history, or reports. + +For staging, keep the signing key process-local and out of the filesystem: + +```sh +node dev/scripts/load-relay-controls.mjs \ + --director-origin https://relay-staging.onorca.dev \ + --auth-origin https://auth-staging.onorca.dev \ + --signing-key-file <(gcloud secrets versions access latest \ + --secret=orca-cloud-auth-signing-key \ + --project=onorca-cloud-staging) \ + --controls 840 \ + --ramp-seconds 210 \ + --duration-seconds 900 +``` + +Against a local or legacy combined service only: + +```sh +pnpm load:relay:controls -- \ + --target-origin http://127.0.0.1:8080 \ + --auth-origin http://127.0.0.1:8081 \ + --signing-key-file "$KEY_FILE" \ + --controls 800 \ + --ramp-seconds 210 \ + --duration-seconds 900 +``` + +Against the stable director and stamped cells: + +```sh +ORCA_RELAY_LOAD_ACCESS_TOKEN="$ACCESS_TOKEN" pnpm load:relay:controls -- \ + --director-origin https://relay-staging.onorca.dev \ + --auth-origin https://auth-staging.onorca.dev \ + --controls 800 \ + --ramp-seconds 210 \ + --duration-seconds 900 +``` + +The director form performs a real assignment for every generated relayHostId and follows the returned cell URL/epoch. Stamped GCE cells require this durable assignment; `--target-origin` cannot bypass it. Fresh identities must not exceed `hardCap - controlRebindReserve - unobservedBound` for the target cell. At the reviewed 1,000/60 policy, that placement ceiling is 840 even though the cell can hold 900 already-assigned ordinary controls. + +## Sharding + +Run one process per shard when the client machine becomes the bottleneck. Every shard receives a disjoint host/user index space: + +```sh +pnpm load:relay:controls -- \ + --director-origin https://relay-staging.onorca.dev \ + --auth-origin https://auth-staging.onorca.dev \ + --controls 1000 \ + --shard-count 4 \ + --shard-index 0 +``` + +Start shard indexes `0..3` with the same count and timing. `--controls` is per shard. Compare the combined client reports with Cloud Monitoring rather than summing only successful connection messages. + +## Required observations + +Capture these before and throughout a launch-gate run: + +- active controls, total connections, pending and active splices; +- auth successes/failures and refresh arrival rate; +- reconnect arrival distribution and close codes; +- SQL query failures and maximum interval latency; +- process queued bytes, heap, and event-loop p99; +- GCE instance CPU/memory, LB request/backend latency and 5xx metrics, plus Cloud Run director request/concurrency metrics; +- Cloud SQL CPU, connections, locks, and failover state. + +The real harness fails when fewer than 95% of requested controls become active, fewer than 95% remain active after ramp-up, or any steady-state connection, excess ramp retry, excess unexpected close, protocol, refresh, or socket error occurs. Public-load tests may set explicit small ramp-retry and close budgets; both default to zero and remain visible in the aggregate report. Use `--ramp-start-delay-ms` to desynchronize independent client shards. Failure reasons are reported only as bounded aggregate categories. `--allow-partial` exists only for diagnosing a known capacity boundary; a run using it cannot satisfy a launch gate. + +For hard-capped candidates, separately verify director placement stops at +`hardCap - controlRebindReserve - unobservedBound` and target ordinary socket +admission stops at `hardCap - controlRebindReserve`. Then overlap up to 100 +same-host replacement sockets that present valid authorization, assignment, +generation, and resume data and receive an encrypted host challenge, without +admitting unrelated controls into that reserve. The boundary mode proves +pre-activation socket headroom; activation and generation replacement remain +separate protocol tests. Above 100 concurrent replacements, verify the deployed +clients use the recorded bounded retry path. + +Use two independent staging proofs. The temporary 1,000/0 policy permits 900 +fresh assignments and exposes the exact physical boundary: + +```sh +node dev/scripts/load-relay-controls.mjs \ + --director-origin https://relay-staging.onorca.dev \ + --auth-origin https://auth-staging.onorca.dev \ + --signing-key-file <(gcloud secrets versions access latest \ + --secret=orca-cloud-auth-signing-key \ + --project=onorca-cloud-staging) \ + --controls 900 \ + --rebind-probes 100 \ + --rebind-hold-ms 4000 \ + --ramp-seconds 210 \ + --duration-seconds 900 +``` + +The command must keep all 100 replacements open for the complete hold, require +the next physical socket to receive HTTP 503, and pass the 15-minute soak. + +Then apply the final 1,000/60 policy and start a separate 840-control process. +During its explicit delay, rerun the reviewed 1,000/60 transition so the no-op +cell plan performs one exact C3 restart. The boundary checks begin only after +all 840 controls recover: + +```sh +CAPACITY_SA="$(gh variable get STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT --env staging)" +ORCA_RELAY_ADMIN_ID_TOKEN="$(gcloud auth print-identity-token \ + --impersonate-service-account="${CAPACITY_SA}" \ + --project=onorca-cloud-staging \ + --include-email \ + --audiences=https://relay-staging.onorca.dev/v1/admin/drain)" \ +node dev/scripts/load-relay-controls.mjs \ + --director-origin https://relay-staging.onorca.dev \ + --auth-origin https://auth-staging.onorca.dev \ + --signing-key-file <(gcloud secrets versions access latest \ + --secret=orca-cloud-auth-signing-key \ + --project=onorca-cloud-staging) \ + --controls 840 \ + --placement-overflow-probes 1 \ + --capacity-cell-id staging-gce-c3 \ + --capacity-hard-cap 1000 \ + --capacity-unobserved-bound 60 \ + --rebind-probes 100 \ + --allow-planned-transition-retries \ + --skip-rebind-overflow-check \ + --rebind-delay-seconds 1800 \ + --rebind-hold-ms 4000 \ + --ramp-seconds 210 \ + --duration-seconds 900 +``` + +The inline environment assignment keeps the one-hour admin token process-local; do +not export, print, or persist it. Before probing, the harness requires two advancing director +heartbeats after recovery with exact 840-connection telemetry, no pending connection +reservations, and the reviewed 1,000/60 policy. This run requires the 841st fresh +placement to receive the capacity-specific HTTP 503 response, then requires a newer +exact director heartbeat while all 840 ordinary controls remain unchanged. This proves at +least 100 replacement sockets remain available, and completes the normal soak. +The explicit planned-transition flag permits only connection retries before the +recovery gate. Expected drain closes and transition retries remain separately +counted; any missing recovery or steady-state error fails the proof. Never +persist generated host keys or resume credentials. + +## Cap transition and rollback order + +Changing the reviewed cap is a fail-closed operation. The capacity workflow +first builds and prunes a capacity-protocol-2 director rollback pair. It drains +C3, validates the saved Terraform plan, updates the director inventory, and +requires the old cell heartbeat to become stale. It then validates and applies +only C3's instance-template replacement and MIG pointer. A fresh heartbeat must +report the exact cap and unobserved bound before C3 returns to general +placement. Ordinary director deploys do not prune historical revisions. + +The staging proof uses reversible admission states. C3 moves to +`migration-only` and drains before replacement. After its fresh heartbeat, C3 +becomes the sole general cell while C2 moves temporarily to `migration-only`, +so fresh load identities can only land on C3. The restore mode first makes C2 +general as a safe fallback, verifies that C3 is healthy and non-draining, then +restores the reviewed C2/C3 general set. It never uses irreversible +`existing-only` for temporary isolation. + +The cell rollout is the forward point of no return. To restore 600, keep the +new director code active, configure 600 there first, then roll the empty cell +back to 600 and require its fresh matching heartbeat. Only after that may an +older director image be considered. Never shift director traffic to a +pre-protocol-2 image while any cell is configured above 600. + +## Soak profiles + +For the accelerated three-cycle gate, use a test deployment whose lease/rotation intervals are explicitly shortened together and run at both modeled 4k and 10k aggregate levels. Keep enough shards/cells that no client or cell exceeds its reviewed ceiling. Record listener, timer, socket, queued-byte, heap, SQL, and event-loop bounds at every cycle. + +For each real-time load-balancer canary, use exact GCE cell hosts through the public external HTTPS LB with the production 86,400-second backend timeout: + +- run a low-scale socket past the actual 86,400-second cap to prove the observed LB termination and configured-director recovery path; +- run a separate production-jitter canary spanning at least three proactive rotations before that cap; +- force at least three fixed-one GCE instance terminations and record recovery through strictly newer director assignments. + +The control harness is one input to those canaries. The full canary must also run real phone/data pairs and verify earlier-leg deadline handling, close/1006/502/503/504 recovery, subscription replay, deterministic rejection of ordinary in-flight RPCs, and idempotent credential reconciliation. These long-running canaries are public-launch gates; modest served load remains sufficient for implementation iteration. + +## Legacy recovery-wave gate + +After an isolated local or staging exercise, evaluate its aggregate report: + +```sh +pnpm load:relay:recovery-gate -- --report "$AGGREGATE_REPORT" +``` + +The evaluator has no HTTP or GCP client and rejects production project IDs, +production origins, unknown fields, malformed values, and reports larger than +1 MiB. It exits nonzero unless the report proves all numeric gates: + +- 760–840 draining desktops under 10,450–11,550 background requests/minute, + including the observed assignment-heavy mix and 8,500–10,500 assignment + `503` responses/minute; +- two targets, each capped at and observed below 600 connections, with every + desktop recovered; the report must separate physical sockets, in-flight + upgrades, pending host-data units, and their enforced sum; +- concurrent boundary probes proving every accepted phone can consume its + reserved host-data leg, control rebind remains available, established + sockets stay open, and failed upgrades release capacity exactly once; +- a 2-vCPU database, pool size 3, two total public slots, and one + resolve-priority slot; +- the production one-to-two-instance director range, Cloud Run concurrency 80, + and an observed two-instance peak during the wave; +- a separate old/new-revision rollout-overlap result that reaches four + processes and eight public operations while preserving the same resolve, + readiness, pool-wait, and database-CPU gates; +- zero migration expirations, migration aborts, retry exhaustion, readiness + failures, and non-public maintenance failures; +- one key-proven target registration per desktop, at least ten minutes on the + oldest migration lease at drain, and full-wave registration within five + minutes; +- at least 90% of baseline assignment throughput, at least 95% eligible + resolve success, and below 1% resolve overload; +- pool-wait p95 below 500 ms, pool-wait max below 5 seconds, database CPU p95 + below 70%, and database CPU max below 85%; and +- recovery within 14 minutes, one minute before the migration lease expires. + +The report must come from a controlled production-shaped run that joins client +counts with relay metrics and Cloud SQL monitoring. The gate validates evidence; +it does not generate the reconnect wave or prove that supplied observations are +authentic. Never use a hand-authored passing report as launch evidence. + +The rollout-overlap fields may come from a separate isolated sub-run. A +steady-state two-instance wave cannot substitute for the four-process overlap +created while old and new Cloud Run revisions coexist. The old side must run +the current production-equivalent shared gate with zero resolve-priority +slots; the new side must run the candidate two-public-slot scheduler with one +resolve-priority slot. diff --git a/cloud/docs/orca-relay-operations.md b/cloud/docs/orca-relay-operations.md new file mode 100644 index 00000000000..0f8eb7a50fb --- /dev/null +++ b/cloud/docs/orca-relay-operations.md @@ -0,0 +1,481 @@ +# Orca Relay operations runbook + +This runbook applies to the stable Cloud Run director and the production-shaped GCE cells in both environments. It does not authorize a full Terraform apply: staging and production contain unrelated drift, so inspect a saved targeted plan and its destroy count before every apply. + +The relay is automatically active for entitled signed-in desktops. There is no rollout flag, cohort, or user toggle. The emergency product kill switch is the auth plane refusing relay-token exchange; use cell drains only to move or terminate existing data-plane work. + +## Safety rules + +- Never put relay JWTs, access tokens, invite/resume credentials, or signing keys in URLs, shell history, logs, or reports. +- Mint the admin identity token with the exact configured audience, which is the stable director origin plus `/v1/admin/drain`, even when calling another admin path. +- A drain response contains only `recovery: resolve-director`. Never provide a recovery URL from a cell. +- Start the target revision and commit its assignment before asking the source control to drain. Keep the source revision alive while the desktop registers the verified target; existing source splices, installs, and confirmations stay origin-owned until their leases settle or grace ends. +- Treat `4401`, `4404`, and `4429` as endpoint-scoped. `4409`, `4503`, transport failure, `1006`, `503`, and `504` recover through the configured director. +- Resume recovery uses bounded `POST /v1/resolve`; an unexpired invite may use only the director compatibility WebSocket. +- Public assignment and resume recovery share two public database slots. A bounded + resolve waiter takes the next slot ahead of new assignments, leaving the third + director-pool connection for non-public work. +- Stop if a plan or command includes an unrelated service, database, bucket, destroy, or a `REPLACE_*` value. + +Set environment-specific values without printing the resulting token: + +```sh +export PROJECT_ID=onorca-cloud-staging +export REGION=us-central1 +export DIRECTOR_ORIGIN=https://relay-staging.onorca.dev +export DEPLOY_SERVICE_ACCOUNT=orca-cloud-staging-gha-deploy@onorca-cloud-staging.iam.gserviceaccount.com +export ADMIN_AUDIENCE="${DIRECTOR_ORIGIN}/v1/admin/drain" +ADMIN_TOKEN="$(gcloud auth print-identity-token \ + --impersonate-service-account="${DEPLOY_SERVICE_ACCOUNT}" \ + --audiences="${ADMIN_AUDIENCE}")" +``` + +Unset `ADMIN_TOKEN` when the operation finishes. + +## Preflight and stop conditions + +Before any rebalance, evacuation, deploy, or game day: + +1. Confirm `/health` on the stable director and every native target service URL. +2. Confirm the target cell is enabled, has a fresh heartbeat, and has reservation headroom for source units plus its migration lease. +3. Check active controls, splices, pending splices, queued bytes, auth failures, reconnects, SQL failures/latency, heap, and event-loop delay. +4. Confirm Cloud SQL is healthy and the auth JWKS endpoint is serving the expected current and rotation keys. +5. Start a timestamped operator log containing only resource names, epochs, aggregate counts, and response codes. + +Stop and roll back or leave the source intact if the target cannot register, source-owned operations do not drain, SQL errors rise, queue/heap alerts fire, or reconnect arrivals form a cliff. + +## Staging sleep and wake + +The `Power Relay Staging` workflow may stop staging outside internal test windows; it never targets +the production project. Its nightly 09:00 UTC sleep is guarded and fails safely when any cell +reports an active lease, request unit, migration, or observed connection. A successful sleep +disables cell admission, proves quiescence again, makes the active auth/director Cloud Run revisions +scale-to-zero compatible, scales every staging MIG to zero, and finally stops Cloud SQL. +After selector generation 1, scheduled sleep fails closed because waking would +require prohibited re-enablement. Selector-era wake restores every retained +cell process and verifies admission without rewriting it. + +Before staging testing, dispatch `wake` with the default `configured` selection. It starts Cloud +SQL, c1, and c2, waits for health, readiness, and authenticated heartbeats, and restores the +Terraform-declared admission cells; disabled c3 stays off. Before any staging Terraform apply or +candidate workflow, wake `all` cells. The local apply command deliberately rejects a stopped or +partially awake staging topology. + +Use `status` for a read-only view of SQL activation, MIG target sizes, and the minimum-instance +setting of the active Cloud Run revisions. If a sleep fails after admission was disabled, the +workflow attempts to restore previously enabled cells and leaves SQL and MIGs running. If status +shows SQL stopped while a MIG is nonzero, do not retry sleep: wake staging and inspect the partial +state first. + +## Legacy staging tagged-revision deployment + +The blue/green script remains for historical protocol fixtures, but live staging now uses GCE cells. This procedure is staging-only and must not be used to replace a production GCE cell. For each stamped cell the script: + +1. Adds a drain tag to the exact current 100%-traffic revision and queries its actual tag URL. +2. Registers the candidate cell as disabled, deploys its tagged revision with no traffic and the tag as its cryptographically bound public origin, queries that tag URL, and passes `/health`. +3. Clones the old image into a no-traffic previous-tag keeper with the old cell ID and its tag as the bound public origin. This is the safe return target for recently dormant assignments; the exact original revision remains the source of live controls. +4. Atomically records the keeper URL while disabling new source assignments, enables the candidate, starts bounded aggregate evacuation batches, and drains the exact original revision so desktops resolve the committed target. +5. Waits for every migration lease to report a key-proven target registration, shifts service traffic only after the verified count matches, then completes migrations only as source activity reaches zero. + +The workflow obtains Google identity tokens in memory and emits aggregate counts only. It retains drain/previous tags and disabled source-cell rows so live origin-owned work can finish and recently dormant assignments remain routable until normal dormant-TTL reassignment. Do not delete old tags while any assignment still resolves to their cell IDs. + +If the workflow stops before source disable, leave the disabled no-traffic candidate in place for inspection. If it stops after migrations begin, do not edit epochs or reservations and do not blindly re-enable the source. Follow the rollback rules below; a migration that never registered a target automatically returns to the source after its lease expires with another newer epoch. + +## Production GCE candidate deployment + +Production publishes an immutable relay image first, then declares a distinct disabled candidate cell in a reviewed targeted Terraform change. The candidate must have a new cell ID, exact hostname, route, backend service, fixed-one `RECREATE` MIG, and durable incarnation. Never replace the backend behind an existing cell origin. + +Run `Deploy Relay Production Candidate` in `preflight` mode before any mutation. The workflow verifies exact-host TLS, `/health`, dependency-backed `/ready`, a fresh authenticated heartbeat, the runtime service account, served digest, fixed-one topology, private-only networking, and survivor request-unit headroom. All production cells remain disabled until an explicit go-live decision. + +After launch, `execute` additionally requires the exact `EVACUATE` confirmation. It enables the proven candidate, disables new source assignment, and performs bounded target-first evacuation. Preserve the source route, backend, and MIG until every source-owned splice, install, confirmation, assignment, reservation, and migration lease is drained. Remove those resources only in a later reviewed Terraform change. + +## Production multi-target evacuation + +Use `Deploy Relay Production Multi-Target` when one candidate cannot hold the +source below the reviewed connection ceiling. Targets are sorted by cell ID, +and active assignments are apportioned deterministically before any mutation. +The workflow rejects a plan when projected or observed target connections +reach 600, request-unit reservations do not fit, or target topology and served +digests do not match Terraform. Current targets expose counts through their +authenticated runtime status. A legacy source without that field must have a +runtime-metrics log no older than 90 seconds; missing or stale telemetry is a +hard stop. + +Preflight separately proves that every source assignment can reserve one target +control. It subtracts enforced units and pending reservations from each target's +normal-admission pause instead of using the physical hard cap. Missing, stale, or +internally inconsistent connection-capacity telemetry is a hard stop. + +Only new-image cells explicitly configured with +`ORCA_RELAY_CELL_CONNECTION_HARD_CAP=600` and +`ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=` use the hard +cap. Configure the same values in the director's cell inventory. Existing +cells without both values retain their current 900-unit admission behavior. +Never add the limit to an old-image cell. + +For a limited cell, authenticated status and heartbeat report physical +connections, in-flight upgrades, pending host-data units, and their enforced +sum. The director admits one new control only while: + +```text +enforced connection units + + durable pending-control leases + + configured unobserved-arrival bound + < 600 - 100 control-rebind reserve +``` + +Missing, stale, wrong-incarnation, mismatched-cap, or internally inconsistent +telemetry makes that cell ineligible. The unobserved bound is the isolated +test's maximum arrivals over one heartbeat plus reaction latency and maximum +pre-auth/in-flight units, with 20% safety margin. Record p99 separately as an +SLO, not the safety bound. + +The target reserves two units for each accepted phone: the phone socket and +its future host-data socket. The second leg transfers that reservation instead +of competing for capacity. Exactly 100 units are exclusive to same-host +control rotation/rebind overlap; first controls, phones, and unreserved data +sockets stop at the target's exact 500-unit ordinary limit and cannot borrow +them. The director stops placement earlier at +`500 - unobserved-arrival bound`. Authenticated runtime status publishes both +thresholds so rollout automation can gate their exact values. At the hard ceiling the +target returns HTTP 503 for new work and leaves established sockets open. A +503 causes clients to consult the configured director, but a healthy assignment +normally resolves to the same cell; it is backpressure, not an automatic +migration. Rebind rejection within the tested 100-concurrent replacement +bound, or a new-phone rejection rate above the load-proven threshold, stops +rollout; excess simultaneous rebinds use the tested bounded retry path. + +Run `preflight` first with every target admission-disabled. `execute` requires +`EVACUATE_MULTI`, disables the source, enables targets serially, publishes each +target's fixed quota serially, and then rechecks all targets. It refuses +`/drain` unless the oldest active migration has at least ten minutes remaining. + +Any `/drain` attempt is rollback-unsafe because a lost response cannot prove +the old source did not accept it. Before that attempt, the workflow may restore +source admission only when no target registered. After it, keep the source +disabled and use `recover-forward`; never restore its old assignment epochs. +If the durable attempt is still exactly `prepared`, recover-forward may acquire +the original send permit once and send its original trace and grace. A +`send-may-have-started` attempt without an application receipt remains +ambiguous and must freeze; it is never resent. +For immutable legacy c3, the director durably records the exact cell +incarnation before the workflow sends `{v:1,graceMs:120000}`. A lost response +is never retried before the grace deadline plus 30 seconds. If authenticated +legacy aggregate counts still show active controls/splices then, +`recover-forward` may durably record and send at most one +`{v:1,graceMs:0}` request. It does not add unsupported fields to c3. +Before recover-forward, run the 15-minute production monitor with its explicit +`recover-forward` migration policy and exact existing-only source cell. That +signed policy permits only already registered migrations from that source whose +target control is inactive, which the recovery drain is intended to reconnect. +Recovery registers a conservative capacity snapshot and catches up newly active +source assignments for at most five passes; it still requires exact zero before drain. +Ordinary execute evidence remains strict, and +blocked or expired/unregistered migrations, inactive target runtimes, selector +drift, stale telemetry, and all capacity and database thresholds still stop +both the gate and immediate live recheck. + +If the old source has zero controls, splices, pending splices, activity leases, +and reserved units but closing transports prevent completion, run +`fence-source` with `FENCE_SOURCE`. It additionally requires every migration to +have a key-proven active target and zero source activity. The workflow resizes +the fixed-one source MIG to zero, proves no instance remains and its heartbeat +is stale, records a short-lived exact-incarnation fence attestation, then +completes through the shared strict database guard. A new heartbeat invalidates +the attestation. +If the resize response or workflow is interrupted, rerun the same fence mode; +an already-zero source resumes only after the retained topology and database +guards pass. + +Terraform intentionally ignores only operational MIG `target_size` drift, so a +later apply cannot recreate a fenced source. All other MIG topology remains +managed and preflighted. Do not resize a relay MIG directly: the guarded +workflow is the only supported zero-size path. + +## Dormant return and dormant rebalance + +A normal assignment may move only after every activity count is zero and the bounded dormant TTL has elapsed. A returning desktop with a stale epoch resolves/assigns through the director and accepts only the newer authenticated epoch. + +For an explicitly selected dormant assignment: + +```sh +curl --fail-with-body --request POST "${DIRECTOR_ORIGIN}/v1/admin/rebalance-dormant" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data "{\"v\":1,\"userId\":\"${USER_ID}\",\"relayHostId\":\"${RELAY_HOST_ID}\",\"targetCellId\":\"${TARGET_CELL_ID}\"}" +``` + +`assignment_active` is a stop result, not permission to clear counters. Investigate leases and wait; do not manually rewrite assignment rows. + +Validate that source reservations fall, target reservations rise by exactly one pending control, and the next returning desktop registers the returned epoch on the target. + +## Cell admission selector + +The director has three durable admission states: + +- `existing-only` preserves current assignments and sticky resolution but receives no ordinary, dormant, or dead-cell placement. +- `migration-only` receives only explicitly published evacuation assignments. +- `general` receives ordinary placement and may also receive explicit evacuation assignments. + +Before the selector boundary, the legacy `enabled` admin field remains compatible: +`false` means `existing-only` and `true` means `general`. Use the explicit +`state` field to stage migration-only targets. Once selector generation 1 is +committed, direct cell-state and cell-config admission changes fail closed. +Every later change must use the selector CAS. + +Deploy the selector-compatible director before cutover. Blue/green deployment +creates a separate `selector-rollback` revision at minimum instances zero, +promotes a second compatible revision, and retains older revisions through the +stabilization window. Cutover removes every older revision and refuses to +proceed unless only the compatible active and rollback revisions remain. + +First inspect the current generation and exact membership: + +```sh +curl --fail-with-body --request POST \ + "${DIRECTOR_ORIGIN}/v1/admin/admission-selector/status" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data '{"v":1}' +``` + +Apply one exact, complete partition of every configured cell: + +```sh +curl --fail-with-body --request POST \ + "${DIRECTOR_ORIGIN}/v1/admin/admission-selector/apply" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data @selector-request.json +``` + +`selector-request.json` must contain `v:1`, a unique 8–128 character +`attemptId`, the inspected `expectedGeneration`, and `membership` arrays named +`existingOnly`, `migrationOnly`, and `general`. A cell must appear exactly +once. For generation zero it must also contain `expectedMembershipSha256`: the +lowercase SHA-256 of the inspected membership's canonical UTF-8 JSON, with keys +in `existingOnly`, `migrationOnly`, `general` order and each array sorted. Hash +the current inspected membership, not the desired membership. Prefer the +audited `cutover-admission` operation, which derives this fingerprint directly +from its inspection. The transaction updates the compatibility boolean and all +tri-state membership atomically. + +If the apply response is lost or ambiguous, do not create a new attempt. +Inspect status with the same `attemptId`. Continue only when it reports either +`committed` with the exact intended generation and membership or `unchanged` +with the exact previous generation and membership. `diverged` is a freeze and +review result. Reapplying the identical attempt is idempotent. + +After the boundary is active, register new empty migration targets only through +`POST /v1/admin/admission-selector/add-migration-cells`. The request contains +one durable attempt ID, the exact current generation, and the complete new cell +configs, including canonical origins and reviewed connection limits. The +operation requires every cell and origin to be new, inserts inventory, +admission, and limit rows, extends migration-only membership, and advances the +selector exactly once in one transaction. Replay the same request after an +ambiguous response; changing its config or reusing its attempt ID fails closed. +One cell may be registered alone; evacuation and supersession still require +their reviewed multi-cell target sets. + +Use `retire-migration-cell` to stop exactly one migration-only cell from +receiving new migrations before fencing it. Supply only that cell as the target, +an exact durable selector attempt ID, and `RETIRE_MIGRATION_CELL`. The mode +requires the cell to be migration-only and applies one generation-bound CAS +that moves it to existing-only while preserving every other membership. + +For production, publish and blue/green deploy the selector-version-2 director +first. Add the disabled fixed-one Terraform cells, review a targeted plan with +zero destroys and replacements, and apply only their templates, MIGs, backend +services, and exact URL-map routes. Then run `Deploy Relay Production +Multi-Target` in `add-migration-cells` mode with `ADD_MIGRATION_CELLS`, a stable +attempt ID, and the reviewed unobserved bound. Wait for exact TLS, health, +readiness, digest, topology, and heartbeat evidence before including the new +generation in a preflight or monitoring gate. + +After generation 1, an `existing-only` cell can never return to +`migration-only` or `general`. A proven migration-only cell may transition to +general as additive capacity through a later CAS. Compatible director +restarts verify this boundary and do not restore legacy admission. + +Production cutover uses `Deploy Relay Production Multi-Target` in +`cutover-admission` mode with `CUTOVER_SELECTOR`. Supply the complete proven +general pool and migration-only target set. Every other Terraform cell becomes +existing-only in the single CAS. A lost response is inspected by attempt ID; +only the exact committed result is accepted. + +Also supply the exact `unobserved-connection-bound` produced by the passing +incident load gate. Before pruning old director revisions, the workflow proves +every proposed general or migration-only cell is a healthy fixed-one Terraform +deployment serving its pinned digest with a fresh heartbeat. Each must expose +the integrated runtime connection-capacity contract: hard cap 600, control +rebind reserve 100, the supplied unobserved bound, and the corresponding normal +admission pause. Cutover also requires fewer than 45 pre-auth connections and +enough pause headroom for current enforced units plus the director's durable +pending control reservations. + +The cutover workflow depends on the hard-cap release exposing +`connectionCapacity` from cell runtime status and +`pendingControlReservations` from director cell status. Missing or mismatched +evidence is a hard stop; do not weaken the gate to deploy the selector branch +alone. + +After cutover, candidate and multi-target workflows require existing-only +sources and migration-only targets. Pre-drain failure preserves that selector +membership and never restores legacy general admission. + +## Hot-cell rebalance + +1. Stop sending new assignments to a demonstrably unhealthy cell without changing already-issued client URLs: + +```sh +curl --fail-with-body --request POST "${DIRECTOR_ORIGIN}/v1/admin/cell-state" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data "{\"v\":1,\"cellId\":\"${SOURCE_CELL_ID}\",\"enabled\":false}" +``` + +The disabled state survives director restart. Before selector generation 1, +explicitly re-enable the cell through the same endpoint only after health and +capacity checks pass. After generation 1, use a reviewed selector CAS and +never re-enable a legacy existing-only cell. +2. Move fully dormant assignments first with `rebalance-dormant`. +3. Recompute source/target request-unit headroom. One control costs one unit; one splice costs two; invite/install/confirmation/migration leases also reserve their contract units. +4. Move active assignments one at a time or in a bounded batch using the target-first evacuation below. +5. Pause whenever target total connections approach 800, queued bytes exceed 48 MiB, or any runtime/SQL alert fires. + +Do not infer required cells from DAU or phones. Use measured standing controls, splice/reservation distribution, startup/login bursts, and explicit safety margin. + +## Target-first active evacuation + +Start a durable migration on the director: + +```sh +curl --fail-with-body --request POST "${DIRECTOR_ORIGIN}/v1/admin/evacuate" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data "{\"v\":1,\"userId\":\"${USER_ID}\",\"relayHostId\":\"${RELAY_HOST_ID}\",\"targetCellId\":\"${TARGET_CELL_ID}\"}" +``` + +Record the returned `sourceCellId`, `targetCellId`, and strictly newer `assignmentEpoch`. The transaction reserves the target before publishing the epoch. + +Ask the source control to resolve the committed assignment while the source revision remains alive: + +```sh +curl --fail-with-body --request POST "${SOURCE_NATIVE_OR_TAG_ORIGIN}/v1/admin/drain" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data '{"v":1,"graceMs":120000}' +``` + +Wait for the desktop to register a key-proven target control. Do not proceed on the basis of a socket open alone. Confirm the migration's target registration and that source-owned splices/install/confirmation work remains on the source control. + +Complete each migration only after source activity reaches zero: + +```sh +curl --fail-with-body --request POST "${DIRECTOR_ORIGIN}/v1/admin/migration-complete" \ + --header "Authorization: Bearer ${ADMIN_TOKEN}" \ + --header 'Content-Type: application/json' \ + --data "{\"v\":1,\"userId\":\"${USER_ID}\",\"relayHostId\":\"${RELAY_HOST_ID}\",\"assignmentEpoch\":${ASSIGNMENT_EPOCH}}" +``` + +`migration_target_not_registered`, `migration_source_still_active`, and `migration_target_not_active` are safety stops. Do not bypass them. + +### Dead-source completion + +Use only `Deploy Relay Production Multi-Target` in `fence-source` mode. The +workflow is the authority that proves the source MIG is durably zero, no +instance remains, retained topology matches Terraform, and the exact +incarnation heartbeat is stale before creating the short-lived database +attestation. There is no per-assignment dead-source operator endpoint. + +The aggregate operation is idempotent. A missing or expired attestation, new +source heartbeat, source activity, stale target heartbeat, inactive target +control, changed admission, request-unit drift, or aggregate reservation +mismatch is a stop result. + +### Registered-target supersession + +Use `Deploy Relay Production Multi-Target` in `supersede-target` mode with +`SUPERSEDE_TARGET`. The workflow requires the original source to remain +disabled, proves the failed target's retained topology, disabled admission, +MIG desired size zero, zero instances, and stale exact-incarnation heartbeat, +then records a short-lived fence attestation. It proves replacement health and +conservative connection headroom before enabling the replacement and invoking +the aggregate database operation. Rerun the same mode after a lost resize or +workflow response; it resumes from durable GCE and database state. + +Target supersession runs only through the IAM-authenticated private broker. The +requester cannot access Terraform state, Compute mutations, or director +mutation routes. The broker permits one configured cell triple, binds its +gitless checkout to the immutable image commit, and holds a durable GCS lease +through the exact-plan operation. It retains that lease after a failure so only +the same request can resume before expiry. This repair does not consume or +weaken the normal 15-minute source-recovery gate; run that unchanged gate only +after failed-target contention is gone. + +The transaction removes only the proven-dead target's leases, preserves +original-source activity, reserves replacement capacity, publishes exactly +`assignmentEpoch + 1`, and retires the old migration atomically. A repeated +identical request returns the same successor. A different concurrent +replacement, live current target, unavailable replacement, unexpected lease +topology, or accounting mismatch fails closed. + +## Dead cell + +Do not call an unreachable cell or trust it to supply a target. For each affected assignment: + +1. Verify another cell has capacity. +2. Start the director evacuation to publish a newer target epoch. +3. Let desktop `1006`/transport recovery resolve the configured director and register the target. +4. Phones resolve through the director; unexpired invites use the director compatibility route. +5. Complete only after expired source activity leases release and the target control is active. + +If the dead cell returns, leave its stale epoch fenced. Never decrement an epoch or restore an old assignment row. + +## Global admission or memory pressure + +At connection headroom, queue, heap, or event-loop alerts: + +1. Preserve established controls/splices and reject new pre-auth work through the existing admission reserves. +2. Identify whether pressure is source-local, cell-wide, auth/JWKS, SQL, or a synchronized reconnect event using aggregate metrics only. +3. Move dormant work first. Evacuate active work only when a target has measured request and memory headroom. +4. If every cell is unsafe, make the auth plane refuse new relay-token exchanges, then issue authenticated drains with full-jitter client recovery. This is the kill switch; do not add a flag. + +Never raise runtime admission or queued-byte limits during an incident without a load result proving memory headroom. + +## Drain and SIGTERM + +Planned drain uses the authenticated admin endpoint and a reviewed grace that permits target-first registration and origin-owned operations to settle. Phones and desktops always recover through the configured director. + +SIGTERM is a degraded path. It may advertise zero grace, rejects new work, closes phone/data pairs together with `4503`, and cannot guarantee target registration first. Game-day this separately from planned drain. + +After either path, verify close/`1006`/`504` observations, bounded reconnect arrival, subscription replay, deterministic ordinary in-flight rejection, and idempotent install/confirmation reconciliation. + +## Rollback + +Before target registration, an expired migration automatically releases target reservations and returns the assignment to the source with another strictly newer epoch. Verify: + +- the migration is marked aborted; +- target pending-control and migration leases are gone; +- source reservation is restored exactly once; +- resolved epoch is newer than both the original and failed target epochs. + +Once a target control is registered, do not force the pre-registration rollback. Keep the source control and operations alive, repair the target, or perform a new target-first evacuation to a healthy cell. Never edit epochs or reservation counters manually. + +After a deployment traffic shift, preserve the old revision/tag until metrics and live reconnect checks pass. If the new revision is unhealthy, shift traffic back only while old controls are still valid, then issue a strictly newer director migration rather than reusing a prior epoch. + +## Game-day matrix + +Run and record each scenario in staging before launch: + +- kill a cell instance mid-session and observe configured-director recovery; +- send zero-grace SIGTERM and compare it with authenticated drain; +- hold an invite install and a resume confirmation across drain; +- wedge a slow receiver until the bounded queue closes only that splice; +- delay mobile Blob conversion and confirm text/binary counter order; +- rotate JWKS while controls refresh; +- make auth unavailable through expiry plus 60-second existing-splice-only grace, then recover with distributed jitter; +- fail SQL during install/confirm and verify only `not-found` or the one committed result is externally visible; +- return a dormant host, overload a cell, kill a cell, evacuate active work, and exercise pre-registration rollback. + +The served black-box relay suite validates the protocol/state transitions used by these procedures. The physical-device and real-GFE canaries remain separate launch gates; unit/black-box success cannot replace them. diff --git a/cloud/docs/relay-incident-monitor.md b/cloud/docs/relay-incident-monitor.md new file mode 100644 index 00000000000..3e5fc04836e --- /dev/null +++ b/cloud/docs/relay-incident-monitor.md @@ -0,0 +1,152 @@ +# Relay incident monitor + +Status: monitor core and dedicated identity boundaries are implemented locally; +the targeted production bootstrap and live negative-permission proof remain +required before dispatch. + +This monitor is read-only. It probes Relay endpoints and reads Cloud Monitoring, Compute +inventory, and authenticated aggregate director status. It does not drain, restart, resize, +deploy, change admission, or write to Google Cloud. + +## Production workflow + +Run `Monitor Relay Production` manually. Choose: + +- `dry-run` for the required 15-minute pre-drain gate. +- `monitor` for a 90-minute incident watch. + +Use the default `strict` migration policy for ordinary mutations. Select +`recover-forward` only after a drain attempt has durably registered migrations +and enter its exact existing-only recovery source cell. Recovery evidence tolerates +only the aggregate count of registered migrations whose target control is not +currently active for that source. Blocked or expired/unregistered migrations and every other +health threshold remain unchanged. + +Enter the exact selector generation and all three disjoint membership sets: +`existing-only`, `migration-only`, and `general`. Every configured cell must +appear exactly once. Generation zero is the only mode that reads the legacy +boolean admission field; selector-era generations use only the durable +tri-state membership. Any generation or membership mismatch freezes the gate. +The workflow verifies dependencies and restored evidence before +authentication. It has no shared-deploy fallback and accepts only the dedicated +monitor provider and service-account production environment variables. Do not +dispatch it until the targeted identity bootstrap is applied and its live +negative-permission checks pass. + +The job polls every 60 seconds and writes private aggregate evidence at 0, 5, 15, 30, 45, 60, 75, +and 90 minutes where applicable. It uploads state JSON, checkpoint JSONL, and Markdown for 14 +days. No tokens, request bodies, logs, user IDs, host IDs, or relay device IDs are recorded. +Reruns keep one stable incident ID, restore the immediately preceding private +artifact, verify its commit/run/attempt provenance and content hashes, and pass +`--restart`. A missing or mismatched artifact fails closed. A missing, stale, +or collector-failed sample is durably recorded and resets the active continuous +window. The next fresh sample starts a new 15- or 90-minute window under the +same incident lineage. + +Exit code `2` means the gate froze or a dry run failed. Missing, stale, malformed, unauthorized, or +unavailable telemetry fails closed. + +## Local use + +The active `gcloud` identity must be a service account that can mint an ID token for the exact +audience `https://relay.onorca.dev/v1/admin/drain`, and it needs read access to the monitored GCP +resources. A user account normally needs Token Creator on an approved service account. + +For a short run, an already minted JWT may instead be supplied through +`ORCA_RELAY_ADMIN_ID_TOKEN`. The token must remain valid for the whole run and is never persisted. +Use refreshable service-account credentials for the 90-minute mode. + +```sh +pnpm incident:relay -- \ + --environment production \ + --incident-id relay-incident-20260728 \ + --expected-selector-generation 2 \ + --expected-existing-only-cells production-gce-c1 \ + --expected-migration-only-cells production-gce-c2 \ + --expected-general-cells production-gce-c3,production-gce-c4,production-gce-c5,production-gce-c6 +``` + +Add `--pre-drain-dry-run --duration-minutes 15` for an ordinary gate. Add +`--migration-policy recover-forward --recovery-source-cell-id ` only +for a committed forward-recovery gate. Durable files default to +`.relay-incidents/`. If the process stops, rerun the identical command with `--restart`. A runner +gap resets the active window at the next fresh sample and preserves the prior +window evidence. A threshold freeze never clears automatically. + +A production candidate or multi-target mutation must download the exact +dry-run artifact by workflow run ID and attempt. It verifies the artifact +hashes and provenance, requires a green completed 15-minute state no older +than five minutes, then rechecks the live selector and one complete fresh +sample of every safety signal immediately before running the mutation command. +The signed state binds `strict` evidence to ordinary mutations and +`recover-forward` evidence to the exact recover-forward source; neither can +authorize the other. +The monitor and mutation jobs share one production lock. A passing dry-run is +durably marked consumed before mutation and cannot authorize another run. + +## Freeze thresholds + +| Signal | Freeze condition | +| --- | ---: | +| Active probe age | over 60 seconds | +| Cloud/log data age | over 180 seconds | +| Cell heartbeat age | over 45 seconds | +| Endpoint latency | over 2,000 ms | +| Cloud SQL CPU | over 80% | +| Cloud SQL memory | over 90% | +| Cloud SQL backends | over 250 (62% of the verified 400-connection ceiling) | +| Cloud SQL waiting backends | over 20 | +| Cloud SQL deadlocks | over 0 | +| Relay pool waiters | over 800 | +| Relay pool wait | over 2,500 ms | +| PostgreSQL retries in five minutes | over 300 | +| Exhausted PostgreSQL retries | over 0 | +| Director instances | outside 5–6 | +| Director CPU or memory | over 80% | +| Director concurrency | over 64 | +| Unexpected director 5xx or auth 5xx in five minutes | over 0 | +| Connections per cell process | over 500 | +| Queued bytes per cell process | over 48 MiB | +| Blocked or expired/unregistered migration | over 0 | +| Registered migration with inactive target | over 0, except in `recover-forward` evidence | + +Expected enabled cells must also have a powered runtime, healthy and ready endpoints, fresh +heartbeats, and matching live admission. + +## Implementation log + +- Recalibrated the relay pool freezes from 30 waiters / 1,000 ms to + 800 waiters / 2,500 ms (2026-08-27). Basis, measured from + `orca_relay_runtime_metrics` (`databasePoolWaitersMax`, + `databasePoolWaitMsMax`): healthy fleet-wide bursts reach 43 waiters and + 2.03 s several times an hour (52 burst-minutes over three days), a cell + roll's reconnect surge peaks at 676 waiters, and the 2026-08-23 incident + peaked at 356 waiters without ever crossing 2.5 s — amplitude does not + separate incident from routine operation in either direction, and a + 15-minute gate had roughly one-in-six odds of freezing on a burst. The + retry signals discriminate that incident at ~10x separation and keep their + thresholds; the pool bars now fence only unbounded queueing. +- Recalibrated the Cloud SQL backends freeze from 160 to 250 (2026-08-26). + Basis, measured from `cloudsql.googleapis.com/database/postgresql/num_backends` + latest-sum over 24 healthy hours: mean ~100, 1-minute spikes to 216, with + 10 minutes over the old bar of 160 — enough to freeze roughly one in ten + 15-minute pre-drain gates on baseline noise. 250 clears measured healthy + peaks and still fires well before the verified 400-connection ceiling; + pool waiters, pool wait latency, and exhausted retries keep their strict + thresholds. +- Recalibrated the PostgreSQL-retry freeze from 20 to 300 per five minutes + (2026-08-26). Basis, measured from + `jsonPayload.event="orca_relay_postgres_transaction_retry"` in production + logs: healthy-day bursts reach 234/5min with zero exhausted retries and 26% + of five-minute windows over 20, while the 2026-08-23 lock-contention + incident ran roughly 2,200–3,000/5min. Exhausted retries stay at zero + tolerance. +- Added a fail-closed state machine with latched threshold freezes, + generation-scoped checkpoint boundaries, continuity-reset evidence, cadence + accounting, restart-gap recovery, and the 15-minute pre-drain gate. +- Added Cloud Monitoring, active-probe, relay runtime, and authenticated director collectors. +- Added exact Cloud SQL instance and Cloud Run service filters, five-minute + DELTA aggregation, and serialized aggregate admin reads so monitoring cannot + load the director's three-connection database pool. +- Added private atomic state, idempotent JSONL checkpoints, and secret-safe Markdown evidence. +- Added the manual production workflow. It has not been dispatched. diff --git a/cloud/docs/relay-terraform-fencing-plan.md b/cloud/docs/relay-terraform-fencing-plan.md new file mode 100644 index 00000000000..7d8664f19f5 --- /dev/null +++ b/cloud/docs/relay-terraform-fencing-plan.md @@ -0,0 +1,149 @@ +# Relay Terraform Fencing Recovery Plan + +Status: private broker implementation in review; no production mutation performed + +## Current status + +- `2d7dc31` commits Terraform-owned per-cell zero/one desired size and removes + the lifecycle ignore. +- The follow-up slice implements private exact saved plans, SHA-256 binding, + durable attempt evidence, lost-response inspection, resume, proven + pre-apply abort, and exact-incarnation attestation. +- Focused Terraform-fence, multi-target, relay database, assignment-store, and + black-box tests pass locally. +- Production remains untouched. A rollout still requires review, CI, the + director/schema deployment, an initial output-state refresh, and a separate + reviewed fence-set commit for any cell selected during an incident. +- The private Cloud Run broker owns the exact Terraform checkout, saved plans, + state access, Compute mutation, and a durable GCS mutation lease. The + workflow requester can invoke the broker and read aggregate evidence but + cannot mutate Terraform state, Compute, or director fence routes. + +## Goal + +Make production relay-cell fencing a Terraform-managed, resumable operation. +The workflow must never attest a cell as fenced until Terraform state, GCE, +runtime identity, and retained routing all prove the same exact cell +incarnation is offline. + +## Safety boundary + +- Do not deploy, mutate GCP, push, or create a pull request during implementation. +- Keep the cell route and backend while its MIG is fenced at size zero. +- Treat any Terraform apply whose start cannot be disproved as recover-forward. +- Never restore a fenced cell automatically after an apply may have started. +- Store plans in a private temporary directory and remove them on every exit. +- Never print, upload, or commit complete plan JSON. +- Fence modes are fail-closed until a private mutation broker owns the narrow + state/plan and exact-cell Compute permissions. The exact-workflow GHA fence + account must never receive those permissions or director mutation routes + directly; it has only aggregate reads and fence-attempt status. +- The monitor identity has no Terraform state access. It can call only the + aggregate selector, cell, evacuation, and runtime status routes. +- The broker accepts only its configured source, failed target, and + replacement target. Its image commit must equal the reviewed fence commit; + callers cannot override topology, Terraform paths, commands, or identities. + +## State model + +1. Add `relay_gce_fenced_cells` as the committed set of cell IDs whose desired + MIG size is zero. +2. Derive every cell's desired size from that set: fenced is zero, otherwise + one. +3. Reject unknown fenced IDs and any topology outside the zero-or-one + invariant. +4. Remove the MIG `target_size` lifecycle ignore so Terraform owns the fence. +5. Persist one durable attempt record keyed by an unguessable attempt ID. It + binds environment, cell ID, exact cell incarnation, MIG/generation + identity, fence commit, saved-plan digest, GCE operation, creation time, + expiry, and terminal status. + +## Operation sequence + +### 1. Prepare and guard + +1. Require the explicit fence confirmation and a clean checkout at the exact + fence commit. +2. Confirm the committed production fence set contains the requested cell. +3. Read Terraform state and live topology, then bind the attempt to the exact + MIG, instance group, backend, origin, and cell incarnation. +4. Run the existing admission, assignment, lease, migration, connection, + heartbeat, backend, and route guards before planning. + +### 2. Create and validate the exact plan + +1. Create a private temporary directory with mode `0700` under `umask 077`. +2. Write the saved plan there and compute its SHA-256 digest. +3. Inspect only narrow fields from `terraform show -json`. +4. Require exactly one relevant in-place MIG update from target size one to + zero. +5. Reject create, delete, replace, unrelated update, route/backend removal, or + any different cell action. +6. Persist the pending attempt evidence before apply. + +### 3. Apply or recover forward + +1. Recheck the pre-apply guards and saved-plan digest immediately before apply. +2. Apply the exact saved plan with Terraform locking. +3. If apply fails or its response is lost, inspect Terraform state, live MIG + size, remaining instances, and the relevant GCE operation. +4. If apply may have started, retain the committed fence set and keep polling + forward until the MIG converges to zero or a bounded, diagnosable failure is + recorded. +5. Resume idempotently from durable evidence after workflow or runner loss. + +### 4. Abort before apply + +1. Allow abort cleanup only when evidence proves apply never began. +2. Require Terraform state and live MIG to remain at one with the original + identity and topology. +3. Mark the attempt aborted, remove the cell from the committed fence set + through a separately reviewed commit, and delete local plan artifacts. +4. If apply start is ambiguous, refuse abort and recover forward. + +### 5. Attest + +1. Require Terraform state target size zero. +2. Require live MIG target size zero and no managed instances. +3. Require the exact MIG/generation identity and retained route/backend + topology to match the attempt. +4. Require admission disabled and the exact runtime heartbeat stale. +5. Require unexpired durable attempt evidence and the same saved-plan digest, + fence commit, and GCE operation. +6. Record the exact-incarnation cell fence and complete the attempt in one + database transaction. + +## Implementation slices + +- Terraform: variable, per-cell desired size, validation, outputs, tfvars, IAM. +- Relay contract: durable attempt schema, store methods, admin endpoints, exact + attestation binding, retention/expiry. +- Deployment tooling: private plan lifecycle, digest and selector validation, + apply/recovery/abort state machine, GCE operation inspection. +- Workflow: explicit prepare/apply/resume/abort modes and no direct MIG resize. +- Runbooks: committed fence-set, exact-plan, recover-forward, and reviewed + abort procedures. + +## Required tests + +- Normal fence plan and apply. +- Lost or ambiguous apply response recovers forward. +- Resume when Terraform state and live MIG are already zero. +- Pre-apply guard failure performs no mutation. +- Proven pre-apply abort permits committed fence-set cleanup. +- Ambiguous apply refuses abort cleanup. +- No attestation before every Terraform, GCE, route/backend, heartbeat, and + exact-incarnation condition passes. +- Saved-plan validation rejects unrelated, replacement, create, and destroy + actions. +- Attempt evidence rejects wrong environment, cell, incarnation, MIG, + generation, commit, digest, operation, expiry, and terminal state. + +## Validation + +- Focused Node and relay contract tests. +- `pnpm test`, `pnpm typecheck`, and `pnpm lint`. +- Sequential relay contract and relay builds after schema/API changes. +- Terraform 1.15.8 `fmt -check -recursive`, `init -backend=false`, and + `validate`. +- `git diff --check`, local commit, and clean worktree. diff --git a/cloud/docs/relay-workflows.md b/cloud/docs/relay-workflows.md new file mode 100644 index 00000000000..14bb2a25d7c --- /dev/null +++ b/cloud/docs/relay-workflows.md @@ -0,0 +1,402 @@ +# Relay GitHub Actions Configuration + +The `cloud-*` workflows in `.github/workflows/` are the Relay deploy and +operate surface. Every one of them is gated on the repository variable +`ORCA_CLOUD_OPERATIONS_ENABLED == 'true'` and does nothing until the repository +owner sets it. The app and auth deploy workflows this document once also +covered stay in the private `stablyai/orca-cloud` repository. + +Set these staging environment variables before running the staging deploy workflow: + +```text +STAGING_GCP_REGION +STAGING_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER +STAGING_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT +STAGING_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER +STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT +STAGING_GCP_RELAY_ASIA_TOPOLOGY_WORKLOAD_IDENTITY_PROVIDER +STAGING_GCP_RELAY_ASIA_TOPOLOGY_SERVICE_ACCOUNT +STAGING_GCP_RELAY_ASIA_PROOF_WORKLOAD_IDENTITY_PROVIDER +STAGING_GCP_RELAY_ASIA_PROOF_SERVICE_ACCOUNT +``` + +`STAGING_GCP_REGION` exists today only as a repository variable. Create it as a +staging **environment** variable before deleting any repository-level variable; +`Deploy Relay Staging` gates its whole job on it being non-empty, so a +delete-before-create silently skips it. + +The Relay deploy, capacity, and Asia values come from the matching staging +Terraform outputs after the targeted identity bootstrap: + +```sh +terraform -chdir=infra/terraform output -raw github_staging_relay_deploy_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_staging_relay_deploy_service_account + +gh variable set STAGING_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER --env staging --body '' +gh variable set STAGING_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT --env staging --body '' + +terraform -chdir=infra/terraform output -raw github_staging_relay_capacity_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_staging_relay_capacity_service_account + +gh variable set STAGING_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER --env staging --body '' +gh variable set STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT --env staging --body '' +terraform -chdir=infra/terraform output -raw github_relay_asia_proof_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_relay_asia_proof_service_account +gh variable set STAGING_GCP_RELAY_ASIA_PROOF_WORKLOAD_IDENTITY_PROVIDER --env staging --body '' +gh variable set STAGING_GCP_RELAY_ASIA_PROOF_SERVICE_ACCOUNT --env staging --body '' +``` + +The capacity provider accepts only this repository's capacity proof and +bootstrap workflows on `main` with the `staging` environment. It does not fall +back to the shared deploy identity. + +The Relay deploy provider accepts exactly five workflows on `main` with the +`staging` environment: Bootstrap Relay Staging Capacity, Deploy Relay Staging, +Deploy Relay Staging GCE Candidate, Operate Relay Asia Admission, and Power +Relay Staging. Set both `STAGING_GCP_RELAY_DEPLOY_*` variables before merging +the workflow repoint; the job gates read them and skip while they are unset. + +Set these separately before enabling production deploys: + +```text +PRODUCTION_GCP_REGION +PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER +PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT +PRODUCTION_GCP_RELAY_MONITOR_WORKLOAD_IDENTITY_PROVIDER +PRODUCTION_GCP_RELAY_MONITOR_SERVICE_ACCOUNT +PRODUCTION_GCP_RELAY_FENCE_WORKLOAD_IDENTITY_PROVIDER +PRODUCTION_GCP_RELAY_FENCE_SERVICE_ACCOUNT +PRODUCTION_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER +PRODUCTION_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT +PRODUCTION_GCP_RELAY_ASIA_TOPOLOGY_WORKLOAD_IDENTITY_PROVIDER +PRODUCTION_GCP_RELAY_ASIA_TOPOLOGY_SERVICE_ACCOUNT +PRODUCTION_GCP_RELAY_DIRECTOR_RUNTIME_SERVICE_ACCOUNT +PRODUCTION_GCP_RELAY_RUNTIME_SERVICE_ACCOUNT +``` + +The Relay operations values come from matching Terraform outputs. Set them as +production GitHub environment variables, not repository fallbacks. Every one of +them is relay-owned in `infra/terraform`: + +```sh +terraform -chdir=infra/terraform output -raw github_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_deploy_service_account +terraform -chdir=infra/terraform output -raw github_relay_monitor_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_relay_monitor_service_account +terraform -chdir=infra/terraform output -raw github_relay_fence_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_relay_fence_service_account +terraform -chdir=infra/terraform output -raw github_production_relay_capacity_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_production_relay_capacity_service_account +terraform -chdir=infra/terraform output -raw relay_director_runtime_service_account +terraform -chdir=infra/terraform output -raw relay_runtime_service_account + +gh variable set PRODUCTION_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_MONITOR_WORKLOAD_IDENTITY_PROVIDER --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_MONITOR_SERVICE_ACCOUNT --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_FENCE_WORKLOAD_IDENTITY_PROVIDER --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_FENCE_SERVICE_ACCOUNT --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_ASIA_TOPOLOGY_WORKLOAD_IDENTITY_PROVIDER --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_ASIA_TOPOLOGY_SERVICE_ACCOUNT --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_DIRECTOR_RUNTIME_SERVICE_ACCOUNT --env production --body '' +gh variable set PRODUCTION_GCP_RELAY_RUNTIME_SERVICE_ACCOUNT --env production --body '' +``` + +Run those commands only from the audited operator session after the targeted +identity bootstrap apply. The providers require their exact workflows on +`refs/heads/main` with the `production` environment. Missing values fail +closed; no dedicated operations identity falls back to the shared deploy identity. + +The shared production identity is restricted to seven named direct Relay callers plus the exact +regional-rehome and same-cap reusable wrapper/job pairs on `main` in the `production` environment. +Its Artifact Registry and Cloud Run mutation permissions are scoped to the Orca repository, Relay +director, and Relay fence broker; it cannot mutate the API or auth services. + +Bootstrap the production capacity identity only after its reviewed commit is on +`main`. Reinitialize the production backend explicitly, save the exact targeted +plan, require **9 additions, 0 changes, and 0 deletions**, then apply that saved +plan. `manage_artifact_dns=false` keeps the unimported Cloudflare records out of +this GCP-only operation. + +```sh +export GOOGLE_OAUTH_ACCESS_TOKEN="$(gcloud auth print-access-token)" +terraform -chdir=infra/terraform init -reconfigure \ + -backend-config=backend/production.hcl -input=false +terraform -chdir=infra/terraform plan -input=false -lock-timeout=30s \ + -var-file=environments/production.tfvars -var manage_artifact_dns=false \ + -target=google_iam_workload_identity_pool_provider.github_production_relay_capacity \ + -target=google_service_account.github_production_relay_capacity \ + -target=google_service_account_iam_member.github_production_relay_capacity_workload_identity_user \ + -target=google_project_iam_custom_role.github_production_relay_capacity_mutation \ + -target=google_project_iam_member.github_production_relay_capacity_mutation \ + -target=google_project_iam_member.github_production_relay_capacity_viewer \ + -target=google_project_iam_member.github_production_relay_capacity_artifact_reader \ + -target=google_storage_bucket_iam_member.github_production_relay_capacity_state \ + -target=google_service_account_iam_member.github_production_relay_capacity_runtime_user \ + -out=/tmp/orca-relay-production-capacity-identity.tfplan +terraform -chdir=infra/terraform show /tmp/orca-relay-production-capacity-identity.tfplan +terraform -chdir=infra/terraform apply /tmp/orca-relay-production-capacity-identity.tfplan +unlink /tmp/orca-relay-production-capacity-identity.tfplan +``` + +Confirm a second targeted plan is empty before setting the two production +environment variables from reviewed Terraform outputs. + +`Deploy Relay Asia Topology` is the only workflow allowed to add the reviewed +`asia-east2` network and fixed-one cell topology. Its dedicated identity is +bound to that exact workflow, `main`, `workflow_dispatch`, and the selected +GitHub environment. The workflow always saves a targeted plan, rejects any +delete, replacement, US-resource, SQL, DNS, certificate, or unrelated change, +and applies only the exact validated plan. It always passes +`manage_artifact_dns=false`; observability and IAM are separate targeted +operations. + +Before the first admission operation, publish and deploy a compatible director image while the +topology remains unchanged. Verify the exact serving digest, health, readiness, and rollback tag; +older directors reject the generation-zero membership fingerprint. After a topology apply, use +`Operate Relay Asia Admission` in `inspect` mode to read the exact live selector generation. If and +only if it is generation 0, run +the explicit `initialize` mode with the exact membership SHA-256 printed by +`inspect` and `INITIALIZE_ADMISSION_SELECTOR`; the director checks both under its +database lock, so this freezes the existing membership without adding, removing, +or moving a cell and rejects intervening drift. Then +atomically register the new cells as migration-only, binding every mutation to +the exact live selector generation and a durable attempt ID. Deploy and verify +the director configuration only after registration, then promote C27 alone before C28/C29. +Rollback returns +Asia cells to migration-only; it does not destroy the network or use +existing-only. The production topology dispatch remains unavailable until the +published compatible image is committed for C27-C29. + +`Prove Relay Asia Staging` runs from a dedicated ephemeral repository runner in +`asia-east2` with the `relay-asia-east2-load` label. It promotes only staging C4, runs four bounded +load shards at the exact 3,000/6,000 shape, validates continuous cell/director/Cloud SQL evidence, +and always returns C4 to migration-only before publishing evidence. Register the runner with +`--ephemeral` immediately before dispatch so it accepts one proof job and then removes itself. Each +shard exchanges its +exact workflow OIDC identity for a ten-minute in-memory staging token; no load +credential, signing key, or raw load output is stored or uploaded. + +Production promotion evidence must prove the exact production manifest, not an independent rebuild. +Before refreshing C4, target and apply only +`google_artifact_registry_repository_iam_member.github_production_relay_staging_mirror_writer` +from the staging state with `manage_artifact_dns=false`. Run `Publish Relay Production Image` in +`mirror-staging` mode with the exact digest and typed confirmation, then run `Deploy Relay Staging` +with that digest. The mirror validates identical source and target manifest digests; the staging +deploy binds the request to C4's checked-in image and deploys the director by that same digest. +Only then refresh empty migration-only C4 and run the proof. The proof and production promotion both +reject a serving director whose runtime digest differs from the cell/evidence digest. + +Expected project IDs: + +```text +staging: onorca-cloud-staging +production: onorca-cloud +``` + +Production relay delivery keeps the stable director separate from GCE cell rollout. `Publish Relay +Production Image` builds and prints an immutable digest. `Deploy Relay Production Director` accepts +only that digest, performs a health-gated Cloud Run director update, and never deploys data-plane +cell stamps. Its explicitly confirmed prune option retains only the serving and cold rollback pair, +and runs only after both compatible revisions pass the capacity-protocol health gate. Use that gate +before adding Asia cells so an incompatible dormant revision cannot be routed later. A reviewed +Terraform candidate pins the same digest on a distinct disabled GCE cell; +`Deploy Relay Production Candidate` then runs read-only preflight or an explicitly confirmed +target-first evacuation. Staging uses the same GCE data-plane shape as production; `Deploy Relay +Staging GCE Candidate` exercises the reviewed GCE preflight and evacuation state machine before +production use. + +`Prove Relay Staging Capacity` is the only cap-transition path. Apply mode +reversibly moves `staging-gce-c3` to migration-only, drains it, validates the +saved director and C3 Terraform plans, updates the director first, and requires +stale telemetry before replacing the exact C3 template and MIG. A fresh +matching heartbeat is required before C3 becomes the sole general placement +cell; C2 remains recoverable in migration-only. Restore mode does not depend on +Terraform or image agreement: it restores C2 first, then restores C3 only after +a fresh, healthy, non-draining capacity check. The same transition order +restores 600 before an older director image can be used. + +Its bounded C4 refresh mode keeps Asia admission migration-only, accepts only an exact predecessor +or already-applied target digest, validates a saved two-resource image-only plan, fences and proves +C4 empty before replacement, and requires an empty targeted readback afterward. It cannot change +C4 capacity, routing, trust configuration, or any production resource. + +`Recover Relay Staging C4 Image` runs independently after a failed, timed-out, or cancelled C4 +refresh and can also be dispatched with `RECOVER_STAGING_ASIA_C4_IMAGE`. It verifies the triggering +job and both exact Terraform end states, preserves a fully converged ready target or predecessor, +and fences partial state before restoring the pinned predecessor through an exact saved two-resource +plan. A separate no-credential supervisor requeues a recovery cancelled while waiting for the shared +staging mutation lane. Admin credentials are refreshed around Terraform. Before the first refresh, +target only +`google_iam_workload_identity_pool_provider.github_staging_relay_capacity`, require exactly one +in-place condition update, apply the saved plan, and require an empty targeted readback. + +This identity cannot bootstrap its own Relay authorization. Before the first +capacity dispatch, use the existing audited staging blue/green deploy path to +roll the compatible image and verified `ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT` +onto both director revisions. Roll C2/C3 through saved, validated cell plans +with `Bootstrap Relay Staging Capacity` while they remain at 600/60. That +workflow keeps the deploy identity only for Relay admin calls and uses the +capacity identity for Terraform and GCP mutations. It isolates, drains, rolls, +verifies, and restores one cell at a time, with the other cell as its failure +fallback. Commit the matching +C2/C3 image and capacity pairs in staging tfvars, then require the capacity workflow's read-only 600/60 +verification to pass. Only a later reviewed configuration commit may select +1,000/0 or 1,000/60. The capacity workflow carries the reviewed director +topology through blue/green; it never targets the drifted director or its Cloud +SQL dependencies with Terraform. + +The one-time bootstrap recognizes only the exact pre-capacity C2/C3 image and +its five-field runtime-status response. It first proves C2 as the general fallback, +then isolates and drains C3 and requires zero durable activity plus two fresh, +instance-bound zero runtime metrics. After restarting the fixed-one MIG, it +requires two new zero samples and a new director heartbeat incarnation started +after the restart before restoring C3. This clears the legacy process's +unreported drain flag without treating missing runtime fields as proof. Any +other image, response shape, activity, or stale evidence fails closed with C2 +preserved as the general fallback. Reruns classify partial C2/C3 progress. A +legacy target may carry only no capacity record or the exact stale 600/60 record +before the idempotent director update; afterward the exact stale record is +required until that cell is replaced. + +`Power Relay Staging` lowers the staging bill when no internal testing is underway. It runs a +guarded sleep attempt at 09:00 UTC every day and also supports manual `status`, `wake`, and `sleep` +dispatches. Manual mutations require the exact `WAKE_STAGING` or `SLEEP_STAGING` confirmation. +Sleep refuses to stop a cell with active Relay work, disables admission and checks again, then +scales the three GCE MIGs to zero and stops the shared staging Cloud SQL instance. Wake starts SQL, +waits for healthy workers and authenticated heartbeats, then restores only the admission state +declared in Terraform. The default wake starts c1/c2; choose `all` before a Terraform apply or GCE +candidate operation so the complete Terraform-owned topology is running. + +`Deploy Relay Production Multi-Target` handles a source that cannot fit on one +candidate. It serializes deterministic per-target quotas, enforces each +target's reviewed 600- or 1,000-connection gate and the ten-minute lease gate, +and treats a drain attempt as the +rollback point of no return. Fence and fence-abort remain fail-closed. +It also registers one additive migration cell and retires exactly one +migration-only cell through explicit, generation-bound selector operations. +Registered-target supersession invokes the IAM-only private broker, which owns +the durable mutation lease, exact Terraform checkout, saved plans, state, and +narrow Compute mutation. The workflow requester has read and broker-invocation +authority only; it never receives those mutation permissions directly. +`Deploy Relay Fence Broker` updates only that service's immutable image and +requires the digest to carry the exact `sha-${GITHUB_SHA}` tag. Terraform +continues to own its identity, scaling, IAM, environment, and deletion +protection. +Its `add-migration-cells` mode is the selector-safe path for newly provisioned +empty targets after generation 1. It requires `ADD_MIGRATION_CELLS` and a +stable selector attempt ID, but no pre-drain artifact because it moves no +assignments. Run the fresh 15-minute gate only after the new cells are +registered and healthy. + +## Cloud SQL rollout lease + +Every workflow that mints a Cloud Run revision or applies a relay instance template against a shared +Cloud SQL instance takes the compare-and-swap lease in `.github/actions/cloud-sql-rollout-lease` +immediately after `google-github-actions/setup-gcloud`. The per-repository `concurrency` groups +(`production-cloud-sql-rollout`, `relay-staging-mutation`) only serialize runs inside one repository; +once the relay workflows live in `stablyai/orca` there are two queues pointed at one instance, and +`relay-cloud-sql-connection-budget.mjs` computes `rolloutOverlap` as a `Math.max` that is only sound +with one rollout in flight. Keep both the groups and the lease. + +| Environment | Bucket | Object | +| ----------- | -------------------------------------- | --------------------------------------------------- | +| production | `onorca-cloud-terraform-state` | `terraform/state/cloud-sql-rollout/production.lock` | +| staging | `onorca-cloud-staging-terraform-state` | `terraform/state/cloud-sql-rollout/staging.lock` | + +`Deploy Relay Asia Topology` and `Operate Relay Asia Admission` pick the pair from +`inputs.environment`. `Deploy Relay Production Capacity` and `Deploy Relay Production Same-Cap` call +their reusable job several times per run, so every wave job acquires with `release: 'false'` under +the run-scoped default holder key and a single `if: always()` `release_lease` job frees it once every +wave has finished. + +`Monitor Relay Production` stays off the lease. It is read-only, holds only viewer roles, and putting +it on a durable lease would let monitoring block a rollout and a rollout block monitoring. +`dev/scripts/production-cloud-sql-rollout-lock.test.mjs` enforces the group, the lease wiring, and a +content-derived census of every rollout candidate against +`dev/scripts/cloud-sql-rollout-lock-census.mjs`. + +`Monitor Relay Production` is manual and read-only. Its `dry-run` mode enforces the 15-minute +pre-drain gate; `monitor` records a 90-minute incident watch. Both require the +operator to enter the exact selector generation and tri-state membership. The +workflow must use a dedicated identity for aggregate monitoring and +exact-audience read-only Relay-admin calls. Do not dispatch it until that +monitor identity, exact workflow-bound WIF trust, and read-only admin-route +authorization have been bootstrapped. +Capacity-transition monitoring binds the evidence to one exact general cell. It +still blocks all migration failures and any inactive registered migration from +that cell or another serving cell; it permits only inactive rows +from unrelated existing-only cells because a capacity restart neither creates +nor advances assignment migrations. +Reruns restore hash-verified private state from the prior attempt. Production +candidate and multi-target mutations require a fresh dry-run artifact and +recheck its exact selector and every live safety signal before any mutation +command. All three workflows share the production deployment lock, and each +passing dry-run artifact is marked consumed before the mutation starts. +The dry-run lineage fails closed after 25 total minutes, so continuity resets cannot extend the +15-minute gate indefinitely. +Missing or stale telemetry fails closed, and the workflow uploads only private aggregate +Markdown/JSON evidence. + +`Deploy Relay Production Capacity` is the only production cap-transition path. It runs only +from `main` and accepts exactly the current general rollout set: C7-C10, C13-C16, and C19-C26. +C17/C18 and every existing-only, draining, fenced, or disabled cell are excluded in code. Its +Terraform/GCE phase uses the dedicated exact-workflow capacity identity. Read-only checks, selector +isolation, drain, and the audited director blue/green update use the existing shared production +deploy identity; the production environment and common deployment lock still gate those steps. +Apply mode consumes a fresh 15-minute monitor gate bound to the selected cell, moves only that cell +from general to migration-only, drains it, updates only its director capacity entry, and applies a +saved validated plan for only its template and MIG. Previously completed 1,000 cells remain +unchanged while later 600 cells roll. The selected cell returns to general only after a fresh +matching 1,000/60 heartbeat. The restart gate waits up to 15 minutes for genuine activity to finish +while preserving every zero-work check. Rollback performs the same isolated sequence to 600/60 +without waiting on a cell that may already be unhealthy. If the selected cell cannot answer the +drain call, rollback instead requires two stale-heartbeat snapshots with zero durable activity +before replacing it. Its typed confirmation includes the exact selected cell so a form-selection +mistake cannot downgrade another cell. Interrupted Terraform applies resume only when the planned +current template has the exact reviewed image, capacity, and identity and the remaining change is +that selected MIG update or obsolete-template deletion. Production configuration pins only the +approved serving set to the compatible image and 1,000/60; the transition classifier accepts only +the reviewed mixed 600/1,000 envelope until every selected cell converges. Every GCP-only Terraform +command disables artifact DNS. Any failed mutation leaves only the selected cell migration-only and +never changes another cell's selector state. + +After multiple production cells pass the canary path, `wave-apply` may raise two to four reviewed +600/60 serving cells under one fresh 15-minute capacity-transition gate. The first cell is bound to +the sealed evidence; every later cell derives the exact expected selector generation and reruns the +complete live preflight before mutation. After the first cell, continuation preflights retry only +missing or stale signal evidence for at most one minute; health, threshold, selector, and migration +failures stop immediately. Cells still drain, restart, and verify sequentially. A +failed cell stays isolated and prevents every later wave job from starting; earlier completed cells +remain general at 1,000/60. The workflow lock, single-use evidence marker, exact predecessor check, +targeted Terraform plan, and per-cell heartbeat/admission oracle are unchanged. + +`Deploy Relay Production Same-Cap` rolls only the reviewed US 1,000/60 and Asia 3,000/60 serving +sets without changing a cell's connection shape. Use `canary-apply` for exactly one cell. A successful canary +seals its commit, target and rollback digests, selector generation, and durable rehome generation; +`batch-apply` accepts only that same authority and rolls two to four cells sequentially. Each cell is +isolated, drained to two restart-safe samples, replaced from a targeted saved plan, and restored only +after a new incarnation reports the exact digest, cap, heartbeat, and rehome protocol. The durable +worker must remain disabled throughout. The post-restart trust check is application-mediated by the +director; the workflow never receives or mints a director or stamped-cell runtime token. A failure +keeps only the selected cell migration-only, while the exact rollback digest remains dispatchable via +the same workflow's `rollback` mode. + +The first compatible director rollout uses `bootstrap-runtime-identity=true` with +`BOOTSTRAP_RELAY_DIRECTOR_REHOME_IDENTITY`. That one-time path requires the exact stamped-cell +predecessor identity, creates both the cold rollback and candidate on the distinct director identity, +and proves the disabled durable control through those compatible revisions before moving traffic. +Later director deploys reject the predecessor identity and verify the disabled control on the serving, +rollback, and candidate revisions. + +`Operate Relay Production Rehome` is the only durable worker control. `inspect` is read-only; +`enable` is selector-, director-digest-, rollback-digest-, and control-generation-bound, starts at +exactly 10 hosts per minute, consumes the fresh 15-minute safety monitor, and seals 24 hourly buckets +of aggregate requested-region, selected-region, fallback, and unavailable-region evidence with +positive Asia requests and selections. `pause` and `disable` apply their generation CAS immediately +after checkout and authentication, before package installation, revision checks, or log diagnostics. +Their typed confirmations are `PAUSE_REGIONAL_REHOMING` and `DISABLE_REGIONAL_REHOMING`. Keep the +default 3,600,000 ms drain grace so existing splices can finish. The job summary contains only fresh +aggregate active, receipt, registration, completion, and abort counts. diff --git a/cloud/docs/staging-relay-deploy-identity-rollout.md b/cloud/docs/staging-relay-deploy-identity-rollout.md new file mode 100644 index 00000000000..81fd3efe44a --- /dev/null +++ b/cloud/docs/staging-relay-deploy-identity-rollout.md @@ -0,0 +1,177 @@ +# Staging Relay deploy identity rollout + +Moves the five staging Relay workflows off the apps-owned `github_deploy` +account and onto the relay-owned `github_staging_relay_deploy` account +(`orca-cloud-staging-gha-relay`). This is a **rollout**, not state surgery, so +it lives beside [`terraform-root-split-runbook.md`](./terraform-root-split-runbook.md) +rather than inside it: that runbook is state-only and runs no apply, and every +step below applies. + +Production is untouched. `local.relay_github_deploy_service_account_email` +still renders `orca-cloud-gha-deploy` there, and the production relay plan slice +is byte-identical to the pre-split single-root baseline. + +## What moves, and what it costs + +The staging Relay runtime allowlists exactly one deploy account +(`ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT`, `apps/relay/src/admin-token-verifier.ts`), +so the account, the director env, the four cell startup scripts, and the +workflow variables have to change together. Plan on a staging window in which +no Relay workflow runs. + +| Change | Cost | +| --- | --- | +| 10 new identity resources | Create only. No compute. | +| 6 relay bindings repoint to the new account | Delete + create. The old account loses them. | +| `ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT` on the director | New Cloud Run revision. | +| Cells c1–c4 startup metadata | Four instance-template replacements, one MIG repoint each. | + +## Preconditions + +- [ ] PR 13 is merged, so both accounts are declared and the census test passes. +- [ ] No staging Relay workflow is running or queued. All five share the + `relay-staging-mutation` concurrency group; `prove-relay-staging-capacity` + and `recover-relay-staging-c4-image` are in it too. +- [ ] Staging is awake, or you accept waking it as part of step (e). + +## (a) Compute-free identity apply + +Targeted apply on the staging relay root. Verified against a post-surgery state +copy: **10 creates, nothing else**. + +```sh +terraform -chdir=infra/terraform apply \ + -var-file=environments/staging.tfvars \ + -target='google_service_account.github_staging_relay_deploy[0]' \ + -target='google_iam_workload_identity_pool_provider.github_staging_relay_deploy[0]' \ + -target='google_service_account_iam_member.github_staging_relay_deploy_workload_identity_user[0]' \ + -target='google_storage_bucket_iam_member.github_staging_relay_deploy_state_list[0]' \ + -target='google_storage_bucket_iam_member.github_staging_relay_deploy_state[0]' \ + -target='google_artifact_registry_repository_iam_member.github_staging_relay_deploy_artifact_reader[0]' \ + -target='google_cloud_run_v2_service_iam_member.github_staging_relay_deploy_director_developer[0]' \ + -target='google_cloud_run_v2_service_iam_member.github_staging_relay_deploy_auth_developer[0]' \ + -target='google_service_account_iam_member.github_staging_relay_deploy_auth_runtime_user[0]' \ + -target='google_project_iam_member.github_staging_relay_deploy_compute_viewer[0]' +``` + +Reject the plan if it shows anything but those ten creates. + +## (b) Publish the two variables + +```sh +terraform -chdir=infra/terraform output -raw github_staging_relay_deploy_workload_identity_provider +terraform -chdir=infra/terraform output -raw github_staging_relay_deploy_service_account + +gh variable set STAGING_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER --env staging --body '' +gh variable set STAGING_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT --env staging --body '' +``` + +Nothing reads them until (c) merges, so this step is reversible on its own. + +## (c) Repoint the workflows and flip the relay bindings + +The workflow repoint ships in PR 13. Merging it and applying the six repointed +bindings is one step, because the new account cannot operate without them and +the old account must not keep them: + +```sh +terraform -chdir=infra/terraform apply \ + -var-file=environments/staging.tfvars \ + -target='google_project_iam_member.github_staging_relay_power[0]' \ + -target='google_service_account_iam_member.github_relay_runtime_service_account_user[0]' \ + -target='google_service_account_iam_member.github_relay_director_runtime_service_account_user[0]' \ + -target='google_secret_manager_secret_iam_member.relay_regional_placement_deploy_accessor[0]' \ + -target='google_secret_manager_secret_iam_member.relay_regional_placement_deploy_adder[0]' \ + -target='google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer[0]' +``` + +Expect **six replacements plus one create**: the target on +`github_relay_director_runtime_service_account_user` drags +`google_service_account.relay_director_runtime`, which staging has never +created. That create is additive and does not move the director onto the new +identity; only applying `google_cloud_run_v2_service.relay` does that. Drop +that one `-target` if you would rather leave it to the staging drift +remediation, and accept that the new account then has no `serviceAccountUser` +on the director runtime account. + +The director's `ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT` changes only through +`google_cloud_run_v2_service.relay`, and applying that address in staging also +carries the whole staging director backlog: the identity swap to +`orca-cloud-staging-relay-dir`, `timeout` 3600s to 30s, concurrency 1000 to 80, +the four-cell topology, and roughly twenty new env entries. Do it as part of +the staging identity remediation in the drift plan, in this same window, with +that plan reviewed on its own terms. + +## (d) Roll the cells, one at a time + +Use **Prove Relay Staging Capacity**. It authenticates only as +`STAGING_GCP_RELAY_CAPACITY_*`, that is `github_staging_relay_capacity`, never +the new deploy account, and the capacity identity's admin routes +(`RELAY_CAPACITY_ADMIN_ROUTES`) cover every call the roll makes. It is +therefore unaffected by this cutover in either direction. + +Per cell the targeted apply is: + +```text +-target='google_compute_instance_template.relay_gce_cell["staging-gce-c"]' +-target='google_compute_instance_group_manager.relay_gce_cell["staging-gce-c"]' +``` + +That plan is one template replacement plus one MIG update, and it also drags an +in-place update of `google_compute_health_check.relay_gce_liveness[0]` from the +standing staging log_config drift. Confirm it is in place, not a replacement. + +Two gaps to plan around: + +- `prove-relay-staging-capacity` has arms for **c3 and c4** only. +- `bootstrap-relay-staging-capacity` rolls **c2 and c3**, but it mints its admin + token as the deploy account. Running it across the email change fails: it + verifies a rolled cell with a token that cell no longer allowlists. +- **c1 has no workflow roller.** Roll c1, and c2 if you do not use bootstrap, + by the same two-target apply under human review inside the window. + +Wait for each MIG to report stable before starting the next cell. + +## (e) Verify + +Dispatch **Power Relay Staging** in `status` mode. It exercises the new +credential end to end: Workload Identity exchange on the new provider, the +state read, `gcloud sql instances describe` and the MIG reads through +`orcaRelayStagingPower`, `run services describe` on both the director and the +shared staging auth service, and an admin `cell-status` call that only succeeds +if the director allowlists the new account. Then run a `sleep` and a `wake` to +exercise the mutation paths. + +## (f) Retire the staging Relay grants on the shared account + +Follow-up PR against `infra/terraform-apps`. Once (e) passes, the staging +`github_deploy` account no longer needs the Relay-only reach it has today. +Narrow, in the apps root: + +- `google_project_iam_member.github_compute_viewer` — kept only for the Relay + candidate preflight; no staging app workflow reads GCE topology. +- `google_project_iam_member.github_cloud_run_developer` — project-wide today; + the Relay services are now covered by the two service-scoped grants above, so + the apps root can scope it to the API and auth services. +- `google_project_iam_member.github_artifact_writer` — still needed by the app + deploys; verify before touching. + +Leave alone: the AR mirror writer +(`google_artifact_registry_repository_iam_member.github_production_relay_staging_mirror_writer`) +names the **production** deploy account by literal and is unrelated to this +change; `github_runtime_service_account_user` and +`github_auth_runtime_service_account_user` are still used by the staging app +deploys. + +## Rollback + +Before (c): revert the two variables, or unset them. The job gates skip while +they are empty, and the old account still holds every grant. + +After (c) but before the cells are rolled: re-apply the six bindings from the +previous commit, which points them back at `orca-cloud-staging-gha-deploy`, and +revert the workflow repoint. The new account and its provider can stay; they +grant nothing the old path needs. + +After the cells are rolled: roll forward. Rolling four templates back costs the +same as rolling them forward and leaves the same window. diff --git a/cloud/infra/terraform/.terraform.lock.hcl b/cloud/infra/terraform/.terraform.lock.hcl new file mode 100644 index 00000000000..13cb0e4644e --- /dev/null +++ b/cloud/infra/terraform/.terraform.lock.hcl @@ -0,0 +1,67 @@ +# This file is maintained automatically by "terraform init". +# Manual edits may be lost in future updates. + +provider "registry.terraform.io/hashicorp/external" { + version = "2.4.0" + constraints = "~> 2.3" + hashes = [ + "h1:AmY6ZeIvqoTT5ZjzD+P49PeQH6Va1QLMkX+7MUQfYoA=", + "h1:gXyK3ZkweDqkwEV9waEDWpljUY4yZXi/nRUSEuJn83k=", + "zh:0772afb42b658468ac5e15df33bf2080456f8f0b8ab163bfe9c50d2b2ea02135", + "zh:0ac31a9aaa43dfcff5944b791596cdc94e153348e4bb4642282d034dff548134", + "zh:32d8492b1bdcc956ca3c6d00c6392d0a83942ff11d4820c7ee63ca6796e06950", + "zh:3c0482e894429f528ce6655a76ab0d8a9f7c0dacc6c828865e1515d4a7dbb852", + "zh:61e68100b4db2f930b31491f23c602126382fd5e51252be1b551f0e17f8ddbee", + "zh:6d60f615a0ad85eb962c9eb94f25e3eba7a72684ce276ba5dfb23f36b295a8f8", + "zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3", + "zh:9ced2745eb5f1346203027d2dd7bf856ad1d279a25730ff7dbc6eec187aaca0c", + "zh:a8378558a177d43f55aa0d79d4fae91a704695122a1b109668c1daa8fb76f09d", + "zh:aadd98086133d3ebea67437d56512fdcc6dfb3bd34dfc23f276c0db9272e27b4", + "zh:beff701b653841e70441978137768f54e7dc6c27e7bf12a4589087f01f5bbcee", + "zh:c91c2223b29fdbc0044d20e1936ccc051d010727a13f2ff1e75e51f09bff33a3", + "zh:d491f9c2d32a39dc4031628469ae7c8aec0074312a7c1f0286b173cdcf854a54", + ] +} + +provider "registry.terraform.io/hashicorp/google" { + version = "6.50.0" + constraints = "~> 6.0" + hashes = [ + "h1:79CwMTsp3Ud1nOl5hFS5mxQHyT0fGVye7pqpU0PPlHI=", + "h1:mhrmzHgoQoall8+7hA9Lpy0HAnjNC1N5+sPDp6bGizM=", + "zh:1f3513fcfcbf7ca53d667a168c5067a4dd91a4d4cccd19743e248ff31065503c", + "zh:3da7db8fc2c51a77dd958ea8baaa05c29cd7f829bd8941c26e2ea9cb3aadc1e5", + "zh:3e09ac3f6ca8111cbb659d38c251771829f4347ab159a12db195e211c76068bb", + "zh:7bb9e41c568df15ccf1a8946037355eefb4dfb4e35e3b190808bb7c4abae547d", + "zh:81e5d78bdec7778e6d67b5c3544777505db40a826b6eb5abe9b86d4ba396866b", + "zh:8d309d020fb321525883f5c4ea864df3d5942b6087f6656d6d8b3a1377f340fc", + "zh:93e112559655ab95a523193158f4a4ac0f2bfed7eeaa712010b85ebb551d5071", + "zh:d3efe589ffd625b300cef5917c4629513f77e3a7b111c9df65075f76a46a63c7", + "zh:d4a4d672bbef756a870d8f32b35925f8ce2ef4f6bbd5b71a3cb764f1b6c85421", + "zh:e13a86bca299ba8a118e80d5f84fbdd708fe600ecdceea1a13d4919c068379fe", + "zh:f569b65999264a9416862bca5cd2a6177d94ccb0424f3a4ef424428912b9cb3c", + "zh:fec30c095647b583a246c39d557704947195a1b7d41f81e369ba377d997faef6", + ] +} + +provider "registry.terraform.io/hashicorp/random" { + version = "3.9.0" + constraints = "~> 3.6" + hashes = [ + "h1:OO+IuvQJSPmWdN8AyyIEvPJbLvDQpgX/zbktoa9KsJE=", + "h1:lVDv+0AjDjrLfpmaJbWqUmIw/k3/AHXLc3N4m55SNdo=", + "zh:161ad0bd9a75768c82f53fb6e7172a9d8be2d4889b012645a34795031aaf1bf1", + "zh:19dc9a5b17729725ccfc4f45b0500af0ee5bc6b6b160c7adb8f2bf617d2c80ea", + "zh:269eda8fe42daa7974d5a34d166c3ba9defe80cde86c01e4dadcfdf2e1f05e5f", + "zh:373f7c65566f8f2cc7f45d698654feb9d988996957e1266a69ca00c52d6d16d0", + "zh:5599d16804c41c83009ec621b6d6b6f74e102f5827678a4750f8809055546b61", + "zh:583be0440469a22bff70dcfa56593b01566860b29607437264adb51060cf46fc", + "zh:5f211d8ec3f2e1f414870d9584bfe26e6995560ef81c748f8447a48164767398", + "zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3", + "zh:7b547fd16216761ef86efc3ed516ac5ac0c5c42b7c7eb24a08cef2d93f69ed5e", + "zh:7e7c0679daf2a382151d05068c8c3f0dae6b7b7dccf818827b73dd08638df2ef", + "zh:8089dec888a8038b9b4fb23b3df7e1057293dbc5b60b42cc47ff690d69d4b61b", + "zh:c51f15a031edfd6f23ce8ced3446ca7f8d8d647e2499890d7d5d10d5016d7257", + "zh:c94784f005708890dc6895afd53636ec00ec1e430b15d41e5aebfb1d4b39bd04", + ] +} diff --git a/cloud/infra/terraform/README.md b/cloud/infra/terraform/README.md new file mode 100644 index 00000000000..2e2b9b02011 --- /dev/null +++ b/cloud/infra/terraform/README.md @@ -0,0 +1,409 @@ +# Terraform + +This root manages the Orca Cloud relay and nothing else. It requires Terraform >= 1.7 +(`removed` blocks); OpenTofu at that floor works too. + +## Three roots + +Orca Cloud is three Terraform roots sharing one project and one state bucket per environment, +with a different prefix each. They are separate so the relay can be extracted into a public +repository without carrying the app plane, its database passwords, or its Cloudflare credential +with it. + +| Root | Directory | State prefix | Owns | +| --- | --- | --- | --- | +| foundation | `infra/terraform-foundation` | `terraform/foundation` | Project service enablement, the Artifact Registry repository, the API runtime service account, the Cloud SQL instance, the GitHub Workload Identity pool, the Cloud SQL rollout lease grant | +| apps | `infra/terraform-apps` | `terraform/apps` | The API and auth services, the artifact and skill-package buckets, the skill plane and its observability, auth and artifact DNS, the app deploy identities | +| relay | `infra/terraform` | `terraform/state` | Everything relay: the director, GCE cells, the fence broker, relay observability, and the relay operator identities | + +Apply order on a greenfield project is **foundation first**, then relay and apps in either order. +The other two roots reach foundation only by literal or by `data` lookup, never through +`terraform_remote_state`: foundation state holds the Cloud SQL instance and apps state holds +generated database passwords in cleartext, and the relay root must not acquire a read path into +either once it is public. Each root's substitution for a foundation value is pinned by +`dev/scripts/terraform-root-partition.test.mjs`, which asserts every declared resource family is +owned by exactly one root per environment. + +Three families are owned per environment rather than outright: the shared deploy service account, +its Workload Identity provider, and its WIF binding live in the relay root for production and in +the apps root for staging, with complementary counts. Every IAM binding on that account follows +it. `dev/fixtures/terraform-root-partition/families.json` is the authority. + +### The carve is complete + +Both state surgeries have run (`docs/terraform-root-split-runbook.md`), so this root's state holds +only relay families and the `removed` guard blocks from the window are gone. The shared deploy +identity (`google_service_account.github_deploy`, its provider, and its bindings) is declared here +with production-only counts; staging's copies are declared by `infra/terraform-apps`. An untargeted +plan is orderable again; the `Plan:` line still reflects the standing cell-template drift backlog. + +### Dual-accept Workload Identity during the public extraction + +While the relay source moves to the public `stablyai/orca` repository, every relay Workload +Identity provider accepts the same workflows from both repositories. `github_accepted_repositories` +lists the extra repositories; `relay-github-workflow-trust.tf` renders one parenthesised OR arm per +accepted repository, each arm carrying that repository's own `repository`, `repository_id`, and +`repository_owner_id` claims plus its exact workflow refs. `ref`, `environment`, and `event_name` +stay outside the OR. Workflow files keep their names in the private repo and take the +`workflow_file_prefix` (`cloud-`) in the public one. + +Adding a repository is a tfvars edit: no provider block changes, and the rendered strings are +pinned by `dev/scripts/workload-identity-attribute-conditions.test.mjs`. An empty list renders +byte-identically to the single-repository form, which is what makes the arms reviewable against +the pre-extraction condition. + +Closing the cutover is an owner step, in this order: + +1. Retire the private workflows, so nothing runs from `stablyai/orca-cloud` any more. +2. Point `github_owner`, `github_repo`, `github_repo_id`, and `github_owner_id` at + `stablyai/orca` (`1183888342`, owner `127256420`), and set `workflow_file_prefix` for it by + moving the surviving entry's prefix onto the primary: the public files keep the `cloud-` names, + so the primary prefix becomes `cloud-` unless the files are renamed back. +3. Empty `github_accepted_repositories` in both `environments/*.tfvars`. +4. Re-render and update the pinned conditions, then apply. Each provider goes back to a single + arm, and `google_service_account_iam_member.github_accepted_repository_workload_identity_user` + is destroyed as the primary `attribute.repository` binding takes over. + +Step 2 and step 3 must land in the same apply: dropping the accepted entry before repointing the +primary would revoke the public repository mid-flight. + +### `ORCA_RELAY_IMAGE_DIGEST` is not Terraform-owned + +`deploy-relay-blue-green.mjs` sets `ORCA_RELAY_IMAGE_DIGEST` on the director container at deploy +time, but `relay.tf` does not declare it and the director's `ignore_changes` cannot name a single +list element. A director apply from this root therefore strips that variable. Terraform is not the +owner today: deploy through the director workflow, and treat any direct +`google_cloud_run_v2_service.relay` apply as something that needs the next deploy to restore the +digest. Giving Terraform the variable (a declared input the deploy script writes through) is +tracked as follow-up work in the split checklist, not in this change. + +Select a root with `--root`; omitting it keeps the relay root, so existing callers are unchanged. + +```sh +pnpm infra:init --env staging --root foundation +pnpm infra:plan --env staging --root apps +``` + +## Bootstrap Remote State + +The GCS backend bucket must exist before `init`. + +Staging: + +```sh +gcloud storage buckets create gs://onorca-cloud-staging-terraform-state --project onorca-cloud-staging --location us +gcloud storage buckets update gs://onorca-cloud-staging-terraform-state --versioning +``` + +Production: + +```sh +gcloud storage buckets create gs://onorca-cloud-terraform-state --project onorca-cloud --location us +gcloud storage buckets update gs://onorca-cloud-terraform-state --versioning +``` + +One bucket per environment holds all three roots' state under separate prefixes, so this is a +one-time step for the whole project. + +Do not commit `.tfstate`, `.tfplan`, or `.terraform` files. + +## Production app deploy identity: moved + +The production app deploy identity, the skill alert channel guard, and every +other app-plane resource now live in `infra/terraform-apps`. Their bootstrap +procedure moved with them; run it with `-chdir=infra/terraform-apps`. This root +no longer declares the API service, the auth service, the artifact or +skill-package buckets, the Cloud SQL databases, or the app DNS records, and it +no longer needs a Cloudflare or 1Password credential. + +## Staging Relay capacity identity bootstrap + +Before the capacity workflow can mutate staging, create a saved targeted plan +containing only `github_staging_relay_capacity` providers, accounts, roles, +bindings, and outputs. Apply it with backend locking, then require a targeted +refresh/no-op plan. The identity is bound to the exact workflow on `main` and +the `staging` environment. Its state write access is limited to the default +staging state and lock object prefix. + +Copy these outputs into same-named staging GitHub environment variables: + +1. `github_staging_relay_capacity_workload_identity_provider` +2. `github_staging_relay_capacity_service_account` + +Before dispatch, prove it can read the saved state and reviewed Relay +resources, but cannot change unrelated Cloud Run services, templates, managed +instance groups, databases, DNS, or secrets. + +The identity bootstrap alone does not authorize Relay admin routes. Publish the +compatible image, then use the existing staging blue/green deploy path to carry +and verify the capacity service account on both director revisions. Use saved, +validated cell plans through `Bootstrap Relay Staging Capacity` to roll C2/C3 +at 600/60. The reviewed staging tfvars must pin the same image and capacity. +Require a read-only 600/60 capacity-workflow +run before reviewing either 1,000-policy configuration. Do not target the +director with Terraform: its dependency closure includes unrelated live drift. + +## Production Relay incident identity bootstrap + +Before running the production monitor, create a saved targeted plan containing +only the two dedicated service accounts, their exact-workflow +providers/bindings, and monitor read roles. Reject any Cloud Run, GCE, +database, network, runtime-service-account, or unrelated IAM change. Apply +that plan with backend locking, then run a targeted refresh/no-op plan. + +Copy these outputs, in order, into same-named production GitHub environment +variables documented in `.github/workflows/README.md`: + +1. `github_relay_monitor_workload_identity_provider` +2. `github_relay_monitor_service_account` +3. `github_relay_fence_workload_identity_provider` +4. `github_relay_fence_service_account` + +Use an audited operator session for the GitHub variable writes. Before +dispatch, prove the monitor account can read required aggregate telemetry but +cannot mutate Relay or state. Fence modes remain disabled until a separate +private broker owns and validates the exact state, plan, cell, and durable +attempt boundary; never grant direct Compute update or Terraform-state write +access or director mutations to the GHA fence account. + +## Relay Asia topology identity bootstrap + +Bootstrap each environment's Asia topology identity with operator credentials +before dispatching its workflow. IAM cannot bootstrap itself. Reinitialize the +exact backend and save a targeted plan that +contains only these twelve additive resources: + +1. `google_iam_workload_identity_pool_provider.github_relay_asia_topology` +2. `google_service_account.github_relay_asia_topology` +3. `google_service_account_iam_member.github_relay_asia_topology_workload_identity_user` +4. `google_project_iam_custom_role.github_relay_asia_topology_mutation` +5. `google_project_iam_member.github_relay_asia_topology_mutation` +6. `google_project_iam_custom_role.github_relay_asia_topology_read` +7. `google_project_iam_member.github_relay_asia_topology_read` +8. `google_artifact_registry_repository_iam_member.github_relay_asia_topology_artifact_reader` +9. `google_storage_bucket_iam_member.github_relay_asia_topology_state` +10. `google_project_iam_custom_role.github_relay_asia_topology_state_list` +11. `google_storage_bucket_iam_member.github_relay_asia_topology_state_list` +12. `google_service_account_iam_member.github_relay_asia_topology_runtime_user` + +The state-list role contains only `storage.objects.list`. Terraform's GCS backend +needs that bucket-level permission before it can access the exact state and lock +objects protected by the conditional object-admin binding. + +Production also requires one exact in-place update to +`google_iam_workload_identity_pool_provider.github[0]` so the existing deploy +identity accepts `operate-relay-asia-admission.yml`; staging's shared provider +already accepts repository workflows. Reject every other change. Apply only +that saved plan, then require the same targeted plan to be empty. Publish the two +`github_relay_asia_topology_*` outputs as the matching staging or production +GitHub environment variables documented in `.github/workflows/README.md`. + +Apply observability separately from IAM and topology. The topology identity +has no IAM, logging-metric, alert-policy, Cloud SQL, DNS, certificate, global +IP, or deletion permission. Its read role includes `serviceusage.services.list` +because the Google provider lists managed APIs while refreshing targeted plans. +Its mutation role includes `compute.networks.updatePolicy`, which Compute requires +to attach the reviewed Asia subnet and router to the existing Relay VPC. +It also includes `compute.healthChecks.useReadOnly`, which backend creation requires +to reference the existing Relay readiness health check. +Managed-group creation additionally requires `compute.instanceGroups.create`; adding +that group as a backend requires `compute.instanceGroups.use` and `compute.instances.use`. + +Bootstrap the staging Asia proof identity separately before its director roll. +Its targeted plan contains only the proof provider, service account, +workload-identity binding, logging/monitoring viewer bindings, and two outputs. The provider +accepts only `prove-relay-asia-staging.yml` on `main` in the staging environment; +the account has no Compute, Cloud SQL, Secret Manager, Terraform-state, or +Cloud Run mutation permission. Publish its provider and account outputs as +`STAGING_GCP_RELAY_ASIA_PROOF_WORKLOAD_IDENTITY_PROVIDER` and +`STAGING_GCP_RELAY_ASIA_PROOF_SERVICE_ACCOUNT`, then deploy the compatible +staging director so it accepts that exact account for bounded capacity routes. + +## Relay regional-placement switch bootstrap + +Before the first director deployment that references the regional-placement +switch, apply its Secret Manager resources with operator credentials. Save a +targeted plan containing exactly these six additions and no other changes: + +1. `google_secret_manager_secret.relay_regional_placement_enabled` +2. `google_secret_manager_secret_version.relay_regional_placement_enabled` +3. `google_secret_manager_secret_iam_member.relay_regional_placement_runtime_accessor` +4. `google_secret_manager_secret_iam_member.relay_regional_placement_deploy_accessor[0]` +5. `google_secret_manager_secret_iam_member.relay_regional_placement_deploy_adder[0]` +6. `google_secret_manager_secret_iam_member.relay_regional_placement_deploy_viewer[0]` + +Pass the exact environment tfvars, apply only +the saved plan, then require the same targeted plan to be empty. Verify the +runtime and deploy identities can access the secret without printing its value, and the deploy +identity can read version metadata without gaining broader mutation rights. +Only then deploy a director revision. Every director revision pins one exact +numeric secret version; the audited director workflow preserves the serving +version by default and creates a new boolean version only for an explicit +enable or disable. The traffic move is therefore the switch commit, and a +failed candidate cannot change the value used by serving instances. +Terraform reads and preserves the currently served director's exact numeric +version, falling back to the bootstrap version only before the setting exists. +It still owns the secret name and every environment field; an unrelated apply +therefore cannot revert a later audited switch version. + +## Relay director runtime identity bootstrap + +Before the regional-rehome director rollout, create the distinct director +runtime identity with operator credentials. Reinitialize the exact environment +backend, export a fresh `GOOGLE_OAUTH_ACCESS_TOKEN` without printing it, pass +the environment tfvars, and save a targeted +plan containing only the applicable resources below: + +1. `google_service_account.relay_director_runtime` +2. `google_project_iam_member.relay_director_runtime_cloudsql_client` +3. `google_secret_manager_secret_iam_member.relay_assignment_signing_key_director_accessor` +4. `google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor` +5. `google_secret_manager_secret_iam_member.relay_database_url_director_accessor` +6. `google_service_account_iam_member.github_relay_director_runtime_service_account_user[0]` +7. `google_iam_workload_identity_pool_provider.github[0]` in production when its + condition adds the exact regional-rehome and same-cap workflow/job pairs +8. `google_iam_workload_identity_pool_provider.github_production_relay_capacity[0]` + in production when its condition adds the exact same-cap workflow/job pair + +Reject Cloud Run, GCE template, database, network, DNS, or any other change. +Apply only the reviewed saved plan, then require the same targeted plan to be +empty. Publish `relay_director_runtime_service_account` and +`relay_runtime_service_account` as the matching GitHub environment variables +documented in `.github/workflows/README.md`. The identity bootstrap does not +authorize a rollout by itself; use the one-time director workflow mode so both +candidate and rollback revisions move together while rehoming remains durably +disabled. + +## Usage + +```sh +pnpm infra:init --env staging +pnpm infra:plan --env staging +pnpm infra:apply --env staging +``` + +Add `--root foundation` or `--root apps` for the other two roots; the default is the relay root. +On a greenfield project apply foundation before either of the others. + +Run staging first. Production should only follow after staging has a successful `/health` smoke test. + +## Relay staging topology + +The stable Cloud Run service is the director. Staging uses fixed-one GCE cells so its data plane +matches production; Cloud Run stamped cells are no longer retained in the live environment. + +Staging is intentionally allowed to drift to a powered-off runtime state between internal test +windows. Use the `Power Relay Staging` GitHub Actions workflow to inspect, wake, or sleep it. A +normal `pnpm infra:apply --env staging` refuses while Cloud SQL is stopped or any staging MIG is +scaled below its Terraform-owned size of one. Dispatch `wake` with `wake-cells: all`, wait for its +health checks, and only then apply a reviewed staging plan. Do not use Terraform to wake staging: +that can mix infrastructure changes with a partial power transition. + +The staging tfvars keep both Cloud Run services at zero minimum instances. Requests wake the auth +service and director when SQL is running; the power workflow separately controls SQL and the GCE +MIGs. The staging-only GitHub service-account role can resize those MIGs and change the SQL +activation policy. Terraform does not create that role in production. + +## Relay GCE production data plane + +`relay_gce_domain` creates the shared private network/NAT, LB address, and Certificate Manager +wildcard authorization used by fixed-one GCE cell MIGs. Cells use exact hosts one label below the +domain, such as `c1.relay-staging.onorca.dev`; future cells therefore reuse one DNS-only wildcard +A record while the HTTPS URL map still admits only Terraform-configured exact hosts. + +After the foundation apply, publish both Terraform outputs and leave them in place for renewal: + +1. `relay_gce_certificate_dns_authorization`: the exact Certificate Manager CNAME. +2. `relay_gce_wildcard_dns_record`: the DNS-only wildcard A record to the reserved LB address. + +Every `relay_gce_cells` entry is one durable cell generation and must pin both its exact COS boot +image and its Artifact Registry relay image. Terraform creates one private COS instance template, one size-one zonal +MIG, and one backend service for that exact host. The MIG uses `RECREATE`, zero surge, and one +unavailable worker; `/health` alone drives autoheal while SQL/JWKS-backed `/ready` controls LB +admission. The backend timeout is 86,400 seconds with connection draining, and the URL map aborts +unknown wildcard hosts before they reach a worker. The startup script obtains short-lived metadata +credentials, fetches the two relay secrets without logging them, and runs a digest-pinned Cloud SQL +Auth Proxy beside the digest-pinned relay image. + +The primary `us-central1` subnet, router, and NAT retain their original +Terraform addresses. `relay_gce_additional_region_subnetwork_cidrs` creates +only additive regional resources; cells select the subnet from their declared +region. Every cell also declares an explicit database pool maximum in startup +metadata and deployment outputs. The initial Asia shape is `e2-standard-4`, +3,000 physical connections, 60 unobserved connections, 6,000 request units, +and a database pool maximum of 10. + +Provision the complete identical Asia wave in one `Deploy Relay Asia Topology` +saved plan. Its validator permits only the additive subnet/router/NAT, reviewed +cell templates/MIGs/backends, and exact shared URL-map host additions. It +rejects deletes, replacements, loss of an existing host route, US-resource +changes, and unrelated drift. Do not add production C27-C29 until the +compatible image has been published and each entry can pin its immutable +digest. + +Topology creation intentionally does not apply the director resource. Once all +MIGs and backends are healthy, register every new cell atomically as +migration-only through `Operate Relay Asia Admission` with the exact live +selector generation and a durable attempt ID. Only then may a director +deployment list the new cells. Verify that configuration and fresh heartbeats +before using the same workflow to promote the canary. A failed canary returns to +migration-only; do not delete the Asia network during rollout recovery. + +`relay_gce_cells` takes precedence in the director's configured-cell list. Adding or replacing a +generation requires a new map key and hostname; do not change an active cell's image in place. Run +`terraform fmt -check -recursive` and `terraform validate` locally; the same non-credentialed +checks run on every pull request. + +GCE deployments must add a distinct cell ID, host, backend, and MIG for every candidate generation; +never update an existing generation's image behind its origin. + +For a post-launch worker replacement, first publish an immutable image with the production image +workflow. Add that digest as a distinct `relay_gce_cells` entry with +`initially_enabled = false`, review/apply the Terraform change, and deploy the compatible director. +The production candidate workflow then reads the remote-state topology and defaults to a read-only +preflight. It verifies the exact TLS origin, `/health`, dependency-backed `/ready`, authenticated +heartbeat, served digest, private fixed-one MIG, runtime identity, dedicated backend, 86,400-second +timeout, and authoritative request-unit headroom. `execute` requires the literal `EVACUATE` +confirmation, disables the source only after preflight, enables the candidate, performs bounded +target-first evacuation, drains the exact source origin, and verifies aggregate completion. Keep +both source and candidate Terraform routes until a later reviewed removal proves the old origin has +no assignments, activity leases, or migrations. + +After any failure following target registration, use `audit` before `recover-forward`; never retry +`execute` or reverse admission. Forward recovery retries only bounded idempotent status operations. +If every remaining migration belongs to a registered target whose desktop is currently offline, it +emits `candidate_forward_pending` and stops without retiring those rows. Keep both origins intact +and rerun recovery only after a fresh audit shows target controls have returned. + +The production multi-target workflow is the reviewed path for evacuations that +need more than one candidate. It enforces deterministic serialized quotas, +target connection ceilings, and the oldest-migration lease gate before drain. +After selector generation 1, add new disabled targets without changing the +director resource in the targeted apply. Deploy the selector-version-2 +director, apply only the new cell templates, MIGs, backends, and URL-map +routes, then use `add-migration-cells` to register their exact configs as +migration-only in one selector generation. A single additive target is valid; +ordinary evacuation and supersession retain their multi-target requirements. +Use `retire-migration-cell` with an exact attempt ID to move one +migration-only cell to existing-only before its reviewed fence. +The director does not depend on the GCE forwarding-rule graph; keep director +configuration plans scoped away from immutable cell generations. +Its guarded `fence-source` mode applies an exact private Terraform saved plan +for a fully quiescent cell already listed in `relay_gce_fenced_cells`. The plan +must contain only that MIG's in-place target-size change from one to zero, so +the origin, backend, and generation remain retained. An interrupted apply is +always recovered forward unless Terraform state, live GCE state, and operation +history prove it never began. `abort-fence-source` records that proven +pre-apply abort; remove the cell from the fence set only in a later reviewed +commit. Never resize a production relay MIG directly. +Before the first fencing workflow rollout, apply this schema with an empty +fence set so remote-state topology contains `generation_identity`, +`fenced`, and `desired_target_size`. Only then commit a cell ID into the +production fence set. +The same workflow's `supersede-target` mode is the only supported path for a +failed registered target: it proves the failed MIG is zero with no instances +before recording an exact-incarnation fence and publishing newer epochs. +It invokes an IAM-authenticated max-one Cloud Run broker. The broker runtime +alone can access the exact state/saved-plan/lease object prefixes and update a +Relay MIG; the GitHub requester can read aggregate safety evidence and invoke +that service but cannot perform either mutation directly. diff --git a/cloud/infra/terraform/backend/production.hcl b/cloud/infra/terraform/backend/production.hcl new file mode 100644 index 00000000000..e482b091273 --- /dev/null +++ b/cloud/infra/terraform/backend/production.hcl @@ -0,0 +1,3 @@ +bucket = "onorca-cloud-terraform-state" +prefix = "terraform/state" + diff --git a/cloud/infra/terraform/backend/staging.hcl b/cloud/infra/terraform/backend/staging.hcl new file mode 100644 index 00000000000..fbd0f0c17b2 --- /dev/null +++ b/cloud/infra/terraform/backend/staging.hcl @@ -0,0 +1,3 @@ +bucket = "onorca-cloud-staging-terraform-state" +prefix = "terraform/state" + diff --git a/cloud/infra/terraform/environments/production.tfvars b/cloud/infra/terraform/environments/production.tfvars new file mode 100644 index 00000000000..60d36e3d832 --- /dev/null +++ b/cloud/infra/terraform/environments/production.tfvars @@ -0,0 +1,418 @@ +project_id = "onorca-cloud" +environment = "production" +name_prefix = "orca-cloud" +region = "us-central1" + +artifact_repository_id = "orca-cloud" + +# Dual accept while the relay source moves to the public stablyai/orca repository: the same +# workflows are trusted from both repos, and the public copies carry a `cloud-` file prefix. +# Remove this entry once the private workflows are retired and point github_owner/github_repo, +# github_repo_id, and github_owner_id at the surviving repository. +github_accepted_repositories = [ + { + owner = "stablyai" + repo = "orca" + repo_id = "1183888342" + owner_id = "127256420" + workflow_file_prefix = "cloud-" + } +] + +# Our first-party auth service. auth.onorca.dev is PropelAuth's prod domain, so +# our service lives at login.onorca.dev (desktop points ORCA_CLOUD_API_URL here). +auth_base_url = "https://login.onorca.dev" + +relay_cloud_run_service_name = "orca-cloud-relay" +relay_base_url = "https://relay.onorca.dev" +# Why: public admission is a per-instance semaphore, so fleet assignment capacity is +# concurrency x instances. Scaling to 2 instances took placement failures 35% -> 70%. +relay_min_instances = 5 +relay_max_instances = 5 +# Production cells run only on fixed-one GCE MIGs; Cloud Run remains the director. +relay_cells = {} +manage_relay_domain_mapping = true + +# Production GCE cells use exact hosts such as c1.relay.onorca.dev. +# The wildcard only handles DNS/TLS; the load balancer rejects unknown hosts. +relay_gce_domain = "relay.onorca.dev" +relay_gce_subnetwork_cidr = "10.42.0.0/24" +relay_gce_additional_region_subnetwork_cidrs = { + "asia-east2" = "10.42.1.0/24" +} +relay_gce_fenced_cells = ["production-gce-c1", "production-gce-c2", "production-gce-c3", "production-gce-c6", "production-gce-c11", "production-gce-c12"] +# Initial cells stay admission-disabled until production preflight and go-live approval. +relay_gce_cells = { + "production-gce-c1" = { + hostname = "c1" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:36a56b106c9e6d5897135c6af829085ab8fd53c85466406d9e887a7a5cfe9a02" + initially_enabled = false + } + "production-gce-c2" = { + hostname = "c2" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:36a56b106c9e6d5897135c6af829085ab8fd53c85466406d9e887a7a5cfe9a02" + initially_enabled = false + } + "production-gce-c3" = { + hostname = "c3" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:19ef4e6e4a043f63d78011d1c29a395a9002ec077d03ee6931f25479fe66f349" + initially_enabled = false + } + # Distinct origins let each existing cell drain without an in-place image swap. + "production-gce-c4" = { + hostname = "c4" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:36a56b106c9e6d5897135c6af829085ab8fd53c85466406d9e887a7a5cfe9a02" + initially_enabled = false + } + "production-gce-c5" = { + hostname = "c5" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:0e83408b0dc08531f1e8182019dc151afc38d63ddde4ad5cc01e40247ef3681d" + initially_enabled = false + } + "production-gce-c6" = { + hostname = "c6" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:36a56b106c9e6d5897135c6af829085ab8fd53c85466406d9e887a7a5cfe9a02" + initially_enabled = false + } + "production-gce-c7" = { + hostname = "c7" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c8" = { + hostname = "c8" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c9" = { + hostname = "c9" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + # Canary for the control-activation fence (PR #207, main b253fcd). + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c10" = { + hostname = "c10" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c11" = { + hostname = "c11" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:e592371013188b8297e395979c70a8b42c39f4bb5f90b01190f0778279cbaef5" + initially_enabled = false + connection_hard_cap = 600 + connection_unobserved_bound = 60 + } + "production-gce-c12" = { + hostname = "c12" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:3d8b388dcbf190be20491ce9c14eeafa0dccd0afbb2725712f6f9d9a754838dc" + initially_enabled = false + connection_hard_cap = 600 + connection_unobserved_bound = 60 + } + "production-gce-c13" = { + hostname = "c13" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c14" = { + hostname = "c14" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c15" = { + hostname = "c15" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c16" = { + hostname = "c16" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c17" = { + hostname = "c17" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:0e83408b0dc08531f1e8182019dc151afc38d63ddde4ad5cc01e40247ef3681d" + initially_enabled = false + connection_hard_cap = 600 + connection_unobserved_bound = 60 + } + # Canary for halved control lease renewal (PR #253, main 11256b5). Chosen as the + # smallest live cell: ~9 connections, so a replace costs 9 reconnects, not ~400. + "production-gce-c18" = { + hostname = "c18" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:0e83408b0dc08531f1e8182019dc151afc38d63ddde4ad5cc01e40247ef3681d" + initially_enabled = false + connection_hard_cap = 600 + connection_unobserved_bound = 60 + } + "production-gce-c19" = { + hostname = "c19" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c20" = { + hostname = "c20" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + # First full-size cell on the halved lease renewal, after the C18 canary measured + # 9.15 -> 5.32 queries per connection with zero failures. C21 also carries the + # control-close churn we still need to attribute to a client build. + "production-gce-c21" = { + hostname = "c21" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c22" = { + hostname = "c22" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c23" = { + hostname = "c23" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c24" = { + hostname = "c24" + zone = "us-central1-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c25" = { + hostname = "c25" + zone = "us-central1-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c26" = { + hostname = "c26" + zone = "us-central1-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "production-gce-c27" = { + hostname = "c27" + region = "asia-east2" + zone = "asia-east2-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 6000 + database_pool_max = 10 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 3000 + connection_unobserved_bound = 60 + } + "production-gce-c28" = { + hostname = "c28" + region = "asia-east2" + zone = "asia-east2-b" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 6000 + database_pool_max = 10 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 3000 + connection_unobserved_bound = 60 + } + "production-gce-c29" = { + hostname = "c29" + region = "asia-east2" + zone = "asia-east2-c" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 6000 + database_pool_max = 10 + image = "us-central1-docker.pkg.dev/onorca-cloud/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 3000 + connection_unobserved_bound = 60 + } +} + +relay_region_rehome_source_cell_ids = [ + "production-gce-c7", + "production-gce-c8", + "production-gce-c9", + "production-gce-c10", + "production-gce-c13", + "production-gce-c14", + "production-gce-c15", + "production-gce-c16", + "production-gce-c19", + "production-gce-c20", + "production-gce-c21", + "production-gce-c22", + "production-gce-c23", + "production-gce-c24", + "production-gce-c25", + "production-gce-c26" +] + +# Slack #orca-relay-alerts, created out of band on 2026-08-05. Declared here because an apply +# was otherwise going to strip it from every policy, leaving the alerts firing at nobody. +relay_alert_notification_channels = ["projects/onorca-cloud/notificationChannels/4879431412695417284"] diff --git a/cloud/infra/terraform/environments/staging.tfvars b/cloud/infra/terraform/environments/staging.tfvars new file mode 100644 index 00000000000..3ee108fe874 --- /dev/null +++ b/cloud/infra/terraform/environments/staging.tfvars @@ -0,0 +1,91 @@ +project_id = "onorca-cloud-staging" +environment = "staging" +name_prefix = "orca-cloud-staging" +region = "us-central1" + +artifact_repository_id = "orca-cloud" + +# Dual accept while the relay source moves to the public stablyai/orca repository: the same +# workflows are trusted from both repos, and the public copies carry a `cloud-` file prefix. +# Remove this entry once the private workflows are retired and point github_owner/github_repo, +# github_repo_id, and github_owner_id at the surviving repository. +github_accepted_repositories = [ + { + owner = "stablyai" + repo = "orca" + repo_id = "1183888342" + owner_id = "127256420" + workflow_file_prefix = "cloud-" + } +] + +auth_base_url = "https://auth-staging.onorca.dev" + +relay_cloud_run_service_name = "orca-cloud-relay-staging" +relay_staging_power_auth_service_name = "orca-cloud-auth-staging" +relay_base_url = "https://relay-staging.onorca.dev" +relay_min_instances = 0 +relay_max_instances = 2 +# Staging now exercises the production-shaped GCE data plane exclusively. +relay_cells = {} +# Keep the stable director mapping Terraform-owned while removing cell mappings. +manage_relay_domain_mapping = true + +# GCE cells use exact hosts below this wildcard, for example +# c1.relay-staging.onorca.dev. Cloudflare records remain out-of-band. +relay_gce_domain = "relay-staging.onorca.dev" +relay_gce_subnetwork_cidr = "10.42.0.0/24" +relay_gce_additional_region_subnetwork_cidrs = { + "asia-east2" = "10.42.1.0/24" +} +relay_gce_fenced_cells = [] +relay_gce_cells = { + "staging-gce-c1" = { + hostname = "c1" + zone = "us-central1-b" + machine_type = "e2-standard-2" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:2d0f6e6db2b0eb9d6aba188698de8330f8c30b4e76badfcf0fac3f3eb9508a87" + } + "staging-gce-c2" = { + hostname = "c2" + zone = "us-central1-c" + machine_type = "e2-standard-2" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:1239830d0946dc92ded3c9edde1c0b827f584a7a2be5c177beed900056d76f69" + connection_hard_cap = 600 + connection_unobserved_bound = 60 + } + "staging-gce-c3" = { + hostname = "c3" + zone = "us-central1-a" + machine_type = "e2-standard-2" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" + capacity_requests = 4000 + image = "us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:9fba2a189ab3fa29853800830e77c7551ab3aaa8f43f3cd9adbdea28b876a8b9" + initially_enabled = false + connection_hard_cap = 1000 + connection_unobserved_bound = 60 + } + "staging-gce-c4" = { + hostname = "c4" + region = "asia-east2" + zone = "asia-east2-a" + machine_type = "e2-standard-4" + boot_disk_gb = 30 + boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-21" + capacity_requests = 6000 + database_pool_max = 10 + image = "us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" + initially_enabled = false + connection_hard_cap = 3000 + connection_unobserved_bound = 60 + } +} + +relay_region_rehome_source_cell_ids = ["staging-gce-c2", "staging-gce-c3"] diff --git a/cloud/infra/terraform/outputs.tf b/cloud/infra/terraform/outputs.tf new file mode 100644 index 00000000000..220aa5cf94f --- /dev/null +++ b/cloud/infra/terraform/outputs.tf @@ -0,0 +1,191 @@ +output "github_deploy_service_account" { + value = try(google_service_account.github_deploy[0].email, null) + description = "Service account email to use in the GitHub Actions deploy workflow." +} + +output "github_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github[0].name, null) + description = "Workload Identity provider resource name for GitHub Actions." +} + +output "github_relay_monitor_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_monitor[0].name, null) + description = "Exact-workflow provider for PRODUCTION_GCP_RELAY_MONITOR_WORKLOAD_IDENTITY_PROVIDER." +} + +output "github_relay_monitor_service_account" { + value = try(google_service_account.github_monitor[0].email, null) + description = "Read-only identity for PRODUCTION_GCP_RELAY_MONITOR_SERVICE_ACCOUNT." +} + +output "github_relay_fence_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_fence[0].name, null) + description = "Exact-workflow provider for PRODUCTION_GCP_RELAY_FENCE_WORKLOAD_IDENTITY_PROVIDER." +} + +output "github_relay_fence_service_account" { + value = try(google_service_account.github_fence[0].email, null) + description = "Narrow fencing identity for PRODUCTION_GCP_RELAY_FENCE_SERVICE_ACCOUNT." +} + +output "github_staging_relay_capacity_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_staging_relay_capacity[0].name, null) + description = "Exact-workflow provider for STAGING_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER." +} + +output "github_staging_relay_capacity_service_account" { + value = try(google_service_account.github_staging_relay_capacity[0].email, null) + description = "Narrow transition identity for STAGING_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT." +} + +output "github_staging_relay_deploy_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_staging_relay_deploy[0].name, null) + description = "Exact-workflow provider for STAGING_GCP_RELAY_DEPLOY_WORKLOAD_IDENTITY_PROVIDER." +} + +output "github_staging_relay_deploy_service_account" { + value = try(google_service_account.github_staging_relay_deploy[0].email, null) + description = "Account for STAGING_GCP_RELAY_DEPLOY_SERVICE_ACCOUNT." +} + +output "github_production_relay_capacity_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_production_relay_capacity[0].name, null) + description = "Exact-workflow provider for PRODUCTION_GCP_RELAY_CAPACITY_WORKLOAD_IDENTITY_PROVIDER." +} + +output "github_production_relay_capacity_service_account" { + value = try(google_service_account.github_production_relay_capacity[0].email, null) + description = "Narrow transition identity for PRODUCTION_GCP_RELAY_CAPACITY_SERVICE_ACCOUNT." +} + +output "github_relay_asia_topology_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_relay_asia_topology[0].name, null) + description = "Workflow-bound provider for validated additive Relay Asia topology plans." +} + +output "github_relay_asia_topology_service_account" { + value = try(google_service_account.github_relay_asia_topology[0].email, null) + description = "Dedicated identity for validated additive Relay Asia topology plans." +} + +output "github_relay_asia_proof_workload_identity_provider" { + value = try(google_iam_workload_identity_pool_provider.github_relay_asia_proof[0].name, null) + description = "Workflow-bound staging Relay Asia proof Workload Identity provider." +} + +output "github_relay_asia_proof_service_account" { + value = try(google_service_account.github_relay_asia_proof[0].email, null) + description = "Least-privilege staging Relay Asia proof service account." +} + +output "relay_fence_broker_service_uri" { + value = try(google_cloud_run_v2_service.relay_fence_broker[0].uri, null) + description = "IAM-authenticated private Relay fence broker URI." +} + +output "relay_fence_broker_service_account" { + value = try(google_service_account.relay_fence_broker[0].email, null) + description = "Runtime identity that owns exact Relay fence mutations." +} + + +output "relay_cloud_run_service_uri" { + value = google_cloud_run_v2_service.relay.uri + description = "Default relay service URI for pre-domain smoke tests." +} + +output "relay_runtime_service_account" { + value = google_service_account.relay_runtime.email + description = "Runtime identity for stamped Relay cells." +} + +output "relay_director_runtime_service_account" { + value = google_service_account.relay_director_runtime.email + description = "Runtime and regional rehoming caller identity for the Relay director." +} + +output "relay_cell_cloud_run_service_uris" { + value = { + for cell_id, service in google_cloud_run_v2_service.relay_cell : + cell_id => service.uri + } + description = "Native Cloud Run URIs for stamped relay cells." +} + +output "relay_database_name" { + value = google_sql_database.relay.name + description = "Database isolated for durable relay state." +} + +output "relay_gce_load_balancer_ip" { + value = try(google_compute_global_address.relay_gce[0].address, null) + description = "Reserved IPv4 address for the shared GCE relay HTTPS load balancer." +} + +output "relay_gce_wildcard_dns_record" { + value = var.relay_gce_domain == "" ? null : { + name = "*.${var.relay_gce_domain}" + type = "A" + data = try(google_compute_global_address.relay_gce[0].address, null) + } + description = "DNS-only wildcard record that routes future cell hosts to the shared LB." +} + +output "relay_gce_certificate_dns_authorization" { + value = try(google_certificate_manager_dns_authorization.relay_gce[0].dns_resource_record[0], null) + description = "Certificate Manager DNS record that must remain published for renewal." +} + +output "relay_gce_cell_origins" { + value = local.relay_gce_cell_urls + description = "Exact public origins admitted by the shared GCE relay load balancer." +} + +output "relay_gce_cell_instance_groups" { + value = { + for cell_id, manager in google_compute_instance_group_manager.relay_gce_cell : + cell_id => manager.instance_group + } + description = "Terraform-sized managed instance groups backing each GCE relay cell." +} + +output "relay_gce_cell_backend_services" { + value = { + for cell_id, backend in google_compute_backend_service.relay_gce_cell : + cell_id => backend.id + } + description = "Non-overlapping backend service for each exact relay cell host." +} + +output "relay_gce_cell_deployments" { + value = { + for cell_id, cell in var.relay_gce_cells : cell_id => { + origin = local.relay_gce_cell_urls[cell_id] + region = cell.region + zone = cell.zone + mig_name = google_compute_instance_group_manager.relay_gce_cell[cell_id].name + instance_group = google_compute_instance_group_manager.relay_gce_cell[cell_id].instance_group + backend_name = google_compute_backend_service.relay_gce_cell[cell_id].name + backend_id = google_compute_backend_service.relay_gce_cell[cell_id].id + url_map_name = google_compute_url_map.relay_gce[0].name + generation_identity = google_compute_instance_template.relay_gce_cell[cell_id].self_link + image = cell.image + capacity_requests = cell.capacity_requests + database_pool_max = cell.database_pool_max + connection_hard_cap = cell.connection_hard_cap + connection_unobserved_bound = cell.connection_unobserved_bound + initially_enabled = cell.initially_enabled + fenced = contains(var.relay_gce_fenced_cells, cell_id) + desired_target_size = local.relay_gce_cell_target_sizes[cell_id] + target_size = google_compute_instance_group_manager.relay_gce_cell[cell_id].target_size + } + } + description = "Non-secret candidate deployment topology consumed by the GCE preflight workflow." + + precondition { + condition = alltrue([ + for cell_id in var.relay_gce_fenced_cells : contains(keys(var.relay_gce_cells), cell_id) + ]) + error_message = "relay_gce_fenced_cells may contain only configured relay_gce_cells keys." + } +} diff --git a/cloud/infra/terraform/relay-asia-proof-iam.tf b/cloud/infra/terraform/relay-asia-proof-iam.tf new file mode 100644 index 00000000000..e1ff687677b --- /dev/null +++ b/cloud/infra/terraform/relay-asia-proof-iam.tf @@ -0,0 +1,79 @@ +locals { + create_relay_asia_proof_identity = ( + local.relay_create_github_deploy_identity && var.environment == "staging" + ) + github_relay_asia_proof_workflow_file = "prove-relay-asia-staging.yml" + github_relay_asia_proof_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "assertion.workflow_ref == '${prefix}${local.github_relay_asia_proof_workflow_file}@refs/heads/main'" + ] + relay_asia_proof_service_account_email = try( + google_service_account.github_relay_asia_proof[0].email, + "" + ) +} + +resource "google_iam_workload_identity_pool_provider" "github_relay_asia_proof" { + count = local.create_relay_asia_proof_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-asia-proof" + display_name = "GitHub Relay Asia staging proof" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.environment" = "assertion.environment" + "attribute.event_name" = "assertion.event_name" + "attribute.ref" = "assertion.ref" + "attribute.repository" = "assertion.repository" + "attribute.repository_id" = "assertion.repository_id" + "attribute.repository_owner_id" = "assertion.repository_owner_id" + "attribute.workflow_ref" = "assertion.workflow_ref" + "attribute.relay_ops_identity" = "'staging-asia-proof'" + } + + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'staging'", + "assertion.event_name == 'workflow_dispatch'", + local.relay_github_workflow_conditions["github_relay_asia_proof"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_service_account" "github_relay_asia_proof" { + count = local.create_relay_asia_proof_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-aproof" + display_name = "Orca Relay Asia staging proof" + description = "Reads staging telemetry and performs only Relay's bounded Asia proof operations." +} + +resource "google_service_account_iam_member" "github_relay_asia_proof_workload_identity_user" { + count = local.create_relay_asia_proof_identity ? 1 : 0 + + service_account_id = google_service_account.github_relay_asia_proof[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/staging-asia-proof" +} + +resource "google_project_iam_member" "github_relay_asia_proof_logging_viewer" { + count = local.create_relay_asia_proof_identity ? 1 : 0 + + project = var.project_id + role = "roles/logging.viewer" + member = google_service_account.github_relay_asia_proof[0].member +} + +resource "google_project_iam_member" "github_relay_asia_proof_monitoring_viewer" { + count = local.create_relay_asia_proof_identity ? 1 : 0 + + project = var.project_id + role = "roles/monitoring.viewer" + member = google_service_account.github_relay_asia_proof[0].member +} diff --git a/cloud/infra/terraform/relay-asia-topology-iam.tf b/cloud/infra/terraform/relay-asia-topology-iam.tf new file mode 100644 index 00000000000..eea36e26247 --- /dev/null +++ b/cloud/infra/terraform/relay-asia-topology-iam.tf @@ -0,0 +1,207 @@ +locals { + create_relay_asia_topology_identity = local.relay_create_github_deploy_identity + github_relay_asia_topology_workflow_file = "deploy-relay-asia-topology.yml" + github_relay_asia_topology_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "assertion.workflow_ref == '${prefix}${local.github_relay_asia_topology_workflow_file}@refs/heads/main'" + ] +} + +resource "google_iam_workload_identity_pool_provider" "github_relay_asia_topology" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-asia" + display_name = "GitHub Relay Asia topology" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.environment" = "assertion.environment" + "attribute.event_name" = "assertion.event_name" + "attribute.ref" = "assertion.ref" + "attribute.repository" = "assertion.repository" + "attribute.repository_id" = "assertion.repository_id" + "attribute.repository_owner_id" = "assertion.repository_owner_id" + "attribute.workflow_ref" = "assertion.workflow_ref" + "attribute.relay_ops_identity" = "'${var.environment}-asia-topology'" + } + + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == '${var.environment}'", + "assertion.event_name == 'workflow_dispatch'", + local.relay_github_workflow_conditions["github_relay_asia_topology"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_service_account" "github_relay_asia_topology" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-asia" + display_name = "Orca Relay Asia topology" + description = "Applies only validated additive Relay Asia topology plans." +} + +resource "google_service_account_iam_member" "github_relay_asia_topology_workload_identity_user" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + service_account_id = google_service_account.github_relay_asia_topology[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/${var.environment}-asia-topology" +} + +resource "google_project_iam_custom_role" "github_relay_asia_topology_mutation" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayAsiaTopology" + title = "Orca Relay Asia topology" + description = "Creates additive Relay Asia network and cell topology and updates its shared URL map." + permissions = [ + "compute.backendServices.create", + "compute.backendServices.get", + "compute.backendServices.update", + "compute.backendServices.use", + "compute.disks.create", + "compute.globalOperations.get", + "compute.healthChecks.use", + "compute.healthChecks.useReadOnly", + "compute.images.useReadOnly", + "compute.instanceGroupManagers.create", + "compute.instanceGroupManagers.get", + "compute.instanceGroupManagers.update", + "compute.instanceGroups.create", + "compute.instanceGroups.get", + "compute.instanceGroups.use", + "compute.instances.create", + "compute.instances.setLabels", + "compute.instances.setMetadata", + "compute.instances.setTags", + "compute.instances.use", + "compute.instanceTemplates.create", + "compute.instanceTemplates.get", + "compute.instanceTemplates.useReadOnly", + "compute.networks.get", + "compute.networks.updatePolicy", + "compute.networks.use", + "compute.regionOperations.get", + "compute.routers.create", + "compute.routers.get", + "compute.routers.update", + "compute.subnetworks.create", + "compute.subnetworks.get", + "compute.subnetworks.setPrivateIpGoogleAccess", + "compute.subnetworks.use", + "compute.urlMaps.get", + "compute.urlMaps.update", + "compute.zoneOperations.get" + ] +} + +resource "google_project_iam_member" "github_relay_asia_topology_mutation" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + role = google_project_iam_custom_role.github_relay_asia_topology_mutation[0].id + member = google_service_account.github_relay_asia_topology[0].member +} + +resource "google_project_iam_custom_role" "github_relay_asia_topology_read" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayAsiaTopologyRead" + title = "Orca Relay Asia topology read" + description = "Refreshes only resource types required by validated Relay Asia topology plans." + permissions = [ + "artifactregistry.repositories.get", + "cloudsql.instances.get", + "compute.backendServices.get", + "compute.healthChecks.get", + "compute.instanceGroupManagers.get", + "compute.instanceGroups.get", + "compute.instanceTemplates.get", + "compute.instances.get", + "compute.networks.get", + "compute.routers.get", + "compute.subnetworks.get", + "compute.urlMaps.get", + "iam.serviceAccounts.get", + "iam.serviceAccounts.getIamPolicy", + "resourcemanager.projects.get", + "resourcemanager.projects.getIamPolicy", + "run.revisions.get", + "run.services.get", + "secretmanager.secrets.get", + "secretmanager.secrets.getIamPolicy", + "serviceusage.services.get", + "serviceusage.services.list" + ] +} + +resource "google_project_iam_member" "github_relay_asia_topology_read" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + role = google_project_iam_custom_role.github_relay_asia_topology_read[0].id + member = google_service_account.github_relay_asia_topology[0].member +} + +resource "google_artifact_registry_repository_iam_member" "github_relay_asia_topology_artifact_reader" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + location = var.region + repository = var.artifact_repository_id + role = "roles/artifactregistry.reader" + member = google_service_account.github_relay_asia_topology[0].member +} + +resource "google_storage_bucket_iam_member" "github_relay_asia_topology_state" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectAdmin" + member = google_service_account.github_relay_asia_topology[0].member + + condition { + title = "relay_asia_topology_state" + description = "Limits the Asia topology workflow to the environment Terraform state and lock." + expression = join(" || ", [ + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tfstate'", + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tflock'" + ]) + } +} + +resource "google_project_iam_custom_role" "github_relay_asia_topology_state_list" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayAsiaStateList" + title = "Orca Relay Asia state list" + description = "Lists the environment state bucket so Terraform can initialize its backend." + permissions = ["storage.objects.list"] +} + +resource "google_storage_bucket_iam_member" "github_relay_asia_topology_state_list" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = google_project_iam_custom_role.github_relay_asia_topology_state_list[0].id + member = google_service_account.github_relay_asia_topology[0].member +} + +resource "google_service_account_iam_member" "github_relay_asia_topology_runtime_user" { + count = local.create_relay_asia_topology_identity ? 1 : 0 + + service_account_id = google_service_account.relay_runtime.name + role = "roles/iam.serviceAccountUser" + member = google_service_account.github_relay_asia_topology[0].member +} diff --git a/cloud/infra/terraform/relay-database.tf b/cloud/infra/terraform/relay-database.tf new file mode 100644 index 00000000000..5dbbd71aa31 --- /dev/null +++ b/cloud/infra/terraform/relay-database.tf @@ -0,0 +1,54 @@ +# Relay credentials and assignment state share the existing Cloud SQL instance +# with auth, but use an isolated database and principal. +resource "google_sql_database" "relay" { + project = var.project_id + name = "orca_relay" + instance = local.relay_database_instance_name +} + +resource "random_password" "relay_database" { + length = 32 + special = false +} + +resource "google_sql_user" "relay" { + project = var.project_id + name = "orca_relay" + instance = local.relay_database_instance_name + password = random_password.relay_database.result +} + +resource "google_secret_manager_secret" "relay_database_url" { + project = var.project_id + secret_id = "orca-cloud-relay-database-url" + labels = local.relay_shared_labels + + replication { + auto {} + } +} + +resource "google_secret_manager_secret_version" "relay_database_url" { + secret = google_secret_manager_secret.relay_database_url.id + secret_data = format( + "postgresql://%s:%s@/%s?host=/cloudsql/%s", + google_sql_user.relay.name, + random_password.relay_database.result, + google_sql_database.relay.name, + local.relay_database_connection_name + ) +} + +resource "google_secret_manager_secret_iam_member" "relay_database_url_accessor" { + project = var.project_id + secret_id = google_secret_manager_secret.relay_database_url.secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.relay_runtime.member +} + +resource "google_secret_manager_secret_iam_member" "relay_database_url_director_accessor" { + project = var.project_id + secret_id = google_secret_manager_secret.relay_database_url.secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.relay_director_runtime.member +} diff --git a/cloud/infra/terraform/relay-dns.tf b/cloud/infra/terraform/relay-dns.tf new file mode 100644 index 00000000000..6e1895da925 --- /dev/null +++ b/cloud/infra/terraform/relay-dns.tf @@ -0,0 +1,47 @@ +# Relay custom domains. The Cloud Run domain mapping is the whole story here: Google issues and +# renews the certificate, and the DNS records that point at ghs.googlehosted.com are managed +# outside this root (the auth and artifact records moved to the apps root with their services). + +locals { + relay_fqdn = replace(replace(var.relay_base_url, "https://", ""), "http://", "") + relay_cell_fqdns = { + for cell_id, cell in var.relay_cells : + cell_id => replace(replace(cell.url, "https://", ""), "http://", "") + } +} + +resource "google_cloud_run_domain_mapping" "relay" { + count = var.manage_relay_domain_mapping ? 1 : 0 + location = var.region + name = local.relay_fqdn + + metadata { + namespace = var.project_id + } + + spec { + route_name = google_cloud_run_v2_service.relay.name + } + + # gcloud-created mappings report an empty legacy certificate_mode even + # though Google provisions the same automatic certificate; replacing it + # would reset issuance for no behavioral change. + lifecycle { + ignore_changes = [spec[0].certificate_mode] + } +} + +resource "google_cloud_run_domain_mapping" "relay_cell" { + for_each = var.manage_relay_domain_mapping ? var.relay_cells : {} + + location = var.region + name = local.relay_cell_fqdns[each.key] + + metadata { + namespace = var.project_id + } + + spec { + route_name = google_cloud_run_v2_service.relay_cell[each.key].name + } +} diff --git a/cloud/infra/terraform/relay-fence-broker.tf b/cloud/infra/terraform/relay-fence-broker.tf new file mode 100644 index 00000000000..0bcff95af9a --- /dev/null +++ b/cloud/infra/terraform/relay-fence-broker.tf @@ -0,0 +1,243 @@ +locals { + create_relay_fence_broker = local.relay_create_production_ops_identity + relay_fence_state_bucket = "${var.project_id}-terraform-state" + relay_fence_state_prefix = "projects/_/buckets/${local.relay_fence_state_bucket}/objects/terraform/state" +} + +resource "google_service_account" "relay_fence_broker" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-relay-fence" + display_name = "Orca Relay fence broker" + description = "Owns exact reviewed Terraform cell fences behind an authenticated broker." +} + +resource "google_project_iam_custom_role" "relay_fence_broker_mutation" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayFenceBroker" + title = "Orca Relay fence broker" + description = "Updates only reviewed Relay MIG sizes and inspects their zone operations." + permissions = [ + "compute.instanceGroupManagers.update" + ] +} + +resource "google_project_iam_member" "relay_fence_broker_mutation" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + role = google_project_iam_custom_role.relay_fence_broker_mutation[0].id + member = google_service_account.relay_fence_broker[0].member +} + +resource "google_project_iam_member" "relay_fence_broker_compute_viewer" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + role = "roles/compute.viewer" + member = google_service_account.relay_fence_broker[0].member +} + +resource "google_project_iam_member" "relay_fence_broker_logging_viewer" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + role = "roles/logging.viewer" + member = google_service_account.relay_fence_broker[0].member +} + +resource "google_project_iam_member" "relay_fence_broker_artifact_reader" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + role = "roles/artifactregistry.reader" + member = google_service_account.relay_fence_broker[0].member +} + +resource "google_storage_bucket_iam_member" "relay_fence_broker_bucket_reader" { + count = local.create_relay_fence_broker ? 1 : 0 + + bucket = local.relay_fence_state_bucket + role = "roles/storage.legacyBucketReader" + member = google_service_account.relay_fence_broker[0].member +} + +resource "google_storage_bucket_iam_member" "relay_fence_broker_state_objects" { + count = local.create_relay_fence_broker ? 1 : 0 + + bucket = local.relay_fence_state_bucket + role = "roles/storage.objectAdmin" + member = google_service_account.relay_fence_broker[0].member + + condition { + title = "relay_fence_exact_objects" + description = "Main state, private saved plans, and the durable broker lease only." + expression = join(" || ", [ + "resource.name.startsWith('${local.relay_fence_state_prefix}/default')", + "resource.name.startsWith('${local.relay_fence_state_prefix}/relay-fence-plans/production/')", + "resource.name.startsWith('${local.relay_fence_state_prefix}/relay-fence-broker/')" + ]) + } +} + +resource "google_service_account_iam_member" "relay_fence_broker_requester_token_creator" { + count = local.create_relay_fence_broker ? 1 : 0 + + service_account_id = google_service_account.github_fence[0].name + role = "roles/iam.serviceAccountTokenCreator" + member = google_service_account.relay_fence_broker[0].member +} + +resource "google_service_account_iam_member" "github_relay_fence_broker_service_account_user" { + count = local.create_relay_fence_broker ? 1 : 0 + + service_account_id = google_service_account.relay_fence_broker[0].name + role = "roles/iam.serviceAccountUser" + member = local.relay_github_deploy_service_account_member +} + +resource "google_cloud_run_v2_service" "relay_fence_broker" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + name = var.relay_fence_broker_service_name + location = var.region + ingress = "INGRESS_TRAFFIC_ALL" + deletion_protection = var.environment == "production" + labels = local.relay_shared_labels + + template { + service_account = google_service_account.relay_fence_broker[0].email + timeout = "1800s" + max_instance_request_concurrency = 1 + + scaling { + min_instance_count = 0 + max_instance_count = 1 + } + + containers { + image = var.relay_fence_broker_image + + ports { + container_port = 8080 + } + + env { + name = "ORCA_RELAY_FENCE_PROJECT" + value = var.project_id + } + + env { + name = "ORCA_RELAY_FENCE_STATE_BUCKET" + value = local.relay_fence_state_bucket + } + + env { + name = "ORCA_RELAY_FENCE_LEASE_OBJECT" + value = "terraform/state/relay-fence-broker/production.lock" + } + + env { + name = "ORCA_RELAY_FENCE_DIRECTOR_ORIGIN" + value = var.relay_base_url + } + + env { + name = "ORCA_RELAY_FENCE_ADMIN_AUDIENCE" + value = "${var.relay_base_url}/v1/admin/drain" + } + + env { + name = "ORCA_RELAY_FENCE_REQUESTER_SERVICE_ACCOUNT" + value = google_service_account.github_fence[0].email + } + + env { + name = "ORCA_RELAY_FENCE_RUNTIME_SERVICE_ACCOUNT" + value = google_service_account.relay_runtime.email + } + + env { + name = "ORCA_RELAY_FENCE_SOURCE_CELL_ID" + value = var.relay_fence_source_cell_id + } + + env { + name = "ORCA_RELAY_FENCE_FAILED_TARGET_CELL_ID" + value = var.relay_fence_failed_target_cell_id + } + + env { + name = "ORCA_RELAY_FENCE_REPLACEMENT_TARGET_CELL_ID" + value = var.relay_fence_replacement_target_cell_id + } + + env { + name = "ORCA_RELAY_FENCE_UNOBSERVED_CONNECTION_BOUND" + value = tostring(var.relay_fence_unobserved_connection_bound) + } + + resources { + limits = { + cpu = "1" + memory = "1Gi" + } + + cpu_idle = false + } + + startup_probe { + failure_threshold = 12 + initial_delay_seconds = 0 + period_seconds = 5 + timeout_seconds = 2 + + http_get { + path = "/healthz" + port = 8080 + } + } + } + } + + lifecycle { + ignore_changes = [ + client, + client_version, + template[0].containers[0].image + ] + } + + depends_on = [ + google_project_iam_member.relay_fence_broker_artifact_reader, + google_project_iam_member.relay_fence_broker_compute_viewer, + google_project_iam_member.relay_fence_broker_logging_viewer, + google_project_iam_member.relay_fence_broker_mutation, + google_storage_bucket_iam_member.relay_fence_broker_bucket_reader, + google_storage_bucket_iam_member.relay_fence_broker_state_objects + ] +} + +resource "google_cloud_run_v2_service_iam_member" "relay_fence_broker_invoker" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + location = var.region + name = google_cloud_run_v2_service.relay_fence_broker[0].name + role = "roles/run.invoker" + member = google_service_account.github_fence[0].member +} + +resource "google_cloud_run_v2_service_iam_member" "relay_fence_broker_deploy_invoker" { + count = local.create_relay_fence_broker ? 1 : 0 + + project = var.project_id + location = var.region + name = google_cloud_run_v2_service.relay_fence_broker[0].name + role = "roles/run.invoker" + member = local.relay_github_deploy_service_account_member +} diff --git a/cloud/infra/terraform/relay-gce-cells.tf b/cloud/infra/terraform/relay-gce-cells.tf new file mode 100644 index 00000000000..d6b7f3351f9 --- /dev/null +++ b/cloud/infra/terraform/relay-gce-cells.tf @@ -0,0 +1,393 @@ +locals { + relay_runtime_service_account_email = "${var.name_prefix}-relay@${var.project_id}.iam.gserviceaccount.com" + relay_director_runtime_service_account_email = "${var.name_prefix}-relay-dir@${var.project_id}.iam.gserviceaccount.com" + relay_capacity_service_account_email = var.environment == "production" ? try( + google_service_account.github_production_relay_capacity[0].email, + "" + ) : try(google_service_account.github_staging_relay_capacity[0].email, "") + relay_gce_cells_enabled = local.relay_gce_configured && length(var.relay_gce_cells) > 0 + relay_gce_subnetworks = merge( + { (var.region) = try(google_compute_subnetwork.relay_gce[0].id, null) }, + { for region, subnet in google_compute_subnetwork.relay_gce_additional : region => subnet.id } + ) + relay_gce_topology = { + max_surge = 0 + max_unavailable = 1 + backend_group_count = 1 + public_access_config_count = 0 + backend_timeout_seconds = 86400 + connection_drain_seconds = 300 + } + relay_gce_cell_urls = { + for cell_id, cell in var.relay_gce_cells : + cell_id => "https://${cell.hostname}.${var.relay_gce_domain}" + } + relay_gce_cell_target_sizes = { + for cell_id in keys(var.relay_gce_cells) : + cell_id => contains(var.relay_gce_fenced_cells, cell_id) ? 0 : 1 + } + relay_director_cells = local.relay_gce_cells_enabled ? { + for cell_id, cell in var.relay_gce_cells : cell_id => { + url = local.relay_gce_cell_urls[cell_id] + region = cell.region + capacity_requests = cell.capacity_requests + initially_enabled = cell.initially_enabled + connection_hard_cap = cell.connection_hard_cap + connection_unobserved_bound = cell.connection_unobserved_bound + } + } : { + for cell_id, cell in var.relay_cells : cell_id => { + url = cell.url + region = "us-central1" + capacity_requests = cell.capacity_requests + initially_enabled = true + connection_hard_cap = null + connection_unobserved_bound = null + } + } + relay_director_cells_json = jsonencode([ + for cell_id, cell in local.relay_director_cells : merge( + { + id = cell_id + url = cell.url + region = cell.region + capacityRequests = cell.capacity_requests + initiallyEnabled = cell.initially_enabled + }, + try(cell.connection_hard_cap, null) == null ? {} : { + connectionHardCap = cell.connection_hard_cap + connectionUnobservedBound = cell.connection_unobserved_bound + } + ) + ]) +} + +check "relay_gce_fixed_one_topology" { + assert { + condition = alltrue([ + for region in keys(var.relay_gce_additional_region_subnetwork_cidrs) : region != var.region + ]) && alltrue([ + for cell in values(var.relay_gce_cells) : + cell.region == var.region || contains(keys(var.relay_gce_additional_region_subnetwork_cidrs), cell.region) + ]) + error_message = "Additional Relay regions must differ from the primary region, and every cell region needs a configured subnetwork." + } + + assert { + condition = alltrue([ + for cell_id in var.relay_gce_fenced_cells : contains(keys(var.relay_gce_cells), cell_id) + ]) + error_message = "relay_gce_fenced_cells may contain only configured relay_gce_cells keys." + } + + assert { + condition = alltrue([ + for cell_id in var.relay_region_rehome_source_cell_ids : try( + var.relay_gce_cells[cell_id].region == var.region && + var.relay_gce_cells[cell_id].connection_hard_cap != null && + !contains(var.relay_gce_fenced_cells, cell_id), + false + ) + ]) + error_message = "Regional rehome sources must be configured, unfenced primary-region GCE cells with explicit connection limits." + } + + assert { + condition = length(var.relay_gce_cells) == 0 || var.relay_gce_domain != "" + error_message = "relay_gce_domain is required when GCE cells are configured." + } + + assert { + condition = ( + alltrue([ + for cell_id, target_size in local.relay_gce_cell_target_sizes : + contains(var.relay_gce_fenced_cells, cell_id) ? target_size == 0 : target_size == 1 + ]) && + local.relay_gce_topology.max_surge == 0 && + local.relay_gce_topology.max_unavailable == 1 && + local.relay_gce_topology.backend_group_count == 1 && + local.relay_gce_topology.public_access_config_count == 0 && + local.relay_gce_topology.backend_timeout_seconds == 86400 + ) + error_message = "Relay cells require fixed-one RECREATE MIGs, one non-public backend, and the 86,400-second WebSocket timeout." + } +} + +resource "google_compute_health_check" "relay_gce_liveness" { + count = local.relay_gce_cells_enabled ? 1 : 0 + + project = var.project_id + name = "${local.relay_gce_name}-health" + check_interval_sec = 10 + timeout_sec = 5 + healthy_threshold = 2 + unhealthy_threshold = 3 + + http_health_check { + port = 8080 + request_path = "/health" + } + + log_config { + enable = true + } + + # A project's first Compute API enablement can return before health-check + # creation is accepted, so keep this independent root behind the service. +} + +resource "google_compute_health_check" "relay_gce_readiness" { + count = local.relay_gce_cells_enabled ? 1 : 0 + + project = var.project_id + name = "${local.relay_gce_name}-ready" + check_interval_sec = 10 + timeout_sec = 5 + healthy_threshold = 2 + unhealthy_threshold = 2 + + http_health_check { + port = 8080 + request_path = "/ready" + } + + log_config { + enable = true + } + + # This check has no other Compute dependency to serialize initial API use. +} + +resource "google_compute_instance_template" "relay_gce_cell" { + for_each = var.relay_gce_cells + + project = var.project_id + name_prefix = "${substr("${local.relay_gce_name}-${each.value.hostname}", 0, 52)}-" + machine_type = each.value.machine_type + can_ip_forward = false + tags = ["orca-relay-cell"] + labels = merge( + local.relay_shared_labels, + { + orca-relay-role = "cell" + orca-relay-cell = each.key + }, + each.value.region == var.region ? {} : { orca-relay-region = each.value.region } + ) + + disk { + auto_delete = true + boot = true + device_name = "persistent-disk-0" + disk_size_gb = each.value.boot_disk_gb + disk_type = "pd-balanced" + source_image = each.value.boot_image + } + + network_interface { + subnetwork = local.relay_gce_subnetworks[each.value.region] + + # An empty dynamic block makes the no-public-IP invariant machine-checkable. + dynamic "access_config" { + for_each = range(local.relay_gce_topology.public_access_config_count) + content {} + } + } + + service_account { + email = local.relay_runtime_service_account_email + scopes = ["cloud-platform"] + } + + scheduling { + automatic_restart = true + on_host_maintenance = "MIGRATE" + provisioning_model = "STANDARD" + } + + shielded_instance_config { + enable_secure_boot = true + enable_vtpm = true + enable_integrity_monitoring = true + } + + metadata = { + block-project-ssh-keys = "TRUE" + enable-oslogin = "TRUE" + google-logging-enabled = "TRUE" + } + + metadata_startup_script = templatefile("${path.module}/relay-gce-startup.sh.tftpl", { + project_id = var.project_id + database_secret = google_secret_manager_secret.relay_database_url.secret_id + assignment_secret = google_secret_manager_secret.relay_assignment_signing_key.secret_id + cell_id = each.key + cell_region = each.value.region + include_cell_region = each.value.region != var.region + cell_url = local.relay_gce_cell_urls[each.key] + capacity_requests = each.value.capacity_requests + database_pool_max = each.value.database_pool_max + include_database_pool_max = each.value.region != var.region || each.value.database_pool_max != 10 + connection_hard_cap = each.value.connection_hard_cap + connection_unobserved_bound = each.value.connection_unobserved_bound + auth_issuer = var.auth_base_url + director_url = var.relay_base_url + deploy_service_account = local.relay_github_deploy_service_account_email + capacity_service_account = local.relay_capacity_service_account_email + asia_proof_service_account = local.relay_asia_proof_service_account_email + runtime_service_account = local.relay_runtime_service_account_email + rehome_source_enabled = contains(var.relay_region_rehome_source_cell_ids, each.key) + rehome_director_service_account = local.relay_director_runtime_service_account_email + rehome_audience = "${var.relay_base_url}/v1/admin/host-drain" + artifact_registry_host = "${var.region}-docker.pkg.dev" + relay_image = each.value.image + cloud_sql_proxy_image = var.relay_gce_cloud_sql_proxy_image + # Keep cell-only plans independent from unrelated database configuration drift. + cloud_sql_connection_name = local.relay_database_connection_name + }) + + lifecycle { + create_before_destroy = true + } + + depends_on = [ + google_project_iam_member.relay_runtime_artifact_reader, + google_project_iam_member.relay_runtime_cloudsql_client, + google_project_iam_member.relay_runtime_log_writer, + google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor, + google_secret_manager_secret_iam_member.relay_database_url_accessor + ] +} + +resource "google_compute_instance_group_manager" "relay_gce_cell" { + for_each = var.relay_gce_cells + + project = var.project_id + name = "${local.relay_gce_name}-${each.value.hostname}" + zone = each.value.zone + base_instance_name = "relay-${each.value.hostname}" + target_size = local.relay_gce_cell_target_sizes[each.key] + + version { + name = "primary" + instance_template = google_compute_instance_template.relay_gce_cell[each.key].self_link + } + + named_port { + name = "relay" + port = 8080 + } + + auto_healing_policies { + health_check = google_compute_health_check.relay_gce_liveness[0].id + initial_delay_sec = 180 + } + + update_policy { + type = "PROACTIVE" + minimal_action = "REPLACE" + most_disruptive_allowed_action = "REPLACE" + replacement_method = "RECREATE" + max_surge_fixed = local.relay_gce_topology.max_surge + max_unavailable_fixed = local.relay_gce_topology.max_unavailable + } + + lifecycle { + precondition { + condition = contains([0, 1], local.relay_gce_cell_target_sizes[each.key]) + error_message = "A relay cell MIG must be fenced at zero or active at exactly one." + } + } +} + +resource "google_compute_backend_service" "relay_gce_cell" { + for_each = var.relay_gce_cells + + project = var.project_id + name = "${local.relay_gce_name}-${each.value.hostname}" + protocol = "HTTP" + port_name = "relay" + load_balancing_scheme = "EXTERNAL_MANAGED" + timeout_sec = local.relay_gce_topology.backend_timeout_seconds + connection_draining_timeout_sec = local.relay_gce_topology.connection_drain_seconds + health_checks = [google_compute_health_check.relay_gce_readiness[0].id] + session_affinity = "NONE" + + backend { + group = google_compute_instance_group_manager.relay_gce_cell[each.key].instance_group + balancing_mode = "UTILIZATION" + max_utilization = 0.8 + capacity_scaler = 1 + } + + # Per-connection client IP/status/latency for the data plane; a WebSocket logs once, at close. + log_config { + enable = true + sample_rate = var.relay_gce_cell_log_sample_rate + } + + lifecycle { + precondition { + condition = local.relay_gce_topology.backend_group_count == 1 + error_message = "Each exact relay host must route to one non-overlapping fixed-one MIG." + } + } +} + +resource "google_compute_url_map" "relay_gce" { + count = local.relay_gce_cells_enabled ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + default_service = google_compute_backend_service.relay_gce_cell[sort(keys(var.relay_gce_cells))[0]].id + + # Unknown wildcard hosts fail at the LB and can never fall through to a cell. + default_route_action { + fault_injection_policy { + abort { + http_status = 404 + percentage = 100 + } + } + } + + dynamic "host_rule" { + for_each = var.relay_gce_cells + iterator = cell + content { + hosts = ["${cell.value.hostname}.${var.relay_gce_domain}"] + path_matcher = "cell-${cell.value.hostname}" + } + } + + dynamic "path_matcher" { + for_each = var.relay_gce_cells + iterator = cell + content { + name = "cell-${cell.value.hostname}" + default_service = google_compute_backend_service.relay_gce_cell[cell.key].id + } + } +} + +resource "google_compute_target_https_proxy" "relay_gce" { + count = local.relay_gce_cells_enabled ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + url_map = google_compute_url_map.relay_gce[0].id + certificate_map = "//certificatemanager.googleapis.com/${google_certificate_manager_certificate_map.relay_gce[0].id}" + quic_override = "NONE" +} + +resource "google_compute_global_forwarding_rule" "relay_gce" { + count = local.relay_gce_cells_enabled ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + ip_address = google_compute_global_address.relay_gce[0].id + port_range = "443" + target = google_compute_target_https_proxy.relay_gce[0].id + load_balancing_scheme = "EXTERNAL_MANAGED" + network_tier = "PREMIUM" +} diff --git a/cloud/infra/terraform/relay-gce-foundation.tf b/cloud/infra/terraform/relay-gce-foundation.tf new file mode 100644 index 00000000000..aab3b4579eb --- /dev/null +++ b/cloud/infra/terraform/relay-gce-foundation.tf @@ -0,0 +1,200 @@ +locals { + relay_gce_configured = var.relay_gce_domain != "" + relay_gce_name = "${var.name_prefix}-relay-gce" +} + +resource "google_compute_network" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + auto_create_subnetworks = false + routing_mode = "REGIONAL" +} + +resource "google_compute_subnetwork" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + region = var.region + network = google_compute_network.relay_gce[0].id + ip_cidr_range = var.relay_gce_subnetwork_cidr + private_ip_google_access = true + stack_type = "IPV4_ONLY" +} + +resource "google_compute_router" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + region = var.region + network = google_compute_network.relay_gce[0].id +} + +resource "google_compute_router_nat" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + region = var.region + router = google_compute_router.relay_gce[0].name + nat_ip_allocate_option = "AUTO_ONLY" + source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + + subnetwork { + name = google_compute_subnetwork.relay_gce[0].id + source_ip_ranges_to_nat = ["ALL_IP_RANGES"] + } + + log_config { + enable = true + filter = "ERRORS_ONLY" + } +} + +# The primary US resources above retain their production addresses. New regions are additive. +resource "google_compute_subnetwork" "relay_gce_additional" { + for_each = local.relay_gce_configured ? var.relay_gce_additional_region_subnetwork_cidrs : {} + + project = var.project_id + name = "${local.relay_gce_name}-${each.key}" + region = each.key + network = google_compute_network.relay_gce[0].id + ip_cidr_range = each.value + private_ip_google_access = true + stack_type = "IPV4_ONLY" +} + +resource "google_compute_router" "relay_gce_additional" { + for_each = local.relay_gce_configured ? var.relay_gce_additional_region_subnetwork_cidrs : {} + + project = var.project_id + name = "${local.relay_gce_name}-${each.key}" + region = each.key + network = google_compute_network.relay_gce[0].id +} + +resource "google_compute_router_nat" "relay_gce_additional" { + for_each = local.relay_gce_configured ? var.relay_gce_additional_region_subnetwork_cidrs : {} + + project = var.project_id + name = "${local.relay_gce_name}-${each.key}" + region = each.key + router = google_compute_router.relay_gce_additional[each.key].name + nat_ip_allocate_option = "AUTO_ONLY" + source_subnetwork_ip_ranges_to_nat = "LIST_OF_SUBNETWORKS" + + subnetwork { + name = google_compute_subnetwork.relay_gce_additional[each.key].id + source_ip_ranges_to_nat = ["ALL_IP_RANGES"] + } + + log_config { + enable = true + filter = "ERRORS_ONLY" + } +} + +resource "google_compute_firewall" "relay_gce_load_balancer" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = "${local.relay_gce_name}-lb" + network = google_compute_network.relay_gce[0].name + direction = "INGRESS" + source_ranges = ["35.191.0.0/16", "130.211.0.0/22"] + target_tags = ["orca-relay-cell"] + + allow { + protocol = "tcp" + ports = ["8080"] + } +} + +resource "google_compute_firewall" "relay_gce_iap_ssh" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = "${local.relay_gce_name}-iap-ssh" + network = google_compute_network.relay_gce[0].name + direction = "INGRESS" + source_ranges = ["35.235.240.0/20"] + target_tags = ["orca-relay-cell"] + + allow { + protocol = "tcp" + ports = ["22"] + } +} + +resource "google_project_iam_member" "relay_runtime_artifact_reader" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + role = "roles/artifactregistry.reader" + member = google_service_account.relay_runtime.member +} + +resource "google_project_iam_member" "relay_runtime_log_writer" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + role = "roles/logging.logWriter" + member = google_service_account.relay_runtime.member +} + +resource "google_compute_global_address" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + address_type = "EXTERNAL" + ip_version = "IPV4" +} + +resource "google_certificate_manager_dns_authorization" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + domain = var.relay_gce_domain + location = "global" + type = "PER_PROJECT_RECORD" + description = "DNS authorization for Orca Relay wildcard cell certificates." + labels = local.relay_shared_labels +} + +resource "google_certificate_manager_certificate" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + location = "global" + description = "Wildcard certificate for exact-routed Orca Relay GCE cells." + labels = local.relay_shared_labels + + managed { + domains = ["*.${var.relay_gce_domain}"] + dns_authorizations = [google_certificate_manager_dns_authorization.relay_gce[0].id] + } +} + +resource "google_certificate_manager_certificate_map" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = local.relay_gce_name + description = "Certificate map for the shared Orca Relay HTTPS load balancer." +} + +resource "google_certificate_manager_certificate_map_entry" "relay_gce" { + count = local.relay_gce_configured ? 1 : 0 + + project = var.project_id + name = "${local.relay_gce_name}-wildcard" + map = google_certificate_manager_certificate_map.relay_gce[0].name + hostname = "*.${var.relay_gce_domain}" + certificates = [google_certificate_manager_certificate.relay_gce[0].id] +} diff --git a/cloud/infra/terraform/relay-gce-startup.sh.tftpl b/cloud/infra/terraform/relay-gce-startup.sh.tftpl new file mode 100644 index 00000000000..f593d94e9e5 --- /dev/null +++ b/cloud/infra/terraform/relay-gce-startup.sh.tftpl @@ -0,0 +1,136 @@ +#!/bin/bash +set -euo pipefail + +readonly metadata_url="http://metadata.google.internal/computeMetadata/v1" +readonly state_dir="/var/lib/orca-relay" +readonly cloudsql_dir="$${state_dir}/cloudsql" +readonly docker_config_dir="$${state_dir}/docker" +readonly env_file="$${state_dir}/relay.env" + +# COS mounts /root read-only, and neither the plaintext environment nor pull credential should survive bootstrap. +trap 'rm -f "$${env_file}"; rm -rf "$${docker_config_dir}"' EXIT + +mkdir -p "$${cloudsql_dir}" "$${docker_config_dir}" +chmod 0700 "$${state_dir}" +chmod 0700 "$${docker_config_dir}" +chmod 0777 "$${cloudsql_dir}" +export DOCKER_CONFIG="$${docker_config_dir}" + +# Startup metadata can run before COS has made the Docker socket usable. +systemctl start docker +for _ in $(seq 1 60); do + if docker info >/dev/null 2>&1; then + break + fi + sleep 1 +done +docker info >/dev/null + +metadata_access_token() { + curl --fail --silent --show-error \ + --header 'Metadata-Flavor: Google' \ + "$${metadata_url}/instance/service-accounts/default/token" \ + | sed -n 's/.*"access_token":"\([^"]*\)".*/\1/p' +} + +read_secret() { + local secret_name="$1" + local token="$2" + local encoded + encoded="$(curl --fail --silent --show-error \ + --header "Authorization: Bearer $${token}" \ + "https://secretmanager.googleapis.com/v1/projects/${project_id}/secrets/$${secret_name}/versions/latest:access" \ + | sed -n 's/.*"data":[[:space:]]*"\([^"]*\)".*/\1/p')" + test -n "$${encoded}" + printf '%s' "$${encoded}" | tr '_-' '/+' | base64 --decode +} + +access_token="$(metadata_access_token)" +test -n "$${access_token}" +database_url="$(read_secret '${database_secret}' "$${access_token}")" +assignment_key="$(read_secret '${assignment_secret}' "$${access_token}")" + +umask 077 +{ + printf 'DATABASE_URL=%s\n' "$${database_url}" + printf 'ORCA_RELAY_ASSIGNMENT_SIGNING_KEY=%s\n' "$${assignment_key}" + printf 'ORCA_RELAY_PUBLIC_URL=%s\n' '${cell_url}' + printf 'ORCA_RELAY_CELL_URL=%s\n' '${cell_url}' + printf 'ORCA_RELAY_AUTH_ISSUER=%s\n' '${auth_issuer}' + printf 'ORCA_RELAY_AUTH_AUDIENCE=orca-relay\n' + printf 'ORCA_RELAY_JWKS_URL=%s/.well-known/jwks.json\n' '${auth_issuer}' + printf 'ORCA_RELAY_ROLE=cell\n' + printf 'ORCA_RELAY_CELL_ID=%s\n' '${cell_id}' +%{ if include_cell_region ~} + printf 'ORCA_RELAY_REGION=%s\n' '${cell_region}' +%{ endif ~} + printf 'ORCA_RELAY_CELL_CAPACITY=%s\n' '${capacity_requests}' +%{ if include_database_pool_max ~} + printf 'ORCA_RELAY_DATABASE_POOL_MAX=%s\n' '${database_pool_max}' +%{ endif ~} +%{ if connection_hard_cap != null ~} + printf 'ORCA_RELAY_CELL_CONNECTION_HARD_CAP=%s\n' '${connection_hard_cap}' + printf 'ORCA_RELAY_CELL_CONNECTION_UNOBSERVED_BOUND=%s\n' '${connection_unobserved_bound}' +%{ endif ~} + printf 'ORCA_RELAY_CELLS_JSON=[]\n' + printf 'ORCA_RELAY_ADMIN_AUDIENCE=%s/v1/admin/drain\n' '${director_url}' + printf 'ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT=%s\n' '${deploy_service_account}' + printf 'ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT=%s\n' '${capacity_service_account}' +%{ if asia_proof_service_account != "" ~} + printf 'ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT=%s\n' '${asia_proof_service_account}' +%{ endif ~} + printf 'ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT=%s\n' '${runtime_service_account}' +%{ if rehome_source_enabled ~} + printf 'ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT=%s\n' '${rehome_director_service_account}' + printf 'ORCA_RELAY_REHOME_AUDIENCE=%s\n' '${rehome_audience}' +%{ endif ~} + printf 'ORCA_RELAY_DIRECTOR_URL=%s\n' '${director_url}' + printf 'ORCA_RELAY_HEARTBEAT_AUDIENCE=%s/v1/admin/cell-heartbeat\n' '${director_url}' + printf 'ORCA_RELAY_IMAGE_DIGEST=%s\n' '${trimprefix(regex("@sha256:[a-f0-9]{64}$", relay_image), "@")}' +} > "$${env_file}" +unset database_url assignment_key + +# COS has no long-lived registry credential; use a short metadata token only for the pull. +printf '%s' "$${access_token}" \ + | docker login --username oauth2accesstoken --password-stdin 'https://${artifact_registry_host}' +docker pull '${relay_image}' +docker logout '${artifact_registry_host}' >/dev/null 2>&1 || true +unset access_token +docker pull '${cloud_sql_proxy_image}' + +docker rm --force orca-relay cloud-sql-proxy >/dev/null 2>&1 || true +# The persistent COS state directory can retain a dead proxy's socket across VM reboots. +find "$${cloudsql_dir}" -type s -name '.s.PGSQL.5432' -delete +docker run --detach \ + --name cloud-sql-proxy \ + --restart always \ + --security-opt no-new-privileges \ + --cap-drop ALL \ + --user 0:0 \ + --volume "$${cloudsql_dir}:/cloudsql" \ + '${cloud_sql_proxy_image}' \ + --unix-socket=/cloudsql \ + '${cloud_sql_connection_name}' + +# Readiness depends on the proxy socket, so do not start the relay into a known SQL failure. +for _ in $(seq 1 60); do + if find "$${cloudsql_dir}" -type s -name '.s.PGSQL.5432' -print -quit | grep -q .; then + break + fi + sleep 1 +done +find "$${cloudsql_dir}" -type s -name '.s.PGSQL.5432' -print -quit | grep -q . + +docker run --detach \ + --name orca-relay \ + --restart always \ + --stop-timeout 300 \ + --security-opt no-new-privileges \ + --cap-drop ALL \ + --publish 8080:8080 \ + --volume "$${cloudsql_dir}:/cloudsql" \ + --env-file "$${env_file}" \ + '${relay_image}' + +# GCE cells feed the same privacy-safe aggregate metrics as Cloud Run without app credentials. +systemctl start logging-agent.target diff --git a/cloud/infra/terraform/relay-github-actions.tf b/cloud/infra/terraform/relay-github-actions.tf new file mode 100644 index 00000000000..450ea64cc0a --- /dev/null +++ b/cloud/infra/terraform/relay-github-actions.tf @@ -0,0 +1,669 @@ +# GitHub Actions identities the relay root owns. +# +# The shared deploy account, its Workload Identity provider, and everything bound to it are +# environment-conditional under amendment A1: production is relay-owned, staging is apps-owned. +# Both state surgeries are done, so their counts are production-only here; the staging copies +# are declared by infra/terraform-apps and live in its state. +# +# The Workload Identity pool itself is foundation-owned and reached by literal through +# relay-shared.tf, never by reference. + +locals { + github_monitor_caller_workflow_file = "monitor-relay-production.yml" + github_monitor_workflow_file = "monitor-relay-production-job.yml" + github_fence_workflow_file = "deploy-relay-production-multi-target.yml" + github_production_relay_workflow_files = [ + "deploy-relay-fence-broker.yml", + "deploy-relay-production-capacity.yml", + "deploy-relay-production-director.yml", + "deploy-relay-production-multi-target.yml", + "deploy-relay-production.yml", + "operate-relay-asia-admission.yml", + "publish-relay-production.yml" + ] + github_production_relay_capacity_workflow_file = "deploy-relay-production-capacity.yml" + github_production_relay_capacity_job_workflow_file = "deploy-relay-production-capacity-job.yml" + github_production_relay_same_cap_workflow_file = "deploy-relay-production-same-cap.yml" + github_production_relay_same_cap_job_workflow_file = "deploy-relay-production-same-cap-job.yml" + github_production_relay_rehome_workflow_file = "operate-relay-production-rehome.yml" + github_production_relay_rehome_job_workflow_file = "operate-relay-production-rehome-job.yml" + github_staging_relay_capacity_workflow_files = [ + "bootstrap-relay-staging-capacity.yml", + "prove-relay-staging-capacity.yml", + "recover-relay-staging-c4-image.yml" + ] + + # The same-cap caller admits its own jobs too: release_lease runs in the caller file and presents + # the caller as job_workflow_ref, so pinning only the reusable job would refuse it. + github_production_relay_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "((${join(" || ", [for workflow_file in local.github_production_relay_workflow_files : "assertion.workflow_ref == '${prefix}${workflow_file}@refs/heads/main'"])}) || (assertion.workflow_ref == '${prefix}${local.github_production_relay_rehome_workflow_file}@refs/heads/main' && assertion.job_workflow_ref == '${prefix}${local.github_production_relay_rehome_job_workflow_file}@refs/heads/main') || (assertion.workflow_ref == '${prefix}${local.github_production_relay_same_cap_workflow_file}@refs/heads/main' && (assertion.job_workflow_ref == '${prefix}${local.github_production_relay_same_cap_job_workflow_file}@refs/heads/main' || assertion.job_workflow_ref == '${prefix}${local.github_production_relay_same_cap_workflow_file}@refs/heads/main')))" + ] + github_monitor_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "assertion.workflow_ref == '${prefix}${local.github_monitor_caller_workflow_file}@refs/heads/main' && assertion.job_workflow_ref == '${prefix}${local.github_monitor_workflow_file}@refs/heads/main'" + ] + github_fence_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "assertion.workflow_ref == '${prefix}${local.github_fence_workflow_file}@refs/heads/main' && assertion.job_workflow_ref == '${prefix}${local.github_fence_workflow_file}@refs/heads/main'" + ] + github_production_relay_capacity_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "((assertion.workflow_ref == '${prefix}${local.github_production_relay_capacity_workflow_file}@refs/heads/main' && assertion.job_workflow_ref == '${prefix}${local.github_production_relay_capacity_job_workflow_file}@refs/heads/main') || (assertion.workflow_ref == '${prefix}${local.github_production_relay_same_cap_workflow_file}@refs/heads/main' && assertion.job_workflow_ref == '${prefix}${local.github_production_relay_same_cap_job_workflow_file}@refs/heads/main'))" + ] + github_staging_relay_capacity_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "(${join(" || ", [for workflow_file in local.github_staging_relay_capacity_workflow_files : "assertion.workflow_ref == '${prefix}${workflow_file}@refs/heads/main'"])})" + ] + + create_staging_relay_power_role = ( + local.relay_create_github_deploy_identity && var.environment == "staging" + ) + create_staging_relay_capacity_identity = ( + local.relay_create_github_deploy_identity && var.environment == "staging" + ) + create_production_relay_capacity_identity = ( + local.relay_create_github_deploy_identity && var.environment == "production" + ) +} + +resource "google_iam_workload_identity_pool_provider" "github" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github" + display_name = "GitHub" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.actor" = "assertion.actor" + "attribute.environment" = "assertion.environment" + "attribute.ref" = "assertion.ref" + "attribute.repository" = "assertion.repository" + "attribute.repository_id" = "assertion.repository_id" + "attribute.repository_owner_id" = "assertion.repository_owner_id" + "attribute.workflow_ref" = "assertion.workflow_ref" + } + + # Production-only, so the condition is unconditional. The staging copy of this provider is + # declared by infra/terraform-apps/github-actions.tf and pinned there. + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'production'", + local.relay_github_workflow_conditions["github"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_iam_workload_identity_pool_provider" "github_monitor" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-monitor" + display_name = "GitHub Relay production monitor" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.repository" = "assertion.repository" + "attribute.ref" = "assertion.ref" + "attribute.environment" = "assertion.environment" + "attribute.job_workflow_ref" = "assertion.job_workflow_ref" + "attribute.relay_ops_identity" = "'monitor'" + } + + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'production'", + local.relay_github_workflow_conditions["github_monitor"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_iam_workload_identity_pool_provider" "github_fence" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-fence" + display_name = "GitHub Relay production fence" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.repository" = "assertion.repository" + "attribute.ref" = "assertion.ref" + "attribute.environment" = "assertion.environment" + "attribute.job_workflow_ref" = "assertion.job_workflow_ref" + "attribute.relay_ops_identity" = "'fence'" + } + + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'production'", + local.relay_github_workflow_conditions["github_fence"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_iam_workload_identity_pool_provider" "github_staging_relay_capacity" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-capacity" + display_name = "GitHub Relay staging capacity" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.repository" = "assertion.repository" + "attribute.ref" = "assertion.ref" + "attribute.environment" = "assertion.environment" + "attribute.workflow_ref" = "assertion.workflow_ref" + "attribute.relay_ops_identity" = "'staging-capacity'" + } + + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'staging'", + local.relay_github_workflow_conditions["github_staging_relay_capacity"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_iam_workload_identity_pool_provider" "github_production_relay_capacity" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-capacity" + display_name = "GitHub Relay production capacity" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.repository" = "assertion.repository" + "attribute.ref" = "assertion.ref" + "attribute.environment" = "assertion.environment" + "attribute.workflow_ref" = "assertion.workflow_ref" + "attribute.job_workflow_ref" = "assertion.job_workflow_ref" + "attribute.relay_ops_identity" = "'production-capacity'" + } + + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'production'", + local.relay_github_workflow_conditions["github_production_relay_capacity"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_service_account" "github_deploy" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-deploy" + display_name = "Orca Cloud GitHub deploy" + description = "Deploys Orca Cloud from GitHub Actions." +} + +resource "google_service_account" "github_monitor" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-monitor" + display_name = "Orca Relay production monitor" + description = "Reads aggregate Relay production telemetry." +} + +resource "google_service_account" "github_fence" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-fence" + display_name = "Orca Relay production fence requester" + description = "Requests exact reviewed Relay cell fences through the private broker." +} + +resource "google_service_account" "github_staging_relay_capacity" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-cap" + display_name = "Orca Relay staging capacity transition" + description = "Runs the exact reviewed Relay staging capacity workflow." +} + +resource "google_service_account" "github_production_relay_capacity" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-cap" + display_name = "Orca Relay production capacity transition" + description = "Runs the exact reviewed Relay production capacity workflow." +} + +resource "google_service_account_iam_member" "github_workload_identity_user" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + service_account_id = google_service_account.github_deploy[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.repository/${local.relay_github_repository}" +} + +# The `github` provider maps assertion.repository straight through, so the binding above admits +# only the primary repository. Each additional accepted repository needs its own principalSet +# before its workflows can mint this account; with an empty list this creates nothing. +resource "google_service_account_iam_member" "github_accepted_repository_workload_identity_user" { + for_each = toset( + local.relay_create_production_ops_identity + ? slice(local.relay_github_accepted_repository_names, 1, length(local.relay_github_accepted_repository_names)) + : [] + ) + + service_account_id = google_service_account.github_deploy[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.repository/${each.key}" +} + +resource "google_service_account_iam_member" "github_monitor_workload_identity_user" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + service_account_id = google_service_account.github_monitor[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/monitor" +} + +resource "google_service_account_iam_member" "github_fence_workload_identity_user" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + service_account_id = google_service_account.github_fence[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/fence" +} + +resource "google_service_account_iam_member" "github_staging_relay_capacity_workload_identity_user" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + service_account_id = google_service_account.github_staging_relay_capacity[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/staging-capacity" +} + +resource "google_service_account_iam_member" "github_production_relay_capacity_workload_identity_user" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + service_account_id = google_service_account.github_production_relay_capacity[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/production-capacity" +} + +resource "google_artifact_registry_repository_iam_member" "github_production_relay_writer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + location = var.region + repository = var.artifact_repository_id + role = "roles/artifactregistry.writer" + member = local.relay_github_deploy_service_account_member +} + +resource "google_artifact_registry_repository_iam_member" "github_production_relay_staging_mirror_writer" { + count = local.relay_create_github_deploy_identity && var.environment == "staging" ? 1 : 0 + + project = var.project_id + location = var.region + repository = var.artifact_repository_id + role = "roles/artifactregistry.writer" + member = "serviceAccount:orca-cloud-gha-deploy@onorca-cloud.iam.gserviceaccount.com" +} + +resource "google_cloud_run_v2_service_iam_member" "github_production_relay_director_developer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + location = var.region + name = var.relay_cloud_run_service_name + role = "roles/run.developer" + member = local.relay_github_deploy_service_account_member +} + +resource "google_cloud_run_v2_service_iam_member" "github_production_relay_fence_broker_developer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + location = var.region + name = var.relay_fence_broker_service_name + role = "roles/run.developer" + member = local.relay_github_deploy_service_account_member +} + +# Candidate preflight reads GCE/LB topology; Terraform fencing keeps mutation narrowly scoped. +resource "google_project_iam_member" "github_compute_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/compute.viewer" + member = google_service_account.github_deploy[0].member +} + +# Incident monitoring needs aggregate telemetry and inventory without mutation permissions. +resource "google_project_iam_member" "github_monitoring_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/monitoring.viewer" + member = google_service_account.github_deploy[0].member +} + +resource "google_project_iam_member" "github_logging_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/logging.viewer" + member = google_service_account.github_deploy[0].member +} + +resource "google_project_iam_member" "github_cloudsql_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/cloudsql.viewer" + member = google_service_account.github_deploy[0].member +} + +resource "google_project_iam_member" "github_relay_monitor_monitoring_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/monitoring.viewer" + member = google_service_account.github_monitor[0].member +} + +resource "google_project_iam_member" "github_relay_monitor_logging_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/logging.viewer" + member = google_service_account.github_monitor[0].member +} + +resource "google_project_iam_member" "github_relay_monitor_cloudsql_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/cloudsql.viewer" + member = google_service_account.github_monitor[0].member +} + +resource "google_project_iam_member" "github_monitor_compute_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/compute.viewer" + member = google_service_account.github_monitor[0].member +} + +resource "google_project_iam_member" "github_fence_monitoring_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/monitoring.viewer" + member = google_service_account.github_fence[0].member +} + +resource "google_project_iam_member" "github_fence_logging_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/logging.viewer" + member = google_service_account.github_fence[0].member +} + +resource "google_project_iam_member" "github_fence_cloudsql_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/cloudsql.viewer" + member = google_service_account.github_fence[0].member +} + +resource "google_project_iam_member" "github_fence_compute_viewer" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + project = var.project_id + role = "roles/compute.viewer" + member = google_service_account.github_fence[0].member +} + +# Staging may sleep between test windows. Keep its workflow narrower than a +# general Compute/Cloud SQL editor and never create this role in production. +resource "google_project_iam_custom_role" "github_staging_relay_power" { + count = local.create_staging_relay_power_role ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayStagingPower" + title = "Orca Relay staging power operator" + description = "Scales staging Relay MIGs and starts or stops its shared staging database." + permissions = [ + "cloudsql.instances.get", + "cloudsql.instances.update", + "cloudsql.operations.get", + "compute.instanceGroupManagers.get", + "compute.instanceGroupManagers.update", + "compute.zoneOperations.get" + ] +} + +resource "google_project_iam_member" "github_staging_relay_power" { + count = local.create_staging_relay_power_role ? 1 : 0 + + project = var.project_id + role = google_project_iam_custom_role.github_staging_relay_power[0].id + member = local.relay_github_deploy_service_account_member +} + +resource "google_project_iam_custom_role" "github_staging_relay_capacity_mutation" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayStagingCapacity" + title = "Orca Relay staging capacity transition" + description = "Replaces one staging Relay template and restarts its managed instance group." + permissions = [ + "compute.disks.create", + "compute.healthChecks.use", + "compute.images.useReadOnly", + "compute.instanceGroupManagers.get", + "compute.instanceGroupManagers.update", + "compute.instances.create", + "compute.instances.setLabels", + "compute.instances.setMetadata", + "compute.instances.setTags", + "compute.instanceTemplates.create", + "compute.instanceTemplates.delete", + "compute.instanceTemplates.get", + "compute.instanceTemplates.useReadOnly", + "compute.networks.use", + "compute.subnetworks.use", + "compute.zoneOperations.get" + ] +} + +resource "google_project_iam_member" "github_staging_relay_capacity_mutation" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role = google_project_iam_custom_role.github_staging_relay_capacity_mutation[0].id + member = google_service_account.github_staging_relay_capacity[0].member +} + +resource "google_project_iam_member" "github_staging_relay_capacity_viewer" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role = "roles/viewer" + member = google_service_account.github_staging_relay_capacity[0].member +} + +resource "google_project_iam_member" "github_staging_relay_capacity_artifact_reader" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role = "roles/artifactregistry.reader" + member = google_service_account.github_staging_relay_capacity[0].member +} + +resource "google_cloud_run_v2_service_iam_member" "github_staging_relay_capacity_developer" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + project = var.project_id + location = var.region + name = var.relay_cloud_run_service_name + role = "roles/run.developer" + member = google_service_account.github_staging_relay_capacity[0].member +} + +resource "google_storage_bucket_iam_member" "github_staging_relay_capacity_state" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectAdmin" + member = google_service_account.github_staging_relay_capacity[0].member + + condition { + title = "relay_capacity_state" + description = "Limits the staging capacity workflow to the default Terraform state and lock." + expression = join(" || ", [ + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tfstate'", + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tflock'" + ]) + } +} + +resource "google_service_account_iam_member" "github_staging_relay_capacity_runtime_user" { + count = local.create_staging_relay_capacity_identity ? 1 : 0 + + service_account_id = google_service_account.relay_runtime.name + role = "roles/iam.serviceAccountUser" + member = google_service_account.github_staging_relay_capacity[0].member +} + +resource "google_project_iam_custom_role" "github_production_relay_capacity_mutation" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role_id = "orcaRelayProductionCapacity" + title = "Orca Relay production capacity transition" + description = "Replaces exactly one production Relay template and its managed instance." + permissions = [ + "compute.disks.create", + "compute.healthChecks.use", + "compute.images.useReadOnly", + "compute.instanceGroupManagers.get", + "compute.instanceGroupManagers.update", + "compute.instances.create", + "compute.instances.setLabels", + "compute.instances.setMetadata", + "compute.instances.setTags", + "compute.instanceTemplates.create", + "compute.instanceTemplates.delete", + "compute.instanceTemplates.get", + "compute.instanceTemplates.useReadOnly", + "compute.networks.use", + "compute.subnetworks.use", + "compute.zoneOperations.get" + ] +} + +resource "google_project_iam_member" "github_production_relay_capacity_mutation" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role = google_project_iam_custom_role.github_production_relay_capacity_mutation[0].id + member = google_service_account.github_production_relay_capacity[0].member +} + +resource "google_project_iam_member" "github_production_relay_capacity_viewer" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role = "roles/viewer" + member = google_service_account.github_production_relay_capacity[0].member +} + +resource "google_project_iam_member" "github_production_relay_capacity_artifact_reader" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + project = var.project_id + role = "roles/artifactregistry.reader" + member = google_service_account.github_production_relay_capacity[0].member +} + +resource "google_storage_bucket_iam_member" "github_production_relay_capacity_state" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectAdmin" + member = google_service_account.github_production_relay_capacity[0].member + + condition { + title = "relay_production_capacity_state" + description = "Limits the production capacity workflow to the default Terraform state and lock." + expression = join(" || ", [ + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tfstate'", + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tflock'" + ]) + } +} + +resource "google_service_account_iam_member" "github_production_relay_capacity_runtime_user" { + count = local.create_production_relay_capacity_identity ? 1 : 0 + + service_account_id = google_service_account.relay_runtime.name + role = "roles/iam.serviceAccountUser" + member = google_service_account.github_production_relay_capacity[0].member +} + +# Candidate preflight reads reviewed Terraform outputs, but must not be able to mutate state. +resource "google_storage_bucket_iam_member" "github_terraform_state_reader" { + count = local.relay_create_production_ops_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectViewer" + member = google_service_account.github_deploy[0].member +} + + +# Relay deploys replace only the image while retaining the Terraform-owned +# runtime identity and service shape. +resource "google_service_account_iam_member" "github_relay_runtime_service_account_user" { + count = local.relay_create_github_deploy_identity ? 1 : 0 + + service_account_id = google_service_account.relay_runtime.name + role = "roles/iam.serviceAccountUser" + member = local.relay_github_deploy_service_account_member +} + +resource "google_service_account_iam_member" "github_relay_director_runtime_service_account_user" { + count = local.relay_create_github_deploy_identity ? 1 : 0 + + service_account_id = google_service_account.relay_director_runtime.name + role = "roles/iam.serviceAccountUser" + member = local.relay_github_deploy_service_account_member +} + diff --git a/cloud/infra/terraform/relay-github-workflow-trust.tf b/cloud/infra/terraform/relay-github-workflow-trust.tf new file mode 100644 index 00000000000..9cf57f05198 --- /dev/null +++ b/cloud/infra/terraform/relay-github-workflow-trust.tf @@ -0,0 +1,38 @@ +# The workflow half of every relay Workload Identity condition, one clause per accepted +# repository. +# +# Each provider file builds its own clause list from local.relay_github_workflow_ref_prefixes, so +# a workflow allowlist stays next to the identity it authorizes. This file only names those lists +# and renders them: with a single accepted repository the clause is spliced in unchanged, and with +# more it becomes one parenthesised OR arm per repository, each arm carrying its own repository +# claims. The repository-independent claims (ref, environment, event_name) stay outside the OR in +# the provider blocks. +locals { + relay_github_workflow_clauses = { + github = local.github_production_relay_workflow_clauses + github_monitor = local.github_monitor_workflow_clauses + github_fence = local.github_fence_workflow_clauses + github_production_relay_capacity = local.github_production_relay_capacity_workflow_clauses + github_staging_relay_capacity = local.github_staging_relay_capacity_workflow_clauses + github_staging_relay_deploy = local.github_staging_relay_deploy_workflow_clauses + github_relay_asia_topology = local.github_relay_asia_topology_workflow_clauses + github_relay_asia_proof = local.github_relay_asia_proof_workflow_clauses + } + + relay_github_workflow_arms = { + for name, clauses in local.relay_github_workflow_clauses : + name => [ + for index, clause in clauses : + "(${local.relay_github_accepted_repository_claims[index]} && ${clause})" + ] + } + + relay_github_workflow_conditions = { + for name, clauses in local.relay_github_workflow_clauses : + name => ( + local.relay_github_single_repository + ? clauses[0] + : "(${join(" || ", local.relay_github_workflow_arms[name])})" + ) + } +} diff --git a/cloud/infra/terraform/relay-observability.tf b/cloud/infra/terraform/relay-observability.tf new file mode 100644 index 00000000000..a8fa4276eb2 --- /dev/null +++ b/cloud/infra/terraform/relay-observability.tf @@ -0,0 +1,525 @@ +locals { + relay_service_names = concat( + [var.relay_cloud_run_service_name], + [for cell in values(var.relay_cells) : cell.service_name] + ) + relay_service_log_filter = join(" OR ", [ + for name in local.relay_service_names : "resource.labels.service_name=\"${name}\"" + ]) + relay_runtime_log_filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR (resource.type=\"gce_instance\" AND jsonPayload.role=\"cell\")) AND jsonPayload.event=\"orca_relay_runtime_metrics\"" + relay_gce_connection_warning_thresholds = { + for cell_id, cell in var.relay_gce_cells : + cell_id => cell.connection_hard_cap == null ? 550 : floor( + (cell.connection_hard_cap - 100 - cell.connection_unobserved_bound) * 0.85 + ) + } + relay_gce_connection_warning_groups = { + for threshold in distinct(values(local.relay_gce_connection_warning_thresholds)) : + tostring(threshold) => sort([ + for cell_id, cell_threshold in local.relay_gce_connection_warning_thresholds : + cell_id if cell_threshold == threshold + ]) + } + relay_incident_metrics = { + assignment_5xx = { + description = "Director assignment requests returning a server error." + filter = "resource.type=\"cloud_run_revision\" AND resource.labels.service_name=\"${var.relay_cloud_run_service_name}\" AND httpRequest.requestMethod=\"POST\" AND httpRequest.requestUrl=~\"/v1/assign$\" AND httpRequest.status>=500" + } + assignment_edge_429 = { + description = "Director assignment or resolve requests rejected before an instance was available." + filter = "resource.type=\"cloud_run_revision\" AND resource.labels.service_name=\"${var.relay_cloud_run_service_name}\" AND httpRequest.requestMethod=\"POST\" AND httpRequest.requestUrl=~\"/v1/(assign|resolve)$\" AND httpRequest.status=429" + } + postgres_retries = { + description = "Relay PostgreSQL transactions recovered after a retryable abort." + filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_retry\"" + } + postgres_retry_exhausted = { + description = "Relay PostgreSQL transactions that exhausted bounded retry." + filter = "((resource.type=\"cloud_run_revision\" AND (${local.relay_service_log_filter})) OR resource.type=\"gce_instance\") AND jsonPayload.event=\"orca_relay_postgres_transaction_exhausted\"" + } + } + + relay_runtime_metrics = { + total_connections = { field = "totalConnections", description = "Open relay WebSocket requests per process." } + controls = { field = "controls", description = "Authenticated standing desktop controls per process." } + splices = { field = "splices", description = "Active phone-to-desktop ciphertext splices per process." } + pending_splices = { field = "pendingSplices", description = "Phone connections waiting for their host data leg." } + queued_bytes = { field = "queuedBytes", description = "Process-wide relay backpressure bytes." } + http_latency_ms = { field = "httpLatencyMsMax", description = "Maximum director/administrative HTTP latency in the interval." } + sql_latency_ms = { field = "sqlLatencyMsMax", description = "Maximum observed SQL operation latency in the interval." } + control_renewal_latency_ms_p50 = { field = "controlRenewalLatencyMsP50", description = "Control renewal latency p50 in the interval." } + control_renewal_latency_ms_p95 = { field = "controlRenewalLatencyMsP95", description = "Control renewal latency p95 in the interval." } + control_renewal_latency_ms_max = { field = "controlRenewalLatencyMsMax", description = "Maximum control renewal latency in the interval." } + control_renewals = { field = "controlRenewalsDelta", description = "Control renewal attempts in the interval." } + control_renewal_successes = { field = "controlRenewalSuccessesDelta", description = "Successful control renewals in the interval." } + control_renewal_lease_misses = { field = "controlRenewalLeaseMissesDelta", description = "Control renewals that found their activity lease missing." } + control_activity_recoveries = { field = "controlActivityRecoveriesDelta", description = "Control activity leases recovered after a renewal miss." } + control_activity_recovery_failures = { field = "controlActivityRecoveryFailuresDelta", description = "Control activity lease recovery attempts that failed." } + heap_used_bytes = { field = "heapUsedBytes", description = "Node.js heap bytes used by the relay process." } + event_loop_ms_p99 = { field = "eventLoopDelayMsP99", description = "Node.js event-loop delay p99 in milliseconds." } + forwarded_bytes = { field = "forwardedBytesDelta", description = "Ciphertext bytes admitted for forwarding." } + auth_successes = { field = "authSuccessesDelta", description = "Successful outer relay authentication stages." } + auth_failures = { field = "authFailuresDelta", description = "Rejected or timed-out outer relay authentication stages." } + reconnects = { field = "reconnectsDelta", description = "Desktop control rebinds or generation replacements." } + sql_queries = { field = "sqlQueriesDelta", description = "Completed relay SQL operations." } + sql_failures = { field = "sqlFailuresDelta", description = "Failed relay SQL operations." } + db_pool_total = { field = "databasePoolTotal", description = "Open PostgreSQL connections in the process pool." } + db_pool_idle = { field = "databasePoolIdle", description = "Idle PostgreSQL connections in the process pool." } + db_pool_waiting = { field = "databasePoolWaiting", description = "Current requests queued for a PostgreSQL connection." } + db_waiters_max = { field = "databasePoolWaitersMax", description = "Maximum requests queued for a PostgreSQL connection during the interval." } + db_oldest_wait_ms = { field = "databasePoolOldestWaitMs", description = "Current oldest PostgreSQL pool waiter age." } + db_wait_ms_max = { field = "databasePoolWaitMsMax", description = "Maximum PostgreSQL pool wait during the interval." } + } + relay_custom_alerts = { + connection_headroom = { + pages_oncall = true + metric = "total_connections" + # Live values, set by hand on 2026-08-05. The cell arm is ~85% of the 440 usable + # connection units (600 hard cap - 100 rebind reserve - 60 unobserved bound); 800 + # exceeded the 600 cap outright and could never fire. + threshold_run = 550 + threshold_gce = 374 + duration = "120s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "A relay process is above its reviewed connection warning point; stop new assignment to the cell and follow the hot-cell runbook." + } + queue_pressure = { + pages_oncall = false + metric = "queued_bytes" + threshold_run = 50331648 + threshold_gce = 50331648 + duration = "120s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Process-wide queued ciphertext exceeded 75% of the 64 MiB hard budget. Investigate slow receivers before admission starts rejecting." + } + auth_failures = { + pages_oncall = false + metric = "auth_failures" + threshold_run = 20 + threshold_gce = 20 + duration = "0s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Outer authentication failures exceeded the per-process interval threshold. Check auth/JWKS health and abuse sources without logging bearer values." + } + reconnects = { + pages_oncall = false + metric = "reconnects" + threshold_run = 100 + threshold_gce = 100 + duration = "0s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Relay control reconnects exceeded the per-process interval threshold. Check revision churn, GFE terminations, and reconnect jitter." + } + sql_failures = { + pages_oncall = true + # Tuned live on 2026-08-05 to stop Slack alert noise; codified here so an apply cannot revert it. + metric = "sql_failures" + threshold_run = 0 + threshold_gce = 0 + duration = "300s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "A relay SQL operation failed. Check Cloud SQL availability, connection pressure, and transaction retry outcomes." + } + sql_latency = { + pages_oncall = false + metric = "sql_latency_ms" + threshold_run = 500 + threshold_gce = 500 + duration = "120s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Relay SQL operations remained above 500 ms. Inspect lock contention and Cloud SQL health before migrations or drains." + } + database_pool_waiters = { + pages_oncall = true + # Tuned live on 2026-08-05 to stop Slack alert noise; codified here so an apply cannot revert it. + metric = "db_waiters_max" + threshold_run = 5 + threshold_gce = 5 + duration = "300s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "A relay process queued work for a PostgreSQL connection. Check public-assignment admission, pool saturation, and Cloud SQL latency before scaling." + } + database_pool_wait = { + pages_oncall = true + # Tuned live on 2026-08-05 to stop Slack alert noise; codified here so an apply cannot revert it. + metric = "db_wait_ms_max" + threshold_run = 500 + threshold_gce = 500 + duration = "300s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "A relay process waited over 500 ms for a PostgreSQL connection. Check long transactions and lock contention before migrations or drains." + } + http_latency = { + pages_oncall = false + metric = "http_latency_ms" + threshold_run = 2000 + threshold_gce = 2000 + duration = "120s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Relay HTTP handling remained above two seconds. Check director assignment/resolve latency and SQL contention; WebSocket lifetimes are intentionally excluded." + } + heap_pressure = { + pages_oncall = false + metric = "heap_used_bytes" + threshold_run = 419430400 + threshold_gce = 419430400 + duration = "120s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Node.js heap stayed above 400 MiB on a 512 MiB container. Drain the affected cell and investigate connection or queue retention." + } + event_loop_delay = { + pages_oncall = false + metric = "event_loop_ms_p99" + threshold_run = 250 + threshold_gce = 250 + duration = "120s" + aligner = "ALIGN_PERCENTILE_99" + reducer = "REDUCE_MAX" + documentation = "Relay event-loop p99 delay stayed above 250 ms. Check CPU, SQL callbacks, synchronized heartbeats, and queue pressure." + } + } +} + +resource "google_logging_metric" "relay_snapshot" { + for_each = local.relay_runtime_metrics + + project = var.project_id + name = "orca_relay_${each.key}" + description = each.value.description + filter = local.relay_runtime_log_filter + value_extractor = "EXTRACT(jsonPayload.${each.value.field})" + label_extractors = { + role = "EXTRACT(jsonPayload.role)" + cell_id = "EXTRACT(jsonPayload.cellId)" + region = "EXTRACT(jsonPayload.region)" + } + + metric_descriptor { + metric_kind = "DELTA" + value_type = "DISTRIBUTION" + unit = contains(["sql_latency_ms", "control_renewal_latency_ms_p50", "control_renewal_latency_ms_p95", "control_renewal_latency_ms_max", "http_latency_ms", "event_loop_ms_p99", "db_oldest_wait_ms", "db_wait_ms_max"], each.key) ? "ms" : each.key == "queued_bytes" || each.key == "heap_used_bytes" || each.key == "forwarded_bytes" ? "By" : "1" + + labels { + key = "role" + value_type = "STRING" + description = "Relay process role." + } + + labels { + key = "cell_id" + value_type = "STRING" + description = "Durable relay cell identifier." + } + + labels { + key = "region" + value_type = "STRING" + description = "Coarse Relay region." + } + } + + bucket_options { + exponential_buckets { + num_finite_buckets = 24 + growth_factor = 2 + scale = 1 + } + } +} + +resource "google_logging_metric" "relay_incident" { + for_each = local.relay_incident_metrics + + project = var.project_id + name = "orca_relay_${each.key}" + description = each.value.description + filter = each.value.filter + + metric_descriptor { + metric_kind = "DELTA" + value_type = "INT64" + unit = "1" + } +} + +resource "google_monitoring_alert_policy" "relay_custom" { + for_each = local.relay_custom_alerts + + project = var.project_id + display_name = "Orca Relay: ${replace(each.key, "_", " ")}" + combiner = "OR" + enabled = true + # Why: only the reviewed critical alerts page; the rest stay in the console. Applying one + # list to every policy would have put the noisy ones back into Slack. + notification_channels = each.value.pages_oncall ? var.relay_alert_notification_channels : [] + + conditions { + display_name = each.key + + condition_threshold { + filter = "resource.type=\"cloud_run_revision\" AND metric.type=\"logging.googleapis.com/user/orca_relay_${each.value.metric}\"" + comparison = "COMPARISON_GT" + threshold_value = each.value.threshold_run + duration = each.value.duration + + aggregations { + alignment_period = "300s" + per_series_aligner = each.value.aligner + cross_series_reducer = each.value.reducer + group_by_fields = ["resource.label.\"service_name\""] + } + + trigger { + count = 1 + } + } + } + + dynamic "conditions" { + for_each = each.key == "connection_headroom" ? [] : [each.value] + content { + display_name = "${each.key} (GCE cell)" + + condition_threshold { + filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_${conditions.value.metric}\"" + comparison = "COMPARISON_GT" + threshold_value = conditions.value.threshold_gce + duration = conditions.value.duration + + aggregations { + alignment_period = "300s" + per_series_aligner = conditions.value.aligner + cross_series_reducer = conditions.value.reducer + group_by_fields = ["resource.label.\"instance_id\""] + } + + trigger { + count = 1 + } + } + } + } + + documentation { + content = each.value.documentation + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_snapshot] +} + +resource "google_monitoring_alert_policy" "relay_assignment_5xx" { + project = var.project_id + display_name = "Orca Relay: assignment 5xx" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "assignment 5xx" + + condition_threshold { + filter = "resource.type=\"cloud_run_revision\" AND metric.type=\"logging.googleapis.com/user/orca_relay_assignment_5xx\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "0s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"service_name\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "A desktop could not obtain a Relay assignment. Check PostgreSQL retry/exhaustion signals and director SQL latency; do not restart GCE cells or invalidate pairings." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_gce_connection_headroom" { + for_each = local.relay_gce_connection_warning_groups + + project = var.project_id + display_name = "Orca Relay: connection headroom (GCE ${each.key})" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "connection headroom (GCE ${each.key})" + + condition_threshold { + filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_total_connections\" AND (${join(" OR ", [for cell_id in each.value : "metric.label.\"cell_id\"=\"${cell_id}\""])})" + comparison = "COMPARISON_GT" + threshold_value = tonumber(each.key) + duration = "120s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_PERCENTILE_99" + cross_series_reducer = "REDUCE_MAX" + group_by_fields = ["resource.label.\"instance_id\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "A Relay GCE cell is above 85% of its configured ordinary placement ceiling; stop new assignment to the cell and follow the hot-cell runbook." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_snapshot] +} + +resource "google_monitoring_alert_policy" "relay_assignment_edge_429" { + project = var.project_id + display_name = "Orca Relay: assignment edge 429" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "assignment edge 429" + + condition_threshold { + filter = "resource.type=\"cloud_run_revision\" AND metric.type=\"logging.googleapis.com/user/orca_relay_assignment_edge_429\"" + comparison = "COMPARISON_GT" + threshold_value = 100 + duration = "0s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"service_name\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Cloud Run rejected a sustained assignment burst before an instance was available. Check public-assignment admission, director concurrency, and database latency before scaling." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_postgres_retry_exhausted" { + project = var.project_id + display_name = "Orca Relay: PostgreSQL retry exhausted" + combiner = "OR" + enabled = true + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "PostgreSQL retry exhausted (Cloud Run)" + + condition_threshold { + filter = "resource.type=\"cloud_run_revision\" AND metric.type=\"logging.googleapis.com/user/orca_relay_postgres_retry_exhausted\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "180s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"service_name\""] + } + + trigger { + count = 1 + } + } + } + + conditions { + display_name = "PostgreSQL retry exhausted (GCE cell)" + + condition_threshold { + filter = "resource.type=\"gce_instance\" AND metric.type=\"logging.googleapis.com/user/orca_relay_postgres_retry_exhausted\"" + comparison = "COMPARISON_GT" + threshold_value = 0 + duration = "180s" + + aggregations { + alignment_period = "300s" + per_series_aligner = "ALIGN_SUM" + cross_series_reducer = "REDUCE_SUM" + group_by_fields = ["resource.label.\"instance_id\""] + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "A Relay database transaction remained unsuccessful after bounded whole-transaction retry. Check Cloud SQL deadlocks/locks and customer-visible request failures before changing cell admission." + mime_type = "text/markdown" + } + + depends_on = [google_logging_metric.relay_incident] +} + +resource "google_monitoring_alert_policy" "relay_cloud_sql_backends" { + project = var.project_id + display_name = "Orca Relay: Cloud SQL connection headroom" + combiner = "OR" + enabled = true + + notification_channels = var.relay_alert_notification_channels + + conditions { + display_name = "Cloud SQL backends above 320" + + condition_threshold { + filter = "resource.type=\"cloudsql_database\" AND resource.label.\"database_id\"=\"${var.project_id}:${local.relay_database_instance_name}\" AND metric.type=\"cloudsql.googleapis.com/database/postgresql/num_backends\"" + comparison = "COMPARISON_GT" + threshold_value = 320 + duration = "300s" + + aggregations { + alignment_period = "60s" + per_series_aligner = "ALIGN_MAX" + } + + trigger { + count = 1 + } + } + } + + documentation { + content = "Cloud SQL connections exceeded 80% of the 400-connection ceiling. Pause Relay pool or cell growth and inspect pool waits before the modeled 385-connection operating maximum is reached." + mime_type = "text/markdown" + } +} diff --git a/cloud/infra/terraform/relay-shared.tf b/cloud/infra/terraform/relay-shared.tf new file mode 100644 index 00000000000..d4f0e5dac15 --- /dev/null +++ b/cloud/infra/terraform/relay-shared.tf @@ -0,0 +1,90 @@ +# Values the relay root shares with the foundation and apps roots, expressed as literals or +# data lookups so the relay never references another root's resources. Every literal here +# renders byte-identically to the resource attribute it replaces; the partition test pins that. +locals { + relay_shared_labels = { + app = "orca-cloud" + environment = var.environment + managed_by = "terraform" + } + relay_github_repository = "${var.github_owner}/${var.github_repo}" + relay_github_repository_claims = [ + "assertion.repository == '${local.relay_github_repository}'", + "assertion.repository_id == '${var.github_repo_id}'", + "assertion.repository_owner_id == '${var.github_owner_id}'", + ] + + # Dual accept during the public extraction: the primary repository first, then every repository + # var.github_accepted_repositories adds. Each one renders its own OR arm in every provider + # condition, so both repos can run the same workflows through the same identities. A repository + # that imports these workflows may rename the files, hence the per-repository prefix. + relay_github_accepted_repositories = concat([{ + owner = var.github_owner + repo = var.github_repo + repo_id = var.github_repo_id + owner_id = var.github_owner_id + workflow_file_prefix = "" + }], var.github_accepted_repositories) + + relay_github_single_repository = length(local.relay_github_accepted_repositories) == 1 + + relay_github_accepted_repository_names = [ + for repository in local.relay_github_accepted_repositories : + "${repository.owner}/${repository.repo}" + ] + + # Everything before the workflow file name, per accepted repository. + relay_github_workflow_ref_prefixes = [ + for repository in local.relay_github_accepted_repositories : + "${repository.owner}/${repository.repo}/.github/workflows/${repository.workflow_file_prefix}" + ] + + # The three repository claims as one conjunction, per accepted repository, for the OR arms. + relay_github_accepted_repository_claims = [ + for repository in local.relay_github_accepted_repositories : + join(" && ", [ + "assertion.repository == '${repository.owner}/${repository.repo}'", + "assertion.repository_id == '${repository.repo_id}'", + "assertion.repository_owner_id == '${repository.owner_id}'" + ]) + ] + + # With one accepted repository the claims lead each condition exactly as they always have. With + # more they move inside the arms, because a leading claim would contradict the other arm. + relay_github_leading_repository_claims = ( + local.relay_github_single_repository ? local.relay_github_repository_claims : [] + ) + + relay_create_github_deploy_identity = var.github_owner != "" && var.github_repo != "" + relay_create_production_ops_identity = local.relay_create_github_deploy_identity && var.environment == "production" + + # google_service_account.github_deploy lives in the relay root in production and in the apps + # root in staging, so its email is derived rather than read, matching the runtime accounts above. + # Staging Relay workflows authenticate as the relay-owned github_staging_relay_deploy account + # instead, so every relay binding, the cell startup metadata, and the director env follow it. + relay_github_deploy_service_account_email = ( + var.environment == "production" + ? "${var.name_prefix}-gha-deploy@${var.project_id}.iam.gserviceaccount.com" + : "${var.name_prefix}-gha-relay@${var.project_id}.iam.gserviceaccount.com" + ) + relay_github_deploy_service_account_member = "serviceAccount:${local.relay_github_deploy_service_account_email}" + + relay_workload_identity_pool_id = "${var.name_prefix}-github" + relay_workload_identity_pool_name = "projects/${data.google_project.relay.number}/locations/global/workloadIdentityPools/${local.relay_workload_identity_pool_id}" + + # Cloud SQL instance is foundation-owned; cell plans already derive the connection name so + # they stay independent of database drift. + relay_database_instance_name = "${var.name_prefix}-auth-db" + relay_database_connection_name = "${var.project_id}:${var.region}:${local.relay_database_instance_name}" +} + +data "google_project" "relay" { + project_id = var.project_id +} + +# Existence checks only: nothing in a plan value depends on these reads. +data "google_artifact_registry_repository" "relay_images" { + project = var.project_id + location = var.region + repository_id = var.artifact_repository_id +} diff --git a/cloud/infra/terraform/relay-staging-deploy-iam.tf b/cloud/infra/terraform/relay-staging-deploy-iam.tf new file mode 100644 index 00000000000..ac5f5632d57 --- /dev/null +++ b/cloud/infra/terraform/relay-staging-deploy-iam.tf @@ -0,0 +1,170 @@ +# The staging Relay deploy identity. +# +# In production the shared `github_deploy` account is relay-owned; in staging it belongs to +# infra/terraform-apps and four app workflows can also mint it. These resources give the five +# staging Relay workflows their own account, provider, and bindings so the relay root owns every +# credential its own workflows use. +# +# Bindings that already name local.relay_github_deploy_service_account_member follow that local +# (relay-shared.tf) and are NOT repeated here: the staging power custom role, serviceAccountUser +# on relay_runtime and relay_director_runtime, and the three regional-placement secret bindings. +# Everything below replaces a grant the apps root still makes to `github_deploy`. + +locals { + create_staging_relay_deploy_identity = ( + local.relay_create_github_deploy_identity && var.environment == "staging" + ) + github_staging_relay_deploy_workflow_files = [ + "bootstrap-relay-staging-capacity.yml", + "deploy-relay-staging-gce-candidate.yml", + "deploy-relay-staging.yml", + "operate-relay-asia-admission.yml", + "power-relay-staging.yml" + ] + github_staging_relay_deploy_workflow_clauses = [ + for prefix in local.relay_github_workflow_ref_prefixes : + "(${join(" || ", [for workflow_file in local.github_staging_relay_deploy_workflow_files : "assertion.workflow_ref == '${prefix}${workflow_file}@refs/heads/main'"])})" + ] + # Apps-owned account the shared staging power workflow scales to zero alongside the director. + staging_auth_runtime_service_account_email = "${var.name_prefix}-auth@${var.project_id}.iam.gserviceaccount.com" +} + +resource "google_service_account" "github_staging_relay_deploy" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + project = var.project_id + account_id = "${var.name_prefix}-gha-relay" + display_name = "Orca Relay staging deploy" + description = "Runs the exact reviewed Relay staging deploy, candidate, power, and admission workflows." +} + +resource "google_iam_workload_identity_pool_provider" "github_staging_relay_deploy" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + project = var.project_id + workload_identity_pool_id = local.relay_workload_identity_pool_id + workload_identity_pool_provider_id = "github-relay-deploy" + display_name = "GitHub Relay staging deploy" + + attribute_mapping = { + "google.subject" = "assertion.sub" + "attribute.repository" = "assertion.repository" + "attribute.repository_id" = "assertion.repository_id" + "attribute.repository_owner_id" = "assertion.repository_owner_id" + "attribute.ref" = "assertion.ref" + "attribute.environment" = "assertion.environment" + "attribute.workflow_ref" = "assertion.workflow_ref" + "attribute.relay_ops_identity" = "'staging-deploy'" + } + + # An allowlist of exact workflow refs, never a prefix: a namespace grant would hand this account + # to any future staging workflow with no Terraform diff. + attribute_condition = join(" && ", concat(local.relay_github_leading_repository_claims, [ + "assertion.ref == 'refs/heads/main'", + "assertion.environment == 'staging'", + local.relay_github_workflow_conditions["github_staging_relay_deploy"] + ])) + + oidc { + issuer_uri = "https://token.actions.githubusercontent.com" + } +} + +resource "google_service_account_iam_member" "github_staging_relay_deploy_workload_identity_user" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + service_account_id = google_service_account.github_staging_relay_deploy[0].name + role = "roles/iam.workloadIdentityUser" + member = "principalSet://iam.googleapis.com/${local.relay_workload_identity_pool_name}/attribute.relay_ops_identity/staging-deploy" +} + +# Every one of the five runs `terraform init` (or `infra.mjs init --env staging`) first, and the +# GCS backend lists the bucket before it can open the state. Bucket metadata only, no object +# access; this mirrors relay_fence_broker_bucket_reader. +resource "google_storage_bucket_iam_member" "github_staging_relay_deploy_state_list" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.legacyBucketReader" + member = google_service_account.github_staging_relay_deploy[0].member +} + +# Reads reviewed topology: `terraform output -json relay_gce_cell_deployments` and the +# `terraform console` binds in Deploy Relay Staging. No step in the five applies, so this is +# objectViewer, not the capacity identity's objectAdmin, over the same two objects. +resource "google_storage_bucket_iam_member" "github_staging_relay_deploy_state" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + bucket = "${var.project_id}-terraform-state" + role = "roles/storage.objectViewer" + member = google_service_account.github_staging_relay_deploy[0].member + + condition { + title = "relay_staging_deploy_state" + description = "Limits the staging Relay deploy workflows to the default Terraform state and lock." + expression = join(" || ", [ + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tfstate'", + "resource.name == 'projects/_/buckets/${var.project_id}-terraform-state/objects/terraform/state/default.tflock'" + ]) + } +} + +# Deploy Relay Staging step "Require the mirrored immutable image" runs +# `gcloud artifacts docker images describe`; nothing in the five writes to the repository. +resource "google_artifact_registry_repository_iam_member" "github_staging_relay_deploy_artifact_reader" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + project = var.project_id + location = var.region + repository = var.artifact_repository_id + role = "roles/artifactregistry.reader" + member = google_service_account.github_staging_relay_deploy[0].member +} + +# Deploy Relay Staging step "Deploy director blue/green", Operate Relay Asia Admission step +# "Deploy the registered additive director topology", and Power Relay Staging's scale-to-zero all +# drive `gcloud run services update|update-traffic` against the staging director. +resource "google_cloud_run_v2_service_iam_member" "github_staging_relay_deploy_director_developer" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + project = var.project_id + location = var.region + name = var.relay_cloud_run_service_name + role = "roles/run.developer" + member = google_service_account.github_staging_relay_deploy[0].member +} + +# Power Relay Staging step "Inspect or change staging power state" scales both entries of +# CLOUD_RUN_SERVICES in power-staging-relay.mjs to zero, and the second one is the shared staging +# auth service. The same workflow already stops the shared staging database through +# orcaRelayStagingPower, so this stays with the power operator rather than the apps root. +resource "google_cloud_run_v2_service_iam_member" "github_staging_relay_deploy_auth_developer" { + count = local.create_staging_relay_deploy_identity && var.relay_staging_power_auth_service_name != "" ? 1 : 0 + + project = var.project_id + location = var.region + name = var.relay_staging_power_auth_service_name + role = "roles/run.developer" + member = google_service_account.github_staging_relay_deploy[0].member +} + +# That same scale-to-zero mints a revision when the latest ready one is not already at zero, and +# Cloud Run gates a revision on actAs over the service's runtime account. +resource "google_service_account_iam_member" "github_staging_relay_deploy_auth_runtime_user" { + count = local.create_staging_relay_deploy_identity && var.relay_staging_power_auth_service_name != "" ? 1 : 0 + + service_account_id = "projects/${var.project_id}/serviceAccounts/${local.staging_auth_runtime_service_account_email}" + role = "roles/iam.serviceAccountUser" + member = google_service_account.github_staging_relay_deploy[0].member +} + +# Deploy Relay Staging GCE Candidate preflight reads MIG, instance, and backend-service topology +# (deploy-relay-gce-candidate.mjs inspectCell), which the power role's instanceGroupManagers.get +# does not cover. +resource "google_project_iam_member" "github_staging_relay_deploy_compute_viewer" { + count = local.create_staging_relay_deploy_identity ? 1 : 0 + + project = var.project_id + role = "roles/compute.viewer" + member = google_service_account.github_staging_relay_deploy[0].member +} diff --git a/cloud/infra/terraform/relay.tf b/cloud/infra/terraform/relay.tf new file mode 100644 index 00000000000..7a5124a00b1 --- /dev/null +++ b/cloud/infra/terraform/relay.tf @@ -0,0 +1,577 @@ +resource "google_service_account" "relay_runtime" { + project = var.project_id + account_id = "${var.name_prefix}-relay" + display_name = var.environment == "staging" ? "Orca Relay" : "Orca Relay cells" + description = var.environment == "staging" ? ( + "Runtime identity for the Orca Relay director and stamped cells." + ) : "Runtime identity for stamped Orca Relay cells." +} + +resource "google_service_account" "relay_director_runtime" { + project = var.project_id + account_id = "${var.name_prefix}-relay-dir" + display_name = "Orca Relay director" + description = "Runtime and regional rehoming caller identity for the Orca Relay director." +} + +resource "google_project_iam_member" "relay_runtime_cloudsql_client" { + project = var.project_id + role = "roles/cloudsql.client" + member = google_service_account.relay_runtime.member +} + +resource "google_project_iam_member" "relay_director_runtime_cloudsql_client" { + project = var.project_id + role = "roles/cloudsql.client" + member = google_service_account.relay_director_runtime.member +} + +resource "random_password" "relay_assignment_signing_key" { + length = 48 + special = false +} + +resource "google_secret_manager_secret" "relay_assignment_signing_key" { + project = var.project_id + secret_id = "orca-cloud-relay-assignment-signing-key" + labels = local.relay_shared_labels + + replication { + auto {} + } +} + +resource "google_secret_manager_secret_version" "relay_assignment_signing_key" { + secret = google_secret_manager_secret.relay_assignment_signing_key.id + secret_data = random_password.relay_assignment_signing_key.result +} + +resource "google_secret_manager_secret_iam_member" "relay_assignment_signing_key_accessor" { + project = var.project_id + secret_id = google_secret_manager_secret.relay_assignment_signing_key.secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.relay_runtime.member +} + +resource "google_secret_manager_secret_iam_member" "relay_assignment_signing_key_director_accessor" { + project = var.project_id + secret_id = google_secret_manager_secret.relay_assignment_signing_key.secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.relay_director_runtime.member +} + +resource "google_secret_manager_secret" "relay_regional_placement_enabled" { + project = var.project_id + secret_id = "orca-cloud-relay-regional-placement-enabled" + labels = local.relay_shared_labels + + replication { + auto {} + } +} + +resource "google_secret_manager_secret_version" "relay_regional_placement_enabled" { + secret = google_secret_manager_secret.relay_regional_placement_enabled.id + secret_data = tostring(var.relay_regional_placement_enabled) +} + +resource "google_secret_manager_secret_iam_member" "relay_regional_placement_runtime_accessor" { + project = var.project_id + secret_id = google_secret_manager_secret.relay_regional_placement_enabled.secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.relay_runtime.member +} + +resource "google_secret_manager_secret_iam_member" "relay_regional_placement_director_accessor" { + project = var.project_id + secret_id = google_secret_manager_secret.relay_regional_placement_enabled.secret_id + role = "roles/secretmanager.secretAccessor" + member = google_service_account.relay_director_runtime.member +} + +resource "google_secret_manager_secret_iam_member" "relay_regional_placement_deploy_accessor" { + count = local.relay_create_github_deploy_identity ? 1 : 0 + + project = var.project_id + secret_id = google_secret_manager_secret.relay_regional_placement_enabled.secret_id + role = "roles/secretmanager.secretAccessor" + member = local.relay_github_deploy_service_account_member +} + +resource "google_secret_manager_secret_iam_member" "relay_regional_placement_deploy_adder" { + count = local.relay_create_github_deploy_identity ? 1 : 0 + + project = var.project_id + secret_id = google_secret_manager_secret.relay_regional_placement_enabled.secret_id + role = "roles/secretmanager.secretVersionAdder" + member = local.relay_github_deploy_service_account_member +} + +resource "google_secret_manager_secret_iam_member" "relay_regional_placement_deploy_viewer" { + count = local.relay_create_github_deploy_identity ? 1 : 0 + + project = var.project_id + secret_id = google_secret_manager_secret.relay_regional_placement_enabled.secret_id + role = "roles/secretmanager.viewer" + member = local.relay_github_deploy_service_account_member +} + +data "external" "relay_serving_regional_placement_version" { + program = [ + "node", + "${path.module}/../../dev/scripts/read-relay-serving-regional-placement-version.mjs" + ] + query = { + project = var.project_id + region = var.region + service = var.relay_cloud_run_service_name + bootstrap_version = google_secret_manager_secret_version.relay_regional_placement_enabled.version + } +} + +resource "google_cloud_run_v2_service" "relay" { + project = var.project_id + name = var.relay_cloud_run_service_name + location = var.region + ingress = "INGRESS_TRAFFIC_ALL" + invoker_iam_disabled = true + labels = local.relay_shared_labels + + template { + service_account = google_service_account.relay_director_runtime.email + timeout = "${var.relay_director_request_timeout_seconds}s" + max_instance_request_concurrency = var.relay_director_concurrency + + scaling { + min_instance_count = var.relay_min_instances + max_instance_count = var.relay_max_instances + } + + volumes { + name = "cloudsql" + + cloud_sql_instance { + instances = [local.relay_database_connection_name] + } + } + + containers { + image = var.relay_cloud_run_image + + env { + name = "ORCA_RELAY_REGIONAL_PLACEMENT_ENABLED" + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.relay_regional_placement_enabled.secret_id + version = data.external.relay_serving_regional_placement_version.result.version + } + } + } + + ports { + container_port = 8080 + } + + volume_mounts { + name = "cloudsql" + mount_path = "/cloudsql" + } + + env { + name = "DATABASE_URL" + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.relay_database_url.secret_id + version = "latest" + } + } + } + + env { + name = "ORCA_RELAY_ASSIGNMENT_SIGNING_KEY" + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.relay_assignment_signing_key.secret_id + version = "latest" + } + } + } + + env { + name = "ORCA_RELAY_PUBLIC_URL" + value = var.relay_base_url + } + + env { + name = "ORCA_RELAY_CELL_URL" + value = var.relay_base_url + } + + env { + name = "ORCA_RELAY_AUTH_ISSUER" + value = var.auth_base_url + } + + env { + name = "ORCA_RELAY_AUTH_AUDIENCE" + value = "orca-relay" + } + + env { + name = "ORCA_RELAY_JWKS_URL" + value = "${var.auth_base_url}/.well-known/jwks.json" + } + + env { + name = "ORCA_RELAY_ROLE" + value = "director" + } + + env { + name = "ORCA_RELAY_ADMISSION_SELECTOR_VERSION" + value = "3" + } + + env { + name = "ORCA_RELAY_CELL_ID" + value = "director" + } + + env { + name = "ORCA_RELAY_CELLS_JSON" + value = local.relay_director_cells_json + } + + env { + name = "ORCA_RELAY_ADMIN_AUDIENCE" + value = "${var.relay_base_url}/v1/admin/drain" + } + + env { + name = "ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT" + value = local.relay_github_deploy_service_account_email + } + + env { + name = "ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT" + value = local.relay_capacity_service_account_email + } + + env { + name = "ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT" + value = local.relay_asia_proof_service_account_email + } + + env { + name = "ORCA_RELAY_MONITOR_SERVICE_ACCOUNT" + value = try(google_service_account.github_monitor[0].email, "") + } + + env { + name = "ORCA_RELAY_FENCE_SERVICE_ACCOUNT" + value = try(google_service_account.github_fence[0].email, "") + } + + env { + name = "ORCA_RELAY_FENCE_BROKER_SERVICE_ACCOUNT" + value = try(google_service_account.relay_fence_broker[0].email, "") + } + + env { + name = "ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT" + value = google_service_account.relay_runtime.email + } + + env { + name = "ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT" + value = google_service_account.relay_director_runtime.email + } + + env { + name = "ORCA_RELAY_REHOME_AUDIENCE" + value = "${var.relay_base_url}/v1/admin/host-drain" + } + + env { + name = "ORCA_RELAY_HEARTBEAT_AUDIENCE" + value = "${var.relay_base_url}/v1/admin/cell-heartbeat" + } + + env { + name = "ORCA_RELAY_PUBLIC_ASSIGNMENTS_ENABLED" + value = tostring(var.relay_public_assignments_enabled) + } + + env { + name = "ORCA_RELAY_PUBLIC_ASSIGNMENT_CONCURRENCY" + value = tostring(var.relay_public_assignment_concurrency) + } + + env { + name = "ORCA_RELAY_PUBLIC_ASSIGNMENT_RETRY_AFTER_SECONDS" + value = tostring(var.relay_public_assignment_retry_after_seconds) + } + + env { + name = "ORCA_RELAY_PUBLIC_ASSIGNMENT_QUEUE_MAX" + value = tostring(var.relay_public_assignment_queue_max) + } + + env { + name = "ORCA_RELAY_PUBLIC_ASSIGNMENT_WAIT_MS" + value = tostring(var.relay_public_assignment_wait_ms) + } + + env { + name = "ORCA_RELAY_DATABASE_POOL_MAX" + value = tostring(var.relay_director_database_pool_max) + } + + env { + name = "ORCA_RELAY_PUBLIC_STICKY_CONCURRENCY" + value = tostring(var.relay_public_sticky_concurrency) + } + + env { + name = "ORCA_RELAY_PUBLIC_STICKY_QUEUE_MAX" + value = tostring(var.relay_public_sticky_queue_max) + } + + env { + name = "ORCA_RELAY_PUBLIC_STICKY_WAIT_MS" + value = tostring(var.relay_public_sticky_wait_ms) + } + + env { + name = "ORCA_RELAY_PUBLIC_STICKY_RETRY_AFTER_SECONDS" + value = tostring(var.relay_public_sticky_retry_after_seconds) + } + + resources { + limits = { + cpu = var.relay_cloud_run_cpu + memory = var.relay_cloud_run_memory + } + + cpu_idle = false + } + } + } + + # Deploys update immutable images; Terraform owns service shape and IAM. + lifecycle { + # Why: the relay refuses to boot unless both admission lanes fit the pool, so catch it + # at plan time rather than as a crash loop on the candidate revision. + precondition { + condition = (var.relay_public_assignment_concurrency + + var.relay_public_sticky_concurrency) <= var.relay_director_database_pool_max + error_message = "Placement plus reconnect admission must fit the director database pool." + } + + ignore_changes = [ + client, + client_version, + template[0].containers[0].image + ] + } + + depends_on = [ + data.google_artifact_registry_repository.relay_images, + google_project_iam_member.relay_director_runtime_cloudsql_client, + google_secret_manager_secret_iam_member.relay_assignment_signing_key_director_accessor, + google_secret_manager_secret_iam_member.relay_regional_placement_director_accessor, + google_secret_manager_secret_iam_member.relay_database_url_director_accessor, + google_secret_manager_secret_version.relay_assignment_signing_key, + google_secret_manager_secret_version.relay_regional_placement_enabled, + google_secret_manager_secret_version.relay_database_url + ] +} + +resource "google_cloud_run_v2_service" "relay_cell" { + for_each = var.relay_cells + + project = var.project_id + name = each.value.service_name + location = var.region + ingress = "INGRESS_TRAFFIC_ALL" + invoker_iam_disabled = true + # Require an explicit configuration change before a stamped cell can be decommissioned. + deletion_protection = each.value.deletion_protection + labels = merge(local.relay_shared_labels, { + "orca-relay-role" = "cell" + "orca-relay-cell" = each.key + }) + + template { + service_account = google_service_account.relay_runtime.email + timeout = "${var.relay_request_timeout_seconds}s" + max_instance_request_concurrency = var.relay_concurrency + + scaling { + min_instance_count = each.value.min_instances + max_instance_count = each.value.max_instances + } + + volumes { + name = "cloudsql" + + cloud_sql_instance { + instances = [local.relay_database_connection_name] + } + } + + containers { + image = var.relay_cloud_run_image + + ports { + container_port = 8080 + } + + volume_mounts { + name = "cloudsql" + mount_path = "/cloudsql" + } + + env { + name = "DATABASE_URL" + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.relay_database_url.secret_id + version = "latest" + } + } + } + + env { + name = "ORCA_RELAY_ASSIGNMENT_SIGNING_KEY" + + value_source { + secret_key_ref { + secret = google_secret_manager_secret.relay_assignment_signing_key.secret_id + version = "latest" + } + } + } + + env { + name = "ORCA_RELAY_PUBLIC_URL" + value = each.value.url + } + + env { + name = "ORCA_RELAY_CELL_URL" + value = each.value.url + } + + env { + name = "ORCA_RELAY_AUTH_ISSUER" + value = var.auth_base_url + } + + env { + name = "ORCA_RELAY_AUTH_AUDIENCE" + value = "orca-relay" + } + + env { + name = "ORCA_RELAY_JWKS_URL" + value = "${var.auth_base_url}/.well-known/jwks.json" + } + + env { + name = "ORCA_RELAY_ROLE" + value = "cell" + } + + env { + name = "ORCA_RELAY_CELL_ID" + value = each.key + } + + env { + name = "ORCA_RELAY_CELL_CAPACITY" + value = tostring(each.value.capacity_requests) + } + + env { + name = "ORCA_RELAY_CELLS_JSON" + value = "[]" + } + + env { + name = "ORCA_RELAY_ADMIN_AUDIENCE" + value = "${var.relay_base_url}/v1/admin/drain" + } + + env { + name = "ORCA_RELAY_DEPLOY_SERVICE_ACCOUNT" + value = local.relay_github_deploy_service_account_email + } + + env { + name = "ORCA_RELAY_CAPACITY_SERVICE_ACCOUNT" + value = local.relay_capacity_service_account_email + } + + env { + name = "ORCA_RELAY_ASIA_PROOF_SERVICE_ACCOUNT" + value = local.relay_asia_proof_service_account_email + } + + env { + name = "ORCA_RELAY_RUNTIME_SERVICE_ACCOUNT" + value = google_service_account.relay_runtime.email + } + + env { + name = "ORCA_RELAY_REHOME_DIRECTOR_SERVICE_ACCOUNT" + value = contains(var.relay_region_rehome_source_cell_ids, each.key) ? google_service_account.relay_director_runtime.email : "" + } + + env { + name = "ORCA_RELAY_REHOME_AUDIENCE" + value = contains(var.relay_region_rehome_source_cell_ids, each.key) ? "${var.relay_base_url}/v1/admin/host-drain" : "" + } + + env { + name = "ORCA_RELAY_DIRECTOR_URL" + value = var.relay_base_url + } + + env { + name = "ORCA_RELAY_HEARTBEAT_AUDIENCE" + value = "${var.relay_base_url}/v1/admin/cell-heartbeat" + } + + resources { + limits = { + cpu = var.relay_cloud_run_cpu + memory = var.relay_cloud_run_memory + } + + cpu_idle = false + } + } + } + + lifecycle { + ignore_changes = [ + client, + client_version, + template[0].containers[0].image + ] + } + + depends_on = [ + data.google_artifact_registry_repository.relay_images, + google_project_iam_member.relay_runtime_cloudsql_client, + google_secret_manager_secret_iam_member.relay_assignment_signing_key_accessor, + google_secret_manager_secret_iam_member.relay_database_url_accessor, + google_secret_manager_secret_version.relay_assignment_signing_key, + google_secret_manager_secret_version.relay_database_url + ] +} diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf new file mode 100644 index 00000000000..c3e06ee280f --- /dev/null +++ b/cloud/infra/terraform/variables.tf @@ -0,0 +1,468 @@ +variable "artifact_repository_id" { + type = string + description = "Artifact Registry Docker repository ID." +} + +variable "environment" { + type = string + description = "Deployment environment." + + validation { + condition = contains(["staging", "production"], var.environment) + error_message = "environment must be staging or production." + } +} + +variable "github_owner" { + type = string + description = "GitHub owner allowed to deploy through Workload Identity Federation." + default = "stablyai" +} + +variable "github_repo" { + type = string + description = "GitHub repo allowed to deploy through Workload Identity Federation." + default = "orca-cloud" +} + +# Numeric IDs survive a rename or transfer of the repository; every provider pins them next to the name. +variable "github_repo_id" { + type = string + description = "Numeric GitHub repository ID of github_owner/github_repo." + default = "1273841466" + + validation { + condition = can(regex("^[0-9]+$", var.github_repo_id)) + error_message = "github_repo_id must be the numeric repository ID." + } +} + +variable "github_owner_id" { + type = string + description = "Numeric GitHub owner ID of github_owner." + default = "127256420" + + validation { + condition = can(regex("^[0-9]+$", var.github_owner_id)) + error_message = "github_owner_id must be the numeric owner ID." + } +} + +# Additional repositories whose identical workflows the same identities must accept while the +# public extraction runs. Each entry renders its own OR arm in every provider condition, so the +# private repo keeps working while the public one takes over. `workflow_file_prefix` is the rename +# the importing repository applies to the workflow files it copies. Empty is the steady state: +# the final step of the cutover is to empty this list again and point github_owner/github_repo, +# github_repo_id, and github_owner_id at the surviving repository. +variable "github_accepted_repositories" { + type = list(object({ + owner = string + repo = string + repo_id = string + owner_id = string + workflow_file_prefix = string + })) + description = "Extra repositories accepted alongside github_owner/github_repo during the public extraction." + default = [] + + validation { + condition = alltrue([ + for repository in var.github_accepted_repositories : + can(regex("^[0-9]+$", repository.repo_id)) && can(regex("^[0-9]+$", repository.owner_id)) + ]) + error_message = "github_accepted_repositories entries must carry numeric repo_id and owner_id values." + } + + validation { + condition = alltrue([ + for repository in var.github_accepted_repositories : + can(regex("^[a-z0-9-]*$", repository.workflow_file_prefix)) + ]) + error_message = "github_accepted_repositories workflow_file_prefix must be lowercase letters, digits, or hyphens." + } +} + +variable "name_prefix" { + type = string + description = "Prefix used for named resources." +} + +variable "project_id" { + type = string + description = "GCP project ID." +} + +variable "region" { + type = string + description = "GCP region for regional resources." + default = "us-central1" +} + +variable "auth_base_url" { + type = string + description = "Public base URL of the auth service; OAuth callbacks and JWT issuer derive from it." +} + +variable "manage_relay_domain_mapping" { + type = bool + description = "Manage the Google Cloud Run mapping independently of the Cloudflare record." + default = false +} + +variable "relay_base_url" { + type = string + description = "Public TLS origin of the stable relay director." +} + +variable "relay_cloud_run_service_name" { + type = string + description = "Cloud Run service name for Orca Relay." +} + +variable "relay_staging_power_auth_service_name" { + type = string + description = "Shared staging auth Cloud Run service that Power Relay Staging scales to zero; empty outside staging." + default = "" +} + +variable "relay_cloud_run_image" { + type = string + description = "Initial image for the Terraform-created relay Cloud Run service." + default = "us-docker.pkg.dev/cloudrun/container/hello" +} + +variable "relay_cloud_run_cpu" { + type = string + description = "CPU limit for the relay container." + default = "1" +} + +variable "relay_cloud_run_memory" { + type = string + description = "Memory limit for the relay container." + default = "512Mi" +} + +variable "relay_fence_broker_service_name" { + type = string + description = "Private Cloud Run service that owns reviewed Relay Terraform fences." + default = "orca-cloud-relay-fence" +} + +variable "relay_fence_broker_image" { + type = string + description = "Immutable image for the private Relay fence broker." + default = "us-docker.pkg.dev/cloudrun/container/hello" + + validation { + condition = ( + var.relay_fence_broker_image == "us-docker.pkg.dev/cloudrun/container/hello" || + can(regex("^[a-z0-9.-]+/[a-z0-9._/-]+@sha256:[a-f0-9]{64}$", var.relay_fence_broker_image)) + ) + error_message = "relay_fence_broker_image must be the bootstrap image or an immutable digest." + } +} + +variable "relay_fence_source_cell_id" { + type = string + description = "Exact incident source cell accepted by the private fence broker." + default = "production-gce-c3" +} + +variable "relay_fence_failed_target_cell_id" { + type = string + description = "Exact failed registered target accepted by the private fence broker." + default = "production-gce-c12" +} + +variable "relay_fence_replacement_target_cell_id" { + type = string + description = "Exact replacement target accepted by the private fence broker." + default = "production-gce-c13" +} + +variable "relay_fence_unobserved_connection_bound" { + type = number + description = "Reviewed unobserved connection bound enforced during supersession." + default = 60 + + validation { + condition = ( + var.relay_fence_unobserved_connection_bound >= 0 && + var.relay_fence_unobserved_connection_bound < 500 + ) + error_message = "relay_fence_unobserved_connection_bound must be between zero and 499." + } +} + +variable "relay_director_concurrency" { + type = number + description = "Cloud Run concurrency for short-lived director HTTP requests." + default = 80 +} + +variable "relay_director_request_timeout_seconds" { + type = number + description = "Cloud Run timeout for short-lived director HTTP requests." + default = 30 +} + +variable "relay_concurrency" { + type = number + description = "Cloud Run cell concurrency; every WebSocket leg counts." + default = 1000 +} + +variable "relay_request_timeout_seconds" { + type = number + description = "Cloud Run cell request timeout for standing WebSocket legs." + default = 3600 +} + +variable "relay_public_assignments_enabled" { + type = bool + description = "Emergency switch for public assignment and resolve requests." + default = true +} + +variable "relay_regional_placement_enabled" { + type = bool + description = "Initial preferred-region placement state; audited director deploys own later changes." + default = true +} + +variable "relay_region_rehome_source_cell_ids" { + type = set(string) + description = "Reviewed US Relay cells allowed to advertise and accept the regional rehome source protocol." + default = [] +} + +variable "relay_public_assignment_concurrency" { + type = number + description = "Per-director public assignment operations allowed to reach shared state." + default = 2 +} + +variable "relay_public_assignment_retry_after_seconds" { + type = number + description = "Minimum retry interval enforced per relay host during assignment recovery." + default = 5 +} + +# Why: these three match the application defaults today. Pinning them keeps a code-side +# default change from silently re-tuning production on the next unrelated apply. +variable "relay_public_assignment_queue_max" { + type = number + description = "Queued public assignment operations allowed per director instance." + default = 128 +} + +variable "relay_public_assignment_wait_ms" { + type = number + description = "Milliseconds a public assignment waits for an admission slot before 503." + default = 4000 +} + +# Why: the sticky (reconnect) lane shared the assignment pool but lived only as a code +# default, so Terraform could not see it. Raising placement concurrency alone then pushed +# placement + sticky past the pool and the director refused to boot. +variable "relay_public_sticky_concurrency" { + type = number + description = "Per-director reconnect-lane operations allowed to reach shared state." + default = 1 +} + +variable "relay_public_sticky_queue_max" { + type = number + description = "Queued reconnect-lane operations allowed per director instance." + default = 64 +} + +variable "relay_public_sticky_wait_ms" { + type = number + description = "Milliseconds a reconnect waits for an admission slot before 503." + default = 2000 +} + +variable "relay_public_sticky_retry_after_seconds" { + type = number + description = "Minimum retry interval enforced per relay host during reconnect recovery." + default = 2 +} + +variable "relay_director_database_pool_max" { + type = number + description = "Director database pool size; must fit placement plus sticky admission slots." + default = 3 + + validation { + condition = var.relay_director_database_pool_max >= 3 + error_message = "The director pool must fit both placement and sticky admission slots." + } +} + +variable "relay_min_instances" { + type = number + description = "Minimum instances for the stable relay director." + default = 1 +} + +variable "relay_max_instances" { + type = number + description = "Maximum instances for the stateless stable relay director." + default = 2 + + validation { + condition = var.relay_max_instances >= 1 + error_message = "The relay director needs at least one instance." + } +} + +variable "relay_cells" { + type = map(object({ + service_name = string + url = string + capacity_requests = number + min_instances = number + max_instances = number + deletion_protection = optional(bool, true) + })) + description = "Explicit stamped max-one relay cells keyed by durable cell ID." + default = {} + + validation { + condition = alltrue([ + for cell in values(var.relay_cells) : + cell.max_instances == 1 && + cell.min_instances >= 0 && + cell.min_instances <= cell.max_instances && + cell.capacity_requests >= 1 && + cell.capacity_requests <= 1000 && + can(regex("^https://[^/]+$", cell.url)) + ]) + error_message = "Relay cells must use HTTPS origins, capacity 1..1000, and max exactly one." + } +} + +variable "relay_alert_notification_channels" { + type = list(string) + description = "Cloud Monitoring notification-channel resource names for Orca Relay alerts. Empty keeps policies visible without paging." + default = [] +} + +variable "relay_gce_domain" { + type = string + description = "Parent DNS name for GCE relay cells; each cell is one exact host below it." + default = "" + + validation { + condition = var.relay_gce_domain == "" || ( + can(regex("^[a-z0-9](?:[a-z0-9-]*[a-z0-9])?(?:\\.[a-z0-9](?:[a-z0-9-]*[a-z0-9])?)+$", var.relay_gce_domain)) && + !startswith(var.relay_gce_domain, "*.") + ) + error_message = "relay_gce_domain must be empty or a lowercase DNS name without a wildcard or scheme." + } +} + +variable "relay_gce_subnetwork_cidr" { + type = string + description = "Private IPv4 range dedicated to GCE relay cells." + default = "10.42.0.0/24" + + validation { + condition = can(cidrhost(var.relay_gce_subnetwork_cidr, 1)) + error_message = "relay_gce_subnetwork_cidr must be a valid IPv4 CIDR." + } +} + +variable "relay_gce_additional_region_subnetwork_cidrs" { + type = map(string) + description = "Private IPv4 ranges for additive Relay regions; the primary region keeps its legacy resources." + default = {} + + validation { + condition = alltrue([ + for region, cidr in var.relay_gce_additional_region_subnetwork_cidrs : + contains(["asia-east2"], region) && + can(cidrhost(cidr, 1)) + ]) + error_message = "Additional Relay regions must be allowlisted and use valid IPv4 CIDRs." + } +} + +variable "relay_gce_cells" { + type = map(object({ + hostname = string + region = optional(string, "us-central1") + zone = string + machine_type = string + boot_disk_gb = number + boot_image = string + capacity_requests = number + database_pool_max = optional(number, 10) + image = string + initially_enabled = optional(bool, true) + connection_hard_cap = optional(number) + connection_unobserved_bound = optional(number) + })) + description = "Private GCE relay cells keyed by durable cell ID; unfenced cells remain fixed-one." + default = {} + + validation { + condition = alltrue([ + for cell_id, cell in var.relay_gce_cells : + can(regex("^[a-z][a-z0-9-]{0,39}$", cell_id)) && + can(regex("^[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?$", cell.hostname)) && + contains(["us-central1", "asia-east2"], cell.region) && + startswith(cell.zone, "${cell.region}-") && + can(regex("^[a-z0-9-]+$", cell.machine_type)) && + can(regex("^https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-[a-z0-9-]+$", cell.boot_image)) && + cell.boot_disk_gb >= 20 && + cell.boot_disk_gb <= 100 && + cell.capacity_requests >= 1 && + cell.capacity_requests <= 100000 && + cell.database_pool_max >= 1 && + cell.database_pool_max <= 100 && + ( + (cell.connection_hard_cap == null && + cell.connection_unobserved_bound == null) || + try( + contains([600, 1000, 3000], cell.connection_hard_cap) && + cell.connection_unobserved_bound >= 0 && + cell.connection_unobserved_bound < cell.connection_hard_cap - 100, + false + ) + ) && + can(regex("^[a-z0-9.-]+/[a-z0-9._/-]+@sha256:[a-f0-9]{64}$", cell.image)) + ]) && length(distinct([for cell in values(var.relay_gce_cells) : cell.hostname])) == length(var.relay_gce_cells) + error_message = "GCE cells need an allowlisted region and matching zone, unique DNS labels, a pinned COS boot image, bounded machine/disk/capacity/pool values, paired supported connection limits with rebind headroom, and digest-pinned relay images." + } +} + +variable "relay_gce_cell_log_sample_rate" { + type = number + description = "Fraction of relay cell load-balancer requests written to Cloud Logging; 1 keeps assign-to-connection joins exact." + default = 1 + + validation { + condition = var.relay_gce_cell_log_sample_rate >= 0 && var.relay_gce_cell_log_sample_rate <= 1 + error_message = "relay_gce_cell_log_sample_rate must be between 0 and 1." + } +} + +variable "relay_gce_fenced_cells" { + type = set(string) + description = "Reviewed relay GCE cell IDs whose Terraform-owned MIG target size is zero." + default = [] +} + +variable "relay_gce_cloud_sql_proxy_image" { + type = string + description = "Digest-pinned Cloud SQL Auth Proxy image used by private relay workers." + default = "gcr.io/cloud-sql-connectors/cloud-sql-proxy@sha256:fc224915ef435afeb5b2a9421260a0d31986d5c8b7c7f5783c7f5d5885700cd2" + + validation { + condition = can(regex("^[a-z0-9.-]+/[a-z0-9._/-]+@sha256:[a-f0-9]{64}$", var.relay_gce_cloud_sql_proxy_image)) + error_message = "relay_gce_cloud_sql_proxy_image must be pinned by sha256 digest." + } +} diff --git a/cloud/infra/terraform/versions.tf b/cloud/infra/terraform/versions.tf new file mode 100644 index 00000000000..730f785a87b --- /dev/null +++ b/cloud/infra/terraform/versions.tf @@ -0,0 +1,27 @@ +terraform { + required_version = ">= 1.7.0" + + required_providers { + google = { + source = "hashicorp/google" + version = "~> 6.0" + } + + random = { + source = "hashicorp/random" + version = "~> 3.6" + } + + external = { + source = "hashicorp/external" + version = "~> 2.3" + } + } + + backend "gcs" {} +} + +provider "google" { + project = var.project_id + region = var.region +} diff --git a/cloud/package.json b/cloud/package.json new file mode 100644 index 00000000000..c75d027d3d2 --- /dev/null +++ b/cloud/package.json @@ -0,0 +1,33 @@ +{ + "name": "orca-cloud", + "private": true, + "version": "0.0.0", + "packageManager": "pnpm@10.24.0", + "engines": { + "node": ">=24 <27", + "pnpm": ">=10" + }, + "scripts": { + "build": "pnpm -r build", + "dev": "pnpm --filter @orca-cloud/relay dev", + "incident:relay": "pnpm --filter @orca-cloud/relay-ops incident:monitor", + "incident:relay-preflight": "pnpm --filter @orca-cloud/relay-ops incident:preflight", + "infra:apply": "node dev/scripts/infra.mjs apply", + "infra:init": "node dev/scripts/infra.mjs init", + "infra:plan": "node dev/scripts/infra.mjs plan", + "lint": "pnpm -r lint", + "load:relay:controls": "node dev/scripts/load-relay-controls.mjs", + "load:relay:model": "node dev/scripts/run-relay-load-model.mjs", + "load:relay:recovery-gate": "node dev/scripts/run-relay-recovery-wave-gate.mjs", + "ops:relay": "pnpm --filter @orca-cloud/relay-ops dev", + "pretest": "node --test dev/scripts/capture-terraform-plan-baseline.test.mjs dev/scripts/operate-relay-asia-admission.test.mjs dev/scripts/prepare-relay-asia-director-cells.test.mjs dev/scripts/prepare-relay-asia-topology-input.test.mjs dev/scripts/production-cloud-sql-rollout-lock.test.mjs dev/scripts/read-relay-serving-regional-placement-version.test.mjs dev/scripts/relay-asia-admission-workflow.test.mjs dev/scripts/relay-asia-rollout-evidence.test.mjs dev/scripts/relay-asia-topology-workflow.test.mjs dev/scripts/relay-cloud-sql-connection-budget.test.mjs dev/scripts/relay-load-reader-evidence.test.mjs dev/scripts/relay-staging-deploy-identity.test.mjs dev/scripts/sanitize-relay-asia-admission-result.test.mjs dev/scripts/terraform-root-partition.test.mjs dev/scripts/validate-relay-asia-topology-plan.test.mjs ../.github/actions/cloud-sql-rollout-lease/action-contract.test.mjs ../.github/actions/cloud-sql-rollout-lease/storage-lease.test.mjs", + "test": "pnpm -r test && node --test dev/scripts/classify-relay-production-capacity-director.test.mjs dev/scripts/classify-relay-staging-bootstrap.test.mjs dev/scripts/deploy-relay-blue-green.test.mjs dev/scripts/deploy-relay-gce-candidate.test.mjs dev/scripts/deploy-relay-gce-multi-target.test.mjs dev/scripts/github-smoke-token.test.mjs dev/scripts/infra.test.mjs dev/scripts/operate-relay-regional-rehome.test.mjs dev/scripts/power-staging-relay.test.mjs dev/scripts/prepare-relay-capacity-canary.test.mjs dev/scripts/prepare-relay-production-capacity-canary.test.mjs dev/scripts/probe-relay-legacy-admission.test.mjs dev/scripts/probe-relay-rehome-trust.test.mjs dev/scripts/production-cell-image-digest-consistency.test.mjs dev/scripts/read-relay-production-capacity-identity.test.mjs dev/scripts/relay-admission-selector.test.mjs dev/scripts/relay-gce-terraform-fence.test.mjs dev/scripts/relay-load-connection-failure.test.mjs dev/scripts/relay-load-control-peer.test.mjs dev/scripts/relay-load-director-capacity-gate.test.mjs dev/scripts/relay-load-model.test.mjs dev/scripts/relay-load-phase-barrier.test.mjs dev/scripts/relay-load-placement-boundary.test.mjs dev/scripts/relay-load-profile.test.mjs dev/scripts/relay-load-rebind-boundary.test.mjs dev/scripts/relay-load-region-behavior.test.mjs dev/scripts/relay-load-request-unit-boundary.test.mjs dev/scripts/relay-load-run-lifecycle.test.mjs dev/scripts/relay-monitor-evidence.test.mjs dev/scripts/relay-production-capacity-wave.test.mjs dev/scripts/relay-production-capacity-workflow.test.mjs dev/scripts/relay-production-identity-boundaries.test.mjs dev/scripts/relay-production-same-cap-wave.test.mjs dev/scripts/relay-public-workflow-contract.test.mjs dev/scripts/relay-recovery-wave-gate.test.mjs dev/scripts/relay-region-observation-evidence.test.mjs dev/scripts/relay-regional-rehome-workflow.test.mjs dev/scripts/relay-rehome-aggregate-evidence.test.mjs dev/scripts/relay-repository.test.mjs dev/scripts/relay-staging-c4-refresh-workflow.test.mjs dev/scripts/relay-staging-capacity-identity.test.mjs dev/scripts/staging-relay-apply-guard.test.mjs dev/scripts/validate-relay-capacity-plan.test.mjs dev/scripts/verify-relay-capacity-transition.test.mjs dev/scripts/verify-relay-legacy-bootstrap.test.mjs dev/scripts/workload-identity-attribute-conditions.test.mjs", + "typecheck": "pnpm -r typecheck" + }, + "devDependencies": { + "@types/node": "^24.10.0", + "tsx": "^4.21.0", + "typescript": "^5.9.3", + "vitest": "^4.0.8" + } +} diff --git a/cloud/packages/relay-contract/package.json b/cloud/packages/relay-contract/package.json new file mode 100644 index 00000000000..4c224b51085 --- /dev/null +++ b/cloud/packages/relay-contract/package.json @@ -0,0 +1,23 @@ +{ + "name": "@orca-cloud/relay-contract", + "private": true, + "version": "0.0.0", + "type": "module", + "main": "dist/index.js", + "types": "dist/index.d.ts", + "scripts": { + "build": "pnpm clean && tsc -p tsconfig.build.json", + "clean": "node -e \"require('fs').rmSync('dist', { recursive: true, force: true })\"", + "lint": "tsc -p tsconfig.json --noEmit", + "test": "vitest run", + "typecheck": "tsc -p tsconfig.json --noEmit" + }, + "dependencies": { + "zod": "^3.25.76" + }, + "devDependencies": { + "@types/node": "^24.10.0", + "typescript": "^5.9.3", + "vitest": "^4.0.8" + } +} diff --git a/cloud/packages/relay-contract/src/admission-budgets.ts b/cloud/packages/relay-contract/src/admission-budgets.ts new file mode 100644 index 00000000000..6dcf18ec3af --- /dev/null +++ b/cloud/packages/relay-contract/src/admission-budgets.ts @@ -0,0 +1,76 @@ +export const RELAY_ADMISSION_BUDGETS = { + cloudRunConcurrency: 900, + maxPreAuthConnections: 45, + maxPreAuthPerSource: 4, + maxPreAuthAttemptsPerSourcePerMinute: 30, + reservedHostControls: 100, + reservedHostDataSockets: 150, + maxProcessQueuedBytes: 64 * 1024 * 1024, + spliceLowWaterBytes: 64 * 1024, + spliceHighWaterBytes: 256 * 1024, + // Why: must admit at least one maxFrameBytes frame for a backpressured + // peer, or large catalog responses close the splice with LIMIT_EXCEEDED. + spliceHardQueuedBytes: 8 * 1024 * 1024 + 256 * 1024, + spliceWedgedTimeoutMs: 10 * 1000 +} as const + +export const RELAY_CELL_CONNECTION_HARD_CAPS = [600, 1_000, 3_000] as const +export type RelayCellConnectionHardCap = (typeof RELAY_CELL_CONNECTION_HARD_CAPS)[number] +export const RELAY_CELL_CONNECTION_HARD_CAP: RelayCellConnectionHardCap = 600 + +export function isRelayCellConnectionHardCap( + value: unknown +): value is RelayCellConnectionHardCap { + return RELAY_CELL_CONNECTION_HARD_CAPS.some((hardCap) => hardCap === value) +} + +// Legacy/default bounds remain available for fixed-600 operational gates. +export const RELAY_CELL_ADMISSION_BOUNDS = { + hardCap: RELAY_CELL_CONNECTION_HARD_CAP, + // Ordinary sockets stop here; the remainder is held for same-host control rebinds. + socketAdmissionCeiling: + RELAY_CELL_CONNECTION_HARD_CAP - RELAY_ADMISSION_BUDGETS.reservedHostControls, + // A cell must leave at least one unit admissible after its unobserved allowance. + maxUnobservedBound: + RELAY_CELL_CONNECTION_HARD_CAP - RELAY_ADMISSION_BUDGETS.reservedHostControls - 1 +} as const + +export function relayCellAdmissionBounds(hardCap: RelayCellConnectionHardCap): { + hardCap: RelayCellConnectionHardCap + socketAdmissionCeiling: number + maxUnobservedBound: number +} { + return { + hardCap, + socketAdmissionCeiling: hardCap - RELAY_ADMISSION_BUDGETS.reservedHostControls, + maxUnobservedBound: hardCap - RELAY_ADMISSION_BUDGETS.reservedHostControls - 1 + } +} + +export const RELAY_MAX_CELL_CONNECTION_UNOBSERVED_BOUND = relayCellAdmissionBounds( + RELAY_CELL_CONNECTION_HARD_CAPS.at(-1)! +).maxUnobservedBound + +// Director placement stops short of the socket ceiling by the cell's own unobserved allowance. +export function cellPlacementCeiling( + connectionHardCap: RelayCellConnectionHardCap, + connectionUnobservedBound: number +): number { + return relayCellAdmissionBounds(connectionHardCap).socketAdmissionCeiling - connectionUnobservedBound +} + +export function hasAdmissionCapacity(input: { + totalRequests: number + preAuthConnections: number + sourcePreAuthConnections: number + totalRequestCeiling?: number +}): boolean { + if (input.preAuthConnections >= RELAY_ADMISSION_BUDGETS.maxPreAuthConnections) return false + if (input.sourcePreAuthConnections >= RELAY_ADMISSION_BUDGETS.maxPreAuthPerSource) return false + const totalRequestCeiling = + input.totalRequestCeiling ?? + RELAY_ADMISSION_BUDGETS.cloudRunConcurrency - + RELAY_ADMISSION_BUDGETS.reservedHostControls - + RELAY_ADMISSION_BUDGETS.reservedHostDataSockets + return input.totalRequests < totalRequestCeiling +} diff --git a/cloud/packages/relay-contract/src/assignment-invariants.ts b/cloud/packages/relay-contract/src/assignment-invariants.ts new file mode 100644 index 00000000000..dc60a1ec7b8 --- /dev/null +++ b/cloud/packages/relay-contract/src/assignment-invariants.ts @@ -0,0 +1,56 @@ +import { z } from 'zod' +import { EpochMsSchema, GenerationSchema, OpaqueIdSchema, RelayHostIdSchema } from './wire-scalars.js' + +export const ASSIGNMENT_LIMITS = { + activityLeaseMs: 90 * 1000, + dormantTtlMs: 24 * 60 * 60 * 1000, + migrationLeaseMs: 15 * 60 * 1000 +} as const + +export const AssignmentActivitySchema = z + .object({ + relayHostId: RelayHostIdSchema, + cellId: OpaqueIdSchema, + assignmentEpoch: GenerationSchema, + leaseExpiresAt: EpochMsSchema, + lastActivityAt: EpochMsSchema, + reservedControls: z.number().int().nonnegative(), + reservedSplices: z.number().int().nonnegative(), + reservedInvites: z.number().int().nonnegative(), + pendingInstalls: z.number().int().nonnegative(), + pendingConfirmations: z.number().int().nonnegative(), + migrationLeases: z.number().int().nonnegative() + }) + .strict() + +export function hasAssignmentActivity(record: z.infer): boolean { + return ( + record.reservedControls + + record.reservedSplices + + record.reservedInvites + + record.pendingInstalls + + record.pendingConfirmations + + record.migrationLeases > + 0 + ) +} + +export function mayNormallyReassign( + record: z.infer, + now: number +): boolean { + return !hasAssignmentActivity(record) && now >= record.lastActivityAt + ASSIGNMENT_LIMITS.dormantTtlMs +} + +export const EvacuationCommitSchema = z + .object({ + relayHostId: RelayHostIdSchema, + sourceCellId: OpaqueIdSchema, + targetCellId: OpaqueIdSchema, + previousEpoch: GenerationSchema, + assignmentEpoch: GenerationSchema, + targetCapacityReserved: z.literal(true) + }) + .strict() + .refine((move) => move.sourceCellId !== move.targetCellId, 'target cell must differ') + .refine((move) => move.assignmentEpoch === move.previousEpoch + 1, 'epoch must increment exactly once') diff --git a/cloud/packages/relay-contract/src/close-codes.ts b/cloud/packages/relay-contract/src/close-codes.ts new file mode 100644 index 00000000000..cb2c6307b90 --- /dev/null +++ b/cloud/packages/relay-contract/src/close-codes.ts @@ -0,0 +1,10 @@ +export const RELAY_CLOSE_CODE = { + BAD_OUTER_CREDENTIAL: 4401, + HOST_OFFLINE: 4404, + PEER_DROPPED: 4408, + WRONG_CELL: 4409, + LIMIT_EXCEEDED: 4429, + DRAINING: 4503 +} as const + +export type RelayCloseCode = (typeof RELAY_CLOSE_CODE)[keyof typeof RELAY_CLOSE_CODE] diff --git a/cloud/packages/relay-contract/src/contract.test.ts b/cloud/packages/relay-contract/src/contract.test.ts new file mode 100644 index 00000000000..805cb8ea698 --- /dev/null +++ b/cloud/packages/relay-contract/src/contract.test.ts @@ -0,0 +1,347 @@ +import { describe, expect, it } from 'vitest' +import { RELAY_CLOSE_CODE } from './close-codes.js' +import { + AuthRefreshSchema, + DrainSchema, + HostChallengeSchema, + HostDataAuthSchema, + HostHelloAckSchema, + HostHelloSchema +} from './control-messages.js' +import { + DeviceCredentialInstallSchema, + DeviceResumeConfirmSchema, + RelayAuthSchema, + RelayHelloSchema +} from './credential-messages.js' +import { + AssignmentRequestSchema, + AssignmentResponseSchema, + isTrustedNewerMove, + RelayMovedSchema, + ResolveRequestSchema +} from './director-messages.js' +import { RELAY_PROTOCOL_LIMITS } from './protocol-limits.js' +import { RelayRegionCatalogResponseSchema } from './relay-regions.js' +import { + buildHostChallengePlaintext, + buildHostProofMacInput, + buildHostProofTranscript +} from './host-proof-transcript.js' +import { ConfirmableResumeTupleSchema } from './resume-confirmation-contract.js' +import { + canAdvanceSplice, + mayAcknowledgeClient, + SPLICE_STATE +} from './splice-state-machine.js' + +const TOKEN = 'abcdefghijklmnopqrstuvwxyzABCDEFGH012345678' +const NONCE = 'abcdefghijklmnopqrstuvwxyzABCDEF' +const KEY_B64 = Buffer.alloc(32, 1).toString('base64') + +describe('relay protocol contract', () => { + it('locks close codes and fixed normative limits', () => { + expect(RELAY_CLOSE_CODE).toEqual({ + BAD_OUTER_CREDENTIAL: 4401, + HOST_OFFLINE: 4404, + PEER_DROPPED: 4408, + WRONG_CELL: 4409, + LIMIT_EXCEEDED: 4429, + DRAINING: 4503 + }) + expect(RELAY_PROTOCOL_LIMITS).toMatchObject({ + firstFrameDeadlineMs: 2_000, + maxHttpBodyBytes: 4_096, + maxFrameBytes: 8 * 1024 * 1024, + maxConnectionsPerHost: 8, + idleTimeoutMs: 600_000, + inviteMaxAttempts: 5, + inviteReservationLeaseMs: 15_000, + hostAttachDeadlineMs: 10_000, + resumeConfirmationDeadlineMs: 30_000 + }) + }) + + it('strictly validates host registration and continuity fields', () => { + const hello = { + v: 1, + relayHostId: 'abcdefghijklmnop', + assignmentEpoch: 7, + hostPublicKeyB64: KEY_B64, + appVersion: '1.2.3' + } + expect(HostHelloSchema.safeParse(hello).success).toBe(true) + expect(HostHelloSchema.safeParse({ ...hello, userId: 'injected' }).success).toBe(false) + expect( + HostHelloAckSchema.safeParse({ + v: 1, + generation: 8, + controlResumeSecret: TOKEN, + leaseExpiresAt: 1_800_000_000_000, + activeConnIds: [], + pendingConns: [] + }).success + ).toBe(true) + }) + + it('locks bounded director assignment and resume-only resolve payloads', () => { + const assignment = { v: 1, relayHostId: 'abcdefghijklmnop' } + expect(AssignmentRequestSchema.safeParse(assignment).success).toBe(true) + expect( + AssignmentRequestSchema.safeParse({ ...assignment, preferredRegion: 'asia-east2' }).success + ).toBe(true) + expect( + AssignmentRequestSchema.safeParse({ ...assignment, preferredRegion: 'europe-west1' }).success + ).toBe(false) + expect(AssignmentRequestSchema.safeParse({ ...assignment, relayJwt: 'url-secret' }).success).toBe( + false + ) + expect( + AssignmentResponseSchema.safeParse({ + v: 1, + cellUrl: 'https://relay-c1.onorca.dev', + assignmentEpoch: 3, + lease: 'signed-lease' + }).success + ).toBe(true) + expect( + ResolveRequestSchema.safeParse({ + v: 1, + relayHostId: 'abcdefghijklmnop', + resumeToken: TOKEN + }).success + ).toBe(true) + }) + + it('accepts only unique regions with unique canonical HTTPS probe origins', () => { + const catalog = { + v: 1, + regions: [ + { + region: 'us-central1', + probeOrigins: ['https://relay-c1.example.test', 'https://relay-c2.example.test'] + }, + { region: 'asia-east2', probeOrigins: ['https://relay-c27.example.test'] } + ] + } + expect(RelayRegionCatalogResponseSchema.safeParse(catalog).success).toBe(true) + expect( + RelayRegionCatalogResponseSchema.safeParse({ + ...catalog, + regions: [catalog.regions[0], { ...catalog.regions[0] }] + }).success + ).toBe(false) + expect( + RelayRegionCatalogResponseSchema.safeParse({ + ...catalog, + regions: [ + catalog.regions[0], + { region: 'asia-east2', probeOrigins: ['https://relay-c1.example.test'] } + ] + }).success + ).toBe(false) + for (const origin of [ + 'http://relay-c1.example.test', + 'https://relay-c1.example.test/', + 'https://relay-c1.example.test/health', + 'https://relay-c1.example.test?probe=1' + ]) { + expect( + RelayRegionCatalogResponseSchema.safeParse({ + v: 1, + regions: [{ region: 'us-central1', probeOrigins: [origin] }] + }).success + ).toBe(false) + } + }) + + it('accepts moves only from the configured director at a strictly newer epoch', () => { + const move = RelayMovedSchema.parse({ + v: 1, + cellUrl: 'https://relay-c2.onorca.dev', + assignmentEpoch: 4 + }) + const base = { + configuredDirectorOrigin: 'https://relay.onorca.dev', + currentAssignmentEpoch: 3, + move + } + expect(isTrustedNewerMove({ ...base, sourceOrigin: 'https://relay.onorca.dev' })).toBe(true) + expect(isTrustedNewerMove({ ...base, sourceOrigin: move.cellUrl })).toBe(false) + expect( + isTrustedNewerMove({ + ...base, + sourceOrigin: 'https://relay.onorca.dev', + currentAssignmentEpoch: 4 + }) + ).toBe(false) + }) + + it('bounds challenge material and fixes director-only drain recovery', () => { + expect( + HostChallengeSchema.safeParse({ + challengeId: 'challenge-1', + relayEphemeralPublicKeyB64: KEY_B64, + nonceB64: NONCE, + ciphertextB64: KEY_B64, + expiresAt: 1_800_000_000_000 + }).success + ).toBe(true) + expect(DrainSchema.safeParse({ graceMs: 0, recovery: 'resolve-director' }).success).toBe(true) + expect(DrainSchema.safeParse({ graceMs: 0, recovery: 'follow-server-url' }).success).toBe(false) + expect(AuthRefreshSchema.safeParse({ relayJwt: 'jwt', identity: 'injected' }).success).toBe(false) + }) + + it('strictly binds one-shot host data tickets to a control generation', () => { + expect(HostDataAuthSchema.safeParse({ v: 1, connTicket: TOKEN, generation: 3 }).success).toBe( + true + ) + expect( + HostDataAuthSchema.safeParse({ + v: 1, + connTicket: TOKEN, + generation: 3, + relayDeviceId: 'injected' + }).success + ).toBe(false) + }) + + it('cannot acknowledge a splice before forwarding handlers exist', () => { + expect( + canAdvanceSplice(SPLICE_STATE.PRE_AUTH_ADMITTED, SPLICE_STATE.CREDENTIAL_LEASE_RESERVED) + ).toBe(true) + expect(canAdvanceSplice(SPLICE_STATE.PRE_AUTH_ADMITTED, SPLICE_STATE.SPLICED)).toBe(false) + expect(mayAcknowledgeClient(SPLICE_STATE.HOST_ATTACHED, false)).toBe(false) + expect(mayAcknowledgeClient(SPLICE_STATE.HOST_ATTACHED, true)).toBe(true) + }) + + it('locks the immutable server-owned resume tuple shape', () => { + const tuple = { + basisConnId: 'conn-1', + owningControlGeneration: 4, + relayDeviceId: 'device-1', + acceptedCredentialVersion: 2, + acceptedAs: 'current', + confirmDeadline: 1_800_000_000_000 + } + expect(ConfirmableResumeTupleSchema.safeParse(tuple).success).toBe(true) + expect(ConfirmableResumeTupleSchema.safeParse({ ...tuple, userId: 'caller-value' }).success).toBe( + false + ) + }) + + it('locks the complete host key-possession transcript', () => { + const transcript = buildHostProofTranscript({ + relayOrigin: 'https://relay.onorca.dev', + relayEphemeralPublicKey: new Uint8Array(32).fill(1), + challengeNonce: new Uint8Array(24).fill(4), + challengeId: 'challenge-1', + issuedAt: 1_700_000_000_000, + expiresAt: 1_700_000_010_000, + userId: 'user-1', + profileId: 'profile-1', + organizationId: 'org-1', + relayHostId: 'abcdefghijklmnop', + hostPublicKey: new Uint8Array(32).fill(2), + assignmentEpoch: 7, + previousGeneration: 6, + resumeRequested: true + }) + const challenge = buildHostChallengePlaintext(transcript, new Uint8Array(32).fill(3)) + const proofInput = buildHostProofMacInput(transcript) + + expect(Buffer.from(transcript).toString('base64url')).toBe( + 'AAAACHByb3RvY29sAAAAGG9yY2EtcmVsYXktaG9zdC1wcm9vZi92MQAAAAd2ZXJzaW9uAAAAAQEAAAALcmVsYXlPcmlnaW4AAAAYaHR0cHM6Ly9yZWxheS5vbm9yY2EuZGV2AAAAF3JlbGF5RXBoZW1lcmFsUHVibGljS2V5AAAAIAEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAAAADmNoYWxsZW5nZU5vbmNlAAAAGAQEBAQEBAQEBAQEBAQEBAQEBAQEBAQEBAAAAAtjaGFsbGVuZ2VJZAAAAAtjaGFsbGVuZ2UtMQAAAAhpc3N1ZWRBdAAAAAgAAAGLz-VoAAAAAAlleHBpcmVzQXQAAAAIAAABi8_ljxAAAAAGdXNlcklkAAAABnVzZXItMQAAAAlwcm9maWxlSWQAAAAJcHJvZmlsZS0xAAAADm9yZ2FuaXphdGlvbklkAAAABW9yZy0xAAAAC3JlbGF5SG9zdElkAAAAEGFiY2RlZmdoaWprbG1ub3AAAAANaG9zdFB1YmxpY0tleQAAACACAgICAgICAgICAgICAgICAgICAgICAgICAgICAgICAgAAAA9hc3NpZ25tZW50RXBvY2gAAAAIAAAAAAAAAAcAAAAScHJldmlvdXNHZW5lcmF0aW9uAAAACAAAAAAAAAAGAAAAD3Jlc3VtZVJlcXVlc3RlZAAAAAEB' + ) + expect(challenge.byteLength).toBe(transcript.byteLength + 65) + expect(proofInput.byteLength).toBe(transcript.byteLength + 29) + expect(() => buildHostChallengePlaintext(transcript, new Uint8Array(31))).toThrow( + 'challengeSecret must be 32 bytes' + ) + expect(() => + buildHostProofTranscript({ + relayOrigin: 'https://relay.onorca.dev', + relayEphemeralPublicKey: new Uint8Array(32), + challengeNonce: new Uint8Array(32), + challengeId: 'challenge-1', + issuedAt: 1, + expiresAt: 2, + userId: 'user-1', + profileId: 'profile-1', + organizationId: 'org-1', + relayHostId: 'abcdefghijklmnop', + hostPublicKey: new Uint8Array(32), + assignmentEpoch: 1, + resumeRequested: false + }) + ).toThrow('challengeNonce must be 24 bytes') + }) + + it('accepts only the bounded first-frame credential shape', () => { + expect(RelayAuthSchema.parse({ v: 1, mode: 'connect', credential: TOKEN })).toEqual({ + v: 1, + mode: 'connect', + credential: TOKEN + }) + expect( + RelayAuthSchema.safeParse({ v: 1, mode: 'connect', credential: TOKEN, extra: true }).success + ).toBe(false) + }) + + it('makes install authorization modes mutually exclusive', () => { + const base = { + v: 1, + reqId: 'req-1', + relayDeviceId: 'device-1', + newResumeTokenHash: TOKEN + } + expect( + DeviceCredentialInstallSchema.safeParse({ + ...base, + authorization: { mode: 'relay-basis', basisConnId: 'conn-1' } + }).success + ).toBe(true) + expect( + DeviceCredentialInstallSchema.safeParse({ + ...base, + authorization: { + mode: 'relay-basis', + basisConnId: 'conn-1', + directAuthId: 'injected' + } + }).success + ).toBe(false) + }) + + it('does not let invite attach observations impersonate resume state', () => { + expect( + RelayHelloSchema.safeParse({ + ok: true, + credentialKind: 'invite', + leaseExpiresAt: 1_800_000_000_000 + }).success + ).toBe(true) + expect( + RelayHelloSchema.safeParse({ + ok: true, + credentialKind: 'invite', + leaseExpiresAt: 1_800_000_000_000, + acceptedCredentialVersion: 7, + acceptedAs: 'current', + resumeExpiresAt: 1_800_000_000_000 + }).success + ).toBe(false) + }) + + it('rejects caller-supplied device and credential metadata on resume confirmation', () => { + const confirmation = { v: 1, reqId: 'req-1', basisConnId: 'conn-1' } + expect(DeviceResumeConfirmSchema.safeParse(confirmation).success).toBe(true) + expect( + DeviceResumeConfirmSchema.safeParse({ + ...confirmation, + relayDeviceId: 'injected', + acceptedCredentialVersion: 99 + }).success + ).toBe(false) + }) +}) diff --git a/cloud/packages/relay-contract/src/control-continuity.ts b/cloud/packages/relay-contract/src/control-continuity.ts new file mode 100644 index 00000000000..11821fc537e --- /dev/null +++ b/cloud/packages/relay-contract/src/control-continuity.ts @@ -0,0 +1,43 @@ +export const CONTROL_GENERATION_STATE = { + ACTIVE: 'active', + ORPHANED: 'orphaned', + DRAIN_ONLY: 'drain-only', + FENCED: 'fenced', + CLOSED: 'closed' +} as const + +export const CONTROL_CONTINUITY_LIMITS = { + orphanGraceMs: 30 * 1000, + authRefreshMinBeforeExpiryMs: 60 * 1000, + authRefreshMaxBeforeExpiryMs: 120 * 1000, + expiredAuthExistingSpliceGraceMs: 60 * 1000 +} as const + +export function controlLossDisposition(input: { + matchingResumeSecret: boolean + competingGeneration: boolean +}): 'rebind' | 'fence-old' | 'orphan-grace' { + if (input.matchingResumeSecret) return 'rebind' + if (input.competingGeneration) return 'fence-old' + return 'orphan-grace' +} + +export function mayStartNewRelayWork(state: string, authExpired: boolean): boolean { + return state === CONTROL_GENERATION_STATE.ACTIVE && !authExpired +} + +export interface RelayAuthIdentity { + sub: string + prof: string + org?: string + relayHostId: string +} + +export function preservesRelayAuthIdentity(previous: RelayAuthIdentity, refreshed: RelayAuthIdentity): boolean { + return ( + previous.sub === refreshed.sub && + previous.prof === refreshed.prof && + previous.org === refreshed.org && + previous.relayHostId === refreshed.relayHostId + ) +} diff --git a/cloud/packages/relay-contract/src/control-messages.ts b/cloud/packages/relay-contract/src/control-messages.ts new file mode 100644 index 00000000000..0d5f8d1b851 --- /dev/null +++ b/cloud/packages/relay-contract/src/control-messages.ts @@ -0,0 +1,119 @@ +import { z } from 'zod' +import { + Base6432ByteSchema, + Base64Raw24ByteSchema, + Base64Url32ByteSchema, + EpochMsSchema, + GenerationSchema, + OpaqueIdSchema, + PositiveDurationMsSchema, + RelayHostIdSchema +} from './wire-scalars.js' + +const AppVersionSchema = z.string().min(1).max(128) +const BoundedCiphertextSchema = z + .string() + .min(1) + .max(16 * 1024) + .regex(/^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/) +const ConnectionKindSchema = z.enum(['invite', 'resume']) + +export const HostHelloSchema = z + .object({ + v: z.literal(1), + relayHostId: RelayHostIdSchema, + assignmentEpoch: GenerationSchema, + hostPublicKeyB64: Base6432ByteSchema, + appVersion: AppVersionSchema, + previousGeneration: GenerationSchema.optional(), + controlResumeSecret: Base64Url32ByteSchema.optional() + }) + .strict() + +export const HostChallengeSchema = z + .object({ + challengeId: OpaqueIdSchema, + relayEphemeralPublicKeyB64: Base6432ByteSchema, + nonceB64: Base64Raw24ByteSchema, + ciphertextB64: BoundedCiphertextSchema, + expiresAt: EpochMsSchema + }) + .strict() + +export const HostChallengeAckSchema = z + .object({ challengeId: OpaqueIdSchema, proofB64: Base6432ByteSchema }) + .strict() + +const PendingConnectionSchema = z + .object({ connId: OpaqueIdSchema, connTicket: Base64Url32ByteSchema }) + .strict() + +export const HostHelloAckSchema = z + .object({ + v: z.literal(1), + generation: GenerationSchema, + controlResumeSecret: Base64Url32ByteSchema, + leaseExpiresAt: EpochMsSchema, + activeConnIds: z.array(OpaqueIdSchema).max(8), + pendingConns: z.array(PendingConnectionSchema).max(8) + }) + .strict() + +export const ConnectionOpenSchema = z + .object({ + connId: OpaqueIdSchema, + connTicket: Base64Url32ByteSchema, + kind: ConnectionKindSchema, + relayDeviceId: OpaqueIdSchema, + attachDeadlineMs: PositiveDurationMsSchema + }) + .strict() + +export const HostDataAuthSchema = z + .object({ + v: z.literal(1), + connTicket: Base64Url32ByteSchema, + generation: GenerationSchema + }) + .strict() + +export const InviteCreateSchema = z + .object({ reqId: OpaqueIdSchema, relayDeviceId: OpaqueIdSchema }) + .strict() + +export const InviteCreatedSchema = z + .object({ + reqId: OpaqueIdSchema, + inviteToken: Base64Url32ByteSchema, + expiresAt: EpochMsSchema, + maxAttempts: z.number().int().positive().max(16) + }) + .strict() + +export const DeviceRevokeSchema = z + .object({ reqId: OpaqueIdSchema, relayDeviceId: OpaqueIdSchema }) + .strict() + +export const AuthRefreshSchema = z.object({ relayJwt: z.string().min(1).max(8 * 1024) }).strict() + +export const DrainSchema = z + .object({ + graceMs: z.number().int().nonnegative().max(60 * 60 * 1000), + recovery: z.literal('resolve-director') + }) + .strict() + +export const HeartbeatSchema = z.object({ t: EpochMsSchema }).strict() + +export type HostHello = z.infer +export type HostChallenge = z.infer +export type HostChallengeAck = z.infer +export type HostHelloAck = z.infer +export type ConnectionOpen = z.infer +export type HostDataAuth = z.infer +export type InviteCreate = z.infer +export type InviteCreated = z.infer +export type DeviceRevoke = z.infer +export type AuthRefresh = z.infer +export type Drain = z.infer +export type Heartbeat = z.infer diff --git a/cloud/packages/relay-contract/src/credential-messages.ts b/cloud/packages/relay-contract/src/credential-messages.ts new file mode 100644 index 00000000000..658d2f25f89 --- /dev/null +++ b/cloud/packages/relay-contract/src/credential-messages.ts @@ -0,0 +1,105 @@ +import { z } from 'zod' +import { Base64Url32ByteSchema, EpochMsSchema, OpaqueIdSchema } from './wire-scalars.js' + +export const RelayAuthSchema = z + .object({ v: z.literal(1), mode: z.literal('connect'), credential: Base64Url32ByteSchema }) + .strict() + +const RelayErrorCodeSchema = z.union([ + z.literal(4401), + z.literal(4404), + z.literal(4408), + z.literal(4409), + z.literal(4429), + z.literal(4503) +]) + +export const RelayHelloSchema = z.union([ + z.object({ ok: z.literal(false), code: RelayErrorCodeSchema }).strict(), + z + .object({ + ok: z.literal(true), + credentialKind: z.literal('invite'), + leaseExpiresAt: EpochMsSchema + }) + .strict(), + z + .object({ + ok: z.literal(true), + credentialKind: z.literal('resume'), + leaseExpiresAt: EpochMsSchema, + acceptedCredentialVersion: z.number().int().positive(), + acceptedAs: z.enum(['current', 'grace']), + resumeExpiresAt: EpochMsSchema, + graceExpiresAt: EpochMsSchema.optional() + }) + .strict() +]) + +export const DeviceCredentialInstallSchema = z + .object({ + v: z.literal(1), + reqId: OpaqueIdSchema, + relayDeviceId: OpaqueIdSchema, + newResumeTokenHash: Base64Url32ByteSchema, + expectedCurrentHash: Base64Url32ByteSchema.optional(), + authorization: z.discriminatedUnion('mode', [ + z.object({ mode: z.literal('relay-basis'), basisConnId: OpaqueIdSchema }).strict(), + z.object({ mode: z.literal('authenticated-direct'), directAuthId: OpaqueIdSchema }).strict() + ]) + }) + .strict() + +export const DeviceCredentialInstallStatusSchema = z + .object({ v: z.literal(1), reqId: OpaqueIdSchema, relayDeviceId: OpaqueIdSchema }) + .strict() + +export const DeviceResumeConfirmSchema = z + .object({ v: z.literal(1), reqId: OpaqueIdSchema, basisConnId: OpaqueIdSchema }) + .strict() + +export const DeviceCredentialInstalledSchema = z + .object({ + v: z.literal(1), + reqId: OpaqueIdSchema, + authorizationMode: z.enum(['relay-basis', 'authenticated-direct']), + currentVersion: z.number().int().positive(), + resumeExpiresAt: EpochMsSchema, + graceExpiresAt: EpochMsSchema.optional() + }) + .strict() + +export const DeviceCredentialInstallStatusResultSchema = z.union([ + z.object({ v: z.literal(1), reqId: OpaqueIdSchema, state: z.literal('not-found') }).strict(), + z + .object({ + v: z.literal(1), + reqId: OpaqueIdSchema, + state: z.literal('committed'), + result: DeviceCredentialInstalledSchema + }) + .strict() +]) + +export const DeviceResumeConfirmedSchema = z + .object({ + v: z.literal(1), + reqId: OpaqueIdSchema, + currentVersion: z.number().int().positive(), + acceptedAs: z.enum(['current', 'grace']), + renewed: z.boolean(), + resumeExpiresAt: EpochMsSchema, + graceExpiresAt: EpochMsSchema.optional() + }) + .strict() + +export type RelayAuth = z.infer +export type RelayHello = z.infer +export type DeviceCredentialInstall = z.infer +export type DeviceCredentialInstalled = z.infer +export type DeviceCredentialInstallStatus = z.infer +export type DeviceCredentialInstallStatusResult = z.infer< + typeof DeviceCredentialInstallStatusResultSchema +> +export type DeviceResumeConfirm = z.infer +export type DeviceResumeConfirmed = z.infer diff --git a/cloud/packages/relay-contract/src/director-messages.ts b/cloud/packages/relay-contract/src/director-messages.ts new file mode 100644 index 00000000000..e697135b68e --- /dev/null +++ b/cloud/packages/relay-contract/src/director-messages.ts @@ -0,0 +1,75 @@ +import { z } from 'zod' +import { + Base64Url32ByteSchema, + CanonicalHttpsOriginSchema, + EpochMsSchema, + GenerationSchema, + RelayHostIdSchema +} from './wire-scalars.js' +import { RelayRegionSchema } from './relay-regions.js' + +const SignedAssignmentLeaseSchema = z.string().min(1).max(8 * 1024) + +export const AssignmentRequestSchema = z + .object({ + v: z.literal(1), + relayHostId: RelayHostIdSchema, + // Client-declared reconnection; the director verifies it against the + // durable assignment before granting fast-lane admission. + reconnect: z.boolean().optional(), + preferredRegion: RelayRegionSchema.optional() + }) + .strict() + +export const AssignmentResponseSchema = z + .object({ + v: z.literal(1), + cellUrl: CanonicalHttpsOriginSchema, + assignmentEpoch: GenerationSchema, + lease: SignedAssignmentLeaseSchema + }) + .strict() + +export const ResolveRequestSchema = z + .object({ + v: z.literal(1), + relayHostId: RelayHostIdSchema, + resumeToken: Base64Url32ByteSchema + }) + .strict() + +export const ResolveResponseSchema = z + .object({ + v: z.literal(1), + cellUrl: CanonicalHttpsOriginSchema, + assignmentEpoch: GenerationSchema, + leaseExpiresAt: EpochMsSchema + }) + .strict() + +export const RelayMovedSchema = z + .object({ + v: z.literal(1), + cellUrl: CanonicalHttpsOriginSchema, + assignmentEpoch: GenerationSchema + }) + .strict() + +export function isTrustedNewerMove(input: { + sourceOrigin: string + configuredDirectorOrigin: string + currentAssignmentEpoch: number + move: z.infer +}): boolean { + // Why: cells and stale director responses must never redirect a credential-bearing client. + return ( + input.sourceOrigin === input.configuredDirectorOrigin && + input.move.assignmentEpoch > input.currentAssignmentEpoch + ) +} + +export type AssignmentRequest = z.infer +export type AssignmentResponse = z.infer +export type ResolveRequest = z.infer +export type ResolveResponse = z.infer +export type RelayMoved = z.infer diff --git a/cloud/packages/relay-contract/src/host-proof-transcript.ts b/cloud/packages/relay-contract/src/host-proof-transcript.ts new file mode 100644 index 00000000000..853986f23f2 --- /dev/null +++ b/cloud/packages/relay-contract/src/host-proof-transcript.ts @@ -0,0 +1,103 @@ +const textEncoder = new TextEncoder() + +export const HOST_PROOF_TRANSCRIPT_DOMAIN = 'orca-relay-host-proof/v1' +export const HOST_CHALLENGE_PLAINTEXT_DOMAIN = 'orca-relay-host-challenge/v1' +export const HOST_CHALLENGE_BOX_ALGORITHM = 'Curve25519-XSalsa20-Poly1305' +export const HOST_PROOF_ALGORITHM = 'HMAC-SHA-256' + +export interface HostProofTranscriptInput { + relayOrigin: string + relayEphemeralPublicKey: Uint8Array + challengeNonce: Uint8Array + challengeId: string + issuedAt: number + expiresAt: number + userId: string + profileId: string + organizationId: string + relayHostId: string + hostPublicKey: Uint8Array + assignmentEpoch: number + previousGeneration?: number + resumeRequested: boolean +} + +function uint32(value: number): Uint8Array { + const bytes = new Uint8Array(4) + new DataView(bytes.buffer).setUint32(0, value, false) + return bytes +} + +function uint64(value: number): Uint8Array { + const bytes = new Uint8Array(8) + new DataView(bytes.buffer).setBigUint64(0, BigInt(value), false) + return bytes +} + +function concat(parts: readonly Uint8Array[]): Uint8Array { + const output = new Uint8Array(parts.reduce((total, part) => total + part.byteLength, 0)) + let offset = 0 + for (const part of parts) { + output.set(part, offset) + offset += part.byteLength + } + return output +} + +function field(name: string, value: Uint8Array): Uint8Array { + const encodedName = textEncoder.encode(name) + return concat([uint32(encodedName.byteLength), encodedName, uint32(value.byteLength), value]) +} + +function text(value: string): Uint8Array { + return textEncoder.encode(value) +} + +function requireByteLength(value: Uint8Array, expected: number, name: string): void { + if (value.byteLength !== expected) throw new Error(`${name} must be ${expected} bytes`) +} + +export function buildHostProofTranscript(input: HostProofTranscriptInput): Uint8Array { + requireByteLength(input.relayEphemeralPublicKey, 32, 'relayEphemeralPublicKey') + requireByteLength(input.challengeNonce, 24, 'challengeNonce') + requireByteLength(input.hostPublicKey, 32, 'hostPublicKey') + return concat([ + field('protocol', text(HOST_PROOF_TRANSCRIPT_DOMAIN)), + field('version', new Uint8Array([1])), + field('relayOrigin', text(input.relayOrigin)), + field('relayEphemeralPublicKey', input.relayEphemeralPublicKey), + field('challengeNonce', input.challengeNonce), + field('challengeId', text(input.challengeId)), + field('issuedAt', uint64(input.issuedAt)), + field('expiresAt', uint64(input.expiresAt)), + field('userId', text(input.userId)), + field('profileId', text(input.profileId)), + field('organizationId', text(input.organizationId)), + field('relayHostId', text(input.relayHostId)), + field('hostPublicKey', input.hostPublicKey), + field('assignmentEpoch', uint64(input.assignmentEpoch)), + field( + 'previousGeneration', + input.previousGeneration === undefined ? new Uint8Array() : uint64(input.previousGeneration) + ), + field('resumeRequested', new Uint8Array([input.resumeRequested ? 1 : 0])) + ]) +} + +export function buildHostChallengePlaintext( + transcript: Uint8Array, + challengeSecret: Uint8Array +): Uint8Array { + if (challengeSecret.byteLength !== 32) throw new Error('challengeSecret must be 32 bytes') + // Why: the encrypted random secret makes the public transcript insufficient to forge the ack. + return concat([ + text(`${HOST_CHALLENGE_PLAINTEXT_DOMAIN}\0`), + uint32(transcript.byteLength), + transcript, + challengeSecret + ]) +} + +export function buildHostProofMacInput(transcript: Uint8Array): Uint8Array { + return concat([text(`${HOST_PROOF_TRANSCRIPT_DOMAIN}\0ack\0`), transcript]) +} diff --git a/cloud/packages/relay-contract/src/index.ts b/cloud/packages/relay-contract/src/index.ts new file mode 100644 index 00000000000..2a7d7d0feda --- /dev/null +++ b/cloud/packages/relay-contract/src/index.ts @@ -0,0 +1,14 @@ +export * from './close-codes.js' +export * from './admission-budgets.js' +export * from './assignment-invariants.js' +export * from './control-messages.js' +export * from './control-continuity.js' +export * from './credential-messages.js' +export * from './director-messages.js' +export * from './host-proof-transcript.js' +export * from './persistence-invariants.js' +export * from './protocol-limits.js' +export * from './resume-confirmation-contract.js' +export * from './relay-regions.js' +export * from './splice-state-machine.js' +export * from './wire-scalars.js' diff --git a/cloud/packages/relay-contract/src/persistence-invariants.test.ts b/cloud/packages/relay-contract/src/persistence-invariants.test.ts new file mode 100644 index 00000000000..291fa7462da --- /dev/null +++ b/cloud/packages/relay-contract/src/persistence-invariants.test.ts @@ -0,0 +1,156 @@ +import { describe, expect, it } from 'vitest' +import { hasAdmissionCapacity, RELAY_ADMISSION_BUDGETS } from './admission-budgets.js' +import { + ASSIGNMENT_LIMITS, + AssignmentActivitySchema, + EvacuationCommitSchema, + hasAssignmentActivity, + mayNormallyReassign +} from './assignment-invariants.js' +import { + CONTROL_GENERATION_STATE, + controlLossDisposition, + mayStartNewRelayWork, + preservesRelayAuthIdentity +} from './control-continuity.js' +import { + CredentialInstallIdentitySchema, + decideResumeCommit, + INSTALL_TRANSACTION_CONTRACT, + INSTALL_STATUS, + InviteRecordSchema, + INVITE_STATE, + nextCredentialVersion, + PAIRING_RECOVERY_MATRIX, + transitionInvite, + TokenFamilySchema +} from './persistence-invariants.js' + +const HASH = 'abcdefghijklmnopqrstuvwxyzABCDEFGH012345678' + +describe('relay persistence invariants', () => { + it('admits pre-auth sockets only outside every reserved budget', () => { + expect(RELAY_ADMISSION_BUDGETS.maxProcessQueuedBytes).toBe(64 * 1024 * 1024) + expect(hasAdmissionCapacity({ totalRequests: 649, preAuthConnections: 44, sourcePreAuthConnections: 3 })).toBe(true) + expect(hasAdmissionCapacity({ totalRequests: 650, preAuthConnections: 0, sourcePreAuthConnections: 0 })).toBe(false) + expect(hasAdmissionCapacity({ totalRequests: 0, preAuthConnections: 45, sourcePreAuthConnections: 0 })).toBe(false) + expect(hasAdmissionCapacity({ totalRequests: 0, preAuthConnections: 0, sourcePreAuthConnections: 4 })).toBe(false) + expect(hasAdmissionCapacity({ + totalRequests: 2_899, + preAuthConnections: 0, + sourcePreAuthConnections: 0, + totalRequestCeiling: 2_900 + })).toBe(true) + expect(hasAdmissionCapacity({ + totalRequests: 2_900, + preAuthConnections: 0, + sourcePreAuthConnections: 0, + totalRequestCeiling: 2_900 + })).toBe(false) + }) + + it('represents persisted invite leases without an intermediate install status', () => { + expect(Object.values(INSTALL_STATUS)).toEqual(['not-found', 'committed']) + expect(INSTALL_TRANSACTION_CONTRACT.atomicEffects).toEqual([ + 'idempotency-result', 'token-family', 'affected-invites' + ]) + expect(nextCredentialVersion(undefined)).toBe(1) + expect(nextCredentialVersion(3)).toBe(4) + expect( + InviteRecordSchema.safeParse({ + relayDeviceId: 'device-1', tokenHash: HASH, state: INVITE_STATE.RESERVED, + attemptCount: 1, maxAttempts: 5, expiresAt: 100, reservationId: 'reservation-1', + reservationExpiresAt: 50 + }).success + ).toBe(true) + expect( + InviteRecordSchema.safeParse({ + relayDeviceId: 'device-1', tokenHash: HASH, state: INVITE_STATE.RESERVED, + attemptCount: 1, maxAttempts: 5, expiresAt: 100 + }).success + ).toBe(false) + expect( + CredentialInstallIdentitySchema.safeParse({ + userId: 'user-1', relayHostId: 'abcdefghijklmnop', relayDeviceId: 'device-1', reqId: 'req-1' + }).success + ).toBe(true) + }) + + it('leases, cools down, rolls back, consumes, invalidates, and cleans invites durably', () => { + const available = InviteRecordSchema.parse({ + relayDeviceId: 'device-1', tokenHash: HASH, state: INVITE_STATE.AVAILABLE, + attemptCount: 0, maxAttempts: 2, expiresAt: 1_000 + }) + const reserved = transitionInvite( + available, + { type: 'reserve', reservationId: 'reservation-1', reservationExpiresAt: 50 }, + 10 + ) + expect(reserved.state).toBe(INVITE_STATE.RESERVED) + expect(transitionInvite(reserved, { type: 'transaction-rollback' }, 20)).toBe(reserved) + const cooldown = transitionInvite(reserved, { type: 'lease-expired', cooldownUntil: 60 }, 50) + expect(cooldown.state).toBe(INVITE_STATE.COOLDOWN) + const second = transitionInvite( + cooldown, + { type: 'reserve', reservationId: 'reservation-2', reservationExpiresAt: 80 }, + 60 + ) + expect(transitionInvite(second, { type: 'attach-failed', cooldownUntil: 90 }, 70).state).toBe( + INVITE_STATE.INVALIDATED + ) + expect(transitionInvite(reserved, { type: 'provision-committed' }, 20).state).toBe( + INVITE_STATE.CONSUMED + ) + expect(transitionInvite(available, { type: 'direct-install-committed' }, 20).state).toBe( + INVITE_STATE.INVALIDATED + ) + expect(transitionInvite(available, { type: 'cleanup' }, 1_000).state).toBe(INVITE_STATE.EXPIRED) + expect(PAIRING_RECOVERY_MATRIX).toHaveLength(4) + }) + + it('makes commit-time current, grace, retired, expired, and revoked outcomes exact', () => { + const family = TokenFamilySchema.parse({ + currentVersion: 3, currentHash: HASH, currentExpiresAt: 200, + graceVersion: 2, graceHash: HASH, graceExpiresAt: 150 + }) + expect(decideResumeCommit(family, 3, 100)).toBe('renew-current') + expect(decideResumeCommit(family, 2, 100)).toBe('return-unchanged-grace') + expect(decideResumeCommit(family, 1, 100)).toBe('reject-retired') + expect(decideResumeCommit(family, 2, 151)).toBe('reject-expired') + expect(decideResumeCommit({ ...family, revokedAt: 90 }, 3, 100)).toBe('reject-revoked') + }) + + it('distinguishes same-process rebind, replacement fencing, and drain-only work', () => { + expect(controlLossDisposition({ matchingResumeSecret: true, competingGeneration: true })).toBe('rebind') + expect(controlLossDisposition({ matchingResumeSecret: false, competingGeneration: true })).toBe('fence-old') + expect(controlLossDisposition({ matchingResumeSecret: false, competingGeneration: false })).toBe('orphan-grace') + expect(mayStartNewRelayWork(CONTROL_GENERATION_STATE.DRAIN_ONLY, false)).toBe(false) + expect(mayStartNewRelayWork(CONTROL_GENERATION_STATE.ACTIVE, true)).toBe(false) + const identity = { sub: 'u', prof: 'p', org: 'o', relayHostId: 'abcdefghijklmnop' } + expect(preservesRelayAuthIdentity(identity, identity)).toBe(true) + expect(preservesRelayAuthIdentity(identity, { ...identity, org: 'other' })).toBe(false) + }) + + it('counts every durable activity class before normal reassignment', () => { + const record = AssignmentActivitySchema.parse({ + relayHostId: 'abcdefghijklmnop', cellId: 'cell-1', assignmentEpoch: 1, + leaseExpiresAt: 100, lastActivityAt: 100, reservedControls: 0, reservedSplices: 0, + reservedInvites: 0, pendingInstalls: 0, pendingConfirmations: 0, migrationLeases: 1 + }) + expect(hasAssignmentActivity(record)).toBe(true) + expect(mayNormallyReassign(record, 100 + ASSIGNMENT_LIMITS.dormantTtlMs)).toBe(false) + expect(mayNormallyReassign({ ...record, migrationLeases: 0 }, 100 + ASSIGNMENT_LIMITS.dormantTtlMs)).toBe(true) + expect( + EvacuationCommitSchema.safeParse({ + relayHostId: 'abcdefghijklmnop', sourceCellId: 'cell-1', targetCellId: 'cell-2', + previousEpoch: 1, assignmentEpoch: 2, targetCapacityReserved: true + }).success + ).toBe(true) + expect( + EvacuationCommitSchema.safeParse({ + relayHostId: 'abcdefghijklmnop', sourceCellId: 'cell-1', targetCellId: 'cell-2', + previousEpoch: 1, assignmentEpoch: 3, targetCapacityReserved: true + }).success + ).toBe(false) + }) +}) diff --git a/cloud/packages/relay-contract/src/persistence-invariants.ts b/cloud/packages/relay-contract/src/persistence-invariants.ts new file mode 100644 index 00000000000..0f07250d008 --- /dev/null +++ b/cloud/packages/relay-contract/src/persistence-invariants.ts @@ -0,0 +1,172 @@ +import { z } from 'zod' +import { Base64Url32ByteSchema, EpochMsSchema, OpaqueIdSchema, RelayHostIdSchema } from './wire-scalars.js' + +export const INVITE_STATE = { + AVAILABLE: 'available', + RESERVED: 'reserved', + COOLDOWN: 'cooldown', + CONSUMED: 'consumed', + EXPIRED: 'expired', + INVALIDATED: 'invalidated' +} as const + +export const InviteRecordSchema = z + .object({ + relayDeviceId: OpaqueIdSchema, + tokenHash: Base64Url32ByteSchema, + state: z.nativeEnum(INVITE_STATE), + attemptCount: z.number().int().nonnegative(), + maxAttempts: z.number().int().positive(), + expiresAt: EpochMsSchema, + reservationId: OpaqueIdSchema.optional(), + reservationExpiresAt: EpochMsSchema.optional(), + cooldownUntil: EpochMsSchema.optional() + }) + .strict() + .superRefine((invite, context) => { + if ( + invite.state === INVITE_STATE.RESERVED && + (!invite.reservationId || !invite.reservationExpiresAt) + ) { + context.addIssue({ code: z.ZodIssueCode.custom, message: 'reserved invite requires a lease' }) + } + if (invite.state === INVITE_STATE.COOLDOWN && invite.cooldownUntil === undefined) { + context.addIssue({ + code: z.ZodIssueCode.custom, + message: 'cooldown invite requires cooldownUntil' + }) + } + }) + +export type InviteRecord = z.infer + +export type InviteEvent = + | { type: 'reserve'; reservationId: string; reservationExpiresAt: number } + | { type: 'attach-failed'; cooldownUntil: number } + | { type: 'lease-expired'; cooldownUntil: number } + | { type: 'provision-committed' } + | { type: 'direct-install-committed' } + | { type: 'transaction-rollback' } + | { type: 'cleanup' } + +function withoutLease(invite: InviteRecord): InviteRecord { + const { reservationId: _reservationId, reservationExpiresAt: _reservationExpiresAt, ...rest } = invite + return rest +} + +export function transitionInvite(invite: InviteRecord, event: InviteEvent, now: number): InviteRecord { + if (event.type === 'transaction-rollback') return invite + if (event.type === 'direct-install-committed') { + return { ...withoutLease(invite), state: INVITE_STATE.INVALIDATED } + } + if (event.type === 'provision-committed') { + return { ...withoutLease(invite), state: INVITE_STATE.CONSUMED } + } + if (event.type === 'cleanup') { + if (now >= invite.expiresAt) return { ...withoutLease(invite), state: INVITE_STATE.EXPIRED } + if ( + invite.state === INVITE_STATE.RESERVED && + invite.reservationExpiresAt !== undefined && + now >= invite.reservationExpiresAt + ) { + const state = + invite.attemptCount >= invite.maxAttempts ? INVITE_STATE.INVALIDATED : INVITE_STATE.COOLDOWN + return { ...withoutLease(invite), state, cooldownUntil: now } + } + return invite + } + if (event.type === 'reserve') { + const available = + invite.state === INVITE_STATE.AVAILABLE || + (invite.state === INVITE_STATE.COOLDOWN && (invite.cooldownUntil ?? 0) <= now) + if (!available || now >= invite.expiresAt || invite.attemptCount >= invite.maxAttempts) return invite + return { + ...invite, + state: INVITE_STATE.RESERVED, + attemptCount: invite.attemptCount + 1, + reservationId: event.reservationId, + reservationExpiresAt: event.reservationExpiresAt, + cooldownUntil: undefined + } + } + if (invite.state !== INVITE_STATE.RESERVED) return invite + const state = + invite.attemptCount >= invite.maxAttempts ? INVITE_STATE.INVALIDATED : INVITE_STATE.COOLDOWN + return { ...withoutLease(invite), state, cooldownUntil: event.cooldownUntil } +} + +export const PAIRING_RECOVERY_MATRIX = [ + 'direct-commit-ack', + 'direct-commit-response-lost', + 'direct-no-commit-invite-fallback', + 'late-direct-versus-invite-global-key' +] as const + +export const CredentialInstallIdentitySchema = z + .object({ + userId: OpaqueIdSchema, + relayHostId: RelayHostIdSchema, + relayDeviceId: OpaqueIdSchema, + reqId: OpaqueIdSchema + }) + .strict() + +export const INSTALL_STATUS = { + NOT_FOUND: 'not-found', + COMMITTED: 'committed' +} as const + +export const INSTALL_TRANSACTION_CONTRACT = { + idempotencyKeyFields: ['userId', 'relayHostId', 'relayDeviceId', 'reqId'], + transactionDeadlineMs: 10 * 1000, + lockAttemptTimeoutMs: 2 * 1000, + maxDeadlockRetries: 3, + atomicEffects: ['idempotency-result', 'token-family', 'affected-invites'] +} as const + +export function nextCredentialVersion(currentVersion: number | undefined): number { + return currentVersion === undefined ? 1 : currentVersion + 1 +} + +export const TokenFamilySchema = z + .object({ + currentVersion: z.number().int().positive(), + currentHash: Base64Url32ByteSchema, + currentExpiresAt: EpochMsSchema, + graceVersion: z.number().int().positive().optional(), + graceHash: Base64Url32ByteSchema.optional(), + graceExpiresAt: EpochMsSchema.optional(), + revokedAt: EpochMsSchema.optional() + }) + .strict() + .superRefine((family, context) => { + const graceFields = [family.graceVersion, family.graceHash, family.graceExpiresAt] + if (graceFields.some((value) => value !== undefined) && graceFields.some((value) => value === undefined)) { + context.addIssue({ code: z.ZodIssueCode.custom, message: 'grace fields must be all present or absent' }) + } + if (family.graceVersion !== undefined && family.graceVersion >= family.currentVersion) { + context.addIssue({ code: z.ZodIssueCode.custom, message: 'grace version must precede current' }) + } + }) + +export type ResumeCommitDecision = + | 'renew-current' + | 'return-unchanged-grace' + | 'reject-retired' + | 'reject-expired' + | 'reject-revoked' + +export function decideResumeCommit( + family: z.infer, + acceptedVersion: number, + now: number +): ResumeCommitDecision { + if (family.revokedAt !== undefined && family.revokedAt <= now) return 'reject-revoked' + if (acceptedVersion === family.currentVersion) { + return family.currentExpiresAt > now ? 'renew-current' : 'reject-expired' + } + if (acceptedVersion === family.graceVersion) { + return (family.graceExpiresAt ?? 0) > now ? 'return-unchanged-grace' : 'reject-expired' + } + return 'reject-retired' +} diff --git a/cloud/packages/relay-contract/src/protocol-limits.ts b/cloud/packages/relay-contract/src/protocol-limits.ts new file mode 100644 index 00000000000..6257f9cecb7 --- /dev/null +++ b/cloud/packages/relay-contract/src/protocol-limits.ts @@ -0,0 +1,23 @@ +export const RELAY_PROTOCOL_LIMITS = { + firstFrameDeadlineMs: 2_000, + maxHttpBodyBytes: 4 * 1024, + // Why: the splice is an opaque E2EE stream; the desktop's worktree catalog + // response already exceeds 1MiB on large workspaces (~775KiB at 415 + // worktrees, growing), and an oversized frame kills the session on every + // reconnect. 8MiB buys years of headroom; catalog pagination is the + // long-term fix on the desktop side. + maxFrameBytes: 8 * 1024 * 1024, + maxConnectionsPerHost: 8, + idleTimeoutMs: 10 * 60 * 1000, + inviteTtlMs: 10 * 60 * 1000, + inviteMaxAttempts: 5, + inviteReservationLeaseMs: 15 * 1000, + inviteAttemptCooldownMs: 2 * 1000, + hostAttachDeadlineMs: 10 * 1000, + resumeConfirmationDeadlineMs: 30 * 1000, + resumeTtlMs: 30 * 24 * 60 * 60 * 1000, + relayTokenTtlMs: 5 * 60 * 1000, + expiredAuthExistingSpliceGraceMs: 60 * 1000, + controlPingIntervalMs: 15 * 1000, + controlSilenceTimeoutMs: 75 * 1000 +} as const diff --git a/cloud/packages/relay-contract/src/relay-regions.ts b/cloud/packages/relay-contract/src/relay-regions.ts new file mode 100644 index 00000000000..38ac36cd738 --- /dev/null +++ b/cloud/packages/relay-contract/src/relay-regions.ts @@ -0,0 +1,62 @@ +import { z } from 'zod' + +export const RELAY_REGIONS = ['us-central1', 'asia-east2'] as const + +export const RelayRegionSchema = z.enum(RELAY_REGIONS) + +export type RelayRegion = z.infer + +export const RELAY_DEFAULT_REGION: RelayRegion = 'us-central1' + +const RelayProbeOriginSchema = z.string().url().max(2_048).refine(isCanonicalHttpsOrigin) + +export const RelayRegionCatalogResponseSchema = z + .object({ + v: z.literal(1), + regions: z + .array( + z + .object({ + region: RelayRegionSchema, + probeOrigins: z.array(RelayProbeOriginSchema).min(1).max(2) + }) + .strict() + ) + .max(RELAY_REGIONS.length) + }) + .strict() + .superRefine((catalog, context) => { + const regions = new Set() + const origins = new Set() + for (const [regionIndex, entry] of catalog.regions.entries()) { + if (regions.has(entry.region)) { + context.addIssue({ + code: 'custom', + message: 'duplicate relay region', + path: ['regions', regionIndex, 'region'] + }) + } + regions.add(entry.region) + for (const [originIndex, origin] of entry.probeOrigins.entries()) { + if (origins.has(origin)) { + context.addIssue({ + code: 'custom', + message: 'duplicate relay probe origin', + path: ['regions', regionIndex, 'probeOrigins', originIndex] + }) + } + origins.add(origin) + } + } + }) + +export type RelayRegionCatalogResponse = z.infer + +function isCanonicalHttpsOrigin(value: string): boolean { + try { + const url = new URL(value) + return url.protocol === 'https:' && url.origin === value + } catch { + return false + } +} diff --git a/cloud/packages/relay-contract/src/resume-confirmation-contract.ts b/cloud/packages/relay-contract/src/resume-confirmation-contract.ts new file mode 100644 index 00000000000..0f429fc9e81 --- /dev/null +++ b/cloud/packages/relay-contract/src/resume-confirmation-contract.ts @@ -0,0 +1,23 @@ +import { z } from 'zod' +import { EpochMsSchema, GenerationSchema, OpaqueIdSchema } from './wire-scalars.js' + +export const ConfirmableResumeTupleSchema = z + .object({ + basisConnId: OpaqueIdSchema, + owningControlGeneration: GenerationSchema, + relayDeviceId: OpaqueIdSchema, + acceptedCredentialVersion: z.number().int().positive(), + acceptedAs: z.enum(['current', 'grace']), + confirmDeadline: EpochMsSchema + }) + .strict() + +export const RESUME_CONFIRMATION_COMMIT_OUTCOME = { + RENEW_CURRENT: 'renew-current', + RETURN_UNCHANGED_GRACE: 'return-unchanged-grace', + REJECT_RETIRED: 'reject-retired', + REJECT_EXPIRED: 'reject-expired', + REJECT_REVOKED: 'reject-revoked' +} as const + +export type ConfirmableResumeTuple = z.infer diff --git a/cloud/packages/relay-contract/src/splice-state-machine.ts b/cloud/packages/relay-contract/src/splice-state-machine.ts new file mode 100644 index 00000000000..3944df74584 --- /dev/null +++ b/cloud/packages/relay-contract/src/splice-state-machine.ts @@ -0,0 +1,34 @@ +export const SPLICE_STATE = { + PRE_AUTH_ADMITTED: 'pre-auth-admitted', + CREDENTIAL_LEASE_RESERVED: 'credential-lease-reserved', + HOST_NOTIFIED: 'host-notified', + ATTACH_PENDING: 'attach-pending', + HOST_ATTACHED: 'host-attached', + CLIENT_ACKNOWLEDGED: 'client-acknowledged', + SPLICED: 'spliced', + E2EE_CONFIRMABLE: 'e2ee-confirmable', + TEARDOWN: 'teardown' +} as const + +export type SpliceState = (typeof SPLICE_STATE)[keyof typeof SPLICE_STATE] + +export const SPLICE_FORWARD_TRANSITIONS: Readonly> = { + [SPLICE_STATE.PRE_AUTH_ADMITTED]: [SPLICE_STATE.CREDENTIAL_LEASE_RESERVED, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.CREDENTIAL_LEASE_RESERVED]: [SPLICE_STATE.HOST_NOTIFIED, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.HOST_NOTIFIED]: [SPLICE_STATE.ATTACH_PENDING, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.ATTACH_PENDING]: [SPLICE_STATE.HOST_ATTACHED, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.HOST_ATTACHED]: [SPLICE_STATE.CLIENT_ACKNOWLEDGED, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.CLIENT_ACKNOWLEDGED]: [SPLICE_STATE.SPLICED, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.SPLICED]: [SPLICE_STATE.E2EE_CONFIRMABLE, SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.E2EE_CONFIRMABLE]: [SPLICE_STATE.TEARDOWN], + [SPLICE_STATE.TEARDOWN]: [] +} + +export function canAdvanceSplice(from: SpliceState, to: SpliceState): boolean { + return SPLICE_FORWARD_TRANSITIONS[from].includes(to) +} + +export function mayAcknowledgeClient(state: SpliceState, forwardingHandlersInstalled: boolean): boolean { + // Why: success before both forwarding handlers exist can strand a client on a fake splice. + return state === SPLICE_STATE.HOST_ATTACHED && forwardingHandlersInstalled +} diff --git a/cloud/packages/relay-contract/src/wire-scalars.ts b/cloud/packages/relay-contract/src/wire-scalars.ts new file mode 100644 index 00000000000..27dd3a8b30f --- /dev/null +++ b/cloud/packages/relay-contract/src/wire-scalars.ts @@ -0,0 +1,20 @@ +import { z } from 'zod' + +export const Base64Url32ByteSchema = z.string().regex(/^[A-Za-z0-9_-]{43}$/) +export const Base64Url24ByteSchema = z.string().regex(/^[A-Za-z0-9_-]{32}$/) +export const Base6432ByteSchema = z.string().regex(/^(?:[A-Za-z0-9+/]{4}){10}[A-Za-z0-9+/]{3}=$/) +export const Base64Raw24ByteSchema = z.string().regex(/^(?:[A-Za-z0-9+/]{4}){8}$/) +export const RelayHostIdSchema = z.string().regex(/^[A-Za-z0-9_-]{16}$/) +export const OpaqueIdSchema = z.string().min(1).max(128) +export const EpochMsSchema = z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER) +export const GenerationSchema = z.number().int().nonnegative().max(Number.MAX_SAFE_INTEGER) +export const PositiveDurationMsSchema = z.number().int().positive().max(24 * 60 * 60 * 1000) + +export const CanonicalHttpsOriginSchema = z.string().max(2048).refine((value) => { + try { + const url = new URL(value) + return url.protocol === 'https:' && url.origin === value && url.pathname === '/' + } catch { + return false + } +}, 'must be a canonical HTTPS origin') diff --git a/cloud/packages/relay-contract/tsconfig.build.json b/cloud/packages/relay-contract/tsconfig.build.json new file mode 100644 index 00000000000..94c84b60803 --- /dev/null +++ b/cloud/packages/relay-contract/tsconfig.build.json @@ -0,0 +1,11 @@ +{ + "extends": "./tsconfig.json", + "compilerOptions": { + "declaration": true, + "emitDeclarationOnly": false, + "noEmit": false, + "outDir": "dist", + "rootDir": "src" + }, + "exclude": ["src/**/*.test.ts"] +} diff --git a/cloud/packages/relay-contract/tsconfig.json b/cloud/packages/relay-contract/tsconfig.json new file mode 100644 index 00000000000..a552e34dbe9 --- /dev/null +++ b/cloud/packages/relay-contract/tsconfig.json @@ -0,0 +1,5 @@ +{ + "extends": "../../tsconfig.base.json", + "compilerOptions": { "noEmit": true }, + "include": ["src/**/*.ts"] +} diff --git a/cloud/pnpm-lock.yaml b/cloud/pnpm-lock.yaml new file mode 100644 index 00000000000..27fdd29071a --- /dev/null +++ b/cloud/pnpm-lock.yaml @@ -0,0 +1,1336 @@ +lockfileVersion: '9.0' + +settings: + autoInstallPeers: true + excludeLinksFromLockfile: false + +importers: + + .: + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + tsx: + specifier: ^4.21.0 + version: 4.22.4 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + + apps/relay: + dependencies: + '@hono/node-server': + specifier: ^1.19.14 + version: 1.19.14(hono@4.12.27) + '@orca-cloud/relay-contract': + specifier: workspace:* + version: link:../../packages/relay-contract + hono: + specifier: ^4.12.27 + version: 4.12.27 + jose: + specifier: ^6.1.3 + version: 6.2.3 + pg: + specifier: ^8.22.0 + version: 8.22.0 + tweetnacl: + specifier: ^1.0.3 + version: 1.0.3 + ws: + specifier: ^8.18.3 + version: 8.21.0 + zod: + specifier: ^3.25.76 + version: 3.25.76 + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + '@types/pg': + specifier: ^8.20.0 + version: 8.20.0 + '@types/ws': + specifier: ^8.18.1 + version: 8.18.1 + tsx: + specifier: ^4.21.0 + version: 4.22.4 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + + apps/relay-fence-broker: + dependencies: + '@hono/node-server': + specifier: ^1.19.14 + version: 1.19.14(hono@4.12.27) + hono: + specifier: ^4.12.27 + version: 4.12.27 + zod: + specifier: ^3.25.76 + version: 3.25.76 + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + tsx: + specifier: ^4.21.0 + version: 4.22.4 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + + apps/relay-ops: + dependencies: + '@hono/node-server': + specifier: ^1.19.14 + version: 1.19.14(hono@4.12.27) + hono: + specifier: ^4.12.27 + version: 4.12.27 + zod: + specifier: ^3.25.76 + version: 3.25.76 + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + tsx: + specifier: ^4.21.0 + version: 4.22.4 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + + packages/relay-contract: + dependencies: + zod: + specifier: ^3.25.76 + version: 3.25.76 + devDependencies: + '@types/node': + specifier: ^24.10.0 + version: 24.13.2 + typescript: + specifier: ^5.9.3 + version: 5.9.3 + vitest: + specifier: ^4.0.8 + version: 4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + +packages: + + '@emnapi/core@1.10.0': + resolution: {integrity: sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw==} + + '@emnapi/runtime@1.10.0': + resolution: {integrity: sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA==} + + '@emnapi/wasi-threads@1.2.1': + resolution: {integrity: sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w==} + + '@esbuild/aix-ppc64@0.28.1': + resolution: {integrity: sha512-Svl7tq8k/08+p6CXPpRjQ1fKX+1odH/BQbb48fV6fj3CWHhsoIOoY87w1oHXm0qEpkIK3ZfVgp0hed3XBXzXMQ==} + engines: {node: '>=18'} + cpu: [ppc64] + os: [aix] + + '@esbuild/android-arm64@0.28.1': + resolution: {integrity: sha512-34EGEbCIAgosYz6goLcopX6Mo7NyGv9tfwEM2/7Ce2VcVRk568iSvniGWcUXIy7wEDR1wzolcxcriFVrWYcwBg==} + engines: {node: '>=18'} + cpu: [arm64] + os: [android] + + '@esbuild/android-arm@0.28.1': + resolution: {integrity: sha512-0k2F129Xdio1TdJfzJ8sy1Q47vUD2NnwdhiAf7drUN1EBTfPf4hsFCtmMgu/6m8JSzsBrlmVjudMBQqOfG8usQ==} + engines: {node: '>=18'} + cpu: [arm] + os: [android] + + '@esbuild/android-x64@0.28.1': + resolution: {integrity: sha512-dbwY7ltSMDWsRatcRpCnES4F+im88OCUgGZjy52shC7GqHRE/cYlxNbB4Z4UpJswpcc4Qxd2oE/ufM0p61IKng==} + engines: {node: '>=18'} + cpu: [x64] + os: [android] + + '@esbuild/darwin-arm64@0.28.1': + resolution: {integrity: sha512-TZbWkQY7kvTAXbXUT7uVACR5cMHsDiSz9z7ZKAX/RTq/WJEk3QyRr0wZpNhBDX+/0CtdqUIJlOiodQcta6tY3Q==} + engines: {node: '>=18'} + cpu: [arm64] + os: [darwin] + + '@esbuild/darwin-x64@0.28.1': + resolution: {integrity: sha512-zfdzgK9ACBNZLI/CyHTOx81SyNbM6YXn7rxSgX97VjyiPl9W1i4Ka4fgKECEoFCKGpvBj5qArWIGgQjOwkgskQ==} + engines: {node: '>=18'} + cpu: [x64] + os: [darwin] + + '@esbuild/freebsd-arm64@0.28.1': + resolution: {integrity: sha512-wG2EA8ENdEI0qhkSZMjfqrdY+ziCYCPMmtZjjIwOmXFjmyzEHn+UUxk5of+SYsjtfs3VpnlC7QLzSI5hY/rOAw==} + engines: {node: '>=18'} + cpu: [arm64] + os: [freebsd] + + '@esbuild/freebsd-x64@0.28.1': + resolution: {integrity: sha512-i7dZ9vQgnvSCzi/rYCXNgtF/U+eKZNJBzu3eTQbRgHnM7tNSizLOkRFAl3qzVc/Op/u5YkHHa4pf/3DOYHthLQ==} + engines: {node: '>=18'} + cpu: [x64] + os: [freebsd] + + '@esbuild/linux-arm64@0.28.1': + resolution: {integrity: sha512-yHs+0uc8+nvEAfAfxrWQKK5peSNzBc4PegcMO0EJ2hT71uA7vB8Ihg2e77R2P7SG5uYjPbHlLLmve4LLLRCf0g==} + engines: {node: '>=18'} + cpu: [arm64] + os: [linux] + + '@esbuild/linux-arm@0.28.1': + resolution: {integrity: sha512-qVXBOHQS+d5Y722GwJzJUtOLlX7km3CraOaGormF1pDtPd2C/l1SHRPgjLunLGe51Sh5YYWKMFDyV4SxgMQYTQ==} + engines: {node: '>=18'} + cpu: [arm] + os: [linux] + + '@esbuild/linux-ia32@0.28.1': + resolution: {integrity: sha512-d1z4ZuP0ajrfz/FhGT4vv278rX8KnPPJx8i5+AtK7TYbx9Le9F1hyzurZpkEyjkGa9dUGhQow4C1NmeGvqxN2w==} + engines: {node: '>=18'} + cpu: [ia32] + os: [linux] + + '@esbuild/linux-loong64@0.28.1': + resolution: {integrity: sha512-M5sRjUVZrkm1OAPR3dlOYzNmN+loZKGVi1VUQGrwuqLcbR6qeAz+famMhjASeH3YVKvZz+zT1jlh/keC3Rj/lg==} + engines: {node: '>=18'} + cpu: [loong64] + os: [linux] + + '@esbuild/linux-mips64el@0.28.1': + resolution: {integrity: sha512-mRObBZeHh2OxcBFPWE/FjylkRgZdYuiTR3vaTozquCGOH14iP9oN4x4Ge81CoIDYQrXmIxpFumJBu5MtZpnQJQ==} + engines: {node: '>=18'} + cpu: [mips64el] + os: [linux] + + '@esbuild/linux-ppc64@0.28.1': + resolution: {integrity: sha512-slScBsMAb3GFDcdrCgLwZtPYRoH2H/youv10QiZyRjmsP48fznoveWytSgCI/R0ZcUgpc0ZhIUEx6LHts8yrfQ==} + engines: {node: '>=18'} + cpu: [ppc64] + os: [linux] + + '@esbuild/linux-riscv64@0.28.1': + resolution: {integrity: sha512-kw0owk1o0GFETUJyW0jc0G4Yzs0BHZn0JDZ8JRT088vjJYX777BAs1fDGxAC+q831qOs2DTC96mNsG2opdfyyQ==} + engines: {node: '>=18'} + cpu: [riscv64] + os: [linux] + + '@esbuild/linux-s390x@0.28.1': + resolution: {integrity: sha512-/lAIjX8aYFRByhh6L5rYtPEDRqa9de/4V/juOXcta5frjvzXO4/sqEtyytse0g3zZFuWu5cDN0MkLz2qRDD2Ag==} + engines: {node: '>=18'} + cpu: [s390x] + os: [linux] + + '@esbuild/linux-x64@0.28.1': + resolution: {integrity: sha512-u/anNYF2mmVOEDwLtnQ1wOr3EZ9sTNGLWrsYGYwHWzGA3Si84IOkHXlbWTD1NB+9/1lcnweYKO54uhxZydNzfA==} + engines: {node: '>=18'} + cpu: [x64] + os: [linux] + + '@esbuild/netbsd-arm64@0.28.1': + resolution: {integrity: sha512-oks0DYbLwWMmaakTsCb+zL4E+aHRVLom9IJZOAthMQEPiQmydXHkziYEsGYRx0uNV/IjEKGAV941JzH02pflqw==} + engines: {node: '>=18'} + cpu: [arm64] + os: [netbsd] + + '@esbuild/netbsd-x64@0.28.1': + resolution: {integrity: sha512-aeL6lAnN89Hz43Mlh1G8ARasbuoYvSITDEx0tHh5b7jJnHcssqgjy9Yx430GDpmCa6OyrKoS0aNRjKundRizGg==} + engines: {node: '>=18'} + cpu: [x64] + os: [netbsd] + + '@esbuild/openbsd-arm64@0.28.1': + resolution: {integrity: sha512-MEFJe5C3R8pwXdZ5Y21oo6m7ePiS0d9pWucn99O/wvyJZChoIQKrQDxKrGeW8F5+T0okTHesAmDeiHDTIq0V/Q==} + engines: {node: '>=18'} + cpu: [arm64] + os: [openbsd] + + '@esbuild/openbsd-x64@0.28.1': + resolution: {integrity: sha512-i/ZLIOafE0Z8cI/XANJAixoJL/uRAoS2xOA3rb0xN+KK0K177cMAsQYkzHtBrtMXAKuAc7HGgcWiZ/sRC1Nxgw==} + engines: {node: '>=18'} + cpu: [x64] + os: [openbsd] + + '@esbuild/openharmony-arm64@0.28.1': + resolution: {integrity: sha512-ge+Z7EXFNt2BO1oAMsVpiQ8EwndV9i1xXerAeTIK7AtPs3bKFXQM7nlRxDSIUIMeueR1CNXxqztLzdNeReKBJg==} + engines: {node: '>=18'} + cpu: [arm64] + os: [openharmony] + + '@esbuild/sunos-x64@0.28.1': + resolution: {integrity: sha512-BEjgtECkL3vY+SaSQ6nzVfiALUeFxpawyp8Jmf5PtYhf1Ug40N1h/hxlhts+f1FvSvarEigdxS3BlSMI2PJLcQ==} + engines: {node: '>=18'} + cpu: [x64] + os: [sunos] + + '@esbuild/win32-arm64@0.28.1': + resolution: {integrity: sha512-lCv9eK/H6ZJWbE7bh2nw54CZ9M2nupBxJcTsdk/QQnWkdSjKGuxmmH8/GWrlT1eMmZfn4dGcCjRte397WqfQXA==} + engines: {node: '>=18'} + cpu: [arm64] + os: [win32] + + '@esbuild/win32-ia32@0.28.1': + resolution: {integrity: sha512-zvb/mB2bSCoJOpoCBgYKKpX6YM6mJBlBUVUtVj41DlZJVEB6/0CKlRYxP5wWl1C1ILiCoAU5wZZ4q1P3qeS6Eg==} + engines: {node: '>=18'} + cpu: [ia32] + os: [win32] + + '@esbuild/win32-x64@0.28.1': + resolution: {integrity: sha512-bm4Mowrv+GXMlpWX++EcXw/iLyd1o3+bJkC2DkWXYVvgZCqD/bSj9ctZeAMC3cIxgjRVR2Dufaiu4YPxr5gW1A==} + engines: {node: '>=18'} + cpu: [x64] + os: [win32] + + '@hono/node-server@1.19.14': + resolution: {integrity: sha512-GwtvgtXxnWsucXvbQXkRgqksiH2Qed37H9xHZocE5sA3N8O8O8/8FA3uclQXxXVzc9XBZuEOMK7+r02FmSpHtw==} + engines: {node: '>=18.14.1'} + peerDependencies: + hono: ^4 + + '@jridgewell/sourcemap-codec@1.5.5': + resolution: {integrity: sha512-cYQ9310grqxueWbl+WuIUIaiUaDcj7WOq5fVhEljNVgRfOUhY9fy2zTvfoqWsnebh8Sl70VScFbICvJnLKB0Og==} + + '@napi-rs/wasm-runtime@1.1.5': + resolution: {integrity: sha512-AWPoBRJ9tsnVhor4sjO7rkni+7p+2IAEFj6cx06UgP10jkQHqay/36uRV/bFkgrh18D9vb4cr8Q0Pthskgzy+Q==} + peerDependencies: + '@emnapi/core': ^1.7.1 + '@emnapi/runtime': ^1.7.1 + + '@oxc-project/types@0.133.0': + resolution: {integrity: sha512-KzkdCd6Uxqnf6l3HOw1xfatAlUURA0g14cvBYFyJ5SaNOQbOUvBr9PKArcPcrNIeRsBdgcUzOGrhKveVpvOIGA==} + + '@rolldown/binding-android-arm64@1.0.3': + resolution: {integrity: sha512-454rs7jHngixp/NMxd5srYD57OnzSlZ/eFTETjORQHLwJG1lRtmNOJcBerZlfu4GjKqeq8aCCIQrMdHyhI51Hw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [android] + + '@rolldown/binding-darwin-arm64@1.0.3': + resolution: {integrity: sha512-PcAhP+ynjURNyy8SKGl5DQP94aGuB/7JrXJb/t7P+hanXvQVMWzUvRRhBAcg/lNRadBhoUPqSoP4xw5tR/KBEA==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [darwin] + + '@rolldown/binding-darwin-x64@1.0.3': + resolution: {integrity: sha512-9YpfeUvSE2RS7wysJ81uOZkXJz7f7Q55H2Gvp3VEw/EsahqDtrphrZ0EwDLK5vvKOzaCrBsjF8JmnMLcUt78Gg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [darwin] + + '@rolldown/binding-freebsd-x64@1.0.3': + resolution: {integrity: sha512-yB1IlAsSNHncV6SCTL27/MVGR5htvQsoGxIv5KMGXALp+Ll1wYsn+x98M9MW7qa+NdSbvrrY7ANI4wLJ0n1e6g==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [freebsd] + + '@rolldown/binding-linux-arm-gnueabihf@1.0.3': + resolution: {integrity: sha512-Yi30IVAAfLUCy2MseFjbB1jAMDl1VMCAas5StnYp8da9+CKvMd2H2cbEjWcw5NPaPqzvYkVIaF1nNUG+b7u/sw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm] + os: [linux] + + '@rolldown/binding-linux-arm64-gnu@1.0.3': + resolution: {integrity: sha512-jsO7R8To+AdlYgUmN5sHSCZbfhtMBkO0WUx8iORQnPcMMdgr7qM2DQmMwgabs3GhNztdmoKkMKQFHD6DTMCIQw==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [linux] + + '@rolldown/binding-linux-arm64-musl@1.0.3': + resolution: {integrity: sha512-VWkUHwWriDciit80wleYwKILoR/KMvxh/IdwS/paX+ZgpuRpCrKLUdadJbc0NpBEiyhpYawsJ73j9aCvOH+f7Q==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [linux] + + '@rolldown/binding-linux-ppc64-gnu@1.0.3': + resolution: {integrity: sha512-5f1laC0SlIR0yDbFCd8acUhvJIag6N3zC5P7oUPN6wX0aOma+uKJ0wBDH5aq7I1PVI2ttTlhJwzwRIBnLiSGEg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [ppc64] + os: [linux] + + '@rolldown/binding-linux-s390x-gnu@1.0.3': + resolution: {integrity: sha512-Iq4ko0r4XsgbrF/LunNgHtAGLRRVE2kXonAXQ/MV0mC6jQpMOhW1SvtZja2EhC/kd05++bP78dsqBeIQyYJ6Yg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [s390x] + os: [linux] + + '@rolldown/binding-linux-x64-gnu@1.0.3': + resolution: {integrity: sha512-B8m6tD5+/N5FeNQFbKlLA/2yVq9ycQP1SeedyEYYKWBNR3ZQbkvIUcNnDNM03lO1l5F2roiiFJGgvoLLyZXtSg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [linux] + + '@rolldown/binding-linux-x64-musl@1.0.3': + resolution: {integrity: sha512-pSdpdUJHkuCxun9LE7jvgUB9qsRgaiyNNCX7m/AvHTcq67AiT/Yhoxvw5zPfhrM8k/BfP8ce/hMOpthKDpEUow==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [linux] + + '@rolldown/binding-openharmony-arm64@1.0.3': + resolution: {integrity: sha512-OXXS3RKJgX2uLwM+gYyuH5omcH8fL1LJs96pZGgtetVCahON57+d4SJHzTgZiOjxgGkSnpXpOsWuPDGAKAigEg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [openharmony] + + '@rolldown/binding-wasm32-wasi@1.0.3': + resolution: {integrity: sha512-JTtb8BWFynicNSoPrehsCzBtOKjZ6jhMiPFEmOiuXg1Fl8dn2KHQob+GuPSGR0dryQa1PQJbzjF3dqO/whhjLg==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [wasm32] + + '@rolldown/binding-win32-arm64-msvc@1.0.3': + resolution: {integrity: sha512-gEdFFEN70A/jxb2svrWsN3aDL7OUtmvlOy+6fa2jxG8K0wQ1ZbdeLGnidov6Yu5/733dI5ySfzFlQ/cb0bSz1g==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [arm64] + os: [win32] + + '@rolldown/binding-win32-x64-msvc@1.0.3': + resolution: {integrity: sha512-eXB7CHuaQdqmJcc3koCNtNPmT/bj2gc999kUFgBxG8Ac0NdgXc4rkCHhqrgrhN3zddvvvrgzj1e90SuSfmyIXA==} + engines: {node: ^20.19.0 || >=22.12.0} + cpu: [x64] + os: [win32] + + '@rolldown/pluginutils@1.0.1': + resolution: {integrity: sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw==} + + '@standard-schema/spec@1.1.0': + resolution: {integrity: sha512-l2aFy5jALhniG5HgqrD6jXLi/rUWrKvqN/qJx6yoJsgKhblVd+iqqU4RCXavm/jPityDo5TCvKMnpjKnOriy0w==} + + '@tybys/wasm-util@0.10.2': + resolution: {integrity: sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg==} + + '@types/chai@5.2.3': + resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} + + '@types/deep-eql@4.0.2': + resolution: {integrity: sha512-c9h9dVVMigMPc4bwTvC5dxqtqJZwQPePsWjPlpSOnojbor6pGqdk541lfA7AqFQr5pB1BRdq0juY9db81BwyFw==} + + '@types/estree@1.0.9': + resolution: {integrity: sha512-GhdPgy1el4/ImP05X05Uw4cw2/M93BCUmnEvWZNStlCzEKME4Fkk+YpoA5OiHNQmoS7Cafb8Xa3Pya8m1Qrzeg==} + + '@types/node@24.13.2': + resolution: {integrity: sha512-fRa09kZTgu8o71KFcDjUFuc7F+dEbZYZmkI0mg5YBTRs0yMKjYHsq/c0urDKeDb+D5qVgXOdFcuu+DZPKOITwA==} + + '@types/pg@8.20.0': + resolution: {integrity: sha512-bEPFOaMAHTEP1EzpvHTbmwR8UsFyHSKsRisLIHVMXnpNefSbGA1bD6CVy+qKjGSqmZqNqBDV2azOBo8TgkcVow==} + + '@types/ws@8.18.1': + resolution: {integrity: sha512-ThVF6DCVhA8kUGy+aazFQ4kXQ7E1Ty7A3ypFOe0IcJV8O/M511G99AW24irKrW56Wt44yG9+ij8FaqoBGkuBXg==} + + '@vitest/expect@4.1.9': + resolution: {integrity: sha512-vl/rYsUKcBr3SnQn166+XR5ZQcgMx3DQhFWdfli/cWpLnLUmbxZvyrJZotLFUryib+LtArYMSTJ5RbQ57ZqrlA==} + + '@vitest/mocker@4.1.9': + resolution: {integrity: sha512-EVkXzBjrPGM+cK8/ANWgBrkUCfJfb38/EfTSO8h7pWvKkyPkpWxvR7BkD2MyItMF62C97zAEoqdpUixwR/e+Rw==} + peerDependencies: + msw: ^2.4.9 + vite: ^6.0.0 || ^7.0.0 || ^8.0.0 + peerDependenciesMeta: + msw: + optional: true + vite: + optional: true + + '@vitest/pretty-format@4.1.9': + resolution: {integrity: sha512-s0iufns3iIFitdgm+YR7g1whCAaGtXz459VS9/PqyKDEEFgYIhsHOQmXgIgDuYCt7DeQmiZT0Qe2OA2p4ZPu5A==} + + '@vitest/runner@4.1.9': + resolution: {integrity: sha512-KXLMDtc7oe70+3mJfGrPUWPesswH+3sTxAMAMl8DG7I8IUQT4XW718dY5ID3vPUcmlu27CcKfY4P3h3I29SLJg==} + + '@vitest/snapshot@4.1.9': + resolution: {integrity: sha512-Jc7RKGNBo8Z28WYIm0Niej4xdSPByRf6mU58VpHQkd6Zh05rlnA+twjbK5HyeIGHxrzsc3mJgS43uM0CZKzaIA==} + + '@vitest/spy@4.1.9': + resolution: {integrity: sha512-fHpsS6mIi+PiEW+vcRVOMkX1oSaPKne3VOclSFICPcGOmfKgXPU5iAah+wcNcj2xPrCCmfq99IDGf+EojhhvhA==} + + '@vitest/utils@4.1.9': + resolution: {integrity: sha512-A51o8ymO5PpqlWNnBP9ZHPXDIpuMtTLlGSjN7la4US+LJzoUMyhwjA5QXlm39JexgwHKW4Xjs8Z2d3dLCXOeuA==} + + assertion-error@2.0.1: + resolution: {integrity: sha512-Izi8RQcffqCeNVgFigKli1ssklIbpHnCYc6AknXGYoB6grJqyeby7jv12JUQgmTAnIDnbck1uxksT4dzN3PWBA==} + engines: {node: '>=12'} + + chai@6.2.2: + resolution: {integrity: sha512-NUPRluOfOiTKBKvWPtSD4PhFvWCqOi0BGStNWs57X9js7XGTprSmFoz5F0tWhR4WPjNeR9jXqdC7/UpSJTnlRg==} + engines: {node: '>=18'} + + convert-source-map@2.0.0: + resolution: {integrity: sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg==} + + detect-libc@2.1.2: + resolution: {integrity: sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==} + engines: {node: '>=8'} + + es-module-lexer@2.1.0: + resolution: {integrity: sha512-n27zTYMjYu1aj4MjCWzSP7G9r75utsaoc8m61weK+W8JMBGGQybd43GstCXZ3WNmSFtGT9wi59qQTW6mhTR5LQ==} + + esbuild@0.28.1: + resolution: {integrity: sha512-HrJrvZv5ayxBzPfwphOoNzkzOIIlifzk0KJrGK2c8R4+LKpMtpYLQeUdjnwjWv/LZlkH2laZk+4w78pi99D4Vw==} + engines: {node: '>=18'} + hasBin: true + + estree-walker@3.0.3: + resolution: {integrity: sha512-7RUKfXgSMMkzt6ZuXmqapOurLGPPfgj6l9uRZ7lRGolvk0y2yocc35LdcxKC5PQZdn2DMqioAQ2NoWcrTKmm6g==} + + expect-type@1.3.0: + resolution: {integrity: sha512-knvyeauYhqjOYvQ66MznSMs83wmHrCycNEN6Ao+2AeYEfxUIkuiVxdEa1qlGEPK+We3n0THiDciYSsCcgW/DoA==} + engines: {node: '>=12.0.0'} + + fdir@6.5.0: + resolution: {integrity: sha512-tIbYtZbucOs0BRGqPJkshJUYdL+SDH7dVM8gjy+ERp3WAUjLEFJE+02kanyHtwjWOnwrKYBiwAmM0p4kLJAnXg==} + engines: {node: '>=12.0.0'} + peerDependencies: + picomatch: ^3 || ^4 + peerDependenciesMeta: + picomatch: + optional: true + + fsevents@2.3.3: + resolution: {integrity: sha512-5xoDfX+fL7faATnagmWPpbFtwh/R77WmMMqqHGS65C3vvB0YHrgF+B1YmZ3441tMj5n63k0212XNoJwzlhffQw==} + engines: {node: ^8.16.0 || ^10.6.0 || >=11.0.0} + os: [darwin] + + hono@4.12.27: + resolution: {integrity: sha512-1yrb/+w6HWQJrUCLkJ2IF5jNIPvvFkblV5RNOYl6bV+OA6p9GLcMpHFFGTosSvHvcAUibuUukRqhlYI4z32C7Q==} + engines: {node: '>=16.9.0'} + + jose@6.2.3: + resolution: {integrity: sha512-YYVDInQKFJfR/xa3ojUTl8c2KoTwiL1R5Wg9YCydwH0x0B9grbzlg5HC7mMjCtUJjbQ/YnGEZIhI5tCgfTb4Hw==} + + lightningcss-android-arm64@1.32.0: + resolution: {integrity: sha512-YK7/ClTt4kAK0vo6w3X+Pnm0D2cf2vPHbhOXdoNti1Ga0al1P4TBZhwjATvjNwLEBCnKvjJc2jQgHXH0NEwlAg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [android] + + lightningcss-darwin-arm64@1.32.0: + resolution: {integrity: sha512-RzeG9Ju5bag2Bv1/lwlVJvBE3q6TtXskdZLLCyfg5pt+HLz9BqlICO7LZM7VHNTTn/5PRhHFBSjk5lc4cmscPQ==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [darwin] + + lightningcss-darwin-x64@1.32.0: + resolution: {integrity: sha512-U+QsBp2m/s2wqpUYT/6wnlagdZbtZdndSmut/NJqlCcMLTWp5muCrID+K5UJ6jqD2BFshejCYXniPDbNh73V8w==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [darwin] + + lightningcss-freebsd-x64@1.32.0: + resolution: {integrity: sha512-JCTigedEksZk3tHTTthnMdVfGf61Fky8Ji2E4YjUTEQX14xiy/lTzXnu1vwiZe3bYe0q+SpsSH/CTeDXK6WHig==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [freebsd] + + lightningcss-linux-arm-gnueabihf@1.32.0: + resolution: {integrity: sha512-x6rnnpRa2GL0zQOkt6rts3YDPzduLpWvwAF6EMhXFVZXD4tPrBkEFqzGowzCsIWsPjqSK+tyNEODUBXeeVHSkw==} + engines: {node: '>= 12.0.0'} + cpu: [arm] + os: [linux] + + lightningcss-linux-arm64-gnu@1.32.0: + resolution: {integrity: sha512-0nnMyoyOLRJXfbMOilaSRcLH3Jw5z9HDNGfT/gwCPgaDjnx0i8w7vBzFLFR1f6CMLKF8gVbebmkUN3fa/kQJpQ==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [linux] + + lightningcss-linux-arm64-musl@1.32.0: + resolution: {integrity: sha512-UpQkoenr4UJEzgVIYpI80lDFvRmPVg6oqboNHfoH4CQIfNA+HOrZ7Mo7KZP02dC6LjghPQJeBsvXhJod/wnIBg==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [linux] + + lightningcss-linux-x64-gnu@1.32.0: + resolution: {integrity: sha512-V7Qr52IhZmdKPVr+Vtw8o+WLsQJYCTd8loIfpDaMRWGUZfBOYEJeyJIkqGIDMZPwPx24pUMfwSxxI8phr/MbOA==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [linux] + + lightningcss-linux-x64-musl@1.32.0: + resolution: {integrity: sha512-bYcLp+Vb0awsiXg/80uCRezCYHNg1/l3mt0gzHnWV9XP1W5sKa5/TCdGWaR/zBM2PeF/HbsQv/j2URNOiVuxWg==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [linux] + + lightningcss-win32-arm64-msvc@1.32.0: + resolution: {integrity: sha512-8SbC8BR40pS6baCM8sbtYDSwEVQd4JlFTOlaD3gWGHfThTcABnNDBda6eTZeqbofalIJhFx0qKzgHJmcPTnGdw==} + engines: {node: '>= 12.0.0'} + cpu: [arm64] + os: [win32] + + lightningcss-win32-x64-msvc@1.32.0: + resolution: {integrity: sha512-Amq9B/SoZYdDi1kFrojnoqPLxYhQ4Wo5XiL8EVJrVsB8ARoC1PWW6VGtT0WKCemjy8aC+louJnjS7U18x3b06Q==} + engines: {node: '>= 12.0.0'} + cpu: [x64] + os: [win32] + + lightningcss@1.32.0: + resolution: {integrity: sha512-NXYBzinNrblfraPGyrbPoD19C1h9lfI/1mzgWYvXUTe414Gz/X1FD2XBZSZM7rRTrMA8JL3OtAaGifrIKhQ5yQ==} + engines: {node: '>= 12.0.0'} + + magic-string@0.30.21: + resolution: {integrity: sha512-vd2F4YUyEXKGcLHoq+TEyCjxueSeHnFxyyjNp80yg0XV4vUhnDer/lvvlqM/arB5bXQN5K2/3oinyCRyx8T2CQ==} + + nanoid@3.3.13: + resolution: {integrity: sha512-sPdqC6ByMVVGvF1ynvvMo0/o+oD1VX7DaHhijt1bFgjvBkHBib4t49GoNDhf2NDta4oeUNlaGbSt5K7qjZ955Q==} + engines: {node: ^10 || ^12 || ^13.7 || ^14 || >=15.0.1} + hasBin: true + + obug@2.1.3: + resolution: {integrity: sha512-9miFgM2OFba7hB+pRgvtV84pYTBaoTHohvmIgiRt6dRIzbwEOIaNaP+dIlGs2fNFoB0SeISs0Jz5WFVRid6Xyg==} + engines: {node: '>=12.20.0'} + + pathe@2.0.3: + resolution: {integrity: sha512-WUjGcAqP1gQacoQe+OBJsFA7Ld4DyXuUIjZ5cc75cLHvJ7dtNsTugphxIADwspS+AraAUePCKrSVtPLFj/F88w==} + + pg-cloudflare@1.4.0: + resolution: {integrity: sha512-Vo7z/6rrQYxpNRylp4Tlob2elzbh+N/MOQbxFVWCxS7oEx6jF53GTJFxK2WWpKuBRkmiin4Mt+xofFDjx09R0A==} + + pg-connection-string@2.14.0: + resolution: {integrity: sha512-XwWDGcLRGCXAR8F/AM5bG7Q+A3Wm2s6QeEjlOKZLlH3UYcguiqCWKyWXVag5TLTIjR7oOJUY8kcADaZgWPyLeg==} + + pg-int8@1.0.1: + resolution: {integrity: sha512-WCtabS6t3c8SkpDBUlb1kjOs7l66xsGdKpIPZsg4wR+B3+u9UAum2odSsF9tnvxg80h4ZxLWMy4pRjOsFIqQpw==} + engines: {node: '>=4.0.0'} + + pg-pool@3.14.0: + resolution: {integrity: sha512-gKtPkFdQPU3DksooVLi9LsjZxrsBUZIpa+7aVx+LV5pNh0KzP4Zleud2po+ConrxbuXGBJ6Hfer6hdgpIBpBaw==} + peerDependencies: + pg: '>=8.0' + + pg-protocol@1.15.0: + resolution: {integrity: sha512-cq9sECI5s0+uPUXjbz8ioyPJni6RzsRib0US67i5IoTZKw8fNeYlVE7u8F4dG7vEJJtc5wdD1K189lCCUwqWTQ==} + + pg-types@2.2.0: + resolution: {integrity: sha512-qTAAlrEsl8s4OiEQY69wDvcMIdQN6wdz5ojQiOy6YRMuynxenON0O5oCpJI6lshc6scgAY8qvJ2On/p+CXY0GA==} + engines: {node: '>=4'} + + pg@8.22.0: + resolution: {integrity: sha512-8wih1vVIBMxoUM2oB4soJsD9tDnDpLv4OXBJ+EJzFsvycD+lfyIreC2gGHq78f8jbLLt+bvlPTFdFZfJkOuzAA==} + engines: {node: '>= 16.0.0'} + peerDependencies: + pg-native: '>=3.0.1' + peerDependenciesMeta: + pg-native: + optional: true + + pgpass@1.0.5: + resolution: {integrity: sha512-FdW9r/jQZhSeohs1Z3sI1yxFQNFvMcnmfuj4WBMUTxOrAyLMaTcE1aAMBiTlbMNaXvBCQuVi0R7hd8udDSP7ug==} + + picocolors@1.1.1: + resolution: {integrity: sha512-xceH2snhtb5M9liqDsmEw56le376mTZkEX/jEb/RxNFyegNul7eNslCXP9FDj/Lcu0X8KEyMceP2ntpaHrDEVA==} + + picomatch@4.0.4: + resolution: {integrity: sha512-QP88BAKvMam/3NxH6vj2o21R6MjxZUAd6nlwAS/pnGvN9IVLocLHxGYIzFhg6fUQ+5th6P4dv4eW9jX3DSIj7A==} + engines: {node: '>=12'} + + postcss@8.5.15: + resolution: {integrity: sha512-FfR8sjd4em2T6fb3I2MwAJU7HWVMr9zba+enmQeeWFfCbm+UOC/0X4DS8XtpUTMwWMGbjKYP7xjfNekzyGmB3A==} + engines: {node: ^10 || ^12 || >=14} + + postgres-array@2.0.0: + resolution: {integrity: sha512-VpZrUqU5A69eQyW2c5CA1jtLecCsN2U/bD6VilrFDWq5+5UIEVO7nazS3TEcHf1zuPYO/sqGvUvW62g86RXZuA==} + engines: {node: '>=4'} + + postgres-bytea@1.0.1: + resolution: {integrity: sha512-5+5HqXnsZPE65IJZSMkZtURARZelel2oXUEO8rH83VS/hxH5vv1uHquPg5wZs8yMAfdv971IU+kcPUczi7NVBQ==} + engines: {node: '>=0.10.0'} + + postgres-date@1.0.7: + resolution: {integrity: sha512-suDmjLVQg78nMK2UZ454hAG+OAW+HQPZ6n++TNDUX+L0+uUlLywnoxJKDou51Zm+zTCjrCl0Nq6J9C5hP9vK/Q==} + engines: {node: '>=0.10.0'} + + postgres-interval@1.2.0: + resolution: {integrity: sha512-9ZhXKM/rw350N1ovuWHbGxnGh/SNJ4cnxHiM0rxE4VN41wsg8P8zWn9hv/buK00RP4WvlOyr/RBDiptyxVbkZQ==} + engines: {node: '>=0.10.0'} + + rolldown@1.0.3: + resolution: {integrity: sha512-i00lAJ2ks1BYr7rjNjKC7BcqAS7nVfiT3QX1SI5aY+AFHblCmaUf9OE9dbdzDvW6dJxbi2ZCZiy9v3CcwOiX3g==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + + siginfo@2.0.0: + resolution: {integrity: sha512-ybx0WO1/8bSBLEWXZvEd7gMW3Sn3JFlW3TvX1nREbDLRNQNaeNN8WK0meBwPdAaOI7TtRRRJn/Es1zhrrCHu7g==} + + source-map-js@1.2.1: + resolution: {integrity: sha512-UXWMKhLOwVKb728IUtQPXxfYU+usdybtUrK/8uGE8CQMvrhOpwvzDBwj0QhSL7MQc7vIsISBG8VQ8+IDQxpfQA==} + engines: {node: '>=0.10.0'} + + split2@4.2.0: + resolution: {integrity: sha512-UcjcJOWknrNkF6PLX83qcHM6KHgVKNkV62Y8a5uYDVv9ydGQVwAHMKqHdJje1VTWpljG0WYpCDhrCdAOYH4TWg==} + engines: {node: '>= 10.x'} + + stackback@0.0.2: + resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==} + + std-env@4.1.0: + resolution: {integrity: sha512-Rq7ybcX2RuC55r9oaPVEW7/xu3tj8u4GeBYHBWCychFtzMIr86A7e3PPEBPT37sHStKX3+TiX/Fr/ACmJLVlLQ==} + + tinybench@2.9.0: + resolution: {integrity: sha512-0+DUvqWMValLmha6lr4kD8iAMK1HzV0/aKnCtWb9v9641TnP/MFb7Pc2bxoxQjTXAErryXVgUOfv2YqNllqGeg==} + + tinyexec@1.2.4: + resolution: {integrity: sha512-SHf/r48b7vOrjve9PxJo3MN5v5yuyjHvdUcrQffT3WXMUfnGmHDVbC4k3sHJaJTgZCwpUplIaAo5ANtMyp3YHg==} + engines: {node: '>=18'} + + tinyglobby@0.2.17: + resolution: {integrity: sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g==} + engines: {node: '>=12.0.0'} + + tinyrainbow@3.1.0: + resolution: {integrity: sha512-Bf+ILmBgretUrdJxzXM0SgXLZ3XfiaUuOj/IKQHuTXip+05Xn+uyEYdVg0kYDipTBcLrCVyUzAPz7QmArb0mmw==} + engines: {node: '>=14.0.0'} + + tslib@2.8.1: + resolution: {integrity: sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==} + + tsx@4.22.4: + resolution: {integrity: sha512-X8EX+XV4QR5xCsrgxaED954zTDfY8KqlDtskKEL0cHhyS/P8b4IFOvGDQpsC9Q1XnLq915wEfwwY/zzskCtmhg==} + engines: {node: '>=18.0.0'} + hasBin: true + + tweetnacl@1.0.3: + resolution: {integrity: sha512-6rt+RN7aOi1nGMyC4Xa5DdYiukl2UWCbcJft7YhxReBGQD7OAM8Pbxw6YMo4r2diNEA8FEmu32YOn9rhaiE5yw==} + + typescript@5.9.3: + resolution: {integrity: sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==} + engines: {node: '>=14.17'} + hasBin: true + + undici-types@7.18.2: + resolution: {integrity: sha512-AsuCzffGHJybSaRrmr5eHr81mwJU3kjw6M+uprWvCXiNeN9SOGwQ3Jn8jb8m3Z6izVgknn1R0FTCEAP2QrLY/w==} + + vite@8.0.16: + resolution: {integrity: sha512-h9bXPmJichP5fLmVQo3PyaGSDE2n3aPuomeAlVRm0JLmt4rY6zmPKd59HYI4LNW8oTK7tlTsuC7l/m7awx9Jcw==} + engines: {node: ^20.19.0 || >=22.12.0} + hasBin: true + peerDependencies: + '@types/node': ^20.19.0 || >=22.12.0 + '@vitejs/devtools': ^0.1.18 + esbuild: ^0.27.0 || ^0.28.0 + jiti: '>=1.21.0' + less: ^4.0.0 + sass: ^1.70.0 + sass-embedded: ^1.70.0 + stylus: '>=0.54.8' + sugarss: ^5.0.0 + terser: ^5.16.0 + tsx: ^4.8.1 + yaml: ^2.4.2 + peerDependenciesMeta: + '@types/node': + optional: true + '@vitejs/devtools': + optional: true + esbuild: + optional: true + jiti: + optional: true + less: + optional: true + sass: + optional: true + sass-embedded: + optional: true + stylus: + optional: true + sugarss: + optional: true + terser: + optional: true + tsx: + optional: true + yaml: + optional: true + + vitest@4.1.9: + resolution: {integrity: sha512-nE3/LEyc0z87uHYLZebqCUOaJr2hdtuPp7BQ4BosVFnfltxgAvMG08NyrSGlPpOUWvR27c5flSmYFTNr78L9GQ==} + engines: {node: ^20.0.0 || ^22.0.0 || >=24.0.0} + hasBin: true + peerDependencies: + '@edge-runtime/vm': '*' + '@opentelemetry/api': ^1.9.0 + '@types/node': ^20.0.0 || ^22.0.0 || >=24.0.0 + '@vitest/browser-playwright': 4.1.9 + '@vitest/browser-preview': 4.1.9 + '@vitest/browser-webdriverio': 4.1.9 + '@vitest/coverage-istanbul': 4.1.9 + '@vitest/coverage-v8': 4.1.9 + '@vitest/ui': 4.1.9 + happy-dom: '*' + jsdom: '*' + vite: ^6.0.0 || ^7.0.0 || ^8.0.0 + peerDependenciesMeta: + '@edge-runtime/vm': + optional: true + '@opentelemetry/api': + optional: true + '@types/node': + optional: true + '@vitest/browser-playwright': + optional: true + '@vitest/browser-preview': + optional: true + '@vitest/browser-webdriverio': + optional: true + '@vitest/coverage-istanbul': + optional: true + '@vitest/coverage-v8': + optional: true + '@vitest/ui': + optional: true + happy-dom: + optional: true + jsdom: + optional: true + + why-is-node-running@2.3.0: + resolution: {integrity: sha512-hUrmaWBdVDcxvYqnyh09zunKzROWjbZTiNy8dBEjkS7ehEDQibXJ7XvlmtbwuTclUiIyN+CyXQD4Vmko8fNm8w==} + engines: {node: '>=8'} + hasBin: true + + ws@8.21.0: + resolution: {integrity: sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g==} + engines: {node: '>=10.0.0'} + peerDependencies: + bufferutil: ^4.0.1 + utf-8-validate: '>=5.0.2' + peerDependenciesMeta: + bufferutil: + optional: true + utf-8-validate: + optional: true + + xtend@4.0.2: + resolution: {integrity: sha512-LKYU1iAXJXUgAXn9URjiu+MWhyUXHsvfp7mcuYm9dSUKK0/CjtrUwFAxD82/mCWbtLsGjFIad0wIsod4zrTAEQ==} + engines: {node: '>=0.4'} + + zod@3.25.76: + resolution: {integrity: sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==} + +snapshots: + + '@emnapi/core@1.10.0': + dependencies: + '@emnapi/wasi-threads': 1.2.1 + tslib: 2.8.1 + optional: true + + '@emnapi/runtime@1.10.0': + dependencies: + tslib: 2.8.1 + optional: true + + '@emnapi/wasi-threads@1.2.1': + dependencies: + tslib: 2.8.1 + optional: true + + '@esbuild/aix-ppc64@0.28.1': + optional: true + + '@esbuild/android-arm64@0.28.1': + optional: true + + '@esbuild/android-arm@0.28.1': + optional: true + + '@esbuild/android-x64@0.28.1': + optional: true + + '@esbuild/darwin-arm64@0.28.1': + optional: true + + '@esbuild/darwin-x64@0.28.1': + optional: true + + '@esbuild/freebsd-arm64@0.28.1': + optional: true + + '@esbuild/freebsd-x64@0.28.1': + optional: true + + '@esbuild/linux-arm64@0.28.1': + optional: true + + '@esbuild/linux-arm@0.28.1': + optional: true + + '@esbuild/linux-ia32@0.28.1': + optional: true + + '@esbuild/linux-loong64@0.28.1': + optional: true + + '@esbuild/linux-mips64el@0.28.1': + optional: true + + '@esbuild/linux-ppc64@0.28.1': + optional: true + + '@esbuild/linux-riscv64@0.28.1': + optional: true + + '@esbuild/linux-s390x@0.28.1': + optional: true + + '@esbuild/linux-x64@0.28.1': + optional: true + + '@esbuild/netbsd-arm64@0.28.1': + optional: true + + '@esbuild/netbsd-x64@0.28.1': + optional: true + + '@esbuild/openbsd-arm64@0.28.1': + optional: true + + '@esbuild/openbsd-x64@0.28.1': + optional: true + + '@esbuild/openharmony-arm64@0.28.1': + optional: true + + '@esbuild/sunos-x64@0.28.1': + optional: true + + '@esbuild/win32-arm64@0.28.1': + optional: true + + '@esbuild/win32-ia32@0.28.1': + optional: true + + '@esbuild/win32-x64@0.28.1': + optional: true + + '@hono/node-server@1.19.14(hono@4.12.27)': + dependencies: + hono: 4.12.27 + + '@jridgewell/sourcemap-codec@1.5.5': {} + + '@napi-rs/wasm-runtime@1.1.5(@emnapi/core@1.10.0)(@emnapi/runtime@1.10.0)': + dependencies: + '@emnapi/core': 1.10.0 + '@emnapi/runtime': 1.10.0 + '@tybys/wasm-util': 0.10.2 + optional: true + + '@oxc-project/types@0.133.0': {} + + '@rolldown/binding-android-arm64@1.0.3': + optional: true + + '@rolldown/binding-darwin-arm64@1.0.3': + optional: true + + '@rolldown/binding-darwin-x64@1.0.3': + optional: true + + '@rolldown/binding-freebsd-x64@1.0.3': + optional: true + + '@rolldown/binding-linux-arm-gnueabihf@1.0.3': + optional: true + + '@rolldown/binding-linux-arm64-gnu@1.0.3': + optional: true + + '@rolldown/binding-linux-arm64-musl@1.0.3': + optional: true + + '@rolldown/binding-linux-ppc64-gnu@1.0.3': + optional: true + + '@rolldown/binding-linux-s390x-gnu@1.0.3': + optional: true + + '@rolldown/binding-linux-x64-gnu@1.0.3': + optional: true + + '@rolldown/binding-linux-x64-musl@1.0.3': + optional: true + + '@rolldown/binding-openharmony-arm64@1.0.3': + optional: true + + '@rolldown/binding-wasm32-wasi@1.0.3': + dependencies: + '@emnapi/core': 1.10.0 + '@emnapi/runtime': 1.10.0 + '@napi-rs/wasm-runtime': 1.1.5(@emnapi/core@1.10.0)(@emnapi/runtime@1.10.0) + optional: true + + '@rolldown/binding-win32-arm64-msvc@1.0.3': + optional: true + + '@rolldown/binding-win32-x64-msvc@1.0.3': + optional: true + + '@rolldown/pluginutils@1.0.1': {} + + '@standard-schema/spec@1.1.0': {} + + '@tybys/wasm-util@0.10.2': + dependencies: + tslib: 2.8.1 + optional: true + + '@types/chai@5.2.3': + dependencies: + '@types/deep-eql': 4.0.2 + assertion-error: 2.0.1 + + '@types/deep-eql@4.0.2': {} + + '@types/estree@1.0.9': {} + + '@types/node@24.13.2': + dependencies: + undici-types: 7.18.2 + + '@types/pg@8.20.0': + dependencies: + '@types/node': 24.13.2 + pg-protocol: 1.15.0 + pg-types: 2.2.0 + + '@types/ws@8.18.1': + dependencies: + '@types/node': 24.13.2 + + '@vitest/expect@4.1.9': + dependencies: + '@standard-schema/spec': 1.1.0 + '@types/chai': 5.2.3 + '@vitest/spy': 4.1.9 + '@vitest/utils': 4.1.9 + chai: 6.2.2 + tinyrainbow: 3.1.0 + + '@vitest/mocker@4.1.9(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4))': + dependencies: + '@vitest/spy': 4.1.9 + estree-walker: 3.0.3 + magic-string: 0.30.21 + optionalDependencies: + vite: 8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4) + + '@vitest/pretty-format@4.1.9': + dependencies: + tinyrainbow: 3.1.0 + + '@vitest/runner@4.1.9': + dependencies: + '@vitest/utils': 4.1.9 + pathe: 2.0.3 + + '@vitest/snapshot@4.1.9': + dependencies: + '@vitest/pretty-format': 4.1.9 + '@vitest/utils': 4.1.9 + magic-string: 0.30.21 + pathe: 2.0.3 + + '@vitest/spy@4.1.9': {} + + '@vitest/utils@4.1.9': + dependencies: + '@vitest/pretty-format': 4.1.9 + convert-source-map: 2.0.0 + tinyrainbow: 3.1.0 + + assertion-error@2.0.1: {} + + chai@6.2.2: {} + + convert-source-map@2.0.0: {} + + detect-libc@2.1.2: {} + + es-module-lexer@2.1.0: {} + + esbuild@0.28.1: + optionalDependencies: + '@esbuild/aix-ppc64': 0.28.1 + '@esbuild/android-arm': 0.28.1 + '@esbuild/android-arm64': 0.28.1 + '@esbuild/android-x64': 0.28.1 + '@esbuild/darwin-arm64': 0.28.1 + '@esbuild/darwin-x64': 0.28.1 + '@esbuild/freebsd-arm64': 0.28.1 + '@esbuild/freebsd-x64': 0.28.1 + '@esbuild/linux-arm': 0.28.1 + '@esbuild/linux-arm64': 0.28.1 + '@esbuild/linux-ia32': 0.28.1 + '@esbuild/linux-loong64': 0.28.1 + '@esbuild/linux-mips64el': 0.28.1 + '@esbuild/linux-ppc64': 0.28.1 + '@esbuild/linux-riscv64': 0.28.1 + '@esbuild/linux-s390x': 0.28.1 + '@esbuild/linux-x64': 0.28.1 + '@esbuild/netbsd-arm64': 0.28.1 + '@esbuild/netbsd-x64': 0.28.1 + '@esbuild/openbsd-arm64': 0.28.1 + '@esbuild/openbsd-x64': 0.28.1 + '@esbuild/openharmony-arm64': 0.28.1 + '@esbuild/sunos-x64': 0.28.1 + '@esbuild/win32-arm64': 0.28.1 + '@esbuild/win32-ia32': 0.28.1 + '@esbuild/win32-x64': 0.28.1 + + estree-walker@3.0.3: + dependencies: + '@types/estree': 1.0.9 + + expect-type@1.3.0: {} + + fdir@6.5.0(picomatch@4.0.4): + optionalDependencies: + picomatch: 4.0.4 + + fsevents@2.3.3: + optional: true + + hono@4.12.27: {} + + jose@6.2.3: {} + + lightningcss-android-arm64@1.32.0: + optional: true + + lightningcss-darwin-arm64@1.32.0: + optional: true + + lightningcss-darwin-x64@1.32.0: + optional: true + + lightningcss-freebsd-x64@1.32.0: + optional: true + + lightningcss-linux-arm-gnueabihf@1.32.0: + optional: true + + lightningcss-linux-arm64-gnu@1.32.0: + optional: true + + lightningcss-linux-arm64-musl@1.32.0: + optional: true + + lightningcss-linux-x64-gnu@1.32.0: + optional: true + + lightningcss-linux-x64-musl@1.32.0: + optional: true + + lightningcss-win32-arm64-msvc@1.32.0: + optional: true + + lightningcss-win32-x64-msvc@1.32.0: + optional: true + + lightningcss@1.32.0: + dependencies: + detect-libc: 2.1.2 + optionalDependencies: + lightningcss-android-arm64: 1.32.0 + lightningcss-darwin-arm64: 1.32.0 + lightningcss-darwin-x64: 1.32.0 + lightningcss-freebsd-x64: 1.32.0 + lightningcss-linux-arm-gnueabihf: 1.32.0 + lightningcss-linux-arm64-gnu: 1.32.0 + lightningcss-linux-arm64-musl: 1.32.0 + lightningcss-linux-x64-gnu: 1.32.0 + lightningcss-linux-x64-musl: 1.32.0 + lightningcss-win32-arm64-msvc: 1.32.0 + lightningcss-win32-x64-msvc: 1.32.0 + + magic-string@0.30.21: + dependencies: + '@jridgewell/sourcemap-codec': 1.5.5 + + nanoid@3.3.13: {} + + obug@2.1.3: {} + + pathe@2.0.3: {} + + pg-cloudflare@1.4.0: + optional: true + + pg-connection-string@2.14.0: {} + + pg-int8@1.0.1: {} + + pg-pool@3.14.0(pg@8.22.0): + dependencies: + pg: 8.22.0 + + pg-protocol@1.15.0: {} + + pg-types@2.2.0: + dependencies: + pg-int8: 1.0.1 + postgres-array: 2.0.0 + postgres-bytea: 1.0.1 + postgres-date: 1.0.7 + postgres-interval: 1.2.0 + + pg@8.22.0: + dependencies: + pg-connection-string: 2.14.0 + pg-pool: 3.14.0(pg@8.22.0) + pg-protocol: 1.15.0 + pg-types: 2.2.0 + pgpass: 1.0.5 + optionalDependencies: + pg-cloudflare: 1.4.0 + + pgpass@1.0.5: + dependencies: + split2: 4.2.0 + + picocolors@1.1.1: {} + + picomatch@4.0.4: {} + + postcss@8.5.15: + dependencies: + nanoid: 3.3.13 + picocolors: 1.1.1 + source-map-js: 1.2.1 + + postgres-array@2.0.0: {} + + postgres-bytea@1.0.1: {} + + postgres-date@1.0.7: {} + + postgres-interval@1.2.0: + dependencies: + xtend: 4.0.2 + + rolldown@1.0.3: + dependencies: + '@oxc-project/types': 0.133.0 + '@rolldown/pluginutils': 1.0.1 + optionalDependencies: + '@rolldown/binding-android-arm64': 1.0.3 + '@rolldown/binding-darwin-arm64': 1.0.3 + '@rolldown/binding-darwin-x64': 1.0.3 + '@rolldown/binding-freebsd-x64': 1.0.3 + '@rolldown/binding-linux-arm-gnueabihf': 1.0.3 + '@rolldown/binding-linux-arm64-gnu': 1.0.3 + '@rolldown/binding-linux-arm64-musl': 1.0.3 + '@rolldown/binding-linux-ppc64-gnu': 1.0.3 + '@rolldown/binding-linux-s390x-gnu': 1.0.3 + '@rolldown/binding-linux-x64-gnu': 1.0.3 + '@rolldown/binding-linux-x64-musl': 1.0.3 + '@rolldown/binding-openharmony-arm64': 1.0.3 + '@rolldown/binding-wasm32-wasi': 1.0.3 + '@rolldown/binding-win32-arm64-msvc': 1.0.3 + '@rolldown/binding-win32-x64-msvc': 1.0.3 + + siginfo@2.0.0: {} + + source-map-js@1.2.1: {} + + split2@4.2.0: {} + + stackback@0.0.2: {} + + std-env@4.1.0: {} + + tinybench@2.9.0: {} + + tinyexec@1.2.4: {} + + tinyglobby@0.2.17: + dependencies: + fdir: 6.5.0(picomatch@4.0.4) + picomatch: 4.0.4 + + tinyrainbow@3.1.0: {} + + tslib@2.8.1: + optional: true + + tsx@4.22.4: + dependencies: + esbuild: 0.28.1 + optionalDependencies: + fsevents: 2.3.3 + + tweetnacl@1.0.3: {} + + typescript@5.9.3: {} + + undici-types@7.18.2: {} + + vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4): + dependencies: + lightningcss: 1.32.0 + picomatch: 4.0.4 + postcss: 8.5.15 + rolldown: 1.0.3 + tinyglobby: 0.2.17 + optionalDependencies: + '@types/node': 24.13.2 + esbuild: 0.28.1 + fsevents: 2.3.3 + tsx: 4.22.4 + + vitest@4.1.9(@types/node@24.13.2)(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)): + dependencies: + '@vitest/expect': 4.1.9 + '@vitest/mocker': 4.1.9(vite@8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4)) + '@vitest/pretty-format': 4.1.9 + '@vitest/runner': 4.1.9 + '@vitest/snapshot': 4.1.9 + '@vitest/spy': 4.1.9 + '@vitest/utils': 4.1.9 + es-module-lexer: 2.1.0 + expect-type: 1.3.0 + magic-string: 0.30.21 + obug: 2.1.3 + pathe: 2.0.3 + picomatch: 4.0.4 + std-env: 4.1.0 + tinybench: 2.9.0 + tinyexec: 1.2.4 + tinyglobby: 0.2.17 + tinyrainbow: 3.1.0 + vite: 8.0.16(@types/node@24.13.2)(esbuild@0.28.1)(tsx@4.22.4) + why-is-node-running: 2.3.0 + optionalDependencies: + '@types/node': 24.13.2 + transitivePeerDependencies: + - msw + + why-is-node-running@2.3.0: + dependencies: + siginfo: 2.0.0 + stackback: 0.0.2 + + ws@8.21.0: {} + + xtend@4.0.2: {} + + zod@3.25.76: {} diff --git a/cloud/pnpm-workspace.yaml b/cloud/pnpm-workspace.yaml new file mode 100644 index 00000000000..cc61e60464c --- /dev/null +++ b/cloud/pnpm-workspace.yaml @@ -0,0 +1,4 @@ +packages: + - apps/* + - packages/* + diff --git a/cloud/tsconfig.base.json b/cloud/tsconfig.base.json new file mode 100644 index 00000000000..d977ad5c9b9 --- /dev/null +++ b/cloud/tsconfig.base.json @@ -0,0 +1,19 @@ +{ + "compilerOptions": { + "allowSyntheticDefaultImports": true, + "declaration": true, + "esModuleInterop": true, + "forceConsistentCasingInFileNames": true, + "isolatedModules": true, + "lib": ["ES2024"], + "module": "NodeNext", + "moduleResolution": "NodeNext", + "noUncheckedIndexedAccess": true, + "outDir": "dist", + "resolveJsonModule": true, + "skipLibCheck": true, + "strict": true, + "target": "ES2024" + } +} + diff --git a/config/oxlint-react-doctor.json b/config/oxlint-react-doctor.json index 3c91b01d58d..c39f0d51bf2 100644 --- a/config/oxlint-react-doctor.json +++ b/config/oxlint-react-doctor.json @@ -25,5 +25,5 @@ "react-doctor/zustand-no-mutating-state": "warn", "react-doctor/zustand-no-whole-store-destructure": "warn" }, - "ignorePatterns": ["**/node_modules", "**/dist", "**/out"] + "ignorePatterns": ["**/node_modules", "**/dist", "**/out", "cloud/**"] } diff --git a/config/scripts/check-changed-code-quality.mjs b/config/scripts/check-changed-code-quality.mjs index 4427c6b7f38..eedf3dbda78 100644 --- a/config/scripts/check-changed-code-quality.mjs +++ b/config/scripts/check-changed-code-quality.mjs @@ -7,6 +7,7 @@ import { resolvePullRequestDiffBase } from './git-pull-request-diff-base.mjs' import { resolveOxlintInvocation } from './oxlint-cli-invocation.mjs' const SOURCE_FILE_PATTERN = /\.(?:[cm]?[jt]sx?)$/ +const ROOT_CODE_QUALITY_IGNORED_PREFIXES = ['cloud/'] export const OXLINT_SCANS = [ { // Why: no --config, so Oxlint keeps discovering nested configs. Pinning the root @@ -73,6 +74,10 @@ function splitNullDelimited(output) { return output.split('\0').filter(Boolean) } +export function isRootCodeQualityPath(file) { + return !ROOT_CODE_QUALITY_IGNORED_PREFIXES.some((prefix) => file.startsWith(prefix)) +} + function resolveBase(root, requestedBase) { for (const candidate of [ requestedBase, @@ -107,7 +112,11 @@ export function collectAddedLineRanges(root, requestedBase) { const rangesByFile = new Map() for (const file of changedFiles) { - if (!SOURCE_FILE_PATTERN.test(file) || !existsSync(path.join(root, file))) { + if ( + !isRootCodeQualityPath(file) || + !SOURCE_FILE_PATTERN.test(file) || + !existsSync(path.join(root, file)) + ) { continue } const diff = runGit(root, ['diff', '--unified=0', '--no-color', comparisonBase, '--', file]) @@ -119,7 +128,11 @@ export function collectAddedLineRanges(root, requestedBase) { for (const file of untrackedFiles) { const absolutePath = path.join(root, file) - if (!SOURCE_FILE_PATTERN.test(file) || !existsSync(absolutePath)) { + if ( + !isRootCodeQualityPath(file) || + !SOURCE_FILE_PATTERN.test(file) || + !existsSync(absolutePath) + ) { continue } const lineCount = readFileSync(absolutePath, 'utf8').split(/\r?\n/).length diff --git a/config/scripts/check-changed-code-quality.test.mjs b/config/scripts/check-changed-code-quality.test.mjs index 76a25802e5c..3a88cf1b02e 100644 --- a/config/scripts/check-changed-code-quality.test.mjs +++ b/config/scripts/check-changed-code-quality.test.mjs @@ -3,6 +3,7 @@ import { OXLINT_SCANS, diagnosticTouchesAddedLines, isMovedCode, + isRootCodeQualityPath, overlapsAddedLines, parseAddedLineRanges } from './check-changed-code-quality.mjs' @@ -52,6 +53,11 @@ describe('changed-code quality line matching', () => { expect(scan.args).not.toContain('--config') expect(scan.args).not.toContain('--disable-nested-config') }) + + it('leaves Cloud source to the independent Cloud quality checks', () => { + expect(isRootCodeQualityPath('cloud/apps/relay/src/index.ts')).toBe(false) + expect(isRootCodeQualityPath('src/main/index.ts')).toBe(true) + }) }) describe('moved-code exemption', () => { diff --git a/config/scripts/check-root-directory-entries.test.mjs b/config/scripts/check-root-directory-entries.test.mjs index 15276e8aa82..dcc5596771b 100644 --- a/config/scripts/check-root-directory-entries.test.mjs +++ b/config/scripts/check-root-directory-entries.test.mjs @@ -105,6 +105,15 @@ describe('root directory guard', () => { expect(output).toContain('new-root.md') }) + it('allows the reviewed cloud workspace directory', () => { + const fixture = makeFixture() + const head = commitFiles(fixture.root, [['cloud/package.json', '{}\n']]) + + const result = runGuard({ ...fixture, head }) + + expect(result.status).toBe(0) + }) + it('rejects a new top-level directory', () => { const fixture = makeFixture() const head = commitFiles(fixture.root, [['new-folder/file.txt', 'too prominent\n']]) diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index 67d4f565afa..8bc10fc5b72 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -237,6 +237,8 @@ const WINDOWS_PACKAGE_TESTS = [ const DESKTOP_IRRELEVANT_PREFIXES = [ 'mobile/', + 'cloud/', + '.github/workflows/cloud-', '.github/workflows/mobile.yml', '.github/workflows/mobile-ios-release.yml', '.github/workflows/mobile-android-release.yml' diff --git a/config/scripts/pr-code-change-scope.test.mjs b/config/scripts/pr-code-change-scope.test.mjs index 9a1c9e649b6..1fe296af265 100644 --- a/config/scripts/pr-code-change-scope.test.mjs +++ b/config/scripts/pr-code-change-scope.test.mjs @@ -86,6 +86,17 @@ describe('docs-only path classification', () => { it('does not start desktop PR Checks for mobile-only diffs', () => { expect(shouldRunPrChecks(['mobile/src/App.tsx', 'mobile/package.json'])).toBe(false) }) + + it('does not start desktop PR Checks for cloud-only diffs', () => { + expect( + shouldRunPrChecks([ + 'cloud/apps/relay/src/index.ts', + 'cloud/package.json', + 'cloud/.gitleaks.toml', + '.github/workflows/cloud-verify.yml' + ]) + ).toBe(false) + }) }) describe('per-job path classification', () => { diff --git a/package.json b/package.json index 005a1e12b0d..179ffd1a82a 100644 --- a/package.json +++ b/package.json @@ -293,12 +293,12 @@ "windows-native-registry": "3.2.2" }, "lint-staged": { - "*.{ts,tsx,js,jsx,mjs,mts,cts}": [ + "{*.{ts,tsx,js,jsx,mjs,mts,cts},!(cloud)/**/*.{ts,tsx,js,jsx,mjs,mts,cts}}": [ "oxlint", "oxlint --config config/oxlint-react-doctor.json", "oxfmt --write" ], - "*.{json,css}": [ + "{*.{json,css},!(cloud)/**/*.{json,css}}": [ "oxfmt --write" ] }, From 3de1b9d0583cb0657b2eb40dcca4366db2de9d12 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 07:05:21 -0400 Subject: [PATCH 179/398] fix(cloud): stop asking setup-node to cache the pnpm store in the relay workflows (#18432) setup-node's cache: pnpm runs 'pnpm store path' from the repository root, where packageManager pins pnpm 12; the shim it downloads fails to execute on the runner, so the step dies before auth. Cloud Verify never used the cache and passes; the six relay workflows that copied it from orca-cloud (root pnpm 10 there) now match. --- .../workflows/cloud-deploy-relay-production-capacity-job.yml | 2 -- .github/workflows/cloud-deploy-relay-production-capacity.yml | 2 -- .../workflows/cloud-deploy-relay-production-multi-target.yml | 2 -- .../workflows/cloud-deploy-relay-production-same-cap-job.yml | 2 -- .github/workflows/cloud-deploy-relay-production.yml | 2 -- .github/workflows/cloud-monitor-relay-production-job.yml | 2 -- 6 files changed, 12 deletions(-) diff --git a/.github/workflows/cloud-deploy-relay-production-capacity-job.yml b/.github/workflows/cloud-deploy-relay-production-capacity-job.yml index a98f910ada4..f7a5f248b05 100644 --- a/.github/workflows/cloud-deploy-relay-production-capacity-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-capacity-job.yml @@ -164,8 +164,6 @@ jobs: - uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - cache-dependency-path: cloud/pnpm-lock.yaml - run: pnpm install --frozen-lockfile diff --git a/.github/workflows/cloud-deploy-relay-production-capacity.yml b/.github/workflows/cloud-deploy-relay-production-capacity.yml index 98985eebd6a..5f6a1897d73 100644 --- a/.github/workflows/cloud-deploy-relay-production-capacity.yml +++ b/.github/workflows/cloud-deploy-relay-production-capacity.yml @@ -136,8 +136,6 @@ jobs: - uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - cache-dependency-path: cloud/pnpm-lock.yaml - run: pnpm install --frozen-lockfile diff --git a/.github/workflows/cloud-deploy-relay-production-multi-target.yml b/.github/workflows/cloud-deploy-relay-production-multi-target.yml index 8f27f568b50..8753cf81895 100644 --- a/.github/workflows/cloud-deploy-relay-production-multi-target.yml +++ b/.github/workflows/cloud-deploy-relay-production-multi-target.yml @@ -193,8 +193,6 @@ jobs: - uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - cache-dependency-path: cloud/pnpm-lock.yaml - run: pnpm install --frozen-lockfile diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index a8d8c68fafd..0b4700e16ef 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -99,8 +99,6 @@ jobs: - uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - cache-dependency-path: cloud/pnpm-lock.yaml - run: pnpm install --frozen-lockfile diff --git a/.github/workflows/cloud-deploy-relay-production.yml b/.github/workflows/cloud-deploy-relay-production.yml index 6de990e4c90..4c8691b9510 100644 --- a/.github/workflows/cloud-deploy-relay-production.yml +++ b/.github/workflows/cloud-deploy-relay-production.yml @@ -90,8 +90,6 @@ jobs: - uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - cache-dependency-path: cloud/pnpm-lock.yaml - run: pnpm install --frozen-lockfile diff --git a/.github/workflows/cloud-monitor-relay-production-job.yml b/.github/workflows/cloud-monitor-relay-production-job.yml index 4e97fb1bb80..4278b1420bc 100644 --- a/.github/workflows/cloud-monitor-relay-production-job.yml +++ b/.github/workflows/cloud-monitor-relay-production-job.yml @@ -64,8 +64,6 @@ jobs: - uses: actions/setup-node@v4 with: node-version: 24 - cache: pnpm - cache-dependency-path: cloud/pnpm-lock.yaml - run: pnpm install --frozen-lockfile From 67e22345daf882190911355eed152ef33e051c5e Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 07:18:29 -0400 Subject: [PATCH 180/398] fix(cloud): stop passing manage_artifact_dns to the relay root (#18442) The relay root does not declare it (it belongs to the private apps root), and Terraform rejects an undeclared -var, so the first public Deploy Relay Staging run failed at the C4 image bind. --- .github/workflows/cloud-deploy-relay-asia-topology.yml | 2 -- .../cloud-deploy-relay-production-capacity-job.yml | 5 ----- .../cloud-deploy-relay-production-same-cap-job.yml | 6 ++---- .github/workflows/cloud-deploy-relay-staging.yml | 2 +- .../workflows/cloud-prove-relay-staging-capacity.yml | 6 +++--- .../workflows/cloud-recover-relay-staging-c4-image.yml | 10 +++++----- .../dev/scripts/relay-asia-topology-workflow.test.mjs | 2 +- .../relay-production-capacity-workflow.test.mjs | 2 +- .../scripts/relay-staging-c4-refresh-workflow.test.mjs | 3 ++- 9 files changed, 15 insertions(+), 23 deletions(-) diff --git a/.github/workflows/cloud-deploy-relay-asia-topology.yml b/.github/workflows/cloud-deploy-relay-asia-topology.yml index f15fc5ae0e2..62e5ebb1426 100644 --- a/.github/workflows/cloud-deploy-relay-asia-topology.yml +++ b/.github/workflows/cloud-deploy-relay-asia-topology.yml @@ -176,7 +176,6 @@ jobs: mapfile -t targets < "${{ steps.targets.outputs.file }}" terraform -chdir=infra/terraform plan -input=false -lock-timeout=30s \ -var-file="${TF_VARS}" \ - -var manage_artifact_dns=false \ "${targets[@]}" -out="${plan}" terraform -chdir=infra/terraform show -json "${plan}" > "${plan_json}" committed="${RUNNER_TEMP}/relay-committed-asia-topology.json" @@ -223,7 +222,6 @@ jobs: plan_json="${RUNNER_TEMP}/relay-asia-topology-readback.json" terraform -chdir=infra/terraform plan -input=false -lock-timeout=30s \ -var-file="${TF_VARS}" \ - -var manage_artifact_dns=false \ "${targets[@]}" -out="${plan}" terraform -chdir=infra/terraform show -json "${plan}" > "${plan_json}" result="$(node dev/scripts/validate-relay-asia-topology-plan.mjs \ diff --git a/.github/workflows/cloud-deploy-relay-production-capacity-job.yml b/.github/workflows/cloud-deploy-relay-production-capacity-job.yml index f7a5f248b05..4cf39571371 100644 --- a/.github/workflows/cloud-deploy-relay-production-capacity-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-capacity-job.yml @@ -274,7 +274,6 @@ jobs: CELL_ORIGIN="https://${TARGET_HOSTNAME}.relay.onorca.dev" CELLS_JSON="$(terraform -chdir=infra/terraform console \ -var-file=environments/production.tfvars \ - -var manage_artifact_dns=false \ <<< 'jsonencode(var.relay_gce_cells)' | jq -er '.')" OVERRIDE_CELLS_JSON="$(jq -ce \ --arg cell "${TARGET_CELL_ID}" \ @@ -287,19 +286,16 @@ jobs: '{relay_gce_cells:$cells}' > "${RUNNER_TEMP}/relay-capacity.tfvars.json" BASE_CELLS_JSON="$(terraform -chdir=infra/terraform console \ -var-file=environments/production.tfvars \ - -var manage_artifact_dns=false \ <<< 'local.relay_director_cells_json' | jq -er '.')" IMAGE_EXPRESSION="var.relay_gce_cells[\"${TARGET_CELL_ID}\"].image" ZONE_EXPRESSION="var.relay_gce_cells[\"${TARGET_CELL_ID}\"].zone" DESIRED_IMAGE="$(terraform -chdir=infra/terraform console \ -var-file=environments/production.tfvars \ -var-file="${RUNNER_TEMP}/relay-capacity.tfvars.json" \ - -var manage_artifact_dns=false \ <<< "${IMAGE_EXPRESSION}" | jq -r '.')" TARGET_ZONE="$(terraform -chdir=infra/terraform console \ -var-file=environments/production.tfvars \ -var-file="${RUNNER_TEMP}/relay-capacity.tfvars.json" \ - -var manage_artifact_dns=false \ <<< "${ZONE_EXPRESSION}" | jq -r '.')" MIG_NAME="$(terraform -chdir=infra/terraform output -json relay_gce_cell_deployments \ | jq -r --arg cell "${TARGET_CELL_ID}" '.[$cell].mig_name')" @@ -700,7 +696,6 @@ jobs: terraform -chdir=infra/terraform plan \ -var-file=environments/production.tfvars \ -var-file="${RUNNER_TEMP}/relay-capacity.tfvars.json" \ - -var manage_artifact_dns=false \ "-target=google_compute_instance_template.relay_gce_cell[\"${TARGET_CELL_ID}\"]" \ "-target=google_compute_instance_group_manager.relay_gce_cell[\"${TARGET_CELL_ID}\"]" \ -out="${RUNNER_TEMP}/relay-capacity-cell.tfplan" diff --git a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml index 0b4700e16ef..6430c3f4793 100644 --- a/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml +++ b/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml @@ -218,10 +218,10 @@ jobs: EXPECTED_UNOBSERVED_BOUND=60 CELL_ORIGIN="https://${TARGET_HOSTNAME}.relay.onorca.dev" CELLS_JSON="$(terraform -chdir=infra/terraform console \ - -var-file=environments/production.tfvars -var manage_artifact_dns=false \ + -var-file=environments/production.tfvars \ <<< 'jsonencode(var.relay_gce_cells)' | jq -er '.')" SOURCE_CELLS="$(terraform -chdir=infra/terraform console \ - -var-file=environments/production.tfvars -var manage_artifact_dns=false \ + -var-file=environments/production.tfvars \ <<< 'jsonencode(var.relay_region_rehome_source_cell_ids)' | jq -er '.')" if test "${EXPECTED_REGION}" = us-central1; then jq -e --arg cell "${TARGET_CELL_ID}" 'index($cell) != null' \ @@ -452,7 +452,6 @@ jobs: terraform -chdir=infra/terraform plan \ -var-file=environments/production.tfvars \ -var-file="${RUNNER_TEMP}/relay-same-cap.tfvars.json" \ - -var manage_artifact_dns=false \ "-target=google_compute_instance_template.relay_gce_cell[\"${TARGET_CELL_ID}\"]" \ "-target=google_compute_instance_group_manager.relay_gce_cell[\"${TARGET_CELL_ID}\"]" \ -out="${RUNNER_TEMP}/relay-same-cap-resume.tfplan" @@ -502,7 +501,6 @@ jobs: terraform -chdir=infra/terraform plan \ -var-file=environments/production.tfvars \ -var-file="${RUNNER_TEMP}/relay-same-cap.tfvars.json" \ - -var manage_artifact_dns=false \ "-target=google_compute_instance_template.relay_gce_cell[\"${TARGET_CELL_ID}\"]" \ "-target=google_compute_instance_group_manager.relay_gce_cell[\"${TARGET_CELL_ID}\"]" \ -out="${RUNNER_TEMP}/relay-same-cap.tfplan" diff --git a/.github/workflows/cloud-deploy-relay-staging.yml b/.github/workflows/cloud-deploy-relay-staging.yml index cdde494f2ca..7ddfb9d7c73 100644 --- a/.github/workflows/cloud-deploy-relay-staging.yml +++ b/.github/workflows/cloud-deploy-relay-staging.yml @@ -63,7 +63,7 @@ jobs: terraform -chdir=infra/terraform init -reconfigure \ -backend-config=backend/staging.hcl -input=false IMAGE="$(terraform -chdir=infra/terraform console \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false \ + -var-file=environments/staging.tfvars \ <<< 'var.relay_gce_cells["staging-gce-c4"].image' | jq -er '.')" test "${IMAGE}" = \ "${GCP_REGION}-docker.pkg.dev/${GCP_PROJECT_ID}/${REPOSITORY_ID}/${IMAGE_NAME}@${EXPECTED_IMAGE_DIGEST}" diff --git a/.github/workflows/cloud-prove-relay-staging-capacity.yml b/.github/workflows/cloud-prove-relay-staging-capacity.yml index ee4fb5ed9b1..52a7538d40b 100644 --- a/.github/workflows/cloud-prove-relay-staging-capacity.yml +++ b/.github/workflows/cloud-prove-relay-staging-capacity.yml @@ -655,7 +655,7 @@ jobs: run: | set -euo pipefail cells="$(terraform -chdir=infra/terraform console \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false \ + -var-file=environments/staging.tfvars \ <<< 'jsonencode(var.relay_gce_cells)' | jq -er '.')" shape="$(jq -cer --arg cell "${TARGET_CELL_ID}" '.[$cell]' <<< "${cells}")" test "$(jq -r '.hostname' <<< "${shape}")" = c4 @@ -690,7 +690,7 @@ jobs: set -euo pipefail plan="${RUNNER_TEMP}/relay-c4-image-refresh.tfplan" terraform -chdir=infra/terraform plan \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false \ + -var-file=environments/staging.tfvars \ '-target=google_compute_instance_template.relay_gce_cell["staging-gce-c4"]' \ '-target=google_compute_instance_group_manager.relay_gce_cell["staging-gce-c4"]' \ -out="${plan}" @@ -905,7 +905,7 @@ jobs: set -euo pipefail plan="${RUNNER_TEMP}/relay-c4-image-readback.tfplan" terraform -chdir=infra/terraform plan \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false \ + -var-file=environments/staging.tfvars \ '-target=google_compute_instance_template.relay_gce_cell["staging-gce-c4"]' \ '-target=google_compute_instance_group_manager.relay_gce_cell["staging-gce-c4"]' \ -out="${plan}" diff --git a/.github/workflows/cloud-recover-relay-staging-c4-image.yml b/.github/workflows/cloud-recover-relay-staging-c4-image.yml index 42cbaf0e374..244cee453f6 100644 --- a/.github/workflows/cloud-recover-relay-staging-c4-image.yml +++ b/.github/workflows/cloud-recover-relay-staging-c4-image.yml @@ -165,13 +165,13 @@ jobs: --project "${GCP_PROJECT_ID}" --format='value(image_summary.digest)')" test "${served_digest}" = "${PREDECESSOR_IMAGE_DIGEST}" cells="$(terraform -chdir=infra/terraform console \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false \ + -var-file=environments/staging.tfvars \ <<< 'jsonencode(var.relay_gce_cells)' | jq -er '.')" test "$(jq -r --arg cell "${TARGET_CELL_ID}" '.[$cell].image' <<< "${cells}")" = \ "${target_image}" target_plan="${RUNNER_TEMP}/relay-c4-image-target.tfplan" terraform -chdir=infra/terraform plan \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false -lock-timeout=5m \ + -var-file=environments/staging.tfvars -lock-timeout=5m \ '-target=google_compute_instance_template.relay_gce_cell["staging-gce-c4"]' \ '-target=google_compute_instance_group_manager.relay_gce_cell["staging-gce-c4"]' \ -out="${target_plan}" @@ -193,7 +193,7 @@ jobs: terraform -chdir=infra/terraform plan \ -var-file=environments/staging.tfvars \ -var-file="${RUNNER_TEMP}/relay-c4-recovery.tfvars.json" \ - -var manage_artifact_dns=false -lock-timeout=5m \ + -lock-timeout=5m \ '-target=google_compute_instance_template.relay_gce_cell["staging-gce-c4"]' \ '-target=google_compute_instance_group_manager.relay_gce_cell["staging-gce-c4"]' \ -out="${plan}" @@ -339,7 +339,7 @@ jobs: other_image="${image_repository}@${PREDECESSOR_IMAGE_DIGEST}" fi cells="$(terraform -chdir=infra/terraform console \ - -var-file=environments/staging.tfvars -var manage_artifact_dns=false \ + -var-file=environments/staging.tfvars \ <<< 'jsonencode(var.relay_gce_cells)' | jq -er '.')" recovery_cells="$(jq -ce --arg cell "${TARGET_CELL_ID}" --arg image "${recovery_image}" \ '.[$cell].image = $image' <<< "${cells}")" @@ -349,7 +349,7 @@ jobs: terraform -chdir=infra/terraform plan \ -var-file=environments/staging.tfvars \ -var-file="${RUNNER_TEMP}/relay-c4-readback.tfvars.json" \ - -var manage_artifact_dns=false -lock-timeout=5m \ + -lock-timeout=5m \ '-target=google_compute_instance_template.relay_gce_cell["staging-gce-c4"]' \ '-target=google_compute_instance_group_manager.relay_gce_cell["staging-gce-c4"]' \ -out="${readback}" diff --git a/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs b/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs index 965a3ce142c..1927d9016ef 100644 --- a/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs +++ b/cloud/dev/scripts/relay-asia-topology-workflow.test.mjs @@ -36,7 +36,7 @@ test('uses only its exact workflow-bound topology identity', () => { }) test('plans only additive Asia topology and applies the saved plan', () => { - assert.equal((workflow.match(/manage_artifact_dns=false/g) ?? []).length, 2) + assert.doesNotMatch(workflow, /manage_artifact_dns/) for (const target of [ 'relay_gce_additional', 'google_compute_instance_template.relay_gce_cell', diff --git a/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs b/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs index 1c9e3aee5f6..d482d8cd856 100644 --- a/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs +++ b/cloud/dev/scripts/relay-production-capacity-workflow.test.mjs @@ -265,7 +265,7 @@ test('Terraform mutation targets only the selected cell and has fail-closed reco assert.match(workflow, /test "\$\{MUTATION_STARTED:-false\}" = true \|\| exit 0/) assert.match(workflow, /--mode isolate/) assert.doesNotMatch(workflow, /rolling-action restart/) - assert.equal(workflow.match(/manage_artifact_dns=false/g)?.length, 5) + assert.doesNotMatch(workflow, /manage_artifact_dns/) assert.match(workflow, /OFFLINE_ROLLBACK=true/) assert.match(workflow, /--runtime unavailable/) assert.equal(workflow.match(/--expected-image-digests/g)?.length, 7) diff --git a/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs b/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs index 1f0e6fce3ad..75bbb9ac8d5 100644 --- a/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs +++ b/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs @@ -111,7 +111,8 @@ test('recovers a failed or cancelled C4 refresh from an independent workflow', ( assert.match(recoveryWorkflow, /expected_digests="\$\{expected_digests\},\$\{current_digest\}"/) assert.match(recoveryWorkflow, /--timeout-ms 240000/) assert.doesNotMatch(recoveryWorkflow, /--timeout-ms 900000/) - assert.match(recoveryWorkflow, /-var manage_artifact_dns=false -lock-timeout=5m/) + assert.match(recoveryWorkflow, /-var-file=environments\/staging.tfvars -lock-timeout=5m/) + assert.doesNotMatch(recoveryWorkflow, /manage_artifact_dns/) assert.match(recoveryWorkflow, /--mode same-cap-image/) assert.match(recoveryWorkflow, /test "\$\(jq -r '\.changes'/) assert.equal(recoveryWorkflow.match(/\*:replacement-with-obsolete-template/g)?.length, 2) From 0f22e1e9051663a627bee78d2be91e44459fa5b2 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Thu, 3 Sep 2026 11:16:40 -0700 Subject: [PATCH 181/398] Fix table header transparency with opaque background (#18499) Replace the translucent bg-muted/25 with an opaque color-mix blend (40% muted on background) to ensure scrolled rows don't show through the sticky header. Add test coverage for header styling and layout. --- .../AutomationListTableHeader.test.tsx | 45 +++++++++++++++++++ src/renderer/src/lib/list-table-layout.ts | 4 +- 2 files changed, 47 insertions(+), 2 deletions(-) create mode 100644 src/renderer/src/components/automations/AutomationListTableHeader.test.tsx diff --git a/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx b/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx new file mode 100644 index 00000000000..5c5bbe8e568 --- /dev/null +++ b/src/renderer/src/components/automations/AutomationListTableHeader.test.tsx @@ -0,0 +1,45 @@ +// @vitest-environment happy-dom + +import { cleanup, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it } from 'vitest' +import { AutomationListTableHeader } from './AutomationListTableHeader' +import { + LIST_TABLE_HEADER_CLASS, + LIST_TABLE_STICKY_HEADER_CELL_CLASS +} from '@/lib/list-table-layout' + +describe('AutomationListTableHeader', () => { + afterEach(cleanup) + + it('renders all expected columns', () => { + render() + + expect(screen.getByText('Name')).toBeDefined() + expect(screen.getByText('Schedule')).toBeDefined() + expect(screen.getByText('Project')).toBeDefined() + expect(screen.getByText('Host')).toBeDefined() + expect(screen.getByText('Next run')).toBeDefined() + expect(screen.getByText('Last run')).toBeDefined() + expect(screen.getByText('Status')).toBeDefined() + expect(screen.getByText('Agent')).toBeDefined() + expect(screen.getByText('Actions')).toBeDefined() + }) + + it('uses opaque background and sticky positioning on the header row', () => { + const { container } = render() + const header = container.firstElementChild as HTMLElement + + expect(header.className).toContain(LIST_TABLE_HEADER_CLASS) + expect(header.className).toContain('sticky') + expect(header.className).toContain('top-0') + expect(header.className).toContain('bg-[color-mix(in_srgb,var(--muted)_40%,var(--background))]') + expect(header.className).not.toContain('bg-muted/25') + }) + + it('applies sticky cell styling to the first column', () => { + render() + const nameCell = screen.getByText('Name') + + expect(nameCell.className).toBe(LIST_TABLE_STICKY_HEADER_CELL_CLASS) + }) +}) diff --git a/src/renderer/src/lib/list-table-layout.ts b/src/renderer/src/lib/list-table-layout.ts index c40607ee944..89b85d0aab5 100644 --- a/src/renderer/src/lib/list-table-layout.ts +++ b/src/renderer/src/lib/list-table-layout.ts @@ -5,9 +5,9 @@ */ export const LIST_TABLE_CONTAINER_CLASS = 'rounded-md border border-border/50 bg-muted/20' -// Why: z-30 must beat the rows' sticky first cells (z-20) so the header still covers them. +// Why: z-30 and opaque wash ensure scrolled rows cannot show through the sticky header. export const LIST_TABLE_HEADER_CLASS = - 'sticky top-0 z-30 h-8 items-center gap-3 border-b border-border/50 bg-muted/25 px-3 text-[11px] font-medium uppercase tracking-[0.08em] text-muted-foreground' + 'sticky top-0 z-30 h-8 items-center gap-3 border-b border-border/50 bg-[color-mix(in_srgb,var(--muted)_40%,var(--background))] px-3 text-[11px] font-medium uppercase tracking-[0.08em] text-muted-foreground' // Why: keep keyboard-selected rows clear of the sticky table header. export const LIST_TABLE_ROW_CLASS = From 3cd817c250e41e91fa050e0e5ffe8c4c8757b032 Mon Sep 17 00:00:00 2001 From: "github-actions[bot]" <41898282+github-actions[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 18:31:46 +0000 Subject: [PATCH 182/398] Update README downloads badge --- docs/assets/readme-downloads.svg | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/assets/readme-downloads.svg b/docs/assets/readme-downloads.svg index 39fbcf45af1..ef8ebb61bb4 100644 --- a/docs/assets/readme-downloads.svg +++ b/docs/assets/readme-downloads.svg @@ -1,5 +1,5 @@ - - downloads: 37m + + downloads: 38m @@ -15,7 +15,7 @@ downloads downloads - 37m - 37m + 38m + 38m From fbea749d07adee7840f1bb5c8b9f33ddf90c908a Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 15:36:41 -0400 Subject: [PATCH 183/398] chore(cloud): pin staging relay c3 to the director's image (#18508) * chore(cloud): pin staging relay c3 to the director's image Mirrors stablyai/orca-cloud#468. c3 stayed on sha-c91439af after the director and c4 moved to sha-e3e92d95, so the staging capacity proof's compatible-director-image check has failed since 2026-08-14. * test(cloud): scope the launch-image pin to staging C4 now that C3 shares the digest * test(cloud): keep the public workflow assertions; scope only the launch-image pin to C4 --- .../relay-staging-c4-refresh-workflow.test.mjs | 11 +++++++---- cloud/infra/terraform/environments/staging.tfvars | 2 +- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs b/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs index 75bbb9ac8d5..0e777b06c38 100644 --- a/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs +++ b/cloud/dev/scripts/relay-staging-c4-refresh-workflow.test.mjs @@ -39,14 +39,17 @@ const launchDigest = '5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70ca // so a file-wide count no longer isolates Asia. const asiaCells = ['production-gce-c27', 'production-gce-c28', 'production-gce-c29'] -function productionCell(cellId) { - const start = productionTfvars.indexOf(`"${cellId}"`) +function cellBlock(tfvars, cellId) { + const start = tfvars.indexOf(`"${cellId}"`) assert.notEqual(start, -1, `${cellId} is missing`) - return productionTfvars.slice(start, productionTfvars.indexOf('\n }', start)) + return tfvars.slice(start, tfvars.indexOf('\n }', start)) } +const productionCell = (cellId) => cellBlock(productionTfvars, cellId) + +// Scoped to C4 by name: staging C3 serves this digest too since its 2026-09-03 re-pin. test('pins staging C4 and all production Asia cells to the same launch image', () => { - assert.equal(stagingTfvars.match(new RegExp(launchDigest, 'g'))?.length, 1) + assert.match(cellBlock(stagingTfvars, 'staging-gce-c4'), new RegExp(`relay@sha256:${launchDigest}"`)) for (const cellId of asiaCells) { assert.match(productionCell(cellId), new RegExp(`relay@sha256:${launchDigest}"`), cellId) } diff --git a/cloud/infra/terraform/environments/staging.tfvars b/cloud/infra/terraform/environments/staging.tfvars index 3ee108fe874..bb894744612 100644 --- a/cloud/infra/terraform/environments/staging.tfvars +++ b/cloud/infra/terraform/environments/staging.tfvars @@ -67,7 +67,7 @@ relay_gce_cells = { boot_disk_gb = 30 boot_image = "https://www.googleapis.com/compute/v1/projects/cos-cloud/global/images/cos-stable-121-18867-528-7" capacity_requests = 4000 - image = "us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:9fba2a189ab3fa29853800830e77c7551ab3aaa8f43f3cd9adbdea28b876a8b9" + image = "us-central1-docker.pkg.dev/onorca-cloud-staging/orca-cloud/relay@sha256:5aedbca5c86de24c8b4d4bf7e3b444b76c712f281ede916cb9d90f70cad1e563" initially_enabled = false connection_hard_cap = 1000 connection_unobserved_bound = 60 From 0746d82c019aa710c1ef19c78e70edb0d7ca7d63 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 15:57:00 -0400 Subject: [PATCH 184/398] chore(cloud): close the Workload Identity cutover onto stablyai/orca (#18509) Mirrors stablyai/orca-cloud#470. The private relay workflows are retired, so the dual accept has one live arm left. Add `github_workflow_file_prefix` for the primary repository's workflow filenames, point `github_repo`/ `github_repo_id` at `stablyai/orca` (`1183888342`), and empty `github_accepted_repositories` in both environments. Every relay provider goes back to a single arm naming `cloud-` prefixed workflow refs. `cloud/infra/terraform` stays byte-identical to the private branch. The two identity tests diverge here as they already did, so they take the same change rather than the same bytes: both now render the trusted ref head from the Terraform variable instead of this checkout's own workflow filenames, which is what lets the length pin be the same 791 characters in either repository. --- cloud/dev/scripts/relay-repository.mjs | 9 ++- cloud/dev/scripts/relay-repository.test.mjs | 3 + .../relay-staging-deploy-identity.test.mjs | 15 ++-- ...oad-identity-attribute-conditions.test.mjs | 69 +++++++++---------- cloud/infra/terraform/README.md | 43 +++++------- .../terraform/environments/production.tfvars | 18 ++--- .../terraform/environments/staging.tfvars | 18 ++--- cloud/infra/terraform/relay-shared.tf | 10 +-- cloud/infra/terraform/variables.tf | 28 +++++--- 9 files changed, 101 insertions(+), 112 deletions(-) diff --git a/cloud/dev/scripts/relay-repository.mjs b/cloud/dev/scripts/relay-repository.mjs index bf41bed8012..7e8b01e4799 100644 --- a/cloud/dev/scripts/relay-repository.mjs +++ b/cloud/dev/scripts/relay-repository.mjs @@ -15,9 +15,16 @@ export function relayWorkflowFile(name) { return `${RELAY_WORKFLOW_FILE_PREFIX}${name}` } +// Repository-relative path for a repository that renames its copies with `prefix`. Terraform's +// trusted prefix is a variable and need not be this checkout's, so callers rendering a +// workflow_ref from Terraform pass it in rather than assuming the local one. +export function prefixedRelayWorkflowPath(prefix, name) { + return `.github/workflows/${prefix}${name}` +} + // Repository-relative path, the shape GitHub reports in workflow_ref and evidence payloads. export function relayWorkflowPath(name) { - return `.github/workflows/${relayWorkflowFile(name)}` + return prefixedRelayWorkflowPath(RELAY_WORKFLOW_FILE_PREFIX, name) } export function relayWorkflowUrl(name) { diff --git a/cloud/dev/scripts/relay-repository.test.mjs b/cloud/dev/scripts/relay-repository.test.mjs index 56383db4a1e..cf33f869773 100644 --- a/cloud/dev/scripts/relay-repository.test.mjs +++ b/cloud/dev/scripts/relay-repository.test.mjs @@ -5,6 +5,7 @@ import { fileURLToPath } from 'node:url' import { RELAY_GITHUB_REPOSITORY, RELAY_WORKFLOW_FILE_PREFIX, + prefixedRelayWorkflowPath, readRelayWorkflow, relayWorkflowFile, relayWorkflowPath, @@ -21,6 +22,8 @@ test('workflow identity is derived, never restated', () => { assert.equal(relayWorkflowFile('deploy-relay-staging.yml'), `${RELAY_WORKFLOW_FILE_PREFIX}deploy-relay-staging.yml`) assert.equal(relayWorkflowPath('deploy-relay-staging.yml'), `.github/workflows/${relayWorkflowFile('deploy-relay-staging.yml')}`) assert.ok(relayWorkflowUrl('deploy-relay-staging.yml').pathname.endsWith(relayWorkflowPath('deploy-relay-staging.yml'))) + // A caller rendering Terraform's trusted ref supplies that prefix instead of this checkout's. + assert.equal(prefixedRelayWorkflowPath('cloud-', 'deploy-relay-staging.yml'), '.github/workflows/cloud-deploy-relay-staging.yml') assert.match(readRelayWorkflow('deploy-relay-staging.yml'), /^name:/m) assert.match(RELAY_GITHUB_REPOSITORY, /^[\w.-]+\/[\w.-]+$/) }) diff --git a/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs b/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs index fece62f9ea2..8afbfb5ca94 100644 --- a/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs +++ b/cloud/dev/scripts/relay-staging-deploy-identity.test.mjs @@ -2,11 +2,7 @@ import assert from 'node:assert/strict' import { readFileSync } from 'node:fs' import test from 'node:test' import { readWorkflow, workflowFiles } from './cloud-sql-rollout-lock-census.mjs' -import { - RELAY_WORKFLOW_FILE_PREFIX, - relayWorkflowFile, - relayWorkflowPath -} from './relay-repository.mjs' +import { prefixedRelayWorkflowPath, relayWorkflowFile } from './relay-repository.mjs' const identity = readFileSync( new URL('../../infra/terraform/relay-staging-deploy-iam.tf', import.meta.url), @@ -128,8 +124,11 @@ test('the rendered attribute condition stays inside the provider limit', () => { `assertion.repository_id == '${variableDefault('github_repo_id')}'`, `assertion.repository_owner_id == '${variableDefault('github_owner_id')}'` ] + // The prefix is the Terraform variable, not this checkout's own workflow filenames: the + // condition names the files as the trusted repository carries them. + const prefix = variableDefault('github_workflow_file_prefix') const workflowRefs = providerWorkflowFiles().map( - (file) => `${repository}/${relayWorkflowPath(file)}@refs/heads/main` + (file) => `${repository}/${prefixedRelayWorkflowPath(prefix, file)}@refs/heads/main` ) const rendered = [ ...claims, @@ -138,9 +137,7 @@ test('the rendered attribute condition stays inside the provider limit', () => { `(${workflowRefs.map((ref) => `assertion.workflow_ref == '${ref}'`).join(' || ')})` ].join(' && ') assert.ok(rendered.length < 4096, `rendered condition is ${rendered.length} characters`) - // 797 is the private repository's rendered length. This copy prefixes every workflow filename, - // which is the only difference, so the pin still moves the moment a workflow is added or dropped. - assert.equal(rendered.length, 797 + workflowRefs.length * RELAY_WORKFLOW_FILE_PREFIX.length) + assert.equal(rendered.length, 791) }) // Why: the census is the point. A binding added here without a workflow step behind it, or one diff --git a/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs b/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs index f25e442d8c0..1d3f3ce4d79 100644 --- a/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs +++ b/cloud/dev/scripts/workload-identity-attribute-conditions.test.mjs @@ -13,13 +13,13 @@ const EXPECTED_CONDITIONS = { staging: { relay: { github_staging_relay_capacity: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/prove-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/recover-relay-staging-c4-image.yml@refs/heads/main')) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-prove-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-recover-relay-staging-c4-image.yml@refs/heads/main')))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-prove-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-recover-relay-staging-c4-image.yml@refs/heads/main')", github_staging_relay_deploy: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-staging-gce-candidate.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/power-relay-staging.yml@refs/heads/main')) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-staging-gce-candidate.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-power-relay-staging.yml@refs/heads/main')))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-bootstrap-relay-staging-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-staging-gce-candidate.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-staging.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-power-relay-staging.yml@refs/heads/main')", github_relay_asia_topology: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && assertion.event_name == 'workflow_dispatch' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-asia-topology.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-asia-topology.yml@refs/heads/main'))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && assertion.event_name == 'workflow_dispatch' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-asia-topology.yml@refs/heads/main'", github_relay_asia_proof: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && assertion.event_name == 'workflow_dispatch' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/prove-relay-asia-staging.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-prove-relay-asia-staging.yml@refs/heads/main'))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'staging' && assertion.event_name == 'workflow_dispatch' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-prove-relay-asia-staging.yml@refs/heads/main'", }, // The relay root creates this provider only in production, so staging has exactly one // definition and it lives here. @@ -31,15 +31,15 @@ const EXPECTED_CONDITIONS = { production: { relay: { github: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-fence-broker.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-director.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-multi-target.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/publish-relay-production.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-production-rehome.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/operate-relay-production-rehome-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap.yml@refs/heads/main' && (assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap-job.yml@refs/heads/main' || assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap.yml@refs/heads/main')))) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-fence-broker.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-director.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-publish-relay-production.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && (assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main' || assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main')))))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-fence-broker.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-director.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-asia-admission.yml@refs/heads/main' || assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-publish-relay-production.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-operate-relay-production-rehome-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && (assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main' || assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main')))", github_monitor: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/monitor-relay-production.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/monitor-relay-production-job.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-monitor-relay-production.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-monitor-relay-production-job.yml@refs/heads/main'))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-monitor-relay-production.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-monitor-relay-production-job.yml@refs/heads/main'", github_fence: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-multi-target.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-multi-target.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main'))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-multi-target.yml@refs/heads/main'", github_production_relay_capacity: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-capacity.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-capacity-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-production-same-cap-job.yml@refs/heads/main'))) || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main'))))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && ((assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-capacity-job.yml@refs/heads/main') || (assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap.yml@refs/heads/main' && assertion.job_workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-production-same-cap-job.yml@refs/heads/main'))", github_relay_asia_topology: - "assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && assertion.event_name == 'workflow_dispatch' && ((assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca-cloud/.github/workflows/deploy-relay-asia-topology.yml@refs/heads/main') || (assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-asia-topology.yml@refs/heads/main'))", + "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420' && assertion.ref == 'refs/heads/main' && assertion.environment == 'production' && assertion.event_name == 'workflow_dispatch' && assertion.workflow_ref == 'stablyai/orca/.github/workflows/cloud-deploy-relay-asia-topology.yml@refs/heads/main'", }, apps: { github_production_app_deploy: @@ -48,20 +48,21 @@ const EXPECTED_CONDITIONS = { }, } -// Every repository the relay root accepts while the public extraction runs, with the workflow-ref -// head each one contributes. The apps root is not part of the dual accept. -const ACCEPTED_REPOSITORIES = [ - { - claims: - "assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420'", - workflowHead: 'stablyai/orca-cloud/.github/workflows/' - }, - { +// The one repository each root trusts, with the workflow-ref head it contributes. The relay root +// moved to the public repository, where the workflow files carry the `cloud-` prefix; the apps +// root still deploys from the private one. +const ROOT_REPOSITORIES = { + relay: { claims: "assertion.repository == 'stablyai/orca' && assertion.repository_id == '1183888342' && assertion.repository_owner_id == '127256420'", workflowHead: 'stablyai/orca/.github/workflows/cloud-' + }, + apps: { + claims: + "assertion.repository == 'stablyai/orca-cloud' && assertion.repository_id == '1273841466' && assertion.repository_owner_id == '127256420'", + workflowHead: 'stablyai/orca-cloud/.github/workflows/' } -] +} // [root, provider, condition] for every provider the environment creates, across all roots. async function flatten(environment) { @@ -103,9 +104,7 @@ for (const environment of Object.keys(EXPECTED_CONDITIONS)) { test(`${environment} pins repository, branch, and environment on every provider`, async () => { for (const [root, provider, condition] of await flatten(environment)) { for (const pin of [ - "assertion.repository == 'stablyai/orca-cloud'", - "assertion.repository_id == '1273841466'", - "assertion.repository_owner_id == '127256420'", + ROOT_REPOSITORIES[root].claims, "assertion.ref == 'refs/heads/main'", `assertion.environment == '${environment}'` ]) { @@ -131,27 +130,23 @@ for (const environment of Object.keys(EXPECTED_CONDITIONS)) { }) } -// Why: the dual accept is only safe if each OR arm carries its own repository claims. An arm that -// inherited them, or a workflow ref that named the other repository, would let one repository's -// workflows run under the other's proof. +// Why: the cutover left one arm per relay provider. A leftover `stablyai/orca-cloud` claim or +// workflow ref would keep trusting a repository whose relay workflows are retired, and an unprefixed +// ref would name a file the public repository does not have. for (const environment of Object.keys(EXPECTED_CONDITIONS)) { - test(`${environment} admits both repositories through every relay provider`, async () => { + test(`${environment} admits only the public repository through every relay provider`, async () => { + const { claims, workflowHead } = ROOT_REPOSITORIES.relay const rendered = await renderAttributeConditions(environment) for (const [provider, condition] of Object.entries(rendered.relay)) { - assert.ok( - condition.startsWith("assertion.ref == 'refs/heads/main' && "), - `${provider} does not lead with the repository-independent claims` - ) + assert.ok(condition.startsWith(`${claims} && `), `${provider} does not lead with the claims`) + assert.doesNotMatch(condition, /stablyai\/orca-cloud|1273841466/, `${provider} keeps an old arm`) const refs = [...condition.matchAll(/(?:job_)?workflow_ref == '([^']+)'/g)].map( (match) => match[1] ) - const perRepository = ACCEPTED_REPOSITORIES.map((repository) => { - assert.ok(condition.includes(`(${repository.claims} && `), `${provider} misses an arm`) - return refs.filter((ref) => ref.startsWith(repository.workflowHead)).length - }) - assert.equal(refs.length, perRepository[0] + perRepository[1], `${provider} names a stray ref`) - assert.equal(perRepository[0], perRepository[1], `${provider} arms are not the same size`) - assert.ok(perRepository[0] > 0, `${provider} names no workflow`) + assert.ok(refs.length > 0, `${provider} names no workflow`) + for (const ref of refs) { + assert.ok(ref.startsWith(workflowHead), `${provider} names a stray ref ${ref}`) + } } }) } diff --git a/cloud/infra/terraform/README.md b/cloud/infra/terraform/README.md index 2e2b9b02011..8aa7aa0a95c 100644 --- a/cloud/infra/terraform/README.md +++ b/cloud/infra/terraform/README.md @@ -37,35 +37,26 @@ identity (`google_service_account.github_deploy`, its provider, and its bindings with production-only counts; staging's copies are declared by `infra/terraform-apps`. An untargeted plan is orderable again; the `Plan:` line still reflects the standing cell-template drift backlog. -### Dual-accept Workload Identity during the public extraction +### Workload Identity trusts the public repository -While the relay source moves to the public `stablyai/orca` repository, every relay Workload -Identity provider accepts the same workflows from both repositories. `github_accepted_repositories` -lists the extra repositories; `relay-github-workflow-trust.tf` renders one parenthesised OR arm per -accepted repository, each arm carrying that repository's own `repository`, `repository_id`, and -`repository_owner_id` claims plus its exact workflow refs. `ref`, `environment`, and `event_name` -stay outside the OR. Workflow files keep their names in the private repo and take the -`workflow_file_prefix` (`cloud-`) in the public one. +The cutover closed on 2026-09-03. Every relay Workload Identity provider now accepts exactly one +repository, `stablyai/orca` (`1183888342`, owner `127256420`), and every workflow ref it names is +built from `github_workflow_file_prefix` (`cloud-`), which is the rename the public repo applies to +the workflow files it carries. `github_repo`, `github_repo_id`, and that prefix are set in both +`environments/*.tfvars` as well as defaulted here, and `github_accepted_repositories` is empty. +Nothing in this root trusts `stablyai/orca-cloud` any more; the apps and foundation roots still do, +because the app workflows still live there. -Adding a repository is a tfvars edit: no provider block changes, and the rendered strings are -pinned by `dev/scripts/workload-identity-attribute-conditions.test.mjs`. An empty list renders -byte-identically to the single-repository form, which is what makes the arms reviewable against -the pre-extraction condition. +`github_accepted_repositories` stays available for the next repository move. Each entry renders its +own parenthesised OR arm in `relay-github-workflow-trust.tf`, carrying that repository's own +`repository`, `repository_id`, and `repository_owner_id` claims plus its exact workflow refs, while +`ref`, `environment`, and `event_name` stay outside the OR. An empty list renders byte-identically +to the single-repository form, so adding and removing a repository is a tfvars edit with no provider +block change. The rendered strings are pinned by +`dev/scripts/workload-identity-attribute-conditions.test.mjs`. -Closing the cutover is an owner step, in this order: - -1. Retire the private workflows, so nothing runs from `stablyai/orca-cloud` any more. -2. Point `github_owner`, `github_repo`, `github_repo_id`, and `github_owner_id` at - `stablyai/orca` (`1183888342`, owner `127256420`), and set `workflow_file_prefix` for it by - moving the surviving entry's prefix onto the primary: the public files keep the `cloud-` names, - so the primary prefix becomes `cloud-` unless the files are renamed back. -3. Empty `github_accepted_repositories` in both `environments/*.tfvars`. -4. Re-render and update the pinned conditions, then apply. Each provider goes back to a single - arm, and `google_service_account_iam_member.github_accepted_repository_workload_identity_user` - is destroyed as the primary `attribute.repository` binding takes over. - -Step 2 and step 3 must land in the same apply: dropping the accepted entry before repointing the -primary would revoke the public repository mid-flight. +Repointing the primary and emptying the list must land in the same apply: dropping the accepted +entry before repointing the primary would revoke the surviving repository mid-flight. ### `ORCA_RELAY_IMAGE_DIGEST` is not Terraform-owned diff --git a/cloud/infra/terraform/environments/production.tfvars b/cloud/infra/terraform/environments/production.tfvars index 60d36e3d832..8e442c75900 100644 --- a/cloud/infra/terraform/environments/production.tfvars +++ b/cloud/infra/terraform/environments/production.tfvars @@ -5,19 +5,11 @@ region = "us-central1" artifact_repository_id = "orca-cloud" -# Dual accept while the relay source moves to the public stablyai/orca repository: the same -# workflows are trusted from both repos, and the public copies carry a `cloud-` file prefix. -# Remove this entry once the private workflows are retired and point github_owner/github_repo, -# github_repo_id, and github_owner_id at the surviving repository. -github_accepted_repositories = [ - { - owner = "stablyai" - repo = "orca" - repo_id = "1183888342" - owner_id = "127256420" - workflow_file_prefix = "cloud-" - } -] +# The relay source lives in the public stablyai/orca repository, where the workflows carry a +# `cloud-` file prefix. github_owner and github_owner_id keep their defaults. +github_repo = "orca" +github_repo_id = "1183888342" +github_workflow_file_prefix = "cloud-" # Our first-party auth service. auth.onorca.dev is PropelAuth's prod domain, so # our service lives at login.onorca.dev (desktop points ORCA_CLOUD_API_URL here). diff --git a/cloud/infra/terraform/environments/staging.tfvars b/cloud/infra/terraform/environments/staging.tfvars index bb894744612..4a32458fcd5 100644 --- a/cloud/infra/terraform/environments/staging.tfvars +++ b/cloud/infra/terraform/environments/staging.tfvars @@ -5,19 +5,11 @@ region = "us-central1" artifact_repository_id = "orca-cloud" -# Dual accept while the relay source moves to the public stablyai/orca repository: the same -# workflows are trusted from both repos, and the public copies carry a `cloud-` file prefix. -# Remove this entry once the private workflows are retired and point github_owner/github_repo, -# github_repo_id, and github_owner_id at the surviving repository. -github_accepted_repositories = [ - { - owner = "stablyai" - repo = "orca" - repo_id = "1183888342" - owner_id = "127256420" - workflow_file_prefix = "cloud-" - } -] +# The relay source lives in the public stablyai/orca repository, where the workflows carry a +# `cloud-` file prefix. github_owner and github_owner_id keep their defaults. +github_repo = "orca" +github_repo_id = "1183888342" +github_workflow_file_prefix = "cloud-" auth_base_url = "https://auth-staging.onorca.dev" diff --git a/cloud/infra/terraform/relay-shared.tf b/cloud/infra/terraform/relay-shared.tf index d4f0e5dac15..b1afc4d1890 100644 --- a/cloud/infra/terraform/relay-shared.tf +++ b/cloud/infra/terraform/relay-shared.tf @@ -14,16 +14,16 @@ locals { "assertion.repository_owner_id == '${var.github_owner_id}'", ] - # Dual accept during the public extraction: the primary repository first, then every repository - # var.github_accepted_repositories adds. Each one renders its own OR arm in every provider - # condition, so both repos can run the same workflows through the same identities. A repository - # that imports these workflows may rename the files, hence the per-repository prefix. + # The primary repository first, then every repository var.github_accepted_repositories adds. + # Each one renders its own OR arm in every provider condition, so a repository move can trust + # both repos at once. A repository that imports these workflows may rename the files, hence the + # per-repository prefix; the primary's is var.github_workflow_file_prefix. relay_github_accepted_repositories = concat([{ owner = var.github_owner repo = var.github_repo repo_id = var.github_repo_id owner_id = var.github_owner_id - workflow_file_prefix = "" + workflow_file_prefix = var.github_workflow_file_prefix }], var.github_accepted_repositories) relay_github_single_repository = length(local.relay_github_accepted_repositories) == 1 diff --git a/cloud/infra/terraform/variables.tf b/cloud/infra/terraform/variables.tf index c3e06ee280f..68f6c555bd3 100644 --- a/cloud/infra/terraform/variables.tf +++ b/cloud/infra/terraform/variables.tf @@ -22,14 +22,14 @@ variable "github_owner" { variable "github_repo" { type = string description = "GitHub repo allowed to deploy through Workload Identity Federation." - default = "orca-cloud" + default = "orca" } # Numeric IDs survive a rename or transfer of the repository; every provider pins them next to the name. variable "github_repo_id" { type = string description = "Numeric GitHub repository ID of github_owner/github_repo." - default = "1273841466" + default = "1183888342" validation { condition = can(regex("^[0-9]+$", var.github_repo_id)) @@ -48,12 +48,24 @@ variable "github_owner_id" { } } -# Additional repositories whose identical workflows the same identities must accept while the -# public extraction runs. Each entry renders its own OR arm in every provider condition, so the -# private repo keeps working while the public one takes over. `workflow_file_prefix` is the rename -# the importing repository applies to the workflow files it copies. Empty is the steady state: -# the final step of the cutover is to empty this list again and point github_owner/github_repo, -# github_repo_id, and github_owner_id at the surviving repository. +# The rename the relay repository applies to the workflow files it carries. The public repo keeps +# the workflows under `cloud-` names, so every relay workflow_ref is built from this head. +variable "github_workflow_file_prefix" { + type = string + description = "Filename prefix on github_owner/github_repo's copies of the relay workflows." + default = "cloud-" + + validation { + condition = can(regex("^[a-z0-9-]*$", var.github_workflow_file_prefix)) + error_message = "github_workflow_file_prefix must be lowercase letters, digits, or hyphens." + } +} + +# Additional repositories whose identical workflows the same identities must accept during a +# repository move. Each entry renders its own OR arm in every provider condition, so both repos +# can run the same workflows through the same identities. `workflow_file_prefix` is the rename the +# importing repository applies to the workflow files it copies. Empty is the steady state, and is +# where the public extraction left it: stablyai/orca is now the primary and only repository. variable "github_accepted_repositories" { type = list(object({ owner = string From 16e26245783744e232fb06695b2b8bb8844312b9 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:02:46 -0400 Subject: [PATCH 185/398] fix(terminal): flush xterm's parked renderer resize when releasing the pause latch (#18510) --- ...render-pause-release-parked-resize.test.ts | 113 ++++++++++++++++++ .../terminal-render-pause-release.ts | 36 ++++-- 2 files changed, 140 insertions(+), 9 deletions(-) create mode 100644 src/renderer/src/lib/pane-manager/terminal-render-pause-release-parked-resize.test.ts diff --git a/src/renderer/src/lib/pane-manager/terminal-render-pause-release-parked-resize.test.ts b/src/renderer/src/lib/pane-manager/terminal-render-pause-release-parked-resize.test.ts new file mode 100644 index 00000000000..7ad57fa7126 --- /dev/null +++ b/src/renderer/src/lib/pane-manager/terminal-render-pause-release-parked-resize.test.ts @@ -0,0 +1,113 @@ +import { describe, expect, it, vi } from 'vitest' +import { + forceFullViewportPresent, + forceRepaintThroughRenderPause, + requestFullViewportPresent +} from './terminal-render-pause-release' + +// Why a separate file: the parked-resize contract is one hazard shared by all +// three helpers, and the main spec is already at the max-lines budget. + +type FakeRenderService = { + _isPaused: boolean + _needsFullRefresh: boolean + _pausedResizeTask?: { flush: ReturnType } | null + refreshRows: ReturnType + _renderer?: { value?: { renderRows?: ReturnType } } +} + +function createPausedTerminal(options: { + synchronizedOutput?: boolean + withoutTask?: boolean + flushThrows?: boolean +}): { terminal: unknown; service: FakeRenderService; order: string[] } { + const order: string[] = [] + const flush = vi.fn(() => { + order.push('flush') + if (options.flushThrows) { + throw new Error('renderer disposed') + } + }) + const service: FakeRenderService = { + _isPaused: true, + _needsFullRefresh: true, + _pausedResizeTask: options.withoutTask ? null : { flush }, + refreshRows: vi.fn(() => order.push('refreshRows')), + _renderer: { value: { renderRows: vi.fn(() => order.push('renderRows')) } } + } + const terminal = { + rows: 24, + _core: { + _renderService: service, + coreService: { decPrivateModes: { synchronizedOutput: options.synchronizedOutput === true } } + } + } + return { terminal, service, order } +} + +const helpers = [ + ['forceRepaintThroughRenderPause', forceRepaintThroughRenderPause], + ['requestFullViewportPresent', requestFullViewportPresent], + ['forceFullViewportPresent', forceFullViewportPresent] +] as const + +describe.each(helpers)('%s parked renderer resize', (_name, present) => { + it('flushes the resize xterm parked while paused before presenting', () => { + // A resize that lands under _isPaused only parks WebglRenderer.handleResize; + // xterm flushes it solely from the observer callback we are pre-empting. + const { terminal, service, order } = createPausedTerminal({}) + + expect(present(terminal)).toBe(true) + expect(service._pausedResizeTask?.flush).toHaveBeenCalledTimes(1) + expect(order[0]).toBe('flush') + expect(order).toHaveLength(2) + expect(service._isPaused).toBe(false) + expect(service._needsFullRefresh).toBe(false) + }) + + it('flushes before a DEC 2026 present too', () => { + const { terminal, service, order } = createPausedTerminal({ synchronizedOutput: true }) + + expect(present(terminal)).toBe(true) + expect(service._pausedResizeTask?.flush).toHaveBeenCalledTimes(1) + expect(order[0]).toBe('flush') + }) + + it('still presents when the parked-task internal is unavailable', () => { + const { terminal, service } = createPausedTerminal({ withoutTask: true }) + + expect(present(terminal)).toBe(true) + expect(service._isPaused).toBe(false) + }) + + it('still presents when the parked resize throws', () => { + const { terminal, order } = createPausedTerminal({ flushThrows: true }) + + expect(present(terminal)).toBe(true) + expect(order).toEqual(['flush', expect.any(String)]) + }) +}) + +describe('parked renderer resize on an unpaused terminal', () => { + it('is left to xterm when the pause latch is not set', () => { + const flush = vi.fn() + const service = { + _isPaused: false, + _needsFullRefresh: false, + _pausedResizeTask: { flush }, + refreshRows: vi.fn() + } + const terminal = { + rows: 24, + _core: { + _renderService: service, + coreService: { decPrivateModes: { synchronizedOutput: true } } + } + } + + expect(requestFullViewportPresent(terminal)).toBe(true) + expect(forceFullViewportPresent(terminal)).toBe(true) + expect(forceRepaintThroughRenderPause(terminal)).toBe(false) + expect(flush).not.toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/lib/pane-manager/terminal-render-pause-release.ts b/src/renderer/src/lib/pane-manager/terminal-render-pause-release.ts index 552939e6cd8..9f823e84630 100644 --- a/src/renderer/src/lib/pane-manager/terminal-render-pause-release.ts +++ b/src/renderer/src/lib/pane-manager/terminal-render-pause-release.ts @@ -24,6 +24,7 @@ type MaybeWebglRenderer = { type MaybePausableRenderService = { _isPaused?: boolean _needsFullRefresh?: boolean + _pausedResizeTask?: { flush?: () => void } | null refreshRows?: (start: number, end: number, sync?: boolean) => void _renderer?: { value?: MaybeWebglRenderer | null } | MaybeWebglRenderer | null } @@ -41,6 +42,29 @@ type TerminalWithRenderService = { } } +/** + * Clears xterm's observer-pause latches and runs the renderer resize xterm parked + * while paused. + * + * Why the flush: `RenderService.handleResize` under `_isPaused` only parks the + * WebGL renderer's own resize on an idle task, and xterm flushes that task solely + * from the observer callback gated on `_needsFullRefresh`. Clearing the latch + * without flushing lets the present below paint the new grid through the old + * canvas/model geometry (misplaced fragments, stray bars until a user resize). + */ +function releaseRenderPause(service: PausableRenderService): void { + // Why: leave the latch as if the pending full refresh was serviced — we are + // about to service it — so the observer's next callback doesn't queue a + // redundant second full repaint. + service._isPaused = false + service._needsFullRefresh = false + try { + service._pausedResizeTask?.flush?.() + } catch { + // Why: a resize that throws mid-dispose must not block the present. + } +} + function getRenderService(terminal: unknown): PausableRenderService | null { const service = (terminal as TerminalWithRenderService | null)?._core?._renderService return service && typeof service.refreshRows === 'function' @@ -66,11 +90,7 @@ export function forceRepaintThroughRenderPause(terminal: unknown): boolean { return false } - // Why: leave the latch as if the pending full refresh was serviced — we are - // about to service it — so the observer's next callback doesn't queue a - // redundant second full repaint. - service._isPaused = false - service._needsFullRefresh = false + releaseRenderPause(service) try { service.refreshRows(0, rows - 1, true) return true @@ -102,8 +122,7 @@ export function requestFullViewportPresent(terminal: unknown): boolean { } if (paused) { - service._isPaused = false - service._needsFullRefresh = false + releaseRenderPause(service) } try { @@ -160,8 +179,7 @@ export function forceFullViewportPresent(terminal: unknown): boolean { } if (paused) { - service._isPaused = false - service._needsFullRefresh = false + releaseRenderPause(service) } const renderer = getRenderer(service) From 91c5a615c5dd46568d232d32fa72c1dc13bfb220 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Thu, 3 Sep 2026 13:22:42 -0700 Subject: [PATCH 186/398] fix(settings): indent the Agent sleep "Sleep after" sub-setting (#18379) "Sleep after" rendered flush with its parent toggle, unlike the Agent Dashboard and Chat UI sub-settings which sit inside the indented, left-bordered group. Reuse that same wrapper and drop the row's extra vertical padding so the block matches its siblings. Co-authored-by: Merge Sim --- .../components/settings/ExperimentalPane.tsx | 51 ++++++++++--------- .../settings/SettingsFormControls.tsx | 5 +- 2 files changed, 31 insertions(+), 25 deletions(-) diff --git a/src/renderer/src/components/settings/ExperimentalPane.tsx b/src/renderer/src/components/settings/ExperimentalPane.tsx index 2ef34d943c8..e755db7e41b 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.tsx @@ -190,30 +190,33 @@ export function ExperimentalPane({ />
    {agentHibernationEnabled ? ( - - updateSettings({ - // Why: settings persist the planner contract, not the display unit. - agentHibernationIdleMs: minutes * MS_PER_MINUTE - }) - } - /> +
    + + updateSettings({ + // Why: settings persist the planner contract, not the display unit. + agentHibernationIdleMs: minutes * MS_PER_MINUTE + }) + } + /> +
    ) : null} ) : null} diff --git a/src/renderer/src/components/settings/SettingsFormControls.tsx b/src/renderer/src/components/settings/SettingsFormControls.tsx index bb38bcc8d4e..9a0993849f6 100644 --- a/src/renderer/src/components/settings/SettingsFormControls.tsx +++ b/src/renderer/src/components/settings/SettingsFormControls.tsx @@ -270,6 +270,7 @@ type NumberFieldProps = { integer?: boolean onChange: (value: number) => void suffix?: string + className?: string } export function ColorField({ @@ -315,7 +316,8 @@ export function NumberField({ step = 1, integer = false, onChange, - suffix + suffix, + className }: NumberFieldProps): React.JSX.Element { const [draft, setDraft] = useState(Number.isFinite(value) ? String(value) : '') const [prevValue, setPrevValue] = useState(value) @@ -346,6 +348,7 @@ export function NumberField({ return ( From a35451f5b91e693ef05cbc80dcc94c75518c6e85 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:37:42 -0400 Subject: [PATCH 187/398] fix(relay): stop self-closing the control socket on unknown messages (#18400) * fix(relay): stop self-closing the control socket on unknown messages The desktop control client tore its own relay control WebSocket down with code 4401 "unknown control message" for any well-formed control frame it did not recognize. handleMessage() funneled everything that was not ping / conn-open / drain / a tracked request reply into failProtocol('unknown control message'), which closes the socket and orphans the origin. Three real frames hit that branch: - A relay reply that arrives after the desktop's 10s request deadline already deleted the pending entry. Relay control operations run DB transactions that can exceed 10s under load, so resolveMessage() finds no waiter and returns false. - A control-error carrying no reqId (or an unknown one), including the relay's own 'unknown_control_message' reply to a host command it could not route. - A newer relay's opcode that this build predates. Fleet telemetry shows ~15 of these closes per day across app versions 1.4.175..1.4.197, so it is version-agnostic. The self-close was also far more costly than the message that caused it: the relay session dropped to 'orphaned' and answered the phone with HOST_OFFLINE (4404) for the orphan grace window, then the desktop had to re-register through the director's 503 reconnect throttle, stretching a single stray frame into minutes of mobile downtime. Per docs/reference/remote-wire-compatibility.md Rule 2, an unknown but well-formed control frame must be dropped, not treated as fatal. Log and ignore it; malformed JSON, binary frames, and messages before activation still close as protocol violations. Adds unit tests for the unknown-opcode drop, the timed-out-reply drop, and the preserved malformed-frame teardown. * docs(relay): correct the ignore rationale, drop the Rule 2 misattribution Rule 2 of remote-wire-compatibility governs the SENDER of a new terminal- stream opcode and treats the receiver's silent drop as a hazard, not a mandate. Reframe the comment around the actual justification: the decoder convention of dropping unknown frames, the control channel's lack of an opcode negotiation step, and the incident cost asymmetry. --- .../relay/relay-control-client.test.ts | 45 +++++++++++++++++++ .../runtime/relay/relay-control-client.ts | 13 +++++- 2 files changed, 57 insertions(+), 1 deletion(-) diff --git a/src/main/runtime/relay/relay-control-client.test.ts b/src/main/runtime/relay/relay-control-client.test.ts index 2975b67e631..d235f0ebacd 100644 --- a/src/main/runtime/relay/relay-control-client.test.ts +++ b/src/main/runtime/relay/relay-control-client.test.ts @@ -4,6 +4,7 @@ import { afterEach, describe, expect, it, vi } from 'vitest' import nacl from 'tweetnacl' import { WebSocketServer, type WebSocket } from 'ws' import type { E2EEKeypair } from '../e2ee-keypair' +import { MOBILE_RELAY_CLOSE_CODE } from '../../../shared/mobile-relay-close-codes' import { RelayControlClient } from './relay-control-client' const encoder = new TextEncoder() @@ -550,4 +551,48 @@ describe('RelayControlClient scripted-socket lifecycle', () => { vi.advanceTimersByTime(91_000) expect(client.isLive()).toBe(false) }) + + it('ignores an unrecognized control message without closing the active control', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const { client, socket, onClose } = scriptedControl() + await client.connect() + expect(client.isLive()).toBe(true) + + // A newer relay opcode the desktop schema does not know. Rule 2 of + // remote-wire-compatibility: an unknown-but-well-formed frame is dropped, + // never fatal to a live control. + socket.deliver({ type: 'relay-hint', v: 2, hint: 'future-feature' }) + + expect(client.isLive()).toBe(true) + expect(socket.readyState).toBe(1) + expect(onClose).not.toHaveBeenCalled() + warn.mockRestore() + }) + + it('ignores a reply whose request already timed out instead of self-closing', async () => { + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + const { client, socket, onClose } = scriptedControl() + await client.connect() + + // A relay control-error carrying a reqId with no live waiter — e.g. a late + // reply that arrived after the desktop's request deadline deleted it, or the + // relay's no-op error for a command it could not route. Must not be fatal. + socket.deliver({ type: 'control-error', reqId: 'expired-req', code: 'unknown_control_message' }) + + expect(client.isLive()).toBe(true) + expect(socket.readyState).toBe(1) + expect(onClose).not.toHaveBeenCalled() + warn.mockRestore() + }) + + it('still tears down a malformed (non-JSON) control frame', async () => { + const { client, socket, onClose } = scriptedControl() + await client.connect() + + socket.emit('message', 'not-json{', false) + + expect(client.isLive()).toBe(false) + expect(socket.readyState).toBe(3) + expect(onClose).toHaveBeenCalledWith(MOBILE_RELAY_CLOSE_CODE.BAD_OUTER_CREDENTIAL) + }) }) diff --git a/src/main/runtime/relay/relay-control-client.ts b/src/main/runtime/relay/relay-control-client.ts index 76816c81a3a..7e742173f72 100644 --- a/src/main/runtime/relay/relay-control-client.ts +++ b/src/main/runtime/relay/relay-control-client.ts @@ -234,7 +234,18 @@ export class RelayControlClient { if (this.requests.resolveMessage(message)) { return } - this.failProtocol('unknown control message') + // Drop a well-formed control message we do not recognize, matching how every + // other Orca decoder treats an unknown frame (see the silent-drop convention + // in docs/reference/remote-wire-compatibility.md). The control channel has no + // opcode negotiation step, so this reaches either a newer relay's message + // this build predates, or a reply whose request already timed out and has no + // waiter (relay control ops run DB transactions that can exceed the request + // deadline under load). Self-closing here was strictly worse than ignoring: + // it orphaned the relay session, which answered the phone with HOST_OFFLINE + // for the orphan-grace window plus the director's reconnect throttle — minutes + // of outage from a single stray frame. + const messageType = typeof message.type === 'string' ? message.type : 'unknown' + console.warn(`[relay] ignoring unrecognized control message type=${messageType}`) } private handleProofMessage(message: Record): void { From aa78d4af17c19549ab003738c2f5fae117e7e6a2 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:53:32 -0400 Subject: [PATCH 188/398] fix(release): restore version and harden staging confirmation Resolves release scan blockers STA-6611 and STA-6612. --- .../cloud-prove-relay-asia-staging.yml | 4 ++- config/scripts/release-blocker-fixes.test.mjs | 35 +++++++++++++++++++ package.json | 2 +- 3 files changed, 39 insertions(+), 2 deletions(-) create mode 100644 config/scripts/release-blocker-fixes.test.mjs diff --git a/.github/workflows/cloud-prove-relay-asia-staging.yml b/.github/workflows/cloud-prove-relay-asia-staging.yml index 9a66e967b57..56677a98600 100644 --- a/.github/workflows/cloud-prove-relay-asia-staging.yml +++ b/.github/workflows/cloud-prove-relay-asia-staging.yml @@ -55,9 +55,11 @@ jobs: - name: Validate the exact staging proof request shell: bash + env: + CONFIRMATION: ${{ inputs.confirmation }} run: | set -euo pipefail - test "${{ inputs.confirmation }}" = PROVE_ASIA_STAGING + test "${CONFIRMATION}" = PROVE_ASIA_STAGING [[ "${IMAGE_DIGEST}" =~ ^sha256:[0-9a-f]{64}$ ]] [[ "${INITIAL_SELECTOR_GENERATION}" =~ ^[1-9][0-9]*$ ]] [[ "${PROMOTE_ATTEMPT_ID}" =~ ^[A-Za-z0-9_-]{8,128}$ ]] diff --git a/config/scripts/release-blocker-fixes.test.mjs b/config/scripts/release-blocker-fixes.test.mjs new file mode 100644 index 00000000000..bccd631a266 --- /dev/null +++ b/config/scripts/release-blocker-fixes.test.mjs @@ -0,0 +1,35 @@ +import { readFileSync } from 'node:fs' +import { resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { parse } from 'yaml' + +const projectDir = resolve(import.meta.dirname, '../..') + +describe('release blocker safeguards', () => { + it('keeps the root package version on the current stable release line', () => { + const packageJson = JSON.parse(readFileSync(resolve(projectDir, 'package.json'), 'utf8')) + const match = /^(\d+)\.(\d+)\.(\d+)(?:-[0-9A-Za-z.-]+)?$/.exec(packageJson.version) + expect(match).not.toBeNull() + const version = match.slice(1, 4).map(Number) + const isAtLeastStable = + version[0] > 1 || + (version[0] === 1 && (version[1] > 4 || (version[1] === 4 && version[2] >= 196))) + expect(isAtLeastStable).toBe(true) + }) + + it('passes the staging confirmation through the step environment', () => { + const workflow = parse( + readFileSync( + resolve(projectDir, '.github/workflows/cloud-prove-relay-asia-staging.yml'), + 'utf8' + ) + ) + const step = workflow.jobs.prove.steps.find( + ({ name }) => name === 'Validate the exact staging proof request' + ) + + expect(step.env.CONFIRMATION).toBe('${{ inputs.confirmation }}') + expect(step.run).toContain('test "${CONFIRMATION}" = PROVE_ASIA_STAGING') + expect(step.run).not.toContain('${{ inputs.confirmation }}') + }) +}) diff --git a/package.json b/package.json index 179ffd1a82a..58f4805d937 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "orca", - "version": "1.4.178-rc.2", + "version": "1.4.197", "description": "Next-gen IDE for parallel agentic development", "homepage": "https://github.com/stablyai/orca", "author": "stablyai", From 4d24fb340b6fd6596fac5ed37d25fe875f4d77bd Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 17:16:35 -0400 Subject: [PATCH 189/398] fix(mobile): stage-aware relay dial bound so a slow cell is not hung up on (#18518) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A phone returning to foreground on 2026-09-03 logged "replacement session authentication timed out" five dials in a row while the desktop's relay control was live. The cell (production-gce-c27) had taken relay-auth but its assignment/reservation transactions were lock-contended (55P03 retries, 14–16s per accept); the phone's flat 12s migrateTo bound closed the socket 2–4s before the cell finished (cell logged host_data_reservation_already_bound), and because the timeout counted as a director-class failure the phone re-resolved the same cell and waited 12s again before logging — every retry landed in the same contended window. - MobileRelayE2eeLink reports onOpen once relay-auth is on the wire; MobileRelayRpcSession exposes a dial stage (opening → awaiting-hello → handshaking → confirming). - waitForAuthenticated keeps the caller's bound until the socket opens, then re-arms a per-stage budget (30s awaiting-hello, 12s handshaking, 35s confirming) so a reachable, slow cell is not treated as a black hole. - The timeout error carries the stalled stage and shows up in the "relay dial failed" log line; a stall past the open socket no longer triggers the director re-resolve round. Phone-local only: no wire change. Claude-Session: https://claude.ai/code/session_01JNnE9qzUZMMnqpZWCqM3nb --- ...e-endpoint-supervisor-stalled-cell.test.ts | 80 +++++++++++ .../mobile-endpoint-supervisor-support.ts | 6 + .../mobile-endpoint-supervisor-test-fakes.ts | 5 + .../transport/mobile-relay-e2ee-link.test.ts | 55 ++++++++ .../src/transport/mobile-relay-e2ee-link.ts | 4 + .../mobile-relay-rpc-session.test.ts | 30 +++++ .../src/transport/mobile-relay-rpc-session.ts | 24 ++-- .../mobile-relay-runtime-failover.test.ts | 5 + mobile/src/transport/relay-dial-stage.ts | 64 +++++++++ ...replacement-session-authentication.test.ts | 127 ++++++++++++++++++ .../replacement-session-authentication.ts | 58 +++++++- .../stable-logical-rpc-client.test.ts | 29 ++++ 12 files changed, 472 insertions(+), 15 deletions(-) create mode 100644 mobile/src/transport/mobile-endpoint-supervisor-stalled-cell.test.ts create mode 100644 mobile/src/transport/relay-dial-stage.ts create mode 100644 mobile/src/transport/replacement-session-authentication.test.ts diff --git a/mobile/src/transport/mobile-endpoint-supervisor-stalled-cell.test.ts b/mobile/src/transport/mobile-endpoint-supervisor-stalled-cell.test.ts new file mode 100644 index 00000000000..be4050cee5b --- /dev/null +++ b/mobile/src/transport/mobile-endpoint-supervisor-stalled-cell.test.ts @@ -0,0 +1,80 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { MobileEndpointSupervisor } from './mobile-endpoint-supervisor' +import { + dependencies, + FakeLogicalClient, + FakeRelaySession, + host, + relay +} from './mobile-endpoint-supervisor-test-fakes' +import { ReplacementAuthenticationTimeoutError } from './replacement-session-authentication' +import type { RpcClient } from './rpc-client' + +vi.mock('react-native', () => ({ Platform: { OS: 'ios' } })) +vi.mock('expo-secure-store', () => ({ WHEN_UNLOCKED_THIS_DEVICE_ONLY: 'when-unlocked' })) +vi.mock('expo-crypto', () => ({ getRandomBytes: (length: number) => new Uint8Array(length) })) + +// The 2026-09-03 incident: five consecutive "authentication timed out" dials against a +// live desktop while the cell's assignment tables were lock-contended. Each logged +// failure was two dials — the timeout counted as a director-class failure, so the phone +// re-resolved the same cell and waited the full bound again. +describe('relay dial against a cell that took the dial and stalled', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.spyOn(console, 'log').mockImplementation(() => {}) + }) + afterEach(() => { + vi.useRealTimers() + vi.restoreAllMocks() + }) + + function timingOut(logical: FakeLogicalClient, error: Error): void { + logical.migrateTo.mockImplementation(async (session: RpcClient) => { + session.close() + throw error + }) + } + + it('does not re-resolve the director and names the stalled stage', async () => { + const logical = new FakeLogicalClient('disconnected', 'lan') + timingOut(logical, new ReplacementAuthenticationTimeoutError('awaiting-hello', 30_000)) + const openRelay = vi.fn(() => new FakeRelaySession('connecting')) + const resolveRelay = vi.fn(async () => relay) + const onLog = vi.fn() + const supervisor = new MobileEndpointSupervisor( + logical, + host, + dependencies({ openRelay, resolveRelay, onLog }) + ) + + await supervisor.start() + + expect(openRelay).toHaveBeenCalledOnce() + expect(resolveRelay).not.toHaveBeenCalled() + expect(onLog).toHaveBeenCalledWith( + expect.objectContaining({ + code: 'relay-dial-failed', + detail: expect.stringContaining('timed out (awaiting-hello, 30s)') + }) + ) + supervisor.stop() + }) + + it('still re-resolves the director when the cell socket never opened', async () => { + const logical = new FakeLogicalClient('disconnected', 'lan') + timingOut(logical, new ReplacementAuthenticationTimeoutError('opening', 12_000)) + const openRelay = vi.fn(() => new FakeRelaySession('connecting')) + const resolveRelay = vi.fn(async () => relay) + const supervisor = new MobileEndpointSupervisor( + logical, + host, + dependencies({ openRelay, resolveRelay }) + ) + + await supervisor.start() + + expect(resolveRelay).toHaveBeenCalledOnce() + expect(openRelay).toHaveBeenCalledTimes(2) + supervisor.stop() + }) +}) diff --git a/mobile/src/transport/mobile-endpoint-supervisor-support.ts b/mobile/src/transport/mobile-endpoint-supervisor-support.ts index 6ee0a6b42cb..1a2c00f12de 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-support.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-support.ts @@ -1,5 +1,6 @@ import { RelayOuterError } from './mobile-relay-e2ee-link' import { MobileE2EEAuthenticationError } from './mobile-e2ee-v2-physical-channel' +import { ReplacementAuthenticationTimeoutError } from './replacement-session-authentication' import type { RelayReconnectController } from './mobile-relay-reconnect-controller' import type { StableLogicalRpcClient } from './stable-logical-rpc-client' import type { HostProfile } from './types' @@ -68,6 +69,11 @@ export async function dialRelayThroughDirectorFallback(args: { } export function isDirectorResolutionFailure(error: Error): boolean { + // Why: a cell that took relay-auth and went quiet is the right cell working slowly; + // re-resolving it just doubles the wait against the same contended window. + if (error instanceof ReplacementAuthenticationTimeoutError) { + return error.stage === null || error.stage === 'opening' + } return ( !(error instanceof MobileE2EEAuthenticationError) && (!(error instanceof RelayOuterError) || [4409, 4503, 1006].includes(error.code)) diff --git a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts index 4023a1a8e39..1dc1473d9db 100644 --- a/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts +++ b/mobile/src/transport/mobile-endpoint-supervisor-test-fakes.ts @@ -1,6 +1,7 @@ import { vi } from 'vitest' import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bundle' import type { MobileRelayRpcSession } from './mobile-relay-rpc-session' +import { RelayDialStageTracker, type RelayDialStage } from './relay-dial-stage' import type { MobileEndpointSupervisorDependencies } from './mobile-endpoint-supervisor' import type { RpcClient } from './rpc-client' import type { MobileConnectionPath, StableLogicalRpcClient } from './stable-logical-rpc-client' @@ -51,6 +52,10 @@ export class FakeRelaySession extends FakeSession implements MobileRelayRpcSessi // Why: production-realistic defaults — fictional fake values hid three // live defects in this subsystem (latch, churn, int32 timer overflow). getAttachDeadlineAt = () => Date.now() + 10_000 + readonly dialStage = new RelayDialStageTracker() + getDialStage = () => this.dialStage.getDialStage() + onDialStageChange = (listener: (stage: RelayDialStage) => void) => + this.dialStage.onDialStageChange(listener) getResumeExpiresAt = () => this.resumeExpiry getResumeConfirmation = () => ({ v: 1 as const, diff --git a/mobile/src/transport/mobile-relay-e2ee-link.test.ts b/mobile/src/transport/mobile-relay-e2ee-link.test.ts index aa2d42b13f0..965135511eb 100644 --- a/mobile/src/transport/mobile-relay-e2ee-link.test.ts +++ b/mobile/src/transport/mobile-relay-e2ee-link.test.ts @@ -60,6 +60,61 @@ describe('MobileRelayE2eeLink', () => { expect(socket.close).toHaveBeenCalledOnce() }) + it('reports open only once relay-auth is on the wire', () => { + const socket = new ThrowingSocket() + const onOpen = vi.fn() + const sent: string[] = [] + socket.send.mockImplementation((frame: string) => { + sent.push(frame) + }) + new MobileRelayE2eeLink({ + endpoint: { + cellUrl: 'https://relay-c1.onorca.dev', + relayHostId: 'AbCdEf0123_-xyZ9' + }, + credential: 'credential', + expectedCredentialKind: 'resume', + deviceToken: 'device-token', + desktopPublicKeyB64: 'desktop-key', + onAuthenticated: vi.fn(), + onText: vi.fn(), + onBinary: vi.fn(), + onOpen, + onError: vi.fn(), + createSocket: () => socket as unknown as WebSocket + }) + + expect(onOpen).not.toHaveBeenCalled() + socket.onopen?.() + expect(sent).toHaveLength(1) + expect(JSON.parse(sent[0]!)).toMatchObject({ type: 'relay-auth', mode: 'connect' }) + expect(onOpen).toHaveBeenCalledOnce() + }) + + it('does not report open when the relay-auth write fails', () => { + const socket = new ThrowingSocket() + const onOpen = vi.fn() + new MobileRelayE2eeLink({ + endpoint: { + cellUrl: 'https://relay-c1.onorca.dev', + relayHostId: 'AbCdEf0123_-xyZ9' + }, + credential: 'credential', + expectedCredentialKind: 'resume', + deviceToken: 'device-token', + desktopPublicKeyB64: 'desktop-key', + onAuthenticated: vi.fn(), + onText: vi.fn(), + onBinary: vi.fn(), + onOpen, + onError: vi.fn(), + createSocket: () => socket as unknown as WebSocket + }) + + socket.onopen?.() + expect(onOpen).not.toHaveBeenCalled() + }) + it('keeps a typed close code when transport error precedes close', () => { const socket = new ThrowingSocket() const onError = vi.fn() diff --git a/mobile/src/transport/mobile-relay-e2ee-link.ts b/mobile/src/transport/mobile-relay-e2ee-link.ts index f19417a60f1..7743deb23dd 100644 --- a/mobile/src/transport/mobile-relay-e2ee-link.ts +++ b/mobile/src/transport/mobile-relay-e2ee-link.ts @@ -26,6 +26,8 @@ type MobileRelayE2eeLinkOptions = { onText: (plaintext: string) => void onBinary: (plaintext: Uint8Array) => void onHello?: (hello: Extract) => void + // Fired once relay-auth is on the wire: from here the cell owns the wait. + onOpen?: () => void onError: (error: Error) => void createSocket?: (url: string) => WebSocket } @@ -96,7 +98,9 @@ export class MobileRelayE2eeLink { ) } catch (error) { this.fail(asError(error)) + return } + this.options.onOpen?.() } this.socket.onmessage = (event) => { this.inboundChain = this.inboundChain diff --git a/mobile/src/transport/mobile-relay-rpc-session.test.ts b/mobile/src/transport/mobile-relay-rpc-session.test.ts index d5b547885cf..7f98436c63b 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.test.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.test.ts @@ -11,6 +11,7 @@ const fakes = vi.hoisted(() => ({ endpoint: { cellUrl: string; relayHostId: string } credential: string expectedCredentialKind: string + onOpen(): void onHello(value: unknown): void onAuthenticated(): void onText(value: string): void @@ -123,6 +124,35 @@ describe('mobile relay RPC session', () => { expect(session.getAttachDeadlineAt()).toEqual(expect.any(Number)) }) + // Why: ConnectionState stays 'connecting' until relay-hello, so the migration bound + // needs a separate signal to tell "cell never answered the upgrade" from "cell took + // relay-auth and is still resolving the assignment". + it('reports the dial stage as the link opens, receives hello, and authenticates', async () => { + const session = openSession() + const stages: string[] = [] + session.onDialStageChange((stage) => stages.push(stage)) + expect(session.getDialStage()).toBe('opening') + + fakes.linkOptions!.onOpen() + expect(session.getDialStage()).toBe('awaiting-hello') + expect(session.getState()).toBe('connecting') + fakes.linkOptions!.onHello({ + type: 'relay-hello', + ok: true, + credentialKind: 'resume', + leaseExpiresAt: Date.now() + 10_000, + acceptedCredentialVersion: 3, + acceptedAs: 'current', + resumeExpiresAt: Date.now() + 300_000 + }) + expect(session.getDialStage()).toBe('handshaking') + fakes.linkOptions!.onAuthenticated() + expect(session.getDialStage()).toBe('confirming') + await vi.waitFor(() => expect(fakes.sendText).toHaveBeenCalledOnce()) + expect(stages).toEqual(['awaiting-hello', 'handshaking', 'confirming']) + session.close() + }) + it('rejects a mismatched outer credential version and closes the physical link', () => { const session = openSession() fakes.linkOptions!.onHello({ diff --git a/mobile/src/transport/mobile-relay-rpc-session.ts b/mobile/src/transport/mobile-relay-rpc-session.ts index f242fce07ba..40b93927139 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.ts @@ -9,6 +9,7 @@ import { MobileE2EEAuthenticationError } from './mobile-e2ee-v2-physical-channel import { markRpcDeliveryUnknown } from './rpc-delivery-ambiguity' import { openRpcRequestBudget, resolvePostConnectRequestTimeout } from './rpc-request-budget' import { isRpcResponse } from './rpc-response-shape' +import { RelayDialStageTracker, type RelayDialStageSource } from './relay-dial-stage' import { RpcSessionLivenessWatchdog } from './rpc-session-liveness-watchdog' import type { RpcClient } from './rpc-client' import type { ConnectionLogSink, ConnectionState, RpcResponse } from './types' @@ -24,14 +25,15 @@ type PendingRequest = { timer: ReturnType } -export type MobileRelayRpcSession = RpcClient & { - // The cell's attach-reservation deadline (~10s). Diagnostics only — never - // schedule anything from it; rotation keys off getResumeExpiresAt(). - getAttachDeadlineAt(): number | null - getResumeExpiresAt(): number | null - getResumeConfirmation(): DeviceResumeConfirmed | null - getFailure(): Error | null -} +export type MobileRelayRpcSession = RpcClient & + RelayDialStageSource & { + // The cell's attach-reservation deadline (~10s). Diagnostics only — never + // schedule anything from it; rotation keys off getResumeExpiresAt(). + getAttachDeadlineAt(): number | null + getResumeExpiresAt(): number | null + getResumeConfirmation(): DeviceResumeConfirmed | null + getFailure(): Error | null + } export function connectMobileRelayRpcSession(args: { relay: MobileRelayEndpoint @@ -58,6 +60,7 @@ export function connectMobileRelayRpcSession(args: { let logSequence = 0 const logSessionId = `${Date.now().toString(36)}-${(++relayRpcSessionSequence).toString(36)}` const livenessIdentity = {} + const dialStage = new RelayDialStageTracker() const streams = new MobileRelayRpcStreams({ nextId, sendFrame, @@ -71,6 +74,7 @@ export function connectMobileRelayRpcSession(args: { deviceToken: args.deviceToken, desktopPublicKeyB64: args.desktopPublicKeyB64, createSocket: args.createSocket, + onOpen: () => dialStage.advance('awaiting-hello'), onHello: (hello) => { if ( hello.credentialKind !== 'resume' || @@ -81,6 +85,7 @@ export function connectMobileRelayRpcSession(args: { } attachDeadlineAt = hello.leaseExpiresAt resumeExpiresAt = hello.resumeExpiresAt + dialStage.advance('handshaking') publishState('handshaking') }, onAuthenticated: () => void confirmResume(), @@ -136,6 +141,8 @@ export function connectMobileRelayRpcSession(args: { streams.clear() publishState('disconnected') }, + getDialStage: () => dialStage.getDialStage(), + onDialStageChange: (listener) => dialStage.onDialStageChange(listener), getAttachDeadlineAt: () => attachDeadlineAt, getResumeExpiresAt: () => resumeExpiresAt, getResumeConfirmation: () => resumeConfirmation, @@ -165,6 +172,7 @@ export function connectMobileRelayRpcSession(args: { return client async function confirmResume(): Promise { + dialStage.advance('confirming') try { const response = await sendRpc( 'pairing.getEndpoints', diff --git a/mobile/src/transport/mobile-relay-runtime-failover.test.ts b/mobile/src/transport/mobile-relay-runtime-failover.test.ts index 795f2618dfb..01f4d45feb0 100644 --- a/mobile/src/transport/mobile-relay-runtime-failover.test.ts +++ b/mobile/src/transport/mobile-relay-runtime-failover.test.ts @@ -9,6 +9,7 @@ import { MobileE2EEAuthenticationError } from './mobile-e2ee-v2-physical-channel import { RelayOuterError } from './mobile-relay-e2ee-link' import type { MobileRelayCredentialBundle } from './mobile-relay-credential-bundle' import type { MobileRelayRpcSession } from './mobile-relay-rpc-session' +import { RelayDialStageTracker, type RelayDialStage } from './relay-dial-stage' import { MobileEndpointSupervisor, type MobileEndpointSupervisorDependencies @@ -81,6 +82,10 @@ class FakeRelaySession extends FakeSession implements MobileRelayRpcSession { // Why: production-realistic constants — fictional fake values hid three // live defects in this subsystem (latch, churn, int32 timer overflow). getAttachDeadlineAt = () => Date.now() + 10_000 + readonly dialStage = new RelayDialStageTracker() + getDialStage = () => this.dialStage.getDialStage() + onDialStageChange = (listener: (stage: RelayDialStage) => void) => + this.dialStage.onDialStageChange(listener) getResumeExpiresAt = () => Date.now() + 30 * 24 * 3_600_000 getResumeConfirmation = () => null getFailure = () => this.failure diff --git a/mobile/src/transport/relay-dial-stage.ts b/mobile/src/transport/relay-dial-stage.ts new file mode 100644 index 00000000000..c4a743f84f4 --- /dev/null +++ b/mobile/src/transport/relay-dial-stage.ts @@ -0,0 +1,64 @@ +// Where a relay dial is waiting, so a bound can tell "the cell never answered the +// upgrade" from "the cell took the dial and is slow" — the two look identical from +// ConnectionState, which stays 'connecting' until relay-hello arrives. +export type RelayDialStage = + // WebSocket upgrade not yet open. + | 'opening' + // Socket open and relay-auth sent; the cell is resolving/reserving and asking the + // desktop to attach before it can answer with relay-hello. + | 'awaiting-hello' + // relay-hello accepted; E2EE handshake with the desktop in flight. + | 'handshaking' + // E2EE authenticated; waiting on the desktop's resume confirmation. + | 'confirming' + +export type RelayDialStageSource = { + getDialStage(): RelayDialStage + onDialStageChange(listener: (stage: RelayDialStage) => void): () => void +} + +export function relayDialStageSource(session: object): RelayDialStageSource | null { + const candidate = session as Partial + return typeof candidate.getDialStage === 'function' && + typeof candidate.onDialStageChange === 'function' + ? (candidate as RelayDialStageSource) + : null +} + +export class RelayDialStageTracker implements RelayDialStageSource { + private stage: RelayDialStage = 'opening' + private readonly listeners = new Set<(stage: RelayDialStage) => void>() + + getDialStage(): RelayDialStage { + return this.stage + } + + onDialStageChange(listener: (stage: RelayDialStage) => void): () => void { + this.listeners.add(listener) + return () => this.listeners.delete(listener) + } + + advance(stage: RelayDialStage): void { + if (this.stage === stage) { + return + } + this.stage = stage + for (const listener of this.listeners) { + listener(stage) + } + } +} + +// Budget per stage once the cell holds the dial. awaiting-hello covers the cell's +// assignment/reservation transactions (observed 14–16s under lock contention) plus its +// 10s host-attach deadline; handshaking is two E2EE round trips; confirming is bounded +// by the session's own 30s resume-confirmation request, with slack so that error wins. +const RELAY_DIAL_STAGE_BUDGET_MS: Record, number> = { + 'awaiting-hello': 30_000, + handshaking: 12_000, + confirming: 35_000 +} + +export function relayDialStageBudgetMs(stage: Exclude): number { + return RELAY_DIAL_STAGE_BUDGET_MS[stage] +} diff --git a/mobile/src/transport/replacement-session-authentication.test.ts b/mobile/src/transport/replacement-session-authentication.test.ts new file mode 100644 index 00000000000..096721fa02d --- /dev/null +++ b/mobile/src/transport/replacement-session-authentication.test.ts @@ -0,0 +1,127 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { RelayDialStageTracker } from './relay-dial-stage' +import { + ReplacementAuthenticationTimeoutError, + waitForAuthenticated +} from './replacement-session-authentication' +import type { RpcClient } from './rpc-client' +import type { ConnectionState } from './types' + +class FakeSession implements RpcClient { + readonly sendRequest = vi.fn() + readonly subscribe = vi.fn(() => () => {}) + readonly updateTerminalSubscriptionViewport = vi.fn() + readonly notifyForeground = vi.fn() + readonly close = vi.fn() + private readonly listeners = new Set<(state: ConnectionState) => void>() + constructor(private state: ConnectionState = 'connecting') {} + getState = () => this.state + getReconnectAttempt = () => 0 + getLastConnectedAt = () => null + onStateChange = (listener: (state: ConnectionState) => void) => { + this.listeners.add(listener) + return () => this.listeners.delete(listener) + } + setState(state: ConnectionState): void { + this.state = state + for (const listener of this.listeners) { + listener(state) + } + } +} + +class FakeRelaySession extends FakeSession { + readonly dialStage = new RelayDialStageTracker() + getDialStage = () => this.dialStage.getDialStage() + onDialStageChange = this.dialStage.onDialStageChange.bind(this.dialStage) +} + +// Why: fake timers are active, so "still pending" is decided on the microtask queue. +async function settle( + promise: Promise +): Promise<{ status: 'pending' | 'settled'; error?: Error }> { + let outcome: { status: 'pending' | 'settled'; error?: Error } = { status: 'pending' } + void promise.then( + () => (outcome = { status: 'settled' }), + (error: Error) => (outcome = { status: 'settled', error }) + ) + await Promise.resolve() + await Promise.resolve() + return outcome +} + +describe('waitForAuthenticated', () => { + beforeEach(() => vi.useFakeTimers()) + afterEach(() => vi.useRealTimers()) + + it('keeps the flat bound for a session that reports no dial stages', async () => { + const session = new FakeSession() + const waiting = waitForAuthenticated(session, 12_000) + waiting.catch(() => {}) + await vi.advanceTimersByTimeAsync(11_999) + expect((await settle(waiting)).status).toBe('pending') + await vi.advanceTimersByTimeAsync(1) + const outcome = await settle(waiting) + expect(outcome.error).toBeInstanceOf(ReplacementAuthenticationTimeoutError) + expect(outcome.error?.message).toBe('replacement session authentication timed out') + }) + + // The 2026-09-03 incident: the cell accepted relay-auth and spent 14–16s in its + // lock-contended assignment transactions. The flat 12s bound hung up 2–4s before + // the cell finished, five dials in a row, while the desktop was live the whole time. + it('re-arms the bound per stage once the cell holds the dial', async () => { + const session = new FakeRelaySession() + const waiting = waitForAuthenticated(session, 12_000) + waiting.catch(() => {}) + await vi.advanceTimersByTimeAsync(11_000) + session.dialStage.advance('awaiting-hello') + await vi.advanceTimersByTimeAsync(5_000) + expect((await settle(waiting)).status).toBe('pending') + session.dialStage.advance('handshaking') + session.setState('handshaking') + await vi.advanceTimersByTimeAsync(11_000) + expect((await settle(waiting)).status).toBe('pending') + session.dialStage.advance('confirming') + await vi.advanceTimersByTimeAsync(20_000) + session.setState('connected') + await expect(waiting).resolves.toBeUndefined() + }) + + it('bounds a cell that took the dial and never answers, naming the stage', async () => { + const session = new FakeRelaySession() + const waiting = waitForAuthenticated(session, 12_000) + waiting.catch(() => {}) + await vi.advanceTimersByTimeAsync(2_000) + session.dialStage.advance('awaiting-hello') + await vi.advanceTimersByTimeAsync(29_999) + expect((await settle(waiting)).status).toBe('pending') + await vi.advanceTimersByTimeAsync(1) + const outcome = await settle(waiting) + expect(outcome.error).toBeInstanceOf(ReplacementAuthenticationTimeoutError) + expect((outcome.error as ReplacementAuthenticationTimeoutError).stage).toBe('awaiting-hello') + expect(outcome.error?.message).toBe( + 'replacement session authentication timed out (awaiting-hello, 30s)' + ) + }) + + it('keeps the caller bound while the socket never opens', async () => { + const session = new FakeRelaySession() + const waiting = waitForAuthenticated(session, 12_000) + waiting.catch(() => {}) + await vi.advanceTimersByTimeAsync(12_000) + const outcome = await settle(waiting) + expect((outcome.error as ReplacementAuthenticationTimeoutError).stage).toBe('opening') + expect(outcome.error?.message).toBe( + 'replacement session authentication timed out (opening, 12s)' + ) + }) + + it('ignores stage advances after the wait has settled', async () => { + const session = new FakeRelaySession() + const waiting = waitForAuthenticated(session, 12_000) + session.setState('disconnected') + await expect(waiting).rejects.toThrow('replacement session disconnected') + session.dialStage.advance('awaiting-hello') + expect(vi.getTimerCount()).toBe(0) + }) +}) diff --git a/mobile/src/transport/replacement-session-authentication.ts b/mobile/src/transport/replacement-session-authentication.ts index 0ba54a9d611..5ef9a5f1f46 100644 --- a/mobile/src/transport/replacement-session-authentication.ts +++ b/mobile/src/transport/replacement-session-authentication.ts @@ -1,21 +1,43 @@ import type { RpcClient } from './rpc-client' +import { + relayDialStageBudgetMs, + relayDialStageSource, + type RelayDialStage +} from './relay-dial-stage' + +export class ReplacementAuthenticationTimeoutError extends Error { + constructor( + readonly stage: RelayDialStage | null, + budgetMs: number + ) { + super( + stage + ? `replacement session authentication timed out (${stage}, ${Math.round(budgetMs / 1000)}s)` + : 'replacement session authentication timed out' + ) + this.name = 'ReplacementAuthenticationTimeoutError' + } +} // Why: a migration must not cut over to a session that has only opened a socket — the -// replacement has to reach 'connected' (E2EE authenticated) first, and a relay dial can -// sit in handshaking for seconds, so the wait is bounded by the caller's timeout. +// replacement has to reach 'connected' (E2EE authenticated) first. The caller's bound +// covers reaching an open socket; a relay session that reports dial stages re-arms a +// per-stage budget on every advance, so a cell that accepted the dial and is working +// slowly (lock-contended assignment tables) is not hung up on like a black hole — the +// retry would land in the same window and burn a director round on the way. export function waitForAuthenticated(session: RpcClient, timeoutMs: number): Promise { if (session.getState() === 'connected') { return Promise.resolve() } + const stages = relayDialStageSource(session) return new Promise((resolve, reject) => { let settled = false let unsubscribe: (() => void) | null = null + let unsubscribeStage: (() => void) | null = null + let timer: ReturnType | null = null // Why: armed before subscribing — a synchronous notification during registration // must find a timer to clear, or a settled wait leaves it running for 12s. - const timer = setTimeout(() => { - finish() - reject(new Error('replacement session authentication timed out')) - }, timeoutMs) + arm(stages?.getDialStage() ?? null) unsubscribe = session.onStateChange((state) => { if (state === 'connected') { finish() @@ -29,6 +51,23 @@ export function waitForAuthenticated(session: RpcClient, timeoutMs: number): Pro // Why: the notification fired inside onStateChange, before we held the handle. unsubscribe() unsubscribe = null + } else if (stages) { + unsubscribeStage = stages.onDialStageChange((stage) => arm(stage)) + } + + function arm(stage: RelayDialStage | null): void { + if (settled) { + return + } + if (timer) { + clearTimeout(timer) + } + const budgetMs = + stage === null || stage === 'opening' ? timeoutMs : relayDialStageBudgetMs(stage) + timer = setTimeout(() => { + finish() + reject(new ReplacementAuthenticationTimeoutError(stage, budgetMs)) + }, budgetMs) } function finish(): void { @@ -36,9 +75,14 @@ export function waitForAuthenticated(session: RpcClient, timeoutMs: number): Pro return } settled = true - clearTimeout(timer) + if (timer) { + clearTimeout(timer) + timer = null + } unsubscribe?.() unsubscribe = null + unsubscribeStage?.() + unsubscribeStage = null } }) } diff --git a/mobile/src/transport/stable-logical-rpc-client.test.ts b/mobile/src/transport/stable-logical-rpc-client.test.ts index 0ba7a1f2ea7..faa236a88ca 100644 --- a/mobile/src/transport/stable-logical-rpc-client.test.ts +++ b/mobile/src/transport/stable-logical-rpc-client.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it, vi } from 'vitest' +import { RelayDialStageTracker, type RelayDialStage } from './relay-dial-stage' import type { ConnectionState, RpcResponse } from './types' import type { RpcClient } from './rpc-client' import { isRpcDeliveryUnknown, markRpcDeliveryUnknown } from './rpc-delivery-ambiguity' @@ -399,6 +400,34 @@ describe('stable logical RPC client', () => { expect(client.getPendingPath()).toBeNull() }) + // Pins the shipping wiring: migrateTo's bound honors the replacement's dial stages. + it('outlives the flat bound when the relay cell holds the dial', async () => { + vi.useFakeTimers() + try { + const oldSession = new FakeSession('connected') + const replacement = Object.assign(new FakeSession('connecting'), { + dialStage: new RelayDialStageTracker(), + getDialStage(): RelayDialStage { + return this.dialStage.getDialStage() + }, + onDialStageChange(listener: (stage: RelayDialStage) => void) { + return this.dialStage.onDialStageChange(listener) + } + }) + const client = createStableLogicalRpcClient(oldSession, 'lan') + const migrating = client.migrateTo(replacement, 'relay', 12_000) + await vi.advanceTimersByTimeAsync(1_000) + replacement.dialStage.advance('awaiting-hello') + await vi.advanceTimersByTimeAsync(20_000) + expect(replacement.close).not.toHaveBeenCalled() + replacement.setState('connected') + await migrating + expect(client.getActivePath()).toBe('relay') + } finally { + vi.useRealTimers() + } + }) + it('closes a replacement that fails authentication and preserves the active session', async () => { const oldSession = new FakeSession('connected') const replacement = new FakeSession('connecting') From d66386bc8278637a62c59502da459ecbdf6c93c6 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 17:27:57 -0400 Subject: [PATCH 190/398] fix(cloud): bound and yield the relay's global cell-inventory lock (#18521) Mirrors stablyai/orca-cloud#471 (squash c3354e8), byte-identical under cloud/. The relay's cell-inventory lock (SELECT ... FROM relay_cells FOR UPDATE over all 23 rows) is one global critical section shared by the assignment hot path and every director sweep; with the pool's 1s lock_timeout a blocked waiter held a pooled client for a full second, producing ~690 55P03 retries per 5 minutes in production. Request paths now bound the wait at 500ms with a SET LOCAL that is restored to the pool default before the next statement; director-only sweeps take the lock NOWAIT and skip the tick; sweep timers are jittered; hold time is exported as additive runtime-metrics fields so the bound can be tuned. --- .../relay-ops/src/incident-monitor.test.ts | 13 + cloud/apps/relay-ops/src/incident-monitor.ts | 14 + cloud/apps/relay/src/assignment-store.ts | 196 +++++-- .../relay/src/cell-inventory-hold-samples.ts | 46 ++ .../src/cell-inventory-lock-census.test.ts | 164 ++++++ .../cell-inventory-lock-contention.test.ts | 541 ++++++++++++++++++ cloud/apps/relay/src/database.ts | 103 +++- cloud/apps/relay/src/index.ts | 7 +- .../relay/src/regional-rehome-store.test.ts | 337 ++++++++++- .../apps/relay/src/regional-rehome-worker.ts | 7 +- cloud/apps/relay/src/relay-observability.ts | 5 +- .../relay/src/relay-sweep-schedule.test.ts | 55 ++ cloud/apps/relay/src/relay-sweep-schedule.ts | 13 + 13 files changed, 1439 insertions(+), 62 deletions(-) create mode 100644 cloud/apps/relay/src/cell-inventory-hold-samples.ts create mode 100644 cloud/apps/relay/src/cell-inventory-lock-census.test.ts create mode 100644 cloud/apps/relay/src/cell-inventory-lock-contention.test.ts create mode 100644 cloud/apps/relay/src/relay-sweep-schedule.test.ts create mode 100644 cloud/apps/relay/src/relay-sweep-schedule.ts diff --git a/cloud/apps/relay-ops/src/incident-monitor.test.ts b/cloud/apps/relay-ops/src/incident-monitor.test.ts index 51153bb63e1..ea5ad55b645 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.test.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.test.ts @@ -123,6 +123,19 @@ describe('incident monitor evaluator', () => { }) }) + // Why: sweeps no longer reach the retry wrapper, so any exhaustion left in this + // counter is a request path that terminally failed. It must still freeze. + it('freezes on a single exhausted request-path transaction', () => { + const sample = healthySample() + sample.sources['relay-logs']!.signals['relay.postgres_retry_exhausted'] = signal(1) + expect(evaluateIncidentSample(sample, startedAt)).toMatchObject({ + status: 'freeze', + failures: [ + expect.objectContaining({ signal: 'relay.postgres_retry_exhausted', threshold: 0 }) + ] + }) + }) + it('allows missing auth readiness and legacy existing-only connections', () => { const sample = healthySample() const legacySelector = { diff --git a/cloud/apps/relay-ops/src/incident-monitor.ts b/cloud/apps/relay-ops/src/incident-monitor.ts index 6073e351511..a1936f5df6c 100644 --- a/cloud/apps/relay-ops/src/incident-monitor.ts +++ b/cloud/apps/relay-ops/src/incident-monitor.ts @@ -38,6 +38,20 @@ export const INCIDENT_MONITOR_THRESHOLDS = { // margin; relayPostgresRetryExhausted below stays at zero tolerance, so any // transaction that terminally fails still freezes the gate. relayPostgresRetries: 300, + // Why: this bar stays at zero. Cell-inventory contention reaches the retry + // wrapper from exactly two kinds of caller, and neither is a sweep tick that + // can shrug the failure off: + // - request paths, which take a wait bounded at CELL_INVENTORY_LOCK_TIMEOUT_MS + // (assignment, control activation, activity, admin drain/evacuate/supersede); + // - sweep-reachable code that a request also enters, which keeps the pool + // lock_timeout so it cannot fail faster than before this change: the + // completeEvacuation site that waits, reconcileReservationAccounting, and + // placement re-entered from evacuateDeadCells. + // Sweep-only sites take the inventory NOWAIT, so their contention becomes + // database_lock_unavailable, which is not a retryable abort and never reaches + // this counter. The relay's cell-inventory-lock census test holds that split. + // Splitting the metric by the phase label PR #423 put on the log payload would + // need a labelled log-based metric, which this signal's counter does not carry. relayPostgresRetryExhausted: 0, // Why: public admission is a per-instance semaphore, so fleet assignment capacity is // concurrency x instances. A floor of 1 let the 2026-08-04 collapse from five instances diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index 0b2b1ef72a9..3571b57cd22 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -31,7 +31,12 @@ import { } from './assignment-connection-headroom-query.js' import { AssignmentIdentityQueue } from './assignment-identity-queue.js' import type { RelayCellConfig } from './config.js' -import type { RelayDatabase, RelayTransactionOptions, SqlRow } from './database.js' +import type { + RelayDatabase, + RelayLockOptions, + RelayTransactionOptions, + SqlRow +} from './database.js' import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' import { combineRegionalRehomeSafety, @@ -316,6 +321,25 @@ const ACTIVITY_REQUEST_UNITS: Record = { } const ASSIGNMENT_LOCK_RETRY_DEADLINE_MS = 15_000 +// Why: one global FOR UPDATE over a 23-row table serialises every director and +// cell. At the 1s pool lock_timeout each blocked waiter also holds a pooled +// client for a full second, so the queue converts contention into pool +// exhaustion. The lock is held to COMMIT and the assignment path runs many +// statements after taking it, and no hold-time telemetry existed before this +// change, so 500ms is a first value to tune once cellInventoryHoldMsMax lands. +export const CELL_INVENTORY_LOCK_TIMEOUT_MS = 500 + +// The same inventory lock is taken by live requests and by background sweeps, +// and the right failure mode differs per caller. +export type CellInventoryLockMode = + // Bound the wait so a blocked request stops occupying a pooled client. + | 'request' + // Never queue: the caller handles database_lock_unavailable and moves on. + | 'nowait' + // A sweep can enter here, so keep the pool default. Failing sooner would turn + // ordinary contention into a 55P03 the retry wrapper reports as terminal, and + // one terminal failure freezes the incident gate. + | 'pool-default' // Why: stranded detection (issue #225) needs a grant old enough that a real // attach would have registered (the 90s activity lease covers dial + // activation), yet recent enough to prove an active retry loop rather than @@ -536,29 +560,35 @@ export class RelayAssignmentStore { async assign( identity: AssignmentIdentity, preferredRegion?: RelayRegion, - placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION + placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION, + // evacuateDeadCells re-enters placement from a sweep; it must not take the + // bounded wait, whose 55P03 would surface as a terminal sweep failure. + lockMode: CellInventoryLockMode = 'request' ): Promise { - const sticky = await this.assignStickyWithLockRetry(identity, preferredRegion) + const sticky = await this.assignStickyWithLockRetry(identity, lockMode, preferredRegion) if (sticky) return sticky // Only placement needs the global inventory critical section; queueing those // attempts locally avoids turning true placement bursts into NOWAIT storms. return await this.serializeAssignment( - async () => await this.assignWithLockRetry(identity, preferredRegion, placementRegion) + async () => + await this.assignWithLockRetry(identity, lockMode, preferredRegion, placementRegion) ) } private async assignStickyWithLockRetry( identity: AssignmentIdentity, + lockMode: CellInventoryLockMode, preferredRegion?: RelayRegion ): Promise { return await this.withAssignmentLockRetry( async (inventoryFirst) => - await this.assignStickyOnce(identity, inventoryFirst, preferredRegion) + await this.assignStickyOnce(identity, inventoryFirst, lockMode, preferredRegion) ) } private async assignWithLockRetry( identity: AssignmentIdentity, + lockMode: CellInventoryLockMode, preferredRegion?: RelayRegion, placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION ): Promise { @@ -566,7 +596,13 @@ export class RelayAssignmentStore { let inventoryScope: AssignmentInventoryScope = 'none' while (true) { try { - return await this.assignOnce(identity, inventoryScope, preferredRegion, placementRegion) + return await this.assignOnce( + identity, + inventoryScope, + lockMode, + preferredRegion, + placementRegion + ) } catch (error) { if (error instanceof AssignmentInventoryScopeChanged) { inventoryScope = 'all' @@ -604,12 +640,13 @@ export class RelayAssignmentStore { private async assignStickyOnce( identity: AssignmentIdentity, inventoryFirst: boolean, + lockMode: CellInventoryLockMode, preferredRegion?: RelayRegion ): Promise { const now = this.now() return await this.database.transaction(async (transaction) => { const lockedCells = inventoryFirst - ? await this.lockCellInventory(transaction) + ? await this.lockCellInventory(transaction, lockMode) : undefined const existing = await this.assignmentRow(transaction, identity, inventoryFirst) if (!existing) return null @@ -751,6 +788,7 @@ export class RelayAssignmentStore { private async assignOnce( identity: AssignmentIdentity, inventoryScope: AssignmentInventoryScope, + lockMode: CellInventoryLockMode, preferredRegion?: RelayRegion, placementRegion: RelayRegion = preferredRegion ?? RELAY_DEFAULT_REGION ): Promise { @@ -760,9 +798,9 @@ export class RelayAssignmentStore { return await this.database.transaction(async (transaction) => { let lockedCells = inventoryScope === 'all' - ? await this.lockCellInventory(transaction) + ? await this.lockCellInventory(transaction, lockMode) : inventoryScope === 'general' - ? await this.lockGeneralCellInventory(transaction) + ? await this.lockGeneralCellInventory(transaction, lockMode) : undefined const existing = await this.assignmentRow( transaction, @@ -779,7 +817,7 @@ export class RelayAssignmentStore { let connectionHeadroomReassignment = false let strandedReassignment = false if (existing && !mayNormallyReassign(activity(existing), now)) { - lockedCells ??= await this.lockCellInventory(transaction, true) + lockedCells ??= await this.lockCellInventory(transaction, 'nowait') const admission = await cellAdmissionStates(transaction) const currentRow = lockedCells.find( (row) => text(row, 'cell_id') === text(existing, 'cell_id') @@ -859,8 +897,8 @@ export class RelayAssignmentStore { } lockedCells ??= existing - ? await this.lockCellInventory(transaction, true) - : await this.lockGeneralCellInventory(transaction, true) + ? await this.lockCellInventory(transaction, 'nowait') + : await this.lockGeneralCellInventory(transaction, 'nowait') const target = await this.leastLoadedCell( transaction, lockedCells, @@ -2114,7 +2152,7 @@ export class RelayAssignmentStore { ORDER BY migration.user_id, migration.relay_host_id`, [input.cellId] ) - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'request') for (const migrationRow of migrations) { const identity = { userId: text(migrationRow, 'user_id'), @@ -2608,10 +2646,12 @@ export class RelayAssignmentStore { let moved = 0 for (const row of rows) { try { - const assignment = await this.assign({ - userId: text(row, 'user_id'), - relayHostId: text(row, 'relay_host_id') - }) + const assignment = await this.assign( + { userId: text(row, 'user_id'), relayHostId: text(row, 'relay_host_id') }, + undefined, + undefined, + 'pool-default' + ) if (assignment.cellId !== text(row, 'cell_id')) moved++ } catch (error) { if (!(error instanceof Error && error.message === 'relay_capacity_exhausted')) throw error @@ -3162,7 +3202,7 @@ export class RelayAssignmentStore { ) const requestDelta = ACTIVITY_REQUEST_UNITS[kind] * (after - before) if (requestDelta !== 0) { - await this.lockCellInventory(transaction) + await this.lockCellInventory(transaction, 'request') await this.adjustCellReservation(transaction, text(row, 'cell_id'), requestDelta) } }) @@ -3223,7 +3263,7 @@ export class RelayAssignmentStore { } const units = ACTIVITY_REQUEST_UNITS[input.kind] if (existing) { - await this.lockCellInventory(transaction) + await this.lockCellInventory(transaction, 'request') await this.removeActivityLease(transaction, identity, existing, now) await this.adjustCellReservation(transaction, input.cellId, units) } @@ -3540,7 +3580,7 @@ export class RelayAssignmentStore { ) await this.touchAssignment(transaction, identity, expiresAt, now) } else { - await this.lockCellInventory(transaction) + await this.lockCellInventory(transaction, 'request') await this.adjustCellReservation(transaction, input.cellId, 1) await this.adjustActivityCount(transaction, identity, 'control', 1, expiresAt, now) await transaction.query( @@ -3614,7 +3654,7 @@ export class RelayAssignmentStore { } if (sourceCellId === targetCellId) throw new Error('target_matches_source') await this.lockAssignmentActivities(transaction, identity) - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'request') const target = cells.find((row) => text(row, 'cell_id') === targetCellId) if (!target || integer(target, 'enabled') !== 1) throw new Error('target_cell_unavailable') if (!(await this.cellIsLive(transaction, targetCellId, now))) { @@ -3822,7 +3862,7 @@ export class RelayAssignmentStore { let lockedCells: SqlRow[] | undefined if (inventoryFirst) { try { - lockedCells = await this.lockCellInventory(transaction) + lockedCells = await this.lockCellInventory(transaction, 'request') } catch (error) { if (isDatabaseLockTimeout(error)) { throw new Error('database_lock_unavailable') @@ -3863,7 +3903,7 @@ export class RelayAssignmentStore { if (activityUnitsForCell(activityLeases, input.sourceCellId) > 0) { throw new Error('migration_source_still_active') } - const cells = lockedCells ?? (await this.lockCellInventory(transaction, true)) + const cells = lockedCells ?? (await this.lockCellInventory(transaction, 'nowait')) const source = cells.find((cell) => text(cell, 'cell_id') === input.sourceCellId) const target = cells.find((cell) => text(cell, 'cell_id') === input.targetCellId) if (!source || integer(source, 'enabled') !== 0) { @@ -3967,7 +4007,7 @@ export class RelayAssignmentStore { const now = this.now() return await this.database.transaction(async (transaction) => { const lockedCells = inventoryFirst - ? await this.lockCellInventory(transaction) + ? await this.lockCellInventory(transaction, 'request') : undefined const assignment = await this.assignmentRow(transaction, identity, inventoryFirst) const existing = ( @@ -4041,7 +4081,7 @@ export class RelayAssignmentStore { ) { throw new Error('migration_activity_topology_mismatch') } - const cells = lockedCells ?? (await this.lockCellInventory(transaction, true)) + const cells = lockedCells ?? (await this.lockCellInventory(transaction, 'nowait')) const source = cells.find((cell) => text(cell, 'cell_id') === input.sourceCellId) const currentTarget = cells.find( (cell) => text(cell, 'cell_id') === input.currentTargetCellId @@ -4458,7 +4498,7 @@ export class RelayAssignmentStore { throw new Error('migration_activity_topology_mismatch') } } - if (obsoleteLeases.length > 0) await this.lockCellInventory(transaction) + if (obsoleteLeases.length > 0) await this.lockCellInventory(transaction, 'request') for (const lease of obsoleteLeases) { await this.removeActivityLease(transaction, identity, lease, now) } @@ -4634,7 +4674,7 @@ export class RelayAssignmentStore { ) { throw new Error('migration_activity_lease_shape_mismatch') } - await this.lockCellInventory(transaction) + await this.lockCellInventory(transaction, 'request') await this.adjustCellReservation( transaction, input.currentTargetCellId, @@ -4723,7 +4763,7 @@ export class RelayAssignmentStore { if (this.requireLiveCells) { let cells: SqlRow[] try { - cells = await this.lockCellInventory(transaction, true) + cells = await this.lockCellInventory(transaction, 'nowait') } catch (error) { if (isDatabaseLockUnavailable(error)) { // Mixed-version workers may still hold a cell-first lock; defer @@ -4779,7 +4819,7 @@ export class RelayAssignmentStore { ) if (!targetIsActive) throw new Error('migration_target_not_active') const lease = activityLeaseById(activityLeases, migrationActivityId(assignmentEpoch)) - if (lease && !cellsLocked) await this.lockCellInventory(transaction) + if (lease && !cellsLocked) await this.lockCellInventory(transaction, 'pool-default') if (lease) await this.removeActivityLease(transaction, identity, lease, now) await transaction.query( `UPDATE relay_assignment_migrations SET completed_at = ?, updated_at = ? @@ -4801,7 +4841,7 @@ export class RelayAssignmentStore { const sourceCellId = text(assignment, 'cell_id') if (sourceCellId === targetCellId) throw new Error('target_matches_source') await this.lockAssignmentActivities(transaction, identity) - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'request') const admission = await cellAdmissionStates(transaction) const targetRow = cells.find( (row) => @@ -5051,7 +5091,13 @@ export class RelayAssignmentStore { } this.pendingRegionalRehomeDisableLog = null const candidateSkips: RegionalRehomeCandidateSkip[] = [] + // A Postgres transaction is unusable after a NOWAIT abort, so a contended + // tick abandons the candidate it stopped on plus every one behind it. + let candidatesTotal = 0 + let candidatesFinished = 0 const claimResult = await this.database.transaction(async (transaction) => { + candidatesTotal = 0 + candidatesFinished = 0 candidateSkips.length = 0 await this.initializeRegionalRehomeControl(transaction, now) const control = ( @@ -5122,6 +5168,7 @@ export class RelayAssignmentStore { ) )[0] if (retry) { + candidatesTotal = 1 const fleetSafety = await this.lockedRegionalRehomeFleetSafety(transaction, now) if ( !(await this.regionalRehomeSafetyAllowsClaim( @@ -5190,6 +5237,7 @@ export class RelayAssignmentStore { ) )[0] if (redrain) { + candidatesTotal = 1 const fleetSafety = await this.lockedRegionalRehomeFleetSafety(transaction, now) if ( !(await this.regionalRehomeSafetyAllowsClaim( @@ -5253,6 +5301,7 @@ export class RelayAssignmentStore { LIMIT 10`, [preferenceCutoff, now - this.heartbeatTtlMs, now] ) + candidatesTotal = candidates.length for (const candidate of candidates) { const claimed = await this.startRegionalRehomeCandidate(transaction, { identity: { @@ -5268,6 +5317,7 @@ export class RelayAssignmentStore { now, skips: candidateSkips }) + candidatesFinished++ if (!claimed) continue await this.markRegionalRehomeDispatchClaimed( transaction, @@ -5283,6 +5333,21 @@ export class RelayAssignmentStore { await this.markRegionalRehomeTickSkipped(transaction, now, intervalMs) } return null + }).catch((error: unknown): RegionalRehomeAttempt | null => { + // Only inventory contention is swallowed here; every other failure keeps + // its existing propagation and its dispatch-failure accounting. + if (!isDatabaseLockUnavailable(error)) throw error + // The dispatch tick runs every second; losing one to inventory contention + // costs a second of latency and never loses durable rehome state. The + // rolled-back transaction never disabled anything, so its pending disable + // log would describe a decision that did not happen. + candidateSkips.length = 0 + this.pendingRegionalRehomeDisableLog = null + warnSweepCellInventoryBusy( + 'claim-regional-rehome', + Math.max(1, candidatesTotal - candidatesFinished) + ) + return null }) const pendingDisableLog = this.pendingRegionalRehomeDisableLog this.pendingRegionalRehomeDisableLog = null @@ -5343,7 +5408,7 @@ export class RelayAssignmentStore { } const activityLeases = await this.lockAssignmentActivities(transaction, input.identity) assertAssignmentActivityCounts(assignment, activityLeases, 0) - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'nowait') const admission = await cellAdmissionStates(transaction) const regions = new Map( (await transaction.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ @@ -5630,7 +5695,7 @@ export class RelayAssignmentStore { transaction: RelayDatabase, now: number ): Promise { - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'nowait') const admission = await cellAdmissionStates(transaction) const regions = new Map( (await transaction.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ @@ -5874,6 +5939,7 @@ export class RelayAssignmentStore { [...quarantined, limit] ) let completed = 0 + let inventoryBusy = 0 for (const candidate of candidates) { // One poisoned row must not stall every later candidate: an invariant // throw here blocked fleet completions head-of-line in production. @@ -5890,9 +5956,14 @@ export class RelayAssignmentStore { if (changed) completed++ this.regionalRehomeCandidateQuarantine.delete(attemptId) } catch (error) { + if (isDatabaseLockUnavailable(error)) { + inventoryBusy++ + continue + } this.recordRegionalRehomeCandidateFailure('complete', attemptId, now, error) } } + warnSweepCellInventoryBusy('complete-ready-regional-rehomes', inventoryBusy) return completed } @@ -6112,7 +6183,7 @@ export class RelayAssignmentStore { leases, migration ) - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'nowait') const target = cells.find((cell) => text(cell, 'cell_id') === targetCellId) const admission = await cellAdmissionStates(transaction) if ( @@ -6261,6 +6332,7 @@ export class RelayAssignmentStore { [now - REGIONAL_REHOME_MAX_REFRESH_MS, ...quarantined, limit] ) let aborted = 0 + let inventoryBusy = 0 for (const candidate of candidates) { const identity = { userId: text(candidate, 'user_id'), @@ -6318,7 +6390,7 @@ export class RelayAssignmentStore { integer(lease, 'expires_at') > now ) if (targetActive) return false - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'nowait') const source = cells.find((cell) => text(cell, 'cell_id') === sourceCellId) const admission = await cellAdmissionStates(transaction) if ( @@ -6376,10 +6448,12 @@ export class RelayAssignmentStore { }) this.regionalRehomeCandidateQuarantine.delete(attemptId) } catch (error) { - this.recordRegionalRehomeCandidateFailure('abort', attemptId, now, error) + if (isDatabaseLockUnavailable(error)) inventoryBusy++ + else this.recordRegionalRehomeCandidateFailure('abort', attemptId, now, error) } if (changed) aborted++ } + warnSweepCellInventoryBusy('abort-expired-regional-rehomes', inventoryBusy) return aborted } @@ -6396,6 +6470,7 @@ export class RelayAssignmentStore { [now, now, abandonedBefore, abandonedBefore] ) let aborted = 0 + let inventoryBusy = 0 for (const candidate of candidates) { const didAbort = await this.database.transaction(async (transaction) => { const identity = { @@ -6480,7 +6555,7 @@ export class RelayAssignmentStore { ] .map((activityId) => activityLeaseById(activityLeases, activityId)) .filter((lease): lease is SqlRow => lease !== undefined) - if (obsoleteLeases.length > 0) await this.lockCellInventory(transaction) + if (obsoleteLeases.length > 0) await this.lockCellInventory(transaction, 'nowait') for (const lease of obsoleteLeases) { await this.removeActivityLease(transaction, identity, lease, now) } @@ -6498,7 +6573,7 @@ export class RelayAssignmentStore { ) return true } - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'nowait') const sourceCellId = text(row, 'source_cell_id') const admissionRows = await transaction.query( `SELECT cell_id, admission_state, updated_at FROM relay_cell_admission @@ -6595,9 +6670,15 @@ export class RelayAssignmentStore { [now, now, identity.userId, identity.relayHostId, assignmentEpoch] ) return true + }).catch((error: unknown): boolean => { + // Expiry is durable; another director settling this row is not a failure. + if (!isDatabaseLockUnavailable(error)) throw error + inventoryBusy++ + return false }) if (didAbort) aborted++ } + warnSweepCellInventoryBusy('abort-expired-evacuations', inventoryBusy) return aborted } @@ -6665,7 +6746,7 @@ export class RelayAssignmentStore { const activityLeases = await this.lockAssignmentActivities(transaction, identity, true) const lease = activityLeaseById(activityLeases, text(candidate, 'activity_id')) if (!lease || integer(lease, 'expires_at') > now) return false - await this.lockCellInventory(transaction, true) + await this.lockCellInventory(transaction, 'nowait') await this.removeActivityLease(transaction, identity, lease, now) return true }) @@ -6709,7 +6790,7 @@ export class RelayAssignmentStore { [now], { failIfUnavailable: true } ) - if (expired.length > 0) await this.lockCellInventory(transaction, true) + if (expired.length > 0) await this.lockCellInventory(transaction, 'nowait') for (const row of expired) { await this.adjustCellReservation(transaction, text(row, 'cell_id'), -requestUnits(row)) await transaction.query( @@ -6782,7 +6863,7 @@ export class RelayAssignmentStore { targetCellId ] ) - const cells = await this.lockCellInventory(transaction) + const cells = await this.lockCellInventory(transaction, 'pool-default') const assignmentKeys = new Set( assignments.map((row) => assignmentKey(text(row, 'user_id'), text(row, 'relay_host_id')) @@ -6861,30 +6942,32 @@ export class RelayAssignmentStore { private async lockCellInventory( database: RelayDatabase, - failIfUnavailable = false + mode: CellInventoryLockMode ): Promise { // Every capacity-changing assignment takes the tiny cell inventory in one // order; dynamically locking only the selected target allowed cross-cell cycles. - return await database.queryLocked( + const rows = await database.queryLocked( `SELECT * FROM relay_cells ORDER BY cell_id ASC`, [], - { failIfUnavailable } + cellInventoryLockOptions(mode) ) + return rows } private async lockGeneralCellInventory( database: RelayDatabase, - failIfUnavailable = false + mode: CellInventoryLockMode ): Promise { - return await database.queryLocked( + const rows = await database.queryLocked( `SELECT * FROM relay_cells WHERE cell_id IN ( SELECT cell_id FROM relay_cell_admission WHERE admission_state = 'general' ) ORDER BY cell_id ASC`, [], - { failIfUnavailable } + cellInventoryLockOptions(mode) ) + return rows } private async leastLoadedCell( @@ -6892,7 +6975,7 @@ export class RelayAssignmentStore { lockedCells: SqlRow[] | undefined, preferredRegion: RelayRegion ): Promise { - const rows = lockedCells ?? (await this.lockCellInventory(database)) + const rows = lockedCells ?? (await this.lockCellInventory(database, 'pool-default')) const regions = new Map( (await database.query(`SELECT cell_id, region FROM relay_cell_regions`)).map((row) => [ text(row, 'cell_id'), @@ -7507,7 +7590,7 @@ export class RelayAssignmentStore { ) { throw new Error('activity_lease_shape_mismatch') } - const cells = await this.lockCellInventory(database) + const cells = await this.lockCellInventory(database, 'request') await database.query( `DELETE FROM relay_assignment_activity_leases WHERE user_id = ? AND relay_host_id = ? AND activity_kind = 'control' @@ -7886,6 +7969,23 @@ function isDatabaseLockUnavailable(error: unknown): boolean { return error instanceof Error && error.message === 'database_lock_unavailable' } +function cellInventoryLockOptions(mode: CellInventoryLockMode): RelayLockOptions { + if (mode === 'nowait') return { failIfUnavailable: true, measureHoldMs: true } + if (mode === 'pool-default') return { measureHoldMs: true } + return { lockTimeoutMs: CELL_INVENTORY_LOCK_TIMEOUT_MS, measureHoldMs: true } +} + +// Background sweeps take the cell inventory NOWAIT so they never queue ahead of +// assignment traffic. A skipped candidate is re-derived from durable state on +// the next tick, so it is ordinary contention, not a sweep failure: one summary +// line per tick, never an error and never a quarantine. +function warnSweepCellInventoryBusy(sweep: string, skipped: number): void { + if (skipped === 0) return + console.warn( + JSON.stringify({ event: 'orca_relay_sweep_cell_inventory_busy', sweep, skipped }) + ) +} + function isDatabaseLockTimeout(error: unknown): boolean { return String((error as { code?: unknown }).code) === '55P03' } diff --git a/cloud/apps/relay/src/cell-inventory-hold-samples.ts b/cloud/apps/relay/src/cell-inventory-hold-samples.ts new file mode 100644 index 00000000000..14941032d80 --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-hold-samples.ts @@ -0,0 +1,46 @@ +// Why: the cell inventory lock is held to COMMIT, and the assignment path runs +// many statements after taking it. Tuning the request-path wait bound needs the +// hold distribution, and no runtime metric carried it before this change. +export type CellInventoryHoldCounts = { + cellInventoryHoldMsMax: number + cellInventoryHoldMsP95: number + cellInventoryHolds: number +} + +// Bounded so a flush interval with heavy assignment traffic cannot grow the array +// without limit; the reservoir keeps the most recent holds. +const MAX_SAMPLES = 2_048 + +export function emptyCellInventoryHoldCounts(): CellInventoryHoldCounts { + return { cellInventoryHoldMsMax: 0, cellInventoryHoldMsP95: 0, cellInventoryHolds: 0 } +} + +export class CellInventoryHoldSamples { + private samples: number[] = [] + + record(holdMs: number): void { + if (!Number.isFinite(holdMs) || holdMs < 0) return + if (this.samples.length === MAX_SAMPLES) this.samples.shift() + this.samples.push(holdMs) + } + + consumeCounts(): CellInventoryHoldCounts { + const counts = this.readCounts() + this.samples = [] + return counts + } + + readCounts(): CellInventoryHoldCounts { + if (this.samples.length === 0) return emptyCellInventoryHoldCounts() + const sorted = [...this.samples].sort((left, right) => left - right) + return { + cellInventoryHoldMsMax: round(sorted[sorted.length - 1]!), + cellInventoryHoldMsP95: round(sorted[Math.ceil(0.95 * sorted.length) - 1] ?? 0), + cellInventoryHolds: sorted.length + } + } +} + +function round(value: number): number { + return Number(value.toFixed(3)) +} diff --git a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts new file mode 100644 index 00000000000..d0527534935 --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts @@ -0,0 +1,164 @@ +import { readFileSync } from 'node:fs' +import { describe, expect, it } from 'vitest' +import type { CellInventoryLockMode } from './assignment-store.js' + +// Which entry points can reach a call site. A site a sweep can enter must never +// take the bounded wait: its 55P03 becomes a terminal transaction failure, and +// the incident monitor freezes on a single one. +type Reachability = 'request' | 'sweep' | 'both' + +// 'caller' is not a CellInventoryLockMode: those sites take the mode threaded +// from `assign`, which is 'request' for a client and 'pool-default' for the +// evacuateDeadCells sweep. +type CensusMode = CellInventoryLockMode | 'caller' + +type CensusEntry = { method: string; mode: CensusMode; reach: Reachability } + +// Every lockCellInventory / lockGeneralCellInventory call site in +// assignment-store.ts, in source order. A new site fails this test until it is +// classified here, which is the point. +const CENSUS: CensusEntry[] = [ + { method: 'assignStickyOnce', mode: 'caller', reach: 'both' }, + { method: 'assignOnce', mode: 'caller', reach: 'both' }, + { method: 'assignOnce', mode: 'caller', reach: 'both' }, + { method: 'assignOnce', mode: 'nowait', reach: 'both' }, + { method: 'assignOnce', mode: 'nowait', reach: 'both' }, + { method: 'assignOnce', mode: 'nowait', reach: 'both' }, + { method: 'refreshDrainMigrationLeasesOnce', mode: 'request', reach: 'request' }, + { method: 'changeActivity', mode: 'request', reach: 'request' }, + { method: 'acquireActivity', mode: 'request', reach: 'request' }, + { method: 'activateControl', mode: 'request', reach: 'request' }, + { method: 'startEvacuation', mode: 'request', reach: 'request' }, + { method: 'completeEvacuationFromDeadSourceOnce', mode: 'request', reach: 'request' }, + { method: 'completeEvacuationFromDeadSourceOnce', mode: 'nowait', reach: 'request' }, + { method: 'supersedeRegisteredEvacuationOnce', mode: 'request', reach: 'request' }, + { method: 'supersedeRegisteredEvacuationOnce', mode: 'nowait', reach: 'request' }, + { method: 'prepareRegisteredCellSupersession', mode: 'request', reach: 'request' }, + { method: 'prepareRegisteredCellSupersession', mode: 'request', reach: 'request' }, + { method: 'completeEvacuation', mode: 'nowait', reach: 'both' }, + { method: 'completeEvacuation', mode: 'pool-default', reach: 'both' }, + { method: 'rebalanceDormant', mode: 'request', reach: 'request' }, + { method: 'startRegionalRehomeCandidate', mode: 'nowait', reach: 'sweep' }, + { method: 'lockedRegionalRehomeFleetSafety', mode: 'nowait', reach: 'sweep' }, + { method: 'completeRegionalRehomeCandidate', mode: 'nowait', reach: 'sweep' }, + { method: 'abortExpiredRegionalRehomes', mode: 'nowait', reach: 'sweep' }, + { method: 'abortExpiredEvacuations', mode: 'nowait', reach: 'sweep' }, + { method: 'abortExpiredEvacuations', mode: 'nowait', reach: 'sweep' }, + { method: 'releaseExpiredActivityLeases', mode: 'nowait', reach: 'sweep' }, + { method: 'releaseExpiredActivity', mode: 'nowait', reach: 'sweep' }, + { method: 'reconcileReservationAccounting', mode: 'pool-default', reach: 'both' }, + { method: 'leastLoadedCell', mode: 'pool-default', reach: 'both' }, + { method: 'removeSupersededSameCellControls', mode: 'request', reach: 'request' } +] + +// The background sweeps, and nothing else. A method reachable from one of these +// can be entered by a sweep tick, whatever else can also enter it. +const SWEEP_ROOTS = [ + 'refreshRegionalRehomeLeases', + 'completeReadyEvacuations', + 'completeReadyRegionalRehomes', + 'abortExpiredEvacuations', + 'abortExpiredRegionalRehomes', + 'reapRegionalRehomeAttempts', + 'releaseExpiredActivityLeases', + 'releaseExpiredActivity', + 'releaseExpiredRegionPreferences', + 'evacuateDeadCells', + 'claimRegionalRehome', + 'recordRegionalRehomeDispatchFailure' +] + +const DECLARATION = /^ {2}(?:private |public )?(?:static )?(?:async )?([A-Za-z_][\w]*)[(<]/ + +function storeSource(): string[] { + return readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8').split('\n') +} + +// Why: a hand-written reachability column is a claim, not a check. Derive it, so +// a new sweep edge into a bounded site fails here instead of in production. +function sweepReachableMethods(lines: string[]): Set { + const bounds: { name: string; start: number }[] = [] + lines.forEach((line, index) => { + const declaration = DECLARATION.exec(line) + if (declaration) bounds.push({ name: declaration[1]!, start: index }) + }) + const callees = new Map>() + bounds.forEach((method, index) => { + const end = bounds[index + 1]?.start ?? lines.length + const names = callees.get(method.name) ?? new Set() + for (const call of lines.slice(method.start, end).join('\n').matchAll( + /this\.([A-Za-z_][\w]*)\s*\(/g + )) { + names.add(call[1]!) + } + callees.set(method.name, names) + }) + const reached = new Set() + const pending = [...SWEEP_ROOTS] + while (pending.length > 0) { + const name = pending.pop()! + if (reached.has(name)) continue + reached.add(name) + for (const callee of callees.get(name) ?? []) if (!reached.has(callee)) pending.push(callee) + } + return reached +} + +function readCallSites(): { method: string; mode: CensusMode }[] { + const sites: { method: string; mode: CensusMode }[] = [] + let method = '' + for (const line of storeSource()) { + const declaration = DECLARATION.exec(line) + if (declaration) method = declaration[1]! + if (/private async lock(General)?CellInventory\(/.test(line)) continue + const call = /lock(?:General)?CellInventory\(\s*\w+\s*,\s*(?:'([a-z-]+)'|(\w+))\s*\)/.exec(line) + if (!call) continue + sites.push({ method, mode: (call[1] ?? 'caller') as CensusMode }) + } + return sites +} + +describe('cell inventory lock call-site census', () => { + it('classifies every call site exactly as recorded', () => { + expect(readCallSites()).toEqual( + CENSUS.map(({ method, mode }) => ({ method, mode })) + ) + }) + + it('leaves no call site taking the inventory without naming a mode', () => { + const source = readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8') + const unclassified = source + .split('\n') + .filter((line) => /lock(?:General)?CellInventory\(\s*\w+\s*\)/.test(line)) + .filter((line) => !line.includes('private async')) + + expect(unclassified).toEqual([]) + }) + + it('derives the same reachability the census claims', () => { + const reached = sweepReachableMethods(storeSource()) + const derived = readCallSites().map(({ method }) => reached.has(method)) + + expect(derived).toEqual(CENSUS.map((entry) => entry.reach !== 'request')) + }) + + // Why: this is the whole point of the classification. A shorter wait on a + // sweep-reachable site turns contention into a terminal transaction failure, + // and relayPostgresRetryExhausted freezes the incident gate at zero. + it('never puts a sweep-reachable site on the bounded wait', () => { + const reached = sweepReachableMethods(storeSource()) + const bounded = readCallSites().filter( + (site) => site.mode === 'request' && reached.has(site.method) + ) + + expect(bounded).toEqual([]) + }) + + it('routes every sweep-only site to NOWAIT so it can skip the tick', () => { + const queueing = CENSUS.filter( + (entry) => entry.reach === 'sweep' && entry.mode !== 'nowait' + ) + + expect(queueing).toEqual([]) + }) +}) diff --git a/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts b/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts new file mode 100644 index 00000000000..783d8a62e8b --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-lock-contention.test.ts @@ -0,0 +1,541 @@ +import { readFileSync } from 'node:fs' +import { afterEach, describe, expect, it, vi } from 'vitest' + +const fakes = vi.hoisted(() => ({ + statements: [] as string[], + query: vi.fn(async (sql: string) => { + fakes.statements.push(sql) + return { rows: [], rowCount: 0 } + }), + release: vi.fn(), + end: vi.fn(async () => undefined) +})) + +vi.mock('pg', () => ({ + default: { + Pool: class { + totalCount = 1 + idleCount = 1 + waitingCount = 0 + end = fakes.end + on = vi.fn() + connect = vi.fn(async () => ({ query: fakes.query, release: fakes.release })) + } + } +})) + +const { CELL_INVENTORY_LOCK_TIMEOUT_MS, RelayAssignmentStore } = await import( + './assignment-store.js' +) +const { consumeRelayCellInventoryHold, openInMemoryRelayDatabase, openRelayDatabase, POSTGRES_LOCK_TIMEOUT_MS } = + await import('./database.js') +const RESTORE = `SET LOCAL lock_timeout = '${POSTGRES_LOCK_TIMEOUT_MS}ms'` +type RelayDatabase = import('./database.js').RelayDatabase +type RelayLockOptions = import('./database.js').RelayLockOptions +type RelayTransactionOptions = import('./database.js').RelayTransactionOptions +type SqlRow = import('./database.js').SqlRow + +const CELL_INVENTORY_SQL = 'SELECT * FROM relay_cells ORDER BY cell_id ASC' + +// The assignment path locks the general-admission subset; both forms are the +// same ordered scan of the same 23-row table and share its lock queue. +function locksCellInventory(sql: string): boolean { + return sql.trim().startsWith('SELECT * FROM relay_cells') && sql.includes('ORDER BY cell_id ASC') +} +const CELLS = [ + { id: 'cell-a', url: 'https://relay-a.example.com', capacityRequests: 10 }, + { id: 'cell-b', url: 'https://relay-b.example.com', capacityRequests: 10 } +] +const identity = { userId: 'user-a', relayHostId: 'host000000000001' } + +async function openFakePostgres(): Promise { + const database = await openRelayDatabase({ + databaseUrl: 'postgresql://relay:secret@127.0.0.1:5432/relay', + dataDir: './unused' + }) + fakes.statements.length = 0 + return database +} + +afterEach(() => { + fakes.statements.length = 0 + fakes.query.mockReset() + fakes.query.mockImplementation(async (sql: string) => { + fakes.statements.push(sql) + return { rows: [], rowCount: 0 } + }) +}) + +describe('bounded cell-inventory lock wait', () => { + // Why: a bound at or above the pool default would fence nothing, and one far + // below the hold time would convert ordinary contention into terminal failures. + it('keeps the request bound strictly inside the pool default', () => { + expect(CELL_INVENTORY_LOCK_TIMEOUT_MS).toBe(500) + expect(CELL_INVENTORY_LOCK_TIMEOUT_MS).toBeLessThan(POSTGRES_LOCK_TIMEOUT_MS) + }) + + // Why: SET LOCAL lasts to COMMIT. Left in place it would govern every later + // locked statement in the transaction and misattribute their 55P03s. + it('restores the pool default before the next statement in the transaction', async () => { + const database = await openFakePostgres() + + await database.transaction(async (transaction) => { + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { lockTimeoutMs: 150 }) + await transaction.queryLocked('SELECT * FROM relay_assignments', []) + }) + + expect(fakes.statements).toEqual([ + 'BEGIN', + "SET LOCAL lock_timeout = '150ms'", + `${CELL_INVENTORY_SQL} FOR UPDATE`, + RESTORE, + 'SELECT * FROM relay_assignments FOR UPDATE', + 'COMMIT' + ]) + await database.close() + }) + + it('restores the pool default when the bounded lock itself times out', async () => { + const database = await openFakePostgres() + fakes.query.mockImplementation(async (sql: string) => { + fakes.statements.push(sql) + if (sql.includes('FOR UPDATE')) { + throw Object.assign(new Error('lock timeout'), { code: '55P03' }) + } + return { rows: [], rowCount: 0 } + }) + + await expect( + database.transaction(async (transaction) => { + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { lockTimeoutMs: 150 }) + }) + ).rejects.toMatchObject({ code: '55P03' }) + + // The retry wrapper makes three attempts; each one must leave the default back. + expect(fakes.statements.filter((sql) => sql.startsWith('SET LOCAL'))).toEqual( + Array.from({ length: 3 }, () => ["SET LOCAL lock_timeout = '150ms'", RESTORE]).flat() + ) + await database.close() + }) + + it('rejects a lock bound that is not a positive whole number of milliseconds', async () => { + const database = await openFakePostgres() + + for (const lockTimeoutMs of [0, -1, 1.5, Number.NaN]) { + await expect( + database.transaction( + async (transaction) => + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { lockTimeoutMs }) + ) + ).rejects.toThrow('invalid_lock_timeout') + } + await database.close() + }) + + it('skips the timeout for a NOWAIT lock, which never queues', async () => { + const database = await openFakePostgres() + + await database.transaction(async (transaction) => { + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { + failIfUnavailable: true, + lockTimeoutMs: 150 + }) + }) + + expect(fakes.statements.filter((sql) => sql.startsWith('SET LOCAL'))).toEqual([]) + await database.close() + }) + + it('skips the timeout outside a transaction, where SET LOCAL cannot survive', async () => { + const database = await openFakePostgres() + + await database.queryLocked(CELL_INVENTORY_SQL, [], { lockTimeoutMs: 150 }) + + expect(fakes.statements).toEqual([`${CELL_INVENTORY_SQL} FOR UPDATE`]) + await database.close() + }) + + it('ignores the timeout on SQLite, which has no SET LOCAL', async () => { + const database = await openInMemoryRelayDatabase() + + const rows = await database.transaction( + async (transaction) => + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { lockTimeoutMs: 150 }) + ) + + expect(rows).toEqual([]) + await database.close() + }) + + // Why: testing the helper alone would pass with the store still queueing for + // the pool's one-second default. + // Why: testing the helper alone would pass with the request path still queueing + // for the pool's full second. + it('never lets a request path take the unbounded wait', async () => { + const database = await openInMemoryRelayDatabase() + const probe = new InventoryLockProbe(database) + const store = new RelayAssignmentStore(probe, () => 1_000) + await store.reconcileCells(CELLS) + probe.inventoryLocks.length = 0 + + // Assignment takes the general-admission subset; evacuation takes them all. + await store.assign(identity) + const generalLocks = probe.inventoryLocks.length + await store.startEvacuation(identity, 'cell-b') + + expect(generalLocks).toBeGreaterThan(0) + expect(probe.inventoryLocks.length).toBeGreaterThan(generalLocks) + for (const options of probe.inventoryLocks) { + const bounded = options?.lockTimeoutMs === CELL_INVENTORY_LOCK_TIMEOUT_MS + expect(bounded || options?.failIfUnavailable === true).toBe(true) + } + await database.close() + }) + + // Why: evacuateDeadCells re-enters placement from a sweep. A 55P03 there would + // be reported as a terminal sweep failure and freeze the incident gate. + it('keeps the pool default when a sweep re-enters placement', async () => { + const requestModes = await recordAssignInventoryModes(async (store) => { + await store.assign(identity) + }) + const sweepModes = await recordAssignInventoryModes(async (store) => { + await store.assign(identity, undefined, undefined, 'pool-default') + }) + + // The inventory-first retry is the lane that carries the caller's mode. + expect(requestModes).toContain(CELL_INVENTORY_LOCK_TIMEOUT_MS) + expect(sweepModes).not.toContain(CELL_INVENTORY_LOCK_TIMEOUT_MS) + expect(sweepModes.filter((mode) => mode === 'nowait').length).toBe( + requestModes.filter((mode) => mode === 'nowait').length + ) + }) + + it('sends the sweep that re-enters placement down the unbounded lane', async () => { + const database = await openInMemoryRelayDatabase() + const probe = new InventoryLockProbe(database) + let now = 1_000 + const store = new RelayAssignmentStore(probe, () => now, { + requireLiveCells: true, + heartbeatTtlMs: 45_000 + }) + await store.reconcileCells(CELLS) + for (const cell of CELLS) { + await store.recordCellHeartbeat({ + cellId: cell.id, + cellUrl: cell.url, + cellIncarnation: `1111111${cell.id.slice(-1)}-1111-4111-8111-111111111111`, + startedAt: 50, + ready: true, + observedRequests: 0 + }) + } + await store.assign(identity) + // Let every heartbeat lapse so the sweep sees the assigned cell as dead. + now += 45_001 + probe.inventoryLocks.length = 0 + probe.failActivityLockOnce = true + + await store.evacuateDeadCells() + + expect(probe.inventoryLocks).not.toEqual([]) + for (const options of probe.inventoryLocks) { + expect(options?.lockTimeoutMs).toBeUndefined() + } + await database.close() + }) + + // Why: the SQLite hold test cannot reach PostgresDatabase.transaction, which is + // the only path production ever takes. + it('records the hold on the PostgreSQL transaction path', async () => { + const database = await openFakePostgres() + + await database.transaction(async (transaction) => { + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { + lockTimeoutMs: 150, + measureHoldMs: true + }) + }) + + expect(consumeRelayCellInventoryHold(database).cellInventoryHolds).toBe(1) + await database.close() + }) + + it('records no hold for a PostgreSQL transaction that took no measured lock', async () => { + const database = await openFakePostgres() + + await database.transaction(async (transaction) => { + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { lockTimeoutMs: 150 }) + }) + + expect(consumeRelayCellInventoryHold(database).cellInventoryHolds).toBe(0) + await database.close() + }) + + // Why: index.ts boots a server on import, so its wiring can only be read. An + // unspread hold metric is invisible: the flush simply omits the fields. + it('spreads the hold counts into the runtime metrics flush', () => { + const source = readFileSync(new URL('./index.ts', import.meta.url), 'utf8') + const flush = /observability\.start\(\(\) => \(\{([^}]*)\}\)\)/.exec(source) + + expect(flush?.[1]).toContain('...consumeRelayCellInventoryHold(database)') + }) + + // Why: 500ms is a first value, not a measurement. Tuning it needs the hold + // distribution, which no runtime metric carried. + it('reports how long the inventory lock was held to COMMIT', async () => { + const database = await openInMemoryRelayDatabase() + const store = new RelayAssignmentStore(database, () => 1_000) + await store.reconcileCells(CELLS) + consumeRelayCellInventoryHold(database) + + await store.assign(identity) + + const counts = consumeRelayCellInventoryHold(database) + expect(counts.cellInventoryHolds).toBeGreaterThan(0) + expect(counts.cellInventoryHoldMsMax).toBeGreaterThanOrEqual(counts.cellInventoryHoldMsP95) + expect(counts.cellInventoryHoldMsMax).toBeGreaterThan(0) + // Consuming resets the window so the next flush reports its own holds. + expect(consumeRelayCellInventoryHold(database).cellInventoryHolds).toBe(0) + await database.close() + }) +}) + +// Why: the incident monitor freezes at zero exhausted transactions. A sweep that +// steps aside must not spend the retry budget or report a terminal failure. +describe('sweep lock skips stay off the transaction retry counters', () => { + it('reports neither a retry nor an exhaustion when NOWAIT finds the lock held', async () => { + const database = await openFakePostgres() + fakes.query.mockImplementation(async (sql: string) => { + fakes.statements.push(sql) + if (sql.includes('FOR UPDATE NOWAIT')) { + throw Object.assign(new Error('could not obtain lock'), { code: '55P03' }) + } + return { rows: [], rowCount: 0 } + }) + const events: string[] = [] + const warn = vi.spyOn(console, 'warn').mockImplementation((line: unknown) => { + try { + events.push(String((JSON.parse(line as string) as { event?: unknown }).event)) + } catch { + // non-JSON lines are not transaction telemetry + } + }) + + try { + await expect( + database.transaction(async (transaction) => { + await transaction.queryLocked(CELL_INVENTORY_SQL, [], { failIfUnavailable: true }) + }) + ).rejects.toThrow('database_lock_unavailable') + } finally { + warn.mockRestore() + } + + expect(events).not.toContain('orca_relay_postgres_transaction_retry') + expect(events).not.toContain('orca_relay_postgres_transaction_exhausted') + expect(fakes.statements.filter((sql) => sql === 'BEGIN')).toHaveLength(1) + await database.close() + }) +}) + +describe('background sweeps skip a contended cell inventory', () => { + it('takes the inventory NOWAIT and skips the tick instead of queueing', async () => { + const database = await openInMemoryRelayDatabase() + const probe = new InventoryLockProbe(database) + let now = 1_000 + const store = new RelayAssignmentStore(probe, () => now) + await store.reconcileCells(CELLS) + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + now += 24 * 60 * 60_000 + probe.inventoryLocks.length = 0 + probe.failNoWait = true + const warnings = collectWarnings('orca_relay_sweep_cell_inventory_busy') + + let aborted: number + try { + aborted = await store.abortExpiredEvacuations() + } finally { + warnings.restore() + } + + expect(aborted).toBe(0) + expect(probe.inventoryLocks).not.toEqual([]) + expect(probe.inventoryLocks.every((options) => options?.failIfUnavailable === true)).toBe( + true + ) + expect(warnings.entries).toEqual([ + { event: 'orca_relay_sweep_cell_inventory_busy', sweep: 'abort-expired-evacuations', skipped: 1 } + ]) + await database.close() + }) + + // Why: a summary line on every quiet tick would bury the contended ones. + it('says nothing on a tick that skipped no candidate', async () => { + const database = await openInMemoryRelayDatabase() + let now = 1_000 + const store = new RelayAssignmentStore(database, () => now) + await store.reconcileCells(CELLS) + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + now += 24 * 60 * 60_000 + const warnings = collectWarnings('orca_relay_sweep_cell_inventory_busy') + + let aborted: number + try { + aborted = await store.abortExpiredEvacuations() + } finally { + warnings.restore() + } + + expect(aborted).toBe(1) + expect(warnings.entries).toEqual([]) + await database.close() + }) + + it('still aborts the expired evacuation once the inventory is free', async () => { + const database = await openInMemoryRelayDatabase() + const probe = new InventoryLockProbe(database) + let now = 1_000 + const store = new RelayAssignmentStore(probe, () => now) + await store.reconcileCells(CELLS) + const assignment = await store.assign(identity) + await store.activateControl(identity, { + cellId: assignment.cellId, + assignmentEpoch: assignment.assignmentEpoch, + generation: 1 + }) + await store.startEvacuation(identity, 'cell-b') + now += 24 * 60 * 60_000 + + expect(await store.abortExpiredEvacuations()).toBe(1) + await database.close() + }) +}) + +// Returns each inventory lock the run took, as its bound or 'nowait'. +async function recordAssignInventoryModes( + drive: (store: InstanceType) => Promise +): Promise<(number | 'nowait' | 'pool-default')[]> { + const database = await openInMemoryRelayDatabase() + const probe = new InventoryLockProbe(database) + const store = new RelayAssignmentStore(probe, () => 1_000) + await store.reconcileCells(CELLS) + probe.inventoryLocks.length = 0 + probe.failActivityLockOnce = true + await drive(store) + await database.close() + return probe.inventoryLocks.map((options) => + options?.failIfUnavailable ? 'nowait' : (options?.lockTimeoutMs ?? 'pool-default') + ) +} + +function collectWarnings(event: string) { + const entries: Record[] = [] + const original = console.warn + console.warn = (line: unknown, ...rest: unknown[]) => { + try { + const parsed = JSON.parse(line as string) as Record + if (parsed.event === event) return void entries.push(parsed) + } catch { + // fall through to the real console for non-JSON lines + } + original(line, ...rest) + } + return { entries, restore: () => (console.warn = original) } +} + +const ACTIVITY_LEASE_SQL = 'SELECT * FROM relay_assignment_activity_leases' + +class InventoryLockProbe implements RelayDatabase { + readonly inventoryLocks: (RelayLockOptions | undefined)[] = [] + failNoWait = false + // Forces the next assign attempt down its inventory-first retry, the only lane + // that reaches the threaded lock mode. + failActivityLockOnce = false + + constructor(private readonly delegate: RelayDatabase) {} + + async query(sql: string, params?: unknown[]): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + if (locksCellInventory(sql)) { + this.inventoryLocks.push(options) + if (this.failNoWait && options?.failIfUnavailable) { + throw new Error('database_lock_unavailable') + } + } + if (this.failActivityLockOnce && sql.trim().startsWith(ACTIVITY_LEASE_SQL) && options?.failIfUnavailable) { + this.failActivityLockOnce = false + throw new Error('database_lock_unavailable') + } + return await this.delegate.queryLocked(sql, params, options) + } + + async transaction( + operation: (transaction: RelayDatabase) => Promise, + options?: RelayTransactionOptions + ): Promise { + return await this.delegate.transaction( + async (transaction) => await operation(new InventoryLockProbeTransaction(transaction, this)), + options + ) + } + + async close(): Promise {} +} + +class InventoryLockProbeTransaction implements RelayDatabase { + constructor( + private readonly delegate: RelayDatabase, + private readonly probe: InventoryLockProbe + ) {} + + async query(sql: string, params?: unknown[]): Promise { + return await this.delegate.query(sql, params) + } + + async queryLocked( + sql: string, + params?: unknown[], + options?: RelayLockOptions + ): Promise { + if (locksCellInventory(sql)) { + this.probe.inventoryLocks.push(options) + if (this.probe.failNoWait && options?.failIfUnavailable) { + throw new Error('database_lock_unavailable') + } + } + if ( + this.probe.failActivityLockOnce && + sql.trim().startsWith(ACTIVITY_LEASE_SQL) && + options?.failIfUnavailable + ) { + this.probe.failActivityLockOnce = false + throw new Error('database_lock_unavailable') + } + return await this.delegate.queryLocked(sql, params, options) + } + + async transaction(operation: (transaction: RelayDatabase) => Promise): Promise { + return await operation(this) + } + + async close(): Promise {} +} diff --git a/cloud/apps/relay/src/database.ts b/cloud/apps/relay/src/database.ts index f7208863f42..326ab010ccb 100644 --- a/cloud/apps/relay/src/database.ts +++ b/cloud/apps/relay/src/database.ts @@ -1,4 +1,5 @@ import { mkdirSync } from 'node:fs' +import { performance } from 'node:perf_hooks' import { join } from 'node:path' import { DatabaseSync } from 'node:sqlite' import pg from 'pg' @@ -8,9 +9,37 @@ import { type PostgresPoolPressureCounts } from './postgres-pool-pressure.js' import { applyPostgresSchema } from './postgres-schema-startup.js' +import { + CellInventoryHoldSamples, + emptyCellInventoryHoldCounts, + type CellInventoryHoldCounts +} from './cell-inventory-hold-samples.js' + +export const POSTGRES_LOCK_TIMEOUT_MS = 1_000 + +function setLocalLockTimeout(milliseconds: number): string { + if (!Number.isInteger(milliseconds) || milliseconds < 1) { + throw new Error('invalid_lock_timeout') + } + return `SET LOCAL lock_timeout = '${milliseconds}ms'` +} export type SqlRow = Record -export type RelayLockOptions = { failIfUnavailable?: boolean } +export type RelayLockOptions = { + failIfUnavailable?: boolean + // Only honoured inside a transaction: SET LOCAL is a no-op in autocommit. + lockTimeoutMs?: number + // Report how long this lock is held to COMMIT. The hold, not the wait, is what + // forms the queue, and nothing measured it before. + measureHoldMs?: boolean +} + +// A transaction that can report how long it held a measured lock before COMMIT. +type HoldMeasuringTransaction = { consumeHoldMs(): number | undefined } + +function measuredHoldMs(transaction: unknown): number | undefined { + return (transaction as HoldMeasuringTransaction).consumeHoldMs?.() +} export type RelayTransactionOptions = { reportRetries?: boolean } export interface RelayDatabase { @@ -612,9 +641,23 @@ function postgresTransactionErrorPhase(error: unknown): string { class SqliteTransaction implements RelayDatabase { readonly dialect = 'sqlite' as const + private heldFromMs: number | undefined constructor(protected readonly database: DatabaseSync) {} + consumeHoldMs(): number | undefined { + if (this.heldFromMs === undefined) return undefined + const holdMs = performance.now() - this.heldFromMs + this.heldFromMs = undefined + return holdMs + } + + protected noteHeld(options: RelayLockOptions): void { + if (options.measureHoldMs && this.heldFromMs === undefined) { + this.heldFromMs = performance.now() + } + } + async query(sql: string, params: unknown[] = []): Promise { const statement = this.database.prepare(sql) const bound = params.map((value) => (value === undefined ? null : value)) as never[] @@ -626,9 +669,11 @@ class SqliteTransaction implements RelayDatabase { async queryLocked( sql: string, params: unknown[] = [], - _options: RelayLockOptions = {} + options: RelayLockOptions = {} ): Promise { - return await this.query(sql, params) + const rows = await this.query(sql, params) + this.noteHeld(options) + return rows } async transaction( @@ -643,6 +688,11 @@ class SqliteTransaction implements RelayDatabase { class SqliteDatabase extends SqliteTransaction { private tail: Promise = Promise.resolve() + private readonly holds = new CellInventoryHoldSamples() + + consumeHoldCounts(): CellInventoryHoldCounts { + return this.holds.consumeCounts() + } override async query(sql: string, params: unknown[] = []): Promise { await this.tail @@ -655,9 +705,11 @@ class SqliteDatabase extends SqliteTransaction { this.tail = new Promise((resolve) => (release = resolve)) await previous this.database.exec('BEGIN IMMEDIATE') + const transaction = new SqliteTransaction(this.database) try { - const result = await operation(new SqliteTransaction(this.database)) + const result = await operation(transaction) this.database.exec('COMMIT') + this.holds.record(measuredHoldMs(transaction) ?? Number.NaN) return result } catch (error) { this.database.exec('ROLLBACK') @@ -675,9 +727,17 @@ class SqliteDatabase extends SqliteTransaction { class PostgresTransaction implements RelayDatabase { readonly dialect = 'postgres' as const + private heldFromMs: number | undefined constructor(protected readonly client: pg.PoolClient) {} + consumeHoldMs(): number | undefined { + if (this.heldFromMs === undefined) return undefined + const holdMs = performance.now() - this.heldFromMs + this.heldFromMs = undefined + return holdMs + } + async query(sql: string, params: unknown[] = []): Promise { try { const result = await this.client.query(postgresSql(sql), params) @@ -693,11 +753,21 @@ class PostgresTransaction implements RelayDatabase { params: unknown[] = [], options: RelayLockOptions = {} ): Promise { + // SET LOCAL lasts to COMMIT, so a bound left in place would silently govern + // every later locked statement in the transaction and misattribute its 55P03s. + const bounded = options.lockTimeoutMs !== undefined && !options.failIfUnavailable try { - return await this.query( + // A blocked waiter holds its pooled client for the whole lock_timeout, so + // hot tiny-table locks bound their own wait well under the pool default. + if (bounded) await this.query(setLocalLockTimeout(options.lockTimeoutMs!)) + const rows = await this.query( `${sql} FOR UPDATE${options.failIfUnavailable ? ' NOWAIT' : ''}`, params ) + if (options.measureHoldMs && this.heldFromMs === undefined) { + this.heldFromMs = performance.now() + } + return rows } catch (error) { if ( options.failIfUnavailable && @@ -706,6 +776,10 @@ class PostgresTransaction implements RelayDatabase { throw new Error('database_lock_unavailable') } throw error + } finally { + // Restore on the error path too: the transaction may still be retried or + // continue with unrelated locks after a caught lock failure. + if (bounded) await this.query(setLocalLockTimeout(POSTGRES_LOCK_TIMEOUT_MS)).catch(() => undefined) } } @@ -723,7 +797,6 @@ const POSTGRES_TRANSACTION_ATTEMPTS = 3 const POSTGRES_RETRY_MAX_DELAY_MS = 25 const POSTGRES_CONNECTION_TIMEOUT_MS = 2_000 const POSTGRES_STATEMENT_TIMEOUT_MS = 5_000 -const POSTGRES_LOCK_TIMEOUT_MS = 1_000 const POSTGRES_IDLE_TRANSACTION_TIMEOUT_MS = 5_000 function retryablePostgresTransactionError(error: unknown): boolean { @@ -749,6 +822,11 @@ async function waitForPostgresRetry(random: () => number = Math.random): Promise class PostgresDatabase implements RelayDatabase { readonly dialect = 'postgres' as const private readonly pressure: PostgresPoolPressure + private readonly holds = new CellInventoryHoldSamples() + + consumeHoldCounts(): CellInventoryHoldCounts { + return this.holds.consumeCounts() + } constructor(private readonly pool: pg.Pool) { this.pressure = new PostgresPoolPressure(pool) @@ -770,6 +848,8 @@ class PostgresDatabase implements RelayDatabase { options: RelayLockOptions = {} ): Promise { try { + // No transaction here, so options.lockTimeoutMs cannot apply: SET LOCAL + // would be discarded at the autocommit boundary before the lock is taken. return await this.query( `${sql} FOR UPDATE${options.failIfUnavailable ? ' NOWAIT' : ''}`, params @@ -791,10 +871,12 @@ class PostgresDatabase implements RelayDatabase { ): Promise { for (let attempt = 1; attempt <= POSTGRES_TRANSACTION_ATTEMPTS; attempt++) { const client = await this.pressure.connect() + const transaction = new PostgresTransaction(client) try { await client.query('BEGIN') - const result = await operation(new PostgresTransaction(client)) + const result = await operation(transaction) await client.query('COMMIT') + this.holds.record(measuredHoldMs(transaction) ?? Number.NaN) return result } catch (error) { await client.query('ROLLBACK').catch(() => undefined) @@ -852,6 +934,13 @@ export function consumeRelayDatabasePoolPressure( : emptyPostgresPoolPressureCounts() } +export function consumeRelayCellInventoryHold( + database: RelayDatabase +): CellInventoryHoldCounts { + const holder = database as { consumeHoldCounts?: () => CellInventoryHoldCounts } + return holder.consumeHoldCounts?.() ?? emptyCellInventoryHoldCounts() +} + export function readRelayDatabasePoolPressure( database: RelayDatabase ): PostgresPoolPressureCounts { diff --git a/cloud/apps/relay/src/index.ts b/cloud/apps/relay/src/index.ts index 635f3ae9b38..541884362c2 100644 --- a/cloud/apps/relay/src/index.ts +++ b/cloud/apps/relay/src/index.ts @@ -10,12 +10,14 @@ import { roleOwnsAssignmentMaintenance } from './cell-admission-startup.js' import { + consumeRelayCellInventoryHold, consumeRelayDatabasePoolPressure, openRelayDatabase, readRelayDatabasePoolPressure } from './database.js' import { runAssignmentCleanup } from './assignment-cleanup-steps.js' import { runRelayBackgroundOperation } from './relay-background-operation.js' +import { jitteredSweepIntervalMs } from './relay-sweep-schedule.js' import { observedRelayRequests } from './relay-observability.js' import { startRegionalRehomeWorker } from './regional-rehome-worker.js' import { createRelayServer } from './relay-server.js' @@ -54,7 +56,7 @@ const cleanupTimer = setInterval( const assignmentCleanupTimer = roleOwnsAssignmentMaintenance(config.role) ? setInterval(() => { void runAssignmentCleanup(assignments) - }, 30_000) + }, jitteredSweepIntervalMs(30_000)) : null const inventorySnapshotTimer = roleOwnsAssignmentMaintenance(config.role) ? setInterval(() => { @@ -78,7 +80,8 @@ inventorySnapshotTimer?.unref() migrationInventoryTimer?.unref() observability.start(() => ({ ...runtimeCounts(), - ...consumeRelayDatabasePoolPressure(database) + ...consumeRelayDatabasePoolPressure(database), + ...consumeRelayCellInventoryHold(database) })) const regionalRehomeWorker = startRegionalRehomeWorker(config, assignments, { safetySnapshot: () => ({ diff --git a/cloud/apps/relay/src/regional-rehome-store.test.ts b/cloud/apps/relay/src/regional-rehome-store.test.ts index 43f293b1131..6f57283396f 100644 --- a/cloud/apps/relay/src/regional-rehome-store.test.ts +++ b/cloud/apps/relay/src/regional-rehome-store.test.ts @@ -5,7 +5,12 @@ import { REGIONAL_REHOME_QUARANTINE_MS, REGIONAL_REHOME_REDRAIN_SEND_LIMIT } from './assignment-store.js' -import { openInMemoryRelayDatabase, type RelayDatabase, type SqlRow } from './database.js' +import { + openInMemoryRelayDatabase, + type RelayDatabase, + type RelayLockOptions, + type SqlRow +} from './database.js' import { REGIONAL_REHOME_SQL_FAILURES_LIMIT, REGIONAL_REHOME_SQL_FAILURES_PER_CELL_LIMIT @@ -556,6 +561,290 @@ describe('regional rehome assignment state', () => { await context.database.close() }) + it('skips a rehome dispatch tick on a contended cell inventory', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + probe.reset() + probe.failNoWait = true + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + + let attempt: unknown + try { + attempt = await context.store.claimRegionalRehome() + } finally { + busy.restore() + } + + expect(attempt).toBeNull() + expect(probe.locks).not.toEqual([]) + expect(probe.locks.every((options) => options?.failIfUnavailable === true)).toBe(true) + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'claim-regional-rehome', + skipped: 1 + } + ]) + + probe.failNoWait = false + expect(await context.store.claimRegionalRehome()).toMatchObject({ + sourceCellId: source.id, + targetCellId: target.id + }) + await context.database.close() + }) + + // Why: the redrain lane reaches the inventory through the fleet-safety read + // rather than through candidate selection, so it needs its own coverage. + // Why: one contended candidate must cost its own tick, not the whole page. The + // sweeps are explicitly per-candidate isolated for exactly this reason. + it('completes the candidates behind a contended one', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + const identities = [ + { userId: 'user-1', relayHostId: 'abcdefghijklmnop' }, + { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' } + ] + for (const identity of identities) { + // Dispatch is rate limited, so each claim needs its own interval. + context.advance(60_000) + await freshHeartbeats(context) + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.store.releaseActivity(identity, sourceControl) + } + probe.reset() + probe.failNoWaitOnce = true + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + + let completed: number + try { + completed = await context.store.completeReadyRegionalRehomes() + } finally { + busy.restore() + } + + expect(completed).toBe(1) + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'complete-ready-regional-rehomes', + skipped: 1 + } + ]) + await context.database.close() + }) + + // Why: only inventory contention is ordinary. Every other failure must keep its + // existing propagation and its dispatch-failure accounting. + it('propagates a claim failure that is not inventory contention', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + await activatePreferredSource(context, { userId: 'user-1', relayHostId: 'abcdefghijklmnop' }) + probe.reset() + probe.failWith = new Error('relay_capacity_exhausted') + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + + try { + await expect(context.store.claimRegionalRehome()).rejects.toThrow( + 'relay_capacity_exhausted' + ) + } finally { + busy.restore() + } + + expect(busy.entries).toEqual([]) + await context.database.close() + }) + + // Why: the transaction dies at the first contended candidate, so every + // candidate behind it is abandoned too. Reporting one would understate the tick. + it('reports every candidate the contended tick abandoned', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + await activatePreferredSource(context, { userId: 'user-1', relayHostId: 'abcdefghijklmnop' }) + await activatePreferredSource(context, { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' }) + await activatePreferredSource(context, { userId: 'user-3', relayHostId: 'aaaabbbbccccdddd' }) + probe.reset() + probe.failNoWait = true + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + + try { + expect(await context.store.claimRegionalRehome()).toBeNull() + } finally { + busy.restore() + } + + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'claim-regional-rehome', + skipped: 3 + } + ]) + await context.database.close() + }) + + it('skips a redrain tick on a contended cell inventory', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + context.advance(60 * 60_000 + 1) + await freshHeartbeats(context) + probe.reset() + probe.failNoWait = true + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + + let redrain: unknown + try { + redrain = await context.store.claimRegionalRehome() + } finally { + busy.restore() + } + + expect(redrain).toBeNull() + expect(probe.locks).not.toEqual([]) + expect(probe.locks.every((options) => options?.failIfUnavailable === true)).toBe(true) + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'claim-regional-rehome', + skipped: 1 + } + ]) + + probe.failNoWait = false + expect(await context.store.claimRegionalRehome()).toMatchObject({ + attemptId: attempt!.attemptId, + sendAttempts: 2 + }) + await context.database.close() + }) + + it('skips a completion tick on a contended cell inventory without quarantining it', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.store.releaseActivity(identity, sourceControl) + probe.reset() + probe.failNoWait = true + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + const failures = collectCandidateFailureWarnings() + + let completed: number + try { + completed = await context.store.completeReadyRegionalRehomes() + } finally { + failures.restore() + busy.restore() + } + + expect(completed).toBe(0) + expect(probe.locks).not.toEqual([]) + expect(probe.locks.every((options) => options?.failIfUnavailable === true)).toBe(true) + expect(failures.entries).toEqual([]) + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'complete-ready-regional-rehomes', + skipped: 1 + } + ]) + + probe.failNoWait = false + expect(await context.store.completeReadyRegionalRehomes()).toBe(1) + await context.database.close() + }) + + // Why: a contended inventory is another director settling the same row, not a + // poisoned candidate. Quarantining on it would exclude a healthy attempt from + // the sweep's LIMIT pages for 15 minutes. + it('skips an abort tick on a contended cell inventory without quarantining it', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + const targetControl = await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.store.releaseActivity(identity, sourceControl) + await context.store.releaseActivity(identity, targetControl) + context.advance(24 * 60 * 60_000) + await heartbeat(context.store, source, sourceIncarnation, 1, 2) + probe.reset() + probe.failNoWait = true + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + const failures = collectCandidateFailureWarnings() + + let aborted: number + try { + aborted = await context.store.abortExpiredRegionalRehomes() + } finally { + failures.restore() + busy.restore() + } + + expect(aborted).toBe(0) + expect(probe.locks).not.toEqual([]) + expect(probe.locks.every((options) => options?.failIfUnavailable === true)).toBe(true) + expect(failures.entries).toEqual([]) + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'abort-expired-regional-rehomes', + skipped: 1 + } + ]) + + probe.failNoWait = false + expect(await context.store.abortExpiredRegionalRehomes()).toBe(1) + await context.database.close() + }) + it('rolls back an inactive registered target only after the 24-hour bound', async () => { const context = await setup() const identity = { userId: 'user-1', relayHostId: 'abcdefghijklmnop' } @@ -1208,10 +1497,12 @@ function collectDisableWarnings() { } } -async function setup(options: { sourceProtocol?: number } = {}) { +async function setup( + options: { sourceProtocol?: number; wrap?: (database: RelayDatabase) => RelayDatabase } = {} +) { let clock = 1_000_000 const database = await openInMemoryRelayDatabase() - const store = new RelayAssignmentStore(database, () => clock, { + const store = new RelayAssignmentStore(options.wrap?.(database) ?? database, () => clock, { requireLiveCells: true, heartbeatTtlMs: 45_000 }) @@ -1397,3 +1688,43 @@ async function heartbeat( } }) } + +class CellInventoryLockProbe { + readonly locks: (RelayLockOptions | undefined)[] = [] + failNoWait = false + // Contends one candidate only, so the sweep must carry on to the next. + failNoWaitOnce = false + failWith: Error | null = null + + reset(): void { + this.locks.length = 0 + } + + wrap(database: RelayDatabase): RelayDatabase { + const probe = this + const decorate = (delegate: RelayDatabase): RelayDatabase => ({ + query: async (sql, params) => await delegate.query(sql, params), + queryLocked: async (sql, params, options) => { + if (sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC') { + probe.locks.push(options) + if (probe.failWith) throw probe.failWith + if (options?.failIfUnavailable && probe.failNoWaitOnce) { + probe.failNoWaitOnce = false + throw new Error('database_lock_unavailable') + } + if (probe.failNoWait && options?.failIfUnavailable) { + throw new Error('database_lock_unavailable') + } + } + return await delegate.queryLocked(sql, params, options) + }, + transaction: async (operation, options) => + await delegate.transaction( + async (transaction) => await operation(decorate(transaction)), + options + ), + close: async () => undefined + }) + return decorate(database) + } +} diff --git a/cloud/apps/relay/src/regional-rehome-worker.ts b/cloud/apps/relay/src/regional-rehome-worker.ts index 97c63a61025..47a2748cff4 100644 --- a/cloud/apps/relay/src/regional-rehome-worker.ts +++ b/cloud/apps/relay/src/regional-rehome-worker.ts @@ -3,6 +3,7 @@ import type { RelayAssignmentStore } from './assignment-store.js' import type { RelayConfig } from './config.js' import { googleMetadataIdentityToken } from './google-metadata-identity-token.js' import type { RegionalRehomeSafetySnapshot } from './relay-observability.js' +import { jitteredSweepIntervalMs } from './relay-sweep-schedule.js' type RegionalRehomeWorkerOptions = { fetch?: typeof fetch @@ -10,6 +11,7 @@ type RegionalRehomeWorkerOptions = { now?: () => number intervalMs?: number requestTimeoutMs?: number + random?: () => number safetySnapshot?: () => RegionalRehomeSafetySnapshot } @@ -109,7 +111,10 @@ export function startRegionalRehomeWorker( inFlight = false } } - const timer = setInterval(() => void run(), options.intervalMs ?? 1_000) + const timer = setInterval( + () => void run(), + options.intervalMs ?? jitteredSweepIntervalMs(1_000, options.random) + ) timer.unref() void run() return { diff --git a/cloud/apps/relay/src/relay-observability.ts b/cloud/apps/relay/src/relay-observability.ts index 6125ede8d1a..2266217d607 100644 --- a/cloud/apps/relay/src/relay-observability.ts +++ b/cloud/apps/relay/src/relay-observability.ts @@ -1,6 +1,7 @@ import { monitorEventLoopDelay, performance } from 'node:perf_hooks' import type { RelayRegion } from '@orca-cloud/relay-contract' import type { ControlRenewalOutcome } from './assignment-store.js' +import type { CellInventoryHoldCounts } from './cell-inventory-hold-samples.js' import type { PostgresPoolPressureCounts } from './postgres-pool-pressure.js' import type { RelayReadinessObservation } from './relay-readiness.js' @@ -20,7 +21,9 @@ export function observedRelayRequests(counts: RelayRuntimeCounts): number { return counts.preAuthConnections + counts.controls + counts.splices + counts.pendingSplices } -export type RelayProcessCounts = RelayRuntimeCounts & PostgresPoolPressureCounts +export type RelayProcessCounts = RelayRuntimeCounts & + PostgresPoolPressureCounts & + Partial export type RegionalRehomeRuntimeSafety = { observedAt: number diff --git a/cloud/apps/relay/src/relay-sweep-schedule.test.ts b/cloud/apps/relay/src/relay-sweep-schedule.test.ts new file mode 100644 index 00000000000..d5ef450cc43 --- /dev/null +++ b/cloud/apps/relay/src/relay-sweep-schedule.test.ts @@ -0,0 +1,55 @@ +import { readFileSync } from 'node:fs' +import { describe, expect, it, vi } from 'vitest' +import { startRegionalRehomeWorker } from './regional-rehome-worker.js' +import { jitteredSweepIntervalMs, SWEEP_JITTER_FRACTION } from './relay-sweep-schedule.js' + +describe('sweep schedule jitter', () => { + it('spreads instances across a bounded window above the base period', () => { + expect(jitteredSweepIntervalMs(30_000, () => 0)).toBe(30_000) + expect(jitteredSweepIntervalMs(30_000, () => 0.5)).toBe(33_000) + // Math.random() never returns 1, so the open bound is the real ceiling. + expect(jitteredSweepIntervalMs(30_000, () => 0.999)).toBeLessThan(36_000) + }) + + // Why: a shorter period would raise the very lock traffic the offset spreads. + it('never schedules a sweep sooner than its base period', () => { + for (const random of [0, 0.25, 0.5, 0.75, 0.999]) { + expect(jitteredSweepIntervalMs(1_000, () => random)).toBeGreaterThanOrEqual(1_000) + } + expect(SWEEP_JITTER_FRACTION).toBeGreaterThan(0) + }) + + it('jitters the regional rehome dispatch tick, which every director runs each second', () => { + const timers: number[] = [] + const setIntervalSpy = vi + .spyOn(globalThis, 'setInterval') + .mockImplementation(((_handler: unknown, delayMs?: number) => { + timers.push(delayMs ?? 0) + return { unref: () => undefined, [Symbol.dispose]: () => undefined } as never + }) as never) + + try { + startRegionalRehomeWorker( + { + role: 'director', + rehomeAudience: 'https://rehome.example.test', + rehomeDirectorServiceAccount: 'rehome@example.test' + } as never, + { claimRegionalRehome: async () => null } as never, + { random: () => 0.5, safetySnapshot: () => ({}) as never } + ) + } finally { + setIntervalSpy.mockRestore() + } + + expect(timers).toEqual([1_100]) + }) + + // Why: index.ts boots a server on import, so its wiring can only be read. + it('jitters the director assignment cleanup tick', () => { + const source = readFileSync(new URL('./index.ts', import.meta.url), 'utf8') + const cleanup = /runAssignmentCleanup\(assignments\)\s*\},\s*([^\n]*?)\)\n/.exec(source) + + expect(cleanup?.[1]).toBe('jitteredSweepIntervalMs(30_000)') + }) +}) diff --git a/cloud/apps/relay/src/relay-sweep-schedule.ts b/cloud/apps/relay/src/relay-sweep-schedule.ts new file mode 100644 index 00000000000..f73e6b69ead --- /dev/null +++ b/cloud/apps/relay/src/relay-sweep-schedule.ts @@ -0,0 +1,13 @@ +// Why: every director instance boots from the same rollout, so its periodic +// sweeps land on the same wall-clock second across instances and pile onto the +// one global cell-inventory lock together. A per-process offset spreads the +// arrivals; the sweeps are idempotent, so a slightly longer period is free. +export const SWEEP_JITTER_FRACTION = 0.2 + +export function jitteredSweepIntervalMs( + baseMs: number, + random: () => number = Math.random +): number { + // Only ever longer: a shorter period would raise the very load being spread. + return baseMs + Math.floor(random() * baseMs * SWEEP_JITTER_FRACTION) +} From 53adf5e2e630fa47002e70f7d909117fec57a2b4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:42:54 -0700 Subject: [PATCH 191/398] fix(git): share one failed-command error-text reader between local and the SSH relay (#18398) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(git): share one error-text reader between the local and relay branch-delete fallbacks The relay and the desktop each carried their own `getErrorText`, and they had drifted: the relay read `message` + `stderr` + `stdout`, the desktop only `message` + `stderr`. A `git branch -d` refusal arriving on `stdout` therefore routed the SSH removal through prune-and-retry while the local removal gave up and preserved the branch. Against a real binary the two agree, because Git prints the refusal through `error()` on every supported version — verified on 2.25.1, 2.38.1, 2.49.1 and 2.55.0, none of which put a byte of it on stdout. What the desktop copy actually missed is that Orca classifies errors it built itself, with the Git output on `.stdout`: `worktree remove`'s submodule retry attaches `git status --porcelain` that way on both paths. The stdout-reading form is also already the shared spelling — `isSubmoduleWorktreeRemovalRefusal` uses it for both hosts — so this converges on it rather than on the shorter one. Move the reader to src/shared/git-command-failure-text.ts and the predicate it feeds to src/shared/git-branch-delete-refusal.ts, and delete all three copies. The predicate carries both refusal wordings live in the supported range: Git through 2.40 says "checked out at", 2.43+ says "used by worktree at". The real-binary contract now pins that boundary: the refusal is recognized, it lands on stderr, and stdout stays empty on every Git in the matrix. * fix(test): consolidate the duplicate worktree import in the parity test --- src/main/git/worktree-branch-removal.ts | 7 +- src/main/git/worktree-operation-options.ts | 23 +- .../git-branch-delete-refusal-parity.test.ts | 205 ++++++++++++++++++ src/relay/git-handler-worktree-remove.ts | 24 +- src/shared/git-binary-compatibility.test.ts | 24 ++ src/shared/git-branch-delete-refusal.ts | 16 ++ src/shared/git-command-failure-text.ts | 27 +++ src/shared/worktree/submodule-removal.ts | 18 +- 8 files changed, 281 insertions(+), 63 deletions(-) create mode 100644 src/relay/git-branch-delete-refusal-parity.test.ts create mode 100644 src/shared/git-branch-delete-refusal.ts create mode 100644 src/shared/git-command-failure-text.ts diff --git a/src/main/git/worktree-branch-removal.ts b/src/main/git/worktree-branch-removal.ts index 35385254bca..dc09311596c 100644 --- a/src/main/git/worktree-branch-removal.ts +++ b/src/main/git/worktree-branch-removal.ts @@ -7,12 +7,9 @@ import { withLocalGitCapabilityCacheForExecution } from './git-capability-state' import { withRepoRefMaintenancePaused } from './local-repo-ref-maintenance' import { gitExecFileAsync } from './runner' import { parseWorktreeList } from '../../shared/git-worktree-porcelain-parser' +import { isBranchCheckedOutInWorktreeError } from '../../shared/git-branch-delete-refusal' import type { GitWorktreeExecOptions, RemoveWorktreeOptions } from './worktree-operation-options' -import { - gitExecOptions, - isBranchCheckedOutInWorktreeError, - normalizeLocalBranchRef -} from './worktree-operation-options' +import { gitExecOptions, normalizeLocalBranchRef } from './worktree-operation-options' export async function deleteBranchAfterWorktreeRemoval( repoPath: string, diff --git a/src/main/git/worktree-operation-options.ts b/src/main/git/worktree-operation-options.ts index 9376fe63f85..6f956c6c422 100644 --- a/src/main/git/worktree-operation-options.ts +++ b/src/main/git/worktree-operation-options.ts @@ -2,6 +2,7 @@ import type { LocalBaseRefRefreshResult, LocalBaseRefUpdateSuggestion } from '../../shared/worktree/base-ref-drift-types' +import { readGitCommandFailureText } from '../../shared/git-command-failure-text' import type { RemoveWorktreeResult } from '../../shared/worktree/create-types' import type { GitWorktreeInfo } from '../../shared/worktree/types' @@ -95,28 +96,8 @@ export function getErrorCode(error: unknown): string | undefined { : undefined } -function getErrorText(error: unknown): string { - if (typeof error === 'object' && error !== null) { - const parts: string[] = [] - if ('message' in error && typeof error.message === 'string') { - parts.push(error.message) - } - if ('stderr' in error && typeof error.stderr === 'string') { - parts.push(error.stderr) - } - return parts.join('\n') - } - return String(error) -} - export function isNotGitRepositoryError(error: unknown): boolean { - return /not a git repository/i.test(getErrorText(error)) -} - -export function isBranchCheckedOutInWorktreeError(error: unknown): boolean { - return /cannot delete branch .*(?:used by worktree|checked out)|branch .*is checked out/i.test( - getErrorText(error) - ) + return /not a git repository/i.test(readGitCommandFailureText(error)) } export function normalizeLocalBranchRef(branch: string): string { diff --git a/src/relay/git-branch-delete-refusal-parity.test.ts b/src/relay/git-branch-delete-refusal-parity.test.ts new file mode 100644 index 00000000000..71344bca940 --- /dev/null +++ b/src/relay/git-branch-delete-refusal-parity.test.ts @@ -0,0 +1,205 @@ +/** + * The relay and the desktop each carried their own `getErrorText`, and they had + * drifted: the relay read `message` + `stderr` + `stdout`, the desktop only + * `message` + `stderr`. So a `git branch -d` refusal that arrived on `stdout` + * routed the SSH removal through prune-and-retry while the local removal gave up + * and preserved the branch. + * + * These tests push the same failure through both published removal entry points — + * `removeWorktreeOp` (what `git.removeWorktree` runs on the host) and `removeWorktree` + * (the local runner) — and require the same branch-deletion commands and the same + * `RemoveWorktreeResult`. A second error-text reader on either side fails here. + */ +import type * as FsPromises from 'node:fs/promises' +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const { gitExecFileAsyncMock, resolveGitDirMock, moveWorktreeDirectoryToTrashMock } = vi.hoisted( + () => ({ + gitExecFileAsyncMock: vi.fn(), + resolveGitDirMock: vi.fn(), + moveWorktreeDirectoryToTrashMock: vi.fn() + }) +) + +vi.mock('../main/worktree-trash', () => ({ + moveWorktreeDirectoryToTrash: moveWorktreeDirectoryToTrashMock, + restoreWorktreeDirectoryFromTrash: vi.fn(async () => true), + scheduleWorktreeTrashDeletion: vi.fn() +})) + +vi.mock('../main/git/runner', () => ({ + gitExecFileAsync: gitExecFileAsyncMock, + gitExecFileSync: vi.fn(), + translateWslOutputPaths: (output: string) => output +})) + +vi.mock('../main/git/status', () => ({ + resolveGitDir: resolveGitDirMock, + runWithGitReadCacheInvalidation: (run: () => Promise) => run() +})) + +vi.mock('fs/promises', async () => { + const actual = await vi.importActual('fs/promises') + return { + ...actual, + stat: vi.fn(async () => { + throw enoent() + }), + readFile: vi.fn() + } +}) + +import { GitCapabilityCache } from '../shared/git-capability-cache' +import type { RemoveWorktreeResult } from '../shared/worktree/create-types' +import { clearGitCapabilityStateForTests } from '../main/git/git-capability-state' +import { _resetWorktreeScanCacheForTests, removeWorktree } from '../main/git/worktree' +import { __resetSparseCheckoutStateCacheForTests } from '../main/git/worktree-sparse-checkout-cache' +import type { GitExec } from './git-handler-ops' +import { removeWorktreeOp } from './git-handler-worktree-ops' + +const REPO_PATH = '/repo' +const WORKTREE_PATH = '/repo-feature' +const BRANCH = 'feature/test' + +function enoent(): Error { + return Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) +} + +/** Only the branch-deletion phase; the two entry points legitimately reach it by different routes. */ +function branchDeletionCalls(calls: string[][]): string[] { + return calls + .map((args) => args.join(' ')) + .filter((call) => call.startsWith('branch ') || call === 'worktree prune') +} + +function worktreeListPorcelain(withFeature: boolean): string { + const blocks = [[`worktree ${REPO_PATH}`, 'HEAD abc123', 'branch refs/heads/main']] + if (withFeature) { + blocks.push([`worktree ${WORKTREE_PATH}`, 'HEAD def456', `branch refs/heads/${BRANCH}`]) + } + return `${blocks.map((block) => block.join('\n')).join('\n\n')}\n` +} + +type RefusalStream = 'stdout' | 'stderr' + +const REFUSAL_TEXT = `error: cannot delete branch '${BRANCH}' used by worktree at '/repo-stale'` + +/** + * A `branch -d` rejection carrying the refusal on exactly one stream. `message` stays + * generic so the assertion is about the stream, not about Node's stderr echo. + */ +function branchDeleteRefusal(stream: RefusalStream): Error { + return Object.assign(new Error('Command failed: git branch -d'), { + code: 1, + stdout: stream === 'stdout' ? REFUSAL_TEXT : '', + stderr: stream === 'stderr' ? REFUSAL_TEXT : '' + }) +} + +/** Refuses the first `branch -d`, accepts the retry that follows `worktree prune`. */ +function scriptRelayGit(stream: RefusalStream): { + git: GitExec + calls: string[][] +} { + const calls: string[][] = [] + let branchDeleteCount = 0 + const git = vi.fn(async (args) => { + calls.push(args) + if (args[0] === 'rev-parse') { + return { stdout: `${REPO_PATH}/.git\n`, stderr: '' } + } + if (args[0] === 'worktree' && args[1] === 'list') { + return { stdout: worktreeListPorcelain(true), stderr: '' } + } + if (args[0] === 'branch' && args[1] === '-d') { + branchDeleteCount += 1 + if (branchDeleteCount === 1) { + throw branchDeleteRefusal(stream) + } + return { stdout: '', stderr: '' } + } + return { stdout: '', stderr: '' } + }) + return { git, calls } +} + +function scriptDesktopGit(stream: RefusalStream): string[][] { + const calls: string[][] = [] + let branchDeleteCount = 0 + gitExecFileAsyncMock.mockImplementation(async (args: string[]) => { + calls.push(args) + if (args[0] === 'worktree' && args[1] === 'list') { + return { stdout: worktreeListPorcelain(branchDeleteCount === 0), stderr: '' } + } + if (args[0] === 'branch' && args[1] === '-d') { + branchDeleteCount += 1 + if (branchDeleteCount === 1) { + throw branchDeleteRefusal(stream) + } + return { stdout: '', stderr: '' } + } + return { stdout: '', stderr: '' } + }) + return calls +} + +async function removeOverRelay( + stream: RefusalStream +): Promise<{ result: RemoveWorktreeResult; branchCalls: string[] }> { + const { git, calls } = scriptRelayGit(stream) + const result = await removeWorktreeOp( + git, + { worktreePath: WORKTREE_PATH }, + new GitCapabilityCache() + ) + return { result, branchCalls: branchDeletionCalls(calls) } +} + +async function removeLocally( + stream: RefusalStream +): Promise<{ result: RemoveWorktreeResult; branchCalls: string[] }> { + const calls = scriptDesktopGit(stream) + const result = await removeWorktree(REPO_PATH, WORKTREE_PATH) + return { result, branchCalls: branchDeletionCalls(calls) } +} + +beforeEach(() => { + clearGitCapabilityStateForTests() + _resetWorktreeScanCacheForTests() + __resetSparseCheckoutStateCacheForTests() + gitExecFileAsyncMock.mockReset() + resolveGitDirMock.mockReset() + resolveGitDirMock.mockImplementation(async (worktreePath: string) => `${worktreePath}/.git`) + moveWorktreeDirectoryToTrashMock.mockReset() + // Default: the checkout cannot be renamed aside, so removal runs `worktree remove` in place. + moveWorktreeDirectoryToTrashMock.mockResolvedValue(undefined) +}) + +describe('relay/desktop branch-delete refusal parity', () => { + it('prunes and retries on both paths when the refusal arrives on stdout', async () => { + const relay = await removeOverRelay('stdout') + const local = await removeLocally('stdout') + + expect(relay.branchCalls).toEqual(local.branchCalls) + expect(relay.result).toEqual(local.result) + expect(local.branchCalls).toEqual([ + `branch -d -- ${BRANCH}`, + 'worktree prune', + `branch -d -- ${BRANCH}` + ]) + expect(local.result).toEqual({}) + }) + + it('prunes and retries on both paths when the refusal arrives on stderr, as real Git sends it', async () => { + const relay = await removeOverRelay('stderr') + const local = await removeLocally('stderr') + + expect(relay.branchCalls).toEqual(local.branchCalls) + expect(relay.result).toEqual(local.result) + expect(local.branchCalls).toEqual([ + `branch -d -- ${BRANCH}`, + 'worktree prune', + `branch -d -- ${BRANCH}` + ]) + }) +}) diff --git a/src/relay/git-handler-worktree-remove.ts b/src/relay/git-handler-worktree-remove.ts index 474e03bab88..bf8e65a066e 100644 --- a/src/relay/git-handler-worktree-remove.ts +++ b/src/relay/git-handler-worktree-remove.ts @@ -1,5 +1,6 @@ import * as path from 'node:path' import type { RemoveWorktreeResult } from '../shared/worktree/create-types' +import { isBranchCheckedOutInWorktreeError } from '../shared/git-branch-delete-refusal' import { assertWorktreeUnlockedForRemoval } from '../shared/worktree/removal' import { isSubmoduleWorktreeRemovalRefusal } from '../shared/worktree/submodule-removal' import { deleteAlreadyMergedRelayBranchAfterSafeDeleteFailure } from './git-handler-branch-cleanup' @@ -7,29 +8,6 @@ import type { GitExec } from './git-handler-ops' import type { GitCapabilityCache } from '../shared/git-capability-cache' import { readRelayWorktreeList } from './git-handler-worktree-list' -function getErrorText(error: unknown): string { - if (typeof error === 'object' && error !== null) { - const parts: string[] = [] - if ('message' in error && typeof error.message === 'string') { - parts.push(error.message) - } - if ('stderr' in error && typeof error.stderr === 'string') { - parts.push(error.stderr) - } - if ('stdout' in error && typeof error.stdout === 'string') { - parts.push(error.stdout) - } - return parts.join('\n') - } - return String(error) -} - -function isBranchCheckedOutInWorktreeError(error: unknown): boolean { - return /cannot delete branch .*(?:used by worktree|checked out)|branch .*is checked out/i.test( - getErrorText(error) - ) -} - function normalizeLocalBranchRef(branch: string): string { return branch.replace(/^refs\/heads\//, '') } diff --git a/src/shared/git-binary-compatibility.test.ts b/src/shared/git-binary-compatibility.test.ts index 5ad37398d14..6387f200b4c 100644 --- a/src/shared/git-binary-compatibility.test.ts +++ b/src/shared/git-binary-compatibility.test.ts @@ -8,6 +8,7 @@ import { isUnsupportedMergeTreeMergeBaseError, isUnsupportedMergeTreeWriteTreeError } from './git-merge-tree-capability' +import { isBranchCheckedOutInWorktreeError } from './git-branch-delete-refusal' import { isForEachRefExcludeUnsupportedError } from './git-ref-command-capabilities' import { isNoWriteFetchHeadUnsupportedError } from './git-fetch-head-capability' import { @@ -140,6 +141,29 @@ describeBinaryCompatibility('real Git binary compatibility', () => { ).resolves.toBeDefined() }) + // Why pin this: worktree removal decides whether to prune and retry `branch -d` by + // matching Git's refusal text, and the wording moved inside the supported range + // (<=2.40 "Cannot delete branch 'x' checked out at", >=2.43 "cannot delete branch 'x' + // used by worktree at"). It is also the only evidence that the refusal is a stderr + // message on every supported Git rather than something a caller could read off stdout. + it('refuses to delete a branch another worktree holds, on stderr, in a recognized wording', async () => { + await runGit(['worktree', 'add', '-b', 'compat-held', 'held-wt']) + try { + const refusal = await runGit(['branch', '-d', '--', 'compat-held']).then( + () => null, + (error: unknown) => error + ) + expect(refusal).not.toBeNull() + expect(isBranchCheckedOutInWorktreeError(refusal)).toBe(true) + const streams = refusal as { stdout?: string; stderr?: string } + expect(streams.stderr ?? '').toMatch(/delete branch .*compat-held/i) + expect(streams.stdout ?? '').toBe('') + } finally { + await runGit(['worktree', 'remove', '--force', 'held-wt']) + await runGit(['branch', '-D', 'compat-held']) + } + }) + it('deregisters a worktree whose directory was renamed away', async () => { // Orca renames the checkout into a trash directory and then clears the registration, so every // supported Git must accept `worktree remove --force` on the now-missing path. diff --git a/src/shared/git-branch-delete-refusal.ts b/src/shared/git-branch-delete-refusal.ts new file mode 100644 index 00000000000..30d03746c2d --- /dev/null +++ b/src/shared/git-branch-delete-refusal.ts @@ -0,0 +1,16 @@ +import { readGitCommandFailureText } from './git-command-failure-text' + +/** + * `git branch -d/-D` refused because the branch is the HEAD of some worktree. + * + * Both wordings are live in Orca's supported range: Git through 2.40 says + * "Cannot delete branch 'x' checked out at ''", and 2.43+ says "cannot delete + * branch 'x' used by worktree at ''". Every version prints it through `error()`, + * so it arrives on stderr. Callers treat a match as "the blocker may be a stale + * worktree record", prune, and retry once. + */ +export function isBranchCheckedOutInWorktreeError(error: unknown): boolean { + return /cannot delete branch .*(?:used by worktree|checked out)|branch .*is checked out/i.test( + readGitCommandFailureText(error) + ) +} diff --git a/src/shared/git-command-failure-text.ts b/src/shared/git-command-failure-text.ts new file mode 100644 index 00000000000..1ebfd322cfe --- /dev/null +++ b/src/shared/git-command-failure-text.ts @@ -0,0 +1,27 @@ +/** + * The text a failed Git invocation left behind, for the predicates that classify a + * failure by what Git said. + * + * Why all three streams and not just `message` + `stderr`: the errors Orca classifies + * do not all come straight out of `execFile`. Node puts Git's stderr in both `message` + * and `stderr`, but Orca also throws its own failures with the Git output on `stdout` + * (`worktree remove`'s submodule retry attaches `git status --porcelain` output that + * way on both the local runner and the relay). Reading all three is what keeps the + * local and relay classifiers from disagreeing about the same error object. + * + * Against a real binary this reads no differently: Git emits every refusal this module + * classifies through `error()`/`die()`, i.e. stderr only, on 2.25 through 2.55. + */ +export function readGitCommandFailureText(error: unknown): string { + if (typeof error !== 'object' || error === null) { + return String(error) + } + const parts: string[] = [] + for (const field of ['message', 'stderr', 'stdout'] as const) { + const value = (error as Record)[field] + if (typeof value === 'string' && value) { + parts.push(value) + } + } + return parts.join('\n') +} diff --git a/src/shared/worktree/submodule-removal.ts b/src/shared/worktree/submodule-removal.ts index 8a6f91a2cf8..2b9306c4650 100644 --- a/src/shared/worktree/submodule-removal.ts +++ b/src/shared/worktree/submodule-removal.ts @@ -1,16 +1,4 @@ -function getErrorText(error: unknown): string { - if (typeof error === 'object' && error !== null) { - const parts: string[] = [] - for (const field of ['message', 'stderr', 'stdout'] as const) { - const value = (error as Record)[field] - if (typeof value === 'string' && value) { - parts.push(value) - } - } - return parts.join('\n') - } - return String(error) -} +import { readGitCommandFailureText } from '../git-command-failure-text' // Why: `git worktree remove` (non-force) categorically refuses any worktree // containing an initialised submodule, even when parent and submodule are @@ -18,5 +6,7 @@ function getErrorText(error: unknown): string { // cleanliness and retry with --force. Both the local runner and the relay pin // English git output (UNTRANSLATED_GIT_OUTPUT_ENV), so text matching is stable. export function isSubmoduleWorktreeRemovalRefusal(error: unknown): boolean { - return /working trees containing submodules cannot be moved or removed/i.test(getErrorText(error)) + return /working trees containing submodules cannot be moved or removed/i.test( + readGitCommandFailureText(error) + ) } From 5a626dcdf48e723962f268c8f7adca77c4fd5a32 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:42:58 -0700 Subject: [PATCH 192/398] refactor(git): share push-target resolution between local and the SSH relay (#18406) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `src/relay/git-handler-push-target.ts` and `src/main/git/remote.ts` carried identical ~160-line copies of the resolver that decides which remote a plain `git push` hits. Identical today is exactly when to share it: the cost of a future divergence is pushing to the wrong remote, which retrying does not undo. Move the resolver to src/shared/git-push-target-resolution.ts, parameterized on a `(args) => Promise<{ stdout }>` runner — the only thing the two hosts actually differ in — and delete both copies. The relay entry point keeps only the work that is genuinely relay-side: re-validating an explicit target that arrived over the wire and running `check-ref-format` on it. No behavior change on either path, and nothing new or different is published, so this engages no rule in remote-wire-compatibility. No git command changes. src/relay/git-push-target-local-parity.test.ts scripts one repository's config and requires `git.push` over the real relay dispatcher and the desktop's `gitPush` to emit the same push argv, plus the argv each case should produce. --- src/main/git/remote.ts | 158 +------------ src/relay/git-handler-push-target.ts | 160 +------------- .../git-push-target-local-parity.test.ts | 209 ++++++++++++++++++ src/shared/git-push-target-resolution.ts | 140 ++++++++++++ 4 files changed, 361 insertions(+), 306 deletions(-) create mode 100644 src/relay/git-push-target-local-parity.test.ts create mode 100644 src/shared/git-push-target-resolution.ts diff --git a/src/main/git/remote.ts b/src/main/git/remote.ts index 20cf8415d04..aa7932687ca 100644 --- a/src/main/git/remote.ts +++ b/src/main/git/remote.ts @@ -3,8 +3,7 @@ import { runPullWithDivergenceFallback } from '../../shared/git-remote-error' import { resolveEffectiveGitUpstream } from '../../shared/git-effective-upstream' -import { gitRefTargetsBranchOnRemote } from '../../shared/git-remote-branch-name' -import { findGitRemoteNameByFetchUrl } from '../../shared/git-remote-url-index' +import { resolveConfiguredGitPushTarget } from '../../shared/git-push-target-resolution' import type { GitPushTarget } from '../../shared/worktree/types' import type { GitRuntimeOptions } from './git-runtime-options' import { gitOptionsForWorktree } from './git-runtime-options' @@ -20,157 +19,6 @@ import { runWithGitWorktreeOperationLock } from '../../shared/git-worktree-opera export { gitPullRebaseFromBase } from './remote-rebase' -async function getConfiguredPushTarget( - worktreePath: string, - options: GitRuntimeOptions = {} -): Promise<{ remote: string; refspec: string } | null> { - try { - const { stdout: branchStdout } = await gitExecFileAsync( - ['symbolic-ref', '--quiet', '--short', 'HEAD'], - gitOptionsForWorktree(worktreePath, options) - ) - const branch = branchStdout.trim() - if (!branch) { - return null - } - - const [pushRemote, { stdout: mergeStdout }] = await Promise.all([ - getConfiguredPushRemote(worktreePath, branch, options), - gitExecFileAsync( - ['config', '--get', `branch.${branch}.merge`], - gitOptionsForWorktree(worktreePath, options) - ) - ]) - const remote = pushRemote?.remote - const mergeRef = mergeStdout.trim() - const branchRef = mergeRef.replace(/^refs\/heads\//, '') - if (!remote || !branchRef || remote === '.' || branchRef === mergeRef) { - return null - } - if (await branchMergeTargetsConfiguredBase(worktreePath, branch, remote, branchRef, options)) { - return null - } - if (!canPushConfiguredMergeBranch(pushRemote, branch, branchRef)) { - return null - } - return { remote, refspec: `HEAD:${branchRef}` } - } catch { - return null - } -} - -async function getConfigValue( - worktreePath: string, - key: string, - options: GitRuntimeOptions = {} -): Promise { - try { - const { stdout } = await gitExecFileAsync( - ['config', '--get', key], - gitOptionsForWorktree(worktreePath, options) - ) - const value = stdout.trim() - return value || null - } catch { - return null - } -} - -function isUrlValuedRemote(remote: string): boolean { - return /^[A-Za-z][A-Za-z0-9+.-]*:\/\//.test(remote) || /^[^@/:]+@[^:]+:.+/.test(remote) -} - -type ConfiguredPushRemote = { - remote: string - branchRemote: string | null -} - -// One `git remote -v` instead of `git remote` plus a serial `git remote get-url` -// per remote; both print the same insteadOf-expanded fetch URL. -async function findRemoteNameForUrl( - worktreePath: string, - remoteUrl: string, - options: GitRuntimeOptions = {} -): Promise { - try { - const { stdout } = await gitExecFileAsync( - ['remote', '-v'], - gitOptionsForWorktree(worktreePath, options) - ) - return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => candidateUrl === remoteUrl) - } catch { - return null - } -} - -async function normalizePushRemote( - worktreePath: string, - remote: string, - options: GitRuntimeOptions = {} -): Promise { - if (!isUrlValuedRemote(remote)) { - return remote - } - return (await findRemoteNameForUrl(worktreePath, remote, options)) ?? remote -} - -async function getConfiguredPushRemote( - worktreePath: string, - branch: string, - options: GitRuntimeOptions = {} -): Promise { - const branchRemote = await getConfigValue(worktreePath, `branch.${branch}.remote`, options) - const remote = - (await getConfigValue(worktreePath, `branch.${branch}.pushRemote`, options)) ?? - (await getConfigValue(worktreePath, 'remote.pushDefault', options)) ?? - branchRemote - if (!remote) { - return null - } - const normalizedRemote = await normalizePushRemote(worktreePath, remote, options) - // The two usually name the same URL; resolving it twice reads the remote table twice. - if (!branchRemote) { - return { remote: normalizedRemote, branchRemote: null } - } - return { - remote: normalizedRemote, - branchRemote: - branchRemote === remote - ? normalizedRemote - : await normalizePushRemote(worktreePath, branchRemote, options) - } -} - -async function branchMergeTargetsConfiguredBase( - worktreePath: string, - branch: string, - remote: string, - branchRef: string, - options: GitRuntimeOptions = {} -): Promise { - return gitRefTargetsBranchOnRemote( - await getConfigValue(worktreePath, `branch.${branch}.base`, options), - remote, - branchRef - ) -} - -function canPushConfiguredMergeBranch( - pushRemote: ConfiguredPushRemote | null, - branch: string, - branchRef: string -): boolean { - if (!pushRemote) { - return false - } - if (branchRef === branch) { - return true - } - // Why: branch.merge belongs to branch.remote. A pushDefault fork must not - // inherit origin/main as its destination branch. - return pushRemote.remote !== 'origin' && pushRemote.branchRemote === pushRemote.remote -} - function explicitPushTarget(target: GitPushTarget): { remote: string; refspec: string } { return { remote: target.remoteName, refspec: `HEAD:${target.branchName}` } } @@ -197,7 +45,9 @@ export async function gitPush( // from worktree config, not the upstream relationship. const target = pushTarget ? explicitPushTarget(pushTarget) - : await getConfiguredPushTarget(worktreePath, options) + : await resolveConfiguredGitPushTarget((args) => + gitExecFileAsync(args, gitOptionsForWorktree(worktreePath, options)) + ) const args = [ 'push', ...(options.forceWithLease ? ['--force-with-lease'] : []), diff --git a/src/relay/git-handler-push-target.ts b/src/relay/git-handler-push-target.ts index d81111d45d3..6663b5b3ad3 100644 --- a/src/relay/git-handler-push-target.ts +++ b/src/relay/git-handler-push-target.ts @@ -1,168 +1,24 @@ import { assertGitPushTargetShape } from '../shared/git-push-target-validation' -import { gitRefTargetsBranchOnRemote } from '../shared/git-remote-branch-name' -import { findGitRemoteNameByFetchUrl } from '../shared/git-remote-url-index' +import { + resolveConfiguredGitPushTarget, + type ResolvedGitPushTarget +} from '../shared/git-push-target-resolution' import type { GitPushTarget } from '../shared/worktree/types' type RelayGit = (args: string[], cwd: string) => Promise<{ stdout: string; stderr: string }> -export type ResolvedPushTarget = { - remote: string - refspec: string -} - -async function getConfiguredPushTarget( - git: RelayGit, - worktreePath: string -): Promise { - try { - const { stdout: branchStdout } = await git( - ['symbolic-ref', '--quiet', '--short', 'HEAD'], - worktreePath - ) - const branch = branchStdout.trim() - if (!branch) { - return null - } - const [pushRemote, { stdout: mergeStdout }] = await Promise.all([ - getConfiguredPushRemote(git, worktreePath, branch), - git(['config', '--get', `branch.${branch}.merge`], worktreePath) - ]) - const remote = pushRemote?.remote - const mergeRef = mergeStdout.trim() - const branchRef = mergeRef.replace(/^refs\/heads\//, '') - if (!remote || !branchRef || remote === '.' || branchRef === mergeRef) { - return null - } - if (await branchMergeTargetsConfiguredBase(git, worktreePath, branch, remote, branchRef)) { - return null - } - if (!canPushConfiguredMergeBranch(pushRemote, branch, branchRef)) { - return null - } - return { remote, refspec: `HEAD:${branchRef}` } - } catch { - return null - } -} - -async function getConfigValue( - git: RelayGit, - worktreePath: string, - key: string -): Promise { - try { - const { stdout } = await git(['config', '--get', key], worktreePath) - const value = stdout.trim() - return value || null - } catch { - return null - } -} - -function isUrlValuedRemote(remote: string): boolean { - return /^[A-Za-z][A-Za-z0-9+.-]*:\/\//.test(remote) || /^[^@/:]+@[^:]+:.+/.test(remote) -} - -type ConfiguredPushRemote = { - remote: string - branchRemote: string | null -} - -// Host-side twin of `src/main/git/remote.ts`: one `git remote -v` instead of -// `git remote` plus a serial `git remote get-url` per remote. -async function findRemoteNameForUrl( - git: RelayGit, - worktreePath: string, - remoteUrl: string -): Promise { - try { - const { stdout } = await git(['remote', '-v'], worktreePath) - return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => candidateUrl === remoteUrl) - } catch { - return null - } -} - -async function normalizePushRemote( - git: RelayGit, - worktreePath: string, - remote: string -): Promise { - if (!isUrlValuedRemote(remote)) { - return remote - } - return (await findRemoteNameForUrl(git, worktreePath, remote)) ?? remote -} - -async function getConfiguredPushRemote( - git: RelayGit, - worktreePath: string, - branch: string -): Promise { - // Why: mirror the local gitPush resolver so SSH worktrees do not drift to a - // different target when branch.pushRemote or remote.pushDefault is present. - const branchRemote = await getConfigValue(git, worktreePath, `branch.${branch}.remote`) - const remote = - (await getConfigValue(git, worktreePath, `branch.${branch}.pushRemote`)) ?? - (await getConfigValue(git, worktreePath, 'remote.pushDefault')) ?? - branchRemote - if (!remote) { - return null - } - const normalizedRemote = await normalizePushRemote(git, worktreePath, remote) - // The two usually name the same URL; resolving it twice reads the remote table twice. - if (!branchRemote) { - return { remote: normalizedRemote, branchRemote: null } - } - return { - remote: normalizedRemote, - branchRemote: - branchRemote === remote - ? normalizedRemote - : await normalizePushRemote(git, worktreePath, branchRemote) - } -} - -async function branchMergeTargetsConfiguredBase( - git: RelayGit, - worktreePath: string, - branch: string, - remote: string, - branchRef: string -): Promise { - return gitRefTargetsBranchOnRemote( - await getConfigValue(git, worktreePath, `branch.${branch}.base`), - remote, - branchRef - ) -} - -function canPushConfiguredMergeBranch( - pushRemote: ConfiguredPushRemote | null, - branch: string, - branchRef: string -): boolean { - if (!pushRemote) { - return false - } - if (branchRef === branch) { - return true - } - // Why: branch.merge belongs to branch.remote. A pushDefault fork must not - // inherit origin/main as its destination branch. - return pushRemote.remote !== 'origin' && pushRemote.branchRemote === pushRemote.remote -} - export async function resolveRelayPushTarget( git: RelayGit, worktreePath: string, pushTarget: unknown -): Promise { +): Promise { if (pushTarget === undefined) { - return getConfiguredPushTarget(git, worktreePath) + return resolveConfiguredGitPushTarget((args) => git(args, worktreePath)) } assertGitPushTargetShape(pushTarget) const explicitTarget: GitPushTarget = pushTarget + // Why here and not in the shared resolver: an explicit target arrives over the wire, + // so the host re-validates its shape and asks Git to vet the branch name itself. await git(['check-ref-format', '--branch', explicitTarget.branchName], worktreePath) return { remote: explicitTarget.remoteName, diff --git a/src/relay/git-push-target-local-parity.test.ts b/src/relay/git-push-target-local-parity.test.ts new file mode 100644 index 00000000000..152b062d88e --- /dev/null +++ b/src/relay/git-push-target-local-parity.test.ts @@ -0,0 +1,209 @@ +/** + * Push-target resolution decides which remote a plain `git push` hits, and a wrong + * answer is not recoverable by retrying. The relay and the desktop used to carry + * identical ~160-line copies of it; they now share one implementation. + * + * These tests script one repository's Git config and require `git.push` over the real + * relay dispatcher and the desktop's `gitPush` to emit the *same push argv*, plus the + * argv each case is supposed to produce — so a second implementation on either side + * fails here even if it is wrong in the same direction on both. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const { gitExecFileAsyncMock } = vi.hoisted(() => ({ gitExecFileAsyncMock: vi.fn() })) + +vi.mock('../main/git/runner', () => ({ + gitExecFileAsync: gitExecFileAsyncMock +})) + +import { gitPush } from '../main/git/remote' +import { RelayContext } from './context' +import { GitHandler } from './git-handler' +import { createMockDispatcher, type RelayDispatcher } from './git-handler-test-setup' + +const WORKTREE_PATH = '/worktree' + +type GitConfigFixture = { + /** Empty means detached HEAD: `symbolic-ref --quiet --short HEAD` prints nothing. */ + branch: string + merge?: string + branchRemote?: string + pushRemote?: string + pushDefault?: string + base?: string + /** remote name -> fetch URL, as `git remote -v` prints it. */ + remotes?: Record +} + +type GitSpyTarget = { + git(args: string[], cwd: string): Promise<{ stdout: string; stderr: string }> +} + +/** One scripted repository, driven identically by both hosts. */ +function scriptGit(fixture: GitConfigFixture) { + const configValues = new Map() + const put = (key: string, value: string | undefined): void => { + if (value !== undefined) { + configValues.set(key, value) + } + } + put(`branch.${fixture.branch}.merge`, fixture.merge) + put(`branch.${fixture.branch}.remote`, fixture.branchRemote) + put(`branch.${fixture.branch}.pushRemote`, fixture.pushRemote) + put(`branch.${fixture.branch}.base`, fixture.base) + put('remote.pushDefault', fixture.pushDefault) + + const calls: string[][] = [] + return { + calls, + run: async (args: string[]): Promise<{ stdout: string; stderr: string }> => { + calls.push(args) + if (args[0] === 'symbolic-ref') { + return { stdout: `${fixture.branch}\n`, stderr: '' } + } + if (args[0] === 'config' && args[1] === '--get') { + const value = configValues.get(args[2] ?? '') + // Why throw: `git config --get` exits 1 for a missing key, and the resolver's + // fallback chain reads that rejection, not an empty string. + if (value === undefined) { + throw Object.assign(new Error('missing config key'), { code: 1 }) + } + return { stdout: `${value}\n`, stderr: '' } + } + if (args[0] === 'remote' && args[1] === '-v') { + const lines = Object.entries(fixture.remotes ?? {}).flatMap(([name, url]) => [ + `${name}\t${url} (fetch)`, + `${name}\t${url} (push)` + ]) + return { stdout: `${lines.join('\n')}\n`, stderr: '' } + } + if (args[0] === 'push') { + return { stdout: '', stderr: '' } + } + throw new Error(`Unexpected git command: ${args.join(' ')}`) + } + } +} + +function pushArgv(calls: string[][]): string[] { + const push = calls.find((args) => args[0] === 'push') + if (!push) { + throw new Error('no push command was issued') + } + return push +} + +async function pushOverRelay(fixture: GitConfigFixture): Promise { + const dispatcher = createMockDispatcher() + const handler = new GitHandler(dispatcher as unknown as RelayDispatcher, new RelayContext()) + const script = scriptGit(fixture) + vi.spyOn(handler as unknown as GitSpyTarget, 'git').mockImplementation((args) => script.run(args)) + await dispatcher.callRequest('git.push', { worktreePath: WORKTREE_PATH }) + return pushArgv(script.calls) +} + +async function pushLocally(fixture: GitConfigFixture): Promise { + const script = scriptGit(fixture) + gitExecFileAsyncMock.mockImplementation((args: string[]) => script.run(args)) + await gitPush(WORKTREE_PATH) + return pushArgv(script.calls) +} + +async function expectSamePushArgv(fixture: GitConfigFixture, expected: string[]): Promise { + const relayArgv = await pushOverRelay(fixture) + const localArgv = await pushLocally(fixture) + expect(relayArgv).toEqual(localArgv) + expect(localArgv).toEqual(expected) +} + +const FIRST_PUBLISH = ['push', '--set-upstream', 'origin', 'HEAD'] + +beforeEach(() => { + gitExecFileAsyncMock.mockReset() +}) + +describe('relay/desktop push-target parity', () => { + it('sends a review branch to the fork its pushDefault names', async () => { + await expectSamePushArgv( + { + branch: 'review/pr-1738', + merge: 'refs/heads/contributor/fix', + branchRemote: 'fork', + pushDefault: 'fork' + }, + ['push', '--set-upstream', 'fork', 'HEAD:contributor/fix'] + ) + }) + + it('refuses to inherit origin/main as a destination for a differently named branch', async () => { + // branch.merge belongs to branch.remote; a branch tracking origin/main must + // first-publish under its own name rather than push onto main. + await expectSamePushArgv( + { + branch: 'feature/fix', + merge: 'refs/heads/main', + branchRemote: 'origin' + }, + FIRST_PUBLISH + ) + }) + + it('refuses a pushDefault fork whose branch.remote names a different remote', async () => { + await expectSamePushArgv( + { + branch: 'review/pr-1738', + merge: 'refs/heads/contributor/fix', + branchRemote: 'origin', + pushDefault: 'fork' + }, + FIRST_PUBLISH + ) + }) + + it('refuses when branch.base names the same remote branch as branch.merge', async () => { + await expectSamePushArgv( + { + branch: 'feature/fix', + merge: 'refs/heads/release', + branchRemote: 'fork', + pushRemote: 'fork', + base: 'fork/release' + }, + FIRST_PUBLISH + ) + }) + + it('resolves a URL-valued pushRemote back to its remote name', async () => { + await expectSamePushArgv( + { + branch: 'review/pr-1738', + merge: 'refs/heads/contributor/fix', + branchRemote: 'git@example.invalid:contributor/repo.git', + pushRemote: 'git@example.invalid:contributor/repo.git', + remotes: { + origin: 'git@example.invalid:upstream/repo.git', + fork: 'git@example.invalid:contributor/repo.git' + } + }, + ['push', '--set-upstream', 'fork', 'HEAD:contributor/fix'] + ) + }) + + it('treats a local-repository remote as no configured target', async () => { + await expectSamePushArgv( + { + branch: 'feature/fix', + merge: 'refs/heads/feature/fix', + branchRemote: '.' + }, + FIRST_PUBLISH + ) + }) + + it('first-publishes a branch with no configured remote at all', async () => { + await expectSamePushArgv( + { branch: 'feature/fix', merge: 'refs/heads/feature/fix' }, + FIRST_PUBLISH + ) + }) +}) diff --git a/src/shared/git-push-target-resolution.ts b/src/shared/git-push-target-resolution.ts new file mode 100644 index 00000000000..63fe7828d9e --- /dev/null +++ b/src/shared/git-push-target-resolution.ts @@ -0,0 +1,140 @@ +import type { GitCommandRunner } from './git-effective-upstream' +import { gitRefTargetsBranchOnRemote } from './git-remote-branch-name' +import { findGitRemoteNameByFetchUrl } from './git-remote-url-index' + +export type ResolvedGitPushTarget = { + remote: string + refspec: string +} + +async function getConfigValue(runGit: GitCommandRunner, key: string): Promise { + try { + const { stdout } = await runGit(['config', '--get', key]) + const value = stdout.trim() + return value || null + } catch { + return null + } +} + +function isUrlValuedRemote(remote: string): boolean { + return /^[A-Za-z][A-Za-z0-9+.-]*:\/\//.test(remote) || /^[^@/:]+@[^:]+:.+/.test(remote) +} + +type ConfiguredPushRemote = { + remote: string + branchRemote: string | null +} + +// One `git remote -v` instead of `git remote` plus a serial `git remote get-url` +// per remote; both print the same insteadOf-expanded fetch URL. +async function findRemoteNameForUrl( + runGit: GitCommandRunner, + remoteUrl: string +): Promise { + try { + const { stdout } = await runGit(['remote', '-v']) + return findGitRemoteNameByFetchUrl(stdout, (candidateUrl) => candidateUrl === remoteUrl) + } catch { + return null + } +} + +async function normalizePushRemote(runGit: GitCommandRunner, remote: string): Promise { + if (!isUrlValuedRemote(remote)) { + return remote + } + return (await findRemoteNameForUrl(runGit, remote)) ?? remote +} + +async function getConfiguredPushRemote( + runGit: GitCommandRunner, + branch: string +): Promise { + const branchRemote = await getConfigValue(runGit, `branch.${branch}.remote`) + const remote = + (await getConfigValue(runGit, `branch.${branch}.pushRemote`)) ?? + (await getConfigValue(runGit, 'remote.pushDefault')) ?? + branchRemote + if (!remote) { + return null + } + const normalizedRemote = await normalizePushRemote(runGit, remote) + // The two usually name the same URL; resolving it twice reads the remote table twice. + if (!branchRemote) { + return { remote: normalizedRemote, branchRemote: null } + } + return { + remote: normalizedRemote, + branchRemote: + branchRemote === remote ? normalizedRemote : await normalizePushRemote(runGit, branchRemote) + } +} + +async function branchMergeTargetsConfiguredBase( + runGit: GitCommandRunner, + branch: string, + remote: string, + branchRef: string +): Promise { + return gitRefTargetsBranchOnRemote( + await getConfigValue(runGit, `branch.${branch}.base`), + remote, + branchRef + ) +} + +function canPushConfiguredMergeBranch( + pushRemote: ConfiguredPushRemote | null, + branch: string, + branchRef: string +): boolean { + if (!pushRemote) { + return false + } + if (branchRef === branch) { + return true + } + // Why: branch.merge belongs to branch.remote. A pushDefault fork must not + // inherit origin/main as its destination branch. + return pushRemote.remote !== 'origin' && pushRemote.branchRemote === pushRemote.remote +} + +/** + * Which remote and refspec a plain `git push` from this worktree should hit, or `null` + * to fall back to first-publish (`origin HEAD`). + * + * Why shared: this decides where commits land, and a wrong answer is not recoverable by + * retrying. The local runner and the SSH relay must never be able to answer differently + * for the same repository — they differ only in how `runGit` reaches the Git binary. + */ +export async function resolveConfiguredGitPushTarget( + runGit: GitCommandRunner +): Promise { + try { + const { stdout: branchStdout } = await runGit(['symbolic-ref', '--quiet', '--short', 'HEAD']) + const branch = branchStdout.trim() + if (!branch) { + return null + } + const [pushRemote, { stdout: mergeStdout }] = await Promise.all([ + getConfiguredPushRemote(runGit, branch), + runGit(['config', '--get', `branch.${branch}.merge`]) + ]) + const remote = pushRemote?.remote + const mergeRef = mergeStdout.trim() + const branchRef = mergeRef.replace(/^refs\/heads\//, '') + if (!remote || !branchRef || remote === '.' || branchRef === mergeRef) { + return null + } + if (await branchMergeTargetsConfiguredBase(runGit, branch, remote, branchRef)) { + return null + } + if (!canPushConfiguredMergeBranch(pushRemote, branch, branchRef)) { + return null + } + return { remote, refspec: `HEAD:${branchRef}` } + } catch { + return null + } +} From 232d04f5414edef2fc219f7726d15b9b15fd0d9b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:43:01 -0700 Subject: [PATCH 193/398] fix(dashboard): open remote sessions from every agent reveal path (#18403) Three reveal paths called bare setActiveWorktree + activateTabAndFocusPane, skipping setActiveView('terminal'), ensureWorktreeHasInitialTerminal and resumeSleepingAgentSessionsForWorktree. A parked SSH workspace has no resident tab until those run, so the reveal landed on a workspace with no terminal. Route all three through the incumbent activateAndRevealWorkspace dispatcher (which the sidebar and "Jump to workspace" already use, and which also handles folder workspaces). The Activity row-click additionally early-returned when the thread's tab was absent from tabsByWorktree/unifiedTabsByWorktree, which made a cold-parked remote thread a silent no-op; residency is now probed after activation, so a revived tab is focused and a genuinely retained thread still activates its workspace instead of doing nothing. Also stop asserting `exited` from an absence of local state: SshPtyProvider reports no authoritative buffer snapshot and the relay has no snapshot RPC, so a null preview snapshot for a remote pty is loss of contact. The preview and the no-pty dialog branch now say the remote preview is unavailable rather than claiming the pane closed. Adding the relay snapshot RPC stays out of scope -- it needs capability negotiation. Fixes #16731 --- .../activity/activity-thread-actions.test.ts | 106 ++++++++++++------ .../activity/activity-thread-actions.ts | 40 +++---- .../AgentTerminalDialog.test.tsx | 26 +++++ .../dashboard-popout/AgentTerminalDialog.tsx | 6 +- .../AgentTerminalPreview.test.tsx | 8 ++ .../dashboard-popout/AgentTerminalPreview.tsx | 7 +- ...rminal-preview-unavailable-message.test.ts | 18 +++ .../terminal-preview-unavailable-message.ts | 27 +++++ .../dashboard/AgentDashboardDrawer.test.tsx | 67 ++++++++--- .../dashboard/AgentDashboardDrawer.tsx | 5 +- .../dashboard/reveal-dashboard-agent.ts | 23 ++++ .../useDashboardPopoutBridge.test.tsx | 61 +++++++++- .../dashboard/useDashboardPopoutBridge.ts | 5 +- src/renderer/src/i18n/locales/en.json | 1 + 14 files changed, 308 insertions(+), 92 deletions(-) create mode 100644 src/renderer/src/components/dashboard-popout/terminal-preview-unavailable-message.test.ts create mode 100644 src/renderer/src/components/dashboard-popout/terminal-preview-unavailable-message.ts create mode 100644 src/renderer/src/components/dashboard/reveal-dashboard-agent.ts diff --git a/src/renderer/src/components/activity/activity-thread-actions.test.ts b/src/renderer/src/components/activity/activity-thread-actions.test.ts index ce2de01fe02..91375ec974b 100644 --- a/src/renderer/src/components/activity/activity-thread-actions.test.ts +++ b/src/renderer/src/components/activity/activity-thread-actions.test.ts @@ -49,12 +49,23 @@ describe('activity thread host routing', () => { const setActiveWorktree = vi.fn() const acknowledgeAgents = vi.fn() const setSelectedPaneKey = vi.fn() + let state: Record + + function makeActions(): ReturnType { + return createActivityThreadActions({ + getMarkAllReadThreads: () => [thread], + acknowledgeAgents, + unacknowledgeAgents: vi.fn(), + setSelectedPaneKey + }) + } beforeEach(() => { vi.clearAllMocks() mocks.activateStructuredAgentSessionTab.mockReturnValue(false) + mocks.activateAndRevealWorkspace.mockReturnValue({ primaryTabId: null }) getKnownWorktreeById.mockReturnValue(thread.worktree) - mocks.getState.mockReturnValue({ + state = { getKnownWorktreeById, worktreesByRepo: { [thread.worktree.repoId]: [thread.worktree] }, detectedWorktreesByRepo: {}, @@ -77,21 +88,19 @@ describe('activity thread host routing', () => { setActiveRepo: vi.fn(), setActiveWorktree, setActiveTabType: vi.fn() - }) + } + mocks.getState.mockImplementation(() => state) }) - it('selects the matching host when the same workspace id is active elsewhere', () => { - const actions = createActivityThreadActions({ - getMarkAllReadThreads: () => [thread], - acknowledgeAgents, - unacknowledgeAgents: vi.fn(), - setSelectedPaneKey + it('routes the row click through the full activation sequence for the matching host', () => { + makeActions().selectThread(thread) + + // Bare setActiveWorktree skips setActiveView('terminal'), initial-terminal seeding and + // sleeping-session resume — the workspace dispatcher is the only path that runs them. + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith(thread.worktree.id, { + executionHostId: REMOTE_HOST }) - - actions.selectThread(thread) - - expect(getKnownWorktreeById).toHaveBeenCalledWith(thread.worktree.id, REMOTE_HOST) - expect(setActiveWorktree).toHaveBeenCalledWith(thread.worktree.id, REMOTE_HOST) + expect(setActiveWorktree).not.toHaveBeenCalled() expect(mocks.activateTabAndFocusPane).toHaveBeenCalledWith( thread.tab.id, '11111111-1111-4111-8111-111111111111', @@ -99,23 +108,56 @@ describe('activity thread host routing', () => { ) }) - it('activates a structured agent session instead of looking for a terminal pane', () => { - mocks.activateStructuredAgentSessionTab.mockReturnValue(true) - mocks.getState.mockReturnValue({ - ...mocks.getState(), - tabsByWorktree: { [thread.worktree.id]: [] }, - unifiedTabsByWorktree: { - [thread.worktree.id]: [{ id: thread.tab.id, contentType: 'agent-session' }] - } - }) - const actions = createActivityThreadActions({ - getMarkAllReadThreads: () => [thread], - acknowledgeAgents, - unacknowledgeAgents: vi.fn(), - setSelectedPaneKey + it('opens a cold-parked remote thread whose tab activation revives', () => { + // The reported SSH symptom: the tab is not resident because the session was never + // revived, so a residency probe before activation made the click a silent no-op. + state.tabsByWorktree = {} + mocks.activateAndRevealWorkspace.mockImplementation(() => { + state.tabsByWorktree = { [thread.worktree.id]: [thread.tab] } + return { primaryTabId: thread.tab.id } }) - actions.selectThread(thread) + makeActions().selectThread(thread) + + expect(setSelectedPaneKey).toHaveBeenCalledWith(thread.paneKey) + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith(thread.worktree.id, { + executionHostId: REMOTE_HOST + }) + expect(mocks.activateTabAndFocusPane).toHaveBeenCalledWith( + thread.tab.id, + '11111111-1111-4111-8111-111111111111', + { flashFocusedPane: true, scrollToBottomIfOutputSinceLastView: true } + ) + }) + + it('still activates the workspace when a retained thread has no tab to focus', () => { + state.tabsByWorktree = {} + + makeActions().selectThread(thread) + + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith(thread.worktree.id, { + executionHostId: REMOTE_HOST + }) + expect(mocks.activateTabAndFocusPane).not.toHaveBeenCalled() + }) + + it('focuses nothing when the workspace itself is gone', () => { + mocks.activateAndRevealWorkspace.mockReturnValue(false) + + makeActions().selectThread(thread) + + expect(mocks.activateStructuredAgentSessionTab).not.toHaveBeenCalled() + expect(mocks.activateTabAndFocusPane).not.toHaveBeenCalled() + }) + + it('activates a structured agent session instead of looking for a terminal pane', () => { + mocks.activateStructuredAgentSessionTab.mockReturnValue(true) + state.tabsByWorktree = { [thread.worktree.id]: [] } + state.unifiedTabsByWorktree = { + [thread.worktree.id]: [{ id: thread.tab.id, contentType: 'agent-session' }] + } + + makeActions().selectThread(thread) expect(mocks.activateStructuredAgentSessionTab).toHaveBeenCalledWith({ worktreeId: thread.worktree.id, @@ -126,14 +168,8 @@ describe('activity thread host routing', () => { it('jumps to and probes the matching host-qualified workspace', () => { expect(hasActivityThreadWorkspace(thread)).toBe(true) - const actions = createActivityThreadActions({ - getMarkAllReadThreads: () => [thread], - acknowledgeAgents, - unacknowledgeAgents: vi.fn(), - setSelectedPaneKey - }) - actions.jumpToWorkspace(thread) + makeActions().jumpToWorkspace(thread) expect(acknowledgeAgents).toHaveBeenCalledWith([thread.paneKey]) expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith(thread.worktree.id, { diff --git a/src/renderer/src/components/activity/activity-thread-actions.ts b/src/renderer/src/components/activity/activity-thread-actions.ts index b0971da7ea7..f9f77587a1e 100644 --- a/src/renderer/src/components/activity/activity-thread-actions.ts +++ b/src/renderer/src/components/activity/activity-thread-actions.ts @@ -1,5 +1,6 @@ import { activateTabAndFocusPane } from '@/lib/activate-tab-and-focus-pane' import { activateStructuredAgentSessionTab } from '@/lib/structured-agent-session-tab-activation' +import { activateAndRevealWorkspace } from '@/lib/worktree-activation' import { jumpToWorktreeFromSidebar } from '@/lib/worktree-jump-navigation' import { useAppStore } from '@/store' import { @@ -74,38 +75,31 @@ export function createActivityThreadActions({ } const activateThreadTarget = (thread: AgentPaneThread): void => { - const state = useAppStore.getState() const executionHostId = getActivityThreadExecutionHostId( thread, - getSettingsFocusedExecutionHostId(state.settings) + getSettingsFocusedExecutionHostId(useAppStore.getState().settings) ) - const worktree = state.getKnownWorktreeById(thread.worktree.id, executionHostId) - if (!worktree) { + // Why the full sequence (not bare setActiveWorktree): a cold-parked thread — the normal + // state of an SSH session that was never revived — has no resident tab until + // resumeSleepingAgentSessionsForWorktree/ensureWorktreeHasInitialTerminal run inside here. + // Probing tab residency first is what made a remote row click a silent no-op (#16731). + if (activateAndRevealWorkspace(thread.worktree.id, { executionHostId }) === false) { return } - const liveTabs = state.tabsByWorktree[worktree.id] ?? [] - const hasLiveTerminal = liveTabs.some((tab) => tab.id === thread.tab.id) - const hasLiveAgentSession = (state.unifiedTabsByWorktree?.[worktree.id] ?? []).some( - (tab) => tab.id === thread.tab.id && tab.contentType === 'agent-session' - ) - // Why: retained threads can outlive their target; reorienting the workspace for a - // dead terminal or structured session would just confuse the user. - if (!hasLiveTerminal && !hasLiveAgentSession) { - return - } - if (state.activeRepoId !== worktree.repoId) { - state.setActiveRepo(worktree.repoId) - } if ( - state.activeWorktreeId !== worktree.id || - state.activeWorkspaceExecutionHostId !== executionHostId + activateStructuredAgentSessionTab({ worktreeId: thread.worktree.id, tabId: thread.tab.id }) ) { - state.setActiveWorktree(worktree.id, executionHostId) - } - if (activateStructuredAgentSessionTab({ worktreeId: worktree.id, tabId: thread.tab.id })) { return } - state.setActiveTabType('terminal') + // Read post-activation: the tab this thread points at may have only just been revived. + const activated = useAppStore.getState() + const liveTabs = activated.tabsByWorktree[thread.worktree.id] ?? [] + if (!liveTabs.some((tab) => tab.id === thread.tab.id)) { + // Retained threads outlive their tab; the workspace is still activated, but there is + // no pane to focus and focusing a sibling would be worse than focusing nothing. + return + } + activated.setActiveTabType('terminal') const parsed = parsePaneKey(thread.paneKey) activateTabAndFocusPane( thread.tab.id, diff --git a/src/renderer/src/components/dashboard-popout/AgentTerminalDialog.test.tsx b/src/renderer/src/components/dashboard-popout/AgentTerminalDialog.test.tsx index 7022d7e5731..770e1974625 100644 --- a/src/renderer/src/components/dashboard-popout/AgentTerminalDialog.test.tsx +++ b/src/renderer/src/components/dashboard-popout/AgentTerminalDialog.test.tsx @@ -91,6 +91,32 @@ describe('AgentTerminalDialog', () => { expect(screen.getByTestId('preview')).toHaveAttribute('data-terminal-input', 'null') }) + it('does not claim a remote pane closed when the card carries no live pty', () => { + render( + {}} + onReveal={() => {}} + /> + ) + + // Loss of contact with an SSH host is `unverifiable`, never `exited`. + expect(screen.getByText(/remote session/)).toBeInTheDocument() + expect(screen.queryByText(/pane has closed/)).not.toBeInTheDocument() + }) + + it('still reports a closed pane for a local card with no live pty', () => { + render( + {}} + onReveal={() => {}} + /> + ) + + expect(screen.getByText(/pane has closed/)).toBeInTheDocument() + }) + it('labels acknowledged completions idle without review or pin controls', () => { render( ) : (
    - {translate( - 'dashboardPopout.terminal.closed', - "No live terminal — this agent's pane has closed." - )} + {terminalPreviewUnavailableMessage({ hostKind: card.hostKind })}
    )}
    diff --git a/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.test.tsx b/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.test.tsx index 4c91a22a3b8..c4856a23834 100644 --- a/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.test.tsx +++ b/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.test.tsx @@ -612,6 +612,14 @@ describe('AgentTerminalPreview', () => { expect(unsubscribe).toHaveBeenCalledWith('pty-1') }) + it('does not claim a remote pane closed when no snapshot can exist for it', async () => { + connect.mockResolvedValueOnce({ snapshot: null, replay: [] }) + const view = render() + + await waitFor(() => expect(view.getByText(/remote session/)).toBeInTheDocument()) + expect(view.queryByText(/pane has closed/)).not.toBeInTheDocument() + }) + it('connects a replacement pty after the previous pty was gone', async () => { connect.mockResolvedValueOnce({ snapshot: null, replay: [] }).mockResolvedValueOnce({ snapshot: { data: 'replacement', cols: 80, rows: 24, seq: 1 }, diff --git a/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.tsx b/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.tsx index f2a702782e9..ec05a7105a5 100644 --- a/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.tsx +++ b/src/renderer/src/components/dashboard-popout/AgentTerminalPreview.tsx @@ -17,7 +17,7 @@ import { installPreviewTerminalCompatibility } from './preview-terminal-compatib import { createPreviewClipboardPaster } from './preview-terminal-paste' import { installPreviewImeBridge, type PreviewImeBridge } from './preview-terminal-ime-bridge' import type { DashboardCardTerminalInput } from '../../../../shared/dashboard-snapshot' -import { translate } from '@/i18n/i18n' +import { terminalPreviewUnavailableMessage } from './terminal-preview-unavailable-message' import { getBuiltinTheme, resolveEffectiveTerminalAppearance } from '@/lib/terminal-theme' import { cn } from '@/lib/utils' import { useAppStore } from '@/store' @@ -430,10 +430,7 @@ export function AgentTerminalPreview({ > {ptyGone ? (
    - {translate( - 'dashboardPopout.terminal.closed', - "No live terminal — this agent's pane has closed." - )} + {terminalPreviewUnavailableMessage({ ptyId })}
    ) : null}
    { + it('claims the pane closed only for a pty the client could have observed', () => { + expect(terminalPreviewUnavailableMessage({ ptyId: 'pty-1' })).toMatch(/pane has closed/) + expect(terminalPreviewUnavailableMessage({ hostKind: 'local' })).toMatch(/pane has closed/) + }) + + it('reports an unobservable remote preview instead of asserting the pane exited', () => { + // SshPtyProvider provides no authoritative buffer snapshot and the relay has no snapshot + // RPC, so a null snapshot is loss of contact. See docs/reference/ssh-execution-boundary.md. + const fromPtyId = terminalPreviewUnavailableMessage({ ptyId: 'ssh:devbox@@pty-3' }) + expect(fromPtyId).toMatch(/remote session/) + expect(fromPtyId).not.toMatch(/pane has closed/) + expect(terminalPreviewUnavailableMessage({ hostKind: 'ssh' })).toBe(fromPtyId) + }) +}) diff --git a/src/renderer/src/components/dashboard-popout/terminal-preview-unavailable-message.ts b/src/renderer/src/components/dashboard-popout/terminal-preview-unavailable-message.ts new file mode 100644 index 00000000000..07080084bfc --- /dev/null +++ b/src/renderer/src/components/dashboard-popout/terminal-preview-unavailable-message.ts @@ -0,0 +1,27 @@ +import { translate } from '@/i18n/i18n' +import type { DashboardCardHostKind } from '../../../../shared/dashboard-snapshot' +import { parseAppSshPtyId } from '../../../../shared/ssh-pty-id' + +/** + * A missing buffer snapshot only proves the pane exited when the client could have + * observed it. `SshPtyProvider` reports no authoritative buffer snapshot and the relay + * exposes no snapshot RPC, so for a remote pty the absence is loss of contact — + * `unverifiable`, never `exited`. See docs/reference/ssh-execution-boundary.md. + */ +export function terminalPreviewUnavailableMessage(source: { + ptyId?: string | null + hostKind?: DashboardCardHostKind +}): string { + const isRemote = + source.hostKind === 'ssh' || + (typeof source.ptyId === 'string' && parseAppSshPtyId(source.ptyId) !== null) + return isRemote + ? translate( + 'dashboardPopout.terminal.remotePreviewUnavailable', + 'No preview for this remote session — open the workspace to view the terminal.' + ) + : translate( + 'dashboardPopout.terminal.closed', + "No live terminal — this agent's pane has closed." + ) +} diff --git a/src/renderer/src/components/dashboard/AgentDashboardDrawer.test.tsx b/src/renderer/src/components/dashboard/AgentDashboardDrawer.test.tsx index 75b76e396b4..97c31c4747a 100644 --- a/src/renderer/src/components/dashboard/AgentDashboardDrawer.test.tsx +++ b/src/renderer/src/components/dashboard/AgentDashboardDrawer.test.tsx @@ -8,13 +8,18 @@ const mocks = vi.hoisted(() => ({ useLiveDashboardSnapshot: vi.fn(() => ({ generatedAt: 1, cards: [] })), blockingOverlay: false, boardProps: null as Record | null, - activateTabAndFocusPane: vi.fn() + activateTabAndFocusPane: vi.fn(), + activateAndRevealWorkspace: vi.fn(() => ({ primaryTabId: null }) as unknown) })) vi.mock('@/lib/activate-tab-and-focus-pane', () => ({ activateTabAndFocusPane: mocks.activateTabAndFocusPane })) +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorkspace: mocks.activateAndRevealWorkspace +})) + vi.mock('./useLiveDashboardSnapshot', () => ({ useLiveDashboardSnapshot: mocks.useLiveDashboardSnapshot })) @@ -49,6 +54,9 @@ beforeEach(() => { false ) mocks.useLiveDashboardSnapshot.mockClear() + mocks.activateTabAndFocusPane.mockClear() + mocks.activateAndRevealWorkspace.mockClear() + mocks.activateAndRevealWorkspace.mockReturnValue({ primaryTabId: null }) mocks.blockingOverlay = false mocks.boardProps = null ;(window as unknown as { api: unknown }).api = { @@ -95,34 +103,63 @@ describe('AgentDashboardDrawer', () => { expect(mocks.boardProps?.initialView).toBeUndefined() }) - it('reveals a colliding worktree on the card execution host', () => { - const setActiveWorktree = vi.spyOn(useAppStore.getState(), 'setActiveWorktree') + type RevealAgent = (args: { + repoId: string + worktreeId: string + executionHostId?: string + tabId: string + leafId: string | null + }) => void + + function revealFromBoard(executionHostId: string): void { render() act(() => useAppStore.setState({ agentDashboardDrawerOpen: true })) const onRevealAgent = mocks.boardProps?.onRevealAgent expect(onRevealAgent).toBeTypeOf('function') - act(() => { - ;( - onRevealAgent as (args: { - repoId: string - worktreeId: string - executionHostId?: string - tabId: string - leafId: string | null - }) => void - )({ + ;(onRevealAgent as RevealAgent)({ repoId: 'repo-1', worktreeId: 'shared-worktree', - executionHostId: 'runtime:env-1', + executionHostId, tabId: 'tab-1', leafId: 'leaf-1' }) }) + } - expect(setActiveWorktree).toHaveBeenCalledWith('shared-worktree', 'runtime:env-1') + it('reveals a colliding worktree on the card execution host', () => { + const setActiveWorktree = vi.spyOn(useAppStore.getState(), 'setActiveWorktree') + + revealFromBoard('runtime:env-1') + + // Bare setActiveWorktree skips the terminal view switch, initial-terminal seeding and + // sleeping-session resume the shared dispatcher runs. + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('shared-worktree', { + executionHostId: 'runtime:env-1' + }) + expect(setActiveWorktree).not.toHaveBeenCalled() expect(mocks.activateTabAndFocusPane).toHaveBeenCalledWith('tab-1', 'leaf-1', { flashFocusedPane: true }) }) + + it('activates a parked SSH workspace before reaching for its pane', () => { + revealFromBoard('ssh:devbox') + + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('shared-worktree', { + executionHostId: 'ssh:devbox' + }) + // Ordering is the fix: a parked remote tab only exists after activation revives it. + expect(mocks.activateAndRevealWorkspace.mock.invocationCallOrder[0]).toBeLessThan( + mocks.activateTabAndFocusPane.mock.invocationCallOrder[0] as number + ) + }) + + it('skips pane focus when the revealed workspace is gone', () => { + mocks.activateAndRevealWorkspace.mockReturnValue(false) + + revealFromBoard('ssh:devbox') + + expect(mocks.activateTabAndFocusPane).not.toHaveBeenCalled() + }) }) diff --git a/src/renderer/src/components/dashboard/AgentDashboardDrawer.tsx b/src/renderer/src/components/dashboard/AgentDashboardDrawer.tsx index a5df13c2395..347255324fd 100644 --- a/src/renderer/src/components/dashboard/AgentDashboardDrawer.tsx +++ b/src/renderer/src/components/dashboard/AgentDashboardDrawer.tsx @@ -1,7 +1,7 @@ import { useCallback, useEffect, useRef, useState } from 'react' import { useAppStore } from '@/store' import { Sheet, SheetContent, SheetTitle } from '@/components/ui/sheet' -import { activateTabAndFocusPane } from '@/lib/activate-tab-and-focus-pane' +import { revealDashboardAgent } from './reveal-dashboard-agent' import { AgentKanbanBoard } from '../dashboard-popout/AgentKanbanBoard' import type { AgentRevealArgs } from '../dashboard-popout/AgentTerminalDialog' import { @@ -47,8 +47,7 @@ function AgentDashboardDrawerBody({ }, []) const handleRevealAgent = useCallback( (args: AgentRevealArgs) => { - useAppStore.getState().setActiveWorktree(args.worktreeId, args.executionHostId) - activateTabAndFocusPane(args.tabId, args.leafId, { flashFocusedPane: true }) + revealDashboardAgent(args) onClose() }, [onClose] diff --git a/src/renderer/src/components/dashboard/reveal-dashboard-agent.ts b/src/renderer/src/components/dashboard/reveal-dashboard-agent.ts new file mode 100644 index 00000000000..9da364c72ef --- /dev/null +++ b/src/renderer/src/components/dashboard/reveal-dashboard-agent.ts @@ -0,0 +1,23 @@ +import { activateTabAndFocusPane } from '@/lib/activate-tab-and-focus-pane' +import { activateAndRevealWorkspace } from '@/lib/worktree-activation' +import type { DashboardRevealAgentArgs } from '../../../../shared/dashboard-snapshot' + +/** + * Click-to-focus from either Agent Dashboard surface (pop-out relay or in-window drawer). + * + * Why the workspace dispatcher rather than a bare `setActiveWorktree`: only the shared + * sequence switches the view back to terminal, resumes sleeping agent sessions, and seeds a + * terminal surface. A parked SSH workspace has no resident tab until those run, so the bare + * call revealed a workspace with nothing in it (#16731). + */ +export function revealDashboardAgent(args: DashboardRevealAgentArgs): boolean { + const activated = activateAndRevealWorkspace( + args.worktreeId, + args.executionHostId ? { executionHostId: args.executionHostId } : undefined + ) + if (activated === false) { + return false + } + activateTabAndFocusPane(args.tabId, args.leafId, { flashFocusedPane: true }) + return true +} diff --git a/src/renderer/src/components/dashboard/useDashboardPopoutBridge.test.tsx b/src/renderer/src/components/dashboard/useDashboardPopoutBridge.test.tsx index aa427ed55c5..e80bf56d778 100644 --- a/src/renderer/src/components/dashboard/useDashboardPopoutBridge.test.tsx +++ b/src/renderer/src/components/dashboard/useDashboardPopoutBridge.test.tsx @@ -21,7 +21,9 @@ const mocks = vi.hoisted(() => ({ offRevealAgent: vi.fn(), offAckAgent: vi.fn(), offPopoutOpenChanged: vi.fn(), - offSnapshotRequested: vi.fn() + offSnapshotRequested: vi.fn(), + activateTabAndFocusPane: vi.fn(), + activateAndRevealWorkspace: vi.fn() })) vi.mock('@/store', () => ({ @@ -35,7 +37,11 @@ vi.mock('@/store', () => ({ })) vi.mock('@/lib/activate-tab-and-focus-pane', () => ({ - activateTabAndFocusPane: vi.fn() + activateTabAndFocusPane: mocks.activateTabAndFocusPane +})) + +vi.mock('@/lib/worktree-activation', () => ({ + activateAndRevealWorkspace: mocks.activateAndRevealWorkspace })) vi.mock('./build-dashboard-snapshot', () => ({ @@ -159,7 +165,8 @@ describe('useDashboardPopoutBridge', () => { expect(mocks.buildDashboardSnapshot).toHaveBeenCalledTimes(1) }) - it('reveals the agent on its exact execution host', async () => { + it('reveals the agent on its exact execution host through the full activation', async () => { + mocks.activateAndRevealWorkspace.mockReturnValue({ primaryTabId: null }) await act(async () => root.render()) await act(async () => @@ -172,7 +179,53 @@ describe('useDashboardPopoutBridge', () => { }) ) - expect(mocks.setActiveWorktree).toHaveBeenCalledWith('shared-worktree', 'runtime:env-1') + // Bare setActiveWorktree skips the terminal view switch, initial-terminal seeding and + // sleeping-session resume, so a parked pane is never revived (#16731). + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('shared-worktree', { + executionHostId: 'runtime:env-1' + }) + expect(mocks.setActiveWorktree).not.toHaveBeenCalled() + expect(mocks.activateTabAndFocusPane).toHaveBeenCalledWith('tab-1', 'leaf-1', { + flashFocusedPane: true + }) + }) + + it('activates a parked SSH workspace before reaching for its pane', async () => { + mocks.activateAndRevealWorkspace.mockReturnValue({ primaryTabId: 'tab-1' }) + await act(async () => root.render()) + + await act(async () => + mocks.onRevealAgent.mock.calls[0][0]({ + repoId: 'repo-1', + worktreeId: 'remote-worktree', + executionHostId: 'ssh:devbox', + tabId: 'tab-1', + leafId: 'leaf-1' + }) + ) + + expect(mocks.activateAndRevealWorkspace).toHaveBeenCalledWith('remote-worktree', { + executionHostId: 'ssh:devbox' + }) + expect(mocks.activateAndRevealWorkspace.mock.invocationCallOrder[0]).toBeLessThan( + mocks.activateTabAndFocusPane.mock.invocationCallOrder[0] as number + ) + }) + + it('skips pane focus when the revealed workspace is gone', async () => { + mocks.activateAndRevealWorkspace.mockReturnValue(false) + await act(async () => root.render()) + + await act(async () => + mocks.onRevealAgent.mock.calls[0][0]({ + repoId: 'repo-1', + worktreeId: 'deleted-worktree', + tabId: 'tab-1', + leafId: 'leaf-1' + }) + ) + + expect(mocks.activateTabAndFocusPane).not.toHaveBeenCalled() }) it('ignores unrelated store writes while retaining every snapshot input', () => { diff --git a/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts b/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts index e6163a70773..f80de74a748 100644 --- a/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts +++ b/src/renderer/src/components/dashboard/useDashboardPopoutBridge.ts @@ -1,6 +1,6 @@ import { useEffect } from 'react' import { useAppStore, type AppState } from '@/store' -import { activateTabAndFocusPane } from '@/lib/activate-tab-and-focus-pane' +import { revealDashboardAgent } from './reveal-dashboard-agent' import { runSleepWorktree } from '../sidebar/sleep-worktree-flow' import type { RepoIcon } from '../../../../shared/repo-icon' import { buildDashboardSnapshot, type DashboardSnapshotState } from './build-dashboard-snapshot' @@ -133,8 +133,7 @@ export function useDashboardPopoutBridge(enabled: boolean): void { return } return window.api.dashboard.onRevealAgent((args) => { - useAppStore.getState().setActiveWorktree(args.worktreeId, args.executionHostId) - activateTabAndFocusPane(args.tabId, args.leafId, { flashFocusedPane: true }) + revealDashboardAgent(args) }) }, [enabled]) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 2ba5712f4fa..28149e7ca70 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -17230,6 +17230,7 @@ }, "terminal": { "closed": "No live terminal — this agent's pane has closed.", + "remotePreviewUnavailable": "No preview for this remote session — open the workspace to view the terminal.", "focusWorktree": "Open worktree", "close": "Close" }, From 9bed758e36951fd2ab9a1d9a7178729591a7048a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:43:05 -0700 Subject: [PATCH 194/398] fix(cli): reject runtime selectors on `host list` and `environment list` (#18405) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `orca host list --environment m4air` was not ignoring the flag — it was applying it to half the answer. `shouldIgnoreRemoteSelection` never pinned the `host` family, so the SSH-target lookup was routed to m4air while paired servers were still read from this machine's own pairing store, and the handler stamped the envelope `_meta.runtimeId: "local"` regardless. The result was one listing describing two hosts: the openclaw row silently disappeared, which reads as "m4air has no SSH targets". `environment list --environment X` had the pin but no guard, so the flag vanished with no signal at all. Reject rather than route. `host list` answers "what can this machine target and with what flag"; its paired-server half comes from a client-local store and cannot be routed at all, so any routed answer is necessarily half-substituted — rule 1 of docs/reference/ssh-execution-boundary.md. `environment list` is entirely client-local, so there is no other host to ask. This matches the `account` and `artifacts` precedent, the only two pinned families that already paired the pin with a rejection guard. - pin the `host` family so an ambient ORCA_ENVIRONMENT cannot produce the same two-machine listing with no flag to reject; `runtimeId: "local"` is now true - extract the duplicated `rejectRemoteSelectionFlags` from account.ts and artifacts.ts into src/cli/remote-selection-flag-rejection.ts - `environment show` / `environment rm` / `environment add` are untouched: there `--environment` and `--pairing-code` name the row to act on, not a route --- src/cli/handlers/account.ts | 19 +- src/cli/handlers/artifacts.ts | 25 ++- src/cli/handlers/environment.ts | 32 ++- .../index-local-command-routing-flags.test.ts | 184 ++++++++++++++++++ src/cli/index.ts | 4 + src/cli/remote-selection-flag-rejection.ts | 29 +++ src/cli/specs/environment.ts | 8 +- 7 files changed, 272 insertions(+), 29 deletions(-) create mode 100644 src/cli/index-local-command-routing-flags.test.ts create mode 100644 src/cli/remote-selection-flag-rejection.ts diff --git a/src/cli/handlers/account.ts b/src/cli/handlers/account.ts index a6ee5d73246..5a8bd4d3c6c 100644 --- a/src/cli/handlers/account.ts +++ b/src/cli/handlers/account.ts @@ -7,6 +7,7 @@ import type { CommandHandler, HandlerContext } from '../dispatch' import { printResult } from '../format' import { RuntimeClientError } from '../runtime-client' import { stripElectronRunAsNode } from '../runtime/launch' +import { rejectRemoteSelectionFlags } from '../remote-selection-flag-rejection' import { deleteActiveClaudeKeychainCredentialsStrict, readActiveClaudeKeychainCredentialsStrict, @@ -276,15 +277,11 @@ async function addCodexAccount({ client, json }: HandlerContext): Promise * mistake this feature exists to avoid. A `--help` note does not reach someone who * already typed the flag. */ -function rejectRemoteSelectionFlags(ctx: HandlerContext, command: string): void { - for (const flag of ['environment', 'pairing-code']) { - if (ctx.flags.has(flag)) { - throw new RuntimeClientError( - 'invalid_argument', - `\`--${flag}\` does not retarget \`${command}\`. Run it on the host whose accounts you want to manage.` - ) - } - } +function rejectAccountRemoteSelectionFlags(ctx: HandlerContext, command: string): void { + rejectRemoteSelectionFlags( + ctx.flags, + `\`${command}\`. Run it on the host whose accounts you want to manage.` + ) } async function assertAccountImportSupported({ client }: HandlerContext): Promise { @@ -316,14 +313,14 @@ export const ACCOUNT_HANDLERS: Record = { `Unsupported --agent "${agent}". Use "claude" or "codex".` ) } - rejectRemoteSelectionFlags(ctx, 'orca account add') + rejectAccountRemoteSelectionFlags(ctx, 'orca account add') // Why: fail on runtime version skew before burning a full OAuth round trip. await assertAccountImportSupported(ctx) await ctx.client.call('accounts.list', { refreshUsage: false }) await (agent === 'claude' ? addClaudeAccount(ctx) : addCodexAccount(ctx)) }, 'account list': async (ctx) => { - rejectRemoteSelectionFlags(ctx, 'orca account list') + rejectAccountRemoteSelectionFlags(ctx, 'orca account list') const { client, json } = ctx // Why: this command renders no usage numbers, so skip the forced provider // refresh — it is one serial network round-trip per managed account. diff --git a/src/cli/handlers/artifacts.ts b/src/cli/handlers/artifacts.ts index 2306dcc406a..8546d7997f3 100644 --- a/src/cli/handlers/artifacts.ts +++ b/src/cli/handlers/artifacts.ts @@ -18,6 +18,7 @@ import { ARTIFACT_SHARING_DISABLED_NEXT_STEPS } from '../../shared/artifact-sharing-gate' import type { CommandHandler, HandlerContext } from '../dispatch' +import { rejectRemoteSelectionFlags } from '../remote-selection-flag-rejection' import { RuntimeClientError } from '../runtime-client' import { formatArtifactListPage, formatArtifactShared } from '../artifact-format' import { printResult } from '../format' @@ -44,15 +45,11 @@ function cloudOptions(ctx: HandlerContext): ArtifactCloudOptions { } } -function rejectRemoteSelectionFlags(ctx: HandlerContext): void { - for (const flag of ['environment', 'pairing-code']) { - if (ctx.flags.has(flag)) { - throw new RuntimeClientError( - 'invalid_argument', - `\`--${flag}\` does not retarget artifact commands; artifacts use the signed-in desktop account.` - ) - } - } +function rejectArtifactRemoteSelectionFlags(ctx: HandlerContext): void { + rejectRemoteSelectionFlags( + ctx.flags, + 'artifact commands; artifacts use the signed-in desktop account.' + ) } function artifactContentType(path: string): ArtifactWriteRequest['contentType'] | null { @@ -165,7 +162,7 @@ function requireOperation(operation: ArtifactCloudOperation): T { export const ARTIFACT_HANDLERS: Record = { 'artifacts list': async (ctx) => { - rejectRemoteSelectionFlags(ctx) + rejectArtifactRemoteSelectionFlags(ctx) const cursor = stringFlag(ctx, 'cursor') const response = await ctx.client.call>( 'artifacts.list', @@ -178,7 +175,7 @@ export const ARTIFACT_HANDLERS: Record = { printResult({ ...response, result: value }, ctx.json, formatArtifactListPage) }, 'artifacts share': async (ctx) => { - rejectRemoteSelectionFlags(ctx) + rejectArtifactRemoteSelectionFlags(ctx) const response = await ctx.client.call>( 'artifacts.share', await readArtifactRequest(ctx) @@ -187,7 +184,7 @@ export const ARTIFACT_HANDLERS: Record = { printResult({ ...response, result: value }, ctx.json, formatArtifactShared) }, 'artifacts update': async (ctx) => { - rejectRemoteSelectionFlags(ctx) + rejectArtifactRemoteSelectionFlags(ctx) const response = await ctx.client.call>( 'artifacts.update', await readArtifactRequest(ctx) @@ -196,7 +193,7 @@ export const ARTIFACT_HANDLERS: Record = { printResult({ ...response, result: value }, ctx.json, formatArtifactShared) }, 'artifacts unshare': async (ctx) => { - rejectRemoteSelectionFlags(ctx) + rejectArtifactRemoteSelectionFlags(ctx) const remoteInput = parseRemoteArtifactInput(process.env[REMOTE_ARTIFACT_INPUT_ENV]) const sourceKey = remoteInput?.sourceKey ?? resolve(ctx.cwd, requireStringFlag(ctx, 'file')) const response = await ctx.client.call>('artifacts.unshare', { @@ -207,7 +204,7 @@ export const ARTIFACT_HANDLERS: Record = { printResult({ ...response, result: { deleted: true } }, ctx.json, () => 'Artifact deleted.') }, 'artifacts delete': async (ctx) => { - rejectRemoteSelectionFlags(ctx) + rejectArtifactRemoteSelectionFlags(ctx) const response = await ctx.client.call>('artifacts.delete', { id: requireStringFlag(ctx, 'id'), ...cloudOptions(ctx) diff --git a/src/cli/handlers/environment.ts b/src/cli/handlers/environment.ts index 2a433021769..37b437af2c3 100644 --- a/src/cli/handlers/environment.ts +++ b/src/cli/handlers/environment.ts @@ -3,6 +3,7 @@ import { formatEnvironment, formatEnvironmentList, formatHostList, printResult } import { listSshTargets } from '../host-selector-alternatives' import { getDefaultUserDataPath, RuntimeClientError } from '../runtime-client' import type { RuntimeRpcSuccess } from '../runtime-client' +import { rejectRemoteSelectionFlags } from '../remote-selection-flag-rejection' import { redactRuntimeEnvironment } from '../../shared/runtime-environments' import { addEnvironmentFromPairingCode, @@ -33,7 +34,12 @@ export const ENVIRONMENT_HANDLERS: Record = { // Why: an agent told "run it on " had nowhere to look. `orca environment list` showed // paired servers only, and nothing in the CLI listed SSH targets at all, so the wrong-axis // guess was the only move available. This is the one place that answers both. - 'host list': async ({ client, json }) => { + 'host list': async ({ client, flags, json }) => { + rejectLocalPairingStoreRetargeting( + flags, + '`orca host list`. It answers from this machine\u2019s own pairing store, so a routed answer would name servers paired with a different machine.', + 'Run `orca host list` on that machine to see the SSH targets registered there.' + ) const environments = listEnvironments(getDefaultUserDataPath()).map((environment) => ({ kind: 'environment' as const, name: environment.name, @@ -53,7 +59,12 @@ export const ENVIRONMENT_HANDLERS: Record = { ] printResult(localSuccess({ hosts }), json, formatHostList) }, - 'environment list': async ({ json }) => { + 'environment list': async ({ flags, json }) => { + rejectLocalPairingStoreRetargeting( + flags, + '`orca environment list`. Paired servers are stored on this machine, so there is no other host to ask.', + 'Run `orca environment list` on that machine to see the servers paired with it.' + ) const environments = listEnvironments(getDefaultUserDataPath()).map(redactRuntimeEnvironment) printResult(localSuccess({ environments }), json, formatEnvironmentList) }, @@ -78,6 +89,23 @@ export const ENVIRONMENT_HANDLERS: Record = { } } +/** + * These two listings are pinned local by `shouldIgnoreRemoteSelection`, so a runtime selector is + * dropped for routing. It used to still reach the SSH half of `host list` through the routed + * client, producing a listing whose SSH rows came from the named server and whose paired-server + * rows came from this machine — one answer describing two hosts, stamped `runtimeId: local`. + * Failing is the only answer that is true of a single machine. + */ +function rejectLocalPairingStoreRetargeting( + flags: Map, + suffix: string, + crossHostNextStep: string +): void { + rejectRemoteSelectionFlags(flags, suffix, { + nextSteps: [crossHostNextStep, 'Drop the flag to answer for this machine.'] + }) +} + function getRequiredStringFlag(flags: Map, name: string): string { const value = flags.get(name) if (typeof value !== 'string' || value.length === 0) { diff --git a/src/cli/index-local-command-routing-flags.test.ts b/src/cli/index-local-command-routing-flags.test.ts new file mode 100644 index 00000000000..b8db44915d2 --- /dev/null +++ b/src/cli/index-local-command-routing-flags.test.ts @@ -0,0 +1,184 @@ +import { describe, expect, it, vi } from 'vitest' + +const { + callMock, + runtimeClientConstructorMock, + serveOrcaAppMock, + getDefaultUserDataPathMock, + addEnvironmentFromPairingCodeMock, + listEnvironmentsMock, + removeEnvironmentMock, + resolveEnvironmentMock, + spawnMock +} = vi.hoisted(() => ({ + callMock: vi.fn(), + runtimeClientConstructorMock: vi.fn(), + serveOrcaAppMock: vi.fn(), + getDefaultUserDataPathMock: vi.fn(() => '/tmp/orca-user-data'), + addEnvironmentFromPairingCodeMock: vi.fn(), + listEnvironmentsMock: vi.fn(), + removeEnvironmentMock: vi.fn(), + resolveEnvironmentMock: vi.fn(), + spawnMock: vi.fn() +})) + +vi.mock('./runtime-client', async () => { + const { createRuntimeClientModuleMock } = await import('./index-test-harness.js') + return createRuntimeClientModuleMock({ + callMock, + runtimeClientConstructorMock, + serveOrcaAppMock, + getDefaultUserDataPathMock + }) +}) + +vi.mock('./runtime/environments', () => ({ + addEnvironmentFromPairingCode: addEnvironmentFromPairingCodeMock, + listEnvironments: listEnvironmentsMock, + removeEnvironment: removeEnvironmentMock, + resolveEnvironment: resolveEnvironmentMock +})) + +vi.mock('child_process', async () => { + const { createChildProcessModuleMock } = await import('./index-test-harness.js') + return createChildProcessModuleMock(spawnMock) +}) + +import { main } from './index' +import { okFixture, queueFixtures } from './test-fixtures' +import { pairRuntimeEnvironment, useWorktreeAwarenessEnvironment } from './index-test-harness' + +const SSH_TARGET = { id: 'ssh-1777360569033-yvz2mp', label: 'openclaw' } + +/** Every SSH-target lookup answers with the one target only this machine's runtime knows about. */ +function queueSshTargetLookups(count: number): void { + queueFixtures( + callMock, + ...Array.from({ length: count }, () => okFixture('req_ssh_targets', { targets: [SSH_TARGET] })) + ) +} + +describe('runtime-selector flags on locally pinned CLI commands', () => { + useWorktreeAwarenessEnvironment({ + callMock, + serveOrcaAppMock, + getDefaultUserDataPathMock, + addEnvironmentFromPairingCodeMock, + listEnvironmentsMock, + spawnMock + }) + + it('answers `host list` from this machine and stamps the runtime that actually answered', async () => { + pairRuntimeEnvironment(listEnvironmentsMock, 'env-m4air', 'm4air') + queueSshTargetLookups(1) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['host', 'list', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed._meta.runtimeId).toBe('local') + expect(printed.result.hosts.map((host: { id: string }) => host.id)).toEqual([ + 'local', + SSH_TARGET.id, + 'env-m4air' + ]) + // The tell: `runtimeId: local` is only honest if no routed client was ever built. + expect(runtimeClientConstructorMock).toHaveBeenCalledWith(null, null) + }) + + it('rejects `host list --environment` instead of answering with a half-routed listing', async () => { + // Why: pre-fix this routed the SSH lookup to m4air while reading paired servers from this + // machine, dropped the openclaw row, and still stamped `_meta.runtimeId: "local"` — one + // listing describing two hosts, which reads as "m4air has no SSH targets". + pairRuntimeEnvironment(listEnvironmentsMock, 'env-m4air', 'm4air') + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['host', 'list', '--environment', 'm4air', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed.ok).toBe(false) + expect(printed.error.code).toBe('invalid_argument') + expect(printed.error.message).toContain('`--environment` does not retarget `orca host list`') + expect(process.exitCode).toBe(1) + expect(callMock).not.toHaveBeenCalled() + expect(runtimeClientConstructorMock).not.toHaveBeenCalledWith(null, 'm4air') + process.exitCode = 0 + }) + + it('rejects `environment list --environment` rather than repeating the local answer', async () => { + pairRuntimeEnvironment(listEnvironmentsMock, 'env-m4air', 'm4air') + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['environment', 'list', '--environment', 'm4air', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed.ok).toBe(false) + expect(printed.error.code).toBe('invalid_argument') + expect(printed.error.message).toContain( + '`--environment` does not retarget `orca environment list`' + ) + process.exitCode = 0 + }) + + it('rejects `--pairing-code` on both listings for the same reason', async () => { + pairRuntimeEnvironment(listEnvironmentsMock, 'env-m4air', 'm4air') + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['host', 'list', '--pairing-code', 'orca://pair?code=x', '--json'], '/tmp/repo') + await main( + ['environment', 'list', '--pairing-code', 'orca://pair?code=x', '--json'], + '/tmp/repo' + ) + + for (const call of logSpy.mock.calls) { + const printed = JSON.parse(String(call[0])) + expect(printed.ok).toBe(false) + expect(printed.error.message).toContain('`--pairing-code` does not retarget') + } + expect(callMock).not.toHaveBeenCalled() + process.exitCode = 0 + }) + + it('keeps `host list` local when ORCA_ENVIRONMENT is set ambiently', async () => { + // Why: the ambient variable produced the same two-machine listing as the explicit flag, with + // no flag to reject. Pinning the family is what makes `runtimeId: local` true in both cases. + process.env.ORCA_ENVIRONMENT = 'm4air' + pairRuntimeEnvironment(listEnvironmentsMock, 'env-m4air', 'm4air') + queueSshTargetLookups(1) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['host', 'list', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed.ok).toBe(true) + expect(printed.result.hosts.some((host: { id: string }) => host.id === SSH_TARGET.id)).toBe( + true + ) + expect(runtimeClientConstructorMock).toHaveBeenCalledWith(null, null) + expect(runtimeClientConstructorMock).not.toHaveBeenCalledWith(undefined, undefined) + }) + + it('still treats --environment as the selector argument on `environment show` and `rm`', async () => { + // Why: the guard must not fire where the flag names the row to act on rather than a route. + const environment = { + id: 'env-m4air', + name: 'm4air', + createdAt: 1, + updatedAt: 1, + lastUsedAt: null, + runtimeId: null, + endpoints: [], + preferredEndpointId: null + } + resolveEnvironmentMock.mockReturnValue(environment) + removeEnvironmentMock.mockReturnValue(environment) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['environment', 'show', '--environment', 'm4air', '--json'], '/tmp/repo') + await main(['environment', 'rm', '--environment', 'm4air', '--json'], '/tmp/repo') + + for (const call of logSpy.mock.calls) { + expect(JSON.parse(String(call[0])).ok).toBe(true) + } + }) +}) diff --git a/src/cli/index.ts b/src/cli/index.ts index c29bfff7060..9389113b195 100644 --- a/src/cli/index.ts +++ b/src/cli/index.ts @@ -31,6 +31,10 @@ function shouldIgnoreRemoteSelection(commandPath: string[]): boolean { commandPath[0] === 'account' || commandPath[0] === 'artifacts' || commandPath[0] === 'environment' || + // Why: `host list` answers "what can this machine target, and with what flag". Half of that + // answer (paired servers) is read from this machine's own pairing store and cannot be routed, + // so routing the other half produced one listing describing two machines at once. + commandPath[0] === 'host' || commandPath[0] === 'serve' || commandPath[0] === 'agent' || commandPath[0] === 'vm' || diff --git a/src/cli/remote-selection-flag-rejection.ts b/src/cli/remote-selection-flag-rejection.ts new file mode 100644 index 00000000000..e2609fa604a --- /dev/null +++ b/src/cli/remote-selection-flag-rejection.ts @@ -0,0 +1,29 @@ +import { RuntimeClientError } from './runtime/types' + +/** + * The flags that pick which runtime answers a command. `shouldIgnoreRemoteSelection` + * in `src/cli/index.ts` pins some command families to the local runtime, which drops + * these silently — so every pinned family pairs the pin with this rejection instead. + */ +export const REMOTE_SELECTION_FLAGS = ['environment', 'pairing-code'] as const + +/** + * Fails a pinned command that was given a runtime selector, rather than answering + * for a machine the caller did not name. `suffix` completes "`--` does not + * retarget …" and should say what the command answers for and where to run it. + */ +export function rejectRemoteSelectionFlags( + flags: ReadonlyMap, + suffix: string, + data?: Record +): void { + for (const flag of REMOTE_SELECTION_FLAGS) { + if (flags.has(flag)) { + throw new RuntimeClientError( + 'invalid_argument', + `\`--${flag}\` does not retarget ${suffix}`, + data + ) + } + } +} diff --git a/src/cli/specs/environment.ts b/src/cli/specs/environment.ts index d6ceb795028..7bf90615270 100644 --- a/src/cli/specs/environment.ts +++ b/src/cli/specs/environment.ts @@ -10,7 +10,8 @@ export const ENVIRONMENT_COMMAND_SPECS: CommandSpec[] = [ notes: [ 'Answers "what can I target and what do I pass" in one place: this machine, the SSH targets registered on it, and the Orca servers paired with it.', 'The three kinds are reached differently. A paired Orca server is a connection, selected with --environment . An SSH target is a machine the connected Orca host reaches, selected with --host ssh:. Passing one where the other belongs is the most common way to get an empty or missing-host answer.', - "SSH targets are read from the Orca host you are currently connected to, so this lists that host's targets and not another server's." + "SSH targets are read from this machine's own Orca runtime, so this lists that machine's targets and not another server's. Run `orca host list` on the other machine to see the targets registered there.", + '--environment and --pairing-code are rejected rather than ignored: paired servers come from this machine\u2019s pairing store, so a routed answer would describe two machines at once.' ], examples: ['orca host list', 'orca host list --json'] }, @@ -25,7 +26,10 @@ export const ENVIRONMENT_COMMAND_SPECS: CommandSpec[] = [ path: ['environment', 'list'], summary: 'List saved Orca runtime environments', usage: 'orca environment list [--json]', - allowedFlags: [...GLOBAL_FLAGS] + allowedFlags: [...GLOBAL_FLAGS], + notes: [ + 'Answers from this machine\u2019s pairing store. --environment and --pairing-code are rejected rather than ignored, because there is no other host that could answer.' + ] }, { path: ['environment', 'show'], From 95eed528012dfd44b7244b3ef17beaefb8390568 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:43:09 -0700 Subject: [PATCH 195/398] fix(cli): report which hosts a worktree listing covered, and stop the cap starving remote ones (#18417) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `orca worktree list` returned zero of 24 SSH worktrees at the default limit (#18104). Rows are resolved repo by repo, so every SSH repo's rows land contiguously at the end of the fleet order — the 24 remote rows sat at indices 496-520 of 521 and a plain `slice(0, 200)` never reached them. The omission was not fully silent: text output printed `truncated: showing 200 of 521` and JSON carried `totalCount` / `truncated`. What was missing is that the omission was *categorically every remote host* — no host column, no `hostScope`, nothing to distinguish "200 of 521" from "one host is entirely absent". Per docs/reference/ssh-execution-boundary.md, a listing that does not name its scope reads as absolute. Adopt the mechanism `terminal list` already has rather than inventing a second one: - `RuntimeTerminalListHostScope` becomes an alias of a shared `RuntimeListingHostScope`, now also carried (optional, so old hosts are unaffected) on `worktree.list` and `worktree.ps` results. - `src/shared/host-balanced-listing-page.ts` round-robins the row cap across hosts and returns the survivors in the caller's original relative order, so the page stays a subsequence of the unbounded listing and nothing downstream re-sorts. An uncapped listing is returned unchanged. - `worktree list` / `worktree ps` text output gains a `host=` column and the same trailing `scope:` line `terminal list` prints. Third defect, same mechanism: `hostScope.omittedHostIds` is built from the runtime's own bookkeeping, so it names `runtime:` ids for servers that are no longer paired — 6 of 9 in the recorded QA run hard-error when queried. Since `hostScope` is *the* documented way to complete a partial listing, that makes the mechanism unreliable for its intended use. Annotate rather than filter. Dropping an id would shrink what the listing admits it did not cover, and the boundary doc requires a listing to name its gaps — the gap is real whether or not this machine can name the host that owns it. `src/cli/omitted-host-scope-selectors.ts` resolves each omitted id against this machine's pairing store and the runtime's SSH-target registry and attaches the exact flag that reaches it, or `null` marked "not selectable from this machine". This is a client-side annotation: nothing new goes over the wire, it answers "can I select it" and never "is it up", and the SSH round trip is only paid when an `ssh:` host was actually omitted. No `--host` filter was added; the host column plus scope line covers the reported need without a new selector axis. --- src/cli/handlers/terminal.ts | 20 +- src/cli/handlers/worktree.ts | 24 +- ...index-omitted-host-scope-selectors.test.ts | 246 ++++++++++++++++++ .../index-terminal-list-host-scope.test.ts | 5 +- src/cli/omitted-host-scope-selectors.ts | 126 +++++++++ src/cli/specs/core.ts | 7 +- src/cli/specs/worktree-listing-scope-notes.ts | 6 + src/cli/terminal-format.ts | 20 +- src/cli/workspace-format.ts | 27 +- .../runtime/orca-runtime-get-worktree-ps.ts | 17 +- .../orca-runtime-stop-requested-pty-ids.ts | 3 +- ...creation-and-orchestration-part-04.spec.ts | 2 + ...me-managed-worktree-metadata-sweep.test.ts | 3 +- .../runtime-managed-worktree-queries.test.ts | 3 +- .../runtime-managed-worktree-queries.ts | 13 +- .../runtime/worktree-list-host-scope.test.ts | 170 ++++++++++++ .../runtime/worktree-listing-host-scope.ts | 64 +++++ .../runtime/worktree-ps-host-scope.test.ts | 131 ++++++++++ src/shared/host-balanced-listing-page.ts | 49 ++++ src/shared/runtime-listing-host-scope.ts | 12 + src/shared/runtime-terminal-contracts.ts | 7 +- src/shared/runtime-worktree-contracts.ts | 5 + 22 files changed, 895 insertions(+), 65 deletions(-) create mode 100644 src/cli/index-omitted-host-scope-selectors.test.ts create mode 100644 src/cli/omitted-host-scope-selectors.ts create mode 100644 src/cli/specs/worktree-listing-scope-notes.ts create mode 100644 src/main/runtime/worktree-list-host-scope.test.ts create mode 100644 src/main/runtime/worktree-listing-host-scope.ts create mode 100644 src/main/runtime/worktree-ps-host-scope.test.ts create mode 100644 src/shared/host-balanced-listing-page.ts create mode 100644 src/shared/runtime-listing-host-scope.ts diff --git a/src/cli/handlers/terminal.ts b/src/cli/handlers/terminal.ts index 72e59b86655..3ff6142275a 100644 --- a/src/cli/handlers/terminal.ts +++ b/src/cli/handlers/terminal.ts @@ -32,6 +32,10 @@ import { getOptionalStringFlag, getRequiredStringFlag } from '../flags' +import { + annotateOmittedHostScope, + type WithAnnotatedHostScope +} from '../omitted-host-scope-selectors' import { RuntimeClientError } from '../runtime-client' import { getBrowserWorktreeSelector, @@ -90,12 +94,16 @@ const terminalFocusHandler: CommandHandler = async ({ flags, client, cwd, json } export const TERMINAL_HANDLERS: Record = { 'terminal list': async ({ flags, client, cwd, json }) => { - const result = await client.call('terminal.list', { - worktree: await getOptionalWorktreeSelector(flags, 'worktree', cwd, client), - limit: getOptionalPositiveIntegerFlag(flags, 'limit'), - // Why: agent JSON calls dominate; topology stays available through an explicit opt-in. - includeVisualLayouts: !json || flags.has('include-visual-layouts') - }) + const result = await client.call>( + 'terminal.list', + { + worktree: await getOptionalWorktreeSelector(flags, 'worktree', cwd, client), + limit: getOptionalPositiveIntegerFlag(flags, 'limit'), + // Why: agent JSON calls dominate; topology stays available through an explicit opt-in. + includeVisualLayouts: !json || flags.has('include-visual-layouts') + } + ) + await annotateOmittedHostScope(client, result.result) printResult(result, json, formatTerminalList) }, 'terminal show': async ({ flags, client, cwd, json }) => { diff --git a/src/cli/handlers/worktree.ts b/src/cli/handlers/worktree.ts index 484bcf9ea6d..262599234f0 100644 --- a/src/cli/handlers/worktree.ts +++ b/src/cli/handlers/worktree.ts @@ -7,6 +7,10 @@ import type { } from '../../shared/runtime-types' import type { CommandHandler } from '../dispatch' import { formatWorktreeList, formatWorktreePs, formatWorktreeShow, printResult } from '../format' +import { + annotateOmittedHostScope, + type WithAnnotatedHostScope +} from '../omitted-host-scope-selectors' import { RuntimeClientError } from '../runtime-client' import { getOptionalNullableNumberFlag, @@ -171,16 +175,22 @@ async function getCreateRepoSelector( export const WORKTREE_HANDLERS: Record = { 'worktree ps': async ({ flags, client, json }) => { - const result = await client.call('worktree.ps', { - limit: getOptionalPositiveIntegerFlag(flags, 'limit') - }) + const result = await client.call>( + 'worktree.ps', + { limit: getOptionalPositiveIntegerFlag(flags, 'limit') } + ) + await annotateOmittedHostScope(client, result.result) printResult(result, json, formatWorktreePs) }, 'worktree list': async ({ flags, client, json }) => { - const result = await client.call('worktree.list', { - repo: getOptionalStringFlag(flags, 'repo'), - limit: getOptionalPositiveIntegerFlag(flags, 'limit') - }) + const result = await client.call>( + 'worktree.list', + { + repo: getOptionalStringFlag(flags, 'repo'), + limit: getOptionalPositiveIntegerFlag(flags, 'limit') + } + ) + await annotateOmittedHostScope(client, result.result) printResult(result, json, formatWorktreeList) }, 'worktree show': async ({ flags, client, cwd, json }) => { diff --git a/src/cli/index-omitted-host-scope-selectors.test.ts b/src/cli/index-omitted-host-scope-selectors.test.ts new file mode 100644 index 00000000000..1fbd77289f5 --- /dev/null +++ b/src/cli/index-omitted-host-scope-selectors.test.ts @@ -0,0 +1,246 @@ +import { describe, expect, it, vi } from 'vitest' + +const { + callMock, + runtimeClientConstructorMock, + serveOrcaAppMock, + getDefaultUserDataPathMock, + addEnvironmentFromPairingCodeMock, + listEnvironmentsMock, + spawnMock +} = vi.hoisted(() => ({ + callMock: vi.fn(), + runtimeClientConstructorMock: vi.fn(), + serveOrcaAppMock: vi.fn(), + getDefaultUserDataPathMock: vi.fn(() => '/tmp/orca-user-data'), + addEnvironmentFromPairingCodeMock: vi.fn(), + listEnvironmentsMock: vi.fn(), + spawnMock: vi.fn() +})) + +vi.mock('./runtime-client', async () => { + const { createRuntimeClientModuleMock } = await import('./index-test-harness.js') + return createRuntimeClientModuleMock({ + callMock, + runtimeClientConstructorMock, + serveOrcaAppMock, + getDefaultUserDataPathMock + }) +}) + +vi.mock('./runtime/environments', () => ({ + addEnvironmentFromPairingCode: addEnvironmentFromPairingCodeMock, + listEnvironments: listEnvironmentsMock, + removeEnvironment: vi.fn(), + resolveEnvironment: vi.fn() +})) + +vi.mock('child_process', async () => { + const { createChildProcessModuleMock } = await import('./index-test-harness.js') + return createChildProcessModuleMock(spawnMock) +}) + +import { main } from './index' +import { okFixture, queueFixtures } from './test-fixtures' +import { pairRuntimeEnvironment, useWorktreeAwarenessEnvironment } from './index-test-harness' + +const TERMINAL_ROW = { + handle: 'term_1', + ptyId: 'pty-1', + worktreeId: 'repo::/wt', + worktreePath: '/wt', + branch: 'main', + tabId: 'tab-1', + leafId: 'leaf-1', + title: 'worker', + connected: true, + writable: true, + lastOutputAt: null, + preview: '', + executionHostId: 'local' +} + +describe('omittedHostIds selector annotation', () => { + useWorktreeAwarenessEnvironment({ + callMock, + serveOrcaAppMock, + getDefaultUserDataPathMock, + addEnvironmentFromPairingCodeMock, + listEnvironmentsMock, + spawnMock + }) + + it('marks a stale runtime host that no caller can select', async () => { + // Why: `omittedHostIds` is built from the runtime's own bookkeeping, so it names `runtime:` + // ids for servers that are no longer paired. An agent looping over the list to complete a + // partial listing hard-errors on those — 6 of 9 in the recorded QA run. + pairRuntimeEnvironment(listEnvironmentsMock, 'env-paired', 'm4air') + queueFixtures( + callMock, + okFixture('req_terminal_list', { + terminals: [TERMINAL_ROW], + totalCount: 1, + truncated: false, + hostScope: { + hostIds: ['local'], + omittedHostIds: ['runtime:env-paired', 'runtime:env-retired'] + } + }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['terminal', 'list', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed.result.hostScope.omittedHostIds).toEqual([ + 'runtime:env-paired', + 'runtime:env-retired' + ]) + expect(printed.result.hostScope.omittedHostSelectors).toEqual([ + { hostId: 'runtime:env-paired', selector: '--environment m4air' }, + { hostId: 'runtime:env-retired', selector: null } + ]) + }) + + it('says which omitted hosts are not selectable in the human listing', async () => { + pairRuntimeEnvironment(listEnvironmentsMock, 'env-paired', 'm4air') + queueFixtures( + callMock, + okFixture('req_terminal_list', { + terminals: [TERMINAL_ROW], + totalCount: 1, + truncated: false, + hostScope: { + hostIds: ['local'], + omittedHostIds: ['runtime:env-paired', 'runtime:env-retired'] + } + }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['terminal', 'list'], '/tmp/repo') + + const printed = String(logSpy.mock.calls[0]?.[0]) + expect(printed).toContain('runtime:env-paired (--environment m4air)') + expect(printed).toContain('runtime:env-retired (not selectable from this machine)') + }) + + it('resolves an omitted SSH host against the targets the runtime actually knows', async () => { + listEnvironmentsMock.mockReturnValue([]) + queueFixtures( + callMock, + okFixture('req_terminal_list', { + terminals: [TERMINAL_ROW], + totalCount: 1, + truncated: false, + hostScope: { hostIds: ['local'], omittedHostIds: ['ssh:box-1', 'ssh:box-gone'] } + }), + okFixture('req_ssh_targets', { targets: [{ id: 'box-1', label: 'openclaw' }] }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['terminal', 'list', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed.result.hostScope.omittedHostSelectors).toEqual([ + { hostId: 'ssh:box-1', selector: '--host ssh:box-1' }, + { hostId: 'ssh:box-gone', selector: null } + ]) + }) + + it('never keeps a host id out of omittedHostIds', async () => { + // Why: filtering the unreachable ones would shrink what the listing admits it did not cover. + // The gap is real whether or not this machine can name the host that owns it. + listEnvironmentsMock.mockReturnValue([]) + queueFixtures( + callMock, + okFixture('req_terminal_list', { + terminals: [], + totalCount: 0, + truncated: false, + hostScope: { hostIds: [], omittedHostIds: ['runtime:env-retired'] } + }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['terminal', 'list', '--json'], '/tmp/repo') + + const printed = JSON.parse(String(logSpy.mock.calls[0]?.[0])) + expect(printed.result.hostScope.omittedHostIds).toEqual(['runtime:env-retired']) + }) + + it('costs no extra round trip when nothing was omitted', async () => { + queueFixtures( + callMock, + okFixture('req_terminal_list', { + terminals: [TERMINAL_ROW], + totalCount: 1, + truncated: false, + hostScope: { hostIds: ['local'], omittedHostIds: [] } + }) + ) + vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['terminal', 'list', '--json'], '/tmp/repo') + + expect(callMock).toHaveBeenCalledTimes(1) + }) +}) + +describe('worktree listings report their host coverage', () => { + useWorktreeAwarenessEnvironment({ + callMock, + serveOrcaAppMock, + getDefaultUserDataPathMock, + addEnvironmentFromPairingCodeMock, + listEnvironmentsMock, + spawnMock + }) + + it('prints a host column and the scope line for `worktree list`', async () => { + listEnvironmentsMock.mockReturnValue([]) + queueFixtures( + callMock, + okFixture('req_worktree_list', { + worktrees: [ + { + id: 'repo-ssh::/remote/wt', + branch: 'main', + path: '/remote/wt', + hostId: 'ssh:box-1', + displayName: 'remote', + parentWorktreeId: null, + childWorktreeIds: [], + linkedIssue: null, + comment: '' + } + ], + totalCount: 521, + truncated: true, + hostScope: { hostIds: ['ssh:box-1'], omittedHostIds: ['runtime:env-retired'] } + }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['worktree', 'list'], '/tmp/repo') + + const printed = String(logSpy.mock.calls[0]?.[0]) + expect(printed).toContain('host=ssh:box-1') + expect(printed).toContain('scope: ssh:box-1') + expect(printed).toContain('runtime:env-retired (not selectable from this machine)') + expect(printed).toContain('truncated: showing 1 of 521') + }) + + it('does not claim a scope for `worktree ps` when the host reported none', async () => { + queueFixtures( + callMock, + okFixture('req_worktree_ps', { worktrees: [], totalCount: 0, truncated: false }) + ) + const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}) + + await main(['worktree', 'ps'], '/tmp/repo') + + const printed = String(logSpy.mock.calls[0]?.[0]) + expect(printed).toContain('scope: unverifiable') + }) +}) diff --git a/src/cli/index-terminal-list-host-scope.test.ts b/src/cli/index-terminal-list-host-scope.test.ts index b38cd71451f..04b486df671 100644 --- a/src/cli/index-terminal-list-host-scope.test.ts +++ b/src/cli/index-terminal-list-host-scope.test.ts @@ -88,7 +88,10 @@ describe('orca terminal list host scope', () => { expect(printed.result.terminals[0].executionHostId).toBe('ssh:box-1') expect(printed.result.hostScope).toEqual({ hostIds: ['ssh:box-1'], - omittedHostIds: ['local'] + omittedHostIds: ['local'], + // The CLI annotates each omitted host with the flag that reaches it; see + // index-omitted-host-scope-selectors.test.ts. + omittedHostSelectors: [{ hostId: 'local', selector: '--host local' }] }) }) diff --git a/src/cli/omitted-host-scope-selectors.ts b/src/cli/omitted-host-scope-selectors.ts new file mode 100644 index 00000000000..2666b116375 --- /dev/null +++ b/src/cli/omitted-host-scope-selectors.ts @@ -0,0 +1,126 @@ +import { + parseExecutionHostId, + type ExecutionHostId, + type ParsedExecutionHost +} from '../shared/execution-host' +import type { RuntimeListingHostScope } from '../shared/runtime-listing-host-scope' +import { + findEnvironmentByName, + findSshTargetByName, + listSshTargets, + type SshTargetSummary +} from './host-selector-alternatives' +import type { RuntimeClient } from './runtime-client' + +export type OmittedHostScopeSelector = { + hostId: ExecutionHostId + /** The flag that routes a follow-up query to this host, or null when it names nothing here. */ + selector: string | null +} + +/** A host scope annotated on this machine. The runtime never sends `omittedHostSelectors`. */ +export type ListingHostScopeWithSelectors = RuntimeListingHostScope & { + omittedHostSelectors?: OmittedHostScopeSelector[] +} + +export type WithAnnotatedHostScope = Omit & { + hostScope?: ListingHostScopeWithSelectors +} + +/** + * Resolves how to reach each host a listing did not cover. + * + * `hostScope` is the documented way to complete a partial listing, but `omittedHostIds` is built + * from the runtime's own bookkeeping — repos, folder workspaces, and workspace sessions — so it + * names `runtime:` ids for servers that are no longer paired. An agent looping over the list to + * finish the job hard-errors on those. + * + * The ids are kept rather than filtered: dropping one would shrink what the listing admits it did + * not cover, and `docs/reference/ssh-execution-boundary.md` requires a listing to name its gaps. + * A `null` selector marks the ones this machine cannot name, which is the part a caller needs. + * Only the local pairing store and SSH-target registry are consulted, so this answers "can I + * select it", never "is it up" — no host is claimed live or exited on this path. + */ +export async function resolveOmittedHostScopeSelectors( + client: RuntimeClient, + omittedHostIds: readonly ExecutionHostId[] +): Promise { + const parsed = omittedHostIds.map((hostId) => ({ + hostId, + host: parseExecutionHostId(hostId) + })) + const environments = parsed.some((entry) => entry.host?.kind === 'runtime') + ? await listPairedEnvironments() + : [] + // Why: SSH targets need a round trip, so only pay for it when an ssh host was actually omitted. + const sshTargets = parsed.some((entry) => entry.host?.kind === 'ssh') + ? await listSshTargets(client) + : [] + return parsed.map(({ hostId, host }) => ({ + hostId, + selector: resolveSelector(host, environments, sshTargets) + })) +} + +async function listPairedEnvironments(): Promise<{ id: string; name: string }[]> { + const [{ listEnvironments }, { getDefaultUserDataPath }] = await Promise.all([ + import('./runtime/environments.js'), + import('./runtime-client.js') + ]) + return listEnvironments(getDefaultUserDataPath()).map((environment) => ({ + id: environment.id, + name: environment.name + })) +} + +function resolveSelector( + host: ParsedExecutionHost | null, + environments: readonly { id: string; name: string }[], + sshTargets: readonly SshTargetSummary[] +): string | null { + if (host?.kind === 'local') { + return '--host local' + } + if (host?.kind === 'ssh') { + return findSshTargetByName(sshTargets, host.targetId) ? `--host ssh:${host.targetId}` : null + } + if (host?.kind === 'runtime') { + const environment = findEnvironmentByName(environments, host.environmentId) + return environment ? `--environment ${environment.name}` : null + } + return null +} + +/** Renders a scope line; an absent scope means the host never reported one, not full coverage. */ +export function formatListingHostScope(scope: ListingHostScopeWithSelectors | undefined): string { + if (!scope) { + return 'scope: unverifiable — this host does not report which hosts it lists' + } + const covered = scope.hostIds.length > 0 ? scope.hostIds.join(', ') : 'none' + if (scope.omittedHostIds.length === 0) { + return `scope: ${covered}` + } + const selectorByHostId = new Map( + (scope.omittedHostSelectors ?? []).map((entry) => [entry.hostId, entry.selector]) + ) + const omitted = scope.omittedHostIds.map((hostId) => { + if (!selectorByHostId.has(hostId)) { + return hostId + } + const selector = selectorByHostId.get(hostId) + return selector ? `${hostId} (${selector})` : `${hostId} (not selectable from this machine)` + }) + return `scope: ${covered} — not covered: ${omitted.join(', ')}` +} + +/** Attaches the resolved selectors in place; a listing with no omitted hosts pays nothing. */ +export async function annotateOmittedHostScope( + client: RuntimeClient, + result: { hostScope?: ListingHostScopeWithSelectors } +): Promise { + const scope = result.hostScope + if (!scope || scope.omittedHostIds.length === 0) { + return + } + scope.omittedHostSelectors = await resolveOmittedHostScopeSelectors(client, scope.omittedHostIds) +} diff --git a/src/cli/specs/core.ts b/src/cli/specs/core.ts index 112d9528d2e..f2236ef86e9 100644 --- a/src/cli/specs/core.ts +++ b/src/cli/specs/core.ts @@ -1,5 +1,6 @@ import type { CommandSpec } from '../args' import { GLOBAL_FLAGS } from '../args' +import { WORKTREE_LISTING_SCOPE_NOTES } from './worktree-listing-scope-notes' import { SERVE_COMMAND_SPECS } from './serve' import { TERMINAL_CLOSE_COMMAND_SPEC } from './terminal-close' @@ -65,7 +66,8 @@ export const CORE_COMMAND_SPECS: CommandSpec[] = [ path: ['worktree', 'list'], summary: 'List Orca-managed worktrees', usage: 'orca worktree list [--repo ] [--limit ] [--json]', - allowedFlags: [...GLOBAL_FLAGS, 'repo', 'limit'] + allowedFlags: [...GLOBAL_FLAGS, 'repo', 'limit'], + notes: [...WORKTREE_LISTING_SCOPE_NOTES] }, { path: ['worktree', 'show'], @@ -180,7 +182,8 @@ export const CORE_COMMAND_SPECS: CommandSpec[] = [ path: ['worktree', 'ps'], summary: 'Show a compact orchestration summary across worktrees', usage: 'orca worktree ps [--limit ] [--json]', - allowedFlags: [...GLOBAL_FLAGS, 'limit'] + allowedFlags: [...GLOBAL_FLAGS, 'limit'], + notes: [...WORKTREE_LISTING_SCOPE_NOTES] }, { path: ['terminal', 'list'], diff --git a/src/cli/specs/worktree-listing-scope-notes.ts b/src/cli/specs/worktree-listing-scope-notes.ts new file mode 100644 index 00000000000..91449cc6584 --- /dev/null +++ b/src/cli/specs/worktree-listing-scope-notes.ts @@ -0,0 +1,6 @@ +/** Shared by `worktree list` and `worktree ps`, which report host coverage the same way. */ +export const WORKTREE_LISTING_SCOPE_NOTES: readonly string[] = [ + 'Each row carries the execution host that owns it (`host=`), and the trailing `scope:` line names every host the page covers plus the ones it does not.', + 'A host named under `not covered` may still have workspaces; an empty answer for it is not evidence that it has none. Each is annotated with the flag that reaches it, or marked not selectable from this machine.', + 'The row cap is shared across hosts, so a host whose rows sort last is not starved out of the page.' +] diff --git a/src/cli/terminal-format.ts b/src/cli/terminal-format.ts index edf4cbaa22c..e61a2e48b76 100644 --- a/src/cli/terminal-format.ts +++ b/src/cli/terminal-format.ts @@ -1,10 +1,10 @@ import { PTY_LIVE_NOTE, describeUnconfirmedStop } from '../shared/pty-liveness-verdict' import { structuredChatPtyWriteRefusalCopy } from '../shared/agent-session-pty-write-refusal-copy' +import { formatListingHostScope, type WithAnnotatedHostScope } from './omitted-host-scope-selectors' import type { RuntimeTerminalClose, RuntimeTerminalCreate, RuntimeTerminalFocus, - RuntimeTerminalListHostScope, RuntimeTerminalListResult, RuntimeTerminalVisualLayout, RuntimeTerminalVisualLayoutNode, @@ -18,8 +18,10 @@ import type { RuntimeTerminalWait } from '../shared/runtime-types' -export function formatTerminalList(result: RuntimeTerminalListResult): string { - const scope = formatTerminalListHostScope(result.hostScope) +export function formatTerminalList( + result: WithAnnotatedHostScope +): string { + const scope = formatListingHostScope(result.hostScope) if (result.terminals.length === 0) { return `No terminals listed.\n${scope}` } @@ -37,18 +39,6 @@ export function formatTerminalList(result: RuntimeTerminalListResult): string { : bodyWithScope } -// Why: a listing that does not say what it covers reads as absolute, and an -// absent scope means the host is too old to know — not that it covered everything. -function formatTerminalListHostScope(scope: RuntimeTerminalListHostScope | undefined): string { - if (!scope) { - return 'scope: unverifiable — this host does not report which hosts it lists' - } - const covered = scope.hostIds.length > 0 ? scope.hostIds.join(', ') : 'none' - const omitted = - scope.omittedHostIds.length > 0 ? ` — not covered: ${scope.omittedHostIds.join(', ')}` : '' - return `scope: ${covered}${omitted}` -} - function formatTerminalVisualLayouts( layouts: readonly RuntimeTerminalVisualLayout[] | undefined ): string | null { diff --git a/src/cli/workspace-format.ts b/src/cli/workspace-format.ts index 8cdfce86b74..51a0369978b 100644 --- a/src/cli/workspace-format.ts +++ b/src/cli/workspace-format.ts @@ -7,6 +7,7 @@ import type { RuntimeWorktreeRecord } from '../shared/runtime-types' import type { MemorySnapshot, WorktreeMemory } from '../shared/process-stats-types' +import { formatListingHostScope, type WithAnnotatedHostScope } from './omitted-host-scope-selectors' export function formatMemorySnapshot(snapshot: MemorySnapshot): string { const topWorktrees = [...snapshot.worktrees].sort((a, b) => b.memory - a.memory).slice(0, 10) @@ -130,19 +131,21 @@ export function formatEnvironment(environment: PublicKnownRuntimeEnvironment): s ].join('\n') } -export function formatWorktreePs(result: RuntimeWorktreePsResult): string { +export function formatWorktreePs(result: WithAnnotatedHostScope): string { + const scope = formatListingHostScope(result.hostScope) if (result.worktrees.length === 0) { - return 'No worktrees found.' + return `No worktrees found.\n${scope}` } const body = result.worktrees .map( (worktree) => - `${worktree.repo} ${worktree.branch} live:${worktree.liveTerminalCount} pty:${worktree.hasAttachedPty ? 'yes' : 'no'} unread:${worktree.unread ? 'yes' : 'no'}\n${worktree.path}${worktree.preview ? `\npreview: ${worktree.preview}` : ''}` + `${worktree.repo} ${worktree.branch} host=${worktree.hostId ?? 'unverifiable'} live:${worktree.liveTerminalCount} pty:${worktree.hasAttachedPty ? 'yes' : 'no'} unread:${worktree.unread ? 'yes' : 'no'}\n${worktree.path}${worktree.preview ? `\npreview: ${worktree.preview}` : ''}` ) .join('\n\n') + const bodyWithScope = `${body}\n\n${scope}` return result.truncated - ? `${body}\n\ntruncated: showing ${result.worktrees.length} of ${result.totalCount}` - : body + ? `${bodyWithScope}\ntruncated: showing ${result.worktrees.length} of ${result.totalCount}` + : bodyWithScope } export function formatRepoList(result: RuntimeRepoList): string { @@ -168,19 +171,23 @@ export function formatRepoRefs(result: RuntimeRepoSearchRefs): string { return result.truncated ? `${result.refs.join('\n')}\n\ntruncated: yes` : result.refs.join('\n') } -export function formatWorktreeList(result: RuntimeWorktreeListResult): string { +export function formatWorktreeList( + result: WithAnnotatedHostScope +): string { + const scope = formatListingHostScope(result.hostScope) if (result.worktrees.length === 0) { - return 'No worktrees found.' + return `No worktrees found.\n${scope}` } const body = result.worktrees .map((worktree) => { const childCount = worktree.childWorktreeIds?.length ?? 0 - return `${String(worktree.id)} ${String(worktree.branch)} ${String(worktree.path)}\ndisplayName: ${String(worktree.displayName ?? '')}\nparentWorktreeId: ${String(worktree.parentWorktreeId ?? 'null')}\nchildWorktreeIds: ${childCount > 0 ? worktree.childWorktreeIds.join(',') : '[]'}\nlinkedIssue: ${String(worktree.linkedIssue ?? 'null')}\ncomment: ${String(worktree.comment ?? '')}` + return `${String(worktree.id)} ${String(worktree.branch)} host=${String(worktree.hostId ?? 'unverifiable')} ${String(worktree.path)}\ndisplayName: ${String(worktree.displayName ?? '')}\nparentWorktreeId: ${String(worktree.parentWorktreeId ?? 'null')}\nchildWorktreeIds: ${childCount > 0 ? worktree.childWorktreeIds.join(',') : '[]'}\nlinkedIssue: ${String(worktree.linkedIssue ?? 'null')}\ncomment: ${String(worktree.comment ?? '')}` }) .join('\n\n') + const bodyWithScope = `${body}\n\n${scope}` return result.truncated - ? `${body}\n\ntruncated: showing ${result.worktrees.length} of ${result.totalCount}` - : body + ? `${bodyWithScope}\ntruncated: showing ${result.worktrees.length} of ${result.totalCount}` + : bodyWithScope } export function formatWorktreeShow(result: { worktree: RuntimeWorktreeRecord }): string { diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 94fc77f8158..42c9c7ff6d3 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -1,7 +1,7 @@ // @ts-nocheck -- mechanically split from OrcaRuntimeService; behavior is covered by AST equivalence and characterization tests. import { OrcaRuntimeWithStructuredAgentSessionRecoverTuiOwner } from './orca-runtime-structured-agent-session-recover-tui-owner' import { DEFAULT_WORKTREE_PS_LIMIT } from './orca-runtime-postlude' -import type { RuntimeWorktreePsSummary } from '../../shared/runtime-types' +import type { RuntimeWorktreePsResult } from '../../shared/runtime-types' import { buildRuntimeWorktreePsSummaries } from './runtime-worktree-ps-summaries' import { buildRuntimeWorktreeSummaryPathIndex } from './runtime-worktree-summary-paths' import { @@ -15,6 +15,7 @@ import { enrichMissingRepoGitRemoteIdentities } from '../repo-git-remote-identit import { ensureStructuredAgentSessionHost as installStructuredAgentSessionHost } from './structured-agent-session-runtime' import { getProfileUserDataPath } from '../orca-profiles/profile-storage-paths' import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { buildWorktreeListingPage } from './worktree-listing-host-scope' import { resolveTuiAgentLaunchArgs, resolveTuiAgentLaunchEnv @@ -30,11 +31,7 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent async getWorktreePs( limit = DEFAULT_WORKTREE_PS_LIMIT, sourceDefaultsSupported = true - ): Promise<{ - worktrees: RuntimeWorktreePsSummary[] - totalCount: number - truncated: boolean - }> { + ): Promise { if (!Number.isInteger(limit) || limit <= 0) { throw new Error('invalid_limit') } @@ -111,11 +108,9 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent }) const sorted = [...summaries.values()].sort(compareWorktreePs) - return { - worktrees: sorted.slice(0, limit), - totalCount: sorted.length, - truncated: sorted.length > limit - } + // Why: the same cap starvation as worktree.list — a host whose rows all sort last gets no + // page at all, which is indistinguishable from it having no workspaces (#18104). + return buildWorktreeListingPage(sorted, limit, this.listKnownExecutionHostIds()) } listRepos(): Repo[] { diff --git a/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts b/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts index 278ccd28a74..346006cd5fd 100644 --- a/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts +++ b/src/main/runtime/orca-runtime-stop-requested-pty-ids.ts @@ -136,7 +136,8 @@ export class OrcaRuntimeWithStopRequestedPtyIds extends OrcaRuntimeWithRuntimeId listResolved: () => this.listResolvedWorktrees(), resolveRepo: (selector) => this.resolveRepoSelector(selector), selectRepos: (selector) => this.selectReposBySelector(selector), - scanRepo: (repo) => this.listRepoWorktreesForResolution(repo) + scanRepo: (repo) => this.listRepoWorktreesForResolution(repo), + listKnownHostIds: () => this.listKnownExecutionHostIds() }) protected readonly ptyForegroundAgent = new RuntimePtyForegroundAgent({ diff --git a/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-04.spec.ts b/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-04.spec.ts index 2d6042c280e..965a1a1bc3f 100644 --- a/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-04.spec.ts +++ b/src/main/runtime/orca-runtime-tests/mobile-creation-and-orchestration-part-04.spec.ts @@ -168,6 +168,8 @@ describe('OrcaRuntimeService', () => { agents: [] } ], + // Why: the summary now names the hosts it covered; an absent scope would read as absolute. + hostScope: { hostIds: ['local'], omittedHostIds: [] }, totalCount: 1, truncated: false }) diff --git a/src/main/runtime/runtime-managed-worktree-metadata-sweep.test.ts b/src/main/runtime/runtime-managed-worktree-metadata-sweep.test.ts index 6df19517d64..4913b31bd7d 100644 --- a/src/main/runtime/runtime-managed-worktree-metadata-sweep.test.ts +++ b/src/main/runtime/runtime-managed-worktree-metadata-sweep.test.ts @@ -44,7 +44,8 @@ function queries( listResolved: async () => [], resolveRepo: async () => repo, selectRepos: () => [repo], - scanRepo: async () => ({ ok, worktrees: [...worktrees] }) + scanRepo: async () => ({ ok, worktrees: [...worktrees] }), + listKnownHostIds: () => [] }) } diff --git a/src/main/runtime/runtime-managed-worktree-queries.test.ts b/src/main/runtime/runtime-managed-worktree-queries.test.ts index 354b6653324..01df53cbd1d 100644 --- a/src/main/runtime/runtime-managed-worktree-queries.test.ts +++ b/src/main/runtime/runtime-managed-worktree-queries.test.ts @@ -46,7 +46,8 @@ function queries(store: RuntimeStore): RuntimeManagedWorktreeQueries { listResolved: async () => [], resolveRepo: async () => store.getRepos()[0]!, selectRepos: () => store.getRepos(), - scanRepo: async () => ({ ok: true, worktrees: [] }) + scanRepo: async () => ({ ok: true, worktrees: [] }), + listKnownHostIds: () => [] }) } diff --git a/src/main/runtime/runtime-managed-worktree-queries.ts b/src/main/runtime/runtime-managed-worktree-queries.ts index b0ed2bc4a3b..5a5812236af 100644 --- a/src/main/runtime/runtime-managed-worktree-queries.ts +++ b/src/main/runtime/runtime-managed-worktree-queries.ts @@ -1,7 +1,8 @@ import type { DetectedWorktreeListResult, Worktree } from '../../shared/worktree/types' import type { Repo } from '../../shared/repo-types' import type { RuntimeWorktreeListResult } from '../../shared/runtime-types' -import { getRepoExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' +import { buildWorktreeListingPage } from './worktree-listing-host-scope' import { readWorktreeMetaForHost } from '../persistence/host-qualified-worktree-meta' import { getRepoOwnedWorktreeMeta } from '../worktree-metadata-ownership' import type { WorktreeMeta } from '../../shared/worktree/meta-types' @@ -38,6 +39,8 @@ type Dependencies = { resolveRepo(selector: string): Promise selectRepos(selector: string): Repo[] scanRepo(repo: Repo): Promise + /** Hosts this runtime has repos or workspaces on, so a host with no rows is still named. */ + listKnownHostIds(): Iterable } /** @@ -100,11 +103,9 @@ export class RuntimeManagedWorktreeQueries { (!repoId || worktree.repoId === repoId) && this.isVisible(worktree, matchers.get(worktree.repoId), sourceDefaultsSupported) ) - return { - worktrees: worktrees.slice(0, limit), - totalCount: worktrees.length, - truncated: worktrees.length > limit - } + // Why: a `--repo` listing was scoped by the caller, so naming every configured host as + // omitted would report a gap the caller deliberately excluded. + return buildWorktreeListingPage(worktrees, limit, repoId ? [] : this.deps.listKnownHostIds()) } resolveRepoForConnection(selector: string, connectionId?: string | null): Promise { diff --git a/src/main/runtime/worktree-list-host-scope.test.ts b/src/main/runtime/worktree-list-host-scope.test.ts new file mode 100644 index 00000000000..e497028f4c4 --- /dev/null +++ b/src/main/runtime/worktree-list-host-scope.test.ts @@ -0,0 +1,170 @@ +import { describe, expect, it, vi } from 'vitest' +import type { ExecutionHostId } from '../../shared/execution-host' +import type { Repo } from '../../shared/repo-types' +import { selectHostBalancedPage } from '../../shared/host-balanced-listing-page' +import { RuntimeManagedWorktreeQueries } from './runtime-managed-worktree-queries' +import type { ResolvedWorktree } from './runtime-worktree-path-identity' +import type { RuntimeStore } from './runtime-store-contract' + +const LOCAL_REPO: Repo = { + id: 'repo-local', + path: '/workspace/app', + displayName: 'app', + badgeColor: '#000000', + addedAt: 1 +} + +const SSH_REPO: Repo = { + ...LOCAL_REPO, + id: 'repo-ssh', + connectionId: 'box-1', + displayName: 'app (remote)' +} + +const settings = { + workspaceDir: '/worktrees', + nestWorkspaces: true, + refreshLocalBaseRefOnWorktreeCreate: false, + branchPrefix: 'none', + branchPrefixCustom: '' +} + +function worktree(repoId: string, path: string, hostId: string): ResolvedWorktree { + return { + id: `${repoId}::${path}`, + repoId, + path, + branch: 'main', + hostId, + displayName: path, + comment: '', + linkedIssue: null, + parentWorktreeId: null, + childWorktreeIds: [], + lineage: null, + git: { path, head: 'abc', branch: 'main', isBare: false, isMainWorktree: false } + } as unknown as ResolvedWorktree +} + +/** The reproduced shape from #18104: every remote row lands contiguously at the end. */ +function fleet(localCount: number, sshCount: number): ResolvedWorktree[] { + return [ + ...Array.from({ length: localCount }, (_, index) => + worktree(LOCAL_REPO.id, `/worktrees/local-${index}`, 'local') + ), + ...Array.from({ length: sshCount }, (_, index) => + worktree(SSH_REPO.id, `/remote/wt-${index}`, 'ssh:box-1') + ) + ] +} + +function queries( + resolved: ResolvedWorktree[], + knownHostIds: ExecutionHostId[] = ['local', 'ssh:box-1'] +): RuntimeManagedWorktreeQueries { + const store = { + getRepos: () => [LOCAL_REPO, SSH_REPO], + getRepo: () => LOCAL_REPO, + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: vi.fn(), + getAllWorktreeLineage: () => ({}), + getSettings: () => settings + } as unknown as RuntimeStore + return new RuntimeManagedWorktreeQueries({ + getStore: () => store, + listResolved: async () => resolved, + resolveRepo: async () => SSH_REPO, + selectRepos: () => [SSH_REPO], + scanRepo: async () => ({ ok: true, worktrees: [] }), + listKnownHostIds: () => knownHostIds + }) +} + +describe('worktree.list host coverage under the row cap', () => { + it('returns remote rows that sit entirely past the cap', async () => { + // Why #18104: 497 local + 24 SSH rows, SSH at indices 496-520, and a 200-row cap returned + // `{local: 200}` — zero of 24 remote worktrees, with nothing saying the gap was a whole host. + const result = await queries(fleet(497, 24)).list(undefined, 200) + + expect(result.totalCount).toBe(521) + expect(result.truncated).toBe(true) + expect(result.worktrees).toHaveLength(200) + const remote = result.worktrees.filter((row) => row.hostId === 'ssh:box-1') + expect(remote).toHaveLength(24) + expect(result.hostScope).toEqual({ hostIds: ['local', 'ssh:box-1'], omittedHostIds: [] }) + }) + + it('keeps the page a subsequence of the unbounded listing', async () => { + // Why: balancing decides which rows survive the cap, never how the survivors are ordered. + const resolved = fleet(497, 24) + const result = await queries(resolved).list(undefined, 200) + + const positions = result.worktrees.map((row) => resolved.findIndex((it) => it.id === row.id)) + expect(positions).toEqual([...positions].sort((left, right) => left - right)) + }) + + it('names a configured host that contributed no rows at all', async () => { + // Why: a repo whose scan failed contributes zero rows exactly like a host with no worktrees. + // docs/reference/ssh-execution-boundary.md forbids the listing from reading as absolute there. + const result = await queries(fleet(3, 0), ['local', 'ssh:box-1', 'runtime:paired']).list( + undefined, + 200 + ) + + expect(result.hostScope).toEqual({ + hostIds: ['local'], + omittedHostIds: ['runtime:paired', 'ssh:box-1'] + }) + }) + + it('does not report configured hosts as omitted from a --repo listing', async () => { + // Why: the caller scoped this themselves, so naming the hosts they excluded is noise. + const result = await queries(fleet(0, 5)).list('id:repo-ssh', 200) + + expect(result.hostScope).toEqual({ hostIds: ['ssh:box-1'], omittedHostIds: [] }) + }) + + it('leaves an uncapped listing byte-identical', async () => { + const resolved = fleet(4, 2) + const result = await queries(resolved).list(undefined, 200) + + expect(result.worktrees.map((row) => row.id)).toEqual(resolved.map((row) => row.id)) + expect(result.truncated).toBe(false) + }) +}) + +describe('selectHostBalancedPage', () => { + it('gives every host a share of the cap rather than filling it from the first', () => { + const rows = [ + ...Array.from({ length: 10 }, (_, index) => ({ host: 'local', index })), + ...Array.from({ length: 10 }, (_, index) => ({ host: 'ssh:box-1', index: index + 10 })) + ] + + const page = selectHostBalancedPage(rows, 4, (row) => row.host) + + expect(page.map((row) => row.host)).toEqual(['local', 'local', 'ssh:box-1', 'ssh:box-1']) + }) + + it('fills the cap from the remaining hosts when one runs out of rows', () => { + const rows = [ + { host: 'local', id: 'a' }, + { host: 'local', id: 'b' }, + { host: 'local', id: 'c' }, + { host: 'ssh:box-1', id: 'd' } + ] + + const page = selectHostBalancedPage(rows, 3, (row) => row.host) + + expect(page.map((row) => row.id)).toEqual(['a', 'b', 'd']) + }) + + it('buckets rows with no host together instead of dropping them', () => { + const rows = [{ id: 'a' }, { id: 'b' }, { id: 'c' }] + + expect(selectHostBalancedPage(rows, 2, () => undefined).map((row) => row.id)).toEqual([ + 'a', + 'b' + ]) + }) +}) diff --git a/src/main/runtime/worktree-listing-host-scope.ts b/src/main/runtime/worktree-listing-host-scope.ts new file mode 100644 index 00000000000..99650882670 --- /dev/null +++ b/src/main/runtime/worktree-listing-host-scope.ts @@ -0,0 +1,64 @@ +import type { ExecutionHostId } from '../../shared/execution-host' +import { selectHostBalancedPage } from '../../shared/host-balanced-listing-page' +import type { RuntimeListingHostScope } from '../../shared/runtime-listing-host-scope' + +/** + * Applies a worktree listing's row cap and reports which hosts the resulting page covers. + * + * Rows are resolved repo by repo, so every SSH repo's rows land contiguously at the end of the + * fleet order: 24 remote worktrees sat at indices 496-520 of 521 and a 200-row cap returned zero + * of them (#18104). Balancing the page across hosts fixes the starvation; the scope is what makes + * the remaining gap legible, because a host with no rows in the page is otherwise indistinguishable + * from a host with no worktrees — which `docs/reference/ssh-execution-boundary.md` forbids a + * listing from implying. + */ +export function buildWorktreeListingPage( + rows: readonly TRow[], + limit: number, + knownHostIds: Iterable +): { + worktrees: TRow[] + hostScope: RuntimeListingHostScope + totalCount: number + truncated: boolean +} { + const page = selectHostBalancedPage(rows, limit, (row) => row.hostId) + return { + worktrees: page, + hostScope: buildWorktreeListingHostScope({ + pageHostIds: page.map((row) => row.hostId), + matchedHostIds: rows.map((row) => row.hostId), + knownHostIds + }), + totalCount: rows.length, + truncated: rows.length > limit + } +} + +/** + * The worktree-listing counterpart of `buildTerminalListHostScope`: names the hosts the returned + * page covers, and every host it does not — including a configured repo whose scan failed, which + * contributes zero rows exactly like a host with no worktrees. + */ +export function buildWorktreeListingHostScope(args: { + /** Hosts of the rows actually returned. */ + pageHostIds: Iterable + /** Hosts of every row that matched, including those the cap dropped. */ + matchedHostIds: Iterable + /** Hosts this runtime has configured repos or workspaces on, even if they contributed no rows. */ + knownHostIds: Iterable +}): RuntimeListingHostScope { + const covered = new Set() + for (const hostId of args.pageHostIds) { + if (hostId) { + covered.add(hostId) + } + } + const omitted = new Set() + for (const hostId of [...args.matchedHostIds, ...args.knownHostIds]) { + if (hostId && !covered.has(hostId)) { + omitted.add(hostId) + } + } + return { hostIds: [...covered].sort(), omittedHostIds: [...omitted].sort() } +} diff --git a/src/main/runtime/worktree-ps-host-scope.test.ts b/src/main/runtime/worktree-ps-host-scope.test.ts new file mode 100644 index 00000000000..cf718cd267f --- /dev/null +++ b/src/main/runtime/worktree-ps-host-scope.test.ts @@ -0,0 +1,131 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const electronMocks = vi.hoisted(() => { + const ipcMain = { + on: vi.fn(() => ipcMain), + removeListener: vi.fn(() => ipcMain), + emit: vi.fn(() => true) + } + return { + BrowserWindow: { fromId: vi.fn((): unknown => null) }, + webContents: { fromId: vi.fn((): unknown => null) }, + ipcMain, + app: { getPath: vi.fn(() => '/tmp'), isPackaged: false } + } +}) +vi.mock('electron', () => electronMocks) + +const getSshGitProviderMock = vi.hoisted(() => vi.fn()) +vi.mock('../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock, + getSshGitProviderGeneration: vi.fn(() => 0), + SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE: 'unavailable', + requireSshGitProvider: (connectionId: string) => getSshGitProviderMock(connectionId) +})) + +const listWorktreesStrictMock = vi.hoisted(() => vi.fn()) +vi.mock('../git/worktree', async (importOriginal) => ({ + ...(await importOriginal>()), + listWorktreesStrict: listWorktreesStrictMock +})) + +import { OrcaRuntimeService } from './orca-runtime' + +const LOCAL_REPO_ID = 'repo-local' +const LOCAL_REPO_PATH = '/Users/me/dev/app' +const SSH_REPO_ID = 'repo-ssh' +const SSH_REPO_PATH = '/home/user/app' +const SSH_CONNECTION_ID = 'box-1' + +function gitWorktree(path: string, isMain = false) { + return { path, head: 'abc', branch: 'main', isBare: false, isMainWorktree: isMain } +} + +/** Local rows sort ahead of the remote ones, mirroring the fleet order that starves the cap. */ +function makeStore() { + const metaById: Record = {} + return { + getRepo: (id: string) => + makeStore() + .getRepos() + .find((repo) => repo.id === id), + getRepos: () => [ + { + id: LOCAL_REPO_ID, + path: LOCAL_REPO_PATH, + displayName: 'app', + badgeColor: 'blue', + addedAt: 1 + }, + { + id: SSH_REPO_ID, + path: SSH_REPO_PATH, + displayName: 'app remote', + badgeColor: 'blue', + addedAt: 2, + connectionId: SSH_CONNECTION_ID + } + ], + getAllWorktreeMeta: () => metaById, + getWorktreeMeta: (id: string) => metaById[id], + setWorktreeMeta: (id: string, meta: Record) => { + metaById[id] = { ...(metaById[id] as object), ...meta } + return metaById[id] + }, + removeWorktreeMeta: () => {}, + getAllWorktreeLineage: () => ({}), + getAllWorkspaceLineage: () => ({}), + removeWorktreeLineage: vi.fn(), + removeWorkspaceLineage: vi.fn(), + getGitHubCache: () => undefined as never, + getSettings: () => ({ + workspaceDir: '/tmp/workspaces', + nestWorkspaces: false, + refreshLocalBaseRefOnWorktreeCreate: false, + branchPrefix: 'none', + branchPrefixCustom: '' + }), + getProjects: () => [] + } +} + +describe('worktree.ps host coverage', () => { + beforeEach(() => { + getSshGitProviderMock.mockReset() + listWorktreesStrictMock.mockReset() + listWorktreesStrictMock.mockResolvedValue([ + gitWorktree(LOCAL_REPO_PATH, true), + gitWorktree(`${LOCAL_REPO_PATH}-a`), + gitWorktree(`${LOCAL_REPO_PATH}-b`), + gitWorktree(`${LOCAL_REPO_PATH}-c`) + ]) + getSshGitProviderMock.mockReturnValue({ + listWorktrees: vi.fn(async () => [ + gitWorktree(SSH_REPO_PATH, true), + gitWorktree(`${SSH_REPO_PATH}-a`) + ]) + }) + }) + + it('names every host the page covers', async () => { + const runtime = new OrcaRuntimeService(makeStore() as never) + + const result = await runtime.getWorktreePs(10_000) + + expect(result.hostScope?.hostIds).toEqual(['local', `ssh:${SSH_CONNECTION_ID}`]) + expect(result.hostScope?.omittedHostIds).toEqual([]) + }) + + it('keeps a remote row in the page when the cap cannot hold every local row', async () => { + const runtime = new OrcaRuntimeService(makeStore() as never) + + const result = await runtime.getWorktreePs(2) + + expect(result.truncated).toBe(true) + expect(result.worktrees).toHaveLength(2) + expect(result.worktrees.map((worktree) => worktree.hostId)).toContain( + `ssh:${SSH_CONNECTION_ID}` + ) + expect(result.hostScope?.hostIds).toEqual(['local', `ssh:${SSH_CONNECTION_ID}`]) + }) +}) diff --git a/src/shared/host-balanced-listing-page.ts b/src/shared/host-balanced-listing-page.ts new file mode 100644 index 00000000000..f771d98fdd4 --- /dev/null +++ b/src/shared/host-balanced-listing-page.ts @@ -0,0 +1,49 @@ +/** + * Chooses which rows survive a listing's row cap so that no execution host is starved by it. + * + * Worktree rows are resolved repo by repo, so every SSH repo's rows land contiguously at the end + * of the fleet order — 24 remote worktrees sat at indices 496-520 of 521 and a 200-row cap + * returned zero of them (#18104). A per-host round robin gives each host a share of the cap. + * + * Chosen rows keep the caller's original relative order, so the page stays a subsequence of the + * unbounded listing and nothing downstream has to re-sort. An uncapped listing is returned as-is. + */ +export function selectHostBalancedPage( + rows: readonly TRow[], + limit: number, + getHostId: (row: TRow) => string | null | undefined +): TRow[] { + if (rows.length <= limit) { + return [...rows] + } + // Insertion order is first-appearance order per host, so the round robin is deterministic. + const indicesByHost = new Map() + rows.forEach((row, index) => { + const hostId = getHostId(row) ?? '' + const bucket = indicesByHost.get(hostId) + if (bucket) { + bucket.push(index) + } else { + indicesByHost.set(hostId, [index]) + } + }) + const buckets = [...indicesByHost.values()] + const cursors = buckets.map(() => 0) + const chosen: number[] = [] + while (chosen.length < limit) { + let advanced = false + for (let bucket = 0; bucket < buckets.length && chosen.length < limit; bucket += 1) { + const cursor = cursors[bucket] ?? 0 + const index = buckets[bucket]?.[cursor] + if (index !== undefined) { + chosen.push(index) + cursors[bucket] = cursor + 1 + advanced = true + } + } + if (!advanced) { + break + } + } + return chosen.sort((left, right) => left - right).map((index) => rows[index] as TRow) +} diff --git a/src/shared/runtime-listing-host-scope.ts b/src/shared/runtime-listing-host-scope.ts new file mode 100644 index 00000000000..6232b0a4259 --- /dev/null +++ b/src/shared/runtime-listing-host-scope.ts @@ -0,0 +1,12 @@ +import type { ExecutionHostId } from './execution-host' + +/** + * What a bounded listing did and did not cover, by execution host. An absent scope means the + * host is too old to report one — not that it covered everything. See + * `docs/reference/ssh-execution-boundary.md`: a listing is only evidence about the hosts it + * actually covered, so an empty answer for a host that is missing here proves nothing. + */ +export type RuntimeListingHostScope = { + hostIds: ExecutionHostId[] + omittedHostIds: ExecutionHostId[] +} diff --git a/src/shared/runtime-terminal-contracts.ts b/src/shared/runtime-terminal-contracts.ts index d2d293c3a72..a75a2256bdb 100644 --- a/src/shared/runtime-terminal-contracts.ts +++ b/src/shared/runtime-terminal-contracts.ts @@ -6,6 +6,7 @@ import type { import type { StartupCommandDelivery } from './codex-startup-delivery' import type { ExecutionHostId } from './execution-host' import type { PtyIncarnationId } from './pty-incarnation' +import type { RuntimeListingHostScope } from './runtime-listing-host-scope' import type { RuntimeMobileSessionTabsResult } from './runtime-session-contracts' import type { TabGroupLayoutNode } from './tab-types' import type { TerminalExitCause } from './terminal-exit-cause' @@ -83,10 +84,8 @@ export type RuntimeTerminalVisualLayout = { root: RuntimeTerminalVisualLayoutNode } -export type RuntimeTerminalListHostScope = { - hostIds: ExecutionHostId[] - omittedHostIds: ExecutionHostId[] -} +/** The shared listing-scope shape, kept under its incumbent name for existing consumers. */ +export type RuntimeTerminalListHostScope = RuntimeListingHostScope export type RuntimeTerminalListResult = { terminals: RuntimeTerminalSummary[] diff --git a/src/shared/runtime-worktree-contracts.ts b/src/shared/runtime-worktree-contracts.ts index 1a053a91d1b..df0d4c68742 100644 --- a/src/shared/runtime-worktree-contracts.ts +++ b/src/shared/runtime-worktree-contracts.ts @@ -6,6 +6,7 @@ import type { WorktreeLineage, WorktreeLineageWarning } from './worktree/lineage-types' +import type { RuntimeListingHostScope } from './runtime-listing-host-scope' import type { GitWorktreeInfo, Worktree } from './worktree/types' export type RuntimeWorktreeAgentRow = { @@ -125,6 +126,8 @@ export type RuntimeWorktreePsResult = { worktrees: RuntimeWorktreePsSummary[] totalCount: number truncated: boolean + /** Absent from hosts that predate the field; treat that scope as unverifiable. */ + hostScope?: RuntimeListingHostScope } export type RuntimeWorktreePsSnapshotResult = RuntimeWorktreePsResult & { snapshotId: string } @@ -150,4 +153,6 @@ export type RuntimeWorktreeListResult = { worktrees: RuntimeWorktreeRecord[] totalCount: number truncated: boolean + /** Absent from hosts that predate the field; treat that scope as unverifiable. */ + hostScope?: RuntimeListingHostScope } From f974e981628ab1b113492b75c4454f10595bf05e Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 17:43:40 -0400 Subject: [PATCH 196/398] test(cloud): derive both reachability directions for the relay inventory census (#18524) Mirrors stablyai/orca-cloud#472 (c82f98f), byte-identical under cloud/. --- cloud/apps/relay/src/assignment-store.ts | 2 +- .../src/cell-inventory-hold-samples.test.ts | 70 +++++++++++++ .../src/cell-inventory-lock-census.test.ts | 97 +++++++++++++------ .../relay/src/regional-rehome-store.test.ts | 61 +++++++++++- 4 files changed, 196 insertions(+), 34 deletions(-) create mode 100644 cloud/apps/relay/src/cell-inventory-hold-samples.test.ts diff --git a/cloud/apps/relay/src/assignment-store.ts b/cloud/apps/relay/src/assignment-store.ts index 3571b57cd22..96d66eb7dc2 100644 --- a/cloud/apps/relay/src/assignment-store.ts +++ b/cloud/apps/relay/src/assignment-store.ts @@ -7969,7 +7969,7 @@ function isDatabaseLockUnavailable(error: unknown): boolean { return error instanceof Error && error.message === 'database_lock_unavailable' } -function cellInventoryLockOptions(mode: CellInventoryLockMode): RelayLockOptions { +export function cellInventoryLockOptions(mode: CellInventoryLockMode): RelayLockOptions { if (mode === 'nowait') return { failIfUnavailable: true, measureHoldMs: true } if (mode === 'pool-default') return { measureHoldMs: true } return { lockTimeoutMs: CELL_INVENTORY_LOCK_TIMEOUT_MS, measureHoldMs: true } diff --git a/cloud/apps/relay/src/cell-inventory-hold-samples.test.ts b/cloud/apps/relay/src/cell-inventory-hold-samples.test.ts new file mode 100644 index 00000000000..abd37dcbb29 --- /dev/null +++ b/cloud/apps/relay/src/cell-inventory-hold-samples.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it } from 'vitest' +import { + CellInventoryHoldSamples, + emptyCellInventoryHoldCounts +} from './cell-inventory-hold-samples.js' + +// Nearest rank, computed in integer arithmetic so it cannot inherit the float +// error the implementation's `0.95 * n` could in principle carry. +function nearestRankP95(sorted: number[]): number { + return sorted[Math.ceil((95 * sorted.length) / 100) - 1]! +} + +function samplesOf(values: number[]): CellInventoryHoldSamples { + const samples = new CellInventoryHoldSamples() + for (const value of values) samples.record(value) + return samples +} + +describe('cell inventory hold samples', () => { + it('reports nothing before the first hold', () => { + expect(new CellInventoryHoldSamples().readCounts()).toEqual( + emptyCellInventoryHoldCounts() + ) + }) + + // Why: the 500ms bound will be tuned against this percentile, so an off-by-one + // here reads as a hold the fleet never had. + it('places p95 at the nearest rank for every window size', () => { + for (let size = 1; size <= 400; size++) { + const values = Array.from({ length: size }, (_, index) => index + 1) + const shuffled = [...values].reverse() + + const counts = samplesOf(shuffled).readCounts() + + expect(counts.cellInventoryHoldMsP95).toBe(nearestRankP95(values)) + expect(counts.cellInventoryHoldMsMax).toBe(size) + expect(counts.cellInventoryHolds).toBe(size) + } + }) + + it('never reports a p95 above the max', () => { + for (let size = 1; size <= 200; size++) { + const counts = samplesOf(Array.from({ length: size }, (_, i) => i + 1)).readCounts() + + expect(counts.cellInventoryHoldMsP95).toBeLessThanOrEqual(counts.cellInventoryHoldMsMax) + } + }) + + it('ignores a hold that is not a finite, non-negative duration', () => { + const samples = samplesOf([Number.NaN, Number.POSITIVE_INFINITY, -1]) + + expect(samples.readCounts()).toEqual(emptyCellInventoryHoldCounts()) + }) + + // Why: the reservoir is bounded, so a heavy flush interval keeps the most + // recent holds rather than growing without limit or freezing on the oldest. + it('keeps the most recent holds once the reservoir is full', () => { + const counts = samplesOf(Array.from({ length: 2_100 }, (_, index) => index + 1)).readCounts() + + expect(counts.cellInventoryHolds).toBe(2_048) + expect(counts.cellInventoryHoldMsMax).toBe(2_100) + }) + + it('resets the window on consume so each flush reports its own holds', () => { + const samples = samplesOf([5, 10]) + + expect(samples.consumeCounts().cellInventoryHolds).toBe(2) + expect(samples.consumeCounts()).toEqual(emptyCellInventoryHoldCounts()) + }) +}) diff --git a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts index d0527534935..b2b5b65684c 100644 --- a/cloud/apps/relay/src/cell-inventory-lock-census.test.ts +++ b/cloud/apps/relay/src/cell-inventory-lock-census.test.ts @@ -1,11 +1,11 @@ import { readFileSync } from 'node:fs' import { describe, expect, it } from 'vitest' -import type { CellInventoryLockMode } from './assignment-store.js' +import { cellInventoryLockOptions, type CellInventoryLockMode } from './assignment-store.js' // Which entry points can reach a call site. A site a sweep can enter must never // take the bounded wait: its 55P03 becomes a terminal transaction failure, and // the incident monitor freezes on a single one. -type Reachability = 'request' | 'sweep' | 'both' +type Reachability = 'request' | 'sweep' | 'both' | 'orphan' // 'caller' is not a CellInventoryLockMode: those sites take the mode threaded // from `assign`, which is 'request' for a client and 'pool-default' for the @@ -25,7 +25,8 @@ const CENSUS: CensusEntry[] = [ { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'assignOnce', mode: 'nowait', reach: 'both' }, { method: 'refreshDrainMigrationLeasesOnce', mode: 'request', reach: 'request' }, - { method: 'changeActivity', mode: 'request', reach: 'request' }, + // Reachable from neither: changeActivity has no production callers, only tests. + { method: 'changeActivity', mode: 'request', reach: 'orphan' }, { method: 'acquireActivity', mode: 'request', reach: 'request' }, { method: 'activateControl', mode: 'request', reach: 'request' }, { method: 'startEvacuation', mode: 'request', reach: 'request' }, @@ -52,20 +53,15 @@ const CENSUS: CensusEntry[] = [ ] // The background sweeps, and nothing else. A method reachable from one of these -// can be entered by a sweep tick, whatever else can also enter it. -const SWEEP_ROOTS = [ - 'refreshRegionalRehomeLeases', - 'completeReadyEvacuations', - 'completeReadyRegionalRehomes', - 'abortExpiredEvacuations', - 'abortExpiredRegionalRehomes', - 'reapRegionalRehomeAttempts', - 'releaseExpiredActivityLeases', - 'releaseExpiredActivity', - 'releaseExpiredRegionPreferences', - 'evacuateDeadCells', - 'claimRegionalRehome', - 'recordRegionalRehomeDispatchFailure' +// can be entered by a sweep tick, whatever else can also enter it. Both lists are +// read from source, so a new sweep step or a new route widens the derivation here +// instead of silently widening what a bounded wait can be entered from. +const SWEEP_ENTRY_FILES = ['./assignment-cleanup-steps.ts', './regional-rehome-worker.ts'] +const REQUEST_ENTRY_FILES = [ + './app.ts', + './relay-server.ts', + './host-session-registry.ts', + './cell-admission-startup.ts' ] const DECLARATION = /^ {2}(?:private |public )?(?:static )?(?:async )?([A-Za-z_][\w]*)[(<]/ @@ -74,9 +70,18 @@ function storeSource(): string[] { return readFileSync(new URL('./assignment-store.ts', import.meta.url), 'utf8').split('\n') } -// Why: a hand-written reachability column is a claim, not a check. Derive it, so -// a new sweep edge into a bounded site fails here instead of in production. -function sweepReachableMethods(lines: string[]): Set { +function entryPoints(files: string[]): string[] { + return files.flatMap((file) => + [ + ...readFileSync(new URL(file, import.meta.url), 'utf8').matchAll( + /assignments\.([A-Za-z_][\w]*)\(/g + ) + ].map((call) => call[1]!) + ) +} + +// Same-class call graph: store methods only ever reach each other through `this.`. +function storeCallGraph(lines: string[]): Map> { const bounds: { name: string; start: number }[] = [] lines.forEach((line, index) => { const declaration = DECLARATION.exec(line) @@ -93,8 +98,12 @@ function sweepReachableMethods(lines: string[]): Set { } callees.set(method.name, names) }) + return callees +} + +function closure(callees: Map>, roots: string[]): Set { const reached = new Set() - const pending = [...SWEEP_ROOTS] + const pending = [...roots] while (pending.length > 0) { const name = pending.pop()! if (reached.has(name)) continue @@ -104,6 +113,23 @@ function sweepReachableMethods(lines: string[]): Set { return reached } +// Why: a hand-written reachability column is a claim, not a check. Derive both +// directions, so a new sweep edge into a bounded site fails here instead of in +// production, and so 'sweep' and 'both' stop being asserted by hand. +function derivedReachability(lines: string[]): (method: string) => Reachability { + const callees = storeCallGraph(lines) + const sweep = closure(callees, entryPoints(SWEEP_ENTRY_FILES)) + const request = closure(callees, entryPoints(REQUEST_ENTRY_FILES)) + return (method) => + sweep.has(method) + ? request.has(method) + ? 'both' + : 'sweep' + : request.has(method) + ? 'request' + : 'orphan' +} + function readCallSites(): { method: string; mode: CensusMode }[] { const sites: { method: string; mode: CensusMode }[] = [] let method = '' @@ -136,27 +162,42 @@ describe('cell inventory lock call-site census', () => { }) it('derives the same reachability the census claims', () => { - const reached = sweepReachableMethods(storeSource()) - const derived = readCallSites().map(({ method }) => reached.has(method)) + const reachOf = derivedReachability(storeSource()) - expect(derived).toEqual(CENSUS.map((entry) => entry.reach !== 'request')) + expect(readCallSites().map(({ method }) => reachOf(method))).toEqual( + CENSUS.map((entry) => entry.reach) + ) }) // Why: this is the whole point of the classification. A shorter wait on a // sweep-reachable site turns contention into a terminal transaction failure, // and relayPostgresRetryExhausted freezes the incident gate at zero. + // Why: the hold distribution is what the 500ms bound will be tuned against, so + // a mode that stops asking for it goes unmeasured in exactly the lane that + // matters. Nothing else in the suite reads the pool-default branch. + it('measures the hold in every lock mode', () => { + const modes: CellInventoryLockMode[] = ['request', 'nowait', 'pool-default'] + + expect(modes.map((mode) => cellInventoryLockOptions(mode).measureHoldMs)).toEqual([ + true, + true, + true + ]) + }) + it('never puts a sweep-reachable site on the bounded wait', () => { - const reached = sweepReachableMethods(storeSource()) + const reachOf = derivedReachability(storeSource()) const bounded = readCallSites().filter( - (site) => site.mode === 'request' && reached.has(site.method) + (site) => site.mode === 'request' && ['sweep', 'both'].includes(reachOf(site.method)) ) expect(bounded).toEqual([]) }) it('routes every sweep-only site to NOWAIT so it can skip the tick', () => { - const queueing = CENSUS.filter( - (entry) => entry.reach === 'sweep' && entry.mode !== 'nowait' + const reachOf = derivedReachability(storeSource()) + const queueing = readCallSites().filter( + (site) => reachOf(site.method) === 'sweep' && site.mode !== 'nowait' ) expect(queueing).toEqual([]) diff --git a/cloud/apps/relay/src/regional-rehome-store.test.ts b/cloud/apps/relay/src/regional-rehome-store.test.ts index 6f57283396f..26f711189ee 100644 --- a/cloud/apps/relay/src/regional-rehome-store.test.ts +++ b/cloud/apps/relay/src/regional-rehome-store.test.ts @@ -626,7 +626,7 @@ describe('regional rehome assignment state', () => { await context.store.releaseActivity(identity, sourceControl) } probe.reset() - probe.failNoWaitOnce = true + probe.failNoWaitTimes = 1 const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') let completed: number @@ -647,6 +647,57 @@ describe('regional rehome assignment state', () => { await context.database.close() }) + // Why: with `continue` replaced by `break` a single contended candidate drops + // the rest of the page. Two in a row prove the sweep resumes, not just that it + // survived one, and that the summary counts both. + it('completes a candidate behind two contended ones', async () => { + const probe = new CellInventoryLockProbe() + const context = await setup({ wrap: (database) => probe.wrap(database) }) + const identities = [ + { userId: 'user-1', relayHostId: 'abcdefghijklmnop' }, + { userId: 'user-2', relayHostId: 'ponmlkjihgfedcba' }, + { userId: 'user-3', relayHostId: 'aaaabbbbccccdddd' } + ] + for (const identity of identities) { + // Dispatch is rate limited, so each claim needs its own interval. + context.advance(60_000) + await freshHeartbeats(context) + const sourceControl = await activatePreferredSource(context, identity) + const attempt = await context.store.claimRegionalRehome() + await context.store.recordRegionalRehomeDrainReceipt(attempt!.attemptId, 'accepted') + await context.store.activateControl(identity, { + cellId: target.id, + assignmentEpoch: 2, + generation: 1 + }) + await context.store.markMigrationTargetRegistered(identity, { + cellId: target.id, + assignmentEpoch: 2 + }) + await context.store.releaseActivity(identity, sourceControl) + } + probe.reset() + probe.failNoWaitTimes = 2 + const busy = collectEventWarnings('orca_relay_sweep_cell_inventory_busy') + + let completed: number + try { + completed = await context.store.completeReadyRegionalRehomes() + } finally { + busy.restore() + } + + expect(completed).toBe(1) + expect(busy.entries).toEqual([ + { + event: 'orca_relay_sweep_cell_inventory_busy', + sweep: 'complete-ready-regional-rehomes', + skipped: 2 + } + ]) + await context.database.close() + }) + // Why: only inventory contention is ordinary. Every other failure must keep its // existing propagation and its dispatch-failure accounting. it('propagates a claim failure that is not inventory contention', async () => { @@ -1692,8 +1743,8 @@ async function heartbeat( class CellInventoryLockProbe { readonly locks: (RelayLockOptions | undefined)[] = [] failNoWait = false - // Contends one candidate only, so the sweep must carry on to the next. - failNoWaitOnce = false + // Contends the first N candidates only, so the sweep must carry on past them. + failNoWaitTimes = 0 failWith: Error | null = null reset(): void { @@ -1708,8 +1759,8 @@ class CellInventoryLockProbe { if (sql.trim() === 'SELECT * FROM relay_cells ORDER BY cell_id ASC') { probe.locks.push(options) if (probe.failWith) throw probe.failWith - if (options?.failIfUnavailable && probe.failNoWaitOnce) { - probe.failNoWaitOnce = false + if (options?.failIfUnavailable && probe.failNoWaitTimes > 0) { + probe.failNoWaitTimes-- throw new Error('database_lock_unavailable') } if (probe.failNoWait && options?.failIfUnavailable) { From f35015d0c8974e5c94e4701fe8f1e9bcedc94fac Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:44:32 -0700 Subject: [PATCH 197/398] fix(ssh): measure pane idleness in the unit the sweep's kill operates on (#18415) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The orphan-relay-PTY sweep authorizes `pty.shutdown { immediate: true }`, which runs `forceKillPosixPtyProcessGroups`: collect every process group on the pane's tty, then `killpg` each one. The blast radius is therefore (groups on the tty) x (members of those groups, wherever they are). The idleness evidence measured only the first factor, so three shapes read as idle and were SIGKILLed: - with job control off (`set +m`) a background job keeps the SHELL's pgid, so the tty carries exactly one process group and that group is running the user's build; - a child that drops the controlling terminal (`ioctl(TIOCNOTTY)` without `setsid`) keeps the pgid, reports `tpgid == -1`, and never appears in `ps -t `; - a double-forked grandchild keeps the pgid and tty but reparents to pid 1, so the `ppid` walk cannot reach it and the named-process backstop never fires. `shellOwnsEveryTtyProcessGroup` now also requires the shell's own process group to hold no other member anywhere in the table, indexed in the same single pass. A pids-per-tty set would catch the first and third but not the second, which is why the count is pgid-wide rather than tty-scoped. The wire field keeps its tty-shaped name: the value only ever became stricter, so an old client skips more, never less. Second, unrelated-in-mechanism but same file family: `foregroundSkipReason` summed `capturedAgeMs + evidenceAgeSinceListingMs` without validating either. A non-numeric `capturedAgeMs` makes the sum `NaN`, and `NaN > 5000` is false, so a malformed record PASSED the freshness gate and proceeded toward the stop — the one place in the file that defaulted toward kill. Nothing validated it on this path (`mapSshPtyProcessList` checks the ownership fields and spreads the rest through; `PtyProcessListAdmission` is not on the sweep path). It now runs `isForegroundProcessEvidence` and fails closed. Verified on real Linux, not only in mocks: a container drives `bash -i` on a real pty, builds each construction, runs the real publisher and planner, and then calls the real `forceKillPosixPtyProcessGroups`. Before, all three published `shellOwnsEveryTtyProcessGroup: true`, planned SWEEP, and the planted pid was gone after the signal. After, all three skip and survive, and an idle shell is still reclaimed. Residuals are written down at the predicate and in ssh-execution-boundary.md: the capture is a snapshot (bounded by the evidence-age budget, not removed), and a process the host's own `ps` cannot enumerate stays unobservable while `killpg` still reaches it. --- docs/reference/ssh-execution-boundary.md | 15 +++ .../agent-foreground-process-batch.ts | 91 ++++++++++++----- ...h-orphan-sweep-pane-state-verdicts.test.ts | 99 +++++++++++++++++++ src/shared/foreground-process-evidence.ts | 16 ++- .../ssh-relay-pty-ownership-proof.test.ts | 52 ++++++++++ src/shared/ssh-relay-pty-ownership-proof.ts | 34 +++++-- 6 files changed, 271 insertions(+), 36 deletions(-) diff --git a/docs/reference/ssh-execution-boundary.md b/docs/reference/ssh-execution-boundary.md index cc88cf39a17..070aae61d66 100644 --- a/docs/reference/ssh-execution-boundary.md +++ b/docs/reference/ssh-execution-boundary.md @@ -70,6 +70,21 @@ A verdict needs evidence from the host that owns the process. Apply these tests Anything short of positive host evidence is `unverifiable`. Reporting it as `exited` is the error this document exists to prevent: it orphans live work and can cold-start a duplicate over the same worktree. +## Deciding a remote pane is idle + +The orphan-PTY sweep is the one flow that turns an observation into a SIGKILL, so its idleness evidence has to be measured against the same thing the signal reaches. It is not the terminal. + +`forceKillPosixPtyProcessGroups` (`src/main/pty/posix-pty-process-groups.ts`) collects every process group on the pane's tty and `killpg`s each one. The blast radius is therefore _(process groups on the tty) × (members of those groups, wherever they are)_, and the second factor is not bounded by the terminal at all. Two facts make that gap reachable: + +- **Job control can be off.** With `set +m` a background job does not get its own process group — it keeps the shell's. `ps` then shows one process group on the tty, running a build. Nothing in a tty-shaped predicate can see it. +- **A group member can leave the terminal.** `ioctl(TIOCNOTTY)` without `setsid` drops the controlling terminal but keeps the pgid, so the process reports `tpgid == -1`, never appears in `ps -t `, and is still killed by `killpg(shellPgid)`. A double-forked grandchild similarly keeps the pgid while reparenting to pid 1, so no walk by `ppid` from the PTY root can name it either. + +So `shellOwnsEveryTtyProcessGroup` (`src/main/providers/agent-foreground-process-batch.ts`) requires both measurements: every process group on the tty is the shell's own with none stopped, **and** the shell's own process group has no other member anywhere in the host's process table. The name is tty-shaped for wire-compatibility reasons only. + +Two residuals remain, and neither is removable here. The capture is a snapshot, so work started between the `ps` and the signal is invisible — bounded by `RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS` on the reading side, not eliminated. And a process the host's own `ps` cannot enumerate (another PID namespace, `hidepid=2`, a table truncated by a permission boundary) is unobservable while `killpg` still reaches it. + +The general rule this instantiates: **evidence must be measured in the unit the destructive action operates on.** Evidence in a different unit is `unverifiable` no matter how precise it looks. + ## Reading artifacts instead of process state Artifacts are stronger evidence than liveness signals, but they answer a narrower question than they appear to. diff --git a/src/main/providers/agent-foreground-process-batch.ts b/src/main/providers/agent-foreground-process-batch.ts index 12b60164446..414e57afcb3 100644 --- a/src/main/providers/agent-foreground-process-batch.ts +++ b/src/main/providers/agent-foreground-process-batch.ts @@ -29,7 +29,10 @@ export type BatchedForegroundProcessResult = { processName: string | null reason?: string /** Set only when the table was readable: every process group attached to this PTY's terminal is - * the shell's own, and none of them is stopped. Left absent when we could not observe it. */ + * the shell's own, none of them is stopped, AND that group's only member is the shell itself. + * Left absent when we could not observe it. Keeps the tty-shaped name because it is on the wire + * (`ForegroundProcessEvidence`); the value only ever got stricter, so an old client reading it + * skips more, never less. */ shellOwnsEveryTtyProcessGroup?: boolean } @@ -62,30 +65,45 @@ export type BatchedForegroundProcessOptions = { stats?: ProcessTableIndexStats } -/** Which process groups occupy each controlling terminal, and which terminals hold a stopped - * process. */ -type TtyOccupancy = { +/** The two units a forced stop can reach, indexed from one capture: which process groups occupy + * each controlling terminal (which terminals hold a stopped process), and how many rows belong to + * each process group anywhere on the host. */ +type PaneOccupancy = { processGroupsByTty: ReadonlyMap> stoppedTtys: ReadonlySet + /** Rows per `pgid`, counted over the WHOLE table with no tty filter — that is the point of it. + * A member that shares the shell's group but has no controlling terminal is reachable by + * `killpg` and invisible to every tty-shaped index. */ + rowsByProcessGroup: ReadonlyMap + /** True when some row carried no `pgid`, so the group counts are incomplete and cannot support + * an idleness claim. */ + processGroupsIncomplete: boolean } -const ttyOccupancyByCapture = new WeakMap() +const paneOccupancyByCapture = new WeakMap() -/** Index the capture by controlling terminal. +/** Index the capture by controlling terminal and by process group. * - * Keyed on `tpgid` because the snapshot carries no tty column and does not need one: a process - * group belongs to exactly one session, a session to at most one controlling terminal, so two - * rows reporting the same live `tpgid` are on the same tty. Memoized per capture, since the - * per-pane cadence poll and `pty.listProcesses` share one TTL-cached table. */ -function getTtyOccupancy(rows: readonly ProcessTableRow[]): TtyOccupancy { - const cached = ttyOccupancyByCapture.get(rows) + * The tty half is keyed on `tpgid` because the snapshot carries no tty column and does not need + * one: a process group belongs to exactly one session, a session to at most one controlling + * terminal, so two rows reporting the same live `tpgid` are on the same tty. Memoized per capture, + * since the per-pane cadence poll and `pty.listProcesses` share one TTL-cached table. */ +function getPaneOccupancy(rows: readonly ProcessTableRow[]): PaneOccupancy { + const cached = paneOccupancyByCapture.get(rows) if (cached) { return cached } const processGroupsByTty = new Map>() const stoppedTtys = new Set() + const rowsByProcessGroup = new Map() + let processGroupsIncomplete = false for (const row of rows) { - if (row.pgid === undefined || row.tpgid === undefined || row.tpgid <= 0) { + if (row.pgid === undefined) { + processGroupsIncomplete = true + continue + } + rowsByProcessGroup.set(row.pgid, (rowsByProcessGroup.get(row.pgid) ?? 0) + 1) + if (row.tpgid === undefined || row.tpgid <= 0) { continue } let groups = processGroupsByTty.get(row.tpgid) @@ -99,8 +117,13 @@ function getTtyOccupancy(rows: readonly ProcessTableRow[]): TtyOccupancy { stoppedTtys.add(row.tpgid) } } - const occupancy: TtyOccupancy = { processGroupsByTty, stoppedTtys } - ttyOccupancyByCapture.set(rows, occupancy) + const occupancy: PaneOccupancy = { + processGroupsByTty, + stoppedTtys, + rowsByProcessGroup, + processGroupsIncomplete + } + paneOccupancyByCapture.set(rows, occupancy) return occupancy } @@ -161,7 +184,7 @@ export function resolveAgentForegroundProcessesFromIndex( } } - const occupancy = getTtyOccupancy(index.rows) + const occupancy = getPaneOccupancy(index.rows) return requests.map((request) => { const root = lookupProcessTableIndex(index, (value) => value.byPid.get(request.rootPid)) if (!root) { @@ -185,20 +208,40 @@ export function resolveAgentForegroundProcessesFromIndex( reason: 'no_controlling_tty' } } - // The only host-observable "nothing is running here" signal, and it has to be read off the - // whole tty rather than off `tpgid === pgid`. A backgrounded `pnpm build &` and a Ctrl-Z'd - // editor both leave the shell owning the foreground group, byte-identical to an idle prompt; - // what separates them is a second process group attached to the pane's terminal. That is also - // exactly the blast radius of the stop this attests to — `forceKillPosixPtyProcessGroups` - // SIGKILLs every process group on the tty — so the evidence and the kill now measure the same - // thing. A reader may treat `false` as "busy" and must never treat absence as "idle". + // The only host-observable "nothing is running here" signal, and it takes TWO measurements + // because the stop it authorizes has two units. `forceKillPosixPtyProcessGroups` collects every + // process group on the pane's tty and then `killpg`s each one, so the blast radius is + // (groups on the tty) x (members of those groups, wherever they are). Neither half implies the + // other, so both are required: + // + // tty: a backgrounded `pnpm build &` and a Ctrl-Z'd editor both hand the terminal back, so + // the shell's row is byte-identical to an idle prompt. What separates them is a second + // process group attached to the pane's terminal. + // group: with job control off (`set +m`, common in non-interactive and dumb-terminal shells, + // and settable by the user at the prompt) a background job KEEPS the shell's pgid, so + // the tty shows one group and that group is running a build. Same for a child that + // drops the controlling terminal without `setsid` (`tpgid == -1`, absent from every + // tty index, still reachable by `killpg`) and for a double-forked grandchild that + // reparents to pid 1 and so never appears in the ppid walk below. + // + // Residual after both, written down because the predicate cannot see it: the capture is a + // snapshot, so work started between the `ps` and the signal is invisible — bounded, not + // removed, by RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS on the reading side; and a process the host's + // own `ps` cannot enumerate (another PID namespace, `hidepid=2`, a table truncated by a + // permission boundary) is unobservable here while `killpg` still reaches it. + // + // A reader may treat `false` as "busy" and must never treat absence as "idle". const ttyProcessGroups = occupancy.processGroupsByTty.get(root.tpgid) const shellOwnsEveryTtyProcessGroup = root.tpgid === root.pgid && ttyProcessGroups !== undefined && ttyProcessGroups.size === 1 && ttyProcessGroups.has(root.pgid) && - !occupancy.stoppedTtys.has(root.tpgid) + !occupancy.stoppedTtys.has(root.tpgid) && + !occupancy.processGroupsIncomplete && + // The root always counts itself, so exactly one row in its group means the group IS the + // shell — no separate leader check, and no set of pids retained per capture. + occupancy.rowsByProcessGroup.get(root.pgid) === 1 const allCandidates = rowsByOwner.get(root.pid) ?? [] const foregroundCandidates = allCandidates.filter((row) => row.pgid === root.tpgid) const fallbackProcess = request.fallbackProcess diff --git a/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts b/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts index ab069240681..560d18b8b62 100644 --- a/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts +++ b/src/main/ssh/ssh-orphan-sweep-pane-state-verdicts.test.ts @@ -68,6 +68,44 @@ const CAPTURES = { ' 3159 3158 3159 3158 T sleep 300', ' 3160 1 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' ] + }, + /** `set +m; sleep 300 &`. With job control OFF the job does not get its own process group — it + * keeps the SHELL's pgid. So the tty carries exactly one process group, and that group is + * running a build. Reproduced independently on a real Ubuntu host through an Orca pane. */ + setMinusMBackground: { + rootPid: 12, + table: [ + ' 1 0 1 -1 Ss /bin/bash /work/run.sh', + ' 11 1 1 -1 S python3 /work/pty-scenario.py setm_background', + ' 12 11 12 12 Ss+ bash -i', + ' 13 12 12 12 S+ sleep 300', + ' 14 11 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** A `set +m` job that drops its controlling terminal (`ioctl(TIOCNOTTY)` with no `setsid`). It + * keeps the shell's pgid, reports `tpgid == -1`, and is absent from `ps -t ` and from every + * tty-keyed index — while `killpg(shellPgid)` still reaches it. */ + nottyGroupMember: { + rootPid: 16, + table: [ + ' 1 0 1 -1 Ss /bin/bash /work/run.sh', + ' 15 1 1 -1 S python3 /work/pty-scenario.py notty_member', + ' 16 15 16 16 Ss+ bash -i', + ' 17 16 16 -1 S python3 -c import fcntl,os,time;fd=os.open("/dev/tty",os.O_RDWR);fcntl.ioctl(fd,0x5422);os.close(fd);time.sleep(300)', + ' 18 15 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] + }, + /** A `set +m` job that double-forks. pid 22 keeps the shell's pgid and tty but reparented to pid + * 1, so the ppid walk from `rootPid` never reaches it and it can never be named. */ + doubleForkedGroupMember: { + rootPid: 20, + table: [ + ' 1 0 1 -1 Ss /bin/bash /work/run.sh', + ' 19 1 1 -1 S python3 /work/pty-scenario.py double_fork', + ' 20 19 20 20 Ss+ bash -i', + ' 22 1 20 20 S+ python3 -c import os,sys,time;p=os.fork() if p: print("GRANDCHILD:%d"%p);sys.stdout.flush();os._exit(0) time.sleep(300)', + ' 23 19 1 -1 R ps -axo pid=,ppid=,pgid=,tpgid=,stat=,command=' + ] } } as const @@ -147,6 +185,15 @@ describe('what the host publishes about a pane, read by the sweep', () => { expect(shellShape(CAPTURES.background)).toBe(shellShape(CAPTURES.idle)) expect(shellShape(CAPTURES.ctrlz)).toBe(shellShape(CAPTURES.idle)) expect(shellShape(CAPTURES.foreground)).not.toBe(shellShape(CAPTURES.idle)) + + // Same premise for the `set +m` captures, minus `ppid`: their harness keeps its parent alive + // rather than reparenting the shell to init, and the ppid is the one field of the shape the + // predicate never reads. + const paneShape = (capture: { rootPid: number; table: readonly string[] }): string => + shellShape(capture).split(' ').slice(1).join(' ') + expect(paneShape(CAPTURES.setMinusMBackground)).toBe(paneShape(CAPTURES.idle)) + expect(paneShape(CAPTURES.nottyGroupMember)).toBe(paneShape(CAPTURES.idle)) + expect(paneShape(CAPTURES.doubleForkedGroupMember)).toBe(paneShape(CAPTURES.idle)) }) it('sweeps an idle shell', async () => { @@ -191,6 +238,58 @@ describe('what the host publishes about a pane, read by the sweep', () => { expect(skipReason(plan)).toBe('host does not attest an idle shell') }) + // The tty is not the unit the stop operates on. `forceKillPosixPtyProcessGroups` collects the + // groups on the tty and then `killpg`s each one, so anything sharing the shell's pgid dies with + // it — including members the tty index cannot see at all. All three captures below reproduce on + // real Linux: before the group-membership half of the predicate they published + // `shellOwnsEveryTtyProcessGroup: true`, planned a SWEEP, and the planted pid was GONE after the + // real `forceKillPosixPtyProcessGroups` call. + it('never sweeps a pane whose background job shares the shell pgid under `set +m`', async () => { + // pid 13 is `sleep 300` — stand in `pnpm build`. Its pgid IS the shell's, so the tty carries + // exactly one process group and the tty half of the predicate reads the pane as idle. + const rows = parseStrictProcessTableRows(CAPTURES.setMinusMBackground.table.join('\n')) + const tty = rows.filter((row) => row.tpgid === CAPTURES.setMinusMBackground.rootPid) + expect(new Set(tty.map((row) => row.pgid))).toEqual(new Set([12])) + expect(tty.map((row) => row.pid)).toEqual([12, 13]) + + const evidence = await publish(CAPTURES.setMinusMBackground) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.setMinusMBackground) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('never sweeps a pane whose group member dropped the controlling terminal', async () => { + // pid 17 kept the shell's pgid and called `ioctl(TIOCNOTTY)`, so it reports `tpgid == -1`, + // never appears in `ps -t `, and no tty-shaped index — not process groups, not pids — + // can observe it. `killpg(16)` reaches it regardless. + const rows = parseStrictProcessTableRows(CAPTURES.nottyGroupMember.table.join('\n')) + expect(rows.filter((row) => row.tpgid === 16).map((row) => row.pid)).toEqual([16]) + expect(rows.filter((row) => row.pgid === 16).map((row) => row.pid)).toEqual([16, 17]) + + const evidence = await publish(CAPTURES.nottyGroupMember) + expect(evidence).toMatchObject({ shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.nottyGroupMember) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + + it('never sweeps a pane whose group member double-forked away from the shell', async () => { + // pid 22 reparented to pid 1, so the ppid walk from rootPid cannot reach it and the named- + // process backstop can never fire. It still holds the shell's pgid. + const rows = parseStrictProcessTableRows(CAPTURES.doubleForkedGroupMember.table.join('\n')) + expect(rows.find((row) => row.pid === 22)).toMatchObject({ ppid: 1, pgid: 20, tpgid: 20 }) + + const evidence = await publish(CAPTURES.doubleForkedGroupMember) + expect(evidence).toMatchObject({ processName: null, shellOwnsEveryTtyProcessGroup: false }) + + const plan = await planFor(CAPTURES.doubleForkedGroupMember) + expect(plan.sweep).toEqual([]) + expect(skipReason(plan)).toBe('host does not attest an idle shell') + }) + it('refuses an observation older than the pass it would authorize', async () => { // Same idle capture that sweeps above; only its age differs. Staleness degrades to "leave it // running", never to "stop it". diff --git a/src/shared/foreground-process-evidence.ts b/src/shared/foreground-process-evidence.ts index a9d36fe557c..975fbe388ff 100644 --- a/src/shared/foreground-process-evidence.ts +++ b/src/shared/foreground-process-evidence.ts @@ -17,14 +17,20 @@ export type ForegroundProcessEvidence = | ({ verdict: 'live' processName: string | null - /** True only when the host observed every process group attached to this PTY's terminal to be - * the shell's own, with none of them stopped — i.e. nothing is running in the pane, in the - * foreground OR the background, and nothing sits suspended. + /** True only when the host observed BOTH units a forced stop can reach to hold nothing but + * the shell: every process group attached to this PTY's terminal is the shell's own with + * none of them stopped, AND the shell's own process group has no other member anywhere on + * the host. I.e. nothing is running in the pane, in the foreground OR the background, and + * nothing sits suspended. * * Deliberately not `tpgid === pgid`: a job the user backgrounded with `&` and a job the user * suspended with Ctrl-Z both hand the terminal back to the shell, so a foreground-only - * predicate reads them as idle. This one is measured against the same set of process groups - * a forced stop would SIGKILL. + * predicate reads them as idle. Deliberately not the tty alone either: with job control off + * (`set +m`) a background job keeps the shell's pgid, and a child that drops the controlling + * terminal leaves every tty index entirely — both are still inside `killpg`'s reach. + * + * The name is tty-shaped for wire reasons only. It shipped that way and old clients read it; + * the value has only ever become stricter, which makes an old client skip more, never less. * * False means something IS running, named or not. Absent from a host that predates the * field, which is neither: a reader deciding whether the pane is idle must require `true` diff --git a/src/shared/ssh-relay-pty-ownership-proof.test.ts b/src/shared/ssh-relay-pty-ownership-proof.test.ts index 1abfc0ef7f5..b17f96808f9 100644 --- a/src/shared/ssh-relay-pty-ownership-proof.test.ts +++ b/src/shared/ssh-relay-pty-ownership-proof.test.ts @@ -146,6 +146,58 @@ describe('planRelayPtySweep', () => { expect(reasonFor(plan, 'pty-1')).toBe('host attests another client created it') }) + // The age gate is the one comparison in the file that a malformed field defaults toward the + // kill: the sum goes `NaN`, and `NaN > budget` is FALSE, so the entry PASSES the freshness gate + // and proceeds toward the stop. Nothing validated this record on the sweep path — + // `mapSshPtyProcessList` checks the ownership fields and spreads the rest through. + it.each([ + ['missing', undefined], + ['a string', '0' as unknown], + ['NaN', Number.NaN], + ['Infinity', Number.POSITIVE_INFINITY], + ['negative', -1], + ['fractional', 1.5] + ])('never sweeps when the host stamped capturedAgeMs %s', (_label, capturedAgeMs) => { + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...idleShell(), + capturedAgeMs + } as unknown as ForegroundProcessEvidence + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host foreground observation is malformed') + }) + + it('never sweeps on an evidence record whose other host stamps are malformed', () => { + const plan = planRelayPtySweep( + [ + orphan({ + foregroundProcessEvidence: { + ...idleShell(), + authorityGeneration: '' + } as ForegroundProcessEvidence + }) + ], + context() + ) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('host foreground observation is malformed') + }) + + it('never sweeps when this client cannot compute an age budget', () => { + const plan = planRelayPtySweep([orphan()], context({ evidenceAgeSinceListingMs: Number.NaN })) + + expect(plan.sweep).toEqual([]) + expect(reasonFor(plan, 'pty-1')).toBe('sweep has no usable evidence-age budget') + }) + it('never sweeps a PTY younger than the floor', () => { const plan = planRelayPtySweep( [orphan({ hostAgeMs: RELAY_PTY_SWEEP_MIN_AGE_MS - 1 })], diff --git a/src/shared/ssh-relay-pty-ownership-proof.ts b/src/shared/ssh-relay-pty-ownership-proof.ts index ed87be13610..866adbfb3e8 100644 --- a/src/shared/ssh-relay-pty-ownership-proof.ts +++ b/src/shared/ssh-relay-pty-ownership-proof.ts @@ -1,4 +1,7 @@ -import type { ForegroundProcessEvidence } from './foreground-process-evidence' +import { + isForegroundProcessEvidence, + type ForegroundProcessEvidence +} from './foreground-process-evidence' /** Which relay PTYs a client may prove it orphaned, and therefore may stop (#9819). * @@ -104,9 +107,10 @@ export const RELAY_PTY_SWEEP_MAX_PER_PASS = 8 export const RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS = 5_000 /** The host's own answer to "is anything running in this pane?". Only a positive "no" clears the - * sweep; every other shape — an older host, an unreadable process table, an observation too old to - * describe now, a named foreground process, any other process group on the pane's terminal — is a - * reason to leave the process alone. */ + * sweep; every other shape — an older host, a malformed record, an unreadable process table, an + * observation too old to describe now, a named foreground process, any other process group on the + * pane's terminal, any other member of the shell's own process group — is a reason to leave the + * process alone. */ function foregroundSkipReason( evidence: ForegroundProcessEvidence | undefined, context: RelayPtySweepContext @@ -116,6 +120,20 @@ function foregroundSkipReason( // observation is not the observation of absence. return 'host published no foreground-process observation' } + // The record reaches this decision straight off the wire — `mapSshPtyProcessList` validates the + // ownership fields and spreads the rest through, and `PtyProcessListAdmission` is not on the + // sweep path. Shape-check it here, because the age gate below is the one comparison in this file + // that a malformed field defaults toward the kill: a non-numeric `capturedAgeMs` makes the sum + // `NaN`, and `NaN > budget` is FALSE, so the entry would pass the freshness gate. + if (!isForegroundProcessEvidence(evidence)) { + return 'host foreground observation is malformed' + } + if ( + !Number.isFinite(context.evidenceAgeSinceListingMs) || + !Number.isFinite(context.maximumEvidenceAgeMs) + ) { + return 'sweep has no usable evidence-age budget' + } // Before anything is read out of it: an observation is only a claim about the instant it was // taken. Age is checked on both verdicts because a stale `unverifiable` is no better. if (evidence.capturedAgeMs + context.evidenceAgeSinceListingMs > context.maximumEvidenceAgeMs) { @@ -130,9 +148,11 @@ function foregroundSkipReason( return 'host observes a named foreground process' } if (evidence.shellOwnsEveryTtyProcessGroup !== true) { - // Something other than the shell's own process group is attached to the pane's terminal — a - // foreground command, a job backgrounded with `&`, a Ctrl-Z'd editor — or this host predates - // the field. The stop would SIGKILL that group, so none of those is a pane to reclaim. + // The host saw work inside the stop's blast radius: another process group on the pane's + // terminal (a foreground command, a job backgrounded with `&`, a Ctrl-Z'd editor), or another + // member of the shell's OWN process group (a `set +m` background job, a child that dropped the + // controlling terminal) — or this host predates the field. `killpg` reaches all of it, so none + // of those is a pane to reclaim. return 'host does not attest an idle shell' } return null From b1186c6beb58adbf796d1352040c0d6877045926 Mon Sep 17 00:00:00 2001 From: Jinjing <6427696+AmethystLiang@users.noreply.github.com> Date: Thu, 3 Sep 2026 14:45:58 -0700 Subject: [PATCH 198/398] Fix scope of workspace-creation-project tour target (#18502) * fix: scope workspace-creation-project tour target to project picker only The tour target was previously applied to a container that included both the project picker and the run target picker below it. Restructure the layout to scope the target to only the project-related section, and add a test to verify the tour target does not span into the run target picker. * fix: scope workspace-creation-project tour target to project picker only Move the tour target attribute from the outer project section to an inner wrapper around just the combobox and its messages, excluding the header label and "Add project" button. Update tests to verify the narrower scope. --- .../NewWorkspaceComposerCard.test.tsx | 153 ++++++++---------- .../NewWorkspaceComposerProjectSection.tsx | 112 ++++++------- 2 files changed, 129 insertions(+), 136 deletions(-) diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.test.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.test.tsx index 8206354f916..1456f7e30ad 100644 --- a/src/renderer/src/components/NewWorkspaceComposerCard.test.tsx +++ b/src/renderer/src/components/NewWorkspaceComposerCard.test.tsx @@ -104,7 +104,7 @@ vi.mock('@/components/new-workspace/ProjectCombobox', () => ({ value: string | null onValueChange: (value: string) => void }) => ( -
    +
    {options.map((option) => ( + + + {translate('auto.components.NewWorkspaceComposerCard.d6b0a96f32', 'Add project')} + + + ) : null} +
    +
    + + {projectError ? ( +

    + {projectError} +

    + ) : projectOptions.length === 0 ? ( +

    + {emptyProjectMessage ?? + translate( + 'auto.components.NewWorkspaceComposerCard.addProjectBeforeWorkspace', + 'Add a project before creating a workspace.' )} - > - - - - - {translate('auto.components.NewWorkspaceComposerCard.d6b0a96f32', 'Add project')} - - - ) : null} +

    + ) : null} +
    - - {projectError ? ( -

    - {projectError} -

    - ) : projectOptions.length === 0 ? ( -

    - {emptyProjectMessage ?? - translate( - 'auto.components.NewWorkspaceComposerCard.addProjectBeforeWorkspace', - 'Add a project before creating a workspace.' - )} -

    - ) : null} {shouldShowRunTargetPicker ? (
      - {CAPABILITIES.map((line) => ( -
    • + {CAPABILITIES.map(({ key, fallback }) => ( +
    • - {line} + {translate(key, fallback)}
    • ))}
    diff --git a/src/renderer/src/components/onboarding/OnboardingFlow.tsx b/src/renderer/src/components/onboarding/OnboardingFlow.tsx index 9e82e42547c..c57e2556014 100644 --- a/src/renderer/src/components/onboarding/OnboardingFlow.tsx +++ b/src/renderer/src/components/onboarding/OnboardingFlow.tsx @@ -90,11 +90,31 @@ const stepCopy = { } as const const stepTooltipLabels = { - agent: 'Default Agent', - theme: 'Appearance', - windows_terminal: 'Windows Terminal', - notifications: 'Notifications', - integrations: 'Integrations' + agent: { + get value() { + return translate('components.onboarding.flow.stepTooltip.agent', 'Default Agent') + } + }, + theme: { + get value() { + return translate('components.onboarding.flow.stepTooltip.theme', 'Appearance') + } + }, + windows_terminal: { + get value() { + return translate('components.onboarding.flow.stepTooltip.windowsTerminal', 'Windows Terminal') + } + }, + notifications: { + get value() { + return translate('components.onboarding.flow.stepTooltip.notifications', 'Notifications') + } + }, + integrations: { + get value() { + return translate('components.onboarding.flow.stepTooltip.integrations', 'Integrations') + } + } } as const type OnboardingFlowProps = { @@ -113,7 +133,10 @@ export default function OnboardingFlow({ const shouldShowSkipToProjectSetup = currentStep.id !== 'notifications' const shouldShowFooterBusy = Boolean(busyLabel) const footerPrimaryLabel = - busyLabel ?? (currentStep.id === 'notifications' ? 'Add your first project' : 'Continue') + busyLabel ?? + (currentStep.id === 'notifications' + ? translate('components.onboarding.flow.actions.addFirstProject', 'Add your first project') + : translate('components.onboarding.flow.actions.continue', 'Continue')) const [skipConfirmOpen, setSkipConfirmOpen] = useState(false) const skipConfirmAdvancedViaRef = useRef<'button' | 'keyboard'>('button') const { next: flowNext, dismissOnboarding: flowDismissOnboarding } = flow @@ -218,6 +241,7 @@ export default function OnboardingFlow({ {flow.progressSteps.map(({ step, index: realStepIndex }, progressIdx) => { const isActive = realStepIndex === stepIndex const isDone = realStepIndex < stepIndex + const stepTooltipLabel = stepTooltipLabels[step.id].value return ( @@ -236,14 +260,14 @@ export default function OnboardingFlow({ aria-label={translate( 'auto.components.onboarding.OnboardingFlow.adaa0aa627', 'Go to onboarding step {{value0}}: {{value1}}', - { value0: progressIdx + 1, value1: stepTooltipLabels[step.id] } + { value0: progressIdx + 1, value1: stepTooltipLabel } )} aria-current={isActive ? 'step' : undefined} onClick={() => flow.jumpToStep(realStepIndex)} /> - {stepTooltipLabels[step.id]} + {stepTooltipLabel} ) diff --git a/src/renderer/src/components/onboarding/OnboardingSkipConfirmationDialog.tsx b/src/renderer/src/components/onboarding/OnboardingSkipConfirmationDialog.tsx index 1a721a6d774..6851ea83399 100644 --- a/src/renderer/src/components/onboarding/OnboardingSkipConfirmationDialog.tsx +++ b/src/renderer/src/components/onboarding/OnboardingSkipConfirmationDialog.tsx @@ -22,8 +22,12 @@ export const ONBOARDING_SKIP_CONFIRMATION_COPY = { "It won't take long!" ) }, - skipLabel: 'Skip', - keepGoingLabel: 'No, keep going' + get skipLabel() { + return translate('components.onboarding.skipConfirmation.skip', 'Skip') + }, + get keepGoingLabel() { + return translate('components.onboarding.skipConfirmation.keepGoing', 'No, keep going') + } } as const export function OnboardingSkipConfirmationDialog(props: { diff --git a/src/renderer/src/components/onboarding/ThemeStep.tsx b/src/renderer/src/components/onboarding/ThemeStep.tsx index bb797a12a6e..0929aba0d59 100644 --- a/src/renderer/src/components/onboarding/ThemeStep.tsx +++ b/src/renderer/src/components/onboarding/ThemeStep.tsx @@ -195,19 +195,19 @@ export function ThemeStep({ theme, onThemeChange, settings, updateSettings }: Th { id: 'system', label: translate('auto.components.onboarding.ThemeStep.827ea7b4a2', 'System'), - hint: 'Match OS', + hint: translate('components.onboarding.theme.hints.system', 'Match OS'), icon: Monitor }, { id: 'dark', label: translate('auto.components.onboarding.ThemeStep.fa7b673ea9', 'Dark'), - hint: 'Easy on the eyes', + hint: translate('components.onboarding.theme.hints.dark', 'Easy on the eyes'), icon: Moon }, { id: 'light', label: translate('auto.components.onboarding.ThemeStep.ad192706e6', 'Light'), - hint: 'Bright & crisp', + hint: translate('components.onboarding.theme.hints.light', 'Bright & crisp'), icon: Sun } ] diff --git a/src/renderer/src/components/onboarding/use-onboarding-flow-actions.ts b/src/renderer/src/components/onboarding/use-onboarding-flow-actions.ts index 9216de04e61..9fac360b162 100644 --- a/src/renderer/src/components/onboarding/use-onboarding-flow-actions.ts +++ b/src/renderer/src/components/onboarding/use-onboarding-flow-actions.ts @@ -117,7 +117,12 @@ export function useOnboardingFlowActions({ if (result.ok) { trackCurrentStepCompleted(advancedVia) if (currentStep.id === 'notifications') { - setBusyLabel('Opening Add Project...') + setBusyLabel( + translate( + 'components.onboarding.flow.actions.openingAddProject', + 'Opening Add Project...' + ) + ) const closed = await closeWith('completed', ONBOARDING_FINAL_STEP, 'add_project_modal') if (closed) { openModal('add-repo') @@ -190,7 +195,9 @@ export function useOnboardingFlowActions({ const stepId = currentStep.id const stepNumber = currentStep.stepNumber const valueKind = currentStep.valueKind - setBusyLabel('Opening Add Project...') + setBusyLabel( + translate('components.onboarding.flow.actions.openingAddProject', 'Opening Add Project...') + ) try { const closed = await closeWith('completed', ONBOARDING_FINAL_STEP, 'add_project_modal') if (!closed) { diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 712df329873..57db52974bd 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16820,6 +16820,41 @@ } }, "components": { + "onboarding": { + "flow": { + "stepTooltip": { + "agent": "Default Agent", + "theme": "Appearance", + "windowsTerminal": "Windows Terminal", + "notifications": "Notifications", + "integrations": "Integrations" + }, + "actions": { + "addFirstProject": "Add your first project", + "continue": "Continue", + "openingAddProject": "Opening Add Project..." + } + }, + "skipConfirmation": { + "skip": "Skip", + "keepGoing": "No, keep going" + }, + "theme": { + "hints": { + "system": "Match OS", + "dark": "Easy on the eyes", + "light": "Bright & crisp" + } + }, + "integrations": { + "capabilities": { + "startWorkspaceFromIssue": "Start a workspace from any GitHub issue or pull request, prefilled with its title and context", + "browseIssues": "Browse GitHub issues and pull requests in the Tasks view without leaving Orca", + "reviewStatus": "See issue state, review status, and CI checks on every worktree", + "managePullRequests": "Read, comment on, and merge pull requests without leaving Orca" + } + } + }, "jiraUserPicker": { "select": "Select {{value0}}", "search": "Search users", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index 47d9e48fd26..d05754d176e 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -14616,6 +14616,41 @@ } }, "components": { + "onboarding": { + "flow": { + "stepTooltip": { + "agent": "기본 Agent", + "theme": "테마", + "windowsTerminal": "Windows 터미널", + "notifications": "알림", + "integrations": "연동" + }, + "actions": { + "addFirstProject": "첫 프로젝트 추가", + "continue": "계속", + "openingAddProject": "프로젝트 추가 화면을 여는 중…" + } + }, + "skipConfirmation": { + "skip": "건너뛰기", + "keepGoing": "아니요, 계속하기" + }, + "theme": { + "hints": { + "system": "OS 설정에 맞춤", + "dark": "눈이 편안한 화면", + "light": "밝고 선명한 화면" + } + }, + "integrations": { + "capabilities": { + "startWorkspaceFromIssue": "GitHub 이슈나 PR에서 제목과 컨텍스트가 미리 채워진 워크스페이스를 시작합니다", + "browseIssues": "Orca를 떠나지 않고 작업 화면에서 GitHub 이슈와 PR을 탐색합니다", + "reviewStatus": "모든 워크트리에서 이슈 상태, 리뷰 상태 및 CI 체크를 확인합니다", + "managePullRequests": "Orca를 떠나지 않고 PR을 읽고, 댓글을 달고, 병합합니다" + } + } + }, "native-chat": { "composer": { "imageUnsupported": "이 에이전트는 이미지 붙여넣기를 지원하지 않습니다.", From 98e77ef1a70e2f619ca7f3b216db13cd20f02461 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Thu, 3 Sep 2026 15:19:26 -0700 Subject: [PATCH 202/398] feat(mobile): structured native Codex chat (#18074) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(mobile): finalize structured native Codex chat * fix(mobile): close structured chat lifecycle gaps * wip(mobile): fence stale structured inventory and bound operation-id retention Fence local structured-session inventory and subscription responses with a sync generation so a toggle-off clear, reconnect restore, or retry cannot apply a mirror from a superseded instance. Bound mobile ambiguous operation-ID retention at 128 with unmount cleanup. Staged on the reconcile branch only: the sync module is now 312 lines and needs a real split before this can reach the PR head. * fix(ci): split the structured session-tabs sync and give static analysis mobile types The local structured session-tabs sync module outgrew the 300-line cap once it took on generation fencing, so split it along its real seams instead of raising the cap: the generation/cursor fence, snapshot projection, snapshot apply, inventory refresh, and the subscription loop. The original path stays as a barrel so no importer moves. Repoint the host-session-mirror settle census at the apply module, which owns two receipts now — the snapshot it mirrors in, and the toggle-off teardown that retracts what it published. The teardown receipt is named rather than anonymous so the pin says which direction it settles. The changed-code quality gate lints mobile files and resolves their types from mobile/node_modules, but mobile is a separate pnpm project that the root install never populates, so every mobile type degraded to an `error` type and the gate reported phantom findings. Install mobile dependencies in static analysis when the diff touches mobile, gated on a new classifier output. * fix(mobile): let a slow capability handshake still reach connected The mobile capability update is an advisory whose result is discarded, yet an unanswered one was fatal while an explicit rejection was tolerated. A 5s timeout on the direct client force-closed the socket, and on the relay path it failed `confirmResume` before `connected` was ever published, so a consistently slow link redialled forever. Both paths now share one helper that settles every ambiguous outcome (timeout, mid-flight drop) like a rejection and rejects only when the frame never reached the wire — the one case nothing else recovers from, since the socket's own desync force-close is gated on already being connected. The generation guard still keeps a replaced session from connecting. Retained structured-session operation ids were capped at 128 with oldest-first eviction, but every retained id belongs to a send whose outcome is unknown, so eviction turned a user's retry into a second message on the host. Bound the map by expiry against the id's own embedded timestamp instead, mirroring the host's operation ledger, so no id is released while the host would still honour it. Also give the mobile CI install the root install's lockfile drift guard (mobile's lockfile carries patchedDependencies a silent rewrite would drop), gate mobile_dependencies on should_run, and key the pnpm store cache on both lockfiles. * refactor(mobile): extract the relay pending-request registry The merge composed two independently-sized changes — this branch's capability handshake settle and main's dial-stage tracking — pushing the relay session file to 304 lines against a 300 cap. Neither side broke it alone. Move the in-flight request registry (id generation, tracking, settlement, and reject-all with its delivery-ambiguity marking) into RelayPendingRequests, matching the existing collaborator pattern alongside RelayDialStageTracker and RpcSessionLivenessWatchdog. No behavior change. --------- Co-authored-by: Merge Sim --- .../install-node-dependencies/action.yml | 9 + .github/workflows/pr.yml | 20 + config/scripts/pr-code-change-scope.mjs | 9 + config/scripts/pr-code-change-scope.test.mjs | 32 + .../src/session/MobileNativeChatQuestion.tsx | 81 +- .../session/MobileSessionActiveContent.tsx | 4 +- mobile/src/session/MobileSessionHeader.tsx | 1 + mobile/src/session/MobileSessionSheets.tsx | 10 + mobile/src/session/mobile-file-tap-open.ts | 4 +- .../mobile-native-chat-controller-contract.ts | 7 +- .../mobile-native-chat-eligibility.test.ts | 24 + .../session/mobile-native-chat-eligibility.ts | 24 +- .../mobile-native-chat-image-scope-state.ts | 18 + .../mobile-native-chat-question.test.ts | 10 + .../session/mobile-native-chat-question.ts | 14 + .../mobile-session-route-parity.test.ts | 22 +- .../src/session/mobile-session-route-types.ts | 10 +- .../mobile-structured-agent-prompts.ts | 251 ++++++ ...le-structured-agent-session-launch.test.ts | 142 +++ .../mobile-structured-agent-session-launch.ts | 158 ++++ .../mobile-structured-agent-session-rpc.ts | 152 ++++ ...ctured-session-operation-retention.test.ts | 58 ++ .../session/mobile-terminal-records.test.ts | 14 + mobile/src/session/mobile-terminal-records.ts | 10 + .../session/mobile-terminal-tab-agent.test.ts | 13 + .../src/session/mobile-terminal-tab-agent.ts | 3 + .../session/opened-mobile-session-tab.test.ts | 15 + .../src/session/opened-mobile-session-tab.ts | 19 +- .../use-mobile-file-tap-handlers.test.ts | 27 + .../session/use-mobile-file-tap-handlers.ts | 10 +- ...se-mobile-native-chat-active-resolution.ts | 83 ++ .../use-mobile-native-chat-controller.test.ts | 163 +++- .../use-mobile-native-chat-controller.ts | 206 +++-- ...se-mobile-native-chat-image-attachments.ts | 167 ++-- .../use-mobile-native-chat-image-upload.ts | 126 +++ ...e-native-chat-session-option-controller.ts | 100 +++ .../session/use-mobile-session-attachments.ts | 4 +- .../use-mobile-session-file-actions.ts | 6 +- ...-mobile-session-image-attachments.test.tsx | 123 +++ .../use-mobile-session-image-attachments.ts | 13 +- ...se-mobile-session-native-chat-dictation.ts | 7 + .../use-mobile-session-screen-state.ts | 24 +- .../use-mobile-session-tab-action-targets.ts | 85 ++ .../use-mobile-session-tab-switching.ts | 3 + ...le-session-terminal-create-actions.test.ts | 231 +++++ ...-mobile-session-terminal-create-actions.ts | 31 + ...se-mobile-session-terminal-send-actions.ts | 28 +- .../use-mobile-structured-agent-options.ts | 161 ++++ ...e-mobile-structured-agent-session.test.tsx | 849 ++++++++++++++++++ .../use-mobile-structured-agent-session.ts | 315 +++++++ .../use-mobile-structured-agent-state.ts | 198 ++++ ...bile-structured-native-chat-send-bridge.ts | 100 +++ .../cellular-connecting-label-stall.test.ts | 4 + mobile/src/transport/direct-connection-log.ts | 4 + mobile/src/transport/direct-rpc-client.ts | 22 +- .../foreground-stale-dial-restart.test.ts | 4 + .../mobile-relay-rpc-session-liveness.test.ts | 10 + .../mobile-relay-rpc-session.test.ts | 59 +- .../src/transport/mobile-relay-rpc-session.ts | 57 +- ...ile-runtime-capability-negotiation.test.ts | 69 ++ .../mobile-runtime-capability-negotiation.ts | 57 ++ .../mobile-runtime-client-capabilities.ts | 44 + .../src/transport/relay-pending-requests.ts | 53 ++ .../transport/rpc-client-capabilities.test.ts | 161 ++++ .../rpc-client-connect-wait-replay.test.ts | 4 + .../rpc-client-delivery-ambiguity.test.ts | 4 + .../rpc-client-request-deadline.test.ts | 4 + .../transport/rpc-client-request-tracker.ts | 21 +- .../rpc-client-runtime-events.test.ts | 4 + ...ient-synthesized-close-diagnostics.test.ts | 4 + .../rpc-client-terminal-reconnect.test.ts | 4 + .../rpc-client-unauthorized-close.test.ts | 4 + mobile/src/transport/rpc-client.test.ts | 41 +- .../rpc-session-liveness-integration.test.ts | 4 + src/main/runtime/mobile-rpc-allowlist.test.ts | 31 +- ...time-close-structured-agent-session-tab.ts | 10 +- src/main/runtime/orca-runtime-state-fields.ts | 5 + ...me-structured-native-chat-settings.test.ts | 21 + ...runtime-structured-session-restore.test.ts | 12 + src/main/runtime/rpc/core.ts | 2 + .../runtime/rpc/dispatcher-stream-options.ts | 1 + src/main/runtime/rpc/dispatcher.ts | 4 +- .../runtime/rpc/methods/client-ui.test.ts | 19 + src/main/runtime/rpc/methods/index.ts | 2 + .../runtime-client-capabilities.test.ts | 74 ++ .../methods/runtime-client-capabilities.ts | 24 + ...ion-tab-agent-capability-mutations.test.ts | 42 + ...ession-tab-agent-status-projection.test.ts | 24 + .../session-tab-agent-status-projection.ts | 14 +- .../rpc/methods/session-tab-close-methods.ts | 11 +- .../methods/session-tab-mutation-methods.ts | 21 +- .../rpc/methods/session-tabs-inventory.ts | 27 +- .../runtime/rpc/methods/session-tabs.test.ts | 47 +- src/main/runtime/rpc/methods/session-tabs.ts | 26 +- .../methods/structured-agent-session-gate.ts | 7 +- .../structured-agent-session-policy.ts | 47 + .../methods/structured-agent-session.test.ts | 76 +- .../rpc/methods/structured-agent-session.ts | 25 +- .../methods/structured-session-tab-restore.ts | 12 +- .../runtime/rpc/rpc-streaming-dispatcher.ts | 2 + src/main/runtime/runtime-client-settings.ts | 2 + .../runtime-rpc-mobile-method-allowlist.ts | 17 + .../runtime-rpc-websocket-dispatch.ts | 6 + src/main/runtime/runtime-store-contract.ts | 2 + .../app-shell/use-app-startup-hydration.ts | 8 +- src/renderer/src/app-startup-routing.test.ts | 10 + ...ctured-agent-session-message-projection.ts | 37 +- .../use-structured-agent-session-hold.ts | 8 +- .../host-session-mirror-settle-census.test.ts | 14 +- ...local-structured-session-tab-retirement.ts | 64 ++ ...local-structured-session-tabs-sync.test.ts | 90 ++ .../local-structured-session-tabs-sync.ts | 301 +------ .../inventory-generation-fence.ts | 57 ++ .../inventory-refresh.ts | 37 + .../snapshot-apply.ts | 118 +++ .../snapshot-projection.ts | 37 + .../subscription.ts | 122 +++ src/shared/structured-agent-session-holder.ts | 6 + ...ctured-agent-session-message-projection.ts | 29 + .../structured-agent-session-reducer.ts | 3 +- ...ss-version-agent-session-wire.unit.test.ts | 44 + .../versioned-agent-session-wire.ts | 1 + 122 files changed, 5667 insertions(+), 764 deletions(-) create mode 100644 mobile/src/session/mobile-native-chat-image-scope-state.ts create mode 100644 mobile/src/session/mobile-structured-agent-prompts.ts create mode 100644 mobile/src/session/mobile-structured-agent-session-launch.test.ts create mode 100644 mobile/src/session/mobile-structured-agent-session-launch.ts create mode 100644 mobile/src/session/mobile-structured-agent-session-rpc.ts create mode 100644 mobile/src/session/mobile-structured-session-operation-retention.test.ts create mode 100644 mobile/src/session/use-mobile-native-chat-active-resolution.ts create mode 100644 mobile/src/session/use-mobile-native-chat-image-upload.ts create mode 100644 mobile/src/session/use-mobile-native-chat-session-option-controller.ts create mode 100644 mobile/src/session/use-mobile-session-image-attachments.test.tsx create mode 100644 mobile/src/session/use-mobile-session-tab-action-targets.ts create mode 100644 mobile/src/session/use-mobile-session-terminal-create-actions.test.ts create mode 100644 mobile/src/session/use-mobile-structured-agent-options.ts create mode 100644 mobile/src/session/use-mobile-structured-agent-session.test.tsx create mode 100644 mobile/src/session/use-mobile-structured-agent-session.ts create mode 100644 mobile/src/session/use-mobile-structured-agent-state.ts create mode 100644 mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts create mode 100644 mobile/src/transport/mobile-runtime-capability-negotiation.test.ts create mode 100644 mobile/src/transport/mobile-runtime-capability-negotiation.ts create mode 100644 mobile/src/transport/mobile-runtime-client-capabilities.ts create mode 100644 mobile/src/transport/relay-pending-requests.ts create mode 100644 mobile/src/transport/rpc-client-capabilities.test.ts create mode 100644 src/main/runtime/orca-runtime-structured-native-chat-settings.test.ts create mode 100644 src/main/runtime/rpc/methods/runtime-client-capabilities.test.ts create mode 100644 src/main/runtime/rpc/methods/runtime-client-capabilities.ts create mode 100644 src/main/runtime/rpc/methods/structured-agent-session-policy.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tab-retirement.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-generation-fence.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-projection.ts create mode 100644 src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts create mode 100644 src/shared/structured-agent-session-holder.ts create mode 100644 src/shared/structured-agent-session-message-projection.ts diff --git a/.github/actions/install-node-dependencies/action.yml b/.github/actions/install-node-dependencies/action.yml index 46edfc54111..e36ec4c65d8 100644 --- a/.github/actions/install-node-dependencies/action.yml +++ b/.github/actions/install-node-dependencies/action.yml @@ -39,6 +39,9 @@ runs: with: install: false + # Why both lockfiles: setup-node keys the pnpm store on the root lockfile alone, so + # jobs that also install mobile restored a store with none of the React Native tree + # in it and re-downloaded the lot on every run. - name: Setup Node.js id: default-node if: inputs.node-version == '' @@ -46,6 +49,9 @@ runs: with: node-version-file: package.json cache: pnpm + cache-dependency-path: | + pnpm-lock.yaml + mobile/pnpm-lock.yaml - name: Setup requested Node.js id: requested-node @@ -54,6 +60,9 @@ runs: with: node-version: ${{ inputs.node-version }} cache: pnpm + cache-dependency-path: | + pnpm-lock.yaml + mobile/pnpm-lock.yaml - name: Validate native runtime shell: bash diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index bde0b5e05d8..d21b1a23784 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -28,6 +28,7 @@ jobs: outputs: should_run: ${{ steps.filter.outputs.should_run }} native_cache_changed: ${{ steps.filter.outputs.native_cache_changed }} + mobile_dependencies: ${{ steps.filter.outputs.mobile_dependencies }} static_analysis: ${{ steps.filter.outputs.static_analysis }} typecheck: ${{ steps.filter.outputs.typecheck }} git_compatibility: ${{ steps.filter.outputs.git_compatibility }} @@ -95,6 +96,25 @@ jobs: - name: Enforce type-aware code-quality baseline run: pnpm run audit:code-quality:type-aware + # Why: the changed-code gate lints mobile files too, and its type-aware pass + # resolves types from mobile/node_modules. Mobile is a separate pnpm project, + # so the root install above leaves it empty and every mobile type degrades to + # an `error` type — reported as phantom findings against the changed lines. + # Why no --ignore-scripts, unlike the root install: mobile's postinstall generates + # the gitignored terminal/mermaid webview engine modules that tracked source imports, + # and skipping it degrades those very types the step exists to resolve. The drift + # guard mirrors the root install so a stale mobile lockfile fails by name — mobile's + # lockfile carries patchedDependencies that a silent rewrite would drop. + - name: Install mobile dependencies + if: needs.code_paths.outputs.mobile_dependencies == 'true' + working-directory: mobile + run: | + pnpm install --frozen-lockfile + if [ "$(git -C "$GITHUB_WORKSPACE" rev-parse --is-inside-work-tree 2>/dev/null)" = true ]; then + git -C "$GITHUB_WORKSPACE" diff --exit-code -- \ + mobile/package.json mobile/pnpm-lock.yaml mobile/pnpm-workspace.yaml + fi + - name: Enforce changed-code quality run: pnpm run check:code-quality:changed -- "${{ github.event.pull_request.base.sha }}" diff --git a/config/scripts/pr-code-change-scope.mjs b/config/scripts/pr-code-change-scope.mjs index 8bc10fc5b72..7111531c35c 100644 --- a/config/scripts/pr-code-change-scope.mjs +++ b/config/scripts/pr-code-change-scope.mjs @@ -263,6 +263,14 @@ export function shouldRunPrChecks(changedFiles) { return changedFiles.some((file) => !isDocsOnlyPath(file) && !isDesktopIrrelevantPath(file)) } +export function needsMobileDependencies(changedFiles) { + // Why: static analysis lints CHANGED files, mobile ones included, and its + // type-aware pass resolves types from mobile/node_modules. Mobile is a + // separate pnpm project, so without this the root-only install leaves every + // mobile type an `error` type and the gate reports phantom findings. + return changedFiles.length === 0 || changedFiles.some((file) => file.startsWith('mobile/')) +} + export function classifyPrJobs(changedFiles) { const emptyDiff = changedFiles.length === 0 const shouldRun = shouldRunPrChecks(changedFiles) @@ -276,6 +284,7 @@ export function classifyPrJobs(changedFiles) { return { should_run: shouldRun, native_cache_changed: shouldRun && (emptyDiff || changedFiles.some(isNativeCacheInputPath)), + mobile_dependencies: shouldRun && needsMobileDependencies(changedFiles), ...jobs } } diff --git a/config/scripts/pr-code-change-scope.test.mjs b/config/scripts/pr-code-change-scope.test.mjs index 1fe296af265..4642372135c 100644 --- a/config/scripts/pr-code-change-scope.test.mjs +++ b/config/scripts/pr-code-change-scope.test.mjs @@ -316,6 +316,24 @@ describe('per-job path classification', () => { } }) + // Why: static analysis lints changed mobile files with a type-aware pass, and + // mobile is a separate pnpm project. Without its node_modules every mobile type + // resolves to an `error` type and the changed-code gate fails on phantom + // findings, which is exactly how a react-test-renderer union broke a PR. + it('installs mobile dependencies exactly when mobile files change', () => { + expect(classifyPrJobs([]).mobile_dependencies).toBe(true) + expect(classifyPrJobs(['README.md']).mobile_dependencies).toBe(false) + expect(classifyPrJobs(['src/main/index.ts']).mobile_dependencies).toBe(false) + expect( + classifyPrJobs(['src/main/index.ts', 'mobile/src/session/a.test.ts']).mobile_dependencies + ).toBe(true) + // Why false: a mobile-only diff skips every desktop job, so the install step's own + // job never runs and claiming the install is needed contradicts should_run. + expect(classifyPrJobs(['mobile/package.json']).mobile_dependencies).toBe(false) + expect(classifyPrJobs(['mobile/package.json']).should_run).toBe(false) + expect(classifyPrJobs(['README.md', 'mobile/src/a.ts']).mobile_dependencies).toBe(false) + }) + it('keeps unit-test-only diffs out of packaging', () => { expectClassification(['src/main/git/git-status.test.ts'], { git_compatibility: true @@ -354,6 +372,20 @@ describe('PR Checks skip wiring', () => { } }) + it('gives static analysis the mobile types its type-aware pass resolves', () => { + expect(prWorkflow.jobs.code_paths.outputs.mobile_dependencies).toBe( + '${{ steps.filter.outputs.mobile_dependencies }}' + ) + const steps = prWorkflow.jobs.static_analysis.steps + const install = steps.findIndex((step) => step.name === 'Install mobile dependencies') + const gate = steps.findIndex((step) => step.name === 'Enforce changed-code quality') + expect(install).toBeGreaterThan(-1) + expect(install).toBeLessThan(gate) + expect(steps[install].if).toBe("needs.code_paths.outputs.mobile_dependencies == 'true'") + expect(steps[install]['working-directory']).toBe('mobile') + expect(steps[install].run).toContain('--frozen-lockfile') + }) + it('keeps the cheap root-directory guard on docs-only PRs', () => { expect(prWorkflow.jobs.root_directory_guard.if).toBeUndefined() expect(prWorkflow.jobs.root_directory_guard.needs).toBeUndefined() diff --git a/mobile/src/session/MobileNativeChatQuestion.tsx b/mobile/src/session/MobileNativeChatQuestion.tsx index f470214dbed..f4a34494328 100644 --- a/mobile/src/session/MobileNativeChatQuestion.tsx +++ b/mobile/src/session/MobileNativeChatQuestion.tsx @@ -2,7 +2,11 @@ import { useMemo, useRef, useState } from 'react' import { Pressable, StyleSheet, Text, TextInput, View } from 'react-native' import { ArrowUp, Check, CircleHelp } from 'lucide-react-native' import { colors, radii, spacing, typography } from '../theme/mobile-theme' -import { formatQuestionAnswer, type MobileChatQuestion } from './mobile-native-chat-question' +import { + formatQuestionAnswer, + formatQuestionFreeTextAnswer, + type MobileChatQuestion +} from './mobile-native-chat-question' type Props = { question: MobileChatQuestion @@ -18,6 +22,7 @@ export function MobileNativeChatQuestion({ question, onAnswer }: Props): React.J const [freeText, setFreeText] = useState('') const [sending, setSending] = useState(false) const sendingRef = useRef(false) + const allowOther = question.allowOther !== false const hasOptions = question.options.length > 0 const trimmedFreeText = freeText.trim() @@ -42,8 +47,9 @@ export function MobileNativeChatQuestion({ question, onAnswer }: Props): React.J } } - const answerSingle = async (option: string): Promise => { - await sendAnswer(formatQuestionAnswer(question, [option])) + const answerSingle = async (option: string, optionIndex: number): Promise => { + const token = question.optionTokens[optionIndex] + await sendAnswer(token && token.length > 0 ? token : formatQuestionAnswer(question, [option])) } const submitMulti = async (): Promise => { @@ -57,14 +63,13 @@ export function MobileNativeChatQuestion({ question, onAnswer }: Props): React.J if (trimmedFreeText.length === 0) { return } - // Free text is an unknown entry; formatQuestionAnswer passes it through. - if (await sendAnswer(formatQuestionAnswer(question, [trimmedFreeText]))) { + if (await sendAnswer(formatQuestionFreeTextAnswer(question, trimmedFreeText))) { setFreeText('') } } const canSubmitMulti = selected.length > 0 && !sending - const canSendFreeText = trimmedFreeText.length > 0 && !sending + const canSendFreeText = allowOther && trimmedFreeText.length > 0 && !sending // Stable keys for option rows even if an agent repeats a label. const optionRows = useMemo( @@ -81,7 +86,7 @@ export function MobileNativeChatQuestion({ question, onAnswer }: Props): React.J {hasOptions ? ( - {optionRows.map(({ label, key }) => { + {optionRows.map(({ label, key }, optIndex) => { const isSelected = selected.includes(label) return ( (question.multiSelect ? toggle(label) : answerSingle(label))} + onPress={() => + question.multiSelect ? toggle(label) : answerSingle(label, optIndex) + } > {question.multiSelect ? ( @@ -124,35 +131,37 @@ export function MobileNativeChatQuestion({ question, onAnswer }: Props): React.J ) : null} - - - [ - styles.freeSend, - !canSendFreeText && styles.freeSendDisabled, - pressed && canSendFreeText && styles.pressed - ]} - onPress={submitFreeText} - disabled={!canSendFreeText} - > - + - - + [ + styles.freeSend, + !canSendFreeText && styles.freeSendDisabled, + pressed && canSendFreeText && styles.pressed + ]} + onPress={submitFreeText} + disabled={!canSendFreeText} + > + + + + ) : null} ) } diff --git a/mobile/src/session/MobileSessionActiveContent.tsx b/mobile/src/session/MobileSessionActiveContent.tsx index 7dd09977c45..019e83c6a99 100644 --- a/mobile/src/session/MobileSessionActiveContent.tsx +++ b/mobile/src/session/MobileSessionActiveContent.tsx @@ -38,7 +38,7 @@ export function MobileSessionActiveContent({ browserScreencastSupported, showToast, nativeChatSendError, - nativeChatInputLockReason, + nativeChatOverlayInputLockReason, nativeChatController, dictation, handleDictationToggle, @@ -240,7 +240,7 @@ export function MobileSessionActiveContent({ dictationMode={dictationMode} onMicPressIn={handleDictationPressIn} onMicPressOut={handleDictationPressOut} - inputLockReason={nativeChatInputLockReason} + inputLockReason={nativeChatOverlayInputLockReason} sendErrorMessage={nativeChatSendError.message} onClearSendError={nativeChatSendError.clear} sendSurfaceId={controller.nativeChatScopeKey ?? ''} diff --git a/mobile/src/session/MobileSessionHeader.tsx b/mobile/src/session/MobileSessionHeader.tsx index 1ddc7cb1d83..552f507a787 100644 --- a/mobile/src/session/MobileSessionHeader.tsx +++ b/mobile/src/session/MobileSessionHeader.tsx @@ -168,6 +168,7 @@ export function MobileSessionHeader({ controller }: { controller: MobileSessionC {t.type === 'file' && ( )} + {t.type === 'agent-session' && } {t.type === 'terminal' && (() => { const agentId = resolveMobileTerminalTabAgentId(t) diff --git a/mobile/src/session/MobileSessionSheets.tsx b/mobile/src/session/MobileSessionSheets.tsx index 48dcabed23c..0aac2bb9d42 100644 --- a/mobile/src/session/MobileSessionSheets.tsx +++ b/mobile/src/session/MobileSessionSheets.tsx @@ -43,6 +43,8 @@ export function MobileSessionSheets({ controller }: { controller: MobileSessionC setFileActionTarget, browserActionTarget, setBrowserActionTarget, + agentSessionActionTarget, + setAgentSessionActionTarget, discardMarkdownTarget, setDiscardMarkdownTarget, leaveDrafts, @@ -261,6 +263,14 @@ export function MobileSessionSheets({ controller }: { controller: MobileSessionC onCloseTab={handleCloseSessionTab} bulkCloseActions={bulkCloseActions} /> + + setAgentSessionActionTarget(null) + )} + onClose={() => setAgentSessionActionTarget(null)} + /> = { activated: boolean activationSeq: number latestActivationSeq: number - sourceTerminalHandle: string + sourceTerminalHandle: string | null activeTerminalHandle: string | null + sourceSessionTabId?: string | null + activeSessionTabId?: string | null activeTabType: string | null } switchSessionTab: (tab: T) => void diff --git a/mobile/src/session/mobile-native-chat-controller-contract.ts b/mobile/src/session/mobile-native-chat-controller-contract.ts index 2283256da05..890a3a1562e 100644 --- a/mobile/src/session/mobile-native-chat-controller-contract.ts +++ b/mobile/src/session/mobile-native-chat-controller-contract.ts @@ -58,7 +58,12 @@ export type MobileNativeChatController = { handleNativeChatSendWithOutcome: ( text: string, images?: string[], - deadline?: number + deadline?: number, + attachments?: readonly { + id?: string + path: string + previewUri: string + }[] ) => Promise /** Launch-context text still parked on the agent's TUI input line, or null. * Image sends read it to size their leading clear (one Ctrl+U per line). */ diff --git a/mobile/src/session/mobile-native-chat-eligibility.test.ts b/mobile/src/session/mobile-native-chat-eligibility.test.ts index 7daa4babea6..e1bd97cad8f 100644 --- a/mobile/src/session/mobile-native-chat-eligibility.test.ts +++ b/mobile/src/session/mobile-native-chat-eligibility.test.ts @@ -123,6 +123,30 @@ describe('resolveMobileNativeChat', () => { expect(resolveMobileNativeChat({ type: 'browser', launchAgent: 'claude' })).toBeNull() }) + it('resolves Codex structured agent-session tabs directly', () => { + expect( + resolveMobileNativeChat({ + type: 'agent-session', + sessionId: 'structured-1', + agent: 'codex' + }) + ).toEqual({ + agent: 'codex', + sessionId: 'structured-1', + transcriptPath: null + }) + }) + + it('rejects non-Codex structured agent-session tabs', () => { + expect( + resolveMobileNativeChat({ + type: 'agent-session', + sessionId: 'structured-1', + agent: 'claude' + } as never) + ).toBeNull() + }) + it('canShowMobileNativeChat mirrors resolution', () => { expect(canShowMobileNativeChat({ type: 'terminal', launchAgent: 'claude' })).toBe(true) expect(canShowMobileNativeChat(null)).toBe(false) diff --git a/mobile/src/session/mobile-native-chat-eligibility.ts b/mobile/src/session/mobile-native-chat-eligibility.ts index abda64b04ab..a3f66eb14aa 100644 --- a/mobile/src/session/mobile-native-chat-eligibility.ts +++ b/mobile/src/session/mobile-native-chat-eligibility.ts @@ -32,6 +32,8 @@ export type MobileNativeChatTab = { /** Host-provided launch context still parked as an unsent TUI-input draft. */ launchDraft?: string launchDraftCreatedAt?: number + sessionId?: string | null + agent?: string | null } /** Resolve a session tab to the transcript identity native chat needs, or @@ -42,7 +44,15 @@ export function resolveMobileNativeChat( tab: MobileNativeChatTab | null, nativeChatTranscriptIsLocalReadable = false ): MobileNativeChatResolution | null { - if (!tab || tab.type !== 'terminal') { + if (!tab) { + return null + } + if (tab.type === 'agent-session') { + return tab.sessionId && tab.agent === 'codex' + ? { agent: tab.agent, sessionId: tab.sessionId, transcriptPath: null } + : null + } + if (tab.type !== 'terminal') { return null } const liveAgent = tab.agentStatus?.agentType ?? null @@ -71,3 +81,15 @@ export function canShowMobileNativeChat( ): boolean { return resolveMobileNativeChat(tab, nativeChatTranscriptIsLocalReadable) !== null } + +export function resolveMobileNativeChatFileSessionId( + tab: MobileNativeChatTab | null +): string | null { + if (tab?.type === 'agent-session') { + return tab.sessionId ?? null + } + if (tab?.type === 'terminal') { + return tab.agentStatus?.providerSession?.id ?? null + } + return null +} diff --git a/mobile/src/session/mobile-native-chat-image-scope-state.ts b/mobile/src/session/mobile-native-chat-image-scope-state.ts new file mode 100644 index 00000000000..8d7de510e3a --- /dev/null +++ b/mobile/src/session/mobile-native-chat-image-scope-state.ts @@ -0,0 +1,18 @@ +import type { PendingNativeChatImage } from './mobile-native-chat-image-attachment' + +export const NO_NATIVE_CHAT_IMAGE_ATTACHMENTS: PendingNativeChatImage[] = [] + +export type MobileNativeChatImagesByScope = Record + +export function withScopeAttachments( + byScope: MobileNativeChatImagesByScope, + scope: string, + next: PendingNativeChatImage[] +): MobileNativeChatImagesByScope { + if (next.length > 0) { + return { ...byScope, [scope]: next } + } + const remaining = { ...byScope } + delete remaining[scope] + return remaining +} diff --git a/mobile/src/session/mobile-native-chat-question.test.ts b/mobile/src/session/mobile-native-chat-question.test.ts index 94fbcf055a9..079e661e545 100644 --- a/mobile/src/session/mobile-native-chat-question.test.ts +++ b/mobile/src/session/mobile-native-chat-question.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from 'vitest' import { formatQuestionAnswer, + formatQuestionFreeTextAnswer, mobileChatQuestionKey, parseAgentQuestion, type MobileChatQuestion @@ -141,6 +142,12 @@ describe('formatQuestionAnswer', () => { expect(formatQuestionAnswer(numbered, [])).toBe('') expect(formatQuestionAnswer(numbered, [' '])).toBe('') }) + + it('prefixes free-text answers with an opaque prompt token when provided', () => { + expect( + formatQuestionFreeTextAnswer({ ...numbered, freeTextToken: 'target' }, ' hi there ') + ).toBe(`target:${encodeURIComponent('hi there')}`) + }) }) describe('mobileChatQuestionKey', () => { @@ -154,5 +161,8 @@ describe('mobileChatQuestionKey', () => { expect(mobileChatQuestionKey({ ...first, options: ['A', 'C'] })).not.toBe( mobileChatQuestionKey(first) ) + expect(mobileChatQuestionKey({ ...first, freeTextToken: 'target-2' })).not.toBe( + mobileChatQuestionKey(first) + ) }) }) diff --git a/mobile/src/session/mobile-native-chat-question.ts b/mobile/src/session/mobile-native-chat-question.ts index 8a1f06dfbf7..5d4e65a46ff 100644 --- a/mobile/src/session/mobile-native-chat-question.ts +++ b/mobile/src/session/mobile-native-chat-question.ts @@ -7,10 +7,14 @@ export type MobileChatQuestion = { question: string options: string[] multiSelect: boolean + /** Structured questions hide the free-text row when the provider does not accept it. */ + allowOther?: boolean /** Per-option leading marker ("1", "b", …) when the source line carried one, * parallel to `options`. Null where the option was a plain bullet. Used to * echo the exact choice the agent listed back to the terminal. */ optionTokens: (string | null)[] + /** Opaque prefix used when free-text answers must target a specific prompt. */ + freeTextToken?: string } export function mobileChatQuestionKey(question: MobileChatQuestion): string { @@ -152,3 +156,13 @@ export function formatQuestionAnswer(question: MobileChatQuestion, selected: str return parts.join(question.multiSelect ? ', ' : ' ') } + +export function formatQuestionFreeTextAnswer(question: MobileChatQuestion, text: string): string { + const trimmed = text.trim() + if (trimmed.length === 0) { + return '' + } + return question.freeTextToken + ? `${question.freeTextToken}:${encodeURIComponent(trimmed)}` + : formatQuestionAnswer(question, [trimmed]) +} diff --git a/mobile/src/session/mobile-session-route-parity.test.ts b/mobile/src/session/mobile-session-route-parity.test.ts index 5134cd373d4..b6abab8001e 100644 --- a/mobile/src/session/mobile-session-route-parity.test.ts +++ b/mobile/src/session/mobile-session-route-parity.test.ts @@ -62,15 +62,15 @@ const HOST_COMPONENT_NAMES = new Set([ 'View' ]) -const HEAD_MAIN_HOOK_SHA256 = '5c475b904928f418c76a7885afdbed7adbfea3fe3ea05e85d956dc22f958a302' -const HEAD_HOOK_BINDING_SHA256 = '028f99dd14fea2110cff446418ee71513aeed38484c2dcea68bf0da8eff377c0' +const HEAD_MAIN_HOOK_SHA256 = '10071240ef9edafc2b9c8bed73be83dceaf7828e3b29f17dab55da020a7697a6' +const HEAD_HOOK_BINDING_SHA256 = 'ecd4c1dad066cf13698447b8ffb61f82e6cc3ebe7d484f71189626efed430272' const HEAD_CALLBACK_IDENTITY_SHA256 = - 'd60ffe53f8d77f2dd3ebd14a5de162bb399113c170b59bdc917de6318ec433ec' -const HEAD_CALLBACK_BODY_SHA256 = '69dfda53fd700f4395a18a37ffdaa530e187bc24b4986d8fdc0184127c00b52d' + 'df073bc13d94a93e7fbd8b1fca2b57eaf43cbf7ca799a649e0ebb783e5b8eecc' +const HEAD_CALLBACK_BODY_SHA256 = '690e3069e08ecf805af726b658e900c973565259160f25e3a643175e2ab1bc75' const HEAD_EFFECT_SHA256 = '346d384ea0bf2f8f926c5092c5bf57bc2a03494f49f9639e9d6b8a2c51c9f882' const HEAD_CONTENT_HOOK_SHA256 = '9c3b612fef3f370d66873aefdbe1d701f20cb64ded31fef5cc45fde6f8189581' const HEAD_NESTED_FUNCTION_SHA256 = - 'b562c117eb1e4532dd656d8bdd3ca3bc58ce65d78a7ed740dbd866a48d4d8dbe' + '6a13919ede2a8033436fb03e0ff7c426fbed97f470875a7b21b00aaada17fb73' const HEAD_NATIVE_REGISTRATION_SHA256 = 'cab85e4e4a3f43289ba93ddea9ccce57aea83e0bf14fd1620a965aad0c1cb49e' const HEAD_NATIVE_REMOVAL_SHA256 = @@ -79,9 +79,9 @@ const HEAD_TIMER_CREATION_SHA256 = '1a31b625e2174c3db77272249843196d2b6b06ab1e654a96d8f7858e3082e66b' const HEAD_TIMER_CLEANUP_SHA256 = 'c73f1d1c2cc89642f3d727d6f3b6b81860a9d6f34234541a2065ec3d1a8cd116' const HEAD_RUNTIME_STRING_SHA256 = - 'ad0def23206f08d0523c155fe730e86824876e67cf1db6b597541b9c35b54447' + '1cb95fe0095c1c57e1b0629472e1cce5328eb7f5bfeca38095f41f4612a37887' const HEAD_HOST_JSX_SHA256 = '390405926b1695fa3a33686f0bc192b432f5468d8576499d7cafbb4922defbb5' -const HEAD_LEAF_JSX_SHA256 = 'b070e25c47b3e298be02a4ffe1572b36e204446fc161bad894690e9939403f54' +const HEAD_LEAF_JSX_SHA256 = '21dba981875e173f692590bf910d60964660c5f4cbb79f3a377c7e54f6a1f016' const HEAD_STYLE_REFERENCE_SHA256 = '295a3501c2c6d7bea7c8bbf38b3f3534f01344cd7e1b91bb8e07c040821d596a' const HEAD_IDENTITY_FIELD_SHA256 = @@ -472,10 +472,10 @@ describe('mobile session route extraction parity', () => { const contentBindings = CONTENT_COMPONENT_NAMES.flatMap( (name) => readHookFacts(name, definitions).bindings ) - expect(main.hooks).toHaveLength(269) + expect(main.hooks).toHaveLength(266) expect(hash(main.hooks)).toBe(HEAD_MAIN_HOOK_SHA256) expect(hash(main.bindings)).toBe(HEAD_HOOK_BINDING_SHA256) - expect(main.callbacks).toHaveLength(78) + expect(main.callbacks).toHaveLength(77) expect(hash(main.callbacks)).toBe(HEAD_CALLBACK_IDENTITY_SHA256) expect(hash(main.callbackBodies)).toBe(HEAD_CALLBACK_BODY_SHA256) expect(main.effects).toHaveLength(24) @@ -517,12 +517,12 @@ describe('mobile session route extraction parity', () => { it('preserves runtime strings, styles, and the expanded JSX tree', () => { const strings = readRuntimeStrings() - expect(strings).toHaveLength(537) + expect(strings).toHaveLength(545) expect(hash(strings)).toBe(HEAD_RUNTIME_STRING_SHA256) const jsx = readJsxFacts(readDefinitions()) expect(jsx.host).toHaveLength(124) expect(hash(jsx.host)).toBe(HEAD_HOST_JSX_SHA256) - expect(jsx.leaf).toHaveLength(59) + expect(jsx.leaf).toHaveLength(61) expect(hash(jsx.leaf)).toBe(HEAD_LEAF_JSX_SHA256) expect(jsx.styleReferences).toHaveLength(172) expect(hash(jsx.styleReferences)).toBe(HEAD_STYLE_REFERENCE_SHA256) diff --git a/mobile/src/session/mobile-session-route-types.ts b/mobile/src/session/mobile-session-route-types.ts index a61b653a0e3..36c90b0a29d 100644 --- a/mobile/src/session/mobile-session-route-types.ts +++ b/mobile/src/session/mobile-session-route-types.ts @@ -9,7 +9,7 @@ import type { TerminalRecord } from './mobile-terminal-records' export type Terminal = TerminalRecord -export type MobileSessionTabType = 'terminal' | 'markdown' | 'file' | 'browser' +export type MobileSessionTabType = 'terminal' | 'markdown' | 'file' | 'browser' | 'agent-session' export type MobileSessionTab = | { @@ -30,6 +30,14 @@ export type MobileSessionTab = terminalTheme?: MobileTerminalTheme isActive: boolean } + | { + type: 'agent-session' + id: string + title: string + sessionId: string + agent: 'codex' + isActive: boolean + } | { type: 'markdown' id: string diff --git a/mobile/src/session/mobile-structured-agent-prompts.ts b/mobile/src/session/mobile-structured-agent-prompts.ts new file mode 100644 index 00000000000..84cb7033d30 --- /dev/null +++ b/mobile/src/session/mobile-structured-agent-prompts.ts @@ -0,0 +1,251 @@ +import type { AgentJournalRenderItem } from '../../../src/shared/agent-session-journal-types' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import type { MobileChatQuestion } from './mobile-native-chat-question' + +export type StructuredApprovalItem = AgentJournalRenderItem & { + body: Extract +} + +export type StructuredQuestionItem = AgentJournalRenderItem & { + body: Extract +} + +export type StructuredPromptResponseTarget = { + itemId: string + expectedRevision: number + optionId: string +} + +type PromptTokenPayload = + | { + kind: 'approval' + itemId: string + revision: number + optionId: string + } + | { + kind: 'question-option' + itemId: string + revision: number + optionId: string + } + | { + kind: 'question-free-text' + itemId: string + revision: number + questionId: string + } + +const STRUCTURED_PROMPT_TOKEN_PREFIX = 'structured-agent-prompt:' + +export function pendingStructuredApproval( + item: AgentJournalRenderItem +): item is StructuredApprovalItem { + return item.body.kind === 'approval' && item.body.resolution.state === 'pending' +} + +export function pendingStructuredQuestion( + item: AgentJournalRenderItem +): item is StructuredQuestionItem { + return item.body.kind === 'question' && item.body.resolution.state === 'pending' +} + +function encodeQuestionAnswer(questionId: string, answer: string): string { + return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` +} + +function encodePromptToken(payload: PromptTokenPayload): string { + return `${STRUCTURED_PROMPT_TOKEN_PREFIX}${encodeURIComponent(JSON.stringify(payload))}` +} + +function decodePromptToken(value: string): PromptTokenPayload | null { + if (!value.startsWith(STRUCTURED_PROMPT_TOKEN_PREFIX)) { + return null + } + try { + const decoded = JSON.parse( + decodeURIComponent(value.slice(STRUCTURED_PROMPT_TOKEN_PREFIX.length)) + ) as Record + if ( + typeof decoded.itemId !== 'string' || + typeof decoded.revision !== 'number' || + !Number.isFinite(decoded.revision) + ) { + return null + } + if (decoded.kind === 'approval' && typeof decoded.optionId === 'string') { + return { + kind: decoded.kind, + itemId: decoded.itemId, + revision: decoded.revision, + optionId: decoded.optionId + } + } + if (decoded.kind === 'question-option' && typeof decoded.optionId === 'string') { + return { + kind: decoded.kind, + itemId: decoded.itemId, + revision: decoded.revision, + optionId: decoded.optionId + } + } + if (decoded.kind === 'question-free-text' && typeof decoded.questionId === 'string') { + return { + kind: decoded.kind, + itemId: decoded.itemId, + revision: decoded.revision, + questionId: decoded.questionId + } + } + } catch { + return null + } + return null +} + +function decodeQuestionFreeTextAnswer(value: string): { + payload: Extract + answer: string +} | null { + if (!value.startsWith(STRUCTURED_PROMPT_TOKEN_PREFIX)) { + return null + } + const separator = value.indexOf(':', STRUCTURED_PROMPT_TOKEN_PREFIX.length) + if (separator === -1) { + return null + } + const payload = decodePromptToken(value.slice(0, separator)) + if (payload?.kind !== 'question-free-text') { + return null + } + return { payload, answer: decodeURIComponent(value.slice(separator + 1)) } +} + +export function projectStructuredPermission( + prompt: StructuredApprovalItem | null +): MobileChatPermission | null { + if (prompt?.body.kind !== 'approval') { + return null + } + return { + title: prompt.body.title, + ...(prompt.body.detail ? { detail: prompt.body.detail } : {}), + options: prompt.body.options.map((option) => ({ + label: option.label, + send: encodePromptToken({ + kind: 'approval', + itemId: prompt.itemId, + revision: prompt.revision, + optionId: option.id + }) + })) + } +} + +export function projectStructuredQuestion( + prompt: StructuredQuestionItem | null +): MobileChatQuestion | null { + if (prompt?.body.kind !== 'question') { + return null + } + return { + question: prompt.body.question, + options: prompt.body.options.map((option) => option.label), + multiSelect: false, + allowOther: Boolean(prompt.body.freeTextQuestionId), + optionTokens: prompt.body.options.map((option) => + encodePromptToken({ + kind: 'question-option', + itemId: prompt.itemId, + revision: prompt.revision, + optionId: option.id + }) + ), + ...(prompt.body.freeTextQuestionId + ? { + freeTextToken: encodePromptToken({ + kind: 'question-free-text', + itemId: prompt.itemId, + revision: prompt.revision, + questionId: prompt.body.freeTextQuestionId + }) + } + : {}) + } +} + +export function structuredApprovalResponseTarget( + response: string, + currentPrompt: StructuredApprovalItem | null +): StructuredPromptResponseTarget | null { + const token = decodePromptToken(response) + if (token?.kind === 'approval') { + return { + itemId: token.itemId, + expectedRevision: token.revision, + optionId: token.optionId + } + } + if (token) { + return null + } + const option = currentPrompt?.body.options.find( + (candidate) => candidate.id === response || candidate.label === response + ) + return currentPrompt && option + ? { + itemId: currentPrompt.itemId, + expectedRevision: currentPrompt.revision, + optionId: option.id + } + : null +} + +export function structuredQuestionResponseTarget( + response: string, + currentPrompt: StructuredQuestionItem | null +): StructuredPromptResponseTarget | null { + const token = decodePromptToken(response) + if (token?.kind === 'question-option') { + return { + itemId: token.itemId, + expectedRevision: token.revision, + optionId: token.optionId + } + } + if (token) { + return null + } + const freeText = decodeQuestionFreeTextAnswer(response) + if (freeText) { + const answer = freeText.answer.trim() + return answer.length > 0 + ? { + itemId: freeText.payload.itemId, + expectedRevision: freeText.payload.revision, + optionId: encodeQuestionAnswer(freeText.payload.questionId, answer) + } + : null + } + if (!currentPrompt) { + return null + } + const trimmed = response.trim() + const option = currentPrompt.body.options.find( + (candidate) => candidate.id === response || candidate.label === trimmed + ) + if (option) { + return { + itemId: currentPrompt.itemId, + expectedRevision: currentPrompt.revision, + optionId: option.id + } + } + return currentPrompt.body.freeTextQuestionId && trimmed + ? { + itemId: currentPrompt.itemId, + expectedRevision: currentPrompt.revision, + optionId: encodeQuestionAnswer(currentPrompt.body.freeTextQuestionId, trimmed) + } + : null +} diff --git a/mobile/src/session/mobile-structured-agent-session-launch.test.ts b/mobile/src/session/mobile-structured-agent-session-launch.test.ts new file mode 100644 index 00000000000..54f9b5cbe88 --- /dev/null +++ b/mobile/src/session/mobile-structured-agent-session-launch.test.ts @@ -0,0 +1,142 @@ +import { describe, expect, it, vi } from 'vitest' +import type { RpcClient } from '../transport/rpc-client' +import { markRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' +import { createMobileStructuredCodexSession } from './mobile-structured-agent-session-launch' + +function clientReturning( + ...responses: unknown[] +): RpcClient & { sendRequest: ReturnType } { + let responseIndex = 0 + const sendRequest = vi.fn(async () => responses[responseIndex++]) + return { sendRequest } as unknown as RpcClient & { sendRequest: ReturnType } +} + +const acceptedCreateResult = { + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-1', sequence: 0 }, + value: { + sessionId: 'codex_session_1', + fence: 1, + page: { + sessionId: 'codex_session_1', + epoch: 'epoch-1', + direction: 'tail', + items: [], + removedItemIds: [], + submissions: [], + window: { oldest: null, newest: null, nextCursor: { epoch: 'epoch-1', sequence: 0 } }, + liveCursor: { epoch: 'epoch-1', sequence: 0 }, + hasOlder: false, + hasNewer: false + }, + unconfirmedClientMessageIds: [] + } +} +const acceptedCreate = { ok: true, result: acceptedCreateResult } + +describe('mobile structured Codex launch', () => { + it('creates through the structured agent-session intent after support is confirmed', async () => { + const client = clientReturning({ ok: true, result: { supported: true } }, acceptedCreate) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toMatchObject({ + kind: 'created', + sessionId: expect.stringMatching(/^codex_[A-Za-z0-9_]{8,128}$/) + }) + expect(client.sendRequest).toHaveBeenNthCalledWith(1, 'agentSession.createSupport', { + worktree: 'id:workspace-1', + agent: 'codex' + }) + expect(client.sendRequest).toHaveBeenNthCalledWith( + 2, + 'agentSession.create', + expect.objectContaining({ + worktree: 'id:workspace-1', + agent: 'codex', + envelope: expect.objectContaining({ expectedRuntimeFence: null }) + }), + expect.objectContaining({ budgetSpansConnect: true }) + ) + const params = client.sendRequest.mock.calls[1]?.[1] as { + envelope: { sessionId: string; payloadFingerprint: string } + worktree: string + agent: 'codex' + } + expect(params.envelope.payloadFingerprint).toMatch(/^[0-9a-f]{64}$/) + expect(params.envelope.sessionId).toMatch(/^codex_[A-Za-z0-9_]{8,128}$/) + }) + + it('reports unsupported without creating a terminal when the structured path is unavailable', async () => { + const client = clientReturning({ ok: true, result: { supported: false, reason: 'remote' } }) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toEqual({ + kind: 'unsupported', + reason: 'remote' + }) + expect(client.sendRequest).toHaveBeenCalledTimes(1) + }) + + it('keeps an unknown create outcome distinct so callers do not create a duplicate terminal', async () => { + const client = clientReturning({ ok: true, result: { supported: true } }) + client.sendRequest.mockImplementationOnce(async () => ({ + ok: true, + result: { supported: true } + })) + client.sendRequest.mockRejectedValue(markRpcDeliveryUnknown(new Error('response lost'))) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toMatchObject({ + kind: 'unknown' + }) + expect(client.sendRequest.mock.calls.map(([method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create', + 'agentSession.create' + ]) + expect(client.sendRequest.mock.calls[1]?.[1]).toBe(client.sendRequest.mock.calls[2]?.[1]) + }) + + it('keeps the outcome unknown when the idempotent retry cannot be sent', async () => { + const client = clientReturning({ ok: true, result: { supported: true } }) + client.sendRequest.mockImplementationOnce(async () => ({ + ok: true, + result: { supported: true } + })) + client.sendRequest.mockRejectedValueOnce(markRpcDeliveryUnknown(new Error('response lost'))) + client.sendRequest.mockRejectedValueOnce(new Error('connection interrupted')) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toMatchObject({ + kind: 'unknown' + }) + }) + + it('never creates a legacy sibling after an unclassified create exception', async () => { + const client = clientReturning({ ok: true, result: { supported: true } }) + client.sendRequest.mockImplementationOnce(async () => ({ + ok: true, + result: { supported: true } + })) + client.sendRequest.mockRejectedValue(new Error('internal error after commit')) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toMatchObject({ + kind: 'unknown' + }) + expect(client.sendRequest.mock.calls.map(([method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create', + 'agentSession.create' + ]) + expect(client.sendRequest.mock.calls[1]?.[1]).toBe(client.sendRequest.mock.calls[2]?.[1]) + }) + + it('treats malformed structured responses as unknown', async () => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { ok: true, result: { ok: true, value: { sessionId: '' } } } + ) + + await expect(createMobileStructuredCodexSession(client, 'workspace-1')).resolves.toMatchObject({ + kind: 'unknown' + }) + }) +}) diff --git a/mobile/src/session/mobile-structured-agent-session-launch.ts b/mobile/src/session/mobile-structured-agent-session-launch.ts new file mode 100644 index 00000000000..ecad0410dfd --- /dev/null +++ b/mobile/src/session/mobile-structured-agent-session-launch.ts @@ -0,0 +1,158 @@ +import type { + AgentSessionAttachResult, + AgentSessionMutationResult +} from '../../../src/shared/agent-session-wire' +import { structuredAgentSessionPayloadFingerprint } from '../../../src/shared/structured-agent-session-mutation' +import type { RpcClient } from '../transport/rpc-client' +import { structuredSessionOperationId } from './mobile-structured-agent-session-rpc' + +type StructuredCreateSupport = { + supported?: boolean + reason?: 'agent' | 'remote' | 'wsl' +} + +export type MobileStructuredCodexLaunchResult = + | { kind: 'created'; sessionId: string } + | { kind: 'unsupported'; reason?: StructuredCreateSupport['reason'] } + | { kind: 'failed'; message: string } + | { kind: 'unknown'; message: string } + +type StructuredCreateParams = { + envelope: { + sessionId: string + clientOperationId: string + expectedRuntimeFence: null + payloadFingerprint: string + } + worktree: string + agent: 'codex' +} + +function createStructuredCodexSessionId(): string { + return `codex_${createRandomUuid().replaceAll('-', '_')}` +} + +function createRandomUuid(): string { + if (typeof globalThis.crypto?.randomUUID === 'function') { + return globalThis.crypto.randomUUID() + } + return Array.from({ length: 32 }, () => Math.floor(Math.random() * 16).toString(16)).join('') +} + +function createStructuredCodexSessionParams(worktreeId: string): StructuredCreateParams { + const sessionId = createStructuredCodexSessionId() + const worktree = `id:${worktreeId}` + const fields = { worktree, agent: 'codex' as const } + return { + envelope: { + sessionId, + clientOperationId: structuredSessionOperationId(), + expectedRuntimeFence: null, + payloadFingerprint: structuredAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId, + fields + }) + }, + ...fields + } +} + +function unknownCreateResult(error: unknown): MobileStructuredCodexLaunchResult { + const message = error instanceof Error ? error.message.trim() : '' + return { + kind: 'unknown', + message: message || 'The Codex chat result could not be confirmed.' + } +} + +export async function createMobileStructuredCodexSession( + client: RpcClient, + worktreeId: string +): Promise { + const worktree = `id:${worktreeId}` + let supportResponse + try { + supportResponse = await client.sendRequest('agentSession.createSupport', { + worktree, + agent: 'codex' + }) + } catch { + // A support probe has no side effect; an unavailable probe safely degrades to terminal chat. + return { kind: 'unsupported' } + } + if ( + !supportResponse || + typeof supportResponse !== 'object' || + typeof supportResponse.ok !== 'boolean' || + !supportResponse.ok + ) { + return { kind: 'unsupported' } + } + const support = supportResponse.result as StructuredCreateSupport | null + if (!support || typeof support !== 'object' || support.supported !== true) { + return { kind: 'unsupported', reason: support?.reason } + } + + const params = createStructuredCodexSessionParams(worktreeId) + let response + try { + response = await client.sendRequest('agentSession.create', params, { + timeoutMs: 15_000, + budgetSpansConnect: true + }) + } catch { + // Replay the durable envelope once so a lost acknowledgement cannot create a sibling. + try { + response = await client.sendRequest('agentSession.create', params, { + timeoutMs: 15_000, + budgetSpansConnect: true + }) + } catch (retryError) { + // A second transport error cannot disprove the first attempt committed. + return unknownCreateResult(retryError) + } + } + + if (!response || typeof response !== 'object' || typeof response.ok !== 'boolean') { + return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) + } + if (!response.ok) { + if ( + !response.error || + typeof response.error !== 'object' || + typeof response.error.code !== 'string' + ) { + return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) + } + if (response.error.code === 'agent_session_operation_unknown') { + return unknownCreateResult(new Error(response.error.message)) + } + return { kind: 'failed', message: response.error.message || 'Could not open Codex chat.' } + } + const result = response.result as AgentSessionMutationResult + if (!result || typeof result !== 'object' || typeof result.ok !== 'boolean') { + return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) + } + if (!result.ok) { + if ( + !result.refusal || + typeof result.refusal !== 'object' || + typeof result.refusal.code !== 'string' + ) { + return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) + } + if (result.refusal.code === 'agent_session_operation_unknown') { + return unknownCreateResult(new Error(result.refusal.message)) + } + return { kind: 'failed', message: result.refusal.message || 'Could not open Codex chat.' } + } + if ( + !result.value || + typeof result.value.sessionId !== 'string' || + !result.value.sessionId.trim() + ) { + return unknownCreateResult(new Error('The Codex chat result could not be confirmed.')) + } + return { kind: 'created', sessionId: result.value.sessionId } +} diff --git a/mobile/src/session/mobile-structured-agent-session-rpc.ts b/mobile/src/session/mobile-structured-agent-session-rpc.ts new file mode 100644 index 00000000000..a602122978e --- /dev/null +++ b/mobile/src/session/mobile-structured-agent-session-rpc.ts @@ -0,0 +1,152 @@ +import { + AGENT_SESSION_MAX_NEW_OPERATION_AGE_MS, + parseAgentSessionOperationTimestamp +} from '../../../src/shared/agent-session-host-authority' +import type { AgentSessionMutationResult } from '../../../src/shared/agent-session-wire' +import { + createStructuredAgentSessionOperationId, + structuredAgentSessionPayloadFingerprint +} from '../../../src/shared/structured-agent-session-mutation' +import { isRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' +import type { RpcClient } from '../transport/rpc-client' +import { isLogicalClientCutoverError } from '../transport/stable-logical-rpc-client' +import { MOBILE_NATIVE_CHAT_MIN_WRITE_TIMEOUT_MS } from './mobile-native-chat-send' + +export const STRUCTURED_SEND_TIMEOUT_MS = 15_000 + +export type StructuredAgentSessionMutationCallResult = + | { status: 'accepted'; value: TValue } + | { status: 'refused'; message: string } + | { status: 'failed'; message: string } + | { status: 'unknown' } + +export type StructuredAgentSessionMutationResult = + | { status: 'accepted'; value: TValue; sameFence: boolean } + | { status: 'rejected' } + | { status: 'unknown' } + +export type StructuredAgentSessionMutate = ( + method: string, + fingerprintMethod: string, + fields: Record +) => Promise> + +export async function callAgentSession( + client: RpcClient, + method: string, + params: unknown, + timeoutMs = STRUCTURED_SEND_TIMEOUT_MS, + options?: { failWhenDisconnected?: boolean } +): Promise { + const response = await client.sendRequest(method, params, { + timeoutMs, + budgetSpansConnect: true, + ...(options?.failWhenDisconnected ? { failWhenDisconnected: true } : {}) + }) + if (!response.ok) { + throw new Error(response.error.message) + } + return response.result as TResult +} + +export function structuredSessionOperationId(): string { + const randomUuid = + typeof globalThis.crypto?.randomUUID === 'function' + ? () => globalThis.crypto.randomUUID() + : () => { + return Array.from({ length: 32 }, () => Math.floor(Math.random() * 16).toString(16)).join( + '' + ) + } + return createStructuredAgentSessionOperationId(randomUuid) +} + +/** + * Bounded by expiry, never by count: every retained id belongs to a send whose outcome is still + * unknown, so dropping one turns the user's retry into a second message on the host. Only an id + * the host would already refuse — unparseable, or past the window in which it can be admitted — + * is safe to release, which matches the host's own tombstone retention. + */ +export function retainStructuredSessionOperationId( + operationIds: Map, + key: string, + operationId = structuredSessionOperationId(), + now: number = Date.now() +): string { + operationIds.delete(key) + operationIds.set(key, operationId) + for (const [retainedKey, retainedId] of operationIds) { + if (retainedKey === key) { + continue + } + const timestamp = parseAgentSessionOperationTimestamp(retainedId) + if (timestamp === null || now - timestamp > AGENT_SESSION_MAX_NEW_OPERATION_AGE_MS) { + operationIds.delete(retainedKey) + } + } + return operationId +} + +export function timeoutForDeadline(deadline: number | undefined): number | null { + if (deadline === undefined) { + return STRUCTURED_SEND_TIMEOUT_MS + } + const timeoutMs = deadline - Date.now() + return timeoutMs >= MOBILE_NATIVE_CHAT_MIN_WRITE_TIMEOUT_MS ? timeoutMs : null +} + +export async function requestStructuredAgentSessionMutation(args: { + client: RpcClient + method: string + fingerprintMethod: string + sessionId: string + expectedRuntimeFence: number + fields: Record + clientOperationId?: string + retryUnknown?: boolean + timeoutMs?: number +}): Promise> { + const { + client, + method, + fingerprintMethod, + sessionId, + expectedRuntimeFence, + fields, + clientOperationId, + retryUnknown, + timeoutMs + } = args + try { + const result = await callAgentSession>( + client, + method, + { + envelope: { + sessionId, + clientOperationId: clientOperationId ?? structuredSessionOperationId(), + expectedRuntimeFence, + payloadFingerprint: structuredAgentSessionPayloadFingerprint({ + method: fingerprintMethod, + sessionId, + fields + }) + }, + ...(retryUnknown ? { retryUnknown: true } : {}), + ...fields + }, + timeoutMs + ) + return result.ok + ? { status: 'accepted', value: result.value } + : { status: 'refused', message: result.refusal.message } + } catch (error) { + if (isRpcDeliveryUnknown(error) || isLogicalClientCutoverError(error)) { + return { status: 'unknown' } + } + return { + status: 'failed', + message: error instanceof Error ? error.message : 'Request not sent' + } + } +} diff --git a/mobile/src/session/mobile-structured-session-operation-retention.test.ts b/mobile/src/session/mobile-structured-session-operation-retention.test.ts new file mode 100644 index 00000000000..209f28f65c9 --- /dev/null +++ b/mobile/src/session/mobile-structured-session-operation-retention.test.ts @@ -0,0 +1,58 @@ +import { describe, expect, it } from 'vitest' +import { AGENT_SESSION_MAX_NEW_OPERATION_AGE_MS } from '../../../src/shared/agent-session-host-authority' +import { retainStructuredSessionOperationId } from './mobile-structured-agent-session-rpc' + +const NOW = 1_900_000_000_000 + +function operationIdAt(timestamp: number, entropy: string): string { + return `${timestamp}-${entropy.repeat(32).slice(0, 32)}` +} + +describe('structured session operation retention', () => { + it('keeps every unconfirmed operation id past the old 128-entry cap', () => { + const operationIds = new Map() + for (let index = 0; index < 400; index += 1) { + retainStructuredSessionOperationId( + operationIds, + `request-${index}`, + operationIdAt(NOW, 'a'), + NOW + ) + } + + expect(operationIds.size).toBe(400) + // Why: the first send is exactly the one a retry would duplicate if it were evicted. + expect(operationIds.get('request-0')).toBe(operationIdAt(NOW, 'a')) + }) + + it('releases only ids the host would already refuse as expired', () => { + const operationIds = new Map() + const expired = operationIdAt(NOW - AGENT_SESSION_MAX_NEW_OPERATION_AGE_MS - 1, 'b') + const admissible = operationIdAt(NOW - AGENT_SESSION_MAX_NEW_OPERATION_AGE_MS, 'c') + retainStructuredSessionOperationId(operationIds, 'stale', expired, NOW) + retainStructuredSessionOperationId(operationIds, 'live', admissible, NOW) + + retainStructuredSessionOperationId(operationIds, 'fresh', operationIdAt(NOW, 'd'), NOW) + + expect(operationIds.has('stale')).toBe(false) + expect(operationIds.get('live')).toBe(admissible) + expect(operationIds.get('fresh')).toBe(operationIdAt(NOW, 'd')) + }) + + it('drops ids the host could never admit and re-keys a repeated send', () => { + const operationIds = new Map() + retainStructuredSessionOperationId(operationIds, 'unparseable', 'not-an-operation-id', NOW) + const reused = retainStructuredSessionOperationId( + operationIds, + 'send', + operationIdAt(NOW, 'e'), + NOW + ) + + // A retry of the same send reuses the retained id rather than minting a duplicate. + expect( + retainStructuredSessionOperationId(operationIds, 'send', operationIds.get('send'), NOW) + ).toBe(reused) + expect(operationIds.has('unparseable')).toBe(false) + }) +}) diff --git a/mobile/src/session/mobile-terminal-records.test.ts b/mobile/src/session/mobile-terminal-records.test.ts index e4ce55a84aa..bcc9510d0e8 100644 --- a/mobile/src/session/mobile-terminal-records.test.ts +++ b/mobile/src/session/mobile-terminal-records.test.ts @@ -182,6 +182,20 @@ describe('mobile terminal records', () => { ).toBe(false) }) + it('treats structured agent-session identity changes as session-tab changes', () => { + const base = { + type: 'agent-session' as const, + id: 'agent-tab-1', + title: 'Codex', + sessionId: 'session-1', + agent: 'codex', + isActive: true + } + + expect(mobileSessionTabsEqual([base], [{ ...base }])).toBe(true) + expect(mobileSessionTabsEqual([base], [{ ...base, sessionId: 'session-2' }])).toBe(false) + }) + const record = (over: Partial & { handle: string }): TerminalRecord => ({ title: 'Terminal', terminalTheme: undefined, diff --git a/mobile/src/session/mobile-terminal-records.ts b/mobile/src/session/mobile-terminal-records.ts index 09426863b99..f03a31acf41 100644 --- a/mobile/src/session/mobile-terminal-records.ts +++ b/mobile/src/session/mobile-terminal-records.ts @@ -62,6 +62,14 @@ type MobileSessionTabLike = canGoForward?: boolean isActive?: boolean } + | { + type: 'agent-session' + id: string + title?: string + sessionId?: string + agent?: string + isActive?: boolean + } export function mobileTerminalThemesEqual( left: MobileTerminalTheme | null | undefined, @@ -152,6 +160,8 @@ function mobileSessionTabEqual( a.canGoBack === b.canGoBack && a.canGoForward === b.canGoForward ) + case 'agent-session': + return b.type === 'agent-session' && a.sessionId === b.sessionId && a.agent === b.agent } } diff --git a/mobile/src/session/mobile-terminal-tab-agent.test.ts b/mobile/src/session/mobile-terminal-tab-agent.test.ts index 5177981f0e6..6ad034335ad 100644 --- a/mobile/src/session/mobile-terminal-tab-agent.test.ts +++ b/mobile/src/session/mobile-terminal-tab-agent.test.ts @@ -120,4 +120,17 @@ describe('getMobileSessionTabTitle', () => { expect(getMobileSessionTabTitle(blankBrowserTab)).toBe('New Browser') }) + + it('labels structured agent-session tabs without terminal decoration rules', () => { + expect( + getMobileSessionTabTitle({ + type: 'agent-session', + id: 'agent-tab-1', + title: 'Codex Chat', + sessionId: 'session-1', + agent: 'codex', + isActive: true + }) + ).toBe('Codex Chat') + }) }) diff --git a/mobile/src/session/mobile-terminal-tab-agent.ts b/mobile/src/session/mobile-terminal-tab-agent.ts index 20326d41c98..dfc7be812e8 100644 --- a/mobile/src/session/mobile-terminal-tab-agent.ts +++ b/mobile/src/session/mobile-terminal-tab-agent.ts @@ -62,6 +62,9 @@ export function getMobileSessionTabTitle(tab: MobileSessionTab): string { if (tab.type === 'file') { return tab.title || 'File' } + if (tab.type === 'agent-session') { + return tab.title || 'Chat' + } // Why: strip the leading agent status glyph (✳ etc.) once the tab shows the // provider icon. Mobile falls back for glyph-only titles because iOS can // render the bare status glyph as a stray colored box beside the icon. diff --git a/mobile/src/session/opened-mobile-session-tab.test.ts b/mobile/src/session/opened-mobile-session-tab.test.ts index 988a3ea313f..dbeed5ed830 100644 --- a/mobile/src/session/opened-mobile-session-tab.test.ts +++ b/mobile/src/session/opened-mobile-session-tab.test.ts @@ -378,4 +378,19 @@ describe('shouldActivateOpenedMobileSessionTab', () => { }) ).toBe(false) }) + + it('allows a structured agent-session tab to anchor chat file activation', () => { + expect( + shouldActivateOpenedMobileSessionTab({ + activated: false, + activationSeq: 2, + latestActivationSeq: 2, + sourceTerminalHandle: null, + activeTerminalHandle: null, + sourceSessionTabId: 'agent-tab-1', + activeSessionTabId: 'agent-tab-1', + activeTabType: 'agent-session' + }) + ).toBe(true) + }) }) diff --git a/mobile/src/session/opened-mobile-session-tab.ts b/mobile/src/session/opened-mobile-session-tab.ts index 17c8cff9848..f76571eceef 100644 --- a/mobile/src/session/opened-mobile-session-tab.ts +++ b/mobile/src/session/opened-mobile-session-tab.ts @@ -9,8 +9,10 @@ export type OpenedMobileSessionTabActivationState = { activated: boolean activationSeq: number latestActivationSeq: number - sourceTerminalHandle: string + sourceTerminalHandle: string | null activeTerminalHandle: string | null + sourceSessionTabId?: string | null + activeSessionTabId?: string | null activeTabType: string | null } @@ -114,12 +116,15 @@ export async function activateOpenedSourceControlDiffTab( diff --git a/mobile/src/session/use-mobile-file-tap-handlers.test.ts b/mobile/src/session/use-mobile-file-tap-handlers.test.ts index 21f07562459..6d0adbbf8df 100644 --- a/mobile/src/session/use-mobile-file-tap-handlers.test.ts +++ b/mobile/src/session/use-mobile-file-tap-handlers.test.ts @@ -142,4 +142,31 @@ describe('useMobileFileTapHandlers', () => { ) expect(options.reportChatTapFailure).toHaveBeenCalledWith("Couldn't open mobile/src/x.ts:12") }) + + it('lets structured chat file taps resolve without a backing terminal handle', async () => { + const sendRequest = vi.fn(async () => ok({ exists: false, isDirectory: false })) + const options = { + ...createOptions(sendRequest), + activeHandleRef: { current: null as string | null }, + getActiveSessionTabId: () => 'agent-tab-1', + getActiveSessionTabType: () => 'agent-session' + } + act(() => { + renderer = create(createElement(Harness, { options })) + }) + + handlers!.handleNativeChatFileTap('src/app.ts') + await act(async () => {}) + + expect(sendRequest).toHaveBeenCalledWith( + 'files.resolveTerminalPath', + { + worktree: 'id:wt-1', + pathText: 'src/app.ts', + crossWorkspace: true, + nativeChatContext: { tabId: 'agent-tab-1', sessionId: 'session-1' } + }, + { timeoutMs: 10_000 } + ) + }) }) diff --git a/mobile/src/session/use-mobile-file-tap-handlers.ts b/mobile/src/session/use-mobile-file-tap-handlers.ts index 67995115f45..5bfe71f839b 100644 --- a/mobile/src/session/use-mobile-file-tap-handlers.ts +++ b/mobile/src/session/use-mobile-file-tap-handlers.ts @@ -141,15 +141,13 @@ export function useMobileFileTapHandlers( const handleNativeChatFileTap = useCallback((pathText: string) => { const current = optionsRef.current - // The chat overlay rides on its backing terminal tab; that handle anchors - // the activation gate even though resolution ignores the terminal's cwd. const sourceTerminalHandle = current.activeHandleRef.current - if (!current.client || !sourceTerminalHandle) { + const nativeChatSessionId = current.nativeChatSessionId + const nativeChatTabId = current.getActiveSessionTabId() + if (!current.client || (!sourceTerminalHandle && !(nativeChatSessionId && nativeChatTabId))) { return } const activationSeq = ++activationSeqRef.current - const nativeChatSessionId = current.nativeChatSessionId - const nativeChatTabId = current.getActiveSessionTabId() openMobileNativeChatFileTap({ client: current.client, hostId: current.hostId, @@ -172,6 +170,8 @@ export function useMobileFileTapHandlers( latestActivationSeq: activationSeqRef.current, sourceTerminalHandle, activeTerminalHandle: current.activeHandleRef.current, + sourceSessionTabId: nativeChatTabId, + activeSessionTabId: current.getActiveSessionTabId(), activeTabType: current.getActiveSessionTabType() }), switchSessionTab: current.switchSessionTab, diff --git a/mobile/src/session/use-mobile-native-chat-active-resolution.ts b/mobile/src/session/use-mobile-native-chat-active-resolution.ts new file mode 100644 index 00000000000..ea7f923de47 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-active-resolution.ts @@ -0,0 +1,83 @@ +import { useLayoutEffect, useRef, type MutableRefObject } from 'react' +import { encodeNativeChatTranscriptIdentity } from '../../../src/shared/native-chat-transcript-retention' +import { resolveMobileNativeChat, type MobileNativeChatTab } from './mobile-native-chat-eligibility' +import { useMobileSessionViewMode } from './use-mobile-session-view-mode' + +export function useMobileNativeChatActiveResolution(args: { + hostId: string + worktreeId: string + activeSessionTab: MobileNativeChatTab | null + activeSessionTabId: string | null + activeHandleRef: MutableRefObject + nativeChatTranscriptIsLocalReadable: boolean +}): { + isTabChatView: (tabId: string) => boolean + toggleTabChatView: (tabId: string) => void + showNativeChat: boolean + showNativeChatRef: MutableRefObject + activeChatAgent: string | null + activeChatAgentRef: MutableRefObject + activeChatSessionId: string | null + activeChatStructured: boolean + activeChatResolution: ReturnType + activeTabAgentWorking: boolean + nativeChatStatus: MobileNativeChatTab['agentStatus'] | null + sourceIdentity: string + streamIdentity: string + streamScopeKey: string +} { + const { + activeHandleRef, + activeSessionTab, + activeSessionTabId, + hostId, + nativeChatTranscriptIsLocalReadable, + worktreeId + } = args + const { isTabChatView, toggleTabChatView } = useMobileSessionViewMode({ hostId, worktreeId }) + const tabWantsChat = + activeSessionTab?.type === 'agent-session' || + (activeSessionTabId ? isTabChatView(activeSessionTabId) : false) + const activeChatResolution = + activeSessionTab && activeSessionTabId && tabWantsChat + ? resolveMobileNativeChat(activeSessionTab, nativeChatTranscriptIsLocalReadable) + : null + const showNativeChat = activeChatResolution != null + const showNativeChatRef = useRef(showNativeChat) + const activeChatAgent = activeChatResolution?.agent ?? null + const activeChatAgentRef = useRef(activeChatAgent) + + useLayoutEffect(() => { + showNativeChatRef.current = showNativeChat + activeChatAgentRef.current = activeChatAgent + }, [activeChatAgent, showNativeChat]) + + const activeChatSessionId = activeChatResolution?.sessionId ?? null + const activeChatStructured = + activeChatResolution != null && activeSessionTab?.type === 'agent-session' + const activeTabStatus = activeSessionTab?.agentStatus + const activeTabAgentWorking = + activeTabStatus?.state === 'working' && activeTabStatus.workingMode !== 'monitoring' + const nativeChatStatus = activeChatResolution && !activeChatStructured ? activeTabStatus : null + const routeKey = `${hostId}\0${worktreeId}\0${activeSessionTabId ?? ''}` + const streamIdentity = `${routeKey}\0${activeChatSessionId ?? ''}\0${activeHandleRef.current ?? ''}` + const providerSessionId = activeSessionTab?.agentStatus?.providerSession?.id ?? '' + const streamScopeKey = `${routeKey}\0${activeChatSessionId ?? providerSessionId}\0${activeHandleRef.current ?? ''}` + + return { + isTabChatView, + toggleTabChatView, + showNativeChat, + showNativeChatRef, + activeChatAgent, + activeChatAgentRef, + activeChatSessionId, + activeChatStructured, + activeChatResolution, + activeTabAgentWorking, + nativeChatStatus, + sourceIdentity: encodeNativeChatTranscriptIdentity([hostId, worktreeId]), + streamIdentity, + streamScopeKey + } +} diff --git a/mobile/src/session/use-mobile-native-chat-controller.test.ts b/mobile/src/session/use-mobile-native-chat-controller.test.ts index 40937adc0e0..83f8075d914 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.test.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.test.ts @@ -1,6 +1,7 @@ import { createElement } from 'react' import { act, create, type ReactTestRenderer } from 'react-test-renderer' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { SessionOptionDescriptor } from '../../../src/shared/native-chat-session-options' import type { RpcClient } from '../transport/rpc-client' import type { ConnectionState } from '../transport/types' @@ -14,6 +15,55 @@ const holdUnconfirmedSend = vi.fn() // and transcript state; defaults keep the send-seam tests unchanged. const viewMode = { isTabChatView: (_tabId: string) => true } const sessionState = { messages: [] as unknown[], status: 'ready', transcriptLoading: false } +const structuredSendWithOutcome = vi.fn() +const structuredCancel = vi.fn() +const structuredRespondPermission = vi.fn(async () => true) +const structuredRespondQuestion = vi.fn(async () => true) +const structuredSetOption = vi.fn(async () => true) +const structuredInvokeOption = vi.fn(async () => true) +const structuredOptionSnapshot: SessionOptionDescriptor[] = [ + { + id: 'model', + label: 'Model', + category: 'model', + kind: { + type: 'select', + currentValue: 'gpt-fast', + choices: [{ value: 'gpt-fast', label: 'GPT Fast' }] + }, + valueSource: 'reported', + settable: true + } +] +const structuredOptionSurface = { + getSnapshot: () => structuredOptionSnapshot, + setOption: async () => ({ snapshot: structuredOptionSnapshot }), + invokeAction: async () => ({ snapshot: structuredOptionSnapshot }), + subscribe: () => () => {} +} +const structuredPermission = { + title: 'Allow Bash?', + detail: 'rm -rf build', + options: [ + { label: 'Allow once', send: 'allow-once' }, + { label: 'Deny', send: 'deny' } + ] +} +const structuredQuestion = { + question: 'Pick destination', + options: ['Choice A', 'Choice B'], + allowOther: true, + optionTokens: ['choice-a', 'choice-b'] +} +const structuredSessionState = { + messages: [] as unknown[], + status: 'ready', + transcriptLoading: false, + error: undefined, + hasMore: false, + loadingEarlier: false, + loadEarlier: vi.fn() +} const draftsArgs: Record[] = [] const promptsState = { permission: null as unknown, @@ -33,6 +83,24 @@ vi.mock('./use-mobile-session-view-mode', () => ({ vi.mock('./use-mobile-native-chat-session', () => ({ useMobileNativeChatSession: () => sessionState })) +vi.mock('./use-mobile-structured-agent-session', () => ({ + useMobileStructuredAgentSession: () => ({ + session: structuredSessionState, + isWorking: false, + turnId: null, + sendWithOutcome: structuredSendWithOutcome, + cancel: structuredCancel, + permission: structuredPermission, + question: structuredQuestion, + optionSnapshot: structuredOptionSnapshot, + optionSurface: structuredOptionSurface, + pendingOptionId: 'model', + respondPermission: structuredRespondPermission, + respondQuestion: structuredRespondQuestion, + setStructuredOption: structuredSetOption, + invokeStructuredOption: structuredInvokeOption + }) +})) vi.mock('./use-mobile-native-chat-drafts', () => ({ useMobileNativeChatDrafts: (args: Record) => { draftsArgs.push(args) @@ -110,18 +178,28 @@ describe('useMobileNativeChatController handleNativeChatSend', () => { // itself is mocked above). const clientStub = { sendRequest: vi.fn() } - function Harness({ connState = 'connected' }: { connState?: ConnectionState }): null { + function Harness({ + connState = 'connected', + tab = null, + activeHandle = 'term-1', + inputLeaseReady = true + }: { + connState?: ConnectionState + tab?: unknown + activeHandle?: string | null + inputLeaseReady?: boolean + }): null { controller = useMobileNativeChatController({ client: clientStub as unknown as RpcClient, connState, hostId: 'h', worktreeId: 'w', - activeSessionTab: null, - activeSessionTabId: 'tab-1', - activeHandleRef: { current: 'term-1' }, + activeSessionTab: tab as never, + activeSessionTabId: (tab as { id?: string } | null)?.id ?? 'tab-1', + activeHandleRef: { current: activeHandle }, deviceTokenRef: { current: null }, nativeChatTranscriptIsLocalReadable: true, - nativeChatInputLeaseReady: true, + nativeChatInputLeaseReady: inputLeaseReady, onSendError, onSendResolved }) @@ -138,6 +216,7 @@ describe('useMobileNativeChatController handleNativeChatSend', () => { }) resetMobileNativeChatStaleInputForTests() captureSendOrigin.mockReturnValue(ORIGIN) + structuredSendWithOutcome.mockResolvedValue('accepted') act(() => { renderer = create(createElement(Harness)) }) @@ -233,6 +312,80 @@ describe('useMobileNativeChatController handleNativeChatSend', () => { expect(restoreRejectedDraft).not.toHaveBeenCalled() }) + it('routes structured agent-session sends away from terminal/nativeChat transports', async () => { + await act(async () => { + renderer?.update( + createElement(Harness, { + tab: { + type: 'agent-session', + id: 'agent-tab-1', + title: 'Codex Chat', + sessionId: 'session-structured', + agent: 'codex', + isActive: true + }, + activeHandle: null, + inputLeaseReady: false + }) + ) + }) + + let accepted = false + await act(async () => { + accepted = await controller!.handleNativeChatSend('look') + }) + + expect(accepted).toBe(true) + expect(structuredSendWithOutcome).toHaveBeenCalledWith('look') + expect(sendWithOutcome).not.toHaveBeenCalled() + expect(clientStub.sendRequest).not.toHaveBeenCalled() + }) + + it('exposes structured prompt cards and session options on structured tabs', async () => { + await act(async () => { + renderer?.update( + createElement(Harness, { + tab: { + type: 'agent-session', + id: 'agent-tab-1', + title: 'Codex Chat', + sessionId: 'session-structured', + agent: 'codex', + isActive: true + }, + activeHandle: null, + inputLeaseReady: false + }) + ) + }) + + expect(controller!.nativeChatPermission).toEqual(structuredPermission) + expect(controller!.nativeChatQuestion).toEqual(structuredQuestion) + expect(controller!.nativeChatSessionOptions).not.toBeNull() + expect(controller!.nativeChatSessionOptions?.controller.snapshot).toEqual( + structuredOptionSnapshot + ) + + await act(async () => { + expect(await controller!.handleNativeChatRespondPermission('allow-once')).toBe(true) + }) + expect(structuredRespondPermission).toHaveBeenCalledWith('allow-once') + expect(sendWithOutcome).not.toHaveBeenCalled() + + await act(async () => { + expect(await controller!.handleNativeChatQuestionAnswer('choice-a')).toBe(true) + }) + expect(structuredRespondQuestion).toHaveBeenCalledWith('choice-a') + expect(clientStub.sendRequest).not.toHaveBeenCalled() + + await act(async () => { + expect( + await controller!.nativeChatSessionOptions!.controller.setOption('model', 'gpt-fast') + ).toBe(true) + }) + expect(structuredSetOption).toHaveBeenCalledWith('model', 'gpt-fast') + }) + it('pre-clears separately for a text-only send but never for an image send', async () => { // The image path pastes the image behind its OWN leading Ctrl+U and then calls // this send; a second clear here wipes the image off the input line and the diff --git a/mobile/src/session/use-mobile-native-chat-controller.ts b/mobile/src/session/use-mobile-native-chat-controller.ts index 109dea93ec3..bf944398e87 100644 --- a/mobile/src/session/use-mobile-native-chat-controller.ts +++ b/mobile/src/session/use-mobile-native-chat-controller.ts @@ -1,9 +1,7 @@ -import { useCallback, useLayoutEffect, useRef, type MutableRefObject } from 'react' -import { encodeNativeChatTranscriptIdentity } from '../../../src/shared/native-chat-transcript-retention' -import { useMobileSessionViewMode } from './use-mobile-session-view-mode' +import { useLayoutEffect, useRef, type MutableRefObject } from 'react' import type { RpcClient } from '../transport/rpc-client' import type { ConnectionState } from '../transport/types' -import { type MobileNativeChatTab, resolveMobileNativeChat } from './mobile-native-chat-eligibility' +import type { MobileNativeChatTab } from './mobile-native-chat-eligibility' import { useMobileNativeChatPermissionSend } from './mobile-native-chat-permission-send' import { useMobileNativeChatAnswerSend } from './use-mobile-native-chat-answer-send' import { useMobileNativeChatAskDismiss } from './use-mobile-native-chat-ask-dismiss' @@ -11,15 +9,17 @@ import { useMobileNativeChatCancelAsk } from './use-mobile-native-chat-cancel-as import { useMobileNativeChatDrafts } from './use-mobile-native-chat-drafts' import { useMobileNativeChatFileSearch } from './use-mobile-native-chat-file-search' import { useMobileNativeChatMessageSend } from './use-mobile-native-chat-message-send' -import { mobileNativeChatScopeKey } from './mobile-native-chat-scope-key' import { mobileNativeChatStreamPreview } from './mobile-native-chat-streaming-gate' import { useMobileNativeChatSession } from './use-mobile-native-chat-session' -import { useMobileNativeChatSessionOptions } from './use-mobile-native-chat-session-options' +import { useMobileNativeChatSessionOptionController } from './use-mobile-native-chat-session-option-controller' +import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' +import { useMobileStructuredNativeChatSendBridge } from './use-mobile-structured-native-chat-send-bridge' import { useMobileNativeChatPrompts } from './use-mobile-native-chat-prompts' import { useMobileNativeChatStop } from './use-mobile-native-chat-stop' import { useNativeChatAcceptedAction } from './use-native-chat-action-outcomes' import { useThrottledLatestValue } from './use-throttled-latest-value' import type { MobileNativeChatController } from './mobile-native-chat-controller-contract' +import { useMobileNativeChatActiveResolution } from './use-mobile-native-chat-active-resolution' export type { MobileNativeChatController } from './mobile-native-chat-controller-contract' @@ -58,36 +58,51 @@ export function useMobileNativeChatController(args: { onSendError, onSendResolved } = args - const { isTabChatView, toggleTabChatView } = useMobileSessionViewMode({ hostId, worktreeId }) - - const activeChatResolution = - activeSessionTab && activeSessionTabId && isTabChatView(activeSessionTabId) - ? resolveMobileNativeChat(activeSessionTab, nativeChatTranscriptIsLocalReadable) - : null - const showNativeChat = activeChatResolution != null - const showNativeChatRef = useRef(showNativeChat) - const activeChatAgent = activeChatResolution?.agent ?? null - const activeChatAgentRef = useRef(activeChatAgent) - useLayoutEffect(() => { - showNativeChatRef.current = showNativeChat - activeChatAgentRef.current = activeChatAgent - }, [activeChatAgent, showNativeChat]) - - const activeChatSessionId = activeChatResolution?.sessionId ?? null - const routeKey = `${hostId}\0${worktreeId}\0${activeSessionTabId ?? ''}` - const streamIdentity = `${routeKey}\0${activeChatSessionId ?? ''}\0${activeHandleRef.current ?? ''}` - // Same chat, but keyed off the tab rather than the view-gated resolution: - // `streamIdentity` goes session-less the moment the user peeks at the terminal, - // and a scope that flips on a view toggle throws the gate's baseline away. - const streamScopeKey = `${routeKey}\0${activeSessionTab?.agentStatus?.providerSession?.id ?? ''}\0${activeHandleRef.current ?? ''}` - - const nativeChatSession = useMobileNativeChatSession({ - client, - sourceIdentity: encodeNativeChatTranscriptIdentity([hostId, worktreeId]), - agent: activeChatResolution?.agent ?? null, - sessionId: activeChatSessionId, - transcriptPath: activeChatResolution?.transcriptPath ?? null + const { + activeChatAgent, + activeChatAgentRef, + activeChatResolution, + activeChatSessionId, + activeChatStructured, + activeTabAgentWorking, + isTabChatView, + nativeChatStatus, + showNativeChat, + showNativeChatRef, + sourceIdentity, + streamIdentity, + streamScopeKey, + toggleTabChatView + } = useMobileNativeChatActiveResolution({ + hostId, + worktreeId, + activeSessionTab, + activeSessionTabId, + activeHandleRef, + nativeChatTranscriptIsLocalReadable }) + + const legacyNativeChatSession = useMobileNativeChatSession({ + client, + sourceIdentity, + agent: activeChatStructured ? null : (activeChatResolution?.agent ?? null), + sessionId: activeChatStructured ? null : activeChatSessionId, + transcriptPath: activeChatStructured ? null : (activeChatResolution?.transcriptPath ?? null) + }) + const structuredNativeChat = useMobileStructuredAgentSession({ + client, + sessionId: activeChatStructured ? activeChatSessionId : null, + sourceIdentity, + enabled: showNativeChat, + // Holds are connection-scoped; dropping this on transport loss lets the hook + // reacquire the provider without clearing the cached transcript. + connected: connState === 'connected', + agent: activeChatStructured ? activeChatAgent : null, + onSendError + }) + const nativeChatSession = activeChatStructured + ? structuredNativeChat.session + : legacyNativeChatSession const { composerText: chatComposerText, setComposerText: setChatComposerText, @@ -117,27 +132,29 @@ export function useMobileNativeChatController(args: { transcriptSettled: nativeChatSession.status === 'ready' }) - const activeTabStatus = activeSessionTab?.agentStatus - const activeTabAgentWorking = - activeTabStatus?.state === 'working' && activeTabStatus.workingMode !== 'monitoring' - const nativeChatStatus = activeChatResolution ? activeTabStatus : null - const nativeChatAgentWorking = activeChatResolution != null && activeTabAgentWorking + const nativeChatAgentWorking = activeChatStructured + ? structuredNativeChat.isWorking + : activeChatResolution != null && activeTabAgentWorking // Deliberately not gated on the chat view being visible: the streaming gate // has to tell "hidden mid-turn" from "the turn ended". - const nativeChatStreamLive = activeTabAgentWorking + const nativeChatStreamLive = activeChatStructured + ? structuredNativeChat.isWorking + : activeTabAgentWorking // Throttle the streaming bubble: OpenCode emits a status frame per streamed // part, and each one re-renders and re-parses the whole accumulated markdown. const nativeChatStreamingText = useThrottledLatestValue( - mobileNativeChatStreamPreview(nativeChatStatus, nativeChatAgentWorking), + activeChatStructured + ? undefined + : mobileNativeChatStreamPreview(nativeChatStatus, nativeChatAgentWorking), NATIVE_CHAT_STREAM_THROTTLE_MS ) const { - permission: nativeChatPermission, - question: nativeChatQuestion, + permission: legacyNativeChatPermission, + question: legacyNativeChatQuestion, detectedAsk: nativeChatDetectedAsk, ask: nativeChatAskPrompt } = useMobileNativeChatPrompts({ - enabled: activeChatResolution != null, + enabled: activeChatResolution != null && !activeChatStructured, status: nativeChatStatus, messages: nativeChatSession.messages, transcriptLoading: nativeChatSession.transcriptLoading @@ -146,8 +163,6 @@ export function useMobileNativeChatController(args: { const nativeChatTranscriptSettled = nativeChatSession.status === 'ready' || (nativeChatSession.status === 'error' && nativeChatSession.messages.length > 0) - const nativeChatAskObservable = - showNativeChat && (nativeChatDetectedAsk != null || nativeChatTranscriptSettled) const { askKey: nativeChatAskKey, showAsk: showNativeChatAsk, @@ -157,17 +172,19 @@ export function useMobileNativeChatController(args: { detectedAsk: nativeChatDetectedAsk, scopeKey: activeSessionTabId, sessionKey: activeChatSessionId, - observing: nativeChatAskObservable + observing: showNativeChat && (nativeChatDetectedAsk != null || nativeChatTranscriptSettled) }) // Every chat write gates on both: the lease proves the input floor is ours, and // `connState` collapses a render before the lease does on disconnect. - const inputSendable = nativeChatInputLeaseReady && connState === 'connected' + const inputSendable = activeChatStructured + ? client != null && activeChatSessionId != null && connState === 'connected' + : nativeChatInputLeaseReady && connState === 'connected' const { answerAsk: handleNativeChatAnswerAsk, cancelPending: cancelNativeChatAnswer } = useMobileNativeChatAnswerSend({ client, - enabled: inputSendable, + enabled: inputSendable && !activeChatStructured, handleRef: activeHandleRef, deviceTokenRef, agentRef: activeChatAgentRef, @@ -178,16 +195,16 @@ export function useMobileNativeChatController(args: { const handleNativeChatCancelAsk = useMobileNativeChatCancelAsk({ client, - enabled: inputSendable, + enabled: inputSendable && !activeChatStructured, handleRef: activeHandleRef, deviceTokenRef, cancelPending: cancelNativeChatAnswer, onSendError }) - const handleNativeChatRespondPermission = useMobileNativeChatPermissionSend({ + const legacyHandleNativeChatRespondPermission = useMobileNativeChatPermissionSend({ client, - enabled: inputSendable, + enabled: inputSendable && !activeChatStructured, handleRef: activeHandleRef, deviceTokenRef, onSendError @@ -195,7 +212,7 @@ export function useMobileNativeChatController(args: { const handleNativeChatStop = useMobileNativeChatStop({ client, - enabled: inputSendable, + enabled: inputSendable && !activeChatStructured, handleRef: activeHandleRef, deviceTokenRef, streamIdentity, @@ -216,11 +233,11 @@ export function useMobileNativeChatController(args: { const { send: handleNativeChatSend, sendWithOutcome: handleNativeChatSendWithOutcome, - answerQuestion: handleNativeChatQuestionAnswer, + answerQuestion: legacyHandleNativeChatQuestionAnswer, dispatchCommand: handleNativeChatDispatchCommand } = useMobileNativeChatMessageSend({ client, - enabled: inputSendable, + enabled: inputSendable && !activeChatStructured, handleRef: activeHandleRef, deviceTokenRef, agentRef: activeChatAgentRef, @@ -234,26 +251,44 @@ export function useMobileNativeChatController(args: { onSendError }) - // Bring the terminal view forward when an agent-owned picker command is used. - const handleAgentPicker = useCallback(() => { - if (activeSessionTabId && isTabChatView(activeSessionTabId)) { - toggleTabChatView(activeSessionTabId) - } - }, [activeSessionTabId, isTabChatView, toggleTabChatView]) - - const sessionOptions = useMobileNativeChatSessionOptions({ - agent: activeChatResolution?.agent ?? null, - scopeKey: mobileNativeChatScopeKey(hostId, worktreeId, activeSessionTabId), - reportedModel: activeSessionTab?.agentStatus?.model ?? null, - dispatchCommand: handleNativeChatDispatchCommand, - onAgentPicker: handleAgentPicker + const structuredNativeChatSend = useMobileStructuredNativeChatSendBridge({ + sendStructured: structuredNativeChat.sendWithOutcome, + captureSendOrigin, + clearDraftForSend, + acceptSend, + holdUnconfirmedSend, + restoreRejectedDraft, + onSendError }) + + const { nativeChatSessionOptions, recordCommand: recordNativeChatSessionOptionCommand } = + useMobileNativeChatSessionOptionController({ + activeChatStructured, + activeSessionTabId, + agent: activeChatResolution?.agent ?? null, + dispatchCommand: handleNativeChatDispatchCommand, + hostId, + isTabChatView, + isWorking: nativeChatAgentWorking, + reportedModel: activeSessionTab?.agentStatus?.model ?? null, + structured: { + snapshot: structuredNativeChat.optionSnapshot, + pendingId: structuredNativeChat.pendingOptionId, + setOption: structuredNativeChat.setStructuredOption, + invokeAction: structuredNativeChat.invokeStructuredOption + }, + toggleTabChatView, + worktreeId + }) useLayoutEffect(() => { - recordSessionOptionCommandRef.current = sessionOptions.recordCommand - }, [sessionOptions.recordCommand]) + recordSessionOptionCommandRef.current = recordNativeChatSessionOptionCommand + }, [recordNativeChatSessionOptionCommand]) // Card actions retire the route's held failure banner too, not just sends. const answerAsk = useNativeChatAcceptedAction(handleNativeChatAnswerAsk, onSendResolved) const cancelAsk = useNativeChatAcceptedAction(handleNativeChatCancelAsk, onSendResolved) + const handleNativeChatRespondPermission = activeChatStructured + ? structuredNativeChat.respondPermission + : legacyHandleNativeChatRespondPermission const respond = useNativeChatAcceptedAction(handleNativeChatRespondPermission, onSendResolved) return { @@ -272,24 +307,31 @@ export function useMobileNativeChatController(args: { nativeChatStreamingText, nativeChatStreamLive, nativeChatStreamScopeKey: streamScopeKey, - nativeChatPermission, - nativeChatQuestion, - nativeChatAsk: showNativeChatAsk ? nativeChatAskPrompt : null, + nativeChatPermission: activeChatStructured + ? structuredNativeChat.permission + : legacyNativeChatPermission, + nativeChatQuestion: activeChatStructured + ? structuredNativeChat.question + : legacyNativeChatQuestion, + nativeChatAsk: !activeChatStructured && showNativeChatAsk ? nativeChatAskPrompt : null, nativeChatAskKey, dismissNativeChatAsk, handleNativeChatAnswerAsk: answerAsk, handleNativeChatCancelAsk: cancelAsk, handleNativeChatRespondPermission: respond, - handleNativeChatStop, + handleNativeChatStop: activeChatStructured ? structuredNativeChat.cancel : handleNativeChatStop, nativeChatFilePaths, loadNativeChatFiles, - handleNativeChatQuestionAnswer, - handleNativeChatSend, - handleNativeChatSendWithOutcome, + handleNativeChatQuestionAnswer: activeChatStructured + ? structuredNativeChat.respondQuestion + : legacyHandleNativeChatQuestionAnswer, + handleNativeChatSend: activeChatStructured + ? structuredNativeChatSend.send + : handleNativeChatSend, + handleNativeChatSendWithOutcome: activeChatStructured + ? structuredNativeChatSend.sendWithOutcome + : handleNativeChatSendWithOutcome, readSeededLaunchDraft, - nativeChatSessionOptions: - sessionOptions.snapshot.length > 0 - ? { controller: sessionOptions, isWorking: nativeChatAgentWorking } - : null + nativeChatSessionOptions } } diff --git a/mobile/src/session/use-mobile-native-chat-image-attachments.ts b/mobile/src/session/use-mobile-native-chat-image-attachments.ts index dbcb97df527..77d839adf34 100644 --- a/mobile/src/session/use-mobile-native-chat-image-attachments.ts +++ b/mobile/src/session/use-mobile-native-chat-image-attachments.ts @@ -1,18 +1,17 @@ import { useCallback, useRef, useState } from 'react' -import { CLIPBOARD_IMAGE_TOO_LARGE_ERROR } from '../../../src/shared/clipboard-image' import { buildAgentTuiClearInputForText } from '../../../src/shared/agent-tui-input-clear' import type { RpcClient } from '../transport/rpc-client' import type { ConnectionState } from '../transport/types' -import { - ImageLibraryPermissionError, - pickMobileImages, - type MobileImageSource -} from './mobile-image-source-picker' +import type { MobileImageSource } from './mobile-image-source-picker' import { appendPendingNativeChatImages, - uploadMobileNativeChatImages, type PendingNativeChatImage } from './mobile-native-chat-image-attachment' +import { + NO_NATIVE_CHAT_IMAGE_ATTACHMENTS, + withScopeAttachments, + type MobileNativeChatImagesByScope +} from './mobile-native-chat-image-scope-state' import { MOBILE_NATIVE_CHAT_IMAGE_SETTLE_MS, pasteMobileNativeChatImagePaths @@ -31,6 +30,7 @@ import { acquireMobileNativeChatTerminalWrite, releaseMobileNativeChatTerminalWrite } from './mobile-native-chat-terminal-write-lock' +import { useMobileNativeChatImageUpload } from './use-mobile-native-chat-image-upload' type CurrentRef = { readonly current: T } type ShowToast = (message: string, durationMs?: number) => void @@ -60,8 +60,11 @@ type Args = { readonly baseSend: ( text: string, imagePreviewUris?: string[], - deadline?: number + deadline?: number, + attachments?: readonly PendingNativeChatImage[] ) => Promise + /** Structured sessions send attachments without the terminal paste path. */ + readonly structuredNativeChat: boolean /** Launch-context text parked on the agent's TUI input line, or null. The * paste's leading clear must cover every line of it, or the draft's earlier * lines survive and ride along with the image. */ @@ -83,21 +86,6 @@ export type MobileNativeChatImageAttachments = { readonly sendNativeChat: (text: string) => Promise } -const NO_ATTACHMENTS: PendingNativeChatImage[] = [] - -function withScopeAttachments( - byScope: Record, - scope: string, - next: PendingNativeChatImage[] -): Record { - if (next.length > 0) { - return { ...byScope, [scope]: next } - } - const remaining = { ...byScope } - delete remaining[scope] - return remaining -} - const defaultSleep = (ms: number): Promise => new Promise((resolve) => setTimeout(resolve, ms)) @@ -112,98 +100,40 @@ export function useMobileNativeChatImageAttachments({ showToast, onSendError, baseSend, + structuredNativeChat, readSeededLaunchDraft, onAttachSuccess, onError, sleep = defaultSleep }: Args): MobileNativeChatImageAttachments { - const [attachmentsByScope, setAttachmentsByScope] = useState< - Record - >({}) - const [isAttaching, setIsAttaching] = useState(false) + const [attachmentsByScope, setAttachmentsByScope] = useState({}) const idCounter = useRef(0) - // Count in-flight uploads so an overlapping attach can't clear the flag early. - const attachingCount = useRef(0) - // Live connState for attachImage's catch: the closure's value was already - // checked 'connected' at entry, so only a ref can see a mid-upload disconnect. - const connStateRef = useRef(connState) - connStateRef.current = connState + const attachments = + (scopeKey ? attachmentsByScope[scopeKey] : undefined) ?? NO_NATIVE_CHAT_IMAGE_ATTACHMENTS - const attachments = (scopeKey ? attachmentsByScope[scopeKey] : undefined) ?? NO_ATTACHMENTS - - const attachImage = useCallback( - async (source: MobileImageSource): Promise => { - // The chip lands in the scope that initiated the pick, even if the user - // switches tabs while the upload is in flight. - const scope = scopeKey - if (!client || !scope || !activeHandleRef.current || connState !== 'connected') { - return - } - // Only this call's own increment may be undone in `finally`; a cancelled - // pick or pre-upload error never ran `onUploadStart`, so decrementing the - // shared counter would clear a concurrent upload's in-flight flag early. - let started = false - const uploadedImages: Omit[] = [] - let uploadError: unknown = null - try { - await uploadMobileNativeChatImages(source, { - client, - getConnectionId: getActiveWorktreeConnectionId, - pickImages: pickMobileImages, - onImageUploaded: (image) => uploadedImages.push(image), - onUploadStart: () => { - started = true - attachingCount.current += 1 - setIsAttaching(true) - } - }) - } catch (error) { - uploadError = error - } finally { - if (started) { - attachingCount.current -= 1 - if (attachingCount.current === 0) { - setIsAttaching(false) - } - } - } - if (uploadedImages.length > 0) { - setAttachmentsByScope((prev) => ({ - ...prev, - [scope]: appendPendingNativeChatImages(prev[scope] ?? [], uploadedImages, idCounter) - })) - onAttachSuccess?.() - } - if (uploadError !== null) { - const message = uploadError instanceof Error ? uploadError.message : String(uploadError) - onError?.() - if (connStateRef.current !== 'connected') { - showToast('Attach failed (disconnected)', 1500) - return - } - if (uploadError instanceof ImageLibraryPermissionError) { - showToast('Photo permission denied', 1500) - return - } - if (message === CLIPBOARD_IMAGE_TOO_LARGE_ERROR) { - showToast('Image too large to attach', 1500) - return - } - showToast('Attach failed', 1500) - } + const addUploadedImages = useCallback( + (scope: string, uploadedImages: Omit[]) => { + setAttachmentsByScope((prev) => ({ + ...prev, + [scope]: appendPendingNativeChatImages(prev[scope] ?? [], uploadedImages, idCounter) + })) }, - [ - activeHandleRef, - client, - connState, - getActiveWorktreeConnectionId, - onAttachSuccess, - onError, - scopeKey, - showToast - ] + [] ) + const { attachImage, isAttaching } = useMobileNativeChatImageUpload({ + client, + activeHandleRef, + getActiveWorktreeConnectionId, + connState, + scopeKey, + structuredNativeChat, + showToast, + onImagesUploaded: addUploadedImages, + onAttachSuccess, + onError + }) + const removeAttachment = useCallback( (id: string): void => { const scope = scopeKey @@ -238,7 +168,32 @@ export function useMobileNativeChatImageAttachments({ const deadline = openMobileNativeChatSendBudget() try { const scope = scopeKey - const pendingImages = (scope ? attachmentsByScope[scope] : undefined) ?? NO_ATTACHMENTS + const pendingImages = + (scope ? attachmentsByScope[scope] : undefined) ?? NO_NATIVE_CHAT_IMAGE_ATTACHMENTS + if (structuredNativeChat && pendingImages.length > 0 && scope) { + if (!client || !enabled || connState !== 'connected') { + onError?.() + onSendError('Message not sent (disconnected)') + return false + } + const outcome = await baseSend( + text, + pendingImages.map((attachment) => attachment.previewUri), + deadline, + pendingImages + ) + if (outcome !== 'rejected') { + const sentIds = new Set(pendingImages.map((attachment) => attachment.id)) + setAttachmentsByScope((prev) => + withScopeAttachments( + prev, + scope, + (prev[scope] ?? []).filter((attachment) => !sentIds.has(attachment.id)) + ) + ) + } + return outcome !== 'rejected' + } if (pendingImages.length === 0 || !scope) { // Heal a previously failed paste: a text-only send to that terminal would // otherwise glue the stale image paste onto this message. Best-effort — diff --git a/mobile/src/session/use-mobile-native-chat-image-upload.ts b/mobile/src/session/use-mobile-native-chat-image-upload.ts new file mode 100644 index 00000000000..01567b5c727 --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-image-upload.ts @@ -0,0 +1,126 @@ +import { useCallback, useLayoutEffect, useRef, useState } from 'react' +import { CLIPBOARD_IMAGE_TOO_LARGE_ERROR } from '../../../src/shared/clipboard-image' +import type { RpcClient } from '../transport/rpc-client' +import type { ConnectionState } from '../transport/types' +import { + ImageLibraryPermissionError, + pickMobileImages, + type MobileImageSource +} from './mobile-image-source-picker' +import { + uploadMobileNativeChatImages, + type PendingNativeChatImage +} from './mobile-native-chat-image-attachment' + +type CurrentRef = { readonly current: T } +type UploadedNativeChatImage = Omit +type ShowToast = (message: string, durationMs?: number) => void + +export function useMobileNativeChatImageUpload(args: { + client: RpcClient | null + activeHandleRef: CurrentRef + getActiveWorktreeConnectionId: () => Promise + connState: ConnectionState + scopeKey: string | null + structuredNativeChat: boolean + showToast: ShowToast + onImagesUploaded: (scope: string, images: UploadedNativeChatImage[]) => void + onAttachSuccess?: () => void + onError?: () => void +}): { + attachImage: (source: MobileImageSource) => Promise + isAttaching: boolean +} { + const { + activeHandleRef, + client, + connState, + getActiveWorktreeConnectionId, + onAttachSuccess, + onError, + onImagesUploaded, + scopeKey, + showToast, + structuredNativeChat + } = args + const [isAttaching, setIsAttaching] = useState(false) + const attachingCount = useRef(0) + const connStateRef = useRef(connState) + useLayoutEffect(() => { + connStateRef.current = connState + }, [connState]) + + const attachImage = useCallback( + async (source: MobileImageSource): Promise => { + const scope = scopeKey + if ( + !client || + !scope || + connState !== 'connected' || + (!activeHandleRef.current && !structuredNativeChat) + ) { + return + } + let started = false + const uploadedImages: UploadedNativeChatImage[] = [] + let uploadError: unknown = null + try { + await uploadMobileNativeChatImages(source, { + client, + getConnectionId: getActiveWorktreeConnectionId, + pickImages: pickMobileImages, + onImageUploaded: (image) => uploadedImages.push(image), + onUploadStart: () => { + started = true + attachingCount.current += 1 + setIsAttaching(true) + } + }) + } catch (error) { + uploadError = error + } finally { + if (started) { + attachingCount.current -= 1 + if (attachingCount.current === 0) { + setIsAttaching(false) + } + } + } + if (uploadedImages.length > 0) { + onImagesUploaded(scope, uploadedImages) + onAttachSuccess?.() + } + if (uploadError !== null) { + const message = uploadError instanceof Error ? uploadError.message : String(uploadError) + onError?.() + if (connStateRef.current !== 'connected') { + showToast('Attach failed (disconnected)', 1500) + return + } + if (uploadError instanceof ImageLibraryPermissionError) { + showToast('Photo permission denied', 1500) + return + } + if (message === CLIPBOARD_IMAGE_TOO_LARGE_ERROR) { + showToast('Image too large to attach', 1500) + return + } + showToast('Attach failed', 1500) + } + }, + [ + activeHandleRef, + client, + connState, + getActiveWorktreeConnectionId, + onAttachSuccess, + onError, + onImagesUploaded, + scopeKey, + showToast, + structuredNativeChat + ] + ) + + return { attachImage, isAttaching } +} diff --git a/mobile/src/session/use-mobile-native-chat-session-option-controller.ts b/mobile/src/session/use-mobile-native-chat-session-option-controller.ts new file mode 100644 index 00000000000..aa61bdd85ff --- /dev/null +++ b/mobile/src/session/use-mobile-native-chat-session-option-controller.ts @@ -0,0 +1,100 @@ +import { useCallback, useMemo } from 'react' +import type { + SessionOptionDescriptor, + SessionOptionValue +} from '../../../src/shared/native-chat-session-options' +import { mobileNativeChatScopeKey } from './mobile-native-chat-scope-key' +import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' +import type { MobileNativeChatSessionOptionPickersProps } from './MobileNativeChatSessionOptionPickers' +import { + useMobileNativeChatSessionOptions, + type MobileNativeChatSessionOptionsController +} from './use-mobile-native-chat-session-options' + +export function useMobileNativeChatSessionOptionController(args: { + activeChatStructured: boolean + activeSessionTabId: string | null + agent: string | null + dispatchCommand: (text: string) => Promise + hostId: string + isTabChatView: (tabId: string) => boolean + isWorking: boolean + reportedModel: string | null + structured: { + snapshot: SessionOptionDescriptor[] + pendingId: string | null + setOption: (id: string, value: SessionOptionValue) => Promise + invokeAction: (id: string) => Promise + } + toggleTabChatView: (tabId: string) => void + worktreeId: string +}): { + nativeChatSessionOptions: MobileNativeChatSessionOptionPickersProps | null + recordCommand: (command: string) => void +} { + const { + activeChatStructured, + activeSessionTabId, + agent, + dispatchCommand, + hostId, + isTabChatView, + isWorking, + reportedModel, + structured, + toggleTabChatView, + worktreeId + } = args + const { + invokeAction: invokeStructuredAction, + pendingId: structuredPendingId, + setOption: setStructuredOption, + snapshot: structuredSnapshot + } = structured + + const handleAgentPicker = useCallback(() => { + if (activeSessionTabId && isTabChatView(activeSessionTabId)) { + toggleTabChatView(activeSessionTabId) + } + }, [activeSessionTabId, isTabChatView, toggleTabChatView]) + + const sessionOptions = useMobileNativeChatSessionOptions({ + agent: activeChatStructured ? null : agent, + scopeKey: mobileNativeChatScopeKey(hostId, worktreeId, activeSessionTabId), + reportedModel, + dispatchCommand, + onAgentPicker: handleAgentPicker + }) + const structuredController = useMemo( + () => + activeChatStructured && structuredSnapshot.length > 0 + ? { + snapshot: structuredSnapshot, + pendingId: structuredPendingId, + setOption: setStructuredOption, + invokeAction: invokeStructuredAction, + recordCommand: () => {} + } + : null, + [ + activeChatStructured, + invokeStructuredAction, + setStructuredOption, + structuredPendingId, + structuredSnapshot + ] + ) + const nativeChatSessionOptions = useMemo( + () => + activeChatStructured + ? structuredController + ? { controller: structuredController, isWorking } + : null + : sessionOptions.snapshot.length > 0 + ? { controller: sessionOptions, isWorking } + : null, + [activeChatStructured, isWorking, sessionOptions, structuredController] + ) + + return { nativeChatSessionOptions, recordCommand: sessionOptions.recordCommand } +} diff --git a/mobile/src/session/use-mobile-session-attachments.ts b/mobile/src/session/use-mobile-session-attachments.ts index 66691545118..841d3984b8f 100644 --- a/mobile/src/session/use-mobile-session-attachments.ts +++ b/mobile/src/session/use-mobile-session-attachments.ts @@ -36,7 +36,8 @@ export function useMobileSessionAttachments(scope: MobileSessionAccessorySelecti nativeChatInputLeaseReady, nativeChatController, getActiveWorktreeConnectionId, - refreshCanPaste + refreshCanPaste, + activeSessionTab } = scope const handlePaste = useMobileTerminalPaste({ client, @@ -80,6 +81,7 @@ export function useMobileSessionAttachments(scope: MobileSessionAccessorySelecti getActiveWorktreeConnectionId, beforeTerminalSend: flushPendingLiveInputBeforeAttachmentSend, nativeChatBaseSend: nativeChatController.handleNativeChatSendWithOutcome, + structuredNativeChat: activeSessionTab?.type === 'agent-session', readSeededLaunchDraft: nativeChatController.readSeededLaunchDraft, showToast, onNativeChatSendError: nativeChatSendError.show, diff --git a/mobile/src/session/use-mobile-session-file-actions.ts b/mobile/src/session/use-mobile-session-file-actions.ts index 7ba21770a21..56aa2b049d8 100644 --- a/mobile/src/session/use-mobile-session-file-actions.ts +++ b/mobile/src/session/use-mobile-session-file-actions.ts @@ -1,6 +1,7 @@ import { useRef, useCallback } from 'react' import { Linking } from 'react-native' import { useMobileFileTapHandlers } from './use-mobile-file-tap-handlers' +import { resolveMobileNativeChatFileSessionId } from './mobile-native-chat-eligibility' import { activateOpenedSourceControlDiffTab } from './opened-mobile-session-tab' import type { MobileSessionTab } from './mobile-session-route-types' import type { MobileSessionTerminalSendActionsModel } from './use-mobile-session-terminal-send-actions' @@ -31,10 +32,7 @@ export function useMobileSessionFileActions(scope: MobileSessionTerminalSendActi hostId, worktreeId, worktreeName: routeWorktreeName, - nativeChatSessionId: - activeSessionTab?.type === 'terminal' - ? (activeSessionTab.agentStatus?.providerSession?.id ?? null) - : null, + nativeChatSessionId: resolveMobileNativeChatFileSessionId(activeSessionTab), activeHandleRef, terminalCwdRef, openBrowser: (url) => void handleCreateBrowserRef.current?.(url), diff --git a/mobile/src/session/use-mobile-session-image-attachments.test.tsx b/mobile/src/session/use-mobile-session-image-attachments.test.tsx new file mode 100644 index 00000000000..68aeafaea6f --- /dev/null +++ b/mobile/src/session/use-mobile-session-image-attachments.test.tsx @@ -0,0 +1,123 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { RpcClient } from '../transport/rpc-client' +import { useMobileSessionImageAttachments } from './use-mobile-session-image-attachments' + +const mocks = vi.hoisted(() => ({ + useMobileImageAttachment: vi.fn(), + useMobileNativeChatImageAttachments: vi.fn() +})) + +vi.mock('./use-mobile-image-attachment', () => ({ + useMobileImageAttachment: mocks.useMobileImageAttachment +})) + +vi.mock('./use-mobile-native-chat-image-attachments', () => ({ + useMobileNativeChatImageAttachments: mocks.useMobileNativeChatImageAttachments +})) + +type HookArgs = Parameters[0] + +function baseArgs(overrides: Partial = {}): HookArgs { + return { + client: {} as RpcClient, + activeHandle: 'term-1', + activeHandleRef: { current: null }, + canSend: true, + connState: 'connected', + deviceTokenRef: { current: null }, + nativeChatScopeKey: 'scope-1', + nativeChatInputLeaseReady: false, + getActiveWorktreeConnectionId: async () => 'conn-1', + beforeTerminalSend: async () => true, + nativeChatBaseSend: vi.fn().mockResolvedValue('accepted'), + structuredNativeChat: true, + readSeededLaunchDraft: () => null, + showToast: vi.fn(), + onNativeChatSendError: vi.fn(), + onSuccess: vi.fn(), + onError: vi.fn(), + ...overrides + } +} + +describe('useMobileSessionImageAttachments', () => { + let renderer: ReactTestRenderer | null = null + + function Harness({ args }: { args: HookArgs }): null { + useMobileSessionImageAttachments(args) + return null + } + + beforeEach(() => { + mocks.useMobileImageAttachment.mockReturnValue({ + attachImage: vi.fn(), + isAttaching: false + }) + mocks.useMobileNativeChatImageAttachments.mockReturnValue({ + attachments: [], + isAttaching: false, + attachImage: vi.fn(), + removeAttachment: vi.fn(), + sendNativeChat: vi.fn() + }) + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + vi.clearAllMocks() + }) + + function render(args: HookArgs): void { + act(() => { + renderer = create(createElement(Harness, { args })) + }) + } + + it('enables native-chat image sends for connected structured sessions without a terminal lease', () => { + render(baseArgs()) + + expect(mocks.useMobileNativeChatImageAttachments).toHaveBeenCalledWith( + expect.objectContaining({ + enabled: true, + structuredNativeChat: true + }) + ) + }) + + it('keeps terminal-backed native-chat image sends gated on the input lease', () => { + render( + baseArgs({ + activeHandleRef: { current: 'term-1' }, + nativeChatInputLeaseReady: false, + structuredNativeChat: false + }) + ) + + expect(mocks.useMobileNativeChatImageAttachments).toHaveBeenCalledWith( + expect.objectContaining({ + enabled: false, + structuredNativeChat: false + }) + ) + }) + + it('disables structured native-chat image sends while disconnected', () => { + render( + baseArgs({ + connState: 'connecting', + nativeChatInputLeaseReady: true, + structuredNativeChat: true + }) + ) + + expect(mocks.useMobileNativeChatImageAttachments).toHaveBeenCalledWith( + expect.objectContaining({ + enabled: false, + structuredNativeChat: true + }) + ) + }) +}) diff --git a/mobile/src/session/use-mobile-session-image-attachments.ts b/mobile/src/session/use-mobile-session-image-attachments.ts index 9b5a010df9f..07edac51f39 100644 --- a/mobile/src/session/use-mobile-session-image-attachments.ts +++ b/mobile/src/session/use-mobile-session-image-attachments.ts @@ -29,8 +29,15 @@ type Args = { readonly nativeChatBaseSend: ( text: string, images?: string[], - deadline?: number + deadline?: number, + attachments?: readonly { + id: string + path: string + previewUri: string + }[] ) => Promise + /** Structured agent sessions do not have a terminal paste path. */ + readonly structuredNativeChat: boolean /** Launch-context text parked on the agent's TUI input line, or null — sizes * the image paste's leading clear so a multi-line draft cannot ride along. */ readonly readSeededLaunchDraft: () => string | null @@ -57,6 +64,7 @@ export function useMobileSessionImageAttachments({ getActiveWorktreeConnectionId, beforeTerminalSend, nativeChatBaseSend, + structuredNativeChat, readSeededLaunchDraft, showToast, onNativeChatSendError, @@ -86,7 +94,8 @@ export function useMobileSessionImageAttachments({ getActiveWorktreeConnectionId, connState, scopeKey: nativeChatScopeKey, - enabled: nativeChatInputLeaseReady, + enabled: structuredNativeChat ? connState === 'connected' : nativeChatInputLeaseReady, + structuredNativeChat, showToast, onSendError: onNativeChatSendError, baseSend: nativeChatBaseSend, diff --git a/mobile/src/session/use-mobile-session-native-chat-dictation.ts b/mobile/src/session/use-mobile-session-native-chat-dictation.ts index 7942233dbfd..6046cba1059 100644 --- a/mobile/src/session/use-mobile-session-native-chat-dictation.ts +++ b/mobile/src/session/use-mobile-session-native-chat-dictation.ts @@ -77,6 +77,12 @@ export function useMobileSessionNativeChatDictation( }) const { toggleTabChatView, showNativeChat, showNativeChatRef } = nativeChatController nativeChatSendError.bannerMountedRef.current = showNativeChat + const nativeChatOverlayInputLockReason = + activeSessionTab?.type === 'agent-session' + ? connState === 'connected' + ? null + : 'disconnected' + : nativeChatInputLockReason const routeKey = nativeChatScopeKey ?? `${hostId}\0${worktreeId}` const getSendCompletionGeneration = useMobileSendCompletionGeneration({ onBlur: resetLiveInputFocus, @@ -211,6 +217,7 @@ export function useMobileSessionNativeChatDictation( nativeChatInputLeaseReady, nativeChatInputLeaseReadyRef, nativeChatInputLockReason, + nativeChatOverlayInputLockReason, markNativeChatInputLeaseReady, clearNativeChatInputLease, nativeChatController, diff --git a/mobile/src/session/use-mobile-session-screen-state.ts b/mobile/src/session/use-mobile-session-screen-state.ts index 107ac3181de..6e2f82124b3 100644 --- a/mobile/src/session/use-mobile-session-screen-state.ts +++ b/mobile/src/session/use-mobile-session-screen-state.ts @@ -23,6 +23,7 @@ import type { MobileSessionTab, Terminal } from './mobile-session-route-types' +import { useMobileSessionTabActionTargets } from './use-mobile-session-tab-action-targets' import type { MobileSessionFoundationModel } from './use-mobile-session-foundation' export function useMobileSessionScreenState(scope: MobileSessionFoundationModel) { @@ -90,19 +91,7 @@ export function useMobileSessionScreenState(scope: MobileSessionFoundationModel) const [createTabAgentOptions, setCreateTabAgentOptions] = useState([]) const [showCreateBrowserModal, setShowCreateBrowserModal] = useState(false) const [showHeaderMoreActions, setShowHeaderMoreActions] = useState(false) - const [actionTarget, setActionTarget] = useState(null) - const [markdownActionTarget, setMarkdownActionTarget] = useState | null>(null) - const [fileActionTarget, setFileActionTarget] = useState | null>(null) - const [browserActionTarget, setBrowserActionTarget] = useState | null>(null) + const sessionTabActionTargets = useMobileSessionTabActionTargets() const [discardMarkdownTarget, setDiscardMarkdownTarget] = useState +type FileTab = Extract +type BrowserTab = Extract +type AgentSessionTab = Extract +type SetActionTarget = Dispatch> + +export function useMobileSessionTabActionTargets() { + const [actionTarget, setActionTarget] = useState(null) + const [markdownActionTarget, setMarkdownActionTarget] = useState(null) + const [fileActionTarget, setFileActionTarget] = useState(null) + const [browserActionTarget, setBrowserActionTarget] = useState(null) + const [agentSessionActionTarget, setAgentSessionActionTarget] = useState( + null + ) + + return { + actionTarget, + agentSessionActionTarget, + browserActionTarget, + fileActionTarget, + markdownActionTarget, + setActionTarget, + setAgentSessionActionTarget, + setBrowserActionTarget, + setFileActionTarget, + setMarkdownActionTarget + } +} + +export function useMobileSessionTabActionSheetOpener(args: { + activeHandleRef: MutableRefObject + setActionTarget: SetActionTarget + setMarkdownActionTarget: SetActionTarget + setFileActionTarget: SetActionTarget + setBrowserActionTarget: SetActionTarget + setAgentSessionActionTarget: SetActionTarget +}): (tab: MobileSessionTab) => void { + const { + activeHandleRef, + setActionTarget, + setAgentSessionActionTarget, + setBrowserActionTarget, + setFileActionTarget, + setMarkdownActionTarget + } = args + return useCallback( + (tab: MobileSessionTab) => { + if (tab.type === 'terminal') { + if (typeof tab.terminal !== 'string') { + return + } + setActionTarget({ + handle: tab.terminal, + title: tab.title, + isActive: tab.terminal === activeHandleRef.current + }) + } else if (tab.type === 'markdown') { + setMarkdownActionTarget(tab) + } else if (tab.type === 'file') { + setFileActionTarget(tab) + } else if (tab.type === 'agent-session') { + setAgentSessionActionTarget(tab) + } else { + setBrowserActionTarget(tab) + } + }, + [ + activeHandleRef, + setActionTarget, + setAgentSessionActionTarget, + setBrowserActionTarget, + setFileActionTarget, + setMarkdownActionTarget + ] + ) +} diff --git a/mobile/src/session/use-mobile-session-tab-switching.ts b/mobile/src/session/use-mobile-session-tab-switching.ts index 21095b613dd..48c48c2f994 100644 --- a/mobile/src/session/use-mobile-session-tab-switching.ts +++ b/mobile/src/session/use-mobile-session-tab-switching.ts @@ -136,6 +136,9 @@ export function useMobileSessionTabSwitching(scope: MobileSessionKeyboardStateMo void readFileTab(tab) return } + if (tab.type === 'agent-session') { + return + } const cached = markdownDocs.get(tab.id) if (cached?.status === 'ready' && cached.isDirty) { return diff --git a/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts new file mode 100644 index 00000000000..c0a4e8368c5 --- /dev/null +++ b/mobile/src/session/use-mobile-session-terminal-create-actions.test.ts @@ -0,0 +1,231 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RpcClient } from '../transport/rpc-client' +import { markRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' +import { useMobileSessionTerminalCreateActions } from './use-mobile-session-terminal-create-actions' + +vi.mock('../platform/haptics', () => ({ + triggerSuccess: vi.fn(), + triggerError: vi.fn() +})) + +function clientReturning(...responses: unknown[]): RpcClient { + let responseIndex = 0 + return { + sendRequest: vi.fn(async () => responses[responseIndex++]) + } as unknown as RpcClient +} + +function terminalCreateResponse() { + return { + ok: true, + result: { + tab: { + type: 'terminal', + id: 'terminal-tab-1', + title: 'Codex', + terminal: 'terminal-1', + isActive: true + } + } + } +} + +function createScope(client: RpcClient) { + return { + worktreeId: 'workspace-1', + client, + connState: 'connected', + setTerminals: vi.fn(), + terminalsRef: { current: [] }, + setSessionTabs: vi.fn(), + defaultTerminalHandlesToLiveInput: vi.fn(), + setActiveHandle: vi.fn(), + activeSessionTabId: 'existing-tab', + activeSessionTabIdRef: { current: 'existing-tab' }, + setActiveSessionTabId: vi.fn(), + setCreating: vi.fn(), + creatingTerminalRef: { current: false }, + creatingBrowser: false, + creatingMarkdown: false, + setCreateError: vi.fn(), + deviceTokenRef: { current: null }, + initializedHandlesRef: { current: new Set() }, + activeHandleRef: { current: 'existing-terminal' }, + activeSessionTabTypeRef: { current: 'terminal' }, + pendingActiveSessionTabIdRef: { current: null }, + pendingActiveTerminalHandleRef: { current: null }, + scheduleDelayedAction: vi.fn(), + showToast: vi.fn(), + unsubscribeTerminal: vi.fn(), + subscribeToTerminal: vi.fn(), + fetchSessionTabs: vi.fn(async () => {}) + } +} + +describe('mobile + Codex tab creation routing', () => { + let renderer: ReactTestRenderer | undefined + afterEach(() => renderer?.unmount()) + + it('uses the structured agent-session path for a bare Codex launch', async () => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: true, + value: { sessionId: 'codex_session_1' } + } + } + ) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + expect(client.sendRequest).toHaveBeenNthCalledWith(1, 'agentSession.createSupport', { + worktree: 'id:workspace-1', + agent: 'codex' + }) + expect(client.sendRequest).toHaveBeenNthCalledWith( + 2, + 'agentSession.create', + expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }), + expect.anything() + ) + expect(client.sendRequest).not.toHaveBeenCalledWith( + 'session.tabs.createTerminal', + expect.anything() + ) + expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('agent-session:codex_session_1') + expect(scope.setActiveHandle).toHaveBeenCalledWith(null) + expect(scope.unsubscribeTerminal).toHaveBeenCalledWith('existing-terminal') + }) + + it('keeps the legacy terminal path when structured support is disabled', async () => { + const client = clientReturning( + { ok: false, error: { code: 'structured_agent_session_unsupported', message: 'off' } }, + terminalCreateResponse() + ) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + expect(client.sendRequest).toHaveBeenNthCalledWith( + 2, + 'session.tabs.createTerminal', + expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }) + ) + expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('terminal-tab-1') + }) + + it('falls back to a terminal when structured creation is refused', async () => { + const client = clientReturning( + { ok: true, result: { supported: true } }, + { + ok: true, + result: { + ok: false, + refusal: { code: 'agent_session_ownership_unknown', message: 'provider unavailable' } + } + }, + terminalCreateResponse() + ) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + expect(client.sendRequest).toHaveBeenNthCalledWith( + 3, + 'session.tabs.createTerminal', + expect.objectContaining({ worktree: 'id:workspace-1', agent: 'codex' }) + ) + expect(scope.setActiveSessionTabId).toHaveBeenCalledWith('terminal-tab-1') + }) + + it('keeps prompted Codex launches on the legacy terminal path', async () => { + const client = clientReturning(terminalCreateResponse(), { + ok: true, + result: { send: { accepted: true } } + }) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex', { initialPrompt: 'Inspect this diff' }) + }) + + expect(client.sendRequest).toHaveBeenCalledWith( + 'session.tabs.createTerminal', + expect.objectContaining({ agent: 'codex' }) + ) + expect(client.sendRequest).not.toHaveBeenCalledWith( + 'agentSession.createSupport', + expect.anything() + ) + }) + + it('does not create a legacy sibling after an unknown structured outcome', async () => { + const client = clientReturning({ ok: true, result: { supported: true } }) + const sendRequest = client.sendRequest as unknown as ReturnType + sendRequest.mockImplementationOnce(async () => ({ + ok: true, + result: { supported: true } + })) + sendRequest.mockRejectedValueOnce(markRpcDeliveryUnknown(new Error('response lost'))) + sendRequest.mockRejectedValueOnce(markRpcDeliveryUnknown(new Error('still unknown'))) + const scope = createScope(client) + let actions: ReturnType | undefined + function Harness() { + actions = useMobileSessionTerminalCreateActions(scope as never) + return null + } + await act(async () => { + renderer = create(createElement(Harness)) + }) + await act(async () => { + await actions?.handleCreateTerminal('codex') + }) + + expect(sendRequest.mock.calls.map(([method]) => method)).toEqual([ + 'agentSession.createSupport', + 'agentSession.create', + 'agentSession.create' + ]) + expect(scope.setCreateError).toHaveBeenCalledWith('still unknown') + expect(scope.showToast).toHaveBeenCalledWith('still unknown', 1800) + }) +}) diff --git a/mobile/src/session/use-mobile-session-terminal-create-actions.ts b/mobile/src/session/use-mobile-session-terminal-create-actions.ts index 895b9a5a256..0ccd3591011 100644 --- a/mobile/src/session/use-mobile-session-terminal-create-actions.ts +++ b/mobile/src/session/use-mobile-session-terminal-create-actions.ts @@ -10,6 +10,7 @@ import type { MobileNewTabAgentOption } from './mobile-new-tab-agent-options' import type { TerminalQuickCommand } from '../../../src/shared/terminal-quick-command-types' import type { Terminal, TerminalCreateResult } from './mobile-session-route-types' import type { MobileSessionAttachmentsModel } from './use-mobile-session-attachments' +import { createMobileStructuredCodexSession } from './mobile-structured-agent-session-launch' export function useMobileSessionTerminalCreateActions(scope: MobileSessionAttachmentsModel) { const { @@ -22,6 +23,7 @@ export function useMobileSessionTerminalCreateActions(scope: MobileSessionAttach defaultTerminalHandlesToLiveInput, setActiveHandle, activeSessionTabId, + activeSessionTabIdRef, setActiveSessionTabId, setCreating, creatingTerminalRef, @@ -61,6 +63,35 @@ export function useMobileSessionTerminalCreateActions(scope: MobileSessionAttach .slice(2, 10)}` try { + // Bare Codex launches follow structured support; prompted launches keep their startup semantics. + if (agent === 'codex' && options === undefined) { + const structured = await createMobileStructuredCodexSession(client, worktreeId) + if (structured.kind === 'created') { + const previous = activeHandleRef.current + if (previous) { + unsubscribeTerminal(previous) + initializedHandlesRef.current.delete(previous) + } + const tabId = `agent-session:${structured.sessionId}` + pendingActiveSessionTabIdRef.current = tabId + pendingActiveTerminalHandleRef.current = null + activeSessionTabTypeRef.current = 'agent-session' + activeSessionTabIdRef.current = tabId + setActiveSessionTabId(tabId) + activeHandleRef.current = null + setActiveHandle(null) + // Refresh if the create response beats its published tab frame. + scheduleDelayedAction(() => void fetchSessionTabs(), 500) + return + } + if (structured.kind === 'unknown') { + // Never create a legacy sibling when the host may already have committed. + setCreateError(structured.message) + triggerError() + showToast(structured.message, 1800) + return + } + } const response = await client.sendRequest('session.tabs.createTerminal', { worktree: `id:${worktreeId}`, afterTabId: activeSessionTabId ?? undefined, diff --git a/mobile/src/session/use-mobile-session-terminal-send-actions.ts b/mobile/src/session/use-mobile-session-terminal-send-actions.ts index c80edee9271..6909f71ca63 100644 --- a/mobile/src/session/use-mobile-session-terminal-send-actions.ts +++ b/mobile/src/session/use-mobile-session-terminal-send-actions.ts @@ -16,6 +16,7 @@ import { import { normalizeTerminalTextInput } from '../terminal/terminal-text-input-normalization' import { useAgentSendKeyboardDismissal } from './use-agent-send-keyboard-dismissal' import type { MobileSessionTab } from './mobile-session-route-types' +import { useMobileSessionTabActionSheetOpener } from './use-mobile-session-tab-action-targets' import type { MobileSessionTerminalWebviewModel } from './use-mobile-session-terminal-webview' export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminalWebviewModel) { @@ -27,6 +28,7 @@ export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminal setMarkdownActionTarget, setFileActionTarget, setBrowserActionTarget, + setAgentSessionActionTarget, keyboardHeight, deviceTokenRef, clientRef, @@ -175,24 +177,14 @@ export function useMobileSessionTerminalSendActions(scope: MobileSessionTerminal sessionTabActionSheetKeyboardHideSubRef.current = null }, []) - const openSessionTabActionSheet = useCallback((tab: MobileSessionTab) => { - if (tab.type === 'terminal') { - if (typeof tab.terminal !== 'string') { - return - } - setActionTarget({ - handle: tab.terminal, - title: tab.title, - isActive: tab.terminal === activeHandleRef.current - }) - } else if (tab.type === 'markdown') { - setMarkdownActionTarget(tab) - } else if (tab.type === 'file') { - setFileActionTarget(tab) - } else { - setBrowserActionTarget(tab) - } - }, []) + const openSessionTabActionSheet = useMobileSessionTabActionSheetOpener({ + activeHandleRef, + setActionTarget, + setMarkdownActionTarget, + setFileActionTarget, + setBrowserActionTarget, + setAgentSessionActionTarget + }) const openSessionTabActionSheetAfterKeyboardDismiss = useCallback( (tab: MobileSessionTab) => { diff --git a/mobile/src/session/use-mobile-structured-agent-options.ts b/mobile/src/session/use-mobile-structured-agent-options.ts new file mode 100644 index 00000000000..108275223be --- /dev/null +++ b/mobile/src/session/use-mobile-structured-agent-options.ts @@ -0,0 +1,161 @@ +import { useCallback, useEffect, useMemo, useRef, useState } from 'react' +import { getAgentSessionOptionCatalog } from '../../../src/shared/agent-session-option-catalog' +import type { + AgentSessionOptionResult, + AgentSessionOptionsResult +} from '../../../src/shared/agent-session-wire' +import type { + SessionOptionDescriptor, + SessionOptionsSurface, + SessionOptionValue +} from '../../../src/shared/native-chat-session-options' +import { + applyStructuredAgentSessionOptions, + canSetStructuredAgentSessionOption, + commitStructuredAgentSessionOption, + commitStructuredAgentSessionOptionValues, + createStructuredAgentSessionOptionState, + structuredAgentSessionOptionSnapshot +} from '../../../src/shared/structured-agent-session-options' +import type { RpcClient } from '../transport/rpc-client' +import { + callAgentSession, + type StructuredAgentSessionMutate +} from './mobile-structured-agent-session-rpc' + +type StructuredOptionsController = { + optionSnapshot: SessionOptionDescriptor[] + optionSurface: SessionOptionsSurface + pendingOptionId: string | null + setStructuredOption: (id: string, value: SessionOptionValue) => Promise + invokeStructuredOption: (id: string) => Promise +} + +export function useMobileStructuredAgentOptions(args: { + agent: string | null + client: RpcClient | null + sessionId: string | null + enabled: boolean + fence: number | null + mutate: StructuredAgentSessionMutate +}): StructuredOptionsController { + const { agent, client, enabled, fence, mutate, sessionId } = args + const [optionState, setOptionState] = useState(() => + createStructuredAgentSessionOptionState(agent ?? 'codex') + ) + const activeOptionRecordRef = useRef(optionState.record) + const optionCatalog = useMemo( + () => (agent === 'claude' || agent === 'codex' ? getAgentSessionOptionCatalog(agent) : null), + [agent] + ) + + useEffect(() => { + const next = createStructuredAgentSessionOptionState(agent ?? 'codex') + activeOptionRecordRef.current = next.record + setOptionState(next) + }, [agent, enabled, fence, sessionId]) + + useEffect(() => { + if (!client || !sessionId || !enabled || !optionCatalog) { + return + } + let stale = false + void callAgentSession(client, 'agentSession.options', { sessionId }) + .then((result) => { + if (!stale) { + setOptionState((current) => + current.record === activeOptionRecordRef.current + ? applyStructuredAgentSessionOptions(current, optionCatalog, result) + : current + ) + } + }) + .catch(() => undefined) + return () => { + stale = true + } + }, [client, enabled, optionCatalog, sessionId, fence]) + + const optionSnapshot = useMemo( + () => structuredAgentSessionOptionSnapshot(optionState), + [optionState] + ) + + const setStructuredOption = useCallback( + async (id: string, value: SessionOptionValue): Promise => { + if ( + !canSetStructuredAgentSessionOption(optionState, id, value) || + typeof value !== 'string' + ) { + return false + } + const targetRecord = optionState.record + setOptionState((current) => ({ ...current, pendingId: id })) + try { + const result = await mutate( + 'agentSession.setOption', + 'agentSession.setOption', + { key: id, value } + ) + if (activeOptionRecordRef.current !== targetRecord) { + return result.status !== 'rejected' + } + if (result.status === 'accepted') { + setOptionState((current) => + current.record === targetRecord && result.sameFence + ? commitStructuredAgentSessionOptionValues( + current, + result.value.options ?? { [id]: value } + ) + : current + ) + return true + } + if (result.status === 'unknown') { + setOptionState((current) => + current.record === targetRecord + ? commitStructuredAgentSessionOption(current, id, value) + : current + ) + return true + } + return false + } finally { + setOptionState((current) => + current.record === targetRecord && current.pendingId === id + ? { ...current, pendingId: null } + : current + ) + } + }, + [mutate, optionState] + ) + + const invokeStructuredOption = useCallback(async () => false, []) + + const setOption = useCallback( + async (id: string, value: SessionOptionValue) => { + await setStructuredOption(id, value) + return { snapshot: optionSnapshot } + }, + [optionSnapshot, setStructuredOption] + ) + + const optionSurface = useMemo( + () => ({ + getSnapshot: () => optionSnapshot, + setOption, + invokeAction: async () => ({ snapshot: optionSnapshot }), + subscribe: () => () => {} + }), + [optionSnapshot, setOption] + ) + + return { + optionSnapshot, + optionSurface, + pendingOptionId: optionState.pendingId, + setStructuredOption, + invokeStructuredOption + } +} diff --git a/mobile/src/session/use-mobile-structured-agent-session.test.tsx b/mobile/src/session/use-mobile-structured-agent-session.test.tsx new file mode 100644 index 00000000000..83562839363 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-agent-session.test.tsx @@ -0,0 +1,849 @@ +import { createElement } from 'react' +import { act, create, type ReactTestRenderer } from 'react-test-renderer' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalRenderItem, + AgentJournalResolution +} from '../../../src/shared/agent-session-journal-types' +import type { AgentSessionSubscribeEvent } from '../../../src/shared/agent-session-wire' +import type { RpcClient } from '../transport/rpc-client' +import { markRpcDeliveryUnknown } from '../transport/rpc-delivery-ambiguity' +import { formatQuestionFreeTextAnswer } from './mobile-native-chat-question' +import { useMobileStructuredAgentSession } from './use-mobile-structured-agent-session' + +function ok(result: unknown) { + return { ok: true, result, _meta: { runtimeId: 'runtime-1' } } +} + +function snapshotEvent(fence = 3): AgentSessionSubscribeEvent { + return { + type: 'snapshot', + sessionId: 'session-1', + fence, + page: { + sessionId: 'session-1', + epoch: 'epoch-1', + fence, + direction: 'tail', + items: [], + removedItemIds: [], + submissions: [], + window: { + oldest: null, + newest: null, + nextCursor: { epoch: 'epoch-1', sequence: 0 } + }, + liveCursor: { epoch: 'epoch-1', sequence: 0 }, + hasOlder: false, + hasNewer: false + } + } +} + +function snapshotWithMessage(): AgentSessionSubscribeEvent { + const event = snapshotEvent() + return { + ...event, + page: { + ...event.page, + items: [ + { + itemId: 'msg-1', + revision: 1, + sequence: 1, + observedAt: 10, + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'sent before the blip' }] + } + } + ], + window: { + oldest: { epoch: 'epoch-1', sequence: 1 }, + newest: { epoch: 'epoch-1', sequence: 1 }, + nextCursor: { epoch: 'epoch-1', sequence: 2 } + }, + liveCursor: { epoch: 'epoch-1', sequence: 1 } + } + } as AgentSessionSubscribeEvent +} + +function pendingResolution(): AgentJournalResolution { + return { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null + } +} + +function approvalItem(): AgentJournalRenderItem { + return { + itemId: 'approval-1', + revision: 2, + sequence: 1, + observedAt: 10, + body: { + kind: 'approval', + title: 'Allow Bash?', + detail: 'rm -rf build', + options: [ + { id: 'allow-once', label: 'Allow once' }, + { id: 'deny', label: 'Deny' } + ], + resolution: pendingResolution() + } + } +} + +function approvalItemWithIdentity(itemId: string, revision: number): AgentJournalRenderItem { + return { ...approvalItem(), itemId, revision } +} + +function questionItem(): AgentJournalRenderItem { + return { + itemId: 'question-1', + revision: 7, + sequence: 2, + observedAt: 12, + body: { + kind: 'question', + question: 'Pick destination', + freeTextQuestionId: 'free-q', + options: [ + { id: 'choice-a', label: 'Choice A' }, + { id: 'choice-b', label: 'Choice B' } + ], + resolution: pendingResolution() + } + } +} + +function questionItemWithIdentity(itemId: string, revision: number): AgentJournalRenderItem { + return { ...questionItem(), itemId, revision } +} + +function runningStatusItem(): AgentJournalRenderItem { + return { + itemId: 'status-1', + revision: 1, + sequence: 3, + observedAt: 14, + body: { + kind: 'status', + text: 'Working', + turnLifecycle: { turnId: 'turn-1', state: 'running' } + } + } +} + +function defaultSendRequest(method: string, params?: Record) { + if (method === 'agentSession.send') { + return ok({ + ok: true, + replayed: false, + fence: 3, + cursor: { epoch: 'epoch-1', sequence: 1 }, + value: { turnId: 'turn-1' } + }) + } + if (method === 'agentSession.options') { + return ok({ + models: [ + { + id: 'gpt-fast', + label: 'GPT Fast', + isDefault: true, + defaultEffort: 'low', + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }, + { + id: 'gpt-slow', + label: 'GPT Slow', + isDefault: false, + defaultEffort: 'high', + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + } + ], + current: { + model: 'gpt-fast', + effort: 'low' + } + }) + } + if (method === 'agentSession.setOption') { + return ok({ + ok: true, + replayed: false, + fence: 3, + cursor: { epoch: 'epoch-1', sequence: 2 }, + value: { + key: 'model', + value: 'gpt-fast', + options: { model: 'gpt-fast' } + } + }) + } + if (method === 'agentSession.respondToApproval' || method === 'agentSession.respondToQuestion') { + return ok({ + ok: true, + replayed: false, + fence: 3, + cursor: { epoch: 'epoch-1', sequence: 3 }, + value: { + itemId: String(params?.itemId ?? ''), + revision: 2, + resolution: { + state: 'resolved', + selectedOptionId: String(params?.optionId ?? ''), + resolvedBy: 'mobile', + resolvedAt: 123 + } + } + }) + } + return ok({}) +} + +describe('useMobileStructuredAgentSession', () => { + let renderer: ReactTestRenderer | null = null + let hook: ReturnType | null = null + let listener: ((value: unknown) => void) | null = null + const onSendError = vi.fn() + const unsubscribe = vi.fn() + const sendRequest = vi.fn(defaultSendRequest) + const subscribe = vi.fn((_method: string, _params: unknown, onData: (value: unknown) => void) => { + listener = onData + return unsubscribe + }) + const client = { + sendRequest, + subscribe + } as unknown as RpcClient + + function Harness({ + sessionId = 'session-1', + agent = 'codex', + connected = true, + sourceIdentity = 'host-a\0workspace-a' + }: { + sessionId?: string | null + agent?: string | null + connected?: boolean + sourceIdentity?: string + }): null { + hook = useMobileStructuredAgentSession({ + client, + sessionId, + sourceIdentity, + enabled: true, + connected, + agent, + onSendError + } as never) + return null + } + + beforeEach(() => { + vi.clearAllMocks() + sendRequest.mockImplementation(defaultSendRequest) + listener = null + }) + + afterEach(() => { + act(() => renderer?.unmount()) + renderer = null + hook = null + }) + + it('subscribes and holds structured sessions without nativeChat or terminal RPCs', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + + await vi.waitFor(() => + expect(subscribe).toHaveBeenCalledWith( + 'agentSession.subscribe', + { sessionId: 'session-1' }, + expect.any(Function) + ) + ) + await vi.waitFor(() => + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.hold', + expect.objectContaining({ sessionId: 'session-1', holderId: expect.any(String) }), + expect.any(Object) + ) + ) + expect(sendRequest).not.toHaveBeenCalledWith( + expect.stringMatching(/^(nativeChat|terminal)\./), + expect.anything(), + expect.anything() + ) + }) + + it('re-holds after a reconnect that outlives the host release grace', async () => { + act(() => { + renderer = create(createElement(Harness, { connected: true })) + }) + await vi.waitFor(() => + expect( + sendRequest.mock.calls.filter(([method]) => method === 'agentSession.hold') + ).toHaveLength(1) + ) + await vi.waitFor(() => expect(subscribe).toHaveBeenCalledTimes(1)) + + // A transport loss retires the connection-scoped hold; after the host's 15s grace + // it may evict the provider child. Reconnect must acquire before replaying the stream. + await act(async () => { + renderer?.update(createElement(Harness, { connected: false })) + }) + expect(unsubscribe).toHaveBeenCalledTimes(1) + await act(async () => { + renderer?.update(createElement(Harness, { connected: true })) + }) + + await vi.waitFor(() => + expect( + sendRequest.mock.calls.filter(([method]) => method === 'agentSession.hold') + ).toHaveLength(2) + ) + await vi.waitFor(() => expect(subscribe).toHaveBeenCalledTimes(2)) + const holdOrders = sendRequest.mock.calls + .map((call, index) => + call[0] === 'agentSession.hold' ? sendRequest.mock.invocationCallOrder[index] : null + ) + .filter((order): order is number => order !== null) + const subscribeOrders = subscribe.mock.invocationCallOrder + const secondHoldOrder = holdOrders[1] + const secondSubscribeOrder = subscribeOrders[1] + if (secondHoldOrder === undefined || secondSubscribeOrder === undefined) { + throw new Error('reconnect calls were not recorded') + } + expect(secondHoldOrder).toBeLessThan(secondSubscribeOrder) + }) + + it('sends with the shared structured mutation envelope after the stream fence lands', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotEvent())) + + let outcome: 'accepted' | 'unknown' | 'rejected' = 'rejected' + await act(async () => { + outcome = await hook!.sendWithOutcome('hello') + }) + + expect(outcome).toBe('accepted') + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.send', + expect.objectContaining({ + envelope: expect.objectContaining({ + sessionId: 'session-1', + expectedRuntimeFence: 3, + clientOperationId: expect.stringMatching(/^\d{13}-[0-9a-f]{32}$/), + payloadFingerprint: expect.any(String) + }), + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'hello' }] + } + }), + expect.any(Object) + ) + }) + + it('surfaces structured prompt cards and option snapshots', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotEvent(3))) + act(() => listener?.(snapshotEvent(3))) + act(() => + listener?.({ + ...snapshotEvent(3), + page: { + ...snapshotEvent(3).page, + items: [approvalItem(), questionItem()] + } + }) + ) + + if (!hook) { + throw new Error('hook not ready') + } + + await vi.waitFor(() => expect(hook.permission).not.toBeNull()) + await vi.waitFor(() => expect(hook.question).not.toBeNull()) + await vi.waitFor(() => expect(hook.optionSnapshot.length).toBeGreaterThan(0)) + + expect(hook.permission).toMatchObject({ + title: 'Allow Bash?', + detail: 'rm -rf build', + options: [ + { label: 'Allow once', send: expect.any(String) }, + { label: 'Deny', send: expect.any(String) } + ] + }) + expect(hook.question).toMatchObject({ + question: 'Pick destination', + allowOther: true, + optionTokens: [expect.any(String), expect.any(String)], + freeTextToken: expect.any(String) + }) + expect(hook.optionSurface.getSnapshot()).toEqual(hook.optionSnapshot) + + await act(async () => { + expect(await hook.setStructuredOption('model', 'gpt-fast')).toBe(true) + }) + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.setOption', + expect.objectContaining({ + envelope: expect.objectContaining({ + sessionId: 'session-1', + expectedRuntimeFence: 3, + clientOperationId: expect.any(String), + payloadFingerprint: expect.any(String) + }), + key: 'model', + value: 'gpt-fast' + }), + expect.any(Object) + ) + + await act(async () => { + expect(await hook.respondPermission(hook.permission!.options[0]!.send)).toBe(true) + }) + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.respondToApproval', + expect.objectContaining({ + envelope: expect.objectContaining({ + sessionId: 'session-1', + expectedRuntimeFence: 3 + }), + itemId: 'approval-1', + optionId: 'allow-once' + }), + expect.any(Object) + ) + + await act(async () => { + expect( + await hook.respondQuestion(formatQuestionFreeTextAnswer(hook.question!, 'custom answer')) + ).toBe(true) + }) + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.respondToQuestion', + expect.objectContaining({ + envelope: expect.objectContaining({ + sessionId: 'session-1', + expectedRuntimeFence: 3 + }), + itemId: 'question-1', + optionId: `${encodeURIComponent('free-q')}:${encodeURIComponent('custom answer')}` + }), + expect.any(Object) + ) + }) + + it('sends structured image attachments in the message body', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotEvent(3))) + + let outcome: 'accepted' | 'unknown' | 'rejected' = 'rejected' + await act(async () => { + outcome = await hook.sendWithOutcome('look at this', undefined, undefined, [ + { path: '/tmp/a.png', previewUri: 'file:///a.jpg' } + ]) + }) + + expect(outcome).toBe('accepted') + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.send', + expect.objectContaining({ + envelope: expect.objectContaining({ + sessionId: 'session-1', + expectedRuntimeFence: 3, + clientOperationId: expect.any(String), + payloadFingerprint: expect.any(String) + }), + body: { + kind: 'message', + role: 'user', + blocks: [ + { type: 'text', text: 'look at this' }, + { type: 'image-ref', path: '/tmp/a.png' } + ] + } + }), + expect.any(Object) + ) + }) + + it('rejects preview-only structured image URIs instead of sending them as host paths', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotEvent(3))) + sendRequest.mockClear() + + let outcome: 'accepted' | 'unknown' | 'rejected' = 'accepted' + await act(async () => { + outcome = await hook!.sendWithOutcome('look at this', ['file:///a.jpg']) + }) + + expect(outcome).toBe('rejected') + expect(onSendError).toHaveBeenCalledWith('Message not sent') + expect(sendRequest).not.toHaveBeenCalledWith( + 'agentSession.send', + expect.objectContaining({ + body: expect.objectContaining({ + blocks: expect.arrayContaining([{ type: 'image-ref', path: 'file:///a.jpg' }]) + }) + }), + expect.any(Object) + ) + }) + + it('answers the prompt captured by a structured card after a newer prompt lands', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => + listener?.({ + ...snapshotEvent(3), + page: { + ...snapshotEvent(3).page, + items: [ + approvalItemWithIdentity('approval-old', 4), + questionItemWithIdentity('question-old', 8) + ] + } + }) + ) + const approvalToken = hook!.permission!.options[0]!.send + const questionToken = hook!.question!.optionTokens[0]! + const freeText = formatQuestionFreeTextAnswer(hook!.question!, 'old answer') + + act(() => + listener?.({ + ...snapshotEvent(3), + page: { + ...snapshotEvent(3).page, + items: [ + approvalItemWithIdentity('approval-new', 9), + questionItemWithIdentity('question-new', 10) + ] + } + }) + ) + sendRequest.mockClear() + + await act(async () => { + expect(await hook!.respondPermission(approvalToken)).toBe(true) + expect(await hook!.respondQuestion(questionToken)).toBe(true) + expect(await hook!.respondQuestion(freeText)).toBe(true) + }) + + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.respondToApproval', + expect.objectContaining({ + itemId: 'approval-old', + expectedRevision: 4, + optionId: 'allow-once' + }), + expect.any(Object) + ) + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.respondToQuestion', + expect.objectContaining({ + itemId: 'question-old', + expectedRevision: 8, + optionId: 'choice-a' + }), + expect.any(Object) + ) + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.respondToQuestion', + expect.objectContaining({ + itemId: 'question-old', + expectedRevision: 8, + optionId: `${encodeURIComponent('free-q')}:${encodeURIComponent('old answer')}` + }), + expect.any(Object) + ) + }) + + it('surfaces unknown structured prompt responses as unconfirmed', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => + listener?.({ + ...snapshotEvent(3), + page: { + ...snapshotEvent(3).page, + items: [approvalItem(), questionItem()] + } + }) + ) + sendRequest.mockImplementation(async (method, params) => { + if (method === 'agentSession.respondToApproval') { + throw markRpcDeliveryUnknown(new Error('Connection closed')) + } + return defaultSendRequest(method, params) + }) + onSendError.mockClear() + + await act(async () => { + expect(await hook!.respondPermission(hook!.permission!.options[0]!.send)).toBe(false) + }) + expect(onSendError).toHaveBeenCalledWith('Response unconfirmed — check chat before retrying') + + sendRequest.mockImplementation(async (method, params) => { + if (method === 'agentSession.respondToQuestion') { + throw markRpcDeliveryUnknown(new Error('Connection closed')) + } + return defaultSendRequest(method, params) + }) + onSendError.mockClear() + + await act(async () => { + expect(await hook!.respondQuestion(hook!.question!.optionTokens[0]!)).toBe(false) + }) + expect(onSendError).toHaveBeenCalledWith('Answer unconfirmed — check chat before retrying') + }) + + it('uses a fresh operation id when a prompt response delivery is unknown', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => + listener?.({ + ...snapshotEvent(3), + page: { ...snapshotEvent(3).page, items: [approvalItem()] } + }) + ) + let attempts = 0 + sendRequest.mockImplementation(async (method, params) => { + if (method === 'agentSession.respondToApproval' && attempts++ === 0) { + throw markRpcDeliveryUnknown(new Error('Connection closed')) + } + return defaultSendRequest(method, params) + }) + + const token = hook!.permission!.options[0]!.send + await act(async () => { + expect(await hook!.respondPermission(token)).toBe(false) + expect(await hook!.respondPermission(token)).toBe(true) + }) + + const calls = sendRequest.mock.calls.filter( + ([method]) => method === 'agentSession.respondToApproval' + ) + expect(calls).toHaveLength(2) + const firstId = (calls[0]![1] as { envelope: { clientOperationId: string } }).envelope + .clientOperationId + const retryId = (calls[1]![1] as { envelope: { clientOperationId: string } }).envelope + .clientOperationId + expect(firstId).toMatch(/^\d{13}-[0-9a-f]{32}$/) + expect(retryId).toMatch(/^\d{13}-[0-9a-f]{32}$/) + expect(retryId).not.toBe(firstId) + }) + + it('marks a retried send as retryUnknown after ambiguous delivery', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotEvent(3))) + let attempts = 0 + sendRequest.mockImplementation(async (method, params) => { + if (method === 'agentSession.send' && attempts++ === 0) { + throw markRpcDeliveryUnknown(new Error('Connection closed')) + } + return defaultSendRequest(method, params) + }) + + await act(async () => { + expect(await hook!.sendWithOutcome('retry me')).toBe('unknown') + expect(await hook!.sendWithOutcome('retry me')).toBe('accepted') + }) + + const calls = sendRequest.mock.calls.filter(([method]) => method === 'agentSession.send') + expect(calls).toHaveLength(2) + expect(calls[0]![1]).not.toHaveProperty('retryUnknown') + expect(calls[1]![1]).toMatchObject({ retryUnknown: true }) + const firstId = (calls[0]![1] as { envelope: { clientOperationId: string } }).envelope + .clientOperationId + const retryId = (calls[1]![1] as { envelope: { clientOperationId: string } }).envelope + .clientOperationId + expect(retryId).toBe(firstId) + }) + + it('keeps structured option changes dispatched after unknown delivery', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotEvent(3))) + await vi.waitFor(() => expect(hook!.optionSnapshot.length).toBeGreaterThan(0)) + sendRequest.mockImplementation(async (method, params) => { + if (method === 'agentSession.setOption') { + throw markRpcDeliveryUnknown(new Error('Connection closed')) + } + return defaultSendRequest(method, params) + }) + onSendError.mockClear() + + await act(async () => { + expect(await hook!.setStructuredOption('model', 'gpt-slow')).toBe(true) + }) + + const model = hook!.optionSnapshot.find((descriptor) => descriptor.id === 'model') + expect(model).toMatchObject({ + valueSource: 'dispatched', + kind: expect.objectContaining({ currentValue: 'gpt-slow' }) + }) + expect(onSendError).not.toHaveBeenCalled() + }) + + it('reports structured Stop as unconfirmed after unknown delivery', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => + listener?.({ + ...snapshotEvent(3), + page: { + ...snapshotEvent(3).page, + items: [runningStatusItem()] + } + }) + ) + sendRequest.mockImplementation(async (method, params) => { + if (method === 'agentSession.cancel') { + throw markRpcDeliveryUnknown(new Error('Connection closed')) + } + return defaultSendRequest(method, params) + }) + onSendError.mockClear() + + await act(async () => { + hook!.cancel() + await Promise.resolve() + }) + + expect(onSendError).toHaveBeenCalledWith('Stop unconfirmed — check chat before retrying') + }) + + it('releases a landed hold when the structured tab unmounts', async () => { + act(() => { + renderer = create(createElement(Harness)) + }) + await vi.waitFor(() => + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.hold', + expect.objectContaining({ sessionId: 'session-1' }), + expect.any(Object) + ) + ) + const held = sendRequest.mock.calls.find((call) => call[0] === 'agentSession.hold')?.[1] as { + holderId: string + } + + act(() => renderer?.unmount()) + + await vi.waitFor(() => + expect(sendRequest).toHaveBeenCalledWith( + 'agentSession.release', + { sessionId: 'session-1', holderId: held.holderId }, + expect.any(Object) + ) + ) + }) + + it('keeps the transcript visible while reconnecting', async () => { + await act(async () => { + renderer = create(createElement(Harness, { connected: true })) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotWithMessage())) + expect(hook?.session.messages).toHaveLength(1) + + await act(async () => { + renderer?.update(createElement(Harness, { connected: false })) + }) + expect(hook?.session.messages).toHaveLength(1) + expect(hook?.session.status).toBe('ready') + + await act(async () => { + renderer?.update(createElement(Harness, { connected: true })) + }) + expect(hook?.session.messages).toHaveLength(1) + }) + + it('restores the correct cached transcript when switching tabs offline', async () => { + await act(async () => { + renderer = create(createElement(Harness, { connected: true, sessionId: 'session-1' })) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotWithMessage())) + expect(hook?.session.messages).toHaveLength(1) + + await act(async () => { + renderer?.update(createElement(Harness, { connected: false, sessionId: 'session-2' })) + }) + expect(hook?.session.messages).toEqual([]) + expect(hook?.session.status).toBe('idle') + + await act(async () => { + renderer?.update(createElement(Harness, { connected: false, sessionId: 'session-1' })) + }) + expect(hook?.session.messages).toHaveLength(1) + }) + + it('isolates matching provider session ids across host and workspace sources', async () => { + await act(async () => { + renderer = create( + createElement(Harness, { + connected: true, + sessionId: 'session-1', + sourceIdentity: 'host-a\0workspace-a' + }) + ) + }) + await vi.waitFor(() => expect(listener).toEqual(expect.any(Function))) + act(() => listener?.(snapshotWithMessage())) + expect(hook?.session.messages).toHaveLength(1) + + await act(async () => { + renderer?.update( + createElement(Harness, { + connected: false, + sessionId: 'session-1', + sourceIdentity: 'host-b\0workspace-b' + }) + ) + }) + expect(hook?.session.messages).toEqual([]) + }) +}) diff --git a/mobile/src/session/use-mobile-structured-agent-session.ts b/mobile/src/session/use-mobile-structured-agent-session.ts new file mode 100644 index 00000000000..d9cabf1f2d0 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-agent-session.ts @@ -0,0 +1,315 @@ +import { useCallback, useEffect, useMemo, useRef } from 'react' +import type { + AgentSessionCancelResult, + AgentSessionPromptResult, + AgentSessionSendResult +} from '../../../src/shared/agent-session-wire' +import type { + SessionOptionDescriptor, + SessionOptionsSurface, + SessionOptionValue +} from '../../../src/shared/native-chat-session-options' +import { + structuredAgentSessionSendBody, + type StructuredAgentSessionAttachment +} from '../../../src/shared/structured-agent-session-outbox' +import { encodeNativeChatTranscriptIdentity } from '../../../src/shared/native-chat-transcript-retention' +import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' +import { projectStructuredAgentSessionMessages } from '../../../src/shared/structured-agent-session-message-projection' +import { activeStructuredAgentSessionTurnId } from '../../../src/shared/structured-agent-session-projection' +import { + pendingStructuredApproval, + pendingStructuredQuestion, + projectStructuredPermission, + projectStructuredQuestion, + structuredApprovalResponseTarget, + structuredQuestionResponseTarget +} from './mobile-structured-agent-prompts' +import { + requestStructuredAgentSessionMutation, + retainStructuredSessionOperationId as retainStructuredOpId, + timeoutForDeadline, + type StructuredAgentSessionMutationResult +} from './mobile-structured-agent-session-rpc' +import type { RpcClient } from '../transport/rpc-client' +import type { MobileChatPermission } from './mobile-native-chat-permission' +import type { MobileChatQuestion } from './mobile-native-chat-question' +import type { MobileNativeChatSession } from './use-mobile-native-chat-session' +import { useMobileStructuredAgentState } from './use-mobile-structured-agent-state' +import { useMobileStructuredAgentOptions } from './use-mobile-structured-agent-options' + +type StructuredMobileAttachment = StructuredAgentSessionAttachment & { id?: string } + +type StructuredMobileSession = { + session: MobileNativeChatSession + isWorking: boolean + turnId: string | null + sendWithOutcome: ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredMobileAttachment[] + ) => Promise + cancel: () => void + permission: MobileChatPermission | null + question: MobileChatQuestion | null + optionSnapshot: SessionOptionDescriptor[] + optionSurface: SessionOptionsSurface + pendingOptionId: string | null + respondPermission: (optionId: string) => Promise + respondQuestion: (answer: string) => Promise + setStructuredOption: (id: string, value: SessionOptionValue) => Promise + invokeStructuredOption: (id: string) => Promise +} + +export function useMobileStructuredAgentSession(args: { + client: RpcClient | null + sessionId: string | null + /** Host/workspace scope used to keep same provider ids isolated. */ + sourceIdentity?: string + enabled: boolean + /** Live transport only; gates the connection-scoped hold, nothing else. */ + connected: boolean + agent: string | null + onSendError: (message: string) => void +}): StructuredMobileSession { + const { agent, client, connected, sessionId, sourceIdentity = '', enabled, onSendError } = args + const sessionKey = encodeNativeChatTranscriptIdentity([sourceIdentity, agent, sessionId]) + const operationIdsRef = useRef(new Map()) + useEffect(() => () => operationIdsRef.current.clear(), []) + const retainOperationId = (key: string, operationId?: string): string => + retainStructuredOpId(operationIdsRef.current, key, operationId) + const stateArgs = { client, sessionId, sessionKey, enabled, connected } + const { state, stateRef, loadingOlder, loadEarlier } = useMobileStructuredAgentState(stateArgs) + + const mutate = useCallback( + async ( + method: string, + fingerprintMethod: string, + fields: Record + ): Promise> => { + const current = stateRef.current + if (!client || !sessionId || !enabled || current.fence === null) { + return { status: 'rejected' } + } + const targetFence = current.fence + const key = `${sessionKey}:${fingerprintMethod}:${JSON.stringify(fields)}` + const clientOperationId = retainOperationId(key, operationIdsRef.current.get(key)) + const result = await requestStructuredAgentSessionMutation({ + client, + method, + fingerprintMethod, + sessionId, + expectedRuntimeFence: targetFence, + fields, + clientOperationId + }) + if (result.status === 'accepted') { + operationIdsRef.current.delete(key) + return { + status: 'accepted', + value: result.value, + sameFence: stateRef.current.fence === targetFence + } + } + if (result.status === 'unknown') { + // Prompt/option/cancel plans cannot redispatch an unknown ledger row; + // issue a fresh id so a retry can be admitted after the user checks the + // stream. Sends opt into explicit retryUnknown below. + operationIdsRef.current.delete(key) + return result + } + operationIdsRef.current.delete(key) + onSendError(result.message) + return { status: 'rejected' } + }, + [client, enabled, onSendError, sessionId, sessionKey] + ) + + const { + invokeStructuredOption, + optionSnapshot, + optionSurface, + pendingOptionId, + setStructuredOption + } = useMobileStructuredAgentOptions({ + agent, + client, + sessionId, + enabled, + fence: state.fence, + mutate + }) + + const sendWithOutcome = useCallback( + async ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredMobileAttachment[] + ): Promise => { + const currentFence = stateRef.current.fence + if (!client || !sessionId || !enabled || currentFence === null) { + onSendError('Message not sent (disconnected)') + return 'rejected' + } + const timeoutMs = timeoutForDeadline(deadline) + if (timeoutMs === null) { + onSendError('Message not sent') + return 'rejected' + } + if (attachments === undefined && images !== undefined && images.length > 0) { + onSendError('Message not sent') + return 'rejected' + } + const sendAttachments = attachments ?? [] + const body = structuredAgentSessionSendBody(text, sendAttachments) + if (body.blocks.length === 0) { + return 'rejected' + } + const fields = { body } + const key = `${sessionKey}:agentSession.send:${JSON.stringify(fields)}` + const priorOperationId = operationIdsRef.current.get(key) + const clientOperationId = retainOperationId(key, priorOperationId) + const result = await requestStructuredAgentSessionMutation({ + client, + method: 'agentSession.send', + fingerprintMethod: 'agentSession.send', + sessionId, + expectedRuntimeFence: currentFence, + fields, + clientOperationId, + ...(priorOperationId ? { retryUnknown: true } : {}), + timeoutMs + }) + if (result.status === 'accepted') { + operationIdsRef.current.delete(key) + return 'accepted' + } + if (result.status === 'unknown') { + return 'unknown' + } + operationIdsRef.current.delete(key) + onSendError(result.message === 'Request not sent' ? 'Message not sent' : result.message) + return 'rejected' + }, + [client, enabled, onSendError, sessionId, sessionKey] + ) + + const respondPermission = useCallback( + async (optionId: string): Promise => { + const target = structuredApprovalResponseTarget( + optionId, + stateRef.current.items.find(pendingStructuredApproval) ?? null + ) + if (!target) { + return false + } + const result = await mutate( + 'agentSession.respondToApproval', + 'agentSession.respondTo:approval', + target + ) + if (result.status === 'unknown') { + onSendError('Response unconfirmed — check chat before retrying') + return false + } + return result.status === 'accepted' + }, + [mutate, onSendError] + ) + + const respondQuestion = useCallback( + async (answer: string): Promise => { + const target = structuredQuestionResponseTarget( + answer, + stateRef.current.items.find(pendingStructuredQuestion) ?? null + ) + if (!target) { + return false + } + const result = await mutate( + 'agentSession.respondToQuestion', + 'agentSession.respondTo:question', + target + ) + if (result.status === 'unknown') { + onSendError('Answer unconfirmed — check chat before retrying') + return false + } + return result.status === 'accepted' + }, + [mutate, onSendError] + ) + + const cancel = useCallback(() => { + const current = stateRef.current + const turnId = activeStructuredAgentSessionTurnId(current.items) + if (!client || !sessionId || !enabled || current.fence === null || !turnId) { + onSendError('Stop not sent') + return + } + const fields = { turnId } + const key = `${sessionKey}:agentSession.cancel:${JSON.stringify(fields)}` + const clientOperationId = retainOperationId(key, operationIdsRef.current.get(key)) + void requestStructuredAgentSessionMutation({ + client, + method: 'agentSession.cancel', + fingerprintMethod: 'agentSession.cancel', + sessionId, + expectedRuntimeFence: current.fence, + fields, + clientOperationId + }).then((result) => { + if (result.status !== 'unknown') { + operationIdsRef.current.delete(key) + } + if (result.status === 'unknown') { + onSendError('Stop unconfirmed — check chat before retrying') + } else if (result.status === 'refused') { + onSendError(result.message) + } else if (result.status === 'failed') { + onSendError(result.message === 'Request not sent' ? 'Stop not sent' : result.message) + } + }) + }, [client, enabled, onSendError, sessionId, sessionKey]) + + const messages = useMemo( + () => projectStructuredAgentSessionMessages(state.items, [], state.submissions), + [state.items, state.submissions] + ) + const status = state.status === 'idle' ? 'idle' : state.status + const approvalPrompt = useMemo( + () => state.items.find(pendingStructuredApproval) ?? null, + [state.items] + ) + const questionPrompt = useMemo( + () => state.items.find(pendingStructuredQuestion) ?? null, + [state.items] + ) + + return { + session: { + messages, + status, + transcriptLoading: status === 'loading', + error: state.error, + hasMore: state.hasOlder, + loadingEarlier: loadingOlder, + loadEarlier + }, + isWorking: activeStructuredAgentSessionTurnId(state.items) !== null, + turnId: activeStructuredAgentSessionTurnId(state.items), + sendWithOutcome, + cancel, + permission: projectStructuredPermission(approvalPrompt), + question: projectStructuredQuestion(questionPrompt), + optionSnapshot, + optionSurface, + pendingOptionId, + respondPermission, + respondQuestion, + setStructuredOption, + invokeStructuredOption + } +} diff --git a/mobile/src/session/use-mobile-structured-agent-state.ts b/mobile/src/session/use-mobile-structured-agent-state.ts new file mode 100644 index 00000000000..52aefab24aa --- /dev/null +++ b/mobile/src/session/use-mobile-structured-agent-state.ts @@ -0,0 +1,198 @@ +import { useCallback, useEffect, useLayoutEffect, useRef, useState } from 'react' +import type { + AgentSessionHistoryResult, + AgentSessionSubscribeEvent +} from '../../../src/shared/agent-session-wire' +import { AGENT_SESSION_HISTORY_MAX_LIMIT } from '../../../src/shared/agent-session-wire' +import { structuredAgentSessionHolderId } from '../../../src/shared/structured-agent-session-holder' +import { + EMPTY_STRUCTURED_AGENT_SESSION, + oldestStructuredAgentSessionCursor, + reduceStructuredAgentSession, + type StructuredAgentSessionAction, + type StructuredAgentSessionState +} from '../../../src/shared/structured-agent-session-reducer' +import type { RpcClient } from '../transport/rpc-client' +import { callAgentSession } from './mobile-structured-agent-session-rpc' + +const MAX_RETAINED_SESSION_STATES = 32 + +function isSubscribeEvent(value: unknown): value is AgentSessionSubscribeEvent { + if (typeof value !== 'object' || value === null) { + return false + } + const type = (value as { type?: unknown }).type + return type === 'snapshot' || type === 'batch' || type === 'reset' || type === 'end' +} + +export function useMobileStructuredAgentState(args: { + client: RpcClient | null + sessionId: string | null + sessionKey: string | null + enabled: boolean + /** Live transport only. The hold dies with the connection and has to be retaken, + * but the transcript must survive the outage rather than blank out with it. */ + connected: boolean +}): { + state: StructuredAgentSessionState + stateRef: { readonly current: StructuredAgentSessionState } + loadingOlder: boolean + loadEarlier: () => void +} { + const { client, connected, enabled, sessionId, sessionKey } = args + // Keep a bounded cache so offline tab switches select the right transcript + // synchronously without growing for the lifetime of the app. + const [sessionStates, setSessionStates] = useState>( + () => new Map() + ) + const state = + enabled && sessionKey + ? (sessionStates.get(sessionKey) ?? EMPTY_STRUCTURED_AGENT_SESSION) + : EMPTY_STRUCTURED_AGENT_SESSION + const [loadingOlder, setLoadingOlder] = useState(false) + const stateRef = useRef(state) + const sessionKeyRef = useRef(sessionKey) + const streamGenerationRef = useRef(0) + useLayoutEffect(() => { + stateRef.current = state + sessionKeyRef.current = sessionKey + }, [sessionKey, state]) + + const apply = useCallback( + (action: StructuredAgentSessionAction) => { + if (!sessionKey) { + return + } + setSessionStates((current) => { + const previous = current.get(sessionKey) ?? EMPTY_STRUCTURED_AGENT_SESSION + const next = reduceStructuredAgentSession(previous, action) + if (next === previous) { + return current + } + const updated = new Map(current) + updated.delete(sessionKey) + updated.set(sessionKey, next) + while (updated.size > MAX_RETAINED_SESSION_STATES) { + const oldest = updated.keys().next().value + if (oldest === undefined) { + break + } + updated.delete(oldest) + } + return updated + }) + }, + [sessionKey] + ) + + useEffect(() => { + streamGenerationRef.current += 1 + sessionKeyRef.current = sessionKey + setLoadingOlder(false) + if (!client || !sessionId || !enabled) { + return + } + if (!connected) { + // The cleanup above drops the dead hold and stream; keyed state keeps this + // session's transcript visible while another tab can be selected. + return + } + apply({ type: 'loading' }) + const holderId = structuredAgentSessionHolderId('mobile-chat') + let cancelled = false + let unsubscribe = (): void => {} + const held = callAgentSession(client, 'agentSession.hold', { + sessionId, + holderId + }) + void held + .then(() => { + if (cancelled) { + return + } + unsubscribe = client.subscribe('agentSession.subscribe', { sessionId }, (raw) => { + if ( + typeof raw === 'object' && + raw !== null && + (raw as { type?: unknown }).type === 'error' + ) { + apply({ type: 'error', message: String((raw as { message?: unknown }).message ?? '') }) + return + } + if (isSubscribeEvent(raw)) { + apply({ type: 'event', event: raw }) + } + }) + }) + .catch((error: unknown) => { + if (!cancelled) { + apply({ type: 'error', message: error instanceof Error ? error.message : String(error) }) + } + }) + return () => { + cancelled = true + unsubscribe() + void held + .then(() => + callAgentSession( + client, + 'agentSession.release', + { + sessionId, + holderId + }, + undefined, + { failWhenDisconnected: true } + ).catch(() => undefined) + ) + .catch(() => undefined) + } + }, [apply, client, connected, enabled, sessionId, sessionKey]) + + const loadEarlier = useCallback(() => { + const current = stateRef.current + if (!client || !sessionId || !sessionKey || loadingOlder || !current.hasOlder) { + return + } + const cursor = oldestStructuredAgentSessionCursor(current) + if (!cursor) { + return + } + const requestSessionKey = sessionKey + const requestGeneration = streamGenerationRef.current + setLoadingOlder(true) + void callAgentSession(client, 'agentSession.history', { + sessionId, + direction: 'before', + cursor, + limit: AGENT_SESSION_HISTORY_MAX_LIMIT + }) + .then((result) => { + if ( + result.ok && + sessionKeyRef.current === requestSessionKey && + streamGenerationRef.current === requestGeneration + ) { + apply({ type: 'older-page', requestedEpoch: cursor.epoch, page: result.page }) + } + }) + .catch((error: unknown) => { + if ( + sessionKeyRef.current === requestSessionKey && + streamGenerationRef.current === requestGeneration + ) { + apply({ type: 'error', message: error instanceof Error ? error.message : String(error) }) + } + }) + .finally(() => { + if ( + sessionKeyRef.current === requestSessionKey && + streamGenerationRef.current === requestGeneration + ) { + setLoadingOlder(false) + } + }) + }, [apply, client, loadingOlder, sessionId, sessionKey]) + + return { state, stateRef, loadingOlder, loadEarlier } +} diff --git a/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts b/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts new file mode 100644 index 00000000000..fa786867fc7 --- /dev/null +++ b/mobile/src/session/use-mobile-structured-native-chat-send-bridge.ts @@ -0,0 +1,100 @@ +import { useCallback } from 'react' +import type { MobileNativeChatSendOutcome } from './mobile-native-chat-send' +import type { MobileNativeChatSendOrigin } from './use-mobile-native-chat-drafts' + +type StructuredNativeChatAttachment = { + id?: string + path: string + previewUri: string +} + +export function useMobileStructuredNativeChatSendBridge(args: { + sendStructured: ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredNativeChatAttachment[] + ) => Promise + captureSendOrigin: (text: string) => MobileNativeChatSendOrigin | null + clearDraftForSend: (origin: MobileNativeChatSendOrigin, text: string) => void + acceptSend: (origin: MobileNativeChatSendOrigin, text: string, images?: string[]) => void + holdUnconfirmedSend: ( + origin: MobileNativeChatSendOrigin, + text: string, + onUnconfirmed: () => void + ) => void + restoreRejectedDraft: (origin: MobileNativeChatSendOrigin, text: string) => void + onSendError: (message: string) => void +}): { + send: (text: string, images?: string[]) => Promise + sendWithOutcome: ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredNativeChatAttachment[] + ) => Promise +} { + const { + acceptSend, + captureSendOrigin, + clearDraftForSend, + holdUnconfirmedSend, + onSendError, + restoreRejectedDraft, + sendStructured + } = args + const sendWithOutcome = useCallback( + async ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredNativeChatAttachment[] + ): Promise => { + const origin = captureSendOrigin(text.trimEnd()) + if (!origin) { + onSendError('Message not sent (disconnected)') + return 'rejected' + } + clearDraftForSend(origin, text) + const outcome = + attachments !== undefined + ? await sendStructured(text, images, deadline, attachments) + : deadline !== undefined + ? await sendStructured(text, images, deadline) + : images !== undefined + ? await sendStructured(text, images) + : await sendStructured(text) + if (outcome === 'accepted') { + acceptSend(origin, text.trimEnd(), images) + return 'accepted' + } + if (outcome === 'unknown') { + holdUnconfirmedSend(origin, text.trimEnd(), () => + onSendError('Delivery unconfirmed — check chat before retrying') + ) + return 'unknown' + } + restoreRejectedDraft(origin, text) + return 'rejected' + }, + [ + acceptSend, + captureSendOrigin, + clearDraftForSend, + holdUnconfirmedSend, + onSendError, + restoreRejectedDraft, + sendStructured + ] + ) + const send = useCallback( + async ( + text: string, + images?: string[], + deadline?: number, + attachments?: readonly StructuredNativeChatAttachment[] + ) => (await sendWithOutcome(text, images, deadline, attachments)) !== 'rejected', + [sendWithOutcome] + ) + return { send, sendWithOutcome } +} diff --git a/mobile/src/transport/cellular-connecting-label-stall.test.ts b/mobile/src/transport/cellular-connecting-label-stall.test.ts index e9a46d3b406..358b0abab41 100644 --- a/mobile/src/transport/cellular-connecting-label-stall.test.ts +++ b/mobile/src/transport/cellular-connecting-label-stall.test.ts @@ -19,6 +19,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + type CarrierBehavior = // Carrier silently drops the SYN to a LAN/CGNAT destination: the socket sits // CONNECTING until the client's 12s connect timeout fires. diff --git a/mobile/src/transport/direct-connection-log.ts b/mobile/src/transport/direct-connection-log.ts index ef805e643c5..2630b008459 100644 --- a/mobile/src/transport/direct-connection-log.ts +++ b/mobile/src/transport/direct-connection-log.ts @@ -43,4 +43,8 @@ export class DirectConnectionLog { { code: 'liveness-timeout' } ) } + + connected = (): void => { + this.emit('success', 'Authenticated', 'Channel ready for RPC', { code: 'direct-connected' }) + } } diff --git a/mobile/src/transport/direct-rpc-client.ts b/mobile/src/transport/direct-rpc-client.ts index 36ed573aa5b..16306f95cdd 100644 --- a/mobile/src/transport/direct-rpc-client.ts +++ b/mobile/src/transport/direct-rpc-client.ts @@ -18,6 +18,7 @@ import { import { RpcSessionLivenessWatchdog } from './rpc-session-liveness-watchdog' import { isStaleForegroundDial } from './rpc-stale-dial' import type { ConnectionState, ForegroundNudgeReason, RpcResponse } from './types' +import { negotiateMobileRuntimeCapabilities } from './mobile-runtime-capability-negotiation' const LIVENESS_REQUEST_ID_PREFIX = 'mobile-liveness-' @@ -226,17 +227,22 @@ export class DirectRpcClient implements RpcClient { } private handleAuthenticated(session: RpcClientSocketSession): void { - console.log('[net] e2ee_authenticated — connected', { streamCount: this.streams.size() }) this.livenessSession = session this.liveness.start(session) - this.authenticationGeneration++ - this.reconnect.authenticated() - this.authenticationRetry.accepted() - this.connectionState.publish('connected') - this.connectionLog.emit('success', 'Authenticated', 'Channel ready for RPC', { - code: 'direct-connected' + const generation = ++this.authenticationGeneration + negotiateMobileRuntimeCapabilities({ + sendRequest: (method, params) => + this.requests.sendAuthenticatedRequest(method, params, 5_000), + current: () => this.socketSession === session && this.authenticationGeneration === generation, + onReady: () => { + this.reconnect.authenticated() + this.authenticationRetry.accepted() + this.connectionState.publish('connected') + this.connectionLog.connected() + this.streams.replayAfterAuthentication() + }, + onFailure: () => this.socketClose.forceClose(session) }) - this.streams.replayAfterAuthentication() } private handleRpcResponse(response: RpcResponse): void { diff --git a/mobile/src/transport/foreground-stale-dial-restart.test.ts b/mobile/src/transport/foreground-stale-dial-restart.test.ts index 8c9ebd94d7e..39410e8fc3f 100644 --- a/mobile/src/transport/foreground-stale-dial-restart.test.ts +++ b/mobile/src/transport/foreground-stale-dial-restart.test.ts @@ -25,6 +25,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + // Mirrors React Native's WebSocket: readyState lives in JS and only advances on // a delivered event, so a socket the OS killed while the app was suspended stays // CONNECTING forever from the client's point of view. diff --git a/mobile/src/transport/mobile-relay-rpc-session-liveness.test.ts b/mobile/src/transport/mobile-relay-rpc-session-liveness.test.ts index a3885a226c0..b811721e562 100644 --- a/mobile/src/transport/mobile-relay-rpc-session-liveness.test.ts +++ b/mobile/src/transport/mobile-relay-rpc-session-liveness.test.ts @@ -74,6 +74,16 @@ async function authenticateSession(onLog?: ConnectionLogSink) { _meta: { runtimeId: 'runtime-1' } }) ) + await vi.waitFor(() => expect(fakes.sendText).toHaveBeenCalledTimes(2)) + const capabilities = sentRequests()[1]! + fakes.linkOptions!.onText( + JSON.stringify({ + id: capabilities.id, + ok: true, + result: {}, + _meta: { runtimeId: 'runtime-1' } + }) + ) await vi.waitFor(() => expect(session.getState()).toBe('connected')) fakes.sendText.mockClear() return session diff --git a/mobile/src/transport/mobile-relay-rpc-session.test.ts b/mobile/src/transport/mobile-relay-rpc-session.test.ts index 7f98436c63b..5887dffc73d 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.test.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.test.ts @@ -55,7 +55,7 @@ function openSession() { }) } -async function authenticateSession() { +async function confirmResume() { const session = openSession() fakes.linkOptions!.onHello({ type: 'relay-hello', @@ -93,9 +93,39 @@ async function authenticateSession() { _meta: { runtimeId: 'runtime-1' } }) ) + await vi.waitFor(() => expect(fakes.sendText).toHaveBeenCalledTimes(2)) + const capabilityRequest = JSON.parse(fakes.sendText.mock.calls[1]![0] as string) as { + id: string + method: string + deviceToken: string + params: { clientCapabilities?: string[] } + } + return { session, confirmationRequest: request, capabilityRequest } +} + +async function authenticateSession(capabilitySupported = true) { + const { session, confirmationRequest, capabilityRequest } = await confirmResume() + expect(session.getState()).toBe('handshaking') + fakes.linkOptions!.onText( + JSON.stringify( + capabilitySupported + ? { + id: capabilityRequest.id, + ok: true, + result: capabilityRequest.params, + _meta: { runtimeId: 'runtime-1' } + } + : { + id: capabilityRequest.id, + ok: false, + error: { code: 'method_not_found', message: 'Unknown method' }, + _meta: { runtimeId: 'runtime-1' } + } + ) + ) await vi.waitFor(() => expect(session.getState()).toBe('connected')) fakes.sendText.mockClear() - return { session, confirmationRequest: request } + return { session, confirmationRequest, capabilityRequest } } describe('mobile relay RPC session', () => { @@ -107,7 +137,7 @@ describe('mobile relay RPC session', () => { afterEach(() => vi.useRealTimers()) it('requires exact resume observations and confirms by request ID before becoming connected', async () => { - const { session, confirmationRequest } = await authenticateSession() + const { session, confirmationRequest, capabilityRequest } = await authenticateSession() expect(fakes.linkOptions).toMatchObject({ endpoint: relay, @@ -121,9 +151,32 @@ describe('mobile relay RPC session', () => { }) expect(confirmationRequest.params).not.toHaveProperty('relayDeviceId') expect(confirmationRequest.params).not.toHaveProperty('acceptedCredentialVersion') + expect(capabilityRequest).toMatchObject({ + method: 'runtime.clientCapabilities.update', + params: { + clientCapabilities: expect.arrayContaining(['agent-session.structured.v1']) + }, + deviceToken: 'device-token' + }) expect(session.getAttachDeadlineAt()).toEqual(expect.any(Number)) }) + it('connects when an older runtime rejects capability negotiation', async () => { + const { session } = await authenticateSession(false) + + expect(session.getState()).toBe('connected') + expect(session.getFailure()).toBeNull() + }) + + it('connects when the relay never answers capability negotiation', async () => { + const { session } = await confirmResume() + + // Why: the advisory's own deadline used to fail confirmResume, so a link too slow to + // answer within the request timeout never published 'connected' — it just redialled. + await vi.waitFor(() => expect(session.getState()).toBe('connected'), { timeout: 5_000 }) + expect(session.getFailure()).toBeNull() + }) + // Why: ConnectionState stays 'connecting' until relay-hello, so the migration bound // needs a separate signal to tell "cell never answered the upgrade" from "cell took // relay-auth and is still resolving the assignment". diff --git a/mobile/src/transport/mobile-relay-rpc-session.ts b/mobile/src/transport/mobile-relay-rpc-session.ts index 40b93927139..947a1d23ce8 100644 --- a/mobile/src/transport/mobile-relay-rpc-session.ts +++ b/mobile/src/transport/mobile-relay-rpc-session.ts @@ -10,7 +10,9 @@ import { markRpcDeliveryUnknown } from './rpc-delivery-ambiguity' import { openRpcRequestBudget, resolvePostConnectRequestTimeout } from './rpc-request-budget' import { isRpcResponse } from './rpc-response-shape' import { RelayDialStageTracker, type RelayDialStageSource } from './relay-dial-stage' +import { RelayPendingRequests } from './relay-pending-requests' import { RpcSessionLivenessWatchdog } from './rpc-session-liveness-watchdog' +import { settleMobileRuntimeCapabilities } from './mobile-runtime-capability-negotiation' import type { RpcClient } from './rpc-client' import type { ConnectionLogSink, ConnectionState, RpcResponse } from './types' @@ -19,12 +21,6 @@ const RELAY_MISSED_PROBE_LIMIT = 2 const RELAY_FOREGROUND_PROBE_MIN_INTERVAL_MS = 10_000 let relayRpcSessionSequence = 0 -type PendingRequest = { - resolve: (response: RpcResponse) => void - reject: (error: Error) => void - timer: ReturnType -} - export type MobileRelayRpcSession = RpcClient & RelayDialStageSource & { // The cell's attach-reservation deadline (~10s). Diagnostics only — never @@ -47,10 +43,9 @@ export function connectMobileRelayRpcSession(args: { onLog?: ConnectionLogSink }): MobileRelayRpcSession { const requestTimeoutMs = args.requestTimeoutMs ?? 30_000 - const pending = new Map() + const pending = new RelayPendingRequests() const stateListeners = new Set<(state: ConnectionState) => void>() let state: ConnectionState = 'connecting' - let requestCounter = 0 let lastConnectedAt: number | null = null let attachDeadlineAt: number | null = null let resumeExpiresAt: number | null = null @@ -62,7 +57,7 @@ export function connectMobileRelayRpcSession(args: { const livenessIdentity = {} const dialStage = new RelayDialStageTracker() const streams = new MobileRelayRpcStreams({ - nextId, + nextId: () => pending.nextId(), sendFrame, waitForConnected: () => waitForConnected() }) @@ -137,7 +132,7 @@ export function connectMobileRelayRpcSession(args: { closed = true livenessWatchdog.stop(livenessIdentity) link.close() - rejectPending(new Error('Client closed')) + pending.rejectAll(new Error('Client closed')) streams.clear() publishState('disconnected') }, @@ -155,7 +150,8 @@ export function connectMobileRelayRpcSession(args: { missedProbeLimit: RELAY_MISSED_PROBE_LIMIT, voluntaryProbeMinIntervalMs: RELAY_FOREGROUND_PROBE_MIN_INTERVAL_MS, sendProbe: () => - state === 'connected' && sendFrame({ id: nextId(), method: 'status.get', params: undefined }), + state === 'connected' && + sendFrame({ id: pending.nextId(), method: 'status.get', params: undefined }), onTimeout: (evidence) => { args.onLog?.({ id: `relay-liveness-${logSessionId}-${++logSequence}`, @@ -190,6 +186,10 @@ export function connectMobileRelayRpcSession(args: { resumeConfirmation = result.resumeConfirmation resumeExpiresAt = result.resumeConfirmation.resumeExpiresAt lastConnectedAt = Date.now() + // Why: an unanswered advisory must not keep a slow relay from ever reaching connected. + await settleMobileRuntimeCapabilities((method, params) => + sendRpc(method, params, requestTimeoutMs, true) + ) livenessWatchdog.start(livenessIdentity) publishState('connected') } catch (error) { @@ -206,17 +206,17 @@ export function connectMobileRelayRpcSession(args: { if (closed || (!beforeConnected && state !== 'connected')) { return Promise.reject(new Error('relay session not connected')) } - const id = nextId() + const id = pending.nextId() return new Promise((resolve, reject) => { const timer = setTimeout(() => { - pending.delete(id) + pending.drop(id) // Why: the frame was written long ago — the desktop may have processed it. reject(markRpcDeliveryUnknown(new Error(`relay RPC timed out: ${method}`))) }, timeoutMs) - pending.set(id, { resolve, reject, timer }) + pending.track(id, { resolve, reject, timer }) if (!sendFrame({ id, method, params })) { clearTimeout(timer) - pending.delete(id) + pending.drop(id) reject(new Error('relay E2EE channel not ready')) } }) @@ -236,11 +236,7 @@ export function connectMobileRelayRpcSession(args: { if (!isRpcResponse(value)) { return } - const request = pending.get(value.id) - if (request) { - clearTimeout(request.timer) - pending.delete(value.id) - request.resolve(value) + if (pending.settle(value)) { return } streams.handleResponse(value) @@ -296,28 +292,9 @@ export function connectMobileRelayRpcSession(args: { failure = error livenessWatchdog.stop(livenessIdentity) link.close() - rejectPending(error) + pending.rejectAll(error) publishState(error instanceof MobileE2EEAuthenticationError ? 'auth-failed' : 'disconnected') } - - function rejectPending(error: Error): void { - if (pending.size === 0) { - return - } - // Why: pending entries only exist after their frame reached the authenticated - // link (sendFrame failures delete them synchronously), so the desktop may - // have processed them — mark the ambiguity for callers. - markRpcDeliveryUnknown(error) - for (const request of pending.values()) { - clearTimeout(request.timer) - request.reject(error) - } - pending.clear() - } - - function nextId(): string { - return `relay-rpc-${++requestCounter}-${Date.now()}` - } } function asError(error: unknown): Error { diff --git a/mobile/src/transport/mobile-runtime-capability-negotiation.test.ts b/mobile/src/transport/mobile-runtime-capability-negotiation.test.ts new file mode 100644 index 00000000000..6eb659d14a6 --- /dev/null +++ b/mobile/src/transport/mobile-runtime-capability-negotiation.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it, vi } from 'vitest' +import { negotiateMobileRuntimeCapabilities } from './mobile-runtime-capability-negotiation' +import { markRpcDeliveryUnknown } from './rpc-delivery-ambiguity' +import type { RpcResponse } from './types' + +function negotiate(args: { reject: unknown; current?: boolean }): { + onReady: ReturnType + onFailure: ReturnType +} { + const onReady = vi.fn() + const onFailure = vi.fn() + negotiateMobileRuntimeCapabilities({ + sendRequest: () => Promise.reject(args.reject), + current: () => args.current ?? true, + onReady, + onFailure + }) + return { onReady, onFailure } +} + +describe('mobile runtime capability negotiation', () => { + it('proceeds when the host never answers, so a slow link still reaches connected', async () => { + const timedOut = markRpcDeliveryUnknown( + new Error('Request timed out: runtime.clientCapabilities.update') + ) + const { onReady, onFailure } = negotiate({ reject: timedOut }) + + await vi.waitFor(() => expect(onReady).toHaveBeenCalledTimes(1)) + expect(onFailure).not.toHaveBeenCalled() + }) + + it('proceeds when the socket drops the request mid-flight', async () => { + const interrupted = markRpcDeliveryUnknown(new Error('Connection interrupted')) + const { onReady, onFailure } = negotiate({ reject: interrupted }) + + await vi.waitFor(() => expect(onReady).toHaveBeenCalledTimes(1)) + expect(onFailure).not.toHaveBeenCalled() + }) + + it('fails a socket that could not put the advisory on the wire', async () => { + const { onReady, onFailure } = negotiate({ reject: new Error('Connection interrupted') }) + + await vi.waitFor(() => expect(onFailure).toHaveBeenCalledTimes(1)) + expect(onReady).not.toHaveBeenCalled() + }) + + it('leaves a replaced session alone on an unanswered request', async () => { + const timedOut = markRpcDeliveryUnknown(new Error('Request timed out')) + const { onReady, onFailure } = negotiate({ reject: timedOut, current: false }) + + await vi.waitFor(() => expect(onReady).not.toHaveBeenCalled()) + expect(onFailure).not.toHaveBeenCalled() + }) + + it('leaves a replaced session alone on a successful response', async () => { + const onReady = vi.fn() + const onFailure = vi.fn() + negotiateMobileRuntimeCapabilities({ + sendRequest: () => + Promise.resolve({ id: 'capability-1', ok: true, result: {} } as RpcResponse), + current: () => false, + onReady, + onFailure + }) + + await vi.waitFor(() => expect(onReady).not.toHaveBeenCalled()) + expect(onFailure).not.toHaveBeenCalled() + }) +}) diff --git a/mobile/src/transport/mobile-runtime-capability-negotiation.ts b/mobile/src/transport/mobile-runtime-capability-negotiation.ts new file mode 100644 index 00000000000..7221f8e085a --- /dev/null +++ b/mobile/src/transport/mobile-runtime-capability-negotiation.ts @@ -0,0 +1,57 @@ +import { + MOBILE_RUNTIME_CLIENT_CAPABILITY_UPDATE_METHOD, + mobileRuntimeClientCapabilityUpdateParams +} from './mobile-runtime-client-capabilities' +import { isRpcDeliveryUnknown } from './rpc-delivery-ambiguity' +import type { RpcResponse } from './types' + +type CapabilityRequest = (method: string, params: unknown) => Promise + +/** + * The advisory is one-way and its result is discarded, so an unanswered request says nothing about + * the link — only a frame that never reached the wire proves the socket cannot carry traffic. + * Everything else (timeout, mid-flight drop) settles like an explicit rejection: capabilities + * unavailable, proceed. Rejects for the unsent case alone. + */ +export async function settleMobileRuntimeCapabilities( + sendRequest: CapabilityRequest +): Promise { + let response: RpcResponse + try { + response = await sendRequest( + MOBILE_RUNTIME_CLIENT_CAPABILITY_UPDATE_METHOD, + mobileRuntimeClientCapabilityUpdateParams() + ) + } catch (error) { + if (!isRpcDeliveryUnknown(error)) { + throw error + } + console.warn('[net] mobile capability negotiation unanswered — proceeding', error) + return + } + if (!response.ok) { + console.warn('[net] mobile capability negotiation unavailable', response.error.code) + } +} + +export function negotiateMobileRuntimeCapabilities(args: { + sendRequest: CapabilityRequest + current: () => boolean + onReady: () => void + onFailure: () => void +}): void { + void settleMobileRuntimeCapabilities(args.sendRequest) + .then(() => { + if (args.current()) { + args.onReady() + } + }) + .catch((error: unknown) => { + if (!args.current()) { + return + } + // Why: nothing else force-closes a socket that cannot send before `connected` is published. + console.warn('[net] mobile capability negotiation could not be sent', error) + args.onFailure() + }) +} diff --git a/mobile/src/transport/mobile-runtime-client-capabilities.ts b/mobile/src/transport/mobile-runtime-client-capabilities.ts new file mode 100644 index 00000000000..5b3dc977240 --- /dev/null +++ b/mobile/src/transport/mobile-runtime-client-capabilities.ts @@ -0,0 +1,44 @@ +import { + STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../../src/shared/protocol-version' +import { remoteRuntimeClientCapabilities } from '../../../src/shared/remote-runtime-client-capabilities' + +export const MOBILE_RUNTIME_CLIENT_CAPABILITIES = remoteRuntimeClientCapabilities([ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_HOLD_RUNTIME_CAPABILITY +]) + +export const MOBILE_RUNTIME_CLIENT_CAPABILITY_UPDATE_METHOD = + 'runtime.clientCapabilities.update' as const + +export function mobileRuntimeClientCapabilityUpdateParams(): { + clientCapabilities: string[] +} { + return { clientCapabilities: [...MOBILE_RUNTIME_CLIENT_CAPABILITIES] } +} + +export function mobileRuntimeClientCapabilityUpdateRequest(args: { + id: string + deviceToken: string +}): { + id: string + deviceToken: string + method: typeof MOBILE_RUNTIME_CLIENT_CAPABILITY_UPDATE_METHOD + params: { clientCapabilities: string[] } +} { + return { + id: args.id, + deviceToken: args.deviceToken, + method: MOBILE_RUNTIME_CLIENT_CAPABILITY_UPDATE_METHOD, + params: mobileRuntimeClientCapabilityUpdateParams() + } +} + +export function advertiseMobileRuntimeClientCapabilities( + send: (request: unknown) => boolean | void, + id: string, + deviceToken: string +): void { + send(mobileRuntimeClientCapabilityUpdateRequest({ id, deviceToken })) +} diff --git a/mobile/src/transport/relay-pending-requests.ts b/mobile/src/transport/relay-pending-requests.ts new file mode 100644 index 00000000000..8260869d73c --- /dev/null +++ b/mobile/src/transport/relay-pending-requests.ts @@ -0,0 +1,53 @@ +import { markRpcDeliveryUnknown } from './rpc-delivery-ambiguity' +import type { RpcResponse } from './types' + +type PendingRequest = { + resolve: (response: RpcResponse) => void + reject: (error: Error) => void + timer: ReturnType +} + +/** In-flight relay RPC requests awaiting their response frame, keyed by request id. */ +export class RelayPendingRequests { + private readonly pending = new Map() + private requestCounter = 0 + + nextId(): string { + return `relay-rpc-${++this.requestCounter}-${Date.now()}` + } + + track(id: string, request: PendingRequest): void { + this.pending.set(id, request) + } + + drop(id: string): void { + this.pending.delete(id) + } + + /** Settle the waiter for this response; false when no request owns it. */ + settle(response: RpcResponse): boolean { + const request = this.pending.get(response.id) + if (!request) { + return false + } + clearTimeout(request.timer) + this.pending.delete(response.id) + request.resolve(response) + return true + } + + rejectAll(error: Error): void { + if (this.pending.size === 0) { + return + } + // Why: pending entries only exist after their frame reached the authenticated + // link (sendFrame failures delete them synchronously), so the desktop may + // have processed them — mark the ambiguity for callers. + markRpcDeliveryUnknown(error) + for (const request of this.pending.values()) { + clearTimeout(request.timer) + request.reject(error) + } + this.pending.clear() + } +} diff --git a/mobile/src/transport/rpc-client-capabilities.test.ts b/mobile/src/transport/rpc-client-capabilities.test.ts new file mode 100644 index 00000000000..7107ae6717e --- /dev/null +++ b/mobile/src/transport/rpc-client-capabilities.test.ts @@ -0,0 +1,161 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { connect } from './rpc-client' + +vi.mock('./e2ee', () => ({ + generateKeyPair: () => ({ + publicKey: new Uint8Array(32), + secretKey: new Uint8Array(32) + }), + deriveSharedKey: () => new Uint8Array(32), + publicKeyFromBase64: () => new Uint8Array(32), + publicKeyToBase64: () => 'client-public-key', + encrypt: (plaintext: string) => `encrypted:${plaintext}`, + decrypt: (raw: string) => raw.replace(/^encrypted:/, ''), + decryptBytes: (bytes: Uint8Array) => bytes +})) + +class MockWebSocket { + static CONNECTING = 0 + static OPEN = 1 + static CLOSING = 2 + static CLOSED = 3 + + readonly CONNECTING = MockWebSocket.CONNECTING + readonly OPEN = MockWebSocket.OPEN + readonly CLOSING = MockWebSocket.CLOSING + readonly CLOSED = MockWebSocket.CLOSED + + readyState = MockWebSocket.CONNECTING + onopen: (() => void) | null = null + onmessage: ((event: { data: unknown }) => void) | null = null + onclose: (() => void) | null = null + sent: string[] = [] + + constructor(readonly endpoint: string) { + mockSockets.push(this) + } + + send(payload: string): void { + this.sent.push(payload) + } + + close(): void { + this.readyState = MockWebSocket.CLOSED + this.onclose?.() + } + + open(): void { + this.readyState = MockWebSocket.OPEN + this.onopen?.() + } + + receive(payload: unknown): void { + this.onmessage?.({ data: payload }) + } +} + +type SentRpcRequest = { id: string; method: string; params?: unknown } + +const mockSockets: MockWebSocket[] = [] +const originalWebSocket = globalThis.WebSocket + +function sentRequest(socket: MockWebSocket, method: string): SentRpcRequest { + const request = socket.sent + .map((payload) => JSON.parse(payload.replace(/^encrypted:/, '')) as SentRpcRequest) + .find((candidate) => candidate.method === method) + if (!request) { + throw new Error(`Request not sent: ${method}`) + } + return request +} + +describe('mobile rpc-client capabilities', () => { + beforeEach(() => { + mockSockets.length = 0 + globalThis.WebSocket = MockWebSocket as unknown as typeof WebSocket + }) + + afterEach(() => { + globalThis.WebSocket = originalWebSocket + }) + + it('waits for mobile capability acknowledgement before replaying streams', async () => { + const client = connect('ws://desktop.invalid', 'token', 'server-key') + const socket = mockSockets[0]! + client.subscribe('session.tabs.subscribe', { worktree: 'id:wt-1' }, () => {}) + + socket.open() + socket.receive(JSON.stringify({ type: 'e2ee_ready' })) + socket.receive('encrypted:{"type":"e2ee_authenticated"}') + + const capabilityRequest = sentRequest(socket, 'runtime.clientCapabilities.update') + expect(capabilityRequest.params).toMatchObject({ + clientCapabilities: expect.arrayContaining(['agent-session.structured.v1']) + }) + expect(socket.sent.some((payload) => payload.includes('session.tabs.subscribe'))).toBe(false) + + socket.receive( + `encrypted:${JSON.stringify({ + id: capabilityRequest.id, + ok: true, + result: capabilityRequest.params, + _meta: { runtimeId: 'runtime-1' } + })}` + ) + + await vi.waitFor(() => expect(sentRequest(socket, 'session.tabs.subscribe')).toBeDefined()) + + client.close() + }) + + it('replays streams when an older runtime rejects capability negotiation', async () => { + const client = connect('ws://desktop.invalid', 'token', 'server-key') + const socket = mockSockets[0]! + client.subscribe('session.tabs.subscribe', { worktree: 'id:wt-1' }, () => {}) + + socket.open() + socket.receive(JSON.stringify({ type: 'e2ee_ready' })) + socket.receive('encrypted:{"type":"e2ee_authenticated"}') + + const capabilityRequest = sentRequest(socket, 'runtime.clientCapabilities.update') + socket.receive( + `encrypted:${JSON.stringify({ + id: capabilityRequest.id, + ok: false, + error: { code: 'method_not_found', message: 'Unknown method' }, + _meta: { runtimeId: 'runtime-1' } + })}` + ) + + await vi.waitFor(() => expect(sentRequest(socket, 'session.tabs.subscribe')).toBeDefined()) + expect(client.getState()).toBe('connected') + + client.close() + }) + + it('reaches connected when a slow host never answers capability negotiation', async () => { + vi.useFakeTimers() + try { + const client = connect('ws://desktop.invalid', 'token', 'server-key') + const socket = mockSockets[0]! + client.subscribe('session.tabs.subscribe', { worktree: 'id:wt-1' }, () => {}) + + socket.open() + socket.receive(JSON.stringify({ type: 'e2ee_ready' })) + socket.receive('encrypted:{"type":"e2ee_authenticated"}') + sentRequest(socket, 'runtime.clientCapabilities.update') + + // Why: the 5s capability deadline used to force-close the socket, so a link + // this slow never left 'connecting' — it just redialled forever. + await vi.advanceTimersByTimeAsync(5_001) + + expect(client.getState()).toBe('connected') + expect(sentRequest(socket, 'session.tabs.subscribe')).toBeDefined() + expect(socket.readyState).toBe(MockWebSocket.OPEN) + + client.close() + } finally { + vi.useRealTimers() + } + }) +}) diff --git a/mobile/src/transport/rpc-client-connect-wait-replay.test.ts b/mobile/src/transport/rpc-client-connect-wait-replay.test.ts index 24015ce829e..55a9dc60311 100644 --- a/mobile/src/transport/rpc-client-connect-wait-replay.test.ts +++ b/mobile/src/transport/rpc-client-connect-wait-replay.test.ts @@ -14,6 +14,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static CONNECTING = 0 static OPEN = 1 diff --git a/mobile/src/transport/rpc-client-delivery-ambiguity.test.ts b/mobile/src/transport/rpc-client-delivery-ambiguity.test.ts index 6fb0f7d3610..eca1eca03d1 100644 --- a/mobile/src/transport/rpc-client-delivery-ambiguity.test.ts +++ b/mobile/src/transport/rpc-client-delivery-ambiguity.test.ts @@ -15,6 +15,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static CONNECTING = 0 static OPEN = 1 diff --git a/mobile/src/transport/rpc-client-request-deadline.test.ts b/mobile/src/transport/rpc-client-request-deadline.test.ts index 2ed349f375a..05fe81d7944 100644 --- a/mobile/src/transport/rpc-client-request-deadline.test.ts +++ b/mobile/src/transport/rpc-client-request-deadline.test.ts @@ -14,6 +14,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static CONNECTING = 0 static OPEN = 1 diff --git a/mobile/src/transport/rpc-client-request-tracker.ts b/mobile/src/transport/rpc-client-request-tracker.ts index 34edd2631e2..d96d1e97609 100644 --- a/mobile/src/transport/rpc-client-request-tracker.ts +++ b/mobile/src/transport/rpc-client-request-tracker.ts @@ -42,9 +42,28 @@ export class RpcClientRequestTracker { }) } + return this.sendConnectedRequest( + method, + params, + resolvePostConnectRequestTimeout(budget, REQUEST_TIMEOUT_MS) + ) + } + + sendAuthenticatedRequest( + method: string, + params: unknown, + timeoutMs = REQUEST_TIMEOUT_MS + ): Promise { + return this.sendConnectedRequest(method, params, timeoutMs) + } + + private sendConnectedRequest( + method: string, + params: unknown, + timeoutMs: number + ): Promise { return new Promise((resolve, reject) => { const id = this.options.nextId() - const timeoutMs = resolvePostConnectRequestTimeout(budget, REQUEST_TIMEOUT_MS) const timeout = setTimeout(() => { this.pending.delete(id) console.log('[net] sendRequest TIMEOUT', { diff --git a/mobile/src/transport/rpc-client-runtime-events.test.ts b/mobile/src/transport/rpc-client-runtime-events.test.ts index d18739423f7..04b2a90039e 100644 --- a/mobile/src/transport/rpc-client-runtime-events.test.ts +++ b/mobile/src/transport/rpc-client-runtime-events.test.ts @@ -14,6 +14,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class RuntimeEventTestSocket { static CONNECTING = 0 static OPEN = 1 diff --git a/mobile/src/transport/rpc-client-synthesized-close-diagnostics.test.ts b/mobile/src/transport/rpc-client-synthesized-close-diagnostics.test.ts index 832180ad6c1..5898685bb70 100644 --- a/mobile/src/transport/rpc-client-synthesized-close-diagnostics.test.ts +++ b/mobile/src/transport/rpc-client-synthesized-close-diagnostics.test.ts @@ -17,6 +17,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + // Why: close() deliberately never fires onclose — that is the wedged-transport bug being modelled. class WedgedWebSocket { static CONNECTING = 0 diff --git a/mobile/src/transport/rpc-client-terminal-reconnect.test.ts b/mobile/src/transport/rpc-client-terminal-reconnect.test.ts index f382068b735..83f40113cae 100644 --- a/mobile/src/transport/rpc-client-terminal-reconnect.test.ts +++ b/mobile/src/transport/rpc-client-terminal-reconnect.test.ts @@ -15,6 +15,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static CONNECTING = 0 static OPEN = 1 diff --git a/mobile/src/transport/rpc-client-unauthorized-close.test.ts b/mobile/src/transport/rpc-client-unauthorized-close.test.ts index 464d0b95808..1ba79015331 100644 --- a/mobile/src/transport/rpc-client-unauthorized-close.test.ts +++ b/mobile/src/transport/rpc-client-unauthorized-close.test.ts @@ -17,6 +17,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static CONNECTING = 0 static OPEN = 1 diff --git a/mobile/src/transport/rpc-client.test.ts b/mobile/src/transport/rpc-client.test.ts index a7f3bb9d82a..e88e6c220e6 100644 --- a/mobile/src/transport/rpc-client.test.ts +++ b/mobile/src/transport/rpc-client.test.ts @@ -15,6 +15,11 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +// Capability ordering has dedicated coverage; keep connection tests focused on socket behavior. +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static CONNECTING = 0 static OPEN = 1 @@ -64,36 +69,20 @@ class MockWebSocket { const mockSockets: MockWebSocket[] = [] const originalWebSocket = globalThis.WebSocket -function sentRequest(socket: MockWebSocket, method: string): { id: string; params?: unknown } { - for (const payload of socket.sent) { - const decoded = JSON.parse(payload.replace(/^encrypted:/, '')) as { - id: string - method: string - params?: unknown - } - if (decoded.method === method) { - return { id: decoded.id, params: decoded.params } - } +type SentRpcRequest = { id: string; method: string; params?: unknown } + +function sentRequest(socket: MockWebSocket, method: string): SentRpcRequest { + const request = sentRequests(socket, method)[0] + if (request) { + return request } throw new Error(`Request not sent: ${method}`) } -function sentRequests( - socket: MockWebSocket, - method: string -): Array<{ id: string; params?: unknown }> { - const requests: Array<{ id: string; params?: unknown }> = [] - for (const payload of socket.sent) { - const decoded = JSON.parse(payload.replace(/^encrypted:/, '')) as { - id: string - method: string - params?: unknown - } - if (decoded.method === method) { - requests.push({ id: decoded.id, params: decoded.params }) - } - } - return requests +function sentRequests(socket: MockWebSocket, method: string): SentRpcRequest[] { + return socket.sent + .map((payload) => JSON.parse(payload.replace(/^encrypted:/, '')) as SentRpcRequest) + .filter((request) => request.method === method) } function encodeBrowserFrame(): Uint8Array { diff --git a/mobile/src/transport/rpc-session-liveness-integration.test.ts b/mobile/src/transport/rpc-session-liveness-integration.test.ts index c446477c365..88e71a0862c 100644 --- a/mobile/src/transport/rpc-session-liveness-integration.test.ts +++ b/mobile/src/transport/rpc-session-liveness-integration.test.ts @@ -15,6 +15,10 @@ vi.mock('./e2ee', () => ({ decryptBytes: (bytes: Uint8Array) => bytes })) +vi.mock('./mobile-runtime-capability-negotiation', () => ({ + negotiateMobileRuntimeCapabilities: (args: { onReady: () => void }) => args.onReady() +})) + class MockWebSocket { static readonly CONNECTING = 0 static readonly OPEN = 1 diff --git a/src/main/runtime/mobile-rpc-allowlist.test.ts b/src/main/runtime/mobile-rpc-allowlist.test.ts index adf9223b39a..684d6699fc1 100644 --- a/src/main/runtime/mobile-rpc-allowlist.test.ts +++ b/src/main/runtime/mobile-rpc-allowlist.test.ts @@ -36,7 +36,13 @@ const MOBILE_DYNAMIC_RPC_METHODS = [ 'github.resolveReviewThread', 'github.project.updateIssueCommentBySlug', 'github.project.deleteIssueCommentBySlug', - 'hostedReview.forBranch' + 'hostedReview.forBranch', + 'runtime.clientCapabilities.update', + 'agentSession.send', + 'agentSession.cancel', + 'agentSession.history', + 'agentSession.hold', + 'agentSession.release' ] const MOBILE_STREAMING_CLEANUP_RPC_METHODS = [ @@ -145,9 +151,28 @@ describe('mobile RPC allowlist', () => { ).toEqual([]) }) - it('does not expose structured agent sessions to mobile credentials', () => { + it('exposes only the mobile structured agent-session surface', () => { expect( [...mobileRpcAllowlist()].filter((method) => method.startsWith('agentSession.')) - ).toEqual([]) + ).toEqual([ + 'agentSession.createSupport', + 'agentSession.create', + 'agentSession.ensure', + 'agentSession.send', + 'agentSession.cancel', + 'agentSession.close', + 'agentSession.respondToApproval', + 'agentSession.respondToQuestion', + 'agentSession.setOption', + 'agentSession.handoffStatus', + 'agentSession.options', + 'agentSession.history', + 'agentSession.subscribe', + 'agentSession.unsubscribe', + 'agentSession.hold', + 'agentSession.release' + ]) + expect(mobileRpcAllowlist().has('agentSession.attach')).toBe(false) + expect(mobileRpcAllowlist().has('agentSession.requestHandoff')).toBe(false) }) }) diff --git a/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts b/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts index 9c6e5cda532..bd282b6575d 100644 --- a/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts +++ b/src/main/runtime/orca-runtime-close-structured-agent-session-tab.ts @@ -21,8 +21,10 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi tab: RuntimeMobileSessionAgentTab ): Promise { const host = getStructuredAgentSessionHost() - if (typeof host?.setSessionTabVisibility === 'function') { - await host.setSessionTabVisibility(tab.sessionId, false) + if (host) { + if (typeof host.setSessionTabVisibility === 'function') { + await host.setSessionTabVisibility(tab.sessionId, false) + } } const nextTabs = snapshot.tabs.filter((candidate) => candidate.id !== tab.id) const active = nextTabs.find((candidate) => candidate.isActive) ?? nextTabs[0] ?? null @@ -41,6 +43,10 @@ export class OrcaRuntimeWithCloseStructuredAgentSessionTab extends OrcaRuntimeWi } this.storeMobileSessionSnapshot(worktreeId, nextSnapshot) this.emitMobileSessionTabsSnapshot(nextSnapshot) + // Retire durable visibility and the runtime snapshot before stopping the provider. + if (typeof host?.close === 'function') { + await host.close(tab.sessionId) + } } // Why: a refused echoed close means the echoing client already pruned its diff --git a/src/main/runtime/orca-runtime-state-fields.ts b/src/main/runtime/orca-runtime-state-fields.ts index 770ce0517a8..2f95dfeada1 100644 --- a/src/main/runtime/orca-runtime-state-fields.ts +++ b/src/main/runtime/orca-runtime-state-fields.ts @@ -87,6 +87,11 @@ export class OrcaRuntimeWithStateFields extends OrcaRuntimeWithLinearCommands { ) { super() this.store = store + store?.onSettingsChanged?.((updates) => { + if ('experimentalStructuredNativeChat' in updates) { + this.notifyMobileSessionTabsChanged() + } + }) const runtime = this as RuntimeCommandSurfaceHost installRuntimeFileCommandSurface(runtime, this.fileCommands) installRuntimeGitCommandSurface(runtime, this.gitCommands) diff --git a/src/main/runtime/orca-runtime-structured-native-chat-settings.test.ts b/src/main/runtime/orca-runtime-structured-native-chat-settings.test.ts new file mode 100644 index 00000000000..90be5c61ba2 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-native-chat-settings.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' + +describe('structured native chat settings', () => { + it('republishes mobile session tabs when the host visibility setting changes', () => { + const settingsListeners: ((updates: Record) => void)[] = [] + const runtime = new OrcaRuntimeService({ + onSettingsChanged: vi.fn((listener) => { + settingsListeners.push(listener as (updates: Record) => void) + return vi.fn() + }) + } as never) + const notify = vi.spyOn(runtime, 'notifyMobileSessionTabsChanged').mockImplementation(() => {}) + + settingsListeners[0]?.({ compactWorktreeCards: true }) + expect(notify).not.toHaveBeenCalled() + + settingsListeners[0]?.({ experimentalStructuredNativeChat: true }) + expect(notify).toHaveBeenCalledTimes(1) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-session-restore.test.ts b/src/main/runtime/orca-runtime-structured-session-restore.test.ts index 52db9673705..9e752330445 100644 --- a/src/main/runtime/orca-runtime-structured-session-restore.test.ts +++ b/src/main/runtime/orca-runtime-structured-session-restore.test.ts @@ -205,6 +205,11 @@ describe('structured session cold restoration', () => { it('normalizes a restored tab id and removes it when closed', async () => { const runtime = new OrcaRuntimeService() const closeSessionTab = vi.fn(async () => undefined) + const closeStructuredSession = vi.fn(async () => { + const snapshot = await runtime.listMobileSessionTabs('id:workspace-1') + expect(snapshot.tabs.some((tab) => tab.type === 'agent-session')).toBe(false) + }) + const setSessionTabVisibility = vi.fn(async () => undefined) runtime.setNotifier({ closeSessionTab } as never) const internal = runtime as unknown as { hasPersistedStructuredAgentSessionStore(): boolean @@ -221,6 +226,8 @@ describe('structured session cold restoration', () => { setStructuredAgentSessionHost({ reconcileRestartLeases: async () => undefined, restoreReadableSessions: async () => undefined, + close: closeStructuredSession, + setSessionTabVisibility, listSessionTabs: () => [ { sessionId: 'agent-session:agent-session:restored-session', @@ -304,6 +311,11 @@ describe('structured session cold restoration', () => { 'structured-agent-session-restored-session', 'workspace-1' ) + expect(closeStructuredSession).toHaveBeenCalledWith('restored-session') + expect(setSessionTabVisibility).toHaveBeenCalledWith('restored-session', false) + expect(setSessionTabVisibility.mock.invocationCallOrder[0]).toBeLessThan( + closeStructuredSession.mock.invocationCallOrder[0]! + ) const closed = await runtime.listMobileSessionTabs('id:workspace-1') expect(closed.tabs.map((tab) => tab.id)).toEqual([ diff --git a/src/main/runtime/rpc/core.ts b/src/main/runtime/rpc/core.ts index 975988d203c..5e669ab702e 100644 --- a/src/main/runtime/rpc/core.ts +++ b/src/main/runtime/rpc/core.ts @@ -77,6 +77,8 @@ export type RpcContext = { clientKind?: 'mobile' | 'runtime' // Why: negotiation is bound to the authenticated socket, never asserted by a destructive request. clientCapabilities?: readonly RuntimeCapability[] + // Why: mobile v2 auth is exact-key validated; capability upgrades must mutate only the authenticated socket after auth. + updateClientCapabilities?: (capabilities: readonly RuntimeCapability[]) => void // Why: Dispatch authority rides in the authenticated RPC envelope, never in user payload fields. orchestrationCapability?: string // Why: long-lived mutations such as ask can durably expose acceptance before their waiter settles. diff --git a/src/main/runtime/rpc/dispatcher-stream-options.ts b/src/main/runtime/rpc/dispatcher-stream-options.ts index cc8322373b1..e3151c66b0e 100644 --- a/src/main/runtime/rpc/dispatcher-stream-options.ts +++ b/src/main/runtime/rpc/dispatcher-stream-options.ts @@ -10,6 +10,7 @@ export type RpcDispatchStreamingOptions = { pairedDeviceId?: string clientKind?: 'mobile' | 'runtime' clientCapabilities?: readonly RuntimeCapability[] + updateClientCapabilities?: (capabilities: readonly RuntimeCapability[]) => void pairing?: PairingRpcContext sendBinary?: (bytes: Uint8Array) => boolean | void registerBinaryStreamHandler?: ( diff --git a/src/main/runtime/rpc/dispatcher.ts b/src/main/runtime/rpc/dispatcher.ts index 122e18f078a..c3febab1d62 100644 --- a/src/main/runtime/rpc/dispatcher.ts +++ b/src/main/runtime/rpc/dispatcher.ts @@ -29,8 +29,7 @@ import { RpcStreamingDispatcher } from './rpc-streaming-dispatcher' export type DispatcherOptions = { runtime: OrcaRuntimeService; methods?: readonly RpcAnyMethod[] } -// oxfmt-ignore -type DispatchCallOptions = Pick +type DispatchCallOptions = RpcDispatchStreamingOptions export class RpcDispatcher { private readonly runtime: OrcaRuntimeService @@ -131,6 +130,7 @@ export class RpcDispatcher { clientId: options?.clientId, clientKind: options?.clientKind, clientCapabilities: options?.clientCapabilities, + updateClientCapabilities: options?.updateClientCapabilities, orchestrationCapability: request.orchestrationCapability, authenticatedCallerFingerprint: mutation?.identity.callerFingerprint ?? diff --git a/src/main/runtime/rpc/methods/client-ui.test.ts b/src/main/runtime/rpc/methods/client-ui.test.ts index 3bfb7303bd0..34596835294 100644 --- a/src/main/runtime/rpc/methods/client-ui.test.ts +++ b/src/main/runtime/rpc/methods/client-ui.test.ts @@ -31,6 +31,7 @@ describe('client UI RPC methods', () => { visibleTaskProviders: ['github', 'gitlab'], defaultRepoSelection: ['repo-1'], defaultLinearTeamSelection: ['team-1'], + experimentalStructuredNativeChat: true, compactWorktreeCards: true, minimaxGroupId: 'group-42', minimaxUsageModels: 'general,abab6.5', @@ -60,6 +61,24 @@ describe('client UI RPC methods', () => { expect(response).toMatchObject({ ok: true, result: { settings } }) }) + it('rejects paired attempts to mutate the host-owned structured chat setting', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + updateClientSettings: vi.fn() + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: CLIENT_UI_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('settings.update', { experimentalStructuredNativeChat: true }) + ) + + expect(response).toMatchObject({ + ok: false, + error: { code: 'invalid_argument' } + }) + expect(runtime.updateClientSettings).not.toHaveBeenCalled() + }) + it('persists the runtime host task source settings for mobile Tasks', async () => { const settings = { defaultTuiAgent: null, diff --git a/src/main/runtime/rpc/methods/index.ts b/src/main/runtime/rpc/methods/index.ts index 1bdaf224397..ba77b94803e 100644 --- a/src/main/runtime/rpc/methods/index.ts +++ b/src/main/runtime/rpc/methods/index.ts @@ -38,6 +38,7 @@ import { PLUGIN_METHODS } from './plugins' import { SKILL_METHODS } from './skills' import { CLIPBOARD_METHODS } from './clipboard' import { HOST_CAPABILITY_METHODS } from './host-capabilities' +import { RUNTIME_CLIENT_CAPABILITY_METHODS } from './runtime-client-capabilities' import { EMULATOR_METHODS } from './emulator' import { PAIRING_METHODS } from './pairing' import { UPDATER_METHODS } from './updater' @@ -91,6 +92,7 @@ export const ALL_RPC_METHODS: readonly RpcAnyMethod[] = [ ...SKILL_METHODS, ...CLIPBOARD_METHODS, ...HOST_CAPABILITY_METHODS, + ...RUNTIME_CLIENT_CAPABILITY_METHODS, ...CLIENT_EVENT_METHODS, ...CLIENT_UI_METHODS, ...EMULATOR_METHODS, diff --git a/src/main/runtime/rpc/methods/runtime-client-capabilities.test.ts b/src/main/runtime/rpc/methods/runtime-client-capabilities.test.ts new file mode 100644 index 00000000000..c0fb2f250bd --- /dev/null +++ b/src/main/runtime/rpc/methods/runtime-client-capabilities.test.ts @@ -0,0 +1,74 @@ +import { describe, expect, it, vi } from 'vitest' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { OrcaRuntimeService } from '../../orca-runtime' +import type { RpcRequest } from '../core' +import { RpcDispatcher } from '../dispatcher' +import { RUNTIME_CLIENT_CAPABILITY_METHODS } from './runtime-client-capabilities' + +function makeRequest(params: unknown): RpcRequest { + return { + id: 'req-1', + authToken: 'tok', + method: 'runtime.clientCapabilities.update', + params + } +} + +function dispatcher(): RpcDispatcher { + return new RpcDispatcher({ + runtime: { getRuntimeId: () => 'runtime-1' } as unknown as OrcaRuntimeService, + methods: RUNTIME_CLIENT_CAPABILITY_METHODS + }) +} + +describe('runtime.clientCapabilities.update', () => { + it('updates the authenticated socket capability set after auth', async () => { + const updateClientCapabilities = vi.fn() + + const response = await dispatcher().dispatch( + makeRequest({ + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + }), + { clientKind: 'mobile', updateClientCapabilities } + ) + + expect(response).toMatchObject({ + ok: true, + result: { clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] } + }) + expect(updateClientCapabilities).toHaveBeenCalledWith([ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ]) + }) + + it('rejects malformed upgrades without mutating authenticated state', async () => { + const updateClientCapabilities = vi.fn() + + const response = await dispatcher().dispatch( + makeRequest({ + clientCapabilities: [42] + }), + { clientKind: 'mobile', updateClientCapabilities } + ) + + expect(response).toMatchObject({ + ok: false, + error: { code: 'invalid_argument' } + }) + expect(updateClientCapabilities).not.toHaveBeenCalled() + }) + + it('fails closed when a transport has no post-auth updater', async () => { + const response = await dispatcher().dispatch( + makeRequest({ + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + }), + { clientKind: 'runtime' } + ) + + expect(response).toMatchObject({ + ok: false, + error: { message: 'client_capabilities_update_unsupported' } + }) + }) +}) diff --git a/src/main/runtime/rpc/methods/runtime-client-capabilities.ts b/src/main/runtime/rpc/methods/runtime-client-capabilities.ts new file mode 100644 index 00000000000..a1ab53267b3 --- /dev/null +++ b/src/main/runtime/rpc/methods/runtime-client-capabilities.ts @@ -0,0 +1,24 @@ +import { z } from 'zod' +import type { RuntimeCapability } from '../../../../shared/protocol-version' +import { defineMethod, type RpcAnyMethod } from '../core' + +const ClientCapabilitiesUpdate = z + .object({ + clientCapabilities: z.array(z.string().min(1).max(128)).max(64) + }) + .strict() + +export const RUNTIME_CLIENT_CAPABILITY_METHODS: RpcAnyMethod[] = [ + defineMethod({ + name: 'runtime.clientCapabilities.update', + params: ClientCapabilitiesUpdate, + handler: (params, { updateClientCapabilities }) => { + if (!updateClientCapabilities) { + throw new Error('client_capabilities_update_unsupported') + } + const clientCapabilities = params.clientCapabilities as RuntimeCapability[] + updateClientCapabilities(clientCapabilities) + return { clientCapabilities } + } + }) +] diff --git a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts index 488ab69fd1e..226277f6ebd 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts @@ -75,6 +75,48 @@ describe('session tab structured capability mutations', () => { expect(fixture.calls[method.runtimeMethod]).not.toHaveBeenCalled() }) } + + it.each(['session.tabs.close', 'session.tabs.closeLifecycle'] as const)( + 'allows capable mobile clients to close structured tabs when the experiment is enabled (%s)', + async (method) => { + const snapshot = agentSnapshot() + const closeMobileSessionTab = vi.fn().mockResolvedValue({ closed: true }) + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + listMobileSessionTabs: vi.fn().mockResolvedValue(snapshot), + closeMobileSessionTab + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + const replies: string[] = [] + await dispatcher.dispatchStreaming( + { + id: 'request-1', + authToken: 'token', + method, + params: + method === 'session.tabs.close' + ? { worktree: 'id:wt-1', tabId: 'codex-session', reason: 'user' } + : { + worktree: 'id:wt-1', + tabId: 'codex-session', + reason: 'cleanup', + publicationEpoch: 'epoch-1', + terminal: 'pty-1' + } + }, + (response) => replies.push(response), + { + clientKind: 'mobile', + pairedDeviceId: 'paired-mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(JSON.parse(replies[0]!).ok).toBe(true) + expect(closeMobileSessionTab).toHaveBeenCalledOnce() + } + ) }) function createFixture(capabilities: RuntimeCapability[]) { diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts index e713f74f057..4a61a99bc20 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts @@ -96,6 +96,22 @@ describe('projectSessionTabAgentStatus', () => { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ]) ).toEqual(oldClient) + expect( + projectSessionTabAgentStatus( + snapshot, + 'mobile', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + false + ) + ).toEqual(oldClient) + + const capableMobile = projectSessionTabAgentStatus( + snapshot, + 'mobile', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ) + expect(capableMobile).toBe(snapshot) const capable = projectSessionTabAgentStatus(snapshot, 'runtime', [ STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY @@ -133,6 +149,14 @@ describe('projectSessionTabAgentStatus', () => { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY ]).tabs.map((tab) => tab.id) ).toEqual(['agent-session:codex']) + expect( + projectSessionTabAgentStatus( + snapshot, + 'mobile', + [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], + true + ).tabs.map((tab) => tab.id) + ).toEqual(['agent-session:codex']) }) it('withholds session boundaries from legacy paired clients', () => { diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts index ac8cc0b2164..375b3b499d5 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts @@ -1,6 +1,5 @@ import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' import type { @@ -9,18 +8,21 @@ import type { RuntimeMobileSessionTabsSnapshot } from '../../../../shared/runtime-types' import type { TabGroupLayoutNode } from '../../../../shared/tab-types' +import { structuredNativeChatProjectionEnabled } from './structured-agent-session-policy' type SessionTabsPayload = RuntimeMobileSessionTabsResult | RuntimeMobileSessionTabsSnapshot export function projectSessionTabAgentStatus( payload: TPayload, clientKind: 'mobile' | 'runtime' | undefined, - clientCapabilities: readonly RuntimeCapability[] | undefined + clientCapabilities: readonly RuntimeCapability[] | undefined, + structuredNativeChatEnabled?: boolean ): TPayload { - const structuredVisible = - clientKind !== 'mobile' && - (clientKind === undefined || - (clientCapabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) ?? false)) + const structuredVisible = structuredNativeChatProjectionEnabled({ + clientKind, + clientCapabilities, + structuredNativeChatEnabled + }) let projected = structuredVisible ? payload : projectAgentSessionTabsOut(payload, () => true) if (structuredVisible && clientKind !== undefined) { projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') diff --git a/src/main/runtime/rpc/methods/session-tab-close-methods.ts b/src/main/runtime/rpc/methods/session-tab-close-methods.ts index 50e56144f29..bd60ecd6ddf 100644 --- a/src/main/runtime/rpc/methods/session-tab-close-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-close-methods.ts @@ -4,6 +4,7 @@ import { defineMethod, type RpcAnyMethod } from '../core' import { CloseLifecycleTab, CloseTab } from './session-tabs-schemas' import { assertProjectedSessionTabVisible } from './session-tab-browser-placement-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' +import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ defineMethod({ @@ -14,7 +15,10 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ const visible = projectSessionTabsForClient( await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), context.clientKind, - context.clientCapabilities + context.clientCapabilities, + context.clientKind === 'mobile' + ? isStructuredNativeChatEnabled(context.runtime) + : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) } @@ -80,7 +84,10 @@ export const SESSION_TAB_CLOSE_METHODS: RpcAnyMethod[] = [ const visible = projectSessionTabsForClient( await context.runtime.listMobileSessionTabs(params.worktree, context.pairedDeviceId), context.clientKind, - context.clientCapabilities + context.clientCapabilities, + context.clientKind === 'mobile' + ? isStructuredNativeChatEnabled(context.runtime) + : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) } diff --git a/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts b/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts index 6b9e953e4e3..ba7c41000d0 100644 --- a/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts +++ b/src/main/runtime/rpc/methods/session-tab-mutation-methods.ts @@ -6,6 +6,7 @@ import { translateProjectedSessionTabMove } from './session-tab-browser-placement-projection' import { projectSessionTabsForClient } from './session-tabs-inventory' +import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' import { ActivateTab, MoveTab, SetTabProps, UpdatePaneLayout } from './session-tabs-schemas' export const SESSION_TAB_MUTATION_METHODS: RpcAnyMethod[] = [ @@ -17,7 +18,8 @@ export const SESSION_TAB_MUTATION_METHODS: RpcAnyMethod[] = [ const visible = projectSessionTabsForClient( await runtime.listMobileSessionTabs(params.worktree, pairedDeviceId), clientKind, - clientCapabilities + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined ) assertProjectedSessionTabVisible(visible, params.tabId) } @@ -36,7 +38,12 @@ export const SESSION_TAB_MUTATION_METHODS: RpcAnyMethod[] = [ }) } ) - return projectSessionTabsForMutationClient(result, clientKind, clientCapabilities) + return projectSessionTabsForMutationClient( + result, + clientKind, + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined + ) } }), defineMethod({ @@ -46,7 +53,12 @@ export const SESSION_TAB_MUTATION_METHODS: RpcAnyMethod[] = [ let translated: Parameters[2] = params if (clientKind) { const raw = await runtime.listMobileSessionTabs(params.worktree, pairedDeviceId) - const projected = projectSessionTabsForClient(raw, clientKind, clientCapabilities) + const projected = projectSessionTabsForClient( + raw, + clientKind, + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined + ) translated = translateProjectedSessionTabMove(raw, projected, params) } const base = { tabId: translated.tabId, targetGroupId: translated.targetGroupId } @@ -129,7 +141,8 @@ async function assertVisibleMutationTab( const visible = projectSessionTabsForClient( await runtime.listMobileSessionTabs(worktree, pairedDeviceId), clientKind, - clientCapabilities + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined ) assertProjectedSessionTabVisible(visible, tabId) } diff --git a/src/main/runtime/rpc/methods/session-tabs-inventory.ts b/src/main/runtime/rpc/methods/session-tabs-inventory.ts index fba9a460e86..5ab29ae51b5 100644 --- a/src/main/runtime/rpc/methods/session-tabs-inventory.ts +++ b/src/main/runtime/rpc/methods/session-tabs-inventory.ts @@ -4,6 +4,7 @@ import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime- import type { RpcContext } from '../core' import { projectSessionTabAgentStatus } from './session-tab-agent-status-projection' import { projectSessionTabBrowserPlacements } from './session-tab-browser-placement-projection' +import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' type SessionTabsInventory = { snapshots: RuntimeMobileSessionTabsResult[] @@ -26,21 +27,38 @@ function clientUnderstandsAuthoritativeInventory(context: RpcContext): boolean { export function projectSessionTabsForClient( snapshot: RuntimeMobileSessionTabsResult, clientKind: 'mobile' | 'runtime' | undefined, - clientCapabilities: Parameters[2] + clientCapabilities: Parameters[2], + structuredNativeChatEnabled?: boolean ): RuntimeMobileSessionTabsResult { return projectSessionTabBrowserPlacements( - projectSessionTabAgentStatus(snapshot, clientKind, clientCapabilities), + projectSessionTabAgentStatus( + snapshot, + clientKind, + clientCapabilities, + structuredNativeChatEnabled + ), clientCapabilities ) } +function structuredNativeChatEnabledForContext(context: RpcContext): boolean | undefined { + return context.clientKind === 'mobile' + ? isStructuredNativeChatEnabled(context.runtime) + : undefined +} + function projectInventory( inventory: SessionTabsInventory, context: RpcContext ): SessionTabsInventory { return { snapshots: inventory.snapshots.map((snapshot) => - projectSessionTabsForClient(snapshot, context.clientKind, context.clientCapabilities) + projectSessionTabsForClient( + snapshot, + context.clientKind, + context.clientCapabilities, + structuredNativeChatEnabledForContext(context) + ) ), ...(inventory.authoritative && clientUnderstandsAuthoritativeInventory(context) ? { authoritative: true as const } @@ -109,7 +127,8 @@ export async function subscribeSessionTabsInventory( projectSessionTabsForClient( snapshot, context.clientKind, - context.clientCapabilities + context.clientCapabilities, + structuredNativeChatEnabledForContext(context) ) as SessionTabsChange const withoutNavigationIntent = (snapshot: SessionTabsChange): SessionTabsChange => { if (snapshot.navigationIntent === undefined) { diff --git a/src/main/runtime/rpc/methods/session-tabs.test.ts b/src/main/runtime/rpc/methods/session-tabs.test.ts index be61fc55edf..29131869fa0 100644 --- a/src/main/runtime/rpc/methods/session-tabs.test.ts +++ b/src/main/runtime/rpc/methods/session-tabs.test.ts @@ -2,7 +2,10 @@ import { describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' import type { RpcRequest } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' -import { SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import { + SESSION_TAB_CLOSE_INTENT_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../../../shared/protocol-version' import { SESSION_TAB_METHODS } from './session-tabs' function makeRequest(method: string, params?: unknown): RpcRequest { @@ -10,6 +13,48 @@ function makeRequest(method: string, params?: unknown): RpcRequest { } describe('session tab RPC methods', () => { + it('does not restore structured tabs for mobile while the host setting is off', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: false })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).not.toHaveBeenCalled() + }) + + it('restores structured tabs for mobile only after capability and setting are present', async () => { + const runtime = { + getRuntimeId: () => 'test-runtime', + getClientSettings: vi.fn(() => ({ experimentalStructuredNativeChat: true })), + restoreStructuredAgentSessionTabs: vi.fn(), + listMobileSessionTabs: vi.fn().mockResolvedValue(visibleSnapshot()) + } as unknown as OrcaRuntimeService + const dispatcher = new RpcDispatcher({ runtime, methods: SESSION_TAB_METHODS }) + + const response = await dispatcher.dispatch( + makeRequest('session.tabs.list', { worktree: 'id:wt-1' }), + { + clientKind: 'mobile', + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + } + ) + + expect(response.ok).toBe(true) + expect(runtime.restoreStructuredAgentSessionTabs).toHaveBeenCalledTimes(1) + }) + it('routes mobile-only activation without notifying desktop clients', async () => { const runtime = { getRuntimeId: () => 'test-runtime', diff --git a/src/main/runtime/rpc/methods/session-tabs.ts b/src/main/runtime/rpc/methods/session-tabs.ts index 3a322e33ed7..34d50a2a76b 100644 --- a/src/main/runtime/rpc/methods/session-tabs.ts +++ b/src/main/runtime/rpc/methods/session-tabs.ts @@ -15,6 +15,7 @@ import { import { SESSION_TAB_MARKDOWN_METHODS } from './session-tab-markdown-methods' import { SESSION_TAB_MUTATION_METHODS } from './session-tab-mutation-methods' import { restoreStructuredTabsIfSupported } from './structured-session-tab-restore' +import { isStructuredNativeChatEnabled } from './structured-agent-session-policy' import { assertLegacyAiVaultResumeCommandAllowed } from '../../../ai-vault/structured-session-ownership' export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ @@ -22,11 +23,12 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ name: 'session.tabs.list', params: WorktreeTabSelector, handler: async (params, { runtime, pairedDeviceId, clientKind, clientCapabilities }) => { - await restoreStructuredTabsIfSupported(runtime, clientCapabilities) + await restoreStructuredTabsIfSupported({ runtime, clientKind, clientCapabilities }) return projectSessionTabsForClient( await runtime.listMobileSessionTabs(params.worktree, pairedDeviceId), clientKind, - clientCapabilities + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined ) } }), @@ -34,7 +36,7 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ name: 'session.tabs.listAll', params: null, handler: async (_params, context) => { - await restoreStructuredTabsIfSupported(context.runtime, context.clientCapabilities) + await restoreStructuredTabsIfSupported(context) return listSessionTabsInventory(context) } }), @@ -89,7 +91,7 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ let unsubscribe = (): void => {} let closed = false let initialized = false - await restoreStructuredTabsIfSupported(runtime, clientCapabilities) + await restoreStructuredTabsIfSupported({ runtime, clientKind, clientCapabilities }) const initial = await runtime.listMobileSessionTabs(params.worktree, pairedDeviceId) if (closed) { return @@ -115,7 +117,12 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ } emit({ type: 'snapshot', - ...projectSessionTabsForClient(initial, clientKind, clientCapabilities) + ...projectSessionTabsForClient( + initial, + clientKind, + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined + ) }) initialized = true if (closed) { @@ -126,7 +133,12 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ if (snapshot.worktree === subscribedWorktree) { emit({ type: 'updated', - ...projectSessionTabsForClient(snapshot, clientKind, clientCapabilities) + ...projectSessionTabsForClient( + snapshot, + clientKind, + clientCapabilities, + clientKind === 'mobile' ? isStructuredNativeChatEnabled(runtime) : undefined + ) }) } }, pairedDeviceId) @@ -157,7 +169,7 @@ export const SESSION_TAB_METHODS: RpcAnyMethod[] = [ name: 'session.tabs.subscribeAll', params: null, handler: async (_params, context, emit) => { - await restoreStructuredTabsIfSupported(context.runtime, context.clientCapabilities) + await restoreStructuredTabsIfSupported(context) return subscribeSessionTabsInventory(context, emit) } }), diff --git a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts index 2dd317e08b0..33018c21f4d 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-gate.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-gate.ts @@ -5,21 +5,18 @@ // handed a session it cannot render or drive — and, just as importantly, cannot make the host EXIST // by calling into it, which is an observable side effect. -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import { getStructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-registry' import type { StructuredAgentSessionHost } from '../../../native-chat/agent-session-wire/structured-agent-session-host' import type { StructuredAgentSessionCaller } from '../../../native-chat/agent-session-wire/structured-agent-session-host-types' import type { RpcContext } from '../core' +import { supportsStructuredAgentSessions } from './structured-agent-session-policy' /** * In-process callers are the same build as the host, so they carry no negotiated * capability list; every remote client must say it can read structured sessions. */ export function supportsStructuredSessions(ctx: RpcContext): boolean { - return ( - ctx.clientKind === undefined || - (ctx.clientCapabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) ?? false) - ) + return supportsStructuredAgentSessions(ctx) } export function requireStructuredCapability(ctx: RpcContext): void { diff --git a/src/main/runtime/rpc/methods/structured-agent-session-policy.ts b/src/main/runtime/rpc/methods/structured-agent-session-policy.ts new file mode 100644 index 00000000000..4fe38474ec6 --- /dev/null +++ b/src/main/runtime/rpc/methods/structured-agent-session-policy.ts @@ -0,0 +1,47 @@ +import { + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + type RuntimeCapability +} from '../../../../shared/protocol-version' +import type { OrcaRuntimeService } from '../../orca-runtime' +import type { RpcContext } from '../core' + +type StructuredPolicyContext = Pick & { + runtime?: Pick + structuredNativeChatEnabled?: boolean +} + +export function isStructuredNativeChatEnabled( + runtime: Pick +): boolean { + try { + return runtime.getClientSettings().experimentalStructuredNativeChat === true + } catch { + return false + } +} + +export function supportsStructuredAgentSessions(context: StructuredPolicyContext): boolean { + if (context.clientKind === undefined) { + return true + } + const hasCapability = + context.clientCapabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) === true + if (!hasCapability) { + return false + } + if (context.clientKind !== 'mobile') { + return true + } + return ( + context.structuredNativeChatEnabled === true || + (context.runtime ? isStructuredNativeChatEnabled(context.runtime) : false) + ) +} + +export function structuredNativeChatProjectionEnabled(args: { + clientKind: 'mobile' | 'runtime' | undefined + clientCapabilities: readonly RuntimeCapability[] | undefined + structuredNativeChatEnabled?: boolean +}): boolean { + return supportsStructuredAgentSessions(args) +} diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index c698cc7229c..b65e6eff825 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -111,7 +111,7 @@ function hostStub(): StructuredAgentSessionHost { return hostCalls as unknown as StructuredAgentSessionHost } -function dispatcher(): RpcDispatcher { +function dispatcher(runtimeOverrides: Record = {}): RpcDispatcher { runtimeCalls = { getStructuredAgentSessionCreateSupport: vi.fn(async () => ({ supported: true })), resolveStructuredAgentSessionCreateIntent: vi.fn(async (params) => ({ @@ -134,7 +134,8 @@ function dispatcher(): RpcDispatcher { registerSubscriptionCleanup: vi.fn(), cleanupSubscription: vi.fn(), cleanupSubscriptionsByPrefix: vi.fn(), - ...runtimeCalls + ...runtimeCalls, + ...runtimeOverrides } return new RpcDispatcher({ runtime: runtime as unknown as OrcaRuntimeService, @@ -151,10 +152,11 @@ async function call( clientId?: string clientKind?: 'mobile' | 'runtime' clientCapabilities?: string[] - } + }, + runtimeOverrides: Record = {} ): Promise { const replies: RpcResponse[] = [] - await dispatcher().dispatchStreaming( + await dispatcher(runtimeOverrides).dispatchStreaming( request(method, params), (raw) => replies.push(JSON.parse(raw) as RpcResponse), client @@ -170,6 +172,10 @@ const STRUCTURED_CLIENT = { clientKind: 'runtime' as const, clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] } +const STRUCTURED_MOBILE_CLIENT = { + clientKind: 'mobile' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} beforeEach(() => { setStructuredAgentSessionHost(hostStub()) @@ -186,6 +192,18 @@ describe('capability gating', () => { expect(response).toMatchObject({ ok: true, result: { ok: true } }) expect(hostCalls.close).toHaveBeenCalledWith(SESSION) expect(hostCalls.setSessionTabVisibility).toHaveBeenCalledWith(SESSION, false) + expect(hostCalls.setSessionTabVisibility.mock.invocationCallOrder[0]).toBeLessThan( + hostCalls.close.mock.invocationCallOrder[0]! + ) + }) + + it('does not stop the provider when durable tab retirement fails', async () => { + hostCalls.setSessionTabVisibility.mockRejectedValueOnce(new Error('visibility write failed')) + + const response = await call('agentSession.close', { sessionId: SESSION }, STRUCTURED_CLIENT) + + expect(response).toMatchObject({ ok: false }) + expect(hostCalls.close).not.toHaveBeenCalled() }) it('advertises the capability without bumping the protocol version', () => { @@ -250,6 +268,25 @@ describe('capability gating', () => { expect(hostCalls.send).toHaveBeenCalledTimes(1) }) + it('requires the host structured-chat setting for mobile clients', async () => { + const response = await call('agentSession.send', sendParams(), STRUCTURED_MOBILE_CLIENT, { + getClientSettings: () => ({ experimentalStructuredNativeChat: false }) + }) + expect(response).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + expect(hostCalls.send).not.toHaveBeenCalled() + }) + + it('serves mobile clients only after capability and setting negotiation', async () => { + const response = await call('agentSession.send', sendParams(), STRUCTURED_MOBILE_CLIENT, { + getClientSettings: () => ({ experimentalStructuredNativeChat: true }) + }) + expect(response).toMatchObject({ ok: true }) + expect(hostCalls.send).toHaveBeenCalledTimes(1) + }) + it('serves an in-process caller, which negotiates no capabilities at all', async () => { const response = await call('agentSession.send', sendParams()) expect(response).toMatchObject({ ok: true }) @@ -292,6 +329,37 @@ describe('method routing', () => { ) }) + it('reports an unknown create outcome when attach commits before tab publication fails', async () => { + const worktree = 'id:workspace-1' + const params = { + envelope: envelope({ + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree, agent: 'codex' } + }) + }), + worktree, + agent: 'codex' + } + + const response = await call('agentSession.create', params, STRUCTURED_CLIENT, { + publishStructuredAgentSessionTab: vi.fn(async () => { + throw new Error('publish failed') + }) + }) + + expect(hostCalls.attach).toHaveBeenCalledOnce() + expect(response).toMatchObject({ + ok: true, + result: { + ok: false, + refusal: { code: 'agent_session_operation_unknown' } + } + }) + }) + it('separates create from ensure by the fence the client may declare', async () => { const created = await call('agentSession.create', attachParams()) expect(created).toMatchObject({ ok: true }) diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index 8ea85aa0e87..ffd23499a3e 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -91,12 +91,23 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } }) if (result.ok && resolved.agent === 'codex') { - await ctx.runtime.publishStructuredAgentSessionTab({ - workspaceId: resolved.location.workspaceId, - sessionId: result.value.sessionId, - agent: 'codex', - activate: true - }) + try { + await ctx.runtime.publishStructuredAgentSessionTab({ + workspaceId: resolved.location.workspaceId, + sessionId: result.value.sessionId, + agent: 'codex', + activate: true + }) + } catch (error) { + console.warn('[agent-session] create committed before tab publication failed', error) + return { + ok: false, + refusal: { + code: 'agent_session_operation_unknown', + message: 'The Codex chat may have been created, but its tab could not be confirmed.' + } + } + } } return result } @@ -129,11 +140,11 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ params: OptionsParams, handler: async (params, ctx) => { const host = requireHost(ctx) - await host.close(params.sessionId) // Terminal-disposal closes use this RPC without the session-tabs retirement RPC. if (typeof host.setSessionTabVisibility === 'function') { await host.setSessionTabVisibility(params.sessionId, false) } + await host.close(params.sessionId) return { ok: true as const } } }), diff --git a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts index c1f265cc4c8..4333713a445 100644 --- a/src/main/runtime/rpc/methods/structured-session-tab-restore.ts +++ b/src/main/runtime/rpc/methods/structured-session-tab-restore.ts @@ -1,11 +1,13 @@ -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { RpcContext } from '../core' +import { supportsStructuredAgentSessions } from './structured-agent-session-policy' export async function restoreStructuredTabsIfSupported( - runtime: RpcContext['runtime'], - capabilities: readonly string[] | undefined + context: Pick ): Promise { - if (capabilities?.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY)) { - await runtime.restoreStructuredAgentSessionTabs() + if ( + supportsStructuredAgentSessions(context) && + typeof context.runtime.restoreStructuredAgentSessionTabs === 'function' + ) { + await context.runtime.restoreStructuredAgentSessionTabs() } } diff --git a/src/main/runtime/rpc/rpc-streaming-dispatcher.ts b/src/main/runtime/rpc/rpc-streaming-dispatcher.ts index 75938d6573c..6eddcb748be 100644 --- a/src/main/runtime/rpc/rpc-streaming-dispatcher.ts +++ b/src/main/runtime/rpc/rpc-streaming-dispatcher.ts @@ -113,6 +113,7 @@ export class RpcStreamingDispatcher { pairedDeviceId: options?.pairedDeviceId, clientKind: options?.clientKind, clientCapabilities: options?.clientCapabilities, + updateClientCapabilities: options?.updateClientCapabilities, orchestrationCapability: request.orchestrationCapability, authenticatedCallerFingerprint: mutation?.identity.callerFingerprint ?? @@ -165,6 +166,7 @@ export class RpcStreamingDispatcher { pairedDeviceId: options?.pairedDeviceId, clientKind: options?.clientKind, clientCapabilities: options?.clientCapabilities, + updateClientCapabilities: options?.updateClientCapabilities, orchestrationCapability: request.orchestrationCapability, pairing: options?.pairing, sendBinary: options?.sendBinary, diff --git a/src/main/runtime/runtime-client-settings.ts b/src/main/runtime/runtime-client-settings.ts index 41a251ce649..b9e6959c8da 100644 --- a/src/main/runtime/runtime-client-settings.ts +++ b/src/main/runtime/runtime-client-settings.ts @@ -32,6 +32,7 @@ export type RuntimeClientSettings = Pick< | 'defaultLinearTeamSelection' | 'githubProjects' | 'experimentalNewWorktreeCardStyle' + | 'experimentalStructuredNativeChat' | 'compactWorktreeCards' | 'minimaxGroupId' | 'minimaxUsageModels' @@ -97,6 +98,7 @@ export class RuntimeClientSettingsController { defaultLinearTeamSelection: settings.defaultLinearTeamSelection ?? null, githubProjects: settings.githubProjects, experimentalNewWorktreeCardStyle: settings.experimentalNewWorktreeCardStyle === true, + experimentalStructuredNativeChat: settings.experimentalStructuredNativeChat === true, compactWorktreeCards: settings.compactWorktreeCards === true, minimaxGroupId: settings.minimaxGroupId ?? '', minimaxUsageModels: settings.minimaxUsageModels ?? 'general', diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts b/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts index 57f5a61af2f..767ca885234 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-mobile-method-allowlist.ts @@ -188,6 +188,7 @@ export const MOBILE_RPC_METHOD_ALLOWLIST = new Set([ 'repo.searchRefs', 'repo.sparsePresets', 'repo.update', + 'runtime.clientCapabilities.update', 'runtime.clientEvents.subscribe', 'runtime.clientEvents.unsubscribe', 'session.tabs.activate', @@ -201,6 +202,22 @@ export const MOBILE_RPC_METHOD_ALLOWLIST = new Set([ 'session.tabs.subscribeAll', 'session.tabs.unsubscribe', 'session.tabs.unsubscribeAll', + 'agentSession.createSupport', + 'agentSession.create', + 'agentSession.ensure', + 'agentSession.send', + 'agentSession.cancel', + 'agentSession.close', + 'agentSession.respondToApproval', + 'agentSession.respondToQuestion', + 'agentSession.setOption', + 'agentSession.handoffStatus', + 'agentSession.options', + 'agentSession.history', + 'agentSession.subscribe', + 'agentSession.unsubscribe', + 'agentSession.hold', + 'agentSession.release', 'nativeChat.readSession', 'nativeChat.subscribe', 'nativeChat.unsubscribe', diff --git a/src/main/runtime/runtime-rpc/runtime-rpc-websocket-dispatch.ts b/src/main/runtime/runtime-rpc/runtime-rpc-websocket-dispatch.ts index 86732e7430c..dd714b77528 100644 --- a/src/main/runtime/runtime-rpc/runtime-rpc-websocket-dispatch.ts +++ b/src/main/runtime/runtime-rpc/runtime-rpc-websocket-dispatch.ts @@ -141,6 +141,12 @@ export class RuntimeRpcWebSocketDispatch extends RuntimeRpcRequestAdmission { // Why: gates the mobile-only payload diet so full-screen web/desktop clients aren't truncated. clientKind: device.scope, clientCapabilities: authenticatedSocket?.clientCapabilities, + updateClientCapabilities: + authenticatedSocket && device.scope === 'mobile' + ? (clientCapabilities) => { + authenticatedSocket.clientCapabilities = clientCapabilities + } + : undefined, pairing: pairingContext, signal: abortRegistration?.signal, sendBinary, diff --git a/src/main/runtime/runtime-store-contract.ts b/src/main/runtime/runtime-store-contract.ts index 649a8ac49b7..6b9858bda0c 100644 --- a/src/main/runtime/runtime-store-contract.ts +++ b/src/main/runtime/runtime-store-contract.ts @@ -87,6 +87,7 @@ export type RuntimeStore = { terminalWindowsShell?: GlobalSettings['terminalWindowsShell'] floatingTerminalEnabled?: GlobalSettings['floatingTerminalEnabled'] agentStatusHooksEnabled?: GlobalSettings['agentStatusHooksEnabled'] + experimentalStructuredNativeChat?: GlobalSettings['experimentalStructuredNativeChat'] defaultTaskSource?: GlobalSettings['defaultTaskSource'] defaultTaskViewPreset?: GlobalSettings['defaultTaskViewPreset'] visibleTaskProviders?: GlobalSettings['visibleTaskProviders'] @@ -122,4 +123,5 @@ export type RuntimeStore = { updates: Partial, options?: { notifyListeners?: boolean; originWebContentsId?: number } ) => unknown + onSettingsChanged?: Store['onSettingsChanged'] } diff --git a/src/renderer/src/app-shell/use-app-startup-hydration.ts b/src/renderer/src/app-shell/use-app-startup-hydration.ts index 91db232081c..77da19ffd20 100644 --- a/src/renderer/src/app-shell/use-app-startup-hydration.ts +++ b/src/renderer/src/app-shell/use-app-startup-hydration.ts @@ -274,9 +274,11 @@ export function useAppStartupHydration(onOnboardingLoaded: (state: OnboardingSta await timeRendererStartupStep('recover-legacy-worker-terminals-post-reconnect', () => window.api.app.recoverLegacyWorkerTerminalsForRendererStartup() ) - await timeRendererStartupStep('project-structured-session-tabs', () => - restoreLocalStructuredSessionTabsOnce() - ) + if (useAppStore.getState().settings?.experimentalStructuredNativeChat === true) { + await timeRendererStartupStep('project-structured-session-tabs', () => + restoreLocalStructuredSessionTabsOnce() + ) + } if (cancelled) { return } diff --git a/src/renderer/src/app-startup-routing.test.ts b/src/renderer/src/app-startup-routing.test.ts index d1aad35829a..fead2f6c7bb 100644 --- a/src/renderer/src/app-startup-routing.test.ts +++ b/src/renderer/src/app-startup-routing.test.ts @@ -357,6 +357,16 @@ describe('renderer startup runtime routing', () => { expect(reconnectIndex).toBeGreaterThan(capabilityIndex) }) + it('skips startup structured tab projection while the host setting is off', () => { + const source = readSource(STARTUP_HYDRATION_PATH) + const projectIndex = source.indexOf("timeRendererStartupStep('project-structured-session-tabs'") + + expect(projectIndex).toBeGreaterThanOrEqual(0) + expect(source.slice(projectIndex - 180, projectIndex)).toContain( + 'settings?.experimentalStructuredNativeChat === true' + ) + }) + it('orders packaged restoration before adoption, projection, and default creation', () => { // Why this file: the startup sequence moved out of App.tsx into the hydration hook; // the ordering it asserts is unchanged, only the module that now spells it out. diff --git a/src/renderer/src/components/native-chat/structured-agent-session-message-projection.ts b/src/renderer/src/components/native-chat/structured-agent-session-message-projection.ts index aebc51c90e0..15c92d2efb7 100644 --- a/src/renderer/src/components/native-chat/structured-agent-session-message-projection.ts +++ b/src/renderer/src/components/native-chat/structured-agent-session-message-projection.ts @@ -1,36 +1 @@ -import type { - AgentJournalRenderItem, - AgentJournalSubmission -} from '../../../../shared/agent-session-journal-types' -import { agentJournalSubmissionKey } from '../../../../shared/agent-session-journal-item-key' -import type { NativeChatMessage } from '../../../../shared/native-chat-types' -import { - reconcileStructuredAgentSessionOutbox, - type StructuredAgentSessionOutboxEntry -} from '../../../../shared/structured-agent-session-outbox' -import { projectStructuredItemsToNativeChat } from '../../../../shared/structured-agent-session-projection' - -export function projectStructuredAgentSessionMessages( - items: readonly AgentJournalRenderItem[], - outbox: readonly StructuredAgentSessionOutboxEntry[], - submissions: readonly AgentJournalSubmission[] -): NativeChatMessage[] { - const optimistic = reconcileStructuredAgentSessionOutbox(outbox, submissions) - // Why: the host renders its own bubble off the submission WAL row, which lands - // while the dispatch is still `pending`. Reconciliation only retires the echo on - // `accepted`, so keying visibility on that alone double-rendered the bubble for - // the whole provider round trip. The entry itself stays for retry/unconfirmed. - const journalled = new Set(items.map((item) => item.itemId)) - return [ - ...projectStructuredItemsToNativeChat(items), - ...optimistic - .filter((entry) => !journalled.has(agentJournalSubmissionKey(entry.clientMessageId))) - .map((entry): NativeChatMessage => ({ - id: agentJournalSubmissionKey(entry.clientMessageId), - role: 'user', - source: 'transcript', - timestamp: entry.queuedAt, - blocks: entry.body.blocks - })) - ] -} +export { projectStructuredAgentSessionMessages } from '../../../../shared/structured-agent-session-message-projection' diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session-hold.ts b/src/renderer/src/components/native-chat/use-structured-agent-session-hold.ts index b7288d4a2a3..2c91621d30a 100644 --- a/src/renderer/src/components/native-chat/use-structured-agent-session-hold.ts +++ b/src/renderer/src/components/native-chat/use-structured-agent-session-hold.ts @@ -10,16 +10,10 @@ // would otherwise release a hold that has not landed yet, and the late hold would never be undone. import { useEffect, useRef } from 'react' +import { structuredAgentSessionHolderId } from '../../../../shared/structured-agent-session-holder' import type { RuntimeClientTarget } from '@/runtime/runtime-rpc-client' import { callStructuredAgentSession } from '@/runtime/structured-agent-session-client' -let holderOrdinal = 0 - -export function structuredAgentSessionHolderId(surface: string): string { - holderOrdinal += 1 - return `${surface}:${holderOrdinal}` -} - export function useStructuredAgentSessionHold(args: { sessionId: string target: RuntimeClientTarget diff --git a/src/renderer/src/runtime/host-session-mirror-settle-census.test.ts b/src/renderer/src/runtime/host-session-mirror-settle-census.test.ts index b1480259599..7d6a0eadc3e 100644 --- a/src/renderer/src/runtime/host-session-mirror-settle-census.test.ts +++ b/src/renderer/src/runtime/host-session-mirror-settle-census.test.ts @@ -150,8 +150,9 @@ describe('host-session-mirror settle census', () => { 'runtime/web-session-tabs-sync/visibility-resume-repair.ts': 1, // The eager post-create session.tabs.list refresh. 'runtime/web-runtime-session-snapshot.ts': 1, - // The local structured-session inventory/subscription frame. - 'runtime/local-structured-session-tabs-sync.ts': 1 + // The local structured-session mirror owns two: the inventory/subscription + // frame, and the toggle-off teardown that retracts the tabs it published. + 'runtime/local-structured-session-tabs-sync/snapshot-apply.ts': 2 }) }) @@ -196,14 +197,19 @@ describe('host-session-mirror settle census', () => { // Hydration and mirror receipts remain pinned by their extracted owners: // the global singular frame owns two hydration completions and the global // inventory frame one, initial loading owns one, active subscription owns - // two mirror settles, and visibility resume repair owns one. + // two mirror settles, and visibility resume repair owns one. The local + // structured-session apply module owns one settle per direction: the + // snapshot it mirrors in, and the teardown that retracts it. 'runtime/web-session-tabs-sync/active-session-subscription.ts': { settle: 2 }, 'runtime/web-session-tabs-sync/global-session-events.ts': { settleHydration: 2 }, 'runtime/web-session-tabs-sync/global-session-inventory-event.ts': { settleHydration: 1 }, 'runtime/web-session-tabs-sync/load-initial.ts': { settleHydration: 1 }, 'runtime/web-session-tabs-sync/visibility-resume-repair.ts': { settle: 1 }, 'runtime/web-runtime-session-snapshot.ts': { settleMirror: 1 }, - 'runtime/local-structured-session-tabs-sync.ts': { settleStructuredSessionMirror: 1 } + 'runtime/local-structured-session-tabs-sync/snapshot-apply.ts': { + settleStructuredSessionClear: 1, + settleStructuredSessionMirror: 1 + } }) }) diff --git a/src/renderer/src/runtime/local-structured-session-tab-retirement.ts b/src/renderer/src/runtime/local-structured-session-tab-retirement.ts new file mode 100644 index 00000000000..81a8d0d5ef5 --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tab-retirement.ts @@ -0,0 +1,64 @@ +import type { WorktreeRuntimeOwnerState } from '../lib/worktree-runtime-owner' +import { folderWorkspaceKey } from '../../../shared/workspace-scope' +import { applyWebSessionTabsSnapshot } from './web-session-tabs-sync' +import type { WebSessionTabsSyncState } from './web-session-tabs-sync' + +export type StructuredSessionTabPublicationVersion = { + publicationEpoch: string + snapshotVersion: number +} + +export function knownStructuredSessionWorktreeIds( + state: WebSessionTabsSyncState & WorktreeRuntimeOwnerState +): Set { + const ids = new Set(Object.keys(state.unifiedTabsByWorktree)) + for (const worktrees of Object.values(state.worktreesByRepo ?? {})) { + for (const worktree of worktrees) { + ids.add(worktree.id) + } + } + for (const detected of Object.values(state.detectedWorktreesByRepo ?? {})) { + for (const worktree of detected.worktrees) { + ids.add(worktree.id) + } + } + for (const workspace of state.folderWorkspaces ?? []) { + ids.add(folderWorkspaceKey(workspace.id)) + } + return ids +} + +export function removeStructuredSessionTabsForVersions< + State extends WebSessionTabsSyncState & WorktreeRuntimeOwnerState +>( + state: State, + versions: Iterable, + owner: string, + now: number +): State { + let next = state + for (const [worktree, version] of versions) { + const patch = applyWebSessionTabsSnapshot( + next, + { + worktree, + publicationEpoch: version.publicationEpoch, + snapshotVersion: version.snapshotVersion + 1, + activeGroupId: null, + activeTabId: null, + activeTabType: null, + tabGroups: [], + tabs: [] + }, + owner, + now, + { + contentScope: 'agent-session', + preserveLocalLayout: true, + terminalPtyMode: 'local' + } + ) + next = patch === next ? next : ({ ...next, ...patch } as State) + } + return next +} diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts index 59be9904e2a..7a6c713f37f 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts @@ -10,7 +10,10 @@ import { buildPersistedUnifiedTabSessionData } from '../lib/workspace-session-un import { buildHydratedTabState } from '../store/slices/tabs-hydration' import { applyLocalStructuredSessionTabSnapshots, + clearLocalStructuredSessionTabs, projectLocalStructuredSessionTabs, + removeLocalStructuredSessionTabs, + refreshLocalStructuredSessionTabs, resetLocalStructuredSessionVersionForTests, startLocalStructuredSessionTabsSync } from './local-structured-session-tabs-sync' @@ -160,6 +163,19 @@ function expectExactSplit(state: { } describe('local structured session tab projection', () => { + it('removes only locally mirrored structured tabs when the feature is disabled', () => { + const mirrored = applyLocalStructuredSessionTabSnapshots(createSnapshot(), [ + structuredInventory('epoch-1', 1, 'codex-1') + ]) + + const disabled = removeLocalStructuredSessionTabs(mirrored) + + expect(disabled.unifiedTabsByWorktree[WORKTREE_ID]).toEqual([ + expect.objectContaining({ id: TERMINAL_ID, contentType: 'terminal' }) + ]) + expect(disabled.activeTabTypeByWorktree[WORKTREE_ID]).toBe('terminal') + }) + it('reconnects after a streaming subscription reports an error', async () => { vi.useFakeTimers() const priorApi = window.api @@ -217,6 +233,80 @@ describe('local structured session tab projection', () => { } }) + it('ignores an in-flight inventory response after toggle-off clears the mirror', async () => { + let resolveInventory: ((response: unknown) => void) | undefined + const pendingInventory = new Promise((resolve) => { + resolveInventory = resolve + }) + const priorApi = window.api + Object.defineProperty(window, 'api', { + configurable: true, + value: { + runtime: { + call: vi.fn().mockReturnValue(pendingInventory) + } + } + }) + try { + const refresh = refreshLocalStructuredSessionTabs() + clearLocalStructuredSessionTabs() + resolveInventory?.({ + ok: true, + result: { snapshots: [structuredInventory('epoch-1', 8, 'stale-session')] } + }) + await refresh + + const fresh = applyLocalStructuredSessionTabSnapshots(createSnapshot(), [ + structuredInventory('epoch-1', 1, 'fresh-session') + ]) + expect(fresh.unifiedTabsByWorktree[WORKTREE_ID]).toEqual( + expect.arrayContaining([expect.objectContaining({ entityId: 'fresh-session' })]) + ) + } finally { + Object.defineProperty(window, 'api', { configurable: true, value: priorApi }) + } + }) + + it('ignores a subscription frame after toggle-off clears the mirror', async () => { + const callbacks: ((response: unknown) => void)[] = [] + const priorApi = window.api + Object.defineProperty(window, 'api', { + configurable: true, + value: { + runtime: { + getStatus: vi.fn().mockResolvedValue({ + capabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + }), + call: vi.fn().mockResolvedValue({ ok: true, result: { snapshots: [] } }), + subscribe: vi.fn(async (_args: unknown, callback: (response: unknown) => void) => { + callbacks.push(callback) + return { unsubscribe: vi.fn() } + }) + } + } + }) + let unsubscribe = (): void => {} + try { + await startLocalStructuredSessionTabsSync({ + isDisposed: () => false, + setUnsubscribe: (next) => { + unsubscribe = next + } + }) + clearLocalStructuredSessionTabs() + callbacks[0]?.({ ok: true, result: structuredInventory('epoch-1', 8, 'stale-session') }) + const fresh = applyLocalStructuredSessionTabSnapshots(createSnapshot(), [ + structuredInventory('epoch-1', 1, 'fresh-session') + ]) + expect(fresh.unifiedTabsByWorktree[WORKTREE_ID]).toEqual( + expect.arrayContaining([expect.objectContaining({ entityId: 'fresh-session' })]) + ) + } finally { + unsubscribe() + Object.defineProperty(window, 'api', { configurable: true, value: priorApi }) + } + }) + it('accepts a newer session after merged content returns to the base epoch', () => { const state = createSnapshot() const base = { diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync.ts index f5a9fa9807c..722bd93af8b 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync.ts @@ -1,293 +1,36 @@ import { useEffect } from 'react' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../shared/protocol-version' -import type { RuntimeMobileSessionTabsResult } from '../../../shared/runtime-types' -import { folderWorkspaceKey } from '../../../shared/workspace-scope' import { useAppStore } from '../store' -import type { WorktreeRuntimeOwnerState } from '../lib/worktree-runtime-owner' -import { getExecutionHostIdForWorktree } from '../lib/worktree-runtime-owner' -import { applyWebSessionTabsSnapshot, applyWebSessionTabsStorePatch } from './web-session-tabs-sync' -import type { WebSessionTabsSyncState } from './web-session-tabs-sync' -import { - noteRetiredValue, - sameSessionTabsPublicationLineage -} from './web-session-tabs-sync/publisher-identity-fences' -import type { SessionTabsPublicationEpochHistory } from './web-session-tabs-sync/state' -import { refreshLocalRuntimeCapabilities } from './local-runtime-capabilities' +import { clearLocalStructuredSessionTabs } from './local-structured-session-tabs-sync/snapshot-apply' +import { startLocalStructuredSessionTabsSync } from './local-structured-session-tabs-sync/subscription' -export const LOCAL_STRUCTURED_SESSION_OWNER = 'local-structured-session' -let localStructuredSessionTabsRestorePromise: Promise | null = null -const localStructuredSessionVersionByWorktree = new Map< - string, - { publicationEpoch: string; snapshotVersion: number } ->() -const localStructuredSessionEpochHistoryByWorktree = new Map< - string, - SessionTabsPublicationEpochHistory ->() - -export function resetLocalStructuredSessionVersionForTests(): void { - localStructuredSessionVersionByWorktree.clear() - localStructuredSessionEpochHistoryByWorktree.clear() -} - -type SessionTabsEvent = - | (RuntimeMobileSessionTabsResult & { type: 'snapshot' | 'updated' }) - | { type: 'snapshots'; snapshots: RuntimeMobileSessionTabsResult[] } - | { type: 'end' } - -export function projectLocalStructuredSessionTabs( - snapshot: RuntimeMobileSessionTabsResult -): RuntimeMobileSessionTabsResult { - const structuredIds = new Set( - snapshot.tabs.filter((tab) => tab.type === 'agent-session').map((tab) => tab.id) - ) - const visibleHostTabIds = structuredIds - const visibleIds = structuredIds - const projectedTabGroups = snapshot.tabGroups - ?.map((group) => ({ - ...group, - tabOrder: group.tabOrder.filter((id) => visibleHostTabIds.has(id)), - activeTabId: - group.activeTabId && visibleHostTabIds.has(group.activeTabId) ? group.activeTabId : null, - recentTabIds: group.recentTabIds?.filter((id) => visibleHostTabIds.has(id)) - })) - .filter((group) => group.tabOrder.length > 0) - - return { - ...snapshot, - activeTabId: visibleIds.has(snapshot.activeTabId ?? '') ? snapshot.activeTabId : null, - activeTabType: - snapshot.activeTabId && visibleIds.has(snapshot.activeTabId) ? snapshot.activeTabType : null, - activeGroupId: - snapshot.activeGroupId && - projectedTabGroups?.some((group) => group.id === snapshot.activeGroupId) - ? snapshot.activeGroupId - : (projectedTabGroups?.[0]?.id ?? null), - tabs: snapshot.tabs.filter((tab) => visibleIds.has(tab.id)), - tabGroups: projectedTabGroups, - // Why: group membership locates chats; the renderer's split tree remains locally authoritative. - tabGroupLayout: undefined - } -} - -export function applyStructuredSessionTabSnapshots( - snapshots: readonly RuntimeMobileSessionTabsResult[], - owner = LOCAL_STRUCTURED_SESSION_OWNER -): void { - const settleStructuredSessionMirror = applyWebSessionTabsStorePatch( - (state) => applyLocalStructuredSessionTabSnapshots(state, snapshots, owner), - { frames: [] } - ) - settleStructuredSessionMirror() -} - -export function applyLocalStructuredSessionTabSnapshots< - State extends WebSessionTabsSyncState & WorktreeRuntimeOwnerState ->( - state: State, - snapshots: readonly RuntimeMobileSessionTabsResult[], - owner = LOCAL_STRUCTURED_SESSION_OWNER, - now = Date.now() -): State { - let next = state - for (const snapshot of snapshots) { - // Why: the execution host owns its tabs; local inventory must not rewrite paired or SSH panes. - if (getExecutionHostIdForWorktree(next, snapshot.worktree) !== 'local') { - continue - } - const prior = localStructuredSessionVersionByWorktree.get(snapshot.worktree) - const sharesLineage = Boolean( - prior && sameSessionTabsPublicationLineage(prior.publicationEpoch, snapshot.publicationEpoch) - ) - const epochHistory = localStructuredSessionEpochHistoryByWorktree.get(snapshot.worktree) - if (epochHistory?.retired.includes(snapshot.publicationEpoch) && !sharesLineage) { - continue - } - if (prior && sharesLineage && snapshot.snapshotVersion <= prior.snapshotVersion) { - continue - } - const patch = applyWebSessionTabsSnapshot( - next, - projectLocalStructuredSessionTabs(snapshot), - owner, - now, - { - contentScope: 'agent-session', - preserveLocalLayout: true, - terminalPtyMode: 'local' - } - ) - next = patch === next ? next : ({ ...next, ...patch } as State) - localStructuredSessionVersionByWorktree.set(snapshot.worktree, { - publicationEpoch: snapshot.publicationEpoch, - snapshotVersion: snapshot.snapshotVersion - }) - localStructuredSessionEpochHistoryByWorktree.set( - snapshot.worktree, - noteRetiredValue(epochHistory, snapshot.publicationEpoch, 8) - ) - } - // Drop publisher cursors for worktrees that no longer exist. Without this, - // every deleted worktree leaves an entry for the lifetime of the renderer. - const knownWorktreeIds = new Set(Object.keys(next.unifiedTabsByWorktree)) - for (const worktrees of Object.values(next.worktreesByRepo ?? {})) { - for (const worktree of worktrees) { - knownWorktreeIds.add(worktree.id) - } - } - for (const detected of Object.values(next.detectedWorktreesByRepo ?? {})) { - for (const worktree of detected.worktrees) { - knownWorktreeIds.add(worktree.id) - } - } - for (const workspace of next.folderWorkspaces ?? []) { - knownWorktreeIds.add(folderWorkspaceKey(workspace.id)) - } - for (const worktreeId of localStructuredSessionVersionByWorktree.keys()) { - if (!knownWorktreeIds.has(worktreeId)) { - localStructuredSessionVersionByWorktree.delete(worktreeId) - localStructuredSessionEpochHistoryByWorktree.delete(worktreeId) - } - } - return next -} - -export function restoreLocalStructuredSessionTabsOnce(): Promise { - localStructuredSessionTabsRestorePromise ??= refreshLocalRuntimeCapabilities() - .then(() => refreshLocalStructuredSessionTabs()) - .then(() => undefined) - .catch((error) => { - localStructuredSessionTabsRestorePromise = null - throw error - }) - return localStructuredSessionTabsRestorePromise -} - -/** Fetch the current host inventory even after the startup restore has settled. */ -export function refreshLocalStructuredSessionTabs(): Promise { - return window.api.runtime - .call({ method: 'session.tabs.listAll', params: {} }) - .then((response) => { - if (!response.ok) { - throw new Error('structured session inventory unavailable') - } - const result = response.result as { snapshots?: RuntimeMobileSessionTabsResult[] } - const snapshots = result.snapshots ?? [] - applyStructuredSessionTabSnapshots(snapshots) - return snapshots - }) -} - -export async function startLocalStructuredSessionTabsSync(args: { - isDisposed: () => boolean - setUnsubscribe: (unsubscribe: () => void) => void -}): Promise { - const capabilities = await refreshLocalRuntimeCapabilities() - if (args.isDisposed()) { - return - } - const supported = capabilities.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) - await restoreLocalStructuredSessionTabsOnce() - if (args.isDisposed()) { - return - } - if (!supported) { - return - } - let subscriptionGeneration = 0 - let reconnectTimer: ReturnType | null = null - let reconnectAttempt = 0 - let activeHandle: { unsubscribe: () => void } | null = null - const scheduleSubscribeRetry = (): void => { - if (args.isDisposed() || reconnectTimer !== null) { - return - } - const reconnectDelay = Math.min(250 * 2 ** reconnectAttempt, 5000) - reconnectAttempt += 1 - reconnectTimer = setTimeout(() => { - reconnectTimer = null - void refreshLocalStructuredSessionTabs() - .catch((error) => console.warn('[structured-session-tabs] resync failed', error)) - .finally(() => { - if (!args.isDisposed()) { - void subscribeCurrent().catch((error) => { - console.warn('[structured-session-tabs] resubscribe failed', error) - scheduleSubscribeRetry() - }) - } - }) - }, reconnectDelay) - } - const subscribeCurrent = async (): Promise => { - if (args.isDisposed()) { - return - } - const generation = ++subscriptionGeneration - let handle: { unsubscribe: () => void } | null = null - handle = await window.api.runtime.subscribe( - { method: 'session.tabs.subscribeAll', params: {} }, - (response) => { - if (args.isDisposed() || generation !== subscriptionGeneration) { - return - } - if (!response.ok) { - // A streaming RPC can terminate with an error response before its - // handle resolves; fence that generation and retry the subscription. - subscriptionGeneration += 1 - handle?.unsubscribe() - if (activeHandle === handle) { - activeHandle = null - } - scheduleSubscribeRetry() - return - } - const event = response.result as SessionTabsEvent - if (event.type === 'snapshots') { - applyStructuredSessionTabSnapshots(event.snapshots) - } else if (event.type === 'snapshot' || event.type === 'updated') { - applyStructuredSessionTabSnapshots([event]) - } else if (event.type === 'end' && generation === subscriptionGeneration) { - // Reattach with one refresh so a runtime-restart boundary cannot strand stale tabs. - subscriptionGeneration += 1 - handle?.unsubscribe() - if (activeHandle === handle) { - activeHandle = null - } - if (reconnectTimer !== null) { - clearTimeout(reconnectTimer) - } - scheduleSubscribeRetry() - } - } - ) - if (args.isDisposed() || generation !== subscriptionGeneration) { - handle.unsubscribe() - } else { - activeHandle = handle - } - } - args.setUnsubscribe(() => { - if (reconnectTimer !== null) { - clearTimeout(reconnectTimer) - reconnectTimer = null - } - activeHandle?.unsubscribe() - activeHandle = null - }) - void subscribeCurrent().catch((error) => { - console.warn('[structured-session-tabs] subscribe failed', error) - scheduleSubscribeRetry() - }) -} +export { resetLocalStructuredSessionVersionForTests } from './local-structured-session-tabs-sync/inventory-generation-fence' +export { + refreshLocalStructuredSessionTabs, + restoreLocalStructuredSessionTabsOnce +} from './local-structured-session-tabs-sync/inventory-refresh' +export { + applyLocalStructuredSessionTabSnapshots, + applyStructuredSessionTabSnapshots, + clearLocalStructuredSessionTabs, + LOCAL_STRUCTURED_SESSION_OWNER, + removeLocalStructuredSessionTabs +} from './local-structured-session-tabs-sync/snapshot-apply' +export { projectLocalStructuredSessionTabs } from './local-structured-session-tabs-sync/snapshot-projection' +export { startLocalStructuredSessionTabsSync } from './local-structured-session-tabs-sync/subscription' export function useLocalStructuredSessionTabsSync(): void { const ready = useAppStore( (state) => state.workspaceSessionReady && state.terminalStartupRestorationReady ) + const enabled = useAppStore((state) => state.settings?.experimentalStructuredNativeChat === true) useEffect(() => { if (!ready) { return } + if (!enabled) { + clearLocalStructuredSessionTabs() + return + } let disposed = false let unsubscribe = (): void => {} void startLocalStructuredSessionTabsSync({ @@ -300,5 +43,5 @@ export function useLocalStructuredSessionTabsSync(): void { disposed = true unsubscribe() } - }, [ready]) + }, [enabled, ready]) } diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-generation-fence.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-generation-fence.ts new file mode 100644 index 00000000000..8ab68385f9e --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-generation-fence.ts @@ -0,0 +1,57 @@ +import type { SessionTabsPublicationEpochHistory } from '../web-session-tabs-sync/state' +import type { StructuredSessionTabPublicationVersion } from '../local-structured-session-tab-retirement' + +// Everything a toggle-off must invalidate: which publisher instance the renderer +// is listening to, which publication it already accepted per worktree, and the +// one-shot startup restore. A response in flight for a superseded instance must +// never reach the mirror, so every async entry point carries the generation it +// was started under and re-checks it before applying. +let syncGeneration = 0 +let restorePromise: Promise | null = null + +export const localStructuredSessionVersionByWorktree = new Map< + string, + StructuredSessionTabPublicationVersion +>() +export const localStructuredSessionEpochHistoryByWorktree = new Map< + string, + SessionTabsPublicationEpochHistory +>() + +export function localStructuredSessionGeneration(): number { + return syncGeneration +} + +export function isCurrentLocalStructuredSessionGeneration(generation: number): boolean { + return generation === syncGeneration +} + +/** Retire the current publisher instance: responses already in flight stop applying. */ +export function supersedeLocalStructuredSessionGeneration(): void { + syncGeneration += 1 +} + +// Separate from superseding because a teardown still has to publish the retiring +// cursors as retracted tabs before it may forget them. +export function forgetLocalStructuredSessionPublicationCursors(): void { + localStructuredSessionVersionByWorktree.clear() + localStructuredSessionEpochHistoryByWorktree.clear() +} + +export function dropLocalStructuredSessionRestoreLatch(): void { + restorePromise = null +} + +/** Latch the startup restore, releasing it on failure so a retry can re-run it. */ +export function latchLocalStructuredSessionRestore(start: () => Promise): Promise { + restorePromise ??= start().catch((error: unknown) => { + restorePromise = null + throw error + }) + return restorePromise +} + +export function resetLocalStructuredSessionVersionForTests(): void { + supersedeLocalStructuredSessionGeneration() + forgetLocalStructuredSessionPublicationCursors() +} diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts new file mode 100644 index 00000000000..d0f29ec0cf8 --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts @@ -0,0 +1,37 @@ +import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import { refreshLocalRuntimeCapabilities } from '../local-runtime-capabilities' +import { + isCurrentLocalStructuredSessionGeneration, + latchLocalStructuredSessionRestore, + localStructuredSessionGeneration +} from './inventory-generation-fence' +import { applyStructuredSessionTabSnapshots } from './snapshot-apply' + +export function restoreLocalStructuredSessionTabsOnce( + expectedGeneration = localStructuredSessionGeneration() +): Promise { + return latchLocalStructuredSessionRestore(() => + refreshLocalRuntimeCapabilities() + .then(() => refreshLocalStructuredSessionTabs(expectedGeneration)) + .then(() => undefined) + ) +} + +/** Fetch the current host inventory even after the startup restore has settled. */ +export function refreshLocalStructuredSessionTabs( + expectedGeneration = localStructuredSessionGeneration() +): Promise { + return window.api.runtime + .call({ method: 'session.tabs.listAll', params: {} }) + .then((response) => { + if (!response.ok) { + throw new Error('structured session inventory unavailable') + } + const result = response.result as { snapshots?: RuntimeMobileSessionTabsResult[] } + const snapshots = result.snapshots ?? [] + if (isCurrentLocalStructuredSessionGeneration(expectedGeneration)) { + applyStructuredSessionTabSnapshots(snapshots) + } + return snapshots + }) +} diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts new file mode 100644 index 00000000000..fc254de62dc --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-apply.ts @@ -0,0 +1,118 @@ +import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import type { WorktreeRuntimeOwnerState } from '../../lib/worktree-runtime-owner' +import { getExecutionHostIdForWorktree } from '../../lib/worktree-runtime-owner' +import { + applyWebSessionTabsSnapshot, + applyWebSessionTabsStorePatch +} from '../web-session-tabs-sync' +import type { WebSessionTabsSyncState } from '../web-session-tabs-sync' +import { + noteRetiredValue, + sameSessionTabsPublicationLineage +} from '../web-session-tabs-sync/publisher-identity-fences' +import { + knownStructuredSessionWorktreeIds, + removeStructuredSessionTabsForVersions +} from '../local-structured-session-tab-retirement' +import { + dropLocalStructuredSessionRestoreLatch, + forgetLocalStructuredSessionPublicationCursors, + localStructuredSessionEpochHistoryByWorktree, + localStructuredSessionVersionByWorktree, + supersedeLocalStructuredSessionGeneration +} from './inventory-generation-fence' +import { projectLocalStructuredSessionTabs } from './snapshot-projection' + +export const LOCAL_STRUCTURED_SESSION_OWNER = 'local-structured-session' + +export function applyStructuredSessionTabSnapshots( + snapshots: readonly RuntimeMobileSessionTabsResult[], + owner = LOCAL_STRUCTURED_SESSION_OWNER +): void { + const settleStructuredSessionMirror = applyWebSessionTabsStorePatch( + (state) => applyLocalStructuredSessionTabSnapshots(state, snapshots, owner), + { frames: [] } + ) + settleStructuredSessionMirror() +} + +export function removeLocalStructuredSessionTabs< + State extends WebSessionTabsSyncState & WorktreeRuntimeOwnerState +>(state: State, owner = LOCAL_STRUCTURED_SESSION_OWNER, now = Date.now()): State { + return removeStructuredSessionTabsForVersions( + state, + localStructuredSessionVersionByWorktree, + owner, + now + ) +} + +export function clearLocalStructuredSessionTabs(): void { + // Fence responses from the previous enabled instance before clearing its mirror. + supersedeLocalStructuredSessionGeneration() + const settleStructuredSessionClear = applyWebSessionTabsStorePatch( + (state) => removeLocalStructuredSessionTabs(state), + { frames: [] } + ) + settleStructuredSessionClear() + dropLocalStructuredSessionRestoreLatch() + forgetLocalStructuredSessionPublicationCursors() +} + +export function applyLocalStructuredSessionTabSnapshots< + State extends WebSessionTabsSyncState & WorktreeRuntimeOwnerState +>( + state: State, + snapshots: readonly RuntimeMobileSessionTabsResult[], + owner = LOCAL_STRUCTURED_SESSION_OWNER, + now = Date.now() +): State { + let next = state + for (const snapshot of snapshots) { + // Why: the execution host owns its tabs; local inventory must not rewrite paired or SSH panes. + if (getExecutionHostIdForWorktree(next, snapshot.worktree) !== 'local') { + continue + } + const prior = localStructuredSessionVersionByWorktree.get(snapshot.worktree) + const sharesLineage = Boolean( + prior && sameSessionTabsPublicationLineage(prior.publicationEpoch, snapshot.publicationEpoch) + ) + const epochHistory = localStructuredSessionEpochHistoryByWorktree.get(snapshot.worktree) + if (epochHistory?.retired.includes(snapshot.publicationEpoch) && !sharesLineage) { + continue + } + if (prior && sharesLineage && snapshot.snapshotVersion <= prior.snapshotVersion) { + continue + } + const patch = applyWebSessionTabsSnapshot( + next, + projectLocalStructuredSessionTabs(snapshot), + owner, + now, + { + contentScope: 'agent-session', + preserveLocalLayout: true, + terminalPtyMode: 'local' + } + ) + next = patch === next ? next : ({ ...next, ...patch } as State) + localStructuredSessionVersionByWorktree.set(snapshot.worktree, { + publicationEpoch: snapshot.publicationEpoch, + snapshotVersion: snapshot.snapshotVersion + }) + localStructuredSessionEpochHistoryByWorktree.set( + snapshot.worktree, + noteRetiredValue(epochHistory, snapshot.publicationEpoch, 8) + ) + } + // Drop publisher cursors for worktrees that no longer exist. Without this, + // every deleted worktree leaves an entry for the lifetime of the renderer. + const knownWorktreeIds = knownStructuredSessionWorktreeIds(next) + for (const worktreeId of localStructuredSessionVersionByWorktree.keys()) { + if (!knownWorktreeIds.has(worktreeId)) { + localStructuredSessionVersionByWorktree.delete(worktreeId) + localStructuredSessionEpochHistoryByWorktree.delete(worktreeId) + } + } + return next +} diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-projection.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-projection.ts new file mode 100644 index 00000000000..b6d7afc5048 --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/snapshot-projection.ts @@ -0,0 +1,37 @@ +import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' + +/** Narrow a host inventory snapshot to the structured agent-session tabs it publishes. */ +export function projectLocalStructuredSessionTabs( + snapshot: RuntimeMobileSessionTabsResult +): RuntimeMobileSessionTabsResult { + const structuredIds = new Set( + snapshot.tabs.filter((tab) => tab.type === 'agent-session').map((tab) => tab.id) + ) + const visibleHostTabIds = structuredIds + const visibleIds = structuredIds + const projectedTabGroups = snapshot.tabGroups + ?.map((group) => ({ + ...group, + tabOrder: group.tabOrder.filter((id) => visibleHostTabIds.has(id)), + activeTabId: + group.activeTabId && visibleHostTabIds.has(group.activeTabId) ? group.activeTabId : null, + recentTabIds: group.recentTabIds?.filter((id) => visibleHostTabIds.has(id)) + })) + .filter((group) => group.tabOrder.length > 0) + + return { + ...snapshot, + activeTabId: visibleIds.has(snapshot.activeTabId ?? '') ? snapshot.activeTabId : null, + activeTabType: + snapshot.activeTabId && visibleIds.has(snapshot.activeTabId) ? snapshot.activeTabType : null, + activeGroupId: + snapshot.activeGroupId && + projectedTabGroups?.some((group) => group.id === snapshot.activeGroupId) + ? snapshot.activeGroupId + : (projectedTabGroups?.[0]?.id ?? null), + tabs: snapshot.tabs.filter((tab) => visibleIds.has(tab.id)), + tabGroups: projectedTabGroups, + // Why: group membership locates chats; the renderer's split tree remains locally authoritative. + tabGroupLayout: undefined + } +} diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts new file mode 100644 index 00000000000..b074cfb1c41 --- /dev/null +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/subscription.ts @@ -0,0 +1,122 @@ +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' +import type { RuntimeMobileSessionTabsResult } from '../../../../shared/runtime-types' +import { refreshLocalRuntimeCapabilities } from '../local-runtime-capabilities' +import { + isCurrentLocalStructuredSessionGeneration, + localStructuredSessionGeneration +} from './inventory-generation-fence' +import { + refreshLocalStructuredSessionTabs, + restoreLocalStructuredSessionTabsOnce +} from './inventory-refresh' +import { applyStructuredSessionTabSnapshots } from './snapshot-apply' + +type SessionTabsEvent = + | (RuntimeMobileSessionTabsResult & { type: 'snapshot' | 'updated' }) + | { type: 'snapshots'; snapshots: RuntimeMobileSessionTabsResult[] } + | { type: 'end' } + +export async function startLocalStructuredSessionTabsSync(args: { + isDisposed: () => boolean + setUnsubscribe: (unsubscribe: () => void) => void +}): Promise { + const syncGeneration = localStructuredSessionGeneration() + const isCurrent = (): boolean => + !args.isDisposed() && isCurrentLocalStructuredSessionGeneration(syncGeneration) + const capabilities = await refreshLocalRuntimeCapabilities() + if (!isCurrent()) { + return + } + const supported = capabilities.includes(STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + await restoreLocalStructuredSessionTabsOnce(syncGeneration) + if (!isCurrent()) { + return + } + if (!supported) { + return + } + let subscriptionGeneration = 0 + let reconnectTimer: ReturnType | null = null + let reconnectAttempt = 0 + let activeHandle: { unsubscribe: () => void } | null = null + const scheduleSubscribeRetry = (): void => { + if (!isCurrent() || reconnectTimer !== null) { + return + } + const reconnectDelay = Math.min(250 * 2 ** reconnectAttempt, 5000) + reconnectAttempt += 1 + reconnectTimer = setTimeout(() => { + reconnectTimer = null + void refreshLocalStructuredSessionTabs(syncGeneration) + .catch((error) => console.warn('[structured-session-tabs] resync failed', error)) + .finally(() => { + if (isCurrent()) { + void subscribeCurrent().catch((error) => { + console.warn('[structured-session-tabs] resubscribe failed', error) + scheduleSubscribeRetry() + }) + } + }) + }, reconnectDelay) + } + const subscribeCurrent = async (): Promise => { + if (!isCurrent()) { + return + } + const generation = ++subscriptionGeneration + let handle: { unsubscribe: () => void } | null = null + handle = await window.api.runtime.subscribe( + { method: 'session.tabs.subscribeAll', params: {} }, + (response) => { + if (!isCurrent() || generation !== subscriptionGeneration) { + return + } + if (!response.ok) { + // A streaming RPC can terminate with an error response before its + // handle resolves; fence that generation and retry the subscription. + subscriptionGeneration += 1 + handle?.unsubscribe() + if (activeHandle === handle) { + activeHandle = null + } + scheduleSubscribeRetry() + return + } + const event = response.result as SessionTabsEvent + if (event.type === 'snapshots') { + applyStructuredSessionTabSnapshots(event.snapshots) + } else if (event.type === 'snapshot' || event.type === 'updated') { + applyStructuredSessionTabSnapshots([event]) + } else if (event.type === 'end' && generation === subscriptionGeneration) { + // Reattach with one refresh so a runtime-restart boundary cannot strand stale tabs. + subscriptionGeneration += 1 + handle?.unsubscribe() + if (activeHandle === handle) { + activeHandle = null + } + if (reconnectTimer !== null) { + clearTimeout(reconnectTimer) + } + scheduleSubscribeRetry() + } + } + ) + if (!isCurrent() || generation !== subscriptionGeneration) { + handle.unsubscribe() + } else { + activeHandle = handle + } + } + args.setUnsubscribe(() => { + if (reconnectTimer !== null) { + clearTimeout(reconnectTimer) + reconnectTimer = null + } + activeHandle?.unsubscribe() + activeHandle = null + }) + void subscribeCurrent().catch((error) => { + console.warn('[structured-session-tabs] subscribe failed', error) + scheduleSubscribeRetry() + }) +} diff --git a/src/shared/structured-agent-session-holder.ts b/src/shared/structured-agent-session-holder.ts new file mode 100644 index 00000000000..6da3be15611 --- /dev/null +++ b/src/shared/structured-agent-session-holder.ts @@ -0,0 +1,6 @@ +let holderOrdinal = 0 + +export function structuredAgentSessionHolderId(surface: string): string { + holderOrdinal += 1 + return `${surface}:${holderOrdinal}` +} diff --git a/src/shared/structured-agent-session-message-projection.ts b/src/shared/structured-agent-session-message-projection.ts new file mode 100644 index 00000000000..c6735a8c772 --- /dev/null +++ b/src/shared/structured-agent-session-message-projection.ts @@ -0,0 +1,29 @@ +import type { AgentJournalRenderItem, AgentJournalSubmission } from './agent-session-journal-types' +import { agentJournalSubmissionKey } from './agent-session-journal-item-key' +import type { NativeChatMessage } from './native-chat-types' +import { + reconcileStructuredAgentSessionOutbox, + type StructuredAgentSessionOutboxEntry +} from './structured-agent-session-outbox' +import { projectStructuredItemsToNativeChat } from './structured-agent-session-projection' + +export function projectStructuredAgentSessionMessages( + items: readonly AgentJournalRenderItem[], + outbox: readonly StructuredAgentSessionOutboxEntry[], + submissions: readonly AgentJournalSubmission[] +): NativeChatMessage[] { + const optimistic = reconcileStructuredAgentSessionOutbox(outbox, submissions) + const journalled = new Set(items.map((item) => item.itemId)) + return [ + ...projectStructuredItemsToNativeChat(items), + ...optimistic + .filter((entry) => !journalled.has(agentJournalSubmissionKey(entry.clientMessageId))) + .map((entry): NativeChatMessage => ({ + id: agentJournalSubmissionKey(entry.clientMessageId), + role: 'user', + source: 'transcript', + timestamp: entry.queuedAt, + blocks: entry.body.blocks + })) + ] +} diff --git a/src/shared/structured-agent-session-reducer.ts b/src/shared/structured-agent-session-reducer.ts index 24b500fc2b3..48029b5910e 100644 --- a/src/shared/structured-agent-session-reducer.ts +++ b/src/shared/structured-agent-session-reducer.ts @@ -95,7 +95,8 @@ export function reduceStructuredAgentSession( action: StructuredAgentSessionAction ): StructuredAgentSessionState { if (action.type === 'loading') { - return { ...EMPTY_STRUCTURED_AGENT_SESSION, status: 'loading' } + // Keep the last transcript visible while a reconnect rehydrates the stream. + return { ...state, status: 'loading', error: undefined } } if (action.type === 'error') { return { ...state, status: 'error', error: action.message } diff --git a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts index 0ce15ee6563..d79c99b679d 100644 --- a/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts +++ b/tests/e2e/cross-version-wire/cross-version-agent-session-wire.unit.test.ts @@ -35,6 +35,7 @@ const SESSION = 'session-alpha' const WORKSPACE = 'workspace-1' const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' const NOW = 1_800_000_000_000 +const CLIENT_CAPABILITY_UPDATE_METHOD = 'runtime.clientCapabilities.update' /** Every method the structured surface publishes: the host method it must reach, * and the result it must hand back. A gate that hides one method and leaks @@ -484,6 +485,49 @@ describe('cross-version structured agent sessions', () => { ) }) + describe('post-auth mobile capability negotiation', () => { + it('is an additive method that lets the current host record mobile capabilities', async () => { + const updates: string[][] = [] + + const replies = await callBuild( + current, + CLIENT_CAPABILITY_UPDATE_METHOD, + { clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] }, + { + clientKind: 'mobile', + clientCapabilities: [], + updateClientCapabilities: (capabilities) => updates.push([...capabilities]) + } + ) + + expect(current.methodNames).toContain(CLIENT_CAPABILITY_UPDATE_METHOD) + expect(current.protocolVersion).toBe(baseline.protocolVersion) + expect(replies).toHaveLength(1) + expect(replies[0]).toMatchObject({ + ok: true, + result: { clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] } + }) + expect(updates).toEqual([[STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]]) + }) + + it('gets a normal answer from an old host instead of changing the auth shape', async () => { + const replies = await callBuild( + baseline, + CLIENT_CAPABILITY_UPDATE_METHOD, + { clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] }, + { clientKind: 'mobile', clientCapabilities: [] } + ) + + expect(replies).toHaveLength(1) + if (!baseline.methodNames.includes(CLIENT_CAPABILITY_UPDATE_METHOD)) { + expect(replies[0]).toMatchObject({ + ok: false, + error: { code: 'method_not_found' } + }) + } + }) + }) + describe('an old client against a structured-owned AI Vault row', () => { let root: string let store: AgentSessionRecordStore diff --git a/tests/e2e/cross-version-wire/versioned-agent-session-wire.ts b/tests/e2e/cross-version-wire/versioned-agent-session-wire.ts index 1b0dc297f81..4d9e2de68ec 100644 --- a/tests/e2e/cross-version-wire/versioned-agent-session-wire.ts +++ b/tests/e2e/cross-version-wire/versioned-agent-session-wire.ts @@ -29,6 +29,7 @@ export type RpcReply = { export type RpcClientIdentity = { clientKind?: 'mobile' | 'runtime' clientCapabilities?: readonly string[] + updateClientCapabilities?: (capabilities: readonly string[]) => void connectionId?: string clientId?: string } From 7d27c841b4d032e2f9220f7b2c8924e152419a9a Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 18:38:33 -0400 Subject: [PATCH 203/398] fix(cloud): run the rehome control job under pipefail (#18537) The five `node ... | tee` steps in cloud-operate-relay-production-rehome-job.yml reported tee's exit code, so a thrown inspect or apply passed green. The Aug 28 21:25Z and Aug 29 inspects and today's first inspect all printed "director returned an invalid regional rehome control" (the durable control had moved to generation 12 when the Aug 28 rehome aborted) and still succeeded. `shell: bash` adds `-o pipefail`. A test pins the default and the tee count. --- .../cloud-operate-relay-production-rehome-job.yml | 3 +++ cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs | 7 +++++++ 2 files changed, 10 insertions(+) diff --git a/.github/workflows/cloud-operate-relay-production-rehome-job.yml b/.github/workflows/cloud-operate-relay-production-rehome-job.yml index 1ceadcece12..682953af5e7 100644 --- a/.github/workflows/cloud-operate-relay-production-rehome-job.yml +++ b/.github/workflows/cloud-operate-relay-production-rehome-job.yml @@ -26,6 +26,9 @@ permissions: defaults: run: + # `shell: bash` adds pipefail; without it `node ... | tee` reports tee's exit code and a + # thrown inspect/apply passed green (Aug 28-29 and Sep 3 2026 runs). + shell: bash working-directory: cloud jobs: diff --git a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs index 5eab757255a..e31403f6dd6 100644 --- a/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs +++ b/cloud/dev/scripts/relay-regional-rehome-workflow.test.mjs @@ -191,3 +191,10 @@ test('director rollout has a strict one-time identity bootstrap', () => { assert.ok(candidateProof > 0 && candidateProof < trafficMove) assert.equal(script.indexOf('verifyRehomeDisabled', trafficMove), -1) }) + +test('rehome job pipes every control result through tee under pipefail', () => { + const job = workflow('operate-relay-production-rehome-job.yml') + // Without `shell: bash` the step exit code is tee's, so a thrown inspect/apply passes green. + assert.match(job, /defaults:\n run:\n(?: #.*\n)* shell: bash\n/) + assert.ok((job.match(/\| tee "\$\{RUNNER_TEMP\}/g) ?? []).length >= 5) +}) From a5d6114baf3daa09dca84a245af7176cf083a163 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:13:03 -0700 Subject: [PATCH 204/398] fix(ssh): stop pane adoption certifying a death from the relay's not-found union (#18531) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(ssh): stop pane adoption certifying a death from the relay's not-found union `attachStablePaneOwner` was the last reader that synthesised a runtime exit from a reattach refusal, and it published code `0` — which `orca-runtime-on-pty-exit` records as `rememberPtyLivenessVerdict(exited)`, a death certificate whose only legitimate writer is a host-delivered exit frame. The refusal it acted on is a union. `pty.attach` answers `PTY "" not found` both for a pid the relay probed with `isProcessAlive` and for an id its session map simply never had — which, because ids carry a per-start mint epoch, is every id minted before a relay restart, checked against nothing. So a relay restart plus a reconnect certified a shell that was still running under the old daemon's orphaned process tree, retired the pane binding, and cold-started a second agent onto the same transcript. The sibling `handlePtyReattachFailure` has always refused to certify from that union; this path did not. - The relay marks the one refusal it backed with a liveness check (`PTY_ATTACH_PROVEN_EXITED_MARKER`). The marker is additive, so an unmarked answer — including an older relay's — stays ambiguous, which is the safe direction. - The client mints that half as `SshPtyProvenExitedOnRelayError`, a subclass so every existing `isSshPtyAbsentFromRelayError` consumer is unchanged. - Pane adoption publishes `UNVERIFIED_PROCESS_EXIT_CODE` (-1), the sentinel its sibling publishes, and passes `hostExitConfirmed` only for evidence that observed the process: the marked relay refusal, or `SessionNotFoundError` from the registry that owns the PTY. The ambiguous half now records `unverifiable` instead of `exited`. - The gone-branch keys on the error type rather than the bare `PTY ".+" not found` text, so an untyped string can no longer authorise abandoning a binding — the discriminator `pty-connect-limits.ts` already documented. Refs docs/reference/ssh-execution-boundary.md * test(pty): make the pane-adoption fixtures throw what real providers throw These four fixtures rejected with bare `new Error('Session not found: ...')` and `new Error('PTY "..." not found')`. No provider produces either untyped: `local-pty-spawn` and `decodeDaemonResponseError` both mint `SessionNotFoundError`, and the SSH reattach path types the relay's wire text before any pane sees it. Fixtures that skip the type were the reason a message-shaped gate looked adequate. The exit-code expectations move with it: the pane path now publishes the -1 stop sentinel plus `hostExitConfirmed`, so a certificate follows the evidence rather than a synthesized zero. --- docs/reference/ssh-execution-boundary.md | 2 + src/main/ipc/pty-dead-owner-respawn.test.ts | 23 ++- .../pty-pane-reservation-settlement.test.ts | 8 +- .../pty-persisted-incarnation-repair.test.ts | 14 +- src/main/ipc/pty/pane/stable-owner.ts | 17 +- ...ble-pane-absence-death-certificate.test.ts | 185 ++++++++++++++++++ src/main/ipc/pty/provider/liveness.ts | 31 ++- src/main/providers/ssh-pty-errors.ts | 37 +++- ...ty-reattach-absence-discrimination.test.ts | 17 ++ .../providers/ssh-pty-session-reattach.ts | 9 + .../ssh-expired-lease-pane-readoption.test.ts | 5 +- src/relay/pty-handler-attach-replay.test.ts | 21 +- src/relay/pty-handler.ts | 7 +- src/shared/pty-attach-absence-evidence.ts | 17 ++ 14 files changed, 362 insertions(+), 31 deletions(-) create mode 100644 src/main/ipc/pty/pane/stable-pane-absence-death-certificate.test.ts create mode 100644 src/shared/pty-attach-absence-evidence.ts diff --git a/docs/reference/ssh-execution-boundary.md b/docs/reference/ssh-execution-boundary.md index 070aae61d66..88a4a3c0a0e 100644 --- a/docs/reference/ssh-execution-boundary.md +++ b/docs/reference/ssh-execution-boundary.md @@ -66,6 +66,8 @@ A verdict needs evidence from the host that owns the process. Apply these tests **Does the termination event match the current identity?** A host-delivered exit for the live PTY incarnation and provider generation, while its siblings still report, establishes `exited`. A stale event, an event for a superseded incarnation, or one quiet terminal with no host evidence does not. +**Did the answer carry its evidence, or only the same wording?** `pty.attach` refuses with `PTY "" not found` both for a pid the relay probed and found gone and for an id its session map never had — which is every id minted before a relay restart, since ids carry a per-start mint epoch. Only the probed refusal carries `PTY_ATTACH_PROVEN_EXITED_MARKER` (`src/shared/pty-attach-absence-evidence.ts`) and reaches the client as `SshPtyProvenExitedOnRelayError`; the unmarked union arrives as `SshPtyAbsentFromRelayError`, which licenses retiring the client's own route to the PTY and nothing more. A missing marker is never evidence — an older relay omits it too. + **Is a returned status actually a claim of success?** An operation that reports failure may have succeeded, and one that reports success may not have run — check the durable state it should have changed rather than trusting the return. Anything short of positive host evidence is `unverifiable`. Reporting it as `exited` is the error this document exists to prevent: it orphans live work and can cold-start a duplicate over the same worktree. diff --git a/src/main/ipc/pty-dead-owner-respawn.test.ts b/src/main/ipc/pty-dead-owner-respawn.test.ts index 8f47fa6fcdf..4669d7e5528 100644 --- a/src/main/ipc/pty-dead-owner-respawn.test.ts +++ b/src/main/ipc/pty-dead-owner-respawn.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from 'vitest' import { setupPtyIpcSuite } from './pty-ipc-test-harness' +import { SessionNotFoundError } from '../daemon/daemon-errors' import { makePaneKey } from '../../shared/stable-pane-id' import { registerPtyHandlers, setLocalPtyProvider } from './pty' @@ -59,7 +60,7 @@ describe('registerPtyHandlers', () => { const providerSpawn = vi.fn( async (options: { attachOnly?: boolean; command?: string; sessionId?: string }) => { if (options.attachOnly) { - throw new Error('Session not found: pty-proven-absent-owner') + throw new SessionNotFoundError('pty-proven-absent-owner') } return { id: 'pty-fresh-proven', incarnationId: 'inc-fresh-proven' } } @@ -161,10 +162,13 @@ describe('registerPtyHandlers', () => { expect(providerSpawn.mock.calls[1]?.[0]).toMatchObject({ command: 'codex resume proven-absent-session' }) + // The registry that owns the PTY answered, so this exit is confirmed — but the code stays the + // -1 sentinel; a synthesized zero would be indistinguishable from a clean shell exit. expect(runtime.onPtyExit).toHaveBeenCalledWith( 'pty-proven-absent-owner', - 0, - 'inc-proven-absent-owner' + -1, + 'inc-proven-absent-owner', + { hostExitConfirmed: true } ) expect(store.setWorkspaceSession).toHaveBeenCalledOnce() expect(store.flushOrThrow).toHaveBeenCalledOnce() @@ -178,7 +182,7 @@ describe('registerPtyHandlers', () => { const providerSpawn = vi.fn( async (options: { attachOnly?: boolean; command?: string; sessionId?: string }) => { if (options.attachOnly) { - throw new Error('Session not found: pty-probe-blip-owner') + throw new SessionNotFoundError('pty-probe-blip-owner') } return { id: 'pty-fresh-probe-blip', incarnationId: 'inc-fresh-probe-blip' } } @@ -282,8 +286,9 @@ describe('registerPtyHandlers', () => { expect(providerSpawn).toHaveBeenCalledTimes(2) expect(runtime.onPtyExit).toHaveBeenCalledWith( 'pty-probe-blip-owner', - 0, - 'inc-probe-blip-owner' + -1, + 'inc-probe-blip-owner', + { hostExitConfirmed: true } ) }) // Why: a parked pane (stopped with keepHistory) leaves the runtime holding the binding while @@ -299,7 +304,7 @@ describe('registerPtyHandlers', () => { const providerSpawn = vi.fn( async (options: { attachOnly?: boolean; command?: string; sessionId?: string }) => { if (options.attachOnly) { - throw new Error('Session not found: pty-already-retired-owner') + throw new SessionNotFoundError('pty-already-retired-owner') } return { id: 'pty-fresh-already-retired', incarnationId: 'inc-fresh-already-retired' } } @@ -420,6 +425,8 @@ describe('registerPtyHandlers', () => { expect(providerSpawn.mock.calls[1]?.[0]).toMatchObject({ command: 'codex resume already-retired-session' }) - expect(runtime.onPtyExit).toHaveBeenCalledWith('pty-already-retired-owner', 0, undefined) + expect(runtime.onPtyExit).toHaveBeenCalledWith('pty-already-retired-owner', -1, undefined, { + hostExitConfirmed: true + }) }) }) diff --git a/src/main/ipc/pty-pane-reservation-settlement.test.ts b/src/main/ipc/pty-pane-reservation-settlement.test.ts index c894eeaeddd..64a4441f376 100644 --- a/src/main/ipc/pty-pane-reservation-settlement.test.ts +++ b/src/main/ipc/pty-pane-reservation-settlement.test.ts @@ -2,6 +2,10 @@ import { describe, expect, it, vi } from 'vitest' import { spawnMock, registerPtyMock } from './pty-ipc-mock-registry' import { setupPtyIpcSuite } from './pty-ipc-test-harness' import { makePaneKey } from '../../shared/stable-pane-id' +import { + SSH_SESSION_EXPIRED_ERROR, + SshPtyProvenExitedOnRelayError +} from '../providers/ssh-pty-errors' import { registerPtyHandlers, registerSshPtyProvider, @@ -67,7 +71,9 @@ describe('registerPtyHandlers', () => { const freshPtyId = `ssh:${connectionId}@@fresh-relay-pty` const remoteSpawn = vi.fn(async (options: { attachOnly?: boolean; command?: string }) => { if (options.attachOnly) { - throw new Error('PTY "dead-relay-pty" not found') + // The relay's raw wire text never reaches a pane untyped; the SSH reattach path mints the + // proven-exit class for the one refusal the relay backed with a pid probe. + throw new SshPtyProvenExitedOnRelayError(`${SSH_SESSION_EXPIRED_ERROR}: dead-relay-pty`) } return { id: freshPtyId, incarnationId: 'inc-fresh-ssh-owner' } }) diff --git a/src/main/ipc/pty-persisted-incarnation-repair.test.ts b/src/main/ipc/pty-persisted-incarnation-repair.test.ts index 8f565f1c08e..743963d0189 100644 --- a/src/main/ipc/pty-persisted-incarnation-repair.test.ts +++ b/src/main/ipc/pty-persisted-incarnation-repair.test.ts @@ -1,7 +1,7 @@ import { describe, expect, it, vi } from 'vitest' import { statSyncMock } from './pty-ipc-mock-registry' import { setupPtyIpcSuite } from './pty-ipc-test-harness' -import { TerminalSessionOwnerUnverifiedError } from '../daemon/daemon-errors' +import { SessionNotFoundError, TerminalSessionOwnerUnverifiedError } from '../daemon/daemon-errors' import { makePaneKey } from '../../shared/stable-pane-id' import { registerPtyHandlers, clearProviderPtyState, setLocalPtyProvider } from './pty' @@ -230,7 +230,7 @@ describe('registerPtyHandlers', () => { const providerSpawn = vi.fn( async (options: { attachOnly?: boolean; command?: string; sessionId?: string }) => { if (options.attachOnly) { - throw new Error('Session not found: pty-dead-persisted-owner') + throw new SessionNotFoundError('pty-dead-persisted-owner') } return { id: 'pty-fresh-recovery', incarnationId: 'inc-fresh-recovery' } } @@ -352,8 +352,9 @@ describe('registerPtyHandlers', () => { expect(store.setWorkspaceSession).toHaveBeenCalledOnce() expect(runtime.onPtyExit).toHaveBeenCalledWith( 'pty-dead-persisted-owner', - 0, - 'inc-dead-persisted-owner' + -1, + 'inc-dead-persisted-owner', + { hostExitConfirmed: true } ) return } @@ -376,8 +377,9 @@ describe('registerPtyHandlers', () => { expect(store.flushOrThrow).toHaveBeenCalledOnce() expect(runtime.onPtyExit).toHaveBeenCalledWith( 'pty-dead-persisted-owner', - 0, - 'inc-dead-persisted-owner' + -1, + 'inc-dead-persisted-owner', + { hostExitConfirmed: true } ) } ) diff --git a/src/main/ipc/pty/pane/stable-owner.ts b/src/main/ipc/pty/pane/stable-owner.ts index 9442e914e1f..731065e59ab 100644 --- a/src/main/ipc/pty/pane/stable-owner.ts +++ b/src/main/ipc/pty/pane/stable-owner.ts @@ -1,5 +1,6 @@ import { toSshExecutionHostId } from '../../../../shared/execution-host' import { makePaneKey, parsePaneKey } from '../../../../shared/stable-pane-id' +import { UNVERIFIED_PROCESS_EXIT_CODE } from '../../../../shared/terminal-exit-cause' import type { Store } from '../../../persistence' import { retireTerminalSurfaceFromPersistence } from '../../../runtime/mobile-session-terminal-persistence-retirement' import type { OrcaRuntimeService } from '../../../runtime/orca-runtime' @@ -11,7 +12,7 @@ import { TerminalSessionOwnerUnverifiedError } from '../../../daemon/daemon-errors' import { ptyIncarnationById, ptyOwnership } from '../provider/ownership-state' -import { isPtyAlreadyGoneError } from '../provider/liveness' +import { isHostReportedPtyAbsenceError, isObservedPtyExitEvidence } from '../provider/liveness' import { clearProviderPtyState } from '../provider/state-cleanup' export type StablePaneOwner = { @@ -239,7 +240,7 @@ export async function attachStablePaneOwner( if (isDaemonEndpointGoneError(error)) { throw new TerminalHostGoneError() } - if (!isPtyAlreadyGoneError(error)) { + if (!isHostReportedPtyAbsenceError(error)) { throw error } const ownerBeforeRetire = args.resolveOwner?.() @@ -252,7 +253,17 @@ export async function attachStablePaneOwner( ) { throw new Error('terminal_pane_owner_changed') } - runtime?.onPtyExit(owner.ptyId, 0, owner.incarnationId) + // `pty.attach` answers absent both for a pid the relay probed and found gone and for an id its + // session map never had — every id minted before a relay restart, checked against nothing. Only + // the marked half observed the process, so only it may certify a death; the rest publishes the + // stop sentinel its sibling handlePtyReattachFailure publishes, which every reader resolves to + // `stop_unverified` (docs/reference/ssh-execution-boundary.md). + runtime?.onPtyExit( + owner.ptyId, + UNVERIFIED_PROCESS_EXIT_CODE, + owner.incarnationId, + isObservedPtyExitEvidence(error) ? { hostExitConfirmed: true } : {} + ) clearProviderPtyState(owner.ptyId) ptyOwnership.delete(owner.ptyId) if ( diff --git a/src/main/ipc/pty/pane/stable-pane-absence-death-certificate.test.ts b/src/main/ipc/pty/pane/stable-pane-absence-death-certificate.test.ts new file mode 100644 index 00000000000..972f479b492 --- /dev/null +++ b/src/main/ipc/pty/pane/stable-pane-absence-death-certificate.test.ts @@ -0,0 +1,185 @@ +// `attachStablePaneOwner` is the last reader that synthesised a runtime exit from a reattach +// refusal, and it published code 0 — which `orca-runtime-on-pty-exit` records as a death +// certificate. The refusal it acts on is a union: `pty.attach` answers absent both for a pid the +// relay probed and found gone, and for an id its session map never had, which is every id minted +// before a relay restart. Certifying the union orphans a live remote shell and cold-starts a second +// agent onto its transcript (docs/reference/ssh-execution-boundary.md). +// +// The sibling handlePtyReattachFailure has always refused to certify from that union. These pin the +// same rule here, and pin that the marked half — the one refusal the relay backed with a pid probe +// — still earns the certificate, so a genuinely dead PTY is not left `unverifiable` forever. +import { describe, expect, it, vi } from 'vitest' +import { getDefaultWorkspaceSession } from '../../../../shared/constants' +import { makePaneKey } from '../../../../shared/stable-pane-id' +import { SSH_EXIT_UNCONFIRMED_REASON } from '../../../../shared/pty-liveness-verdict' +import type { WorkspaceSessionState } from '../../../../shared/workspace-session-state-types' +import { SessionNotFoundError } from '../../../daemon/daemon-errors' +import type { Store } from '../../../persistence' +import { + SSH_SESSION_EXPIRED_ERROR, + SshPtyAbsentFromRelayError, + SshPtyProvenExitedOnRelayError +} from '../../../providers/ssh-pty-errors' +import type { IPtyProvider } from '../../../providers/types' +import { OrcaRuntimeService } from '../../../runtime/orca-runtime' +import { resolvePersistedStablePaneOwner, spawnForStablePane } from './stable-owner' + +const CONNECTION = 'conn-1' +const WORKTREE = 'repo-1::/tmp/pane-absence' +const TAB = 'tab-1' +const LEAF = '1b3f2c4d-5e6a-4b7c-8d9e-0f1a2b3c4d5e' +const SIBLING_LEAF = '2c4d3e5f-6a7b-4c8d-9e0f-1a2b3c4d5e6f' +// Ids carry the relay's per-start mint epoch, so this one names a PTY the CURRENT relay never minted. +const PTY_ID = 'ssh:conn-1@@pty2:epoch-a:1' +const OWNER = { tabId: TAB, leafId: LEAF, ptyId: PTY_ID, hasPersistedBinding: true as const } + +function paneStore(): { store: Store; read: () => WorkspaceSessionState } { + let session = { + ...getDefaultWorkspaceSession(), + tabsByWorktree: { + [WORKTREE]: [{ id: TAB, type: 'terminal', worktreeId: WORKTREE, ptyId: PTY_ID }] + }, + terminalLayoutsByTabId: { + [TAB]: { + root: { + type: 'split', + direction: 'row', + first: { type: 'leaf', leafId: LEAF }, + second: { type: 'leaf', leafId: SIBLING_LEAF } + }, + activeLeafId: LEAF, + ptyIdsByLeafId: { [LEAF]: PTY_ID, [SIBLING_LEAF]: 'ssh:conn-1@@pty2:epoch-a:2' } + } + } + } as unknown as WorkspaceSessionState + return { + read: () => session, + store: { + getWorkspaceSession: () => session, + setWorkspaceSession: (next: WorkspaceSessionState) => { + session = next + }, + flushOrThrow: () => {}, + getRepos: () => [ + { + id: 'repo-1', + path: '/tmp/pane-absence', + displayName: 'pane-absence', + badgeColor: '#000000', + addedAt: 0 + } + ], + getAllWorktreeMeta: () => ({}), + getWorktreeMeta: () => undefined, + setWorktreeMeta: () => {}, + removeWorktreeMeta: () => {}, + getSettings: () => ({ workspaceDir: '/tmp/workspaces' }), + getProjects: () => [] + } as unknown as Store + } +} + +function runtimeOwning(store: Store): OrcaRuntimeService { + const runtime = new OrcaRuntimeService(store as never) + runtime.setPtyController({ + write: () => true, + kill: () => true, + hasPty: () => null, + listProcesses: async () => [], + getForegroundProcess: async () => null + } as never) + runtime.attachWindow(1) + runtime.syncWindowGraph(1, { tabs: [], leaves: [] }) + runtime.registerPty(PTY_ID, WORKTREE, CONNECTION) + return runtime +} + +async function adoptAfterAttachRefusal(error: unknown): Promise<{ + runtime: OrcaRuntimeService + store: Store + read: () => WorkspaceSessionState + spawn: ReturnType +}> { + const { store, read } = paneStore() + const runtime = runtimeOwning(store) + const spawn = vi + .fn() + .mockRejectedValueOnce(error) + .mockResolvedValueOnce({ id: 'ssh:conn-1@@pty2:epoch-b:1', isReattach: false }) + await spawnForStablePane({ + runtime, + store, + provider: { spawn } as unknown as IPtyProvider, + spawnOptions: { cols: 80, rows: 24 }, + owner: OWNER, + worktreeId: WORKTREE, + connectionId: CONNECTION, + resolveOwner: () => null + }) + return { runtime, store, read, spawn } +} + +describe('a stable pane whose reattach was refused', () => { + it('records no death certificate when the relay merely does not know the id', async () => { + const { runtime, spawn } = await adoptAfterAttachRefusal( + new SshPtyAbsentFromRelayError(`${SSH_SESSION_EXPIRED_ERROR}: pty2:epoch-a:1`) + ) + + expect(spawn).toHaveBeenCalledTimes(2) + // The shell may well still be running under the previous daemon's orphaned process tree, so the + // register must keep saying "we could not observe it" — not "it ended". + expect(runtime.getPtyLivenessVerdict(PTY_ID)).toEqual({ + status: 'unverifiable', + reason: SSH_EXIT_UNCONFIRMED_REASON + }) + }) + + it('still certifies the death the relay proved with a pid probe', async () => { + const { runtime, store, spawn } = await adoptAfterAttachRefusal( + new SshPtyProvenExitedOnRelayError(`${SSH_SESSION_EXPIRED_ERROR}: pty2:epoch-a:1`) + ) + + expect(spawn).toHaveBeenCalledTimes(2) + expect(runtime.getPtyLivenessVerdict(PTY_ID)).toEqual({ status: 'exited' }) + // If nothing ever retired, a proven-dead pane would reattach to a corpse on every adoption. + expect( + resolvePersistedStablePaneOwner(store, makePaneKey(TAB, LEAF), WORKTREE, CONNECTION) + ).toBeNull() + }) + + it('certifies an absence reported by the process registry that owns the PTY', async () => { + // The daemon (or the in-process map) answering here is the owner of the process, and an + // endpoint that had gone raises TerminalHostGoneError above, so this absence is an observation + // rather than a lost route. + const { runtime } = await adoptAfterAttachRefusal(new SessionNotFoundError(PTY_ID)) + + expect(runtime.getPtyLivenessVerdict(PTY_ID)).toEqual({ status: 'exited' }) + }) + + it('refuses to abandon the binding on an untyped "not found" string', async () => { + // The relay's raw wire wording. The SSH reattach path types it before any pane sees it, so an + // untyped one reached this gate having lost every distinction the type carries — including + // whether the answer came from the host that owns the process at all. + const { store, read } = paneStore() + const before = JSON.stringify(read()) + const runtime = runtimeOwning(store) + const spawn = vi.fn().mockRejectedValue(new Error(`PTY "pty2:epoch-a:1" not found`)) + + await expect( + spawnForStablePane({ + runtime, + store, + provider: { spawn } as unknown as IPtyProvider, + spawnOptions: { cols: 80, rows: 24 }, + owner: OWNER, + worktreeId: WORKTREE, + connectionId: CONNECTION, + resolveOwner: () => null + }) + ).rejects.toThrow('not found') + + expect(spawn).toHaveBeenCalledTimes(1) + expect(runtime.getPtyLivenessVerdict(PTY_ID)).toBeNull() + expect(JSON.stringify(read())).toBe(before) + }) +}) diff --git a/src/main/ipc/pty/provider/liveness.ts b/src/main/ipc/pty/provider/liveness.ts index cc764143384..8a362b74aaf 100644 --- a/src/main/ipc/pty/provider/liveness.ts +++ b/src/main/ipc/pty/provider/liveness.ts @@ -2,10 +2,12 @@ import { isRemoteAgentHooksEnabled } from '../../../../shared/agent-hook-relay' import type { AgentSessionOwnerBinding } from '../../../../shared/agent-session-host-authority' import { agentSessionOwnerBindingsEqual } from '../../../../shared/claimed-agent-pty-owner' import { addNodePtyRecoveryHint } from '../../../daemon/node-pty-error-hints' +import { SessionNotFoundError } from '../../../daemon/daemon-errors' import type { Store } from '../../../persistence' import { isSshPtyAbsentFromRelayError, - isSshPtyNotFoundError + isSshPtyNotFoundError, + isSshPtyProvenExitedOnRelayError } from '../../../providers/ssh-pty-errors' import type { IPtyProvider } from '../../../providers/types' import { markClaudePtyExited } from '../../../claude-accounts/live-pty-gate' @@ -66,6 +68,33 @@ export function isPtyAlreadyGoneError(err: unknown): boolean { ) } +/** + * Narrower than {@link isPtyAlreadyGoneError}, for the one caller that retires a durable pane + * binding rather than just releasing in-memory state: only a typed answer from the host that owns + * the process may authorise that. The bare `PTY ".+" not found` text is the relay's raw wire + * wording, which the SSH reattach path always types before it reaches a pane; matching the text + * instead would let any untyped string carrying that phrase unbind a live pane + * (docs/reference/ssh-execution-boundary.md). + */ +export function isHostReportedPtyAbsenceError(err: unknown): boolean { + return isSshPtyAbsentFromRelayError(err) || err instanceof SessionNotFoundError +} + +/** + * The half of {@link isHostReportedPtyAbsenceError} that actually observed the process, and so the + * only half that may certify an exit. + * + * The relay's plain absence answer is excluded because `pty.attach` gives it for an id its session + * map never had as readily as for a pid it probed — after a relay restart, every id the previous + * one minted. `SessionNotFoundError` is included because the process answering is the one that owns + * the PTY: the in-process registry itself, or a daemon whose endpoint is live (a gone endpoint + * raises `isDaemonEndpointGoneError` instead), so its absence is an observation rather than a lost + * route (docs/reference/ssh-execution-boundary.md). + */ +export function isObservedPtyExitEvidence(err: unknown): boolean { + return isSshPtyProvenExitedOnRelayError(err) || err instanceof SessionNotFoundError +} + export function delay(ms: number): Promise { return new Promise((resolve) => { const timer = setTimeout(resolve, ms) diff --git a/src/main/providers/ssh-pty-errors.ts b/src/main/providers/ssh-pty-errors.ts index 92ee84941ad..4d312caa8b1 100644 --- a/src/main/providers/ssh-pty-errors.ts +++ b/src/main/providers/ssh-pty-errors.ts @@ -20,12 +20,16 @@ export function isSshPtyIdentityMismatchError(error: unknown): boolean { } /** - * A reachable relay answered for this exact PTY id and reported it absent — positive evidence of - * absence from the execution host, so `exited` rather than `unverifiable` - * (docs/reference/ssh-execution-boundary.md). Deliberately NOT raised for a transport failure, a - * request timeout, a disposed multiplexer, an identity mismatch (the id names a live PTY belonging - * to another pane), or `restoreRequired` (the PTY is live, only its source stream is not) — none of - * those observe the process, and treating them as absence orphans live remote work. + * A reachable relay answered for this exact PTY id and reported it absent, so the client may retire + * its own route to it. Deliberately NOT raised for a transport failure, a request timeout, a + * disposed multiplexer, an identity mismatch (the id names a live PTY belonging to another pane), + * or `restoreRequired` (the PTY is live, only its source stream is not) — none of those observe the + * process, and treating them as absence orphans live remote work. + * + * This is NOT itself a death certificate. `pty.attach` answers absent for an id its session map + * never had as readily as for a pid it probed — and after a relay restart that is every id the + * previous one minted. Certifying `exited` needs {@link SshPtyProvenExitedOnRelayError} + * (docs/reference/ssh-execution-boundary.md). * * Carries the same `SSH_SESSION_EXPIRED` message so message-based consumers are unaffected; only * callers that can act on the stronger verdict test the class. @@ -40,3 +44,24 @@ export class SshPtyAbsentFromRelayError extends Error { export function isSshPtyAbsentFromRelayError(error: unknown): boolean { return error instanceof SshPtyAbsentFromRelayError } + +/** + * The narrow half of {@link SshPtyAbsentFromRelayError}: the relay probed the pid and found it gone + * before answering absent, so this is the one attach refusal that observed the process and the only + * one that may certify a death. + * + * The parent class is raised for the whole union, which also contains "this session map has no such + * id" — every id minted before a relay restart, checked against nothing. Callers that only release + * client-side bookkeeping keep testing the parent; a caller about to record `exited` must test this + * (docs/reference/ssh-execution-boundary.md). + */ +export class SshPtyProvenExitedOnRelayError extends SshPtyAbsentFromRelayError { + constructor(message: string) { + super(message) + this.name = 'SshPtyProvenExitedOnRelayError' + } +} + +export function isSshPtyProvenExitedOnRelayError(error: unknown): boolean { + return error instanceof SshPtyProvenExitedOnRelayError +} diff --git a/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts b/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts index 79ed97a8f19..bfcc174a4b6 100644 --- a/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts +++ b/src/main/providers/ssh-pty-reattach-absence-discrimination.test.ts @@ -14,9 +14,11 @@ import { describe, expect, it, vi } from 'vitest' import { isSshPtyAbsentFromRelayError, + isSshPtyProvenExitedOnRelayError, SSH_PTY_SOURCE_RESTORE_REQUIRED_ERROR, SSH_SESSION_EXPIRED_ERROR } from './ssh-pty-errors' +import { PTY_ATTACH_PROVEN_EXITED_MARKER } from '../../shared/pty-attach-absence-evidence' import { reattachSshPtySessionForSpawn } from './ssh-pty-session-reattach' import type { SshChannelMultiplexer } from '../ssh/ssh-channel-multiplexer' @@ -55,6 +57,20 @@ describe('an SSH reattach refusal says whether the host observed the PTY', () => expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) expect(isSshPtyAbsentFromRelayError(error)).toBe(true) + // ...but absence from the relay is a union. An unmarked answer is also what a restarted relay + // gives for every id the previous one minted, checked against nothing, so it may not certify a + // death (docs/reference/ssh-execution-boundary.md). + expect(isSshPtyProvenExitedOnRelayError(error)).toBe(false) + }) + + it('separates the refusal the relay backed with a pid probe', async () => { + const error = await refusalFrom(async () => { + throw new Error(`PTY "${SESSION}" not found (${PTY_ATTACH_PROVEN_EXITED_MARKER})`) + }) + + expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) + expect(isSshPtyAbsentFromRelayError(error)).toBe(true) + expect(isSshPtyProvenExitedOnRelayError(error)).toBe(true) }) it('gives a restoreRequired refusal its own token instead of the expiry text', async () => { @@ -79,5 +95,6 @@ describe('an SSH reattach refusal says whether the host observed the PTY', () => expect(error.message).toContain(SSH_SESSION_EXPIRED_ERROR) expect(isSshPtyAbsentFromRelayError(error)).toBe(false) + expect(isSshPtyProvenExitedOnRelayError(error)).toBe(false) }) }) diff --git a/src/main/providers/ssh-pty-session-reattach.ts b/src/main/providers/ssh-pty-session-reattach.ts index 530135ad310..f05373a03dc 100644 --- a/src/main/providers/ssh-pty-session-reattach.ts +++ b/src/main/providers/ssh-pty-session-reattach.ts @@ -5,9 +5,11 @@ import { SSH_PTY_SOURCE_RESTORE_REQUIRED_ERROR, SSH_SESSION_EXPIRED_ERROR, SshPtyAbsentFromRelayError, + SshPtyProvenExitedOnRelayError, isSshPtyIdentityMismatchError, isSshPtyNotFoundError } from './ssh-pty-errors' +import { isProvenExitedPtyAttachRefusal } from '../../shared/pty-attach-absence-evidence' import { toAppSshPtyId, toRelaySshPtyId } from './ssh-pty-id' import type { PtySpawnOptions, PtySpawnResult } from './types' import type { SshPtySpawnExitRaceTracker } from './ssh-pty-spawn-exit-race' @@ -238,6 +240,13 @@ export async function reattachSshPtySession(args: { // Why the class: the relay answered for this exact id, so callers holding a pane binding may // retire it and spawn fresh. Plain `SSH_SESSION_EXPIRED` cannot say that — a restarted relay // renumbers from pty-1, so the message alone is indistinguishable from a lost link. + // + // Why the subclass: the relay marks the one refusal it backed with a pid probe. Without the + // marker the answer is the "no such id" union, which is not evidence the shell ended, so the + // narrow class is minted only when the relay said so (docs/reference/ssh-execution-boundary.md). + if (isProvenExitedPtyAttachRefusal(error)) { + throw new SshPtyProvenExitedOnRelayError(`${SSH_SESSION_EXPIRED_ERROR}: ${relaySessionId}`) + } throw new SshPtyAbsentFromRelayError(`${SSH_SESSION_EXPIRED_ERROR}: ${relaySessionId}`) } throw error diff --git a/src/main/ssh-expired-lease-pane-readoption.test.ts b/src/main/ssh-expired-lease-pane-readoption.test.ts index 4c2c34b5f62..11bf5fe2083 100644 --- a/src/main/ssh-expired-lease-pane-readoption.test.ts +++ b/src/main/ssh-expired-lease-pane-readoption.test.ts @@ -7,6 +7,7 @@ import { resolvePersistedStablePaneOwner } from './ipc/pty/pane/stable-owner' import { adoptStablePane } from './ipc/pty/pane/adopt-stable' import { sshProviders } from './ipc/pty/provider/registry' import type { IPtyProvider } from './providers/types' +import { SSH_SESSION_EXPIRED_ERROR, SshPtyAbsentFromRelayError } from './providers/ssh-pty-errors' import { testState, createStore, makeTerminalTab } from './persistence-test-harness' import { TEST_LEAF_1 } from './persistence-session-fixtures' @@ -141,11 +142,13 @@ describe('recovery through createTerminal reattaches before it respawns', () => expect(adopted?.owner).toMatchObject({ ptyId: APP_PTY_ID, hasPersistedBinding: true }) }) + // The typed refusal is what the SSH reattach path raises; the raw `PTY "…" not found` wire text + // never reaches a pane untyped, and an untyped one no longer authorises abandoning the binding. it('falls through to a fresh spawn once the host answers that the PTY is absent', async () => { const store = storeWithBoundRemotePane() store.markSshRemotePtyLease(TARGET, APP_PTY_ID, 'expired') const spawn = vi.fn(async () => { - throw new Error('PTY "remote-pty" not found') + throw new SshPtyAbsentFromRelayError(`${SSH_SESSION_EXPIRED_ERROR}: remote-pty`) }) sshProviders.set(TARGET, { spawn } as unknown as IPtyProvider) diff --git a/src/relay/pty-handler-attach-replay.test.ts b/src/relay/pty-handler-attach-replay.test.ts index 6af9faec047..e289547bf0c 100644 --- a/src/relay/pty-handler-attach-replay.test.ts +++ b/src/relay/pty-handler-attach-replay.test.ts @@ -1,5 +1,9 @@ import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest' import * as ptyShellUtils from './pty-shell-utils' +import { + PTY_ATTACH_PROVEN_EXITED_MARKER, + isProvenExitedPtyAttachRefusal +} from '../shared/pty-attach-absence-evidence' const { mockPtySpawn, mockPtyInstance, mockCreateShellPromptReadinessProbe } = vi.hoisted(() => ({ mockPtySpawn: vi.fn(), @@ -87,7 +91,7 @@ describe('PtyHandler', () => { try { await expect( dispatcher.callRequest('pty.attach', { id: PTY_1, suppressReplayNotification: true }) - ).rejects.toThrow(`PTY "${PTY_1}" not found`) + ).rejects.toThrow(`PTY "${PTY_1}" not found (${PTY_ATTACH_PROVEN_EXITED_MARKER})`) } finally { aliveSpy.mockRestore() } @@ -96,9 +100,18 @@ describe('PtyHandler', () => { // is freed so a later attach also cleanly reports not-found. expect(exits).toEqual([{ id: PTY_1, paneKey: 'tab-dead:0' }]) expect(handler.activePtyCount).toBe(0) - await expect( - dispatcher.callRequest('pty.attach', { id: PTY_1, suppressReplayNotification: true }) - ).rejects.toThrow(`PTY "${PTY_1}" not found`) + const unknownId = await dispatcher + .callRequest('pty.attach', { id: PTY_1, suppressReplayNotification: true }) + .then( + () => new Error('expected the attach to be refused'), + (error: Error) => error + ) + + // The second refusal is the shape a restarted relay gives for every id the previous one minted: + // same words, no liveness check behind them. Only the probed one may be read as a death + // (docs/reference/ssh-execution-boundary.md). + expect(unknownId.message).toContain(`PTY "${PTY_1}" not found`) + expect(isProvenExitedPtyAttachRefusal(unknownId)).toBe(false) }) it('settles concurrent immediate shutdown when attach proves the shell exited', async () => { diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 1c213325bb8..42f8bdc0a77 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -35,6 +35,7 @@ import { type RelaySpawnCwdResolution } from './pty-spawn-cwd' import { PhysicalExitTracker } from '../shared/physical-exit-tracker' +import { PTY_ATTACH_PROVEN_EXITED_MARKER } from '../shared/pty-attach-absence-evidence' import { SHELL_READY_MARKER_PREFIX } from '../main/shell-ready-marker-scanner' import { createShellStartupOutputScanState, @@ -2055,7 +2056,11 @@ export class PtyHandler { // Why: verify liveness because shells can exit without node-pty onExit. if (this.reapPtyProvenExited(managed)) { - throw new Error(`PTY "${id}" not found`) + // Why the marker: this is the ONLY not-found answer backed by a liveness check. The unmarked + // one above is also thrown for an id this session map never had — every id minted before a + // relay restart — so a client that cannot tell them apart certifies deaths it never observed + // (docs/reference/ssh-execution-boundary.md). + throw new Error(`PTY "${id}" not found (${PTY_ATTACH_PROVEN_EXITED_MARKER})`) } // Why: legacy `pty-N` ids repeated across relay generations; reject conflicting identities. diff --git a/src/shared/pty-attach-absence-evidence.ts b/src/shared/pty-attach-absence-evidence.ts new file mode 100644 index 00000000000..2b306ca40f0 --- /dev/null +++ b/src/shared/pty-attach-absence-evidence.ts @@ -0,0 +1,17 @@ +/** + * `pty.attach` refuses with `PTY "" not found` for two unrelated situations: a pid the relay + * probed and found gone, and an id its session map simply never had — which is every id minted + * before a relay restart, since ids carry a per-start mint epoch. Only the first observes the + * process, so only the first carries this marker. + * + * The marker is additive on purpose: an answer without it means "ambiguous", which is also what an + * older relay's unmarked answer means, so a client may never read a missing marker as evidence of + * anything (docs/reference/ssh-execution-boundary.md). + */ +export const PTY_ATTACH_PROVEN_EXITED_MARKER = 'process exited' + +const PROVEN_EXITED_ATTACH_REFUSAL = /PTY ".+" not found \(process exited\)/i + +export function isProvenExitedPtyAttachRefusal(error: unknown): boolean { + return PROVEN_EXITED_ATTACH_REFUSAL.test(error instanceof Error ? error.message : String(error)) +} From 04ae62202ae05036ad7332e71083d37ae29f8f6a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:13:07 -0700 Subject: [PATCH 205/398] fix(ssh): close the macOS relay's per-terminal pty fd leak (#18534) The relay asset from #17920 only rewrote the forkpty `default:` call site, which sits in the `#else` arm of PtyFork's `#if defined(__APPLE__)`. macOS takes `pty_posix_spawn`, so the asset had never patched anything a Mac executes -- and `applyNodePtyMasterCloexecPatch` returned 'fixed' for any non-Linux host without running the script at all, which is what publishes a tree to the shared native-deps cache. Stock `pty_posix_spawn` opens up to three throwaway ptys to push the real master off fds 0-2 and never closes them: the cleanup loop is `for (; count > 0; count--)`, but the first `posix_openpt()` in a running process already returns >= 2, so it breaks with `count == 0` and the body never runs -- and where it does run it closes `low_fds[count]`, never `low_fds[0]`. One orphaned /dev/ptmx fd per terminal, for the life of the relay. Ports the `low_fds` fix and the Apple-branch `pty_cloexec(master)` call from the app's `config/patches/node-pty@1.1.0.patch`, byte-identical, and runs the gate on darwin. macOS needs a different build layout than Linux: it has no `build/` at all, so the fallback moved aside is `prebuilds/darwin-` -- which is also what makes node-pty's install script fall through from "prebuild found" to node-gyp -- and the compile writes a `build/Release` the loader checks first. Verification is per-platform too: Linux's leak is inheritance (/proc), macOS's is self-held (lsof). Also corrects the asset's claim that "macOS re-opens the tty through uv_tty_init's cloexec dup". Measured false: FD_CLOEXEC is not set on the master. What protects it is POSIX_SPAWN_CLOEXEC_DEFAULT, one option away from gone since uid/gid drops libuv back to fork()/exec() -- so the master is now marked there too. Measured on darwin-arm64, one PTY per open/close cycle in a relay-shaped dir running the relay's own commands: before cycle:ptmx 1:1 2:2 3:3 ... 10:10 (10 after a settle) after cycle:ptmx 1:0 2:0 3:0 ... 10:0 (0 after a settle) Linux re-verified in docker node:22: inherited before, isolated after, `already-patched` on the second run. Refs #17915 Refs #8362 --- .../node-pty-1.1.0-master-cloexec-patch.cjs | 202 ++++++++++++++---- .../node-pty-master-cloexec-patch.test.mjs | 138 ++++++++++-- src/main/ssh/ssh-relay-deploy.ts | 37 ++-- ...h-relay-pty-master-cloexec-install.test.ts | 55 +++-- 4 files changed, 343 insertions(+), 89 deletions(-) diff --git a/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs b/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs index f4f4f87619a..40bc0350d86 100644 --- a/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs +++ b/config/relay-assets/node-pty-1.1.0-master-cloexec-patch.cjs @@ -1,16 +1,34 @@ /** - * Relay-side pty-master close-on-exec patch for node-pty 1.1.0 (#17915). + * Relay-side pty fd-leak patch for node-pty 1.1.0 (#17915). * * The app gets this through pnpm `patchedDependencies`; the relay installs stock - * node-pty from npm onto the host, where no pnpm patch reaches. Without it every - * later child of the relay -- pty children, git helpers, probes, agent CLIs -- - * inherits each live master fd and keeps its /dev/pts device alive for the life - * of the relay (#8362). + * node-pty from npm onto the host, where no pnpm patch reaches. Stock 1.1.0 leaks + * a pty fd on both Unix relay platforms, by two unrelated bugs on two code paths. * - * Linux only, deliberately: it is the only relay platform that takes forkpty()'s - * no-atomic-O_CLOEXEC path, and the only one that already compiles node-pty at - * install time, so the rebuild costs a second compile rather than a first one. - * macOS re-opens the tty through uv_tty_init's cloexec dup and Windows has no fds. + * Linux takes forkpty(), which has no atomic O_CLOEXEC, so every later child of + * the relay -- pty children, git helpers, probes, agent CLIs -- inherits each live + * master and keeps its /dev/pts device alive for the life of the relay (#8362). + * + * macOS takes pty_posix_spawn(), which opens up to three throwaway ptys to push + * the real master off fds 0-2 and then never closes them: the cleanup loop is + * `for (; count > 0; count--)`, but in any running process the first posix_openpt() + * already returns >= 2, so the loop breaks with count == 0 and its body never runs + * -- and where it does run it closes low_fds[count], never low_fds[0]. Measured on + * darwin-arm64: one orphaned /dev/ptmx fd per terminal, never returned. + * + * macOS does not inherit the master into spawned children today, but not because it + * is marked: FD_CLOEXEC is not set on it (`lsof +fg` shows R,W,NB, no CX). What + * closes it is POSIX_SPAWN_CLOEXEC_DEFAULT in pty_posix_spawn's spawn flags, an + * Apple-only flag that closes every fd in the child. That is one option away from + * gone -- setting uid/gid drops libuv back to fork()/exec(), which honors nothing + * but FD_CLOEXEC -- so the master is marked on the Apple path too, exactly as the + * app's pnpm patch marks it. Windows has no fds and is excluded. + * + * The compile it buys differs by platform. Linux relays already run node-gyp at + * install time (1.1.0 ships no linux prebuild), so this is a second compile on a + * path that already compiles. macOS runs the shipped darwin prebuild and has no + * build/ at all, so this is its first compile -- the price of the only fix there + * is, since the bug is in the source that prebuild was built from. * * Non-fatal by construction: the working build is moved aside before anything is * touched and moved back on any failure, and a failed attempt drops a skip marker @@ -31,7 +49,7 @@ const { dirname, join, resolve } = require('node:path') const EXPECTED_NODE_PTY_VERSION = '1.1.0' const ORIGINAL_SOURCE_SHA256 = '5e1005d6bdcfbe97b486ee415419fe7adae99035047f07340fbad36419e0bae6' -const PATCHED_SOURCE_SHA256 = '97dea52199216c01b62070758f0f38621ae53adc16c221271dd35ae2d8ee3482' +const PATCHED_SOURCE_SHA256 = '3e6bc1a688aae187d231687130cfc0a11781c672f5f616d73183d471ee8ee65c' const STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' const SKIP_MARKER_FILENAME = '.node-pty-cloexec-skip' @@ -97,7 +115,56 @@ const FORKPTY_CALL_SITE = [ ` ] -const REPLACEMENTS = [FORWARD_DECLARATION, DEFINITION, FORKPTY_CALL_SITE] +// Apple never reaches FORKPTY_CALL_SITE: `default:` sits in the `#else` arm of PtyFork's +// `#if defined(__APPLE__)`, so before this pair the asset patched nothing macOS executes. +const POSIX_SPAWN_CALL_SITE = [ + ` if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } +#else +`, + ` if (pty_nonblock(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to nonblocking."); + } + if (pty_cloexec(master) == -1) { + throw Napi::Error::New(napiEnv, "Could not set master fd to close-on-exec."); + } +#else +` +] + +// The throwaway ptys pty_posix_spawn opens to keep the real master off fds 0-2. Byte-identical to +// the app's pnpm patch, so both trees compile the same cleanup. +const LOW_FDS_DECLARATION = [ + ` int low_fds[3]; + size_t count = 0; +`, + ` int low_fds[3] = {-1, -1, -1}; + size_t count = 0; +` +] + +const LOW_FDS_CLEANUP = [ + ` for (; count > 0; count--) { + close(low_fds[count]); + } +`, + ` for (size_t i = 0; i <= count && i < 3; i++) { + if (low_fds[i] != -1) { + close(low_fds[i]); + } + } +` +] + +const REPLACEMENTS = [ + FORWARD_DECLARATION, + DEFINITION, + POSIX_SPAWN_CALL_SITE, + FORKPTY_CALL_SITE, + LOW_FDS_DECLARATION, + LOW_FDS_CLEANUP +] function sourceSha256(source) { return createHash('sha256').update(source).digest('hex') @@ -188,10 +255,12 @@ function rebuildNodePty(relayDir) { } } -// Why a child: a bad build can abort the process on require, which would strand the -// moved-aside working build. Why the reachability check: a host without /proc cannot -// show inheritance, and an unobservable flag is not evidence the rebuild was wrong. -const VERIFY_SCRIPT = ` +// Why a child, for both scripts below: a bad build can abort the process on require, which would +// strand the moved-aside working build. Why each ends in a reachability check: a host that cannot +// show its fds says nothing, and an unobservable flag is not evidence the rebuild was wrong. +// +// Linux's leak is inheritance, so the observation is a later plain child's /proc/self/fd. +const VERIFY_INHERITANCE_SCRIPT = ` const pty = require(process.argv[1]); const term = pty.spawn('/bin/sh', ['-c', 'exit 0'], { name: 'xterm-256color', cols: 80, rows: 24, cwd: process.cwd(), env: process.env @@ -200,13 +269,39 @@ const probe = require('node:child_process').spawnSync('/bin/sh', ['-c', 'ls -l / try { term.kill() } catch {} const listing = probe.stdout || ''; if (probe.status !== 0 || !listing.includes('->')) { console.log('UNVERIFIED'); process.exit(0) } -console.log(listing.includes('ptmx') ? 'INHERITED' : 'ISOLATED'); +console.log(listing.includes('ptmx') ? 'LEAKED' : 'ISOLATED'); process.exit(0); ` -/** 'isolated' when a later plain child no longer inherits the master, 'unverified' when /proc cannot say. */ -function verifyMasterNotInheritedByLaterChild(relayDir) { - const result = spawnSync(process.execPath, ['-e', VERIFY_SCRIPT, nodePtyDir(relayDir)], { +// Apple's leak is self-held, not inherited, so the observation is this process's own fd table: +// N live ptys must account for exactly N /dev/ptmx rows. A stock build shows 2N -- the master plus +// the throwaway pty_posix_spawn opened and never closed. lsof, not /proc, because macOS has no +// /proc; a host without lsof cannot say, which is 'unverified', not a failed patch. +const VERIFY_SELF_FDS_SCRIPT = ` +const pty = require(process.argv[1]); +const terms = []; +for (let i = 0; i < 3; i++) { + terms.push(pty.spawn('/bin/sh', ['-c', 'sleep 30'], { + name: 'xterm-256color', cols: 80, rows: 24, cwd: process.cwd(), env: process.env + })); +} +const probe = require('node:child_process').spawnSync('/bin/sh', ['-c', 'lsof -p ' + process.pid], { encoding: 'utf8', maxBuffer: 1 << 24 }); +for (const term of terms) { try { term.kill() } catch {} } +const rows = (probe.stdout || '').split('\\n').filter((line) => line.includes('/dev/ptmx')); +if (probe.status !== 0 || rows.length < terms.length) { console.log('UNVERIFIED'); process.exit(0) } +console.log(rows.length > terms.length ? 'LEAKED' : 'ISOLATED'); +process.exit(0); +` + +const LEAK_MESSAGE = { + darwin: 'rebuilt node-pty still leaks a throwaway pty fd per spawn', + linux: 'rebuilt node-pty still leaks the pty master into later children' +} + +/** 'isolated' when the platform's leak is gone, 'unverified' when the host cannot show it. */ +function verifyNoPtyFdLeak(relayDir, platform) { + const script = platform === 'darwin' ? VERIFY_SELF_FDS_SCRIPT : VERIFY_INHERITANCE_SCRIPT + const result = spawnSync(process.execPath, ['-e', script, nodePtyDir(relayDir)], { cwd: relayDir, encoding: 'utf8', timeout: VERIFY_TIMEOUT_MS, @@ -219,22 +314,53 @@ function verifyMasterNotInheritedByLaterChild(relayDir) { `rebuilt node-pty did not load: ${tail || result.error?.message || result.signal}` ) } - if (output.includes('INHERITED')) { - throw new Error('rebuilt node-pty still leaks the pty master into later children') + if (output.includes('LEAKED')) { + throw new Error(LEAK_MESSAGE[platform] || LEAK_MESSAGE.linux) } return output.includes('ISOLATED') ? 'isolated' : 'unverified' } -function rollback(relayDir, releaseDir, backupDir) { - rmSync(releaseDir, { recursive: true, force: true }) +/** + * What gets moved aside before the compile, and where the compile writes. + * + * Linux ships no prebuild, so `build/Release` is both the working build and the compile's output, + * and moving it aside only arms the rollback. macOS runs `prebuilds/darwin-` and has no + * `build/` at all, so the compile writes a new `build/Release` -- which node-pty's loader checks + * ahead of `prebuilds`. Moving `prebuilds` aside does double duty there: it arms the rollback and + * it is what makes node-pty's install script fall through from "prebuild found" to `node-gyp + * rebuild`. Deliberately not `npm_config_build_from_source`, which deletes the prebuilds outright + * and would leave nothing to roll back to. + */ +function buildLayout(relayDir, platform, arch) { + const ptyDir = nodePtyDir(relayDir) + const compiledDir = join(ptyDir, 'build', 'Release') + if (platform === 'darwin') { + const prebuildsDir = join(ptyDir, 'prebuilds') + return { + compiledDir, + movedDir: prebuildsDir, + workingBuildPath: join(prebuildsDir, `darwin-${arch}`, 'pty.node'), + missingStatus: 'skipped:no-prebuild' + } + } + return { + compiledDir, + movedDir: compiledDir, + workingBuildPath: join(compiledDir, 'pty.node'), + missingStatus: 'skipped:no-compiled-build' + } +} + +function rollback(relayDir, layout, backupDir) { + rmSync(layout.compiledDir, { recursive: true, force: true }) try { revertNodePtyMasterCloexecSource(relayDir) } catch { // The build that is about to be restored predates the patch either way. } if (existsSync(backupDir)) { - mkdirSync(dirname(releaseDir), { recursive: true }) - renameSync(backupDir, releaseDir) + mkdirSync(dirname(layout.movedDir), { recursive: true }) + renameSync(backupDir, layout.movedDir) } } @@ -244,16 +370,17 @@ function rollback(relayDir, releaseDir, backupDir) { */ function applyNodePtyMasterCloexecPatch(relayDir = process.cwd(), options = {}) { const platform = options.platform || process.platform + const arch = options.arch || process.arch const rebuild = options.rebuild || rebuildNodePty - const verify = options.verify || verifyMasterNotInheritedByLaterChild - if (platform !== 'linux') { - return 'skipped:not-linux' + const verify = options.verify || verifyNoPtyFdLeak + if (platform !== 'linux' && platform !== 'darwin') { + return 'skipped:unsupported-platform' } const skipMarkerPath = join(relayDir, SKIP_MARKER_FILENAME) if (existsSync(skipMarkerPath)) { return 'skipped:earlier-attempt-failed' } - const releaseDir = join(nodePtyDir(relayDir), 'build', 'Release') + const layout = buildLayout(relayDir, platform, arch) const backupDir = join(nodePtyDir(relayDir), BACKUP_DIRNAME) // A backup stranded by a connection that died mid-rebuild is stale by definition: // whatever repaired node-pty since built from the source now on disk. @@ -272,25 +399,28 @@ function applyNodePtyMasterCloexecPatch(relayDir = process.cwd(), options = {}) if (hash !== ORIGINAL_SOURCE_SHA256) { return 'skipped:unexpected-source' } - // No compiled build means the host runs a prebuild or nothing at all; rebuilding - // could only take away the artifact the probe just proved loadable. - if (!existsSync(join(releaseDir, 'pty.node'))) { - return 'skipped:no-compiled-build' + // Nothing to fall back on means the host runs neither a compile nor the prebuild + // this platform expects; rebuilding could only take away the artifact the probe + // just proved loadable. + if (!existsSync(layout.workingBuildPath)) { + return layout.missingStatus } try { - renameSync(releaseDir, backupDir) + renameSync(layout.movedDir, backupDir) } catch (err) { return `skipped:${err.message}` } try { patchNodePtyMasterCloexecSource(relayDir) rebuild(relayDir) - const verdict = verify(relayDir) + const verdict = verify(relayDir, platform) + // Discarded, not restored: a tree that gets published must hold no unpatched binary the + // loader could still fall back to. A later repair recompiles from the patched source. rmSync(backupDir, { recursive: true, force: true }) return verdict === 'isolated' ? 'patched' : 'patched-unverified' } catch (err) { - rollback(relayDir, releaseDir, backupDir) + rollback(relayDir, layout, backupDir) // Bounded on purpose: one compile attempt per relay directory, never a retry loop. try { writeFileSync(skipMarkerPath, `${new Date().toISOString()} ${err.message}\n`) diff --git a/config/scripts/node-pty-master-cloexec-patch.test.mjs b/config/scripts/node-pty-master-cloexec-patch.test.mjs index 16013cbf0a3..e323c5d61c8 100644 --- a/config/scripts/node-pty-master-cloexec-patch.test.mjs +++ b/config/scripts/node-pty-master-cloexec-patch.test.mjs @@ -27,7 +27,7 @@ afterEach(() => { } }) -describe('SSH relay node-pty pty-master close-on-exec patch', () => { +describe('SSH relay node-pty pty fd-leak patch', () => { it('adds the forkpty close-on-exec call and reverts to the published bytes', () => { const fixture = writeRelayFixture() @@ -44,6 +44,27 @@ describe('SSH relay node-pty pty-master close-on-exec patch', () => { expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) }) + it('rewrites the Apple branch, which is the only one macOS executes', () => { + const fixture = writeRelayFixture() + patchNodePtyMasterCloexecSource(fixture.root) + const patched = readFileSync(fixture.sourcePath, 'utf8') + + // Stock's cleanup never runs: the first posix_openpt() already returns >= 2, so the loop + // breaks with count == 0 -- and where it does run it closes low_fds[count], never low_fds[0]. + expect(STOCK_SOURCE).toContain('for (; count > 0; count--) {') + expect(patched).not.toContain('for (; count > 0; count--) {') + expect(patched).toContain('int low_fds[3] = {-1, -1, -1};') + expect(patched).toContain('for (size_t i = 0; i <= count && i < 3; i++) {') + + // `default:` sits in the `#else` arm of PtyFork's `#if defined(__APPLE__)`, so marking only + // the forkpty call site left the master macOS actually opens unmarked. + expect(patched).toContain( + ' if (pty_cloexec(master) == -1) {\n' + + ' throw Napi::Error::New(napiEnv, "Could not set master fd to close-on-exec.");\n' + + ' }\n#else\n' + ) + }) + it('refuses a different node-pty version or an unrecognized source', () => { const wrongVersion = writeRelayFixture({ version: '1.2.0-beta.4' }) expect(() => patchNodePtyMasterCloexecSource(wrongVersion.root)).toThrow('expected 1.1.0') @@ -139,19 +160,81 @@ describe('SSH relay node-pty pty-master close-on-exec patch', () => { expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) }) - it('never compiles on a platform that does not leak', () => { - for (const platform of ['darwin', 'win32']) { - const fixture = writeRelayFixture() - const calls = [] - const status = applyNodePtyMasterCloexecPatch(fixture.root, { - platform, - rebuild: () => calls.push('rebuild'), - verify: () => 'isolated' - }) - expect(status).toBe('skipped:not-linux') - expect(calls).toEqual([]) - expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) - } + it('never compiles on a platform with no pty fds to leak', () => { + const fixture = writeRelayFixture() + const calls = [] + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'win32', + rebuild: () => calls.push('rebuild'), + verify: () => 'isolated' + }) + expect(status).toBe('skipped:unsupported-platform') + expect(calls).toEqual([]) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + }) + + it('compiles a macOS install out from under its shipped prebuild', () => { + // macOS has no build/ at all: node-pty runs `prebuilds/darwin-`, built from the leaky + // source. Moving `prebuilds` aside is what both arms the rollback and makes node-pty's own + // install script fall through from "prebuild found" to node-gyp. + const fixture = writeRelayFixture({ platform: 'darwin' }) + const prebuildsPresentDuringRebuild = [] + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'darwin', + arch: fixture.arch, + rebuild: () => { + prebuildsPresentDuringRebuild.push(existsSync(fixture.prebuildsDir)) + writeCompiledBuild(fixture, 'patched-build') + }, + verify: () => 'isolated' + }) + + expect(status).toBe('patched') + expect(prebuildsPresentDuringRebuild).toEqual([false]) + expect(readFileSync(fixture.compiledPath, 'utf8')).toBe('patched-build') + // The published tree must hold no unpatched binary: node-pty's loader checks build/Release + // first, but falls back to a prebuild if that ever fails to load. + expect(existsSync(fixture.prebuildsDir)).toBe(false) + expect(existsSync(fixture.backupDir)).toBe(false) + }) + + it('restores the macOS prebuild when the first compile fails', () => { + // A macOS host has no toolchain guarantee at all, so this is the common failure, not the rare + // one -- and the relay has to come back on the prebuild exactly as it was installed. + const fixture = writeRelayFixture({ platform: 'darwin' }) + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'darwin', + arch: fixture.arch, + rebuild: () => { + writeCompiledBuild(fixture, 'half-built') + throw new Error('npm rebuild node-pty exited 1: no C++ toolchain') + }, + verify: () => 'isolated' + }) + + expect(status).toContain('failed:') + expect(readFileSync(fixture.buildPath, 'utf8')).toBe('stock-build') + expect(existsSync(fixture.compiledPath)).toBe(false) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) + expect(existsSync(fixture.skipMarkerPath)).toBe(true) + }) + + it('will not rebuild a macOS install that has no prebuild to fall back on', () => { + const fixture = writeRelayFixture({ platform: 'darwin', build: false }) + const calls = [] + + const status = applyNodePtyMasterCloexecPatch(fixture.root, { + platform: 'darwin', + arch: fixture.arch, + rebuild: () => calls.push('rebuild'), + verify: () => 'isolated' + }) + + expect(status).toBe('skipped:no-prebuild') + expect(calls).toEqual([]) + expect(readFileSync(fixture.sourcePath, 'utf8')).toBe(STOCK_SOURCE) }) it('leaves an already patched install alone', () => { @@ -201,19 +284,35 @@ describe('SSH relay node-pty pty-master close-on-exec patch', () => { }) }) -function writeRelayFixture({ version = '1.1.0', source = STOCK_SOURCE, build = true } = {}) { +/** + * `buildPath` is the working build the patch has to be able to fall back on, which differs by + * platform: Linux compiles into build/Release at install time, macOS runs a shipped prebuild and + * has no build/ at all. `compiledPath` is where the rebuild writes on either. + */ +function writeRelayFixture({ + version = '1.1.0', + source = STOCK_SOURCE, + build = true, + platform = 'linux', + arch = 'arm64' +} = {}) { const root = mkdtempSync(join(projectDir, '.node-pty-cloexec-patch-test-')) cleanupDirs.push(root) const nodePtyDir = join(root, 'node_modules', 'node-pty') const sourcePath = join(nodePtyDir, 'src', 'unix', 'pty.cc') - const buildPath = join(nodePtyDir, 'build', 'Release', 'pty.node') + const compiledPath = join(nodePtyDir, 'build', 'Release', 'pty.node') + const prebuildsDir = join(nodePtyDir, 'prebuilds') mkdirSync(join(nodePtyDir, 'src', 'unix'), { recursive: true }) writeFileSync(join(nodePtyDir, 'package.json'), JSON.stringify({ version })) writeFileSync(sourcePath, source) const fixture = { root, + arch, sourcePath, - buildPath, + compiledPath, + prebuildsDir, + buildPath: + platform === 'darwin' ? join(prebuildsDir, `darwin-${arch}`, 'pty.node') : compiledPath, backupDir: join(nodePtyDir, '.orca-cloexec-prepatch-release'), skipMarkerPath: join(root, SKIP_MARKER_FILENAME) } @@ -227,3 +326,8 @@ function writeBuild(fixture, contents) { mkdirSync(resolve(fixture.buildPath, '..'), { recursive: true }) writeFileSync(fixture.buildPath, contents) } + +function writeCompiledBuild(fixture, contents) { + mkdirSync(resolve(fixture.compiledPath, '..'), { recursive: true }) + writeFileSync(fixture.compiledPath, contents) +} diff --git a/src/main/ssh/ssh-relay-deploy.ts b/src/main/ssh/ssh-relay-deploy.ts index c5cb2dd552c..257eb984caa 100644 --- a/src/main/ssh/ssh-relay-deploy.ts +++ b/src/main/ssh/ssh-relay-deploy.ts @@ -52,6 +52,7 @@ import { RELAY_DEPLOY_TIMEOUT_MS } from './ssh-relay-deploy-timing' import { createSshOperationAbortError, shellEscape } from './ssh-connection-utils' +import { isWindowsRelayPlatform } from '../../shared/relay-artifacts' import { probeBuildToolchain, formatMissingToolchainError, @@ -745,27 +746,29 @@ const NODE_PTY_CONSOLE_LIST_PATCH_FILENAME = 'node-pty-1.1.0-console-list-agent- const NODE_PTY_MASTER_CLOEXEC_PATCH_FILENAME = 'node-pty-1.1.0-master-cloexec-patch.cjs' const NODE_PTY_CLOEXEC_STATUS_PREFIX = 'ORCA-NPTY-CLOEXEC:' /** - * Whether the tree the patch left behind still leaks the pty master into every later child. - * `fixed` is the only outcome a shared cache entry may be published from. + * Whether the tree the patch left behind still leaks a pty fd -- the master into every later child + * on Linux, a throwaway /dev/ptmx per spawn on macOS. `fixed` is the only outcome a shared cache + * entry may be published from. */ type NodePtyMasterCloexecOutcome = 'fixed' | 'unfixed' /** * The statuses that leave a non-leaking tree. Deliberately an allowlist, not a `failed:` denylist: - * the script's `skipped:` family is mixed. `skipped:not-linux` is a platform that never leaks, but - * `skipped:earlier-attempt-failed`, `skipped:no-compiled-build`, `skipped:unexpected-source` and - * the two `skipped:` forms all mean the patch was refused and the leaky build is still on - * disk -- indistinguishable from `failed:` as far as what gets published. + * the script's `skipped:` family is mixed. `skipped:unsupported-platform` is a platform that never + * leaks, but `skipped:earlier-attempt-failed`, `skipped:no-compiled-build`, `skipped:no-prebuild`, + * `skipped:unexpected-source` and the two `skipped:` forms all mean the patch was refused + * and the leaky build is still on disk -- indistinguishable from `failed:` as far as what gets + * published. */ const NODE_PTY_CLOEXEC_FIXED_STATUSES: ReadonlySet = new Set([ 'patched', - // The rebuild ran from patched source; only the isolation check could not observe the result. - // An unobservable check is not a failed patch, and treating it as one would disable the shared - // cache on every host without `lsof`. + // The rebuild ran from patched source; only the leak check could not observe the result. An + // unobservable check is not a failed patch, and treating it as one would disable the shared + // cache on every host without `/proc` or `lsof`. 'patched-unverified', 'already-patched', - // Unreachable while the platform gate below short-circuits first, but it is the one `skipped:` - // that means "nothing to fix" rather than "would not fix it". - 'skipped:not-linux' + // Unreachable while the platform gate below short-circuits Windows first, but it is the one + // `skipped:` that means "nothing to fix" rather than "would not fix it". + 'skipped:unsupported-platform' ]) // Exported for the relay-native-dependency-coverage test, which asserts every // native addon the relay bundle imports is either installed here or explicitly @@ -1300,7 +1303,7 @@ async function installNativeDeps( } /** - * Re-apply the pty-master FD_CLOEXEC patch the app gets from pnpm to the host's npm copy (#17915). + * Re-apply the pty fd-leak patch the app gets from pnpm to the host's npm copy (#17915). * * Why it is safe to rebuild under a live relay: this only runs from installNativeDeps, so only on a * freshly created directory or a locked repair, and a relay already serving PTYs has pty.node mapped @@ -1324,9 +1327,11 @@ async function applyNodePtyMasterCloexecPatch( nodePath: string, signal?: AbortSignal ): Promise { - // Linux is the only relay platform that takes forkpty()'s no-O_CLOEXEC path; macOS and Windows - // ship prebuilds, so forcing a rebuild there would add a first compile to fix nothing. - if (isWindowsRemoteHost(hostPlatform) || !platform.startsWith('linux')) { + // Both Unix relay platforms leak, by different bugs: Linux inherits the master through forkpty()'s + // no-O_CLOEXEC path, macOS orphans one throwaway /dev/ptmx fd per spawn in pty_posix_spawn. Only + // Windows, which has no fds, is short-circuited -- and answering 'fixed' from a gate that ran + // nothing is exactly how a leaking darwin tree got published to the shared cache. + if (isWindowsRemoteHost(hostPlatform) || isWindowsRelayPlatform(platform)) { return 'fixed' } try { diff --git a/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts b/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts index 66af01fa43c..b0ca39ec2b4 100644 --- a/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts +++ b/src/main/ssh/ssh-relay-pty-master-cloexec-install.test.ts @@ -88,11 +88,12 @@ import { const PATCH_ASSET = 'node-pty-1.1.0-master-cloexec-patch.cjs' /** - * The relay installs stock node-pty from npm, so the app's pnpm patch never reaches it and every - * later child of the relay inherits a live pty master (#17915). The compile that closes it sits on + * The relay installs stock node-pty from npm, so the app's pnpm patch never reaches it and the + * relay leaks a pty fd per terminal (#17915) -- the master into every later child on Linux, an + * orphaned /dev/ptmx throwaway in `pty_posix_spawn` on macOS. The compile that closes both sits on * the connect path, so what these specs pin is the blast radius, not the patch itself. */ -describe('relay pty-master close-on-exec patch on the install path', () => { +describe('relay pty fd-leak patch on the install path', () => { const sftpCapture: SftpWriteCapture = { paths: [], contents: {}, @@ -211,9 +212,9 @@ describe('relay pty-master close-on-exec patch on the install path', () => { }) it('does not publish a tree the patch refused to touch', async () => { - // `skipped:` is not one verdict. Every form except `skipped:not-linux` means the patch was - // declined and the leaky build is still on disk, which is indistinguishable from `failed:` - // as far as what would get published. + // `skipped:` is not one verdict. Every form except `skipped:unsupported-platform` means the + // patch was declined and the leaky build is still on disk, which is indistinguishable from + // `failed:` as far as what would get published. const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) try { const conn = makeMockConnection(sftpCapture) @@ -281,25 +282,39 @@ describe('relay pty-master close-on-exec patch on the install path', () => { expect(patchCommands()).toEqual([]) }) - it('never adds a compile to a macOS relay, which does not leak the master', async () => { + it('runs the patch on a macOS relay, which orphans a /dev/ptmx fd per spawn', async () => { + // macOS takes `pty_posix_spawn`, not forkpty, so the asset's original replacements rewrote + // nothing macOS executes -- and this gate answered 'fixed' without running anything, which is + // exactly what publishes to the shared cache. Every later host on the machine then linked a + // tree that leaks one /dev/ptmx fd per terminal, measured +1 per open/close cycle on + // darwin-arm64. macOS pays a first compile here, unlike Linux's second, and that is the price. vi.mocked(parseUnameToRelayPlatform).mockReturnValue('darwin-arm64') const conn = makeMockConnection(sftpCapture) - feed([ - ...makeStagedFirstInstallExecPrefix(), - '', // npm install native deps - '', // chmod prebuilds - 'ORCA-NPTY-PROBE-OK\n', - '', // rm probe stderr - '', // promote into the shared native-deps cache - '', // clean stage root - 'DEAD', - '', // publish the per-launch credential - 'READY' - ]) + feed(makeExecResponses({ npmInstall: 'ok', probe: 'ok' })) await deployAndLaunchRelay(conn) - expect(patchCommands()).toEqual([]) + expect(patchCommands()).toHaveLength(1) + expect(promoted()).toBe(true) + }) + + it('does not publish a macOS tree whose compile failed', async () => { + // The darwin rollback restores the shipped prebuild, so the relay still works -- and that is + // precisely why the status, not the exit code, has to decide publishability: a rolled-back + // macOS tree probes loadable and still leaks. + const warn = vi.spyOn(console, 'warn').mockImplementation(() => {}) + try { + vi.mocked(parseUnameToRelayPlatform).mockReturnValue('darwin-arm64') + const conn = makeMockConnection(sftpCapture) + feed(firstInstallReporting('failed:npm rebuild node-pty exited 1: gyp ERR! not ok')) + + await deployAndLaunchRelay(conn) + + expect(patchCommands()).toHaveLength(1) + expect(promoted()).toBe(false) + } finally { + warn.mockRestore() + } }) it('connects anyway when the patch command fails outright', async () => { From 3c914048206a333aa106eba3598464ea5efa5b83 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:21:52 -0700 Subject: [PATCH 206/398] fix(worktrees): route managed worktree removal by resolved execution host (#18529) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `removeManagedWorktree` resolved its host once — for the metadata prune (`cleanupHostId ?? getRepoExecutionHostId(repo)`) — and then read raw `repo.connectionId` for every step that touches the filesystem: the `git worktree list` deciding whether the path is registered, the provider handed to the unregistered-removal branch, the registered-remote-vs-local fork, and the PTY/history teardown. One function, two spellings. For a row naming its owner only as `executionHostId: 'ssh:'` — the exact class #18296 names — the list ran on the client against a remote path, `removeRuntimeUnregisteredWorktree` was entered with `provider: null`, and the metadata was pruned under `ssh:` while a same-named *local* directory was the one considered for deletion (#11163). #18358 made this reachable: it migrated the cleanup scan, so `executionHostId`-only rows now surface as removable candidates, but removal did not move with it. Routing is now one answer for the whole removal, taken from the host the prune already used, through #18296's host-keyed dispatch. The ambiguous `provider: SshGitProvider | null` carrier is deleted from the callees rather than supplemented, so every remaining reader is a compile error in the typed modules that do the destructive work (`runtime-unregistered-worktree-removal`, `runtime-registered-remote-worktree-removal`, `runtime-worktree-filesystem`). The orchestrator itself carries `@ts-nocheck` from its mechanical split, so that guarantee does not reach it — tests cover it instead. Every change is in the refusing direction; nothing became more aggressive: - an `ssh:` host with no registered provider throws instead of deleting a client-side path (`requireSshGitProvider` already threw for rows that spelled the same host as `connectionId`); - `runtime:` throws rather than dialling a same-named target in this client's namespace, matching `workspace-cleanup-git-route` and `runtime-git-command-target`; - the folder-workspace teardown resolves its connection instead of reading the raw field, so a `runtime:` row stops dialling the wrong namespace. No wire or persistence change: `removeWorktreeMetadataAndHistory` already took the resolved host, and the removal RPC result shape is untouched. --- .../orca-runtime-remove-managed-worktree.ts | 49 ++-- ...untime-remove-orphan-or-folder-worktree.ts | 6 +- ...removal-and-reconciliation-part-03.spec.ts | 64 ++++-- .../worktree-removal-execution-host.spec.ts | 216 ++++++++++++++++++ src/main/runtime/orca-runtime.test.ts | 1 + ...time-registered-remote-worktree-removal.ts | 5 +- .../runtime-unregistered-worktree-removal.ts | 45 ++-- .../runtime/runtime-worktree-filesystem.ts | 20 +- ...ktree-removal-execution-host-route.test.ts | 100 ++++++++ .../worktree-removal-execution-host-route.ts | 81 +++++++ 10 files changed, 521 insertions(+), 66 deletions(-) create mode 100644 src/main/runtime/orca-runtime-tests/worktree-removal-execution-host.spec.ts create mode 100644 src/main/worktree-removal-execution-host-route.test.ts create mode 100644 src/main/worktree-removal-execution-host-route.ts diff --git a/src/main/runtime/orca-runtime-remove-managed-worktree.ts b/src/main/runtime/orca-runtime-remove-managed-worktree.ts index 5b45abd13fa..25a294c2a90 100644 --- a/src/main/runtime/orca-runtime-remove-managed-worktree.ts +++ b/src/main/runtime/orca-runtime-remove-managed-worktree.ts @@ -10,8 +10,7 @@ import { preservedBranchCleanupScopeKey } from '../../shared/preserved-branch-cl import { getRuntimeWorktreeRemovalOptionsKey } from './runtime-worktree-selection' import { withWorktreeSpan } from '../observability/instrumentation' import { invalidateAuthorizedRootsCache } from '../ipc/filesystem-auth' -import { requireSshGitProvider } from '../providers/ssh-git-dispatch' -import { getSshFilesystemProvider } from '../providers/ssh-filesystem-dispatch' +import { resolveWorktreeRemovalRoute } from '../worktree-removal-execution-host-route' import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options' import { listWorktreesStrict } from '../git/worktree' import { findRegisteredDeletableWorktree } from '../worktree-removal-safety' @@ -79,22 +78,24 @@ export class OrcaRuntimeWithRemoveManagedWorktree extends OrcaRuntimeWithCreateM if (orphanOrFolderResult) { return orphanOrFolderResult } - const provider = repo.connectionId ? requireSshGitProvider(repo.connectionId) : null - const fsProvider = repo.connectionId ? getSshFilesystemProvider(repo.connectionId) : null - const localWorktreeGitOptions = repo.connectionId - ? {} - : getLocalProjectWorktreeGitOptions(this.requireStore(), repo) + // One host for the whole removal. Listing on a different host from the one the prune and + // the delete use is how an `executionHostId: 'ssh:*'`-only row got listed remotely and + // deleted here; the route refuses rather than falling back to this machine. + const route = resolveWorktreeRemovalRoute(removalHostId) + const localWorktreeGitOptions = + route.kind === 'ssh' ? {} : getLocalProjectWorktreeGitOptions(this.requireStore(), repo) const hasLocalWorktreeGitOptions = Object.keys(localWorktreeGitOptions).length > 0 - const registeredWorktrees = repo.connectionId - ? await provider!.listWorktrees(repo.path) - : hasLocalWorktreeGitOptions - ? await listWorktreesStrict(repo.path, localWorktreeGitOptions) - : await listWorktreesStrict(repo.path) + const registeredWorktrees = + route.kind === 'ssh' + ? await route.provider.listWorktrees(repo.path) + : hasLocalWorktreeGitOptions + ? await listWorktreesStrict(repo.path, localWorktreeGitOptions) + : await listWorktreesStrict(repo.path) const removedMeta = resolveWorktreeRemovalMetadata( store, removalTarget.repoId, removalTarget.id, - cleanupHostId ?? getRepoExecutionHostId(repo) + removalHostId ) const removedPushTarget = removedMeta?.pushTarget ?? removalTarget.pushTarget const registeredWorktree = findRegisteredDeletableWorktree( @@ -111,8 +112,7 @@ export class OrcaRuntimeWithRemoveManagedWorktree extends OrcaRuntimeWithCreateM removedPushTarget, force, allowUnverifiedPtyStop, - provider, - fsProvider: fsProvider ?? null, + route, localOptions: localWorktreeGitOptions, store, acquireWatcherRemoval: this.acquireFileWatcherRemoval, @@ -123,7 +123,7 @@ export class OrcaRuntimeWithRemoveManagedWorktree extends OrcaRuntimeWithCreateM }), deleteHistory: () => deleteRemoteWorktreeHistory( - repo.connectionId ? this.getSshProviderFn?.(repo.connectionId) : undefined, + route.kind === 'ssh' ? this.getSshProviderFn?.(route.connectionId) : undefined, removalTarget.id ), finishRemoval: () => { @@ -145,13 +145,17 @@ export class OrcaRuntimeWithRemoveManagedWorktree extends OrcaRuntimeWithCreateM throw new Error(formatWorktreeRemovalError(error, canonicalWorktreePath, force)) } if ( - !repo.connectionId && + route.kind === 'local' && force === true && process.platform === 'win32' && (isWindowsAbsolutePathLike(canonicalWorktreePath) || !!localWorktreeGitOptions.wslDistro) && removedMeta && - (await isRuntimeWorktreePathMissing(repo, canonicalWorktreePath, localWorktreeGitOptions)) + (await isRuntimeWorktreePathMissing( + route.hostId, + canonicalWorktreePath, + localWorktreeGitOptions + )) ) { const removalResult = await removeStaleLocalWorktreeRegistrationAfterFilesystemRemoval({ canonicalWorktreePath, @@ -182,26 +186,27 @@ export class OrcaRuntimeWithRemoveManagedWorktree extends OrcaRuntimeWithCreateM this.notifyWorktreesChanged(repo.id) return removalResult ?? {} } - if (repo.connectionId) { + if (route.kind === 'ssh') { return removeRuntimeRegisteredRemoteWorktree({ repo, target: removalTarget, registeredWorktree, removedPushTarget, store, - provider: provider!, + provider: route.provider, + connectionId: route.connectionId, force, allowUnverifiedPtyStop, deleteBranch, acquireWatcherRemoval: this.acquireFileWatcherRemoval, stopPtys: () => this.stopPtysForDestructiveWorktreeRemoval(removalTarget.id, { - connectionId: repo.connectionId!, + connectionId: route.connectionId, allowUnverifiedStop: allowUnverifiedPtyStop }), deleteHistory: () => deleteRemoteWorktreeHistory( - this.getSshProviderFn?.(repo.connectionId!), + this.getSshProviderFn?.(route.connectionId), removalTarget.id ), preserveBranchHead: (result, fallbackHead) => diff --git a/src/main/runtime/orca-runtime-remove-orphan-or-folder-worktree.ts b/src/main/runtime/orca-runtime-remove-orphan-or-folder-worktree.ts index d49dc410cd5..01f4f305803 100644 --- a/src/main/runtime/orca-runtime-remove-orphan-or-folder-worktree.ts +++ b/src/main/runtime/orca-runtime-remove-orphan-or-folder-worktree.ts @@ -91,7 +91,11 @@ export async function removeOrphanOrFolderWorktree({ if (removalTarget.id === getRuntimeFolderWorkspaceRootId(repo)) { throw new Error('Cannot delete the project root workspace. Remove the folder project instead.') } - const folderConnectionId = repo.connectionId?.trim() || null + // Resolved, not raw: a folder repo naming its owner only as `executionHostId: 'ssh:*'` used to + // tear down its PTYs and history on the client. A `runtime:` host answers null — its nested + // target is addressable only inside that environment, never from this client's SSH table. + const folderHost = parseExecutionHostId(removalHostId) + const folderConnectionId = folderHost?.kind === 'ssh' ? folderHost.targetId : null const folderSshPtyProvider = folderConnectionId ? runtime.getSshProviderFn?.(folderConnectionId) : undefined diff --git a/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation-part-03.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation-part-03.spec.ts index beaa91036d0..06982cf179e 100644 --- a/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation-part-03.spec.ts +++ b/src/main/runtime/orca-runtime-tests/worktree-removal-and-reconciliation-part-03.spec.ts @@ -193,41 +193,69 @@ describe('OrcaRuntimeService', () => { }) it('does not coalesce concurrent same-id removals on different hosts', async () => { + const baseRepo = store.getRepos()[0]! + // The second owner names its host only in the migrated spelling, so its removal must reach the + // SSH host rather than joining the local one and running `git worktree remove` here. + const remoteRepo = { ...baseRepo, path: '/remote/repo', executionHostId: 'ssh:host-b' } const runtimeStore = { ...store, - getRepos: () => [ - { ...store.getRepos()[0], executionHostId: 'local' }, - { ...store.getRepos()[0], executionHostId: 'runtime:env-1' } - ] + getRepos: () => [{ ...baseRepo, executionHostId: 'local' }, remoteRepo] } const runtime = createWorktreeRemovalRuntime(runtimeStore) vi.spyOn(runtime, 'acquireFileWatcherRemoval').mockResolvedValue({ finish: vi.fn() }) const bothStarted = deferred() const finishRemovals = deferred() let startedCount = 0 - vi.mocked(removeWorktree).mockImplementation(async () => { + const startRemoval = async (): Promise> => { startedCount += 1 if (startedCount === 2) { bothStarted.resolve() } await finishRemovals.promise return {} - }) + } + vi.mocked(removeWorktree).mockImplementation(startRemoval) + const provider = { + exec: vi.fn().mockResolvedValue({ stdout: '', stderr: '' }), + listWorktrees: vi.fn().mockResolvedValue([ + { + path: remoteRepo.path, + head: 'main', + branch: 'main', + isBare: false, + isMainWorktree: true + }, + { + path: TEST_WORKTREE_PATH, + head: 'def456', + branch: 'feature/test', + isBare: false, + isMainWorktree: false + } + ]), + removeWorktree: vi.fn().mockImplementation(startRemoval) + } + registerSshGitProvider('host-b', provider as never) - const local = runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'local') - const paired = runtime.removeManagedWorktree( - TEST_WORKTREE_ID, - true, - false, - false, - 'runtime:env-1' - ) + try { + const local = runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'local') + const remote = runtime.removeManagedWorktree( + TEST_WORKTREE_ID, + true, + false, + false, + 'ssh:host-b' + ) - await bothStarted.promise - expect(removeWorktree).toHaveBeenCalledTimes(2) + await bothStarted.promise + expect(removeWorktree).toHaveBeenCalledTimes(1) + expect(provider.removeWorktree).toHaveBeenCalledTimes(1) - finishRemovals.resolve() - await expect(Promise.all([local, paired])).resolves.toEqual([{}, {}]) + finishRemovals.resolve() + await expect(Promise.all([local, remote])).resolves.toEqual([{}, {}]) + } finally { + unregisterSshGitProvider('host-b') + } }) it('rejects concurrent runtime worktree removals for the same id with different options', async () => { diff --git a/src/main/runtime/orca-runtime-tests/worktree-removal-execution-host.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-removal-execution-host.spec.ts new file mode 100644 index 00000000000..8ec38e0e5e8 --- /dev/null +++ b/src/main/runtime/orca-runtime-tests/worktree-removal-execution-host.spec.ts @@ -0,0 +1,216 @@ +import { + listWorktrees, + listWorktreesStrict, + registerSshFilesystemProvider, + registerSshGitProvider, + removeWorktree, + unregisterSshFilesystemProvider, + unregisterSshGitProvider +} from '../orca-runtime-test-mocks.spec' +import type { WorktreeMeta } from '../orca-runtime-test-mocks.spec' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + TEST_WORKTREE_ID, + TEST_WORKTREE_PATH, + makeWorktreeMeta, + store +} from '../orca-runtime-test-fixtures.spec' +import { createWorktreeRemovalRuntime } from '../orca-runtime-test-scenario-builders.spec' +import type { ExecutionHostId } from '../../../shared/execution-host' + +const REMOTE_REPO_PATH = '/remote/repo' + +function missingPath(): never { + throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) +} + +function makeGitProvider(worktrees: readonly unknown[]) { + return { + exec: vi.fn().mockResolvedValue({ stdout: '', stderr: '' }), + listWorktrees: vi.fn().mockResolvedValue(worktrees), + removeWorktree: vi.fn().mockResolvedValue({}) + } +} + +function makeRemoteRepoStore( + executionHostId: ExecutionHostId, + extraRepoFields: Record = {}, + metaOverrides: Partial = {} +) { + const repo = { + ...store.getRepos()[0]!, + path: REMOTE_REPO_PATH, + executionHostId, + ...extraRepoFields + } + const metaById: Record = { + [TEST_WORKTREE_ID]: makeWorktreeMeta({ hostId: executionHostId, ...metaOverrides }) + } + const removeWorktreeMeta = vi.fn((worktreeId: string, hostId?: string) => { + if (!hostId || metaById[worktreeId]?.hostId === hostId) { + delete metaById[worktreeId] + } + }) + return { + repo, + metaById, + removeWorktreeMeta, + runtimeStore: { + ...store, + getRepos: () => [repo], + getRepo: (id: string) => (id === repo.id ? repo : undefined), + getAllWorktreeMeta: () => metaById, + getWorktreeMeta: (worktreeId: string) => metaById[worktreeId], + setWorktreeMeta: (worktreeId: string, meta: Partial) => { + metaById[worktreeId] = { ...(metaById[worktreeId] ?? makeWorktreeMeta()), ...meta } + return metaById[worktreeId] + }, + removeWorktreeMeta + } + } +} + +const REPO_ROOT_ENTRY = { + path: REMOTE_REPO_PATH, + head: 'main', + branch: 'main', + isBare: false, + isMainWorktree: true +} + +const REGISTERED_ENTRY = { + path: TEST_WORKTREE_PATH, + head: 'def456', + branch: 'feature/test', + isBare: false, + isMainWorktree: false +} + +describe('OrcaRuntimeService worktree removal execution host', () => { + beforeEach(() => { + vi.mocked(listWorktrees).mockClear() + vi.mocked(listWorktreesStrict).mockClear() + vi.mocked(removeWorktree).mockClear() + }) + + it('removes a migrated-spelling SSH row on its own host, never this machine', async () => { + // No `connectionId` at all: the row names its owner only as `executionHostId: 'ssh:target-a'`. + const { runtimeStore, removeWorktreeMeta, metaById } = makeRemoteRepoStore('ssh:target-a') + const provider = makeGitProvider([REPO_ROOT_ENTRY, REGISTERED_ENTRY]) + registerSshGitProvider('target-a', provider as never) + const runtime = createWorktreeRemovalRuntime(runtimeStore) + vi.spyOn(runtime, 'acquireFileWatcherRemoval').mockResolvedValue({ finish: vi.fn() }) + + try { + await runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'ssh:target-a') + + expect(provider.listWorktrees).toHaveBeenCalledWith(REMOTE_REPO_PATH) + expect(provider.removeWorktree).toHaveBeenCalledWith(TEST_WORKTREE_PATH, true) + expect(listWorktreesStrict).not.toHaveBeenCalled() + expect(removeWorktree).not.toHaveBeenCalled() + expect(removeWorktreeMeta).toHaveBeenCalledWith(TEST_WORKTREE_ID, 'ssh:target-a') + expect(metaById[TEST_WORKTREE_ID]).toBeUndefined() + } finally { + unregisterSshGitProvider('target-a') + } + }) + + it('runs the cleanup path for a migrated-spelling row entirely on its host', async () => { + // #18358 made an `executionHostId`-only row a removable cleanup candidate. Every step of the + // removal it starts — list, existence probe, delete, prune — must name the same host. + const { runtimeStore, removeWorktreeMeta } = makeRemoteRepoStore('ssh:target-a') + const provider = makeGitProvider([REPO_ROOT_ENTRY]) + const fsProvider = { stat: vi.fn(missingPath), deletePath: vi.fn() } + registerSshGitProvider('target-a', provider as never) + registerSshFilesystemProvider('target-a', fsProvider as never) + const runtime = createWorktreeRemovalRuntime(runtimeStore) + + try { + await expect( + runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'ssh:target-a') + ).resolves.toEqual({}) + + expect(provider.listWorktrees).toHaveBeenCalledWith(REMOTE_REPO_PATH) + expect(fsProvider.stat).toHaveBeenCalledWith(TEST_WORKTREE_PATH) + expect(listWorktreesStrict).not.toHaveBeenCalled() + expect(removeWorktree).not.toHaveBeenCalled() + expect(fsProvider.deletePath).not.toHaveBeenCalled() + expect(removeWorktreeMeta).toHaveBeenCalledWith(TEST_WORKTREE_ID, 'ssh:target-a') + } finally { + unregisterSshFilesystemProvider('target-a') + unregisterSshGitProvider('target-a') + } + }) + + it('keeps two simultaneously registered SSH hosts off each other paths', async () => { + const { runtimeStore } = makeRemoteRepoStore('ssh:target-b') + const providerA = makeGitProvider([REPO_ROOT_ENTRY, REGISTERED_ENTRY]) + const providerB = makeGitProvider([REPO_ROOT_ENTRY, REGISTERED_ENTRY]) + registerSshGitProvider('target-a', providerA as never) + registerSshGitProvider('target-b', providerB as never) + const runtime = createWorktreeRemovalRuntime(runtimeStore) + vi.spyOn(runtime, 'acquireFileWatcherRemoval').mockResolvedValue({ finish: vi.fn() }) + + try { + await runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'ssh:target-b') + + expect(providerB.removeWorktree).toHaveBeenCalledWith(TEST_WORKTREE_PATH, true) + expect(providerA.listWorktrees).not.toHaveBeenCalled() + expect(providerA.removeWorktree).not.toHaveBeenCalled() + } finally { + unregisterSshGitProvider('target-b') + unregisterSshGitProvider('target-a') + } + }) + + it('refuses an SSH row whose host is unreachable instead of deleting here', async () => { + const { runtimeStore, metaById } = makeRemoteRepoStore('ssh:target-a') + const runtime = createWorktreeRemovalRuntime(runtimeStore) + + await expect( + runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'ssh:target-a') + ).rejects.toThrow('Remote connection dropped') + + expect(listWorktreesStrict).not.toHaveBeenCalled() + expect(removeWorktree).not.toHaveBeenCalled() + expect(metaById[TEST_WORKTREE_ID]).toBeDefined() + }) + + it('refuses a runtime row with no nested SSH target rather than deleting locally', async () => { + const { runtimeStore, metaById } = makeRemoteRepoStore('runtime:env-1') + const runtime = createWorktreeRemovalRuntime(runtimeStore) + + await expect( + runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'runtime:env-1') + ).rejects.toThrow('not dispatched by this process') + + expect(listWorktreesStrict).not.toHaveBeenCalled() + expect(removeWorktree).not.toHaveBeenCalled() + expect(metaById[TEST_WORKTREE_ID]).toBeDefined() + }) + + it('refuses a runtime row whose nested SSH target is dialable in this namespace', async () => { + // `connectionId: 'target-a'` names a target inside env-1, not the one registered here. The raw + // read dialled this client's same-named host and removed a worktree on the wrong machine. + const { runtimeStore, metaById } = makeRemoteRepoStore('runtime:env-1', { + connectionId: 'target-a' + }) + const provider = makeGitProvider([REPO_ROOT_ENTRY, REGISTERED_ENTRY]) + registerSshGitProvider('target-a', provider as never) + const runtime = createWorktreeRemovalRuntime(runtimeStore) + + try { + await expect( + runtime.removeManagedWorktree(TEST_WORKTREE_ID, true, false, false, 'runtime:env-1') + ).rejects.toThrow('not dispatched by this process') + + // Selector resolution still lists through the raw field before removal begins — a read on + // the wrong namespace, tracked separately. Nothing destructive reaches it. + expect(provider.removeWorktree).not.toHaveBeenCalled() + expect(removeWorktree).not.toHaveBeenCalled() + expect(metaById[TEST_WORKTREE_ID]).toBeDefined() + } finally { + unregisterSshGitProvider('target-a') + } + }) +}) diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index aabfeb899c8..573c5ba979d 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -104,5 +104,6 @@ await import('./orca-runtime-tests/worktree-removal-and-reconciliation.spec') await import('./orca-runtime-tests/worktree-removal-and-reconciliation-part-02.spec') await import('./orca-runtime-tests/worktree-removal-and-reconciliation-part-03.spec') await import('./orca-runtime-tests/worktree-removal-and-reconciliation-part-04.spec') +await import('./orca-runtime-tests/worktree-removal-execution-host.spec') await import('./orca-runtime-tests/targeting-and-resilience.spec') await import('./orca-runtime-tests/worktree-scan-cache-ttl.spec') diff --git a/src/main/runtime/runtime-registered-remote-worktree-removal.ts b/src/main/runtime/runtime-registered-remote-worktree-removal.ts index 09c7ac362f4..6eec18d52c8 100644 --- a/src/main/runtime/runtime-registered-remote-worktree-removal.ts +++ b/src/main/runtime/runtime-registered-remote-worktree-removal.ts @@ -13,6 +13,8 @@ export async function removeRuntimeRegisteredRemoteWorktree(args: { removedPushTarget: GitPushTarget | undefined store: RuntimeStore provider: SshGitProvider + /** From the resolved removal route; `repo.connectionId!` answered null for an `ssh:`-only row. */ + connectionId: string force: boolean allowUnverifiedPtyStop: boolean deleteBranch: boolean @@ -28,8 +30,7 @@ export async function removeRuntimeRegisteredRemoteWorktree(args: { ) => RemoveWorktreeResult finishRemoval: (result: RemoveWorktreeResult) => void }): Promise { - const { repo, target, registeredWorktree, provider } = args - const connectionId = repo.connectionId! + const { repo, target, registeredWorktree, provider, connectionId } = args const removeOptions = !args.deleteBranch ? { deleteBranch: args.deleteBranch } : {} const gate = await args.acquireWatcherRemoval(registeredWorktree.path, connectionId) let rawResult: RemoveWorktreeResult | undefined diff --git a/src/main/runtime/runtime-unregistered-worktree-removal.ts b/src/main/runtime/runtime-unregistered-worktree-removal.ts index 3d21998ef55..01de8820c1c 100644 --- a/src/main/runtime/runtime-unregistered-worktree-removal.ts +++ b/src/main/runtime/runtime-unregistered-worktree-removal.ts @@ -2,8 +2,10 @@ import type { GitPushTarget, GitWorktreeInfo } from '../../shared/worktree/types import type { Repo } from '../../shared/repo-types' import type { WorktreeMeta } from '../../shared/worktree/meta-types' import type { LocalProjectWorktreeGitOptions } from '../project-runtime-git-options' -import type { IFilesystemProvider } from '../providers/types' -import type { SshGitProvider } from '../providers/ssh-git-provider' +import { + getWorktreeRemovalConnectionId, + type WorktreeRemovalRoute +} from '../worktree-removal-execution-host-route' import { getLocalWorktreePathAccess, removeLocalWorktreePath, @@ -39,8 +41,8 @@ export async function removeRuntimeUnregisteredWorktree(args: { removedPushTarget: GitPushTarget | undefined force: boolean allowUnverifiedPtyStop: boolean - provider: SshGitProvider | null - fsProvider: IFilesystemProvider | null + /** One resolved host for the whole removal, replacing the `provider` whose `null` also meant local. */ + route: WorktreeRemovalRoute localOptions: LocalProjectWorktreeGitOptions store: RuntimeStore acquireWatcherRemoval: (path: string, connectionId?: string) => Promise @@ -48,21 +50,23 @@ export async function removeRuntimeUnregisteredWorktree(args: { deleteHistory: () => Promise finishRemoval: () => void }): Promise<{}> { - const { repo, target, registeredWorktrees, removedMeta } = args + const { repo, target, registeredWorktrees, removedMeta, route } = args let canCleanOrphanedDirectory = false if (canCleanupUnregisteredOrcaWorktreeDirectory({ meta: removedMeta })) { - if (repo.connectionId) { - if (!args.fsProvider) { + if (route.kind === 'ssh') { + const fsProvider = route.fsProvider + if (!fsProvider) { throw new Error('SSH filesystem provider unavailable') } - if (!args.fsProvider.lstat) { + const lstat = fsProvider.lstat + if (!lstat) { throw new Error('SSH filesystem provider lstat unavailable') } canCleanOrphanedDirectory = await canSafelyRemoveOrphanedWorktreeDirectory( target.path, repo.path, - (path) => args.fsProvider!.lstat!(path), - (path) => args.fsProvider!.readFile(path) + (path) => lstat(path), + (path) => fsProvider.readFile(path) ) } else { const access = getLocalWorktreePathAccess(args.localOptions) @@ -85,7 +89,7 @@ export async function removeRuntimeUnregisteredWorktree(args: { args.finishRemoval() return {} } - if (!repo.connectionId) { + if (route.kind === 'local') { const access = getLocalWorktreePathAccess(args.localOptions) const runtimeWorktreePath = toLocalWorktreeRuntimePath(target.path, args.localOptions) if ( @@ -108,7 +112,7 @@ export async function removeRuntimeUnregisteredWorktree(args: { return {} } } - if (await isRuntimeWorktreePathMissing(repo, target.path, args.localOptions)) { + if (await isRuntimeWorktreePathMissing(route.hostId, target.path, args.localOptions)) { if (!args.force && !removedMeta) { throw new Error(UNREGISTERED_MISSING_WORKTREE_MESSAGE) } @@ -123,14 +127,19 @@ export async function removeRuntimeUnregisteredWorktree(args: { async function deleteUnregisteredDirectory( args: Parameters[0] ): Promise { - const connectionId = args.repo.connectionId?.trim() || undefined + const route = args.route + const connectionId = getWorktreeRemovalConnectionId(route) const gate = await args.acquireWatcherRemoval(args.target.path, connectionId) let completed = false try { await args.stopPtys(args.target.id, connectionId, args.allowUnverifiedPtyStop) - await (connectionId - ? args.fsProvider!.deletePath(args.target.path, true) - : removeLocalWorktreePath(args.target.path, args.localOptions)) + if (route.kind === 'local') { + await removeLocalWorktreePath(args.target.path, args.localOptions) + } else if (route.fsProvider) { + await route.fsProvider.deletePath(args.target.path, true) + } else { + throw new Error('SSH filesystem provider unavailable') + } completed = true } finally { await gate.finish(completed) @@ -142,9 +151,9 @@ async function deleteUnregisteredDirectory( async function cleanupPushTarget( args: Parameters[0] ): Promise { - await (args.repo.connectionId + await (args.route.kind === 'ssh' ? cleanupUnusedWorktreePushTargetRemoteSsh( - args.provider!, + args.route.provider, args.repo.path, args.target.id, args.removedPushTarget, diff --git a/src/main/runtime/runtime-worktree-filesystem.ts b/src/main/runtime/runtime-worktree-filesystem.ts index c2c5ac711c5..4f29919ca95 100644 --- a/src/main/runtime/runtime-worktree-filesystem.ts +++ b/src/main/runtime/runtime-worktree-filesystem.ts @@ -9,9 +9,12 @@ import { getLocalWorktreePathAccess, toLocalWorktreeRuntimePath } from '../local-worktree-filesystem' -import { getSshFilesystemProvider } from '../providers/ssh-filesystem-dispatch' +import { + ExecutionHostNotDispatchableError, + resolveFilesystemRouteForHost +} from '../providers/execution-host-provider-dispatch' import { isWorktreePathMissing } from '../worktree-removal-safety' -import { getRepoExecutionHostId } from '../../shared/execution-host' +import { getRepoExecutionHostId, type ExecutionHostId } from '../../shared/execution-host' import { getRepoOwnedWorktreeMeta } from '../worktree-metadata-ownership' import type { WorktreeMeta } from '../../shared/worktree/meta-types' import { @@ -22,19 +25,26 @@ import { import type { RuntimeStore } from './runtime-store-contract' import { gitStatusErrorMeansNotRepository } from './runtime-worktree-selection' +// Takes the resolved host rather than the repo: reading `repo.connectionId` answered "stat this on +// the client" for a row that names its owner only as `executionHostId: 'ssh:'`, which is +// the evidence a forced removal prunes registrations on. export async function isRuntimeWorktreePathMissing( - repo: Repo, + hostId: ExecutionHostId, worktreePath: string, localWorktreeGitOptions: { wslDistro?: string } = {} ): Promise { - if (!repo.connectionId) { + const route = resolveFilesystemRouteForHost(hostId) + if (route.kind === 'runtime') { + throw new ExecutionHostNotDispatchableError(route.hostId) + } + if (route.kind === 'local') { const access = getLocalWorktreePathAccess(localWorktreeGitOptions) return isWorktreePathMissing( toLocalWorktreeRuntimePath(worktreePath, localWorktreeGitOptions), access.statPath ) } - const fsProvider = getSshFilesystemProvider(repo.connectionId) + const fsProvider = route.provider return fsProvider ? isWorktreePathMissing(worktreePath, (path) => fsProvider.stat(path)) : false } diff --git a/src/main/worktree-removal-execution-host-route.test.ts b/src/main/worktree-removal-execution-host-route.test.ts new file mode 100644 index 00000000000..5fbbfbf9540 --- /dev/null +++ b/src/main/worktree-removal-execution-host-route.test.ts @@ -0,0 +1,100 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { + registerSshGitProvider, + SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE, + unregisterSshGitProvider +} from './providers/ssh-git-dispatch' +import { + registerSshFilesystemProvider, + unregisterSshFilesystemProvider +} from './providers/ssh-filesystem-dispatch' +import { ExecutionHostNotDispatchableError } from './providers/execution-host-provider-dispatch' +import { + getWorktreeRemovalConnectionId, + resolveWorktreeRemovalRoute +} from './worktree-removal-execution-host-route' + +const HOST_A = 'target-a' +const HOST_B = 'target-b' + +function gitProvider(name: string): never { + return { name } as never +} + +function fsProvider(name: string): never { + return { name } as never +} + +afterEach(() => { + unregisterSshGitProvider(HOST_A) + unregisterSshGitProvider(HOST_B) + unregisterSshFilesystemProvider(HOST_A) + unregisterSshFilesystemProvider(HOST_B) +}) + +describe('resolveWorktreeRemovalRoute', () => { + it('routes a local host to this machine with no connection', () => { + const route = resolveWorktreeRemovalRoute('local') + + expect(route).toEqual({ kind: 'local', hostId: 'local' }) + expect(getWorktreeRemovalConnectionId(route)).toBeUndefined() + }) + + it('keeps two simultaneously registered SSH hosts on their own providers', () => { + registerSshGitProvider(HOST_A, gitProvider('git-a')) + registerSshGitProvider(HOST_B, gitProvider('git-b')) + registerSshFilesystemProvider(HOST_A, fsProvider('fs-a')) + registerSshFilesystemProvider(HOST_B, fsProvider('fs-b')) + + const routeA = resolveWorktreeRemovalRoute('ssh:target-a') + const routeB = resolveWorktreeRemovalRoute('ssh:target-b') + + expect(routeA).toMatchObject({ + kind: 'ssh', + hostId: 'ssh:target-a', + connectionId: HOST_A, + provider: { name: 'git-a' }, + fsProvider: { name: 'fs-a' } + }) + expect(routeB).toMatchObject({ + kind: 'ssh', + hostId: 'ssh:target-b', + connectionId: HOST_B, + provider: { name: 'git-b' }, + fsProvider: { name: 'fs-b' } + }) + expect(getWorktreeRemovalConnectionId(routeA)).toBe(HOST_A) + expect(getWorktreeRemovalConnectionId(routeB)).toBe(HOST_B) + }) + + it('carries a null filesystem provider without falling back to the local one', () => { + registerSshGitProvider(HOST_A, gitProvider('git-a')) + + expect(resolveWorktreeRemovalRoute('ssh:target-a')).toMatchObject({ + kind: 'ssh', + fsProvider: null + }) + }) + + it('refuses an unreachable SSH host instead of answering local', () => { + expect(() => resolveWorktreeRemovalRoute('ssh:target-a')).toThrow( + SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE + ) + }) + + it('refuses a runtime host with no nested SSH target', () => { + expect(() => resolveWorktreeRemovalRoute('runtime:env-1')).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it('refuses a runtime host even when a same-named target is dialable here', () => { + // The nested target lives in the environment's namespace; a same-named local one is a + // different machine, and removing a worktree through it deletes the wrong checkout. + registerSshGitProvider(HOST_A, gitProvider('git-a')) + + expect(() => resolveWorktreeRemovalRoute('runtime:target-a')).toThrow( + ExecutionHostNotDispatchableError + ) + }) +}) diff --git a/src/main/worktree-removal-execution-host-route.ts b/src/main/worktree-removal-execution-host-route.ts new file mode 100644 index 00000000000..529eed71af5 --- /dev/null +++ b/src/main/worktree-removal-execution-host-route.ts @@ -0,0 +1,81 @@ +/** + * Which execution host a destructive worktree removal runs against. + * + * `removeManagedWorktree` resolved its host once, for metadata pruning + * (`cleanupHostId ?? getRepoExecutionHostId(repo)`), and then read raw `repo.connectionId` for + * every step that actually touches the filesystem: the `git worktree list` that decides whether the + * path is registered, the provider handed to the unregistered-removal branch, the + * registered-remote-vs-local fork, and the PTY/history teardown. One function, two spellings — + * so a row naming its owner only as `executionHostId: 'ssh:'` listed a *remote* checkout on + * this client, entered the unregistered branch with `provider: null`, and deleted a same-named + * local directory while metadata was pruned under `ssh:` (#11163). #18358 made that + * reachable by migrating the cleanup scan, so those rows now surface as removable candidates. + * + * Routing is now one answer for the whole removal, taken from the host the prune already used, so + * list, remove and prune cannot disagree. The ambiguous `provider: SshGitProvider | null` carrier + * is deleted from the callees rather than supplemented, which makes every remaining reader a + * compile error in the typed modules that do the destructive work. + * + * `runtime:` is not a variant. Its files live on that environment's own server and the SSH + * target on its repo row is that server's nested one, addressable only as the pair + * (environmentId, targetId); handing it to this client's SSH table would `git worktree remove` a + * same-named path on the wrong machine. It throws, matching `workspace-cleanup-git-route` and + * `runtime-git-command-target`. + * + * An `ssh:` host with no registered provider also throws. Loss of contact is never evidence that + * the checkout is local (docs/reference/ssh-execution-boundary.md); refusing leaves a remote + * worktree in place, while the incumbent fallback deleted a client-side path. + */ + +import type { ExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' +import { + ExecutionHostNotDispatchableError, + resolveFilesystemRouteForHost, + resolveGitRouteForHost +} from './providers/execution-host-provider-dispatch' +import { SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE } from './providers/ssh-git-dispatch' +import type { SshGitProvider } from './providers/ssh-git-provider' +import type { IFilesystemProvider } from './providers/types' + +export type WorktreeRemovalRoute = + | { kind: 'local'; hostId: typeof LOCAL_EXECUTION_HOST_ID } + | { + kind: 'ssh' + hostId: `ssh:${string}` + connectionId: string + provider: SshGitProvider + /** + * Still nullable: the incumbent read `getSshFilesystemProvider` (not `require…`) and the + * directory branches raise their own message when they need it. Narrowing it here would + * refuse removals that never touch the filesystem provider. + */ + fsProvider: IFilesystemProvider | null + } + +export function resolveWorktreeRemovalRoute(hostId: ExecutionHostId): WorktreeRemovalRoute { + const route = resolveGitRouteForHost(hostId) + switch (route.kind) { + case 'local': + return { kind: 'local', hostId: route.hostId } + case 'runtime': + throw new ExecutionHostNotDispatchableError(route.hostId) + case 'ssh': { + if (!route.provider) { + throw new Error(SSH_GIT_PROVIDER_UNAVAILABLE_MESSAGE) + } + const fsRoute = resolveFilesystemRouteForHost(hostId) + return { + kind: 'ssh', + hostId: route.hostId, + connectionId: route.connectionId, + provider: route.provider, + fsProvider: fsRoute.kind === 'ssh' ? fsRoute.provider : null + } + } + } +} + +/** The connection to teardown PTYs, watchers and history against — `undefined` on a local host. */ +export function getWorktreeRemovalConnectionId(route: WorktreeRemovalRoute): string | undefined { + return route.kind === 'ssh' ? route.connectionId : undefined +} From 5d8532f6d3eb2afdc2fb7dc7ec910e15535f189b Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:27:13 -0700 Subject: [PATCH 207/398] fix(worktrees): resolve the execution host at both worktree-create entry points (#18545) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two entry points create the same workspace and disagreed about how to read its host. `orca-runtime-create-managed-worktree.ts:63` resolved through `getRepoSshConnectionId` and then normalized the row; the `worktrees:create` IPC handler branched on raw `repo.connectionId` (`register-worktree-create-handlers.ts:66-69`). So a repo naming its owner only as `executionHostId: 'ssh:'` created remotely through the runtime and ran `git worktree add` on the client against a remote path through IPC (#11163). Same repo, two entry points, different answers. Both now take one route, resolved through the existing layer (`getRepoExecutionHostId` -> #18296's `resolveGitRouteForHost`). No new resolver. The row normalization on the `ssh` variant is kept, and it is a **workaround, not the pattern**. `createRemoteWorktree` and its callees re-read `repo.connectionId!` at five depths in `ipc/worktree-remote.ts` (1627, 1847, 1848, 1865, 2029), so the resolved connection has to reach them through the field they already read. It travels only as far as that object does — anything downstream that re-reads the row from the store still sees the unnormalized one, and it cannot express the `runtime:` refusal on its own. Proper fix, deliberately not done here: give that pipeline an explicit connection parameter and delete `repo.connectionId!` from it so every reader becomes a compile error, the technique #18307/#18325 used. That is a change inside a 2800-line module plus its callers, and it wants its own PR. Three answers that used to collapse into one, now distinct at both entry points: - `executionHostId: 'ssh:*'` with no `connectionId` -> that SSH host (IPC used to create locally); - `executionHostId: 'local'` with a surviving `connectionId` -> local, since a local row cannot nest an SSH namespace. This is what `getRepoSshConnectionId` and therefore the runtime sibling already answered; IPC used to go remote; - `runtime:` -> refused. Its worktree is created by that environment's own server and the SSH target on its repo row is that server's nested one, addressable only as (environmentId, targetId). The renderer already routes runtime-environment creates over `worktree.create` RPC rather than this IPC channel, so reaching either entry point with one is a routing mistake. Matches `workspace-cleanup-git-route` and `runtime-git-command-target`. Folder-workspace creation is untouched on both sides: it is a registration, not a filesystem create, so the route is resolved after that branch on the IPC side, and on the runtime side only the agent trust write consumes it — where a `runtime:` host now yields `null` instead of the nested target, so the write stops going to a same-named target in this client's table. No wire or persistence change: the normalized row is a local value passed to the create pipeline, never stored, and `CreateWorktreeResult` is untouched. --- ...rees-create-execution-host-routing.test.ts | 229 ++++++++++++++++++ .../register-worktree-create-handlers.ts | 19 +- .../orca-runtime-create-managed-worktree.ts | 44 ++-- .../worktree-create-execution-host.spec.ts | 61 +++++ src/main/runtime/orca-runtime.test.ts | 1 + ...rktree-create-execution-host-route.test.ts | 106 ++++++++ .../worktree-create-execution-host-route.ts | 69 ++++++ 7 files changed, 505 insertions(+), 24 deletions(-) create mode 100644 src/main/ipc/worktrees-create-execution-host-routing.test.ts create mode 100644 src/main/runtime/orca-runtime-tests/worktree-create-execution-host.spec.ts create mode 100644 src/main/worktree-create-execution-host-route.test.ts create mode 100644 src/main/worktree-create-execution-host-route.ts diff --git a/src/main/ipc/worktrees-create-execution-host-routing.test.ts b/src/main/ipc/worktrees-create-execution-host-routing.test.ts new file mode 100644 index 00000000000..36f1e816ea6 --- /dev/null +++ b/src/main/ipc/worktrees-create-execution-host-routing.test.ts @@ -0,0 +1,229 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + addWorktreeMock, + getActiveMultiplexerMock, + getSshGitProviderMock, + listWorktreesMock +} from './worktrees-test-module-mocks' +import { handlers, setupWorktreeHandlers, store } from './worktrees-test-harness' + +vi.mock('electron', async () => + (await import('./worktrees-test-module-mocks')).electronModuleMock() +) +vi.mock('../git/worktree', async () => + (await import('./worktrees-test-module-mocks')).gitWorktreeModuleMock() +) +vi.mock('../git/runner', async () => + (await import('./worktrees-test-module-mocks')).gitRunnerModuleMock() +) +vi.mock('../git/repo', async () => + (await import('./worktrees-test-module-mocks')).gitRepoModuleMock() +) +vi.mock('../git/git-username', async (importOriginal) => ({ + ...(await importOriginal>()), + resolveLocalGitUsername: (await import('./worktrees-test-module-mocks')) + .resolveLocalGitUsernameMock +})) +vi.mock('../github/client', async () => + (await import('./worktrees-test-module-mocks')).githubClientModuleMock() +) +vi.mock('../source-control/hosted-review', async () => + (await import('./worktrees-test-module-mocks')).hostedReviewModuleMock() +) +vi.mock('../providers/ssh-git-dispatch', async () => + (await import('./worktrees-test-module-mocks')).sshGitDispatchModuleMock() +) +vi.mock('../providers/ssh-filesystem-dispatch', async () => + (await import('./worktrees-test-module-mocks')).sshFilesystemDispatchModuleMock() +) +vi.mock('./worktree-symlinks', async () => + (await import('./worktrees-test-module-mocks')).worktreeSymlinksModuleMock() +) +vi.mock('./ssh', async () => (await import('./worktrees-test-module-mocks')).sshModuleMock()) +vi.mock('../ssh/ssh-target-registry', async () => + (await import('./worktrees-test-module-mocks')).sshTargetRegistryModuleMock() +) +vi.mock('../hooks', async () => (await import('./worktrees-test-module-mocks')).hooksModuleMock()) +vi.mock('../setup-runner-script-text', async (importOriginal) => + (await import('./worktrees-test-module-mocks')).setupRunnerScriptTextModuleMock( + (await importOriginal()) as Record + ) +) +vi.mock('../worktree-runner-script', async (importOriginal) => + (await import('./worktrees-test-module-mocks')).worktreeRunnerScriptModuleMock( + (await importOriginal()) as Record + ) +) +vi.mock('../effective-hook-config', async (importOriginal) => + (await import('./worktrees-test-module-mocks')).effectiveHookConfigModuleMock( + (await importOriginal()) as Record + ) +) +vi.mock('../setup-hook-env-vars', async (importOriginal) => + (await import('./worktrees-test-module-mocks')).setupHookEnvVarsModuleMock( + (await importOriginal()) as Record + ) +) +vi.mock('./worktree-logic', async (importOriginal) => + (await import('./worktrees-test-module-mocks')).worktreeLogicModuleMock( + (await importOriginal()) as Record + ) +) +vi.mock('../terminal-history-deletion', async () => + (await import('./worktrees-test-module-mocks')).terminalHistoryDeletionModuleMock() +) +vi.mock('../ports/advertised-url-watcher', async () => + (await import('./worktrees-test-module-mocks')).advertisedUrlWatcherModuleMock() +) +vi.mock('../workspace-cleanup-scan-snapshot', async () => + (await import('./worktrees-test-module-mocks')).workspaceCleanupScanSnapshotModuleMock() +) +vi.mock('../workspace-space-analysis-snapshot', async () => + (await import('./worktrees-test-module-mocks')).workspaceSpaceAnalysisSnapshotModuleMock() +) +vi.mock('../workspace-cleanup-removal-snapshot-prune', async () => + (await import('./worktrees-test-module-mocks')).workspaceCleanupRemovalSnapshotPruneModuleMock() +) +vi.mock('../runtime/worktree-teardown', async () => + (await import('./worktrees-test-module-mocks')).worktreeTeardownModuleMock() +) +vi.mock('./pty', async () => (await import('./worktrees-test-module-mocks')).ptyModuleMock()) + +const REMOTE_REPO_PATH = '/remote/repo' + +function makeRepo(fields: Record) { + return { + id: 'repo-1', + path: REMOTE_REPO_PATH, + displayName: 'repo', + badgeColor: '#000', + addedAt: 0, + worktreeBaseRef: 'origin/main', + ...fields + } +} + +function makeProvider(worktreePath: string) { + return { + exec: vi.fn().mockImplementation(async (args: string[]) => { + if (args[0] === 'remote') { + return { stdout: 'origin\n', stderr: '' } + } + if (args[0] === 'show-ref') { + // A hit here reads as "branch already exists"; the create loop would then rename. + throw Object.assign(new Error('missing exact ref'), { code: 1 }) + } + return { stdout: '', stderr: '' } + }), + fetchRemoteTrackingRef: vi.fn().mockResolvedValue(undefined), + addWorktree: vi.fn().mockResolvedValue(undefined), + listWorktrees: vi.fn().mockResolvedValue([ + { + path: worktreePath, + head: 'abc123', + branch: 'refs/heads/wt', + isBare: false, + isMainWorktree: false + } + ]) + } +} + +function useRepo(repo: ReturnType): void { + store.getRepos.mockReturnValue([repo]) + store.getRepo.mockReturnValue(repo) + store.setWorktreeMeta.mockImplementation((_worktreeId: string, meta: unknown) => meta) + getActiveMultiplexerMock.mockReturnValue({ + request: vi.fn().mockResolvedValue(undefined), + notify: vi.fn() + }) +} + +describe('worktrees:create execution host routing', () => { + beforeEach(() => { + setupWorktreeHandlers() + }) + + it('creates on the SSH host for a row that names it only as executionHostId', async () => { + // No `connectionId`: the raw read answered "local" and ran `git worktree add` on the client + // against `/remote/repo`. The runtime sibling already resolved this row remotely. + useRepo(makeRepo({ executionHostId: 'ssh:target-a' })) + const provider = makeProvider('/remote/repo-wt') + getSshGitProviderMock.mockImplementation((connectionId: string) => + connectionId === 'target-a' ? provider : undefined + ) + + await handlers['worktrees:create'](null, { repoId: 'repo-1', name: 'wt' }) + + expect(provider.addWorktree).toHaveBeenCalledTimes(1) + expect(addWorktreeMock).not.toHaveBeenCalled() + }) + + it('keeps two simultaneously registered SSH hosts apart', async () => { + useRepo(makeRepo({ executionHostId: 'ssh:target-b' })) + const providerA = makeProvider('/remote/repo-wt-a') + const providerB = makeProvider('/remote/repo-wt-b') + getSshGitProviderMock.mockImplementation((connectionId: string) => + connectionId === 'target-a' ? providerA : connectionId === 'target-b' ? providerB : undefined + ) + + await handlers['worktrees:create'](null, { repoId: 'repo-1', name: 'wt' }) + + expect(providerB.addWorktree).toHaveBeenCalledTimes(1) + expect(providerA.addWorktree).not.toHaveBeenCalled() + expect(addWorktreeMock).not.toHaveBeenCalled() + }) + + it('refuses a runtime row with no nested SSH target instead of creating locally', async () => { + useRepo(makeRepo({ executionHostId: 'runtime:env-1' })) + + await expect( + handlers['worktrees:create'](null, { repoId: 'repo-1', name: 'wt' }) + ).rejects.toThrow('not dispatched by this process') + + expect(addWorktreeMock).not.toHaveBeenCalled() + }) + + it('refuses a runtime row whose nested SSH target is dialable in this namespace', async () => { + // `target-a` names a target inside env-1. A same-named one registered here is another machine, + // so creating through it lands the checkout on the wrong host. + useRepo(makeRepo({ executionHostId: 'runtime:env-1', connectionId: 'target-a' })) + const provider = makeProvider('/remote/repo-wt') + getSshGitProviderMock.mockImplementation((connectionId: string) => + connectionId === 'target-a' ? provider : undefined + ) + + await expect( + handlers['worktrees:create'](null, { repoId: 'repo-1', name: 'wt' }) + ).rejects.toThrow('not dispatched by this process') + + expect(provider.addWorktree).not.toHaveBeenCalled() + expect(addWorktreeMock).not.toHaveBeenCalled() + }) + + it('answers local for a row that declares itself local while carrying a connection', async () => { + // A contradictory row: `getRepoSshConnectionId` lets `local` win, and the runtime sibling has + // always read it that way. The raw field sent it remote, so the two entry points disagreed. + useRepo( + makeRepo({ path: '/workspace/repo', executionHostId: 'local', connectionId: 'target-a' }) + ) + listWorktreesMock.mockResolvedValue([ + { + path: '/workspace/wt', + head: 'abc123', + branch: 'wt', + isBare: false, + isMainWorktree: false + } + ]) + const provider = makeProvider('/remote/repo-wt') + getSshGitProviderMock.mockImplementation((connectionId: string) => + connectionId === 'target-a' ? provider : undefined + ) + + await handlers['worktrees:create'](null, { repoId: 'repo-1', name: 'wt' }) + + expect(addWorktreeMock).toHaveBeenCalledTimes(1) + expect(provider.addWorktree).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts b/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts index f371529963c..1f94598208b 100644 --- a/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts +++ b/src/main/ipc/worktrees/create/register-worktree-create-handlers.ts @@ -29,6 +29,7 @@ import { normalizeLinkedWorkItemFields } from '../ipc-context-schemas' import type { CreateWorktreeArgsWithSystemProvenance } from '../ipc-context-schemas' import { createFolderWorkspace } from './folder-workspace-creation' import { findExactRepoOwner, isCapturedRepoCurrent } from '../listing/worktree-host-ownership' +import { requireWorktreeCreateRoute } from '../../../worktree-create-execution-host-route' import type { WorktreeIpcContext } from '../worktree-ipc-context' export function registerWorktreeCreateHandlers(context: WorktreeIpcContext): void { @@ -62,11 +63,19 @@ export function registerWorktreeCreateHandlers(context: WorktreeIpcContext): voi let result: CreateWorktreeResult try { // Why: wrap only the helpers; the pre-validation throws above are IPC-shape bugs, not the git/filesystem failures the funnel tracks. - result = isFolderRepo(repo) - ? createFolderWorkspace(createArgs, repo, store) - : repo.connectionId - ? await createRemoteWorktree(createArgs, repo, store, mainWindow) - : await createLocalWorktree(createArgs, repo, store, mainWindow, runtime) + if (isFolderRepo(repo)) { + // A folder workspace is a registration, not a filesystem create, so it is host-agnostic. + result = createFolderWorkspace(createArgs, repo, store) + } else { + // Resolve the host rather than reading the raw field: an `executionHostId: 'ssh:*'`-only + // row read as local here and ran `git worktree add` on the client against a remote path, + // while the runtime sibling on the same repo already resolved. + const createRoute = requireWorktreeCreateRoute(repo) + result = + createRoute.kind === 'ssh' + ? await createRemoteWorktree(createArgs, createRoute.repo, store, mainWindow) + : await createLocalWorktree(createArgs, repo, store, mainWindow, runtime) + } } catch (error) { releaseAutomationWorkspaceProvenanceRequest(args.automationProvenanceRequest) track('workspace_create_failed', { diff --git a/src/main/runtime/orca-runtime-create-managed-worktree.ts b/src/main/runtime/orca-runtime-create-managed-worktree.ts index f9116f73405..f17a2285b83 100644 --- a/src/main/runtime/orca-runtime-create-managed-worktree.ts +++ b/src/main/runtime/orca-runtime-create-managed-worktree.ts @@ -4,7 +4,8 @@ import type { RuntimeManagedWorktreeCreateArgs } from './runtime-managed-worktre import type { CreateWorktreeResult } from '../../shared/worktree/create-types' import { isTuiAgentEnabled } from '../../shared/tui-agent-selection' import { isFolderRepo } from '../../shared/repo-kind' -import { getRepoSshConnectionId } from '../../shared/execution-host' +import { resolveWorktreeCreateRoute } from '../worktree-create-execution-host-route' +import { ExecutionHostNotDispatchableError } from '../providers/execution-host-provider-dispatch' import { createRuntimeFolderWorktree } from './runtime-folder-worktree-create' import { createRuntimeLocalManagedWorktree } from './runtime-local-worktree-create' import { prepareRuntimeLocalWorktreeSetup } from './runtime-local-worktree-setup' @@ -57,10 +58,14 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork draftStartup?.agent ?? (requestedAgentEnabled ? requestedAgent : undefined)) const effectiveDraftPaste = args.startupDraftPaste ?? draftStartup?.draftPaste - // Resolve the execution host once: SSH ownership has two spellings, and reading the raw - // `connectionId` field routes an `executionHostId: 'ssh:*'`-only repo down the local path, - // which runs `git worktree add` on the client against a remote path. - const sshConnectionId = getRepoSshConnectionId(repo) + // Resolve the execution host once, shared with the `worktrees:create` IPC entry point so the + // two cannot answer differently for the same repo. Reading the raw `connectionId` field routes + // an `executionHostId: 'ssh:*'`-only repo down the local path, which runs `git worktree add` on + // the client against a remote path. + const createRoute = resolveWorktreeCreateRoute(repo) + // `null` on a `runtime:` host is deliberate: its nested target is addressable only inside that + // environment, so the trust write must not go to a same-named target in this client's table. + const sshConnectionId = createRoute.kind === 'ssh' ? createRoute.connectionId : null if (isFolderRepo(repo)) { // A folder workspace is a registration, not a filesystem create, so it is host-agnostic — // except for the agent trust write, which must land on the host that will run the agent. @@ -97,20 +102,21 @@ export class OrcaRuntimeWithCreateManagedWorktree extends OrcaRuntimeWithGetWork const lineageInput = args.lineage || args.comment ? { ...args.lineage, comment: args.comment } : undefined const lineageResolution = await this.resolveLineageForWorktreeCreate(lineageInput) - if (sshConnectionId) { - // Why normalize the row: the remote-create pipeline reads `repo.connectionId!` at every - // depth, so hand it the connection the resolved host actually names. - const result = await this.createManagedRemoteWorktree( - { ...repo, connectionId: sshConnectionId }, - { - ...args, - activate: args.activate, - ...(effectiveStartup ? { startup: effectiveStartup } : {}), - ...(effectiveStartupFollowup ? { startupFollowup: effectiveStartupFollowup } : {}), - ...(effectiveCreatedWithAgent ? { createdWithAgent: effectiveCreatedWithAgent } : {}), - ...(effectiveDraftPaste ? { startupDraftPaste: effectiveDraftPaste } : {}) - } - ) + if (createRoute.kind === 'runtime') { + throw new ExecutionHostNotDispatchableError(createRoute.hostId) + } + if (createRoute.kind === 'ssh') { + // `createRoute.repo` carries the resolved connection in `connectionId`, because the + // remote-create pipeline still reads `repo.connectionId!` at every depth. See the workaround + // note in worktree-create-execution-host-route.ts. + const result = await this.createManagedRemoteWorktree(createRoute.repo, { + ...args, + activate: args.activate, + ...(effectiveStartup ? { startup: effectiveStartup } : {}), + ...(effectiveStartupFollowup ? { startupFollowup: effectiveStartupFollowup } : {}), + ...(effectiveCreatedWithAgent ? { createdWithAgent: effectiveCreatedWithAgent } : {}), + ...(effectiveDraftPaste ? { startupDraftPaste: effectiveDraftPaste } : {}) + }) const recordedLineage = this.recordCreatedWorktreeLineage(result.worktree, lineageResolution) this.emitWorktreeLifecycle({ kind: 'created', diff --git a/src/main/runtime/orca-runtime-tests/worktree-create-execution-host.spec.ts b/src/main/runtime/orca-runtime-tests/worktree-create-execution-host.spec.ts new file mode 100644 index 00000000000..981e9860463 --- /dev/null +++ b/src/main/runtime/orca-runtime-tests/worktree-create-execution-host.spec.ts @@ -0,0 +1,61 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + OrcaRuntimeService, + addWorktree, + registerSshGitProvider, + unregisterSshGitProvider +} from '../orca-runtime-test-mocks.spec' +import { store } from '../orca-runtime-test-fixtures.spec' + +const RUNTIME_REPO_PATH = '/remote/repo' + +function makeRuntimeHostedStore(extraRepoFields: Record = {}) { + const repo = { + ...store.getRepos()[0]!, + path: RUNTIME_REPO_PATH, + executionHostId: 'runtime:env-1', + ...extraRepoFields + } + return { + ...store, + getRepos: () => [repo], + getRepo: (id: string) => (id === repo.id ? repo : undefined) + } +} + +describe('OrcaRuntimeService worktree create execution host', () => { + beforeEach(() => { + vi.mocked(addWorktree).mockClear() + }) + + it('refuses to create for a runtime-hosted repo with no nested SSH target', async () => { + const runtime = new OrcaRuntimeService(makeRuntimeHostedStore() as never) + + await expect( + runtime.createManagedWorktree({ repoSelector: 'id:repo-1', name: 'wt' }) + ).rejects.toThrow('not dispatched by this process') + + expect(addWorktree).not.toHaveBeenCalled() + }) + + it('refuses a runtime-hosted repo whose nested SSH target is dialable in this namespace', async () => { + // `target-a` names a target inside env-1. The same-named one registered here is another + // machine, so creating through it would put the checkout on the wrong host. + const provider = { exec: vi.fn(), addWorktree: vi.fn(), listWorktrees: vi.fn() } + registerSshGitProvider('target-a', provider as never) + const runtime = new OrcaRuntimeService( + makeRuntimeHostedStore({ connectionId: 'target-a' }) as never + ) + + try { + await expect( + runtime.createManagedWorktree({ repoSelector: 'id:repo-1', name: 'wt' }) + ).rejects.toThrow('not dispatched by this process') + + expect(provider.addWorktree).not.toHaveBeenCalled() + expect(addWorktree).not.toHaveBeenCalled() + } finally { + unregisterSshGitProvider('target-a') + } + }) +}) diff --git a/src/main/runtime/orca-runtime.test.ts b/src/main/runtime/orca-runtime.test.ts index 573c5ba979d..9d223233ef2 100644 --- a/src/main/runtime/orca-runtime.test.ts +++ b/src/main/runtime/orca-runtime.test.ts @@ -18,6 +18,7 @@ await import('./orca-runtime-tests/terminal-listing.spec') await import('./orca-runtime-tests/worktree-selector-resolution.spec') await import('./orca-runtime-tests/local-worktree-creation.spec') await import('./orca-runtime-tests/local-worktree-creation-part-02.spec') +await import('./orca-runtime-tests/worktree-create-execution-host.spec') await import('./orca-runtime-tests/ssh-worktree-lifecycle.spec') await import('./orca-runtime-tests/ssh-worktree-lifecycle-part-02.spec') await import('./orca-runtime-tests/ssh-worktree-lifecycle-part-03.spec') diff --git a/src/main/worktree-create-execution-host-route.test.ts b/src/main/worktree-create-execution-host-route.test.ts new file mode 100644 index 00000000000..ec4ecd2899e --- /dev/null +++ b/src/main/worktree-create-execution-host-route.test.ts @@ -0,0 +1,106 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { registerSshGitProvider, unregisterSshGitProvider } from './providers/ssh-git-dispatch' +import { ExecutionHostNotDispatchableError } from './providers/execution-host-provider-dispatch' +import type { Repo } from '../shared/repo-types' +import { + requireWorktreeCreateRoute, + resolveWorktreeCreateRoute +} from './worktree-create-execution-host-route' + +const HOST_A = 'target-a' +const HOST_B = 'target-b' + +function repoRow(fields: Partial): Repo { + return { + id: 'repo-1', + path: '/remote/repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0, + ...fields + } as Repo +} + +afterEach(() => { + unregisterSshGitProvider(HOST_A) + unregisterSshGitProvider(HOST_B) +}) + +describe('resolveWorktreeCreateRoute', () => { + it('routes a row that names its host only as executionHostId to that SSH target', () => { + registerSshGitProvider(HOST_A, { name: 'git-a' } as never) + + expect(resolveWorktreeCreateRoute(repoRow({ executionHostId: 'ssh:target-a' }))).toMatchObject({ + kind: 'ssh', + hostId: 'ssh:target-a', + connectionId: HOST_A, + repo: { connectionId: HOST_A } + }) + }) + + it('normalizes the row for a legacy connectionId-only repo without changing its answer', () => { + expect(resolveWorktreeCreateRoute(repoRow({ connectionId: HOST_A }))).toMatchObject({ + kind: 'ssh', + connectionId: HOST_A, + repo: { connectionId: HOST_A } + }) + }) + + it('keeps two simultaneously registered SSH hosts on their own connections', () => { + registerSshGitProvider(HOST_A, { name: 'git-a' } as never) + registerSshGitProvider(HOST_B, { name: 'git-b' } as never) + + expect(resolveWorktreeCreateRoute(repoRow({ executionHostId: 'ssh:target-a' }))).toMatchObject({ + connectionId: HOST_A, + repo: { connectionId: HOST_A } + }) + expect(resolveWorktreeCreateRoute(repoRow({ executionHostId: 'ssh:target-b' }))).toMatchObject({ + connectionId: HOST_B, + repo: { connectionId: HOST_B } + }) + }) + + it('lets an explicit local host win over a surviving connectionId', () => { + // A contradictory row. `getRepoExecutionHostId` answers `local`, which is what the runtime + // create sibling has always done; the raw read sent it remote. + expect( + resolveWorktreeCreateRoute(repoRow({ executionHostId: 'local', connectionId: HOST_A })) + ).toEqual({ kind: 'local', hostId: 'local' }) + }) + + it('answers runtime for a runtime row with no nested SSH target', () => { + expect(resolveWorktreeCreateRoute(repoRow({ executionHostId: 'runtime:env-1' }))).toEqual({ + kind: 'runtime', + hostId: 'runtime:env-1', + environmentId: 'env-1' + }) + }) + + it('answers runtime for a runtime row whose nested target is dialable here', () => { + registerSshGitProvider(HOST_A, { name: 'git-a' } as never) + + expect( + resolveWorktreeCreateRoute( + repoRow({ executionHostId: 'runtime:env-1', connectionId: HOST_A }) + ) + ).toMatchObject({ kind: 'runtime', environmentId: 'env-1' }) + }) +}) + +describe('requireWorktreeCreateRoute', () => { + it('refuses a runtime host rather than creating through this client', () => { + expect(() => requireWorktreeCreateRoute(repoRow({ executionHostId: 'runtime:env-1' }))).toThrow( + ExecutionHostNotDispatchableError + ) + }) + + it('passes local and SSH hosts through unchanged', () => { + registerSshGitProvider(HOST_A, { name: 'git-a' } as never) + + expect(requireWorktreeCreateRoute(repoRow({}))).toEqual({ kind: 'local', hostId: 'local' }) + expect(requireWorktreeCreateRoute(repoRow({ executionHostId: 'ssh:target-a' }))).toMatchObject({ + kind: 'ssh', + connectionId: HOST_A + }) + }) +}) diff --git a/src/main/worktree-create-execution-host-route.ts b/src/main/worktree-create-execution-host-route.ts new file mode 100644 index 00000000000..2831a2fcbeb --- /dev/null +++ b/src/main/worktree-create-execution-host-route.ts @@ -0,0 +1,69 @@ +/** + * Which execution host a worktree create runs on. + * + * Two entry points create the same workspace and disagreed about how to read its host. The runtime + * path resolved (`orca-runtime-create-managed-worktree.ts`) and then normalized the row; the IPC + * handler branched on raw `repo.connectionId`, so a row naming its owner only as + * `executionHostId: 'ssh:'` ran `git worktree add` on the client against a remote path + * (#11163). Same repo, two entry points, two answers. + * + * Both now take this one route. + * + * The `repo` on the `ssh` variant is a normalization, and it is a workaround rather than the + * pattern: `createRemoteWorktree` and its callees re-read `repo.connectionId!` at five depths + * (`ipc/worktree-remote.ts`), so the resolved connection has to be handed to them through the field + * they already read. It travels only as far as this object does — anything downstream that re-reads + * the row from the store still sees the unnormalized one. The real fix is to give that pipeline an + * explicit connection parameter and delete `repo.connectionId!` from it, which is a separate change. + */ + +import { getRepoExecutionHostId, type LOCAL_EXECUTION_HOST_ID } from '../shared/execution-host' +import type { Repo } from '../shared/repo-types' +import { + ExecutionHostNotDispatchableError, + resolveGitRouteForHost +} from './providers/execution-host-provider-dispatch' + +export type WorktreeCreateRoute = + | { kind: 'local'; hostId: typeof LOCAL_EXECUTION_HOST_ID } + | { + kind: 'ssh' + hostId: `ssh:${string}` + connectionId: string + /** The row with `connectionId` set to the resolved target; see the workaround note above. */ + repo: Repo + } + | { kind: 'runtime'; hostId: `runtime:${string}`; environmentId: string } + +export function resolveWorktreeCreateRoute(repo: Repo): WorktreeCreateRoute { + const route = resolveGitRouteForHost(getRepoExecutionHostId(repo)) + switch (route.kind) { + case 'local': + return { kind: 'local', hostId: route.hostId } + case 'ssh': + return { + kind: 'ssh', + hostId: route.hostId, + connectionId: route.connectionId, + repo: { ...repo, connectionId: route.connectionId } + } + case 'runtime': + return { kind: 'runtime', hostId: route.hostId, environmentId: route.environmentId } + } +} + +/** + * For the two create forks that put files on a host. `runtime:` is not one of them: the + * environment's own server creates the worktree, and the SSH target on its repo row is that + * server's nested one, addressable only as (environmentId, targetId). Creating through this + * client's SSH table would `git worktree add` on a same-named target on the wrong machine. + */ +export function requireWorktreeCreateRoute( + repo: Repo +): Exclude { + const route = resolveWorktreeCreateRoute(repo) + if (route.kind === 'runtime') { + throw new ExecutionHostNotDispatchableError(route.hostId) + } + return route +} From e85ebb0086e7b867e64ada9feae3aa99d9402e1c Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:28:45 -0700 Subject: [PATCH 208/398] feat(native-chat): restore the terminal/chat switcher for bridge chat only (#18532) * feat(native-chat): restore the terminal/chat switcher for bridge chat only #16729 removed every user-facing terminal<->chat switching affordance as a side effect of the structured Codex restructure ("renderer switching affordances and their dead leftovers"). That was right for structured Codex sessions, which render their own transcript with no live TUI underneath, but it also took the switcher away from bridge native chat, which still reads the terminal and has one to return to. Restore all four surfaces, each gated so structured sessions keep the removal: - pane header chat/terminal button (TerminalPaneHeaderOverlay) - pane context-menu "Switch to chat/terminal view" (TerminalContextMenu) - tab context-menu equivalent (SortableTabContextMenu) - the keyboard chord, whose hook had survived uncalled since #16729 Gating is one rule in one place: `canSwitchNativeChatView` refuses whenever a `structuredSessionId` is present, over the existing `canToggleNativeChat` eligibility. Standalone structured tabs are already excluded by the `contentType === 'terminal'` check; the new guard covers a terminal tab that adopted a structured session. The shortcut hook applies the same rule. The state plumbing (`viewMode`, `setTabViewMode`, `toggleTabViewMode`, host mirroring, `native_chat_toggled` telemetry) was never removed, so this rewires live actions rather than reintroducing logic. SortableTab.tsx sat exactly at its 400-line cap, so its inline-rename state and the window rename-request listener move to `use-sortable-tab-rename.ts` to make room. No behavior change; its rename tests pass unmodified. Two ratchets move for real, explained in place: - store-subscription budget: per-pane listeners stay pinned at 17 (the folded action bundle is still one listener); only the counterfactual pre-fold constant grows 48 -> 49 for the added `toggleTabViewMode` key. - hook-order parity: 204 -> 208 hooks for the four added `useCallback`s, useMemo count unchanged at 8. * fix(native-chat): restore bridge chat escape hatch * test: update pane agent identity inventory --------- Co-authored-by: Merge Sim --- .../native-chat/NativeChatResolvedView.tsx | 1 + .../native-chat-availability.test.ts | 36 +++++- .../native-chat/native-chat-availability.ts | 13 +++ .../use-native-chat-context-menu.test.tsx | 110 ++++++++++++++++++ .../use-native-chat-context-menu.tsx | 22 +++- .../use-native-chat-toggle-shortcut.ts | 15 ++- .../src/components/tab-bar/SortableTab.tsx | 98 +++++----------- .../tab-bar/SortableTabContextMenu.tsx | 47 +++++++- .../native-chat-tab-agent-evidence.test.ts | 38 ++++++ .../tab-bar/native-chat-tab-agent-evidence.ts | 18 +++ .../tab-bar/tab-bar-item-surface.tsx | 37 +++++- .../tab-bar/use-sortable-tab-rename.ts | 94 +++++++++++++++ .../TerminalContextMenu.test.tsx | 3 + .../terminal-pane/TerminalContextMenu.tsx | 29 ++++- .../TerminalPaneHeaderOverlay.tsx | 62 +++++++++- .../TerminalPaneOverlayLayer.tsx | 3 + .../terminal-pane/TerminalPaneSurface.tsx | 12 ++ .../terminal-pane-hook-order-parity.test.ts | 7 +- ...al-pane-store-subscription-budget.test.tsx | 8 +- .../use-terminal-pane-chat-state.ts | 50 +++++++- .../use-terminal-pane-projection.ts | 23 +++- .../use-terminal-pane-store-actions.ts | 2 + .../pane-agent-identity-inventory.test.ts | 2 +- 23 files changed, 637 insertions(+), 93 deletions(-) create mode 100644 src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx create mode 100644 src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.test.ts create mode 100644 src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.ts create mode 100644 src/renderer/src/components/tab-bar/use-sortable-tab-rename.ts diff --git a/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx b/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx index bc170a98a2b..dc152df4c70 100644 --- a/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx +++ b/src/renderer/src/components/native-chat/NativeChatResolvedView.tsx @@ -130,6 +130,7 @@ export function NativeChatResolvedView({ }) const contextMenu = useNativeChatContextMenu({ rootRef, + onSwitchToTerminal, actions: { onPaste: pasteClipboardIntoComposer, ...(contextMenuActions ?? emptyNativeChatContextMenuActions) diff --git a/src/renderer/src/components/native-chat/native-chat-availability.test.ts b/src/renderer/src/components/native-chat/native-chat-availability.test.ts index a08bc476d2f..409f8c7329f 100644 --- a/src/renderer/src/components/native-chat/native-chat-availability.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-availability.test.ts @@ -1,5 +1,5 @@ import { describe, it, expect } from 'vitest' -import { canToggleNativeChat } from './native-chat-availability' +import { canSwitchNativeChatView, canToggleNativeChat } from './native-chat-availability' import { isNativeChatTranscriptLocalReadable } from '@/lib/native-chat-transcript-readability' describe('canToggleNativeChat', () => { @@ -225,3 +225,37 @@ describe('canToggleNativeChat', () => { ).toBe(false) }) }) + +describe('canSwitchNativeChatView', () => { + it('allows bridge chat to expose a terminal/chat switcher', () => { + expect( + canSwitchNativeChatView({ + experimentalNativeChatEnabled: true, + contentType: 'terminal', + launchAgent: 'claude' + }) + ).toBe(true) + }) + + it('keeps structured sessions free of terminal/chat switchers', () => { + expect( + canSwitchNativeChatView({ + experimentalNativeChatEnabled: true, + contentType: 'terminal', + launchAgent: 'codex', + structuredSessionId: 'thread-1' + }) + ).toBe(false) + }) + + it('keeps structured sessions hidden even when toggling back', () => { + expect( + canSwitchNativeChatView({ + experimentalNativeChatEnabled: true, + contentType: 'terminal', + isChatViewMode: true, + structuredSessionId: 'thread-1' + }) + ).toBe(false) + }) +}) diff --git a/src/renderer/src/components/native-chat/native-chat-availability.ts b/src/renderer/src/components/native-chat/native-chat-availability.ts index 3cad7fc157f..1f883630f13 100644 --- a/src/renderer/src/components/native-chat/native-chat-availability.ts +++ b/src/renderer/src/components/native-chat/native-chat-availability.ts @@ -58,3 +58,16 @@ export function canToggleNativeChat(input: NativeChatAvailabilityInput): boolean } return isNativeChatSupportedAgent(agent) } + +/** Whether a user-facing terminal⇄chat switcher may be offered. A structured + * session IS the conversation — it owns the surface with no live TUI beneath + * it — so only terminal-backed (bridge) chat, which renders a terminal we can + * return to, gets the switch. */ +export function canSwitchNativeChatView( + input: NativeChatAvailabilityInput & { structuredSessionId?: string | null } +): boolean { + if (input.structuredSessionId) { + return false + } + return canToggleNativeChat(input) +} diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx new file mode 100644 index 00000000000..f4e8b3efc72 --- /dev/null +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.test.tsx @@ -0,0 +1,110 @@ +/** + * @vitest-environment happy-dom + */ +import React, { createRef, type ReactNode } from 'react' +import { renderToStaticMarkup } from 'react-dom/server' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { + emptyNativeChatContextMenuActions, + useNativeChatContextMenu, + type NativeChatContextMenuActions +} from './use-native-chat-context-menu' + +type ItemProps = { onSelect?: () => void; children?: ReactNode } + +const items = vi.hoisted(() => ({ list: [] as ItemProps[] })) + +vi.mock('@/components/ui/dropdown-menu', () => ({ + DropdownMenu: ({ children }: { children?: ReactNode }) => children, + DropdownMenuContent: ({ children }: { children?: ReactNode }) => children, + DropdownMenuItem: (props: ItemProps) => { + items.list.push(props) + return props.children + }, + DropdownMenuLabel: ({ children }: { children?: ReactNode }) => children, + DropdownMenuSeparator: () => null, + DropdownMenuShortcut: ({ children }: { children?: ReactNode }) => children, + DropdownMenuSub: ({ children }: { children?: ReactNode }) => children, + DropdownMenuSubContent: ({ children }: { children?: ReactNode }) => children, + DropdownMenuSubTrigger: ({ children }: { children?: ReactNode }) => children, + DropdownMenuTrigger: ({ children }: { children?: ReactNode }) => children +})) + +vi.mock('lucide-react', () => { + const Icon = () => null + return { + Clipboard: Icon, + Copy: Icon, + GitFork: Icon, + Maximize2: Icon, + MessageSquarePlus: Icon, + Minimize2: Icon, + PanelBottomClose: Icon, + PanelsTopLeft: Icon, + PanelRightClose: Icon, + Pencil: Icon, + SquareTerminal: Icon, + X: Icon + } +}) + +vi.mock('@/i18n/i18n', () => ({ + translate: (_key: string, fallback: string) => fallback +})) + +function childrenText(children: ReactNode): string { + return React.Children.toArray(children) + .map((child) => { + if (typeof child === 'string') { + return child + } + return React.isValidElement<{ children?: ReactNode }>(child) + ? childrenText(child.props.children) + : '' + }) + .join('') +} + +function Harness({ onSwitchToTerminal }: { onSwitchToTerminal?: () => void }) { + const rootRef = createRef() + const { menu } = useNativeChatContextMenu({ + rootRef, + onSwitchToTerminal, + actions: { + ...emptyNativeChatContextMenuActions, + onPaste: vi.fn() + } satisfies NativeChatContextMenuActions + }) + return menu +} + +describe('useNativeChatContextMenu', () => { + beforeEach(() => { + items.list = [] + }) + + it('restores the bridge switch-to-terminal action when supplied', () => { + const onSwitchToTerminal = vi.fn() + + renderToStaticMarkup() + + // Keep the assertions tied to the mocked menu item's semantic children. + const labels = items.list.map((candidate) => childrenText(candidate.children)) + + expect(labels.some((label) => label.startsWith('Switch to terminal view'))).toBe(true) + const item = items.list.find((candidate) => + childrenText(candidate.children).startsWith('Switch to terminal view') + ) + expect(item).toBeDefined() + item?.onSelect?.() + expect(onSwitchToTerminal).toHaveBeenCalledTimes(1) + }) + + it('does not render a terminal switch action without a bridge callback', () => { + renderToStaticMarkup() + + expect( + items.list.some((candidate) => childrenText(candidate.children) === 'Switch to terminal view') + ).toBe(false) + }) +}) diff --git a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx index 8e939401a47..a2d7dae4c93 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx +++ b/src/renderer/src/components/native-chat/use-native-chat-context-menu.tsx @@ -17,6 +17,7 @@ import { PanelsTopLeft, PanelRightClose, Pencil, + SquareTerminal, X } from 'lucide-react' import { @@ -28,7 +29,7 @@ import { DropdownMenuTrigger } from '@/components/ui/dropdown-menu' import { translate } from '@/i18n/i18n' -import { isMacPlatform } from './native-chat-shortcut' +import { isMacPlatform, nativeChatToggleShortcutLabel } from './native-chat-shortcut' type NativeChatContextMenuState = { open: boolean @@ -38,6 +39,8 @@ type NativeChatContextMenuState = { type UseNativeChatContextMenuArgs = { rootRef: RefObject + /** Bridge-only escape hatch; structured sessions never mount this menu. */ + onSwitchToTerminal?: () => void actions: NativeChatContextMenuActions } @@ -83,7 +86,11 @@ export const emptyNativeChatContextMenuActions: Omit {} } -export function useNativeChatContextMenu({ rootRef, actions }: UseNativeChatContextMenuArgs): { +export function useNativeChatContextMenu({ + rootRef, + onSwitchToTerminal, + actions +}: UseNativeChatContextMenuArgs): { onContextMenuCapture: MouseEventHandler onSelectionCapture: () => void menu: React.JSX.Element @@ -95,6 +102,7 @@ export function useNativeChatContextMenu({ rootRef, actions }: UseNativeChatCont point: { x: 0, y: 0 }, selectedText: '' }) + const shortcutLabel = nativeChatToggleShortcutLabel(isMacPlatform()) const rememberCurrentSelection = useCallback(() => { const selectedText = getNativeChatSelectedText(rootRef.current) @@ -161,6 +169,16 @@ export function useNativeChatContextMenu({ rootRef, actions }: UseNativeChatCont {translate('auto.components.terminal.pane.TerminalContextMenu.0a917b591a', 'Paste')} + {onSwitchToTerminal ? ( + + + {translate( + 'components.tab.bar.SortableTabContextMenu.switchToTerminalView', + 'Switch to terminal view' + )} + {shortcutLabel} + + ) : null} {actions.canContinueAgentSessionInNewSession ? ( diff --git a/src/renderer/src/components/native-chat/use-native-chat-toggle-shortcut.ts b/src/renderer/src/components/native-chat/use-native-chat-toggle-shortcut.ts index 90c84470cee..272e0ba81eb 100644 --- a/src/renderer/src/components/native-chat/use-native-chat-toggle-shortcut.ts +++ b/src/renderer/src/components/native-chat/use-native-chat-toggle-shortcut.ts @@ -2,7 +2,7 @@ import { useEffect } from 'react' import { useAppStore } from '../../store' import type { AgentType } from '../../../../shared/agent-status-types' import type { TerminalLayoutSnapshot } from '../../../../shared/terminal-tab-types' -import { resolveCommittedTitleAgentType } from '@/lib/pane-agent-evidence' +import { resolveNativeChatTabAgentEvidence } from '../tab-bar/native-chat-tab-agent-evidence' import { canToggleNativeChat } from './native-chat-availability' import { isNativeChatTranscriptLocalReadable } from '@/lib/native-chat-transcript-readability' import { isMacPlatform, matchesNativeChatToggleShortcut } from './native-chat-shortcut' @@ -58,7 +58,10 @@ export function useNativeChatToggleShortcut(worktreeId: string, isWorktreeActive const tab = (state.unifiedTabsByWorktree[worktreeId] ?? []).find( (candidate) => candidate.id === group.activeTabId ) - if (!tab || tab.contentType !== 'terminal') { + // contentType gates out standalone structured (agent-session) tabs; + // structuredSessionId gates out a terminal tab that adopted one, which + // renders the structured surface with no TUI to switch back to. + if (!tab || tab.contentType !== 'terminal' || tab.structuredSessionId) { return } const terminalTab = (state.tabsByWorktree[worktreeId] ?? []).find( @@ -75,10 +78,10 @@ export function useNativeChatToggleShortcut(worktreeId: string, isWorktreeActive terminalLayout, agentStatusByPaneKey: state.agentStatusByPaneKey }) - const titleFallbackAgent = tabWideFallbackSafe - ? (resolveCommittedTitleAgentType(tab.label ?? '') ?? - (terminalTab ? resolveCommittedTitleAgentType(terminalTab.title) : null)) - : null + const titleFallbackAgent = + tabWideFallbackSafe && terminalTab + ? resolveNativeChatTabAgentEvidence(terminalTab, tab) + : null if ( !canToggleNativeChat({ experimentalNativeChatEnabled: state.settings?.experimentalNativeChat === true, diff --git a/src/renderer/src/components/tab-bar/SortableTab.tsx b/src/renderer/src/components/tab-bar/SortableTab.tsx index cb5966d1e23..8e793f96773 100644 --- a/src/renderer/src/components/tab-bar/SortableTab.tsx +++ b/src/renderer/src/components/tab-bar/SortableTab.tsx @@ -1,4 +1,4 @@ -import { useCallback, useEffect, useRef, useState } from 'react' +import { useCallback, useEffect, useState } from 'react' import { useSortable } from '@dnd-kit/sortable' import { X, Minimize2, Pin } from 'lucide-react' import { stripLeadingAgentTitleDecoration } from '../../../../shared/agent-title-decoration' @@ -17,10 +17,7 @@ import { type DropIndicator } from './drop-indicator' import { preventMiddleButtonDefault } from './middle-button-default-guard' -import { - RENAME_TERMINAL_TAB_EVENT, - type RenameTerminalTabDetail -} from './terminal-tab-rename-request' +import { useSortableTabRename } from './use-sortable-tab-rename' import { SortableTabContextMenu } from './SortableTabContextMenu' import { translate } from '@/i18n/i18n' import { TAB_CONTAINER_WIDTH_CLASSES, TAB_LABEL_WIDTH_CLASSES } from './tab-width-rules' @@ -55,6 +52,12 @@ type SortableTabProps = { dragData: TabDragItemData dropIndicator?: DropIndicator includeTopTabBorder?: boolean + /** True when this agent terminal can switch between the terminal and native chat views; surfaces the "Switch view" context-menu item. */ + canToggleViewMode?: boolean + /** True when the tab is currently showing the native chat view. */ + isChatView?: boolean + /** Toggle the tab between terminal and native chat view. */ + onToggleViewMode?: () => void } export const CLOSE_ALL_CONTEXT_MENUS_EVENT = 'orca-close-all-context-menus' @@ -80,7 +83,10 @@ export default function SortableTab({ onToggleExpand, dragData, dropIndicator, - includeTopTabBorder = true + includeTopTabBorder = true, + canToggleViewMode = false, + isChatView = false, + onToggleViewMode }: SortableTabProps): React.JSX.Element { // Why: agent-completion unread exists even with terminal-attention off; collapse both sources to one primitive so unrelated tabs don't re-render. const hasUnreadActivity = useAppStore((s) => @@ -121,72 +127,23 @@ export default function SortableTab({ // Why: no transform/transition/opacity so tabs stay anchored during drag, only the insertion bar moves (see TabBar.tsx). const [menuOpen, setMenuOpen] = useState(false) const [menuPoint, setMenuPoint] = useState({ x: 0, y: 0 }) - const [isEditing, setIsEditing] = useState(false) + const { + isEditing, + renameValue, + setRenameValue, + handleRenameOpen, + commitRename, + cancelRename, + setRenameInputElement + } = useSortableTabRename({ + tabId: tab.id, + title: tab.title, + customTitle: tab.customTitle, + onSetCustomTitle + }) // Why: a live working/needs-input state is newer than a prior-turn unread, so it owns the icon until the turn ends. const showUnreadActivity = hasUnreadActivity && !isEditing && !isTerminalTabActivityLive(activityStatus) - const [renameValue, setRenameValue] = useState('') - const renameFocusFrameRef = useRef(null) - // Why: onBlur fires during Input unmount; mark rename resolved so it can't re-commit and overwrite discarded edits. - const committedOrCancelledRef = useRef(false) - - const handleRenameOpen = useCallback(() => { - committedOrCancelledRef.current = false - // Why: snapshot title once; don't refresh if tab.title changes mid-edit (e.g. OSC) so the user's edits aren't overwritten. - setRenameValue(tab.customTitle ?? tab.title) - setIsEditing(true) - }, [tab.customTitle, tab.title]) - - const commitRename = useCallback(() => { - if (committedOrCancelledRef.current) { - return - } - committedOrCancelledRef.current = true - const trimmed = renameValue.trim() - onSetCustomTitle(tab.id, trimmed.length > 0 ? trimmed : null) - setIsEditing(false) - }, [renameValue, onSetCustomTitle, tab.id]) - - const cancelRename = useCallback(() => { - committedOrCancelledRef.current = true - setIsEditing(false) - }, []) - - const setRenameInputElement = useCallback((input: HTMLInputElement | null) => { - if (renameFocusFrameRef.current !== null) { - cancelAnimationFrame(renameFocusFrameRef.current) - renameFocusFrameRef.current = null - } - if (!input) { - return - } - // Why: defer past Radix menu teardown/focus restore; key off input mount so title updates don't re-select edited text. - renameFocusFrameRef.current = requestAnimationFrame(() => { - renameFocusFrameRef.current = null - input.focus() - input.select() - }) - }, []) - - // Why the ref: keeps the listener subscribed to tab.id alone, so OSC title churn can't - // resubscribe it mid-edit. Written from an Effect, not in render -- a render React discards - // must not leave a stale handler behind for the next commit to fire. - const handleRenameOpenRef = useRef(handleRenameOpen) - useEffect(() => { - handleRenameOpenRef.current = handleRenameOpen - }, [handleRenameOpen]) - - useEffect(() => { - const onRenameRequest = (event: Event): void => { - const detail = (event as CustomEvent).detail - if (detail?.tabId !== tab.id) { - return - } - handleRenameOpenRef.current() - } - window.addEventListener(RENAME_TERMINAL_TAB_EVENT, onRenameRequest) - return () => window.removeEventListener(RENAME_TERMINAL_TAB_EVENT, onRenameRequest) - }, [tab.id]) useEffect(() => { const closeMenu = (): void => setMenuOpen(false) @@ -439,6 +396,9 @@ export default function SortableTab({ onRenameOpen={handleRenameOpen} onSetTabColor={onSetTabColor} onTogglePin={onTogglePin} + canToggleViewMode={canToggleViewMode} + isChatView={isChatView} + onToggleViewMode={onToggleViewMode} /> ) diff --git a/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx b/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx index 53b31012d39..5b6caaecb55 100644 --- a/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx +++ b/src/renderer/src/components/tab-bar/SortableTabContextMenu.tsx @@ -1,4 +1,14 @@ -import { PanelLeftClose, PanelRightClose, Pin, PinOff, Pencil, X, ListX } from 'lucide-react' +import { + MessageSquare, + PanelLeftClose, + PanelRightClose, + Pin, + PinOff, + Pencil, + SquareTerminal, + X, + ListX +} from 'lucide-react' import { DropdownMenu, DropdownMenuContent, @@ -97,6 +107,15 @@ type SortableTabContextMenuProps = { onRenameOpen: () => void onSetTabColor: (tabId: string, color: string | null) => void onTogglePin: () => void + /** True when this tab is an agent terminal that can switch between the terminal + * and native chat views; gates the "Switch view" menu item. Structured + * sessions never qualify — they have no terminal underneath. */ + canToggleViewMode?: boolean + /** True when the tab is currently showing the native chat view (drives the + * item's label/icon between "chat" and "terminal"). */ + isChatView?: boolean + /** Toggle the tab between terminal and native chat view. */ + onToggleViewMode?: () => void } export function SortableTabContextMenu({ @@ -118,7 +137,10 @@ export function SortableTabContextMenu({ onCloseToLeft, onRenameOpen, onSetTabColor, - onTogglePin + onTogglePin, + canToggleViewMode = false, + isChatView = false, + onToggleViewMode }: SortableTabContextMenuProps): React.JSX.Element { const keybindings = useAppStore((state) => state.keybindings) const splitRightShortcut = formatShortcutLabel('terminal.splitRight', keybindings) @@ -147,6 +169,27 @@ export function SortableTabContextMenu({ splitRightShortcut={splitRightShortcut} splitDownShortcut={splitDownShortcut} /> + {canToggleViewMode && onToggleViewMode ? ( + <> + + + {isChatView ? ( + + ) : ( + + )} + {isChatView + ? translate( + 'components.tab.bar.SortableTabContextMenu.switchToTerminalView', + 'Switch to terminal view' + ) + : translate( + 'components.tab.bar.SortableTabContextMenu.switchToChatView', + 'Switch to chat view' + )} + + + ) : null} {isPinned ? ( diff --git a/src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.test.ts b/src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.test.ts new file mode 100644 index 00000000000..0ebf19e6e9b --- /dev/null +++ b/src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.test.ts @@ -0,0 +1,38 @@ +import { describe, expect, it } from 'vitest' +import { resolveNativeChatTabAgentEvidence } from './native-chat-tab-agent-evidence' + +describe('resolveNativeChatTabAgentEvidence', () => { + it('uses the retained provider identity when a generated title masks the process title', () => { + expect( + resolveNativeChatTabAgentEvidence( + { + title: 'Summarize recent commits', + aiVaultTitle: null + }, + { + label: 'Summarize recent commits', + aiVaultTitle: { + agent: 'codex', + sessionId: 'thread-1', + title: 'Summarize recent commits' + } + } + ) + ).toBe('codex') + }) + + it('keeps the committed process-title signal ahead of retained metadata', () => { + expect( + resolveNativeChatTabAgentEvidence( + { + title: 'Claude Code', + aiVaultTitle: { agent: 'codex', sessionId: 'thread-1', title: 'Old title' } + }, + { + label: 'Old title', + aiVaultTitle: null + } + ) + ).toBe('claude') + }) +}) diff --git a/src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.ts b/src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.ts new file mode 100644 index 00000000000..54b97ea775b --- /dev/null +++ b/src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.ts @@ -0,0 +1,18 @@ +import type { Tab } from '../../../../shared/tab-types' +import type { TerminalTab } from '../../../../shared/terminal-tab-types' +import type { TuiAgent } from '../../../../shared/tui-agent' +import { resolveCommittedTitleAgentType } from '@/lib/pane-agent-evidence' + +/** Resolve durable tab metadata used before a live pane status arrives. */ +export function resolveNativeChatTabAgentEvidence( + tab: Pick, + unifiedTab?: Pick +): TuiAgent | null { + return ( + resolveCommittedTitleAgentType(unifiedTab?.label ?? '') ?? + resolveCommittedTitleAgentType(tab.title) ?? + unifiedTab?.aiVaultTitle?.agent ?? + tab.aiVaultTitle?.agent ?? + null + ) +} diff --git a/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx b/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx index c0af60cf6fc..24114f42a16 100644 --- a/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx +++ b/src/renderer/src/components/tab-bar/tab-bar-item-surface.tsx @@ -4,6 +4,8 @@ import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { TuiAgent } from '../../../../shared/tui-agent' import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import type { OpenFile } from '../../store/slices/editor' +import { canSwitchNativeChatView } from '../native-chat/native-chat-availability' +import { resolveNativeChatTabAgentEvidence } from './native-chat-tab-agent-evidence' import SortableTab from './SortableTab' import EditorFileTab from './EditorFileTab' import BrowserTab from './BrowserTab' @@ -56,7 +58,17 @@ export function renderTabBarItems({ onCloseAllFiles, onMakePreviewFilePermanent } = props - const { resolvedGroupId, generatedTabTitlesEnabled, statusByRelativePath } = runtime + const { + resolvedGroupId, + generatedTabTitlesEnabled, + unifiedTabByVisibleId, + nativeChatEnabled, + tabAgentTypesByTabId, + nativeChatTabWideFallbackUnsafeTabsById, + nativeChatTranscriptIsLocalReadable, + toggleTabViewMode, + statusByRelativePath + } = runtime // A selected client-hosted row covers the pane, so the tab it covers must stop looking active — // the group's own activeTabId never moves for it, and two underlines would show at once. @@ -89,6 +101,24 @@ export function renderTabBarItems({ ...item.data, title: resolveTerminalTabTitle(item.data, generatedTabTitlesEnabled, item.data.title) } + const unifiedTabForItem = unifiedTabByVisibleId.get(item.id) + // Carry the agent *identity* (not just "an agent exists") so the native-chat gate can reject agents like Grok. + const resolvedAgent = resolveNativeChatTabAgentEvidence(terminalTab, unifiedTabForItem) + // Key the live-agent lookup by the backing terminal tab id: agent-status pane keys use it, not the unified tab id. + const detectedAgent = tabAgentTypesByTabId[terminalTab.id] ?? null + const tabWideFallbackSafe = nativeChatTabWideFallbackUnsafeTabsById[terminalTab.id] !== true + const canToggleViewMode = + unifiedTabForItem !== undefined && + canSwitchNativeChatView({ + experimentalNativeChatEnabled: nativeChatEnabled, + contentType: 'terminal', + launchAgent: tabWideFallbackSafe ? terminalTab.launchAgent : null, + detectedAgent, + resolvedAgent: tabWideFallbackSafe ? resolvedAgent : null, + nativeChatTranscriptIsLocalReadable, + isChatViewMode: unifiedTabForItem.viewMode === 'chat', + structuredSessionId: unifiedTabForItem.structuredSessionId ?? null + }) return ( toggleTabViewMode(unifiedTabForItem.id) : undefined + } hasTabsToRight={index < items.length - 1} hasTabsToLeft={index > 0} isActive={ diff --git a/src/renderer/src/components/tab-bar/use-sortable-tab-rename.ts b/src/renderer/src/components/tab-bar/use-sortable-tab-rename.ts new file mode 100644 index 00000000000..05b85d226bb --- /dev/null +++ b/src/renderer/src/components/tab-bar/use-sortable-tab-rename.ts @@ -0,0 +1,94 @@ +import { useCallback, useEffect, useRef, useState } from 'react' +import { + RENAME_TERMINAL_TAB_EVENT, + type RenameTerminalTabDetail +} from './terminal-tab-rename-request' + +/** Inline tab-title rename: snapshots the title on open so mid-edit OSC churn + * cannot overwrite the user's text, commits at most once, and answers the + * window rename request addressed to this tab. */ +export function useSortableTabRename({ + tabId, + title, + customTitle, + onSetCustomTitle +}: { + tabId: string + title: string + customTitle?: string | null + onSetCustomTitle: (tabId: string, title: string | null) => void +}) { + const [isEditing, setIsEditing] = useState(false) + const [renameValue, setRenameValue] = useState('') + const renameFocusFrameRef = useRef(null) + // Why: onBlur fires during Input unmount; mark rename resolved so it can't re-commit and overwrite discarded edits. + const committedOrCancelledRef = useRef(false) + + const handleRenameOpen = useCallback(() => { + committedOrCancelledRef.current = false + // Why: snapshot title once; don't refresh if tab.title changes mid-edit (e.g. OSC) so the user's edits aren't overwritten. + setRenameValue(customTitle ?? title) + setIsEditing(true) + }, [customTitle, title]) + + const commitRename = useCallback(() => { + if (committedOrCancelledRef.current) { + return + } + committedOrCancelledRef.current = true + const trimmed = renameValue.trim() + onSetCustomTitle(tabId, trimmed.length > 0 ? trimmed : null) + setIsEditing(false) + }, [renameValue, onSetCustomTitle, tabId]) + + const cancelRename = useCallback(() => { + committedOrCancelledRef.current = true + setIsEditing(false) + }, []) + + const setRenameInputElement = useCallback((input: HTMLInputElement | null) => { + if (renameFocusFrameRef.current !== null) { + cancelAnimationFrame(renameFocusFrameRef.current) + renameFocusFrameRef.current = null + } + if (!input) { + return + } + // Why: defer past Radix menu teardown/focus restore; key off input mount so title updates don't re-select edited text. + renameFocusFrameRef.current = requestAnimationFrame(() => { + renameFocusFrameRef.current = null + input.focus() + input.select() + }) + }, []) + + // Why the ref: keeps the listener subscribed to tabId alone, so OSC title churn can't + // resubscribe it mid-edit. Written from an Effect, not in render -- a render React discards + // must not leave a stale handler behind for the next commit to fire. + const handleRenameOpenRef = useRef(handleRenameOpen) + useEffect(() => { + handleRenameOpenRef.current = handleRenameOpen + }, [handleRenameOpen]) + + useEffect(() => { + const onRenameRequest = (event: Event): void => { + const detail = (event as CustomEvent).detail + if (detail?.tabId !== tabId) { + return + } + handleRenameOpenRef.current() + } + window.addEventListener(RENAME_TERMINAL_TAB_EVENT, onRenameRequest) + return () => window.removeEventListener(RENAME_TERMINAL_TAB_EVENT, onRenameRequest) + }, [tabId]) + + return { + isEditing, + renameValue, + setRenameValue, + handleRenameOpen, + commitRename, + cancelRename, + setRenameInputElement + } +} diff --git a/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx b/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx index d8dc54c954b..58a39f28a50 100644 --- a/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalContextMenu.test.tsx @@ -77,6 +77,9 @@ function renderMenu(overrides: Record = {}): string { canContinueAgentSessionInNewSession: false, onContinueAgentSessionInNewSession: vi.fn(), onForkAgentSession: vi.fn(), + canToggleNativeChat: false, + isNativeChatView: false, + onToggleNativeChat: vi.fn(), onCopyAgentSessionContext: vi.fn(), quickCommandHosts: [ { hostId: 'local' as const, label: 'Local Linux', repoCommands: [], globalCommands: [] } diff --git a/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx b/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx index 8311fb3df9d..2cc76cd7164 100644 --- a/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalContextMenu.tsx @@ -6,11 +6,13 @@ import { Eraser, GitFork, Maximize2, + MessageSquare, Minimize2, PanelBottomClose, PanelsTopLeft, PanelRightClose, Pencil, + SquareTerminal, TextSelect, X } from 'lucide-react' @@ -28,6 +30,7 @@ import type { ExecutionHostId } from '../../../../shared/execution-host' import { formatPrimaryShortcutLabel } from '@/hooks/useShortcutLabel' import type { KeybindingOverrides } from '../../../../shared/keybindings' import { translate } from '@/i18n/i18n' +import { isMacPlatform, nativeChatToggleShortcutLabel } from '../native-chat/native-chat-shortcut' import { AgentSessionContinuationMenuItem } from './AgentSessionContinuationMenuItem' import type { TerminalQuickCommandMenuHost } from '@/hooks/use-terminal-quick-command-hosts' import { TerminalQuickCommandsSubmenu } from './TerminalQuickCommandsSubmenu' @@ -53,6 +56,11 @@ type TerminalContextMenuProps = { canContinueAgentSessionInNewSession: boolean onContinueAgentSessionInNewSession: () => void onForkAgentSession: () => void + /** True when this pane may switch between the terminal and native chat views. + * Structured sessions are excluded — they have no terminal underneath. */ + canToggleNativeChat: boolean + isNativeChatView: boolean + onToggleNativeChat: () => void onCopyAgentSessionContext: () => void quickCommandHosts: TerminalQuickCommandMenuHost[] quickCommandHostLoadFailed: boolean @@ -91,6 +99,9 @@ export default function TerminalContextMenu({ canContinueAgentSessionInNewSession, onContinueAgentSessionInNewSession, onForkAgentSession, + canToggleNativeChat, + isNativeChatView, + onToggleNativeChat, onCopyAgentSessionContext, quickCommandHosts, quickCommandHostLoadFailed, @@ -119,7 +130,8 @@ export default function TerminalContextMenu({ expand: formatPrimaryShortcutLabel('terminal.expandPane', keybindings), setTitle: formatPrimaryShortcutLabel('terminal.setTitle', keybindings), clearPaneTitle: formatPrimaryShortcutLabel('terminal.clearPaneTitle', keybindings), - close: formatPrimaryShortcutLabel('terminal.closePane', keybindings) + close: formatPrimaryShortcutLabel('terminal.closePane', keybindings), + nativeChat: nativeChatToggleShortcutLabel(isMacPlatform()) }), [keybindings] ) @@ -209,6 +221,21 @@ export default function TerminalContextMenu({ 'Copy Context' )} + {canToggleNativeChat ? ( + + {isNativeChatView ? : } + {isNativeChatView + ? translate( + 'components.tab.bar.SortableTabContextMenu.switchToTerminalView', + 'Switch to terminal view' + ) + : translate( + 'components.tab.bar.SortableTabContextMenu.switchToChatView', + 'Switch to chat view' + )} + {shortcuts.nativeChat} + + ) : null} diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneHeaderOverlay.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneHeaderOverlay.tsx index 6b246701c43..6256bc64c07 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneHeaderOverlay.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneHeaderOverlay.tsx @@ -1,5 +1,11 @@ import type { CSSProperties, RefObject } from 'react' -import { MessageSquarePlus, SquareSplitVertical, X } from 'lucide-react' +import { + MessageSquare, + MessageSquarePlus, + SquareSplitVertical, + SquareTerminal, + X +} from 'lucide-react' import type { ManagedPane, PaneManager } from '@/lib/pane-manager/pane-manager' import { Button } from '@/components/ui/button' import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' @@ -36,6 +42,16 @@ type TerminalPaneHeaderOverlayProps = { hiddenStartupStyle: CSSProperties managerRef: RefObject paneTransportsRef: RefObject> + /** When true, this pane can switch between the terminal and the native chat + * view; renders a chat/terminal toggle as the first button in the pane header + * actions row (beside split/close). The caller gates it to the active pane to + * avoid duplicating it across splits, and to bridge chat only — a structured + * session has no terminal underneath to switch to. */ + canToggleNativeChat?: boolean + /** True when the active pane is currently showing the native chat view. */ + isChatViewMode?: boolean + /** Flip the active pane between the terminal and the native chat view. */ + onToggleNativeChat?: () => void canContinueAgentSessionInNewSession?: boolean onContinueAgentSessionInNewSession?: (pane: ManagedPane) => void onSplitPane: (pane: ManagedPane, direction: 'vertical' | 'horizontal') => void @@ -71,6 +87,9 @@ export default function TerminalPaneHeaderOverlay({ hiddenStartupStyle, managerRef, paneTransportsRef, + canToggleNativeChat, + isChatViewMode, + onToggleNativeChat, canContinueAgentSessionInNewSession, onContinueAgentSessionInNewSession, onSplitPane, @@ -255,6 +274,47 @@ export default function TerminalPaneHeaderOverlay({ ) : null} + {canToggleNativeChat && isActivePane ? ( + + + + + + {isChatViewMode + ? translate('components.native-chat.toggle.showTerminal', 'Show terminal') + : translate('components.native-chat.toggle.showChat', 'Show chat view')} + + + ) : null} {showAlwaysOnHeaders && showSplitButton ? ( diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneOverlayLayer.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneOverlayLayer.tsx index 79d0f42a14e..2cb20783f53 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneOverlayLayer.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneOverlayLayer.tsx @@ -8,6 +8,7 @@ import { type ActivityTerminalPortalTarget } from '../activity/activity-terminal-portal' import { shouldMountBackgroundWorktreeTab } from '../terminal/background-terminal-worktree-mount' +import { useNativeChatToggleShortcut } from '../native-chat/use-native-chat-toggle-shortcut' import { TerminalOverlaySlot } from './TerminalOverlaySlot' import { useTerminalTabColdParking } from './use-terminal-tab-cold-parking' @@ -59,6 +60,8 @@ const TerminalPaneOverlayLayer = memo(function TerminalPaneOverlayLayer({ const setActiveWorktree = useAppStore((state) => state.setActiveWorktree) const reconcileWorktreeTabModel = useAppStore((state) => state.reconcileWorktreeTabModel) + useNativeChatToggleShortcut(worktreeId, isWorktreeActive) + const leaveWorktreeIfEmpty = useCallback(() => { const state = useAppStore.getState() if (state.activeWorktreeId !== worktreeId) { diff --git a/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx b/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx index fc33cab3f18..1773aa48d20 100644 --- a/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx +++ b/src/renderer/src/components/terminal-pane/TerminalPaneSurface.tsx @@ -32,6 +32,8 @@ export function TerminalPaneSurface({ const { activePane, activePaneCanContinueInNewSession, + activePaneCanToggleChat, + activePaneIsChatLeaf, activatePaneTitleInteraction, agentSessionContinuation, agentSessionFork, @@ -39,6 +41,8 @@ export function TerminalPaneSurface({ closeTerminalLinkActions, contextMenu, contextMenuCanContinueInNewSession, + contextMenuCanToggleChat, + contextMenuIsChatView, cwd, daemonActions, dismissTerminalError, @@ -46,6 +50,7 @@ export function TerminalPaneSurface({ expandedPaneId, handleCancelClose, handleConfirmClose, + handleContextMenuToggleNativeChat, handlePrimarySelectionAuxClick, handlePrimarySelectionMiddleMouseDown, handleRemoveTitle, @@ -54,6 +59,7 @@ export function TerminalPaneSurface({ handleRenameSubmit, handleRequestClosePane, handleStartRename, + handleToggleNativeChat, hiddenStartupStyle, isActive, keybindings, @@ -231,6 +237,9 @@ export function TerminalPaneSurface({ canContinueAgentSessionInNewSession={contextMenuCanContinueInNewSession} onContinueAgentSessionInNewSession={contextMenu.onContinueAgentSessionInNewSession} onForkAgentSession={() => void contextMenu.onForkAgentSession()} + canToggleNativeChat={contextMenuCanToggleChat} + isNativeChatView={contextMenuIsChatView} + onToggleNativeChat={handleContextMenuToggleNativeChat} onCopyAgentSessionContext={() => void contextMenu.onCopyAgentSessionContext()} quickCommandHosts={visibleQuickCommandHosts} quickCommandHostLoadFailed={quickCommandHostLoadFailed} @@ -303,6 +312,9 @@ export function TerminalPaneSurface({ hiddenStartupStyle={hiddenStartupStyle} managerRef={managerRef} paneTransportsRef={paneTransportsRef} + canToggleNativeChat={activePaneCanToggleChat} + isChatViewMode={activePaneIsChatLeaf} + onToggleNativeChat={handleToggleNativeChat} canContinueAgentSessionInNewSession={activePaneCanContinueInNewSession} onContinueAgentSessionInNewSession={(pane) => contextMenu.runForPane(pane.id, contextMenu.onContinueAgentSessionInNewSession) diff --git a/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts b/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts index 6151594ff83..a6b96168047 100644 --- a/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts +++ b/src/renderer/src/components/terminal-pane/terminal-pane-hook-order-parity.test.ts @@ -11,8 +11,11 @@ const TERMINAL_PANE_HOOK_SOURCE_PATTERN = // split-cwd changes; the pane session-ID projection added one render hook (230 hooks). // Then 27 stable-action `useAppStore` subscriptions folded into four // `useTerminalPaneStoreActions()` calls, each one `useMemo` (204 hooks, 8 useMemo). +// Restoring the terminal/chat switcher added four `useCallback`s -- three in +// chat-state (can-toggle, toggle-for-leaf, toggle-active) and the context-menu +// toggle in projection (208 hooks, still 8 useMemo). const PRE_REFACTOR_HOOK_ORDER_SHA256 = - 'b541d26ff66a1db0b7150ced2a4340c5cf025aba79a651a2ad8d2fa68d403ae3' + '983ad067c9feca82c5435eb1b865674344489c368ec2007dc7bb40c81aef037c' const sourceFiles = readdirSync(__dirname) .filter((name) => TERMINAL_PANE_HOOK_SOURCE_PATTERN.test(name)) @@ -77,7 +80,7 @@ function readFlattenedHookOrder(): string[] { describe('TerminalPane refactor hook parity', () => { it('preserves the recursively flattened render hook order', () => { const hooks = readFlattenedHookOrder() - expect(hooks).toHaveLength(204) + expect(hooks).toHaveLength(208) expect(hooks.filter((hook) => hook === 'useMemo')).toHaveLength(8) expect(createHash('sha256').update(hooks.join('\n')).digest('hex')).toBe( PRE_REFACTOR_HOOK_ORDER_SHA256 diff --git a/src/renderer/src/components/terminal-pane/terminal-pane-store-subscription-budget.test.tsx b/src/renderer/src/components/terminal-pane/terminal-pane-store-subscription-budget.test.tsx index 9562c46c772..2418a5da6d3 100644 --- a/src/renderer/src/components/terminal-pane/terminal-pane-store-subscription-budget.test.tsx +++ b/src/renderer/src/components/terminal-pane/terminal-pane-store-subscription-budget.test.tsx @@ -4,8 +4,8 @@ * synchronously on every publication, so the per-pane subscription count is a * direct multiplier on agent-status burn (docs/reference/renderer-agent-status-performance.md). * - * On `main` one mounted pane opened 48 listeners; 31 of them selected values that - * can never change — 27 store actions and 4 duplicate reads of one unified tab. + * On `main` one mounted pane opened 49 listeners; 32 of them selected values that + * can never change — 28 store actions and 4 duplicate reads of one unified tab. */ import { act, createRef, type ReactNode } from 'react' import { createRoot, type Root } from 'react-dom/client' @@ -27,7 +27,7 @@ import { */ const TERMINAL_PANE_LISTENER_BUDGET = 17 /** What the same mount cost before the stable-action and unified-tab folds. */ -const PRE_FOLD_LISTENERS_PER_PANE = 48 +const PRE_FOLD_LISTENERS_PER_PANE = 49 const originalState = useAppStore.getState() @@ -103,7 +103,7 @@ describe('TerminalPane store subscription budget', () => { expect(perPane).toBe(TERMINAL_PANE_LISTENER_BUDGET) expect(perPane).toBeLessThan(PRE_FOLD_LISTENERS_PER_PANE) - // 27 stable actions plus four duplicate unified-tab reads. + // 28 stable actions plus four duplicate unified-tab reads. expect(PRE_FOLD_LISTENERS_PER_PANE - perPane).toBe(TERMINAL_PANE_STORE_ACTION_KEYS.length + 4) unmount() diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-chat-state.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-chat-state.ts index 92bc444e299..2fd75f3f093 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-pane-chat-state.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-chat-state.ts @@ -37,7 +37,8 @@ export function useTerminalPaneChatState(controller: TerminalPaneTitleController setTabCanExpandPane, setTabPaneExpanded, setTabViewMode, - suppressPtyExit + suppressPtyExit, + toggleTabViewMode } = useTerminalPaneStoreActions() const pendingCodexPaneRestartIds = useAppStore((store) => store.pendingCodexPaneRestartIds) // Why one selector: five separate subscriptions each re-read the same unified @@ -207,6 +208,50 @@ export function useTerminalPaneChatState(controller: TerminalPaneTitleController onAgentExitedRef.current = handleConfirmedAgentExit // oxlint-disable-next-line react-hooks/exhaustive-deps -- Preserve the pre-split dependency contract. }, [handleConfirmedAgentExit]) + const canToggleChatForLeaf = useCallback( + (leafId: string | null): boolean => { + // A structured session renders its own transcript with no TUI beneath it, + // so the switcher stays off for it while bridge chat keeps it. + if (structuredSessionId) { + return false + } + // Scope the "always allow toggling back" rule to the leaf showing chat; must not make an unsupported sibling look eligible. + const isChatViewForLeaf = effectiveChatViewMode && leafId !== null && chatLeafId === leafId + return (nativeChatEnabled && isChatViewForLeaf) || isChatEligibleForLeaf(leafId) + }, + [ + chatLeafId, + effectiveChatViewMode, + isChatEligibleForLeaf, + nativeChatEnabled, + structuredSessionId + ] + ) + const toggleNativeChatForLeaf = useCallback( + (leafId: string) => { + if (!unifiedTabId) { + return + } + if (effectiveChatViewMode && chatLeafId === leafId) { + setChatLeafId(null) + toggleTabViewMode(unifiedTabId) + return + } + setChatLeafId(leafId) + if (!effectiveChatViewMode) { + toggleTabViewMode(unifiedTabId) + } + }, + [chatLeafId, effectiveChatViewMode, setChatLeafId, toggleTabViewMode, unifiedTabId] + ) + const handleToggleNativeChat = useCallback(() => { + const activeLeafId = managerRef.current?.getActivePane()?.leafId ?? null + if (!activeLeafId) { + return + } + toggleNativeChatForLeaf(activeLeafId) + // oxlint-disable-next-line react-hooks/exhaustive-deps -- managerRef is a stable ref container. + }, [toggleNativeChatForLeaf]) const switchNativeChatToTerminal = useCallback(() => { if (chatLeafId && unifiedTabId) { setChatLeafId(null) @@ -249,6 +294,9 @@ export function useTerminalPaneChatState(controller: TerminalPaneTitleController getTabWideAgentHintLeafIdRef, resolveTitleAgentForLeaf, isChatEligibleForLeaf, + canToggleChatForLeaf, + toggleNativeChatForLeaf, + handleToggleNativeChat, applyNativeChatLeafRoute, switchNativeChatToTerminal, readNativeChatTerminalScreen diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts index ea179b6925e..7ef23834873 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-projection.ts @@ -1,4 +1,4 @@ -import { useEffect, useMemo } from 'react' +import { useCallback, useEffect, useMemo } from 'react' import type { CSSProperties } from 'react' import { DEFAULT_TERMINAL_DIVIDER_DARK, @@ -23,10 +23,13 @@ import { resolvePaneAgentSessionId } from './pane-agent-session-id' export function useTerminalPaneProjection(controller: TerminalPaneMobileController) { const { applyNativeChatLeafRoute, + canToggleChatForLeaf, chatLeafId, chatPaneDispatchStatus, contextMenu, contextMenuLeafId, + effectiveChatViewMode, + getContextMenuLeafId, getNativeChatLeafIds, getTabWideAgentHintLeafId, isActive, @@ -34,6 +37,7 @@ export function useTerminalPaneProjection(controller: TerminalPaneMobileControll isChatViewMode, isVisible, managerRef, + toggleNativeChatForLeaf, paneTitles, paneTransportsRef, resolveTitleAgentForLeaf, @@ -177,6 +181,17 @@ export function useTerminalPaneProjection(controller: TerminalPaneMobileControll const contextMenuCanContinueInNewSession = canContinueAgentSessionInNewSession( resolveAgentForLeaf(contextMenuLeafId) ) + // Each switcher gates on its own leaf (header=active, menu=opened-over), so mixed splits show it only where chat can render. + const activePaneCanToggleChat = canToggleChatForLeaf(activePane?.leafId ?? null) + const contextMenuCanToggleChat = canToggleChatForLeaf(contextMenuLeafId) + const contextMenuIsChatView = effectiveChatViewMode && contextMenuLeafId === chatLeafId + const handleContextMenuToggleNativeChat = useCallback(() => { + const leafId = getContextMenuLeafId() + if (!leafId) { + return + } + toggleNativeChatForLeaf(leafId) + }, [getContextMenuLeafId, toggleNativeChatForLeaf]) return { effectiveAppearance, terminalBackground, @@ -204,7 +219,11 @@ export function useTerminalPaneProjection(controller: TerminalPaneMobileControll activePaneIsChatLeaf, resolveAgentForLeaf, activePaneCanContinueInNewSession, - contextMenuCanContinueInNewSession + contextMenuCanContinueInNewSession, + activePaneCanToggleChat, + contextMenuCanToggleChat, + contextMenuIsChatView, + handleContextMenuToggleNativeChat } } diff --git a/src/renderer/src/components/terminal-pane/use-terminal-pane-store-actions.ts b/src/renderer/src/components/terminal-pane/use-terminal-pane-store-actions.ts index 5eb20c1559d..02c58e96049 100644 --- a/src/renderer/src/components/terminal-pane/use-terminal-pane-store-actions.ts +++ b/src/renderer/src/components/terminal-pane/use-terminal-pane-store-actions.ts @@ -37,6 +37,7 @@ export function useTerminalPaneStoreActions() { setTabLayout: state.setTabLayout, setTabPaneExpanded: state.setTabPaneExpanded, setTabViewMode: state.setTabViewMode, + toggleTabViewMode: state.toggleTabViewMode, suppressPtyExit: state.suppressPtyExit, updateSettings: state.updateSettings, updateTabPtyId: state.updateTabPtyId, @@ -72,6 +73,7 @@ export const TERMINAL_PANE_STORE_ACTION_KEYS = [ 'setTabLayout', 'setTabPaneExpanded', 'setTabViewMode', + 'toggleTabViewMode', 'suppressPtyExit', 'updateSettings', 'updateTabPtyId', diff --git a/src/shared/pane-agent-identity-inventory.test.ts b/src/shared/pane-agent-identity-inventory.test.ts index 739ec6987f8..6871a38eb6d 100644 --- a/src/shared/pane-agent-identity-inventory.test.ts +++ b/src/shared/pane-agent-identity-inventory.test.ts @@ -161,7 +161,7 @@ const INVENTORY: readonly InventoryGroup[] = [ helper: 'resolveCommittedTitleAgentType', classification: 'action-consumer', paths: [ - ['src/renderer/src/components/native-chat/use-native-chat-toggle-shortcut.ts', 3], + ['src/renderer/src/components/tab-bar/native-chat-tab-agent-evidence.ts', 3], ['src/renderer/src/components/terminal-pane/pty-connection/connect-pane-pty.ts', 2], ['src/renderer/src/components/terminal-pane/terminal-ctrl-enter.ts', 2], ['src/renderer/src/components/terminal-pane/terminal-windows-shift-enter.ts', 2], From e42c60e8a399831263dd8628ecd085ef34b2c9eb Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 16:47:32 -0700 Subject: [PATCH 209/398] fix(ssh): resolve a pane's binding from the target partition, not the stale local copy (#18546) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit One SSH pane accumulated one extra reattachable lease per relay restart (2, 3, 4, 5, 6 across five), and every one of them costs a `pty.attach` round trip on every later connect, forever. Nothing prunes `sshRemotePtyLeases`, so the fan-out only grows. `supersedeSiblingLeasesForPane` is fenced on the PTY the pane is durably bound to, and `durablyBoundPtyIdForPane` read `state.workspaceSession` (local) before `workspaceSessionsByHostId['ssh:']`. But `persistPtyBinding(binding, hostId)` updates ONLY the host partition: AFTER-PERSIST local= ssh:t@@pty2:old:1 host= ssh:t@@pty2:new:1 So for the length of a reconnect the local copy still names the predecessor, the fence resolved to it, supersession took an already-`expired` lease as its winner, and returned having marked nothing. Both partitions agree again once the renderer republishes its layout, which is why the settled store looks consistent and hid this. Read both partitions as an ordered list, target's own first, and test the fence by membership rather than by equality with whichever was read first. Pick the winner preferring a lease this client still has a route to, since the stale partition names an expired one. Never retire a lease that is both bound and live, so a partition disagreement can't strand a running remote process. Superseded predecessors stay `expired` and are never `terminated`: losing a lease is not evidence the shell died (docs/reference/ssh-execution-boundary.md). A pane with no binding is skipped rather than pruned, so a genuine orphan stays askable. Also re-runs supersession from the binding side after each spawn commit's binding write, so the lease/binding order at a call site no longer decides, and reconciles every pane for a target immediately before `reattachKnownPtys` reads the set it feeds to `pty.attach` — that repairs stores which already accumulated these rows. The guard suite could not catch this: every assertion bound the pane BEFORE upserting the lease, an order no caller uses. Rewritten to the spawn commits' real order (lease, then binding, then the binding-side trigger); it fails 8 assertions without this change. Added a suite that drives the real `persistPtyIpcSpawnCommit` rather than the store primitives, including the exact stale-partition state written by production's own binding writer. Verified on the Docker SSH lane: five `relay.js` SIGKILLs with recovery between each, reattachable leases flat at one per pane. Note: this bounds the reattach SET, not the store. `sshRemotePtyLeases` still has no cap or TTL and rows still accumulate; pruning is left alone deliberately, since an `expired` row without `supersededBy` is a genuine orphan and must not be dropped on age. --- .../pty-controller-ownership-routing.test.ts | 1 + .../pty-daemon-spawn-session-identity.test.ts | 1 + .../pty-pane-reservation-settlement.test.ts | 2 + ...ty-runtime-ssh-binding-persistence.test.ts | 5 + .../pty-serializer-settlement-mapping.test.ts | 2 + ...pty-session-liveness-and-ownership.test.ts | 1 + src/main/ipc/pty/ipc/spawn-commit-persist.ts | 7 + ...spawn-commit-ssh-lease-cardinality.test.ts | 289 ++++++++++++++++++ src/main/ipc/pty/pane/ssh-pane-lease-claim.ts | 44 +++ src/main/ipc/pty/runtime/spawn-commit.ts | 27 +- src/main/ipc/ssh-ipc-test-harness.ts | 2 + .../ssh-pty-lease-operations.ts | 88 +----- .../ssh-pty-pane-supersession.ts | 201 ++++++++++++ .../ssh-lease-recovery-operations.ts | 26 ++ .../ssh-reattach-pane-cardinality.test.ts | 60 ++-- .../ssh/ssh-orphan-relay-pty-sweep.test.ts | 3 +- ...ay-session-agent-hooks.integration.test.ts | 1 + .../ssh-relay-session-terminal-error.test.ts | 1 + .../ssh/ssh-relay-session-test-fixtures.ts | 1 + src/main/ssh/ssh-relay-session.ts | 6 + ...ssh-docker-transport-drop-recovery.spec.ts | 146 ++++++++- 21 files changed, 777 insertions(+), 137 deletions(-) create mode 100644 src/main/ipc/pty/ipc/spawn-commit-ssh-lease-cardinality.test.ts create mode 100644 src/main/ipc/pty/pane/ssh-pane-lease-claim.ts create mode 100644 src/main/persistence/leasing-ssh-ptys/ssh-pty-pane-supersession.ts diff --git a/src/main/ipc/pty-controller-ownership-routing.test.ts b/src/main/ipc/pty-controller-ownership-routing.test.ts index 0d236e63cb2..2101d17f26b 100644 --- a/src/main/ipc/pty-controller-ownership-routing.test.ts +++ b/src/main/ipc/pty-controller-ownership-routing.test.ts @@ -324,6 +324,7 @@ describe('registerPtyHandlers', () => { } const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), diff --git a/src/main/ipc/pty-daemon-spawn-session-identity.test.ts b/src/main/ipc/pty-daemon-spawn-session-identity.test.ts index 90ea56049fa..b1ead181530 100644 --- a/src/main/ipc/pty-daemon-spawn-session-identity.test.ts +++ b/src/main/ipc/pty-daemon-spawn-session-identity.test.ts @@ -345,6 +345,7 @@ describe('registerPtyHandlers', () => { ) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn() } registerSshPtyProvider('ssh-1', { diff --git a/src/main/ipc/pty-pane-reservation-settlement.test.ts b/src/main/ipc/pty-pane-reservation-settlement.test.ts index 64a4441f376..06e8679ace4 100644 --- a/src/main/ipc/pty-pane-reservation-settlement.test.ts +++ b/src/main/ipc/pty-pane-reservation-settlement.test.ts @@ -124,6 +124,7 @@ describe('registerPtyHandlers', () => { flushOrThrow: vi.fn(), persistPtyBinding: vi.fn(), upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), clearSshRemotePtyKillIntent: vi.fn() @@ -482,6 +483,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), diff --git a/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts b/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts index 3fd1f89b11d..d1078477d12 100644 --- a/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts +++ b/src/main/ipc/pty-runtime-ssh-binding-persistence.test.ts @@ -154,6 +154,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), @@ -260,6 +261,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn() } let controller: RuntimeSpawnController | null = null @@ -370,6 +372,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(() => { throw new Error('disk full') }), @@ -471,6 +474,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), @@ -577,6 +581,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), diff --git a/src/main/ipc/pty-serializer-settlement-mapping.test.ts b/src/main/ipc/pty-serializer-settlement-mapping.test.ts index 4af42f1dbd5..53755214238 100644 --- a/src/main/ipc/pty-serializer-settlement-mapping.test.ts +++ b/src/main/ipc/pty-serializer-settlement-mapping.test.ts @@ -108,6 +108,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), removeSshRemotePtyLease: vi.fn(), markSshRemotePtyLease: vi.fn(), @@ -212,6 +213,7 @@ describe('registerPtyHandlers', () => { } as never) const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(() => { throw new Error('disk full') }), diff --git a/src/main/ipc/pty-session-liveness-and-ownership.test.ts b/src/main/ipc/pty-session-liveness-and-ownership.test.ts index fb88fc4b6cb..574e83c1220 100644 --- a/src/main/ipc/pty-session-liveness-and-ownership.test.ts +++ b/src/main/ipc/pty-session-liveness-and-ownership.test.ts @@ -386,6 +386,7 @@ describe('registerPtyHandlers', () => { it('ignores fire-and-forget IPC for detached SSH PTYs without a provider', async () => { const store = { upsertSshRemotePtyLease: vi.fn(), + supersedeSshRemotePtyLeasesForBoundPane: vi.fn(), persistPtyBinding: vi.fn(), markSshRemotePtyLease: vi.fn(), clearSshRemotePtyKillIntent: vi.fn() diff --git a/src/main/ipc/pty/ipc/spawn-commit-persist.ts b/src/main/ipc/pty/ipc/spawn-commit-persist.ts index be9cb31394a..540ada9397d 100644 --- a/src/main/ipc/pty/ipc/spawn-commit-persist.ts +++ b/src/main/ipc/pty/ipc/spawn-commit-persist.ts @@ -134,6 +134,13 @@ export async function persistPtyIpcSpawnCommit(ctx: PtyIpcSpawnState): Promise<{ }) } } + // Why here and not at the upsert: this path leases before it binds, so supersession fenced on the + // pane's binding still named the predecessor and bailed on every reconnect — one more reattachable + // lease, and one more `pty.attach`, per reconnect forever. Runs after whichever binding write this + // commit made, so the lease/binding order no longer decides. + if (ctx.deps.store && args.connectionId && ctx.validatedLeafId !== null) { + ctx.deps.store.supersedeSshRemotePtyLeasesForBoundPane(args.connectionId, ctx.validatedLeafId) + } // Why: when the renderer has declared it will own the serializer for this paneKey, suppress the daemon-snapshot seed so its hydration path is sole authority (keyed on paneKey since the ptyId isn't known yet). See docs/mobile-prefer-renderer-scrollback.md. const rendererPreSignaled = ctx.validatedPaneKey ? pendingByPaneKey.has(ctx.validatedPaneKey) diff --git a/src/main/ipc/pty/ipc/spawn-commit-ssh-lease-cardinality.test.ts b/src/main/ipc/pty/ipc/spawn-commit-ssh-lease-cardinality.test.ts new file mode 100644 index 00000000000..6266faf0c02 --- /dev/null +++ b/src/main/ipc/pty/ipc/spawn-commit-ssh-lease-cardinality.test.ts @@ -0,0 +1,289 @@ +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest' +import { rmSync, mkdtempSync } from 'node:fs' +import { join } from 'node:path' +import { tmpdir } from 'node:os' +import { testState, createStore } from '../../../persistence-test-harness' +import { TEST_LEAF_1, TEST_LEAF_2 } from '../../../persistence-session-fixtures' +import { sshRemotePtyLeaseAllowsReattach } from '../../../../shared/ssh-types' +import { toAppSshPtyId } from '../../../providers/ssh-pty-id' +import { toSshExecutionHostId } from '../../../../shared/execution-host' +import type { PtySpawnIpcArgs, PtySpawnIpcDeps } from './spawn-types' +import { createPtyIpcSpawnState } from './spawn-state' +import { persistPtyIpcSpawnCommit } from './spawn-commit-persist' + +vi.mock('electron', () => ({ + app: { getPath: () => testState.dir }, + safeStorage: { isEncryptionAvailable: () => false } +})) + +const TARGET = 'ssh-1' +const WORKTREE = 'repo1::/worktree' +const TAB = 'tab-1' + +/** + * Drives the shipped IPC spawn commit rather than the store primitives it calls. + * + * The store-level suite could not catch this: it exercised bind-then-upsert, and this path does the + * opposite — it writes the lease row first so a force-quit in the renderer's debounce window cannot + * strand a running remote shell without one, then binds the pane. Supersession is fenced on the + * pane's binding, so under this real order it bailed on the predecessor every time and never re-ran, + * and each reconnect left one more reattachable lease for `reattachKnownPtys` to `pty.attach`. + */ +async function commitSshSpawn( + store: ReturnType, + args: { relayPtyId: string; leafId: string } +): Promise { + const deps = { store } as unknown as PtySpawnIpcDeps + const spawnArgs = { + cols: 80, + rows: 24, + worktreeId: WORKTREE, + tabId: TAB, + leafId: args.leafId, + connectionId: TARGET + } as unknown as PtySpawnIpcArgs + const ctx = createPtyIpcSpawnState(deps, spawnArgs) + ctx.result = { id: toAppSshPtyId(TARGET, args.relayPtyId) } + ctx.validatedLeafId = args.leafId + await persistPtyIpcSpawnCommit(ctx) +} + +/** One pane's layout, so the two host partitions can be given different bindings for one leaf. */ +function sessionBinding(ptyId: string) { + return { + activeRepoId: 'repo1', + activeWorktreeId: WORKTREE, + activeTabId: TAB, + tabsByWorktree: {}, + terminalLayoutsByTabId: { + [TAB]: { + root: { type: 'leaf' as const, leafId: TEST_LEAF_1 }, + activeLeafId: TEST_LEAF_1, + expandedLeafId: null, + ptyIdsByLeafId: { [TEST_LEAF_1]: ptyId } + } + } + } +} + +function bulkReattachPtyIds(store: ReturnType): string[] { + return store + .getSshRemotePtyLeases(TARGET) + .filter(sshRemotePtyLeaseAllowsReattach) + .map((lease) => lease.ptyId) + .sort() +} + +describe('the IPC spawn commit keeps one reattachable lease per SSH pane', () => { + beforeEach(() => { + testState.dir = mkdtempSync(join(tmpdir(), 'orca-test-')) + }) + afterEach(() => { + rmSync(testState.dir, { recursive: true, force: true }) + }) + + // QA's measurement, driven through the real path: five relay restarts, one pane, N+1 leases. + it('holds the reattach set flat across five reconnects of one pane', async () => { + const store = await createStore() + + for (let reconnect = 0; reconnect < 5; reconnect++) { + // A relay renumbers from `pty-1` on every start; a reconnect therefore re-leases the same + // pane under an id it has never used before. + await commitSshSpawn(store, { relayPtyId: `pty-${reconnect}`, leafId: TEST_LEAF_1 }) + } + + expect(bulkReattachPtyIds(store)).toEqual(['pty-4']) + }) + + it('retires each predecessor as `expired` with the winner recorded, never `terminated`', async () => { + const store = await createStore() + + await commitSshSpawn(store, { relayPtyId: 'pty-0', leafId: TEST_LEAF_1 }) + await commitSshSpawn(store, { relayPtyId: 'pty-1', leafId: TEST_LEAF_1 }) + + const predecessor = store.getSshRemotePtyLeases(TARGET).find((entry) => entry.ptyId === 'pty-0') + // `expired`, not `terminated`: losing the lease is not evidence the remote shell died, and the + // process is deliberately left running (docs/reference/ssh-execution-boundary.md). + expect(predecessor).toMatchObject({ state: 'expired', supersededBy: 'pty-1' }) + }) + + // The failure that would be worse than the fan-out: over-superseding strands a live remote + // process behind a pane that can no longer find it. + it('leaves a genuine orphan reattachable while superseding the pane that re-leased', async () => { + const store = await createStore() + + await commitSshSpawn(store, { relayPtyId: 'orphan-pty', leafId: TEST_LEAF_2 }) + // The orphan's client lost its route; nothing observed the shell, so it stays askable. + store.markSshRemotePtyLease(TARGET, 'orphan-pty', 'expired') + + await commitSshSpawn(store, { relayPtyId: 'pty-0', leafId: TEST_LEAF_1 }) + await commitSshSpawn(store, { relayPtyId: 'pty-1', leafId: TEST_LEAF_1 }) + + const orphan = store.getSshRemotePtyLeases(TARGET).find((entry) => entry.ptyId === 'orphan-pty') + expect(orphan?.supersededBy).toBeUndefined() + expect(bulkReattachPtyIds(store)).toEqual(['orphan-pty', 'pty-1']) + }) + + /** + * The shape the Docker lane exposed, and the reason a spawn-time trigger is not enough on its + * own. When the spawn commit writes no binding, the renderer's debounced layout publish does it + * later — so at commit time the pane still names the predecessor and supersession correctly + * declines. Nothing revisited it afterwards, and the predecessor stayed reattachable forever. + * + * Measured rows agreed on target, worktree, tab and leaf and still carried no `supersededBy`. + */ + it('retires a predecessor whose successor bound the pane after the spawn commit', async () => { + const store = await createStore() + + await commitSshSpawn(store, { relayPtyId: 'pty2:aaa:1', leafId: TEST_LEAF_1 }) + // What `handlePtyReattachFailure` writes when a restarted relay disowns the id. + store.markSshRemotePtyLease(TARGET, 'pty2:aaa:1', 'expired') + + // The successor leases without binding the pane; the binding catches up afterwards, exactly as + // the renderer's debounced publish does. + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'pty2:bbb:1', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'attached' + }) + expect(bulkReattachPtyIds(store)).toEqual(['pty2:aaa:1', 'pty2:bbb:1']) + store.persistPtyBinding({ + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + ptyId: toAppSshPtyId(TARGET, 'pty2:bbb:1') + }) + + // What the connect path does before reading the set it feeds to `pty.attach`. + store.reconcileSshRemotePtyLeasesForTarget(TARGET) + + expect(bulkReattachPtyIds(store)).toEqual(['pty2:bbb:1']) + }) + + // Reconciliation must not invent evidence: with no binding naming the pane, nothing says which + // shell owns it, so every lease stays askable. + it('leaves leases reattachable when no binding names the pane', async () => { + const store = await createStore() + + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'unbound-a', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'expired' + }) + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'unbound-b', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'expired' + }) + + store.reconcileSshRemotePtyLeasesForTarget(TARGET) + + expect(bulkReattachPtyIds(store)).toEqual(['unbound-a', 'unbound-b']) + }) + + /** + * The measured defect, reduced to its cause. + * + * Main writes an SSH pane's binding to the `ssh:` partition, but a stale copy of the same + * leaf survives in `local`. Reading `local` first named the PREDECESSOR as the pane's current + * PTY, so supersession took an already-expired lease as its winner and returned having marked + * nothing — once per relay restart, forever. Both partitions name the same PTY again once the + * renderer republishes, which is why the finished store looks consistent and hides this. + */ + it('supersedes when the local partition still names the predecessor', async () => { + const store = await createStore() + const hostId = toSshExecutionHostId(TARGET) + const predecessor = toAppSshPtyId(TARGET, 'pty2:old:1') + const successor = toAppSshPtyId(TARGET, 'pty2:new:1') + + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'pty2:old:1', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'expired' + }) + // Both partitions start on the predecessor, as they do before a relay restart. + store.setWorkspaceSession(sessionBinding(predecessor)) + store.setWorkspaceSession(sessionBinding(predecessor), hostId) + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'pty2:new:1', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'attached' + }) + // Production's writer for an SSH pane binding, and the whole point: it updates ONLY the host + // partition, so `local` is left naming the predecessor until the renderer republishes. + store.persistPtyBinding( + { worktreeId: WORKTREE, tabId: TAB, leafId: TEST_LEAF_1, ptyId: successor }, + hostId + ) + expect( + store.getWorkspaceSession().terminalLayoutsByTabId?.[TAB]?.ptyIdsByLeafId?.[TEST_LEAF_1] + ).toBe(predecessor) + + store.supersedeSshRemotePtyLeasesForBoundPane(TARGET, TEST_LEAF_1) + + const retired = store.getSshRemotePtyLeases(TARGET).find((l) => l.ptyId === 'pty2:old:1') + expect(retired).toMatchObject({ state: 'expired', supersededBy: 'pty2:new:1' }) + expect(bulkReattachPtyIds(store)).toEqual(['pty2:new:1']) + }) + + // The mirror: a live shell the pane is still bound to must never be retired, whichever partition + // names it. Over-superseding strands a running remote process. + it('never retires a live lease the pane is still bound to', async () => { + const store = await createStore() + const live = toAppSshPtyId(TARGET, 'pty2:live:1') + + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'pty2:live:1', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'attached' + }) + // Bound BEFORE the stray lease arrives, which is the order that makes the binding meaningful: + // with no binding at all, an arriving lease is the only evidence there is and does win. + store.setWorkspaceSession(sessionBinding(live)) + store.setWorkspaceSession(sessionBinding(live), toSshExecutionHostId(TARGET)) + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: 'pty2:other:1', + worktreeId: WORKTREE, + tabId: TAB, + leafId: TEST_LEAF_1, + state: 'attached' + }) + + store.supersedeSshRemotePtyLeasesForBoundPane(TARGET, TEST_LEAF_1) + + const stillLive = store.getSshRemotePtyLeases(TARGET).find((l) => l.ptyId === 'pty2:live:1') + expect(stillLive).toMatchObject({ state: 'attached' }) + expect(stillLive?.supersededBy).toBeUndefined() + }) + + // Panes are independent, and supersession keys on the leaf: a second live pane on the same + // target must survive its neighbour reconnecting. + it('does not touch a sibling pane on the same target', async () => { + const store = await createStore() + + await commitSshSpawn(store, { relayPtyId: 'sibling-pty', leafId: TEST_LEAF_2 }) + await commitSshSpawn(store, { relayPtyId: 'pty-0', leafId: TEST_LEAF_1 }) + await commitSshSpawn(store, { relayPtyId: 'pty-1', leafId: TEST_LEAF_1 }) + + expect(bulkReattachPtyIds(store)).toEqual(['pty-1', 'sibling-pty']) + }) +}) diff --git a/src/main/ipc/pty/pane/ssh-pane-lease-claim.ts b/src/main/ipc/pty/pane/ssh-pane-lease-claim.ts new file mode 100644 index 00000000000..5746814a5f7 --- /dev/null +++ b/src/main/ipc/pty/pane/ssh-pane-lease-claim.ts @@ -0,0 +1,44 @@ +import { isTerminalLeafId } from '../../../../shared/stable-pane-id' +import { getRelayPtyId } from '../provider/registry' +import type { Store } from '../../../persistence' + +/** + * Claim a remote PTY for a pane: record the lease, then retire the pane's predecessors. + * + * The lease keeps the RELAY id, because reconnect calls `pty.attach` with target-local ids, while + * the pane binding keeps the app-facing id used for hydration. + * + * Supersession is a second step rather than something `upsertSshRemotePtyLease` finishes on its own + * because it is fenced on the pane's durable binding — it refuses to retire a predecessor the pane + * is still bound to, which would detach a live pane. A caller that leases BEFORE it binds therefore + * trips that fence on every reconnect and, with the upsert as the only trigger, never re-runs: one + * more reattachable lease, and one more `pty.attach` round trip on every later connect, forever. + * Re-running it here from the binding side is what makes the two writes commute. + */ +export function claimSshPaneLease(args: { + store: Store | undefined + connectionId: string | null | undefined + ptyId: string + worktreeId: string | undefined + tabId: string | undefined + leafId: string | undefined +}): void { + const { store, connectionId } = args + if (!store || !connectionId) { + return + } + const leafId = + typeof args.leafId === 'string' && isTerminalLeafId(args.leafId) ? args.leafId : null + store.upsertSshRemotePtyLease({ + targetId: connectionId, + ptyId: getRelayPtyId(connectionId, args.ptyId), + ...(typeof args.worktreeId === 'string' ? { worktreeId: args.worktreeId } : {}), + ...(typeof args.tabId === 'string' ? { tabId: args.tabId } : {}), + ...(leafId ? { leafId } : {}), + state: 'attached', + lastAttachedAt: Date.now() + }) + if (leafId) { + store.supersedeSshRemotePtyLeasesForBoundPane(connectionId, leafId) + } +} diff --git a/src/main/ipc/pty/runtime/spawn-commit.ts b/src/main/ipc/pty/runtime/spawn-commit.ts index 0b4e8804ac0..23592604bea 100644 --- a/src/main/ipc/pty/runtime/spawn-commit.ts +++ b/src/main/ipc/pty/runtime/spawn-commit.ts @@ -1,8 +1,6 @@ import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' -import { isTerminalLeafId } from '../../../../shared/stable-pane-id' import { ptyOwnership, ptyIncarnationById, deletePtyOwnership } from '../provider/ownership-state' import { ptySizes } from '../delivery/visibility-state' -import { getRelayPtyId } from '../provider/registry' import { shouldSkipCodexHomeEnvForWindowsShell, recordCodexPaneAccountForSpawn, @@ -25,6 +23,7 @@ import { requestKindSchema } from '../../../../shared/telemetry-events' import { persistAdmittedStablePaneBinding } from '../pane/stable-owner' +import { claimSshPaneLease } from '../pane/ssh-pane-lease-claim' import { isNativeWindowsLocalPtySpawn, markNativeWindowsConptyPty @@ -114,23 +113,15 @@ export async function commitRuntimePtySpawn(ctx: RuntimePtySpawnState) { ) { markNativeWindowsConptyPty(ctx.result.id) } - const persistSshLease = (): void => { - if (!ctx.deps.store || !args.connectionId) { - return - } - // Why: SSH leases keep relay ids for remote reconciliation, while session bindings keep app-facing ids for hydration. - ctx.deps.store.upsertSshRemotePtyLease({ - targetId: args.connectionId, - ptyId: getRelayPtyId(args.connectionId, ctx.result.id), - ...(typeof args.worktreeId === 'string' ? { worktreeId: args.worktreeId } : {}), - ...(typeof args.tabId === 'string' ? { tabId: args.tabId } : {}), - ...(typeof args.leafId === 'string' && isTerminalLeafId(args.leafId) - ? { leafId: args.leafId } - : {}), - state: 'attached', - lastAttachedAt: Date.now() + const persistSshLease = (): void => + claimSshPaneLease({ + store: ctx.deps.store, + connectionId: args.connectionId, + ptyId: ctx.result.id, + worktreeId: args.worktreeId, + tabId: args.tabId, + leafId: args.leafId }) - } if (!ctx.hostSessionBinding) { persistSshLease() } diff --git a/src/main/ipc/ssh-ipc-test-harness.ts b/src/main/ipc/ssh-ipc-test-harness.ts index 84802fe83a0..581fcc916ca 100644 --- a/src/main/ipc/ssh-ipc-test-harness.ts +++ b/src/main/ipc/ssh-ipc-test-harness.ts @@ -25,6 +25,7 @@ export type SshLeaseStoreMock = { upsertSshPtyConsumerRecovery: Mock removeSshPtyConsumerRecovery: Mock getSshRemotePtyLeases: Mock + reconcileSshRemotePtyLeasesForTarget: Mock markSshRemotePtyLease: Mock markSshRemotePtyLeases: Mock markSshRemotePtyLeasesAsync: Mock @@ -102,6 +103,7 @@ export function createSshIpcHarness(mocks: SshIpcMocks): SshIpcHarness { upsertSshPtyConsumerRecovery: vi.fn(), removeSshPtyConsumerRecovery: vi.fn(), getSshRemotePtyLeases: vi.fn().mockReturnValue([]), + reconcileSshRemotePtyLeasesForTarget: vi.fn(), markSshRemotePtyLease: vi.fn(), markSshRemotePtyLeases: vi.fn(), markSshRemotePtyLeasesAsync: vi.fn(), diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts index 46db8c4f0c7..f0628cb25e5 100644 --- a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts @@ -1,9 +1,8 @@ -import { toSshExecutionHostId } from '../../../shared/execution-host' import type { PersistedState } from '../../../shared/persisted-state-types' import type { SshRemotePtyLease } from '../../../shared/ssh-types' import { isTerminalLeafId } from '../../../shared/stable-pane-id' -import type { WorkspaceSessionState } from '../../../shared/workspace-session-state-types' import { invalidateLocalWorktreeMetadataPruneInputs } from '../../local-worktree-metadata-prune-gate' +import { supersedeSiblingLeasesForPane } from './ssh-pty-pane-supersession' export type SshPtyLeaseOperations = { state: PersistedState @@ -15,91 +14,6 @@ export type SshPtyLeaseOperations = { flushDurableStateOrThrowAsync: () => Promise } -/** - * The PTY a pane is durably bound to, keyed on the leaf alone — the only remint-stable half of a - * pane key, since `detachTerminalPaneToTab` moves a live pane and leaves its lease naming the tab - * it left. - * - * Reads both partitions deliberately. Main writes some SSH pane bindings to `ssh:` and - * some to `local`, so a reader that consulted one would see "unbound" for a live pane and expire - * its lease. Reading both makes this fence correct whichever partition the binding landed in. - */ -function durablyBoundPtyIdForPane( - operations: SshPtyLeaseOperations, - targetId: string, - leafId: string -): string | undefined { - const findLeafBinding = (session: WorkspaceSessionState | undefined): string | undefined => - Object.values(session?.terminalLayoutsByTabId ?? {}).find( - (layout) => layout?.ptyIdsByLeafId?.[leafId] - )?.ptyIdsByLeafId?.[leafId] - const boundPtyId = - findLeafBinding(operations.state.workspaceSession) ?? - findLeafBinding(operations.state.workspaceSessionsByHostId?.[toSshExecutionHostId(targetId)]) - return boundPtyId ? operations.toComparablePtyId(targetId, boundPtyId) : undefined -} - -/** - * One pane owns at most one live remote PTY. Lease identity is `(targetId, ptyId)` alone, so a - * pane re-leasing under a new relay id leaves its predecessor live with nothing to retire it and - * the next reattach fans out over both — the reported 2 -> 19 -> 20 across three reconnects. - * - * Superseded leases are marked `expired`, never `terminated`: losing a lease is not evidence the - * shell died, so the remote process is deliberately left running. They also carry `supersededBy`, - * which is what keeps them out of the bulk reattach set now that plain `expired` no longer does — - * the winner's ptyId is already in hand here, so recording it needs no relay-start identity. - */ -function supersedeSiblingLeasesForPane( - operations: SshPtyLeaseOperations, - winner: SshRemotePtyLease, - now: number -): void { - if (!winner.worktreeId || !winner.leafId) { - return - } - if (winner.state === 'terminated' || winner.state === 'expired') { - return - } - // At upsert time the arriving lease may not be the one the pane is bound to yet. Expiring the - // bound predecessor would detach a live pane, so leave both live and let reattach arbitrate - // with the binding in hand. - const boundPtyId = durablyBoundPtyIdForPane(operations, winner.targetId, winner.leafId) - if (boundPtyId && boundPtyId !== winner.ptyId) { - return - } - const superseded: SshRemotePtyLease[] = [] - for (const lease of operations.state.sshRemotePtyLeases ?? []) { - if ( - lease.ptyId === winner.ptyId || - lease.targetId !== winner.targetId || - lease.worktreeId !== winner.worktreeId || - // Leaf only: a lease freezes its tabId, so a pane broken out into a new tab would otherwise - // never compete with its own predecessor — which is the reported cardinality growth. - lease.leafId !== winner.leafId || - lease.state === 'terminated' - ) { - continue - } - if (lease.state === 'expired') { - // An already-expired predecessor is superseded by the same evidence, and marking it is what - // bounds the reattach set: without this, every past orphan for this pane stays reattachable - // forever. `updatedAt` stays put — bumping it would make a stale lease look recent to - // `getRecentExpiredSshLease`. - lease.supersededBy = winner.ptyId - continue - } - lease.state = 'expired' - lease.supersededBy = winner.ptyId - lease.updatedAt = now - superseded.push(lease) - } - if (superseded.length > 0) { - // Why: matching on lease ptyId first means this scrubs only the predecessor's stale binding — - // the winner's own binding cannot match and is left intact. - operations.clearBindingsForLeases(winner.targetId, superseded) - } -} - /** * Only `terminated` unbinds a pane. It is the operator-close state and the one written after a * host-acknowledged stop; `expired` records that the CLIENT lost its route and says nothing about diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-pane-supersession.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-pane-supersession.ts new file mode 100644 index 00000000000..61d6b934db5 --- /dev/null +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-pane-supersession.ts @@ -0,0 +1,201 @@ +import { toSshExecutionHostId } from '../../../shared/execution-host' +import type { SshRemotePtyLease } from '../../../shared/ssh-types' +import { isTerminalLeafId } from '../../../shared/stable-pane-id' +import type { WorkspaceSessionState } from '../../../shared/workspace-session-state-types' +import type { SshPtyLeaseOperations } from './ssh-pty-lease-operations' + +/** + * Every PTY id any partition binds to this pane, most authoritative first. + * + * Keyed on the leaf alone — the only remint-stable half of a pane key, since + * `detachTerminalPaneToTab` moves a live pane and leaves its lease naming the tab it left. + * + * Returns a LIST, and reads the target's own partition first, because the two partitions disagree + * for the length of a reconnect and this resolved that disagreement backwards. Main writes an SSH + * pane's binding to `ssh:`, while a stale copy of the same leaf survives in `local`; + * consulting `local` first therefore named the PREDECESSOR as the pane's current PTY on every relay + * restart. Supersession then took that expired predecessor as its winner and returned without + * marking anything — the per-reconnect lease growth. Both partitions are still read, because a + * reader that consulted only one would see "unbound" for a live pane and expire its lease. + */ +function durablyBoundPtyIdsForPane( + operations: SshPtyLeaseOperations, + targetId: string, + leafId: string +): string[] { + const findLeafBindings = (session: WorkspaceSessionState | undefined): string[] => + Object.values(session?.terminalLayoutsByTabId ?? {}) + .map((layout) => layout?.ptyIdsByLeafId?.[leafId]) + .filter((ptyId): ptyId is string => Boolean(ptyId)) + const ordered = [ + ...findLeafBindings( + operations.state.workspaceSessionsByHostId?.[toSshExecutionHostId(targetId)] + ), + ...findLeafBindings(operations.state.workspaceSession) + ] + return [...new Set(ordered.map((ptyId) => operations.toComparablePtyId(targetId, ptyId)))] +} + +/** A lease this client still holds a route to, as opposed to one it has already lost. */ +function isLiveLeaseState(state: SshRemotePtyLease['state']): boolean { + return state === 'attached' || state === 'detached' +} + +/** + * One pane owns at most one live remote PTY. Lease identity is `(targetId, ptyId)` alone, so a + * pane re-leasing under a new relay id leaves its predecessor live with nothing to retire it and + * the next reattach fans out over both — the reported 2 -> 19 -> 20 across three reconnects. + * + * Superseded leases are marked `expired`, never `terminated`: losing a lease is not evidence the + * shell died, so the remote process is deliberately left running. They also carry `supersededBy`, + * which is what keeps them out of the bulk reattach set now that plain `expired` no longer does — + * the winner's ptyId is already in hand here, so recording it needs no relay-start identity. + */ +export function supersedeSiblingLeasesForPane( + operations: SshPtyLeaseOperations, + winner: SshRemotePtyLease, + now: number +): boolean { + if (!winner.worktreeId || !winner.leafId) { + return false + } + if (winner.state === 'terminated' || winner.state === 'expired') { + return false + } + // At upsert time the arriving lease may not be the one the pane is bound to yet. Expiring the + // bound predecessor would detach a live pane, so leave both live and let reattach arbitrate + // with the binding in hand. `supersedeSshRemotePtyLeasesForBoundPane` re-runs this once the + // binding write lands, so a caller that upserts before it binds is not left bailed forever. + // Membership rather than equality: during a reconnect the two partitions name different PTYs for + // the same leaf, and requiring the winner to match the FIRST one read is what made this bail. + const boundPtyIds = durablyBoundPtyIdsForPane(operations, winner.targetId, winner.leafId) + if (boundPtyIds.length > 0 && !boundPtyIds.includes(winner.ptyId)) { + return false + } + let marked = false + const superseded: SshRemotePtyLease[] = [] + for (const lease of operations.state.sshRemotePtyLeases ?? []) { + if ( + lease.ptyId === winner.ptyId || + lease.targetId !== winner.targetId || + lease.worktreeId !== winner.worktreeId || + // Leaf only: a lease freezes its tabId, so a pane broken out into a new tab would otherwise + // never compete with its own predecessor — which is the reported cardinality growth. + lease.leafId !== winner.leafId || + lease.state === 'terminated' || + // Never retire a shell the pane is BOTH still bound to and still routable to. The stale + // partition can name a predecessor, and retiring that is the point; retiring a live one + // would strand a running remote process behind a pane that can no longer reach it. + (boundPtyIds.includes(lease.ptyId) && isLiveLeaseState(lease.state)) + ) { + continue + } + if (lease.state === 'expired') { + // An already-expired predecessor is superseded by the same evidence, and marking it is what + // bounds the reattach set: without this, every past orphan for this pane stays reattachable + // forever. `updatedAt` stays put — bumping it would make a stale lease look recent to + // `getRecentExpiredSshLease`. + marked ||= lease.supersededBy !== winner.ptyId + lease.supersededBy = winner.ptyId + continue + } + lease.state = 'expired' + lease.supersededBy = winner.ptyId + lease.updatedAt = now + marked = true + superseded.push(lease) + } + if (superseded.length > 0) { + // Why: matching on lease ptyId first means this scrubs only the predecessor's stale binding — + // the winner's own binding cannot match and is left intact. + operations.clearBindingsForLeases(winner.targetId, superseded) + } + return marked +} + +/** + * Supersede from the lease the pane's binding names — preferring a LIVE one when the partitions + * disagree, since a reconnect leaves the stale partition naming an already-expired predecessor and + * an expired winner supersedes nothing. + */ +function supersedeFromBoundPane( + operations: SshPtyLeaseOperations, + targetId: string, + leafId: string, + now: number +): boolean { + if (!isTerminalLeafId(leafId)) { + return false + } + const boundPtyIds = durablyBoundPtyIdsForPane(operations, targetId, leafId) + if (boundPtyIds.length === 0) { + // No binding names this pane, so nothing here is evidence about which shell owns it. Leaving + // every lease reattachable is the deliberate direction: an orphan must stay askable. + return false + } + const candidates = (operations.state.sshRemotePtyLeases ?? []).filter( + (lease) => + lease.targetId === targetId && lease.leafId === leafId && boundPtyIds.includes(lease.ptyId) + ) + const winner = candidates.find((lease) => isLiveLeaseState(lease.state)) + const marked = winner ? supersedeSiblingLeasesForPane(operations, winner, now) : false + return marked +} + +/** + * The binding-side trigger for supersession, and the reason the two writes that together claim a + * pane are commutative. + * + * `upsertSshRemotePtyLease` is the only other trigger, and it bails whenever the pane's durable + * binding still names the predecessor. A spawn path that upserts its lease BEFORE it writes the + * binding therefore bails and never re-runs on its own. Re-resolving the winner from the binding + * is safe in the other direction too: it supersedes only from the lease the pane is actually bound + * to, so it can never strand a live orphan. + */ +export function supersedeSshRemotePtyLeasesForBoundPane( + operations: SshPtyLeaseOperations, + targetId: string, + leafId: string +): void { + if (supersedeFromBoundPane(operations, targetId, leafId, Date.now())) { + operations.flush() + } +} + +/** + * Bound the reattach set to one lease per pane, re-derived from each pane's CURRENT binding. + * + * The spawn-side trigger cannot be sufficient alone, and measuring the shipped path is what showed + * it: a pane's binding has several writers — the spawn commit, the relay's reattach bind, and the + * renderer's debounced layout publish — and the last of those lands well after the spawn commit + * that leased the pty. A predecessor that was still bound when its successor was claimed therefore + * keeps its reattachability forever, because nothing revisits it once the binding catches up. The + * observed rows agreed on target, worktree, tab and leaf and still carried no mark. + * + * Running this immediately before the reattach set is read makes the answer independent of which + * writer bound the pane and when. It also repairs stores written by earlier builds, where these + * rows have already accumulated and no spawn-time trigger would ever revisit them. + * + * Panes with no binding are skipped rather than pruned: absence of a binding is not evidence about + * which shell owns the pane, and a genuine orphan has to stay askable + * (docs/reference/ssh-execution-boundary.md). + */ +export function reconcileSshRemotePtyLeasesForTarget( + operations: SshPtyLeaseOperations, + targetId: string +): void { + const leafIds = new Set() + for (const lease of operations.state.sshRemotePtyLeases ?? []) { + if (lease.targetId === targetId && lease.leafId) { + leafIds.add(lease.leafId) + } + } + const now = Date.now() + let changed = false + for (const leafId of leafIds) { + changed = supersedeFromBoundPane(operations, targetId, leafId, now) || changed + } + if (changed) { + operations.flush() + } +} diff --git a/src/main/persistence/loading-store/ssh-lease-recovery-operations.ts b/src/main/persistence/loading-store/ssh-lease-recovery-operations.ts index 8768a47e50a..75aa19ad24b 100644 --- a/src/main/persistence/loading-store/ssh-lease-recovery-operations.ts +++ b/src/main/persistence/loading-store/ssh-lease-recovery-operations.ts @@ -23,6 +23,10 @@ import { type SshPtyLeaseOperations, upsertSshRemotePtyLease as upsertSshRemotePtyLeaseOperation } from '../leasing-ssh-ptys/ssh-pty-lease-operations' +import { + reconcileSshRemotePtyLeasesForTarget as reconcileSshRemotePtyLeasesForTargetOperation, + supersedeSshRemotePtyLeasesForBoundPane as supersedeSshRemotePtyLeasesForBoundPaneOperation +} from '../leasing-ssh-ptys/ssh-pty-pane-supersession' import { getSshPtyConsumerRecovery as getSshPtyConsumerRecoveryOperation, removeSshPtyConsumerRecovery as removeSshPtyConsumerRecoveryOperation, @@ -95,6 +99,28 @@ export class SshLeaseRecoveryOperations { upsertSshRemotePtyLeaseOperation(getSshPtyLeaseOperations(this), lease) } + /** + * Re-run pane supersession from the binding rather than from an arriving lease. Spawn commits + * call this after their binding write so it does not matter whether the lease or the binding + * landed first; see `supersedeSshRemotePtyLeasesForBoundPane`. + */ + supersedeSshRemotePtyLeasesForBoundPane(targetId: string, leafId: string): void { + supersedeSshRemotePtyLeasesForBoundPaneOperation( + getSshPtyLeaseOperations(this), + targetId, + leafId + ) + } + + /** + * Re-derive one reattachable lease per pane from each pane's current binding. Called on the + * connect path immediately before the reattach set is read; see + * `reconcileSshRemotePtyLeasesForTarget`. + */ + reconcileSshRemotePtyLeasesForTarget(targetId: string): void { + reconcileSshRemotePtyLeasesForTargetOperation(getSshPtyLeaseOperations(this), targetId) + } + markSshRemotePtyLeases(targetId: string, state: SshRemotePtyLease['state']): void { markSshRemotePtyLeasesOperation(getSshPtyLeaseOperations(this), targetId, state) } diff --git a/src/main/ssh-reattach-pane-cardinality.test.ts b/src/main/ssh-reattach-pane-cardinality.test.ts index 5827ec1ee68..d7179703360 100644 --- a/src/main/ssh-reattach-pane-cardinality.test.ts +++ b/src/main/ssh-reattach-pane-cardinality.test.ts @@ -82,24 +82,38 @@ function relayReattachBinds( } /** - * Both binding writers land before the lease upsert — spawn asserts that ordering directly, and - * the relay's reattach binds the pane before `markSshRemotePtyLeasesAttachedAsync`. Supersession - * therefore sees a session already naming the arriving shell. + * One reconnect's worth of writes, in the order the spawn commits actually issue them: the lease + * row first — so a force-quit in the renderer's debounce window cannot leave a running remote shell + * with no lease to reattach it — then the binding, then the binding-side supersession trigger. + * + * This suite used to bind BEFORE upserting, an order no caller uses. Under that order supersession + * always saw a session already naming the arriving shell and passed; under production's order it + * bailed on the predecessor's binding every time and never re-ran, so the guard could not catch the + * per-reconnect lease growth it exists to pin. * * Goes through `persistPtyBinding` rather than `setWorkspaceSession` because that is the writer * production uses; a raw session write is reconciled back to the attached lease's PTY by binding * recovery, which would make the fixture disagree with the real flow. */ -function paneBindsTo( +function paneSpawnCommits( store: ReturnType, - args: { tabId: string; leafId: string; ptyId: string } + args: { tabId: string; leafId: string; ptyId: string; leaseTabId?: string } ): void { + store.upsertSshRemotePtyLease({ + targetId: TARGET, + ptyId: args.ptyId, + worktreeId: WORKTREE, + tabId: args.leaseTabId ?? args.tabId, + leafId: args.leafId, + state: 'attached' + }) store.persistPtyBinding({ worktreeId: WORKTREE, tabId: args.tabId, leafId: args.leafId, ptyId: args.ptyId }) + store.supersedeSshRemotePtyLeasesForBoundPane(TARGET, args.leafId) } function liveLeasePtyIds(store: ReturnType): string[] { @@ -300,8 +314,7 @@ describe('STA-3077: one pane keeps at most one live remote lease', () => { const lease = { targetId: TARGET, worktreeId: WORKTREE, tabId: TAB, leafId: TEST_LEAF_1 } store.upsertSshRemotePtyLease({ ...lease, ptyId: 'pty-1', state: 'attached' }) - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) - store.upsertSshRemotePtyLease({ ...lease, ptyId: 'pty-2', state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) expect(liveLeasePtyIds(store)).toEqual(['pty-2']) }) @@ -314,8 +327,7 @@ describe('STA-3077: one pane keeps at most one live remote lease', () => { const lease = { targetId: TARGET, worktreeId: WORKTREE, tabId: TAB, leafId: TEST_LEAF_1 } store.upsertSshRemotePtyLease({ ...lease, ptyId: 'pty-1', state: 'attached' }) - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) - store.upsertSshRemotePtyLease({ ...lease, ptyId: 'pty-2', state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) const predecessor = store.getSshRemotePtyLeases(TARGET).find((entry) => entry.ptyId === 'pty-1') expect(predecessor?.state).toBe('expired') @@ -325,11 +337,9 @@ describe('STA-3077: one pane keeps at most one live remote lease', () => { it('holds the live lease count flat across ten reconnects of one pane', async () => { const store = await createStore() store.setWorkspaceSession(sessionWithPane({ tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-0' })) - const lease = { targetId: TARGET, worktreeId: WORKTREE, tabId: TAB, leafId: TEST_LEAF_1 } for (let reconnect = 0; reconnect < 10; reconnect++) { - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: `pty-${reconnect}` }) - store.upsertSshRemotePtyLease({ ...lease, ptyId: `pty-${reconnect}`, state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: `pty-${reconnect}` }) } expect(liveLeasePtyIds(store)).toEqual(['pty-9']) @@ -349,17 +359,14 @@ describe('STA-3077: one pane keeps at most one live remote lease', () => { state: 'attached' }) - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) // The successor's lease names the tab the pane sits in NOW; the predecessor's still names the // one it was written in. Only the leaf is common, so keying on the tab would stop the two // competing and leave both live — the cardinality growth. - store.upsertSshRemotePtyLease({ - targetId: TARGET, - ptyId: 'pty-2', - worktreeId: WORKTREE, - tabId: OTHER_TAB, + paneSpawnCommits(store, { + tabId: TAB, leafId: TEST_LEAF_1, - state: 'attached' + ptyId: 'pty-2', + leaseTabId: OTHER_TAB }) expect(liveLeasePtyIds(store)).toEqual(['pty-2']) @@ -394,8 +401,7 @@ describe('STA-3077: one pane keeps at most one live remote lease', () => { const lease = { targetId: TARGET, worktreeId: WORKTREE, tabId: TAB, leafId: TEST_LEAF_1 } store.upsertSshRemotePtyLease({ ...lease, ptyId: 'pty-1', state: 'attached' }) - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) - store.upsertSshRemotePtyLease({ ...lease, ptyId: 'pty-2', state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) expect(liveLeasePtyIds(store).sort()).toEqual(['pty-2', 'sibling-pty']) }) @@ -455,8 +461,7 @@ describe('STA-3077: `expired` separates a superseded sibling from an orphan', () it('never bulk-reattaches a superseded sibling', async () => { const store = await storeWithPane('pty-1') - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) - store.upsertSshRemotePtyLease({ ...paneLease, ptyId: 'pty-2', state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) const predecessor = store.getSshRemotePtyLeases(TARGET).find((entry) => entry.ptyId === 'pty-1') expect(predecessor).toMatchObject({ state: 'expired', supersededBy: 'pty-2' }) @@ -469,8 +474,7 @@ describe('STA-3077: `expired` separates a superseded sibling from an orphan', () store.setWorkspaceSession(sessionWithPane({ tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-0' })) for (let reconnect = 0; reconnect < 10; reconnect++) { - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: `pty-${reconnect}` }) - store.upsertSshRemotePtyLease({ ...paneLease, ptyId: `pty-${reconnect}`, state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: `pty-${reconnect}` }) } expect(bulkReattachPtyIds(store)).toEqual(['pty-9']) @@ -496,8 +500,7 @@ describe('STA-3077: `expired` separates a superseded sibling from an orphan', () store.markSshRemotePtyLease(TARGET, 'pty-1', 'expired') const orphanUpdatedAt = store.getSshRemotePtyLeases(TARGET)[0].updatedAt - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) - store.upsertSshRemotePtyLease({ ...paneLease, ptyId: 'pty-2', state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) const predecessor = store.getSshRemotePtyLeases(TARGET).find((entry) => entry.ptyId === 'pty-1') expect(predecessor).toMatchObject({ state: 'expired', supersededBy: 'pty-2' }) @@ -510,8 +513,7 @@ describe('STA-3077: `expired` separates a superseded sibling from an orphan', () // belongs to the lease that lost, never to whatever claims the id next. it('clears the supersession mark when the id is re-upserted as a live lease', async () => { const store = await storeWithPane('pty-1') - paneBindsTo(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) - store.upsertSshRemotePtyLease({ ...paneLease, ptyId: 'pty-2', state: 'attached' }) + paneSpawnCommits(store, { tabId: TAB, leafId: TEST_LEAF_1, ptyId: 'pty-2' }) // A restarted relay hands `pty-1` to a new shell for a different pane. store.upsertSshRemotePtyLease({ diff --git a/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts b/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts index 3201f8709fe..afb365c075c 100644 --- a/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts +++ b/src/main/ssh/ssh-orphan-relay-pty-sweep.test.ts @@ -53,7 +53,8 @@ function createHarness( shutdown } as unknown as IPtyProvider const store = { - getSshRemotePtyLeases: vi.fn().mockReturnValue(leases) + getSshRemotePtyLeases: vi.fn().mockReturnValue(leases), + reconcileSshRemotePtyLeasesForTarget: vi.fn() } as unknown as Store return { provider, store, shutdown } } diff --git a/src/main/ssh/ssh-relay-session-agent-hooks.integration.test.ts b/src/main/ssh/ssh-relay-session-agent-hooks.integration.test.ts index 49fc1e98ccd..76d92e77e9a 100644 --- a/src/main/ssh/ssh-relay-session-agent-hooks.integration.test.ts +++ b/src/main/ssh/ssh-relay-session-agent-hooks.integration.test.ts @@ -157,6 +157,7 @@ function createSession(targetId: string): InstanceType { upsertSshPtyConsumerRecovery: vi.fn(), removeSshPtyConsumerRecovery: vi.fn(), getSshRemotePtyLeases: vi.fn().mockReturnValue([]), + reconcileSshRemotePtyLeasesForTarget: vi.fn(), markSshRemotePtyLease: vi.fn(), markSshRemotePtyLeases: vi.fn(), markSshRemotePtyLeasesAsync: vi.fn(), diff --git a/src/main/ssh/ssh-relay-session-terminal-error.test.ts b/src/main/ssh/ssh-relay-session-terminal-error.test.ts index 9bb14618319..4cfa657e401 100644 --- a/src/main/ssh/ssh-relay-session-terminal-error.test.ts +++ b/src/main/ssh/ssh-relay-session-terminal-error.test.ts @@ -104,6 +104,7 @@ function createMockDeps(): { upsertSshPtyConsumerRecovery: vi.fn(), removeSshPtyConsumerRecovery: vi.fn(), getSshRemotePtyLeases: vi.fn().mockReturnValue([]), + reconcileSshRemotePtyLeasesForTarget: vi.fn(), markSshRemotePtyLease: vi.fn(), markSshRemotePtyLeases: vi.fn(), markSshRemotePtyLeasesAsync: vi.fn(), diff --git a/src/main/ssh/ssh-relay-session-test-fixtures.ts b/src/main/ssh/ssh-relay-session-test-fixtures.ts index efdee31c406..ba0782b3a20 100644 --- a/src/main/ssh/ssh-relay-session-test-fixtures.ts +++ b/src/main/ssh/ssh-relay-session-test-fixtures.ts @@ -21,6 +21,7 @@ export function createMockDeps(): SshRelaySessionTestDeps { upsertSshPtyConsumerRecovery: vi.fn(), removeSshPtyConsumerRecovery: vi.fn(), getSshRemotePtyLeases: vi.fn().mockReturnValue([]), + reconcileSshRemotePtyLeasesForTarget: vi.fn(), getWorkspaceSession: vi.fn(), markSshRemotePtyLease: vi.fn(), markSshRemotePtyLeases: vi.fn(), diff --git a/src/main/ssh/ssh-relay-session.ts b/src/main/ssh/ssh-relay-session.ts index bca3632b83b..99a2ce9fbf9 100644 --- a/src/main/ssh/ssh-relay-session.ts +++ b/src/main/ssh/ssh-relay-session.ts @@ -2322,6 +2322,12 @@ export class SshRelaySession { if (!shouldContinue()) { return } + // Why immediately before the read: a pane's binding is written by several writers, and the + // renderer's debounced layout publish lands long after the spawn commit that leased the pty — + // so a predecessor that was still bound at spawn time never gets marked by a spawn-side + // trigger. Re-deriving from each pane's CURRENT binding here is what actually bounds this set, + // and it repairs stores that already accumulated these rows. + this.store.reconcileSshRemotePtyLeasesForTarget(this.targetId) // Why not `state !== 'expired'`: that state covers both a superseded sibling (re-adopting it is // the 2 -> 19 -> 20 fan-out) and an orphan whose reattach merely lost contact. Only the first // carries a retirement mark, and only it has to be skipped. diff --git a/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts b/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts index 9f8434ed51a..e18336d12c5 100644 --- a/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts +++ b/tests/e2e/ssh-docker-transport-drop-recovery.spec.ts @@ -1,5 +1,9 @@ -import type { Page } from '@playwright/test' +import path from 'node:path' +import { readFileSync } from 'node:fs' +import type { ElectronApplication, Page } from '@playwright/test' import { test, expect } from './helpers/orca-app' +import { DEFAULT_LOCAL_ORCA_PROFILE_ID } from '../../src/shared/orca-profiles' +import { sshRemotePtyLeaseAllowsReattach, type SshRemotePtyLease } from '../../src/shared/ssh-types' import { ensureTerminalVisible, waitForActiveWorktree, waitForSessionReady } from './helpers/store' import { execInTerminal, @@ -49,6 +53,59 @@ async function readSshStatus(orcaPage: Page, targetId: string) { ) } +/** + * Every lease `reattachKnownPtys` would feed to `pty.attach` on the next connect, read from the + * durable store rather than from the renderer — leases are main-owned and never published. + * + * Goes through the shipped `sshRemotePtyLeaseAllowsReattach` predicate so the measurement cannot + * drift from the fan-out it exists to bound. + */ +function readSshLeases(userDataDir: string, targetId: string): SshRemotePtyLease[] { + const dataPath = path.join( + userDataDir, + 'profiles', + DEFAULT_LOCAL_ORCA_PROFILE_ID, + 'orca-data.json' + ) + const parsed = JSON.parse(readFileSync(dataPath, 'utf8')) as { + sshRemotePtyLeases?: SshRemotePtyLease[] + } + return (parsed.sshRemotePtyLeases ?? []).filter((lease) => lease.targetId === targetId) +} + +function readReattachablePtyIds(userDataDir: string, targetId: string): string[] { + return readSshLeases(userDataDir, targetId) + .filter(sshRemotePtyLeaseAllowsReattach) + .map((lease) => lease.ptyId) + .sort() +} + +/** + * Everything a cardinality failure needs to be diagnosable from the report alone. + * + * Worth keeping rather than reducing to a count: when this first failed, the count said only "2", + * and it was the per-row fields that ruled out the obvious causes — the rows agreed on worktree, + * tab and leaf, so the pane identity was never the problem. + */ +function describeSshLeases(userDataDir: string, targetId: string): string { + return JSON.stringify( + readSshLeases(userDataDir, targetId).map((lease) => ({ + ptyId: lease.ptyId, + state: lease.state, + worktreeId: lease.worktreeId, + leafId: lease.leafId, + tabId: lease.tabId, + supersededBy: lease.supersededBy, + relayIdRecycled: lease.relayIdRecycled, + reattachable: sshRemotePtyLeaseAllowsReattach(lease) + })) + ) +} + +function readUserDataDir(electronApp: ElectronApplication): Promise { + return electronApp.evaluate(({ app }) => app.getPath('userData')) +} + /** * Not covered here on purpose: park-then-reveal after a reconnect. ssh-terminal-parking already * covers the park/reveal round trip, and driving a park deterministically from this lane proved @@ -117,7 +174,9 @@ test.describe('SSH transport drop recovery', () => { // run after the flood produces no output within the poll budget. Same shape as #18018 (deaf pane // after a stalled host resumes), and not caused by this spec. Tracked there; the three verdict // assertions around it stay enforced. - test.fixme('stays bounded when a disconnected shell floods its pty', async ({ orcaPage }, testInfo) => { + test.fixme('stays bounded when a disconnected shell floods its pty', async ({ + orcaPage + }, testInfo) => { test.slow() // Timeouts here are deliberately generous: this guards memory, not latency. A 48MB flood plus a // reconnect lands near 60s wall-clock end to end, so a 60s bind timeout was marginal and made @@ -261,6 +320,89 @@ test.describe('SSH transport drop recovery', () => { } }) + /** + * The cardinality half of the same fault, which the verdict test above cannot see: it asserts the + * pane is re-backed, not what the pane's PREVIOUS shells left behind in the store. + * + * A pane re-leases under a new relay pty id on every relay restart, and nothing else retires the + * predecessor. When supersession fails, each generation leaves one more `expired`-but-unsuperseded + * lease that `reattachKnownPtys` still asks about — one extra `pty.attach` round trip on every + * later connect, forever, growing linearly with reconnect count. Measured as leases rather than + * as latency because latency hides the growth until it is already large. + * + * The reattachable set must stay at exactly one per pane. It must not go to zero either: a lease + * wrongly superseded is a running remote shell the pane can no longer find, which is the worse + * failure (docs/reference/ssh-execution-boundary.md). + */ + test('keeps one reattachable lease per pane across repeated relay restarts', async ({ + orcaPage, + electronApp + }, testInfo) => { + test.slow() + let target: DockerSshRelayTarget | null = null + try { + target = startDockerSshRelayTarget(testInfo) + enableDockerSshRelayTargetShellTitle(target) + await waitForSessionReady(orcaPage) + await waitForActiveWorktree(orcaPage) + const remote = await connectDockerSshRelayTarget(orcaPage, target) + await ensureTerminalVisible(orcaPage, 45_000) + await waitForActiveTerminalManager(orcaPage, 60_000) + await waitForActivePanePtyId(orcaPage, 60_000) + + const userDataDir = await readUserDataDir(electronApp) + const generations: string[][] = [] + + for (let generation = 1; generation <= 5; generation++) { + expect( + killDockerSshRelayDaemon(target), + 'no relay process was found to kill' + ).toBeGreaterThan(0) + await expect + .poll(() => readSshStatus(orcaPage, remote.targetId), { + timeout: 120_000, + message: `SSH target never reconnected after relay kill ${generation}` + }) + .toBe('connected') + await waitForActiveTerminalManager(orcaPage, 120_000) + // The pane must be usable again before the count is meaningful: recovery is what mints the + // successor lease that retires the generation before it. + const ptyId = await waitForActivePanePtyId(orcaPage, 120_000) + const marker = `LEASE_GEN_${generation}_${Date.now()}` + await execInTerminal(orcaPage, ptyId, `printf '%s\\n' ${marker}`) + await waitForTerminalOutput(orcaPage, marker, 60_000) + + try { + await expect + .poll(() => readReattachablePtyIds(userDataDir, remote.targetId).length, { + timeout: 60_000 + }) + .toBe(1) + } catch (error) { + // Why re-thrown with the rows: the count alone cannot say WHICH predecessor stayed + // reattachable, and the user-data dir is torn down before the report is read. + throw new Error( + `reattachable lease count never settled at 1 in generation ${generation}; leases: ${describeSshLeases(userDataDir, remote.targetId)}`, + { cause: error } + ) + } + generations.push(readReattachablePtyIds(userDataDir, remote.targetId)) + } + + // Stated as the whole sequence so a regression reports the growth, not just its endpoint — + // the reported shape was 2, 3, 4, 5, 6 across five restarts. + expect( + generations.map((ptyIds) => ptyIds.length), + `reattachable lease count per generation: ${JSON.stringify(generations)}` + ).toEqual([1, 1, 1, 1, 1]) + } finally { + if (target) { + clearDockerSshRelayFaults(target) + cleanupDockerSshRelayTarget(target) + } + } + }) + /** * The third fault shape: silence with the socket still established. `docker pause` freezes the * container, so nothing is closed or reset — the client simply stops hearing from a host that is From dec1a1d788899499ddf0c15217ac685b62c9e662 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:32:48 -0400 Subject: [PATCH 210/398] fix(i18n): ship onboarding integration capability strings in boot catalog Jinwoo-H <73622457+Jinwoo-H@users.noreply.github.com> --- src/renderer/src/i18n/en-runtime-required.json | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/src/renderer/src/i18n/en-runtime-required.json b/src/renderer/src/i18n/en-runtime-required.json index 99b7f1a7870..6f3c49a21e3 100644 --- a/src/renderer/src/i18n/en-runtime-required.json +++ b/src/renderer/src/i18n/en-runtime-required.json @@ -3717,6 +3717,16 @@ "readyOneOne": "1 workspace found, with 1 cleanup suggestion." } } + }, + "onboarding": { + "integrations": { + "capabilities": { + "browseIssues": "Browse GitHub issues and pull requests in the Tasks view without leaving Orca", + "managePullRequests": "Read, comment on, and merge pull requests without leaving Orca", + "reviewStatus": "See issue state, review status, and CI checks on every worktree", + "startWorkspaceFromIssue": "Start a workspace from any GitHub issue or pull request, prefilled with its title and context" + } + } } }, "dashboard": { From 912463c2780237d7f51b254e50dd271684302631 Mon Sep 17 00:00:00 2001 From: Trevin Chow Date: Thu, 3 Sep 2026 17:32:51 -0700 Subject: [PATCH 211/398] docs: document localization workflow tmchow <517103+tmchow@users.noreply.github.com> --- AGENTS.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/AGENTS.md b/AGENTS.md index 8b0156ba6b1..6915b246bcc 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -53,6 +53,18 @@ Orca targets macOS, Linux, and Windows. Keep all platform-dependent behavior beh - **WSL commands**: build argv with `buildWslExecArgs` (always `--exec` — under `--`, `wsl.exe` expands `$name` in every argument and silently rewrites the script), and fence anything whose stdout you parse with `buildWslCapturedLoginShellCommand`, because the interactive login shell prints the distro banner to stdout. See [`docs/reference/wsl-command-execution.md`](./docs/reference/wsl-command-execution.md). - **Linux native modules**: keep the glibc floor at Ubuntu 20.04 / glibc 2.31. A module compiled from source on a newer runner can reference symbol versions absent on the floor and crash the app on startup. See [`docs/reference/linux-glibc-compatibility.md`](./docs/reference/linux-glibc-compatibility.md); packaging fails if a bundled native binary needs newer glibc. +## Localization (i18n) + +All user-facing copy is localized. `src/renderer/src/i18n/locales/en.json` is the source of truth; the `zh`, `ja`, `ko`, and `es` catalogs mirror its keys. Strings reach the UI through `translate('auto.', 'English fallback')` — never hardcode display text. + +When you touch user-facing copy, keep all five catalogs in sync: + +- **New strings** — wrap them in `translate(...)` with an English fallback, run `pnpm sync:localization-catalog` to register the keys in `en.json` and add placeholders to every other locale, then `pnpm bootstrap:-catalog` (e.g. `bootstrap:ja-catalog`) to translate the placeholders. +- **Reworded strings** — changing the value of an existing key updates only `en.json`. The other locales keep the key with its old translation, and **the lint checks will not catch this**: `verify:localization-catalog` enforces key _parity_, not translation _freshness_. Update the same key in `zh/ja/ko/es` by hand, or re-translate it via `bootstrap:-catalog`. +- **Removed strings** — delete the key from _every_ locale; the parity check rejects a key that exists in one catalog but not another. + +Before pushing copy changes, run `pnpm verify:localization-catalog` and `pnpm verify:localization-coverage` (both also run in `pnpm lint`). + ## SSH Use Case All changes must consider the SSH use case. Don't assume local-only execution. Before changing anything that reports on, stops, or lists remote work, follow [`docs/reference/ssh-execution-boundary.md`](./docs/reference/ssh-execution-boundary.md): the execution host owns everything that touches execution, and loss of contact is never evidence of process death — the verdict vocabulary is `live` / `unverifiable` / `exited`, with no synonyms. From 48cb575db58a8e66f63c8e8c375ecce5f144df4a Mon Sep 17 00:00:00 2001 From: hwantage <82494320+hwantage@users.noreply.github.com> Date: Fri, 4 Sep 2026 09:32:55 +0900 Subject: [PATCH 212/398] feat(i18n): localize Orca Account settings and navigation to Korean hwantage <82494320+hwantage@users.noreply.github.com> --- src/renderer/src/i18n/locales/ko.json | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index d05754d176e..f494ff29a58 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -10289,6 +10289,28 @@ "updateServer": "소스 기본값을 구성하려면 이 서버를 업데이트하세요.", "updateServerDefaults": "표시 기본값을 구성하려면 이 서버를 업데이트하세요." }, + "orcaAccount": { + "connected": "연결됨", + "reconnectRequired": "세션이 만료되었습니다. 클라우드 기능을 사용하려면 다시 로그인하세요.", + "unavailable": "이 빌드에서는 Orca 로그인을 사용할 수 없습니다.", + "signedOut": "로그인하여 아티팩트 및 Orca Relay를 비롯한 클라우드 기능으로 Orca를 확장하세요.", + "checking": "계정 상태 확인 중…", + "account": "Orca 계정", + "signOut": "로그아웃", + "signingIn": "로그인 중…", + "signInAgain": "다시 로그인", + "signIn": "Orca 로그인", + "title": "Orca 계정", + "description": "작업을 즉시 공유하고 어디서나 Orca Mobile로 데스크톱에 연결하세요.", + "searchDescription": "아티팩트 및 Orca Relay에서 사용하는 계정에 로그인하거나 로그아웃합니다.", + "benefitsTitle": "계정에 포함된 기능", + "artifactsTitle": "아티팩트 공유", + "artifactsDescription": "HTML 및 Markdown 파일을 게시하고 Orca에서 모든 공유 링크를 관리합니다.", + "relayTitle": "Orca Relay", + "relayDescription": "셀룰러 또는 모든 Wi-Fi 네트워크를 통해 Orca Mobile을 이 데스크톱에 연결합니다.", + "skillsTitle": "스킬 공유", + "skillsDescription": "단일 스킬 또는 전체 세트를 비공개 링크로 공유하고 사용하는 모든 기기에 설치하세요." + }, "automations": { "title": "자동화", "description": "에이전트 작업을 예약하고 사이드바에 자동화 표시 여부를 선택합니다.", From 49d6d35b16c998ee2e135385e6e710405b3148a7 Mon Sep 17 00:00:00 2001 From: foXaCe Date: Fri, 4 Sep 2026 02:32:59 +0200 Subject: [PATCH 213/398] feat(i18n): add French UI locale foXaCe <290678+foXaCe@users.noreply.github.com> --- .gitignore | 1 + config/scripts/bootstrap-locale-catalog.mjs | 5 + config/scripts/locale-translation-policy.mjs | 39 +- src/main/i18n/main-i18n.ts | 1 + src/renderer/src/i18n/i18n.ts | 1 + src/renderer/src/i18n/locales/en.json | 3 +- src/renderer/src/i18n/locales/es.json | 3 +- src/renderer/src/i18n/locales/fr.json | 16601 +++++++++++++++++ src/renderer/src/i18n/locales/ja.json | 3 +- src/renderer/src/i18n/locales/ko.json | 3 +- src/renderer/src/i18n/locales/zh.json | 3 +- src/renderer/src/i18n/supported-languages.ts | 7 +- src/shared/ui-language.test.ts | 4 +- src/shared/ui-language.ts | 5 +- src/shared/ui-locale.test.ts | 16 +- src/shared/ui-locale.ts | 6 +- 16 files changed, 16685 insertions(+), 16 deletions(-) create mode 100644 src/renderer/src/i18n/locales/fr.json diff --git a/.gitignore b/.gitignore index 3fb72a6486a..8be3fc5b6f4 100644 --- a/.gitignore +++ b/.gitignore @@ -158,6 +158,7 @@ src/renderer/src/i18n/locales/.zh-catalog-cache.json src/renderer/src/i18n/locales/.ko-catalog-cache.json src/renderer/src/i18n/locales/.ja-catalog-cache.json src/renderer/src/i18n/locales/.es-catalog-cache.json +src/renderer/src/i18n/locales/.fr-catalog-cache.json # Bench result JSONs are working artifacts tests/tools/benchmarks/results/terminal-pipeline-*.json diff --git a/config/scripts/bootstrap-locale-catalog.mjs b/config/scripts/bootstrap-locale-catalog.mjs index 05739b4da9c..e5c5fb69a81 100644 --- a/config/scripts/bootstrap-locale-catalog.mjs +++ b/config/scripts/bootstrap-locale-catalog.mjs @@ -34,6 +34,11 @@ const LOCALE_CONFIG = { targetLanguage: 'es', displayName: 'Spanish', cacheFile: '.es-catalog-cache.json' + }, + fr: { + targetLanguage: 'fr', + displayName: 'French', + cacheFile: '.fr-catalog-cache.json' } } diff --git a/config/scripts/locale-translation-policy.mjs b/config/scripts/locale-translation-policy.mjs index 9fd4350ee45..cec2ebf63ad 100644 --- a/config/scripts/locale-translation-policy.mjs +++ b/config/scripts/locale-translation-policy.mjs @@ -217,10 +217,41 @@ export const NEVER_TRANSLATE_VALUES = new Set([ ]) export const NATIVE_PICKER_LABELS = { - zh: { chinese: '中文(简体)', korean: '한국어', japanese: '日本語', spanish: 'Español' }, - ko: { chinese: '中文(简体)', korean: '한국어', japanese: '日本語', spanish: 'Español' }, - ja: { chinese: '中文(简体)', korean: '한국어', japanese: '日本語', spanish: 'Español' }, - es: { chinese: '中文(简体)', korean: '한국어', japanese: '日本語', spanish: 'Español' } + zh: { + chinese: '中文(简体)', + korean: '한국어', + japanese: '日本語', + spanish: 'Español', + french: 'Français' + }, + ko: { + chinese: '中文(简体)', + korean: '한국어', + japanese: '日本語', + spanish: 'Español', + french: 'Français' + }, + ja: { + chinese: '中文(简体)', + korean: '한국어', + japanese: '日本語', + spanish: 'Español', + french: 'Français' + }, + es: { + chinese: '中文(简体)', + korean: '한국어', + japanese: '日本語', + spanish: 'Español', + french: 'Français' + }, + fr: { + chinese: '中文(简体)', + korean: '한국어', + japanese: '日本語', + spanish: 'Español', + french: 'Français' + } } const CJK_LATIN_SPACED_TERM_PATTERN = CJK_LATIN_SPACED_TERMS.join('|') diff --git a/src/main/i18n/main-i18n.ts b/src/main/i18n/main-i18n.ts index 8ebe21e0545..4fb548986b4 100644 --- a/src/main/i18n/main-i18n.ts +++ b/src/main/i18n/main-i18n.ts @@ -25,6 +25,7 @@ const LAZY_LOCALE_LOADERS: Record< () => Promise<{ default: Record }> > = { es: () => import('../../renderer/src/i18n/locales/es.json'), + fr: () => import('../../renderer/src/i18n/locales/fr.json'), ja: () => import('../../renderer/src/i18n/locales/ja.json'), ko: () => import('../../renderer/src/i18n/locales/ko.json'), zh: () => import('../../renderer/src/i18n/locales/zh.json') diff --git a/src/renderer/src/i18n/i18n.ts b/src/renderer/src/i18n/i18n.ts index f7fd94f2a55..725a40e9aaa 100644 --- a/src/renderer/src/i18n/i18n.ts +++ b/src/renderer/src/i18n/i18n.ts @@ -26,6 +26,7 @@ const NON_DEFAULT_LOCALE_LOADERS: Record< () => Promise<{ default: Record }> > = { es: () => import('./locales/es.json'), + fr: () => import('./locales/fr.json'), ja: () => import('./locales/ja.json'), ko: () => import('./locales/ko.json'), zh: () => import('./locales/zh.json') diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 57db52974bd..87fcb50a2a1 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -119,7 +119,8 @@ "chinese": "中文(简体)", "korean": "한국어", "japanese": "日本語", - "spanish": "Español" + "spanish": "Español", + "french": "Français" }, "statusBar": { "claudeToggleDescription": "Show Claude token and cost usage for the active workspace.", diff --git a/src/renderer/src/i18n/locales/es.json b/src/renderer/src/i18n/locales/es.json index 0ac5d624ede..9999634daee 100644 --- a/src/renderer/src/i18n/locales/es.json +++ b/src/renderer/src/i18n/locales/es.json @@ -20,7 +20,8 @@ "chinese": "中文(简体)", "korean": "한국어", "japanese": "日本語", - "spanish": "Español" + "spanish": "Español", + "french": "Français" }, "statusBar": { "claudeToggleDescription": "Muestra el consumo de tokens y el costo de Claude para el espacio de trabajo activo.", diff --git a/src/renderer/src/i18n/locales/fr.json b/src/renderer/src/i18n/locales/fr.json new file mode 100644 index 00000000000..ed9307b30db --- /dev/null +++ b/src/renderer/src/i18n/locales/fr.json @@ -0,0 +1,16601 @@ +{ + "app": { + "recoverableError": { + "rootTitle": "Orca a rencontré une erreur de rendu.", + "rootDescription": "Le shell de l'application n'a pas pu terminer son rendu. Réessayez pour le remonter, ou relancez Orca si l'erreur persiste.", + "webTitle": "Orca web a rencontré une erreur de rendu.", + "webDescription": "Réessayez le client web ou reconnectez-vous au runtime appairé." + } + }, + "browser": { + "loadFailure": { + "connectionNotSecure": "La connexion n'est pas sécurisée", + "cantReachHost": "Impossible d'accéder à {{value0}}", + "cantLoadPage": "Impossible de charger cette page", + "certificateNameMismatch": "Le certificat ne correspond pas à {{value0}}.", + "certificateDateInvalid": "Le certificat de {{value0}} n'est pas valide à la date et à l'heure actuelles.", + "certificateAuthorityInvalid": "Orca ne fait pas confiance à l'autorité qui a émis le certificat de {{value0}}.", + "certificateVerificationFailed": "Orca n'a pas pu vérifier le certificat de {{value0}}.", + "trustedCertificateGuidance": "Pour le développement local, utilisez si possible un certificat local de confiance.", + "retry": "Réessayer", + "copyAddress": "Copier l'adresse", + "addressCopied": "Adresse de la page actuelle copiée.", + "openExternally": "Ouvrir en externe", + "tryHttps": "Essayer HTTPS", + "proceedUnsafe": "Continuer quand même (non sécurisé)", + "connecting": "Connexion…", + "certificateChallengeExpired": "Cette approbation de certificat a expiré. Rechargez la page pour en demander une nouvelle.", + "certificateChallengeChanged": "La demande de certificat a changé. Rechargez la page et examinez le nouvel avertissement.", + "certificateChallengeUnavailable": "Cette demande de certificat n'est plus disponible. Rechargez la page pour en demander une nouvelle.", + "certificateProceedFailed": "Orca n'a pas pu approuver cette demande de certificat. Rechargez la page et réessayez." + }, + "guestRecovery": { + "title": "Page du navigateur arrêtée", + "failed": "La page du navigateur s'est arrêtée de manière inattendue. Réessayez pour la restaurer." + } + }, + "githubChecks": { + "retrying": "Nouvelle tentative…" + }, + "settings": { + "appearance": { + "language": { + "title": "Langue", + "description": "Choisissez la langue utilisée par l'interface d'Orca.", + "system": "Système", + "english": "English", + "chinese": "中文(简体)", + "korean": "한국어", + "japanese": "日本語", + "spanish": "Español", + "french": "Français" + }, + "statusBar": { + "claudeToggleDescription": "Afficher l'utilisation des tokens et des coûts de Claude pour l'espace de travail actif.", + "codexToggleDescription": "Afficher l'utilisation des tokens et des coûts de Codex pour l'espace de travail actif.", + "geminiToggleDescription": "Afficher l'utilisation des tokens et des coûts de Gemini pour l'espace de travail actif.", + "opencodeGoToggleDescription": "Afficher l'utilisation des tokens et des coûts d'OpenCode Go pour l'espace de travail actif.", + "kimiToggleDescription": "Afficher l'utilisation de l'abonnement Kimi pour l'espace de travail actif.", + "minimaxToggleDescription": "Afficher l'utilisation de l'abonnement MiniMax pour l'espace de travail actif.", + "sshToggleDescription": "Afficher les hôtes SSH et les hôtes Orca distants configurés lorsqu'ils sont disponibles.", + "resourceUsageToggleDescription": "Afficher le gestionnaire de ressources. Cliquez dessus pour le CPU, la mémoire, les sessions, les contrôles des daemons et les analyses disque des espaces de travail.", + "portsToggleDescription": "Afficher les ports actifs de l'espace de travail. Cliquez dessus pour les ports par espace de travail et les écouteurs externes.", + "antigravityToggleDescription": "Afficher l'utilisation de l'abonnement Antigravity pour l'espace de travail actif.", + "grokToggleDescription": "Afficher l'utilisation des crédits de l'abonnement Grok lorsque vous êtes connecté via Grok CLI." + }, + "menuBarIcon": { + "title": "Afficher l'icône dans la barre de menus", + "description": "Garder un raccourci Orca et un indicateur d'activité dans la barre de menus macOS." + } + } + }, + "menu": { + "checkForUpdates": "Rechercher les mises à jour...", + "settings": "Paramètres", + "exploreOrca": "Découvrir Orca", + "gettingStarted": "Premiers pas avec Orca", + "reportCrash": "Signaler un plantage...", + "file": "Fichier", + "exit": "Quitter", + "edit": "Édition", + "appearance": "Apparence", + "toggleLeftSidebar": "Basculer la barre latérale gauche", + "toggleRightSidebar": "Basculer la barre latérale droite", + "showStatusBar": "Afficher la barre d'état", + "showTasksButton": "Afficher le bouton Tâches", + "showAutomationsButton": "Afficher le bouton Automatisations", + "showMobileButton": "Afficher le bouton Orca Mobile", + "showTitlebarAppName": "Afficher le nom de l'app dans la barre de titre", + "view": "Affichage", + "reload": "Recharger", + "forceReload": "Forcer le rechargement", + "resetSize": "Réinitialiser la taille", + "zoomIn": "Zoom avant", + "zoomOut": "Zoom arrière", + "openWorktreePalette": "Ouvrir la palette des worktrees", + "window": "Fenêtre", + "help": "Aide", + "paste": "Coller", + "copy": "Copier", + "selectAll": "Tout sélectionner" + }, + "tray": { + "openOrca": "Ouvrir Orca", + "quit": "Quitter", + "minimizeNotice": { + "body": "Orca continue de tourner dans la zone de notification" + }, + "activityWaiting": "Orca - activité en attente", + "activityWaitingSuffix": "activité en attente" + }, + "worktreeJumpPalette": { + "matchLabel": { + "comment": "Commentaire", + "issue": "Issue", + "mr": "MR", + "port": "Port", + "pr": "PR", + "task": "Tâche", + "automation": "Exécution" + }, + "linearIssue": { + "createLabel": "Créer un worktree à partir de l'issue Linear {{value0}} : {{value1}}", + "pendingLabel": "Créer un worktree à partir de l'issue Linear {{value0}}", + "loadingLabel": "Chargement de l'issue Linear {{value0}}", + "createHint": "Créer un worktree à partir d'une issue Linear", + "loadingHint": "Chargement de l'issue Linear…" + }, + "renderCapOverflow": "{{value0}} autres — faites défiler ou continuez à taper pour affiner", + "filter": { + "emptyTitle": "Aucun résultat ne correspond au filtre actif", + "emptySubtitle": "Effacez le filtre ci-dessus, ou élargissez-le à davantage d'hôtes et de projets.", + "tabKey": "Tab", + "label": "Filtre", + "activeCount": "{{value0}} actif(s)", + "removeChip": "Supprimer le filtre {{value0}}", + "chipOverflow": "+{{value0}}", + "clearAll": "Tout effacer", + "hosts": "Hôtes", + "projects": "Projets", + "trigger": "Filtrer les résultats", + "search": "Filtrer par hôte ou projet...", + "searchHosts": "Filtrer les hôtes...", + "searchProjects": "Filtrer les projets...", + "noOptions": "Aucun hôte ni projet correspondant", + "noHosts": "Aucun hôte correspondant", + "noProjects": "Aucun projet correspondant", + "moreOptions": "{{value0}} autres - continuez à taper pour affiner", + "selectAllMatching": "Sélectionner toutes les correspondances ({{value0}})", + "selectedCollapsed": "{{value0}} sélectionnés — supprimez-les via les chips ou via Effacer", + "clearHosts": "Effacer les hôtes", + "clearProjects": "Effacer les projets", + "back": "Retour", + "ariaActive": "Filtre : {{value0}} actif(s)." + }, + "taskUrl": { + "loadingHint": "Chargement de {{value0}}…", + "createHint": "Créer un worktree à partir de {{value0}}" + } + }, + "auto": { + "App": { + "221a95ba38": "Relancez l'onboarding, ou fermez-le et continuez dans l'app.", + "f02d37278a": "Une erreur est survenue dans l'onboarding.", + "acd66311dc": "Utilisez le menu Aide après avoir réessayé si vous avez toujours besoin d'un diagnostic.", + "722d03aa62": "La boîte de dialogue de rapport de plantage a rencontré une erreur.", + "8a023cea1f": "Réessayez la barre d'état pour remonter ses contrôles.", + "2e8ff36f94": "La barre d'état a rencontré une erreur.", + "7cbfbf622f": "Réessayez l'espace de travail flottant, ou fermez-le et rouvrez-le.", + "1b3024bcd6": "L'espace de travail flottant a rencontré une erreur.", + "8d1e160ed1": "Réessayez la barre latérale ou changez d'onglet pour recharger cette vue.", + "ed6b168d00": "La barre latérale droite a rencontré une erreur.", + "03a14f6b5b": "Réessayez la page ou naviguez vers une autre vue Orca.", + "b7a714db1e": "Cette page a rencontré une erreur.", + "98d4ea2823": "Le rendu du terminal, du navigateur ou de l'éditeur a échoué dans cet espace de travail. Réessayez pour le remonter.", + "5a9519aef0": "Le workbench de l'espace de travail a rencontré une erreur.", + "cba0fafda5": "La page active reste ouverte. Réessayez la liste ou changez de vue.", + "1468601e7b": "La liste des espaces de travail a rencontré une erreur.", + "bdc71dddc9": "L'espace de travail actif reste ouvert. Réessayez la liste ou changez de vue.", + "c1cf0b0e4a": "Réduire le volet", + "8504ddf267": "L'app tourne toujours. Réessayez le shell ou utilisez le menu pour signaler les détails du plantage.", + "df1d56bf87": "Le shell de l'espace de travail a rencontré une erreur.", + "9e0b441a91": "Basculer la barre latérale droite", + "e81217c1b7": "Masquer le nom de l'app", + "5096cbbc86": "Orca", + "8b0b8eb54f": "Menu de l'application", + "caea5b51b9": "Redémarrer maintenant", + "0a9e810705": "Les modifications ne seront enregistrées qu'après redémarrage. Vos onglets précédents sont en sécurité sur le disque.", + "12e77cf12b": "Échec de la restauration de session", + "332dbfa497": "Espace de travail téléversé", + "e960d18540": "Fermer", + "c9d6f98459": "Agrandir", + "66f0a552e5": "Restaurer", + "bbb7f90669": "Réduire", + "d54e66004c": "terminal", + "9f0152563e": "mobile", + "62ca9895a7": "espace", + "844eb0f4f4": "activité", + "3443924e91": "automatisations", + "4f08ae8311": "tâches", + "ca6c6eece7": "skills", + "1b9d9d065f": "paramètres", + "c184e056de": "Basculer la barre latérale droite ({{value0}})", + "f7aa73e785": "Suivant ({{value0}})", + "cf9099fe98": "Suivant", + "fe21e8f6f5": "Précédent ({{value0}})", + "064bd07810": "Précédent", + "ce37cf5279": "Basculer la barre latérale ({{value0}})", + "e4b9e7dff7": "Basculer la barre latérale", + "pluginCommandFailed": "Impossible d'exécuter la commande du plugin." + }, + "web": { + "WebConnect": { + "b411ec0069": "Se connecter", + "2cf9e5a294": "Effacer le serveur enregistré", + "4a4c017be1": "Point de terminaison :", + "27393856e4": "orca://pair?code=...", + "7a566540de": "URL ou code d'appairage", + "cb4d287238": "Nom du serveur", + "3affe7de3a": "Collez une URL d'appairage provenant d'un serveur Orca accessible depuis ce navigateur.", + "e3bcd082ac": "Se connecter à Orca", + "mobileScopeRejected": "Ce code QR accorde un accès limité (mobile). Pour utiliser l'application web complète, ouvrez le lien d'accès navigateur depuis Paramètres → Environnements d'exécution → Partager ce serveur Orca → Nouveau lien." + }, + "webPreloadApi": { + "aiVaultUnavailableForHost": "L'historique des sessions d'agent n'est pas disponible pour cet hôte d'exécution.", + "runtimeEnvironmentManuallyDisconnected": "L'environnement d'exécution est déconnecté manuellement.", + "loopbackPairingBlocked": "Ce lien d'accès pointe vers cet appareil lui-même.", + "remotePairingUnreachable": "Impossible de joindre Orca à l'adresse {{endpoint}}.", + "remotePairingInvalidDetails": "Ce lien d'accès contient des détails de connexion invalides.", + "remotePairingSaveFailed": "Orca a vérifié l'hôte mais n'a pas pu l'enregistrer. Vérifiez le stockage du navigateur et réessayez." + }, + "web": { + "preload": { + "api": { + "31bfe8ae1a": "Indisponible dans le client web.", + "67ec964791": "L'import de cookies est indisponible dans le client web.", + "275a776357": "L'extraction au survol est indisponible dans le client web.", + "8dfcb7a351": "Les captures de sélection sont indisponibles dans le client web.", + "31bea294d5": "Le mode capture est indisponible dans le client web.", + "b8a1618172": "La génération du détail de pull request est indisponible dans le client web.", + "e57c82d276": "La découverte des modèles de messages de commit est indisponible dans le client web.", + "9fc90740b6": "La génération des messages de commit est indisponible dans le client web.", + "52bee9d8a0": "Des raccourcis personnalisés en conflit ont été ignorés : {{value0}}.", + "32f15bdb0f": "Plateforme inconnue « {{value0}} » ignorée.", + "0a69fcd8bc": "« platforms » doit être un objet avec des sections darwin, linux ou win32.", + "10898045f3": "Raccourci pour « {{value0}} » ignoré : utilisez un tableau de chaînes.", + "36761d9604": "Action de raccourci inconnue « {{value0}} » ignorée.", + "d2e43e426a": "{{value0}} doit être un objet.", + "fb290366b2": "Indisponible sur le web.", + "76122208ca": "Raccourci pour « {{value0}} » ignoré : {{value1}}" + } + }, + "runtime": { + "environment": { + "07f788de83": "WebSocket" + } + } + } + }, + "store": { + "slices": { + "browser": { + "d175274b6d": "Nouvel onglet de navigateur", + "08fc23631d": "Navigateur", + "remoteCookieImportUnavailable": "L'import manuel d'un fichier de cookies est indisponible tant qu'un runtime distant est actif." + }, + "editor": { + "dcb521ed29": "Ce fichier est en état de conflit, mais aucun fichier de l'arbre de travail n'est disponible à l'édition.", + "conflictPlaceholderGuidance": "Résolvez le conflit dans Git ou restaurez l'un des deux côtés avant de le rouvrir.", + "51f15c37d3": "Impossible d'ouvrir le répertoire : {{value0}}", + "f2e00db373": "Fichier introuvable : {{value0}}", + "checkRunDetailsUnavailable": "Aucun détail disponible pour cette vérification.", + "checkRunDetailsLoadFailed": "Échec du chargement des détails de la vérification.", + "checkRunDetailsRepoUnavailable": "Les détails du dépôt sont indisponibles pour cette vérification." + }, + "github": { + "f129c42773": "GitHub n'a pas renvoyé le nouveau commentaire.", + "683a21264b": "La ligne n'a pas de owner/repo/number.", + "83f9b126ad": "Le type d'issue ne peut être défini que sur des issues.", + "f963485d37": "Ligne introuvable", + "a967f23983": "Vue de projet non chargée", + "87020f6605": "La ligne n'a pas de owner/repo/number — impossible de modifier l'élément sous-jacent", + "d49ef4b944": "Échec de l'enregistrement de la préférence de source d'issue" + }, + "sparse": { + "presets": { + "ef13e994e6": "Les préréglages doivent être chargés avant l'enregistrement.", + "6ed7d6010a": "Échec de la suppression du préréglage", + "ee434d7941": "Préréglage supprimé", + "c96b770172": "Échec de l'enregistrement du préréglage", + "811be06b57": "Échec de la mise à jour du préréglage", + "0696d13e56": "Préréglage enregistré", + "e10f097822": "Préréglage mis à jour" + } + }, + "store": { + "test": { + "helpers": { + "b9a8117c33": "Terminal 1" + } + } + }, + "workspace": { + "cleanup": { + "9d6e531da6": "L'espace de travail n'existe plus.", + "changedSinceConfirmation": "L'espace de travail a changé après confirmation. Actualisez pour le passer en revue avant de le supprimer.", + "hostCollision": "Erreur : cet espace de travail existe sur plusieurs hôtes au même chemin", + "hostUnresolved": "Orca ne peut pas déterminer quel hôte possède cet espace de travail. Actualisez les projets et examinez-le à nouveau.", + "gitStatusUnavailable": "Orca n'a pas pu vérifier le statut git de cet espace de travail. Réessayez, ou supprimez-le depuis la barre latérale propre à son hôte ou depuis la liste des projets.", + "gitStatusUnavailableOlderPeer": "Orca n'a pas pu rattacher cet échec de statut git à un hôte. Mettez à jour l'homologue connecté le plus ancien, ou supprimez l'espace de travail depuis la barre latérale propre à son hôte ou depuis la liste des projets.", + "forceNeedsApproval": "Examinez et confirmez cet espace de travail avant de le supprimer de force." + } + }, + "worktrees": { + "5a58e03a26": "« {{value0}} » supprimé.", + "d1d78a7baa": "Git n'a pas pu supprimer la branche « {{value0}} »{{value1}} sans risque, Orca l'a donc conservée pour éviter de perdre des commits locaux.", + "4e6496f3d2": "{{value0}} supprimé, branche conservée", + "e50495aae6": "Forcer la suppression de la branche", + "889487d8bb": "Ignorer", + "f4503ca505": "Ouvrez Paramètres > Git et réessayez.", + "34a03a6565": "Garder {{value0}} à jour", + "fa9299a66f": "Votre nouveau worktree est à jour, mais le {{value0}} local est en retard de {{value1}} {{value2}}. Les diffs IA peuvent omettre des commits récents.", + "14bc053a47": "Le {{value0}} local n'a pas été actualisé", + "localBaseRefRefreshFailedForWorktree": "Le {{value0}} local n'a pas été actualisé pour « {{value1}} »", + "4a18052018": "Le {{value0}} local est en retard sur {{value1}}", + "903b51c2ed": "Espace de travail créé à partir de {{value0}}, mais Orca n'a pas pu avancer le {{value1}} local en fast-forward. {{value2}}", + "localBaseRefRefreshFailedDescriptionNamed": "L'espace de travail « {{value0}} » a été créé à partir de {{value1}}, mais Orca n'a pas pu avancer le {{value2}} local en fast-forward. {{value3}}", + "localBaseRefRefreshFailedDetailDirtyNamed": "Le worktree situé dans {{value0}} (où le {{value1}} local est en checkout) contient des changements non commités. Committez-les, stashez-les ou abandonnez-les, puis mettez à jour le {{value1}} local manuellement.", + "localBaseRefRefreshFailedDetailDirty": "Le worktree où le {{value0}} local est en checkout contient des changements non commités. Committez-les, stashez-les ou abandonnez-les, puis mettez à jour le {{value0}} local manuellement.", + "localBaseRefRefreshFailedDetailNotFastForward": "Le {{value0}} local n'existe pas ou ne peut pas être avancé proprement en fast-forward depuis la base distante. Vérifiez les commits présents uniquement en local avant de le mettre à jour manuellement.", + "localBaseRefRefreshFailedDetailError": "Git a renvoyé une erreur lors de la mise à jour du {{value0}} local. Vérifiez que le dépôt n'a pas de refs verrouillées ni d'état de worktree inhabituel, puis mettez à jour le {{value0}} local manuellement.", + "0216895fb5": "Échec de la suppression de la branche", + "c6cf133786": "Cet espace de travail n'est plus disponible.", + "19db0085fb": "Branche locale supprimée", + "2b0afc7f14": "Impossible de garder le {{value0}} local à jour", + "670864ab52": "Mise à jour du {{value0}} local en cours", + "5366d13eec": "Espace de travail supprimé, branche conservée", + "2e17f825d4": "Worktree supprimé, branche conservée", + "78e08cd877": "Git n'a pas pu supprimer la branche « {{value0}} » sans risque, Orca l'a donc conservée pour éviter de perdre des commits locaux.", + "3b57982bf6": "Git n'a pas pu supprimer la branche « {{value0}} » sans risque après la suppression de l'espace de travail « {{value1}} », Orca l'a donc conservée pour éviter de perdre des commits locaux.", + "81f13f48d2": "Git n'a pas pu supprimer la branche « {{value0}} » sans risque après la suppression du worktree « {{value1}} », Orca l'a donc conservée pour éviter de perdre des commits locaux.", + "runtimeScopeForbiddenTitle": "Cette connexion a un accès limité (mobile)", + "runtimeScopeForbiddenDescription": "Les espaces de travail sont indisponibles sur un appairage à périmètre mobile. Reconnectez-vous via le lien d'accès navigateur depuis Paramètres → Environnements d'exécution → Partager ce serveur Orca.", + "a17f4d2e93": "Impossible de mettre à jour cet espace de travail.", + "preservedBranchCleanupHostAmbiguous": "Plusieurs nettoyages de branches conservées sont en attente pour « {{value0}} » ; précisez l'hôte.", + "metadata": { + "worktree": { + "meta": { + "persist": { + "877e3638d8": "Mettez à jour le runtime distant pour modifier l'issue liée à cet espace de travail", + "4367540861": "Mettez à jour le runtime distant pour lier des issues Linear" + } + } + } + } + }, + "repos": { + "2975400634": "Mettez à jour le serveur Orca pour ouvrir des dossiers non Git sur ce runtime.", + "b7e14472ae": "Échec de l'ajout du dossier", + "e649269645": "Utilisez Ajouter un projet pour saisir un chemin sur l'hôte sélectionné.", + "c6e022ddfc": "Échec de l'ajout du projet", + "90d129b48b": "Dossier ajouté", + "8bb3ad7935": "Projet ajouté", + "a8e4b3af5b": "Projet déjà ajouté", + "6d3318e813": "Échec de l'import des dépôts", + "3be0f7df04": "Impossible d'ouvrir le dossier sur le runtime sélectionné", + "15cf5319ec": "{{path}} a été vérifié sur {{hostName}}, mais cet hôte n'a pas signalé de dossier exploitable.", + "2dcd706774": "Le projet existe aussi dans un autre profil", + "presenceProfileOverflow": "{{names}} +{{count}} autres", + "removeProjectFailed": "Échec de la suppression du projet" + }, + "settings": { + "e12dab333b": "Échec du changement de serveur", + "faa8fb83dd": "Enregistrez ou fermez les onglets d'éditeur non enregistrés avant de changer de serveur." + }, + "ui": { + "66e3bd7ce6": "Envoyé à {{value0}}", + "53883b7bc3": "Impossible d'envoyer à {{value0}}" + }, + "jira": { + "856083302c": "La connexion Jira a été remplacée par une requête plus récente." + }, + "linear": { + "37d36984d0": "La connexion Linear a été remplacée par une requête plus récente." + }, + "orca": { + "profiles": { + "612f7f6861": "Échec de la création du profil", + "319d7cf39b": "Profil cloud créé", + "d6e764e7db": "Reconnecter ce profil", + "f0c9e11a6d": "Échec de la création du profil cloud", + "8b8fa73174": "La connexion à Orca Cloud n'est pas configurée", + "33290e88ed": "Échec de la connexion du profil", + "9fcb07a796": "Profil connecté", + "2f6c78a039": "Échec de l'actualisation de l'authentification du profil", + "a37b5e6d37": "Déconnecté du profil", + "83600521e7": "Échec de la déconnexion", + "76deec8f58": "Échec du changement d'organisation", + "7d4bc516ee": "Échec du changement de profil", + "f518e89aa5": "Le projet existe déjà dans ce profil", + "f03ae7f27b": "Échec du transfert du projet" + } + }, + "runtime": { + "status": { + "runtimeHostDisconnectedDescription": "Vérifiez qu'Orca tourne sur ce serveur et que votre connexion réseau fonctionne, puis réessayez.", + "runtimeHostUnreachableNamed": "Impossible de joindre {{hostName}}", + "runtimeHostUnreachable": "Impossible de joindre le serveur Orca", + "tryAgain": "Réessayer" + } + }, + "terminal": { + "quick": { + "command": { + "hosts": { + "5b7d781d67": "Échec de l'enregistrement de la commande rapide" + } + } + } + } + } + }, + "lib": { + "agent": { + "catalog": { + "5dff448636": "OpenClaw", + "8a9ba743cc": "Hermes", + "4e63c7b956": "Rovo Dev", + "bee242fe3d": "Qwen Code", + "ca73055bd0": "Mistral Vibe", + "28810273af": "Kimi", + "739a930554": "Droid", + "667c104cff": "Cursor", + "9e2a9bb87b": "Continuer", + "6f8056a565": "Command Code", + "4238b771b5": "Codebuff", + "cbaf0c2e0b": "Cline", + "1f8a19e9ad": "Autohand Code", + "5e8eff11b3": "Auggie", + "9477377a2a": "Charm", + "e0247254f2": "Kiro", + "918ba4ffed": "Kilocode", + "c73c573939": "Amp", + "8da11d876c": "Goose", + "b32627f09b": "Aider", + "691dd11789": "Antigravity", + "12e6baa4f7": "Gemini", + "09973b4d84": "OMP", + "302934c5d9": "Pi", + "e7a4ca5103": "OpenCode", + "706b0fe68b": "GitHub Copilot", + "0baad2d5d2": "Grok", + "760bc6883d": "Codex", + "a5fc0cb622": "OpenClaude", + "bf53f09bf8": "Claude Agent Teams", + "0708ed89f1": "Claude", + "fc80296033": "Devin", + "da41abbdd4": "Ante", + "060d152fb5": "Trae", + "d443a47995": "Prime Agent", + "mimo_code_label": "MiMo Code" + }, + "skill": { + "cli": { + "prerequisite": { + "79371593b0": "Orca CLI n'est pas encore visible dans le PATH", + "e99d7dc36f": "L'enregistrement d'Orca CLI demande votre attention", + "2db0bd7515": "L'enregistrement d'Orca CLI est indisponible", + "8d6eedf97e": "Échec de l'enregistrement d'Orca CLI dans le PATH.", + "0f116999f1": "Redémarrez votre shell ou ajoutez le répertoire d'Orca CLI au PATH avant la configuration.", + "15cbedc3e3": "Installez Orca CLI avant de lancer la configuration des skills d'agent.", + "windowsPathUnknown": "Orca n'a pas pu vérifier votre PATH utilisateur Windows", + "refreshCliRegistration": "Actualisez le statut d'enregistrement du CLI et réessayez." + } + } + } + }, + "remotePairingCopy": { + "invalidInput": "Saisissez un lien d'accès Orca ou un code d'appairage brut.", + "mobileOnly": "Ce lien accorde un accès mobile uniquement. Générez un lien pour un autre client Orca.", + "invalidDestination": "Ce lien d'accès contient une destination invalide.", + "unsupportedDestination": "Ce lien d'accès contient une destination non prise en charge.", + "nonConnectableDestination": "Ce lien d'accès contient une destination non joignable.", + "loopback": "Boucle locale", + "tailscale": "Adresse Tailscale", + "lan": "Adresse LAN privée", + "public": "Adresse publique", + "custom": "Nom d'hôte personnalisé" + }, + "ensure": { + "simulator": { + "tab": { + "372d21d428": "Émulateur mobile" + } + } + }, + "fix": { + "checks": { + "agent": { + "launch": { + "027228a06b": "Impossible de trouver un espace de travail pour ces vérifications.", + "fb6c294e85": "Impossible de construire la commande de lancement de l'agent.", + "03c1d61f83": "Impossible d'ouvrir l'espace de travail associé à ces vérifications.", + "822bf52295": "Impossible de résoudre la plateforme de lancement de l'espace de travail.", + "dfb4dd7c00": "Impossible de trouver l'espace de travail associé à ces vérifications.", + "9f00d7df0c": "Le prompt de correction des checks est vide. Mettez à jour les paramètres d'IA du contrôle de code source.", + "2ebf794906": "Aucun agent IA activé n'a été détecté sur l'hôte de cet espace de travail.", + "4c7f783a7a": "L'agent de checks enregistré n'est pas disponible sur l'hôte de cet espace de travail." + } + } + } + }, + "floating": { + "workspace": { + "tab": { + "creation": { + "f3785eddc2": "Nouvel onglet de navigateur" + } + } + } + }, + "launch": { + "agent": { + "in": { + "new": { + "tab": { + "11cce5cc77": "Impossible de lancer {{value0}} dans un nouveau terminal.", + "a5a1f7033f": "Votre {{value0}} n'a pas été envoyé — collez-le dès que l'agent est prêt." + } + } + }, + "background": { + "session": { + "4ca0651d56": "Votre prompt d'automatisation n'a pas été envoyé — ouvrez l'espace de travail et collez-le." + } + } + }, + "worktree": { + "background": { + "terminals": { + "setupTitle": "Configuration" + } + } + }, + "work": { + "item": { + "direct": { + "3de6371df3": "Impossible de construire la commande de lancement de l'agent.", + "67e103dd60": "L'espace de travail a été créé mais n'a pas pu être activé.", + "19c7683acf": "L'agent sélectionné n'est pas disponible dans l'espace de travail créé.", + "8bc45efdbc": "Échec de la résolution du head de la pull request.", + "agent": { + "ceeeb509b5": "L'agent a mis trop de temps à démarrer. L'espace de travail est prêt — collez le {{value0}} quand l'agent est inactif." + } + } + } + } + }, + "local": { + "path": { + "open": { + "guard": { + "edc1908653": "L'ouverture de chemins distants dans l'OS local n'est pas disponible." + } + } + } + }, + "open": { + "in": { + "app": { + "catalog": { + "f8b8ca2711": "Zed", + "d62b12e98a": "Cursor", + "173553f73a": "VS Code" + } + } + }, + "mobile": { + "emulator": { + "tab": { + "bf4f2a8a72": "Impossible de démarrer l'émulateur. Vérifiez la configuration de l'émulateur iOS ou Android et essayez un autre appareil." + } + } + } + }, + "orchestration": { + "usage": { + "examples": { + "f91fe27f2a": "Scinder un changement volumineux en pull requests plus petites", + "9e37a5b1b3": "Exécuter des travaux indépendants en parallèle", + "bddc4c09b8": "Exécuter un workflow par phases", + "ab0e9803b7": "Transférer vers un autre worktree", + "5e0d489fe1": "Transférer une tâche active", + "handoffSummary": "Confier la responsabilité à un autre agent avec suffisamment de contexte pour continuer.", + "worktreeHandoffSummary": "Déplacer le travail vers un agent qui tourne déjà dans une autre branche.", + "childSequenceSummary": "Enchaîner des agents enfants les uns après les autres lorsque chaque phase dépend de la précédente.", + "childParallelSummary": "Répartir entre agents enfants des tâches d'investigation ou d'implémentation qui ne se chevauchent pas.", + "prSplitSummary": "Donner à chaque agent enfant son propre worktree pour que l'implémentation en parallèle reste relisible." + } + } + }, + "pr": { + "comment": { + "audience": { + "64deee36a9": "Bots", + "a7150a17bc": "Humains", + "27ce73211c": "Tous", + "empty": { + "bot": "Aucun commentaire de bot.", + "human": "Aucun commentaire humain.", + "all": "Aucun commentaire pour le moment." + } + } + }, + "bot": { + "author": { + "overrides": { + "6d5d52b53f": "Limite de remplacement d'auteur de bot atteinte" + } + } + } + }, + "resume": { + "sleeping": { + "agent": { + "session": { + "f235f604fd": "Cette session d'agent ne peut pas être reprise." + } + } + } + }, + "source": { + "control": { + "agent": { + "action": { + "plan": { + "3f0ea9aa0d": "Impossible de construire la commande de lancement de l'agent.", + "46f1a2c9bd": "La saisie de commande est vide.", + "8eb541cc83": "L'agent sélectionné n'a pas été détecté sur l'hôte de cet espace de travail.", + "b96e091fc9": "L'agent sélectionné est désactivé dans les Paramètres.", + "a7ac8717c7": "Choisissez un agent avant de démarrer." + } + } + }, + "generation": { + "plan": { + "dc480d5897": "La saisie de commande est vide." + } + } + } + }, + "sparse": { + "preset": { + "draft": { + "5915a0a1f6": "Utilisez des répertoires relatifs au dépôt, pas la racine, des chemins absolus ni des segments parent.", + "efc05d1820": "Ajoutez au moins un répertoire." + } + } + }, + "terminal": { + "shortcut": { + "capture": { + "notification": { + "b0536028c9": "Ouvrir les raccourcis", + "141ad6c004": "Raccourci terminal traité", + "0ab0cd001a": "size-4 text-muted-foreground" + } + } + } + }, + "workspace": { + "create": { + "error": { + "format": { + "37cf0bc991": "Orca n'a pas pu résoudre une ref de base exploitable pour cet espace de travail.", + "64555d0014": "Aucune branche de base trouvée" + } + } + }, + "browser": { + "tab": { + "open": { + "urlFailed": "Impossible d'ouvrir l'URL.", + "searchFailed": "Impossible d'effectuer la recherche avec {{value0}}." + } + } + } + }, + "worktree": { + "palette": { + "search": { + "9ccec2316b": "Issue", + "ca40ffcbec": "PR", + "0b01ff98d2": "Port", + "7d732521ec": "Commentaire" + } + }, + "activation": { + "cannotOpenFolderWorkspace": "Impossible d'ouvrir l'espace de travail de type dossier" + } + }, + "folderWorkspacePathStatus": { + "title": { + "missing": "Dossier introuvable", + "notDirectory": "Le chemin n'est pas un dossier", + "ambiguousConnection": "Impossible de déterminer la connexion", + "unavailable": "Impossible de vérifier le dossier", + "unrecognized": "Le dossier n'est pas exploitable" + }, + "description": { + "missing": "Orca ne trouve pas {{path}}. Supprimez puis réimportez cet espace de travail de type dossier.", + "notDirectory": "{{path}} existe, mais ce n'est pas un dossier.", + "ambiguousConnection": "Orca ne peut pas déterminer quelle connexion SSH possède ce périmètre de dossier.", + "unavailable": "Orca ne peut pas vérifier ce dossier pour le moment. Vérifiez le runtime ou la connexion SSH, puis réessayez.", + "unrecognized": "Orca ne peut pas utiliser {{path}}, et cette version ne reconnaît pas la raison. Mettez Orca à jour pour voir les détails." + }, + "createError": { + "title": { + "missing": "Dossier introuvable", + "notDirectory": "Le chemin n'est pas un dossier", + "ambiguousConnection": "Impossible de déterminer la connexion", + "unavailable": "Impossible de vérifier le dossier", + "generic": "Échec de la création de l'espace de travail de type dossier" + }, + "description": { + "missing": "Orca ne trouve pas {{path}}. Supprimez puis réimportez le dossier.", + "notDirectory": "{{path}} existe, mais ce n'est pas un dossier.", + "ambiguousConnection": "Orca ne peut pas déterminer quelle connexion SSH possède ce périmètre de dossier.", + "unavailable": "Orca ne peut pas vérifier ce dossier pour le moment. Vérifiez le runtime ou la connexion SSH, puis réessayez." + } + } + }, + "projectSkillRuntime": { + "wslUnavailable": "Le runtime du projet nécessite WSL avant de pouvoir installer ce skill.", + "distroRequired": "Sélectionnez une distro WSL pour ce projet avant d'installer ce skill.", + "distroMissing": "La distro WSL sélectionnée est indisponible. Choisissez une distro disponible ou basculez ce projet sur Windows.", + "wslDefault": "WSL par défaut" + }, + "sidebarWorktreeActivation": { + "wakeEphemeralVmFailed": "Échec du réveil de l'espace de travail VM éphémère" + }, + "ephemeralVmWorkspaceTarget": { + "projectRootRegistrationFailed": "Échec de l'enregistrement sur le runtime de la racine de projet créée par la recette.", + "provisionedRootRequiresSsh": "Les recettes à racine provisionnée nécessitent actuellement une connexion SSH directe." + }, + "blocked": { + "notification": { + "fallback": { + "de50bef680": "macOS bloque les notifications Orca" + } + } + }, + "linear": { + "usage": { + "examples": { + "readTicket": "Lire le ticket lié", + "postUpdate": "Publier un point d'avancement", + "moveState": "Faire avancer le ticket", + "attachPr": "Joindre le lien de review", + "triageFollowups": "Trier et créer des tickets de suivi", + "readTicketSummary": "Récupérer tout le contexte de l'issue Linear liée avant de commencer le travail.", + "readTicketPrompt": "Utilisez {{value0}} pour lire l'issue Linear liée à ce worktree, puis résumez l'objectif et les critères d'acceptation avant de commencer.", + "postUpdateSummary": "Commenter la progression ou un résumé d'achèvement vers l'issue Linear.", + "postUpdatePrompt": "Utilisez {{value0}} pour publier une mise à jour d'achèvement sur l'issue Linear associée, avec ce qui a changé et comment cela a été vérifié.", + "moveStateSummary": "Faites avancer l'état du workflow Linear au fur et à mesure de la progression des travaux.", + "moveStatePrompt": "Utilisez {{value0}} pour passer l'issue Linear associée à In Review maintenant que la modification est prête.", + "attachPrSummary": "Liez la pull request ou merge request à l'issue Linear au moment de l'ouvrir.", + "attachPrPrompt": "Utilisez {{value0}} pour rattacher cette pull request ou merge request à l'issue Linear associée.", + "triageFollowupsSummary": "Définissez l'assigné, la priorité ou l'estimation, et créez des tickets de suivi rattachés.", + "triageFollowupsPrompt": "Utilisez {{value0}} pour trier l'issue Linear associée — définir la priorité et l'estimation — et créer un ticket de suivi rattaché pour le nettoyage différé." + } + }, + "issue": { + "workspace": { + "open": { + "4f2c1d8a3b": "Impossible d'ouvrir l'espace de travail associé à cette issue." + } + } + } + }, + "codex": { + "session": { + "restart": { + "4bd4a3a9c7": "Valeur par défaut du système", + "9f0b1c2d3e": "Compte Codex" + } + } + }, + "browser": { + "cookie": { + "import": { + "toast": { + "restartFallbackUnavailableNone": "Aucun des {{value0}} cookies n'a pu être chargé, et la solution de repli par redémarrage était indisponible. Les cookies précédents de ce profil ont été remplacés. Réessayez l'importation.", + "restartFallbackUnavailablePartial": "{{value0}} cookies sur {{value1}} ont été importés. Le reste n'a pas pu être chargé, et la solution de repli par redémarrage était indisponible. Réessayez l'importation.", + "googleCookiesSkipped": "Les cookies Google n'ont pas été importés. Ouvrez un navigateur dans Orca sur {{value0}} avec ce profil, puis connectez-vous à Google.", + "undecryptableAppBound": "Orca ne peut pas déchiffrer {{value0}} cookies de ce navigateur car ils utilisent un chiffrement lié à l'application. Vous pouvez importer des cookies depuis un fichier avec « Depuis un fichier… ».", + "undecryptableAppBoundMixed": "Orca ne peut pas déchiffrer {{value0}} cookies de ce navigateur car ils utilisent un chiffrement lié à l'application ; {{value1}} autres n'ont pas pu être déchiffrés pour une autre raison. Vous pouvez importer des cookies depuis un fichier avec « Depuis un fichier… ».", + "undecryptableKeyring": "{{value0}} cookies n'ont pas pu être déchiffrés car le trousseau système était indisponible. Déverrouillez votre trousseau de connexion (ou installez un fournisseur Secret Service tel que gnome-keyring), puis importez à nouveau.", + "undecryptableKeyringMixed": "{{value0}} cookies n'ont pas pu être déchiffrés car le trousseau système était indisponible ; {{value1}} autres n'ont pas pu être déchiffrés pour une autre raison. Déverrouillez votre trousseau de connexion (ou installez un fournisseur Secret Service tel que gnome-keyring), puis importez à nouveau.", + "undecryptableUnknown": "{{value0}} cookies n'ont pas pu être déchiffrés et ont été ignorés. Fermez complètement le navigateur source, puis réessayez l'importation.", + "unrecognizedWarning": "L'importation des cookies s'est terminée avec un avertissement que cette version d'Orca ne reconnaît pas. Mettez Orca à jour pour voir les détails, puis vérifiez ce profil avant de vous fier à ses cookies.", + "undecryptableUnrecognizedReason": "{{value0}} cookies n'ont pas pu être déchiffrés et ont été ignorés pour une raison que cette version d'Orca ne reconnaît pas. Mettez Orca à jour pour voir les détails, puis réessayez l'importation.", + "partitionSkipped": "{{value0}} cookies n'ont pas été importés car leur partition de site n'a pas pu être lue. Connectez-vous à nouveau à ces sites dans Orca." + } + } + } + }, + "pluginCommandKeybindings": { + "group": "Plugins" + }, + "feedback": { + "image": { + "attachments": { + "fallbackName": "Pièce jointe image", + "unsupportedType": "{{fileName}} n'est pas un type d'image pris en charge.", + "empty": "{{fileName}} est vide.", + "tooLarge": "{{fileName}} dépasse {{maxSize}}.", + "tooMany": "Vous pouvez joindre jusqu'à {{maxCount}} images.", + "dimensionsTooLarge": "{{fileName}} a des dimensions trop grandes pour un aperçu sans risque.", + "invalidImage": "{{fileName}} n'est pas une image valide prise en charge.", + "additionalErrors": "{{count}} images supplémentaires n'ont pas pu être jointes." + } + } + }, + "ephemeralVmWorktreeCreation": { + "sparseCheckoutUnsupported": "Les recettes à racine provisionnée ne prennent pas en charge le sparse checkout." + } + }, + "hooks": { + "useAutomationDispatchEvents": { + "59718b120b": "L'espace de travail cible n'est plus disponible.", + "16a21d6413": "La reconnexion SSH nécessite des identifiants interactifs.", + "386db94f3e": "Le projet cible n'est plus disponible.", + "3ad7d77f57": "L'espace de travail cible se trouve sur un hôte différent de la cible de cette exécution d'automatisation." + }, + "useComposerState": { + "7eb3f44ff7": "L'agent sélectionné est désactivé. Choisissez un agent activé avant de créer.", + "b2ead86962": "Échec de la résolution de la base de la PR.", + "a9ff236145": "Certaines pièces jointes n'ont pas pu être envoyées.", + "3db83fc58a": "Aucun chemin de projet n'est disponible sur cet hôte pour les pièces jointes.", + "ba6cb77082": "Échec de la connexion au projet.", + "chooseOrAddProjectBeforeWorkspace": "Choisissez ou ajoutez un projet avant de créer un espace de travail.", + "folderWorkspaceCreateFailedTitle": "Échec de la création de l'espace de travail de dossier", + "folderWorkspaceCreateFailedMessage": "L'espace de travail de dossier n'a pas pu être créé. Vérifiez les détails de l'erreur ci-dessus, puis réessayez.", + "setupAgentStartupPolicySaveFailed": "Échec de l'enregistrement du comportement de démarrage de la configuration.", + "5f3d2c8a1b": "Échec de la résolution de la base de la MR." + }, + "useGlobalFileDrop": { + "38c9f034ff": "Échec de l'envoi des fichiers déposés.", + "d720e2f855": "Certains fichiers déposés n'ont pas pu être envoyés.", + "245faa95b9": "Aucun chemin de espace de travail distant n'est disponible pour les fichiers déposés.", + "nativeDropTooManyPathsDescription": "Déposez {{value0}} fichiers maximum à la fois.", + "nativeDropTooManyPaths": "Le dépôt contient trop de fichiers.", + "nativeDropPathsTooLargeDescription": "Déposez moins de fichiers ou utilisez une liste de chemins plus courte.", + "nativeDropPathsTooLarge": "La liste de chemins du dépôt est trop grande.", + "ownerChanged": "Impossible de vérifier quel hôte possède cet espace de travail. Réessayez après sa reconnexion." + }, + "useIpcEvents": { + "0e3cf53060": "Onglet de navigateur {{value0}} introuvable", + "2f6637fe6c": "L'onglet de navigateur {{value0}} est épinglé", + "a8d2bf8e9e": "Aucun onglet de navigateur actif à fermer", + "291c8ed902": "Les onglets de navigateur sont indisponibles tant qu'un runtime distant est actif", + "f45fa2b03c": "Les profils de navigateur sont indisponibles tant qu'un runtime distant est actif", + "f000b2ff76": "Aucun worktree actif", + "56d3ec4203": "Échec de la création du fichier markdown sans titre.", + "f6300deb8b": "Nouvel onglet de navigateur", + "7a64b31991": "La création d'un terminal local est indisponible tant qu'un runtime distant est actif", + "60428567b4": "L'affichage d'un terminal local est indisponible tant qu'un runtime distant est actif", + "f8aaf2bde3": "Espace de travail téléversé", + "2fe88c2e06": "Synchronisation de l'espace de travail distant indisponible", + "workspaceChangedOnAnotherDevice": "Espace de travail modifié sur un autre appareil", + "2ec42e1c52": "Pas encore de espace de travail distant", + "88214a785b": "La synchronisation de l'espace de travail a attendu l'hydratation de la session locale et a expiré", + "4f78ba5885": "Espace de travail synchronisé", + "ef223fbb6b": "Un appareil a tenté de se connecter mais n'est pas jumelé", + "11992d0337": "S'il s'agissait de votre téléphone ou d'un autre client Orca, réappairez-le depuis Paramètres → Mobile.", + "6573cfe955": "Ouvrir les paramètres mobiles", + "unresolvedTerminalWorktreeOwner": "La création de terminal est indisponible car le propriétaire du worktree n'a pas pu être déterminé" + }, + "useSettingsNavigationMetadata": { + "4a728cd56b": "Nouvelles fonctionnalités encore en cours de construction. Essayez-les.", + "225071c560": "Expérimental", + "e338c507c1": "Paramètres de compatibilité bas niveau pour le dépannage.", + "580a04cd81": "Avancé", + "8400cfe1c1": "Données d'utilisation anonymes et contrôles de télémétrie.", + "3618579df6": "Confidentialité & télémétrie", + "65ec7d1968": "Accès de confidentialité macOS pour les outils de développement lancés depuis le terminal.", + "d91ae31fbd": "Autorisations macOS", + "95a1886d94": "Contrôlez les terminaux et les agents depuis votre téléphone.", + "1cd25673df": "Mobile", + "31e57d1c70": "Utilisez des machines existantes via SSH pour les fichiers, les terminaux, Git et les espaces de travail.", + "94a5afe910": "Hôtes SSH", + "40d80bad8a": "Beta", + "de0c2907a1": "Serveurs Orca distants", + "b351014180": "Statistiques Orca, plus analyses de tokens Claude, Codex, OpenCode et utilisation d'abonnement Grok.", + "d72a58b5b9": "Statistiques & usage", + "dcd0d9b74f": "Raccourcis clavier pour les actions courantes.", + "94295ebfb3": "Raccourcis", + "7682607591": "Notifications natives de bureau pour les événements d'agent et de terminal.", + "2eece16ad1": "Notifications", + "1f452cbd4c": "Comportement de sélection et d'édition.", + "0c6ee88a5f": "Saisie & édition", + "b11a5a48a2": "Thème, zoom, apparence de l'app et du terminal, barres latérales et barre d'état.", + "93d88d20bf": "Apparence", + "2d0659f6f0": "Onglets globaux de terminal, de navigateur et de markdown.", + "65b19f5bde": "Espace de travail flottant", + "3d65d3f1b9": "Configurez la prise en charge des émulateurs mobiles pour Orca et les agents de codage.", + "1e761cff2b": "Émulateur mobile", + "e815fd01bd": "Page d'accueil, routage des liens et cookies de session.", + "8c197f74a1": "Navigateur", + "42ae40842f": "Commandes de terminal enregistrées, globales ou par projet.", + "3fc3db144f": "Commandes rapides", + "c33bfd664c": "Shells, moteur de rendu, sessions et comportement du terminal.", + "a9fb10afca": "Terminal", + "5235c215ca": "Choisissez les fournisseurs de tâches affichés dans la page Tâches et la barre latérale.", + "85f4fd7710": "Sources de tâches", + "ab4b21b58e": "Nommage des branches, refs de base et Git AI Author.", + "09607cb0fe": "Git & gestion de code source", + "33a5e1d597": "Connectez GitHub, GitLab, Linear et les services d'hébergement de sources.", + "2b043783ef": "Intégrations", + "2cd4ea75da": "Valeurs par défaut des espaces de travail, configuration de l'app et maintenance.", + "13241992bd": "Général", + "724c440e72": "prise en main", + "0505d0df29": "démarrer avec Orca", + "ea0b1bc7b8": "guide de configuration", + "17005c73d4": "Ouvrez la checklist d'intégration pour les étapes de configuration et les jalons.", + "ded9e9032f": "Checklist d'intégration", + "5f32ac08f3": "Terminez la checklist d'intégration pour les workflows essentiels d'Orca.", + "8ac3de82f5": "Dictée vocale locale avec modèles embarqués sur l'appareil.", + "6a50cdcd7c": "Voix", + "0059bd17f3": "Permettez aux agents de contrôler n'importe quelle app de votre ordinateur.", + "b35e92364b": "Computer Use", + "cd50cec5d7": "Coordonnez plusieurs agents de codage via Orca.", + "58a868e8e4": "Orchestration", + "7c79d3b7bf": "Facultatif", + "b1c2f8b0ac": "Configuration facultative de changement de compte et d'utilisation pour Claude, Codex, Gemini, OpenCode Go, MiniMax et Grok.", + "f70ac54d38": "Comptes de fournisseurs d'IA", + "4121f7a0a2": "Gérez les agents IA, définissez-en un par défaut et personnalisez les commandes.", + "b49abbd2f7": "Agents", + "dev": "Outils dev", + "devDescription": "Outils réservés au développement pour exercer les états de l'UI.", + "devBadge": "Dev", + "devSearchNotificationPlayground": "Terrain d'essai des notifications", + "devSearchNotificationPlaygroundDescription": "Déclenche des états d'UI représentatifs de toast et de notification.", + "devSearchKeywordDev": "dev", + "devSearchKeywordToast": "toast", + "devSearchKeywordSonner": "sonner", + "devSearchKeywordError": "error", + "devSearchKeywordNotification": "notification", + "projectHostsSummary": "{{value0}} hôtes", + "linearTitle": "Linear", + "linearDescription": "Fonctionnement de Linear dans Orca, checklist de configuration, skill d'agent et exemples de prompts.", + "pluginsTitle": "Plugins", + "pluginsDescription": "Installez et gérez les plugins Orca expérimentaux.", + "tasksDescription": "Connectez des fournisseurs, installez le skill Linear et choisissez ce qui apparaît dans Tâches.", + "artifactsTitle": "Artefacts", + "artifactsDescription": "Partagez des fichiers HTML et Markdown avec votre équipe et gérez leurs liens publics.", + "automationsTitle": "Automatisations", + "automationsDescription": "Planifiez le travail des agents et choisissez si Automatisations apparaît dans la barre latérale.", + "shareSkillsTitle": "Partage de skills", + "shareSkillsDescription": "Partagez vos skills avec un lien non répertorié. Toute personne qui l'a peut les installer.", + "floatingWorkspaceWebDescription": "Onglets globaux de terminal et de markdown." + }, + "useAppMenuPaste": { + "pasteTooLarge": "Le collage est trop volumineux." + }, + "useLargeTextControlPaste": { + "pasteTooLarge": "Le collage est trop volumineux." + }, + "useInstalledAgentSkills": { + "unreadableSkillSource": "Un dossier de skill n'a pas répondu ; cet état peut donc être incomplet." + }, + "useMacosTccPromptNotice": { + "title": "Vous voyez des invites « Orca souhaite accéder à… » ?", + "description": "Des messages d'autorisation macOS peuvent apparaître quand un agent ou un outil de terminal exécuté dans Orca tente d'accéder à des fichiers protégés. Accordez l'Accès complet au disque dans les paramètres pour réduire ces invites.", + "openSettings": "Ouvrir les paramètres", + "dismiss": "Ne plus afficher" + }, + "useMacTccAttributionSeveredNotice": { + "title": "Les autorisations macOS peuvent ne pas s'appliquer aux terminaux Orca", + "description": "Les terminaux Orca en cours d'exécution sont hébergés par un daemon démarré par une installation précédente d'Orca. macOS peut ne pas leur appliquer les autorisations Accessibilité, Automatisation ou fichiers protégés d'Orca. Redémarrez le daemon depuis Gérer les sessions pour restaurer l'accès. Cela fermera tous les terminaux Orca en cours d'exécution.", + "openManageSessions": "Ouvrir Gérer les sessions", + "dismiss": "Ignorer" + } + }, + "components": { + "BrowserCookieImportDisclosure": { + "title": "Connexions Google non importées", + "description": "Connectez-vous à Google directement dans Orca." + }, + "CodexRestartChip": { + "a4c8e1b2f7": "Codex est toujours connecté en tant que {{value0}}", + "c72a5fb234": "Redémarrer", + "9263e75f49": "Codex utilise le compte précédent", + "d3e8a1f4b2": "Compte changé", + "9375620cc3": "Redémarrez cette session pour utiliser {{value0}}. Elle reste sur le compte précédent tant que vous ne l'avez pas fait.", + "6133594b12": "Conserver l'ancien compte", + "8f0d5c92a1": "Configuration Codex modifiée", + "3ea91b5c07": "Cette session Codex utilise une configuration obsolète", + "e6b7139d2a": "Redémarrez cette session pour charger votre configuration Codex actuelle.", + "7b1d20f4c8": "Conserver la session actuelle" + }, + "FirstLaunchBanner": { + "b9e1b966c7": "Masquer l'avis", + "94cc673726": "Compris", + "fc5cc29955": "Refuser", + "d1deebb050": "Politique de confidentialité", + "958d2cc31b": "Des comptages anonymes des fonctionnalités que vous utilisez nous aident à prioriser ce qu'il faut construire. Aucun contenu de fichier, prompt, sortie de terminal ni rien qui vous identifie. Modifiable à tout moment dans Paramètres -> Confidentialité & télémétrie.", + "9784b4d7bc": "Aidez-nous à décider quoi construire ensuite", + "fcbee32f08": "Avis de télémétrie" + }, + "GitHubItemDialog": { + "3ab6ac0fc8": "Prévisualisez et modifiez l'issue ou la pull request GitHub sélectionnée.", + "3cd5ae5b7b": "Aucun fichier modifié.", + "filesUnavailable": "Impossible de charger les fichiers modifiés.", + "filesRetry": "Réessayer", + "999b5ad7d9": "Fichiers", + "4bd1f5b055": "Vérifications", + "e30a5470c9": "Conversation", + "474c59b4b3": "Fermer · Esc", + "45af57999b": "Fermer l'aperçu", + "3fdf777817": "Ouvrir sur GitHub", + "c43fe79ee0": "Copier le lien GitHub", + "0caac1a18f": "Démarrer un espace de travail à partir de la PR", + "8223320f8d": "mis à jour", + "10ef1afb8e": "· mis à jour", + "55962099bc": "a ouvert cette issue", + "0ab4664a8b": "Démarrer un espace de travail à partir de l'issue", + "36182aa57f": "Démarrer un nouveau espace de travail", + "fe6ff12dc2": "Plus d'actions de espace de travail pour l'issue", + "726db41722": "Ouvrir l'espace de travail", + "84855fedd0": "Ouvrir l'espace de travail associé à l'issue", + "b7bf31b8de": "Échec de la synchronisation de l'état consulté avec GitHub.", + "c0253318d6": "Impossible de synchroniser l'état consulté de cette pull request.", + "5fea151559": "Échec de la copie du lien GitHub", + "2e77dc2053": "Lien GitHub copié", + "2ef631437e": "Impossible d'ouvrir l'espace de travail associé à cette issue.", + "bf43425540": "Commentaire", + "0a73f59e85": "Envoyer le commentaire", + "c5c117270e": "Ajouter un commentaire…", + "082515176a": "Échec de l'ajout du commentaire", + "c6f37a563d": "+ Assigné", + "f41ec96c13": "+ Étiquette", + "ab050dffec": "Fermé", + "dc1ca081a8": "Ouvert", + "2e4d806c92": "Espace de travail", + "886a64b081": "Aucun pour l'instant", + "4ba0132f37": "Modifier les étiquettes", + "217e55d87c": "Étiquettes", + "c67de9e2fe": "Personne n'est assigné", + "76adcf5fe2": "Modifier les assignés", + "83ac703dda": "Assignés", + "00ccdf9b5a": "Statut", + "2aa9acdf34": "Modifier les étiquettes sur GitHub", + "e52bed9264": "Aucune vérification signalée pour l'instant", + "90020cc1f3": "Cette pull request n'a aucune vérification signalée pour l'instant.", + "ecffebc251": "Aucune vérification trouvée", + "checkOpenInBrowser": "Ouvrir dans le navigateur", + "744197c84d": "Aucune sortie en ligne n'est disponible pour cette vérification.", + "08d072664d": "Jobs", + "96d8f36798": "Annotations", + "485609c4f2": "vérification #", + "0f478f5efa": "Terminé", + "4812814bc8": "Démarré", + "9c3ba11a05": "Statut :", + "934d87ab96": "Chargement des détails de la vérification…", + "71c11aff84": "Relancer toutes les vérifications", + "e31651a224": "Relancer les vérifications échouées", + "1b56e28faa": "Relancer", + "f4b1292569": "Démarrer l'agent IA par défaut sur ces vérifications", + "9a1004fc76": "Actualiser les vérifications", + "03e542fcfe": "Échec du démarrage d'un agent IA pour les vérifications cassées : {{value0}}", + "28986b3747": "Un agent IA a été démarré pour les vérifications cassées.", + "1690fd7f4a": "Aucune vérification cassée à corriger.", + "9e7c221b8d": "Échec de la relance des vérifications", + "e463ec935f": "Relances des vérifications demandées", + "ddafe851e1": "Relance de la vérification demandée", + "0bbdc673c1": "Échec de l'actualisation des vérifications", + "e7007aa1d8": "Impossible d'actualiser les vérifications sans chemin de dépôt.", + "675bc0d638": "Annuler", + "a18f669c7a": "{{value0}} {{value1}} réaction{{value2}}", + "53fe19aefc": "Ouvrir la zone de merge GitHub", + "a2495e4784": "Pull request", + "ce360fc318": "Échec de la désactivation de l'auto-merge", + "825a8fb8cd": "Échec de l'activation de l'auto-merge", + "4b390bd50d": "Auto-merge désactivé", + "a35ea5a0f6": "Auto-merge activé", + "aba792c8b3": "Échec de la fusion de la pull request", + "dbe5e2448e": "Pull request fusionnée", + "a27ee5ca1a": "Cela mettra à jour la pull request sur GitHub.", + "03d7216d62": "{{value0}} la PR #{{value1}} ?", + "e9b7cb7d17": "Impossible de {{value0}} la PR", + "bd3b4492a0": "Pull request rouverte", + "9f88657c4e": "Pull request fermée", + "b6f1b7adbd": "Cela rouvrira la pull request sur GitHub.", + "de45fedf7b": "Cela fermera la pull request sur GitHub.", + "5a94f3d0e9": "Aucun commentaire pour le moment.", + "1506916c09": "Commentaires", + "9b9cb55994": "Aucune description fournie.", + "52b20b56f7": "Description", + "4d555d3796": "Modifier la description", + "9df4e74bdf": "Enregistrer", + "0ae387d8ca": "par", + "228e2f59d3": "Résolu", + "a154ec5224": "Ouvrir le commentaire sur GitHub", + "bca8eb39ac": "Répondre au commentaire", + "68cb993d61": "résolu", + "10f4ff5be8": "Réponse publiée.", + "745c9089ec": "Impossible de répondre sans chemin de dépôt.", + "58c73cb0d8": "Échec de la mise à jour de la description.", + "5221548274": "Description mise à jour.", + "06c06e58ba": "Afficher plus de lignes en dessous", + "307c98e8e3": "Afficher {{value0}} lignes supplémentaires au-dessus", + "5664681624": "Afficher plus de lignes au-dessus", + "b1574e8ac2": "Réinitialiser le contexte de code", + "d43736d09c": "Contrôles du contexte de code", + "bd7be7b1fd": "commentaire L", + "db61d76cd5": "Chargement du contexte de code…", + "f2d02cdf8c": "fichiers consultés", + "1257d1435d": "Afficher l'arborescence des fichiers", + "a341343303": "Commentaire de review ajouté.", + "d1fa2cf888": "Impossible de commenter sans le SHA head de la PR.", + "829674460a": "Diff indisponible car les SHA des commits de la PR manquent.", + "af924014f8": "Consulté", + "2d89a38d9d": "{{value0}} {{value1}} comme consultés", + "70e84e3d0b": "Aucun reviewer correspondant.", + "1ffce94a8b": "Tous les autres", + "c2b21818e1": "Suggestions", + "a98433e73d": "Chargement...", + "b0b7344684": "Demander jusqu'à 15 reviewers", + "934add88b6": "Reviewer", + "bb42774171": "Saisissez ou choisissez un utilisateur", + "36f9ac4a47": "Aucun reviewer demandé.", + "8b15a5e91c": "Retirer le reviewer {{value0}}", + "6a45771d47": "Chargement des reviewers", + "dc8a092c57": "Reviewers", + "e3243d9376": "A récemment modifié ces fichiers", + "8c45901789": "Demander le reviewer {{value0}}", + "fedc09eeb9": "Retirer la demande pour le reviewer {{value0}}", + "73487fb975": "Échec du retrait du reviewer", + "2e69540652": "Reviewers retirés", + "69515bff81": "Reviewer retiré", + "b4af16bf43": "Aucun contexte de dépôt disponible pour cette pull request.", + "c42d942b75": "Échec de la demande de reviewer", + "c016e4bac3": "Reviewers demandés", + "ea985e657f": "Reviewer demandé", + "12e761610e": "Vous pouvez demander jusqu'à 15 reviewers", + "94ab23a9f9": "Saisissez un reviewer", + "3853476a97": "Élément GitHub", + "68796dafa0": "pr", + "b1157c78ff": "sheet", + "038b3d39b1": "Copied", + "04539beb48": "issue", + "773ff70035": "unknown", + "3e544d966d": "Issue", + "88fd82474d": "page", + "e517b4d641": "closed", + "d0a05e73f5": "compact", + "7d42606f66": "Annotation", + "2511f44bb7": "Corriger les vérifications cassées", + "9157d48ddb": "Corriger les vérifications", + "06482d6190": "Impossible de créer automatiquement un espace de travail de correction.", + "f64dd90102": "Répondre", + "5752c25aff": "Publication…", + "ec5c4b3ab2": "Rouvrir la PR", + "21860b58d0": "Fermer la pull request", + "5932578f51": "Le merge exige un dépôt local enregistré", + "ce8a85d209": "default", + "924c2fe05e": "destructive", + "e2bf3e41a9": "comment", + "28d0d3374f": "thread", + "080d071d48": "Répondre à @{{value0}}", + "86f809e2ce": "Répondre dans ce fil de review", + "136542c9ba": ":L{{value0}}", + "283699bc82": "Échec de la publication de la réponse.", + "d1c0dad471": "-L{{value0}}", + "31770bef03": "Côte à côte", + "6e43a16435": "En ligne", + "d00a0a7f8f": "Tout réduire", + "3c19ec3069": "Tout développer", + "b0b09778c8": "Échec de l'ajout du commentaire de review.", + "16c1abe76c": "Marquer comme consulté", + "ba8e329d92": "Ne plus marquer comme consulté", + "3f79ffc8b7": "Ouvrez les détails de la PR pour voir les reviewers actuels.", + "5c1c973855": "Retirer le reviewer", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque.", + "timeline": { + "completed": "comme terminé", + "notPlanned": "comme non prévu", + "someone": "quelqu'un", + "assigned": "a assigné", + "unassigned": "a désassigné", + "mentioned": "a mentionné ceci", + "in": "dans", + "closed": "a fermé ceci", + "reopened": "a rouvert ceci", + "moved": "a déplacé ceci", + "from": "de", + "to": "vers", + "activity": "Activité", + "noActivity": "Aucune activité pour l'instant." + }, + "checkActionRequiredHint": "Cette vérification nécessite une action manuelle sur GitHub (par exemple, approuver l'exécution du workflow) avant que le merge soit débloqué.", + "e15a8b77ef": "Aucun détail en ligne n'est disponible pour cette vérification.", + "e45324fbed": "Échec du chargement des détails de la vérification.", + "dcb3c546fe": "Réessayer", + "85a2b66f54": "Ouvrir les fichiers sur GitHub", + "86d84a17ca": "Ajouter un commentaire de review", + "d9fa90b625": "Échec du chargement du diff.", + "4aecf121e7": "Fermer", + "8812225174": "Rouvrir", + "09d67a0f9b": "Échec de la fermeture de la pull request", + "88809e79db": "Échec de la réouverture de la pull request", + "00f55cc17b": "L'accès au dépôt est indisponible pour cette pull request." + }, + "GitLabItemDialog": { + "65e784c1f1": "Rouvrir", + "a199eb364b": "Fermer", + "16b3412570": "Merge", + "131865e231": "Créer un espace de travail", + "f2e64d1c20": "Ouvrir dans GitLab", + "84012fa8fb": "Commentaire", + "c08e1d5a57": "Commenter {{value0}}{{value1}}…", + "f11e3e7675": "Aucune exécution de pipeline pour cette MR.", + "808b1ca1ba": "Aucun fichier modifié.", + "007423f585": "Contenu du diff indisponible.", + "a7eb4f4916": "de", + "21f8dde18a": "Commentaire en ligne", + "7a7204417f": "Ligne", + "ceb08a733d": "Fichier", + "85a8170279": "Aucun commentaire pour le moment.", + "14423484db": "Aucune description.", + "da4174b00f": "Édition", + "93f79a3fc1": "Enregistrer", + "f72fad3b16": "Annuler", + "717b706849": "Chargement des étiquettes", + "3c0b6ccca7": "bug, backend", + "dde24ade55": "Étiquettes", + "908d8d2a73": "Description", + "89f3f19368": "Titre", + "7a2117129a": "Ajouter", + "05939e977d": "Ajouter un reviewer", + "474b50d988": "Aucun reviewer.", + "1b19cdc510": "Retirer le reviewer {{value0}}", + "cb55b0390f": "Gérer", + "4f9313984d": "Reviewers", + "02cbe2de44": "Pipeline", + "be3d291837": "Fichiers", + "c996e2962c": "Conversation", + "b3c156dd51": "Actualiser", + "9bfb4a24d7": "par", + "30c97083c2": "Détail de l'élément de travail GitLab", + "e089f62594": "MR !{{value0}} fusionnée", + "865ea2703e": "MR !{{value0}} rouverte", + "9b11cd233f": "MR !{{value0}} fermée", + "60c13320c4": "Commentaire en ligne ajouté", + "ffdd9a78e1": "Les refs de diff de la MR sont indisponibles pour les commentaires en ligne.", + "00d0d25825": "Le fichier, la ligne et le commentaire sont requis.", + "ceaf7c30c7": "L'id de reviewer est indisponible pour cet utilisateur GitLab.", + "f7cb495a12": "{{value0}} relancé(s)", + "98718490e4": "Le titre de la MR est requis.", + "d600c2619a": "Chargement du log", + "028bde664e": "Masquer", + "2f9b27f838": "Log du job", + "032ae1312b": "Ouvrir le job dans GitLab", + "fa3e042203": "Réessayer", + "f23ea85341": "résolu", + "4186685c78": "rouvrir", + "cae2712a23": "clôturer", + "881e522e04": "merge", + "6de8ce0cc6": "{{value0}} requis", + "22511537d2": "Approuvé", + "00f3bab87b": " sur {{value0}} requis", + "11384f99aa": "nombre", + "40c56b95e2": "Il reste {{value0}} approbation{{value1}}", + "3a051b8ade": "Élément de travail", + "32f8bef818": "Aucune sortie de log.", + "4168eb2c51": "Résoudre", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque." + }, + "JiraIssueWorkspace": { + "b0b92666c9": "Commentaire", + "a585fd204e": "Ajouter un commentaire Jira...", + "2441be6f9f": "Démarrer l'espace de travail", + "9178090e26": "Aucun commentaire pour le moment.", + "5cd09beaf9": "Réessayer", + "9a980b06b9": "Commentaires", + "c4889a47e4": "Aucune description fournie.", + "0f3c07a901": "backend, bug", + "aee97b6913": "Étiquettes", + "444865b4a8": "Titre", + "0b6b5646ed": "Non assigné", + "51bed73f88": "Aucune priorité", + "7a96985ca0": "Fermer", + "76513c7898": "Fermer l'aperçu du ticket Jira", + "857bd2f88f": "Prévisualisez, modifiez et démarrez le travail depuis le ticket sélectionné.", + "0cc62bd690": "Copier le prompt", + "80efa101c5": "Copier le nom de branche suggéré", + "38839801e8": "Copier la clé", + "779bb91ee0": "Copier l'URL", + "69da9a208c": "Ouvrir dans Jira", + "fa132c8aed": "Échec de l'ajout du commentaire.", + "ea21952aa3": "Échec de la mise à jour du ticket Jira.", + "6c41a9bcea": "Échec de la copie : {{value0}}", + "2ff69a3545": "{{value0}} copié", + "666cfdd835": "Inconnu", + "9ebee71962": "labels", + "def3d0e824": "titre", + "b8e2079d96": "assigné", + "54649eaeab": "+ Assigné", + "2a829a2f00": "priorité", + "693be070d0": "transition", + "ef21405c6d": "Ticket Jira", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque." + }, + "Landing": { + "76a95f7f47": "Créer", + "f05d237049": "Ajoutez d'abord un projet", + "f9eaa9e12d": "Ajouter un projet", + "6ca6ff404e": "ORCA", + "520304a067": "Logo Orca", + "ce44fad849": "Dépendances manquantes", + "c1cf168479": "Masquer", + "00cee697c1": "Exécutez \"gh auth login\" dans un terminal pour connecter votre compte GitHub.", + "9f96d018b7": "GitHub CLI n'est pas authentifiée", + "73e1ad4282": "Orca utilise la GitHub CLI (gh) pour afficher les pull requests, les tickets et les vérifications.", + "5beaef5f9e": "GitHub CLI n'est pas installée", + "b673e7cf1b": "Git est requis pour les projets Git, le contrôle de version et la gestion des espaces de travail.", + "e5b7296d9d": "Git n'est pas installé", + "cd21242762": "Ajoutez un projet pour commencer.", + "9c00bd4adf": "Sélectionnez un espace de travail dans la barre latérale pour commencer.", + "16e9e3df89": "favori", + "0d0ace8861": "Mettre une étoile sur GitHub", + "ec43b38ba7": "Étoile ajoutée sur GitHub", + "157bb5ecbb": "Ouvrir GitHub", + "preflightDismiss": "Ignorer" + }, + "LinearIssueMarkdownDescriptionEditor": { + "d9c47069ef": "Markdown", + "a7301a11f3": "enregistrer", + "632096eb1c": "Lien", + "340160f4e8": "Supprimer le lien", + "9eaf02ac01": "Citation", + "e2a0267c8c": "Liste de tâches", + "d6b2f3d35b": "Liste numérotée", + "c82917e06e": "Liste à puces", + "ad1869bd54": "Code en ligne", + "28fd951b83": "Barré", + "5666b4493d": "Italique", + "caa88f50d0": "Gras", + "dddaa7a0a6": "Titre 2", + "e3f741d258": "Titre 1", + "68a41d5665": "Texte courant", + "7c52151156": "Mise en forme de la description du ticket", + "5c16ec8f14": "URL du lien", + "4f2fddc2b7": "Aucune description fournie." + }, + "LinearIssueTextEditor": { + "947ba2d6f4": "pour enregistrer", + "04d73b72dc": "Titre du ticket", + "e8ff595db3": "Échec de la mise à jour de {{value0}}", + "1e08a1ec80": "Le titre est obligatoire", + "00fa439dc7": "titre", + "75294b07d1": "description" + }, + "LinearIssueWorkspace": { + "ad5dec37b7": "Prévisualisez, modifiez et démarrez le travail depuis le ticket sélectionné.", + "c23e79e5c0": "Actions", + "b0eac92d85": "Réessayer", + "fabbd3f974": "a mis à jour le ticket ·", + "543970c87a": "Activité", + "df4c86ed12": "Fermer", + "7a4997d8bb": "Fermer l'aperçu du ticket Linear", + "e1e0a9bca9": "Démarrer l'espace de travail", + "30a7f56c0a": "Démarrer un espace de travail à partir de l'issue", + "30c1242f3a": "Copier l'identifiant", + "9e3c49beb8": "Copier l'identifiant du ticket", + "9a9a884236": "Copier l'URL", + "97c19a84f1": "Copier l'URL Linear", + "f63ef94ea8": "Tickets", + "f6c6381593": "Copier le prompt", + "5d670ec8dc": "Copier le nom de branche suggéré", + "937ba6ad9a": "Chargement des projets", + "db3f269d98": "Rechercher des projets", + "b51276c8d6": "Projet", + "8b5b593053": "Échec de la mise à jour du projet", + "f9d4ef9807": "Projet mis à jour", + "38b80780c2": "Échec du chargement des projets", + "42589845bc": "Créer", + "c182e02de5": "Titre du sous-ticket", + "8c55d6696a": "Ajouter des sous-tickets", + "b25e453c9d": "Échec de la création du sous-ticket", + "aeed19d003": "{{value0}} créé", + "9a1317cdd3": "Échec du chargement du sous-ticket", + "9bcbaa2737": "Échec de la copie : {{value0}}", + "7835483c43": "{{value0}} copié", + "61f424f8ca": "Ticket Linear", + "ca8778c124": "Inconnu", + "8a33c85e9c": "Quelqu'un", + "f5a6b38a14": "sheet", + "65239a714b": "Linear", + "af6e02c44a": "page", + "76ffd3c937": "Recherchez un projet à ajouter.", + "c11b4e3cc2": "Aucun projet trouvé.", + "519c3587f3": "Ajouter au projet", + "openAttachedWorkspace": "Ouvrir l'espace de travail associé à l'issue", + "openWorkspace": "Ouvrir l'espace de travail", + "openOnLinear": "Ouvrir dans Linear", + "moreWorkspaceActions": "Plus d'actions de espace de travail pour l'issue", + "startNewWorkspace": "Démarrer un nouveau espace de travail", + "workspaceSection": "Espace de travail", + "noWorkspaceYet": "Aucun pour l'instant" + }, + "LinearItemDrawer": { + "04008e6c46": "Démarrer un espace de travail à partir de l'issue", + "a4fcc57522": "Aucun commentaire pour le moment.", + "fde849b2b6": "Commentaires", + "9dc54172db": "Fermer · Esc", + "0190b760c1": "Ouvrir dans Linear", + "04a442f796": "Prévisualisez et modifiez le ticket Linear sélectionné.", + "d369841269": "Envoyer le commentaire", + "2fcff829a8": "Ajouter un commentaire…", + "2820f0f0f0": "Laisser un commentaire...", + "6ab35eafd5": "Échec de l'ajout du commentaire", + "367f828482": "Aucun label trouvé", + "cddd9b04a7": "Chargement des étiquettes", + "23886c7eec": "Ajouter un label", + "7f7b89b631": "Labels : {{value0}}", + "b2376d0179": "Chargement des membres", + "866316f22c": "Non assigné", + "b5675b0694": "Enregistrer", + "ceeb8c6153": "Effacer", + "fbb90300e2": "Estimation personnalisée", + "780ea6ed89": "Aucun état trouvé", + "59b6cd3706": "Chargement des états", + "64bfffc4dd": "Étiquettes", + "dd304de85a": "Propriétés", + "0be31fef8e": "L'estimation doit être un entier non négatif", + "48e17e8cbd": "Inconnu", + "858d0630da": "Fermer l'aperçu", + "39883467f4": "Ticket Linear", + "fda549766e": "{{value0}} pour commenter", + "d71cd3003e": "+ Assigné", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque.", + "openAttachedWorkspace": "Ouvrir l'espace de travail associé à l'issue", + "openWorkspace": "Ouvrir l'espace de travail", + "moreWorkspaceActions": "Plus d'actions de espace de travail pour l'issue", + "startNewWorkspace": "Démarrer un nouveau espace de travail" + }, + "NewWorkspaceComposerCard": { + "reuseExistingBranch": "Réutiliser la branche", + "reuseExistingBranchHint": "Faites un checkout de la branche existante au lieu d'en créer une nouvelle à partir de celle-ci.", + "createMultiple": "Créer plus", + "cbb47ee0dc": "Disponible uniquement pour les projets Git locaux.", + "d861de981b": "Sparse checkout", + "090cfedeb4": "Écrire une note", + "f8728aa4f9": "Note", + "0ee17638fe": "Nom de l'espace de travail", + "2688050e4b": "Nom", + "f0470c7383": "Avancé", + "ba64270bdb": "Configurer les agents", + "ab63f25397": "Ouvrir les paramètres des agents", + "01d1e8f601": "Agent", + "0c5d6a479c": "[Optionnel]", + "b5a0796911": "Se connecter", + "dccd26d4e4": "Choisir le projet", + "d6b0a96f32": "Ajouter un projet", + "969a8bff66": "Projet", + "23bb365554": "orca.yaml", + "a239038146": "Erreur de connexion SSH", + "9a70e4859e": "Choisissez si le setup doit être exécuté avant de créer cet espace de travail.", + "803b7fe72f": "Vérification de la configuration du setup...", + "92e34f0311": "paramètres locaux", + "326a578923": "orca.yaml + local", + "2132b670da": "les deux", + "0e587e31fb": "yaml", + "ac3748dcda": "Nom ou « Créer à partir de »", + "f660aa1454": "Connexion", + "7711ad5122": "Commande de setup locale", + "e5db1b0419": "Commande de setup combinée", + "runOn": "Exécuter sur", + "addProjectBeforeWorkspace": "Ajoutez un projet avant de créer un espace de travail.", + "connectProjectFirst": "Connectez d'abord ce projet", + "connectSourceLookup": "Se connecter", + "reconnectSourceLookup": "Reconnecter", + "setupHostExistingFolderTitle": "Configurer {{value0}}", + "cloneProjectOnHost": "Cloner le projet", + "cloneUrlPlaceholder": "https://github.com/owner/repo.git", + "cloneDestinationPlaceholder": "/parent/directory/on/host", + "cloningHostSetup": "Clonage...", + "cloneHostSetup": "Cloner", + "importExistingFolderOnHost": "Importer un dossier existant", + "setupHostExistingFolderPlaceholder": "/path/to/project/on/host", + "setupKindGit": "Dépôt Git", + "setupKindFolder": "Dossier", + "setupHostExistingFolderHelp": "Reliez un checkout déjà présent à cet emplacement, puis créez cet espace de travail sur cet hôte.", + "importingHostSetup": "Import...", + "importHostSetup": "Importer", + "sshNotConnected": "SSH non connecté", + "connectingSsh": "Connexion SSH...", + "sshAuthenticationFailed": "Échec de l'authentification SSH", + "preparingSshConnection": "Préparation de la connexion SSH...", + "connected": "Connecté", + "reconnectingSsh": "Reconnexion SSH...", + "sshReconnectionFailed": "Échec de la reconnexion SSH", + "notConnected": "Non connecté", + "notePasteTooLarge": "Le contenu collé est trop volumineux pour le champ de note.", + "waitForSetupBeforeAgent": "Attendre la fin du setup avant de démarrer l'agent", + "waitForSetupBeforeAgentHelp": "Activez cette option quand le setup installe des dépendances, des serveurs MCP ou des fichiers de configuration dont l'agent a besoin au démarrage.", + "destroyDisabled": "destroy désactivé", + "destroyConfigured": "destroy configuré", + "noDestroyConfigured": "pas de destroy", + "ephemeralVm": "Environnement par espace de travail", + "chooseRunTarget": "Choisir la cible", + "noRunTargets": "Aucune cible d'exécution n'est prête pour ce projet.", + "perWorkspaceEnvHint": "Provisionner un environnement à la demande depuis une recette", + "branchName": "Nom de branche", + "branchNamePlaceholder": "feature/my-branch", + "connectTimedOut": "La connexion a expiré. Elle se poursuit peut-être en arrière-plan.", + "connectingHost": "Connexion…", + "connectHost": "Se connecter", + "setLocation": "Définir l'emplacement du projet", + "setLocationOnHost": "Définir l'emplacement du projet sur {{host}}", + "setLocationTooltip": "Choisissez un dossier ou clonez ce projet sur {{host}}.", + "addHost": "Ajouter un hôte", + "addHostHint": "Enregistrer une autre machine ou un serveur Orca", + "addSshHost": "Ajouter un hôte SSH", + "addSshHostHint": "Utiliser une machine existante via SSH", + "addRemoteOrcaServer": "Ajouter un serveur Orca distant", + "addRemoteOrcaServerHint": "Appairer un autre runtime Orca", + "hostConnectionFailed": "Échec de la connexion" + }, + "NewWorkspaceComposerModal": { + "createWorktree": "Créer un worktree", + "createWorkspace": "Créer un espace de travail", + "fa90f739a5": "Choisissez le projet, le nom de l'espace de travail et l'agent avant de créer l'espace de travail." + }, + "PullRequestPage": { + "2560588245": "Échec de la demande de reviewer", + "3450247584": "Commentaire", + "6ad2c1ab9c": "Aucun fichier modifié.", + "filesUnavailable": "Impossible de charger les fichiers modifiés.", + "filesRetry": "Réessayer", + "4d18310d55": "Fichiers modifiés", + "94d95cf1f7": "Vérifications", + "9e8d45700e": "Conversation", + "e6996f4024": "· mis à jour", + "dd5d9a4f17": "a mis à jour {{value0}}", + "71a3c0f9d2": "Démarrer l'espace de travail", + "00b7b82329": "branche source", + "e1f3641bfd": "de", + "c44b70352b": "branche cible", + "b0e80f083d": "veut merger dans", + "8ecda455a0": "Ouvrir sur GitHub", + "1a2570e18e": "Démarrer un nouveau espace de travail", + "57c13a5aa4": "Plus d'actions d'espace de travail pour la PR", + "25690a3855": "Démarrer un espace de travail à partir de la PR", + "a459866967": "Reprendre l'espace de travail attaché à la PR", + "347034903a": "Copier le lien GitHub", + "5a01ca7253": "Échec de la synchronisation de l'état consulté avec GitHub.", + "996a1897d2": "Impossible de synchroniser l'état consulté de cette pull request.", + "e0b15c793f": "Échec de la copie du lien GitHub", + "992e799227": "Lien GitHub copié", + "61bfc81ada": "Impossible d'ouvrir l'espace de travail attaché à cette pull request.", + "161d91ef02": "Envoyer le commentaire", + "d2030fc8cd": "Ajouter un commentaire…", + "1208347ac0": "Échec de l'ajout du commentaire", + "61452f2143": "Démarrer un espace de travail à partir de l'issue", + "14c9fc70ed": "+ Assigné", + "bc215fea4d": "+ Étiquette", + "b936cc51a4": "Fermé", + "7b8f6bf6d8": "Ouvert", + "a18d01cda3": "Aucune vérification signalée pour l'instant", + "3912daf310": "Cette pull request n'a aucune vérification signalée pour l'instant.", + "45877f5089": "Aucune vérification trouvée", + "85e62c5266": "Un agent IA a été démarré pour les vérifications cassées.", + "ddfd42f460": "Vérifiez le prompt avant de démarrer un agent.", + "a053bdd082": "Corriger les vérifications en échec avec l'IA", + "1b14d0a69c": "Ouvrir dans GitHub", + "1550675e5f": "Aucune sortie en ligne n'est disponible pour cette vérification.", + "7720c9c3f5": "Jobs", + "8432d17901": "Annotations", + "f01bf79a79": "vérification #", + "000f90afcf": "Terminé", + "76551b1161": "Démarré", + "662bc2998d": "Statut :", + "d8e82b7f15": "Chargement des détails de la vérification…", + "54cddd1858": "Relancer toutes les vérifications", + "68605516dd": "Relancer les vérifications échouées", + "522d9353e1": "Relancer", + "0fa8b8faec": "Démarrer l'agent IA par défaut sur ces vérifications", + "5d0f42766d": "Actualiser les vérifications", + "98583589c6": "Échec du démarrage d'un agent IA pour les vérifications cassées : {{value0}}", + "51c65c0265": "Aucune vérification cassée à corriger.", + "788a782bb0": "Échec de la relance des vérifications", + "18f2af42ac": "Relances des vérifications demandées", + "5963a6a852": "Relance de la vérification demandée", + "246b2c6456": "Échec de l'actualisation des vérifications", + "c057f2fcb0": "Impossible d'actualiser les vérifications sans chemin de dépôt.", + "6591b1fa82": "Annuler", + "42c36d9166": "{{value0}} {{value1}} réaction{{value2}}", + "7df8d5fc60": "Ouvrir la zone de merge GitHub", + "1939d0f663": "Pull request", + "973ef2fac9": "Échec de la désactivation de l'auto-merge", + "d31f4b508c": "Échec de l'activation de l'auto-merge", + "0f5821b035": "Auto-merge désactivé", + "5edbe7eefa": "Auto-merge activé", + "aae645d36d": "Échec de la fusion de la pull request", + "c57873d721": "Pull request fusionnée", + "a63b3c159c": "Cela mettra à jour la pull request sur GitHub.", + "eec3706a6a": "{{value0}} la PR #{{value1}} ?", + "710e47aa06": "Pull request rouverte", + "7aa3b5f706": "Pull request fermée", + "3d77438c92": "Cela rouvrira la pull request sur GitHub.", + "5a65651096": "Cela fermera la pull request sur GitHub.", + "d2d589556c": "Aucun commentaire pour le moment.", + "3463d10a63": "Commentaires", + "c8ea6c7c4c": "Aucune description fournie.", + "778683ec84": "Description", + "da9aaa8bcf": "Modifier la description", + "4a337ac05f": "Enregistrer", + "169a93b29a": "mis à jour", + "3c891789f6": "par", + "f4fe47c2bb": "Résolu", + "0ac19bb52e": "Ouvrir le commentaire sur GitHub", + "d6c6679de7": "Répondre au commentaire", + "76b2a0ac5b": "résolu", + "11505c7a71": "Réponse publiée.", + "6885c619e7": "Impossible de répondre sans chemin de dépôt.", + "d94810f652": "Échec de la mise à jour de la description.", + "9b4190dc98": "Description mise à jour.", + "51ed0cf38b": "Afficher plus de lignes en dessous", + "e295a78c11": "Afficher {{value0}} lignes supplémentaires au-dessus", + "c9de94b07a": "Afficher plus de lignes au-dessus", + "5f3e293517": "Réinitialiser le contexte de code", + "85d119be40": "Contrôles du contexte de code", + "791ddede19": "commentaire L", + "4b960e5978": "Chargement du contexte de code…", + "89e80af1c7": "fichiers consultés", + "319cf2d54b": "Afficher l'arborescence des fichiers", + "eff839f438": "Commentaire de review ajouté.", + "d8c3ba91c4": "Impossible de commenter sans le SHA head de la PR.", + "74660bd80b": "Diff indisponible car les SHA des commits de la PR manquent.", + "2e528e1c2d": "Consulté", + "ff84e1f54c": "{{value0}} {{value1}} comme consultés", + "5ad00c7a0e": "Aucun reviewer correspondant.", + "2760fa29a4": "Tous les autres", + "828f045847": "Suggestions", + "57750f4a8c": "Chargement...", + "805cb72cd4": "Demander jusqu'à 15 reviewers", + "a04c137bb7": "Reviewer", + "3bde131f49": "Saisissez ou choisissez un utilisateur", + "d10b6d5209": "Aucun reviewer demandé.", + "ae9a38fd4a": "Retirer le reviewer {{value0}}", + "acbd110867": "Chargement des reviewers", + "00d3be6bcd": "Reviewers", + "f4a4b3fd9f": "A récemment modifié ces fichiers", + "41d275d3ec": "Demander le reviewer {{value0}}", + "36b514a457": "Retirer la demande pour le reviewer {{value0}}", + "c798fa0ec7": "Échec du retrait du reviewer", + "1e6d089420": "Reviewers retirés", + "2c1d93da43": "Reviewer retiré", + "1ae11c905c": "Aucun contexte de dépôt disponible pour cette pull request.", + "102d3d177f": "Reviewers demandés", + "03282ff3b9": "Reviewer demandé", + "8f369a6b6b": "Vous pouvez demander jusqu'à 15 reviewers", + "dace0d1a9f": "Saisissez un reviewer", + "77d9388fb0": "unknown", + "c9e7094a7b": "Reprendre l'espace de travail", + "3b6886b2ee": "Copied", + "b6bda618cf": "compact", + "35a0573f41": "Annotation", + "61a8c69a33": "page", + "a4541fd3db": "Corriger les vérifications cassées", + "c808db1dd1": "Corriger les vérifications", + "c4c02ea23e": "Impossible de créer automatiquement un espace de travail de correction.", + "f119e5f5ef": "Répondre", + "894cfd884b": "Publication…", + "9d5425918e": "Rouvrir la PR", + "96d013ed28": "Fermer la pull request", + "d65f70786e": "closed", + "eca289e593": "Le merge exige un dépôt local enregistré", + "6568ae8ece": "default", + "19f19560d5": "destructive", + "aae99c6c04": "pr", + "e01e34f5fa": "comment", + "345b68254c": "thread", + "31a7b202f2": "Répondre à @{{value0}}", + "408e634fbb": "Répondre dans ce fil de review", + "34b9f7c264": ":L{{value0}}", + "5821aab360": "Échec de la publication de la réponse.", + "84fc40769a": "-L{{value0}}", + "1378d79e83": "Côte à côte", + "e5f4a24f78": "En ligne", + "dd94111c18": "Tout réduire", + "eb722a5a8c": "Tout développer", + "19628e058d": "Échec de l'ajout du commentaire de review.", + "50b8fb290f": "Marquer comme consulté", + "2b4fdb880c": "Ne plus marquer comme consulté", + "56ec6eafb7": "Ouvrez les détails de la PR pour voir les reviewers actuels.", + "7f964a365a": "Retirer le reviewer", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque.", + "8ff5ae8866": "Assignés", + "82c87eceb9": "Modifier les assignés", + "1ff5d979df": "Personne n'est assigné", + "checkActionRequiredHint": "Cette vérification nécessite une action manuelle sur GitHub (par exemple, approuver l'exécution du workflow) avant que le merge soit débloqué.", + "6b1d5ee3e4": "Aucun détail en ligne n'est disponible pour cette vérification.", + "e04c027d98": "Échec du chargement des détails de la vérification.", + "5df7c41d2a": "Réessayer", + "77482513f8": "Fermer", + "2f5195c6a0": "Rouvrir", + "4b8ae7303f": "Échec du chargement du diff.", + "closePullRequestFailed": "Échec de la fermeture de la pull request", + "reopenPullRequestFailed": "Échec de la réouverture de la pull request", + "diffLoadTimedOut": "Délai dépassé lors du chargement de ce diff." + }, + "QuickOpen": { + "1dbd3f59ff": "Déplacer", + "73b2c581f1": "Fermer", + "95fccbae88": "Esc", + "61b1c871a6": "Ouvert", + "250e5b2dfb": "Enter", + "74e2e1b3e4": "Aucun fichier correspondant.", + "722a21e1a8": "Chargement des fichiers...", + "1cb6ef47b7": "Accéder au fichier...", + "9e97f08d0f": "Rechercher un fichier à ouvrir", + "ec31e058f7": "Accéder au fichier", + "73b44e7bde": "Copier la commande d'installation", + "1cf8561ab4": "sur le remote pour activer un listage rapide respectant .gitignore :", + "5d80dc39bb": "ripgrep", + "2ca749c15d": "Installer", + "4725b0e931": "Scan Quick Open trop volumineux (", + "b227d88520": "{{value0}} fichiers trouvés", + "995be8ea22": "Copier", + "cf144856dc": "Copied", + "344f8a48dd": "sur l'hôte exécutant le scan Quick Open pour activer un listage rapide respectant .gitignore :" + }, + "SelectedTextCopyMenu": { + "9b40d7b018": "Copier" + }, + "StarNagCard": { + "92b0f9d921": "soit authentifiée et réessayez.", + "cd8c34aac1": "gh", + "cf82170065": "Impossible de mettre une étoile au dépôt. Vérifiez que", + "30c36231c1": "Orca est open source. Si ça vous a aidé aujourd'hui, une étoile GitHub aide d'autres développeurs à le découvrir.", + "b5e685e4d9": "Ignorer", + "5f6df21046": "Vous appréciez Orca ?", + "2d67b6c849": "Mettre une étoile sur GitHub", + "af3c9bbb37": "Ajout de l'étoile…", + "68a41bc3aa": "Impossible de mettre une étoile avec", + "996bf76e46": "Ouvrez GitHub pour terminer dans votre navigateur.", + "d32015fec7": "Ouverture...", + "157bb5ecbb": "Ouvrir GitHub", + "8c967b4d15": "Plus tard", + "73dfd4eb8d": "Ne plus demander" + }, + "TaskPage": { + "513cddfa7a": "Vérification…", + "ff69a30681": "Annuler", + "2abe22ef76": "Votre token est chiffré via le trousseau du système d'exploitation et stocké localement.", + "246c2b3dd3": "Paramètres du compte Atlassian", + "59c14d34a2": "Créez un token dans", + "b95623e93f": "Token API Atlassian", + "68df347677": "you@example.com", + "163df31e0e": "https://example.atlassian.net", + "33fc2bcb30": "Utilisez l'URL d'un site Jira Cloud, un e-mail Atlassian et un token API pour parcourir les tickets.", + "60f806ce99": "Connecter le site Jira", + "8ff6fdc368": "Création…", + "fc0d8a1fa4": "pour valider.", + "919a20dd5b": "Saisissez {{value0}}", + "56cdb413a2": "Valeurs séparées par des virgules", + "1f0fce91e3": "Sélectionner {{value0}}", + "cbcdcbe244": "Chargement des champs Jira obligatoires…", + "34d97ca682": "Que se passe-t-il ?", + "f161bf9ede": "Description (facultatif)", + "578f730c16": "Résumé court", + "16cba35bee": "Titre", + "ae592fee62": "Type de ticket", + "7d63e2626e": "Chargement...", + "93c57f15e5": "Aucun projet trouvé.", + "cfb56a7868": "Rechercher des projets...", + "00022ec0ba": "Projet", + "0c11ca0b6d": "Nouveau ticket Jira", + "d0ca4aa1d0": "Étiquettes", + "1742eafc14": "Aucun projet", + "69591944e7": "Basse", + "7fd59c18d8": "Moyenne", + "345b169f1f": "Haute", + "f373ab1a4f": "Urgente", + "713179dfdc": "Aucune priorité", + "c8d5bec5f7": "Priorité", + "42a9160321": "Non assigné", + "d2a876ca53": "Assigné", + "154b0fa623": "Statut", + "9bc8aea407": "Ajouter une description...", + "d9151fd4e9": "Titre du ticket", + "4f3cb99f41": "Changer d'équipe", + "c11105dac5": "Nouveau ticket", + "1b59a07674": "Création...", + "cf72580c04": "Rédigez une description, une note de projet ou rassemblez des idées...", + "2ea1c701b6": "Date cible", + "7da41c9225": "Cible", + "09623359b9": "Date de début", + "7d08e8be0f": "Début", + "af9e877f30": "Aucun label", + "d6cda23ef1": "Membres", + "cfaadb6b22": "Aucun responsable", + "34da8ac06c": "Responsable", + "579f98afcd": "Ajoutez un résumé court...", + "ecbcc83140": "Nom du projet", + "b6795e65fd": "Fermer", + "a98cbe7664": "Équipe", + "02f67c0d09": "Nouveau projet", + "bdebffcbfe": "Créez un projet Linear pour l'équipe sélectionnée.", + "1361275ec3": "Nouveau projet Linear", + "7f3f7b4c18": "Description (facultatif, markdown)", + "9f2b4c03a6": "Création dans", + "d3d0998b7d": "Nouveau ticket GitHub", + "d1e243795c": "tickets", + "be8cf68d9f": "voir les tickets", + "67662ade50": "tickets du projet", + "6244a02f46": "Ouvrir dans Linear", + "606a85c774": "Ouvert", + "5e8061b088": "Démarrer un espace de travail depuis {{value0}} {{value1}}", + "592a55611b": "Essayez de sélectionner plus d'équipes ou d'actualiser ; les filtres d'équipe s'appliquent aux tickets déjà récupérés.", + "618107fab3": "Aucun ticket récupéré ne correspond aux équipes sélectionnées", + "903c7af49f": "Aucun ticket Linear trouvé", + "5ed38a49e5": "Consultez l'erreur d'espace de travail ci-dessous, puis actualisez.", + "cc8795e07c": "Impossible de charger les tickets Linear", + "f362667d55": "Mis à jour", + "b1eaa18ace": "Issue", + "37e7ee311e": "Clé", + "b7bae28b6a": "affichés", + "a26a48252e": "Propriétés affichées", + "5d2d835467": "Tri", + "5659da12fc": "Regroupement", + "9c57663908": "Affichage", + "af377b13b1": "Vue {{value0}}", + "d47248df4d": "Mode d'affichage Linear", + "f397d513e3": "Retour", + "b39fe6511d": "projets", + "8675cd6188": "Linear", + "733b8f2421": "Linear / Vues", + "bc06ed0fb0": "Retour aux vues", + "3cb855080f": "vues", + "b4e10f096e": "Propriétaire", + "a04fe7ba73": "Visibilité", + "0aa8525950": "Modèle", + "dfc0c79bd8": "Tickets", + "8a07f21e76": "Santé", + "851017590d": "Ajouter l'accès Linear", + "228b25028f": "Parcourez vos tickets Linear assignés et démarrez le travail directement depuis ici.", + "6d56559467": "Connectez votre compte Linear", + "eee68073b2": "Ouvrir dans Jira", + "9497f2787c": "Démarrer l'espace de travail", + "eba87f2edb": "Aucun ticket Jira trouvé", + "63b2abd3aa": "Tickets Jira", + "e7115334aa": "Masquer Jira", + "83bce6be5c": "Connecter Jira", + "b518ae6307": "Parcourez, modifiez, créez des tickets Jira et démarrez le travail directement depuis ici.", + "a150c59da7": "Connectez votre site Jira", + "bcdc1330b2": "Ouvrir dans GitLab", + "00b7ffb952": "Type / État", + "eb10c32872": "ID", + "e9b6955dcd": "Ticket #{{value0}}", + "a0544fb653": "MR !{{value0}}", + "8396825a14": "Action", + "c1d1600362": "Ouvrir dans le navigateur", + "b6329379ca": "Démarrer un nouveau espace de travail", + "054bf695cc": "Brouillon", + "285bc21dc5": "Modifiez la requête ou effacez-la.", + "d0e3c8f933": "Aucun travail GitHub correspondant", + "5b6b2af943": "Nouvelle tentative…", + "0c0de0fc0e": "Impossible de charger les tickets depuis", + "d1766fd62d": "projets n'ont pas pu être chargés", + "7762f4b03a": "sur", + "443f7dd928": "Merge", + "a7396b05c6": "Vérifications", + "f6fa3c97d0": "Reviewers", + "8aba10579d": "Assignés", + "5eccb3c841": "Titre / Contexte", + "d4c2830063": "Actualiser les éléments de travail GitLab", + "c679af7ad9": "Actualiser mes todos", + "dfd72673e7": "Échec de l'enregistrement de la sélection de projets.", + "b797bdd7c3": "Effacer la recherche", + "99c2755218": "JQL Jira, ex. project = ABC AND statusCategory != Done", + "2ff9fd71fd": "Actualiser les tickets Jira", + "0b65d3fb2c": "Rechercher des projets Linear...", + "eec0c5c079": "Rechercher des tickets Linear...", + "8964184a8b": "Actualiser Linear", + "3feb524d42": "Nouveau ticket Linear", + "0cbf7e5cf3": "Mode tâches Linear", + "ff53631e6f": "Actualiser le travail GitHub", + "6ffa6be99f": "Actualisation du travail GitHub", + "b15ceb409d": "Rechercher des tickets GitHub...", + "eee4df4c66": "Rechercher des PR GitHub...", + "e592d99051": "Tous les sites Jira", + "d09b7631b7": "Échec du changement de site Jira.", + "8029e2bd4d": "Sélectionnez une équipe Linear à ouvrir dans Linear", + "609532fae7": "Échec de l'enregistrement de la source de tâches par défaut.", + "4826fd1ad8": "Fermer · Esc", + "1a06219d5c": "Fermer les tâches", + "3f594861a5": "Échec de l'enregistrement de la sélection d'équipe.", + "d0d570b306": "Échec du changement d'espace de travail Linear.", + "cb98f0350c": "{{value0}} créé", + "1e1b2ad8f2": "Sélectionnez une équipe depuis l'espace de travail du projet avant de créer ce ticket.", + "3ca9b424a3": "Échec de la création du projet.", + "3f9604efc7": "Ticket #{{value0}} ouvert", + "585dba2989": "Impossible d'ouvrir l'espace de travail associé à cette issue.", + "534a9c6017": "Impossible d'ouvrir l'espace de travail attaché à cette pull request.", + "fe380f306c": "Échec de l'enregistrement de la vue de tâches par défaut.", + "af2a8371de": "Échec du chargement des types de tickets Jira.", + "6775c05483": "Échec de la mise à jour de l'état Linear", + "745ae567d4": "« {{value0}} » n'est pas disponible pour {{value1}}", + "669e419d65": "Contexte d'espace de travail manquant pour la vue Linear.", + "cba2a2b7fb": "Contexte d'espace de travail manquant pour le projet Linear.", + "f4374519ae": "Votre source de tickets préférée (upstream) n'est plus configurée pour {{value0}}. Utilisation d'origin.", + "e9139db03f": "Échec du masquage de {{value0}}.", + "b73717af92": "Suivant", + "0c8df28045": "Page suivante", + "ae859c816b": "Page {{value0}}", + "cd171f3391": "...", + "297a805b64": "Précédent", + "6cd6b3ae6a": "Page précédente", + "e65757a338": "Pagination", + "37d60046e3": "Ouvrir la zone de merge GitHub", + "1a9ea003dc": "Échec de la désactivation de l'auto-merge", + "a3318684bc": "Échec de l'activation de l'auto-merge", + "a5bf86defe": "Auto-merge désactivé", + "fed317634c": "Auto-merge activé", + "88f478cdef": "Échec de la fusion de la pull request", + "a161925adc": "Pull request fusionnée", + "0506a78337": "Cela mettra à jour la pull request sur GitHub.", + "844dc193c7": "{{value0}} la PR #{{value1}} ?", + "995dd6af9b": "Ouvrir les vérifications de la PR", + "8a22eb3f7b": "Aucun reviewer correspondant.", + "67755a83a1": "Tous les autres", + "3ace2e6bcf": "Suggestions", + "0eacf48491": "Chargement…", + "0b9b04f4b5": "Saisissez ou choisissez un utilisateur", + "62c7bd789f": "Demander jusqu'à 15 reviewers", + "5d4fd69a6a": "Actifs récemment dans cette pull request", + "ed1daeb49a": "Échec du retrait du reviewer", + "837bb901ec": "Reviewers retirés", + "f9191d1714": "Reviewer retiré", + "dc67f69962": "Échec de la demande de reviewer", + "8f06dbb9e5": "Reviewer demandé", + "969e26577c": "Vous pouvez demander jusqu'à 15 reviewers", + "d00571d9b1": "Saisissez un reviewer", + "edf4bc4135": "Aucun utilisateur assignable.", + "53e002d895": "Le ticket n'a pas de slug de dépôt.", + "7f94eb6395": "Assigner le ticket", + "bb63046423": "Assigné à {{value0}}", + "ca63694b4c": "Échec de la mise à jour des assignés.", + "b36f4bf9de": "Aucun label.", + "5ebff3a0aa": "Aucun", + "d09bf34db7": "Fermé", + "1c893195ac": "Échec de la mise à jour de l'état", + "afc68824ff": "Aucun état trouvé", + "cc13109b5d": "Chargement des états", + "d45a910c4a": "Changer l'état Linear depuis {{value0}}", + "d8a517ad89": "Identifiant", + "50387522d7": "Aucun regroupement", + "d747aed72f": "Tableau", + "a6f7e93d7f": "Liste", + "e78ec261ed": "Vues", + "727069bee5": "Projets", + "137e2a8a01": "PRs", + "18451e99df": "Terminé", + "4b6e40e42c": "Tous les ouverts", + "bd9965df51": "Signalés", + "1301d376f1": "Assignés", + "9cd11ba218": "Jira", + "11a828abf8": "GitLab", + "acef77f7ca": "GitHub", + "524f095d55": "En attente de relecture", + "7698af5263": "À moi", + "94f0339621": "Assignés à moi", + "c2268a9982": "Tous", + "37a82eaaf8": "Fusionnés", + "887efe9140": "Se connecter", + "a70153f583": "connexion en cours", + "6459faa8b3": "error", + "e15ba2d2eb": "Créer un ticket", + "3b11c8e8fc": "tableau", + "e178c0a953": "Choisissez un projet Jira avant de créer le ticket.", + "0f7b0d964a": "Crée un nouveau ticket dans {{value0}}.", + "eff9800d4b": "{{value0}} label{{value1}}", + "d7f16d0e32": "Sélectionner l'équipe", + "5301ca0f20": "Créer le projet", + "7719d8daa9": "{{value0}} membre{{value1}}", + "5af6f0ae5b": "Sélectionner l'équipe", + "cfb730b73e": "issue", + "3659f9792a": "tableau", + "d079be2dc8": "Aucun ticket assigné. Essayez de rechercher autre chose.", + "2bdefbcac3": "Essayez une autre requête de recherche.", + "25ff84769a": "Aucun ticket ne correspond à ce contexte Linear.", + "cbce2bc9cd": "aucun", + "51411113df": "liste", + "60f68a2ef4": "Issues Linear", + "b2007ba885": "projet", + "6edf402e11": "vue d'ensemble", + "9ae151b26b": "linear", + "94d900518d": "Aucune issue ne correspond au préréglage sélectionné.", + "f51e254d35": "Essayez une requête JQL différente.", + "4645a7814f": "jira", + "e224d76876": "MR", + "bbec4717ee": "mr", + "d6d08c1650": "Sélectionnez un projet pour voir les work items GitLab.", + "f294c500ef": "Aucun work GitLab ne correspond à ce filtre.", + "cd7dc432a3": "Aucune MR GitLab ne correspond à ce filtre.", + "171b7739d8": "mrs", + "a9f256ecea": "Aucune issue GitLab ne correspond à ce filtre.", + "2007e14d95": "gitlab", + "456b8512da": "MergeRequest", + "03da966159": "Sélectionnez un projet afin que nous puissions nous authentifier auprès de GitLab.", + "d591aac6ae": "Aucun todo en attente. Vous êtes à jour !", + "33a4bd7f5c": "todos", + "66ae7330f6": "Plus d'actions", + "93d5f21fc1": "pr", + "e104fa3d3d": "Démarrer un espace de travail à partir de l'issue", + "2193a99ec1": "Ouvrir l'espace de travail associé à l'issue", + "7deb9e59a5": "Plus d'actions sur la PR", + "7753652524": "Reprendre", + "e4b29c5bcf": "Démarrer un espace de travail à partir de la PR", + "67d881244c": "Reprendre l'espace de travail attaché à la PR", + "6430594b18": "auteur inconnu", + "7799ad9ab6": "brouillon", + "0bfbf62f75": "Réessayer", + "38139edb52": "github", + "31f81cc334": "Actualisation des works GitHub…", + "3d93316bb0": "prs", + "937b29fa35": "éléments", + "bc46d8204e": "Sélectionnez un seul projet à ouvrir dans GitHub", + "d1132848f8": "Sélectionnez un seul projet GitHub à ouvrir dans GitHub", + "2af3ab5c58": "Sélectionnez une seule équipe à ouvrir dans Linear", + "aec5feeb69": "Échec de la création de l'issue Jira.", + "7437e340b4": "Échec de la création de l'issue.", + "9e03c17847": "Ouvrez les détails de la PR pour voir les reviewers actuels.", + "3b7f34282f": "closed", + "246bd64aed": "Ouvrir {{value0}} dans Linear", + "ff90d0abc7": "Démarrer un espace de travail à partir de {{value0}}", + "fe28c9821f": "vue", + "8d1e17a3ef": "Ouvrir {{value0}} dans GitHub", + "4ac8ff2275": "Ouvrir {{value0}} dans Jira", + "40eaf2c27c": "Détails", + "closeAsCompleted": "Fermer comme terminée", + "closeAsCompletedDescription": "Terminée, fermée, corrigée, résolue", + "closeAsNotPlanned": "Fermer comme non planifiée", + "closeAsNotPlannedDescription": "Ne sera pas corrigée, non reproductible, obsolète", + "closeAsDuplicate": "Fermer comme doublon", + "closeAsDuplicateDescription": "Doublon d'une autre issue de ce dépôt", + "duplicateIssueNumberPlaceholder": "Numéro d'issue", + "closeAsDuplicateSubmit": "Fermer le doublon", + "duplicateIssueMissing": "Saisissez un numéro d'issue de ce dépôt.", + "duplicateIssueNotInteger": "Utilisez un numéro d'issue entier.", + "duplicateIssueNotPositive": "Utilisez un numéro d'issue positif.", + "duplicateIssueSameIssue": "Choisissez une autre issue.", + "repository": "Dépôt", + "backToCloseReasons": "Retour", + "searchIssues": "Rechercher des issues", + "useIssueNumber": "Utiliser l'issue #{{value0}}", + "noMatchingIssuesLoaded": "Aucune issue correspondante n'a été chargée.", + "editReviewersWithCurrent": "Modifier les reviewers : {{value0}}", + "linearEmptyAttributeFilter": "Aucune issue ne correspond aux filtres sélectionnés. Effacez un filtre ou essayez d'autres critères.", + "linearEmptyUnfilteredScope": "Aucune issue dans le périmètre de cet espace de travail. Essayez de rechercher ou d'ajuster les équipes.", + "linearFetchMore": "Charger plus", + "jiraSortAscending": "croissant", + "jiraSortDescending": "décroissant", + "jiraSortBy": "Trier par", + "jiraToggleSortDirection": "Trier : {{value0}}", + "75a38d7df8": "Les données GitHub sont temporairement indisponibles. Son API est peut-être en panne, limitée en débit ou injoignable. Réessayez dans quelques instants.", + "noGithubSourceDetected": "Aucune source GitHub détectée pour", + "noGithubSourceDetectedHint": "il n'a peut-être aucun remote GitHub, ou la source n'a pas pu être résolue.", + "jiraLinkSourceUnavailable": "Impossible de lier cette issue Jira. Reconnectez Jira ou sélectionnez le site correspondant, puis réessayez.", + "loadPageUnreachable": "La page {{value0}} dépasse ce que la recherche GitHub peut renvoyer.", + "loadPageFailed": "La page {{value0}} n'a pas pu être chargée depuis GitHub.", + "loadPageNoMoreResults": "Aucun autre résultat sur la page {{value0}}.", + "linearHasWorktreeLoadFailed": "Impossible de charger les issues Linear liées à un espace de travail Orca.", + "linearHasWorktreePartialLoadFailed": "Certaines issues Linear liées à un espace de travail Orca n'ont pas pu être chargées. Actualisez pour réessayer.", + "linearHasWorktreeSearchPlaceholder": "Filtrer les issues liées à un espace de travail Orca...", + "linearModeHasWorktree": "Avec espace de travail", + "linearModeHasWorktreeTooltip": "Tickets Linear liés à un espace de travail Orca", + "linearEmptyHasWorktree": "Aucun ticket Linear n'est encore lié à un espace de travail Orca. Démarrez le travail depuis une issue Linear pour le voir ici.", + "linearOpenAttachedWorkspace": "Ouvrir l'espace de travail attaché à {{value0}}", + "linearWorktreesColumn": "Espaces de travail" + }, + "Terminal": { + "73768427cf": "Fermer", + "f82e9f02df": "Annuler", + "7958465754": "Des terminaux locaux exécutent des processus. Fermer quand même la fenêtre ?", + "2fa9c69ff3": "Fermer la fenêtre ?", + "cd51e28d8b": "Enregistrer", + "0037b21794": "Ne pas enregistrer", + "21295c6b8c": "Modifications non enregistrées", + "5c1d2a32bb": "Chargement de l'éditeur...", + "f0600556b3": "Échec de la création du fichier markdown sans titre.", + "37da0d736f": "Nouvel onglet de navigateur", + "a2a279b32a": "L'enregistrement a expiré ou échoué. Corrigez les erreurs avant de fermer.", + "46e08bc5c8": "Ce fichier contient des modifications non enregistrées.", + "61ed600d29": "« {{value0}} » contient des modifications non enregistrées. Voulez-vous enregistrer avant de fermer ?", + "cdc9ac4b2d": "éditeur", + "e57db40c11": "Impossible de construire la commande de lancement pour {{value0}}.", + "5b2c1a9e44": "Aucun CLI d'agent détecté — installez-en un ou choisissez un agent par défaut dans les paramètres." + }, + "TerminalSearch": { + "db234b7519": "Fermer", + "7cb40c04eb": "Correspondance suivante", + "0f3066256e": "Correspondance précédente", + "42e466b9f1": "Regex", + "90c61387d9": "Respecter la casse", + "e07012f26e": "Rechercher..." + }, + "UpdateCard": { + "68b235d264": "Redémarrer pour mettre à jour", + "6714206e5a": "Orca v{{value0}} est téléchargée. Redémarrez quand vous êtes prêt.", + "93794ea932": "Orca v{{value0}} est en cours de téléchargement.", + "8acbdd3961": "Réduire dans la barre d'état", + "17412483da": "Prêt à installer", + "47126bcf57": "Télécharger manuellement", + "3553a8672f": "Dernière erreur", + "90559b14e3": "Active un commutateur réseau Electron à l'échelle du processus après redémarrage. Utilisez-le pour les VPN d'entreprise ou les proxys qui rejettent les téléchargements de mises à jour en HTTP/2.", + "6e45bfa2e0": "Téléchargement...", + "558842597d": "Téléchargement de la mise à jour", + "f58b5c57a6": "Nouveau :", + "ec8fe71cfc": "Mettre à jour", + "44324ef542": "Notes de version", + "fdd4a364fa": "Les sessions ne seront pas interrompues.", + "05ad78a6d1": "Orca v{{value0}} est prête.", + "318d3b4bc7": "Ignorer la mise à jour", + "9abc59f814": "Mise à jour disponible", + "aad383aecc": "Lire toutes les notes de version", + "ccd8b0a793": "en plus depuis votre dernière mise à jour", + "b1d867f4fb": "Vos sessions de terminal ne seront pas interrompues pendant la mise à jour.", + "09a55c39b5": "Installation...", + "ea2a41adbe": "Vous utilisez déjà la dernière version.", + "ba5ffc949c": "Recherche de mises à jour...", + "2c2d3e03ca": "Réessayer", + "4cf109845a": "Erreur de mise à jour", + "6b0085010d": "Revérifier", + "48565a32bc": "Réessayer le téléchargement", + "933c6fdf5b": "Activer et redémarrer", + "1339b82cee": "Téléchargement HTTP/2 bloqué", + "7274ef6e59": "Ignorer l'astuce", + "a726967bd3": "Ignorer", + "d5253b54af": "error", + "7ffc08506e": "check", + "522df222b9": "spinner", + "e944c2de43": "Vérification de la mise à jour bloquée", + "5b309b19f3": "La mise à jour n'a pas été installée", + "092f09fc14": "L'éditeur de l'installateur ne correspond pas à Orca ; la mise à jour a donc été stoppée. N'installez pas ce téléchargement ; consultez les versions officielles pour obtenir une version corrigée.", + "c9ff9b9ec2": "Consulter les versions officielles", + "a05992a26b": "La vérification de signature n'a pas pu s'exécuter — généralement parce qu'un logiciel antivirus l'a bloquée. Relancez le téléchargement ou récupérez l'installateur depuis nos versions officielles.", + "5194358929": "Masquer les détails", + "8bc9e17d8f": "Afficher les détails", + "8cf17b10af": "Erreur de build local", + "a4650b0dc4": "Impossible d'utiliser le build local", + "b1e390250d": "Impossible de finaliser le basculement vers le build local.", + "d29740d175": "Le build sélectionné n'a pas pu être utilisé.", + "37d45c9ec1": "Choisir un autre build" + }, + "WorktreeJumpPalette": { + "ac037cfac2": "Déplacer", + "75499e01d9": "Fermer", + "66b5a67bee": "Esc", + "45def60329": "Ouvert", + "f65d992a11": "Enter", + "c5081f2814": "Worktree actuel", + "52404f8096": "Onglet actuel", + "739bda980c": "principal", + "556e7232ca": "Actuel", + "684e8d7bc2": "Collecte de vos worktrees récents et de vos onglets ouverts.", + "ff908adfe9": "Chargement des cibles de navigation", + "4ee378034d": "Aller à...", + "f7fda8d562": "Créez un worktree ou ouvrez un onglet dans Orca pour commencer.", + "1628fd7dfa": "Aucun worktree actif, paramètre, action ni onglet ouvert", + "b781ae05e3": "Saisissez pour rechercher parmi les worktrees, paramètres, onglets et actions.", + "f60f8730be": "Aucun autre worktree vers lequel basculer", + "c4afa68159": "Essayez un worktree, un paramètre, une action, un titre d'onglet, un prompt d'agent, une URL, une PR ou un port.", + "dbd9d87eec": "Aucun résultat ne correspond à votre recherche", + "2c38630a01": "L'espace de travail n'existe plus", + "7726ce9970": "L'onglet émulateur mobile n'existe plus", + "d7d496a451": "La page du navigateur n'existe plus", + "50a1d11d5b": "Onglets ouverts", + "088d66d980": "Actions et paramètres", + "dabd819ca1": "Saisissez pour voir les {{value0}} worktrees", + "20af998bff": "{{value0}} éléments disponibles{{value1}}", + "bb72c08e63": "{{value0}} résultats trouvés{{value1}}", + "34c8fbb46e": "Remote SSH", + "63c2be1914": "SSH déconnecté", + "95be6587d3": "Créer le worktree « {{value0}} »", + "worktreesHeader": "Worktrees", + "recentWorktreesHeader": "Worktrees récents", + "settingsBadge": "Paramètres", + "actionBadge": "Action", + "paletteHostBadge": "Hôte : {{value0}}", + "workspaceTabMissing": "L'onglet n'existe plus", + "projectsGroupsHeader": "Projets et groupes", + "projectBadge": "Projet", + "repoGroupBadge": "Groupe de dépôts", + "pluginCommandFailed": "Impossible d'exécuter la commande du plugin.", + "recentChatsTerminalsHeader": "Chats et terminaux récents", + "27f10cca63": "Rechercher chats, terminaux, worktrees, paramètres et actions...", + "2770f02910": "Rechercher chats, terminaux, worktrees, paramètres et actions", + "paletteOpenTabBranch": "Nom de branche", + "paletteOpenTabWorkspace": "Nom de l'espace de travail", + "lastActiveTime": "Dernière activité il y a {{value0}}" + }, + "github": { + "pr": { + "merge": { + "state": { + "a80132573b": "GitHub calcule encore le statut de merge de cette pull request", + "f958920f3a": "Vérification", + "9bd983ce8f": "GitHub indique que cette PR peut merger, mais des checks sont encore en cours", + "4e2507176b": "Vérifications en attente", + "1432ecff30": "GitHub indique que cette PR peut merger, mais certains checks ont échoué", + "87fa36ac83": "Vérifications échouées", + "1766eb46ba": "GitHub signale que cette pull request est bloquée", + "bf5e4c6c92": "Bloquée", + "c614e2660a": "Mettez à jour la branche avant de merger", + "039c072f94": "En retard", + "b37d45bca9": "GitHub signale des conflits de merge", + "7e8bbe3cd7": "Conflits", + "09896aad26": "Statut de merge indisponible pour cette PR", + "bd4f27b50e": "Merge", + "35ec24bc43": "Cette branche de base utilise la merge queue GitHub", + "b289646bcd": "GitHub signale des changements demandés sur cette pull request", + "c606463dc2": "Modifications demandées", + "a20db875ed": "GitHub exige une review approuvée avant que cette pull request puisse merger", + "1f8eb81c0e": "Approbation requise", + "f03028e055": "Cette pull request est toujours en brouillon", + "ec8e2cebaa": "Brouillon", + "820fd21663": "Cette pull request est fermée", + "4f976d3450": "Fermé", + "62eb8d39da": "Cette pull request est déjà mergée", + "83ecdbb4a6": "Fusionnés", + "331ebe1170": "Ajouter cette pull request à la merge queue GitHub", + "b169f943e1": "Merger quand prêt", + "62703b1dc4": "L'auto-merge GitHub est activé pour cette pull request", + "48d75ae118": "Désactiver l'auto-merge", + "a5b66afb58": "Checks réussis", + "fbd4f57f0a": "Checks réussis. L'éligibilité au merge sera revérifiée avant le merge.", + "4ab19a62ef": "Activer l'auto-merge", + "8f6cb3772f": "Merger automatiquement cette pull request une fois les conditions remplies" + } + } + }, + "project": { + "ColumnResizeHandle": { + "1304289353": "Redimensionner la colonne" + }, + "GhAuthErrorHelp": { + "7e800068d8": "Recharger", + "baa006f9af": "Docs", + "3fefeebde4": "Copier la commande d'actualisation", + "9c2da6353b": "Copier la commande de connexion", + "b436c586d1": "Copier la commande", + "8a7f6bf5dc": "Échec de la copie", + "224c9d0ae8": "Copié dans le presse-papiers", + "891a7d4616": "Retirer pour ce shell", + "fd17b3019f": "Retirer (PowerShell, persistant)", + "ae43542893": "Trouver où elle est définie", + "df636f5886": "Vérifier si elle est définie (PowerShell)" + }, + "ProjectCell": { + "4b5b871da8": "Aucun label dans ce dépôt.", + "2219e945ef": "Chargement…", + "54cac64427": "La ligne n'a pas de slug de dépôt.", + "8ae56a88a6": "Étiquettes", + "f7cdb78efb": "Assignés", + "ebde486e3c": "Effacer", + "191905e20e": "Actuels et à venir", + "e17bb96881": "Terminé", + "943b3dadc9": "Ce dépôt n'a aucun type d'issue.", + "c7b059cf07": "Type de ticket", + "c5f949e489": "Issue", + "8d669084f6": "Restreint", + "6efdc0d920": "Brouillon", + "d0d0e13a5a": "PR", + "af5d8c912a": "Élément restreint", + "bb7ebc11e3": "Ajouter un nombre", + "9cb1a0c984": "Ajouter du texte", + "2e26a06c70": "Ajouter un label", + "36341ffc66": "Assigner", + "e369bf4fec": "Sélectionner", + "ffeff79861": "PULL_REQUEST" + }, + "ProjectGroupHeader": { + "82a22d2079": "Actuel", + "244c9e7d06": "Tous" + }, + "ProjectItemSlugDialog": { + "e55a5c4e68": "Aperçu de la ligne de projet.", + "4450efea9c": "Élément GitHub" + }, + "ProjectPicker": { + "96739284c3": "Collez ci-dessous l'URL d'un projet pour accéder à ceux qui manquent.", + "9b36829267": "Aucune vue trouvée.", + "72a05c04a6": "Chargement des vues…", + "9bf55fa1e8": "Choisir une vue", + "a51b3337ab": "← Retour", + "8ab5447c64": "Épingler", + "5009ffc2f3": "Retirer l'épingle", + "fce99a24a7": "Ajouter", + "5113ecc298": "Ajouter par URL ou propriétaire/numéro", + "7b6d39627e": "Chargement…", + "b787682111": "Tout parcourir", + "ba0ab9a117": "Tout parcourir (chargement…)", + "b3044b7a25": "Récents", + "707843206c": "Épinglés", + "f492e1b539": "Rechercher des projets", + "44b2c6326b": "Échec du chargement des vues : {{value0}}", + "ab1a2c357d": "Roadmap (non prise en charge)", + "d34ef9b554": "Board (non pris en charge)", + "43a88ae574": "BOARD_LAYOUT", + "1a2b8e512e": "Tableau", + "cafb908f34": "TABLE_LAYOUT" + }, + "ProjectRow": { + "75b5d816e3": "Démarrer le travail", + "e12be8b4d4": "Ouvrir dans GitHub", + "c3b81ddea2": "DRAFT_ISSUE" + }, + "ProjectViewList": { + "989f81dc2a": "Colonnes", + "f949f5b2b7": "Configurer les colonnes", + "eddfc7a794": "Trier par {{value0}}", + "4f57d2e0b1": "Aucun élément ne correspond au filtre de cette vue." + }, + "ProjectViewWrapper": { + "463f1205c0": "Chargement de la vue du projet", + "23b87ba9f7": "Ouvrir dans GitHub", + "4d2a77a119": "Soumettre une demande de fonctionnalité", + "1bf8c01c8b": "Passez en vue Tableau pour travailler sur ce projet dans Orca.", + "55de4fb57a": "{{value0}}. {{value1}} Soumettez une demande de fonctionnalité sur {{value2}}.", + "2edf5e7e77": "{{value0}} — Orca ne prend pas encore en charge les vues de projet {{value1}}. Soumettez une demande de fonctionnalité sur {{value2}}.", + "7245c3d7ac": "Effacer la recherche", + "c5bc7ec007": "Filtre de la vue : {{value0}}", + "840c268665": "Ajouter un dépôt", + "dffa899f36": "Annuler", + "7037c8f5f1": "Dépôt absent d'Orca", + "512fc171d6": "Choisissez un projet pour commencer.", + "71fb69926c": "Actualiser", + "a8fa0d2bf5": "Actualisation", + "fd15491034": "Ouvrir la vue dans GitHub", + "22df63c393": "Les données de sub-issues ne sont pas disponibles avec votre token.", + "067119985c": "Recherche GitHub, ex. assignee:@me is:open", + "1850fceac8": "{{value0}}/{{value1}} n'est pas ajouté à Orca. Ajoutez-le pour démarrer le travail, ou ouvrez-le dans GitHub.", + "1aa7c952b9": "Vue de projet", + "f352abf7c3": "La liste des dépôts est en cours de mise à jour.", + "1ce21b8cff": "Cet élément est hors des dépôts sélectionnés.", + "030de75bc5": "Cet élément correspond à plusieurs dépôts sélectionnés." + }, + "slug": { + "dialog": { + "AssigneesEditor": { + "529fec247b": "Chargement…", + "98914e6b36": "Assignés :", + "94a4e6e4fa": "aucun" + }, + "Comments": { + "fd5cccd138": "Commentaire", + "1c95937c8b": "Écrire un commentaire…", + "c0e576e96b": "Annuler", + "c3e829b4d9": "Enregistrer", + "463d030ae4": "Supprimer", + "8564f58542": "Édition", + "5f104bf855": "Aucun commentaire pour le moment.", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque." + }, + "LabelsEditor": { + "34dd57d6c8": "Chargement…", + "a7b182fcda": "Labels :", + "1a5366b5be": "aucun" + }, + "SlugDialogBody": { + "598ad6a517": "Commentaires", + "41169e41fb": "Ajouter une description…", + "a91735d19f": "Annuler", + "e64f6c3eff": "Enregistrer", + "e4ef8281e9": "Chargement…", + "ae98897edf": "Fermer", + "69caf40ae8": "Ouvrir dans GitHub", + "7c302f8174": "Sans titre" + } + } + } + }, + "GitHubMarkdownComposer": { + "015b4e607d": "Annuler", + "e3bd59143c": "Insérer", + "f24783f470": "https://...", + "ec6310b731": "Utilisez une URL d'image en http:// ou https://.", + "b7e4a1c902": "Collez, déposez ou cliquez pour ajouter des fichiers", + "8f1c2d4e6a": "Rien à prévisualiser", + "c91f0a2b14": "Écrire", + "d82b1e3f05": "Aperçu", + "imageUrlTooLarge": "L'URL de l'image est trop volumineuse." + }, + "IssueSourceSelector": { + "d6aeb2012b": "Affichage des issues de", + "787c970baf": "Source des issues", + "643d7e9496": "upstream", + "51d1608920": "Origine", + "cdc9bd64fa": "compact", + "30b2c9df91": "Upstream" + }, + "PRFilterDropdowns": { + "979be3cf6b": "Assigné", + "b27b7e526c": "Review de", + "7f1ba66c3e": "Review effectuée par", + "9d0f2eda6d": "Label", + "01f3f3d161": "Auteur", + "13b3ac0a84": "Statut", + "79c54552f7": "Filtres", + "8a2ffbf9b3": "Retirer le filtre {{value0}}", + "19bb6f115f": "reviewed-by" + }, + "PRFilterPickers": { + "fdf387297c": "Effacer (", + "472c12ae03": "Effacer", + "2d1f58eda6": "Utiliser" + }, + "PRFilterSections": { + "a00830d3f7": "Aucun utilisateur", + "0103e1cb18": "Review effectuée par", + "94b42b0edf": "Review demandée", + "de26e2eb06": "Aucun label", + "458ea3602b": "Aucun auteur", + "b69fa4fa20": "Retour", + "30ebb6ca44": "Effacer tous les filtres", + "8177eda37e": "Filtre", + "ea3416d646": "Assigné", + "b1d9fdea08": "Label", + "24754c44ad": "Auteur", + "764a0b4ce1": "Statut", + "f0cf6dd591": "désactivé", + "1e9b5244f2": "activé", + "b930de7194": "Brouillons uniquement", + "e0002f1eba": "sélectionnés", + "2b2f019091": "Tout état", + "0fd3249e2e": "Fermé", + "d78b60b5c2": "Ouvert", + "bd162b7d5a": "Fusionnés", + "2e639b84fa": "reviewer", + "712c5abdbf": "label", + "4e50c7bc03": "assigné", + "7bf3a6e5ac": "auteur", + "66256a73b3": "statut", + "3ce4d5e96e": "prs" + }, + "github": { + "rate": { + "limit": { + "display": { + "5509443543": "Chargement du quota d'API GitHub…", + "34973d4695": "Le quota d'API GitHub est indisponible.", + "d12d3d6f33": "Actualiser le quota d'API GitHub", + "d5e5de9070": "Orca utilise REST, Search et GraphQL via GitHub CLI.", + "58c5f88216": "Quota d'API GitHub", + "6da1858354": "restant · réinitialisation dans", + "f42790d150": "sur", + "01f7323e58": "API GraphQL", + "1daf0f22a9": "GraphQL", + "1f2f28a4de": "API Search", + "c377a4f06a": "Recherche", + "c392c749a6": "API REST", + "bb227706a6": "REST", + "budget_scope_prefix": "Périmètre du quota" + } + } + } + }, + "CloseReasonDropdown": { + "e1f2a3b4c5": "Choisir le motif de clôture" + }, + "GitHubIssueCommentComposer": { + "082515176a": "Échec de l'ajout du commentaire", + "9f88657c4e": "Issue fermée", + "e9b7cb7d17": "Échec de la fermeture de l'issue", + "bd3b4492a0": "Issue rouverte", + "f2a8c1d903": "Échec de la réouverture de l'issue", + "a1b2c3d4e5": "Ajouter un commentaire", + "c5c117270e": "Ajoutez votre commentaire ici, soyez bienveillant", + "f6a7b8c9d0": "Fermer l'issue", + "b1c2d3e4f5": "Rouvrir l'issue", + "0a73f59e85": "Envoyer le commentaire", + "bf43425540": "Commentaire", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque." + }, + "GitHubWorkItemAssigneePopoverContent": { + "cddd9b04a7": "Chargement des assignés", + "a00830d3f7": "Aucun utilisateur", + "4f8b6f2c1d": "Filtrer les assignés..." + }, + "GitHubWorkItemLabelPopoverContent": { + "2aa9acdf34": "Modifier les étiquettes sur GitHub", + "cddd9b04a7": "Chargement des étiquettes", + "de26e2eb06": "Aucun label", + "8b0d52ee3a": "Filtrer les labels..." + }, + "githubIssueCloseReasons": { + "completed": { + "label": "Fermer comme terminée", + "description": "Terminée, fermée, corrigée, résolue" + }, + "notPlanned": { + "label": "Fermer comme non planifiée", + "description": "Ne sera pas corrigée, non reproductible, obsolète" + }, + "duplicate": { + "label": "Fermer comme doublon", + "description": "Doublon d'une autre issue" + } + }, + "CommentReactions": { + "addReaction": "Ajouter une réaction", + "removeNamedReaction": "Retirer la réaction {{value0}}", + "addNamedReaction": "Ajouter la réaction {{value0}}" + } + }, + "linear": { + "api": { + "key": { + "dialog": { + "834a52c084": "Vérification...", + "f8f704a019": "Annuler", + "e603ee9156": "Paramètres d'API de l'espace de travail", + "dc7ccb0f7c": "Clés d'API personnelles", + "e3100b36b9": "Si les clés d'API des membres sont bloquées, demandez à un administrateur de l'espace de travail de les autoriser depuis les paramètres d'API de l'espace de travail.", + "d56d3629f4": "Privilégiez l'accès complet lorsqu'Orca doit afficher toutes les équipes accessibles au compte dans cet espace de travail. Les clés restreintes n'exposent que les équipes autorisées, et les équipes privées exigent que le propriétaire de la clé y ait accès.", + "af52a6227f": "Créez une clé d'API personnelle depuis Account > Security & Access.", + "edec49dfae": "lin_api_...", + "7d498f653c": "Clé d'API personnelle", + "57a66522c8": "connexion en cours", + "c9889a09f8": "Utilisez Linear pour choisir l'espace de travail souhaité avant de créer la clé.", + "e689a4d0a6": "error" + } + } + }, + "priority": { + "icon": { + "c43d3e065b": "Priorité :" + } + }, + "project": { + "view": { + "surfaces": { + "8bbecb2510": "Aucun", + "e1fa97d21d": "Sélectionnez un projet pour afficher sa vue d'ensemble.", + "1748d3b9af": "Étiquettes", + "65bda65159": "Membres", + "c5f79616c3": "Équipes", + "25a2196732": "Cible", + "3fb6473111": "Début", + "111bef9aa8": "Responsable", + "3be47aed6f": "Priorité", + "f5ef24cf46": "Santé", + "9ddb58edbd": "Statut", + "0a6a5a7dd6": "Dernière mise à jour", + "c8db98b73b": "Ressources", + "bb1405eff8": "Jalons", + "5d99315fb8": "Planification", + "3ad562bdf4": "issues du périmètre", + "563501f191": "Progression", + "bb5664d456": "Aucune description de projet.", + "7b147907dc": "Linear", + "a9785c7158": "Actualiser", + "ee3d2caabd": "Tickets", + "5f79bc76b0": "Retour aux projets", + "aac9a4afc6": "Ouvrir dans Linear", + "7616c986c6": "Ouvrir {{value0}} issues", + "93e1f6bfca": "Chargement", + "98730088a6": ". Recherchez ou ouvrez Linear pour voir l'ensemble complet.", + "06b887d622": "Affichage des premières", + "2c4b1c2c08": "nombre", + "f2cc1e0ff6": "Linear / Projets", + "906b5e4cb8": "Linear / Projets / {{value0}}", + "85607ff793": "Projet", + "20b9d09b7d": "Inconnu", + "f059181bd9": "Privé", + "27d91cb1a6": "Partagé", + "9f0f51fd9e": "Créez ou enregistrez des vues dans Linear, puis actualisez.", + "f4c79cff5f": "Consultez l'erreur d'espace de travail ci-dessous, puis actualisez.", + "ef90b21366": "Aucune vue trouvée", + "c0a50f96a4": "Impossible de charger les vues", + "df4bd63c1d": "Non assigné", + "30402d2c6e": "Essayez la recherche ou actualisez.", + "a2f31c4cd6": "Aucun projet Linear trouvé", + "c9b6e9f90d": "Impossible de charger les projets Linear" + } + } + }, + "scope": { + "selector": { + "91c8871dad": "Ajouter un accès d'équipe", + "7783361266": "Toutes les équipes", + "e1ae6bebb0": "Équipes", + "a14ce4df2b": "Tous les espaces de travail", + "05baa5ae90": "Espace de travail", + "89f6580dbf": "Rechercher des équipes...", + "b3488fad3c": "Aucune équipe n'a été récupérée. L'accès peut dépendre du périmètre de la clé, de l'appartenance à des équipes privées, d'équipes archivées, des permissions ou d'un échec de récupération.", + "405b33c378": "Aucune équipe récupérée ne correspond à votre recherche." + } + } + }, + "notification": { + "sound": { + "options": { + "e38b0a2e68": "Beep", + "0acd3d384e": "Clack", + "79919c832d": "Ding", + "2b44847d8d": "Blop", + "020826ef17": "Sonar", + "588c90487d": "Blip", + "1e4b81d892": "Thump", + "86af8d938c": "Bong", + "80f7cc95b3": "Two Tone", + "017abebfa6": "Son système par défaut" + } + } + }, + "worktree": { + "creation": { + "WorktreeCreationPanel": { + "dabd226118": "Ignorer", + "34dd5ee38b": "Réessayer", + "ed2a664f8b": "Impossible de créer le worktree", + "a3346fc6ed": "Annuler la création du worktree", + "532aea14ce": "Annuler", + "767951265d": "Un problème est survenu lors de la création du worktree.", + "vmProvisioningTitle": "Provisionnement de la VM", + "cancelProvisioning": "Annuler", + "vmProvisioningLogEmpty": "En attente de la sortie de la recette…" + } + } + }, + "workspace": { + "space": { + "WorkspaceSpacePage": { + "8d0048e1cb": "Utilisation du disque de l'espace de travail et stockage worktree récupérable.", + "e8d6ba11ab": "Beta", + "45f6302dbc": "Espace", + "ecf72fdc3b": "Retour" + } + }, + "cleanup": { + "WorkspaceCleanupDialog": { + "3828408538": "Supprimer {{value0}}", + "352f15d6fc": "Dernière activité", + "cbf2f664e2": "Supprimer", + "b6bae1eed1": "Annuler", + "592fbab446": "Trié par activité la plus ancienne", + "dba753e94f": "à supprimer", + "38ca0b1400": "Cette opération supprime définitivement leurs fichiers locaux. Vous ne pourrez pas annuler.", + "c4f4782c02": "Non suggéré", + "0a2e3c7cba": "À examiner", + "e97e4580c7": "Modifié", + "9623a5107d": "Commits non poussés", + "e8b3741ff7": "Ignoré", + "a9957007eb": "Ignorer {{value0}}", + "1bffc07ba7": "Voir {{value0}}", + "bef0adef9b": "Branche", + "0b1766738a": "Dépôt", + "bbb1ab6a6f": "Sélectionner {{value0}}", + "d1094dd529": "En attente de relecture", + "4b93a235d8": "Suggéré", + "f68d538c63": "Aucun espace de travail dans cet ensemble de nettoyage.", + "4719327c9c": "Toutes les suggestions de nettoyage sont ignorées.", + "a19040cd67": "Aucun espace de travail inactif ne correspond aux dépôts sélectionnés.", + "97c772c4fe": "Aucun espace de travail inactif trouvé dans les dépôts cochés.", + "d3eef9463d": "Aucun espace de travail inactif à supprimer.", + "aaee139eab": "Restaurer les suggestions ignorées", + "06cf78521e": "Tout sélectionner dans {{value0}}", + "73690b0031": "Tout désélectionner dans {{value0}}", + "b771c92598": "Supprimer la sélection", + "37ab28277e": "non suggéré", + "1b18868569": "à vérifier", + "b299f201b9": "suppression sans risque", + "2b31bf68de": "inactif", + "ac5ba84cc1": "sélectionnés", + "8b74d4ea6e": "Analyse des worktrees et de l'état Git, puis combinaison des signaux d'onglets ouverts, de terminaux, d'agents actifs et de disponibilité distante avant de suggérer les suppressions.", + "7eee951968": "Analyse des espaces de travail", + "191f0bc98e": "Fermer", + "7ae2ad30f4": "Actualiser", + "e0b5a4deaa": "Vérifiez les espaces de travail inactifs avant de supprimer leurs fichiers locaux et leur état Orca.", + "b2c1331844": "Supprimer les espaces de travail inactifs", + "41d594d01e": "Impossible de supprimer {{value0}} espace de travail{{value1}}", + "0f00612b6d": "Suppression de {{value0}} espace de travail{{value1}} effectuée", + "7f451a3e2c": "Impossible d'ignorer la suggestion de nettoyage", + "662b8ec3f8": "Échec du scan de nettoyage des espaces de travail", + "bc43c37faf": "masqué", + "0c6672f5e3": "Suggestions de nettoyage ignorées", + "fc49f79434": "mixte", + "2ddbd6fe8a": "coché", + "ee81adfcef": "Affichage", + "4d0b72481c": "Ignorer", + "9cc26c019d": "Supprimer", + "0e2d235c63": "Analyse des espaces de travail prête", + "4a35c08764": "À examiner", + "47123d0108": "Collecte des informations sur les espaces de travail. Vous pouvez fermer cette fenêtre et revenir plus tard.", + "9a3be9f2df": "Analyse des espaces de travail en cours. Les nouvelles lignes apparaissent ici au fur et à mesure. Vous pouvez fermer cette fenêtre et revenir plus tard.", + "3d957ff117": "Aucun espace de travail ne correspond à ces filtres.", + "e94b1f8bb4": "Réinitialiser les filtres", + "efb3843e75": "Filtrer et trier les espaces de travail", + "93b7381d50": "Filtres", + "a615e24679": "Trier", + "4cc5b73efe": "Recherche des espaces de travail...", + "5bf2e88480": "{{value0}}/{{value1}} {{value2}} analysés", + "searchPlaceholder": "Rechercher des espaces de travail", + "ageFilter": "Âge", + "reviewFilter": "À examiner", + "gitFilter": "Git", + "contextFilter": "Contexte", + "sortBy": "Trier par", + "sortDirection": "Direction", + "1d3503357d": "Vous pouvez fermer cette fenêtre et revenir pendant que la suppression continue.", + "4c2990886e": "{{value0}}/{{value1}} supprimés", + "86ba852118": "{{value0}}, {{value1}} en échec", + "7b7bde5181": "Espaces de travail vérifiés jusqu'à présent : {{value0}}", + "deletingCount": "Suppression des espaces de travail : {{value0}}", + "deleteCount": "Supprimer les espaces de travail : {{value0}} ?", + "selectedForDeletionCount": "Sélectionnés pour suppression : {{value0}}", + "deleteButtonCount": "Supprimer {{value0}}", + "archivedStatus": "Archivé", + "readyStatus": "Prêt", + "f637f63882": "Le Resource Manager compte {{value0}} ; cette liste en a trouvé {{value1}}. Ce compteur repose uniquement sur le journal d'activité d'Orca, tandis que cette analyse vérifie aussi l'historique Git de chaque espace de travail et ignore les remotes déconnectés.", + "74f6c16279": "Retour" + }, + "backgroundRemoval": { + "removed": "Espaces de travail supprimés : {{value0}}", + "failed": "Espaces de travail non supprimés : {{value0}}", + "error": "Échec du nettoyage des espaces de travail", + "skippedAncestor": "Ignoré car un espace de travail imbriqué n'a pas pu être supprimé.", + "timedOut": "La suppression de {{value0}} prend plus de temps que prévu. Elle continuera en arrière-plan.", + "stillRemoving": "Espaces de travail encore en cours de suppression : {{value0}}", + "skippedPendingAncestor": "Ignoré car un espace de travail imbriqué n'a pas terminé sa suppression." + }, + "candidateRow": { + "gitLabel": "Git", + "commitsLabel": "Commits", + "contextLabel": "Contexte", + "flagsLabel": "Indicateurs", + "collapseDetails": "Réduire les détails", + "expandDetails": "Développer les détails", + "cleanGit": "Git propre", + "dirtyGit": "Git modifié", + "unpushedCommits": "Commits non poussés", + "gitUnknown": "Git inconnu", + "gitStatusUnknown": "État Git inconnu", + "noUnpushedCommits": "Aucun commit non poussé", + "unpushedCommitsCount": "Commits non poussés : {{value0}}", + "uncommittedChanges": "Modifications non validées", + "terminalTabsCount": "Onglets de terminal : {{value0}}", + "editorTabsCount": "Onglets d'éditeur : {{value0}}", + "browserTabsCount": "Onglets de navigateur : {{value0}}", + "diffNotesCount": "Notes de diff : {{value0}}", + "completedAgentsCount": "Agents terminés : {{value0}}", + "contextCount": "Contexte : {{value0}}", + "mainWorkspaceBlocker": "Espace de travail principal", + "folderProjectBlocker": "Projet de dossier", + "pinnedBlocker": "Épinglés", + "activeWorkspaceBlocker": "Espace de travail actif", + "runningTerminalBlocker": "Processus de terminal en cours", + "terminalLivenessUnknownBlocker": "Activité du terminal inconnue", + "dirtyEditorBufferBlocker": "Tampon d'éditeur non enregistré", + "volatileLocalContextBlocker": "Contexte local volatil", + "recentVisibleContextBlocker": "Onglets visités récemment", + "liveAgentBlocker": "Agent actif", + "sshDisconnectedBlocker": "Distant indisponible", + "gitStatusErrorBlocker": "État Git indisponible", + "dirtyFilesBlocker": "Fichiers modifiés", + "unknownBaseBlocker": "Impossible de vérifier les commits non poussés", + "dismissedBlocker": "Ignoré" + }, + "workspace": { + "cleanup": { + "candidate": { + "row": { + "b5d2b33e47": "Suppression…", + "e1135728e3": "En attente de suppression" + } + } + } + } + } + }, + "ui": { + "color": { + "picker": { + "ebcf6ba29e": "Couleur hexadécimale invalide.", + "faa855a582": "Hex", + "1cec618bcc": "Sélecteur {{value0}}" + } + }, + "dialog": { + "f26c4baeda": "Fermer" + }, + "repo": { + "multi": { + "combobox": { + "286ce70256": "SSH", + "4471d4a1c0": "Aucun projet ne correspond à votre recherche.", + "bfd8ce21c6": "Tous les projets", + "a58a0cd100": "Rechercher des projets...", + "65a3dae41d": "Aucun projet" + } + } + }, + "sheet": { + "1189e9fe0a": "Fermer" + } + }, + "terminal": { + "quick": { + "commands": { + "TerminalQuickCommandActionToggle": { + "b0d58e37ed": "Prompt d'agent", + "b5ea4d64f6": "Commande de terminal", + "terminal_short": "Terminal", + "agent_short": "Agent" + }, + "TerminalQuickCommandAppendEnterSwitch": { + "e4e5fed3b3": "Activer/désactiver « Ajouter Entrée »", + "c936c2d6d2": "Envoie immédiatement au lieu de simplement insérer le texte.", + "5fa607d807": "Ajouter Entrée", + "767e4be3e3": "Ajouter Entrée — exécution immédiate" + }, + "TerminalQuickCommandDialog": { + "925b8e0f6e": "Avancé", + "97e96cc027": "/goal", + "e604bd40d6": "Prend en charge les skills, les chemins de fichiers et les commandes intégrées comme", + "79af0c0841": "npm run dev", + "577a342c7d": "Demander à l'agent d'examiner cet espace de travail", + "026cfb232a": "Ne prend pas en charge les commandes de prompt", + "346d409ab2": "Choisir un agent", + "0adba8fa0c": "Agent", + "ec8f081919": "Action", + "ed04233b3e": "Les éléments enregistrés apparaissent dans le menu de la barre d'onglets pour un lancement en un clic.", + "ca414324ee": "Texte de la commande", + "dc921c17ee": "Prompt", + "5b3f634a55": "Ajouter une commande rapide", + "f9b184fc16": "Modifier la commande rapide", + "6751598542": "modifier", + "command_label": "Commande", + "agent_toolbar_hint": "Prend en charge /goal, les skills et les chemins", + "agent_footer_hint": "Les prompts multi-lignes conviennent — restez concis.", + "resize_hint": "Glissez le coin pour redimensionner" + }, + "TerminalQuickCommandDialogFooter": { + "2e2b958dfc": "Enregistrer", + "8dff838dea": "Enregistrer ({{value0}})", + "28370f16b9": "Annuler" + }, + "TerminalQuickCommandLabelField": { + "66ea254301": "Démarrer le serveur de dev", + "db17f1e41e": "Label" + }, + "TerminalQuickCommandScopeField": { + "2db6edede7": "L'enregistrement conserve la portée de projet actuelle, sauf si vous en choisissez une autre.", + "2496523a6f": "Choisir le projet", + "2264edd5d3": "Projet absent de la liste", + "3834d24243": "Projet", + "b83efc79e2": "Global", + "c25cf350ef": "Portée", + "f0631e4999": "dépôt" + } + } + }, + "pane": { + "CloseTerminalDialog": { + "ebd2fa844d": "Fermer", + "1d1a7a9c1f": "Annuler", + "6b9a6975f8": "Le terminal exécute toujours un processus. Si vous fermez le terminal, le processus sera interrompu.", + "78b79d854d": "Fermer le terminal ?", + "stop_agent_title": "Arrêter cet agent ?", + "stop_command_title": "Arrêter la commande en cours ?", + "stop_agent_description": "Fermer ce terminal interrompra le travail en cours de l'agent.", + "stop_command_description": "Fermer ce terminal interrompra la commande qui s'y exécute.", + "dont_ask_again": "Ne plus demander pour les terminaux en cours d'exécution", + "stop_agent_confirm": "Arrêter l'agent", + "stop_command_confirm": "Arrêter et fermer" + }, + "MobileDriverOverlay": { + "c6460cf584": "Reprendre", + "c8f2e1a4b9": "Reprendre ce terminal", + "c44659e09f": "Pilotage par téléphone", + "7cffad954c": "Réduire", + "3eed73394f": "Votre clavier est en pause", + "faa367dc74": "Votre téléphone a laissé ce terminal en taille téléphone", + "54f7d6f69d": "Reprendre tous les terminaux", + "b3d8e1f42a": "Restaurer ce terminal", + "e8c4f2a91b": "Restaurer tous les terminaux", + "f2a8b9c1d3": "Depuis votre téléphone", + "c7e4a2b8f1": "Votre téléphone a le contrôle", + "d9f3c6e2a4": "Le clavier de l'ordinateur est en pause. Reprenez ce terminal pour taper ici, reprenez tous les terminaux contrôlés par votre téléphone, ou réduisez pour continuer à observer.", + "a6b1d8f3e2": "Votre session téléphone est terminée. Restaurez ce terminal à la taille bureau, ou tous les terminaux que votre téléphone a laissés en taille téléphone." + }, + "TerminalAgentSessionForkDialog": { + "17fc841e59": "Copier le contexte", + "0c8a8629b1": "Le fork apparaît comme son propre espace de travail, et non comme un enfant imbriqué. Le nouvel agent reçoit une transcription limitée sous forme de brouillon modifiable.", + "620461df22": "Fork de premier niveau", + "619b5a35d2": "Créer un fork de espace de travail au premier niveau et démarrer un nouvel onglet d'agent avec le contexte capturé.", + "64e292e8e3": "Forker la session d'agent", + "9d25de2920": "Créer un fork", + "2b10412cfc": "Création..." + }, + "TerminalContextMenu": { + "b4cdd9314e": "Effacer l'écran", + "8c17d6786d": "Fermer le volet", + "copyTerminalId": "Copier l'ID du terminal", + "2cf85a6a55": "Copier l'ID du volet", + "39809d152f": "Définir le titre…", + "clearPaneTitle": "Effacer le titre du volet", + "06c2b0f043": "Égaliser les tailles des volets", + "98bccf4fa2": "Scinder le terminal vers le bas", + "20e565d865": "Scinder le terminal vers la droite", + "8a7ddb8b8a": "Forker la session d'agent…", + "0a82b0608c": "Ajouter une commande rapide…", + "9528a65ef8": "Aucune commande rapide", + "3ce594a4a0": "Global", + "ec85df5914": "Commandes rapides", + "0a917b591a": "Coller", + "f3eeb1de13": "Copier", + "selectAll": "Tout sélectionner", + "c2f0b72b8d": "Insérer", + "925f49f210": "Agrandir le volet", + "df766809e0": "Réduire le volet", + "cff67afad1": "Copier le contexte", + "15dd899676": "Ajouter à {{value0}}…" + }, + "TerminalErrorToast": { + "e4aa243f8c": "Redémarrer le daemon", + "a7e2fd2699": "signaler un problème", + "5c8ce20be6": "Si le problème persiste, veuillez", + "cc6d997c65": "Redémarrez le daemon de terminal depuis ici pour effacer un état de daemon obsolète.", + "7ee11bc0db": "Orca n'a pas pu confirmer si la session précédente de ce terminal tourne encore ; il a donc laissé la session intacte. Rouvrez ce volet pour réessayer.", + "e16012e31e": "Le daemon de terminal propriétaire de cette session s'est arrêté ; la session et son historique de défilement n'ont pas pu être récupérés. Ouvrez un nouveau terminal pour continuer.", + "sessionUnavailable": "Orca n'a pas pu se rattacher à la session de terminal de ce volet sur l'hôte. Ouvrez un nouveau terminal pour continuer." + }, + "TerminalProcessExitOverlay": { + "capacityTitle": "Limite de consoles Git Bash atteinte", + "capacityDetail": "Git Bash a atteint sa limite de 128 consoles. Fermez les terminaux Git Bash inutilisés, puis redémarrez ce terminal.", + "failedTitle": "Terminal terminé", + "failedDetail": "Le processus shell s'est terminé avec le code de sortie {{code}}. Sa sortie est conservée.", + "close": "Fermer", + "restart": "Redémarrer" + }, + "TerminalPane": { + "ac112e9036": "Retirer le titre", + "f984ab2a30": "Retirer le titre du volet : {{value0}}", + "cc5a2dc706": "Modifier le titre du volet : {{value0}}", + "7dbbfcbecc": "Titre du volet" + }, + "TerminalLinkActionPopover": { + "openLink": "Ouvrir le lien", + "openFile": "Ouvrir le fichier", + "switchWorkspace": "Changer de espace de travail", + "openInFinder": "Afficher dans le Finder", + "openFolder": "Ouvrir le dossier", + "openWithDefaultApp": "Ouvrir avec l'application par défaut", + "switchTerminal": "Changer de terminal", + "openTaskTerminal": "Ouvrir le terminal de tâches", + "systemBrowser": "Navigateur système", + "orcaBrowser": "Navigateur Orca", + "terminalLinkSettings": "Paramètres des liens du terminal", + "copyLink": "Copier le lien", + "copied": "Copied", + "copiedLink": "Lien copié", + "copyLinkFailed": "Échec de la copie du lien" + }, + "TerminalSessionStateSaveFailureDialog": { + "6bee0c8f17": "Ouvrir l'Analyseur d'espace disque", + "ae20d0ffc2": "Ignorer", + "38c282a2c4": "L'analyseur s'ouvre directement depuis ici. Vous pouvez aussi l'ouvrir plus tard depuis le menu d'outils en bas à gauche, en choisissant Analyseur d'espace.", + "e2fcf07c0d": "Orca n'a pas pu enregistrer cette session de terminal car le stockage local est plein ou inaccessible en écriture. Ouvrez l'analyseur d'espace disque pour trouver du stockage de espace de travail à nettoyer.", + "678c780a2c": "Espace disque indisponible" + }, + "osc52": { + "clipboard": { + "blocked": { + "toast": { + "97c98f1afe": "Ouvrir le paramètre", + "7cf51f74fd": "Activez l'écriture du presse-papiers par les TUI dans les paramètres du terminal pour copier depuis SSH, Zellij, tmux, Neovim, fzf ou Grok.", + "89eaa3e80b": "Écriture du presse-papiers du terminal bloquée" + } + }, + "default": { + "on": { + "notice": { + "title": "L'écriture du presse-papiers par les TUI est désormais activée par défaut", + "description": "Zellij, tmux, Neovim et les autres programmes de terminal peuvent désormais copier vers votre presse-papiers. Désactivez l'option dans les paramètres du terminal.", + "action": "Ouvrir le paramètre" + } + } + }, + "failed": { + "toast": { + "62a0af2cb4": "La copie du presse-papiers du terminal n'a pas pu être confirmée", + "fdd3e7e977": "L'application de terminal a demandé une copie, mais Orca n'a pas pu confirmer qu'elle a bien atteint le presse-papiers système." + } + } + } + }, + "stale": { + "agent": { + "row": { + "ad991ece5c": "Le volet de l'agent n'est plus disponible.", + "090d607412": "stale-agent-row-{{value0}}" + } + } + }, + "terminal": { + "agent": { + "session": { + "fork": { + "2317900211": "Échec de la copie du contexte du fork.", + "88e34d00eb": "Fork de session de premier niveau ouvert dans un nouveau espace de travail", + "fd3d12a1e1": "Échec de la création de l'espace de travail fork.", + "38e41edc6e": "Cet espace de travail ne peut pas être forké en worktree Git.", + "f867385bb5": "Impossible de trouver l'espace de travail source de ce fork.", + "046e8d853c": "Aucun contexte de terminal à forker", + "c00421d320": "Contexte du fork copié. Lancez un agent et collez-le pour démarrer le fork.", + "f62b40e2c7": "Aucun contexte de terminal à copier", + "373a3103e7": "Contexte copié", + "3fc568a49d": "Échec de la copie du contexte." + } + } + }, + "drop": { + "handler": { + "1e072f611e": "Échec de l'envoi de {{value0}} {{value1}}.", + "53f015fd85": "Ignoré : {{value0}} symlink{{value1}}.", + "29c031b49a": "Envoi de {{value0}} fichier{{value1}} vers le runtime…", + "0c77693641": "Worktree pas encore prêt — réessayez dans un instant.", + "ce8248b835": "Chemin du worktree indisponible.", + "internalTooManyPaths": "Le contenu déposé contient trop de chemins pour un collage sûr dans le terminal.", + "internalPathsTooLarge": "La liste de chemins déposée est trop volumineuse pour un collage sûr dans le terminal.", + "b4cf68e889": "Ignoré : {{value0}} {{value1}}.", + "writeTimeout": "Glisser-déposer de fichiers annulé : le terminal n'a pas accepté le chemin avant le délai de sécurité.", + "writeRejected": "Glisser-déposer de fichiers annulé : le terminal n'a pas pu accepter le chemin." + } + } + }, + "use": { + "terminal": { + "pane": { + "context": { + "menu": { + "a29b9faa01": "ID du volet copié", + "pane": { + "id": { + "copy": { + "failed": "Impossible de copier l'ID du volet" + } + } + }, + "terminal": { + "id": { + "copied": "ID du terminal copié", + "copy": { + "failed": "Impossible de copier l'ID du terminal" + } + } + } + } + } + } + } + }, + "PinnedTabCloseDialog": { + "6c190f295a": "Fermer l'onglet épinglé ?", + "0d1963f4a6": "Cet onglet est épinglé. Voulez-vous vraiment le fermer ?", + "dont_ask_again": "Ne plus demander pour les onglets épinglés", + "0b38ee2f86": "Annuler", + "c337c9d75c": "Fermer" + }, + "TerminalSshReconnectOverlay": { + "authFailed": "Échec de l'authentification pour {{value0}}. Reconnectez-vous pour poursuivre cette session de terminal.", + "reconnectFailed": "La connexion SSH à {{value0}} a échoué. Reconnectez-vous pour poursuivre cette session de terminal.", + "connecting": "Connexion à {{value0}} en cours. Ce terminal reprendra dès que l'hôte sera disponible.", + "connected": "SSH est connecté.", + "disconnected": "Ce terminal attend {{value0}}. Connectez-vous pour poursuivre cette session SSH.", + "connectFailed": "Échec de la connexion SSH", + "title": "Connexion SSH requise", + "connectingButton": "Connexion...", + "connectButton": "Se connecter", + "removedTitle": "Hôte SSH supprimé", + "removedBody": "L'hôte SSH de cet espace de travail a été supprimé ; il ne peut plus se connecter. Supprimez l'espace de travail pour l'effacer — les fichiers distants restent intacts.", + "removeWorkspaceButton": "Supprimer l'espace de travail" + }, + "TerminalRemoteRuntimeReconnectBanner": { + "retryingTitle": "Reconnexion au runtime distant", + "disconnectedTitle": "Runtime distant déconnecté", + "retryingBody": "Orca réessaiera pendant une minute au maximum. Ce terminal reprendra si la connexion revient.", + "disconnectedBody": "Les tentatives automatiques sont interrompues. Reconnectez-vous pour reprendre cette session de terminal.", + "reconnectButton": "Reconnecter" + }, + "TerminalQuickCommandsSubmenu": { + "3ccc7981bb": "Hôte indisponible", + "54f29b7c0d": "Chargement de l'hôte…" + } + } + }, + "tab": { + "group": { + "TabGroupPanel": { + "814fb04c43": "Chargement de l'éditeur...", + "f7d6ce445e": "Fermer le groupe", + "0db2081805": "Scinder vers le haut", + "30137df7d0": "Scinder vers la gauche", + "4df2a06d36": "Scinder vers le bas", + "ab1e2bff04": "Scinder vers la droite", + "9acaf92093": "Actions du volet", + "1bce81dba6": "simulateur", + "1ff1c77616": "navigateur", + "586d2ac445": "terminal", + "addSplitPane": "Ajouter un volet scindé", + "closePaneColumn": "Fermer le volet scindé" + }, + "AiVaultSessionDropLayer": { + "dropOntoTerminalPane": "Déposez sur un volet de terminal pour reprendre cette session.", + "couldNotReadPayload": "Impossible de lire les données de glisser-déposer de la session.", + "localWorkspacesOnly": "La reprise depuis l'historique n'est disponible que dans les espaces de travail locaux.", + "openLocalWorkspace": "Ouvrez un espace de travail local avant de reprendre une session.", + "sessionQueued": "Session en attente", + "openSupportedWorkspace": "Ouvrez un espace de travail avant de reprendre une session.", + "sessionHostMismatchUnsupported": "Cette session appartient à un autre hôte. Déposez-la sur un espace de travail du même hôte.", + "localSessionSshWorkspaceUnsupported": "L'historique de cette session est stocké sur cette machine ; la reprise est impossible dans un espace de travail SSH. Déposez-la plutôt sur un espace de travail local." + }, + "TabGroupDropOverlay": { + "paneColumnLabel": "Nouvelle scission" + } + }, + "bar": { + "BrowserTab": { + "6e0bc8f3a8": "Ouvrir dans le navigateur", + "9dd880bd56": "Fermer les onglets à droite", + "1611a1324b": "Fermer", + "5d6e89891f": "Dupliquer l'onglet", + "966feb9ad5": "Scinder vers la droite", + "7e8106899f": "Scinder vers la gauche", + "2186a8407c": "Scinder vers le bas", + "96354ed249": "Scinder vers le haut", + "911542656f": "Épingler l'onglet", + "c5aaee8c39": "Détacher l'onglet" + }, + "EditorFileTab": { + "3da7445c84": "Renommer le fichier {{value0}}" + }, + "EditorFileTabContextMenu": { + "52ce4f4605": "Copier le chemin relatif", + "5b85754786": "Copier le chemin", + "bfd5797ef4": "Ouvrir l'aperçu Markdown", + "e5ff31ccaf": "Fermer les onglets à droite", + "ba1369dd24": "Fermer tous les onglets de l'éditeur", + "1ba8492c5b": "Fermer", + "68cc610e7f": "Renommer", + "f7c3d7d5af": "Scinder vers la droite", + "e3ff145b98": "Scinder vers la gauche", + "1d04b1630b": "Scinder vers le bas", + "6b3efb106e": "Scinder vers le haut", + "fdd29eb669": "Épingler l'onglet", + "8e9d603a09": "Détacher l'onglet" + }, + "QuickLaunchButton": { + "348a04c1ad": "Paramètres de l'agent…", + "ec2adf093e": "Lancer {{value0}} dans un nouveau terminal", + "465e432ef1": "Impossible de construire la commande de lancement pour {{value0}}.", + "e518f544b1": "Aucun agent détecté", + "8dea9b5cdf": "Aucun agent activé" + }, + "RecentTabSwitcher": { + "329638ff6f": "Changer d'onglet", + "07ad4cd0b7": "Basculer entre les onglets" + }, + "SortableTab": { + "ab19f603eb": "Renommer l'onglet {{value0}}", + "6df69d9388": "Fermer l'onglet {{value0}}", + "fdb2691425": "Réduire le volet", + "95db5f2f7d": "Fermer l'onglet" + }, + "SortableTabContextMenu": { + "35e8892fd0": "Couleur de l'onglet", + "2f697b3c31": "Modifier le titre", + "c1ee099c7e": "Fermer les onglets à droite", + "8d16f9cd30": "Fermer les autres", + "89359a36f7": "Fermer", + "21132389e9": "Scinder vers la droite", + "0ce4bae39d": "Scinder vers la gauche", + "af80ed83c1": "Scinder vers le bas", + "591f9b12c1": "Scinder vers le haut", + "7703990447": "Gris", + "845576bed1": "Sarcelle", + "be905e9b0a": "Vert", + "69682e2ce4": "Jaune", + "a47629b3cf": "Orange", + "620aec6729": "Rouge", + "03cf6dab1a": "Rose", + "c2d8b0991f": "Violet", + "cb3eadefd2": "Bleu", + "20baa43c05": "Aucun", + "60f958ec75": "Épingler l'onglet", + "417722e9c2": "Détacher l'onglet", + "splitTerminalRight": "Scinder le terminal vers la droite", + "splitTerminalDown": "Scinder le terminal vers le bas" + }, + "TabBar": { + "b1a132357f": "Nouvel onglet", + "4f327c8b3d": "Ouvrir un Markdown...", + "3d5d6c960d": "Nouveau Markdown", + "fd2b42aaa3": "Nouvel émulateur mobile", + "aea43b5748": "Ouvrir l'onglet émulateur existant.", + "b426bb2615": "Aller à l'émulateur mobile", + "4833fb2cbe": "Nouvel onglet de navigateur", + "d364f3c8d4": "Nouveau terminal", + "7c1313d237": "Nouveau terminal :", + "d1afac112b": "WSL", + "efb33546ff": "Git Bash", + "1a8af49530": "Invite CMD", + "2148f65e04": "PowerShell", + "ab589350e5": "Impossible de construire la commande de lancement pour {{value0}}.", + "7a9b4af2af": "Faire défiler les onglets vers la gauche", + "232e075b07": "Faire défiler les onglets vers la droite" + }, + "TabBarCreateEntry": { + "d62d63b807": "Créer un fichier", + "25dc1cd653": "Ouvrir le fichier", + "7cdf8ee0c8": "Ouvrir une URL", + "b27864279e": "Lancer un agent", + "0e5b7a3f16": "Rechercher parmi les onglets ouverts, fichiers, URL et agents…", + "8f0a1c4d92": "Basculer vers l'onglet", + "2c38630a01": "L'espace de travail n'existe plus", + "4f0d9a71c2": "L'onglet n'existe plus", + "d7d496a451": "La page du navigateur n'existe plus", + "7726ce9970": "L'onglet émulateur mobile n'existe plus", + "chooseAction": "Choisissez une action.", + "searchProvider": "Rechercher {{value0}}" + }, + "TabBarQuickCommandsButton": { + "a2c7a33831": "Commande", + "20bbd75896": "Aucune commande", + "b82e237a4b": "Plus de commandes rapides", + "85482c57bc": "Exécuter la commande rapide", + "b775303755": "Exécuter la commande rapide : {{value0}}", + "196593b6a9": "Supprimer {{value0}}", + "15529ede69": "Modifier {{value0}}", + "1d411fb6a5": "Enregistrer une commande rapide pour ce dépôt", + "8f1e971966": "Ajouter une commande rapide", + "3220e2da27": "Cette commande rapide sera retirée de votre liste enregistrée.", + "e8e1a52edb": "Supprimer « {{value0}} » ?", + "37e1bb90ce": "Exécuter : {{value0}}", + "77ac113df0": "Démarrer {{value0}} : {{value1}}", + "7b1c9d6ae1": "Exécution", + "c781f992e4": "destructive", + "be8f0ff166": "Supprimer", + "f3a8c2d1e7": "Rechercher des commandes rapides...", + "b4e7f9a2c1": "Aucune commande ne correspond", + "8d525e5f15": "Copied", + "53b17a4b1b": "Impossible de copier", + "a9a564b7e7": "Copier {{value0}}", + "69a1441a21": "Rien à copier", + "192a4616a5": "Actions de la commande rapide" + }, + "shell": { + "icons": { + "d4ceaa227c": "Git", + "e9b2e70613": "WSL" + } + }, + "tab": { + "create": { + "entry": { + "classifier": { + "42e6262ae9": "Aucune action disponible.", + "097a982ee0": "Chargement des fichiers...", + "c41f8d20b7": "Rechercher parmi les onglets ouverts, fichiers, URL et agents…", + "90eb94dc48": "Saisissez une URL http:// ou https://.", + "5553b283ce": "Saisissez une URL ou un chemin de fichier.", + "queryTooLarge": "Le texte recherché est trop long.", + "absolutePathRemoteBlocked": "Les chemins absolus nécessitent un espace de travail local." + } + }, + "menu": { + "options": { + "5501c2fb7a": "terminal", + "9630dd5494": "shell", + "a094576900": "nouveau terminal", + "4f23f4d01d": "nouveau shell", + "4f2a91e15b": "navigateur", + "6d0e6a4b7a": "nouveau navigateur", + "c87ad57785": "onglet de navigateur", + "cce7ef1d2c": "web", + "5f17fb9d0c": "markdown", + "44caaf7b36": "md", + "fb50e3d874": "nouveau markdown", + "6d8b6b4117": "nouveau fichier", + "b330f72434": "mark", + "37ff3ddca1": "ouvrir un markdown", + "164c394bab": "ouvrir un fichier", + "bbaf4f85a4": "émulateur mobile", + "3784b83bd4": "émulateur", + "a63847a742": "simulateur", + "1baeb07c17": "simulateur iOS", + "8a580f88cf": "iPhone", + "7ecdc5ef08": "iPad", + "14965cc123": "mobile" + } + } + } + }, + "TabWorkspaceLayoutMenuSection": { + "right": "Droite", + "left": "Gauche", + "down": "Bas", + "up": "Haut", + "moveToPaneColumn": "Déplacer l'onglet vers la scission" + }, + "EditorFileTabCloseButton": { + "4655cf570e": "Fermer l'onglet", + "a768f428f1": "Fermer l'onglet" + }, + "TerminalTabSplitMenuSection": { + "splitTerminal": "Scinder le terminal" + }, + "TerminalTabLeadingIcon": { + "7ab2964bea": "Achèvement d'agent non lu" + }, + "TabBarQuickCommandAddActions": { + "45a2f36d51": "Commande", + "b856c833ae": "Commande sur {{value0}}" + }, + "TabBarQuickCommandHostLoadStatus": { + "82e294f3ca": "Hôte indisponible", + "7c129b08ff": "Chargement de l'hôte…" + } + } + }, + "status": { + "bar": { + "PetStatusSegment": { + "3668339495": "Supprimer {{value0}}", + "cd8c6c654c": "Paramètres du compagnon…", + "ed176ad68f": "Importer un bundle .codex-pet…", + "59b5955621": "Téléverser le vôtre…", + "0608ad02a2": "Choisir un compagnon", + "b75484a01a": "Taille du compagnon", + "c6aa805b1b": "px", + "2f7bbaa457": "Taille", + "34c25dfe9c": "Compagnon", + "aec479308a": "Menu du compagnon", + "cef0ab4636": "Échec de l'importation du bundle de compagnon", + "2021d4f6db": "L'importation d'un bundle de compagnon nécessite un redémarrage complet de l'app (pas seulement un rechargement).", + "f395c9a685": "Échec de l'importation du fichier", + "e6234bcc17": "L'ajout d'un compagnon personnalisé nécessite un redémarrage complet de l'app (pas seulement un rechargement).", + "6d0a8cd179": "Afficher le compagnon", + "1fbc51cc77": "Masquer le compagnon" + }, + "PortsStatusSegment": { + "4ebf90c12e": "Aucun port externe détecté", + "7dac3ecc9d": "Ports externes", + "95495019ed": "Scan des ports indisponible sur {{value0}} : {{value1}}", + "a8e4bdb412": " · {{value0}} externe(s)", + "9aa11005bf": "workspace ·", + "c22ea609fd": "Ports", + "a11ed266ce": "espace de travail", + "ca41be2802": "Ports — espace de travail {{value0}} {{value1}}{{value2}}", + "b8bc3e420a": "Ports, espace de travail {{value0}} {{value1}}", + "3a87d54dfb": "Aucun port de espace de travail détecté", + "c174bbbfed": "Recherche des ports de espace de travail...", + "8caaa86e9a": "ports", + "45834a9ace": "port", + "4ae65d871a": "externes", + "2b84c4d11f": "{{value0}} espace de travail · {{value1}} externes" + }, + "ResourceUsageStatusSegment": { + "946d9f94d0": "Annuler", + "67c4ecda49": "Force la fermeture de ce terminal. Tout travail non enregistré dans le volet est perdu. Action irréversible.", + "4bb076fa89": "Tuer", + "996295bff2": "terminal orphelin", + "92924a14e3": "Nettoyer les espaces de travail", + "27a74f91f0": "Rien en cours d'exécution pour le moment", + "1b24a32d3a": "Mémoire", + "298f4be7f2": "CPU", + "2aa2de6cb9": "Nom", + "30ff2c3c31": "{{value0}} orphelin", + "6449a95c78": "Part de la RAM physique de cette machine occupée par les processus suivis par Orca.", + "e7ccce7e87": "de la RAM système", + "9e2525c89f": "Mémoire résidente utilisée par Orca ainsi que par les processus des terminaux de chaque worktree.", + "1fedf94eae": "Charge CPU cumulée. Une valeur supérieure à 100 % signifie que plusieurs cœurs travaillent simultanément.", + "e7cf14ec78": "Sessions de terminal indisponibles. La liste est peut-être obsolète.", + "93b0de3c21": "Redémarrer", + "f85af9cda6": "Instantanés de ressources et sessions de terminal indisponibles.", + "f8e0d794b4": "Le daemon ne répond pas", + "bd19fd7a59": "Tuer toutes les sessions", + "c9382662bb": "Redémarrer le daemon", + "59f178fe11": "{{value0}}, daemon injoignable", + "21cacb16d1": "· distant", + "73a3fd68a9": "Réduire le dépôt", + "b12e31dfcb": "Développer le dépôt", + "d659d71d2d": "Reprendre l'espace de travail {{value0}}", + "bbcd9b7b85": "Réduire l'espace de travail", + "c4a8968bdd": "Développer l'espace de travail", + "b10695d6ce": "Tuer la session", + "288a4dd177": "Orca", + "53dd5560ae": "Réduire Orca", + "e419d27083": "Développer Orca", + "41ae4fa725": "Arrêt forcé…", + "138b99bd80": "cette session", + "888dad8c55": "Chargement…", + "6d9793d4bc": "Gestionnaire de ressources - Terminaux", + "ca95d077db": "Daemon injoignable", + "a82253b458": "Supprimer l'espace de travail.", + "946724a70a": "L'espace de travail principal ne peut pas être supprimé.", + "16bc3c998a": "Supprimer l'espace de travail {{value0}}", + "0f9e50eb07": "Autre", + "d406915b78": "Renderer", + "81cd37af99": "Main", + "fa6d36758d": "Tuer la session {{value0}}", + "b8f4a2c1d0e3": "{{value0}} orphelins", + "c7e3b1a0d9f2": "Tuer {{value0}} terminal orphelin", + "d8f4c2b1e0a3": "Tuer {{value0}} terminaux orphelins", + "e9a5d3c2b1f0": "Tuer {{value0}} ?" + }, + "SshStatusSegment": { + "3ad70e0365": "Gérer les hôtes distants…", + "6e8a9a4242": "Hôtes distants", + "d09ec41831": "Hôtes distants", + "fdc57e9970": "État de connexion de l'hôte distant", + "59b553e2aa": "Déconnecter", + "63f36455cc": "Se connecter", + "bf07aee59e": "Échec de la déconnexion", + "2c29e2de68": "Échec de la connexion", + "bc5a3fd41a": "partiel", + "3d0128b105": "connecté", + "fd9a3c600e": "error", + "fbb3f9f05e": "conflit", + "95e4ff5b4b": "push en cours", + "63a2b965f6": "pull en cours", + "remote_server": "Serveur distant", + "runtime_checking": "Vérification", + "runtime_online": "Connecté", + "runtime_unavailable": "Déconnecté", + "runtime_connect_unavailable": "Hôte distant injoignable", + "runtime_disconnect_failed": "Échec de la déconnexion", + "runtime_reconnecting": "Reconnexion", + "runtime_last_close_reason": "Fermé : {{value0}}", + "runtime_reconnect_attempt": "Tentative {{value0}}", + "runtime_workspace_window_closed": "Fenêtre de l'espace de travail fermée", + "connectedHostCount_one": "{{count}} hôte", + "connectedHostCount_other": "{{count}} hôtes", + "connecting": "Connexion…", + "workspaceConflict": "Conflit de espace de travail", + "workspaceSyncError": "Erreur de synchronisation de l'espace de travail" + }, + "StatusBar": { + "9659e38343": "Ports", + "d1e1a7a6bf": "Gestionnaire de ressources", + "24ac89df1a": "Hôtes distants", + "5e59007df4": "Utilisation Kimi", + "8c86cd77b0": "Utilisation OpenCode Go", + "c1df0d67ec": "Utilisation Gemini", + "c0909c686e": "Utilisation Codex", + "3885eb74d8": "Utilisation Claude", + "c8857b40f7": "Actualiser les données d'utilisation", + "75ded02687": "Gérer les comptes…", + "ff0fbe9311": "Actif", + "7657e3db9c": "Compte Codex", + "38b5647724": "Runtime d'utilisation Codex", + "ba55303942": "Ouvrir les détails Codex et le sélecteur de compte", + "4dff061aab": "·", + "2483c60695": "···", + "c35af53b73": "Se connecter", + "f19a63e7cd": "Se connecter pour voir l'utilisation", + "5c938d39ac": "sem.", + "54e8d6bb2d": "Fable", + "d79c3362c4": "5 h", + "a79c64f87e": "Fable", + "8295903d17": "Redémarrez les terminaux Claude actifs avant de reprendre d'anciennes conversations après un changement de compte.", + "c98ea88392": "Aucun autre compte", + "9332ba8684": "Basculer vers", + "d450654fa2": "Compte Claude", + "11e2354daf": "Runtime d'utilisation Claude", + "3dd7ddfae1": "Ouvrir les détails Claude et le sélecteur de compte", + "59c6e7b4e0": "Les sessions visibles redémarrent maintenant. Les autres redémarrent quand leur worktree devient actif.", + "c676918adc": "Valeur par défaut du système", + "3325d996cb": "Actualiser les limites de débit", + "fda8146810": "Ouvrir les détails d'utilisation Kimi", + "629251f4b6": "Ouvrir les détails d'utilisation OpenCode Go", + "d2375976eb": "Ouvrir les détails d'utilisation Gemini", + "3d79122c3f": "kimi", + "d7a0668acc": "opencode-go", + "68efc0345c": "gemini", + "76b06d4da5": "claude", + "a28a5dd9b1": "error", + "cd9d7b40ff": "Redémarrer {{value0}} sessions", + "6cd6650b4c": "Redémarrer la session", + "1446d0d8a0": "{{value0}} sessions Codex sont toujours sur l'ancien compte.", + "605901a495": "1 session Codex est encore sur l'ancien compte", + "5e5f9f5160": "1 réinitialisation de limite de débit disponible", + "5ecae9197c": "{{value0}} réinitialisations de limite de débit disponibles", + "25d8bbde69": "Utilisation de la réinitialisation…", + "e159fc1fd7": "Réinitialiser maintenant", + "972a1ff497": "Réinitialiser les limites Codex ?", + "6d1042aa6f": "Cela consomme un crédit de réinitialisation de limite de débit Codex pour le compte actif et réinitialise immédiatement toute fenêtre d'utilisation éligible.", + "f077f586db": "Ne plus demander", + "c0e972d726": "Annuler", + "06741a2f3d": "Ouvrir les détails d'utilisation MiniMax", + "3bbf140864": "Utilisation MiniMax", + "remoteServerLabel": "Serveur distant", + "antigravityUsage": "Utilisation Antigravity", + "antigravityUsageDetails": "Ouvrir les détails d'utilisation Antigravity", + "grokUsageAria": "Ouvrir les détails d'utilisation Grok", + "grokUsageMenu": "Utilisation Grok", + "floatingTerminalNewActivity": "{{label}}, nouvelle activité", + "codexSignInSuccess": "Connecté à Codex", + "codexSignInError": "Échec de la connexion à Codex. Veuillez réessayer." + }, + "StatusBarUsageEmptyCta": { + "828c764a79": "Connecter un compte", + "caa0f39811": "Prend en charge :", + "97957ad3a3": "Connectez vos comptes de fournisseurs d'IA pour voir leur utilisation en temps réel et basculer facilement entre les comptes.", + "9a542f46c7": "Masquer de la barre d'état", + "84c3b15dca": "Limites d'utilisation des agents", + "d663430cf9": "Configurer le suivi d'utilisation" + }, + "SkillUpdateStatusSegment": { + "runningLabel": "Mise à jour des skills", + "runningOne": "Mise à jour de {{value0}}…", + "runningMany": "Mise à jour des skills de {{value0}}…", + "runningAria": "Mise à jour des skills en cours. Cliquez pour ouvrir les détails.", + "successLabel": "Skills mis à jour", + "successOne": "{{value0}} mis à jour", + "successMany": "Skills de {{value0}} mis à jour", + "successAria": "Skills mis à jour. Cliquez pour ouvrir les détails.", + "errorLabel": "Échec de la mise à jour", + "errorTooltip": "Échec de la mise à jour des skills — cliquez pour voir les détails", + "errorAria": "Échec de la mise à jour des skills. Cliquez pour ouvrir les détails.", + "stoppingLabel": "Arrêt de la mise à jour", + "stoppingTooltip": "Arrêt de la mise à jour des skills…", + "stoppingAria": "Arrêt de la mise à jour des skills. Cliquez pour ouvrir les détails." + }, + "UpdateStatusSegment": { + "5cd13105a3": "Échec de la mise à jour. Cliquez pour déplier.", + "2201df6987": "Échec de la mise à jour — cliquez pour voir les détails", + "8533c12c3c": "Échec de la mise à jour", + "962404f68e": "Mise à jour prête à installer. Cliquez pour déplier.", + "248ee5d8ef": "Téléchargement d'Orca v{{value0}}… {{value1}} %", + "57a29c3b0e": "Mise à jour prête", + "fd1d3b3a1d": "Téléchargement de la mise à jour, {{value0}} pour cent. Cliquez pour déplier.", + "9d13213a56": "Orca v{{value0}} prête à installer" + }, + "WorkspaceSpaceCompactPanel": { + "a471aa9c24": "Mis à jour", + "9be86c46a0": "Libérable", + "f4d2651498": "Analysé", + "6a5dc3c61a": "À examiner", + "c361440dc0": "Beta", + "8ff597593d": "Espace", + "0582df6d2e": "Analyser", + "f5e1a84d79": "Actualiser", + "2af2174d6d": "Annuler", + "5691353a21": "Arrêt en cours", + "2837dc7c72": "annulation", + "0583c806ac": "L'espace disque des espaces de travail n'a pas été analysé.", + "39786e3b73": "Analyse de la taille des espaces de travail.", + "bef4dc0457": "{{value0}} récupérables · {{value1}} indisponibles", + "3d8d47ce77": "{{value0}} · dernier résultat conservé" + }, + "WorkspaceSpaceManagerPanel": { + "2965415393": "Échec de la suppression forcée", + "e031e93219": "Aucun espace de travail correspondant.", + "a02d84d2d2": "Analyse des espaces de travail. Vous pouvez quitter cette page.", + "be37293b10": "État", + "33aef3e9cc": "Taille", + "81f14d9924": "Dépôt", + "e4aebea158": "Espace de travail", + "1d0f8300d1": "Sélectionner les espaces de travail supprimables visibles", + "697d60c456": "Effacer la sélection visible", + "81aaf1de65": "Afficher uniquement les espaces de travail supprimables", + "d7ac56452e": "Activité", + "243287ac60": "Nom", + "6f8f6a6b04": "Filtrer les espaces de travail", + "5caccea440": "Supprimer la sélection", + "e4a12c455b": "Effacer", + "0cb1501ccf": "récupérable", + "65402b7192": "sélectionnés", + "43171f3e60": "Espaces de travail", + "83f1a0a932": "Récupérable", + "09960d86bd": "Analysé", + "1cc6cd4c0f": "espaces de travail", + "02b27c2230": "espace de travail", + "63efebe0e6": "{{value0}} {{value1}} supprimés de Space.", + "eee5240810": "Espaces de travail supprimés", + "9afc97f9a3": "Espace de travail supprimé", + "792a214457": "Supprimer l'espace de travail", + "a998501630": "Forcer", + "9155381019": "Sparse", + "f39d291997": "Sélectionner", + "16988df079": "Aucun fichier trouvé.", + "b25c2c1086": "éléments de premier niveau", + "d3f9c69ddc": "Zoom", + "ef890d31b9": "Tous", + "c28643d3da": "Aller à l'espace de travail", + "66870929fb": "Issue", + "fb2069acb7": "À examiner", + "b9b4a3a25d": "Branche", + "c432278ec7": "Buffers de l'éditeur", + "0bc756efaf": "Modifications Git", + "e9528a89b3": "Terminaux", + "a8d9e0de79": "Agents", + "d384a4ce9f": "Décision de suppression", + "7d7745bb8f": "Supprimable", + "720870a18e": "Conserver : lié", + "cbc343a7a8": "Conserver : en cours d'utilisation", + "2055bc6a5a": "Conserver : modifications non enregistrées", + "ec7b076a75": "Conserver : état Git non vérifié", + "7ab8d7e2d7": "Conserver : fichiers modifiés", + "7f7895514e": "Conserver : actif", + "2b501ee391": "Conserver : principal", + "39801484e0": "Échec", + "33653dbac2": "Suppression en cours", + "52b629eb84": "Mis à jour", + "e91dd2a9ae": "Lancez une analyse pour inspecter la taille des espaces de travail.", + "61e25239da": "Aucune ligne de espace de travail n'a été fournie par l'analyse.", + "8194a4fb29": "L'analyse a échoué avant la collecte de la moindre taille de espace de travail.", + "b2f82ed5ae": "Supprimable", + "20a4204dce": "Les derniers résultats valides restent visibles.", + "8c7c57fbf8": "Analyser", + "508673bac0": "Actualiser", + "8dc9ddac8a": "Annuler", + "1fce91d1b9": "Arrêt en cours", + "d254f04097": "annulation", + "265d956765": "{{value0}}. Vous pouvez quitter cette page.", + "d595295d7d": "{{value0}} peuvent être récupérés depuis les worktrees liés.", + "34174bd83d": "{{value0}}. Vous pouvez quitter cette page ; le dernier résultat reste visible.", + "433bb7f595": "ok", + "0ba046fbc5": "Échec de l'analyse.", + "5c6d25720c": "Sélectionnez un espace de travail à inspecter.", + "c5135e7e4a": "Analyse de la taille des espaces de travail. Vous pouvez quitter cette page.", + "0990a63160": "Aucune taille de espace de travail analysée pour le moment.", + "977bdf9a36": "Aucun élément de premier niveau à afficher.", + "131662ac65": "{{value0}} ouvert(s)", + "0d1c78d749": "Sélectionner {{value0}}" + }, + "ports": { + "status": { + "popover": { + "rows": { + "a49ea79246": "Aller au worktree", + "f2b813345f": "Espace de travail indisponible", + "0e72c8d9fb": "Arrêter le processus", + "536d48a5dc": "Copier {{value0}}", + "085f4f0334": "Ouvrir dans le navigateur", + "e4a709548c": "Échec de l'actualisation des ports", + "acdb6df590": "Processus arrêté sur {{value0}}", + "480d8f2347": "{{value0}} copié", + "b854ec9ff5": "Échec de l'ouverture du navigateur" + } + } + } + }, + "tooltip": { + "cedb7b99e3": "% utilisé", + "6d6df77f41": "Aucune donnée disponible", + "7f7f208060": "Mensuel", + "252c096536": "Hebdomadaire", + "a79c64f87e": "Fable", + "94038ad2fa": "Session", + "2c35eca8d4": "Impossible de récupérer l'utilisation", + "1292d4f2ee": "Indisponible", + "7567cd1c6b": "Utilisation indisponible", + "a9a318b7a3": "Échec de l'actualisation — affichage des données en cache", + "7ad719c4bf": "Limité", + "e740f92596": "Échec de l'actualisation", + "8418ec448d": "L'utilisation de {{value0}} n'a pas pu être actualisée. Des sessions d'agent sont peut-être encore connectées.", + "45198c7d95": "1 réinitialisation de limite de débit disponible", + "bce421cba3": "{{value0}} réinitialisations de limite de débit disponibles", + "7ec6e030a0": "La prochaine expire maintenant", + "d1e442a9e5": "Expire maintenant", + "6cf9eaed10": "La prochaine expire dans {{value0}}", + "20ad66aed1": "Expire dans {{value0}}", + "0d8d7cfe15": "En attente d'une session Claude", + "1804cd8c3f": "Actualisation de la connexion", + "f8f0f9d8cc": "Problème de réseau", + "bf2e739f18": "Connexion indisponible", + "f8b8dbed85": "Utilisation indisponible", + "3d3c9c0c1f": "L'utilisation de Claude s'actualisera une fois que le terminal Claude actif aura renouvelé ses identifiants.", + "42fdd4da1d": "La connexion Claude est en cours d'actualisation. Des sessions d'agent sont peut-être encore connectées.", + "c06c1d215d": "L'utilisation de Claude n'a pas pu être actualisée car la requête réseau a échoué.", + "cabdc2a9e0": "Les identifiants de connexion Claude n'ont pas pu être lus.", + "a7517cccb6": "L'utilisation de Claude est indisponible pour le moment.", + "e2c6a4f917": "Exécutez Grok pour actualiser", + "d1b7f509ac": "Exécutez grok dans un terminal sur l'ordinateur qui fait tourner Orca et attendez son démarrage. Si demandé, terminez la connexion, puis réessayez l'utilisation. Inutile d'envoyer un message.", + "f90b3d7a16": "Exécutez Kimi pour actualiser", + "a37e8c15d4": "Exécutez kimi dans un terminal sur l'ordinateur qui fait tourner Orca et attendez son démarrage, puis réessayez l'utilisation." + }, + "SshTargetStatusRow": { + "sshHost": "Hôte SSH" + }, + "usagePercentageLabel": { + "used": "{{value0}} % utilisé", + "remaining": "{{value0}} % restants" + }, + "UsagePercentageDisplayChangeNotice": { + "title": "L'utilisation affiche désormais le pourcentage consommé", + "body": "Vous préférez le restant ? Changez ce choix dans les paramètres.", + "dismiss": "Ignorer", + "openSettings": "Ouvrir les paramètres", + "gotIt": "Compris" + }, + "UsageRosterPanel": { + "title": "Utilisation", + "openDetails": "Ouvrir les détails d'utilisation", + "notSignedIn": "non connecté", + "allAgents": "tous les agents", + "usageDetails": "Détails et historique d'utilisation", + "loadingUsage": "Chargement de l'utilisation…", + "usageUnavailable": "Utilisation indisponible", + "noUsageData": "Aucune donnée d'utilisation", + "detailed": "Détaillé", + "compact": "Compact", + "detailedTooltip": "Utilisation complète avec barres, libellés et pourcentages", + "compactTooltip": "Utilisation condensée : uniquement la fenêtre la plus contrainte", + "footerDetailAria": "Détail du pied de page d'utilisation" + }, + "RemoteServerUpdateStatusSegment": { + "updating": "Mise à jour de {{value0}}/{{value1}}", + "updatingTooltip": "Des mises à jour de serveurs Orca distants sont en cours", + "failed": "Échec de la mise à jour de {{value0}} serveurs", + "failedTooltip": "Ouvrez les mises à jour des serveurs Orca distants pour vérifier et relancer", + "updated": "{{value0}} serveurs mis à jour", + "updatedTooltip": "Mises à jour des serveurs Orca distants terminées", + "failedOne": "Échec de la mise à jour d'un serveur", + "updatedOne": "1 serveur mis à jour" + }, + "resource": { + "memory": { + "metric": { + "workingSetDescription": "Somme des working sets (WS). Les pages partagées peuvent apparaître dans plusieurs processus.", + "rssDescription": "Somme des tailles résidentes (RSS). Les pages partagées ou aliasées peuvent apparaître dans plusieurs processus." + } + }, + "manager": { + "terminal": { + "copy": { + "terminalSessionCount_one": "{{count}} session de terminal", + "terminalSessionCount_other": "{{count}} sessions de terminal", + "memoryUnavailable": "mémoire indisponible", + "tooltipSummary": "Gestionnaire de ressources - {{memory}} - {{sessions}}", + "spaceScanReady": "Analyse Space prête", + "sessionsGroupedByWorkspace": "Les sessions de terminal sont regroupées par espace de travail.", + "noTerminalSessions": "Aucune session de terminal pour le moment.", + "ariaLabel": "Gestionnaire de ressources, {{sessions}}", + "ariaLabelWithSpaceScan": "Gestionnaire de ressources, {{sessions}}, {{spaceScan}}" + } + } + } + }, + "CaffeinateStatusSegment": { + "active": "Actif", + "inactive": "Inactif", + "ariaLabel": "{{title}}, {{status}}", + "onDescription": "Maintenir cet ordinateur éveillé en permanence", + "autoDescription": "Rester éveillé pendant qu'un agent travaille", + "offDescription": "Autoriser le comportement de veille normal du système" + } + } + }, + "stats": { + "ClaudeUsageDailyChart": { + "2a6360c7cb": "Écriture en cache", + "61c58f8976": "Lecture du cache", + "7d2efeff5e": "Sortie", + "d7fb787e6b": "Entrée", + "a7902d3c1d": "tokens", + "059945f71d": "Totaux quotidiens d'entrées, de sorties, de lectures et d'écritures en cache.", + "c9f7cd30e9": "Utilisation quotidienne" + }, + "ClaudeUsagePane": { + "21ea00bfa8": "Cache", + "a8b7487ff7": "Sortie", + "faf3444859": "Entrée", + "0f03975d59": "Tours", + "1afc25eb06": "Modèle", + "c17bed0416": "Projet", + "01476891c7": "Dernière activité", + "abfc4a4943": "Taux de réutilisation du cache :", + "7e76c84153": "Sessions récentes", + "32176e1d44": "tours", + "02a046792e": "sessions •", + "f97435845c": "Projet principal :", + "7dc9e5613b": "Par projet", + "c3fdbc5474": "Modèle principal :", + "0f394c24e3": "Par modèle", + "51ae85fa00": "Le taux de réutilisation du cache est calculé ainsi : tokens de lecture du cache / (tokens d'entrée + tokens de lecture du cache).", + "b26d4ddb58": "Coût estimé équivalent API", + "0f3e696ca9": "Sessions / Tours", + "8cc23be4a3": "Tours sans lecture du cache", + "1634c4f404": "Taux de réutilisation du cache", + "b786fb4a70": "Écriture en cache", + "268cf0af51": "Lecture du cache", + "2b8a2f14aa": "Tokens de sortie", + "ea71fae8fc": "Tokens d'entrée", + "7dde9331fd": "Aucune utilisation locale de Claude trouvée pour l'instant dans ce périmètre.", + "424cd50412": "Activer les statistiques d'utilisation Claude", + "8d18bbb771": "Actualiser", + "c5b9b344d0": "Actualiser l'utilisation Claude", + "505be9aac4": "Période", + "f61cffb9c8": "Portée", + "dd29209b21": "Filtres", + "e9bf9fce0e": "Options d'utilisation Claude", + "6afacbee37": "Suivi de l'utilisation Claude", + "0cb1a36d7d": "Lit les journaux locaux d'utilisation de Claude pour afficher les statistiques de tokens, de modèles et de sessions.", + "5ce4842c2c": "Toute l'utilisation locale de Claude", + "4f8368c272": "Worktrees Orca uniquement", + "cfe2282ffa": "Inconnu", + "7765a4c3e1": "n/d", + "2d41fd45c6": " • Dernière erreur d'analyse : {{value0}}", + "rangeLast7Days": "7 derniers jours", + "rangeLast30Days": "30 derniers jours", + "rangeLast90Days": "90 derniers jours", + "rangeAllTime": "Depuis toujours" + }, + "CodexUsageDailyChart": { + "1e6f62d7e3": "Raisonnement", + "c646e1783c": "Entrées en cache", + "7b596a88b2": "Sortie", + "99a91d3143": "Entrée", + "e4bdcf0071": "tokens", + "c756cda6a8": "Totaux quotidiens d'entrées, d'entrées en cache, de sorties et de raisonnement.", + "609aa96e8b": "Utilisation quotidienne" + }, + "CodexUsagePane": { + "e0b988599d": "Total", + "bbd20344b8": "Sortie", + "3acc582214": "Entrée", + "bd0822ca47": "Événements", + "c2478bcc3c": "Modèle", + "1a65900aea": "Projet", + "0c36b100be": "Dernière activité", + "0bd8655475": "Sessions locales Codex les plus récentes dans ce périmètre.", + "0cb0983c07": "Sessions récentes", + "79a69522a5": "événements", + "bf1bf2f674": "sessions •", + "829ee743f2": "Projet principal :", + "b98718aaab": "Par projet", + "95d2d89285": "Modèle principal :", + "5a0d1d69cd": "Par modèle", + "94ac1f1ee7": "Les tokens de raisonnement sont affichés à titre indicatif, mais le coût est calculé uniquement à partir des entrées hors cache, des entrées en cache et des sorties.", + "1a18fbd56b": "Coût estimé équivalent API", + "907b31865f": "Sessions / Événements", + "6e18146e9b": "Sortie de raisonnement", + "a9ac0f423a": "Entrées en cache", + "5d8eba87bd": "Tokens de sortie", + "e365eaa6fd": "Tokens d'entrée", + "4c865393b4": "Aucune utilisation locale de Codex trouvée pour l'instant dans ce périmètre.", + "f7c1affbd5": "Activer les statistiques d'utilisation Codex", + "3022cda443": "Actualiser", + "ec4d270e2c": "Actualiser l'utilisation Codex", + "89162e019b": "Période", + "6d68e8399a": "Portée", + "1af1a39b2f": "Filtres", + "70b5b8581f": "Options d'utilisation Codex", + "408210470c": "Suivi de l'utilisation Codex", + "13badcd8f2": "Lit les journaux locaux d'utilisation de Codex pour afficher les statistiques de tokens, de modèles et de sessions.", + "4fe8820098": "Toute l'utilisation locale de Codex", + "201766b754": "Worktrees Orca uniquement", + "bf6cf2d4dd": "Inconnu", + "ae255c3dba": "n/d", + "247c93ca92": "• tarification déduite", + "8a6655f7a2": " • Dernière erreur d'analyse : {{value0}}", + "rangeLast7Days": "7 derniers jours", + "rangeLast30Days": "30 derniers jours", + "rangeLast90Days": "90 derniers jours", + "rangeAllTime": "Depuis toujours" + }, + "OpenCodeUsagePane": { + "349f7c3f5c": "Total", + "dfc4513657": "Sortie", + "0f2f266c9d": "Entrée", + "d416f5cf92": "Événements", + "08c78441b7": "Modèle", + "a4738de041": "Projet", + "d97bdf6e27": "Dernière activité", + "81817a641a": "Sessions locales OpenCode les plus récentes dans ce périmètre.", + "4799177b1c": "Sessions récentes", + "1e5d410df0": "événements", + "bc0cb89901": "sessions •", + "048ffe4d65": "Projet principal :", + "0f0a1684bb": "Par projet", + "a15206a63a": "Modèle principal :", + "040c044d39": "Par modèle", + "e5bb23d85e": "Le coût provient de la base de données locale d'OpenCode lorsque le message de l'assistant en enregistre un.", + "15c34d4b08": "Coût enregistré", + "7e9433469a": "Sessions / Événements", + "5a65d68b77": "Sortie de raisonnement", + "603504ee3b": "Entrées en cache", + "7aa4d8ce35": "Tokens de sortie", + "d637a892ed": "Tokens d'entrée", + "bb6363e08c": "Aucune utilisation locale d'OpenCode trouvée pour l'instant dans ce périmètre.", + "f04131b3be": "Activer les statistiques d'utilisation OpenCode", + "603cd138dc": "Actualiser", + "bed558df0b": "Actualiser l'utilisation OpenCode", + "b5ed5c9fd0": "Période", + "40d283c837": "Portée", + "01583b30aa": "Filtres", + "230d6de108": "Options d'utilisation OpenCode", + "bea80ceae0": "Suivi de l'utilisation OpenCode", + "b8b3522436": "Lit les journaux locaux d'utilisation d'OpenCode pour afficher les statistiques de tokens, de modèles et de sessions.", + "144a6050e9": "Toute l'utilisation locale d'OpenCode", + "e04c58327c": "Worktrees Orca uniquement", + "362231082f": "Inconnu", + "8095a63426": "n/d", + "6cc7782458": " • Dernière erreur d'analyse : {{value0}}", + "rangeLast7Days": "7 derniers jours", + "rangeLast30Days": "30 derniers jours", + "rangeLast90Days": "90 derniers jours", + "rangeAllTime": "Depuis toujours" + }, + "ShareUsageButton": { + "7d6b25323d": "Partager sur X", + "b295c1c75d": "Copier l'image", + "bd82c76a70": "Copied", + "bce08eccb9": "Partager l'utilisation", + "cecefa7c32": "Partager" + }, + "ShareUsageCard": { + "4a4c6c79a3": "sessions ·", + "66c83284cf": "Tokens quotidiens", + "b760c0b622": "Modèle principal", + "2d9eb39264": "Total de tokens", + "beb6f24f37": "Coût estimé", + "da62578d9d": "Utilisation", + "0eb31e79ee": "IDE Orca", + "960324e9b8": "événements", + "6adac63cfe": "tours" + }, + "StatsPane": { + "42d3e0bdf7": "Fournisseur de statistiques d'utilisation : {{value0}}", + "c79f073d4c": "Statistiques d'utilisation", + "a58aba506f": "PRs créées", + "1c96f433e2": "Temps de travail des agents", + "9dbec9e675": "Agents lancés", + "73ed07859c": "Lancez votre premier agent pour commencer le suivi", + "1e696db2f6": "OpenCode", + "7d26110cea": "Codex", + "85457c02fe": "Claude", + "b2cf4310ce": "Vue d'ensemble", + "908c470587": "codex", + "eb6a066185": "claude", + "eee19cfade": "vue d'ensemble", + "grokUsageTab": "Grok", + "trackingSince": "Suivi depuis {{value0}}" + }, + "UsageOverviewPane": { + "22ed1b7669": "sessions", + "444585cb41": "avec données", + "ecb0cd8a4c": "activés -", + "33f7b043d2": "Fournisseurs", + "60002bb22f": "Aucune utilisation locale de Claude, Codex ou OpenCode trouvée pour l'instant. La vue d'ensemble se remplira après la prochaine session d'agent qui écrira des journaux de tokens.", + "70f36452d4": "Part du cache", + "327603fe8b": "Jours actifs", + "0eaf937335": "Coût estimé", + "3887b94ce5": "Total de tokens", + "2d13e57f72": "Activer OpenCode", + "2f1ee2878b": "Activer Codex", + "0ea0cae435": "Activer Claude", + "6c00c46815": "Activez un fournisseur pour analyser les journaux locaux des agents et construire le registre combiné de tokens.", + "49405ccc8d": "Commencer le suivi des tokens", + "ca6bc5fded": "Actualiser", + "e06d1baf5c": "Actualiser la vue d'ensemble de l'utilisation", + "c760c481c5": "Vue d'ensemble de l'utilisation", + "55c910f4f1": "- certains prix de modèles sont indisponibles" + }, + "share": { + "card": { + "utils": { + "19f4b4dc75": "github.com/stablyai/orca", + "d864fc5f98": "sortie", + "5d66fdd7c2": "entrée", + "7080aeaebb": "Raisonnement", + "4ee864629a": "Entrées en cache", + "33d38e2177": "Sortie", + "c2d7b23d57": "Entrée", + "9d166247ee": "Écriture en cache", + "cc28cb965e": "Lecture du cache" + } + } + }, + "stats": { + "search": { + "cb6a9f0334": "cache", + "eaf251e183": "tokens", + "6953af58e6": "opencode", + "b77826fca3": "codex", + "e9dc37d889": "claude", + "8efeae0b22": "suivi", + "5acbe1fdf2": "temps", + "ef8bbf7739": "prs", + "ce8533f02e": "agents", + "0bba8ca244": "statistiques", + "0e2a0b6431": "utilisation", + "372debfac0": "stats", + "26bb901fcd": "Statistiques Orca, plus analyses de tokens Claude, Codex, OpenCode et utilisation d'abonnement Grok.", + "cb2430ae6a": "Statistiques & usage", + "f8a1b2c3d4": "grok", + "e7f0a1b2c3": "abonnement", + "d6e9f0a1b2": "crédits", + "a3b6c7d8e9": "utilisation grok", + "9f2a3b4c5d": "xai" + } + }, + "usage": { + "overview": { + "model": { + "bc474051e5": "OpenCode", + "eb220d193b": "Codex", + "544d6d4c16": "Claude" + }, + "sections": { + "9564a3b21b": "sessions -", + "32330a6e66": "{{value0}} : {{value1}} tokens", + "57d1448ef8": "Activer", + "f6df0d7d6d": "Plus", + "1dd166c920": "Moins", + "52d9221dc0": "Heatmap d'activité récente des tokens", + "c424eb3f8e": "Meilleur :", + "f28ff1f852": "Activité combinée récente des tokens Claude, Codex et OpenCode.", + "69e2b50427": "Intensité quotidienne", + "3a795542fa": "Répartition combinée des tokens", + "e65084cb4b": "raisonnement", + "3bc4a01b24": "Tokens d'entrée, de sortie et de cache cumulés sur les fournisseurs activés.", + "4ff104da47": "Répartition des tokens", + "0015facc1f": "Cache", + "7f270458af": "Sortie", + "9365b14a4e": "Nouvelle entrée", + "3de9bf87fc": "Aucun modèle pour l'instant", + "6762f6a682": "tokens", + "a7f937fb29": "{{value0}} sessions - {{value1}} {{value2}}", + "c8f3a2d1e0b4": "tours", + "d9a4b3e2f1c5": "événements", + "statusScanning": "Analyse en cours", + "statusEnabled": "Activé", + "statusOff": "Désactivé" + } + } + }, + "UsageBreakdownSection": { + "7765a4c3e1": "n/d", + "247c93ca92": "• tarification déduite", + "32176e1d44": "tours", + "79a69522a5": "événements", + "02a046792e": "sessions •" + }, + "UsageSessionsTable": { + "1afc25eb06": "Tours", + "0f03975d59": "Événements", + "21ea00bfa8": "Cache", + "e0b988599d": "Total", + "01476891c7": "Dernière activité", + "c17bed0416": "Projet", + "f6a2c8d019": "Modèle", + "faf3444859": "Entrée", + "a8b7487ff7": "Sortie", + "cfe2282ffa": "Inconnu" + }, + "GrokUsagePane": { + "g8h9i0j1k2": "Utilisation Grok", + "b2d3e4f5c6": "Crédits hebdomadaires de l'abonnement via OAuth du Grok CLI (~/.grok/auth.json). Même source que la barre d'état.", + "c3e4f5a6b7": "Configurer dans Comptes", + "h9i0j1k2l3": " • {{value0}}", + "i0j1k2l3m4": "Actualiser l'utilisation Grok", + "d4f5a6b7c8": "Actualiser", + "e5a6b7c8d9": "Crédits hebdomadaires utilisés", + "f6b7c8d9e0": "Réinitialisation de la période de facturation", + "a7b8c9d0e1": "Paramètres du compte Grok" + } + }, + "sparse": { + "SparseCheckoutPresetSelect": { + "c4ac80151d": "Nouveau préréglage", + "7c3275d307": "Modifier {{value0}}", + "c7f9b3f0c1": "Désactivé", + "8b12c0850a": "Enregistrer", + "de8fce5854": "Annuler", + "ddbcaef7be": "src/renderer packages/ui", + "0e9ad9c798": "Répertoires", + "064c1e2d12": "UI du renderer", + "b3a500c623": "Nom", + "16223dde6a": "Charger les préréglages", + "a683a4bc8e": "Réessayer le chargement des préréglages", + "14952d451e": "{{value0}} répertoires", + "e9283eb171": "1 répertoire", + "69c020eddc": "Modifier le préréglage", + "bd6cec2056": "nouveau" + } + }, + "source": { + "control": { + "SourceControlActionVariableChips": { + "1b77798d5f": "Variables", + "6b921a0ac2": "Exemple", + "4bf6d88039": "(vide)", + "7377483644": "Cet espace de travail" + } + } + }, + "skills": { + "SkillsPage": { + "cb142070b4": "Actualiser", + "984405683f": "Plugin", + "4d177feabd": "Intégré", + "aa59462502": "Dépôt", + "571c5818c1": "Home", + "0bc1379f4c": "Toutes les sources", + "38e0951c3a": "Skills d'agent", + "fb6bf60b52": "Claude", + "426be2aac6": "Codex", + "39b6998ddb": "Tous les fournisseurs", + "a68dee6a32": "Rechercher des skills", + "e46e162e2e": "de", + "b088e0785d": "Beta", + "f43ad6edf3": "Skills", + "7e828fb2c6": "Retour", + "ea72d6185b": "Impossible d'analyser les skills", + "dc4c3328ee": "Afficher le fichier", + "9963dff6d3": "Aucune description trouvée.", + "995fde8337": "Impossible d'afficher le fichier du skill", + "ab5b777350": "Dossiers de skills home, dépôt, intégrés et plugins passés en revue.", + "08a321a984": "Ajustez la recherche ou les filtres.", + "4acd6d68ec": "Aucun skill trouvé", + "6a62a0168c": "Aucun résultat", + "cd7893fbc1": "Analyse des skills", + "35b9a724a0": "Disponibles", + "0c74e7ff34": "Installés", + "c13b82793c": "Gérer les installations", + "aee7b99cc6": "Installer depuis un lien", + "filterProvider": "Filtrer par agent", + "filterSource": "Filtrer par source", + "allSources": "Tous", + "clearFilters": "Réinitialiser les filtres", + "closeSkills": "Fermer les skills", + "closeTooltip": "Fermer · Esc", + "moreActions": "Plus d'actions", + "sharedLinks": "Liens partagés", + "emptyCopy": "Les dossiers de skills analysés sont vides. Installez un bundle partagé, ou actualisez après avoir ajouté un skill.", + "retry": "Réessayer", + "remoteShareNotice": "Ces skills se trouvent sur {{host}}. Ouvrez Skills sur cette machine pour les partager.", + "viewSwitch": "Afficher", + "searchLinks": "Rechercher des liens", + "deleteSkills": "Supprimer des skills…" + }, + "SkillFreshnessNudge": { + "titleOne": "Un skill Orca installé n'est pas à jour", + "titleMany": "{{value0}} skills Orca installés ne sont pas à jour", + "description": "Mettez à jour {{value0}} pour que les agents suivent les instructions actuelles de cette version d'Orca.", + "updateOne": "Mettre à jour le skill", + "updateMany": "Mettre à jour les skills" + }, + "SkillFreshnessRow": { + "statusUpdateAvailable": "Mise à jour disponible", + "statusCantUpdate": "Ignoré", + "cantUpdateReason": "Orca a exclu ce skill de la commande de mise à jour.", + "skippedReasonUnrecognized": "La copie ici ne correspond pas à la version officielle — elle a peut-être été modifiée, ou il s'agit d'un autre skill portant le même nom. Orca l'a exclue de la mise à jour pour ne pas l'écraser. Supprimez-la si vous voulez qu'Orca mette à jour ce skill.", + "skippedReasonReadOnly": "Cette copie se trouve dans un emplacement en lecture seule, donc Orca l'a exclue de la mise à jour. Modifiez ses permissions pour permettre à Orca de la mettre à jour.", + "skippedReasonInaccessible": "Orca n'a pas pu lire cette copie, il a donc exclu le skill de la mise à jour.", + "skippedReasonInRepo": "C'est un skill de projet, pas un skill global — Orca ne met à jour que vos skills globaux, il a donc exclu celui-ci de la mise à jour.", + "skippedReasonPluginCache": "Un plugin gère ce skill, donc Orca l'a exclu de la mise à jour — mettez plutôt à jour le plugin.", + "skippedReasonExternalLink": "Cette copie est un raccourci pointant hors des dossiers de skills d'Orca, donc Orca l'a exclue de la mise à jour.", + "skippedReasonBrokenLink": "Cette copie est un raccourci vers quelque chose qui n'existe plus, donc Orca l'a exclue — vous pouvez la supprimer sans risque.", + "chipCurrent": "Actuel", + "chipUnrecognized": "Non reconnu", + "chipInaccessible": "Inaccessible", + "chipDuplicate": "Doublon", + "chipExternalLink": "Lien externe", + "chipBrokenLink": "Lien cassé", + "chipReadOnly": "Lecture seule", + "chipInRepo": "Dans un dépôt", + "chipPluginCache": "Cache de plugins", + "tipCurrent": "Cette copie correspond à la version officielle actuelle.", + "tipUnrecognized": "Cette copie ne correspond à aucune version officielle — elle a peut-être été modifiée, ou il s'agit d'un autre skill portant le même nom.", + "tipInaccessible": "Orca n'a pas pu lire cette copie (erreur de permissions ou de fichier).", + "tipDuplicate": "Une copie séparée de ce skill, installée à part de la principale.", + "tipExternalLink": "Un raccourci pointant hors des dossiers de skills d'Orca.", + "tipBrokenLink": "Un raccourci vers quelque chose qui n'existe plus.", + "tipReadOnly": "Cette copie se trouve dans un emplacement en lecture seule.", + "tipInRepo": "Cette copie vit dans un projet, pas dans vos skills globaux.", + "tipPluginCache": "Cette copie est gérée par un plugin.", + "skippedReasonDuplicate": "Il s'agit d'une copie séparée : la mise à jour ne l'atteindra pas — la commande ne rafraîchit que la copie principale. Supprimez cette copie puis réinstallez le skill pour que cet emplacement suive le principal.", + "skippedReasonNewer": "Cette copie est plus récente que celle fournie avec cette build d'Orca ; Orca l'a donc laissée telle quelle plutôt que de la rétrograder. La mise à jour d'Orca remettra les deux en phase.", + "skippedReasonStaleRecord": "Le programme de mise à jour des skills n'a aucun enregistrement exploitable pour cette copie ; il signale donc le skill comme déjà à jour sans rien modifier. Réinstallez-le pour remettre l'enregistrement en cohérence : {{value0}}", + "chipNewer": "Plus récent", + "tipNewer": "Cette copie est plus récente que celle fournie avec cette build d'Orca." + }, + "SkillFreshnessUpdateDialog": { + "title": "Mettre à jour les skills", + "checking": "Vérification des skills Orca installés…", + "none": "Aucun skill Orca installé trouvé.", + "updateOne": "1 mise à jour disponible", + "updateMany": "{{value0}} mises à jour disponibles", + "blockedOne": "1 skill ne peut pas être mis à jour automatiquement.", + "blockedMany": "{{value0}} skills ne peuvent pas être mis à jour automatiquement.", + "success": "Tous les skills Orca installés sont à jour.", + "attention": "Certains skills Orca installés ont été exclus de la mise à jour.", + "runningOne": "Mise à jour d'1 skill…", + "runningMany": "Mise à jour des skills de {{value0}}…", + "runningDescription": "Vous pouvez fermer cette fenêtre — le traitement continue en arrière-plan.", + "progressAria": "Mise à jour des skills", + "updatedOne": "1 skill mis à jour", + "updatedMany": "Skills de {{value0}} mis à jour", + "updatedPartial": "{{value0}} skills mis à jour sur {{value1}}", + "errorTitle": "La mise à jour ne s'est pas terminée", + "retry": "Réessayer", + "copyCommand": "Copier la commande", + "copied": "Copied", + "showLog": "Afficher le journal", + "updating": "Mise à jour…", + "updateActionOne": "Mettre à jour 1 skill", + "updateActionMany": "Mettre à jour {{value0}} skills", + "checkNow": "Revérifier", + "done": "Terminé", + "close": "Fermer", + "stop": "Arrêter", + "stopping": "Arrêt…", + "stoppingHeadline": "Arrêt de la mise à jour…", + "scanIncomplete": "Orca n'a pas pu terminer la vérification des skills gérés par plugin.", + "scanDepthLimit": "Orca a atteint sa limite de profondeur d'analyse des plugins avant de vérifier ce dossier.", + "scanEntryLimit": "Orca a atteint sa limite d'entrées d'analyse des plugins avant de vérifier le reste de ce cache.", + "scanCandidateLimit": "Orca a trouvé plus de dossiers de skills homonymes qu'il ne peut en inspecter sans risque.", + "scanManifestLimit": "Orca a ignoré ce manifeste de plugin car il dépassait une limite de sécurité.", + "scanOutsideRoot": "Orca a ignoré ce chemin de plugin car il pointe hors du cache de plugins.", + "scanIoErrorWithCode": "Orca n'a pas pu lire ce chemin de plugin ({{value0}}).", + "scanIoError": "Orca n'a pas pu lire ce chemin de plugin.", + "scanIssueLimit": "Orca a trouvé trop de dossiers de plugins ignorés pour pouvoir les lister un par un." + }, + "SkillUpdateResultRows": { + "stillOutdated": "Toujours pas à jour après l'exécution de la mise à jour." + }, + "SkillUpdateRow": { + "oneLocation": "1 emplacement", + "manyLocations": "{{value0}} emplacements" + }, + "SkillFreshnessStatusPill": { + "updateAvailable": "Mise à jour disponible", + "upToDate": "À jour", + "installed": "Installés", + "details": "Détails", + "needsAttention": "Examiner le skill", + "checking": "Vérification...", + "checkFailed": "Échec de la vérification" + }, + "SkillShareDialog": { + "reconnect": "Reconnectez votre compte Orca avant de partager.", + "unconfigured": "Connectez un compte Orca Cloud avant de partager.", + "prepareFailed": "Impossible de préparer ce skill au partage.", + "peopleRequired": "Sélectionnez au moins un membre de l'équipe.", + "publishCancelled": "Téléversement annulé. La copie préparée reste disponible pour une nouvelle tentative.", + "publishFailed": "Impossible de publier ce skill. La copie préparée reste disponible pour une nouvelle tentative.", + "copied": "Lien de partage copié", + "preparing": "Préparation de l'aperçu…", + "ready": "Lien du skill prêt", + "title": "Partager le skill", + "readyDescription": "Les destinataires s'authentifient auprès d'Orca avant de pouvoir l'inspecter ou l'installer.", + "newVersionDescription": "Vérifiez les fichiers exacts, puis publiez une version immuable dans le package Cloud existant.", + "description": "Vérifiez les fichiers exacts, choisissez qui peut y accéder, puis publiez une version immuable.", + "0f07fa2a79": "Publier le skill", + "7aa4ba0dba": "Publier une nouvelle version", + "3a51d0f34f": "Annuler le téléversement", + "e9d652ae3d": "Annulation…", + "30985d4fc0": "Annuler", + "3af85f6add": "Terminé", + "readyDescriptionV2": "Toute personne disposant de ce lien non répertorié peut inspecter et installer les skills.", + "descriptionV2": "Vérifiez les fichiers exacts, puis publiez une version immuable protégée par un lien non répertorié.", + "publishBundle": "Publier le bundle" + }, + "SkillCard": { + "d25a1b8ae6": "Partager le skill", + "01c5a16e01": "Sélectionner {{value0}}" + }, + "SkillCloudManagementActions": { + "ed753624c0": "Supprimer le package Cloud", + "640bc6b92e": "Confirmer la suppression du package", + "6b6adf68c1": "Supprimer la version Cloud sélectionnée", + "bbec37d8f8": "Confirmer la suppression de la version", + "7a80285dad": "Ne plus partager", + "0ac5c175fd": "Confirmer l'arrêt du partage", + "9e6bd31487": "Aucun lien de partage actif.", + "c7f2b69122": "L'arrêt du partage bloque les installations futures. Les copies déjà installées sur une machine y restent.", + "8cac3b9362": "Liens actifs", + "cc8e4ef6ba": "Enregistrer les accès", + "eb603f7888": "en dehors de la liste actuelle seront conservés.", + "3d346ddce8": "destinataire existant", + "2b8c569ea7": "Organisation actuelle", + "2552c12fea": "Accès", + "505cd6105c": "Contrôles de partage Cloud", + "linkCopied": "Lien de partage copié", + "activeLinkBearerDescription": "Toute personne disposant d'un lien actif peut inspecter et installer les skills. La révocation bloque les accès futurs mais laisse inchangées les copies installées.", + "copyLink": "Copier le lien" + }, + "SkillInstallDialog": { + "39acb9e8f4": "Installer le skill", + "59c3b76cdd": "Réessayer l'installation", + "241e72f9d6": "Installation…", + "05588076a9": "Annuler l'installation", + "d198ec91e5": "Fermer", + "4a00d133c5": "Orca vérifie l'accès et l'identité du package avant de modifier la machine sélectionnée.", + "fcbec627cc": "Installer le skill partagé", + "01c5a14e01": "Installer les skills partagés", + "opening": "Ouverture de ce lien…" + }, + "SkillInstallManagementDialog": { + "8095927ff3": "Fermer", + "e91af0079f": "Supprimer", + "470d8d2476": "Confirmer la suppression", + "c1d03ee50d": "Annuler l'installation", + "561e49ccd1": "Installer la version sélectionnée", + "2ae587d39c": "Réessayer les installations incomplètes", + "f7c5075e77": "Abandonner les modifications et supprimer", + "e1884c812e": "Abandonner les modifications et installer la version", + "070654c6d1": "Orca les conservera, sauf si vous abandonnez explicitement les modifications locales.", + "74b70892ea": "Fichiers locaux modifiés", + "86ed219b55": "Choisir une version", + "9c04bd0120": "Version installée", + "64c71cf7b9": "Aucune installation de skill gérée par Orca n'a été trouvée sur cette machine.", + "0900db719a": "— déconnecté", + "176fef9516": "· SSH", + "6cb1fbe039": "Cet ordinateur", + "3677ae58e7": "Mettez à jour, revenez en arrière ou supprimez sans risque les versions installées par Orca.", + "44d118a8f7": "Gérer les skills installés", + "1c9e7f420a": "Skills de bundle installés", + "34c2ef9e71": "Installer {{count}} skills", + "a0cab67c2f": "Supprimer {{count}} skills", + "f970d8088d": "Confirmer la suppression de {{count}} skills", + "dab29e4b54": "{{installed}} installés · {{updated}} mis à jour · {{keptLocal}} conservés en local", + "installAnotherMachine": "Installer sur une autre machine" + }, + "SkillInstallReviewContent": { + "89e2601162": "Abandonner et remplacer", + "a5675fb371": "de contenu et l'a laissée intacte. Conservez-la, ou abandonnez explicitement et remplacez-la par cette version.", + "37d990b94c": "modifié", + "2a31912f14": "Orca a détecté", + "651b7d8a57": "La copie locale nécessite une décision", + "98ed90e523": "Un skill contient des instructions et peut inclure des scripts. Considérez-le comme du code provenant de son auteur.", + "f72ee4022a": "SHA-256", + "3d8421ca2f": "exécutable", + "87137bcb8d": "scripts", + "fab8fce842": "fichiers", + "8f9833d509": "Version immuable", + "66270286ac": "· Vous pouvez réessayer sans risque.", + "1b6ad2ca5c": "vérifié(s).", + "3fc62a61eb": "emplacement", + "157de228b4": "Inspecter le skill", + "69236de8d6": "Vérification…", + "27672470d9": "Ouvrir le lien n'installe rien. Examinez d'abord la version immuable.", + "66cff7a804": "https://app.orca.dev/skills/share/…", + "93eb0fe8c7": "Lien de skill Orca", + "releaseNotes": "Notes de version :", + "targetHeader": "Cible d'installation" + }, + "SkillInstallTargetFields": { + "8e6a972229": "Aucun espace de travail connu sur cette machine.", + "7a366323e7": "Dossier", + "d628c416a2": "Worktree Git", + "5845cfe543": "Choisir un worktree ou un dossier", + "0e5b43a9e3": "Espace de travail", + "0c10a406fb": "WSL ·", + "e5b0d15e64": "Système d'exploitation hôte", + "cb47652227": "Environnement d'exécution", + "a4dfd33095": "Un seul espace de travail", + "c779621aa0": "Skills globaux", + "63cc9e31fe": "Destination", + "85d85880df": "· SSH", + "71eefd7660": "— déconnecté", + "0785d0a503": "— mise à jour requise", + "8562dd1e6e": "Cet ordinateur", + "b8a0b706ad": "Machine" + }, + "SkillInstallWorkspaceCombobox": { + "search": "Rechercher des espaces de travail...", + "empty": "Aucun espace de travail trouvé." + }, + "SkillShareReviewContent": { + "e3caf6baeb": "La révocation de ce lien bloque les accès futurs. Elle ne retire pas les copies déjà installées sur les machines des destinataires.", + "6d6233a3a4": "Copier le lien", + "0142581727": "Téléversement…", + "ad940367a5": "Validation et publication…", + "bf02d6ed9e": "Quoi de neuf dans cette version ?", + "f0c0411549": "Notes de version", + "4895d3e0ee": "Aucun membre de l'équipe disponible.", + "8188f5d765": "peuvent accéder au lien.", + "fe204e06f0": "l'organisation", + "0cd4b3e396": "Tous les membres actuels de", + "f25429e266": "Personnes sélectionnées", + "0a49d901ab": "Organisation", + "6817a7f6f8": "Accès aux skills", + "c15d90c10b": "Un compte Orca Cloud connecté est requis.", + "266527d295": "Publication en tant que {{value0}}{{value1}}.", + "78d863235c": "Accès", + "b3b1d4b911": "SHA-256", + "77f636eac3": "exécutable", + "8edd32622f": "scripts", + "3121f44358": "fichiers", + "2dca0b720b": "Publier une nouvelle version du skill", + "01c5a17e01": "{{value0}} skills", + "01c5a17e02": "Examiner les skills inclus", + "01c5a17e03": "{{value0}} fichiers · {{value1}}", + "unlistedLinkTitle": "Lien non répertorié", + "unlistedPublishingAs": "Publication en tant que {{value0}}. Toute personne disposant du lien peut inspecter et installer ces skills.", + "unlistedLinkDetails": "Le lien n'est ni recherchable ni listé publiquement. Révoquez-le pour bloquer les accès futurs ; les copies installées restent en place.", + "bundleReady": "Lien du bundle de skills prêt", + "publishBundleVersion": "Publier une nouvelle version du bundle de skills", + "shareBundle": "Partager le bundle de skills", + "publishingLink": "Publication du lien…", + "verifyingPackage": "Vérification du package…" + }, + "SkillBundleInstallFlow": { + "01c5a13e01": "Fermer", + "01c5a13e02": "Annuler l'installation", + "01c5a13e03": "Installation…", + "01c5a13e04": "Réessayer {{value0}} skills", + "01c5a13e05": "Installer {{value0}} skills" + }, + "SkillBundleInstallOutcome": { + "01c5a12e01": "Installés", + "01c5a12e02": "Mis à jour", + "01c5a12e03": "Inchangé", + "01c5a12e04": "Conservé en local", + "01c5a12e05": "Échec", + "01c5a12e06": "Annulé", + "01c5a12e07": "L'installation du bundle nécessite votre attention.", + "01c5a12e08": "Skills installés et vérifiés.", + "01c5a12e09": "{{value0}} skills sélectionnés vérifiés.", + "01c5a12e10": "Réessai nécessaire" + }, + "SkillBundleInstallReview": { + "01c5a11e01": "Version immuable", + "01c5a11e02": "{{value0}} skills", + "01c5a11e03": "{{value0}} fichiers", + "01c5a11e04": "{{value0}} scripts", + "01c5a11e05": "{{value0}} exécutable", + "01c5a11e06": "SHA-256", + "01c5a11e09": "Notes de version :", + "01c5a11e0a": "Les skills contiennent des instructions et peuvent inclure des scripts. Considérez-les comme du code provenant de leur auteur.", + "01c5a11e0b": "Skills à installer", + "01c5a11e0c": "Choisissez n'importe quel sous-ensemble de ce bundle.", + "01c5a11e0d": "Tout sélectionner", + "01c5a11e0e": "Aucune description", + "01c5a11e0f": "{{value0}} fichiers · {{value1}} scripts", + "01c5a11e10": "{{value0}} exécutable", + "01c5a11e11": "Les copies locales nécessitent une décision", + "01c5a11e12": "Par défaut, Orca conserve ces copies locales. Sélectionnez uniquement celles que vous voulez abandonner et remplacer.", + "01c5a11e13": "Remplacer {{value0}} ({{value1}})", + "01c5a11e14": "{{value0}} nouveaux", + "01c5a11e15": "{{value0}} inchangés", + "01c5a11e16": "{{value0}} mises à jour", + "01c5a11e17": "{{value0}} conflits" + }, + "SkillShareSelectionControls": { + "01c5a15e05": "Partager {{value0}} skills", + "01c5a15e01": "Annuler le partage", + "01c5a15e02": "Partager des skills", + "01c5a15e03": "{{value0}} sélectionnés", + "01c5a15e04": "Sélectionner tous les résultats", + "01c5a15e06": "Ouvrez ce skill sur la machine qui l'héberge pour le partager.", + "01c5a15e07": "Installez ce skill avant de le partager.", + "01c5a15e08": "Seuls les skills home et espace de travail peuvent être partagés.", + "01c5a15e09": "Un skill portant ce nom est déjà sélectionné depuis une autre source." + }, + "SkillManagedInstallList": { + "86c76cb262": "{{count}} bundle de skills" + }, + "skill-install-progress-state": { + "currentSkill": "Installation de {{value0}} sur {{value1}} : {{value2}}…" + }, + "SkillRow": { + "updatedUnknown": "Aucune date", + "pathCopied": "Chemin copié", + "copyPath": "Copier le chemin", + "notShareable": "Non partageable", + "detailPath": "Chemin", + "detailSource": "Source", + "detailContents": "Contenu", + "skillActions": "Actions pour {{value0}}", + "viewDetails": "Voir les détails", + "deleteSkill": "Supprimer…", + "notDeletable": "Non supprimable" + }, + "SkillSelectionBar": { + "selectAll": "Sélectionner les {{count}} éligibles", + "clear": "Effacer" + }, + "SkillsList": { + "listLabel": "Skills" + }, + "sourceStatus": { + "missing": "Dossier introuvable", + "remoteRepo": "Dépôt distant — non analysé", + "unavailable": "Non analysé" + }, + "sources": { + "heading": "Dossiers de skills" + }, + "sourceKind": { + "home": "Home", + "workspace": "Espace de travail", + "bundled": "Intégré", + "plugin": "Plugin" + }, + "count": { + "skillOne": "{{count}} skill", + "skillOther": "{{count}} skills", + "sourceOne": "{{count}} source", + "sourceOther": "{{count}} sources", + "fileOne": "{{count}} fichier", + "fileOther": "{{count}} fichiers", + "resultOne": "{{count}} résultat", + "resultOther": "{{count}} résultats", + "selected": "{{count}} sélectionnés", + "shareOne": "Partager {{count}} skill", + "shareOther": "Partager {{count}} skills", + "installOne": "Installer {{count}} skill", + "installOther": "Installer {{count}} skills", + "retryOne": "Réessayer {{count}} skill", + "retryOther": "Réessayer {{count}} skills", + "linkOne": "{{count}} lien", + "linkOther": "{{count}} liens", + "deleteOne": "Supprimer {{count}} skill", + "deleteOther": "Supprimer {{count}} skills", + "deletedOne": "{{count}} skill supprimé", + "deletedOther": "{{count}} skills supprimés", + "deleteFolderOne": "{{count}} dossier", + "deleteFolderOther": "{{count}} dossiers", + "deleteLinkOne": "{{count}} lien", + "deleteLinkOther": "{{count}} liens" + }, + "host": { + "local": "Cette machine", + "remote": "Runtime connecté" + }, + "share": { + "fileExecutable": "exécutable", + "fileScript": "script", + "reviewFiles": "Examiner les fichiers exécutables", + "reviewSkills": "Examiner les skills inclus", + "releaseNotesVersion": "Notes de version", + "releaseNotesOptional": "Ajouter des notes de version (facultatif)", + "releaseNotesFirst": "Décrivez cette version", + "readyDescription": "Copiez le lien pour le partager.", + "newVersionDescription": "Publie une version immuable dans le package existant.", + "description": "Publie une version immuable protégée par un lien non répertorié.", + "accessSummary": "Toute personne disposant du lien peut l'installer. Il n'est listé nulle part, et le révoquer bloque les nouvelles installations — pas les copies déjà installées. Publication en tant que {{value0}}.", + "manageLinks": "Gérer ou révoquer ce lien dans les Paramètres", + "scriptOne": "{{count}} script", + "scriptOther": "{{count}} scripts", + "executableOne": "{{count}} exécutable", + "executableOther": "{{count}} exécutables", + "noExecutableContent": "Aucun script ni exécutable", + "newVersionDescriptionPlain": "Ajoute une nouvelle version au lien existant.", + "accessSummaryShort": "Lien non répertorié — toute personne qui le possède peut l'installer. Publication en tant que {{value0}}.", + "accessSummaryPlain": "Lien non répertorié — toute personne qui le possède peut l'installer." + }, + "description": { + "showLess": "Afficher moins", + "showMore": "Afficher plus" + }, + "install": { + "agentsLabel": "Agents", + "agentsCanonical": "Toujours installé pour {{value0}}, qui lisent {{value1}}.", + "agentsNoneChosen": "Aucun agent supplémentaire", + "agentsSummary": "Également : {{value0}}", + "agentNotInstalled": "Non installé", + "bundleDestinationSummary": "{{value0}} nouveaux · {{value1}} déjà installés · {{value2}} nécessitent une décision", + "trustNote": "Les skills sont des instructions et du code provenant de leur auteur. N'installez que ce en quoi vous avez confiance.", + "agentsNone": "Aucun agent sélectionné", + "agentsSelected": "Installation pour : {{value0}}", + "agentAlways": "Toujours", + "agentsCanonicalNote": "{{value0}} lisent {{value1}}, que chaque installation écrit.", + "linkHint": "Ouvrir un lien n'installe jamais rien — vous l'examinez d'abord.", + "rowNeedsDecision": "Nécessite une décision", + "rowInstalled": "Déjà installé", + "rowUpdate": "Mettre à jour", + "rowNew": "Nouveau", + "chooseSkills": "Choisissez ce que vous voulez installer depuis ce lien.", + "singleRunnableWarning": "Ce skill inclut des scripts ou des fichiers binaires.", + "runnableWarning": "{{affectedCount}} des {{selectedCount}} skills sélectionnés incluent des scripts ou des fichiers binaires : {{affected}}.", + "reviewRunnableFiles": "Inclut des scripts ou des fichiers binaires", + "reviewSupportingFiles": "Inclut des fichiers annexes", + "reviewInstructions": "À propos de ce skill", + "supportingFileWarning": "Les skills sélectionnés incluent {{fileCount}} fichiers en plus de SKILL.md.", + "agentAccessWarning": "Les skills contiennent des instructions que votre agent pourrait suivre. Ne continuez que si vous faites confiance à la source de ce lien de partage.", + "fileBinary": "binaire", + "fileRunnable": "exécutable", + "agentsCountBadge": "{{count}} sélectionnés", + "agentsCountLabel": "{{count}} {{label}}", + "targetAgentsHeader": "Agents ciblés", + "deselectAll": "Désélectionner les facultatifs", + "selectAll": "Tout sélectionner", + "canonicalHeader": "Agents standard (toujours inclus)", + "canonicalExplanation": "Ces agents lisent nativement {{root}}, où Orca installe par défaut :", + "additionalAgentsHeader": "Répertoires d'agents supplémentaires" + }, + "filter": { + "allAgents": "Tous les agents", + "sharedAgent": "Partagé (.agents)" + }, + "SkillsSelectionHeader": { + "exit": "Quitter la sélection", + "exitTooltip": "Quitter la sélection · Esc", + "title": "Sélectionner les skills à partager", + "selectAll": "Sélectionner les {{count}} éligibles", + "clear": "Effacer", + "deleteTitle": "Sélectionner les skills à supprimer" + }, + "SkillDetailDialog": { + "agents": "Agents", + "updated": "Mis à jour", + "copy": "Copier" + }, + "SkillSharedLinkRow": { + "contentsUnavailable": "Impossible de charger le contenu de ce lien.", + "loading": "Chargement du contenu…", + "deleteFailed": "Orca n'a pas pu supprimer cet élément du Cloud.", + "deleted": "Supprimé du Cloud", + "confirmDelete": "Confirmer la suppression", + "moreActions": "Autres actions pour {{name}}", + "deletePackage": "Supprimer du Cloud" + }, + "SkillSharedLinksView": { + "loading": "Chargement des liens partagés…", + "noMatches": "Aucun lien ne correspond à cette recherche." + }, + "SkillsSharedLinksHeader": { + "back": "Retour aux skills", + "backTooltip": "Retour aux skills · Esc", + "unlisted": "Non répertorié — seules les personnes disposant d'un lien peuvent l'ouvrir." + }, + "managedInstall": { + "versionLabel": "Version", + "editedWarning": "Vos modifications sont conservées, sauf si vous choisissez de les abandonner.", + "finishInstall": "Terminer l'installation", + "useVersion": "Utiliser cette version", + "reinstall": "Réinstaller", + "cancel": "Annuler", + "sendToMachine": "Envoyer vers une autre machine", + "confirmRemove": "Confirmer la suppression", + "remove": "Supprimer", + "discardEditsRemove": "Abandonner mes modifications et supprimer", + "discardEdits": "Abandonner mes modifications et réinstaller", + "titleMore": "{{name}} +{{count}}", + "scopeWorkspace": "Cet espace de travail", + "scopeGlobal": "Partout", + "installedOn": "Installé le {{date}}", + "stateEdited": "Modifié après l'installation", + "stateMissing": "Des fichiers sont manquants", + "skillEdited": "Modifié", + "skillMissing": "Manquant", + "versionCurrent": "{{date}} (installé)", + "installElsewhere": "Installer ailleurs…" + }, + "skillWarningPreview": { + "bundleDescription": "Un bundle d'aperçu couvrant tous les niveaux d'avertissement.", + "instructionsOnlyDescription": "Instructions entièrement contenues dans SKILL.md.", + "supportingFilesDescription": "Inclut des fichiers de référence lisibles en plus des instructions principales.", + "runnableFilesDescription": "Inclut des scripts qu'un agent peut exécuter avec vos accès.", + "binaryFilesDescription": "Inclut des ressources opaques et un binaire exécutable." + }, + "SkillDelete": { + "dismissResults": "Fermer", + "reasonBundled": "Intégré à Orca — il serait restauré", + "reasonPlugin": "Installé par un plugin — supprimez plutôt le plugin", + "reasonUnowned": "Ce skill vit hors des dossiers de skills d'Orca — supprimez-le là où il est stocké", + "reasonMissing": "Ce skill n'est plus présent sur le disque", + "reasonStale": "Ce skill a changé depuis le chargement de la liste — actualisez puis réessayez", + "blockedLine": "{{count}} × {{reason}}", + "resultLine": "{{count}} × {{label}}", + "statusSkipped": "Ignoré", + "statusBusy": "Occupé — une autre opération de skill est en cours", + "statusPartial": "Partiellement supprimé — certains fichiers restent sur le disque sous un nom caché", + "statusFailed": "Échec — rien n'a changé", + "statusDeleted": "Supprimé", + "confirmHost": "Sur {{host}}.", + "confirmPermanent": "Action irréversible.", + "failed": "Impossible de supprimer les skills", + "hostUnavailable": "Impossible de joindre la machine sélectionnée. Actualisez puis réessayez.", + "hostUnresolved": "La machine qui héberge ces skills est encore en cours d'identification.", + "hostUpdateRequired": "Mettez à jour Orca sur la machine sélectionnée pour supprimer des skills.", + "nothingOne": "Ce skill ne peut pas être supprimé depuis ici.", + "nothingOther": "Aucun des {{count}} skills sélectionnés ne peut être supprimé depuis ici.", + "placementSummaryParts": "Supprime {{parts}} répartis sur {{roots}}.", + "placementJoin": " et ", + "retainedSource": "Le skill lui-même reste à {{path}}." + } + }, + "sidebar": { + "local": { + "base": { + "ref": { + "suggestion": { + "toast": { + "670864ab52": "Mise à jour du {{value0}} local en cours", + "84c62e4d7f": "Impossible d'activer {{value0}}", + "442552c656": "Ouvrez les paramètres et réessayez.", + "f15fd80989": "Votre nouveau worktree est à jour, mais le {{value0}} local est en retard de {{value1}} {{value2}}, les diffs IA peuvent donc se comparer à un historique obsolète. Laissez Orca le maintenir à jour automatiquement. Modifiable à tout moment dans", + "3d260e1a5d": "Paramètres › {{value0}}", + "34a03a6565": "Garder {{value0}} à jour", + "4a18052018": "Le {{value0}} local est en retard sur {{value1}}", + "commit": "commit", + "commits": "commits" + } + } + } + } + }, + "AddProjectFromFolderDialog": { + "7d1f51678c": "Ajouter un projet", + "7726a16374": "Annuler", + "046751dbfb": "Ajoutez ce dossier comme projet Orca distinct.", + "e643b30398": "Projet ajouté sur l'hôte SSH" + }, + "AddRepoCreateStep": { + "0ae45b8238": "my-project", + "a8149a3a5a": "Nom", + "11fd2a7db8": "Dépôt Git", + "c7b9f94456": "Créer un nouveau projet", + "b100311784": "Nommez-le et Orca créera un vrai projet avec des valeurs par défaut sensées.", + "685b5eefe1": "{{kind}} dans {{parent}}", + "2a762f3b19": "Vérification de Git sur cet hôte...", + "fe1e616c5b": "Git est requis pour créer un projet.", + "c234df77f7": "Choisissez ou saisissez un dossier parent sur l'hôte avant de créer.", + "3a13f6e88b": "emplacement non sélectionné", + "6ed14c0281": "dossier hôte non sélectionné", + "5e97f0c4b9": "Projet créé", + "2c12db1511": "Projet déjà ajouté", + "875dda0995": "Saisissez un chemin parent sur l'hôte.", + "ssh_parent_manual": "Saisissez un chemin parent SSH.", + "45b7c26034": "Créer le projet", + "85085d74d2": "Création…" + }, + "AddRepoHostSelector": { + "host": "Hôte", + "local": "Local", + "runtime": "Serveur", + "ssh": "SSH", + "connecting": "Connexion", + "connect": "Se connecter", + "addSshHost": "Ajouter un hôte SSH", + "addRemoteServer": "Ajouter un serveur distant", + "addRemoteHost": "Ajouter un hôte distant", + "addRemoteHostTooltip": "Choisissez SSH pour une machine sur laquelle vous pouvez vous connecter, ou serveur distant pour un serveur Orca.", + "addSshHostDetail": "Utiliser une machine existante via SSH.", + "addRemoteServerDetail": "Se jumeler avec Orca exécuté sur un autre ordinateur.", + "addRemoteHostDetail": "Hôte SSH ou serveur Orca" + }, + "AddRepoNestedImportStep": { + "496f68cf8c": "Analyse des dépôts en cours. Cliquez pour arrêter.", + "a32bef9516": "Arrêter l'analyse", + "2f8298f3c3": "Interrompre l'analyse", + "5f857ba8e6": "dans", + "4df0d08cc5": "Trouvé(s)", + "8db50afe1a": "Importer des dépôts depuis un dossier", + "5b2e6fe3c8": "Importer séparément", + "cf9d382ca1": "Importer", + "220dd32d83": "Analyse...", + "fb33359f69": "Regrouper ces dépôts ?", + "d75170194e": "Choisissez ceci si ces projets vont ensemble — un monorepo, ou juste un ensemble de dépôts liés. Orca les regroupera et vous permettra de travailler depuis le dossier parent.", + "39d51212cc": "Nom du groupe", + "aa0247680d": "Non, importer séparément", + "a0bc4d1f8e": "Oui, importer comme groupe", + "8401a7a0d0": "1 dépôt", + "d4f1df62ef": "{{value0}} dépôts", + "b4263a2ac4": "{{value0}} trouvé(s) dans {{value1}}.", + "24eda6c8b2": "Analyse... {{value0}}", + "6149d5203f": "Aucun dépôt sélectionné. Ouvrez plutôt le dossier parent pour utiliser l'éditeur, le terminal et la recherche sans les fonctionnalités Git.", + "e52454b7f6": "Ouvrir comme dossier" + }, + "AddRepoRemoteStep": { + "5b205b5281": "Interrompre l'analyse", + "6680289908": "/home/user/project", + "ef410aa881": "Chemin sur l'hôte", + "0416bde073": "Ajouter dans les paramètres", + "df6fbcf880": "Aucune cible SSH configurée.", + "44637f43bd": "Cible SSH", + "80557be85a": "Choisissez une cible SSH connectée et saisissez le chemin vers un dépôt Git.", + "91b93a90a4": "Ouvrir un projet sur l'hôte SSH", + "007651bdf9": "Naviguez jusqu'à un répertoire et cliquez sur Sélectionner pour le choisir.", + "dd3ff65486": "Parcourir le système de fichiers distant", + "36d427bb66": "Ajouter un projet sur l'hôte SSH", + "35831a7312": "Ajout...", + "lockedDescription": "Saisissez le chemin vers un dépôt Git sur {{value0}}.", + "lockedDisconnected": "{{value0}} est déconnecté.", + "93e0221434": "Se connecter" + }, + "AddRepoServerStartStep": { + "ae990c86a0": "Retour aux options d'ajout", + "e1710bf831": "Ouvrir comme dossier", + "8da4d1a5be": "Ajouter un projet Git", + "ac66a3ed2d": "Parcourir le système de fichiers de l'hôte", + "92d25420a0": "/home/user/project", + "867692f505": "Chemin sur l'hôte", + "423b5d3d31": "Ajoutez un dépôt Git ou un dossier déjà présent sur l'hôte sélectionné.", + "3d0c035483": "Ouvrir un projet de l'hôte", + "438493f214": "Ou saisissez manuellement un chemin sur l'hôte", + "6b9958492a": "Vous voulez importer plusieurs dépôts d'un coup ? Parcourez jusqu'au dossier parent.", + "d40d751517": "Nouveau dépôt ou dossier", + "a81ffa0a99": "Créer sur l'hôte", + "a2ea37d549": "Dépôt Git distant", + "47759c9491": "Cloner depuis une URL", + "516187414c": "Projet ou dossier existant", + "0adf083af7": "Parcourir l'hôte", + "8efa930eb5": "Ajoutez un autre projet depuis l'hôte sélectionné.", + "39bd249b3a": "Ajouter un projet", + "0f8aba944c": "Naviguez jusqu'à un répertoire et cliquez sur Sélectionner pour le choisir." + }, + "AddRepoStartSteps": { + "87596c1446": "Autres moyens d'ajout", + "acf895cb42": "Ajoutez un projet pour démarrer avec Orca.", + "d13757911c": "Ajouter un projet", + "d301db1c9a": "Analyse des dépôts en cours. Cliquez pour arrêter.", + "69ea7f8dc4": "Arrêter l'analyse", + "9906cae183": "Interrompre l'analyse" + }, + "AddRepoStepIndicator": { + "3bb655c117": "Retour" + }, + "AddRepoSteps": { + "569326d9cc": "Choisir un dossier", + "a93ef169b5": "Parcourir le système de fichiers de l'hôte", + "2ce3f6edf8": "/path/to/destination", + "04a4c4e84a": "Emplacement du clone", + "b698a4a29d": "https://github.com/user/repo.git", + "3d4acbe693": "URL Git", + "5b2ea674b1": "Saisissez l'URL Git et choisissez où le cloner.", + "c05f88a31f": "Cloner depuis une URL", + "fe8e629fe3": "Naviguez jusqu'à un répertoire et cliquez sur Sélectionner pour le choisir.", + "df8b0e6c22": "Projet ajouté sur l'hôte SSH", + "3e64e8a70d": "Échec de la connexion", + "32a7256d85": "Cloner", + "69f5b5380d": "Clonage...", + "cloneOnHostDescription": "Saisissez l'URL Git et choisissez où le cloner sur {{value0}}.", + "cloneParentFolder": "Dossier parent", + "remoteCloneParentPlaceholder": "/home/user/projects" + }, + "AutoRenameFailedDialog": { + "aed1623b1e": "Fermer", + "eab8b45238": "Copier l'erreur", + "a23b22d16f": "Copied", + "74fc00776f": "Détails de l'erreur", + "3afcad0497": "à partir du premier message de l'agent.", + "ff62a18580": "Orca n'a pas pu générer un nom de branche pour", + "ca3b225195": "Échec du nommage automatique de la branche" + }, + "CreateProjectLocationField": { + "95548e33bf": "Choisir un dossier parent...", + "632b456b1b": "Modifier", + "afaf54f245": "Modifier le dossier parent", + "f520f83a97": "Parcourir le système de fichiers de l'hôte", + "2a20a603a3": "/home/user/projects", + "134e37f711": "Emplacement", + "b589b77997": "Naviguez jusqu'à un répertoire et cliquez sur Sélectionner pour le choisir." + }, + "DeleteWorktreeDialog": { + "ff2a74ac0e": "et", + "91492c9ad6": "Supprimer", + "4f6750ca7b": "Échec de la suppression de l'espace de travail", + "42e610d6cf": "Échec de la suppression forcée", + "5cc1a6701c": "Ouvrir les paramètres", + "2b56b35f53": "Vous pouvez changer cela dans les paramètres.", + "dd3a45bbbd": "Cette confirmation sera ignorée la prochaine fois.", + "fc23c4cbdf": "Supprimer l'espace de travail", + "86f0ae1257": "Supprimer les espaces de travail" + }, + "DeleteWorktreeDirtyChangeHint": { + "8e2994ce28": "Supprimer cet espace de travail efface définitivement ces modifications du disque." + }, + "DeleteWorktreeLineageNotice": { + "ad407c2d55": "de plus", + "a940f3c96e": "Les espaces de travail enfants seront supprimés", + "29b98bf9cd": "La suppression de cet espace de travail supprime aussi {{value0}} espaces de travail enfants.", + "66798cc6a2": "La suppression de cet espace de travail supprime aussi 1 espace de travail enfant." + }, + "DeleteWorktreeSkipConfirmOption": { + "29aefb7e52": "Ne plus demander" + }, + "DeleteWorktreeWarningPanels": { + "026738155a": "(le répertoire de clone d'origine).", + "c4f96a6e18": "worktree principal", + "e3be9eba15": "Ceci est le" + }, + "ImportedWorktreesVisibilityLine": { + "b7a87dc32f": "Afficher dans la liste des worktrees", + "ad99f4eea9": "Garder masqué", + "9f4f14e821": "Modifiable plus tard depuis le menu du projet.", + "b2bc47c080": "autres emplacements", + "b47ba1a9d2": "Aperçu de {{value0}}", + "2251d41ebb": "Groupes de worktrees masqués", + "f54f2bec5d": "{{value0}} worktrees masqués pour {{value1}}", + "5a9688802a": "Afficher {{value0}} de plus", + "294de4aeb2": "Afficher moins" + }, + "NonGitFolderDialog": { + "e52454b7f6": "Ouvrir comme dossier", + "05b33a17a9": "Annuler", + "8fba4b8cbb": "Ce dossier n'est pas un dépôt Git. Vous aurez l'éditeur, le terminal et la recherche, mais les fonctionnalités basées sur Git ne seront pas disponibles.", + "c49fb13492": "Échec de l'ajout du dossier sur cet hôte", + "9a766f33ac": "Ce chemin a été vérifié sur l'hôte SSH.", + "79fd02cf5f": "Ce chemin a été vérifié sur {{hostName}}.", + "8851b77327": "Ce chemin a été vérifié localement." + }, + "OrcaYamlTrustDialog": { + "f3e2b868fb": "Exécuter les hooks", + "43b7bec4cd": "Ne pas exécuter", + "c494b3ccb1": "dans", + "79afc6772b": "orca.yaml", + "531689199b": "Toujours faire confiance", + "bf800b7e04": ". N'exécutez que si vous faites confiance", + "831f2cd9f0": "s'exécute sur votre machine", + "aa3ffb33fb": "De ce dépôt, le", + "c55beddbf8": "a changé depuis votre dernière approbation. Revérifiez avant son exécution", + "95bf974a1a": "script {{value0}}", + "9e52effffd": "Nouveau script {{value0}}", + "e4a51dc4b3": "Exécuter {{value0}} de {{value1}} ?", + "02b0ede5ad": "Le {{value1}} de {{value0}} a changé — exécuter la nouvelle version ?" + }, + "PendingWorktreeRow": { + "af21e953d1": "Annuler la création du worktree", + "188f6922a0": "Annuler" + }, + "ProjectGroupDeleteDialog": { + "ca65b78f78": "Annuler", + "9be10d49ea": "et dissocier ses projets.", + "69f5cb97d0": "Supprimer", + "591f330288": "Supprimer le groupe de projets", + "2c14ce677a": "Suppression...", + "0e0e6764af": "Projets contenus", + "ad407c2d55": "de plus", + "removeContainedProjectSingular": "Retirer 1 projet contenu", + "removeContainedProjectPlural": "Retirer {{value0}} projets contenus", + "eeabb8e8e4": "Retirer {{value0}} {{value1}} d'Orca", + "55f75628c0": "Les dossiers de projets sur le disque ne sont pas supprimés.", + "897e5d3d4c": "Supprimer le groupe et retirer les projets", + "fec7e9c8ae": "Supprimer le groupe" + }, + "ProjectGroupNameDialog": { + "d99a034073": "Annuler", + "83dfbc5313": "Nom du groupe", + "4a64e78822": "Enregistrement..." + }, + "RemoteFileBrowser": { + "2300612806": "Tapez pour filtrer ou saisissez un chemin…", + "9e060f5815": "Sélectionner un dossier", + "f8b1deb1a4": "Annuler", + "51001182e3": "Répertoire vide", + "971d85cc84": "S'ouvre comme projet sur cet hôte · {{value0}}", + "00c4235c10": "Aucun résultat pour « {{value0}} »", + "largeInputNoMatches": "Aucun résultat pour cette saisie longue" + }, + "RemoveFolderDialog": { + "4dc5b5065b": "Supprimer", + "d36883e046": "Annuler", + "b79b39d865": "Retirer le projet", + "removeDescriptionSsh": "Ceci retire uniquement {{name}} d'Orca. Ses fichiers restent sur {{host}} — rajoutez cet hôte SSH pour le récupérer.", + "removeDescriptionLocal": "Ceci retire uniquement {{name}} d'Orca. Il est toujours sur votre disque.", + "removeDescriptionVmRecipe": "Ceci retire {{name}} d'Orca. Sa recette de VM détermine si l'environnement et ses fichiers sont définitivement supprimés." + }, + "ScrollToCurrentWorkspaceToolbarButton": { + "23989bb663": "Révéler l'espace de travail actif" + }, + "SetupGuideSidebarEntry": { + "b0a7bfc34c": "Masquer de la barre latérale", + "88d402b71d": "Checklist d'intégration" + }, + "SetupScriptPromptCard": { + "ff1e819a11": "Ajouter un script de setup", + "70715947fb": "Le script de setup ne peut pas être vide", + "888b83bf78": "Échec de l'enregistrement du script de setup", + "a49196d538": "S'exécute quand Orca crée un nouveau worktree.", + "d9f2db2738": "paramètres du projet", + "a5bb8c5135": "Enregistré dans les", + "dcaa645da5": "ok" + }, + "SetupScriptPromptCardViews": { + "96a7f4198c": "Enregistrer le setup local", + "3933401d28": "Configurer", + "31b8b01a45": "Paramètres", + "4a98f907ae": "Réessayer", + "0a98169776": "Ajoutez une commande de setup à exécuter quand Orca crée de nouveaux worktrees.", + "8349e3fa4c": ". Enregistrez-la pour l'exécuter sur les nouveaux worktrees.", + "b56d1322f7": "Une commande de setup a été trouvée dans", + "aef6c0a213": "Enregistrez la commande détectée pour l'exécuter chaque fois qu'Orca crée un worktree.", + "660cdc17f8": "scripts de setup partagés. Ajoutez une commande locale, ou changez la source dans les paramètres.", + "8f6be51aa1": "orca.yaml", + "bb879db364": "Ce dépôt ignore les", + "0155fb9ed3": "Impossible de vérifier le script de setup de ce dépôt pour le moment.", + "eefa756190": "Configurer manuellement", + "ca4efcbc25": "Enregistrer", + "d02e6a42b1": "Détecté depuis", + "fdbc6cb064": "Script de setup détecté", + "7275f674cc": "Setup détecté", + "822ff300ad": "Ignorer", + "5bfd5c8779": "Ignorer les scripts de setup" + }, + "SidebarFeedbackDialog": { + "8bf619e4cf": "Annuler", + "8de03e23c5": "Envoyez avec votre commentaire texte uniquement, ou connectez `gh` pour inclure votre identité GitHub.", + "d20439c560": "Vérification de l'identité GitHub…", + "5b120b9634": "Envoyer anonymement", + "c9e5ea0791": "GitHub :", + "d46ddd66fc": "Que pouvons-nous améliorer ?", + "3460258a54": "Suivre sur X", + "26108d3699": "Rejoindre Discord", + "d245c4ef6c": "Issues GitHub", + "9b33530b3d": "Autres moyens de nous contacter", + "a828fa4aee": "Partagez ce qui fonctionne, ce qui est cassé, ou ce qu'Orca devrait faire ensuite.", + "0eb643f07f": "Envoyer un commentaire", + "60b721e857": "Échec de l'envoi du commentaire. Réessayez.", + "7a46c228b8": "Merci pour votre commentaire.", + "a2fd890d9e": "Saisissez un commentaire avant d'envoyer.", + "f2e42e1307": "Envoyer", + "69969ba364": "Envoi…", + "imageReadFailed": "Impossible de lire les images jointes. Essayez de les joindre à nouveau.", + "attachWhileSending": "Attendez la fin de l'envoi du commentaire en cours avant de joindre d'autres images.", + "imagesNotDelivered": "Commentaire envoyé, mais la livraison des images n'a pas pu être confirmée." + }, + "SidebarFilter": { + "e3b3898218": "Ajouter un projet", + "92a23e6d07": "Réinitialiser les filtres", + "81ded53722": "SSH", + "b9e8802e73": "Aucun projet ne correspond", + "779b7ba05d": "Effacer", + "139877b384": "Tout sélectionner", + "5f7085a077": "Projets", + "e5cb32a898": "Masquer la branche par défaut", + "638a2d221d": "Masquer les espaces en veille", + "f506a1262a": "Filtrer les espaces de travail", + "75405270ed": "Modifier les filtres ({{value0}} actifs)", + "489d1c8c9f": "Rechercher des projets...", + "ee240a39eb": "Modifier les filtres", + "automationCreated": "Masquer ceux créés par l'automatisation", + "cliCreated": "Masquer ceux créés par le CLI", + "detachedHead": "Masquer les HEAD détachés", + "keepDefaultBranch": "Sauf la branche par défaut", + "keepDefaultBranchAria": "Garder la branche par défaut visible tout en masquant les espaces de travail en veille" + }, + "SidebarHeader": { + "25a95899c9": "Ajouter un projet", + "92154beb7e": "Nouvel espace de travail", + "49f62c5665": "Tableau des espaces de travail", + "5c9c7c16aa": "Ajoutez un projet pour créer des espaces de travail", + "ca6f729da2": "Nouvel espace de travail ({{value0}})", + "a30e34eb5c": "Fermer le tableau des espaces de travail" + }, + "SidebarNav": { + "80611a8b10": "Recherche", + "0c3395fd32": "Rechercher parmi les worktrees et les onglets du navigateur", + "c86d83b5c3": "Nouveau", + "1b5c41caee": "Orca Mobile", + "9c95e1ce91": "Agents", + "f323383e9a": "Automatisations", + "e7ad3c540d": "Ouvrir les tâches Jira", + "c39ab10000": "Ouvrir les tâches Linear", + "196c1b5362": "Ouvrir les tâches GitLab", + "0ccba862b8": "Ouvrir les tâches GitHub", + "fee535205b": "Tâches", + "d599269755": "Masquer de la barre latérale", + "artifacts": "Artefacts", + "skills": "Skills" + }, + "SidebarRepositoryFilterSection": { + "d3a9c4cea1": "Effacer", + "7679f0c268": "Projets", + "f10ca29601": "Retirer le filtre {{value0}}", + "2656053db4": "SSH", + "83a820fa71": "Filtrer les projets...", + "5a273fbfce": "Ajouter un projet...", + "4815c70605": "Aucun projet ne correspond", + "bbbc6e8e3b": "Aucun projet non sélectionné ne correspond", + "allProjects": "Tous les projets", + "selectedProjectsCount": "{{value0}} projets" + }, + "SidebarSettingsHelpMenu": { + "ad3d3ed7f1": "Redémarrer Orca", + "29c56f30ee": "Rechercher des mises à jour", + "eb9884e55b": "Discord", + "5687ab246a": "GitHub", + "5f83d86d92": "Journal des modifications", + "cdc87f897e": "Docs", + "e565171a7c": "Raccourcis clavier", + "4cf5b868d7": "Envoyer un commentaire", + "a428c25998": "Paramètres", + "2991a0106c": "Aide", + "4e8f5710d3": "Impossible de redémarrer Orca.", + "5161eef55d": "Redémarrage d'Orca…", + "d396773ef0": "vérification", + "f8a2c91d4e": "Jalons", + "b7e4d2a19c": "Prise en main", + "c4f8e1b72a": "X" + }, + "SidebarToolbar": { + "87d0064026": "Tableau des espaces de travail déplacé dans la barre inférieure", + "a30e34eb5c": "Fermer le tableau des espaces de travail", + "49f62c5665": "Tableau des espaces de travail", + "19e32d0e5f": "Ouvrir le sélecteur de dossiers pour ajouter un projet", + "abc62b6328": "Ajouter un projet" + }, + "SidebarWorkspaceFilterSection": { + "c3fa13dc2e": "Masquer la branche par défaut", + "ed1611b65b": "Masquer les espaces en veille", + "82594419ba": "Filtres", + "automationCreated": "Masquer ceux créés par l'automatisation", + "cliCreated": "Masquer ceux créés par le CLI", + "detachedHead": "Masquer les HEAD détachés", + "keepDefaultBranch": "Sauf la branche par défaut", + "keepDefaultBranchAria": "Garder la branche par défaut visible tout en masquant les espaces de travail en veille", + "otherClients": "Masquer les espaces de travail d'autres clients", + "otherClientsAria": "Masquer les espaces de travail créés depuis d'autres clients Orca sur des serveurs distants partagés" + }, + "sidebarHostOptions": { + "3e102f111c": "Tous les hôtes", + "visibleHostsCount": "{{value0}} hôtes" + }, + "SidebarHostScopeStrip": { + "scopedTo": "{{value0}} visible(s)", + "backToAll": "Tous les hôtes" + }, + "SidebarWorkspaceOptionsMenu": { + "hosts": "Hôtes", + "showSection": "Afficher", + "allHostsDetail": "Afficher tous les hôtes", + "configuredSshHost": "SSH configuré", + "projectSshHost": "SSH du projet", + "activeRuntimeHost": "Serveur actif", + "projectRuntimeHost": "Serveur du projet", + "95c9754653": "Disposition de l'activité des agents", + "3d4b9c4997": "Survol", + "ba87080fb7": "Afficher les propriétés", + "320b675c9a": "Disposition en cartes", + "newCardDisplay": { + "title": "Affichage des cartes" + }, + "09faabd875": "Ordre des projets", + "7bada3b1ab": "Trier par", + "dc0bb670bc": "Grouper par", + "631b97eea9": "Portée des hôtes", + "9919ae1082": "Options des espaces de travail", + "bc96dbd041": "Options des espaces de travail ({{value0}})", + "af9249c505": "Activité d'espace de travail la plus récente", + "b451c8b162": "Récents", + "6664282a7b": "Glissez les projets pour les ranger", + "7b316bdd51": "Manuel", + "7153d07485": "Glissez les espaces de travail pour les ranger au sein de chaque groupe.", + "2170d553cf": "Projet", + "b759bb87ee": "Agents nécessitant une attention, puis activité la plus récente.", + "503462f2b4": "Activité des agents", + "3728165cdd": "Nom", + "2a81e07366": "Liste complète", + "25105b28cb": "Compact", + "d7084e8bc8": "Activité des agents", + "automation": "Automatisation", + "b64d8bcca0": "Ports", + "26c71e536c": "Notes", + "b8dcc6f321": "Lien PR/MR", + "ca4d3c522e": "Ticket Linear", + "jiraIssue": "Ticket Jira", + "91dfc653e8": "Ticket GitHub", + "bdd23b4e07": "Issues GitHub", + "44713a5d04": "Issues Linear", + "jiraIssues": "Tickets Jira", + "cc17bd443b": "Détaillé", + "0f9b959b31": "PR", + "e029a2d775": "Statut", + "c2c7a45cda": "Aucun", + "680043342f": "compact", + "c7591b6014": "dépôt", + "2d4f0eb933": "Par défaut", + "1a0eec0d35": "Statut", + "b5536d5a88": "Tâches", + "8d62c68b35": "Notes", + "2d74665a56": "Ports", + "65a9820bd1": "Statuts des agents", + "219ebf1961": "Nom de branche", + "folderPathIdentity": "Branche / chemin de dossier", + "cli": "Orca CLI" + }, + "SshTargetRow": { + "4677394048": "Connexion…", + "75ad429b5d": "Se connecter" + }, + "WorkspaceKanbanCard": { + "cefae8983e": "Épinglés" + }, + "WorkspaceKanbanDrawerHeader": { + "f369f5c5a3": "Fermer", + "e1a34450fc": "Organisez les espaces de travail par statut et ouvrez les cartes d'espace de travail.", + "81870af08f": "sélectionnés", + "c6a77ab0f4": "Tableau des espaces de travail" + }, + "WorkspaceKanbanPinDropTarget": { + "c30151c5ee": "Déposez ici pour épingler sans changer de statut.", + "8fae2d0862": "Épinglés" + }, + "WorkspaceKanbanSettingsMenu": { + "79eb990aa4": "Ajouter un statut", + "054cb50df7": "Supprimer {{value0}}", + "b45b350eb0": "Déplacer {{value0}} vers la gauche", + "8ce44af9a8": "Renommer {{value0}}", + "395e541d5d": "Statuts", + "34f03eb0de": "Paramètres du tableau", + "26cbc92150": "Paramètres du tableau des espaces de travail", + "87d24a0c2f": "Synchroniser le tableau et le statut des issues", + "4c2eaa78cc": "Déplacer un espace de travail lié met à jour le statut de son issue Linear quand un état de workflow correspondant existe." + }, + "WorkspaceKanbanStatusLane": { + "8ad104642b": "Vide", + "3611d1ae7f": "Redimensionner les colonnes du tableau des espaces de travail", + "2df01a03ff": "Aucun résultat" + }, + "WorkspaceStatusAppearancePopover": { + "514be2f569": "Définir la couleur de {{value0}} sur {{value1}}", + "8be427206b": "Icône", + "2ac106f6b2": "Couleur", + "74b1413279": "Apparence", + "ccbd1e2c69": "Personnaliser l'apparence de {{value0}}" + }, + "WorktreeCard": { + "a88c92d0e3": "{{value0}}/{{value1}} existe déjà.", + "6f09f58541": "Supprimer l'espace de travail", + "0777de5970": "Worktree principal (répertoire de clone d'origine)", + "0f33af979b": "Checkout partiel. Les fichiers hors de ces chemins ne sont pas sur le disque.", + "4f964d5e8c": "sparse", + "7d517f82e2": "principal", + "4eba2ea99e": "Échec du nommage automatique. Cliquez pour voir les détails.", + "74522ee457": "échec du renommage", + "02e19349f4": "Échec du renommage auto : voir l'erreur", + "691ccfd622": "Suppression…", + "35ccfe2475": "Projet {{value0}}", + "1d66d84f0b": "string", + "57eaa61b55": "Masquer les espaces de travail enfants", + "8cb634cda6": "Afficher les espaces de travail enfants", + "01f45d3d8a": "barre latérale", + "93aebe4529": "Dossier", + "0d224eff10": "Worktree principal", + "ca74db7550": "Projet sur hôte SSH", + "021538e1d1": "SSH déconnecté", + "runtimeHostDisconnected": "Serveur déconnecté", + "runtimeHostDisconnectedNamed": "{{hostName}} déconnecté", + "runtimeHostProject": "Projet sur serveur Orca", + "runtimeHostProjectNamed": "Projet sur {{hostName}}", + "automationCreated": "Créé par automatisation", + "branchIdentity": "Branche", + "branchFolderPathIdentity": "Branche ou chemin de dossier", + "ef18787206": "En attente de suppression" + }, + "WorktreeCardAgents": { + "1b0a156717": "Agents" + }, + "WorktreeCardReviewDetailSection": { + "copyLink": "Copier le lien", + "reviewHeader": "{{value0}} #{{value1}}" + }, + "WorktreeCardMeta": { + "3e65e11cc6": "Métadonnées de l'espace de travail", + "c7fa72ead0": "Modifier les notes", + "93cbea12c2": "Notes", + "eace1d2cf6": "neutre", + "dbe2d18972": "Plus d'actions {{value0}}", + "ae76907ca6": "Délier {{value0}}", + "ad25c3ff05": "Voir sur {{value0}}", + "2c67730e07": "Ouvrir dans Orca", + "e42941631a": "Voir sur Linear", + "5e982e6128": "Linear {{value0}}", + "807b13b9ec": "Modifier l'issue", + "b22f058067": "Voir sur GitHub", + "e97d8f2876": "Ticket #{{value0}}", + "moreIssueActions": "Plus d'actions d'issue", + "copyLink": "Copier le lien", + "copyLinkSuccess": "{{value0}} copié", + "copyLinkFailure": "Échec de la copie du lien", + "issueLinkLabel": "Lien d'issue", + "reviewLinkLabel": "Lien {{value0}}", + "3ea2702e62": "Lié à {{value0}} #{{value1}}", + "b105fd3057": "Lié à Linear {{value0}}", + "3f2649eeb8": "Issue #{{value0}} liée", + "fe075cb851": "Notes de l'espace de travail", + "automationHeader": "Automatisation", + "openAutomation": "Ouvrir l'automatisation", + "openAutomationRun": "Ouvrir l'exécution", + "automationCreated": "Créé par automatisation", + "checkingAutomationAvailability": "Vérification de la disponibilité de l'automatisation...", + "automationMissing": "Cette automatisation n'est plus disponible.", + "automationRunMissing": "L'historique d'exécution n'est plus disponible.", + "automationAvailabilityUnavailable": "La disponibilité de l'automatisation n'a pas pu être vérifiée.", + "cliHeader": "Orca CLI", + "cliCreatedFromAgent": "Créé par un agent via `orca worktree create`", + "cliCreatedFromShell": "Créé via `orca worktree create`", + "cliStartupAgent": "Démarré avec {{value0}}", + "cliCreated": "Créé par Orca CLI", + "jiraIssue": "Jira {{value0}}", + "viewOnJira": "Voir sur Jira", + "linkedJira": "Lié à Jira {{value0}}" + }, + "WorktreeCardMetadataStatusBadges": { + "fe188062a1": "État : ouvert", + "2931b42b09": "État : brouillon {{value0}}", + "e888362def": "État : fermé", + "f394b3e86e": "État : fusionné", + "af2b07bda5": "État : {{value0}}", + "29df45afa2": "MR" + }, + "WorktreeCardPorts": { + "34f733dda2": "Aller au worktree", + "3240f320d7": "Ports actifs", + "3e5f66564e": "Espace de travail indisponible", + "2f854442ff": "Arrêter le processus", + "c8067a829a": "Copier {{value0}}", + "33bc7d7495": "Ouvrir dans le navigateur", + "9950fe2d20": "Échec de l'actualisation des ports", + "5d1a5d51bb": "Processus arrêté sur {{value0}}", + "c89f290e25": "{{value0}} copié", + "d1113f4660": "Échec de l'ouverture du navigateur", + "fed49903c9": "{{value0}} {{value1}} actif(s)" + }, + "WorktreeContextMenu": { + "c39c37676a": "Créez un groupe et déplacez-y ce projet.", + "6664418e98": "Nouveau groupe de projets", + "e091caab15": "Le projet est introuvable", + "439fa94d53": "Mettre à jour", + "579b1a8e61": "Retirer du parent", + "8d9cd19d09": "Ouvrir le worktree parent", + "d35dfeae58": "Retirer du groupe", + "76865d827f": "Déplacer vers un groupe", + "503ec0f8e6": "Nouveau groupe à partir du projet", + "3350101edb": "Copier le chemin", + "f4475537d8": "Supprimer", + "f5ac91531d": "Retirer le projet d'Orca", + "b42391d8bf": "Suppression…", + "0918b35e4f": "Fermez tous les panneaux actifs de cet espace de travail pour libérer de la mémoire et du CPU.", + "7d190f7d2b": "Fermez tous les panneaux actifs des espaces de travail sélectionnés pour libérer de la mémoire et du CPU.", + "84cdbb7e30": "Déplacer vers le statut", + "56cde9e8e6": "Déplacer les statuts vers", + "f50603c6b2": "Marquer comme non lu", + "8dacff1fe0": "Marquer comme lu", + "3baa7d6507": "Épingler", + "697d0f6e1b": "Désépingler", + "250de158fd": "Retirer l'espace de travail", + "changeParentWorkspace": "Changer le worktree parent...", + "setParentWorkspace": "Définir le worktree parent...", + "workspaceSection": "Espace de travail", + "primaryDeleteDisabled": "Worktree principal — impossible à supprimer. Retirez plutôt le projet.", + "deleteWorktree": "Supprimer le worktree", + "sleepWithDescendants": "Mettre en veille avec les descendants ({{value0}})", + "sleepWithDescendantsDescription": "Fermez les panneaux actifs de cet espace de travail et de chaque descendant imbriqué pour libérer de la mémoire et du CPU.", + "deleteWithDescendants": "Supprimer avec les descendants…" + }, + "WorktreeList": { + "7a8b9c0d1e": "Mise à jour requise", + "hostAuthNeeded": "Authentification requise", + "hostDisconnected": "Déconnecté", + "d880ea0744": "Créez un groupe et déplacez-y ce projet.", + "bc1460beb3": "Modifiez le nom du groupe affiché dans la barre latérale.", + "13757c053c": "Nouveau groupe de projets", + "f9dc6cc5d3": "Renommer le groupe de projets", + "370c6a55dd": "Réinitialiser les filtres", + "b7acbf038b": "Aucun espace de travail trouvé", + "0c6ee14f23": "enfant", + "5fc9d1891b": "Suppression…", + "c83968f87f": "Retirer le projet", + "64e55f7f01": "Retirer du groupe", + "4a08fb55f2": "Déplacer vers un groupe", + "cbfd565f83": "Nouveau groupe à partir du projet", + "e82d3589a1": "Changer l'icône du projet", + "2cdffbc728": "Paramètres du projet", + "2ef41bf9a7": "Actions du projet", + "609633a9e6": "Actions du projet pour {{value0}}", + "902115cdbe": "Supprimer le groupe", + "4d7b73658c": "Renommer le groupe", + "79465e9034": "Actions du groupe pour {{value0}}", + "bfbedc547b": "Worktrees", + "45fbfe0335": "renommer", + "ebc5c7dcef": "Masquer les espaces de travail enfants", + "84a2238242": "Afficher les espaces de travail enfants", + "045a8aed48": "enfants", + "2ca6e29a3c": "dépôt", + "bb85cd86ba": "Créer un espace de travail pour {{value0}}", + "ebadb7eadb": "{{value0}} {{value1}} enfant {{value2}}", + "20bebf9c7f": "Afficher {{value0}} espace de travail enfant", + "c1f4a31623": "Afficher {{value0}} espaces de travail enfants", + "e97297cb75": "Masquer {{value0}} espace de travail enfant", + "0cd15956d4": "Masquer {{value0}} espaces de travail enfants", + "bd37a57ac8": "Créer un espace de travail pour {{value0}}", + "b667b59632": "Certains projets n'ont pas pu être retirés d'Orca", + "f94466bc39": "{{value0}} projet{{value2}} sur {{value1}} est resté après la suppression du groupe.", + "groupDeleteFailed": "Échec de la suppression du groupe", + "groupDeleteFailedDesc": "Une erreur est survenue lors de la suppression du groupe. Aucun projet n'a été retiré.", + "groupRenameFailed": "Échec du renommage du groupe", + "groupRenameFailedDesc": "Orca n'a pas pu confirmer le nouveau nom auprès de l'hôte du groupe. Revérifiez le groupe après reconnexion.", + "failedNestWorkspace": "Échec de l'imbrication de l'espace de travail", + "sidebarRowMissing": "La cible n'existe plus", + "failedUnnestWorkspace": "Échec de la désimbrication de l'espace de travail" + }, + "WorktreeMetaDialog": { + "3db0a2a593": "Annuler", + "b48c271d39": "pour enregistrer, Maj+Entrée pour un retour à la ligne.", + "7f0be5e9a6": "Prend en charge **markdown** — gras, listes, `code`, liens. Appuyez sur Entrée ou", + "030d484fc0": "Notes sur ce worktree...", + "9c1d1e9b71": "Commentaire", + "5ae06f40fd": "Collez une URL de pull request ou saisissez un numéro. Laissez vide pour supprimer le lien.", + "077a4f7b5c": "N° de PR ou URL GitHub", + "1b91db7e14": "PR GH", + "7c454be4c5": "Collez une URL d'issue ou saisissez un numéro. Laissez vide pour supprimer le lien.", + "029ea5ec57": "Ouvrir l'issue GitHub", + "741279e7b7": "N° d'issue ou URL GitHub", + "645fa4a0fd": "Issue GH", + "459ad7f650": "Change uniquement le nom affiché dans la barre latérale — le dossier sur disque reste identique. Laissez vide pour utiliser le nom de branche ou de dossier.", + "7f21e0464f": "Nom d'affichage personnalisé...", + "ad5e4e514f": "Nom d'affichage", + "65770ad0f0": "Modifiez les liens GitHub et les notes de cet espace de travail.", + "382fd11a3e": "Modifier les détails du worktree", + "2174f17011": "Enregistrer", + "61d6f612cf": "Enregistrement...", + "a0d191b7a7": "Modifiez les liens d'issues, de pull requests et les notes de cet espace de travail." + }, + "WorktreeOpenInMenu": { + "localOnly": "Local uniquement", + "remoteSsh": "SSH distant", + "remoteRuntimeUnsupported": "L'ouverture de ce chemin dans une application locale n'est pas disponible.", + "remoteRuntimeUnsupportedDetail": "Passez à un espace de travail local ou SSH, puis réessayez.", + "sshTargetNotFound": "L'hôte SSH n'est plus disponible.", + "sshTargetNotFoundDetail": "Actualisez les espaces de travail ou reconnectez l'hôte, puis réessayez.", + "sshTargetInvalid": "La configuration de l'hôte SSH est incomplète.", + "sshTargetInvalidDetail": "Modifiez l'hôte SSH ou reconnectez-le, puis réessayez.", + "sshAliasRequired": "VS Code a besoin d'un alias de config SSH pour cet hôte.", + "sshAliasRequiredDetail": "Ajoutez un alias Host pour {{host}}:{{port}} dans votre config SSH locale, reconnectez l'espace de travail, puis réessayez.", + "remoteEditorUnsupported": "Cette application ne peut pas ouvrir les espaces de travail SSH.", + "remoteEditorUnsupportedDetail": "Choisissez VS Code ou utilisez l'application localement.", + "remotePathInvalid": "Le chemin n'est pas valide pour l'hôte SSH.", + "remotePathInvalidDetail": "Actualisez l'espace de travail avant de réessayer.", + "remoteLaunchFailed": "Impossible d'ouvrir le chemin dans VS Code.", + "remoteLaunchFailedDetail": "Vérifiez la commande VS Code configurée sur cette machine.", + "1417fd8380": "Personnaliser les applications...", + "8009ab69a6": "Ouvrir dans", + "bd0e8159f8": "Vérifiez la commande de l'éditeur ou la configuration du gestionnaire de fichiers sur cette machine.", + "9a5381eb09": "Impossible d'ouvrir le dossier de l'espace de travail.", + "0bed8727db": "Il a peut-être été déplacé ou supprimé. Actualisez les espaces de travail ou retirez-le d'Orca.", + "3921d3d9a5": "Le dossier de l'espace de travail est introuvable.", + "f387af445b": "Le chemin de l'espace de travail n'est pas un chemin local valide.", + "3ec372b664": "file-manager" + }, + "WorktreeTitleInlineRename": { + "2f42ae024f": "Non lus :", + "bff3bdd00c": "Renommer l'espace de travail", + "8df295a78d": "Échec du renommage de l'espace de travail." + }, + "WorktreeVisibilityHelpPopover": { + "c41f2d7e90": "Quels worktrees sont masqués par défaut ?", + "8db4e19a26": "Ce paramètre ne masque jamais les worktrees créés via Orca.", + "ec1e6a10fb": "Les autres worktrees démarrent masqués pour éviter un encombrement inattendu de la barre latérale.", + "1c68c9cf77": "Activez une source pour tous les worktrees actuels et futurs, ou affichez des worktrees individuels ci-dessous." + }, + "HiddenWorktreeRecoveryList": { + "64e6f53f05": "Affichez-en un sans activer sa source.", + "showWorktree": "Afficher {{value0}} à {{value1}}" + }, + "WorktreeVisibilityDialog": { + "83a5ba8dd1": "Worktrees non Orca", + "f1f71b9f02": "Toujours afficher", + "759371df43": "Masquer", + "25ddf19920": "{{value0}} actuellement masqués", + "8372e4bbd9": "{{value0}} actuellement affichés", + "5d02a5647f": "Masqué dans la barre latérale", + "3e045d4cb8": "Affiché dans la barre latérale", + "a3f19c07d2": "Vérification…", + "b8d24e61f5": "Impossible de lister les worktrees de ce dépôt.", + "c5e70a93b1": "Réessayer", + "7d21c5e848": "Worktrees masqués ({{value0}})", + "2f80cd4b97": "Affichage…", + "e64b81d3a9": "Afficher", + "d40d436fc2": "Impossible de mettre à jour la visibilité du worktree. Réessayez.", + "unsupportedHost": "Cet hôte ne prend pas en charge la visibilité des worktrees par source. Mettez à jour Orca sur l'hôte pour modifier ce paramètre.", + "search": "Rechercher des worktrees masqués", + "searchPlaceholder": "Rechercher parmi {{value0}} worktrees…", + "noMatches": "Aucun worktree correspondant", + "allShown": "Tous les worktrees détectés sont affichés", + "noneFound": "Aucun worktree non Orca trouvé", + "tryDifferentSearch": "Essayez un autre nom ou chemin.", + "disableSource": "Désactivez une source pour gérer ses worktrees individuellement.", + "appearWhenDetected": "Les nouveaux worktrees apparaîtront ici quand Orca les détectera.", + "openGlobalSettings": "Gérer dans les paramètres globaux", + "globalSettingsSources": "Ces sources ont un paramètre global que vous pouvez remplacer ici :" + }, + "add": { + "repo": { + "local": { + "start": { + "actions": { + "d72789705e": "Démarrer depuis un dossier vide", + "c709860596": "Créer un nouveau projet", + "5f9ffac036": "Cloner un dépôt Git distant", + "7edb8ebe24": "Cloner depuis une URL", + "a6c20dca96": "Ouvrir un projet depuis une cible SSH", + "sshCreateUnavailable": "Pas encore disponible pour les hôtes SSH", + "3d162cc76f": "Projet sur hôte SSH", + "fb4fc5380e": "Projet local, dépôt Git ou dossier contenant plusieurs dépôts", + "2281fdc8c7": "Parcourir le dossier", + "sshBrowseTitle": "Ouvrir un projet sur l'hôte SSH", + "sshBrowseDescription": "Dépôt Git existant ou dossier sur cet hôte SSH", + "runtimeBrowseDescription": "Dépôt Git existant ou dossier sur cette machine" + } + } + } + } + }, + "delete": { + "worktree": { + "flow": { + "b81b4e40ca": "Actualisez l'espace et réessayez si la liste des espaces de travail semble obsolète.", + "7243145cd6": "Aucun espace de travail supprimable sélectionné", + "ae57cbf6e4": "Échec de la suppression de l'espace de travail", + "7488ed8711": "Affichage", + "2b20ce87b3": "Forcer la suppression", + "4f3876c0f5": "Échec de la suppression forcée", + "workspaceListChanged": "Liste des espaces de travail modifiée" + }, + "toast": { + "1d0fa5c0a5": "Échec de la suppression de l'espace de travail {{value0}}", + "ead7b8ee15": "Il contient des fichiers modifiés. Utilisez Forcer la suppression pour le supprimer quand même.", + "905fc8efac": "Git a déjà supprimé cet espace de travail. Utilisez Forcer la suppression pour le nettoyer d'Orca.", + "0899ebdb28": "Git a déjà oublié cet espace de travail, mais son dossier est toujours sur le disque. Utilisez Forcer la suppression pour retirer le dossier orphelin.", + "locked": "Cet espace de travail est verrouillé par Git. Exécutez git worktree unlock depuis son dépôt, puis relancez la suppression.", + "lockedReason": "Cet espace de travail est verrouillé par Git. Git a signalé : {{value0}}. Exécutez git worktree unlock depuis son dépôt, puis relancez la suppression.", + "unstoppedPty": "Orca n'a pas pu confirmer que tous les terminaux de cet espace de travail sont fermés, il s'est donc arrêté avant de supprimer des fichiers. Utilisez Forcer la suppression pour le retirer quand même.", + "unstoppedPtyLive": "Cet espace de travail a encore des terminaux actifs, Orca s'est donc arrêté avant de supprimer des fichiers. Forcer la suppression les arrêtera et abandonnera tout travail non commité qu'ils contiennent." + } + } + }, + "remote": { + "file": { + "browser": { + "helpers": { + "4dbd72a7d7": "{{value0}} n'est pas un dossier dans {{value1}}", + "be266af66c": "{{value0}} correspond à plusieurs dossiers dans {{value1}}" + } + } + } + }, + "repo": { + "header": { + "create": { + "state": { + "992cfbc44b": "Créer un nouveau worktree pour {{value0}}", + "3a70acd808": "Reconnectez la cible SSH avant de créer des espaces de travail pour {{value0}}", + "6d022563a8": "Reconnectez la cible SSH avant de créer des espaces de travail", + "62e71f2d5d": "Créer un espace de travail pour {{value0}}" + } + } + } + }, + "sidebar": { + "project": { + "drop": { + "669e12dd97": "Dossiers locaux et dépôts Git", + "ffc769ca29": "Déposez un dossier pour ajouter un projet", + "740e8d0d46": "Utilisez « Ajouter un projet » pour les chemins d'hôte", + "e344666fb8": "Runtime serveur actif", + "d0f8943f8b": "Préparation du flux d'ajout de projet", + "18d3cf40e9": "Vérification du dossier" + } + } + }, + "sleep": { + "worktree": { + "flow": { + "c460fecc4a": "Échec de mise en veille de certains espaces de travail", + "8bc3fc0671": "Échec de mise en veille de l'espace de travail", + "legacy": { + "unverified": "L'ancien runtime d'hôte n'a pas pu confirmer l'arrêt des terminaux. L'espace de travail est resté ouvert ; mettez à jour l'hôte et réessayez." + }, + "host": { + "unverified": "L'hôte n'a pas pu confirmer l'arrêt des terminaux. L'espace de travail est resté ouvert ; vérifiez la connexion et réessayez." + }, + "retry": "L'espace de travail est resté ouvert. Réessayez ; si le problème persiste, vérifiez la connexion de l'hôte." + } + } + }, + "useAddRepoCloneFlow": { + "4d0013cc93": "Dépôt cloné", + "0dc4d1b657": "Saisissez un chemin d'hôte pour la destination du clone." + }, + "useAddRepoLocalFolderFlow": { + "7ab10e4974": "Utilisez un chemin d'hôte pour ajouter des projets depuis un hôte distant.", + "skippedBatchFolders": "Certains dossiers ont été ignorés", + "skippedBatchFoldersDescription": "Ajoutez les dossiers ignorés un par un pour les examiner ou les confirmer." + }, + "useAddRepoNestedImportFlow": { + "680cac2c82": "{{value0}} a échoué", + "cbfbc7a797": "Certains dépôts n'ont pas pu être importés", + "1b33c5f090": "Aucun dépôt importé" + }, + "useSidebarProjectDrop": { + "f34a286c0d": "Impossible d'ajouter le dossier déposé.", + "451a4638db": "Déposez un dossier pour l'ajouter comme projet.", + "5ccb56c7be": "Utilisez « Ajouter un projet » pour saisir un chemin d'hôte.", + "849ef13dc0": "Le dépôt de dossiers locaux n'est pas disponible pour les runtimes serveur.", + "c0315153d1": "Déposez un seul dossier à la fois." + }, + "workspace": { + "status": { + "cb387159f6": "En cours", + "6c1efa2cf8": "En revue", + "6b8285b8dd": "Terminé", + "93ac840dcb": "Bloquée", + "2c19d1db33": "Lecture", + "111db162bf": "En pause", + "642da473f2": "Alerte", + "6380517b10": "Drapeau", + "251c817bdd": "Minuteur", + "409528031f": "À examiner", + "5f9ca31a84": "En attente", + "821d156f54": "Tirets", + "226d1e7773": "Progression", + "a702bc08d4": "Point", + "b4a7101fe1": "Cercle", + "1a9383112b": "Progression Conductor", + "caebe3c10f": "Revue Conductor", + "895f381714": "Conductor terminé", + "caabd5ca85": "Zinc", + "7adb43ecf0": "Rose", + "ddf25b6262": "Émeraude", + "7cebab6d4a": "Ambre", + "1b81da243a": "Violet", + "6437a8c253": "Azur", + "fc3b92756c": "Bleu", + "52e3c6e2a4": "Neutre" + } + }, + "worktree": { + "card": { + "compact": { + "agents": { + "a128d7006b": "{{value0}} {{value1}} enfant {{value2}}", + "289a1d2ca7": "Développer {{value0}}. {{value1}}", + "0c1debfe84": "Réduire {{value0}}" + } + } + }, + "list": { + "groups": { + "0ed04075b8": "Tous", + "4aeefc5996": "Épinglés", + "682ed5d551": "Fermé", + "7c2f009786": "En cours", + "6798dc7c94": "En revue", + "5076efc3d2": "Terminé" + } + } + }, + "CacheTimer": { + "07729cc155": "expiré" + }, + "DeleteWorktreeDialogFooter": { + "c0e972d726": "Annuler", + "cf95e3b5bb": "Fermer" + }, + "index": { + "b826a98b6f": "occupé" + }, + "HostRemoveDialog": { + "1a2b3c4d5e": "{{value0}} retiré", + "2b3c4d5e6f": "Échec du retrait de l'hôte", + "3c4d5e6f7a": "Retirer {{value0}} ?", + "4d5e6f7a8b": "Ceci ouvre les paramètres des serveurs d'Orca où vous pouvez retirer ce serveur.", + "5e6f7a8b9c": "Ceci retire l'hôte SSH enregistré et ses identifiants de cet ordinateur. Les fichiers distants ne sont pas supprimés.", + "6f7a8b9c0d": "Annuler", + "7a8b9c0d1e": "Ouvrir les paramètres", + "8b9c0d1e2f": "Retirer l'hôte", + "workspacesFailed": "Impossible de retirer {{count}} espaces de travail de cet hôte. L'hôte a été conservé pour vous permettre de réessayer.", + "oneWorkspace": "1 espace de travail", + "manyWorkspaces": "{{count}} espaces de travail", + "hostHasWorkspacesDefault": "Retire {{value0}} et ses identifiants de cet ordinateur. Ses {{value1}} restent dans Orca — les fichiers distants ne sont pas touchés.", + "alsoDeleteRemote": "Supprimer aussi ces {{value0}} sur {{value1}}", + "alsoForgetLocal": "Retirer aussi ces {{value0}} d'Orca", + "advanced": "Avancé", + "alsoDeleteRemoteHint": "Supprime définitivement les worktrees Git distants et leurs branches. Irréversible.", + "alsoForgetLocalHint": "Les retire uniquement d'Orca. Les fichiers, worktrees et branches distants restent intacts." + }, + "HostRenameDialog": { + "1a2b3c4d5e": "Renommer l'hôte", + "2b3c4d5e6f": "Ce libellé n'est affiché que sur cet ordinateur. Laissez vide pour utiliser le nom par défaut.", + "3c4d5e6f7a": "Nom d'affichage", + "4d5e6f7a8b": "Rétablir la valeur par défaut", + "5e6f7a8b9c": "Annuler", + "6f7a8b9c0d": "Enregistrer" + }, + "HostSectionHeaderMenu": { + "5b8b4b6a01": "Mise à jour du serveur requise", + "9b3c1d2e44": "Mise à jour du client requise", + "2c29e2de68": "Échec de la connexion", + "bf07aee59e": "Échec de la déconnexion", + "7f1a2b3c4d": "{{value0}} est joignable", + "4f2c8a9b10": "Actions d'hôte pour {{value0}}", + "6b7c8d9e10": "Actions d'hôte", + "8d1e2f3a4b": "Renommer…", + "63f36455cc": "Reconnecter", + "59b553e2aa": "Déconnecter", + "2d3e4f5a6b": "Vérifier la connexion", + "3c4d5e6f7a": "Gérer l'hôte…", + "6e7f8a9b0c": "Retirer l'hôte…" + }, + "LinearAgentSkillSetupPrompt": { + "missingCliAndSkill": "La CLI Orca et le skill d'agent Linear sont manquants.", + "modalTitle": "Activer l'accès aux tickets Linear", + "modalDescription": "Installez le skill Linear depuis un terminal.", + "modalPrompt": "Permet aux agents de lire et de modifier le ticket Linear attaché.", + "dontShowAgain": "Ne plus afficher", + "missingBoth": "La CLI Orca et le skill d'agent Linear sont manquants.", + "missingCli": "La CLI Orca est manquante.", + "missingSkill": "Le skill d'agent Linear est manquant.", + "title": "Configurer le skill d'agent Linear", + "remoteCopy": "Ceci installe la configuration côté hôte ; les environnements d'agents distants peuvent nécessiter une configuration séparée.", + "hostCopy": "Installez-le pour les passations d'agents hôtes depuis le travail Linear lié.", + "dismiss": "Ignorer la configuration du skill d'agent Linear", + "setup": "Configurer", + "recheck": "Revérifier", + "panelTitle": "Skill d'agent Linear", + "panelDescription": "Installez le skill d'agent hôte pour les passations de tâches Linear liées.", + "terminalTitle": "Installer le skill d'agent Linear", + "terminalAria": "Terminal d'installation du skill d'agent Linear", + "install": "Installer la CLI et le skill", + "successTitle": "L'accès aux tickets Linear est prêt", + "successDescription": "Les agents peuvent désormais lire et mettre à jour les tickets Linear liés depuis cet espace de travail.", + "successDescriptionWsl": "Les agents WSL peuvent désormais utiliser les tickets Linear liés depuis cet espace de travail.", + "successDescriptionRemote": "Les agents de l'hôte peuvent désormais utiliser les tickets Linear liés. Les environnements d'agents distants peuvent encore nécessiter leur propre configuration.", + "successStatus": "Accès aux tickets Linear prêt", + "done": "Terminé", + "wslCopy": "Installez-le pour les passations d'agents WSL depuis le travail Linear lié.", + "wslLabel": "WSL par défaut", + "toastMissingCliAndSkill": "La CLI Orca et le skill Linear sont manquants", + "toastMissingCli": "La CLI Orca est manquante", + "toastMissingSkill": "Le skill Linear est manquant", + "toastInstallCliAndSkillDescription": "Installez la CLI Orca et le skill Linear pour permettre à vos agents de lire et de modifier les tâches Linear.", + "toastInstallCliDescription": "Installez la CLI Orca pour permettre à vos agents de lire et de modifier les tâches Linear.", + "toastInstallSkillDescription": "Installez le skill Linear pour permettre à vos agents de lire et de modifier les tâches Linear via la CLI Orca.", + "toastRemoteDescription": "{{value0}} Les environnements d'agents distants peuvent nécessiter leur propre configuration.", + "toastWslDescription": "{{value0}} Cette configuration s'exécute dans le runtime d'agent WSL sélectionné." + }, + "FolderWorkspaceComposerDialog": { + "connectFailed": "Échec de la connexion au projet.", + "noRepos": "Ajoutez un projet Git sous ce dossier pour attacher des tâches GitHub ou GitLab.", + "title": "Créer un espace de travail dossier", + "create": "Créer un espace de travail", + "sourceProject": "Source des tâches", + "chooseSourceProject": "Choisir la source des tâches", + "createFailed": "Échec de la création de l'espace de travail dossier." + }, + "ProjectOrderManualDefaultNotice": { + "a1f4c2d8e0": "L'ordre manuel des projets est désormais le réglage par défaut", + "822ff300ad": "Ignorer", + "b7e3a91c4f": "Faites glisser les en-têtes de projets pour les réordonner, ou passez à", + "e8c1f4a2b9": "dans les options de l'espace de travail." + }, + "WorkspaceKanbanDrawer": { + "1975a4e480": "Échec de la synchronisation du statut des tâches", + "e02b0d92ff": "Synchronisation du statut des tâches ignorée", + "c1d2e3f4a5": "L'issue Linear {{value0}} n'a pas pu être lue.", + "d2e3f4a5b6": "Aucun état de workflow Linear ne correspond à {{value0}}.", + "e3f4a5b6c7": "Plusieurs états de workflow Linear correspondent à {{value0}}.", + "f4a5b6c7d8": "Impossible de mettre à jour l'issue Linear {{value0}}.", + "a5b6c7d8e9": "Impossible de synchroniser l'issue Linear {{value0}}.", + "b6c7d8e9f0": "La synchronisation du statut des tâches n'a pas pu aboutir.", + "c7d8e9f0a1": "{{value0}} mis à jour", + "d8e9f0a1b2": "{{value0}} ignoré", + "e9f0a1b2c3": "{{value0}} a échoué" + }, + "WorktreeParentPickerPopover": { + "failedSetParent": "Échec de la définition du worktree parent", + "current": "Actuel", + "searchPlaceholder": "Rechercher des worktrees...", + "empty": "Aucun worktree éligible correspondant.", + "setParentFor": "Définir le parent pour" + }, + "WorktreeCardStatusSlot": { + "branchIdentity": "Branche" + }, + "NewExternalWorktreesInboxLine": { + "9f2d4c8b17": "Masquer définitivement les worktrees externes pour {{value0}}", + "5e1b8d3f62": "Masquer définitivement les worktrees externes", + "2a6f31d8c7": "worktree masqué", + "5b90e4a2f6": "worktrees masqués", + "7f18c5b0d3": "Examiner {{value0}} worktree masqué dans {{value1}}", + "4e2b7a9c05": "Examiner {{value0}} worktrees masqués dans {{value1}}", + "c3e8a1f4b2": "Ne plus afficher", + "6c07f3a91e": "{{value0}} sur {{value1}}" + }, + "NoticeHostGlyph": { + "hostDisconnected": "{{hostName}} déconnecté", + "sshHostProject": "Projet sur l'hôte SSH {{hostName}}", + "localHostProject": "Projet sur cette machine", + "runtimeHostProject": "Projet sur {{hostName}}" + }, + "newExternalWorktreesInboxActions": { + "a11c2f6d89": "Impossible de garder les worktrees externes masqués. Réessayez.", + "b7e4d1a062": "Impossible d'importer les worktrees externes. Réessayez.", + "c94f0b3a15": "Impossible de masquer définitivement les worktrees externes. Réessayez." + }, + "SuppressExternalWorktreeInboxDialog": { + "a4c2d8f1b0": "Masquer les worktrees externes ?", + "6e91b3c4d2": "Les worktrees externes ne seront plus affichés dans la barre latérale ni dans cette liste pour {{value0}}, y compris ceux créés ultérieurement.", + "1f8a5d9e73": "Vous pourrez réactiver cela plus tard depuis les paramètres du projet.", + "8c0b2e7a41": "Ouvrir les paramètres des worktrees non Orca", + "5d1c9f0a82": "Annuler", + "3b7e4a1c96": "Masquer les worktrees externes" + }, + "useAddRepoHostSelection": { + "connectionFailed": "Échec de la connexion SSH." + }, + "AddRemoteHostDialog": { + "sshHostRequired": "Un hôte ou un alias de config SSH est requis.", + "sshPortInvalid": "Le port doit être compris entre 1 et 65535.", + "sshRelayGraceInvalid": "Le délai d'expiration du terminal doit être compris entre 60 et {{value0}} secondes.", + "sshSaved": "Hôte SSH ajouté.", + "sshSaveFailed": "Échec de l'ajout de l'hôte SSH.", + "serverFieldsRequired": "Le nom du serveur et le code d'appairage sont requis.", + "serverSaved": "Serveur distant ajouté.", + "serverSaveFailed": "Échec de l'ajout du serveur distant.", + "serverTitle": "Ajouter un serveur distant", + "sshTitle": "Ajouter un hôte SSH", + "serverDescription": "Se jumeler avec Orca exécuté sur un autre ordinateur.", + "sshDescription": "Ajoutez une machine permanente sur laquelle vous pouvez vous connecter en SSH.", + "cancel": "Annuler", + "saving": "Enregistrement...", + "save": "Enregistrer", + "label": "Label", + "sshLabelPlaceholder": "Machine de dev", + "sshHost": "Hôte ou alias", + "sshHostPlaceholder": "deploy@server:22", + "username": "Nom d'utilisateur", + "usernamePlaceholder": "deploy", + "port": "Port", + "identityFile": "Fichier d'identité", + "identityFilePlaceholder": "~/.ssh/id_ed25519 (facultatif)", + "sshPersistenceDefault": "Les terminaux distants de cet hôte restent actifs jusqu'à ce que vous les fermiez ou réinitialisiez le relais.", + "serverName": "Nom du serveur", + "serverNamePlaceholder": "Machine de dev", + "pairingCode": "Code d'appairage", + "pairingCodePlaceholder": "orca://pair?code=...", + "pairingHelpPrefix": "Exécution", + "pairingCommand": "orca serve --pairing-address ", + "pairingHelpSuffix": "sur le serveur et collez l'URL d'appairage affichée.", + "sshImportAlreadySynced": "~/.ssh/config est déjà à jour.", + "sshImportSynced": "Ajout de {{value0}} hôte{{value1}} à Orca.", + "sshImportFailed": "Échec de l'import de la config SSH.", + "sshAlreadyExists": "Cet hôte SSH est déjà dans Orca.", + "advanced": "Avancé", + "linkDestination": "Destination du lien", + "sshTunnel": "J'utilise un tunnel SSH", + "sshTunnelHelp": "Sinon, ce lien pointe vers cet appareil et ne peut pas identifier l'autre ordinateur.", + "loopbackBlocked": "Activez l'option de tunnel SSH ou créez un nouveau lien utilisant l'adresse Tailscale ou LAN de l'autre hôte.", + "sshConfigPickerLoadFailed": "Échec de la lecture de ~/.ssh/config.", + "sshConfigPickerResolveFailed": "Échec de la résolution de cet hôte de la config SSH.", + "sshConfigPickerRestartRequired": "Redémarrez Orca pour finaliser l'application de la mise à jour du sélecteur de config SSH.", + "sshConfigPickerFilled": "Rempli depuis {{value0}}. Vérifiez puis enregistrez.", + "sshConfigPickerTitle": "Choisir depuis ~/.ssh/config", + "sshConfigPickerDescription": "Choisissez un hôte pour remplir le formulaire. Rien n'est enregistré tant que vous n'avez pas cliqué sur Enregistrer.", + "sshConfigPickerFilter": "Filtrer les hôtes…", + "sshConfigPickerHostsLabel": "Hôtes de la config SSH", + "sshConfigPickerLoading": "Lecture de ~/.ssh/config…", + "sshConfigPickerEmpty": "Aucun hôte dans ~/.ssh/config", + "sshConfigPickerNoMatch": "Aucun hôte correspondant", + "sshConfigPickerEmptyHint": "Ajoutez-y une entrée Host, ou revenez en arrière pour saisir les détails manuellement.", + "sshConfigPickerNoMatchHint": "Essayez un autre filtre, ou revenez en arrière pour saisir manuellement.", + "sshConfigPickerMoreResults": "Affichage des {{value0}} premières correspondances. Affinez le filtre pour en trouver davantage.", + "sshConfigPickerInOrca": "Dans Orca", + "sshConfigPickerPreviouslyRemoved": "Retiré d'Orca", + "sshConfigPickerBulkHint": "La synchronisation groupée reste dans Paramètres → SSH", + "sshConfigPickerBack": "Retour", + "fillFromSshConfig": "Remplir depuis ~/.ssh/config…", + "sshConfigPickerAddingAll": "Ajout des hôtes…", + "sshConfigPickerAddAll": "Ajouter tous les {{value0}} à Orca", + "sshConfigPickerNoNewHosts": "Aucun nouvel hôte à ajouter", + "sshConfigPickerAddAllEmpty": "Tout ajouter à Orca", + "identityFileFromConfigHint": "Volontairement laissé vide : Orca utilise toutes les clés résolues par ~/.ssh/config pour {{value0}}. Saisissez un chemin pour n'utiliser que cette clé.", + "sshConfigPickerRetry": "Réessayer", + "sshConfigPickerResolving": "Lecture…" + }, + "ForgetSshWorkspaceDialog": { + "reconnectFailed": "Échec de la reconnexion", + "forgetBody": "Retire uniquement cet espace de travail d'Orca. Les fichiers, le worktree Git et les branches sur {{host}} restent intacts.", + "title": "Supprimer « {{name}} » ?", + "disconnectedBody": "L'hôte SSH de cet espace de travail n'est pas connecté. Reconnectez-le pour le supprimer aussi à distance, ou retirez-le seulement d'Orca.", + "ghostBody": "{{host}} n'est plus un hôte SSH enregistré ; cet espace de travail n'est donc plus connecté à un hôte actif. Il ne peut être que retiré d'Orca — les fichiers et branches distants restent intacts.", + "cancel": "Annuler", + "forget": "Retirer d'Orca", + "reconnectAndDelete": "Reconnecter et supprimer" + }, + "MarkdownImageLightbox": { + "image": "Image", + "expand": "Développer l'image", + "close": "Fermer" + }, + "WorktreeDeveloperMenu": { + "developer": "Développeur", + "parkTerminal": "Parquer le terminal" + }, + "SidebarFeedbackImageAttachments": { + "screenshotsHint": "Joignez jusqu'à {count} captures d'écran", + "attachImages": "Joindre", + "removeImage": "Supprimer {{fileName}}" + }, + "WorkspaceKanbanSearchField": { + "bdb753c78d": "Aucun espace de travail correspondant", + "4d96c209d6": "{{value0}} espaces de travail sur {{value1}} correspondent", + "c0cd6bdf6c": "Rechercher des espaces de travail", + "3b7ea51793": "Effacer la recherche", + "7f1c2e94a5": "Texte de recherche trop long — le tableau n'est pas filtré", + "9a4d0f6b21": "Trop long" + }, + "WorktreeIssueLinkField": { + "25852bfc59": "Linear", + "5b440069e6": "GitHub", + "161b2d053a": "Ouvrir l'issue liée", + "d4785f9954": "Les liens d'issue sont définis à la création d'un espace de travail dossier et ne peuvent pas encore être modifiés ici.", + "964d9bc00a": "Ni une clé d'issue Linear ni une URL d'issue linear.app.", + "0a7a2c6efd": "Ni un numéro d'issue GitHub ni une URL d'issue.", + "72486800ff": "Enregistrer dissociera {{first}} et {{second}} — un espace de travail ne suit qu'une seule issue.", + "2c245ac134": "Enregistrer dissociera {{link}} — un espace de travail ne suit qu'une seule issue.", + "d8c8a30d1f": "Impossible d'ouvrir cette issue. Vérifiez l'identifiant et votre connexion Linear.", + "269198eeda": "Impossible d'ouvrir cette issue. Vérifiez le numéro et votre connexion GitHub.", + "f047887705": "Collez une URL GitHub ou Linear, ou saisissez un numéro. Laissez vide pour supprimer le lien.", + "ad78f9bee2": "Issue", + "662ae142f8": "N° d'issue, ou URL GitHub ou Linear", + "929c98d05a": "Fournisseur d'issues" + }, + "worktreeIssueDisplacement": { + "3f61c0a8d2": "Linear {{value}}", + "9c4b7e1f60": "GitHub #{{value}}" + }, + "WorktreeCardSshHostControl": { + "connectFailed": "Échec de la connexion SSH", + "removedTooltip": "Hôte SSH retiré — reconnexion impossible", + "removedName": "L'hôte SSH {{value0}} a été retiré", + "connectedTooltip": "Projet sur hôte SSH", + "connectedName": "Projet sur l'hôte SSH {{value0}}", + "connectingName": "Connexion à l'hôte SSH {{value0}}", + "authFailedName": "Reconnexion à l'hôte SSH {{value0}} — échec d'authentification", + "retryName": "Relancer la connexion SSH vers {{value0}}", + "connectName": "Se connecter à l'hôte SSH {{value0}}", + "authFailedTooltip": "{{value0}} · échec d'authentification", + "failedTooltip": "{{value0}} · échec de connexion" + }, + "preserved": { + "branch": { + "batch": { + "toast": { + "a3cdd9d9e6": "Git a conservé {{count}} branches locales car elles peuvent contenir des commits non fusionnés. Les branches conservées ne gardent pas les dossiers des espaces de travail ; leurs commits restent dans le dépôt. Orca peut continuer à libérer de l'espace disque en arrière-plan.", + "a3cdd9d9e6_one": "Git a conservé {{count}} branche locale car elle peut contenir des commits non fusionnés. Les branches conservées ne gardent pas les dossiers des espaces de travail ; leurs commits restent dans le dépôt. Orca peut continuer à libérer de l'espace disque en arrière-plan.", + "a3cdd9d9e6_other": "Git a conservé {{count}} branches locales car elles peuvent contenir des commits non fusionnés. Les branches conservées ne gardent pas les dossiers des espaces de travail ; leurs commits restent dans le dépôt. Orca peut continuer à libérer de l'espace disque en arrière-plan.", + "6310412304": "Examiner {{count}} branches", + "6310412304_one": "Examiner {{count}} branche", + "6310412304_other": "Examiner {{count}} branches", + "cea24c2b7d": "{{count}} espaces de travail supprimés", + "cea24c2b7d_one": "{{count}} espace de travail supprimé", + "cea24c2b7d_other": "{{count}} espaces de travail supprimés", + "0e0379f24a": "{{count}} branches conservées", + "0e0379f24a_one": "{{count}} branche conservée", + "0e0379f24a_other": "{{count}} branches conservées", + "4cf75caab7": "{{value0}}, {{value1}}", + "e61d78054f": "Suppression des branches locales : {{value0}}", + "1e1a5f6763": "Branches locales supprimées : {{value0}}", + "43d9395605": "{{value0}} supprimées, {{value1}} non supprimées", + "d42f1f14e0": "Réessayer {{count}} branches", + "d42f1f14e0_one": "Réessayer {{count}} branche", + "d42f1f14e0_other": "Réessayer {{count}} branches" + } + } + } + }, + "PreservedBranchBatchReviewDialog": { + "c4bf8e7eaf": "Examiner les branches conservées", + "f21976c9a8": "Sélectionnez les branches locales à supprimer de force. Les branches non sélectionnées restent dans leurs dépôts.", + "38c947f7c5": "Tout sélectionner", + "9602129d38": "{{value0}} sur {{value1}} sélectionnées", + "ee39e872d5": "Possiblement non fusionnée", + "676db406fd": "Head indisponible", + "285e1e4882": "Annuler", + "a0f9863597": "Forcer la suppression de {{count}} branches", + "a0f9863597_one": "Forcer la suppression de {{count}} branche", + "a0f9863597_other": "Forcer la suppression de {{count}} branches" + }, + "WorktreeListScrollToTopButton": { + "jumpToTop": "Remonter en haut" + }, + "WorktreeVisibilitySourceList": { + "claude": "Claude Code", + "gsd": "GSD", + "other": "Autres emplacements", + "custom": "Emplacement personnalisé", + "otherPath": "Hors des sources listées", + "invalidPath": "Saisissez un chemin absolu pour cet hôte.", + "duplicatePath": "Cet emplacement est déjà listé.", + "limit": "Retirez un emplacement personnalisé avant d'en ajouter un autre.", + "sources": "Sources", + "sourcesDescription": "Les sources affichées incluent les worktrees actuels et futurs dans la barre latérale.", + "found": "{{value0}} trouvés", + "remove": "Supprimer {{value0}}", + "removeLocation": "Retirer l'emplacement personnalisé", + "addLocation": "Ajouter un emplacement", + "worktreeRoot": "Racine des worktrees", + "add": "Ajouter", + "rootHelp": "Orca reconnaîtra les worktrees sous ce dossier.", + "visibility": "Visibilité pour {{value0}}", + "show": "Afficher", + "hide": "Masquer", + "overridingGlobal": "Remplace le paramètre global : {{value0}}", + "projectOnly": "Ajouté dans ce projet uniquement.", + "useGlobalFor": "Utiliser la valeur globale pour {{value0}}", + "useGlobal": "Utiliser la valeur globale" + } + }, + "shared": { + "useDaemonActions": { + "01af244097": "Annuler", + "28c8e53176": "Force l'arrêt de tous les volets de terminal en cours d'exécution dans tous les espaces de travail. Tout travail non enregistré de ces sessions est perdu. Le daemon continue de tourner et de nouveaux terminaux peuvent être ouverts immédiatement. Irréversible.", + "1bbea41a77": "Forcer l'arrêt de toutes les sessions de terminal ?", + "01d6b7c64e": "Tue tous les volets de terminal en cours d'exécution et redémarre le processus daemon. Les volets affichent \"Process exited\" et peuvent être rouverts immédiatement. Les sessions au protocole hérité d'une version précédente de l'application sont préservées. Irréversible.", + "922548bc66": "Redémarrer le daemon de terminal ?", + "2b4efdc162": "Impossible de tuer les sessions.", + "d18f3005c2": "Fermeture refusée pour {{value0}} session{{value1}}.", + "baad8cd651": "Aucune session en cours.", + "fe2ab66d45": "{{value0}} sessions sur {{value1}} tuées. Fermeture refusée pour {{value2}}.", + "d762b41f41": "Échec du redémarrage.", + "b5954e12d3": "Échec du redémarrage — consultez les logs.", + "0e9da1b98e": "Daemon redémarré.", + "d6372cc797": "Arrêt forcé de {{value0}} session{{value1}}.", + "87412c2a68": "Arrêt forcé de {{value0}} session.", + "a2f040ac1c": "{{value0}} sessions terminées.", + "63520148e2": "La session {{value0}} a refusé de se terminer.", + "cc0a26cb14": "{{value0}} sessions ont refusé de se terminer.", + "71a8d342b0": "Onglets de terminal absents : {{value0}}/{{value1}}. Tentatives de fermeture échouées : {{value2}}. Demandes d'arrêt PTY exactes acceptées : {{value3}} ; échecs : {{value4}}.", + "2e57c1a940": "Le résultat de l'arrêt du daemon est non vérifié car sa requête de gestion a échoué.", + "993af6052c": "Le gestionnaire de daemons signale : arrêtés : {{value0}}/{{value1}} ; encore présents avant le nettoyage précis : {{value2}}.", + "1f0d8ac762": "Le nettoyage des terminaux s'est terminé avec des erreurs.", + "80b6ea14cf": "Le nettoyage des terminaux s'est terminé avec des avertissements.", + "47cd2a50e9": "Aucune session ni onglet de terminal signalé.", + "c34fb1098d": "Onglets de terminal fermés et arrêt demandé.", + "d9657ac204": "Arrêt de la session de terminal demandé.", + "e8f25bd903": "Impossible de terminer le nettoyage des terminaux.", + "a702d4196e": "Ceci ferme tous les onglets de terminal de tous les espaces de travail et demande l'arrêt des sessions de terminal en cours. Tout travail non enregistré dans un terminal est perdu. Le daemon continue de tourner et de nouveaux terminaux peuvent être ouverts immédiatement. Cette action est irréversible." + } + }, + "setup": { + "guide": { + "SetupGuideModal": { + "3598a3ca0c": "Terminez les workflows essentiels qui rendent Orca utile au travail parallèle avec des agents.", + "48a9e5ef2d": "Premiers pas", + "28cf59fcb4": "Cela masquera la checklist de la barre latérale", + "f3b5ffb2a6": "Masquer la checklist de la barre latérale" + }, + "SetupGuideProgressRing": { + "dac3a4724a": "{{value0}} étapes de configuration sur {{value1}} terminées" + } + } + }, + "settings": { + "keep": { + "local": { + "main": { + "up": { + "to": { + "date": { + "setting": { + "f8bda25f29": "Garder la branche main locale à jour" + } + } + } + } + } + } + }, + "AccountsPane": { + "c2d2751587": "Supprimer le compte", + "dbb9626ed1": "Annuler", + "854ebbcc45": "Orca supprimera l'authentification Claude gérée pour ce compte enregistré. S'il est actuellement actif, Orca revient à la connexion Claude par défaut du système.", + "63843e37e2": "Supprimer le compte Claude ?", + "380a7736cc": "La suppression de ce compte efface définitivement son home Codex géré, y compris tout l'historique de sessions Codex et les connexions MCP stockés dedans. Cette action est irréversible. Si le compte est actuellement actif, Orca revient à la connexion Codex par défaut du système.", + "0d47394635": "Supprimer le compte Codex ?", + "ae3b21eb6c": "opencode.ai/workspace/wrk_…/go", + "51c9104e13": "Trouvez-le dans l'URL après connexion à opencode.ai (par ex.", + "b398b834c9": "Effacer", + "316ca4e610": "Oublier le cookie", + "a122332371": "wrk_… (laisser vide pour une recherche automatique)", + "dbdb0b0bd8": "Remplacement de l'ID de espace de travail", + "d70a5287a4": "Remplacement facultatif de l'ID de espace de travail si la recherche automatique échoue.", + "02cb127710": "ID de espace de travail OpenCode Go", + "7ce0e1907c": "). Trouvez-le dans les DevTools de votre navigateur → Réseau → une requête opencode.ai → en-tête Cookie. L'authentification OpenCode Go est web et partagée entre les terminaux Windows et WSL.", + "8951c5309f": "auth=Fe26.2**…", + "338820326a": ") ou l'en-tête cookie complet (par ex.", + "922b51e02d": "Fe26.2**…", + "0023cc336e": "Collez soit la valeur brute du jeton (par ex.", + "a7e38affcd": "jeton Fe26.2**… ou en-tête auth=Fe26.2**…", + "67e3c33670": "Cookie de session OpenCode Go", + "b2b1aa936d": "Collez votre cookie de session opencode.ai pour récupérer les rate limits.", + "36223200ac": "Cookie de session OpenCode Go", + "ea631977b5": "Configurer les paramètres du fournisseur OpenCode Go.", + "4ac10b4d08": "OpenCode Go", + "c2aee76420": "Extrait les identifiants OAuth de votre installation locale de Gemini CLI pour vous authentifier auprès de Google pour {{value0}}. Utilise les identifiants émis pour l'app Gemini CLI, pas pour Orca. Peut cesser de fonctionner si Google met à jour la CLI. À utiliser à vos risques et périls.", + "96f3649526": "Utiliser les identifiants Gemini CLI (expérimental)", + "d676c41fc6": "Extrait les identifiants OAuth de votre installation locale de Gemini CLI pour vous authentifier auprès de Google. Utilise les identifiants émis pour l'app Gemini CLI, pas pour Orca. Peut cesser de fonctionner si Google met à jour la CLI. À utiliser à vos risques et périls.", + "0c7f915b01": "Utiliser les identifiants Gemini CLI", + "973741a871": "Configurer les paramètres du fournisseur Gemini.", + "0c64dc2a64": "Gemini", + "db209ee572": "Supprimer", + "8a0f870153": "Réauthentifier", + "3d245ef7d9": "Codex signale que cette connexion est périmée", + "589eba1eee": "Réauthentification requise", + "e74831fb6b": "Actif", + "b4c9450319": "Aucun compte Codex géré pour {{value0}}. Orca utilisera la connexion Codex par défaut du système de cet environnement jusqu'à ce que vous en ajoutiez un ici.", + "93c47b333a": "Connexion requise", + "f2a265f8c7": "Valeur par défaut du système", + "b0e948a4f9": "Ajouter un compte", + "c0a52abfc5": "Affiche les comptes pour {{value0}}. Les nouveaux comptes y sont ajoutés.", + "94d351af4a": "Comptes", + "d0d53b7eb0": "Gérer le compte Codex qu'Orca utilise pour récupérer les rate limits en direct.", + "3180536c7a": "Comptes Codex", + "340d6f7a85": "Chaque compte garde son propre contexte de connexion local dans Orca. L'authentification des comptes reste sur cet appareil.", + "cedfab35ab": "Facultatif. Orca peut utiliser votre connexion Codex habituelle ; ajoutez des comptes seulement si vous voulez changer rapidement de compte dans Orca.", + "ef91cfa06b": "Codex", + "3fe7862418": "Aucun compte Claude géré pour {{value0}}. Orca utilisera la connexion Claude par défaut du système de cet environnement jusqu'à ce que vous en ajoutiez un ici.", + "3455cf43fa": " Claude.", + "fcc4093fc1": "Utilisez votre connexion Codex {{value0}} actuelle.", + "79e484c3b2": "Sélecteur de comptes facultatif pour les fichiers d'authentification Claude partagés.", + "8bbfd74556": "Comptes Claude", + "72b36ea174": "Facultatif. Orca peut utiliser votre connexion Claude habituelle ; ajoutez des comptes seulement si vous voulez changer rapidement sans déplacer les sessions de chat.", + "26ef4b55be": "Claude", + "2743cdc0af": "Échec de la mise à jour du compte Claude.", + "b15ce90870": "{{value0}} -> {{value1}}. Redémarrez les terminaux Claude actifs avant de poursuivre les anciennes sessions.", + "f921d32606": "Compte Claude mis à jour.", + "5bf8764953": "Échec de la mise à jour du compte Codex.", + "9baf45d071": "Cet appareil", + "2358ac71d2": "WSL par défaut", + "ad47a33f72": "Chargement de WSL", + "8619f9afa9": "WSL", + "46cf7e7495": "Emplacement des comptes", + "0b4591ff93": "Choisissez l'environnement local à inspecter et l'emplacement où les nouveaux comptes Claude et Codex gérés sont ajoutés.", + "0c67a2a1aa": "WSL n'est pas disponible sur cette machine.", + "2cd197025c": "Choisissez si les comptes des fournisseurs sont inspectés et ajoutés dans {{value0}} ou WSL.", + "f54b4fbd71": "Emplacement des comptes", + "9107406589": "Impossible de charger les comptes Claude.", + "b8c2905c2b": "Impossible de charger les comptes Codex.", + "fd62f37c24": "Codex signale que cette connexion {{value0}} est périmée.", + "b10cb4f696": "ajout", + "e4a28e8894": "Codex signale que la connexion {{value0}} doit être refaite. Reconnectez-vous avant de démarrer de nouvelles sessions Codex.", + "75ca9b718e": "Codex signale que le compte actif doit être reconnecté. Réauthentifiez-le avant de démarrer de nouvelles sessions Codex.", + "b11078a9c2": "wsl", + "350b2a1aa7": "Utilisez votre connexion ", + "e05d0ff737": "Utilisez votre connexion Claude {{value0}} actuelle.", + "2f24f244a4": "Le cookie MiniMax est requis.", + "8e6f0cb1d8": "Le cookie MiniMax n'a pas été enregistré.", + "8d61637a77": "Cookie MiniMax enregistré.", + "b43e761fe5": "Échec de la mise à jour du cookie MiniMax.", + "5d63bbfbec": "MiniMax", + "15e831350e": "Configurer le suivi de consommation MiniMax depuis platform.minimax.io.", + "21d6eb141e": "Cookie de session MiniMax", + "33bba5ad83": "Collez votre cookie de session MiniMax pour récupérer les rate limits localement.", + "73ea15f24b": "Enregistré", + "23afe8f226": "Non enregistré", + "566d9a99ab": "_token=…; minimax_group_id_v2=…", + "f38b9cc4bd": "Remplacer", + "590a3130f9": "Enregistrer", + "79418c782a": "Ouvrez platform.minimax.io/console/usage dans votre navigateur, connectez-vous, puis copiez l'en-tête de requête Cookie depuis les DevTools (Réseau → une requête remains → Cookie).", + "9dd50d3f75": "Avancé", + "174fb408f9": "Ne touchez pas à ces valeurs par défaut, sauf si l'actualisation de consommation MiniMax vise le mauvais espace de travail ou modèle.", + "bf160bb6c0": "Remplacement du group ID", + "b1e2743313": "Facultatif. Laissez vide pour utiliser minimax_group_id_v2 du cookie.", + "0747d6391a": "Utiliser le group ID du cookie", + "4ff2af7524": "Noms de modèles pour la consommation", + "5cf4b0f85f": "Noms de modèles facultatifs, séparés par des virgules. Laissez general sauf si MiniMax renvoie une erreur propre à un modèle.", + "3c92b0d31c": "general", + "0d8e77bc40": "Ouvrir la console", + "0b8c1c7e02": "Stocké localement", + "1fd1b1b6b4": "Cookie non défini", + "5e08b0fe57": "Stocké localement et envoyé uniquement à platform.minimax.io pour les actualisations de consommation.", + "43d7a45b97": "Comment copier", + "b8a4f21c3e": "Collez l'en-tête Cookie depuis les DevTools", + "53f7b8c7a2": "Dernière actualisation : {{value0}}", + "31d24a4e87": "Le cookie expire quand vous vous déconnectez dans le navigateur.", + "3a30aaf526": "à l'instant", + "f5d8d2a6a1": "Ouvrez platform.minimax.io/console/usage dans votre navigateur et connectez-vous.", + "24560fe830": "Ouvrez les DevTools.", + "4cab0fa42d": "Allez dans l'onglet Network et activez Preserve log.", + "bee4e63e1c": "Rechargez la page.", + "87f814af6f": "Filtrez sur remains et sélectionnez la requête coding_plan/remains.", + "435df0ee51": "Sous Request Headers, copiez la valeur Cookie.", + "7492fb3bba": "Collez-la ici et cliquez sur Enregistrer.", + "9fec52de4b": "Comment copier le cookie", + "4e32e030b2": "Stocké localement. Orca ne l'envoie qu'à platform.minimax.io pour les actualisations de consommation.", + "remoteServerFallback": "le serveur distant", + "loadAccountsFailed": "Impossible de charger les comptes du fournisseur.", + "remoteScopeAccounts": "Affiche les comptes gérés par {{value0}}. Ajoutez ou réauthentifiez les comptes sur ce serveur.", + "accountScopePrefix": "Portée des comptes", + "accountScopeRemoteServerUnnamed": "Serveur distant", + "remoteScopeLocalAccountsKept": "Les comptes gérés sur ce bureau restent inchangés. Repassez le runtime par défaut sur Local desktop pour les voir.", + "remoteScopeAuthContext": "Chaque compte garde son propre contexte de connexion sur {{value0}}.", + "remoteEmptyClaudeAccounts": "Aucun compte Claude géré sur {{value0}}. Il utilise sa connexion Claude par défaut du système ; ajoutez des comptes sur ce serveur.", + "remoteEmptyCodexAccounts": "Aucun compte Codex géré sur {{value0}}. Il utilise sa connexion Codex par défaut du système ; ajoutez des comptes sur ce serveur.", + "codexSystemDefaultCustomProvider": "Fournisseur personnalisé — aucune consommation suivie.", + "codexSystemDefaultNeedsSignIn": "Aucune connexion Codex trouvée pour {{value0}}.", + "codexConfigSyncMissingSource": "Codex utilise toujours les derniers paramètres synchronisés car {{value0}} est absent. Restaurez ce fichier pour reprendre la synchronisation.", + "codexConfigSyncBlankSource": "Codex utilise toujours les derniers paramètres synchronisés car {{value0}} est vide. C'est attendu tant qu'un dossier synchronisé finit de se télécharger.", + "codexConfigSyncManagedHomeUnavailable": "Orca n'a pas réussi à lire les fichiers Codex de ce compte à l'instant ; les paramètres peuvent donc ne plus se synchroniser. Cela se résout généralement tout seul — un antivirus ou un outil de sauvegarde les verrouille brièvement.", + "codexConfigSyncUnreadableSource": "Codex utilise toujours les derniers paramètres synchronisés car {{value0}} n'a pas pu être lu. Vérifiez les permissions de ce fichier." + }, + "AdvancedPane": { + "40b29e0bf3": "Redémarrer", + "87a2cb2ac8": "Orca applique ce mode réseau au démarrage.", + "89958d7edf": "Redémarrage requis", + "b3ad629640": "À utiliser uniquement quand un VPN d'entreprise ou un proxy fait échouer les téléchargements de mises à jour avec des erreurs de protocole HTTP/2. Affecte tout le réseau d'Electron après redémarrage.", + "6627e75c92": "Expliquer la compatibilité HTTP/1.1", + "e9506d3377": "Compatibilité HTTP/1.1", + "8b7a8df299": "Contournements de bas niveau pour le diagnostic du support.", + "8d8d8ac599": "Compatibilité", + "network": "Réseau", + "networkDescription": "Routage réseau au niveau de l'app pour proxys et environnements d'entreprise." + }, + "AgentLocationSetting": { + "92f4238f1a": "WSL par défaut", + "fc806485ae": "Chargement de WSL", + "43663b5e69": "WSL", + "9bccf48906": "Emplacement des agents", + "d00949e59b": "Affiche les agents installés depuis {{value0}}. L'actualisation revérifie PATH dans cet environnement.", + "c7c516946f": "WSL n'est pas disponible sur cette machine.", + "f97b986b7f": "wsl" + }, + "AgentSkillSetupPanel": { + "0b810ec59f": "Appuyez sur Entrée pour lancer la commande.", + "ed197f59a2": "Copier la commande", + "817d3f9f18": "Copier la commande", + "5289300939": "Non installé", + "9fcebceb2a": "Installés", + "68a468752e": "Vérification...", + "c689392435": "Revérifier", + "a31e2aa302": "Échec de la copie de la commande.", + "378ad26865": "Commande copiée.", + "copiedCommand": "Commande copiée.", + "failedToCopyCommand": "Échec de la copie de la commande.", + "copyCommandAria": "Copier la commande", + "runCommandDescription": "Appuyez sur Entrée pour lancer la commande.", + "5f818f12ab": "Préparation...", + "4c05b9d7cb": "Préparation du terminal de configuration.", + "installLabel": "Installer", + "updateLabel": "Mettre à jour", + "setupCommandFailed": "La commande de configuration s'est terminée avec le code {{value0}}. Cette erreur disparaîtra après une nouvelle tentative réussie.", + "setupFailed": "Échec de la configuration", + "retrySetup": "Réessayer" + }, + "AgentsPane": { + "d83834f5e6": "Détection des agents installés…", + "024bd95089": "agents", + "e8da2af684": "Disponibles à l'installation", + "ed3e110e61": "détecté", + "03e1a5081a": "sur {{value0}}", + "25a41a9aad": "Redétecter les agents installés sur le serveur actif", + "remoteDetectionFailed": "Impossible de détecter les agents installés. Vérifiez la connexion à l'hôte puis réessayez.", + "retryDetection": "Réessayer", + "02e0143be5": "Installés", + "110b74b022": "Aucun agent (terminal vierge)", + "92033495ff": "Auto", + "9b175d0f5e": "Agent présélectionné à l'ouverture d'un nouveau espace de travail.", + "385212c7a1": "Agent par défaut", + "f9f127d664": "Remplacez le chemin ou le nom du binaire, et modifiez les arguments de lancement ou l'environnement par défaut de cet agent.", + "f95b5c79b8": "Installer", + "fe4d630c94": "Docs", + "8dc0192e48": "Désactivé", + "df123171d1": "Non installé", + "c8794e622e": "Détecté", + "5200dac9da": "Réinitialiser", + "2e45ca29b6": "Commande", + "d4d2a45d63": "Activé", + "1c9a9679ec": "Disponibilité de {{value0}}", + "0d9e293a02": "Actualiser", + "c9b33eb5c0": "Actualisation…", + "13647f9f80": "Relire le PATH de votre shell et redétecter les agents installés", + "dc4a2ffdc0": "Déplier le remplacement de commande", + "cea7d97be1": "Replier le remplacement de commande", + "db9e9e5887": "Personnaliser la commande", + "959b67385b": "Définir par défaut", + "24e032fa34": "Par défaut", + "5f986a9b92": "Définir par défaut", + "d7625cf8b2": "Agent par défaut", + "cfb3f35775": "Arguments", + "6f99bf5dd0": "Aucun argument par défaut", + "8fbe1f37c1": "Environnement", + "2d133152fa": "Aucun environnement par défaut", + "agentPermissions": "Permissions des agents", + "agentPermissionsInfo": "Infos permissions des agents", + "agentPermissionsTooltip": "Ne s'applique pas aux agents dont vous avez remplacé les arguments de lancement.", + "agentPermissionsDescription": "Choisissez si Orca lance les agents avec moins de demandes de permission ou avec des vérifications manuelles.", + "agentPermissionsYolo": "Yolo", + "agentPermissionsManual": "Manuel", + "3f1bdf3cb4": "Le texte d'environnement est trop volumineux pour être analysé sans risque.", + "codexSessionSource": "Home Codex d'où importer", + "codexSessionSourceInfo": "À propos de l'import de l'historique Codex", + "codexSessionSourceTooltip": "Orca exécute Codex dans un home isolé. Pointez ceci vers votre home Codex existant pour en importer l'historique de sessions. Si vide, ~/.codex est utilisé.", + "storedDefaultUndetected": "Enregistré par défaut, mais non détecté actuellement", + "noAgentsDetected": "Aucun agent détecté. Si un agent est installé, la détection a peut-être expiré." + }, + "AppIconSelector": { + "d5a112dc9b": "Icône suivante", + "415fa76f64": "Icône d'app sélectionnée", + "5f5142a62a": "Icône précédente" + }, + "AppearancePane": { + "3057983501": "Non assigné", + "872af9556e": "Zone de notification", + "2edf606c46": "Réduire dans la zone de notification à la fermeture", + "b707773a0d": "Quand activé, fermer la fenêtre laisse Orca tourner dans la zone de notification au lieu de quitter.", + "0cd9b8228f": "Choisissez l'icône d'app affichée dans le Dock et le sélecteur de fenêtres.", + "ca1590d42f": "Icône de l'app", + "61d842eca0": "Afficher le raccourci Orca Mobile dans la barre latérale. Il reste accessible depuis la Toolbox.", + "9da1020447": "Afficher le bouton Orca Mobile", + "5db6ba961f": "Afficher le bouton Orca Mobile en haut de la barre latérale gauche.", + "fa882a3e6b": "Afficher le bouton Automatisations en haut de la barre latérale gauche.", + "511f270ebb": "Afficher le bouton Automatisations", + "661942ab7f": "Afficher le bouton Tâches en haut de la barre latérale gauche.", + "cf81907069": "Afficher le bouton Tâches", + "dc29f3cc0d": "Barre latérale", + "ea943d0db0": "Choisissez les indicateurs affichés en bas de la fenêtre. Un clic droit sur la barre d'état donne accès aux mêmes options.", + "3e4175e5c6": "Barre d'état", + "2df8f79aa5": "Afficher Orca dans la barre de titre.", + "9868f39007": "Nom de l'app dans la barre de titre", + "4de76f6902": "Contrôler ce qui apparaît dans la barre de titre de l'application.", + "6a272ca553": "Barre de titre", + "e9f2ca5582": "Désactivez pour masquer les fichiers correspondant à .gitignore dans l'explorateur de fichiers.", + "0fafabcf35": "Afficher les fichiers ignorés par Git", + "75f07ab60c": "Afficher les fichiers correspondant à .gitignore dans l'explorateur de fichiers.", + "d496901cd0": "Explorateur de fichiers", + "42554f615f": "Choisissez la police utilisée par l'interface d'Orca.", + "102d6b5f9b": "Police de l'IDE", + "ef89200c1f": "en dehors d'un panneau de terminal.", + "f687711a9b": "Mettez toute l'interface de l'application à l'échelle. Utilisez", + "5e6d7aba8d": "Zoom de l'interface", + "622e1c3465": "Met toute l'interface de l'application à l'échelle.", + "fd89b5487c": "Clair", + "7d26ccabe8": "Sombre", + "fb0e0b4453": "Système", + "932ff1fbff": "Thème", + "0f28e7b30c": "Choisissez l'apparence d'Orca dans la fenêtre de l'app.", + "leftSidebarAppearance": { + "title": "Apparence de la barre latérale gauche", + "rowDescription": "Accordez la barre latérale gauche à votre terminal, gardez-la par défaut ou appliquez une teinte.", + "default": "Par défaut", + "matchTerminal": "Comme le terminal", + "tinted": "Teintée", + "tintColor": "Teinte de la barre latérale", + "tintColorDescription": "Couleur mélangée à la surface de la barre latérale gauche.", + "tintOpacity": "Intensité de la teinte", + "tintOpacityDescription": "Contrôle l'intensité avec laquelle la teinte est mélangée à la barre latérale." + }, + "workspaceCardLayoutGuidance": "Géré depuis la barre latérale des espaces de travail.", + "interfaceDefaultFont": "Police par défaut", + "terminalDefaultFont": "Police par défaut", + "interfaceTitle": "Interface", + "terminalTitle": "Terminal", + "windowSidebarTitle": "Fenêtre & barre latérale", + "windowSidebarSummary": "Barre latérale, barre d'état et explorateur de fichiers", + "statusBarCount": "{{value0}} indicateurs visibles.", + "gitIgnoredGlossary": "Fichiers correspondant à .gitignore.", + "statusBarDescription": "Choisissez les indicateurs affichés dans la barre d'état.", + "showPinnedWorktreesInGroups": { + "title": "Afficher aussi les worktrees épinglés dans leurs listes d'origine", + "description": "Les worktrees épinglés restent dans Épinglés et apparaissent aussi dans Tous, Projet, Statut et PR." + } + }, + "AutoRenameBranchFromWorkSetting": { + "1626524572": "Nautilus", + "0de9fda203": "Abandonner", + "c71770c455": "{basePrompt}", + "f19a56498d": "; votre paramètre de préfixe de branche s'applique toujours.", + "800edb1e54": "fix-login-flow", + "5d569f5199": ". Orca ne génère que le dernier segment, comme", + "570817d126": "et", + "56580dcf60": ". Vous pouvez aussi référencer", + "9c9b54e4ea": "l'invite intégrée de nom de branche d'Orca", + "69bf4830c2": "pour inclure", + "9241b59bf5": "Utiliser", + "a869d0edd8": "Modèle de commande de nom de branche", + "e784ea62dc": "Avancé", + "d9b65054ef": ") en un nom court résumant la tâche. Seules les branches qu'Orca a lui-même nommées sont renommées, et jamais après avoir été poussées.", + "12ea4a408d": "Quand un agent commence à travailler dans un nouveau espace de travail, Orca renomme sa branche générée automatiquement (par ex.", + "ef787db0e3": "Renommage auto de la branche", + "6a051586d2": "Renommer la branche générée automatiquement d'après le travail dès qu'un agent démarre.", + "ec3e0c388e": "Enregistrer", + "cfd82406dd": "Enregistrement...", + "40e7be7850": "Enregistré", + "7c7e34a66d": "Modifications non enregistrées", + "a4fa380b67": "{assistantMessage}", + "2ee2779c05": "{firstPrompt}" + }, + "AutoRenameBranchPromptEditor": { + "63121132c0": "Abandonner", + "4416b25d29": "Privilégiez les noms du domaine de la tâche, évitez les IDs de ticket et gardez des noms faciles à relire.", + "39278f4411": "; votre paramètre de préfixe de branche s'applique toujours.", + "ebb942a2ec": "fix-login-flow", + "af2d9a2cc6": ". Orca ne génère que le dernier segment, comme", + "182d419b97": "l'invite intégrée de nom de branche d'Orca", + "2f5dc661fe": "Ajouté à", + "7d6176f506": "Prompt", + "5968112152": "Enregistrer", + "54ac229ad4": "Enregistrement...", + "af0831a590": "Enregistré", + "0691753cf2": "Modifications non enregistrées" + }, + "BaseRefPicker": { + "1b8e54151f": "Aucune branche correspondante.", + "d166ff883d": "Actuel", + "a4a9372eb2": "Recherche des branches...", + "7db7fb87e5": "Rechercher des branches par nom...", + "773a5687a3": "Utiliser la principale", + "ade9a5bb03": ") pour restreindre les résultats.", + "b468f46726": "upstream/main", + "80f7c82303": ") ou un ref complet (par ex.", + "915ad97875": "upstream", + "a5c16712c1": "Plusieurs remotes détectés. Tapez un nom de remote (par ex.", + "9a14ec7400": "Choisissez une branche de base ci-dessous", + "086ce7f369": "Suit la branche principale ({{value0}})", + "2f3cda96f5": "Épinglé pour ce dépôt", + "ee110e1830": "Aucun ref de base par défaut" + }, + "BrowserDefaultZoomSetting": { + "2622126877": "Niveau de zoom appliqué aux nouveaux onglets du navigateur.", + "bbeec087d3": "Appliqué aux nouveaux onglets du navigateur.", + "265597101f": "Zoom par défaut" + }, + "BrowserHomePageSetting": { + "d4ddcd0056": "Enregistrer", + "37a30c5bfd": "https://google.com", + "c6cbd1c105": "Page d'accueil enregistrée.", + "6a37540f4b": "URL ouverte à la création d'un nouvel onglet du navigateur. Laissez vide pour un onglet vierge.", + "70224e37b1": "Page d'accueil par défaut" + }, + "BrowserPane": { + "81ff774667": "Annuler", + "7d4c0a2aa4": "Nom du profil", + "612f7f6861": "Échec de la création du profil.", + "8f22b7580d": "Profil « {{value0}} » créé.", + "8481ee0331": "Nouveau profil de navigateur", + "c0f85056d9": "Profils de navigateur sur ce serveur Orca.", + "86b7c83fee": "Cet ordinateur", + "6480776a03": "Profils de navigateur pour l'hôte sélectionné.", + "5e19a692f7": "Hôte", + "6f2584b39e": "Ajouter un profil", + "e4aaf8051b": "menu de la barre d'outils.", + "cd47bc9622": "Sélectionnez un profil par défaut pour les nouveaux onglets du navigateur. Importez des cookies et changez de profil par onglet via le ", + "2d66a6efb5": "Session & cookies", + "aa1074bfe9": "Gérez les profils de navigateur et importez les cookies depuis Chrome, Edge, Comet ou d'autres navigateurs.", + "113cd2dc9b": "Session & cookies", + "d3eb69c0aa": "Routage des liens", + "3e46903ad4": "Utilisé quand on saisit du texte autre qu'une URL dans la barre d'adresse.", + "0d9c987f21": "Moteur de recherche par défaut", + "7b225c78f5": "Moteur de recherche utilisé quand on saisit du texte autre qu'une URL dans la barre d'adresse.", + "64898ecdab": "Créer", + "7b649a578a": "Création…", + "4399c77caa": "Par défaut", + "4af9a17947": "kagi" + }, + "BrowserProfileRow": { + "8e636cae25": "Profil « {{value0}} » supprimé.", + "2d4bea7f35": "Cookies par défaut effacés.", + "ebb78dfd6f": "Depuis un fichier…", + "7df818977e": "Depuis", + "cdec84552f": "Importer des cookies", + "796d846483": "Aucun cookie importé", + "c29648fe5b": "Actif", + "d420c43729": "{{value0}} cookies importés depuis {{value1}}{{value2}} vers {{value3}}.", + "b4c167764d": "{{value0}} cookies importés depuis un fichier vers {{value1}}.", + "a3f8c2d1e0b4": "{{value0}} cookies importés depuis {{value1}} ({{value2}}) vers {{value3}}.", + "b4e9d3f2a1c5": "{{value0}} cookies importés depuis {{value1}} vers {{value2}}.", + "c5a273a809": "Depuis {{value0}}", + "b5c0479e21": "User agent non modifié" + }, + "BrowserUseComputerUseNotice": { + "15b5e680ba": "Ouvrir Computer Use", + "79209b37b9": "Si l'import de cookies ne convient pas, Computer Use peut contrôler des apps locales et, le cas échéant, réutiliser des sessions de navigateur déjà connectées. Installez la skill Computer Use ; macOS exige aussi des autorisations de confidentialité.", + "333984cf90": "Utiliser une session de navigateur existante" + }, + "BrowserUseEnableSwitch": { + "aea3f45349": "Activer Agent Browser Use" + }, + "BrowserUseExamples": { + "1199258ace": "Copier", + "1188e56af4": "Copier l'exemple de prompt", + "b84807f228": "\"", + "59722f31b4": "\"", + "c5325e91f6": "Collez l'un de ces exemples dans Claude Code, Codex ou un autre agent, dans un projet où la skill est installée.", + "2a180694f7": "Essayez — exemples de prompts", + "5ec620ccc4": "Échec de la copie.", + "a602d43069": "{{value0}} copié." + }, + "BrowserUsePane": { + "be6df68384": "Depuis un fichier…", + "e44c5d681e": "Depuis", + "67d9a53f47": "Gérer les profils pour des connexions distinctes", + "112f70adc4": "Dernier import depuis {{value0}}", + "72d4815523": "Importez vos connexions existantes dans Orca pour que les agents accèdent aux pages authentifiées. Importe dans le profil par défaut.", + "2eb906706c": "Importer les cookies du navigateur", + "af8c83ed61": "Importez les cookies depuis Chrome, Edge ou d'autres navigateurs pour que les agents réutilisent vos connexions.", + "68ea76eb71": "Installez la skill Browser Use pour que les agents pilotent le navigateur d'Orca.", + "2d6ead9ab2": "Installer la skill Browser Use", + "e9f3f3b488": "Installé dans", + "9fca1f7f5d": "Enregistre la commande CLI d'Orca pour que les agents orchestrent le navigateur depuis leur shell.", + "c6065d205d": "Activer Orca CLI", + "c79eff0213": "Enregistrer Orca CLI pour que les agents pilotent le navigateur.", + "702488a5f7": "Permettez aux agents de code de piloter ce navigateur avec vos connexions. Terminez les trois étapes ci-dessous.", + "b8a1f2d84d": "Agent Browser Use", + "96b91c6349": "Permettez aux agents de code de piloter ce navigateur avec vos connexions.", + "2ea4617e3a": "{{value0}} cookies importés depuis {{value1}}{{value2}}.", + "721aee31b4": "Orca CLI enregistré dans PATH.", + "180a9abf3a": "Échec du chargement de l'état de la CLI.", + "2ccfc9cff8": "Importer", + "0462565413": "Réimporter", + "de9b2f32f3": "Activer", + "ad8cb0ee22": "Corriger PATH", + "0289434ed6": "Activé", + "8b3054dac7": "Enregistrement...", + "8f2675c2f3": "{{value0}} cookies importés depuis un fichier.", + "5301857d88": "Depuis {{value0}}" + }, + "BrowserUseSkillStep": { + "0871b6998d": "Permet aux agents de naviguer et de vérifier des pages dans le navigateur d'Orca.", + "459e24eebc": "Skill Browser Use" + }, + "CliSection": { + "8671e406f0": "Annuler", + "a4aafe46e3": "Chemin cible :", + "e8012c03a1": "Permet aux agents d'utiliser les commandes espace de travail, terminal et progression d'Orca.", + "6053cf736c": "Skill CLI", + "36a6f919ba": "Donne aux agents des workflows espace de travail, terminal et progression propres à Orca.", + "04873eea3e": "Skills des agents", + "7f2747f7dd": "n'est pas actuellement visible dans PATH pour ce shell.", + "b0c310ab46": "Cible du lanceur existante :", + "15eaad0d31": "Chemin de la commande :", + "5dae812f50": "Actualiser", + "52e640f3a0": "Actualiser l'état de la CLI", + "38edbb5721": "Commande shell", + "6930feda9e": "Utilisez Orca depuis votre terminal pour ouvrir l'app, gérer les worktrees et interagir avec les terminaux Orca.", + "c5c0f2641d": "Orca CLI", + "d77352f2df": "Échec de la suppression de `{{value0}}` de PATH.", + "af5540930c": "`{{value0}}` supprimé de PATH.", + "a2b13efa94": "Échec de l'enregistrement de `{{value0}}` dans PATH.", + "9cbcd31338": "`{{value0}}` enregistré dans PATH.", + "7baec27029": "Échec du chargement de l'état de la CLI.", + "d00df2e397": "Enregistrer", + "9a5f8a4568": "Supprimer", + "b0fca411a0": "Enregistrement…", + "4c7e3e4c5f": "installer", + "068552b191": "Suppression…", + "8d96213669": "supprimer", + "aa6536977e": "Orca va enregistrer {{value0}} pour que la commande fonctionne depuis votre terminal.", + "a030816e3e": "Ceci supprime le lien symbolique de la commande shell. Orca reste installé.", + "fa87db3d6e": "Enregistrer `{{value0}}` dans PATH ?", + "14444243ba": "Supprimer `{{value0}}` de PATH ?", + "5d432fe44d": "installé", + "8a9b784c60": "périmé", + "d363e5929b": "Vérification de l'enregistrement de la CLI…", + "cliSkillTerminalTitle": "Configuration de la skill CLI", + "cliSkillTerminalAria": "Terminal d'installation de la skill CLI" + }, + "CliSkillRuntimeSetup": { + "04325573f8": "WSL", + "a58ba464ad": "Emplacement de la skill", + "0ed08febc5": "Échec de l'enregistrement de la commande shell WSL.", + "3728a94fb6": "La commande shell WSL demande votre attention", + "775a4cfbb8": "L'enregistrement de la commande shell WSL est indisponible", + "c47127f222": "WSL par défaut", + "0c9f3cf9da": "Choisissez où Orca vérifie et installe les skills globales des agents.", + "f00d6aa9b5": "WSL n'est pas disponible sur cette machine.", + "7c776ff9d8": "wsl", + "fc0fcf72fd": "Enregistrez la commande shell WSL avant la configuration des skills.", + "windowsPathUnknown": "Le PATH de la commande shell WSL n'a pas pu être vérifié", + "refreshCliRegistration": "Actualisez le statut d'enregistrement du CLI et réessayez." + }, + "CommitMessageAiPane": { + "841ed9884a": "Utilisé par les dépôts qui n'ont pas personnalisé Source Control AI.", + "ad66ff886d": "Valeurs par défaut de Source Control AI", + "347094560b": "Utilisé par les dépôts qui héritent des valeurs par défaut globales de hosted review.", + "2dafc7646e": "Valeurs par défaut de création de hosted review", + "e9d46a544d": "Valeurs par défaut utilisées à l'ouverture du composer de hosted review.", + "b125eabffa": "Ouvrir la hosted review créée dans votre navigateur après l'envoi.", + "7662715213": "Ouvrir la hosted review après création", + "b27b0809f3": "Lance la génération des détails de hosted review une fois, à l'ouverture du composer.", + "d5f0de6309": "Générer les détails à l'ouverture de Create PR", + "6278c0ce43": "Préférer les templates de pull request du dépôt quand aucune description n'est définie.", + "d8b6764d79": "Utiliser le template de review quand disponible", + "e001734396": "Créer les hosted reviews en brouillon, sauf changement dans le composer.", + "6ba48f07a4": "Brouillon par défaut", + "15b60d54b2": "par ex. ollama run llama3.1 {prompt}", + "3f1b26cc91": "pour passer l'entrée de la commande en argument ; sinon Orca la redirige sur stdin.", + "4f722a5f53": "Utilisé par les recettes de message de commit, de pull request et de nom de branche qui sélectionnent Commande personnalisée. Utilisez", + "47e45cbd5a": "Commande personnalisée", + "1ef29f8c29": "Ligne de commande qu'Orca exécute quand une recette texte utilise Commande personnalisée.", + "2339a89104": "Ajoute des boutons IA qui exécutent l'agent sélectionné avec le modèle de commande de cette action.", + "d5b45a3628": "Afficher les actions Source Control AI", + "7bcad2b200": "Ajoute des recettes d'actions pour les actions commit, pull request, nom de branche et correction de Source Control.", + "d54c64163d": "activé", + "4ec89c319e": "agent", + "34d0348e34": "générer", + "8cd2be0948": "message", + "ca433708cb": "commit", + "0b7eafe55f": "ai", + "2c5436c018": "ouvrir", + "6c84ba6de3": "template", + "ebed4d2a29": "brouillon", + "02bab6542c": "pr", + "fdee745b87": "merge request", + "b388463881": "pull request", + "19e10a12bb": "hosted review", + "b8b6fd55b4": "{prompt}", + "fc1a525fa5": "placeholder", + "a69e1fe91a": "prompt", + "1df7d71313": "binaire", + "407d28bde6": "cli", + "54038660e0": "command", + "25350d670f": "custom" + }, + "ComputerUsePane": { + "1735461723": "Permet aux agents d'inspecter et de piloter des applications de bureau locales.", + "93255aaf18": "Compétence Computer Use", + "45f8e22c2e": "Ouvert", + "d95d1cfab8": "Actualiser", + "0c29da5805": "Prêt", + "3383ea1aab": "Impossible de réinitialiser les autorisations Computer Use", + "f189f448a3": "Réinitialiser l'accès Computer Use", + "5c45349665": "Impossible d'ouvrir les autorisations Computer Use", + "7801ac08ec": "Les autorisations Computer Use ne sont requises que sur macOS", + "740766c291": "La configuration de Computer Use est déjà terminée", + "697005758f": "Panneau Confidentialité et sécurité de macOS ouvert", + "2168fa5ab0": "Impossible de charger les autorisations Computer Use", + "0c9a33f468": "Capture les fenêtres des applications pour que les agents puissent inspecter leur état visuel.", + "07bbe4c4cb": "Captures d'écran", + "4d03dec2d0": "Lit les arborescences d'interface des applications et effectue les actions demandées.", + "6b5a2cd3a5": "Accessibilité", + "6b17602073": "Réinitialiser l'accès", + "506f2acf7a": "Réinitialisation de l'accès...", + "4b65070096": "darwin", + "statusGranted": "Accordée", + "statusUnsupported": "macOS uniquement", + "statusNotEnabled": "Non activé" + }, + "DeveloperPermissionsPane": { + "4c17304beb": "Actualiser", + "6326a4c5cc": "Utilisez ces contrôles quand une CLI, une application locale ou un outil d'automatisation a besoin d'une autorisation de confidentialité macOS. Orca ne le demande pas au démarrage.", + "6f011b9bf6": "Les outils de terminal héritent du périmètre de confidentialité macOS d'Orca.", + "bfa3402305": "Impossible de demander l'autorisation", + "66e94d6cf3": "Demande d'autorisation envoyée", + "fa809e8ada": "Panneau Confidentialité et sécurité de macOS ouvert", + "48d87edcd2": "Autorisation accordée", + "a552887288": "Impossible de charger les autorisations développeur", + "4cfaa7e98a": "Outils pour appareils Bluetooth et expérimentations matérielles locales.", + "b2210b1b4f": "Bluetooth", + "dfbc12c8c8": "Débogage matériel et outils d'appareils communiquant avec des périphériques USB.", + "bf51e4a542": "Périphériques USB", + "f903bf20b5": "Permet aux terminaux et outils de développement de se connecter aux services de votre réseau local. macOS ne signale pas à Orca l'état actuel de cette autorisation.", + "e7bb06007c": "Réseau local", + "4a73f5217a": "Événements Apple pour les scripts qui contrôlent d'autres applications locales.", + "e119f0d66b": "Automatisation", + "7ca17b62c8": "macOS désigne Orca quand les agents qu'il exécute lisent les données d'autres applications, car Orca est le processus responsable des commandes du terminal. Accordez cette autorisation à Orca pour réduire ces invites. Puis quittez et rouvrez Orca.", + "c566bca278": "Accès complet au disque", + "9f35980756": "Outils d'injection de frappes clavier, de contrôle de fenêtres et d'automatisation d'UI.", + "5b2f22ca2d": "Accessibilité", + "0639db5496": "Outils de capture d'écran, d'automatisation visuelle et d'inspection d'UI.", + "f24f31a884": "Enregistrement de l'écran", + "550cfa3750": "Capture webcam et applications de test locales utilisant la caméra.", + "e5b5f3d6b9": "Caméra", + "cc8151d9fa": "Saisie vocale, transcription, enregistrement audio, CLI sox, ffmpeg et Whisper.", + "16381e040a": "Microphone", + "dac08ec03e": "Traitement...", + "actionRequest": "Demander", + "actionOpenSettings": "Ouvrir les paramètres", + "actionTriggerPrompt": "Déclencher l'invite", + "statusGranted": "Accordée", + "statusDenied": "Refusée", + "statusNotRequested": "Non demandée", + "statusRestricted": "Restreint", + "statusUnsupported": "macOS uniquement", + "statusEntitled": "Habilitée", + "statusCheckManually": "Vérifier manuellement", + "actionRequestAccess": "Demander l'accès", + "localNetworkPromptCheck": "Vérifiez la présence d'une invite macOS", + "localNetworkPromptGuidance": "Si une invite apparaît, choisissez Autoriser. Si aucune invite n'apparaît, ouvrez Réglages Système et activez Orca sous Confidentialité et sécurité → Réseau local.", + "localNetworkOpenSettings": "Ouvrir Réglages Système", + "openSettingsFailed": "Impossible d'ouvrir Réglages Système", + "localNetworkOpenSystemSettings": "Ouvrir Réglages Système", + "statusManagedByMacOS": "Géré par macOS", + "connectionTestInvalidTarget": "Saisissez un nom d'hôte ou une adresse IP privée du réseau local, et un port compris entre 1 et 65535.", + "connectionTestTimeout": "Délai de connexion dépassé. Vérifiez la cible, le service et les réglages Réseau local de macOS.", + "connectionTestRefused": "L'hôte a répondu, mais le port a refusé la connexion.", + "connectionTestUnreachable": "Impossible d'atteindre la cible.", + "connectionTestUnresolved": "Le nom d'hôte n'a pas pu être résolu.", + "connectionTestUnsupported": "Le test de connexion est disponible dans l'application de bureau macOS.", + "connectionTestFailed": "Le test de connexion n'a pas pu être effectué.", + "connectionTestTitle": "Tester la connexion", + "connectionTestDescription": "Saisissez un service situé sur un autre appareil de votre réseau local. Orca teste le même chemin réseau que celui utilisé par les outils de terminal.", + "connectionTestLastVerified": "Dernière vérification", + "connectionTestNotYetVerified": "Aucun test réussi enregistré.", + "connectionTestHost": "Hôte", + "connectionTestPort": "Port", + "connectionTestRunning": "Test en cours...", + "connectionTestAction": "Tester la connexion" + }, + "ExperimentalPane": { + "9762364929": "Utilise le clonage APFS sur macOS quand c'est possible, sinon crée des liens symboliques vers les dossiers ou fichiers configurés dans les worktrees créés.", + "24416f42cd": "Chemins partagés sur les worktrees", + "fb82ea1d7a": "Matérialise automatiquement les fichiers ou dossiers configurés dans les worktrees nouvellement créés.", + "a20d5ea365": "Maintient un surlignage au niveau du volet après une sonnerie de terminal ou la fin d'un agent, jusqu'à ce que vous interagissiez avec ce volet. Expérimental pendant que nous ajustons le signal.", + "ec897e8d89": "Attention du terminal", + "88b7613afb": "Surlignage persistant du volet pour les sonneries de terminal et les fins d'agents.", + "0277901cf7": "Ajoute une entrée Agents à la barre latérale gauche avec un flux groupé par worktree pour les agents terminés, les questions bloquantes, l'état non lu et les événements de création de worktree. Expérimental — le modèle d'événements et l'interface peuvent changer.", + "a05bcdaf57": "Vue Agents", + "f63ea281e3": "Flux groupé dans la barre latérale gauche pour les achèvements d'agents et les états bloquants.", + "agentHibernation": { + "copy": "Arrête les terminaux d'agents en arrière-plan inactifs après la fenêtre d'inactivité configurée et reprend les sessions prises en charge quand vous les rouvrez. La mise en veille des agents conserve les options de lancement pour les agents démarrés par Orca. Les agents démarrés manuellement peuvent reprendre avec vos valeurs par défaut Orca actuelles. Expérimental pendant que nous ajustons le modèle de sécurité.", + "description": "Arrête les terminaux d'agents en arrière-plan inactifs après la fenêtre d'inactivité configurée et reprend les sessions prises en charge quand vous les rouvrez.", + "idleMinutesDescription": "Nombre de minutes d'inactivité qu'un agent d'arrière-plan terminé doit attendre avant qu'Orca puisse le mettre en veille.", + "idleMinutesLabel": "Mise en veille après", + "idleMinutesSuffix": "minutes", + "title": "Mise en veille des agents", + "toggleLabel": "Activer la mise en veille des agents" + }, + "newWorktreeCardStyle": { + "copy": "Prévisualise la nouvelle mise en page des cartes de worktree, l'emplacement des métadonnées, les options du menu d'affichage des cartes et la présentation des statuts.", + "description": "Prévisualise la nouvelle mise en page des cartes de worktree, l'emplacement des métadonnées, les options du menu d'affichage des cartes et la présentation des statuts.", + "title": "Nouveau style de carte", + "toggleLabel": "Activer le nouveau style de carte" + }, + "ca2219fe5e": "Affiche un petit compagnon animé épinglé dans le coin inférieur droit. Choisissez un personnage (Claudino, OpenCode, Gremlin) ou importez votre propre PNG, APNG, GIF, WebP, JPG ou SVG depuis le menu compagnon de la barre d'état. Masquez-le à tout moment depuis le même menu sans désactiver ce réglage.", + "dd6f0a1d45": "Compagnon", + "0e89a574ae": "Compagnon animé flottant dans le coin inférieur droit.", + "nativeChat": { + "title": "Chat UI", + "description": "Prévisualisez la surface de chat de bureau pour les sessions de terminal d'agents prises en charge.", + "copy": "Ajoute une vue Chat UI accessible depuis les volets de terminal d'agents pris en charge. Expérimental pendant que nous peaufinons la fidélité de la transcription, le streaming et la parité avec le terminal.", + "toggleLabel": "Activer Chat UI", + "defaultTitle": "Vue par défaut", + "defaultCopy": "Choisissez comment s'ouvrent les nouveaux onglets de terminal d'agents pris en charge.", + "defaultViewLabel": "Vue Chat UI par défaut", + "defaultViewTerminal": "Chat dans le terminal", + "defaultViewNative": "Chat UI" + }, + "agentDashboard": { + "title": "Tableau de bord des agents", + "description": "Tableau Kanban pour surveiller les agents de tous les worktrees, dans la fenêtre ou en fenêtre indépendante.", + "copy": "Ajoute une entrée Tableau de bord des agents à la barre latérale gauche. Surveillez les agents qui ont besoin de vous, qui travaillent ou qui ont terminé, avec en option les agents inactifs.", + "toggleLabel": "Activer le tableau de bord des agents", + "modeLabel": "Mode d'affichage", + "modeCopy": "Affiche le tableau de bord dans la fenêtre, à côté de la barre latérale, ou dans une fenêtre indépendante.", + "modeAriaLabel": "Mode d'ouverture du tableau de bord des agents", + "modeInWindow": "Dans la fenêtre", + "modePopout": "Fenêtre indépendante" + } + }, + "FloatingWorkspacePane": { + "aeaf76fda9": "Barre d'état", + "9fb225f2d7": "Bouton flottant", + "3c900e26e5": "Le raccourci clavier fonctionne quel que soit l'emplacement du bouton d'activation.", + "5e5a8da236": "Emplacement du bouton d'activation", + "505001823e": "Choisir le dossier de l'espace de travail flottant", + "81afb79785": "C'est ici que démarrent les nouveaux onglets de terminal flottants. Les notes Markdown sont enregistrées dans l'espace de travail flottant propre à l'application Orca.", + "12aa09f10c": "Dossier du terminal", + "41eb95f7f0": "Affiche le bouton et le panneau de l'espace de travail flottant.", + "5136813663": "Activer l'espace de travail flottant", + "37df688d6f": "Activez l'espace de travail flottant et choisissez où démarrent les nouveaux onglets.", + "1f67f39384": "Espace de travail flottant" + }, + "AgentCacheTimerSection": { + "05de84a104": "1 heure", + "54395ecd7c": "5 minutes", + "8b9e202e0a": "Alignez cette durée sur le TTL de cache de votre fournisseur. Par défaut : 5 minutes.", + "a2a8962138": "Durée du minuteur", + "b4e7302944": "Minuteur de cache", + "487b176240": "Afficher un compte à rebours dans la barre latérale quand un agent Claude devient inactif.", + "9c20253679": "Afficher un compte à rebours quand un agent Claude devient inactif.", + "fe590653c1": "Claude met votre conversation en cache pour réduire les coûts. Après une trop longue inactivité, le cache expire et le message suivant renvoie tout le contexte, à un coût plus élevé. Ce compte à rebours vous indique quand reprendre.", + "a137f8854d": "Minuteur de cache de prompt", + "80c454e8a6": "Alignez cette durée sur le TTL de cache de votre fournisseur." + }, + "GeneralEditorSettingsSection": { + "f80603d293": "Afficher les contrôles des notes markdown locales en mode éditeur enrichi et dans les actions de passation aux agents.", + "4edc104f0f": "Notes de revue Markdown", + "5f02e6fb21": "Afficher les contrôles des notes de revue markdown locales en mode éditeur enrichi.", + "51161d1647": "Afficher la minimap lors de l'édition d'un fichier.", + "6690b1ffb9": "Minimap", + "5a1ea6eaa2": "Masquée", + "73a09aad63": "Affichée", + "1de48ad940": "Arborescence de fichiers des diffs par défaut", + "1b87897af9": "Afficher ou masquer l'arborescence de fichiers à l'ouverture des vues de diff combinées.", + "12cbc0d0d6": "Côte à côte", + "05b6df93b3": "En ligne", + "7311f67ee7": "Vue de diff par défaut", + "b492397d34": "Format de présentation préféré pour l'affichage par défaut des diffs git.", + "a5db1d3975": "ms", + "fc5c5306ff": "ms.", + "8112cd6dcf": "Délai d'attente d'Orca après votre dernière modification avant l'enregistrement automatique. Au premier lancement, la valeur par défaut est de", + "d6cf227ca0": "Délai d'enregistrement automatique", + "1bec6d8318": "Délai d'attente d'Orca après votre dernière modification avant l'enregistrement automatique.", + "70bb30feb1": "Enregistre automatiquement les modifications de l'éditeur et des diffs éditables après une courte pause.", + "0df2e4fd12": "Enregistrement automatique des fichiers", + "d21136d9ef": "Configure la façon dont Orca enregistre les modifications de fichiers.", + "45c6e85c4d": "Éditeur", + "8f1afdfbd8": "Retour à la ligne dans les diffs", + "4aa4d9fb73": "Renvoie à la ligne les lignes longues dans les éditeurs de diff au lieu d'imposer un défilement horizontal.", + "bf16ef0af2": "Désactivé", + "3f6892f307": "Activé", + "b82f86d7d2": "Vérification orthographique du Markdown enrichi", + "5195f0b9ef": "Affiche les soulignements d'orthographe et les suggestions du navigateur pendant l'édition du Markdown enrichi.", + "7ddd66fede": "Retour à la ligne dans l'éditeur", + "9b18de6eea": "Renvoie à la ligne les lignes longues dans les éditeurs de fichiers au lieu d'imposer un défilement horizontal." + }, + "AdvancedNetworkSettingsSection": { + "3e431564b5": "localhost, 127.0.0.1, *.internal", + "33ee3ca3af": "Facultatif. Séparez les hôtes par des virgules, des points-virgules ou des sauts de ligne.", + "f6d76cc8f4": "Règles de contournement du proxy", + "fb7130dcb9": "Hôtes qui doivent contourner le proxy HTTP configuré.", + "0adfce9fa7": "Prend en charge les URL http, https, socks, socks4 et socks5.", + "476f302aca": "http://proxy.example.com:8080", + "1e214e265a": "Laissez vide pour utiliser les réglages de proxy du système et les variables d'environnement de proxy héritées.", + "f00daf6324": "Proxy HTTP", + "823e0f15b1": "URL de proxy pour les requêtes réseau d'Orca et les processus enfants des terminaux locaux.", + "d93c7cd531": "Configure le routage réseau au niveau de l'application.", + "c46cdbbd4e": "Réseau", + "configureProxy": "Configurer le proxy" + }, + "GeneralPane": { + "d58fccfd84": "Navigation", + "5cb5475664": "Confirmer avant de fermer les onglets épinglés", + "36b2a5dc6d": "Afficher une boîte de dialogue de confirmation avant de fermer un onglet épinglé.", + "projectRuntime": "Runtime des projets", + "projectRuntimeDescription": "Runtime par défaut pour les projets Windows locaux qui ne le remplacent pas." + }, + "GeneralSupportSection": { + "af7d9f4396": "Merci pour votre soutien !", + "6922c1fa2b": "Ajouter une étoile à Orca sur GitHub", + "511782265b": "Soutenez le projet avec une étoile GitHub via la CLI gh.", + "55a87e5fd1": "Soutenir Orca", + "964acc6bb4": "Ajouter une étoile", + "73b327e793": "Réessayer", + "c9f96d4234": "error", + "397719bee5": "Ajout de l'étoile...", + "1e29570462": "ajout de l'étoile", + "9d181300e3": "favori", + "5c49f02662": "masqué", + "b3f0584f5d": "chargement", + "cb65c75b11": "Ouverture...", + "f2d4f877b2": "Ouvrir GitHub" + }, + "GeneralUpdateSettingsSection": { + "8a52ca1d02": "Notes de version", + "d89806cc89": "est prête à être installée.", + "a6b37929dc": "Version", + "8311da27ba": "est disponible. Cliquez sur « Installer la mise à jour » pour la télécharger et l'installer.", + "f44299636f": "Redémarrer pour mettre à jour (", + "42717918f4": "Installer la mise à jour (", + "02dc082e70": "Impossible de démarrer le téléchargement de la mise à jour.", + "e1a647adc5": "Rechercher des mises à jour", + "ceb579abaf": "Recherche les mises à jour de l'application et installe une version plus récente d'Orca.", + "d91ebfb87e": "Version actuelle : {{value0}}", + "f2b1ccc12a": "Mises à jour", + "bd79d412f0": "Échec de la vérification des mises à jour. {{value0}}", + "b9ad70c30d": "Erreur de mise à jour. {{value0}}", + "6405510b92": "error", + "a0832ccdb1": "téléchargée", + "2a48034c4c": "Téléchargement v{{value0}}... {{value1}} %", + "4c1c001813": "téléchargement", + "f40d88390d": "Vous utilisez la dernière version.", + "90eb7309d7": "not-available", + "82465b2444": "disponible", + "31fd7150cf": "Recherche de mises à jour...", + "3394d1f663": "vérification", + "d69a09b672": "Les mises à jour sont vérifiées automatiquement au lancement.", + "7173352632": "idle" + }, + "GeneralWorkspaceSettingsSection": { + "3d538a98f7": "Choisissez les applications proposées dans le menu « Ouvrir dans » d'un espace de travail.", + "008f92085f": "Applications « Ouvrir dans »", + "824b98a0d9": "Demander confirmation avant de supprimer des automatisations et leur historique d'exécution.", + "ea98373cd8": "Confirmer avant de supprimer les automatisations", + "d2dd2ca2e3": "Afficher une boîte de dialogue de confirmation avant de supprimer une automatisation et son historique d'exécution.", + "28bc3d085e": "Demander confirmation avant de supprimer un espace de travail depuis le menu contextuel. En cas d'échec, l'option « Forcer la suppression » reste disponible.", + "9f380934cf": "Confirmer avant de supprimer les espaces de travail", + "5734db82af": "Afficher une boîte de dialogue de confirmation avant de supprimer un espace de travail.", + "4fbf910ded": "Créer les espaces de travail dans un sous-dossier nommé d'après le dépôt.", + "ba3480642f": "Imbriquer les espaces de travail", + "a246f5ce6f": "Dossier racine où sont créés les dossiers des espaces de travail.", + "5567191a6e": "Parcourir", + "0e9fc0eadc": "Dossier des espaces de travail", + "e2955d9ccb": "Configure où sont créés les nouveaux espaces de travail.", + "7511097c5d": "Espace de travail", + "31e300af1c": "Confirmer avant de supprimer les artefacts", + "fb29a73a17": "Afficher une boîte de dialogue de confirmation avant de supprimer un artefact partagé et de rompre son lien public.", + "bf46474e33": "Demander confirmation avant de supprimer un artefact partagé. Quiconque détient son lien public perd l'accès.", + "externalWorktrees": "Worktrees externes", + "externalWorktreesDescription": "Choisissez si les worktrees créés en dehors d'Orca apparaissent par défaut.", + "show": "Afficher", + "hide": "Masquer" + }, + "GhosttyImportModal": { + "9d3e56ca36": "Appliquer les modifications", + "f96688b6bc": "Annuler", + "b7ddae600c": "Terminé", + "e4bda7ce6f": "Aucune configuration Ghostty trouvée sur ce système.", + "b58d4c9051": "Clés non prises en charge", + "674b5ccd6b": "Aucun nouveau réglage à importer — vos réglages actuels correspondent déjà.", + "a4c5dec640": "Réglages à mettre à jour", + "4466f4cdaa": "Importation terminée", + "023a52c1f7": "Chargement de l'aperçu…", + "2763b0c045": "Consultez les réglages qui seront importés depuis votre configuration Ghostty.", + "d2f33670a9": "Importer depuis Ghostty", + "273e7e81fe": "Configurations", + "1f744a72f4": "Configuration" + }, + "WarpThemeImportModal": { + "title": "Importer depuis Warp", + "description": "Importez des thèmes Warp comme thèmes de terminal Orca.", + "yaml_title": "Importer un thème YAML", + "yaml_description": "Importez des fichiers YAML de thèmes (format Warp) comme thèmes de terminal Orca.", + "yaml_no_themes_found": "Aucun thème trouvé dans les fichiers sélectionnés.", + "choose_file": "Choisir un fichier", + "choose_folder": "Choisir un dossier", + "loading": "Chargement des thèmes Warp...", + "found_theme_one": "1 thème trouvé", + "found_theme_other": "{{value0}} thèmes trouvés", + "found_in_source": " dans {{value0}}", + "clear_all": "Tout effacer", + "select_all": "Tout sélectionner", + "colors_only": "Couleurs uniquement", + "no_themes_found": "Aucun thème Warp personnalisé trouvé.", + "builtin_themes_hint": "Les thèmes préinstallés de Warp font partie de l'application Warp et ne peuvent pas être lus depuis le disque. Orca inclut déjà la plupart d'entre eux, comme Dracula, Gruvbox, Solarized et Tokyo Night.", + "custom_theme_yaml_hint": "Les thèmes personnalisés et communautaires doivent exister sous forme de fichiers YAML dans un dossier de thèmes Warp pour que l'import automatique puisse les détecter. Si vous avez cloné le dépôt public de thèmes de Warp, utilisez « Choisir un dossier » pour importer ce clone.", + "choose_manually": "Choisissez un fichier ou un dossier YAML de thèmes à importer manuellement.", + "skipped_files": "Fichiers ignorés", + "more_skipped_files": "{{value0}} autres fichiers ignorés.", + "cancel": "Annuler", + "import_theme_one": "Importer 1 thème", + "import_theme_other": "Importer {{value0}} thèmes", + "import_themes": "Importer des thèmes" + }, + "useWarpThemeImport": { + "unknown_error": "Erreur inconnue", + "imported_one": "1 thème importé", + "imported_other": "{{value0}} thèmes importés", + "import_failed": "Échec de l'importation des thèmes", + "over_limit_one": "Importer ces thèmes dépasserait la limite de {{value0}} thèmes de terminal personnalisés. Désélectionnez 1 nouveau thème et réessayez.", + "over_limit_other": "Importer ces thèmes dépasserait la limite de {{value0}} thèmes de terminal personnalisés. Désélectionnez {{value1}} nouveaux thèmes et réessayez." + }, + "YamlThemeImportButton": { + "label": "Importer depuis YAML" + }, + "BranchPrefixFeedback": { + "6c40c0908f": "Le préfixe ne peut pas contenir d'espaces ni de caractères spéciaux comme ~ ^ : ? * [ \\", + "64d70b156a": "Les branches seront nommées {{example}}", + "808f9a726e": "Aucun préfixe ne sera appliqué" + }, + "GitPane": { + "895d3f70b8": "gh", + "32dca11189": "github", + "c4f610d057": "En-têtes de limite de débit REST actuels de la CLI GitLab, lorsqu'ils sont disponibles.", + "0de4ae556c": "Budget API GitLab", + "cdd793134e": "budget api", + "b9c011fbc2": "limite de débit", + "3072428ac7": "glab", + "8a527d48e3": "gitlab", + "aa204f185f": "Limites de débit REST, Search et GraphQL actuelles de la CLI GitHub.", + "612a440e57": "Quota d'API GitHub", + "2cde9044a8": "graphql", + "36e3de3619": "de comparer avec un historique périmé. Orca ignore la mise à jour si cette branche contient des modifications non commitées ou des commits uniquement locaux.", + "d072a12995": "git diff main...HEAD", + "db3a127eb1": ". Cela permet de garder des commandes comme", + "3ae3de8898": "master", + "5bf885be48": "ou", + "ffba483bae": "main", + "976afc6b3e": "Quand vous créez un espace de travail, Orca actualise la base distante et fait avancer en toute sécurité votre branche locale correspondante (fast-forward), telle que", + "1ec5c91e1d": "Choisissez si les noms de branches utilisent votre nom d'utilisateur Git, un préfixe personnalisé ou aucun préfixe.", + "330f584b50": "Préfixe de branche", + "1ffaadf0a0": "Préfixe ajouté aux noms de branches à la création des worktrees.", + "813e15b346": "custom", + "2351aa5a31": "nom d'utilisateur git", + "cc63fce906": "nommage des branches", + "b559bf9899": "p. ex. feature", + "aefa1ecb59": "Aucun nom d'utilisateur git configuré", + "f35007e6e8": "git-username", + "3d172725cc": "Aucun", + "1f32ba27a6": "Personnalisé", + "a182c5125e": "Nom d'utilisateur Git", + "sourceControlGroupOrderTitle": "Ordre des groupes du contrôle de code source", + "sourceControlGroupOrderDescription": "Choisissez si Modifications, Modifications indexées ou Fichiers non suivis apparaissent en premier dans le contrôle de code source.", + "changesFirst": "Modifications d'abord", + "stagedFirst": "Indexées d'abord", + "untrackedFirst": "Non suivis d'abord", + "compareAgainstUpstreamTitle": "Base de comparaison par défaut", + "compareAgainstUpstreamDescription": "Choisissez la base que le contrôle de code source utilise par défaut pour comparer les changements commités. L'amont de la branche suit automatiquement la branche courante et revient à la branche par défaut du dépôt en l'absence d'amont. Vous pouvez toujours changer la base de comparaison d'un worktree depuis son panneau Git. Les cibles de pull request et de rebase ne changent pas.", + "compareBaseRepositoryDefault": "Valeur par défaut du dépôt", + "compareBaseBranchUpstream": "Amont de la branche" + }, + "HiddenExperimentalGroup": { + "d0f914a528": "Bascule fictive", + "1014ddbfaf": "Sans effet aujourd'hui. Réservée comme premier emplacement pour les options expérimentales masquées.", + "232cf83de8": "Bascules non répertoriées pour les tests internes. Rien ici n'est pris en charge.", + "3e9e827ca5": "Expérimental masqué" + }, + "InputPane": { + "db15068196": "Activé par défaut sur Linux et macOS. Linux utilise le presse-papiers de sélection du système ; les autres plateformes utilisent un tampon privé.", + "ad31c3c5fb": "Collage de la sélection au clic milieu" + }, + "IntegrationsPane": { + "2122e15517": "Chaque espace de travail Linear connecté possède une clé stockée par le runtime actif. Les clés à accès complet peuvent couvrir toutes les équipes auxquelles le propriétaire de la clé a accès ; les clés restreintes peuvent être remplacées à tout moment.", + "e7b2dd46f9": "Test en cours…", + "fe4d378dc4": "Vérifié", + "f5c5246514": "Ajouter l'accès Linear", + "6432f6522e": "Connecté", + "077844591a": "Ajouter un accès à l'espace de travail", + "264a9b6128": "Linear", + "4831ba1083": "Revérifier", + "01f6c7582e": "En savoir plus", + "1a62c295c6": "Les identifiants Gitea sont configurés mais l'authentification a échoué. Vérifiez le token, l'URL de base de l'API et les permissions du dépôt, puis redémarrez Orca si les variables d'environnement ont changé.", + "5a1f86225a": "uniquement quand Orca ne peut pas déduire l'URL de l'API depuis le remote.", + "6193444689": "ORCA_GITEA_API_BASE_URL", + "2c0330ec3e": "pour les dépôts privés, et définissez", + "e678d89e8c": "ORCA_GITEA_TOKEN", + "d9467ab026": "Les dépôts publics sont détectés via leur remote git. Définissez", + "4ab9b96925": "Gitea", + "953b7bf6f7": "Les identifiants Azure DevOps sont configurés mais l'authentification a échoué. Vérifiez le token, l'URL de base de l'API et les permissions du dépôt, puis redémarrez Orca si les variables d'environnement ont changé.", + "6f317f5132": "uniquement quand Orca ne peut pas déduire l'URL de base de l'API depuis le remote git.", + "ae6b7f5f40": "ORCA_AZURE_DEVOPS_API_BASE_URL", + "67a9f26a80": ". Définissez", + "8f960935c1": "ORCA_AZURE_DEVOPS_ACCESS_TOKEN", + "ce3c58cd63": ", ou définissez", + "5ee6ef6405": "ORCA_AZURE_DEVOPS_TOKEN", + "4ee74d1470": "Définissez", + "5efce6953d": "Azure DevOps", + "3c3cf05c63": "Les identifiants Bitbucket sont configurés mais l'authentification a échoué. Vérifiez le token et les permissions du dépôt, puis redémarrez Orca si les variables d'environnement ont changé.", + "6e0ff3403e": "ORCA_BITBUCKET_ACCESS_TOKEN", + "44cde4aa01": "ORCA_BITBUCKET_API_TOKEN", + "a6c2816115": "et", + "b8a7efb3f6": "ORCA_BITBUCKET_EMAIL", + "8489c0aa49": "Bitbucket", + "e74de656ce": "glab auth login", + "05e5245af7": "La CLI GitLab est installée mais non authentifiée. Exécutez cette commande dans un terminal :", + "a83cac5726": "Installer la CLI GitLab", + "35a3379372": "Installez la CLI GitLab pour activer les merge requests, les issues et les pipelines.", + "ea160a9978": "CLI.", + "a3326f6f1b": "glab", + "027440e1cb": "Merge requests, issues, todos et pipelines via la", + "513abfe47d": "GitLab", + "51000487c4": "gh auth login", + "09285e9fe6": "La CLI GitHub est installée mais non authentifiée. Exécutez cette commande dans un terminal :", + "399cf46867": "Installer la CLI GitHub", + "c0c8575e05": "Installez la CLI GitHub pour activer les pull requests, les issues et les checks.", + "f36365ed45": "gh", + "de6a0d13ab": "Pull requests, issues et checks via la", + "70c5f74f36": "GitHub", + "8e078e480c": "Déconnecter {{value0}}", + "95b9a87e7e": "Tester", + "62b20292de": "error", + "ae38fc62a8": "ok", + "33ae9730a8": "Ajoutez un accès Linear pour parcourir et lier des issues.", + "98ded79cd7": "Connecté{{value1}} : {{value0}} espace de travail", + "4bdc6fe4f5": "not-configured", + "4972f3c95d": "configuré", + "3614887c40": "vérification", + "45bf5e6e4b": "Échec de l'authentification", + "e1bd5364e6": "Configuration facultative", + "e7a961e1c5": "Configuré", + "6bd148dcb5": "Pull requests et statuts de commits via l'API REST Gitea.", + "6355fe585e": "Pull requests et statuts de commits pour les dépôts détectés", + "1fac9b4910": "{{value0}} · Pull requests et statuts de commits", + "f92fbf11aa": "Non configuré", + "6791d7af95": "Pull requests et statuts de builds via des tokens de l'API REST Azure DevOps.", + "e3d5a24979": "Pull requests et statuts de builds pour les Azure Repos détectés", + "277fc23929": "{{value0}} · Pull requests et statuts de builds", + "295154e54e": "connecté", + "0879860c58": "Pull requests et statuts de builds via des tokens de l'API Bitbucket Cloud.", + "9707523939": "Pull requests et statuts de builds", + "a565377c38": "not-installed", + "15cf990798": "Non authentifié", + "f7eb5f0b24": "Non installé", + "3ba07f933b": "Connectez les trackers d'issues qu'Orca peut utiliser pour parcourir les tâches et démarrer des espaces de travail avec le contexte associé.", + "70e885705b": "Fournisseurs de tâches", + "1683acbac4": "Connectez les hébergeurs de code qu'Orca peut utiliser pour les pull requests, merge requests, checks et statuts de revue.", + "298c65ecac": "Fournisseurs de revue" + }, + "KagiSessionLinkForm": { + "92f0b4e472": "Effacer", + "9f741627a7": "Lien de session Kagi effacé.", + "d5c8b94c5b": "Enregistrer", + "ff450194cd": "Lien de session privée Kagi", + "e383683485": "https://kagi.com/search?token=...", + "81409d9362": "Lien de session privée facultatif pour l'authentification Kagi.", + "3e5b7c6c25": "Lien de session Kagi enregistré.", + "0911d5fa4c": "Saisissez un lien de session privée Kagi de la forme https://kagi.com/search?token=..." + }, + "KeybindingsFileActions": { + "abc49853fb": "Recharger depuis le disque", + "a8a8d6b9d3": "Révéler dans le gestionnaire de fichiers", + "9e24c0e858": "Ouvrir dans Cursor", + "1637f64033": "Ouvrir dans VS Code", + "98f1a23e1c": "Ouvrir avec l'application par défaut", + "400397a10d": "Ouvrir le menu du fichier de raccourcis clavier", + "1c2be2b2c6": "Modifier le fichier dans Orca", + "c5886a31cc": "Échec de l'ouverture de l'éditeur externe.", + "cdf794f46d": "Le fichier de raccourcis clavier n'est pas disponible.", + "dd532a01ce": "Échec de l'ouverture des raccourcis clavier dans Orca." + }, + "ManageSessionKillDialog": { + "6bf4627168": "Annuler", + "ad9832aa26": ". Tout travail non enregistré dans ce volet sera perdu. L'action est irréversible.", + "8401328fed": "Force la fermeture de", + "87dcafc85c": "Tuer cette session ?", + "0b0db4c68c": "Tuer la session", + "d3dba51b15": "Arrêt forcé…" + }, + "ManageSessionsSection": { + "33c2a1e1b4": "Tuer la session {{value0}}", + "2896a50f50": "Aller au terminal {{value0}}", + "e26a60d9eb": "Aucune session.", + "39c53d6d74": "Chargement…", + "5ed15e778c": "Redémarrer le daemon", + "3282db098c": "Tuer toutes les sessions", + "b3b1cc5708": "Actualiser", + "a795a9552a": "Sessions", + "7c4889a724": "Récupérez un terminal gelé ou qui se comporte mal en tuant des sessions ou en redémarrant le démon sous-jacent.", + "d1b80fd5cd": "Gérer les sessions", + "9c940434af": "Repassez au runtime local pour redémarrer ou tuer les sessions du démon local.", + "ad467eaadc": "La gestion des sessions est indisponible tant qu'un serveur runtime distant est actif.", + "8dbd96b463": "Impossible de tuer la session.", + "0735b7a586": "Impossible de tuer la session — elle a peut-être déjà disparu.", + "bfba05dccd": "Session tuée.", + "c535cbdd09": "Impossible de charger les sessions.", + "e3d1fbe008": "Redémarrer", + "a06ababda0": "killAll" + }, + "McpConfigFileRow": { + "b145eb6009": "env:", + "e720c139cd": "Ouvert", + "845ae248e8": "valid" + }, + "McpConfigSection": { + "4d16a0d9ac": "Vérifié", + "b900cd6282": "Aucune configuration MCP trouvée. Ajoutez une configuration d'espace de travail vide si vous voulez que ce dépôt définisse ses propres serveurs MCP.", + "3b224167ff": "serveur", + "251b96564a": "détecté ·", + "f34c152dc0": "Actualiser les configurations MCP", + "6bac9ddfc6": "Les dépôts SSH sont lus via le système de fichiers distant. La création de starters est limitée à la configuration racine de l'espace de travail.", + "96f5609b04": "Inspectez les définitions de serveurs MCP utilisables par les agents travaillant dans ce dépôt.", + "55eea3ef47": "Configurations MCP", + "9ee215caf6": ".mcp.json", + "1f3665e35a": "Configuration MCP créée", + "82436439eb": "Ajouter une configuration MCP", + "0a5c1ead54": "Créer une configuration vide" + }, + "MobileEmulatorAgentControlRow": { + "1861982430": "Échec du chargement de l'état de la CLI.", + "8af7a8bc38": "Les commandes ciblent l'émulateur actif du worktree courant. Les coordonnées sont normalisées entre 0 et 1.", + "c7f3fe0a6e": "Commandes d'émulateur courantes", + "d94ca6a623": "Permet aux agents d'utiliser les commandes CLI d'Orca, y compris le contrôle de l'émulateur mobile.", + "67e19ee03c": "Compétence CLI Orca", + "aaf62a3dd2": "Installé dans", + "2fef055608": "Enregistre la commande CLI Orca pour que les agents puissent contrôler l'émulateur actif depuis leur shell.", + "4f2205f3b6": "Activer Orca CLI", + "ff4b7e65d6": "Laisser les agents de codage contrôler l'émulateur mobile actif avec des commandes CLI Orca.", + "2a674aa810": "Contrôle de l'émulateur mobile par les agents", + "cdeaed9e37": "Orca CLI enregistré dans PATH.", + "3d34423e88": "Enregistrement de la CLI Orca", + "3be27641c9": "afin que les commandes d'émulateur puissent s'exécuter depuis les shells des agents.", + "3941719a56": "Vérification de la CLI Orca avant d'ouvrir la configuration du skill." + }, + "MobileEmulatorExamples": { + "edf13dd03b": "Copier", + "c12b253997": "Copier l'exemple de prompt", + "d151e25078": "\"", + "b525ff2b12": "\"", + "4daa95f25a": "Collez l'un de ces prompts dans Claude Code, Codex ou un autre agent, dans un projet où le skill CLI Orca est installé.", + "0820b3f84f": "Essayez — exemples de prompts", + "1f608e7d60": "Échec de la copie du prompt.", + "2b077b5544": "Prompt copié." + }, + "MobileEmulatorSettingsPane": { + "19d39113b6": "Laisser les agents de codage contrôler l'émulateur mobile actif avec des commandes CLI Orca.", + "f2f8d97bb6": "Contrôle de l'émulateur mobile par les agents", + "143961d031": "Appareil par défaut", + "8aec2f99a0": "Actualiser la disponibilité des émulateurs", + "ae1612c58c": "Disponibilité", + "f9af91ea26": "Affiche l'action Nouvel émulateur mobile et permet aux agents de se rattacher à l'émulateur actif.", + "700ddbf9b1": "Activer l'émulateur mobile", + "bc39d0f115": "Configurez la prise en charge des émulateurs mobiles pour Orca et les agents de codage.", + "6593c9ddd3": "Émulateur mobile", + "a4f1c82d90": "Désactivé", + "b5e2d93e01": "Vérification...", + "c6f3ea4f12": "Prêt", + "d704fb5023": "Configuration requise", + "06b06429c6": "Vérification de la prise en charge du SDK Android et du simulateur iOS.", + "6d1483d4a0": "1 appareil émulateur détecté.", + "0a452d4d3b": "{{value0}} appareils émulateur détectés.", + "f62a1bb759": "Orca sélectionnera automatiquement un appareil émulateur une fois les appareils détectés.", + "b2fd62ea75": "Appareil par défaut pour les nouveaux onglets d'émulateur et les commandes d'attachement des agents. La sélection automatique privilégie un appareil déjà démarré." + }, + "MobileNetworkInterfaceSection": { + "63d5e4ae1e": "Régénérez le QR code et scannez-le depuis l'app mobile Orca.", + "87985ba6f5": "Dans ce menu Interface réseau, choisissez l'adresse Tailscale, généralement une IP en 100.x.y.z.", + "1f7c26d36a": "Connectez-vous au même tailnet sur les deux appareils.", + "668016be7a": "sur votre ordinateur et votre téléphone.", + "1dc87a7fbc": "Tailscale", + "51d29927eb": "Installer", + "9fc5d203ff": "Orca Mobile se connecte directement à cet ordinateur. Pour l'utiliser loin du même réseau local, placez votre ordinateur et votre téléphone sur le même réseau privé superposé (overlay), puis générez le QR code avec cette adresse réseau sélectionnée.", + "39fad211d9": "Se connecter hors de votre Wi-Fi grâce à un tailnet", + "a9db5d771d": "Actualiser les interfaces réseau", + "b2c384cfd6": "Aucune interface trouvée", + "d536b5e20d": "Choisissez l'adresse réseau à annoncer dans le QR code. Utilisez votre adresse LAN pour un appairage sur le même réseau, ou une adresse de réseau superposé (Tailscale, ZeroTier) pour un accès entre réseaux distincts.", + "406a35121c": "Interface réseau", + "c541f67790": "Générer le QR code", + "1e64659126": "Régénérer" + }, + "MobilePane": { + "dd3cd78d04": "Scanner avec Orca Mobile", + "35100bca5d": "Pendant que vous utilisez un terminal sur votre téléphone, Orca le réduit pour tenir sur l'écran du téléphone. À la fermeture de l'app ou quand vous passez à autre chose, ce réglage détermine s'il garde la taille téléphone (pour que les outils CLI interactifs ne se reforment pas) ou reprend sa taille bureau. Vous pouvez toujours utiliser Restaurer ce terminal ou Restaurer tous les terminaux sur la bannière pour redimensionner manuellement.", + "ee56f1c7e4": "Quand vous quittez l'app mobile", + "3939fd062c": "La révocation d'un appareil le déconnecte immédiatement.", + "254a6d09e4": "Appairé", + "d7ce676270": "Appareils appairés", + "e778ecb209": "Ou collez ce code dans l'app mobile :", + "310924ad2c": "Scannez ce code avec l'app mobile Orca. Chaque code crée un jeton d'appareil unique.", + "870e1b5ca5": "Échec de la révocation de l'appareil", + "2e3dd0bc29": "Appareil révoqué", + "711231348f": "Échec de la copie du code d'appairage", + "e3c427e020": "Échec de la génération du QR code", + "cb9067c1c1": "Le transport WebSocket n'est pas actif", + "d714614dbf": "Échec de l'actualisation des interfaces réseau", + "ff865419dc": "Après 30 minutes", + "d4ba07d914": "Après 5 minutes", + "c474aa09d8": "Après 1 minute", + "aa1263e881": "Conserver à la taille téléphone (par défaut)", + "6436e56546": "QR code pour l'appairage mobile", + "1b1b70279a": "Aucun appareil appairé pour le moment.", + "1592afcc7a": "Aucun appareil appairé pour le moment. Scannez le QR code avec l'app mobile Orca.", + "relayDegradedNotice": "Le relais est injoignable — ce code ne fonctionne que sur votre LAN ou via Tailscale. Régénérez-le pour réessayer.", + "pairingQrError": "Ce code d'appairage n'a pas pu être rendu sous forme de QR code. Copiez-le plutôt dans Orca Mobile.", + "pairingCodeReady": "Code d'appairage prêt", + "copyPairingCode": "Copier le code d'appairage", + "diagnosticsCopied": "Diagnostics copiés", + "diagnosticsCopyFailed": "Échec de la copie des diagnostics" + }, + "MobileSettingsPane": { + "9a3c280e49": "GitHub Releases", + "b0088412a1": "ou l'APK Android depuis", + "b5a2ed83ff": "App Store", + "installIntro": "Installez Orca Mobile depuis", + "installOutro": ", puis appairez ci-dessous.", + "androidApkLabel": "APK Android", + "c8491c17ef": "Contrôlez Orca depuis votre téléphone en scannant un QR code. Bêta / avant-première — attendez-vous à des bugs et à des changements incompatibles. Obtenez l'app iOS depuis", + "e7a3ae8c4e": "Mobile", + "174f4a3c6d": "Contrôlez les terminaux et les agents depuis votre téléphone.", + "1de96ec8a6": "Afficher le bouton Orca Mobile", + "682293cadf": "Afficher le bouton Orca Mobile en haut de la barre latérale gauche.", + "d4f2b65f30": "Afficher le raccourci Orca Mobile dans la barre latérale." + }, + "NotificationsPane": { + "906b4afebf": "Envoyer une notification de test", + "2772d2f257": "Ignorer les notifications quand le worktree déclencheur est déjà visible.", + "00cd406dbb": "Masquer quand la fenêtre est active", + "2a42dd8d6f": "Volume du son de notification", + "4aa5085cd7": "Personnalisé :", + "c258cb96dc": "Choisir le son de notification", + "2a2033c388": "Choisissez l'alerte jouée par Orca lors d'une notification bureau.", + "88686e6ca8": "Son de notification", + "b6fc369244": "Un terminal en arrière-plan émet un caractère de sonnerie.", + "591fe605b9": "Sonnerie du terminal", + "55f901a59b": "Un agent de codage termine et devient inactif.", + "ca76d06fd2": "Tâche d'agent terminée", + "deff6d30da": "Notifications système natives pour les événements en arrière-plan.", + "841c8c549f": "Activer les notifications", + "0fadad17ce": "Impossible de lire le son de notification", + "406feb0aa6": "La notification de test n'a pas été délivrée", + "6fc3781729": "Les notifications sont désactivées", + "4676a95bc3": "Vérifiez les réglages de notification bureau d'Orca.", + "0cb93240b8": "Le système n'a pas affiché la notification", + "145227ca2b": "Ouvrir les paramètres", + "d3d54e0915": "Notification de test envoyée", + "115437bc35": "Si aucune bannière macOS n'est apparue, activez « Autoriser les notifications » pour Orca.", + "7f45542625": "Notification de test demandée", + "98d70fb261": "Impossible de lire le son de notification personnalisé", + "c83b05a055": "Les notifications ne sont pas prises en charge sur ce système", + "274af61bc0": "système", + "6e6df3a09a": "Choisir un fichier personnalisé", + "76e02467b8": "Changer de fichier personnalisé", + "d3756cf5bc": "désactivé" + }, + "OpenAiTranscriptionKeyDialog": { + "fa83512e48": "Enregistrer la clé", + "07b26f2742": "Effacer la clé", + "d246b2bdb3": "Les clés locales d'exécution sont stockées dans ~/.orca via le stockage chiffré d'Electron quand celui-ci est disponible.", + "c3380e4ca5": "sk-...", + "2f797018f0": "Clé API configurée", + "16015322f9": "Clé API", + "07ed3e512e": "L'audio n'est envoyé à OpenAI que lorsqu'un modèle vocal OpenAI est sélectionné.", + "439e91879e": "Transcription OpenAI" + }, + "OpenAiTranscriptionSettingsRow": { + "85c589cd61": "Ajouter une clé API", + "ae2df8f511": "Déconnecter la clé API OpenAI", + "a622bc3b37": "Remplacer la clé", + "3b0ab3fc0b": "Connecté", + "27e0cb656d": "Transcription OpenAI", + "893790e13b": "Ajoutez une clé API OpenAI avant de sélectionner des modèles de transcription vocale cloud.", + "b59b9b2b51": "Clé API configurée pour les modèles de transcription vocale cloud." + }, + "OpenInMenuSetting": { + "03b00b1f64": "App personnalisée", + "c1d817e027": "Ajoutée", + "e4064916aa": "Ajouter une app", + "9d0413817d": "Choisissez les applications proposées dans le menu « Ouvrir dans » d'un espace de travail.", + "6ed52fe71e": "Applications « Ouvrir dans »", + "eb55b87570": "La commande à saisir dans Terminal pour ouvrir cette app.", + "ba1422ee07": "Commande de terminal", + "e1fc0085c6": "Libellé du menu", + "a261931d29": "Supprimer l'app", + "af7d1c3656": "Modifier l'app", + "494ed535cd": "Réduire les détails de l'app", + "810ef39b56": "cursor", + "3ebe650f74": "Nom de l'app", + "3743ed080c": "Définir la commande", + "f79084947b": "Nouvelle app" + }, + "OrchestrationPane": { + "52e0634e2c": "Demandez à un agent coordinateur d'utiliser l'orchestration pour les passations, les transferts de worktree et les agents enfants séquentiels ou parallèles.", + "ae79504732": "Comment l'utiliser", + "7bc082f4de": "Copier la commande d'installation", + "832f1f3ee6": "Vous préférez votre propre terminal ?", + "9bedd2a6e5": "Permet aux agents de transmettre le contexte et de coordonner le travail via Orca.", + "07641b9768": "Skill d'orchestration", + "2aacdb0517": "Coordonnez les agents de codage entre passations, transferts de worktree et travaux d'agents enfants.", + "191ac34567": "Orchestration d'agents" + }, + "OrchestrationSetupCard": { + "e7d2a5146c": "Permet aux agents de transmettre le contexte et de coordonner le travail via Orca.", + "2777ff0fdc": "Skill d'orchestration" + }, + "OrchestrationSkillAgentCoverage": { + "6dec5ce2d2": "Couverture des agents", + "ffe13e36fb": "Manquant", + "1e8f8d8fae": "Prêt", + "checking": "Vérification des agents installés et des chemins des skills…", + "noAgents": "Aucune CLI d'agent détectée sur le PATH. Installez des agents dans Paramètres → Agents, puis relancez la vérification.", + "fullCoverage_one": "L'unique agent détecté possède le skill.", + "fullCoverage_other": "Tous les {{value0}} agents détectés possèdent le skill.", + "noCoverage": "Installez le skill ci-dessus, puis relancez la vérification.", + "partialCoverage": "{{value0}} des {{value1}} agents détectés possèdent le skill." + }, + "OrchestrationSkillPromptDialog": { + "f08d45293d": "Copier la commande", + "35550f3b3b": "Terminé", + "1bdce1911e": "Copier la commande d'installation du skill d'orchestration", + "b99f375eb2": "Exécutez cette commande dans un terminal pour installer le skill d'orchestration pour vos agents.", + "2914abcfa2": "Installer le skill d'orchestration", + "d3dc559225": "Échec de la copie de la commande d'installation.", + "239bf9132b": "Commande d'installation copiée." + }, + "PrivacyDiagnosticBundleControls": { + "dc8404a930": "Créer le fichier de diagnostic", + "a5acaffdb6": "Abandonner", + "aca2c8a367": "Envoyer au support", + "798b6f0be5": "Ouvrir le fichier de révision", + "2ae9a6b63e": "Terminé", + "7f14a1733c": "Supprimer le fichier envoyé", + "2801d4ce22": "Copier l'ID de référence", + "d8be621237": "Ouvrez d'abord le fichier de révision.", + "61676df223": "Diagnostics envoyés. Communiquez cet ID de référence au support : {{value0}}.", + "fd7b3891af": "Vous avez ouvert le fichier de révision ({{value0}}). Envoyez ce fichier au support, ou abandonnez-le.", + "62340d4439": "Votre fichier de révision est prêt ({{value0}}). Ouvrez-le pour voir ce qui serait envoyé, puis choisissez de l'envoyer ou non au support.", + "19ec5e29b3": "Collecte l'activité récente de l'app et les erreurs dans un fichier expurgé que vous pouvez consulter avant l'envoi. Rien n'est téléversé tant que vous n'avez pas choisi de l'envoyer." + }, + "PrivacyDiagnosticsSection": { + "af2fc82cde": "Envoyer les diagnostics de l'app au support", + "c18cbe45df": "Diagnostics envoyés supprimés", + "7a4944595b": "Impossible de copier l'ID de référence", + "13eb2c65a1": "ID de référence copié", + "860bca9ec9": "Fichier de révision abandonné", + "49fc6c80e8": "Diagnostics envoyés", + "db3228e01a": "Fichier de révision ouvert", + "a2b3505c77": "Fichier de révision créé" + }, + "PrivacyPane": { + "36e0e2e63b": "variable d'environnement. Supprimez-la et redémarrez pour réactiver.", + "79a0f3c16c": "La télémétrie est désactivée par la", + "e3970bbbf5": "La télémétrie est désactivée car une variable d'environnement CI est définie. Supprimez-la et redémarrez.", + "fe904ac984": "Partager les données d'utilisation anonymes", + "77410e0566": "Politique de confidentialité", + "8bfdd23a88": "Aidez-nous à décider quoi développer ensuite. Orca envoie anonymement des comptages des fonctionnalités que vous utilisez et des endroits où ça casse.", + "afec8b03be": "ci" + }, + "QuickCommandsList": { + "noSearchMatches": "Aucune commande ne correspond à cette recherche." + }, + "QuickCommandsToolbar": { + "searchLabel": "Rechercher des commandes" + }, + "QuickCommandsPane": { + "8764c6e9e4": "Supprimer {{value0}}", + "7d90fd5299": "Modifier {{value0}}", + "8c877dec41": "Global", + "c6b155911b": "Toutes les commandes", + "5aacc8f7dc": "Ajouter une commande", + "c36912efd5": "Exécutez-les depuis le bouton Commandes rapides de la barre d'onglets, ou faites un clic droit dans n'importe quel terminal.", + "f91b649324": "Commandes enregistrées", + "3d9dc558e8": "Cette commande rapide sera retirée de votre liste enregistrée.", + "3edf3deaf8": "Supprimer « {{value0}} » ?", + "9fcfc29519": "Insérer", + "9b3e338d62": "Enter", + "4ccc63da87": "Agent", + "0252ddd578": "Aucun texte de commande", + "7784912ed6": "dépôt", + "2bb9e38e93": "Sans titre", + "3eb9897ab0": "Aucune commande dans les portées sélectionnées.", + "38d61927e6": "Aucune commande rapide enregistrée.", + "44923dd982": "destructive", + "ec1ed99e70": "Supprimer", + "d1d0976320": "Aucun", + "89f7e57fcc": "Enregistrée le", + "d59bd333c3": "Mettez à jour ce serveur Orca pour gérer ses commandes rapides.", + "f2bf411640": "Impossible de charger les commandes depuis cet hôte.", + "7ecfee5b8e": "Réessayer", + "601d6af51f": "Chargement des commandes…", + "923ba89646": "Impossible d'actualiser les commandes depuis cet hôte. Affichage des dernières commandes chargées.", + "8d525e5f15": "Copied", + "53b17a4b1b": "Impossible de copier", + "a9a564b7e7": "Copier {{value0}}", + "69a1441a21": "Rien à copier" + }, + "RecentTabOrderControl": { + "3b17c81ede": "Ordre de la barre d'onglets", + "6e6a3fcc61": "Plus récents", + "7a546f2309": "Ordre des onglets", + "a867a0889f": "Récents ou barre d'onglets." + }, + "RepositoryHooksSection": { + "af49e2a19e": "Le fichier est présent, mais Orca n'a pas trouvé de définitions valides de `scripts` ou d'`issueCommand`.", + "3397879bee": "et des commandes locales existent, choisissez celles qui s'exécutent.", + "39da2ae12f": "orca.yaml", + "ac9038d2cc": "Lorsque les deux", + "32fec28f5b": "Source des commandes", + "bbbd6e0bc4": "Source des commandes et orca.yaml", + "c9bc1bfd8f": "Avancé", + "610d90fdbd": "Détails sur la source des commandes et orca.yaml.", + "52aef29e69": "Laissez vide pour utiliser la valeur par défaut du dépôt définie dans", + "4084720f47": "Terminer {{artifact_url}}", + "13394103bd": "Commande d'issue GitHub personnalisée", + "70ad20f883": "pour l'URL de l'issue ou de la pull request liée.", + "b997331366": "Remplacement facultatif. Utilisez", + "2cc27dc12b": "Remplacement facultatif par utilisateur pour la commande d'issue liée.", + "b91a0f297d": "Scripts locaux et partagés exécutés avant l'archivage d'un worktree.", + "9a100323ff": "Script d'archivage", + "21fb607a87": "Comportement par défaut à la création d'un nouveau worktree.", + "793dcee97d": "Moment d'exécution", + "63e1783173": "Choisissez le comportement par défaut quand un script de setup est disponible.", + "fb6bebcf7e": "Quand exécuter le setup", + "30d555acd2": "Scripts locaux et partagés exécutés après la création d'un nouveau worktree.", + "52b31baf02": "Script de setup", + "8567127a40": "Scripts exécutés à la création ou à l'archivage des worktrees. Les scripts locaux sont stockés sur cette machine ; les scripts `orca.yaml` sont partagés avec votre équipe.", + "ff082fe7c6": "Hooks de worktree", + "5d940bde5c": "Ajouter un script local", + "8c2893fae0": "S'exécute comme un script shell unique. Enregistré sur cette machine.", + "40a446ae16": "- rien que pour vous, sur cette machine", + "2d03a514db": "local", + "7e4427b4a2": "pour changer.", + "b113344b6a": "Édition", + "f828e1de19": "- partagés avec votre équipe", + "673a7fd10e": "Vérification...", + "5426ecbdcb": "Les scripts locaux ne s'exécuteront pas", + "b2b06c7ce8": "Variables d'environnement disponibles (survolez pour les détails) :", + "95a0411b3e": "template", + "175daba180": "Exemple", + "b20c5df6ca": "Ajoutez un fichier `orca.yaml` pour activer des valeurs par défaut partagées de setup, d'archivage ou d'automatisation d'issues pour ce dépôt. Exemple de modèle :", + "56f9a4a1d0": "Utilisation de `orca.yaml`", + "623e0c9f31": "`orca.yaml` n'a pas pu être analysé", + "5a67e4793d": "Aucun `orca.yaml` détecté", + "07ba35bc68": "Vérifiez l'indentation sous `scripts:`. Les clés de hook doivent être indentées de deux espaces, et les lignes de commande de quatre.", + "787ca433ef": "Définissez uniquement les clés prises en charge : `scripts`, `setup`, `archive` et `issueCommand`.", + "ecc73d9125": "Comparez votre fichier au modèle fonctionnel ci-dessous et recopiez cette structure si besoin.", + "925f9e0dc4": "text-foreground", + "0cc712b823": "Le fichier de configuration principal existe à la racine du dépôt, mais Orca n'a pas encore pu analyser les définitions de hooks prises en charge.", + "c90b858573": "text-amber-700 dark:text-amber-300", + "aba825233f": "Le fichier contient des clés de configuration que cette version d'Orca ne reconnaît pas. Vous devrez peut-être mettre à jour Orca, ou vérifier le fichier pour écarter une faute de frappe.", + "ca424ff135": "Les hooks partagés et les valeurs par défaut d'automatisation d'issues sont définis dans le dépôt et accessibles à tous ceux qui l'utilisent.", + "32f417fe17": "text-emerald-700 dark:text-emerald-300", + "8bfe65fc60": "Utiliser les commandes locales", + "8d6c56bff8": "Exécuter les deux", + "0fa21e19ec": "Nom de l'espace de travail, généralement basé sur le nom de branche.", + "54c73d88d0": "Chemin du worktree en cours de création. Les commandes de setup s'exécutent depuis ce répertoire.", + "30952c4aa4": "Chemin vers le checkout principal du dépôt. Utile pour copier des fichiers partagés, comme .env, dans un worktree.", + "9b821fa19d": "# ex. echo \"Nettoyage de $ORCA_WORKSPACE_NAME\"", + "6f90ebe3fd": "S'exécute avant qu'un worktree soit archivé ou supprimé.", + "a3fc966677": "# ex. pnpm install cp \"$ORCA_ROOT_PATH/.env\" \"$ORCA_WORKTREE_PATH/.env\"", + "f0710e1c83": "S'exécute après la création d'un nouveau worktree : installer les dépendances, copier les fichiers env, lancer les migrations.", + "8561b0665f": "d'abord orca.yaml, puis vos commandes locales.", + "0e8b2a520d": "Ignore orca.yaml ; exécute uniquement vos commandes locales.", + "83dc78202a": "Local uniquement", + "29397e8bbc": "Exécute uniquement les commandes commises du dépôt ; ignore les commandes locales.", + "d88b6ff88f": "orca.yaml uniquement", + "99e3264a49": "N'exécute le setup que si vous le choisissez.", + "15debc1fd9": "Ignorer par défaut", + "022ba10cf2": "Exécute le setup automatiquement.", + "d3ef1ab247": "Exécuter par défaut", + "90b1f50137": "Demander confirmation avant d'exécuter le setup.", + "e03d9a8f38": "Demander à chaque fois", + "8dbe6bedf5": "invalide", + "0e0dd5b9a5": "chargé", + "9b12f15b1e": "lorsqu'il en existe un.", + "c85c2c88a2": "{{artifact_url}}", + "fac13f8c1e": "faisant foi", + "0518758f38": "les deux", + "d2b3016c20": "partagé", + "4611b78617": "source des commandes", + "c5a55a2d2e": "avancé", + "b5e3e77e89": "action", + "0ce113fd7b": "Les scripts locaux sont enregistrés, mais la source des scripts est réglée sur orca.yaml uniquement.", + "7f78e5eea6": "Les scripts locaux sont enregistrés. Orca analyse encore orca.yaml avant de pouvoir recommander la source de scripts à utiliser.", + "2b6356e744": "Enregistré", + "81057d5f71": "Enregistrement...", + "da37d6f10e": "Copier", + "3149964b66": "Copied", + "waitForSetupBeforeAgent": "Attendre la fin du setup avant de démarrer l'agent", + "waitForSetupBeforeAgentHelp": "Activez cette option quand le setup installe des dépendances, des serveurs MCP ou des fichiers de configuration dont l'agent a besoin au démarrage." + }, + "RepositoryIconPicker": { + "2b7d27b93c": "Utiliser la couleur de dépôt {{value0}}", + "fde066a63b": "Les PNG téléversés doivent peser 256 Ko au maximum.", + "cc1286e263": "Favicon", + "03ca1a4e9b": "example.com", + "381b4844fd": "Téléverser un PNG", + "7da623abcc": "Utilisé par défaut — GitHub en fournit toujours un, même quand le propriétaire n'a pas défini d'image personnalisée.", + "39da8a10bf": "Utiliser l'avatar GitHub", + "c490787d24": "Émoji", + "b2d7fd2116": "Icône", + "2d8bd302fa": "Avatar", + "913c55833d": "Couleur de dépôt personnalisée {{value0}}", + "0e5f0693c1": "Choisir une couleur de dépôt personnalisée", + "642dc29c6d": "Couleur", + "549d126081": "Réinitialiser", + "4e2a14f967": "Icône du dépôt", + "d71df44587": "Échec de la résolution du dépôt GitHub.", + "f79972271a": "Aucun remote GitHub trouvé pour ce dépôt.", + "4d039317f4": "Favicon du site web", + "acf31559a0": "Saisissez une URL de site web valide.", + "868c5c9b56": "Échec de l'importation de l'icône du dépôt", + "emojiTooLongForRepoIcon": "Cet émoji ne peut pas servir d'icône de dépôt.", + "currentEmojiSelection": "Actuel : {{value0}}", + "searchEmojiPlaceholder": "Rechercher un émoji" + }, + "RepositoryPane": { + "15a99d9b9f": "Les chemins relatifs sont résolus depuis la racine de ce projet.", + "8ccacbeb5a": "Utiliser le réglage global", + "e9bd57a336": "Emplacement des worktrees", + "e63bb96a9b": "Répertoire spécifique au projet pour les nouveaux worktrees.", + "f88db4fece": "Base de worktree par défaut", + "8984d06520": "Branche ou ref de base par défaut à la création des worktrees.", + "e641c359de": "Icône et couleur du projet utilisées dans la barre latérale et les onglets.", + "26fef02bf3": "Icône du projet", + "c7ef4415de": "Nom d'affichage", + "b0a0c14a1c": "Détails d'affichage propres au projet pour la barre latérale et les onglets.", + "removeProjectAllHosts": "Retirer ce projet d'Orca sur tous les hôtes configurés.", + "0909e5d650": "Retirer le projet", + "ee5a290616": "Ouvert en tant que dossier. Les fonctions Git sont indisponibles pour cet espace de travail.", + "323debba71": "Type :", + "availableHosts": "Hôtes disponibles", + "availableHostsDescription": "Hôtes où ce projet est configuré.", + "availableHostsHelp": "Les chemins de projet et les réglages de worktree sont propres à chaque hôte ; la création d'un espace de travail peut cibler toute configuration prête.", + "viewingHost": "Hôte affiché", + "currentSetup": "Actuel", + "hostSetupStateReady": "Prêt", + "hostSetupStateNotSetUp": "Non configuré", + "hostSetupStateSettingUp": "Configuration en cours", + "hostSetupStateError": "Erreur", + "hostSetupStateUnsupported": "Non pris en charge", + "setupPathPending": "Chemin en attente", + "openSetup": "Ouvert", + "removeSetup": "Supprimer", + "hostSetupBlockedVersion": "Version du serveur Orca incompatible", + "hostSetupMissingCapability": "Mettez à jour Orca sur cet hôte pour configurer des projets", + "hostSetupConnectionRequired": "Connectez cet hôte avant d'importer ou de cloner le projet", + "setupProjectOnHost": "Configurer sur un autre hôte", + "setupProjectOnHostHelp": "Choisissez un hôte, puis importez un checkout existant, clonez-y le dépôt, ou suivez une configuration qui sera provisionnée plus tard.", + "setupExistingFolder": "Importer un dossier existant", + "setupExistingFolderHelp": "Rendez ce projet disponible sur un autre hôte en reliant un checkout qui y existe déjà.", + "setupExistingFolderPathPlaceholder": "/path/to/project/on/host", + "cloneUrlPlaceholder": "URL du dépôt", + "cloneDestinationPlaceholder": "/destination/on/host", + "setupKindGit": "Dépôt Git", + "setupKindFolder": "Dossier", + "settingUpHost": "Import...", + "setupHost": "Importer", + "cloningHost": "Clonage...", + "cloneHost": "Cloner", + "creatingPendingSetup": "Création...", + "createPendingSetup": "Suivre la configuration", + "499a437335": "Identité", + "hostSetupCheckingCapability": "Vérification des capacités de l'hôte", + "hostAvailability": "Disponibilité de l'hôte", + "hostAvailabilityHelp": "Ajoutez ce même projet sur un autre hôte connecté.", + "addToAnotherHost": "Ajouter à un autre hôte", + "addProjectHost": "Ajouter le projet à l'hôte", + "addProjectHostHelp": "Choisissez où ce projet doit aussi être disponible.", + "closeHostSetup": "Fermer", + "setupHostLabel": "Hôte", + "browseFolder": "Parcourir le dossier", + "browseFolderHelp": "Utilisez un checkout ou un dossier existant sur cet hôte.", + "otherWaysToAdd": "Autres moyens d'ajout", + "cloneFromUrl": "Cloner depuis une URL", + "cloneFromUrlHelp": "Clonez ce dépôt sur l'hôte sélectionné.", + "addPlannedHost": "Placeholder pour l'ajout d'hôte", + "addPlannedHostHelp": "Mémorisez cet hôte et terminez l'ajout du projet plus tard.", + "existingFolder": "Dossier existant", + "addPlannedHostToHost": "Ajouter {{host}}", + "addPlannedHostConfirm": "Ceci enregistre seulement que le projet doit être disponible sur cet hôte. Vous pourrez ajouter le dossier ou cloner plus tard.", + "projectRuntime": "Runtime des projets", + "projectRuntimeDescription": "Choisissez si ce projet s'exécute sous Windows ou WSL.", + "hostStateDisconnected": "Déconnecté", + "hostStateUnknown": "Inconnu", + "nestedHostLabel": "{{value0}} via {{value1}}", + "hostStateWorkspaceWindowClosed": "Fenêtre de l'espace de travail fermée", + "hostWorkspaceWindowClosedHelp": "Le serveur est joignable mais sa fenêtre Orca est fermée. Ouvrez Orca sur {{value0}} pour utiliser cette configuration.", + "externalWorktrees": "Worktrees externes", + "externalWorktreesDescription": "Définir si les worktrees créés hors d'Orca apparaissent pour ce projet.", + "externalWorktreesInherited": "Utilise le réglage global : {{value0}}", + "show": "Afficher", + "hide": "Masquer", + "externalWorktreesGlobal": "Réglage global par défaut : {{value0}}", + "useGlobal": "Utiliser la valeur globale" + }, + "RepositorySourceControlAiActionRows": { + "548a6e1281": "Modèle de commande", + "7a3a8e431d": "Arguments CLI", + "2b2f38652b": "Commande personnalisée", + "0ffb081b3a": "Utiliser l'agent par défaut", + "f4310cf63f": "Agent", + "1cd88d470a": "Personnaliser", + "403876bb48": "Utiliser la valeur globale", + "f0aa2cfaea": "Recettes d'action" + }, + "RepositorySourceControlAiCustomCommand": { + "0704dd55cd": "Commande du dépôt", + "e56668c291": "Utiliser la valeur globale", + "fbb77e122a": "Repli du dépôt pour les actions de texte qui sélectionnent Commande personnalisée.", + "ebffc5a28c": "Commande personnalisée", + "f9941f0caf": "par ex. ollama run llama3.1 {prompt}" + }, + "RepositorySourceControlAiEnablement": { + "84233d1bb3": "Désactivé", + "bea897eec2": "Activé", + "62511a575d": "Utiliser la valeur globale", + "30ae6dcce8": "La valeur globale par défaut est", + "cf5959c834": "IA du contrôle de code source activée", + "show": "Afficher", + "hide": "Masquer", + "showActionsLabel": "Afficher les actions Source Control AI", + "visibilityHelper": "Détermine si les boutons IA du contrôle de code source sont affichés pour ce dépôt. La génération utilisée par des fonctionnalités distinctes suit les réglages de ces fonctionnalités. Réglage global par défaut : {{value0}}." + }, + "RepositorySourceControlAiHostedReviewDefaults": { + "053ccfbf52": "Désactivé", + "777443bf89": "Activé", + "ffc3b26b26": "Utiliser la valeur globale", + "a68849a859": "La valeur globale par défaut est", + "aa6ee4b7d6": "Valeurs par défaut de création de hosted review", + "629ed8a9d3": "Ouvrir la hosted review après création", + "14f1eb99d0": "Générer les détails à l'ouverture de Create PR", + "d32b87e754": "Utiliser le template de review quand disponible", + "981eae7e14": "Brouillon par défaut" + }, + "RepositorySourceControlAiSection": { + "67b3ff5467": "Abandonner", + "8b8bc5913a": "Recettes d'actions du dépôt. Les réglages globaux s'appliquent tant que ce dépôt ne les personnalise pas.", + "71b003b62b": "IA du contrôle de code source", + "152268c295": "Enregistrer", + "57e6e9d4b1": "Enregistrement...", + "ccb07dd027": "Enregistré", + "e57dde9d93": "Modifications non enregistrées" + }, + "RuntimeAccessGrantList": { + "8b82879581": "Toute personne disposant d'une autorisation active peut se connecter jusqu'à sa révocation. Révoquer l'accès partagé déconnecte immédiatement les clients actifs.", + "68ec21309f": "Révoquer l'accès", + "6f6d5188ed": "Révoquer {{value0}}", + "87b16cd11d": "Créé", + "434e4a6af6": "Lien actuel", + "fd83b94095": "Aucun accès serveur partagé pour le moment.", + "27cf8507ad": "Actualiser l'accès partagé", + "f031182867": "Accès serveur partagé", + "df142657a5": "Pas encore utilisé", + "b18d1764ef": "Dernière utilisation {{value0}}" + }, + "RuntimeEnvironmentsPane": { + "aeb26635d2": "Supprimer {{value0}}", + "af53761f31": "Annuler", + "bb90dd6487": "Supprimer le serveur", + "d2e00809e4": "Basculer", + "05e0fc3ebf": "Basculer vers", + "b2290ed203": "Orca mettra cet hôte au premier plan et chargera ses projets. Les terminaux et onglets de navigateur existants sur les autres hôtes restent actifs.", + "d570c35a99": "Changer de serveur", + "f3a3d6d834": "Capacités de {{value0}}", + "0ef838094a": "Protocole {{value0}}", + "9a91c4a0eb": "Compatible", + "86ed75bec8": "Mettre à jour le serveur", + "62ac182a27": "Mettre à jour le client", + "c8791efc45": "Statut indisponible", + "5120beaac6": "Vérification…", + "84b9b2be05": "Créez une autorisation d'accès révocable pour qu'un navigateur ou un autre client Orca puisse se connecter.", + "6e1280ca55": "Partager ce serveur Orca", + "9a3758d983": "Aucun serveur enregistré.", + "9bee6bbeeb": "Ajouter un serveur", + "55fcc964cd": "sur le serveur et collez l'URL d'appairage affichée.", + "960e901ae4": "orca serve --pairing-address ", + "163671f7b5": "Exécution", + "c3d772c514": "orca://pair?code=...", + "9bc9b83474": "Code d'appairage", + "e038625857": "Machine de dev", + "54ebacc600": "Nom du serveur", + "1826bd0608": "Serveurs enregistrés", + "6ce4664003": "Actualiser les serveurs", + "b07070ed3c": "Aucun serveur connecté", + "78692becbd": "Bureau local", + "64b6bea541": "Serveur actif", + "99ac81fb43": "Passé à {{value0}}.", + "b5b5114cb0": "{{value0}} supprimé.", + "6cb6eae14f": "Échec de l'enregistrement de l'environnement d'exécution.", + "7b5986c8df": "{{value0}} enregistré. Utilisez « Serveur actif » pour basculer quand vous le souhaitez.", + "5ef712f407": "Un serveur nommé « {{value0}} » existe déjà.", + "0c55a47480": "Le nom et le code d'appairage sont requis.", + "e6410d72c3": "Échec du chargement des environnements d'exécution.", + "6ef71985da": "Aucun endpoint", + "ed3e3f069d": "Ceci supprime le serveur enregistré d'Orca. Cela ne change pas le serveur actif.", + "b2fda48c39": "Supprimer le serveur actif déconnecte ce navigateur de cet hôte. Les sessions existantes de l'hôte ne sont pas touchées.", + "9f7665a01b": "Supprimer le serveur actif fait d'abord repasser Orca sur Bureau local. Les sessions existantes de l'hôte ne sont pas touchées.", + "3595fd1948": "Nouveau lien", + "54dee18f5c": "Masquer le formulaire", + "8cf8790697": "Les serveurs enregistrés acheminent ce navigateur via un runtime Orca appairé.", + "f75ce1c7a5": "Local conserve le comportement bureau actuel. Les serveurs enregistrés acheminent les appels client pris en charge via le runtime distant.", + "d25f0688b1": "Supprimer", + "4b5c6d7e8f": "Aucune capacité signalée", + "hostModelCapabilityUnknown": "Prise en charge des modèles côté hôte : vérification des capacités du serveur", + "hostModelCapabilitySupported": "Prise en charge des modèles côté hôte : prête", + "hostModelCapabilityMissing": "Prise en charge des modèles côté hôte : mettre à jour le serveur pour {{value0}}", + "hostModelCapabilityProjectSetup": "configuration de projet", + "hostModelCapabilityTaskSourceContext": "contexte de source de tâche", + "hostModelCapabilityWorkspaceRunContext": "contexte d'exécution de l'espace de travail", + "3f67e8078a": "Utilise cet ordinateur par défaut. Ne choisissez un serveur enregistré que si vous voulez faire passer projets, fichiers, terminaux, vérifications de fournisseurs et passation navigateur/mobile pris en charge par ce serveur.", + "2c85efb3e8": "Sélectionner un serveur enregistré fait de ce runtime Orca appairé l'Hôte par défaut de ce navigateur.", + "serverConnected": "Connecté", + "serverChecking": "Vérification…", + "serverDisconnected": "Déconnecté", + "disconnectedServer": "Déconnecté de {{value0}}.", + "connectToRemoteServers": "Se connecter aux serveurs distants", + "connectToRemoteServersHelp": "Appairez un autre runtime Orca, puis connectez-le ou déconnectez-le ici.", + "activeServerRowHelp": "Serveur actif pour les projets, terminaux et vérifications de fournisseurs routés via le serveur.", + "disconnect": "Déconnecter", + "connect": "Se connecter", + "advanced": "Avancé", + "serverDetails": "Détails du serveur", + "advertiseThisApp": "Annoncer cette app comme serveur", + "advertiseThisAppHelp": "Créez des liens d'accès pour que des navigateurs, des clients mobiles ou un autre client Orca se connectent à cette app en cours d'exécution.", + "runtimeReachable": "{{value0}} est joignable.", + "updateAvailableOne": "1 mise à jour disponible", + "updatesAvailable": "{{value0}} mises à jour disponibles", + "versionUnavailable": "Version d'Orca indisponible", + "updateServer": "Mettre à jour", + "reviewServerUpdates": "Rechercher des mises à jour du serveur", + "updatingServers": "Mise à jour des serveurs…", + "orcaVersion": "Orca v{{value0}}", + "removeActiveServerBlocked": "Choisissez un autre Serveur actif dans Avancé avant de supprimer ce serveur.", + "removeActiveServerDescription": "Choisissez un autre Serveur actif dans Avancé avant de supprimer ce serveur. Les sessions existantes de l'hôte ne sont pas touchées.", + "workflow": "Flux de travail avec serveur distant", + "connectWorkflow": "Se connecter à un hôte", + "connectWorkflowHelp": "Cette app rejoint une autre machine", + "shareWorkflow": "Partager cet hôte", + "shareWorkflowHelp": "D'autres appareils rejoignent cette machine", + "troubleshootWorkflow": "Dépannage de la connexion", + "sshTunnelRequired": "Tunnel SSH requis", + "troubleshootTitle": "Créez un nouveau lien sur l'autre hôte", + "troubleshootDescription": "Un lien qui utilise 127.0.0.1 pointe vers l'appareil qui l'ouvre, pas vers l'ordinateur qui l'a créé.", + "troubleshootStepShare": "Sur l'autre ordinateur, ouvrez « Partager cet hôte ».", + "troubleshootStepAddress": "Choisissez « Autre appareil » et sélectionnez son adresse Tailscale ou LAN.", + "troubleshootStepRegenerate": "Générez un nouveau lien d'accès et utilisez ici uniquement le lien le plus récent.", + "troubleshootTunnel": "Vous utilisez une redirection locale SSH ? Revenez à « Se connecter à un hôte », collez le lien loopback, puis activez « J'utilise un tunnel SSH » sous « Avancé ».", + "cloudVmWorkflow": "VM cloud", + "cloudVmWorkflowHelp": "Gérer les machines cloud créées par recette" + }, + "RuntimePairingGeneratedUrlRows": { + "0495f68959": "Copier {{value0}}" + }, + "RuntimePairingUrlGenerator": { + "849825e829": "Collez cette URL d'appairage dans un autre client Orca.", + "2e5c4e3c93": "Appairer un autre client Orca", + "f7cafdc9f3": "Lien navigateur indisponible dans cette version. L'URL d'appairage reste fonctionnelle pour les clients Orca.", + "6b9ca3e69b": "Ouvrir dans le navigateur", + "1ca2e5194d": "Utilisez cette URL depuis un navigateur capable de joindre l'adresse sélectionnée.", + "8de0f84fff": "Générer un lien d'accès", + "279e0dcb57": "127.0.0.1 ne fonctionne que sur cet ordinateur. Utilisez une adresse LAN, Tailscale ou personnalisée pour un autre appareil.", + "45cf476df3": "hôte, hôte:port ou wss://host/path", + "4531ea3158": "Adresse personnalisée", + "360c548cf3": "Actualiser les adresses de connexion", + "de6d5cff95": "Cet ordinateur (", + "de77eb1b65": "Adresse de connexion", + "ff80904fc4": "Créez un droit d'accès révocable pour les clients navigateur ou bureau.", + "f8500e134a": "Partager ce serveur Orca", + "d6c081adf4": "Échec de la copie de l'URL.", + "df0aa45a86": "URL d'appairage copiée.", + "13704d635e": "URL du client web copiée.", + "e8d83f2b0f": "Échec de la révocation de l'accès partagé.", + "9f8e037c4a": "Accès partagé révoqué.", + "d797f516b1": "L'accès partagé était déjà révoqué.", + "2ed55c841a": "Échec de la génération de l'URL d'appairage.", + "11d5248e62": "URL d'appairage générée.", + "6dd594a507": "URL du client web générée.", + "2752126f3e": "L'appairage à l'exécution est indisponible.", + "95b8be4cea": "Échec de l'actualisation des interfaces réseau.", + "1b4e0bbcc5": "Échec du chargement des droits d'accès partagés.", + "b91e36a986": "web", + "custom-option": "{{address}} (personnalisée)", + "add-custom": "Ajouter une adresse personnalisée…", + "custom-title": "Adresse de connexion personnalisée", + "custom-description": "Annoncez une adresse joignable par un autre appareil — un hôte LAN ou Tailscale, ou une URL ws(s):// complète.", + "custom-hint": "Saisissez un hôte, un hôte:port ou une URL ws(s)://.", + "custom-cancel": "Annuler", + "custom-use": "Utiliser l'adresse", + "intentQuestion": "Où ce lien sera-t-il ouvert ?", + "anotherDevice": "Autre appareil", + "anotherDeviceHelp": "Tailscale, LAN ou autre adresse joignable", + "localOnly": "Cet ordinateur uniquement", + "localOnlyHelp": "Un navigateur ou un client Orca sur cet ordinateur", + "customAddress": "Adresse personnalisée", + "customAddressHelp": "Tunnel SSH, proxy inverse ou nom d'hôte personnalisé", + "recommended": "Recommandé", + "localLink": "Lien local uniquement", + "localLinkHelp": "Ce lien ne fonctionne que dans un navigateur ou un client Orca exécuté sur cet ordinateur.", + "noExternalAddress": "Aucune adresse pour un autre appareil n'a été trouvée. Connectez cet ordinateur à un LAN ou à Tailscale, actualisez, ou choisissez « Adresse personnalisée ».", + "staleAddress": "L'adresse de connexion a changé. Générez un nouveau lien pour {{address}}.", + "customInvalid": "Saisissez un hôte, un hôte:port, une adresse IPv6 ou une URL ws(s):// valide." + }, + "Settings": { + "3bf149e873": "Paramètres du projet > {{value0}}", + "075341c763": "Nouvelles fonctionnalités encore en cours de construction. Essayez-les.", + "8b017f2506": "Expérimental", + "499c1cd7f9": "Paramètres de compatibilité bas niveau pour le dépannage.", + "1c87f8d024": "Avancé", + "c1b43dc4e2": "Données d'utilisation anonymes et contrôles de télémétrie.", + "d7e3f62d70": "Confidentialité & télémétrie", + "9b83cc62c2": "Accès de confidentialité macOS pour les outils de développement lancés depuis le terminal.", + "65660d4548": "Autorisations macOS", + "c6c01ac209": "Contrôlez les terminaux et les agents depuis votre téléphone.", + "c40dadaac8": "Mobile", + "c2ee313198": "Utilisez des machines existantes via SSH pour les fichiers, les terminaux, Git et les espaces de travail.", + "9b02492d1f": "Hôtes SSH", + "b5ee17826b": "Appairez des runtimes Orca distants pour des sessions persistantes, un état distant plus riche et un transfert web ou mobile.", + "7686cb5c36": "Connectez ce navigateur à un serveur Orca enregistré.", + "bd0181eeca": "Serveurs Orca distants", + "8acf3f22e0": "Statistiques Orca, plus analyses de tokens Claude, Codex, OpenCode et utilisation d'abonnement Grok.", + "954a8f5aef": "Statistiques & usage", + "a737a4bb22": "Raccourcis clavier pour les actions courantes.", + "23bf7a1ad4": "Raccourcis", + "7210ac09c4": "Notifications natives du bureau pour l'activité des agents et les événements du terminal.", + "9907545fa3": "Notifications", + "d0b7021d64": "Comportement de sélection et d'édition.", + "d7a3e635b6": "Saisie & édition", + "6d1a27e193": "Thème, zoom, apparence de l'app et du terminal, barres latérales et barre d'état.", + "2b4474780a": "Apparence", + "3d9adfe6a5": "Onglets globaux de terminal, de navigateur et de markdown.", + "3eb22a3ada": "Espace de travail flottant", + "01f9d36292": "Configurez la prise en charge des émulateurs mobiles pour Orca et les agents de codage.", + "f75daf1002": "Émulateur mobile", + "ad9788036f": "Page d'accueil, routage des liens et cookies de session.", + "c46215ea03": "Navigateur", + "6742c7932c": "Commandes de terminal enregistrées, globales ou par projet.", + "13d4fe30ad": "Commandes rapides", + "b79b5b31e9": "Shells, moteur de rendu, sessions et comportement du terminal.", + "3de4bbb841": "Terminal", + "dd72ed437a": "Choisissez les fournisseurs de tâches affichés dans la page Tâches et la barre latérale.", + "11faa2f7dd": "Sources de tâches", + "cfa34f4465": "Nommage des branches, refs de base et Git AI Author.", + "70100f94c7": "Git & gestion de code source", + "b07041697f": "Connectez GitHub, GitLab, Linear et les services d'hébergement de sources.", + "c9ca101a3b": "Intégrations", + "f9b77539fd": "Valeurs par défaut des espaces de travail, configuration de l'app et maintenance.", + "7807c11c4d": "Général", + "6855b0f77d": "Terminez les workflows essentiels qui rendent Orca utile au travail parallèle avec des agents.", + "6d119427ef": "Checklist d'intégration", + "eb1176a14e": "Dictée vocale locale avec modèles embarqués sur l'appareil.", + "5063bb47a5": "Voix", + "7118953f14": "Permettez aux agents de contrôler n'importe quelle app de votre ordinateur.", + "c9841721cb": "Computer Use", + "475980f53d": "Coordonnez plusieurs agents de codage via Orca.", + "00c3a7950d": "Orchestration", + "21f09426ea": "Facultatif. Orca fonctionne avec vos connexions de fournisseurs existantes ; ajoutez des comptes uniquement si vous voulez qu'Orca aide à basculer entre eux.", + "ad6c529693": "Comptes de fournisseurs d'IA", + "ec1ba547f7": "Gérez les agents IA, définissez-en un par défaut et personnalisez les commandes.", + "8afa676615": "Agents", + "add3b97ee6": "\"", + "3c88ec55d6": "Aucun paramètre trouvé pour \"", + "c7ad095d96": "Chargement des paramètres...", + "acc7bbdefd": "Appuyez à nouveau sur ESC pour quitter les paramètres", + "43b68e10f0": "Vous avez des modifications Git AI Author non enregistrées. Quitter les abandonnera.", + "17bdee4ff1": "Abandonner les modifications Git AI Author non enregistrées ?", + "084d8fac5b": "Confidentialité et sécurité", + "23931df7e8": "Hôtes distants", + "mobile_group": "Mobile", + "8bd117d669": "Interface", + "e1578cd4bc": "Workflows", + "9abb9be3bc": "Configurer", + "23c6874fdf": "Capacités IA", + "2309068a6f": "destructive", + "65358016ea": "Abandonner", + "dev": "Outils dev", + "devDescription": "Outils réservés au développement pour exercer les états de l'UI.", + "linearTitle": "Linear", + "linearDescription": "Fonctionnement de Linear dans Orca, checklist de configuration, skill d'agent et exemples de prompts.", + "tasksDescription": "Connectez des fournisseurs, installez le skill Linear et choisissez ce qui apparaît dans Tâches." + }, + "SettingsFormControls": { + "42a4d15a30": "Aucune police correspondante.", + "b55371ea18": "Polices", + "c766f8ac75": "Activer/désactiver les suggestions de polices", + "74bcecd5ec": "Effacer", + "a4ff6143f8": "Effacer la sélection de police", + "b661b034ec": "· Par défaut :", + "builtin_themes": "Intégrée", + "ceefb9d7f1": "Aucun thème trouvé.", + "imported_from": "Importé depuis {{value0}}", + "imported_themes": "Importé", + "9119fb2268": "Actuel", + "4e11f87ca6": "Affichage de", + "fbb428db98": "Sélectionné :", + "fac59213fc": "Rechercher parmi les thèmes intégrés", + "search_terminal_themes": "Rechercher des thèmes de terminal", + "cb330ef7f8": " sur {{value0}}", + "c822571b2e": " correspondant à \"{{value0}}\"", + "3119c012a5": "string" + }, + "SettingsSidebar": { + "e0900f83e7": "SSH", + "5c9669ff9c": "Projets", + "dbceaa8840": "Rechercher dans les paramètres", + "60f8a673a7": "Retour à l'application", + "6503182299": "Checklist d'intégration", + "82db1b7de4": "Checklist d'intégration, {{value0}} sur {{value1}} terminés. Afficher le guide de configuration.", + "df38d612b7": "Aucun projet ajouté pour le moment.", + "3e483e256b": "Aucun paramètre de projet correspondant." + }, + "ShortcutFilterRail": { + "28b63545bf": "Statut", + "8a1e78c14b": "Filtres d'état des raccourcis", + "df8466f3fc": "Effacer la recherche de raccourcis", + "f733c4b89f": "Rechercher une commande ou des touches", + "02dc7d4251": "Trouver des raccourcis", + "1d5634ba31": "Raccourci {{value0}}" + }, + "ShortcutRowsList": { + "4ce3cd24d9": "Aucun raccourci ne correspond à ces filtres." + }, + "ShortcutTerminalPolicyControl": { + "0762983d13": "Terminal en priorité", + "63308571d8": "Orca en priorité", + "c43c7ff5f9": "Décidez qui intercepte en premier les raccourcis", + "c3a554288e": "Raccourcis dans le terminal", + "0f55c6f15c": "Choisissez si Orca ou le terminal ciblé l'emporte lorsque des raccourcis se chevauchent." + }, + "ShortcutsPane": { + "4b7ae34062": "directement.", + "38e86e206a": "Personnalisez les raccourcis visuellement ou modifiez", + "47f8f7aef9": "Raccourcis clavier", + "f0b35b0b2e": "Désactivé lorsqu'un terminal ou une TUI a le focus clavier.", + "5c65d5db9d": "Terminal en priorité", + "dfa8ff612f": "S'exécute également lorsqu'un terminal ou une TUI a le focus clavier.", + "2a0e8aeccf": "Orca en priorité", + "3c0fac059a": "S'exécute quand même lorsqu'un terminal a le focus clavier.", + "25b0004fbf": "Terminal actif", + "781cb74d22": "S'exécute depuis les volets de terminal.", + "cb02e00202": "Terminal", + "d8c988dab4": "~/.orca/keybindings.json", + "shortcutUnavailable": "Le raccourci n'est plus disponible." + }, + "SourceControlAiActionRecipeDefaults": { + "2576299196": "args", + "b3914ecbbc": "Abandonner", + "fb09da4345": "Modèle de commande", + "2cb4bb7e5d": "Arguments CLI", + "0740d30915": "Commande personnalisée", + "ee0e5c2a48": "Utiliser l'agent par défaut", + "bf84dea6af": "Utilisez les variables uniquement si vous voulez qu'Orca injecte du contexte. Laissez l'agent sur « Par défaut » pour suivre votre préférence d'agent habituelle.", + "a79c567194": "Recettes d'action", + "cf01d41bce": "Agent, arguments CLI et modèle de commande utilisés par chaque bouton IA de contrôle de code source.", + "a9359c8aa9": "Erreur inconnue", + "b5f46664d3": "Échec de l'enregistrement de l'action IA par défaut du contrôle de code source : {{value0}}", + "d18d665e12": "Enregistrer", + "4f549a5fa8": "Enregistrement...", + "9d3cc627f8": "Enregistré", + "817128d94e": "Modifications non enregistrées", + "7ab1437a12": "pull request", + "e5b24893ba": "commit", + "06a9dab64d": "checks", + "cb67b938c5": "fix", + "2037c78a6f": "template", + "eb7e8f3b39": "model", + "d74fdc776c": "command", + "673369fe0c": "cli", + "db9bd75d10": "arguments", + "926d58e87f": "agent" + }, + "SparsePresetSettingsSection": { + "6fa754d20f": "Supprimer", + "fe1f2c6572": "Modifier {{value0}}", + "88bfbf1a9c": "Aucun préréglage sparse enregistré pour ce dépôt.", + "d7565029a9": "Nouveau préréglage", + "17f8c4ce10": "Gérez les ensembles de répertoires enregistrés pour la création de worktrees sparse.", + "388513be2d": "Préréglages de sparse checkout", + "a05bc9183f": "Enregistrer le préréglage", + "2d7d45e991": "Annuler", + "c240a16f25": "Utilisez des chemins relatifs au dépôt, comme packages/web ou apps/api.", + "fde7ff2cc3": "packages/web shared/ui", + "caf33029cc": "Répertoires", + "3b6f1abd3e": "ex. web-only", + "a6fcdd9e3c": "Nom", + "b9922ec194": "Annuler la modification du préréglage", + "694cc55ecb": "Les répertoires enregistrés sont utilisés lors de la création de worktrees sparse pour ce dépôt.", + "8b64731aaf": "+{{value0}} autres", + "755c6a1a0d": "Confirmer", + "a7bcf206b1": "Suppression en cours", + "ba9ad2d4cd": "Date de mise à jour inconnue", + "568d7e1e49": "{{value0}} mis à jour", + "d7b3f0bdc3": "{{value0}} répertoires", + "9d3c087fc0": "1 répertoire", + "8deb7024ab": "Chargement des préréglages sparse...", + "92c08ccae3": "Impossible de charger les préréglages sparse.", + "3dfa765ca7": "{{value0}} répertoires seront enregistrés.", + "b532b9c17d": "1 répertoire sera enregistré.", + "623b4cf910": "Modifier le préréglage", + "68bbcd864a": "nouveau", + "2ef2b2674b": "Supprimer {{value0}}" + }, + "SshDestructiveActionDialog": { + "895b216267": "Annuler" + }, + "SshPane": { + "c0f1c80166": "Aucune cible SSH configurée.", + "639ceb3698": "Ajouter une cible", + "51d7dba44d": "Importer", + "a7d28dff81": "Ajoutez un hôte distant pour vous y connecter depuis Orca.", + "94c5284560": "Cibles", + "f495689b82": "Échec de l'importation", + "f8050f6307": "Synchronisation terminée : {{value0}} serveur{{value1}}", + "68c13b4589": "Échec du test", + "81d08bcddf": "Connexion réussie", + "2c4ee7332b": "Échec de la réinitialisation du relais distant", + "db2e48975e": "Relais distant réinitialisé", + "025e107643": "Échec de l'arrêt des terminaux distants", + "90e308c98b": "Terminaux distants arrêtés", + "a43de1d3ee": "Échec de la déconnexion", + "e95d5ae10e": "Échec de la connexion", + "c2a69510e3": "Échec de la suppression de la cible", + "a0237eb1ca": "Cible supprimée", + "2227ce47b6": "Échec de l'enregistrement de la cible", + "f602009125": "Cible ajoutée", + "b4ba0ce33d": "Cible mise à jour", + "3879cbaa52": "Le délai d'inactivité du terminal doit être compris entre 60 et {{value0}} secondes, ou maintenez les terminaux actifs jusqu'à la réinitialisation.", + "4db9afce1c": "Le port doit être compris entre 1 et 65535", + "0e5aa04161": "Un hôte ou un alias de config SSH est requis", + "f1fc50dad2": "Échec du chargement des cibles SSH", + "0cda732f43": "Échec du test de connexion" + }, + "SshPassphraseDialog": { + "d5a234456f": "Annuler", + "c3ce71aad6": "Saisissez la phrase secrète", + "abaa0dc653": "Saisissez le mot de passe", + "ce4fdf7914": "Saisissez la phrase secrète pour", + "dbf9b6f2d0": "Saisissez le mot de passe pour", + "c55f105262": "Échec de l'annulation de la demande d'identifiants SSH", + "b8e88fd0de": "Échec de l'envoi des identifiants SSH", + "405066423c": "Déverrouiller", + "bec2c1318f": "Se connecter", + "8a349e3fac": "Phrase secrète pour {{value0}}", + "cab3d5f5a5": "Mot de passe pour {{value0}}", + "1f3dde805d": "Phrase secrète de la clé SSH", + "106bd57f4a": "Mot de passe SSH" + }, + "SshTargetCard": { + "a883f5a00f": "délai d'inactivité du terminal : {{value0}}", + "8ce71262f4": "terminaux jusqu'à la réinitialisation", + "ec6543cee9": "Se connecter", + "0e53e9f8e8": "Tester", + "1810b51482": "Connexion", + "4c86f30877": "Déconnecter", + "7f7b3d7ab4": "Supprimer la cible", + "3d21a22d0e": "Suppression de la cible", + "3d8af2949f": "Modifier la cible", + "762a48c662": "Réinitialiser le relais distant", + "97dea4e8cf": "Réinitialisation du relais distant", + "da16e108e6": "Arrêter les terminaux distants", + "c77f1abfe3": "Arrêt des terminaux distants", + "18968ede9e": "Erreur", + "f0871e6bfb": "connect", + "47e94bd6ba": "connecté" + }, + "SshTargetDestructiveActions": { + "7e66942808": "Ceci arrêtera les sessions de terminal actives sur cette cible SSH. La reconnexion ne les restaurera pas.", + "accf177a03": "Arrêter les terminaux distants ?", + "26be00392d": "Ceci force l'arrêt du relais distant pour cette cible SSH. Les terminaux distants actifs et les redirections de port de cette cible seront terminés.", + "570a7a0574": "Réinitialiser le relais distant ?", + "3bb0cf0ee4": "Ceci supprimera la cible et arrêtera tous les terminaux distants actifs.", + "4808966c41": "Supprimer la cible SSH" + }, + "SshTargetForm": { + "fea9cb402e": "Annuler", + "1b19b00e93": "Les délais limités doivent être compris entre 60 secondes et 7 jours.", + "7c13f58c91": "Jusqu'à la réinitialisation", + "55c56cf2c7": "Délai après déconnexion (secondes)", + "137e88ce8d": "Les terminaux distants continuent de s'exécuter après la déconnexion d'Orca de cet hôte.", + "b574994adc": "Utilisez « Arrêter les terminaux distants » ou « Réinitialiser le relais » quand vous voulez les arrêter.", + "71fc546097": "Garder les terminaux actifs jusqu'à la réinitialisation", + "92f80edbfd": "Persistance des terminaux distants", + "feae1d1e69": "Facultatif. Équivalent à ProxyJump / ssh -J.", + "11bcb4507a": "bastion.example.com", + "b2ab248ded": "Hôte de rebond", + "3b01ca44a0": "Facultatif. Utilisé pour le tunneling (ex. Cloudflare Access, ProxyCommand).", + "f42d844544": "ex. cloudflared access ssh --hostname %h", + "c7d0e18ecb": "Commande de proxy", + "cb91f6375c": "Facultatif. L'agent SSH est utilisé par défaut.", + "d6a5f2ee5c": "~/.ssh/id_ed25519 (laisser vide pour l'agent SSH)", + "63c0c145c1": "Fichier d'identité", + "c94cfa634c": "Port", + "47e082bc17": "deploy", + "dc1dc52aaa": "Nom d'utilisateur", + "2ee9bcd2e8": "server, deploy@server:2222, ssh://server", + "ce370ce674": "Hôte ou alias *", + "b8dab0aa7b": "Mon serveur", + "298de87a88": "Label", + "9518545cb6": "Ajouter une cible", + "a62b4cb39a": "Enregistrer les modifications", + "29af933cd5": "Nouvelle cible SSH", + "f2331ce599": "Modifier la cible SSH", + "4a342f44c1": "Connexion avancée", + "e9609ddca6": "Proxy, hôte de rebond et réutilisation de la connexion", + "8c922dffba": "Réutiliser la connexion SSH pour une configuration plus rapide", + "53e9aabfc0": "Utilise le multiplexage OpenSSH quand il est disponible. Désactivez pour les hôtes soumis à des restrictions SSH particulières.", + "editTitle": "Modifier l'hôte SSH", + "addTitle": "Ajouter un hôte SSH", + "editDescription": "Mettez à jour les détails de connexion de cette machine. Les modifications s'appliquent à la prochaine connexion.", + "addDescription": "Ajoutez une machine permanente sur laquelle vous pouvez vous connecter en SSH.", + "editingPrefix": "Modification" + }, + "TasksPane": { + "f71d8a9dd3": "Fournisseurs de tâches", + "6b23a34f6d": "Jira", + "09ae2d7c51": "Linear", + "7c5d7fdc20": "GitLab", + "e14063e727": "GitHub", + "connectProviderTitle": "Connecter {{provider}}", + "connectCodeHostDescription": "Installez et authentifiez la CLI sous Intégrations pour qu'Orca puisse charger les issues.", + "connectionCheckUnavailable": "Orca n'a pas pu vérifier cette connexion. Réessayez, ou ouvrez Intégrations pour le détail de la configuration.", + "retryConnection": "Réessayer", + "openIntegrations": "Intégrations", + "connectInIntegrations": "Configurer dans Intégrations", + "connectJiraTitle": "Connecter Jira", + "connectJiraDescription": "Ajoutez un site Jira Cloud ou une instance auto-hébergée avec un jeton d'API ou un PAT.", + "manageJira": "Gérer les clés", + "addJira": "Ajouter un accès Jira", + "githubDescription": "Parcourez les issues GitHub et lancez des espaces de travail à partir de celles-ci.", + "gitlabDescription": "Parcourez les issues GitLab et lancez des espaces de travail à partir de celles-ci.", + "linearDescription": "Connectez Linear, installez le skill d'agent et affichez-le dans Tâches.", + "jiraDescription": "Connectez Jira Cloud ou Jira auto-hébergé et affichez-le dans Tâches.", + "setupTitle": "Configuration de la gestion des tâches", + "setupDescription": "Terminez connexion + visibilité pour chaque fournisseur au même endroit. Linear nécessite aussi le skill d'agent afin que les agents de codage puissent lire et mettre à jour les tickets. Au moins un fournisseur doit rester visible.", + "incompleteBannerTitle": "Certains fournisseurs visibles doivent encore être configurés", + "incompleteBannerBody": "Masquez les fournisseurs que vous n'utilisez pas, ou dépliez une carte et terminez ses étapes.", + "providersDescription": "Chaque carte décrit la connexion (et le skill pour Linear), ainsi que son affichage dans Tâches.", + "integrationsHint": "Les identifiants de tous les fournisseurs se trouvent aussi sous", + "integrationsLink": "Intégrations", + "skillHint": ". Une fois Linear connecté, des exemples d'utilisation restent disponibles sous Paramètres → Linear.", + "incompleteBannerBodyWithLinear": "Masquez les fournisseurs que vous n'utilisez pas, ou dépliez une carte et terminez ses étapes. Pour Linear : accès API, skill d'agent et Afficher dans Tâches." + }, + "TerminalAppearanceSection": { + "a14a427ae4": "Épaisseur de la ligne de séparation des volets.", + "f27a99978d": "Épaisseur du séparateur", + "db632cb50e": "Opacité appliquée aux volets qui ne sont pas actuellement actifs.", + "a6fdd6a3b1": "Opacité des volets inactifs", + "1b79379d4f": "Contrôle l'assombrissement des volets inactifs et l'épaisseur du séparateur.", + "e1a5c25555": "Volets de terminal", + "04cdf85dec": "Opacité du curseur du terminal.", + "b9f1804422": "Opacité du curseur", + "2de6b5a699": "Utilise la variante clignotante de la forme de curseur sélectionnée.", + "74736cc9b1": "Curseur clignotant", + "2e5aec3cf6": "Souligné", + "52854a5608": "Bloc", + "e070e8aeba": "Barre", + "db270cc9a9": "Forme du curseur", + "d455f2ef4f": "Apparence par défaut du curseur pour les volets de terminal Orca.", + "abcb4dd019": "Curseur du terminal", + "70beb1bbc7": "Aperçu", + "31f6e61085": "Les ligatures sont actuellement", + "870377082f": "Désactivé", + "84bd22f2cd": "Activé", + "bc9ff84d61": "Auto", + "be8da35e7f": "Ligatures de police", + "4b1f29598e": "Auto - désactivées pour \"{{value0}}\".", + "400e950ca5": "Auto - activées pour \"{{value0}}\".", + "04569feb07": "Toujours désactivées, même pour les polices qui en proposent.", + "7234abcd08": "Toujours activées. Les polices sans ligatures s'affichent simplement telles quelles.", + "7233d594bf": "Affiche les ligatures de programmation (ex. =>, !=, ===) pour les polices qui en proposent. \"Auto\" n'active les ligatures que pour les polices réputées pour leurs ligatures (Fira Code, JetBrains Mono, Cascadia Code, Iosevka, etc.).", + "bafc80efbc": "Contrôle le multiplicateur de hauteur de ligne du terminal.", + "c084eb7d4c": "Hauteur de ligne", + "36af8ad94c": "Contrôle la graisse de la police du texte du terminal.", + "4aae5db258": "Graisse de la police", + "f04b17a50e": "Famille de polices du terminal par défaut pour les nouveaux volets et les mises à jour en direct.", + "a408266e67": "Famille de polices", + "855a76343a": "Importer depuis Ghostty", + "711e589f18": "Typographie du terminal par défaut pour les nouveaux volets et les mises à jour en direct.", + "048aac8a64": "Typographie du terminal", + "4415beb958": "désactivé", + "4e7d41a9f0": "activé", + "e90afcc44f": "désactivé", + "16c471ee03": "activé", + "typographyAdvanced": "Typographie", + "dimUnfocusedPanes": "Assombrir les volets sans focus." + }, + "TerminalAdvancedTypographyControls": { + "f5fa1a08f1": "Graisse de la police en gras", + "0e0d1ee15a": "À ajuster indépendamment de Graisse de la police. Certaines polices associent plusieurs valeurs à un même dessin ; baissez la graisse ou choisissez une autre police si le gras semble inchangé." + }, + "TerminalFontSizeSetting": { + "9b5252c85a": "px", + "0f4c92e595": "Taille de police du terminal par défaut pour les nouveaux volets et les mises à jour en direct.", + "a4a352b1e9": "Taille de police" + }, + "TerminalPane": { + "4263e940e0": "L'appui sur la touche Yen JIS (¥) envoie un antislash (\\\\) à la place.", + "19f4935159": "Yen JIS (¥) vers antislash (\\\\)", + "1c337bef4a": "Contrôle si l'appui sur la touche Yen JIS (¥) envoie un antislash (\\\\) à la place.", + "3fe1c5bfe0": "Désactivé", + "c73d510938": "Droite", + "e7aec1fd60": "Gauche", + "badb1219fc": "Les deux", + "43c2ff7b0e": "Auto", + "0a10420e1a": "Option comme Alt", + "ce3aadf0b2": "La touche Option {{value0}} envoie Alt/Esc ; l'autre compose des caractères spéciaux.", + "b62373091a": "Les deux touches Option envoient des séquences Alt/Esc.", + "d8998bb328": "Option compose des caractères spéciaux selon votre disposition de clavier.", + "d21c493808": "Auto — détecté : {{value0}}.", + "2561d3fc1b": "Contrôle si la touche Option macOS envoie des séquences Alt/Esc ou compose des caractères.", + "96be03b8eb": "PowerShell 7+", + "d26174e1dd": "Windows PowerShell", + "fe20f79dd1": "Version de PowerShell", + "822f62ddcd": "Télécharger PowerShell 7+", + "a016ffbeed": "Auto utilise Windows PowerShell pour le moment et bascule vers PowerShell 7+ une fois installé.", + "5ed5c95344": "Choisissez entre Windows PowerShell et PowerShell 7+ pour les nouveaux volets de terminal.", + "3d88af864d": "Choisissez si l'option de shell PowerShell lance Windows PowerShell ou PowerShell 7+ pour les nouveaux volets de terminal.", + "8a956cc91e": "Caractères traités comme séparateurs de mots pour la sélection par double-clic.", + "4bebcc2b2c": "Séparateurs de mots", + "12e06178fa": "lignes", + "907b0b9d3e": "Personnalisé", + "5336c096af": "{{value0}} lignes", + "81d86b2dd2": "Lignes de terminal de bureau conservées pour les nouveaux volets et les volets ouverts.", + "9df53f7c14": "Lignes de scrollback", + "c3810b2b42": "Lignes de terminal de bureau conservées.", + "267d020745": "Défilement, séparateurs de mots et comportements de terminal propres à chaque plateforme.", + "5e5f06c82c": "Avancé", + "003df129fe": "Fractionner horizontalement", + "623e62df99": "Fractionner horizontalement", + "332e8a2872": "Fractionner verticalement", + "691ce810e0": "Fractionner verticalement", + "1158f8fd55": "Nouvel onglet", + "6c6a054a1c": "Exécuter dans un nouvel onglet", + "a9d47451d1": "« Nouvel onglet » ouvre la commande de configuration dans un onglet en arrière-plan intitulé « Configuration », sans voler le focus.", + "d23b43c5be": "Emplacement du script de configuration", + "34a0dfa06e": "Endroit où s'exécute le script de configuration du dépôt lors de la création d'un nouveau espace de travail.", + "21f8da2078": "Script de configuration de l'espace de travail", + "6e6480a7df": "Permet aux programmes du terminal (Zellij, tmux, Neovim, fzf, Grok, SSH) de copier vers le presse-papiers système.", + "3338dcf8c1": "Autoriser les écritures presse-papiers des TUI (OSC 52)", + "69c64a479c": "Permet à Zellij, tmux, Neovim, fzf et Grok de copier vers le presse-papiers système via le PTY (y compris via SSH).", + "4729c645fc": "Copier automatiquement les sélections du terminal vers le presse-papiers.", + "902f5dee1f": "Copie à la sélection", + "9129b7e805": "Survoler un volet de terminal l'active sans avoir à cliquer.", + "8eefeaa3da": "Le focus suit la souris", + "96fe15def8": "Comportement souris et presse-papiers des volets de terminal.", + "45721f3e67": "Interaction avec le terminal", + "9c0b1c1792": "Activé", + "c1fc9e9444": "Accélération GPU", + "e0996d141a": "Auto tente WebGL, avec repli DOM pour les moteurs de rendu non pris en charge ou risqués.", + "7eaccc1424": "WebGL est toujours tenté pour les volets de terminal.", + "fe4acf36c6": "WebGL désactivé ; moteur de rendu DOM pour une compatibilité maximale.", + "f07dfb4466": "Contrôle si le terminal utilise le rendu WebGL xterm.js. Auto tente WebGL lorsque le moteur est pris en charge, avec repli Linux conservateur pour les rendus logiciels ou GPU inconnus.", + "72bc9334a0": "Comportement du moteur de rendu du terminal pour les volets existants et les nouveaux volets.", + "2fba319f21": "Rendu", + "cc8c5ca224": "Défaut Windows", + "d78fc4fdef": "Chargement des distributions", + "219aaa59f4": "Distribution WSL", + "2503f1e86b": "Utilisée pour les nouveaux volets de terminal WSL et la détection des agents locaux quand l'espace de travail actif n'est pas déjà dans WSL.", + "5fe79a5e56": "Choisissez la distribution WSL utilisée par les nouveaux terminaux WSL et les analyses d'agents locaux.", + "b637dd57a7": "WSL", + "f61ac77f16": "Git Bash", + "0f1b8669e6": "Invite de commandes", + "eb7fc4d98a": "PowerShell", + "27e301f22c": "Shell par défaut", + "09bf02de9a": "Shell utilisé à l'ouverture d'un nouveau volet de terminal. Prend effet pour les nouveaux terminaux.", + "bd68f3170d": "Choisissez le shell par défaut pour les nouveaux volets de terminal sous Windows.", + "a55eee649f": "Shell par défaut des nouveaux volets de terminal sous Windows.", + "87e678a8af": "Shell Windows", + "05efc0bada": "true", + "348246b06f": "false", + "5936387ddd": "auto", + "adbafefe56": "custom", + "16753eea48": "Le clic droit colle le presse-papiers. Ctrl+clic droit ouvre le menu contextuel.", + "9c178cf8aa": "Clic droit pour coller", + "af0c3b6e39": "Le clic droit colle le presse-papiers dans le terminal. Utilisez Ctrl+clic droit pour ouvrir le menu contextuel.", + "29154326bb": "activé", + "ab20575a8a": "désactivé", + "ab3a1f9068": "wsl.exe", + "ask_before_closing_running_terminals_title": "Demander avant de fermer les terminaux en cours d'exécution", + "ask_before_closing_running_terminals_description": "Afficher une confirmation avant de fermer un terminal où tourne une commande ou un agent.", + "scrollSpeed": { + "title": "Vitesse de défilement", + "description": "Ajustez le défilement normal du terminal, le défilement rapide avec modificateur et la vitesse de molette des TUI plein écran.", + "helper": "Ajustez le ressenti de la molette dans le scrollback et dans les applications de terminal sensibles à la souris.", + "reset": "Réinitialiser", + "normal": "Normal", + "normalDescription": "Multiplicateur de molette du scrollback.", + "fast": "Rapide", + "fastDescription": "Multiplicateur supplémentaire lors du défilement avec une touche modificateur.", + "tui": "TUI", + "tuiDescription": "Rapports de molette discrets pour les applications de terminal plein écran." + } + }, + "TerminalSettingsPreview": { + "a63953a48a": "Prévisualiser le thème {{value0}}", + "2c248fcc27": "Prévisualiser le thème", + "f8931d407d": "Afficher le séparateur de volets dans l'aperçu", + "50419052fe": "Séparateur de volets", + "d06664e889": "sombre" + }, + "TerminalThemeSections": { + "db210115c5": "Aperçu du mode clair", + "5e0c24b5c8": "Contrôle la ligne de séparation entre volets en mode clair.", + "ec2e33ad80": "Couleur du séparateur en mode clair", + "d56af60e6f": "Choisissez le thème utilisé quand Orca est en mode clair.", + "8273bc75d7": "Thème clair", + "bc8e8a251a": "Aperçu du mode sombre", + "cbe56a0f79": "Contrôle la ligne de séparation entre volets en mode sombre.", + "b739d2abfe": "Couleur du séparateur en mode sombre", + "7add204bd5": "Choisissez le thème de terminal utilisé en mode sombre.", + "9499ad1dc4": "Thème sombre", + "f012172e21": "Choisissez le thème utilisé pour les volets de terminal en mode sombre.", + "catalog_title": "Thèmes du terminal", + "catalog_description": "Choisissez les thèmes de terminal et les couleurs de séparateur pour les modes sombre et clair.", + "target_title": "Mode de thème", + "target_label": "Mode de thème", + "target_aria": "Mode de thème du terminal", + "target_dark": "Sombre", + "target_light": "Clair", + "match_dark_mode": "Aligner sur le mode sombre", + "match_dark_mode_description": "Partage le thème de terminal sombre et la couleur de séparateur en mode clair.", + "light_preview_description": "Affiche l'apparence effective du terminal en mode clair.", + "dark_preview_description": "Affiche l'apparence effective du terminal en mode sombre." + }, + "TerminalWindowSection": { + "1705318506": "Couleur ANSI magenta", + "03c855d15f": "Réinitialiser toutes les surcharges de couleurs", + "63f8d9336e": "Surcharges de couleurs", + "e86e09b5c7": "Remplacer individuellement les couleurs du terminal.", + "1d1920dc8a": "Masquer le curseur de la souris pendant la saisie dans le terminal.", + "3530908ef9": "Masquer la souris pendant la saisie", + "1846f6ee6a": "Marge verticale autour de la grille du terminal, en pixels.", + "1afcc1d973": "Marge verticale", + "25e2f8e8e1": "Marge horizontale autour de la grille du terminal, en pixels.", + "36b8402015": "Marge horizontale", + "53ce336e15": "Redémarrez Orca pour appliquer la modification du flou de fenêtre.", + "c65bb9ce63": "Redémarrage requis", + "97950bb087": "Applique un flou d'arrière-plan à la fenêtre du terminal. Nécessite un redémarrage.", + "2b82242f43": "Flou de fenêtre", + "809f37738d": "Contrôle la transparence de l'arrière-plan du terminal. 1 est totalement opaque, 0 totalement transparent.", + "ea7b1a158e": "Opacité de l'arrière-plan", + "03acb60aa0": "Contrôle la transparence de l'arrière-plan du terminal.", + "00eaa6b881": "Réglages d'apparence et d'arrière-plan de la fenêtre.", + "b96ba13ed1": "Fenêtre", + "42e01a6055": "Couleur ANSI blanc vif", + "16948119cb": "Blanc vif", + "1601140f03": "Couleur ANSI cyan vif", + "f94adc4113": "Cyan vif", + "fe4d89ef85": "Couleur ANSI magenta vif", + "e56e7d6ea0": "Magenta vif", + "bef6c0f6bf": "Couleur ANSI bleu vif", + "66820332fa": "Bleu vif", + "e2ef5f4ab7": "Couleur ANSI jaune vif", + "936a326be3": "Jaune vif", + "0ffb02f921": "Couleur ANSI vert vif", + "7dafd57730": "Vert vif", + "667de68863": "Couleur ANSI rouge vif", + "32b1b6acd7": "Rouge vif", + "f30c492769": "Couleur ANSI noir vif", + "260d69ce9a": "Noir vif", + "1be593d3e8": "ANSI vif", + "28846b1ca6": "Couleur blanche ANSI", + "0cb4459fb8": "Blanc", + "bd4c759327": "Couleur cyan ANSI", + "fb8bb4eb1f": "Cyan", + "d5e92fcd94": "Magenta", + "9635a71c51": "Couleur bleue ANSI", + "292a4c7316": "Bleu", + "09c1c6b096": "Couleur jaune ANSI", + "bb516de873": "Jaune", + "8a673d4206": "Couleur verte ANSI", + "8f2092b315": "Vert", + "b41270f5ca": "Couleur rouge ANSI", + "3a78f30b50": "Rouge", + "cf4437a2f7": "Couleur noire ANSI", + "adfdee23cb": "Noir", + "68e9f07de0": "ANSI normal", + "862e463f7f": "Texte en gras", + "b2c0857c49": "Couleur du texte sélectionné", + "8b450b5305": "Premier plan de la sélection", + "74d8555f85": "Couleur d'arrière-plan du texte sélectionné", + "40c3cfd30a": "Arrière-plan de la sélection", + "7f4063076c": "Couleur du texte sous le curseur (curseur bloc)", + "a2d9f095a7": "Texte du curseur", + "cd0700762b": "Couleur du curseur", + "c9e1fdf42f": "Cursor", + "da64e8f4c1": "Couleur d'arrière-plan du terminal", + "cc1b2ffeb2": "Arrière-plan", + "026a0b8013": "Couleur du texte principal", + "79f6bfb76e": "Premier plan", + "cf37ff69f6": "Base", + "8abdab9f7c": "Redémarrer maintenant", + "907131d741": "Redémarrage…", + "605a35d600": "Non encore appliqué par le moteur de rendu du terminal — xterm.js n'a pas d'emplacement dédié aux couleurs en gras. Une valeur enregistrée est conservée." + }, + "UIZoomControl": { + "c2c64b24d0": "Réinitialiser" + }, + "VoicePane": { + "6fa734ed95": "Supprimer {{value0}}", + "68de13f72c": "Échec de la suppression du modèle.", + "1ba81c0ff0": "recommandé", + "cfde55c7b0": "Échec du téléchargement du modèle.", + "43fd4f454b": "Modèle vocal", + "7cf715f891": "est maintenu.", + "295d84b849": "une fois pour démarrer, à nouveau pour arrêter. Maintien : dictez tant que", + "ff9a680010": "Bascule : appuyez", + "ba4a900d1d": "Mode de dictée", + "0121960365": "Activer la dictée vocale", + "366e1b4f36": "pour dicter du texte dans n'importe quel volet actif.", + "4465596675": "Appuyez", + "62d2a84d31": "Échec de l'effacement de la clé API OpenAI", + "37aba8bb63": "Clé API OpenAI effacée", + "8572bbb537": "Échec de l'enregistrement de la clé API OpenAI", + "506df81ba6": "Clé API OpenAI enregistrée", + "91980ce124": "{{value0}} Mo", + "61a16c8141": "Extraction...", + "b6536a1d12": "extraction", + "8f4d2a51d7": "hors ligne", + "d504ab05f0": "streaming", + "fbe5990716": "Sélectionner un modèle", + "e24f7d43d2": "Sélectionnez un modèle vocal. Les modèles locaux fonctionnent hors ligne ; les modèles cloud nécessitent une clé API.", + "174da92062": "Maintien", + "118b3c2dee": "Bascule", + "901985625d": "bascule", + "ad5d036ecc": "Impossible de demander l'autorisation d'accès au micro. La dictée vocale n'a pas été activée.", + "f9a9cf6928": "L'autorisation d'accès au micro est requise avant d'activer la dictée vocale.", + "1eac933202": "Panneau Confidentialité et sécurité de macOS ouvert. Réactivez la dictée après avoir accordé l'accès.", + "cd9fe37556": "Autorisation d'accès au micro accordée" + }, + "WorktreeSymlinksSection": { + "1c1e35b219": "Supprimer {{value0}}", + "b814c618e2": "Chemins liés", + "31ebab5403": "Aucun chemin partagé configuré pour ce dépôt.", + "ea06227efa": "ajouté", + "b2429aeb31": "Ajouter", + "ab40b8a5f1": "Aucun résultat. Continuez à saisir pour ajouter un chemin personnalisé.", + "4cd2a4c077": "Saisissez un chemin (ex. .env ou node_modules)…", + "241325302c": "Ajouter un chemin", + "7ff265071d": "Lors de la création d'un nouveau worktree, chaque chemin listé ici est copié par clone APFS sur macOS quand c'est possible, sinon créé en lien symbolique depuis le checkout principal.", + "4755f120b6": "Chemins partagés des worktrees", + "b07ef5a8b6": "Chemins à matérialiser depuis le checkout principal vers les nouveaux worktrees.", + "d72ba8dc68": "{{value0}} chemins", + "9ea912d811": "1 chemin" + }, + "WslCliRegistration": { + "c6f6f89d7c": "Annuler", + "119fef6cd2": "Chemin cible :", + "1dbb0377d9": "Cible du lanceur existante :", + "554305956d": "Chemin de la commande :", + "9b6627522c": "Actualiser", + "ab6b022a5c": "Actualiser l'état de la CLI WSL", + "d9c6880dbd": "Commande shell WSL", + "52d990420e": "Échec de la suppression de `{{value0}}` de WSL.", + "89c7414cf5": "`{{value0}}` supprimé de WSL.", + "6f91ad1333": "Échec de l'enregistrement de `{{value0}}` dans WSL.", + "951536dda5": "`{{value0}}` enregistré dans WSL.", + "26b4b3b00f": "Échec du chargement de l'état de la CLI WSL.", + "290bfff3ab": "Enregistrer", + "f951f85196": "Supprimer", + "4c4a9178a3": "Enregistrement...", + "41a1480d3e": "installer", + "4598b18464": "Suppression...", + "7c3bb36706": "supprimer", + "7ee4e52b99": "Orca enregistrera {{value0}} pour que la commande fonctionne depuis les terminaux WSL.", + "d8216eb22e": "Cela supprime la commande shell WSL. Orca lui-même reste installé sous Windows.", + "e49688f67f": "Enregistrer `{{value0}}` dans WSL ?", + "61ac55278e": "Supprimer `{{value0}}` de WSL ?", + "e2b0ee267f": "périmé", + "7aa456a460": "Enregistre `orca-ide` dans ~/.local/bin au sein de WSL.", + "0307677bb9": "Vérification de l'enregistrement de la CLI WSL..." + }, + "accounts": { + "search": { + "86edc96bc9": "barre d'état", + "e949b08ffb": "limite de débit", + "7e67d7d1b6": "wrk", + "421c6be25e": "id", + "be8b621bdc": "espace de travail", + "8dcbef1856": "opencode", + "38d22ff8d6": "Remplacement facultatif de l'ID de workspace si la recherche automatique échoue.", + "4ee2029e9c": "ID de workspace OpenCode Go", + "9c4e40cf6b": "session", + "61f7d1fcbe": "cookie", + "d1d2ae383c": "Collez votre cookie de session opencode.ai pour récupérer les rate limits.", + "6ed1401020": "Cookie de session OpenCode Go", + "b7c2cee442": "expérimental", + "7118d2f908": "identifiants", + "933deaf732": "oauth", + "8630464352": "cli", + "e8e1ff3887": "gemini", + "bada4a3218": "Extrait les identifiants OAuth de votre installation locale de Gemini CLI pour l'authentification auprès de Google.", + "d819755b02": "Utiliser les identifiants Gemini CLI", + "35b461d817": "connexion", + "f2d666a886": "facultatif", + "8b06729e0f": "actif", + "5b3f18ef4a": "bascule", + "06662af91e": "compte", + "70d1b8def5": "codex", + "87a4a8584e": "Choisissez le compte Codex facultatif enregistré qui alimente la lecture des quotas en temps réel.", + "a4bcfd6f86": "Compte Codex actif", + "042885c07c": "périmé", + "02c438bc7b": "expiré", + "77e32a2ad3": "réauthentifier", + "c759741d77": "quota", + "b40d5b6570": "Changement de compte facultatif pour Codex et récupération en direct des limites de débit.", + "17c5d244eb": "Comptes Codex", + "e14049e1a8": "claude", + "dd75a73991": "Changement de compte facultatif pour Claude tout en préservant le contexte de conversation partagé.", + "75682e1b62": "Comptes Claude", + "e02c136ad0": "auth", + "9f70aa706c": "fournisseur", + "488a7e9206": "linux", + "0b4d948eb5": "wsl", + "bdbd1e668e": "windows", + "593720c17f": "emplacement", + "b84a5b0c8a": "Choisissez si les comptes des fournisseurs sont inspectés et ajoutés sur cet appareil ou dans WSL.", + "d09fb5ca92": "Emplacement des comptes", + "733f9e2a93": "Utilisation MiniMax", + "f8374c3151": "Collez votre cookie de session platform.minimax.io pour récupérer localement les limites de débit.", + "f4a8c2e1b7": "Utilisation Grok (xAI)", + "e3b7d1f9a2": "Connexion OAuth via Grok CLI (grok login) pour l'utilisation hebdomadaire des crédits.", + "d2c6a0e8f1": "grok", + "c1b5f9d7e0": "xai", + "b0a4e8c6d9": "oauth", + "a9f3d7b5c8": "connexion" + } + }, + "advanced": { + "search": { + "a7002e1ac4": "programme de mise à jour", + "e61ed8ab33": "mises à jour", + "6576fce4d2": "dépannage", + "79e0947e95": "assistance", + "4383251647": "vpn", + "f98a60af11": "proxy", + "65bf6af262": "compatibilité", + "621233008b": "http/1.1", + "f8ff125ebe": "http1", + "a0f71bd909": "http/2", + "4b4ae4345a": "http2", + "48a1c8f534": "http", + "4d44352eea": "réseau", + "2b4d26d11e": "connectivité réseau", + "e04e9db503": "avancé", + "585f56fae0": "Utiliser HTTP/1.1 pour la mise en réseau d'Electron lorsque HTTP/2 échoue derrière un proxy.", + "11eea3da72": "Compatibilité HTTP/1.1" + } + }, + "agents": { + "search": { + "2814401339": "installé", + "719f53350c": "chemin", + "839e82c81f": "détection", + "f622b8eb2a": "linux", + "d608654c03": "wsl", + "77c02fa3c3": "windows", + "d2952dfd74": "emplacement", + "96ba2373b6": "agent", + "cbdd7f3b9e": "Choisissez si les agents installés sont détectés sur cet appareil ou dans WSL.", + "ef804b7337": "Emplacement des agents", + "01926b9d8c": "Configurer les agents de codage IA, l'agent par défaut et les remplacements de commandes.", + "bb9ad95777": "Agents", + "d8f3a8b8a0": "default", + "167daeb5e9": "command", + "be59907510": "remplacement", + "a6d594c17d": "installer", + "f2932bf22b": "détecté", + "2afd3b5858": "activer", + "60393e1b17": "désactiver", + "2e188c771c": "masquer", + "87fffe6c20": "afficher", + "e2b7c0dcd7": "github", + "66b6b82eb4": "éveil", + "dbc8aca6b0": "veille", + "845ad9128a": "alimentation", + "48f84d10f1": "en cours", + "affbf130f6": "travaille", + "0d1c334987": "couvercle", + "ff8de8a2ad": "écran", + "0d752916f8": "hooks", + "6984d4291a": "statut", + "13b20636a6": "en attente", + "8599603496": "terminé", + "ea71995548": "supprimer", + "c1317fe641": "restaurer", + "5963143e00": "paramètres", + "042c551bc5": "config", + "f412abbba5": "claude", + "5ded38b843": "codex", + "be7ea3553b": "onglet", + "6956646a1e": "titre", + "32836788b0": "titre généré", + "966890236d": "nom", + "848dcae8d3": "généré", + "52115d0d7c": "auto", + "c64059f50d": "prompt", + "5784ae8c43": "renommer", + "8a17fd6026": "stable", + "a79d266f71": "session", + "afbf35be68": "session stable", + "agentPermissions": "Permissions des agents", + "agentPermissionsDescription": "Basculer les autorisations d'agent par défaut entre Yolo et Manuelle.", + "agentRuntime": "Runtime des agents", + "agentRuntimeDescription": "Choisissez si les agents sont détectés et lancés par défaut sous Windows ou dans WSL." + } + }, + "appearance": { + "search": { + "9a115966d3": "Réduire dans la zone de notification à la fermeture", + "4d5b9427b5": "Quand activé, fermer la fenêtre laisse Orca tourner dans la zone de notification au lieu de quitter.", + "468448bba4": "aquarelle", + "f586abfa35": "bleu", + "651f35b2c6": "sélecteur", + "e5bc35d59e": "fenêtre", + "d18b54ca90": "dock", + "1f2880a9d5": "orca", + "2cfb3420c0": "icône de l'app", + "e80c2af428": "Choisissez l'icône d'app affichée dans le Dock et le sélecteur de fenêtres.", + "2b313598c6": "Icône de l'app", + "839fb1e3ed": "boîte à outils", + "ac79fe4a04": "afficher", + "648eeada79": "masquer", + "6cf5f54ce1": "bouton", + "5bff6a2ef0": "barre latérale", + "5e5b8878bf": "téléphone", + "74618577c7": "mobile", + "682293cadf": "Afficher le bouton Orca Mobile en haut de la barre latérale gauche.", + "1de96ec8a6": "Afficher le bouton Orca Mobile", + "4c920ab2d1": "planification", + "58f4e22fa2": "automatisation", + "b186f3cefb": "automatisations", + "ae13a0d340": "Afficher le bouton Automatisations en haut de la barre latérale gauche.", + "caa27e1a8e": "Afficher le bouton Automatisations", + "6b846424cc": "linear", + "2ee4810f38": "github", + "0d5a74b606": "tâches", + "9a248333c7": "Afficher le bouton Tâches en haut de la barre latérale gauche.", + "155a1e7438": "Afficher le bouton Tâches", + "a895d0f938": "marque", + "51f957ce39": "nom", + "36e006efc1": "app", + "bed343b03e": "barre de titre", + "18b4c4c30b": "Afficher Orca dans la barre de titre.", + "fdd31b00d0": "Nom de l'app dans la barre de titre", + "c1bca1885a": "explorateur de fichiers", + "9f2df826ac": "ignoré", + "08c86bf58e": "gitignore", + "bce3ac317a": "git", + "7164edf71a": "Atténuer les fichiers correspondant à .gitignore dans l'explorateur de fichiers.", + "f8129fb544": "Afficher les fichiers ignorés par Git", + "2f12e1aa3a": "ui", + "5095258df2": "interface", + "fab91464dd": "ide", + "8b36fb3f64": "typographie", + "a0e09aed9c": "police", + "24094af355": "police", + "07c7c38fac": "Choisissez la police utilisée par l'interface d'Orca.", + "ddb991024d": "Police de l'IDE", + "0c83659f48": "raccourci", + "0952091186": "échelle", + "3ae5de6101": "zoom", + "adddb91a3d": "Met toute l'interface de l'application à l'échelle.", + "c5e933970f": "Zoom de l'interface", + "3a9b69d734": "système", + "44d873fd18": "clair", + "262fe1d24f": "sombre", + "0709c794f7": "Choisissez l'apparence d'Orca dans la fenêtre de l'app.", + "71e06350b4": "Thème", + "dc02c8759d": "espace de travail", + "43cfba3b95": "serveur", + "46d21eef62": "localhost", + "006e67b279": "ports", + "896eb53fd4": "barre d'état", + "0ececfa190": "Afficher les ports actifs des espaces de travail dans la barre d'état.", + "cf409b6c4d": "Ports", + "cb1cc62cf8": "espace", + "90bdc043ea": "disque", + "96b4fb0064": "terminal", + "4ddbde4999": "cpu", + "4355f18ac6": "mémoire", + "9c4d5f0894": "gestionnaire", + "c690a15849": "ressource", + "81ef5abc2f": "Afficher l'utilisation CPU, mémoire et disque des espaces de travail ainsi que les sessions de terminal dans la barre d'état.", + "7cf005b29f": "Gestionnaire de ressources", + "fe192b060e": "hôte", + "f4997e0f8a": "connexion", + "a278406ed5": "distant", + "6ecad74eb3": "ssh", + "f17d66d0d2": "Afficher l'état de connexion à l'hôte distant dans la barre d'état.", + "57fb424c56": "Hôtes distants", + "35565867cb": "moonshot", + "de586def95": "abonnement", + "00a028f25f": "utilisation", + "40e5c3c285": "kimi", + "c927a155d5": "Afficher l'utilisation de l'abonnement Kimi dans la barre d'état.", + "3a6c028ea8": "Utilisation Kimi", + "25e51b62ee": "limite de débit", + "d9e7cef86f": "cookie", + "d16378a88f": "minimax", + "e46178eb1b": "Afficher l'utilisation de l'abonnement MiniMax dans la barre d'état.", + "0f08f6b483": "Utilisation MiniMax", + "edbf0f63a0": "coût", + "afbb6a3767": "tokens", + "d77537b580": "opencode-go", + "a9d56852eb": "opencode", + "7f72de7cbe": "Afficher l'utilisation des tokens et des coûts d'OpenCode Go dans la barre d'état.", + "bc046e7899": "Utilisation OpenCode Go", + "51b0ccd6a2": "google", + "2804a920ad": "gemini", + "9660c5b2f1": "Afficher l'utilisation des tokens et des coûts de Gemini dans la barre d'état.", + "5bfb874d05": "Utilisation Gemini", + "97957e374e": "openai", + "8dfd676c28": "codex", + "e9e4412545": "Afficher l'utilisation des tokens et des coûts de Codex dans la barre d'état.", + "54b1acf24f": "Utilisation Codex", + "dea0a9a665": "anthropic", + "c9fe3a7876": "claude", + "de50c6f516": "Afficher l'utilisation des tokens et des coûts de Claude dans la barre d'état.", + "9dc15020d7": "Utilisation Claude", + "language": { + "locale": "paramètres régionaux", + "i18n": "i18n", + "translation": "traduction" + }, + "leftSidebarAppearance": { + "title": "Apparence de la barre latérale gauche", + "description": "Accordez la barre latérale gauche à votre terminal, gardez-la par défaut ou appliquez une teinte." + }, + "workspaceCardLayout": { + "title": "Disposition des cartes d'espace de travail", + "description": "Les cartes d'espace de travail peuvent adopter une disposition compacte ou détaillée.", + "compact": "compact", + "compactDisplay": "affichage compact", + "workspaceCards": "cartes d'espace de travail", + "worktreeCards": "cartes de worktree", + "cardLayout": "disposition des cartes", + "workspaceOptions": "options d'espace de travail", + "detailed": "détaillé" + }, + "showPinnedWorktreesInGroups": { + "title": "Afficher aussi les worktrees épinglés dans leurs listes d'origine", + "description": "Les worktrees épinglés restent dans Épinglés et apparaissent aussi dans Tous, Projet, Statut et PR." + }, + "antigravityUsageTitle": "Utilisation Antigravity", + "antigravityUsageDescription": "Afficher l'utilisation de l'abonnement Antigravity dans la barre d'état.", + "antigravityKeyword": "antigravity", + "f8e2a1c4b6": "Utilisation Grok", + "e7d1b0f3a5": "Afficher l'utilisation hebdomadaire des crédits Grok issue de l'OAuth de Grok CLI.", + "d6c0a9e2f4": "grok", + "c5b9f8d1e3": "xai", + "usagePercentageDisplayTitle": "Pourcentages d'utilisation", + "usagePercentageDisplayDescription": "Choisissez si les limites des fournisseurs affichent le pourcentage utilisé ou restant." + } + }, + "auto": { + "rename": { + "branch": { + "search": { + "0971762141": "kebab-case", + "a482f6a423": "slug", + "7adefcdd94": "template", + "10485c4fc5": "command", + "50139297e6": "prompt intégré", + "502aa57681": "instructions", + "40d21f2efc": "prompt", + "672387fb77": "Modèle de commande d'agent utilisé lors de la génération des noms de branche.", + "722551c5b3": "Modèle de commande de nom de branche", + "f41833025e": "générer", + "ed677944cc": "worktree", + "3ef3cbe98c": "agent", + "f0acf64301": "nom de créature", + "7803423877": "auto", + "55a1860e47": "renommer", + "9319bd9827": "branche", + "ea94b9da8a": "Renommer la branche générée automatiquement d'après le travail dès qu'un agent démarre.", + "427f2cd1eb": "Renommage auto de la branche" + } + } + } + }, + "browser": { + "search": { + "7539f6336c": "profil", + "1c1e097985": "arc", + "533a253deb": "edge", + "75a0d435b7": "chrome", + "854ef6ce83": "connexion", + "3910a41f32": "auth", + "2e7f951773": "importer", + "66dd641a47": "session", + "29193a51d5": "cookies", + "2d2d995c58": "navigateur", + "060ac1fcba": "Importer les cookies de Chrome, Edge ou d'autres navigateurs pour utiliser vos sessions existantes dans Orca.", + "96afedcb5c": "Session & cookies", + "a7a07d5415": "éditeur", + "8dd4805991": "fichier", + "68d1db8929": "markdown", + "90425d313c": "shift", + "linkRoutingModifier": { + "routing": "routage", + "modifier": "modificateur", + "invert": "inverser", + "opposite": "opposé" + }, + "terminalLinkActions": { + "terminal": "terminal", + "click": "clic", + "actions": "actions", + "popover": "popover", + "menu": "menu", + "disable": "désactiver" + }, + "72c58f7792": "webview", + "82ba1c80ea": "localhost", + "bea27bac4b": "liens", + "44d14df30d": "aperçu", + "5cb082b3e3": "Routage des liens", + "95944898e0": "pourcentage", + "483a0eb5e0": "nouvel onglet", + "726f2a8556": "zoom de page", + "5448f4097b": "default", + "54f4ea55f7": "échelle", + "4a98ed195f": "zoom", + "c3d89ed4d0": "Niveau de zoom appliqué aux nouveaux onglets du navigateur.", + "072b7f5c1f": "Zoom par défaut", + "0bb34eacc9": "requête", + "8b8ed06e4b": "omnibox", + "3538b3aaeb": "token", + "0732ebe6fb": "privé", + "e1c2a57f07": "kagi", + "ad40e75d13": "bing", + "1f8153acfb": "duckduckgo", + "8a489aab8d": "google", + "72b4b89970": "moteur", + "16bd69cd82": "recherche", + "0628d5943b": "Moteur de recherche utilisé quand on saisit du texte autre qu'une URL dans la barre d'adresse.", + "5e755920c9": "Moteur de recherche par défaut", + "4596a52cf7": "landing", + "5164c47e31": "vierge", + "4fda4fb066": "url", + "0dbb1eaf4e": "page d'accueil", + "291f480a5e": "accueil", + "a942905148": "URL ouverte à la création d'un nouvel onglet du navigateur. Laissez vide pour un onglet vierge.", + "c3903322d2": "Page d'accueil par défaut", + "19ea5607cf": "Libellés localhost des worktrees", + "4e0fdf0a3f": "Ouvrir les ports des espaces de travail sous forme d'URL localhost Orca propres à chaque worktree, pour mieux distinguer les onglets du navigateur." + }, + "use": { + "search": { + "3f4c559deb": "profil Arc", + "d5afa54d21": "profil Edge", + "22fb801af8": "profil Chrome", + "59968bb9b4": "navigateur authentifié", + "62e2a790c0": "session existante", + "63a66da648": "navigateur système", + "20c1323d1e": "computer use", + "ab349a2dd0": "arc", + "2e1b09897b": "edge", + "088e7a9012": "chrome", + "96ce3d2de2": "auth", + "48557f639c": "connexion", + "d5ad1f7aad": "importer", + "02837ee497": "session", + "fb8178824f": "cookies", + "ba4eb53b72": "browser use", + "2fb24d17db": "Importez les cookies depuis Chrome, Edge ou d'autres navigateurs pour que les agents réutilisent vos connexions.", + "614c756ab1": "Importer les cookies du navigateur", + "cee44fb442": "automatisation", + "a57c2172dc": "agent-browser", + "6ea88e5206": "npx", + "f5b8fdddf5": "orca-cli", + "e5a784bc54": "installer", + "9d97446873": "agent", + "a2d489263e": "skill", + "a7e82445fa": "Installez la skill Browser Use pour que les agents pilotent le navigateur d'Orca.", + "a1414dcefb": "Installer la skill Browser Use", + "e56c7b55c9": "configuration", + "034c5e8d7f": "activer", + "7e0dcb257a": "shell", + "3ffafc9b95": "command", + "30c74aaa1f": "chemin", + "ff05cbc344": "orca", + "85fab5e12c": "cli", + "890ddf943d": "Enregistrer Orca CLI pour que les agents pilotent le navigateur.", + "50f0860e18": "Activer Orca CLI" + } + } + }, + "commit": { + "message": { + "ai": { + "search": { + "3766941527": "agent", + "181cdb0637": "ouvrir", + "8e9cc598d7": "générer", + "b7d50da4d8": "template", + "7e264b926b": "brouillon", + "b261c88609": "pr", + "110be48b81": "pull request", + "001ca3f2af": "Valeurs par défaut appliquées à l'ouverture du composeur de création de PR.", + "eefd33788c": "Valeurs par défaut de création de PR", + "d32936bb2a": "branche", + "127d512e75": "commit", + "d22a6459e4": "conflits", + "53e8504fb2": "ci", + "c46e665f7e": "checks", + "37c65bbb44": "fix", + "402f101af8": "prompt", + "8e0bcc5d99": "model", + "f4731b22bf": "command", + "57c851a68c": "cli", + "61117e57f3": "args", + "0f29331fed": "arguments", + "18b6d38835": "Agent, arguments CLI et modèle de commande utilisés par chaque bouton IA de contrôle de code source.", + "3c4e5e5938": "Recettes d'action", + "ee14a9e9f7": "activé", + "82109d627d": "contrôle de code source", + "542e1a00a7": "codex", + "f121bec167": "claude", + "93e5210da8": "message", + "c33cb1b982": "ai", + "0b946b2abe": "Ajoute des recettes d'actions pour les actions commit, pull request, nom de branche et correction de Source Control.", + "24dbdfca78": "Afficher les actions Source Control AI" + } + } + } + }, + "computer": { + "use": { + "search": { + "6e88da3508": "skill", + "798be54d7e": "automatisation", + "e27f8bafbf": "capture d'écran", + "26c1290d83": "enregistrement d'écran", + "82f01c2d2c": "accessibilité", + "fefb452f5b": "computer use", + "9210db582b": "Autoriser les agents à analyser des captures d'écran et à piloter des applications locales à votre demande.", + "442bec10fe": "Computer Use" + } + } + }, + "developer": { + "permissions": { + "search": { + "3363889768": "LAN, USB et Bluetooth", + "6c82846f66": "appareil", + "11653d3f42": "mdns", + "78a10b826f": "bonjour", + "e3fbc48083": "bluetooth", + "c4a4a02ea4": "usb", + "fa3239cd42": "réseau local", + "87620e6416": "lan", + "acad3d4743": "Autoriser les outils d'appareils et l'accès LAN utilisés depuis les sessions de terminal.", + "3e0131e45d": "icloud", + "ce07159ff5": "bureau", + "a0c19119fb": "téléchargements", + "4438f81bfa": "documents", + "c10e36cbd1": "accès complet au disque", + "05ab708ee5": "Ouvrir le volet de confidentialité de macOS pour l'accès aux fichiers protégés des projets et worktrees.", + "bbf543a3a1": "Accès complet au disque", + "7f145a3984": "fenêtre", + "5610022e1e": "automatisation", + "0a467b750e": "capture d'écran", + "08f8039ca9": "accessibilité", + "3cd51d18a1": "enregistrement d'écran", + "2e5d98ab56": "Autoriser les captures d'écran, l'inspection de l'écran, la saisie clavier et l'automatisation des fenêtres.", + "39bb49e662": "Enregistrement d'écran et accessibilité", + "00e954319e": "whisper", + "1e6e27b202": "ffmpeg", + "f061f08b7b": "sox", + "a765112513": "vidéo", + "b192432ef0": "audio", + "af122938a3": "voix", + "259b829b84": "caméra", + "ed7c12bdb4": "micro", + "6eca1636b7": "Autoriser les outils de voix, de transcription, de webcam et de capture multimédia.", + "302c0c42f9": "Micro et caméra", + "4e225e7c56": "outils développeur", + "6db4fca386": "macos", + "0c13b249e3": "tcc", + "2270ccff3f": "confidentialité", + "a98aa11a9c": "autorisations", + "bc8ac95310": "Autorisations macOS pour les outils développeur lancés depuis le terminal.", + "e92cb0896d": "Autorisations développeur" + } + } + }, + "experimental": { + "search": { + "44c7f209d5": "node_modules", + "4ad605f222": "env", + "3021571c30": "partagé", + "f082788cfe": "liens", + "3028f0bd3a": "lien", + "bff1ff7768": "liens symboliques", + "c387565812": "lien symbolique", + "10b52f79c1": "worktrees", + "d23ae13990": "worktree", + "0d24759f14": "expérimental", + "603d29ed74": "Matérialiser automatiquement les fichiers ou dossiers configurés dans les nouveaux worktrees créés, par clone APFS sur macOS quand c'est possible, sinon par liens symboliques.", + "78c2a8dc74": "Chemins partagés sur les worktrees", + "7b79081695": "non lu", + "f10d307468": "achèvement", + "5f067ba0f9": "agent", + "7695fd30e9": "notification", + "8facf10138": "cloche", + "edc49480a1": "volet", + "268e99d957": "surlignage", + "01567f19ca": "attention", + "9bb3bd5098": "terminal", + "11877246fc": "Surlignage persistant du volet pour les sonneries de terminal et les fins d'agents.", + "9e4ddf776d": "Attention du terminal", + "agentHibernation": { + "agent": "agent", + "agents": "agents", + "description": "Interrompt les terminaux d'agents en arrière-plan devenus inactifs après la durée d'inactivité configurée et reprend les sessions prises en charge lors de leur réouverture. La veille des agents conserve les options de lancement des agents démarrés par Orca ; les agents démarrés manuellement peuvent reprendre avec les valeurs par défaut actuelles d'Orca.", + "minutes": "minutes", + "sleep": "veille", + "terminal": "terminal", + "title": "Mise en veille des agents" + }, + "newWorktreeCardStyle": { + "card": "carte", + "cards": "cartes", + "description": "Prévisualise la nouvelle mise en page des cartes de worktree, l'emplacement des métadonnées, les options du menu d'affichage des cartes et la présentation des statuts.", + "menu": "menu", + "metadata": "métadonnées", + "status": "statut", + "title": "Nouveau style de carte", + "worktree": "worktree", + "worktrees": "worktrees" + }, + "fe5688b761": "barre latérale", + "ca5d1f3f46": "chronologie", + "d01b3882ba": "notifications", + "244a0ecd3d": "activité", + "92a9357d1f": "vue des agents", + "fa72e71f05": "agents", + "4d63251595": "Flux groupé dans la barre latérale gauche pour les achèvements d'agents et les états bloquants.", + "ccc5548ac5": "Vue Agents", + "9af7a518db": "personnage", + "791fefc0b0": "coin", + "65df471ab2": "animé", + "9f5609bfb8": "superposition", + "2a33975d72": "mascotte", + "b54cea709b": "compagnon", + "051203d37c": "animal", + "6b5a56ac35": "Compagnon animé flottant dans le coin inférieur droit.", + "87d99e634b": "Compagnon", + "nativeChat": { + "title": "Chat UI", + "description": "Prévisualisez la surface de chat de bureau pour les sessions de terminal d'agents prises en charge.", + "grok": "grok" + }, + "agentDashboard": { + "title": "Tableau de bord des agents", + "description": "Tableau Kanban pour surveiller les agents de tous les worktrees, dans la fenêtre ou en fenêtre indépendante." + } + } + }, + "floating": { + "workspace": { + "search": { + "94f4d013c8": "barre d'état", + "a452146574": "bouton de bascule", + "6765b85e48": "répertoire de lancement", + "a38bfc3f77": "panneau rapide", + "52db6e3baf": "notes", + "156ffeee08": "note", + "884e5e6132": "markdown", + "49db74a92d": "navigateur", + "6410fe83d8": "terminal", + "2b5efa55c9": "global", + "ebeedb2f6a": "terminal rapide", + "6f183fa1b9": "terminal flottant", + "a08e482f6d": "espace de travail flottant", + "b96b5ee6cf": "Activer l'espace de travail flottant, choisir où s'ouvrent les nouveaux onglets et où apparaît le bouton de bascule.", + "b2b60e7163": "Workspace flottant" + } + } + }, + "general": { + "search": { + "bdfb6dc21b": "j'aime", + "e6b01c8e30": "commentaires", + "b65665703a": "assistance", + "06ea5a69a6": "github", + "e4fb4516d0": "étoile", + "e0b8c8bc25": "Soutenez le projet avec une étoile GitHub via la CLI gh.", + "36a72f0d9e": "Ajouter une étoile à Orca sur GitHub", + "c61b14be7c": "grok", + "5d9ba08673": "copilot", + "f472e97440": "aider", + "3c30fe2d51": "gemini", + "5fdf1dc2d1": "omp", + "9b0bc30160": "pi", + "882c4896fd": "opencode", + "27d9b996ba": "codex", + "5baf51c4d9": "open claude", + "aea7d2cccb": "openclaude", + "95b63edde7": "claude", + "41c2f9a025": "default", + "8ea37a05bc": "agent", + "e2da948f59": "Présélectionner un agent de codage IA dans le composeur de nouvel espace de travail.", + "db11502270": "Agent par défaut", + "3462308bd3": "tokens", + "660528b048": "coût", + "585beac3f8": "ttl", + "0efc9d96ad": "prompt", + "939b80f5fd": "minuteur", + "b2601a778c": "cache", + "40c9585e43": "Minuteur à rebours affichant le temps restant avant l'expiration du cache de prompt (agents Claude).", + "1e0f28c6f1": "Minuteur de cache de prompt", + "e49e739a59": "téléchargement", + "c9d8c1ce66": "notes de version", + "9e86ccd05c": "version", + "f89a94773c": "mise à jour", + "79ff46776e": "Recherche les mises à jour de l'application et installe une version plus récente d'Orca.", + "e15af4eb64": "Rechercher des mises à jour", + "6382fe9724": "npx", + "baa263d6d8": "agents", + "bda108e66c": "skill", + "244e3fb4c8": "Installer le skill Orca pour que les agents sachent utiliser la CLI Orca.", + "2d9f7b42df": "Skill d'agent", + "0a00691c06": "commande shell", + "dbeb1f348e": "command", + "88d3df9ce9": "terminal", + "fb4f338a3d": "chemin", + "924a660a78": "cli", + "ca529079bf": "Enregistrer ou supprimer la commande CLI Orca.", + "327e3fa70d": "Orca CLI", + "fb84767421": "bascule", + "f8f0ac213a": "séquentiel", + "12ecc640a8": "mru", + "54ba13831a": "récent", + "750420dd9a": "contrôle", + "fe62b3f09f": "ctrl", + "2a254b725e": "onglet", + "ca812803ea": "ordre récent des onglets", + "e53d585ed6": "Récents ou barre d'onglets.", + "256d92554d": "Ordre des onglets", + "22572e99c1": "annotations", + "1ff67ba40c": "notes", + "4dd5684836": "revue", + "d05f629d2c": "markdown", + "694613d47f": "Afficher les contrôles des notes de revue markdown locales en mode éditeur enrichi.", + "128bc09325": "Notes de revue Markdown", + "a0014961ae": "défilement", + "3ca5ab78a5": "code", + "e3919429c0": "vue d'ensemble", + "9c72990db8": "minimap", + "716a4dfb1f": "Afficher la minimap lors de l'édition d'un fichier.", + "6f584fcb48": "Minimap", + "19baae651b": "barre latérale", + "973ed6bfbf": "diff combiné", + "0a02059549": "arborescence de fichiers", + "2f42852568": "arborescence", + "3b5733573e": "diff", + "dec71988f0": "Afficher ou masquer l'arborescence de fichiers à l'ouverture des vues de diff combinées.", + "adec13f2ef": "Arborescence de fichiers des diffs par défaut", + "be24c7cd67": "scindé", + "233f7e2f37": "côte à côte", + "0a5fa65926": "en ligne", + "2b463f0bf9": "vue", + "ecb9415c80": "Format de présentation préféré pour l'affichage par défaut des diffs git.", + "2760c9933f": "Vue de diff par défaut", + "b2799ba622": "millisecondes", + "146728ac2c": "délai", + "86f54575c7": "enregistrement automatique", + "8ea61ad55c": "Délai d'attente d'Orca après votre dernière modification avant l'enregistrement automatique.", + "14e46c745b": "Délai d'enregistrement automatique", + "4469b6fa4e": "enregistrer", + "e9d948d3c3": "Enregistre automatiquement les modifications de l'éditeur et des diffs éditables après une courte pause.", + "ae21e806ce": "Enregistrement automatique des fichiers", + "c56cb6f1c2": "réseau", + "3566fce83f": "localhost", + "91a46caafc": "no_proxy", + "3a73054565": "contourner", + "20b711ac9e": "proxy", + "eb8946b2c9": "Hôtes qui doivent contourner le proxy HTTP configuré.", + "8436ff6f8e": "Règles de contournement du proxy", + "e55d62dfa4": "launchpad", + "9da6c875e5": "dock", + "b9096a44cf": "https_proxy", + "8f03d44672": "http_proxy", + "e3b1d42f95": "URL de proxy pour les requêtes réseau d'Orca et les processus enfants des terminaux locaux.", + "c29f23ab57": "Proxy HTTP", + "6c2ce8457c": "explorateur de fichiers", + "c9d9636f24": "Finder", + "68d03d9980": "vscode", + "ebf8f056b5": "zed", + "0cb3d94f00": "cursor", + "8fb00fcd05": "lanceur", + "e1ee631696": "éditeur", + "5a9df5566f": "ouvrir le menu", + "b8093e9a93": "ouvrir dans", + "a916662068": "Choisissez les applications proposées dans le menu « Ouvrir dans » d'un espace de travail.", + "451d4af994": "Applications « Ouvrir dans »", + "7e9b556873": "ignorer", + "ca86dd6e27": "boîte de dialogue", + "9f8558233a": "confirmer", + "7edf4f69e2": "automatisation", + "84c67d0108": "supprimer", + "a0c44061ee": "Afficher une boîte de dialogue de confirmation avant de supprimer une automatisation et son historique d'exécution.", + "d0a65b27fd": "Confirmer avant de supprimer les automatisations", + "df10666259": "worktree", + "ae98c9cf36": "Afficher une boîte de dialogue de confirmation avant de supprimer un espace de travail.", + "913242091d": "Confirmer avant de supprimer les espaces de travail", + "93f6ec5e70": "répertoire", + "9bde064915": "sous-dossier", + "ec5049e510": "imbriqué", + "b9cffd374d": "Créer les espaces de travail dans un sous-dossier nommé d'après le dépôt.", + "141f71c69f": "Imbriquer les espaces de travail", + "7887a2c262": "dossier", + "7baf524b04": "espace de travail", + "d0bc793689": "Dossier racine où sont créés les dossiers des espaces de travail.", + "4c95d08fa2": "Dossier des espaces de travail", + "161a86a9da": "Confirmer avant de fermer les onglets épinglés", + "8e593f04fc": "Afficher une boîte de dialogue de confirmation avant de fermer un onglet épinglé.", + "defaultProjectRuntime": "Runtime par défaut des projets", + "defaultProjectRuntimeDescription": "Choisissez le runtime hérité par les projets Windows locaux.", + "d2d2d929c0": "Vérification orthographique du Markdown enrichi", + "4497e2e2bb": "Affiche les soulignements d'orthographe et les suggestions du navigateur pendant l'édition du Markdown enrichi.", + "e61157e926": "Retour à la ligne dans l'éditeur", + "005be5c699": "Renvoie à la ligne les lignes longues dans les éditeurs de fichiers au lieu d'imposer un défilement horizontal.", + "editorFontFamily": "Police de l'éditeur", + "editorFontFamilyDesc": "Police utilisée par les éditeurs de fichiers et les vues de diff. Laissez vide pour suivre la police du terminal.", + "externalWorktrees": "Worktrees externes", + "externalWorktreesDescription": "Choisissez si les worktrees créés en dehors d'Orca apparaissent par défaut." + } + }, + "git": { + "search": { + "16f53f7323": "gh", + "d088806071": "github", + "40f9b815fd": "budget api", + "b7e52124c7": "limite de débit", + "ead733645f": "glab", + "4808f065b3": "gitlab", + "2b4a72885d": "En-têtes de limite de débit REST actuels de la CLI GitLab, lorsqu'ils sont disponibles.", + "83ecb3f470": "Budget API GitLab", + "65b69d9f80": "graphql", + "1139f61512": "Limites de débit REST, Search et GraphQL actuelles de la CLI GitHub.", + "ff86e354c4": "Quota d'API GitHub", + "035134fcd9": "worktree", + "0c75583ca9": "sans risque", + "bae91effdd": "base à jour", + "de06e9d105": "ref de base", + "ab0e22c9f6": "actualiser le main local", + "d9f70d51a0": "main obsolète", + "0849b571fe": "à jour", + "c41e345153": "en retard sur main", + "6ee3cfff02": "git diff", + "564942ffc5": "origin/main", + "28192e3a63": "master", + "e3e9adde59": "main", + "0e993bf00f": "Lorsque vous créez un espace de travail, Orca actualise la base distante et fait avancer en fast-forward, sans risque, votre branche locale correspondante, telle que main ou master. Cela évite que des commandes comme git diff main...HEAD comparent avec un historique obsolète. Orca ignore cette mise à jour si cette branche contient des modifications non committées ou des commits locaux uniquement.", + "f8bda25f29": "Garder la branche main locale à jour", + "769ddd7f81": "custom", + "1d2fae1fa2": "nom d'utilisateur git", + "f83c8937c4": "nommage des branches", + "5ecd91c5ef": "Préfixe ajouté aux noms de branches à la création des worktrees.", + "68bd65fdb8": "Préfixe de branche", + "sourceControlGroupOrderTitle": "Ordre des groupes du contrôle de code source", + "sourceControlGroupOrderDescription": "Choisissez si Modifications, Modifications indexées ou Fichiers non suivis apparaissent en premier dans le contrôle de code source.", + "groupOrder": "ordre des groupes", + "changesFirst": "modifications d'abord", + "stagedFirst": "staged d'abord", + "untrackedFirst": "untracked d'abord", + "sourceControl": "contrôle de code source", + "gitChanges": "modifications git", + "compareAgainstUpstreamTitle": "Base de comparaison par défaut", + "compareAgainstUpstreamDescription": "Choisissez la base que le contrôle de code source utilise par défaut pour comparer les changements commités. L'amont de la branche suit automatiquement la branche courante et revient à la branche par défaut du dépôt en l'absence d'amont. Vous pouvez toujours changer la base de comparaison d'un worktree depuis son panneau Git. Les cibles de pull request et de rebase ne changent pas.", + "compareBase": "base de comparaison", + "defaultCompareBase": "base de comparaison par défaut", + "defaultBranch": "branche par défaut", + "repositoryDefault": "défaut du dépôt", + "branchUpstream": "upstream de la branche", + "currentBranch": "branche actuelle", + "upstream": "upstream", + "localChanges": "modifications locales", + "originMaster": "origin/master", + "committedChanges": "modifications committées" + } + }, + "input": { + "search": { + "886597d6b3": "macos", + "26c83b06c5": "linux", + "71905435dd": "x11", + "7059cfb00a": "presse-papiers", + "c4440c3986": "coller", + "5fb84ba77f": "bouton du milieu", + "31ba58c8ae": "clic milieu", + "de51e18ee9": "sélection primaire", + "e5cd0e7a46": "sélection", + "e25165320e": "édition", + "b51d47ceb7": "entrée", + "874d88f4a6": "Activé par défaut sur Linux et macOS. Linux utilise le presse-papiers de sélection du système ; les autres plateformes utilisent un tampon privé.", + "d952ce9b46": "Collage de la sélection au clic milieu" + } + }, + "integrations": { + "search": { + "a626990bd2": "déconnexion", + "3c3d3d8ffa": "connect", + "faa0b5a0d9": "clé API", + "c450244ad7": "intégration", + "7319e3015b": "linear", + "16a486a49d": "Connectez Linear pour parcourir et lier des issues.", + "b027b4b318": "Intégration Linear", + "20540996ef": "identifiants", + "2ec2bd328c": "jeton API", + "7345b7c3e6": "atlassian", + "e1263dd748": "jira", + "76f6af7c57": "Connectez Jira Cloud ou mettez à jour les identifiants du jeton API Jira.", + "617603509b": "Intégration Jira", + "8c568d761c": "pull request", + "33180e8c10": "auto-hébergé", + "129fc59aa8": "gitea", + "d0d019dc29": "Authentification Gitea via des variables d'environnement de jeton API.", + "aab86d64e5": "Intégration Gitea", + "03a7b275be": "ado", + "ed63380247": "azure repos", + "b38b5d27f1": "azure devops", + "7b1f3984bb": "Authentification Azure DevOps Repos via des variables d'environnement de jeton.", + "af6611fa6e": "Intégration Azure DevOps", + "50d20817f7": "bitbucket", + "c97d58a0f3": "Authentification Bitbucket Cloud via des variables d'environnement de jeton API.", + "67a2a0e868": "Intégration Bitbucket", + "371ee914d2": "merge request", + "581844769a": "mr", + "b40cbe5de4": "glab", + "b939695c69": "gitlab", + "6e2ab619c6": "Authentification GitLab via la CLI glab.", + "b50b71ef9d": "Intégration GitLab", + "41ccade05c": "gh", + "b79c21bd42": "github", + "7166b9090c": "Authentification GitHub via la CLI gh.", + "f16e41cc72": "Intégration GitHub", + "760e2446e0": "Authentification Bitbucket Cloud avec un jeton API enregistré ou des variables d'environnement.", + "af5ae87847": "jeton d'accès" + } + }, + "jira": { + "integration": { + "card": { + "8ff73fef62": "Les jetons Jira sont chiffrés par le runtime actif et stockés localement. Ressaisir la même URL de site et le même e-mail remplace le jeton API de ce site.", + "9046a20d4c": "Déconnecter {{value0}}", + "eaffa454e9": "Mettre à jour", + "cec06a0f79": "Test en cours…", + "ab350991b8": "Vérifié", + "d914d7ab70": "Vérification…", + "5936977fcd": "Annuler", + "1666f8d562": "Créer un jeton API Atlassian", + "1ab7f551f3": "Token API Atlassian", + "09d310e42d": "you@example.com", + "27dae4ab60": "https://example.atlassian.net", + "a28f417220": "Connecter Jira", + "9bb34706ca": "Connecté", + "efaab83c5d": "Ajouter un site", + "09742875cd": "Jira", + "255bfe98ec": "Tester", + "5fb1562315": "error", + "3df81cb0ac": "ok", + "2e8bb790fd": "Se connecter", + "33a8b261ee": "Mettre à jour les identifiants", + "d5c5b47bb9": "mise à jour", + "fb854902d9": "connexion en cours", + "9a9f8d4910": "Connectez Jira Cloud pour parcourir, créer et lier des issues.", + "74f3063026": "{{value0}} site{{value1}} connecté", + "statusConnected": "Connecté", + "statusNotConnected": "Non connecté" + } + } + }, + "mobile": { + "emulator": { + "search": { + "2348045036": "Choisissez l'appareil émulateur qu'Orca ouvre par défaut.", + "2bb2e09225": "skill mobile", + "bbe4267416": "type d'émulateur", + "64494f03c3": "attach de l'émulateur", + "6f728f1456": "tap de l'émulateur", + "f8b871d655": "CLI de l'agent", + "2e0b45b2ba": "Utilisez les commandes CLI d'Orca pour lister, attacher, taper et saisir du texte dans un émulateur mobile.", + "ea3eac39bb": "Contrôle par CLI d'agent", + "8ef0f08d36": "runtime", + "27397fe8e9": "outils de ligne de commande Xcode", + "7650063d17": "simctl", + "3211e7acf9": "xcrun", + "42bfab45d8": "disponibilité", + "ea1f51b980": "Vérifiez que Xcode, simctl, serve-sim et les appareils émulateurs sont prêts.", + "0b95dfd5b3": "Disponibilité de l'émulateur", + "04c5f5d901": "appareil", + "25d7bfbcd4": "udid", + "ec3c4043fd": "iPad par défaut", + "1dc8c52ffa": "iPhone par défaut", + "ab4814f3c5": "simulateur par défaut", + "54184cb9c5": "Appareil émulateur par défaut", + "b8ddd13195": "émulateur de l'agent", + "1ad6fb6230": "appareil par défaut", + "ac0a985873": "skill émulateur", + "9353854ff3": "émulateur Orca", + "d4b7833894": "CLI Orca", + "84e5706975": "serve-sim", + "7c5a8a2bee": "xcode", + "bec7231663": "iPad", + "49727355a3": "iPhone", + "6b6407dc1f": "émulateur", + "2d67f708ce": "simulateur", + "c5eca29310": "simulateur iOS", + "25159de808": "émulateur mobile", + "9595354cff": "Configurez la prise en charge des émulateurs mobiles pour Orca et les agents de codage.", + "cdd3c31918": "Émulateur mobile" + } + }, + "pane": { + "search": { + "dbccde3a60": "clôturer", + "9e16be01d6": "arrière-plan", + "3a5e31e84b": "laisser", + "8015fd9523": "conserver", + "aa3f736042": "redimensionner", + "356c31d6dc": "largeur", + "fadcbfdd99": "ajuster", + "ad08035c5f": "téléphone", + "6cd2bfdb0e": "restaurer", + "b34ad5b3a7": "terminal", + "6db86f445f": "mobile", + "707fc78052": "Choisissez ce qu'il advient des terminaux consultés sur mobile après la fermeture de l'app ou un changement de contexte.", + "1e711aca11": "Quand vous quittez l'app mobile", + "126afc5dbd": "distant", + "70f505f3c3": "lan", + "1802188b5d": "Wi-Fi", + "dd6e671aa9": "adresse", + "1f70d63998": "IP", + "d0c89bc4a9": "superposition", + "87711f4b8f": "vpn", + "16bff559a0": "tailnet", + "c690e3ee38": "tailscale", + "a023683767": "interface", + "7b37c2e557": "réseau", + "3190ef67a4": "Choisissez l'adresse réseau à utiliser pour l'appairage mobile.", + "d96c315227": "Interface réseau", + "7d01f93ec0": "connecté", + "5e8fda4d7f": "appairé", + "905c65a308": "révoquer", + "82783d9b71": "appareils", + "13419718b3": "Gérez les appareils mobiles appairés.", + "9d3a9397ba": "Appareils connectés", + "2128a21096": "scanner", + "e518cbd61c": "appairer", + "4a0c826f3d": "code", + "3c1807a81a": "QR", + "7fb728fb2b": "Appairez un appareil mobile en scannant un code QR.", + "d49925710a": "Appairage mobile" + } + }, + "settings": { + "search": { + "b730ff7049": "expérimental", + "8d4ba0ef09": "bêta", + "6bfa001752": "APK", + "a7eececc1d": "android", + "7e801801ac": "distant", + "0b7e585cb9": "scanner", + "59b1d75fd1": "code", + "87816d1c59": "QR", + "cf2c93b479": "appairer", + "f4ed142753": "téléphone", + "f213400800": "mobile", + "671eb4173c": "Contrôlez les terminaux et les agents depuis votre téléphone.", + "ffd52a96e4": "Mobile", + "1de96ec8a6": "Afficher le bouton Orca Mobile", + "682293cadf": "Afficher le bouton Orca Mobile en haut de la barre latérale gauche.", + "e4f4daea0e": "relais", + "5d5af8e041": "iPhone" + } + } + }, + "notifications": { + "search": { + "aa288005c3": "test", + "ca8faa40d7": "notifications", + "4e30b1925e": "Déclenchez une notification de bureau d'exemple via le chemin de livraison natif.", + "ef9b311346": "Envoyer une notification de test", + "ecdeff4993": "niveau sonore", + "d58b64dddf": "volume", + "dc7d7c07cd": "son", + "eeb6f77322": "Volume de lecture des sons de notification non système.", + "aace1a62c6": "Volume des notifications", + "ef86a782cc": "bong", + "3014ad1b8f": "ding", + "079c29aeb5": "flac", + "722face52f": "aac", + "6ecb8418cb": "m4a", + "d16ae23645": "ogg", + "57e34a31cd": "wav", + "5362074f19": "mp3", + "6e08f78315": "audio", + "c718793e95": "Choisissez le fichier audio intégré, système ou local qu'Orca lit pour les notifications de bureau.", + "ea8cb8d9ce": "Son de notification", + "4ada6bfde9": "filtrage", + "fa60d8e4ab": "masquer", + "a4c3b29a3c": "focus", + "7247b97a31": "Évitez de notifier quand Orca est au premier plan sur le worktree actif.", + "96562a72c6": "Masquer quand la fenêtre est active", + "a2ab73b325": "attention", + "ae0487f8fd": "cloche", + "c638ae989d": "terminal", + "d3f1c48677": "Notifiez quand un terminal en arrière-plan émet un caractère de cloche (bell).", + "a5edee1d99": "Sonnerie du terminal", + "193e1f107c": "tâche", + "dd9d3e5f0f": "idle", + "5f7472d3fb": "terminé", + "7fa07e9600": "agent", + "10d83ef8dc": "Notifiez quand un agent de codage passe de l'état actif à l'état inactif.", + "bdc1edaeb4": "Tâche d'agent terminée", + "adbc3a0fcf": "natif", + "72539aede4": "système", + "51ae2183e1": "bureau", + "0534c76311": "Interrupteur principal des notifications de bureau d'Orca.", + "4a210b2f72": "Activer les notifications" + } + }, + "orchestration": { + "search": { + "f5d39af41e": "agents enfants", + "c766a01978": "handoff", + "08c65b12a2": "exemples", + "f278fd04db": "codex", + "32c5098e7b": "claude", + "21c28ccdf7": "coordinateur", + "741dfc03fa": "worker", + "ca54c69806": "DAG", + "7ad948b714": "tâche", + "eee028ae14": "dispatch", + "9a5ebdca31": "messagerie", + "91fc8ab7e5": "coordination", + "13ba5c6cbd": "agents", + "d86705ba77": "multi-agents", + "a7f76b4ca7": "orchestration", + "e05ff36753": "Coordonnez plusieurs agents de codage via messagerie, DAG de tâches, dispatch et points de décision.", + "c34045764e": "Orchestration d'agents" + } + }, + "privacy": { + "search": { + "3922051573": "données", + "e8bc614a18": "désactiver", + "d8191ae5ca": "variable d'environnement", + "94e04427f6": "env", + "664f1a8984": "intégration continue", + "5854a5c752": "ci", + "69637f4dc4": "orca_telemetry_disabled", + "058550f6bc": "do_not_track", + "83a6cd79b3": "do not track", + "f7a2d9f137": "Variables d'environnement qui désactivent l'envoi des données de télémétrie.", + "e058a3c98d": "Variables d'environnement de télémétrie", + "1686c07fee": "assistance", + "c0494ff48a": "diagnostics", + "8b08f32366": "Diagnostics de l'app et contrôles de partage avec le support.", + "6d258d2ed6": "Diagnostics", + "ead1deded2": "partager", + "27a27b2f63": "refuser", + "4d4bb76bf4": "accepter", + "b021b9cb81": "anonyme", + "79c319948b": "utilisation", + "77d3180def": "télémétrie", + "b707cc3981": "Aidez à améliorer Orca en envoyant des événements anonymes d'utilisation des fonctionnalités.", + "57b283461a": "Partager les données d'utilisation anonymes", + "2b5a5c312f": "posthog", + "4104f6f0f3": "analytique", + "10124159f1": "confidentialité", + "aa3b794c17": "Données d'utilisation anonymes du produit, diagnostics et contrôles de télémétrie.", + "5c508bad41": "Confidentialité & télémétrie" + } + }, + "quick": { + "commands": { + "search": { + "3c316e6ef8": "yarn", + "b86c727100": "npm", + "b949a7c0a0": "pnpm", + "0b78c4a165": "lancer", + "2d8aff42be": "exécuter", + "1c5bdcd0f2": "dépôt", + "89d2a9ad9f": "dépôt", + "f58b92a48f": "projet", + "8bf43c2dad": "global", + "a26ecdb77b": "snippet", + "d07d130849": "raccourci", + "0073cf8ce9": "terminal", + "cfffa6cdb6": "commandes", + "fecb031823": "command", + "236d4cfac8": "rapide", + "d691c4e8d8": "Commandes de terminal enregistrées, lançables depuis n'importe quel terminal, à portée globale ou propre à un projet donné.", + "4c8945952b": "Commandes rapides" + } + } + }, + "repository": { + "search": { + "bc7e504b8e": ".orca/issue-command", + "603c68b68c": "orca.yaml", + "9dc60d7f6d": "github", + "ec70364df2": "workflow", + "66b584bd6c": "commande d'issue", + "2011a6a4f2": "commande d'issue GitHub", + "d42d1e49c0": "Commande d'issue liée définie dans un fichier, configurée via orca.yaml et une surcharge locale facultative.", + "d86ea12d16": "Commande d'issue GitHub personnalisée", + "c5e8bdbcbb": "ignorée par défaut", + "a69c5cbe90": "exécutée par défaut", + "80c490b012": "demander", + "f9d84b7971": "politique d'exécution du setup", + "c00a549e03": "Choisissez le comportement par défaut quand un script de setup est disponible.", + "cdfe398068": "Quand exécuter le setup", + "5e9445bbfd": "faisant foi", + "f1e1bfa89f": "source", + "1d90a6cfbb": "les deux", + "fcb8fa8144": "partagé", + "0432d2fb7c": "local", + "ed269fad69": "source des commandes", + "19f58d6d89": "avancé", + "d141897c90": "Détails sur la source des commandes et orca.yaml.", + "cc11699c3d": "Avancé", + "bf460fded8": "yaml", + "9cad92fe77": "hooks orca.yaml", + "6b80f7d3c8": "scripts des paramètres locaux", + "a1a4c51d58": "commande d'archive", + "fbfd2386e8": "script d'archive", + "4c17787d7b": "archiver", + "8655e3387b": "hooks", + "acd1157f0c": "Scripts locaux et partagés exécutés avant l'archivage d'un worktree.", + "bce0ca23c6": "Script d'archivage", + "491b05d6e6": "commande de setup", + "a31b43a7f8": "script de setup", + "5590388dfa": "configuration", + "baaf70bb37": "Scripts locaux et partagés exécutés après la création d'un nouveau worktree.", + "b79df26937": "Script de setup", + "d73fb47b45": ".claude/mcp.json", + "db11b337c4": ".claude.json", + "26f42fe773": ".cursor/mcp.json", + "e760e3fae7": ".mcp.json", + "16dc7a4637": "model context protocol", + "343f0a508c": "mcp", + "3c31801626": "Inspectez les fichiers de configuration des serveurs MCP au niveau du projet.", + "31bd0a2420": "Configurations MCP", + "84da7fa2d7": "node_modules", + "0a3a582794": "env", + "3c180a251c": "lien", + "f1c53f2820": "worktree", + "7e228fc439": "liens symboliques", + "c06adcf136": "lien symbolique", + "copy": "copier", + "clone": "cloner", + "apfs": "apfs", + "ed885e589f": "Chemins à matérialiser depuis le checkout principal vers les nouveaux worktrees.", + "01b3377ebc": "Chemins partagés des worktrees", + "fff8834983": "prompt", + "fa3131f223": "model", + "130d76dc16": "renommer", + "917dce844a": "nom de branche", + "8068d8d0f1": "pr", + "5ff7fe1ade": "pull request", + "eec39b3de6": "message de commit", + "cfad7ce5f3": "ai", + "a47f51127e": "contrôle de code source", + "6cc5c65e64": "Surcharges de génération git propres au projet.", + "eec3995dc6": "Auteur IA Git", + "availableHosts": "Hôtes disponibles", + "availableHostsDescription": "Hôtes où ce projet est configuré.", + "host": "hôte", + "ssh": "ssh", + "remote": "distant", + "vm": "VM", + "cc876ca5f2": "dépôt", + "6469de5368": "projet", + "3067595d82": "supprimer", + "c86478c3d8": "Retirez ce projet d'Orca.", + "c5266c2c9d": "Retirer le projet", + "4b9a18a56d": "monorepo", + "4e2529722c": "répertoires", + "1ff4f12c0c": "répertoire", + "9f5ae26ccd": "préréglages", + "095fca94fe": "préréglage", + "aa42616e3d": "checkout", + "4f3c0230c2": "sparse", + "90a331fd68": "Ensembles de répertoires enregistrés pour la création de worktrees sparse.", + "1f0f20bbb6": "Préréglages de sparse checkout", + "4733ec2395": "../worktrees", + "58d8bca414": "relatif", + "a325a89dff": "chemin de l'espace de travail", + "f3e6dee5fe": "chemin du worktree", + "cd33a5525e": "Répertoire spécifique au projet pour les nouveaux worktrees.", + "443d127b5a": "Emplacement des worktrees", + "9811f3d152": "branche", + "f41cef5083": "ref de base", + "f571081ec4": "Branche ou ref de base par défaut à la création des worktrees.", + "094adbe930": "Base de worktree par défaut", + "27733eb6c1": "favicon", + "1e73e840ff": "emoji", + "cb4b4de666": "avatar", + "c1075178cf": "badge", + "6d8de2f090": "hex", + "8d045419b1": "couleur", + "b2546efab5": "icône du dépôt", + "6438a94c63": "icône du projet", + "a1f3a2bd47": "Icône et couleur du projet utilisées dans la barre latérale et les onglets.", + "b24f00294a": "Icône du projet", + "cd73b976d7": "nom du dépôt", + "92af66c7ce": "nom du projet", + "883aad2801": "Détails d'affichage propres au projet pour la barre latérale et les onglets.", + "7e1e456a95": "Nom d'affichage", + "keepForkUpToDate": "Garder le fork à jour", + "keepForkUpToDateDescription": "Faites avancer ce fork en fast-forward depuis upstream, sans risque.", + "projectRuntime": "Runtime des projets", + "projectRuntimeDescription": "Choisissez si ce projet s'exécute sous Windows ou WSL.", + "externalWorktrees": "Worktrees externes", + "externalWorktreesDescription": "Définir si les worktrees créés hors d'Orca apparaissent pour ce projet." + } + }, + "runtime": { + "environments": { + "search": { + "772e3b4753": "VM", + "45501ff2c3": "cloud", + "2bd988d041": "code d'appairage", + "5cd7dca3b8": "distant", + "d760866285": "client", + "09568ccc65": "serveur", + "ebd5369acf": "environnement", + "d198440ce3": "runtime", + "baec27aa8f": "Connectez ce navigateur à un serveur Orca enregistré.", + "3517fb2ec0": "Serveur actif", + "c6e5a03aa0": "dev box", + "f1575f1e09": "client web", + "81444c4102": "URL d'appairage", + "104f4d7dbd": "appairage", + "4575341c77": "Choisissez le bureau local, ajoutez un serveur Orca distant enregistré ou générez une URL d'appairage." + } + } + }, + "shortcuts": { + "search": { + "ca6a0c2df7": "raccourci", + "groupShortcut": "Raccourci {{value0}}", + "4811a8264a": "terminal d'abord", + "afda131738": "Orca d'abord", + "0ecfc47434": "conflit", + "0f8cb15582": "agent", + "f1adebbe8c": "shell", + "7f1b38f59a": "TUI", + "7e3fc707aa": "terminal", + "0ecba9aa5f": "clavier", + "ebd7d81e1d": "Choisissez si Orca ou le terminal ciblé l'emporte lorsque des raccourcis se chevauchent.", + "f052906167": "Raccourcis dans le terminal" + } + }, + "ssh": { + "search": { + "d41f296f64": "ping", + "237b391f7c": "connexion", + "8cb870b109": "test", + "7efd17e816": "ssh", + "96ca5d9a0b": "Testez la connectivité vers une cible SSH.", + "a3058f3605": "Tester la connexion", + "2cd40ba0d0": "hôtes", + "5220501141": "config", + "3b12e064a4": "importer", + "7f251a45a8": "Importez les hôtes depuis ~/.ssh/config.", + "41a3127094": "Importer depuis la config SSH", + "f9493b80c0": "serveur", + "8fb1cc87cc": "hôte", + "09395490af": "cible", + "00d1fda01a": "nouveau", + "f7b6383aec": "ajouter", + "62826efbe9": "Ajoutez une nouvelle cible SSH distante.", + "f5a691bb6c": "Ajouter une cible SSH", + "d4bcd497c7": "distant", + "74c6d90d78": "Gérez les cibles SSH distantes.", + "380a788da7": "Connexions SSH" + } + }, + "tasks": { + "search": { + "58cda6f9c0": "masquer", + "44083ae418": "écran", + "604d8e4089": "atlassian", + "5430396e11": "jira", + "412ec3c702": "linear", + "11f001cdd4": "gitlab", + "c10ac2125e": "github", + "3d81c26d78": "source", + "cf0e3e0c2f": "fournisseur", + "2ec54bee51": "tâches", + "5b8e4aace5": "Fournisseurs de tâches", + "providersDescription": "Connectez des fournisseurs de tâches, installez le skill d'agent Linear et choisissez ce qui apparaît dans Tâches.", + "setup": "configuration", + "apiKey": "clé API", + "skill": "skill", + "connect": "connect" + } + }, + "terminal": { + "clipboard": { + "search": { + "5fb3512e8c": "coller", + "a38508c419": "copier", + "d106f44fb4": "distant", + "043b32faa1": "ssh", + "9fda309db9": "fzf", + "64533e30cc": "nvim", + "2061d8db1a": "neovim", + "5ffcd13c90": "tmux", + "10d73e22d3": "presse-papiers", + "9dfc125cd3": "osc52", + "62d1208b90": "osc 52", + "459fea094a": "Laissez les programmes du terminal copier vers le presse-papiers système via OSC 52, y compris par-dessus SSH.", + "74db8721e4": "Autoriser les écritures presse-papiers des TUI (OSC 52)", + "4043e294d2": "gnome", + "cf83ac3dbd": "linux", + "737cef6de1": "x11", + "e87c6d776d": "automatique", + "664789b73a": "auto", + "c38c18be15": "sélection", + "797fdfe4ca": "sélectionner", + "603818e8d8": "Copiez automatiquement les sélections du terminal vers le presse-papiers dès qu'une sélection est faite.", + "3bdc84f059": "Copie à la sélection" + } + }, + "search": { + "c047f398cc": "lancer", + "b872de3926": "emplacement", + "fd6c24313d": "nouveau", + "f44643328e": "onglet", + "18ce996647": "vertical", + "54a9b3725b": "horizontal", + "de7bc1d5f5": "scindé", + "7a48c7715b": "espace de travail", + "6b659fff2a": "script", + "4529806908": "configuration", + "2610ee3b56": "Où s'exécute le script de setup du dépôt lorsqu'un nouvel espace de travail est créé : un fractionnement vertical (par défaut), un fractionnement horizontal ou un onglet en arrière-plan intitulé « Setup ».", + "5be2d67678": "Emplacement du script de configuration", + "0ce176909a": "thème", + "4ba8623632": "palette", + "11fd3fbcf2": "ansi", + "d8bd6182b8": "remplacement", + "674b7c8436": "couleur", + "3023e01415": "Remplacer individuellement les couleurs du terminal.", + "aed2a4b4eb": "Surcharges de couleurs", + "6eaf7ee0e4": "cursor", + "34fe1af39d": "saisie", + "ee611ae238": "masquer", + "ea364ce6e4": "souris", + "77201c0bb2": "Masquer le curseur de la souris pendant la saisie dans le terminal.", + "d1fe5f99ff": "Masquer la souris pendant la saisie", + "f25d948664": "marge", + "b2f52cb96c": "espacement", + "e8baf0d12c": "marge interne", + "4655567c37": "Marge verticale autour de la grille du terminal, en pixels.", + "692c4ad032": "Marge verticale", + "75691e4911": "Marge horizontale autour de la grille du terminal, en pixels.", + "b4f182f24d": "Marge horizontale", + "6c2f9f05c8": "vibrance", + "4f7f8f28ca": "transparence", + "f6dd9ff606": "arrière-plan", + "71eb45e293": "flou", + "0838b3717b": "fenêtre", + "bc2054657a": "Applique un flou d'arrière-plan à la fenêtre du terminal. Nécessite un redémarrage.", + "72d0482137": "Flou de fenêtre", + "7db59c4738": "alpha", + "46d99ef4bb": "opacité", + "4c643695aa": "Contrôle la transparence de l'arrière-plan du terminal.", + "b36fd2416d": "Opacité de l'arrière-plan", + "d4daf4f612": "dégeler", + "88561b3499": "figé", + "0a05629060": "récupérer", + "f66a7cf715": "terminal", + "6892fb1019": "restart", + "cde233f5da": "scrollback", + "3982d88725": "historique", + "456da64d4d": "effacer", + "920573d65b": "tout tuer", + "a3e5297c10": "tuer", + "a8d2784214": "gérer", + "d802a578bf": "sessions", + "9f2dda133c": "pty", + "f35400f7e8": "daemon", + "f72abc493c": "Récupérez des terminaux figés en tuant des sessions, en effaçant le scrollback enregistré ou en redémarrant le daemon.", + "6f5d486a68": "Gérer les sessions", + "10f9fb6fea": "paramètres", + "2ade3ea490": "config", + "fd752b3cac": "importer", + "73e9422f19": "Import unique des paramètres de terminal Ghostty pris en charge.", + "a979df0083": "Importer depuis Ghostty", + "warp_import": { + "title": "Importer depuis Warp", + "description": "Importez des thèmes Warp comme thèmes de terminal Orca.", + "keyword_warp": "warp", + "keyword_themes": "thèmes", + "keyword_yaml": "yaml", + "keyword_legacy_title": "Importer des thèmes depuis Warp" + }, + "yaml_import": { + "title": "Importer depuis YAML", + "description": "Importez des fichiers YAML de thèmes comme thèmes de terminal Orca.", + "keyword_yaml": "yaml", + "keyword_custom": "custom" + }, + "4cec42dbf7": "intl", + "b495dc6a9f": "jis", + "d8d6f7a3c5": "macos", + "1ab57a0fbd": "Mac", + "abaa24752d": "clavier", + "24f7977756": "japonais", + "98059d0944": "antislash", + "9c35f56625": "yen", + "063914c486": "Détermine si la touche Yen JIS (¥) envoie un antislash (\\) à la place.", + "694b8764ac": "Yen JIS (¥) en antislash (\\)", + "fae142a354": "readline", + "b3b94cfcb5": "international", + "dd4f6cb541": "allemand", + "983d45cf4c": "composer", + "7ace5beec9": "Meta", + "38f1b4f4cb": "touche", + "c4427dc5ff": "Alt", + "b37edfc65a": "Option", + "1f8b00f5ce": "Détermine si la touche Option de macOS envoie des séquences Alt/Esc ou compose des caractères. Équivaut à macos-option-as-alt de Ghostty.", + "9bd7229927": "Option comme Alt", + "affb14efd4": "sélection", + "d2a366c7f9": "double-clic", + "4ed3e239a8": "limite", + "d4aeafac10": "séparateur", + "7286cd2566": "mot", + "3ab64c47d8": "Caractères traités comme séparateurs de mots pour la sélection par double-clic.", + "957a0203fc": "Séparateurs de mots", + "56fff3d113": "mémoire", + "fffdff40a7": "buffer", + "rows": "lignes", + "f7d56b6281": "Lignes de terminal de bureau conservées.", + "7674e758e1": "Lignes de scrollback", + "411229c636": "clair", + "781f49d942": "diviseur", + "77d9f9cd55": "Contrôle la ligne de séparation entre volets en mode clair.", + "595b97b446": "Couleur du séparateur en mode clair", + "7718d70356": "aperçu", + "1dee533bd9": "Choisissez le thème utilisé quand Orca est en mode clair.", + "1d89457764": "Thème clair", + "da864e6cec": "mode clair", + "f268092ee3": "Le mode clair peut utiliser son propre thème de terminal.", + "match_dark_mode_title": "Aligner sur le mode sombre", + "f785374072": "sombre", + "9c32726f47": "Contrôle la ligne de séparation entre volets en mode sombre.", + "8987db7ff2": "Couleur du séparateur en mode sombre", + "13f6310dd3": "Choisissez le thème de terminal utilisé en mode sombre.", + "ec07ce9b02": "Thème sombre", + "f036794286": "actif", + "846a7a1204": "volet", + "d1fa00a9cb": "survol", + "b5116e7b12": "suit", + "f5d1e3d472": "focus", + "17cc3ea102": "Le survol d'un volet de terminal l'active sans clic nécessaire. Équivaut au paramètre focus-follows-mouse de Ghostty. Les sélections et le changement de fenêtre restent sûrs.", + "c6178a2b4d": "Le focus suit la souris", + "f637a7dee9": "épaisseur", + "e58d4040d0": "Épaisseur de la ligne de séparation des volets.", + "2d5ab88b7c": "Épaisseur du séparateur", + "6c4c85ba43": "atténuation", + "18dd5026c6": "Opacité appliquée aux volets qui ne sont pas actuellement actifs.", + "72bbcbd1dd": "Opacité des volets inactifs", + "d4f7d1ce5c": "Opacité du curseur du terminal.", + "7f1e356a54": "Opacité du curseur", + "25f606d9e5": "clignotement", + "a27f6edf52": "Utilise la variante clignotante de la forme de curseur sélectionnée.", + "b03d01fd49": "Curseur clignotant", + "eefd1d8332": "souligné", + "015c82349f": "bloc", + "a6e9dcc829": "barre", + "275a9d6395": "Apparence par défaut du curseur pour les volets de terminal Orca.", + "97bcfff662": "Forme du curseur", + "1abcf4d7de": "linux", + "7d924d870d": "graphismes", + "bc7ae1f7c0": "rendu", + "fffa9ab980": "moteur de rendu", + "6cddc858ba": "WebGL", + "4b4e80d850": "accélération", + "db82cb13b0": "GPU", + "8f9f953de7": "Détermine si le terminal utilise le rendu WebGL de xterm.js. Auto essaie WebGL quand le moteur de rendu est pris en charge, avec repli prudent pour un rendu logiciel ou un GPU inconnu.", + "13a2502dfc": "Accélération GPU", + "d5e6c7fab1": "fonctionnalités de police", + "a16224d16a": "calt", + "6ded6297fe": "iosevka", + "e3aeea308e": "cascadia code", + "35c2311a33": "jetbrains mono", + "7f7640c29e": "fira code", + "7ab424c4d3": "ligature", + "afc8d5f790": "ligatures", + "103cdb862f": "typographie", + "893aa92997": "Affiche les ligatures de programmation (p. ex. => → ≠ ≥) pour les polices qui en proposent. « Auto » n'active les ligatures que pour les polices à ligatures connues (Fira Code, JetBrains Mono, Cascadia Code, Iosevka, etc.).", + "58da1ae45d": "Ligatures de police", + "7341e3d00e": "hauteur de ligne", + "36a1b38bc8": "Contrôle le multiplicateur de hauteur de ligne du terminal.", + "0f2fb0cb74": "Hauteur de ligne", + "20ce287cc6": "graisse", + "98c18f2c77": "Contrôle la graisse de la police du texte du terminal.", + "28ea41bd2d": "Graisse de la police", + "b0bb76ae6b": "police", + "0acdc17891": "Famille de polices du terminal par défaut pour les nouveaux volets et les mises à jour en direct.", + "e989914ad6": "Famille de polices", + "33031c1465": "taille du texte", + "0fe0073f0c": "Taille de police du terminal par défaut pour les nouveaux volets et les mises à jour en direct.", + "5930244899": "Taille de police", + "ask_before_closing_running_terminals_title": "Demander avant de fermer les terminaux en cours d'exécution", + "ask_before_closing_running_terminals_description": "Afficher une confirmation avant de fermer un terminal où tourne une commande ou un agent.", + "scrollSpeed": { + "title": "Vitesse de défilement", + "description": "Ajustez le défilement normal du terminal, le défilement rapide avec modificateur et la vitesse de molette des TUI plein écran." + }, + "theme_target": { + "title": "Mode de thème", + "keyword_target": "cible", + "keyword_editing": "édition" + } + }, + "windows": { + "search": { + "4d09141a42": "menu contextuel", + "fcfa53920b": "coller", + "e55186fe2b": "clic droit", + "28ff08ed35": "windows", + "e7d2793b03": "terminal", + "8ba875c132": "Le clic droit colle le presse-papiers dans le terminal. Utilisez Ctrl+clic droit pour ouvrir le menu contextuel.", + "f0b8448570": "Clic droit pour coller", + "04994f6929": "default", + "fc564eadaf": "debian", + "4ee2579c32": "ubuntu", + "5074ad8b5f": "distro", + "2b4a340ce0": "distribution", + "02c772582a": "linux", + "6e3adf4cba": "wsl", + "978457945b": "Choisissez la distribution WSL utilisée par les nouveaux terminaux WSL et les analyses d'agents locaux.", + "1f402b3651": "Distribution WSL", + "d57f870938": "avancé", + "4af2f7526e": "version", + "d414022016": "pwsh", + "768613e483": "PowerShell 7", + "f9162f0b8e": "Windows PowerShell", + "2d99cd91be": "powershell", + "41a69bc24d": "Choisissez si l'option de shell PowerShell lance Windows PowerShell ou PowerShell 7+ pour les nouveaux volets de terminal.", + "860e0e6402": "Version de PowerShell", + "07ec155fb6": "bash.exe", + "5a2db98d23": "bash", + "591912177b": "Git Bash", + "12519edb5d": "invite de commandes", + "6cd20b9e64": "cmd", + "7c7056940a": "shell", + "713c4a2f92": "Choisissez le shell par défaut pour les nouveaux volets de terminal sous Windows.", + "13715f9d23": "Shell par défaut" + } + } + }, + "voice": { + "pane": { + "search": { + "f6e0dfa61c": "cloud", + "2d206de105": "clé API", + "04c25a6fb0": "openai", + "b9dee49cd7": "téléchargement", + "10d45a9fce": "STT", + "3d8b853963": "parole", + "080202facb": "model", + "7640ed9848": "voix", + "56defcd6c3": "Sélectionnez un modèle de reconnaissance vocale local ou cloud à utiliser pour la dictée.", + "7e62cd7c41": "Modèle vocal", + "931b1a9e53": "push-to-talk", + "064a9bd94a": "conserver", + "6fa48bcd41": "bascule", + "d86f5600da": "mode", + "089d31a45b": "dictée", + "748b33e531": "Comportement de dictée : bascule ou maintien pour parler.", + "6a3abb4338": "Mode de dictée", + "e360027a65": "micro", + "698376a38d": "Interrupteur principal des fonctions de dictée vocale.", + "20574cbc72": "Activer la dictée vocale", + "322d457a0d": "transcription", + "dcc7846641": "Configurez la clé d'API OpenAI utilisée pour les modèles de reconnaissance vocale dans le cloud.", + "ebfd0b32e5": "Transcription OpenAI", + "microphoneTitle": "Microphone", + "microphoneDescription": "Choisissez le périphérique d'entrée utilisé par la dictée vocale.", + "micInput": "entrée", + "micDevice": "appareil", + "micAirpods": "airpods", + "micDefault": "système par défaut" + } + } + }, + "source": { + "control": { + "action": { + "recipe": { + "options": { + "commitMessage": "Génère le message de commit à partir des changements indexés.", + "pullRequest": "Génère le titre et la description de la hosted review.", + "branchName": "Renomme les branches créées par Orca à partir de la tâche initiale de l'agent.", + "fixCommitFailure": "Démarre un agent lorsqu'un hook de commit ou un git commit échoue.", + "fixChecks": "Démarre un agent à partir des vérifications de hosted review en échec.", + "resolveConflicts": "Démarre un agent pour les conflits de merge locaux ou de hosted review.", + "customCommand": "Commande personnalisée", + "supportedAgents": "Agents pris en charge pour cette recette : {{value0}}.", + "unsupportedSavedAgent": "{{value0}} ne peut pas exécuter cette recette de génération de texte. Choisissez l'un des agents pris en charge ci-dessous.", + "resolveComments": "Démarre un agent à partir des commentaires PR ou MR non résolus sélectionnés.", + "fixPushFailure": "Démarre un agent lorsqu'un hook pre-push ou un git push échoue." + } + } + } + } + }, + "WorkspaceDirectorySetting": { + "1a2b3c4d5e": "Valeur par défaut du client", + "2b3c4d5e6f": "Appliquer à", + "3c4d5e6f7a": "Remplace la valeur par défaut du client", + "4d5e6f7a8b": "Hérite de la valeur par défaut du client", + "5e6f7a8b9c": "Réinitialiser", + "6f7a8b9cad": "Utilisez un chemin relatif (ex. .orca/worktrees) pour un emplacement par projet, ou un chemin absolu pour un dossier partagé unique." + }, + "agent-awake-copy": { + "e5995ce268": "Empêcher la mise en veille de l'ordinateur pendant que les agents travaillent", + "95d3031db2": "Maintient cet ordinateur et cet écran éveillés pendant que les agents travaillent. Le comportement à la fermeture du capot suit les paramètres d'alimentation de cet appareil.", + "a42f6fbdd8": "Maintient cet ordinateur et cet écran éveillés pendant que les agents travaillent. Orca demande également à cet appareil de rester éveillé lorsque le capot est fermé, sous réserve de sa politique d'alimentation.", + "modeTitle": "Empêcher la mise en veille de l'ordinateur", + "modeDescriptionWindows": "Choisissez Activé, Agent ou Désactivé. Le mode Agent reste éveillé tant que des agents travaillent ; le comportement à la fermeture du capot suit les paramètres d'alimentation de cet appareil.", + "modeDescriptionDefault": "Choisissez Activé, Agent ou Désactivé. Le mode Agent reste éveillé tant que des agents travaillent. Orca demande également à cet appareil de rester éveillé lorsque le capot est fermé, sous réserve de sa politique d'alimentation." + }, + "agent-status-hooks-copy": { + "7707c15abb": "Hooks de statut des agents", + "a68a642835": "Affiche les états en cours, en attente et terminé dans Orca. Désactivez pour supprimer les hooks gérés par Orca et arrêter de les réinstaller." + }, + "agent-generated-tab-title-copy": { + "19ad21615a": "Générer automatiquement les titres des onglets", + "b036c7a409": "Déduit des noms d'onglets courts et stables à partir de la première requête connue de l'agent. Les renommages manuels ont toujours priorité." + }, + "cli": { + "source": { + "control": { + "integration": { + "cards": { + "d5b3be8ecd": "Revérifier", + "8cbc39f862": "En savoir plus", + "707180d09c": "glab auth login", + "4be0616873": "La CLI GitLab est installée mais non authentifiée. Exécutez cette commande dans un terminal :", + "54a640af7a": "Installer la CLI GitLab", + "b56fd5676a": "Installez la CLI GitLab pour activer les merge requests, les issues et les pipelines.", + "faddeb763d": "Le statut de GitLab CLI n'est pas encore disponible dans ce runtime.", + "a47f71e357": "CLI.", + "2a6b359e75": "glab", + "1f2b347bd3": "Merge requests, issues, todos et pipelines via la", + "8d90249d22": "gh auth login", + "2e44dda68a": "La CLI GitHub est installée mais non authentifiée. Exécutez cette commande dans un terminal :", + "7755c28af5": "Installer la CLI GitHub", + "23cb5a0dee": "Installez la CLI GitHub pour activer les pull requests, les issues et les checks.", + "6f30fc4216": "Le statut de GitHub CLI n'est pas encore disponible dans ce runtime.", + "6b2cfb52b4": "gh", + "b4d900e7f1": "Pull requests, issues et checks via la", + "account_scope_prefix": "Portée des comptes", + "statusConnected": "Connecté", + "statusUnavailable": "Indisponible", + "statusNotInstalled": "Non installé", + "statusNotAuthenticated": "Non authentifié" + } + } + } + } + }, + "task": { + "tracker": { + "integration": { + "cards": { + "c90f2ef419": "Revérifier", + "dd3529015d": "Déconnecter {{value0}}", + "8b2408a8e5": "Jira est connecté pour ce runtime. Relancez la vérification si la liste des sites connectés semble obsolète.", + "8c20e76308": "Chaque site Jira connecté dispose d'un jeton stocké par le runtime actif.", + "c24e56c532": "Tester", + "3e7c10d286": "Test en cours...", + "a2c0015fb8": "Vérifié", + "e2ff968276": "Connecter Jira", + "60996beda6": "Ajouter un site Jira", + "7ca5ffffdb": "Parcourez, créez et démarrez du travail depuis les tickets Jira Cloud.", + "a1093a06c7": "Vérification de l'accès Jira avant d'afficher les actions de configuration.", + "9fa04a032e": "{{value0}} site{{value1}} connecté", + "cef18762a2": "Ajoutez l'accès avec une clé d'API personnelle depuis vos paramètres Linear. Les clés à accès complet peuvent voir toutes les équipes accessibles au propriétaire de la clé.", + "6224fe9d34": "Chaque espace de travail Linear connecté possède une clé stockée par le runtime actif. Les clés à accès complet peuvent couvrir toutes les équipes auxquelles le propriétaire de la clé a accès ; les clés restreintes peuvent être remplacées à tout moment.", + "1a12e33fe5": "Ajouter l'accès Linear", + "622c224082": "Ajouter un accès à l'espace de travail", + "eae4a9f16b": "Ajoutez un accès Linear pour parcourir et lier des issues.", + "fe9231215b": "Vérification de l'accès Linear avant d'afficher les actions de configuration.", + "e1f5e6424c": "Connecté{{value1}} : {{value0}} espace de travail", + "disconnect_all": "Déconnecter", + "account_scope_prefix": "Portée des comptes", + "2d60ec7921": "Connectez un site Jira Cloud avec un jeton d'API, ou un Jira auto-hébergé avec un jeton d'accès personnel ou un nom d'utilisateur et mot de passe. Les identifiants sont envoyés au runtime distant sélectionné et y sont stockés avec le chiffrement pris en charge par le runtime.", + "977e360b71": "Connectez un site Jira Cloud avec un jeton d'API, ou un Jira auto-hébergé avec un jeton d'accès personnel ou un nom d'utilisateur et mot de passe. Les identifiants sont stockés localement et chiffrés lorsque le stockage du runtime local le prend en charge.", + "statusConnected": "Connecté", + "statusNotConnected": "Non connecté" + } + } + } + }, + "token": { + "source": { + "control": { + "integration": { + "cards": { + "793a06e899": "Revérifier", + "1a9475dace": "En savoir plus", + "19fb419c12": "Les identifiants Gitea sont configurés mais l'authentification a échoué. Vérifiez le token, l'URL de base de l'API et les permissions du dépôt, puis redémarrez Orca si les variables d'environnement ont changé.", + "60708f23da": "uniquement quand Orca ne peut pas déduire l'URL de l'API depuis le remote.", + "709057ad91": "ORCA_GITEA_API_BASE_URL", + "6da9dfa5de": "pour les dépôts privés, et définissez", + "6d5c2a3005": "ORCA_GITEA_TOKEN", + "fcbe0469fd": "Les dépôts publics sont détectés via leur remote git. Définissez", + "0613928cb3": "Le statut de Gitea n'est pas encore disponible dans ce runtime.", + "05863d2599": "Pull requests et statuts de commits via l'API REST Gitea.", + "52f75876be": "Pull requests et statuts de commits pour les dépôts détectés", + "0b5242f8a2": "{{value0}} · Pull requests et statuts de commits", + "40f678df73": "Les identifiants Azure DevOps sont configurés mais l'authentification a échoué. Vérifiez le token, l'URL de base de l'API et les permissions du dépôt, puis redémarrez Orca si les variables d'environnement ont changé.", + "7bd345e3f6": "uniquement quand Orca ne peut pas déduire l'URL de base de l'API depuis le remote git.", + "186a6689df": "ORCA_AZURE_DEVOPS_API_BASE_URL", + "b8a10b07c1": ". Définissez", + "fbfd237f5e": "ORCA_AZURE_DEVOPS_ACCESS_TOKEN", + "087feb92f1": ", ou définissez", + "48842720d2": "ORCA_AZURE_DEVOPS_TOKEN", + "7bbc9c64f0": "Définissez", + "f3f47dc7de": "Le statut d'Azure DevOps n'est pas encore disponible dans ce runtime.", + "0eb50d5593": "Pull requests et statuts de builds via des tokens de l'API REST Azure DevOps.", + "54636c65d4": "Pull requests et statuts de builds pour les Azure Repos détectés", + "ea204f5e03": "{{value0}} · Pull requests et statuts de builds", + "6154b02093": "Les identifiants Bitbucket sont configurés mais l'authentification a échoué. Vérifiez le token et les permissions du dépôt, puis redémarrez Orca si les variables d'environnement ont changé.", + "e63fe8f627": "ORCA_BITBUCKET_ACCESS_TOKEN", + "19416c874c": "ORCA_BITBUCKET_API_TOKEN", + "fc71a0e7aa": "et", + "63a7f47392": "ORCA_BITBUCKET_EMAIL", + "24ac1c69dc": "Le statut de Bitbucket n'est pas encore disponible dans ce runtime.", + "a924e8dcd1": "Pull requests et statuts de builds via des tokens de l'API Bitbucket Cloud.", + "0fa5629dad": "Pull requests et statuts de builds", + "statusConnected": "Connecté", + "statusUnavailable": "Indisponible", + "statusNotConfigured": "Non configuré", + "statusAuthFailed": "Échec de l'authentification", + "statusConfigured": "Configuré", + "statusOptionalSetup": "Configuration facultative" + } + } + } + } + }, + "ProviderHostScopeControl": { + "scope_label": "{{value0}} : {{value1}}", + "change_host": "Ouvrir les serveurs distants" + }, + "computerUseSkillRuntime": { + "thisDevice": "Cet appareil" + }, + "computerUseSummary": { + "checkingTitle": "Vérification de l'accès Computer Use.", + "checkingDescription": "Orca vérifie les autorisations de confidentialité macOS pour l'assistant Computer Use.", + "unavailableTitle": "Computer Use est indisponible.", + "unavailableDescription": "Les autorisations Computer Use sont indisponibles car {{value0}}.", + "readyTitle": "Computer Use est prêt.", + "readyDescription": "Les agents peuvent inspecter et manipuler des fenêtres d'applications à votre demande.", + "permissionsTitle": "Terminez la configuration pour utiliser les applications locales.", + "permissionsRequired_one": "1 autorisation requise avant que les agents puissent manipuler des fenêtres d'applications.", + "permissionsRequired_other": "{{value0}} autorisations requises avant que les agents puissent manipuler des fenêtres d'applications." + }, + "providerAccountScope": { + "remoteServer": "Serveur distant : {{value0}}", + "remoteServerCredentials": "Les identifiants et vérifications de compte pour ce fournisseur appartiennent à ce serveur distant. Utilisez Paramètres > Serveurs distants Orca > Avancé pour modifier la portée d'un autre runtime par défaut.", + "localMac": "Mac local", + "localCredentials": "Les identifiants et vérifications de compte pour ce fournisseur appartiennent à ce client de bureau. Utilisez Paramètres > Serveurs distants Orca > Avancé pour modifier les identifiants détenus par le serveur.", + "remoteServerRateLimit": "Le budget d'API {{value0}} est récupéré via la CLI sur ce serveur distant. Utilisez Paramètres > Serveurs distants Orca > Avancé pour afficher le budget d'un autre runtime par défaut.", + "localRateLimit": "Le budget d'API {{value0}} est récupéré via la CLI sur ce client de bureau. Utilisez Paramètres > Serveurs distants Orca > Avancé pour afficher les budgets détenus par le serveur." + }, + "settingOwnership": { + "clientDefault": "Valeur par défaut du client", + "sourceControlAiDefaults": "Recettes, prompts et valeurs par défaut de hosted review sont partagés par ce client ; les choix de modèle et la découverte restent limités à l'hôte où l'agent s'exécute.", + "projectOnThisHost": "Projet sur cette machine", + "repositorySourceControlAi": "Ces substitutions s'appliquent à cette configuration de projet et héritent des valeurs par défaut IA du contrôle source côté client tant qu'elles ne sont pas personnalisées.", + "agentLaunchDefaults": "L'agent par défaut, les substitutions de commandes, les arguments CLI et l'environnement de lancement sont des préférences du client. Les lancements SSH et serveur distant valident toujours la disponibilité de l'hôte à l'exécution.", + "clientDefaultProjectScopes": "Par défaut du client + portées projet", + "terminalQuickCommands": "Les commandes sont enregistrées sur ce client, puis définies globalement ou pour une configuration de projet afin de s'exécuter depuis le contexte de terminal sélectionné.", + "hostOverride": "Substitution par hôte", + "workspaceDirectory": "La valeur par défaut du client est héritée tant qu'un hôte n'a pas besoin de son propre répertoire de worktrees.", + "providerHost": "Hôte du fournisseur", + "providerAccounts": "Les identifiants et vérifications de compte appartiennent au client local ou au serveur distant sélectionné qui détient l'intégration du fournisseur.", + "hostCollectionProjectScopes": "Collection de l'hôte + portées projet", + "terminalQuickCommandHostCollections": "Les commandes sont enregistrées sur l'hôte Orca sélectionné, puis définies globalement ou pour une configuration de projet. Les commandes de cet appareil restent également disponibles dans les espaces de travail distants." + }, + "RepositoryForkSyncSection": { + "defaultBranch": "branche par défaut", + "synced": "Fork mis à jour", + "syncedDescriptionSingular": "Avance rapide de {{branch}} de 1 commit.", + "syncedDescriptionPlural": "Avance rapide de {{branch}} de {{count}} commits.", + "upToDate": "Fork déjà à jour", + "upToDateDescription": "{{branch}} correspond déjà à upstream.", + "missingOrigin": "Le remote origin est absent.", + "missingUpstream": "Le remote upstream est absent.", + "upstreamMismatch": "Le remote upstream ne correspond plus à ce fork.", + "missingUpstreamBranch": "Impossible de déterminer la branche par défaut d'upstream.", + "missingOriginBranch": "origin n'a pas la branche par défaut d'upstream.", + "diverged": "origin contient des commits absents d'upstream.", + "blocked": "Synchronisation du fork ignorée", + "blockedFallback": "Orca n'a pas pu avancer ce fork en fast-forward en toute sécurité.", + "failed": "Échec de la synchronisation du fork", + "title": "Garder le fork à jour", + "description": "Faites avancer ce fork en fast-forward depuis upstream, sans risque.", + "longDescription": "Quand ce fork est en retard sur upstream, Orca peut avancer sa branche par défaut en fast-forward sans risque. Orca ignore la mise à jour si la branche contient des commits uniquement locaux ou des conflits.", + "forkOf": "Fork de {{owner}}/{{repo}}", + "syncing": "Synchronisation en cours", + "syncNow": "Synchroniser maintenant", + "modeLabel": "Mode de synchronisation du fork", + "ask": "Demander", + "safeAuto": "Auto sécurisé", + "off": "Désactivé" + }, + "DefaultWindowsProjectRuntimeSetting": { + "defaultRuntime": "Runtime de projet par défaut", + "windows": "Windows", + "wsl": "WSL", + "selectDistro": "Sélectionner une distro", + "windowsDescription": "Les projets héritent de Windows sauf si un projet le remplace.", + "wslUnavailable": "WSL n'est pas disponible. Les projets qui héritent de WSL devront être réparés.", + "distroRequired": "Choisissez une distro WSL avant que les projets puissent hériter de WSL.", + "wslDescription": "Les projets héritent de {{value0}} via WSL sauf si un projet le remplace." + }, + "ProjectWindowsRuntimeSetting": { + "projectRuntime": "Runtime du projet", + "defaultRuntime": "Par défaut ({{value0}})", + "windows": "Windows", + "wsl": "WSL", + "selectDistro": "Sélectionner une distro", + "runtimeChangeHelp": "Les changements de runtime s'appliquent aux nouveaux terminaux, aux vérifications d'agents et à la découverte de skills pour ce projet. Les terminaux existants conservent leur runtime actuel.", + "wslUnavailable": "WSL n'est pas disponible. Basculez ce projet sur Windows ou réparez WSL.", + "distroMissing": "{{value0}} n'est pas installé dans WSL. Choisissez une distro installée ou basculez ce projet sur Windows.", + "distroRequired": "Choisissez une distro WSL ou basculez ce projet sur Windows.", + "inheritedWsl": "Aucune substitution au niveau projet. Les paramètres généraux sélectionnent {{value0}} via WSL.", + "projectWsl": "Ce projet s'exécute dans {{value0}} via WSL.", + "inheritedWindows": "Aucune substitution au niveau projet. Les paramètres généraux sélectionnent Windows.", + "projectWindows": "Ce projet s'exécute sous Windows.", + "liveTerminalSingular": "{{count}} terminal actif", + "liveTerminalPlural": "{{count}} terminaux actifs", + "activeTaskSingular": "{{count}} tâche active", + "activeTaskPlural": "{{count}} tâches actives", + "runtimeSessionJoin": "{{value0}} et {{value1}}", + "runtimeSessionWarning": "{{value0}} continuera de s'exécuter dans le runtime actuel. Laissez les tâches se terminer ou redémarrez les terminaux avant de continuer.", + "pendingRuntimeChange": "Changement de runtime en attente. Le nouveau travail sur le projet utilisera le runtime sélectionné après application.", + "cancel": "Annuler", + "applyRuntimeChange": "Appliquer le changement de runtime" + }, + "PrivacyDiagnosticsRows": { + "5a7cbe069a": "DO_NOT_TRACK=1 est défini — la création et l'envoi de fichiers de diagnostic sont désactivés.", + "63d03261d1": "ORCA_TELEMETRY_DISABLED=1 est défini — la création et l'envoi de fichiers de diagnostic sont désactivés.", + "d37e92a06b": "ORCA_DIAGNOSTICS_DISABLED=1 est défini — les diagnostics de l'app sont désactivés.", + "5ebb31e1fb": "Exécution en CI — diagnostics désactivés.", + "e27c8d45bf": "Les diagnostics sont désactivés par une variable d'environnement." + }, + "ShortcutCommandBlock": { + "eb72c52c28": "Appuyez sur un raccourci, ou appuyez deux fois sur une touche modificatrice (ex. {{value0}}). Échap annule.", + "70b5d25583": "Raccourci {{value0}}", + "287e07ddde": "Modifié", + "3c83cd7d1c": "Désactivé", + "07939d084e": "Réinitialiser {{value0}} à la valeur par défaut", + "9b02917027": "Rétablir la valeur par défaut", + "a799f90f82": "Désactiver {{value0}}", + "25e6e76618": "Désactiver le raccourci", + "035a822ef0": "Ajouter un raccourci", + "a0e2ef0e61": "Ajouter un autre raccourci pour {{value0}}", + "245c83af24": "Ajouter un autre raccourci", + "482a60225d": "Activer {{value0}}", + "6287677c37": "Activer", + "01481b964c": "Ajouter un raccourci pour {{value0}}" + }, + "ShortcutRecorderButton": { + "1a13bb054d": "Appuyez sur les touches du raccourci pour {{value0}}. Échap annule.", + "3732775d74": "Ajouter un raccourci pour {{value0}}", + "88764af2c1": "Modifier le raccourci pour {{value0}}", + "30feb099d6": "Modifier le raccourci {{value0}} sur {{value1}} pour {{value2}}", + "5d982a2a1f": "En écoute du raccourci", + "152e0bcd64": "Ajouter un raccourci", + "5bd56445da": "Modifier le raccourci", + "f5ed5dcbf6": "Appuyez sur des touches…" + }, + "ShortcutRemoveButton": { + "9e29aff18b": "Supprimer le raccourci {{value1}} de {{value0}}", + "2a9588b1c2": "Supprimer cette association" + }, + "BrowserLocalhostWorktreeLabelsSetting": { + "8ac8c3ad19": "Libellés localhost des worktrees", + "1db3c8b983": "Ouvrir les ports des espaces de travail sous forme d'URL localhost Orca propres à chaque worktree, pour mieux distinguer les onglets du navigateur." + }, + "AgentRuntimeSetting": { + "label": "Runtime des agents", + "wsl": "WSL", + "loadingWsl": "Chargement de WSL", + "selectDistro": "Sélectionner une distro", + "windowsDescription": "Détecte et lance les agents sous Windows pour les projets qui ne remplacent pas leur runtime.", + "wslUnavailable": "WSL n'est pas disponible sur cette machine.", + "distroRequired": "Choisissez une distro WSL avant que les projets puissent hériter de WSL.", + "wslDescription": "Détecte et lance les agents dans {{value0}} via WSL pour les projets qui ne remplacent pas leur runtime." + }, + "MobileEmulatorSdkStatus": { + "536026130e": "SDK d'émulateurs", + "dde0ec1cd8": "Chaînes d'outils qu'Orca utilise pour exécuter des émulateurs. Android fonctionne sur tout OS via le SDK Android ; les simulateurs iOS nécessitent Xcode sur macOS.", + "027cbf668a": "Android SDK", + "f6d080d128": "Utilise le chemin configuré", + "7fe4bd5907": "Détecté à", + "2784f0b22d": "Introuvable. Installez Android Studio, puis créez un Virtual Device.", + "b94ff260e6": "Télécharger Android Studio", + "18925b082d": "Localiser le dossier du SDK", + "8c52684db8": "Effacer", + "76eb88b88e": "Simulateur iOS (Xcode)", + "c6f3ea4f12": "Prêt", + "e4f14b50d7": "Installez Xcode et ajoutez un runtime de simulateur iOS.", + "63fe73a1ea": "Impossible de mettre à jour le dossier du SDK Android." + }, + "AppearanceAdvancedDisclosure": { + "advanced": "Avancé" + }, + "DevToolsPane": { + "nativeActionLayoutCheck": "Vérification de la disposition des actions natives", + "secondary": "Secondaire", + "secondaryActionClicked": "Action secondaire cliquée", + "primaryAction": "Action principale", + "primaryActionClicked": "Action principale cliquée", + "branchHasChanges": "la branche a des modifications", + "viewChangesClicked": "« Voir les modifications » cliqué", + "forceDeleteClicked": "« Forcer la suppression » cliqué", + "devOnlyCallback": "Callback réservé au développement.", + "infoToast": "Toast d'information", + "infoToastDescription": "Texte informatif long sans action explicite.", + "localCanaryBehind": "Le canary local est en retard sur origin/canary", + "successToast": "Toast de succès", + "successToastDescription": "Court état de confirmation.", + "settingsSaved": "Paramètres enregistrés", + "shortSuccessCopy": "Court texte de succès.", + "errorToast": "Toast d'erreur", + "errorToastDescription": "Texte d'erreur long sans actions de récupération.", + "failedToSyncWorkspaceMetadata": "Échec de la synchronisation des métadonnées de l'espace de travail", + "nativeActionToast": "Toast avec action native", + "nativeActionToastDescription": "Utilise les emplacements action et cancel de Sonner.", + "deleteFailureToast": "Toast d'échec de suppression", + "deleteFailureToastDescription": "Pied personnalisé avec « Afficher » et « Forcer la suppression ».", + "behindBaseRefToast": "Toast de retard sur la ref de base", + "behindBaseRefToastDescription": "Invite persistante avec un lien Paramètres intégré et une action en pied de page.", + "notificationPlayground": "Terrain d'essai des notifications", + "notificationPlaygroundDescription": "Déclencheurs réservés au développement pour vérifier la disposition des toasts, les actions de récupération et le retour à la ligne des textes longs.", + "devOnly": "Réservé au développement", + "orcaCloud": "Orca Cloud", + "orcaCloudDescription": "Aperçu réservé au développement de la connexion au cloud officiel. Masqué en production ; en dev, il apparaît aussi dans le sélecteur de comptes de la barre latérale dès que ORCA_CLOUD_API_URL et ORCA_CLOUD_CLIENT_ID sont définis.", + "orcaCloudStatus": "Statut", + "orcaCloudSignOut": "Se déconnecter", + "orcaCloudConnect": "Connecter le profil", + "orcaCloudRefresh": "Actualiser le statut", + "orcaCloudNotConfigured": "Définissez ORCA_CLOUD_API_URL et ORCA_CLOUD_CLIENT_ID pour prévisualiser la connexion à Orca Cloud dans ce build de développement." + }, + "EphemeralVmRecipeRow": { + "useInWorkspace": "Utiliser dans l'espace de travail" + }, + "EphemeralVmRuntimesSection": { + "cleanupFailed": "Échec du nettoyage", + "cleanupRunning": "Nettoyage en cours", + "cleanupDisabled": "Nettoyage désactivé", + "running": "En cours", + "failed": "Échec", + "loadFailed": "Impossible de charger les runtimes de VM temporaires.", + "cleanupFailedToast": "Impossible de nettoyer le runtime de VM temporaire.", + "markedCleaned": "Runtime de VM temporaire marqué comme nettoyé.", + "cleaned": "Runtime de VM temporaire nettoyé.", + "copiedCleanupCommand": "Commande de nettoyage copiée.", + "copiedCleanupPayload": "Contenu de nettoyage copié.", + "copyCleanupFailed": "Impossible de copier la commande de nettoyage.", + "title": "Runtimes de VM temporaires", + "description": "Les runtimes créés par recette appartiennent à l'espace de travail. Nettoyez les entrées obsolètes après un plantage, une création échouée ou une récupération manuelle.", + "refresh": "Actualiser les runtimes de VM temporaires", + "loading": "Vérification des runtimes de VM temporaires…", + "empty": "Aucun runtime de VM temporaire à nettoyer.", + "copyCleanup": "Copier la commande", + "retry": "Réessayer le nettoyage", + "cleanup": "Nettoyage", + "cloudVmLoadFailed": "Impossible de charger les runtimes de VM Cloud.", + "cloudVmCleanupFailedToast": "Impossible de nettoyer le runtime de VM Cloud.", + "cloudVmMarkedCleaned": "Runtime de VM Cloud marqué comme nettoyé.", + "cloudVmCleaned": "Runtime de VM Cloud nettoyé.", + "cloudVmTitle": "Runtimes de VM Cloud", + "cloudVmRefresh": "Actualiser les runtimes de VM Cloud", + "cloudVmLoading": "Vérification des runtimes de VM Cloud…", + "cloudVmEmptyWithSetup": "Aucun runtime de VM Cloud pour l'instant. Créez-en un depuis un espace de travail avec une recette d'environnement.", + "stoppingCleanup": "Arrêt…", + "stopCleanup": "Arrêter le nettoyage", + "cleanupStopped": "Nettoyage arrêté", + "stopCleanupFailed": "Impossible d'arrêter le nettoyage de la VM Cloud." + }, + "ephemeralVms": { + "search": { + "description": "Découvrez comment les recettes détenues par le dépôt offrent à chaque workspace son propre environnement jetable à la demande.", + "cloudVmTitle": "VM cloud" + } + }, + "EphemeralVmsPane": { + "loadError": "Impossible de charger les recettes.", + "copyError": "Impossible de copier le prompt.", + "skillDescription": "Configure, construit, authentifie et valide les recettes d'environnement détenues par le dépôt.", + "whatTitle": "Ce que fait la skill, avec vous", + "whatScaffold": "Écrit la recette et les scripts pour votre fournisseur — connecté via un serveur Orca ou SSH.", + "whatBuild": "Construit une image de base réutilisable et connecte votre agent.", + "whatValidate": "La valide pour que vous puissiez créer un espace de travail dessus.", + "promptHint": "Dans n'importe quel espace de travail, demandez à votre agent :", + "copy": "Copier", + "copied": "Copied", + "recipes": "Recettes", + "recipesHelp": "Les recettes issues d'orca.yaml et des plugins activés apparaissent ici, prêtes à lancer un espace de travail.", + "checking": "Vérification des recettes...", + "none": "Aucune recette trouvée pour l'instant.", + "cloudVmSkillTitle": "Skill de configuration de VM Cloud", + "cloudVmRefresh": "Actualiser les recettes de VM Cloud", + "cloudVmTerminalTitle": "Configuration de VM Cloud", + "cloudVmTerminalAriaLabel": "Terminal d'installation de la skill VM Cloud" + }, + "ephemeralVmsExperimentalSetting": { + "description": "Affiche les contrôles de configuration et les cibles d'exécution espace de travail pour les environnements détenus par le dépôt, à la demande.", + "cloudVmToggleLabel": "Activer/désactiver Cloud VM" + }, + "SourceControlActionRepoOverrideNote": { + "agent": "Agent", + "agentArgs": "Arguments CLI", + "commandTemplate": "Modèle de commande", + "more": "+{{count}} autres", + "plural": "Les enregistrements globaux ne modifieront pas {{count}} dépôts ayant leurs propres recettes.", + "recipe": "Recette", + "review": "À examiner", + "reviewFirst": "Vérifier d'abord", + "singular": "Les enregistrements globaux ne modifieront pas 1 dépôt ayant sa propre recette.", + "tooltipTitle": "Substitutions au niveau dépôt" + }, + "linear": { + "agent": { + "skill": { + "install": { + "cta": { + "copiedCommand": "Commande copiée.", + "copyFailed": "Échec de la copie de la commande.", + "skillLabel": "Skill d'agent :", + "checking": "Vérification...", + "installed": "Installés", + "notInstalled": "Non installé", + "recheck": "Revérifier", + "installedDescription": "Skill d'agent installée. Pour la mettre à jour, exécutez :", + "description": "Permet aux agents de lire et modifier les tâches Linear. La configuration guidée complète (connexion + skill + visibilité) se trouve dans Paramètres → Task Sources.", + "copyCommand": "Copier la commande" + } + }, + "search": { + "title": "Linear", + "description": "Statut de la skill Linear, exemples d'utilisation et liens vers la configuration de Task Sources." + } + } + } + }, + "GrokAccountsSection": { + "a1b2c3d4e5": "Grok (xAI)", + "f6e5d4c3b2": "Affiche l'utilisation hebdomadaire de crédits issue de votre connexion Grok CLI (fichier de session ~/.grok/auth.json).", + "0d8e77bc40": "Documentation Grok CLI", + "ad47a33f72": "Chargement…", + "b2c3d4e5f6": "Connecté", + "c3d4e5f6a7": "Connecté. Orca lit uniquement ce fichier sur disque — relancez grok login si l'affichage de l'utilisation échoue.", + "d4e5f6a7b8": "Session expirée — exécutez grok login dans un terminal pour l'actualiser.", + "e5f6a7b8c9": "Non connecté à Grok CLI", + "f6a7b8c9d0": "Dans un terminal, exécutez grok login, puis cliquez ici sur Actualiser l'utilisation.", + "3325d996cb": "Actualiser l'utilisation", + "a8f3e2c1b4": "Crédits hebdomadaires", + "b7e2d9f0a3": "Même % de crédits hebdomadaires que l'écran grok /usage dans le terminal.", + "c6d1a8f4e2": "Réinitialisation {{when}}", + "e6dadc1e2b": "Utilisation mensuelle", + "75e396bf42": "Utilisation mensuelle incluse pour les comptes Grok à facturation unifiée.", + "b36fa2c908": "Connecté. Orca lit la session Grok CLI stockée sur disque.", + "f08c41de73": "Session expirée — lancez grok sur l'ordinateur qui exécute Orca et attendez son démarrage. Complétez la connexion si une invite apparaît, puis cliquez sur Actualiser l'utilisation. Aucun message dans le chat n'est nécessaire." + }, + "AppearanceWindowSidebarSection": { + "usagePercentageDisplayUsed": "Utilisé", + "usagePercentageDisplayRemaining": "Restant" + }, + "TerminalInteractionSection": { + "567633ff50": "Le clic droit colle le presse-papiers dans le terminal. Control-clic pour ouvrir le menu contextuel.", + "c64497148a": "Le clic droit colle le presse-papiers. Control-clic ouvre le menu contextuel." + }, + "MobileRelayStatusSection": { + "registered": "Enregistré", + "connecting": "Connexion", + "reconnecting": "Reconnexion", + "offline": "Hors ligne", + "title": "Orca Relay", + "automatic": "Connexion depuis n'importe où quand un téléphone est jumelé", + "signInPrompt": "Connectez-vous sur cet ordinateur pour accéder depuis n'importe où", + "directStillAvailable": "Les connexions LAN et Tailscale restent disponibles.", + "directNeedsNoAccount": "Le jumelage LAN et Tailscale fonctionne toujours sans compte.", + "signIn": "Se connecter", + "unavailable": "Indisponible", + "standby": "Veille — aucun appareil via Relay" + }, + "MobilePairingConnectionOptions": { + "ready": "Prêt", + "connecting": "Connexion", + "available": "Disponibles", + "reconnecting": "Reconnexion", + "unavailable": "Indisponible", + "pathGroup": "Comment le téléphone joint cet ordinateur", + "anywhereTitle": "Orca Relay", + "anywhereDescription": "Le téléphone peut être en données mobiles ou sur n'importe quel Wi‑Fi. Connexion requise uniquement pour Relay.", + "signInRequired": "Relay uniquement — le LAN n'a pas besoin de compte.", + "relayUnavailable": "Orca Relay n'est pas disponible dans ce build. Utilisez le LAN.", + "signIn": "Se connecter pour Relay", + "signInAgain": "Se reconnecter pour Relay", + "localTitle": "LAN", + "localDescription": "Le téléphone doit être sur ce Wi‑Fi ou connecté via Tailscale. Aucun compte requis.", + "retrying": "Nouvelle tentative" + }, + "MobilePairingSetupSection": { + "title": "Jumeler un téléphone", + "overview": "Générez un QR code, puis scannez-le dans Orca Mobile sous Jumeler le bureau.", + "step1Title": "Connexion", + "step2Title": "Adresse de cet ordinateur", + "step2LocalDescription": "Le téléphone doit pouvoir joindre cette adresse via Tailscale ou le Wi‑Fi.", + "regenerate": "Régénérer le QR code", + "generate": "Générer le QR code", + "refresh": "Actualiser les interfaces réseau", + "step2RelayDescription": "Facultatif. Choisissez l'adresse Wi‑Fi ou Tailscale que votre téléphone utilisera à proximité — généralement plus rapide que Relay. Relay reste fonctionnel quand vous êtes loin.", + "step2RelayDisclosure": "Utiliser aussi un chemin local plus rapide" + }, + "MobileRelayBetaAvailability": { + "about": "À propos de la bêta d'Orca Relay", + "beta": "Beta", + "availability": "Disponible sur", + "testFlight": "TestFlight", + "androidApk": "APK Android" + }, + "SkillUsageExampleDialog": { + "copiedPrompt": "Prompt d'exemple copié.", + "copyFailed": "Échec de la copie du prompt.", + "copyExampleAria": "Copier le prompt d'exemple {{value0}}", + "done": "Terminé", + "copyPrompt": "Copier le prompt" + }, + "LinearAgentSkillPane": { + "title": "Linear", + "description": "Fonctionnement de Linear dans Orca : parcourez les tickets, démarrez des espaces de travail liés et laissez les agents mettre à jour les tickets avec /orca-linear.", + "skillTitle": "Skill Linear", + "howToUse": "Prompts d'exemple", + "howToUseDescription": "Cliquez sur une carte pour copier un prompt. Utilisez-les dans un worktree lié à Linear après l'installation de la skill.", + "terminalTitle": "Configuration de la skill Linear", + "terminalAriaLabel": "Terminal d'installation de la skill Linear", + "manageConnectionHint": "Consultez les espaces de travail Linear connectés et les clés d'API dans", + "manageConnectionLink": "Intégrations" + }, + "MobileRelayBetaNotice": { + "notice": "Orca Relay est en bêta." + }, + "EditorFontFamilySetting": { + "title": "Police de l'éditeur", + "description": "Police utilisée par les éditeurs de fichiers et les vues de diff. Laissez vide pour suivre la police du terminal.", + "placeholder": "Identique à la police du terminal" + }, + "GeneralRemoteServerUpdates": { + "serverCount": "{{value0}} serveurs jumelés", + "availableCount": "{{value0}} prêts pour la mise à jour", + "currentCount": "{{value0}} à jour", + "manualCount": "{{value0}} manuels", + "offlineCount": "{{value0}} hors ligne", + "title": "Serveurs Orca distants", + "description": "Vérifiez et mettez à jour les serveurs Orca jumelés depuis ce client.", + "updating": "Mise à jour des serveurs…", + "reviewUpdates": "{{value0}} mises à jour disponibles", + "reviewServers": "Rechercher des mises à jour du serveur", + "serverCountOne": "1 serveur jumelé", + "reviewUpdateOne": "1 mise à jour disponible" + }, + "RemoteServerUpdateDialog": { + "versionUnavailable": "Version indisponible", + "restartingHelp": "En attente de la reconnexion du serveur de remplacement sur la nouvelle version.", + "retry": "Réessayer", + "update": "Mettre à jour ce serveur", + "title": "Mettre à jour les serveurs distants Orca", + "description": "Consultez les serveurs jumelés et mettez à jour les installations prises en charge depuis ce client Orca.", + "restartWarning": "La mise à jour redémarre ces serveurs. {{value0}} onglets actifs et {{value1}} volets de terminal peuvent se déconnecter brièvement.", + "checking": "Vérification des serveurs jumelés…", + "empty": "Aucun serveur distant Orca jumelé.", + "checkAgain": "Rechercher des mises à jour du serveur", + "updating": "Mise à jour des serveurs…", + "updateAll": "Mettre à jour tous les {{value0}} serveurs", + "noUpdates": "Tous les serveurs sont à jour.", + "updateOne": "Mettre à jour le serveur", + "downloadProgress": "Progression du téléchargement de {{value0}}", + "liveTabOne": "1 onglet actif", + "liveTabs": "{{value0}} onglets actifs", + "livePaneOne": "1 volet actif", + "livePanes": "{{value0}} volets actifs" + }, + "RemoteServerUpdateStatus": { + "checking": "Vérification…", + "available": "Mise à jour disponible", + "current": "À jour", + "manual": "Mise à jour manuelle", + "offline": "Hors ligne", + "queued": "En file d'attente", + "checkingUpdate": "Vérification de la mise à jour…", + "downloading": "Téléchargement…", + "restarting": "Redémarrage…", + "updated": "Mis à jour", + "failed": "Échec de la mise à jour", + "serviceManagerHelp": "Mettez à jour Orca via le gestionnaire de services qui démarre ce serveur.", + "unpackedHelp": "Les builds de développement doivent être mis à jour depuis leur checkout source.", + "legacyHelp": "Mettez à jour ce serveur une fois manuellement pour activer les mises à jour à distance." + }, + "BrowserLinkRoutingSetting": { + "description": "Ouvre les liens http(s) dans le navigateur intégré d'Orca — depuis le terminal, le markdown et l'éditeur. {{shortcut}} utilise toujours votre navigateur système.", + "descriptionBase": "Ouvre les liens http(s) dans le navigateur intégré d'Orca — depuis le terminal, le markdown et l'éditeur." + }, + "BrowserLinkRoutingModifierSetting": { + "titleSystem": "Maintenez Shift pour ouvrir dans votre navigateur web", + "titleOrca": "Maintenez Shift pour ouvrir dans Orca", + "descriptionSystem": "Les liens s'ouvrent dans Orca ; {{chord}}+clic les envoie plutôt à votre navigateur système.", + "descriptionOrca": "Les liens s'ouvrent dans votre navigateur système. Une fois activé, {{chord}}+clic en ouvre un dans le navigateur intégré d'Orca." + }, + "BrowserTerminalLinkActionsSetting": { + "title": "Afficher les actions de lien du terminal", + "description": "Affiche les actions disponibles quand vous cliquez sur un lien du terminal. Désactivez pour exiger un {{modifier}}+clic." + }, + "PluginConsentDialog": { + "workerTrust": "Worker en arrière-plan — exécute son propre processus", + "instructionalTrust": "Contenu instructif — s'exécute plus tard sous l'autorité de l'utilisateur ou d'un agent", + "declarativeTrust": "Contenu déclaratif — aucun code de plugin", + "panelTrust": "Contenu de panneau ou intégré à l'hôte — aucun processus worker", + "trustShortWorker": "Worker", + "trustShortInstructional": "Instructif", + "trustShortPanel": "Panneau", + "trustShortDeclarative": "Déclaratif", + "reviewTitle": "Examiner le plugin", + "title": "Examiner les autorisations", + "mixedTitle": "Examiner l'accès et le contenu", + "instructionalTitle": "Examiner le contenu du plugin", + "subtitle": "{{value0}} v{{value1}} · {{value2}}", + "reconsent": "Les autorisations, le niveau de confiance du worker ou le contenu instructif ont changé depuis votre dernière revue de ce plugin. Revoyez-le avant qu'il puisse s'exécuter.", + "capabilities": "Ce plugin peut", + "warning": "Ces autorisations limitent la façon dont le plugin utilise l'API d'Orca. Son worker s'exécute néanmoins comme un processus normal sur votre ordinateur, avec un accès complet à vos fichiers, votre réseau et vos autres processus.", + "instructionalWarning": "Ce plugin n'a pas de processus worker. Son contenu instructif peut malgré tout déclencher des actions quand vous ou un agent l'utilisez. Examinez les instructions et commandes ci-dessous avant de l'activer.", + "panelWarning": "Ces autorisations limitent la façon dont le plugin utilise l'API d'Orca. Ce plugin n'a pas de worker en arrière-plan.", + "declarativeWarning": "Ce plugin fournit uniquement du contenu validé. Il n'exécute pas de worker en arrière-plan ni ne reçoit d'accès à l'API d'Orca.", + "keepDisabled": "Garder désactivé", + "enable": "Activer le plugin", + "capability": { + "workspaceRead": "Lire le nom, la branche et la liste des terminaux de votre worktree actif", + "terminalSend": "Saisir du texte dans un terminal visible (toujours un terminal spécifique)", + "notificationsShow": "Afficher des notifications de bureau libellées avec le nom du plugin", + "storage": "Stocker des données dans le dossier de stockage propre au plugin", + "secrets": "Stocker et lire des secrets dans le coffre chiffré propre au plugin", + "eventsSubscribe": "Être notifié quand des worktrees sont créés ou supprimés et quand le statut d'un agent change", + "settingsOwn": "Lire et modifier les paramètres propres au plugin" + }, + "decisionFailed": "Impossible d'enregistrer la décision d'autorisation. Réessayez." + }, + "PluginConsentProvenance": { + "official": "Officiel", + "bundled": "Fourni avec Orca", + "local": "Dossier local", + "community": "Communauté", + "source": "Source", + "sourceLabel": "Source", + "commit": "Commit épinglé", + "localCommit": "Dossier local — aucun commit", + "indexCommit": "Commit d'index" + }, + "PluginDevelopmentSection": { + "saveFailed": "Impossible d'enregistrer les chemins de plugins de développement.", + "pathRequired": "Saisissez un chemin de dossier de plugin.", + "title": "Développement", + "help": "Chargez les plugins directement depuis des dossiers de cet ordinateur pendant leur développement. Les plugins de développement exigent toujours une revue des autorisations. Les workers s'exécutent sur cet hôte de bureau ; les actions SSH sur les espaces de travail passent par Orca, donc les chemins indiqués ici sont des chemins du bureau.", + "remove": "Supprimer", + "pathLabel": "Chemin du dossier de plugin de développement", + "placeholder": "/Users/you/plugins/my-plugin or C:\\Users\\you\\plugins\\my-plugin", + "add": "Ajouter un chemin" + }, + "PluginInstallDialog": { + "localRequired": "Saisissez le chemin du dossier du plugin.", + "gitUrlRequired": "Saisissez une URL de dépôt.", + "gitUrlInvalid": "Utilisez une URL Git HTTPS ou SSH. Les protocoles de helper Git exécutables ne sont pas autorisés.", + "gitRefRequired": "Ajoutez une #ref explicite (tag ou commit) pour épingler l'installation — par exemple #v0.1.0.", + "title": "Installer le plugin", + "description": "L'installation copie le plugin dans Orca et affiche ses autorisations pour revue. Aucun code de plugin ne s'exécute tant que vous ne l'activez pas.", + "source": "Source d'installation", + "localTab": "Dossier local", + "gitTab": "URL Git", + "localLabel": "Chemin du dossier du plugin", + "localPlaceholder": "/Users/you/plugins/my-plugin or C:\\Users\\you\\plugins\\my-plugin", + "localHelp": "Chemin complet vers un dossier contenant orca-plugin.json sur cet ordinateur. Le chemin est utilisé exactement tel que saisi.", + "gitLabel": "URL de dépôt avec #ref", + "gitPlaceholder": "https://git.example/acme/orca-notes#v0.1.0", + "gitHelp": "Ajoutez une #ref explicite — un tag ou un commit — pour épingler l'installation. Fonctionne avec GitHub, GitLab et tout hôte git.", + "cancel": "Annuler", + "installing": "Installation…", + "install": "Installer", + "installFailed": "Échec de l'installation du plugin. Vérifiez la source et réessayez." + }, + "PluginKeybindingConsentPreview": { + "heading": "Raccourcis clavier", + "worktree": "Ne fonctionne que lorsqu'un espace de travail est actif.", + "global": "Fonctionne dans l'app sans espace de travail actif requis.", + "shadows": "Remplace : {{value0}}" + }, + "PluginMarketplaceBrowser": { + "loadFailed": "Impossible de charger les plugins du marketplace.", + "refreshFailed": "Impossible d'actualiser les marketplaces. Les listes en cache restent disponibles.", + "previewFailed": "Impossible de préparer ce plugin pour revue. Actualisez le marketplace et réessayez.", + "installFailed": "Impossible d'installer ce plugin. La source examinée a peut-être changé.", + "manageSources": "Gérer les sources", + "refreshing": "Actualisation…", + "refresh": "Actualiser", + "noInstalledTitle": "Aucun plugin installé", + "noInstalled": "Les plugins que vous installez apparaissent ici.", + "loading": "Chargement des plugins du marketplace…", + "tryAgain": "Réessayer", + "noSourcesTitle": "Aucun marketplace configuré", + "noSources": "Ajoutez un marketplace Git officiel, communautaire ou privé pour parcourir les plugins.", + "addSource": "Ajouter un marketplace", + "noResultsTitle": "Aucun plugin correspondant", + "noResults": "Aucun plugin du marketplace ne correspond à cette recherche.", + "clearSearch": "Effacer la recherche", + "emptyTitle": "Rien de listé pour l'instant", + "empty": "Les marketplaces configurés ne listent aucun plugin." + }, + "PluginMarketplaceListingRow": { + "official": "Officiel", + "installed": "Installés", + "noDescription": "Aucune description fournie.", + "blocked": "Bloqué par la liste de sécurité d'Orca : {{value0}}", + "blockedAction": "Bloquée", + "checkUpdate": "Vérifier les mises à jour", + "install": "Installer" + }, + "PluginMarketplacePreviewDialog": { + "languagePacksOne": "1 pack de langue", + "languagePacks": "{{value0}} packs de langue", + "commandsOne": "1 commande", + "commands": "{{value0}} commandes", + "keybindingsOne": "1 raccourci clavier", + "keybindings": "{{value0}} raccourcis clavier", + "vmRecipesOne": "1 recette de VM", + "vmRecipes": "{{value0}} recettes de VM", + "panelsOne": "1 panneau", + "panels": "{{value0}} panneaux", + "eventsOne": "1 abonnement aux événements", + "events": "{{value0}} abonnements aux événements", + "worker": "Worker en arrière-plan", + "versionLine": "v{{value0}} · {{value1}}", + "includes": "Inclut", + "noContributions": "Métadonnées du manifeste uniquement", + "capabilities": "Accès demandé", + "workerWarning": "Les capabilities limitent la façon dont ce plugin utilise l'API d'Orca. Son worker s'exécute toujours comme un processus normal sur cet ordinateur, avec un accès complet à vos fichiers, au réseau et aux autres processus.", + "blocked": "La liste de sécurité d'Orca bloque ce plugin : {{value0}}", + "current": "Ce contenu exact de plugin est déjà installé.", + "close": "Fermer", + "cancel": "Annuler", + "update": "Mettre à jour le plugin", + "install": "Installer le plugin" + }, + "PluginMarketplaceSourceDialog": { + "addFailed": "Impossible d'ajouter ce marketplace. Vérifiez l'URL Git, la ref et vos identifiants Git.", + "refreshFailed": "Impossible d'actualiser ce marketplace. Son dernier index valide en cache reste disponible.", + "removeFailed": "Impossible de supprimer ce marketplace.", + "title": "Sources du marketplace", + "description": "Les marketplaces sont des dépôts Git épinglés. Orca utilise vos identifiants Git système existants pour les dépôts privés.", + "urlLabel": "URL Git", + "urlDescription": "Utilisez une URL de dépôt HTTPS ou SSH contenant orca-marketplace.json.", + "urlPlaceholder": "https://git.example.com/team/plugins.git", + "refLabel": "Ref Git", + "refDescription": "Choisissez une branche, un tag ou un commit. Chaque index récupéré est enregistré à un commit précis.", + "adding": "Ajout…", + "add": "Ajouter une source", + "configured": "Sources configurées", + "empty": "Aucune source de marketplace configurée.", + "official": "Officiel", + "owner": "Propriétaire : {{value0}}", + "pinnedCommit": "Épinglé à {{value0}}", + "stale": "Échec de l'actualisation. Navigation dans le dernier index valide en cache.", + "refreshLabel": "Actualiser {{value0}}", + "removeLabel": "Supprimer {{value0}}", + "done": "Terminé" + }, + "PluginRemoveDialog": { + "title": "Supprimer le plugin ?", + "description": "Ceci supprime {{value0}} et ses données de plugin stockées de cet ordinateur. Vous pourrez le réinstaller plus tard.", + "cancel": "Annuler", + "remove": "Supprimer le plugin" + }, + "PluginRollbackDialog": { + "title": "Restaurer le plugin ?", + "description": "Ceci désactive {{value0}} et restaure sa version immuable précédente. Si cette version demande un accès ou un contenu d'instructions différents, Orca exigera une nouvelle revue.", + "cancel": "Annuler", + "confirm": "Restaurer le plugin" + }, + "PluginsSettingsSection": { + "systemLabel": "Système de plugins", + "systemDescription": "Détecte les plugins installés et permet de les activer individuellement. Rien ne s'exécute avant examen et activation. Les workers s'exécutent toujours sur cet ordinateur ; les actions sur les espaces de travail SSH passent par Orca.", + "featureOff": "Activez le système de plugins pour voir et gérer les plugins installés. Tout ce qui est déjà installé reste sur disque et désactivé tant que le système est coupé.", + "loading": "Chargement des plugins…", + "noInstalledResultsTitle": "Aucun plugin correspondant", + "noInstalledResults": "Aucun plugin installé ne correspond à cette recherche.", + "emptyTitle": "Aucun plugin installé pour l'instant", + "empty": "Parcourez l'onglet Tous pour installer des plugins depuis un marketplace.", + "loadFailed": "Impossible de charger les plugins.", + "settingsUpdateFailed": "Impossible d'enregistrer les paramètres du plugin.", + "install": "Installer le plugin", + "title": "Plugins", + "experimental": "Expérimental", + "description": "Installez et gérez les plugins Orca. Les plugins s'exécutent sur cet ordinateur, même pour les espaces de travail SSH.", + "logsFailed": "Impossible de charger les logs du plugin.", + "rollbackFailed": "Impossible de restaurer ce plugin. Une version immuable précédente n'est peut-être pas disponible." + }, + "PluginSettingsRow": { + "blocked": "Bloquée", + "needsReview": "En attente de relecture", + "restarting": "Redémarrage", + "invalid": "Non valide", + "error": "Erreur", + "disabled": "Désactivé", + "running": "En cours", + "enabled": "Activé", + "loadingLogs": "Chargement des logs…", + "noLogs": "Aucune ligne de log enregistrée.", + "logCount": "{{value0}} dernières lignes sur un maximum de 200 lignes conservées", + "reviewAndEnable": "Examiner & activer", + "official": "Officiel", + "dev": "Dev", + "bundled": "Intégré", + "noDescription": "Aucune description fournie.", + "killListMessage": "La liste de sécurité d'Orca a désactivé ce plugin : {{value0}}", + "viewAdvisory": "Consulter l'avis de sécurité", + "runtimeError": "Le plugin s'est arrêté après une erreur d'activation ou de worker.", + "restartCount": " · {{value0}} redémarrages", + "moreActions": "Plus d'actions pour {{value0}}", + "hideLogs": "Masquer les logs", + "viewLogs": "Afficher les logs", + "rollback": "Restaurer", + "remove": "Supprimer", + "disableLabel": "Désactiver {{value0}}", + "enableLabel": "Activer {{value0}}", + "invalidPluginError": "Le manifeste du plugin ou les fichiers installés sont non valides. Corrigez le plugin, puis actualisez." + }, + "PluginVmRecipeConsentPreview": { + "create": "Créer", + "suspend": "Suspendre", + "resume": "Reprendre", + "destroy": "Détruire", + "heading": "Commandes de recette de VM", + "commandLabel": "{{value0}} · {{value1}} commande" + }, + "pluginError": { + "installManifestMissing": "Aucun orca-plugin.json lisible trouvé. Choisissez le dossier racine du plugin.", + "installManifestInvalid": "orca-plugin.json est non valide. Demandez à l'auteur du plugin de corriger le manifeste.", + "incompatible": "Ce plugin nécessite une autre version d'Orca.", + "installUnsafePath": "Le plugin contient un chemin de fichier ou un symlink dangereux et n'a pas été installé.", + "installLimit": "Le plugin dépasse les limites d'installation d'Orca en taille ou en nombre de fichiers.", + "installGit": "Orca n'a pas pu récupérer la révision Git épinglée. Vérifiez l'URL, la #ref, l'accès et la configuration Git du système.", + "invalidManifestMissing": "orca-plugin.json manque à la racine du plugin. Ajoutez-le, puis actualisez les plugins.", + "invalidManifest": "orca-plugin.json est non valide. Corrigez-le, puis actualisez les plugins.", + "invalidArtifact": "Un fichier worker ou panneau déclaré est manquant ou dangereux. Corrigez les fichiers du plugin, puis actualisez.", + "consentChanged": "Le plugin a changé pendant votre examen. Fermez cette boîte de dialogue et examinez les permissions mises à jour." + }, + "plugins": { + "search": { + "title": "Plugins", + "description": "Installez et gérez les plugins Orca expérimentaux.", + "install": "installer le plugin", + "permissions": "permissions du plugin", + "logs": "logs du plugin", + "development": "plugins de développement" + } + }, + "LinearAgentSkillGuide": { + "setupConnectTitle": "1. Connecter Linear", + "setupConnectBody": "Clé API personnelle pour qu'Orca liste les issues et ouvre les espaces de travail liés.", + "manageKeys": "Gérer les clés", + "addAccess": "Ajouter l'accès", + "setupSkillTitle": "2. Installer la skill de l'agent", + "setupSkillBody": "Donne aux agents de codage /orca-linear pour la lecture, les mises à jour, le tri et la jointure de pull requests ou merge requests.", + "setupVisibleTitle": "3. Afficher Linear dans Tasks", + "setupVisibleBody": "Conserve Linear dans le sélecteur de sources Tasks et les raccourcis de la barre latérale.", + "openTaskSources": "Sources de tâches", + "setupTitle": "Checklist de configuration", + "setupBody": "Les trois sont requis pour la boucle complète Tasks + agent. Le parcours de première configuration figure aussi sous Task Sources.", + "setupReady": "Tout est prêt", + "setupProgress": "{{done}} sur {{total}} prêts", + "notesTitle": "Bon à savoir", + "notesIntro": "Quelques rappels une fois Linear connecté et la skill installée.", + "noteLinkedTitle": "Partir d'une issue Linear", + "noteLinkedBody": "Les actions sur les tickets fonctionnent mieux dans un worktree créé depuis Tasks, pour que l'issue reste liée comme contexte.", + "noteSlashTitle": "Mentionner /orca-linear", + "noteSlashBody": "Dans le chat, utilisez /orca-linear (ou demandez en langage naturel) pour que l'agent charge la skill à ce tour.", + "noteKeysTitle": "Les clés suivent le runtime", + "noteKeysBody": "Les clés API et les espaces de travail sont stockés pour le runtime actif.", + "noteVisibilityTitle": "Masquer ≠ déconnecter", + "noteVisibilityBody": "Masquer Linear dans Task Sources le retire seulement du sélecteur. Cela ne supprime ni votre clé ni votre skill.", + "setupChecking": "Vérification…" + }, + "TaskSourceLinearSetup": { + "connectTitle": "Connecter Linear", + "connectDescription": "Ajoutez une clé API personnelle pour qu'Orca puisse parcourir les issues et ouvrir des espaces de travail avec le contexte du ticket.", + "manageAccess": "Gérer les clés", + "addAccess": "Ajouter l'accès Linear", + "connectedHint": "Les espaces de travail et les clés sont stockés pour le runtime actif. Vous pouvez ajouter d'autres accès à tout moment.", + "recheck": "Revérifier la connexion", + "skillTitle": "Installer le skill d'agent Linear", + "skillDescription": "Donne aux agents /orca-linear pour lire les tickets, publier des mises à jour, changer les états et joindre des pull requests ou merge requests.", + "skillBlocked": "Connectez d'abord Linear, puis installez la skill pour les agents.", + "skillPanelTitle": "Skill Linear", + "terminalTitle": "Configuration de la skill Linear", + "terminalAriaLabel": "Terminal d'installation de la skill Linear", + "showDescription": "Incluez Linear dans le sélecteur de sources de la page Tasks et les raccourcis de la barre latérale." + }, + "TaskSourceProviderCard": { + "statusChecking": "Vérification…", + "statusReady": "Prêt", + "statusConnectRequired": "Connexion requise", + "statusSkillRequired": "Skill requise", + "statusUnavailable": "Statut indisponible", + "statusHidden": "Masqué dans Tasks", + "statusIncomplete": "Configuration requise", + "collapseSetup": "Réduire les étapes de configuration de {{provider}}", + "expandSetup": "Afficher les étapes de configuration de {{provider}}" + }, + "TaskSourceShowInTasksStep": { + "shown": "Affichée", + "show": "Afficher", + "hide": "Masquer", + "hideProviderAction": "Masquer {{provider}} dans Tasks", + "showProviderAction": "Afficher {{provider}} dans Tasks", + "lastProviderAction": "{{provider}} est affiché dans Tasks. Au moins un fournisseur doit rester visible.", + "lastProviderHint": "Au moins un fournisseur doit rester visible dans Tasks.", + "title": "Afficher dans Tasks", + "description": "Inclure ce fournisseur dans le sélecteur de sources Tasks et les raccourcis de la barre latérale." + }, + "RuntimeHostAccessForm": { + "getLink": "Obtenir un lien d'accès depuis l'autre hôte", + "stepOpenShare": "Ouvrez Paramètres → Serveurs Orca distants → Partager cet hôte.", + "stepChooseAddress": "Choisissez Autre appareil et sélectionnez une adresse joignable.", + "stepCopyLink": "Générez le lien, puis copiez le lien « Associer un autre client Orca ».", + "name": "Nom dans Orca", + "namePlaceholder": "Station de travail Linux", + "nameHelp": "Cela change uniquement la façon dont l'ordinateur apparaît dans Orca.", + "accessLink": "Lien d'accès", + "accessLinkPlaceholder": "orca://pair?code=...", + "accessLinkHelp": "Orca affiche la destination avant de se connecter. Les identifiants restent masqués.", + "destination": "Destination du lien", + "loopbackTitle": "Ce lien pointe vers cet appareil lui-même", + "loopbackDescription": "Il utilise {{endpoint}}, qui pointe vers l'appareil qui ouvre le lien — et non vers l'autre ordinateur qui l'a créé.", + "loopbackRecovery": "Sur l'autre ordinateur, créez un nouveau lien avec Autre appareil et choisissez son adresse Tailscale ou LAN.", + "identityMismatch": "L'hôte Orca rejoint ne correspond pas à ce lien d'accès", + "identityMismatchHelp": "Orca a rejoint {{endpoint}}, mais cet hôte ne correspond pas à ce lien. Générez un nouveau lien sur l'autre hôte.", + "invalidLink": "Ce lien d'accès n'est plus valide", + "invalidLinkHelp": "Générez un nouveau lien d'accès sur l'autre hôte et réessayez.", + "incompatible": "Les versions d'Orca ne sont pas compatibles", + "incompatibleHelp": "Mettez à jour Orca sur cet appareil et sur l'autre hôte, puis réessayez.", + "interrupted": "Connexion interrompue", + "interruptedHelp": "La connexion s'est arrêtée pendant la vérification. Vérifiez le réseau ou le tunnel SSH et réessayez.", + "unavailable": "Hôte indisponible", + "unavailableHelp": "Vérifiez qu'Orca tourne sur l'autre hôte et que le réseau ou le tunnel SSH peut joindre {{endpoint}}.", + "advanced": "Avancé", + "sshTunnel": "J'utilise un tunnel SSH vers cette adresse locale", + "sshTunnelHelp": "Gardez le tunnel actif tant que cette connexion est utilisée.", + "headlessHelp": "Vous utilisez orca serve sans interface ? Exécutez orca serve --pairing-address sur l'autre ordinateur.", + "cancel": "Annuler", + "addWithTunnel": "Ajouter l'hôte via un tunnel", + "addHost": "Ajouter un hôte", + "connectionDetails": "Détails de la connexion", + "endpointKind": "Type d'endpoint", + "networkConnection": "Connexion réseau", + "notAttempted": "Non tenté", + "saveFailed": "Impossible d'enregistrer l'hôte", + "saveFailedHelp": "L'hôte a été vérifié, mais Orca n'a pas pu l'enregistrer. Vérifiez le nom et le stockage local des paramètres, puis réessayez." + }, + "CloudVmSetupGuide": { + "title": "Créer une VM cloud", + "description": "Les VM cloud sont créées à partir de recettes d'environnement lors de la création d'un espace de travail.", + "setupRecipe": "Configurez une recette d'environnement pour votre fournisseur cloud.", + "createWorkspace": "Créez un espace de travail et sélectionnez cette recette sous Run on.", + "openSetup": "Configurer les recettes d'environnement" + }, + "ReleaseChannelSection": { + "switchFailed": "Impossible de passer à ce build.", + "title": "Canal de release", + "description": "Changez de canal de mise à jour ou passez à n'importe quel build publié, y compris plus ancien. Les retours en arrière sont permis et les builds non vérifiés peuvent être défectueux.", + "devOnly": "Réservé au développement", + "channelAriaLabel": "Canal de mise à jour", + "hourlyWarning": "Les builds horaires sortent directement de main sans validation par tests, et ceux pour Windows ne sont pas signés. Gardez un build stable sous la main.", + "dailyWarning": "Les builds quotidiens sortent directement de main sans validation par tests, et ceux pour Windows ne sont pas signés. Gardez un build stable sous la main.", + "adhocWarning": "Les builds ad hoc viennent d'une branche non intégrée, et ceux pour Windows ne sont pas signés. Leur auteur peut les abandonner — gardez un build stable sous la main.", + "loadingBuilds": "Chargement des builds…", + "noBuilds": "Aucun build trouvé", + "refresh": "Actualiser la liste des builds", + "switchTo": "Basculer vers le build", + "alreadyRunning": "C'est le build que vous utilisez.", + "willSwitch": "{{value0}} → {{value1}}", + "webUnavailable": "Le changement de build n'est disponible que dans l'app de bureau.", + "devChannelUnsupportedAria": "{{value0}} ({{value1}} uniquement)", + "devChannelUnsupported": "Les builds {{value0}} ne sont produits que pour {{value1}}. Linux reste sur Stable ou RC.", + "downloadInstaller": "Télécharger l'installateur", + "manualInstallHint": "Les builds {{value0}} ne sont pas signés sous Windows ; la mise à jour intégrée ne peut donc pas installer l'un d'eux par-dessus un build signé. Exécutez une fois l'installateur téléchargé — Windows avertira d'un éditeur inconnu — puis tous les changements ultérieurs, y compris le retour à Stable, fonctionnent depuis ici." + }, + "VoiceMicrophoneSetting": { + "systemDefault": "Valeur par défaut du système", + "unavailable": "indisponible", + "label": "Microphone", + "description": "Périphérique d'entrée utilisé pour la dictée vocale. Le défaut système suit le réglage du micro de l'OS.", + "accessHint": "Autorisez l'accès au microphone pour lister les périphériques d'entrée.", + "allowAccess": "Autoriser l'accès" + }, + "TerminalTccAttributionNotice": { + "body": "Le daemon du terminal a été démarré par une installation d'Orca qui n'existe plus, donc macOS ne peut pas attribuer ses commandes à Orca — les autorisations Accessibilité et Automatisation sont ignorées silencieusement (osascript échoue avec l'erreur -25211). Redémarrer le daemon corrige le problème ; les sessions de terminal en cours seront fermées.", + "openManageSessions": "Ouvrir Gérer les sessions", + "title": "Les autorisations macOS n'arrivent pas aux terminaux" + }, + "artifacts": { + "enable": "Activer Artifacts", + "enableDescription": "Ajouter Artifacts à la barre latérale pour ouvrir et supprimer les fichiers partagés.", + "account": "Compte Orca", + "connected": "Connecté", + "signInRequired": "La connexion est requise pour téléverser et gérer les artifacts.", + "signingIn": "Connexion…", + "signIn": "Se connecter à Orca", + "title": "Artefacts", + "description": "Partagez des fichiers HTML et Markdown avec votre équipe et gérez leurs liens publics.", + "howToTitle": "Comment utiliser Artifacts", + "howToDescription": "Publiez des fichiers HTML ou Markdown sous forme de liens publics, puis partagez-les avec votre équipe.", + "shareStepTitle": "Choisir un fichier à partager", + "shareStepDescription": "Ouvrez un fichier HTML ou Markdown et sélectionnez Partager comme artifact, ou demandez à un agent de le partager.", + "linkStepTitle": "Copier le lien public", + "linkStepDescription": "Après publication, copiez le lien et envoyez-le à votre équipe.", + "manageStepTitle": "Gérer dans Orca", + "manageStepDescription": "Ouvrez Artifacts depuis la barre latérale pour prévisualiser ou supprimer des liens.", + "openArtifacts": "Ouvrir Artifacts", + "openArtifactsDescription": "Consultez et supprimez les liens partagés via votre compte.", + "showButton": "Afficher le bouton Artifacts", + "showButtonDescription": "Afficher le raccourci Artifacts dans la barre latérale.", + "openArtifactsDescriptionV2": "Prévisualisez, copiez et gérez les liens partagés via votre compte.", + "signInTitle": "Se connecter pour partager des artifacts", + "signInDescription": "Utilisez votre compte Orca pour téléverser des artifacts et gérer leurs liens publics.", + "signInAgain": "Se reconnecter", + "enableStepTitle": "Activer le partage d'artifacts", + "enableStepWebDescription": "Ouvrez Paramètres → Artifacts dans l'app Orca de bureau sur l'appareil hôte et activez la publication.", + "enableStepDescription": "Activez « Autoriser la publication de liens d'artifacts publics » ci-dessus.", + "allowPublishing": "Autoriser la publication de liens d'artifacts publics", + "allowPublishingWebDescription": "Bureau uniquement. Ouvrez Paramètres → Artifacts sur l'appareil hôte pour modifier ce paramètre.", + "allowPublishingDescription": "Publiez des fichiers HTML et Markdown sous forme de liens ouvrables par quiconque possède l'URL. Les liens existants demeurent jusqu'à leur suppression dans Artifacts.", + "howToDescriptionDisabled": "Activez le partage d'artifacts ci-dessus pour publier des fichiers HTML ou Markdown sous forme de liens publics.", + "allowPublishingSearchDescription": "Autoriser Orca à publier des fichiers HTML et Markdown sous forme de liens publics." + }, + "orcaAccount": { + "connected": "Connecté", + "reconnectRequired": "Votre session a expiré. Reconnectez-vous pour utiliser les fonctions cloud.", + "unavailable": "La connexion Orca n'est pas disponible dans ce build.", + "signedOut": "Connectez-vous pour étendre Orca avec des fonctions cloud, dont Artifacts et Orca Relay.", + "checking": "Vérification de l'état du compte…", + "account": "Compte Orca", + "signOut": "Se déconnecter", + "signingIn": "Connexion…", + "signInAgain": "Se reconnecter", + "signIn": "Se connecter à Orca", + "title": "Compte Orca", + "description": "Partagez instantanément votre travail et accédez à votre poste depuis Orca Mobile, où que vous soyez.", + "searchDescription": "Connectez-vous ou déconnectez-vous du compte utilisé par Artifacts et Orca Relay.", + "benefitsTitle": "Inclus avec votre compte", + "artifactsTitle": "Partage d'artifacts", + "artifactsDescription": "Publiez des fichiers HTML et Markdown, puis gérez chaque lien partagé depuis Orca.", + "relayTitle": "Orca Relay", + "relayDescription": "Connectez Orca Mobile à ce poste via réseau mobile ou n'importe quel Wi-Fi.", + "skillsTitle": "Partage de skills", + "skillsDescription": "Partagez une skill ou tout un ensemble derrière un lien non répertorié, et installez-les sur toutes les machines que vous utilisez." + }, + "automations": { + "showButton": "Afficher le bouton Automatisations", + "showButtonDescription": "Afficher le raccourci Automations dans la barre latérale.", + "howItWorksTitle": "Fonctionnement des Automations", + "howItWorksDescription": "Planifiez le travail d'un agent une fois, puis laissez Orca créer chaque exécution et regrouper ses résultats.", + "defineStepTitle": "Décrire le travail", + "defineStepDescription": "Choisissez un projet, un agent, un prompt et une planification.", + "runStepTitle": "Orca lance chaque exécution", + "runStepDescription": "L'agent sélectionné reçoit un espace de travail neuf quand la planification arrive à échéance.", + "reviewStepTitle": "Consulter les résultats", + "reviewStepDescription": "Inspectez les exécutions récentes et poursuivez le travail dès que nécessaire.", + "openAutomations": "Ouvrir Automations", + "openAutomationsDescription": "Créez des planifications et inspectez les exécutions récentes.", + "title": "Automatisations", + "description": "Planifiez le travail des agents et choisissez si Automatisations apparaît dans la barre latérale." + }, + "AgentAwakeSetting": { + "on": "Activé", + "auto": "Agent", + "off": "Désactivé" + }, + "shareSkills": { + "title": "Partage de skills", + "description": "Partagez vos skills avec un lien non répertorié. Toute personne qui l'a peut les installer.", + "allowAgentPublishing": "Autoriser les agents et la CLI Orca à publier des liens de skills", + "allowAgentPublishingDescription": "Permet aux commandes de publier des skills installées nommées explicitement. Les dossiers de skills peuvent contenir des scripts, de la configuration ou des secrets, c'est donc désactivé par défaut.", + "allowAgentPublishingWebDescription": "Bureau uniquement. Ouvrez Paramètres → Partage de skills sur l'appareil hôte pour modifier ce paramètre.", + "selectTitle": "Sélectionnez une ou plusieurs skills", + "selectDescription": "Ouvrez Skills, choisissez Partager des skills, puis sélectionnez les skills à regrouper derrière un seul lien.", + "reviewTitle": "Vérifier et publier", + "reviewDescription": "Vérifiez les fichiers, scripts et exécutables inclus avant de téléverser le bundle immuable.", + "copyTitle": "Copier le lien non répertorié", + "copyDescription": "Quiconque possède le lien peut inspecter et installer toutes les skills ou celles sélectionnées, sans se connecter.", + "manageTitle": "Gérer ou révoquer les liens", + "manageDescription": "La révocation bloque tout accès futur. Les skills déjà installées sur une autre machine y restent.", + "linkTitle": "Liens de skills non répertoriés", + "linkDescription": "Les bundles partagés ne sont ni recherchables ni listés dans Orca. Le lien tient lieu d'identifiant : envoyez-le uniquement à des personnes de confiance.", + "signInTitle": "Se connecter pour partager des skills", + "signInWebDescription": "La publication et la gestion des liens sont disponibles dans l'app Orca de bureau.", + "signInDescription": "Utilisez votre compte Orca pour publier des bundles et gérer leurs liens. Les destinataires n'ont pas besoin de compte.", + "signingIn": "Connexion…", + "signInAgain": "Se reconnecter", + "signIn": "Se connecter à Orca", + "howToTitle": "Comment partager des skills", + "howToDescription": "Publiez une seule skill ou un bundle, par exemple une collection de 30 skills, derrière un seul lien.", + "openSkills": "Ouvrir Skills", + "openSkillsDescription": "Publiez un bundle, installez depuis un lien ou gérez les skills installées et partagées.", + "searchDescription": "Partagez vos skills avec un lien non répertorié. Toute personne qui l'a peut les installer.", + "linksReconnect": "Reconnectez-vous pour gérer les liens partagés.", + "linksUnavailable": "Les liens partagés sont indisponibles pour le moment.", + "linkCopied": "Lien de partage copié", + "linkRevoked": "Lien révoqué", + "revokeFailed": "Orca n'a pas pu révoquer ce lien.", + "activeLinks": "Liens partagés actifs", + "activeLinksDescription": "Seules les personnes disposant du lien peuvent l'ouvrir. Annulez le partage d'un lien pour bloquer toute inspection et installation futures.", + "refreshLinks": "Actualiser", + "noActiveLinks": "Aucun lien actif. Publiez un bundle de skills depuis Skills pour en créer un.", + "copyLink": "Copier le lien", + "confirmUnshare": "Confirmer l'arrêt du partage", + "unshare": "Ne plus partager", + "showButton": "Afficher le bouton Skills", + "showButtonDescription": "Afficher le raccourci Skills dans la barre latérale.", + "manageInSkills": "Gérer dans Skills" + }, + "bitbucket": { + "credentials": { + "dialog": { + "connectFailed": "Échec de la connexion", + "title": "Connecter Bitbucket", + "description": "Utilisez un identifiant Bitbucket Cloud pour parcourir les pull requests et les statuts de build. Orca le vérifie avant de l'enregistrer.", + "remoteRuntime": "Les identifiants Bitbucket enregistrés ici ne sont stockés que sur cette machine locale. Définissez plutôt les variables d'environnement ORCA_BITBUCKET_* sur le runtime distant.", + "environmentManaged": "Bitbucket est déjà configuré via les variables d'environnement ORCA_BITBUCKET_*, qui ont priorité. Désactivez-les pour enregistrer un identifiant dans Orca.", + "authModeLabel": "Méthode d'authentification Bitbucket", + "modeBasic": "E-mail & token API", + "modeToken": "Token d'accès", + "accessToken": "Token d'accès", + "accessTokenPlaceholder": "Token d'accès de dépôt, projet ou espace de travail", + "email": "E-mail du compte Atlassian", + "emailPlaceholder": "you@example.com", + "apiToken": "Token API", + "apiTokenPlaceholder": "Token API Atlassian", + "baseUrl": "URL de base de l'API (facultatif)", + "baseUrlPlaceholder": "https://api.bitbucket.org/2.0", + "tokenHint": "Les tokens d'accès de dépôt, projet et espace de travail se créent depuis la page de paramètres Bitbucket correspondante et nécessitent un accès en lecture aux pull requests.", + "basicHint": "Créez un token API Atlassian pour votre compte, puis associez-le à l'adresse e-mail qui en est propriétaire.", + "docsLink": "Documentation des tokens API Bitbucket", + "storageNote": "Stocké sur cette machine avec chiffrement quand le trousseau de l'OS est disponible. Les variables d'environnement ORCA_BITBUCKET_* ont toujours priorité sur ce que vous enregistrez ici.", + "cancel": "Annuler", + "verifying": "Vérification...", + "connect": "Se connecter" + } + }, + "integration": { + "card": { + "description": "Pull requests et statuts de build pour Bitbucket Cloud.", + "edit": "Modifier les identifiants", + "connect": "Se connecter", + "accountUnknown": "Bitbucket Cloud", + "authModeToken": "Token d'accès", + "authModeBasic": "E-mail & token API", + "disconnect": "Déconnecter Bitbucket", + "envManaged": "Configuré via des variables d'environnement. Désactivez les variables ORCA_BITBUCKET_* pour gérer cet identifiant dans Orca.", + "storedAuthFailed": "L'identifiant Bitbucket enregistré n'a pas réussi à s'authentifier. Modifiez-le ou vérifiez que le token garde l'accès aux pull requests.", + "storedCredential": "Enregistré dans Orca sur cette machine. Les variables d'environnement ORCA_BITBUCKET_* ont priorité quand elles sont définies.", + "notConfigured": "Connectez un compte Bitbucket Cloud avec un token API Atlassian ou un token d'accès. Les variables d'environnement ORCA_BITBUCKET_* marchent aussi et ont priorité.", + "disconnectFailed": "Impossible de supprimer l'identifiant Bitbucket enregistré." + } + } + }, + "GlobalWorktreeVisibilitySourcesSetting": { + "saveFailed": "Impossible d'enregistrer les valeurs de visibilité par défaut.", + "updateServer": "Mettez à jour ce serveur pour configurer les sources par défaut.", + "updateServerDefaults": "Mettez à jour ce serveur pour configurer les visibilités par défaut." + }, + "EphemeralVmCleanupStopDialog": { + "title": "Arrêter le nettoyage ?", + "description": "La VM peut continuer à tourner et générer des frais. Vous pourrez relancer le nettoyage plus tard.", + "cancel": "Continuer le nettoyage", + "stopping": "Arrêt…", + "confirm": "Arrêter le nettoyage" + }, + "shortcutDefinitionCatalog": { + "missionControlConflict": "Bloqué par Mission Control. Remappez ici ou changez-le dans Réglages Système." + } + }, + "right": { + "sidebar": { + "BulkActionBar": { + "79a9f5f712": "Retirer du staging (", + "ef5f5bd06e": "Ajouter au staging (", + "60ed678138": "sélectionnés" + }, + "ChecksPanel": { + "2ef90c9819": "Un agent IA a été démarré pour les vérifications cassées.", + "a0181a8d76": "Un agent IA a été lancé pour les conflits.", + "34464d00b9": "mis à jour", + "058039787c": "Annuler", + "2ab7fd4b6d": "Enregistrer", + "dda5924a40": "Les checks exigent une branche Git et un contexte de revue hébergée", + "976cefd02f": "Checks indisponibles", + "b5dd73a105": "Sélectionnez un espace de travail pour voir les checks", + "a4ef4e0832": "Aucun espace de travail sélectionné", + "5594400d73": "Aucune vérification cassée à corriger.", + "abf59262fb": "Vérifiez le prompt avant de démarrer un agent.", + "4ede779461": "Résoudre les conflits de revue avec l'IA", + "3b203c62f8": "Le commentaire sera définitivement retiré de la PR.", + "ea9b649ce3": "Supprimer le commentaire ?", + "5788d1059d": "Impossible de mettre à jour le fil de revue. Vérifiez le budget de l'API GitHub.", + "07871c0589": "Lier une autre PR", + "7202f4a40a": "délier la PR", + "7f4489f370": "Actualiser", + "5c88c6db07": "Ouvrir sur {{value0}}", + "7fad8509fe": "Corriger avec l'IA", + "71026ca2cb": "Actualisation…", + "889cdfba04": "Créer {{value0}}", + "98f4c37b33": "Pusher & créer {{value0}}", + "b6ce28da5b": "{{value0}} #{{value1}} est déjà ouvert", + "cf9e69f3be": "{{value0}} est déjà ouvert", + "192e686e57": "Ouvrir sur {{value0}}", + "6633c7a1fb": "Publier la branche", + "fdb27637f2": "Publication…", + "e56c42122e": "destructive", + "786e3c143f": "Supprimer", + "653c105ecc": "Plus d'actions sur la PR", + "f316a8ca2b": "Aucun commentaire non résolu sélectionné.", + "d00ebdc402": "Résoudre {{value0}} commentaires avec l'IA", + "5eb2163b6b": "Vérifiez le prompt avant de lancer un agent. Une fois le prompt livré, Orca résout les fils sélectionnés côté hôte et répond aux commentaires qu'il ne peut pas résoudre.", + "f273f2271c": "Agent lancé. {{value0}} marqués résolus, {{value1}} réponses envoyées, {{value2}} ignorés, {{value3}} échecs.{{value4}}", + "aa95b81a3a": "Agent lancé. {{value0}} marqués résolus, {{value1}} réponses envoyées, {{value2}} ignorés, {{value3}} échecs.", + "495b2f8c4b": "Agent lancé, mais impossible de résoudre ou de répondre aux commentaires sélectionnés.", + "3c3ad3a1d2": "Agent lancé. Aucun des commentaires sélectionnés ne peut être marqué résolu côté hôte.", + "review": { + "auto_retry": "Orca réessaiera à {{time}}.", + "open_review": "Ouvrir la revue", + "retry": "Réessayer" + }, + "sync": { + "pending": "Synchronisation…", + "branch": "Synchroniser la branche" + }, + "7e4b2a19c0": "Impossible d'identifier la PR GitHub sur laquelle répondre.", + "430f1a62d4": "Impossible de résoudre le fil sélectionné côté hôte.", + "updateReactionFailed": "Échec de la mise à jour de la réaction." + }, + "CreatePullRequestDialog": { + "2bc1b4345e": "Annuler", + "27ef4b195c": "Choisissez une autre branche de base avant de créer {{value0}}.", + "7ef56f3efe": "Créer en brouillon", + "0c9f9a568c": "Prend en charge le formatage Markdown. Utilisez Générer avec l'IA pour le remplir automatiquement à partir de vos modifications.", + "02b2ce911f": "Description (facultatif)", + "1cd53359db": "Description", + "68314b4369": "Titre", + "694550a610": "main", + "0fad57a14c": "Recherchez des branches distantes ou saisissez un nom de branche.", + "8584ccb43c": "Branche de base", + "6f5f1962b6": "Branche source", + "b504b3ceb1": "détails avant de créer la revue hébergée.", + "f658ff2455": "Confirmez la branche cible et les détails de {{value0}} avant de créer la revue hébergée.", + "b7f43474d7": "Créer {{value0}}", + "7a21f0dae8": "Ouvrir sur {{value0}}", + "edc35a7027": "{{value0}} #{{value1}} est déjà ouvert", + "a154fe55e6": "Pusher & créer {{value0}}", + "21c7a1daa0": "{{value0}} est déjà ouvert", + "db9cee18f7": "Créer {{value0}}" + }, + "CreateHostedReviewComposer": { + "741ff8a0d2": "Pusher & créer {{value0}}" + }, + "CreatePullRequestGenerateButton": { + "4012459f8a": "Générer avec l'IA", + "a0501572c1": "Générer les détails de {{value0}} avec l'IA", + "d47fd63012": "Génération des détails de {{value0}}. Cliquez pour arrêter.", + "bdf83ccb15": "Génération de {{value0}}", + "f5513bdeb1": "Génération", + "a6ea6dc3aa": "Génération…", + "e61d7e7ad4": "Arrêter la génération des détails de {{value0}}", + "e041998cad": "Arrêter la génération" + }, + "FileExplorer": { + "79b1537dd3": "Sélectionnez un espace de travail pour parcourir les fichiers", + "4da4d89845": "Retour à l'explorateur", + "6ed5ce817b": "Recherche", + "2f4483d6c4": "Aucun fichier ne correspond à ce filtre" + }, + "FileExplorerBackgroundMenu": { + "3b5e2dcb8d": "Nouveau dossier", + "21fe46ed36": "Nouveau fichier" + }, + "FileExplorerRow": { + "addc01145f": "Supprimer", + "fc747429bf": "Renommer", + "0df0e5abac": "Rechercher dans le dossier", + "d6a25618aa": "Réduire le dossier", + "d87a4c42e1": "Ouvrir l'aperçu Markdown", + "c2112579f6": "Télécharger", + "7ac885bd2f": "Télécharger le dossier", + "dd112c81d2": "Ouvrir dans le navigateur Orca", + "1bb9be455c": "Ajouter comme projet...", + "0fec99bfd7": "Doublon", + "f61af83316": "Nouveau dossier", + "37c875d827": "Nouveau fichier", + "b3e288bf41": "Échec du téléchargement de « {{value0}} ».", + "f729bcd97d": "Échec du téléchargement du dossier « {{value0}} ».", + "1a3df04ae1": "Ouvert", + "bce4d4e44f": "« {{value0}} » téléchargé", + "a4029c996b": "Dossier « {{value0}} » téléchargé", + "e26010014a": "Ignoré par .gitignore", + "a06551beee": "Enter", + "128a99ed5e": "Non assigné", + "2de3b21934": "markdown", + "66a29dde82": "Copier le chemin relatif", + "42e10cbf57": "Copier les chemins relatifs", + "b5d436aa30": "Copier le chemin", + "98a79948b3": "Copier", + "b234ab25b4": "Impossible de copier le fichier dans le presse-papiers", + "f9d7ca753d": "Copier les chemins", + "3161c4e425": "dossier", + "e887fa4b2e": "Ouvrir dans le terminal", + "1d8e182c32": "Afficher le fichier", + "clipboardStagingUnavailable": "Impossible de copier le fichier car le stockage temporaire d'Orca est indisponible" + }, + "FileExplorerToolbar": { + "d238264654": "Afficher les fichiers ignorés par Git", + "78f133232c": "Afficher les dotfiles", + "31b4c3195d": "Plus d'actions de l'explorateur", + "d95e30fe28": "Actualiser l'explorateur", + "6026b16950": "Tout réduire", + "693cbeadd0": "Recherche", + "c1f3f3ec70": "Rechercher dans le contenu des fichiers" + }, + "FileExplorerTreeStatus": { + "ce03835e1f": "Aucun fichier dans cet espace de travail", + "c76693e456": "Impossible de charger les fichiers de cet espace de travail :" + }, + "GitHistoryPanel": { + "cf7cad58d2": "Aucun commit pour le moment", + "781a8bcf7b": "Chargement du graphe...", + "d0fb0f4bf2": "Actualiser les commits", + "9f7535d22b": "Les refs sont des noms de branche ou de tag pointant vers ce commit exact. Elles n'apparaissent que là où Git possède une ref nommée pour ce commit.", + "9289ba0cb9": "Que sont les refs ?", + "d836037d02": "Commits", + "8232c8b2f2": "Ouvrir le commit {{value0}} : {{value1}}", + "9a8b85882d": "chargement", + "62e685d5ec": "idle", + "111e1d0db4": "error", + "e5e81e59a6": "Redimensionner les commits", + "6d1e0a7c3b": "Échec du chargement des fichiers du commit" + }, + "HostedReviewActions": { + "4d5fb5a284": "Fermer", + "9845a71e17": "Plus d'actions", + "2bfaf4379c": "Plus d'actions {{value0}}", + "377269db6f": "Réouverture de {{value0}} effectuée", + "fa3ee9a515": "closed", + "closedToast": "Fermeture de {{value0}} effectuée", + "78f5ff294c": "Cela rouvrira {{value0}}.", + "a3d572a4de": "Cela fermera {{value0}}.", + "e4aca40024": "Supprimer l'espace de travail", + "eefd50457e": "Suppression...", + "3ce211ece6": "Rouvrir {{value0}}", + "6645ac7dd1": "Réouverture...", + "b25f63edd7": "ouvrir", + "d2ca293f3d": "Traitement...", + "ef064cb7c3": "default", + "59b4dccf70": "destructive", + "9a41a687b7": "Mettre en file via #{{pr}} · {{count}} PR", + "38a1bccb14": "Mettre en file via #{{pr}} · {{count}} PRs", + "3de88351c5": "GitHub ajoutera cette pull request et toutes celles situées en dessous à la file de merge.", + "a32fe6dba6": "GitHub mergera cette pull request et toutes celles situées en dessous dans la pile.", + "73e0e1819d": "Mise en file de la pile...", + "e555a41d32": "Merge de la pile..." + }, + "PortsPanel": { + "3ea4a02a8f": "Annuler", + "4eb801ce93": "dev-server", + "8dfed0a15c": "Libellé (facultatif)", + "17bea6e391": "localhost", + "a3721a50b0": "Hôte distant", + "d57545ff92": "Identique au distant", + "b950b1948b": "Port local", + "9e5a4118b0": "Port distant", + "c9d106547a": "Transférer", + "c7e920aa7c": "annoncé comme {{value0}}", + "e740075063": "Supprimer", + "b3548e59f4": "Édition", + "fe2730d050": "Copier {{value0}}", + "b22b128b2a": "Ouvrir dans le navigateur", + "75aeea592f": "Ouvrir {{value0}} dans le navigateur", + "de349d4560": "ouvre {{value0}}", + "907eb53ed2": "Transférer un port", + "04efd3dad4": "Transférez un port pour accéder aux services distants depuis votre machine locale.", + "1f0d2a24f9": "Aucun port transféré", + "36b1b2984a": "Détecté", + "ddbe58d74e": "Transféré", + "a103dae837": "Ajouter", + "6bc058dbe1": "Ports", + "d4c3cd679c": "Reconnexion...", + "a2f1a47f42": "Connexion SSH perdue", + "409afcc145": "Aucun espace de travail sélectionné pour le navigateur.", + "153145e675": "Preuve", + "c7b4702b7b": "Espace de travail", + "57d930fa45": "PID", + "5dd86dcf2f": "Processus", + "b1ff94fa27": "Protocole", + "729be0b4e5": "Type", + "0f1d8cd324": "Bind", + "1c1c18cefc": "Adresse", + "f9528da632": "Arrêter le processus", + "a223459512": "Afficher les détails", + "bdac206faf": "Copier les détails", + "792baeb7ed": "Copier l'adresse", + "d41a8241ec": "Port", + "a2a9fc6899": "Aucun port local détecté", + "f59c783b7a": "Scan des ports indisponible sur {{value0}} : {{value1}}", + "7822e3edc6": "Actualiser les ports", + "c1b115c375": "Aucun espace de travail sélectionné", + "98e9a414f8": "Échec de l'ouverture du navigateur", + "a00f3a2840": "Échec de l'actualisation des ports", + "97b562d21d": "Processus arrêté sur :{{value0}}", + "9079776663": "Enregistrer", + "c57eda6822": "modifier", + "9f475dc994": "Transfert en cours...", + "d7c83cfd24": "Enregistrement...", + "31e80cff2d": "Transférez un port distant vers votre machine locale.", + "10360598a4": "Mettez à jour la configuration de transfert de port.", + "80206251c8": "Modifier le transfert de port", + "4bc9b00912": "espace de travail", + "3e13cb63ee": "Inconnu", + "472054d94c": "Port :{{value0}}", + "1119f90ad7": "conteneur", + "d32820d3e2": "Externe", + "4db4b5e435": "Autres espaces de travail", + "38b16cfbef": "Aucun port détecté", + "0d63d94db3": "Analyse...", + "935dda7718": "Espace de travail actif", + "740aca88ab": "Échec du scan des ports de l'espace de travail.", + "5be4f7f727": "Menu du port {{value0}}", + "7550998473": "Copier", + "1004af16ab": "Copier {{value0}}" + }, + "Search": { + "1abfb25a66": "Saisissez pour rechercher dans les fichiers", + "d56d140747": "Appuyez sur Entrée pour rechercher", + "0b8104eaf2": "fichier", + "4107975b3a": "dans", + "6aeda362ed": "résultat", + "98c8435e36": "Sélectionnez un espace de travail pour rechercher", + "1ec640c9c7": "correspondance", + "dcc294f28d": "(résultats tronqués)" + }, + "SearchFilters": { + "01e4671ccf": "fichiers à exclure (ex. *.min.js, dist/**)", + "0a6412a895": "Fichiers à exclure", + "8a77efcbd1": "fichiers à inclure (ex. *.ts, src/**)", + "a69ee1bd0e": "Fichiers à inclure" + }, + "SearchHeader": { + "6234a5ef85": "Utiliser une expression régulière", + "4567e6e0b6": "Mot entier uniquement", + "464ae3974f": "Respecter la casse", + "693cbeadd0": "Recherche" + }, + "SearchQueryRow": { + "queryLabel": "Rechercher dans les fichiers", + "clearLabel": "Effacer la recherche" + }, + "SearchResultItems": { + "cc06595a3b": "Copier le chemin de la ligne", + "3596b9668d": "Copier le chemin" + }, + "SourceControlEntryContextMenu": { + "a1f2c8d901": "Affichage" + }, + "SourceControl": { + "1406954883": "Effacer toutes les notes...", + "conflictsSection": "Conflits", + "cc05b2d088": "Ouvrir dans l'explorateur de fichiers", + "03194cfff4": "État de session local issu d'un conflit que vous avez ouvert ici.", + "413a3ba113": "conflit", + "27a50fe970": "Examiner les conflits", + "f6cb48b6fe": "Résoudre avec l'IA", + "3eeccbb221": "Les fichiers résolus repassent dans les modifications normales une fois sortis de l'état de conflit actif.", + "c321542ee2": "Supprimer la note de la ligne {{value0}}", + "b656381c18": "Supprimer la note", + "c085946bda": "Copier la note de la ligne {{value0}}", + "1623bf4e19": "Copier la note", + "655633c08a": "Envoyé", + "3eb9b2805e": "Ouvrir la note sur {{value0}}", + "0d963bf982": "Ouvrir {{value0}}", + "59654650d3": "Effacer les notes de {{value0}}", + "ac8cbe3bf5": "Survolez une ligne dans la vue diff et cliquez sur le + pour ajouter une note.", + "286dbda4d6": "Réessayer", + "476b77745b": "Modifier la ref de base", + "ed34038d0d": "Actualiser la comparaison de branches", + "493f963029": "Modifier la ref de base", + "3278b2767b": "en avance", + "11b5dd8e41": "Comparaison avec", + "783a808870": "Fermer", + "a9bf7c171a": "Échec du commit", + "03d238218c": "Détails", + "011f9713fc": "Commit bloqué", + "cc199ccc5f": "Autres actions de commit et de remote", + "4d6e1fd7f3": "Plus d'actions", + "37a81f29ad": "Génération du message de commit. Cliquez pour arrêter.", + "b94112eb9e": "Message de commit", + "0d0a8359d3": "Message", + "15b7f210d7": "Vérifiez le prompt avant de démarrer un agent.", + "054ead86b1": "Corriger l'échec du commit avec l'IA", + "9e5ccd00aa": "Contexte de l'échec du commit indisponible", + "f0a2dc9e46": "Personnaliser le lancement...", + "ec7bfced55": "Choisir l'agent pour corriger l'échec du commit", + "dd43c47089": "Choisissez un agent pour cet échec de commit", + "30b8d4f181": "Corriger l'échec du commit avec l'IA", + "4b37ae99b0": "Démarrer l'agent IA par défaut pour corriger cet échec de commit", + "ae743199cd": "Choisissez une autre branche de base avant de créer {{value0}}.", + "318e2a7f88": "Attendez la fin de la génération par l'IA.", + "f76307c1f7": "Choisissez une branche de base.", + "4f76c0a9de": "La branche de base doit être différente de la branche courante.", + "c5e4175139": "Autres actions {{value0}} et de remote", + "78ddfd0bb4": "Créer en brouillon", + "e64a632456": "main", + "6055949c50": "Branche de base {{value0}}", + "1f7119f604": "Base", + "9484270f45": "Génération du titre et de la description…", + "a0dc20fc93": "Description (facultatif)", + "a8873e1d62": "Description de {{value0}}", + "7d6a8f0082": "Titre", + "a6eda33521": "Titre de {{value0}}", + "02d8c04339": "Générer les détails de {{value0}} avec l'IA", + "aee92f8684": "Générer", + "e868cec4e1": "Génération…", + "b355e740b2": "Arrêter la génération des détails de {{value0}}", + "527e130b6f": "Arrêter la génération", + "e1970d327d": "Nouvelle {{value0}}", + "f4c766f1ca": "Choisissez l'agent et le modèle de commande pour cette exécution.", + "1a6a6e0bc5": "Générer les détails de la revue hébergée", + "6b122529d4": "Générer le message de commit", + "e48caaf0dd": "Un agent IA a été lancé pour les conflits.", + "901140f47d": "Vérifiez le prompt avant de démarrer un agent.", + "19652ddd76": "Résoudre les conflits avec l'IA", + "c9ad22888e": "Choisissez la cible de comparaison de branches pour ce dépôt.", + "574d2f4413": "Effacer les notes", + "05bb8f4a48": "Annuler", + "48db37cca9": "Tout afficher", + "78ce2d37ac": "Pousse vers le fork", + "c05fe04839": "Pousse vers le fork situé à {{value0}} (pas origin)", + "c35baf2f1e": "Filtrer les fichiers…", + "2fe2a67580": "Autres actions de note", + "eae2d051af": "Copier toutes les notes", + "cc474e0b8c": "Notes", + "e131cd7128": "Le contrôle de code source n'est disponible que pour les dépôts Git", + "c07b236287": "Sélectionnez un espace de travail pour afficher les modifications", + "dc5a6465fc": "{{value0}} (ex. {{value1}}{{value2}})", + "8eb3782a0c": "Échec de l'abandon de {{value0}} fichier{{value1}}", + "a5e5a11090": "Échec de l'abandon global — impossible de sortir les fichiers de l'index avant l'abandon", + "8a5ba6a988": "Échec du chargement du diff du commit", + "fe5bd1a610": "Création de {{value0}}...", + "812cb992ee": "Ouvrir sur {{value0}}", + "eef5446523": "{{value0}} #{{value1}} est déjà ouvert", + "0453ca3a9a": "Création de {{value0}} effectuée, mais Orca n'a pas encore pu l'actualiser.", + "f99560ab29": "Échec de l'abandon de {{value0}}", + "eae7a1da5f": "Échec de l'effacement des notes.", + "657e0c90ad": "{{value0}} note{{value1}}", + "df5040e3c3": "Déstager", + "8cde1a2fb0": "Stager", + "d54dd48b0b": "Abandonner les modifications", + "989f3d5e34": "Restaurer le fichier", + "2830dd64a2": "supprimé", + "11463f7a98": "Supprimer le fichier non suivi", + "d62bc0c7d8": "non suivi", + "ab31221779": "Déstager le dossier", + "bfe9011a0e": "Stager le dossier", + "6d7f2a47e5": "Abandonner le dossier", + "9b367363b6": "Supprimer les non suivis du dossier", + "540ca8f78c": "Abandonner le merge", + "425f138269": "Abandonner le rebase", + "04832d8047": "rebase", + "c105a61960": "merge", + "d7a5942e41": "{{value0}} : {{value1}} non résolus", + "c56ba7fa06": "Diff", + "94c42b252e": "MD", + "e59bca888a": "markdown", + "b6922abb13": "Impossible de charger la comparaison de branches.", + "715d229c86": "Comparaison de branches indisponible", + "97d8b03cdf": "Échec de la comparaison de branches", + "424ee0e5bf": "error", + "834cb3f23d": "Corriger avec l'IA", + "60bd988f0b": "Correction IA", + "461575b9bc": "Générer le message de commit avec l'IA", + "b16b8f0e4b": "message commit IA", + "ddc1fbd690": "Arrêter la génération du message de commit", + "5acbcedc1a": "Créer {{value0}}", + "aaf1451654": "Créer un brouillon de {{value0}}", + "26511c22b4": "Création...", + "7a09d7f9d2": "base", + "383cf92c73": "arborescence", + "d7ae61269b": "Committé sur la branche", + "48a003c1b1": "Modifications stagées", + "d4ef4bafc5": "Modifications", + "522f44dce5": "Fichiers non suivis", + "3636d0f686": "prêt", + "d2e9189866": "tout", + "a0cc0e6b4e": "chargement", + "9339382454": "Déstager tout", + "24d2598eff": "Stager tout", + "ce41708855": "Tout abandonner", + "2f609a2e7c": "Supprimer tous les non suivis", + "9bb062a886": "non committé", + "9febd8ab5f": "create_pr", + "f62ce91ade": "origin", + "3a231c845b": "unknown", + "3baf6c77b4": "Copier toutes les notes dans le presse-papiers", + "72f2bea3f4": "Déplier les notes", + "d13edef890": "Replier les notes", + "0fad573938": "Non committé", + "77afaa8152": "Tous", + "d6fb1df5fe": "{{value0}} est déjà ouvert", + "05838cfdeb": "Conflit {{value0}}", + "d206117f90": "Conflit {{value0}} ({{value1}})", + "0b5b8c234c": "Ouvrir {{value0}} ({{value1}})", + "d97ef8f221": "lignes {{value0}}-{{value1}}", + "6f8bfa0eb9": "ligne {{value0}}", + "c569d29a02": "modifié des deux côtés", + "ea7287d84f": "ajouté des deux côtés", + "bd0151ef7b": "supprimé chez nous", + "44594e8c61": "supprimé chez eux", + "24773ee581": "ajouté chez nous", + "c03d7c952f": "ajouté chez eux", + "5b176fa431": "supprimé des deux côtés", + "31f6d46278": "Non résolu", + "2c417432b7": "Résolu localement", + "f3a8b2c1d0e5": "Saisissez un titre de {{value0}}.", + "e2b7a1c0d9f4": "Échec de la création de {{value0}}", + "hugeRepoIgnorePrompt": "Ce dépôt contient trop de modifications actives. Ajouter « {{value0}} » à .gitignore ?", + "hugeRepoIgnoreAction": "Ajouter à .gitignore", + "tooManyChanges": "Trop de modifications détectées. Seules les {{value0}} premières modifications sont affichées.", + "submoduleTruncated": "D'autres modifications de sous-modules ont été omises", + "bf5082de46": "{{value0}} copié", + "c06193ef57": "Échec de la copie : {{value0}}", + "d172a4f068": "Hash du commit", + "e283b50179": "Message de commit", + "f394c6128a": "Aucun agent disponible pour expliquer ce commit", + "04a5d7239b": "Ce dépôt n'a aucun remote web pris en charge", + "15b6e834ac": "Échec de l'ouverture du commit dans le navigateur", + "d37e68f61d": "Préparation de la branche pour la revue…", + "8d8f5c6c94": "Génération du message de commit…", + "fda060d6ce": "Relisez le message de commit, puis réessayez Créer une PR.", + "b75cb1fd0c": "Commit des modifications…", + "995c5e67ec": "La configuration de revue nécessite votre attention.", + "d7492cafce": "Impossible d'actualiser le contrôle de code source. Réessayez Créer une PR.", + "473f18758e": "Paramètres IA du contrôle de code source", + "createPrIntentConfigureAi": "Ajoutez un message de commit ou configurez les paramètres IA du contrôle de code source.", + "createPrIntentGenerateFailed": "Impossible de générer un message de commit. Ajoutez-en un et réessayez.", + "createPrIntentCommitFailed": "Impossible de committer les modifications. Corrigez le problème, puis réessayez Créer une PR.", + "createPrIntentNeedsSync": "Synchronisez cette branche avant de créer une revue.", + "createPrIntentBranchNotReady": "La branche n'est pas encore prête pour la création d'une revue.", + "createPrIntentPublishing": "Publication de la branche…", + "createPrIntentForcePushing": "Force push avec lease…", + "createPrIntentPushing": "Push des commits…", + "createPrIntentFastForwarding": "Mise à jour de la branche…", + "createPrIntentRemoteFailed": "Impossible de mettre à jour la branche distante. Réessayez Créer une PR.", + "createPrIntentGeneratingDetails": "Génération des détails de revue…", + "createPrIntentBranchChangedDuringDetails": "La branche a changé pendant la génération des détails de revue. Réessayez Créer une PR.", + "createPrIntentCreatingReview": "Création de la revue…", + "a91f8e2b01": "Afficher en liste", + "b82e9f3c12": "Afficher en arbre", + "f71c4a8d90": "Plus d'actions du contrôle de code source", + "c8e4a1f902": "Filtre : {{value0}}", + "b3c8f1a902": "Filtrer les fichiers par nom", + "d4f8c2a901": "Effacer et fermer le filtre", + "e8a1c4b203": "vs", + "f9b2441bb6": "1 commit d'avance sur {{value0}}", + "b715ef615b": "{{value0}} commits d'avance sur {{value1}}", + "c1a8f3e204": "1 commit de retard sur {{value0}}", + "d2b9g4f315": "{{value0}} commits de retard sur {{value1}}", + "4b4a7de138": "Ouvrir la page de revue dans le navigateur", + "createPrIntentCommitBlockedSummary": "Commit bloqué : {{value0}} Corrigez le problème, puis réessayez Créer une PR.", + "pushRecovery": { + "4b37ae99b0": "Démarrer l'agent IA par défaut pour corriger cet échec de push", + "30b8d4f181": "Corriger l'échec de push avec l'IA", + "dd43c47089": "Choisissez un agent pour cet échec de push", + "ec7bfced55": "Choisir l'agent pour corriger l'échec de push", + "9e5ccd00aa": "Contexte de l'échec de push indisponible", + "054ead86b1": "Corriger l'échec de push avec l'IA", + "15b7f210d7": "Choisissez l'agent et modifiez la commande complète avant le lancement.", + "011f9713fc": "Push bloqué", + "60bd988f0b": "Correction IA", + "03d238218c": "Détails", + "a9bf7c171a": "Échec du push", + "834cb3f23d": "Corriger avec l'IA", + "783a808870": "Fermer" + }, + "97e7124eac": "Impossible d'actualiser le contrôle de code source. Réessayez.", + "b8c2e1a904": "{{value0}} → {{value1}}", + "a4e93c21d7": "Branche actuelle : {{value0}}", + "c7d4e2f801": "Modifier la ref de base : {{value0}}", + "f3a1b8c204": "upstream", + "createPrIntentEmptyGeneratedBody": "Les détails de revue générés n'incluent pas de description. Réessayez Créer une PR.", + "createPrIntentGenerateDetailsFailed": "Impossible de générer les détails de revue. Réessayez Créer une PR." + }, + "SourceControlAgentActionDialog": { + "8e856842d1": "Impossible de démarrer l'agent sélectionné.", + "c075d00de1": "Impossible de résoudre la connexion de l'espace de travail.", + "38b899cc02": "Tous les dépôts", + "808cfe0a3b": "Ce dépôt", + "994cddd1f7": "Ne pas enregistrer" + }, + "SourceControlAgentActionDialogForm": { + "013c9ac04a": "Enregistrer pour", + "1bb611240f": "Utilisez {basePrompt} pour le prompt par défaut d'Orca.", + "23280cbab1": "Ce modèle n'inclut pas {basePrompt}, l'agent ne recevra donc pas le prompt par défaut d'Orca.", + "5421a96acb": "Enregistrer et démarrer l'agent", + "6cefcdfba1": "Vous pourrez le modifier plus tard dans les paramètres IA du contrôle de code source.", + "c29f9cf266": "Enregistrer ce prompt et ne plus afficher cette revue la prochaine fois", + "d8f40128ee": "{basePrompt} est le prompt par défaut d'Orca.", + "ea4788705e": "Annuler", + "7ec6abbf2a": "Réinitialiser", + "f4f3c9ca4a": "Modèle de prompt", + "1bc0bdbb5e": "Lancement :", + "fe119187bb": "--model sonnet", + "bc8dc39f4b": "Arguments CLI", + "b99c33cec5": "Paramètres", + "15c5d85706": "Agent", + "3e8f21954f": "error", + "74168d7ada": "idle", + "1d47db9bf0": "Aucun agent activé", + "c7ff8cef11": "Détection des agents...", + "b0da3a4d3e": "Recette de lancement déjà enregistrée", + "bff4795a6d": "Modifiez l'agent, les arguments ou le modèle de prompt pour mettre à jour la recette enregistrée.", + "5c75b24735": "Personnalisez ce que reçoit l'agent avant qu'Orca ne le démarre.", + "repoAgentOverrideNote": "Ce dépôt remplace votre défaut global ({{global}}) et exécute actuellement {{effective}}. Enregistrez dans ce dépôt pour changer ce qui s'exécute ici." + }, + "SourceControlTextGenerationDialog": { + "c5b7fa7cb6": "Enregistrer comme défaut global", + "7f1ec309a4": "Enregistrer comme défaut pour tous les dépôts", + "5959da1e4d": "Enregistrer pour ce dépôt uniquement", + "d054d5e0a0": "Les paramètres ne sont pas chargés." + }, + "SourceControlTextGenerationDialogForm": { + "25fcd8e49a": "Enregistrer les défauts", + "d91b0a189d": "Enregistrer la recette", + "1f6fcfb6cf": "Modèle de commande", + "551ffd111b": "--model sonnet", + "4eab815004": "Arguments CLI", + "914c8f6ac2": "Commande personnalisée", + "cce2cbd01d": "Choisir un agent", + "9c14186dd2": "Agent" + }, + "activity": { + "bar": { + "buttons": { + "1fd284e931": "Autres onglets de la barre latérale", + "f1132ea95d": "neutre" + } + } + }, + "checks": { + "panel": { + "content": { + "3916814392": "en retard (commit de base :", + "755be805f6": "Aucun commentaire", + "751f7c6e5c": "Affichage des 100 premiers commentaires par source", + "94557d68e2": "Commentaires", + "3fff651d32": "Ajouter un commentaire de PR", + "ea9fd5ed6a": "Démarrer une conversation...", + "0fc6f743b3": "par", + "8987d5a3dd": "Résolu", + "ba20d1a896": "Répondre à {{value0}}", + "f6a40263ff": "Enregistrer", + "b062f55f29": "Annuler", + "c1f6fc006a": "Répondre", + "2ba0a32bdd": "bot", + "6cc6eace26": "Supprimer", + "03ca88f623": "Édition", + "d3923d18fe": "Aller au commentaire", + "1abb17aac9": "Plus", + "74c6885b8a": "Plus d'actions de commentaire", + "cbcc4ab3db": "Affichage des 100 premières vérifications", + "0dca6bfab5": "Ouvrir les détails de la vérification", + "991f50c7e4": "Aucune vérification configurée", + "9ad98f2a17": "en attente", + "5e52f4ef7f": "en échec", + "02ca4f9074": "en succès", + "checksUnresolvedChip": "non résolu", + "checksUnresolvedStripHint": "Ces vérifications se sont terminées sans verdict de succès ni d'échec.", + "e15a8b77ef": "Aucun détail en ligne n'est disponible pour cette vérification.", + "dcb3c546fe": "Réessayer", + "679bf2093c": "Copier l'extrait de log", + "d713f500b2": "Extrait de log", + "a916648574": "Ouvrir les détails", + "07eccfa397": "Aucun détail disponible pour cette vérification.", + "49731703ea": "Jobs", + "f2fe8a4e8f": "Annotations", + "d098e5529a": "Sortie", + "2dd5ddabc4": "workflow #", + "aa8494ae3c": "vérification #", + "00e1c1658a": "Terminé", + "fd46a70f1a": "Démarré", + "a54ae21c6f": "Statut :", + "e4e3af15ee": "Voir tous les détails", + "b8c4e2a1f7": "Voir tous les logs", + "a2fb3f4408": "Affichage des 100 premiers jobs", + "df137989b3": "Affichage des 20 premières annotations", + "1f2b980522": "Chargement des détails de la vérification…", + "0c96cd25e5": "Résoudre", + "3a71a6ed0b": "Résolvez les conflits pour que les vérifications et le merge puissent aboutir.", + "60186d8498": "Des conflits bloquent ceci", + "c16762ac8c": "Les vérifications et commentaires ci-dessous montrent le contexte récupéré actuellement.", + "9d0e7bcefc": "Aucune action PR bloquante", + "5856874b59": "Orca actualisera les vérifications tant que ce panneau reste ouvert.", + "5341023167": "check", + "b45db92d0e": "Corriger", + "5d4ebf9391": "Inspectez les détails ou lancez une passe de correction IA.", + "b652f38caf": "vérification en échec", + "87cd07c69a": "Cette branche contient des conflits qui doivent être résolus", + "0975eeaaef": "Fichiers en conflit", + "6fa7f8723f": "commit", + "2b2be92919": "Ajouter un commentaire", + "7440d09d2c": "Démarrer une conversation", + "b37ebdc51c": "Commentaires indisponibles.", + "90206b6353": "comment", + "95ad090b01": "thread", + "365254cc1b": "Marquer comme non résolu", + "7f793b571d": "Glisser pour redimensionner les vérifications", + "ee07b33924": "unknown", + "cdbfda4dec": "Annotation", + "066fedd446": "Jobs en échec", + "ae8a04ef17": "Les détails du fichier en conflit sont indisponibles", + "73d0675356": "Actualisation des détails de conflit…", + "f5bc5c4cf1": "Le fournisseur d'hébergement signale des conflits, mais Git en local ne les a pas reproduits. Actualisez la revue ou poussez la branche pour recalculer la possibilité de merge.", + "5bc9bda2af": "Exécuter depuis ce worktree", + "e87fb3d929": "Copier les commandes de recalcul du merge", + "1e53e45072": "Copied", + "084c516efb": "Copier les commandes", + "5dc3af25c0": "Sélectionner le commentaire", + "d7a2f9c401": "Envoyer les {{value0}} commentaires non résolus", + "d91f2a6c39": "Envoyer les {{value0}} commentaires en file à l'IA", + "a6de3e5a20": "Vider les commentaires en file", + "49ea0937e4": "Ajouter le commentaire à la liste de résolution", + "9fecebb29d": "Ajouter", + "f8a2c91d04": "Mettre en file pour l'agent", + "b4e8a1c902": "En file d'attente", + "7c1f0a2b11": "Ouvert", + "e8b4c1a903": "Résolu · {{value0}}", + "c3a8e5d710": "À examiner · {{value0}}", + "a7f0c7e8d1": "File", + "8a621a2c4f": "Groupés", + "b13f85d75c": "Chronologie", + "f5cf324efa": "Options d'affichage des commentaires", + "5e6e5a13fa": "Affichage", + "actionRequiredHint": "Cette vérification nécessite une action manuelle sur GitHub (par exemple, approuver l'exécution du workflow) avant que le merge soit débloqué.", + "b3195cba33": "Retirer le marqueur bot de l'auteur", + "f588b46a6c": "Marquer l'auteur comme bot", + "e45324fbed": "Échec du chargement des détails de la vérification." + }, + "empty": { + "state": { + "3322603418": "Statut de la pull request indisponible", + "5b0cfae9a5": "Créez une {{value0}} pour lancer les vérifications et la revue.", + "13e1c7d5ed": "Aucune {{value0}} trouvée", + "d372072df1": "L'actualisation GitHub est suspendue par le budget de rate-limit actuel", + "7c299df37b": "Aucune pull request trouvée", + "3d4af82ff4": "Actualisation du statut GitHub pour cette branche", + "938b5606a6": "Recherche de pull request", + "6ba2440770": "En attente de l'actualisation du statut GitHub pour cette branche", + "2bdd7aaf2d": "Le statut GitHub n'a pas pu être actualisé. Les données en cache ont été conservées.", + "5f478ab3d3": "Impossible d'actualiser la pull request", + "6ce9d4e069": "Poussez votre branche avant de créer une {{value0}}.", + "76e15946a9": "La branche contient des commits non poussés", + "f8543140cc": "Publiez cette branche avant de créer une {{value0}}.", + "41252bc53f": "Branche non publiée", + "05e4aec17b": "Les vérifications {{value0}} seront disponibles une fois l'opération terminée", + "d77c513c1e": "{{value0}} en cours", + "b597440265": "Actualisez le statut GitHub de cette branche pour charger les vérifications et la revue." + } + }, + "review": { + "detail": { + "positive": "Orca dispose également d'informations {{reviewLabel}} enregistrées qu'il n'a pas pu vérifier.", + "rate_limited": "Orca n'a pas non plus pu vérifier le statut de la {{reviewLabel}} car {{provider}} limite temporairement les requêtes.", + "network": "Orca n'a pas non plus pu vérifier le statut de la {{reviewLabel}} car cet environnement n'a pas pu joindre {{provider}}.", + "untyped": "Orca n'a pas non plus pu confirmer si cette branche possède déjà une {{reviewLabel}}." + }, + "positive": { + "title": "Détails de la {{reviewLabelCap}} indisponibles", + "body": "Orca possède des informations {{reviewLabel}} enregistrées pour cette branche, mais n'a pas pu confirmer son statut actuel." + }, + "no_review": { + "title": "Aucune {{reviewLabel}} trouvée", + "body": "Créez une {{reviewLabel}} pour lancer les vérifications et la revue." + }, + "active": { + "title": "Vérification du statut de la {{reviewLabel}}", + "body": "Orca interroge {{provider}} pour trouver une {{reviewLabel}} sur cette branche." + }, + "git_loading": { + "title": "Vérification du statut de la branche", + "body": "Orca vérifie cette branche avant d'afficher les actions de création ou de publication." + }, + "git_error": { + "title": "Impossible de vérifier le statut de la branche", + "body": "Orca n'a pas pu confirmer l'upstream de cette branche depuis cet environnement. Réessayez avant de publier ou de créer une {{reviewLabel}}." + }, + "unknown": { + "title": "Statut de la {{reviewLabelCap}} indisponible", + "body": "Orca n'a pas confirmé le statut de la {{reviewLabel}} pour cette branche. Réessayez pour revérifier." + }, + "paused": { + "title": "Actualisation {{provider}} suspendue", + "body": "{{provider}} limite temporairement les requêtes. Cela peut se produire même quand le quota API affiché n'est pas épuisé." + }, + "network": { + "title": "Impossible de joindre {{provider}}", + "body": "Cet environnement n'a pas pu joindre {{provider}}. Vérifiez sa connexion, puis réessayez." + }, + "unknown_error": { + "title": "Impossible de vérifier le statut de la {{reviewLabel}}", + "body": "La recherche a échoué : Orca n'a pas pu confirmer si cette branche possède déjà une {{reviewLabel}}." + }, + "untyped": { + "title": "Statut de la {{reviewLabelCap}} indisponible", + "body": "Orca n'a pas pu confirmer si cette branche possède déjà une {{reviewLabel}}. Réessayez pour revérifier." + }, + "auth": { + "title": "Échec de l'authentification {{provider}}", + "body": "{{provider}} n'a pas pu authentifier les identifiants disponibles dans cet environnement. Vérifiez le login {{provider}} ou le token d'environnement, puis réessayez." + }, + "permission": { + "title": "Accès {{provider}} refusé", + "body": "Les identifiants {{provider}} actuels ne permettent pas de lire les {{reviewLabel}}s de ce dépôt. Vérifiez le compte, les scopes du token et l'accès au dépôt, puis réessayez." + }, + "repo": { + "title": "Dépôt {{provider}} indisponible", + "body": "{{provider}} n'a pas pu résoudre ni accéder au dépôt pour le remote et le compte actuels. Vérifiez le remote et l'accès au dépôt, puis réessayez." + }, + "cli": { + "title": "CLI {{provider}} indisponible", + "body": "Orca n'a pas pu exécuter la CLI {{provider}} dans cet environnement. Configurez-la ici, puis réessayez." + }, + "skipped": { + "disconnected": { + "title": "Hôte déconnecté", + "body": "L'hôte d'exécution de ce dépôt est déconnecté ; Orca ne peut donc pas actualiser le statut de la {{reviewLabel}}." + }, + "bare": { + "title": "Dépôt bare", + "body": "Ce dépôt est bare ; le statut de la {{reviewLabel}} n'est donc pas disponible ici." + }, + "archived": { + "title": "Dépôt archivé", + "body": "Ce dépôt est archivé ; Orca n'actualise donc pas le statut de la {{reviewLabel}}." + }, + "not_git": { + "title": "N'est pas un dépôt Git", + "body": "Orca n'a pas pu traiter ce dossier comme un dépôt Git pour le statut de {{reviewLabel}}." + }, + "remote": { + "title": "Contexte distant uniquement", + "body": "Orca n'a pas pu actualiser le statut de {{reviewLabel}} pour ce contexte distant. Réessayez une fois l'hôte disponible." + } + }, + "no_upstream": { + "title": "Aucun upstream configuré", + "body": "Publiez cette branche pour définir son upstream avant de créer une {{reviewLabel}}." + }, + "needs_sync": { + "title": "La branche doit être synchronisée", + "body": "Synchronisez cette branche avec son upstream avant de créer une {{reviewLabel}}." + }, + "auth_required": { + "title": "Connecter {{provider}}", + "body": "{{provider}} doit être connecté dans cet environnement avant qu'Orca puisse créer une {{reviewLabel}}." + }, + "needs_push": { + "title": "La branche contient des commits non poussés", + "body": "Poussez les derniers commits avant de créer une {{reviewLabel}}." + }, + "existing": { + "title": "{{reviewLabelCap}} existe déjà", + "body": "Orca a trouvé une {{reviewLabel}} existante pour cette branche." + }, + "detached": { + "title": "Aucune branche courante", + "body": "Effectuez un checkout d'une branche avant de créer une {{reviewLabel}}." + }, + "dirty": { + "title": "Commitez d'abord vos modifications", + "body": "Commitez ou stashez vos modifications avant de créer une {{reviewLabel}}." + }, + "default_branch": { + "title": "Sur la branche par défaut", + "body": "Passez sur une branche de fonctionnalité avant de créer une {{reviewLabel}}." + }, + "fork": { + "title": "Tête de fork non prise en charge", + "body": "Orca ne peut pas créer une {{reviewLabel}} à partir de cette tête de fork ici." + }, + "base_missing": { + "title": "Branche de base absente du remote", + "body": "La base de cette branche n'est pas encore sur le remote ; une {{reviewLabel}} ne peut donc pas la cibler." + }, + "unsupported": { + "title": "{{reviewLabelCap}} non prise en charge ici", + "body": "Ce fournisseur de dépôt ne prend pas en charge la création d'une {{reviewLabel}} depuis Orca." + } + } + } + }, + "gitlab": { + "mr": { + "merge": { + "state": { + "04a3015a12": "Peut être fusionné", + "53c6d3b7e9": "GitLab indique que cette MR peut être fusionnée, mais le pipeline est toujours en cours", + "65c847ad1e": "Vérifications en attente", + "b41fbc180c": "GitLab indique que cette MR peut être fusionnée, mais certains jobs du pipeline ont échoué", + "49ac4fec10": "Vérifications échouées", + "22b7e50621": "GitLab signale des conflits de fusion", + "96b05e374c": "Conflits", + "d63bb6f76e": "Cette demande de fusion est encore au stade de brouillon", + "b2715092c6": "Brouillon", + "2388413f28": "Cette demande de fusion est fermée", + "88d044c42f": "Fermé", + "ee482a2bad": "Cette demande de fusion est déjà fusionnée", + "fae95ae20d": "Fusionnés", + "5105b0e584": "Approbation requise", + "46dc85711b": "GitLab exige une approbation avant que cette MR puisse être fusionnée", + "893a999c8c": "Modifications demandées", + "c65408a99d": "Un relecteur a demandé des modifications sur cette demande de fusion", + "01ab632a21": "Fils non résolus", + "4191fdfcec": "GitLab exige que les discussions non résolues soient résolues avant la fusion", + "bf628700f6": "Les vérifications doivent passer", + "37d8637c70": "GitLab exige que le pipeline réussisse avant que cette MR puisse être fusionnée", + "049ea91a82": "Le pipeline est toujours en cours", + "382c70faeb": "En retard", + "2c61100b3a": "Mettez à jour la branche avant de merger", + "195917574e": "Vérification", + "ff6db2db63": "GitLab calcule encore le statut de cette demande de fusion", + "4ecc0d90ee": "Bloquée", + "a0f3de7023": "GitLab signale que cette demande de fusion est bloquée", + "804ecf93e8": "GitLab n'a pas communiqué de statut de fusion final", + "a5c049afe4": "GitLab indique que cette MR peut être fusionnée" + } + } + } + }, + "index": { + "70893f017b": "Côté", + "7b415c39e9": "Haut", + "864111caa2": "Position de la barre d'activité", + "e8e2e4ce74": "Basculer la barre latérale droite", + "441733b630": "Ports", + "83a10e3c44": "Vérifications", + "0314901467": "Contrôle de code source", + "06219e4cb1": "Recherche", + "8bc2bbc3a0": "Explorateur", + "45b78f03bc": "côté", + "34af8aadf5": "haut", + "9fffaf17c1": "Basculer la barre latérale droite ({{value0}})", + "b37ff4a89a": "ports", + "9f83375839": "checks", + "6306b48afd": "source-control", + "ef182dcb12": "recherche", + "fc3095d2ed": "explorateur", + "aiVaultSessionHistory": "Agents", + "folderWorkspaces": "Worktrees attachés", + "parentPrChecks": "Vérifications de la PR" + }, + "right": { + "panel": { + "comment": { + "composer": { + "9bca633dee": "Annuler", + "cf5a7aba6f": "Liste", + "d6d9c3c947": "Citation", + "f49e0a21e0": "Code", + "542bf6a7e2": "Italique", + "256300f8ea": "Gras", + "87aff03d63": "Envoi…", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque." + } + } + } + }, + "source": { + "control": { + "ai": { + "commit": { + "failure": { + "launch": { + "a8b97d2318": "Un agent IA a été lancé pour l'échec du commit.", + "5540ff50cc": "Impossible de construire la commande de lancement de l'agent.", + "9bbd9077a2": "Aucun agent IA activé. Configurez les agents dans les paramètres.", + "d481ab22f9": "L'agent IA enregistré est indisponible. Utilisez « Personnaliser le lancement » pour choisir un autre agent.", + "f2b47026e8": "Le prompt d'échec de commit est vide. Mettez à jour les paramètres IA du contrôle de code source.", + "4f4e0418a0": "Impossible de construire le prompt de l'agent.", + "216f762bd7": "Impossible de résoudre la connexion de l'espace de travail." + } + } + }, + "push": { + "failure": { + "launch": { + "216f762bd7": "Impossible de résoudre la connexion de l'espace de travail.", + "4f4e0418a0": "Impossible de construire le prompt de l'agent.", + "f2b47026e8": "Le prompt d'échec de push est vide. Mettez à jour les paramètres IA du contrôle de code source.", + "d481ab22f9": "L'agent IA enregistré est indisponible. Utilisez « Personnaliser le lancement » pour choisir un autre agent.", + "9bbd9077a2": "Aucun agent IA activé. Configurez les agents dans les paramètres.", + "5540ff50cc": "Impossible de construire la commande de lancement de l'agent.", + "a8b97d2318": "Un agent IA a été lancé pour l'échec du push." + } + } + }, + "recovery": { + "launch": { + "4f4e0418a0": "Impossible de construire le prompt de l'agent.", + "push": { + "empty": "Le prompt d'échec de push est vide. Mettez à jour les paramètres IA du contrôle de code source." + }, + "commit": { + "empty": "Le prompt d'échec de commit est vide. Mettez à jour les paramètres IA du contrôle de code source." + }, + "d481ab22f9": "L'agent IA enregistré est indisponible. Utilisez « Personnaliser le lancement » pour choisir un autre agent.", + "9bbd9077a2": "Aucun agent IA activé. Configurez les agents dans les paramètres.", + "5540ff50cc": "Impossible de construire la commande de lancement de l'agent.", + "216f762bd7": "Impossible de résoudre la connexion de l'espace de travail.", + "success": "Un agent IA a été lancé pour l'échec de {{value0}}." + } + } + }, + "discard": { + "confirmation": { + "2ae5a785b3": "Abandonner toutes les modifications non indexées ?", + "ddf36f291c": "Cela désindexera et annulera toutes les modifications indexées. Les nouveaux fichiers indexés seront supprimés. Cette action est irréversible.", + "5ddd8cac7f": "Abandonner toutes les modifications indexées ?", + "1426c2efff": "Cela annulera toutes les modifications de ce fichier. Cette action est irréversible.", + "d4df3a61df": "Abandonner les modifications de « {{value0}} » ?", + "40e9357b2a": "Cela restaurera le fichier depuis HEAD et annulera la suppression. Cette action est irréversible.", + "5c0bdbc4cb": "Restaurer « {{value0}} » ?", + "d97bf697c9": "Cela supprimera définitivement ce fichier. Cette action est irréversible.", + "96c772bee9": "Supprimer « {{value0}} » ?" + }, + "dialog": { + "3bc61dc989": "Annuler", + "15efa778e3": "Abandonner", + "6de99d162b": "entrée", + "42f89dd030": "fichiers", + "e7611dca35": "fichier", + "48c5ef95d9": "zone", + "0d2d88cba5": "Cette action est irréversible.", + "1551c14668": "Abandonner les modifications ?" + } + }, + "dropdown": { + "items": { + "7aad2c0240": "Opération de revue hébergée en cours…", + "9e779995dd": "Créer {{value0}}", + "226b85a3a7": "Fetch", + "323bb614aa": "Commit & Sync", + "2b8e6595fd": "Commit", + "04d709801d": "Fetch depuis le remote sans fusionner" + } + }, + "primary": { + "action": { + "ed93b4f14f": "Commit", + "946a8a05ea": "Créer une {{value0}} pour cette branche", + "e7ffa46946": "Créer {{value0}}", + "95550cff15": "Push", + "d64292a938": "Pull", + "795f1509c5": "Synchroniser", + "390abeab93": "Force Push", + "1884cf34af": "Publier cette branche vers origin", + "7b4d02e6b8": "Publier la branche", + "3d5dccef0b": "Rien à commiter. La PR est déjà fusionnée.", + "41d4bcf157": "Vérification du statut de la PR…", + "acce237921": "Rien à commiter. La branche n'a aucune modification à publier.", + "fa3bd4f40c": "Indexez au moins un fichier pour commiter", + "5a477d80cb": "Indexer toutes les modifications", + "18a0fca877": "Tout indexer", + "f01f16d77f": "Saisissez un message de commit pour commiter", + "ab41fb926b": "Committer les modifications indexées", + "2d8f185fbc": "Indexez toutes les modifications avant de commiter des fichiers partiellement indexés", + "a6457b46a7": "Résolvez les conflits avant de commiter", + "484f45c439": "{{value0}} en cours…", + "6f7a8b9c0d": "Opération distante en cours…", + "7f8a9b0c1d": "Opération distante en cours — réessayez une fois celle-ci terminée", + "74fc171e99": "Force Push en cours…", + "16aee3a5c1": "Commit en cours…", + "e61b0d7a3c": "Effectuez un checkout d'une branche avant de publier des commits.", + "1d47e850cf": "Poussez les mises à jour vers la branche de revue liée", + "c39d0c75c3": "La cible de la branche de revue liée est indisponible.", + "8c6d15a07d": "Créer une PR", + "d37e68f61d": "Préparation de la branche pour la revue…", + "c72e5e65d1": "Préparez cette branche et créez une {{value0}}", + "b8e4f2a901": "Attendez la fin de l'opération distante.", + "c9f3a1b802": "Résolvez les conflits avant de créer une {{value0}}.", + "d2a8c4e703": "Aucune modification sur cette branche à inclure dans une {{value0}}.", + "e3b9d5f814": "Impossible de créer une {{value0}} depuis la branche par défaut.", + "f4c0e6a925": "Commitez vos modifications avant de créer une {{value0}}.", + "a5d1f7b036": "Publiez les commits avant de créer une {{value0}}.", + "b6e2a8c147": "Poussez les commits avant de créer une {{value0}}.", + "c7f3b9d258": "Synchronisez cette branche avant de créer une {{value0}}.", + "d8a4c0e369": "Authentifiez-vous avant de créer une {{value0}}.", + "e9b5d1f470": "Effectuez un checkout d'une branche avant de créer une {{value0}}.", + "f0c6e2a581": "Cette branche n'est pas encore prête pour une {{value0}}.", + "8f9a0b1c2d": "Rien à commiter. La branche est à jour.", + "h3i4j5k607": "Vérification de la possibilité de créer une {{value0}} pour cette branche…" + } + }, + "branch": { + "line": { + "total": { + "chip": { + "daa8e8e59b": "{{value0}} lignes ajoutées, {{value1}} lignes supprimées", + "8a9b97b666": "{{value0}} lignes ajoutées", + "52c366d88d": "{{value0}} lignes supprimées", + "4c1f70ba92": "{{value0}} — code de test : {{value1}} lignes ajoutées, {{value2}} lignes supprimées", + "6b2d0f14a7": "Tests", + "9e4a3c5081": "Hors tests", + "3d7e9b1042": "Non généré", + "a1b2c3d4e5": "Répartition du code", + "b2c3d4e5f6": "Lignes de code", + "7f3e1a9c24": "{{value0}} — généré : {{value1}} lignes ajoutées, {{value2}} lignes supprimées", + "c8d5b21e07": "Source", + "7a04c6f8b3": "Généré" + } + } + } + }, + "compare": { + "summary": { + "dd72a6fd37": "{{value0}} commit{{value1}} d'avance sur {{value2}}" + } + }, + "conflict": { + "status": { + "cards": { + "5302a1ddba": "Conflits de fusion", + "7f3af87549": "Conflits de rebase", + "6a8e9ad490": "Conflits de cherry-pick", + "bdf8772106": "Conflits", + "edc2d82a2b": "Fusion en cours", + "5c3707aa44": "Rebase en cours", + "ffe53a1da6": "Cherry-pick en cours", + "35eb76d323": "Opération en cours" + } + } + }, + "content": { + "status": { + "3f425c239c": "Aucune modification sur cette branche", + "640f6fdb36": "Cet espace de travail est propre et cette branche n'a aucune modification en avance sur {{value0}}", + "8deb86bbec": "base", + "978eba351e": "Le texte de recherche est trop volumineux", + "0bce43409f": "Utilisez un filtre de fichiers plus court.", + "1b6caf533d": "Aucun fichier correspondant", + "00c07771b7": "Aucun fichier modifié ne correspond à « {{value0}} »" + } + }, + "diff": { + "comments": { + "list": { + "cefacd0ec7": "fichier entier" + } + } + }, + "commit": { + "area": { + "cc5739bd1d": "Génération du message de commit…", + "4cbd0cd9d2": "Commit en cours…", + "e7876a2bde": "Indexez au moins un fichier pour générer un message.", + "0904be2505": "Effacez le message pour le régénérer.", + "1ea9ba37aa": "Choisissez un agent dans Paramètres -> Git -> IA du contrôle de code source." + } + } + } + }, + "use": { + "source": { + "control": { + "ai": { + "cfafa92509": "Aucun conflit non résolu à envoyer." + }, + "bulk": { + "actions": { + "2f67630884": "Échec de l'indexation/désindexation groupée" + } + } + } + } + }, + "fileExplorerOperationOwner": { + "unresolved": "Impossible de déterminer quel hôte possède cet espace de travail. Vérifiez la connexion et réessayez." + }, + "useFileDeletion": { + "72691dfebc": "Échec : {{value0}} « {{value1}} ».", + "96affe1302": "« {{value0}} » déplacé vers {{value1}}", + "74727df633": "« {{value0}} » supprimé", + "d979a4fbb5": "Supprimer définitivement « {{value0}} » ?", + "a76c74f105": "destructive", + "92276aceb7": "Supprimer", + "7fb9435c86": "Cette action supprime définitivement le répertoire et son contenu sur l'hôte distant. Elle est irréversible.", + "23e98f192f": "Cette action supprime définitivement le fichier sur l'hôte distant. Elle est irréversible.", + "8b8ee9d22f": "Impossible de déterminer quel hôte possède ce fichier. Vérifiez la connexion de l'espace de travail et réessayez.", + "af1270b90d": "Supprimer définitivement {{count}} éléments ?", + "dd029aa5cd": "Cette action supprime définitivement les éléments sélectionnés ainsi que le contenu des répertoires sur l'hôte distant. Elle est irréversible.", + "77fdc36183": "Supprimer {{count}} éléments ?", + "fca915a67a": "Les éléments distants sont définitivement supprimés, sans retour arrière. Les éléments locaux sont déplacés dans {{value0}}." + }, + "useFileExplorerHandlers": { + "32cd9fd991": "Impossible d'ouvrir la cible du symlink" + }, + "useFileExplorerImport": { + "25919b2050": "Ignoré : {{value0}} {{value1}}.", + "132fd0e1e9": "Échec de l'importation de {{value0}} {{value1}}." + }, + "useFileExplorerKeys": { + "8adb953095": "Échec de l'opération" + }, + "GitHistoryGraphSvg": { + "47eff48230": "HEAD" + }, + "create": { + "pull": { + "request": { + "review": { + "copy": { + "a1f8c3d2e4": "Push réussi, mais la création de {{value0}} a échoué : {{value1}}" + } + } + } + }, + "hosted": { + "review": { + "button": { + "label": { + "96ae7358e0": "Push et création de PR dans la stack", + "8e8149a0bf": "Créer une PR brouillon dans la stack", + "8df1a05952": "Créer une PR dans la stack" + } + } + } + } + }, + "AiVaultPanel": { + "resumeCommandCopied": "Commande de reprise copiée", + "valueCopied": "{{value0}} copié", + "valueCopyFailed": "Impossible de copier {{value0}}", + "openWorkspaceBeforeResuming": "Ouvrez un espace de travail avant de reprendre une session.", + "localWorkspacesOnly": "La reprise depuis l'historique n'est disponible que dans les espaces de travail locaux.", + "agentSessionQueued": "Session {{value0}} en file d'attente", + "sessionHistory": "Historique des sessions d'agent", + "agents": "Agents", + "shownRecent": "{{value0}} affichées · {{value1}} récentes", + "sessionsShownCompact": "{{value0}} affichées", + "resumePastSessions": "Reprendre les sessions passées", + "refreshSessionHistory": "Actualiser l'historique des sessions", + "searchSessions": "Rechercher des sessions", + "clearSearch": "Effacer la recherche", + "remoteBrowseLocalHistory": "Les espaces de travail distants peuvent parcourir l'historique local. Les actions de reprise s'exécutent depuis les espaces de travail locaux.", + "transcriptsSkipped": "{{count}} transcript ignoré", + "noAgentSessionsFound": "Aucune session d'agent trouvée", + "noSessionsMatchFilters": "Aucune session ne correspond aux filtres actuels", + "noAgentsSelected": "Aucun agent sélectionné", + "sessionId": "ID de session", + "logPath": "Chemin du log", + "originalPaneUnavailable": "Le volet d'origine n'est plus disponible.", + "worktreeUnavailable": "Le worktree n'est plus disponible.", + "openSupportedWorkspace": "Ouvrez un espace de travail avant de reprendre une session.", + "sessionHostMismatchUnsupported": "Cette session appartient à un autre hôte. Ouvrez un espace de travail sur le même hôte pour la reprendre.", + "localSessionSshWorkspaceUnsupported": "L'historique de cette session est stocké sur cette machine ; impossible de la reprendre dans un espace de travail SSH. Ouvrez plutôt un espace de travail local.", + "prepareSessionResumeFailed": "Impossible de préparer cette session pour la reprise.", + "sessionDeleted": "Session supprimée", + "sessionDeleteFailed": "Impossible de supprimer la session" + }, + "AiVaultPanelControls": { + "scanningSessions": "Analyse des sessions", + "scopeAriaLabel": "Portée de l'historique des sessions : {{value0}}", + "currentWorkspaceLower": "espace de travail actuel", + "currentWorktreeLower": "worktree actuel", + "allSessionsLower": "toutes les sessions", + "thisScope": "Actuel", + "allScope": "Tous", + "scope": "Portée", + "currentWorkspace": "Espace de travail actuel", + "allSessions": "Toutes les sessions", + "viewOptionsAriaLabel": "Options d'affichage de l'historique des sessions", + "viewOptions": "Options d'affichage", + "agents": "Agents", + "selectAllAgents": "Tout sélectionner", + "clearAgents": "Effacer", + "sort": "Trier", + "lastUpdated": "Dernière mise à jour", + "created": "Créé", + "group": "Regroupement", + "folder": "Dossier", + "agent": "Agent", + "resetView": "Réinitialiser la vue", + "hideEmptySessions": "Masquer les sessions vides", + "workspaceScope": "Espace de travail", + "worktreeScope": "Worktree", + "globalScope": "Global", + "projectScope": "Projet", + "currentProjectLower": "projet actuel", + "project": "Projet", + "hostScopeAriaLabel": "Hôte de l'historique des sessions : {{value0}}", + "host": "Hôte" + }, + "AiVaultSessionDetails": { + "originalAsk": "Demande initiale", + "latestTurns": "Derniers échanges", + "noPreviewAvailable": "Aucun aperçu de conversation disponible", + "conversationNotSaved": "Conversation non enregistrée", + "recoverableEmptyDetail": "Cette session n'a pas de conversation enregistrée, mais {{value0}} élément(s) récupérable(s) subsistent.", + "recoverableEmptyOpenLogHint": "Ouvrez le log pour les récupérer.", + "emptyConversationDetail": "Cette session n'a pas de conversation enregistrée et ne peut pas être reprise.", + "queuedMessages": "{{value0}} message(s) en file d'attente", + "subagentTranscripts": "{{value0}} transcript(s) de sous-agent", + "messageCount": "{{value0}} msg", + "updated": "Mis à jour", + "created": "Créé", + "workingDir": "Répertoire de travail", + "unknownLocation": "Emplacement inconnu", + "branch": "Branche", + "model": "Modèle", + "usage": "Utilisation", + "usageValue": "{{value0}} msg{{value1}}", + "tokenSuffix": " · {{value0}} tok", + "session": "Session", + "sessionId": "ID de session", + "copyDetailValue": "Copier {{value0}}", + "latestLog": "Dernier log", + "noReadablePreview": "Aucun aperçu de message lisible dans ce transcript.", + "resumeCommand": "Commande de reprise", + "sessionActions": "Actions de la session {{value0}}", + "resumeInNewTab": "Reprendre dans un nouvel onglet", + "resumeInWorktree": "Reprendre dans le worktree", + "viewLog": "Afficher le log", + "copyResumeCommand": "Copier la commande de reprise", + "openLog": "Ouvrir le log", + "revealLog": "Révéler le log", + "openWorkingDirectory": "Ouvrir le répertoire de travail", + "copySessionId": "Copier l'ID de session", + "copyLogPath": "Copier le chemin du log", + "unknownTime": "Heure inconnue", + "unknown": "Inconnu", + "user": "Utilisateur", + "assistant": "Assistant", + "tool": "Outil", + "system": "Système", + "log": "Log", + "justNow": "À l'instant", + "minutesAgo": "il y a {{value0}} min", + "hoursAgo": "il y a {{value0}} h", + "daysAgo": "il y a {{value0}} j", + "monthsAgo": "il y a {{value0}} mois", + "yearsAgo": "il y a {{value0}} an(s)", + "userRole": "Vous", + "agentRole": "Agent", + "toolRole": "Outil", + "systemRole": "Système", + "sessionRole": "Session", + "jumpToOriginalPane": "Aller au volet d'origine", + "worktree": "Worktree", + "jumpToWorktree": "Aller au worktree", + "prompt": "Prompt", + "firstPrompt": "Premier prompt", + "recentPrompt": "Prompt récent", + "firstPromptCopied": "Premier prompt copié", + "recentPromptCopied": "Prompt récent copié", + "copyFirstPrompt": "Copier le premier prompt", + "copyRecentPrompt": "Copier le prompt récent", + "copyPrompt": "Copier le prompt", + "copied": "Copied", + "copy": "Copier", + "noFirstPromptAvailable": "Aucun premier prompt disponible", + "loadingFirstPrompt": "Chargement du premier prompt…" + }, + "AiVaultSessionRow": { + "noPreviewAvailable": "Aucun aperçu de conversation disponible", + "recoverableBadge": "Non enregistré", + "dragToResume": "Glissez pour reprendre dans un nouvel onglet", + "resumeAgentSession": "Reprendre la session {{value0}}", + "resumeInNewTab": "Reprendre dans un nouvel onglet", + "toggleSessionDetails": "Détails de la session {{value0}}", + "copyResumeCommand": "Copier la commande de reprise", + "openLog": "Ouvrir le log", + "revealLog": "Révéler le log", + "openWorkingDirectory": "Ouvrir le répertoire de travail", + "copySessionId": "Copier l'ID de session", + "copyLogPath": "Copier le chemin du log", + "messageCount": "{{value0}} msg", + "tokenCount": "{{value0}} tok", + "hideDetails": "Masquer les détails", + "showDetails": "Afficher les détails", + "moreSessionActions": "Autres actions de session", + "moreActions": "Plus d'actions", + "userRole": "Vous", + "agentRole": "Agent", + "toolRole": "Outil", + "systemRole": "Système", + "sessionRole": "Session", + "jumpToOriginalPane": "Aller au volet d'origine", + "jumpToWorktree": "Aller au worktree", + "subagentCountSingular": "1 sous-agent", + "subagentCountPlural": "{{value0}} sous-agents", + "delete": "Supprimer", + "deleteReasonNonLocalHost": "Seules les sessions de cet appareil peuvent être supprimées.", + "deleteReasonSyntheticPath": "Cette session ne peut pas être supprimée depuis Orca.", + "deleteReasonUnsupportedAgent": "{{value0}} sessions ne peuvent pas être supprimées depuis Orca." + }, + "AiVaultSessionDeleteDialog": { + "title": "Supprimer cette session ?", + "description": "« {{value0}} » sera supprimée. Une fois supprimée, elle ne pourra plus être reprise depuis la ligne de commande de {{value1}} non plus.", + "confirm": "Supprimer" + }, + "FileExplorerNameFilter": { + "26fb73c6e3": "Rechercher des fichiers", + "4d5a6b2a49": "Effacer le filtre de fichiers", + "7a9fb1e6aa": "Contenu" + }, + "FileExplorerViewSwitch": { + "c4e9a2b713": "Noms", + "b3c8f1a902": "Filtrer les fichiers par nom", + "f8a2c4d1e0": "Mode de recherche de l'explorateur" + }, + "GitHistoryCommitFiles": { + "a1b2c3d4e5": "Chargement des fichiers…", + "b2c3d4e5f6": "Aucune modification de fichier dans ce commit", + "c3d4e5f6a7": "Ouvrir toutes les modifications ensemble" + }, + "GitHistoryRow": { + "2f9c41ab07": "Afficher les fichiers du commit {{value0}} : {{value1}}", + "4a8d9e0c1f": "Masquer les fichiers du commit {{value0}} : {{value1}}" + }, + "GitHistoryCommitContextMenu": { + "7b1c4e9a02": "Ouvrir le commit dans le navigateur", + "8c2d5fab13": "Copier le hash du commit", + "9d3e60bc24": "Copier le message de commit", + "ae4f71cd35": "Expliquer les modifications" + }, + "pull": { + "policy": { + "notice": { + "merge": "Merge", + "mergeDescription": "Crée un commit de fusion quand il y a des changements à la fois en local et sur le remote.", + "rebase": "Rebase", + "rebaseDescription": "Rejoue les commits locaux au-dessus de la branche distante.", + "fastForwardOnly": "Fast-forward uniquement", + "fastForwardOnlyDescription": "Ne fait un pull que lorsqu'aucune fusion ni rebase n'est nécessaire.", + "title": "Le pull nécessite une stratégie", + "diverged": "Divergée", + "body": "Cette branche contient des commits locaux et distants. Exécutez une commande dans ce worktree ou sur l'hôte SSH, puis réessayez Pull ou Sync.", + "copyAria": "Copier la commande de pull pour la stratégie {{value0}}", + "copied": "Copied", + "copyCommand": "Copier la commande" + } + } + }, + "AiVaultSessionWorktree": { + "currentWorktree": "Worktree actuel", + "activeWorktree": "Worktree actif", + "archivedWorktree": "Worktree archivé", + "unavailableWorktree": "Worktree indisponible", + "jumpToWorktree": "Aller au worktree", + "noRecordedWorktree": "Aucun worktree n'a été enregistré pour cette session.", + "archivedJumpUnavailable": "Cette session se trouve dans un worktree archivé.", + "noActiveWorktreeMatch": "Aucun worktree actif ne correspond à cette session.", + "noActiveWorktreeTarget": "Aucun worktree actif n'est disponible." + }, + "AiVaultSessionSubagents": { + "subagentsCount": "Sous-agents ({{value0}})", + "messageCount": "{{value0}} msg", + "viewLog": "Afficher le log" + }, + "aiVaultSessionLogOpen": { + "workspaceGone": "Impossible d'ouvrir le log — l'espace de travail n'est plus disponible.", + "notAuthorized": "Impossible d'ouvrir le log — chemin non autorisé.", + "alreadyEditable": "Le log est déjà ouvert en édition." + }, + "github": { + "refresh": { + "error": { + "copy": { + "580025e7b7": "GitHub est indisponible", + "01c85b5770": "L'API de GitHub est temporairement indisponible. Ce panneau se recharge automatiquement dès qu'elle revient.", + "d1a9f2b165": "Impossible de joindre GitHub", + "7d01d42a3a": "GitHub est injoignable pour le moment. Vérifiez votre connexion, puis réessayez dans quelques instants.", + "e9d681894a": "Limite de requêtes GitHub atteinte", + "8c77434d6f": "GitHub limite actuellement les requêtes. Ce panneau s'actualisera dès que la limite sera réinitialisée.", + "79aa06bb2c": "Actualisation impossible. L'API de GitHub est temporairement indisponible. Affichage du dernier statut connu.", + "6ec12cee0c": "Actualisation impossible. GitHub est injoignable pour le moment. Affichage du dernier statut connu.", + "de088015e8": "Actualisation impossible. GitHub limite les requêtes. Affichage du dernier statut connu.", + "d9dd7c6687": "Actualisation depuis GitHub impossible. Affichage du dernier statut connu." + } + } + }, + "pr": { + "stack": { + "confirmation": { + "84f6f5b9eb": "Inclus : {{numbers}}. ", + "541984b2eb": "Ajouter à la file via #{{pr}} ?", + "4809f55cdb": "{{included}}GitHub ajoutera {{count}} pull request à la file de fusion en une seule fois. La file choisit la méthode de fusion et peut les fusionner en groupes séparés.", + "be8f2621be": "{{included}}GitHub ajoutera {{count}} pull requests à la file de fusion en une seule fois. La file choisit la méthode de fusion et peut les fusionner en groupes séparés.", + "92ca033e72": "Mettre {{count}} PR en file", + "478a527b15": "Mettre {{count}} PR en file", + "1feef35ca4": "Fusionner via #{{pr}} ?", + "c3e036c99f": "{{included}}GitHub fusionnera {{count}} pull request de manière atomique avec {{method}}. Si elle ne peut pas être fusionnée, rien ne le sera.", + "369aba4b32": "{{included}}GitHub fusionnera {{count}} pull requests de manière atomique avec {{method}}. Si l'une ne peut pas être fusionnée, aucune ne le sera.", + "493c78f521": "Fusionner {{count}} PR", + "eb7051d268": "Fusionner {{count}} PR" + }, + "merge": { + "55ae29b907": "Fusionner via #{{pr}} · {{count}} PR", + "b8446f6ec2": "Fusionner via #{{pr}} · {{count}} PR", + "189d0ec614": "#{{pr}} est toujours un brouillon.", + "640fb50d9c": "#{{pr}} est fermée.", + "46ffcbda75": "#{{pr}} présente des conflits de fusion.", + "6dabefd63e": "#{{pr}} a demandé des modifications.", + "2bb21fc326": "#{{pr}} attend encore une approbation de relecture.", + "c23faf74df": "#{{pr}} doit être mise à jour.", + "f561e80968": "#{{pr}} est bloquée." + } + } + } + }, + "PluginPanel": { + "unavailable": "Ce panneau de plugin n'est plus disponible.", + "loading": "Chargement du panneau de plugin…", + "unresponsive": "Ce panneau de plugin a cessé de répondre et a été suspendu.", + "loadFailed": "Le panneau de plugin n'a pas pu être chargé." + }, + "activityBar": { + "error": "Erreur" + }, + "AiVaultSessionLimitMenu": { + "historyDepth": "Profondeur de l'historique : {{value0}}", + "performanceWarning": "Des historiques plus longs peuvent ralentir toute l'application, en particulier sur des hôtes distants. « Illimité » analyse tout l'historique disponible.", + "recommended": "Recommandé", + "mayBeSlower": "Peut être plus lent", + "slowest": "Le plus lent", + "unlimited": "Illimité" + }, + "GitHubPRStackMap": { + "3511405914": "closed", + "8a9bdc36c0": "fusionnée", + "568c647ccd": "brouillon", + "bea9ade223": "conflits", + "838aadf512": "vérifications échouées", + "316039b5db": "vérifications en attente", + "4b1e5ee9d3": "modifications demandées", + "9a17b5255c": "relecture requise", + "d3d97cf3f2": "approuvée", + "e6cb964305": "ouvrir", + "7737bd66be": "Réduire la stack #{{value0}}", + "0c1645ebd0": "Développer la stack #{{value0}}", + "e3ee2daa32": "Stack #{{value0}}", + "cb440931b7": "{{value0}} sur {{value1}} · {{value2}}", + "525259fa17": "Les détails de la stack sont temporairement indisponibles." + }, + "CreateHostedReviewBasePicker": { + "205ef284fa": "Branche de base", + "bb4b41d563": "à partir de {{value0}}", + "5a9315b61a": "Aucune branche ne correspond à « {{value0}} ». Appuyez sur Entrée pour l'utiliser quand même.", + "da4d57c9c2": "Laissez vide pour utiliser {{value0}}." + }, + "CreateHostedReviewComposerFields": { + "90cabf6cfc": "Empiler cette PR au-dessus de #{{value0}}", + "ff81473a57": "Crée une GitHub Stack ou étend la stack existante du parent.", + "29732f2fb0": "nouvelle PR" + } + } + }, + "repo": { + "NestedRepoChecklist": { + "f7e1170567": "sélectionnés", + "ea54c7bf8f": "sur", + "91b5bcadb6": "Tout sélectionner", + "929734aea5": "Tout désélectionner" + }, + "NestedRepoScanLimitNotice": { + "642a43c139": "Limites du scan des dépôts imbriqués", + "574eb5408b": "Affichage de résultats de scan partiels.", + "03e9beab7b": "Le scan s'est arrêté prématurément." + }, + "RepoCombobox": { + "b3e15f4525": "Ajouter un projet", + "b4a235e886": "Ajout d'un projet", + "3639fd9da2": "SSH", + "e7ed739236": "Aucun projet/dossier ne correspond à votre recherche.", + "a0c48f5f29": "Rechercher des projets/dossiers…", + "116812151a": "Ajout du projet…" + }, + "repo": { + "icon": { + "0ad395d475": "Boîte", + "857977b901": "Formes", + "b1b8d99fc4": "AI", + "137bdb1856": "Métriques", + "d202c659a3": "Design", + "c4fd14299d": "Entreprise", + "4ab9433660": "Travail", + "febfbe0cd5": "Outils", + "ecf63ec3ef": "Lancement", + "31826b712e": "API", + "70bef15d40": "Calques", + "b5fac337aa": "Calcul", + "d37b4e2641": "Serveur", + "3c5a593bc8": "Web", + "477b28c948": "Base de données", + "787490e9bd": "Paquet", + "07012dc113": "Agent", + "3eba7387ab": "Terminal", + "65b437c381": "Code", + "bed2674f9d": "Dossier" + } + } + }, + "pet": { + "pet": { + "models": { + "7433516faf": "Gremlin", + "a84d5677ff": "OpenCode", + "2528586aa7": "Claudino" + } + } + }, + "onboarding": { + "AgentStep": { + "e6a369bd04": "Agents populaires", + "d7b3ef168b": "Détectés sur votre système", + "9c163bb0e0": "Instructions d'installation", + "69af7e9c1c": "n'est pas encore dans votre PATH. Orca le définira comme valeur par défaut et vous pourrez l'installer à tout moment.", + "1eee1c7bd8": "Aucun agent détecté dans votre PATH. Choisissez-en un à installer plus tard, ou continuez avec un terminal vide.", + "hideAgents": "Masquer les agents", + "showMoreAgents": "Afficher {{value0}} autres agents→", + "yoloPermissionsLabel": "Yolo / Ignorer dangereusement les permissions", + "yoloPermissionsInfo": "Informations sur les permissions des agents", + "yoloPermissionsTooltip": "Ignorer les vérifications de permissions des agents pour moins d'interruptions" + }, + "FeatureSetupInlineTerminal": { + "789b59936e": "Appuyez sur Entrée pour exécuter la commande et confirmez npx si demandé. Vous pouvez aussi configurer cela plus tard dans les paramètres.", + "47fc6cc6dc": "Commande de configuration du skill", + "c767ab7061": "Configuration du skill" + }, + "IntegrationsStep": { + "277f30eb34": "Linear, GitLab, Bitbucket, Azure DevOps, Gitea et Jira se trouvent dans Paramètres > Intégrations.", + "3a3e360289": "Autres sources de tâches", + "80e3ce0bc9": "Revérifier", + "04ef416712": "Ajouter l'accès Linear", + "dd9c186a8b": "Ajouter un accès à l'espace de travail", + "c91a5782f1": "Connecté", + "27743304b1": "Linear", + "af69f42372": "Appuyez sur Entrée pour lancer l'authentification GitHub CLI. Revérifiez GitHub une fois le flux navigateur ou appareil terminé.", + "f9d2e12d17": "Commande de connexion à GitHub", + "6d469169f2": "Configuration de GitHub", + "bd5d976fb2": "Installer gh", + "50db38cf4b": "Pull requests, issues et statut des vérifications.", + "c1547656f0": "Vérification…", + "8405043962": "Connexion requise", + "5c115cb713": "CLI non installé", + "217beb0658": "GitHub", + "4983ae7433": "Ajoutez l'accès à Linear avec une clé d'API personnelle. Les clés à accès complet peuvent afficher toutes les équipes accessibles au propriétaire de la clé.", + "b08a6ac93c": "Espace de travail {{value0}}{{value1}} associé. Ajoutez un autre espace de travail ou remplacez une clé restreinte à tout moment.", + "93f0c49ad1": "not-authenticated", + "a3bcf13694": "connecté", + "d6e5dba05a": "Se connecter", + "0b4a7d23ab": "Connexion en cours", + "a74c6d6b18": "not-installed" + }, + "NotificationStep": { + "3bede04483": "Envoyer une notification de test", + "dc897423e1": "Choisir le son de notification", + "53aaffe49a": "Son de notification", + "0fe570690c": "Choisissez l'alerte jouée par Orca après l'envoi d'une notification bureau.", + "0af746e41f": "Choisir un son", + "8124d085a6": "Ouvrir les réglages Mac", + "aa36281b00": "Ouvrez Réglages Système et vérifiez qu'Orca est autorisé à envoyer des notifications.", + "d2dba86837": "Autoriser Orca dans macOS", + "e52aacf380": "Chargement des réglages de notification…", + "3cd5374e22": "Les réglages de notification sont encore en cours de chargement", + "b6a994e36e": "Impossible de lire le son de notification", + "c0692baa52": "Choisir un fichier personnalisé", + "ac80d97e02": "Changer de fichier personnalisé", + "56b836215c": "Vérification de l'autorisation de notification…", + "fd84d3e9b8": "Les notifications sont activées", + "4f7bce5644": "macOS vous alertera quand un agent aura terminé ou qu'un terminal demandera votre attention.", + "95d99b52fa": "Autoriser les notifications pour Orca", + "94562ba367": "macOS demande l'autorisation. Cliquez sur Autoriser dans la boîte de dialogue : cette étape se mettra à jour automatiquement.", + "4f6a1da718": "Ouvrir Réglages Système", + "90b5d2e363": "macOS ne délivre pas les notifications d'Orca", + "2c47f5465f": "Activez « Autoriser les notifications pour Orca » dans Réglages Système. Cette étape se met à jour automatiquement une fois l'option activée." + }, + "OnboardingFlow": { + "1b5e182e9f": "Bienvenue dans Orca", + "4db04f2f57": "sur", + "adaa0aa627": "Aller à l'étape {{value0}} de la prise en main : {{value1}}", + "a249f81538": "Orca", + "277ba45540": "Prise en main d'Orca", + "97c42cda00": "Installez le CLI GitHub pour :", + "ae3b00ca82": "Configurer les tâches GitHub", + "ff92d15436": "Orca vous préviendra quand les agents auront terminé ou auront besoin d'aide.", + "b054332836": "Configurer les notifications", + "04ae28d8ca": "Choisissez le look que vous allez contempler pendant des heures.", + "f396db9f20": "Donnez-lui vos couleurs", + "322fc50a18": "Orca fonctionne avec tous les agents CLI. Choisissez celui que vous solliciterez le plus. Changez quand vous voulez.", + "198b148b3c": "Choisissez votre agent par défaut", + "a5e5da02f7": "intégrations", + "35bbaf5ae0": "notifications", + "984338477a": "thème", + "c47e1bd149": "agent", + "windowsTerminalTitle": "Définir les valeurs par défaut du terminal Windows", + "windowsTerminalSubtitle": "Choisissez le shell PAR DÉFAUT des nouveaux volets et le comportement du clic droit dans le terminal." + }, + "OnboardingFooter": { + "ba58547306": "Retour", + "111d3f8d92": "Passer à la configuration du projet" + }, + "OnboardingInlineCommandTerminal": { + "4123609efd": "Démarrage du terminal..." + }, + "OnboardingSkipConfirmationDialog": { + "9f47f345a4": "Ça ne sera pas long !", + "e4726b2d50": "Passer la prise en main ?" + }, + "ThemeStep": { + "a4b254779d": "Touche Option macOS", + "6c51398942": "Souris", + "8ca01945f2": "Séparateurs", + "b3a99a2d29": "Fenêtre", + "86c0f1caa2": "Marge interne", + "06a24f4f2d": "Couleurs", + "c021e9dddd": "Palette du thème", + "ab2a583a97": "Cursor", + "cc1858e19e": "Police", + "248c812283": "Importer", + "7ee9234e54": "Configuration Ghostty détectée.", + "78b6386140": "Importé depuis Ghostty.", + "2c3aa538f8": "Recherche d'une configuration Ghostty…", + "94b9dc561d": "Réglages → Terminal", + "dd5c16ad1b": "D'autres options de terminal, notamment police, curseur et palette, dans", + "ad192706e6": "Clair", + "fa7b673ea9": "Sombre", + "827ea7b4a2": "Système", + "699ddf83c2": "Échec de l'import des réglages Ghostty", + "16a9f0446a": "Aucun réglage Ghostty à importer", + "ad19e5c916": "Importation…", + "906c4373fe": "paramètres" + }, + "use": { + "onboarding": { + "flow": { + "52acfbef51": "Impossible d'enregistrer la progression" + } + } + }, + "WindowsTerminalStep": { + "powerShell": "PowerShell", + "powerShellPwsh": "Utilise PowerShell 7+ quand il est disponible, avec Windows PowerShell en repli.", + "powerShellInbox": "Utilise Windows PowerShell, présent sur toutes les installations de Windows prises en charge.", + "commandPrompt": "Invite de commandes", + "commandPromptDescription": "Ouvre les nouveaux volets de terminal avec le comportement classique de cmd.exe.", + "gitBash": "Git Bash", + "gitBashDescription": "Utilise le bash.exe de Git for Windows pour les workflows shell de type Unix.", + "gitBashUnavailable": "Sélectionné, mais Git Bash n'a pas été détecté sur cette machine.", + "wsl": "WSL", + "wslDescription": "Démarre les nouveaux volets de terminal dans la distribution par défaut de votre Windows Subsystem for Linux.", + "wslUnavailable": "Sélectionné, mais WSL n'a pas été détecté sur cette machine.", + "rightClickPaste": "Coller au clic droit", + "rightClickPasteDescription": "Le clic droit colle le presse-papiers. Ctrl+clic droit ouvre le menu contextuel.", + "rightClickMenu": "Ouvrir le menu contextuel", + "rightClickMenuDescription": "Le clic droit ouvre le menu du terminal. Collez depuis le menu ou le clavier.", + "loading": "Chargement des réglages du terminal...", + "defaultShell": "Shell par défaut", + "defaultShellDescription": "Choisissez le shell qu'Orca ouvre pour les nouveaux volets de terminal sous Windows.", + "wslDistribution": "Distribution WSL", + "wslDistributionDescription": "Utilisez la distribution Windows par défaut ou choisissez une distribution installée précise.", + "loadingDistros": "Chargement des distributions", + "windowsDefault": "Défaut Windows", + "rightClickBehavior": "Comportement du clic droit", + "rightClickBehaviorDescription": "Choisissez le comportement souris du terminal qui correspond à vos réflexes Windows." + }, + "mac": { + "notification": { + "permission": { + "card": { + "f696515944": "Cliquez sur Autoriser dans la boîte de dialogue macOS.", + "3d18cf71f9": "Se met à jour automatiquement.", + "721d2bedb6": "Activez « Autoriser les notifications pour Orca » dans Réglages Système." + } + } + } + } + }, + "new": { + "workspace": { + "SetProjectLocationDialog": { + "title": "Définir l'emplacement du projet", + "description": "Choisissez l'emplacement de {{project}} sur {{host}}.", + "browseFolder": "Parcourir le dossier", + "browseFolderHelp": "Utilisez un checkout ou un dossier existant sur cet hôte.", + "cloneFromUrl": "Cloner depuis une URL", + "cloneFromUrlHelp": "Clonez ce dépôt sur {{host}}.", + "saveLocation": "Définir l'emplacement" + }, + "SmartWorkspaceNameField": { + "2a0d535f69": "Créer une nouvelle branche", + "a44229ce4d": "comme nom d'espace de travail", + "766083a596": "\"", + "34ca97bce3": "\"", + "b1a7d679ba": "Utiliser", + "e57c53727c": "Ajouter un projet...", + "a76fcb4fa0": "Basculer vers", + "eadf877af5": "Conserver", + "6859e2896c": "Annuler", + "9ef1a7c4b0": ", qui est différent du projet sélectionné.", + "ad188067ae": "L'URL GitHub pointe vers", + "4bd98f1091": "Changer de projet ?", + "0c9e668e3a": "Effacer", + "7199ff19c7": "Effacer la source sélectionnée", + "370a1faf67": "Ouvrir dans le navigateur", + "2c69728c2a": "Ouvrir le lien dans le navigateur", + "6f07a18604": "Nom", + "2e4c7c95fe": "Branche", + "2cfc6be192": "GitLab", + "7a47af0565": "Linear", + "0a180280bd": "GitHub", + "b3c60c2b7c": "Intelligent", + "26824f60dd": "Tous", + "6fad211c66": "Fermé", + "2319d87718": "Fusionnés", + "622864b52a": "Ouvert", + "fda67f0b61": "projet actuel", + "3e8bb1176a": "Connectez Linear dans les réglages pour rechercher des issues.", + "69ce292138": "linear", + "9c004911c3": "gitlab", + "switchTaskSourceTitle": "Changer de source de tâches ?", + "differentTaskSource": ", qui est différent de la source de tâches sélectionnée.", + "currentTaskSource": "source de tâches actuelle", + "placeholderNameOrLinearUrl": "Saisissez un nom, une URL Linear ou une URL Jira", + "placeholderWorkspaceName": "Saisissez un nom d'espace de travail", + "placeholderSmartWithBranchGitLabLinear": "Saisissez un nom, #1234, une branche, une URL GitHub/GitLab, Linear ou Jira", + "placeholderSmartGitLabLinear": "Saisissez un nom, #1234, une URL GitHub/GitLab, Linear ou Jira", + "placeholderSmartWithBranchGitLab": "Saisissez un nom, #1234, une branche, une URL GitHub, GitLab ou Jira", + "placeholderSmartGitLab": "Saisissez un nom, #1234, une URL GitHub, GitLab ou Jira", + "unavailable": "Indisponible", + "searchGitHub": "Rechercher des PR et issues GitHub", + "searchGitLab": "Rechercher des MR et issues GitLab", + "searchBranches": "Rechercher des branches", + "searchLinear": "Rechercher des issues Linear", + "workspaceName": "Nom de l'espace de travail", + "loadingJira": "Chargement de l'issue Jira…", + "jiraDisconnected": "Connectez Jira dans les réglages pour associer cette issue", + "jiraSiteNotConnected": "Ce site Jira n'est pas connecté", + "jiraRuntimeUpdate": "Mettez à jour le runtime distant pour associer Jira", + "jiraReadFailed": "Impossible de charger cette issue Jira", + "chooseJiraAccount": "Choisir un compte Jira", + "jiraLoaded": "Issue Jira chargée", + "openSettings": "Paramètres", + "retryJira": "Réessayer", + "searchJira": "Recherchez des issues Jira ou collez l'URL d'une issue", + "jiraMode": "Jira", + "jiraSelectBindFailed": "Impossible d'associer cette issue Jira. Sélectionnez le site correspondant ou reconnectez Jira, puis réessayez.", + "emoji": "Émoji", + "loadingLinearIssue": "Chargement de l'issue Linear…" + }, + "ProjectCombobox": { + "empty": "Aucun projet ne correspond à votre recherche.", + "addProject": "Ajouter un nouveau projet", + "label": "Projet", + "browse": "Parcourir les projets", + "listLabel": "Projets", + "noProjects": "Aucun projet pour le moment." + }, + "RunTargetCombobox": { + "listLabel": "Cibles d'exécution", + "label": "Exécuter sur", + "browse": "Parcourir les cibles d'exécution" + } + } + }, + "mobile": { + "MobileHero": { + "a8fb43cf1c": "Continuer", + "3f90dbd274": "Terminé", + "b622eba64d": "Retour", + "65b3f2e8bc": "Génération…", + "27735e5f4e": "QR d'appairage", + "bb0074ce11": "Code QR d'appairage", + "010dddcf27": "Copier le code d'appairage", + "4c1df4eba7": "Impossible de scanner ?", + "85067b9e06": "Actualiser les interfaces réseau", + "ca85e595a7": "Aucune interface trouvée", + "79d2f480da": "Interface réseau à annoncer", + "dfd2aa9d5d": "Réseau", + "2f077ef4eb": ", puis scannez le code.", + "3aa7bb2d8b": "Appairer l'ordinateur", + "d1495e5e64": "Ouvrez Orca Mobile, touchez", + "3960f5c339": "Étape 2 sur 2", + "3241f3c26a": "QR d'installation", + "7af266b80d": "Code QR d'installation", + "aa97420ba4": "Copier le lien d'installation", + "ac1eb64952": "Android", + "711e6f4b47": "iOS", + "e75647ace0": "Scannez le code QR avec votre téléphone ou ouvrez le lien d'installation pour récupérer Orca Mobile.", + "0d9b33299e": "Récupérez l'app.", + "92ddfdfa1f": "Étape 1 sur 2", + "ff48d9d520": "Appairer un autre appareil", + "f9cbf4bb53": "Révoquer l'appareil", + "34f878d04f": "Révoquer {{value0}}", + "94829abdb1": "Appairé", + "266c18c105": "Ouvrez Orca Mobile pour reprendre là où vous en étiez, ou appairez un autre appareil.", + "5410d55d79": "Orca Mobile", + "10d27b4cba": "Commencer", + "da1d5e5ed0": "Disponible sur", + "ec0607bf66": "Plateformes mobiles prises en charge", + "b4ccce5cb7": "Pilotez Orca depuis votre téléphone. Suivez les agents, examinez les changements et lancez des tâches même loin de votre bureau.", + "cd4e5e816f": "Vos espaces de travail, dans votre poche.", + "a6cffbbb0b": "Générer le code", + "e59a252eca": "Régénérer le code", + "d0b52871ce": "Vos téléphones sont appairés.", + "051978a785": "Votre téléphone est appairé.", + "channel": { + "group": "Canal de release", + "preview": "Aperçu", + "stable": "Stable" + }, + "relayDegradedNotice": "Relay injoignable — ce code ne fonctionne que sur votre LAN ou Tailscale.", + "pairingQrError": "Ce code d'appairage n'a pas pu être rendu sous forme de QR code. Copiez-le plutôt dans Orca Mobile.", + "noRelayCode": "Aucun code d'appairage disponible", + "noPairingCode": "Aucun code d'appairage disponible", + "qrSignInRequired": "Connectez-vous pour créer un code d'appairage Relay", + "qrRenderFailed": "Impossible d'afficher le code QR — copiez le code ci-dessous", + "qrGeneratePrompt": "Générez un code d'appairage pour continuer", + "pairingCodeReady": "Code d'appairage prêt", + "pairThisMac": "Appairez ce Mac.", + "pairThisPc": "Appairez ce PC.", + "pairThisComputer": "Appairez cet ordinateur.", + "directAddressDisclosure": "Utiliser aussi un chemin local plus rapide", + "directAddressHint": "Facultatif. Choisissez l'adresse Wi‑Fi ou Tailscale que votre téléphone utilisera à proximité — généralement plus rapide que Relay. Relay reste fonctionnel quand vous êtes loin.", + "androidHelp": { + "guide": "Guide d'installation" + } + }, + "MobilePage": { + "e17393c6a3": "Aperçu du téléphone", + "baea63c445": "Échec de la copie du lien", + "fad833de8d": "Lien d'installation copié", + "6a66e38943": "Échec de la copie du code d'appairage", + "3c1f7168bb": "Code d'appairage copié", + "4c8bd11c1a": "Échec de la génération du code d'appairage", + "b353e18de1": "Le transport WebSocket n'est pas actif", + "4e1eb5d55c": "Échec de la révocation de l'appareil", + "255372e6e8": "Appareil révoqué", + "1b4509a8a1": "appairé", + "c5909374cf": "intro", + "diagnosticsCopied": "Diagnostics copiés", + "diagnosticsCopyFailed": "Échec de la copie des diagnostics" + }, + "MobilePageToolbar": { + "ad2284a9e2": "Fermer · Esc", + "9883b58693": "Fermer Orca Mobile", + "fb5f28330e": "Afficher dans la barre latérale", + "c669abcf8f": "Masquer de la barre latérale", + "e1c7b4a92d": "À configurer dans Réglages > Mobile.", + "a4f8c2d91e": "Masquer de la barre latérale", + "b7e3d1c84a": "Afficher dans la barre latérale", + "f3d8e5b71a": "Remet le raccourci dans la barre latérale." + }, + "PhoneCarousel": { + "96d651cb87": "Session de terminal", + "93217b41c1": "Liste des worktrees", + "89c7713645": "Écran d'accueil d'Orca Mobile" + }, + "mobile": { + "platform": { + "copy": { + "2a532d6fd7": "Scannez avec l'appareil photo de votre Android pour télécharger le dernier APK depuis GitHub Releases.", + "432db52b73": "Scannez avec l'appareil photo de votre iPhone pour ouvrir l'App Store.", + "preview": { + "tagline": "Fonctionnalités les plus récentes, mises à jour chaque jour." + }, + "stable": { + "tagline": "La version publique, mise à jour chaque semaine." + } + } + } + }, + "slides": { + "HomeSlide": { + "a7d9e2c44d": "7 j", + "a3d5476811": "5 h", + "8a350a4784": "Utilisation du compte", + "e27fdaee51": "Nouvel espace de travail", + "4405f3c440": "Appairer l'ordinateur", + "0b00c98506": "Actions rapides", + "0bad5b07c8": "GitHub et Linear", + "d047197480": "GitHub · Linear", + "a4c3f7b7aa": "Tâches", + "d33d7a9c29": "orca · feat/mobile-page", + "25d6e8a491": "feat/mobile-page", + "c791677f2f": "Reprendre", + "cf3f98fa3f": "Déconnecté", + "091355da3d": "M1 Mini · home", + "0bc1881bc4": "Connecté · 40 worktrees · 5 actifs", + "19c212e25e": "MacBook Pro", + "2f1a1d10c4": "Ordinateurs", + "156db8a68a": "PRs créées", + "4a40af029b": "Temps agent", + "00a6903322": "Agents lancés", + "c0e2e9dcd9": "Bon retour", + "af761a0c0d": "Paramètres", + "5d94e8ddcc": "Orca" + }, + "TerminalSlide": { + "0bb39f8fe6": "Envoyer", + "69334b4b10": "Dictée vocale", + "29f2d13839": "Saisissez une commande…", + "817090af40": "Ctrl+C", + "53ff909568": "Tab", + "4930eaaae7": "Esc", + "fa22927f13": "Coller", + "985373052e": "Basculer en mode téléphone", + "58a9ee6003": "avec mise en forme des tool calls. Je vous ajoute le diff ensuite ?", + "aa64b519c6": "écran de terminal. Palette Tokyonight, police Menlo, véritable claude", + "e75112c834": "J'ai remplacé la diapositive pair-scan par un", + "3ce3e8c892": "14 réussis, 1 ignoré (1,8 s)", + "4b3666f9a9": "src/cache/worktree-cache.test.ts", + "1d448b69f7": "PASS", + "d39445686a": "src/transport/host-store.test.ts", + "a6e7cdc688": "pnpm test --filter mobile", + "21b67dfc92": "Bash", + "d6d1041a1c": "⎿ Diapositive pair-scan remplacée par une session de terminal", + "336c0e070e": "mobile/orca-mobile-sidebar-mock-v3.html", + "6d4ebd5833": "Édition", + "fc83e0d5ef": "⎿ Lecture de 2103 lignes", + "80cc356591": "Lecture", + "2c10d43745": "claude", + "e0f98be657": "orca/feat-mobile-page", + "2defc05141": "dev@mac", + "da121ba48d": "PLAN.md", + "e4befee569": "shell", + "606aa93192": "Fichiers", + "94febb0976": "Contrôle de code source", + "8d6516312d": "2 terminaux · claude actif", + "8432787c4e": "feat/mobile-page", + "8fd998acd3": "Retour" + }, + "WorktreeListSlide": { + "357a519567": "Actif", + "79a24ff530": "Épinglés", + "22971156df": "Dépôt", + "17f9e0d226": "Récents", + "0e3e809a4b": "Filtre", + "b4271864bd": "MacBook Pro", + "cefd048225": "Retour", + "c5ad56786d": "spinner" + } + }, + "CustomNetworkAddressDialog": { + "title": "Adresse réseau personnalisée", + "description": "Annoncez une adresse que votre téléphone peut joindre — par exemple un nom d'hôte Tailscale, une adresse IP ou une URL de reverse proxy.", + "label": "Adresse", + "placeholder": "home.example.com:8443 or https://example.com/orca", + "hint": "Saisissez une adresse IPv4/IPv6, un nom d'hôte ou une URL HTTP(S)/WebSocket complète. Les ports sont facultatifs.", + "cancel": "Annuler", + "use": "Utiliser l'adresse", + "confirmationError": "Cette adresse n'a pas permis de générer un code d'appairage scannable. Vérifiez l'adresse et réessayez." + }, + "NetworkInterfacePicker": { + "no-address-selected": "Aucune adresse sélectionnée", + "trigger-label": "Adresse réseau à annoncer", + "custom-option": "{{address}} (personnalisée)", + "add-custom": "Ajouter une adresse personnalisée…", + "custom-section": "Personnalisé", + "remove-custom": "Supprimer {{address}}" + }, + "WindowsFirewallNotice": { + "repair-success": "Le Pare-feu Windows autorise désormais Orca Mobile sur les réseaux privés", + "repair-failed": "Impossible de mettre à jour les règles du Pare-feu Windows", + "public-title": "Windows marque ce réseau comme public", + "missing-title": "Autoriser les connexions des téléphones via le Pare-feu Windows", + "public-description": "Définissez ce réseau Wi-Fi approuvé comme Privé avant d'autoriser les connexions Orca Mobile.", + "missing-description": "Windows peut bloquer le serveur d'appairage. Ajoutez une règle pour cette application Orca et le port TCP {{port}} sur les réseaux privés.", + "open-settings": "Ouvrir les paramètres réseau", + "waiting": "Attente de Windows…", + "allow": "Autoriser les connexions des téléphones", + "repair-unverified": "Impossible de vérifier l'accès au Pare-feu Windows", + "blocked-title": "Windows bloque peut-être Orca Mobile", + "blocked-description": "Une règle entrante de blocage existante peut prendre le pas sur l'exception d'appairage. La réparation supprime les règles TCP conflictuelles de cette application Orca, puis autorise le port {{port}} sur les réseaux privés.", + "repair": "Réparer l'accès pare-feu", + "relay-note": "L'appairage fonctionne toujours via Orca Relay — cette autorisation ajoute seulement la connexion locale, plus rapide." + }, + "MobileRelayMintFailureNotice": { + "retryingTitle": "Nouvelle tentative via Orca Relay…", + "unavailableTitle": "Orca Relay n'est pas disponible sur ce poste de bureau.", + "title": "Impossible de créer un code d'appairage Relay.", + "retryingBody": "Création d'un nouveau code d'appairage. Cela peut prendre un moment via une connexion distante.", + "unavailableBody": "Utilisez le LAN pour appairer via Tailscale ou le même Wi-Fi.", + "body": "Réessayez, ou utilisez le LAN pour appairer via Tailscale ou le même Wi-Fi.", + "useLan": "Utiliser le LAN", + "retrying": "Nouvelle tentative…", + "retry": "Réessayer Relay", + "copyDiagnostics": "Copier les diagnostics" + } + }, + "gitlab": { + "gitlab": { + "rate": { + "limit": { + "display": { + "ebc0e8ecf1": "Chargement du quota de l'API GitLab...", + "a2d3d1fdde": "Le quota de l'API GitLab est indisponible.", + "a2f68645ac": "Actualiser le quota de l'API GitLab", + "2f9c16d6c3": "Orca utilise REST via le CLI GitLab.", + "14e144f7a7": "Budget API GitLab", + "3e2c982cfa": "restants, réinitialisation dans", + "ea8ad0bae8": "sur", + "0a891e8935": "API REST", + "953f7c6062": "Cet hôte GitLab n'a pas renvoyé d'en-têtes de limite de débit.", + "budget_scope_prefix": "Périmètre du quota" + } + } + } + } + }, + "floating": { + "terminal": { + "FloatingTerminalIconContextMenu": { + "8e7d775287": "Masquer l'espace de travail flottant", + "763f5fa2c1": "Déplacer vers le bouton flottant", + "0ee79e0674": "Déplacer vers la barre d'état" + }, + "FloatingTerminalOrchestrationDialog": { + "f726054620": "Permet aux agents de transmettre le contexte et de coordonner le travail via Orca.", + "1cd3f8af64": "Skill d'orchestration", + "6f0aed26b8": "Installez le CLI Orca et la skill d'orchestration pour que les agents puissent se coordonner via Orca.", + "05d7aabc20": "Non installé", + "630c0ac8c8": "Installés", + "dfd021ce46": "Vérification...", + "543f325a14": "Activer l'orchestration" + }, + "FloatingTerminalPanel": { + "fc1042e92b": "Réduire", + "8b07759314": "Nouveau navigateur", + "88ffb502e5": "Ouvrir une note Markdown", + "629528690b": "Nouvelle note Markdown", + "3215fc73e9": "Nouveau terminal", + "da508bd7f5": "Enregistrer", + "918c2139f3": "Ne pas enregistrer", + "e7bf09d4d4": "Annuler", + "690b6fb98a": "Modifications non enregistrées", + "bbc177f98f": "Activer", + "adc281394d": "Ignorer", + "8cf80db43b": "Configurez le CLI Orca et la skill d'agent pour que les agents puissent se coordonner via Orca.", + "2a3c5ddf5e": "Activer l'orchestration", + "d6b563ae24": "Chargement de l'éditeur...", + "8b14ba6c17": "Nouvel onglet de navigateur", + "b085fb58b5": "Ce fichier contient des modifications non enregistrées.", + "5ddc688c52": "« {{value0}} » contient des modifications non enregistrées. Voulez-vous enregistrer avant de fermer ?", + "25d7817f79": "terminal" + }, + "FloatingTerminalToggleButton": { + "3b04b065b5": "Afficher l'espace de travail flottant", + "4cb418b991": "Afficher l'espace de travail flottant, nouvelle activité", + "5785dd9148": "Réduire l'espace de travail flottant", + "bfe7809a70": "Espace de travail flottant {{value0}} ({{value1}})" + }, + "FloatingTerminalWindowControls": { + "2f6054342c": "Réduire", + "1bbaa0302f": "Réduire l'espace de travail flottant", + "3f4ca29961": "Agrandir l'espace de travail flottant", + "1c79cba25d": "Restaurer l'espace de travail flottant", + "648352c51f": "Ouvrir {{value0}} dans l'espace de travail flottant", + "82da3701e7": "Impossible de construire la commande de lancement pour {{value0}}.", + "109870e023": "Agrandir", + "b5686fee1e": "Restaurer", + "1e502f1284": "Ouvert" + } + } + }, + "feature": { + "wall": { + "AgentCapabilitiesSetupAction": { + "b8dc9dd8a2": "Installés", + "1b51644c2d": "Laissez les agents contrôler le bureau : déplacer le curseur, cliquer et taper dans n'importe quelle application.", + "362a07517d": "Computer Use", + "5e8fe5a72d": "Donnez aux agents un accès direct au navigateur d'Orca pour tester des pages, capturer des captures d'écran et agir sur ce qu'ils voient.", + "e638da007a": "Agent Browser Use", + "c61c91e642": "Laissez les agents se coordonner via Orca pour faire avancer jusqu'au bout les grandes tâches en plusieurs étapes.", + "ac07f8887f": "Orchestration d'agents", + "e9eb197e12": "Autorisations Computer Use ouvertes", + "3a59452a67": "Commande de skill copiée et insérée ci-dessous pour vérification.", + "c605f51f2b": "Configuration des capacités prête", + "1aa657d8f4": "Certaines configurations de capacités demandent votre attention", + "c89534cbe9": "Installer CLI & Skills" + }, + "AiCommitPrSettingsCard": { + "8d4152701a": "ex. ollama run llama3.1 {{value0}}", + "9ee54037a4": "Commande personnalisée", + "4b2fc4b80c": "Effort de réflexion", + "be8917699e": "Modèle", + "4d9b6d84df": "non pris en charge. Choisissez Claude, Codex ou Personnalisé.", + "560d4feb00": "Personnalisé", + "29d119fe95": "Agent", + "f9382b48a1": "Activer l'auteur IA", + "1c0cb4fabb": "Auteur IA", + "bd14e9c42a": "Non configuré", + "1f9468c5c9": "{{value0}} non pris en charge" + }, + "BrowserAnimatedVisual": { + "46df009982": "Démarrer l'essai gratuit", + "25f15c2219": "Pro", + "59ae327405": "Starter", + "9e0f530390": "Tarifs", + "0ce7c24b4d": "Traitement…", + "f2034c4930": ">", + "6e4616d039": "Claude", + "0f8481e1a7": "Envoyer à Claude", + "3d2352f94b": "Décrivez la modification…", + "d8856b604a": "div.pricing-grid > div.card.starter:nth-of-type(1) > a.cta", + "7da6eed7bf": "localhost:3000", + "0a2bd01c02": "Nouvel onglet de navigateur", + "04096318ab": "Terminal 1", + "eb88125c6f": "✓ Vérifié — Essai gratuit toujours fonctionnel.", + "051c97d15a": ".pp-card[data-card=\"starter\"] .pp-cta", + "4fa59ca545": "✓ Mis à jour", + "1bec24acc1": "@keyframes browserFlash { 0% { opacity: 0; } 20% { opacity: 0.85; } 100% { opacity: 0; } } @keyframes browserTabIn { from { opacity: 0; transform: translateY(-2px); } to { opacity: 1; transform: none; } } @keyframes browserViewIn { from { opacity: 0; transform: translateY(4px); } to { opacity: 1; transform: none; } }", + "73bbb46073": "/pricing", + "f39be6ca14": "/signup" + }, + "BrowserUseSkillSetupCard": { + "cbc45022d4": "Permet aux agents de naviguer et de vérifier des pages dans le navigateur d'Orca.", + "d5bb1cd4ba": "Skill Browser Use" + }, + "ComputerUseAnimatedVisual": { + "d8401975b1": "approuvée", + "f27676a92c": "statut :", + "6804cb356f": "clic envoyé", + "1719b28a81": "trouvé « Approve »", + "79445f7512": "approuver la note dans mon app", + "99a8624bcb": ">", + "2adb561b44": "Session Claude Code démarrée", + "94787f01f8": "Claude Code", + "9cddfe96b2": "Application locale", + "9634d870d1": "Approuver", + "3cc2df3671": "Terminé", + "bdd5312213": "En attente", + "c11dda000b": "Approuvé" + }, + "EditorAnimatedVisual": { + "7a763daf2f": "italique", + "8521536429": "gras ·", + "8341391520": "pour les blocs ·", + "3fe42a1da0": "Tapez", + "8268b2376b": "Bloc de code", + "37fa4948ce": "Liste à puces", + "f25687c588": "Citation", + "abbdeea15d": "Blocs de base", + "a26a68d30c": "Titre 2", + "722170663a": "Titre 1", + "1fb29ad710": "Titres", + "4426aab46f": "Mettre à jour l'index de la doc dès que la nouvelle tuile arrive.", + "95f0c3a46f": "Tester le parcours d'installation sur une machine vierge.", + "22ae7b4d9d": "Note rapide pour l'équipe — le point sur ce qui reste avant la sortie.", + "5a55c00a81": "Plan de lancement", + "218503f9f3": "enregistrement auto", + "cda56c5915": "notes / launch-plan.md", + "e16479c1c5": "[data-slash-menu] [data-slash-row].slash-active { background: rgba(24,24,27,0.07); box-shadow: inset 0 0 0 1px rgba(24,24,27,0.06); } [data-md-active-line][data-role=\"active\"] { color: rgb(113 113 122); font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 12.5px; } [data-md-active-line][data-role=\"h1\"] { color: inherit; font-family: inherit; font-size: 18px; font-weight: 700; letter-spacing: -0.01em; line-height: 1.2; margin-top: 6px; } [data-md-caret] { display: inline-block; width: 1.5px; height: 1em; background: currentColor; vertical-align: -2px; margin-left: 1px; animation: md-caret-blink 1.05s steps(1) infinite; } @keyframes md-caret-blink { 0%, 50% { opacity: 1 } 51%, 100% { opacity: 0 } } @keyframes md-block-in { from { opacity: 0; transform: translateY(-2px); } to { opacity: 1; transform: none; } } @keyframes md-cursor-ripple { 0% { transform: scale(0.4); opacity: 0.9; } 100% { transform: scale(1.4); opacity: 0; } } [data-clicking=\"1\"] [data-cursor-ripple] { animation: md-cursor-ripple 460ms ease-out forwards; }" + }, + "FeatureWallBody": { + "25ec5356d6": "Configuration" + }, + "FeatureWallModal": { + "33dca8bbbe": "Une brève visite d'Orca, workflow par workflow.", + "3567e147c8": "Découvrez Orca" + }, + "FeatureWallPreview": { + "a666384798": "Aussi dans ce workflow" + }, + "FeatureWallRail": { + "69ea857689": "Terminé", + "7593d15f94": "Workflows" + }, + "FeatureWallSetupChecklist": { + "1a6a7d6c80": "Configuration", + "713cc529a5": "Jalons", + "b1f1981c5e": "Voir les tâches", + "0235b268b2": "Pas encore terminé", + "13294d3405": "Terminé" + }, + "FeatureWallSetupWorkflowActions": { + "486c2f4d8d": "Ajoutez d'abord un projet git, puis configurez le script de setup de ce dépôt.", + "00078a6134": "Voir dans les réglages", + "14327073cc": "Enregistrer", + "88469e926b": "Script de setup", + "5c5b65044e": "pnpm install", + "a7463915b6": "Échec de l'enregistrement du script de setup", + "6299297dac": "Script de setup enregistré", + "f0bbf7da77": "Essayez", + "522cce9e33": "Ajouter un projet" + }, + "FeatureWallTourPanel": { + "af7d622f6f": "Facultatif" + }, + "KeepAwakeCard": { + "209713d3c7": "Facultatif" + }, + "ReviewNotesAnimatedVisual": { + "5dbd27c4c2": "Codex", + "09094f25e2": "Claude Code", + "294aaff104": "Envoyer les notes à", + "ea4e45b71b": "Ajouter une note", + "271ea0cbf3": "Annuler", + "a7a89d8f94": "Ligne", + "5cb213f967": "Notes IA", + "1eee3a397e": "src/server/migrate.ts (diff)" + }, + "ReviewPRViewAnimatedVisual": { + "7c2808ecff": "troncation avant le merge.", + "c2062da7ec": "stderr", + "6f4c2d7cb7": "Ajouter un cas de couverture pour", + "71828fba75": "Peut-on inclure la commande qui échoue dans la charge utile de diagnostic ?", + "fb1a856b6d": "ouvrir", + "7a8b896e11": "Commentaires", + "ca36f7b27c": "Réussi", + "25f6838e43": "lint", + "2ef0b97954": "typecheck", + "8ed213397c": "En cours", + "d340c052fb": "verify", + "9a097cae12": "1 en attente", + "2f37142229": "Squasher et merger", + "0aab7ab84a": "Ajouter le suivi d'erreurs des diagnostics locaux", + "dfe313e0c9": "OPEN", + "6e3f5223c5": "Explorateur", + "ab2901bce6": "Vérifications", + "d7f80060ca": "Contrôle de code source", + "8e715588e4": "Recherche", + "a6c8b9e32f": "Checks réussis", + "f4d5e1a7b2": "3 vérifications" + }, + "ReviewShipAnimatedVisual": { + "4d99496b8c": "Créer une PR", + "62544e0852": "Annuler", + "bcd5cae3c4": "Description de la pull request", + "3774b80eae": "Description", + "07da9245cc": "Titre de la pull request", + "54a093c52d": "Titre", + "3b9b96d6a6": "main", + "ce7d5d3a18": "Branche de base", + "e4473d438f": "Générer avec l'IA", + "c30cd930ff": "Créer la pull request", + "ea0100dd15": "Tout afficher", + "e725000cd7": "Modifications", + "a079083a6c": "Commit", + "7347fa5839": "Message", + "d1a7f15876": "Générer le message de commit avec l'IA", + "cd8a3a39d7": "3 commits d'avance" + }, + "TasksAnimatedVisual": { + "efba6f77eb": "Lecture de l'issue #", + "b68c92fbdc": "Démarrer l'espace de travail", + "4331c4d0f8": "Ouvert", + "b13375617e": "Le sélecteur de worktrees tronque les noms", + "72f9e516a3": "en appuyant sur", + "fe47c9c9e8": "Espace de travail prêt", + "61ffda7601": "Création de l'espace de travail" + }, + "WorkbenchAnimatedVisual": { + "633a91e358": "Réflexion…", + "932c4b3a97": ">", + "ca2cfbf188": "Scinder le terminal vers le bas", + "e370fa8c2b": "Scinder le terminal vers la droite", + "b85eab49dd": "src/auth/session.ts", + "99f5224f1e": "Édition", + "0d93c298a7": "throw src/auth", + "17cfdc3344": "Grep", + "9923847785": "Lecture", + "c0eb94125e": "revue des cas limites d'authentification", + "431ca9842a": "Session Claude Code démarrée", + "000106adfe": "claude", + "7d9f1d5f7d": "(0,8 s)", + "944199e54a": "› la mise à jour du total du panier", + "623881d72e": "checkout.spec.ts", + "5c5cbd783f": "(1,2 s)", + "3261c6853b": "› la connexion fonctionne", + "defe550fe2": "login.spec.ts", + "0b20782e0f": "Exécution de 12 tests avec 4 workers", + "4371cc9931": "pnpm playwright test", + "16877e038d": "scinde vers le bas", + "a2b114dad0": "scinde à droite ·", + "0bc9ad0cd1": "Même volet :", + "fc84f17fe7": "Session Codex démarrée" + }, + "agent": { + "capability": { + "setup": { + "status": { + "8eccfcb314": "Installés", + "21d4f79c93": "cliquez sur Installer CLI & Skills pour ouvrir les réglages d'accès macOS", + "5c9293e51a": "vérification de l'accès à l'app", + "aae94eeb52": "Cliquez sur Installer CLI & Skills", + "aa8e143a2f": "Impossible de vérifier l'installation", + "9b33e7fb13": "Vérification de l'installation", + "4c8e1f92a7": "ouvrez Orca Desktop sur ce Mac", + "6d2b0a84e1": "Indisponible dans ce build" + } + } + } + }, + "feature": { + "wall": { + "usage": { + "tracking": { + "b94ec70eda": "Suivi non configuré", + "cc39a87288": "Connecté · Par défaut du système", + "00087eecb2": "Connecté · {{value0}}" + } + } + } + }, + "review": { + "animated": { + "visual": { + "shared": { + "e7894927a2": "Codex", + "9deecb021c": "Claude" + }, + "notes": { + "styles": { + "db6691aa0a": ".ravs-window { position: absolute; inset: 0; --ravs-soft-surface: color-mix(in srgb, var(--foreground) 2%, var(--card)); --ravs-soft-fill: color-mix(in srgb, var(--foreground) 6%, transparent); --ravs-panel-border: color-mix(in srgb, var(--foreground) 18%, var(--border)); --ravs-emphasis-border: color-mix(in srgb, var(--foreground) 44%, var(--border)); --ravs-floating-shadow: 0 14px 30px rgb(0 0 0 / 0.22), 0 2px 6px rgb(0 0 0 / 0.12); background: var(--card); border: 1px solid var(--border); border-radius: 10px; overflow: hidden; display: flex; flex-direction: column; box-shadow: 0 1px 2px rgb(0 0 0 / 0.08); } .ravs-difftoolbar { display: flex; align-items: center; gap: 8px; padding: 6px 10px; border-bottom: 1px solid var(--border); background: var(--ravs-soft-surface); font-size: 11px; color: var(--muted-foreground); } .ravs-diff-path { min-width: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; color: var(--foreground); } .ravs-ai-chip { margin-left: auto; display: inline-flex; align-items: stretch; overflow: hidden; border-radius: 6px; border: 1px solid var(--border); background: var(--ravs-soft-surface); opacity: 0; transform: translateY(-2px); transition: opacity 320ms ease, transform 320ms ease; } .ravs-ai-chip.is-visible { opacity: 1; transform: none; } .ravs-ai-chip .ravs-count-btn, .ravs-ai-chip .ravs-send-btn { display: inline-flex; align-items: center; gap: 5px; padding: 3px 8px; font-size: 11px; color: var(--muted-foreground); background: transparent; line-height: 1; } .ravs-ai-chip .ravs-count-btn { border-right: 1px solid var(--border); } .ravs-ai-chip .ravs-send-btn { padding: 3px 7px; position: relative; } .ravs-send-glow { position: absolute; inset: 0; background: rgba(34, 197, 94, 0.18); opacity: 0; transition: opacity 280ms ease; pointer-events: none; } .ravs-ai-chip .ravs-send-btn.is-flash .ravs-send-glow { opacity: 1; } .ravs-count-num { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; color: var(--foreground); font-weight: 600; } .ravs-diffbody { flex: 1; min-height: 0; position: relative; background: var(--editor-surface, var(--card)); } .ravs-diffscroll { position: absolute; inset: 0; overflow: hidden; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 11.5px; line-height: 1.55; color: var(--foreground); padding: 4px 0 8px; transition: opacity 240ms ease; } .ravs-diffscroll.is-hidden { opacity: 0; pointer-events: none; } .ravs-term { position: absolute; inset: 0; background: var(--editor-surface, var(--card)); color: var(--foreground); font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 11px; line-height: 1.45; overflow: hidden; display: flex; flex-direction: column; opacity: 0; pointer-events: none; transition: opacity 240ms ease; z-index: 4; } .ravs-term.is-visible { opacity: 1; } .ravs-term-body { flex: 1; min-height: 0; padding: 10px 12px; overflow: hidden; display: flex; flex-direction: column; gap: 6px; } .ravs-term-line { white-space: pre-wrap; word-break: break-word; line-height: 1.45; } .ravs-term-muted { color: var(--muted-foreground); } .ravs-term-glyph { color: rgb(217 119 6); margin-right: 6px; } .ravs-term-check { color: rgb(16 185 129); font-weight: 700; margin-right: 6px; } .ravs-term-spinner { display: inline-block; width: 8px; height: 8px; margin-right: 6px; border-radius: 999px; border: 1.5px solid color-mix(in srgb, var(--foreground) 20%, transparent); border-top-color: var(--foreground); vertical-align: -1px; animation: ravs-term-spin 0.9s linear infinite; } @keyframes ravs-term-spin { to { transform: rotate(360deg) } } .ravs-hunk-header { display: grid; grid-template-columns: 36px 36px 16px minmax(0,1fr); align-items: center; padding: 1px 8px 1px 0; background: rgba(99, 102, 241, 0.06); color: var(--muted-foreground); font-size: 10.5px; border-top: 1px solid var(--border); border-bottom: 1px solid var(--border); } .ravs-hunk-header .ravs-text { grid-column: 4 / -1; white-space: nowrap; overflow: hidden; text-overflow: ellipsis; color: rgb(99 102 241); font-size: 10.5px; } .ravs-diff-line { display: grid; grid-template-columns: 36px 36px 16px minmax(0,1fr); align-items: stretch; position: relative; } .ravs-ln { text-align: right; padding: 0 6px 0 0; color: var(--muted-foreground); font-size: 10.5px; user-select: none; opacity: 0.85; } .ravs-marker { text-align: center; color: var(--muted-foreground); font-weight: 700; opacity: 0.7; } .ravs-text-cell { padding-right: 8px; white-space: pre; overflow: hidden; } .ravs-tok-kw { color: #a855f7; } .ravs-tok-id { color: #2563eb; } .ravs-tok-str { color: #16a34a; } .ravs-diff-line.is-add { background: color-mix(in srgb, var(--git-decoration-added) 14%, transparent); } .ravs-diff-line.is-add .ravs-marker { color: color-mix(in srgb, var(--git-decoration-added) 72%, transparent); opacity: 1; } .ravs-diff-line.is-rem { background: color-mix(in srgb, var(--git-decoration-deleted) 14%, transparent); } .ravs-diff-line.is-rem .ravs-marker { color: color-mix(in srgb, var(--git-decoration-deleted) 72%, transparent); opacity: 1; } .ravs-add-note-btn { position: absolute; left: 4px; width: 18px; height: 18px; display: inline-flex; align-items: center; justify-content: center; padding: 0; border: 1px solid color-mix(in srgb, currentColor 22%, var(--border)); border-radius: 4px; background: var(--ravs-soft-fill); color: var(--foreground); z-index: 5; opacity: 0; box-shadow: 0 1px 2px rgb(0 0 0 / 0.14); pointer-events: none; transition: opacity 160ms ease; } .ravs-add-note-btn.is-visible { opacity: 1; } .ravs-note-row { padding: 4px 8px 4px 0; max-height: 0; overflow: hidden; opacity: 0; transition: max-height 360ms cubic-bezier(.4,0,.2,1), opacity 280ms ease 60ms, padding 360ms cubic-bezier(.4,0,.2,1); } .ravs-note-row.is-visible { max-height: 90px; opacity: 1; } .ravs-note-card { margin: 0 12px; position: relative; border: 1px solid var(--ravs-panel-border); border-left: 3px solid var(--ravs-emphasis-border); border-radius: 6px; background-color: var(--card); padding: 5px 8px 5px 10px; box-shadow: 0 1px 2px rgb(0 0 0 / 0.16); } .ravs-note-meta { font-size: 9.5px; font-weight: 600; text-transform: uppercase; letter-spacing: 0.04em; color: var(--muted-foreground); } .ravs-note-body { font-size: 11.5px; color: var(--foreground); line-height: 1.35; margin-top: 2px; } .ravs-popover { position: absolute; left: 12px; right: 12px; max-width: none; z-index: 20; padding: 8px 10px; border: 1px solid var(--ravs-panel-border); border-left: 3px solid var(--ravs-emphasis-border); border-radius: 6px; background-color: var(--card); color: var(--foreground); box-shadow: var(--ravs-floating-shadow); display: flex; flex-direction: column; gap: 6px; opacity: 0; transform: translateY(-4px) scale(0.985); pointer-events: none; transition: opacity 180ms ease, transform 180ms ease; } .ravs-popover.is-visible { opacity: 1; transform: none; pointer-events: auto; } .ravs-pop-label { font-size: 10px; font-weight: 600; text-transform: uppercase; letter-spacing: 0.04em; color: var(--muted-foreground); } .ravs-pop-input { min-height: 38px; max-height: 80px; padding: 6px 8px; border: 1px solid var(--border); border-radius: 4px; background: var(--editor-surface, var(--card)); font-size: 12px; line-height: 1.4; color: var(--foreground); white-space: pre-wrap; word-break: break-word; overflow: hidden; } .ravs-pop-footer { display: flex; justify-content: flex-end; gap: 6px; } .ravs-pop-btn { font-size: 11px; font-weight: 500; padding: 4px 9px; border-radius: 5px; line-height: 1; border: 1px solid transparent; display: inline-flex; align-items: center; gap: 5px; } .ravs-pop-btn.is-cancel { color: var(--muted-foreground); background: transparent; } .ravs-pop-btn.is-add { color: var(--primary-foreground); background: var(--primary); } .ravs-send-menu { position: absolute; z-index: 30; right: 8px; top: 6px; min-width: 200px; background: var(--popover); color: var(--popover-foreground); border: 1px solid var(--border); border-radius: 8px; padding: 4px; box-shadow: var(--ravs-floating-shadow); opacity: 0; transform: translateY(-4px) scale(0.985); pointer-events: none; transition: opacity 180ms ease, transform 180ms ease; } .ravs-send-menu.is-visible { opacity: 1; transform: none; pointer-events: auto; } .ravs-menu-section { padding: 4px 8px 2px; font-size: 9.5px; font-weight: 700; text-transform: uppercase; letter-spacing: 0.06em; color: var(--muted-foreground); } .ravs-menu-row { display: grid; grid-template-columns: 16px minmax(0,1fr); align-items: center; gap: 8px; padding: 6px 8px; border-radius: 5px; font-size: 12px; color: var(--popover-foreground); } .ravs-menu-row.is-hot { background: var(--accent); box-shadow: inset 0 0 0 1px var(--border); } .ravs-cursor { position: absolute; z-index: 40; pointer-events: none; transition: transform 600ms cubic-bezier(.45,.05,.2,1), opacity 200ms ease; transform: translate(-30px, 220px); opacity: 0; } .ravs-cursor.is-visible { opacity: 1; } .ravs-cursor .ravs-ripple { position: absolute; left: -6px; top: -6px; width: 28px; height: 28px; border-radius: 999px; border: 2px solid color-mix(in srgb, var(--foreground) 52%, transparent); opacity: 0; } .ravs-cursor.is-clicking .ravs-ripple { animation: ravs-ripple 460ms ease-out forwards; } @keyframes ravs-ripple { 0% { transform: scale(0.4); opacity: 0.9; } 100% { transform: scale(1.4); opacity: 0; } } .ravs-caret { display: inline-block; width: 1.5px; height: 1em; background: currentColor; vertical-align: -2px; margin-left: 1px; animation: ravs-caret-blink 1.05s steps(1) infinite; } @keyframes ravs-caret-blink { 0%, 50% { opacity: 1 } 51%, 100% { opacity: 0 } }" + } + }, + "pr": { + "view": { + "styles": { + "fc9a23c83d": ".ravpr-stage { position: absolute; inset: 0; overflow: hidden; } .ravpr-stack { position: absolute; inset: 0; display: flex; justify-content: flex-end; padding: 4px 34px 4px 2px; overflow: hidden; } .ravpr-sidebar, .ravpr-card { position: absolute; top: 4px; right: 2px; width: 464px; height: calc(100% - 8px); background: var(--card, #fff); border: 1px solid var(--border); border-radius: 10px; color: var(--foreground, #18181b); overflow: hidden; box-shadow: 0 1px 2px rgba(24,24,27,0.04); } .ravpr-sidebar { opacity: 0; transition: opacity 220ms ease; } .ravpr-sidebar.is-visible { opacity: 1; } .ravpr-sidebar.is-hiding { opacity: 0; } .ravpr-card { display: flex; flex-direction: column; min-width: 0; opacity: 0; transition: opacity 260ms ease; } .ravpr-card.is-visible { opacity: 1; } .ravpr-tabs { position: relative; display: flex; align-items: center; gap: 14px; height: 36px; padding: 0 14px; background: rgba(24,24,27,0.015); color: var(--muted-foreground, #71717a); } .ravpr-tab { position: relative; width: 18px; height: 18px; display: inline-flex; align-items: center; justify-content: center; color: var(--muted-foreground, #71717a); } .ravpr-tab.is-active, .ravpr-tab.is-hovered { color: var(--foreground, #18181b); } .ravpr-tab.is-active::after { content: ''; position: absolute; left: -5px; right: -5px; bottom: -10px; height: 1px; background: var(--foreground, #18181b); } .ravpr-tooltip { position: absolute; top: 34px; left: 106px; z-index: 6; padding: 7px 11px; border-radius: 8px; background: var(--card, #fff); color: var(--foreground, #18181b); font-size: 12px; line-height: 1; box-shadow: 0 8px 22px rgba(0,0,0,0.22); opacity: 0; transform: translateY(-3px); pointer-events: none; transition: opacity 160ms ease, transform 160ms ease; } .ravpr-tooltip.is-visible { opacity: 1; transform: translateY(0); } .ravpr-explorer { padding: 10px 12px 12px; } .ravpr-heading { color: var(--muted-foreground, #71717a); font-size: 10px; font-weight: 600; letter-spacing: 0.05em; text-transform: uppercase; } .ravpr-file-list { margin-top: 8px; display: flex; flex-direction: column; gap: 2px; } .ravpr-file { display: grid; grid-template-columns: 14px minmax(0,1fr) 22px; align-items: center; gap: 8px; min-height: 28px; padding: 4px 6px; border-radius: 6px; } .ravpr-file.is-active { background: rgba(24,24,27,0.06); box-shadow: inset 0 0 0 1px rgba(24,24,27,0.06); } .ravpr-file-icon { width: 12px; height: 12px; border-radius: 3px; background: rgba(24,24,27,0.14); } .ravpr-file-name { height: 8px; border-radius: 999px; background: rgba(24,24,27,0.14); } .ravpr-file-status { width: 14px; height: 8px; border-radius: 999px; background: rgba(24,24,27,0.12); } .ravpr-body { padding: 10px 12px 18px; display: flex; flex-direction: column; gap: 5px; min-height: 0; } .ravpr-number-row { display: flex; align-items: center; gap: 7px; } .ravpr-number { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 12px; font-weight: 700; } .ravpr-open { display: inline-flex; align-items: center; justify-content: center; height: 18px; padding: 0 7px; border-radius: 5px; background: rgba(16,185,129,0.10); border: 1px solid rgba(16,185,129,0.28); color: rgb(4 120 87); font-size: 9px; font-weight: 700; line-height: 1; } .ravpr-title { font-size: 12px; font-weight: 600; line-height: 1.35; color: var(--foreground, #18181b); margin-bottom: 2px; } .ravpr-merge { display: inline-flex; align-items: center; justify-content: center; gap: 6px; height: 30px; min-height: 30px; flex: 0 0 30px; border-radius: 7px; background: rgb(22 163 74); color: #fff; font-size: 11.5px; font-weight: 700; margin-bottom: 2px; box-shadow: 0 1px 2px rgba(22,163,74,0.18); transition: box-shadow 220ms ease, filter 220ms ease; } .ravpr-merge.is-ready { box-shadow: 0 0 0 3px rgba(34,197,94,0.22), 0 1px 2px rgba(22,163,74,0.18); } .ravpr-section-row, .ravpr-check-row { display: grid; grid-template-columns: 18px minmax(0,1fr) auto; align-items: center; gap: 7px; padding: 5px 0; font-size: 11.5px; color: var(--foreground, #18181b); } .ravpr-check-row { padding: 5px 7px; font-size: 10.5px; } .ravpr-label { white-space: nowrap; overflow: hidden; text-overflow: ellipsis; } .ravpr-meta, .ravpr-check-state { color: var(--muted-foreground, #71717a); font-size: 10.5px; } .ravpr-check-state { font-size: 10px; } .ravpr-ring { display: inline-block; width: 14px; height: 14px; border-radius: 999px; border: 2px solid rgba(245,158,11,0.35); border-top-color: rgb(245 158 11); animation: ravpr-spin 1.1s linear infinite; } .ravpr-check { width: 15px; height: 15px; border-radius: 999px; display: none; align-items: center; justify-content: center; background: rgba(34,197,94,0.14); color: rgb(22 163 74); } .ravpr-section-row.is-done .ravpr-ring, .ravpr-check-row.is-done .ravpr-ring { display: none; } .ravpr-section-row.is-done .ravpr-check, .ravpr-check-row.is-done .ravpr-check { display: inline-flex; } .ravpr-reveal { display: flex; flex-direction: column; gap: 3px; opacity: 0; transform: translateY(4px); transition: opacity 260ms ease, transform 260ms ease; pointer-events: none; } .ravpr-reveal.is-visible { opacity: 1; transform: translateY(0); pointer-events: auto; } .ravpr-check-list, .ravpr-comment-list { display: flex; flex-direction: column; gap: 4px; min-height: 0; } .ravpr-comment-card { border: 1px solid var(--border); border-radius: 8px; background: var(--card, #fff); overflow: hidden; opacity: 0; transform: translateY(4px); transition: opacity 260ms ease, transform 260ms ease; } .ravpr-comment-card.is-visible { opacity: 1; transform: translateY(0); } .ravpr-comment-head { display: grid; grid-template-columns: 18px minmax(0,1fr) auto; align-items: center; gap: 7px; padding: 5px 7px; background: rgba(24,24,27,0.015); } .ravpr-avatar { width: 16px; height: 16px; border-radius: 999px; background: rgba(24,24,27,0.16); } .ravpr-author { width: 78px; height: 8px; border-radius: 999px; background: rgba(24,24,27,0.18); } .ravpr-comment-path { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 9.5px; color: var(--muted-foreground, #71717a); } .ravpr-comment-body { padding: 5px 7px 6px; font-size: 11px; line-height: 1.32; color: var(--foreground, #18181b); } .ravpr-comment-body code { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 10.5px; padding: 1px 4px; border-radius: 4px; background: rgba(24,24,27,0.06); } .ravpr-cursor { position: absolute; z-index: 40; pointer-events: none; transition: transform 600ms cubic-bezier(.45,.05,.2,1), opacity 200ms ease; transform: translate(-30px, 220px); opacity: 0; } .ravpr-cursor.is-visible { opacity: 1; } .ravpr-ripple { position: absolute; left: -6px; top: -6px; width: 28px; height: 28px; border-radius: 999px; border: 2px solid rgba(24,24,27,0.5); opacity: 0; } .ravpr-cursor.is-clicking .ravpr-ripple { animation: ravpr-ripple 460ms ease-out forwards; } @keyframes ravpr-ripple { 0% { transform: scale(0.4); opacity: 0.9; } 100% { transform: scale(1.4); opacity: 0; } } @keyframes ravpr-spin { to { transform: rotate(360deg); } }" + } + } + }, + "ship": { + "styles": { + "90cdcd2ecc": ".ravs-ship-root { position: absolute; inset: 0; } .ravs-ship-stack { position: absolute; inset: 0; display: grid; grid-template-columns: 232px minmax(0,1fr); gap: 14px; padding: 4px 2px; /* Why: cards size to their content rather than stretch to the parent's full height, so the two cards don't show empty space below their content. */ align-items: start; } /* Source Control mini-sidebar — ahead-count header, commit textarea + split Commit button, then a CHANGES section with file rows. The file rows are the surface the \"reading\" pulse animates over. */ .ravs-sc-card { display: flex; flex-direction: column; background: var(--card, #fff); border: 1px solid var(--border); border-radius: 10px; overflow: hidden; box-shadow: 0 1px 2px rgba(24,24,27,0.04); } /* Both card headers share the same fixed height so the SC card and PR dialog align across the top edge regardless of header content. */ .ravs-sc-header, .ravs-pr-head { height: 36px; box-sizing: border-box; } .ravs-sc-header { display: flex; align-items: center; justify-content: space-between; padding: 0 10px; border-bottom: 1px solid var(--border); } .ravs-sc-ahead { display: inline-flex; align-items: center; gap: 5px; font-size: 11px; font-weight: 500; color: var(--foreground, #18181b); } .ravs-sc-ahead svg { color: var(--muted-foreground, #71717a); } .ravs-sc-commit-area { display: flex; flex-direction: column; gap: 6px; padding: 8px 10px; } .ravs-sc-textarea { position: relative; border: 1px solid var(--border); border-radius: 6px; background: var(--editor-surface, var(--card)); padding: 6px 26px 6px 8px; min-height: 56px; font-size: 12px; line-height: 1.45; color: var(--foreground, #18181b); white-space: pre-wrap; word-break: break-word; overflow: hidden; } .ravs-sc-textarea .ravs-placeholder { color: rgba(113,113,122,0.7); } .ravs-sc-sparkle { position: absolute; right: 6px; top: 6px; width: 20px; height: 20px; display: inline-flex; align-items: center; justify-content: center; border-radius: 4px; color: var(--muted-foreground, #71717a); background: transparent; transition: color 160ms ease, background 160ms ease; } .ravs-sc-sparkle.is-scanning { color: rgb(109 40 217); background: color-mix(in srgb, rgb(139 92 246) 18%, transparent); } .ravs-sc-split { display: flex; align-items: stretch; } /* Why: Commit + Create PR are surrounding chrome — the violet AI affordances are the focal points. Render them as quiet secondary buttons so they don't compete with the sparkle/scan signals. */ .ravs-sc-split .ravs-primary { flex: 1; display: inline-flex; align-items: center; justify-content: center; gap: 5px; padding: 5px 10px; background: var(--secondary, #f5f5f5); color: var(--secondary-foreground, #171717); font-size: 11px; font-weight: 500; border-radius: 6px 0 0 6px; border: 1px solid var(--border); transition: background 240ms ease, border-color 240ms ease, color 240ms ease; } .ravs-sc-split .ravs-chev { display: inline-flex; align-items: center; justify-content: center; width: 22px; background: var(--secondary, #f5f5f5); color: var(--muted-foreground, #71717a); border-radius: 0 6px 6px 0; border: 1px solid var(--border); border-left: 1px solid var(--border); transition: background 240ms ease, border-color 240ms ease, color 240ms ease; } /* Why: when AI has filled the commit message, tint the Commit button green to signal \"ready to commit\". Uses the same success-green family as the PR flash so the two beats rhyme. Mix is intentionally strong (~28%) — at 14% it disappeared next to the violet sparkle and PR flash, so users only saw the PR change color. */ .ravs-sc-split.is-ready .ravs-primary, .ravs-sc-split.is-ready .ravs-chev { background: color-mix(in srgb, rgb(34 197 94) 28%, var(--secondary, #f5f5f5)); border-color: rgb(34 197 94); color: rgb(21 128 61); transition: background 220ms ease, border-color 220ms ease, color 220ms ease; } .ravs-sc-split.is-ready .ravs-chev { border-left-color: rgba(34, 197, 94, 0.55); } .ravs-sc-changes-header { display: flex; align-items: center; justify-content: space-between; padding: 8px 10px 4px; font-size: 10px; font-weight: 600; text-transform: uppercase; letter-spacing: 0.05em; color: var(--muted-foreground, #71717a); } .ravs-sc-changes-count { color: var(--foreground, #18181b); font-weight: 600; margin-left: 2px; } .ravs-sc-view-all { font-size: 10px; font-weight: 500; text-transform: none; letter-spacing: 0; color: var(--muted-foreground, #71717a); } .ravs-sc-files { display: flex; flex-direction: column; padding: 2px 6px 8px; flex: 1; min-height: 0; overflow: hidden; } .ravs-sc-file { display: grid; grid-template-columns: 14px minmax(0,1fr) 12px; align-items: center; gap: 6px; padding: 3px 6px; border-radius: 4px; font-size: 11px; line-height: 1.35; color: var(--foreground, #18181b); position: relative; transition: background 220ms ease; } .ravs-sc-ficon { color: rgb(180 83 9); display: inline-flex; } .ravs-sc-fname { min-width: 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } .ravs-sc-fmark { font-family: ui-monospace, SFMono-Regular, Menlo, monospace; font-size: 10px; text-align: right; color: rgb(180 83 9); } .ravs-sc-file.is-reading { background: color-mix(in srgb, rgb(139 92 246) 14%, transparent); box-shadow: inset 0 0 0 1px color-mix(in srgb, rgb(139 92 246) 28%, transparent); } /* PR dialog — matches the .ravs-sc-card chrome (same border, radius, elevation) so the two cards read as one design language. */ .ravs-pr-dialog { background: var(--card, #fff); border: 1px solid var(--border); border-radius: 10px; box-shadow: 0 1px 2px rgba(24,24,27,0.04); display: flex; flex-direction: column; min-width: 0; overflow: hidden; } .ravs-pr-head { display: flex; align-items: center; justify-content: space-between; gap: 6px; padding: 0 10px; border-bottom: 1px solid var(--border); } .ravs-pr-title-text { font-size: 11px; font-weight: 500; color: var(--foreground, #18181b); } /* Icon-only AI-assist chip — mirrors .ravs-sc-sparkle so the affordance reads identically across both cards. */ .ravs-pr-gen-btn { display: inline-flex; align-items: center; justify-content: center; width: 22px; height: 22px; padding: 0; border-radius: 4px; color: var(--muted-foreground, #71717a); background: transparent; border: 0; cursor: pointer; transition: color 160ms ease, background 160ms ease; } .ravs-pr-gen-btn:hover { background: rgba(24,24,27,0.06); color: var(--foreground, #18181b); } .ravs-pr-gen-btn.is-scanning { color: rgb(109 40 217); background: color-mix(in srgb, rgb(139 92 246) 18%, transparent); } .ravs-pr-body { display: flex; flex-direction: column; gap: 8px; padding: 10px; } .ravs-pr-field { display: flex; flex-direction: column; gap: 4px; } .ravs-pr-field-label { font-size: 10px; font-weight: 600; text-transform: uppercase; letter-spacing: 0.05em; color: var(--muted-foreground, #71717a); } .ravs-pr-base { display: inline-flex; align-items: center; gap: 5px; padding: 4px 9px; border: 1px solid var(--border); border-radius: 6px; font-size: 11px; font-family: ui-monospace, SFMono-Regular, Menlo, monospace; color: var(--foreground, #18181b); background: var(--editor-surface, var(--card)); align-self: flex-start; } .ravs-pr-base svg { color: var(--muted-foreground, #71717a); } .ravs-pr-input { position: relative; padding: 6px 8px; border: 1px solid var(--border); border-radius: 6px; background: var(--editor-surface, var(--card)); min-height: 28px; font-size: 12px; line-height: 1.45; color: var(--foreground, #18181b); white-space: pre-wrap; word-break: break-word; overflow: hidden; } .ravs-pr-input.is-body { min-height: 50px; font-size: 11px; line-height: 1.4; } .ravs-pr-input .ravs-placeholder { color: rgba(113,113,122,0.7); } .ravs-pr-footer { display: flex; align-items: center; gap: 6px; justify-content: flex-end; margin-top: 2px; } .ravs-pr-btn { font-size: 11px; font-weight: 500; padding: 5px 10px; border-radius: 6px; line-height: 1; border: 1px solid transparent; outline: none; } .ravs-pr-btn:focus, .ravs-pr-btn:focus-visible { outline: none; } /* Cancel reads as a quiet ghost button so it doesn't compete with the affirmative Create PR action. */ .ravs-pr-btn.is-outline { background: transparent; color: var(--muted-foreground, #71717a); border-color: transparent; } .ravs-pr-btn.is-outline:hover { background: rgba(24,24,27,0.05); color: var(--foreground, #18181b); } /* Quiet secondary fill — see the .ravs-sc-split note above. The flash ring still uses success-green so the \"PR created\" beat reads. The Create-PR button is slightly larger than Cancel so the affirmative action remains the bigger target. */ .ravs-pr-btn.is-solid { background: var(--secondary, #f5f5f5); color: var(--secondary-foreground, #171717); border-color: var(--border); font-size: 12px; padding: 7px 14px; transition: background 220ms ease, border-color 220ms ease, color 220ms ease, box-shadow 220ms ease; } .ravs-pr-btn.is-solid.is-ready { background: color-mix(in srgb, rgb(34 197 94) 28%, var(--secondary, #f5f5f5)); border-color: rgb(34 197 94); color: rgb(21 128 61); } .ravs-pr-btn.is-solid.is-flash { box-shadow: 0 0 0 3px rgba(34, 197, 94, 0.30); } .ravs-cursor { position: absolute; z-index: 40; pointer-events: none; transition: transform 600ms cubic-bezier(.45,.05,.2,1), opacity 200ms ease; transform: translate(-30px, 220px); opacity: 0; } .ravs-cursor.is-visible { opacity: 1; } .ravs-cursor .ravs-ripple { position: absolute; left: -6px; top: -6px; width: 28px; height: 28px; border-radius: 999px; border: 2px solid rgba(24,24,27,0.5); opacity: 0; } .ravs-cursor.is-clicking .ravs-ripple { animation: ravs-ripple 460ms ease-out forwards; } @keyframes ravs-ripple { 0% { transform: scale(0.4); opacity: 0.9; } 100% { transform: scale(1.4); opacity: 0; } } .ravs-caret { display: inline-block; width: 1.5px; height: 1em; background: currentColor; vertical-align: -2px; margin-left: 1px; animation: ravs-caret-blink 1.05s steps(1) infinite; } @keyframes ravs-caret-blink { 0%, 50% { opacity: 1 } 51%, 100% { opacity: 0 } }" + } + } + } + }, + "notes": { + "diff": { + "rows": { + "f621c734f8": "Note · ligne" + } + } + } + }, + "agents": { + "orchestration": { + "OrchestrationPage": { + "30b509a467": "2 enfants", + "862605d066": "2 espaces de travail enfants", + "coordinatorName": "refonte du flux d'authentification", + "childPr1Name": "PR 1/2 : migration de users.sql", + "childPr2Name": "PR 2/2 : middleware withSession" + }, + "types": { + "coordinatorInitial": "Fractionnement de la refonte de l'authentification en 2 PR…", + "codexInitial": "Écriture de la migration de la table users…", + "claudeInitial": "Ébauche du middleware withSession…", + "codexBeat1": "Ajout de la colonne email_verified…", + "claudeBeat1": "Branchement du middleware withSession…", + "coordinatorBeat1": "PR 1/2 prête", + "coordinatorBeat2": "PR 2/2 prête" + }, + "steps": { + "name": "Orchestration", + "subtitle": "Orchestration", + "description": "Permettez aux agents de gérer et de coordonner les espaces de travail Orca pour exécuter des tâches plus importantes." + }, + "StatusesPage": { + "2f549fc0ba": "src/auth/session.test.ts", + "139e3d7458": "Mis à jour", + "7b26349cb2": "pnpm migrate latest", + "78f0318ac1": "Souhaite exécuter", + "79971d1539": "refonte du flux d'authentification" + }, + "UsageAccountsCard": { + "6986b36708": "Affichez les limites de débit et changez de compte directement.", + "d90d2e1f6d": "Suivez l'utilisation par session et par semaine.", + "8919321417": "Échec de la connexion Codex.", + "c7b90c140b": "Compte Codex ajouté.", + "4e71d72912": "Échec de la connexion Claude.", + "9ddeb558f9": "Compte Claude ajouté.", + "29d0653961": "Se connecter", + "945865332e": "Connexion en cours" + }, + "UsagePage": { + "64265cb295": "29 % utilisés sur 5 h", + "be5a165875": "Basculer vers", + "277a9c65a9": "Compte Codex", + "4dce5ca3aa": "Réinitialisation dans 4 j 3 h", + "05ce4ecdd3": "38 % utilisés", + "0470aaed99": "Hebdomadaire", + "f421abf962": "Session", + "5e45fb1238": "Mis à jour il y a 1 min", + "6a4b1d3c38": "Codex" + }, + "orchestration": { + "cards": { + "da6f1f97c9": "travaille" + } + } + } + }, + "ReviewAnimatedVisual": { + "8df4d52b68": "pr-view", + "8ab622e4d6": "notes" + }, + "FeatureWallBrowserAction": { + "5022c43a88": "Impossible d'ouvrir le navigateur", + "c9eb68b474": "Aucun groupe d'espaces de travail n'est encore disponible pour ce worktree.", + "c9728107c5": "Essayez", + "25dd101f15": "La configuration du navigateur demande votre attention", + "e02b11e6b0": "Configuration du navigateur prête", + "d6d15077df": "Commande de skill copiée et insérée ci-dessous pour vérification.", + "78e65f19d9": "Échec de la configuration du navigateur", + "b7345c18db": "Une erreur inattendue s'est produite.", + "5f97caf76b": "Installation…", + "c2df599513": "Installer la CLI et le skill" + }, + "ConnectIntegrationsList": { + "3dddb2d565": "connecté pour les tâches", + "33b650af52": "Connectez l'outil où votre équipe suit son travail. Orca démarre les espaces de travail avec le titre de l'issue, le lien et le contexte déjà rattachés.", + "5b3577a492": "connecté pour le statut de revue", + "list_end": ", et ", + "list_pair": " et ", + "list_mid": ", ", + "code_host_tasks_summary": "issues disponibles comme tâches · ajoutez Linear ou Jira si votre équipe y planifie son travail", + "code_host_tasks_caption": "Les issues de votre hébergeur de code servent aussi de tâches.", + "review_step_title": "Voyez le statut des PR pendant que les agents travaillent", + "review_step_description": "Connectez un fournisseur de revue pour qu'Orca affiche le statut des PR ou MR, les vérifications et les revues.", + "task_step_title": "Lancez des agents sur vos tâches sans quitter Orca" + }, + "connect": { + "integration": { + "step": { + "0f47ff17c6": "Modifier", + "5538eb6743": "Terminé", + "open_step": "Ouvert", + "close_step": "Fermer" + } + } + }, + "FullDiskAccessSetupPrompt": { + "bbb3f1e404": "Vérification", + "48d87edcd2": "Accordée", + "6db9a69f4e": "Recommandé", + "fa809e8ada": "Panneau Confidentialité et sécurité de macOS ouvert", + "bfa3402305": "Impossible de demander l'autorisation", + "c566bca278": "Accès complet au disque", + "0d6efe9cf4": "Recommandé sur macOS quand les projets ou les worktrees se trouvent dans des dossiers protégés.", + "dac08ec03e": "Ouverture...", + "6e3d62b816": "Ouvrir l'accès complet au disque" + } + }, + "tips": { + "CliFeatureTipVisual": { + "badb4fc342": ">", + "22e62f3bab": "Session Claude Code démarrée" + }, + "CliSkillSetupTerminal": { + "1953e90447": "Appuyez sur Entrée pour installer la skill d'orchestration Orca CLI destinée à vos agents.", + "43b60ec5c3": "Terminal d'installation de l'Orca CLI et de la skill d'orchestration", + "84e9576dac": "Configuration du skill", + "5c3aee22c0": "Copier la commande", + "5eca672aac": "Copier la commande d'installation de la skill", + "6ff813fc1d": "Impossible de copier la commande de la skill.", + "b8ad063571": "Commande d'installation de la skill copiée." + }, + "CmdJPaletteTipDialog": { + "c0bb9f869b": "Paramètres → Raccourcis", + "8241897205": "Réassignez le raccourci à tout moment dans" + }, + "FeatureTipActions": { + "eb04abece8": "Peut-être plus tard" + }, + "FeatureTipsModal": { + "c169298e4d": "Terminé", + "3c6c478462": "X a terminé, envoyez-lui la tâche de revue. »", + "298301b7a0": "worktree", + "864e2db28f": "« Quand l'agent dans", + "7fc6f02099": "et créez une pull request pour chacune. »", + "27c567a89c": "worktrees", + "55846c7f95": "« Divisez cette pull request en deux", + "4795ac2d4a": "Essayez de demander :", + "53905bd076": "Aperçu de développement : ouverture du terminal de configuration des skills.", + "1da82af45b": "L'Orca CLI requiert votre attention", + "ce13a742d0": "`orca` enregistré dans le PATH.", + "d1a86c7eb5": "Ouvrez les paramètres pour terminer la configuration du CLI." + }, + "CmdJPaletteFeatureTipVisual": { + "ab94e16d44": "Créer le worktree « {{value0}} »", + "d20ccf1e61": "terminé", + "379d776971": "saisie", + "0418f9becc": "ouvrir" + } + } + }, + "error": { + "boundaries": { + "RecoverableRenderErrorBoundary": { + "55001880db": "Réessayer", + "34a189ae0f": "Le reste de l'application fonctionne toujours. Réessayez ici, ou changez de vue puis revenez.", + "ab855c11f4": "Cette partie d'Orca a rencontré une erreur." + } + } + }, + "emulator": { + "pane": { + "EmulatorPane": { + "59b08fa031": "Aucun émulateur connecté" + }, + "emulator": { + "device": { + "frame": { + "9406c15775": "Écran de l'émulateur", + "8f25ffaf8a": "Écran de l'émulateur, clavier capturé. Appuyez sur Échap pour le libérer.", + "0022420df0": "téléphone" + } + }, + "pane": { + "toolbar": { + "06e10d7356": "Éteindre l'émulateur", + "e7a0d1897e": "Home", + "6bd8dff42a": "Pivoter", + "3d836b879c": "Choisir un émulateur", + "81b3571a07": "Se connecter", + "868c0f2938": "Traitement…" + } + }, + "screen": { + "stream": { + "content": { + "8b1a0d8694": "Aperçu de l'émulateur", + "36841af608": "Flux déconnecté", + "5f818f12ab": "Connexion à l'émulateur…", + "5ee64cd44e": "Écran de l'émulateur" + } + } + }, + "unavailable": { + "pane": { + "f630b9ca9f": "L'émulateur mobile nécessite un Mac avec Xcode et le runtime iOS Simulator. Sous Linux ou Windows, utilisez un appareil physique ou un hôte de build Mac distant.", + "b2c268a0b9": "L'émulateur mobile est disponible uniquement sur macOS" + } + } + }, + "use": { + "emulator": { + "frame": { + "stream": { + "f1c0179002": "Le flux ne produit aucune image." + } + } + }, + "mobile": { + "emulator": { + "agent": { + "setup": { + "state": { + "fdcca1ec75": "Enregistrement...", + "69fb2c2289": "Activé", + "c6705092ba": "Corriger PATH", + "7c1b6bdb1e": "Activer", + "51074ccb05": "Échec du chargement de l'état de la CLI.", + "35dea1ae12": "Le contrôle des agents est prêt.", + "9dff3a6338": "La skill est installée. Activez l'Orca CLI pour terminer la configuration.", + "15986a1080": "L'Orca CLI est prêt. Installez la skill pour terminer la configuration.", + "4c26913def": "Toujours pas configuré. Terminez les deux étapes pour activer le contrôle des agents.", + "c94ff11e91": "Impossible de revérifier l'état de la configuration.", + "2b519eed94": "Orca CLI enregistré dans PATH." + } + } + }, + "tab": { + "intro": { + "actions": { + "68a5dc6604": "Impossible de masquer l'émulateur mobile." + } + } + } + } + } + }, + "MobileEmulatorAgentSetupGuide": { + "2fda9ff015": "Configurer le contrôle des agents", + "0ac0fef514": "Le contrôle des agents est prêt.", + "2bdfff8763": "Contrôle des agents (facultatif).", + "72736b051f": "Configurez Orca CLI + skill quand vous voulez que des agents pilotent ce simulateur.", + "d10ae98046": "Terminé", + "3756cbeca7": "Pas maintenant", + "6d950431d2": "Masquer", + "ebceac65a4": "Configurer", + "3f003507f4": "Ouvrir la configuration complète dans les paramètres" + }, + "MobileEmulatorAgentSetupGuideSteps": { + "9b49d892e3": "Activer Orca CLI", + "3d8dc52c93": "Enregistre la commande orca de contrôle de l'émulateur dans les shells des agents.", + "21f5687c07": "Compétence CLI Orca", + "64fb057667": "Apprend aux agents les commandes orca de l'émulateur pour ce worktree.", + "5c59ea96ca": "Configuration de la skill Orca CLI de l'émulateur mobile", + "bff5341ac3": "Terminal d'installation de la skill Orca CLI de l'émulateur mobile", + "3d34423e88": "Enregistrement de la CLI Orca", + "3be27641c9": "afin que les commandes d'émulateur puissent s'exécuter depuis les shells des agents.", + "3941719a56": "Vérification de la CLI Orca avant d'ouvrir la configuration du skill." + }, + "MobileEmulatorTabIntroCallout": { + "1924982130": "Ignorer", + "5789936d9a": "Prévisualisez les simulateurs iOS pendant que des agents pilotent l'écran.", + "8014b4b80b": "Conserver", + "6e051a40b7": "Masquer" + }, + "mobile": { + "emulator": { + "hidden": { + "toast": { + "e8f098a870": "Émulateur mobile masqué", + "c46c979c1d": "Réactivez l'émulateur mobile à tout moment dans", + "600f9a745a": "Paramètres › Émulateur mobile" + } + } + } + }, + "useEmulatorScreenKeyboard": { + "pasteTooLarge": "Le collage est trop volumineux pour la saisie clavier de l'émulateur.", + "unsupportedPasteText": "Le collage via le clavier de l'émulateur prend uniquement en charge le texte de clavier US.", + "pasteTargetUnavailable": "Le collage via le clavier de l'émulateur a échoué car l'appareil n'est pas prêt." + } + } + }, + "editor": { + "ChangesModeView": { + "ef25ae2d09": "Aucune modification non commitée.", + "052c184f24": "Le diff texte est indisponible pour ce fichier.", + "7dffb0f563": "Fichier binaire", + "54e0035b15": "Chargement du diff..." + }, + "CodeBlockCopyButton": { + "28921f5bf9": "Copied", + "1f9f4def45": "Copier le code" + }, + "CombinedDiffFileTree": { + "f984289373": "Aucun fichier ne correspond aux filtres actuels.", + "eafe1aeb53": "Réinitialiser les filtres", + "be119cb9d1": "Fichiers consultés", + "c00020f081": "Extensions de fichiers", + "cd0e0ed79e": "Filtrer les fichiers du diff", + "4cc7b83ffe": "Filtrer les fichiers...", + "21783df79f": "Réduire l'arborescence des fichiers", + "481e63ca52": "Fichiers", + "resizeFileTree": "Redimensionner l'arborescence des fichiers", + "d5ac717d65": "non committé", + "39b6b9e4e4": "Committé sur la branche" + }, + "CombinedDiffViewer": { + "35cc27aeb2": "dans le contrôle de code source", + "e3b9a6ce02": "de plus", + "1da745c551": "Envoyé", + "84898c548d": "Effacer", + "88b70d0ef5": "Copier", + "bb84b4c374": "Notes IA", + "948a5fd6c8": "Effacer les notes", + "0f806a2ab1": "Annuler", + "80a286d8f5": "de ce worktree ?", + "7e7ca60816": "fichiers modifiés", + "b6c3b84476": "Afficher l'arborescence des fichiers", + "39f8007549": "Examiner les conflits", + "39e73e7181": "ont été exclus de cette vue de diff.", + "689b99f8ad": "conflit non résolu", + "820ec01f24": "Les fichiers en conflit sont examinés séparément", + "fd8892b120": "Aucune modification à afficher", + "eb5f40e49c": "Cette vue de diff exclut les conflits non résolus, car le pipeline de diff bidirectionnel habituel n'est pas conçu pour gérer les conflits.", + "45cf23b418": "Échec de l'effacement des notes.", + "0fb870a0fe": "notes", + "8ab3248fd8": "note", + "ec5053c7f5": "Côte à côte", + "f786fd54e1": "En ligne", + "ea08dae15b": "Tout réduire", + "19c45cfdc0": "Tout développer", + "982d14bfa5": "Ouvrir toutes les modifications", + "3d909843bb": "Ouvrir le diff de branche", + "8368d256ec": "combined-branch", + "8f68ad9ca9": "Afficher {{value0}} IA {{value1}}", + "724a13568d": " dans {{value0}}", + "6094135eec": " vs {{value0}}", + "a4420ca1f7": "Retour à la ligne activé", + "dde325ddfe": "Retour à la ligne désactivé" + }, + "ConflictComponents": { + "f338288514": "Chargement du contenu des conflits...", + "90d576adb2": "Actualiser", + "a1ce36f77d": "Instantané capturé le", + "4be41eaafc": "conflit non résolu", + "c8ca989aea": "Afficher l'arborescence des fichiers", + "58ad5ad431": "Ignorer", + "28e7db4a90": "Contrôle de code source", + "31931dec46": "Cet instantané de revue ne contient plus aucun conflit non résolu actif.", + "992145ff5a": "Tous les conflits sont résolus", + "d5edd81755": "Renommé depuis", + "6e459867ad": "État de continuité valable pour la session. Git ne signale plus ce fichier comme non fusionné.", + "9c2901ef8a": "Conflit suivant", + "41d9af2e7a": "Conflit précédent", + "55d61a0ccd": "conflit ·", + "da539359b6": "Aucun fichier de l'arbre de travail n'est modifiable pour ce conflit." + }, + "ConflictReviewFileTree": { + "3449521a8c": "Aucun conflit dans cet instantané.", + "a54551c5a6": "Réduire l'arborescence des fichiers", + "99496bab6e": "Fichiers", + "496e28a932": "Disparu", + "8528a5eaf5": "Résolu", + "69d4e210bb": "Non résolu" + }, + "CsvViewer": { + "eedd0d37a7": "colonnes", + "ac31d2cd60": "lignes", + "a233d55b77": "Fichier vide" + }, + "DiffNotesSendMenu": { + "f1aa04b5cf": "Ce fichier", + "8b87612461": "Toutes les notes non envoyées" + }, + "DiffSectionBody": { + "35d6afb5be": "Fichier binaire modifié", + "cef4cf0ff5": "Réessayer", + "f5cf81cec2": "Chargement du diff...", + "72f71f52eb": "Le diff texte est indisponible pour ce fichier.", + "7ce8436458": "Le diff texte est indisponible pour ce fichier dans la comparaison de branches.", + "bdbf02d5df": "binaire", + "b5675b0694": "Enregistrer", + "593f2193f6": "Ce brouillon dépasse la limite d'affichage sécurisée, mais il peut toujours être enregistré." + }, + "DiffSectionHeader": { + "8915726e93": "Copier le chemin" + }, + "EditorContent": { + "56dba34e1a": "(modifier en mode source)", + "e4b074749d": "Front Matter", + "9640d1d3db": "Aperçu de la version modifiée de ce diff. Passez en mode source pour examiner les modifications.", + "78541e254e": "Fichier binaire modifié", + "c88c73a0d3": "Chargement du diff...", + "b9de81ba52": "Fichier binaire — impossible d'afficher", + "b2735221f5": "Chargement...", + "8608ce4cb1": "L'aperçu Markdown est indisponible pour les fichiers binaires.", + "37a0e81fa6": "Chargement de l'aperçu...", + "8b1a605bae": "Ce fichier est en état de conflit, mais aucun fichier de l'arbre de travail n'est disponible à l'édition.", + "2a512bb46a": "Réessayer", + "39f018b052": "Impossible de charger le fichier", + "8a0898ae4c": "Le diff texte est indisponible pour ce fichier.", + "3c6e71df22": "Le diff texte est indisponible pour ce fichier dans la comparaison de branches.", + "d07e4b8553": "branche", + "d16e037f40": "enrichi", + "6c4f1a8d2e": "Les détails de la vérification sont indisponibles." + }, + "EditorPanelHeader": { + "fb8331694e": "Ouvrir l'aperçu sur le côté", + "4157f3cbf3": "Ouvrir l'aperçu Markdown", + "269ce4842b": "Copier le chemin relatif", + "7c08a1f990": "Copier le chemin", + "84cdc0794b": "Renommer", + "1bb1e226ec": "Renommer le fichier {{value0}}", + "5447c4f68f": "Table des matières", + "146cb5473c": "La table des matières est disponible en mode enrichi ou aperçu", + "e836faacfa": "Passer au diff côte à côte", + "94756f08ba": "Passer au diff en ligne", + "c98ce191da": "Ce diff n'a aucun fichier du côté modifié à ouvrir", + "9b80bbe1de": "Ouvrir un onglet de fichier", + "f0fd4174b5": "Ouvrez un onglet de fichier pour utiliser l'édition Markdown enrichie", + "a10d9b8337": "Ouvrir le fichier", + "2076ecfc9c": "Modification précédente", + "631dab0df3": "Modification suivante" + }, + "EditorPanelMarkdownActionsMenu": { + "3e0ce48c24": "Exporter en PDF", + "561251019a": "Plus d'actions", + "8c8b7f5ff5": "Afficher le front matter", + "10c39d58c1": "Masquer le front matter", + "1eef809708": "Retour à la ligne" + }, + "EditorPanelShell": { + "e2c4dec350": "Chargement de l'éditeur..." + }, + "EditorViewToggle": { + "b3410cd5e0": "Notebook", + "e408aa9cd5": "Tableau", + "167f45888c": "Modifications non validées", + "4837f3f578": "Modifications", + "ac3bb87913": "Édition", + "0d193dc03c": "Aperçu", + "aff15f94f5": "Éditeur enrichi", + "4d6ccb7ba6": "Source" + }, + "ImageDiffViewer": { + "a651be62b0": "Modifié", + "57aac3979a": "Original", + "fb0ae4f3c0": "Aucun aperçu" + }, + "ImageViewer": { + "3c9217f5a6": "Zoom avant", + "6c89c73d9f": "Réinitialiser le zoom", + "be27304574": "Zoom arrière", + "77bfc9b35a": "Ouvrir l'image dans une fenêtre indépendante", + "3ef9551ba2": "Chargement de l'aperçu...", + "d9d2944855": "Échec du chargement de l'aperçu du fichier" + }, + "ImageViewerPopup": { + "0ef78475e7": "Appuyez sur Échap pour fermer", + "535f4e2b56": "Fermer", + "9e27b2ecaf": "Aperçu de l'image en taille réelle" + }, + "IpynbViewer": { + "859bf9fc21": "Exécuter la cellule", + "7f0d7077c6": "Annuler", + "10ed04a685": "Les cellules du notebook exécutent du Python en local sur cette machine depuis le dossier du notebook. N'exécutez que des cellules issues de fichiers de confiance.", + "9e06ae5d36": "Exécuter le code du notebook ?", + "d6f37a640b": "Notebook vide", + "8c3b21369a": "nbformat", + "329764e9fc": "BÊTA", + "15ec40a735": "Enregistrer le notebook", + "07e7d96612": "cellules", + "c1601b23b2": "Impossible d'afficher le notebook", + "66a3f7d330": "Sortie HTML du notebook", + "781abd6926": "Supprimer la cellule", + "b42f6a9547": "Insérer une cellule Markdown ci-dessous", + "ffc1ac2699": "Insérer une cellule Markdown ci-dessus", + "b4208cad7e": "Insérer une cellule de code ci-dessous", + "53b839b8a0": "Insérer une cellule de code ci-dessus", + "27e064e2db": "Déplacer la cellule vers le bas", + "fd8ac707bc": "Déplacer la cellule vers le haut", + "3e4cbf15ea": "Brut", + "1833dbbc43": "Markdown", + "7005960d73": "Code", + "59b6cd874b": "code", + "ba149053d5": "markdown" + }, + "MarkdownPreview": { + "e4683f70c4": "Annuler", + "d737791433": "Ajouter une note pour l'IA", + "b1bfc04034": "Texte sélectionné", + "f37b98999e": "Cette note", + "2b2b31382c": "Front Matter", + "bb629de58a": "Copier les notes pour l'agent", + "322afab6ff": "Notes de revue", + "0f9969a159": "Aller à la première note de revue", + "12052c639c": "Fermer la recherche", + "b42c41bd0d": "Correspondance suivante", + "1febd97f5c": "Correspondance précédente", + "ec77985138": "Rechercher dans l'aperçu Markdown", + "517aea303b": "Rechercher dans l'aperçu", + "f961e94057": "Copier la note pour l'agent", + "94b520a96a": "Note copiée", + "13f94d760c": "Ajouter une note", + "ddf087d12e": "Toutes les notes non envoyées", + "d652c87c91": "Enregistrement…", + "c5dc92cfe3": "Aucun résultat", + "6c043947ae": "Fichier introuvable : {{value0}}", + "759463a221": "Impossible d'ouvrir le répertoire : {{value0}}" + }, + "MarkdownTableOfContentsPanel": { + "de3928b6e4": "Aucun titre", + "bbe8369097": "Fermer la table des matières", + "4680a4b808": "Réduire jusqu'à H{{value0}}", + "a5daadd68b": "Tout développer", + "111e66b85d": "Réduire au niveau de titre {{value0}}", + "f3de856175": "Développer tous les niveaux de titre", + "0dc7b2f05a": "Réduire par niveau", + "06357eea60": "Table des matières", + "27d0a9c49a": "Table des matières", + "65b036a6c8": "Développer {{value0}}", + "97ad46f11f": "Réduire {{value0}}", + "8f4d2c1a9b": "Redimensionner la table des matières" + }, + "MarkdownTemplatePicker": { + "22cd94426f": "untitled.md", + "6e2e6c04ad": "Markdown vierge", + "df667919ca": "Aucun modèle correspondant.", + "22fd4890ad": "Rechercher des modèles...", + "7b458e0b7f": "Choisissez un modèle Markdown.", + "1829437fce": "Nouveau Markdown" + }, + "MermaidBlock": { + "dcc132e691": "Erreur de diagramme :" + }, + "MonacoEditor": { + "68cb83f4a7": "Ajouter une note sur le texte sélectionné", + "fd68ae03b3": "Rechercher dans les fichiers", + "largePasteTooLarge": "Le collage est trop volumineux." + }, + "MonacoGutterContextMenu": { + "7b57b1b468": "Copier l'URL du remote", + "2e0b1cdc05": "Copier le chemin relatif jusqu'à la ligne", + "4eaa991bde": "Copier le chemin jusqu'à la ligne" + }, + "NotesSendMenu": { + "44dc5e60a6": "Envoyer les notes", + "433928cd9f": "Envoyer {{value0}} à un agent" + }, + "PdfFind": { + "cd65b1d6b0": "Fermer", + "eeba2547a1": "Correspondance suivante", + "30de726ad0": "Correspondance précédente", + "2fc3ba0ea8": "Rechercher dans la page...", + "d080ab37d6": "Aucun résultat", + "db56fcd6d2": "{{value0}} sur {{value1}}" + }, + "PdfViewer": { + "3e98d500d2": "Aperçu PDF", + "069ff59932": "Rechercher dans le PDF ({{value0}})", + "2b6eb1ccd6": "Zoom avant", + "c0119616d6": "Ajuster à la largeur", + "fa5d096b00": "Zoom arrière" + }, + "ReviewNotesSendMenuContent": { + "a49800405b": "Nouvel agent", + "e84705f223": "Session d'agent active", + "03378aea75": "Envoyer les notes à", + "f5096c6e4e": "Impossible d'envoyer les notes à l'agent actif.", + "bb9c69a0c9": "Notes envoyées à l'agent actif.", + "50f7e753ea": "Envoi des notes à l'agent actif..." + }, + "RichMarkdownAnnotationOverlay": { + "069b5677b8": "Texte sélectionné", + "6f2f3a6001": "Ajouter une note de revue" + }, + "RichMarkdownCodeBlock": { + "232d9ed853": "Copied", + "c72beafc0f": "Copier le code", + "74eab1d9b2": "YAML", + "5ef5605cb7": "XML", + "88d777bc07": "TypeScript", + "9e384d48dc": "Swift", + "3009f722b9": "SQL", + "d01f55be57": "Shell", + "5af8251002": "SCSS", + "e72e6b03f4": "Rust", + "96182a2f64": "Ruby", + "2391f9cda9": "Python", + "89d6cc14fb": "Mermaid", + "983b9576b4": "Markdown", + "bcb236e2d8": "Kotlin", + "78eba32de4": "JSON", + "a209c57063": "JavaScript", + "36536ad539": "Java", + "8c4a3fa02d": "HTML", + "706fd85738": "GraphQL", + "edfcc64182": "Go", + "bf6ee5caaa": "Diff", + "026653f21f": "CSS", + "4daed43ae3": "C++", + "4227cf50fe": "Bash", + "13822cdfda": "Texte brut" + }, + "RichMarkdownDocLinkMenu": { + "e17b987473": "↑↓ naviguer  ↵ sélectionner  esc fermer", + "90c5f0e1e4": "sur", + "2aaf7d9678": "Affichage de", + "63ced7cb9b": "Aucun document trouvé", + "0e8489bc11": "Liens de documents Markdown", + "142a7d51cd": "document" + }, + "RichMarkdownErrorBoundary": { + "aad0998127": "Réessayer", + "4a5de9f2f0": "Passez en mode source, ou cliquez sur Réessayer pour recharger la vue enrichie.", + "dfdf1cacd4": "L'éditeur Markdown enrichi a rencontré une erreur inattendue et a été réinitialisé pour que le reste d'Orca reste réactif." + }, + "RichMarkdownLinkBubble": { + "1c99b726e0": "Supprimer le lien", + "cdfe166f6f": "Modifier le lien", + "bfc813e909": "Ouvrir le lien", + "7b0b945fdc": "Collez ou saisissez un lien…", + "copyLink": "Copier le lien" + }, + "RichMarkdownReviewNoteLayer": { + "f3ef92952b": "Cette note", + "9cde7ad994": "Copier la note pour l'agent", + "117432e2c6": "Note copiée", + "3ababd949d": "Notes de revue" + }, + "RichMarkdownReviewRailActions": { + "636394af72": "Copier les notes pour l'agent", + "a807596997": "Notes copiées", + "8aaf2c4c69": "Afficher les notes de revue", + "af02dc2456": "Masquer les notes de revue" + }, + "RichMarkdownSearchBar": { + "de68b75bde": "Fermer la recherche", + "f7bcecbe26": "Correspondance suivante", + "32ae8d7d57": "Correspondance précédente", + "158c645829": "Rechercher dans l'éditeur Markdown enrichi", + "98b89276f3": "Rechercher dans l'éditeur enrichi", + "a86958d508": "Aucun résultat", + "e8c147435f": "Masquer le remplacement", + "9cdc38be33": "Afficher/masquer le remplacement", + "482b637099": "Respecter la casse", + "68d090241d": "Mot entier uniquement", + "fd97c7e585": "Remplacer", + "44682b4159": "Remplacer dans l'éditeur Markdown enrichi", + "c2884f5e95": "Tout remplacer", + "preservedRichContentReadOnly": "Le contenu enrichi préservé est en lecture seule en mode enrichi." + }, + "RichMarkdownSlashMenu": { + "82c6816ff8": "Aucun bloc trouvé", + "dbdd2ad15f": "Rechercher des blocs...", + "550189b06c": "Rechercher des blocs", + "2e0400b958": "Commandes slash", + "e2e12b0e98": "composant" + }, + "RichMarkdownToolbar": { + "e935c6b61e": "Image", + "6d52624712": "Lien", + "f6a51cb9af": "Citation", + "f97031be09": "Liste de tâches", + "31630ed66e": "Liste numérotée", + "5d1539e5a9": "Liste à puces", + "0bea19a988": "Barré", + "6b4ccf9493": "Italique", + "4f9e789fe0": "Gras", + "cf5817d827": "Titre 3", + "d34a2021c8": "Titre 2", + "abb5100a3d": "Titre 1", + "b462641ed2": "Texte courant", + "91a843fb43": "Plus de blocs", + "2cd9e0bbb3": "Titres", + "b05e14620d": "Titre 4", + "6bbf827ef5": "Titre 5", + "d1bbf9a835": "Section repliable" + }, + "UntitledFileRenameDialog": { + "a7dd27b0bc": "Enregistrer", + "949711deb4": "Annuler", + "725868c75d": "Parcourir les dossiers", + "5e7f0d8a80": "Sélecteur de dossiers indisponible pour les fichiers distants", + "30099dca46": "Dossier", + "2d7d39dc63": ".md", + "c8ac7868e6": "nom de fichier", + "b6ed807cc6": "Nom", + "e365f3c638": "Nommez votre fichier Markdown et choisissez un dossier.", + "674b046582": "Enregistrer sous" + }, + "export": { + "active": { + "markdown": { + "51c4244904": "Exporté vers {{value0}}", + "d4a901e0ad": "Export du PDF...", + "eda2cea3ad": "Échec de l'export du PDF" + } + } + }, + "markdown": { + "rich": { + "mode": { + "7a8ce7c7da": "Modifiable uniquement en mode code, car ce fichier contient des notes de bas de page.", + "2fd2b44073": "Modifiable uniquement en mode code, car ce fichier contient des liens par référence.", + "57128b73e1": "Modifiable uniquement en mode code, car ce fichier contient du HTML, du JSX ou du MDX." + } + } + }, + "rich": { + "markdown": { + "editor": { + "click": { + "routing": { + "2d5fb9335d": "Fichier introuvable : {{value0}}" + } + } + }, + "slash": { + "commands": { + "07e1b32396": "Insérer un emoji Unicode simple.", + "8a30cbaeca": "Émoji", + "3324eb391a": "Insérer une image depuis votre ordinateur.", + "572be8e524": "Image", + "ae7d0f3f37": "Insérer une formule LaTeX en bloc.", + "6993a38ad1": "Bloc mathématique", + "565907cf7a": "Insérer une formule LaTeX en ligne.", + "2bf5544faf": "Maths en ligne", + "0ed9a7b38c": "Insérer un bloc de code Mermaid.", + "e516d3f6e3": "Diagramme Mermaid", + "67faab829b": "Insérer un tableau Markdown 3x3.", + "19ea597868": "Tableau", + "fae45ef4d3": "Insérer une ligne horizontale.", + "ae8377cf6b": "Séparateur", + "89e327e054": "Insérer un bloc de code délimité.", + "624b50cf25": "Bloc de code", + "972ef9aeea": "Créer une section de texte repliable.", + "f82c78a2ee": "Texte repliable", + "9a7fe896dc": "Commencer un paragraphe normal.", + "58abdb9d41": "Paragraphe", + "d766f44867": "Créer une liste de cases à cocher.", + "d0d2cdfbdb": "Liste de cases à cocher", + "c9b9e826b8": "Créer une liste à puces.", + "56ff3237e7": "Liste à puces", + "8e00aba296": "Créer une liste ordonnée.", + "ed4cf0ebce": "Liste numérotée", + "6a3def14de": "Insérer un bloc de citation.", + "c4c775778b": "Citation", + "4920740259": "Petit titre de section.", + "30566ee962": "Titre 3", + "45cf7ceb3f": "Titre de section moyen.", + "c209a116b7": "Titre 2", + "3294a2c0cc": "Créer une section repliable avec un grand titre de résumé.", + "41482b15ce": "Titre repliable H1", + "570611864e": "Grand titre de section.", + "e66e7f04c6": "Titre 1", + "5f9a0ed7c4": "Titre 4", + "01a71dbbdd": "Titre de section imbriqué.", + "8440fa4acf": "Titre 5", + "b287b93c66": "Titre de section profond.", + "7a2c1f9b04": "Titre repliable H2", + "b3e5d8a1c6": "Créer une section repliable avec un titre de résumé moyen.", + "2f9d6b4e10": "Titre repliable H3", + "8c1a3e7d52": "Créer une section repliable avec un petit titre de résumé.", + "5e0b9c2a71": "Titre repliable H4", + "d4f16a8b39": "Créer une section repliable avec un titre de résumé imbriqué.", + "21d8c463e5": "Titre repliable H5", + "dc239b41ad": "Créer une section repliable avec un titre de résumé profond." + } + } + } + }, + "useContextualCopySetup": { + "059bfb0d94": "Contexte copié" + }, + "useLocalImagePick": { + "175cb8b8ce": "Échec de l'insertion de l'image.", + "91d835dc88": "Chemin du worktree indisponible." + }, + "useRichMarkdownReviewData": { + "f9d2acd6b0": "Toutes les notes non envoyées" + }, + "LargeDiffFallback": { + "a3c74f8a21": "nombre de lignes supérieur à la limite d'affichage sécurisée", + "fd92fbde46": "nombre de caractères supérieur à la limite d'affichage sécurisée", + "7d424bb761": "Ce diff est trop volumineux pour être affiché sans risque.", + "28aa2cc90b": "Lignes d'origine", + "20857938dd": "Lignes modifiées", + "e5f0d2182e": "Caractères", + "877c25a02f": "Raison", + "5fca073b72": "Limites", + "f1d136a163": "lignes par côté", + "23433fcdea": "caractères cumulés", + "7944ed9fb8": "Non compté" + }, + "DiffViewer": { + "b5675b0694": "Enregistrer", + "593f2193f6": "Ce brouillon dépasse la limite d'affichage sécurisée, mais il peut toujours être enregistré." + }, + "CheckRunDetailsPanel": { + "8f2d0f5a91": "Réussi", + "4c8e1b2d73": "Échec", + "91a4c7e2b0": "Annulé", + "2f6d8a1c45": "Délai dépassé", + "7b3e9d4f12": "Ignoré", + "5a1c8e3d67": "Neutre", + "3d9f2b8e14": "En attente", + "b7f5e2c91a": "Actualiser", + "a54ae21c6f": "Statut :", + "fd46a70f1a": "Démarré", + "00e1c1658a": "Terminé", + "aa8494ae3c": "vérification #", + "2dd5ddabc4": "workflow #", + "1f2b980522": "Chargement des détails de la vérification…", + "d098e5529a": "Sortie", + "f2fe8a4e8f": "Annotations", + "cdbfda4dec": "Annotation", + "5e2a9c3f88": "Ouvrir le fichier à cette ligne", + "066fedd446": "Jobs en échec", + "49731703ea": "Jobs", + "ee07b33924": "unknown", + "07eccfa397": "Aucun détail disponible pour cette vérification.", + "a916648574": "Ouvrir les détails", + "834cb3f23d": "Corriger avec l'IA", + "c8f1a2d4e7": "Choisissez l'agent et modifiez la commande complète avant le lancement.", + "b3e7f9a1c2": "Contexte de correction de la vérification indisponible", + "d5a8c2f1b9": "Démarrer l'agent IA par défaut pour corriger cette vérification", + "e2b4d7c8a1": "Choisir un agent pour cette vérification", + "f1c9e3a6d4": "Choisir un agent pour corriger la vérification", + "actionRequired": "Action requise" + }, + "CheckRunJobs": { + "1c0a4d7e02": "réussi", + "2d3b8f1a55": "ignoré", + "3e6c9a2b71": "en attente", + "4f7d0c3e88": "étapes en échec", + "5a8e1d4f23": " · " + }, + "check": { + "run": { + "details": { + "fix": { + "with": { + "ai": { + "1a8c4e2b90": "Sélectionnez un espace de travail avant de lancer une action IA.", + "4f2d9a8c17": "Sélectionnez un dépôt avant de lancer une action IA.", + "7c3e1b5d42": "Ouvrez une pull request ou une MR avant de lancer une correction IA.", + "9b2f6d4a81": "Cette vérification n'est pas en échec.", + "2ef90c9819": "Un agent IA a été démarré pour cette vérification." + } + } + } + } + } + }, + "richMarkdownLargeTextPaste": { + "tooLarge": "Le collage est trop volumineux." + }, + "ExternalFileChangeBanner": { + "7c41e90d12": "Ce fichier a été modifié sur le disque alors que vous avez des modifications non enregistrées. L'enregistrement écrasera le contenu plus récent présent sur le disque.", + "3fa2b8d417": "Recharger depuis le disque", + "a95d02c644": "Garder mes modifications", + "5c02de9b31": "Rechargé depuis le disque", + "d1e830fa22": "Annuler", + "90b2ce7d43": "Comparer" + }, + "ExternalFileChangeCompareDialog": { + "4b8de20a11": "Fichier modifié sur le disque", + "90cc31e4d7": "Version du disque à gauche, vos modifications non enregistrées à droite.", + "8fe30ab254": "Lecture du fichier depuis le disque...", + "e2b1cd0393": "Impossible de lire le fichier depuis le disque : {{value0}}", + "b6cf20d514": "Le fichier sur le disque est binaire — aucune comparaison de texte disponible.", + "3fa2b8d417": "Recharger depuis le disque", + "a95d02c644": "Garder mes modifications", + "2c8f1e07b9": "Chargement de la comparaison..." + }, + "RichMarkdownEditor": { + "citationLinkAvailable": "{{value0}}, lien vers {{value1}}. Appuyez sur Entrée pour ouvrir, ou Tab pour les actions du lien.", + "citationFallbackLabel": "Citation", + "citationLinkUnavailable": "{{value0}}, lien de citation indisponible. {{value1}}", + "tabForCitationActions": "Tab pour les actions disponibles.", + "noCitationActions": "Aucune action de lien disponible." + }, + "richMarkdownHtmlSuperscriptLink": { + "availableAriaLabel": "{{value0}}, lien vers {{value1}}", + "unavailableAriaLabel": "{{value0}}, lien de citation indisponible" + }, + "richMarkdownLinkClipboard": { + "copiedLink": "Lien copié", + "copyLinkFailed": "Échec de la copie du lien" + }, + "richMarkdownSourceOwningCutFeedback": { + "selectLessContent": "Sélectionnez moins de contenu ou utilisez le mode code pour couper les citations HTML préservées." + }, + "editor": { + "save": { + "failure": { + "notice": { + "8c59ce5075": "Échec de l'enregistrement du fichier. Veuillez réessayer." + } + } + } + }, + "RichMarkdownTableControls": { + "deleteTable": "Supprimer le tableau", + "tableActions": "Actions du tableau", + "addColumn": "Ajouter une colonne", + "addRow": "Ajouter une ligne", + "rowActions": "Actions de ligne", + "columnActions": "Actions de colonne", + "insertRowAbove": "Insérer une ligne au-dessus", + "insertColumnLeft": "Insérer une colonne à gauche", + "insertRowBelow": "Insérer une ligne en dessous", + "insertColumnRight": "Insérer une colonne à droite", + "deleteRow": "Supprimer la ligne", + "deleteColumn": "Supprimer la colonne" + } + }, + "diff": { + "comments": { + "DiffCommentCard": { + "109a791e7b": "Enregistrer", + "bb0a55f856": "Enregistrement…", + "0203bed775": "Annuler", + "6978871a3d": "Ouvert", + "cce596969e": "Supprimer la note", + "cad3384faa": "Modifier la note", + "508ee678a5": "Ouvrir dans le navigateur" + }, + "DiffCommentPopover": { + "2b3ce6d394": "Annuler", + "e05063cfc1": "Ligne {{value0}}", + "c845170b3b": "Lignes {{value0}}-{{value1}}", + "commentTooLarge": "Le commentaire est trop volumineux pour être envoyé sans risque." + }, + "useDiffCommentDecorator": { + "995fa28b50": "Cette note" + } + } + }, + "dictation": { + "DictationController": { + "de136f1199": "Erreur vocale : {{value0}}", + "7afff43472": "La dictée est terminée, mais aucun champ de texte n'avait le focus.", + "55127a3706": "Échec de la dictée : {{value0}}", + "bb7f599ee7": "Ouvrir les paramètres", + "2d5b9fabf9": "Accès au micro refusé. Accordez l'autorisation dans les réglages système, puis redémarrez Orca.", + "5d2c3e7ae3": "Aucune parole détectée.", + "micFallback": "Micro sélectionné indisponible. Utilisation du micro par défaut du système.", + "micDisconnected": "Micro déconnecté. Dictée arrêtée." + }, + "DictationIndicator": { + "335e1bc6cb": "Arrêter la dictée", + "7f3660a7ba": "Démarrage du micro…", + "f082d0cb9d": "Traitement…", + "3de5a129e7": "Écoute en cours", + "4977162383": "Trop fort", + "25f2b7a6a5": "Vous parlez" + } + }, + "dashboard": { + "DashboardAgentChildDisclosure": { + "1b57ce9fa4": "{{value0}} {{value1}} enfant {{value2}}" + }, + "DashboardAgentRow": { + "912e136cd9": "Envoyer", + "a743da52ff": "Développer les détails", + "a41fb5376e": "Réduire les détails", + "5ae84475cc": "Ignorer", + "b06e13fcf7": "Rejeter l'agent", + "0272969e28": "Envoyer à cet agent", + "92a7017987": "envoi", + "019b74d93a": "éligible" + }, + "DashboardAgentRowMessage": { + "0a01046763": "interrompu", + "1ec01cef03": "Interrompu par l'utilisateur" + } + }, + "crash": { + "report": { + "CrashReportDialog": { + "b4951cd27c": "Envoyer le rapport", + "88fea8e84e": "Ne pas envoyer", + "50b00dc327": "Copier les détails", + "6d3ebe216a": "Texte de diagnostic", + "835037edc9": "· Orca", + "56a3dfa283": "Échec de l'envoi du rapport de plantage.", + "8e24fe4f75": "Rapport de plantage envoyé.", + "8b8473c544": "Rapport de plantage copié.", + "b175e90213": "Aucun rapport de plantage disponible.", + "765591798d": "Recherche de rapports de plantage...", + "b2e36f53a1": "Échec de l'envoi du rapport de plantage. Le ticket de diagnostic {{value0}} a été téléversé, mais pas associé.", + "ead6fc0510": "Aucun rapport de plantage automatique n'a été capturé. Vous pouvez toujours envoyer des détails et inclure les journaux de diagnostic récents lorsqu'ils sont disponibles.", + "b082f27490": "Joindre les journaux de diagnostic récents", + "e59f0b9427": "Envoie avec le rapport un lot de journaux anonymisés et de taille limitée." + }, + "submit": { + "notice": { + "unknownError": "La demande de rapport de plantage a échoué avant de renvoyer une raison.", + "ticketUploaded": "Le ticket de diagnostic {{value0}} a été téléversé, mais pas associé.", + "uncheckDiagnostics": "Décochez « Joindre les journaux de diagnostic récents » et réessayez, ou copiez les détails.", + "checkConnection": "Vérifiez votre connexion et réessayez, ou copiez les détails.", + "notSent": "Le rapport de plantage n'a pas été envoyé", + "copyDetails": "Copier les détails", + "sentWithoutDiagnostics": "Rapport de plantage envoyé sans journaux de diagnostic", + "diagnosticsReason": "Journaux de diagnostic non joints : {{value0}}" + } + }, + "copy": { + "copyFailed": "Impossible de copier les détails du rapport de plantage." + } + } + }, + "contextual": { + "tours": { + "ContextualTourControl": { + "186eecc34f": "Nommer automatiquement l'espace de travail d'après le premier message de l'agent", + "02e8373219": "Génère automatiquement un nouveau nom lorsque vous laissez cette zone de texte vide.", + "731c5573df": "Nom automatique d'après le premier message" + }, + "ContextualTourOverlaySurface": { + "4a9568f773": "Retour", + "4f86e2a10b": "Passer la visite guidée", + "d974f32a83": "Fermer la visite guidée", + "ffa4412b66": "suivant", + "complete": "Terminé" + }, + "ContextualTourProgressDots": { + "7734cb8ad3": "sur", + "dcd6e6b03e": "Étape {{value0}} sur {{value1}}" + }, + "contextual": { + "tour": { + "overlay": { + "measurement": { + "38b3155418": "Suivant", + "automations": { + "intro": { + "title": "Qu'est-ce qu'une automatisation ?", + "body": "Les automatisations exécutent le travail des agents de façon planifiée. Ajoutez une automatisation en cliquant sur ce bouton." + }, + "results": { + "title": "Trouver les résultats", + "body": "Les exécutions montrent quand les automatisations ont tourné, ce qui s'est passé et où inspecter leur sortie." + } + } + } + } + } + } + } + }, + "cmd": { + "j": { + "quick": { + "actions": { + "c884a6398e": "Créer une commande de terminal enregistrée.", + "a43ab56fc1": "Ajouter une commande rapide", + "54853d52a2": "Supprimer le worktree actuel.", + "9537b910fe": "Supprimer le worktree", + "0b1f25f796": "Démarrer un nouveau worktree.", + "52ac9da671": "Créer un worktree", + "f70812764a": "Ouvrir un onglet de terminal dans l'espace de travail actif.", + "34980395d4": "Nouvel onglet de terminal", + "f2a1b33f8d": "Créer un fichier markdown sans titre dans l'espace de travail actif.", + "25349b66fc": "Nouveau fichier markdown", + "784812ca24": "Ouvrir un onglet de navigateur dans l'espace de travail actif.", + "892bfa9339": "Nouvel onglet de navigateur", + "verbs": { + "newBrowser": "nouveau navigateur", + "newBrowserTab": "nouvel onglet de navigateur", + "openBrowser": "ouvrir le navigateur", + "browserTab": "onglet de navigateur", + "newMarkdown": "nouveau markdown", + "newMarkdownFile": "nouveau fichier markdown", + "newMark": "nouveau mark", + "newFile": "nouveau fichier", + "markdownFile": "fichier markdown", + "newTerminal": "nouveau terminal", + "newTerminalTab": "nouvel onglet de terminal", + "newShell": "nouveau shell", + "terminalTab": "onglet de terminal", + "createWorktree": "créer un worktree", + "addWorktree": "ajouter un worktree", + "newWorktree": "nouveau worktree", + "deleteWorktree": "supprimer le worktree", + "deleteCurrentWorktree": "supprimer le worktree actuel", + "removeWorktree": "retirer le worktree", + "trashWorktree": "mettre le worktree à la corbeille", + "addQuickCommand": "ajouter une commande rapide", + "newQuickCommand": "nouvelle commande rapide" + } + } + }, + "palette": { + "project": { + "results": { + "repoGroup": "Groupe de dépôts", + "project": "Projet" + } + } + }, + "pluginQuickActions": { + "description": "commande du plugin {{value0}}", + "keyword": "commande de plugin" + } + } + }, + "browser": { + "pane": { + "BrowserFind": { + "c9d5f63fdc": "Fermer", + "5c0c02ae76": "Correspondance suivante", + "ca7aebbd7f": "Correspondance précédente", + "636a69cd66": "Rechercher dans la page...", + "7baca7b1b8": "Aucun résultat", + "fc63f336aa": "{{value0}} sur {{value1}}" + }, + "BrowserImportHintButton": { + "05e675fe96": "Masquer l'astuce", + "77351d22f5": "Paramètres du navigateur", + "e0e125e074": "Depuis un fichier…", + "0c6d254eca": "Depuis {{value0}}", + "244266c122": "Importer…", + "e52a955e6f": "Vous retrouverez toujours cette option dans Paramètres > Navigateur.", + "4f5ffaa6a1": "Importer les données du navigateur", + "b24fef25be": "Importer", + "02e89014c5": "{{value0}} cookies importés depuis {{value1}}{{value2}}.", + "d40d584769": "{{value0}} cookies importés depuis un fichier." + }, + "BrowserMobileDriverOverlay": { + "a6914ee43f": "Reprendre", + "f4ecd61552": "Cet onglet est contrôlé depuis votre téléphone. Reprenez-le pour l'utiliser sur ordinateur.", + "d9768ec642": "Saisie du navigateur en pause", + "20539eca03": "Le mobile contrôle ce navigateur", + "7c31b0da94": "Impossible de reprendre cet onglet. Vérifiez la session sur le téléphone, puis réessayez." + }, + "BrowserPane": { + "1ded0d3168": "Copier la capture d'écran", + "f30d2d35a7": "Capture effectuée", + "fa6ea61de3": "Annuler", + "c2ef0359b9": "Copier le contenu", + "f2d0c22d67": "Supprimer l'annotation {{value0}}", + "11c5084aa2": "Effacer les annotations", + "734e4343ec": "Effacer les annotations du navigateur", + "95af781091": "Envoyer un retour à un agent", + "ac39b9366b": "Envoyer", + "a3508d7e6e": "{{value0}} annotation{{value1}} prête. Sélectionnez un autre élément ou copiez tous les retours.", + "f796c774a4": "Saisissez une URL ci-dessus pour commencer à naviguer.", + "366bf5d62c": "Nouvel onglet", + "1c78adc73d": "Ouvrir en externe", + "da68d35f7b": "Ouvrir la page en erreur dans le navigateur par défaut", + "93be92f8d1": "Copier l'adresse", + "3c085f638d": "Copier l'URL de la page en erreur", + "c6be71329e": "Actualiser", + "781d6459ad": "Réessayer", + "2fdca7df09": "Ignorer", + "8b6fab9ffa": "Enregistrer", + "0f41bf80c7": "Ouvrir dans le navigateur par défaut", + "ec75d0c412": "Ouvrir les devtools du navigateur", + "fc9be38f6f": "Annoter un élément de la page", + "fdfc7fe0ef": "Capturer un élément de la page", + "a8f37f70c3": "Inspecter la page", + "1b179ab561": "Copier l'URL de la page", + "f7ab83f7ed": "Ouvrir la page dans le navigateur par défaut", + "0e080d820e": "Recharger", + "a1f3c2e4b5": "Rechargement forcé", + "b7e4d9c1a2": "Arrêter", + "250a9b3e42": "Transférer", + "40edfa75cb": "Retour", + "efb0e8f7f3": "Copier l'adresse du lien", + "8ce4f6b12e": "Ouvrir le lien dans le navigateur par défaut", + "b5b87d6cbb": "Ouvrir le lien dans le navigateur Orca", + "87eb75f7d2": "Saisissez une URL http(s) ou localhost valide.", + "27d863542c": "Annotations du navigateur", + "e48569ac6d": "Impossible d'accéder à ce site.", + "bbe8f15e83": "Ce volet est rendu depuis le serveur runtime actif.", + "8b7e6d1f5a": "Les annotations du navigateur ne sont disponibles que dans les onglets locaux.", + "deb5293610": "Annotations du navigateur indisponibles dans un runtime distant", + "90d021f2ad": "Ajouter", + "0cb3bd6221": "Intention d'annotation", + "8f87e6c2e5": "Intention", + "532bac48c5": "Décrivez ce que l'agent doit modifier ici...", + "d2a7092e6e": "Commentaire d'annotation", + "b472c5fe03": "Ajouter une annotation au navigateur", + "b5ba6085de": "Question", + "143204e423": "Modifier", + "b71dc3d930": "Reconnecter", + "e7ca5a098c": "succès", + "d51ef37351": "Copier", + "6f4ab3592b": "Copied", + "b2856516e2": "Impossible de charger cette page", + "db325a7eeb": "Impossible d'accéder à {{value0}}", + "499b31b84e": "Tout copier", + "e72dfa268a": "annoter", + "168350ae6a": "Cliquez ou survolez un élément, puis appuyez sur C pour copier ou S pour capturer.", + "e852e20cea": "Copié — appuyez sur S pour capturer, ou sélectionnez un autre élément", + "a5dcd0fd1d": "confirmation", + "777b5bc4ec": "Cliquez sur un élément pour ajouter un retour pour l'agent.", + "b733a91bd9": "Ajouter un retour pour l'élément sélectionné.", + "4328a0a062": "Échec de la capture : {{value0}}", + "26615e116b": "error", + "8aec5bc044": "idle", + "759f32af29": "Téléchargement", + "c8bc7f1f9e": "demandé", + "4300f38145": "Téléchargement depuis {{value0}}{{value1}}", + "31375046b7": "Télécharger depuis {{value0}}", + "acbe79fd01": "Capturer un élément de la page ({{value0}})", + "572046436a": "Navigateur distant", + "b313a7275b": "Ouverture du navigateur distant", + "5f66313863": "annotation", + "ea6af700da": "{{value0}} annotation", + "c13693fe27": "{{value0}} annotations", + "074f0ed10b": "{{value0}} annotation prête. Sélectionnez un autre élément ou copiez tous les retours.", + "a2164a6e5a": "{{value0}} annotations prêtes. Sélectionnez un autre élément ou copiez tous les retours.", + "9f6f2e8c19": "Le chemin du fichier téléchargé est indisponible.", + "0c79b7634d": "Impossible d'ouvrir le fichier téléchargé. Il a peut-être été déplacé ou supprimé.", + "397d9dc923": "Impossible d'afficher le fichier téléchargé. Il a peut-être été déplacé ou supprimé.", + "39c04fed61": "Téléchargement en pause", + "5c3d530a68": "Téléchargé", + "4bb7424d6b": "Annulé", + "6e776f9ef9": "Échec du téléchargement", + "756bfc25c9": "Ouvert", + "09a9489aa5": "Afficher", + "2a4c4b8e1f": "Copier" + }, + "BrowserToolbarMenu": { + "429ef481f9": "Annuler", + "64f448fb6e": "Nom du profil", + "67e9b9fcd6": "Nouveau profil de navigateur", + "58f2c81542": "Basculer", + "a38f217b46": "Changer de profil rechargera cette page. Toutes les données de formulaire non enregistrées seront perdues.", + "fe683eb3b4": "Changer de profil", + "a771c2b6c8": "Paramètres du navigateur…", + "ed8f54509d": "Par défaut", + "e5d31de1a9": "Taille de la zone d'affichage", + "56f94f4ffa": "Depuis un fichier…", + "eb280bfb11": "Depuis {{value0}}", + "2293adf620": "Importer des cookies", + "cf7cdc67ef": "Nouveau profil…", + "7b838540c7": "Menu du navigateur", + "6aa42813e4": "{{value0}} cookies importés depuis {{value1}}{{value2}}.", + "a7a86702b3": "Profil {{value0}} créé et activé", + "4d2f9f13a7": "Échec de la création du profil.", + "3ccd29d771": "Profil {{value0}} activé", + "569bce8eb1": "Créer", + "bf648471c5": "Création…", + "53bbe3dab4": "{{value0}} cookies importés depuis un fichier.", + "c5f0e4d3b2a1": "{{value0}} cookies importés depuis {{value1}} ({{value2}}).", + "d6a1f5e4c3b2": "{{value0}} cookies importés depuis {{value1}}." + }, + "GrabConfirmationSheet": { + "314a0aaa5b": "Joindre à l'IA", + "7095e98362": "Copier la capture d'écran", + "26fd87f4df": "Copier", + "87d97bdd6d": "Annuler", + "effd75e330": "Contexte proche", + "7d1480fbf1": "HTML", + "9098b118ab": "Page", + "eb98a0971a": "\"", + "d053db279d": "role=", + "a759d8f866": "Élément sélectionné", + "9c6ce0632a": "Capture d'écran de l'élément sélectionné", + "50f7114f99": "Vérifiez avant de joindre. Le contexte de page capturé peut inclure du contenu visible du site.", + "f3575229df": "Capturer", + "405bb315da": "Sans titre" + }, + "browser": { + "address": { + "bar": { + "suggestions": { + "87fcdd0da9": "Recherche {{value0}}" + } + } + } + }, + "annotate": { + "use": { + "browser": { + "page": { + "grab": { + "annotations": { + "0c7b9b2b7a": "Copied", + "c937229f19": "Capture effectuée", + "1f5cb19034": "Annotation ajoutée" + } + } + } + } + } + }, + "navigate": { + "use": { + "browser": { + "page": { + "navigation": { + "downloads": { + "8683b84b9e": "La page du navigateur n'est pas prête pour le dépôt de fichiers.", + "22272f2784": "Déposez les fichiers sur la page du navigateur, pas sur la barre d'outils." + } + } + } + } + } + } + }, + "profile": { + "user": { + "agent": { + "option": { + "04af3dc12b": "Utiliser le user agent non modifié", + "5bf47a3c91": "Peut améliorer la connexion à Google, mais réduire la compatibilité avec les sites protégés contre les bots." + } + } + } + }, + "webauthn": { + "account": { + "fallback": "Clé d'accès", + "title": "Choisir une clé d'accès", + "description": "Choisissez le compte à utiliser avec cette clé de sécurité.", + "site": "Site", + "cancel": "Annuler" + } + } + }, + "automations": { + "AutomationCustomCronPanel": { + "3e3b2c369f": "Expression cron", + "e81a02d61b": "Saisissez une expression cron à cinq champs valide avant d'enregistrer.", + "968e66d686": "Saisissez une expression cron à cinq champs.", + "cadb7b0bc9": "invalide", + "f6ca30da23": "Cron personnalisé valide", + "a226dbdd40": "Minute", + "ec9c1e35df": "Heure", + "2d82246d23": "Jour", + "0e1de0358b": "Mois", + "77e96bded6": "Jour de la semaine" + }, + "AutomationPromptDisclosure": { + "showLess": "Afficher moins", + "showMore": "Afficher plus" + }, + "AutomationDetail": { + "007c8ad874": "Prompt", + "a1d52c2189": "Couverture d'utilisation", + "449fc83bf7": "Tokens", + "401f40ae79": "Dépense est.", + "a7c312430d": "Dernière exécution", + "2df8970cd5": "Agent", + "e353ab9516": "Pré-vérification", + "620b22145e": "Marge", + "15ea446b93": "Session", + "5405a09b1f": "Lieu d'exécution", + "2f8baf5360": "Créer depuis", + "578ff46987": "Prochaine exécution", + "18763ded26": "Planification", + "dbef8dc110": "Cette automatisation SSH ne s'exécute que si Orca peut joindre l'hôte SSH. Si la reconnexion exige des identifiants interactifs ou si l'hôte est indisponible, l'exécution est enregistrée comme ignorée.", + "1f6026358e": "Supprimer l'automatisation", + "d79452fb30": "Reprendre l'automatisation", + "91a4155e95": "Mettre l'automatisation en pause", + "4b1ea02d2e": "Modifier l'automatisation", + "2fb1605beb": "Exécuter maintenant", + "221916d93c": "Créez une automatisation pour planifier le travail des agents.", + "de0fedac06": "new_per_run", + "51a470b966": "ssh", + "b09b2384fd": "En pause", + "eaa02014f8": "Activé", + "29baf8f4c2": "Source" + }, + "AutomationEditorDialog": { + "fb1896a5e7": "Annuler", + "57b722cbba": "Agent", + "6ff66f9012": "Nouvelle exécution", + "a2e688226d": "Worktree", + "6f9610e667": "Les exécutions se font dans un worktree de l'espace de travail sélectionné. Chaque nouvelle exécution crée un espace de travail neuf depuis la branche sélectionnée.", + "2c3fd9bfa1": "Aide sur le mode espace de travail", + "b28b140eaf": "Espace de travail", + "0d17f4ca8f": "Sélectionner un projet", + "02d351877e": "Projet", + "a4ac8fcc62": "/goal", + "827b25a81e": "Prend en charge les skills, les chemins de fichiers et les commandes intégrées comme", + "6d778190b7": "Exécuter l'audit hebdomadaire des dépendances et résumer les changements à risque.", + "058c23cb3f": "Prompt", + "c4b19094c2": "Planification", + "e46c1aa9ad": "Créer", + "a9d9dccf77": "Enregistrer", + "777548c2d6": "Enregistrer les modifications", + "e8c2a14f70": "Une fois enregistrée, s'exécute automatiquement jusqu'à sa mise en pause.", + "ff5db28639": "existant" + }, + "AutomationEditorDialogHeader": { + "31f9253920": "Utiliser un modèle", + "7e35393632": "Hermes", + "6f309eef8d": "Orca", + "58f56b73d9": "Nom de l'automatisation", + "1d9826933e": "Audit du dépôt en semaine", + "4133d33862": "Créer une automatisation", + "0a75e5e2fa": "Créer une automatisation Hermes", + "03142e7721": "Modifier une automatisation Hermes", + "17086b48ee": "Modifier l'automatisation", + "4c8e1a72b9": "Une tâche d'agent récurrente" + }, + "AutomationMissedRunGraceField": { + "0f4459e91d": "48 heures", + "adbab51feb": "24 heures", + "ba50e2a230": "12 heures", + "2dc9ee84d0": "3 heures", + "521f77cd58": "1 heure", + "e5ad263ae5": "30 minutes", + "529dc6c0b7": "Pas de marge", + "3d70c185c8": "Si Orca ou l'hôte d'exécution était indisponible à l'heure prévue, Orca exécute une occurrence manquée dès qu'il redevient disponible dans cette fenêtre. Les exécutions manquées plus anciennes sont ignorées.", + "3df53d554a": "Aide sur la marge des exécutions manquées", + "fc089e5fde": "Marge" + }, + "AutomationEditorPromptSection": { + "a7c3e91b04": "Modifier le nom" + }, + "AutomationPrecheckFields": { + "d2a2ac89ac": "10 min", + "bf49585b3c": "5 min", + "d84d3765fd": "2 min", + "c820119736": "1 min", + "51e28cdad9": "30 s", + "bb2dfb3629": "Délai d'expiration", + "99a577306c": "gh pr list --json number -q '.[0].number'", + "c2a762a180": "Pré-vérification" + }, + "AutomationRunHistory": { + "402651bfb6": "Aucune exécution pour l'instant.", + "9974a2b429": "Statut", + "13988187b3": "Tokens", + "86a248187e": "Dépense", + "149c0b49c7": "Espace de travail", + "8faaa00726": "Exécution", + "53fc5f07ab": "Historique des exécutions", + "a00e38d1a3": "n/d", + "fdb3caa8fb": "connu" + }, + "AutomationRunPageFrame": { + "40a511bed4": "Contexte d'exécution", + "33741dd973": "Retour aux exécutions" + }, + "AutomationSchedulePicker": { + "9e677335b0": "Minute", + "d90981f766": "Heure", + "6b914c5fbb": "Jour", + "233b8c94b6": "Cadence", + "55b2ef82a4": "Toutes les heures", + "f0202f3a89": "Chaque jour", + "57e83307d0": "En semaine", + "837d902bba": "Hebdomadaire", + "ddba78647e": "Cron personnalisé" + }, + "AutomationTimeField": { + "39ec1383f6": "AM ou PM", + "32a5e4e35e": "Minute", + "aa593eb5e2": "Heure" + }, + "AutomationSessionField": { + "f3c76dce51": "Réutiliser", + "c90888ee94": "Nouvelle", + "b675112193": "Réutiliser envoie les futures exécutions vers la précédente session d'automatisation encore vivante. Si cette session n'existe plus, Orca en démarre une nouvelle.", + "4bdce31f37": "Aide sur la réutilisation de session", + "5ad314118e": "Session" + }, + "AutomationsPage": { + "2695883141": "supprimer", + "c3a28c9793": "Sélectionnez une automatisation pour voir ses exécutions.", + "295698292f": "Relancer", + "0e110a3469": "Exécutions", + "bb1b2cd31e": "Vue d'ensemble", + "97ff587ee3": "Connectez cette source pour détecter les automatisations Hermes dans le profil distant.", + "aaa007846f": "source indisponible", + "25060635c6": "Ajouter", + "d207ab4c25": "Partir d'un modèle", + "15e0bfb13b": "Supprimer", + "f4612e3f78": "Édition", + "2faecab10b": "Exécuter maintenant", + "82eb6cb933": "source", + "13118faadf": "Projet inconnu", + "587a4b205c": "Suivant", + "761a35834d": "Automatisation", + "73f630b49d": "Annuler", + "1b586f0e2b": "activé", + "02a33e3204": "de", + "9adfab2596": "Supprimer l'automatisation externe", + "1e2e41392f": "Ne plus demander", + "b264564427": "et son historique d'exécution. Les espaces de travail créés par les exécutions précédentes ne sont pas supprimés.", + "080dcb5fbb": "Supprimer l'automatisation", + "19a6e30eae": "Actualiser les automatisations", + "8d1afa8269": "Ajouter une automatisation", + "77c2778945": "Automatisations", + "0329f9bef1": "Fermer · Esc", + "67c7ff795b": "Fermer les automatisations", + "e1bf9b1512": "L'espace de travail n'est pas disponible.", + "3e42a5cc1b": "Échec de la connexion SSH.", + "9f2855677c": "SSH connecté.", + "126d726546": "L'action sur l'automatisation externe a échoué.", + "37288942f0": "Automatisation externe reprise.", + "77c518a34b": "Automatisation externe mise en pause.", + "4d7878402c": "Automatisation externe en file d'attente.", + "4c22bc9913": "Automatisation externe supprimée.", + "3a4c476aa0": "Échec de la nouvelle exécution de l'automatisation.", + "a1bdb57008": "Exécution de l'automatisation en file d'attente.", + "8a3226f172": "Ouvrir les paramètres", + "d2a01b0b6f": "Vous pouvez changer cela dans les paramètres.", + "690b94da54": "Cette confirmation sera ignorée la prochaine fois.", + "b11170a008": "Échec de l'enregistrement de l'automatisation.", + "2a20596d6b": "Automatisation enregistrée.", + "244727e655": "Automatisation mise à jour.", + "77b81bc4ac": "Automatisation Hermes créée.", + "08efc3ae12": "Automatisation Hermes mise à jour.", + "e431bb85d4": "Choisissez un espace de travail sur le même hôte que cette automatisation Hermes.", + "32534e7c9c": "Choisissez un espace de travail disponible avant d'enregistrer.", + "2360ffc956": "Choisissez un agent activé avant d'enregistrer.", + "6e91dab317": "Saisissez une planification avancée valide avant d'enregistrer.", + "64bdb2304f": "Choisissez une planification prise en charge avant d'enregistrer.", + "2430fecf53": "Choisissez un lieu d'exécution et saisissez un prompt avant d'enregistrer.", + "7934ee0d81": "Connecter SSH", + "f93ed7a6f8": "Connexion...", + "8705757e27": "tâche", + "376631ef2b": "Reprendre", + "b457436d6a": "Pause", + "0ae52dd760": "hermes", + "e059042585": "Lecture seule", + "aecdc3681f": "Gérable", + "8500baacb4": "source externe", + "36f71740a7": "Espace de travail sélectionné", + "cd8397cc32": "Nouvel espace de travail à chaque exécution", + "dd0bc7a1ba": "new_per_run", + "7b2e285552": "Les connexions SSH sont indisponibles dans ce client.", + "d441032f7e": "pause", + "5918020edc": "exécuter", + "a21f6c33ad": "Source des automatisations actualisée.", + "53f06f0ad5": "Réessayer la source", + "pendingAutomationMissing": "Cette automatisation n'est plus disponible.", + "pendingAutomationRunMissing": "L'historique d'exécution n'est plus disponible.", + "noSearchMatches": "Aucune automatisation ne correspond à votre recherche.", + "paused": "En pause", + "runCount": "{{count}} exécutions", + "runCount_one": "{{count}} exécution", + "runCount_other": "{{count}} exécutions", + "projectDefaultBaseRef": "défaut du projet", + "createFromBaseRef": "Créer depuis {{baseRef}}", + "missingWorkspace": "Espace de travail manquant", + "runUsageSummary": "{{cost}} est. · {{tokens}} tokens", + "usageUnavailable": "Utilisation indisponible", + "noRunUsageYet": "Aucune utilisation d'exécution pour l'instant", + "noWorkspace": "Aucun espace de travail", + "latestSavedOutput": "Dernière sortie enregistrée", + "backToList": "Toutes les automatisations", + "tableName": "Nom", + "tableProject": "Projet", + "tableStatus": "Statut", + "tableActions": "Actions", + "newAutomation": "Nouvelle automatisation", + "rowActions": "Actions d'automatisation", + "tableLastRun": "Dernière exécution", + "noListMatches": "Aucune automatisation ne correspond." + }, + "AutomationsPageSkeleton": { + "55527b7bcf": "Chargement des automatisations" + }, + "CreateFromPicker": { + "f061f49e3f": "Rechercher des branches du dépôt...", + "dd3841b442": "Brancher depuis", + "ef6d762538": "Défaut du projet", + "e53d306056": "{{value0}} (par défaut)", + "79512f22a7": "Aucune branche trouvée.", + "9ce96621f4": "Recherche des branches..." + }, + "ExternalAutomationManagers": { + "e02f970595": "Aucun gestionnaire d'automatisations externes trouvé.", + "6da3bfba4b": "automatisations trouvées.", + "3d58d5b67d": "Non", + "a42bf2b27e": "Supprimer l'automatisation externe", + "1c3bfd38fe": "Reprendre l'automatisation externe", + "0def1693bb": "Mettre l'automatisation externe en pause", + "1df491fd00": "Modifier l'automatisation externe", + "cc77ba88ff": "Exécuter l'automatisation externe", + "5820648765": "Dernière", + "844f1acb72": "trouvé(s)", + "20fd7a3a15": "suivant", + "c6695e6fbd": "Automatisations externes", + "5524365227": "OpenClaw", + "766abf833c": "Hermes", + "bf5f67b590": "hermes", + "e66091daf4": "exécutions", + "8e9165af08": "exécuter", + "2b0adbce21": "En pause", + "b3feba84c7": "Actif", + "92405f1431": "Indisponible", + "dbdcec22bd": "Lecture seule", + "0a2d4359a8": "Gérable", + "330b3c32e8": "disponible", + "e2532150ed": "automatisations", + "701515f010": "automatisation" + }, + "ExternalAutomationRunTable": { + "0ba9c0a95c": "Page d'exécutions suivante", + "52d468a0b8": "Page d'exécutions précédente", + "7475c0ce96": "sur", + "be551397ca": "Statut", + "a813df9808": "Aperçu", + "d4b34feb66": "Heure d'exécution", + "2d4388a908": "Exécutions", + "9c080765ff": "Aucune exécution Hermes trouvée pour l'instant.", + "8ea934cacf": "Chargement des exécutions...", + "d5527d8fe7": "exécutions", + "872d032d05": "exécuter" + }, + "HermesCronOutputView": { + "e27c716b43": "Prompt", + "4557213074": "Réponse", + "05affc68e3": "Erreur", + "88d48157fc": "default" + }, + "WorkspaceCombobox": { + "ee5b280eba": "Aucun espace de travail trouvé.", + "8e9c8cc6b5": "Rechercher des espaces de travail...", + "66a0cd9628": "Sélectionner un espace de travail" + }, + "automation": { + "templates": { + "37571fcb16": "Rechercher les travaux bloqués, les fichiers générés obsolètes et les validations locales en échec.", + "8a0228bea3": "Vérification horaire de la file d'attente", + "3b7281c75f": "Analyser le travail récent et signaler les risques de justesse, d'UX et de couverture de tests.", + "6023075b27": "Revue quotidienne des changements", + "513401db93": "Préparer un résumé hebdomadaire des risques de release à partir de l'état actuel du projet.", + "39ed39280a": "Préparation de la release", + "a7fbd32ddb": "Vérifier chaque jour de semaine les dépendances, les tests en échec et les changements ouverts à risque.", + "b84757677d": "Audit du dépôt en semaine", + "repoHealth": { + "category": "Santé du dépôt", + "name": "Audit du dépôt en semaine", + "prompt": "Passer en revue la santé du dépôt. Vérifier les mises à jour de dépendances, les tests en échec, l'état du lint/typecheck et les changements ouverts à risque. Résumer les constats et suggérer l'action suivante." + }, + "releasePrep": { + "category": "Préparation de release", + "name": "Revue de préparation de release", + "prompt": "Préparer un résumé de préparation de release. Rechercher les bloqueurs, les changements à risque non mergés, les validations manquantes et les lacunes de documentation. Terminer par une recommandation concise : release ou pas." + }, + "recurringReview": { + "category": "Revue récurrente", + "name": "Revue quotidienne des changements", + "prompt": "Passer en revue les changements récents de cet espace de travail. Se concentrer sur les risques de justesse, les régressions UX, les tests manquants et les tâches de suivi. Garder le rapport court et actionnable." + }, + "maintenance": { + "category": "Maintenance", + "name": "Vérification de maintenance horaire", + "prompt": "Rechercher les travaux bloqués, les fichiers générés obsolètes, les validations en échec et tout ce qui requiert une attention humaine. Ne signaler que les problèmes actionnables." + } + }, + "list": { + "last": { + "run": { + "done": "Terminé", + "failed": "Échec" + } + } + }, + "schedule": { + "label": { + "086a5a9fe2": "Planification invalide", + "ba20c92073": "Planification personnalisée", + "a95afb7483": "Toutes les heures à :{{minute}}", + "280ccd2701": "Chaque jour à {{time}}", + "3f1422adc1": "En semaine à {{time}}", + "cc71e252ba": "{{day}}s à {{time}}" + } + } + }, + "external": { + "automation": { + "schedule": { + "display": { + "a8e92b815a": "Planification indisponible" + } + } + } + }, + "AutomationProjectCombobox": { + "search": "Rechercher des projets/dossiers…", + "empty": "Aucun projet/dossier ne correspond à votre recherche.", + "chooseHost": "Choisir l'hôte d'automatisation", + "adding": "Ajout du projet…", + "addProject": "Ajouter un projet" + }, + "AutomationSetupDecisionField": { + "5a7863909c": "Exécuter le setup pour chaque nouvel espace de travail", + "18f000ad4e": "Avancé", + "874b72195b": "Quand cette automatisation crée un espace de travail, le prépare comme lors de la création manuelle d'un worktree — exécute le setup du projet et ouvre ses onglets de terminal." + }, + "AutomationListSearchField": { + "tooLong": "Le texte recherché est trop long — liste non filtrée", + "label": "Rechercher des automatisations", + "placeholder": "Rechercher...", + "tooLongShort": "Trop long", + "clear": "Effacer la recherche" + }, + "AutomationListLocalRows": { + "c92c9463c6": "Actions d'automatisation" + }, + "AutomationListFilterMenu": { + "926e785e4d": "Rechercher des agents...", + "491043ee45": "Aucun agent ne correspond à votre recherche.", + "removeFilter": "Retirer le filtre {{value0}}", + "failed": "Échec", + "succeeded": "Réussie", + "neverRan": "Jamais exécutée", + "filters": "Filtres", + "all": "Tous", + "clear": "Réinitialiser les filtres", + "agent": "Agent", + "lastRun": "Dernière exécution" + }, + "AutomationListSortHeader": { + "sortedAscending": "{{value0}}, tri croissant", + "sortedDescending": "{{value0}}, tri décroissant" + } + }, + "agent": { + "AgentCombobox": { + "19522e25ee": "Gérer les agents", + "986f946354": "Terminal vierge", + "579c768bde": "Aucun agent ne correspond à votre recherche.", + "48c6a5a9b4": "Rechercher des agents...", + "9c6b59fe58": "Définir par défaut", + "1b0d6965fa": "Défaut actuel" + }, + "AgentSettingsDialog": { + "50cdb57c03": "Gérez les agents IA, définissez-en un par défaut et personnalisez les commandes.", + "fc0268e4ed": "Agents" + } + }, + "activity": { + "ActivityPrototypePage": { + "cf780197a1": "Sélectionnez un agent pour voir son activité", + "e3db9892f6": "Aucune activité pour l'instant.", + "1b633f5c1e": "Connexion au terminal...", + "8de7c5beaa": "Terminal indisponible", + "866083500b": "Faire glisser pour redimensionner", + "443690186e": "Redimensionner la liste des fils d'activité", + "7cd632006b": "Aucune activité d'agent ne correspond à ces filtres.", + "a2b4437bfb": "Activité de {{value0}}", + "023ff75afe": "Tout marquer comme lu", + "f70e4bec47": "Mode compact", + "a472a14700": "Plus d'options", + "db8a1878b5": "Options de la liste des fils", + "d1a88df9a8": "Afficher uniquement les fils non lus", + "f6396e1f85": "Agent", + "b29191b3e0": "Worktree", + "8c3b621ddf": "Projet", + "4a3986b200": "Statut", + "770d458144": "Grouper l'activité des agents par", + "795cbf26e2": "Filtrer...", + "4616ea39fd": "Aller à l'espace de travail", + "59b131fbd9": "Marquer le fil comme non lu", + "beb2c19173": "Non lus", + "5651b216c6": "Projet inconnu", + "22b22034bc": "Terminal autonome indisponible dans Activité.", + "afdc2139a8": "Terminal de l'agent fermé. Ouvrez un nouveau terminal dans cet espace de travail pour continuer." + }, + "ActivityTitlebarControls": { + "f915168c8e": "non lu", + "d6a8de3934": "agents", + "dc708f3eff": "Fermer les agents" + } + }, + "confirmation": { + "dialog": { + "8490e5d36a": "Confirmer", + "56f5c60e0c": "Annuler", + "92bac3217e": "Ne plus demander" + }, + "skip": { + "saved": "Cette confirmation sera ignorée la prochaine fois.", + "savedDescription": "Vous pouvez changer cela dans les paramètres.", + "openSettings": "Ouvrir les paramètres", + "preference": { + "0b0cb6e3f9": "Impossible d'enregistrer la préférence de confirmation." + } + } + }, + "jira": { + "connect": { + "dialog": { + "63ce735809": "Se connecter", + "4a2ab52781": "Vérification…", + "79e7aaed39": "Annuler", + "fdd26d81cc": "Paramètres du compte Atlassian", + "8090504a3e": "Créez un token dans", + "7b3967c12f": "Token API Atlassian", + "3d81bf3ab3": "Token API", + "e91b9a4073": "you@example.com", + "2849ddb295": "E-mail Atlassian", + "70fcd360c4": "https://example.atlassian.net", + "e176f9d0c5": "URL du site Jira Cloud", + "d785c42b8b": "Utilisez l'URL d'un site Jira Cloud, un e-mail Atlassian et un token API pour parcourir les tickets.", + "8388bdea2b": "Connecter le site Jira", + "2e2b69e48e": "Utilisez une URL de base Jira auto-hébergée et un jeton d'accès personnel pour parcourir les tickets.", + "b67e919bd5": "Type d'instance Jira", + "17787d6e4b": "Atlassian Cloud", + "bc7a831773": "Auto-hébergé", + "3489e186d6": "URL du site Jira", + "cbc27fa599": "https://jira.example.com", + "730d973bae": "Jeton d'accès personnel", + "8b9c7b9e7b": "Jeton d'accès personnel Jira", + "ccfb086d3e": "Créez un jeton d'accès personnel dans votre profil Jira, rubrique Personal Access Tokens.", + "1d947a07ab": "Utilisez une URL de base Jira auto-hébergée, un nom d'utilisateur et un mot de passe pour parcourir les tickets.", + "f49708c369": "Méthode d'authentification Jira", + "84a810dd0e": "Nom d'utilisateur et mot de passe", + "8d1223fa5c": "Nom d'utilisateur", + "be9eba0a1b": "nom d'utilisateur", + "70035652d7": "Mot de passe", + "c50abbf340": "Mot de passe du compte Jira", + "d8737db691": "Utilisez le nom d'utilisateur et le mot de passe de votre compte Jira Server ou Data Center." + } + } + }, + "rightSidebar": { + "FolderWorkspaceWorktreesPanel": { + "unavailable": "Les espaces de travail ne sont affichés que pour les espaces de travail de type dossier.", + "label": "Espaces de travail", + "description": "Affiche les worktrees attachés à cet espace de travail de type dossier.", + "countOne": "1 worktree attaché", + "countMany": "{{value0}} worktrees attachés", + "emptyTitle": "Aucun worktree attaché pour l'instant", + "emptyCopy": "Les worktrees créés depuis cet espace de travail apparaîtront ici." + }, + "FolderWorkspacePrChecksPanel": { + "unavailable": "Les vérifications PR ne sont affichées que pour les espaces de travail de type dossier.", + "refresh": "Actualiser les vérifications PR", + "emptyTitle": "Aucun worktree attaché pour l'instant", + "emptyCopy": "Les vérifications PR apparaîtront ici après l'attachement de worktrees à cet espace de travail de type dossier.", + "openChecksTab": "Ouvrir l'onglet Vérifications de {{value0}}", + "openReviewExternally": "Ouvrir {{value0}} en externe", + "summary": "{{value0}} attachés · {{value1}} avec PR/MR · {{value2}} attention requise · {{value3}} en attente · {{value4}} réussies · {{value5}} sans PR · {{value6}} inconnu", + "showDetails": "Afficher les détails des vérifications PR de {{value0}}", + "hideDetails": "Masquer les détails des vérifications PR de {{value0}}", + "reviewChecks": "Vérifications de revue", + "allChecksPassing": "toutes les vérifications passent", + "oneWorktree": "1 worktree", + "worktreeCount": "{{value0}} worktrees", + "oneFailing": "1 en échec", + "failingCount": "{{value0}} en échec", + "onePending": "1 en attente", + "pendingCount": "{{value0}} en attente" + }, + "parentPrChecks": { + "rowSummary": { + "failingCount": "{{value0}} en échec", + "pendingCount": "{{value0}} en attente", + "checksFailing": "Vérifications en échec", + "mergeConflicts": "Conflits de fusion", + "checksPending": "Vérifications en attente", + "checksPassing": "Vérifications réussies", + "merged": "Fusionnés", + "closedWithoutMerge": "Fermée sans merge", + "draftReview": "Revue de brouillon", + "noCheckSignal": "Aucun signal de vérification", + "reviewUnavailable": "Statut de revue indisponible", + "noPrLinked": "Aucune PR liée", + "detailsUnavailable": "Détails de revue indisponibles", + "refreshFailed": "Échec de l'actualisation", + "checking": "Vérification du statut de revue…", + "notFetched": "Statut pas encore récupéré", + "unavailableWorktree": "Indisponible pour ce worktree" + }, + "groups": { + "needsAttention": "Attention requise", + "pending": "En attente", + "merged": "Fusionnés", + "passing": "Réussies", + "draftOrNoChecks": "Brouillon / sans vérification", + "noPr": "Sans PR", + "unavailable": "Indisponible" + } + }, + "pluginPanelBridgeHost": { + "actionsUnavailable": "Les actions de plugin ne sont pas disponibles dans ce client.", + "messageTooLarge": "Le message dépasse la taille limite.", + "tooManyRequests": "Trop de requêtes." + } + }, + "link": { + "routing": { + "preference": { + "dialog": { + "badge": "Lien de terminal", + "preview": "Aperçu", + "keep": { + "title": "Ouvrir les liens de terminal dans le navigateur d'Orca ?", + "description": "Ou utilisez votre navigateur système par défaut.", + "orca": { + "button": "Garder Orca" + } + }, + "title": "Ouvrir les liens du terminal dans le navigateur d'Orca ?", + "description": "Utilisez le navigateur d'Orca pour les liens du terminal, ou gardez votre navigateur système.", + "link": { + "label": "Lien" + }, + "orca": { + "note": "Orca peut utiliser les cookies importés pour les sites où vous êtes connecté.", + "button": "Ouvrir dans Orca" + }, + "settings": { + "note": "Modifiable plus tard dans Paramètres → Navigateur." + }, + "shortcut": { + "note": { + "prefix": "Quand les liens s'ouvrent dans Orca,", + "suffix": "un clic ouvre le navigateur système une seule fois." + } + }, + "system": { + "button": "Utiliser le navigateur système" + } + } + } + } + }, + "task": { + "project": { + "source": { + "combobox": { + "noProjects": "Aucun projet", + "allProjects": "Tous les projets", + "hostCount": "{{value0}} hôtes", + "searchProjects": "Rechercher des projets...", + "noMatches": "Aucun projet ne correspond à votre recherche.", + "chooseSource": "Choisir la source des tâches" + } + } + } + }, + "taskPageEmptyState": { + "noProjectSourcesTitle": "Aucune source de projet sélectionnée", + "noProjectSourcesDescription": "Sélectionnez au moins une source de projet pour qu'Orca sache sur quel hôte/compte récupérer les tâches.", + "noMatchingGitHubWorkTitle": "Aucun travail GitHub correspondant", + "changeQueryDescription": "Modifiez la requête ou effacez-la.", + "noGitLabIssuesTitle": "Aucun ticket GitLab", + "noGitLabIssuesDescription": "Aucune issue GitLab ne correspond à ce filtre.", + "noGitLabMrsTitle": "Aucune merge request GitLab", + "noGitLabMrsDescription": "Aucune MR GitLab ne correspond à ce filtre.", + "noGitLabWorkTitle": "Aucun travail GitLab", + "noGitLabWorkDescription": "Aucun work GitLab ne correspond à ce filtre." + }, + "taskSourceContextSummary": { + "sourceUnavailable": "Source {{value0}} indisponible : {{value1}}", + "someSourceHostsUnavailable": "Certains hôtes de la source {{value0}} sont indisponibles : {{value1}}", + "reconnectOrUpdateTitle": "Reconnectez ou mettez à jour {{value0}} pour charger cette source." + }, + "ShortcutKeyCombo": { + "07eb4985a1": "Appuyez deux fois sur {{value0}}" + }, + "star": { + "nag": { + "StarNagToastHost": { + "starredThanks": "Étoile ajoutée — merci !", + "githubOpened": "GitHub ouvert", + "opening": "Ouverture…", + "starring": "Ajout de l'étoile…", + "openGithub": "Ouvrir GitHub", + "starOnGithub": "Mettre une étoile sur GitHub", + "onboardingCompleted": "Prise en main terminée !", + "body": "Si Orca vous plaît jusqu'ici, une étoile GitHub aide d'autres développeurs à le découvrir.", + "dismiss": "Ignorer", + "later": "Plus tard" + } + } + }, + "nativeChat": { + "contextMenu": { + "copy": "Copier" + } + }, + "orca": { + "profiles": { + "signout": { + "confirm": { + "title": "Se déconnecter d'Orca ?", + "description": "Les artifacts et Orca Relay seront indisponibles jusqu'à votre reconnexion. Vos projets et worktrees locaux ne seront pas affectés.", + "cancel": "Annuler", + "action": "Se déconnecter" + } + } + } + }, + "linear-issue-attribute-filter-dropdowns": { + "removeFilter": "Retirer le filtre {{value0}}", + "teamRequired": "Sélectionnez une équipe pour charger les statuts, responsables et étiquettes de cet espace de travail.", + "filters": "Filtres", + "allWorkspacesTitle": "Sélectionnez un espace de travail", + "allWorkspacesBody": "Les filtres de statut, responsable et étiquette utilisent des identifiants issus d'un seul espace de travail Linear. Choisissez-en un pour filtrer selon ces attributs.", + "optionsFromTeam": "Options de {{team}}", + "clearAll": "Effacer tous les filtres" + }, + "linear-issue-attribute-filter-sections": { + "status": "Statut", + "priority": "Priorité", + "assignee": "Assigné", + "unassigned": "Non assigné", + "labels": "Étiquettes", + "countSelected": "{{count}} sélectionnés", + "selected": "sélectionnés", + "searchPriority": "Filtrer par priorité…", + "searchStatus": "Filtrer par statut…", + "searchLabels": "Filtrer par étiquette…", + "searchAssignee": "Filtrer par responsable…", + "back": "Retour" + }, + "browser-pane": { + "markup": { + "cancel": "Annuler", + "clear": "Tout effacer", + "copiedToast": "Markup copié — collez-le dans votre agent ({{value0}})", + "copy": "Copier le markup", + "drawButton": "Dessiner sur la capture d'écran", + "drawHint": "Dessinez sur la page, puis copiez le markup pour le coller dans votre agent.", + "drawHintBadge": "Nouveau", + "drawHintDismiss": "Compris", + "drawHintTry": "Essayer", + "errorAttach": "Impossible de joindre la capture d'écran annotée.", + "errorCapture": "Impossible de capturer la page à annoter.", + "errorUnavailable": "Le markup de capture d'écran n'est pas disponible sur cette page.", + "fontSize": "Taille de police", + "hint": "Dessinez sur la page, puis copiez le markup pour le coller dans votre agent.", + "redo": "Rétablir", + "style": "Couleur et épaisseur", + "textInput": "Texte d'annotation", + "tool": { + "arrow": "Flèche", + "ellipse": "Ellipse", + "highlight": "Surligneur", + "pen": "Stylo", + "rect": "Rectangle", + "text": "Texte" + }, + "undo": "Annuler", + "widthOption": "{{value0}} px" + }, + "grab": { + "errorNotReady": "Cette page n'est pas encore prête pour la sélection d'éléments.", + "errorNotAuthorized": "L'accès au navigateur a été refusé.", + "errorAlreadyActive": "La sélection d'éléments est déjà active.", + "errorInjectionFailed": "Impossible de démarrer la sélection d'éléments sur cette page." + } + }, + "pluginCatalog": { + "PluginCatalogLayout": { + "title": "Plugins", + "filter": "Filtre de plugins", + "all": "Tous", + "installed": "Installés", + "searchLabel": "Rechercher des plugins", + "searchPlaceholder": "Rechercher des plugins, catégories ou éditeurs" + } + }, + "terminalPane": { + "useManualTerminalWorktreeParking": { + "cannotPark": "Ces terminaux ne peuvent pas être mis en attente en toute sécurité." + } + }, + "WorktreeBaseFallbackDialog": { + "title": "Espace de travail créé depuis une base locale", + "description": "La ref de suivi distante « {{value0}} » était indisponible ; Orca a donc utilisé la locale « {{value1}} ». Cet espace de travail pourrait ne pas inclure les derniers changements distants.", + "dismiss": "Compris" + }, + "LinuxPackageInstallRecoveryCard": { + "e3de29c86a": "Afficher le paquet", + "3da99454c6": "Réessayer l'installation automatique", + "55c86654b7": "Copier la commande d'installation", + "53e1559f99": "Échec de l'installation automatique", + "a7ac6ec78b": "Orca a téléchargé la mise à jour mais n'a pas pu installer le paquet système automatiquement.", + "82c6dbea00": "Copiez la commande et exécutez-la dans un terminal système sur l'ordinateur où Orca est installé. Une fois terminée, quittez puis rouvrez Orca pour utiliser la nouvelle version.", + "53c4b8e148": "Aucun agent d'authentification exploitable n'a répondu à la demande d'installation privilégiée.", + "c732bcbf8f": "Vérification du paquet...", + "aa57fa4f80": "Commande copiée. Exécutez-la dans un terminal système pour installer {{value0}}, puis quittez et rouvrez Orca.", + "b7e7c5bc95": "Orca vérifie le fichier téléchargé par rapport aux métadonnées de version au moment où il construit cette commande. Le paquet système lui-même n'est pas vérifié par signature, et Orca ne peut pas garantir le fichier au-delà de ce point." + }, + "pr-check-counts": { + "passingChip": "{{value0}} réussis", + "failingChip": "{{value0}} en échec", + "pendingChip": "{{value0}} en attente", + "needsActionChip": "{{value0}} action requise", + "unresolvedChip": "{{value0}} non résolus" + }, + "BrowserPane": { + "streamConnectionLost": "Connexion au serveur distant perdue.", + "streamConnectionUnreachable": "Impossible de joindre le serveur distant.", + "streamRestartFailed": "Échec du redémarrage du flux du navigateur distant.", + "streamCapabilityUnsupported": "Le runtime sélectionné ne prend pas en charge le streaming du navigateur distant." + }, + "artifacts": { + "ArtifactsPage": { + "signInAgain": "Reconnectez-vous à Orca pour charger les artifacts.", + "loadFailed": "Impossible de charger les artifacts.", + "deleteTitle": "Supprimer l'artifact ?", + "deleteDescription": "« {{name}} » ne sera plus disponible via son lien public.", + "delete": "Supprimer", + "deleteFailed": "Impossible de supprimer l'artifact.", + "title": "Artefacts", + "refresh": "Actualiser", + "signInHeading": "Se connecter à Orca", + "signInCopy": "Connectez-vous pour voir et gérer les artifacts partagés via votre compte.", + "signingIn": "Connexion…", + "signIn": "Se connecter à Orca", + "empty": "Aucun artifact partagé", + "emptyCopy": "Ouvrez un fichier HTML ou Markdown et choisissez Partager comme artifact, ou demandez à votre agent de le partager.", + "openArtifact": "Ouvrir l'artifact", + "deleteArtifact": "Supprimer l'artifact", + "moreAvailable": "Davantage d'artifacts sont disponibles", + "moreAvailableCopy": "Chargez la page suivante pour continuer.", + "loadMoreFailed": "Impossible de charger davantage d'artifacts.", + "publishingOff": "La publication est désactivée", + "publishingOffCopy": "Rien sur cet appareil ne peut encore créer de lien d'artifact public. Autorisez la publication dans Paramètres → Artifacts, puis partagez depuis un fichier HTML ou Markdown ouvert, ou demandez à votre agent.", + "openArtifactsSettings": "Ouvrir Paramètres → Artifacts", + "retry": "Réessayer", + "reconnectHeading": "Se reconnecter à Orca", + "reconnectCopy": "Reconnectez-vous pour voir et gérer les artifacts partagés via votre compte.", + "signInAgainAction": "Se reconnecter", + "unconfiguredCopy": "La connexion au compte Orca n'est pas encore configurée sur cette machine.", + "openAccountSettings": "Ouvrir les paramètres du compte" + }, + "copySuccess": "Lien de l'artifact copié", + "copyFailed": "Impossible de copier le lien de l'artifact", + "copyLink": "Copier le lien", + "openInBrowser": "Ouvrir dans le navigateur", + "preview": "Aperçu de l'artifact", + "previewUnavailable": "Aperçu indisponible", + "previewUnavailableDescription": "Ouvrez cet artifact dans votre navigateur pour le consulter.", + "actions": "Actions de l'artifact", + "ArtifactActions": { + "more": "Plus d'actions sur l'artifact" + }, + "ArtifactCollection": { + "loadMore": "Charger plus", + "noMatches": "Aucun résultat" + }, + "ArtifactDetailHeader": { + "publicLink": "Toute personne disposant de ce lien peut le consulter", + "close": "Fermer" + }, + "updatedAt": "Mis à jour {{when}}", + "updatedRecently": "récemment", + "expiryUnknown": "Expiration inconnue", + "expired": "Lien expiré", + "expires": "Le lien expire {{when}}", + "ArtifactPublishButton": { + "a4a49da6af": "Partager comme artifact", + "confirmTitle": "Partager comme artifact", + "confirmDescription": "Ceci publie le fichier actuel à un lien consultable par toute personne disposant de l'URL.", + "accountTitle": "Compte Orca", + "accountDescription": "Connectez-vous pour créer et gérer ce lien.", + "signingIn": "Connexion…", + "signInAgain": "Se reconnecter", + "signIn": "Se connecter", + "publishingOffTitle": "Le partage d'artifacts est désactivé", + "publishingOffDescription": "Découvrez les liens publics et activez le partage dans les paramètres.", + "openSettings": "Ouvrir les paramètres Artifacts", + "sharing": "Partage…", + "sharePublicLink": "Partager le lien public", + "publishedDescription": "Toute personne disposant de ce lien peut consulter le fichier partagé.", + "checkingLink": "Recherche d'un lien existant…", + "checkFailed": "Impossible de rechercher un lien existant.", + "tryAgain": "Réessayer" + }, + "artifact-publish-flow": { + "9a078a0c65": "Le partage d'artifacts est indisponible", + "bba20daa6d": "Connectez-vous à Orca puis réessayez.", + "54b1805328": "Impossible de partager l'artifact", + "430019efd0": "Artifact partagé", + "2fc727c831": "Artifact mis à jour", + "5cb4f5ec36": "Copier le lien", + "fbb5018602": "Ce fichier est vide.", + "6112db5a1c": "Les artifacts partagés depuis Orca doivent peser moins de 800 Ko.", + "e2ed5acd8c": "Orca n'a pas pu lire ce fichier. Ouvrez-le depuis un espace de travail et réessayez.", + "6d475e9b25": "Seuls les fichiers HTML et Markdown locaux peuvent être partagés comme artifacts.", + "29a406be09": "Les artifacts doivent contenir du texte." + }, + "ArtifactPublishedLinkPanel": { + "copyLink": "Copier le lien", + "openLink": "Ouvrir le lien", + "updating": "Mise à jour…", + "update": "Mettre à jour le contenu partagé" + }, + "ArtifactDetailDrawer": { + "description": "Prévisualisez et gérez cet artifact partagé." + }, + "ArtifactListSearchField": { + "label": "Rechercher des artifacts", + "placeholder": "Rechercher...", + "clear": "Effacer la recherche" + }, + "ArtifactListTableHeader": { + "name": "Nom", + "type": "Tapez", + "size": "Taille", + "updated": "Mis à jour", + "expires": "Expiration", + "actions": "Actions" + }, + "ArtifactsPageSkeleton": { + "loading": "Chargement des artifacts" + }, + "expiredCompact": "Expiré", + "typeMarkdown": "Markdown", + "typeHtml": "HTML" + } + }, + "i18n": { + "hostedReview": { + "copy": { + "f0a4b8c2d1": "PR", + "e9f3a7b1c0": "pull request", + "d8e2f6a0b9": "Pull Request", + "c7d1e5f9a8": "GitHub", + "c4e8f1a2b9": "MR", + "b3d7e0f1a8": "merge request", + "a2c6d9e0f7": "Merge Request", + "91b5c8d7e6": "GitLab" + } + } + }, + "runtime": { + "remoteServerUpdateErrors": { + "manualRequired": "Ce serveur doit être mis à jour manuellement via son gestionnaire de services.", + "notAvailable": "Le serveur ne signale plus de mise à jour disponible. Revérifiez.", + "notDownloaded": "Le téléchargement de la mise à jour du serveur n'est pas terminé.", + "legacyServer": "Mettez à jour ce serveur une fois manuellement pour activer les mises à jour à distance.", + "updaterTimeout": "Délai dépassé en attendant le programme de mise à jour du serveur.", + "requestedVersionUnavailable": "Le programme de mise à jour du serveur n'a pas proposé la version d'Orca demandée.", + "updateUnavailable": "Le serveur n'a pas signalé de mise à jour disponible.", + "downloadIncomplete": "La mise à jour du serveur n'a pas fini de se télécharger.", + "reconnectTimeout": "Le serveur ne s'est pas reconnecté avec la version mise à jour." + }, + "webRuntimeSession": { + "remoteHostDisconnected": "L'espace de travail n'est pas connecté à un hôte Orca distant." + }, + "gitlabJobTraceClient": { + "loadFailed": "Échec du chargement du journal du job GitLab.", + "emptyTrace": "Aucun journal disponible pour ce job GitLab.", + "timedOut": "Délai dépassé lors du chargement du journal du job GitLab." + }, + "githubCheckDetailsTimeout": { + "timedOut": "Délai dépassé lors du chargement des détails des vérifications." + } + }, + "ssh": { + "sshConnectVerb": { + "reconnect": "Reconnecter", + "retry": "Réessayer", + "connect": "Se connecter", + "connecting": "Connexion…" + } + }, + "main": { + "window": { + "editableContextMenu": { + "table": "Tableau", + "insertRowAbove": "Insérer une ligne au-dessus", + "insertRowBelow": "Insérer une ligne en dessous", + "deleteRow": "Supprimer la ligne", + "insertColumnLeft": "Insérer une colonne à gauche", + "insertColumnRight": "Insérer une colonne à droite", + "deleteColumn": "Supprimer la colonne", + "deleteTable": "Supprimer le tableau" + } + } + } + }, + "components": { + "native-chat": { + "composer": { + "imageUnsupported": "Le collage d'image n'est pas pris en charge pour cet agent.", + "send": "Envoyer", + "noPty": "Aucun terminal actif — rebasculez pour vous reconnecter.", + "locked": "La saisie est détenue par un autre appareil.", + "placeholder": "Envoyer un message…", + "mentionHint": "Fichier référencé :", + "attach": "Joindre un fichier", + "startDictation": "Démarrer la dictée", + "stopDictation": "Arrêter la dictée", + "noSkills": "Aucune skill correspondante", + "noCommandsOrSkills": "Aucune commande ou skill correspondante", + "noCommands": "Aucune commande correspondante", + "commands": "Commandes", + "skills": "Skills", + "loadingSkills": "Chargement des skills...", + "skillsLoaded": "Skills chargées", + "skillsUnavailableHost": "Les skills sont indisponibles pour cet hôte", + "skillsLoadFailed": "Impossible de charger les skills depuis cet hôte", + "retrySkills": "Réessayer", + "skillCommandCollision": "Aussi un nom de skill - l'agent décide", + "skillMultipleSources": "{{sourceCount}} sources - l'agent décide", + "skillScopeProject": "Projet", + "skillScopePersonal": "Personnel", + "skillScopeBuiltIn": "Intégrée", + "skillScopePlugin": "Plugin", + "localAttachmentUnsupported": "Les pièces jointes locales ne sont pas disponibles pour les sessions distantes.", + "removeAttachment": "Retirer la pièce jointe", + "pastedImageLabel": "Image collée", + "imagePasteFailed": "Échec du collage de l'image.", + "worktreeNotReady": "Worktree pas encore prêt — réessayez dans un instant.", + "uploadingAttachments": "Envoi de {{value0}} fichier(s) vers le distant…", + "model": "Modèle", + "effort": "Effort", + "fastMode": "Mode rapide", + "thinking": "Réflexion", + "options": "Options", + "sessionOptions": "Options de session", + "chooseInAgentPicker": "Choisir dans le sélecteur d'agents…", + "toggleOption": "Basculer {{value0}}", + "pillAccessibleName": "{{value0}} {{value1}}", + "valueUnknown": "Valeur actuelle inconnue — choisissez Activé ou Désactivé", + "sentNotConfirmed": "Envoyé à l'agent — non confirmé", + "setWhenSessionStarts": "Défini au démarrage de la session.", + "availableAfterSessionStarts": "Disponible après le démarrage de la session.", + "optionUpdateFailed": "Impossible de mettre à jour l'option", + "optionValue": { + "fast": "Rapide", + "minimal": "Minimal", + "low": "Basse", + "medium": "Moyenne", + "high": "Haute", + "xhigh": "Extra élevé", + "max": "Max", + "ultra": "Ultra", + "on": "Activé", + "off": "Désactivé" + } + }, + "tool": { + "running": "En cours…", + "result": "Résultat", + "countOne": "1 appel d'outil", + "countN": "{{value0}} appels d'outils" + }, + "status": { + "responding": "L'agent répond" + }, + "jumpToLatest": "Aller au dernier message", + "toggle": { + "showTerminal": "Afficher le terminal", + "showChat": "Afficher la vue chat" + }, + "state": { + "loading": { + "title": "Chargement de la conversation…", + "subtitle": "Lecture de la transcription de l'agent." + }, + "error": { + "title": "Impossible de charger la conversation", + "subtitle": "La transcription n'a pas pu être lue. Rebasculez sur le terminal pour continuer à travailler." + }, + "pairHost": "Associez un hôte pour consulter l'historique de chat des agents.", + "notAgent": { + "title": "Aucune conversation ici", + "subtitle": "Ce terminal n'exécute aucun agent de code reconnu." + }, + "empty": { + "title": "Démarrer un chat avec {{value0}}", + "subtitle": "Demandez à {{value0}} d'inspecter du code, d'expliquer une sortie ou de faire une modification." + } + }, + "stop": "Arrêter l'agent", + "copyMessage": { + "copied": "Copied", + "copy": "Copier le message" + }, + "scrollMessageToTop": "Revenir au début du message", + "loadingEarlier": "Chargement…", + "loadEarlier": "Charger les messages précédents", + "question": { + "step": "Étape {{value0}}", + "other": "Autre…", + "otherPlaceholder": "Saisissez votre réponse", + "cancel": "Annuler", + "send": "Valider", + "next": "Suivant", + "skip": "Ignorer", + "sending": "Envoi…" + }, + "approval": { + "title": "Autoriser {{value0}} ?", + "allow": "Autoriser", + "deny": "Refuser" + }, + "launchPromptNotDelivered": "Non livré — vérifiez le terminal" + }, + "tab": { + "bar": { + "SortableTabContextMenu": { + "switchToTerminalView": "Passer à la vue terminal", + "switchToChatView": "Passer à la vue chat", + "closeTabsToLeft": "Fermer les onglets à gauche" + }, + "BrowserTab": { + "closeOthers": "Fermer les autres", + "closeTabsToLeft": "Fermer les onglets à gauche" + }, + "EditorFileTabContextMenu": { + "closeOthers": "Fermer les autres", + "closeTabsToLeft": "Fermer les onglets à gauche" + } + } + }, + "workspace": { + "cleanup": { + "presentationFixtures": { + "reviewAlphaCleanup": "Nettoyage alpha de la revue" + }, + "presentation": { + "gitlabMergeRequestNumber": "MR #{{value0}}", + "githubPullRequestNumber": "PR #{{value0}}" + }, + "browse": { + "noSizes": "Aucune taille d'espace de travail mesurée pour l'instant.", + "noSizesDescription": "Les tailles apparaissent une fois l'utilisation disque mesurée.", + "measureSizes": "Analyser", + "measureSizesTitle": "Analyser les tailles des espaces de travail", + "measureSizesDescription": "Analysez l'utilisation disque pour comparer, trier et filtrer les espaces de travail par taille.", + "sizeOnDisk": "Taille sur disque : {{value0}}", + "checkingGitRow": "Vérification de l'état git", + "searchPlaceholder": "Rechercher par nom, dépôt, branche, chemin, hôte", + "searchLabel": "Rechercher des espaces de travail", + "filters": "Filtres", + "showingCount": "Affichage de {{value0}} sur {{value1}}", + "clearFilters": "Réinitialiser les filtres", + "checkingGit": "Vérification de l'état git : {{value0}} restants", + "facet": { + "git": "Git", + "review": "À examiner", + "ticket": "Tickets", + "location": "Emplacement", + "safety": "Sécurité", + "activity": "Activité", + "size": "Taille sur disque", + "status": "Statut de l'espace de travail", + "agent": "Agent", + "context": "Contexte local" + }, + "minAhead": "Commits d'avance ≥", + "minBehind": "Commits de retard ≥", + "branchQuery": "La branche contient", + "prunable": "Nettoyable", + "locked": "Verrouillé", + "reviewPresence": "PR / MR", + "reviewProvider": "Fournisseur", + "ticketPresence": "Ticket lié", + "host": "Hôte", + "repo": "Dépôt", + "pathPrefix": "Chemin commençant par", + "tier": { + "ready": "Prêt", + "review": "En attente de relecture", + "protected": "Protégé" + }, + "blockerMode": "Correspondance de bloqueur", + "blockerModeAny": "Au moins un", + "blockerModeNone": "Aucun", + "blockers": "Bloqueurs", + "dismissed": "Ignoré", + "selectableOnly": "Uniquement les espaces de travail que je peux supprimer maintenant", + "idleSignal": { + "lastVisited": "Non ouvert", + "lastActivity": "Aucune activité", + "created": "Créé avant" + }, + "idleMinDays": "Inactif depuis au moins", + "days": "jours", + "neverVisited": "Jamais ouvert", + "minSize": "Au moins", + "maxSize": "Au plus", + "includeUnsized": "Inclure les espaces de travail non mesurés", + "matchStatusless": "Inclure les espaces de travail sans statut", + "archived": "Archivé", + "pinned": "Épinglés", + "unread": "Non lus", + "comment": "A un commentaire", + "retainedDoneAgents": "Transcriptions d'agents terminés", + "contextPresence": "Onglets, terminaux, commentaires ouverts", + "completelyEmpty": "Plus rien à perdre", + "noWorkspaces": "Aucun espace de travail trouvé.", + "noMatches": "Aucun espace de travail ne correspond à ces filtres.", + "noMatchesDescription": "Tous les espaces de travail sont dans une même liste — élargissez une facette ou effacez les filtres.", + "selectAll": "Sélectionner tous les espaces de travail correspondants", + "sortBy": "Trier par", + "sortByField": "Trier par {{value0}}", + "sort": { + "lastActivity": "Dernière activité", + "lastVisited": "Dernière ouverture", + "created": "Créé", + "size": "Taille", + "name": "Espace de travail", + "repo": "Dépôt", + "path": "Chemin", + "host": "Hôte", + "workspaceStatus": "Statut", + "agent": "Agent", + "git": "Git", + "ahead": "En avance", + "behind": "En retard", + "branch": "Branche", + "review": "À examiner", + "ticket": "Ticket", + "localContext": "Contexte ouvert", + "tier": "Sécurité", + "blockerCount": "Bloqueurs" + }, + "git": { + "clean": "Propre", + "dirty": "Modifications non validées", + "unpushed": "Commits non poussés", + "unknown": "Non vérifié" + }, + "agent": { + "working": "En activité", + "permission": "Vous attend", + "idle": "Inactif" + }, + "review": { + "open": "Ouvert", + "draft": "Brouillon", + "merged": "Fusionnés", + "closed": "Fermé", + "unknown": "Inconnu", + "otherProvider": "Autre" + }, + "ticket": { + "workItem": "Élément de travail", + "linear": "Linear", + "issue": "Issue" + }, + "triState": { + "any": "Tous", + "only": "Uniquement", + "exclude": "Exclure" + }, + "presence": { + "any": "Tous", + "some": "Un", + "none": "Aucun" + }, + "tierField": "Palier de nettoyage", + "idleSignalField": "Signal d'inactivité", + "selectionVanishedOne": "1 espace de travail sélectionné n'existe plus.", + "selectionVanished": "{{value0}} espaces de travail sélectionnés n'existent plus.", + "updatedAgo": "{{value0}} mis à jour", + "refreshing": "Actualisation…", + "refreshingProgress": "Actualisation {{value0}}/{{value1}}", + "notMeasured": "Non mesuré", + "openWorkspaceNamed": "Ouvrir {{value0}}", + "openWorkspace": "Ouvrir l'espace de travail", + "workspaceStatus": "Statut de l'espace de travail : {{value0}}", + "measuringSizesProgress": "Analyse {{value0}}/{{value1}}", + "measuringSizes": "Analyse des tailles", + "measureSizesFailed": "Impossible d'analyser les tailles des espaces de travail", + "selectedCount": "{{value0}} sélectionnés", + "gitStatusCheckFailed": "Échec de la vérification de l'état git", + "gitStatusUnverified": "L'état git n'a pas pu être vérifié", + "deleteAnyway": "Supprimer quand même", + "forceDeleteProjectionOne": "{{count}} espace de travail présente actuellement un risque et peut nécessiter une suppression forcée", + "forceDeleteProjectionMany": "{{count}} espaces de travail présentent actuellement un risque et peuvent nécessiter une suppression forcée", + "selectionWithheldOne": "1 espace de travail sélectionné est masqué par les filtres actuels et a été désélectionné.", + "selectionWithheld": "{{value0}} espaces de travail sélectionnés sont masqués par les filtres actuels et ont été désélectionnés.", + "selectAllCountOne": "Sélectionner 1 espace de travail vérifié côté sécurité", + "selectAllCount": "Sélectionner les {{value0}} espaces de travail vérifiés côté sécurité", + "appliedFilters": "Filtres appliqués", + "removeFilter": "Supprimer le filtre {{value0}}", + "chip": { + "idleDays": "Inactif {{value0}} j+", + "neverVisited": "Jamais visité", + "minSize": "Au moins {{value0}} Mo", + "maxSize": "Au plus {{value0}} Mo", + "excludesUnsized": "Mesurés uniquement", + "excludesStatusless": "A un statut", + "list": "{{value0}} : {{value1}}", + "triState": "{{value0}} : {{value1}}", + "only": "Uniquement", + "exclude": "Exclus", + "minAhead": "{{value0}}+ d'avance", + "minBehind": "{{value0}}+ de retard", + "branchQuery": "Branche : {{value0}}", + "pathPrefix": "Chemin : {{value0}}", + "presence": "{{value0}} : {{value1}}", + "has": "A", + "none": "Aucun", + "completelyEmpty": "Rien à perdre", + "selectableOnly": "Supprimables uniquement", + "kind": { + "status": "Statut", + "agent": "Agent", + "git": "Git", + "review": "À examiner", + "reviewState": "État de revue", + "reviewProvider": "Fournisseur", + "ticket": "Ticket", + "ticketSource": "Source des tickets", + "context": "Contexte", + "host": "Hôte", + "repo": "Dépôt", + "blocker": "Bloqueur", + "tier": "Sécurité", + "dismissed": "Ignoré", + "archived": "Archivé", + "pinned": "Épinglés", + "unread": "Non lus", + "comment": "Commentaire", + "prunable": "Nettoyable", + "locked": "Verrouillé", + "retainedAgents": "Agents terminés" + } + } + }, + "scan": { + "progress": "Analyse des espaces de travail ({{value0}}/{{value1}})", + "singleError": "Impossible de vérifier {{value0}} : {{value1}}. Certains espaces de travail peuvent manquer. Actualisez pour réessayer.", + "moreErrors": ", +{{value0}} autres", + "multipleErrors": "Impossible de vérifier les dépôts {{value0}} ({{value1}}{{value2}}). Certains espaces de travail peuvent manquer. Actualisez pour réessayer.", + "noWorkspaces": "Aucun espace de travail trouvé.", + "readyOneOne": "1 espace de travail trouvé, avec 1 suggestion de nettoyage.", + "readyOneMany": "1 espace de travail trouvé, avec {{value0}} suggestions de nettoyage.", + "readyManyOne": "{{value0}} espaces de travail trouvés, avec 1 suggestion de nettoyage.", + "readyManyMany": "{{value0}} espaces de travail trouvés, avec {{value1}} suggestions de nettoyage.", + "fallbackRepository": "un dépôt", + "gitListFailed": "Git n'a pas pu lister les worktrees", + "readyOne": "1 espace de travail trouvé.", + "readyMany": "{{value0}} espaces de travail trouvés." + }, + "relativeTime": { + "never": "Jamais", + "justNow": "À l'instant", + "minutesAgo": "il y a {{value0}} min", + "hoursAgo": "il y a {{value0}} h", + "daysAgo": "il y a {{value0}} j" + }, + "host": { + "label": "Hôte : {{value0}}", + "unknown": "Hôte inconnu", + "candidateName": "{{value0}} sur {{value1}}" + } + } + }, + "agentSessionContinuation": { + "continueInNewSession": "Continuer dans une nouvelle session…", + "dialogTitle": "Continuer dans une nouvelle session", + "dialogDescription": "Démarrez une nouvelle session d'Agent depuis ce point d'arrêt. La session d'origine reste inchangée.", + "untitledSession": "Session actuelle", + "originalAgent": "Agent d'origine : {{agent}}", + "agent": "Agent", + "selectAgent": "Sélectionner un Agent", + "detectingAgents": "Détection des Agents sur cet hôte d'espace de travail…", + "detectionFailed": "Impossible de détecter des Agents sur cet hôte d'espace de travail.", + "noAgents": "Aucun Agent activé n'a été détecté sur cet hôte d'espace de travail.", + "context": "Contexte", + "modeFocused": "Transfert ciblé (recommandé)", + "modeFocusedDescription": "Utilise le statut le plus récent et l'espace de travail actuel, en ne lisant les détails anciens de la transcription qu'en cas de besoin.", + "modeFull": "Transcription complète de la session", + "modeFullDescription": "Demande au nouvel Agent de lire la session sauvegardée complète avant de continuer. Cela peut prendre plus de temps et consommer beaucoup de contexte, d'utilisation du plan ou de crédits API.", + "startsIn": "Démarre dans :", + "startSession": "Démarrer une nouvelle session", + "starting": "Démarrage…", + "noContext": "Aucun contexte de session disponible pour continuer dans une nouvelle session.", + "agentDisabled": "{{agent}} est désactivé dans les paramètres des Agents.", + "agentUnavailable": "{{agent}} n'a pas été détecté sur cet hôte d'espace de travail.", + "sent": "Contexte de session envoyé à {{agent}} dans une nouvelle session.", + "deliveryFailed": "La nouvelle session {{agent}} a démarré, mais son contexte n'a pas pu être envoyé.", + "launchFailed": "Impossible de démarrer une nouvelle session {{agent}}." + }, + "status": { + "bar": { + "workspaceSpace": { + "otherTopLevelItems": "Autres éléments de premier niveau ({{value0}})" + } + } + }, + "cmd-j": { + "paletteSessionAge": { + "minutes": "{{value0}} min", + "hours": "{{value0}} h", + "days": "{{value0}} j", + "underOneMinute": "<1 min" + } + } + }, + "dashboardPopout": { + "view": { + "label": "Vue tableau de bord", + "board": "Tableau de bord", + "map": "Carte des agents" + }, + "map": { + "host": { + "local": "Local", + "ssh": "SSH", + "wsl": "WSL", + "remote": "Distant" + }, + "filters": { + "showStates": "États des agents", + "agentlessWorkspaces": "Espaces de travail sans agents", + "orchestrationLinks": "Liens d'orchestration", + "orchestrationLinksHidden": "Liens d'orchestration masqués", + "workspaceVisibility": "Contenu de la carte", + "title": "Contrôles de la carte", + "reset": "Réinitialiser", + "ofTotalAgents": "sur {{total}} agents affichés", + "quickViews": "Vues rapides", + "agents": "Agents", + "time": "Heure", + "lifespan": "Durée de vie de la session", + "sinceMessage": "Depuis le dernier message", + "timeInState": "Temps dans l'état actuel", + "timeMinimum": "{{label}} minimum", + "timeMaximum": "{{label}} maximum", + "timeAny": "tous", + "timeRangeCount": "{{count}} plages", + "resetRanges": "Réinitialiser les plages", + "summaryAll": "Tous", + "summaryCount": "{{shown}} sur {{total}}", + "summarySelected": "{{count}} sélectionnés", + "workspace": "Espace de travail", + "stateChip": "État : {{states}}" + }, + "liveContainmentMap": "Carte d'imbrication en temps réel", + "empty": "Aucun agent ne correspond aux filtres actuels.", + "canvasLabel": "Carte imbriquée des projets, espaces de travail et agents", + "zoomOut": "Zoom arrière", + "zoomIn": "Zoom avant", + "fit": "Ajuster", + "openWorktree": "Ouvrir les détails du worktree {{worktree}}", + "openFolderWorkspace": "Ouvrir les détails de l'espace de travail de type dossier {{workspace}}", + "worktreeSummary": "{{total}} agents · {{active}} actifs · {{done}} terminés", + "worktreeSummary_one": "{{total}} agent · {{active}} actif · {{done}} terminé", + "worktreeSummary_other": "{{total}} agents · {{active}} actifs · {{done}} terminés", + "runningAgents": "Agents", + "spawnAgent": "Démarrer un nouvel agent", + "noLaunchableAgents": "Aucun agent activé détecté.", + "sleepWorkspace": "Veille", + "projectCount": "{{agents}} agents · {{workspaces}} workspaces", + "agentCount": "{{count}} agents", + "agentCount_one": "{{count}} agent", + "agentCount_other": "{{count}} agents", + "noWorkspaceAgents": "Aucun agent dans cet espace de travail.", + "status": { + "doneSeen": "Terminé, vu" + }, + "quickView": { + "everything": "Tout", + "attention": "Me concerne", + "stuck": "Bloqué", + "unread": "Non lus", + "recent": "30 dernières minutes", + "longRunning": "Exécutions longues", + "stale": "Inactif > 3 j", + "orchestration": "Orchestration" + } + }, + "placeholder": { + "title": "Tableau de bord des agents", + "description": "C'est ici que tous vos agents s'afficheront d'un coup d'œil. Le tableau arrive bientôt." + }, + "recoverableError": { + "title": "Le tableau de bord Orca a rencontré une erreur.", + "description": "Le tableau de bord n'a pas pu terminer son rendu. Réessayez pour le recharger, ou rouvrez-le." + }, + "bucket": { + "attention": "Vous concerne", + "working": "En activité", + "idle": "Inactif", + "empty": "Aucun", + "done": "Terminé" + }, + "title": "Agents", + "total": "{{count}} au total", + "close": "Fermer le tableau de bord", + "settings": { + "showIdle": "Afficher les agents inactifs", + "showIdleCopy": "Inclure les agents restés silencieux pendant 30 minutes sans signaler leur achèvement. Masqués par défaut." + }, + "settingsTooltip": "Paramètres du tableau", + "card": { + "you": "Vous", + "time": { + "justNow": "à l'instant", + "minutes": "{{count}} m", + "hours": "{{count}} h", + "days": "{{count}} j" + }, + "review": { + "open": "Ouvrir la review", + "draft": "Revue de brouillon", + "merged": "Review fusionnée", + "closed": "Review fermée" + }, + "subagents_one": "{{count}} sous-agent", + "subagents_other": "{{count}} sous-agents" + }, + "terminal": { + "closed": "Pas de terminal en direct — le volet de cet agent est fermé.", + "focusWorktree": "Ouvrir le worktree", + "close": "Fermer" + }, + "filters": { + "remove": "Retirer le filtre {{label}}", + "active": "Filtres", + "clear": "Effacer", + "review": { + "open": "Ouvert", + "draft": "Brouillon", + "merged": "Fusionnés", + "closed": "Fermé", + "none": "Pas de review" + }, + "reviewChip": "Review : {{state}}", + "label": "Filtre", + "project": "Projet", + "workspaceStatus": "Statut de l'espace de travail", + "reviewStatus": "Statut PR / MR", + "clearAll": "Effacer tous les filtres", + "removeChip": "Retirer {{filter}}" + }, + "search": { + "placeholder": "Rechercher un worktree, un projet ou un agent…", + "label": "Rechercher des agents", + "clear": "Effacer la recherche", + "results": "{{shown}} sur {{total}} affichés" + }, + "settingsLabel": "Paramètres du tableau de bord des agents", + "host": { + "sshNamed": "Hôte SSH · {{host}}", + "ssh": "Hôte SSH", + "remoteNamed": "Hôte Orca distant · {{host}}", + "remote": "Hôte Orca distant" + } + }, + "dashboard": { + "sidebar": { + "label": "Tableau de bord des agents" + } + }, + "runtimeRpc": { + "startupFailure": { + "unknownCause": "Aucun détail supplémentaire sur l'erreur n'était disponible.", + "continueButton": "Continuer sans CLI", + "title": "CLI Orca indisponible", + "message": "Orca n'a pas pu démarrer son transport de commandes local.", + "detail": "Orca continuera de fonctionner, mais les commandes telles que orca status, orca terminal et l'orchestration sont indisponibles pour cette session.\n\n{{guidance}}\n\nCause : {{cause}}", + "guidance": { + "permissionDenied": "Orca n'a pas pu écrire son fichier runtime. Vérifiez les permissions du dossier de données d'Orca, puis redémarrez.", + "storageUnavailable": "Votre disque est peut-être plein ou en lecture seule. Libérez de l'espace, puis redémarrez Orca.", + "invalidPath": "Le dossier de données d'Orca est peut-être manquant, déplacé, ou situé dans un chemin trop long. Restaurez-le ou utilisez un chemin plus court, puis redémarrez Orca.", + "addressInUse": "Un autre processus occupe peut-être le port. Redémarrez Orca pour réessayer.", + "unknown": "Redémarrez Orca pour réessayer." + } + } + }, + "featureTips": { + "voice": { + "demoPrompt": "Relisez ce diff à la recherche de cas limites et ajoutez des tests pour tout ce que vous trouvez.", + "startDictation": "Démarrer la dictée", + "agentPromptTitle": "Prompt d'agent", + "listening": "Écoute...", + "focusPaneInstruction": "Placez le focus sur un terminal, un éditeur ou un prompt d'agent, puis appuyez sur", + "startInstruction": "pour lancer la dictée vocale. Appuyez de nouveau sur", + "stopInstruction": "pour arrêter.", + "unassignedInstruction": "Assignez un raccourci de dictée avant de lancer la dictée vocale dans un volet actif.", + "settingsInstruction": "Modifiez le modèle, le mode de dictée ou le raccourci à tout moment dans", + "settingsLink": "Paramètres → Voix" + } + }, + "quickOpen": { + "moreMatchesAvailable": "D'autres correspondances sont peut-être disponibles. Affinez votre recherche pour réduire les résultats." + } +} diff --git a/src/renderer/src/i18n/locales/ja.json b/src/renderer/src/i18n/locales/ja.json index d3af25ad697..b3c30da9446 100644 --- a/src/renderer/src/i18n/locales/ja.json +++ b/src/renderer/src/i18n/locales/ja.json @@ -20,7 +20,8 @@ "chinese": "中文(简体)", "korean": "한국어", "japanese": "日本語", - "spanish": "Español" + "spanish": "Español", + "french": "Français" }, "statusBar": { "claudeToggleDescription": "アクティブなワークスペースの Claude トークンとコストの使用状況を表示します。", diff --git a/src/renderer/src/i18n/locales/ko.json b/src/renderer/src/i18n/locales/ko.json index f494ff29a58..ac75cca2209 100644 --- a/src/renderer/src/i18n/locales/ko.json +++ b/src/renderer/src/i18n/locales/ko.json @@ -20,7 +20,8 @@ "chinese": "中文(简体)", "korean": "한국어", "japanese": "日本語", - "spanish": "Español" + "spanish": "Español", + "french": "Français" }, "statusBar": { "claudeToggleDescription": "활성 워크스페이스에 대한 Claude 토큰 및 비용 사용량을 표시합니다.", diff --git a/src/renderer/src/i18n/locales/zh.json b/src/renderer/src/i18n/locales/zh.json index 2bbaee9d1bb..7a3b47c8f2a 100644 --- a/src/renderer/src/i18n/locales/zh.json +++ b/src/renderer/src/i18n/locales/zh.json @@ -20,7 +20,8 @@ "chinese": "中文(简体)", "korean": "한국어", "japanese": "日本語", - "spanish": "Español" + "spanish": "Español", + "french": "Français" }, "statusBar": { "claudeToggleDescription": "显示当前工作区的 Claude Token 与费用消耗情况。", diff --git a/src/renderer/src/i18n/supported-languages.ts b/src/renderer/src/i18n/supported-languages.ts index 446a3a0419c..eee26df0e5f 100644 --- a/src/renderer/src/i18n/supported-languages.ts +++ b/src/renderer/src/i18n/supported-languages.ts @@ -2,6 +2,7 @@ import { DEFAULT_UI_LOCALE, resolveRendererUiLocale } from '../../../shared/ui-l import { UI_LANGUAGE_CHINESE, UI_LANGUAGE_ENGLISH, + UI_LANGUAGE_FRENCH, UI_LANGUAGE_JAPANESE, UI_LANGUAGE_KOREAN, UI_LANGUAGE_SPANISH, @@ -25,7 +26,8 @@ export const UI_LANGUAGE_CHOICES: UiLanguageChoice[] = [ { value: UI_LANGUAGE_CHINESE, labelKey: 'settings.appearance.language.chinese' }, { value: UI_LANGUAGE_KOREAN, labelKey: 'settings.appearance.language.korean' }, { value: UI_LANGUAGE_JAPANESE, labelKey: 'settings.appearance.language.japanese' }, - { value: UI_LANGUAGE_SPANISH, labelKey: 'settings.appearance.language.spanish' } + { value: UI_LANGUAGE_SPANISH, labelKey: 'settings.appearance.language.spanish' }, + { value: UI_LANGUAGE_FRENCH, labelKey: 'settings.appearance.language.french' } ] const UI_LANGUAGE_CHOICE_FALLBACKS: Record = { @@ -34,7 +36,8 @@ const UI_LANGUAGE_CHOICE_FALLBACKS: Record = { [UI_LANGUAGE_CHINESE]: '中文(简体)', [UI_LANGUAGE_KOREAN]: '한국어', [UI_LANGUAGE_JAPANESE]: '日本語', - [UI_LANGUAGE_SPANISH]: 'Español' + [UI_LANGUAGE_SPANISH]: 'Español', + [UI_LANGUAGE_FRENCH]: 'Français' } export function getUiLanguageChoiceLabel( diff --git a/src/shared/ui-language.test.ts b/src/shared/ui-language.test.ts index b27ab0c8312..cad48bcf396 100644 --- a/src/shared/ui-language.test.ts +++ b/src/shared/ui-language.test.ts @@ -3,6 +3,7 @@ import { describe, expect, it } from 'vitest' import { UI_LANGUAGE_CHINESE, UI_LANGUAGE_ENGLISH, + UI_LANGUAGE_FRENCH, UI_LANGUAGE_JAPANESE, UI_LANGUAGE_KOREAN, UI_LANGUAGE_SPANISH, @@ -18,10 +19,11 @@ describe('normalizeUiLanguage', () => { expect(normalizeUiLanguage(UI_LANGUAGE_KOREAN)).toBe('ko') expect(normalizeUiLanguage(UI_LANGUAGE_JAPANESE)).toBe('ja') expect(normalizeUiLanguage(UI_LANGUAGE_SPANISH)).toBe('es') + expect(normalizeUiLanguage(UI_LANGUAGE_FRENCH)).toBe('fr') }) it('falls back unknown values to system', () => { - expect(normalizeUiLanguage('fr')).toBe('system') + expect(normalizeUiLanguage('de')).toBe('system') expect(normalizeUiLanguage(null)).toBe('system') }) }) diff --git a/src/shared/ui-language.ts b/src/shared/ui-language.ts index 77002e81f89..0f60b9aaba5 100644 --- a/src/shared/ui-language.ts +++ b/src/shared/ui-language.ts @@ -4,6 +4,7 @@ export const UI_LANGUAGE_CHINESE = 'zh' export const UI_LANGUAGE_KOREAN = 'ko' export const UI_LANGUAGE_JAPANESE = 'ja' export const UI_LANGUAGE_SPANISH = 'es' +export const UI_LANGUAGE_FRENCH = 'fr' export type BuiltInUiLanguage = | typeof UI_LANGUAGE_SYSTEM @@ -12,6 +13,7 @@ export type BuiltInUiLanguage = | typeof UI_LANGUAGE_KOREAN | typeof UI_LANGUAGE_JAPANESE | typeof UI_LANGUAGE_SPANISH + | typeof UI_LANGUAGE_FRENCH export type PluginUiLanguage = `plugin:${string}` export type UiLanguage = BuiltInUiLanguage | PluginUiLanguage @@ -22,7 +24,8 @@ const UI_LANGUAGE_VALUES = new Set([ UI_LANGUAGE_CHINESE, UI_LANGUAGE_KOREAN, UI_LANGUAGE_JAPANESE, - UI_LANGUAGE_SPANISH + UI_LANGUAGE_SPANISH, + UI_LANGUAGE_FRENCH ]) const PLUGIN_UI_LANGUAGE_RE = diff --git a/src/shared/ui-locale.test.ts b/src/shared/ui-locale.test.ts index 272506e26a9..3c2de4a262d 100644 --- a/src/shared/ui-locale.test.ts +++ b/src/shared/ui-locale.test.ts @@ -4,6 +4,7 @@ import { normalizeSupportedUiLocale, resolveUiLocale, resolveRendererUiLocale } import { UI_LANGUAGE_CHINESE, UI_LANGUAGE_ENGLISH, + UI_LANGUAGE_FRENCH, UI_LANGUAGE_JAPANESE, UI_LANGUAGE_KOREAN, UI_LANGUAGE_SPANISH, @@ -34,8 +35,14 @@ describe('ui-locale', () => { expect(normalizeSupportedUiLocale('es')).toBe('es') }) + it('normalizes French locale prefixes', () => { + expect(normalizeSupportedUiLocale('fr-FR')).toBe('fr') + expect(normalizeSupportedUiLocale('fr-CA')).toBe('fr') + expect(normalizeSupportedUiLocale('fr')).toBe('fr') + }) + it('falls back unsupported locales to English', () => { - expect(normalizeSupportedUiLocale('fr-FR')).toBe('en') + expect(normalizeSupportedUiLocale('de-DE')).toBe('en') }) it('does not map Traditional Chinese to Simplified yet', () => { @@ -64,6 +71,10 @@ describe('ui-locale', () => { expect(resolveUiLocale(UI_LANGUAGE_SPANISH, 'en-US')).toBe('es') }) + it('resolves explicit French independently of system locale', () => { + expect(resolveUiLocale(UI_LANGUAGE_FRENCH, 'en-US')).toBe('fr') + }) + it('preserves a selected plugin language bundle id', () => { expect(resolveUiLocale('plugin:orca-samples.portuguese/pt-BR')).toBe( 'plugin:orca-samples.portuguese/pt-BR' @@ -76,7 +87,7 @@ describe('ui-locale', () => { expect(resolveUiLocale(UI_LANGUAGE_SYSTEM, 'ko-KR')).toBe('ko') expect(resolveUiLocale(UI_LANGUAGE_SYSTEM, 'ja-JP')).toBe('ja') expect(resolveUiLocale(UI_LANGUAGE_SYSTEM, 'es-MX')).toBe('es') - expect(resolveUiLocale(UI_LANGUAGE_SYSTEM, 'fr-FR')).toBe('en') + expect(resolveUiLocale(UI_LANGUAGE_SYSTEM, 'fr-FR')).toBe('fr') }) it('uses renderer system locale only for the system setting', () => { @@ -85,5 +96,6 @@ describe('ui-locale', () => { expect(resolveRendererUiLocale(UI_LANGUAGE_KOREAN)).toBe('ko') expect(resolveRendererUiLocale(UI_LANGUAGE_JAPANESE)).toBe('ja') expect(resolveRendererUiLocale(UI_LANGUAGE_SPANISH)).toBe('es') + expect(resolveRendererUiLocale(UI_LANGUAGE_FRENCH)).toBe('fr') }) }) diff --git a/src/shared/ui-locale.ts b/src/shared/ui-locale.ts index bc84491b7b2..70bffcca8c8 100644 --- a/src/shared/ui-locale.ts +++ b/src/shared/ui-locale.ts @@ -1,6 +1,7 @@ import { UI_LANGUAGE_CHINESE, UI_LANGUAGE_ENGLISH, + UI_LANGUAGE_FRENCH, UI_LANGUAGE_JAPANESE, UI_LANGUAGE_KOREAN, UI_LANGUAGE_SPANISH, @@ -9,7 +10,7 @@ import { type UiLanguage } from './ui-language' -export const SUPPORTED_UI_LOCALES = ['en', 'zh', 'ko', 'ja', 'es'] as const +export const SUPPORTED_UI_LOCALES = ['en', 'zh', 'ko', 'ja', 'es', 'fr'] as const export type SupportedUiLocale = (typeof SUPPORTED_UI_LOCALES)[number] export const DEFAULT_UI_LOCALE: SupportedUiLocale = 'en' @@ -54,6 +55,9 @@ export function resolveUiLocale( if (language === UI_LANGUAGE_SPANISH) { return 'es' } + if (language === UI_LANGUAGE_FRENCH) { + return 'fr' + } return normalizeSupportedUiLocale(systemLocale) } From 963839aa4f3cbd932dac3229364b115a77bfb620 Mon Sep 17 00:00:00 2001 From: Ihor Date: Fri, 4 Sep 2026 03:33:02 +0300 Subject: [PATCH 214/398] docs: add Ukrainian README translation iho <4000375+iho@users.noreply.github.com> --- README.md | 2 +- docs/readme/README.es.md | 2 +- docs/readme/README.fr.md | 2 +- docs/readme/README.ja.md | 2 +- docs/readme/README.ko.md | 2 +- docs/readme/README.pt.md | 2 +- docs/readme/README.uk.md | 269 ++++++++++++++++++++++++++++++++++++ docs/readme/README.zh-CN.md | 2 +- 8 files changed, 276 insertions(+), 7 deletions(-) create mode 100644 docs/readme/README.uk.md diff --git a/README.md b/README.md index 7a3cbe2360c..805c94295cc 100644 --- a/README.md +++ b/README.md @@ -12,7 +12,7 @@

    - 中文 · 日本語 · 한국어 · Español · Français · Português + 中文 · 日本語 · 한국어 · Español · Français · Português · Українська

    diff --git a/docs/readme/README.es.md b/docs/readme/README.es.md index f2247e0900d..c8333ec31c9 100644 --- a/docs/readme/README.es.md +++ b/docs/readme/README.es.md @@ -12,7 +12,7 @@

    - English · 中文 · 日本語 · 한국어 · Français · Português + English · 中文 · 日本語 · 한국어 · Français · Português · Українська

    diff --git a/docs/readme/README.fr.md b/docs/readme/README.fr.md index 97c78d4e713..7a69e1acffe 100644 --- a/docs/readme/README.fr.md +++ b/docs/readme/README.fr.md @@ -12,7 +12,7 @@

    - English · 中文 · 日本語 · 한국어 · Español · Português + English · 中文 · 日本語 · 한국어 · Español · Português · Українська

    diff --git a/docs/readme/README.ja.md b/docs/readme/README.ja.md index cce2032a67c..a256f5239c4 100644 --- a/docs/readme/README.ja.md +++ b/docs/readme/README.ja.md @@ -12,7 +12,7 @@

    - English · 中文 · 한국어 · Español · Français · Português + English · 中文 · 한국어 · Español · Français · Português · Українська

    diff --git a/docs/readme/README.ko.md b/docs/readme/README.ko.md index 4a75722ff8c..d2dd2a62747 100644 --- a/docs/readme/README.ko.md +++ b/docs/readme/README.ko.md @@ -12,7 +12,7 @@

    - English · 中文 · 日本語 · Español · Français · Português + English · 中文 · 日本語 · Español · Français · Português · Українська

    diff --git a/docs/readme/README.pt.md b/docs/readme/README.pt.md index 86d998a4e5f..667cebb685f 100644 --- a/docs/readme/README.pt.md +++ b/docs/readme/README.pt.md @@ -12,7 +12,7 @@

    - English · 中文 · 日本語 · 한국어 · Español · Français + English · 中文 · 日本語 · 한국어 · Español · Français · Українська

    diff --git a/docs/readme/README.uk.md b/docs/readme/README.uk.md new file mode 100644 index 00000000000..b3e2a42b868 --- /dev/null +++ b/docs/readme/README.uk.md @@ -0,0 +1,269 @@ +

    + Orca Orca +

    + +

    + Зірки на GitHub + Загальна кількість завантажень усіх релізів + Ліцензія: MIT + Приєднатися до Discord Orca + Стежити за Orca в X + Підтримувані платформи: macOS, Windows і Linux +

    + +

    + English · 中文 · 日本語 · 한국어 · Español · Français · Português +

    + +

    + AI-оркестратор для розробників рівня 100x.
    + Запускайте Codex, Claude Code, OpenCode або Pi паралельно — кожен у власному worktree, усі під контролем в одному місці. +

    + +

    Завантажити Orca

    + +

    + Десктопний застосунок Orca запускає агентів у паралельних worktree, у кутку — супутній мобільний застосунок Orca +

    + +## Можливості + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
    + +### Супутній мобільний застосунок + +Стежте за агентами та керуйте ними з телефону — отримуйте сповіщення про завершення роботи агента та надсилайте подальші вказівки, де б ви не були. + +[App Store для iOS](https://apps.apple.com/us/app/orca-ide/id6766130217) · [TestFlight](https://testflight.apple.com/join/YjeGMQBA) · [Android APK 0.0.44](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.44/app-release.apk) · [Документація →](https://www.onorca.dev/docs/mobile) + + + Десктоп Orca із супутнім мобільним застосунком +
    + +### Паралельні worktree + +Надішліть один промпт одразу п’ятьом агентам, кожен із яких працюватиме у власному ізольованому git worktree, — порівняйте результати та виконайте злиття найкращого з них. + +[Документація →](https://www.onorca.dev/docs/model/worktrees) + + + Оркестрація паралельних worktree +
    + +### Розділені термінали + +Термінали рівня Ghostty з рендерингом на WebGL, необмеженою кількістю розділень і буфером прокручування, який зберігається після перезапуску. + +[Документація →](https://www.onorca.dev/docs/terminal) + + + Розділені термінали +
    + +### Режим дизайну + +Клацніть на будь-якому елементі інтерфейсу у справжньому вікні Chromium, щоб надіслати його HTML, CSS і обрізаний скриншот прямо в промпт агента. + +[Документація →](https://www.onorca.dev/docs/browser/design-mode) + + + Вбудований браузер і режим дизайну +
    + +### GitHub і Linear, нативно + +Переглядайте PR, issue та дошки проєктів прямо в застосунку — відкривайте worktree з будь-якої задачі та рев'юйте без перемикання контексту. + +[Документація →](https://www.onorca.dev/docs/review/linear) + + + Робочі процеси GitHub і Linear в Orca +
    + +### SSH worktree + +Запускайте агентів на потужній віддаленій машині з повноцінним редагуванням файлів, git і терміналами — з автоперепідключенням і прокиданням портів. + +[Документація →](https://www.onorca.dev/docs/ssh) + + + Віддалені worktree через SSH +
    + +### Анотуйте diff-и агентів + +Залишайте коментарі на будь-якому рядку diff-у й надсилайте їх агенту — рев'юйте, редагуйте та комітьте, не виходячи з Orca. + +[Документація →](https://www.onorca.dev/docs/review/annotate-ai-diff) + + + Анотування diff-ів, згенерованих AI +
    + +### Перетягуйте файли агентам + +Редактор на базі VS Code з автозбереженням усюди — перетягуйте файли чи зображення прямо в промпт агента. + +[Документація →](https://www.onorca.dev/docs/editing/file-explorer) + + + Перетягування файлів і зображень у промпт агента +
    + +### Orca CLI + +Агенти теж керують Orca — автоматизуйте будь-який робочий процес командами `orca worktree create`, `snapshot`, `click` і `fill`. + +[Документація →](https://www.onorca.dev/docs/cli/overview) + + + Керування Orca з CLI +
    + +**Також у комплекті:** + +- **[Швидкий пошук](https://www.onorca.dev/docs/model/quick-open)** — Шукайте серед worktree, файлів, агентів, команд і контексту репозиторію, не відриваючись від роботи. +- **[Перемикач акаунтів і відстеження використання](https://www.onorca.dev/docs/agents/usage-tracking)** — Стежте за використанням Claude і Codex та скиданням лімітів, перемикайте акаунти на льоту без повторного входу. +- **[Розширені перегляди репозиторію](https://www.onorca.dev/docs/editing/markdown)** — Переглядайте Markdown, зображення, PDF та документацію репозиторію прямо в робочому просторі. +- **[Computer Use](https://www.onorca.dev/docs/cli/computer-use)** — Дозвольте агентам керувати десктопними застосунками та видимим інтерфейсом, коли робочий процес потребує реальної взаємодії. +- **[Сповіщення та статус непрочитаного](https://www.onorca.dev/docs/notifications)** — Дізнавайтеся, коли агент завершив роботу або потребує уваги, і позначайте треди як непрочитані, щоб повернутися пізніше. +- **І багато іншого** — ми випускаємо оновлення щодня, тож цей список завжди відстає. Справжній перелік можливостей — це [changelog](https://github.com/stablyai/orca/releases). + +--- + +## Підтримувані агенти + +Працює з **будь-яким CLI-агентом** — якщо він запускається в терміналі, він запуститься і в Orca. + +

    + Claude Code logo Claude Code   + Codex logo Codex   + Grok logo Grok   + Cursor logo Cursor   + GitHub Copilot logo GitHub Copilot   + OpenCode logo OpenCode   + MiMo Code logo MiMo Code   + Amp logo Amp   + OpenClaude logo OpenClaude   + Antigravity logo Antigravity   + Pi logo Pi   + oh-my-pi logo oh-my-pi   + Hermes Agent logo Hermes Agent   + Devin logo Devin   + Goose logo Goose   + Auggie logo Auggie   + Autohand Code logo Autohand Code   + Charm logo Charm   + Cline logo Cline   + Codebuff logo Codebuff   + Command Code logo Command Code   + Continue logo Continue   + Droid logo Droid   + Kilocode logo Kilocode   + Kimi logo Kimi   + Kiro logo Kiro   + Mistral Vibe logo Mistral Vibe   + Qwen Code logo Qwen Code   + Rovo Dev logo Rovo Dev   + + будь-який CLI-агент +

    + +--- + +## Встановлення + +### Десктоп — macOS, Windows, Linux + +- **[Завантажити з onOrca.dev](https://onorca.dev/download)** +- Або завантажте білд напряму: [macOS Apple Silicon](https://github.com/stablyai/orca/releases/latest/download/orca-macos-arm64.dmg) · [macOS Intel](https://github.com/stablyai/orca/releases/latest/download/orca-macos-x64.dmg) · [Windows (.exe)](https://github.com/stablyai/orca/releases/latest/download/orca-windows-setup.exe) · [Linux AppImage](https://github.com/stablyai/orca/releases/latest/download/orca-linux.AppImage) · [Усі білди](https://github.com/stablyai/orca/releases/latest) +- Запускаєте `orca serve` на headless Linux-сервері? Дивіться [посібник із headless Linux-сервера](../reference/headless-linux-server.md). + +_Або через пакетний менеджер:_ + +```bash +# macOS (Homebrew) +brew install --cask stablyai/orca/orca + +# Arch Linux (AUR) — або stably-orca-git для збірки з джерела +yay -S stably-orca-bin +``` + +### Супутній мобільний застосунок — iOS, Android + +Під’єднайте мобільний застосунок до десктопного, щоб стежити за агентами та керувати ними з телефону. + +- **iOS:** [Завантажити з App Store](https://apps.apple.com/us/app/orca-ide/id6766130217) або [приєднатися до TestFlight](https://testflight.apple.com/join/YjeGMQBA) +- **Android:** [Завантажити APK 0.0.44](https://github.com/stablyai/orca/releases/download/mobile-android-v0.0.44/app-release.apk) · [Інструкція зі встановлення](https://www.onorca.dev/docs/android-apk) + +--- + +## Спільнота та підтримка + +- **Discord:** Приєднуйтеся до спільноти в **[Discord](https://discord.gg/fzjDKHxv8Q)**. +- **Twitter / X:** Стежте за **[@orca_build](https://x.com/orca_build)**, щоб бути в курсі оновлень і анонсів. +- **WeChat:** Відскануйте QR-код, щоб приєднатися до групи № 7 спільноти Orca у WeChat. Якщо вона заповнена, приєднайтеся до групи № 8. + + QR-код групи WeChat 7 спільноти Orca   + QR-код групи WeChat 8 спільноти Orca + +- **Зворотний зв'язок та ідеї:** Ми випускаємо оновлення швидко. Чогось бракує? [Запропонуйте нову функцію](https://github.com/stablyai/orca/issues). +- **Конфіденційність:** Перегляньте [документацію про конфіденційність і телеметрію](https://www.onorca.dev/docs/telemetry), щоб дізнатися, які анонімні дані про використання збирає Orca і як від цього відмовитися. +- **Підтримайте нас:** Поставте [зірку](https://github.com/stablyai/orca) цьому репозиторію, щоб стежити за нашими щоденними релізами. + +--- + +## Розробка + +Хочете зробити внесок або запустити проєкт локально? Перегляньте наш посібник [CONTRIBUTING.md](../../.github/CONTRIBUTING.md). + + + Контриб'ютори Orca + + +

    + Графік історії зірок на GitHub для stablyai/orca +

    + +## Підписані білди +Підписання коду для Windows надано за підтримки [SignPath.io](https://signpath.io), сертифікат надано [SignPath Foundation](https://signpath.org). + +## Ліцензія + +Orca — безкоштовний проєкт із відкритим кодом за ліцензією [MIT](../../LICENSE). diff --git a/docs/readme/README.zh-CN.md b/docs/readme/README.zh-CN.md index d7bae3fba9e..160f5b05611 100644 --- a/docs/readme/README.zh-CN.md +++ b/docs/readme/README.zh-CN.md @@ -12,7 +12,7 @@

    - English · 日本語 · 한국어 · Español · Français · Português + English · 日本語 · 한국어 · Español · Français · Português · Українська

    From 8463dcb7b97614a75d91e6d71c9cd83c9712daa2 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 17:59:16 -0700 Subject: [PATCH 215/398] fix(terminal): make wrapped-line search rewind iterative and bound its scans (#18402) Patches @xterm/addon-search so one very long un-newlined line no longer overflows the stack, freezes the renderer, or goes unsearched. Submitted upstream as xtermjs/xterm.js#6149 (issue #6148); drop the patch once a release ships it. See the PR for measurements and the differential fuzz. --- .github/workflows/pr.yml | 2 +- ...@xterm__addon-search@0.17.0-beta.300.patch | 275 +++++++++++++ ...rm__addon-search@0.17.0-beta.300.src.patch | 231 +++++++++++ config/patches/xterm-upstream.json | 27 ++ docs/reference/xterm-patch-regeneration.md | 10 +- pnpm-lock.yaml | 5 +- pnpm-workspace.yaml | 1 + .../terminal-search-long-wrapped-line.test.ts | 362 ++++++++++++++++++ 8 files changed, 905 insertions(+), 8 deletions(-) create mode 100644 config/patches/@xterm__addon-search@0.17.0-beta.300.patch create mode 100644 config/patches/xterm-src/@xterm__addon-search@0.17.0-beta.300.src.patch create mode 100644 src/renderer/src/components/terminal-search-long-wrapped-line.test.ts diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml index d21b1a23784..ed37cefb434 100644 --- a/.github/workflows/pr.yml +++ b/.github/workflows/pr.yml @@ -380,7 +380,7 @@ jobs: - uses: ./.github/actions/install-node-dependencies # Why: the check rebuilds every package in the manifest from a pinned upstream - # commit — @xterm/xterm and the two addons, each built twice (once unmodified to + # commit — @xterm/xterm and its three addons, each built twice (once unmodified to # prove the toolchain still reproduces the published bundles, once patched). Caching # the npm metadata and the shallow clone keeps the repeated cost to the builds # themselves; the key is the manifest, so a commit, package or toolchain bump diff --git a/config/patches/@xterm__addon-search@0.17.0-beta.300.patch b/config/patches/@xterm__addon-search@0.17.0-beta.300.patch new file mode 100644 index 00000000000..c2843ec6b3d --- /dev/null +++ b/config/patches/@xterm__addon-search@0.17.0-beta.300.patch @@ -0,0 +1,275 @@ +diff --git a/lib/addon-search.js b/lib/addon-search.js +index d939cf1a65f3de449059efbf4fc5c3c3515f533f..8d2b66d265c256d4ebd507624d6a29b8075929ce 100644 +--- a/lib/addon-search.js ++++ b/lib/addon-search.js +@@ -1,2 +1,2 @@ +-!function(e,t){"object"==typeof exports&&"object"==typeof module?module.exports=t():"function"==typeof define&&define.amd?define([],t):"object"==typeof exports?exports.SearchAddon=t():e.SearchAddon=t()}(globalThis,()=>(()=>{"use strict";var e={578(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.IntervalTimer=t.MicrotaskTimer=t.TimeoutTimer=void 0,t.timeout=function(e){return new Promise(t=>setTimeout(t,e))},t.disposableTimeout=function(e,t=0,s){const r=setTimeout(()=>{e(),s&&n.dispose()},t),n=(0,i.toDisposable)(()=>{clearTimeout(r)});return s?.add(n),n};const i=s(426);t.TimeoutTimer=class{constructor(){this._token=-1,this._isDisposed=!1}dispose(){this.cancel(),this._isDisposed=!0}cancel(){-1!==this._token&&(clearTimeout(this._token),this._token=-1)}cancelAndSet(e,t){if(this._isDisposed)throw new Error("Calling cancelAndSet on a disposed TimeoutTimer");this.cancel(),this._token=setTimeout(()=>{this._token=-1,e()},t)}setIfNotSet(e,t){if(this._isDisposed)throw new Error("Calling setIfNotSet on a disposed TimeoutTimer");-1===this._token&&(this._token=setTimeout(()=>{this._token=-1,e()},t))}},t.MicrotaskTimer=class{constructor(){this._isScheduled=!1,this._isDisposed=!1}dispose(){this.cancel(),this._isDisposed=!0}cancel(){this._isScheduled=!1}set(e){if(this._isDisposed)throw new Error("Calling set on a disposed MicrotaskTimer");this._isScheduled||(this._isScheduled=!0,queueMicrotask(()=>{this._isScheduled&&(this._isScheduled=!1,e())}))}},t.IntervalTimer=class{constructor(){this._isDisposed=!1}cancel(){this._disposable?.dispose(),this._disposable=void 0}cancelAndSet(e,t,s=globalThis){if(this._isDisposed)throw new Error("Calling cancelAndSet on a disposed IntervalTimer");this.cancel();const i=s.setInterval(()=>{e()},t);this._disposable={dispose:()=>{s.clearInterval(i),this._disposable=void 0}}}dispose(){this.cancel(),this._isDisposed=!0}}},414(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.EventUtils=t.Emitter=void 0;const i=s(426);var r;t.Emitter=class{constructor(){this._listeners=[],this._disposed=!1}get event(){return this._event||(this._event=(e,t,s)=>{if(this._disposed)return(0,i.toDisposable)(()=>{});const r={fn:e,thisArgs:t};this._listeners=this._listeners.slice(),this._listeners.push(r);const n=(0,i.toDisposable)(()=>{const e=this._listeners.indexOf(r);-1!==e&&(this._listeners=this._listeners.slice(),this._listeners.splice(e,1))});return s&&(Array.isArray(s)?s.push(n):s.add(n)),n}),this._event}fire(e){if(this._disposed||!this._listeners.length)return;if(1===this._listeners.length)return void this._listeners[0].fn.call(this._listeners[0].thisArgs,e);const t=this._listeners;for(let s=0,i=t.length;st.fire(e))},e.map=function(e,t){return(s,i,r)=>e(e=>s.call(i,t(e)),void 0,r)},e.any=function(...e){return(t,s,r)=>{const n=new i.DisposableStore;for(const i of e)n.add(i(e=>t.call(s,e)));return r&&(Array.isArray(r)?r.push(n):r.add(n)),n}},e.runAndSubscribe=function(e,t,s){return t(s),e(e=>t(e))}}(r||(t.EventUtils=r={}))},426(e,t){function s(e){return{dispose:e}}function i(e){if(!e)return e;if(Array.isArray(e)){for(const t of e)t.dispose();return[]}return e.dispose(),e}Object.defineProperty(t,"__esModule",{value:!0}),t.MutableDisposable=t.Disposable=t.DisposableStore=void 0,t.toDisposable=s,t.dispose=i,t.combinedDisposable=function(...e){return s(()=>i(e))};class r{constructor(){this._disposables=new Set,this._isDisposed=!1}get isDisposed(){return this._isDisposed}add(e){return this._isDisposed?e.dispose():this._disposables.add(e),e}dispose(){if(!this._isDisposed){this._isDisposed=!0;for(const e of this._disposables)e.dispose();this._disposables.clear()}}clear(){for(const e of this._disposables)e.dispose();this._disposables.clear()}}t.DisposableStore=r;class n{constructor(){this._store=new r}dispose(){this._store.dispose()}_register(e){return this._store.add(e)}}t.Disposable=n,n.None=Object.freeze({dispose(){}}),t.MutableDisposable=class{constructor(){this._isDisposed=!1}get value(){return this._isDisposed?void 0:this._value}set value(e){this._isDisposed||e===this._value||(this._value?.dispose(),this._value=e)}clear(){this.value=void 0}dispose(){this._isDisposed=!0,this._value?.dispose(),this._value=void 0}}},864(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.DecorationManager=void 0;const i=s(426);class r extends i.Disposable{constructor(e){super(),this._terminal=e,this._highlightDecorations=[],this._highlightedLines=new Set,this._register((0,i.toDisposable)(()=>this.clearHighlightDecorations()))}createHighlightDecorations(e,t){this.clearHighlightDecorations();for(const s of e){const e=this._createResultDecorations(s,t,!1);if(e)for(const t of e)this._storeDecoration(t,s)}}createActiveDecoration(e,t){const s=this._createResultDecorations(e,t,!0);if(s)return{decorations:s,match:e,dispose(){(0,i.dispose)(s)}}}clearHighlightDecorations(){(0,i.dispose)(this._highlightDecorations),this._highlightDecorations=[],this._highlightedLines.clear()}_storeDecoration(e,t){this._highlightedLines.add(e.marker.line),this._highlightDecorations.push({decoration:e,match:t,dispose(){e.dispose()}})}_applyStyles(e,t,s){e.classList.contains("xterm-find-result-decoration")||(e.classList.add("xterm-find-result-decoration"),t&&(e.style.outline=`1px solid ${t}`)),s&&e.classList.add("xterm-find-active-result-decoration")}_createResultDecorations(e,t,s){const r=[];let n=e.col,o=e.size,a=-this._terminal.buffer.active.baseY-this._terminal.buffer.active.cursorY+e.row;for(;o>0;){const e=Math.min(this._terminal.cols-n,o);r.push([a,n,e]),n=0,o-=e,a++}const h=[];for(const e of r){const r=this._terminal.registerMarker(e[0]),n=this._terminal.registerDecoration({marker:r,x:e[1],width:e[2],layer:s?"top":"bottom",backgroundColor:s?t.activeMatchBackground:t.matchBackground,overviewRulerOptions:this._highlightedLines.has(r.line)?void 0:{color:s?t.activeMatchColorOverviewRuler:t.matchOverviewRuler,position:"center"}});if(n){const e=[];e.push(r),e.push(n.onRender(e=>this._applyStyles(e,s?t.activeMatchBorder:t.matchBorder,!1))),e.push(n.onDispose(()=>(0,i.dispose)(e))),h.push(n)}}return 0===h.length?void 0:h}}t.DecorationManager=r},615(e,t){Object.defineProperty(t,"__esModule",{value:!0}),t.SearchEngine=void 0,t.SearchEngine=class{constructor(e,t){this._terminal=e,this._lineCache=t}find(e,t,s,i){if(!e||0===e.length)return void this._terminal.clearSelection();if(s>=this._terminal.cols)throw new Error(`Invalid col: ${s} to search in terminal of ${this._terminal.cols} cols`);this._lineCache.initLinesCache();const r={startRow:t,startCol:s};let n=this._findInLine(e,r,i);if(!n)for(let s=t+1;s=0&&(a.startRow=s,h=this._findInLine(e,a,t,o),!h);s--);}if(!h&&r!==this._terminal.buffer.active.baseY+this._terminal.rows-1)for(let s=this._terminal.buffer.active.baseY+this._terminal.rows-1;s>=r&&(a.startRow=s,h=this._findInLine(e,a,t,o),!h);s--);return h}_isWholeWord(e,t,s){return(0===e||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(t[e-1]))&&(e+s.length===t.length||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(t[e+s.length]))}_findInLine(e,t,s={},i=!1){const r=t.startRow,n=t.startCol,o=this._terminal.buffer.active.getLine(r);if(o?.isWrapped)return i?void(t.startCol+=this._terminal.cols):(t.startRow--,t.startCol+=this._terminal.cols,this._findInLine(e,t,s));let a=this._lineCache.getLineFromCache(r);a||(a=this._lineCache.translateBufferLineToStringWithWrap(r,!0),this._lineCache.setLineInCache(r,a));const[h,l]=a,c=this._bufferColsToStringOffset(r,n);let d=e,_=h;s.regex||(d=s.caseSensitive?e:e.toLowerCase(),_=s.caseSensitive?h:h.toLowerCase());let u=-1;if(s.regex){const t=RegExp(d,s.caseSensitive?"g":"gi");let r;if(i)for(;r=t.exec(_.slice(0,c));)u=t.lastIndex-r[0].length,e=r[0],t.lastIndex-=e.length-1;else r=t.exec(_.slice(c)),r&&r[0].length>0&&(u=c+(t.lastIndex-r[0].length),e=r[0])}else i?c-d.length>=0&&(u=_.lastIndexOf(d,c-d.length)):u=_.indexOf(d,c);if(u>=0){if(s.wholeWord&&!this._isWholeWord(u,_,e))return;let t=0;for(;t=l[t+1];)t++;let i=t;for(;i=l[i+1];)i++;const n=u-l[t],o=u+e.length-l[i],a=this._stringLengthToBufferSize(r+t,n);return{term:e,col:a,row:r+t,size:this._stringLengthToBufferSize(r+i,o)-a+this._terminal.cols*(i-t)}}}_stringLengthToBufferSize(e,t){const s=this._terminal.buffer.active.getLine(e);if(!s)return 0;for(let e=0;e1&&(t-=r.length-1);const n=s.getCell(e+1);n&&0===n.getWidth()&&t++}return t}_bufferColsToStringOffset(e,t){let s=e,i=0,r=this._terminal.buffer.active.getLine(s);for(;t>0&&r;){for(let e=0;ethis._destroyLinesCache()))}initLinesCache(){this._linesCache||(this._linesCache=new Array(this._terminal.buffer.active.length),this._linesCacheDisposables.value=(0,i.combinedDisposable)(this._terminal.onLineFeed(()=>this._destroyLinesCache()),this._terminal.onCursorMove(()=>this._destroyLinesCache()),this._terminal.onResize(()=>this._destroyLinesCache()))),this._lastAccessTimestamp=Date.now(),this._linesCacheTimeout.value||this._scheduleLinesCacheTimeout(15e3)}_destroyLinesCache(){this._linesCache=void 0,this._lastAccessTimestamp=0,this._linesCacheDisposables.clear(),this._linesCacheTimeout.clear()}_scheduleLinesCacheTimeout(e){this._linesCacheTimeout.value=(0,r.disposableTimeout)(()=>{if(!this._linesCache)return;const e=Date.now()-this._lastAccessTimestamp;e>=15e3?this._destroyLinesCache():this._scheduleLinesCacheTimeout(15e3-e)},e)}getLineFromCache(e){return this._linesCache?.[e]}setLineInCache(e,t){this._linesCache&&(this._linesCache[e]=t)}translateBufferLineToStringWithWrap(e,t){const s=[],i=[0];let r=this._terminal.buffer.active.getLine(e);for(;r;){const n=this._terminal.buffer.active.getLine(e+1),o=!!n&&n.isWrapped;let a=r.translateToString(!o&&t);if(o&&n){const e=r.getCell(r.length-1);e&&0===e.getCode()&&1===e.getWidth()&&2===n.getCell(0)?.getWidth()&&(a=a.slice(0,-1))}if(s.push(a),!o)break;i.push(i[i.length-1]+a.length),e++,r=n}return[s.join(""),i]}}t.SearchLineCache=n},438(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.SearchResultTracker=void 0;const i=s(414),r=s(426);class n extends r.Disposable{constructor(){super(...arguments),this._searchResults=[],this._onDidChangeResults=this._register(new i.Emitter)}get onDidChangeResults(){return this._onDidChangeResults.event}get searchResults(){return this._searchResults}get selectedDecoration(){return this._selectedDecoration}set selectedDecoration(e){this._selectedDecoration=e}updateResults(e,t){this._searchResults=e.slice(0,t)}clearResults(){this._searchResults=[]}clearSelectedDecoration(){this._selectedDecoration&&(this._selectedDecoration.dispose(),this._selectedDecoration=void 0)}findResultIndex(e){for(let t=0;t0)}didOptionsChange(e){return!this._lastSearchOptions||!!e&&(this._lastSearchOptions.caseSensitive!==e.caseSensitive||this._lastSearchOptions.regex!==e.regex||this._lastSearchOptions.wholeWord!==e.wholeWord)}shouldUpdateHighlighting(e,t){return!!t?.decorations&&(void 0===this._cachedSearchTerm||e!==this._cachedSearchTerm||this.didOptionsChange(t))}clearCachedTerm(){this._cachedSearchTerm=void 0}reset(){this._cachedSearchTerm=void 0,this._lastSearchOptions=void 0}}}},t={};function s(i){var r=t[i];if(void 0!==r)return r.exports;var n=t[i]={exports:{}};return e[i](n,n.exports,s),n.exports}var i={};return(()=>{var e=i;Object.defineProperty(e,"__esModule",{value:!0}),e.SearchAddon=void 0;const t=s(414),r=s(426),n=s(578),o=s(149),a=s(772),h=s(615),l=s(864),c=s(438);class d extends r.Disposable{get onDidChangeResults(){return this._resultTracker.onDidChangeResults}constructor(e){super(),this._highlightTimeout=this._register(new r.MutableDisposable),this._lineCache=this._register(new r.MutableDisposable),this._state=new a.SearchState,this._resultTracker=this._register(new c.SearchResultTracker),this._onAfterSearch=this._register(new t.Emitter),this.onAfterSearch=this._onAfterSearch.event,this._onBeforeSearch=this._register(new t.Emitter),this.onBeforeSearch=this._onBeforeSearch.event,this._highlightLimit=e?.highlightLimit??1e3}activate(e){this._terminal=e,this._lineCache.value=new o.SearchLineCache(e),this._engine=new h.SearchEngine(e,this._lineCache.value),this._decorationManager=new l.DecorationManager(e),this._register(this._terminal.onWriteParsed(()=>this._updateMatches())),this._register(this._terminal.onResize(()=>this._updateMatches())),this._register((0,r.toDisposable)(()=>this.clearDecorations()))}_updateMatches(){this._highlightTimeout.clear(),this._state.cachedSearchTerm&&this._state.lastSearchOptions?.decorations&&(this._highlightTimeout.value=(0,n.disposableTimeout)(()=>{const e=this._state.cachedSearchTerm;this._state.clearCachedTerm(),this.findPrevious(e,{...this._state.lastSearchOptions,incremental:!0},{noScroll:!0})},200))}clearDecorations(e){this._resultTracker.clearSelectedDecoration(),this._decorationManager?.clearHighlightDecorations(),this._resultTracker.clearResults(),e||this._state.clearCachedTerm()}clearActiveDecoration(){this._resultTracker.clearSelectedDecoration()}findNext(e,t,s){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);const i=this._findNextAndSelect(e,t,s);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),i}_highlightAllMatches(e,t){if(!this._terminal||!this._engine||!this._decorationManager)throw new Error("Cannot use addon until it has been loaded");if(!this._state.isValidSearchTerm(e))return void this.clearDecorations();this.clearDecorations(!0);const s=[];let i,r=this._engine.find(e,0,0,t);for(;r&&(i?.row!==r.row||i?.col!==r.col)&&!(s.length>=this._highlightLimit);){i=r,s.push(i);const n=this._terminal.cols;let o=i.col+i.size,a=i.row;o>=n&&(a+=Math.floor(o/n),o%=n),r=this._engine.find(e,a,o,t)}this._resultTracker.updateResults(s,this._highlightLimit),t.decorations&&this._decorationManager.createHighlightDecorations(s,t.decorations)}_findNextAndSelect(e,t,s){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;const i=this._engine.findNextWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(i,t?.decorations,s?.noScroll)}findPrevious(e,t,s){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);const i=this._findPreviousAndSelect(e,t,s);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),i}_fireResults(e){this._resultTracker.fireResultsChanged(!!e?.decorations)}_findPreviousAndSelect(e,t,s){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;const i=this._engine.findPreviousWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(i,t?.decorations,s?.noScroll)}_selectResult(e,t,s){if(!this._terminal||!this._decorationManager)return!1;if(this._resultTracker.clearSelectedDecoration(),!e)return this._terminal.clearSelection(),!1;if(this._terminal.select(e.col,e.row,e.size),t){const s=this._decorationManager.createActiveDecoration(e,t);s&&(this._resultTracker.selectedDecoration=s)}if(!s&&(e.row>=this._terminal.buffer.active.viewportY+this._terminal.rows||e.row(()=>{"use strict";var e={578(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.IntervalTimer=t.MicrotaskTimer=t.TimeoutTimer=void 0,t.timeout=function(e){return new Promise(t=>setTimeout(t,e))},t.disposableTimeout=function(e,t=0,s){const r=setTimeout(()=>{e(),s&&o.dispose()},t),o=(0,i.toDisposable)(()=>{clearTimeout(r)});return s?.add(o),o};const i=s(426);t.TimeoutTimer=class{constructor(){this._token=-1,this._isDisposed=!1}dispose(){this.cancel(),this._isDisposed=!0}cancel(){-1!==this._token&&(clearTimeout(this._token),this._token=-1)}cancelAndSet(e,t){if(this._isDisposed)throw new Error("Calling cancelAndSet on a disposed TimeoutTimer");this.cancel(),this._token=setTimeout(()=>{this._token=-1,e()},t)}setIfNotSet(e,t){if(this._isDisposed)throw new Error("Calling setIfNotSet on a disposed TimeoutTimer");-1===this._token&&(this._token=setTimeout(()=>{this._token=-1,e()},t))}},t.MicrotaskTimer=class{constructor(){this._isScheduled=!1,this._isDisposed=!1}dispose(){this.cancel(),this._isDisposed=!0}cancel(){this._isScheduled=!1}set(e){if(this._isDisposed)throw new Error("Calling set on a disposed MicrotaskTimer");this._isScheduled||(this._isScheduled=!0,queueMicrotask(()=>{this._isScheduled&&(this._isScheduled=!1,e())}))}},t.IntervalTimer=class{constructor(){this._isDisposed=!1}cancel(){this._disposable?.dispose(),this._disposable=void 0}cancelAndSet(e,t,s=globalThis){if(this._isDisposed)throw new Error("Calling cancelAndSet on a disposed IntervalTimer");this.cancel();const i=s.setInterval(()=>{e()},t);this._disposable={dispose:()=>{s.clearInterval(i),this._disposable=void 0}}}dispose(){this.cancel(),this._isDisposed=!0}}},414(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.EventUtils=t.Emitter=void 0;const i=s(426);var r;t.Emitter=class{constructor(){this._listeners=[],this._disposed=!1}get event(){return this._event||(this._event=(e,t,s)=>{if(this._disposed)return(0,i.toDisposable)(()=>{});const r={fn:e,thisArgs:t};this._listeners=this._listeners.slice(),this._listeners.push(r);const o=(0,i.toDisposable)(()=>{const e=this._listeners.indexOf(r);-1!==e&&(this._listeners=this._listeners.slice(),this._listeners.splice(e,1))});return s&&(Array.isArray(s)?s.push(o):s.add(o)),o}),this._event}fire(e){if(this._disposed||!this._listeners.length)return;if(1===this._listeners.length)return void this._listeners[0].fn.call(this._listeners[0].thisArgs,e);const t=this._listeners;for(let s=0,i=t.length;st.fire(e))},e.map=function(e,t){return(s,i,r)=>e(e=>s.call(i,t(e)),void 0,r)},e.any=function(...e){return(t,s,r)=>{const o=new i.DisposableStore;for(const i of e)o.add(i(e=>t.call(s,e)));return r&&(Array.isArray(r)?r.push(o):r.add(o)),o}},e.runAndSubscribe=function(e,t,s){return t(s),e(e=>t(e))}}(r||(t.EventUtils=r={}))},426(e,t){function s(e){return{dispose:e}}function i(e){if(!e)return e;if(Array.isArray(e)){for(const t of e)t.dispose();return[]}return e.dispose(),e}Object.defineProperty(t,"__esModule",{value:!0}),t.MutableDisposable=t.Disposable=t.DisposableStore=void 0,t.toDisposable=s,t.dispose=i,t.combinedDisposable=function(...e){return s(()=>i(e))};class r{constructor(){this._disposables=new Set,this._isDisposed=!1}get isDisposed(){return this._isDisposed}add(e){return this._isDisposed?e.dispose():this._disposables.add(e),e}dispose(){if(!this._isDisposed){this._isDisposed=!0;for(const e of this._disposables)e.dispose();this._disposables.clear()}}clear(){for(const e of this._disposables)e.dispose();this._disposables.clear()}}t.DisposableStore=r;class o{constructor(){this._store=new r}dispose(){this._store.dispose()}_register(e){return this._store.add(e)}}t.Disposable=o,o.None=Object.freeze({dispose(){}}),t.MutableDisposable=class{constructor(){this._isDisposed=!1}get value(){return this._isDisposed?void 0:this._value}set value(e){this._isDisposed||e===this._value||(this._value?.dispose(),this._value=e)}clear(){this.value=void 0}dispose(){this._isDisposed=!0,this._value?.dispose(),this._value=void 0}}},864(e,t,s){Object.defineProperty(t,"__esModule",{value:!0}),t.DecorationManager=void 0;const i=s(426);class r extends i.Disposable{constructor(e){super(),this._terminal=e,this._highlightDecorations=[],this._highlightedLines=new Set,this._register((0,i.toDisposable)(()=>this.clearHighlightDecorations()))}createHighlightDecorations(e,t){this.clearHighlightDecorations();for(const s of e){const e=this._createResultDecorations(s,t,!1);if(e)for(const t of e)this._storeDecoration(t,s)}}createActiveDecoration(e,t){const s=this._createResultDecorations(e,t,!0);if(s)return{decorations:s,match:e,dispose(){(0,i.dispose)(s)}}}clearHighlightDecorations(){(0,i.dispose)(this._highlightDecorations),this._highlightDecorations=[],this._highlightedLines.clear()}_storeDecoration(e,t){this._highlightedLines.add(e.marker.line),this._highlightDecorations.push({decoration:e,match:t,dispose(){e.dispose()}})}_applyStyles(e,t,s){e.classList.contains("xterm-find-result-decoration")||(e.classList.add("xterm-find-result-decoration"),t&&(e.style.outline=`1px solid ${t}`)),s&&e.classList.add("xterm-find-active-result-decoration")}_createResultDecorations(e,t,s){const r=[];let o=e.col,n=e.size,a=-this._terminal.buffer.active.baseY-this._terminal.buffer.active.cursorY+e.row;for(;n>0;){const e=Math.min(this._terminal.cols-o,n);r.push([a,o,e]),o=0,n-=e,a++}const h=[];for(const e of r){const r=this._terminal.registerMarker(e[0]),o=this._terminal.registerDecoration({marker:r,x:e[1],width:e[2],layer:s?"top":"bottom",backgroundColor:s?t.activeMatchBackground:t.matchBackground,overviewRulerOptions:this._highlightedLines.has(r.line)?void 0:{color:s?t.activeMatchColorOverviewRuler:t.matchOverviewRuler,position:"center"}});if(o){const e=[];e.push(r),e.push(o.onRender(e=>this._applyStyles(e,s?t.activeMatchBorder:t.matchBorder,!1))),e.push(o.onDispose(()=>(0,i.dispose)(e))),h.push(o)}}return 0===h.length?void 0:h}}t.DecorationManager=r},615(e,t){Object.defineProperty(t,"__esModule",{value:!0}),t.SearchEngine=void 0,t.SearchEngine=class{constructor(e,t){this._terminal=e,this._lineCache=t}find(e,t,s,i){if(!e||0===e.length)return void this._terminal.clearSelection();if(s>=this._terminal.cols)throw new Error(`Invalid col: ${s} to search in terminal of ${this._terminal.cols} cols`);this._lineCache.initLinesCache();const r={startRow:t,startCol:s};let o=this._findInLine(e,r,i);if(!o)for(let s=t+1;s0&&this._isRowCoveredByEarlierSearch(s)||(n.startRow=s,n.startCol=0,a=this._findInLine(e,n,t),!a));s++);return!a&&i&&(n.startRow=i.start.y,n.startCol=0,a=this._findInLine(e,n,t)),a}findPreviousWithSelection(e,t,s){if(!e||0===e.length)return void this._terminal.clearSelection();const i=this._terminal.getSelectionPosition();this._terminal.clearSelection();let r=this._terminal.buffer.active.baseY+this._terminal.rows-1;const o=this._terminal.cols,n=!0;this._lineCache.initLinesCache();const a={startRow:r,startCol:o};let h;if(i&&(a.startRow=r=i.start.y,a.startCol=i.start.x,s!==e&&(h=this._findInLine(e,a,t,!1),h||(a.startRow=r=i.end.y,a.startCol=i.end.x))),h??=this._findInLine(e,a,t,n),!h){a.startCol=Math.max(a.startCol,this._terminal.cols);for(let s=r-1;s>=0&&(a.startRow=s,h=this._findInLine(e,a,t,n),!h);s--);}if(!h&&r!==this._terminal.buffer.active.baseY+this._terminal.rows-1)for(let s=this._terminal.buffer.active.baseY+this._terminal.rows-1;s>=r&&(a.startRow=s,h=this._findInLine(e,a,t,n),!h);s--);return h}_isWholeWord(e,t,s){return(0===e||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(t[e-1]))&&(e+s.length===t.length||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(t[e+s.length]))}_satisfiesWholeWord(e,t,s,i){return!i.wholeWord||this._isWholeWord(e,t,s)}_isRowCoveredByEarlierSearch(e){return!0===this._terminal.buffer.active.getLine(e)?.isWrapped}_findInLine(e,t,s={},i=!1){if(i){if(t.startRow>0&&this._terminal.buffer.active.getLine(t.startRow)?.isWrapped)return void(t.startCol+=this._terminal.cols)}else for(;t.startRow>0&&this._terminal.buffer.active.getLine(t.startRow)?.isWrapped;)t.startRow--,t.startCol+=this._terminal.cols;const r=t.startRow,o=t.startCol;let n=this._lineCache.getLineFromCache(r);n||(n=this._lineCache.translateBufferLineToStringWithWrap(r,!0),this._lineCache.setLineInCache(r,n));const[a,h]=n,l=this._bufferColsToStringOffset(r,o,h);let c=e,d=a;s.regex||(c=s.caseSensitive?e:e.toLowerCase(),d=s.caseSensitive?a:a.toLowerCase());let _=-1;if(s.regex){const t=RegExp(c,s.caseSensitive?"g":"gi");let r;if(i)for(;r=t.exec(d.slice(0,l));){const i=t.lastIndex-r[0].length;r[0].length>0&&this._satisfiesWholeWord(i,d,r[0],s)&&(_=i,e=r[0]),t.lastIndex=i+1}else for(t.lastIndex=l;r=t.exec(d);){const i=t.lastIndex-r[0].length;if(r[0].length>0&&this._satisfiesWholeWord(i,d,r[0],s)){_=i,e=r[0];break}t.lastIndex=i+1}}else if(i){let e=l-c.length>=0?d.lastIndexOf(c,l-c.length):-1;for(;e>=0&&!this._satisfiesWholeWord(e,d,c,s);)e=e>0?d.lastIndexOf(c,e-1):-1;_=e}else{let e=d.indexOf(c,l);for(;e>=0&&!this._satisfiesWholeWord(e,d,c,s);)e=d.indexOf(c,e+1);_=e}if(_>=0){let t=0;for(;t=h[t+1];)t++;let s=t;for(;s=h[s+1];)s++;const i=_-h[t],o=_+e.length-h[s],n=this._stringLengthToBufferSize(r+t,i);return{term:e,col:n,row:r+t,size:this._stringLengthToBufferSize(r+s,o)-n+this._terminal.cols*(s-t)}}}_stringLengthToBufferSize(e,t){const s=this._terminal.buffer.active.getLine(e);if(!s)return 0;for(let e=0;e1&&(t-=r.length-1);const o=s.getCell(e+1);o&&0===o.getWidth()&&t++}return t}_bufferColsToStringOffset(e,t,s){const i=Math.min(Math.floor(t/this._terminal.cols),s.length-1);let r=s[i];const o=this._terminal.buffer.active.getLine(e+i);if(o){const e=Math.min(t-i*this._terminal.cols,this._terminal.cols);for(let t=0;tthis._destroyLinesCache()))}initLinesCache(){this._linesCache||(this._linesCache=new Array(this._terminal.buffer.active.length),this._linesCacheDisposables.value=(0,i.combinedDisposable)(this._terminal.onLineFeed(()=>this._destroyLinesCache()),this._terminal.onCursorMove(()=>this._destroyLinesCache()),this._terminal.onResize(()=>this._destroyLinesCache()))),this._lastAccessTimestamp=Date.now(),this._linesCacheTimeout.value||this._scheduleLinesCacheTimeout(15e3)}_destroyLinesCache(){this._linesCache=void 0,this._lastAccessTimestamp=0,this._linesCacheDisposables.clear(),this._linesCacheTimeout.clear()}_scheduleLinesCacheTimeout(e){this._linesCacheTimeout.value=(0,r.disposableTimeout)(()=>{if(!this._linesCache)return;const e=Date.now()-this._lastAccessTimestamp;e>=15e3?this._destroyLinesCache():this._scheduleLinesCacheTimeout(15e3-e)},e)}getLineFromCache(e){return this._linesCache?.[e]}setLineInCache(e,t){this._linesCache&&(this._linesCache[e]=t)}translateBufferLineToStringWithWrap(e,t){const s=[],i=[0],r=this._terminal.buffer.active.length;let o=this._terminal.buffer.active.getLine(e);for(;o;){const n=e+10)}didOptionsChange(e){return!this._lastSearchOptions||!!e&&(this._lastSearchOptions.caseSensitive!==e.caseSensitive||this._lastSearchOptions.regex!==e.regex||this._lastSearchOptions.wholeWord!==e.wholeWord)}shouldUpdateHighlighting(e,t){return!!t?.decorations&&(void 0===this._cachedSearchTerm||e!==this._cachedSearchTerm||this.didOptionsChange(t))}clearCachedTerm(){this._cachedSearchTerm=void 0}reset(){this._cachedSearchTerm=void 0,this._lastSearchOptions=void 0}}}},t={};function s(i){var r=t[i];if(void 0!==r)return r.exports;var o=t[i]={exports:{}};return e[i](o,o.exports,s),o.exports}var i={};return(()=>{var e=i;Object.defineProperty(e,"__esModule",{value:!0}),e.SearchAddon=void 0;const t=s(414),r=s(426),o=s(578),n=s(149),a=s(772),h=s(615),l=s(864),c=s(438);class d extends r.Disposable{get onDidChangeResults(){return this._resultTracker.onDidChangeResults}constructor(e){super(),this._highlightTimeout=this._register(new r.MutableDisposable),this._lineCache=this._register(new r.MutableDisposable),this._state=new a.SearchState,this._resultTracker=this._register(new c.SearchResultTracker),this._onAfterSearch=this._register(new t.Emitter),this.onAfterSearch=this._onAfterSearch.event,this._onBeforeSearch=this._register(new t.Emitter),this.onBeforeSearch=this._onBeforeSearch.event,this._highlightLimit=e?.highlightLimit??1e3}activate(e){this._terminal=e,this._lineCache.value=new n.SearchLineCache(e),this._engine=new h.SearchEngine(e,this._lineCache.value),this._decorationManager=new l.DecorationManager(e),this._register(this._terminal.onWriteParsed(()=>this._updateMatches())),this._register(this._terminal.onResize(()=>this._updateMatches())),this._register((0,r.toDisposable)(()=>this.clearDecorations()))}_updateMatches(){this._highlightTimeout.clear(),this._state.cachedSearchTerm&&this._state.lastSearchOptions?.decorations&&(this._highlightTimeout.value=(0,o.disposableTimeout)(()=>{const e=this._state.cachedSearchTerm;this._state.clearCachedTerm(),this.findPrevious(e,{...this._state.lastSearchOptions,incremental:!0},{noScroll:!0})},200))}clearDecorations(e){this._resultTracker.clearSelectedDecoration(),this._decorationManager?.clearHighlightDecorations(),this._resultTracker.clearResults(),e||this._state.clearCachedTerm()}clearActiveDecoration(){this._resultTracker.clearSelectedDecoration()}findNext(e,t,s){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);const i=this._findNextAndSelect(e,t,s);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),i}_highlightAllMatches(e,t){if(!this._terminal||!this._engine||!this._decorationManager)throw new Error("Cannot use addon until it has been loaded");if(!this._state.isValidSearchTerm(e))return void this.clearDecorations();this.clearDecorations(!0);const s=[];let i,r=this._engine.find(e,0,0,t);for(;r&&(i?.row!==r.row||i?.col!==r.col)&&!(s.length>=this._highlightLimit);){i=r,s.push(i);const o=this._terminal.cols;let n=i.col+i.size,a=i.row;n>=o&&(a+=Math.floor(n/o),n%=o),r=this._engine.find(e,a,n,t)}this._resultTracker.updateResults(s,this._highlightLimit),t.decorations&&this._decorationManager.createHighlightDecorations(s,t.decorations)}_findNextAndSelect(e,t,s){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;const i=this._engine.findNextWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(i,t?.decorations,s?.noScroll)}findPrevious(e,t,s){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);const i=this._findPreviousAndSelect(e,t,s);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),i}_fireResults(e){this._resultTracker.fireResultsChanged(!!e?.decorations)}_findPreviousAndSelect(e,t,s){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;const i=this._engine.findPreviousWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(i,t?.decorations,s?.noScroll)}_selectResult(e,t,s){if(!this._terminal||!this._decorationManager)return!1;if(this._resultTracker.clearSelectedDecoration(),!e)return this._terminal.clearSelection(),!1;if(this._terminal.select(e.col,e.row,e.size),t){const s=this._decorationManager.createActiveDecoration(e,t);s&&(this._resultTracker.selectedDecoration=s)}if(!s&&(e.row>=this._terminal.buffer.active.viewportY+this._terminal.rows||e.row {\nreturn ","/**\n * Copyright (c) 2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal async helpers for xterm.js core.\n */\n\nimport { DisposableStore, IDisposable, toDisposable } from './Lifecycle';\n\nexport function timeout(millis: number): Promise {\n return new Promise(resolve => setTimeout(resolve, millis));\n}\n\n/**\n * Creates a timeout that can be disposed using its returned value.\n * @param handler The timeout handler.\n * @param timeout An optional timeout in milliseconds.\n * @param store An optional {@link DisposableStore} that will have the timeout disposable managed\n * automatically.\n */\nexport function disposableTimeout(handler: () => void, timeout = 0, store?: DisposableStore): IDisposable {\n const timer = setTimeout(() => {\n handler();\n if (store) {\n disposable.dispose();\n }\n }, timeout);\n const disposable = toDisposable(() => {\n clearTimeout(timer);\n });\n store?.add(disposable);\n return disposable;\n}\n\nexport class TimeoutTimer implements IDisposable {\n private _token: any = -1;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n if (this._token !== -1) {\n clearTimeout(this._token);\n this._token = -1;\n }\n }\n\n public cancelAndSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed TimeoutTimer');\n }\n this.cancel();\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n\n public setIfNotSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling setIfNotSet on a disposed TimeoutTimer');\n }\n if (this._token !== -1) {\n return;\n }\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n}\n\n/**\n * Schedules a single runner on the microtask queue. Unlike {@link TimeoutTimer}, a scheduled\n * microtask cannot be unqueued; {@link cancel} prevents the runner from executing if it has not\n * run yet.\n */\nexport class MicrotaskTimer implements IDisposable {\n private _isScheduled = false;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n this._isScheduled = false;\n }\n\n public set(runner: () => void): void {\n if (this._isDisposed) {\n throw new Error('Calling set on a disposed MicrotaskTimer');\n }\n if (this._isScheduled) {\n return;\n }\n this._isScheduled = true;\n queueMicrotask(() => {\n if (!this._isScheduled) {\n return;\n }\n this._isScheduled = false;\n runner();\n });\n }\n}\n\nexport class IntervalTimer implements IDisposable {\n private _disposable: IDisposable | undefined;\n private _isDisposed = false;\n\n public cancel(): void {\n this._disposable?.dispose();\n this._disposable = undefined;\n }\n\n public cancelAndSet(runner: () => void, interval: number, context: Window | typeof globalThis = globalThis): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed IntervalTimer');\n }\n this.cancel();\n const handle = context.setInterval(() => {\n runner();\n }, interval);\n this._disposable = {\n dispose: () => {\n context.clearInterval(handle as any);\n this._disposable = undefined;\n }\n };\n }\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n}\n","/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal event utilities for xterm.js core.\n * Simplified from VS Code's event.ts - no leak detection/profiling.\n */\n\nimport { IDisposable, DisposableStore, toDisposable } from './Lifecycle';\n\nexport interface IEvent {\n (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore): IDisposable;\n}\n\nexport class Emitter {\n private _listeners: { fn: (e: T) => any, thisArgs: any }[] = [];\n private _disposed = false;\n private _event: IEvent | undefined;\n\n public get event(): IEvent {\n if (this._event) {\n return this._event;\n }\n this._event = (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n if (this._disposed) {\n return toDisposable(() => {});\n }\n\n const entry = { fn: listener, thisArgs };\n this._listeners = this._listeners.slice();\n this._listeners.push(entry);\n\n const result = toDisposable(() => {\n const idx = this._listeners.indexOf(entry);\n if (idx !== -1) {\n this._listeners = this._listeners.slice();\n this._listeners.splice(idx, 1);\n }\n });\n\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(result);\n } else {\n disposables.add(result);\n }\n }\n\n return result;\n };\n return this._event;\n }\n\n public fire(event: T): void {\n if (this._disposed || !this._listeners.length) {\n return;\n }\n if (this._listeners.length === 1) {\n this._listeners[0].fn.call(this._listeners[0].thisArgs, event);\n return;\n }\n const listeners = this._listeners;\n for (let i = 0, len = listeners.length; i < len; ++i) {\n listeners[i].fn.call(listeners[i].thisArgs, event);\n }\n }\n\n public dispose(): void {\n if (this._disposed) {\n return;\n }\n this._disposed = true;\n this._listeners.length = 0;\n }\n}\n\nexport namespace EventUtils {\n export function forward(from: IEvent, to: Emitter): IDisposable {\n return from(e => to.fire(e));\n }\n\n export function map(event: IEvent, map: (i: I) => O): IEvent {\n return (listener: (e: O) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n return event(i => listener.call(thisArgs, map(i)), undefined, disposables);\n };\n }\n\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent {\n return (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n const store = new DisposableStore();\n for (const event of events) {\n store.add(event(e => listener.call(thisArgs, e)));\n }\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(store);\n } else {\n disposables.add(store);\n }\n }\n return store;\n };\n }\n\n export function runAndSubscribe(event: IEvent, handler: (e: T) => void, initial: T): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void, initial?: T): IDisposable {\n handler(initial);\n return event(e => handler(e));\n }\n}\n","/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal lifecycle utilities for xterm.js core.\n * Simplified from VS Code's lifecycle.ts - no tracking/leak detection.\n */\n\nexport interface IDisposable {\n dispose(): void;\n}\n\nexport function toDisposable(fn: () => void): IDisposable {\n return { dispose: fn };\n}\n\nexport function dispose(disposable: T): T;\nexport function dispose(disposable: T | undefined): T | undefined;\nexport function dispose(disposables: T[]): T[];\nexport function dispose(arg: T | T[] | undefined): T | T[] | undefined {\n if (!arg) {\n return arg;\n }\n if (Array.isArray(arg)) {\n for (const d of arg) {\n d.dispose();\n }\n return [];\n }\n arg.dispose();\n return arg;\n}\n\nexport function combinedDisposable(...disposables: IDisposable[]): IDisposable {\n return toDisposable(() => dispose(disposables));\n}\n\nexport class DisposableStore implements IDisposable {\n private readonly _disposables = new Set();\n private _isDisposed = false;\n\n public get isDisposed(): boolean {\n return this._isDisposed;\n }\n\n public add(o: T): T {\n if (this._isDisposed) {\n o.dispose();\n } else {\n this._disposables.add(o);\n }\n return o;\n }\n\n public dispose(): void {\n if (this._isDisposed) {\n return;\n }\n this._isDisposed = true;\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n\n public clear(): void {\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n}\n\nexport abstract class Disposable implements IDisposable {\n public static readonly None: IDisposable = Object.freeze({ dispose() { } });\n\n protected readonly _store = new DisposableStore();\n\n public dispose(): void {\n this._store.dispose();\n }\n\n protected _register(o: T): T {\n return this._store.add(o);\n }\n}\n\nexport class MutableDisposable implements IDisposable {\n private _value: T | undefined;\n private _isDisposed = false;\n\n public get value(): T | undefined {\n return this._isDisposed ? undefined : this._value;\n }\n\n public set value(value: T | undefined) {\n if (this._isDisposed || value === this._value) {\n return;\n }\n this._value?.dispose();\n this._value = value;\n }\n\n public clear(): void {\n this.value = undefined;\n }\n\n public dispose(): void {\n this._isDisposed = true;\n this._value?.dispose();\n this._value = undefined;\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, IDecoration } from '@xterm/xterm';\nimport type { ISearchDecorationOptions } from '@xterm/addon-search';\nimport { dispose, Disposable, toDisposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a highlight decoration.\n */\ninterface IHighlight extends IDisposable {\n decoration: IDecoration;\n match: ISearchResult;\n}\n\n/**\n * Interface for managing multiple decorations for a single match.\n */\ninterface IMultiHighlight extends IDisposable {\n decorations: IDecoration[];\n match: ISearchResult;\n}\n\n/**\n * Manages visual decorations for search results including highlighting and active selection\n * indicators. This class handles the creation, styling, and disposal of search-related decorations.\n */\nexport class DecorationManager extends Disposable {\n private _highlightDecorations: IHighlight[] = [];\n private _highlightedLines: Set = new Set();\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this.clearHighlightDecorations()));\n }\n\n /**\n * Creates decorations for all provided search results.\n * @param results The search results to create decorations for.\n * @param options The decoration options.\n */\n public createHighlightDecorations(results: ISearchResult[], options: ISearchDecorationOptions): void {\n this.clearHighlightDecorations();\n\n for (const match of results) {\n const decorations = this._createResultDecorations(match, options, false);\n if (decorations) {\n for (const decoration of decorations) {\n this._storeDecoration(decoration, match);\n }\n }\n }\n }\n\n /**\n * Creates decorations for the currently active search result.\n * @param result The active search result.\n * @param options The decoration options.\n * @returns The multi-highlight decoration or undefined if creation failed.\n */\n public createActiveDecoration(result: ISearchResult, options: ISearchDecorationOptions): IMultiHighlight | undefined {\n const decorations = this._createResultDecorations(result, options, true);\n if (decorations) {\n return { decorations, match: result, dispose() { dispose(decorations); } };\n }\n return undefined;\n }\n\n /**\n * Clears all highlight decorations.\n */\n public clearHighlightDecorations(): void {\n dispose(this._highlightDecorations);\n this._highlightDecorations = [];\n this._highlightedLines.clear();\n }\n\n /**\n * Stores a decoration and tracks it for management.\n * @param decoration The decoration to store.\n * @param match The search result this decoration represents.\n */\n private _storeDecoration(decoration: IDecoration, match: ISearchResult): void {\n this._highlightedLines.add(decoration.marker.line);\n this._highlightDecorations.push({ decoration, match, dispose() { decoration.dispose(); } });\n }\n\n /**\n * Applies styles to the decoration when it is rendered.\n * @param element The decoration's element.\n * @param borderColor The border color to apply.\n * @param isActiveResult Whether the element is part of the active search result.\n */\n private _applyStyles(element: HTMLElement, borderColor: string | undefined, isActiveResult: boolean): void {\n if (!element.classList.contains('xterm-find-result-decoration')) {\n element.classList.add('xterm-find-result-decoration');\n if (borderColor) {\n element.style.outline = `1px solid ${borderColor}`;\n }\n }\n if (isActiveResult) {\n element.classList.add('xterm-find-active-result-decoration');\n }\n }\n\n /**\n * Creates a decoration for the result and applies styles\n * @param result the search result for which to create the decoration\n * @param options the options for the decoration\n * @param isActiveResult whether this is the currently active result\n * @returns the decorations or undefined if the marker has already been disposed of\n */\n private _createResultDecorations(result: ISearchResult, options: ISearchDecorationOptions, isActiveResult: boolean): IDecoration[] | undefined {\n // Gather decoration ranges for this match as it could wrap\n const decorationRanges: [number, number, number][] = [];\n let currentCol = result.col;\n let remainingSize = result.size;\n let markerOffset = -this._terminal.buffer.active.baseY - this._terminal.buffer.active.cursorY + result.row;\n while (remainingSize > 0) {\n const amountThisRow = Math.min(this._terminal.cols - currentCol, remainingSize);\n decorationRanges.push([markerOffset, currentCol, amountThisRow]);\n currentCol = 0;\n remainingSize -= amountThisRow;\n markerOffset++;\n }\n\n // Create the decorations\n const decorations: IDecoration[] = [];\n for (const range of decorationRanges) {\n const marker = this._terminal.registerMarker(range[0]);\n const decoration = this._terminal.registerDecoration({\n marker,\n x: range[1],\n width: range[2],\n layer: isActiveResult ? 'top' : 'bottom',\n backgroundColor: isActiveResult ? options.activeMatchBackground : options.matchBackground,\n overviewRulerOptions: this._highlightedLines.has(marker.line) ? undefined : {\n color: isActiveResult ? options.activeMatchColorOverviewRuler : options.matchOverviewRuler,\n position: 'center'\n }\n });\n if (decoration) {\n const disposables: IDisposable[] = [];\n disposables.push(marker);\n disposables.push(decoration.onRender((e) => this._applyStyles(e, isActiveResult ? options.activeMatchBorder : options.matchBorder, false)));\n disposables.push(decoration.onDispose(() => dispose(disposables)));\n decorations.push(decoration);\n }\n }\n\n return decorations.length === 0 ? undefined : decorations;\n }\n}\n\n\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport type { ISearchOptions } from '@xterm/addon-search';\nimport type { SearchLineCache } from './SearchLineCache';\n\n/**\n * Represents the position to start a search from.\n */\ninterface ISearchPosition {\n startCol: number;\n startRow: number;\n}\n\n/**\n * Represents a search result with its position and content.\n */\nexport interface ISearchResult {\n term: string;\n col: number;\n row: number;\n size: number;\n}\n\n/**\n * Configuration constants for the search engine functionality.\n */\nconst enum Constants {\n /**\n * Characters that are considered non-word characters for search boundary detection. These\n * characters are used to determine word boundaries when performing whole-word searches. Includes\n * common punctuation, symbols, and whitespace characters.\n */\n NON_WORD_CHARACTERS = ' ~!@#$%^&*()+`-=[]{}|\\\\;:\"\\',./<>?'\n}\n\n/**\n * Core search engine that handles finding text within terminal content.\n * This class is responsible for the actual search algorithms and position calculations.\n */\nexport class SearchEngine {\n constructor(\n private readonly _terminal: Terminal,\n private readonly _lineCache: SearchLineCache\n ) {}\n\n /**\n * Find the first occurrence of a term starting from a specific position.\n * @param term The search term.\n * @param startRow The row to start searching from.\n * @param startCol The column to start searching from.\n * @param searchOptions Search options.\n * @returns The search result if found, undefined otherwise.\n */\n public find(term: string, startRow: number, startCol: number, searchOptions?: ISearchOptions): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n if (startCol >= this._terminal.cols) {\n throw new Error(`Invalid col: ${startCol} to search in terminal of ${this._terminal.cols} cols`);\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n return result;\n }\n\n /**\n * Find the next occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine incremental behavior.\n * @returns The search result if found, undefined otherwise.\n */\n public findNextWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startCol = 0;\n let startRow = 0;\n if (prevSelectedPos) {\n if (cachedSearchTerm === term) {\n startCol = prevSelectedPos.end.x;\n startRow = prevSelectedPos.end.y;\n } else {\n startCol = prevSelectedPos.start.x;\n startRow = prevSelectedPos.start.y;\n }\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n // If we hit the bottom and didn't search from the very top wrap back up\n if (!result && startRow !== 0) {\n for (let y = 0; y < startRow; y++) {\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n\n // If there is only one result, wrap back and return selection if it exists.\n if (!result && prevSelectedPos) {\n searchPosition.startRow = prevSelectedPos.start.y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n }\n\n return result;\n }\n\n /**\n * Find the previous occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine if expansion should occur.\n * @returns The search result if found, undefined otherwise.\n */\n public findPreviousWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startRow = this._terminal.buffer.active.baseY + this._terminal.rows - 1;\n const startCol = this._terminal.cols;\n const isReverseSearch = true;\n\n this._lineCache.initLinesCache();\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n let result: ISearchResult | undefined;\n if (prevSelectedPos) {\n searchPosition.startRow = startRow = prevSelectedPos.start.y;\n searchPosition.startCol = prevSelectedPos.start.x;\n if (cachedSearchTerm !== term) {\n // Try to expand selection to right first.\n result = this._findInLine(term, searchPosition, searchOptions, false);\n if (!result) {\n // If selection was not able to be expanded to the right, then try reverse search\n searchPosition.startRow = startRow = prevSelectedPos.end.y;\n searchPosition.startCol = prevSelectedPos.end.x;\n }\n }\n }\n\n result ??= this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n\n // Search from startRow - 1 to top\n if (!result) {\n searchPosition.startCol = Math.max(searchPosition.startCol, this._terminal.cols);\n for (let y = startRow - 1; y >= 0; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n // If we hit the top and didn't search from the very bottom wrap back down\n if (!result && startRow !== (this._terminal.buffer.active.baseY + this._terminal.rows - 1)) {\n for (let y = (this._terminal.buffer.active.baseY + this._terminal.rows - 1); y >= startRow; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n\n return result;\n }\n\n /**\n * A found substring is a whole word if it doesn't have an alphanumeric character directly\n * adjacent to it.\n * @param searchIndex starting index of the potential whole word substring\n * @param line entire string in which the potential whole word was found\n * @param term the substring that starts at searchIndex\n */\n private _isWholeWord(searchIndex: number, line: string, term: string): boolean {\n return ((searchIndex === 0) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex - 1]))) &&\n (((searchIndex + term.length) === line.length) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex + term.length])));\n }\n\n /**\n * Searches a line for a search term. Takes the provided terminal line and searches the text line,\n * which may contain subsequent terminal lines if the text is wrapped. If the provided line number\n * is part of a wrapped text line that started on an earlier line then it is skipped since it will\n * be properly searched when the terminal line that the text starts on is searched.\n * @param term The search term.\n * @param searchPosition The position to start the search.\n * @param searchOptions Search options.\n * @param isReverseSearch Whether the search should start from the right side of the terminal and\n * search to the left.\n * @returns The search result if it was found.\n */\n private _findInLine(term: string, searchPosition: ISearchPosition, searchOptions: ISearchOptions = {}, isReverseSearch: boolean = false): ISearchResult | undefined {\n const row = searchPosition.startRow;\n const col = searchPosition.startCol;\n\n // Ignore wrapped lines, only consider on unwrapped line (first row of command string).\n const firstLine = this._terminal.buffer.active.getLine(row);\n if (firstLine?.isWrapped) {\n if (isReverseSearch) {\n searchPosition.startCol += this._terminal.cols;\n return;\n }\n\n // This will iterate until we find the line start.\n // When we find it, we will search using the calculated start column.\n searchPosition.startRow--;\n searchPosition.startCol += this._terminal.cols;\n return this._findInLine(term, searchPosition, searchOptions);\n }\n let cache = this._lineCache.getLineFromCache(row);\n if (!cache) {\n cache = this._lineCache.translateBufferLineToStringWithWrap(row, true);\n this._lineCache.setLineInCache(row, cache);\n }\n const [stringLine, offsets] = cache;\n\n const offset = this._bufferColsToStringOffset(row, col);\n let searchTerm = term;\n let searchStringLine = stringLine;\n if (!searchOptions.regex) {\n searchTerm = searchOptions.caseSensitive ? term : term.toLowerCase();\n searchStringLine = searchOptions.caseSensitive ? stringLine : stringLine.toLowerCase();\n }\n\n let resultIndex = -1;\n if (searchOptions.regex) {\n const searchRegex = RegExp(searchTerm, searchOptions.caseSensitive ? 'g' : 'gi');\n let foundTerm: RegExpExecArray | null;\n if (isReverseSearch) {\n // This loop will get the resultIndex of the _last_ regex match in the range 0..offset\n while (foundTerm = searchRegex.exec(searchStringLine.slice(0, offset))) {\n resultIndex = searchRegex.lastIndex - foundTerm[0].length;\n term = foundTerm[0];\n searchRegex.lastIndex -= (term.length - 1);\n }\n } else {\n foundTerm = searchRegex.exec(searchStringLine.slice(offset));\n if (foundTerm && foundTerm[0].length > 0) {\n resultIndex = offset + (searchRegex.lastIndex - foundTerm[0].length);\n term = foundTerm[0];\n }\n }\n } else {\n if (isReverseSearch) {\n if (offset - searchTerm.length >= 0) {\n resultIndex = searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length);\n }\n } else {\n resultIndex = searchStringLine.indexOf(searchTerm, offset);\n }\n }\n\n if (resultIndex >= 0) {\n if (searchOptions.wholeWord && !this._isWholeWord(resultIndex, searchStringLine, term)) {\n return;\n }\n\n // Adjust the row number and search index if needed since a \"line\" of text can span multiple\n // rows\n let startRowOffset = 0;\n while (startRowOffset < offsets.length - 1 && resultIndex >= offsets[startRowOffset + 1]) {\n startRowOffset++;\n }\n let endRowOffset = startRowOffset;\n while (endRowOffset < offsets.length - 1 && resultIndex + term.length >= offsets[endRowOffset + 1]) {\n endRowOffset++;\n }\n const startColOffset = resultIndex - offsets[startRowOffset];\n const endColOffset = resultIndex + term.length - offsets[endRowOffset];\n const startColIndex = this._stringLengthToBufferSize(row + startRowOffset, startColOffset);\n const endColIndex = this._stringLengthToBufferSize(row + endRowOffset, endColOffset);\n const size = endColIndex - startColIndex + this._terminal.cols * (endRowOffset - startRowOffset);\n\n return {\n term,\n col: startColIndex,\n row: row + startRowOffset,\n size\n };\n }\n }\n\n private _stringLengthToBufferSize(row: number, offset: number): number {\n const line = this._terminal.buffer.active.getLine(row);\n if (!line) {\n return 0;\n }\n for (let i = 0; i < offset; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n // Adjust the searchIndex to normalize emoji into single chars\n const char = cell.getChars();\n if (char.length > 1) {\n offset -= char.length - 1;\n }\n // Adjust the searchIndex for empty characters following wide unicode\n // chars (eg. CJK)\n const nextCell = line.getCell(i + 1);\n if (nextCell && nextCell.getWidth() === 0) {\n offset++;\n }\n }\n return offset;\n }\n\n private _bufferColsToStringOffset(startRow: number, cols: number): number {\n let lineIndex = startRow;\n let offset = 0;\n let line = this._terminal.buffer.active.getLine(lineIndex);\n while (cols > 0 && line) {\n for (let i = 0; i < cols && i < this._terminal.cols; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n if (cell.getWidth()) {\n // Treat null characters as whitespace to align with the translateToString API\n offset += cell.getCode() === 0 ? 1 : cell.getChars().length;\n }\n }\n lineIndex++;\n line = this._terminal.buffer.active.getLine(lineIndex);\n if (line && !line.isWrapped) {\n break;\n }\n cols -= this._terminal.cols;\n }\n return offset;\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport { combinedDisposable, Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\n\nexport type LineCacheEntry = [\n /**\n * The string representation of a line (as opposed to the buffer cell representation).\n */\n lineAsString: string,\n /**\n * The offsets where each line starts when the entry describes a wrapped line.\n */\n lineOffsets: number[]\n];\n\n/**\n * Configuration constants for the search line cache functionality.\n */\nconst enum Constants {\n /**\n * Time-to-live for cached search results in milliseconds. After this duration, cached search\n * results will be invalidated to ensure they remain consistent with terminal content changes.\n */\n LINES_CACHE_TIME_TO_LIVE = 15000\n}\n\nexport class SearchLineCache extends Disposable {\n /**\n * translateBufferLineToStringWithWrap is a fairly expensive call.\n * We memoize the calls into an array that has a time based ttl.\n * _linesCache is also invalidated when the terminal cursor moves.\n */\n private _linesCache: LineCacheEntry[] | undefined;\n private _linesCacheTimeout = this._register(new MutableDisposable());\n private _linesCacheDisposables = this._register(new MutableDisposable());\n // Track access to avoid recreating a timeout on every init call which occurs once per search\n // result (findNext/findPrevious -> _highlightAllMatches -> find loop).\n private _lastAccessTimestamp = 0;\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this._destroyLinesCache()));\n }\n\n /**\n * Sets up a line cache with a ttl\n */\n public initLinesCache(): void {\n if (!this._linesCache) {\n this._linesCache = new Array(this._terminal.buffer.active.length);\n this._linesCacheDisposables.value = combinedDisposable(\n this._terminal.onLineFeed(() => this._destroyLinesCache()),\n this._terminal.onCursorMove(() => this._destroyLinesCache()),\n this._terminal.onResize(() => this._destroyLinesCache())\n );\n }\n\n this._lastAccessTimestamp = Date.now();\n if (!this._linesCacheTimeout.value) {\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE);\n }\n }\n\n private _destroyLinesCache(): void {\n this._linesCache = undefined;\n this._lastAccessTimestamp = 0;\n this._linesCacheDisposables.clear();\n this._linesCacheTimeout.clear();\n }\n\n private _scheduleLinesCacheTimeout(delay: number): void {\n this._linesCacheTimeout.value = disposableTimeout(() => {\n if (!this._linesCache) {\n return;\n }\n const now = Date.now();\n const elapsed = now - this._lastAccessTimestamp;\n if (elapsed >= Constants.LINES_CACHE_TIME_TO_LIVE) {\n this._destroyLinesCache();\n return;\n }\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE - elapsed);\n }, delay);\n }\n\n public getLineFromCache(row: number): LineCacheEntry | undefined {\n return this._linesCache?.[row];\n }\n\n public setLineInCache(row: number, entry: LineCacheEntry): void {\n if (this._linesCache) {\n this._linesCache[row] = entry;\n }\n }\n\n /**\n * Translates a buffer line to a string, including subsequent lines if they are wraps.\n * Wide characters will count as two columns in the resulting string. This\n * function is useful for getting the actual text underneath the raw selection\n * position.\n * @param lineIndex The index of the line being translated.\n * @param trimRight Whether to trim whitespace to the right.\n */\n public translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean): LineCacheEntry {\n const strings = [];\n const lineOffsets = [0];\n let line = this._terminal.buffer.active.getLine(lineIndex);\n while (line) {\n const nextLine = this._terminal.buffer.active.getLine(lineIndex + 1);\n const lineWrapsToNext = nextLine ? nextLine.isWrapped : false;\n let string = line.translateToString(!lineWrapsToNext && trimRight);\n if (lineWrapsToNext && nextLine) {\n const lastCell = line.getCell(line.length - 1);\n const lastCellIsNull = lastCell && lastCell.getCode() === 0 && lastCell.getWidth() === 1;\n // a wide character wrapped to the next line\n if (lastCellIsNull && nextLine.getCell(0)?.getWidth() === 2) {\n string = string.slice(0, -1);\n }\n }\n strings.push(string);\n if (lineWrapsToNext) {\n lineOffsets.push(lineOffsets[lineOffsets.length - 1] + string.length);\n } else {\n break;\n }\n lineIndex++;\n line = nextLine;\n }\n return [strings.join(''), lineOffsets];\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchResultChangeEvent } from '@xterm/addon-search';\nimport type { IDisposable } from '@xterm/xterm';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a currently selected decoration.\n */\ninterface ISelectedDecoration extends IDisposable {\n match: ISearchResult;\n}\n\n/**\n * Tracks search results, manages result indexing, and fires events when results change.\n * This class provides centralized management of search result state and notifications.\n */\nexport class SearchResultTracker extends Disposable {\n private _searchResults: ISearchResult[] = [];\n private _selectedDecoration: ISelectedDecoration | undefined;\n\n private readonly _onDidChangeResults = this._register(new Emitter());\n public get onDidChangeResults(): IEvent { return this._onDidChangeResults.event; }\n\n /**\n * Gets the current search results.\n */\n public get searchResults(): ReadonlyArray {\n return this._searchResults;\n }\n\n /**\n * Gets the currently selected decoration.\n */\n public get selectedDecoration(): ISelectedDecoration | undefined {\n return this._selectedDecoration;\n }\n\n /**\n * Sets the currently selected decoration.\n */\n public set selectedDecoration(decoration: ISelectedDecoration | undefined) {\n this._selectedDecoration = decoration;\n }\n\n /**\n * Updates the search results with a new set of results.\n * @param results The new search results.\n * @param maxResults The maximum number of results to track.\n */\n public updateResults(results: ISearchResult[], maxResults: number): void {\n this._searchResults = results.slice(0, maxResults);\n }\n\n /**\n * Clears all search results.\n */\n public clearResults(): void {\n this._searchResults = [];\n }\n\n /**\n * Clears the selected decoration.\n */\n public clearSelectedDecoration(): void {\n if (this._selectedDecoration) {\n this._selectedDecoration.dispose();\n this._selectedDecoration = undefined;\n }\n }\n\n /**\n * Finds the index of a result in the current results array.\n * @param result The result to find.\n * @returns The index of the result, or -1 if not found.\n */\n public findResultIndex(result: ISearchResult): number {\n for (let i = 0; i < this._searchResults.length; i++) {\n const match = this._searchResults[i];\n if (match.row === result.row && match.col === result.col && match.size === result.size) {\n return i;\n }\n }\n return -1;\n }\n\n /**\n * Fires a result change event with the current state.\n * @param hasDecorations Whether decorations are enabled.\n */\n public fireResultsChanged(hasDecorations: boolean): void {\n if (!hasDecorations) {\n return;\n }\n\n let resultIndex = -1;\n if (this._selectedDecoration) {\n resultIndex = this.findResultIndex(this._selectedDecoration.match);\n }\n\n this._onDidChangeResults.fire({\n resultIndex,\n resultCount: this._searchResults.length\n });\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this.clearSelectedDecoration();\n this.clearResults();\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchOptions } from '@xterm/addon-search';\n\n/**\n * Manages search state including cached search terms, options tracking, and validation.\n * This class provides a centralized way to handle search state consistency and option changes.\n */\nexport class SearchState {\n private _cachedSearchTerm: string | undefined;\n private _lastSearchOptions: ISearchOptions | undefined;\n\n /**\n * Gets the currently cached search term.\n */\n public get cachedSearchTerm(): string | undefined {\n return this._cachedSearchTerm;\n }\n\n /**\n * Sets the cached search term.\n */\n public set cachedSearchTerm(term: string | undefined) {\n this._cachedSearchTerm = term;\n }\n\n /**\n * Gets the last search options used.\n */\n public get lastSearchOptions(): ISearchOptions | undefined {\n return this._lastSearchOptions;\n }\n\n /**\n * Sets the last search options used.\n */\n public set lastSearchOptions(options: ISearchOptions | undefined) {\n this._lastSearchOptions = options;\n }\n\n /**\n * Validates a search term to ensure it's not empty or invalid.\n * @param term The search term to validate.\n * @returns true if the term is valid for searching.\n */\n public isValidSearchTerm(term: string): boolean {\n return !!(term && term.length > 0);\n }\n\n /**\n * Determines if search options have changed compared to the last search.\n * @param newOptions The new search options to compare.\n * @returns true if the options have changed.\n */\n public didOptionsChange(newOptions?: ISearchOptions): boolean {\n if (!this._lastSearchOptions) {\n return true;\n }\n if (!newOptions) {\n return false;\n }\n if (this._lastSearchOptions.caseSensitive !== newOptions.caseSensitive) {\n return true;\n }\n if (this._lastSearchOptions.regex !== newOptions.regex) {\n return true;\n }\n if (this._lastSearchOptions.wholeWord !== newOptions.wholeWord) {\n return true;\n }\n return false;\n }\n\n /**\n * Determines if a new search should trigger highlighting updates.\n * @param term The search term.\n * @param options The search options.\n * @returns true if highlighting should be updated.\n */\n public shouldUpdateHighlighting(term: string, options?: ISearchOptions): boolean {\n if (!options?.decorations) {\n return false;\n }\n return this._cachedSearchTerm === undefined ||\n term !== this._cachedSearchTerm ||\n this.didOptionsChange(options);\n }\n\n /**\n * Clears the cached search term.\n */\n public clearCachedTerm(): void {\n this._cachedSearchTerm = undefined;\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this._cachedSearchTerm = undefined;\n this._lastSearchOptions = undefined;\n }\n}\n","// The module cache\nvar __webpack_module_cache__ = {};\n\n// The require function\nfunction __webpack_require__(moduleId) {\n\t// Check if module is in cache\n\tvar cachedModule = __webpack_module_cache__[moduleId];\n\tif (cachedModule !== undefined) {\n\t\treturn cachedModule.exports;\n\t}\n\t// Create a new module (and put it into the cache)\n\tvar module = __webpack_module_cache__[moduleId] = {\n\t\t// no module.id needed\n\t\t// no module.loaded needed\n\t\texports: {}\n\t};\n\n\t// Execute the module function\n\t__webpack_modules__[moduleId](module, module.exports, __webpack_require__);\n\n\t// Return the exports of the module\n\treturn module.exports;\n}\n\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, ITerminalAddon } from '@xterm/xterm';\nimport type { SearchAddon as ISearchApi, ISearchOptions, ISearchAddonOptions, ISearchResultChangeEvent, ISearchDecorationOptions } from '@xterm/addon-search';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\nimport { SearchLineCache } from './SearchLineCache';\nimport { SearchState } from './SearchState';\nimport { SearchEngine, type ISearchResult } from './SearchEngine';\nimport { DecorationManager } from './DecorationManager';\nimport { SearchResultTracker } from './SearchResultTracker';\n\ninterface IInternalSearchOptions {\n noScroll: boolean;\n}\n\n/**\n * Configuration constants for the search addon functionality.\n */\nconst enum Constants {\n /**\n * Default maximum number of search results to highlight simultaneously. This limit prevents\n * performance degradation when searching for very common terms that would result in excessive\n * highlighting decorations.\n */\n DEFAULT_HIGHLIGHT_LIMIT = 1000\n}\n\nexport class SearchAddon extends Disposable implements ITerminalAddon, ISearchApi {\n private _terminal: Terminal | undefined;\n private _highlightLimit: number;\n private _highlightTimeout = this._register(new MutableDisposable());\n private _lineCache = this._register(new MutableDisposable());\n\n // Component instances\n private _state = new SearchState();\n private _engine: SearchEngine | undefined;\n private _decorationManager: DecorationManager | undefined;\n private _resultTracker = this._register(new SearchResultTracker());\n\n private readonly _onAfterSearch = this._register(new Emitter());\n public readonly onAfterSearch = this._onAfterSearch.event;\n private readonly _onBeforeSearch = this._register(new Emitter());\n public readonly onBeforeSearch = this._onBeforeSearch.event;\n\n public get onDidChangeResults(): IEvent {\n return this._resultTracker.onDidChangeResults;\n }\n\n constructor(options?: Partial) {\n super();\n\n this._highlightLimit = options?.highlightLimit ?? Constants.DEFAULT_HIGHLIGHT_LIMIT;\n }\n\n public activate(terminal: Terminal): void {\n this._terminal = terminal;\n this._lineCache.value = new SearchLineCache(terminal);\n this._engine = new SearchEngine(terminal, this._lineCache.value);\n this._decorationManager = new DecorationManager(terminal);\n this._register(this._terminal.onWriteParsed(() => this._updateMatches()));\n this._register(this._terminal.onResize(() => this._updateMatches()));\n this._register(toDisposable(() => this.clearDecorations()));\n }\n\n private _updateMatches(): void {\n this._highlightTimeout.clear();\n if (this._state.cachedSearchTerm && this._state.lastSearchOptions?.decorations) {\n this._highlightTimeout.value = disposableTimeout(() => {\n const term = this._state.cachedSearchTerm;\n this._state.clearCachedTerm();\n this.findPrevious(term!, { ...this._state.lastSearchOptions, incremental: true }, { noScroll: true });\n }, 200);\n }\n }\n\n public clearDecorations(retainCachedSearchTerm?: boolean): void {\n this._resultTracker.clearSelectedDecoration();\n this._decorationManager?.clearHighlightDecorations();\n this._resultTracker.clearResults();\n if (!retainCachedSearchTerm) {\n this._state.clearCachedTerm();\n }\n }\n\n public clearActiveDecoration(): void {\n this._resultTracker.clearSelectedDecoration();\n }\n\n /**\n * Find the next instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findNext(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findNextAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _highlightAllMatches(term: string, searchOptions: ISearchOptions): void {\n if (!this._terminal || !this._engine || !this._decorationManager) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n if (!this._state.isValidSearchTerm(term)) {\n this.clearDecorations();\n return;\n }\n\n // new search, clear out the old decorations\n this.clearDecorations(true);\n\n const results: ISearchResult[] = [];\n let prevResult: ISearchResult | undefined = undefined;\n let result = this._engine.find(term, 0, 0, searchOptions);\n\n while (result && (prevResult?.row !== result.row || prevResult?.col !== result.col)) {\n if (results.length >= this._highlightLimit) {\n break;\n }\n prevResult = result;\n results.push(prevResult);\n const cols = this._terminal.cols;\n let nextCol = prevResult.col + prevResult.size;\n let nextRow = prevResult.row;\n if (nextCol >= cols) {\n nextRow += Math.floor(nextCol / cols);\n nextCol = nextCol % cols;\n }\n result = this._engine.find(term, nextRow, nextCol, searchOptions);\n }\n\n this._resultTracker.updateResults(results, this._highlightLimit);\n if (searchOptions.decorations) {\n this._decorationManager.createHighlightDecorations(results, searchOptions.decorations);\n }\n }\n\n private _findNextAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findNextWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Find the previous instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findPrevious(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findPreviousAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _fireResults(searchOptions?: ISearchOptions): void {\n this._resultTracker.fireResultsChanged(!!searchOptions?.decorations);\n }\n\n private _findPreviousAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findPreviousWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Selects and scrolls to a result.\n * @param result The result to select.\n * @returns Whether a result was selected.\n */\n private _selectResult(result: ISearchResult | undefined, options?: ISearchDecorationOptions, noScroll?: boolean): boolean {\n if (!this._terminal || !this._decorationManager) {\n return false;\n }\n\n this._resultTracker.clearSelectedDecoration();\n if (!result) {\n this._terminal.clearSelection();\n return false;\n }\n\n this._terminal.select(result.col, result.row, result.size);\n if (options) {\n const activeDecoration = this._decorationManager.createActiveDecoration(result, options);\n if (activeDecoration) {\n this._resultTracker.selectedDecoration = activeDecoration;\n }\n }\n\n if (!noScroll) {\n // If it is not in the viewport then we scroll else it just gets selected\n if (result.row >= (this._terminal.buffer.active.viewportY + this._terminal.rows) || result.row < this._terminal.buffer.active.viewportY) {\n let scroll = result.row - this._terminal.buffer.active.viewportY;\n scroll -= Math.floor(this._terminal.rows / 2);\n this._terminal.scrollLines(scroll);\n }\n }\n return true;\n }\n}\n"],"names":["root","factory","exports","module","define","amd","globalThis","millis","Promise","resolve","setTimeout","handler","timeout","store","timer","disposable","dispose","Lifecycle_1","toDisposable","clearTimeout","add","__webpack_require__","constructor","this","_token","_isDisposed","cancel","cancelAndSet","runner","Error","setIfNotSet","_isScheduled","set","queueMicrotask","_disposable","undefined","interval","context","handle","setInterval","clearInterval","EventUtils","_listeners","_disposed","event","_event","listener","thisArgs","disposables","entry","fn","slice","push","result","idx","indexOf","splice","Array","isArray","fire","length","call","listeners","i","len","forward","from","to","e","map","any","events","DisposableStore","runAndSubscribe","initial","arg","d","_disposables","Set","isDisposed","o","clear","Disposable","_store","_register","None","Object","freeze","value","_value","DecorationManager","_terminal","super","_highlightDecorations","_highlightedLines","clearHighlightDecorations","createHighlightDecorations","results","options","match","decorations","_createResultDecorations","decoration","_storeDecoration","createActiveDecoration","marker","line","_applyStyles","element","borderColor","isActiveResult","classList","contains","style","outline","decorationRanges","currentCol","col","remainingSize","size","markerOffset","buffer","active","baseY","cursorY","row","amountThisRow","Math","min","cols","range","registerMarker","registerDecoration","x","width","layer","backgroundColor","activeMatchBackground","matchBackground","overviewRulerOptions","has","color","activeMatchColorOverviewRuler","matchOverviewRuler","position","onRender","activeMatchBorder","matchBorder","onDispose","_lineCache","find","term","startRow","startCol","searchOptions","clearSelection","initLinesCache","searchPosition","_findInLine","y","rows","findNextWithSelection","cachedSearchTerm","prevSelectedPos","getSelectionPosition","end","start","findPreviousWithSelection","isReverseSearch","max","_isWholeWord","searchIndex","includes","firstLine","getLine","isWrapped","cache","getLineFromCache","translateBufferLineToStringWithWrap","setLineInCache","stringLine","offsets","offset","_bufferColsToStringOffset","searchTerm","searchStringLine","regex","caseSensitive","toLowerCase","resultIndex","searchRegex","RegExp","foundTerm","exec","lastIndex","lastIndexOf","wholeWord","startRowOffset","endRowOffset","startColOffset","endColOffset","startColIndex","_stringLengthToBufferSize","cell","getCell","char","getChars","nextCell","getWidth","lineIndex","getCode","Async_1","SearchLineCache","_linesCacheTimeout","MutableDisposable","_linesCacheDisposables","_lastAccessTimestamp","_destroyLinesCache","_linesCache","combinedDisposable","onLineFeed","onCursorMove","onResize","Date","now","_scheduleLinesCacheTimeout","delay","disposableTimeout","elapsed","trimRight","strings","lineOffsets","nextLine","lineWrapsToNext","string","translateToString","lastCell","join","Event_1","SearchResultTracker","_searchResults","_onDidChangeResults","Emitter","onDidChangeResults","searchResults","selectedDecoration","_selectedDecoration","updateResults","maxResults","clearResults","clearSelectedDecoration","findResultIndex","fireResultsChanged","hasDecorations","resultCount","reset","_cachedSearchTerm","lastSearchOptions","_lastSearchOptions","isValidSearchTerm","didOptionsChange","newOptions","shouldUpdateHighlighting","clearCachedTerm","__webpack_module_cache__","moduleId","cachedModule","__webpack_modules__","SearchLineCache_1","SearchState_1","SearchEngine_1","DecorationManager_1","SearchResultTracker_1","SearchAddon","_resultTracker","_highlightTimeout","_state","SearchState","_onAfterSearch","onAfterSearch","_onBeforeSearch","onBeforeSearch","_highlightLimit","highlightLimit","activate","terminal","_engine","SearchEngine","_decorationManager","onWriteParsed","_updateMatches","clearDecorations","findPrevious","incremental","noScroll","retainCachedSearchTerm","clearActiveDecoration","findNext","internalSearchOptions","_highlightAllMatches","found","_findNextAndSelect","_fireResults","prevResult","nextCol","nextRow","floor","_selectResult","_findPreviousAndSelect","select","activeDecoration","viewportY","scroll","scrollLines"],"sourceRoot":""} +\ No newline at end of file ++{"version":3,"file":"addon-search.js","mappings":"CAAA,SAAAA,EAAAC,GACA,iBAAAC,SAAA,iBAAAC,OACAA,OAAAD,QAAAD,IACA,mBAAAG,QAAAA,OAAAC,IACAD,OAAA,GAAAH,GACA,iBAAAC,QACAA,QAAA,YAAAD,IAEAD,EAAA,YAAAC,GACC,CATD,CASCK,WAAA,2JCAD,SAAwBC,GACtB,OAAO,IAAIC,QAAQC,GAAWC,WAAWD,EAASF,GACpD,sBASA,SAAkCI,EAAqBC,EAAU,EAAGC,GAClE,MAAMC,EAAQJ,WAAW,KACvBC,IACIE,GACFE,EAAWC,WAEZJ,GACGG,GAAa,EAAAE,EAAAC,cAAa,KAC9BC,aAAaL,KAGf,OADAD,GAAOO,IAAIL,GACJA,CACT,EAzBA,MAAAE,EAAAI,EAAA,oBA2BA,iBAAAC,GACUC,KAAAC,QAAe,EACfD,KAAAE,aAAc,CAqCxB,CAnCS,OAAAT,GACLO,KAAKG,SACLH,KAAKE,aAAc,CACrB,CAEO,MAAAC,IACgB,IAAjBH,KAAKC,SACPL,aAAaI,KAAKC,QAClBD,KAAKC,QAAU,EAEnB,CAEO,YAAAG,CAAaC,EAAoBhB,GACtC,GAAIW,KAAKE,YACP,MAAM,IAAII,MAAM,mDAElBN,KAAKG,SACLH,KAAKC,OAASd,WAAW,KACvBa,KAAKC,QAAU,EACfI,KACChB,EACL,CAEO,WAAAkB,CAAYF,EAAoBhB,GACrC,GAAIW,KAAKE,YACP,MAAM,IAAII,MAAM,mDAEG,IAAjBN,KAAKC,SAGTD,KAAKC,OAASd,WAAW,KACvBa,KAAKC,QAAU,EACfI,KACChB,GACL,oBAQF,iBAAAU,GACUC,KAAAQ,cAAe,EACfR,KAAAE,aAAc,CA2BxB,CAzBS,OAAAT,GACLO,KAAKG,SACLH,KAAKE,aAAc,CACrB,CAEO,MAAAC,GACLH,KAAKQ,cAAe,CACtB,CAEO,GAAAC,CAAIJ,GACT,GAAIL,KAAKE,YACP,MAAM,IAAII,MAAM,4CAEdN,KAAKQ,eAGTR,KAAKQ,cAAe,EACpBE,eAAe,KACRV,KAAKQ,eAGVR,KAAKQ,cAAe,EACpBH,OAEJ,mBAGF,iBAAAN,GAEUC,KAAAE,aAAc,CA2BxB,CAzBS,MAAAC,GACLH,KAAKW,aAAalB,UAClBO,KAAKW,iBAAcC,CACrB,CAEO,YAAAR,CAAaC,EAAoBQ,EAAkBC,EAAsC/B,YAC9F,GAAIiB,KAAKE,YACP,MAAM,IAAII,MAAM,oDAElBN,KAAKG,SACL,MAAMY,EAASD,EAAQE,YAAY,KACjCX,KACCQ,GACHb,KAAKW,YAAc,CACjBlB,QAAS,KACPqB,EAAQG,cAAcF,GACtBf,KAAKW,iBAAcC,GAGzB,CAEO,OAAAnB,GACLO,KAAKG,SACLH,KAAKE,aAAc,CACrB,8FCnIF,MAAAR,EAAAI,EAAA,KAoEA,IAAiBoB,YA9DjB,iBAAAnB,GACUC,KAAAmB,WAAqD,GACrDnB,KAAAoB,WAAY,CA0DtB,CAvDE,SAAWC,GACT,OAAIrB,KAAKsB,SAGTtB,KAAKsB,OAAS,CAACC,EAAyBC,EAAgBC,KACtD,GAAIzB,KAAKoB,UACP,OAAO,EAAA1B,EAAAC,cAAa,QAGtB,MAAM+B,EAAQ,CAAEC,GAAIJ,EAAUC,YAC9BxB,KAAKmB,WAAanB,KAAKmB,WAAWS,QAClC5B,KAAKmB,WAAWU,KAAKH,GAErB,MAAMI,GAAS,EAAApC,EAAAC,cAAa,KAC1B,MAAMoC,EAAM/B,KAAKmB,WAAWa,QAAQN,IACvB,IAATK,IACF/B,KAAKmB,WAAanB,KAAKmB,WAAWS,QAClC5B,KAAKmB,WAAWc,OAAOF,EAAK,MAYhC,OARIN,IACES,MAAMC,QAAQV,GAChBA,EAAYI,KAAKC,GAEjBL,EAAY5B,IAAIiC,IAIbA,IA3BA9B,KAAKsB,MA8BhB,CAEO,IAAAc,CAAKf,GACV,GAAIrB,KAAKoB,YAAcpB,KAAKmB,WAAWkB,OACrC,OAEF,GAA+B,IAA3BrC,KAAKmB,WAAWkB,OAElB,YADArC,KAAKmB,WAAW,GAAGQ,GAAGW,KAAKtC,KAAKmB,WAAW,GAAGK,SAAUH,GAG1D,MAAMkB,EAAYvC,KAAKmB,WACvB,IAAK,IAAIqB,EAAI,EAAGC,EAAMF,EAAUF,OAAQG,EAAIC,IAAOD,EACjDD,EAAUC,GAAGb,GAAGW,KAAKC,EAAUC,GAAGhB,SAAUH,EAEhD,CAEO,OAAA5B,GACDO,KAAKoB,YAGTpB,KAAKoB,WAAY,EACjBpB,KAAKmB,WAAWkB,OAAS,EAC3B,GAGF,SAAiBnB,GACCA,EAAAwB,QAAhB,SAA2BC,EAAiBC,GAC1C,OAAOD,EAAKE,GAAKD,EAAGR,KAAKS,GAC3B,EAEgB3B,EAAA4B,IAAhB,SAA0BzB,EAAkByB,GAC1C,MAAO,CAACvB,EAAyBC,EAAgBC,IACxCJ,EAAMmB,GAAKjB,EAASe,KAAKd,EAAUsB,EAAIN,SAAK5B,EAAWa,EAElE,EAIgBP,EAAA6B,IAAhB,YAA0BC,GACxB,MAAO,CAACzB,EAAyBC,EAAgBC,KAC/C,MAAMnC,EAAQ,IAAII,EAAAuD,gBAClB,IAAK,MAAM5B,KAAS2B,EAClB1D,EAAMO,IAAIwB,EAAMwB,GAAKtB,EAASe,KAAKd,EAAUqB,KAS/C,OAPIpB,IACES,MAAMC,QAAQV,GAChBA,EAAYI,KAAKvC,GAEjBmC,EAAY5B,IAAIP,IAGbA,EAEX,EAIgB4B,EAAAgC,gBAAhB,SAAmC7B,EAAkBjC,EAAqC+D,GAExF,OADA/D,EAAQ+D,GACD9B,EAAMwB,GAAKzD,EAAQyD,GAC5B,CACD,CApCD,CAAiB3B,IAAUvC,EAAAuC,WAAVA,EAAU,eChE3B,SAAAvB,EAA6BgC,GAC3B,MAAO,CAAElC,QAASkC,EACpB,CAKA,SAAAlC,EAA+C2D,GAC7C,IAAKA,EACH,OAAOA,EAET,GAAIlB,MAAMC,QAAQiB,GAAM,CACtB,IAAK,MAAMC,KAAKD,EACdC,EAAE5D,UAEJ,MAAO,EACT,CAEA,OADA2D,EAAI3D,UACG2D,CACT,8JAEA,YAAsC3B,GACpC,OAAO9B,EAAa,IAAMF,EAAQgC,GACpC,EAEA,MAAAwB,EAAA,WAAAlD,GACmBC,KAAAsD,aAAe,IAAIC,IAC5BvD,KAAAE,aAAc,CAgCxB,CA9BE,cAAWsD,GACT,OAAOxD,KAAKE,WACd,CAEO,GAAAL,CAA2B4D,GAMhC,OALIzD,KAAKE,YACPuD,EAAEhE,UAEFO,KAAKsD,aAAazD,IAAI4D,GAEjBA,CACT,CAEO,OAAAhE,GACL,IAAIO,KAAKE,YAAT,CAGAF,KAAKE,aAAc,EACnB,IAAK,MAAMmD,KAAKrD,KAAKsD,aACnBD,EAAE5D,UAEJO,KAAKsD,aAAaI,OALlB,CAMF,CAEO,KAAAA,GACL,IAAK,MAAML,KAAKrD,KAAKsD,aACnBD,EAAE5D,UAEJO,KAAKsD,aAAaI,OACpB,sBAGF,MAAAC,EAAA,WAAA5D,GAGqBC,KAAA4D,OAAS,IAAIX,CASlC,CAPS,OAAAxD,GACLO,KAAK4D,OAAOnE,SACd,CAEU,SAAAoE,CAAiCJ,GACzC,OAAOzD,KAAK4D,OAAO/D,IAAI4D,EACzB,iBAVuBE,EAAAG,KAAoBC,OAAOC,OAAO,CAAE,OAAAvE,GAAY,wBAazE,iBAAAM,GAEUC,KAAAE,aAAc,CAuBxB,CArBE,SAAW+D,GACT,OAAOjE,KAAKE,iBAAcU,EAAYZ,KAAKkE,MAC7C,CAEA,SAAWD,CAAMA,GACXjE,KAAKE,aAAe+D,IAAUjE,KAAKkE,SAGvClE,KAAKkE,QAAQzE,UACbO,KAAKkE,OAASD,EAChB,CAEO,KAAAP,GACL1D,KAAKiE,WAAQrD,CACf,CAEO,OAAAnB,GACLO,KAAKE,aAAc,EACnBF,KAAKkE,QAAQzE,UACbO,KAAKkE,YAAStD,CAChB,2FCxGF,MAAAlB,EAAAI,EAAA,KAuBA,MAAAqE,UAAuCzE,EAAAiE,WAIrC,WAAA5D,CAA6BqE,GAC3BC,QAD2BrE,KAAAoE,UAAAA,EAHrBpE,KAAAsE,sBAAsC,GACtCtE,KAAAuE,kBAAiC,IAAIhB,IAI3CvD,KAAK6D,WAAU,EAAAnE,EAAAC,cAAa,IAAMK,KAAKwE,6BACzC,CAOO,0BAAAC,CAA2BC,EAA0BC,GAC1D3E,KAAKwE,4BAEL,IAAK,MAAMI,KAASF,EAAS,CAC3B,MAAMG,EAAc7E,KAAK8E,yBAAyBF,EAAOD,GAAS,GAClE,GAAIE,EACF,IAAK,MAAME,KAAcF,EACvB7E,KAAKgF,iBAAiBD,EAAYH,EAGxC,CACF,CAQO,sBAAAK,CAAuBnD,EAAuB6C,GACnD,MAAME,EAAc7E,KAAK8E,yBAAyBhD,EAAQ6C,GAAS,GACnE,GAAIE,EACF,MAAO,CAAEA,cAAaD,MAAO9C,EAAQ,OAAArC,IAAY,EAAAC,EAAAD,SAAQoF,EAAc,EAG3E,CAKO,yBAAAL,IACL,EAAA9E,EAAAD,SAAQO,KAAKsE,uBACbtE,KAAKsE,sBAAwB,GAC7BtE,KAAKuE,kBAAkBb,OACzB,CAOQ,gBAAAsB,CAAiBD,EAAyBH,GAChD5E,KAAKuE,kBAAkB1E,IAAIkF,EAAWG,OAAOC,MAC7CnF,KAAKsE,sBAAsBzC,KAAK,CAAEkD,aAAYH,QAAO,OAAAnF,GAAYsF,EAAWtF,SAAW,GACzF,CAQQ,YAAA2F,CAAaC,EAAsBC,EAAiCC,GACrEF,EAAQG,UAAUC,SAAS,kCAC9BJ,EAAQG,UAAU3F,IAAI,gCAClByF,IACFD,EAAQK,MAAMC,QAAU,aAAaL,MAGrCC,GACFF,EAAQG,UAAU3F,IAAI,sCAE1B,CASQ,wBAAAiF,CAAyBhD,EAAuB6C,EAAmCY,GAEzF,MAAMK,EAA+C,GACrD,IAAIC,EAAa/D,EAAOgE,IACpBC,EAAgBjE,EAAOkE,KACvBC,GAAgBjG,KAAKoE,UAAU8B,OAAOC,OAAOC,MAAQpG,KAAKoE,UAAU8B,OAAOC,OAAOE,QAAUvE,EAAOwE,IACvG,KAAOP,EAAgB,GAAG,CACxB,MAAMQ,EAAgBC,KAAKC,IAAIzG,KAAKoE,UAAUsC,KAAOb,EAAYE,GACjEH,EAAiB/D,KAAK,CAACoE,EAAcJ,EAAYU,IACjDV,EAAa,EACbE,GAAiBQ,EACjBN,GACF,CAGA,MAAMpB,EAA6B,GACnC,IAAK,MAAM8B,KAASf,EAAkB,CACpC,MAAMV,EAASlF,KAAKoE,UAAUwC,eAAeD,EAAM,IAC7C5B,EAAa/E,KAAKoE,UAAUyC,mBAAmB,CACnD3B,SACA4B,EAAGH,EAAM,GACTI,MAAOJ,EAAM,GACbK,MAAOzB,EAAiB,MAAQ,SAChC0B,gBAAiB1B,EAAiBZ,EAAQuC,sBAAwBvC,EAAQwC,gBAC1EC,qBAAsBpH,KAAKuE,kBAAkB8C,IAAInC,EAAOC,WAAQvE,EAAY,CAC1E0G,MAAO/B,EAAiBZ,EAAQ4C,8BAAgC5C,EAAQ6C,mBACxEC,SAAU,YAGd,GAAI1C,EAAY,CACd,MAAMtD,EAA6B,GACnCA,EAAYI,KAAKqD,GACjBzD,EAAYI,KAAKkD,EAAW2C,SAAU7E,GAAM7C,KAAKoF,aAAavC,EAAG0C,EAAiBZ,EAAQgD,kBAAoBhD,EAAQiD,aAAa,KACnInG,EAAYI,KAAKkD,EAAW8C,UAAU,KAAM,EAAAnI,EAAAD,SAAQgC,KACpDoD,EAAYhD,KAAKkD,EACnB,CACF,CAEA,OAA8B,IAAvBF,EAAYxC,YAAezB,EAAYiE,CAChD,wHC/GF,MACE,WAAA9E,CACmBqE,EACA0D,kBADA1D,kBACA0D,CAChB,CAUI,IAAAC,CAAKC,EAAcC,EAAkBC,EAAkBC,GAC5D,IAAKH,GAAwB,IAAhBA,EAAK3F,OAEhB,YADArC,KAAKoE,UAAUgE,iBAGjB,GAAIF,GAAYlI,KAAKoE,UAAUsC,KAC7B,MAAM,IAAIpG,MAAM,gBAAgB4H,8BAAqClI,KAAKoE,UAAUsC,aAGtF1G,KAAK8H,WAAWO,iBAEhB,MAAMC,EAAkC,CACtCL,WACAC,YAIF,IAAIpG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,GAEpD,IAAKrG,EACH,IAAK,IAAI0G,EAAIP,EAAW,EAAGO,EAAIxI,KAAKoE,UAAU8B,OAAOC,OAAOC,MAAQpG,KAAKoE,UAAUqE,OAC7EzI,KAAK0I,6BAA6BF,KAGtCF,EAAeL,SAAWO,EAC1BF,EAAeJ,SAAW,EAC1BpG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,IAC5CrG,IAPmF0G,KAY3F,OAAO1G,CACT,CASO,qBAAA6G,CAAsBX,EAAcG,EAAgCS,GACzE,IAAKZ,GAAwB,IAAhBA,EAAK3F,OAEhB,YADArC,KAAKoE,UAAUgE,iBAIjB,MAAMS,EAAkB7I,KAAKoE,UAAU0E,uBACvC9I,KAAKoE,UAAUgE,iBAEf,IAAIF,EAAW,EACXD,EAAW,EACXY,IACED,IAAqBZ,GACvBE,EAAWW,EAAgBE,IAAIjC,EAC/BmB,EAAWY,EAAgBE,IAAIP,IAE/BN,EAAWW,EAAgBG,MAAMlC,EACjCmB,EAAWY,EAAgBG,MAAMR,IAIrCxI,KAAK8H,WAAWO,iBAEhB,MAAMC,EAAkC,CACtCL,WACAC,YAIF,IAAIpG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,GAEpD,IAAKrG,EACH,IAAK,IAAI0G,EAAIP,EAAW,EAAGO,EAAIxI,KAAKoE,UAAU8B,OAAOC,OAAOC,MAAQpG,KAAKoE,UAAUqE,OAC7EzI,KAAK0I,6BAA6BF,KAGtCF,EAAeL,SAAWO,EAC1BF,EAAeJ,SAAW,EAC1BpG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,IAC5CrG,IAPmF0G,KAa3F,IAAK1G,GAAuB,IAAbmG,EACb,IAAK,IAAIO,EAAI,EAAGA,EAAIP,IAGdO,EAAI,GAAKxI,KAAK0I,6BAA6BF,KAG/CF,EAAeL,SAAWO,EAC1BF,EAAeJ,SAAW,EAC1BpG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,IAC5CrG,IATwB0G,KAsBhC,OANK1G,GAAU+G,IACbP,EAAeL,SAAWY,EAAgBG,MAAMR,EAChDF,EAAeJ,SAAW,EAC1BpG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,IAG3CrG,CACT,CASO,yBAAAmH,CAA0BjB,EAAcG,EAAgCS,GAC7E,IAAKZ,GAAwB,IAAhBA,EAAK3F,OAEhB,YADArC,KAAKoE,UAAUgE,iBAIjB,MAAMS,EAAkB7I,KAAKoE,UAAU0E,uBACvC9I,KAAKoE,UAAUgE,iBAEf,IAAIH,EAAWjI,KAAKoE,UAAU8B,OAAOC,OAAOC,MAAQpG,KAAKoE,UAAUqE,KAAO,EAC1E,MAAMP,EAAWlI,KAAKoE,UAAUsC,KAC1BwC,GAAkB,EAExBlJ,KAAK8H,WAAWO,iBAChB,MAAMC,EAAkC,CACtCL,WACAC,YAGF,IAAIpG,EAkBJ,GAjBI+G,IACFP,EAAeL,SAAWA,EAAWY,EAAgBG,MAAMR,EAC3DF,EAAeJ,SAAWW,EAAgBG,MAAMlC,EAC5C8B,IAAqBZ,IAEvBlG,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,GAAe,GAC1DrG,IAEHwG,EAAeL,SAAWA,EAAWY,EAAgBE,IAAIP,EACzDF,EAAeJ,SAAWW,EAAgBE,IAAIjC,KAKpDhF,IAAW9B,KAAKuI,YAAYP,EAAMM,EAAgBH,EAAee,IAG5DpH,EAAQ,CACXwG,EAAeJ,SAAW1B,KAAK2C,IAAIb,EAAeJ,SAAUlI,KAAKoE,UAAUsC,MAC3E,IAAK,IAAI8B,EAAIP,EAAW,EAAGO,GAAK,IAC9BF,EAAeL,SAAWO,EAC1B1G,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,EAAee,IAC3DpH,GAH6B0G,KAOrC,CAEA,IAAK1G,GAAUmG,IAAcjI,KAAKoE,UAAU8B,OAAOC,OAAOC,MAAQpG,KAAKoE,UAAUqE,KAAO,EACtF,IAAK,IAAID,EAAKxI,KAAKoE,UAAU8B,OAAOC,OAAOC,MAAQpG,KAAKoE,UAAUqE,KAAO,EAAID,GAAKP,IAChFK,EAAeL,SAAWO,EAC1B1G,EAAS9B,KAAKuI,YAAYP,EAAMM,EAAgBH,EAAee,IAC3DpH,GAHsF0G,KAS9F,OAAO1G,CACT,CASQ,YAAAsH,CAAaC,EAAqBlE,EAAc6C,GACtD,OAAyB,IAAhBqB,GAAuB,qCAA8BC,SAASnE,EAAKkE,EAAc,OACrFA,EAAcrB,EAAK3F,SAAY8C,EAAK9C,QAAY,qCAA8BiH,SAASnE,EAAKkE,EAAcrB,EAAK3F,SACtH,CAGQ,mBAAAkH,CAAoBF,EAAqBlE,EAAc6C,EAAcG,GAC3E,OAAQA,EAAcqB,WAAaxJ,KAAKoJ,aAAaC,EAAalE,EAAM6C,EAC1E,CASQ,4BAAAU,CAA6BpC,GACnC,OAAgE,IAAzDtG,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQnD,IAAMoD,SACpD,CAcQ,WAAAnB,CAAYP,EAAcM,EAAiCH,EAAgC,GAAIe,GAA2B,GAEhI,GAAIA,GAGF,GAAIZ,EAAeL,SAAW,GAAKjI,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQnB,EAAeL,WAAWyB,UAEhG,YADApB,EAAeJ,UAAYlI,KAAKoE,UAAUsC,WAO5C,KAAO4B,EAAeL,SAAW,GAAKjI,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQnB,EAAeL,WAAWyB,WACnGpB,EAAeL,WACfK,EAAeJ,UAAYlI,KAAKoE,UAAUsC,KAG9C,MAAMJ,EAAMgC,EAAeL,SACrBnC,EAAMwC,EAAeJ,SAE3B,IAAIyB,EAAQ3J,KAAK8H,WAAW8B,iBAAiBtD,GACxCqD,IACHA,EAAQ3J,KAAK8H,WAAW+B,oCAAoCvD,GAAK,GACjEtG,KAAK8H,WAAWgC,eAAexD,EAAKqD,IAEtC,MAAOI,EAAYC,GAAWL,EAExBM,EAASjK,KAAKkK,0BAA0B5D,EAAKR,EAAKkE,GACxD,IAAIG,EAAanC,EACboC,EAAmBL,EAClB5B,EAAckC,QACjBF,EAAahC,EAAcmC,cAAgBtC,EAAOA,EAAKuC,cACvDH,EAAmBjC,EAAcmC,cAAgBP,EAAaA,EAAWQ,eAG3E,IAAIC,GAAe,EACnB,GAAIrC,EAAckC,MAAO,CACvB,MAAMI,EAAcC,OAAOP,EAAYhC,EAAcmC,cAAgB,IAAM,MAC3E,IAAIK,EACJ,GAAIzB,EAEF,KAAOyB,EAAYF,EAAYG,KAAKR,EAAiBxI,MAAM,EAAGqI,KAAU,CACtE,MAAMY,EAAaJ,EAAYK,UAAYH,EAAU,GAAGtI,OACpDsI,EAAU,GAAGtI,OAAS,GAAKrC,KAAKuJ,oBAAoBsB,EAAYT,EAAkBO,EAAU,GAAIxC,KAClGqC,EAAcK,EACd7C,EAAO2C,EAAU,IAEnBF,EAAYK,UAAYD,EAAa,CACvC,MAOA,IADAJ,EAAYK,UAAYb,EACjBU,EAAYF,EAAYG,KAAKR,IAAmB,CACrD,MAAMS,EAAaJ,EAAYK,UAAYH,EAAU,GAAGtI,OACxD,GAAIsI,EAAU,GAAGtI,OAAS,GAAKrC,KAAKuJ,oBAAoBsB,EAAYT,EAAkBO,EAAU,GAAIxC,GAAgB,CAClHqC,EAAcK,EACd7C,EAAO2C,EAAU,GACjB,KACF,CAEAF,EAAYK,UAAYD,EAAa,CACvC,CAEJ,MAAO,GAAI3B,EAAiB,CAC1B,IAAI2B,EAAaZ,EAASE,EAAW9H,QAAU,EAAI+H,EAAiBW,YAAYZ,EAAYF,EAASE,EAAW9H,SAAW,EAE3H,KAAOwI,GAAc,IAAM7K,KAAKuJ,oBAAoBsB,EAAYT,EAAkBD,EAAYhC,IAC5F0C,EAAaA,EAAa,EAAIT,EAAiBW,YAAYZ,EAAYU,EAAa,IAAM,EAE5FL,EAAcK,CAChB,KAAO,CACL,IAAIA,EAAaT,EAAiBpI,QAAQmI,EAAYF,GACtD,KAAOY,GAAc,IAAM7K,KAAKuJ,oBAAoBsB,EAAYT,EAAkBD,EAAYhC,IAC5F0C,EAAaT,EAAiBpI,QAAQmI,EAAYU,EAAa,GAEjEL,EAAcK,CAChB,CAEA,GAAIL,GAAe,EAAG,CAGpB,IAAIQ,EAAiB,EACrB,KAAOA,EAAiBhB,EAAQ3H,OAAS,GAAKmI,GAAeR,EAAQgB,EAAiB,IACpFA,IAEF,IAAIC,EAAeD,EACnB,KAAOC,EAAejB,EAAQ3H,OAAS,GAAKmI,EAAcxC,EAAK3F,QAAU2H,EAAQiB,EAAe,IAC9FA,IAEF,MAAMC,EAAiBV,EAAcR,EAAQgB,GACvCG,EAAeX,EAAcxC,EAAK3F,OAAS2H,EAAQiB,GACnDG,EAAgBpL,KAAKqL,0BAA0B/E,EAAM0E,EAAgBE,GAI3E,MAAO,CACLlD,OACAlC,IAAKsF,EACL9E,IAAKA,EAAM0E,EACXhF,KAPkBhG,KAAKqL,0BAA0B/E,EAAM2E,EAAcE,GAC5CC,EAAgBpL,KAAKoE,UAAUsC,MAAQuE,EAAeD,GAQnF,CACF,CAEQ,yBAAAK,CAA0B/E,EAAa2D,GAC7C,MAAM9E,EAAOnF,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQnD,GAClD,IAAKnB,EACH,OAAO,EAET,IAAK,IAAI3C,EAAI,EAAGA,EAAIyH,EAAQzH,IAAK,CAC/B,MAAM8I,EAAOnG,EAAKoG,QAAQ/I,GAC1B,IAAK8I,EACH,MAGF,MAAME,EAAOF,EAAKG,WACdD,EAAKnJ,OAAS,IAChB4H,GAAUuB,EAAKnJ,OAAS,GAI1B,MAAMqJ,EAAWvG,EAAKoG,QAAQ/I,EAAI,GAC9BkJ,GAAoC,IAAxBA,EAASC,YACvB1B,GAEJ,CACA,OAAOA,CACT,CAUQ,yBAAAC,CAA0BjC,EAAkBvB,EAAckF,GAChE,MAAMC,EAAWrF,KAAKC,IAAID,KAAKsF,MAAMpF,EAAO1G,KAAKoE,UAAUsC,MAAOkF,EAAYvJ,OAAS,GACvF,IAAI4H,EAAS2B,EAAYC,GACzB,MAAM1G,EAAOnF,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQxB,EAAW4D,GAC7D,GAAI1G,EAAM,CACR,MAAM4G,EAAYvF,KAAKC,IAAIC,EAAOmF,EAAW7L,KAAKoE,UAAUsC,KAAM1G,KAAKoE,UAAUsC,MACjF,IAAK,IAAIlE,EAAI,EAAGA,EAAIuJ,EAAWvJ,IAAK,CAClC,MAAM8I,EAAOnG,EAAKoG,QAAQ/I,GAC1B,IAAK8I,EACH,MAEEA,EAAKK,aAEP1B,GAA6B,IAAnBqB,EAAKU,UAAkB,EAAIV,EAAKG,WAAWpJ,OAEzD,CACF,CACA,OAAO4H,CACT,yFC/aF,MAAAvK,EAAAI,EAAA,KACAmM,EAAAnM,EAAA,KAwBA,MAAAoM,UAAqCxM,EAAAiE,WAanC,WAAA5D,CAA6BqE,GAC3BC,QAD2BrE,KAAAoE,UAAAA,EANrBpE,KAAAmM,mBAAqBnM,KAAK6D,UAAU,IAAInE,EAAA0M,mBACxCpM,KAAAqM,uBAAyBrM,KAAK6D,UAAU,IAAInE,EAAA0M,mBAG5CpM,KAAAsM,qBAAuB,EAI7BtM,KAAK6D,WAAU,EAAAnE,EAAAC,cAAa,IAAMK,KAAKuM,sBACzC,CAKO,cAAAlE,GACArI,KAAKwM,cACRxM,KAAKwM,YAAc,IAAItK,MAAMlC,KAAKoE,UAAU8B,OAAOC,OAAO9D,QAC1DrC,KAAKqM,uBAAuBpI,OAAQ,EAAAvE,EAAA+M,oBAClCzM,KAAKoE,UAAUsI,WAAW,IAAM1M,KAAKuM,sBACrCvM,KAAKoE,UAAUuI,aAAa,IAAM3M,KAAKuM,sBACvCvM,KAAKoE,UAAUwI,SAAS,IAAM5M,KAAKuM,wBAIvCvM,KAAKsM,qBAAuBO,KAAKC,MAC5B9M,KAAKmM,mBAAmBlI,OAC3BjE,KAAK+M,2BAA0B,KAEnC,CAEQ,kBAAAR,GACNvM,KAAKwM,iBAAc5L,EACnBZ,KAAKsM,qBAAuB,EAC5BtM,KAAKqM,uBAAuB3I,QAC5B1D,KAAKmM,mBAAmBzI,OAC1B,CAEQ,0BAAAqJ,CAA2BC,GACjChN,KAAKmM,mBAAmBlI,OAAQ,EAAAgI,EAAAgB,mBAAkB,KAChD,IAAKjN,KAAKwM,YACR,OAEF,MACMU,EADML,KAAKC,MACK9M,KAAKsM,qBACvBY,GAAO,KACTlN,KAAKuM,qBAGPvM,KAAK+M,2BAA2B,KAAqCG,IACpEF,EACL,CAEO,gBAAApD,CAAiBtD,GACtB,OAAOtG,KAAKwM,cAAclG,EAC5B,CAEO,cAAAwD,CAAexD,EAAa5E,GAC7B1B,KAAKwM,cACPxM,KAAKwM,YAAYlG,GAAO5E,EAE5B,CAUO,mCAAAmI,CAAoCsD,EAAmBC,GAC5D,MAAMC,EAAU,GACVzB,EAAc,CAAC,GAIf0B,EAAetN,KAAKoE,UAAU8B,OAAOC,OAAO9D,OAClD,IAAI8C,EAAOnF,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQ0D,GAChD,KAAOhI,GAAM,CACX,MAAMoI,EAAWJ,EAAY,EAAIG,EAAetN,KAAKoE,UAAU8B,OAAOC,OAAOsD,QAAQ0D,EAAY,QAAKvM,EAChG4M,IAAkBD,GAAWA,EAAS7D,UAC5C,IAAI+D,EAAStI,EAAKuI,mBAAmBF,GAAmBJ,GACxD,GAAII,GAAmBD,EAAU,CAC/B,MAAMI,EAAWxI,EAAKoG,QAAQpG,EAAK9C,OAAS,GACrBsL,GAAmC,IAAvBA,EAAS3B,WAA2C,IAAxB2B,EAAShC,YAEd,IAApC4B,EAAShC,QAAQ,IAAII,aACzC8B,EAASA,EAAO7L,MAAM,GAAI,GAE9B,CAEA,GADAyL,EAAQxL,KAAK4L,IACTD,EAGF,MAFA5B,EAAY/J,KAAK+J,EAAYA,EAAYvJ,OAAS,GAAKoL,EAAOpL,QAIhE8K,IACAhI,EAAOoI,CACT,CACA,MAAO,CAACF,EAAQO,KAAK,IAAKhC,EAC5B,gHCnIF,MAAAiC,EAAA/N,EAAA,KACAJ,EAAAI,EAAA,KAcA,MAAAgO,UAAyCpO,EAAAiE,WAAzC,WAAA5D,uBACUC,KAAA+N,eAAkC,GAGzB/N,KAAAgO,oBAAsBhO,KAAK6D,UAAU,IAAIgK,EAAAI,QA4F5D,CA3FE,sBAAWC,GAAyD,OAAOlO,KAAKgO,oBAAoB3M,KAAO,CAK3G,iBAAW8M,GACT,OAAOnO,KAAK+N,cACd,CAKA,sBAAWK,GACT,OAAOpO,KAAKqO,mBACd,CAKA,sBAAWD,CAAmBrJ,GAC5B/E,KAAKqO,oBAAsBtJ,CAC7B,CAOO,aAAAuJ,CAAc5J,EAA0B6J,GAC7CvO,KAAK+N,eAAiBrJ,EAAQ9C,MAAM,EAAG2M,EACzC,CAKO,YAAAC,GACLxO,KAAK+N,eAAiB,EACxB,CAKO,uBAAAU,GACDzO,KAAKqO,sBACPrO,KAAKqO,oBAAoB5O,UACzBO,KAAKqO,yBAAsBzN,EAE/B,CAOO,eAAA8N,CAAgB5M,GACrB,IAAK,IAAIU,EAAI,EAAGA,EAAIxC,KAAK+N,eAAe1L,OAAQG,IAAK,CACnD,MAAMoC,EAAQ5E,KAAK+N,eAAevL,GAClC,GAAIoC,EAAM0B,MAAQxE,EAAOwE,KAAO1B,EAAMkB,MAAQhE,EAAOgE,KAAOlB,EAAMoB,OAASlE,EAAOkE,KAChF,OAAOxD,CAEX,CACA,OAAQ,CACV,CAMO,kBAAAmM,CAAmBC,GACxB,IAAKA,EACH,OAGF,IAAIpE,GAAe,EACfxK,KAAKqO,sBACP7D,EAAcxK,KAAK0O,gBAAgB1O,KAAKqO,oBAAoBzJ,QAG9D5E,KAAKgO,oBAAoB5L,KAAK,CAC5BoI,cACAqE,YAAa7O,KAAK+N,eAAe1L,QAErC,CAKO,KAAAyM,GACL9O,KAAKyO,0BACLzO,KAAKwO,cACP,wHC1GF,MAOE,oBAAW5F,GACT,OAAO5I,KAAK+O,iBACd,CAKA,oBAAWnG,CAAiBZ,GAC1BhI,KAAK+O,kBAAoB/G,CAC3B,CAKA,qBAAWgH,GACT,OAAOhP,KAAKiP,kBACd,CAKA,qBAAWD,CAAkBrK,GAC3B3E,KAAKiP,mBAAqBtK,CAC5B,CAOO,iBAAAuK,CAAkBlH,GACvB,SAAUA,GAAQA,EAAK3F,OAAS,EAClC,CAOO,gBAAA8M,CAAiBC,GACtB,OAAKpP,KAAKiP,sBAGLG,IAGDpP,KAAKiP,mBAAmB3E,gBAAkB8E,EAAW9E,eAGrDtK,KAAKiP,mBAAmB5E,QAAU+E,EAAW/E,OAG7CrK,KAAKiP,mBAAmBzF,YAAc4F,EAAW5F,UAIvD,CAQO,wBAAA6F,CAAyBrH,EAAcrD,GAC5C,QAAKA,GAASE,mBAGoBjE,IAA3BZ,KAAK+O,mBACL/G,IAAShI,KAAK+O,mBACd/O,KAAKmP,iBAAiBxK,GAC/B,CAKO,eAAA2K,GACLtP,KAAK+O,uBAAoBnO,CAC3B,CAKO,KAAAkO,GACL9O,KAAK+O,uBAAoBnO,EACzBZ,KAAKiP,wBAAqBrO,CAC5B,KCvGF2O,EAAA,GAGA,SAAAzP,EAAA0P,GAEA,IAAAC,EAAAF,EAAAC,GACA,QAAA5O,IAAA6O,EACA,OAAAA,EAAA9Q,QAGA,IAAAC,EAAA2Q,EAAAC,GAAA,CAGA7Q,QAAA,IAOA,OAHA+Q,EAAAF,GAAA5Q,EAAAA,EAAAD,QAAAmB,GAGAlB,EAAAD,OACA,oGCfA,MAAAkP,EAAA/N,EAAA,KACAJ,EAAAI,EAAA,KACAmM,EAAAnM,EAAA,KACA6P,EAAA7P,EAAA,KACA8P,EAAA9P,EAAA,KACA+P,EAAA/P,EAAA,KACAgQ,EAAAhQ,EAAA,KACAiQ,EAAAjQ,EAAA,KAkBA,MAAAkQ,UAAiCtQ,EAAAiE,WAiB/B,sBAAWuK,GACT,OAAOlO,KAAKiQ,eAAe/B,kBAC7B,CAEA,WAAAnO,CAAY4E,GACVN,QAnBMrE,KAAAkQ,kBAAoBlQ,KAAK6D,UAAU,IAAInE,EAAA0M,mBACvCpM,KAAA8H,WAAa9H,KAAK6D,UAAU,IAAInE,EAAA0M,mBAGhCpM,KAAAmQ,OAAS,IAAIP,EAAAQ,YAGbpQ,KAAAiQ,eAAiBjQ,KAAK6D,UAAU,IAAIkM,EAAAjC,qBAE3B9N,KAAAqQ,eAAiBrQ,KAAK6D,UAAU,IAAIgK,EAAAI,SACrCjO,KAAAsQ,cAAgBtQ,KAAKqQ,eAAehP,MACnCrB,KAAAuQ,gBAAkBvQ,KAAK6D,UAAU,IAAIgK,EAAAI,SACtCjO,KAAAwQ,eAAiBxQ,KAAKuQ,gBAAgBlP,MASpDrB,KAAKyQ,gBAAkB9L,GAAS+L,gBAAc,GAChD,CAEO,QAAAC,CAASC,GACd5Q,KAAKoE,UAAYwM,EACjB5Q,KAAK8H,WAAW7D,MAAQ,IAAI0L,EAAAzD,gBAAgB0E,GAC5C5Q,KAAK6Q,QAAU,IAAIhB,EAAAiB,aAAaF,EAAU5Q,KAAK8H,WAAW7D,OAC1DjE,KAAK+Q,mBAAqB,IAAIjB,EAAA3L,kBAAkByM,GAChD5Q,KAAK6D,UAAU7D,KAAKoE,UAAU4M,cAAc,IAAMhR,KAAKiR,mBACvDjR,KAAK6D,UAAU7D,KAAKoE,UAAUwI,SAAS,IAAM5M,KAAKiR,mBAClDjR,KAAK6D,WAAU,EAAAnE,EAAAC,cAAa,IAAMK,KAAKkR,oBACzC,CAEQ,cAAAD,GACNjR,KAAKkQ,kBAAkBxM,QACnB1D,KAAKmQ,OAAOvH,kBAAoB5I,KAAKmQ,OAAOnB,mBAAmBnK,cACjE7E,KAAKkQ,kBAAkBjM,OAAQ,EAAAgI,EAAAgB,mBAAkB,KAC/C,MAAMjF,EAAOhI,KAAKmQ,OAAOvH,iBACzB5I,KAAKmQ,OAAOb,kBACZtP,KAAKmR,aAAanJ,EAAO,IAAKhI,KAAKmQ,OAAOnB,kBAAmBoC,aAAa,GAAQ,CAAEC,UAAU,KAC7F,KAEP,CAEO,gBAAAH,CAAiBI,GACtBtR,KAAKiQ,eAAexB,0BACpBzO,KAAK+Q,oBAAoBvM,4BACzBxE,KAAKiQ,eAAezB,eACf8C,GACHtR,KAAKmQ,OAAOb,iBAEhB,CAEO,qBAAAiC,GACLvR,KAAKiQ,eAAexB,yBACtB,CASO,QAAA+C,CAASxJ,EAAcG,EAAgCsJ,GAC5D,IAAKzR,KAAKoE,YAAcpE,KAAK6Q,QAC3B,MAAM,IAAIvQ,MAAM,6CAGlBN,KAAKuQ,gBAAgBnO,OAErBpC,KAAKmQ,OAAOnB,kBAAoB7G,EAE5BnI,KAAKmQ,OAAOd,yBAAyBrH,EAAMG,IAC7CnI,KAAK0R,qBAAqB1J,EAAMG,GAGlC,MAAMwJ,EAAQ3R,KAAK4R,mBAAmB5J,EAAMG,EAAesJ,GAM3D,OALAzR,KAAK6R,aAAa1J,GAClBnI,KAAKmQ,OAAOvH,iBAAmBZ,EAE/BhI,KAAKqQ,eAAejO,OAEbuP,CACT,CAEQ,oBAAAD,CAAqB1J,EAAcG,GACzC,IAAKnI,KAAKoE,YAAcpE,KAAK6Q,UAAY7Q,KAAK+Q,mBAC5C,MAAM,IAAIzQ,MAAM,6CAElB,IAAKN,KAAKmQ,OAAOjB,kBAAkBlH,GAEjC,YADAhI,KAAKkR,mBAKPlR,KAAKkR,kBAAiB,GAEtB,MAAMxM,EAA2B,GACjC,IAAIoN,EACAhQ,EAAS9B,KAAK6Q,QAAQ9I,KAAKC,EAAM,EAAG,EAAGG,GAE3C,KAAOrG,IAAWgQ,GAAYxL,MAAQxE,EAAOwE,KAAOwL,GAAYhM,MAAQhE,EAAOgE,QACzEpB,EAAQrC,QAAUrC,KAAKyQ,kBADwD,CAInFqB,EAAahQ,EACb4C,EAAQ7C,KAAKiQ,GACb,MAAMpL,EAAO1G,KAAKoE,UAAUsC,KAC5B,IAAIqL,EAAUD,EAAWhM,IAAMgM,EAAW9L,KACtCgM,EAAUF,EAAWxL,IACrByL,GAAWrL,IACbsL,GAAWxL,KAAKsF,MAAMiG,EAAUrL,GAChCqL,GAAoBrL,GAEtB5E,EAAS9B,KAAK6Q,QAAQ9I,KAAKC,EAAMgK,EAASD,EAAS5J,EACrD,CAEAnI,KAAKiQ,eAAe3B,cAAc5J,EAAS1E,KAAKyQ,iBAC5CtI,EAActD,aAChB7E,KAAK+Q,mBAAmBtM,2BAA2BC,EAASyD,EAActD,YAE9E,CAEQ,kBAAA+M,CAAmB5J,EAAcG,EAAgCsJ,GACvE,IAAKzR,KAAKoE,YAAcpE,KAAK6Q,QAC3B,OAAO,EAET,IAAK7Q,KAAKmQ,OAAOjB,kBAAkBlH,GAGjC,OAFAhI,KAAKoE,UAAUgE,iBACfpI,KAAKkR,oBACE,EAGT,MAAMpP,EAAS9B,KAAK6Q,QAAQlI,sBAAsBX,EAAMG,EAAenI,KAAKmQ,OAAOvH,kBACnF,OAAO5I,KAAKiS,cAAcnQ,EAAQqG,GAAetD,YAAa4M,GAAuBJ,SACvF,CASO,YAAAF,CAAanJ,EAAcG,EAAgCsJ,GAChE,IAAKzR,KAAKoE,YAAcpE,KAAK6Q,QAC3B,MAAM,IAAIvQ,MAAM,6CAGlBN,KAAKuQ,gBAAgBnO,OAErBpC,KAAKmQ,OAAOnB,kBAAoB7G,EAE5BnI,KAAKmQ,OAAOd,yBAAyBrH,EAAMG,IAC7CnI,KAAK0R,qBAAqB1J,EAAMG,GAGlC,MAAMwJ,EAAQ3R,KAAKkS,uBAAuBlK,EAAMG,EAAesJ,GAM/D,OALAzR,KAAK6R,aAAa1J,GAClBnI,KAAKmQ,OAAOvH,iBAAmBZ,EAE/BhI,KAAKqQ,eAAejO,OAEbuP,CACT,CAEQ,YAAAE,CAAa1J,GACnBnI,KAAKiQ,eAAetB,qBAAqBxG,GAAetD,YAC1D,CAEQ,sBAAAqN,CAAuBlK,EAAcG,EAAgCsJ,GAC3E,IAAKzR,KAAKoE,YAAcpE,KAAK6Q,QAC3B,OAAO,EAET,IAAK7Q,KAAKmQ,OAAOjB,kBAAkBlH,GAGjC,OAFAhI,KAAKoE,UAAUgE,iBACfpI,KAAKkR,oBACE,EAGT,MAAMpP,EAAS9B,KAAK6Q,QAAQ5H,0BAA0BjB,EAAMG,EAAenI,KAAKmQ,OAAOvH,kBACvF,OAAO5I,KAAKiS,cAAcnQ,EAAQqG,GAAetD,YAAa4M,GAAuBJ,SACvF,CAOQ,aAAAY,CAAcnQ,EAAmC6C,EAAoC0M,GAC3F,IAAKrR,KAAKoE,YAAcpE,KAAK+Q,mBAC3B,OAAO,EAIT,GADA/Q,KAAKiQ,eAAexB,2BACf3M,EAEH,OADA9B,KAAKoE,UAAUgE,kBACR,EAIT,GADApI,KAAKoE,UAAU+N,OAAOrQ,EAAOgE,IAAKhE,EAAOwE,IAAKxE,EAAOkE,MACjDrB,EAAS,CACX,MAAMyN,EAAmBpS,KAAK+Q,mBAAmB9L,uBAAuBnD,EAAQ6C,GAC5EyN,IACFpS,KAAKiQ,eAAe7B,mBAAqBgE,EAE7C,CAEA,IAAKf,IAECvP,EAAOwE,KAAQtG,KAAKoE,UAAU8B,OAAOC,OAAOkM,UAAYrS,KAAKoE,UAAUqE,MAAS3G,EAAOwE,IAAMtG,KAAKoE,UAAU8B,OAAOC,OAAOkM,WAAW,CACvI,IAAIC,EAASxQ,EAAOwE,IAAMtG,KAAKoE,UAAU8B,OAAOC,OAAOkM,UACvDC,GAAU9L,KAAKsF,MAAM9L,KAAKoE,UAAUqE,KAAO,GAC3CzI,KAAKoE,UAAUmO,YAAYD,EAC7B,CAEF,OAAO,CACT","sources":["webpack://SearchAddon/webpack/universalModuleDefinition","webpack://SearchAddon/../src/common/Async.ts","webpack://SearchAddon/../src/common/Event.ts","webpack://SearchAddon/../src/common/Lifecycle.ts","webpack://SearchAddon/./src/DecorationManager.ts","webpack://SearchAddon/./src/SearchEngine.ts","webpack://SearchAddon/./src/SearchLineCache.ts","webpack://SearchAddon/./src/SearchResultTracker.ts","webpack://SearchAddon/./src/SearchState.ts","webpack://SearchAddon/webpack/bootstrap","webpack://SearchAddon/./src/SearchAddon.ts"],"sourcesContent":["(function webpackUniversalModuleDefinition(root, factory) {\n\tif(typeof exports === 'object' && typeof module === 'object')\n\t\tmodule.exports = factory();\n\telse if(typeof define === 'function' && define.amd)\n\t\tdefine([], factory);\n\telse if(typeof exports === 'object')\n\t\texports[\"SearchAddon\"] = factory();\n\telse\n\t\troot[\"SearchAddon\"] = factory();\n})(globalThis, () => {\nreturn ","/**\n * Copyright (c) 2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal async helpers for xterm.js core.\n */\n\nimport { DisposableStore, IDisposable, toDisposable } from './Lifecycle';\n\nexport function timeout(millis: number): Promise {\n return new Promise(resolve => setTimeout(resolve, millis));\n}\n\n/**\n * Creates a timeout that can be disposed using its returned value.\n * @param handler The timeout handler.\n * @param timeout An optional timeout in milliseconds.\n * @param store An optional {@link DisposableStore} that will have the timeout disposable managed\n * automatically.\n */\nexport function disposableTimeout(handler: () => void, timeout = 0, store?: DisposableStore): IDisposable {\n const timer = setTimeout(() => {\n handler();\n if (store) {\n disposable.dispose();\n }\n }, timeout);\n const disposable = toDisposable(() => {\n clearTimeout(timer);\n });\n store?.add(disposable);\n return disposable;\n}\n\nexport class TimeoutTimer implements IDisposable {\n private _token: any = -1;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n if (this._token !== -1) {\n clearTimeout(this._token);\n this._token = -1;\n }\n }\n\n public cancelAndSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed TimeoutTimer');\n }\n this.cancel();\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n\n public setIfNotSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling setIfNotSet on a disposed TimeoutTimer');\n }\n if (this._token !== -1) {\n return;\n }\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n}\n\n/**\n * Schedules a single runner on the microtask queue. Unlike {@link TimeoutTimer}, a scheduled\n * microtask cannot be unqueued; {@link cancel} prevents the runner from executing if it has not\n * run yet.\n */\nexport class MicrotaskTimer implements IDisposable {\n private _isScheduled = false;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n this._isScheduled = false;\n }\n\n public set(runner: () => void): void {\n if (this._isDisposed) {\n throw new Error('Calling set on a disposed MicrotaskTimer');\n }\n if (this._isScheduled) {\n return;\n }\n this._isScheduled = true;\n queueMicrotask(() => {\n if (!this._isScheduled) {\n return;\n }\n this._isScheduled = false;\n runner();\n });\n }\n}\n\nexport class IntervalTimer implements IDisposable {\n private _disposable: IDisposable | undefined;\n private _isDisposed = false;\n\n public cancel(): void {\n this._disposable?.dispose();\n this._disposable = undefined;\n }\n\n public cancelAndSet(runner: () => void, interval: number, context: Window | typeof globalThis = globalThis): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed IntervalTimer');\n }\n this.cancel();\n const handle = context.setInterval(() => {\n runner();\n }, interval);\n this._disposable = {\n dispose: () => {\n context.clearInterval(handle as any);\n this._disposable = undefined;\n }\n };\n }\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n}\n","/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal event utilities for xterm.js core.\n * Simplified from VS Code's event.ts - no leak detection/profiling.\n */\n\nimport { IDisposable, DisposableStore, toDisposable } from './Lifecycle';\n\nexport interface IEvent {\n (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore): IDisposable;\n}\n\nexport class Emitter {\n private _listeners: { fn: (e: T) => any, thisArgs: any }[] = [];\n private _disposed = false;\n private _event: IEvent | undefined;\n\n public get event(): IEvent {\n if (this._event) {\n return this._event;\n }\n this._event = (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n if (this._disposed) {\n return toDisposable(() => {});\n }\n\n const entry = { fn: listener, thisArgs };\n this._listeners = this._listeners.slice();\n this._listeners.push(entry);\n\n const result = toDisposable(() => {\n const idx = this._listeners.indexOf(entry);\n if (idx !== -1) {\n this._listeners = this._listeners.slice();\n this._listeners.splice(idx, 1);\n }\n });\n\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(result);\n } else {\n disposables.add(result);\n }\n }\n\n return result;\n };\n return this._event;\n }\n\n public fire(event: T): void {\n if (this._disposed || !this._listeners.length) {\n return;\n }\n if (this._listeners.length === 1) {\n this._listeners[0].fn.call(this._listeners[0].thisArgs, event);\n return;\n }\n const listeners = this._listeners;\n for (let i = 0, len = listeners.length; i < len; ++i) {\n listeners[i].fn.call(listeners[i].thisArgs, event);\n }\n }\n\n public dispose(): void {\n if (this._disposed) {\n return;\n }\n this._disposed = true;\n this._listeners.length = 0;\n }\n}\n\nexport namespace EventUtils {\n export function forward(from: IEvent, to: Emitter): IDisposable {\n return from(e => to.fire(e));\n }\n\n export function map(event: IEvent, map: (i: I) => O): IEvent {\n return (listener: (e: O) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n return event(i => listener.call(thisArgs, map(i)), undefined, disposables);\n };\n }\n\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent {\n return (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n const store = new DisposableStore();\n for (const event of events) {\n store.add(event(e => listener.call(thisArgs, e)));\n }\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(store);\n } else {\n disposables.add(store);\n }\n }\n return store;\n };\n }\n\n export function runAndSubscribe(event: IEvent, handler: (e: T) => void, initial: T): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void, initial?: T): IDisposable {\n handler(initial);\n return event(e => handler(e));\n }\n}\n","/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal lifecycle utilities for xterm.js core.\n * Simplified from VS Code's lifecycle.ts - no tracking/leak detection.\n */\n\nexport interface IDisposable {\n dispose(): void;\n}\n\nexport function toDisposable(fn: () => void): IDisposable {\n return { dispose: fn };\n}\n\nexport function dispose(disposable: T): T;\nexport function dispose(disposable: T | undefined): T | undefined;\nexport function dispose(disposables: T[]): T[];\nexport function dispose(arg: T | T[] | undefined): T | T[] | undefined {\n if (!arg) {\n return arg;\n }\n if (Array.isArray(arg)) {\n for (const d of arg) {\n d.dispose();\n }\n return [];\n }\n arg.dispose();\n return arg;\n}\n\nexport function combinedDisposable(...disposables: IDisposable[]): IDisposable {\n return toDisposable(() => dispose(disposables));\n}\n\nexport class DisposableStore implements IDisposable {\n private readonly _disposables = new Set();\n private _isDisposed = false;\n\n public get isDisposed(): boolean {\n return this._isDisposed;\n }\n\n public add(o: T): T {\n if (this._isDisposed) {\n o.dispose();\n } else {\n this._disposables.add(o);\n }\n return o;\n }\n\n public dispose(): void {\n if (this._isDisposed) {\n return;\n }\n this._isDisposed = true;\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n\n public clear(): void {\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n}\n\nexport abstract class Disposable implements IDisposable {\n public static readonly None: IDisposable = Object.freeze({ dispose() { } });\n\n protected readonly _store = new DisposableStore();\n\n public dispose(): void {\n this._store.dispose();\n }\n\n protected _register(o: T): T {\n return this._store.add(o);\n }\n}\n\nexport class MutableDisposable implements IDisposable {\n private _value: T | undefined;\n private _isDisposed = false;\n\n public get value(): T | undefined {\n return this._isDisposed ? undefined : this._value;\n }\n\n public set value(value: T | undefined) {\n if (this._isDisposed || value === this._value) {\n return;\n }\n this._value?.dispose();\n this._value = value;\n }\n\n public clear(): void {\n this.value = undefined;\n }\n\n public dispose(): void {\n this._isDisposed = true;\n this._value?.dispose();\n this._value = undefined;\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, IDecoration } from '@xterm/xterm';\nimport type { ISearchDecorationOptions } from '@xterm/addon-search';\nimport { dispose, Disposable, toDisposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a highlight decoration.\n */\ninterface IHighlight extends IDisposable {\n decoration: IDecoration;\n match: ISearchResult;\n}\n\n/**\n * Interface for managing multiple decorations for a single match.\n */\ninterface IMultiHighlight extends IDisposable {\n decorations: IDecoration[];\n match: ISearchResult;\n}\n\n/**\n * Manages visual decorations for search results including highlighting and active selection\n * indicators. This class handles the creation, styling, and disposal of search-related decorations.\n */\nexport class DecorationManager extends Disposable {\n private _highlightDecorations: IHighlight[] = [];\n private _highlightedLines: Set = new Set();\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this.clearHighlightDecorations()));\n }\n\n /**\n * Creates decorations for all provided search results.\n * @param results The search results to create decorations for.\n * @param options The decoration options.\n */\n public createHighlightDecorations(results: ISearchResult[], options: ISearchDecorationOptions): void {\n this.clearHighlightDecorations();\n\n for (const match of results) {\n const decorations = this._createResultDecorations(match, options, false);\n if (decorations) {\n for (const decoration of decorations) {\n this._storeDecoration(decoration, match);\n }\n }\n }\n }\n\n /**\n * Creates decorations for the currently active search result.\n * @param result The active search result.\n * @param options The decoration options.\n * @returns The multi-highlight decoration or undefined if creation failed.\n */\n public createActiveDecoration(result: ISearchResult, options: ISearchDecorationOptions): IMultiHighlight | undefined {\n const decorations = this._createResultDecorations(result, options, true);\n if (decorations) {\n return { decorations, match: result, dispose() { dispose(decorations); } };\n }\n return undefined;\n }\n\n /**\n * Clears all highlight decorations.\n */\n public clearHighlightDecorations(): void {\n dispose(this._highlightDecorations);\n this._highlightDecorations = [];\n this._highlightedLines.clear();\n }\n\n /**\n * Stores a decoration and tracks it for management.\n * @param decoration The decoration to store.\n * @param match The search result this decoration represents.\n */\n private _storeDecoration(decoration: IDecoration, match: ISearchResult): void {\n this._highlightedLines.add(decoration.marker.line);\n this._highlightDecorations.push({ decoration, match, dispose() { decoration.dispose(); } });\n }\n\n /**\n * Applies styles to the decoration when it is rendered.\n * @param element The decoration's element.\n * @param borderColor The border color to apply.\n * @param isActiveResult Whether the element is part of the active search result.\n */\n private _applyStyles(element: HTMLElement, borderColor: string | undefined, isActiveResult: boolean): void {\n if (!element.classList.contains('xterm-find-result-decoration')) {\n element.classList.add('xterm-find-result-decoration');\n if (borderColor) {\n element.style.outline = `1px solid ${borderColor}`;\n }\n }\n if (isActiveResult) {\n element.classList.add('xterm-find-active-result-decoration');\n }\n }\n\n /**\n * Creates a decoration for the result and applies styles\n * @param result the search result for which to create the decoration\n * @param options the options for the decoration\n * @param isActiveResult whether this is the currently active result\n * @returns the decorations or undefined if the marker has already been disposed of\n */\n private _createResultDecorations(result: ISearchResult, options: ISearchDecorationOptions, isActiveResult: boolean): IDecoration[] | undefined {\n // Gather decoration ranges for this match as it could wrap\n const decorationRanges: [number, number, number][] = [];\n let currentCol = result.col;\n let remainingSize = result.size;\n let markerOffset = -this._terminal.buffer.active.baseY - this._terminal.buffer.active.cursorY + result.row;\n while (remainingSize > 0) {\n const amountThisRow = Math.min(this._terminal.cols - currentCol, remainingSize);\n decorationRanges.push([markerOffset, currentCol, amountThisRow]);\n currentCol = 0;\n remainingSize -= amountThisRow;\n markerOffset++;\n }\n\n // Create the decorations\n const decorations: IDecoration[] = [];\n for (const range of decorationRanges) {\n const marker = this._terminal.registerMarker(range[0]);\n const decoration = this._terminal.registerDecoration({\n marker,\n x: range[1],\n width: range[2],\n layer: isActiveResult ? 'top' : 'bottom',\n backgroundColor: isActiveResult ? options.activeMatchBackground : options.matchBackground,\n overviewRulerOptions: this._highlightedLines.has(marker.line) ? undefined : {\n color: isActiveResult ? options.activeMatchColorOverviewRuler : options.matchOverviewRuler,\n position: 'center'\n }\n });\n if (decoration) {\n const disposables: IDisposable[] = [];\n disposables.push(marker);\n disposables.push(decoration.onRender((e) => this._applyStyles(e, isActiveResult ? options.activeMatchBorder : options.matchBorder, false)));\n disposables.push(decoration.onDispose(() => dispose(disposables)));\n decorations.push(decoration);\n }\n }\n\n return decorations.length === 0 ? undefined : decorations;\n }\n}\n\n\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport type { ISearchOptions } from '@xterm/addon-search';\nimport type { SearchLineCache } from './SearchLineCache';\n\n/**\n * Represents the position to start a search from.\n */\ninterface ISearchPosition {\n startCol: number;\n startRow: number;\n}\n\n/**\n * Represents a search result with its position and content.\n */\nexport interface ISearchResult {\n term: string;\n col: number;\n row: number;\n size: number;\n}\n\n/**\n * Configuration constants for the search engine functionality.\n */\nconst enum Constants {\n /**\n * Characters that are considered non-word characters for search boundary detection. These\n * characters are used to determine word boundaries when performing whole-word searches. Includes\n * common punctuation, symbols, and whitespace characters.\n */\n NON_WORD_CHARACTERS = ' ~!@#$%^&*()+`-=[]{}|\\\\;:\"\\',./<>?'\n}\n\n/**\n * Core search engine that handles finding text within terminal content.\n * This class is responsible for the actual search algorithms and position calculations.\n */\nexport class SearchEngine {\n constructor(\n private readonly _terminal: Terminal,\n private readonly _lineCache: SearchLineCache\n ) {}\n\n /**\n * Find the first occurrence of a term starting from a specific position.\n * @param term The search term.\n * @param startRow The row to start searching from.\n * @param startCol The column to start searching from.\n * @param searchOptions Search options.\n * @returns The search result if found, undefined otherwise.\n */\n public find(term: string, startRow: number, startCol: number, searchOptions?: ISearchOptions): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n if (startCol >= this._terminal.cols) {\n throw new Error(`Invalid col: ${startCol} to search in terminal of ${this._terminal.cols} cols`);\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n if (this._isRowCoveredByEarlierSearch(y)) {\n continue;\n }\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n return result;\n }\n\n /**\n * Find the next occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine incremental behavior.\n * @returns The search result if found, undefined otherwise.\n */\n public findNextWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startCol = 0;\n let startRow = 0;\n if (prevSelectedPos) {\n if (cachedSearchTerm === term) {\n startCol = prevSelectedPos.end.x;\n startRow = prevSelectedPos.end.y;\n } else {\n startCol = prevSelectedPos.start.x;\n startRow = prevSelectedPos.start.y;\n }\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n if (this._isRowCoveredByEarlierSearch(y)) {\n continue;\n }\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n // If we hit the bottom and didn't search from the very top wrap back up\n if (!result && startRow !== 0) {\n for (let y = 0; y < startRow; y++) {\n // Row 0 is never skipped: it can be a continuation whose line start was trimmed from the\n // scrollback, and nothing earlier in this loop has searched it.\n if (y > 0 && this._isRowCoveredByEarlierSearch(y)) {\n continue;\n }\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n\n // If there is only one result, wrap back and return selection if it exists.\n if (!result && prevSelectedPos) {\n searchPosition.startRow = prevSelectedPos.start.y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n }\n\n return result;\n }\n\n /**\n * Find the previous occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine if expansion should occur.\n * @returns The search result if found, undefined otherwise.\n */\n public findPreviousWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startRow = this._terminal.buffer.active.baseY + this._terminal.rows - 1;\n const startCol = this._terminal.cols;\n const isReverseSearch = true;\n\n this._lineCache.initLinesCache();\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n let result: ISearchResult | undefined;\n if (prevSelectedPos) {\n searchPosition.startRow = startRow = prevSelectedPos.start.y;\n searchPosition.startCol = prevSelectedPos.start.x;\n if (cachedSearchTerm !== term) {\n // Try to expand selection to right first.\n result = this._findInLine(term, searchPosition, searchOptions, false);\n if (!result) {\n // If selection was not able to be expanded to the right, then try reverse search\n searchPosition.startRow = startRow = prevSelectedPos.end.y;\n searchPosition.startCol = prevSelectedPos.end.x;\n }\n }\n }\n\n result ??= this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n\n // Search from startRow - 1 to top\n if (!result) {\n searchPosition.startCol = Math.max(searchPosition.startCol, this._terminal.cols);\n for (let y = startRow - 1; y >= 0; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n // If we hit the top and didn't search from the very bottom wrap back down\n if (!result && startRow !== (this._terminal.buffer.active.baseY + this._terminal.rows - 1)) {\n for (let y = (this._terminal.buffer.active.baseY + this._terminal.rows - 1); y >= startRow; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n\n return result;\n }\n\n /**\n * A found substring is a whole word if it doesn't have an alphanumeric character directly\n * adjacent to it.\n * @param searchIndex starting index of the potential whole word substring\n * @param line entire string in which the potential whole word was found\n * @param term the substring that starts at searchIndex\n */\n private _isWholeWord(searchIndex: number, line: string, term: string): boolean {\n return ((searchIndex === 0) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex - 1]))) &&\n (((searchIndex + term.length) === line.length) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex + term.length])));\n }\n\n /** `_isWholeWord` gated on the option, so a rejected hit can be stepped past instead of ending the scan. */\n private _satisfiesWholeWord(searchIndex: number, line: string, term: string, searchOptions: ISearchOptions): boolean {\n return !searchOptions.wholeWord || this._isWholeWord(searchIndex, line, term);\n }\n\n /**\n * Whether an earlier `_findInLine` in this same call already scanned this row's line from an\n * equal or lower offset, which makes rescanning it pure O(rows^2) work on one long line. Sound\n * for every option because `_findInLine` returns the first accepted match at or after its\n * offset, which is monotone in that offset. Only valid once such a search has happened — the\n * wrap-around loop starts at row 0, whose line start may have been trimmed from the scrollback.\n */\n private _isRowCoveredByEarlierSearch(row: number): boolean {\n return this._terminal.buffer.active.getLine(row)?.isWrapped === true;\n }\n\n /**\n * Searches a line for a search term. Takes the provided terminal line and searches the text line,\n * which may contain subsequent terminal lines if the text is wrapped. If the provided line number\n * is part of a wrapped text line that started on an earlier line then it is skipped since it will\n * be properly searched when the terminal line that the text starts on is searched.\n * @param term The search term.\n * @param searchPosition The position to start the search.\n * @param searchOptions Search options.\n * @param isReverseSearch Whether the search should start from the right side of the terminal and\n * search to the left.\n * @returns The search result if it was found.\n */\n private _findInLine(term: string, searchPosition: ISearchPosition, searchOptions: ISearchOptions = {}, isReverseSearch: boolean = false): ISearchResult | undefined {\n // Ignore wrapped lines, only consider on unwrapped line (first row of command string).\n if (isReverseSearch) {\n // Reverse search never rewinds: its caller carries startCol down the rows of the line. Row 0\n // is searched even when wrapped, since its line start may have been trimmed from the scrollback.\n if (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) {\n searchPosition.startCol += this._terminal.cols;\n return;\n }\n } else {\n // A loop rather than recursion: one frame per wrapped row overflows the stack on a line long\n // enough to fill the scrollback. Bounded at row 0 because after a reflow the buffer's ring\n // holds stale entries at negative indices, so `getLine(-1)` answers with a wrapped line.\n while (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) {\n searchPosition.startRow--;\n searchPosition.startCol += this._terminal.cols;\n }\n }\n const row = searchPosition.startRow;\n const col = searchPosition.startCol;\n\n let cache = this._lineCache.getLineFromCache(row);\n if (!cache) {\n cache = this._lineCache.translateBufferLineToStringWithWrap(row, true);\n this._lineCache.setLineInCache(row, cache);\n }\n const [stringLine, offsets] = cache;\n\n const offset = this._bufferColsToStringOffset(row, col, offsets);\n let searchTerm = term;\n let searchStringLine = stringLine;\n if (!searchOptions.regex) {\n searchTerm = searchOptions.caseSensitive ? term : term.toLowerCase();\n searchStringLine = searchOptions.caseSensitive ? stringLine : stringLine.toLowerCase();\n }\n\n let resultIndex = -1;\n if (searchOptions.regex) {\n const searchRegex = RegExp(searchTerm, searchOptions.caseSensitive ? 'g' : 'gi');\n let foundTerm: RegExpExecArray | null;\n if (isReverseSearch) {\n // This loop will get the resultIndex of the _last_ regex match in the range 0..offset\n while (foundTerm = searchRegex.exec(searchStringLine.slice(0, offset))) {\n const matchIndex = searchRegex.lastIndex - foundTerm[0].length;\n if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) {\n resultIndex = matchIndex;\n term = foundTerm[0];\n }\n searchRegex.lastIndex = matchIndex + 1;\n }\n } else {\n // Driven over the whole line from `offset` rather than over `slice(offset)`: a slice\n // re-anchors ^ and \\b at whatever column the row happened to wrap at, and only\n // first-accepted-match-at-or-after-offset is monotone in `offset`, which is what lets\n // `_isRowCoveredByEarlierSearch` skip a wrapped row an earlier scan already covered.\n searchRegex.lastIndex = offset;\n while (foundTerm = searchRegex.exec(searchStringLine)) {\n const matchIndex = searchRegex.lastIndex - foundTerm[0].length;\n if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) {\n resultIndex = matchIndex;\n term = foundTerm[0];\n break;\n }\n // A zero-length or rejected match would otherwise repeat forever.\n searchRegex.lastIndex = matchIndex + 1;\n }\n }\n } else if (isReverseSearch) {\n let matchIndex = offset - searchTerm.length >= 0 ? searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length) : -1;\n // `lastIndexOf` clamps a negative fromIndex to 0, so index 0 has to end the walk.\n while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) {\n matchIndex = matchIndex > 0 ? searchStringLine.lastIndexOf(searchTerm, matchIndex - 1) : -1;\n }\n resultIndex = matchIndex;\n } else {\n let matchIndex = searchStringLine.indexOf(searchTerm, offset);\n while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) {\n matchIndex = searchStringLine.indexOf(searchTerm, matchIndex + 1);\n }\n resultIndex = matchIndex;\n }\n\n if (resultIndex >= 0) {\n // Adjust the row number and search index if needed since a \"line\" of text can span multiple\n // rows\n let startRowOffset = 0;\n while (startRowOffset < offsets.length - 1 && resultIndex >= offsets[startRowOffset + 1]) {\n startRowOffset++;\n }\n let endRowOffset = startRowOffset;\n while (endRowOffset < offsets.length - 1 && resultIndex + term.length >= offsets[endRowOffset + 1]) {\n endRowOffset++;\n }\n const startColOffset = resultIndex - offsets[startRowOffset];\n const endColOffset = resultIndex + term.length - offsets[endRowOffset];\n const startColIndex = this._stringLengthToBufferSize(row + startRowOffset, startColOffset);\n const endColIndex = this._stringLengthToBufferSize(row + endRowOffset, endColOffset);\n const size = endColIndex - startColIndex + this._terminal.cols * (endRowOffset - startRowOffset);\n\n return {\n term,\n col: startColIndex,\n row: row + startRowOffset,\n size\n };\n }\n }\n\n private _stringLengthToBufferSize(row: number, offset: number): number {\n const line = this._terminal.buffer.active.getLine(row);\n if (!line) {\n return 0;\n }\n for (let i = 0; i < offset; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n // Adjust the searchIndex to normalize emoji into single chars\n const char = cell.getChars();\n if (char.length > 1) {\n offset -= char.length - 1;\n }\n // Adjust the searchIndex for empty characters following wide unicode\n // chars (eg. CJK)\n const nextCell = line.getCell(i + 1);\n if (nextCell && nextCell.getWidth() === 0) {\n offset++;\n }\n }\n return offset;\n }\n\n /**\n * `cols` counts from the start of the logical line, so summing the cells of every row before the\n * resume point costs O(line) per call and the highlight-all pass makes one call per match.\n * `lineOffsets` already holds the string offset each wrapped row starts at — the same map used\n * above to turn a match index back into a row — so only the last, partial row needs cells. It is\n * also the map the row a match lands on is read from, which the cell sum disagreed with by one\n * for a row whose trailing cell is the null placeholder of a wide character that wrapped.\n */\n private _bufferColsToStringOffset(startRow: number, cols: number, lineOffsets: number[]): number {\n const rowsBack = Math.min(Math.floor(cols / this._terminal.cols), lineOffsets.length - 1);\n let offset = lineOffsets[rowsBack];\n const line = this._terminal.buffer.active.getLine(startRow + rowsBack);\n if (line) {\n const colsInRow = Math.min(cols - rowsBack * this._terminal.cols, this._terminal.cols);\n for (let i = 0; i < colsInRow; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n if (cell.getWidth()) {\n // Treat null characters as whitespace to align with the translateToString API\n offset += cell.getCode() === 0 ? 1 : cell.getChars().length;\n }\n }\n }\n return offset;\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport { combinedDisposable, Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\n\nexport type LineCacheEntry = [\n /**\n * The string representation of a line (as opposed to the buffer cell representation).\n */\n lineAsString: string,\n /**\n * The offsets where each line starts when the entry describes a wrapped line.\n */\n lineOffsets: number[]\n];\n\n/**\n * Configuration constants for the search line cache functionality.\n */\nconst enum Constants {\n /**\n * Time-to-live for cached search results in milliseconds. After this duration, cached search\n * results will be invalidated to ensure they remain consistent with terminal content changes.\n */\n LINES_CACHE_TIME_TO_LIVE = 15000\n}\n\nexport class SearchLineCache extends Disposable {\n /**\n * translateBufferLineToStringWithWrap is a fairly expensive call.\n * We memoize the calls into an array that has a time based ttl.\n * _linesCache is also invalidated when the terminal cursor moves.\n */\n private _linesCache: LineCacheEntry[] | undefined;\n private _linesCacheTimeout = this._register(new MutableDisposable());\n private _linesCacheDisposables = this._register(new MutableDisposable());\n // Track access to avoid recreating a timeout on every init call which occurs once per search\n // result (findNext/findPrevious -> _highlightAllMatches -> find loop).\n private _lastAccessTimestamp = 0;\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this._destroyLinesCache()));\n }\n\n /**\n * Sets up a line cache with a ttl\n */\n public initLinesCache(): void {\n if (!this._linesCache) {\n this._linesCache = new Array(this._terminal.buffer.active.length);\n this._linesCacheDisposables.value = combinedDisposable(\n this._terminal.onLineFeed(() => this._destroyLinesCache()),\n this._terminal.onCursorMove(() => this._destroyLinesCache()),\n this._terminal.onResize(() => this._destroyLinesCache())\n );\n }\n\n this._lastAccessTimestamp = Date.now();\n if (!this._linesCacheTimeout.value) {\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE);\n }\n }\n\n private _destroyLinesCache(): void {\n this._linesCache = undefined;\n this._lastAccessTimestamp = 0;\n this._linesCacheDisposables.clear();\n this._linesCacheTimeout.clear();\n }\n\n private _scheduleLinesCacheTimeout(delay: number): void {\n this._linesCacheTimeout.value = disposableTimeout(() => {\n if (!this._linesCache) {\n return;\n }\n const now = Date.now();\n const elapsed = now - this._lastAccessTimestamp;\n if (elapsed >= Constants.LINES_CACHE_TIME_TO_LIVE) {\n this._destroyLinesCache();\n return;\n }\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE - elapsed);\n }, delay);\n }\n\n public getLineFromCache(row: number): LineCacheEntry | undefined {\n return this._linesCache?.[row];\n }\n\n public setLineInCache(row: number, entry: LineCacheEntry): void {\n if (this._linesCache) {\n this._linesCache[row] = entry;\n }\n }\n\n /**\n * Translates a buffer line to a string, including subsequent lines if they are wraps.\n * Wide characters will count as two columns in the resulting string. This\n * function is useful for getting the actual text underneath the raw selection\n * position.\n * @param lineIndex The index of the line being translated.\n * @param trimRight Whether to trim whitespace to the right.\n */\n public translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean): LineCacheEntry {\n const strings = [];\n const lineOffsets = [0];\n // A single line longer than the whole scrollback leaves every buffer row wrapped, and the\n // buffer's ring answers an out-of-range row by cycling back to the start, so an unbounded walk\n // never reaches an unwrapped line.\n const bufferLength = this._terminal.buffer.active.length;\n let line = this._terminal.buffer.active.getLine(lineIndex);\n while (line) {\n const nextLine = lineIndex + 1 < bufferLength ? this._terminal.buffer.active.getLine(lineIndex + 1) : undefined;\n const lineWrapsToNext = nextLine ? nextLine.isWrapped : false;\n let string = line.translateToString(!lineWrapsToNext && trimRight);\n if (lineWrapsToNext && nextLine) {\n const lastCell = line.getCell(line.length - 1);\n const lastCellIsNull = lastCell && lastCell.getCode() === 0 && lastCell.getWidth() === 1;\n // a wide character wrapped to the next line\n if (lastCellIsNull && nextLine.getCell(0)?.getWidth() === 2) {\n string = string.slice(0, -1);\n }\n }\n strings.push(string);\n if (lineWrapsToNext) {\n lineOffsets.push(lineOffsets[lineOffsets.length - 1] + string.length);\n } else {\n break;\n }\n lineIndex++;\n line = nextLine;\n }\n return [strings.join(''), lineOffsets];\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchResultChangeEvent } from '@xterm/addon-search';\nimport type { IDisposable } from '@xterm/xterm';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a currently selected decoration.\n */\ninterface ISelectedDecoration extends IDisposable {\n match: ISearchResult;\n}\n\n/**\n * Tracks search results, manages result indexing, and fires events when results change.\n * This class provides centralized management of search result state and notifications.\n */\nexport class SearchResultTracker extends Disposable {\n private _searchResults: ISearchResult[] = [];\n private _selectedDecoration: ISelectedDecoration | undefined;\n\n private readonly _onDidChangeResults = this._register(new Emitter());\n public get onDidChangeResults(): IEvent { return this._onDidChangeResults.event; }\n\n /**\n * Gets the current search results.\n */\n public get searchResults(): ReadonlyArray {\n return this._searchResults;\n }\n\n /**\n * Gets the currently selected decoration.\n */\n public get selectedDecoration(): ISelectedDecoration | undefined {\n return this._selectedDecoration;\n }\n\n /**\n * Sets the currently selected decoration.\n */\n public set selectedDecoration(decoration: ISelectedDecoration | undefined) {\n this._selectedDecoration = decoration;\n }\n\n /**\n * Updates the search results with a new set of results.\n * @param results The new search results.\n * @param maxResults The maximum number of results to track.\n */\n public updateResults(results: ISearchResult[], maxResults: number): void {\n this._searchResults = results.slice(0, maxResults);\n }\n\n /**\n * Clears all search results.\n */\n public clearResults(): void {\n this._searchResults = [];\n }\n\n /**\n * Clears the selected decoration.\n */\n public clearSelectedDecoration(): void {\n if (this._selectedDecoration) {\n this._selectedDecoration.dispose();\n this._selectedDecoration = undefined;\n }\n }\n\n /**\n * Finds the index of a result in the current results array.\n * @param result The result to find.\n * @returns The index of the result, or -1 if not found.\n */\n public findResultIndex(result: ISearchResult): number {\n for (let i = 0; i < this._searchResults.length; i++) {\n const match = this._searchResults[i];\n if (match.row === result.row && match.col === result.col && match.size === result.size) {\n return i;\n }\n }\n return -1;\n }\n\n /**\n * Fires a result change event with the current state.\n * @param hasDecorations Whether decorations are enabled.\n */\n public fireResultsChanged(hasDecorations: boolean): void {\n if (!hasDecorations) {\n return;\n }\n\n let resultIndex = -1;\n if (this._selectedDecoration) {\n resultIndex = this.findResultIndex(this._selectedDecoration.match);\n }\n\n this._onDidChangeResults.fire({\n resultIndex,\n resultCount: this._searchResults.length\n });\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this.clearSelectedDecoration();\n this.clearResults();\n }\n}\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchOptions } from '@xterm/addon-search';\n\n/**\n * Manages search state including cached search terms, options tracking, and validation.\n * This class provides a centralized way to handle search state consistency and option changes.\n */\nexport class SearchState {\n private _cachedSearchTerm: string | undefined;\n private _lastSearchOptions: ISearchOptions | undefined;\n\n /**\n * Gets the currently cached search term.\n */\n public get cachedSearchTerm(): string | undefined {\n return this._cachedSearchTerm;\n }\n\n /**\n * Sets the cached search term.\n */\n public set cachedSearchTerm(term: string | undefined) {\n this._cachedSearchTerm = term;\n }\n\n /**\n * Gets the last search options used.\n */\n public get lastSearchOptions(): ISearchOptions | undefined {\n return this._lastSearchOptions;\n }\n\n /**\n * Sets the last search options used.\n */\n public set lastSearchOptions(options: ISearchOptions | undefined) {\n this._lastSearchOptions = options;\n }\n\n /**\n * Validates a search term to ensure it's not empty or invalid.\n * @param term The search term to validate.\n * @returns true if the term is valid for searching.\n */\n public isValidSearchTerm(term: string): boolean {\n return !!(term && term.length > 0);\n }\n\n /**\n * Determines if search options have changed compared to the last search.\n * @param newOptions The new search options to compare.\n * @returns true if the options have changed.\n */\n public didOptionsChange(newOptions?: ISearchOptions): boolean {\n if (!this._lastSearchOptions) {\n return true;\n }\n if (!newOptions) {\n return false;\n }\n if (this._lastSearchOptions.caseSensitive !== newOptions.caseSensitive) {\n return true;\n }\n if (this._lastSearchOptions.regex !== newOptions.regex) {\n return true;\n }\n if (this._lastSearchOptions.wholeWord !== newOptions.wholeWord) {\n return true;\n }\n return false;\n }\n\n /**\n * Determines if a new search should trigger highlighting updates.\n * @param term The search term.\n * @param options The search options.\n * @returns true if highlighting should be updated.\n */\n public shouldUpdateHighlighting(term: string, options?: ISearchOptions): boolean {\n if (!options?.decorations) {\n return false;\n }\n return this._cachedSearchTerm === undefined ||\n term !== this._cachedSearchTerm ||\n this.didOptionsChange(options);\n }\n\n /**\n * Clears the cached search term.\n */\n public clearCachedTerm(): void {\n this._cachedSearchTerm = undefined;\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this._cachedSearchTerm = undefined;\n this._lastSearchOptions = undefined;\n }\n}\n","// The module cache\nvar __webpack_module_cache__ = {};\n\n// The require function\nfunction __webpack_require__(moduleId) {\n\t// Check if module is in cache\n\tvar cachedModule = __webpack_module_cache__[moduleId];\n\tif (cachedModule !== undefined) {\n\t\treturn cachedModule.exports;\n\t}\n\t// Create a new module (and put it into the cache)\n\tvar module = __webpack_module_cache__[moduleId] = {\n\t\t// no module.id needed\n\t\t// no module.loaded needed\n\t\texports: {}\n\t};\n\n\t// Execute the module function\n\t__webpack_modules__[moduleId](module, module.exports, __webpack_require__);\n\n\t// Return the exports of the module\n\treturn module.exports;\n}\n\n","/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, ITerminalAddon } from '@xterm/xterm';\nimport type { SearchAddon as ISearchApi, ISearchOptions, ISearchAddonOptions, ISearchResultChangeEvent, ISearchDecorationOptions } from '@xterm/addon-search';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\nimport { SearchLineCache } from './SearchLineCache';\nimport { SearchState } from './SearchState';\nimport { SearchEngine, type ISearchResult } from './SearchEngine';\nimport { DecorationManager } from './DecorationManager';\nimport { SearchResultTracker } from './SearchResultTracker';\n\ninterface IInternalSearchOptions {\n noScroll: boolean;\n}\n\n/**\n * Configuration constants for the search addon functionality.\n */\nconst enum Constants {\n /**\n * Default maximum number of search results to highlight simultaneously. This limit prevents\n * performance degradation when searching for very common terms that would result in excessive\n * highlighting decorations.\n */\n DEFAULT_HIGHLIGHT_LIMIT = 1000\n}\n\nexport class SearchAddon extends Disposable implements ITerminalAddon, ISearchApi {\n private _terminal: Terminal | undefined;\n private _highlightLimit: number;\n private _highlightTimeout = this._register(new MutableDisposable());\n private _lineCache = this._register(new MutableDisposable());\n\n // Component instances\n private _state = new SearchState();\n private _engine: SearchEngine | undefined;\n private _decorationManager: DecorationManager | undefined;\n private _resultTracker = this._register(new SearchResultTracker());\n\n private readonly _onAfterSearch = this._register(new Emitter());\n public readonly onAfterSearch = this._onAfterSearch.event;\n private readonly _onBeforeSearch = this._register(new Emitter());\n public readonly onBeforeSearch = this._onBeforeSearch.event;\n\n public get onDidChangeResults(): IEvent {\n return this._resultTracker.onDidChangeResults;\n }\n\n constructor(options?: Partial) {\n super();\n\n this._highlightLimit = options?.highlightLimit ?? Constants.DEFAULT_HIGHLIGHT_LIMIT;\n }\n\n public activate(terminal: Terminal): void {\n this._terminal = terminal;\n this._lineCache.value = new SearchLineCache(terminal);\n this._engine = new SearchEngine(terminal, this._lineCache.value);\n this._decorationManager = new DecorationManager(terminal);\n this._register(this._terminal.onWriteParsed(() => this._updateMatches()));\n this._register(this._terminal.onResize(() => this._updateMatches()));\n this._register(toDisposable(() => this.clearDecorations()));\n }\n\n private _updateMatches(): void {\n this._highlightTimeout.clear();\n if (this._state.cachedSearchTerm && this._state.lastSearchOptions?.decorations) {\n this._highlightTimeout.value = disposableTimeout(() => {\n const term = this._state.cachedSearchTerm;\n this._state.clearCachedTerm();\n this.findPrevious(term!, { ...this._state.lastSearchOptions, incremental: true }, { noScroll: true });\n }, 200);\n }\n }\n\n public clearDecorations(retainCachedSearchTerm?: boolean): void {\n this._resultTracker.clearSelectedDecoration();\n this._decorationManager?.clearHighlightDecorations();\n this._resultTracker.clearResults();\n if (!retainCachedSearchTerm) {\n this._state.clearCachedTerm();\n }\n }\n\n public clearActiveDecoration(): void {\n this._resultTracker.clearSelectedDecoration();\n }\n\n /**\n * Find the next instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findNext(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findNextAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _highlightAllMatches(term: string, searchOptions: ISearchOptions): void {\n if (!this._terminal || !this._engine || !this._decorationManager) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n if (!this._state.isValidSearchTerm(term)) {\n this.clearDecorations();\n return;\n }\n\n // new search, clear out the old decorations\n this.clearDecorations(true);\n\n const results: ISearchResult[] = [];\n let prevResult: ISearchResult | undefined = undefined;\n let result = this._engine.find(term, 0, 0, searchOptions);\n\n while (result && (prevResult?.row !== result.row || prevResult?.col !== result.col)) {\n if (results.length >= this._highlightLimit) {\n break;\n }\n prevResult = result;\n results.push(prevResult);\n const cols = this._terminal.cols;\n let nextCol = prevResult.col + prevResult.size;\n let nextRow = prevResult.row;\n if (nextCol >= cols) {\n nextRow += Math.floor(nextCol / cols);\n nextCol = nextCol % cols;\n }\n result = this._engine.find(term, nextRow, nextCol, searchOptions);\n }\n\n this._resultTracker.updateResults(results, this._highlightLimit);\n if (searchOptions.decorations) {\n this._decorationManager.createHighlightDecorations(results, searchOptions.decorations);\n }\n }\n\n private _findNextAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findNextWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Find the previous instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findPrevious(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findPreviousAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _fireResults(searchOptions?: ISearchOptions): void {\n this._resultTracker.fireResultsChanged(!!searchOptions?.decorations);\n }\n\n private _findPreviousAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findPreviousWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Selects and scrolls to a result.\n * @param result The result to select.\n * @returns Whether a result was selected.\n */\n private _selectResult(result: ISearchResult | undefined, options?: ISearchDecorationOptions, noScroll?: boolean): boolean {\n if (!this._terminal || !this._decorationManager) {\n return false;\n }\n\n this._resultTracker.clearSelectedDecoration();\n if (!result) {\n this._terminal.clearSelection();\n return false;\n }\n\n this._terminal.select(result.col, result.row, result.size);\n if (options) {\n const activeDecoration = this._decorationManager.createActiveDecoration(result, options);\n if (activeDecoration) {\n this._resultTracker.selectedDecoration = activeDecoration;\n }\n }\n\n if (!noScroll) {\n // If it is not in the viewport then we scroll else it just gets selected\n if (result.row >= (this._terminal.buffer.active.viewportY + this._terminal.rows) || result.row < this._terminal.buffer.active.viewportY) {\n let scroll = result.row - this._terminal.buffer.active.viewportY;\n scroll -= Math.floor(this._terminal.rows / 2);\n this._terminal.scrollLines(scroll);\n }\n }\n return true;\n }\n}\n"],"names":["root","factory","exports","module","define","amd","globalThis","millis","Promise","resolve","setTimeout","handler","timeout","store","timer","disposable","dispose","Lifecycle_1","toDisposable","clearTimeout","add","__webpack_require__","constructor","this","_token","_isDisposed","cancel","cancelAndSet","runner","Error","setIfNotSet","_isScheduled","set","queueMicrotask","_disposable","undefined","interval","context","handle","setInterval","clearInterval","EventUtils","_listeners","_disposed","event","_event","listener","thisArgs","disposables","entry","fn","slice","push","result","idx","indexOf","splice","Array","isArray","fire","length","call","listeners","i","len","forward","from","to","e","map","any","events","DisposableStore","runAndSubscribe","initial","arg","d","_disposables","Set","isDisposed","o","clear","Disposable","_store","_register","None","Object","freeze","value","_value","DecorationManager","_terminal","super","_highlightDecorations","_highlightedLines","clearHighlightDecorations","createHighlightDecorations","results","options","match","decorations","_createResultDecorations","decoration","_storeDecoration","createActiveDecoration","marker","line","_applyStyles","element","borderColor","isActiveResult","classList","contains","style","outline","decorationRanges","currentCol","col","remainingSize","size","markerOffset","buffer","active","baseY","cursorY","row","amountThisRow","Math","min","cols","range","registerMarker","registerDecoration","x","width","layer","backgroundColor","activeMatchBackground","matchBackground","overviewRulerOptions","has","color","activeMatchColorOverviewRuler","matchOverviewRuler","position","onRender","activeMatchBorder","matchBorder","onDispose","_lineCache","find","term","startRow","startCol","searchOptions","clearSelection","initLinesCache","searchPosition","_findInLine","y","rows","_isRowCoveredByEarlierSearch","findNextWithSelection","cachedSearchTerm","prevSelectedPos","getSelectionPosition","end","start","findPreviousWithSelection","isReverseSearch","max","_isWholeWord","searchIndex","includes","_satisfiesWholeWord","wholeWord","getLine","isWrapped","cache","getLineFromCache","translateBufferLineToStringWithWrap","setLineInCache","stringLine","offsets","offset","_bufferColsToStringOffset","searchTerm","searchStringLine","regex","caseSensitive","toLowerCase","resultIndex","searchRegex","RegExp","foundTerm","exec","matchIndex","lastIndex","lastIndexOf","startRowOffset","endRowOffset","startColOffset","endColOffset","startColIndex","_stringLengthToBufferSize","cell","getCell","char","getChars","nextCell","getWidth","lineOffsets","rowsBack","floor","colsInRow","getCode","Async_1","SearchLineCache","_linesCacheTimeout","MutableDisposable","_linesCacheDisposables","_lastAccessTimestamp","_destroyLinesCache","_linesCache","combinedDisposable","onLineFeed","onCursorMove","onResize","Date","now","_scheduleLinesCacheTimeout","delay","disposableTimeout","elapsed","lineIndex","trimRight","strings","bufferLength","nextLine","lineWrapsToNext","string","translateToString","lastCell","join","Event_1","SearchResultTracker","_searchResults","_onDidChangeResults","Emitter","onDidChangeResults","searchResults","selectedDecoration","_selectedDecoration","updateResults","maxResults","clearResults","clearSelectedDecoration","findResultIndex","fireResultsChanged","hasDecorations","resultCount","reset","_cachedSearchTerm","lastSearchOptions","_lastSearchOptions","isValidSearchTerm","didOptionsChange","newOptions","shouldUpdateHighlighting","clearCachedTerm","__webpack_module_cache__","moduleId","cachedModule","__webpack_modules__","SearchLineCache_1","SearchState_1","SearchEngine_1","DecorationManager_1","SearchResultTracker_1","SearchAddon","_resultTracker","_highlightTimeout","_state","SearchState","_onAfterSearch","onAfterSearch","_onBeforeSearch","onBeforeSearch","_highlightLimit","highlightLimit","activate","terminal","_engine","SearchEngine","_decorationManager","onWriteParsed","_updateMatches","clearDecorations","findPrevious","incremental","noScroll","retainCachedSearchTerm","clearActiveDecoration","findNext","internalSearchOptions","_highlightAllMatches","found","_findNextAndSelect","_fireResults","prevResult","nextCol","nextRow","_selectResult","_findPreviousAndSelect","select","activeDecoration","viewportY","scroll","scrollLines"],"sourceRoot":""} +\ No newline at end of file +diff --git a/lib/addon-search.mjs b/lib/addon-search.mjs +index 5cf231a96b56284705711faebe6b0c492f133546..f2b8b804ac733d737a9bff22b4e1d24778a807c8 100644 +--- a/lib/addon-search.mjs ++++ b/lib/addon-search.mjs +@@ -14,5 +14,5 @@ + * Copyright (c) Microsoft Corporation. All rights reserved. + * Licensed under the MIT License. See License.txt in the project root for license information. + *--------------------------------------------------------------------------------------------*/ +-function _(c){return{dispose:c}}function D(c){if(!c)return c;if(Array.isArray(c)){for(let i of c)i.dispose();return[]}return c.dispose(),c}function E(...c){return _(()=>D(c))}var S=class{constructor(){this._disposables=new Set;this._isDisposed=!1}get isDisposed(){return this._isDisposed}add(i){return this._isDisposed?i.dispose():this._disposables.add(i),i}dispose(){if(!this._isDisposed){this._isDisposed=!0;for(let i of this._disposables)i.dispose();this._disposables.clear()}}clear(){for(let i of this._disposables)i.dispose();this._disposables.clear()}},b=class{constructor(){this._store=new S}dispose(){this._store.dispose()}_register(i){return this._store.add(i)}};b.None=Object.freeze({dispose(){}});var v=class{constructor(){this._isDisposed=!1}get value(){return this._isDisposed?void 0:this._value}set value(i){this._isDisposed||i===this._value||(this._value?.dispose(),this._value=i)}clear(){this.value=void 0}dispose(){this._isDisposed=!0,this._value?.dispose(),this._value=void 0}};var I=class{constructor(){this._listeners=[];this._disposed=!1}get event(){return this._event?this._event:(this._event=(i,e,t)=>{if(this._disposed)return _(()=>{});let r={fn:i,thisArgs:e};this._listeners=this._listeners.slice(),this._listeners.push(r);let s=_(()=>{let n=this._listeners.indexOf(r);n!==-1&&(this._listeners=this._listeners.slice(),this._listeners.splice(n,1))});return t&&(Array.isArray(t)?t.push(s):t.add(s)),s},this._event)}fire(i){if(this._disposed||!this._listeners.length)return;if(this._listeners.length===1){this._listeners[0].fn.call(this._listeners[0].thisArgs,i);return}let e=this._listeners;for(let t=0,r=e.length;t{function c(s,n){return s(a=>n.fire(a))}r.forward=c;function i(s,n){return(a,o,l)=>s(h=>a.call(o,n(h)),void 0,l)}r.map=i;function e(...s){return(n,a,o)=>{let l=new S;for(let h of s)l.add(h(p=>n.call(a,p)));return o&&(Array.isArray(o)?o.push(l):o.add(l)),l}}r.any=e;function t(s,n,a){return n(a),s(o=>n(o))}r.runAndSubscribe=t})(W||={});function T(c,i=0,e){let t=setTimeout(()=>{c(),e&&r.dispose()},i),r=_(()=>{clearTimeout(t)});return e?.add(r),r}var C=class extends b{constructor(e){super();this._terminal=e;this._linesCacheTimeout=this._register(new v);this._linesCacheDisposables=this._register(new v);this._lastAccessTimestamp=0;this._register(_(()=>this._destroyLinesCache()))}initLinesCache(){this._linesCache||(this._linesCache=new Array(this._terminal.buffer.active.length),this._linesCacheDisposables.value=E(this._terminal.onLineFeed(()=>this._destroyLinesCache()),this._terminal.onCursorMove(()=>this._destroyLinesCache()),this._terminal.onResize(()=>this._destroyLinesCache()))),this._lastAccessTimestamp=Date.now(),this._linesCacheTimeout.value||this._scheduleLinesCacheTimeout(15e3)}_destroyLinesCache(){this._linesCache=void 0,this._lastAccessTimestamp=0,this._linesCacheDisposables.clear(),this._linesCacheTimeout.clear()}_scheduleLinesCacheTimeout(e){this._linesCacheTimeout.value=T(()=>{if(!this._linesCache)return;let r=Date.now()-this._lastAccessTimestamp;if(r>=15e3){this._destroyLinesCache();return}this._scheduleLinesCacheTimeout(15e3-r)},e)}getLineFromCache(e){return this._linesCache?.[e]}setLineInCache(e,t){this._linesCache&&(this._linesCache[e]=t)}translateBufferLineToStringWithWrap(e,t){let r=[],s=[0],n=this._terminal.buffer.active.getLine(e);for(;n;){let a=this._terminal.buffer.active.getLine(e+1),o=a?a.isWrapped:!1,l=n.translateToString(!o&&t);if(o&&a){let h=n.getCell(n.length-1);h&&h.getCode()===0&&h.getWidth()===1&&a.getCell(0)?.getWidth()===2&&(l=l.slice(0,-1))}if(r.push(l),o)s.push(s[s.length-1]+l.length);else break;e++,n=a}return[r.join(""),s]}};var x=class{get cachedSearchTerm(){return this._cachedSearchTerm}set cachedSearchTerm(i){this._cachedSearchTerm=i}get lastSearchOptions(){return this._lastSearchOptions}set lastSearchOptions(i){this._lastSearchOptions=i}isValidSearchTerm(i){return!!(i&&i.length>0)}didOptionsChange(i){return this._lastSearchOptions?i?this._lastSearchOptions.caseSensitive!==i.caseSensitive||this._lastSearchOptions.regex!==i.regex||this._lastSearchOptions.wholeWord!==i.wholeWord:!1:!0}shouldUpdateHighlighting(i,e){return e?.decorations?this._cachedSearchTerm===void 0||i!==this._cachedSearchTerm||this.didOptionsChange(e):!1}clearCachedTerm(){this._cachedSearchTerm=void 0}reset(){this._cachedSearchTerm=void 0,this._lastSearchOptions=void 0}};var R=class{constructor(i,e){this._terminal=i;this._lineCache=e}find(i,e,t,r){if(!i||i.length===0){this._terminal.clearSelection();return}if(t>=this._terminal.cols)throw new Error(`Invalid col: ${t} to search in terminal of ${this._terminal.cols} cols`);this._lineCache.initLinesCache();let s={startRow:e,startCol:t},n=this._findInLine(i,s,r);if(!n)for(let a=e+1;a=0&&(o.startRow=h,l=this._findInLine(i,o,e,a),!l);h--);}if(!l&&s!==this._terminal.buffer.active.baseY+this._terminal.rows-1)for(let h=this._terminal.buffer.active.baseY+this._terminal.rows-1;h>=s&&(o.startRow=h,l=this._findInLine(i,o,e,a),!l);h--);return l}_isWholeWord(i,e,t){return(i===0||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(e[i-1]))&&(i+t.length===e.length||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(e[i+t.length]))}_findInLine(i,e,t={},r=!1){let s=e.startRow,n=e.startCol;if(this._terminal.buffer.active.getLine(s)?.isWrapped){if(r){e.startCol+=this._terminal.cols;return}return e.startRow--,e.startCol+=this._terminal.cols,this._findInLine(i,e,t)}let o=this._lineCache.getLineFromCache(s);o||(o=this._lineCache.translateBufferLineToStringWithWrap(s,!0),this._lineCache.setLineInCache(s,o));let[l,h]=o,p=this._bufferColsToStringOffset(s,n),m=i,g=l;t.regex||(m=t.caseSensitive?i:i.toLowerCase(),g=t.caseSensitive?l:l.toLowerCase());let f=-1;if(t.regex){let u=RegExp(m,t.caseSensitive?"g":"gi"),d;if(r)for(;d=u.exec(g.slice(0,p));)f=u.lastIndex-d[0].length,i=d[0],u.lastIndex-=i.length-1;else d=u.exec(g.slice(p)),d&&d[0].length>0&&(f=p+(u.lastIndex-d[0].length),i=d[0])}else r?p-m.length>=0&&(f=g.lastIndexOf(m,p-m.length)):f=g.indexOf(m,p);if(f>=0){if(t.wholeWord&&!this._isWholeWord(f,g,i))return;let u=0;for(;u=h[u+1];)u++;let d=u;for(;d=h[d+1];)d++;let O=f-h[u],k=f+i.length-h[d],y=this._stringLengthToBufferSize(s+u,O),M=this._stringLengthToBufferSize(s+d,k)-y+this._terminal.cols*(d-u);return{term:i,col:y,row:s+u,size:M}}}_stringLengthToBufferSize(i,e){let t=this._terminal.buffer.active.getLine(i);if(!t)return 0;for(let r=0;r1&&(e-=n.length-1);let a=t.getCell(r+1);a&&a.getWidth()===0&&e++}return e}_bufferColsToStringOffset(i,e){let t=i,r=0,s=this._terminal.buffer.active.getLine(t);for(;e>0&&s;){for(let n=0;nthis.clearHighlightDecorations()))}createHighlightDecorations(e,t){this.clearHighlightDecorations();for(let r of e){let s=this._createResultDecorations(r,t,!1);if(s)for(let n of s)this._storeDecoration(n,r)}}createActiveDecoration(e,t){let r=this._createResultDecorations(e,t,!0);if(r)return{decorations:r,match:e,dispose(){D(r)}}}clearHighlightDecorations(){D(this._highlightDecorations),this._highlightDecorations=[],this._highlightedLines.clear()}_storeDecoration(e,t){this._highlightedLines.add(e.marker.line),this._highlightDecorations.push({decoration:e,match:t,dispose(){e.dispose()}})}_applyStyles(e,t,r){e.classList.contains("xterm-find-result-decoration")||(e.classList.add("xterm-find-result-decoration"),t&&(e.style.outline=`1px solid ${t}`)),r&&e.classList.add("xterm-find-active-result-decoration")}_createResultDecorations(e,t,r){let s=[],n=e.col,a=e.size,o=-this._terminal.buffer.active.baseY-this._terminal.buffer.active.cursorY+e.row;for(;a>0;){let h=Math.min(this._terminal.cols-n,a);s.push([o,n,h]),n=0,a-=h,o++}let l=[];for(let h of s){let p=this._terminal.registerMarker(h[0]),m=this._terminal.registerDecoration({marker:p,x:h[1],width:h[2],layer:r?"top":"bottom",backgroundColor:r?t.activeMatchBackground:t.matchBackground,overviewRulerOptions:this._highlightedLines.has(p.line)?void 0:{color:r?t.activeMatchColorOverviewRuler:t.matchOverviewRuler,position:"center"}});if(m){let g=[];g.push(p),g.push(m.onRender(f=>this._applyStyles(f,r?t.activeMatchBorder:t.matchBorder,!1))),g.push(m.onDispose(()=>D(g))),l.push(m)}}return l.length===0?void 0:l}};var L=class extends b{constructor(){super(...arguments);this._searchResults=[];this._onDidChangeResults=this._register(new I)}get onDidChangeResults(){return this._onDidChangeResults.event}get searchResults(){return this._searchResults}get selectedDecoration(){return this._selectedDecoration}set selectedDecoration(e){this._selectedDecoration=e}updateResults(e,t){this._searchResults=e.slice(0,t)}clearResults(){this._searchResults=[]}clearSelectedDecoration(){this._selectedDecoration&&(this._selectedDecoration.dispose(),this._selectedDecoration=void 0)}findResultIndex(e){for(let t=0;tthis._updateMatches())),this._register(this._terminal.onResize(()=>this._updateMatches())),this._register(_(()=>this.clearDecorations()))}_updateMatches(){this._highlightTimeout.clear(),this._state.cachedSearchTerm&&this._state.lastSearchOptions?.decorations&&(this._highlightTimeout.value=T(()=>{let e=this._state.cachedSearchTerm;this._state.clearCachedTerm(),this.findPrevious(e,{...this._state.lastSearchOptions,incremental:!0},{noScroll:!0})},200))}clearDecorations(e){this._resultTracker.clearSelectedDecoration(),this._decorationManager?.clearHighlightDecorations(),this._resultTracker.clearResults(),e||this._state.clearCachedTerm()}clearActiveDecoration(){this._resultTracker.clearSelectedDecoration()}findNext(e,t,r){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);let s=this._findNextAndSelect(e,t,r);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),s}_highlightAllMatches(e,t){if(!this._terminal||!this._engine||!this._decorationManager)throw new Error("Cannot use addon until it has been loaded");if(!this._state.isValidSearchTerm(e)){this.clearDecorations();return}this.clearDecorations(!0);let r=[],s,n=this._engine.find(e,0,0,t);for(;n&&(s?.row!==n.row||s?.col!==n.col)&&!(r.length>=this._highlightLimit);){s=n,r.push(s);let a=this._terminal.cols,o=s.col+s.size,l=s.row;o>=a&&(l+=Math.floor(o/a),o=o%a),n=this._engine.find(e,l,o,t)}this._resultTracker.updateResults(r,this._highlightLimit),t.decorations&&this._decorationManager.createHighlightDecorations(r,t.decorations)}_findNextAndSelect(e,t,r){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;let s=this._engine.findNextWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(s,t?.decorations,r?.noScroll)}findPrevious(e,t,r){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);let s=this._findPreviousAndSelect(e,t,r);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),s}_fireResults(e){this._resultTracker.fireResultsChanged(!!e?.decorations)}_findPreviousAndSelect(e,t,r){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;let s=this._engine.findPreviousWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(s,t?.decorations,r?.noScroll)}_selectResult(e,t,r){if(!this._terminal||!this._decorationManager)return!1;if(this._resultTracker.clearSelectedDecoration(),!e)return this._terminal.clearSelection(),!1;if(this._terminal.select(e.col,e.row,e.size),t){let s=this._decorationManager.createActiveDecoration(e,t);s&&(this._resultTracker.selectedDecoration=s)}if(!r&&(e.row>=this._terminal.buffer.active.viewportY+this._terminal.rows||e.rowD(d))}var S=class{constructor(){this._disposables=new Set;this._isDisposed=!1}get isDisposed(){return this._isDisposed}add(i){return this._isDisposed?i.dispose():this._disposables.add(i),i}dispose(){if(!this._isDisposed){this._isDisposed=!0;for(let i of this._disposables)i.dispose();this._disposables.clear()}}clear(){for(let i of this._disposables)i.dispose();this._disposables.clear()}},g=class{constructor(){this._store=new S}dispose(){this._store.dispose()}_register(i){return this._store.add(i)}};g.None=Object.freeze({dispose(){}});var v=class{constructor(){this._isDisposed=!1}get value(){return this._isDisposed?void 0:this._value}set value(i){this._isDisposed||i===this._value||(this._value?.dispose(),this._value=i)}clear(){this.value=void 0}dispose(){this._isDisposed=!0,this._value?.dispose(),this._value=void 0}};var I=class{constructor(){this._listeners=[];this._disposed=!1}get event(){return this._event?this._event:(this._event=(i,e,t)=>{if(this._disposed)return m(()=>{});let s={fn:i,thisArgs:e};this._listeners=this._listeners.slice(),this._listeners.push(s);let r=m(()=>{let l=this._listeners.indexOf(s);l!==-1&&(this._listeners=this._listeners.slice(),this._listeners.splice(l,1))});return t&&(Array.isArray(t)?t.push(r):t.add(r)),r},this._event)}fire(i){if(this._disposed||!this._listeners.length)return;if(this._listeners.length===1){this._listeners[0].fn.call(this._listeners[0].thisArgs,i);return}let e=this._listeners;for(let t=0,s=e.length;t{function d(r,l){return r(o=>l.fire(o))}s.forward=d;function i(r,l){return(o,n,a)=>r(c=>o.call(n,l(c)),void 0,a)}s.map=i;function e(...r){return(l,o,n)=>{let a=new S;for(let c of r)a.add(c(u=>l.call(o,u)));return n&&(Array.isArray(n)?n.push(a):n.add(a)),a}}s.any=e;function t(r,l,o){return l(o),r(n=>l(n))}s.runAndSubscribe=t})(k||={});function T(d,i=0,e){let t=setTimeout(()=>{d(),e&&s.dispose()},i),s=m(()=>{clearTimeout(t)});return e?.add(s),s}var C=class extends g{constructor(e){super();this._terminal=e;this._linesCacheTimeout=this._register(new v);this._linesCacheDisposables=this._register(new v);this._lastAccessTimestamp=0;this._register(m(()=>this._destroyLinesCache()))}initLinesCache(){this._linesCache||(this._linesCache=new Array(this._terminal.buffer.active.length),this._linesCacheDisposables.value=E(this._terminal.onLineFeed(()=>this._destroyLinesCache()),this._terminal.onCursorMove(()=>this._destroyLinesCache()),this._terminal.onResize(()=>this._destroyLinesCache()))),this._lastAccessTimestamp=Date.now(),this._linesCacheTimeout.value||this._scheduleLinesCacheTimeout(15e3)}_destroyLinesCache(){this._linesCache=void 0,this._lastAccessTimestamp=0,this._linesCacheDisposables.clear(),this._linesCacheTimeout.clear()}_scheduleLinesCacheTimeout(e){this._linesCacheTimeout.value=T(()=>{if(!this._linesCache)return;let s=Date.now()-this._lastAccessTimestamp;if(s>=15e3){this._destroyLinesCache();return}this._scheduleLinesCacheTimeout(15e3-s)},e)}getLineFromCache(e){return this._linesCache?.[e]}setLineInCache(e,t){this._linesCache&&(this._linesCache[e]=t)}translateBufferLineToStringWithWrap(e,t){let s=[],r=[0],l=this._terminal.buffer.active.length,o=this._terminal.buffer.active.getLine(e);for(;o;){let n=e+10)}didOptionsChange(i){return this._lastSearchOptions?i?this._lastSearchOptions.caseSensitive!==i.caseSensitive||this._lastSearchOptions.regex!==i.regex||this._lastSearchOptions.wholeWord!==i.wholeWord:!1:!0}shouldUpdateHighlighting(i,e){return e?.decorations?this._cachedSearchTerm===void 0||i!==this._cachedSearchTerm||this.didOptionsChange(e):!1}clearCachedTerm(){this._cachedSearchTerm=void 0}reset(){this._cachedSearchTerm=void 0,this._lastSearchOptions=void 0}};var w=class{constructor(i,e){this._terminal=i;this._lineCache=e}find(i,e,t,s){if(!i||i.length===0){this._terminal.clearSelection();return}if(t>=this._terminal.cols)throw new Error(`Invalid col: ${t} to search in terminal of ${this._terminal.cols} cols`);this._lineCache.initLinesCache();let r={startRow:e,startCol:t},l=this._findInLine(i,r,s);if(!l)for(let o=e+1;o0&&this._isRowCoveredByEarlierSearch(a))&&(o.startRow=a,o.startCol=0,n=this._findInLine(i,o,e),n));a++);return!n&&s&&(o.startRow=s.start.y,o.startCol=0,n=this._findInLine(i,o,e)),n}findPreviousWithSelection(i,e,t){if(!i||i.length===0){this._terminal.clearSelection();return}let s=this._terminal.getSelectionPosition();this._terminal.clearSelection();let r=this._terminal.buffer.active.baseY+this._terminal.rows-1,l=this._terminal.cols,o=!0;this._lineCache.initLinesCache();let n={startRow:r,startCol:l},a;if(s&&(n.startRow=r=s.start.y,n.startCol=s.start.x,t!==i&&(a=this._findInLine(i,n,e,!1),a||(n.startRow=r=s.end.y,n.startCol=s.end.x))),a??=this._findInLine(i,n,e,o),!a){n.startCol=Math.max(n.startCol,this._terminal.cols);for(let c=r-1;c>=0&&(n.startRow=c,a=this._findInLine(i,n,e,o),!a);c--);}if(!a&&r!==this._terminal.buffer.active.baseY+this._terminal.rows-1)for(let c=this._terminal.buffer.active.baseY+this._terminal.rows-1;c>=r&&(n.startRow=c,a=this._findInLine(i,n,e,o),!a);c--);return a}_isWholeWord(i,e,t){return(i===0||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(e[i-1]))&&(i+t.length===e.length||" ~!@#$%^&*()+`-=[]{}|\\;:\"',./<>?".includes(e[i+t.length]))}_satisfiesWholeWord(i,e,t,s){return!s.wholeWord||this._isWholeWord(i,e,t)}_isRowCoveredByEarlierSearch(i){return this._terminal.buffer.active.getLine(i)?.isWrapped===!0}_findInLine(i,e,t={},s=!1){if(s){if(e.startRow>0&&this._terminal.buffer.active.getLine(e.startRow)?.isWrapped){e.startCol+=this._terminal.cols;return}}else for(;e.startRow>0&&this._terminal.buffer.active.getLine(e.startRow)?.isWrapped;)e.startRow--,e.startCol+=this._terminal.cols;let r=e.startRow,l=e.startCol,o=this._lineCache.getLineFromCache(r);o||(o=this._lineCache.translateBufferLineToStringWithWrap(r,!0),this._lineCache.setLineInCache(r,o));let[n,a]=o,c=this._bufferColsToStringOffset(r,l,a),u=i,f=n;t.regex||(u=t.caseSensitive?i:i.toLowerCase(),f=t.caseSensitive?n:n.toLowerCase());let _=-1;if(t.regex){let h=RegExp(u,t.caseSensitive?"g":"gi"),p;if(s)for(;p=h.exec(f.slice(0,c));){let b=h.lastIndex-p[0].length;p[0].length>0&&this._satisfiesWholeWord(b,f,p[0],t)&&(_=b,i=p[0]),h.lastIndex=b+1}else for(h.lastIndex=c;p=h.exec(f);){let b=h.lastIndex-p[0].length;if(p[0].length>0&&this._satisfiesWholeWord(b,f,p[0],t)){_=b,i=p[0];break}h.lastIndex=b+1}}else if(s){let h=c-u.length>=0?f.lastIndexOf(u,c-u.length):-1;for(;h>=0&&!this._satisfiesWholeWord(h,f,u,t);)h=h>0?f.lastIndexOf(u,h-1):-1;_=h}else{let h=f.indexOf(u,c);for(;h>=0&&!this._satisfiesWholeWord(h,f,u,t);)h=f.indexOf(u,h+1);_=h}if(_>=0){let h=0;for(;h=a[h+1];)h++;let p=h;for(;p=a[p+1];)p++;let b=_-a[h],O=_+i.length-a[p],L=this._stringLengthToBufferSize(r+h,b),W=this._stringLengthToBufferSize(r+p,O)-L+this._terminal.cols*(p-h);return{term:i,col:L,row:r+h,size:W}}}_stringLengthToBufferSize(i,e){let t=this._terminal.buffer.active.getLine(i);if(!t)return 0;for(let s=0;s1&&(e-=l.length-1);let o=t.getCell(s+1);o&&o.getWidth()===0&&e++}return e}_bufferColsToStringOffset(i,e,t){let s=Math.min(Math.floor(e/this._terminal.cols),t.length-1),r=t[s],l=this._terminal.buffer.active.getLine(i+s);if(l){let o=Math.min(e-s*this._terminal.cols,this._terminal.cols);for(let n=0;nthis.clearHighlightDecorations()))}createHighlightDecorations(e,t){this.clearHighlightDecorations();for(let s of e){let r=this._createResultDecorations(s,t,!1);if(r)for(let l of r)this._storeDecoration(l,s)}}createActiveDecoration(e,t){let s=this._createResultDecorations(e,t,!0);if(s)return{decorations:s,match:e,dispose(){D(s)}}}clearHighlightDecorations(){D(this._highlightDecorations),this._highlightDecorations=[],this._highlightedLines.clear()}_storeDecoration(e,t){this._highlightedLines.add(e.marker.line),this._highlightDecorations.push({decoration:e,match:t,dispose(){e.dispose()}})}_applyStyles(e,t,s){e.classList.contains("xterm-find-result-decoration")||(e.classList.add("xterm-find-result-decoration"),t&&(e.style.outline=`1px solid ${t}`)),s&&e.classList.add("xterm-find-active-result-decoration")}_createResultDecorations(e,t,s){let r=[],l=e.col,o=e.size,n=-this._terminal.buffer.active.baseY-this._terminal.buffer.active.cursorY+e.row;for(;o>0;){let c=Math.min(this._terminal.cols-l,o);r.push([n,l,c]),l=0,o-=c,n++}let a=[];for(let c of r){let u=this._terminal.registerMarker(c[0]),f=this._terminal.registerDecoration({marker:u,x:c[1],width:c[2],layer:s?"top":"bottom",backgroundColor:s?t.activeMatchBackground:t.matchBackground,overviewRulerOptions:this._highlightedLines.has(u.line)?void 0:{color:s?t.activeMatchColorOverviewRuler:t.matchOverviewRuler,position:"center"}});if(f){let _=[];_.push(u),_.push(f.onRender(h=>this._applyStyles(h,s?t.activeMatchBorder:t.matchBorder,!1))),_.push(f.onDispose(()=>D(_))),a.push(f)}}return a.length===0?void 0:a}};var y=class extends g{constructor(){super(...arguments);this._searchResults=[];this._onDidChangeResults=this._register(new I)}get onDidChangeResults(){return this._onDidChangeResults.event}get searchResults(){return this._searchResults}get selectedDecoration(){return this._selectedDecoration}set selectedDecoration(e){this._selectedDecoration=e}updateResults(e,t){this._searchResults=e.slice(0,t)}clearResults(){this._searchResults=[]}clearSelectedDecoration(){this._selectedDecoration&&(this._selectedDecoration.dispose(),this._selectedDecoration=void 0)}findResultIndex(e){for(let t=0;tthis._updateMatches())),this._register(this._terminal.onResize(()=>this._updateMatches())),this._register(m(()=>this.clearDecorations()))}_updateMatches(){this._highlightTimeout.clear(),this._state.cachedSearchTerm&&this._state.lastSearchOptions?.decorations&&(this._highlightTimeout.value=T(()=>{let e=this._state.cachedSearchTerm;this._state.clearCachedTerm(),this.findPrevious(e,{...this._state.lastSearchOptions,incremental:!0},{noScroll:!0})},200))}clearDecorations(e){this._resultTracker.clearSelectedDecoration(),this._decorationManager?.clearHighlightDecorations(),this._resultTracker.clearResults(),e||this._state.clearCachedTerm()}clearActiveDecoration(){this._resultTracker.clearSelectedDecoration()}findNext(e,t,s){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);let r=this._findNextAndSelect(e,t,s);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),r}_highlightAllMatches(e,t){if(!this._terminal||!this._engine||!this._decorationManager)throw new Error("Cannot use addon until it has been loaded");if(!this._state.isValidSearchTerm(e)){this.clearDecorations();return}this.clearDecorations(!0);let s=[],r,l=this._engine.find(e,0,0,t);for(;l&&(r?.row!==l.row||r?.col!==l.col)&&!(s.length>=this._highlightLimit);){r=l,s.push(r);let o=this._terminal.cols,n=r.col+r.size,a=r.row;n>=o&&(a+=Math.floor(n/o),n=n%o),l=this._engine.find(e,a,n,t)}this._resultTracker.updateResults(s,this._highlightLimit),t.decorations&&this._decorationManager.createHighlightDecorations(s,t.decorations)}_findNextAndSelect(e,t,s){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;let r=this._engine.findNextWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(r,t?.decorations,s?.noScroll)}findPrevious(e,t,s){if(!this._terminal||!this._engine)throw new Error("Cannot use addon until it has been loaded");this._onBeforeSearch.fire(),this._state.lastSearchOptions=t,this._state.shouldUpdateHighlighting(e,t)&&this._highlightAllMatches(e,t);let r=this._findPreviousAndSelect(e,t,s);return this._fireResults(t),this._state.cachedSearchTerm=e,this._onAfterSearch.fire(),r}_fireResults(e){this._resultTracker.fireResultsChanged(!!e?.decorations)}_findPreviousAndSelect(e,t,s){if(!this._terminal||!this._engine)return!1;if(!this._state.isValidSearchTerm(e))return this._terminal.clearSelection(),this.clearDecorations(),!1;let r=this._engine.findPreviousWithSelection(e,t,this._state.cachedSearchTerm);return this._selectResult(r,t?.decorations,s?.noScroll)}_selectResult(e,t,s){if(!this._terminal||!this._decorationManager)return!1;if(this._resultTracker.clearSelectedDecoration(),!e)return this._terminal.clearSelection(),!1;if(this._terminal.select(e.col,e.row,e.size),t){let r=this._decorationManager.createActiveDecoration(e,t);r&&(this._resultTracker.selectedDecoration=r)}if(!s&&(e.row>=this._terminal.buffer.active.viewportY+this._terminal.rows||e.row void): IDisposable {\n return { dispose: fn };\n}\n\nexport function dispose(disposable: T): T;\nexport function dispose(disposable: T | undefined): T | undefined;\nexport function dispose(disposables: T[]): T[];\nexport function dispose(arg: T | T[] | undefined): T | T[] | undefined {\n if (!arg) {\n return arg;\n }\n if (Array.isArray(arg)) {\n for (const d of arg) {\n d.dispose();\n }\n return [];\n }\n arg.dispose();\n return arg;\n}\n\nexport function combinedDisposable(...disposables: IDisposable[]): IDisposable {\n return toDisposable(() => dispose(disposables));\n}\n\nexport class DisposableStore implements IDisposable {\n private readonly _disposables = new Set();\n private _isDisposed = false;\n\n public get isDisposed(): boolean {\n return this._isDisposed;\n }\n\n public add(o: T): T {\n if (this._isDisposed) {\n o.dispose();\n } else {\n this._disposables.add(o);\n }\n return o;\n }\n\n public dispose(): void {\n if (this._isDisposed) {\n return;\n }\n this._isDisposed = true;\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n\n public clear(): void {\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n}\n\nexport abstract class Disposable implements IDisposable {\n public static readonly None: IDisposable = Object.freeze({ dispose() { } });\n\n protected readonly _store = new DisposableStore();\n\n public dispose(): void {\n this._store.dispose();\n }\n\n protected _register(o: T): T {\n return this._store.add(o);\n }\n}\n\nexport class MutableDisposable implements IDisposable {\n private _value: T | undefined;\n private _isDisposed = false;\n\n public get value(): T | undefined {\n return this._isDisposed ? undefined : this._value;\n }\n\n public set value(value: T | undefined) {\n if (this._isDisposed || value === this._value) {\n return;\n }\n this._value?.dispose();\n this._value = value;\n }\n\n public clear(): void {\n this.value = undefined;\n }\n\n public dispose(): void {\n this._isDisposed = true;\n this._value?.dispose();\n this._value = undefined;\n }\n}\n", "/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal event utilities for xterm.js core.\n * Simplified from VS Code's event.ts - no leak detection/profiling.\n */\n\nimport { IDisposable, DisposableStore, toDisposable } from './Lifecycle';\n\nexport interface IEvent {\n (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore): IDisposable;\n}\n\nexport class Emitter {\n private _listeners: { fn: (e: T) => any, thisArgs: any }[] = [];\n private _disposed = false;\n private _event: IEvent | undefined;\n\n public get event(): IEvent {\n if (this._event) {\n return this._event;\n }\n this._event = (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n if (this._disposed) {\n return toDisposable(() => {});\n }\n\n const entry = { fn: listener, thisArgs };\n this._listeners = this._listeners.slice();\n this._listeners.push(entry);\n\n const result = toDisposable(() => {\n const idx = this._listeners.indexOf(entry);\n if (idx !== -1) {\n this._listeners = this._listeners.slice();\n this._listeners.splice(idx, 1);\n }\n });\n\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(result);\n } else {\n disposables.add(result);\n }\n }\n\n return result;\n };\n return this._event;\n }\n\n public fire(event: T): void {\n if (this._disposed || !this._listeners.length) {\n return;\n }\n if (this._listeners.length === 1) {\n this._listeners[0].fn.call(this._listeners[0].thisArgs, event);\n return;\n }\n const listeners = this._listeners;\n for (let i = 0, len = listeners.length; i < len; ++i) {\n listeners[i].fn.call(listeners[i].thisArgs, event);\n }\n }\n\n public dispose(): void {\n if (this._disposed) {\n return;\n }\n this._disposed = true;\n this._listeners.length = 0;\n }\n}\n\nexport namespace EventUtils {\n export function forward(from: IEvent, to: Emitter): IDisposable {\n return from(e => to.fire(e));\n }\n\n export function map(event: IEvent, map: (i: I) => O): IEvent {\n return (listener: (e: O) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n return event(i => listener.call(thisArgs, map(i)), undefined, disposables);\n };\n }\n\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent {\n return (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n const store = new DisposableStore();\n for (const event of events) {\n store.add(event(e => listener.call(thisArgs, e)));\n }\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(store);\n } else {\n disposables.add(store);\n }\n }\n return store;\n };\n }\n\n export function runAndSubscribe(event: IEvent, handler: (e: T) => void, initial: T): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void, initial?: T): IDisposable {\n handler(initial);\n return event(e => handler(e));\n }\n}\n", "/**\n * Copyright (c) 2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal async helpers for xterm.js core.\n */\n\nimport { DisposableStore, IDisposable, toDisposable } from './Lifecycle';\n\nexport function timeout(millis: number): Promise {\n return new Promise(resolve => setTimeout(resolve, millis));\n}\n\n/**\n * Creates a timeout that can be disposed using its returned value.\n * @param handler The timeout handler.\n * @param timeout An optional timeout in milliseconds.\n * @param store An optional {@link DisposableStore} that will have the timeout disposable managed\n * automatically.\n */\nexport function disposableTimeout(handler: () => void, timeout = 0, store?: DisposableStore): IDisposable {\n const timer = setTimeout(() => {\n handler();\n if (store) {\n disposable.dispose();\n }\n }, timeout);\n const disposable = toDisposable(() => {\n clearTimeout(timer);\n });\n store?.add(disposable);\n return disposable;\n}\n\nexport class TimeoutTimer implements IDisposable {\n private _token: any = -1;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n if (this._token !== -1) {\n clearTimeout(this._token);\n this._token = -1;\n }\n }\n\n public cancelAndSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed TimeoutTimer');\n }\n this.cancel();\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n\n public setIfNotSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling setIfNotSet on a disposed TimeoutTimer');\n }\n if (this._token !== -1) {\n return;\n }\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n}\n\n/**\n * Schedules a single runner on the microtask queue. Unlike {@link TimeoutTimer}, a scheduled\n * microtask cannot be unqueued; {@link cancel} prevents the runner from executing if it has not\n * run yet.\n */\nexport class MicrotaskTimer implements IDisposable {\n private _isScheduled = false;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n this._isScheduled = false;\n }\n\n public set(runner: () => void): void {\n if (this._isDisposed) {\n throw new Error('Calling set on a disposed MicrotaskTimer');\n }\n if (this._isScheduled) {\n return;\n }\n this._isScheduled = true;\n queueMicrotask(() => {\n if (!this._isScheduled) {\n return;\n }\n this._isScheduled = false;\n runner();\n });\n }\n}\n\nexport class IntervalTimer implements IDisposable {\n private _disposable: IDisposable | undefined;\n private _isDisposed = false;\n\n public cancel(): void {\n this._disposable?.dispose();\n this._disposable = undefined;\n }\n\n public cancelAndSet(runner: () => void, interval: number, context: Window | typeof globalThis = globalThis): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed IntervalTimer');\n }\n this.cancel();\n const handle = context.setInterval(() => {\n runner();\n }, interval);\n this._disposable = {\n dispose: () => {\n context.clearInterval(handle as any);\n this._disposable = undefined;\n }\n };\n }\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport { combinedDisposable, Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\n\nexport type LineCacheEntry = [\n /**\n * The string representation of a line (as opposed to the buffer cell representation).\n */\n lineAsString: string,\n /**\n * The offsets where each line starts when the entry describes a wrapped line.\n */\n lineOffsets: number[]\n];\n\n/**\n * Configuration constants for the search line cache functionality.\n */\nconst enum Constants {\n /**\n * Time-to-live for cached search results in milliseconds. After this duration, cached search\n * results will be invalidated to ensure they remain consistent with terminal content changes.\n */\n LINES_CACHE_TIME_TO_LIVE = 15000\n}\n\nexport class SearchLineCache extends Disposable {\n /**\n * translateBufferLineToStringWithWrap is a fairly expensive call.\n * We memoize the calls into an array that has a time based ttl.\n * _linesCache is also invalidated when the terminal cursor moves.\n */\n private _linesCache: LineCacheEntry[] | undefined;\n private _linesCacheTimeout = this._register(new MutableDisposable());\n private _linesCacheDisposables = this._register(new MutableDisposable());\n // Track access to avoid recreating a timeout on every init call which occurs once per search\n // result (findNext/findPrevious -> _highlightAllMatches -> find loop).\n private _lastAccessTimestamp = 0;\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this._destroyLinesCache()));\n }\n\n /**\n * Sets up a line cache with a ttl\n */\n public initLinesCache(): void {\n if (!this._linesCache) {\n this._linesCache = new Array(this._terminal.buffer.active.length);\n this._linesCacheDisposables.value = combinedDisposable(\n this._terminal.onLineFeed(() => this._destroyLinesCache()),\n this._terminal.onCursorMove(() => this._destroyLinesCache()),\n this._terminal.onResize(() => this._destroyLinesCache())\n );\n }\n\n this._lastAccessTimestamp = Date.now();\n if (!this._linesCacheTimeout.value) {\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE);\n }\n }\n\n private _destroyLinesCache(): void {\n this._linesCache = undefined;\n this._lastAccessTimestamp = 0;\n this._linesCacheDisposables.clear();\n this._linesCacheTimeout.clear();\n }\n\n private _scheduleLinesCacheTimeout(delay: number): void {\n this._linesCacheTimeout.value = disposableTimeout(() => {\n if (!this._linesCache) {\n return;\n }\n const now = Date.now();\n const elapsed = now - this._lastAccessTimestamp;\n if (elapsed >= Constants.LINES_CACHE_TIME_TO_LIVE) {\n this._destroyLinesCache();\n return;\n }\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE - elapsed);\n }, delay);\n }\n\n public getLineFromCache(row: number): LineCacheEntry | undefined {\n return this._linesCache?.[row];\n }\n\n public setLineInCache(row: number, entry: LineCacheEntry): void {\n if (this._linesCache) {\n this._linesCache[row] = entry;\n }\n }\n\n /**\n * Translates a buffer line to a string, including subsequent lines if they are wraps.\n * Wide characters will count as two columns in the resulting string. This\n * function is useful for getting the actual text underneath the raw selection\n * position.\n * @param lineIndex The index of the line being translated.\n * @param trimRight Whether to trim whitespace to the right.\n */\n public translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean): LineCacheEntry {\n const strings = [];\n const lineOffsets = [0];\n let line = this._terminal.buffer.active.getLine(lineIndex);\n while (line) {\n const nextLine = this._terminal.buffer.active.getLine(lineIndex + 1);\n const lineWrapsToNext = nextLine ? nextLine.isWrapped : false;\n let string = line.translateToString(!lineWrapsToNext && trimRight);\n if (lineWrapsToNext && nextLine) {\n const lastCell = line.getCell(line.length - 1);\n const lastCellIsNull = lastCell && lastCell.getCode() === 0 && lastCell.getWidth() === 1;\n // a wide character wrapped to the next line\n if (lastCellIsNull && nextLine.getCell(0)?.getWidth() === 2) {\n string = string.slice(0, -1);\n }\n }\n strings.push(string);\n if (lineWrapsToNext) {\n lineOffsets.push(lineOffsets[lineOffsets.length - 1] + string.length);\n } else {\n break;\n }\n lineIndex++;\n line = nextLine;\n }\n return [strings.join(''), lineOffsets];\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchOptions } from '@xterm/addon-search';\n\n/**\n * Manages search state including cached search terms, options tracking, and validation.\n * This class provides a centralized way to handle search state consistency and option changes.\n */\nexport class SearchState {\n private _cachedSearchTerm: string | undefined;\n private _lastSearchOptions: ISearchOptions | undefined;\n\n /**\n * Gets the currently cached search term.\n */\n public get cachedSearchTerm(): string | undefined {\n return this._cachedSearchTerm;\n }\n\n /**\n * Sets the cached search term.\n */\n public set cachedSearchTerm(term: string | undefined) {\n this._cachedSearchTerm = term;\n }\n\n /**\n * Gets the last search options used.\n */\n public get lastSearchOptions(): ISearchOptions | undefined {\n return this._lastSearchOptions;\n }\n\n /**\n * Sets the last search options used.\n */\n public set lastSearchOptions(options: ISearchOptions | undefined) {\n this._lastSearchOptions = options;\n }\n\n /**\n * Validates a search term to ensure it's not empty or invalid.\n * @param term The search term to validate.\n * @returns true if the term is valid for searching.\n */\n public isValidSearchTerm(term: string): boolean {\n return !!(term && term.length > 0);\n }\n\n /**\n * Determines if search options have changed compared to the last search.\n * @param newOptions The new search options to compare.\n * @returns true if the options have changed.\n */\n public didOptionsChange(newOptions?: ISearchOptions): boolean {\n if (!this._lastSearchOptions) {\n return true;\n }\n if (!newOptions) {\n return false;\n }\n if (this._lastSearchOptions.caseSensitive !== newOptions.caseSensitive) {\n return true;\n }\n if (this._lastSearchOptions.regex !== newOptions.regex) {\n return true;\n }\n if (this._lastSearchOptions.wholeWord !== newOptions.wholeWord) {\n return true;\n }\n return false;\n }\n\n /**\n * Determines if a new search should trigger highlighting updates.\n * @param term The search term.\n * @param options The search options.\n * @returns true if highlighting should be updated.\n */\n public shouldUpdateHighlighting(term: string, options?: ISearchOptions): boolean {\n if (!options?.decorations) {\n return false;\n }\n return this._cachedSearchTerm === undefined ||\n term !== this._cachedSearchTerm ||\n this.didOptionsChange(options);\n }\n\n /**\n * Clears the cached search term.\n */\n public clearCachedTerm(): void {\n this._cachedSearchTerm = undefined;\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this._cachedSearchTerm = undefined;\n this._lastSearchOptions = undefined;\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport type { ISearchOptions } from '@xterm/addon-search';\nimport type { SearchLineCache } from './SearchLineCache';\n\n/**\n * Represents the position to start a search from.\n */\ninterface ISearchPosition {\n startCol: number;\n startRow: number;\n}\n\n/**\n * Represents a search result with its position and content.\n */\nexport interface ISearchResult {\n term: string;\n col: number;\n row: number;\n size: number;\n}\n\n/**\n * Configuration constants for the search engine functionality.\n */\nconst enum Constants {\n /**\n * Characters that are considered non-word characters for search boundary detection. These\n * characters are used to determine word boundaries when performing whole-word searches. Includes\n * common punctuation, symbols, and whitespace characters.\n */\n NON_WORD_CHARACTERS = ' ~!@#$%^&*()+`-=[]{}|\\\\;:\"\\',./<>?'\n}\n\n/**\n * Core search engine that handles finding text within terminal content.\n * This class is responsible for the actual search algorithms and position calculations.\n */\nexport class SearchEngine {\n constructor(\n private readonly _terminal: Terminal,\n private readonly _lineCache: SearchLineCache\n ) {}\n\n /**\n * Find the first occurrence of a term starting from a specific position.\n * @param term The search term.\n * @param startRow The row to start searching from.\n * @param startCol The column to start searching from.\n * @param searchOptions Search options.\n * @returns The search result if found, undefined otherwise.\n */\n public find(term: string, startRow: number, startCol: number, searchOptions?: ISearchOptions): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n if (startCol >= this._terminal.cols) {\n throw new Error(`Invalid col: ${startCol} to search in terminal of ${this._terminal.cols} cols`);\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n return result;\n }\n\n /**\n * Find the next occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine incremental behavior.\n * @returns The search result if found, undefined otherwise.\n */\n public findNextWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startCol = 0;\n let startRow = 0;\n if (prevSelectedPos) {\n if (cachedSearchTerm === term) {\n startCol = prevSelectedPos.end.x;\n startRow = prevSelectedPos.end.y;\n } else {\n startCol = prevSelectedPos.start.x;\n startRow = prevSelectedPos.start.y;\n }\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n // If we hit the bottom and didn't search from the very top wrap back up\n if (!result && startRow !== 0) {\n for (let y = 0; y < startRow; y++) {\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n\n // If there is only one result, wrap back and return selection if it exists.\n if (!result && prevSelectedPos) {\n searchPosition.startRow = prevSelectedPos.start.y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n }\n\n return result;\n }\n\n /**\n * Find the previous occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine if expansion should occur.\n * @returns The search result if found, undefined otherwise.\n */\n public findPreviousWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startRow = this._terminal.buffer.active.baseY + this._terminal.rows - 1;\n const startCol = this._terminal.cols;\n const isReverseSearch = true;\n\n this._lineCache.initLinesCache();\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n let result: ISearchResult | undefined;\n if (prevSelectedPos) {\n searchPosition.startRow = startRow = prevSelectedPos.start.y;\n searchPosition.startCol = prevSelectedPos.start.x;\n if (cachedSearchTerm !== term) {\n // Try to expand selection to right first.\n result = this._findInLine(term, searchPosition, searchOptions, false);\n if (!result) {\n // If selection was not able to be expanded to the right, then try reverse search\n searchPosition.startRow = startRow = prevSelectedPos.end.y;\n searchPosition.startCol = prevSelectedPos.end.x;\n }\n }\n }\n\n result ??= this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n\n // Search from startRow - 1 to top\n if (!result) {\n searchPosition.startCol = Math.max(searchPosition.startCol, this._terminal.cols);\n for (let y = startRow - 1; y >= 0; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n // If we hit the top and didn't search from the very bottom wrap back down\n if (!result && startRow !== (this._terminal.buffer.active.baseY + this._terminal.rows - 1)) {\n for (let y = (this._terminal.buffer.active.baseY + this._terminal.rows - 1); y >= startRow; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n\n return result;\n }\n\n /**\n * A found substring is a whole word if it doesn't have an alphanumeric character directly\n * adjacent to it.\n * @param searchIndex starting index of the potential whole word substring\n * @param line entire string in which the potential whole word was found\n * @param term the substring that starts at searchIndex\n */\n private _isWholeWord(searchIndex: number, line: string, term: string): boolean {\n return ((searchIndex === 0) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex - 1]))) &&\n (((searchIndex + term.length) === line.length) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex + term.length])));\n }\n\n /**\n * Searches a line for a search term. Takes the provided terminal line and searches the text line,\n * which may contain subsequent terminal lines if the text is wrapped. If the provided line number\n * is part of a wrapped text line that started on an earlier line then it is skipped since it will\n * be properly searched when the terminal line that the text starts on is searched.\n * @param term The search term.\n * @param searchPosition The position to start the search.\n * @param searchOptions Search options.\n * @param isReverseSearch Whether the search should start from the right side of the terminal and\n * search to the left.\n * @returns The search result if it was found.\n */\n private _findInLine(term: string, searchPosition: ISearchPosition, searchOptions: ISearchOptions = {}, isReverseSearch: boolean = false): ISearchResult | undefined {\n const row = searchPosition.startRow;\n const col = searchPosition.startCol;\n\n // Ignore wrapped lines, only consider on unwrapped line (first row of command string).\n const firstLine = this._terminal.buffer.active.getLine(row);\n if (firstLine?.isWrapped) {\n if (isReverseSearch) {\n searchPosition.startCol += this._terminal.cols;\n return;\n }\n\n // This will iterate until we find the line start.\n // When we find it, we will search using the calculated start column.\n searchPosition.startRow--;\n searchPosition.startCol += this._terminal.cols;\n return this._findInLine(term, searchPosition, searchOptions);\n }\n let cache = this._lineCache.getLineFromCache(row);\n if (!cache) {\n cache = this._lineCache.translateBufferLineToStringWithWrap(row, true);\n this._lineCache.setLineInCache(row, cache);\n }\n const [stringLine, offsets] = cache;\n\n const offset = this._bufferColsToStringOffset(row, col);\n let searchTerm = term;\n let searchStringLine = stringLine;\n if (!searchOptions.regex) {\n searchTerm = searchOptions.caseSensitive ? term : term.toLowerCase();\n searchStringLine = searchOptions.caseSensitive ? stringLine : stringLine.toLowerCase();\n }\n\n let resultIndex = -1;\n if (searchOptions.regex) {\n const searchRegex = RegExp(searchTerm, searchOptions.caseSensitive ? 'g' : 'gi');\n let foundTerm: RegExpExecArray | null;\n if (isReverseSearch) {\n // This loop will get the resultIndex of the _last_ regex match in the range 0..offset\n while (foundTerm = searchRegex.exec(searchStringLine.slice(0, offset))) {\n resultIndex = searchRegex.lastIndex - foundTerm[0].length;\n term = foundTerm[0];\n searchRegex.lastIndex -= (term.length - 1);\n }\n } else {\n foundTerm = searchRegex.exec(searchStringLine.slice(offset));\n if (foundTerm && foundTerm[0].length > 0) {\n resultIndex = offset + (searchRegex.lastIndex - foundTerm[0].length);\n term = foundTerm[0];\n }\n }\n } else {\n if (isReverseSearch) {\n if (offset - searchTerm.length >= 0) {\n resultIndex = searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length);\n }\n } else {\n resultIndex = searchStringLine.indexOf(searchTerm, offset);\n }\n }\n\n if (resultIndex >= 0) {\n if (searchOptions.wholeWord && !this._isWholeWord(resultIndex, searchStringLine, term)) {\n return;\n }\n\n // Adjust the row number and search index if needed since a \"line\" of text can span multiple\n // rows\n let startRowOffset = 0;\n while (startRowOffset < offsets.length - 1 && resultIndex >= offsets[startRowOffset + 1]) {\n startRowOffset++;\n }\n let endRowOffset = startRowOffset;\n while (endRowOffset < offsets.length - 1 && resultIndex + term.length >= offsets[endRowOffset + 1]) {\n endRowOffset++;\n }\n const startColOffset = resultIndex - offsets[startRowOffset];\n const endColOffset = resultIndex + term.length - offsets[endRowOffset];\n const startColIndex = this._stringLengthToBufferSize(row + startRowOffset, startColOffset);\n const endColIndex = this._stringLengthToBufferSize(row + endRowOffset, endColOffset);\n const size = endColIndex - startColIndex + this._terminal.cols * (endRowOffset - startRowOffset);\n\n return {\n term,\n col: startColIndex,\n row: row + startRowOffset,\n size\n };\n }\n }\n\n private _stringLengthToBufferSize(row: number, offset: number): number {\n const line = this._terminal.buffer.active.getLine(row);\n if (!line) {\n return 0;\n }\n for (let i = 0; i < offset; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n // Adjust the searchIndex to normalize emoji into single chars\n const char = cell.getChars();\n if (char.length > 1) {\n offset -= char.length - 1;\n }\n // Adjust the searchIndex for empty characters following wide unicode\n // chars (eg. CJK)\n const nextCell = line.getCell(i + 1);\n if (nextCell && nextCell.getWidth() === 0) {\n offset++;\n }\n }\n return offset;\n }\n\n private _bufferColsToStringOffset(startRow: number, cols: number): number {\n let lineIndex = startRow;\n let offset = 0;\n let line = this._terminal.buffer.active.getLine(lineIndex);\n while (cols > 0 && line) {\n for (let i = 0; i < cols && i < this._terminal.cols; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n if (cell.getWidth()) {\n // Treat null characters as whitespace to align with the translateToString API\n offset += cell.getCode() === 0 ? 1 : cell.getChars().length;\n }\n }\n lineIndex++;\n line = this._terminal.buffer.active.getLine(lineIndex);\n if (line && !line.isWrapped) {\n break;\n }\n cols -= this._terminal.cols;\n }\n return offset;\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, IDecoration } from '@xterm/xterm';\nimport type { ISearchDecorationOptions } from '@xterm/addon-search';\nimport { dispose, Disposable, toDisposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a highlight decoration.\n */\ninterface IHighlight extends IDisposable {\n decoration: IDecoration;\n match: ISearchResult;\n}\n\n/**\n * Interface for managing multiple decorations for a single match.\n */\ninterface IMultiHighlight extends IDisposable {\n decorations: IDecoration[];\n match: ISearchResult;\n}\n\n/**\n * Manages visual decorations for search results including highlighting and active selection\n * indicators. This class handles the creation, styling, and disposal of search-related decorations.\n */\nexport class DecorationManager extends Disposable {\n private _highlightDecorations: IHighlight[] = [];\n private _highlightedLines: Set = new Set();\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this.clearHighlightDecorations()));\n }\n\n /**\n * Creates decorations for all provided search results.\n * @param results The search results to create decorations for.\n * @param options The decoration options.\n */\n public createHighlightDecorations(results: ISearchResult[], options: ISearchDecorationOptions): void {\n this.clearHighlightDecorations();\n\n for (const match of results) {\n const decorations = this._createResultDecorations(match, options, false);\n if (decorations) {\n for (const decoration of decorations) {\n this._storeDecoration(decoration, match);\n }\n }\n }\n }\n\n /**\n * Creates decorations for the currently active search result.\n * @param result The active search result.\n * @param options The decoration options.\n * @returns The multi-highlight decoration or undefined if creation failed.\n */\n public createActiveDecoration(result: ISearchResult, options: ISearchDecorationOptions): IMultiHighlight | undefined {\n const decorations = this._createResultDecorations(result, options, true);\n if (decorations) {\n return { decorations, match: result, dispose() { dispose(decorations); } };\n }\n return undefined;\n }\n\n /**\n * Clears all highlight decorations.\n */\n public clearHighlightDecorations(): void {\n dispose(this._highlightDecorations);\n this._highlightDecorations = [];\n this._highlightedLines.clear();\n }\n\n /**\n * Stores a decoration and tracks it for management.\n * @param decoration The decoration to store.\n * @param match The search result this decoration represents.\n */\n private _storeDecoration(decoration: IDecoration, match: ISearchResult): void {\n this._highlightedLines.add(decoration.marker.line);\n this._highlightDecorations.push({ decoration, match, dispose() { decoration.dispose(); } });\n }\n\n /**\n * Applies styles to the decoration when it is rendered.\n * @param element The decoration's element.\n * @param borderColor The border color to apply.\n * @param isActiveResult Whether the element is part of the active search result.\n */\n private _applyStyles(element: HTMLElement, borderColor: string | undefined, isActiveResult: boolean): void {\n if (!element.classList.contains('xterm-find-result-decoration')) {\n element.classList.add('xterm-find-result-decoration');\n if (borderColor) {\n element.style.outline = `1px solid ${borderColor}`;\n }\n }\n if (isActiveResult) {\n element.classList.add('xterm-find-active-result-decoration');\n }\n }\n\n /**\n * Creates a decoration for the result and applies styles\n * @param result the search result for which to create the decoration\n * @param options the options for the decoration\n * @param isActiveResult whether this is the currently active result\n * @returns the decorations or undefined if the marker has already been disposed of\n */\n private _createResultDecorations(result: ISearchResult, options: ISearchDecorationOptions, isActiveResult: boolean): IDecoration[] | undefined {\n // Gather decoration ranges for this match as it could wrap\n const decorationRanges: [number, number, number][] = [];\n let currentCol = result.col;\n let remainingSize = result.size;\n let markerOffset = -this._terminal.buffer.active.baseY - this._terminal.buffer.active.cursorY + result.row;\n while (remainingSize > 0) {\n const amountThisRow = Math.min(this._terminal.cols - currentCol, remainingSize);\n decorationRanges.push([markerOffset, currentCol, amountThisRow]);\n currentCol = 0;\n remainingSize -= amountThisRow;\n markerOffset++;\n }\n\n // Create the decorations\n const decorations: IDecoration[] = [];\n for (const range of decorationRanges) {\n const marker = this._terminal.registerMarker(range[0]);\n const decoration = this._terminal.registerDecoration({\n marker,\n x: range[1],\n width: range[2],\n layer: isActiveResult ? 'top' : 'bottom',\n backgroundColor: isActiveResult ? options.activeMatchBackground : options.matchBackground,\n overviewRulerOptions: this._highlightedLines.has(marker.line) ? undefined : {\n color: isActiveResult ? options.activeMatchColorOverviewRuler : options.matchOverviewRuler,\n position: 'center'\n }\n });\n if (decoration) {\n const disposables: IDisposable[] = [];\n disposables.push(marker);\n disposables.push(decoration.onRender((e) => this._applyStyles(e, isActiveResult ? options.activeMatchBorder : options.matchBorder, false)));\n disposables.push(decoration.onDispose(() => dispose(disposables)));\n decorations.push(decoration);\n }\n }\n\n return decorations.length === 0 ? undefined : decorations;\n }\n}\n\n\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchResultChangeEvent } from '@xterm/addon-search';\nimport type { IDisposable } from '@xterm/xterm';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a currently selected decoration.\n */\ninterface ISelectedDecoration extends IDisposable {\n match: ISearchResult;\n}\n\n/**\n * Tracks search results, manages result indexing, and fires events when results change.\n * This class provides centralized management of search result state and notifications.\n */\nexport class SearchResultTracker extends Disposable {\n private _searchResults: ISearchResult[] = [];\n private _selectedDecoration: ISelectedDecoration | undefined;\n\n private readonly _onDidChangeResults = this._register(new Emitter());\n public get onDidChangeResults(): IEvent { return this._onDidChangeResults.event; }\n\n /**\n * Gets the current search results.\n */\n public get searchResults(): ReadonlyArray {\n return this._searchResults;\n }\n\n /**\n * Gets the currently selected decoration.\n */\n public get selectedDecoration(): ISelectedDecoration | undefined {\n return this._selectedDecoration;\n }\n\n /**\n * Sets the currently selected decoration.\n */\n public set selectedDecoration(decoration: ISelectedDecoration | undefined) {\n this._selectedDecoration = decoration;\n }\n\n /**\n * Updates the search results with a new set of results.\n * @param results The new search results.\n * @param maxResults The maximum number of results to track.\n */\n public updateResults(results: ISearchResult[], maxResults: number): void {\n this._searchResults = results.slice(0, maxResults);\n }\n\n /**\n * Clears all search results.\n */\n public clearResults(): void {\n this._searchResults = [];\n }\n\n /**\n * Clears the selected decoration.\n */\n public clearSelectedDecoration(): void {\n if (this._selectedDecoration) {\n this._selectedDecoration.dispose();\n this._selectedDecoration = undefined;\n }\n }\n\n /**\n * Finds the index of a result in the current results array.\n * @param result The result to find.\n * @returns The index of the result, or -1 if not found.\n */\n public findResultIndex(result: ISearchResult): number {\n for (let i = 0; i < this._searchResults.length; i++) {\n const match = this._searchResults[i];\n if (match.row === result.row && match.col === result.col && match.size === result.size) {\n return i;\n }\n }\n return -1;\n }\n\n /**\n * Fires a result change event with the current state.\n * @param hasDecorations Whether decorations are enabled.\n */\n public fireResultsChanged(hasDecorations: boolean): void {\n if (!hasDecorations) {\n return;\n }\n\n let resultIndex = -1;\n if (this._selectedDecoration) {\n resultIndex = this.findResultIndex(this._selectedDecoration.match);\n }\n\n this._onDidChangeResults.fire({\n resultIndex,\n resultCount: this._searchResults.length\n });\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this.clearSelectedDecoration();\n this.clearResults();\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, ITerminalAddon } from '@xterm/xterm';\nimport type { SearchAddon as ISearchApi, ISearchOptions, ISearchAddonOptions, ISearchResultChangeEvent, ISearchDecorationOptions } from '@xterm/addon-search';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\nimport { SearchLineCache } from './SearchLineCache';\nimport { SearchState } from './SearchState';\nimport { SearchEngine, type ISearchResult } from './SearchEngine';\nimport { DecorationManager } from './DecorationManager';\nimport { SearchResultTracker } from './SearchResultTracker';\n\ninterface IInternalSearchOptions {\n noScroll: boolean;\n}\n\n/**\n * Configuration constants for the search addon functionality.\n */\nconst enum Constants {\n /**\n * Default maximum number of search results to highlight simultaneously. This limit prevents\n * performance degradation when searching for very common terms that would result in excessive\n * highlighting decorations.\n */\n DEFAULT_HIGHLIGHT_LIMIT = 1000\n}\n\nexport class SearchAddon extends Disposable implements ITerminalAddon, ISearchApi {\n private _terminal: Terminal | undefined;\n private _highlightLimit: number;\n private _highlightTimeout = this._register(new MutableDisposable());\n private _lineCache = this._register(new MutableDisposable());\n\n // Component instances\n private _state = new SearchState();\n private _engine: SearchEngine | undefined;\n private _decorationManager: DecorationManager | undefined;\n private _resultTracker = this._register(new SearchResultTracker());\n\n private readonly _onAfterSearch = this._register(new Emitter());\n public readonly onAfterSearch = this._onAfterSearch.event;\n private readonly _onBeforeSearch = this._register(new Emitter());\n public readonly onBeforeSearch = this._onBeforeSearch.event;\n\n public get onDidChangeResults(): IEvent {\n return this._resultTracker.onDidChangeResults;\n }\n\n constructor(options?: Partial) {\n super();\n\n this._highlightLimit = options?.highlightLimit ?? Constants.DEFAULT_HIGHLIGHT_LIMIT;\n }\n\n public activate(terminal: Terminal): void {\n this._terminal = terminal;\n this._lineCache.value = new SearchLineCache(terminal);\n this._engine = new SearchEngine(terminal, this._lineCache.value);\n this._decorationManager = new DecorationManager(terminal);\n this._register(this._terminal.onWriteParsed(() => this._updateMatches()));\n this._register(this._terminal.onResize(() => this._updateMatches()));\n this._register(toDisposable(() => this.clearDecorations()));\n }\n\n private _updateMatches(): void {\n this._highlightTimeout.clear();\n if (this._state.cachedSearchTerm && this._state.lastSearchOptions?.decorations) {\n this._highlightTimeout.value = disposableTimeout(() => {\n const term = this._state.cachedSearchTerm;\n this._state.clearCachedTerm();\n this.findPrevious(term!, { ...this._state.lastSearchOptions, incremental: true }, { noScroll: true });\n }, 200);\n }\n }\n\n public clearDecorations(retainCachedSearchTerm?: boolean): void {\n this._resultTracker.clearSelectedDecoration();\n this._decorationManager?.clearHighlightDecorations();\n this._resultTracker.clearResults();\n if (!retainCachedSearchTerm) {\n this._state.clearCachedTerm();\n }\n }\n\n public clearActiveDecoration(): void {\n this._resultTracker.clearSelectedDecoration();\n }\n\n /**\n * Find the next instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findNext(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findNextAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _highlightAllMatches(term: string, searchOptions: ISearchOptions): void {\n if (!this._terminal || !this._engine || !this._decorationManager) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n if (!this._state.isValidSearchTerm(term)) {\n this.clearDecorations();\n return;\n }\n\n // new search, clear out the old decorations\n this.clearDecorations(true);\n\n const results: ISearchResult[] = [];\n let prevResult: ISearchResult | undefined = undefined;\n let result = this._engine.find(term, 0, 0, searchOptions);\n\n while (result && (prevResult?.row !== result.row || prevResult?.col !== result.col)) {\n if (results.length >= this._highlightLimit) {\n break;\n }\n prevResult = result;\n results.push(prevResult);\n const cols = this._terminal.cols;\n let nextCol = prevResult.col + prevResult.size;\n let nextRow = prevResult.row;\n if (nextCol >= cols) {\n nextRow += Math.floor(nextCol / cols);\n nextCol = nextCol % cols;\n }\n result = this._engine.find(term, nextRow, nextCol, searchOptions);\n }\n\n this._resultTracker.updateResults(results, this._highlightLimit);\n if (searchOptions.decorations) {\n this._decorationManager.createHighlightDecorations(results, searchOptions.decorations);\n }\n }\n\n private _findNextAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findNextWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Find the previous instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findPrevious(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findPreviousAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _fireResults(searchOptions?: ISearchOptions): void {\n this._resultTracker.fireResultsChanged(!!searchOptions?.decorations);\n }\n\n private _findPreviousAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findPreviousWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Selects and scrolls to a result.\n * @param result The result to select.\n * @returns Whether a result was selected.\n */\n private _selectResult(result: ISearchResult | undefined, options?: ISearchDecorationOptions, noScroll?: boolean): boolean {\n if (!this._terminal || !this._decorationManager) {\n return false;\n }\n\n this._resultTracker.clearSelectedDecoration();\n if (!result) {\n this._terminal.clearSelection();\n return false;\n }\n\n this._terminal.select(result.col, result.row, result.size);\n if (options) {\n const activeDecoration = this._decorationManager.createActiveDecoration(result, options);\n if (activeDecoration) {\n this._resultTracker.selectedDecoration = activeDecoration;\n }\n }\n\n if (!noScroll) {\n // If it is not in the viewport then we scroll else it just gets selected\n if (result.row >= (this._terminal.buffer.active.viewportY + this._terminal.rows) || result.row < this._terminal.buffer.active.viewportY) {\n let scroll = result.row - this._terminal.buffer.active.viewportY;\n scroll -= Math.floor(this._terminal.rows / 2);\n this._terminal.scrollLines(scroll);\n }\n }\n return true;\n }\n}\n"], +- "mappings": ";;;;;;;;;;;;;;;;AAYO,SAASA,EAAaC,EAA6B,CACxD,MAAO,CAAE,QAASA,CAAG,CACvB,CAKO,SAASC,EAA+BC,EAA+C,CAC5F,GAAI,CAACA,EACH,OAAOA,EAET,GAAI,MAAM,QAAQA,CAAG,EAAG,CACtB,QAAWC,KAAKD,EACdC,EAAE,QAAQ,EAEZ,MAAO,CAAC,CACV,CACA,OAAAD,EAAI,QAAQ,EACLA,CACT,CAEO,SAASE,KAAsBC,EAAyC,CAC7E,OAAON,EAAa,IAAME,EAAQI,CAAW,CAAC,CAChD,CAEO,IAAMC,EAAN,KAA6C,CAA7C,cACL,KAAiB,aAAe,IAAI,IACpC,KAAQ,YAAc,GAEtB,IAAW,YAAsB,CAC/B,OAAO,KAAK,WACd,CAEO,IAA2BC,EAAS,CACzC,OAAI,KAAK,YACPA,EAAE,QAAQ,EAEV,KAAK,aAAa,IAAIA,CAAC,EAElBA,CACT,CAEO,SAAgB,CACrB,GAAI,MAAK,YAGT,MAAK,YAAc,GACnB,QAAWJ,KAAK,KAAK,aACnBA,EAAE,QAAQ,EAEZ,KAAK,aAAa,MAAM,EAC1B,CAEO,OAAc,CACnB,QAAWA,KAAK,KAAK,aACnBA,EAAE,QAAQ,EAEZ,KAAK,aAAa,MAAM,CAC1B,CACF,EAEsBK,EAAf,KAAiD,CAAjD,cAGL,KAAmB,OAAS,IAAIF,EAEzB,SAAgB,CACrB,KAAK,OAAO,QAAQ,CACtB,CAEU,UAAiCC,EAAS,CAClD,OAAO,KAAK,OAAO,IAAIA,CAAC,CAC1B,CACF,EAZsBC,EACG,KAAoB,OAAO,OAAO,CAAE,SAAU,CAAE,CAAE,CAAC,EAarE,IAAMC,EAAN,KAAsE,CAAtE,cAEL,KAAQ,YAAc,GAEtB,IAAW,OAAuB,CAChC,OAAO,KAAK,YAAc,OAAY,KAAK,MAC7C,CAEA,IAAW,MAAMC,EAAsB,CACjC,KAAK,aAAeA,IAAU,KAAK,SAGvC,KAAK,QAAQ,QAAQ,EACrB,KAAK,OAASA,EAChB,CAEO,OAAc,CACnB,KAAK,MAAQ,MACf,CAEO,SAAgB,CACrB,KAAK,YAAc,GACnB,KAAK,QAAQ,QAAQ,EACrB,KAAK,OAAS,MAChB,CACF,EClGO,IAAMC,EAAN,KAAiB,CAAjB,cACL,KAAQ,WAAqD,CAAC,EAC9D,KAAQ,UAAY,GAGpB,IAAW,OAAmB,CAC5B,OAAI,KAAK,OACA,KAAK,QAEd,KAAK,OAAS,CAACC,EAAyBC,EAAgBC,IAAkD,CACxG,GAAI,KAAK,UACP,OAAOC,EAAa,IAAM,CAAC,CAAC,EAG9B,IAAMC,EAAQ,CAAE,GAAIJ,EAAU,SAAAC,CAAS,EACvC,KAAK,WAAa,KAAK,WAAW,MAAM,EACxC,KAAK,WAAW,KAAKG,CAAK,EAE1B,IAAMC,EAASF,EAAa,IAAM,CAChC,IAAMG,EAAM,KAAK,WAAW,QAAQF,CAAK,EACrCE,IAAQ,KACV,KAAK,WAAa,KAAK,WAAW,MAAM,EACxC,KAAK,WAAW,OAAOA,EAAK,CAAC,EAEjC,CAAC,EAED,OAAIJ,IACE,MAAM,QAAQA,CAAW,EAC3BA,EAAY,KAAKG,CAAM,EAEvBH,EAAY,IAAIG,CAAM,GAInBA,CACT,EACO,KAAK,OACd,CAEO,KAAKE,EAAgB,CAC1B,GAAI,KAAK,WAAa,CAAC,KAAK,WAAW,OACrC,OAEF,GAAI,KAAK,WAAW,SAAW,EAAG,CAChC,KAAK,WAAW,CAAC,EAAE,GAAG,KAAK,KAAK,WAAW,CAAC,EAAE,SAAUA,CAAK,EAC7D,MACF,CACA,IAAMC,EAAY,KAAK,WACvB,QAASC,EAAI,EAAGC,EAAMF,EAAU,OAAQC,EAAIC,EAAK,EAAED,EACjDD,EAAUC,CAAC,EAAE,GAAG,KAAKD,EAAUC,CAAC,EAAE,SAAUF,CAAK,CAErD,CAEO,SAAgB,CACjB,KAAK,YAGT,KAAK,UAAY,GACjB,KAAK,WAAW,OAAS,EAC3B,CACF,EAEiBI,MAAV,CACE,SAASC,EAAWC,EAAiBC,EAA6B,CACvE,OAAOD,EAAKE,GAAKD,EAAG,KAAKC,CAAC,CAAC,CAC7B,CAFOJ,EAAS,QAAAC,EAIT,SAASI,EAAUT,EAAkBS,EAA6B,CACvE,MAAO,CAAChB,EAAyBC,EAAgBC,IACxCK,EAAME,GAAKT,EAAS,KAAKC,EAAUe,EAAIP,CAAC,CAAC,EAAG,OAAWP,CAAW,CAE7E,CAJOS,EAAS,IAAAK,EAQT,SAASC,KAAUC,EAAgC,CACxD,MAAO,CAAClB,EAAyBC,EAAgBC,IAAkD,CACjG,IAAMiB,EAAQ,IAAIC,EAClB,QAAWb,KAASW,EAClBC,EAAM,IAAIZ,EAAMQ,GAAKf,EAAS,KAAKC,EAAUc,CAAC,CAAC,CAAC,EAElD,OAAIb,IACE,MAAM,QAAQA,CAAW,EAC3BA,EAAY,KAAKiB,CAAK,EAEtBjB,EAAY,IAAIiB,CAAK,GAGlBA,CACT,CACF,CAfOR,EAAS,IAAAM,EAmBT,SAASI,EAAmBd,EAAkBe,EAAqCC,EAA0B,CAClH,OAAAD,EAAQC,CAAO,EACRhB,EAAMQ,GAAKO,EAAQP,CAAC,CAAC,CAC9B,CAHOJ,EAAS,gBAAAU,IAhCDV,IAAA,ICxDV,SAASa,EAAkBC,EAAqBC,EAAU,EAAGC,EAAsC,CACxG,IAAMC,EAAQ,WAAW,IAAM,CAC7BH,EAAQ,EACJE,GACFE,EAAW,QAAQ,CAEvB,EAAGH,CAAO,EACJG,EAAaC,EAAa,IAAM,CACpC,aAAaF,CAAK,CACpB,CAAC,EACD,OAAAD,GAAO,IAAIE,CAAU,EACdA,CACT,CCDO,IAAME,EAAN,cAA8BC,CAAW,CAa9C,YAA6BC,EAAqB,CAChD,MAAM,EADqB,eAAAA,EAN7B,KAAQ,mBAAqB,KAAK,UAAU,IAAIC,CAAmB,EACnE,KAAQ,uBAAyB,KAAK,UAAU,IAAIA,CAAmB,EAGvE,KAAQ,qBAAuB,EAI7B,KAAK,UAAUC,EAAa,IAAM,KAAK,mBAAmB,CAAC,CAAC,CAC9D,CAKO,gBAAuB,CACvB,KAAK,cACR,KAAK,YAAc,IAAI,MAAM,KAAK,UAAU,OAAO,OAAO,MAAM,EAChE,KAAK,uBAAuB,MAAQC,EAClC,KAAK,UAAU,WAAW,IAAM,KAAK,mBAAmB,CAAC,EACzD,KAAK,UAAU,aAAa,IAAM,KAAK,mBAAmB,CAAC,EAC3D,KAAK,UAAU,SAAS,IAAM,KAAK,mBAAmB,CAAC,CACzD,GAGF,KAAK,qBAAuB,KAAK,IAAI,EAChC,KAAK,mBAAmB,OAC3B,KAAK,2BAA2B,IAAkC,CAEtE,CAEQ,oBAA2B,CACjC,KAAK,YAAc,OACnB,KAAK,qBAAuB,EAC5B,KAAK,uBAAuB,MAAM,EAClC,KAAK,mBAAmB,MAAM,CAChC,CAEQ,2BAA2BC,EAAqB,CACtD,KAAK,mBAAmB,MAAQC,EAAkB,IAAM,CACtD,GAAI,CAAC,KAAK,YACR,OAGF,IAAMC,EADM,KAAK,IAAI,EACC,KAAK,qBAC3B,GAAIA,GAAW,KAAoC,CACjD,KAAK,mBAAmB,EACxB,MACF,CACA,KAAK,2BAA2B,KAAqCA,CAAO,CAC9E,EAAGF,CAAK,CACV,CAEO,iBAAiBG,EAAyC,CAC/D,OAAO,KAAK,cAAcA,CAAG,CAC/B,CAEO,eAAeA,EAAaC,EAA6B,CAC1D,KAAK,cACP,KAAK,YAAYD,CAAG,EAAIC,EAE5B,CAUO,oCAAoCC,EAAmBC,EAAoC,CAChG,IAAMC,EAAU,CAAC,EACXC,EAAc,CAAC,CAAC,EAClBC,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQJ,CAAS,EACzD,KAAOI,GAAM,CACX,IAAMC,EAAW,KAAK,UAAU,OAAO,OAAO,QAAQL,EAAY,CAAC,EAC7DM,EAAkBD,EAAWA,EAAS,UAAY,GACpDE,EAASH,EAAK,kBAAkB,CAACE,GAAmBL,CAAS,EACjE,GAAIK,GAAmBD,EAAU,CAC/B,IAAMG,EAAWJ,EAAK,QAAQA,EAAK,OAAS,CAAC,EACtBI,GAAYA,EAAS,QAAQ,IAAM,GAAKA,EAAS,SAAS,IAAM,GAEjEH,EAAS,QAAQ,CAAC,GAAG,SAAS,IAAM,IACxDE,EAASA,EAAO,MAAM,EAAG,EAAE,EAE/B,CAEA,GADAL,EAAQ,KAAKK,CAAM,EACfD,EACFH,EAAY,KAAKA,EAAYA,EAAY,OAAS,CAAC,EAAII,EAAO,MAAM,MAEpE,OAEFP,IACAI,EAAOC,CACT,CACA,MAAO,CAACH,EAAQ,KAAK,EAAE,EAAGC,CAAW,CACvC,CACF,EC5HO,IAAMM,EAAN,KAAkB,CAOvB,IAAW,kBAAuC,CAChD,OAAO,KAAK,iBACd,CAKA,IAAW,iBAAiBC,EAA0B,CACpD,KAAK,kBAAoBA,CAC3B,CAKA,IAAW,mBAAgD,CACzD,OAAO,KAAK,kBACd,CAKA,IAAW,kBAAkBC,EAAqC,CAChE,KAAK,mBAAqBA,CAC5B,CAOO,kBAAkBD,EAAuB,CAC9C,MAAO,CAAC,EAAEA,GAAQA,EAAK,OAAS,EAClC,CAOO,iBAAiBE,EAAsC,CAC5D,OAAK,KAAK,mBAGLA,EAGD,KAAK,mBAAmB,gBAAkBA,EAAW,eAGrD,KAAK,mBAAmB,QAAUA,EAAW,OAG7C,KAAK,mBAAmB,YAAcA,EAAW,UAR5C,GAHA,EAeX,CAQO,yBAAyBF,EAAcC,EAAmC,CAC/E,OAAKA,GAAS,YAGP,KAAK,oBAAsB,QAC3BD,IAAS,KAAK,mBACd,KAAK,iBAAiBC,CAAO,EAJ3B,EAKX,CAKO,iBAAwB,CAC7B,KAAK,kBAAoB,MAC3B,CAKO,OAAc,CACnB,KAAK,kBAAoB,OACzB,KAAK,mBAAqB,MAC5B,CACF,EC9DO,IAAME,EAAN,KAAmB,CACxB,YACmBC,EACAC,EACjB,CAFiB,eAAAD,EACA,gBAAAC,CAChB,CAUI,KAAKC,EAAcC,EAAkBC,EAAkBC,EAA2D,CACvH,GAAI,CAACH,GAAQA,EAAK,SAAW,EAAG,CAC9B,KAAK,UAAU,eAAe,EAC9B,MACF,CACA,GAAIE,GAAY,KAAK,UAAU,KAC7B,MAAM,IAAI,MAAM,gBAAgBA,CAAQ,6BAA6B,KAAK,UAAU,IAAI,OAAO,EAGjG,KAAK,WAAW,eAAe,EAE/B,IAAME,EAAkC,CACtC,SAAAH,EACA,SAAAC,CACF,EAGIG,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EAEjE,GAAI,CAACE,EACH,QAASC,EAAIL,EAAW,EAAGK,EAAI,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,OACjFF,EAAe,SAAWE,EAC1BF,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EACzD,CAAAE,GAJmFC,IAIvF,CAKJ,OAAOD,CACT,CASO,sBAAsBL,EAAcG,EAAgCI,EAAsD,CAC/H,GAAI,CAACP,GAAQA,EAAK,SAAW,EAAG,CAC9B,KAAK,UAAU,eAAe,EAC9B,MACF,CAEA,IAAMQ,EAAkB,KAAK,UAAU,qBAAqB,EAC5D,KAAK,UAAU,eAAe,EAE9B,IAAIN,EAAW,EACXD,EAAW,EACXO,IACED,IAAqBP,GACvBE,EAAWM,EAAgB,IAAI,EAC/BP,EAAWO,EAAgB,IAAI,IAE/BN,EAAWM,EAAgB,MAAM,EACjCP,EAAWO,EAAgB,MAAM,IAIrC,KAAK,WAAW,eAAe,EAE/B,IAAMJ,EAAkC,CACtC,SAAAH,EACA,SAAAC,CACF,EAGIG,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EAEjE,GAAI,CAACE,EACH,QAASC,EAAIL,EAAW,EAAGK,EAAI,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,OACjFF,EAAe,SAAWE,EAC1BF,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EACzD,CAAAE,GAJmFC,IAIvF,CAMJ,GAAI,CAACD,GAAUJ,IAAa,EAC1B,QAASK,EAAI,EAAGA,EAAIL,IAClBG,EAAe,SAAWE,EAC1BF,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EACzD,CAAAE,GAJwBC,IAI5B,CAOJ,MAAI,CAACD,GAAUG,IACbJ,EAAe,SAAWI,EAAgB,MAAM,EAChDJ,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,GAGxDE,CACT,CASO,0BAA0BL,EAAcG,EAAgCI,EAAsD,CACnI,GAAI,CAACP,GAAQA,EAAK,SAAW,EAAG,CAC9B,KAAK,UAAU,eAAe,EAC9B,MACF,CAEA,IAAMQ,EAAkB,KAAK,UAAU,qBAAqB,EAC5D,KAAK,UAAU,eAAe,EAE9B,IAAIP,EAAW,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,KAAO,EACpEC,EAAW,KAAK,UAAU,KAC1BO,EAAkB,GAExB,KAAK,WAAW,eAAe,EAC/B,IAAML,EAAkC,CACtC,SAAAH,EACA,SAAAC,CACF,EAEIG,EAkBJ,GAjBIG,IACFJ,EAAe,SAAWH,EAAWO,EAAgB,MAAM,EAC3DJ,EAAe,SAAWI,EAAgB,MAAM,EAC5CD,IAAqBP,IAEvBK,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,EAAe,EAAK,EAC/DE,IAEHD,EAAe,SAAWH,EAAWO,EAAgB,IAAI,EACzDJ,EAAe,SAAWI,EAAgB,IAAI,KAKpDH,IAAW,KAAK,YAAYL,EAAMI,EAAgBD,EAAeM,CAAe,EAG5E,CAACJ,EAAQ,CACXD,EAAe,SAAW,KAAK,IAAIA,EAAe,SAAU,KAAK,UAAU,IAAI,EAC/E,QAASE,EAAIL,EAAW,EAAGK,GAAK,IAC9BF,EAAe,SAAWE,EAC1BD,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,EAAeM,CAAe,EAC1E,CAAAJ,GAH6BC,IAGjC,CAIJ,CAEA,GAAI,CAACD,GAAUJ,IAAc,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,KAAO,EACtF,QAASK,EAAK,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,KAAO,EAAIA,GAAKL,IAChFG,EAAe,SAAWE,EAC1BD,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,EAAeM,CAAe,EAC1E,CAAAJ,GAHsFC,IAG1F,CAMJ,OAAOD,CACT,CASQ,aAAaK,EAAqBC,EAAcX,EAAuB,CAC7E,OAASU,IAAgB,GAAO,qCAA8B,SAASC,EAAKD,EAAc,CAAC,CAAC,KACvFA,EAAcV,EAAK,SAAYW,EAAK,QAAY,qCAA8B,SAASA,EAAKD,EAAcV,EAAK,MAAM,CAAC,EAC7H,CAcQ,YAAYA,EAAcI,EAAiCD,EAAgC,CAAC,EAAGM,EAA2B,GAAkC,CAClK,IAAMG,EAAMR,EAAe,SACrBS,EAAMT,EAAe,SAI3B,GADkB,KAAK,UAAU,OAAO,OAAO,QAAQQ,CAAG,GAC3C,UAAW,CACxB,GAAIH,EAAiB,CACnBL,EAAe,UAAY,KAAK,UAAU,KAC1C,MACF,CAIA,OAAAA,EAAe,WACfA,EAAe,UAAY,KAAK,UAAU,KACnC,KAAK,YAAYJ,EAAMI,EAAgBD,CAAa,CAC7D,CACA,IAAIW,EAAQ,KAAK,WAAW,iBAAiBF,CAAG,EAC3CE,IACHA,EAAQ,KAAK,WAAW,oCAAoCF,EAAK,EAAI,EACrE,KAAK,WAAW,eAAeA,EAAKE,CAAK,GAE3C,GAAM,CAACC,EAAYC,CAAO,EAAIF,EAExBG,EAAS,KAAK,0BAA0BL,EAAKC,CAAG,EAClDK,EAAalB,EACbmB,EAAmBJ,EAClBZ,EAAc,QACjBe,EAAaf,EAAc,cAAgBH,EAAOA,EAAK,YAAY,EACnEmB,EAAmBhB,EAAc,cAAgBY,EAAaA,EAAW,YAAY,GAGvF,IAAIK,EAAc,GAClB,GAAIjB,EAAc,MAAO,CACvB,IAAMkB,EAAc,OAAOH,EAAYf,EAAc,cAAgB,IAAM,IAAI,EAC3EmB,EACJ,GAAIb,EAEF,KAAOa,EAAYD,EAAY,KAAKF,EAAiB,MAAM,EAAGF,CAAM,CAAC,GACnEG,EAAcC,EAAY,UAAYC,EAAU,CAAC,EAAE,OACnDtB,EAAOsB,EAAU,CAAC,EAClBD,EAAY,WAAcrB,EAAK,OAAS,OAG1CsB,EAAYD,EAAY,KAAKF,EAAiB,MAAMF,CAAM,CAAC,EACvDK,GAAaA,EAAU,CAAC,EAAE,OAAS,IACrCF,EAAcH,GAAUI,EAAY,UAAYC,EAAU,CAAC,EAAE,QAC7DtB,EAAOsB,EAAU,CAAC,EAGxB,MACMb,EACEQ,EAASC,EAAW,QAAU,IAChCE,EAAcD,EAAiB,YAAYD,EAAYD,EAASC,EAAW,MAAM,GAGnFE,EAAcD,EAAiB,QAAQD,EAAYD,CAAM,EAI7D,GAAIG,GAAe,EAAG,CACpB,GAAIjB,EAAc,WAAa,CAAC,KAAK,aAAaiB,EAAaD,EAAkBnB,CAAI,EACnF,OAKF,IAAIuB,EAAiB,EACrB,KAAOA,EAAiBP,EAAQ,OAAS,GAAKI,GAAeJ,EAAQO,EAAiB,CAAC,GACrFA,IAEF,IAAIC,EAAeD,EACnB,KAAOC,EAAeR,EAAQ,OAAS,GAAKI,EAAcpB,EAAK,QAAUgB,EAAQQ,EAAe,CAAC,GAC/FA,IAEF,IAAMC,EAAiBL,EAAcJ,EAAQO,CAAc,EACrDG,EAAeN,EAAcpB,EAAK,OAASgB,EAAQQ,CAAY,EAC/DG,EAAgB,KAAK,0BAA0Bf,EAAMW,EAAgBE,CAAc,EAEnFG,EADc,KAAK,0BAA0BhB,EAAMY,EAAcE,CAAY,EACxDC,EAAgB,KAAK,UAAU,MAAQH,EAAeD,GAEjF,MAAO,CACL,KAAAvB,EACA,IAAK2B,EACL,IAAKf,EAAMW,EACX,KAAAK,CACF,CACF,CACF,CAEQ,0BAA0BhB,EAAaK,EAAwB,CACrE,IAAMN,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQC,CAAG,EACrD,GAAI,CAACD,EACH,MAAO,GAET,QAASkB,EAAI,EAAGA,EAAIZ,EAAQY,IAAK,CAC/B,IAAMC,EAAOnB,EAAK,QAAQkB,CAAC,EAC3B,GAAI,CAACC,EACH,MAGF,IAAMC,EAAOD,EAAK,SAAS,EACvBC,EAAK,OAAS,IAChBd,GAAUc,EAAK,OAAS,GAI1B,IAAMC,EAAWrB,EAAK,QAAQkB,EAAI,CAAC,EAC/BG,GAAYA,EAAS,SAAS,IAAM,GACtCf,GAEJ,CACA,OAAOA,CACT,CAEQ,0BAA0BhB,EAAkBgC,EAAsB,CACxE,IAAIC,EAAYjC,EACZgB,EAAS,EACTN,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQuB,CAAS,EACzD,KAAOD,EAAO,GAAKtB,GAAM,CACvB,QAASkB,EAAI,EAAGA,EAAII,GAAQJ,EAAI,KAAK,UAAU,KAAMA,IAAK,CACxD,IAAMC,EAAOnB,EAAK,QAAQkB,CAAC,EAC3B,GAAI,CAACC,EACH,MAEEA,EAAK,SAAS,IAEhBb,GAAUa,EAAK,QAAQ,IAAM,EAAI,EAAIA,EAAK,SAAS,EAAE,OAEzD,CAGA,GAFAI,IACAvB,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQuB,CAAS,EACjDvB,GAAQ,CAACA,EAAK,UAChB,MAEFsB,GAAQ,KAAK,UAAU,IACzB,CACA,OAAOhB,CACT,CACF,ECzWO,IAAMkB,EAAN,cAAgCC,CAAW,CAIhD,YAA6BC,EAAqB,CAChD,MAAM,EADqB,eAAAA,EAH7B,KAAQ,sBAAsC,CAAC,EAC/C,KAAQ,kBAAiC,IAAI,IAI3C,KAAK,UAAUC,EAAa,IAAM,KAAK,0BAA0B,CAAC,CAAC,CACrE,CAOO,2BAA2BC,EAA0BC,EAAyC,CACnG,KAAK,0BAA0B,EAE/B,QAAWC,KAASF,EAAS,CAC3B,IAAMG,EAAc,KAAK,yBAAyBD,EAAOD,EAAS,EAAK,EACvE,GAAIE,EACF,QAAWC,KAAcD,EACvB,KAAK,iBAAiBC,EAAYF,CAAK,CAG7C,CACF,CAQO,uBAAuBG,EAAuBJ,EAAgE,CACnH,IAAME,EAAc,KAAK,yBAAyBE,EAAQJ,EAAS,EAAI,EACvE,GAAIE,EACF,MAAO,CAAE,YAAAA,EAAa,MAAOE,EAAQ,SAAU,CAAEC,EAAQH,CAAW,CAAG,CAAE,CAG7E,CAKO,2BAAkC,CACvCG,EAAQ,KAAK,qBAAqB,EAClC,KAAK,sBAAwB,CAAC,EAC9B,KAAK,kBAAkB,MAAM,CAC/B,CAOQ,iBAAiBF,EAAyBF,EAA4B,CAC5E,KAAK,kBAAkB,IAAIE,EAAW,OAAO,IAAI,EACjD,KAAK,sBAAsB,KAAK,CAAE,WAAAA,EAAY,MAAAF,EAAO,SAAU,CAAEE,EAAW,QAAQ,CAAG,CAAE,CAAC,CAC5F,CAQQ,aAAaG,EAAsBC,EAAiCC,EAA+B,CACpGF,EAAQ,UAAU,SAAS,8BAA8B,IAC5DA,EAAQ,UAAU,IAAI,8BAA8B,EAChDC,IACFD,EAAQ,MAAM,QAAU,aAAaC,CAAW,KAGhDC,GACFF,EAAQ,UAAU,IAAI,qCAAqC,CAE/D,CASQ,yBAAyBF,EAAuBJ,EAAmCQ,EAAoD,CAE7I,IAAMC,EAA+C,CAAC,EAClDC,EAAaN,EAAO,IACpBO,EAAgBP,EAAO,KACvBQ,EAAe,CAAC,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,OAAO,OAAO,QAAUR,EAAO,IACvG,KAAOO,EAAgB,GAAG,CACxB,IAAME,EAAgB,KAAK,IAAI,KAAK,UAAU,KAAOH,EAAYC,CAAa,EAC9EF,EAAiB,KAAK,CAACG,EAAcF,EAAYG,CAAa,CAAC,EAC/DH,EAAa,EACbC,GAAiBE,EACjBD,GACF,CAGA,IAAMV,EAA6B,CAAC,EACpC,QAAWY,KAASL,EAAkB,CACpC,IAAMM,EAAS,KAAK,UAAU,eAAeD,EAAM,CAAC,CAAC,EAC/CX,EAAa,KAAK,UAAU,mBAAmB,CACnD,OAAAY,EACA,EAAGD,EAAM,CAAC,EACV,MAAOA,EAAM,CAAC,EACd,MAAON,EAAiB,MAAQ,SAChC,gBAAiBA,EAAiBR,EAAQ,sBAAwBA,EAAQ,gBAC1E,qBAAsB,KAAK,kBAAkB,IAAIe,EAAO,IAAI,EAAI,OAAY,CAC1E,MAAOP,EAAiBR,EAAQ,8BAAgCA,EAAQ,mBACxE,SAAU,QACZ,CACF,CAAC,EACD,GAAIG,EAAY,CACd,IAAMa,EAA6B,CAAC,EACpCA,EAAY,KAAKD,CAAM,EACvBC,EAAY,KAAKb,EAAW,SAAUc,GAAM,KAAK,aAAaA,EAAGT,EAAiBR,EAAQ,kBAAoBA,EAAQ,YAAa,EAAK,CAAC,CAAC,EAC1IgB,EAAY,KAAKb,EAAW,UAAU,IAAME,EAAQW,CAAW,CAAC,CAAC,EACjEd,EAAY,KAAKC,CAAU,CAC7B,CACF,CAEA,OAAOD,EAAY,SAAW,EAAI,OAAYA,CAChD,CACF,ECrIO,IAAMgB,EAAN,cAAkCC,CAAW,CAA7C,kCACL,KAAQ,eAAkC,CAAC,EAG3C,KAAiB,oBAAsB,KAAK,UAAU,IAAIC,CAAmC,EAC7F,IAAW,oBAAuD,CAAE,OAAO,KAAK,oBAAoB,KAAO,CAK3G,IAAW,eAA8C,CACvD,OAAO,KAAK,cACd,CAKA,IAAW,oBAAsD,CAC/D,OAAO,KAAK,mBACd,CAKA,IAAW,mBAAmBC,EAA6C,CACzE,KAAK,oBAAsBA,CAC7B,CAOO,cAAcC,EAA0BC,EAA0B,CACvE,KAAK,eAAiBD,EAAQ,MAAM,EAAGC,CAAU,CACnD,CAKO,cAAqB,CAC1B,KAAK,eAAiB,CAAC,CACzB,CAKO,yBAAgC,CACjC,KAAK,sBACP,KAAK,oBAAoB,QAAQ,EACjC,KAAK,oBAAsB,OAE/B,CAOO,gBAAgBC,EAA+B,CACpD,QAASC,EAAI,EAAGA,EAAI,KAAK,eAAe,OAAQA,IAAK,CACnD,IAAMC,EAAQ,KAAK,eAAeD,CAAC,EACnC,GAAIC,EAAM,MAAQF,EAAO,KAAOE,EAAM,MAAQF,EAAO,KAAOE,EAAM,OAASF,EAAO,KAChF,OAAOC,CAEX,CACA,MAAO,EACT,CAMO,mBAAmBE,EAA+B,CACvD,GAAI,CAACA,EACH,OAGF,IAAIC,EAAc,GACd,KAAK,sBACPA,EAAc,KAAK,gBAAgB,KAAK,oBAAoB,KAAK,GAGnE,KAAK,oBAAoB,KAAK,CAC5B,YAAAA,EACA,YAAa,KAAK,eAAe,MACnC,CAAC,CACH,CAKO,OAAc,CACnB,KAAK,wBAAwB,EAC7B,KAAK,aAAa,CACpB,CACF,ECtFO,IAAMC,EAAN,cAA0BC,CAAiD,CAqBhF,YAAYC,EAAwC,CAClD,MAAM,EAnBR,KAAQ,kBAAoB,KAAK,UAAU,IAAIC,CAAgC,EAC/E,KAAQ,WAAa,KAAK,UAAU,IAAIA,CAAoC,EAG5E,KAAQ,OAAS,IAAIC,EAGrB,KAAQ,eAAiB,KAAK,UAAU,IAAIC,CAAqB,EAEjE,KAAiB,eAAiB,KAAK,UAAU,IAAIC,CAAe,EACpE,KAAgB,cAAgB,KAAK,eAAe,MACpD,KAAiB,gBAAkB,KAAK,UAAU,IAAIA,CAAe,EACrE,KAAgB,eAAiB,KAAK,gBAAgB,MASpD,KAAK,gBAAkBJ,GAAS,gBAAkB,GACpD,CARA,IAAW,oBAAuD,CAChE,OAAO,KAAK,eAAe,kBAC7B,CAQO,SAASK,EAA0B,CACxC,KAAK,UAAYA,EACjB,KAAK,WAAW,MAAQ,IAAIC,EAAgBD,CAAQ,EACpD,KAAK,QAAU,IAAIE,EAAaF,EAAU,KAAK,WAAW,KAAK,EAC/D,KAAK,mBAAqB,IAAIG,EAAkBH,CAAQ,EACxD,KAAK,UAAU,KAAK,UAAU,cAAc,IAAM,KAAK,eAAe,CAAC,CAAC,EACxE,KAAK,UAAU,KAAK,UAAU,SAAS,IAAM,KAAK,eAAe,CAAC,CAAC,EACnE,KAAK,UAAUI,EAAa,IAAM,KAAK,iBAAiB,CAAC,CAAC,CAC5D,CAEQ,gBAAuB,CAC7B,KAAK,kBAAkB,MAAM,EACzB,KAAK,OAAO,kBAAoB,KAAK,OAAO,mBAAmB,cACjE,KAAK,kBAAkB,MAAQC,EAAkB,IAAM,CACrD,IAAMC,EAAO,KAAK,OAAO,iBACzB,KAAK,OAAO,gBAAgB,EAC5B,KAAK,aAAaA,EAAO,CAAE,GAAG,KAAK,OAAO,kBAAmB,YAAa,EAAK,EAAG,CAAE,SAAU,EAAK,CAAC,CACtG,EAAG,GAAG,EAEV,CAEO,iBAAiBC,EAAwC,CAC9D,KAAK,eAAe,wBAAwB,EAC5C,KAAK,oBAAoB,0BAA0B,EACnD,KAAK,eAAe,aAAa,EAC5BA,GACH,KAAK,OAAO,gBAAgB,CAEhC,CAEO,uBAA8B,CACnC,KAAK,eAAe,wBAAwB,CAC9C,CASO,SAASD,EAAcE,EAAgCC,EAAyD,CACrH,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAM,IAAI,MAAM,2CAA2C,EAG7D,KAAK,gBAAgB,KAAK,EAE1B,KAAK,OAAO,kBAAoBD,EAE5B,KAAK,OAAO,yBAAyBF,EAAME,CAAa,GAC1D,KAAK,qBAAqBF,EAAME,CAAc,EAGhD,IAAME,EAAQ,KAAK,mBAAmBJ,EAAME,EAAeC,CAAqB,EAChF,YAAK,aAAaD,CAAa,EAC/B,KAAK,OAAO,iBAAmBF,EAE/B,KAAK,eAAe,KAAK,EAElBI,CACT,CAEQ,qBAAqBJ,EAAcE,EAAqC,CAC9E,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,SAAW,CAAC,KAAK,mBAC5C,MAAM,IAAI,MAAM,2CAA2C,EAE7D,GAAI,CAAC,KAAK,OAAO,kBAAkBF,CAAI,EAAG,CACxC,KAAK,iBAAiB,EACtB,MACF,CAGA,KAAK,iBAAiB,EAAI,EAE1B,IAAMK,EAA2B,CAAC,EAC9BC,EACAC,EAAS,KAAK,QAAQ,KAAKP,EAAM,EAAG,EAAGE,CAAa,EAExD,KAAOK,IAAWD,GAAY,MAAQC,EAAO,KAAOD,GAAY,MAAQC,EAAO,MACzE,EAAAF,EAAQ,QAAU,KAAK,kBADwD,CAInFC,EAAaC,EACbF,EAAQ,KAAKC,CAAU,EACvB,IAAME,EAAO,KAAK,UAAU,KACxBC,EAAUH,EAAW,IAAMA,EAAW,KACtCI,EAAUJ,EAAW,IACrBG,GAAWD,IACbE,GAAW,KAAK,MAAMD,EAAUD,CAAI,EACpCC,EAAUA,EAAUD,GAEtBD,EAAS,KAAK,QAAQ,KAAKP,EAAMU,EAASD,EAASP,CAAa,CAClE,CAEA,KAAK,eAAe,cAAcG,EAAS,KAAK,eAAe,EAC3DH,EAAc,aAChB,KAAK,mBAAmB,2BAA2BG,EAASH,EAAc,WAAW,CAEzF,CAEQ,mBAAmBF,EAAcE,EAAgCC,EAAyD,CAChI,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAO,GAET,GAAI,CAAC,KAAK,OAAO,kBAAkBH,CAAI,EACrC,YAAK,UAAU,eAAe,EAC9B,KAAK,iBAAiB,EACf,GAGT,IAAMO,EAAS,KAAK,QAAQ,sBAAsBP,EAAME,EAAe,KAAK,OAAO,gBAAgB,EACnG,OAAO,KAAK,cAAcK,EAAQL,GAAe,YAAaC,GAAuB,QAAQ,CAC/F,CASO,aAAaH,EAAcE,EAAgCC,EAAyD,CACzH,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAM,IAAI,MAAM,2CAA2C,EAG7D,KAAK,gBAAgB,KAAK,EAE1B,KAAK,OAAO,kBAAoBD,EAE5B,KAAK,OAAO,yBAAyBF,EAAME,CAAa,GAC1D,KAAK,qBAAqBF,EAAME,CAAc,EAGhD,IAAME,EAAQ,KAAK,uBAAuBJ,EAAME,EAAeC,CAAqB,EACpF,YAAK,aAAaD,CAAa,EAC/B,KAAK,OAAO,iBAAmBF,EAE/B,KAAK,eAAe,KAAK,EAElBI,CACT,CAEQ,aAAaF,EAAsC,CACzD,KAAK,eAAe,mBAAmB,CAAC,CAACA,GAAe,WAAW,CACrE,CAEQ,uBAAuBF,EAAcE,EAAgCC,EAAyD,CACpI,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAO,GAET,GAAI,CAAC,KAAK,OAAO,kBAAkBH,CAAI,EACrC,YAAK,UAAU,eAAe,EAC9B,KAAK,iBAAiB,EACf,GAGT,IAAMO,EAAS,KAAK,QAAQ,0BAA0BP,EAAME,EAAe,KAAK,OAAO,gBAAgB,EACvG,OAAO,KAAK,cAAcK,EAAQL,GAAe,YAAaC,GAAuB,QAAQ,CAC/F,CAOQ,cAAcI,EAAmClB,EAAoCsB,EAA6B,CACxH,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,mBAC3B,MAAO,GAIT,GADA,KAAK,eAAe,wBAAwB,EACxC,CAACJ,EACH,YAAK,UAAU,eAAe,EACvB,GAIT,GADA,KAAK,UAAU,OAAOA,EAAO,IAAKA,EAAO,IAAKA,EAAO,IAAI,EACrDlB,EAAS,CACX,IAAMuB,EAAmB,KAAK,mBAAmB,uBAAuBL,EAAQlB,CAAO,EACnFuB,IACF,KAAK,eAAe,mBAAqBA,EAE7C,CAEA,GAAI,CAACD,IAECJ,EAAO,KAAQ,KAAK,UAAU,OAAO,OAAO,UAAY,KAAK,UAAU,MAASA,EAAO,IAAM,KAAK,UAAU,OAAO,OAAO,WAAW,CACvI,IAAIM,EAASN,EAAO,IAAM,KAAK,UAAU,OAAO,OAAO,UACvDM,GAAU,KAAK,MAAM,KAAK,UAAU,KAAO,CAAC,EAC5C,KAAK,UAAU,YAAYA,CAAM,CACnC,CAEF,MAAO,EACT,CACF", +- "names": ["toDisposable", "fn", "dispose", "arg", "d", "combinedDisposable", "disposables", "DisposableStore", "o", "Disposable", "MutableDisposable", "value", "Emitter", "listener", "thisArgs", "disposables", "toDisposable", "entry", "result", "idx", "event", "listeners", "i", "len", "EventUtils", "forward", "from", "to", "e", "map", "any", "events", "store", "DisposableStore", "runAndSubscribe", "handler", "initial", "disposableTimeout", "handler", "timeout", "store", "timer", "disposable", "toDisposable", "SearchLineCache", "Disposable", "_terminal", "MutableDisposable", "toDisposable", "combinedDisposable", "delay", "disposableTimeout", "elapsed", "row", "entry", "lineIndex", "trimRight", "strings", "lineOffsets", "line", "nextLine", "lineWrapsToNext", "string", "lastCell", "SearchState", "term", "options", "newOptions", "SearchEngine", "_terminal", "_lineCache", "term", "startRow", "startCol", "searchOptions", "searchPosition", "result", "y", "cachedSearchTerm", "prevSelectedPos", "isReverseSearch", "searchIndex", "line", "row", "col", "cache", "stringLine", "offsets", "offset", "searchTerm", "searchStringLine", "resultIndex", "searchRegex", "foundTerm", "startRowOffset", "endRowOffset", "startColOffset", "endColOffset", "startColIndex", "size", "i", "cell", "char", "nextCell", "cols", "lineIndex", "DecorationManager", "Disposable", "_terminal", "toDisposable", "results", "options", "match", "decorations", "decoration", "result", "dispose", "element", "borderColor", "isActiveResult", "decorationRanges", "currentCol", "remainingSize", "markerOffset", "amountThisRow", "range", "marker", "disposables", "e", "SearchResultTracker", "Disposable", "Emitter", "decoration", "results", "maxResults", "result", "i", "match", "hasDecorations", "resultIndex", "SearchAddon", "Disposable", "options", "MutableDisposable", "SearchState", "SearchResultTracker", "Emitter", "terminal", "SearchLineCache", "SearchEngine", "DecorationManager", "toDisposable", "disposableTimeout", "term", "retainCachedSearchTerm", "searchOptions", "internalSearchOptions", "found", "results", "prevResult", "result", "cols", "nextCol", "nextRow", "noScroll", "activeDecoration", "scroll"] ++ "sourcesContent": ["/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal lifecycle utilities for xterm.js core.\n * Simplified from VS Code's lifecycle.ts - no tracking/leak detection.\n */\n\nexport interface IDisposable {\n dispose(): void;\n}\n\nexport function toDisposable(fn: () => void): IDisposable {\n return { dispose: fn };\n}\n\nexport function dispose(disposable: T): T;\nexport function dispose(disposable: T | undefined): T | undefined;\nexport function dispose(disposables: T[]): T[];\nexport function dispose(arg: T | T[] | undefined): T | T[] | undefined {\n if (!arg) {\n return arg;\n }\n if (Array.isArray(arg)) {\n for (const d of arg) {\n d.dispose();\n }\n return [];\n }\n arg.dispose();\n return arg;\n}\n\nexport function combinedDisposable(...disposables: IDisposable[]): IDisposable {\n return toDisposable(() => dispose(disposables));\n}\n\nexport class DisposableStore implements IDisposable {\n private readonly _disposables = new Set();\n private _isDisposed = false;\n\n public get isDisposed(): boolean {\n return this._isDisposed;\n }\n\n public add(o: T): T {\n if (this._isDisposed) {\n o.dispose();\n } else {\n this._disposables.add(o);\n }\n return o;\n }\n\n public dispose(): void {\n if (this._isDisposed) {\n return;\n }\n this._isDisposed = true;\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n\n public clear(): void {\n for (const d of this._disposables) {\n d.dispose();\n }\n this._disposables.clear();\n }\n}\n\nexport abstract class Disposable implements IDisposable {\n public static readonly None: IDisposable = Object.freeze({ dispose() { } });\n\n protected readonly _store = new DisposableStore();\n\n public dispose(): void {\n this._store.dispose();\n }\n\n protected _register(o: T): T {\n return this._store.add(o);\n }\n}\n\nexport class MutableDisposable implements IDisposable {\n private _value: T | undefined;\n private _isDisposed = false;\n\n public get value(): T | undefined {\n return this._isDisposed ? undefined : this._value;\n }\n\n public set value(value: T | undefined) {\n if (this._isDisposed || value === this._value) {\n return;\n }\n this._value?.dispose();\n this._value = value;\n }\n\n public clear(): void {\n this.value = undefined;\n }\n\n public dispose(): void {\n this._isDisposed = true;\n this._value?.dispose();\n this._value = undefined;\n }\n}\n", "/**\n * Copyright (c) 2024-2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal event utilities for xterm.js core.\n * Simplified from VS Code's event.ts - no leak detection/profiling.\n */\n\nimport { IDisposable, DisposableStore, toDisposable } from './Lifecycle';\n\nexport interface IEvent {\n (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore): IDisposable;\n}\n\nexport class Emitter {\n private _listeners: { fn: (e: T) => any, thisArgs: any }[] = [];\n private _disposed = false;\n private _event: IEvent | undefined;\n\n public get event(): IEvent {\n if (this._event) {\n return this._event;\n }\n this._event = (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n if (this._disposed) {\n return toDisposable(() => {});\n }\n\n const entry = { fn: listener, thisArgs };\n this._listeners = this._listeners.slice();\n this._listeners.push(entry);\n\n const result = toDisposable(() => {\n const idx = this._listeners.indexOf(entry);\n if (idx !== -1) {\n this._listeners = this._listeners.slice();\n this._listeners.splice(idx, 1);\n }\n });\n\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(result);\n } else {\n disposables.add(result);\n }\n }\n\n return result;\n };\n return this._event;\n }\n\n public fire(event: T): void {\n if (this._disposed || !this._listeners.length) {\n return;\n }\n if (this._listeners.length === 1) {\n this._listeners[0].fn.call(this._listeners[0].thisArgs, event);\n return;\n }\n const listeners = this._listeners;\n for (let i = 0, len = listeners.length; i < len; ++i) {\n listeners[i].fn.call(listeners[i].thisArgs, event);\n }\n }\n\n public dispose(): void {\n if (this._disposed) {\n return;\n }\n this._disposed = true;\n this._listeners.length = 0;\n }\n}\n\nexport namespace EventUtils {\n export function forward(from: IEvent, to: Emitter): IDisposable {\n return from(e => to.fire(e));\n }\n\n export function map(event: IEvent, map: (i: I) => O): IEvent {\n return (listener: (e: O) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n return event(i => listener.call(thisArgs, map(i)), undefined, disposables);\n };\n }\n\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent;\n export function any(...events: IEvent[]): IEvent {\n return (listener: (e: T) => any, thisArgs?: any, disposables?: IDisposable[] | DisposableStore) => {\n const store = new DisposableStore();\n for (const event of events) {\n store.add(event(e => listener.call(thisArgs, e)));\n }\n if (disposables) {\n if (Array.isArray(disposables)) {\n disposables.push(store);\n } else {\n disposables.add(store);\n }\n }\n return store;\n };\n }\n\n export function runAndSubscribe(event: IEvent, handler: (e: T) => void, initial: T): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void): IDisposable;\n export function runAndSubscribe(event: IEvent, handler: (e: T | undefined) => void, initial?: T): IDisposable {\n handler(initial);\n return event(e => handler(e));\n }\n}\n", "/**\n * Copyright (c) 2026 The xterm.js authors. All rights reserved.\n * @license MIT\n *\n * Minimal async helpers for xterm.js core.\n */\n\nimport { DisposableStore, IDisposable, toDisposable } from './Lifecycle';\n\nexport function timeout(millis: number): Promise {\n return new Promise(resolve => setTimeout(resolve, millis));\n}\n\n/**\n * Creates a timeout that can be disposed using its returned value.\n * @param handler The timeout handler.\n * @param timeout An optional timeout in milliseconds.\n * @param store An optional {@link DisposableStore} that will have the timeout disposable managed\n * automatically.\n */\nexport function disposableTimeout(handler: () => void, timeout = 0, store?: DisposableStore): IDisposable {\n const timer = setTimeout(() => {\n handler();\n if (store) {\n disposable.dispose();\n }\n }, timeout);\n const disposable = toDisposable(() => {\n clearTimeout(timer);\n });\n store?.add(disposable);\n return disposable;\n}\n\nexport class TimeoutTimer implements IDisposable {\n private _token: any = -1;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n if (this._token !== -1) {\n clearTimeout(this._token);\n this._token = -1;\n }\n }\n\n public cancelAndSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed TimeoutTimer');\n }\n this.cancel();\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n\n public setIfNotSet(runner: () => void, timeout: number): void {\n if (this._isDisposed) {\n throw new Error('Calling setIfNotSet on a disposed TimeoutTimer');\n }\n if (this._token !== -1) {\n return;\n }\n this._token = setTimeout(() => {\n this._token = -1;\n runner();\n }, timeout);\n }\n}\n\n/**\n * Schedules a single runner on the microtask queue. Unlike {@link TimeoutTimer}, a scheduled\n * microtask cannot be unqueued; {@link cancel} prevents the runner from executing if it has not\n * run yet.\n */\nexport class MicrotaskTimer implements IDisposable {\n private _isScheduled = false;\n private _isDisposed = false;\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n\n public cancel(): void {\n this._isScheduled = false;\n }\n\n public set(runner: () => void): void {\n if (this._isDisposed) {\n throw new Error('Calling set on a disposed MicrotaskTimer');\n }\n if (this._isScheduled) {\n return;\n }\n this._isScheduled = true;\n queueMicrotask(() => {\n if (!this._isScheduled) {\n return;\n }\n this._isScheduled = false;\n runner();\n });\n }\n}\n\nexport class IntervalTimer implements IDisposable {\n private _disposable: IDisposable | undefined;\n private _isDisposed = false;\n\n public cancel(): void {\n this._disposable?.dispose();\n this._disposable = undefined;\n }\n\n public cancelAndSet(runner: () => void, interval: number, context: Window | typeof globalThis = globalThis): void {\n if (this._isDisposed) {\n throw new Error('Calling cancelAndSet on a disposed IntervalTimer');\n }\n this.cancel();\n const handle = context.setInterval(() => {\n runner();\n }, interval);\n this._disposable = {\n dispose: () => {\n context.clearInterval(handle as any);\n this._disposable = undefined;\n }\n };\n }\n\n public dispose(): void {\n this.cancel();\n this._isDisposed = true;\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport { combinedDisposable, Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\n\nexport type LineCacheEntry = [\n /**\n * The string representation of a line (as opposed to the buffer cell representation).\n */\n lineAsString: string,\n /**\n * The offsets where each line starts when the entry describes a wrapped line.\n */\n lineOffsets: number[]\n];\n\n/**\n * Configuration constants for the search line cache functionality.\n */\nconst enum Constants {\n /**\n * Time-to-live for cached search results in milliseconds. After this duration, cached search\n * results will be invalidated to ensure they remain consistent with terminal content changes.\n */\n LINES_CACHE_TIME_TO_LIVE = 15000\n}\n\nexport class SearchLineCache extends Disposable {\n /**\n * translateBufferLineToStringWithWrap is a fairly expensive call.\n * We memoize the calls into an array that has a time based ttl.\n * _linesCache is also invalidated when the terminal cursor moves.\n */\n private _linesCache: LineCacheEntry[] | undefined;\n private _linesCacheTimeout = this._register(new MutableDisposable());\n private _linesCacheDisposables = this._register(new MutableDisposable());\n // Track access to avoid recreating a timeout on every init call which occurs once per search\n // result (findNext/findPrevious -> _highlightAllMatches -> find loop).\n private _lastAccessTimestamp = 0;\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this._destroyLinesCache()));\n }\n\n /**\n * Sets up a line cache with a ttl\n */\n public initLinesCache(): void {\n if (!this._linesCache) {\n this._linesCache = new Array(this._terminal.buffer.active.length);\n this._linesCacheDisposables.value = combinedDisposable(\n this._terminal.onLineFeed(() => this._destroyLinesCache()),\n this._terminal.onCursorMove(() => this._destroyLinesCache()),\n this._terminal.onResize(() => this._destroyLinesCache())\n );\n }\n\n this._lastAccessTimestamp = Date.now();\n if (!this._linesCacheTimeout.value) {\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE);\n }\n }\n\n private _destroyLinesCache(): void {\n this._linesCache = undefined;\n this._lastAccessTimestamp = 0;\n this._linesCacheDisposables.clear();\n this._linesCacheTimeout.clear();\n }\n\n private _scheduleLinesCacheTimeout(delay: number): void {\n this._linesCacheTimeout.value = disposableTimeout(() => {\n if (!this._linesCache) {\n return;\n }\n const now = Date.now();\n const elapsed = now - this._lastAccessTimestamp;\n if (elapsed >= Constants.LINES_CACHE_TIME_TO_LIVE) {\n this._destroyLinesCache();\n return;\n }\n this._scheduleLinesCacheTimeout(Constants.LINES_CACHE_TIME_TO_LIVE - elapsed);\n }, delay);\n }\n\n public getLineFromCache(row: number): LineCacheEntry | undefined {\n return this._linesCache?.[row];\n }\n\n public setLineInCache(row: number, entry: LineCacheEntry): void {\n if (this._linesCache) {\n this._linesCache[row] = entry;\n }\n }\n\n /**\n * Translates a buffer line to a string, including subsequent lines if they are wraps.\n * Wide characters will count as two columns in the resulting string. This\n * function is useful for getting the actual text underneath the raw selection\n * position.\n * @param lineIndex The index of the line being translated.\n * @param trimRight Whether to trim whitespace to the right.\n */\n public translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean): LineCacheEntry {\n const strings = [];\n const lineOffsets = [0];\n // A single line longer than the whole scrollback leaves every buffer row wrapped, and the\n // buffer's ring answers an out-of-range row by cycling back to the start, so an unbounded walk\n // never reaches an unwrapped line.\n const bufferLength = this._terminal.buffer.active.length;\n let line = this._terminal.buffer.active.getLine(lineIndex);\n while (line) {\n const nextLine = lineIndex + 1 < bufferLength ? this._terminal.buffer.active.getLine(lineIndex + 1) : undefined;\n const lineWrapsToNext = nextLine ? nextLine.isWrapped : false;\n let string = line.translateToString(!lineWrapsToNext && trimRight);\n if (lineWrapsToNext && nextLine) {\n const lastCell = line.getCell(line.length - 1);\n const lastCellIsNull = lastCell && lastCell.getCode() === 0 && lastCell.getWidth() === 1;\n // a wide character wrapped to the next line\n if (lastCellIsNull && nextLine.getCell(0)?.getWidth() === 2) {\n string = string.slice(0, -1);\n }\n }\n strings.push(string);\n if (lineWrapsToNext) {\n lineOffsets.push(lineOffsets[lineOffsets.length - 1] + string.length);\n } else {\n break;\n }\n lineIndex++;\n line = nextLine;\n }\n return [strings.join(''), lineOffsets];\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchOptions } from '@xterm/addon-search';\n\n/**\n * Manages search state including cached search terms, options tracking, and validation.\n * This class provides a centralized way to handle search state consistency and option changes.\n */\nexport class SearchState {\n private _cachedSearchTerm: string | undefined;\n private _lastSearchOptions: ISearchOptions | undefined;\n\n /**\n * Gets the currently cached search term.\n */\n public get cachedSearchTerm(): string | undefined {\n return this._cachedSearchTerm;\n }\n\n /**\n * Sets the cached search term.\n */\n public set cachedSearchTerm(term: string | undefined) {\n this._cachedSearchTerm = term;\n }\n\n /**\n * Gets the last search options used.\n */\n public get lastSearchOptions(): ISearchOptions | undefined {\n return this._lastSearchOptions;\n }\n\n /**\n * Sets the last search options used.\n */\n public set lastSearchOptions(options: ISearchOptions | undefined) {\n this._lastSearchOptions = options;\n }\n\n /**\n * Validates a search term to ensure it's not empty or invalid.\n * @param term The search term to validate.\n * @returns true if the term is valid for searching.\n */\n public isValidSearchTerm(term: string): boolean {\n return !!(term && term.length > 0);\n }\n\n /**\n * Determines if search options have changed compared to the last search.\n * @param newOptions The new search options to compare.\n * @returns true if the options have changed.\n */\n public didOptionsChange(newOptions?: ISearchOptions): boolean {\n if (!this._lastSearchOptions) {\n return true;\n }\n if (!newOptions) {\n return false;\n }\n if (this._lastSearchOptions.caseSensitive !== newOptions.caseSensitive) {\n return true;\n }\n if (this._lastSearchOptions.regex !== newOptions.regex) {\n return true;\n }\n if (this._lastSearchOptions.wholeWord !== newOptions.wholeWord) {\n return true;\n }\n return false;\n }\n\n /**\n * Determines if a new search should trigger highlighting updates.\n * @param term The search term.\n * @param options The search options.\n * @returns true if highlighting should be updated.\n */\n public shouldUpdateHighlighting(term: string, options?: ISearchOptions): boolean {\n if (!options?.decorations) {\n return false;\n }\n return this._cachedSearchTerm === undefined ||\n term !== this._cachedSearchTerm ||\n this.didOptionsChange(options);\n }\n\n /**\n * Clears the cached search term.\n */\n public clearCachedTerm(): void {\n this._cachedSearchTerm = undefined;\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this._cachedSearchTerm = undefined;\n this._lastSearchOptions = undefined;\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal } from '@xterm/xterm';\nimport type { ISearchOptions } from '@xterm/addon-search';\nimport type { SearchLineCache } from './SearchLineCache';\n\n/**\n * Represents the position to start a search from.\n */\ninterface ISearchPosition {\n startCol: number;\n startRow: number;\n}\n\n/**\n * Represents a search result with its position and content.\n */\nexport interface ISearchResult {\n term: string;\n col: number;\n row: number;\n size: number;\n}\n\n/**\n * Configuration constants for the search engine functionality.\n */\nconst enum Constants {\n /**\n * Characters that are considered non-word characters for search boundary detection. These\n * characters are used to determine word boundaries when performing whole-word searches. Includes\n * common punctuation, symbols, and whitespace characters.\n */\n NON_WORD_CHARACTERS = ' ~!@#$%^&*()+`-=[]{}|\\\\;:\"\\',./<>?'\n}\n\n/**\n * Core search engine that handles finding text within terminal content.\n * This class is responsible for the actual search algorithms and position calculations.\n */\nexport class SearchEngine {\n constructor(\n private readonly _terminal: Terminal,\n private readonly _lineCache: SearchLineCache\n ) {}\n\n /**\n * Find the first occurrence of a term starting from a specific position.\n * @param term The search term.\n * @param startRow The row to start searching from.\n * @param startCol The column to start searching from.\n * @param searchOptions Search options.\n * @returns The search result if found, undefined otherwise.\n */\n public find(term: string, startRow: number, startCol: number, searchOptions?: ISearchOptions): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n if (startCol >= this._terminal.cols) {\n throw new Error(`Invalid col: ${startCol} to search in terminal of ${this._terminal.cols} cols`);\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n if (this._isRowCoveredByEarlierSearch(y)) {\n continue;\n }\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n return result;\n }\n\n /**\n * Find the next occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine incremental behavior.\n * @returns The search result if found, undefined otherwise.\n */\n public findNextWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startCol = 0;\n let startRow = 0;\n if (prevSelectedPos) {\n if (cachedSearchTerm === term) {\n startCol = prevSelectedPos.end.x;\n startRow = prevSelectedPos.end.y;\n } else {\n startCol = prevSelectedPos.start.x;\n startRow = prevSelectedPos.start.y;\n }\n }\n\n this._lineCache.initLinesCache();\n\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n // Search startRow\n let result = this._findInLine(term, searchPosition, searchOptions);\n // Search from startRow + 1 to end\n if (!result) {\n for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) {\n if (this._isRowCoveredByEarlierSearch(y)) {\n continue;\n }\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n // If we hit the bottom and didn't search from the very top wrap back up\n if (!result && startRow !== 0) {\n for (let y = 0; y < startRow; y++) {\n // Row 0 is never skipped: it can be a continuation whose line start was trimmed from the\n // scrollback, and nothing earlier in this loop has searched it.\n if (y > 0 && this._isRowCoveredByEarlierSearch(y)) {\n continue;\n }\n searchPosition.startRow = y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n if (result) {\n break;\n }\n }\n }\n\n // If there is only one result, wrap back and return selection if it exists.\n if (!result && prevSelectedPos) {\n searchPosition.startRow = prevSelectedPos.start.y;\n searchPosition.startCol = 0;\n result = this._findInLine(term, searchPosition, searchOptions);\n }\n\n return result;\n }\n\n /**\n * Find the previous occurrence of a term with wrapping and selection management.\n * @param term The search term.\n * @param searchOptions Search options.\n * @param cachedSearchTerm The cached search term to determine if expansion should occur.\n * @returns The search result if found, undefined otherwise.\n */\n public findPreviousWithSelection(term: string, searchOptions?: ISearchOptions, cachedSearchTerm?: string): ISearchResult | undefined {\n if (!term || term.length === 0) {\n this._terminal.clearSelection();\n return undefined;\n }\n\n const prevSelectedPos = this._terminal.getSelectionPosition();\n this._terminal.clearSelection();\n\n let startRow = this._terminal.buffer.active.baseY + this._terminal.rows - 1;\n const startCol = this._terminal.cols;\n const isReverseSearch = true;\n\n this._lineCache.initLinesCache();\n const searchPosition: ISearchPosition = {\n startRow,\n startCol\n };\n\n let result: ISearchResult | undefined;\n if (prevSelectedPos) {\n searchPosition.startRow = startRow = prevSelectedPos.start.y;\n searchPosition.startCol = prevSelectedPos.start.x;\n if (cachedSearchTerm !== term) {\n // Try to expand selection to right first.\n result = this._findInLine(term, searchPosition, searchOptions, false);\n if (!result) {\n // If selection was not able to be expanded to the right, then try reverse search\n searchPosition.startRow = startRow = prevSelectedPos.end.y;\n searchPosition.startCol = prevSelectedPos.end.x;\n }\n }\n }\n\n result ??= this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n\n // Search from startRow - 1 to top\n if (!result) {\n searchPosition.startCol = Math.max(searchPosition.startCol, this._terminal.cols);\n for (let y = startRow - 1; y >= 0; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n // If we hit the top and didn't search from the very bottom wrap back down\n if (!result && startRow !== (this._terminal.buffer.active.baseY + this._terminal.rows - 1)) {\n for (let y = (this._terminal.buffer.active.baseY + this._terminal.rows - 1); y >= startRow; y--) {\n searchPosition.startRow = y;\n result = this._findInLine(term, searchPosition, searchOptions, isReverseSearch);\n if (result) {\n break;\n }\n }\n }\n\n return result;\n }\n\n /**\n * A found substring is a whole word if it doesn't have an alphanumeric character directly\n * adjacent to it.\n * @param searchIndex starting index of the potential whole word substring\n * @param line entire string in which the potential whole word was found\n * @param term the substring that starts at searchIndex\n */\n private _isWholeWord(searchIndex: number, line: string, term: string): boolean {\n return ((searchIndex === 0) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex - 1]))) &&\n (((searchIndex + term.length) === line.length) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex + term.length])));\n }\n\n /** `_isWholeWord` gated on the option, so a rejected hit can be stepped past instead of ending the scan. */\n private _satisfiesWholeWord(searchIndex: number, line: string, term: string, searchOptions: ISearchOptions): boolean {\n return !searchOptions.wholeWord || this._isWholeWord(searchIndex, line, term);\n }\n\n /**\n * Whether an earlier `_findInLine` in this same call already scanned this row's line from an\n * equal or lower offset, which makes rescanning it pure O(rows^2) work on one long line. Sound\n * for every option because `_findInLine` returns the first accepted match at or after its\n * offset, which is monotone in that offset. Only valid once such a search has happened \u2014 the\n * wrap-around loop starts at row 0, whose line start may have been trimmed from the scrollback.\n */\n private _isRowCoveredByEarlierSearch(row: number): boolean {\n return this._terminal.buffer.active.getLine(row)?.isWrapped === true;\n }\n\n /**\n * Searches a line for a search term. Takes the provided terminal line and searches the text line,\n * which may contain subsequent terminal lines if the text is wrapped. If the provided line number\n * is part of a wrapped text line that started on an earlier line then it is skipped since it will\n * be properly searched when the terminal line that the text starts on is searched.\n * @param term The search term.\n * @param searchPosition The position to start the search.\n * @param searchOptions Search options.\n * @param isReverseSearch Whether the search should start from the right side of the terminal and\n * search to the left.\n * @returns The search result if it was found.\n */\n private _findInLine(term: string, searchPosition: ISearchPosition, searchOptions: ISearchOptions = {}, isReverseSearch: boolean = false): ISearchResult | undefined {\n // Ignore wrapped lines, only consider on unwrapped line (first row of command string).\n if (isReverseSearch) {\n // Reverse search never rewinds: its caller carries startCol down the rows of the line. Row 0\n // is searched even when wrapped, since its line start may have been trimmed from the scrollback.\n if (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) {\n searchPosition.startCol += this._terminal.cols;\n return;\n }\n } else {\n // A loop rather than recursion: one frame per wrapped row overflows the stack on a line long\n // enough to fill the scrollback. Bounded at row 0 because after a reflow the buffer's ring\n // holds stale entries at negative indices, so `getLine(-1)` answers with a wrapped line.\n while (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) {\n searchPosition.startRow--;\n searchPosition.startCol += this._terminal.cols;\n }\n }\n const row = searchPosition.startRow;\n const col = searchPosition.startCol;\n\n let cache = this._lineCache.getLineFromCache(row);\n if (!cache) {\n cache = this._lineCache.translateBufferLineToStringWithWrap(row, true);\n this._lineCache.setLineInCache(row, cache);\n }\n const [stringLine, offsets] = cache;\n\n const offset = this._bufferColsToStringOffset(row, col, offsets);\n let searchTerm = term;\n let searchStringLine = stringLine;\n if (!searchOptions.regex) {\n searchTerm = searchOptions.caseSensitive ? term : term.toLowerCase();\n searchStringLine = searchOptions.caseSensitive ? stringLine : stringLine.toLowerCase();\n }\n\n let resultIndex = -1;\n if (searchOptions.regex) {\n const searchRegex = RegExp(searchTerm, searchOptions.caseSensitive ? 'g' : 'gi');\n let foundTerm: RegExpExecArray | null;\n if (isReverseSearch) {\n // This loop will get the resultIndex of the _last_ regex match in the range 0..offset\n while (foundTerm = searchRegex.exec(searchStringLine.slice(0, offset))) {\n const matchIndex = searchRegex.lastIndex - foundTerm[0].length;\n if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) {\n resultIndex = matchIndex;\n term = foundTerm[0];\n }\n searchRegex.lastIndex = matchIndex + 1;\n }\n } else {\n // Driven over the whole line from `offset` rather than over `slice(offset)`: a slice\n // re-anchors ^ and \\b at whatever column the row happened to wrap at, and only\n // first-accepted-match-at-or-after-offset is monotone in `offset`, which is what lets\n // `_isRowCoveredByEarlierSearch` skip a wrapped row an earlier scan already covered.\n searchRegex.lastIndex = offset;\n while (foundTerm = searchRegex.exec(searchStringLine)) {\n const matchIndex = searchRegex.lastIndex - foundTerm[0].length;\n if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) {\n resultIndex = matchIndex;\n term = foundTerm[0];\n break;\n }\n // A zero-length or rejected match would otherwise repeat forever.\n searchRegex.lastIndex = matchIndex + 1;\n }\n }\n } else if (isReverseSearch) {\n let matchIndex = offset - searchTerm.length >= 0 ? searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length) : -1;\n // `lastIndexOf` clamps a negative fromIndex to 0, so index 0 has to end the walk.\n while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) {\n matchIndex = matchIndex > 0 ? searchStringLine.lastIndexOf(searchTerm, matchIndex - 1) : -1;\n }\n resultIndex = matchIndex;\n } else {\n let matchIndex = searchStringLine.indexOf(searchTerm, offset);\n while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) {\n matchIndex = searchStringLine.indexOf(searchTerm, matchIndex + 1);\n }\n resultIndex = matchIndex;\n }\n\n if (resultIndex >= 0) {\n // Adjust the row number and search index if needed since a \"line\" of text can span multiple\n // rows\n let startRowOffset = 0;\n while (startRowOffset < offsets.length - 1 && resultIndex >= offsets[startRowOffset + 1]) {\n startRowOffset++;\n }\n let endRowOffset = startRowOffset;\n while (endRowOffset < offsets.length - 1 && resultIndex + term.length >= offsets[endRowOffset + 1]) {\n endRowOffset++;\n }\n const startColOffset = resultIndex - offsets[startRowOffset];\n const endColOffset = resultIndex + term.length - offsets[endRowOffset];\n const startColIndex = this._stringLengthToBufferSize(row + startRowOffset, startColOffset);\n const endColIndex = this._stringLengthToBufferSize(row + endRowOffset, endColOffset);\n const size = endColIndex - startColIndex + this._terminal.cols * (endRowOffset - startRowOffset);\n\n return {\n term,\n col: startColIndex,\n row: row + startRowOffset,\n size\n };\n }\n }\n\n private _stringLengthToBufferSize(row: number, offset: number): number {\n const line = this._terminal.buffer.active.getLine(row);\n if (!line) {\n return 0;\n }\n for (let i = 0; i < offset; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n // Adjust the searchIndex to normalize emoji into single chars\n const char = cell.getChars();\n if (char.length > 1) {\n offset -= char.length - 1;\n }\n // Adjust the searchIndex for empty characters following wide unicode\n // chars (eg. CJK)\n const nextCell = line.getCell(i + 1);\n if (nextCell && nextCell.getWidth() === 0) {\n offset++;\n }\n }\n return offset;\n }\n\n /**\n * `cols` counts from the start of the logical line, so summing the cells of every row before the\n * resume point costs O(line) per call and the highlight-all pass makes one call per match.\n * `lineOffsets` already holds the string offset each wrapped row starts at \u2014 the same map used\n * above to turn a match index back into a row \u2014 so only the last, partial row needs cells. It is\n * also the map the row a match lands on is read from, which the cell sum disagreed with by one\n * for a row whose trailing cell is the null placeholder of a wide character that wrapped.\n */\n private _bufferColsToStringOffset(startRow: number, cols: number, lineOffsets: number[]): number {\n const rowsBack = Math.min(Math.floor(cols / this._terminal.cols), lineOffsets.length - 1);\n let offset = lineOffsets[rowsBack];\n const line = this._terminal.buffer.active.getLine(startRow + rowsBack);\n if (line) {\n const colsInRow = Math.min(cols - rowsBack * this._terminal.cols, this._terminal.cols);\n for (let i = 0; i < colsInRow; i++) {\n const cell = line.getCell(i);\n if (!cell) {\n break;\n }\n if (cell.getWidth()) {\n // Treat null characters as whitespace to align with the translateToString API\n offset += cell.getCode() === 0 ? 1 : cell.getChars().length;\n }\n }\n }\n return offset;\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, IDecoration } from '@xterm/xterm';\nimport type { ISearchDecorationOptions } from '@xterm/addon-search';\nimport { dispose, Disposable, toDisposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a highlight decoration.\n */\ninterface IHighlight extends IDisposable {\n decoration: IDecoration;\n match: ISearchResult;\n}\n\n/**\n * Interface for managing multiple decorations for a single match.\n */\ninterface IMultiHighlight extends IDisposable {\n decorations: IDecoration[];\n match: ISearchResult;\n}\n\n/**\n * Manages visual decorations for search results including highlighting and active selection\n * indicators. This class handles the creation, styling, and disposal of search-related decorations.\n */\nexport class DecorationManager extends Disposable {\n private _highlightDecorations: IHighlight[] = [];\n private _highlightedLines: Set = new Set();\n\n constructor(private readonly _terminal: Terminal) {\n super();\n this._register(toDisposable(() => this.clearHighlightDecorations()));\n }\n\n /**\n * Creates decorations for all provided search results.\n * @param results The search results to create decorations for.\n * @param options The decoration options.\n */\n public createHighlightDecorations(results: ISearchResult[], options: ISearchDecorationOptions): void {\n this.clearHighlightDecorations();\n\n for (const match of results) {\n const decorations = this._createResultDecorations(match, options, false);\n if (decorations) {\n for (const decoration of decorations) {\n this._storeDecoration(decoration, match);\n }\n }\n }\n }\n\n /**\n * Creates decorations for the currently active search result.\n * @param result The active search result.\n * @param options The decoration options.\n * @returns The multi-highlight decoration or undefined if creation failed.\n */\n public createActiveDecoration(result: ISearchResult, options: ISearchDecorationOptions): IMultiHighlight | undefined {\n const decorations = this._createResultDecorations(result, options, true);\n if (decorations) {\n return { decorations, match: result, dispose() { dispose(decorations); } };\n }\n return undefined;\n }\n\n /**\n * Clears all highlight decorations.\n */\n public clearHighlightDecorations(): void {\n dispose(this._highlightDecorations);\n this._highlightDecorations = [];\n this._highlightedLines.clear();\n }\n\n /**\n * Stores a decoration and tracks it for management.\n * @param decoration The decoration to store.\n * @param match The search result this decoration represents.\n */\n private _storeDecoration(decoration: IDecoration, match: ISearchResult): void {\n this._highlightedLines.add(decoration.marker.line);\n this._highlightDecorations.push({ decoration, match, dispose() { decoration.dispose(); } });\n }\n\n /**\n * Applies styles to the decoration when it is rendered.\n * @param element The decoration's element.\n * @param borderColor The border color to apply.\n * @param isActiveResult Whether the element is part of the active search result.\n */\n private _applyStyles(element: HTMLElement, borderColor: string | undefined, isActiveResult: boolean): void {\n if (!element.classList.contains('xterm-find-result-decoration')) {\n element.classList.add('xterm-find-result-decoration');\n if (borderColor) {\n element.style.outline = `1px solid ${borderColor}`;\n }\n }\n if (isActiveResult) {\n element.classList.add('xterm-find-active-result-decoration');\n }\n }\n\n /**\n * Creates a decoration for the result and applies styles\n * @param result the search result for which to create the decoration\n * @param options the options for the decoration\n * @param isActiveResult whether this is the currently active result\n * @returns the decorations or undefined if the marker has already been disposed of\n */\n private _createResultDecorations(result: ISearchResult, options: ISearchDecorationOptions, isActiveResult: boolean): IDecoration[] | undefined {\n // Gather decoration ranges for this match as it could wrap\n const decorationRanges: [number, number, number][] = [];\n let currentCol = result.col;\n let remainingSize = result.size;\n let markerOffset = -this._terminal.buffer.active.baseY - this._terminal.buffer.active.cursorY + result.row;\n while (remainingSize > 0) {\n const amountThisRow = Math.min(this._terminal.cols - currentCol, remainingSize);\n decorationRanges.push([markerOffset, currentCol, amountThisRow]);\n currentCol = 0;\n remainingSize -= amountThisRow;\n markerOffset++;\n }\n\n // Create the decorations\n const decorations: IDecoration[] = [];\n for (const range of decorationRanges) {\n const marker = this._terminal.registerMarker(range[0]);\n const decoration = this._terminal.registerDecoration({\n marker,\n x: range[1],\n width: range[2],\n layer: isActiveResult ? 'top' : 'bottom',\n backgroundColor: isActiveResult ? options.activeMatchBackground : options.matchBackground,\n overviewRulerOptions: this._highlightedLines.has(marker.line) ? undefined : {\n color: isActiveResult ? options.activeMatchColorOverviewRuler : options.matchOverviewRuler,\n position: 'center'\n }\n });\n if (decoration) {\n const disposables: IDisposable[] = [];\n disposables.push(marker);\n disposables.push(decoration.onRender((e) => this._applyStyles(e, isActiveResult ? options.activeMatchBorder : options.matchBorder, false)));\n disposables.push(decoration.onDispose(() => dispose(disposables)));\n decorations.push(decoration);\n }\n }\n\n return decorations.length === 0 ? undefined : decorations;\n }\n}\n\n\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { ISearchResultChangeEvent } from '@xterm/addon-search';\nimport type { IDisposable } from '@xterm/xterm';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable } from 'common/Lifecycle';\nimport type { ISearchResult } from './SearchEngine';\n\n/**\n * Interface for managing a currently selected decoration.\n */\ninterface ISelectedDecoration extends IDisposable {\n match: ISearchResult;\n}\n\n/**\n * Tracks search results, manages result indexing, and fires events when results change.\n * This class provides centralized management of search result state and notifications.\n */\nexport class SearchResultTracker extends Disposable {\n private _searchResults: ISearchResult[] = [];\n private _selectedDecoration: ISelectedDecoration | undefined;\n\n private readonly _onDidChangeResults = this._register(new Emitter());\n public get onDidChangeResults(): IEvent { return this._onDidChangeResults.event; }\n\n /**\n * Gets the current search results.\n */\n public get searchResults(): ReadonlyArray {\n return this._searchResults;\n }\n\n /**\n * Gets the currently selected decoration.\n */\n public get selectedDecoration(): ISelectedDecoration | undefined {\n return this._selectedDecoration;\n }\n\n /**\n * Sets the currently selected decoration.\n */\n public set selectedDecoration(decoration: ISelectedDecoration | undefined) {\n this._selectedDecoration = decoration;\n }\n\n /**\n * Updates the search results with a new set of results.\n * @param results The new search results.\n * @param maxResults The maximum number of results to track.\n */\n public updateResults(results: ISearchResult[], maxResults: number): void {\n this._searchResults = results.slice(0, maxResults);\n }\n\n /**\n * Clears all search results.\n */\n public clearResults(): void {\n this._searchResults = [];\n }\n\n /**\n * Clears the selected decoration.\n */\n public clearSelectedDecoration(): void {\n if (this._selectedDecoration) {\n this._selectedDecoration.dispose();\n this._selectedDecoration = undefined;\n }\n }\n\n /**\n * Finds the index of a result in the current results array.\n * @param result The result to find.\n * @returns The index of the result, or -1 if not found.\n */\n public findResultIndex(result: ISearchResult): number {\n for (let i = 0; i < this._searchResults.length; i++) {\n const match = this._searchResults[i];\n if (match.row === result.row && match.col === result.col && match.size === result.size) {\n return i;\n }\n }\n return -1;\n }\n\n /**\n * Fires a result change event with the current state.\n * @param hasDecorations Whether decorations are enabled.\n */\n public fireResultsChanged(hasDecorations: boolean): void {\n if (!hasDecorations) {\n return;\n }\n\n let resultIndex = -1;\n if (this._selectedDecoration) {\n resultIndex = this.findResultIndex(this._selectedDecoration.match);\n }\n\n this._onDidChangeResults.fire({\n resultIndex,\n resultCount: this._searchResults.length\n });\n }\n\n /**\n * Resets all state.\n */\n public reset(): void {\n this.clearSelectedDecoration();\n this.clearResults();\n }\n}\n", "/**\n * Copyright (c) 2017 The xterm.js authors. All rights reserved.\n * @license MIT\n */\n\nimport type { Terminal, IDisposable, ITerminalAddon } from '@xterm/xterm';\nimport type { SearchAddon as ISearchApi, ISearchOptions, ISearchAddonOptions, ISearchResultChangeEvent, ISearchDecorationOptions } from '@xterm/addon-search';\nimport { Emitter, type IEvent } from 'common/Event';\nimport { Disposable, MutableDisposable, toDisposable } from 'common/Lifecycle';\nimport { disposableTimeout } from 'common/Async';\nimport { SearchLineCache } from './SearchLineCache';\nimport { SearchState } from './SearchState';\nimport { SearchEngine, type ISearchResult } from './SearchEngine';\nimport { DecorationManager } from './DecorationManager';\nimport { SearchResultTracker } from './SearchResultTracker';\n\ninterface IInternalSearchOptions {\n noScroll: boolean;\n}\n\n/**\n * Configuration constants for the search addon functionality.\n */\nconst enum Constants {\n /**\n * Default maximum number of search results to highlight simultaneously. This limit prevents\n * performance degradation when searching for very common terms that would result in excessive\n * highlighting decorations.\n */\n DEFAULT_HIGHLIGHT_LIMIT = 1000\n}\n\nexport class SearchAddon extends Disposable implements ITerminalAddon, ISearchApi {\n private _terminal: Terminal | undefined;\n private _highlightLimit: number;\n private _highlightTimeout = this._register(new MutableDisposable());\n private _lineCache = this._register(new MutableDisposable());\n\n // Component instances\n private _state = new SearchState();\n private _engine: SearchEngine | undefined;\n private _decorationManager: DecorationManager | undefined;\n private _resultTracker = this._register(new SearchResultTracker());\n\n private readonly _onAfterSearch = this._register(new Emitter());\n public readonly onAfterSearch = this._onAfterSearch.event;\n private readonly _onBeforeSearch = this._register(new Emitter());\n public readonly onBeforeSearch = this._onBeforeSearch.event;\n\n public get onDidChangeResults(): IEvent {\n return this._resultTracker.onDidChangeResults;\n }\n\n constructor(options?: Partial) {\n super();\n\n this._highlightLimit = options?.highlightLimit ?? Constants.DEFAULT_HIGHLIGHT_LIMIT;\n }\n\n public activate(terminal: Terminal): void {\n this._terminal = terminal;\n this._lineCache.value = new SearchLineCache(terminal);\n this._engine = new SearchEngine(terminal, this._lineCache.value);\n this._decorationManager = new DecorationManager(terminal);\n this._register(this._terminal.onWriteParsed(() => this._updateMatches()));\n this._register(this._terminal.onResize(() => this._updateMatches()));\n this._register(toDisposable(() => this.clearDecorations()));\n }\n\n private _updateMatches(): void {\n this._highlightTimeout.clear();\n if (this._state.cachedSearchTerm && this._state.lastSearchOptions?.decorations) {\n this._highlightTimeout.value = disposableTimeout(() => {\n const term = this._state.cachedSearchTerm;\n this._state.clearCachedTerm();\n this.findPrevious(term!, { ...this._state.lastSearchOptions, incremental: true }, { noScroll: true });\n }, 200);\n }\n }\n\n public clearDecorations(retainCachedSearchTerm?: boolean): void {\n this._resultTracker.clearSelectedDecoration();\n this._decorationManager?.clearHighlightDecorations();\n this._resultTracker.clearResults();\n if (!retainCachedSearchTerm) {\n this._state.clearCachedTerm();\n }\n }\n\n public clearActiveDecoration(): void {\n this._resultTracker.clearSelectedDecoration();\n }\n\n /**\n * Find the next instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findNext(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findNextAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _highlightAllMatches(term: string, searchOptions: ISearchOptions): void {\n if (!this._terminal || !this._engine || !this._decorationManager) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n if (!this._state.isValidSearchTerm(term)) {\n this.clearDecorations();\n return;\n }\n\n // new search, clear out the old decorations\n this.clearDecorations(true);\n\n const results: ISearchResult[] = [];\n let prevResult: ISearchResult | undefined = undefined;\n let result = this._engine.find(term, 0, 0, searchOptions);\n\n while (result && (prevResult?.row !== result.row || prevResult?.col !== result.col)) {\n if (results.length >= this._highlightLimit) {\n break;\n }\n prevResult = result;\n results.push(prevResult);\n const cols = this._terminal.cols;\n let nextCol = prevResult.col + prevResult.size;\n let nextRow = prevResult.row;\n if (nextCol >= cols) {\n nextRow += Math.floor(nextCol / cols);\n nextCol = nextCol % cols;\n }\n result = this._engine.find(term, nextRow, nextCol, searchOptions);\n }\n\n this._resultTracker.updateResults(results, this._highlightLimit);\n if (searchOptions.decorations) {\n this._decorationManager.createHighlightDecorations(results, searchOptions.decorations);\n }\n }\n\n private _findNextAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findNextWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Find the previous instance of the term, then scroll to and select it. If it\n * doesn't exist, do nothing.\n * @param term The search term.\n * @param searchOptions Search options.\n * @returns Whether a result was found.\n */\n public findPrevious(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n throw new Error('Cannot use addon until it has been loaded');\n }\n\n this._onBeforeSearch.fire();\n\n this._state.lastSearchOptions = searchOptions;\n\n if (this._state.shouldUpdateHighlighting(term, searchOptions)) {\n this._highlightAllMatches(term, searchOptions!);\n }\n\n const found = this._findPreviousAndSelect(term, searchOptions, internalSearchOptions);\n this._fireResults(searchOptions);\n this._state.cachedSearchTerm = term;\n\n this._onAfterSearch.fire();\n\n return found;\n }\n\n private _fireResults(searchOptions?: ISearchOptions): void {\n this._resultTracker.fireResultsChanged(!!searchOptions?.decorations);\n }\n\n private _findPreviousAndSelect(term: string, searchOptions?: ISearchOptions, internalSearchOptions?: IInternalSearchOptions): boolean {\n if (!this._terminal || !this._engine) {\n return false;\n }\n if (!this._state.isValidSearchTerm(term)) {\n this._terminal.clearSelection();\n this.clearDecorations();\n return false;\n }\n\n const result = this._engine.findPreviousWithSelection(term, searchOptions, this._state.cachedSearchTerm);\n return this._selectResult(result, searchOptions?.decorations, internalSearchOptions?.noScroll);\n }\n\n /**\n * Selects and scrolls to a result.\n * @param result The result to select.\n * @returns Whether a result was selected.\n */\n private _selectResult(result: ISearchResult | undefined, options?: ISearchDecorationOptions, noScroll?: boolean): boolean {\n if (!this._terminal || !this._decorationManager) {\n return false;\n }\n\n this._resultTracker.clearSelectedDecoration();\n if (!result) {\n this._terminal.clearSelection();\n return false;\n }\n\n this._terminal.select(result.col, result.row, result.size);\n if (options) {\n const activeDecoration = this._decorationManager.createActiveDecoration(result, options);\n if (activeDecoration) {\n this._resultTracker.selectedDecoration = activeDecoration;\n }\n }\n\n if (!noScroll) {\n // If it is not in the viewport then we scroll else it just gets selected\n if (result.row >= (this._terminal.buffer.active.viewportY + this._terminal.rows) || result.row < this._terminal.buffer.active.viewportY) {\n let scroll = result.row - this._terminal.buffer.active.viewportY;\n scroll -= Math.floor(this._terminal.rows / 2);\n this._terminal.scrollLines(scroll);\n }\n }\n return true;\n }\n}\n"], ++ "mappings": ";;;;;;;;;;;;;;;;AAYO,SAASA,EAAaC,EAA6B,CACxD,MAAO,CAAE,QAASA,CAAG,CACvB,CAKO,SAASC,EAA+BC,EAA+C,CAC5F,GAAI,CAACA,EACH,OAAOA,EAET,GAAI,MAAM,QAAQA,CAAG,EAAG,CACtB,QAAWC,KAAKD,EACdC,EAAE,QAAQ,EAEZ,MAAO,CAAC,CACV,CACA,OAAAD,EAAI,QAAQ,EACLA,CACT,CAEO,SAASE,KAAsBC,EAAyC,CAC7E,OAAON,EAAa,IAAME,EAAQI,CAAW,CAAC,CAChD,CAEO,IAAMC,EAAN,KAA6C,CAA7C,cACL,KAAiB,aAAe,IAAI,IACpC,KAAQ,YAAc,GAEtB,IAAW,YAAsB,CAC/B,OAAO,KAAK,WACd,CAEO,IAA2BC,EAAS,CACzC,OAAI,KAAK,YACPA,EAAE,QAAQ,EAEV,KAAK,aAAa,IAAIA,CAAC,EAElBA,CACT,CAEO,SAAgB,CACrB,GAAI,MAAK,YAGT,MAAK,YAAc,GACnB,QAAWJ,KAAK,KAAK,aACnBA,EAAE,QAAQ,EAEZ,KAAK,aAAa,MAAM,EAC1B,CAEO,OAAc,CACnB,QAAWA,KAAK,KAAK,aACnBA,EAAE,QAAQ,EAEZ,KAAK,aAAa,MAAM,CAC1B,CACF,EAEsBK,EAAf,KAAiD,CAAjD,cAGL,KAAmB,OAAS,IAAIF,EAEzB,SAAgB,CACrB,KAAK,OAAO,QAAQ,CACtB,CAEU,UAAiCC,EAAS,CAClD,OAAO,KAAK,OAAO,IAAIA,CAAC,CAC1B,CACF,EAZsBC,EACG,KAAoB,OAAO,OAAO,CAAE,SAAU,CAAE,CAAE,CAAC,EAarE,IAAMC,EAAN,KAAsE,CAAtE,cAEL,KAAQ,YAAc,GAEtB,IAAW,OAAuB,CAChC,OAAO,KAAK,YAAc,OAAY,KAAK,MAC7C,CAEA,IAAW,MAAMC,EAAsB,CACjC,KAAK,aAAeA,IAAU,KAAK,SAGvC,KAAK,QAAQ,QAAQ,EACrB,KAAK,OAASA,EAChB,CAEO,OAAc,CACnB,KAAK,MAAQ,MACf,CAEO,SAAgB,CACrB,KAAK,YAAc,GACnB,KAAK,QAAQ,QAAQ,EACrB,KAAK,OAAS,MAChB,CACF,EClGO,IAAMC,EAAN,KAAiB,CAAjB,cACL,KAAQ,WAAqD,CAAC,EAC9D,KAAQ,UAAY,GAGpB,IAAW,OAAmB,CAC5B,OAAI,KAAK,OACA,KAAK,QAEd,KAAK,OAAS,CAACC,EAAyBC,EAAgBC,IAAkD,CACxG,GAAI,KAAK,UACP,OAAOC,EAAa,IAAM,CAAC,CAAC,EAG9B,IAAMC,EAAQ,CAAE,GAAIJ,EAAU,SAAAC,CAAS,EACvC,KAAK,WAAa,KAAK,WAAW,MAAM,EACxC,KAAK,WAAW,KAAKG,CAAK,EAE1B,IAAMC,EAASF,EAAa,IAAM,CAChC,IAAMG,EAAM,KAAK,WAAW,QAAQF,CAAK,EACrCE,IAAQ,KACV,KAAK,WAAa,KAAK,WAAW,MAAM,EACxC,KAAK,WAAW,OAAOA,EAAK,CAAC,EAEjC,CAAC,EAED,OAAIJ,IACE,MAAM,QAAQA,CAAW,EAC3BA,EAAY,KAAKG,CAAM,EAEvBH,EAAY,IAAIG,CAAM,GAInBA,CACT,EACO,KAAK,OACd,CAEO,KAAKE,EAAgB,CAC1B,GAAI,KAAK,WAAa,CAAC,KAAK,WAAW,OACrC,OAEF,GAAI,KAAK,WAAW,SAAW,EAAG,CAChC,KAAK,WAAW,CAAC,EAAE,GAAG,KAAK,KAAK,WAAW,CAAC,EAAE,SAAUA,CAAK,EAC7D,MACF,CACA,IAAMC,EAAY,KAAK,WACvB,QAASC,EAAI,EAAGC,EAAMF,EAAU,OAAQC,EAAIC,EAAK,EAAED,EACjDD,EAAUC,CAAC,EAAE,GAAG,KAAKD,EAAUC,CAAC,EAAE,SAAUF,CAAK,CAErD,CAEO,SAAgB,CACjB,KAAK,YAGT,KAAK,UAAY,GACjB,KAAK,WAAW,OAAS,EAC3B,CACF,EAEiBI,MAAV,CACE,SAASC,EAAWC,EAAiBC,EAA6B,CACvE,OAAOD,EAAKE,GAAKD,EAAG,KAAKC,CAAC,CAAC,CAC7B,CAFOJ,EAAS,QAAAC,EAIT,SAASI,EAAUT,EAAkBS,EAA6B,CACvE,MAAO,CAAChB,EAAyBC,EAAgBC,IACxCK,EAAME,GAAKT,EAAS,KAAKC,EAAUe,EAAIP,CAAC,CAAC,EAAG,OAAWP,CAAW,CAE7E,CAJOS,EAAS,IAAAK,EAQT,SAASC,KAAUC,EAAgC,CACxD,MAAO,CAAClB,EAAyBC,EAAgBC,IAAkD,CACjG,IAAMiB,EAAQ,IAAIC,EAClB,QAAWb,KAASW,EAClBC,EAAM,IAAIZ,EAAMQ,GAAKf,EAAS,KAAKC,EAAUc,CAAC,CAAC,CAAC,EAElD,OAAIb,IACE,MAAM,QAAQA,CAAW,EAC3BA,EAAY,KAAKiB,CAAK,EAEtBjB,EAAY,IAAIiB,CAAK,GAGlBA,CACT,CACF,CAfOR,EAAS,IAAAM,EAmBT,SAASI,EAAmBd,EAAkBe,EAAqCC,EAA0B,CAClH,OAAAD,EAAQC,CAAO,EACRhB,EAAMQ,GAAKO,EAAQP,CAAC,CAAC,CAC9B,CAHOJ,EAAS,gBAAAU,IAhCDV,IAAA,ICxDV,SAASa,EAAkBC,EAAqBC,EAAU,EAAGC,EAAsC,CACxG,IAAMC,EAAQ,WAAW,IAAM,CAC7BH,EAAQ,EACJE,GACFE,EAAW,QAAQ,CAEvB,EAAGH,CAAO,EACJG,EAAaC,EAAa,IAAM,CACpC,aAAaF,CAAK,CACpB,CAAC,EACD,OAAAD,GAAO,IAAIE,CAAU,EACdA,CACT,CCDO,IAAME,EAAN,cAA8BC,CAAW,CAa9C,YAA6BC,EAAqB,CAChD,MAAM,EADqB,eAAAA,EAN7B,KAAQ,mBAAqB,KAAK,UAAU,IAAIC,CAAmB,EACnE,KAAQ,uBAAyB,KAAK,UAAU,IAAIA,CAAmB,EAGvE,KAAQ,qBAAuB,EAI7B,KAAK,UAAUC,EAAa,IAAM,KAAK,mBAAmB,CAAC,CAAC,CAC9D,CAKO,gBAAuB,CACvB,KAAK,cACR,KAAK,YAAc,IAAI,MAAM,KAAK,UAAU,OAAO,OAAO,MAAM,EAChE,KAAK,uBAAuB,MAAQC,EAClC,KAAK,UAAU,WAAW,IAAM,KAAK,mBAAmB,CAAC,EACzD,KAAK,UAAU,aAAa,IAAM,KAAK,mBAAmB,CAAC,EAC3D,KAAK,UAAU,SAAS,IAAM,KAAK,mBAAmB,CAAC,CACzD,GAGF,KAAK,qBAAuB,KAAK,IAAI,EAChC,KAAK,mBAAmB,OAC3B,KAAK,2BAA2B,IAAkC,CAEtE,CAEQ,oBAA2B,CACjC,KAAK,YAAc,OACnB,KAAK,qBAAuB,EAC5B,KAAK,uBAAuB,MAAM,EAClC,KAAK,mBAAmB,MAAM,CAChC,CAEQ,2BAA2BC,EAAqB,CACtD,KAAK,mBAAmB,MAAQC,EAAkB,IAAM,CACtD,GAAI,CAAC,KAAK,YACR,OAGF,IAAMC,EADM,KAAK,IAAI,EACC,KAAK,qBAC3B,GAAIA,GAAW,KAAoC,CACjD,KAAK,mBAAmB,EACxB,MACF,CACA,KAAK,2BAA2B,KAAqCA,CAAO,CAC9E,EAAGF,CAAK,CACV,CAEO,iBAAiBG,EAAyC,CAC/D,OAAO,KAAK,cAAcA,CAAG,CAC/B,CAEO,eAAeA,EAAaC,EAA6B,CAC1D,KAAK,cACP,KAAK,YAAYD,CAAG,EAAIC,EAE5B,CAUO,oCAAoCC,EAAmBC,EAAoC,CAChG,IAAMC,EAAU,CAAC,EACXC,EAAc,CAAC,CAAC,EAIhBC,EAAe,KAAK,UAAU,OAAO,OAAO,OAC9CC,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQL,CAAS,EACzD,KAAOK,GAAM,CACX,IAAMC,EAAWN,EAAY,EAAII,EAAe,KAAK,UAAU,OAAO,OAAO,QAAQJ,EAAY,CAAC,EAAI,OAChGO,EAAkBD,EAAWA,EAAS,UAAY,GACpDE,EAASH,EAAK,kBAAkB,CAACE,GAAmBN,CAAS,EACjE,GAAIM,GAAmBD,EAAU,CAC/B,IAAMG,EAAWJ,EAAK,QAAQA,EAAK,OAAS,CAAC,EACtBI,GAAYA,EAAS,QAAQ,IAAM,GAAKA,EAAS,SAAS,IAAM,GAEjEH,EAAS,QAAQ,CAAC,GAAG,SAAS,IAAM,IACxDE,EAASA,EAAO,MAAM,EAAG,EAAE,EAE/B,CAEA,GADAN,EAAQ,KAAKM,CAAM,EACfD,EACFJ,EAAY,KAAKA,EAAYA,EAAY,OAAS,CAAC,EAAIK,EAAO,MAAM,MAEpE,OAEFR,IACAK,EAAOC,CACT,CACA,MAAO,CAACJ,EAAQ,KAAK,EAAE,EAAGC,CAAW,CACvC,CACF,EChIO,IAAMO,EAAN,KAAkB,CAOvB,IAAW,kBAAuC,CAChD,OAAO,KAAK,iBACd,CAKA,IAAW,iBAAiBC,EAA0B,CACpD,KAAK,kBAAoBA,CAC3B,CAKA,IAAW,mBAAgD,CACzD,OAAO,KAAK,kBACd,CAKA,IAAW,kBAAkBC,EAAqC,CAChE,KAAK,mBAAqBA,CAC5B,CAOO,kBAAkBD,EAAuB,CAC9C,MAAO,CAAC,EAAEA,GAAQA,EAAK,OAAS,EAClC,CAOO,iBAAiBE,EAAsC,CAC5D,OAAK,KAAK,mBAGLA,EAGD,KAAK,mBAAmB,gBAAkBA,EAAW,eAGrD,KAAK,mBAAmB,QAAUA,EAAW,OAG7C,KAAK,mBAAmB,YAAcA,EAAW,UAR5C,GAHA,EAeX,CAQO,yBAAyBF,EAAcC,EAAmC,CAC/E,OAAKA,GAAS,YAGP,KAAK,oBAAsB,QAC3BD,IAAS,KAAK,mBACd,KAAK,iBAAiBC,CAAO,EAJ3B,EAKX,CAKO,iBAAwB,CAC7B,KAAK,kBAAoB,MAC3B,CAKO,OAAc,CACnB,KAAK,kBAAoB,OACzB,KAAK,mBAAqB,MAC5B,CACF,EC9DO,IAAME,EAAN,KAAmB,CACxB,YACmBC,EACAC,EACjB,CAFiB,eAAAD,EACA,gBAAAC,CAChB,CAUI,KAAKC,EAAcC,EAAkBC,EAAkBC,EAA2D,CACvH,GAAI,CAACH,GAAQA,EAAK,SAAW,EAAG,CAC9B,KAAK,UAAU,eAAe,EAC9B,MACF,CACA,GAAIE,GAAY,KAAK,UAAU,KAC7B,MAAM,IAAI,MAAM,gBAAgBA,CAAQ,6BAA6B,KAAK,UAAU,IAAI,OAAO,EAGjG,KAAK,WAAW,eAAe,EAE/B,IAAME,EAAkC,CACtC,SAAAH,EACA,SAAAC,CACF,EAGIG,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EAEjE,GAAI,CAACE,EACH,QAASC,EAAIL,EAAW,EAAGK,EAAI,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,MAC7E,QAAK,6BAA6BA,CAAC,IAGvCF,EAAe,SAAWE,EAC1BF,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EACzDE,IAPmFC,IACvF,CAWJ,OAAOD,CACT,CASO,sBAAsBL,EAAcG,EAAgCI,EAAsD,CAC/H,GAAI,CAACP,GAAQA,EAAK,SAAW,EAAG,CAC9B,KAAK,UAAU,eAAe,EAC9B,MACF,CAEA,IAAMQ,EAAkB,KAAK,UAAU,qBAAqB,EAC5D,KAAK,UAAU,eAAe,EAE9B,IAAIN,EAAW,EACXD,EAAW,EACXO,IACED,IAAqBP,GACvBE,EAAWM,EAAgB,IAAI,EAC/BP,EAAWO,EAAgB,IAAI,IAE/BN,EAAWM,EAAgB,MAAM,EACjCP,EAAWO,EAAgB,MAAM,IAIrC,KAAK,WAAW,eAAe,EAE/B,IAAMJ,EAAkC,CACtC,SAAAH,EACA,SAAAC,CACF,EAGIG,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EAEjE,GAAI,CAACE,EACH,QAASC,EAAIL,EAAW,EAAGK,EAAI,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,MAC7E,QAAK,6BAA6BA,CAAC,IAGvCF,EAAe,SAAWE,EAC1BF,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EACzDE,IAPmFC,IACvF,CAYJ,GAAI,CAACD,GAAUJ,IAAa,EAC1B,QAASK,EAAI,EAAGA,EAAIL,GAGd,IAAAK,EAAI,GAAK,KAAK,6BAA6BA,CAAC,KAGhDF,EAAe,SAAWE,EAC1BF,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,EACzDE,IATwBC,IAG5B,CAaJ,MAAI,CAACD,GAAUG,IACbJ,EAAe,SAAWI,EAAgB,MAAM,EAChDJ,EAAe,SAAW,EAC1BC,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,CAAa,GAGxDE,CACT,CASO,0BAA0BL,EAAcG,EAAgCI,EAAsD,CACnI,GAAI,CAACP,GAAQA,EAAK,SAAW,EAAG,CAC9B,KAAK,UAAU,eAAe,EAC9B,MACF,CAEA,IAAMQ,EAAkB,KAAK,UAAU,qBAAqB,EAC5D,KAAK,UAAU,eAAe,EAE9B,IAAIP,EAAW,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,KAAO,EACpEC,EAAW,KAAK,UAAU,KAC1BO,EAAkB,GAExB,KAAK,WAAW,eAAe,EAC/B,IAAML,EAAkC,CACtC,SAAAH,EACA,SAAAC,CACF,EAEIG,EAkBJ,GAjBIG,IACFJ,EAAe,SAAWH,EAAWO,EAAgB,MAAM,EAC3DJ,EAAe,SAAWI,EAAgB,MAAM,EAC5CD,IAAqBP,IAEvBK,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,EAAe,EAAK,EAC/DE,IAEHD,EAAe,SAAWH,EAAWO,EAAgB,IAAI,EACzDJ,EAAe,SAAWI,EAAgB,IAAI,KAKpDH,IAAW,KAAK,YAAYL,EAAMI,EAAgBD,EAAeM,CAAe,EAG5E,CAACJ,EAAQ,CACXD,EAAe,SAAW,KAAK,IAAIA,EAAe,SAAU,KAAK,UAAU,IAAI,EAC/E,QAASE,EAAIL,EAAW,EAAGK,GAAK,IAC9BF,EAAe,SAAWE,EAC1BD,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,EAAeM,CAAe,EAC1E,CAAAJ,GAH6BC,IAGjC,CAIJ,CAEA,GAAI,CAACD,GAAUJ,IAAc,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,KAAO,EACtF,QAASK,EAAK,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,KAAO,EAAIA,GAAKL,IAChFG,EAAe,SAAWE,EAC1BD,EAAS,KAAK,YAAYL,EAAMI,EAAgBD,EAAeM,CAAe,EAC1E,CAAAJ,GAHsFC,IAG1F,CAMJ,OAAOD,CACT,CASQ,aAAaK,EAAqBC,EAAcX,EAAuB,CAC7E,OAASU,IAAgB,GAAO,qCAA8B,SAASC,EAAKD,EAAc,CAAC,CAAC,KACvFA,EAAcV,EAAK,SAAYW,EAAK,QAAY,qCAA8B,SAASA,EAAKD,EAAcV,EAAK,MAAM,CAAC,EAC7H,CAGQ,oBAAoBU,EAAqBC,EAAcX,EAAcG,EAAwC,CACnH,MAAO,CAACA,EAAc,WAAa,KAAK,aAAaO,EAAaC,EAAMX,CAAI,CAC9E,CASQ,6BAA6BY,EAAsB,CACzD,OAAO,KAAK,UAAU,OAAO,OAAO,QAAQA,CAAG,GAAG,YAAc,EAClE,CAcQ,YAAYZ,EAAcI,EAAiCD,EAAgC,CAAC,EAAGM,EAA2B,GAAkC,CAElK,GAAIA,GAGF,GAAIL,EAAe,SAAW,GAAK,KAAK,UAAU,OAAO,OAAO,QAAQA,EAAe,QAAQ,GAAG,UAAW,CAC3GA,EAAe,UAAY,KAAK,UAAU,KAC1C,MACF,MAKA,MAAOA,EAAe,SAAW,GAAK,KAAK,UAAU,OAAO,OAAO,QAAQA,EAAe,QAAQ,GAAG,WACnGA,EAAe,WACfA,EAAe,UAAY,KAAK,UAAU,KAG9C,IAAMQ,EAAMR,EAAe,SACrBS,EAAMT,EAAe,SAEvBU,EAAQ,KAAK,WAAW,iBAAiBF,CAAG,EAC3CE,IACHA,EAAQ,KAAK,WAAW,oCAAoCF,EAAK,EAAI,EACrE,KAAK,WAAW,eAAeA,EAAKE,CAAK,GAE3C,GAAM,CAACC,EAAYC,CAAO,EAAIF,EAExBG,EAAS,KAAK,0BAA0BL,EAAKC,EAAKG,CAAO,EAC3DE,EAAalB,EACbmB,EAAmBJ,EAClBZ,EAAc,QACjBe,EAAaf,EAAc,cAAgBH,EAAOA,EAAK,YAAY,EACnEmB,EAAmBhB,EAAc,cAAgBY,EAAaA,EAAW,YAAY,GAGvF,IAAIK,EAAc,GAClB,GAAIjB,EAAc,MAAO,CACvB,IAAMkB,EAAc,OAAOH,EAAYf,EAAc,cAAgB,IAAM,IAAI,EAC3EmB,EACJ,GAAIb,EAEF,KAAOa,EAAYD,EAAY,KAAKF,EAAiB,MAAM,EAAGF,CAAM,CAAC,GAAG,CACtE,IAAMM,EAAaF,EAAY,UAAYC,EAAU,CAAC,EAAE,OACpDA,EAAU,CAAC,EAAE,OAAS,GAAK,KAAK,oBAAoBC,EAAYJ,EAAkBG,EAAU,CAAC,EAAGnB,CAAa,IAC/GiB,EAAcG,EACdvB,EAAOsB,EAAU,CAAC,GAEpBD,EAAY,UAAYE,EAAa,CACvC,KAOA,KADAF,EAAY,UAAYJ,EACjBK,EAAYD,EAAY,KAAKF,CAAgB,GAAG,CACrD,IAAMI,EAAaF,EAAY,UAAYC,EAAU,CAAC,EAAE,OACxD,GAAIA,EAAU,CAAC,EAAE,OAAS,GAAK,KAAK,oBAAoBC,EAAYJ,EAAkBG,EAAU,CAAC,EAAGnB,CAAa,EAAG,CAClHiB,EAAcG,EACdvB,EAAOsB,EAAU,CAAC,EAClB,KACF,CAEAD,EAAY,UAAYE,EAAa,CACvC,CAEJ,SAAWd,EAAiB,CAC1B,IAAIc,EAAaN,EAASC,EAAW,QAAU,EAAIC,EAAiB,YAAYD,EAAYD,EAASC,EAAW,MAAM,EAAI,GAE1H,KAAOK,GAAc,GAAK,CAAC,KAAK,oBAAoBA,EAAYJ,EAAkBD,EAAYf,CAAa,GACzGoB,EAAaA,EAAa,EAAIJ,EAAiB,YAAYD,EAAYK,EAAa,CAAC,EAAI,GAE3FH,EAAcG,CAChB,KAAO,CACL,IAAIA,EAAaJ,EAAiB,QAAQD,EAAYD,CAAM,EAC5D,KAAOM,GAAc,GAAK,CAAC,KAAK,oBAAoBA,EAAYJ,EAAkBD,EAAYf,CAAa,GACzGoB,EAAaJ,EAAiB,QAAQD,EAAYK,EAAa,CAAC,EAElEH,EAAcG,CAChB,CAEA,GAAIH,GAAe,EAAG,CAGpB,IAAII,EAAiB,EACrB,KAAOA,EAAiBR,EAAQ,OAAS,GAAKI,GAAeJ,EAAQQ,EAAiB,CAAC,GACrFA,IAEF,IAAIC,EAAeD,EACnB,KAAOC,EAAeT,EAAQ,OAAS,GAAKI,EAAcpB,EAAK,QAAUgB,EAAQS,EAAe,CAAC,GAC/FA,IAEF,IAAMC,EAAiBN,EAAcJ,EAAQQ,CAAc,EACrDG,EAAeP,EAAcpB,EAAK,OAASgB,EAAQS,CAAY,EAC/DG,EAAgB,KAAK,0BAA0BhB,EAAMY,EAAgBE,CAAc,EAEnFG,EADc,KAAK,0BAA0BjB,EAAMa,EAAcE,CAAY,EACxDC,EAAgB,KAAK,UAAU,MAAQH,EAAeD,GAEjF,MAAO,CACL,KAAAxB,EACA,IAAK4B,EACL,IAAKhB,EAAMY,EACX,KAAAK,CACF,CACF,CACF,CAEQ,0BAA0BjB,EAAaK,EAAwB,CACrE,IAAMN,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQC,CAAG,EACrD,GAAI,CAACD,EACH,MAAO,GAET,QAASmB,EAAI,EAAGA,EAAIb,EAAQa,IAAK,CAC/B,IAAMC,EAAOpB,EAAK,QAAQmB,CAAC,EAC3B,GAAI,CAACC,EACH,MAGF,IAAMC,EAAOD,EAAK,SAAS,EACvBC,EAAK,OAAS,IAChBf,GAAUe,EAAK,OAAS,GAI1B,IAAMC,EAAWtB,EAAK,QAAQmB,EAAI,CAAC,EAC/BG,GAAYA,EAAS,SAAS,IAAM,GACtChB,GAEJ,CACA,OAAOA,CACT,CAUQ,0BAA0BhB,EAAkBiC,EAAcC,EAA+B,CAC/F,IAAMC,EAAW,KAAK,IAAI,KAAK,MAAMF,EAAO,KAAK,UAAU,IAAI,EAAGC,EAAY,OAAS,CAAC,EACpFlB,EAASkB,EAAYC,CAAQ,EAC3BzB,EAAO,KAAK,UAAU,OAAO,OAAO,QAAQV,EAAWmC,CAAQ,EACrE,GAAIzB,EAAM,CACR,IAAM0B,EAAY,KAAK,IAAIH,EAAOE,EAAW,KAAK,UAAU,KAAM,KAAK,UAAU,IAAI,EACrF,QAASN,EAAI,EAAGA,EAAIO,EAAWP,IAAK,CAClC,IAAMC,EAAOpB,EAAK,QAAQmB,CAAC,EAC3B,GAAI,CAACC,EACH,MAEEA,EAAK,SAAS,IAEhBd,GAAUc,EAAK,QAAQ,IAAM,EAAI,EAAIA,EAAK,SAAS,EAAE,OAEzD,CACF,CACA,OAAOd,CACT,CACF,ECxZO,IAAMqB,EAAN,cAAgCC,CAAW,CAIhD,YAA6BC,EAAqB,CAChD,MAAM,EADqB,eAAAA,EAH7B,KAAQ,sBAAsC,CAAC,EAC/C,KAAQ,kBAAiC,IAAI,IAI3C,KAAK,UAAUC,EAAa,IAAM,KAAK,0BAA0B,CAAC,CAAC,CACrE,CAOO,2BAA2BC,EAA0BC,EAAyC,CACnG,KAAK,0BAA0B,EAE/B,QAAWC,KAASF,EAAS,CAC3B,IAAMG,EAAc,KAAK,yBAAyBD,EAAOD,EAAS,EAAK,EACvE,GAAIE,EACF,QAAWC,KAAcD,EACvB,KAAK,iBAAiBC,EAAYF,CAAK,CAG7C,CACF,CAQO,uBAAuBG,EAAuBJ,EAAgE,CACnH,IAAME,EAAc,KAAK,yBAAyBE,EAAQJ,EAAS,EAAI,EACvE,GAAIE,EACF,MAAO,CAAE,YAAAA,EAAa,MAAOE,EAAQ,SAAU,CAAEC,EAAQH,CAAW,CAAG,CAAE,CAG7E,CAKO,2BAAkC,CACvCG,EAAQ,KAAK,qBAAqB,EAClC,KAAK,sBAAwB,CAAC,EAC9B,KAAK,kBAAkB,MAAM,CAC/B,CAOQ,iBAAiBF,EAAyBF,EAA4B,CAC5E,KAAK,kBAAkB,IAAIE,EAAW,OAAO,IAAI,EACjD,KAAK,sBAAsB,KAAK,CAAE,WAAAA,EAAY,MAAAF,EAAO,SAAU,CAAEE,EAAW,QAAQ,CAAG,CAAE,CAAC,CAC5F,CAQQ,aAAaG,EAAsBC,EAAiCC,EAA+B,CACpGF,EAAQ,UAAU,SAAS,8BAA8B,IAC5DA,EAAQ,UAAU,IAAI,8BAA8B,EAChDC,IACFD,EAAQ,MAAM,QAAU,aAAaC,CAAW,KAGhDC,GACFF,EAAQ,UAAU,IAAI,qCAAqC,CAE/D,CASQ,yBAAyBF,EAAuBJ,EAAmCQ,EAAoD,CAE7I,IAAMC,EAA+C,CAAC,EAClDC,EAAaN,EAAO,IACpBO,EAAgBP,EAAO,KACvBQ,EAAe,CAAC,KAAK,UAAU,OAAO,OAAO,MAAQ,KAAK,UAAU,OAAO,OAAO,QAAUR,EAAO,IACvG,KAAOO,EAAgB,GAAG,CACxB,IAAME,EAAgB,KAAK,IAAI,KAAK,UAAU,KAAOH,EAAYC,CAAa,EAC9EF,EAAiB,KAAK,CAACG,EAAcF,EAAYG,CAAa,CAAC,EAC/DH,EAAa,EACbC,GAAiBE,EACjBD,GACF,CAGA,IAAMV,EAA6B,CAAC,EACpC,QAAWY,KAASL,EAAkB,CACpC,IAAMM,EAAS,KAAK,UAAU,eAAeD,EAAM,CAAC,CAAC,EAC/CX,EAAa,KAAK,UAAU,mBAAmB,CACnD,OAAAY,EACA,EAAGD,EAAM,CAAC,EACV,MAAOA,EAAM,CAAC,EACd,MAAON,EAAiB,MAAQ,SAChC,gBAAiBA,EAAiBR,EAAQ,sBAAwBA,EAAQ,gBAC1E,qBAAsB,KAAK,kBAAkB,IAAIe,EAAO,IAAI,EAAI,OAAY,CAC1E,MAAOP,EAAiBR,EAAQ,8BAAgCA,EAAQ,mBACxE,SAAU,QACZ,CACF,CAAC,EACD,GAAIG,EAAY,CACd,IAAMa,EAA6B,CAAC,EACpCA,EAAY,KAAKD,CAAM,EACvBC,EAAY,KAAKb,EAAW,SAAUc,GAAM,KAAK,aAAaA,EAAGT,EAAiBR,EAAQ,kBAAoBA,EAAQ,YAAa,EAAK,CAAC,CAAC,EAC1IgB,EAAY,KAAKb,EAAW,UAAU,IAAME,EAAQW,CAAW,CAAC,CAAC,EACjEd,EAAY,KAAKC,CAAU,CAC7B,CACF,CAEA,OAAOD,EAAY,SAAW,EAAI,OAAYA,CAChD,CACF,ECrIO,IAAMgB,EAAN,cAAkCC,CAAW,CAA7C,kCACL,KAAQ,eAAkC,CAAC,EAG3C,KAAiB,oBAAsB,KAAK,UAAU,IAAIC,CAAmC,EAC7F,IAAW,oBAAuD,CAAE,OAAO,KAAK,oBAAoB,KAAO,CAK3G,IAAW,eAA8C,CACvD,OAAO,KAAK,cACd,CAKA,IAAW,oBAAsD,CAC/D,OAAO,KAAK,mBACd,CAKA,IAAW,mBAAmBC,EAA6C,CACzE,KAAK,oBAAsBA,CAC7B,CAOO,cAAcC,EAA0BC,EAA0B,CACvE,KAAK,eAAiBD,EAAQ,MAAM,EAAGC,CAAU,CACnD,CAKO,cAAqB,CAC1B,KAAK,eAAiB,CAAC,CACzB,CAKO,yBAAgC,CACjC,KAAK,sBACP,KAAK,oBAAoB,QAAQ,EACjC,KAAK,oBAAsB,OAE/B,CAOO,gBAAgBC,EAA+B,CACpD,QAASC,EAAI,EAAGA,EAAI,KAAK,eAAe,OAAQA,IAAK,CACnD,IAAMC,EAAQ,KAAK,eAAeD,CAAC,EACnC,GAAIC,EAAM,MAAQF,EAAO,KAAOE,EAAM,MAAQF,EAAO,KAAOE,EAAM,OAASF,EAAO,KAChF,OAAOC,CAEX,CACA,MAAO,EACT,CAMO,mBAAmBE,EAA+B,CACvD,GAAI,CAACA,EACH,OAGF,IAAIC,EAAc,GACd,KAAK,sBACPA,EAAc,KAAK,gBAAgB,KAAK,oBAAoB,KAAK,GAGnE,KAAK,oBAAoB,KAAK,CAC5B,YAAAA,EACA,YAAa,KAAK,eAAe,MACnC,CAAC,CACH,CAKO,OAAc,CACnB,KAAK,wBAAwB,EAC7B,KAAK,aAAa,CACpB,CACF,ECtFO,IAAMC,EAAN,cAA0BC,CAAiD,CAqBhF,YAAYC,EAAwC,CAClD,MAAM,EAnBR,KAAQ,kBAAoB,KAAK,UAAU,IAAIC,CAAgC,EAC/E,KAAQ,WAAa,KAAK,UAAU,IAAIA,CAAoC,EAG5E,KAAQ,OAAS,IAAIC,EAGrB,KAAQ,eAAiB,KAAK,UAAU,IAAIC,CAAqB,EAEjE,KAAiB,eAAiB,KAAK,UAAU,IAAIC,CAAe,EACpE,KAAgB,cAAgB,KAAK,eAAe,MACpD,KAAiB,gBAAkB,KAAK,UAAU,IAAIA,CAAe,EACrE,KAAgB,eAAiB,KAAK,gBAAgB,MASpD,KAAK,gBAAkBJ,GAAS,gBAAkB,GACpD,CARA,IAAW,oBAAuD,CAChE,OAAO,KAAK,eAAe,kBAC7B,CAQO,SAASK,EAA0B,CACxC,KAAK,UAAYA,EACjB,KAAK,WAAW,MAAQ,IAAIC,EAAgBD,CAAQ,EACpD,KAAK,QAAU,IAAIE,EAAaF,EAAU,KAAK,WAAW,KAAK,EAC/D,KAAK,mBAAqB,IAAIG,EAAkBH,CAAQ,EACxD,KAAK,UAAU,KAAK,UAAU,cAAc,IAAM,KAAK,eAAe,CAAC,CAAC,EACxE,KAAK,UAAU,KAAK,UAAU,SAAS,IAAM,KAAK,eAAe,CAAC,CAAC,EACnE,KAAK,UAAUI,EAAa,IAAM,KAAK,iBAAiB,CAAC,CAAC,CAC5D,CAEQ,gBAAuB,CAC7B,KAAK,kBAAkB,MAAM,EACzB,KAAK,OAAO,kBAAoB,KAAK,OAAO,mBAAmB,cACjE,KAAK,kBAAkB,MAAQC,EAAkB,IAAM,CACrD,IAAMC,EAAO,KAAK,OAAO,iBACzB,KAAK,OAAO,gBAAgB,EAC5B,KAAK,aAAaA,EAAO,CAAE,GAAG,KAAK,OAAO,kBAAmB,YAAa,EAAK,EAAG,CAAE,SAAU,EAAK,CAAC,CACtG,EAAG,GAAG,EAEV,CAEO,iBAAiBC,EAAwC,CAC9D,KAAK,eAAe,wBAAwB,EAC5C,KAAK,oBAAoB,0BAA0B,EACnD,KAAK,eAAe,aAAa,EAC5BA,GACH,KAAK,OAAO,gBAAgB,CAEhC,CAEO,uBAA8B,CACnC,KAAK,eAAe,wBAAwB,CAC9C,CASO,SAASD,EAAcE,EAAgCC,EAAyD,CACrH,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAM,IAAI,MAAM,2CAA2C,EAG7D,KAAK,gBAAgB,KAAK,EAE1B,KAAK,OAAO,kBAAoBD,EAE5B,KAAK,OAAO,yBAAyBF,EAAME,CAAa,GAC1D,KAAK,qBAAqBF,EAAME,CAAc,EAGhD,IAAME,EAAQ,KAAK,mBAAmBJ,EAAME,EAAeC,CAAqB,EAChF,YAAK,aAAaD,CAAa,EAC/B,KAAK,OAAO,iBAAmBF,EAE/B,KAAK,eAAe,KAAK,EAElBI,CACT,CAEQ,qBAAqBJ,EAAcE,EAAqC,CAC9E,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,SAAW,CAAC,KAAK,mBAC5C,MAAM,IAAI,MAAM,2CAA2C,EAE7D,GAAI,CAAC,KAAK,OAAO,kBAAkBF,CAAI,EAAG,CACxC,KAAK,iBAAiB,EACtB,MACF,CAGA,KAAK,iBAAiB,EAAI,EAE1B,IAAMK,EAA2B,CAAC,EAC9BC,EACAC,EAAS,KAAK,QAAQ,KAAKP,EAAM,EAAG,EAAGE,CAAa,EAExD,KAAOK,IAAWD,GAAY,MAAQC,EAAO,KAAOD,GAAY,MAAQC,EAAO,MACzE,EAAAF,EAAQ,QAAU,KAAK,kBADwD,CAInFC,EAAaC,EACbF,EAAQ,KAAKC,CAAU,EACvB,IAAME,EAAO,KAAK,UAAU,KACxBC,EAAUH,EAAW,IAAMA,EAAW,KACtCI,EAAUJ,EAAW,IACrBG,GAAWD,IACbE,GAAW,KAAK,MAAMD,EAAUD,CAAI,EACpCC,EAAUA,EAAUD,GAEtBD,EAAS,KAAK,QAAQ,KAAKP,EAAMU,EAASD,EAASP,CAAa,CAClE,CAEA,KAAK,eAAe,cAAcG,EAAS,KAAK,eAAe,EAC3DH,EAAc,aAChB,KAAK,mBAAmB,2BAA2BG,EAASH,EAAc,WAAW,CAEzF,CAEQ,mBAAmBF,EAAcE,EAAgCC,EAAyD,CAChI,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAO,GAET,GAAI,CAAC,KAAK,OAAO,kBAAkBH,CAAI,EACrC,YAAK,UAAU,eAAe,EAC9B,KAAK,iBAAiB,EACf,GAGT,IAAMO,EAAS,KAAK,QAAQ,sBAAsBP,EAAME,EAAe,KAAK,OAAO,gBAAgB,EACnG,OAAO,KAAK,cAAcK,EAAQL,GAAe,YAAaC,GAAuB,QAAQ,CAC/F,CASO,aAAaH,EAAcE,EAAgCC,EAAyD,CACzH,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAM,IAAI,MAAM,2CAA2C,EAG7D,KAAK,gBAAgB,KAAK,EAE1B,KAAK,OAAO,kBAAoBD,EAE5B,KAAK,OAAO,yBAAyBF,EAAME,CAAa,GAC1D,KAAK,qBAAqBF,EAAME,CAAc,EAGhD,IAAME,EAAQ,KAAK,uBAAuBJ,EAAME,EAAeC,CAAqB,EACpF,YAAK,aAAaD,CAAa,EAC/B,KAAK,OAAO,iBAAmBF,EAE/B,KAAK,eAAe,KAAK,EAElBI,CACT,CAEQ,aAAaF,EAAsC,CACzD,KAAK,eAAe,mBAAmB,CAAC,CAACA,GAAe,WAAW,CACrE,CAEQ,uBAAuBF,EAAcE,EAAgCC,EAAyD,CACpI,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,QAC3B,MAAO,GAET,GAAI,CAAC,KAAK,OAAO,kBAAkBH,CAAI,EACrC,YAAK,UAAU,eAAe,EAC9B,KAAK,iBAAiB,EACf,GAGT,IAAMO,EAAS,KAAK,QAAQ,0BAA0BP,EAAME,EAAe,KAAK,OAAO,gBAAgB,EACvG,OAAO,KAAK,cAAcK,EAAQL,GAAe,YAAaC,GAAuB,QAAQ,CAC/F,CAOQ,cAAcI,EAAmClB,EAAoCsB,EAA6B,CACxH,GAAI,CAAC,KAAK,WAAa,CAAC,KAAK,mBAC3B,MAAO,GAIT,GADA,KAAK,eAAe,wBAAwB,EACxC,CAACJ,EACH,YAAK,UAAU,eAAe,EACvB,GAIT,GADA,KAAK,UAAU,OAAOA,EAAO,IAAKA,EAAO,IAAKA,EAAO,IAAI,EACrDlB,EAAS,CACX,IAAMuB,EAAmB,KAAK,mBAAmB,uBAAuBL,EAAQlB,CAAO,EACnFuB,IACF,KAAK,eAAe,mBAAqBA,EAE7C,CAEA,GAAI,CAACD,IAECJ,EAAO,KAAQ,KAAK,UAAU,OAAO,OAAO,UAAY,KAAK,UAAU,MAASA,EAAO,IAAM,KAAK,UAAU,OAAO,OAAO,WAAW,CACvI,IAAIM,EAASN,EAAO,IAAM,KAAK,UAAU,OAAO,OAAO,UACvDM,GAAU,KAAK,MAAM,KAAK,UAAU,KAAO,CAAC,EAC5C,KAAK,UAAU,YAAYA,CAAM,CACnC,CAEF,MAAO,EACT,CACF", ++ "names": ["toDisposable", "fn", "dispose", "arg", "d", "combinedDisposable", "disposables", "DisposableStore", "o", "Disposable", "MutableDisposable", "value", "Emitter", "listener", "thisArgs", "disposables", "toDisposable", "entry", "result", "idx", "event", "listeners", "i", "len", "EventUtils", "forward", "from", "to", "e", "map", "any", "events", "store", "DisposableStore", "runAndSubscribe", "handler", "initial", "disposableTimeout", "handler", "timeout", "store", "timer", "disposable", "toDisposable", "SearchLineCache", "Disposable", "_terminal", "MutableDisposable", "toDisposable", "combinedDisposable", "delay", "disposableTimeout", "elapsed", "row", "entry", "lineIndex", "trimRight", "strings", "lineOffsets", "bufferLength", "line", "nextLine", "lineWrapsToNext", "string", "lastCell", "SearchState", "term", "options", "newOptions", "SearchEngine", "_terminal", "_lineCache", "term", "startRow", "startCol", "searchOptions", "searchPosition", "result", "y", "cachedSearchTerm", "prevSelectedPos", "isReverseSearch", "searchIndex", "line", "row", "col", "cache", "stringLine", "offsets", "offset", "searchTerm", "searchStringLine", "resultIndex", "searchRegex", "foundTerm", "matchIndex", "startRowOffset", "endRowOffset", "startColOffset", "endColOffset", "startColIndex", "size", "i", "cell", "char", "nextCell", "cols", "lineOffsets", "rowsBack", "colsInRow", "DecorationManager", "Disposable", "_terminal", "toDisposable", "results", "options", "match", "decorations", "decoration", "result", "dispose", "element", "borderColor", "isActiveResult", "decorationRanges", "currentCol", "remainingSize", "markerOffset", "amountThisRow", "range", "marker", "disposables", "e", "SearchResultTracker", "Disposable", "Emitter", "decoration", "results", "maxResults", "result", "i", "match", "hasDecorations", "resultIndex", "SearchAddon", "Disposable", "options", "MutableDisposable", "SearchState", "SearchResultTracker", "Emitter", "terminal", "SearchLineCache", "SearchEngine", "DecorationManager", "toDisposable", "disposableTimeout", "term", "retainCachedSearchTerm", "searchOptions", "internalSearchOptions", "found", "results", "prevResult", "result", "cols", "nextCol", "nextRow", "noScroll", "activeDecoration", "scroll"] + } +diff --git a/src/SearchEngine.ts b/src/SearchEngine.ts +index 1760bc2bd1fd274d23e2032fde631b39c739f0d9..5b3c5cc5e861356b87e8a15c55797f45bac20a5c 100644 +--- a/src/SearchEngine.ts ++++ b/src/SearchEngine.ts +@@ -76,6 +76,9 @@ export class SearchEngine { + // Search from startRow + 1 to end + if (!result) { + for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) { ++ if (this._isRowCoveredByEarlierSearch(y)) { ++ continue; ++ } + searchPosition.startRow = y; + searchPosition.startCol = 0; + result = this._findInLine(term, searchPosition, searchOptions); +@@ -127,6 +130,9 @@ export class SearchEngine { + // Search from startRow + 1 to end + if (!result) { + for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) { ++ if (this._isRowCoveredByEarlierSearch(y)) { ++ continue; ++ } + searchPosition.startRow = y; + searchPosition.startCol = 0; + result = this._findInLine(term, searchPosition, searchOptions); +@@ -138,6 +144,11 @@ export class SearchEngine { + // If we hit the bottom and didn't search from the very top wrap back up + if (!result && startRow !== 0) { + for (let y = 0; y < startRow; y++) { ++ // Row 0 is never skipped: it can be a continuation whose line start was trimmed from the ++ // scrollback, and nothing earlier in this loop has searched it. ++ if (y > 0 && this._isRowCoveredByEarlierSearch(y)) { ++ continue; ++ } + searchPosition.startRow = y; + searchPosition.startCol = 0; + result = this._findInLine(term, searchPosition, searchOptions); +@@ -237,6 +248,22 @@ export class SearchEngine { + (((searchIndex + term.length) === line.length) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex + term.length]))); + } + ++ /** `_isWholeWord` gated on the option, so a rejected hit can be stepped past instead of ending the scan. */ ++ private _satisfiesWholeWord(searchIndex: number, line: string, term: string, searchOptions: ISearchOptions): boolean { ++ return !searchOptions.wholeWord || this._isWholeWord(searchIndex, line, term); ++ } ++ ++ /** ++ * Whether an earlier `_findInLine` in this same call already scanned this row's line from an ++ * equal or lower offset, which makes rescanning it pure O(rows^2) work on one long line. Sound ++ * for every option because `_findInLine` returns the first accepted match at or after its ++ * offset, which is monotone in that offset. Only valid once such a search has happened — the ++ * wrap-around loop starts at row 0, whose line start may have been trimmed from the scrollback. ++ */ ++ private _isRowCoveredByEarlierSearch(row: number): boolean { ++ return this._terminal.buffer.active.getLine(row)?.isWrapped === true; ++ } ++ + /** + * Searches a line for a search term. Takes the provided terminal line and searches the text line, + * which may contain subsequent terminal lines if the text is wrapped. If the provided line number +@@ -250,23 +277,26 @@ export class SearchEngine { + * @returns The search result if it was found. + */ + private _findInLine(term: string, searchPosition: ISearchPosition, searchOptions: ISearchOptions = {}, isReverseSearch: boolean = false): ISearchResult | undefined { +- const row = searchPosition.startRow; +- const col = searchPosition.startCol; +- + // Ignore wrapped lines, only consider on unwrapped line (first row of command string). +- const firstLine = this._terminal.buffer.active.getLine(row); +- if (firstLine?.isWrapped) { +- if (isReverseSearch) { ++ if (isReverseSearch) { ++ // Reverse search never rewinds: its caller carries startCol down the rows of the line. Row 0 ++ // is searched even when wrapped, since its line start may have been trimmed from the scrollback. ++ if (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) { + searchPosition.startCol += this._terminal.cols; + return; + } +- +- // This will iterate until we find the line start. +- // When we find it, we will search using the calculated start column. +- searchPosition.startRow--; +- searchPosition.startCol += this._terminal.cols; +- return this._findInLine(term, searchPosition, searchOptions); ++ } else { ++ // A loop rather than recursion: one frame per wrapped row overflows the stack on a line long ++ // enough to fill the scrollback. Bounded at row 0 because after a reflow the buffer's ring ++ // holds stale entries at negative indices, so `getLine(-1)` answers with a wrapped line. ++ while (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) { ++ searchPosition.startRow--; ++ searchPosition.startCol += this._terminal.cols; ++ } + } ++ const row = searchPosition.startRow; ++ const col = searchPosition.startCol; ++ + let cache = this._lineCache.getLineFromCache(row); + if (!cache) { + cache = this._lineCache.translateBufferLineToStringWithWrap(row, true); +@@ -274,7 +304,7 @@ export class SearchEngine { + } + const [stringLine, offsets] = cache; + +- const offset = this._bufferColsToStringOffset(row, col); ++ const offset = this._bufferColsToStringOffset(row, col, offsets); + let searchTerm = term; + let searchStringLine = stringLine; + if (!searchOptions.regex) { +@@ -289,32 +319,46 @@ export class SearchEngine { + if (isReverseSearch) { + // This loop will get the resultIndex of the _last_ regex match in the range 0..offset + while (foundTerm = searchRegex.exec(searchStringLine.slice(0, offset))) { +- resultIndex = searchRegex.lastIndex - foundTerm[0].length; +- term = foundTerm[0]; +- searchRegex.lastIndex -= (term.length - 1); ++ const matchIndex = searchRegex.lastIndex - foundTerm[0].length; ++ if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) { ++ resultIndex = matchIndex; ++ term = foundTerm[0]; ++ } ++ searchRegex.lastIndex = matchIndex + 1; + } + } else { +- foundTerm = searchRegex.exec(searchStringLine.slice(offset)); +- if (foundTerm && foundTerm[0].length > 0) { +- resultIndex = offset + (searchRegex.lastIndex - foundTerm[0].length); +- term = foundTerm[0]; ++ // Driven over the whole line from `offset` rather than over `slice(offset)`: a slice ++ // re-anchors ^ and \b at whatever column the row happened to wrap at, and only ++ // first-accepted-match-at-or-after-offset is monotone in `offset`, which is what lets ++ // `_isRowCoveredByEarlierSearch` skip a wrapped row an earlier scan already covered. ++ searchRegex.lastIndex = offset; ++ while (foundTerm = searchRegex.exec(searchStringLine)) { ++ const matchIndex = searchRegex.lastIndex - foundTerm[0].length; ++ if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) { ++ resultIndex = matchIndex; ++ term = foundTerm[0]; ++ break; ++ } ++ // A zero-length or rejected match would otherwise repeat forever. ++ searchRegex.lastIndex = matchIndex + 1; + } + } ++ } else if (isReverseSearch) { ++ let matchIndex = offset - searchTerm.length >= 0 ? searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length) : -1; ++ // `lastIndexOf` clamps a negative fromIndex to 0, so index 0 has to end the walk. ++ while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) { ++ matchIndex = matchIndex > 0 ? searchStringLine.lastIndexOf(searchTerm, matchIndex - 1) : -1; ++ } ++ resultIndex = matchIndex; + } else { +- if (isReverseSearch) { +- if (offset - searchTerm.length >= 0) { +- resultIndex = searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length); +- } +- } else { +- resultIndex = searchStringLine.indexOf(searchTerm, offset); ++ let matchIndex = searchStringLine.indexOf(searchTerm, offset); ++ while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) { ++ matchIndex = searchStringLine.indexOf(searchTerm, matchIndex + 1); + } ++ resultIndex = matchIndex; + } + + if (resultIndex >= 0) { +- if (searchOptions.wholeWord && !this._isWholeWord(resultIndex, searchStringLine, term)) { +- return; +- } +- + // Adjust the row number and search index if needed since a "line" of text can span multiple + // rows + let startRowOffset = 0; +@@ -365,12 +409,21 @@ export class SearchEngine { + return offset; + } + +- private _bufferColsToStringOffset(startRow: number, cols: number): number { +- let lineIndex = startRow; +- let offset = 0; +- let line = this._terminal.buffer.active.getLine(lineIndex); +- while (cols > 0 && line) { +- for (let i = 0; i < cols && i < this._terminal.cols; i++) { ++ /** ++ * `cols` counts from the start of the logical line, so summing the cells of every row before the ++ * resume point costs O(line) per call and the highlight-all pass makes one call per match. ++ * `lineOffsets` already holds the string offset each wrapped row starts at — the same map used ++ * above to turn a match index back into a row — so only the last, partial row needs cells. It is ++ * also the map the row a match lands on is read from, which the cell sum disagreed with by one ++ * for a row whose trailing cell is the null placeholder of a wide character that wrapped. ++ */ ++ private _bufferColsToStringOffset(startRow: number, cols: number, lineOffsets: number[]): number { ++ const rowsBack = Math.min(Math.floor(cols / this._terminal.cols), lineOffsets.length - 1); ++ let offset = lineOffsets[rowsBack]; ++ const line = this._terminal.buffer.active.getLine(startRow + rowsBack); ++ if (line) { ++ const colsInRow = Math.min(cols - rowsBack * this._terminal.cols, this._terminal.cols); ++ for (let i = 0; i < colsInRow; i++) { + const cell = line.getCell(i); + if (!cell) { + break; +@@ -380,12 +433,6 @@ export class SearchEngine { + offset += cell.getCode() === 0 ? 1 : cell.getChars().length; + } + } +- lineIndex++; +- line = this._terminal.buffer.active.getLine(lineIndex); +- if (line && !line.isWrapped) { +- break; +- } +- cols -= this._terminal.cols; + } + return offset; + } +diff --git a/src/SearchLineCache.ts b/src/SearchLineCache.ts +index 526f4bfcc74a881bb39b400ec79a25d33d602303..19b22f2f70e50a6b01d07966e15727cc5271c776 100644 +--- a/src/SearchLineCache.ts ++++ b/src/SearchLineCache.ts +@@ -109,9 +109,13 @@ export class SearchLineCache extends Disposable { + public translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean): LineCacheEntry { + const strings = []; + const lineOffsets = [0]; ++ // A single line longer than the whole scrollback leaves every buffer row wrapped, and the ++ // buffer's ring answers an out-of-range row by cycling back to the start, so an unbounded walk ++ // never reaches an unwrapped line. ++ const bufferLength = this._terminal.buffer.active.length; + let line = this._terminal.buffer.active.getLine(lineIndex); + while (line) { +- const nextLine = this._terminal.buffer.active.getLine(lineIndex + 1); ++ const nextLine = lineIndex + 1 < bufferLength ? this._terminal.buffer.active.getLine(lineIndex + 1) : undefined; + const lineWrapsToNext = nextLine ? nextLine.isWrapped : false; + let string = line.translateToString(!lineWrapsToNext && trimRight); + if (lineWrapsToNext && nextLine) { diff --git a/config/patches/xterm-src/@xterm__addon-search@0.17.0-beta.300.src.patch b/config/patches/xterm-src/@xterm__addon-search@0.17.0-beta.300.src.patch new file mode 100644 index 00000000000..c38f1d1278e --- /dev/null +++ b/config/patches/xterm-src/@xterm__addon-search@0.17.0-beta.300.src.patch @@ -0,0 +1,231 @@ +diff --git a/src/SearchEngine.ts b/src/SearchEngine.ts +index 1760bc2bd1fd274d23e2032fde631b39c739f0d9..5b3c5cc5e861356b87e8a15c55797f45bac20a5c 100644 +--- a/src/SearchEngine.ts ++++ b/src/SearchEngine.ts +@@ -76,6 +76,9 @@ export class SearchEngine { + // Search from startRow + 1 to end + if (!result) { + for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) { ++ if (this._isRowCoveredByEarlierSearch(y)) { ++ continue; ++ } + searchPosition.startRow = y; + searchPosition.startCol = 0; + result = this._findInLine(term, searchPosition, searchOptions); +@@ -127,6 +130,9 @@ export class SearchEngine { + // Search from startRow + 1 to end + if (!result) { + for (let y = startRow + 1; y < this._terminal.buffer.active.baseY + this._terminal.rows; y++) { ++ if (this._isRowCoveredByEarlierSearch(y)) { ++ continue; ++ } + searchPosition.startRow = y; + searchPosition.startCol = 0; + result = this._findInLine(term, searchPosition, searchOptions); +@@ -138,6 +144,11 @@ export class SearchEngine { + // If we hit the bottom and didn't search from the very top wrap back up + if (!result && startRow !== 0) { + for (let y = 0; y < startRow; y++) { ++ // Row 0 is never skipped: it can be a continuation whose line start was trimmed from the ++ // scrollback, and nothing earlier in this loop has searched it. ++ if (y > 0 && this._isRowCoveredByEarlierSearch(y)) { ++ continue; ++ } + searchPosition.startRow = y; + searchPosition.startCol = 0; + result = this._findInLine(term, searchPosition, searchOptions); +@@ -237,6 +248,22 @@ export class SearchEngine { + (((searchIndex + term.length) === line.length) || (Constants.NON_WORD_CHARACTERS.includes(line[searchIndex + term.length]))); + } + ++ /** `_isWholeWord` gated on the option, so a rejected hit can be stepped past instead of ending the scan. */ ++ private _satisfiesWholeWord(searchIndex: number, line: string, term: string, searchOptions: ISearchOptions): boolean { ++ return !searchOptions.wholeWord || this._isWholeWord(searchIndex, line, term); ++ } ++ ++ /** ++ * Whether an earlier `_findInLine` in this same call already scanned this row's line from an ++ * equal or lower offset, which makes rescanning it pure O(rows^2) work on one long line. Sound ++ * for every option because `_findInLine` returns the first accepted match at or after its ++ * offset, which is monotone in that offset. Only valid once such a search has happened — the ++ * wrap-around loop starts at row 0, whose line start may have been trimmed from the scrollback. ++ */ ++ private _isRowCoveredByEarlierSearch(row: number): boolean { ++ return this._terminal.buffer.active.getLine(row)?.isWrapped === true; ++ } ++ + /** + * Searches a line for a search term. Takes the provided terminal line and searches the text line, + * which may contain subsequent terminal lines if the text is wrapped. If the provided line number +@@ -250,23 +277,26 @@ export class SearchEngine { + * @returns The search result if it was found. + */ + private _findInLine(term: string, searchPosition: ISearchPosition, searchOptions: ISearchOptions = {}, isReverseSearch: boolean = false): ISearchResult | undefined { +- const row = searchPosition.startRow; +- const col = searchPosition.startCol; +- + // Ignore wrapped lines, only consider on unwrapped line (first row of command string). +- const firstLine = this._terminal.buffer.active.getLine(row); +- if (firstLine?.isWrapped) { +- if (isReverseSearch) { ++ if (isReverseSearch) { ++ // Reverse search never rewinds: its caller carries startCol down the rows of the line. Row 0 ++ // is searched even when wrapped, since its line start may have been trimmed from the scrollback. ++ if (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) { + searchPosition.startCol += this._terminal.cols; + return; + } +- +- // This will iterate until we find the line start. +- // When we find it, we will search using the calculated start column. +- searchPosition.startRow--; +- searchPosition.startCol += this._terminal.cols; +- return this._findInLine(term, searchPosition, searchOptions); ++ } else { ++ // A loop rather than recursion: one frame per wrapped row overflows the stack on a line long ++ // enough to fill the scrollback. Bounded at row 0 because after a reflow the buffer's ring ++ // holds stale entries at negative indices, so `getLine(-1)` answers with a wrapped line. ++ while (searchPosition.startRow > 0 && this._terminal.buffer.active.getLine(searchPosition.startRow)?.isWrapped) { ++ searchPosition.startRow--; ++ searchPosition.startCol += this._terminal.cols; ++ } + } ++ const row = searchPosition.startRow; ++ const col = searchPosition.startCol; ++ + let cache = this._lineCache.getLineFromCache(row); + if (!cache) { + cache = this._lineCache.translateBufferLineToStringWithWrap(row, true); +@@ -274,7 +304,7 @@ export class SearchEngine { + } + const [stringLine, offsets] = cache; + +- const offset = this._bufferColsToStringOffset(row, col); ++ const offset = this._bufferColsToStringOffset(row, col, offsets); + let searchTerm = term; + let searchStringLine = stringLine; + if (!searchOptions.regex) { +@@ -289,32 +319,46 @@ export class SearchEngine { + if (isReverseSearch) { + // This loop will get the resultIndex of the _last_ regex match in the range 0..offset + while (foundTerm = searchRegex.exec(searchStringLine.slice(0, offset))) { +- resultIndex = searchRegex.lastIndex - foundTerm[0].length; +- term = foundTerm[0]; +- searchRegex.lastIndex -= (term.length - 1); ++ const matchIndex = searchRegex.lastIndex - foundTerm[0].length; ++ if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) { ++ resultIndex = matchIndex; ++ term = foundTerm[0]; ++ } ++ searchRegex.lastIndex = matchIndex + 1; + } + } else { +- foundTerm = searchRegex.exec(searchStringLine.slice(offset)); +- if (foundTerm && foundTerm[0].length > 0) { +- resultIndex = offset + (searchRegex.lastIndex - foundTerm[0].length); +- term = foundTerm[0]; ++ // Driven over the whole line from `offset` rather than over `slice(offset)`: a slice ++ // re-anchors ^ and \b at whatever column the row happened to wrap at, and only ++ // first-accepted-match-at-or-after-offset is monotone in `offset`, which is what lets ++ // `_isRowCoveredByEarlierSearch` skip a wrapped row an earlier scan already covered. ++ searchRegex.lastIndex = offset; ++ while (foundTerm = searchRegex.exec(searchStringLine)) { ++ const matchIndex = searchRegex.lastIndex - foundTerm[0].length; ++ if (foundTerm[0].length > 0 && this._satisfiesWholeWord(matchIndex, searchStringLine, foundTerm[0], searchOptions)) { ++ resultIndex = matchIndex; ++ term = foundTerm[0]; ++ break; ++ } ++ // A zero-length or rejected match would otherwise repeat forever. ++ searchRegex.lastIndex = matchIndex + 1; + } + } ++ } else if (isReverseSearch) { ++ let matchIndex = offset - searchTerm.length >= 0 ? searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length) : -1; ++ // `lastIndexOf` clamps a negative fromIndex to 0, so index 0 has to end the walk. ++ while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) { ++ matchIndex = matchIndex > 0 ? searchStringLine.lastIndexOf(searchTerm, matchIndex - 1) : -1; ++ } ++ resultIndex = matchIndex; + } else { +- if (isReverseSearch) { +- if (offset - searchTerm.length >= 0) { +- resultIndex = searchStringLine.lastIndexOf(searchTerm, offset - searchTerm.length); +- } +- } else { +- resultIndex = searchStringLine.indexOf(searchTerm, offset); ++ let matchIndex = searchStringLine.indexOf(searchTerm, offset); ++ while (matchIndex >= 0 && !this._satisfiesWholeWord(matchIndex, searchStringLine, searchTerm, searchOptions)) { ++ matchIndex = searchStringLine.indexOf(searchTerm, matchIndex + 1); + } ++ resultIndex = matchIndex; + } + + if (resultIndex >= 0) { +- if (searchOptions.wholeWord && !this._isWholeWord(resultIndex, searchStringLine, term)) { +- return; +- } +- + // Adjust the row number and search index if needed since a "line" of text can span multiple + // rows + let startRowOffset = 0; +@@ -365,12 +409,21 @@ export class SearchEngine { + return offset; + } + +- private _bufferColsToStringOffset(startRow: number, cols: number): number { +- let lineIndex = startRow; +- let offset = 0; +- let line = this._terminal.buffer.active.getLine(lineIndex); +- while (cols > 0 && line) { +- for (let i = 0; i < cols && i < this._terminal.cols; i++) { ++ /** ++ * `cols` counts from the start of the logical line, so summing the cells of every row before the ++ * resume point costs O(line) per call and the highlight-all pass makes one call per match. ++ * `lineOffsets` already holds the string offset each wrapped row starts at — the same map used ++ * above to turn a match index back into a row — so only the last, partial row needs cells. It is ++ * also the map the row a match lands on is read from, which the cell sum disagreed with by one ++ * for a row whose trailing cell is the null placeholder of a wide character that wrapped. ++ */ ++ private _bufferColsToStringOffset(startRow: number, cols: number, lineOffsets: number[]): number { ++ const rowsBack = Math.min(Math.floor(cols / this._terminal.cols), lineOffsets.length - 1); ++ let offset = lineOffsets[rowsBack]; ++ const line = this._terminal.buffer.active.getLine(startRow + rowsBack); ++ if (line) { ++ const colsInRow = Math.min(cols - rowsBack * this._terminal.cols, this._terminal.cols); ++ for (let i = 0; i < colsInRow; i++) { + const cell = line.getCell(i); + if (!cell) { + break; +@@ -380,12 +433,6 @@ export class SearchEngine { + offset += cell.getCode() === 0 ? 1 : cell.getChars().length; + } + } +- lineIndex++; +- line = this._terminal.buffer.active.getLine(lineIndex); +- if (line && !line.isWrapped) { +- break; +- } +- cols -= this._terminal.cols; + } + return offset; + } +diff --git a/src/SearchLineCache.ts b/src/SearchLineCache.ts +index 526f4bfcc74a881bb39b400ec79a25d33d602303..19b22f2f70e50a6b01d07966e15727cc5271c776 100644 +--- a/src/SearchLineCache.ts ++++ b/src/SearchLineCache.ts +@@ -109,9 +109,13 @@ export class SearchLineCache extends Disposable { + public translateBufferLineToStringWithWrap(lineIndex: number, trimRight: boolean): LineCacheEntry { + const strings = []; + const lineOffsets = [0]; ++ // A single line longer than the whole scrollback leaves every buffer row wrapped, and the ++ // buffer's ring answers an out-of-range row by cycling back to the start, so an unbounded walk ++ // never reaches an unwrapped line. ++ const bufferLength = this._terminal.buffer.active.length; + let line = this._terminal.buffer.active.getLine(lineIndex); + while (line) { +- const nextLine = this._terminal.buffer.active.getLine(lineIndex + 1); ++ const nextLine = lineIndex + 1 < bufferLength ? this._terminal.buffer.active.getLine(lineIndex + 1) : undefined; + const lineWrapsToNext = nextLine ? nextLine.isWrapped : false; + let string = line.translateToString(!lineWrapsToNext && trimRight); + if (lineWrapsToNext && nextLine) { diff --git a/config/patches/xterm-upstream.json b/config/patches/xterm-upstream.json index ec36f65c71d..89afe4fb1fb 100644 --- a/config/patches/xterm-upstream.json +++ b/config/patches/xterm-upstream.json @@ -59,6 +59,33 @@ } ] }, + { + "name": "@xterm/addon-search", + "version": "0.17.0-beta.300", + "packageDir": "addons/addon-search", + "$note": "No versionStampFile: publish.js stamps the addon's package.json, which overlayBuildOutput never patches. The root `build` is required because the addon's own tsgo -p . has empty files/include and only project references, so it emits nothing on its own; `package` is the addon's webpack (CJS half) and the root `esbuild-package` emits the ESM half.", + "$upstream": "Submitted as https://github.com/xtermjs/xterm.js/pull/6149 (issue #6148). Once a release ships it, bump the addon and drop this entry.", + "sourcePatch": "config/patches/xterm-src/@xterm__addon-search@0.17.0-beta.300.src.patch", + "patch": "config/patches/@xterm__addon-search@0.17.0-beta.300.patch", + "generatedPaths": ["lib/"], + "build": [ + { + "cwd": "../..", + "command": "npm", + "args": ["run", "build"] + }, + { + "cwd": ".", + "command": "npm", + "args": ["run", "package"] + }, + { + "cwd": "../..", + "command": "npm", + "args": ["run", "esbuild-package"] + } + ] + }, { "name": "@xterm/addon-serialize", "version": "0.15.0-beta.300", diff --git a/docs/reference/xterm-patch-regeneration.md b/docs/reference/xterm-patch-regeneration.md index bd49dd8ca3d..ecb30913d28 100644 --- a/docs/reference/xterm-patch-regeneration.md +++ b/docs/reference/xterm-patch-regeneration.md @@ -24,11 +24,11 @@ truth. Everything else is derived from it by `config/scripts/regenerate-xterm-patches.mjs`, which is pinned to the exact upstream commit the published tarball was built from. -`@xterm/addon-webgl` and `@xterm/addon-serialize` are generated the same way, -from their own source patches under `config/patches/xterm-src/`. Their entries -differ only in `packageDir` and build steps; everything below applies to all -three. `@xterm/addon-ligatures` is the one patch still written by hand — see -[Known Gaps](#known-gaps). +`@xterm/addon-webgl`, `@xterm/addon-search` and `@xterm/addon-serialize` are +generated the same way, from their own source patches under +`config/patches/xterm-src/`. Their entries differ only in `packageDir` and build +steps; everything below applies to all four. `@xterm/addon-ligatures` is the one +patch still written by hand — see [Known Gaps](#known-gaps). ## Rules diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index eac56401344..a045833577f 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -111,6 +111,7 @@ overrides: patchedDependencies: '@vscode/windows-process-tree@0.8.0': 9217ef36c01ed74127fef5512b0c92089cdbf820fd6c109dd671137eebdc7585 '@xterm/addon-ligatures@0.11.0-beta.300': 47405b9994b5acf1b4e90b49250358c1ca03649854d59560e7732b72fe336920 + '@xterm/addon-search@0.17.0-beta.300': eee5338dd2621ece46e79c61ec06766cd7fadaf79ffdb24e2a8ab68e97ef31f0 '@xterm/addon-serialize@0.15.0-beta.300': 851eac3d75e6d8c013b9f4c053e61d824b23965cb19ecc28e335e05059f3a294 '@xterm/addon-webgl@0.20.0-beta.299': 94687e89a0115e6e6aa102837f986debdc029c091527ee5eb4a4e17ceaf9473e '@xterm/xterm@6.1.0-beta.303': 98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d @@ -319,7 +320,7 @@ importers: version: 0.11.0-beta.300(patch_hash=47405b9994b5acf1b4e90b49250358c1ca03649854d59560e7732b72fe336920)(@xterm/xterm@6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d)) '@xterm/addon-search': specifier: 0.17.0-beta.300 - version: 0.17.0-beta.300(@xterm/xterm@6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d)) + version: 0.17.0-beta.300(patch_hash=eee5338dd2621ece46e79c61ec06766cd7fadaf79ffdb24e2a8ab68e97ef31f0)(@xterm/xterm@6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d)) '@xterm/addon-unicode11': specifier: 0.10.0-beta.300 version: 0.10.0-beta.300(@xterm/xterm@6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d)) @@ -9754,7 +9755,7 @@ snapshots: lru-cache: 11.5.1 opentype.js: 2.0.0 - '@xterm/addon-search@0.17.0-beta.300(@xterm/xterm@6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d))': + '@xterm/addon-search@0.17.0-beta.300(patch_hash=eee5338dd2621ece46e79c61ec06766cd7fadaf79ffdb24e2a8ab68e97ef31f0)(@xterm/xterm@6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d))': dependencies: '@xterm/xterm': 6.1.0-beta.303(patch_hash=98756bcedc402bcdb7c6ab7b015d2e59cd18e97b03a2c06a27e95bb3ba429d9d) diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index 8338c47e126..d241459f884 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -44,6 +44,7 @@ patchedDependencies: node-pty@1.1.0: config/patches/node-pty@1.1.0.patch '@xterm/addon-ligatures@0.11.0-beta.300': config/patches/@xterm__addon-ligatures@0.11.0-beta.300.patch '@xterm/addon-webgl@0.20.0-beta.299': config/patches/@xterm__addon-webgl@0.20.0-beta.299.patch + '@xterm/addon-search@0.17.0-beta.300': config/patches/@xterm__addon-search@0.17.0-beta.300.patch '@xterm/addon-serialize@0.15.0-beta.300': config/patches/@xterm__addon-serialize@0.15.0-beta.300.patch '@xterm/xterm@6.1.0-beta.303': config/patches/@xterm__xterm@6.1.0-beta.303.patch lint-staged@16.4.0: config/patches/lint-staged@16.4.0.patch diff --git a/src/renderer/src/components/terminal-search-long-wrapped-line.test.ts b/src/renderer/src/components/terminal-search-long-wrapped-line.test.ts new file mode 100644 index 00000000000..28be8b5fc7a --- /dev/null +++ b/src/renderer/src/components/terminal-search-long-wrapped-line.test.ts @@ -0,0 +1,362 @@ +// @vitest-environment happy-dom + +import type { ISearchOptions } from '@xterm/addon-search' +import { SearchAddon } from '@xterm/addon-search' +import { Terminal } from '@xterm/xterm' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { + DESKTOP_TERMINAL_SCROLLBACK_ROWS_DEFAULT, + DESKTOP_TERMINAL_SCROLLBACK_ROWS_MAX +} from '../../../shared/terminal-scrollback-policy' +import { safeFind } from './terminal-search-safe-find' + +/** + * Regression for crash report 012eb5be (Orca 1.4.194, win32): searching a pane + * that held one un-newlined line — base64, a minified bundle, a single huge log + * record — threw `RangeError: Maximum call stack size exceeded` out of + * TerminalSearch's effect and tripped the `terminal.workbench` error boundary. + * + * Mechanism, in @xterm/addon-search's SearchEngine (patched in + * config/patches/@xterm__addon-search@*.patch, generated from the source patch + * under config/patches/xterm-src/; submitted upstream as + * https://github.com/xtermjs/xterm.js/pull/6149): `_findInLine` rewound to the first row of a + * wrapped line by calling itself once per wrapped row, so recursion depth equals + * the number of screen rows the logical line occupies. Scrollback reaches + * DESKTOP_TERMINAL_SCROLLBACK_ROWS_MAX rows, which is far past V8's stack. + * + * The rewind is reached on every re-entry into the middle of a wrapped line — + * `_highlightAllMatches` restarting at the row after a match, and `findNext` + * resuming from the current selection — so this drives the real Terminal + + * SearchAddon through Orca's own `safeFind`, which deliberately rethrows + * anything that is not the decoration error. + */ + +/** Reaches the ring behind the public buffer API to pin the negative-index state reflow leaves. */ +type RingBufferProbe = { + _core: { buffer: { lines: { _array: unknown[] } } } +} + +const COLS = 80 +const ROWS = 24 +/** Long enough that the wrap chain outruns V8's stack on any host. */ +const WRAPPED_ROWS = 12_000 +/** A line longer than this scrollback loses its first rows: the bug is eviction, not size. */ +const TRIMMED_HEAD_SCROLLBACK = 100 +const TRIMMED_HEAD_LINE_ROWS = 200 +const NEEDLE = 'needle' +/** + * Rows of one wrapped line per match, and how many matches that line holds. Enough matches to make + * the highlight-all pass — which re-enters the line once per match — the dominant cost, and enough + * rows that a per-match walk of the line is a freeze rather than a slow search. Kept under the + * addon's 1 000-decoration limit so the match count is exact. + */ +const MATCH_ROW_STRIDE = 40 +const MATCHES_IN_LINE = 750 +/** + * A full-buffer scan runs on the renderer's main thread on every keystroke in + * the find bar, so anything near this is a visible freeze rather than a slow + * search. Unfixed it is ~18s for a default-scrollback buffer; fixed, ~6ms. + */ +const FULL_SCAN_BUDGET_MS = 5_000 + +// Matches the decoration options TerminalSearch passes, so the highlight-all +// pass (the crash's entry point) actually runs. +const SEARCH_DECORATIONS = { + matchBackground: '#5c4a00', + matchBorder: '#5c4a00', + matchOverviewRuler: '#ffcc00', + activeMatchBackground: '#c4580e', + activeMatchBorder: '#ffcf6b', + activeMatchColorOverviewRuler: '#ff9900' +} as const + +/** + * Every mode the find bar can put the engine in. Regex and whole word used to be + * excluded from the wrapped-row skip, which left them on the O(rows^2) walk after + * the recursion that used to abort it was gone: a 12 000-row line took 5.4 minutes + * of blocked main thread instead of throwing after 48 seconds. + */ +const SEARCH_MODES = [ + ['plain', {}], + ['regex', { regex: true }], + ['whole word', { wholeWord: true }] +] as const satisfies readonly (readonly [string, ISearchOptions])[] + +function write(terminal: Terminal, data: string): Promise { + return new Promise((resolve) => terminal.write(data, resolve)) +} + +function openTerminalWithSearch(scrollback: number = DESKTOP_TERMINAL_SCROLLBACK_ROWS_MAX): { + terminal: Terminal + search: SearchAddon +} { + const container = document.createElement('div') + document.body.appendChild(container) + const terminal = new Terminal({ cols: COLS, rows: ROWS, scrollback }) + terminal.open(container) + const search = new SearchAddon() + terminal.loadAddon(search) + return { terminal, search } +} + +describe('terminal search inside one very long wrapped line', () => { + beforeEach(() => { + // happy-dom has no canvas text metrics; xterm measures glyphs on open(). + vi.spyOn(HTMLCanvasElement.prototype, 'getContext').mockReturnValue({ + measureText: () => ({ width: 10 }) + } as unknown as CanvasRenderingContext2D) + }) + + afterEach(() => { + vi.restoreAllMocks() + document.body.replaceChildren() + }) + + it.each(SEARCH_MODES)( + 'rewinds to the start of the line without overflowing the stack (%s)', + async (_mode, options) => { + const { terminal, search } = openTerminalWithSearch() + // One line of WRAPPED_ROWS screen rows whose only match ends on the + // second-to-last row, so the highlight pass resumes one row further on and + // has to rewind the whole chain to reach the line start. Space-delimited so + // the whole-word mode has something to find. + await write( + terminal, + `${'x'.repeat(COLS * (WRAPPED_ROWS - 1) - NEEDLE.length - 1)} ${NEEDLE} ${'x'.repeat(COLS - 1)}` + ) + + const find = (): boolean => + safeFind((term, searchOptions) => search.findNext(term, searchOptions), NEEDLE, { + ...options, + decorations: SEARCH_DECORATIONS + }) + + let found: boolean | undefined + expect(() => { + found = find() + }).not.toThrow() + expect(found).toBe(true) + + // Second find resumes from the selection, deep inside the wrapped line. + expect(() => { + found = find() + }).not.toThrow() + expect(found).toBe(true) + } + ) + + it.each(SEARCH_MODES)( + 'scans a long wrapped line once, not once per wrapped row (%s)', + async (_mode, options) => { + const { terminal, search } = openTerminalWithSearch() + await write(terminal, 'x'.repeat(COLS * DESKTOP_TERMINAL_SCROLLBACK_ROWS_DEFAULT)) + + // No match, so the scan visits every row: the shape that froze the pane. + const startedAt = performance.now() + safeFind((term, searchOptions) => search.findNext(term, searchOptions), NEEDLE, { + ...options, + decorations: SEARCH_DECORATIONS + }) + + expect(performance.now() - startedAt).toBeLessThan(FULL_SCAN_BUDGET_MS) + } + ) + + it.each(SEARCH_MODES)( + 'highlights many matches in one wrapped line without re-walking it per match (%s)', + async (_mode, options) => { + const { terminal, search } = openTerminalWithSearch() + // `_highlightAllMatches` calls `SearchEngine.find` once per match, and each call resumes + // deep inside the line. Converting the resume column to a string offset walked every cell + // before it, O(line) per match, so this shape stayed quadratic after the no-match scan was + // bounded: ~11s for a line this long, a renderer freeze rather than a RangeError the + // boundary recovered from. The per-match rewind and case fold are O(rows) and O(chars) + // but measured at well under 1s combined, so they stay simple. + const block = ` ${NEEDLE} ${'x'.repeat(COLS * MATCH_ROW_STRIDE - NEEDLE.length - 2)}` + await write(terminal, block.repeat(MATCHES_IN_LINE)) + + let resultCount = -1 + search.onDidChangeResults((event) => { + resultCount = event.resultCount + }) + const startedAt = performance.now() + const found = safeFind( + (term, searchOptions) => search.findNext(term, searchOptions), + NEEDLE, + { + ...options, + decorations: SEARCH_DECORATIONS + } + ) + + expect(performance.now() - startedAt).toBeLessThan(FULL_SCAN_BUDGET_MS) + expect(found).toBe(true) + expect(resultCount).toBe(MATCHES_IN_LINE) + } + ) + + it('reports every match inside a wrapped line', async () => { + const { terminal, search } = openTerminalWithSearch() + let resultCount = -1 + search.onDidChangeResults((event) => { + resultCount = event.resultCount + }) + // One logical line wrapping over three rows with a match in each, then a + // separate unwrapped line. + const paddedNeedle = NEEDLE + 'x'.repeat(COLS - NEEDLE.length) + await write(terminal, `${paddedNeedle.repeat(3)}\r\nplain ${NEEDLE}\r\n`) + + safeFind((term, options) => search.findNext(term, options), NEEDLE, { + decorations: SEARCH_DECORATIONS + }) + + expect(resultCount).toBe(4) + }) + + it('keeps reaching matches in a wrapped line whose first row was trimmed away', async () => { + const { terminal, search } = openTerminalWithSearch(TRIMMED_HEAD_SCROLLBACK) + // One long line whose head is evicted, so the surviving chain begins on a + // row marked isWrapped and no line start is left in the buffer to cover it. + const paddedNeedle = NEEDLE + 'x'.repeat(COLS - NEEDLE.length) + const lead = 'x'.repeat(COLS * (TRIMMED_HEAD_LINE_ROWS - 3)) + await write(terminal, `${lead}${paddedNeedle.repeat(2)}${'x'.repeat(COLS)}\r\n`) + for (let i = 0; i < 60; i++) { + await write(terminal, `line ${i}\r\n`) + } + const buffer = terminal.buffer.active + expect(buffer.getLine(0)?.isWrapped).toBe(true) + + const matchRows: number[] = [] + for (let y = 0; y < buffer.length; y++) { + if (buffer.getLine(y)?.translateToString().includes(NEEDLE)) { + matchRows.push(y) + } + } + expect(matchRows.length).toBe(2) + + // Cycle far enough to come back round in both directions: the surviving rows must stay + // reachable, not be visited once and then stranded. Reverse search used to return for any + // wrapped row including row 0, so a trimmed-head line was never searched backwards at all. + for (const direction of ['findNext', 'findPrevious'] as const) { + const visits = new Map() + for (let i = 0; i < 12; i++) { + safeFind((term, options) => search[direction](term, options), NEEDLE, { + decorations: SEARCH_DECORATIONS + }) + const row = terminal.getSelectionPosition()?.start.y + if (row !== undefined) { + visits.set(row, (visits.get(row) ?? 0) + 1) + } + } + + for (const row of matchRows) { + expect(visits.get(row) ?? 0, direction).toBeGreaterThan(1) + } + } + }) + + it('searches a line that is longer than the whole scrollback', async () => { + const { terminal, search } = openTerminalWithSearch(TRIMMED_HEAD_SCROLLBACK) + // Every row of the buffer is then a continuation, and the ring answers an out-of-range row by + // cycling back to row 0, so walking forward for the end of the line never terminates. + const paddedNeedle = NEEDLE + 'x'.repeat(COLS - NEEDLE.length) + await write(terminal, `${'x'.repeat(COLS * TRIMMED_HEAD_LINE_ROWS)}${paddedNeedle.repeat(3)}`) + const buffer = terminal.buffer.active + expect(buffer.getLine(0)?.isWrapped).toBe(true) + expect(buffer.getLine(buffer.length - 1)?.isWrapped).toBe(true) + + const startedAt = performance.now() + const found = safeFind((term, options) => search.findNext(term, options), NEEDLE, { + decorations: SEARCH_DECORATIONS + }) + + expect(found).toBe(true) + expect(performance.now() - startedAt).toBeLessThan(FULL_SCAN_BUDGET_MS) + }) + + it('stops rewinding at row 0 when a reflow trims a wrapped line head', async () => { + const { terminal, search } = openTerminalWithSearch(TRIMMED_HEAD_SCROLLBACK) + const paddedNeedle = NEEDLE + 'x'.repeat(COLS - NEEDLE.length) + await write(terminal, `${'x'.repeat(COLS * TRIMMED_HEAD_LINE_ROWS)}${paddedNeedle.repeat(4)}`) + for (let i = 0; i < 20; i++) { + await write(terminal, `line ${i}\r\n`) + } + // Narrowing a pane reflows the buffer, which leaves the ring holding entries at negative + // indices, so `getLine(-1)` answers with a stale wrapped line instead of undefined. The rewind + // has to stop at row 0 or it walks backwards forever and hangs the renderer. + terminal.resize(15, 5) + await write(terminal, '') + expect(terminal.buffer.active.getLine(0)?.isWrapped).toBe(true) + expect( + Object.keys((terminal as unknown as RingBufferProbe)._core.buffer.lines._array) + ).toContain('-1') + + const startedAt = performance.now() + const found = safeFind((term, options) => search.findNext(term, options), NEEDLE, { + decorations: SEARCH_DECORATIONS + }) + + expect(found).toBe(true) + expect(performance.now() - startedAt).toBeLessThan(FULL_SCAN_BUDGET_MS) + }) + + it('keeps scanning a line past a hit whole word rejects', async () => { + const { terminal, search } = openTerminalWithSearch() + // Upstream stopped at the first `indexOf` hit, so `aneedlea` hid the real word + // nine columns later and the find bar reported no match at all. Scanning on is + // also what makes the wrapped-row skip sound for wholeWord. + await write(terminal, `a${NEEDLE}a ${NEEDLE} done\r\n`) + + const found = safeFind((term, options) => search.findNext(term, options), NEEDLE, { + wholeWord: true, + decorations: SEARCH_DECORATIONS + }) + + expect(found).toBe(true) + expect(terminal.getSelectionPosition()?.start).toEqual({ x: 9, y: 0 }) + }) + + it('finds a whole-word match that only matches from a later wrapped row', async () => { + const { terminal, search } = openTerminalWithSearch() + // The first hit on the line is `aneedlea`; the real word is on the second + // wrapped row, which the skip removes from the walk. + const filler = 'x'.repeat(COLS - NEEDLE.length - 2) + await write(terminal, `a${NEEDLE}a${filler} ${NEEDLE} ${filler}`) + + const found = safeFind((term, options) => search.findNext(term, options), NEEDLE, { + wholeWord: true, + decorations: SEARCH_DECORATIONS + }) + + expect(found).toBe(true) + }) + + it('steps past a zero-length regex match instead of abandoning the line', async () => { + const { terminal, search } = openTerminalWithSearch() + await write(terminal, `abc ${NEEDLE} def\r\n`) + + // `^` matches empty at offset 0, which upstream took as the line's only answer. + const found = safeFind((term, options) => search.findNext(term, options), `^|${NEEDLE}`, { + regex: true, + decorations: SEARCH_DECORATIONS + }) + + expect(found).toBe(true) + expect(terminal.getSelectionPosition()?.start).toEqual({ x: 4, y: 0 }) + }) + + it('anchors a regex to the logical line, not to every wrapped row', async () => { + const { terminal, search } = openTerminalWithSearch() + // The needle starts the second row of one wrapped line. A wrap column is a + // rendering artifact, so `^` must not match there. + await write(terminal, 'x'.repeat(COLS) + NEEDLE + 'x'.repeat(COLS - NEEDLE.length)) + + const found = safeFind((term, options) => search.findNext(term, options), `^${NEEDLE}`, { + regex: true, + decorations: SEARCH_DECORATIONS + }) + + expect(found).toBe(false) + }) +}) From 90780acb8561298d1cbb793373e8684c68b87445 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Thu, 3 Sep 2026 18:07:56 -0700 Subject: [PATCH 216/398] refactor(agents): one pane-identity resolver behind six thin adapters (tranche 0) (#18243) * feat(agents): pane-identity canonical adapter, comparison telemetry, inventory ratchet phase 1 * fix(agents): preserve canonical coverage provenance * refactor(agents): unify pane identity adapters for tranche 0 * fix(agents): keep title resolver cache-free after rebase * Fix ladder tranche zero review findings * fix(agents): restore title classifier memoization * fix(agents): fence unknown canonical evidence sources * docs: drop the ladder plan and decision table from the PR Design docs stay out of the shipped tree; the code carries its own comments and the decision table lives in the test fixture. --------- Co-authored-by: Merge Sim --- .../tab-agent-identity-decision-table.test.ts | 198 ++++++++++ src/shared/agent-status-identity.ts | 9 +- src/shared/pane-agent-evidence-sources.ts | 19 + .../pane-agent-identity-adapter.test.ts | 278 ++++++++++++++ src/shared/pane-agent-identity-adapter.ts | 361 ++++++++++++++++++ .../pane-agent-identity-inventory.test.ts | 138 ++++++- .../pane-agent-identity-resolver.test.ts | 20 + src/shared/pane-agent-identity-resolver.ts | 75 +--- ...e-agent-identity-surface-inventory.test.ts | 284 ++++++++++++++ .../pane-agent-identity-title-corpus.test.ts | 156 ++++++++ src/shared/pane-agent-owner.test.ts | 36 ++ src/shared/pane-agent-owner.ts | 7 +- src/shared/terminal-title-agent-type.ts | 23 +- 13 files changed, 1520 insertions(+), 84 deletions(-) create mode 100644 src/renderer/src/lib/tab-agent-identity-decision-table.test.ts create mode 100644 src/shared/pane-agent-evidence-sources.ts create mode 100644 src/shared/pane-agent-identity-adapter.test.ts create mode 100644 src/shared/pane-agent-identity-adapter.ts create mode 100644 src/shared/pane-agent-identity-surface-inventory.test.ts create mode 100644 src/shared/pane-agent-identity-title-corpus.test.ts diff --git a/src/renderer/src/lib/tab-agent-identity-decision-table.test.ts b/src/renderer/src/lib/tab-agent-identity-decision-table.test.ts new file mode 100644 index 00000000000..6c543db3594 --- /dev/null +++ b/src/renderer/src/lib/tab-agent-identity-decision-table.test.ts @@ -0,0 +1,198 @@ +import { writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { + resolveCanonicalPaneAgentIdentity, + type CanonicalPaneAgentIdentity +} from '../../../shared/pane-agent-identity-adapter' +import { resolveTabAgentFromSignals } from './tab-agent-from-signals' +import type { TuiAgent } from '../../../shared/tui-agent' + +const AGENTS: readonly TuiAgent[] = ['claude', 'codex'] +const SLOT_COUNT = 7 +const SHAPE_COUNT = 3 ** SLOT_COUNT * 4 * 2 +const TITLES: readonly string[] = ['', 'zsh', 'Task - claude', 'Task - codex'] + +type Breakdown = Record< + 'launch' | 'completed-hook' | 'sleeping-session' | 'process' | 'sibling' | 'title', + number +> + +function slotValues(mask: number): (TuiAgent | null)[] { + let remaining = mask + return Array.from({ length: SLOT_COUNT }, () => { + const value = remaining % 3 + remaining = Math.floor(remaining / 3) + return value === 0 ? null : AGENTS[value - 1] + }) +} + +function canonicalResult( + values: readonly (TuiAgent | null)[], + title: string, + withProof: boolean +): CanonicalPaneAgentIdentity { + const [hook, siblingHook, completed, siblingCompleted, process, sleeping, launch] = values + return resolveCanonicalPaneAgentIdentity({ + hookAgent: hook, + hookIsLive: hook !== null, + completedHookAgent: completed, + launchAgent: launch, + foregroundAgent: process, + processProof: + withProof && process + ? { + agent: process, + processIncarnation: 'fixture-process', + authorityId: 'fixture-authority', + capturedAgeMs: 10, + validForMs: 1_000 + } + : undefined, + sleepingSessionAgent: sleeping, + siblingAgents: [siblingHook, siblingCompleted].filter( + (agent): agent is TuiAgent => agent !== null + ), + allowSibling: true, + title + }) +} + +function realResult(values: readonly (TuiAgent | null)[], title: string, remote: boolean) { + const [hook, siblingHook, completed, siblingCompleted, process, sleeping, launch] = values + // The seven slots model steady-state observations; this runtime memory bit is intentionally + // held true instead of adding an eighth dimension to the approved 17,496-shape table. + return resolveTabAgentFromSignals({ + hasObservedAgentSignal: true, + isRemote: remote, + title, + hookAgent: hook, + siblingHookAgent: siblingHook, + focusedCompletedHookAgent: completed, + siblingCompletedHookAgent: siblingCompleted, + processAgent: process, + processShellForeground: false, + sleepingSessionAgent: sleeping, + launchAgent: launch ?? undefined + }) +} + +function runDecisionTable(withProof: boolean) { + let disagreements = 0 + let flipped = 0 + const breakdown: Breakdown = { + launch: 0, + 'completed-hook': 0, + 'sleeping-session': 0, + process: 0, + sibling: 0, + title: 0 + } + for (let mask = 0; mask < 3 ** SLOT_COUNT; mask += 1) { + const values = slotValues(mask) + for (const title of TITLES) { + for (const remote of [false, true]) { + const real = realResult(values, title, remote) + const canonical = canonicalResult(values, title, withProof) + if (real !== canonical.agent) { + disagreements += 1 + if (canonical.source !== null) { + breakdown[canonical.source] += 1 + } + } + if (!withProof) { + const proven = canonicalResult(values, title, true) + if ( + canonical.agent !== proven.agent && + proven.source === 'process' && + (canonical.source === 'launch' || + canonical.source === 'completed-hook' || + canonical.source === 'sleeping-session') + ) { + flipped += 1 + } + } + } + } + } + return { disagreements, flipped, breakdown } +} + +describe('renderer ladder decision table', () => { + it('replays the real shipping ladder and records all rung disagreements', () => { + const proofFree = runDecisionTable(false) + const freshProof = runDecisionTable(true) + const result = { + shapes: SHAPE_COUNT, + proofOmitted: proofFree, + freshProof, + flippedByAddingProof: proofFree.flipped + } + writeFileSync( + join(tmpdir(), 'orca-pane-agent-identity-decision-table-real.json'), + `${JSON.stringify(result, null, 2)}\n` + ) + // Re-derived against resolveTabAgentFromSignals (not a hand-written model). These differ from + // the approved 2,520/648 totals and 396/144/72/36 breakdown; see the PR comment. + expect(proofFree).toEqual({ + disagreements: 2_622, + flipped: 1_872, + breakdown: { + launch: 1_884, + 'completed-hook': 478, + 'sleeping-session': 144, + process: 0, + sibling: 54, + title: 6 + } + }) + expect(freshProof).toEqual({ + disagreements: 658, + flipped: 0, + breakdown: { + launch: 588, + 'completed-hook': 46, + 'sleeping-session': 0, + process: 0, + sibling: 6, + title: 2 + } + }) + expect(proofFree.flipped).toBe(1_872) + }) + + it('requires both freshness fields before process evidence can change the no-proof result', () => { + const values = [null, null, null, null, 'codex', null, 'claude'] as const + expect(canonicalResult(values, '', false)).toMatchObject({ + agent: 'claude', + source: 'launch' + }) + expect( + resolveCanonicalPaneAgentIdentity({ + foregroundAgent: 'codex', + processProof: { + agent: 'codex', + processIncarnation: 'fixture-process', + authorityId: 'fixture-authority', + capturedAgeMs: undefined as unknown as number, + validForMs: 1_000 + }, + launchAgent: 'claude' + }) + ).toMatchObject({ agent: 'claude', source: 'launch' }) + expect( + resolveCanonicalPaneAgentIdentity({ + foregroundAgent: 'codex', + processProof: { + agent: 'codex', + processIncarnation: 'fixture-process', + authorityId: 'fixture-authority', + capturedAgeMs: 10, + validForMs: undefined as unknown as number + }, + launchAgent: 'claude' + }) + ).toMatchObject({ agent: 'claude', source: 'launch' }) + }) +}) diff --git a/src/shared/agent-status-identity.ts b/src/shared/agent-status-identity.ts index f3d4b65a86b..9cb81625310 100644 --- a/src/shared/agent-status-identity.ts +++ b/src/shared/agent-status-identity.ts @@ -4,6 +4,8 @@ import { type AgentStatusState, type AgentType } from './agent-status-types' +import { resolveCanonicalPaneAgentIdentity } from './pane-agent-identity-adapter' +import type { TuiAgent } from './tui-agent' type ExistingAgentIdentity = { agentType?: AgentType @@ -63,6 +65,11 @@ export function resolveAgentStatusIdentity(args: { inheritedFromActivePane: false } } + const canonical = resolveCanonicalPaneAgentIdentity({ + hookAgent: incomingAgentType as TuiAgent, + hookIsLive: true, + completedHookAgent: args.existing.state === 'done' ? (existingAgentType as TuiAgent) : undefined + }) if (isActiveExistingIdentity(args.existing, args.now, staleAfterMs)) { return { // Why: child agent CLIs inherit ORCA_PANE_KEY from their parent terminal. @@ -74,7 +81,7 @@ export function resolveAgentStatusIdentity(args: { } return { - agentType: incomingAgentType, + agentType: canonical.agent ?? incomingAgentType, inheritedFromActivePane: false } } diff --git a/src/shared/pane-agent-evidence-sources.ts b/src/shared/pane-agent-evidence-sources.ts new file mode 100644 index 00000000000..32b3648494a --- /dev/null +++ b/src/shared/pane-agent-evidence-sources.ts @@ -0,0 +1,19 @@ +/** Evidence classes in canonical strength order, strongest first. */ +export const PANE_AGENT_EVIDENCE_SOURCES = [ + /** A live provider hook for a turn in progress. The agent is running and said so. */ + 'live-hook', + /** The pane's foreground process, as read on the execution host. */ + 'process', + /** Orca launched, resumed, or accepted a command for this agent. A fact Orca owns. */ + 'launch', + /** A provider hook from a turn that finished. Still authoritative about identity. */ + 'completed-hook', + /** A sleeping session record restored for this pane. */ + 'sleeping-session', + /** Another pane in the same tab. Tab-level surfaces only; never pane-scoped routing. */ + 'sibling', + /** Parsed from the terminal title. A decoration channel; anyone can type an agent's name. */ + 'title' +] as const + +export type PaneAgentEvidenceSource = (typeof PANE_AGENT_EVIDENCE_SOURCES)[number] diff --git a/src/shared/pane-agent-identity-adapter.test.ts b/src/shared/pane-agent-identity-adapter.test.ts new file mode 100644 index 00000000000..d898cb9adcf --- /dev/null +++ b/src/shared/pane-agent-identity-adapter.test.ts @@ -0,0 +1,278 @@ +import { describe, expect, it } from 'vitest' +import { + buildPaneAgentIdentityEvidenceWire, + isForegroundProcessProofFresh, + resolveCanonicalPaneAgentIdentity, + type ForegroundProcessProof +} from './pane-agent-identity-adapter' + +const freshProof: ForegroundProcessProof = { + agent: 'codex', + processIncarnation: 'opaque-pid-token', + authorityId: 'main:test', + capturedAgeMs: 50, + validForMs: 5_000 +} + +describe('per-pane coverage gate', () => { + it('covers a pane from hook, launch, or sleeping-session evidence alone', () => { + expect( + resolveCanonicalPaneAgentIdentity({ hookAgent: 'claude', hookIsLive: true }).coverage + ).toBe('covered') + expect(resolveCanonicalPaneAgentIdentity({ completedHookAgent: 'claude' }).coverage).toBe( + 'covered' + ) + expect(resolveCanonicalPaneAgentIdentity({ launchAgent: 'codex' }).coverage).toBe('covered') + expect(resolveCanonicalPaneAgentIdentity({ sleepingSessionAgent: 'gemini' }).coverage).toBe( + 'covered' + ) + }) + + it('never covers a pane from a title, a sibling, or a bare foreground name', () => { + expect(resolveCanonicalPaneAgentIdentity({ title: 'claude' }).coverage).toBe('uncovered') + expect( + resolveCanonicalPaneAgentIdentity({ siblingAgent: 'claude', allowSibling: true }).coverage + ).toBe('uncovered') + expect(resolveCanonicalPaneAgentIdentity({ foregroundAgent: 'codex' }).coverage).toBe( + 'uncovered' + ) + }) + + it('is computed from evidence, never from a platform or remote flag', () => { + // The input deliberately has no platform/isRemote field to branch on; this pins that a + // hook-covered pane resolves identically regardless of any caller-side host knowledge. + const identity = resolveCanonicalPaneAgentIdentity({ hookAgent: 'claude', hookIsLive: true }) + expect(identity).toMatchObject({ agent: 'claude', source: 'live-hook', coverage: 'covered' }) + }) +}) + +describe('process rung requires a host-stamped proof', () => { + it('rejects a stale or malformed proof and accepts a fresh one', () => { + expect(isForegroundProcessProofFresh(freshProof)).toBe(true) + expect(isForegroundProcessProofFresh({ ...freshProof, capturedAgeMs: 6_000 })).toBe(false) + expect(isForegroundProcessProofFresh({ ...freshProof, capturedAgeMs: -1 })).toBe(false) + expect(isForegroundProcessProofFresh({ ...freshProof, validForMs: 0 })).toBe(false) + expect(isForegroundProcessProofFresh({ ...freshProof, capturedAgeMs: Number.NaN })).toBe(false) + }) + + it('a bare foreground name cannot outrank launch; a proven process can', () => { + const unproven = resolveCanonicalPaneAgentIdentity({ + launchAgent: 'claude', + foregroundAgent: 'codex' + }) + expect(unproven).toMatchObject({ agent: 'claude', source: 'launch' }) + + const proven = resolveCanonicalPaneAgentIdentity({ + launchAgent: 'claude', + foregroundAgent: 'codex', + processProof: freshProof + }) + expect(proven).toMatchObject({ agent: 'codex', source: 'process', coverage: 'covered' }) + }) + + it('an expired proof and a name-mismatched proof both drop the process rung', () => { + const expired = resolveCanonicalPaneAgentIdentity({ + launchAgent: 'claude', + foregroundAgent: 'codex', + processProof: { ...freshProof, capturedAgeMs: 10_000 } + }) + expect(expired).toMatchObject({ agent: 'claude', source: 'launch' }) + + const mismatched = resolveCanonicalPaneAgentIdentity({ + launchAgent: 'claude', + foregroundAgent: 'gemini', + processProof: freshProof + }) + expect(mismatched).toMatchObject({ agent: 'claude', source: 'launch' }) + }) +}) + +describe('uncovered compatibility lane', () => { + it('preserves the caller-provided legacy result verbatim', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + title: 'Fix the parser - grok', + uncoveredFallback: { agent: 'grok', titleOnly: true } + }) + expect(identity).toMatchObject({ + agent: 'grok', + source: 'title', + coverage: 'uncovered', + titleOnly: true + }) + }) + + it('answers from title evidence marked title-only when no fallback is supplied', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + title: 'STA-4011 Linux Antigravity Commit Messages - grok' + }) + expect(identity).toMatchObject({ + agent: 'grok', + source: 'title', + coverage: 'uncovered', + titleOnly: true + }) + }) + + it('a legacy null stays null rather than re-deriving from the title', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + title: 'anything - grok', + uncoveredFallback: { agent: null } + }) + expect(identity).toMatchObject({ agent: null, source: null, coverage: 'uncovered' }) + }) + + it('does not let a legacy title fallback bypass the ambiguity fence', () => { + expect( + resolveCanonicalPaneAgentIdentity({ + title: 'OC | something - grok', + uncoveredFallback: { agent: 'opencode', titleOnly: true } + }) + ).toMatchObject({ agent: null, source: null, ambiguousAt: 'title' }) + expect( + resolveCanonicalPaneAgentIdentity({ + title: 'compare codex with grok', + uncoveredFallback: { agent: 'codex', titleOnly: true } + }) + ).toMatchObject({ agent: null, source: null, coverage: 'uncovered' }) + }) + + it('does not label a foreground-only compatibility answer as title-only', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + foregroundAgent: 'codex', + uncoveredFallback: { agent: 'codex' } + }) + expect(identity).toMatchObject({ + agent: 'codex', + source: null, + coverage: 'uncovered', + titleOnly: false + }) + }) +}) + +describe('canonical ladder inside the covered lane', () => { + it('keeps title last: a covered launch beats a parsed title', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + launchAgent: 'claude', + title: 'STA-4011 Linux Antigravity Commit Messages - grok' + }) + expect(identity).toMatchObject({ agent: 'claude', source: 'launch', titleOnly: false }) + }) + + it('sibling evidence needs the explicit tab-scope opt-in', () => { + const withoutOptIn = resolveCanonicalPaneAgentIdentity({ + launchAgent: 'claude', + siblingAgent: 'codex' + }) + expect(withoutOptIn.agent).toBe('claude') + const optedIn = resolveCanonicalPaneAgentIdentity({ + hookAgent: 'claude', + hookIsLive: true, + siblingAgent: 'codex', + allowSibling: true + }) + expect(optedIn).toMatchObject({ agent: 'claude', source: 'live-hook' }) + }) + + it('surfaces ambiguity instead of picking by array order', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + hookAgent: 'claude', + hookIsLive: false, + completedHookAgent: 'codex' + }) + expect(identity).toMatchObject({ agent: null, ambiguousAt: 'completed-hook' }) + }) +}) + +describe('reclaim-versus-stale-hook discriminator (run keys, not title text)', () => { + const run1 = { authorityId: 'main:a', incarnation: 1 } + const run2 = { authorityId: 'main:a', incarnation: 2 } + const otherAuthority = { authorityId: 'renderer:b', incarnation: 9 } + + it('bug shape: hook and pane share the current run, so the completed hook wins over the title', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + completedHookAgent: 'claude', + completedHookRun: run1, + currentRun: run1, + title: 'STA-4011 Linux Antigravity Commit Messages - grok' + }) + expect(identity).toMatchObject({ agent: 'claude', source: 'completed-hook' }) + }) + + it('reclaim shape: a superseded hook is ineligible and the current title evidence answers', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + completedHookAgent: 'claude', + completedHookRun: run1, + currentRun: run2, + title: 'STA-4011 Linux Antigravity Commit Messages - grok' + }) + expect(identity).toMatchObject({ + agent: 'grok', + source: 'title', + coverage: 'uncovered', + titleOnly: true + }) + expect(identity.supersededSources).toEqual(['completed-hook']) + }) + + it('cross-authority runs are incomparable, so the hook stays eligible', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + completedHookAgent: 'claude', + completedHookRun: otherAuthority, + currentRun: run2, + title: 'STA-4011 Linux Antigravity Commit Messages - grok' + }) + expect(identity).toMatchObject({ agent: 'claude', source: 'completed-hook' }) + }) + + it('an absent run key keeps evidence eligible (old peer), never guessed stale', () => { + const identity = resolveCanonicalPaneAgentIdentity({ + completedHookAgent: 'claude', + currentRun: run2, + title: 'STA-4011 Linux Antigravity Commit Messages - grok' + }) + expect(identity).toMatchObject({ agent: 'claude', source: 'completed-hook' }) + }) +}) + +describe('action floor', () => { + it("minimumSource: 'launch' refuses title and completed-hook answers outright", () => { + const identity = resolveCanonicalPaneAgentIdentity({ + completedHookAgent: 'claude', + title: 'claude', + minimumSource: 'launch' + }) + expect(identity).toMatchObject({ agent: null, source: null, coverage: 'covered' }) + }) +}) + +describe('wire evidence projection', () => { + it('publishes nothing for an absent identity — absence stays absence', () => { + expect( + buildPaneAgentIdentityEvidenceWire(resolveCanonicalPaneAgentIdentity({})) + ).toBeUndefined() + }) + + it('marks the uncovered title-only route explicitly and carries the run key when known', () => { + const wire = buildPaneAgentIdentityEvidenceWire( + resolveCanonicalPaneAgentIdentity({ title: 'claude - claude' }), + { authorityId: 'main:a', incarnation: 3 }, + { capturedAgeMs: 10, validForMs: 1_000 } + ) + expect(wire).toMatchObject({ + coverage: 'uncovered', + titleOnlyActionFallback: true, + authorityId: 'main:a', + incarnation: 3, + freshness: { capturedAgeMs: 10, validForMs: 1_000 } + }) + }) + + it('a covered identity never carries the title-only action marker', () => { + const wire = buildPaneAgentIdentityEvidenceWire( + resolveCanonicalPaneAgentIdentity({ launchAgent: 'claude' }) + ) + expect(wire).toMatchObject({ source: 'launch', coverage: 'covered' }) + expect(wire?.titleOnlyActionFallback).toBeUndefined() + }) +}) diff --git a/src/shared/pane-agent-identity-adapter.ts b/src/shared/pane-agent-identity-adapter.ts new file mode 100644 index 00000000000..c08e1b77dea --- /dev/null +++ b/src/shared/pane-agent-identity-adapter.ts @@ -0,0 +1,361 @@ +import { collectAgentTitleEvidence } from './agent-title-evidence' +import { PANE_AGENT_EVIDENCE_SOURCES } from './pane-agent-evidence-sources' +import type { + PaneAgentEvidence, + PaneAgentIdentity, + PaneAgentIdentityInput, + PaneAgentRunKey +} from './pane-agent-identity-resolver' +import type { PaneAgentEvidenceSource } from './pane-agent-evidence-sources' +import type { TuiAgent } from './tui-agent' + +/** + * Canonical pane identity ranking. All adapters, including the compatibility resolver, delegate to + * this implementation so source precedence, ambiguity, and run eligibility cannot drift. + */ + +/** + * Whether the execution authority proved at least one identity-bearing source for this pane. + * Computed from evidence presence, never from `platform`, `isRemote`, or OS: a remote pane with a + * host-stamped hook is covered; a local pane with only a title is uncovered. + */ +export type PaneAgentCoverage = 'covered' | 'uncovered' + +/** + * Host-stamped proof that a recognized agent process is the pane's foreground process. + * + * A process NAME is not a PID-reuse-safe identity, so a bare foreground read never enters the + * covered process rung. `processIncarnation` is an opaque token the execution host derives from + * the selected PID plus start/creation time (or an equivalent platform-native identity); the raw + * tuple never crosses the renderer/remote wire. The host emits no proof when the start identity + * is unavailable or ambiguous, so that pane reads `uncovered` rather than guessed. + */ +export type ForegroundProcessProof = { + agent: TuiAgent + /** Opaque host-derived PID+start-time token. Compared for equality only, never decoded. */ + processIncarnation: string + ptyIncarnationId?: string + /** The execution authority that stamped the proof (see agent-status-observation.ts). */ + authorityId: string + /** Age on the AUTHORITY's clock at capture. Replicas decay from this plus `validForMs`, + * never by subtracting a host wall clock from a local `Date.now()`. */ + capturedAgeMs: number + validForMs: number +} + +/** + * Positive evidence that a pane/process was REPLACED, required before any consumer may advance a + * pane incarnation. A retired-pane `restart` disposition, an accepted send, an ordinary provider + * turn boundary, a title change, a transport loss, or a renderer-only foreground change is never + * one of these. Defined here so the rebind-gate wave has a contract to be correct against; no + * sequencer call site consumes it yet. + */ +export type PaneReplacementProof = + | { kind: 'accepted-launch'; launchToken: string; ptyIncarnationId: string } + | { kind: 'process-replacement'; processIncarnation: string; authorityId: string } + | { kind: 'provider-session-attach'; providerSessionId: string } + +/** + * The optional wire object a host will publish alongside `agentIdentity` after capability + * negotiation (host-publisher wave, not now). All fields bounded and JSON-safe; old peers ignore + * it. Never inferred from the bare `agentIdentity` string. + */ +export type PaneAgentIdentityEvidenceWire = { + source: PaneAgentEvidenceSource + coverage: PaneAgentCoverage + authorityId?: string + incarnation?: number + freshness?: { capturedAgeMs: number; validForMs: number } + /** Marks the scoped host-published title-only best-effort route (hand-started WSL panes). + * Counted separately, `unverifiable` for liveness, and never relabeled as covered proof. */ + titleOnlyActionFallback?: true +} + +export type CanonicalPaneAgentIdentityInput = { + hookAgent?: TuiAgent | null + hookIsLive?: boolean + hookRun?: PaneAgentRunKey + /** A distinct completed-hook signal for callers that hold live and completed rows separately + * (the tab ladder does); `hookAgent` + `hookIsLive: false` remains the single-slot spelling. */ + completedHookAgent?: TuiAgent | null + completedHookRun?: PaneAgentRunKey + launchAgent?: TuiAgent | null + launchRun?: PaneAgentRunKey + /** + * Foreground process NAME as currently read. Without a fresh `processProof` this is a weak + * hint: it neither enters the covered process rung nor makes the pane covered. + */ + foregroundAgent?: TuiAgent | null + processProof?: ForegroundProcessProof | null + sleepingSessionAgent?: TuiAgent | null + sleepingRun?: PaneAgentRunKey + /** Tab-level display fallback only; ignored unless `allowSibling` opts in. */ + siblingAgent?: TuiAgent | null + /** Additional tab-level sibling observations retained for ambiguity checking. */ + siblingAgents?: readonly TuiAgent[] + allowSibling?: boolean + title?: string | null + currentRun?: PaneAgentRunKey + minimumSource?: PaneAgentEvidenceSource + /** + * The caller's CURRENT ladder result, preserved verbatim while the pane is uncovered. The + * uncovered lane is a temporary compatibility lane, not a new host-specific ranking; absent a + * fallback, an uncovered pane answers from title evidence alone, marked title-only. + */ + uncoveredFallback?: { agent: TuiAgent | null; titleOnly?: boolean } +} + +export type CanonicalPaneAgentIdentity = { + agent: TuiAgent | null + source: PaneAgentEvidenceSource | null + coverage: PaneAgentCoverage + /** True when the answer was derived from a parsed title (the uncovered/title-only marking). */ + titleOnly: boolean + ambiguousAt?: PaneAgentEvidenceSource + supersededSources: readonly PaneAgentEvidenceSource[] +} + +/** Authority order, strongest first. This is the only place precedence is expressed. */ +const SOURCE_RANK: readonly PaneAgentEvidenceSource[] = PANE_AGENT_EVIDENCE_SOURCES + +/** Exported for the source/rank drift ratchet; the rank is the canonical source list itself. */ +export const PANE_AGENT_SOURCE_RANK = SOURCE_RANK + +/** Reject an unrecognised source instead of silently dropping it from the ranking loop. */ +function sourceRankIndex(source: PaneAgentEvidenceSource): number { + const index = SOURCE_RANK.indexOf(source) + if (index === -1) { + throw new Error(`Unknown pane-agent evidence source: ${String(source)}`) + } + return index +} + +/** Run keys only supersede evidence from the same authority; unknown authorities stay eligible. */ +function isPaneAgentRunEligible( + run: PaneAgentRunKey | undefined, + currentRun: PaneAgentRunKey | undefined +): boolean { + return ( + run === undefined || + currentRun === undefined || + run.authorityId !== currentRun.authorityId || + run.incarnation === currentRun.incarnation + ) +} + +/** Shared evidence ranking primitive used by every pane-identity adapter. */ +export function resolveCanonicalPaneAgentEvidence( + input: PaneAgentIdentityInput +): PaneAgentIdentity { + const superseded: PaneAgentEvidenceSource[] = [] + const floor = input.minimumSource ? sourceRankIndex(input.minimumSource) : Number.MAX_SAFE_INTEGER + const eligible = input.evidence.filter((item) => { + if (item.source === 'sibling' && input.allowSibling !== true) { + return false + } + if (sourceRankIndex(item.source) > floor) { + return false + } + if (isPaneAgentRunEligible(item.run, input.currentRun)) { + return true + } + superseded.push(item.source) + return false + }) + + for (const source of SOURCE_RANK) { + const matches = eligible.filter((item) => item.source === source) + if (matches.length === 0) { + continue + } + const agents = new Set(matches.map((item) => item.agent)) + if (agents.size > 1) { + return { agent: null, source: null, ambiguousAt: source, supersededSources: superseded } + } + return { agent: matches[0].agent, source, supersededSources: superseded } + } + return { agent: null, source: null, supersededSources: superseded } +} + +/** Freshness is judged on the authority's own clock: age at capture against its TTL. */ +export function isForegroundProcessProofFresh(proof: ForegroundProcessProof): boolean { + return ( + Number.isFinite(proof.capturedAgeMs) && + Number.isFinite(proof.validForMs) && + proof.capturedAgeMs >= 0 && + proof.validForMs > 0 && + proof.capturedAgeMs <= proof.validForMs + ) +} + +/** A proof only carries identity for the agent it names; a name mismatch is no proof at all. */ +function processEvidenceFromProof( + input: CanonicalPaneAgentIdentityInput +): PaneAgentEvidence | null { + const proof = input.processProof + if (!proof || !isForegroundProcessProofFresh(proof)) { + return null + } + if (input.foregroundAgent && input.foregroundAgent !== proof.agent) { + return null + } + return { source: 'process', agent: proof.agent } +} + +export function resolveCanonicalPaneAgentIdentity( + input: CanonicalPaneAgentIdentityInput +): CanonicalPaneAgentIdentity { + const processEvidence = processEvidenceFromProof(input) + // Coverage comes from authority-bearing sources that are still eligible for this run. A stale + // hook/launch row can remain in the input after a pane is replaced; it must not make a title-only + // answer look covered to a future action consumer. + const covered = Boolean( + (input.hookAgent && isPaneAgentRunEligible(input.hookRun, input.currentRun)) || + (input.completedHookAgent && + isPaneAgentRunEligible(input.completedHookRun, input.currentRun)) || + processEvidence || + (input.launchAgent && isPaneAgentRunEligible(input.launchRun, input.currentRun)) || + (input.sleepingSessionAgent && isPaneAgentRunEligible(input.sleepingRun, input.currentRun)) + ) + // Keep stale evidence in the resolver so diagnostics still report which source was superseded, + // even when it no longer qualifies the pane as covered. + const hasAuthorityEvidence = Boolean( + input.hookAgent || + input.completedHookAgent || + processEvidence || + input.launchAgent || + input.sleepingSessionAgent + ) + const titleEvidence = input.title ? collectAgentTitleEvidence(input.title) : null + const titleAgent = titleEvidence?.agent ?? null + + if (!hasAuthorityEvidence) { + if (input.uncoveredFallback) { + const agent = input.uncoveredFallback.agent + // A legacy title parser may have picked the first token from an ambiguous or + // free-text-only title. Do not let that compatibility value bypass the canonical + // ambiguity fence when the caller marks it as title-only evidence. + const rejectTitleFallback = + input.uncoveredFallback.titleOnly === true && + ((titleEvidence?.reason === 'free-text-only' && + (titleEvidence.freeTextNames?.length ?? 0) > 1) || + titleEvidence?.reason === 'conflicting-anchored-names' || + titleEvidence?.reason === 'conflicting-vendor-markers') + if (rejectTitleFallback) { + return { + agent: null, + source: null, + coverage: 'uncovered', + titleOnly: false, + ...(titleEvidence?.reason === 'free-text-only' ? {} : { ambiguousAt: 'title' as const }), + supersededSources: [] + } + } + const titleOnly = + input.uncoveredFallback.titleOnly ?? (agent !== null && agent === titleAgent) + return { + agent, + source: agent === null ? null : titleOnly ? 'title' : null, + coverage: 'uncovered', + titleOnly, + supersededSources: [] + } + } + const siblingEvidence = [ + ...(input.siblingAgent ? [{ source: 'sibling' as const, agent: input.siblingAgent }] : []), + ...(input.siblingAgents?.map((agent) => ({ source: 'sibling' as const, agent })) ?? []), + ...(titleAgent ? [{ source: 'title' as const, agent: titleAgent }] : []) + ] + const siblingResolved = resolveCanonicalPaneAgentEvidence({ + evidence: siblingEvidence, + allowSibling: input.allowSibling, + minimumSource: input.minimumSource + }) + return { + agent: siblingResolved.agent, + source: siblingResolved.source, + coverage: 'uncovered', + titleOnly: siblingResolved.source === 'title', + ...(siblingResolved.ambiguousAt ? { ambiguousAt: siblingResolved.ambiguousAt } : {}), + supersededSources: siblingResolved.supersededSources + } + } + + const resolved = resolveCanonicalPaneAgentEvidence({ + evidence: [ + ...(input.hookAgent + ? [ + { + source: input.hookIsLive ? ('live-hook' as const) : ('completed-hook' as const), + agent: input.hookAgent, + ...(input.hookRun ? { run: input.hookRun } : {}) + } + ] + : []), + ...(input.completedHookAgent + ? [ + { + source: 'completed-hook' as const, + agent: input.completedHookAgent, + ...(input.completedHookRun ? { run: input.completedHookRun } : {}) + } + ] + : []), + ...(processEvidence ? [processEvidence] : []), + ...(input.launchAgent + ? [ + { + source: 'launch' as const, + agent: input.launchAgent, + ...(input.launchRun ? { run: input.launchRun } : {}) + } + ] + : []), + ...(input.sleepingSessionAgent + ? [ + { + source: 'sleeping-session' as const, + agent: input.sleepingSessionAgent, + ...(input.sleepingRun ? { run: input.sleepingRun } : {}) + } + ] + : []), + ...(input.siblingAgent ? [{ source: 'sibling' as const, agent: input.siblingAgent }] : []), + ...(input.siblingAgents?.map((agent) => ({ source: 'sibling' as const, agent })) ?? []), + ...(titleAgent ? [{ source: 'title' as const, agent: titleAgent }] : []) + ], + currentRun: input.currentRun, + minimumSource: input.minimumSource, + allowSibling: input.allowSibling + }) + return { + agent: resolved.agent, + source: resolved.source, + coverage: covered ? 'covered' : 'uncovered', + titleOnly: resolved.source === 'title', + ...(resolved.ambiguousAt ? { ambiguousAt: resolved.ambiguousAt } : {}), + supersededSources: resolved.supersededSources + } +} + +/** Projects the host-local sidecar onto the optional wire shape. Returns undefined when there is + * nothing to publish — absence stays absence, and a bare `agentIdentity` with no sidecar is + * never treated as covered proof by any consumer. */ +export function buildPaneAgentIdentityEvidenceWire( + identity: CanonicalPaneAgentIdentity, + run?: PaneAgentRunKey, + freshness?: { capturedAgeMs: number; validForMs: number } +): PaneAgentIdentityEvidenceWire | undefined { + if (identity.agent === null || identity.source === null) { + return undefined + } + return { + source: identity.source, + coverage: identity.coverage, + ...(run ? { authorityId: run.authorityId, incarnation: run.incarnation } : {}), + ...(freshness ? { freshness } : {}), + ...(identity.coverage === 'uncovered' && identity.titleOnly + ? { titleOnlyActionFallback: true as const } + : {}) + } +} diff --git a/src/shared/pane-agent-identity-inventory.test.ts b/src/shared/pane-agent-identity-inventory.test.ts index 6871a38eb6d..ee868bfcc16 100644 --- a/src/shared/pane-agent-identity-inventory.test.ts +++ b/src/shared/pane-agent-identity-inventory.test.ts @@ -18,7 +18,13 @@ const HELPERS = [ 'resolveExplicitTerminalTitleAgentType', 'resolveCommittedTitleAgentType', 'resolvePaneAgentOwner', - 'resolveCompatibleAgentTypeForOwner' + 'resolveCompatibleAgentTypeForOwner', + 'classifyTitleActivity', + 'detectAgentStatusFromTitle', + 'resolveAgentTypeFromTerminalTitle', + 'resolvePaneAgentIdentity', + 'resolveCanonicalPaneAgentIdentity', + 'resolvePublishedPaneAgentIdentity' ] as const const TEST_SUPPORT_PATHS = new Set([ @@ -242,6 +248,136 @@ const INVENTORY: readonly InventoryGroup[] = [ ['src/renderer/src/components/terminal-pane/pty-connection/direct-ssh-retry-status.ts', 2], ['src/renderer/src/components/terminal-pane/pty-connection/title-spawn-bell.ts', 2] ] + }, + { + helper: 'classifyTitleActivity', + classification: 'identity-consumer', + paths: [ + ['src/renderer/src/components/sidebar/smart-attention.ts', 3], + ['src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts', 2], + ['src/renderer/src/components/status-bar/workspace-space-presentation.ts', 3], + ['src/renderer/src/lib/active-agent-note-target.ts', 2], + ['src/renderer/src/lib/worktree-status.ts', 3], + ['src/renderer/src/store/slices/terminal-helpers.ts', 2] + ] + }, + { + helper: 'classifyTitleActivity', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/terminal-pane/cache-timer-seeding.ts', 2], + ['src/renderer/src/lib/agent-ready-wait.ts', 2], + ['src/renderer/src/store/terminals/terminal-ephemeral-state.ts', 2] + ] + }, + { + helper: 'classifyTitleActivity', + classification: 'activity-only', + paths: [ + ['src/renderer/src/store/slices/workspace-cleanup-local-evidence.ts', 3], + ['src/renderer/src/store/terminals/terminal-tab-presentation.ts', 4] + ] + }, + { + helper: 'classifyTitleActivity', + classification: 'evidence-producer', + paths: [ + ['src/renderer/src/lib/agent-send-title-status.ts', 2], + ['src/renderer/src/lib/agent-status-terminal-title.ts', 2] + ] + }, + { + helper: 'classifyTitleActivity', + classification: 'parser-implementation', + paths: [ + ['src/renderer/src/lib/agent-status.ts', 5], + 'src/renderer/src/lib/pane-agent-evidence.ts' + ] + }, + { + helper: 'detectAgentStatusFromTitle', + classification: 'evidence-producer', + paths: [ + ['src/main/runtime/orca-runtime-apply-tracked-pty-title.ts', 2], + ['src/main/runtime/orca-runtime-get-pty-record-for-pane-key.ts', 2], + ['src/main/runtime/orca-runtime-get-unpersisted-tracked-title-for-pty.ts', 2], + ['src/main/runtime/orca-runtime-maybe-hydrate-headless-from-renderer.ts', 2], + ['src/main/runtime/orca-runtime-record-agent-prompt-lifecycle-state.ts', 2], + ['src/main/runtime/runtime-terminal-agent-status-query.ts', 3], + ['src/main/runtime/runtime-worktree-status-projection.ts', 4], + ['src/main/runtime/terminal-wait-detection.ts', 2], + ['src/renderer/src/components/terminal-pane/agent-completion-title-observer.ts', 2], + ['src/renderer/src/components/terminal-pane/pty-connection/shell-command-inference.ts', 4], + ['src/renderer/src/components/terminal-pane/pty-output-title-observer.ts', 2], + ['src/shared/terminal-output-side-effects.ts', 3] + ] + }, + { + helper: 'detectAgentStatusFromTitle', + classification: 'action-consumer', + paths: [ + ['src/renderer/src/components/terminal-pane/pty-connection/agent-task-complete-notify.ts', 2], + [ + 'src/renderer/src/components/terminal-pane/pty-connection/command-inferred-pane-agent.ts', + 3 + ], + ['src/renderer/src/components/terminal-pane/pty-connection/interrupt-input-intent.ts', 3] + ] + }, + { + helper: 'detectAgentStatusFromTitle', + classification: 'parser-implementation', + paths: [ + ['src/renderer/src/components/terminal-pane/title-agent-identity.ts', 2], + 'src/renderer/src/lib/agent-status.ts', + ['src/renderer/src/lib/pane-agent-evidence.ts', 3], + ['src/shared/agent-decorative-title-signature.ts', 2], + 'src/shared/agent-detection.ts', + ['src/shared/agent-title-owner.ts', 2], + ['src/shared/agent-title-status.ts', 6] + ] + }, + { + helper: 'resolveAgentTypeFromTerminalTitle', + classification: 'identity-consumer', + paths: [ + ['src/renderer/src/components/sidebar/worktree-agent-row-type.ts', 2], + 'src/renderer/src/components/sidebar/worktree-title-derived-agent-rows.ts', + ['src/renderer/src/lib/worktree-status.ts', 2] + ] + }, + { + helper: 'resolvePaneAgentIdentity', + classification: 'parser-implementation', + paths: ['src/shared/pane-agent-identity-resolver.ts'] + }, + { + helper: 'resolvePaneAgentIdentity', + classification: 'identity-consumer', + paths: [['src/shared/published-pane-agent-identity.ts', 2]] + }, + { + helper: 'resolveCanonicalPaneAgentIdentity', + classification: 'parser-implementation', + paths: ['src/shared/pane-agent-identity-adapter.ts'] + }, + { + helper: 'resolveCanonicalPaneAgentIdentity', + classification: 'identity-consumer', + paths: [['src/shared/agent-status-identity.ts', 2]] + }, + { + helper: 'resolveCanonicalPaneAgentIdentity', + classification: 'identity-consumer', + paths: [['src/shared/terminal-title-agent-type.ts', 2]] + }, + { + helper: 'resolvePublishedPaneAgentIdentity', + classification: 'parser-implementation', + paths: [ + 'src/shared/published-pane-agent-identity.ts', + ['src/main/runtime/orca-runtime-write-orchestration-pointer-pty.ts', 2] + ] } ] diff --git a/src/shared/pane-agent-identity-resolver.test.ts b/src/shared/pane-agent-identity-resolver.test.ts index e28fb0cbd22..2f14a54976a 100644 --- a/src/shared/pane-agent-identity-resolver.test.ts +++ b/src/shared/pane-agent-identity-resolver.test.ts @@ -1,4 +1,5 @@ import { describe, expect, it } from 'vitest' +import { PANE_AGENT_SOURCE_RANK } from './pane-agent-identity-adapter' import { PANE_AGENT_EVIDENCE_SOURCES, type PaneAgentEvidence, @@ -11,6 +12,14 @@ const resolve = (evidence: PaneAgentEvidence[], extra = {}) => const H = 'authority-a' describe('resolvePaneAgentIdentity', () => { + it('keeps every evidence source ranked exactly once', () => { + expect(PANE_AGENT_SOURCE_RANK).toBe(PANE_AGENT_EVIDENCE_SOURCES) + expect(new Set(PANE_AGENT_SOURCE_RANK).size).toBe(PANE_AGENT_SOURCE_RANK.length) + for (const source of PANE_AGENT_EVIDENCE_SOURCES) { + expect(PANE_AGENT_SOURCE_RANK.indexOf(source)).toBeGreaterThanOrEqual(0) + } + }) + describe('a display title is the last thing consulted', () => { it.each(PANE_AGENT_EVIDENCE_SOURCES.filter((s) => s !== 'title' && s !== 'sibling'))( 'lets %s outrank a conflicting title', @@ -153,6 +162,17 @@ describe('resolvePaneAgentIdentity', () => { }) }) + it('fails loudly when an evidence source is missing from the rank', () => { + expect(() => + resolve([ + { + source: 'future-source' as PaneAgentEvidence['source'], + agent: 'codex' + } + ]) + ).toThrow('Unknown pane-agent evidence source') + }) + describe('input order does not decide the answer', () => { it('resolves the same regardless of how evidence is listed', () => { const evidence: PaneAgentEvidence[] = [ diff --git a/src/shared/pane-agent-identity-resolver.ts b/src/shared/pane-agent-identity-resolver.ts index 98342963415..4b03c5de16d 100644 --- a/src/shared/pane-agent-identity-resolver.ts +++ b/src/shared/pane-agent-identity-resolver.ts @@ -1,5 +1,10 @@ +import { resolveCanonicalPaneAgentEvidence } from './pane-agent-identity-adapter' +import type { PaneAgentEvidenceSource } from './pane-agent-evidence-sources' import type { TuiAgent } from './tui-agent' +export { PANE_AGENT_EVIDENCE_SOURCES } from './pane-agent-evidence-sources' +export type { PaneAgentEvidenceSource } from './pane-agent-evidence-sources' + /** * One place that answers "which agent is in this pane". * @@ -23,27 +28,6 @@ import type { TuiAgent } from './tui-agent' * launch, a recognized command at a shell prompt, a host-confirmed foreground change, a new * provider session. Never by a title changing, and never by transport loss. */ -export const PANE_AGENT_EVIDENCE_SOURCES = [ - /** A live provider hook for a turn in progress. The agent is running and said so. */ - 'live-hook', - /** The pane's foreground process, as read on the execution host. */ - 'process', - /** Orca launched, resumed, or accepted a command for this agent. A fact Orca owns. */ - 'launch', - /** A provider hook from a turn that finished. Still authoritative about identity. */ - 'completed-hook', - /** A sleeping session record restored for this pane. */ - 'sleeping-session', - /** Another pane in the same tab. Tab-level surfaces only; never pane-scoped routing. */ - 'sibling', - /** Parsed from the terminal title. A decoration channel; anyone can type an agent's name. */ - 'title' -] as const -export type PaneAgentEvidenceSource = (typeof PANE_AGENT_EVIDENCE_SOURCES)[number] - -/** Authority order, strongest first. Position here is the ONLY place precedence is expressed. */ -const SOURCE_RANK: readonly PaneAgentEvidenceSource[] = PANE_AGENT_EVIDENCE_SOURCES - /** * Which agent run a piece of evidence belongs to. * @@ -112,52 +96,5 @@ export type PaneAgentIdentity = { export function resolvePaneAgentIdentity( input: PaneAgentIdentityInput ): PaneAgentIdentity { - const superseded: PaneAgentEvidenceSource[] = [] - const floor = input.minimumSource - ? SOURCE_RANK.indexOf(input.minimumSource) - : Number.MAX_SAFE_INTEGER - - const eligible = input.evidence.filter((item) => { - if (item.source === 'sibling' && input.allowSibling !== true) { - return false - } - // Why the floor: an action consumer must not be able to act on a title, at any rank. Dropping - // the evidence entirely rather than ranking it lower makes misuse impossible rather than - // unlikely — a caller cannot accidentally consult it by reordering. - if (SOURCE_RANK.indexOf(item.source) > floor) { - return false - } - if (input.currentRun === undefined || item.run === undefined) { - // Why eligible: absence means "this peer does not publish run keys", not "this is stale". - // Treating unknown as superseded would blank every row from an older host. - return true - } - if (item.run.authorityId !== input.currentRun.authorityId) { - // Why eligible and NOT superseded: runs from different authorities are incomparable, not - // older. A restarted main counts from its own floor, so `incarnation` alone would falsely - // equate unrelated runs. Incomparable evidence is treated as unknown, like an absent key. - return true - } - if (item.run.incarnation === input.currentRun.incarnation) { - return true - } - superseded.push(item.source) - return false - }) - - for (const source of SOURCE_RANK) { - const matches = eligible.filter((item) => item.source === source) - if (matches.length === 0) { - continue - } - const agents = new Set(matches.map((item) => item.agent)) - if (agents.size > 1) { - // Why null and not the first: two observations of the same class naming different agents is - // a genuine conflict, and picking one would make the answer depend on array order — the very - // property this resolver exists to remove. Fall through to nothing rather than guess. - return { agent: null, source: null, ambiguousAt: source, supersededSources: superseded } - } - return { agent: matches[0].agent, source, supersededSources: superseded } - } - return { agent: null, source: null, supersededSources: superseded } + return resolveCanonicalPaneAgentEvidence(input) } diff --git a/src/shared/pane-agent-identity-surface-inventory.test.ts b/src/shared/pane-agent-identity-surface-inventory.test.ts new file mode 100644 index 00000000000..77d9edb3b84 --- /dev/null +++ b/src/shared/pane-agent-identity-surface-inventory.test.ts @@ -0,0 +1,284 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { glob } from 'tinyglobby' +import { isTestFile, stripComments } from './source-scan/source-tree-scan' + +/** + * Surface half of the identity inventory ratchet: every consumer decision point from the closed + * 65-row inventory (rows 32–65 — the direct title/native-chat selectors, tab projections, mobile + * sync graph, lifecycle selectors, status/OSC ingress, worktree status, attention, and + * title-reset paths) is pinned to a marker symbol in its file. The helper-name census + * (`pane-agent-identity-inventory.test.ts`) is necessary but not sufficient — these files reach + * identity through direct reads a name census cannot see. Moving or renaming a marker means the + * inventory row must be re-classified, deliberately, before review. + */ + +type SurfaceRow = { + /** Row number in the closed consumer inventory. */ + row: number + path: string + marker: string +} + +const SURFACE_ROWS: readonly SurfaceRow[] = [ + { + row: 32, + path: 'src/renderer/src/components/terminal-pane/native-chat-leaf-title-agent.ts', + marker: 'resolveNativeChatLeafTitleAgent' + }, + { + row: 32, + path: 'src/renderer/src/components/terminal-pane/use-terminal-pane-chat-state.ts', + marker: 'resolveNativeChatLeafTitleAgent' + }, + { + row: 33, + path: 'src/renderer/src/components/terminal-pane/pty-connection/pane-agent-identity.ts', + marker: 'installPaneAgentIdentity' + }, + { row: 34, path: 'src/main/runtime/orchestration/groups.ts', marker: 'terminalIsAgent' }, + { + row: 35, + path: 'src/renderer/src/lib/active-agent-note-target.ts', + marker: 'getActiveTerminalNoteTarget' + }, + { + row: 36, + path: 'src/renderer/src/components/terminal-pane/terminal-agent-paste-bracketing.ts', + marker: 'resolveProtectedMultilinePasteOptionsForPane' + }, + { + row: 37, + path: 'src/renderer/src/components/terminal-pane/command-code-output-ownership.ts', + marker: 'canCommandCodeOutputOwnPane' + }, + { row: 38, path: 'src/renderer/src/lib/agent-ready-wait.ts', marker: 'waitForAgentReady' }, + { + row: 39, + path: 'src/renderer/src/lib/agent-paste-draft.ts', + marker: 'getSettingsForAgentTabRuntimeOwner' + }, + { + row: 40, + path: 'src/renderer/src/lib/agent-followup-delivery.ts', + marker: 'sendFollowupPromptWhenAgentReady' + }, + { + row: 41, + path: 'src/renderer/src/lib/codex-session-restart.ts', + marker: 'markLiveCodexSessionsForRestart' + }, + { + row: 41, + path: 'src/renderer/src/lib/codex-pane-restart-eligibility.ts', + marker: 'isCodexForegroundProcess' + }, + { + row: 42, + path: 'src/renderer/src/components/native-chat/native-chat-availability.ts', + marker: 'canToggleNativeChat' + }, + { + row: 43, + path: 'src/renderer/src/components/native-chat/native-chat-pane-resolution.ts', + marker: 'resolveNativeChatSession' + }, + { + row: 44, + path: 'src/renderer/src/components/terminal-pane/terminal-agent-session-continuation.ts', + marker: 'canContinueAgentSessionInNewSession' + }, + { + row: 45, + path: 'src/renderer/src/components/terminal-pane/terminal-agent-session-fork.ts', + marker: 'prepareAgentSessionForkFromPane' + }, + { + row: 46, + path: 'src/renderer/src/components/terminal-pane/agent-interrupt-inference.ts', + marker: 'isPlainEscapeKeyEvent' + }, + { + row: 46, + path: 'src/renderer/src/components/terminal-pane/agent-question-answered-inference.ts', + marker: 'inferQuestionAnsweredFromCurrentStatus' + }, + { + row: 47, + path: 'src/renderer/src/components/terminal-pane/terminal-keyboard-protocol-pane-agent.ts', + marker: 'resolvePaneKeyboardProtocolAgent' + }, + { + row: 47, + path: 'src/renderer/src/components/terminal-pane/terminal-pane-manager-options.ts', + marker: 'resolvePaneKeyboardProtocolAgent' + }, + { + row: 48, + path: 'src/renderer/src/components/tab-bar/tab-agent-types-by-tab-id.ts', + marker: 'selectTabAgentTypesByTabId' + }, + { + row: 49, + path: 'src/renderer/src/components/terminal-pane/terminal-tab-agent-type-index.ts', + marker: 'createTerminalTabAgentTypeSelector' + }, + { + row: 50, + path: 'src/renderer/src/lib/tab-agent-status-index.ts', + marker: 'selectLiveTabAgentPanes' + }, + { + row: 51, + path: 'src/renderer/src/components/tab-bar/terminal-tab-activity-status.ts', + marker: 'resolveTerminalTabActivityStatus' + }, + { + row: 52, + path: 'src/renderer/src/lib/workspace-tab-agent-metadata.ts', + marker: 'maxAgentActivityAt' + }, + { + row: 52, + path: 'src/renderer/src/lib/workspace-tab-palette-entry-builder.ts', + marker: 'buildSearchableWorkspaceTabEntries' + }, + { + row: 53, + path: 'src/renderer/src/lib/running-agent-targets.ts', + marker: 'deriveRunningAgentSendTargets' + }, + { + row: 54, + path: 'src/renderer/src/runtime/sync-runtime-graph.ts', + marker: 'buildMobileSessionTabSnapshots' + }, + { + row: 55, + path: 'src/renderer/src/lib/agent-hibernation-pane-eligibility.ts', + marker: 'toRuntimePtyId' + }, + { + row: 56, + path: 'src/renderer/src/lib/resume-sleeping-agent-session.ts', + marker: 'resumeSleepingAgentSessionsForWorktree' + }, + { + row: 57, + path: 'src/renderer/src/lib/automation-session-reuse.ts', + marker: 'findReusableAutomationSession' + }, + { + row: 58, + path: 'src/renderer/src/components/terminal-pane/pty-connection/cold-restore-resume-startup.ts', + marker: 'bindBuildColdRestoreAgentResumeStartup' + }, + { + row: 59, + path: 'src/main/agent-hooks/server/server-authority-evidence.ts', + marker: 'recordCurrentAuthorityObservation' + }, + { + row: 59, + path: 'src/main/runtime/orca-runtime-write-orchestration-pointer-pty.ts', + marker: 'resolvePaneAgentIdentityField' + }, + { + row: 59, + path: 'src/renderer/src/hooks/ipc-events/agent-status-event-applicator.ts', + marker: 'createAgentStatusEventApplicator' + }, + { + row: 59, + path: 'src/renderer/src/store/slices/agent-status-authority-actions.ts', + marker: 'transferAgentPaneAuthority' + }, + { + row: 59, + path: 'src/renderer/src/store/slices/pane-foreground-agent.ts', + marker: 'createPaneForegroundAgentSlice' + }, + { + row: 59, + path: 'src/renderer/src/hooks/ipc-events/agent-status-routing.ts', + marker: 'isAgentStatusForRecentlyClosedTab' + }, + { + row: 60, + path: 'src/renderer/src/components/terminal-pane/pty-connection/title-spawn-bell.ts', + marker: 'installTitleSpawnBell' + }, + { row: 61, path: 'src/renderer/src/lib/worktree-status.ts', marker: 'getWorktreeStatus' }, + { + row: 62, + path: 'src/main/runtime/runtime-worktree-status-projection.ts', + marker: 'getLeafWorktreeStatus' + }, + { + row: 63, + path: 'src/renderer/src/components/sidebar/smart-attention.ts', + marker: 'buildAttentionByWorktree' + }, + { + row: 64, + path: 'src/renderer/src/components/status-bar/workspace-space-presentation.ts', + marker: 'countWorkspaceSpaceActiveAgents' + }, + { row: 65, path: 'src/renderer/src/store/slices/terminal-helpers.ts', marker: 'getResetTitle' }, + { + row: 6, + path: 'src/renderer/src/runtime/web-session-tabs-sync.ts', + marker: 'applyWebSessionTabs' + } +] + +describe('pane agent identity surface inventory (rows 6, 32–65)', () => { + it('every pinned surface still carries its marker symbol', () => { + for (const row of SURFACE_ROWS) { + const source = stripComments(readFileSync(join(process.cwd(), row.path), 'utf8')) + expect({ row: row.row, path: row.path, hasMarker: source.includes(row.marker) }).toEqual({ + row: row.row, + path: row.path, + hasMarker: true + }) + } + }) +}) + +/** + * Identity-observation rebind audit. Advancing a pane incarnation without a positive replacement + * proof is how a legitimate reclaim and a stale-hook bug get conflated (see + * `PaneReplacementProof` in pane-agent-identity-adapter.ts). Every existing sequencer `rebind` + * call is pinned here by file and count: today they are the retired-pane `restart` disposition + * (three ingress paths) and the renderer pane-key transfer. Adding a rebind call, or changing + * these, requires updating this audit — and per the migration plan, a `replacementProof`. + */ +const IDENTITY_SEQUENCER_REBIND_RE = /\b(?:observations|rendererAgentStatusObservations)\.rebind\(/g + +const EXPECTED_REBIND_SITES: readonly (readonly [path: string, occurrences: number])[] = [ + ['src/main/agent-hooks/server/server-ingest-normalization.ts', 1], + ['src/main/agent-hooks/server/server-ingest-remote.ts', 1], + ['src/main/agent-hooks/server/server-lifecycle.ts', 1], + ['src/renderer/src/store/slices/agent-status-authority-actions.ts', 1] +] + +describe('identity observation rebind audit', () => { + it('pins every identity-sequencer rebind call site by file and count', async () => { + const files = await glob(['src/**/*.{ts,tsx}', 'mobile/src/**/*.{ts,tsx}'], { + ignore: ['**/*.test.*', '**/*.spec.*'] + }) + const actual: [string, number][] = [] + for (const path of files.sort()) { + if (isTestFile(path)) { + continue + } + const source = stripComments(readFileSync(join(process.cwd(), path), 'utf8')) + const occurrences = source.match(IDENTITY_SEQUENCER_REBIND_RE)?.length ?? 0 + if (occurrences > 0) { + actual.push([path, occurrences]) + } + } + expect(actual).toEqual(EXPECTED_REBIND_SITES.map((site) => [...site])) + }, 30_000) +}) diff --git a/src/shared/pane-agent-identity-title-corpus.test.ts b/src/shared/pane-agent-identity-title-corpus.test.ts new file mode 100644 index 00000000000..f24cd431065 --- /dev/null +++ b/src/shared/pane-agent-identity-title-corpus.test.ts @@ -0,0 +1,156 @@ +import { existsSync, readdirSync, readFileSync } from 'node:fs' +import { homedir } from 'node:os' +import { join } from 'node:path' +import { createHash, randomBytes } from 'node:crypto' +import { describe, expect, it } from 'vitest' +import { collectAgentTitleEvidence } from './agent-title-evidence' +import { resolveCanonicalPaneAgentIdentity } from './pane-agent-identity-adapter' +import type { TuiAgent } from './tui-agent' + +/** + * Title regression gates for the identity-ladder migration. + * + * Two layers: a controlled fixture table that always runs (CI-safe), and a local characterization + * gate over the machine's real recorded corpus. The corpus gate is not a CI prerequisite tied to + * one developer's home — when no history exists it reports `corpus unavailable — skipped` + * explicitly, never a silently green zero-title run. Raw titles never reach logs or failure + * output; changed titles are reported as salted hashes plus old/new agent summaries only. + */ + +const RECORDED_HISTORY_DIR = 'terminal-history' +const QUARANTINE_DIR = '.recovery-quarantine' + +function orcaAppSupportCandidates(): string[] { + if (process.platform === 'darwin') { + return [join(homedir(), 'Library', 'Application Support', 'Orca')] + } + if (process.platform === 'win32') { + return [join(process.env.APPDATA ?? join(homedir(), 'AppData', 'Roaming'), 'Orca')] + } + return [ + join(process.env.XDG_CONFIG_HOME ?? join(homedir(), '.config'), 'Orca'), + join(process.env.XDG_DATA_HOME ?? join(homedir(), '.local', 'share'), 'Orca') + ] +} + +/** Deliberately shallow: `terminal-history//checkpoint.json` only, with the hidden + * quarantine subtree excluded BY NAME so a future recursive rewrite cannot silently turn + * quarantined recovery data into product regressions. */ +function loadRecordedTitleCorpus(): { checkpointCount: number; titles: string[] } | null { + const root = orcaAppSupportCandidates() + .map((candidate) => join(candidate, RECORDED_HISTORY_DIR)) + .find((candidate) => existsSync(candidate)) + if (!root) { + return null + } + let checkpointCount = 0 + const titles = new Set() + for (const entry of readdirSync(root, { withFileTypes: true })) { + if (!entry.isDirectory() || entry.name === QUARANTINE_DIR || entry.name.startsWith('.')) { + continue + } + const checkpointPath = join(root, entry.name, 'checkpoint.json') + if (!existsSync(checkpointPath)) { + continue + } + const parsed: unknown = JSON.parse(readFileSync(checkpointPath, 'utf8')) + checkpointCount += 1 + const lastTitle = (parsed as { lastTitle?: unknown }).lastTitle + if (typeof lastTitle === 'string' && lastTitle.length > 0) { + titles.add(lastTitle) + } + } + return { checkpointCount, titles: [...titles] } +} + +/** What the canonical adapter answers when a title is all a pane has (the uncovered lane). */ +function canonicalTitleOnlyAgent(title: string): TuiAgent | null { + return resolveCanonicalPaneAgentIdentity({ title }).agent +} + +describe('controlled title fixtures (always run)', () => { + const FIXTURES: readonly { name: string; title: string; expected: TuiAgent | null }[] = [ + { + name: 'mandatory adversarial owner suffix beats the agent names in task text', + title: 'STA-4011 Linux Antigravity Commit Messages - grok', + expected: 'grok' + }, + { + name: 'task text mentioning other agents is not identity', + title: 'Compare Antigravity with Gemini 3.7 Flash', + expected: null + }, + { + name: 'owner suffix still answers over mentioned agents', + title: 'Compare Antigravity with Gemini 3.7 Flash… - grok', + expected: 'grok' + }, + { name: 'Claude status sigil is a vendor marker', title: '✳', expected: 'claude' }, + { name: 'Claude management screen is not identity', title: 'claude agents', expected: null }, + { name: 'a shell title names no agent', title: 'zsh', expected: null }, + { name: 'a default worktree-ish title names no agent', title: 'my-claude-fix', expected: null }, + { + name: 'conflicting vendor markers resolve to nothing', + title: '✳ | ✦ two sigils', + expected: null + }, + { + name: 'conflicting anchored names resolve to nothing', + title: 'OC | something… - grok', + expected: null + }, + { name: 'a bare Pi title anchors as Pi', title: 'pi', expected: 'pi' }, + { name: 'an OMP status title anchors as OMP', title: 'omp ready', expected: 'omp' }, + { + // Wrapper-frame π/OMP separators are handled by the synthetic-title path, not this + // evidence parser; pinned so a parser change here is a deliberate decision. + name: 'a π wrapper frame is declined by the evidence parser', + title: 'π : ready', + expected: null + } + ] + + for (const fixture of FIXTURES) { + it(fixture.name, () => { + expect(collectAgentTitleEvidence(fixture.title).agent).toBe(fixture.expected) + // The adapter's title-only lane must give the very same answer — phase 1 changes no + // parser semantics, only provenance. + expect(canonicalTitleOnlyAgent(fixture.title)).toBe(fixture.expected) + }) + } +}) + +describe('recorded title corpus characterization (local gate)', () => { + it('the canonical title-only lane matches the shipped parser on every recorded title', (ctx) => { + const corpus = loadRecordedTitleCorpus() + if (corpus === null) { + console.info('corpus unavailable — skipped (no recorded terminal history on this machine)') + ctx.skip() + return + } + // A machine WITH history must never pass on an empty read — that would be a silently green + // zero-title run, not a characterization. + expect(corpus.checkpointCount).toBeGreaterThan(0) + expect(corpus.titles.length).toBeGreaterThan(0) + console.info( + `corpus: ${corpus.checkpointCount} checkpoints, ${corpus.titles.length} distinct titles` + ) + + const salt = randomBytes(16).toString('hex') + const changed: { titleHash: string; oldAgent: string | null; newAgent: string | null }[] = [] + for (const title of corpus.titles) { + const oldAgent = collectAgentTitleEvidence(title).agent + const newAgent = canonicalTitleOnlyAgent(title) + if (oldAgent !== newAgent) { + changed.push({ + titleHash: createHash('sha256').update(`${salt}:${title}`).digest('hex').slice(0, 16), + oldAgent, + newAgent + }) + } + } + // Report hashes and agent summaries only; a reviewer who needs the raw value inspects the + // protected corpus on the machine that owns it. + expect(changed).toEqual([]) + }, 60_000) +}) diff --git a/src/shared/pane-agent-owner.test.ts b/src/shared/pane-agent-owner.test.ts index 13cfd217c45..b0d61801392 100644 --- a/src/shared/pane-agent-owner.test.ts +++ b/src/shared/pane-agent-owner.test.ts @@ -37,6 +37,42 @@ describe('resolvePaneAgentOwner', () => { ).toBe('omp') }) + it('preserves the pre-tranche precedence for every conflicting owner tier', () => { + expect( + resolvePaneAgentOwnerRecord({ + launchAgent: 'claude', + hookAgent: 'codex', + siblingHookAgent: 'gemini', + completedHookAgent: 'pi', + sleepingSessionAgent: 'omp' + }) + ).toEqual({ agent: 'claude', ownerIsLaunch: true }) + expect( + resolvePaneAgentOwnerRecord({ + hookAgent: 'claude', + siblingHookAgent: 'codex', + completedHookAgent: 'gemini', + siblingCompletedHookAgent: 'pi', + sleepingSessionAgent: 'omp' + }) + ).toEqual({ agent: 'claude', ownerIsLaunch: false }) + expect( + resolvePaneAgentOwnerRecord({ + siblingHookAgent: 'codex', + completedHookAgent: 'claude', + siblingCompletedHookAgent: 'gemini', + sleepingSessionAgent: 'omp' + }) + ).toEqual({ agent: 'codex', ownerIsLaunch: false }) + expect( + resolvePaneAgentOwnerRecord({ + completedHookAgent: 'claude', + siblingCompletedHookAgent: 'codex', + sleepingSessionAgent: 'gemini' + }) + ).toEqual({ agent: 'claude', ownerIsLaunch: false }) + }) + it('returns null when no owner evidence exists', () => { expect(resolvePaneAgentOwner({})).toBeNull() expect(resolvePaneAgentOwner({ launchAgent: null, hookAgent: undefined })).toBeNull() diff --git a/src/shared/pane-agent-owner.ts b/src/shared/pane-agent-owner.ts index 31b878d67ec..5b5563670e8 100644 --- a/src/shared/pane-agent-owner.ts +++ b/src/shared/pane-agent-owner.ts @@ -48,9 +48,10 @@ const PANE_OWNER_RANK: readonly { ] /** - * The single authoritative resolver for "which agent owns this pane", shared by - * the tab-icon resolver, the terminal-pane display/renderer owner, and the - * mirrored-tab title owner so they cannot drift apart. + * Compatibility owner lookup shared by the existing consumer surfaces. + * + * Tranche 0 intentionally preserves this pre-migration precedence byte-for-byte; switching + * these consumers to canonical evidence belongs to tranche 1. * * Why this precedence: launch intent is the authoritative bootstrap before any * process signal exists, so it leads. Once launch metadata is gone — a mirrored diff --git a/src/shared/terminal-title-agent-type.ts b/src/shared/terminal-title-agent-type.ts index cdbce788806..4a000f8be9b 100644 --- a/src/shared/terminal-title-agent-type.ts +++ b/src/shared/terminal-title-agent-type.ts @@ -10,6 +10,7 @@ import { getPiCompatibleSyntheticAgentLabel, isLegacyPiCompatibleTitle } from './pi-compatible-synthetic-title' +import { resolveCanonicalPaneAgentIdentity } from './pane-agent-identity-adapter' import { memoizeTitleClassification } from './terminal-title-classification-memo' import type { TuiAgent } from './tui-agent' @@ -217,10 +218,10 @@ function computeAgentLabel(title: string): string | null { return null } -// Maps getAgentLabel()'s product labels to TuiAgent ids — the fallback for -// agents whose foreground PROCESS name isn't self-identifying (Claude Code runs -// as `node`, but its "✳ Claude Code" title resolves here). Agents whose process -// name already matches (codex, etc.) never reach this path. +/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +export const getAgentLabel: (title: string) => string | null = + memoizeTitleClassification(computeAgentLabel) + const TITLE_LABEL_TO_AGENT: Partial> = { 'Claude Code': 'claude', OpenClaude: 'openclaude', @@ -240,10 +241,6 @@ const TITLE_LABEL_TO_AGENT: Partial> = { OMP: 'omp' } -/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ -export const getAgentLabel: (title: string) => string | null = - memoizeTitleClassification(computeAgentLabel) - function hasGenericClaudeStatusPrefix(title: string): boolean { return ( containsAgentSpinnerGlyph(title) || @@ -266,7 +263,13 @@ function isGenericClaudeStatusClaim(title: string, titleAgent: TuiAgent | null): export function resolveTerminalTitleAgentType(title: string): TuiAgent | null { const label = getAgentLabel(title) - return label ? (TITLE_LABEL_TO_AGENT[label] ?? null) : null + const parsed = label ? (TITLE_LABEL_TO_AGENT[label] ?? null) : null + return resolveCanonicalPaneAgentIdentity({ + title, + // Preserve this public title-parser adapter's historical answer; pane identity + // consumers pass raw titles to the canonical resolver and enforce its fence. + uncoveredFallback: { agent: parsed, titleOnly: false } + }).agent } /** @@ -283,6 +286,6 @@ function computeExplicitTerminalTitleAgentType(title: string): TuiAgent | null { return titleAgent } -/** Pure in `title` — memoized so repeated selector reads skip the regex ladder. */ +/** Pure in `title` — memoized so repeated selector reads skip the canonical/title parse. */ export const resolveExplicitTerminalTitleAgentType: (title: string) => TuiAgent | null = memoizeTitleClassification(computeExplicitTerminalTitleAgentType) From dd9eaa958545de250e504bc5a0ff0e05afc90da1 Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:20:27 -0400 Subject: [PATCH 217/398] fix(cloud): retry the committed-winner collision codes in relay schema startup (#18553) * fix(cloud): retry the committed-winner collision codes in relay schema startup `CREATE TABLE IF NOT EXISTS` only checks the name before the catalog inserts, so the loser of a concurrent CREATE fails in one of two ways depending on timing: on the catalog unique index (23505, which the startup retry already handled) or, when the winner has committed by the time the loser reaches TypeCreate / heap_create_with_catalog, on the name check those routines repeat (42710 duplicate type, 42P07 duplicate relation). The predicate treated the latter as fatal, so a director could fail startup on a table it was about to find present. This is what turned `postgres-schema-concurrency-postgres.test.ts` red on main and on every relay PR (CI's shared runner loses the race more often than a dev box): a throwaway diagnostic run in CI reported 42710 from TypeCreate and 42P07 from heap_create_with_catalog as the only rejection reasons. Treat 42710/42P07 as retryable for `CREATE TABLE IF NOT EXISTS` and 42P07 for `CREATE [UNIQUE] INDEX IF NOT EXISTS`; every other statement shape still fails fast. The concurrency test now runs ten rounds and reports the loser's SQLSTATE instead of a bare boolean. * chore(cloud): allowlist the RFC 6455 example Sec-WebSocket-Key for upgrade tests Cloud Verify's Secret scan runs gitleaks over --all refs, so the raw-socket upgrade test on fix/relay-upgrade-malformed-uri (#18547) trips every cloud PR's scan until its allowlist reaches main. Land the allowlist here first. --- cloud/.gitleaks.toml | 7 ++++ .../src/database-postgres-timeout.test.ts | 33 ++++++++++++++++++ ...stgres-schema-concurrency-postgres.test.ts | 34 +++++++++++-------- .../apps/relay/src/postgres-schema-startup.ts | 34 +++++++++++++++---- 4 files changed, 88 insertions(+), 20 deletions(-) diff --git a/cloud/.gitleaks.toml b/cloud/.gitleaks.toml index fc5c725c37d..0bb1f966fae 100644 --- a/cloud/.gitleaks.toml +++ b/cloud/.gitleaks.toml @@ -13,3 +13,10 @@ description = "Cloud SQL rollout lease holder keys in the action's unit tests" regexTarget = "secret" paths = ['''\.github/actions/cloud-sql-rollout-lease/[a-z-]+\.test\.mjs$'''] regexes = ['''^[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+/[0-9]+$'''] + +# RFC 6455 §1.3 example handshake nonce ("the sample nonce" in base64), sent by the raw-socket +# upgrade tests; the generic key rule reads any base64 header value as a secret. +[[allowlists]] +description = "RFC 6455 example Sec-WebSocket-Key in upgrade tests" +regexTarget = "secret" +regexes = ['''^dGhlIHNhbXBsZSBub25jZQ==$'''] diff --git a/cloud/apps/relay/src/database-postgres-timeout.test.ts b/cloud/apps/relay/src/database-postgres-timeout.test.ts index a04a9eea042..7aba1234f7f 100644 --- a/cloud/apps/relay/src/database-postgres-timeout.test.ts +++ b/cloud/apps/relay/src/database-postgres-timeout.test.ts @@ -118,6 +118,39 @@ describe('PostgreSQL schema startup', () => { expect(query).toHaveBeenCalledTimes(2) }) + it.each([ + ['42710', 'CREATE TABLE IF NOT EXISTS test'], + ['42P07', 'CREATE TABLE IF NOT EXISTS test'], + ['42P07', 'CREATE INDEX IF NOT EXISTS test_index ON test(id)'], + ['42P07', 'CREATE UNIQUE INDEX IF NOT EXISTS test_index ON test(id)'] + ])('retries the committed-winner %s collision for %s', async (code, statement) => { + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + const collision = Object.assign(new Error('already exists'), { code }) + const query = vi + .fn<(statement: string) => Promise>() + .mockRejectedValueOnce(collision) + .mockResolvedValue(undefined) + + await applyPostgresSchema([statement], query, { wait: async () => undefined }) + + expect(query).toHaveBeenCalledTimes(2) + }) + + it.each([ + ['42710', 'CREATE INDEX IF NOT EXISTS test_index ON test(id)'], + ['42710', 'CREATE TABLE test'], + ['42P07', 'CREATE TABLE test'], + ['42P07', 'CREATE INDEX test_index ON test(id)'] + ])('does not retry %s for %s', async (code, statement) => { + const error = Object.assign(new Error('already exists'), { code }) + const query = vi.fn<(statement: string) => Promise>().mockRejectedValue(error) + const pause = vi.fn(async () => undefined) + + await expect(applyPostgresSchema([statement], query, { wait: pause })).rejects.toBe(error) + + expect(pause).not.toHaveBeenCalled() + }) + it.each([ ['pg_type_typname_nsp_index', 'CREATE TABLE test'], ['pg_class_relname_nsp_index', 'CREATE INDEX test_index ON test(id)'] diff --git a/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts b/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts index 5208f137ae8..9ed3cb2f324 100644 --- a/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts +++ b/cloud/apps/relay/src/postgres-schema-concurrency-postgres.test.ts @@ -34,22 +34,28 @@ describePostgres('PostgreSQL schema concurrency', () => { }) it('opens five directors when one new table is absent', async () => { - const initial = await openRelayDatabase({ databaseUrl: scopedUrl, dataDir: '' }) - await initial.query(`DROP TABLE relay_cell_legacy_fence_adoptions`) - await initial.close() + // Which catalog step the race loser fails on depends on scheduling, so run several rounds and + // keep the loser's SQLSTATE in the failure instead of a bare boolean. + for (let round = 0; round < 10; round += 1) { + const initial = await openRelayDatabase({ databaseUrl: scopedUrl, dataDir: '' }) + await initial.query(`DROP TABLE relay_cell_legacy_fence_adoptions`) + await initial.close() - const results = await Promise.allSettled( - Array.from({ length: 5 }, async (): Promise => - await openRelayDatabase({ databaseUrl: scopedUrl, dataDir: '' }) + const results = await Promise.allSettled( + Array.from({ length: 5 }, async (): Promise => + await openRelayDatabase({ databaseUrl: scopedUrl, dataDir: '' }) + ) + ) + const databases = results.flatMap((result) => + result.status === 'fulfilled' ? [result.value] : [] ) - ) - const databases = results.flatMap((result) => - result.status === 'fulfilled' ? [result.value] : [] - ) - try { - expect(results.every((result) => result.status === 'fulfilled')).toBe(true) - } finally { await Promise.all(databases.map(async (database) => await database.close())) + const rejections = results.flatMap((result) => + result.status === 'rejected' + ? [{ round, code: (result.reason as { code?: unknown }).code, message: String(result.reason) }] + : [] + ) + expect(rejections).toEqual([]) } - }) + }, 60_000) }) diff --git a/cloud/apps/relay/src/postgres-schema-startup.ts b/cloud/apps/relay/src/postgres-schema-startup.ts index 5e6260ad2fb..ba9efc6a792 100644 --- a/cloud/apps/relay/src/postgres-schema-startup.ts +++ b/cloud/apps/relay/src/postgres-schema-startup.ts @@ -22,15 +22,37 @@ function wait(delayMs: number): Promise { return new Promise((resolve) => setTimeout(resolve, delayMs)) } +const CREATE_TABLE_IF_NOT_EXISTS = /^\s*CREATE\s+TABLE\s+IF\s+NOT\s+EXISTS\b/i +const CREATE_INDEX_IF_NOT_EXISTS = /^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+IF\s+NOT\s+EXISTS\b/i + +// `IF NOT EXISTS` only checks the name before the catalog inserts, so the loser of a concurrent +// CREATE can fail on the catalog unique index (23505) or, when the winner has already committed by +// the time the loser reaches TypeCreate/heap_create_with_catalog, on the name check those routines +// repeat (42710 duplicate type, 42P07 duplicate relation). Each is a no-op on the next attempt. +function concurrentCreateCollision( + value: { code?: unknown; constraint?: unknown }, + statement: string +): boolean { + if (CREATE_TABLE_IF_NOT_EXISTS.test(statement)) { + return ( + (value.code === '23505' && value.constraint === 'pg_type_typname_nsp_index') || + value.code === '42710' || + value.code === '42P07' + ) + } + if (CREATE_INDEX_IF_NOT_EXISTS.test(statement)) { + return ( + (value.code === '23505' && value.constraint === 'pg_class_relname_nsp_index') || + value.code === '42P07' + ) + } + return false +} + function retryableSchemaError(error: unknown, statement: string): boolean { const value = error as { code?: unknown; constraint?: unknown } return ( - RETRYABLE_SCHEMA_CODES.has(String(value.code)) || - (value.code === '23505' && - ((value.constraint === 'pg_type_typname_nsp_index' && - /^\s*CREATE\s+TABLE\s+IF\s+NOT\s+EXISTS\b/i.test(statement)) || - (value.constraint === 'pg_class_relname_nsp_index' && - /^\s*CREATE\s+(?:UNIQUE\s+)?INDEX\s+IF\s+NOT\s+EXISTS\b/i.test(statement)))) + RETRYABLE_SCHEMA_CODES.has(String(value.code)) || concurrentCreateCollision(value, statement) ) } From 8c1a28d39c283b358e7dd4217ddaf7fb8c9a2250 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 18:24:43 -0700 Subject: [PATCH 218/398] fix(i18n): repair French locale drift breaking static analysis (#18550) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The French UI locale landed with two catalog drifts that fail `static analysis` on every PR in the repo: - `fr.json` carried 15 keys absent from `en.json` (and from every other locale), so `verify:localization-catalog` rejected it. They are stale entries generated against an older `en.json` snapshot; none is referenced anywhere in the source. - `settings.appearance.language.french` had no call site supplying a literal default, which promotes it to a boot-bundle-required entry that `en-runtime-required.json` does not ship, so `verify:localization-runtime-catalog` rejected it. Registering the key in settings search alongside its siblings fixes the runtime-catalog failure at its source and closes the real gap the drift exposed: French was the only supported language not findable in settings search. `en-runtime-required.json` is deliberately untouched — the sync script regenerates it wholesale and would drop 925 entries the check itself documents as harmless. --- config/localization-coverage-allowlist.json | 7 +++++++ .../settings/appearance-search.test.ts | 9 +++++++-- .../components/settings/appearance-search.ts | 2 ++ src/renderer/src/i18n/locales/fr.json | 17 ----------------- 4 files changed, 16 insertions(+), 19 deletions(-) diff --git a/config/localization-coverage-allowlist.json b/config/localization-coverage-allowlist.json index 57694b2033f..a10139d218f 100644 --- a/config/localization-coverage-allowlist.json +++ b/config/localization-coverage-allowlist.json @@ -55,6 +55,13 @@ "dynamic": false, "count": 1 }, + { + "filePath": "src/renderer/src/components/settings/appearance-search.ts", + "kind": "object-property:keywords", + "text": "Langue", + "dynamic": false, + "count": 1 + }, { "filePath": "src/renderer/src/components/settings/terminal-advanced-platform-search.ts", "kind": "object-property:keywords", diff --git a/src/renderer/src/components/settings/appearance-search.test.ts b/src/renderer/src/components/settings/appearance-search.test.ts index 4b8c29781c4..6169a3cccb1 100644 --- a/src/renderer/src/components/settings/appearance-search.test.ts +++ b/src/renderer/src/components/settings/appearance-search.test.ts @@ -7,14 +7,14 @@ import { matchesSettingsSearch } from './settings-search' // Native word for "language" in each supported UI language. These must be // findable no matter which locale the interface is currently rendered in, so a // speaker can locate (and switch to) their language from any starting point. -const NATIVE_LANGUAGE_WORDS = ['语言', '語言', '언어', '言語', 'Idioma'] +const NATIVE_LANGUAGE_WORDS = ['语言', '語言', '언어', '言語', 'Idioma', 'Langue'] describe('getLanguageEntries', () => { afterEach(async () => { await i18n.changeLanguage('en') }) - it.each(['en', 'zh', 'ko', 'ja', 'es'])( + it.each(['en', 'zh', 'ko', 'ja', 'es', 'fr'])( 'indexes every native word for "language" under the %s UI locale', async (locale) => { await i18n.changeLanguage(locale) @@ -29,4 +29,9 @@ describe('getLanguageEntries', () => { await i18n.changeLanguage('en') expect(matchesSettingsSearch('Español', getLanguageEntries()[0])).toBe(true) }) + + it('matches the French native language name in English UI', async () => { + await i18n.changeLanguage('en') + expect(matchesSettingsSearch('Français', getLanguageEntries()[0])).toBe(true) + }) }) diff --git a/src/renderer/src/components/settings/appearance-search.ts b/src/renderer/src/components/settings/appearance-search.ts index 1886409e9e8..726ecaa1b01 100644 --- a/src/renderer/src/components/settings/appearance-search.ts +++ b/src/renderer/src/components/settings/appearance-search.ts @@ -50,6 +50,7 @@ export const getLanguageEntries = createLocalizedCatalog((): SettingsSearchEntry ...translateSearchKeyword('settings.appearance.language.korean', '한국어'), ...translateSearchKeyword('settings.appearance.language.japanese', '日本語'), ...translateSearchKeyword('settings.appearance.language.spanish', 'Español'), + ...translateSearchKeyword('settings.appearance.language.french', 'Français'), // Why: the native word for "language" only reaches search via the localized // title in its own UI locale — index each here so speakers can find (and // switch to) their language whatever the current interface locale is. @@ -58,6 +59,7 @@ export const getLanguageEntries = createLocalizedCatalog((): SettingsSearchEntry '언어', // Korean '言語', // Japanese 'Idioma', // Spanish + 'Langue', // French ...translateSearchKeyword( 'auto.components.settings.appearance.search.language.locale', 'locale' diff --git a/src/renderer/src/i18n/locales/fr.json b/src/renderer/src/i18n/locales/fr.json index ed9307b30db..8eff250a820 100644 --- a/src/renderer/src/i18n/locales/fr.json +++ b/src/renderer/src/i18n/locales/fr.json @@ -2934,7 +2934,6 @@ "a7e2fd2699": "signaler un problème", "5c8ce20be6": "Si le problème persiste, veuillez", "cc6d997c65": "Redémarrez le daemon de terminal depuis ici pour effacer un état de daemon obsolète.", - "7ee11bc0db": "Orca n'a pas pu confirmer si la session précédente de ce terminal tourne encore ; il a donc laissé la session intacte. Rouvrez ce volet pour réessayer.", "e16012e31e": "Le daemon de terminal propriétaire de cette session s'est arrêté ; la session et son historique de défilement n'ont pas pu être récupérés. Ouvrez un nouveau terminal pour continuer.", "sessionUnavailable": "Orca n'a pas pu se rattacher à la session de terminal de ce volet sur l'hôte. Ouvrez un nouveau terminal pour continuer." }, @@ -3228,7 +3227,6 @@ "25dc1cd653": "Ouvrir le fichier", "7cdf8ee0c8": "Ouvrir une URL", "b27864279e": "Lancer un agent", - "0e5b7a3f16": "Rechercher parmi les onglets ouverts, fichiers, URL et agents…", "8f0a1c4d92": "Basculer vers l'onglet", "2c38630a01": "L'espace de travail n'existe plus", "4f0d9a71c2": "L'onglet n'existe plus", @@ -3274,7 +3272,6 @@ "classifier": { "42e6262ae9": "Aucune action disponible.", "097a982ee0": "Chargement des fichiers...", - "c41f8d20b7": "Rechercher parmi les onglets ouverts, fichiers, URL et agents…", "90eb94dc48": "Saisissez une URL http:// ou https://.", "5553b283ce": "Saisissez une URL ou un chemin de fichier.", "queryTooLarge": "Le texte recherché est trop long.", @@ -6999,12 +6996,6 @@ "compareBaseRepositoryDefault": "Valeur par défaut du dépôt", "compareBaseBranchUpstream": "Amont de la branche" }, - "HiddenExperimentalGroup": { - "d0f914a528": "Bascule fictive", - "1014ddbfaf": "Sans effet aujourd'hui. Réservée comme premier emplacement pour les options expérimentales masquées.", - "232cf83de8": "Bascules non répertoriées pour les tests internes. Rien ici n'est pris en charge.", - "3e9e827ca5": "Expérimental masqué" - }, "InputPane": { "db15068196": "Activé par défaut sur Linux et macOS. Linux utilise le presse-papiers de sélection du système ; les autres plateformes utilisent un tampon privé.", "ad31c3c5fb": "Collage de la sélection au clic milieu" @@ -11832,10 +11823,6 @@ "b3c8f1a902": "Filtrer les fichiers par nom", "d4f8c2a901": "Effacer et fermer le filtre", "e8a1c4b203": "vs", - "f9b2441bb6": "1 commit d'avance sur {{value0}}", - "b715ef615b": "{{value0}} commits d'avance sur {{value1}}", - "c1a8f3e204": "1 commit de retard sur {{value0}}", - "d2b9g4f315": "{{value0}} commits de retard sur {{value1}}", "4b4a7de138": "Ouvrir la page de revue dans le navigateur", "createPrIntentCommitBlockedSummary": "Commit bloqué : {{value0}} Corrigez le problème, puis réessayez Créer une PR.", "pushRecovery": { @@ -11858,7 +11845,6 @@ "a4e93c21d7": "Branche actuelle : {{value0}}", "c7d4e2f801": "Modifier la ref de base : {{value0}}", "f3a1b8c204": "upstream", - "createPrIntentEmptyGeneratedBody": "Les détails de revue générés n'incluent pas de description. Réessayez Créer une PR.", "createPrIntentGenerateDetailsFailed": "Impossible de générer les détails de revue. Réessayez Créer une PR." }, "SourceControlAgentActionDialog": { @@ -14038,8 +14024,6 @@ "7e7ca60816": "fichiers modifiés", "b6c3b84476": "Afficher l'arborescence des fichiers", "39f8007549": "Examiner les conflits", - "39e73e7181": "ont été exclus de cette vue de diff.", - "689b99f8ad": "conflit non résolu", "820ec01f24": "Les fichiers en conflit sont examinés séparément", "fd8892b120": "Aucune modification à afficher", "eb5f40e49c": "Cette vue de diff exclut les conflits non résolus, car le pipeline de diff bidirectionnel habituel n'est pas conçu pour gérer les conflits.", @@ -15793,7 +15777,6 @@ }, "LinuxPackageInstallRecoveryCard": { "e3de29c86a": "Afficher le paquet", - "3da99454c6": "Réessayer l'installation automatique", "55c86654b7": "Copier la commande d'installation", "53e1559f99": "Échec de l'installation automatique", "a7ac6ec78b": "Orca a téléchargé la mise à jour mais n'a pas pu installer le paquet système automatiquement.", From 11aace8dec5437ff5d2ad0e5936a049eaa1db9eb Mon Sep 17 00:00:00 2001 From: Jinwoo Hong <73622457+Jinwoo-H@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:31:53 -0400 Subject: [PATCH 219/398] fix(relay): reject malformed percent-escapes on upgrade instead of throwing (#18547) decodeURIComponent on the /v1/connect/ and /v1/host/data/ path segments threw URIError out of the http 'upgrade' listener, which is uncaught and kills the relay process. Any client that sends GET /v1/connect/% could take down a cell (and every connection on it) or a director instance. Pre-existing since the splice landed (orca-cloud #20); not introduced by the import. A malformed escape now takes the existing 4xx reject branch. The blackbox test sends three malformed connect targets and one host-data target to the real server and asserts no uncaughtException fires and a well-formed upgrade still gets 101 afterwards; reverting either site fails it. --- cloud/apps/relay/src/relay-server.ts | 16 ++- ...lay-upgrade-malformed-uri.blackbox.test.ts | 122 ++++++++++++++++++ 2 files changed, 135 insertions(+), 3 deletions(-) create mode 100644 cloud/apps/relay/src/relay-upgrade-malformed-uri.blackbox.test.ts diff --git a/cloud/apps/relay/src/relay-server.ts b/cloud/apps/relay/src/relay-server.ts index a14240cfa6a..32a83962d81 100644 --- a/cloud/apps/relay/src/relay-server.ts +++ b/cloud/apps/relay/src/relay-server.ts @@ -31,6 +31,16 @@ import { createRelayTokenVerifier, readBearer } from './relay-token-verifier.js' import { closeRelayWebSocket } from './relay-websocket-close.js' import { ProcessQueuedByteBudget } from './splice-forwarder.js' +// A malformed percent-escape in the request target must be a client error, never a URIError +// thrown out of the `upgrade` listener (which is uncaught and kills the process). +function decodePathSegment(value: string): string | null { + try { + return decodeURIComponent(value) + } catch { + return null + } +} + function rejectUpgrade(socket: NodeJS.WritableStream, status: number, message: string): void { socket.write(`HTTP/1.1 ${status} ${message}\r\nConnection: close\r\nContent-Length: 0\r\n\r\n`) if ('destroy' in socket && typeof socket.destroy === 'function') socket.destroy() @@ -278,8 +288,8 @@ export function createRelayServer( return } if (url.pathname.startsWith('/v1/connect/')) { - const hostId = decodeURIComponent(url.pathname.slice('/v1/connect/'.length)) - if (!/^[A-Za-z0-9_-]{16}$/.test(hostId)) { + const hostId = decodePathSegment(url.pathname.slice('/v1/connect/'.length)) + if (hostId === null || !/^[A-Za-z0-9_-]{16}$/.test(hostId)) { rejectUpgrade(socket, 429, 'Too Many Requests') return } @@ -373,7 +383,7 @@ export function createRelayServer( rejectUpgrade(socket, 404, 'Not Found') return } - const connId = decodeURIComponent(url.pathname.slice('/v1/host/data/'.length)) + const connId = decodePathSegment(url.pathname.slice('/v1/host/data/'.length)) if (!connId || connId.length > 128) { rejectUpgrade(socket, 429, 'Too Many Requests') return diff --git a/cloud/apps/relay/src/relay-upgrade-malformed-uri.blackbox.test.ts b/cloud/apps/relay/src/relay-upgrade-malformed-uri.blackbox.test.ts new file mode 100644 index 00000000000..24635a68e6f --- /dev/null +++ b/cloud/apps/relay/src/relay-upgrade-malformed-uri.blackbox.test.ts @@ -0,0 +1,122 @@ +import { connect, createServer as createNetServer } from 'node:net' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { RelayConfig } from './config.js' +import type { RelayDatabase } from './database.js' +import { createRelayServer } from './relay-server.js' + +async function unusedPort(): Promise { + const server = createNetServer() + await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve)) + const address = server.address() + if (!address || typeof address === 'string') throw new Error('missing test port') + await new Promise((resolve) => server.close(() => resolve())) + return address.port +} + +function rawUpgrade(port: number, target: string): Promise<{ status: string; closed: boolean }> { + return new Promise((resolve, reject) => { + const socket = connect(port, '127.0.0.1') + let data = '' + socket.once('connect', () => { + socket.write( + `GET ${target} HTTP/1.1\r\nHost: 127.0.0.1\r\nConnection: Upgrade\r\n` + + 'Upgrade: websocket\r\nSec-WebSocket-Version: 13\r\n' + + // RFC 6455 §1.3 example nonce; allowlisted in cloud/.gitleaks.toml. + 'Sec-WebSocket-Key: dGhlIHNhbXBsZSBub25jZQ==\r\n\r\n' + ) + }) + socket.on('data', (chunk) => { + data += chunk.toString() + }) + socket.once('close', () => resolve({ status: data.split('\r\n')[0] ?? '', closed: true })) + socket.once('error', reject) + setTimeout(() => { + socket.destroy() + resolve({ status: data.split('\r\n')[0] ?? '', closed: false }) + }, 1_500).unref() + }) +} + +describe('relay upgrade with a malformed request target', () => { + const cleanup: Array<() => Promise | void> = [] + + afterEach(async () => { + for (const close of cleanup.splice(0).reverse()) await close() + vi.restoreAllMocks() + }) + + it('rejects an undecodable /v1/connect path without an uncaught exception', async () => { + const port = await unusedPort() + const relayUrl = `http://127.0.0.1:${port}` + const database: RelayDatabase = { + query: vi.fn(async () => []), + queryLocked: vi.fn(async () => []), + transaction: vi.fn(async (operation) => await operation(database)), + close: vi.fn(async () => undefined) + } + const config = { + port, + publicUrl: relayUrl, + cellUrl: relayUrl, + authIssuer: 'https://auth.example.com', + authAudience: 'orca-relay', + jwksUrl: 'https://auth.example.com/jwks', + assignmentSigningKey: new Uint8Array(32), + role: 'cell', + cellId: 'production-gce-c3', + cells: [{ id: 'production-gce-c3', url: relayUrl, capacityRequests: 4_000 }], + adminAudience: `${relayUrl}/admin`, + deployServiceAccount: 'deploy@example.com', + runtimeServiceAccount: 'runtime@example.com', + connectionHardCap: 600, + connectionUnobservedBound: 60, + adminJwksUrl: 'https://auth.example.com/admin-jwks', + databasePoolMax: 10, + publicAssignmentsEnabled: true, + publicAssignmentConcurrency: 2, + publicAssignmentQueueMax: 128, + publicAssignmentWaitMs: 4_000, + publicResolveConcurrency: 1, + publicResolveWaitMs: 5_000, + publicAssignmentRetryAfterSeconds: 5, + dataDir: './test-data' + } satisfies RelayConfig + const relay = createRelayServer(config, database, { + connectionLedgerLimits: { hardCap: 5, controlReserve: 1 } + }) + relay.server.listen(port, '127.0.0.1') + await new Promise((resolve) => relay.server.once('listening', resolve)) + cleanup.push(() => new Promise((resolve) => relay.server.close(() => resolve()))) + vi.spyOn(console, 'log').mockImplementation(() => undefined) + vi.spyOn(console, 'warn').mockImplementation(() => undefined) + + // Vitest installs its own uncaughtException listener; capture ours first so the test reports + // the exception as a verdict instead of dying with it. + const uncaught: unknown[] = [] + const onUncaught = (error: unknown): void => { + uncaught.push(error) + } + process.prependListener('uncaughtException', onUncaught) + cleanup.push(() => { + process.off('uncaughtException', onUncaught) + }) + + const results = [] + for (const target of [ + '/v1/connect/%', + '/v1/connect/%E0%A4%A', + '/v1/connect/%C0%AF', + '/v1/host/data/%' + ]) { + results.push(await rawUpgrade(port, target)) + } + // A malformed percent-escape must be a client error, never a process-level throw. + expect(uncaught).toEqual([]) + for (const result of results) { + expect(result.status).toMatch(/^HTTP\/1\.1 4\d\d/) + } + // The server must still serve a well-formed upgrade afterwards. + const after = await rawUpgrade(port, '/v1/connect/abcdefghijklmnop') + expect(after.status).toMatch(/^HTTP\/1\.1 101/) + }) +}) From 7106101ed2c913858cc088930158a2d25b441c71 Mon Sep 17 00:00:00 2001 From: Shahar Mor Date: Thu, 3 Sep 2026 18:32:53 -0700 Subject: [PATCH 220/398] fix(mobile): restore terminal input when reopening worktrees (#16239) * fix(mobile): restore terminal input when reopening worktrees * test(mobile): update session parity facts * refactor(mobile): split host client hooks * chore: restore localization formatter scope * fix(mobile): retain RpcClient type import --------- Co-authored-by: Merge Sim --- .../mobile-session-route-parity.test.ts | 14 +-- .../mobile-session-startup-source.test.ts | 21 ++++ .../session/use-mobile-session-foundation.ts | 3 +- .../session/use-mobile-session-lifecycle.ts | 4 +- .../use-mobile-session-terminal-runtime.ts | 9 +- ...se-mobile-session-terminal-subscription.ts | 9 +- mobile/src/transport/client-context.test.ts | 10 +- mobile/src/transport/client-context.tsx | 106 ++---------------- .../transport/host-client-context-state.ts | 1 + mobile/src/transport/host-client-hooks.ts | 92 +++++++++++++++ mobile/src/transport/host-entry-opener.ts | 2 + .../transport/rpc-client-context-contract.ts | 1 + 12 files changed, 157 insertions(+), 115 deletions(-) create mode 100644 mobile/src/transport/host-client-hooks.ts diff --git a/mobile/src/session/mobile-session-route-parity.test.ts b/mobile/src/session/mobile-session-route-parity.test.ts index b6abab8001e..c6e5f434326 100644 --- a/mobile/src/session/mobile-session-route-parity.test.ts +++ b/mobile/src/session/mobile-session-route-parity.test.ts @@ -63,11 +63,11 @@ const HOST_COMPONENT_NAMES = new Set([ ]) const HEAD_MAIN_HOOK_SHA256 = '10071240ef9edafc2b9c8bed73be83dceaf7828e3b29f17dab55da020a7697a6' -const HEAD_HOOK_BINDING_SHA256 = 'ecd4c1dad066cf13698447b8ffb61f82e6cc3ebe7d484f71189626efed430272' +const HEAD_HOOK_BINDING_SHA256 = '1dadb8c3dc0573ea20659ce7251629669e618dd0effaeac3a4536b29c2e865a1' const HEAD_CALLBACK_IDENTITY_SHA256 = - 'df073bc13d94a93e7fbd8b1fca2b57eaf43cbf7ca799a649e0ebb783e5b8eecc' -const HEAD_CALLBACK_BODY_SHA256 = '690e3069e08ecf805af726b658e900c973565259160f25e3a643175e2ab1bc75' -const HEAD_EFFECT_SHA256 = '346d384ea0bf2f8f926c5092c5bf57bc2a03494f49f9639e9d6b8a2c51c9f882' + '2a9e4825df007f6ef53b81aa5004991d6318eee7507b44d625c07e630be432eb' +const HEAD_CALLBACK_BODY_SHA256 = '22103ba85a86e3a3fcb80a7509c7a455d79863010cde3af02db6565b55e3ebe9' +const HEAD_EFFECT_SHA256 = 'd9ebfaabc1e79773cdada7ab370b20459ed972f1f8edce1652199f4d0391cd13' const HEAD_CONTENT_HOOK_SHA256 = '9c3b612fef3f370d66873aefdbe1d701f20cb64ded31fef5cc45fde6f8189581' const HEAD_NESTED_FUNCTION_SHA256 = '6a13919ede2a8033436fb03e0ff7c426fbed97f470875a7b21b00aaada17fb73' @@ -79,13 +79,13 @@ const HEAD_TIMER_CREATION_SHA256 = '1a31b625e2174c3db77272249843196d2b6b06ab1e654a96d8f7858e3082e66b' const HEAD_TIMER_CLEANUP_SHA256 = 'c73f1d1c2cc89642f3d727d6f3b6b81860a9d6f34234541a2065ec3d1a8cd116' const HEAD_RUNTIME_STRING_SHA256 = - '1cb95fe0095c1c57e1b0629472e1cce5328eb7f5bfeca38095f41f4612a37887' + 'ba52a3ede721bd29acbe8593161e90b216b7f361ff896e085927d4d73fa83b2f' const HEAD_HOST_JSX_SHA256 = '390405926b1695fa3a33686f0bc192b432f5468d8576499d7cafbb4922defbb5' const HEAD_LEAF_JSX_SHA256 = '21dba981875e173f692590bf910d60964660c5f4cbb79f3a377c7e54f6a1f016' const HEAD_STYLE_REFERENCE_SHA256 = '295a3501c2c6d7bea7c8bbf38b3f3534f01344cd7e1b91bb8e07c040821d596a' const HEAD_IDENTITY_FIELD_SHA256 = - '6b37a0351795a387a358df76a5ab919a7098ddb76bf25a936c8902c062c8951c' + '91146853930a34dd1f3d80e5c97fbacd7cf19fb93dd26fe8fc6f29169622f9d6' const HEAD_NAVIGATION_SHA256 = '9d96f5dad7de555d6553eac39c0fab00efad507470fd562cb9beaa32db16f512' const HEAD_CAPABILITY_SHA256 = 'ca219f7909a091717110b823d5b94a20770ad3ae51894e0fa765e8628309392d' @@ -517,7 +517,7 @@ describe('mobile session route extraction parity', () => { it('preserves runtime strings, styles, and the expanded JSX tree', () => { const strings = readRuntimeStrings() - expect(strings).toHaveLength(545) + expect(strings).toHaveLength(546) expect(hash(strings)).toBe(HEAD_RUNTIME_STRING_SHA256) const jsx = readJsxFacts(readDefinitions()) expect(jsx.host).toHaveLength(124) diff --git a/mobile/src/session/mobile-session-startup-source.test.ts b/mobile/src/session/mobile-session-startup-source.test.ts index acab1d5e7c6..83b2021b95e 100644 --- a/mobile/src/session/mobile-session-startup-source.test.ts +++ b/mobile/src/session/mobile-session-startup-source.test.ts @@ -29,6 +29,14 @@ const tabReconciliationOwnerSource = readMobileSessionRouteSource( const autoCreateHookSource = readMobileSessionRouteSource( './use-initial-session-terminal-autocreate.ts' ) +const foundationSource = readMobileSessionRouteSource('./use-mobile-session-foundation.ts') +const terminalRuntimeSource = readMobileSessionRouteSource( + './use-mobile-session-terminal-runtime.ts' +) +const terminalSubscriptionSourceForIdentity = readMobileSessionRouteSource( + './use-mobile-session-terminal-subscription.ts' +) +const lifecycleSource = readMobileSessionRouteSource('./use-mobile-session-lifecycle.ts') function sliceBetween(startPattern: string, endPattern: string, targetSource = source): string { const start = targetSource.indexOf(startPattern) @@ -106,6 +114,19 @@ describe('mobile session startup', () => { expect(reconciliationHookSource).toContain('appStateSubscription.remove()') }) + it('binds terminal identity to the shared client before subscription effects run', () => { + expect(foundationSource).toContain('const { client, clientId, state: connState }') + expect(foundationSource).toContain(' clientId,') + expect(terminalRuntimeSource).toContain('useRef(clientId)') + expect(terminalRuntimeSource).toContain('deviceTokenRef.current = clientId') + expect(terminalRuntimeSource).toContain('inputGate.canSend && clientId !== null') + expect(terminalSubscriptionSourceForIdentity).toContain('if (clientId === null)') + expect(terminalSubscriptionSourceForIdentity).toContain( + "client: { id: clientId, type: 'mobile' as const }" + ) + expect(lifecycleSource).not.toContain('deviceTokenRef.current = host.deviceToken') + }) + it('confirms terminal stream teardown with a committed inventory-recovery bridge', () => { expect(terminalSubscriptionSource).toContain( "if (data.type === 'end' || data.type === 'error')" diff --git a/mobile/src/session/use-mobile-session-foundation.ts b/mobile/src/session/use-mobile-session-foundation.ts index 9a7889e10cb..fa2f9607bbc 100644 --- a/mobile/src/session/use-mobile-session-foundation.ts +++ b/mobile/src/session/use-mobile-session-foundation.ts @@ -35,7 +35,7 @@ export function useMobileSessionFoundation() { const router = useRouter() const insets = useSafeAreaInsets() // Why: shared client per host owned by RpcClientProvider (docs/mobile-shared-client-per-host.md). - const { client, state: connState } = useHostClient(hostId) + const { client, clientId, state: connState } = useHostClient(hostId) const reconnectAttempts = useReconnectAttempt(hostId) const lastConnectedAt = useLastConnectedAt(hostId) const forceReconnectHost = useForceReconnect() @@ -96,6 +96,7 @@ export function useMobileSessionFoundation() { router, insets, client, + clientId, connState, reconnectAttempts, lastConnectedAt, diff --git a/mobile/src/session/use-mobile-session-lifecycle.ts b/mobile/src/session/use-mobile-session-lifecycle.ts index 912ebdb4921..7c58e84a3e5 100644 --- a/mobile/src/session/use-mobile-session-lifecycle.ts +++ b/mobile/src/session/use-mobile-session-lifecycle.ts @@ -16,7 +16,6 @@ export function useMobileSessionLifecycle(scope: MobileSessionTabReconciliationM connState, setCustomKeys, setVisibleBuiltInIds, - deviceTokenRef, setHostEndpoint, connStateRef, terminalRefs, @@ -26,7 +25,7 @@ export function useMobileSessionLifecycle(scope: MobileSessionTabReconciliationM unsubscribeTerminal, subscribeToTerminal } = scope - // Why: read deviceToken from host record so code can pass client.id on subscribe/send for driver-state-machine identity. + // Why: the shared client owns authenticated identity; this host read only supplies connection-hint metadata. useEffect(() => { if (!hostId) { return @@ -38,7 +37,6 @@ export function useMobileSessionLifecycle(scope: MobileSessionTabReconciliationM } const host = hosts.find((h) => h.id === hostId) if (host) { - deviceTokenRef.current = host.deviceToken setHostEndpoint(host.endpoint) } }) diff --git a/mobile/src/session/use-mobile-session-terminal-runtime.ts b/mobile/src/session/use-mobile-session-terminal-runtime.ts index 8ef472fb845..5086efe1ba1 100644 --- a/mobile/src/session/use-mobile-session-terminal-runtime.ts +++ b/mobile/src/session/use-mobile-session-terminal-runtime.ts @@ -27,6 +27,7 @@ export function useMobileSessionTerminalRuntime(scope: MobileSessionScreenStateM worktreeId, connState, client, + clientId, sessionTabs, setLiveInputCapture, liveInputTerminalHandles, @@ -42,7 +43,9 @@ export function useMobileSessionTerminalRuntime(scope: MobileSessionScreenStateM const terminalGestureInputInFlightRef = useRef>(new Set()) const terminalCwdRef = useRef>(new Map()) const initialModesSeenRef = useRef>(new Set()) - const deviceTokenRef = useRef(null) + const deviceTokenRef = useRef(clientId) + // Keep the authenticated identity synchronous with the client exposed to downstream hooks. + deviceTokenRef.current = clientId // Why: state (not a ref) so the connection verdict re-renders when the endpoint loads and the Tailscale hint can appear. const [hostEndpoint, setHostEndpoint] = useState(null) const clientRef = useRef(null) @@ -123,11 +126,13 @@ export function useMobileSessionTerminalRuntime(scope: MobileSessionScreenStateM sendLiveTerminalInputRef, setLiveInputCapture }) - const { canCompose, canSend } = resolveMobileTerminalInputGate({ + const inputGate = resolveMobileTerminalInputGate({ connState, activeHandle, activeSessionTabType: activeSessionTab?.type }) + const canCompose = inputGate.canCompose + const canSend = inputGate.canSend && clientId !== null const liveInputEnabled = activeHandle ? liveInputTerminalHandles.has(activeHandle) : false const { focusLiveInput, handleTerminalTap, resetLiveInputFocus } = useTerminalLiveInputFocus({ activeHandleRef, diff --git a/mobile/src/session/use-mobile-session-terminal-subscription.ts b/mobile/src/session/use-mobile-session-terminal-subscription.ts index 8ad0bb7383a..333d472a30e 100644 --- a/mobile/src/session/use-mobile-session-terminal-subscription.ts +++ b/mobile/src/session/use-mobile-session-terminal-subscription.ts @@ -15,9 +15,9 @@ export function useMobileSessionTerminalSubscription( ) { const { client, + clientId, setTerminalModes, terminalCwdRef, - deviceTokenRef, viewportRef, viewportMeasuredRef, terminalUnsubsRef, @@ -49,6 +49,10 @@ export function useMobileSessionTerminalSubscription( logSkippedGate('no-client') return } + if (clientId === null) { + logSkippedGate('no-client-identity') + return + } if (terminalUnsubsRef.current.has(handle)) { logSkippedGate('already-subscribed') return @@ -89,7 +93,7 @@ export function useMobileSessionTerminalSubscription( client, { terminal: handle, - client: { id: deviceTokenRef.current!, type: 'mobile' as const }, + client: { id: clientId, type: 'mobile' as const }, viewport: nativeChatTerminalStream.mobileNativeChatSubscribeViewport( covered, viewportRef.current @@ -263,6 +267,7 @@ export function useMobileSessionTerminalSubscription( }, [ client, + clientId, getTerminalRef, markNativeChatInputLeaseReady, scheduleDelayedAction, diff --git a/mobile/src/transport/client-context.test.ts b/mobile/src/transport/client-context.test.ts index a311f362f85..4a7d5d7b0e8 100644 --- a/mobile/src/transport/client-context.test.ts +++ b/mobile/src/transport/client-context.test.ts @@ -146,8 +146,8 @@ beforeEach(() => { }) describe('useHostClient', () => { - it('rebinds when Expo reuses a screen between two connected cached hosts', async () => { - const host2 = { ...HOST, id: 'host-2', name: 'Host 2' } + it('rebinds the client and its authenticated identity together across cached hosts', async () => { + const host2 = { ...HOST, id: 'host-2', name: 'Host 2', deviceToken: 'token-2' } const client1 = makeFakeClient('connected') const client2 = makeFakeClient('connected') connectMock.mockReturnValueOnce(client1).mockReturnValueOnce(client2) @@ -155,11 +155,13 @@ describe('useHostClient', () => { let selectedHostId = HOST.id let selectedClient: RpcClient | null = null + let selectedClientId: string | null = null let selectedState: ConnectionState = 'disconnected' let renderer: ReactTestRenderer | null = null function Probe(): null { const selected = useHostClient(selectedHostId) selectedClient = selected.client + selectedClientId = selected.clientId selectedState = selected.state useHostClient(host2.id) return null @@ -171,16 +173,18 @@ describe('useHostClient', () => { await Promise.resolve() }) expect(selectedClient).toBe(client1) + expect(selectedClientId).toBe(HOST.deviceToken) expect(selectedState).toBe('connected') selectedHostId = host2.id - client2.emitState('disconnected') await act(async () => { + client2.emitState('disconnected') renderer?.update(createElement(RpcClientProvider, null, createElement(Probe))) await Promise.resolve() }) expect(selectedClient).toBe(client2) + expect(selectedClientId).toBe(host2.deviceToken) expect(selectedState).toBe('disconnected') expect(connectMock).toHaveBeenCalledTimes(2) } finally { diff --git a/mobile/src/transport/client-context.tsx b/mobile/src/transport/client-context.tsx index f83519940c1..76d26c1bab5 100644 --- a/mobile/src/transport/client-context.tsx +++ b/mobile/src/transport/client-context.tsx @@ -7,7 +7,6 @@ import { useEffect, useMemo, useRef, - useState, type ReactNode } from 'react' import type { RpcClient } from './rpc-client' @@ -35,6 +34,15 @@ import { import type { ConnectionState, HostProfile } from './types' import type { RpcClientContextValue } from './rpc-client-context-contract' +export { + useDisconnectHostClient, + useForceReconnect, + useForgetHostClient, + useHostClient, + usePrimeHosts, + useRefreshHostClient +} from './host-client-hooks' + type StoreEntry = HostClientStoreEntry const Ctx = createContext(null) @@ -364,99 +372,3 @@ export function useRpcClientContext(): RpcClientContextValue { } return ctx } - -// Primary hook for screens: acquires the shared client on mount, releases on unmount, re-renders on state change. -export function useHostClient(hostId: string | undefined): { - client: RpcClient | null - state: ConnectionState -} { - const ctx = useRpcClientContext() - const [, force] = useState(0) - // Why: an absent entry at mount is almost always the open racing the render, not a - // dead host — seed amber; a failed open notifies 'disconnected' moments later. - const [state, setState] = useState(() => - hostId ? (ctx.getKnownState(hostId) ?? 'connecting') : 'disconnected' - ) - const clientRef = useRef(null) - const clientHostIdRef = useRef(hostId) - const acquisitionRef = useRef({}) - - useEffect(() => { - if (!hostId) { - clientRef.current = null - clientHostIdRef.current = undefined - setState('disconnected') - return - } - clientHostIdRef.current = hostId - let cancelled = false - // Subscribe before acquire so any state change during open is captured. - const unsub = ctx.subscribeHostState(hostId, (next) => { - if (cancelled) { - return - } - setState(next) - // Why: async open and forceReconnect swap the client object; re-read each state change so screens never drive a stale one. - const found = ctx.getAllClients().find((entry) => entry.hostId === hostId) - if (found && found.client !== clientRef.current) { - clientRef.current = found.client - force((n) => n + 1) - } else if (!found && clientRef.current) { - // Why: disconnect/forget deletes the entry; never retain a dead client (STA-1511). - clientRef.current = null - force((n) => n + 1) - } - }) - const initial = ctx.acquire(hostId, acquisitionRef.current) - clientRef.current = initial - setState(ctx.getKnownState(hostId) ?? 'connecting') - if (initial) { - // Why: two cached hosts can both be connected, so equal state values cannot reveal the replacement client. - force((n) => n + 1) - } - return () => { - cancelled = true - unsub() - ctx.release(hostId, acquisitionRef.current) - clientRef.current = null - clientHostIdRef.current = undefined - } - }, [ctx, hostId]) - - // Why: Expo can reuse the screen before effects bind the next host; never expose the prior host's client or state in that render. - const bound = clientHostIdRef.current === hostId - const boundState = bound - ? state - : hostId - ? (ctx.getKnownState(hostId) ?? 'connecting') - : 'disconnected' - return { client: bound ? clientRef.current : null, state: boundState } -} - -// Why: host-store's removeHost() must close the live client but has no React-side handle; this hook bridges to it. -export function useRefreshHostClient(): (hostId: string) => void { - const ctx = useRpcClientContext() - return ctx.refreshHostClient -} - -export function useForgetHostClient(): (hostId: string) => void { - const ctx = useRpcClientContext() - return ctx.forgetHostClient -} - -export function useDisconnectHostClient(): (hostId: string) => void { - const ctx = useRpcClientContext() - return ctx.disconnectHostClient -} - -// Why: future-proof "Connection issues — try again" affordance. -export function useForceReconnect(): (hostId: string) => Promise { - const ctx = useRpcClientContext() - return ctx.forceReconnect -} - -// Why: primes already-loaded HostProfiles so the provider can skip a second loadHosts()/Keychain pass on cold start. -export function usePrimeHosts(): (hosts: HostProfile[]) => void { - const ctx = useRpcClientContext() - return ctx.primeHosts -} diff --git a/mobile/src/transport/host-client-context-state.ts b/mobile/src/transport/host-client-context-state.ts index 972a810af21..859e5d847f9 100644 --- a/mobile/src/transport/host-client-context-state.ts +++ b/mobile/src/transport/host-client-context-state.ts @@ -79,6 +79,7 @@ export function createHostClientSelectors( return { getKnownState, getState: (hostId: string): ConnectionState => getKnownState(hostId) ?? 'disconnected', + getClientId: (hostId: string): string | null => entries.get(hostId)?.clientId ?? null, getReconnectAttempt: (hostId: string): number => entries.get(hostId)?.client.getReconnectAttempt() ?? 0, getLastConnectedAt: (hostId: string): number | null => diff --git a/mobile/src/transport/host-client-hooks.ts b/mobile/src/transport/host-client-hooks.ts new file mode 100644 index 00000000000..c7f885034c2 --- /dev/null +++ b/mobile/src/transport/host-client-hooks.ts @@ -0,0 +1,92 @@ +import { useEffect, useRef, useState } from 'react' +import type { RpcClient } from './rpc-client' +import type { ConnectionState, HostProfile } from './types' +import type { HostClientAcquisition } from './host-client-acquisition-registry' +import { useRpcClientContext } from './client-context' + +// Primary hook for screens: acquires the shared client on mount, releases on unmount, re-renders on state change. +export function useHostClient(hostId: string | undefined): { + client: RpcClient | null + clientId: string | null + state: ConnectionState +} { + const ctx = useRpcClientContext() + const [, force] = useState(0) + const [state, setState] = useState(() => + hostId ? (ctx.getKnownState(hostId) ?? 'connecting') : 'disconnected' + ) + const clientRef = useRef(null) + const clientHostIdRef = useRef(hostId) + const acquisitionRef = useRef({}) + + useEffect(() => { + if (!hostId) { + clientRef.current = null + clientHostIdRef.current = undefined + setState('disconnected') + return + } + clientHostIdRef.current = hostId + let cancelled = false + const unsub = ctx.subscribeHostState(hostId, (next) => { + if (cancelled) { + return + } + setState(next) + const found = ctx.getAllClients().find((entry) => entry.hostId === hostId) + if (found && found.client !== clientRef.current) { + clientRef.current = found.client + force((n) => n + 1) + } else if (!found && clientRef.current) { + clientRef.current = null + force((n) => n + 1) + } + }) + const initial = ctx.acquire(hostId, acquisitionRef.current) + clientRef.current = initial + setState(ctx.getKnownState(hostId) ?? 'connecting') + if (initial) { + force((n) => n + 1) + } + return () => { + cancelled = true + unsub() + ctx.release(hostId, acquisitionRef.current) + clientRef.current = null + clientHostIdRef.current = undefined + } + }, [ctx, hostId]) + + const bound = clientHostIdRef.current === hostId + const boundClient = bound ? clientRef.current : null + const boundState = bound + ? state + : hostId + ? (ctx.getKnownState(hostId) ?? 'connecting') + : 'disconnected' + return { + client: boundClient, + clientId: boundClient && hostId ? ctx.getClientId(hostId) : null, + state: boundState + } +} + +export function useRefreshHostClient(): (hostId: string) => void { + return useRpcClientContext().refreshHostClient +} + +export function useForgetHostClient(): (hostId: string) => void { + return useRpcClientContext().forgetHostClient +} + +export function useDisconnectHostClient(): (hostId: string) => void { + return useRpcClientContext().disconnectHostClient +} + +export function useForceReconnect(): (hostId: string) => Promise { + return useRpcClientContext().forceReconnect +} + +export function usePrimeHosts(): (hosts: HostProfile[]) => void { + return useRpcClientContext().primeHosts +} diff --git a/mobile/src/transport/host-entry-opener.ts b/mobile/src/transport/host-entry-opener.ts index 6e29f55d985..03d6363c7c7 100644 --- a/mobile/src/transport/host-entry-opener.ts +++ b/mobile/src/transport/host-entry-opener.ts @@ -12,6 +12,7 @@ import type { ConnectionState, HostProfile } from './types' export type HostClientStoreEntry = { client: RpcClient + clientId: string state: ConnectionState refCount: number unsubState: () => void @@ -130,6 +131,7 @@ export async function openHostClientEntry( }) ?? (() => {}) const entry: HostClientStoreEntry = { client, + clientId: host.deviceToken, state: client.getState(), refCount: state.pendingAcquisitions.get(hostId) ?? 0, unsubState, diff --git a/mobile/src/transport/rpc-client-context-contract.ts b/mobile/src/transport/rpc-client-context-contract.ts index 9eb9efd4444..54e25973c7f 100644 --- a/mobile/src/transport/rpc-client-context-contract.ts +++ b/mobile/src/transport/rpc-client-context-contract.ts @@ -18,6 +18,7 @@ export type RpcClientContextValue = { disconnectHostClient: (hostId: string) => void getState: (hostId: string) => ConnectionState getKnownState: (hostId: string) => ConnectionState | null + getClientId: (hostId: string) => string | null getReconnectAttempt: (hostId: string) => number getLastConnectedAt: (hostId: string) => number | null getActivePath: (hostId: string) => MobileConnectionPath From 34222e01376d164530cd7f2e1bdaa04731b15db0 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:10:01 -0700 Subject: [PATCH 221/398] perf(orchestration): project explicit columns so the graph publish stops recompiling SQL (#18420) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(orchestration): cache the prepared statements the graph publish recompiles SyncDatabase refuses to cache any `SELECT *` — node:sqlite can build the first row after a schema change from stale column names — so every wildcard read in the orchestration DB recompiles its SQL on each call. The graph publish runs that fan-out once per pane, ~0.7 times a second, forever. Add a per-connection prepared-statement cache scoped to the orchestration DB, whose schema is frozen in the constructor (createTables/migrate/trigger) and whose resets are DELETE-only, and route the buildByPaneKey -> getForHandle -> getRecent path through it. 5 publishes over 2 panes: 30 compilations -> 2. * perf(orchestration): project explicit columns so the existing cache covers the hot path Replaces the branch's second statement cache. The six graph-publish reads were uncacheable only because they were spelled `SELECT *` / `SELECT t.*`, which SyncDatabase refuses to cache (node:sqlite can build the first row after a schema change from stale column names). Spelling the projection out from type-checked column tuples makes them cacheable by the SyncDatabase LRU that is already merged, already bounded, and already clears on DDL — so the WeakMap and its documented cross-connection ALTER hazard both go away. Drift is caught at build time: `satisfies readonly (keyof Row)[]` plus an `Exclude extends never` assertion pins list vs type at tsc, and a PRAGMA table_info test against a freshly migrated OrchestrationDb pins list vs schema. Same win, verified: 6 compilations per publish -> 2 total then 0, identical to the WeakMap branch; 92/96/91 us CPU per 2-pane publish before, 11-12 us after on both. --- .../db/dispatch-context/dispatch-lookup.ts | 54 +++---- .../db/hot-path-statement-compilation.test.ts | 148 ++++++++++++++++++ .../orchestration/db/row-column-lists.test.ts | 57 +++++++ .../orchestration/db/row-column-lists.ts | 82 ++++++++++ .../orchestration/db/runs/run-lookup.ts | 19 +-- .../orchestration/db/tasks/task-store.ts | 42 ++--- 6 files changed, 346 insertions(+), 56 deletions(-) create mode 100644 src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts create mode 100644 src/main/runtime/orchestration/db/row-column-lists.test.ts create mode 100644 src/main/runtime/orchestration/db/row-column-lists.ts diff --git a/src/main/runtime/orchestration/db/dispatch-context/dispatch-lookup.ts b/src/main/runtime/orchestration/db/dispatch-context/dispatch-lookup.ts index ae1cd4f45d1..c96238f13ee 100644 --- a/src/main/runtime/orchestration/db/dispatch-context/dispatch-lookup.ts +++ b/src/main/runtime/orchestration/db/dispatch-context/dispatch-lookup.ts @@ -6,6 +6,23 @@ import { paneKeyMatchSuffix } from '../pane-key-match' import type { OrchestrationDb } from '../orchestration-db' +import { DISPATCH_CONTEXT_COLUMN_LIST } from '../row-column-lists' + +// Why: hoisted and wildcard-free so the graph-publish fan-out hits the SyncDatabase statement cache. +const ACTIVE_DISPATCH_BY_HANDLE_SQL = + // Why: newest-first like the pane lookups below — an unordered LIMIT 1 could pin a stale row if a handle ever has two active dispatches. + `SELECT ${DISPATCH_CONTEXT_COLUMN_LIST} FROM dispatch_contexts + WHERE assignee_handle = ? AND status IN ('pending', 'dispatched') + ORDER BY rowid DESC LIMIT 1` +const ACTIVE_DISPATCH_BY_PANE_KEY_SQL = `SELECT ${DISPATCH_CONTEXT_COLUMN_LIST} FROM dispatch_contexts + WHERE assignee_pane_key = ? AND status IN ('pending', 'dispatched') + ORDER BY rowid DESC LIMIT 1` +const ACTIVE_DISPATCH_BY_PANE_SUFFIX_SQL = `SELECT ${DISPATCH_CONTEXT_COLUMN_LIST} FROM dispatch_contexts + WHERE assignee_pane_key IS NOT NULL + AND status IN ('pending', 'dispatched') AND instr(assignee_pane_key, ':') > 1 + AND ${DISPATCH_PANE_KEY_MATCH_SUFFIX_SQL} = ? + ORDER BY rowid DESC LIMIT 1` +const LATEST_DISPATCH_BY_HANDLE_SQL = `SELECT ${DISPATCH_CONTEXT_COLUMN_LIST} FROM dispatch_contexts WHERE assignee_handle = ? ORDER BY rowid DESC LIMIT 1` export function getActiveDispatchForTerminal( this: OrchestrationDb, @@ -116,14 +133,9 @@ export function findActiveDispatchForAssignee( assigneeHandle: string, assigneePaneKey?: string ): DispatchContextRow | undefined { - const byHandle = this.db - .prepare( - // Why: newest-first like the pane lookups below — an unordered LIMIT 1 could pin a stale row if a handle ever has two active dispatches. - `SELECT * FROM dispatch_contexts - WHERE assignee_handle = ? AND status IN ('pending', 'dispatched') - ORDER BY rowid DESC LIMIT 1` - ) - .get(assigneeHandle) as DispatchContextRow | undefined + const byHandle = this.db.prepare(ACTIVE_DISPATCH_BY_HANDLE_SQL).get(assigneeHandle) as + | DispatchContextRow + | undefined if (byHandle) { return byHandle } @@ -132,13 +144,9 @@ export function findActiveDispatchForAssignee( return undefined } - const exactPane = this.db - .prepare( - `SELECT * FROM dispatch_contexts - WHERE assignee_pane_key = ? AND status IN ('pending', 'dispatched') - ORDER BY rowid DESC LIMIT 1` - ) - .get(assigneePaneKey) as DispatchContextRow | undefined + const exactPane = this.db.prepare(ACTIVE_DISPATCH_BY_PANE_KEY_SQL).get(assigneePaneKey) as + | DispatchContextRow + | undefined if (exactPane) { return exactPane } @@ -146,13 +154,7 @@ export function findActiveDispatchForAssignee( return undefined } return this.db - .prepare( - `SELECT * FROM dispatch_contexts - WHERE assignee_pane_key IS NOT NULL - AND status IN ('pending', 'dispatched') AND instr(assignee_pane_key, ':') > 1 - AND ${DISPATCH_PANE_KEY_MATCH_SUFFIX_SQL} = ? - ORDER BY rowid DESC LIMIT 1` - ) + .prepare(ACTIVE_DISPATCH_BY_PANE_SUFFIX_SQL) .get(paneKeyMatchSuffix(assigneePaneKey)) as DispatchContextRow | undefined } @@ -160,11 +162,9 @@ export function getLatestDispatchForTerminal( this: OrchestrationDb, handle: string ): DispatchContextRow | undefined { - return this.db - .prepare( - 'SELECT * FROM dispatch_contexts WHERE assignee_handle = ? ORDER BY rowid DESC LIMIT 1' - ) - .get(handle) as DispatchContextRow | undefined + return this.db.prepare(LATEST_DISPATCH_BY_HANDLE_SQL).get(handle) as + | DispatchContextRow + | undefined } export type DispatchLookupMethods = { diff --git a/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts b/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts new file mode 100644 index 00000000000..26f82410cfe --- /dev/null +++ b/src/main/runtime/orchestration/db/hot-path-statement-compilation.test.ts @@ -0,0 +1,148 @@ +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { RuntimeAgentOrchestrationProjection } from '../../runtime-agent-orchestration-projection' +import type { OrchestrationCompatibilityTerminalAuthority } from '../../runtime-terminal-contracts' +import type { RuntimeLeafRecord } from '../../runtime-terminal-state-records' +import { OrchestrationDb } from '../db' +import { createRootDispatch } from './root-dispatch-test-fixture' + +const COORDINATOR_HANDLE = 'term_coordinator' +const COORDINATOR_PANE = 'tab_c:leaf_c' +const WORKER_HANDLE = 'term_worker' +const WORKER_PANE = 'tab_w:leaf_w' +const IDLE_HANDLE = 'term_idle' +const IDLE_PANE = 'tab_i:leaf_i' + +// Why: mirrors SyncDatabase's `isStatementCacheable` — aggregate `(*)` is fine, any other `*` is not. +const WILDCARD_PROJECTION = /(? { + for (const db of openDatabases.splice(0)) { + try { + db.close() + } catch { + // already closed by the test + } + } + for (const directory of temporaryDirectories.splice(0)) { + rmSync(directory, { recursive: true, force: true }) + } +}) + +function openDatabase(path: string): OrchestrationDb { + const db = new OrchestrationDb(path) + openDatabases.push(db) + return db +} + +function temporaryDatabasePath(): string { + const directory = mkdtempSync(join(tmpdir(), 'orca-orchestration-hot-path-')) + temporaryDirectories.push(directory) + return join(directory, 'orchestration.db') +} + +/** Counts real SQL compilations by wrapping the node:sqlite handle SyncDatabase prepares against. */ +function trackCompiledSql(db: OrchestrationDb): string[] { + const inner = (db.db as unknown as { db: { prepare(sql: string): unknown } }).db + const original = inner.prepare.bind(inner) + const compiled: string[] = [] + inner.prepare = (sql: string) => { + compiled.push(sql) + return original(sql) + } + return compiled +} + +function seedDispatchedWorker(db: OrchestrationDb): void { + const run = db.createRun({ + objective: 'demo', + coordinatorHandle: COORDINATOR_HANDLE, + coordinatorPaneKey: COORDINATOR_PANE + }) + const task = db.createTask({ + spec: 'ship the thing', + runId: run.id, + createdByTerminalHandle: COORDINATOR_HANDLE, + createdByPaneKey: COORDINATOR_PANE, + createdByProcessIncarnation: 'inc_1', + createdByRunGeneration: run.consumer_generation + }) + createRootDispatch(db, task.id, WORKER_HANDLE, WORKER_PANE) +} + +function buildProjection(db: OrchestrationDb): RuntimeAgentOrchestrationProjection { + const leaves = [{ ptyId: 'pty_w' }, { ptyId: 'pty_i' }] as unknown as RuntimeLeafRecord[] + const handleByLeaf = new Map([ + [leaves[0] as RuntimeLeafRecord, WORKER_HANDLE], + [leaves[1] as RuntimeLeafRecord, IDLE_HANDLE] + ]) + const paneByLeaf = new Map([ + [leaves[0] as RuntimeLeafRecord, WORKER_PANE], + [leaves[1] as RuntimeLeafRecord, IDLE_PANE] + ]) + return new RuntimeAgentOrchestrationProjection({ + getDb: () => db, + getLeaves: () => leaves, + getPtys: () => [], + issueLeafHandle: (leaf) => handleByLeaf.get(leaf) ?? '', + issuePtyHandle: () => '', + makePaneKey: (leaf) => paneByLeaf.get(leaf) ?? '', + getWorktreeId: () => null, + getHandleForPaneKey: (paneKey) => (paneKey === COORDINATOR_PANE ? COORDINATOR_HANDLE : null), + getPaneKey: (handle) => (handle === COORDINATOR_HANDLE ? COORDINATOR_PANE : null), + getDispatchAuthority: (handle) => + handle === COORDINATOR_HANDLE + ? ({ + paneKey: COORDINATOR_PANE, + processIncarnation: 'inc_1' + } as OrchestrationCompatibilityTerminalAuthority) + : null + }) +} + +describe('orchestration hot-path statement compilation', () => { + it('compiles each hot-path SQL exactly once across repeated graph publishes', () => { + const db = openDatabase(':memory:') + seedDispatchedWorker(db) + const projection = buildProjection(db) + const compiled = trackCompiledSql(db) + + const publishes = [projection.buildByPaneKey()] + const compiledByFirstPublish = [...compiled] + for (let publish = 0; publish < 4; publish += 1) { + publishes.push(projection.buildByPaneKey()) + } + + const compilationsPerSql = new Map() + for (const sql of compiled) { + compilationsPerSql.set(sql, (compilationsPerSql.get(sql) ?? 0) + 1) + } + expect([...compilationsPerSql].filter(([, count]) => count > 1)).toEqual([]) + expect(compiled).toEqual(compiledByFirstPublish) + // Why: a cache that changed what the fan-out returns would be worse than the recompiles. + expect(publishes[0]).toBeDefined() + for (const publish of publishes) { + expect(publish).toEqual(publishes[0]) + } + }) + + // Why: `SELECT *` is what made these statements uncacheable, and a retained wildcard is the only + // way node:sqlite could build a row from stale column names after another connection's ALTER. + // Seeds on one connection and publishes on a second so every compilation here is hot-path SQL. + it('publishes without compiling a single wildcard projection', () => { + const path = temporaryDatabasePath() + seedDispatchedWorker(openDatabase(path)) + + const reader = openDatabase(path) + const compiled = trackCompiledSql(reader) + buildProjection(reader).buildByPaneKey() + + expect(compiled.length).toBeGreaterThan(0) + expect(compiled.filter((sql) => WILDCARD_PROJECTION.test(sql))).toEqual([]) + }) +}) diff --git a/src/main/runtime/orchestration/db/row-column-lists.test.ts b/src/main/runtime/orchestration/db/row-column-lists.test.ts new file mode 100644 index 00000000000..2c4041bebaa --- /dev/null +++ b/src/main/runtime/orchestration/db/row-column-lists.test.ts @@ -0,0 +1,57 @@ +import { afterEach, describe, expect, it } from 'vitest' +import { OrchestrationDb } from './orchestration-db' +import { + DISPATCH_CONTEXT_COLUMNS, + RUN_COLUMNS, + selectColumns, + TASK_COLUMNS +} from './row-column-lists' + +let db: OrchestrationDb | undefined + +afterEach(() => { + db?.close() + db = undefined +}) + +function tableColumns(table: string): string[] { + const rows = (db as OrchestrationDb).db.pragma(`table_info(${table})`) as { name: string }[] + return rows.map((row) => row.name).sort() +} + +describe('row column lists', () => { + // Why: these lists replaced `SELECT *`, so a column added to the schema without being listed here + // would silently stop being read. tsc pins list↔type; this pins list↔schema. + it.each([ + ['runs', RUN_COLUMNS], + ['tasks', TASK_COLUMNS], + ['dispatch_contexts', DISPATCH_CONTEXT_COLUMNS] + ])('projects every %s column the migrated schema declares', (table, columns) => { + db = new OrchestrationDb(':memory:') + + expect([...columns].sort()).toEqual(tableColumns(table)) + }) + + it('qualifies each column when the statement joins under an alias', () => { + expect(selectColumns(['id', 'run_id'])).toBe('id, run_id') + expect(selectColumns(['id', 'run_id'], 't')).toBe('t.id, t.run_id') + }) + + // Why: an alias-qualified projection must key the returned row by the bare column name, exactly as + // the `t.*` it replaced did — otherwise every lineage consumer reads undefined. + it('returns bare column names for an alias-qualified projection', () => { + db = new OrchestrationDb(':memory:') + const run = db.createRun({ + objective: 'demo', + coordinatorHandle: 'term_c', + coordinatorPaneKey: 'tab_c:leaf_c' + }) + const task = db.createTask({ spec: 'work', runId: run.id }) + + const row = db.db + .prepare(`SELECT ${selectColumns(TASK_COLUMNS, 't')} FROM tasks t WHERE t.id = ?`) + .get(task.id) as Record + + expect(Object.keys(row).sort()).toEqual([...TASK_COLUMNS].sort()) + }) +}) diff --git a/src/main/runtime/orchestration/db/row-column-lists.ts b/src/main/runtime/orchestration/db/row-column-lists.ts new file mode 100644 index 00000000000..26255fe551a --- /dev/null +++ b/src/main/runtime/orchestration/db/row-column-lists.ts @@ -0,0 +1,82 @@ +import type { DispatchContextRow, RunRow, TaskRow } from '../types' + +// Why: `SyncDatabase` refuses to cache any `SELECT *` (node:sqlite can build the first row after a +// schema change from stale column names), so a wildcard read recompiles its SQL on every call. +// Spelling the projection out makes the hot-path statements cacheable by that existing LRU. +// Drift is caught twice: `satisfies` + the exhaustiveness assertions below pin list↔type at tsc, +// and `row-column-lists.test.ts` pins list↔schema against a freshly migrated database. + +export const RUN_COLUMNS = [ + 'id', + 'objective', + 'home_database', + 'coordinator_handle', + 'coordinator_pane_key', + 'consumer_generation', + 'legacy', + 'created_at', + 'updated_at' +] as const satisfies readonly (keyof RunRow)[] + +export const TASK_COLUMNS = [ + 'id', + 'run_id', + 'parent_id', + 'created_by_terminal_handle', + 'created_by_pane_key', + 'created_by_process_incarnation', + 'created_by_run_generation', + 'task_title', + 'display_name', + 'spec', + 'status', + 'deps', + 'result', + 'created_at', + 'completed_at' +] as const satisfies readonly (keyof TaskRow)[] + +export const DISPATCH_CONTEXT_COLUMNS = [ + 'id', + 'run_id', + 'task_id', + 'contract_version', + 'launch_token_hash', + 'assignee_handle', + 'assignee_pane_key', + 'capability_hash', + 'process_incarnation', + 'capability_revoked_at', + 'status', + 'failure_count', + 'last_failure', + 'termination_reason', + 'depth', + 'dispatched_at', + 'completed_at', + 'created_at', + 'last_heartbeat_at' +] as const satisfies readonly (keyof DispatchContextRow)[] + +// Compile check: a row field added without its column here would silently vanish from the +// projection that used to be `SELECT *`, so the missing key must fail the build. +type UnprojectedRunColumn = Exclude +type UnprojectedTaskColumn = Exclude +type UnprojectedDispatchContextColumn = Exclude< + keyof DispatchContextRow, + (typeof DISPATCH_CONTEXT_COLUMNS)[number] +> +const assertEveryRowColumnProjected: [ + UnprojectedRunColumn extends never ? true : never, + UnprojectedTaskColumn extends never ? true : never, + UnprojectedDispatchContextColumn extends never ? true : never +] = [true, true, true] +void assertEveryRowColumnProjected + +/** Projection list for a `SELECT`; `alias` qualifies each name for a joined table (`t.id, …`). */ +export function selectColumns(columns: readonly string[], alias?: string): string { + return columns.map((column) => (alias ? `${alias}.${column}` : column)).join(', ') +} + +export const RUN_COLUMN_LIST = selectColumns(RUN_COLUMNS) +export const DISPATCH_CONTEXT_COLUMN_LIST = selectColumns(DISPATCH_CONTEXT_COLUMNS) diff --git a/src/main/runtime/orchestration/db/runs/run-lookup.ts b/src/main/runtime/orchestration/db/runs/run-lookup.ts index 061a7b39497..84eeece7374 100644 --- a/src/main/runtime/orchestration/db/runs/run-lookup.ts +++ b/src/main/runtime/orchestration/db/runs/run-lookup.ts @@ -9,12 +9,20 @@ import { exposeRunTimestamps } from '../utc-timestamp' import { encodeRunListCursor, decodeRunListCursor } from '../run-list-cursor' import type { RunListPage } from '../run-list-page' import type { OrchestrationDb } from '../orchestration-db' +import { RUN_COLUMN_LIST } from '../row-column-lists' export type LegacyAdoptedMailboxOwner = { runId: string terminalHandle: string } +// Why: hoisted and wildcard-free so the per-publish run lookups hit the SyncDatabase statement cache. +const RUN_BY_ID_SQL = `SELECT ${RUN_COLUMN_LIST} FROM runs WHERE id = ?` +const RUNS_BOUND_TO_PANE_SQL = `SELECT ${RUN_COLUMN_LIST} FROM runs + WHERE coordinator_pane_key IS NOT NULL AND legacy = 0 + AND ${RUN_PANE_KEY_MATCH_SUFFIX_SQL} = ? + ORDER BY rowid` + export function getRun(this: OrchestrationDb, id: string): RunRow | undefined { const run = this.getRunRaw(id) return run ? exposeRunTimestamps(run) : undefined @@ -103,14 +111,7 @@ export function getCurrentRunForPane(this: OrchestrationDb, paneKey: string): Ru // reminted tab halves keep matching and unparseable keys keep requiring an exact match. export function runsBoundToPane(this: OrchestrationDb, paneKey: string): RunRow[] { return ( - this.db - .prepare( - `SELECT * FROM runs - WHERE coordinator_pane_key IS NOT NULL AND legacy = 0 - AND ${RUN_PANE_KEY_MATCH_SUFFIX_SQL} = ? - ORDER BY rowid` - ) - .all(paneKeyMatchSuffix(paneKey)) as RunRow[] + this.db.prepare(RUNS_BOUND_TO_PANE_SQL).all(paneKeyMatchSuffix(paneKey)) as RunRow[] ).filter( (run) => run.coordinator_pane_key !== null && isEquivalentPaneKey(run.coordinator_pane_key, paneKey) @@ -118,7 +119,7 @@ export function runsBoundToPane(this: OrchestrationDb, paneKey: string): RunRow[ } export function getRunRaw(this: OrchestrationDb, id: string): RunRow | undefined { - return this.db.prepare('SELECT * FROM runs WHERE id = ?').get(id) as RunRow | undefined + return this.db.prepare(RUN_BY_ID_SQL).get(id) as RunRow | undefined } export function unbindOtherRunsForPane( diff --git a/src/main/runtime/orchestration/db/tasks/task-store.ts b/src/main/runtime/orchestration/db/tasks/task-store.ts index 9a0b15259ba..69316765a74 100644 --- a/src/main/runtime/orchestration/db/tasks/task-store.ts +++ b/src/main/runtime/orchestration/db/tasks/task-store.ts @@ -5,6 +5,7 @@ import { LEGACY_RUN_ID } from '../contract-constants' import { generateId } from '../generated-id' import type { TaskRuntimeLineageRow } from '../run-list-page' import type { OrchestrationDb } from '../orchestration-db' +import { selectColumns, TASK_COLUMNS } from '../row-column-lists' // ── Tasks ── @@ -81,24 +82,8 @@ export function createTask( return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow } -// Why: return the active creator Dispatch proof with the Task read; runtime still owns pane/process currency. -export function getTask(this: OrchestrationDb, id: string): TaskRow | undefined -export function getTask( - this: OrchestrationDb, - id: string, - dispatchRunId: string -): TaskRuntimeLineageRow | undefined -export function getTask( - this: OrchestrationDb, - id: string, - dispatchRunId?: string -): TaskRow | TaskRuntimeLineageRow | undefined { - if (dispatchRunId === undefined) { - return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow | undefined - } - return this.db - .prepare( - `SELECT t.*, +// Why: hoisted and wildcard-free so the per-publish lineage lookup hits the SyncDatabase statement cache. +const TASK_RUNTIME_LINEAGE_SQL = `SELECT ${selectColumns(TASK_COLUMNS, 't')}, creator.id AS creator_dispatch_id, creator.run_id AS creator_dispatch_run_id, creator.assignee_pane_key AS creator_dispatch_pane_key, @@ -114,8 +99,25 @@ export function getTask( LIMIT 1 ) WHERE t.id = ?` - ) - .get(dispatchRunId, id) as TaskRuntimeLineageRow | undefined + +// Why: return the active creator Dispatch proof with the Task read; runtime still owns pane/process currency. +export function getTask(this: OrchestrationDb, id: string): TaskRow | undefined +export function getTask( + this: OrchestrationDb, + id: string, + dispatchRunId: string +): TaskRuntimeLineageRow | undefined +export function getTask( + this: OrchestrationDb, + id: string, + dispatchRunId?: string +): TaskRow | TaskRuntimeLineageRow | undefined { + if (dispatchRunId === undefined) { + return this.db.prepare('SELECT * FROM tasks WHERE id = ?').get(id) as TaskRow | undefined + } + return this.db.prepare(TASK_RUNTIME_LINEAGE_SQL).get(dispatchRunId, id) as + | TaskRuntimeLineageRow + | undefined } export function listTasks( From a711cb8b60e9cdf3b2e0aa2480d7ae2a2f1fcfe2 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:25:22 -0700 Subject: [PATCH 222/398] perf(renderer): gate the tab strip's worktree subscriptions and fix the orchestration batch's self-invalidating cache (#18428) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(renderer): gate the tab strip's worktree subscriptions and stop the orchestration batch invalidating itself Two store-subscription hot paths. The tab strip subscribed to projects/repos/worktreesByRepo for the Windows shell menu's local project runtime, which is never built unless that menu is on. On macOS/Linux every worktree write therefore re-rendered and re-committed every mounted tab strip. Gate the three on the condition that already gates their only consumer. The runtime-orchestration batch keyed its cache on agentStatusByPaneKey identity, which `agentStatus:set` replaces by definition, so it missed 100% of the time on the only event that calls it. Key on the paneKey -> worktreeId pairs the batch actually reads instead, and hang the requested-id array off the existing activeWorkspaces memo so the O(worktrees) prologue stops running per event. * refactor(renderer): make the orchestration batch's cache key its build's only inputs buildRuntimeBatch no longer receives agentStatusByPaneKey/retainedAgentsByPaneKey. It takes a RuntimeBatchInputs record whose paneWorktreeIds projection is its whole view of those maps, and that same record is the cache key, so the key cannot drift from the read set. Adds a guard asserting one read per orchestrated pane per map. * refactor(renderer): move the orchestration projection key onto the shared index The batch builder and `worktree-agent-orchestration-index.ts` were near-duplicate implementations of the same attribution walk, and both had the self-invalidating `liveSource === agentStatusByPaneKey` gate. Fixing only the batch left the index — which every mounted WorktreeCard hits on every `agentStatus:set` — still rebuilding per publication. Put `paneWorktreeIds` on the index instead and reduce the batch to a `.get`-compatible view of it. That deletes the whole `requestedWorktreeIds` apparatus the batch fix needed (the `worktreeIds` memo threading, the optional `selectDashboardOrchestration` param, the `uniqueWorktreeIdsByInput` WeakMap and its no-mutation contract, `getRequestedTabMembership`), leaves one builder guarded by the index's randomized oracle test, and extends the fix to the sidebar. The projection is memoised on the live/retained map identities so it is computed once per publication rather than once per card, and a successful ordered compare adopts the new array so the remaining cards compare by identity. --- .../build-dashboard-snapshot.test.ts | 18 +- ...worktree-agent-orchestration-batch.test.ts | 196 +++++++++++-- .../worktree-agent-orchestration-batch.ts | 263 ++---------------- ...worktree-agent-orchestration-index.test.ts | 88 ++++++ .../worktree-agent-orchestration-index.ts | 136 ++++++--- .../TabBar.worktree-write-gate.test.tsx | 111 ++++++++ ...abBar.worktree-write-gate.windows.test.tsx | 84 ++++++ ...-bar-runtime-model-worktree-write-probe.ts | 164 +++++++++++ .../tab-bar/use-tab-bar-runtime-model.ts | 21 +- 9 files changed, 774 insertions(+), 307 deletions(-) create mode 100644 src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.test.tsx create mode 100644 src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.windows.test.tsx create mode 100644 src/renderer/src/components/tab-bar/use-tab-bar-runtime-model-worktree-write-probe.ts diff --git a/src/renderer/src/components/dashboard/build-dashboard-snapshot.test.ts b/src/renderer/src/components/dashboard/build-dashboard-snapshot.test.ts index bae9892736d..698f759d7ab 100644 --- a/src/renderer/src/components/dashboard/build-dashboard-snapshot.test.ts +++ b/src/renderer/src/components/dashboard/build-dashboard-snapshot.test.ts @@ -10,6 +10,7 @@ import { makePaneKey } from '../../../../shared/stable-pane-id' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import type { Worktree } from '../../../../shared/worktree/types' import { selectRuntimeAgentOrchestrationBatch } from '../sidebar/worktree-agent-orchestration-batch' +import { selectRuntimeAgentOrchestrationForWorktree } from '../sidebar/worktree-agent-row-selectors' import type * as DashboardSnapshotWorkspacesModule from './dashboard-snapshot-workspaces' import type * as AgentRowLineageModule from './agent-row-lineage' @@ -753,7 +754,10 @@ describe('buildDashboardSnapshot', () => { expect(snapshot.cards[0].task).toBe('Batched orchestration task') }) - it('releases stale batch references when production moves from multi to singleton to zero', () => { + // Why identity, not release: the batch is a view of the shared orchestration index, which + // mounted sidebar cards read through. A dashboard that drops below two worktrees must not + // invalidate it, and nothing the index reads changed across these transitions. + it('keeps batch records live and correct when production moves from multi to singleton to zero', () => { const secondLeafId = '77777777-7777-4777-8777-777777777777' const firstPaneKey = makePaneKey('tab-w1', LEAF_ID) const secondPaneKey = makePaneKey('tab-w2', secondLeafId) @@ -788,13 +792,17 @@ describe('buildDashboardSnapshot', () => { NOW ) const afterSingleton = selectRuntimeAgentOrchestrationBatch(multiState, requested) - expect(afterSingleton).not.toBe(firstBatch) - expect(afterSingleton.get('w1')).not.toBe(firstW1) + expect(afterSingleton).toBe(firstBatch) + expect(afterSingleton.get('w1')).toBe(firstW1) buildDashboardSnapshot(baseState({ repos: [], worktreesByRepo: {} }), NOW) const afterZero = selectRuntimeAgentOrchestrationBatch(multiState, requested) - expect(afterZero).not.toBe(afterSingleton) - expect(afterZero.get('w1')).not.toBe(afterSingleton.get('w1')) + expect(afterZero).toBe(firstBatch) + for (const worktreeId of requested) { + expect(afterZero.get(worktreeId)).toBe( + selectRuntimeAgentOrchestrationForWorktree(multiState, worktreeId) + ) + } }) it('scans orchestration runtime once for a dashboard snapshot', () => { diff --git a/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.test.ts b/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.test.ts index 665cbb07896..705a1e0c43e 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.test.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.test.ts @@ -10,6 +10,10 @@ import { EMPTY_WORKTREE_AGENT_ORCHESTRATION, selectRuntimeAgentOrchestrationBatch } from './worktree-agent-orchestration-batch' +import { + _getWorktreeAgentOrchestrationIndexBuildCountForTest, + releaseWorktreeAgentOrchestrationIndexCache +} from './worktree-agent-orchestration-index' import { selectRuntimeAgentOrchestrationForWorktree } from './worktree-agent-row-selectors' type BatchState = Parameters[0] @@ -285,7 +289,9 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { expect(getBatchRecord(replacedBatch, 'wt-2')).toBe(firstWt2) }) - it('releases raw and derived caches for empty requests and empty runtime', () => { + // Why this matters now that the batch is a view of the shared index: an empty dashboard must + // not drop a cache that every mounted sidebar card is still reading through. + it('leaves the shared index intact for an empty request and rebuilds after an empty runtime', () => { let tabIdReads = 0 const state = { tabsByWorktree: { @@ -305,14 +311,15 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { const first = getBatchRecord(selectRuntimeAgentOrchestrationBatch(state, ['target']), 'target') expect(tabIdReads).toBe(1) - selectRuntimeAgentOrchestrationBatch(state, []) + expect(selectRuntimeAgentOrchestrationBatch(state, []).size).toBe(0) const afterEmptyRequest = getBatchRecord( selectRuntimeAgentOrchestrationBatch(state, ['target']), 'target' ) - expect(tabIdReads).toBe(2) - expect(afterEmptyRequest).not.toBe(first) + expect(tabIdReads).toBe(1) + expect(afterEmptyRequest).toBe(first) + // An emptied orchestration map is a real change of the index's own domain, so it does drop. selectRuntimeAgentOrchestrationBatch({ ...state, runtimeAgentOrchestrationByPaneKey: {} }, [ 'target' ]) @@ -320,11 +327,11 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { selectRuntimeAgentOrchestrationBatch(state, ['target']), 'target' ) - expect(tabIdReads).toBe(3) - expect(afterEmptyRuntime).not.toBe(afterEmptyRequest) + expect(tabIdReads).toBe(2) + expect(afterEmptyRuntime).not.toBe(first) }) - it('keeps singleton tab work target-local', () => { + it('matches the per-worktree selector for a single requested worktree', () => { const tabCount = 10 const contextCount = 8 const makeCountedState = () => { @@ -403,22 +410,17 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { ) expect(Object.keys(actual)).toEqual(Object.keys(expected)) - // Why the batch stays tighter: it knows which worktrees are on screen. The - // shared index covers all of them, so it saves per *card*, not per worktree. - expect(batched.counts()).toEqual({ - runtimeEnumerations: 1, - runtimeValueReads: contextCount, - contextVisits: contextCount, - targetTabIdReads: 1, - unrelatedTabIdReads: 0 - }) - expect(reference.counts()).toEqual({ + // Why identical: the batch is the shared index, which walks every worktree's tabs once per + // tabs-slice identity — not once per request — so a one-worktree request costs the same. + const singleWorktreeCounts = { runtimeEnumerations: 1, runtimeValueReads: contextCount, contextVisits: contextCount, targetTabIdReads: 1, unrelatedTabIdReads: tabCount - 1 - }) + } + expect(batched.counts()).toEqual(singleWorktreeCounts) + expect(reference.counts()).toEqual(singleWorktreeCounts) }) it('collapses multi-worktree runtime scans and caches unchanged publications', () => { @@ -521,15 +523,19 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { requested ) } + // The batch reads nothing but each orchestrated pane's worktreeId out of the live + // map, so publications that leave those alone never revisit a context at all. expect(batched.counts()).toEqual({ runtimeEnumerations: 1, runtimeValueReads: contextCount, - contextVisits: contextCount * (publicationCount + 1), + contextVisits: contextCount, tabIdReads: worktreeCount }) // Publications that change nothing the index reads cost nothing, however - // many cards call in. + // many cards call in. The warm-up pass is the cost of the batch loop above having left the + // one cache slot on a different fixture store; production has a single store. + selectRuntimeAgentOrchestrationForWorktree(reference.state, requested[0]) const referenceBefore = reference.counts() for (let publication = 0; publication < publicationCount; publication += 1) { for (const worktreeId of requested) { @@ -538,11 +544,9 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { } expect(reference.counts()).toEqual(referenceBefore) - // Why this is the honest claim: a real live-status ping replaces - // agentStatusByPaneKey, so the index does rebuild once per publication. What - // the shared index removes is the mounted-card multiplier, not the - // per-publication rebuild. Tab reads stay flat because tab membership is - // keyed on the tabs slice, which a live-status ping does not replace. + // A real live-status ping replaces agentStatusByPaneKey wholesale. The index is keyed on + // what it reads out of that map, not on its identity, so an unrelated pane's ping costs + // nothing: no rebuild, no context revisit, however many cards call in. const churn = makeCountedState() for (const worktreeId of requested) { selectRuntimeAgentOrchestrationForWorktree(churn.state, worktreeId) @@ -561,8 +565,148 @@ describe('selectRuntimeAgentOrchestrationBatch', () => { expect(churn.counts()).toEqual({ runtimeEnumerations: 1, runtimeValueReads: contextCount, - contextVisits: contextCount * (publicationCount + 1), + contextVisits: contextCount, tabIdReads: worktreeCount }) }) }) + +describe('selectRuntimeAgentOrchestrationBatch live-map churn', () => { + const ORCHESTRATED_CONTEXT = makeContext('orchestrated') + const SECOND_CONTEXT = makeContext('second') + const requested = ['wt-1', 'wt-2'] + // Held by identity so only the live map churns, as it does under `agentStatus:set`. + const TABS_BY_WORKTREE = { 'wt-1': [makeTab('unrelated-tab')], 'wt-2': [] } + const RUNTIME_ONE = { [CHILD_KEY]: ORCHESTRATED_CONTEXT } + const RUNTIME_TWO = { [CHILD_KEY]: ORCHESTRATED_CONTEXT, [SECOND_CHILD_KEY]: SECOND_CONTEXT } + const RETAINED = {} + + function makeChurnState( + agentStatusByPaneKey: Record, + runtimeAgentOrchestrationByPaneKey: Record< + string, + AgentStatusOrchestrationContext + > = RUNTIME_ONE + ): BatchState { + return { + tabsByWorktree: TABS_BY_WORKTREE, + runtimeAgentOrchestrationByPaneKey, + agentStatusByPaneKey, + retainedAgentsByPaneKey: RETAINED + } as BatchState + } + + function builds(): number { + return _getWorktreeAgentOrchestrationIndexBuildCountForTest() + } + + it('rebuilds once across repeated agentStatus:set identity churn on unrelated panes', () => { + releaseWorktreeAgentOrchestrationIndexCache() + const first = selectRuntimeAgentOrchestrationBatch( + makeChurnState({ [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-1') }), + requested + ) + const buildsAfterFirst = builds() + + for (let index = 0; index < 25; index += 1) { + // A fresh live map on every tick, exactly as `agentStatus:set` replaces the slice. + const churned = selectRuntimeAgentOrchestrationBatch( + makeChurnState({ + [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-1'), + [`unrelated-${index}`]: makeEntry(`unrelated-${index}`, 'wt-9') + }), + requested + ) + expect(churned).toBe(first) + } + expect(builds()).toBe(buildsAfterFirst) + expect(getBatchRecord(first, 'wt-1')[CHILD_KEY]).toBe(ORCHESTRATED_CONTEXT) + }) + + // Why this is a structural guard: the cache key is the projection, so anything the build + // reads straight out of the live/retained maps is unkeyed and can go stale. The build no + // longer receives those maps at all, which shows up here as exactly one read per pane. + it('reads each orchestrated pane out of the live and retained maps once per build', () => { + releaseWorktreeAgentOrchestrationIndexCache() + const liveReads: string[] = [] + const retainedReads: string[] = [] + const countReads = (target: Value, reads: string[]): Value => + new Proxy(target, { + get(source, key, receiver) { + if (typeof key === 'string') { + reads.push(key) + } + return Reflect.get(source, key, receiver) + } + }) + const state = { + tabsByWorktree: TABS_BY_WORKTREE, + runtimeAgentOrchestrationByPaneKey: RUNTIME_TWO, + agentStatusByPaneKey: countReads( + { + [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-1'), + [SECOND_CHILD_KEY]: makeEntry(SECOND_CHILD_KEY, 'wt-2') + }, + liveReads + ), + retainedAgentsByPaneKey: countReads( + { [CHILD_KEY]: makeRetained(CHILD_KEY, 'wt-1') }, + retainedReads + ) + } as BatchState + + const buildsBefore = builds() + const batch = selectRuntimeAgentOrchestrationBatch(state, requested) + + expect(builds()).toBe(buildsBefore + 1) + expect(liveReads).toEqual([CHILD_KEY, SECOND_CHILD_KEY]) + expect(retainedReads).toEqual([CHILD_KEY, SECOND_CHILD_KEY]) + expect(getBatchRecord(batch, 'wt-1')[CHILD_KEY]).toBe(ORCHESTRATED_CONTEXT) + expect(getBatchRecord(batch, 'wt-2')[SECOND_CHILD_KEY]).toBe(SECOND_CONTEXT) + }) + + it('rebuilds when an orchestrated pane changes worktree or the entry set changes', () => { + releaseWorktreeAgentOrchestrationIndexCache() + const first = selectRuntimeAgentOrchestrationBatch( + makeChurnState({ [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-1') }), + requested + ) + expect(getBatchRecord(first, 'wt-1')[CHILD_KEY]).toBe(ORCHESTRATED_CONTEXT) + expect(first.has('wt-2')).toBe(false) + + const movedBuilds = builds() + const moved = selectRuntimeAgentOrchestrationBatch( + makeChurnState({ [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-2') }), + requested + ) + expect(builds()).toBe(movedBuilds + 1) + expect(moved).not.toBe(first) + expect(moved.has('wt-1')).toBe(false) + expect(getBatchRecord(moved, 'wt-2')[CHILD_KEY]).toBe(ORCHESTRATED_CONTEXT) + + const addedBuilds = builds() + const added = selectRuntimeAgentOrchestrationBatch( + makeChurnState( + { + [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-2'), + [SECOND_CHILD_KEY]: makeEntry(SECOND_CHILD_KEY, 'wt-1') + }, + RUNTIME_TWO + ), + requested + ) + expect(builds()).toBe(addedBuilds + 1) + expect(Object.keys(getBatchRecord(added, 'wt-1'))).toEqual([SECOND_CHILD_KEY]) + + const removedBuilds = builds() + const removed = selectRuntimeAgentOrchestrationBatch( + makeChurnState({ + [CHILD_KEY]: makeEntry(CHILD_KEY, 'wt-2'), + [SECOND_CHILD_KEY]: makeEntry(SECOND_CHILD_KEY, 'wt-1') + }), + requested + ) + expect(builds()).toBe(removedBuilds + 1) + expect(removed.has('wt-1')).toBe(false) + }) +}) diff --git a/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.ts b/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.ts index d1d66c3be6e..4891632eeb3 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-orchestration-batch.ts @@ -1,6 +1,11 @@ import type { AppState } from '@/store/types' import type { AgentStatusOrchestrationContext } from '../../../../shared/agent-status-types' -import { parsePaneKey } from '../../../../shared/stable-pane-id' +import { + EMPTY_WORKTREE_AGENT_ORCHESTRATION_INDEX, + selectWorktreeAgentOrchestrationIndex +} from './worktree-agent-orchestration-index' + +export { EMPTY_WORKTREE_AGENT_ORCHESTRATION } from './worktree-agent-orchestration-index' type RuntimeOrchestrationState = Pick< AppState, @@ -10,246 +15,28 @@ type RuntimeOrchestrationState = Pick< | 'tabsByWorktree' > -type RuntimeOrchestrationMap = RuntimeOrchestrationState['runtimeAgentOrchestrationByPaneKey'] -type RuntimeOrchestrationRecord = Record - -type RuntimeDomainCache = { - source: RuntimeOrchestrationMap - orderedEntries: [string, AgentStatusOrchestrationContext][] -} - -type RequestedTabMembershipCache = { - tabsSource: RuntimeOrchestrationState['tabsByWorktree'] - requestedWorktreeIds: string[] - requestedIds: Set - worktreeIdsByTabId: Map> -} - -type RuntimeBatchCache = { - runtimeSource: RuntimeOrchestrationMap - tabsSource: RuntimeOrchestrationState['tabsByWorktree'] - liveSource: RuntimeOrchestrationState['agentStatusByPaneKey'] - retainedSource: RuntimeOrchestrationState['retainedAgentsByPaneKey'] - requestedWorktreeIds: string[] - recordsByWorktree: ReadonlyMap -} - -const EMPTY_RUNTIME_ORCHESTRATION: RuntimeOrchestrationMap = {} -const EMPTY_TABS_BY_WORKTREE: RuntimeOrchestrationState['tabsByWorktree'] = {} -const EMPTY_AGENT_STATUS: RuntimeOrchestrationState['agentStatusByPaneKey'] = {} -const EMPTY_RETAINED_AGENTS: RuntimeOrchestrationState['retainedAgentsByPaneKey'] = {} -const EMPTY_BATCH: ReadonlyMap = new Map() - -export const EMPTY_WORKTREE_AGENT_ORCHESTRATION: RuntimeOrchestrationRecord = Object.freeze({}) - -// Why null-prototype: a pane key of `__proto__` is a plain data key here; on a -// normal object the write vanishes into the prototype setter and repoints it. -function createRecord(): RuntimeOrchestrationRecord { - return Object.create(null) as RuntimeOrchestrationRecord -} - -let runtimeDomainCache: RuntimeDomainCache | null = null -let requestedTabMembershipCache: RequestedTabMembershipCache | null = null -let runtimeBatchCache: RuntimeBatchCache | null = null - -export function releaseRuntimeAgentOrchestrationBatchCache(): void { - runtimeDomainCache = null - requestedTabMembershipCache = null - runtimeBatchCache = null -} - -function getOrderedRuntimeEntries( - runtimeAgentOrchestrationByPaneKey: RuntimeOrchestrationMap -): [string, AgentStatusOrchestrationContext][] { - if (runtimeDomainCache?.source === runtimeAgentOrchestrationByPaneKey) { - return runtimeDomainCache.orderedEntries - } - const orderedEntries = Object.entries(runtimeAgentOrchestrationByPaneKey) - runtimeDomainCache = { source: runtimeAgentOrchestrationByPaneKey, orderedEntries } - return orderedEntries -} - -function uniqueWorktreeIds(worktreeIds: readonly string[]): string[] { - const uniqueIds: string[] = [] - const seen = new Set() - for (const worktreeId of worktreeIds) { - if (!seen.has(worktreeId)) { - seen.add(worktreeId) - uniqueIds.push(worktreeId) - } - } - return uniqueIds -} - -function hasSameWorktreeIds(previous: readonly string[], next: readonly string[]): boolean { - if (previous.length !== next.length) { - return false - } - return previous.every((worktreeId, index) => worktreeId === next[index]) -} - -function getRequestedTabMembership( - tabsByWorktree: RuntimeOrchestrationState['tabsByWorktree'], - requestedWorktreeIds: string[] -): RequestedTabMembershipCache { - if ( - requestedTabMembershipCache?.tabsSource === tabsByWorktree && - hasSameWorktreeIds(requestedTabMembershipCache.requestedWorktreeIds, requestedWorktreeIds) - ) { - return requestedTabMembershipCache - } - - const requestedIds = new Set(requestedWorktreeIds) - const worktreeIdsByTabId = new Map>() - for (const worktreeId of requestedWorktreeIds) { - // Why: the batch must not make a singleton dashboard scan unrelated tabs. - for (const tab of tabsByWorktree[worktreeId] ?? []) { - const tabId = tab.id - const existing = worktreeIdsByTabId.get(tabId) - if (existing) { - existing.add(worktreeId) - } else { - worktreeIdsByTabId.set(tabId, new Set([worktreeId])) - } - } - } - requestedTabMembershipCache = { - tabsSource: tabsByWorktree, - requestedWorktreeIds, - requestedIds, - worktreeIdsByTabId - } - return requestedTabMembershipCache -} - -function reuseRecordIfOrderedEqual( - previous: RuntimeOrchestrationRecord | undefined, - next: RuntimeOrchestrationRecord -): RuntimeOrchestrationRecord { - if (!previous) { - return next - } - const previousEntries = Object.entries(previous) - const nextEntries = Object.entries(next) - if (previousEntries.length !== nextEntries.length) { - return next - } - for (let index = 0; index < nextEntries.length; index += 1) { - if ( - previousEntries[index]?.[0] !== nextEntries[index]?.[0] || - previousEntries[index]?.[1] !== nextEntries[index]?.[1] - ) { - return next - } - } - return previous -} - -function buildRuntimeBatch( - requestedWorktreeIds: string[], - orderedRuntimeEntries: [string, AgentStatusOrchestrationContext][], - tabsByWorktree: RuntimeOrchestrationState['tabsByWorktree'], - agentStatusByPaneKey: RuntimeOrchestrationState['agentStatusByPaneKey'], - retainedAgentsByPaneKey: RuntimeOrchestrationState['retainedAgentsByPaneKey'] -): ReadonlyMap { - const { requestedIds, worktreeIdsByTabId } = getRequestedTabMembership( - tabsByWorktree, - requestedWorktreeIds - ) - - const recordsByWorktree = new Map() - for (const [paneKey, orchestration] of orderedRuntimeEntries) { - const targets = new Set() - const parsed = parsePaneKey(paneKey) - const parsedParent = orchestration.parentPaneKey - ? parsePaneKey(orchestration.parentPaneKey) - : null - if (parsed) { - for (const worktreeId of worktreeIdsByTabId.get(parsed.tabId) ?? []) { - targets.add(worktreeId) - } - } - if (parsedParent) { - for (const worktreeId of worktreeIdsByTabId.get(parsedParent.tabId) ?? []) { - targets.add(worktreeId) - } - } - - // Why: exact runtime keys preserve early SSH attribution and ignore stale - // entry.paneKey fields carried by a live or retained row. - const liveWorktreeId = agentStatusByPaneKey[paneKey]?.worktreeId - const retainedWorktreeId = retainedAgentsByPaneKey[paneKey]?.worktreeId - if (typeof liveWorktreeId === 'string' && requestedIds.has(liveWorktreeId)) { - targets.add(liveWorktreeId) - } - if (typeof retainedWorktreeId === 'string' && requestedIds.has(retainedWorktreeId)) { - targets.add(retainedWorktreeId) - } - - for (const worktreeId of targets) { - let record = recordsByWorktree.get(worktreeId) - if (!record) { - record = createRecord() - recordsByWorktree.set(worktreeId, record) - } - record[paneKey] = orchestration - } - } - - const previousRecords = runtimeBatchCache?.recordsByWorktree - for (const [worktreeId, record] of recordsByWorktree) { - recordsByWorktree.set( - worktreeId, - reuseRecordIfOrderedEqual(previousRecords?.get(worktreeId), record) - ) - } - return recordsByWorktree -} +/** + * No-op: the batch has no cache of its own. Kept because the dashboard's singleton and + * zero-worktree branches still announce that they are done with the batch view, and the shared + * index behind it must survive that — mounted sidebar cards are reading the same records. + */ +export function releaseRuntimeAgentOrchestrationBatchCache(): void {} +/** + * The dashboard's multi-worktree orchestration view. + * + * Why this is the shared index verbatim: the batch used to build its own worktree-keyed records + * from the same four slices, restricted to the requested ids. Callers only ever `.get(id)`, so + * the extra keys are unobservable, and one builder means one cache to keep honest and one + * correctness oracle to satisfy. `worktreeIds` survives only as the empty-dashboard + * short-circuit, which keeps the runtime map unread when nothing is on screen. + */ export function selectRuntimeAgentOrchestrationBatch( state: RuntimeOrchestrationState, worktreeIds: readonly string[] -): ReadonlyMap { - const requestedWorktreeIds = uniqueWorktreeIds(worktreeIds) - if (requestedWorktreeIds.length === 0) { - releaseRuntimeAgentOrchestrationBatchCache() - return EMPTY_BATCH +): ReadonlyMap> { + if (worktreeIds.length === 0) { + return EMPTY_WORKTREE_AGENT_ORCHESTRATION_INDEX } - - const runtimeAgentOrchestrationByPaneKey = - state.runtimeAgentOrchestrationByPaneKey ?? EMPTY_RUNTIME_ORCHESTRATION - const orderedRuntimeEntries = getOrderedRuntimeEntries(runtimeAgentOrchestrationByPaneKey) - if (orderedRuntimeEntries.length === 0) { - releaseRuntimeAgentOrchestrationBatchCache() - return EMPTY_BATCH - } - - const tabsByWorktree = state.tabsByWorktree ?? EMPTY_TABS_BY_WORKTREE - const agentStatusByPaneKey = state.agentStatusByPaneKey ?? EMPTY_AGENT_STATUS - const retainedAgentsByPaneKey = state.retainedAgentsByPaneKey ?? EMPTY_RETAINED_AGENTS - if ( - runtimeBatchCache?.runtimeSource === runtimeAgentOrchestrationByPaneKey && - runtimeBatchCache.tabsSource === tabsByWorktree && - runtimeBatchCache.liveSource === agentStatusByPaneKey && - runtimeBatchCache.retainedSource === retainedAgentsByPaneKey && - hasSameWorktreeIds(runtimeBatchCache.requestedWorktreeIds, requestedWorktreeIds) - ) { - return runtimeBatchCache.recordsByWorktree - } - - runtimeBatchCache = { - runtimeSource: runtimeAgentOrchestrationByPaneKey, - tabsSource: tabsByWorktree, - liveSource: agentStatusByPaneKey, - retainedSource: retainedAgentsByPaneKey, - requestedWorktreeIds, - recordsByWorktree: buildRuntimeBatch( - requestedWorktreeIds, - orderedRuntimeEntries, - tabsByWorktree, - agentStatusByPaneKey, - retainedAgentsByPaneKey - ) - } - return runtimeBatchCache.recordsByWorktree + return selectWorktreeAgentOrchestrationIndex(state) } diff --git a/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.test.ts b/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.test.ts index e7881d95aff..2d9ba917520 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.test.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.test.ts @@ -7,6 +7,7 @@ import type { RetainedAgentEntry } from '@/store/slices/agent-status' import type { TerminalTab } from '../../../../shared/terminal-tab-types' import { makePaneKey, parsePaneKey } from '../../../../shared/stable-pane-id' import { + _getWorktreeAgentOrchestrationIndexBuildCountForTest, EMPTY_WORKTREE_AGENT_ORCHESTRATION, releaseWorktreeAgentOrchestrationIndexCache, selectWorktreeAgentOrchestration @@ -247,6 +248,93 @@ describe('selectWorktreeAgentOrchestration', () => { expect(selectWorktreeAgentOrchestration(retainedChurn, 'wt-1')).toBe(first) }) + // Why a build counter and not record identity: `reuseRecordIfOrderedEqual` hides a rebuild + // from every identity assertion, so the wasted O(tabs + contexts) pass under `agentStatus:set` + // — several a second on a busy install, with a fresh live map each time — was invisible. + it('does not rebuild when agentStatus:set replaces the live map without moving a pane', () => { + const paneKey = paneKeyFor('tab-1', 0) + const context = { taskId: 't', dispatchId: 'd' } + const tabsByWorktree = { 'wt-1': [makeTab('tab-1')] } + const runtimeAgentOrchestrationByPaneKey = { [paneKey]: context } + const publish = (agentStatusByPaneKey: Record): IndexState => + ({ + tabsByWorktree, + runtimeAgentOrchestrationByPaneKey, + agentStatusByPaneKey, + retainedAgentsByPaneKey: {} + }) as unknown as IndexState + + const first = selectWorktreeAgentOrchestration( + publish({ [paneKey]: makeEntry(paneKey, 'wt-1') }), + 'wt-1' + ) + const buildsAfterFirst = _getWorktreeAgentOrchestrationIndexBuildCountForTest() + + for (let tick = 0; tick < 25; tick += 1) { + // A fresh live map every tick, exactly as `agentStatus:set` replaces the slice, plus a + // stable entry for the orchestrated pane so the projection is non-trivially equal. + const published = publish({ + [paneKey]: makeEntry(paneKey, 'wt-1'), + [`unrelated-${tick}`]: makeEntry(`unrelated-${tick}`, 'wt-9') + }) + expect(selectWorktreeAgentOrchestration(published, 'wt-1')).toBe(first) + } + expect(_getWorktreeAgentOrchestrationIndexBuildCountForTest()).toBe(buildsAfterFirst) + + // ...and the projection is still load-bearing: moving that pane must re-attribute it. + const moved = publish({ [paneKey]: makeEntry(paneKey, 'wt-2') }) + expect(selectWorktreeAgentOrchestration(moved, 'wt-2')[paneKey]).toBe(context) + expect(_getWorktreeAgentOrchestrationIndexBuildCountForTest()).toBe(buildsAfterFirst + 1) + }) + + // Why counted rather than timed: the projection is the index's per-publication work, and + // recomputing it per card would put the O(contexts) scan back on the per-card path that the + // index exists to remove — which no identity or correctness assertion would notice. + it('projects the live and retained maps once per publication, not once per card', () => { + const cardCount = 8 + const contextCount = 6 + const tabsByWorktree: Record = {} + const runtimeAgentOrchestrationByPaneKey: Record = {} + for (let index = 0; index < cardCount; index += 1) { + tabsByWorktree[`wt-${index}`] = [makeTab(`tab-${index}`)] + } + for (let index = 0; index < contextCount; index += 1) { + runtimeAgentOrchestrationByPaneKey[paneKeyFor(`tab-${index}`, index)] = { + taskId: `t-${index}`, + dispatchId: `d-${index}` + } + } + let liveReads = 0 + let retainedReads = 0 + const countReads = (target: object, onRead: () => void): object => + new Proxy(target, { + get(source, key, receiver) { + if (typeof key === 'string') { + onRead() + } + return Reflect.get(source, key, receiver) + } + }) + const state = { + tabsByWorktree, + runtimeAgentOrchestrationByPaneKey, + agentStatusByPaneKey: countReads({}, () => { + liveReads += 1 + }), + retainedAgentsByPaneKey: countReads({}, () => { + retainedReads += 1 + }) + } as unknown as IndexState + + for (let card = 0; card < cardCount; card += 1) { + selectWorktreeAgentOrchestration(state, `wt-${card}`) + } + expect({ liveReads, retainedReads }).toEqual({ + liveReads: contextCount, + retainedReads: contextCount + }) + }) + it('rebuilds when a source it reads actually changes', () => { const context = { taskId: 't', dispatchId: 'd' } const paneKey = paneKeyFor('tab-1', 0) diff --git a/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.ts b/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.ts index 5cad5ec1f48..00e0c8af9e3 100644 --- a/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.ts +++ b/src/renderer/src/components/sidebar/worktree-agent-orchestration-index.ts @@ -22,11 +22,18 @@ type TabMembershipCache = { worktreeIdsByTabId: Map> } +type PaneWorktreeProjectionCache = { + runtimeSource: OrchestrationIndexState['runtimeAgentOrchestrationByPaneKey'] + liveSource: OrchestrationIndexState['agentStatusByPaneKey'] + retainedSource: OrchestrationIndexState['retainedAgentsByPaneKey'] + paneWorktreeIds: readonly (string | undefined)[] +} + type OrchestrationIndexCache = { runtimeSource: OrchestrationIndexState['runtimeAgentOrchestrationByPaneKey'] tabsSource: OrchestrationIndexState['tabsByWorktree'] - liveSource: OrchestrationIndexState['agentStatusByPaneKey'] - retainedSource: OrchestrationIndexState['retainedAgentsByPaneKey'] + /** @see projectPaneWorktreeIds — the build's whole view of the live and retained maps. */ + paneWorktreeIds: readonly (string | undefined)[] recordsByWorktree: ReadonlyMap } @@ -51,14 +58,79 @@ function createRecord(): RuntimeOrchestrationRecord { let runtimeEntriesCache: RuntimeEntriesCache | null = null let tabMembershipCache: TabMembershipCache | null = null +let paneWorktreeProjectionCache: PaneWorktreeProjectionCache | null = null let orchestrationIndexCache: OrchestrationIndexCache | null = null +let indexBuildCount = 0 export function releaseWorktreeAgentOrchestrationIndexCache(): void { runtimeEntriesCache = null tabMembershipCache = null + paneWorktreeProjectionCache = null orchestrationIndexCache = null } +export function _getWorktreeAgentOrchestrationIndexBuildCountForTest(): number { + return indexBuildCount +} + +/** + * The build's whole view of the live and retained maps: the `worktreeId` each orchestrated pane + * key resolves to, as live,retained pairs in entry order. A status write for any other pane + * cannot change the index, so this projection — not the map identities — is the correct cache + * key, and `agentStatus:set` replaces those maps several times a second. + * + * Why exact runtime keys: this preserves early SSH attribution and ignores stale `entry.paneKey` + * fields carried by a live or retained row. + */ +function projectPaneWorktreeIds( + runtimeSource: OrchestrationIndexState['runtimeAgentOrchestrationByPaneKey'], + runtimeEntries: readonly [string, AgentStatusOrchestrationContext][], + agentStatusByPaneKey: OrchestrationIndexState['agentStatusByPaneKey'], + retainedAgentsByPaneKey: OrchestrationIndexState['retainedAgentsByPaneKey'] +): readonly (string | undefined)[] { + // Why memoised on the map identities: every mounted card calls this selector on the same + // publication, and re-walking the contexts per card is the per-card cost the index removes. + if ( + paneWorktreeProjectionCache?.runtimeSource === runtimeSource && + paneWorktreeProjectionCache.liveSource === agentStatusByPaneKey && + paneWorktreeProjectionCache.retainedSource === retainedAgentsByPaneKey + ) { + return paneWorktreeProjectionCache.paneWorktreeIds + } + const paneWorktreeIds: (string | undefined)[] = [] + for (const [paneKey] of runtimeEntries) { + paneWorktreeIds.push( + agentStatusByPaneKey[paneKey]?.worktreeId, + retainedAgentsByPaneKey[paneKey]?.worktreeId + ) + } + paneWorktreeProjectionCache = { + runtimeSource, + liveSource: agentStatusByPaneKey, + retainedSource: retainedAgentsByPaneKey, + paneWorktreeIds + } + return paneWorktreeIds +} + +function hasSameOrderedValues( + previous: readonly (string | undefined)[], + next: readonly (string | undefined)[] +): boolean { + if (previous === next) { + return true + } + if (previous.length !== next.length) { + return false + } + for (let index = 0; index < next.length; index += 1) { + if (previous[index] !== next[index]) { + return false + } + } + return true +} + function reuseRecordIfOrderedEqual( previous: RuntimeOrchestrationRecord | undefined, next: RuntimeOrchestrationRecord @@ -109,12 +181,13 @@ function getWorktreeIdsByTabId( function buildIndex( runtimeEntries: [string, AgentStatusOrchestrationContext][], tabsByWorktree: OrchestrationIndexState['tabsByWorktree'], - agentStatusByPaneKey: OrchestrationIndexState['agentStatusByPaneKey'], - retainedAgentsByPaneKey: OrchestrationIndexState['retainedAgentsByPaneKey'] + paneWorktreeIds: readonly (string | undefined)[] ): ReadonlyMap { + indexBuildCount += 1 const worktreeIdsByTabId = getWorktreeIdsByTabId(tabsByWorktree) const recordsByWorktree = new Map() + let projectionCursor = 0 for (const [paneKey, orchestration] of runtimeEntries) { const parsed = parsePaneKey(paneKey) const parsedParent = orchestration.parentPaneKey @@ -134,13 +207,12 @@ function buildIndex( targets.add(worktreeId) } } - // Why exact runtime keys: this preserves early SSH attribution and ignores - // stale entry.paneKey fields carried by a live or retained row. - const liveWorktreeId = agentStatusByPaneKey[paneKey]?.worktreeId + const liveWorktreeId = paneWorktreeIds[projectionCursor] + const retainedWorktreeId = paneWorktreeIds[projectionCursor + 1] + projectionCursor += 2 if (typeof liveWorktreeId === 'string') { targets.add(liveWorktreeId) } - const retainedWorktreeId = retainedAgentsByPaneKey[paneKey]?.worktreeId if (typeof retainedWorktreeId === 'string') { targets.add(retainedWorktreeId) } @@ -166,16 +238,15 @@ function buildIndex( } /** - * Worktree-keyed index of runtime agent orchestration contexts, rebuilt only - * when one of its four source maps changes identity. + * Worktree-keyed index of runtime agent orchestration contexts, rebuilt only when the context + * map, the tabs slice, or the per-pane worktree projection of the live/retained maps changes. * - * Why: every mounted worktree card subscribes to its own orchestration slice, - * and Zustand re-runs every subscriber's selector on every store publication. - * Scanning the whole context map per card made that O(cards x contexts). What - * this removes is the per-card multiplier, not the rebuild itself: an agent - * ping replaces the live map, so the index still rebuilds once per publication. - * The first caller through a given store version pays O(tabs + contexts); the - * rest are a Map lookup. + * Why: every mounted worktree card subscribes to its own orchestration slice, and Zustand + * re-runs every subscriber's selector on every store publication. Scanning the whole context + * map per card made that O(cards x contexts). Keying on the live and retained map identities + * then made the index rebuild once per `agentStatus:set` even though a status write for an + * unorchestrated pane cannot change a single record; keying on the projection instead is what + * makes those publications free. */ export function selectWorktreeAgentOrchestrationIndex( state: OrchestrationIndexState @@ -184,7 +255,7 @@ export function selectWorktreeAgentOrchestrationIndex( state.runtimeAgentOrchestrationByPaneKey ?? EMPTY_SOURCE // Why cached separately from the index: enumerating the context map is the // per-publication cost this index exists to remove, and the entry list stays - // valid even when a churning live/retained slice forces an index rebuild. + // valid across the live/retained churn the projection absorbs. if (runtimeEntriesCache?.source !== runtimeAgentOrchestrationByPaneKey) { runtimeEntriesCache = { source: runtimeAgentOrchestrationByPaneKey, @@ -198,35 +269,38 @@ export function selectWorktreeAgentOrchestrationIndex( // Why the entries cache survives: dropping it would re-enumerate the empty // map once per card, which is the per-publication cost this index removes. tabMembershipCache = null + paneWorktreeProjectionCache = null orchestrationIndexCache = null return EMPTY_WORKTREE_AGENT_ORCHESTRATION_INDEX } const tabsByWorktree = state.tabsByWorktree ?? EMPTY_SOURCE - const agentStatusByPaneKey = state.agentStatusByPaneKey ?? EMPTY_SOURCE - const retainedAgentsByPaneKey = state.retainedAgentsByPaneKey ?? EMPTY_SOURCE + const paneWorktreeIds = projectPaneWorktreeIds( + runtimeAgentOrchestrationByPaneKey, + runtimeEntries, + state.agentStatusByPaneKey ?? EMPTY_SOURCE, + state.retainedAgentsByPaneKey ?? EMPTY_SOURCE + ) if ( orchestrationIndexCache?.runtimeSource === runtimeAgentOrchestrationByPaneKey && orchestrationIndexCache.tabsSource === tabsByWorktree && - orchestrationIndexCache.liveSource === agentStatusByPaneKey && - orchestrationIndexCache.retainedSource === retainedAgentsByPaneKey + hasSameOrderedValues(orchestrationIndexCache.paneWorktreeIds, paneWorktreeIds) ) { + // Why adopt the equal array: the remaining cards on this publication then compare by + // identity instead of walking it again. + orchestrationIndexCache.paneWorktreeIds = paneWorktreeIds return orchestrationIndexCache.recordsByWorktree } + // buildIndex reuses the previous build's records, so publish the new cache only after it runs. + const recordsByWorktree = buildIndex(runtimeEntries, tabsByWorktree, paneWorktreeIds) orchestrationIndexCache = { runtimeSource: runtimeAgentOrchestrationByPaneKey, tabsSource: tabsByWorktree, - liveSource: agentStatusByPaneKey, - retainedSource: retainedAgentsByPaneKey, - recordsByWorktree: buildIndex( - runtimeEntries, - tabsByWorktree, - agentStatusByPaneKey, - retainedAgentsByPaneKey - ) + paneWorktreeIds, + recordsByWorktree } - return orchestrationIndexCache.recordsByWorktree + return recordsByWorktree } export function selectWorktreeAgentOrchestration( diff --git a/src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.test.tsx b/src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.test.tsx new file mode 100644 index 00000000000..66e283cf754 --- /dev/null +++ b/src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.test.tsx @@ -0,0 +1,111 @@ +// @vitest-environment happy-dom + +/** + * The tab strip must not wake on worktree writes. `projects`/`repos`/`worktreesByRepo` + * exist in the runtime model only to build the Windows shell menu's local project + * runtime; `worktreesByRepo` gets a new identity on every poller result, head-identity + * refresh and git-status write, so an ungated subscription re-renders and re-commits + * every mounted tab strip continuously on a large install. + * + * Runs as a non-Windows client. The menu-on branch is exercised through a `win32` + * host platform, which is how a paired web client on macOS legitimately gets the menu. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { act, cleanup, render } from '@testing-library/react' +import { + createTabBarProbeStore, + localProjectRuntimeSpy, + probeWindowsCapabilities, + pushWorktreeWrite, + TAB_BAR_PROBE_PROPS, + tabBarRuntimeModelStubs, + tabBarShellStubs, + tabBarSurfaceRenders, + type TabBarProbeStore +} from './use-tab-bar-runtime-model-worktree-write-probe' +import type { TabBarProps } from './tab-bar-props' + +vi.mock('@/store', async () => ({ useAppStore: await createTabBarProbeStore() })) +vi.mock('../../store', async () => ({ useAppStore: await createTabBarProbeStore() })) +vi.mock('@/hooks/useShortcutLabel', () => tabBarRuntimeModelStubs().shortcutLabels()) +vi.mock('@/hooks/useDetectedAgents', () => tabBarRuntimeModelStubs().detectedAgents()) +vi.mock('@/hooks/useAgentDetectionTarget', () => tabBarRuntimeModelStubs().detectionTarget()) +vi.mock('@/lib/connection-context', () => tabBarRuntimeModelStubs().connectionContext()) +vi.mock('@/lib/worktree-runtime-owner', () => tabBarRuntimeModelStubs().runtimeOwner()) +vi.mock('@/runtime/runtime-rpc-client', () => tabBarRuntimeModelStubs().runtimeRpcClient()) +vi.mock('@/lib/native-chat-transcript-readability', () => + tabBarRuntimeModelStubs().nativeChatReadability() +) +vi.mock('@/lib/client-creation-action-policy', () => tabBarRuntimeModelStubs().creationPolicy()) +vi.mock('./tab-agent-types-by-tab-id', () => tabBarRuntimeModelStubs().agentProjections()) +vi.mock('@/lib/local-preflight-context', () => tabBarRuntimeModelStubs().localPreflight()) +vi.mock('@/lib/windows-terminal-capabilities', () => + tabBarRuntimeModelStubs().windowsCapabilities() +) +vi.mock('./tab-bar-surface', () => tabBarShellStubs().surface()) +vi.mock('./use-tab-bar-create-menu-controller', () => tabBarShellStubs().createMenuController()) +vi.mock('./use-tab-bar-item-projection', () => tabBarShellStubs().itemProjection()) +vi.mock('./tab-strip-overflow-navigation', () => tabBarShellStubs().overflowNavigation()) +vi.mock('./tab-strip-drag-scroll', () => tabBarShellStubs().dragScroll()) +vi.mock('@/lib/pane-manager/client-hosted-browser-row-state', () => + tabBarShellStubs().clientHostedBrowserRows() +) + +const WORKTREE_WRITES = 25 + +async function renderTabBar(): Promise { + const { default: TabBar } = await import('./TabBar') + render() +} + +/** One commit per write, not one batched commit, so the render count is the real one. */ +async function pushWorktreeWrites(store: TabBarProbeStore): Promise { + for (let tick = 0; tick < WORKTREE_WRITES; tick += 1) { + await act(async () => { + pushWorktreeWrite(store, tick) + }) + } +} + +describe('TabBar worktree-write gate (non-Windows client)', () => { + beforeEach(async () => { + Object.defineProperty(navigator, 'userAgent', { + configurable: true, + value: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7)' + }) + probeWindowsCapabilities.hostPlatform = 'darwin' + tabBarSurfaceRenders.count = 0 + localProjectRuntimeSpy.mockClear() + ;(await createTabBarProbeStore()).setState({ worktreesByRepo: {}, projects: [], repos: [] }) + }) + + afterEach(() => { + cleanup() + }) + + it('does not re-render the tab strip when worktree writes republish worktreesByRepo', async () => { + const store = await createTabBarProbeStore() + await renderTabBar() + const rendersAtMount = tabBarSurfaceRenders.count + expect(rendersAtMount).toBeGreaterThan(0) + + await pushWorktreeWrites(store) + + expect(tabBarSurfaceRenders.count).toBe(rendersAtMount) + expect(localProjectRuntimeSpy).not.toHaveBeenCalled() + }) + + it('still tracks worktree writes when the Windows shell menu is on', async () => { + const store = await createTabBarProbeStore() + probeWindowsCapabilities.hostPlatform = 'win32' + await renderTabBar() + const rendersAtMount = tabBarSurfaceRenders.count + const runtimeCallsAtMount = localProjectRuntimeSpy.mock.calls.length + expect(runtimeCallsAtMount).toBeGreaterThan(0) + + await pushWorktreeWrites(store) + + expect(tabBarSurfaceRenders.count).toBe(rendersAtMount + WORKTREE_WRITES) + expect(localProjectRuntimeSpy.mock.calls.length).toBe(runtimeCallsAtMount + WORKTREE_WRITES) + }) +}) diff --git a/src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.windows.test.tsx b/src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.windows.test.tsx new file mode 100644 index 00000000000..75a132e47c0 --- /dev/null +++ b/src/renderer/src/components/tab-bar/TabBar.worktree-write-gate.windows.test.tsx @@ -0,0 +1,84 @@ +// @vitest-environment happy-dom + +/** + * Windows half of the tab-strip worktree-write gate. `isWindows` is read once at module + * load, so the two client platforms cannot share a file; see + * TabBar.worktree-write-gate.test.tsx for the non-Windows half and the full rationale. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { act, cleanup, render } from '@testing-library/react' +import { + createTabBarProbeStore, + localProjectRuntimeSpy, + probeWindowsCapabilities, + pushWorktreeWrite, + TAB_BAR_PROBE_PROPS, + tabBarRuntimeModelStubs, + tabBarShellStubs, + tabBarSurfaceRenders +} from './use-tab-bar-runtime-model-worktree-write-probe' +import type { TabBarProps } from './tab-bar-props' + +vi.mock('@/store', async () => ({ useAppStore: await createTabBarProbeStore() })) +vi.mock('../../store', async () => ({ useAppStore: await createTabBarProbeStore() })) +vi.mock('@/hooks/useShortcutLabel', () => tabBarRuntimeModelStubs().shortcutLabels()) +vi.mock('@/hooks/useDetectedAgents', () => tabBarRuntimeModelStubs().detectedAgents()) +vi.mock('@/hooks/useAgentDetectionTarget', () => tabBarRuntimeModelStubs().detectionTarget()) +vi.mock('@/lib/connection-context', () => tabBarRuntimeModelStubs().connectionContext()) +vi.mock('@/lib/worktree-runtime-owner', () => tabBarRuntimeModelStubs().runtimeOwner()) +vi.mock('@/runtime/runtime-rpc-client', () => tabBarRuntimeModelStubs().runtimeRpcClient()) +vi.mock('@/lib/native-chat-transcript-readability', () => + tabBarRuntimeModelStubs().nativeChatReadability() +) +vi.mock('@/lib/client-creation-action-policy', () => tabBarRuntimeModelStubs().creationPolicy()) +vi.mock('./tab-agent-types-by-tab-id', () => tabBarRuntimeModelStubs().agentProjections()) +vi.mock('@/lib/local-preflight-context', () => tabBarRuntimeModelStubs().localPreflight()) +vi.mock('@/lib/windows-terminal-capabilities', () => + tabBarRuntimeModelStubs().windowsCapabilities() +) +vi.mock('./tab-bar-surface', () => tabBarShellStubs().surface()) +vi.mock('./use-tab-bar-create-menu-controller', () => tabBarShellStubs().createMenuController()) +vi.mock('./use-tab-bar-item-projection', () => tabBarShellStubs().itemProjection()) +vi.mock('./tab-strip-overflow-navigation', () => tabBarShellStubs().overflowNavigation()) +vi.mock('./tab-strip-drag-scroll', () => tabBarShellStubs().dragScroll()) +vi.mock('@/lib/pane-manager/client-hosted-browser-row-state', () => + tabBarShellStubs().clientHostedBrowserRows() +) + +const WORKTREE_WRITES = 25 + +describe('TabBar worktree-write gate (Windows client)', () => { + beforeEach(async () => { + Object.defineProperty(navigator, 'userAgent', { + configurable: true, + value: 'Mozilla/5.0 (Windows NT 10.0; Win64; x64)' + }) + // A Windows client gets the shell menu before its capability probe resolves. + probeWindowsCapabilities.hostPlatform = null + tabBarSurfaceRenders.count = 0 + localProjectRuntimeSpy.mockClear() + ;(await createTabBarProbeStore()).setState({ worktreesByRepo: {}, projects: [], repos: [] }) + }) + + afterEach(() => { + cleanup() + }) + + it('keeps recomputing the local project runtime on every worktree write', async () => { + const store = await createTabBarProbeStore() + const { default: TabBar } = await import('./TabBar') + render() + const rendersAtMount = tabBarSurfaceRenders.count + const runtimeCallsAtMount = localProjectRuntimeSpy.mock.calls.length + expect(runtimeCallsAtMount).toBeGreaterThan(0) + + for (let tick = 0; tick < WORKTREE_WRITES; tick += 1) { + await act(async () => { + pushWorktreeWrite(store, tick) + }) + } + + expect(tabBarSurfaceRenders.count).toBe(rendersAtMount + WORKTREE_WRITES) + expect(localProjectRuntimeSpy.mock.calls.length).toBe(runtimeCallsAtMount + WORKTREE_WRITES) + }) +}) diff --git a/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model-worktree-write-probe.ts b/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model-worktree-write-probe.ts new file mode 100644 index 00000000000..028316b9b84 --- /dev/null +++ b/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model-worktree-write-probe.ts @@ -0,0 +1,164 @@ +import { vi } from 'vitest' +import type { WindowsTerminalCapabilities } from '@/lib/windows-terminal-capabilities' + +/** + * Shared rig for the tab-strip worktree-write gate tests. The platform check in + * `use-tab-bar-runtime-model` is read once at module load, so each platform needs + * its own test file; everything but the `vi.mock` calls lives here. + */ +export type TabBarProbeState = { + settings: Record | null + persistedUIReady: boolean + mobileEmulatorTabIntroDismissed: boolean + gitStatusByWorktree: Record + unifiedTabsByWorktree: Record + activeGroupIdByWorktree: Record + activeRepoId: string | null + activeWorktreeId: string | null + projects: unknown[] + repos: unknown[] + worktreesByRepo: Record + sshConnectionStates: Map + pinTab: (tabId: string) => void + unpinTab: (tabId: string) => void + toggleTabViewMode: (tabId: string) => void +} + +export type TabBarProbeStore = { + (selector: (state: TabBarProbeState) => unknown): unknown + getState: () => TabBarProbeState + setState: (partial: Partial) => void +} + +const noop = (): void => {} + +// Both specifiers resolve to the same store module; memoize so they share one instance. +export async function createTabBarProbeStore(): Promise { + const globalKey = '__tabBarRuntimeModelProbeStore' + const globals = globalThis as Record + if (!globals[globalKey]) { + const { create } = await import('zustand') + globals[globalKey] = create(() => ({ + settings: null, + persistedUIReady: true, + mobileEmulatorTabIntroDismissed: true, + gitStatusByWorktree: {}, + unifiedTabsByWorktree: {}, + activeGroupIdByWorktree: {}, + activeRepoId: null, + activeWorktreeId: null, + projects: [], + repos: [], + worktreesByRepo: {}, + sshConnectionStates: new Map(), + pinTab: noop, + unpinTab: noop, + toggleTabViewMode: noop + })) + } + return globals[globalKey] as TabBarProbeStore +} + +/** Mutable so a test can flip the probed host platform without changing identity. */ +export const probeWindowsCapabilities: WindowsTerminalCapabilities = { + wslAvailable: false, + wslDistros: [], + pwshAvailable: false, + gitBashAvailable: false, + hostPlatform: 'darwin', + isLoading: false +} + +export const localProjectRuntimeSpy = vi.fn(() => undefined) + +const AGENT_PROJECTIONS = Object.freeze({ + nativeChatEnabled: false, + tabAgentTypesByTabId: Object.freeze({}), + nativeChatTabWideFallbackUnsafeTabsById: Object.freeze({}) +}) +const CREATION_POLICY = Object.freeze({ + 'managed-browser': { state: 'enabled' }, + 'mobile-emulator': { state: 'enabled' } +}) +const DETECTED_AGENTS = Object.freeze({ detectedIds: Object.freeze([]) }) +const RUNTIME_TARGET = Object.freeze({ kind: 'local' }) +const CREATE_MENU = Object.freeze({}) +const ITEM_PROJECTION = Object.freeze({ + orderedItems: Object.freeze([]), + activeVisibleTabId: null, + tabStripLayoutKey: 'probe' +}) +const OVERFLOW_NAVIGATION = Object.freeze({ + scrollTabStrip: noop, + tabStripOverflowState: Object.freeze({ canScrollStart: false, canScrollEnd: false }) +}) +const DRAG_SCROLL = Object.freeze({ + isTabDragActive: false, + onDragScrollStartEnter: noop, + onDragScrollEndEnter: noop, + onDragScrollLeave: noop +}) + +export function tabBarRuntimeModelStubs(): Record Record> { + return { + shortcutLabels: () => ({ + useShortcutLabel: () => '', + useOptionalShortcutLabel: () => null + }), + detectedAgents: () => ({ useDetectedAgents: () => DETECTED_AGENTS }), + detectionTarget: () => ({ useAgentDetectionTargetForWorktree: () => null }), + connectionContext: () => ({ getConnectionIdFromState: () => null }), + runtimeOwner: () => ({ getRuntimeEnvironmentIdForWorktree: () => null }), + runtimeRpcClient: () => ({ getActiveRuntimeTarget: () => RUNTIME_TARGET }), + nativeChatReadability: () => ({ isNativeChatTranscriptLocalReadable: () => false }), + creationPolicy: () => ({ getClientCreationActionPolicy: () => CREATION_POLICY }), + agentProjections: () => ({ selectTabBarAgentProjections: () => AGENT_PROJECTIONS }), + localPreflight: () => ({ + getLocalProjectExecutionRuntimeContext: localProjectRuntimeSpy + }), + windowsCapabilities: () => ({ + getWindowsTerminalCapabilityOwnerKey: () => 'probe', + useWindowsTerminalCapabilities: () => probeWindowsCapabilities + }) + } +} + +export const tabBarSurfaceRenders = { count: 0 } + +export function tabBarShellStubs(): Record Record> { + return { + surface: () => ({ + renderTabBarSurface: () => { + tabBarSurfaceRenders.count += 1 + return null + } + }), + createMenuController: () => ({ useTabBarCreateMenuController: () => CREATE_MENU }), + itemProjection: () => ({ useTabBarItemProjection: () => ITEM_PROJECTION }), + overflowNavigation: () => ({ useTabStripOverflowNavigation: () => OVERFLOW_NAVIGATION }), + dragScroll: () => ({ useTabStripDragScrollHandlers: () => DRAG_SCROLL }), + clientHostedBrowserRows: () => ({ useActiveClientHostedBrowserRowId: () => null }) + } +} + +export const TAB_BAR_PROBE_PROPS = { + tabs: [], + activeTabId: null, + worktreeId: 'wt-target', + expandedPaneByTabId: {}, + onActivate: noop, + onClose: noop, + onCloseOthers: noop, + onCloseToRight: noop, + onCloseToLeft: noop, + onNewTerminalTab: noop, + onNewBrowserTab: noop, + onSetCustomTitle: noop, + onSetTabColor: noop, + onTogglePaneExpand: noop +} as const + +/** One fresh `worktreesByRepo` identity, exactly as a worktree write publishes it. */ +export function pushWorktreeWrite(store: TabBarProbeStore, tick: number): void { + store.setState({ worktreesByRepo: { 'repo-1': [{ id: `wt-${tick}`, repoId: 'repo-1' }] } }) +} diff --git a/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model.ts b/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model.ts index f4dfb67fb49..3436b2200f2 100644 --- a/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model.ts +++ b/src/renderer/src/components/tab-bar/use-tab-bar-runtime-model.ts @@ -39,6 +39,9 @@ type GitStatusEntries = AppStoreState['gitStatusByWorktree'][string] const EMPTY_GIT_STATUS_ENTRIES: GitStatusEntries = [] const EMPTY_AGENT_CMD_OVERRIDES: Partial> = {} const EMPTY_UNIFIED_TABS: readonly Tab[] = [] +const EMPTY_PROJECTS: AppStoreState['projects'] = [] +const EMPTY_REPOS: AppStoreState['repos'] = [] +const EMPTY_WORKTREES_BY_REPO: AppStoreState['worktreesByRepo'] = {} export function getProjectRuntimeShellMenuMode( projectRuntime: ProjectExecutionRuntimeResolution | undefined @@ -123,10 +126,7 @@ export function useTabBarRuntimeModel({ ) const activeRepoId = useAppStore((s) => s.activeRepoId) const activeWorktreeId = useAppStore((s) => s.activeWorktreeId) - const projects = useAppStore((s) => s.projects) - const repos = useAppStore((s) => s.repos) const settings = useAppStore((s) => s.settings) - const worktreesByRepo = useAppStore((s) => s.worktreesByRepo) // Why: use the worktree's owning host so offered Windows shells match the host that actually runs the terminal. const activeRuntimeEnvironmentId = useAppStore( (s) => getRuntimeEnvironmentIdForWorktree(s, worktreeId)?.trim() || null @@ -188,8 +188,17 @@ export function useTabBarRuntimeModel({ isWindowsClient: isWindows, worktreeHasRemoteConnection: Boolean(worktreeConnectionId) }) + // Why: `projects`/`repos`/`worktreesByRepo` feed nothing but the local runtime context below, and + // `worktreesByRepo` churns on every worktree write; ungated, each write re-renders every tab strip. + const needsLocalProjectRuntime = + showWindowsShellMenu && !activeRuntimeEnvironmentId?.trim() && !worktreeConnectionId + const projects = useAppStore((s) => (needsLocalProjectRuntime ? s.projects : EMPTY_PROJECTS)) + const repos = useAppStore((s) => (needsLocalProjectRuntime ? s.repos : EMPTY_REPOS)) + const worktreesByRepo = useAppStore((s) => + needsLocalProjectRuntime ? s.worktreesByRepo : EMPTY_WORKTREES_BY_REPO + ) const localProjectRuntime = useMemo(() => { - if (!showWindowsShellMenu || activeRuntimeEnvironmentId?.trim() || worktreeConnectionId) { + if (!needsLocalProjectRuntime) { return undefined } return getLocalProjectExecutionRuntimeContext( @@ -207,13 +216,11 @@ export function useTabBarRuntimeModel({ ) }, [ activeRepoId, - activeRuntimeEnvironmentId, activeWorktreeId, + needsLocalProjectRuntime, projects, repos, settings, - showWindowsShellMenu, - worktreeConnectionId, windowsTerminalCapabilities.isLoading, windowsTerminalCapabilities.wslAvailable, windowsTerminalCapabilities.wslDistros, From 07e50e9513d6f0b19ceb11042758ea49c20c7954 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:34:42 -0700 Subject: [PATCH 223/398] perf(terminal): scan only new tail lines for the wait-blocked sentinel (#18437) * perf(terminal): scan only new tail lines for the wait-blocked sentinel The wait-blocked scan must prove a signal is ABSENT, so it could not early-exit and re-tested all 2000 retained lines with a 13-alternative regex on every scan (20/s per streaming PTY) even though only ~20 lines were new. Index the matching line indices per tail-array identity and carry them across appends, testing only the lines each append produced. Also carries the retained character total and the redraw prefix's right-trimmed state across appends, so a saturated tail is no longer re-summed and re-scanned per chunk. * perf(terminal): build the carried tail window and its match index from one constructor --- src/main/runtime/terminal-tail-buffer.test.ts | 157 +++++++ src/main/runtime/terminal-tail-buffer.ts | 237 +++++++--- .../terminal-tail-sentinel-index.test.ts | 428 ++++++++++++++++++ .../runtime/terminal-tail-sentinel-index.ts | 94 ++++ src/main/runtime/terminal-wait-tail-state.ts | 28 +- 5 files changed, 870 insertions(+), 74 deletions(-) create mode 100644 src/main/runtime/terminal-tail-buffer.test.ts create mode 100644 src/main/runtime/terminal-tail-sentinel-index.test.ts create mode 100644 src/main/runtime/terminal-tail-sentinel-index.ts diff --git a/src/main/runtime/terminal-tail-buffer.test.ts b/src/main/runtime/terminal-tail-buffer.test.ts new file mode 100644 index 00000000000..9851bd7fdc6 --- /dev/null +++ b/src/main/runtime/terminal-tail-buffer.test.ts @@ -0,0 +1,157 @@ +import { describe, expect, it, vi } from 'vitest' +import { appendNormalizedToTailBuffer } from './terminal-tail-buffer' +import { MAX_TAIL_LINES } from './terminal-tail-limits' +import type { RetainedTailRedrawCursor } from './terminal-tail-redraw-buffer' + +// Guards the per-chunk prefix work in appendNormalizedToTailBuffer: the retained char total is +// carried across appends and the redraw prefix is not re-scanned, so a saturated tail must not be +// walked once per chunk. Correctness is pinned by the cold/warm differential below — a "cold" run +// hands every append a fresh array so the memo always misses and every total is summed in full. + +type TailSim = { + lines: string[] + partialLine: string + redrawCursor: RetainedTailRedrawCursor | null +} + +function newSim(): TailSim { + return { lines: [], partialLine: '', redrawCursor: null } +} + +type Step = ReturnType + +function feed(sim: TailSim, chunk: string, cold: boolean): Step { + const next = appendNormalizedToTailBuffer( + cold ? [...sim.lines] : sim.lines, + sim.partialLine, + chunk, + sim.redrawCursor + ) + sim.lines = next.lines + sim.partialLine = next.partialLine + sim.redrawCursor = next.redrawCursor + return next +} + +function mulberry32(seed: number): () => number { + let state = seed >>> 0 + return () => { + state = (state + 0x6d2b79f5) >>> 0 + let t = Math.imul(state ^ (state >>> 15), 1 | state) + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } +} + +const ESC = String.fromCharCode(27) + +/** + * `short` fills the 2000-line cap and mixes in TUI redraws; `long` streams lines wide enough to + * hit the 256 KiB character cap first. Both eviction paths adjust the carried character total, so + * both need differential coverage, and a redraw's row truncation would keep `long` off its cap. + */ +function randomChunk(random: () => number, profile: 'short' | 'long'): string { + const roll = random() + if (profile === 'long') { + if (roll < 0.7) { + return `${'w'.repeat(1000 + Math.floor(random() * 3000))}\n` + } + if (roll < 0.8) { + return `\rspinner ${Math.floor(random() * 100)}%` + } + if (roll < 0.9) { + return 'trailing spaces here \n' + } + return roll < 0.95 ? '' : `no newline ${Math.floor(random() * 1000)}` + } + if (roll < 0.22) { + const lines: string[] = [] + for (let index = 0; index < 30; index += 1) { + lines.push(`burst ${Math.floor(random() * 1e6)}${random() < 0.3 ? ' ' : ''}`) + } + return `${lines.join('\n')}\n` + } + if (roll < 0.42) { + return `plain output ${Math.floor(random() * 1e6)}\n` + } + if (roll < 0.56) { + return `${' '.repeat(Math.floor(random() * 3))}\n` + } + if (roll < 0.68) { + const rows = 1 + Math.floor(random() * 12) + return `${ESC}[${rows}A${ESC}[2Kredrawn ${Math.floor(random() * 1000)}\n` + } + if (roll < 0.8) { + return `\rspinner ${Math.floor(random() * 100)}%` + } + if (roll < 0.86) { + return 'trailing spaces here \n' + } + if (roll < 0.92) { + return `multi\nline\nchunk ${Math.floor(random() * 1000)}\n` + } + if (roll < 0.96) { + return '' + } + return `no newline ${Math.floor(random() * 1000)}` +} + +describe('retained tail buffer prefix reuse', () => { + for (const profile of ['short', 'long'] as const) { + for (const seed of [3, 11, 91, 2024]) { + it(`carries the retained char total exactly (${profile}, seed ${seed})`, () => { + const random = mulberry32(seed) + const warm = newSim() + const cold = newSim() + let sawCap = false + for (let step = 0; step < 1400; step += 1) { + const chunk = randomChunk(random, profile) + const lineCountBefore = warm.lines.length + const warmStep = feed(warm, chunk, false) + const coldStep = feed(cold, chunk, true) + expect(warmStep.lines, `step ${step} lines`).toEqual(coldStep.lines) + expect(warmStep.partialLine, `step ${step} partial`).toBe(coldStep.partialLine) + expect(warmStep.truncated, `step ${step} truncated`).toBe(coldStep.truncated) + expect(warmStep.redrawCursor, `step ${step} cursor`).toEqual(coldStep.redrawCursor) + expect(warmStep.newCompleteLines, `step ${step} newCompleteLines`).toBe( + coldStep.newCompleteLines + ) + expect(warmStep.newlyCompletedLines, `step ${step} newlyCompletedLines`).toEqual( + coldStep.newlyCompletedLines + ) + sawCap = + sawCap || + (profile === 'short' + ? warm.lines.length >= MAX_TAIL_LINES + : // Lines dropped below the line cap on an append-only chunk == character-cap eviction. + !chunk.includes(ESC) && + warm.lines.length < MAX_TAIL_LINES && + lineCountBefore + warmStep.newlyCompletedLines.length > warm.lines.length) + } + // Guard against a vacuous pass: the profile's eviction path must have run. + expect(sawCap).toBe(true) + }) + } + } + + it('does not walk the untouched redraw prefix on every chunk', () => { + const sim = newSim() + for (let index = 0; index < MAX_TAIL_LINES + 200; index += 1) { + feed(sim, `streaming build output line ${index}\n`, false) + } + expect(sim.lines.length).toBe(MAX_TAIL_LINES) + + const redrawChunk = `${ESC}[3A${ESC}[2Krewritten row${ESC}[2B\n` + const spy = vi.spyOn(String.prototype, 'charCodeAt') + let prefixTouches = 0 + try { + feed(sim, redrawChunk, false) + prefixTouches = spy.mock.calls.length + } finally { + spy.mockRestore() + } + // Before this change the prefix trailing-space scan alone cost one charCodeAt per retained + // row (~1990); the chunk itself accounts for well under a hundred. + expect(prefixTouches).toBeLessThan(300) + }) +}) diff --git a/src/main/runtime/terminal-tail-buffer.ts b/src/main/runtime/terminal-tail-buffer.ts index 0cf551e92c6..141b14da6ec 100644 --- a/src/main/runtime/terminal-tail-buffer.ts +++ b/src/main/runtime/terminal-tail-buffer.ts @@ -1,4 +1,5 @@ import { containsTerminalVerticalLineControl } from './terminal-ansi-normalization' +import { carryTerminalTailSentinelMatches } from './terminal-tail-sentinel-index' import { applyTerminalLineControls, processTerminalTailCompleteSegments, @@ -11,6 +12,122 @@ import { type RetainedTailRedrawCursor } from './terminal-tail-redraw-buffer' +type RetainedTailLineStats = { + totalChars: number + /** Whether every line is already right-trimmed, so the redraw prefix trim is a no-op. */ + rightTrimmed: boolean +} + +// Why weak + array-keyed: the tail is replaced (never mutated) on every append, so an entry dies +// with the array it describes and only the live tail per PTY is retained. Carrying the char total +// this way replaces a full-tail re-sum on every chunk. +const tailLineStatsByLines = new WeakMap() + +function getRetainedTailLineStats(lines: readonly string[]): RetainedTailLineStats { + const cached = tailLineStatsByLines.get(lines) + if (cached) { + return cached + } + let totalChars = 0 + let rightTrimmed = true + for (const line of lines) { + totalChars += line.length + if (rightTrimmed && trimTerminalLineRight(line) !== line) { + rightTrimmed = false + } + } + const stats = { totalChars, rightTrimmed } + tailLineStatsByLines.set(lines, stats) + return stats +} + +type CarriedTailBuild = { + lines: string[] + /** Whether a retention cap dropped a row. */ + truncated: boolean +} + +/** + * The only way to produce a next tail array: `previousLines[keepStart, keepEnd) ++ appended`, + * capped by `MAX_TAIL_LINES` and — when `charCapPartialChars` is non-null — `MAX_TAIL_CHARS`. + * + * Why a constructor rather than three call sites doing their own arithmetic: the carried-match + * window handed to the sentinel index, the character total, and the array itself are all derived + * here from the same keep bounds, including whatever the caps drop, so they cannot disagree. The + * one thing a caller still has to get right is that every row in `appended` is already + * right-trimmed, which every producer of retained rows does. + */ +function buildCarriedTailLines( + previousLines: string[], + keepStart: number, + keepEnd: number, + appended: readonly string[], + charCapPartialChars: number | null +): CarriedTailBuild { + const keptCount = keepEnd > keepStart ? keepEnd - keepStart : 0 + let totalChars = 0 + let carriedRightTrimmed = true + if (keptCount > 0) { + const previousStats = getRetainedTailLineStats(previousLines) + totalChars = previousStats.totalChars + carriedRightTrimmed = previousStats.rightTrimmed + for (let index = 0; index < keepStart; index += 1) { + totalChars -= previousLines[index]!.length + } + for (let index = keepEnd; index < previousLines.length; index += 1) { + totalChars -= previousLines[index]!.length + } + } + for (const line of appended) { + totalChars += line.length + } + + // Both caps only ever drop from the front, so resolve them against the virtual concatenation + // before the array exists; the surviving keep bounds then define the carried window exactly. + const combinedLength = keptCount + appended.length + let dropCount = combinedLength > MAX_TAIL_LINES ? combinedLength - MAX_TAIL_LINES : 0 + for (let index = 0; index < dropCount; index += 1) { + totalChars -= ( + index < keptCount ? previousLines[keepStart + index]! : appended[index - keptCount]! + ).length + } + if (charCapPartialChars !== null) { + const charBudget = MAX_TAIL_CHARS - charCapPartialChars + while (dropCount < combinedLength && totalChars > charBudget) { + totalChars -= ( + dropCount < keptCount + ? previousLines[keepStart + dropCount]! + : appended[dropCount - keptCount]! + ).length + dropCount += 1 + } + } + + if ( + dropCount === 0 && + appended.length === 0 && + keepStart === 0 && + keepEnd === previousLines.length + ) { + return { lines: previousLines, truncated: false } + } + + const droppedFromCarried = dropCount < keptCount ? dropCount : keptCount + const carriedSourceStart = keepStart + droppedFromCarried + const carriedCount = keptCount - droppedFromCarried + const lines = previousLines.slice(carriedSourceStart, keepEnd) + for (let index = dropCount - droppedFromCarried; index < appended.length; index += 1) { + lines.push(appended[index]!) + } + + tailLineStatsByLines.set(lines, { + totalChars, + rightTrimmed: carriedCount === 0 || carriedRightTrimmed + }) + carryTerminalTailSentinelMatches(previousLines, lines, carriedSourceStart, carriedCount) + return { lines, truncated: dropCount > 0 } +} + export function appendNormalizedToTailBuffer( previousLines: string[], previousPartialLine: string, @@ -52,42 +169,35 @@ export function appendNormalizedToTailBuffer( // Why: status UIs redraw one line via CR/backspace/erase; retain the latest redraw segment instead of appending every spinner frame. const segments = splitRetainedTerminalTailSegments(combinedChunk) const pieces = processTerminalTailCompleteSegments(segments.completeSegments) - const newlyCompletedLines = pieces.map((line) => trimTerminalLineRight(line)) + const newlyCompletedLines: string[] = [] + for (const piece of pieces) { + newlyCompletedLines.push(trimTerminalLineRight(piece)) + } const partialResult = applyTerminalLineControls(segments.partialSegment) const nextPartialLine = trimTerminalLineRight(partialResult.text) const retainedPartialLine = nextPartialLine.slice(-MAX_TAIL_PARTIAL_CHARS) const newCompleteLines = segments.completeLineCount const omittedNewCompleteLines = newCompleteLines - pieces.length - let nextLines = - newCompleteLines > 0 - ? [...(omittedNewCompleteLines > 0 ? [] : previousLines), ...newlyCompletedLines] - : previousLines - let truncated = + + // The plain path only ever appends, so the whole previous tail carries unless it was discarded. + const carriesPreviousLines = newCompleteLines === 0 || omittedNewCompleteLines === 0 + const built = buildCarriedTailLines( + previousLines, + carriesPreviousLines ? 0 : previousLines.length, + previousLines.length, + newlyCompletedLines, + // Why gated: a chunk that neither completes a line nor grows the partial cannot breach the + // character cap, and re-checking it would evict on a tail that has not changed size. + newCompleteLines > 0 || retainedPartialLine.length > previousPartialLine.length + ? retainedPartialLine.length + : null + ) + const nextLines = built.lines + const truncated = previousPartialWasCapped || omittedNewCompleteLines > 0 || - nextPartialLine.length > MAX_TAIL_PARTIAL_CHARS - - if (nextLines.length > MAX_TAIL_LINES) { - nextLines = nextLines.slice(nextLines.length - MAX_TAIL_LINES) - truncated = true - } - - if (newCompleteLines > 0 || retainedPartialLine.length > previousPartialLine.length) { - if (nextLines === previousLines) { - nextLines = [...previousLines] - } - let totalChars = - nextLines.reduce((sum, line) => sum + line.length, 0) + retainedPartialLine.length - let trimStartIndex = 0 - while (trimStartIndex < nextLines.length && totalChars > MAX_TAIL_CHARS) { - totalChars -= nextLines[trimStartIndex].length - trimStartIndex += 1 - } - if (trimStartIndex > 0) { - nextLines = nextLines.slice(trimStartIndex) - truncated = true - } - } + nextPartialLine.length > MAX_TAIL_PARTIAL_CHARS || + built.truncated const redrawCursor = !partialResult.hadControl || partialResult.cursorColumn === nextPartialLine.length @@ -145,13 +255,22 @@ function appendNormalizedToMultilineTailBuffer( const windowRows = maxUpwardCursorReach(normalizedChunk, previousRedrawCursor) + REDRAW_WINDOW_SAFETY_ROWS if (windowRows >= previousLines.length) { - return appendNormalizedToMultilineTailBufferUnwindowed( + const unwindowed = appendNormalizedToMultilineTailBufferUnwindowed( previousLines, boundedPreviousPartialLine, normalizedChunk, previousPartialWasCapped, previousRedrawCursor ) + if (unwindowed.lines === previousLines) { + return unwindowed + } + // Why nothing carries: an unwindowed redraw may rewrite any retained row. Both caps were + // already applied inside the unwindowed builder, so the constructor only registers here. + return { + ...unwindowed, + lines: buildCarriedTailLines(previousLines, 0, 0, unwindowed.lines, null).lines + } } const prefixLength = previousLines.length - windowRows const suffix = previousLines.slice(prefixLength) @@ -162,41 +281,39 @@ function appendNormalizedToMultilineTailBuffer( previousPartialWasCapped, previousRedrawCursor ) - let lines = previousLines.slice(0, prefixLength) - // Why: the shared prefix must match the unwindowed finalize's trailing-space trim without paying a regex per untouched row. - for (let index = 0; index < lines.length; index += 1) { - const line = lines[index]! - const lastChar = line.charCodeAt(line.length - 1) - if (lastChar === 32 || lastChar === 9) { - lines[index] = line.replace(/[ \t]+$/g, '') + // The window provably cannot reach the prefix, so it carries unchanged — unless the tail + // entered un-right-trimmed, in which case the prefix has to be rewritten to match the + // unwindowed finalize's trailing-space trim and is therefore no longer the previous tail's rows. + const previousStats = getRetainedTailLineStats(previousLines) + let keepEnd = prefixLength + let appended: readonly string[] = windowed.lines + if (!previousStats.rightTrimmed) { + const rewritten = previousLines.slice(0, prefixLength) + for (let index = 0; index < rewritten.length; index += 1) { + const line = rewritten[index]! + const lastChar = line.charCodeAt(line.length - 1) + if (lastChar === 32 || lastChar === 9) { + rewritten[index] = line.replace(/[ \t]+$/g, '') + } } + for (const line of windowed.lines) { + rewritten.push(line) + } + keepEnd = 0 + appended = rewritten } - for (const line of windowed.lines) { - lines.push(line) - } - let truncated = windowed.truncated - if (lines.length > MAX_TAIL_LINES) { - lines = lines.slice(lines.length - MAX_TAIL_LINES) - truncated = true - } - let totalChars = windowed.partialLine.length - for (const line of lines) { - totalChars += line.length - } - let dropCount = 0 - while (dropCount < lines.length && totalChars > MAX_TAIL_CHARS) { - totalChars -= lines[dropCount]!.length - dropCount += 1 - } - if (dropCount > 0) { - lines = lines.slice(dropCount) - truncated = true - } + const built = buildCarriedTailLines( + previousLines, + 0, + keepEnd, + appended, + windowed.partialLine.length + ) return { - lines, + lines: built.lines, partialLine: windowed.partialLine, redrawCursor: windowed.redrawCursor, - truncated, + truncated: windowed.truncated || built.truncated, newCompleteLines: windowed.newCompleteLines, newlyCompletedLines: windowed.newlyCompletedLines } diff --git a/src/main/runtime/terminal-tail-sentinel-index.test.ts b/src/main/runtime/terminal-tail-sentinel-index.test.ts new file mode 100644 index 00000000000..2b33b2770ea --- /dev/null +++ b/src/main/runtime/terminal-tail-sentinel-index.test.ts @@ -0,0 +1,428 @@ +import { describe, expect, it, vi } from 'vitest' +import { appendNormalizedToTailBuffer } from './terminal-tail-buffer' +import { MAX_TAIL_CHARS, MAX_TAIL_LINES } from './terminal-tail-limits' +import { buildPreview } from './terminal-tail-state' +import { + getTerminalTailSentinelFullScanCount, + getTerminalTailSentinelMatches, + tailMayContainBlockedSignal +} from './terminal-tail-sentinel-index' +import { computeTerminalTailWaitState } from './terminal-wait-tail-state' +import { TERMINAL_WAIT_BLOCKED_SENTINEL_RE } from './terminal-wait-detection' +import type { RetainedTailRedrawCursor } from './terminal-tail-redraw-buffer' + +// The definition the incremental index must reproduce: does ANY retained line (or the +// partial line) match the sentinel? Written out independently of the implementation. +function referenceMayContainBlockedSignal(lines: string[], partialLine: string): boolean { + for (const line of lines) { + if (TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(line)) { + return true + } + } + return TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(partialLine) +} + +function indexedMayContainBlockedSignal(lines: string[], partialLine: string): boolean { + return tailMayContainBlockedSignal(lines) || TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(partialLine) +} + +type TailSim = { + lines: string[] + partialLine: string + redrawCursor: RetainedTailRedrawCursor | null + preview: string +} + +function newSim(): TailSim { + return { lines: [], partialLine: '', redrawCursor: null, preview: '' } +} + +function feed(sim: TailSim, chunk: string): void { + const next = appendNormalizedToTailBuffer(sim.lines, sim.partialLine, chunk, sim.redrawCursor) + sim.lines = next.lines + sim.partialLine = next.partialLine + sim.redrawCursor = next.redrawCursor + sim.preview = buildPreview(next.lines, next.partialLine) +} + +/** A structurally identical tail the index has never seen, so it takes the full-scan path. */ +function unindexed(sim: TailSim): string[] { + return [...sim.lines] +} + +function assertMatchesFullScan(sim: TailSim): void { + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe( + referenceMayContainBlockedSignal(sim.lines, sim.partialLine) + ) + expect(computeTerminalTailWaitState(sim.lines, sim.partialLine, sim.preview)).toEqual( + computeTerminalTailWaitState(unindexed(sim), sim.partialLine, sim.preview) + ) +} + +const BLOCKED_LINE = 'Update available! Press Enter to continue.' +const ESC = String.fromCharCode(27) + +/** The exact positions a from-scratch scan would record, written out independently. */ +function referenceSentinelMatches(lines: readonly string[]): number[] { + const matches: number[] = [] + for (let index = 0; index < lines.length; index += 1) { + if (TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(lines[index]!)) { + matches.push(index) + } + } + return matches +} + +/** + * The whole contract of the carried window, at position resolution: the index the constructor + * registered for this exact array must equal a from-scratch scan of it. A boolean-only assertion + * would pass on an index whose positions are shifted, doubled, or out of bounds. + */ +function assertIndexedPositionsAreExact(lines: readonly string[]): void { + const indexed = [...getTerminalTailSentinelMatches(lines)] + expect(indexed).toEqual(referenceSentinelMatches(lines)) + for (const position of indexed) { + expect(position).toBeGreaterThanOrEqual(0) + expect(position).toBeLessThan(lines.length) + } +} + +function countSentinelTests(run: () => void): number { + const spy = vi.spyOn(TERMINAL_WAIT_BLOCKED_SENTINEL_RE, 'test') + try { + run() + return spy.mock.calls.length + } finally { + spy.mockRestore() + } +} + +function saturatedSim(): TailSim { + const sim = newSim() + for (let index = 0; index < MAX_TAIL_LINES + 400; index += 1) { + feed(sim, `streaming build output line ${index}\n`) + } + expect(sim.lines.length).toBe(MAX_TAIL_LINES) + return sim +} + +describe('terminal tail sentinel index', () => { + it('tests only the lines an append produced, not the whole retained tail', () => { + const sim = saturatedSim() + // Warm the index for the current tail identity. + computeTerminalTailWaitState(sim.lines, sim.partialLine, sim.preview) + + const incrementalTests = countSentinelTests(() => { + for (let index = 0; index < 20; index += 1) { + feed(sim, `fresh line ${index}\n`) + } + computeTerminalTailWaitState(sim.lines, sim.partialLine, sim.preview) + }) + + const fullScanTests = countSentinelTests(() => { + computeTerminalTailWaitState(unindexed(sim), sim.partialLine, sim.preview) + }) + + expect(fullScanTests).toBeGreaterThanOrEqual(MAX_TAIL_LINES) + // 20 appended lines + one partial-line test per compute call. + expect(incrementalTests).toBeLessThanOrEqual(25) + }) + + it('keeps a retained sentinel visible and drops it exactly when it is evicted', () => { + const sim = saturatedSim() + feed(sim, `${BLOCKED_LINE}\n`) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + assertMatchesFullScan(sim) + + // Push the prompt to the very last retained slot. + for (let index = 0; index < MAX_TAIL_LINES - 1; index += 1) { + feed(sim, `after prompt ${index}\n`) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + } + expect(sim.lines[0]).toBe(BLOCKED_LINE) + + // One more line evicts it. + feed(sim, 'evicting line\n') + expect(sim.lines.includes(BLOCKED_LINE)).toBe(false) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + assertMatchesFullScan(sim) + + // And it stays gone many chunks later. + for (let index = 0; index < 200; index += 1) { + feed(sim, `long after eviction ${index}\n`) + } + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + assertMatchesFullScan(sim) + }) + + it('drops a sentinel evicted by the retained-character cap', () => { + const sim = newSim() + feed(sim, `${BLOCKED_LINE}\n`) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + const bulkLine = `${'x'.repeat(4000)}\n` + for (let index = 0; index * 4001 < MAX_TAIL_CHARS + 20000; index += 1) { + feed(sim, bulkLine) + } + expect(sim.lines.includes(BLOCKED_LINE)).toBe(false) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + assertMatchesFullScan(sim) + }) + + it('finds a sentinel split across two chunks once the line completes', () => { + const sim = saturatedSim() + feed(sim, 'Codex asks: press ent') + // Still only a partial line, and no alternative matches the fragment yet. + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + assertMatchesFullScan(sim) + + feed(sim, 'er to confirm') + // Now complete, but still the partial line — the partial is always tested directly. + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + assertMatchesFullScan(sim) + + feed(sim, '\n') + // And once it becomes a retained line the index carries it. + expect(sim.partialLine).toBe('') + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + assertMatchesFullScan(sim) + }) + + it('full-scans a tail array the index has never seen (seed/restore path)', () => { + // primeWaitBlockedBaselineFromSeededTail reads whatever tail the restore seed installed. + const seeded = ['boot log', BLOCKED_LINE, 'trailing'] + expect(tailMayContainBlockedSignal(seeded)).toBe(true) + const state = computeTerminalTailWaitState(seeded, '', '') + expect(state.fromTail).toBe(true) + expect(state.signal?.reason).toBe('codex-update-prompt') + + const clean = ['boot log', 'no prompt here', 'trailing'] + expect(tailMayContainBlockedSignal(clean)).toBe(false) + expect(computeTerminalTailWaitState(clean, '', '').signal).toBeNull() + }) + + it('reports fromTail from a blank tail without consulting the index', () => { + const sim = newSim() + feed(sim, ' \n\t\n') + expect(computeTerminalTailWaitState(sim.lines, sim.partialLine, '').fromTail).toBe(false) + feed(sim, 'now visible\n') + expect(computeTerminalTailWaitState(sim.lines, sim.partialLine, '').fromTail).toBe(true) + }) +}) + +/** + * `buildCarriedTailLines` is the only producer of a tail array, and it derives the carried-match + * window from the same keep bounds it slices the array out of. These guards pin the four ways + * that derivation could still be written wrong, plus the one way a path could escape it. Each was + * confirmed to fail against a deliberately broken constructor (see the PR body). + */ +describe('terminal tail sentinel index carried window', () => { + it('drops a carried match the moment the constructor evicts its row (no stale match)', () => { + const sim = saturatedSim() + feed(sim, `${BLOCKED_LINE}\n`) + // Walk it to the very first retained slot, checking the position every step: each append + // evicts one row at a saturated tail, so the carried match must shift down by exactly one. + for (let index = 0; index < MAX_TAIL_LINES - 1; index += 1) { + feed(sim, `after prompt ${index}\n`) + expect(getTerminalTailSentinelMatches(sim.lines)).toEqual([MAX_TAIL_LINES - 2 - index]) + } + expect(sim.lines[0]).toBe(BLOCKED_LINE) + + feed(sim, 'evicting line\n') + expect(sim.lines.includes(BLOCKED_LINE)).toBe(false) + assertIndexedPositionsAreExact(sim.lines) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + }) + + it('finds a match a redraw writes into rows the carried prefix does not cover', () => { + const sim = saturatedSim() + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + + // A windowed redraw: the prefix carries, the rewritten suffix must still be scanned. + feed(sim, `${ESC}[3A${ESC}[2K${BLOCKED_LINE}\n`) + assertIndexedPositionsAreExact(sim.lines) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + + // And a redraw that overwrites that same row again must drop it. + feed(sim, `${ESC}[1A${ESC}[2Kplain replacement\n`) + assertIndexedPositionsAreExact(sim.lines) + + // A redraw deep enough to outrun the window carries nothing and rescans in full. + feed(sim, `${ESC}[2500A${ESC}[2K${BLOCKED_LINE}\n`) + assertIndexedPositionsAreExact(sim.lines) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + }) + + it('shifts every carried position by exactly the number of rows evicted', () => { + const sim = newSim() + feed(sim, `first\n${BLOCKED_LINE}\nsecond\n${BLOCKED_LINE}\nthird\n`) + expect(getTerminalTailSentinelMatches(sim.lines)).toEqual([1, 3]) + + // Saturate so the line cap evicts exactly one row per single-line append. + for (let index = 0; index < MAX_TAIL_LINES - 5; index += 1) { + feed(sim, `pad ${index}\n`) + } + expect(sim.lines.length).toBe(MAX_TAIL_LINES) + expect(getTerminalTailSentinelMatches(sim.lines)).toEqual([1, 3]) + + feed(sim, 'evict one\n') + expect(getTerminalTailSentinelMatches(sim.lines)).toEqual([0, 2]) + feed(sim, 'evict two\n') + expect(getTerminalTailSentinelMatches(sim.lines)).toEqual([1]) + assertIndexedPositionsAreExact(sim.lines) + }) + + it('stays in bounds when a single chunk evicts the whole carried window and part of itself', () => { + const sim = saturatedSim() + feed(sim, `${BLOCKED_LINE}\n`) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(true) + + // One chunk with more complete lines than the tail retains: every carried row goes, and so + // does the front of the chunk itself, so nothing may survive from before the cut. + const early: string[] = [BLOCKED_LINE] + for (let index = 0; index < MAX_TAIL_LINES + 500; index += 1) { + early.push(`flood ${index}`) + } + feed(sim, `${early.join('\n')}\n`) + expect(sim.lines.length).toBe(MAX_TAIL_LINES) + assertIndexedPositionsAreExact(sim.lines) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + + // Same shape, but the prompt lands inside the surviving suffix of the chunk. + const late: string[] = [] + for (let index = 0; index < MAX_TAIL_LINES + 500; index += 1) { + late.push(`flood ${index}`) + } + late.push(BLOCKED_LINE) + feed(sim, `${late.join('\n')}\n`) + expect(getTerminalTailSentinelMatches(sim.lines)).toEqual([MAX_TAIL_LINES - 1]) + assertIndexedPositionsAreExact(sim.lines) + + // The character cap drops from the same front, past the carried window and into the chunk. + const bulk = `${'x'.repeat(4000)}\n` + for (let index = 0; index * 4001 < MAX_TAIL_CHARS + 20000; index += 1) { + feed(sim, bulk) + } + assertIndexedPositionsAreExact(sim.lines) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(false) + }) + + it('never leaves a produced tail unindexed, on any append path', () => { + const sim = saturatedSim() + computeTerminalTailWaitState(sim.lines, sim.partialLine, sim.preview) + const fullScansBefore = getTerminalTailSentinelFullScanCount() + + feed(sim, `${BLOCKED_LINE}\n`) + for (let index = 0; index < 200; index += 1) { + feed(sim, `after prompt ${index}\n`) + } + feed(sim, 'partial with no newline') + feed(sim, ' and its completion\n') + feed(sim, '\rspinner 40%') + feed(sim, `${ESC}[3A${ESC}[2Kredrawn\n`) + feed(sim, `${ESC}[2500A${ESC}[2Kdeep redraw\n`) + feed(sim, 'trailing spaces here \n') + feed(sim, `${'z'.repeat(5000)}\n`) + feed(sim, `multi\nline\nchunk\n`) + feed(sim, '') + // Reading the verdict must never trigger a scan of an array the constructor produced. + computeTerminalTailWaitState(sim.lines, sim.partialLine, sim.preview) + + expect(getTerminalTailSentinelFullScanCount()).toBe(fullScansBefore) + assertIndexedPositionsAreExact(sim.lines) + }) +}) + +// Deterministic PRNG so a divergence is reproducible from the seed alone. +function mulberry32(seed: number): () => number { + let state = seed >>> 0 + return () => { + state = (state + 0x6d2b79f5) >>> 0 + let t = Math.imul(state ^ (state >>> 15), 1 | state) + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } +} + +/** + * `streaming` saturates and evicts the retained tail; `tui` trades saturation for redraw + * coverage (cursor-up rewrites of retained rows, and reaches past the redraw window). + */ +function randomChunk(random: () => number, profile: 'streaming' | 'tui'): string { + const roll = random() + if (roll < 0.2) { + const lines: string[] = [] + for (let index = 0; index < 30; index += 1) { + lines.push(`burst line ${Math.floor(random() * 1e6)}`) + } + return `${lines.join('\n')}\n` + } + if (roll < 0.42) { + return `plain output ${Math.floor(random() * 1e6)}\n` + } + if (roll < 0.48) { + return `${' '.repeat(Math.floor(random() * 3))}\n` + } + if (roll < 0.54) { + return `${BLOCKED_LINE}\n` + } + if (roll < 0.58) { + return 'do you trust the files in this folder?\n' + } + if (roll < 0.63) { + // Sentinel split across a chunk boundary. + return random() < 0.5 ? 'Codex asks: press ent' : 'er to confirm\n' + } + if (roll < 0.7) { + // TUI redraw: move the cursor up a few rows and rewrite them. + const rows = 1 + Math.floor(random() * 12) + return `${ESC}[${rows}A${ESC}[2Kredrawn row ${Math.floor(random() * 1000)}\n` + } + if (roll < (profile === 'tui' ? 0.76 : 0.7)) { + // Deep redraw that outruns the window and forces the unwindowed path. + return `${ESC}[${1500 + Math.floor(random() * 800)}A${ESC}[2Kdeep redraw\n` + } + if (roll < 0.82) { + return `\rspinner ${Math.floor(random() * 100)}%` + } + if (roll < 0.87) { + return 'trailing spaces here \n' + } + if (roll < 0.91) { + return `${'y'.repeat(3000)}\n` + } + if (roll < 0.95) { + return `multi\nline\nchunk ${Math.floor(random() * 1000)}\n` + } + if (roll < 0.97) { + return '' + } + return `no newline ${Math.floor(random() * 1000)}` +} + +describe('terminal tail sentinel index property', () => { + for (const profile of ['streaming', 'tui'] as const) { + for (const seed of [1, 7, 42, 1337]) { + it(`matches a full scan on every step of a random ${profile} sequence (seed ${seed})`, () => { + const random = mulberry32(seed) + const sim = newSim() + let sawSentinel = false + let sawSaturation = false + for (let step = 0; step < 1200; step += 1) { + feed(sim, randomChunk(random, profile)) + const expected = referenceMayContainBlockedSignal(sim.lines, sim.partialLine) + expect(indexedMayContainBlockedSignal(sim.lines, sim.partialLine)).toBe(expected) + expect(computeTerminalTailWaitState(sim.lines, sim.partialLine, sim.preview)).toEqual( + computeTerminalTailWaitState(unindexed(sim), sim.partialLine, sim.preview) + ) + sawSentinel = sawSentinel || expected + sawSaturation = sawSaturation || sim.lines.length >= MAX_TAIL_LINES + } + // Guard against a vacuous pass. + expect(sawSentinel).toBe(true) + if (profile === 'streaming') { + expect(sawSaturation).toBe(true) + } + }) + } + } +}) diff --git a/src/main/runtime/terminal-tail-sentinel-index.ts b/src/main/runtime/terminal-tail-sentinel-index.ts new file mode 100644 index 00000000000..c99fd31bac7 --- /dev/null +++ b/src/main/runtime/terminal-tail-sentinel-index.ts @@ -0,0 +1,94 @@ +import { TERMINAL_WAIT_BLOCKED_SENTINEL_RE } from './terminal-wait-detection' + +/** + * Which retained tail lines match the wait-blocked sentinel, memoized per + * lines-array identity. + * + * Why: `computeTerminalTailWaitState` must prove the ABSENCE of a signal, so it + * cannot early-exit and re-tested all 2000 retained lines on every scan (20/s + * per streaming PTY) even though only ~20 lines were new. Keyed weakly by the + * array so an entry dies with the tail it describes; the tail array is replaced + * on every append and never mutated in place, so at most one entry per PTY + * stays live. + */ +const sentinelMatchesByTailLines = new WeakMap() + +function collectSentinelMatches( + lines: readonly string[], + startIndex: number, + into: number[] +): void { + for (let index = startIndex; index < lines.length; index += 1) { + if (TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(lines[index]!)) { + into.push(index) + } + } +} + +/** + * How many arrays have been full-scanned because they arrived without an index entry. + * Every array `appendNormalizedToTailBuffer` produces is registered by `buildCarriedTailLines`, + * so this only advances for tails the index has genuinely never seen (a restore seed, a persisted + * record, a hand-built array). Tests assert it stays flat across the real append paths, which is + * what proves no producer path silently bypasses the constructor. + */ +let sentinelFullScanCount = 0 + +export function getTerminalTailSentinelFullScanCount(): number { + return sentinelFullScanCount +} + +/** Ascending indices of sentinel-matching lines; full-scans an unseen array. */ +export function getTerminalTailSentinelMatches(lines: readonly string[]): readonly number[] { + const cached = sentinelMatchesByTailLines.get(lines) + if (cached) { + return cached + } + sentinelFullScanCount += 1 + const matches: number[] = [] + collectSentinelMatches(lines, 0, matches) + sentinelMatchesByTailLines.set(lines, matches) + return matches +} + +export function tailMayContainBlockedSignal(lines: readonly string[]): boolean { + return getTerminalTailSentinelMatches(lines).length > 0 +} + +/** + * Derive `nextLines`' match index from `previousLines`', testing only the lines + * the append actually produced. + * + * `nextLines[0 … carriedCount)` are the very same strings as + * `previousLines[carriedSourceStart … carriedSourceStart + carriedCount)`, and + * every later line is newly produced. That is not an assumption a caller has to + * uphold by hand: `buildCarriedTailLines` in `terminal-tail-buffer.ts` is the + * sole caller, and it derives this window from the same keep bounds it slices + * `nextLines` out of, so the window and the array cannot disagree. Matches + * outside the carried window are dropped because their lines were evicted or + * rewritten, which is exactly what a full scan would conclude. + */ +export function carryTerminalTailSentinelMatches( + previousLines: readonly string[], + nextLines: readonly string[], + carriedSourceStart: number, + carriedCount: number +): void { + if (nextLines === previousLines) { + return + } + const matches: number[] = [] + if (carriedCount > 0) { + const carriedEnd = carriedSourceStart + carriedCount + for (const index of getTerminalTailSentinelMatches(previousLines)) { + if (index >= carriedEnd) { + break + } + if (index >= carriedSourceStart) { + matches.push(index - carriedSourceStart) + } + } + } + collectSentinelMatches(nextLines, carriedCount, matches) + sentinelMatchesByTailLines.set(nextLines, matches) +} diff --git a/src/main/runtime/terminal-wait-tail-state.ts b/src/main/runtime/terminal-wait-tail-state.ts index 2b9cafee27b..712b301a961 100644 --- a/src/main/runtime/terminal-wait-tail-state.ts +++ b/src/main/runtime/terminal-wait-tail-state.ts @@ -1,5 +1,6 @@ import type { RuntimeTerminalWaitBlockedReason } from '../../shared/runtime-types' import { buildTailLines } from './terminal-tail-state' +import { tailMayContainBlockedSignal } from './terminal-tail-sentinel-index' import { findActionableTerminalWaitBlockedSignal, TERMINAL_WAIT_BLOCKED_SENTINEL_RE @@ -60,23 +61,22 @@ function inspectTerminalWaitTail( lines: string[], partialLine: string ): { fromTail: boolean; mayContainBlockedSignal: boolean } { - let fromTail = false - let mayContainBlockedSignal = false + return { + fromTail: hasVisibleTailLine(lines) || partialLine.trim().length > 0, + // Why the index: proving a signal is ABSENT can't early-exit, so a full re-test of the + // 2000-line tail ran per scan; the index tests only the lines each append produced. + mayContainBlockedSignal: + tailMayContainBlockedSignal(lines) || TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(partialLine) + } +} + +function hasVisibleTailLine(lines: string[]): boolean { for (const line of lines) { - if (!fromTail && line.trim().length > 0) { - fromTail = true - } - if (!mayContainBlockedSignal && TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(line)) { - mayContainBlockedSignal = true + if (line.trim().length > 0) { + return true } } - if (!fromTail && partialLine.trim().length > 0) { - fromTail = true - } - if (!mayContainBlockedSignal && TERMINAL_WAIT_BLOCKED_SENTINEL_RE.test(partialLine)) { - mayContainBlockedSignal = true - } - return { fromTail, mayContainBlockedSignal } + return false } // Why: consumes precomputed wait states so full-tail scans aren't repeated per chunk (replaces the former inline double full-tail scan). From 63aee7f1ee5b006330e755bd2e23d7faa89b4988 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:37:44 -0700 Subject: [PATCH 224/398] perf(terminals): spend one inspection start on a whole cadence round (#18438) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The inspection rate limiter counted panes when it should have counted host observations. `MAX_INSPECTION_STARTS_PER_SECOND = 8` is global, and it was spent one pane at a time, so N due panes meant an effective per-pane period of max(tier, N/8 seconds) — ~37.5s at 300 panes for a pane the code polls at 750ms. Agent-completion latency degraded monotonically as panes were added. Every local pane's inspection resolves out of the same TTL-and-in-flight- deduped process-table capture, so a whole round of them is one host observation. The queue now drains all shared-observation tasks as one round on one start, launched in a single tick. Remote panes each cost their own execution-host round trip and stay admitted one at a time. Both the budget and the cadence tiers are numerically unchanged. Disposed tasks are also compacted out in one pass instead of a splice per drop, so the per-round predicate cost is linear rather than quadratic at pane scale. No IPC, preload, wire, or main-process change: each pane keeps its existing per-pane `pty:inspectProcess` invoke. --- .../agent-completion-process-monitor.ts | 3 + .../agent-process-inspection-queue.ts | 142 +++++++++++++---- .../agent-process-inspection-round.test.ts | 149 ++++++++++++++++++ 3 files changed, 261 insertions(+), 33 deletions(-) create mode 100644 src/renderer/src/components/terminal-pane/agent-process-inspection-round.test.ts diff --git a/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts b/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts index 78c859e4b4b..f34ceffa97e 100644 --- a/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts +++ b/src/renderer/src/components/terminal-pane/agent-completion-process-monitor.ts @@ -101,6 +101,9 @@ export function createAgentCompletionProcessMonitor({ enqueueAgentProcessInspection({ priority, canRun: () => !state.disposed, + // Local reads all resolve out of one process-table capture; remote ones each cost their + // own execution-host round trip and stay admitted one at a time. + sharesHostObservation: options.isRemotePtyId?.(ptyId) !== true, run: async () => { let inspectedRecognizedAgent = false let inspectionSucceeded = false diff --git a/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts b/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts index 6dc1dc4ccf2..8e5aef6cade 100644 --- a/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts +++ b/src/renderer/src/components/terminal-pane/agent-process-inspection-queue.ts @@ -4,29 +4,56 @@ type InspectionTask = { priority: InspectionPriority canRun: () => boolean run: () => Promise + /** + * Reads served by one shared host observation. Every local pane's inspection resolves out of + * the same TTL-and-in-flight-deduped process-table snapshot, so a whole round of them costs + * the host one capture however many panes ride it. + */ + sharesHostObservation?: boolean } const MAX_CONCURRENT_INSPECTIONS = 4 const MAX_INSPECTION_STARTS_PER_SECOND = 8 let activeInspections = 0 +let inspectionPumpQueued = false let inspectionPumpTimer: ReturnType | null = null const inspectionStarts: number[] = [] const inspectionQueue: InspectionTask[] = [] -function canStartInspection(now: number): boolean { +/** + * Host observations still admissible right now. A start is one observation, not one pane: a + * shared-observation round costs one however many panes ride it, an unshared task costs one each. + */ +function availableInspectionStarts(now: number): number { if (inspectionStarts.length > 0 && now < inspectionStarts[0]!) { inspectionStarts.length = 0 } while (inspectionStarts.length > 0 && now - inspectionStarts[0]! >= 1_000) { inspectionStarts.shift() } - return ( - activeInspections < MAX_CONCURRENT_INSPECTIONS && - inspectionStarts.length < MAX_INSPECTION_STARTS_PER_SECOND + return Math.min( + MAX_CONCURRENT_INSPECTIONS - activeInspections, + MAX_INSPECTION_STARTS_PER_SECOND - inspectionStarts.length ) } +/** + * Pump on a microtask, so a synchronous burst of enqueues forms one round. Pumping inline + * spent a start per pane until the concurrency slots filled and then parked the rest of the + * burst on the 100ms retry. + */ +function queueInspectionPump(): void { + if (inspectionPumpQueued) { + return + } + inspectionPumpQueued = true + queueMicrotask(() => { + inspectionPumpQueued = false + pumpInspectionQueue() + }) +} + function scheduleInspectionPump(delayMs = 0): void { if (inspectionPumpTimer !== null) { return @@ -37,44 +64,92 @@ function scheduleInspectionPump(delayMs = 0): void { }, delayMs) } -function pumpInspectionQueue(): void { - // Drop disposed tasks before slot/rate accounting. - for (let index = inspectionQueue.length - 1; index >= 0; index -= 1) { - const task = inspectionQueue[index] - if (task && !task.canRun()) { - inspectionQueue.splice(index, 1) +/** Compact disposed tasks out in one pass; a splice per drop is quadratic at pane scale. */ +function dropDisposedInspections(): void { + let write = 0 + for (let read = 0; read < inspectionQueue.length; read += 1) { + const task = inspectionQueue[read]! + if (task.canRun()) { + inspectionQueue[write] = task + write += 1 } } + inspectionQueue.length = write +} + +function startInspectionRound(tasks: InspectionTask[], now: number): void { + activeInspections += 1 + inspectionStarts.push(now) + let outstanding = tasks.length + const settleOne = (): void => { + outstanding -= 1 + if (outstanding > 0) { + return + } + activeInspections = Math.max(0, activeInspections - 1) + if (inspectionQueue.length > 0) { + scheduleInspectionPump() + } + } + for (const task of tasks) { + // Started synchronously so every read in the round lands in the same tick, hitting one + // process-table capture instead of serializing one capture window apart. + // Why the catch before finally: an unreachable runtime rejects the inspection on a cadence, and a + // bare `.finally()` chain re-raises it as a renderer-global unhandledrejection. Coordinators own + // their own failure/backoff state, so the queue only has to keep its accounting running. + void task + .run() + .catch(() => {}) + .finally(settleOne) + } +} + +/** Take every shared-observation task, in order, leaving the rest queued. */ +function takeSharedObservationRound(): InspectionTask[] { + const round: InspectionTask[] = [] + let write = 0 + for (let read = 0; read < inspectionQueue.length; read += 1) { + const task = inspectionQueue[read]! + if (task.sharesHostObservation === true) { + round.push(task) + } else { + inspectionQueue[write] = task + write += 1 + } + } + inspectionQueue.length = write + return round +} + +function pumpInspectionQueue(): void { + // Drop disposed tasks before slot/rate accounting. + dropDisposedInspections() if (inspectionQueue.length === 0) { return } const now = Date.now() - if (!canStartInspection(now)) { + let starts = availableInspectionStarts(now) + if (starts <= 0) { scheduleInspectionPump(100) return } - - const priorityIndex = inspectionQueue.findIndex((task) => task.priority === 'pending-title') - const next = - priorityIndex !== -1 ? inspectionQueue.splice(priorityIndex, 1)[0] : inspectionQueue.shift() - if (!next) { - return + // The whole shared-observation backlog goes on one start, so a pane's wait is bounded by the + // observation budget rather than by how many other panes are also due. + const sharedRound = takeSharedObservationRound() + if (sharedRound.length > 0) { + startInspectionRound(sharedRound, now) + starts -= 1 + } + while (starts > 0 && inspectionQueue.length > 0) { + const priorityIndex = inspectionQueue.findIndex((task) => task.priority === 'pending-title') + const next = + priorityIndex !== -1 ? inspectionQueue.splice(priorityIndex, 1)[0] : inspectionQueue.shift() + if (!next) { + break + } + startInspectionRound([next], now) + starts -= 1 } - - activeInspections += 1 - inspectionStarts.push(now) - // Why the catch before finally: an unreachable runtime rejects the inspection on a cadence, and a - // bare `.finally()` chain re-raises it as a renderer-global unhandledrejection. Coordinators own - // their own failure/backoff state, so the queue only has to keep its accounting running. - void next - .run() - .catch(() => {}) - .finally(() => { - activeInspections = Math.max(0, activeInspections - 1) - if (inspectionQueue.length > 0) { - scheduleInspectionPump() - } - }) if (inspectionQueue.length > 0) { scheduleInspectionPump() @@ -83,7 +158,7 @@ function pumpInspectionQueue(): void { export function enqueueAgentProcessInspection(task: InspectionTask): void { inspectionQueue.push(task) - pumpInspectionQueue() + queueInspectionPump() } export function resetAgentProcessInspectionQueueForTests(): void { @@ -91,6 +166,7 @@ export function resetAgentProcessInspectionQueueForTests(): void { clearTimeout(inspectionPumpTimer) inspectionPumpTimer = null } + inspectionPumpQueued = false activeInspections = 0 inspectionStarts.length = 0 inspectionQueue.length = 0 diff --git a/src/renderer/src/components/terminal-pane/agent-process-inspection-round.test.ts b/src/renderer/src/components/terminal-pane/agent-process-inspection-round.test.ts new file mode 100644 index 00000000000..216bf304a1c --- /dev/null +++ b/src/renderer/src/components/terminal-pane/agent-process-inspection-round.test.ts @@ -0,0 +1,149 @@ +// Regression guard for the inspection admission budget. The cadence tiers +// (active 750ms / idle 2000 / hidden 3000 / no-evidence 15000) were a per-pane +// promise the queue could not keep: the budget of 8 starts per second was spent +// one pane at a time, so N due panes meant roughly N/8 seconds between +// inspections for each of them and agent-completion latency degraded as the +// user added panes. Local inspections all resolve out of one TTL-and-in-flight- +// deduped process-table capture, so a whole round of them is one host +// observation and rides one start, launched in a single tick. The budget itself +// is unchanged — it just buys the whole round instead of one pane. +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + enqueueAgentProcessInspection, + resetAgentProcessInspectionQueueForTests +} from './agent-process-inspection-queue' + +const PANES = 300 + +beforeEach(() => { + vi.useFakeTimers() +}) + +afterEach(() => { + vi.useRealTimers() + resetAgentProcessInspectionQueueForTests() +}) + +describe('agent process inspection rounds', () => { + it('inspects every pane of a 300-pane round inside the same admission budget', async () => { + const inspected = new Set() + + for (let index = 0; index < PANES; index += 1) { + const ptyId = `pty-${index}` + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + sharesHostObservation: true, + run: async () => { + await Promise.resolve() + inspected.add(ptyId) + } + }) + } + // Well inside one 1s rate-limiter window: pre-fix only the 8 starts that window + // allows are spent, so only 8 of the 300 panes are ever inspected. + await vi.advanceTimersByTimeAsync(200) + + expect(inspected.size).toBe(PANES) + }) + + it('launches the whole round in one tick on one start', async () => { + const launchesPerTick = new Map() + let unshared = 0 + + for (let index = 0; index < PANES; index += 1) { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + sharesHostObservation: true, + run: async () => { + // Fake timers freeze the clock inside a tick, so a shared timestamp is a shared burst. + const tick = Date.now() + launchesPerTick.set(tick, (launchesPerTick.get(tick) ?? 0) + 1) + } + }) + } + // Seven unshared panes still fit, which is what proves the round cost exactly one of + // the eight starts rather than one per pane until the concurrency slots filled. + for (let index = 0; index < 7; index += 1) { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + sharesHostObservation: false, + run: async () => { + unshared += 1 + } + }) + } + await vi.advanceTimersByTimeAsync(200) + + // One synchronous burst carries every pane, so they hit one process-table capture + // rather than serializing across the limiter. + expect([...launchesPerTick.values()]).toEqual([PANES]) + expect(unshared).toBe(7) + }) + + it('keeps a pane whose read is not shared admitted one round trip at a time', async () => { + const started: string[] = [] + + for (let index = 0; index < PANES; index += 1) { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + // Remote panes: each costs its own execution-host round trip, so no round shares them. + sharesHostObservation: false, + run: async () => { + started.push(`ssh-${index}`) + } + }) + } + await vi.advanceTimersByTimeAsync(200) + + expect(started.length).toBeLessThanOrEqual(8) + }) + + it('still serves a pending-title read ahead of the queued cadence backlog', async () => { + const order: string[] = [] + + for (let index = 0; index < 20; index += 1) { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => true, + sharesHostObservation: false, + run: async () => { + order.push(`cadence-${index}`) + } + }) + } + enqueueAgentProcessInspection({ + priority: 'pending-title', + canRun: () => true, + sharesHostObservation: false, + run: async () => { + order.push('pending-title') + } + }) + await vi.advanceTimersByTimeAsync(200) + + expect(order.length).toBeLessThanOrEqual(8) + expect(order).toContain('pending-title') + }) + + it('drops disposed panes out of the round instead of inspecting them', async () => { + const inspected: number[] = [] + + for (let index = 0; index < PANES; index += 1) { + enqueueAgentProcessInspection({ + priority: 'cadence', + canRun: () => index % 2 === 0, + sharesHostObservation: true, + run: async () => { + inspected.push(index) + } + }) + } + await vi.advanceTimersByTimeAsync(200) + + expect(inspected).toEqual(Array.from({ length: PANES / 2 }, (_unused, index) => index * 2)) + }) +}) From 7ed86a98ae97d2e64728cd307f2d546374d52e97 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:41:17 -0700 Subject: [PATCH 225/398] perf(ipc): index worktree owners instead of rescanning the repo list per lookup (#18416) Two hot lookups rescanned a whole table once per repo. `getLocalRepoForRegisteredWorktree` (59 IPC call sites, including Quick Open keystrokes and every File Explorer expand) walked the entire worktree-meta table once per repo. One pass now collects the owning repo ids, built lazily so a repo whose own path matches still never touches the table. `createRepoRowExecutionHostLookup` re-filtered the repo array on every `byId` / `byHost` call. Rows are grouped into a Map once at construction, preserving repo-list order so `rows[0]` still picks the same owner. --- .../local-worktree-runtime-options.test.ts | 142 ++++++++++++++++++ .../ipc/local-worktree-runtime-options.ts | 20 ++- ...worktree-execution-host-resolution.test.ts | 67 ++++++++- .../worktree-execution-host-resolution.ts | 29 +++- 4 files changed, 244 insertions(+), 14 deletions(-) create mode 100644 src/main/ipc/local-worktree-runtime-options.test.ts diff --git a/src/main/ipc/local-worktree-runtime-options.test.ts b/src/main/ipc/local-worktree-runtime-options.test.ts new file mode 100644 index 00000000000..f7a1b0430e1 --- /dev/null +++ b/src/main/ipc/local-worktree-runtime-options.test.ts @@ -0,0 +1,142 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { WORKTREE_ID_SEPARATOR, type ParsedWorktreeId } from '../../shared/worktree/id' +import type * as WorktreeIdModule from '../../shared/worktree/id' + +const counter = vi.hoisted(() => ({ splitCalls: 0 })) + +// Why: `splitWorktreeId` (and the `Object.keys` snapshot around it) is the per-row work the repo +// loop used to repeat once per repo. Counting it makes the O(repos x rows) regression observable. +vi.mock('../../shared/worktree/id', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + splitWorktreeId: (worktreeId: string): ParsedWorktreeId | null => { + counter.splitCalls += 1 + return actual.splitWorktreeId(worktreeId) + } + } +}) + +const { getLocalRepoForRegisteredWorktree } = await import('./local-worktree-runtime-options') + +type TestRepo = { id: string; path: string; connectionId?: string } + +const makeStore = ( + repos: readonly TestRepo[], + worktreeIds: readonly string[] +): { store: never; metaScans: () => number } => { + let metaScans = 0 + const meta = Object.fromEntries(worktreeIds.map((id) => [id, {}])) + const store = { + getRepos: () => repos, + getAllWorktreeMeta: () => { + metaScans += 1 + return meta + } + } + return { store: store as never, metaScans: () => metaScans } +} + +const worktreeId = (repoId: string, path: string): string => + `${repoId}${WORKTREE_ID_SEPARATOR}${path}` + +beforeEach(() => { + counter.splitCalls = 0 +}) + +describe('getLocalRepoForRegisteredWorktree', () => { + it('walks the worktree meta table once, not once per repo', () => { + // Worst case: the owning repo is last, so every earlier repo used to force a full rescan. + const repoCount = 10 + const rowCount = 200 + const repos = Array.from({ length: repoCount }, (_, i) => ({ + id: `repo-${i}`, + path: `/repos/repo-${i}` + })) + const target = '/repos/repo-9/wt-last' + const worktreeIds = Array.from({ length: rowCount }, (_, i) => + worktreeId(`repo-${i % repoCount}`, `/repos/wt-${i}`) + ) + worktreeIds[rowCount - 1] = worktreeId(`repo-${repoCount - 1}`, target) + const { store, metaScans } = makeStore(repos, worktreeIds) + + expect(getLocalRepoForRegisteredWorktree(store, target, target)?.id).toBe('repo-9') + expect(metaScans()).toBe(1) + expect(counter.splitCalls).toBe(rowCount) + }) + + it('never touches the meta table when a repo path matches directly', () => { + const { store, metaScans } = makeStore([{ id: 'repo-a', path: '/repos/a' }], []) + expect(getLocalRepoForRegisteredWorktree(store, '/repos/a', '/repos/a')?.id).toBe('repo-a') + expect(metaScans()).toBe(0) + }) + + describe('equivalence with the per-repo scan', () => { + const repos: TestRepo[] = [ + { id: 'first', path: '/repos/first' }, + { id: 'middle', path: '/repos/middle' }, + { id: 'last', path: '/repos/last' } + ] + + it('finds a worktree owned by the first repo', () => { + const { store } = makeStore(repos, [worktreeId('first', '/wt/one')]) + expect(getLocalRepoForRegisteredWorktree(store, '/wt/one', '/wt/one')?.id).toBe('first') + }) + + it('finds a worktree owned by the last repo', () => { + const { store } = makeStore(repos, [worktreeId('last', '/wt/one')]) + expect(getLocalRepoForRegisteredWorktree(store, '/wt/one', '/wt/one')?.id).toBe('last') + }) + + it('returns undefined when no repo owns the worktree', () => { + const { store } = makeStore(repos, [worktreeId('other', '/wt/elsewhere')]) + expect(getLocalRepoForRegisteredWorktree(store, '/wt/one', '/wt/one')).toBeUndefined() + }) + + it('keeps getRepos precedence when two repos both own the path', () => { + const { store } = makeStore(repos, [ + worktreeId('last', '/wt/shared'), + worktreeId('middle', '/wt/shared') + ]) + // getRepos order decides, not the meta table's insertion order. + expect(getLocalRepoForRegisteredWorktree(store, '/wt/shared', '/wt/shared')?.id).toBe( + 'middle' + ) + }) + + it('excludes an SSH repo even when it owns the registered worktree', () => { + const { store } = makeStore( + [{ id: 'remote', path: '/repos/remote', connectionId: 'm4air' }, ...repos], + [worktreeId('remote', '/wt/one'), worktreeId('middle', '/wt/one')] + ) + expect(getLocalRepoForRegisteredWorktree(store, '/wt/one', '/wt/one')?.id).toBe('middle') + + const onlyRemote = makeStore( + [{ id: 'remote', path: '/repos/remote', connectionId: 'm4air' }], + [worktreeId('remote', '/wt/one')] + ) + expect( + getLocalRepoForRegisteredWorktree(onlyRemote.store, '/wt/one', '/wt/one') + ).toBeUndefined() + }) + + it('matches the resolved path spelling as well as the raw one', () => { + const { store } = makeStore(repos, [worktreeId('middle', '/wt/one')]) + expect(getLocalRepoForRegisteredWorktree(store, '/wt/other', '/wt/one')?.id).toBe('middle') + }) + + it('returns undefined for a folder workspace that is not a registered worktree', () => { + const { store } = makeStore(repos, [worktreeId('middle', '/wt/one')]) + expect( + getLocalRepoForRegisteredWorktree(store, '/folders/notes', '/folders/notes') + ).toBeUndefined() + }) + + it('tolerates a store without getRepos or getAllWorktreeMeta', () => { + expect(getLocalRepoForRegisteredWorktree({} as never, '/wt/one', '/wt/one')).toBeUndefined() + expect( + getLocalRepoForRegisteredWorktree({ getRepos: () => repos } as never, '/wt/one', '/wt/one') + ).toBeUndefined() + }) + }) +}) diff --git a/src/main/ipc/local-worktree-runtime-options.ts b/src/main/ipc/local-worktree-runtime-options.ts index d0d1a67e3cd..07bb9b41c83 100644 --- a/src/main/ipc/local-worktree-runtime-options.ts +++ b/src/main/ipc/local-worktree-runtime-options.ts @@ -19,20 +19,21 @@ function getCandidateLocalWorktreePaths( return new Set([worktreePath, resolvedWorktreePath].map(comparableLocalPath)) } -function hasRegisteredWorktreeMetaForRepo( +/** Repos owning a registered worktree at one of `candidatePaths`, in one pass over the meta table. */ +function collectRepoIdsWithRegisteredWorktreeMeta( store: Store, - repoId: string, candidatePaths: Set -): boolean { +): Set { const worktreeMeta = typeof store.getAllWorktreeMeta === 'function' ? store.getAllWorktreeMeta() : {} + const repoIds = new Set() for (const worktreeId of Object.keys(worktreeMeta)) { const parsed = splitWorktreeId(worktreeId) - if (parsed?.repoId === repoId && candidatePaths.has(comparableLocalPath(parsed.worktreePath))) { - return true + if (parsed && candidatePaths.has(comparableLocalPath(parsed.worktreePath))) { + repoIds.add(parsed.repoId) } } - return false + return repoIds } export function getLocalRepoForRegisteredWorktree( @@ -45,13 +46,18 @@ export function getLocalRepoForRegisteredWorktree( } const candidatePaths = getCandidateLocalWorktreePaths(worktreePath, resolvedWorktreePath) + // Built at most once, and only when a repo actually needs it, so the meta table is never + // rescanned per repo — 59 IPC call sites hit this, some per keystroke. + let repoIdsWithMeta: Set | undefined return store .getRepos() .find( (repo) => !repo.connectionId && (candidatePaths.has(comparableLocalPath(repo.path)) || - hasRegisteredWorktreeMetaForRepo(store, repo.id, candidatePaths)) + (repoIdsWithMeta ??= collectRepoIdsWithRegisteredWorktreeMeta(store, candidatePaths)).has( + repo.id + )) ) } diff --git a/src/shared/worktree-execution-host-resolution.test.ts b/src/shared/worktree-execution-host-resolution.test.ts index 70ee6d304b6..b82cea476eb 100644 --- a/src/shared/worktree-execution-host-resolution.test.ts +++ b/src/shared/worktree-execution-host-resolution.test.ts @@ -1,7 +1,8 @@ import { describe, expect, it } from 'vitest' import { createRepoRowExecutionHostLookup, - resolveWorktreeExecutionHost + resolveWorktreeExecutionHost, + type ExecutionHostOwnerRow } from './worktree-execution-host-resolution' // Why (#11163, #17799): main's terminal launch scope and the renderer's owner index both answer @@ -165,3 +166,67 @@ describe('resolveWorktreeExecutionHost', () => { }) }) }) + +describe('createRepoRowExecutionHostLookup', () => { + /** Rows whose `id` reads are counted, so a rescan of the repo list is observable. */ + const countingRepos = ( + rows: readonly ExecutionHostOwnerRow[] + ): { repos: ExecutionHostOwnerRow[]; idReads: () => number } => { + let idReads = 0 + const repos = rows.map(({ id, ...rest }) => ({ + ...rest, + get id(): string { + idReads += 1 + return id + } + })) + return { repos, idReads: () => idReads } + } + + it('scans the repo list once for the factory, never again per lookup', () => { + const { repos, idReads } = countingRepos([ + { id: 'a' }, + { id: 'b', connectionId: 'm4air' }, + { id: 'c' } + ]) + const lookup = createRepoRowExecutionHostLookup(repos) + // One grouping pass over the list — a Map get plus a set per row — and then never again. + const afterBuild = idReads() + expect(afterBuild).toBeLessThanOrEqual(repos.length * 2) + + for (let i = 0; i < 50; i++) { + lookup.byId('a') + lookup.byId('missing') + lookup.byHost('b', 'ssh:m4air') + } + expect(idReads()).toBe(afterBuild) + }) + + it('answers missing, ambiguous and resolved exactly as a per-call scan would', () => { + expect(createRepoRowExecutionHostLookup([]).byId('r')).toEqual({ kind: 'missing' }) + + const openclaw = { id: 'r', connectionId: 'openclaw' } + const m4air = { id: 'r', connectionId: 'm4air' } + expect(createRepoRowExecutionHostLookup([openclaw, m4air]).byId('r')).toEqual({ + kind: 'ambiguous' + }) + + // Two rows agreeing on one host still resolve to the first in repo-list order. + const first: ExecutionHostOwnerRow = { id: 'r', connectionId: 'm4air' } + const second: ExecutionHostOwnerRow = { id: 'r', executionHostId: 'ssh:m4air' } + expect(createRepoRowExecutionHostLookup([first, second]).byId('r')).toEqual({ + kind: 'resolved', + owner: first + }) + }) + + it('keeps byHost hits, misses and repo-list order', () => { + const openclaw = { id: 'r', connectionId: 'openclaw' } + const m4air = { id: 'r', connectionId: 'm4air' } + const lookup = createRepoRowExecutionHostLookup([openclaw, m4air]) + expect(lookup.byHost('r', 'ssh:m4air')).toBe(m4air) + expect(lookup.byHost('r', 'ssh:openclaw')).toBe(openclaw) + expect(lookup.byHost('r', 'local')).toBeNull() + expect(lookup.byHost('other', 'local')).toBeNull() + }) +}) diff --git a/src/shared/worktree-execution-host-resolution.ts b/src/shared/worktree-execution-host-resolution.ts index 8abe5de72cc..00b66b7f11c 100644 --- a/src/shared/worktree-execution-host-resolution.ts +++ b/src/shared/worktree-execution-host-resolution.ts @@ -88,20 +88,37 @@ export function resolveWorktreeExecutionHost( } } -/** Array-backed lookup for callers holding the whole repo list (main's store). */ +const EMPTY_ROWS: readonly never[] = [] + +/** + * Array-backed lookup for callers holding the whole repo list (main's store). Grouped once at + * construction — a lookup is hit once per worktree key per target, so a per-call `filter` was an + * O(repos) rescan each time. Rows keep repo-list order, which `byId` depends on for `rows[0]`. + */ export function createRepoRowExecutionHostLookup( repos: readonly T[] ): ExecutionHostOwnerLookup { - const rowsFor = (repoId: string): T[] => repos.filter((repo) => repo.id === repoId) + const rowsById = new Map() + for (const repo of repos) { + const rows = rowsById.get(repo.id) + if (rows) { + rows.push(repo) + } else { + rowsById.set(repo.id, [repo]) + } + } + const rowsFor = (repoId: string): readonly T[] => rowsById.get(repoId) ?? EMPTY_ROWS return { byId: (repoId) => { const rows = rowsFor(repoId) - if (rows.length === 0) { + const owner = rows[0] + if (!owner) { return { kind: 'missing' } } - const hostIds = new Set(rows.map((repo) => getRepoExecutionHostId(repo))) - const owner = rows[0] - return hostIds.size > 1 || !owner ? { kind: 'ambiguous' } : { kind: 'resolved', owner } + const ownerHostId = getRepoExecutionHostId(owner) + return rows.some((repo) => getRepoExecutionHostId(repo) !== ownerHostId) + ? { kind: 'ambiguous' } + : { kind: 'resolved', owner } }, byHost: (repoId, hostId) => rowsFor(repoId).find((repo) => getRepoExecutionHostId(repo) === hostId) ?? null From 97e5eb8886ebf48460d4f28ea4795bac4b1a23c4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:54:59 -0700 Subject: [PATCH 226/398] perf(paths): guard the no-op regex passes on the path-comparison hot path (#18418) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(paths): guard the no-op regex passes and hoist the loop-invariant root `normalizeRuntimePathForComparison` ran two whole-string regex passes on every call — `/\/+/g` and `/\/+$/` — that cannot change a path with no doubled slash and no trailing slash, which is nearly every path. `parseWslUncPath` likewise folded backslashes and ran an anchored UNC regex over every POSIX path. Substring/char-code probes skip all of them, and a `createRelativePathInsideRootResolver` factory (mirroring the existing `createNormalizedPathInsideOrEqualMatcher`) folds a fan-out's root once instead of once per candidate. Outputs are unchanged; a seeded 200k-path differential fuzz against a pre-guard copy proves it. * perf(paths): drop the root hoist, land the guards alone The three in-module guards are the whole win: 5000-op batches, CPU time, median of 9 --- normalize 541 -> 239 ns/op, relativePathInsideRoot 1778 -> 899, isPathInsideOrEqual 1076 -> 572, parseWslUncPath 57 -> 14. The loop-invariant root hoist added 176 ns/op on top of that (899 -> 723) at 7 hand-picked call sites, and cost a new exported factory whose input contract is the opposite of the one next to it, plus a function substitution in worktree/ownership.ts. Not worth 0.9 ms per storm. Prod diff: 2 files. New ratchet pins the single-factory surface. * docs(paths): point the fixture header at the real guards test --- src/shared/cross-platform-path-guards.test.ts | 316 ++++++++++++++++++ ...ss-platform-path-unguarded.test-fixture.ts | 267 +++++++++++++++ src/shared/cross-platform-path.ts | 20 +- src/shared/wsl-paths.ts | 17 +- 4 files changed, 616 insertions(+), 4 deletions(-) create mode 100644 src/shared/cross-platform-path-guards.test.ts create mode 100644 src/shared/cross-platform-path-unguarded.test-fixture.ts diff --git a/src/shared/cross-platform-path-guards.test.ts b/src/shared/cross-platform-path-guards.test.ts new file mode 100644 index 00000000000..cf6c729b89b --- /dev/null +++ b/src/shared/cross-platform-path-guards.test.ts @@ -0,0 +1,316 @@ +/** + * Proves the substring/char-code guards added to `cross-platform-path.ts` and `parseWslUncPath` + * are pure fast paths: a seeded differential fuzz against the pre-guard copy in + * `cross-platform-path-unguarded.test-fixture.ts`, plus counters that fail if the guards regress. + */ +import { describe, expect, it, afterEach } from 'vitest' +import * as guarded from './cross-platform-path' +import * as unguarded from './cross-platform-path-unguarded.test-fixture' +import { parseWslUncPath } from './wsl-paths' + +// ─── Deterministic path generator ──────────────────────────────────── + +function createRandom(seed: number): () => number { + let state = seed >>> 0 + return () => { + state = (state + 0x6d2b79f5) >>> 0 + let t = Math.imul(state ^ (state >>> 15), 1 | state) + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t + return ((t ^ (t >>> 14)) >>> 0) / 4294967296 + } +} + +const PREFIXES = [ + '', + '/', + '//', + '///', + '.', + './', + '../', + 'C:/', + 'c:\\', + 'Z:', + '\\\\', + '\\\\wsl.localhost\\Ubuntu', + '//wsl.localhost/Ubuntu-22.04', + '//WSL$/Debian', + '\\\\wsl$\\ubuntu', + '//server/share', + '/mnt/c', + '\\\\wsl.localhost\\Ubuntu\\mnt\\c' +] + +// NFD + KELVIN SIGN are the folds `normalizeRuntimePathForComparison` is built around. +const SEGMENTS = [ + 'home', + 'user', + 'orca', + 'workspaces', + '..', + '.', + '', + 'a', + 'B', + 'wsl$', + 'wsl.localhost', + 'mnt', + 'c', + 'C', + 'répertoire', + 're\u0301pertoire', + '\u212Aelvin', + 'Kelvin', + 'back\\slash', + 'sp ace', + 'Ubuntu' +] + +const JOINERS = ['/', '/', '/', '//', '///', '\\', '\\\\'] +const SUFFIXES = ['', '', '', '/', '//', '\\', '/.', '/..'] + +function generatePath(random: () => number): string { + const pick = (items: readonly T[]): T => items[Math.floor(random() * items.length)] + let path = pick(PREFIXES) + const segmentCount = Math.floor(random() * 5) + for (let index = 0; index < segmentCount; index++) { + path += (path === '' ? '' : pick(JOINERS)) + pick(SEGMENTS) + } + return path + pick(SUFFIXES) +} + +/** Roots that actually contain the candidate, so the matching branches get exercised too. */ +function generateRoot(random: () => number, candidate: string): string { + const roll = random() + if (roll < 0.35) { + const cut = Math.floor(random() * (candidate.length + 1)) + return candidate.slice(0, cut) + } + if (roll < 0.45) { + return candidate + } + return generatePath(random) +} + +// ─── Differential fuzz ─────────────────────────────────────────────── + +const FUZZ_ITERATIONS = 20_000 + +describe('guarded path normalization matches the pre-guard implementation', () => { + it(`agrees on every export across ${FUZZ_ITERATIONS} seeded paths`, () => { + const random = createRandom(0x5eed) + const mismatches: string[] = [] + const record = (label: string, path: string, root: string): void => { + if (mismatches.length < 5) { + mismatches.push(`${label}: candidate=${JSON.stringify(path)} root=${JSON.stringify(root)}`) + } + } + + for (let iteration = 0; iteration < FUZZ_ITERATIONS; iteration++) { + const path = generatePath(random) + const root = generateRoot(random, path) + const distro = Math.floor(random() * 2) === 0 ? 'Ubuntu' : 'debian' + + const singles: [string, (value: string) => unknown, (value: string) => unknown][] = [ + [ + 'isWindowsAbsolutePathLike', + guarded.isWindowsAbsolutePathLike, + unguarded.isWindowsAbsolutePathLike + ], + [ + 'isCaseInsensitiveRuntimeRoot', + guarded.isCaseInsensitiveRuntimeRoot, + unguarded.isCaseInsensitiveRuntimeRoot + ], + [ + 'normalizeRuntimePathSeparators', + guarded.normalizeRuntimePathSeparators, + unguarded.normalizeRuntimePathSeparators + ], + [ + 'normalizeRuntimePathForComparison', + guarded.normalizeRuntimePathForComparison, + unguarded.normalizeRuntimePathForComparison + ], + ['isRuntimePathAbsolute', guarded.isRuntimePathAbsolute, unguarded.isRuntimePathAbsolute], + ['getRuntimePathBasename', guarded.getRuntimePathBasename, unguarded.getRuntimePathBasename] + ] + for (const [label, left, right] of singles) { + if (left(path) !== right(path)) { + record(label, path, root) + } + } + + const identity = guarded.getLocalWindowsWslPathIdentity(path) + const expectedIdentity = unguarded.getLocalWindowsWslPathIdentity(path) + if ( + identity.normalizedPath !== expectedIdentity.normalizedPath || + identity.aliasComparisonPath !== expectedIdentity.aliasComparisonPath || + identity.isWslUnc !== expectedIdentity.isWslUnc + ) { + record('getLocalWindowsWslPathIdentity', path, root) + } + const wslUnc = parseWslUncPath(path) + const expectedWslUnc = unguarded.parseWslUncPath(path) + if ( + wslUnc?.distro !== expectedWslUnc?.distro || + wslUnc?.linuxPath !== expectedWslUnc?.linuxPath + ) { + record('parseWslUncPath', path, root) + } + if ( + guarded.areLocalWindowsWslPathAliases(root, path) !== + unguarded.areLocalWindowsWslPathAliases(root, path) + ) { + record('areLocalWindowsWslPathAliases', path, root) + } + if ( + guarded.isWslUncPathForCallerLinuxPath(root, path, distro) !== + unguarded.isWslUncPathForCallerLinuxPath(root, path, distro) + ) { + record('isWslUncPathForCallerLinuxPath', path, root) + } + if ( + guarded.isWslUncPathForLinuxMountedPath(root, path) !== + unguarded.isWslUncPathForLinuxMountedPath(root, path) + ) { + record('isWslUncPathForLinuxMountedPath', path, root) + } + if (guarded.resolveRuntimePath(root, path) !== unguarded.resolveRuntimePath(root, path)) { + record('resolveRuntimePath', path, root) + } + if (guarded.isPathInsideOrEqual(root, path) !== unguarded.isPathInsideOrEqual(root, path)) { + record('isPathInsideOrEqual', path, root) + } + if ( + guarded.createNormalizedPathInsideOrEqualMatcher(root)( + guarded.normalizeRuntimePathForComparison(path) + ) !== + unguarded.createNormalizedPathInsideOrEqualMatcher(root)( + unguarded.normalizeRuntimePathForComparison(path) + ) + ) { + record('createNormalizedPathInsideOrEqualMatcher', path, root) + } + if ( + guarded.relativePathInsideRoot(root, path) !== unguarded.relativePathInsideRoot(root, path) + ) { + record('relativePathInsideRoot', path, root) + } + } + + expect(mismatches).toEqual([]) + }, 120_000) +}) + +// ─── The guards must not skip work that was actually needed ────────── + +describe('guards still do the work when the fast path does not apply', () => { + it('collapses doubled slashes', () => { + expect(guarded.normalizeRuntimePathForComparison('/a//b///c')).toBe('/a/b/c') + expect(guarded.normalizeRuntimePathSeparators('/a//b')).toBe('/a/b') + expect(guarded.relativePathInsideRoot('/a', '/a//b//c')).toBe('b/c') + }) + + it('trims trailing slashes but keeps bare roots', () => { + expect(guarded.normalizeRuntimePathForComparison('/a/b/')).toBe('/a/b') + expect(guarded.normalizeRuntimePathForComparison('/a/b//')).toBe('/a/b') + expect(guarded.normalizeRuntimePathForComparison('/')).toBe('/') + expect(guarded.normalizeRuntimePathForComparison('C:/')).toBe('c:/') + }) + + it('folds backslashes only on Windows-shaped paths', () => { + expect(guarded.normalizeRuntimePathForComparison('C:\\a\\b')).toBe('c:/a/b') + expect(guarded.normalizeRuntimePathSeparators('C:\\a\\\\b')).toBe('C:/a/b') + // Backslash is a legal POSIX filename character and must survive. + expect(guarded.normalizeRuntimePathForComparison('/a/b\\c')).toBe('/a/b\\c') + }) + + it('still parses both WSL UNC aliases in either separator spelling', () => { + expect(parseWslUncPath('\\\\wsl.localhost\\Ubuntu\\home\\me')).toEqual({ + distro: 'Ubuntu', + linuxPath: '/home/me' + }) + expect(parseWslUncPath('//wsl$/Debian/srv')).toEqual({ distro: 'Debian', linuxPath: '/srv' }) + expect(guarded.normalizeRuntimePathForComparison('\\\\wsl.localhost\\Ubuntu\\Repo')).toBe( + '//wsl/ubuntu/Repo' + ) + expect(parseWslUncPath('/wsl.localhost/Ubuntu/home')).toBeNull() + expect(parseWslUncPath('/')).toBeNull() + expect(parseWslUncPath('')).toBeNull() + }) +}) + +// ─── Regression guards: counted work, not wall clock ───────────────── + +const originalReplace = String.prototype.replace + +afterEach(() => { + String.prototype.replace = originalReplace +}) + +function countReplaceCalls(run: () => void): number { + let calls = 0 + String.prototype.replace = function (this: string, ...args: never[]) { + calls++ + return originalReplace.apply(this, args as never) + } as typeof String.prototype.replace + try { + run() + } finally { + String.prototype.replace = originalReplace + } + return calls +} + +const CLEAN_POSIX_PATH = + '/Users/nwparker/orca/workspaces/orca/perf/src/renderer/src/components/x.ts' + +describe('no-op regex passes stay skipped', () => { + it('runs zero replaces for a path with no doubled slash, trailing slash, or backslash', () => { + expect( + countReplaceCalls(() => guarded.normalizeRuntimePathForComparison(CLEAN_POSIX_PATH)) + ).toBe(0) + expect(countReplaceCalls(() => guarded.normalizeRuntimePathSeparators(CLEAN_POSIX_PATH))).toBe( + 0 + ) + expect(countReplaceCalls(() => parseWslUncPath(CLEAN_POSIX_PATH))).toBe(0) + }) + + it('runs one replace per pass that is genuinely needed', () => { + expect(countReplaceCalls(() => guarded.normalizeRuntimePathForComparison('/a//b'))).toBe(1) + expect(countReplaceCalls(() => guarded.normalizeRuntimePathForComparison('/a/b/'))).toBe(1) + }) +}) + +// ─── One root-bound factory, one input contract ────────────────────── + +/** + * `createNormalizedPathInsideOrEqualMatcher` demands an already-normalized candidate because + * `normalizeRuntimePathForComparison` is not idempotent. A sibling factory on the same root that + * took RAW candidates would put two opposite contracts one line apart, and mixing them up returns + * "outside the root" rather than throwing. Hoisting a root out of a loop is worth ~0.2 us/event; + * this is the price. Keep the raw-candidate entry point the plain `relativePathInsideRoot` call. + */ +describe('cross-platform-path exposes a single root-bound factory', () => { + it('has no raw-candidate sibling to the normalized matcher', () => { + expect(Object.keys(guarded).filter((name) => name.startsWith('create'))).toEqual([ + 'createNormalizedPathInsideOrEqualMatcher' + ]) + }) + + it('shows what mixing the two contracts would cost', () => { + const root = '//wsl.localhost/Ubuntu/Repo' + const candidate = '//wsl.localhost/Ubuntu/Repo/src/App.tsx' + const normalizedCandidate = guarded.normalizeRuntimePathForComparison(candidate) + expect(guarded.normalizeRuntimePathForComparison(normalizedCandidate)).not.toBe( + normalizedCandidate + ) + + const matcher = guarded.createNormalizedPathInsideOrEqualMatcher(root) + expect(matcher(normalizedCandidate)).toBe(true) + // The raw spelling a resolver would accept is silently reported as outside the root. + expect(matcher(candidate)).toBe(false) + expect(guarded.relativePathInsideRoot(root, candidate)).toBe('src/App.tsx') + }) +}) diff --git a/src/shared/cross-platform-path-unguarded.test-fixture.ts b/src/shared/cross-platform-path-unguarded.test-fixture.ts new file mode 100644 index 00000000000..c6be537eea3 --- /dev/null +++ b/src/shared/cross-platform-path-unguarded.test-fixture.ts @@ -0,0 +1,267 @@ +/** + * Verbatim pre-guard copy of `cross-platform-path.ts` and `parseWslUncPath`, kept only so + * `cross-platform-path-guards.test.ts` can differentially fuzz the guarded versions + * against what shipped. Comments were stripped; the code is otherwise unchanged. Update this file + * only when the guarded originals are meant to change behaviour. + */ +import { toWindowsWslPath } from './wsl-paths' + +type WslUncPathInfo = { distro: string; linuxPath: string } + +export function parseWslUncPath(path: string): WslUncPathInfo | null { + const normalized = path.replace(/\\/g, '/') + const match = normalized.match(/^\/\/(wsl\.localhost|wsl\$)\/([^/]+)(\/.*)?$/i) + if (!match) { + return null + } + return { distro: match[2], linuxPath: match[3] || '/' } +} + +function isWslUncPath(path: string): boolean { + return parseWslUncPath(path) !== null +} + +const SLASH_CHAR_CODE = '/'.charCodeAt(0) + +export function isWindowsAbsolutePathLike(value: string): boolean { + return /^[A-Za-z]:[\\/]/.test(value) || value.startsWith('\\\\') || value.startsWith('//') +} + +export function isCaseInsensitiveRuntimeRoot(rootPath: string): boolean { + return isWindowsAbsolutePathLike(rootPath) && !isWslUncPath(rootPath) +} + +export function normalizeRuntimePathSeparators(value: string): string { + const normalized = value.replace(/\\/g, '/').replace(/\/+/g, '/') + if (value.startsWith('\\\\') || value.startsWith('//')) { + return `//${normalized.replace(/^\/+/, '')}` + } + return normalized +} + +export function normalizeRuntimePathForComparison(rawValue: string): string { + const value = rawValue.normalize('NFC') + const isWindowsPath = isWindowsAbsolutePathLike(value) + const normalized = trimRuntimePathTrailingSlash( + isWindowsPath ? normalizeRuntimePathSeparators(value) : value.replace(/\/+/g, '/') + ) + const wslUnc = normalized.match(/^\/\/(?:wsl\.localhost|wsl\$)\/([^/]+)(\/[\s\S]*)?$/i) + if (wslUnc) { + return `//wsl/${wslUnc[1].toLowerCase()}${wslUnc[2] ?? ''}` + } + return isWindowsPath ? normalized.toLowerCase() : normalized +} + +export function isWslUncPathForCallerLinuxPath( + uncPath: string, + linuxPath: string, + callerDistro: string +): boolean { + const parsed = parseWslUncPath(uncPath) + if (!parsed) { + return false + } + return ( + parsed.distro.toLowerCase() === callerDistro.toLowerCase() && + normalizeRuntimePathForComparison(parsed.linuxPath) === + normalizeRuntimePathForComparison(linuxPath) + ) +} + +export function isWslUncPathForLinuxMountedPath(uncPath: string, linuxPath: string): boolean { + const parsed = parseWslUncPath(uncPath) + if (!parsed || !/^\/mnt\/[A-Za-z](?:\/|$)/.test(parsed.linuxPath)) { + return false + } + if (!/^\/mnt\/[A-Za-z](?:\/|$)/.test(linuxPath)) { + return false + } + return ( + normalizeRuntimePathForComparison(toWindowsWslPath(parsed.linuxPath, parsed.distro)) === + normalizeRuntimePathForComparison(toWindowsWslPath(linuxPath, parsed.distro)) + ) +} + +export function areLocalWindowsWslPathAliases(left: string, right: string): boolean { + const leftIdentity = getLocalWindowsWslPathIdentity(left) + const rightIdentity = getLocalWindowsWslPathIdentity(right) + return ( + (leftIdentity.isWslUnc || rightIdentity.isWslUnc) && + leftIdentity.aliasComparisonPath === rightIdentity.aliasComparisonPath + ) +} + +export type LocalWindowsWslPathIdentity = { + normalizedPath: string + aliasComparisonPath: string + isWslUnc: boolean +} + +export function getLocalWindowsWslPathIdentity(value: string): LocalWindowsWslPathIdentity { + const wslPath = parseWslUncPath(value) + const normalizedPath = normalizeRuntimePathForComparison(value) + return { + normalizedPath, + aliasComparisonPath: wslPath + ? normalizeRuntimePathForComparison(toWindowsWslPath(wslPath.linuxPath, wslPath.distro)) + : normalizedPath, + isWslUnc: wslPath !== null + } +} + +export function isRuntimePathAbsolute( + value: string, + pathFlavor: 'posix' | 'windows' = isWindowsPathFlavor(value) ? 'windows' : 'posix' +): boolean { + if (pathFlavor === 'windows') { + return /^[A-Za-z]:[\\/]/.test(value) || value.startsWith('\\') || value.startsWith('/') + } + return value.startsWith('/') +} + +export function resolveRuntimePath(basePath: string, targetPath: string): string { + const pathFlavor = + isWindowsPathFlavor(basePath) || isWindowsPathFlavor(targetPath) ? 'windows' : 'posix' + if (isRuntimePathAbsolute(targetPath, pathFlavor)) { + return normalizeRuntimePathDots(targetPath, pathFlavor) + } + return normalizeRuntimePathDots( + `${trimRuntimePathTrailingSlash(normalizeRuntimePathSeparators(basePath))}/${targetPath}`, + pathFlavor + ) +} + +export function getRuntimePathBasename(value: string): string { + const trimmed = value.replace(/[\\/]+$/g, '') + if (!trimmed) { + return '' + } + return trimmed.split(/[\\/]/).findLast(Boolean) ?? '' +} + +export function createNormalizedPathInsideOrEqualMatcher( + rootPath: string +): (normalizedCandidate: string) => boolean { + const root = normalizeRuntimePathForComparison(rootPath) + const rootWithBoundary = + root === '/' || /^[a-z]:\/$/i.test(root) ? root : `${root.replace(/\/+$/, '')}/` + return (normalizedCandidate) => + normalizedCandidate === root || normalizedCandidate.startsWith(rootWithBoundary) +} + +export function isPathInsideOrEqual(rootPath: string, candidatePath: string): boolean { + return createNormalizedPathInsideOrEqualMatcher(rootPath)( + normalizeRuntimePathForComparison(candidatePath) + ) +} + +export function relativePathInsideRoot(rootPath: string, candidatePath: string): string | null { + const normalizedCandidate = trimRuntimePathTrailingSlash( + isWindowsAbsolutePathLike(candidatePath.normalize('NFC')) + ? normalizeRuntimePathSeparators(candidatePath) + : candidatePath.replace(/\/+/g, '/') + ) + const comparisonRoot = normalizeRuntimePathForComparison(rootPath) + const comparisonCandidate = normalizeRuntimePathForComparison(candidatePath) + + if (comparisonCandidate === comparisonRoot) { + return '' + } + const isRoot = comparisonRoot === '/' || /^[a-z]:\/$/i.test(comparisonRoot) + const comparisonPrefix = isRoot ? comparisonRoot : `${comparisonRoot}/` + if (!comparisonCandidate.startsWith(comparisonPrefix)) { + return null + } + return sliceCandidatePastRootSegments(comparisonRoot, normalizedCandidate) +} + +function sliceCandidatePastRootSegments(root: string, candidate: string): string { + let remainingRootSegments = 0 + let inRootSegment = false + for (let index = 0; index < root.length; index++) { + if (root.charCodeAt(index) === SLASH_CHAR_CODE) { + inRootSegment = false + } else if (!inRootSegment) { + inRootSegment = true + remainingRootSegments++ + } + } + + let inSegment = false + for (let index = 0; index < candidate.length; index++) { + if (candidate.charCodeAt(index) === SLASH_CHAR_CODE) { + inSegment = false + continue + } + if (!inSegment) { + inSegment = true + if (remainingRootSegments-- === 0) { + return candidate.slice(index) + } + } + } + return '' +} + +function trimRuntimePathTrailingSlash(value: string): string { + if (value === '/' || /^[A-Za-z]:\/$/.test(value)) { + return value + } + return value.replace(/\/+$/, '') +} + +function isWindowsPathFlavor(value: string): boolean { + return /^[A-Za-z]:[\\/]/.test(value) || value.includes('\\') || value.startsWith('//') +} + +function normalizeRuntimePathDots(value: string, pathFlavor: 'posix' | 'windows'): string { + const normalized = normalizeRuntimePathSeparators(value) + const { root, rest } = splitRuntimePathRoot(normalized, pathFlavor) + const segments: string[] = [] + for (const segment of rest.split('/')) { + if (!segment || segment === '.') { + continue + } + if (segment === '..') { + if (segments.length > 0 && segments.at(-1) !== '..') { + segments.pop() + } else if (!root) { + segments.push(segment) + } + continue + } + segments.push(segment) + } + const suffix = segments.join('/') + if (!root) { + return suffix || '.' + } + return suffix ? `${root}${suffix}` : trimRuntimePathTrailingSlash(root) +} + +function splitRuntimePathRoot( + value: string, + pathFlavor: 'posix' | 'windows' +): { root: string; rest: string } { + if (pathFlavor === 'windows') { + const drive = value.match(/^([A-Za-z]:)(?:\/|$)/) + if (drive) { + return { root: `${drive[1]}/`, rest: value.slice(drive[0].length) } + } + if (value.startsWith('//')) { + const parts = value.slice(2).split('/') + if (parts.length >= 2 && parts[0] && parts[1]) { + const root = `//${parts[0]}/${parts[1]}/` + return { root, rest: parts.slice(2).join('/') } + } + return { root: '//', rest: value.slice(2) } + } + if (value.startsWith('/')) { + return { root: '/', rest: value.slice(1) } + } + } + if (value.startsWith('/')) { + return { root: '/', rest: value.slice(1) } + } + return { root: '', rest: value } +} diff --git a/src/shared/cross-platform-path.ts b/src/shared/cross-platform-path.ts index 61308a70489..f173914c789 100644 --- a/src/shared/cross-platform-path.ts +++ b/src/shared/cross-platform-path.ts @@ -21,13 +21,23 @@ export function isCaseInsensitiveRuntimeRoot(rootPath: string): boolean { } export function normalizeRuntimePathSeparators(value: string): string { - const normalized = value.replace(/\\/g, '/').replace(/\/+/g, '/') + const normalized = collapseRuntimePathSlashes( + value.includes('\\') ? value.replace(/\\/g, '/') : value + ) if (value.startsWith('\\\\') || value.startsWith('//')) { return `//${normalized.replace(/^\/+/, '')}` } return normalized } +/** + * Why the probe: `/\/+/g` can only change a string that contains `//`, and the scan is the + * dominant cost of every comparison key on the FS-event storm path (`includes` is ~30x cheaper). + */ +function collapseRuntimePathSlashes(value: string): string { + return value.includes('//') ? value.replace(/\/+/g, '/') : value +} + /** * Comparison key only — never return this as, or splice it into, a real path. * @@ -45,7 +55,7 @@ export function normalizeRuntimePathForComparison(rawValue: string): string { // Why: backslash is a valid POSIX filename character; fold it only when the // path itself proves Windows drive/UNC semantics. const normalized = trimRuntimePathTrailingSlash( - isWindowsPath ? normalizeRuntimePathSeparators(value) : value.replace(/\/+/g, '/') + isWindowsPath ? normalizeRuntimePathSeparators(value) : collapseRuntimePathSlashes(value) ) const wslUnc = normalized.match(/^\/\/(?:wsl\.localhost|wsl\$)\/([^/]+)(\/[\s\S]*)?$/i) if (wslUnc) { @@ -190,7 +200,7 @@ export function relativePathInsideRoot(rootPath: string, candidatePath: string): const normalizedCandidate = trimRuntimePathTrailingSlash( isWindowsAbsolutePathLike(candidatePath.normalize('NFC')) ? normalizeRuntimePathSeparators(candidatePath) - : candidatePath.replace(/\/+/g, '/') + : collapseRuntimePathSlashes(candidatePath) ) const comparisonRoot = normalizeRuntimePathForComparison(rootPath) const comparisonCandidate = normalizeRuntimePathForComparison(candidatePath) @@ -242,6 +252,10 @@ function sliceCandidatePastRootSegments(root: string, candidate: string): string } function trimRuntimePathTrailingSlash(value: string): string { + // Nothing to trim, and neither preserved-root case can match, unless the value ends in `/`. + if (!value.endsWith('/')) { + return value + } if (value === '/' || /^[A-Za-z]:\/$/.test(value)) { return value } diff --git a/src/shared/wsl-paths.ts b/src/shared/wsl-paths.ts index ba7f9a12f92..1d4b535812a 100644 --- a/src/shared/wsl-paths.ts +++ b/src/shared/wsl-paths.ts @@ -3,8 +3,23 @@ export type WslUncPathInfo = { linuxPath: string } +const SLASH_CHAR_CODE = '/'.charCodeAt(0) +const BACKSLASH_CHAR_CODE = '\\'.charCodeAt(0) + +function isPathSeparatorCharCode(charCode: number): boolean { + return charCode === SLASH_CHAR_CODE || charCode === BACKSLASH_CHAR_CODE +} + export function parseWslUncPath(path: string): WslUncPathInfo | null { - const normalized = path.replace(/\\/g, '/') + // The match is anchored at `//` after the fold, so only two leading separators can ever reach it. + // Every POSIX path pays the fold + regex otherwise, and this is on the FS-event storm path. + if ( + !isPathSeparatorCharCode(path.charCodeAt(0)) || + !isPathSeparatorCharCode(path.charCodeAt(1)) + ) { + return null + } + const normalized = path.includes('\\') ? path.replace(/\\/g, '/') : path const match = normalized.match(/^\/\/(wsl\.localhost|wsl\$)\/([^/]+)(\/.*)?$/i) if (!match) { return null From 6415b1dc22bee422c2aaefcef37376d8376ea59f Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:56:29 -0700 Subject: [PATCH 227/398] perf(images): probe raster headers instead of decoding whole payloads, memoize repo icon validation (#18421) * perf(images): measure raster headers from a probe and memoize repo icon validation `getRepos()` re-sanitizes every repo on every call, and an uploaded/file repo icon costs a full base64 decode of its data URI each time. Three fixes: - `writeQuartet` destructured a mutable array, which sends V8 through the iterator protocol once per four input characters; index reads plus a length counter produce identical bytes. - `decodeBase64Prefix` decoded the whole payload despite only the first bytes being needed. `exceedsRasterImagePreviewLimits` now probes 64 bytes and widens x16 until the header measures, and only re-runs the original full-payload decode when the verdict would suppress a preview. - `sanitizeRepoIcon`'s src validation is memoized per icon source with a bounded FIFO map, reusing the `memoizeTitleClassification` idiom (now a shared `memoizeByStringKey`). * perf(images): key icon-validation memo on the persisted icon object Replaces the per-source 64-entry FIFO string-key memo with a WeakMap keyed on the persisted repoIcon object that hydrateRepo already receives, storing {src, source, supported} and re-checking both fields on a hit. Retention becomes zero by construction (entries die with state.repos[i].repoIcon), so there is no cap to evict live icons and no dead icon strings held after a repo or icon is replaced. The identity re-check makes an in-place mutation unable to serve a stale verdict. Drops bounded-string-key-memo.ts and reverts the collateral terminal-title-classification-memo refactor. --- .../raster-image-base64-preview.test.ts | 345 ++++++++++++++++++ src/shared/raster-image-base64-preview.ts | 83 +++-- src/shared/repo-icon.test.ts | 99 ++++- src/shared/repo-icon.ts | 32 +- 4 files changed, 535 insertions(+), 24 deletions(-) create mode 100644 src/shared/raster-image-base64-preview.test.ts diff --git a/src/shared/raster-image-base64-preview.test.ts b/src/shared/raster-image-base64-preview.test.ts new file mode 100644 index 00000000000..af4d9de118a --- /dev/null +++ b/src/shared/raster-image-base64-preview.test.ts @@ -0,0 +1,345 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import { decodeBase64Prefix, exceedsRasterImagePreviewLimits } from './raster-image-base64-preview' +import type * as RasterImageDimensionsModule from './raster-image-dimensions' +import { readRasterImageDimensions } from './raster-image-dimensions' +import { + isKnownRasterImageMimeType, + isRasterImagePreviewDimensions, + RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES +} from './raster-image-preview-limits' + +/** The pre-change verdict: one full-payload decode, then one dimension read. */ +function unprobedExceeds(content: string, mimeType: string | undefined): boolean { + if (!isKnownRasterImageMimeType(mimeType)) { + return false + } + const prefix = referenceDecodeBase64Prefix(content, RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES) + if (!prefix) { + return false + } + const dimensions = readRasterImageDimensions(prefix) + return dimensions !== null && !isRasterImagePreviewDimensions(dimensions) +} + +// Why a module mock: the byte length handed to the dimension reader is the direct measure of how +// much of the payload the preview check decoded, and it is the only observable difference between +// the early-stopping probe and the full-payload decode it replaces. +const { dimensionReadLengths } = vi.hoisted(() => ({ dimensionReadLengths: [] as number[] })) + +vi.mock('./raster-image-dimensions', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + readRasterImageDimensions: (bytes: Uint8Array) => { + dimensionReadLengths.push(bytes.byteLength) + return actual.readRasterImageDimensions(bytes) + } + } +}) + +// ── Reference decoder: the pre-change implementation, verbatim ────────────────────────────────── +const BASE64_PADDING = -2 +const INVALID_BASE64 = -1 + +function base64Value(code: number): number { + if (code >= 65 && code <= 90) { + return code - 65 + } + if (code >= 97 && code <= 122) { + return code - 71 + } + if (code >= 48 && code <= 57) { + return code + 4 + } + if (code === 43) { + return 62 + } + if (code === 47) { + return 63 + } + if (code === 61) { + return BASE64_PADDING + } + return INVALID_BASE64 +} + +function isWhitespace(code: number): boolean { + return code === 9 || code === 10 || code === 12 || code === 13 || code === 32 +} + +function referenceWriteQuartet( + output: Uint8Array, + offset: number, + quartet: readonly number[] +): { bytesWritten: number; padded: boolean } | null { + const [a, b, c, d] = quartet + if (a === undefined || b === undefined || a < 0 || b < 0) { + return null + } + if (c === BASE64_PADDING) { + if (d !== BASE64_PADDING) { + return null + } + if (offset < output.length) { + output[offset] = (a << 2) | (b >> 4) + } + return { bytesWritten: Math.min(1, output.length - offset), padded: true } + } + if (c === undefined || c < 0) { + return null + } + if (offset < output.length) { + output[offset] = (a << 2) | (b >> 4) + } + if (offset + 1 < output.length) { + output[offset + 1] = ((b & 15) << 4) | (c >> 2) + } + if (d === BASE64_PADDING) { + return { bytesWritten: Math.min(2, output.length - offset), padded: true } + } + if (d === undefined || d < 0) { + return null + } + if (offset + 2 < output.length) { + output[offset + 2] = ((c & 3) << 6) | d + } + return { bytesWritten: Math.min(3, output.length - offset), padded: false } +} + +function referenceDecodeBase64Prefix(content: string, maxBytes: number): Uint8Array | null { + const capacity = Math.min(maxBytes, Math.ceil(content.length / 4) * 3) + const output = new Uint8Array(capacity) + const quartet: number[] = [] + let outputLength = 0 + let padded = false + + for (let index = 0; index < content.length && outputLength < capacity; index += 1) { + const code = content.charCodeAt(index) + if (isWhitespace(code)) { + continue + } + if (padded) { + return null + } + const value = base64Value(code) + if (value === INVALID_BASE64) { + return null + } + quartet.push(value) + if (quartet.length !== 4) { + continue + } + const decoded = referenceWriteQuartet(output, outputLength, quartet) + if (!decoded) { + return null + } + outputLength += decoded.bytesWritten + padded = decoded.padded + quartet.length = 0 + } + + if (!padded && outputLength < capacity && quartet.length > 0) { + if (quartet.length === 1 || quartet.includes(BASE64_PADDING)) { + return null + } + while (quartet.length < 4) { + quartet.push(BASE64_PADDING) + } + const decoded = referenceWriteQuartet(output, outputLength, quartet) + if (!decoded) { + return null + } + outputLength += decoded.bytesWritten + } + return output.subarray(0, outputLength) +} + +// ── Fixtures ─────────────────────────────────────────────────────────────────────────────────── +function pngBytes(totalBytes: number, width: number, height: number): Buffer { + const bytes = Buffer.alloc(Math.max(totalBytes, 24)) + Buffer.from([137, 80, 78, 71, 13, 10, 26, 10]).copy(bytes) + bytes.writeUInt32BE(13, 8) + bytes.write('IHDR', 12, 'ascii') + bytes.writeUInt32BE(width, 16) + bytes.writeUInt32BE(height, 20) + for (let index = 24; index < bytes.length; index += 1) { + bytes[index] = (index * 31 + 7) & 0xff + } + return bytes +} + +/** SOI, `metadataBytes` of APP2 padding (real cameras chain many segments), then SOF0. */ +function jpegBytes( + metadataBytes: number, + width: number, + height: number, + trailingBytes = 4096 +): Buffer { + const parts: Buffer[] = [Buffer.from([0xff, 0xd8])] + for (let written = 0; written < metadataBytes;) { + const size = Math.min(65_533, metadataBytes - written) + const header = Buffer.alloc(4) + header.writeUInt16BE(0xffe2) + header.writeUInt16BE(size + 2, 2) + parts.push(header, Buffer.alloc(size)) + written += size + } + const sof = Buffer.alloc(11) + sof.writeUInt16BE(0xffc0) + sof.writeUInt16BE(8, 2) + sof[4] = 8 + sof.writeUInt16BE(height, 5) + sof.writeUInt16BE(width, 7) + parts.push(sof, Buffer.alloc(trailingBytes)) + return Buffer.concat(parts) +} + +function gifBytes(width: number, height: number): Buffer { + const gif = Buffer.alloc(64) + gif.write('GIF89a', 0, 'ascii') + gif.writeUInt16LE(width, 6) + gif.writeUInt16LE(height, 8) + return gif +} + +function webpBytes(width: number, height: number): Buffer { + const webp = Buffer.alloc(64) + webp.write('RIFF', 0, 'ascii') + webp.writeUInt32LE(50, 4) + webp.write('WEBP', 8, 'ascii') + webp.write('VP8X', 12, 'ascii') + webp.writeUInt32LE(10, 16) + webp.writeUIntLE(width - 1, 24, 3) + webp.writeUIntLE(height - 1, 27, 3) + return webp +} + +const DECODE_FIXTURES: { label: string; content: string }[] = [ + { label: 'png', content: pngBytes(24, 512, 512).toString('base64') }, + { label: 'png padded once', content: pngBytes(26, 512, 512).toString('base64') }, + { label: 'png padded twice', content: pngBytes(25, 512, 512).toString('base64') }, + { label: 'png 70 KiB', content: pngBytes(70_000, 512, 512).toString('base64') }, + { label: 'jpeg', content: jpegBytes(0, 640, 480).toString('base64') }, + { label: 'jpeg 70 KiB exif', content: jpegBytes(70_000, 4000, 3000).toString('base64') }, + { label: 'gif', content: gifBytes(320, 240).toString('base64') }, + { label: 'webp', content: webpBytes(800, 600).toString('base64') }, + { label: 'empty', content: '' }, + { label: 'one character', content: 'A' }, + { label: 'two characters', content: 'AB' }, + { label: 'three characters', content: 'ABC' }, + { label: 'invalid character', content: 'AB*D' }, + { label: 'invalid tail', content: `${pngBytes(24, 4, 4).toString('base64')}!!!` }, + { label: 'padding mid-payload', content: 'AAAA=AAA' }, + { label: 'lone padding in tail', content: 'AAAAAB=' }, + { label: 'single padding', content: 'AAAAAA==' }, + { label: 'double padding', content: 'AAAAAAA=' }, + { label: 'stray padding after padding', content: 'AAAA====' }, + { + label: 'line-wrapped png', + content: pngBytes(70_000, 512, 512) + .toString('base64') + .replace(/(.{76})/g, '$1\r\n') + }, + { label: 'leading and trailing whitespace', content: `\n\t ${'AAAA'} \r\n` }, + { label: 'truncated png header', content: pngBytes(24, 512, 512).toString('base64').slice(0, 18) } +] + +const DECODE_CAPS = [ + 0, + 1, + 2, + 3, + 4, + 23, + 24, + 25, + 63, + 64, + 65, + 1024, + RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES +] + +describe('decodeBase64Prefix', () => { + it('decodes byte-for-byte identically to the reference implementation', () => { + for (const { label, content } of DECODE_FIXTURES) { + for (const maxBytes of DECODE_CAPS) { + const expected = referenceDecodeBase64Prefix(content, maxBytes) + const actual = decodeBase64Prefix(content, maxBytes) + const detail = `${label} @ maxBytes=${maxBytes}` + if (expected === null) { + expect(actual, detail).toBeNull() + continue + } + expect(actual, detail).not.toBeNull() + expect(Array.from(actual!), detail).toEqual(Array.from(expected)) + } + } + }) + + it('stops at the byte cap instead of decoding the whole payload', () => { + const content = pngBytes(70_000, 512, 512).toString('base64') + expect(decodeBase64Prefix(content, 32)?.byteLength).toBe(32) + }) +}) + +describe('exceedsRasterImagePreviewLimits', () => { + beforeEach(() => { + dimensionReadLengths.length = 0 + }) + + it('measures a large image from its first bytes, not its last', () => { + // Regression guard: before the probe this decoded all 5 MiB before reading 24 bytes of IHDR. + const content = pngBytes(5 * 1024 * 1024, 512, 512).toString('base64') + expect(exceedsRasterImagePreviewLimits(content, 'image/png')).toBe(false) + expect(dimensionReadLengths).toEqual([64]) + }) + + it('widens the probe until a JPEG SOF past its metadata is reachable', () => { + const content = jpegBytes(70_000, 4000, 3000, 3_000_000).toString('base64') + expect(exceedsRasterImagePreviewLimits(content, 'image/jpeg')).toBe(false) + expect(dimensionReadLengths).toEqual([64, 1024, 16_384, 262_144]) + // Far below the ~3 MiB the payload decodes to, and below the 8 MiB fallback cap. + expect(dimensionReadLengths.at(-1)!).toBeLessThan(RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES) + }) + + it('re-reads the whole payload before suppressing an over-limit image', () => { + const content = pngBytes(70_000, 32_769, 1).toString('base64') + expect(exceedsRasterImagePreviewLimits(content, 'image/png')).toBe(true) + // The header answers at 64 bytes, but a suppression verdict is only taken from the same + // full-payload decode the unprobed implementation used, so invalid base64 past the header + // still demotes the answer to "could not measure". + expect(dimensionReadLengths).toEqual([64, 70_000]) + }) + + it('keeps rendering an over-limit header whose payload is not valid base64', () => { + const content = `${pngBytes(70_000, 32_769, 1).toString('base64')}!!!` + expect(exceedsRasterImagePreviewLimits(content, 'image/png')).toBe(false) + }) + + it('agrees with the unprobed implementation on every fixture and mime type', () => { + const mimeTypes = [ + 'image/png', + 'image/jpeg', + 'image/gif', + 'image/webp', + 'image/svg+xml', + undefined + ] + const fixtures = [ + ...DECODE_FIXTURES, + { label: 'over-limit png', content: pngBytes(24, 32_769, 1).toString('base64') }, + { label: 'over-limit pixels png', content: pngBytes(24, 8192, 8192).toString('base64') }, + { label: 'over-limit gif', content: gifBytes(65_535, 65_535).toString('base64') }, + { label: 'over-limit jpeg', content: jpegBytes(70_000, 40_000, 40_000).toString('base64') }, + { label: 'over-limit webp', content: webpBytes(40_000, 40_000).toString('base64') } + ] + for (const { label, content } of fixtures) { + for (const mimeType of mimeTypes) { + expect(exceedsRasterImagePreviewLimits(content, mimeType), `${label} / ${mimeType}`).toBe( + unprobedExceeds(content, mimeType) + ) + } + } + }) +}) diff --git a/src/shared/raster-image-base64-preview.ts b/src/shared/raster-image-base64-preview.ts index b366d737218..78fb10f0d53 100644 --- a/src/shared/raster-image-base64-preview.ts +++ b/src/shared/raster-image-base64-preview.ts @@ -34,13 +34,20 @@ function isWhitespace(code: number): boolean { return code === 9 || code === 10 || code === 12 || code === 13 || code === 32 } +/** `quartetLength` under 4 is a final short group; the missing slots decode as `=` padding. */ function writeQuartet( output: Uint8Array, offset: number, - quartet: readonly number[] + quartet: readonly number[], + quartetLength: number ): { bytesWritten: number; padded: boolean } | null { - const [a, b, c, d] = quartet - if (a === undefined || b === undefined || a < 0 || b < 0) { + // Index reads, not `const [a, b, c, d] = quartet`: destructuring an array runs the iterator + // protocol (Symbol.iterator plus four `.next()` calls) once per four input characters. + const a = quartet[0] + const b = quartet[1] + const c = quartetLength > 2 ? quartet[2] : BASE64_PADDING + const d = quartetLength > 3 ? quartet[3] : BASE64_PADDING + if (quartetLength < 2 || a === undefined || b === undefined || a < 0 || b < 0) { return null } if (c === BASE64_PADDING) { @@ -73,10 +80,13 @@ function writeQuartet( return { bytesWritten: Math.min(3, output.length - offset), padded: false } } -function decodeBase64Prefix(content: string, maxBytes: number): Uint8Array | null { +/** Exported so the decode can be compared byte-for-byte against a reference implementation. */ +export function decodeBase64Prefix(content: string, maxBytes: number): Uint8Array | null { const capacity = Math.min(maxBytes, Math.ceil(content.length / 4) * 3) const output = new Uint8Array(capacity) - const quartet: number[] = [] + // Fixed four slots plus a counter, never resized: `quartet.length = 0` deoptimizes the array. + const quartet = [0, 0, 0, 0] + let quartetLength = 0 let outputLength = 0 let padded = false @@ -92,27 +102,27 @@ function decodeBase64Prefix(content: string, maxBytes: number): Uint8Array | nul if (value === INVALID_BASE64) { return null } - quartet.push(value) - if (quartet.length !== 4) { + quartet[quartetLength] = value + quartetLength += 1 + if (quartetLength !== 4) { continue } - const decoded = writeQuartet(output, outputLength, quartet) + const decoded = writeQuartet(output, outputLength, quartet, 4) if (!decoded) { return null } outputLength += decoded.bytesWritten padded = decoded.padded - quartet.length = 0 + quartetLength = 0 } - if (!padded && outputLength < capacity && quartet.length > 0) { - if (quartet.length === 1 || quartet.includes(BASE64_PADDING)) { - return null + if (!padded && outputLength < capacity && quartetLength > 0) { + for (let index = 0; index < quartetLength; index += 1) { + if (quartet[index] === BASE64_PADDING) { + return null + } } - while (quartet.length < 4) { - quartet.push(BASE64_PADDING) - } - const decoded = writeQuartet(output, outputLength, quartet) + const decoded = writeQuartet(output, outputLength, quartet, quartetLength) if (!decoded) { return null } @@ -121,6 +131,13 @@ function decodeBase64Prefix(content: string, maxBytes: number): Uint8Array | nul return output.subarray(0, outputLength) } +// First probe: past every fixed-offset header (PNG 24, GIF 10, WebP 30, BMP 26) and a JFIF-only +// JPEG's SOF, so an icon or screenshot is measured from its first bytes instead of its last. +const RASTER_IMAGE_HEADER_PROBE_BYTES = 64 +// Growth per miss. JPEG SOF sits past however much EXIF/ICC/MPF the camera wrote, so the probe +// widens geometrically: total decoded stays within ~1.07x of the bytes the header actually needed. +const RASTER_IMAGE_HEADER_PROBE_GROWTH = 16 + /** * Whether the encoded dimensions are known to exceed the preview limits. * @@ -135,10 +152,34 @@ export function exceedsRasterImagePreviewLimits( if (!isKnownRasterImageMimeType(mimeType)) { return false } - const prefix = decodeBase64Prefix(content, RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES) - if (!prefix) { - return false + let probeBytes = RASTER_IMAGE_HEADER_PROBE_BYTES + for (;;) { + const prefix = decodeBase64Prefix(content, probeBytes) + // A short probe only ever fails where the whole payload would: it walks a strict prefix of the + // same characters through the same state machine. + if (!prefix) { + return false + } + // Shorter than asked for means the payload ran out, so a wider probe cannot add bytes. + const exhausted = + prefix.byteLength < probeBytes || probeBytes >= RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES + const dimensions = readRasterImageDimensions(prefix) + if (dimensions !== null) { + const withinLimits = isRasterImagePreviewDimensions(dimensions) + if (withinLimits || exhausted) { + return !withinLimits + } + // About to suppress: redo the decode over the whole payload so the verdict stays the one the + // full read gives, including its rejection of base64 that turns invalid past the header. + probeBytes = RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES + continue + } + if (exhausted) { + return false + } + probeBytes = Math.min( + probeBytes * RASTER_IMAGE_HEADER_PROBE_GROWTH, + RASTER_IMAGE_PREVIEW_HEADER_MAX_BYTES + ) } - const dimensions = readRasterImageDimensions(prefix) - return dimensions !== null && !isRasterImagePreviewDimensions(dimensions) } diff --git a/src/shared/repo-icon.test.ts b/src/shared/repo-icon.test.ts index 566d8209e52..fde215bf943 100644 --- a/src/shared/repo-icon.test.ts +++ b/src/shared/repo-icon.test.ts @@ -1,6 +1,22 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' +import type * as ImageDataUriModule from './image-data-uri' import { githubAvatarIcon, githubAvatarSlug, sanitizeRepoIcon } from './repo-icon' +// Why a module mock: `validateRasterImageDataUri` is the leaf that base64-decodes an inline icon's +// header, so counting its invocations is the direct measure of what re-hydrating a repo costs. +const { dataUriValidations } = vi.hoisted(() => ({ dataUriValidations: { count: 0 } })) + +vi.mock('./image-data-uri', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + validateRasterImageDataUri: (dataUri: string) => { + dataUriValidations.count += 1 + return actual.validateRasterImageDataUri(dataUri) + } + } +}) + const PNG_1X1_BASE64 = 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+/p9sAAAAASUVORK5CYII=' const WEBP_1X1_BASE64 = 'UklGRhoAAABXRUJQVlA4IA4AAAAwAQCdASoBAAEAAQIlSkwAAA==' @@ -188,3 +204,84 @@ describe('githubAvatarSlug', () => { expect(githubAvatarSlug(null, undefined)).toBeNull() }) }) + +describe('repo icon source validation memo', () => { + const HYDRATIONS = 25 + + function uploadIcon(width: number): { type: 'image'; src: string; source: 'upload' } { + return { type: 'image', src: `data:image/png;base64,${pngBase64(width, 1)}`, source: 'upload' } + } + + it('validates each distinct icon src once across repeated hydrations', () => { + const icons = [uploadIcon(2), uploadIcon(3), uploadIcon(4)] + // Warm the memo the way the first hydration would, then measure steady state. + for (const icon of icons) { + sanitizeRepoIcon(icon) + } + dataUriValidations.count = 0 + + for (let hydration = 0; hydration < HYDRATIONS; hydration += 1) { + for (const icon of icons) { + expect(sanitizeRepoIcon(icon)).toEqual(icon) + } + } + + // Unmemoized this is HYDRATIONS x icons full base64 header decodes; memoized an unchanged + // persisted icon costs nothing. + expect(dataUriValidations.count).toBe(0) + }) + + it('re-validates as soon as the src changes', () => { + dataUriValidations.count = 0 + expect(sanitizeRepoIcon(uploadIcon(11))).toEqual(uploadIcon(11)) + expect(sanitizeRepoIcon(uploadIcon(12))).toEqual(uploadIcon(12)) + expect(dataUriValidations.count).toBe(2) + }) + + it('keeps the verdict specific to the icon source', () => { + const src = `data:image/webp;base64,${WEBP_1X1_BASE64}` + expect(sanitizeRepoIcon({ type: 'image', src, source: 'file' })).toEqual({ + type: 'image', + src, + source: 'file' + }) + // WebP is a `file` icon only; sharing one cache across sources would accept it as an upload. + expect(sanitizeRepoIcon({ type: 'image', src, source: 'upload' })).toBeUndefined() + }) + + // Guard for the removed cap: the memo hangs off the persisted icon object, so it holds a verdict + // for every live icon no matter how many there are. A fixed-size map would evict the earliest + // entries here and re-decode them on the next hydration. + it('keeps a verdict for every live icon, however many repos have one', () => { + const LIVE_ICONS = 200 + const icons = Array.from({ length: LIVE_ICONS }, (_, index) => uploadIcon(1000 + index)) + for (const icon of icons) { + sanitizeRepoIcon(icon) + } + dataUriValidations.count = 0 + + for (const icon of icons) { + expect(sanitizeRepoIcon(icon)).toEqual(icon) + } + + expect(dataUriValidations.count).toBe(0) + }) + + // Guard for the hazard object keying introduces: the stored src/source are re-checked on a hit, + // so a persisted icon edited in place can never be served its previous verdict. + it('re-validates an icon object whose src or source is mutated in place', () => { + const icon = { type: 'image', src: `data:image/png;base64,${pngBase64(7, 1)}`, source: 'file' } + expect(sanitizeRepoIcon(icon)).toEqual(icon) + + icon.src = `data:image/png;base64,${pngBase64(8, 1)}` + dataUriValidations.count = 0 + expect(sanitizeRepoIcon(icon)).toEqual(icon) + expect(dataUriValidations.count).toBe(1) + + // WebP is a `file` icon but not an `upload` icon, so the same object must flip verdicts. + icon.src = `data:image/webp;base64,${WEBP_1X1_BASE64}` + expect(sanitizeRepoIcon(icon)).toEqual(icon) + icon.source = 'upload' + expect(sanitizeRepoIcon(icon)).toBeUndefined() + }) +}) diff --git a/src/shared/repo-icon.ts b/src/shared/repo-icon.ts index 1973a2227ac..35f57a8f191 100644 --- a/src/shared/repo-icon.ts +++ b/src/shared/repo-icon.ts @@ -79,7 +79,7 @@ function normalizeGitHubAvatarHost(rawHost?: string): string { } } -function isSupportedImageSrc(src: string, source: RepoIconImageSource): boolean { +function computeIsSupportedImageSrc(src: string, source: RepoIconImageSource): boolean { if (source === 'upload') { return ( /^data:image\/png;base64,[A-Za-z0-9+/=\s]+$/i.test(src) && @@ -112,6 +112,34 @@ function isSupportedImageSrc(src: string, source: RepoIconImageSource): boolean return url.hostname === 'www.google.com' && url.pathname === '/s2/favicons' } +type ImageSrcVerdict = { src: unknown; source: unknown; supported: boolean } + +/** + * Why: `getRepos()` re-hydrates every repo on every call, and validating one inline data URI means + * scanning a 400 KB string twice with a regex and base64-decoding its header. `hydrateRepo` is + * handed the *same* persisted `repoIcon` object every time, so the verdict is cached on that object + * and dies with it — no cap, no eviction, and nothing retained once a repo or an icon is replaced. + * + * `src`/`source` are re-checked on a hit, so mutating the persisted icon in place cannot serve a + * stale verdict. Both are the identical string references in the steady state, so the compare is a + * pointer check, not a 400 KB scan. + */ +const imageSrcVerdicts = new WeakMap() + +function isSupportedImageSrc( + candidate: Record, + src: string, + source: RepoIconImageSource +): boolean { + const cached = imageSrcVerdicts.get(candidate) + if (cached && cached.src === candidate.src && cached.source === candidate.source) { + return cached.supported + } + const supported = computeIsSupportedImageSrc(src, source) + imageSrcVerdicts.set(candidate, { src: candidate.src, source: candidate.source, supported }) + return supported +} + export function sanitizeRepoIcon(value: unknown): RepoIcon | null | undefined { if (value === undefined) { return undefined @@ -146,7 +174,7 @@ export function sanitizeRepoIcon(value: unknown): RepoIcon | null | undefined { if (!isRepoIconImageSource(source) || src.length > MAX_REPO_ICON_DATA_URL_LENGTH) { return undefined } - if (!isSupportedImageSrc(src, source)) { + if (!isSupportedImageSrc(candidate, src, source)) { return undefined } const label = typeof candidate.label === 'string' ? candidate.label.trim().slice(0, 80) : '' From ab32c2c0c58d30822f03d18b00db79e64eeb17a6 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:57:57 -0700 Subject: [PATCH 228/398] perf(startup): stop the persistence milestone from timing its own details closure (#18439) `logPersistenceStartupMilestone` resolved the lazy `details` closure before reading `performance.now()`, so the 1.6 MB `JSON.stringify` that `persistence-load-done` uses to report `workspaceSessionBytes` was billed to the milestone it measures. Snapshot `t` first. Diagnostics output is unchanged; only the recorded timestamp moves. --- ...rsistence-loading-store-extraction.test.ts | 41 +++++++++++++++++++ .../loading-store/loaded-state-parsing.ts | 4 +- 2 files changed, 44 insertions(+), 1 deletion(-) diff --git a/src/main/persistence-loading-store-extraction.test.ts b/src/main/persistence-loading-store-extraction.test.ts index e48971d6c5c..a3c5e248909 100644 --- a/src/main/persistence-loading-store-extraction.test.ts +++ b/src/main/persistence-loading-store-extraction.test.ts @@ -116,6 +116,47 @@ describe('loading Store extraction seams', () => { }) }) + it('timestamps persistence-load-done before resolving its details closure', () => { + const sentinel = 'startup-diagnostics-workspace-session-sentinel-ordering' + vi.stubEnv('ORCA_STARTUP_DIAGNOSTICS', '1') + const state = getDefaultPersistedState(testState.dir) + state.workspaceSession = { ...state.workspaceSession, activeTabId: sentinel } + writeDataFile(state) + + // Fake clock only the details closure advances, so a post-closure timestamp is unambiguous. + let clock = 0 + const nowSpy = vi.spyOn(performance, 'now').mockImplementation(() => clock) + const realStringify = JSON.stringify + const stringifySpy = vi.spyOn(JSON, 'stringify').mockImplementation((( + value: unknown, + ...rest: unknown[] + ) => { + if ( + value && + typeof value === 'object' && + (value as { activeTabId?: unknown }).activeTabId === sentinel + ) { + clock += 1000 + } + return (realStringify as (...args: unknown[]) => string)(value, ...rest) + }) as typeof JSON.stringify) + + try { + const store = createStore() + store.freezeWrites() + } finally { + stringifySpy.mockRestore() + nowSpy.mockRestore() + } + + const loadDoneCall = logStartupDiagnosticMock.mock.calls.find( + ([event]) => event === 'persistence-load-done' + ) + const details = loadDoneCall?.[1] as Record | undefined + expect(details?.workspaceSessionBytes).toEqual(expect.any(Number)) + expect(details?.t).toBe(0) + }) + it('accepts the first JSON-parseable backup even when an older backup has richer state', async () => { mkdirSync(testState.dir, { recursive: true }) writeFileSync(dataFile(), '{{corrupt-primary', 'utf-8') diff --git a/src/main/persistence/loading-store/loaded-state-parsing.ts b/src/main/persistence/loading-store/loaded-state-parsing.ts index 5bd6e572ab1..e4034fa2b17 100644 --- a/src/main/persistence/loading-store/loaded-state-parsing.ts +++ b/src/main/persistence/loading-store/loaded-state-parsing.ts @@ -51,8 +51,10 @@ function logPersistenceStartupMilestone( if (!isStartupDiagnosticsEnabled()) { return } + // Why: snapshot `t` before resolving lazy details — otherwise an expensive details closure is billed to the milestone it measures. + const t = Math.round(performance.now()) const resolvedDetails = typeof details === 'function' ? details() : details - logStartupDiagnostic(event, { t: Math.round(performance.now()), ...resolvedDetails }) + logStartupDiagnostic(event, { t, ...resolvedDetails }) } import type { StoreRuntimeState } from './store-runtime-state' From c11c6878c1ee712f893a02a276ca3ec5e52f7aeb Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 20:59:53 -0700 Subject: [PATCH 229/398] perf(persistence): stop dead SSH leases pinning metadata, retire unreachable tombstones (#18430) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two unbounded-growth fixes in the persisted profile, which is re-serialised in full on every save. `collectPersistedWorkspaceOwners` registered every SSH lease's worktreeId as a live persisted owner with no state filter, so a route-retired lease — the operator-close `terminated` tombstone, or an `expired` row already marked `supersededBy`/`relayIdRecycled` — pinned its worktree's metadata row permanently. The prune gate's own doc names that failure: "Rows pinned by a persisted session are never removable, so the repetition cannot even make progress." Reuses `sshRemotePtyLeaseAllowsReattach`, the predicate that already decides which leases still name a route. `sshRemotePtyLeases` had no pruning path at all: removal happens in three explicit places, none age- or state-based, so `terminated` rows accumulated forever (137 rows / 54 KB on the reported profile, ~38/day from one target). Marking a lease `terminated` scrubs its pane bindings in the same write, so once no persisted binding names the id the row routes nothing — reattach, pane recovery, the orphan sweep, `ssh:reset` and `ssh:terminateSessions` all behave identically on an absent row. Delete it then, gated on that reachability check because a lease freezes its tabId and the tab-qualified scrub cannot reach a pane that was detached into a new tab. `expired` rows are deliberately untouched, superseded ones included: `sweepOrphanedRelayPtys` reads those ids as its leave-alone list, so dropping one would authorize stopping a remote shell that supersession left running on purpose (docs/reference/ssh-execution-boundary.md). --- ...istence-ssh-lease-reattach-reclaim.test.ts | 6 +- ...ence-ssh-lease-tombstone-retention.test.ts | 147 ++++++++++++++++++ .../persistence-ssh-remote-pty-leases.test.ts | 24 +-- .../ssh-pty-lease-operations.ts | 17 +- .../ssh-pty-lease-tombstone-retention.ts | 89 +++++++++++ ...ng-local-worktree-metadata-pruning.test.ts | 58 +++++++ ...missing-local-worktree-metadata-pruning.ts | 8 + 7 files changed, 324 insertions(+), 25 deletions(-) create mode 100644 src/main/persistence-ssh-lease-tombstone-retention.test.ts create mode 100644 src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts diff --git a/src/main/persistence-ssh-lease-reattach-reclaim.test.ts b/src/main/persistence-ssh-lease-reattach-reclaim.test.ts index 9d6c74f710e..b2088fa616b 100644 --- a/src/main/persistence-ssh-lease-reattach-reclaim.test.ts +++ b/src/main/persistence-ssh-lease-reattach-reclaim.test.ts @@ -49,14 +49,16 @@ describe('ssh remote pty lease reclaim after a proven reattach', () => { expect(sshRemotePtyLeaseAllowsReattach(lease)).toBe(true) }) - it('leaves a terminated lease absorbing even when the id appears in a reattach batch', async () => { + it('never lets a reattach batch revive an operator-closed id', async () => { const store = await createStore() store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId: 'pty-1', state: 'attached' }) store.markSshRemotePtyLease('ssh-1', 'pty-1', 'terminated') await store.markSshRemotePtyLeasesAttachedAsync('ssh-1', ['pty-1']) - expect(store.getSshRemotePtyLeases('ssh-1')[0]).toMatchObject({ state: 'terminated' }) + // The unbound tombstone is retired at close, and the batch only ever updates existing rows — + // so the id stays out of the reattach set either way. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) }) it('does not revive an expired lease from an unqualified bulk attach', async () => { diff --git a/src/main/persistence-ssh-lease-tombstone-retention.test.ts b/src/main/persistence-ssh-lease-tombstone-retention.test.ts new file mode 100644 index 00000000000..feafdd766e4 --- /dev/null +++ b/src/main/persistence-ssh-lease-tombstone-retention.test.ts @@ -0,0 +1,147 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { mkdtempSync, rmSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { createStore, testState } from './persistence-test-harness' +import { TEST_LEAF_1 } from './persistence-session-fixtures' + +vi.mock('electron', () => ({ + app: { getPath: () => testState.dir }, + safeStorage: { isEncryptionAvailable: () => false } +})) + +vi.mock('./telemetry/client', () => ({ track: vi.fn() })) +vi.mock('./telemetry/cohort-classifier', () => ({ getCohortAtEmit: () => ({}) })) + +describe('operator-closed SSH lease tombstones', () => { + beforeEach(() => { + testState.dir = mkdtempSync(join(tmpdir(), 'orca-test-')) + }) + + afterEach(() => { + rmSync(testState.dir, { recursive: true, force: true }) + }) + + /** A pane whose lease froze `tab-old` before `detachTerminalPaneToTab` moved it to `tab-new`. + * The binding scrub matches tab-qualified, so it cannot reach this row's binding. */ + async function storeWithDetachedPaneBinding(): Promise>> { + const store = await createStore() + store.upsertSshRemotePtyLease({ + targetId: 'ssh-1', + ptyId: 'remote-pty', + worktreeId: 'wt1', + tabId: 'tab-old', + leafId: TEST_LEAF_1, + state: 'attached' + }) + store.setWorkspaceSession({ + activeRepoId: 'r1', + activeWorktreeId: 'wt1', + activeTabId: 'tab-new', + tabsByWorktree: { + wt1: [ + { + id: 'tab-new', + worktreeId: 'wt1', + title: 'Terminal', + customTitle: null, + color: null, + sortOrder: 0, + createdAt: 1, + ptyId: null + } + ] + }, + terminalLayoutsByTabId: { + 'tab-new': { + root: { type: 'leaf', leafId: TEST_LEAF_1 }, + activeLeafId: TEST_LEAF_1, + expandedLeafId: null, + ptyIdsByLeafId: { [TEST_LEAF_1]: 'ssh:ssh-1@@remote-pty' } + } + } + }) + return store + } + + it('keeps the tombstone while a binding the scrub could not reach still names the pty', async () => { + const store = await storeWithDetachedPaneBinding() + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') + + // `isRestorablePtyBinding` still consults this row to refuse replaying that binding. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ + expect.objectContaining({ ptyId: 'remote-pty', state: 'terminated' }) + ]) + expect(store.getWorkspaceSession().terminalLayoutsByTabId['tab-new'].ptyIdsByLeafId).toEqual({ + [TEST_LEAF_1]: 'ssh:ssh-1@@remote-pty' + }) + }) + + it('keeps an operator-closed lease that still owes an undelivered stop', async () => { + const store = await createStore() + store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId: 'remote-pty', state: 'attached' }) + store.recordSshRemotePtyKillIntent('ssh-1', 'remote-pty', { + incarnationId: 'inc-1', + requestedAt: 1, + attempts: 0 + }) + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') + + expect(store.getSshRemotePtyKillIntents('ssh-1', 2)).toHaveLength(1) + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ + expect.objectContaining({ ptyId: 'remote-pty', state: 'terminated' }) + ]) + }) + + // `expired` is never evidence the shell died, and `sweepOrphanedRelayPtys` reads these ids as its + // leave-alone list, so dropping one would authorize stopping a process left running on purpose. + it('keeps a superseded expired lease when a sibling pane is closed', async () => { + const store = await createStore() + store.upsertSshRemotePtyLease({ + targetId: 'ssh-1', + ptyId: 'remote-pty-1', + worktreeId: 'wt1', + tabId: 'tab1', + leafId: TEST_LEAF_1, + state: 'attached' + }) + store.upsertSshRemotePtyLease({ + targetId: 'ssh-1', + ptyId: 'remote-pty-2', + worktreeId: 'wt1', + tabId: 'tab1', + leafId: TEST_LEAF_1, + state: 'attached' + }) + store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId: 'remote-pty-3', state: 'attached' }) + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty-3', 'terminated') + + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ + expect.objectContaining({ + ptyId: 'remote-pty-1', + state: 'expired', + supersededBy: 'remote-pty-2' + }), + expect.objectContaining({ ptyId: 'remote-pty-2', state: 'attached' }) + ]) + }) + + it('retires every unreachable tombstone for the target, not only the one just closed', async () => { + const store = await createStore() + for (const ptyId of ['remote-pty-1', 'remote-pty-2', 'remote-pty-3']) { + store.upsertSshRemotePtyLease({ targetId: 'ssh-1', ptyId, state: 'terminated' }) + } + store.upsertSshRemotePtyLease({ targetId: 'ssh-2', ptyId: 'other-pty', state: 'terminated' }) + expect(store.getSshRemotePtyLeases()).toHaveLength(4) + + store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty-1', 'terminated') + + // Other targets are untouched: the pass is scoped to the one whose bindings were just scrubbed. + expect(store.getSshRemotePtyLeases()).toEqual([ + expect.objectContaining({ targetId: 'ssh-2', ptyId: 'other-pty' }) + ]) + }) +}) diff --git a/src/main/persistence-ssh-remote-pty-leases.test.ts b/src/main/persistence-ssh-remote-pty-leases.test.ts index 464138f69d4..d4cdc79285c 100644 --- a/src/main/persistence-ssh-remote-pty-leases.test.ts +++ b/src/main/persistence-ssh-remote-pty-leases.test.ts @@ -526,12 +526,9 @@ describe('Store', () => { store.markSshRemotePtyLeases('ssh-1', 'terminated') const session = store.getWorkspaceSession() - expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ - expect.objectContaining({ - ptyId: 'remote-pty', - state: 'terminated' - }) - ]) + // The scrub is what retires the row: with no binding left naming the id, the tombstone routes + // nothing and is dropped in the same write. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) expect(session.tabsByWorktree.wt1[0].ptyId).toBeNull() expect(session.terminalLayoutsByTabId.tab1.ptyIdsByLeafId).toEqual({}) }) @@ -622,12 +619,8 @@ describe('Store', () => { store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') - expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ - expect.objectContaining({ - ptyId: 'remote-pty', - state: 'terminated' - }) - ]) + // An unresolved id would have left the lease `attached`; this unbound row is retired instead. + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) }) // `expired` never means the shell exited — every writer records that the CLIENT lost its route @@ -658,12 +651,7 @@ describe('Store', () => { store.markSshRemotePtyLease('ssh-1', 'ssh:ssh-1@@remote-pty', 'terminated') const session = store.getWorkspaceSession() - expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([ - expect.objectContaining({ - ptyId: 'remote-pty', - state: 'terminated' - }) - ]) + expect(store.getSshRemotePtyLeases('ssh-1')).toEqual([]) expect(session.tabsByWorktree.wt1[0].ptyId).toBeNull() expect(session.terminalLayoutsByTabId.tab1.ptyIdsByLeafId).toEqual({}) }) diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts index f0628cb25e5..f06e5b0ece7 100644 --- a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-operations.ts @@ -2,6 +2,7 @@ import type { PersistedState } from '../../../shared/persisted-state-types' import type { SshRemotePtyLease } from '../../../shared/ssh-types' import { isTerminalLeafId } from '../../../shared/stable-pane-id' import { invalidateLocalWorktreeMetadataPruneInputs } from '../../local-worktree-metadata-prune-gate' +import { pruneRetiredSshRemotePtyLeaseTombstones } from './ssh-pty-lease-tombstone-retention' import { supersedeSiblingLeasesForPane } from './ssh-pty-pane-supersession' export type SshPtyLeaseOperations = { @@ -145,7 +146,11 @@ function updateSshRemotePtyLeaseStates( const bindingsChanged = shouldClearBindings ? operations.clearBindingsForLeases(targetId, leasesToClear) : false - return changed || bindingsChanged + // Why after the scrub: it is the scrub that makes the tombstones unreachable. + const tombstonesPruned = shouldClearBindings + ? pruneRetiredSshRemotePtyLeaseTombstones(operations, targetId) + : false + return changed || bindingsChanged || tombstonesPruned } export function markSshRemotePtyLeases( @@ -215,10 +220,11 @@ export function markSshRemotePtyLease( } const shouldClearBindings = leaseStateWithdrawsBinding(state) if (lease.state === state) { - if ( - (shouldClearBindings && operations.clearBindingsForLeases(targetId, [lease])) || - recycledChanged - ) { + const bindingsCleared = + shouldClearBindings && operations.clearBindingsForLeases(targetId, [lease]) + const tombstonesPruned = + shouldClearBindings && pruneRetiredSshRemotePtyLeaseTombstones(operations, targetId) + if (bindingsCleared || tombstonesPruned || recycledChanged) { operations.flush() } return @@ -233,6 +239,7 @@ export function markSshRemotePtyLease( } if (shouldClearBindings) { operations.clearBindingsForLeases(targetId, [lease]) + pruneRetiredSshRemotePtyLeaseTombstones(operations, targetId) } operations.flush() } diff --git a/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts new file mode 100644 index 00000000000..261162a3825 --- /dev/null +++ b/src/main/persistence/leasing-ssh-ptys/ssh-pty-lease-tombstone-retention.ts @@ -0,0 +1,89 @@ +import type { PersistedState } from '../../../shared/persisted-state-types' +import type { SshRemotePtyLease } from '../../../shared/ssh-types' + +export type SshPtyLeaseTombstoneRetentionOperations = { + state: PersistedState + toComparablePtyId: (targetId: string, ptyId: string) => string +} + +/** A routing tombstone with nothing left to route: the operator closed this PTY and no stop is + * still owed for it. `expired` is deliberately not here — it says only that the CLIENT lost its + * route (docs/reference/ssh-execution-boundary.md), and `sweepOrphanedRelayPtys` reads those ids + * as its leave-alone list, so deleting one would authorize stopping a remote shell that + * supersession left running on purpose. */ +function isRetiredRoutingTombstone(lease: SshRemotePtyLease, targetId: string): boolean { + return ( + lease.targetId === targetId && lease.state === 'terminated' && lease.pendingKill === undefined + ) +} + +/** Every stored-form relay pty id some persisted pane binding still names for this target. + * + * Reads all partitions, not only the two `clearSshRemotePtyBindingsForLeases` scrubs: this answer + * authorizes a delete, so a partition left unscanned would be a binding whose tombstone we dropped. + */ +function boundRelayPtyIds( + operations: SshPtyLeaseTombstoneRetentionOperations, + targetId: string +): Set { + const bound = new Set() + const sessions = [ + operations.state.workspaceSession, + ...Object.values(operations.state.workspaceSessionsByHostId ?? {}) + ] + for (const session of sessions) { + if (!session) { + continue + } + for (const tabs of Object.values(session.tabsByWorktree ?? {})) { + for (const tab of tabs) { + if (tab.ptyId) { + bound.add(operations.toComparablePtyId(targetId, tab.ptyId)) + } + } + } + for (const layout of Object.values(session.terminalLayoutsByTabId ?? {})) { + for (const ptyId of Object.values(layout?.ptyIdsByLeafId ?? {})) { + bound.add(operations.toComparablePtyId(targetId, ptyId)) + } + } + } + return bound +} + +/** + * Deletes the `terminated` rows nothing can reach, bounding an array that otherwise only grew. + * + * `terminated` is written with a binding scrub in the same call, so once no persisted binding names + * the id the row answers no question any reader asks. Reattach refuses it + * (`sshRemotePtyLeaseAllowsReattach`), pane recovery matches on `expired` only, the orphan sweep + * already classes it neither routed nor expired, and `ssh:reset` / `ssh:terminateSessions` skip it + * outright — every one of those behaves identically on an absent row. The one reader that can still + * observe it is `isRestorablePtyBinding`, and only through a binding whose pty id matches, which is + * exactly what the reachability test rules out. A `pendingKill` is an undelivered stop, so those + * rows stay until the replay retires them. + * + * The reachability test is not redundant with the scrub: a lease freezes its `tabId`, so a pane + * broken out into a new tab leaves a binding the scrub's tab-qualified match no longer reaches. + * + * Does not re-arm the local-worktree-metadata prune gate: a `terminated` lease no longer counts as + * a persisted workspace owner, so dropping one cannot make any metadata row more removable. + */ +export function pruneRetiredSshRemotePtyLeaseTombstones( + operations: SshPtyLeaseTombstoneRetentionOperations, + targetId: string +): boolean { + const leases = operations.state.sshRemotePtyLeases ?? [] + if (!leases.some((lease) => isRetiredRoutingTombstone(lease, targetId))) { + return false + } + const bound = boundRelayPtyIds(operations, targetId) + const retained = leases.filter( + (lease) => !isRetiredRoutingTombstone(lease, targetId) || bound.has(lease.ptyId) + ) + if (retained.length === leases.length) { + return false + } + operations.state.sshRemotePtyLeases = retained + return true +} diff --git a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts index cdad14034e2..0588a2f6dae 100644 --- a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts +++ b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.test.ts @@ -287,6 +287,64 @@ describe('pruneSessionlessMissingLocalWorktreeMetadataForRepo', () => { } }) + // A route-retired lease is a tombstone, not a claim: counting one pinned its worktree's metadata + // row for good, so the prune could never make progress on it (#17775). + it('does not let route-retired SSH leases pin a metadata row', () => { + const state = makeState() + const liveIds = Array.from({ length: 3 }, (_, i) => `${REPO_ID}::/workspace/live-${i}`) + const terminatedIds = Array.from({ length: 5 }, (_, i) => `${REPO_ID}::/workspace/closed-${i}`) + const supersededIds = Array.from({ length: 4 }, (_, i) => `${REPO_ID}::/workspace/lost-${i}`) + const recycledIds = [`${REPO_ID}::/workspace/recycled`] + const allIds = [...liveIds, ...terminatedIds, ...supersededIds, ...recycledIds] + for (const worktreeId of allIds) { + state.worktreeMeta[worktreeId] = makeMeta(worktreeId) + } + const lease = (worktreeId: string, index: number, extra: object) => ({ + targetId: 'builder', + ptyId: `pty-${index}`, + worktreeId, + createdAt: 1, + updatedAt: 1, + ...extra + }) + state.sshRemotePtyLeases = [ + ...liveIds.map((id, i) => lease(id, i, { state: 'detached' })), + ...terminatedIds.map((id, i) => lease(id, 100 + i, { state: 'terminated' })), + ...supersededIds.map((id, i) => + lease(id, 200 + i, { state: 'expired', supersededBy: 'pty-9' }) + ), + ...recycledIds.map((id, i) => lease(id, 300 + i, { state: 'expired', relayIdRecycled: true })) + ] as never + + const scan = capture(state) + + expect(pruneCaptured(state, scan, allIds).sort()).toEqual( + [...terminatedIds, ...supersededIds, ...recycledIds].sort() + ) + expect(Object.keys(state.worktreeMeta).sort()).toEqual([...liveIds].sort()) + }) + + // A plain `expired` lease says only that the CLIENT lost its route, so its pane is still + // recoverable and its metadata row is still owned (docs/reference/ssh-execution-boundary.md). + it('keeps a metadata row pinned by an unmarked expired lease', () => { + const state = makeState() + const worktreeId = `${REPO_ID}::/workspace/orphaned` + state.worktreeMeta[worktreeId] = makeMeta(worktreeId) + const scan = capture(state) + state.sshRemotePtyLeases = [ + { + targetId: 'builder', + ptyId: 'pty', + worktreeId, + state: 'expired', + createdAt: 1, + updatedAt: 1 + } + ] + + expect(pruneCaptured(state, scan, [worktreeId])).toEqual([]) + }) + it('preserves canonically equivalent session and top-level owners', () => { const candidateId = `${REPO_ID}::/workspace/Café`.normalize('NFC') const ownerId = candidateId.normalize('NFD') diff --git a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts index 0e03a560a17..db24c25bcf4 100644 --- a/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts +++ b/src/main/persistence/tracking-repos/missing-local-worktree-metadata-pruning.ts @@ -2,6 +2,7 @@ import { isWindowsAbsolutePathLike } from '../../../shared/cross-platform-path' import { getRepoExecutionHostId, LOCAL_EXECUTION_HOST_ID } from '../../../shared/execution-host' import type { PersistedState } from '../../../shared/persisted-state-types' import { getRepoKind } from '../../../shared/repo-kind' +import { sshRemotePtyLeaseAllowsReattach } from '../../../shared/ssh-types' import { worktreeWorkspaceKey } from '../../../shared/workspace-scope' import { FOLDER_WORKSPACE_INSTANCE_SEPARATOR, splitWorktreeId } from '../../../shared/worktree/id' import { isWslUncPath } from '../../../shared/wsl-paths' @@ -40,6 +41,13 @@ function collectPersistedWorkspaceOwners( } } for (const lease of state.sshRemotePtyLeases) { + // A lease that can never be reattached is a routing tombstone, not a claim on a workspace: + // `terminated` is the operator close, and an `expired` row marked `supersededBy` / + // `relayIdRecycled` already lost its pane to a newer lease. Counting them as owners pinned + // their worktree's metadata row permanently, so the prune could never make progress (#17775). + if (!sshRemotePtyLeaseAllowsReattach(lease)) { + continue + } add(lease.worktreeId) } for (const entry of state.migrationUnsupportedPtyEntries) { From 949c9d3353b99ce327e83d344b22f8684b9e8c1a Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:02:11 -0700 Subject: [PATCH 230/398] perf(worktrees): classify each worktree once, defer the SSH meta index, drop the conflict-path probe (#18433) * perf(worktrees): classify each worktree once, defer the SSH meta index, unserialise conflict probes Three redundancies on the worktree-catalog and git-status read paths: - buildDetectedGitWorktrees ran mergeWorktree + toDetectedWorktree twice for every visible row. Discovery backfill returns the same meta object when it wrote nothing, and both builders are pure over it, so skip the second pass on identity. - The SSH worktree-meta index parsed every worktree id on the host, then threw it away whenever the provider was connected. Build it lazily, memoised. - Unmerged `u` records were resolved one fs.access at a time. Resolve the prefix the cap can reach with 8-way concurrency, keyed by record index so Git's output order and error precedence are unchanged. * perf(git): read the porcelain worktree mode instead of probing conflicted paths Every porcelain-v2 `u` record already carries `mW`, the working-tree mode Git stat'ed for that row: `000000` means the conflicted path is absent. Reading it replaces the per-conflict `fs.access`, so the bounded-concurrency resolver, its `= 8` cap, and the order/error-precedence invariant are unnecessary rather than cheaper. `access()` stays only as a fallback for a malformed `mW`, so `parseUnmergedEntry` keeps its signature and neither status-read.ts nor the relay loop changes. Also corrects two fixtures that encoded `mW=100644` for a file that does not exist, which real Git never emits. * fix(test): import the conflict parser statically so the CJS cli project compiles --- src/main/git/status.test.ts | 31 ++- ...tected-provider-listing-meta-index.test.ts | 131 ++++++++++ .../listing/detected-provider-listing.ts | 15 +- .../detected-worktree-classification.test.ts | 234 ++++++++++++++++++ .../listing/ssh-worktree-fallback.ts | 15 +- .../listing/worktree-discovery-metadata.ts | 3 +- src/relay/git-porcelain-local-parity.test.ts | 3 +- .../git-status-conflict-entries.test.ts | 87 +++++++ src/shared/git-status-conflict-entries.ts | 25 +- 9 files changed, 514 insertions(+), 30 deletions(-) create mode 100644 src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts create mode 100644 src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts create mode 100644 src/shared/git-status-conflict-entries.test.ts diff --git a/src/main/git/status.test.ts b/src/main/git/status.test.ts index 4675e3a57a2..7b73739cd3f 100644 --- a/src/main/git/status.test.ts +++ b/src/main/git/status.test.ts @@ -77,6 +77,13 @@ describe('getStatus', () => { gitExecFileAsyncMock.mockResolvedValue({ stdout: '' }) }) + /** `access` targets outside the git dir — i.e. working-tree probes, not conflict-marker reads. */ + function conflictFileProbes(): string[] { + return accessMock.mock.calls + .map(([target]) => String(target).replaceAll('\\', '/')) + .filter((target) => !target.includes('/.git/')) + } + it('parses unmerged porcelain v2 entries into unresolved conflict rows', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') accessMock.mockImplementation(async (target: string) => { @@ -104,11 +111,12 @@ describe('getStatus', () => { ]) }) - it('maps deleted conflicts to deleted when the working tree file is absent', async () => { + // The 7th field of a `u` record is the working-tree mode; `000000` is how Git reports an absent path. + it('maps deleted conflicts to deleted from the porcelain working-tree mode', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: - 'u UD N... 100644 100644 000000 100644 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/deleted.ts\n' + 'u UD N... 100644 100644 000000 000000 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/deleted.ts\n' }) const result = await getStatus('/repo') @@ -120,10 +128,12 @@ describe('getStatus', () => { conflictKind: 'deleted_by_them', conflictStatus: 'unresolved' }) + expect(conflictFileProbes()).toEqual([]) }) - it('falls back to modified when the working-tree probe fails for a non-absence reason', async () => { + it('never re-probes the working tree for a conflict row, whatever the filesystem would say', async () => { readFileMock.mockResolvedValue('gitdir: /repo/.git/worktrees/feature\n') + // Every probe fails ENOENT (beforeEach) or EIO — neither may reach the row's status. accessMock.mockRejectedValue(Object.assign(new Error('EIO'), { code: 'EIO' })) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: @@ -134,19 +144,14 @@ describe('getStatus', () => { expect(result.entries[0]?.status).toBe('modified') expect(result.entries[0]?.conflictKind).toBe('added_by_us') + expect(conflictFileProbes()).toEqual([]) }) // Why both cases normalize separators: git reports the worktree in the WSL guest namespace, and // the assertion is about which path is probed, not which separator this host's `path` emits. - it('probes the conflict working tree through the distro spelling on Windows', async () => { + it('resolves a WSL conflict row without crossing the 9p share', async () => { const platformSpy = vi.spyOn(process, 'platform', 'get').mockReturnValue('win32') readFileMock.mockResolvedValue('gitdir: /home/me/repo/.git/worktrees/feature\n') - accessMock.mockImplementation(async (target: string) => { - if (String(target).endsWith('new.ts')) { - return undefined - } - throw Object.assign(new Error('ENOENT'), { code: 'ENOENT' }) - }) gitExecFileAsyncMock.mockResolvedValueOnce({ stdout: 'u DU N... 100644 100644 100644 100644 aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc src/new.ts\n' @@ -156,10 +161,12 @@ describe('getStatus', () => { const result = await getStatus('/home/me/repo/feature', { wslDistro: 'Ubuntu' }) const probed = accessMock.mock.calls.map(([target]) => String(target).replaceAll('\\', '/')) - expect(probed).toContain('//wsl.localhost/Ubuntu/home/me/repo/feature/src/new.ts') + // No `\\wsl.localhost` round trip per conflict row: the porcelain `mW` field already answered. + expect(probed).not.toContain('//wsl.localhost/Ubuntu/home/me/repo/feature/src/new.ts') + expect(conflictFileProbes()).toEqual([]) expect(result.entries[0]?.status).toBe('modified') expect(result.entries[0]?.conflictKind).toBe('deleted_by_us') - // The conflict-marker probes travel the same way. + // The conflict-marker probes still travel through the distro spelling. expect( probed.filter((target) => target.startsWith('//wsl.localhost/Ubuntu/home/me/repo/.git/worktrees/feature/') diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts b/src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts new file mode 100644 index 00000000000..4bf55e97492 --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-provider-listing-meta-index.test.ts @@ -0,0 +1,131 @@ +/** + * The SSH worktree-meta index is only ever read via `metaIndex.get(repo.id)` on the disconnected + * fallbacks, so a connected listing must not pay `parseWorktreeId` over the whole host snapshot. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { Store } from '../../../persistence/loading-store/store' +import type * as SshWorktreeFallbackModule from './ssh-worktree-fallback' + +const { getSshGitProviderMock, indexBuildSpy } = vi.hoisted(() => ({ + getSshGitProviderMock: vi.fn(), + indexBuildSpy: vi.fn() +})) + +vi.mock('../../../providers/ssh-git-dispatch', () => ({ + getSshGitProvider: getSshGitProviderMock, + requireSshGitProvider: getSshGitProviderMock, + getSshGitProviderGeneration: () => 1 +})) + +vi.mock('./ssh-worktree-fallback', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + // Both builders are counted: the point is that NO index is built on the connected path. + createSshWorktreeMetaIndex: (...args: Parameters) => { + indexBuildSpy('all-hosts', ...args) + return actual.createSshWorktreeMetaIndex(...args) + }, + createSshWorktreeMetaIndexForRepo: ( + ...args: Parameters + ) => { + indexBuildSpy('repo-scoped', ...args) + return actual.createSshWorktreeMetaIndexForRepo(...args) + } + } +}) + +const { listDetectedWorktreesForCapturedRepo } = await import('./detected-provider-listing') + +const repo = { + id: 'repo-1', + path: '/home/user/repo', + displayName: 'repo', + connectionId: 'conn-1' +} as Repo + +const worktreeId = `${repo.id}::/home/user/feature` + +function createStore(): Store { + const rows: Record = { + [worktreeId]: { instanceId: 'instance-1' }, + // Other repos' rows share the host snapshot; only this repo's bucket is ever read back. + 'repo-2::/home/user/other': { instanceId: 'instance-2' } + } + return { + getRepos: () => [repo], + getRepo: () => repo, + getSettings: () => ({}), + getProjectHostSetups: () => [], + getAllWorktreeLineage: () => ({}), + getAllWorktreeMeta: () => rows, + getWorktreeMeta: (id: string) => rows[id], + setWorktreeMeta: vi.fn() + } as unknown as Store +} + +describe('SSH worktree meta index construction', () => { + beforeEach(() => { + indexBuildSpy.mockClear() + getSshGitProviderMock.mockReset() + }) + + it('does not build the index when the provider answers', async () => { + const provider = { + listWorktrees: vi.fn().mockResolvedValue([ + { path: repo.path, head: 'a', branch: 'main', isBare: false, isMainWorktree: true }, + { + path: '/home/user/feature', + head: 'b', + branch: 'feature', + isBare: false, + isMainWorktree: false + } + ]) + } + + const result = await listDetectedWorktreesForCapturedRepo( + createStore(), + repo, + () => true, + provider as never + ) + + expect(result).toMatchObject({ authoritative: true, source: 'git' }) + expect(indexBuildSpy).not.toHaveBeenCalled() + }) + + it('builds the index once when no provider is available', async () => { + const result = await listDetectedWorktreesForCapturedRepo( + createStore(), + repo, + () => true, + undefined + ) + + expect(result).toMatchObject({ authoritative: false, source: 'metadata-fallback' }) + expect(indexBuildSpy).toHaveBeenCalledTimes(1) + expect(indexBuildSpy).toHaveBeenCalledWith('all-hosts', expect.anything()) + expect( + (result as { worktrees: { id: string }[] }).worktrees.map((worktree) => worktree.id) + ).toEqual([worktreeId]) + }) + + it('builds the index once when the provider listing fails', async () => { + const provider = { listWorktrees: vi.fn().mockRejectedValue(new Error('relay down')) } + + const result = await listDetectedWorktreesForCapturedRepo( + createStore(), + repo, + () => true, + provider as never + ) + + expect(result).toMatchObject({ authoritative: false, source: 'metadata-fallback' }) + expect(indexBuildSpy).toHaveBeenCalledTimes(1) + expect( + (result as { worktrees: { id: string }[] }).worktrees.map((worktree) => worktree.id) + ).toEqual([worktreeId]) + }) +}) diff --git a/src/main/ipc/worktrees/listing/detected-provider-listing.ts b/src/main/ipc/worktrees/listing/detected-provider-listing.ts index 51a0618ba66..25c08d262fb 100644 --- a/src/main/ipc/worktrees/listing/detected-provider-listing.ts +++ b/src/main/ipc/worktrees/listing/detected-provider-listing.ts @@ -10,7 +10,8 @@ import type { ListDesktopLineageForHostArgs } from '../../../../shared/host-line import { buildDetectedGitWorktrees, createSshWorktreeMetaIndex, - listDisconnectedSshWorktrees + listDisconnectedSshWorktrees, + type SshWorktreeMetaIndex } from './ssh-worktree-fallback' import { buildDisconnectedDetectedWorktrees, @@ -42,9 +43,11 @@ export async function listDetectedWorktreesForCapturedRepo( const allMeta = isFolderRepo(repo) ? undefined : readAllWorktreeMetaForHost(store, getRepoExecutionHostId(repo)) - const sshWorktreeMetaIndex = repo.connectionId - ? createSshWorktreeMetaIndex(Object.entries(allMeta ?? {})) - : new Map() + // Why: only the disconnected fallbacks read this, so keep parseWorktreeId over the whole host snapshot + // off the connected path entirely. + let cachedSshWorktreeMetaIndex: SshWorktreeMetaIndex | undefined + const sshWorktreeMetaIndex = (): SshWorktreeMetaIndex => + (cachedSshWorktreeMetaIndex ??= createSshWorktreeMetaIndex(Object.entries(allMeta ?? {}))) try { let gitWorktrees: GitWorktreeInfo[] @@ -86,7 +89,7 @@ export async function listDetectedWorktreesForCapturedRepo( if (!isCurrent()) { return null } - const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex) + const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, @@ -158,7 +161,7 @@ export async function listDetectedWorktreesForCapturedRepo( err ) if (repo.connectionId) { - const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex) + const worktrees = listDisconnectedSshWorktrees(store, repo, sshWorktreeMetaIndex()) return { repoId: repo.id, authoritative: false, diff --git a/src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts b/src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts new file mode 100644 index 00000000000..6a25d696c9b --- /dev/null +++ b/src/main/ipc/worktrees/listing/detected-worktree-classification.test.ts @@ -0,0 +1,234 @@ +/** + * Guards the single-classification contract of `buildDetectedGitWorktrees`: every visible worktree + * used to be run through `mergeWorktree` + `toDetectedWorktree` twice per catalog pass. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { Repo } from '../../../../shared/repo-types' +import type { Store } from '../../../persistence/loading-store/store' +import type { WorktreeMeta } from '../../../../shared/worktree/meta-types' +import type { GitWorktreeInfo } from '../../../../shared/worktree/types' +import type * as NodeCryptoModule from 'node:crypto' +import type * as OwnershipModule from '../../../../shared/worktree/ownership' + +const { toDetectedWorktreeSpy } = vi.hoisted(() => ({ toDetectedWorktreeSpy: vi.fn() })) + +vi.mock('../../../../shared/worktree/ownership', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + toDetectedWorktree: (args: Parameters[0]) => { + toDetectedWorktreeSpy(args) + return actual.toDetectedWorktree(args) + } + } +}) + +vi.mock('node:crypto', async (importOriginal) => ({ + ...(await importOriginal()), + randomUUID: () => 'fixed-instance-id' +})) + +const { buildDetectedGitWorktrees } = await import('./ssh-worktree-fallback') +const { getProjectHostSetupWorktreeMeta } = + await import('../../../../shared/project-host-setup-lookup') +const { mergeWorktree } = await import('../../worktree-logic') +const { resolveWorktreeMetaWithDiscoveryBackfill } = await import('./worktree-discovery-metadata') +const ownership = await import('../../../../shared/worktree/ownership') +const { projectResolvedWorktreeLineage } = + await import('../../../../shared/resolved-worktree-lineage') +const { createWorktreeVisibilitySourceMatcher, resolveCustomWorktreeVisibilitySources } = + await import('../../../../shared/worktree/visibility-sources') +const { resolveConfiguredWorktreeBasePaths } = + await import('../../../../shared/worktree/configured-worktree-base-path') +const { dedupeWorktreesByPath } = await import('../../worktree-path-comparison') +const { readWorktreeMetaForHost } = + await import('../../../persistence/host-qualified-worktree-meta') +const { getRepoOwnedWorktreeMeta } = await import('../../../worktree-metadata-ownership') +const { getRepoExecutionHostId } = await import('../../../../shared/execution-host') + +const repo: Repo = { + id: 'repo-1', + path: '/workspace/repo', + displayName: 'repo', + badgeColor: '#000', + addedAt: 0 +} as Repo + +const ownershipMeta = getProjectHostSetupWorktreeMeta([], repo) + +function gitWorktree(path: string): GitWorktreeInfo { + return { + path, + head: 'abc123', + branch: 'refs/heads/feature', + isBare: false, + isMainWorktree: false + } +} + +/** Fully settled metadata: discovery backfill has nothing to write, so it hands the same object back. */ +function settledMeta(overrides: Partial = {}): WorktreeMeta { + return { + ...ownershipMeta, + instanceId: 'instance-settled', + orcaCreatedAt: 1, + lastActivityAt: 5, + ...overrides + } as WorktreeMeta +} + +function createStore(meta: Record, repos: Repo[] = [repo]) { + const rows = { ...meta } + return { + getRepos: () => repos, + getSettings: () => ({ workspaceDir: '/workspace', nestWorkspaces: true }), + getProjectHostSetups: () => [], + getAllWorktreeLineage: () => ({}), + getAllWorktreeMeta: () => rows, + getWorktreeMeta: (id: string) => rows[id], + getWorktreeMetaForHost: (id: string, hostId: string) => + rows[id]?.hostId === hostId ? rows[id] : undefined, + getAllWorktreeMetaForHost: () => rows, + setWorktreeMeta: (id: string, patch: Partial) => { + rows[id] = { ...rows[id], ...patch } as WorktreeMeta + return rows[id] + }, + setWorktreeMetaForHost: (id: string, hostId: string, patch: Partial) => { + rows[id] = { ...rows[id], ...patch, hostId } as WorktreeMeta + return rows[id] + } + } as unknown as Store +} + +/** The pre-change implementation, verbatim, as the equivalence oracle. */ +function buildDetectedGitWorktreesTwoPass( + store: Store, + target: Repo, + gitWorktrees: GitWorktreeInfo[], + allMetaOverride?: Record +) { + const settings = store.getSettings() + const knownOrcaLayouts = ownership.buildKnownOrcaWorkspaceLayouts(settings, target) + const isLegacyRepoForVisibility = ownership.isLegacyRepoForExternalWorktreeVisibility(target) + const liveWorktrees = dedupeWorktreesByPath(gitWorktrees.filter((info) => !info.prunable)) + const worktreeVisibilitySourceMatcher = createWorktreeVisibilitySourceMatcher( + [target.path, ...liveWorktrees.map((worktree) => worktree.path)], + resolveCustomWorktreeVisibilitySources(target, settings.worktreeVisibilityDefaults), + resolveConfiguredWorktreeBasePaths(target) + ) + const allMeta = allMetaOverride ?? store.getAllWorktreeMeta?.() + const repoOwnerCount = store.getRepos().filter((candidate) => candidate.id === target.id).length + const detectedRows = liveWorktrees.map((info) => { + const worktreeId = `${target.id}::${info.path}` + const legacyMeta = store.getWorktreeMeta?.(worktreeId) + const metaById = allMeta ?? (legacyMeta ? { [worktreeId]: legacyMeta } : {}) + let meta = + readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(target)) ?? + getRepoOwnedWorktreeMeta(target, worktreeId, metaById, repoOwnerCount) + const worktree = mergeWorktree(target.id, info, meta, target.displayName) + const detected = ownership.toDetectedWorktree({ + repo: target, + worktree, + meta, + settings, + knownOrcaLayouts, + isLegacyRepoForVisibility, + worktreeVisibilitySourceMatcher + }) + if (!detected.visible) { + return detected + } + meta = resolveWorktreeMetaWithDiscoveryBackfill( + store, + target, + worktreeId, + allMeta, + repoOwnerCount + ) + return ownership.toDetectedWorktree({ + repo: target, + worktree: mergeWorktree(target.id, info, meta, target.displayName), + meta, + settings, + knownOrcaLayouts, + isLegacyRepoForVisibility, + worktreeVisibilitySourceMatcher + }) + }) + return projectResolvedWorktreeLineage(detectedRows, store.getAllWorktreeLineage?.() ?? {}) +} + +describe('buildDetectedGitWorktrees classification passes', () => { + beforeEach(() => { + toDetectedWorktreeSpy.mockClear() + // Discovery backfill stamps lastActivityAt from the clock; freeze it so equivalence is deterministic. + vi.spyOn(Date, 'now').mockReturnValue(1_700_000_000_000) + }) + + it('classifies each visible worktree once per catalog pass, not twice', () => { + const paths = ['/workspace/one', '/workspace/two', '/workspace/three'] + const meta = Object.fromEntries( + paths.map((path) => [`${repo.id}::${path}`, settledMeta({ displayName: path })]) + ) + const store = createStore(meta) + + const detected = buildDetectedGitWorktrees(store, repo, paths.map(gitWorktree), meta) + + expect(detected).toHaveLength(3) + expect(detected.every((row) => row.visible)).toBe(true) + expect(toDetectedWorktreeSpy).toHaveBeenCalledTimes(paths.length) + }) + + it('reads the locator-keyed metadata row only when no host snapshot is available', () => { + const worktreeId = `${repo.id}::/workspace/one` + const meta = { [worktreeId]: settledMeta() } + const store = createStore(meta) + const legacyReads = vi.spyOn(store, 'getWorktreeMeta') + + buildDetectedGitWorktrees(store, repo, [gitWorktree('/workspace/one')], meta) + expect(legacyReads).not.toHaveBeenCalled() + + // Partial stores (compatibility shapes) expose no snapshot, so the locator-keyed lookup must still run. + const partialStore = createStore(meta) as Partial + delete partialStore.getAllWorktreeMeta + delete partialStore.getAllWorktreeMetaForHost + delete partialStore.getWorktreeMetaForHost + const partialLegacyReads = vi.spyOn(partialStore as Store, 'getWorktreeMeta') + + const rows = buildDetectedGitWorktrees( + partialStore as Store, + repo, + [gitWorktree('/workspace/one')], + undefined + ) + expect(partialLegacyReads).toHaveBeenCalledWith(worktreeId) + expect(rows[0]).toMatchObject({ id: worktreeId, lastActivityAt: 5 }) + }) + + it.each([ + ['settled metadata', () => settledMeta()], + ['metadata needing discovery backfill', () => ({ orcaCreatedAt: 1 }) as WorktreeMeta], + ['no metadata at all', () => undefined] + ])('emits a catalog deep-equal to the two-pass build for %s', (_label, makeMeta) => { + const worktreeId = `${repo.id}::/workspace/one` + const seed = makeMeta() + const build = (fn: typeof buildDetectedGitWorktrees) => + fn( + createStore(seed ? { [worktreeId]: seed } : {}), + repo, + [gitWorktree('/workspace/one'), gitWorktree('/workspace/hidden-external')], + seed ? { [worktreeId]: seed } : {} + ) + + expect(build(buildDetectedGitWorktrees)).toEqual(build(buildDetectedGitWorktreesTwoPass)) + }) + + it('emits a catalog deep-equal to the two-pass build for a folder-style listing with no host snapshot', () => { + const worktreeId = `${repo.id}::/workspace/one` + const seed = settledMeta() + const build = (fn: typeof buildDetectedGitWorktrees) => + fn(createStore({ [worktreeId]: seed }), repo, [gitWorktree('/workspace/one')], undefined) + + expect(build(buildDetectedGitWorktrees)).toEqual(build(buildDetectedGitWorktreesTwoPass)) + }) +}) diff --git a/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts b/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts index 7ec2cac1fc9..5c734d8bcd9 100644 --- a/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts +++ b/src/main/ipc/worktrees/listing/ssh-worktree-fallback.ts @@ -155,9 +155,10 @@ export function buildDetectedGitWorktrees( const repoOwnerCount = store.getRepos().filter((candidate) => candidate.id === repo.id).length const detected = liveWorktrees.map((gitWorktree) => { const worktreeId = `${repo.id}::${gitWorktree.path}` - const legacyMeta = store.getWorktreeMeta?.(worktreeId) + // Why: the locator-keyed row is only a stand-in for a missing host snapshot, so don't read it when we have one. + const legacyMeta = allMeta === undefined ? store.getWorktreeMeta?.(worktreeId) : undefined const metaById = allMeta ?? (legacyMeta ? { [worktreeId]: legacyMeta } : {}) - let meta = + const meta = readWorktreeMetaForHost(store, worktreeId, getRepoExecutionHostId(repo)) ?? getRepoOwnedWorktreeMeta(repo, worktreeId, metaById, repoOwnerCount) const worktree = mergeWorktree(repo.id, gitWorktree, meta, repo.displayName) @@ -174,17 +175,21 @@ export function buildDetectedGitWorktrees( return detected } - meta = resolveWorktreeMetaWithDiscoveryBackfill( + const backfilledMeta = resolveWorktreeMetaWithDiscoveryBackfill( store, repo, worktreeId, allMeta, repoOwnerCount ) + // Why: backfill hands back the same object when it wrote nothing, and both builders are pure over it. + if (backfilledMeta === meta) { + return detected + } return toDetectedWorktree({ repo, - worktree: mergeWorktree(repo.id, gitWorktree, meta, repo.displayName), - meta, + worktree: mergeWorktree(repo.id, gitWorktree, backfilledMeta, repo.displayName), + meta: backfilledMeta, settings, knownOrcaLayouts, isLegacyRepoForVisibility, diff --git a/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts b/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts index 6af4d45c3cd..b17677ddfcc 100644 --- a/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts +++ b/src/main/ipc/worktrees/listing/worktree-discovery-metadata.ts @@ -40,8 +40,9 @@ export function resolveWorktreeMetaWithDiscoveryBackfill( repoOwnerCount = store.getRepos().filter((candidate) => candidate.id === repo.id).length ): WorktreeMeta { const executionHostId = getRepoExecutionHostId(repo) - const legacyMeta = store.getWorktreeMeta?.(worktreeId) const allMeta = allMetaOverride ?? store.getAllWorktreeMeta?.() + // Why: the locator-keyed row is only a stand-in for a missing snapshot, so don't read it when we have one. + const legacyMeta = allMeta === undefined ? store.getWorktreeMeta?.(worktreeId) : undefined const existing = readWorktreeMetaForHost(store, worktreeId, executionHostId) ?? getRepoOwnedWorktreeMeta( diff --git a/src/relay/git-porcelain-local-parity.test.ts b/src/relay/git-porcelain-local-parity.test.ts index d4969f6fd0e..9fccc85e7b3 100644 --- a/src/relay/git-porcelain-local-parity.test.ts +++ b/src/relay/git-porcelain-local-parity.test.ts @@ -160,7 +160,8 @@ describe('relay/desktop unmerged-entry porcelain parity', () => { const unmergedLines = [ 'u UU N... 100644 100644 100644 100644 aa bb cc plain.ts', 'u UD N... 100644 100644 000000 100644 aa bb cc "present \\303\\251.ts"', - 'u UD N... 100644 100644 000000 100644 aa bb cc "missing \\303\\251.ts"', + // mW=000000: real Git reports an absent working-tree path this way, and the file is not created below. + 'u UD N... 100644 100644 000000 000000 aa bb cc "missing \\303\\251.ts"', 'u DD N... 100644 100644 000000 000000 aa bb cc both-gone.ts' ] const git = vi.fn(async (args) => { diff --git a/src/shared/git-status-conflict-entries.test.ts b/src/shared/git-status-conflict-entries.test.ts new file mode 100644 index 00000000000..ce6b5ab9e9e --- /dev/null +++ b/src/shared/git-status-conflict-entries.test.ts @@ -0,0 +1,87 @@ +/** + * Asymmetric `u` records used to cost one `fs.access` each — a 9p/network round trip per conflict on + * a WSL or remote worktree. Porcelain v2 already carries the answer in the worktree mode (`mW`), so + * the probe must not come back. + */ +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type * as NodeFsPromisesModule from 'node:fs/promises' + +const { accessMock } = vi.hoisted(() => ({ accessMock: vi.fn() })) + +vi.mock('node:fs/promises', async (importOriginal) => ({ + ...(await importOriginal()), + access: accessMock +})) + +import { parseUnmergedEntry } from './git-status-conflict-entries' + +const WORKTREE = '/repo' + +/** `u

    ` — `mW` is the working-tree mode. */ +function unmergedLine(xy: string, modeWorktree: string, filePath: string): string { + return `u ${xy} N... 100644 100644 100644 ${modeWorktree} aaa bbb ccc ${filePath}` +} + +const ASYMMETRIC_KINDS = ['AU', 'UA', 'DU', 'UD'] as const + +describe('parseUnmergedEntry', () => { + beforeEach(() => { + accessMock.mockReset() + accessMock.mockRejectedValue(new Error('fs.access must not be reached for well-formed records')) + }) + + it('reads the working-tree mode instead of probing the filesystem', async () => { + for (const xy of ASYMMETRIC_KINDS) { + const absent = await parseUnmergedEntry(WORKTREE, unmergedLine(xy, '000000', 'gone.ts')) + const present = await parseUnmergedEntry(WORKTREE, unmergedLine(xy, '100644', 'here.ts')) + + expect(absent?.status, xy).toBe('deleted') + expect(present?.status, xy).toBe('modified') + } + + // The regression this replaces: one probe per asymmetric row, serialised across the status poll. + expect(accessMock).not.toHaveBeenCalled() + }) + + it('treats a symlink left in place of the conflicted file as present', async () => { + const entry = await parseUnmergedEntry(WORKTREE, unmergedLine('UD', '120000', 'link.ts')) + + expect(entry?.status).toBe('modified') + expect(accessMock).not.toHaveBeenCalled() + }) + + it('resolves the symmetric kinds from XY alone, whatever the working-tree mode says', async () => { + const bothModified = await parseUnmergedEntry(WORKTREE, unmergedLine('UU', '000000', 'a.ts')) + const bothAdded = await parseUnmergedEntry(WORKTREE, unmergedLine('AA', '000000', 'b.ts')) + const bothDeleted = await parseUnmergedEntry(WORKTREE, unmergedLine('DD', '100644', 'c.ts')) + + expect(bothModified?.status).toBe('modified') + expect(bothAdded?.status).toBe('modified') + expect(bothDeleted?.status).toBe('deleted') + expect(accessMock).not.toHaveBeenCalled() + }) + + it('drops submodule conflicts without probing', async () => { + const line = 'u UU S... 160000 160000 160000 160000 aa bb cc vendor/sub' + + expect(await parseUnmergedEntry(WORKTREE, line)).toBeNull() + expect(accessMock).not.toHaveBeenCalled() + }) + + it('keeps the working-tree probe as a fallback for a mode no real Git emits', async () => { + accessMock.mockRejectedValueOnce(Object.assign(new Error('nope'), { code: 'ENOENT' })) + const missing = await parseUnmergedEntry(WORKTREE, unmergedLine('UD', 'zzzzzz', 'weird-a.ts')) + + accessMock.mockResolvedValueOnce(undefined) + const found = await parseUnmergedEntry(WORKTREE, unmergedLine('UD', '12345', 'weird-b.ts')) + + accessMock.mockRejectedValueOnce(Object.assign(new Error('denied'), { code: 'EACCES' })) + const unreadable = await parseUnmergedEntry(WORKTREE, unmergedLine('UD', '', 'weird-c.ts')) + + expect(missing?.status).toBe('deleted') + expect(found?.status).toBe('modified') + // Why: an ambiguous fs failure keeps the row visible rather than falsely reading as 'deleted'. + expect(unreadable?.status).toBe('modified') + expect(accessMock).toHaveBeenCalledTimes(3) + }) +}) diff --git a/src/shared/git-status-conflict-entries.ts b/src/shared/git-status-conflict-entries.ts index a50d389e2bb..980b4960d8b 100644 --- a/src/shared/git-status-conflict-entries.ts +++ b/src/shared/git-status-conflict-entries.ts @@ -3,6 +3,8 @@ import * as path from 'node:path' import type { GitConflictKind, GitFileStatus, GitStatusEntry } from './git-status-types' import { decodeGitCQuotedPath } from './git-cquoted-path' +const OCTAL_FILE_MODE = /^[0-7]{6}$/ + export async function parseUnmergedEntry( worktreePath: string, line: string @@ -13,6 +15,7 @@ export async function parseUnmergedEntry( const modeStage1 = parts[3] const modeStage2 = parts[4] const modeStage3 = parts[5] + const modeWorktree = parts[6] const filePath = decodeGitCQuotedPath(parts.slice(10).join(' ')) if (!filePath) { return null @@ -32,7 +35,12 @@ export async function parseUnmergedEntry( return { path: filePath, area: 'unstaged', - status: await getConflictCompatibilityStatus(worktreePath, filePath, conflictKind), + status: await getConflictCompatibilityStatus( + worktreePath, + filePath, + conflictKind, + modeWorktree + ), conflictKind, conflictStatus: 'unresolved' } @@ -60,11 +68,12 @@ function parseConflictKind(xy: string): GitConflictKind | null { } // Why: `status` here is a rendering-compat choice for icon/color plumbing, not semantic; the conflict badge carries the real meaning. -// Why: for deleted_by_*/added_by_* variants Git's result depends on merge strategy, so check the filesystem. +// Why: for deleted_by_*/added_by_* variants Git's result depends on merge strategy, so ask whether the path is in the working tree. async function getConflictCompatibilityStatus( worktreePath: string, filePath: string, - conflictKind: GitConflictKind + conflictKind: GitConflictKind, + modeWorktree: string ): Promise { if (conflictKind === 'both_modified' || conflictKind === 'both_added') { return 'modified' @@ -74,8 +83,14 @@ async function getConflictCompatibilityStatus( return 'deleted' } - // Why async: on a WSL worktree this path is a `\\wsl.localhost\...` share, and a sync probe - // per asymmetric conflict blocks the Electron main thread for a 9p round trip each. + // Why: `mW` is the worktree mode Git already stat'ed for this row — `000000` means absent. Reading + // it costs nothing and stays consistent with the rest of the snapshot, whereas a re-probe here is a + // 9p/network round trip per asymmetric conflict on a WSL or remote worktree. + if (OCTAL_FILE_MODE.test(modeWorktree)) { + return modeWorktree === '000000' ? 'deleted' : 'modified' + } + + // Why: only reachable on output no real Git emits (truncated/malformed `u` record). try { await access(path.join(worktreePath, filePath)) return 'modified' From d247d6441ba09c23987065631e02c44dd4fba8d6 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:04:04 -0700 Subject: [PATCH 231/398] perf(startup): overlap the runtime capability refresh with the session-tabs inventory (#18460) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The startup structured-session restore chained `runtime:getStatus` before `session.tabs.listAll`, but the capability value is discarded at that call site — it only seeds the module cache later launch flows read, and the inventory fetch never reads it. On a profile with 413 worktrees / 801 tabs that serial leg cost a median 109 ms of the did-finish-load -> renderer-startup-hydration-done window. Issue both calls concurrently. `Promise.all` still resolves only after both settle, so the capability cache is populated no later than before. --- ...local-structured-session-tabs-sync.test.ts | 37 +++++++++++++++++++ .../inventory-refresh.ts | 10 +++-- 2 files changed, 44 insertions(+), 3 deletions(-) diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts index 7a6c713f37f..702c5570c0d 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync.test.ts @@ -307,6 +307,43 @@ describe('local structured session tab projection', () => { } }) + it('starts the session-tabs inventory without waiting for the capability refresh', async () => { + const priorApi = window.api + let releaseStatus = (): void => undefined + const statusGate = new Promise((resolve) => { + releaseStatus = resolve + }) + const getStatus = vi.fn(async () => { + await statusGate + return { capabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] } + }) + const call = vi.fn().mockResolvedValue({ ok: true, result: { snapshots: [] } }) + Object.defineProperty(window, 'api', { + configurable: true, + value: { runtime: { getStatus, call } } + }) + try { + vi.resetModules() + const { restoreLocalStructuredSessionTabsOnce } = + await import('./local-structured-session-tabs-sync') + let settled = false + const restored = restoreLocalStructuredSessionTabsOnce().finally(() => { + settled = true + }) + expect(getStatus).toHaveBeenCalledOnce() + // The inventory RPC must already be in flight while the capability refresh is pending. + expect(call).toHaveBeenCalledWith({ method: 'session.tabs.listAll', params: {} }) + // ...and overlapping must not let the restore open the gate before capabilities land. + await new Promise((resolve) => setTimeout(resolve, 0)) + expect(settled).toBe(false) + releaseStatus() + await restored + expect(call).toHaveBeenCalledOnce() + } finally { + Object.defineProperty(window, 'api', { configurable: true, value: priorApi }) + } + }) + it('accepts a newer session after merged content returns to the base epoch', () => { const state = createSnapshot() const base = { diff --git a/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts index d0f29ec0cf8..af756458707 100644 --- a/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts +++ b/src/renderer/src/runtime/local-structured-session-tabs-sync/inventory-refresh.ts @@ -10,10 +10,14 @@ import { applyStructuredSessionTabSnapshots } from './snapshot-apply' export function restoreLocalStructuredSessionTabsOnce( expectedGeneration = localStructuredSessionGeneration() ): Promise { + // Why concurrent: the capability refresh only seeds the module cache that later launch + // flows read; the inventory fetch never reads it, so chaining them only paid a second + // serial IPC round-trip on the startup gate. return latchLocalStructuredSessionRestore(() => - refreshLocalRuntimeCapabilities() - .then(() => refreshLocalStructuredSessionTabs(expectedGeneration)) - .then(() => undefined) + Promise.all([ + refreshLocalRuntimeCapabilities(), + refreshLocalStructuredSessionTabs(expectedGeneration) + ]).then(() => undefined) ) } From 6815fed6d66d2ac0a28123d5e68834b91843564c Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:04:54 -0700 Subject: [PATCH 232/398] perf(worktrees): converge the trash sweep instead of re-walking doomed trees (#18429) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(worktrees): converge the trash sweep instead of re-walking doomed trees `transientLockRemovalOptions()` only asked for `maxRetries` on Windows, and `removeHostTree`'s retry ladder was gated on `process.platform === 'win32'`. A concurrent writer is not Windows-specific: Spotlight/`mds`, a scanner, or a live process writing under the tree surface the same EBUSY/ENOTEMPTY/EPERM on macOS and Linux. So on POSIX the startup sweep got exactly one attempt per entry, failed, and re-issued the same guaranteed-to-fail walk on every launch. - Extend the retry policy to every platform. Windows keeps its error set, its message fallback, and its delays; the message fallback stays Windows-only because POSIX always sets a code. - Persist a per-entry failure ledger in the trash root so a repeatedly failing entry is retried on a 15m/1h/6h ladder rather than on every launch. Nothing is abandoned: the ladder clamps, records are pruned when the entry goes, and a torn ledger fails open to a full sweep. - Defer the sweep behind first paint, so its recursive readdir/rm no longer competes with window creation and worktree-catalog hydration. * fix(worktrees): keep Node's per-level rm retries Windows-only Node's rimraf hands every child back to the retrying entry point (`_rmchildren` -> `rimraf`), so `maxRetries` is applied once per directory level and compounds: a permanently-failing leaf at depth d costs roughly `retryDelay * 36 * 9^(d-1)`. Measured on macOS against one `chflags uchg` file at depth 2, `{recursive, force}` rejected in 1 ms while `{maxRetries: 8, retryDelay: 150}` had not settled after 5 minutes. Handing those options to POSIX removals turned every `removeHostTree` on a worktree residue (`node_modules/.pnpm/...`, a dozen levels deep) into a promise that never settles -- wedging the serialized trash-deletion queue, hanging the sweep on its first failing entry so no backoff is ever recorded, and leaving the unregistered-worktree removal IPC pending forever. Keep the cross-platform retry where this PR put it -- the bounded outer ladders that re-issue one whole `rm` against the same already-chosen path -- and restore `transientLockRemovalOptions()` to Windows-only `maxRetries`. Also guard the deferred first-window task: off whenReady's promise chain a synchronous throw is an uncaughtException, which the pipe-error guard re-throws fatally. * fix(worktrees): make host tree removal see through Electron's asar shim The 267 stranded trash entries were not a concurrent-writer race. Electron patches `fs` so a `*.asar` file reports `isDirectory() === true`, so Node's recursive `rm` descends into the archive, `rmdir`s a real file, and fails the parent with ENOTEMPTY — deterministically, on every attempt. Every worktree that has run `pnpm install` carries a `default_app.asar`, which is why every residue stopped at the same path. Route `removeHostTree` through `original-fs` (Electron's unpatched `fs`, with a `node:fs/promises` fallback outside Electron) instead of retrying a failure that can never succeed. `removalPath`, `rmOptions` and the Windows retry ladder are byte-identical to `main`. Reverts the POSIX retry ladder, the `isTransientRemovalError` widening, the sweep backoff ledger and the inverted `does not retry host removal failures outside Windows` ratchet — none of them were fixing the actual failure. * fix(worktrees): drop the stray orchestration test and bundle the asar guard like production `orchestration-statement-compilation.test.ts` belongs to #18420 and was swept into this branch by accident. It imports `./prepared-statement-cache`, which does not exist here, so `tsc -p config/tsconfig.node.json` failed on this branch. Removed; typecheck is clean again. The Electron asar guard pre-externalized `original-fs` in its own Vite build, which is not what the shipped bundle does. Mirror `isExternalMainModule` from electron.vite.config.ts instead, so the guard also proves the production bundler leaves `createRequire(__filename)('original-fs')` as a runtime require — if that ever became a static import or got folded, production would silently degrade to the shimmed `fs` while the old test kept passing. --- src/main/asar-transparent-fs.test.ts | 43 ++++++ src/main/asar-transparent-fs.ts | 35 +++++ .../host-tree-removal-asar.electron.test.ts | 132 ++++++++++++++++++ src/main/host-tree-removal.ts | 8 +- .../host/deferred-secret-protection-report.ts | 20 +-- src/main/startup/first-window-deferral.ts | 37 +++++ .../startup/main-process-ready-runtime.ts | 20 ++- 7 files changed, 268 insertions(+), 27 deletions(-) create mode 100644 src/main/asar-transparent-fs.test.ts create mode 100644 src/main/asar-transparent-fs.ts create mode 100644 src/main/host-tree-removal-asar.electron.test.ts create mode 100644 src/main/startup/first-window-deferral.ts diff --git a/src/main/asar-transparent-fs.test.ts b/src/main/asar-transparent-fs.test.ts new file mode 100644 index 00000000000..f416450599d --- /dev/null +++ b/src/main/asar-transparent-fs.test.ts @@ -0,0 +1,43 @@ +import { mkdir, mkdtemp, writeFile } from 'node:fs/promises' +import { existsSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { rm } from './asar-transparent-fs' + +// Why not an asar fixture here: plain Node has no asar shim to see through, so the archive case can +// only be settled by the real binary — `host-tree-removal-asar.electron.test.ts` does that. What +// this pins is the other half: outside Electron `original-fs` does not resolve, and the helper has +// to degrade to `node:fs/promises` rather than throw at first use. +const roots: string[] = [] + +afterAll(async () => { + for (const root of roots) { + await rm(root, { recursive: true, force: true }).catch(() => {}) + } +}) + +describe('asar-transparent rm', () => { + it('removes a tree recursively where `original-fs` is unresolvable', async () => { + expect(process.versions.electron).toBeUndefined() + const root = await mkdtemp(join(tmpdir(), 'orca-asar-transparent-')) + roots.push(root) + const target = join(root, 'wt-1700000000000-abcdef01') + await mkdir(join(target, 'nested'), { recursive: true }) + await writeFile(join(target, 'nested', 'file.txt'), 'x', 'utf8') + + await expect(rm(target, { recursive: true, force: true })).resolves.toBeUndefined() + + expect(existsSync(target)).toBe(false) + expect(existsSync(root)).toBe(true) + }) + + it('honours `force: false` rather than swallowing a missing path', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-asar-transparent-')) + roots.push(root) + + await expect(rm(join(root, 'absent'), { recursive: true })).rejects.toMatchObject({ + code: 'ENOENT' + }) + }) +}) diff --git a/src/main/asar-transparent-fs.ts b/src/main/asar-transparent-fs.ts new file mode 100644 index 00000000000..cddab843434 --- /dev/null +++ b/src/main/asar-transparent-fs.ts @@ -0,0 +1,35 @@ +// Why: Electron patches `fs` so a `*.asar` file reports `isDirectory() === true`, so Node's +// recursive `rm` descends into the archive, tries to `rmdir` a real file, and fails the parent with +// ENOTEMPTY. Every worktree that has ever run `pnpm install` carries at least one +// (`node_modules/.pnpm/electron@…/…/Electron.app/Contents/Resources/default_app.asar`), so a +// worktree removal aborts there deterministically — the residue is not a concurrent-writer race and +// no amount of retrying clears it. `original-fs` is Electron's unpatched `fs`; unlike +// `process.noAsar` it is scoped to this call rather than to the whole process, which matters because +// a multi-GB removal runs for seconds while the main process may still be loading modules out of +// `app.asar`. See `cli/appimage-payload-removal.ts` for the same bug at a call site short enough to +// use the process-global flag. + +import { rm as nodeRm } from 'node:fs/promises' +import { createRequire } from 'node:module' + +type Rm = typeof nodeRm + +let resolvedRm: Rm | undefined + +function resolveRm(): Rm { + try { + // Why require and not an import: `original-fs` only exists inside Electron, so vitest, the + // `orca` CLI and the plain-node entrypoints must resolve `node:fs/promises` instead — and there + // the shim does not exist either, so plain `fs` is already asar-transparent. + const originalFs = createRequire(__filename)('original-fs') as { promises?: { rm?: Rm } } + return typeof originalFs.promises?.rm === 'function' ? originalFs.promises.rm : nodeRm + } catch { + return nodeRm + } +} + +/** `fs.promises.rm` that sees a `*.asar` as the file it is rather than as a directory. */ +export const rm: Rm = (path, options) => { + resolvedRm ??= resolveRm() + return resolvedRm(path, options) +} diff --git a/src/main/host-tree-removal-asar.electron.test.ts b/src/main/host-tree-removal-asar.electron.test.ts new file mode 100644 index 00000000000..a41a55035ac --- /dev/null +++ b/src/main/host-tree-removal-asar.electron.test.ts @@ -0,0 +1,132 @@ +import { spawnSync } from 'node:child_process' +import { + copyFileSync, + existsSync, + mkdirSync, + mkdtempSync, + readFileSync, + writeFileSync +} from 'node:fs' +import { createRequire, isBuiltin } from 'node:module' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import { afterAll, describe, expect, it } from 'vitest' +import { removeTreeSync } from '../shared/windows-transient-lock-removal' + +/** + * Why the real binary: Electron patches `fs` so a `*.asar` file reports `isDirectory() === true`, so + * a recursive `rm` descends into the archive, `rmdir`s a real file, and fails the parent with + * ENOTEMPTY. Plain Node has no such shim, so no in-process unit test can reproduce it — and every + * worktree that has run `pnpm install` carries a `default_app.asar`, which is what stranded 267 + * trash entries on the reporting machine. This runs the shipped `removeHostTree` against a real + * archive under the real binary. + */ +const requireFromTest = createRequire(import.meta.url) +const electronBinary = requireFromTest('electron') as string +const electronDist = join(dirname(requireFromTest.resolve('electron/package.json')), 'dist') +const FIXTURE_ASAR = [ + join(electronDist, 'Electron.app/Contents/Resources/default_app.asar'), + join(electronDist, 'resources/default_app.asar') +].find((candidate) => existsSync(candidate)) + +// Mirrors the residue reported on the failing machine, down to the depth of the blocking leaf. +const ENTRY_NAME = 'wt-1700000000000-abcdef01' +const ASAR_PARENT = 'node_modules/.pnpm/electron/node_modules/electron/dist/App/Contents/Resources' + +const roots: string[] = [] + +afterAll(() => { + for (const root of roots) { + try { + removeTreeSync(root) + } catch { + // A fixture the shim strands is exactly what this file is about; never fail teardown on it. + } + } +}) + +type ProbeResult = { failure: string | null; residue: string[] } + +function buildDriver(bundlePath: string, target: string, resultPath: string): string { + return [ + `const fs = require('node:fs')`, + `const { removeHostTree } = require(${JSON.stringify(bundlePath)})`, + // Why noAsar for the read-back: the shim would report the stranded archive as a directory here + // too, so the residue listing has to be taken with real filesystem semantics. + `const withoutAsar = (fn) => { const prev = process.noAsar; process.noAsar = true; try { return fn() } finally { process.noAsar = prev } }`, + `;(async () => {`, + ` let failure = null`, + ` try { await removeHostTree(${JSON.stringify(target)}) } catch (error) { failure = error.code ?? String(error) }`, + ` const residue = withoutAsar(() => fs.existsSync(${JSON.stringify(target)})`, + ` ? fs.readdirSync(${JSON.stringify(target)}, { recursive: true }).map(String)`, + ` : [])`, + ` fs.writeFileSync(${JSON.stringify(resultPath)}, JSON.stringify({ failure, residue }))`, + `})()` + ].join('\n') +} + +async function bundleHostTreeRemoval(outFile: string): Promise { + const { build } = await import('vite') + const result = await build({ + root: process.cwd(), + configFile: false, + logLevel: 'error', + build: { + write: false, + minify: false, + ssr: true, + rollupOptions: { + input: 'src/main/host-tree-removal.ts', + // Why mirror `isExternalMainModule` from electron.vite.config.ts exactly — CJS, and + // `original-fs` deliberately *not* externalized: the shipped bundle does not list it either, + // so if the archive-aware `rm` ever became a static import (or the bundler learned to fold + // `createRequire(...)('original-fs')`) production would silently degrade to the shimmed `fs` + // while a test that pre-externalized it kept passing. + output: { format: 'cjs' }, + external: (id: string) => isBuiltin(id) || id === 'electron' || id.startsWith('electron/') + } + } + }) + const output = (Array.isArray(result) ? result[0] : result) as { output: { code?: string }[] } + const code = output.output[0]?.code + expect(typeof code).toBe('string') + writeFileSync(outFile, code as string, 'utf8') +} + +function buildStrandedTree(root: string): string { + const target = join(root, ENTRY_NAME) + const asarParent = join(target, ...ASAR_PARENT.split('/')) + mkdirSync(asarParent, { recursive: true }) + copyFileSync(FIXTURE_ASAR as string, join(asarParent, 'default_app.asar')) + writeFileSync(join(asarParent, 'plain.txt'), 'x', 'utf8') + return target +} + +describe('removeHostTree against a tree holding an asar archive', () => { + it.runIf(FIXTURE_ASAR)( + 'removes the whole tree under the real Electron binary', + async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-host-tree-asar-')) + roots.push(root) + const bundlePath = join(root, 'host-tree-removal.cjs') + await bundleHostTreeRemoval(bundlePath) + const target = buildStrandedTree(root) + const resultPath = join(root, 'result.json') + const driverPath = join(root, 'driver.cjs') + writeFileSync(driverPath, buildDriver(bundlePath, target, resultPath), 'utf8') + + const run = spawnSync(electronBinary, [driverPath], { + encoding: 'utf8', + env: { ...process.env, ELECTRON_RUN_AS_NODE: '1' }, + timeout: 60_000 + }) + expect(run.status, run.stderr?.slice(-2000)).toBe(0) + + const probe = JSON.parse(readFileSync(resultPath, 'utf8')) as ProbeResult + // Without an asar-transparent `rm` this is `ENOTEMPTY` and the residue stops at the archive, + // on every attempt, forever — it is not a race a retry can win. + expect(probe).toEqual({ failure: null, residue: [] }) + }, + 120_000 + ) +}) diff --git a/src/main/host-tree-removal.ts b/src/main/host-tree-removal.ts index a5d5d447956..f789a9861d0 100644 --- a/src/main/host-tree-removal.ts +++ b/src/main/host-tree-removal.ts @@ -1,10 +1,12 @@ // Why: every recursive host delete Orca performs (worktrees, terminal history, quarantined recovery -// generations) hits the same Windows stickiness — AV/indexers/late handle releases surface transient -// EBUSY/ENOTEMPTY/EPERM on a tree Node just emptied. One helper so no call site forgets the retries. +// generations) hits the same two hazards, so one helper exists so no call site forgets either. +// Windows stickiness — AV/indexers/late handle releases surface transient EBUSY/ENOTEMPTY/EPERM on a +// tree Node just emptied — and Electron's asar shim, which strands any tree holding a `*.asar` +// (see `asar-transparent-fs`). -import { rm } from 'node:fs/promises' import { win32 } from 'node:path' import { setTimeout as delay } from 'node:timers/promises' +import { rm } from './asar-transparent-fs' import { isWindowsAbsolutePathLike } from '../shared/cross-platform-path' import { isWslUncPath } from '../shared/wsl-paths' import { transientLockRemovalOptions } from '../shared/windows-transient-lock-removal' diff --git a/src/main/host/deferred-secret-protection-report.ts b/src/main/host/deferred-secret-protection-report.ts index 8f5be1fb047..7b4ea14367c 100644 --- a/src/main/host/deferred-secret-protection-report.ts +++ b/src/main/host/deferred-secret-protection-report.ts @@ -1,4 +1,4 @@ -import { app, type BrowserWindow } from 'electron' +import { runAfterFirstWindowShown } from '../startup/first-window-deferral' import { reportSecretProtectionGap } from './secret-protection-report' /** @@ -72,21 +72,5 @@ export function scheduleSecretProtectionGapReport({ } } - let ran = false - const run = (): void => { - if (ran) { - return - } - ran = true - clearTimeout(fallback) - // Why setImmediate: keep the blocking keyring probe off the event handler that - // reveals the window, so the reveal paints first. - setImmediate(report) - } - - const fallback = setTimeout(run, REPORT_FALLBACK_MS) - fallback.unref?.() - app.once('browser-window-created', (_event: Electron.Event, window: BrowserWindow) => { - window.once('ready-to-show', run) - }) + runAfterFirstWindowShown(report, REPORT_FALLBACK_MS) } diff --git a/src/main/startup/first-window-deferral.ts b/src/main/startup/first-window-deferral.ts new file mode 100644 index 00000000000..6a144549efe --- /dev/null +++ b/src/main/startup/first-window-deferral.ts @@ -0,0 +1,37 @@ +import { app, type BrowserWindow } from 'electron' + +/** + * Run `task` once the first window can paint, or after `fallbackMs` if it never does. + * + * For startup work nothing on the critical path consumes: a probe or a disk sweep started before the + * window exists competes with window creation for the same main thread and libuv threadpool, and the + * user sees that as the app being slow to open. + * + * Why a fallback as well as the window event: `ready-to-show` can fail to fire at all when the + * GPU/driver cannot present (see main-window-state-lifecycle), and headless serve has no window. + */ +export function runAfterFirstWindowShown(task: () => void, fallbackMs: number): void { + let ran = false + const run = (): void => { + if (ran) { + return + } + ran = true + clearTimeout(fallback) + // Why setImmediate: keep the work off the event handler that reveals the window, so it paints first. + // Why the guard: off whenReady's promise chain a synchronous throw is an uncaughtException, and + // installUncaughtPipeErrorGuard re-throws those fatally — deferred startup chores are never that. + setImmediate(() => { + try { + task() + } catch (error) { + console.warn('[startup] deferred first-window task failed', error) + } + }) + } + const fallback = setTimeout(run, fallbackMs) + fallback.unref?.() + app.once('browser-window-created', (_event: Electron.Event, window: BrowserWindow) => { + window.once('ready-to-show', run) + }) +} diff --git a/src/main/startup/main-process-ready-runtime.ts b/src/main/startup/main-process-ready-runtime.ts index f89f63186a0..75784a4c196 100644 --- a/src/main/startup/main-process-ready-runtime.ts +++ b/src/main/startup/main-process-ready-runtime.ts @@ -32,8 +32,12 @@ import { import { initializeMainProcessAutomations } from './main-process-automations' import { initializeMainProcessPlugins } from './main-process-plugins' import { collectWorktreeTrashSweepRoots, sweepStaleWorktreeTrash } from '../worktree-trash' +import { runAfterFirstWindowShown } from './first-window-deferral' import { logStartupMilestone } from './startup-diagnostics' +// Headless serve never opens a window, so the sweep still has to run off a timer there. +const WORKTREE_TRASH_SWEEP_FALLBACK_MS = 15_000 + export async function initializeReadyRuntimeServices(): Promise { const store = state.store if (!store) { @@ -74,12 +78,16 @@ export async function initializeReadyRuntimeServices(): Promise { state.emulatorBridge = new EmulatorBridge() runtime.setEmulatorBridge(state.emulatorBridge) // Why: worktree deletion renames the checkout aside and deletes it in the background, so a quit or - // crash mid-delete can leave the moved directory on disk. - void sweepStaleWorktreeTrash( - collectWorktreeTrashSweepRoots(store.getRepos(), store.getSettings()) - ).catch((error) => { - console.warn('[worktrees] Failed to sweep leftover worktree directories:', error) - }) + // crash mid-delete can leave the moved directory on disk. Why deferred: the sweep's recursive + // readdir/rm runs on the same libuv threadpool the window's first paint and worktree-catalog + // hydration are reading disk on, and nothing on the startup path consumes its result. + runAfterFirstWindowShown(() => { + void sweepStaleWorktreeTrash( + collectWorktreeTrashSweepRoots(store.getRepos(), store.getSettings()) + ).catch((error) => { + console.warn('[worktrees] Failed to sweep leftover worktree directories:', error) + }) + }, WORKTREE_TRASH_SWEEP_FALLBACK_MS) nativeTheme.themeSource = store.getSettings().theme ?? 'system' // Why (#16441): the real-home grant runs a codex app-server session. It stays // ordered before managed-hook reconciliation — an incapable host must re-arm From 558f57de58958bfaf57dbe9c5df6da5d4f7130b4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:05:55 -0700 Subject: [PATCH 233/398] perf(source-control): sort branch entries before filtering, gate projections by view mode (#18426) * perf(source-control): sort branch entries before filtering, gate projections by view mode Two dead-work fixes in the Source Control file projection. 1. filterAndSortSourceControlPathEntries copied and re-sorted the uncapped branch entry list with Intl.Collator on every keystroke. Sort once on branchEntries, filter after: Array#filter preserves order and compareFileNames is a total order, so filter(sort(x)) === sort(filter(x)). 2. The tree projection was built in list mode and the list projection in tree mode, then discarded. Gate each memo on sourceControlViewMode and return a shared empty projection, matching the combined-diff file tree precedent. * docs(source-control): drop the total-order premise from the projection sort argument The sort-before-filter swap does not need compareFileNames to be a total order. A stable Array#sort places each element by (comparator result, original index) and Array#filter disturbs neither, so filter(sort(x)) === sort(filter(x)) for any self-consistent comparator -- which the previous filter-then-sort already required. Restating that removes a shared-module property (the code-unit tie-break in file-name-sort.ts) from this hook's correctness argument instead of defending it. Also record on the EMPTY_* singletons that the gates and both branching consumers read one sourceControlViewMode prop in one synchronous render, so the off-mode value cannot reach the screen, and warn against deriving the mode from a separate store read. New guard: matches filter-then-sort under a comparator that is not a total order. It ties every path sharing a top-level directory over 300 entries and fails against a correct-but-unstable sort. With the duplicate paths removed from ORDERING_FIXTURE the pre-existing equivalence test passes under that same mutant, so this is the only test that pins stability. No behaviour change: counters over first render + 8 keystrokes at n=2000 are identical before and after (34685 compareFileNames calls, 0 tree builds in list mode). * refactor(source-control): freeze the empty branch-tree singleton Object.freeze([]) matches the other three empty projection singletons and the combined-diff-file-tree precedent; readonly types keep it honest. --- .../source-control-file-filter.test.ts | 21 -- .../right-sidebar/source-control-tree.ts | 2 +- .../source-control/listing/branch-section.tsx | 2 +- .../source-control/listing/file-filter.ts | 10 - .../listing/use-file-projection-work.test.tsx | 323 ++++++++++++++++++ .../listing/use-file-projection.ts | 64 +++- 6 files changed, 379 insertions(+), 43 deletions(-) create mode 100644 src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection-work.test.tsx diff --git a/src/renderer/src/components/right-sidebar/source-control-file-filter.test.ts b/src/renderer/src/components/right-sidebar/source-control-file-filter.test.ts index 6b2419c9195..18d8ad5f3b9 100644 --- a/src/renderer/src/components/right-sidebar/source-control-file-filter.test.ts +++ b/src/renderer/src/components/right-sidebar/source-control-file-filter.test.ts @@ -1,7 +1,6 @@ import { describe, expect, it } from 'vitest' import { SOURCE_CONTROL_FILE_FILTER_QUERY_MAX_BYTES, - filterAndSortSourceControlPathEntries, filterSourceControlGroupedPathEntries, filterSourceControlPathEntries, getSourceControlFileFilterState, @@ -10,26 +9,6 @@ import { } from './source-control/listing/file-filter' describe('source-control-file-filter', () => { - it('naturally orders committed branch rows without mutating store input', () => { - const entries = [ - { path: 'migrations/100.sql' }, - { path: 'migrations/9.sql' }, - { path: 'migrations/99.sql' } - ] - - expect( - filterAndSortSourceControlPathEntries(entries, { - normalizedFilter: '', - tooLarge: false - }).map((entry) => entry.path) - ).toEqual(['migrations/9.sql', 'migrations/99.sql', 'migrations/100.sql']) - expect(entries.map((entry) => entry.path)).toEqual([ - 'migrations/100.sql', - 'migrations/9.sql', - 'migrations/99.sql' - ]) - }) - it('normalizes bounded queries and filters entries by path', () => { const filter = getSourceControlFileFilterState(' SRC/button ') diff --git a/src/renderer/src/components/right-sidebar/source-control-tree.ts b/src/renderer/src/components/right-sidebar/source-control-tree.ts index eb4a01a7ea3..d855201646b 100644 --- a/src/renderer/src/components/right-sidebar/source-control-tree.ts +++ b/src/renderer/src/components/right-sidebar/source-control-tree.ts @@ -165,7 +165,7 @@ export function buildGitStatusSourceControlTree( } export function flattenSourceControlTree( - nodes: SourceControlTreeNode[], + nodes: readonly SourceControlTreeNode[], collapsedDirectoryKeys: ReadonlySet ): SourceControlTreeNode[] { const result: SourceControlTreeNode[] = [] diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/branch-section.tsx b/src/renderer/src/components/right-sidebar/source-control/listing/branch-section.tsx index ab37458adb7..3f2ab4723f7 100644 --- a/src/renderer/src/components/right-sidebar/source-control/listing/branch-section.tsx +++ b/src/renderer/src/components/right-sidebar/source-control/listing/branch-section.tsx @@ -39,7 +39,7 @@ export function SourceControlBranchSection({ collapsedSections: Set toggleSection: (section: string) => void sourceControlViewMode: SourceControlViewMode - visibleBranchTreeRows: SourceControlTreeNode[] + visibleBranchTreeRows: readonly SourceControlTreeNode[] fileListScrollElement: HTMLDivElement | null collapsedTreeDirs: Set toggleTreeDir: (key: string) => void diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/file-filter.ts b/src/renderer/src/components/right-sidebar/source-control/listing/file-filter.ts index 4bdd676bcf5..ae379b51d4c 100644 --- a/src/renderer/src/components/right-sidebar/source-control/listing/file-filter.ts +++ b/src/renderer/src/components/right-sidebar/source-control/listing/file-filter.ts @@ -1,5 +1,4 @@ import { isClipboardTextByteLengthOverLimit } from '../../../../../../shared/clipboard-text' -import { compareFileNames } from '../../../../../../shared/file-name-sort' export const SOURCE_CONTROL_FILE_FILTER_QUERY_MAX_BYTES = 2 * 1024 @@ -49,15 +48,6 @@ export function filterSourceControlPathEntries return entries.filter((entry) => entry.path.toLowerCase().includes(filter.normalizedFilter)) } -export function filterAndSortSourceControlPathEntries( - entries: T[], - filter: SourceControlFileFilterState -): T[] { - return [...filterSourceControlPathEntries(entries, filter)].sort((a, b) => - compareFileNames(a.path, b.path) - ) -} - export function filterSourceControlGroupedPathEntries( grouped: SourceControlGroupedPathEntries, filter: SourceControlFileFilterState diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection-work.test.tsx b/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection-work.test.tsx new file mode 100644 index 00000000000..ed62e05a25a --- /dev/null +++ b/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection-work.test.tsx @@ -0,0 +1,323 @@ +// @vitest-environment happy-dom + +import { renderHook } from '@testing-library/react' +import { beforeEach, describe, expect, it, vi } from 'vitest' +import type { GitBranchChangeEntry } from '../../../../../../shared/git-diff-compare-types' +import type { GitStatusEntry } from '../../../../../../shared/git-status-types' +import type { SourceControlViewMode } from '../../../../../../shared/ui-chrome-types' +import type * as FileNameSortModule from '../../../../../../shared/file-name-sort' +import type * as SourceControlTreeModule from '../../source-control-tree' +import type * as SubmoduleExpansionModule from './submodule-expansion' + +const counters = vi.hoisted(() => ({ + compareFileNames: 0, + buildGitStatusSourceControlTree: 0, + buildSourceControlTree: 0, + flattenSourceControlTree: 0, + injectExpandedSubmoduleRows: 0, + injectExpandedSubmoduleEntries: 0 +})) + +/** Lets a test swap in a comparator that is deliberately not a total order. */ +const comparatorOverride = vi.hoisted(() => ({ + current: null as ((a: string, b: string) => number) | null +})) + +vi.mock('../../../../../../shared/file-name-sort', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + compareFileNames: (a: string, b: string) => { + counters.compareFileNames += 1 + return comparatorOverride.current + ? comparatorOverride.current(a, b) + : actual.compareFileNames(a, b) + } + } +}) + +vi.mock('../../source-control-tree', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + buildGitStatusSourceControlTree: ( + ...args: Parameters + ) => { + counters.buildGitStatusSourceControlTree += 1 + return actual.buildGitStatusSourceControlTree(...args) + }, + buildSourceControlTree: ((...args: unknown[]) => { + counters.buildSourceControlTree += 1 + return (actual.buildSourceControlTree as (...a: unknown[]) => unknown)(...args) + }) as typeof actual.buildSourceControlTree, + flattenSourceControlTree: ((...args: unknown[]) => { + counters.flattenSourceControlTree += 1 + return (actual.flattenSourceControlTree as (...a: unknown[]) => unknown)(...args) + }) as typeof actual.flattenSourceControlTree + } +}) + +vi.mock('./submodule-expansion', async (importOriginal) => { + const actual = await importOriginal() + return { + ...actual, + injectExpandedSubmoduleRows: ((...args: unknown[]) => { + counters.injectExpandedSubmoduleRows += 1 + return (actual.injectExpandedSubmoduleRows as (...a: unknown[]) => unknown)(...args) + }) as typeof actual.injectExpandedSubmoduleRows, + injectExpandedSubmoduleEntries: ((...args: unknown[]) => { + counters.injectExpandedSubmoduleEntries += 1 + return (actual.injectExpandedSubmoduleEntries as (...a: unknown[]) => unknown)(...args) + }) as typeof actual.injectExpandedSubmoduleEntries + } +}) + +const { compareFileNames } = await import('../../../../../../shared/file-name-sort') +const { getSourceControlFileFilterState, filterSourceControlPathEntries } = + await import('./file-filter') +const { useSourceControlFileProjection } = await import('./use-file-projection') + +const NO_ENTRIES: GitStatusEntry[] = [] +const NO_COLLAPSED_TREE_DIRS = new Set() +const NO_EXPANDED_SUBMODULES = new Set() +const NO_COLLAPSED_SECTIONS = new Set() +const NO_SUBMODULE_STATUS = {} +const GROUP_ORDER = ['unstaged', 'staged', 'untracked'] as const + +type ProjectionProps = { + entries: GitStatusEntry[] + branchEntries: GitBranchChangeEntry[] + filterQuery: string + sourceControlViewMode: SourceControlViewMode +} + +function renderProjection(initialProps: ProjectionProps) { + return renderHook( + (props: ProjectionProps) => + useSourceControlFileProjection({ + entries: props.entries, + branchEntries: props.branchEntries, + filterQuery: props.filterQuery, + sourceControlGroupOrder: GROUP_ORDER, + activeWorktreeId: 'wt-1', + worktreePath: '/repo', + isFolder: false, + collapsedTreeDirs: NO_COLLAPSED_TREE_DIRS, + expandedSubmoduleKeys: NO_EXPANDED_SUBMODULES, + submoduleStatusByKey: NO_SUBMODULE_STATUS, + sourceControlViewMode: props.sourceControlViewMode, + collapsedSections: NO_COLLAPSED_SECTIONS + }), + { initialProps } + ) +} + +function makeBranchEntries(count: number): GitBranchChangeEntry[] { + return Array.from({ length: count }, (_, index) => ({ + path: `src/area-${index % 7}/nested/deep-${index % 13}/file-${index}.ts`, + status: 'modified' as const + })) +} + +/** Numeric collation, case variants, unicode, collator ties, and an exact duplicate path. */ +const ORDERING_FIXTURE: GitBranchChangeEntry[] = [ + { path: 'migrations/100.sql', status: 'modified' }, + { path: 'migrations/9.sql', status: 'modified' }, + { path: 'migrations/99.sql', status: 'added' }, + { path: 'migrations/02.sql', status: 'modified' }, + { path: 'migrations/2.sql', status: 'deleted' }, + { path: 'src/Button.tsx', status: 'modified' }, + { path: 'src/button.tsx', status: 'added' }, + { path: 'src/éclair.ts', status: 'modified' }, + { path: 'src/eclair.ts', status: 'deleted' }, + { path: 'src/Éclair.ts', status: 'modified' }, + { path: 'src/日本語.ts', status: 'added' }, + { path: 'src/dup.ts', status: 'modified' }, + { path: 'src/dup.ts', status: 'added' } +] + +/** The pre-change implementation: filter, then copy-and-sort. */ +function legacyFilterThenSort( + entries: GitBranchChangeEntry[], + filterQuery: string +): GitBranchChangeEntry[] { + const state = getSourceControlFileFilterState(filterQuery) + return [...filterSourceControlPathEntries(entries, state)].sort((a, b) => + compareFileNames(a.path, b.path) + ) +} + +function makeStatusEntries(count: number): GitStatusEntry[] { + return Array.from({ length: count }, (_, index) => ({ + path: `src/area-${index % 7}/nested/deep-${index % 13}/file-${index}.ts`, + status: 'modified' as const, + area: (['unstaged', 'staged', 'untracked'] as const)[index % 3] + })) +} + +/** + * Deliberately not a total order: every path under the same top-level directory compares equal, so + * distinct paths tie. Sort-before-filter must still match filter-before-sort under it. + */ +function compareTopLevelDirOnly(a: string, b: string): number { + const dirA = a.slice(0, a.indexOf('/')) + const dirB = b.slice(0, b.indexOf('/')) + return dirA < dirB ? -1 : dirA > dirB ? 1 : 0 +} + +/** Big enough that V8 leaves binary insertion sort for TimSort, where instability would show. */ +function makeTieHeavyEntries(count: number): GitBranchChangeEntry[] { + return Array.from({ length: count }, (_, index) => ({ + // Scrambled so the tie order is not already the sorted order. + path: `dir-${(index * 7) % 3}/file-${(index * 31) % count}.ts`, + status: 'modified' as const + })) +} + +beforeEach(() => { + comparatorOverride.current = null + for (const key of Object.keys(counters) as (keyof typeof counters)[]) { + counters[key] = 0 + } +}) + +describe('useSourceControlFileProjection branch entry ordering', () => { + it('sorts committed branch entries once across many filter changes', () => { + const branchEntries = makeBranchEntries(400) + const { rerender } = renderProjection({ + entries: NO_ENTRIES, + branchEntries, + filterQuery: '', + sourceControlViewMode: 'list' + }) + + const comparesForInitialSort = counters.compareFileNames + expect(comparesForInitialSort).toBeGreaterThan(0) + + for (const filterQuery of ['f', 'fi', 'fil', 'file', 'file-', 'file-1']) { + rerender({ entries: NO_ENTRIES, branchEntries, filterQuery, sourceControlViewMode: 'list' }) + } + + expect(counters.compareFileNames).toBe(comparesForInitialSort) + }) + + it('produces the same order as the previous filter-then-sort for every filter', () => { + const { result, rerender } = renderProjection({ + entries: NO_ENTRIES, + branchEntries: ORDERING_FIXTURE, + filterQuery: '', + sourceControlViewMode: 'list' + }) + + for (const filterQuery of ['', 'src', 'MIGRATIONS', 'é', '9', 'dup', 'no-match']) { + rerender({ + entries: NO_ENTRIES, + branchEntries: ORDERING_FIXTURE, + filterQuery, + sourceControlViewMode: 'list' + }) + expect(result.current.filteredBranchEntries).toEqual( + legacyFilterThenSort(ORDERING_FIXTURE, filterQuery) + ) + } + }) + + // Guards the invariant the sort-before-filter swap actually rests on: a stable sort, not a total + // order. If sortedBranchEntries ever stops preserving the original order of tied paths, this + // diverges from filter-then-sort even though every total-order fixture above still passes. + it('matches filter-then-sort under a comparator that is not a total order', () => { + comparatorOverride.current = compareTopLevelDirOnly + const branchEntries = makeTieHeavyEntries(300) + const { result, rerender } = renderProjection({ + entries: NO_ENTRIES, + branchEntries, + filterQuery: '', + sourceControlViewMode: 'list' + }) + + for (const filterQuery of ['', 'dir-1', 'file-1', 'file-12', '7.ts', 'no-match']) { + rerender({ entries: NO_ENTRIES, branchEntries, filterQuery, sourceControlViewMode: 'list' }) + expect(result.current.filteredBranchEntries.map((entry) => entry.path)).toEqual( + legacyFilterThenSort(branchEntries, filterQuery).map((entry) => entry.path) + ) + } + }) + + it('does not mutate the store-owned branch entry array', () => { + const branchEntries = [...ORDERING_FIXTURE] + renderProjection({ + entries: NO_ENTRIES, + branchEntries, + filterQuery: '', + sourceControlViewMode: 'list' + }) + + expect(branchEntries).toEqual(ORDERING_FIXTURE) + }) +}) + +describe('useSourceControlFileProjection view-mode gating', () => { + const entries = makeStatusEntries(120) + const branchEntries = makeBranchEntries(120) + + it('builds no tree projection in list mode', () => { + const { result } = renderProjection({ + entries, + branchEntries, + filterQuery: '', + sourceControlViewMode: 'list' + }) + + expect(counters.buildGitStatusSourceControlTree).toBe(0) + expect(counters.buildSourceControlTree).toBe(0) + expect(counters.flattenSourceControlTree).toBe(0) + expect(counters.injectExpandedSubmoduleRows).toBe(0) + expect(counters.injectExpandedSubmoduleEntries).toBeGreaterThan(0) + expect(result.current.visibleTreeRowsBySection).toEqual({}) + expect(result.current.visibleBranchTreeRows).toEqual([]) + }) + + it('builds no list projection in tree mode', () => { + const { result } = renderProjection({ + entries, + branchEntries, + filterQuery: '', + sourceControlViewMode: 'tree' + }) + + expect(counters.injectExpandedSubmoduleEntries).toBe(0) + expect(counters.buildGitStatusSourceControlTree).toBeGreaterThan(0) + expect(counters.buildSourceControlTree).toBeGreaterThan(0) + expect(result.current.visibleListRowsBySection).toEqual({}) + }) + + it('has the other mode fully projected on the first render after a switch', () => { + const { result, rerender } = renderProjection({ + entries, + branchEntries, + filterQuery: '', + sourceControlViewMode: 'list' + }) + const listRows = result.current.visibleListRowsBySection + const listSelectionCount = result.current.visibleSelectionEntries.length + expect(listSelectionCount).toBe(entries.length) + + rerender({ entries, branchEntries, filterQuery: '', sourceControlViewMode: 'tree' }) + + expect(result.current.visibleListRowsBySection).toEqual({}) + expect(result.current.visibleBranchTreeRows.length).toBeGreaterThan(0) + expect( + Object.values(result.current.visibleTreeRowsBySection).reduce( + (total, rows) => total + rows.length, + 0 + ) + ).toBeGreaterThan(0) + expect(result.current.visibleSelectionEntries.length).toBe(listSelectionCount) + + rerender({ entries, branchEntries, filterQuery: '', sourceControlViewMode: 'list' }) + + expect(result.current.visibleTreeRowsBySection).toEqual({}) + expect(result.current.visibleBranchTreeRows).toEqual([]) + expect(result.current.visibleListRowsBySection).toEqual(listRows) + }) +}) diff --git a/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection.ts b/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection.ts index c885eaa33dc..66a370dd14d 100644 --- a/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection.ts +++ b/src/renderer/src/components/right-sidebar/source-control/listing/use-file-projection.ts @@ -2,10 +2,11 @@ import { useMemo } from 'react' import type { GitBranchChangeEntry } from '../../../../../../shared/git-diff-compare-types' import type { GitStatusEntry } from '../../../../../../shared/git-status-types' import type { SourceControlViewMode } from '../../../../../../shared/ui-chrome-types' +import { compareFileNames } from '../../../../../../shared/file-name-sort' import { compareGitStatusEntries } from '../../source-control-status-sort' import { - filterAndSortSourceControlPathEntries, filterSourceControlGroupedPathEntries, + filterSourceControlPathEntries, getSourceControlFileFilterState, type SourceControlFileFilterState } from './file-filter' @@ -56,10 +57,28 @@ export type SourceControlFileProjection = { visibleListRowsBySection: Partial< Record > - visibleBranchTreeRows: SourceControlTreeNode[] + visibleBranchTreeRows: readonly SourceControlTreeNode[] visibleSelectionEntries: FlatEntry[] } +// Why: only one view mode is ever rendered, so building the other mode's projection is pure dead +// work (precedent: the combined-diff file tree short-circuits the same way while collapsed). +// The gates below and both branching consumers (section-file-list.tsx, branch-section.tsx) read the +// same sourceControlViewMode prop within one synchronous render, so a mode switch can never show +// these. Keep that single source: deriving the mode from a separate store read would let a consumer +// switch a render before the memos do, and only then could one of these reach the screen. +const EMPTY_TREE_ROOTS_BY_SECTION: Readonly< + Partial> +> = Object.freeze({}) +const EMPTY_TREE_ROWS_BY_SECTION: Readonly< + Partial> +> = Object.freeze({}) +const EMPTY_LIST_ROWS_BY_SECTION: Readonly< + Partial> +> = Object.freeze({}) +const EMPTY_BRANCH_TREE_NODES: readonly SourceControlTreeNode[] = + Object.freeze([]) + export function useSourceControlFileProjection({ entries, branchEntries, @@ -127,12 +146,24 @@ export function useSourceControlFileProjection({ [unfilteredDisplaySections] ) + // Why: sorting before filtering keeps the collator off the keystroke path, and is order-identical + // to the old filter-then-sort for any self-consistent comparator (a total order is not required): + // a stable sort fixes each element's position by (comparator result, original index), and + // Array#filter drops elements without disturbing either, so re-sorting the survivors would + // reproduce the same relative order. filter(sort(x)) === sort(filter(x)). + const sortedBranchEntries = useMemo( + () => [...branchEntries].sort((a, b) => compareFileNames(a.path, b.path)), + [branchEntries] + ) const filteredBranchEntries = useMemo( - () => filterAndSortSourceControlPathEntries(branchEntries, fileFilterState), - [branchEntries, fileFilterState] + () => filterSourceControlPathEntries(sortedBranchEntries, fileFilterState), + [fileFilterState, sortedBranchEntries] ) const treeRootsBySection = useMemo(() => { + if (sourceControlViewMode !== 'tree') { + return EMPTY_TREE_ROOTS_BY_SECTION + } const roots: Partial> = {} for (const section of displaySections) { @@ -148,9 +179,12 @@ export function useSourceControlFileProjection({ : sectionRoots } return roots - }, [displaySections]) + }, [displaySections, sourceControlViewMode]) const visibleTreeRowsBySection = useMemo(() => { + if (sourceControlViewMode !== 'tree') { + return EMPTY_TREE_ROWS_BY_SECTION + } const rows: Partial> = {} for (const section of displaySections) { rows[section.id] = injectExpandedSubmoduleRows( @@ -167,11 +201,15 @@ export function useSourceControlFileProjection({ displaySections, treeRootsBySection, expandedSubmoduleKeys, + sourceControlViewMode, submoduleStatusByKey ]) // List view needs the same lazy submodule expansion as tree view, spliced into the flat entry list. const visibleListRowsBySection = useMemo(() => { + if (sourceControlViewMode !== 'list') { + return EMPTY_LIST_ROWS_BY_SECTION + } const rows: Partial> = {} for (const section of displaySections) { rows[section.id] = injectExpandedSubmoduleEntries( @@ -183,15 +221,21 @@ export function useSourceControlFileProjection({ ) } return rows - }, [displaySections, expandedSubmoduleKeys, submoduleStatusByKey]) + }, [displaySections, expandedSubmoduleKeys, sourceControlViewMode, submoduleStatusByKey]) const branchTreeRoots = useMemo( - () => compactSourceControlTree(buildSourceControlTree('branch', filteredBranchEntries)), - [filteredBranchEntries] + () => + sourceControlViewMode === 'tree' + ? compactSourceControlTree(buildSourceControlTree('branch', filteredBranchEntries)) + : EMPTY_BRANCH_TREE_NODES, + [filteredBranchEntries, sourceControlViewMode] ) const visibleBranchTreeRows = useMemo( - () => flattenSourceControlTree(branchTreeRoots, collapsedTreeDirs), - [branchTreeRoots, collapsedTreeDirs] + () => + sourceControlViewMode === 'tree' + ? flattenSourceControlTree(branchTreeRoots, collapsedTreeDirs) + : EMPTY_BRANCH_TREE_NODES, + [branchTreeRoots, collapsedTreeDirs, sourceControlViewMode] ) const visibleSelectionEntries = useMemo(() => { From 6a5aa1904f2f1c0d5e3329ceed05a006b974a318 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:11:02 -0700 Subject: [PATCH 234/398] perf(renderer): load the project-location and feedback dialogs on click (#18440) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(renderer): load the project-location and feedback dialogs on click Both are reachable only from an explicit click, but their chunks sat on the renderer boot graph and were fetched and parsed on every launch. Route them through the existing `lazy-with-retry` helper, keeping each trigger eager so the click target still exists, and keep the mount sticky once opened so the dialog's own close animation and repeat opens are unaffected. Renderer boot graph 4,473,242 -> 4,424,142 bytes (-49,100 B / -47.9 KiB). Trade-off: the first open per session now waits on a local chunk fetch — measured at ~0.53 ms (project location) and ~0.26 ms (feedback) of read plus V8 parse/compile, warm page cache. * test(renderer): flush the lazy set-location chunk in the ready-target test Without the flush this case only passed because an earlier test in the file had already resolved the shared lazy chunk; it fails under -t filtering. * perf(renderer): warm the lazy dialog chunks on their precursor Both deferred dialogs have a guaranteed, strictly-earlier precursor: the composer only renders "Set location" for a needs-setup host that can take one, and Send Feedback only exists inside an open help menu. Warm each chunk there with a swallowed `import()` (the `preloadCommentMarkdown` pattern) so the fetch/parse happens while the user is reading the picker or the menu, not on the click. Boot graph is unchanged in kind: `import()` never enters modulepreload, so the win holds at -48,958 B (was -49,100 B before the warm; the 142 B is the warm's own source on an already-preloaded chunk). * test(renderer): make the composer warm guard's no-mount assertion real The mock stubbed SetProjectLocationDialog as `() => null`, so the "warming must not mount the dialog" assertion could never fail — the testid it looked for was not rendered under any condition. Render a marker unconditionally instead, matching the sidebar guard, so the assertion actually pins the behaviour. Verified non-vacuous: forcing the lazy element to mount eagerly now fails with "expected
    to be null" rather than passing. * fix(renderer): latch the lazy dialog mounts in state instead of during render React Doctor's ref-mutated-during-render rule failed static analysis on both sticky-mount latches. Use the useState mount-flag idiom already in NewWorkspaceComposerModal (addProjectMounted), set from the open handler. --- ...aceComposerCard.set-location-warm.test.tsx | 123 +++++++++++++++++ ...orkspaceComposerCard.set-location.test.tsx | 130 ++---------------- .../NewWorkspaceComposerCard.test-fixture.tsx | 120 ++++++++++++++++ .../components/NewWorkspaceComposerCard.tsx | 51 +++++-- .../sidebar/SidebarSettingsHelpMenu.test.tsx | 37 ++++- .../sidebar/SidebarSettingsHelpMenu.tsx | 34 ++++- 6 files changed, 359 insertions(+), 136 deletions(-) create mode 100644 src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx create mode 100644 src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx new file mode 100644 index 00000000000..fd1a16f7f0a --- /dev/null +++ b/src/renderer/src/components/NewWorkspaceComposerCard.set-location-warm.test.tsx @@ -0,0 +1,123 @@ +// @vitest-environment happy-dom + +import React from 'react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { hostOptions, renderCard } from './NewWorkspaceComposerCard.test-fixture' +import type { ProjectHostSetupOption } from '@/lib/project-host-setup-options' + +// Counts evaluations of the set-location chunk. A dynamic import evaluates a module once, +// so this only moves when the composer actually reaches for the chunk. +const chunk = vi.hoisted(() => ({ loads: 0 })) + +// Renders a marker unconditionally so the "warming did not mount it" assertion below can +// actually fail; a `() => null` stub would make that check vacuous. +vi.mock('@/components/new-workspace/SetProjectLocationDialog', () => { + chunk.loads += 1 + return { + SetProjectLocationDialog: () =>
    + } +}) + +vi.mock('@/store', () => ({ + useAppStore: Object.assign( + (selector: (state: unknown) => unknown) => + selector({ + closeModal: vi.fn(), + openModal: vi.fn(), + openSettingsPage: vi.fn(), + openSettingsTarget: vi.fn(), + setRuntimeEnvironmentStatus: vi.fn(), + setupProjectExistingFolder: vi.fn(), + setupProjectClone: vi.fn(), + activeModal: 'new-workspace-composer', + settings: { defaultTuiAgent: null, disabledTuiAgents: [] }, + updateSettings: vi.fn(), + projects: [], + repos: [] + }), + { getState: () => ({}) } + ) +})) + +vi.mock('@/components/contextual-tours/use-contextual-tour', () => ({ + useContextualTour: vi.fn() +})) + +vi.mock('@/components/ui/tooltip', () => ({ + Tooltip: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipContent: ({ children }: { children: React.ReactNode }) => <>{children}, + TooltipTrigger: ({ children }: { children: React.ReactNode }) => <>{children} +})) + +vi.mock('@/components/agent/AgentCombobox', () => ({ + default: () => +})) + +vi.mock('@/components/sidebar/AddRemoteHostDialog', () => ({ + AddRemoteHostDialog: () => null +})) + +vi.mock('@/components/sparse/SparseCheckoutPresetSelect', () => ({ + default: () => null +})) + +vi.mock('@/components/new-workspace/SmartWorkspaceNameField', () => ({ + default: () => +})) + +vi.mock('@/components/new-workspace/ProjectCombobox', () => ({ + default: () =>
    +})) + +const readyOnlyHostOptions = hostOptions.filter((option) => option.kind === 'ready') +// A disconnected host is a needs-setup row with no "Set location" action, so it must not warm. +const unavailableHostOptions: ProjectHostSetupOption[] = [ + ...readyOnlyHostOptions, + { + kind: 'needs-setup', + id: 'needs-setup:ssh:offline', + projectId: 'project-group:platform', + hostId: 'ssh:offline', + label: 'Offline box', + detail: 'Not connected', + isAvailable: false, + attention: false, + canSetLocation: false + } +] + +// Declaration order matters here and nowhere else: a module evaluates once, so the +// no-warm cases have to observe the counter before anything warms it. +describe('NewWorkspaceComposerCard set-location chunk warm', () => { + let container: HTMLDivElement | null = null + + afterEach(() => { + container?.remove() + container = null + }) + + it('does not warm the chunk when no host needs its location set', async () => { + container = await renderCard({ projectHostSetupOptions: readyOnlyHostOptions }) + + expect( + [...container.querySelectorAll('button')].some((button) => + button.textContent?.includes('Set project location') + ) + ).toBe(false) + expect(chunk.loads).toBe(0) + }) + + it('does not warm the chunk when the needs-setup host cannot take a location', async () => { + container = await renderCard({ projectHostSetupOptions: unavailableHostOptions }) + + expect(chunk.loads).toBe(0) + }) + + it('warms the chunk on mount for a needs-setup host, before Set project location is clicked', async () => { + container = await renderCard() + + expect(chunk.loads).toBe(1) + // Warming must not mount the dialog; it still waits on an explicit click. + expect(document.body.querySelector('[data-testid="set-project-location-dialog"]')).toBeNull() + }) +}) diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx index 9cdea1d7d54..06718fab91b 100644 --- a/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx +++ b/src/renderer/src/components/NewWorkspaceComposerCard.set-location.test.tsx @@ -1,11 +1,8 @@ // @vitest-environment happy-dom import React, { act } from 'react' -import { createRoot } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import NewWorkspaceComposerCard from './NewWorkspaceComposerCard' -import type { NewWorkspaceProjectOption } from '@/lib/new-workspace-project-options' -import type { ProjectHostSetupOption } from '@/lib/project-host-setup-options' +import { renderCard } from './NewWorkspaceComposerCard.test-fixture' const storeMocks = vi.hoisted(() => ({ closeModal: vi.fn(), @@ -99,118 +96,6 @@ vi.mock('@/components/new-workspace/SetProjectLocationDialog', () => ({ ) : null })) -const projectOptions: NewWorkspaceProjectOption[] = [ - { - kind: 'project-group', - id: 'project-group:platform', - projectGroupId: 'platform', - displayName: 'Platform', - badgeColor: 'var(--muted-foreground)', - detail: '/workspace/platform', - parentPath: '/workspace/platform', - connectionId: null - } -] - -const hostOptions: ProjectHostSetupOption[] = [ - { - kind: 'ready', - id: 'setup-local', - projectId: 'project-group:platform', - hostId: 'local', - repoId: 'repo-a', - label: 'Local Mac', - detail: 'Orca', - path: '/Users/alice/orca' - }, - { - kind: 'needs-setup', - id: 'needs-setup:ssh:devbox', - projectId: 'project-group:platform', - hostId: 'ssh:devbox', - label: 'Devbox', - detail: 'Project location not set', - isAvailable: true, - attention: false, - canSetLocation: true - } -] - -function renderCard( - overrides: Partial> = {} -): HTMLDivElement { - const container = document.createElement('div') - document.body.appendChild(container) - const root = createRoot(container) - act(() => { - root.render( - {}} - eligibleRepos={[]} - repoId="repo-a" - projectOptions={projectOptions} - selectedProjectId="project-group:platform" - selectedRepoIsGit - onRepoChange={() => {}} - onProjectChange={() => {}} - primaryActionLabel="Create workspace" - name="" - onNameValueChange={() => {}} - onSmartGitHubItemSelect={() => {}} - onSmartGitLabItemSelect={() => {}} - onSmartBranchSelect={() => {}} - onSmartLinearIssueSelect={() => {}} - smartNameSelection={null} - onClearSmartNameSelection={() => {}} - canReuseSelectedBranch={false} - reuseSelectedBranch={false} - onReuseSelectedBranchChange={() => {}} - forkPushWarning={null} - detectedAgentIds={null} - onOpenAgentSettings={() => {}} - advancedOpen={false} - onToggleAdvanced={() => {}} - parentWorktreeId={null} - onParentWorktreeIdChange={() => {}} - createDisabled={false} - projectError={null} - creating={false} - onCreate={() => {}} - note="" - onNoteChange={() => {}} - setupConfig={null} - requiresExplicitSetupChoice={false} - setupDecision={null} - onSetupDecisionChange={() => {}} - setupAgentStartupPolicy="start-immediately" - onSetupAgentStartupPolicyChange={() => {}} - shouldWaitForSetupCheck={false} - resolvedSetupDecision={null} - createError={null} - selectedRepoConnectionId={null} - selectedRepoSshStatus={null} - selectedRepoRequiresConnection={false} - selectedRepoConnectInProgress={false} - onConnectSelectedRepo={async () => {}} - canUseSparseCheckout={false} - sparsePresets={[]} - sparseSelectedPresetId={null} - onSparseSelectPreset={() => {}} - branchNameOverride={undefined} - onBranchNameOverrideChange={() => {}} - branchesEnabled={false} - setupControlsEnabled={false} - sparseControlsEnabled={false} - projectHostSetupOptions={hostOptions} - selectedProjectHostSetupId="setup-local" - {...overrides} - /> - ) - }) - return container -} - describe('NewWorkspaceComposerCard set location', () => { let container: HTMLDivElement | null = null @@ -225,9 +110,10 @@ describe('NewWorkspaceComposerCard set location', () => { container = null }) - it('opens set-location over the composer without leaving the create dialog', () => { + // Async because the dialog is a lazy chunk: the click mounts Suspense, the chunk resolves next tick. + it('opens set-location over the composer without leaving the create dialog', async () => { const nestedOpenChanges: boolean[] = [] - container = renderCard({ + container = await renderCard({ onNestedDialogOpenChange: (open) => nestedOpenChanges.push(open) }) @@ -239,6 +125,7 @@ describe('NewWorkspaceComposerCard set location', () => { ) expect(setLocation).toBeTruthy() act(() => setLocation?.click()) + await act(async () => {}) const dialog = document.body.querySelector('[data-testid="set-project-location-dialog"]') expect(dialog?.getAttribute('data-host')).toBe('Devbox') @@ -249,10 +136,12 @@ describe('NewWorkspaceComposerCard set location', () => { expect(storeMocks.openSettingsPage).not.toHaveBeenCalled() }) - it('closes the nested dialog before publishing the ready run target', () => { + // Async for the same reason: without the flush this only passes when an earlier + // test in this file already resolved the shared lazy chunk. + it('closes the nested dialog before publishing the ready run target', async () => { const nestedOpenChanges: boolean[] = [] const setupChanges: string[] = [] - container = renderCard({ + container = await renderCard({ onNestedDialogOpenChange: (open) => nestedOpenChanges.push(open), onProjectHostSetupChange: (setupId) => setupChanges.push(setupId) }) @@ -264,6 +153,7 @@ describe('NewWorkspaceComposerCard set location', () => { (button) => button.textContent?.includes('Set project location') ) act(() => setLocation?.click()) + await act(async () => {}) const complete = [...document.body.querySelectorAll('button')].find( (button) => button.textContent === 'Complete location' ) diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx new file mode 100644 index 00000000000..c8eb2b56f43 --- /dev/null +++ b/src/renderer/src/components/NewWorkspaceComposerCard.test-fixture.tsx @@ -0,0 +1,120 @@ +import React, { act } from 'react' +import { createRoot } from 'react-dom/client' +import NewWorkspaceComposerCard from './NewWorkspaceComposerCard' +import type { NewWorkspaceProjectOption } from '@/lib/new-workspace-project-options' +import type { ProjectHostSetupOption } from '@/lib/project-host-setup-options' + +export const projectOptions: NewWorkspaceProjectOption[] = [ + { + kind: 'project-group', + id: 'project-group:platform', + projectGroupId: 'platform', + displayName: 'Platform', + badgeColor: 'var(--muted-foreground)', + detail: '/workspace/platform', + parentPath: '/workspace/platform', + connectionId: null + } +] + +export const hostOptions: ProjectHostSetupOption[] = [ + { + kind: 'ready', + id: 'setup-local', + projectId: 'project-group:platform', + hostId: 'local', + repoId: 'repo-a', + label: 'Local Mac', + detail: 'Orca', + path: '/Users/alice/orca' + }, + { + kind: 'needs-setup', + id: 'needs-setup:ssh:devbox', + projectId: 'project-group:platform', + hostId: 'ssh:devbox', + label: 'Devbox', + detail: 'Project location not set', + isAvailable: true, + attention: false, + canSetLocation: true + } +] + +export async function renderCard( + overrides: Partial> = {} +): Promise { + const container = document.createElement('div') + document.body.appendChild(container) + const root = createRoot(container) + act(() => { + root.render( + {}} + eligibleRepos={[]} + repoId="repo-a" + projectOptions={projectOptions} + selectedProjectId="project-group:platform" + selectedRepoIsGit + onRepoChange={() => {}} + onProjectChange={() => {}} + primaryActionLabel="Create workspace" + name="" + onNameValueChange={() => {}} + onSmartGitHubItemSelect={() => {}} + onSmartGitLabItemSelect={() => {}} + onSmartBranchSelect={() => {}} + onSmartLinearIssueSelect={() => {}} + smartNameSelection={null} + onClearSmartNameSelection={() => {}} + canReuseSelectedBranch={false} + reuseSelectedBranch={false} + onReuseSelectedBranchChange={() => {}} + forkPushWarning={null} + detectedAgentIds={null} + onOpenAgentSettings={() => {}} + advancedOpen={false} + onToggleAdvanced={() => {}} + parentWorktreeId={null} + onParentWorktreeIdChange={() => {}} + createDisabled={false} + projectError={null} + creating={false} + onCreate={() => {}} + note="" + onNoteChange={() => {}} + setupConfig={null} + requiresExplicitSetupChoice={false} + setupDecision={null} + onSetupDecisionChange={() => {}} + setupAgentStartupPolicy="start-immediately" + onSetupAgentStartupPolicyChange={() => {}} + shouldWaitForSetupCheck={false} + resolvedSetupDecision={null} + createError={null} + selectedRepoConnectionId={null} + selectedRepoSshStatus={null} + selectedRepoRequiresConnection={false} + selectedRepoConnectInProgress={false} + onConnectSelectedRepo={async () => {}} + canUseSparseCheckout={false} + sparsePresets={[]} + sparseSelectedPresetId={null} + onSparseSelectPreset={() => {}} + branchNameOverride={undefined} + onBranchNameOverrideChange={() => {}} + branchesEnabled={false} + setupControlsEnabled={false} + sparseControlsEnabled={false} + projectHostSetupOptions={hostOptions} + selectedProjectHostSetupId="setup-local" + {...overrides} + /> + ) + }) + // Settle the mount-time chunk warm before the click, so the click's import() is not + // overlapping an in-flight one (vitest's module runner serialises those; a browser does not). + await act(async () => {}) + return container +} diff --git a/src/renderer/src/components/NewWorkspaceComposerCard.tsx b/src/renderer/src/components/NewWorkspaceComposerCard.tsx index c6cca5192ea..d17bb8ba56d 100644 --- a/src/renderer/src/components/NewWorkspaceComposerCard.tsx +++ b/src/renderer/src/components/NewWorkspaceComposerCard.tsx @@ -11,7 +11,8 @@ import { AddRemoteHostDialog, type AddRemoteHostMode } from '@/components/sidebar/AddRemoteHostDialog' -import { SetProjectLocationDialog } from '@/components/new-workspace/SetProjectLocationDialog' +import { lazyWithRetry } from '@/lib/lazy-with-retry' +import type * as SetProjectLocationDialogModule from '@/components/new-workspace/SetProjectLocationDialog' import { unwrapRuntimeRpcResult } from '@/runtime/runtime-rpc-client' import { withUiConnectTimeout } from '@/ssh/ssh-connect-ui-timeout' import { isSshConnectInFlight, trackSshConnect } from '@/ssh/ssh-connect-in-flight' @@ -37,6 +38,20 @@ import { import { getSshStatusLabel } from './new-workspace/new-workspace-composer-ssh-status' import { useComposerFileDragOver } from './new-workspace/use-composer-file-drag-over' +// Why lazy: this pulls the ~41 KB project-location browser onto the boot graph, and nothing +// reaches it without an explicit "Set location" click. Shared with the warm below so both hit +// the same module-map entry. +const loadSetProjectLocationDialog = (): Promise => + import('@/components/new-workspace/SetProjectLocationDialog') + +const SetProjectLocationDialog = lazyWithRetry( + () => + loadSetProjectLocationDialog().then((module) => ({ + default: module.SetProjectLocationDialog + })), + { reloadKey: 'set-project-location-dialog' } +) + export default function NewWorkspaceComposerCard( props: NewWorkspaceComposerCardProps ): React.JSX.Element { @@ -83,6 +98,9 @@ export default function NewWorkspaceComposerCard( const [setLocationOption, setSetLocationOption] = React.useState( null ) + // Why sticky: the dialog animates itself closed off its own `option` prop, so unmounting it + // when the option clears would cut that animation short. + const [setLocationDialogMounted, setSetLocationDialogMounted] = React.useState(false) const selectedRepo = eligibleRepos.find((candidate) => candidate.id === repoId) const selectedRepoName = selectedRepo?.displayName ?? selectedRepo?.path ?? 'This project' @@ -96,6 +114,16 @@ export default function NewWorkspaceComposerCard( const needsSetupProjectHostSetupOptions = projectHostSetupOptions.filter( (option) => option.kind === 'needs-setup' ) + // Warm on the precursor: the "Set location" row only renders for a needs-setup host that can + // still take one, so the chunk resolves while the picker is being read rather than on the click. + const hasSetLocationOption = needsSetupProjectHostSetupOptions.some( + (option) => option.canSetLocation + ) + React.useEffect(() => { + if (hasSetLocationOption) { + void loadSetProjectLocationDialog().catch(() => {}) + } + }, [hasSetLocationOption]) const shouldShowRunTargetPicker = readyProjectHostSetupOptions.length > 0 || ephemeralVmRecipes.length > 0 || @@ -177,6 +205,7 @@ export default function NewWorkspaceComposerCard( }, [onAddProjectOverride, openModal]) const handleSetLocation = React.useCallback( (option: NeedsProjectHostOption): void => { + setSetLocationDialogMounted(true) setSetLocationOption(option) onNestedDialogOpenChange?.(true) }, @@ -319,14 +348,18 @@ export default function NewWorkspaceComposerCard( submitShortcutModifierLabel={getScreenSubmitModifierLabel()} /> - + {setLocationDialogMounted ? ( + + + + ) : null}
    ) } diff --git a/src/renderer/src/components/sidebar/SidebarSettingsHelpMenu.test.tsx b/src/renderer/src/components/sidebar/SidebarSettingsHelpMenu.test.tsx index b88bc1c102e..be0cbe27f97 100644 --- a/src/renderer/src/components/sidebar/SidebarSettingsHelpMenu.test.tsx +++ b/src/renderer/src/components/sidebar/SidebarSettingsHelpMenu.test.tsx @@ -14,6 +14,8 @@ const mocks = vi.hoisted(() => ({ updaterCheck: vi.fn(), shellOpenUrl: vi.fn(), useShortcutKeyDetails: vi.fn(), + /** Counts evaluations of the feedback chunk; a dynamic import evaluates it exactly once. */ + feedbackChunkLoads: 0, setupProgress: { ready: true, coreDoneCount: 2, @@ -56,7 +58,18 @@ vi.mock('../setup-guide/SetupGuideProgressRing', () => ({ })) vi.mock('@/components/ui/dropdown-menu', () => ({ - DropdownMenu: ({ children }: { children: ReactNode }) => <>{children}, + DropdownMenu: ({ + children, + onOpenChange + }: { + children: ReactNode + onOpenChange?: (open: boolean) => void + }) => ( + <> +
    - {!providerMissing ? ( + {!providerMissing && !reconnectRequired ? ( + + +
    +
    {title}
    +
    {detected.unavailableReason}
    +
    + {translate( + 'auto.components.sidebar.RepoScanUnavailableIndicator.retained', + 'Existing worktrees are kept until a scan succeeds. Click to retry.' + )} +
    +
    +
    + + ) +} diff --git a/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx b/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx index b4ed149712a..db0c7ec1371 100644 --- a/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx +++ b/src/renderer/src/components/sidebar/worktree-list/rows/SectionHeader.tsx @@ -26,6 +26,7 @@ import { WORKTREE_SECTION_HEADER_PADDING_LEFT } from './indentation' import { FolderPathStatusIndicator } from './FolderPathStatusIndicator' +import { RepoScanUnavailableIndicator } from './RepoScanUnavailableIndicator' import { ProjectGroupCreateWorkspaceButton, ProjectGroupHeaderMenu @@ -334,6 +335,7 @@ export function renderWorktreeSectionHeaderRow(args: {
    + {isRepoHeader ? : null}

    diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index e8bc924f64e..3c468d20929 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -6231,6 +6231,11 @@ "projectOnly": "Added in this project only.", "useGlobalFor": "Use global for {{value0}}", "useGlobal": "Use global" + }, + "RepoScanUnavailableIndicator": { + "title": "Worktree scan failed for {{value0}}", + "retry": "Retry scan", + "retained": "Existing worktrees are kept until a scan succeeds. Click to retry." } }, "shared": { diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts index 073bd712f32..f8078a9925c 100644 --- a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-host-merge.ts @@ -22,6 +22,7 @@ export function mergeDetectedWorktreesForHost( current.repoId === refreshed.repoId && current.authoritative === refreshed.authoritative && current.source === refreshed.source && + current.unavailableReason === refreshed.unavailableReason && current.worktrees === worktrees ) { return current diff --git a/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts new file mode 100644 index 00000000000..da44f24c0d3 --- /dev/null +++ b/src/renderer/src/store/slices/worktrees/listing/detected-worktree-unavailable-reason.test.ts @@ -0,0 +1,31 @@ +import { describe, expect, it } from 'vitest' +import { makeDetectedResult } from '../../worktrees-detected-listing-fixtures' +import { mergeDetectedWorktreesForHost } from './detected-worktree-host-merge' +import { areDetectedWorktreeResultsEqual } from './worktree-catalog-visibility' + +const failed = (unavailableReason?: string) => + makeDetectedResult('repo-1', [], { + authoritative: false, + source: 'metadata-fallback', + ...(unavailableReason ? { unavailableReason } : {}) + }) + +// Why: two failed scans differ only by cause; dropping that from equality would freeze the first +// reason on the header until the listing's rows or authority changed. +describe('detected listing unavailable reason', () => { + it('is part of listing equality', () => { + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed('distro gone'))).toBe(true) + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed('mount hung'))).toBe(false) + expect(areDetectedWorktreeResultsEqual(failed('distro gone'), failed())).toBe(false) + }) + + it('survives the host merge when only the reason changed', () => { + const merged = mergeDetectedWorktreesForHost( + failed('distro gone'), + failed('mount hung'), + 'local' + ) + + expect(merged.unavailableReason).toBe('mount hung') + }) +}) diff --git a/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts b/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts index ce7a13c4b25..c6d3a9636f7 100644 --- a/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts +++ b/src/renderer/src/store/slices/worktrees/listing/worktree-catalog-visibility.ts @@ -14,6 +14,7 @@ export function areDetectedWorktreeResultsEqual( current.repoId === next.repoId && current.authoritative === next.authoritative && current.source === next.source && + current.unavailableReason === next.unavailableReason && catalogRowsEqual(current.worktrees, next.worktrees) ) } diff --git a/src/shared/worktree/types.ts b/src/shared/worktree/types.ts index 3e5c05a65bc..e368716dc01 100644 --- a/src/shared/worktree/types.ts +++ b/src/shared/worktree/types.ts @@ -221,4 +221,6 @@ export type DetectedWorktreeListResult = { authoritative: boolean source: DetectedWorktreeListSource worktrees: DetectedWorktree[] + /** Why a non-authoritative listing could not be scanned; additive, older hosts omit it. */ + unavailableReason?: string } From c16913ab659ca3482fc77dda0698240f7c3fb22e Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Fri, 4 Sep 2026 15:29:36 -0700 Subject: [PATCH 304/398] fix(native-chat): clarify active progress (#18705) Co-authored-by: Merge Sim --- .../NativeChatMessageList.test.tsx | 8 ++--- .../native-chat/NativeChatToolRun.test.tsx | 4 ++- .../native-chat/NativeChatToolRun.tsx | 2 +- .../native-chat/NativeChatWorkingStatus.tsx | 23 ++++++++++--- ...-chat-working-status-shared-clock.test.tsx | 34 ++++++++++++++----- src/renderer/src/i18n/locales/en.json | 4 +-- 6 files changed, 54 insertions(+), 21 deletions(-) diff --git a/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx b/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx index 9695317f3af..860464ac65b 100644 --- a/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatMessageList.test.tsx @@ -215,7 +215,7 @@ describe('NativeChatMessageList assistant messages', () => { ) const user = screen.getByText('Run the checks') - const status = screen.getByText('Working for 0 seconds') + const status = screen.getByText('Working for 0s') const assistant = screen.getByText('I am checking now.') expect(user.compareDocumentPosition(status)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) expect(status.compareDocumentPosition(assistant)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) @@ -252,7 +252,7 @@ describe('NativeChatMessageList assistant messages', () => { /> ) - expect(screen.getByText('Working for 3 seconds')).toBeInTheDocument() + expect(screen.getByText('Working for 3s')).toBeInTheDocument() }) it('keeps the completed duration below the user message', () => { @@ -298,7 +298,7 @@ describe('NativeChatMessageList assistant messages', () => { ) const user = screen.getByText('Complete this task') - const status = screen.getByText('Worked for 3 seconds') + const status = screen.getByText('Worked for 3s') const assistant = screen.getByText('Task complete.') expect(user.compareDocumentPosition(status)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) expect(status.compareDocumentPosition(assistant)).toBe(Node.DOCUMENT_POSITION_FOLLOWING) @@ -326,7 +326,7 @@ describe('NativeChatMessageList assistant messages', () => { /> ) - expect(screen.getByText('Worked for 3 seconds')).toBeInTheDocument() + expect(screen.getByText('Worked for 3s')).toBeInTheDocument() expect(screen.getByText('Thinking')).toBeInTheDocument() }) diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx index 41a70a8457d..965ed0ad176 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.test.tsx @@ -99,7 +99,9 @@ describe('NativeChatToolRun', () => { const { container } = render() - expect(screen.getByText('Running cat package.json')).toBeInTheDocument() + const activeLabel = screen.getByText('Running cat package.json') + expect(activeLabel).toBeInTheDocument() + expect(activeLabel).toHaveClass('animate-pulse', 'motion-reduce:animate-none') expect(screen.queryByText('Running date')).toBeNull() expect(screen.queryByText('Running pwd')).toBeNull() expect(screen.queryByText('Ran 3 commands and used 1 tool')).toBeNull() diff --git a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx index 716d293838e..f16d01331f8 100644 --- a/src/renderer/src/components/native-chat/NativeChatToolRun.tsx +++ b/src/renderer/src/components/native-chat/NativeChatToolRun.tsx @@ -227,7 +227,7 @@ export function NativeChatToolRun({ - + {activeToolLabel(latestActiveCall)} {open ? : null} diff --git a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx index a883429f658..21145e94c94 100644 --- a/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx +++ b/src/renderer/src/components/native-chat/NativeChatWorkingStatus.tsx @@ -3,6 +3,21 @@ import { ChevronRight } from 'lucide-react' import { translate } from '@/i18n/i18n' import { useNow } from '@/hooks/use-now' +/** Format turn time without exposing an ever-growing raw seconds count. */ +export function formatNativeChatDuration(seconds: number): string { + const totalSeconds = Number.isFinite(seconds) ? Math.max(0, Math.floor(seconds)) : 0 + if (totalSeconds < 60) { + return `${totalSeconds}s` + } + const minutes = Math.floor(totalSeconds / 60) + const remainingSeconds = totalSeconds % 60 + if (minutes < 60) { + return `${minutes}m ${remainingSeconds}s` + } + const hours = Math.floor(minutes / 60) + return `${hours}h ${minutes % 60}m ${remainingSeconds}s` +} + export function NativeChatWorkingStatus({ startedAt, thinking, @@ -30,13 +45,13 @@ export function NativeChatWorkingStatus({ const label = workedSeconds != null - ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}} seconds', { - value0: workedSeconds + ? translate('components.native-chat.status.workedFor', 'Worked for {{value0}}', { + value0: formatNativeChatDuration(workedSeconds) }) : thinking ? translate('components.native-chat.status.thinking', 'Thinking') - : translate('components.native-chat.status.workingFor', 'Working for {{value0}} seconds', { - value0: elapsedSeconds + : translate('components.native-chat.status.workingFor', 'Working for {{value0}}', { + value0: formatNativeChatDuration(elapsedSeconds) }) const className = `flex min-h-8 items-center gap-1 text-sm text-muted-foreground${thinking ? '' : ' border-b border-border'}` diff --git a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx index dfea16ae0e4..e4bae292004 100644 --- a/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx +++ b/src/renderer/src/components/native-chat/native-chat-working-status-shared-clock.test.tsx @@ -3,7 +3,7 @@ import { act } from 'react' import { createRoot, type Root } from 'react-dom/client' import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { NativeChatWorkingStatus } from './NativeChatWorkingStatus' +import { formatNativeChatDuration, NativeChatWorkingStatus } from './NativeChatWorkingStatus' let container: HTMLDivElement let root: Root @@ -49,34 +49,50 @@ afterEach(() => { }) describe('native chat working status elapsed clock', () => { + it.each([ + [0, '0s'], + [59, '59s'], + [60, '1m 0s'], + [69, '1m 9s'], + [3_725, '1h 2m 5s'] + ])('formats %s seconds as %s', (seconds, expected) => { + expect(formatNativeChatDuration(seconds)).toBe(expected) + }) + + it('renders the compact duration in the completed status label', () => { + act(() => { + root.render( + + ) + }) + + expect(elapsedLabels()).toEqual(['Worked for 1m 9s']) + }) + it('collapses every in-flight turn onto one shared visibility-gated timer', () => { renderTurns(3, 1_000_000) // One shared 1s clock for all three turns, not one interval per turn. expect(vi.getTimerCount()).toBe(1) act(() => vi.advanceTimersByTime(3_000)) - expect(elapsedLabels()).toEqual([ - 'Working for 3 seconds', - 'Working for 3 seconds', - 'Working for 3 seconds' - ]) + expect(elapsedLabels()).toEqual(['Working for 3s', 'Working for 3s', 'Working for 3s']) }) it('stops ticking while hidden and re-syncs the elapsed value on return', () => { renderTurns(1, 1_000_000) act(() => vi.advanceTimersByTime(3_000)) - expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + expect(elapsedLabels()).toEqual(['Working for 3s']) setDocumentVisibility('hidden') expect(vi.getTimerCount()).toBe(0) // A minute of hidden wall-clock: no callbacks, no commits, label frozen. act(() => vi.advanceTimersByTime(60_000)) - expect(elapsedLabels()).toEqual(['Working for 3 seconds']) + expect(elapsedLabels()).toEqual(['Working for 3s']) // Returning re-derives elapsed from startedAt, so nothing was lost. setDocumentVisibility('visible') - expect(elapsedLabels()).toEqual(['Working for 63 seconds']) + expect(elapsedLabels()).toEqual(['Working for 1m 3s']) expect(vi.getTimerCount()).toBe(1) }) diff --git a/src/renderer/src/i18n/locales/en.json b/src/renderer/src/i18n/locales/en.json index 3c468d20929..8a6e54d2d11 100644 --- a/src/renderer/src/i18n/locales/en.json +++ b/src/renderer/src/i18n/locales/en.json @@ -16957,8 +16957,8 @@ "responding": "Agent is responding", "working": "Working…", "thinking": "Thinking", - "workingFor": "Working for {{value0}} seconds", - "workedFor": "Worked for {{value0}} seconds", + "workingFor": "Working for {{value0}}", + "workedFor": "Worked for {{value0}}", "toggleDetails": "Toggle turn details" }, "jumpToLatest": "Jump to latest", From 264c9ed8d27b6ff50d0e38e9c1dd2540539c7bd4 Mon Sep 17 00:00:00 2001 From: Neil <4138956+nwparker@users.noreply.github.com> Date: Fri, 4 Sep 2026 15:38:00 -0700 Subject: [PATCH 305/398] fix(browser-pane): stop a dead client-hosted guest from killing the workbench (#18334) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(browser): stop a dead client-hosted guest taking down the workbench ClientHostedBrowserPagePane called raw methods from two React effects, so a guest that is gone throws out of a commit phase and unwinds the terminal.workbench error boundary instead of showing the pane's own unavailable notice. Two runtime conditions, two guards: - Guest destroyed in main while the tag is still in the DOM: the tag keeps its guestInstanceId, so every read throws 'Invalid guestInstanceId'. The metadata read is now total and the attach effect degrades to browser_client_page_guest_unavailable. - Retained tag removed from the DOM while the pane stays mounted: contentWindow is null, so focus() throws a TypeError. The activation-focus hook now goes through the BrowserPageGuestFocus wrapper the pane already builds, which has carried that guard since STA-3448. Follow-up, not in this change: the registry's liveness check compares readBrowserClientPageAttachedGuestId(webview) to page.webContentsId, which still matches after main destroys the guest, so a stale 'attached' page can linger. * fix(browser): keep the dead-guest degrade honest — no spinner, no silent swallow Review remediation for the guest guards. - The attach bail now writes `loading: false` before setting browser_client_page_guest_unavailable, matching what retryGuestRecoveryRef already does on the pane's own route into that state. A page that died mid-load carries `loading: true` in the store, so without it the unavailable notice rendered beside a spinner nothing would ever stop. Uses the existing updatePageStateFromGuest effect event, not a new setter. - The dead-guest condition is no longer silent: the metadata read logs the caught error under the subsystem's `[browser-client-page]` warn convention, and the attach bail records a `browser_client_page_guest_unavailable` crash breadcrumb via the existing recordRendererCrashBreadcrumb. Renderer diagnostics only capture window error/rejection, so the breadcrumb is what puts this on a channel crash reports actually carry — the registry liveness defect this change deliberately does not fix stays measurable, and a read failure that is not `Invalid guestInstanceId` is no longer indistinguishable from a dead guest. - onFailLoad no longer pays five sync IPCs per discarded event: resolveBrowserWebviewLoadFailure accepts a lazy fallbackUrl and resolves it after the subframe/ERR_ABORTED filter. Covered by a new case in browser-webview-load-failure.test.ts. Correction to the previous commit's narrative: only the metadata half reaches browser_client_page_guest_unavailable. The focus half reaches no state at all — the guarded BrowserPageGuestFocus wrapper returns false and the pane stays mounted over a webview the registry already removed from the DOM, with no notice and no reopen-on-server escape. It stops the crash; it does not diagnose the page. Not changed, with reasons: - The two sibling attach bails (renderer-unavailable, attach threw) omit the same `loading` write. That predates this branch and neither is reached by the dead-guest path; fixing them is a separate change. - recordHistoryFromGuest still passes a raw `webview.getTitle()`. Substituting `metadata.title` is not behaviour-preserving: metadata.title falls back to the URL, so an untitled page would be filed in history under its URL instead of "New Tab". The call runs only after a five-read succeeded and sits in a DOM event listener, which cannot unwind a React commit. * fix(browser): route every client-hosted guest death to the unavailable notice A dead guest could still leave the pane mute (retained tag fenced on render-process-gone/destroyed with no signal to the pane), spinning forever (did-start-loading read bailing after loading:true), or frozen at stale chrome (navigation reads bailing silently). All of those now go through one watcher that detaches, releases the webview ref and enters the pane's existing browser_client_page_guest_unavailable recovery state, so the user always sees the notice with its reopen-on-server escape. The total catch in the guest reader now records a browser_client_page_guest_read_failed breadcrumb with the error name/message, so a swallowed failure that is not guest death stays distinguishable in diagnostics; the guest_unavailable breadcrumb carries the loss reason. * fix(browser): finish dead guest cleanup and guard history reads --- ...tHostedBrowserPagePane.dead-guest.test.tsx | 310 ++++++++++++++++++ .../ClientHostedBrowserPagePane.tsx | 109 +++--- .../browser-client-page-guest-metadata.ts | 63 +++- .../browser-client-page-guest-loss.ts | 54 +++ ...se-client-hosted-guest-activation-focus.ts | 10 +- .../browser-webview-load-failure.test.ts | 14 +- .../navigate/browser-webview-load-failure.ts | 9 +- 7 files changed, 505 insertions(+), 64 deletions(-) create mode 100644 src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx create mode 100644 src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts diff --git a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx new file mode 100644 index 00000000000..f2fd2ad533b --- /dev/null +++ b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.dead-guest.test.tsx @@ -0,0 +1,310 @@ +// @vitest-environment happy-dom +import { act, cleanup, render, screen } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { BrowserPage } from '../../../../shared/browser-workspace-types' + +const mocks = vi.hoisted(() => ({ + attach: vi.fn(), + detach: vi.fn(), + recordBreadcrumb: vi.fn() +})) + +vi.mock('./browser-client-page-renderer-installation', () => ({ + attachBrowserClientPageToViewport: mocks.attach +})) +vi.mock('@/lib/crash-breadcrumb-recorder', () => ({ + recordRendererCrashBreadcrumb: mocks.recordBreadcrumb +})) +vi.mock('sonner', () => ({ + toast: { error: vi.fn(), success: vi.fn(), loading: vi.fn(), message: vi.fn() } +})) + +import { TooltipProvider } from '@/components/ui/tooltip' +import { installClientHostedPaneApi } from './client-hosted-browser-pane-test-rig' +import { ClientHostedBrowserPagePane } from './ClientHostedBrowserPagePane' + +const PLACEMENT = { + kind: 'client' as const, + browserHostClientId: 'host-a', + browserHostGeneration: 3, + pageHostGeneration: 7 +} + +/** Verbatim from Electron 43.4.1: main destroyed the guest, the tag still holds its id. */ +function invalidGuestInstanceId(): Error { + return new Error('Invalid guestInstanceId: 7') +} + +/** Verbatim from Electron 43.4.1: focus() after the retained tag left the DOM. */ +function nullContentWindowFocus(): TypeError { + return new TypeError("Cannot read properties of null (reading 'focus')") +} + +function page(overrides?: Partial): BrowserPage { + return { + id: 'page-a', + workspaceId: 'workspace-a', + worktreeId: 'worktree-a', + url: 'https://example.internal/', + title: 'Example', + loading: false, + faviconUrl: null, + canGoBack: false, + canGoForward: false, + loadError: null, + createdAt: 1, + ...overrides + } +} + +function createGuest(): Electron.WebviewTag & { + getURL: ReturnType + reload: ReturnType +} { + const webview = document.createElement('webview') as Electron.WebviewTag & { + getURL: ReturnType + reload: ReturnType + } + Object.assign(webview, { + getURL: vi.fn(() => 'https://example.internal/'), + getTitle: vi.fn(() => 'Example'), + isLoading: vi.fn(() => false), + canGoBack: vi.fn(() => false), + canGoForward: vi.fn(() => false), + focus: vi.fn(), + blur: vi.fn(), + goBack: vi.fn(), + goForward: vi.fn(), + reload: vi.fn(), + loadURL: vi.fn(async () => {}) + }) + mocks.attach.mockReturnValue({ + webview, + detach: mocks.detach, + nextMetadataRevision: vi.fn(() => 1) + }) + return webview +} + +function paneElement( + isActive: boolean, + options?: { browserTab?: BrowserPage; onUpdatePageState?: (id: string, state: unknown) => void } +): React.JSX.Element { + return ( + + + + ) +} + +let webview: ReturnType + +beforeEach(() => { + mocks.attach.mockReset() + mocks.detach.mockReset() + mocks.recordBreadcrumb.mockReset() + installClientHostedPaneApi() + webview = createGuest() +}) + +afterEach(() => { + cleanup() + vi.clearAllMocks() +}) + +describe('client-hosted browser pane over a dead guest', () => { + it('degrades to the unavailable notice when the guest was destroyed in main', () => { + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + expect(() => render(paneElement(true))).not.toThrow() + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.detach).toHaveBeenCalled() + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_unavailable', { + browserPageId: 'page-a', + pageHostGeneration: PLACEMENT.pageHostGeneration, + reason: 'unreadable', + tagConnected: false + }) + // Why: the catch is total, so the swallowed error must stay visible to diagnostics. + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_read_failed', { + errorName: 'Error', + errorMessage: 'Invalid guestInstanceId: 7' + }) + }) + + it('stops the spinner it inherited from a page that died mid-load', () => { + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + const onUpdatePageState = vi.fn() + + render(paneElement(true, { browserTab: page({ loading: true }), onUpdatePageState })) + + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + }) + + it('flips to the unavailable notice when the guest renderer goes away after attach', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { browserTab: page({ loading: true }), onUpdatePageState })) + expect(screen.queryByText('Client-hosted browser unavailable')).toBeNull() + onUpdatePageState.mockClear() + + // The registry pulls the tag out of the DOM on this event without telling the pane. + webview.remove() + act(() => { + webview.dispatchEvent(new Event('render-process-gone')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.detach).toHaveBeenCalled() + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith('browser_client_page_guest_unavailable', { + browserPageId: 'page-a', + pageHostGeneration: PLACEMENT.pageHostGeneration, + reason: 'render-process-gone', + tagConnected: false + }) + }) + + it('flips to the unavailable notice when main destroys the guest after attach', () => { + render(paneElement(true)) + + act(() => { + webview.dispatchEvent(new Event('destroyed')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(mocks.recordBreadcrumb).toHaveBeenCalledWith( + 'browser_client_page_guest_unavailable', + expect.objectContaining({ reason: 'destroyed' }) + ) + // The chrome must not keep driving the dead tag: Reload routes to the notice, not a throw. + webview.reload.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + expect(() => act(() => screen.getByRole('button', { name: 'Reload' }).click())).not.toThrow() + expect(webview.reload).not.toHaveBeenCalled() + }) + + it('does not freeze silently when a navigation event finds the guest gone', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + onUpdatePageState.mockClear() + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => { + webview.dispatchEvent(new Event('did-navigate')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(onUpdatePageState).toHaveBeenCalledWith('page-a', { loading: false }) + }) + + it('stops the spinner when the guest dies as a load starts', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + onUpdatePageState.mockClear() + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => { + webview.dispatchEvent(new Event('did-start-loading')) + }) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + // did-start-loading writes loading:true first; the loss must be the last word. + expect(onUpdatePageState.mock.calls.at(-1)).toEqual(['page-a', { loading: false }]) + }) + + it('ignores queued load events after guest loss', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + act(() => webview.dispatchEvent(new Event('destroyed'))) + onUpdatePageState.mockClear() + + act(() => webview.dispatchEvent(new Event('did-start-loading'))) + + expect(onUpdatePageState).not.toHaveBeenCalled() + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + }) + + it('uses the guarded title snapshot if the guest dies immediately afterward', () => { + render(paneElement(true)) + webview.getTitle = vi.fn(() => { + webview.getTitle = vi.fn(() => { + throw invalidGuestInstanceId() + }) + return 'Last live title' + }) + + expect(() => act(() => webview.dispatchEvent(new Event('did-navigate')))).not.toThrow() + expect(webview.getTitle).not.toHaveBeenCalled() + }) + + it('shows unavailability when a load-failure fallback finds the guest gone', () => { + const onUpdatePageState = vi.fn() + render(paneElement(true, { onUpdatePageState })) + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + act(() => webview.dispatchEvent(new Event('did-fail-load'))) + + expect(screen.getByText('Client-hosted browser unavailable')).toBeTruthy() + expect(onUpdatePageState.mock.calls.at(-1)).toEqual(['page-a', { loading: false }]) + }) + + it('removes loss listeners when the initial guest read fails', () => { + const removeListener = vi.spyOn(webview, 'removeEventListener') + webview.getURL.mockImplementation(() => { + throw invalidGuestInstanceId() + }) + + render(paneElement(true)) + + expect(removeListener).toHaveBeenCalledWith('destroyed', expect.any(Function)) + expect(removeListener).toHaveBeenCalledWith('render-process-gone', expect.any(Function)) + }) + + it('stops listening for guest loss once the pane lets go of the tag', () => { + const onUpdatePageState = vi.fn() + const view = render(paneElement(true, { onUpdatePageState })) + view.unmount() + onUpdatePageState.mockClear() + mocks.recordBreadcrumb.mockClear() + + webview.dispatchEvent(new Event('destroyed')) + + expect(onUpdatePageState).not.toHaveBeenCalled() + expect(mocks.recordBreadcrumb).not.toHaveBeenCalled() + }) + + it('survives activation focus after the retained tag left the DOM', () => { + const view = render(paneElement(false)) + webview.focus = vi.fn(() => { + throw nullContentWindowFocus() + }) + + expect(() => + act(() => { + view.rerender(paneElement(true)) + }) + ).not.toThrow() + expect(webview.focus).toHaveBeenCalled() + }) +}) diff --git a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx index f4f698f516b..349886a570f 100644 --- a/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx +++ b/src/renderer/src/components/browser-pane/ClientHostedBrowserPagePane.tsx @@ -7,7 +7,10 @@ import type { } from '../../../../shared/browser-workspace-types' import { toHttpsRecoveryUrl } from '../../../../shared/browser-url' import type { RuntimeBrowserClientPlacement } from '../../../../shared/runtime-browser-placement' -import { readBrowserClientPageGuestMetadata } from './browser-client-page-guest-metadata' +import { + readBrowserClientPageGuestMetadataIfLive, + createBrowserClientPageLoadFailureHandler +} from './browser-client-page-guest-metadata' import { forgetBrowserClientPageMetadataReports, startBrowserClientPageMetadataPublisher @@ -18,6 +21,7 @@ import { useBrowserClientHostedPopupNotices } from './browser-client-hosted-popu import { useBrowserClientHostedPermissionNotices } from './browser-client-hosted-permission-notices' import { useClientHostedBrowserIntroTour } from './use-client-hosted-browser-intro-tour' import { ClientHostedBrowserUnavailableNotice } from './client-hosted-browser-unavailable-notice' +import { watchBrowserClientPageGuestLoss } from './host-guest/browser-client-page-guest-loss' import { useRestoredClientHostedRecoveryWindow } from './restored-client-hosted-recovery-window' import BrowserFind from './assemble-chrome/BrowserFind' import { BrowserNavigationControlRow } from './assemble-chrome/browser-navigation-control-row' @@ -36,7 +40,6 @@ import { BrowserLoadFailureOverlay } from './navigate/browser-load-failure-overl import { useClientHostedPageUrlSubmission } from './navigate/use-client-hosted-page-url-submission' import { convertBrowserPageToWorkspaceDoc } from '@/lib/file-preview' import { useBrowserPageReloadActions } from './navigate/use-browser-page-reload-actions' -import { resolveBrowserWebviewLoadFailure } from './navigate/browser-webview-load-failure' import { resolveActiveBrowserLoadFailure } from './navigate/browser-load-failure-for-url' import { consumeBrowserPageDeferredNavigation } from './navigate/browser-page-deferred-navigation' import { @@ -46,7 +49,6 @@ import { } from './describe-page/browser-page-url-display' import type { BrowserChromeShortcutScope, - BrowserPageFailLoadEvent, BrowserPageUrlSetter, BrowserTabPageState } from './describe-page/browser-page-types' @@ -174,9 +176,7 @@ export function ClientHostedBrowserPagePane({ useLayoutEffect(() => { const viewport = viewportRef.current - // Why: no placement means the host has not minted this page yet. Attaching would throw for an - // id the retained registry has never seen and strand the pane on the unavailable notice, whose - // only exit is reopening on the server — so mount quiet and wait for adoption to supply it. + // Wait for host adoption before attaching an optimistic page the registry has not seen. if ( !viewport || pageHostGeneration === null || @@ -200,6 +200,24 @@ export function ClientHostedBrowserPagePane({ return } const webview = attachment.webview + // Guest loss uses the existing recovery notice and clears pending loading state. + let releaseGuest = (): void => attachment.detach() + const guestLoss = watchBrowserClientPageGuestLoss({ + webview, + webviewRef, + browserPageId: browserTab.id, + pageHostGeneration, + onLost: () => { + releaseGuest() + retryGuestRecoveryRef.current() + } + }) + // Main can destroy the guest while its tag still holds the stale id. + const attachedMetadata = readBrowserClientPageGuestMetadataIfLive(webview) + if (!attachedMetadata) { + guestLoss.lose('unreadable') + return guestLoss.dispose() + } const publisher = startBrowserClientPageMetadataPublisher({ browserPageId: browserTab.id, environmentId: runtimeEnvironmentId, @@ -213,23 +231,21 @@ export function ClientHostedBrowserPagePane({ }) webviewRef.current = webview setAttachmentError(null) - // Why: the failure carried in from the store is hearsay — this pane may be remounting over a - // guest that navigated on while nothing was listening — so it is checked once against where - // the guest actually is. Failures this session observes are trusted as they arrive, because a - // navigation that fails outright often never commits and leaves the guest on the old URL. + // Reconcile restored failures once; failed navigations this session may never commit a URL. activeLoadFailureRef.current = resolveActiveBrowserLoadFailure( activeLoadFailureRef.current, - readBrowserClientPageGuestMetadata(webview).url + attachedMetadata.url ) const syncNavigation = (event?: Event): void => { const eventUrl = (event as (Event & { url?: string }) | undefined)?.url - const metadata = readBrowserClientPageGuestMetadata(webview, eventUrl) - // Why: did-stop-loading fires after did-fail-load, so an unconditional null here would - // wipe the failure the overlay is about to show. + const metadata = readBrowserClientPageGuestMetadataIfLive(webview, eventUrl) + if (!metadata) { + guestLoss.lose('unreadable') + return + } + // did-stop-loading must preserve the preceding did-fail-load overlay. const activeLoadFailure = activeLoadFailureRef.current - // Why: a URL write drops the page's certificate challenge by design (challenges are - // transient across navigation), so a standing failure must not run through one — the - // local pane returns before its own setUrl for the same reason. + // URL writes clear certificate challenges, so preserve them while a failure stands. if (!activeLoadFailure) { setUrlFromGuest(browserTab.id, metadata.url, { preserveLoadError: true @@ -243,26 +259,41 @@ export function ClientHostedBrowserPagePane({ loadError: activeLoadFailure }) publisher.publish(metadata) - // Why: the address bar's suggestions read the client's shared URL history, so a page - // hosted here has to file its navigations there like a local guest does. - recordHistoryFromGuest(metadata.url, getBrowserDisplayTitle(webview.getTitle(), metadata.url)) + // Address-bar suggestions use the client's URL history, including client-hosted pages. + recordHistoryFromGuest(metadata.url, getBrowserDisplayTitle(metadata.title, metadata.url)) setAddressBarValueFromPage(toDisplayUrl(metadata.url)) } const onStart = (): void => { activeLoadFailureRef.current = null updatePageStateFromGuest(browserTab.id, { loading: true, loadError: null }) - publisher.publish(readBrowserClientPageGuestMetadata(webview, undefined, true)) - } - const onFailLoad = (event: Event): void => { - const loadError = resolveBrowserWebviewLoadFailure(event as BrowserPageFailLoadEvent, { - fallbackUrl: webview.getURL() - }) - if (!loadError) { + const startMetadata = readBrowserClientPageGuestMetadataIfLive(webview, undefined, true) + if (!startMetadata) { + guestLoss.lose('unreadable') return } - activeLoadFailureRef.current = loadError - updatePageStateFromGuest(browserTab.id, { loading: false, loadError }) + publisher.publish(startMetadata) } + const onFailLoad = createBrowserClientPageLoadFailureHandler( + webview, + () => guestLoss.lose('unreadable'), + (loadError) => { + activeLoadFailureRef.current = loadError + updatePageStateFromGuest(browserTab.id, { loading: false, loadError }) + } + ) + const cleanupGuest = (): void => { + webview.removeEventListener('did-start-loading', onStart) + webview.removeEventListener('did-stop-loading', syncNavigation) + webview.removeEventListener('did-navigate', syncNavigation) + webview.removeEventListener('did-navigate-in-page', syncNavigation) + webview.removeEventListener('page-title-updated', syncNavigation) + webview.removeEventListener('did-fail-load', onFailLoad) + guestLoss.dispose() + publisher.dispose() + forgetBrowserClientPageMetadataReports(browserTab.id) + attachment.detach() + } + releaseGuest = cleanupGuest webview.addEventListener('did-start-loading', onStart) webview.addEventListener('did-stop-loading', syncNavigation) webview.addEventListener('did-navigate', syncNavigation) @@ -270,26 +301,12 @@ export function ClientHostedBrowserPagePane({ webview.addEventListener('page-title-updated', syncNavigation) webview.addEventListener('did-fail-load', onFailLoad) syncNavigation() - // Why: the user pressed Enter while this page was still an optimistic stage, so the navigation - // was parked rather than sent to a host page that did not exist yet. The guest exists now. + // Resume navigation submitted before host adoption. const deferredUrl = consumeBrowserPageDeferredNavigation(browserTab.id) if (deferredUrl) { runDeferredNavigation(deferredUrl) } - return () => { - webview.removeEventListener('did-start-loading', onStart) - webview.removeEventListener('did-stop-loading', syncNavigation) - webview.removeEventListener('did-navigate', syncNavigation) - webview.removeEventListener('did-navigate-in-page', syncNavigation) - webview.removeEventListener('page-title-updated', syncNavigation) - webview.removeEventListener('did-fail-load', onFailLoad) - if (webviewRef.current === webview) { - webviewRef.current = null - } - publisher.dispose() - forgetBrowserClientPageMetadataReports(browserTab.id) - attachment.detach() - } + return cleanupGuest }, [ browserTab.id, browserHostClientId, @@ -299,7 +316,7 @@ export function ClientHostedBrowserPagePane({ setAddressBarValueFromPage ]) - useClientHostedGuestActivationFocus({ isActive, webviewRef, keepAddressBarFocusRef }) + useClientHostedGuestActivationFocus({ isActive, guestFocus, keepAddressBarFocusRef }) const showFailureOverlay = !attachmentError && Boolean(browserTab.loadError) // Why: the failure is about the URL that failed, not whatever page is still loaded — feeding diff --git a/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts b/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts index 5e4446f978a..93e62de4dd7 100644 --- a/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts +++ b/src/renderer/src/components/browser-pane/browser-client-page-guest-metadata.ts @@ -1,24 +1,67 @@ +import type { BrowserLoadError } from '../../../../shared/browser-workspace-types' +import type { BrowserPageFailLoadEvent } from './describe-page/browser-page-types' +import { resolveBrowserWebviewLoadFailure } from './navigate/browser-webview-load-failure' +import { recordRendererCrashBreadcrumb } from '@/lib/crash-breadcrumb-recorder' import { redactKagiSessionToken } from '../../../../shared/browser-url' import type { BrowserClientPageMetadataSnapshot } from './browser-client-page-metadata-publisher' /** - * What a client-hosted guest currently is, read straight off the webview. + * What a client-hosted guest currently is, read straight off the webview, or null once the tag + * can no longer reach its guest. * * `eventUrl` wins when a navigation event carries one: the tag's own getURL() can still report the * previous page while the event is being delivered. `loading` is forced for did-start-loading, * which fires before isLoading() flips. + * + * Why total rather than throwing: a guest destroyed in main leaves the tag holding its id, so + * every method on it throws `Invalid guestInstanceId` from then on — and every caller reads from + * a React effect, where that unwinds the whole workbench error boundary. */ -export function readBrowserClientPageGuestMetadata( +export function readBrowserClientPageGuestMetadataIfLive( webview: Electron.WebviewTag, eventUrl?: string, loading?: boolean -): BrowserClientPageMetadataSnapshot { - const url = redactKagiSessionToken(eventUrl || webview.getURL() || 'about:blank') - return { - url, - title: webview.getTitle() || url || 'Browser', - loading: loading ?? webview.isLoading(), - canGoBack: webview.canGoBack(), - canGoForward: webview.canGoForward() +): BrowserClientPageMetadataSnapshot | null { + try { + const url = redactKagiSessionToken(eventUrl || webview.getURL() || 'about:blank') + return { + url, + title: webview.getTitle() || url || 'Browser', + loading: loading ?? webview.isLoading(), + canGoBack: webview.canGoBack(), + canGoForward: webview.canGoForward() + } + } catch (error) { + // Why recorded: the catch is total, so a read failure that is NOT guest death would otherwise + // be indistinguishable from one — the breadcrumb carries the error text the console cannot. + console.warn('[browser-client-page] guest read failed, treating the page as gone:', error) + recordRendererCrashBreadcrumb('browser_client_page_guest_read_failed', { + errorName: error instanceof Error ? error.name : typeof error, + errorMessage: error instanceof Error ? error.message : String(error) + }) + return null + } +} + +export function createBrowserClientPageLoadFailureHandler( + webview: Electron.WebviewTag, + onUnavailable: () => void, + onFailure: (error: BrowserLoadError) => void +): (event: Event) => void { + return (event) => { + let guestUnavailable = false + const loadError = resolveBrowserWebviewLoadFailure(event as BrowserPageFailLoadEvent, { + // Discarded ERR_ABORTED/subframe events must not read the guest. + fallbackUrl: () => { + const metadata = readBrowserClientPageGuestMetadataIfLive(webview) + guestUnavailable = metadata === null + return metadata?.url ?? null + } + }) + if (guestUnavailable) { + onUnavailable() + } else if (loadError) { + onFailure(loadError) + } } } diff --git a/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts b/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts new file mode 100644 index 00000000000..b9e2f6a21f6 --- /dev/null +++ b/src/renderer/src/components/browser-pane/host-guest/browser-client-page-guest-loss.ts @@ -0,0 +1,54 @@ +import type { MutableRefObject } from 'react' +import { recordRendererCrashBreadcrumb } from '@/lib/crash-breadcrumb-recorder' + +export type BrowserClientPageGuestLossReason = 'unreadable' | 'destroyed' | 'render-process-gone' + +/** + * Tells a client-hosted pane, once, that its guest is gone. The retained registry fences the tag on + * `destroyed` / `render-process-gone` without telling the pane, which would otherwise sit mute or + * spinning over a tag whose every method throws; a failed guest read is the same verdict. + */ +export function watchBrowserClientPageGuestLoss(options: { + webview: Electron.WebviewTag + /** Released on loss and dispose: every chrome action null-checks it, so a dead tag is never driven. */ + webviewRef: MutableRefObject + browserPageId: string + pageHostGeneration: number + onLost: () => void +}): { lose(reason: BrowserClientPageGuestLossReason): void; dispose(): void } { + const { webview } = options + const releaseWebviewRef = (): void => { + if (options.webviewRef.current === webview) { + options.webviewRef.current = null + } + } + let lost = false + const lose = (reason: BrowserClientPageGuestLossReason): void => { + if (lost) { + return + } + lost = true + // Why the breadcrumb: the crash report this replaces was the only field signal for guest death. + recordRendererCrashBreadcrumb('browser_client_page_guest_unavailable', { + browserPageId: options.browserPageId, + pageHostGeneration: options.pageHostGeneration, + reason, + tagConnected: webview.isConnected + }) + releaseWebviewRef() + options.onLost() + } + const onDestroyed = (): void => lose('destroyed') + const onRendererGone = (): void => lose('render-process-gone') + webview.addEventListener('destroyed', onDestroyed) + webview.addEventListener('render-process-gone', onRendererGone) + return { + lose, + dispose: () => { + lost = true + releaseWebviewRef() + webview.removeEventListener('destroyed', onDestroyed) + webview.removeEventListener('render-process-gone', onRendererGone) + } + } +} diff --git a/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts b/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts index 95494fb043e..ac9fcf87d8b 100644 --- a/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts +++ b/src/renderer/src/components/browser-pane/host-guest/use-client-hosted-guest-activation-focus.ts @@ -1,4 +1,5 @@ import { useEffect, useRef, type RefObject } from 'react' +import type { BrowserPageGuestFocus } from '../assemble-chrome/browser-page-guest-focus' import { useWebviewDragPassthroughActive } from './use-webview-drag-passthrough-active' /** @@ -11,11 +12,12 @@ import { useWebviewDragPassthroughActive } from './use-webview-drag-passthrough- */ export function useClientHostedGuestActivationFocus({ isActive, - webviewRef, + guestFocus, keepAddressBarFocusRef }: { isActive: boolean - webviewRef: RefObject + /** Not the raw tag: a retired page's is out of the DOM, where focus() throws (STA-3448). */ + guestFocus: BrowserPageGuestFocus keepAddressBarFocusRef: RefObject }): void { const dragPassthroughActive = useWebviewDragPassthroughActive() @@ -41,6 +43,6 @@ export function useClientHostedGuestActivationFocus({ if (keepAddressBarFocusRef.current) { return } - webviewRef.current?.focus() - }, [dragPassthroughActive, isActive, keepAddressBarFocusRef, webviewRef]) + guestFocus.focus() + }, [dragPassthroughActive, guestFocus, isActive, keepAddressBarFocusRef]) } diff --git a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts index 44e59fdbedd..1595b69f6aa 100644 --- a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts +++ b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { resolveBrowserWebviewLoadFailure } from './browser-webview-load-failure' describe('resolveBrowserWebviewLoadFailure', () => { @@ -49,6 +49,18 @@ describe('resolveBrowserWebviewLoadFailure', () => { ).toMatchObject({ validatedUrl: 'https://example.com/current' }) }) + it('never reads a lazy fallback URL for an event it discards', () => { + const fallbackUrl = vi.fn(() => 'https://example.com/current') + expect(resolveBrowserWebviewLoadFailure({ errorCode: -3 }, { fallbackUrl })).toBeNull() + expect(fallbackUrl).not.toHaveBeenCalled() + expect( + resolveBrowserWebviewLoadFailure( + { errorCode: -105, errorDescription: 'ERR_NAME_NOT_RESOLVED', validatedURL: '' }, + { fallbackUrl } + ) + ).toMatchObject({ validatedUrl: 'https://example.com/current' }) + }) + it('keeps a usable description when Chromium reports an empty one', () => { expect( resolveBrowserWebviewLoadFailure({ errorCode: -105, errorDescription: '' }) diff --git a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts index c39c5afe53b..03228c3e1d6 100644 --- a/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts +++ b/src/renderer/src/components/browser-pane/navigate/browser-webview-load-failure.ts @@ -8,20 +8,23 @@ import type { BrowserPageFailLoadEvent } from '../describe-page/browser-page-typ * cannot forget the ignore rules or build a differently-shaped BrowserLoadError. * * `fallbackUrl` covers failures that arrive without a validatedURL — pass the webview's - * current URL so the overlay names the page instead of about:blank. + * current URL so the overlay names the page instead of about:blank. Pass it as a function when + * reading it costs anything: discarded events never ask for it. */ export function resolveBrowserWebviewLoadFailure( event: BrowserPageFailLoadEvent, - options: { fallbackUrl?: string | null } = {} + options: { fallbackUrl?: string | null | (() => string | null) } = {} ): BrowserLoadError | null { // Why: Chromium reports redirect/cancel races as ERR_ABORTED (-3) even when the // replacement navigation succeeds; subframe failures never blank the page. if (event.isMainFrame === false || event.errorCode === -3) { return null } + const fallbackUrl = + typeof options.fallbackUrl === 'function' ? options.fallbackUrl() : options.fallbackUrl return { code: event.errorCode ?? -1, description: event.errorDescription || 'Unknown load failure', - validatedUrl: redactKagiSessionToken(event.validatedURL || options.fallbackUrl || 'about:blank') + validatedUrl: redactKagiSessionToken(event.validatedURL || fallbackUrl || 'about:blank') } } From a65332a8bdaec95a48cc0ade4a7c482586bb9370 Mon Sep 17 00:00:00 2001 From: Brennan Benson <79079362+brennanb2025@users.noreply.github.com> Date: Fri, 4 Sep 2026 15:55:20 -0700 Subject: [PATCH 306/398] feat(claude): move structured native chat onto the Claude Agent SDK and enable it on macOS and Linux (#18560) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Join structured attach teardown through journal bind * fix: restore structured chat parity * feat: add Claude structured session adapter * fix: harden Claude structured adapter * fix: close Claude adapter edge cases * fix: start Claude init deadline after launch * feat: wire Claude structured sessions * fix: harden Claude structured runtime * fix: fence Claude structured compatibility * fix: preserve Claude free-text prompt answers * fix: decode addressed Claude prompt text * feat: enable Claude structured chat on mobile * fix(mobile): keep structured chat provider-aware * fix(mobile): negotiate Claude structured tabs * fix: keep scoped RPC tests native-free * fix: secure mobile structured image delivery * fix: close structured session data-loss gaps * fix: prove real Claude structured startup * fix: consume pre-spawn proof before retry * feat(native-chat): add desktop structured sessions * fix(native-chat): satisfy structured session cleanup gates * fix(native-chat): keep structured renders pure * fix(native-chat): open composer pickers upward * fix(native-chat): use existing view for structured sessions * fix: harden structured desktop status projection * fix: close structured desktop lifecycle gaps * fix: fence structured AI Vault resumes * fix: fence structured AI Vault resumes * fix: preserve structured tabs during activation * feat: toggle structured sessions between chat and TUI * fix: harden structured session handoffs * fix: bind structured TUI before rollout proof * fix: complete structured chat round trips * fix: align structured TUI return readiness * fix(native-chat): make reverse handoff transactional * Add Claude structured TUI handoff seams * fix(native-chat): clear sticky handoff recovery * fix(native-chat): complete mobile reverse after TUI exit * fix(native-chat): keep TUI transcripts readable * fix(native-chat): recover TUI transcript gaps * fix(native-chat): recover claimed TUI owners * fix(native-chat): retain cold TUI proof authority * fix(native-chat): preserve Claude handoff authority * fix(native-chat): recover TUI transcripts read-only * fix(native-chat): harden Claude handoff recovery * fix(native-chat): serialize structured handoff recovery * fix(native-chat): close handoff admission races * fix(native-chat): validate pinned launch environment * fix(native-chat): revalidate restored and retried owners * fix(native-chat): gate restart recovery publications * fix(i18n): catalog Claude session controls * fix(native-chat): wait for structured TUI process proof * fix(native-chat): queue stale idle TUI handoffs * fix(native-chat): route structured Codex options directly * fix(native-chat): persist structured session options * fix(native-chat): hydrate resumed structured options * fix(native-chat): preserve options across structured handoffs * fix(native-chat): replay pending option mutations * fix(native-chat): rotate settled handoff operations * fix(native-chat): rotate refused send operations * test(native-chat): derive refusal retry state from host * test(native-chat): give the host-oracle matrix test an explicit timeout * fix(native-chat): keep Claude option controls idle * fix mobile structured first-send hydration race * fix(native-chat): preserve handoff launch authority * fix(native-chat): harden shared handoff recovery * fix(native-chat): serialize structured handoff recovery * fix(native-chat): close handoff admission races * fix(native-chat): validate pinned launch environment * fix(native-chat): revalidate restored and retried owners * fix(native-chat): gate restart recovery publications * fix(i18n): catalog structured session recovery control * fix(native-chat): wait for structured TUI process proof * fix(native-chat): queue stale idle TUI handoffs * fix(native-chat): keep structured recovery provider-neutral * fix(native-chat): drop local terminal topology from structured sync * fix structured outbox and tab restore races * fix(native-chat): preserve Claude question groups * fix structured provider visibility and request handling * fix structured session TUI handoff recovery * fix reverse structured session handoff * fix(native-chat): recover Claude outbox and resume state * chore(mobile): preserve the working-tree lockfile state before the main merge Carries the pre-existing uncommitted mobile/pnpm-lock.yaml modification into history so the main merge cannot overwrite it. Verified benign pnpm drift (babel 7.29.7->7.29.8 transitives plus deprecation metadata); drops no patchedDependencies (the mobile lockfile declares none). * test(native-chat): drop orphaned Claude handoff-auth test left by the main merge 'pins Claude handoff auth through the terminal provider boundary' is absent from main and its production counterpart preserveClaudeAuthEnv no longer exists outside this test - orphaned residue of the terminal/native handoff work this PR excludes by scope. Removed rather than repaired: the failure was a renamed field (providerHome -> providerRoot), and renaming it would have carried out-of-scope handoff code into the merge. Body preserved as evidence and logged in CLAUDE-STRUCTURED-DISPOSITION-TABLE.md. * Fix mobile structured turn state * fix Claude structured session blockers * fix claude structured lane blockers * fix Claude acquisition exit proof * fix(claude): route stream-json launch through process wrapper * fix(claude): gate structured chat support * Fix Claude structured launch gating * fix(claude): split session acquisition and prune mobile scope * test(claude): align structured session fixtures * fix(agent-session): preserve handoff launch arguments * fix(claude): open journals through the factory after origin/main split The journal opener moved to journal-store-factory on main; retarget the Claude structured tests that still imported the old path. * fix(claude): resolve Claude structured launch args, auth, and win32 proof The origin/main merge re-expressed the lane's Claude wiring onto main's split orca-runtime facade and dropped three wires past green typecheck and lint. - resolveLaunchArgs discarded its provider parameter, so structured Claude sessions were launched with Codex app-server flags; Claude exits on --dangerously-bypass-approvals-and-sandbox, and a Codex arg-parse throw could block Claude session creation outright. - resolveClaudeLaunchEnv was no longer supplied, so the launch resolver fell back to the whole process env as configuredEnv and buildClaudeChildProcessEnv re-applied every auth var it had just stripped. The resolver now merges the Claude overlay onto a strip-applied copy of the inherited env, which also keeps PATH intact for withCliRuntimeOnPath. - The windowsProcessStartTimeAvailable producer was gone while the contract field and both consumers survived, so the renderer gate fail-closed and structured native chat was unreachable on every win32 host. Separately, structured Claude pinned CLAUDE_CONFIG_DIR unconditionally. An explicit pin makes the CLI abandon the macOS Keychain even when it names the CLI's own default, so a default claude.ai account could not authenticate where the legacy Claude terminal could. Pin only a home the CLI would not resolve on its own, matching ClaudeRuntimePathResolver, and compare against the env the child would otherwise inherit so a diverging overlay cannot outrank the record's account home. Also await the now-async revealNativeSession in its regression test, and set the native status before revealing so a rejecting reveal cannot leave a session released but never marked native. Claude-Session: https://claude.ai/code/session_013UqKCRB6k5e8UaYhXUHeWY * fix(claude): scrub case-insensitive Windows auth env * fix(native-chat): settle handoff outcome-write failures instead of leaking them A store write failure while recording a handoff outcome escaped the flow runner's catch handler, so the client never received the failure and the flow surfaced as an unhandled rejection (seen as an intermittent agent_session_store_corrupt error in the proven-dead-retry suite, whose teardown raced the flow's trailing outcome write). Record the failed outcome best-effort, and drain the coordinator before that test's teardown removes the store root. Claude-Session: https://claude.ai/code/session_011aXkcHyeiRJuezupQdjZaM * fix(native-chat): make the structured close-failure toast provider-neutral The structuredSessionCloseFailed toast fires for any structured session, but its copy said 'Codex chat', so a Claude structured session that fails to close showed the wrong provider name. The launch-failure toast is only reachable behind the agent === 'codex' gate, so its copy stays as is. Claude-Session: https://claude.ai/code/session_013ugSpCx4AWkySaJb69BQax * fix(native-chat): wire structured handoff proof recovery * fix(native-chat): wire structured handoff proof recovery * fix(native-chat): correct the structured chat opt-in copy The one `experimentalStructuredNativeChat` toggle gates both providers — `useStructuredAgentSessionCreate` runs `canUseStructuredNativeChat` for `'claude'` as well as `'codex'` — but its description named only Codex. Its scope line also said Windows keeps using terminal chat, while the gate refuses win32 only until the host proves it can read a process start time. `structured-native-chat-availability.test.ts` already pins that Windows is allowed once the proof is cached, so the two contradicted each other. Claude-Session: https://claude.ai/code/session_01RJFsidQWmKYFmeoUuVu4Tp * test(claude): pin @anthropic-ai/claude-agent-sdk 0.3.251 contracts against a scripted CLI PR 1 of the SDK migration: dependency + test-only harness, no product wiring. - Pin @anthropic-ai/claude-agent-sdk to exactly 0.3.251 — not the newest release — because 0.3.251 (published 2026-08-28) clears the repo's 3-day minimumReleaseAge supply-chain gate with no exclusion, while the newest release was minutes old and would have required excluding a brand-new publish from the exact control built to catch brand-new malicious publishes. Every contract this design depends on was verified identical on 0.3.251: the full option surface, no pid on SpawnedProcess (custom spawner stays mandatory), env defaulting to process.env when omitted, and --replay-user-messages appearing only via extraArgs. - Exclude all eight bundled CLI platform binaries via ignoredOptionalDependencies. The setting lives in pnpm-workspace.yaml because pnpm 12 no longer reads the package.json "pnpm" field (it warns and ignores it; verified by install ablation). Excluding the binaries is what makes Orca's pathToClaudeCodeExecutable override mandatory rather than merely preferred. Note: pnpm 12.0.0 honors the ignore list when reconciling an existing lockfile but not on fresh resolution of a new dependency, so the lockfile's SDK entry was pinned surgically; both 'pnpm install' and 'pnpm install --frozen-lockfile' verify clean and stable against the committed lockfile. - Contract-pin suite drives the real SDK against a scripted fake CLI and pins: unknown type/field/content-block pass-through (and keep_alive interception), spawner env fidelity plus the omitted-env process.env inheritance sharp edge, extraArgs producing --replay-user-messages, argument parity for every CLAUDE_STRUCTURED_BASE_ARGS entry plus --session-id/--resume/ --resume-session-at, canUseTool wire request_id stability and abort on control_cancel_request, one spawn per query, pathToClaudeCodeExecutable honored by the default spawner, the exact SDK version, and the eight platform binaries staying uninstalled. Claude-Session: https://claude.ai/code/session_01FGCRfYUnb4hbvfTAHGtJKQ * feat(claude): drive the structured transport through the agent SDK Replaces the hand-rolled `claude -p --input-format stream-json` transport with @anthropic-ai/claude-agent-sdk 0.3.251, keeping the existing connection interface for this commit so the acquisition path changes minimally. The control-plane rewrite is a separate change. Orca still supplies the process. `spawnClaudeCodeProcess` routes through `spawnProcess`, retains the child and its pid — the triple the durable lease adjudicates on — drains stderr so exit errors keep their tail, and hands `.cmd` shims to Orca's Windows argument encoder rather than the SDK's plain spawn. `close()` keeps Orca's own bounded tree-kill and exit deadline, so it still resolves true only after an observed exit. Launch resolution emits an SDK options object instead of argv; durable `launchArgs` translate to a typed option where one exists and to `extraArgs` otherwise, refusing a token neither can carry rather than dropping it. The child env is always passed explicitly — omitting it would let the SDK inherit `process.env` and reintroduce the ambient `ANTHROPIC_*` leak. The stdout line parser is deleted; the SDK owns framing, and unknown frames still reach the translator verbatim. Claude-Session: https://claude.ai/code/session_01JMhFjh9HEnkcJ5YTfCdgD3 * fix(claude): settle the frame the SDK pulled but never wrote The SDK's input pump is `for await (frame of prompt) { await transport.write(frame) }`. When that write rejects — the child dies between Orca's liveness guard and the write — the for-await ends abruptly and calls the generator's `return()`, so the code after `yield` never runs. The frame was already shift()ed out of `queued`, so the later `fail()` from the exit path could not reach it and `send()` never settled: `dispatchClaudeTurn` awaits that send before it can return `unknown`, wedging the caller and the durable outbox. The pre-SDK transport rejected on the stdin write callback instead. Retain the in-flight entry and settle it from the generator's cleanup, and let fail() reach it too for the pump that never resumes at all. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * fix(claude): keep the agent SDK behind the structured-Claude boundary The ordinary OrcaRuntimeService graph statically reaches the Claude adapter and so the transport module, whose first line imported @anthropic-ai/claude-agent-sdk. The SDK is evaluated whenever the regular runtime loads, before any structured Claude session is chosen: it sets process.env.NoDefaultCurrentDirectoryInExePath, changing Windows executable resolution for later subprocesses, and a missing or incompatible install would break normal runtime startup — for a user who never leaves the terminal/TUI path. Defer the SDK to the connection, memoized so it loads once per process, and add the import-graph ratchet: a walk from the Electron main entry that fails on any static import of the package, plus a clean-fork check that loading the runtime leaves the Windows search variable untouched and a child-process pin that the side effect is still real. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * fix(claude): answer list_models so the picker stops serving the seed sendControlRequest had no list_models case, so every request hit the default reject; readClaudeStructuredSessionOptions swallows that with .catch(() => null) and falls back to the static catalog. Every structured session therefore served a hardcoded model list with no per-model effort levels, no resolvedModel and no default detection, and nothing surfaced the failure. The pre-SDK transport got the live catalog from the CLI. Route it through the SDK's supportedModels(), wrapped in the { models } envelope the existing parser reads. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * fix(claude): reap the child's descendants before killing it The forced step of the exit ladder went through the Codex helper, which spawns `pkill -KILL -P ` and SIGKILLs the parent in the same tick: the parent usually dies first, the descendants reparent to pid 1, and `-P` matches nothing. An MCP or launcher descendant of a stubborn Claude child was left running. The test named for that requirement declined to assert it and killed the survivor by hand instead, so it could not fail for the thing it was named after. Route the Claude reap through Orca's existing sweep, which snapshots descendants while their parent link still exists and signals them before the root goes, and on Windows uses the identity-gated `taskkill /T /F`. The test now asserts the descendant is dead; the manual kill stays only as a failure-safe. close() still returns true only on an observed exit. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * fix(native-chat): merge the duplicated handoff type import CI's static-analysis lint (`oxlint --config config/oxlint-code-quality-native-plugins.json src config tests mobile --deny-warnings`) exits 1 on the two separate `import type` statements from the same module. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * fix(claude): answer a permission callback whose signal already aborted settleFrom registered the abort listener and then delivered the request. A callback that arrives already aborted never fires that event, so the promise stayed pending behind a durable prompt with no cancel path. Check the signal first, emit the cancel, and resolve the SDK's null sentinel without registering. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * test(claude): wait for the child to record the frame, not just for its report The scripted CLI writes its report at startup, so `until(readReport)` returned a report with no user messages whenever the child had not yet read the line. The assertion then failed under parallel load. Poll for the frame instead of for the file. Claude-Session: https://claude.ai/code/session_01AobxxokqQ3qcxS7sy7ckum * fix(claude): coalesce partial deltas onto one assistant item and stop painting result frames Under --include-partial-messages every stream_event frame carries its own uuid, and the final assistant frame for a block carries yet another; only message.id ties them. The translator keyed each delta by its frame uuid, so a reply painted as one bubble per delta chunk followed by a complete duplicate under the final frame's uuid. The block's first stream frame now mints the claude:(sessionId, uuid) identity, deltas coalesce onto it through the shared 60ms seam, and the final frame reconciles onto that same item. Known SDK bookkeeping no longer reaches the provider-fallback row: result subtypes are catalogued and settled by the turn lifecycle, an empty thinking block (redacted thinking) is a modeled kind, a string-content user replay is a text block, and an empty user frame paints nothing. An unmodeled result subtype or content kind still lands on the bounded fallback row. Claude-Session: https://claude.ai/code/session_01GaP5HpYQbvy2hYehVhwfEW * fix(claude): prove descendant exit at the close boundary instead of on an unref'd timer close() reported proven=true as soon as the direct child exited while the descendant sweep's SIGKILL sat on an unref'd 2 s timer, so a SIGTERM-resistant MCP server outlived the lease release. The reaper now composes the same shared primitives the Codex structured provider uses: snapshot, verified bounded descendant termination on POSIX, taskkill /T /F on Windows. The proof is false whenever descendants outlive the deadline, a retried close re-verifies the retained snapshot rather than trusting the dead root, and the raw pipe child no longer goes through the PTY job sweep it never owned a job for. Measured on macOS: a killed child of a SIGSTOPped parent stays a matching zombie row in ps, so the root is killed while verification runs rather than stopped first as the Codex non-group path does. Claude-Session: https://claude.ai/code/session_0161QFm3KVRNJKfdzWVGVNWk * feat(claude): replace the hand-rolled control plane with the SDK's native surface PR 3 of the Claude structured SDK migration removes the wire-frame scaffolding PR 2 kept, so Orca drives the SDK's typed control surface directly. Inbound permissions move from a rebuilt control_request dispatch to the SDK's canUseTool / onUserDialog callbacks. The prompt registry now carries the callback's own resolver: a decodable can_use_tool becomes a durable prompt whose answer settles the callback; a malformed one is denied without registering; the SDK's abort signal (fired on control_cancel_request, which the SDK matches and dedups itself) forgets the prompt and settles it null, and a late answer after abort finds no prompt and is refused. Closing settles every in-flight callback so no promise dangles. The claude-agent-sdk-control-bridge that rebuilt the wire frame is deleted. Outbound control maps to Query methods: interrupt() for cancel, setModel / setPermissionMode / applyFlagSettings for options, supportedModels for the model list, initializationResult() for init proof, each under Orca's own request deadline and error classification. Cancel is interrupt-receipt aware: a CLI advertising interrupt_cancel_queued_v1 gets cancel_queued in one round trip, otherwise the receipt's still_queued uuids are swept with cancel_async_message so a cancelled turn cannot spawn a later unexpected turn; older CLIs resolve no receipt. Init keeps the 10s deadline and the unauthenticated-startup guidance. Every behavior is failing-first and ablation-proven; the toggle-off import boundary and the accepted loss of unknown-control visibility rows are unchanged. Claude-Session: https://claude.ai/code/session_01Pqjduxt5G4rr9aYvtp7rNm * fix(claude): arm the descendant snapshot before stdin closes and make the tree verdict unproven by default A healthy Claude root leaves within the graceful window, and the close ladder only snapshotted descendants when the root was still alive after that window. So the common close never looked at the tree: `treeExited` stayed null, `!== false` passed it, and close() reported a proven exit with an MCP child still running. A root that died before the walk made the snapshot vacuous too. The proof is now unproven by default. The reaper holds one verdict in Orca's vocabulary (exited / live / unverifiable), assigned in exactly one place from the bounded verification, and close() returns true only on `exited`. The snapshot is armed before stdin closes, while the root can still be walked, and is verified after the root exits; a root that left before any snapshot could be armed stays unverifiable rather than vouching for descendants it never showed us. The shared verifier gains the three-way verdict behind its boolean face, and the connection reports the root and tree verdicts separately along with the child's exit status. One verification per close attempt: the retried close re-verifies, so the intra-attempt re-reap is gone from the teardown budget. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * fix(claude): verify the Windows tree after taskkill instead of trusting that it ran `terminateWindowsProcessTree` resolves from taskkill's callback whatever the error says, so a timeout, an access denial, a recycled root and a surviving descendant all looked identical to the reaper — which then returned a proven exit unconditionally. close() reported true and the lease was released with an MCP descendant potentially still live. The Windows branch now snapshots the root's descendants while it is alive and, after taskkill, polls a fresh process table to a bounded deadline: a row still matching by pid AND creation time is `live`, an unreadable table is `unverifiable`, and only a table with no match is `exited`. Creation time is the PID-reuse guard the POSIX path gets from ps lstart, so a descendant that denied a creation-time query is omitted rather than signalled on a bare pid. A root already observed exited is never taskkilled: `/T /F` on a recycled pid would take an unrelated tree down with it. The captured tree is tagged by platform so neither verifier can be handed the other's rows. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * fix(claude): release a reservation on a first-hand root exit instead of latching it into manual recovery Making close() strict about the descendant tree exposed a second defect at the same boundary. A create-time acquisition has no ownerProcess until publication, so an unproven cleanup mapped to handoffStage `manual-recovery`, and adjudication then refuses every later attach with agent_session_ownership_unknown. A user who was merely signed out, or whose --resume the CLI rejected, wedged the session id permanently. Each question now answers from its own evidence. close() is unchanged and stays strict about the tree. Separately, the lease is keyed on the root's pid and start time, so when Orca's own child handle observed that root exit and no descendant snapshot was ever admissible, the reservation is released and the CLI's exit code and stderr reach the user. A descendant observed still alive, or a root Orca never saw leave, stays unproven and keeps the reservation. The settlement records only what was observed: the released lease says the provider process exited and its descendants were not verifiable, rather than reusing the wording that claims cleanup proved no child remains. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * fix(claude): surface an API error a result frame reports instead of settling the turn on it The SDK models an API failure as a SUCCESS-subtype result whose `result` string is the user-facing error text, with no assistant frame behind it. The translator suppressed every catalogued result subtype as turn bookkeeping, so that turn tombstoned its lifecycle and showed the user a completed, empty reply with no sign anything had failed. Suppression is now by meaning. A result reporting a failure routes to the bounded provider-error surface, leading with the provider's own sentence and keeping the raw frame behind the row's disclosure; ordinary successful results stay off the timeline as before. A turn the user aborted also stays suppressed: its interrupt frame already says so, and its execution diagnostic would only be noise on every stop. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * fix(claude): drop the stream state of turns that never received their final frame Every streamed delta recorded its block's identity, latest text and checkpoint length. Only the final assistant frame removed them, so an interrupted turn left its whole accumulated reply reachable until the session was disposed, and a long session with repeated interruptions grew those maps without bound. The partial text was already journaled by the flush that precedes settlement, so the live copy was pure retention. That state now lives in its own module, named for what it does — grow a streamed block's journal row between its deltas and its final frame — and turn settlement drops every block still awaiting a final. The translator reports how many remain, which is the invariant: a settled turn leaves none. Also makes a timed-out process-table read retryable while the root is still alive. A loaded host can miss the table's one-second deadline, and latching that as "no descendants" both lost the descendant sweep and, on a busy machine, made the close ladder report unproven for a tree it never actually looked at. Only the root's death still makes a missing snapshot final. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * perf(claude): capture the Windows descendant tree from one process-table read The capture walked the descendant tree and then read the table again for the creation times the walk's projection drops. Each read is bounded in seconds and both run inside the close ladder's budget, so the second one cost the worst-case teardown three seconds for data the first read already held. The walk is now exported from the module that owns it and runs over rows the caller has already read, which is also what lets the snapshot keep the PID-reuse guard the projection cannot carry. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * fix(pty): spend the descendant verification window instead of surrendering on one slow table read The verification abandoned the whole check the first time a process-table read missed its own one-second deadline, with seconds of its window still unspent. On a loaded host that reported a tree unverifiable without ever having looked at it, which the Claude close ladder then turned into an unproven close and a retried teardown. It also made the descendant-exit tests flake under a parallel suite run, for the same reason and with the same honest-but-premature verdict. A read that missed its deadline is now simply not an answer: the loop waits and reads again until its own deadline, and only a window that ends without a readable table reports unverifiable. This can only turn a premature verdict into one backed by evidence; it never manufactures a proof. Claude-Session: https://claude.ai/code/session_01BSmXgkWSsNHft8jFkdBFG9 * fix(claude): never let a later failed look collapse an observed live descendant into unverifiable The reaper's single assignment site latched only 'exited', so a second reap whose table reads all missed their deadline overwrote an earlier completed verification's 'live' with 'unverifiable'. The acquisition release gate discriminates on exactly that pair, so a root exit after such a decay released the lease over a descendant that had been observed alive. The latch is now monotone in trust order: exited is final, and live is only ever raised to exited. Claude-Session: https://claude.ai/code/session_01HfdhsvSJucLw4cTZxzg2CP * fix(claude): never prove a Windows tree gone while a descendant denied identification The Windows snapshot dropped rows that denied the creation-time query, and an emptied snapshot was judged exited without any table read: a descendant Orca was refused information about was treated as one that had left. The snapshot now counts the unidentified rows it saw, and verification caps its verdict at unverifiable while any exist. Nothing is ever signalled on a bare pid, as before. Claude-Session: https://claude.ai/code/session_01HfdhsvSJucLw4cTZxzg2CP * fix(claude): classify cleanup after a first-hand exit as a root exit instead of a proven tree When the CLI died between a successful acquire and the host's commit or proof of the lease, handleExit had already removed the session, so releaseAcquisition found nothing and reported true. The attach flow then settled exit-proven with deathEvidence claiming cleanup proved no provider child remains, though the tree was never verified. The adapter now keeps the exit that removed a published session until the session is acquired again; acquisition cleanup runs that connection's close ladder and classifies its verdict exactly as a start-time failure would be, so the record reads root-exit-observed. The wire helper keeps that typed classification and its provider diagnostic instead of wrapping it as unproven, and the router gives up its owner even when the release throws. Claude-Session: https://claude.ai/code/session_01HfdhsvSJucLw4cTZxzg2CP * fix(claude): integrate SDK teardown and picker lifecycle fixes * fix(claude): preserve resume leaf and settle processless spawns * fix(claude): reacquire from persisted resume leaf * fix(native-chat): restore Claude grouped question handling * fix(claude): persist only resumable transcript leaves * fix(claude): recover structured session exits safely * fix(claude): close remaining structured session P1s * fix(claude): harden transcript branch proof * Remove superseded root fix reports * fix(windows): restore indexed descendant row walk * fix(router): forward force-close lifecycle * fix(claude): fence stale turn cancellations * fix(claude): fence cancellation after unknown dispatch * fix(claude): fence replay and option recovery races * fix(claude): block replay fallback after waiter eviction * fix(claude): fence evicted slash results * fix(claude): fence ambiguous results and restore options safely * fix(claude): scrub SDK child env and localize pending launch * fix(claude): pin transcript roots and exit recovery proofs * fix(claude): retain unproven SDK exits * fix(claude): settle retained exit before reacquire * fix(claude): resume from settled retained cursor * chore: remove tracked review artifact * fix: harden Claude SDK transport session cleanup * fix: close Claude sessions safely * fix(claude): close races with fresh child snapshots * fix(claude): fail closed on recycled child identities * fix(claude): gate root cleanup on process identity * fix(claude): fence same-second root identity reuse * fix(claude): restore the root SIGKILL fallback the identity gate took away The direct root kill goes through the handle Node owns, not through a pid: libuv drops that handle in the same turn it reaps, so the signal either reaches the process Orca spawned or reaches nothing at all. Gating it on a process-table probe therefore bought no safety and cost the tree its only fallback whenever the probe declined -- a first capture landing in the fork's own second, a recycled descendant pid voiding the snapshot, or a process table that could not be read on either platform. Identity verification stays where a bare pid is genuinely addressed: Windows `taskkill /T /F`, and the descendant sweep's own revalidation before it signals. Also stops a declined root probe from collapsing an observed `live` or `exited` descendant verdict into `unverifiable`, and stops a successful taskkill from reporting `unverifiable` because a later probe found the root correctly dead. * docs(claude): rewrap the root-kill ordering comment * Match the Claude structured launch to the terminal path's managed-account auth rules The SDK path stripped ambient Anthropic auth unconditionally, let an explicit agentDefaultEnv override beat a pinned managed account, and had no account-switch guard. Reuse the terminal preflight's own predicate and messages so both transports strip, refuse, and report identically, and cover the CLI transcript location that mobile native chat depends on. * Reach the Claude structured chat lane from the desktop UI The main process has had a complete, correctly gated Claude Agent SDK lane for a while, but no renderer ever asked for it: the launch route accepted only `codex`, and the create path was typed `agent: 'codex'` end to end. Widen both to the structured provider union that already exists (`AgentSessionHandleProvider`), and generalize the codex-named create path instead of adding a Claude twin beside it. The pending-launch registry is now keyed by agent as well as workspace — a shared key handed a second caller the first agent's intent, so a Claude and a Codex launch in one worktree collided. Windows, per agent. Codex's client-side win32 refusal is deliberate and settled elsewhere, so it stays exactly as it was. Claude's answer is no longer guessed from the client's platform: a structured session fences its provider child on that child's process start time, and only the executing host knows whether it can read one. `agentSession.createSupport` already answers precisely that, per agent, and had no renderer caller — so the Claude create path asks it before creating and turns a "no", or a probe it cannot get answered, into the definitive refusal the launch fallback already handles. Fail closed either way. That refusal mapping also closes a real gap: the host reports an unsupported location by throwing `structured_agent_session_unsupported`, which reaches the client as a transport rejection rather than a refusal envelope, so `StructuredAgentSessionCreateRefusalError` never fired. The launch would retry the create, strand itself in `visibilityUnknown`, run no legacy fallback, and show an error toast. Close a fail-open hole while Claude and win32 become reachable: `create` with a client-supplied location, and `ensure`, both skip the worktree-resolving support check. They now ask the executing host the same question directly, so a host that cannot fence a provider child no longer creates one on a client's say-so. Also deletes `structured-agent-session-provider-routing.ts`, a duplicate of `structured-agent-session-provider-support.ts` with no importers. WSL, SSH and paired hosts, floating workspaces, draft prompt delivery, explicit TUI customization and initial session options all keep refusing; folder workspaces keep working. * P1-1: make the structured Claude auth policy required and testable The optional dep plus a {stripAuthEnv:false} fallback meant a dropped wiring under-stripped silently. Required at all three hops, asserted at install time for the @ts-nocheck caller, and the settings-to-policy mapping is now a named tested function. * P2-3: mobile's default Claude transcript root must follow CLAUDE_CONFIG_DIR session-file-resolver's default ignored the variable the pinned account home follows, so a CLAUDE_CONFIG_DIR launch wrote one tree and mobile read another. The Task-4 test now resolves with no root override (mobile's own call) and checks the answer against the root the CLI itself reports, instead of mirroring the code under test's own expression. * P2-1/P2-2/P3: close the teardown window, join the live-auth gate, align the refusal P2-1: a switch beginning inside the acquire teardown left a dead chat and no replacement. Past that point the launch waits the swap out and refuses only if it never settles; the entry guard still refuses outright, because nothing is torn down there yet. P2-2: structured children now hold the same OAuth-refresh gate a Claude PTY does, so a managed refresh cannot rotate the token out from under a live turn. P3: the refusal now matches the strip it guards (case-folded on win32, presence not truthiness), and the dead structured-to-TUI builder states its auth policy instead of silently signing a system-auth user out. * Make the live-auth gate tests independent of sibling connection teardown order * Do not offer structured Claude under a WSL-only managed account Structured Claude launches against the ambient Claude config, which the account service keeps in sync with the selected HOST account. A WSL-bound managed account lives inside the distro and is never synced there, so on Windows a structured session would authenticate as whatever the ambient identity happens to be while the UI names the WSL account — the user is told one identity and given another. That was unreachable only because nothing offered structured Claude on win32. Enabling it makes it reachable, so gate it here rather than patching the auth layer: refuse the structured path when the active managed Claude account is WSL-bound, and let the terminal-backed path — which resolves the account per runtime — handle that account shape. The answer rides the agentSession.createSupport seam the renderer already consumes, so no new capability and no renderer knowledge of account internals. A create the host declines becomes the definitive refusal the launch fallback already turns into a legacy native chat tab, with no error toast. Unknown answers refuse. An install with no managed accounts claims no identity and is fine, but an active selection that cannot be resolved — or account state that cannot be read at all — is not evidence that the ambient identity is right. Claude only. Codex resolves its account through a different path and its createSupport answer is untouched, as is every Codex routing decision. * Read the structured Claude account gate through the auth policy's accessor The gate resolved the active account from the account-service snapshot's runtime map; the auth policy resolves it with getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }). Those are two sources and two resolution rules, and they disagree on a legacy settings blob that carries the selection only in the flat activeClaudeManagedAccountId: the accessor falls through to it, a direct read of the runtime map does not. The gate would then refuse a launch the policy would have run under host-1 — and in the mirror case a session could be admitted under a policy computed from a different account than the gate approved. Read the same settings through the same accessor so agreement is structural rather than coincidental, and drop the controller accessor that existed only to reach the snapshot. No behaviour change for any state both already agreed on; Codex is untouched. * Round-3 review fixes: N-1 empty-value regression, N-2 gate leak window, N-4 lost history N-1: my presence-based conflict predicate refused a terminal launch that works today. 'ANTHROPIC_API_KEY=' is how a user blanks a variable and the settings pipeline preserves that empty value; an empty override cannot beat the pinned account and the strip removes the name anyway. Back to truthiness for the value, keeping the win32 case folding. N-2: enter the live-auth gate only after the exit/close handlers that release it, so no throw in between can leave an entry nothing reconciles. N-4: the Claude transcript resolver searches config-dir-then-default and de-dupes, matching the Codex sibling in the same file, so adopting CLAUDE_CONFIG_DIR no longer hides history written before it. * Run the managed-account gate on every Claude acquisition, not just create createSupport gates the create path, but a session's account state can change while it lives. A reacquire after an unexpected child exit re-resolves the launch and re-derives auth, with nothing re-checking the gate — so a session created while supported could come back up in the refused shape. With the strip predicate keyed on there being an active non-WSL account, the WSL-only user's normalized steady state (accounts exist, none active) does not strip, and that reacquire reaches the child with ambient auth while the UI names the account. Gate at resolveLaunch, the one choke point every acquisition passes through, refusing with the pre-spawn error the caller already handles. Same predicate as create-time, now sharing one settings reader so the two cannot drift. Claude only; Codex resolves its account on a different path and is untouched. The runtime class that wires this does not typecheck its own `this` calls — a missing hookup compiles clean — so the wiring is pinned behaviourally rather than trusted to the compiler. * Move the structured Claude gate out of the @ts-nocheck runtime files Both call sites of the managed-account gate sat in files whose first line is `// @ts-nocheck`, so neither was typechecked: three arguments to a one-argument function plus an undeclared identifier compiled clean. New auth-identity decision logic had no compiler behind it. Move the verdict into a checked module that takes the two facts the runtime owns — the adapter's answer and a settings getter — and decides. The runtime class now only forwards. Move the gate reader's construction into the checked installer too, so the nocheck file passes a plain settings closure and never names a gate symbol. Every reference to the gate predicate and its reader now lives in a checked file, so the ablation that used to pass silently is a compile error at both the create-support and reacquire sites. Removing the file-level @ts-nocheck is a separate, larger job and is not attempted here. * Derive the gate test's auth policy from the settings under test A hardcoded stripAuthEnv asserts a gate/policy pairing production cannot produce, and false additionally lets launch.env inherit the runner's real process.env. Derive via claudeStructuredAuthPolicyForSettings instead: the gate settings type is the same Pick the policy takes, and both resolve the account through getSelectedClaudeAccountIdForTarget. * Pin the absent-vs-empty distinction in the managed-account gate An empty claudeManagedAccounts array is a real answer: the user has no managed accounts, nothing claims an identity, and the ambient path is legitimate. A readable settings object with no such field is settings we failed to parse — the same unknown as unreadable — so it refuses. The two are one character apart in the code and the difference is invisible without the reasoning, so record it at the branch and pin both sides. The test fails under the obvious "consistency fix" of treating a missing field as empty. * fix(claude): keep command queue bookkeeping out of the transcript Claude Code 2.1.258 emits a `command_lifecycle` frame for every uuid-stamped command it starts, completes or cancels. The frame carries a command uuid and a state and no content, and the CLI keeps it out of its own transcript -- but it is absent from the SDK's SDKMessage union and so from Orca's frame catalogue, where an uncatalogued kind defaults to a substantive row. Every structured turn therefore painted raw JSON rows into the user-visible transcript. Catalogue it and disposition it as status chrome. The unknown-kind default stays `timeline-substantive`: a kind we have never seen is likelier to carry content than to be chrome, and a visible row we can catalogue later beats content we silently dropped. A lifecycle state that reads as a failure still surfaces, because the payload error check in `classifyProviderFrame` outranks the catalogue. * fix(claude): let a re-walked descendant become eligible for the forced sweep A descendant first observed by a capture inside its own birth second could never be SIGKILLed: `ps lstart` is second-resolution, so that capture cannot rule out a pid recycled later in the same second, and the merge pinned each retained row to the boundary of the walk that first saw it. SIGTERM-resistant children forked in that window were signalled and then never escalated -- they survived close, quit and restart, reparented to init, and had to be killed by hand. Advancing that boundary on any later capture would be unsound: a later capture matching pid, pgid and start-second is exactly what an impostor would also show. But a capture is not a match -- it is a fresh ppid walk from a root Node pins through its own handle, so a row it re-derives is proved ours at that instant without appealing to its start time. Chain the fence from there instead, and take that walk at the close boundary while the root certainly still lives: the root may leave inside the grace window, and the post-timeout refresh never runs. A row absent from the later walk still keeps its earlier boundary, and a row no walk has ever re-derived in a later second is still never escalated. * Treat an absent managed-account list as empty, not as unreadable An empty claudeManagedAccounts array and a missing one are the same answer: this user has no managed Claude accounts, so nothing claims an identity and ambient auth is the truth. Refusing on absence strands any profile that simply never wrote the key, and it disagrees with the auth policy, whose own predicate takes `(accounts ?? [])` for exactly this reason. Only settings that cannot be READ stay unknown, and those still refuse — as do a WSL-bound active account and a selection naming an account the list does not explain. The earlier reasoning treated a missing field as settings we failed to parse. That conflated "not present" with "not readable"; only the second is unknown. * Support structured Claude when accounts are registered but none is selected Registered-but-deselected Claude accounts were refused, which is behaviourally identical to having no accounts at all: the auth policy does not strip, ambient auth is the truth, and the UI names no host identity. A user who deselected their accounts silently got legacy chat with nothing explaining why. Nothing selected for the host runtime is two states the settings cannot tell apart after the fact, because pruneInvalidClaudeRuntimeSelection empties the host slot and persists null in the second one: honest deselection -> ambient auth, UI names nothing -> SUPPORTED the WSL-only steady state -> ambient auth, UI names the WSL account -> REFUSED The presence of any WSL-bound account in the list decides. Simplifying this to "none active -> supported" re-opens the auth-identity misrepresentation, so the tests fail loudly on exactly that: five of them, across the unit rule and the createSupport path. * Stop treating an unanswerable create-support probe as a refusal A worktree is not resolvable for a beat after createWorktree resolves, so a probe fired immediately after creation fails the RPC with selector_not_found instead of answering. The catch collapsed that into `supported = false`, so the composer refused and quietly built a terminal session — the gate never said no, it was never asked successfully. Elapsed time was the only input that decided whether a Claude launch went structured. "Could not answer" and "answered no" are different states and only the second is a verdict. Retry while the host cannot yet resolve the selector, with a bounded backoff that covers the measured window with margin, and keep refusing on the first ask for everything else. Fail-closed is unchanged: a probe that still cannot be answered when the budget is spent refuses. The retry is narrowed with the shared error-code matcher, which classifies a token that transports re-wrap into a longer message without matching prose that merely mentions it. Codex never probes, so this race has never been able to refuse a Codex launch — the race itself is identical for it. Recorded at the early return, because whoever gives Codex a probe inherits the bug. * fix(claude): fence the forced sweep on re-derivation, not on lstart's second A descendant forked in the same wall-clock second as every walk that sees it was signalled with SIGTERM and then never escalated, so a SIGTERM-resistant child survived tab close, app quit and a full relaunch. Two children of one parent 96ms apart across a second boundary took opposite paths. The leak predates this branch: it reproduces with the change reverted. `ps lstart` has one-second resolution, so a walk landing inside a row's birth second can never rule out a pid recycled later in that same second. But a walk is not a match: a ppid walk only reaches what the root actually parents, and the root is pinned by Node's own handle, so a row the walk re-derived is ours whatever second it was born in -- a stranger would have to have been forked into our tree, and then it is not a stranger. Fence the escalation on that. Rows a merge retained from an earlier walk are not re-derived and still answer to the start-time fence, which remains correct for them. Scoped to callers that revalidate identity before signalling, which is the Claude close path. Codex teardown reaches this same verifier and is unchanged; the argument holds there too, but widening it is its own deliberate change. Also reverts two changes from the previous attempt at this leak. Advancing the capture boundary on a later walk is inert once the sweep fences on re-derivation -- both key on the same set of rows, so the new term short-circuits for exactly the rows whose boundary it advanced. The extra ladder refresh was a duplicate full process-table read: close() already awaits tree.refresh() immediately before proveClaudeChildExit, on the only path that reaches it. Known property: the kill lands roughly a grace window after the walk that proved membership, so a pid recycled inside that gap could in principle be signalled. It is bounded -- matchingSnapshotRows already requires the live row to carry the same start-second and pgid, so an impostor must be born in the remainder of that one second, land on that exact pid, and sit in the same process group, and it has already received the unfenced SIGTERM from the same loop. * Run the Claude structured integration suite as a runtime client The suite exercises agentSession.* for Claude, not the mobile surface: nothing in it asserts anything mobile-specific and its sibling integration suites use 'runtime'. Mobile now additionally requires the experimental structured-chat setting, which structured-agent-session.test.ts pins in both states, so the stale 'mobile' fixture was claiming coverage it never had. * fix(claude): report effort from get_settings, which is the only frame that has it The composer's Effort pill rendered blank in every structured session. This is not a missing source: the publication reads `effortLevel` off the `system/init` frame, and that frame has never carried an effort of any kind, while the correct value is already fetched at acquisition and thrown away on the auth diagnostic. Verified two ways -- a live get_settings probe against Claude Code 2.1.258, and the shipped binary's own init frame construction, which lists `model` and no effort. So `reportedOptions.effort` was always empty, the options reader dropped the key, and the pill had no value. Model survived only because `currentModelId()` has a fallback chain. The get_settings call acquisition already makes reports the session's current effort as `effective.effortLevel`; pass that into the publication instead. Selecting an effort already worked, so this is the arrival value only. The legacy PTY path is unaffected and must not be "fixed" to match: it reads its effort by parsing the startup banner (`CLAUDE_MODEL_EFFORT` in src/renderer/src/components/native-chat/claude-terminal-session-options.ts), which is why it shows a value where the structured path does not. Also removes the fixture that hid this: the fake init frame invented `effortLevel: 'high'`, a field the CLI does not send, which is why every gate stayed green over a value that is always empty in production. The fixture's get_settings now returns the real {applied, effective, sources} shape instead of a bare `{env: {}}`, so the two adapter tests that asserted an effort keep asserting it through the path production actually uses. The reader returns null rather than defaulting: an effort nothing measured would repeat the fixture's mistake, and a blank pill is the honest degradation if the provider ever renames the key. * fix(claude): only record an effort the child confirms it adopted apply_flag_settings answers `success` for an effort it then ignores. Measured against Claude Code 2.1.258: applying `bogus-effort-xyz` returns subtype "success" with no error while `applied.effort` stays at its previous value, and a valid `low` moves it. The option write treated the absence of a throw as adoption and recorded the requested value unconditionally, so Orca would show and persist an effort the child was not using, with nothing anywhere reporting a problem. Read the effort back after applying it, through the same reader the arrival value uses, and reject when the child reports a different one. A readback that could not be taken is not evidence of a refusal -- the apply itself succeeded -- so it still records; only a readback that disagrees rejects. Not reachable from today's picker, which offers catalog values only, but the CLI's effort catalog is server-delivered and has changed before, so a retired id would otherwise become a pill confidently displaying a setting that never took. * test(claude): assert the effort contract against the real binary The blank pill survived every gate because the only tests that touched it were fixture-backed, and the fixture invented the field. A test that pins the shape we read cannot catch the provider renaming the key, which is the failure mode that produced this defect. Asserts both halves against a live authenticated CLI: that no frame it publishes carries an effort at all, and that the session's current effort arrives through get_settings. Which frame proves the session varies by host -- this machine proves it with a SessionStart hook rather than a system/init frame -- so the negative half asserts over every published frame rather than picking one. Skips with the rest of the file when no authenticated CLI is present. * fix(claude): stop the synthesised content-part kinds leaking into the transcript Sending an image put a bare `claude · message:user:content:image` row between the user's bubble and the answer. Two causes, and only the second is a family. An image part counted as modelled only when `source.type === 'url'`, but claudeDispatchMessageContent sends a local attachment as a base64 source and the CLI replays that shape back, so every attached image was classified unmodelled. Accept the base64 and file sources Orca itself sends. The family is the real defect. `message::content:` kinds are synthesised at runtime from whatever `part.type` arrives, so unlike the top-level frame catalogue they can never be enumerated ahead of time -- the `?? 'timeline-substantive'` default then prints the synthesised name at a user who cannot act on it. That default is right for top-level frames, where "substantive" means show the frame; here it meant show our own vocabulary, which drops the content AND leaks the opcode. So an unrenderable part now renders a sentence saying exactly that, with the kind and payload still on the row's disclosure. A part that carries its own readable sentence keeps it -- the placeholder is a fallback, not an override. An unknown future part type is therefore visible, never silently dropped and never printed as a kind: the same principle as the effort readback, which records only what the provider confirms. * Declare agentSession.requestHandoff on the cross-version wire surface The manifest is a ratchet for cross-version reachability, so the method is declared with real HandoffParams rather than counted. requestHandoff is capability-gated through requireStructuredHost and has no client caller, so declaring it is the whole of the change. Also model two host capabilities the harness omitted: the stub host's supportsCreate, and the fake adapter's, without which adapterSupportsCreate falls through to a supportsLocation the fake also lacks. Every ensure was refused for the harness's silence rather than for its location. * Gate structured Claude session tabs on the client capability that names them The Claude structured lane deleted the projection's `agent !== 'codex'` filter and added CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY in the same commit, but never wired the constant to anything. Paired clients then received agent-session tabs for Claude, which no shipped client renders -- mobile's resolveMobileNativeChat returns null for every agent but codex, so the row listed and selected into a pane with neither chat nor terminal. Restore the filter behind the declared capability instead of the bare agent name. No client advertises it yet, so this matches main's behaviour today and becomes a negotiation a future client can opt into. * Confirm the structured Claude model against the model the CLI reports set_model answers success for any string, including a model it cannot resolve — the failure only surfaces when the turn runs — and get_settings reports the settings-file model, not the session's. The init frame that opens each turn is the only channel carrying the adopted model, so keep the session's reported model current from it instead of reading it once at acquisition. Also stop rejecting an effort the readback cannot represent: max is session-scoped and excluded from the persisted effortLevel, so a readback reporting the level underneath it is an absence of evidence, not a refusal. * Clear the session-option hedge when the provider confirms the value The pill claimed every option was unconfirmed for the life of the session: the renderer recorded each write as dispatched and nothing ever moved it, so a model the CLI had already reported back still read as unconfirmed. Carry the provider's own confirmation to the surface. Main reports which option ids the provider named rather than merely accepted, and the client re-reads options as a turn changes, because the frame that opens a turn is where the adopted model arrives. A value the provider has not reported stays hedged, including an effort whose readback could not be taken. The confirmed list is optional on the wire: a host that predates it sends nothing and the client keeps hedging, which is the behaviour it had. * Keep the model report current across an acquisition fence bump * Show the picked session-option value and let the provider report correct it The pill showed a "not confirmed" second tooltip line for any value we had sent but not yet seen reported back. Nothing acts on it, and for the PTY lane it was permanent — that transport has no report channel. The pill now shows the picked value immediately and the provider's per-turn report corrects it when the two disagree; a newer local write still outranks a report that precedes it. `dispatched` stays as a provenance member rather than collapsing into `applied`: it is produced independently by the PTY lane, and it is where the `confirmed` wire field lands, which would otherwise be unobservable. Effort keeps its readback and its rejection path. That matters more now, not less: with the hedge gone the rejection is the only user-visible failure signal on this surface, so a spurious one would be the loudest bug here. Skipping the readback for an effort the settings response structurally cannot echo is what prevents it — the response carries the persisted level, so reading it back for a session-scoped value would report the level underneath and fail a valid write. * Hedge a session-option value only when the terminal transport sent it Both lanes emit `dispatched`, so it could never say which one produced a value. The descriptor now carries the transport that built it, set once in the shared snapshot builder from a parameter that is required rather than defaulted — the builder is the only place a descriptor is constructed, so a new producer has to name its lane or fail to compile. The structured lane confirms every value from the provider's own per-turn report, which makes the hedge transient noise there. The terminal lane can only learn an outcome by parsing the screen back, and only for Claude: every other agent's `dispatched` value stays unconfirmed for the life of the session, so the line is the only signal that we sent something we never saw land. * Refuse an effort the session's model advertises no control for * Refuse tab mutations on a Claude row the client never negotiated The branch added a case asserting a client advertising only agent-session.structured.v1 may mutate a claude row. That is the same ungated behaviour the projection gate removes, encoded a second time — mutation authorization reads the projection, so hiding the row refuses the write. Assert that contract instead, and add the positive case for a client that does negotiate Claude rows. * Resolve the Claude session's current model in one place so the effort guard and the pill agree * Record an effort the child did not adopt instead of refusing the write apply_flag_settings answers success for an effort it then ignores, so the readback exists to detect that. Refusing on it made the detection a veto, and a veto is only correct if the readback can never be wrong about which model is current -- which it was, twice. The pre-flight guard already refuses a level the model advertises no control for, so the veto guarded a door that is now locked upstream. Keep the detection, drop the refusal: a disagreement records the child's own answer and omits the option from confirmed, so main stops vouching for a value the provider rejected without blocking the user's write. * Stop a slow whole-machine ps from being read as an absent process `ps -axo ...command=` pays a per-pid argv read: measured 1.15s for 1,948 processes (0.03s without `command=`), and CPU contention stretched the same capture to 6.0s. Two budgets sized for a cheap look then misreport a readable machine. The reader's 3s ceiling killed 6 of 20 consecutive captures at load 27, so every consumer answered "unverifiable" about a table it could read. Raise it to 15s, and stamp the capture instant at ps START so `capturedAgeMs` is the upper bound its contract promises -- a 6s capture used to report itself as freshly taken, understating staleness against a 5s kill gate. The TTL keys on completion so a slow capture still coalesces instead of forking ps per caller. `readStructuredTuiProcessIdentity` then spent its whole 5s wait inside one capture and concluded "no exact child" after a single look taken before the child existed (observed landing at ~3.5s). Absence needs a look that did not race the spawn, so require two captures before the deadline can end the loop. Both surfaced by the real-binary Claude TUI resume test, which failed ~1 in 5 under load; 14/14 now, 8 of those runs containing a capture the old 3s budget would have killed. * Let the desktop renderer negotiate Claude structured tabs The paired-client gate hides agent-session rows an agent the client cannot render. The desktop renderer's own IPC dispatches as clientKind 'runtime' advertising only agent-session.structured.v1, so the gate hid Claude rows from the surface this feature ships on. It renders them; it should say so. * Stop a slow process table from silently blinding every freshness gate Stamping `capturedAgeMs` at ps START made the number honest, and honest broke both consumers that read it. `ps -axo ...command=` measured 2.5-9.0s on an idle 2,002-process laptop and 4.0-18.6s at load 46, so the age it now reports lands past every budget: `planRelayPtySweep` refuses the stop as "too old", and the renderer's `admitRemoteForegroundEvidence` refuses the record outright. That second one is the expensive half and was outside the diff -- a refusal bumps `consecutiveInspectionErrors`, the poll scheduler backs off to its 10s floor, and agent-completion detection stops for the pane. The subsystem went blind on exactly the loaded hosts the honest stamp was meant to serve. The evidence-publishing read now gives up at 1,200ms instead of waiting out `PS_TIMEOUT_MS`. It is one budget for one question: these consumers ask whether an observation describes NOW, and past this it does not -- a late answer is refused by the age gate anyway, having first blocked a polled path for the whole capture, so a prompt `unverifiable` is both the truthful verdict and the cheap one. Both relay call sites already produce it from a rejection, and an admitted `unverifiable` costs a poll where a refusal costs the cadence. Identity proof keeps the full 15s through `getFreshProcessTableSnapshot`, because it asks whether a process EXISTS and must never read slow as absent. The budget bounds the wait, never the capture: the reader coalesces, so an abandoned wait leaves its capture running to fill the cache rather than forking a second whole-machine `ps` on the host that can least afford one. 1,200ms is bracketed rather than picked. The floor is the capture's own cost -- `command=` measured 1.15s for 1,948 processes on an idle host, and a budget under that answers `unverifiable` about a machine nobody is straining. The ceiling is the consumer's: 2,000ms, less the 500ms a TTL-shared capture may already have aged, leaves 1,500ms, and transit takes the rest. That ceiling only fits once the capture stops being charged twice. `ps` runs inside the RPC round trip, so its duration is already in `receiveDelay`, and `capturedAgeMs` is that same duration on the host's clock; summing them halved the budget this gate grants a host from ~2.0s of `ps` to ~1.0s, which is why a 1.2s capture arriving at 1.3s read as 2.5s old and was refused. Admission now takes the larger of the two. The sweep's gate keeps its sum, which is correct there: `evidenceAgeSinceListingMs` is stamped after the listing ARRIVES, so it measures planning time and overlaps nothing. A stated limit rather than an assumed one: 15s is not proven sufficient for identity proof. The same capture reached 18.6s at load 46, so that path can still time out and answer "no exact child" about a host it simply could not read in time. Narrowing it needs a cheaper question than a whole-machine argv read, not a larger number. The one test guarding this field could not fail. `beginPtyHandlerTest` installs fake timers, so `Date.now()` is frozen, the real reader reports exactly +0, and `0 <= 500` held identically for a hardcoded zero, for completion-stamping and for start-stamping -- while the real reader on that host returns thousands of ms. It now drives a measured age in and asserts the handler publishes it rather than restamping; that the reader MEASURES it correctly stays pinned separately, against a controllable clock. Both consumers get boundary coverage either side, and each new gate was ablated red before it went green. * Keep the compatibility fields off the capture the budget just abandoned inspectProcess falls back to processHasChildren and listProcesses to getForegroundProcessName, and both read the same TTL-shared capture with no budget of their own. On a slow host they joined the in-flight capture the budgeted evidence read had just given up on, so the call still blocked for the full 6-18s and the budget bought nothing -- once for inspectProcess and once per managed PTY for listProcesses. Use the degraded answers those helpers already give for an unreadable table, reached promptly. pty.hasChildProcesses keeps its unbudgeted fresh probe: it is a one-shot destructive gate that can afford to wait. --------- Co-authored-by: Merge Sim Co-authored-by: Merge Sim --- .../scripts/verify-localization-catalog.mjs | 7 +- ...bileNativeChatSessionOptionPickers.test.ts | 44 + .../MobileNativeChatSessionOptionPickers.tsx | 9 +- .../use-mobile-native-chat-session-options.ts | 3 +- package.json | 1 + pnpm-lock.yaml | 133 ++- pnpm-workspace.yaml | 14 + .../claude-structured-auth-policy.test.ts | 168 ++++ .../claude-structured-auth-policy.ts | 37 + src/main/claude-accounts/environment.ts | 101 +- src/main/claude-accounts/live-pty-gate.ts | 73 ++ .../runtime-auth/runtime-auth-preparation.ts | 6 +- .../claude-agent-sdk-scripted-cli.mjs | 144 +++ .../claude-agent-sdk-contract-pins.test.ts | 519 ++++++++++ .../claude-agent-sdk-control-requests.ts | 154 +++ ...aude-agent-sdk-exit-proof-identity.test.ts | 104 ++ .../claude-agent-sdk-exit-proof.test.ts | 934 ++++++++++++++++++ .../claude/claude-agent-sdk-exit-proof.ts | 366 +++++++ .../claude-agent-sdk-import-boundary.test.ts | 154 +++ .../claude-agent-sdk-process-spawn.test.ts | 107 ++ .../claude/claude-agent-sdk-process-spawn.ts | 69 ++ ...laude-agent-sdk-root-kill-fallback.test.ts | 190 ++++ ...laude-agent-sdk-user-message-queue.test.ts | 65 ++ .../claude-agent-sdk-user-message-queue.ts | 100 ++ .../claude/claude-child-exit-proof-ladder.ts | 41 + .../claude-child-process-environment.test.ts | 63 ++ .../claude-child-process-environment.ts | 69 ++ .../claude/claude-child-root-termination.ts | 54 + src/main/claude/claude-child-tree-snapshot.ts | 128 +++ .../claude-command-lifecycle-frames.test.ts | 115 +++ src/main/claude/claude-config-dir-pin.test.ts | 34 + src/main/claude/claude-config-dir-pin.ts | 37 + ...ude-descendant-escalation-boundary.test.ts | 124 +++ ...laude-stream-json-connection-close.test.ts | 126 +++ .../claude-stream-json-connection.test.ts | 768 ++++++++++++++ .../claude/claude-stream-json-connection.ts | 283 ++++++ .../claude/claude-streamed-block-identity.ts | 110 +++ .../claude-streamed-text-checkpoints.test.ts | 93 ++ .../claude-streamed-text-checkpoints.ts | 105 ++ .../claude-structured-acquisition-release.ts | 43 + .../claude-structured-auth-parity.test.ts | 235 +++++ .../claude-structured-content-parts.test.ts | 98 ++ .../claude-structured-control-actions.test.ts | 113 +++ .../claude-structured-control-actions.ts | 60 ++ .../claude-structured-dispatch-content.ts | 165 ++++ .../claude/claude-structured-dispatch.test.ts | 598 +++++++++++ src/main/claude/claude-structured-dispatch.ts | 264 +++++ ...claude-structured-effort-reporting.test.ts | 257 +++++ .../claude-structured-inbound-control.test.ts | 164 +++ .../claude-structured-inbound-control.ts | 91 ++ .../claude/claude-structured-init-deadline.ts | 68 ++ .../claude/claude-structured-init-proof.ts | 88 ++ .../claude-structured-item-translation.ts | 179 ++++ ...ude-structured-journal-translation.test.ts | 811 +++++++++++++++ .../claude-structured-journal-translation.ts | 287 ++++++ ...laude-structured-launch-resolution.test.ts | 392 ++++++++ .../claude-structured-launch-resolution.ts | 273 +++++ ...claude-structured-location-support.test.ts | 91 ++ .../claude-structured-location-support.ts | 11 + ...aude-structured-model-confirmation.test.ts | 209 ++++ ...ude-structured-option-confirmation.test.ts | 193 ++++ .../claude/claude-structured-options.test.ts | 51 + src/main/claude/claude-structured-options.ts | 147 +++ .../claude-structured-owner-identity.test.ts | 39 + .../claude-structured-owner-identity.ts | 41 + .../claude-structured-prompt-items.test.ts | 110 +++ .../claude/claude-structured-prompt-items.ts | 134 +++ .../claude-structured-prompt-replies.ts | 297 ++++++ ...laude-structured-provider-fallback.test.ts | 117 +++ .../claude-structured-provider-fallback.ts | 125 +++ .../claude/claude-structured-real-cli.test.ts | 299 ++++++ ...ed-session-acquisition-processless.test.ts | 70 ++ .../claude-structured-session-acquisition.ts | 298 ++++++ .../claude-structured-session-adapter.test.ts | 891 +++++++++++++++++ .../claude-structured-session-adapter.ts | 246 +++++ .../claude-structured-session-close.test.ts | 50 + .../claude/claude-structured-session-close.ts | 256 +++++ .../claude-structured-session-options.ts | 183 ++++ .../claude-structured-session-publication.ts | 68 ++ ...claude-structured-session-recovery.test.ts | 619 ++++++++++++ .../claude/claude-structured-session-state.ts | 276 ++++++ .../claude-structured-session-test-support.ts | 260 +++++ .../claude/claude-transcript-branch-proof.ts | 135 ++- src/main/claude/claude-tui-exit.test.ts | 159 +++ src/main/claude/claude-tui-exit.ts | 119 +++ .../claude/claude-tui-resume-launch.test.ts | 227 +++++ src/main/claude/claude-tui-resume-launch.ts | 101 ++ .../claude/claude-tui-resume-proof.test.ts | 80 ++ src/main/claude/claude-tui-resume-proof.ts | 111 +++ ...tui-resume-real-binary.integration.test.ts | 279 ++++++ .../codex-structured-session-close.test.ts | 35 + src/main/ipc/pty/ipc/spawn-env.ts | 12 +- src/main/ipc/pty/ipc/spawn-preflight.ts | 3 +- src/main/ipc/pty/runtime/spawn-preflight.ts | 14 +- src/main/ipc/runtime.test.ts | 34 + src/main/ipc/runtime.ts | 15 +- .../claude-stream-json-frame-schema.ts | 12 +- .../provider-frame-disposition.test.ts | 23 + .../provider-frame-disposition.ts | 13 +- ...tured-agent-session-adapter-router.test.ts | 114 +++ ...structured-agent-session-adapter-router.ts | 124 +++ .../structured-agent-session-adapter.test.ts | 22 + .../structured-agent-session-adapter.ts | 34 +- ...structured-agent-session-attach-context.ts | 3 - .../structured-agent-session-attach-flow.ts | 13 +- ...ured-agent-session-attach-orchestration.ts | 23 +- .../structured-agent-session-attach.ts | 3 +- ...-session-claude-options-round-trip.test.ts | 182 ++++ ...tured-agent-session-grouped-prompt.test.ts | 166 ++++ ...uctured-agent-session-handoff-admission.ts | 138 +++ ...-agent-session-handoff-flow-runner.test.ts | 98 ++ ...tured-agent-session-handoff-flow-runner.ts | 110 +++ ...nt-session-handoff-operation-guard.test.ts | 221 +++++ ...d-agent-session-handoff-operation-guard.ts | 125 +++ ...ured-agent-session-handoff-options.test.ts | 286 ++++++ ...tructured-agent-session-handoff-options.ts | 23 + ...tured-agent-session-handoff-queue-start.ts | 46 + .../structured-agent-session-handoff-queue.ts | 133 +++ ...tructured-agent-session-handoff-recover.ts | 30 + ...structured-agent-session-handoff-result.ts | 32 + ...ured-agent-session-handoff-revalidation.ts | 52 + ...ured-agent-session-handoff-reverse.test.ts | 100 ++ ...tructured-agent-session-handoff-reverse.ts | 14 +- ...-agent-session-handoff-test-coordinator.ts | 80 ++ ...d-agent-session-handoff-test-identities.ts | 40 + ...red-agent-session-handoff-test-requests.ts | 53 + .../structured-agent-session-handoff.ts | 277 +++++- .../structured-agent-session-host.ts | 17 +- ...tructured-agent-session-manual-recovery.ts | 103 ++ ...ctured-agent-session-option-restoration.ts | 10 +- ...ed-agent-session-proven-dead-retry.test.ts | 178 ++++ ...tured-agent-session-recovery-exits.test.ts | 2 +- .../structured-agent-session-turns-prompt.ts | 20 +- .../unhandled-provider-frame.ts | 5 +- ...structured-managed-account-support.test.ts | 154 +++ ...aude-structured-managed-account-support.ts | 61 ++ ...session-file-resolver-claude-roots.test.ts | 97 ++ .../native-chat/session-file-resolver.test.ts | 303 +++++- src/main/native-chat/session-file-resolver.ts | 40 +- ...tured-agent-session-create-support.test.ts | 83 ++ ...structured-agent-session-create-support.ts | 48 + .../windows-foreground-process-rows.ts | 20 + src/main/pty-descendant-exit-verification.ts | 147 ++- src/main/pty-descendant-termination.test.ts | 146 ++- src/main/pty-descendant-termination.ts | 44 +- ...-session-acquisition-failure-settlement.ts | 76 +- .../agent-session-launch-env-backfill.test.ts | 90 ++ .../agent-session-record-options.test.ts | 14 + .../runtime/agent-session-resume-args.test.ts | 33 + src/main/runtime/agent-session-resume-args.ts | 17 + ...ude-structured-session-integration.test.ts | 747 ++++++++++++++ ...e-get-agent-session-execution-namespace.ts | 21 +- .../runtime/orca-runtime-get-worktree-ps.ts | 41 +- ...lve-recovered-structured-tui-transcript.ts | 54 +- ...tore-structured-agent-session-tabs-once.ts | 8 +- ...ctured-agent-session-create-intent.test.ts | 104 ++ ...ructured-agent-session-launch-args.test.ts | 86 ++ ...ime-structured-agent-session-launch-tui.ts | 7 +- ...ime-structured-claude-account-gate.test.ts | 112 +++ ...time-structured-claude-gate-wiring.test.ts | 64 ++ ...runtime-structured-session-restore.test.ts | 50 + ...runtime-structured-tui-tab-binding.test.ts | 730 ++++++++++++++ src/main/runtime/rpc/e2ee-channel-v2.test.ts | 19 + .../runtime/rpc/methods/clipboard.test.ts | 100 +- src/main/runtime/rpc/methods/clipboard.ts | 72 +- .../methods/mobile-markdown-tab-methods.ts | 22 + ...ion-tab-agent-capability-mutations.test.ts | 14 +- ...ession-tab-agent-status-projection.test.ts | 134 ++- .../session-tab-agent-status-projection.ts | 10 +- .../structured-agent-session-schemas.ts | 13 +- .../methods/structured-agent-session.test.ts | 116 ++- .../rpc/methods/structured-agent-session.ts | 55 +- .../mobile-clipboard-image-provenance.test.ts | 53 + .../rpc/mobile-clipboard-image-provenance.ts | 90 ++ .../rpc/mobile-e2ee-v2-client-capabilities.ts | 21 + .../runtime/rpc/mobile-socket-wiring.test.ts | 82 ++ src/main/runtime/rpc/mobile-socket-wiring.ts | 4 +- ...d-agent-session-integration-replay.test.ts | 2 + ...ructured-agent-session-integration.test.ts | 1 + .../structured-agent-session-owner-probe.ts | 108 ++ ...uctured-agent-session-runtime-exit.test.ts | 3 + .../structured-agent-session-runtime.test.ts | 9 +- .../structured-agent-session-runtime.ts | 170 ++-- ...ructured-claude-auth-policy-wiring.test.ts | 59 ++ .../structured-claude-runtime-adapter.ts | 98 ++ .../structured-tui-process-identity.test.ts | 35 + .../structured-tui-process-identity.ts | 12 +- ...ndows-descendant-exit-verification.test.ts | 175 ++++ .../windows-descendant-exit-verification.ts | 156 +++ .../pty-handler-ownership-attestation.test.ts | 42 +- src/relay/pty-handler-spawn-admission.test.ts | 40 + src/relay/pty-handler.ts | 21 +- .../NativeChatQuestionCard.test.tsx | 69 +- .../native-chat/NativeChatQuestionCard.tsx | 5 +- .../NativeChatSessionOptionPickers.test.tsx | 66 +- .../NativeChatSessionOptionPickers.tsx | 13 +- .../NativeChatStructuredSession.test.tsx | 147 ++- .../NativeChatStructuredSession.tsx | 54 +- ...ructuredAgentSessionHandoffChrome.test.tsx | 58 ++ .../StructuredAgentSessionHandoffChrome.tsx | 225 +++++ .../native-chat-pty-session-options.test.ts | 1 + .../native-chat-pty-session-options.ts | 6 +- .../native-chat-session-option-labels.test.ts | 1 + .../native-chat-session-option-snapshot.ts | 4 +- .../use-structured-agent-session.ts | 8 +- .../settings/ExperimentalPane.test.tsx | 6 +- .../NativeChatExperimentalSetting.tsx | 4 +- .../components/sidebar/NonGitFolderDialog.tsx | 12 +- .../folder-workspace-composer-submit.ts | 9 +- .../components/tab-bar/QuickLaunchButton.tsx | 41 +- .../TabBarCreateEntry.keyboard.test.tsx | 2 +- .../components/tab-bar/TabBarCreateEntry.tsx | 34 +- ...abCloseCommands.structured-session.test.ts | 8 +- .../tab-group/useTabGroupTabCloseCommands.ts | 10 +- ...cturedAgentSessionTerminalReturnButton.tsx | 25 + ...-completion-stale-evidence-backoff.test.ts | 136 +++ .../terminal/terminal-tab-actions.ts | 2 +- .../full-creation-structured-launch.test.ts | 12 +- .../full-creation-structured-launch.ts | 9 +- src/renderer/src/i18n/locales/en.json | 43 +- .../src/lib/agent-launch-routing.test.ts | 72 +- src/renderer/src/lib/agent-launch-routing.ts | 8 +- .../src/lib/launch-agent-in-new-tab.ts | 7 +- ...launch-agent-structured-chat-guard.test.ts | 62 +- .../launch-structured-agent-session.test.ts | 244 +++++ .../lib/launch-structured-agent-session.ts | 159 +++ .../launch-structured-codex-session.test.ts | 87 -- .../lib/launch-structured-codex-session.ts | 87 -- ...nch-work-item-direct-agent-routing.test.ts | 12 +- .../launch-work-item-direct-agent-routing.ts | 9 +- ...structured-agent-session-launch-callers.ts | 10 +- .../structured-agent-session-launch-prompt.ts | 2 +- ...tructured-agent-session-launch-recovery.ts | 22 +- .../structured-agent-session-launch.test.ts | 130 ++- .../lib/structured-agent-session-launch.ts | 73 +- .../src/lib/worktree-creation-flow-execute.ts | 3 +- ...rktree-creation-structured-session.test.ts | 16 +- .../worktree-creation-structured-session.ts | 15 +- .../structured-agent-session-handoff-store.ts | 50 + .../src/store/repos/repo-add-actions.ts | 12 +- .../agent-status-provider-session.test.ts | 21 + src/shared/agent-session-journal-schemas.ts | 20 +- src/shared/agent-session-journal-types.ts | 12 + .../agent-session-question-answer.test.ts | 49 + src/shared/agent-session-question-answer.ts | 84 ++ src/shared/agent-session-wire.ts | 6 + ...ative-chat-session-option-snapshot.test.ts | 64 +- .../native-chat-session-option-snapshot.ts | 12 +- .../native-chat-session-option-state.ts | 15 +- src/shared/native-chat-session-options.ts | 22 +- src/shared/process-table-snapshot-reader.ts | 63 +- src/shared/process-table-snapshot.test.ts | 139 ++- src/shared/protocol-version.ts | 4 + .../remote-foreground-evidence-admission.ts | 11 +- src/shared/remote-foreground-evidence.test.ts | 70 +- .../runtime-mobile-session-tab-contracts.ts | 2 +- .../ssh-relay-pty-ownership-proof.test.ts | 46 + .../structured-agent-session-mutation.ts | 27 + .../structured-agent-session-options.test.ts | 6 +- .../structured-agent-session-options.ts | 12 +- ...ss-version-agent-session-wire.unit.test.ts | 21 + 261 files changed, 26210 insertions(+), 887 deletions(-) create mode 100644 src/main/claude-accounts/claude-structured-auth-policy.test.ts create mode 100644 src/main/claude-accounts/claude-structured-auth-policy.ts create mode 100644 src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs create mode 100644 src/main/claude/claude-agent-sdk-contract-pins.test.ts create mode 100644 src/main/claude/claude-agent-sdk-control-requests.ts create mode 100644 src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts create mode 100644 src/main/claude/claude-agent-sdk-exit-proof.test.ts create mode 100644 src/main/claude/claude-agent-sdk-exit-proof.ts create mode 100644 src/main/claude/claude-agent-sdk-import-boundary.test.ts create mode 100644 src/main/claude/claude-agent-sdk-process-spawn.test.ts create mode 100644 src/main/claude/claude-agent-sdk-process-spawn.ts create mode 100644 src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts create mode 100644 src/main/claude/claude-agent-sdk-user-message-queue.test.ts create mode 100644 src/main/claude/claude-agent-sdk-user-message-queue.ts create mode 100644 src/main/claude/claude-child-exit-proof-ladder.ts create mode 100644 src/main/claude/claude-child-process-environment.test.ts create mode 100644 src/main/claude/claude-child-process-environment.ts create mode 100644 src/main/claude/claude-child-root-termination.ts create mode 100644 src/main/claude/claude-child-tree-snapshot.ts create mode 100644 src/main/claude/claude-command-lifecycle-frames.test.ts create mode 100644 src/main/claude/claude-config-dir-pin.test.ts create mode 100644 src/main/claude/claude-config-dir-pin.ts create mode 100644 src/main/claude/claude-descendant-escalation-boundary.test.ts create mode 100644 src/main/claude/claude-stream-json-connection-close.test.ts create mode 100644 src/main/claude/claude-stream-json-connection.test.ts create mode 100644 src/main/claude/claude-stream-json-connection.ts create mode 100644 src/main/claude/claude-streamed-block-identity.ts create mode 100644 src/main/claude/claude-streamed-text-checkpoints.test.ts create mode 100644 src/main/claude/claude-streamed-text-checkpoints.ts create mode 100644 src/main/claude/claude-structured-acquisition-release.ts create mode 100644 src/main/claude/claude-structured-auth-parity.test.ts create mode 100644 src/main/claude/claude-structured-content-parts.test.ts create mode 100644 src/main/claude/claude-structured-control-actions.test.ts create mode 100644 src/main/claude/claude-structured-control-actions.ts create mode 100644 src/main/claude/claude-structured-dispatch-content.ts create mode 100644 src/main/claude/claude-structured-dispatch.test.ts create mode 100644 src/main/claude/claude-structured-dispatch.ts create mode 100644 src/main/claude/claude-structured-effort-reporting.test.ts create mode 100644 src/main/claude/claude-structured-inbound-control.test.ts create mode 100644 src/main/claude/claude-structured-inbound-control.ts create mode 100644 src/main/claude/claude-structured-init-deadline.ts create mode 100644 src/main/claude/claude-structured-init-proof.ts create mode 100644 src/main/claude/claude-structured-item-translation.ts create mode 100644 src/main/claude/claude-structured-journal-translation.test.ts create mode 100644 src/main/claude/claude-structured-journal-translation.ts create mode 100644 src/main/claude/claude-structured-launch-resolution.test.ts create mode 100644 src/main/claude/claude-structured-launch-resolution.ts create mode 100644 src/main/claude/claude-structured-location-support.test.ts create mode 100644 src/main/claude/claude-structured-location-support.ts create mode 100644 src/main/claude/claude-structured-model-confirmation.test.ts create mode 100644 src/main/claude/claude-structured-option-confirmation.test.ts create mode 100644 src/main/claude/claude-structured-options.test.ts create mode 100644 src/main/claude/claude-structured-options.ts create mode 100644 src/main/claude/claude-structured-owner-identity.test.ts create mode 100644 src/main/claude/claude-structured-prompt-items.test.ts create mode 100644 src/main/claude/claude-structured-prompt-items.ts create mode 100644 src/main/claude/claude-structured-prompt-replies.ts create mode 100644 src/main/claude/claude-structured-provider-fallback.test.ts create mode 100644 src/main/claude/claude-structured-provider-fallback.ts create mode 100644 src/main/claude/claude-structured-real-cli.test.ts create mode 100644 src/main/claude/claude-structured-session-acquisition-processless.test.ts create mode 100644 src/main/claude/claude-structured-session-acquisition.ts create mode 100644 src/main/claude/claude-structured-session-adapter.test.ts create mode 100644 src/main/claude/claude-structured-session-adapter.ts create mode 100644 src/main/claude/claude-structured-session-close.test.ts create mode 100644 src/main/claude/claude-structured-session-close.ts create mode 100644 src/main/claude/claude-structured-session-options.ts create mode 100644 src/main/claude/claude-structured-session-publication.ts create mode 100644 src/main/claude/claude-structured-session-recovery.test.ts create mode 100644 src/main/claude/claude-structured-session-state.ts create mode 100644 src/main/claude/claude-structured-session-test-support.ts create mode 100644 src/main/claude/claude-tui-exit.test.ts create mode 100644 src/main/claude/claude-tui-exit.ts create mode 100644 src/main/claude/claude-tui-resume-launch.test.ts create mode 100644 src/main/claude/claude-tui-resume-launch.ts create mode 100644 src/main/claude/claude-tui-resume-proof.test.ts create mode 100644 src/main/claude/claude-tui-resume-proof.ts create mode 100644 src/main/claude/claude-tui-resume-real-binary.integration.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts create mode 100644 src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts create mode 100644 src/main/native-chat/claude-structured-managed-account-support.test.ts create mode 100644 src/main/native-chat/claude-structured-managed-account-support.ts create mode 100644 src/main/native-chat/session-file-resolver-claude-roots.test.ts create mode 100644 src/main/native-chat/structured-agent-session-create-support.test.ts create mode 100644 src/main/native-chat/structured-agent-session-create-support.ts create mode 100644 src/main/runtime/agent-session-launch-env-backfill.test.ts create mode 100644 src/main/runtime/agent-session-resume-args.test.ts create mode 100644 src/main/runtime/agent-session-resume-args.ts create mode 100644 src/main/runtime/claude-structured-session-integration.test.ts create mode 100644 src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts create mode 100644 src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts create mode 100644 src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts create mode 100644 src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts create mode 100644 src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts create mode 100644 src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts create mode 100644 src/main/runtime/rpc/mobile-clipboard-image-provenance.ts create mode 100644 src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts create mode 100644 src/main/runtime/structured-agent-session-owner-probe.ts create mode 100644 src/main/runtime/structured-claude-auth-policy-wiring.test.ts create mode 100644 src/main/runtime/structured-claude-runtime-adapter.ts create mode 100644 src/main/windows-descendant-exit-verification.test.ts create mode 100644 src/main/windows-descendant-exit-verification.ts create mode 100644 src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx create mode 100644 src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx create mode 100644 src/renderer/src/components/terminal-pane/StructuredAgentSessionTerminalReturnButton.tsx create mode 100644 src/renderer/src/components/terminal-pane/agent-completion-stale-evidence-backoff.test.ts create mode 100644 src/renderer/src/lib/launch-structured-agent-session.test.ts create mode 100644 src/renderer/src/lib/launch-structured-agent-session.ts delete mode 100644 src/renderer/src/lib/launch-structured-codex-session.test.ts delete mode 100644 src/renderer/src/lib/launch-structured-codex-session.ts create mode 100644 src/renderer/src/runtime/structured-agent-session-handoff-store.ts create mode 100644 src/shared/agent-session-question-answer.test.ts create mode 100644 src/shared/agent-session-question-answer.ts diff --git a/config/scripts/verify-localization-catalog.mjs b/config/scripts/verify-localization-catalog.mjs index 002a84360f6..a73e9d5e3cc 100644 --- a/config/scripts/verify-localization-catalog.mjs +++ b/config/scripts/verify-localization-catalog.mjs @@ -11,7 +11,12 @@ import { repairTranslatedValue } from './locale-translation-policy.mjs' const SOURCE_EXTENSIONS = new Set(['.ts', '.tsx', '.js', '.jsx', '.mts', '.cts']) const SKIP_PATH_PARTS = new Set(['.git', 'dist', 'node_modules', 'out', '__snapshots__', 'assets']) -const LOCALIZATION_FUNCTION_NAMES = new Set(['t', 'translate', 'translateMain', 'translateSearchKeyword']) +const LOCALIZATION_FUNCTION_NAMES = new Set([ + 't', + 'translate', + 'translateMain', + 'translateSearchKeyword' +]) const PLACEHOLDER_RE = /\{\{[^}]+\}\}/g const LOCALES_RELATIVE_DIR = path.join('src', 'renderer', 'src', 'i18n', 'locales') export const LOCALIZATION_SOURCE_ROOTS = [ diff --git a/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts b/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts index 4430c60115a..5a959b870bf 100644 --- a/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts +++ b/mobile/src/session/MobileNativeChatSessionOptionPickers.test.ts @@ -41,6 +41,7 @@ const MODEL_DESCRIPTOR: SessionOptionDescriptor = { ] }, valueSource: 'reported', + transport: 'catalog', settable: true } @@ -57,6 +58,7 @@ const EFFORT_DESCRIPTOR: SessionOptionDescriptor = { ] }, valueSource: 'dispatched', + transport: 'catalog', settable: true } @@ -66,6 +68,7 @@ const FAST_MODE_DESCRIPTOR: SessionOptionDescriptor = { category: 'mode', kind: { type: 'boolean', currentValue: false }, valueSource: 'reported', + transport: 'catalog', settable: true } @@ -227,6 +230,7 @@ describe('MobileNativeChatSessionOptionPickers', () => { ...MODEL_DESCRIPTOR, kind: { type: 'select', choices: [] }, valueSource: 'unknown', + transport: 'catalog', action: { type: 'agent-picker' } } ]) @@ -236,6 +240,46 @@ describe('MobileNativeChatSessionOptionPickers', () => { expect(invokeAction).toHaveBeenCalledWith('model') }) + // The terminal transport can only learn the outcome by parsing the screen back, + // so the sheet admits the value is unconfirmed; the structured transport reports + // it every turn, which makes the same caption noise there. + it.each([ + { transport: 'catalog' as const, caption: true }, + { transport: 'agent-session' as const, caption: false } + ])('captions a dispatched value only on the terminal transport', async (scenario) => { + mount([ + MODEL_DESCRIPTOR, + { ...EFFORT_DESCRIPTOR, valueSource: 'dispatched', transport: scenario.transport } + ]) + await act(async () => pill('Model').props.onPress()) + await act(async () => rowByText('Effort').props.onPress()) + const captions = renderer!.root + .findAll((node) => node.type === 'Text') + .filter( + (node) => + (node.props as { children?: unknown }).children === 'Sent to the agent — not confirmed' + ) + expect(captions.length > 0).toBe(scenario.caption) + }) + + it.each(['catalog', 'agent-session'] as const)( + 'does not caption a reported value on the %s transport', + async (transport) => { + mount([MODEL_DESCRIPTOR, { ...EFFORT_DESCRIPTOR, valueSource: 'reported', transport }]) + await act(async () => pill('Model').props.onPress()) + await act(async () => rowByText('Effort').props.onPress()) + expect( + renderer!.root + .findAll((node) => node.type === 'Text') + .some( + (node) => + (node.props as { children?: unknown }).children === + 'Sent to the agent — not confirmed' + ) + ).toBe(false) + } + ) + it('locks the pills while the agent is working', () => { mount([MODEL_DESCRIPTOR, EFFORT_DESCRIPTOR], true) expect(pill('Model').props).toMatchObject({ disabled: true }) diff --git a/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx b/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx index bfa5244a398..c9d641f74ec 100644 --- a/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx +++ b/mobile/src/session/MobileNativeChatSessionOptionPickers.tsx @@ -3,9 +3,10 @@ import { ActivityIndicator, Keyboard, Pressable, StyleSheet, Text, View } from ' import { ChevronLeft, X } from 'lucide-react-native' import { BottomDrawer } from '../components/BottomDrawer' import { colors, radii, spacing, typography } from '../theme/mobile-theme' -import type { - SessionOptionDescriptor, - SessionOptionValue +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor, + type SessionOptionValue } from '../../../src/shared/native-chat-session-options' import { mobileModelPillLabel, @@ -119,7 +120,7 @@ export function MobileNativeChatSessionOptionPickers({ ) : null} - {activeDescriptor.valueSource === 'dispatched' ? ( + {sessionOptionDispatchUnconfirmed(activeDescriptor) ? ( Sent to the agent — not confirmed ) : null} {reason ? {reason} : null} diff --git a/mobile/src/session/use-mobile-native-chat-session-options.ts b/mobile/src/session/use-mobile-native-chat-session-options.ts index 66a929aeb56..6acdf8b3750 100644 --- a/mobile/src/session/use-mobile-native-chat-session-options.ts +++ b/mobile/src/session/use-mobile-native-chat-session-options.ts @@ -168,7 +168,8 @@ export function useMobileNativeChatSessionOptions(args: { models: activeModels(catalog, record), record, mode: 'live', - modelLabel: 'Model' + modelLabel: 'Model', + liveTransport: 'catalog' }) }, [agent, catalog, scopeKey, version]) diff --git a/package.json b/package.json index 58f4805d937..c519d7ea6b1 100644 --- a/package.json +++ b/package.json @@ -153,6 +153,7 @@ "repro:live-remote-realistic-freeze": "node config/scripts/live-remote-realistic-freeze-repro.mjs" }, "dependencies": { + "@anthropic-ai/claude-agent-sdk": "0.3.251", "@electron-toolkit/preload": "^3.0.2", "@electron-toolkit/utils": "^4.0.0", "@floating-ui/dom": "1.7.6", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index d7481f2e556..6b59d23e026 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -122,6 +122,9 @@ importers: .: dependencies: + '@anthropic-ai/claude-agent-sdk': + specifier: 0.3.251 + version: 0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4) '@electron-toolkit/preload': specifier: ^3.0.2 version: 3.0.2(electron@43.4.1(supports-color@7.2.0)) @@ -535,6 +538,23 @@ packages: '@antfu/install-pkg@1.1.0': resolution: {integrity: sha512-MGQsmw10ZyI+EJo45CdSER4zEb+p31LpDAFp2Z3gkSd1yqVZGi0Ebx++YTEMonJy4oChEMLsxZ64j8FH6sSqtQ==} + '@anthropic-ai/claude-agent-sdk@0.3.251': + resolution: {integrity: sha512-DqSi8mH2tQYRlVV0G+lJnQ/WbjJZ/a+8cJ3vPuYoqh8esIIvXHm1ZOXV1UPGsFYRnbBytEoiSGitguEXd+sQ+Q==} + engines: {node: '>=18.0.0'} + peerDependencies: + '@anthropic-ai/sdk': '>=0.93.0' + '@modelcontextprotocol/sdk': ^1.29.0 + zod: ^4.0.0 + + '@anthropic-ai/sdk@0.122.0': + resolution: {integrity: sha512-GGPNftt0caaz9MDlmNQGHX8855Ojaduyy5pm9Sm1h7HalCn0cWNb5/bweadJF+4yzbal+QL6ztBa09WAAOzLmQ==} + hasBin: true + peerDependencies: + zod: ^3.25.0 || ^4.0.0 + peerDependenciesMeta: + zod: + optional: true + '@babel/code-frame@7.29.7': resolution: {integrity: sha512-Aup7aUOfpbAUg2ROOJN6Iw5f9DMBlzu0mIkm/malLQFN/YQgO48wCj0Kxa3sEHJvPVFg7siR+qRInwXd2qhQKw==} engines: {node: '>=6.9.0'} @@ -2628,6 +2648,9 @@ packages: resolution: {integrity: sha512-tlqY9xq5ukxTUZBmoOp+m61cqwQD5pHJtFY3Mn8CA8ps6yghLH/Hw8UPdqg4OLmFW3IFlcXnQNmo/dh8HzXYIQ==} engines: {node: '>=18'} + '@stablelib/base64@1.0.1': + resolution: {integrity: sha512-1bnPQqSxSuc3Ii6MhBysoWCg58j97aUjuCSZrGSmDxNqtytIi0k8utUenAwTZN4V5mXXYGsVUI9zeBqy+jBOSQ==} + '@stablyai/playwright-base@2.1.14': resolution: {integrity: sha512-/iAgMW5tC0ETDo3mFyTzszRrD7rGFIT4fgDgtZxqa9vPhiTLix/1+GeOOBNY0uS+XRLFY0Uc/irsC3XProL47g==} engines: {node: '>=18'} @@ -4493,6 +4516,9 @@ packages: resolution: {integrity: sha512-7MptL8U0cqcFdzIzwOTHoilX9x5BrNqye7Z/LuC7kCMRio1EMSyqRK3BEAUD7sXRq4iT4AzTVuZdhgQ2TCvYLg==} engines: {node: '>=8.6.0'} + fast-sha256@1.3.0: + resolution: {integrity: sha512-n11RGP/lrWEFI/bWdygLxhI+pVeo1ZYIVwvvPkW7azl/rOy+F3HYRZ2K5zeE9mmkhQppyv9sQFx0JM9UabnpPQ==} + fast-string-truncated-width@3.0.3: resolution: {integrity: sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g==} @@ -5039,6 +5065,10 @@ packages: json-parse-even-better-errors@2.3.1: resolution: {integrity: sha512-xyFwyhro/JEof6Ghe2iz2NcXoj2sloNsWr/XsERDK/oiPCfaNhl5ONfp+jQdAZRQQ0IJWNzH9zIZF7li91kh2w==} + json-schema-to-ts@3.1.1: + resolution: {integrity: sha512-+DWg8jCJG2TEnpy7kOm/7/AxaYoaRbjVB4LFZLySZlWn8exGs3A4OLJR966cVvU26N7X9TWxl+Jsw7dzAqKT6g==} + engines: {node: '>=16'} + json-schema-traverse@1.0.0: resolution: {integrity: sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==} @@ -6425,6 +6455,9 @@ packages: stackback@0.0.2: resolution: {integrity: sha512-1XMJE5fQo1jGH6Y/7ebnwPOBEkIEnT4QF32d5R1+VXdXveM0IBMJt8zfaxX1P3QhVwrYe+576+jkANtSS2mBbw==} + standardwebhooks@1.1.1: + resolution: {integrity: sha512-bCbX9ZEyFkWPsRz7Bl3NuQUJohmwGSev/yhr7vhaGPlc4AfIrspIRa6cPTBuI1ItmrTDJ4d/S2hCsfe4+vQGnQ==} + stat-mode@1.0.0: resolution: {integrity: sha512-jH9EhtKIjuXZ2cWxmXS8ZP80XyC3iasQxMDV8jzhNJpfDb7VbQLVW4Wvsxz9QZvzV+G4YoSfBUVKDOyxLzi/sg==} engines: {node: '>= 6'} @@ -6608,6 +6641,9 @@ packages: truncate-utf8-bytes@1.0.2: resolution: {integrity: sha512-95Pu1QXQvruGEhv62XCMO3Mm90GscOCClvrIUwCM0PYOXK3kaF3l3sIHxx71ThJfcbM2O5Au6SO3AWCSEfW4mQ==} + ts-algebra@2.0.0: + resolution: {integrity: sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw==} + ts-dedent@2.2.0: resolution: {integrity: sha512-q5W7tVM71e2xjHZTlgfTDoPF/SmqKG5hddq9SzR49CH2hayqRKJtQ4mtRlSxKaJlR/+9rEM+mnBHf7I2/BQcpQ==} engines: {node: '>=6.10'} @@ -7007,6 +7043,16 @@ packages: zwitch@2.0.4: resolution: {integrity: sha512-bXE4cR/kVZhKZX/RjPEflHaKVhUVl85noU3v6b8apfQEc1x4A+zBxjZ4lN8LqGd6WZ3dl98pY4o717VFmoPp+A==} +ignoredOptionalDependencies: + - '@anthropic-ai/claude-agent-sdk-darwin-arm64' + - '@anthropic-ai/claude-agent-sdk-darwin-x64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64-musl' + - '@anthropic-ai/claude-agent-sdk-linux-x64' + - '@anthropic-ai/claude-agent-sdk-linux-x64-musl' + - '@anthropic-ai/claude-agent-sdk-win32-arm64' + - '@anthropic-ai/claude-agent-sdk-win32-x64' + snapshots: '@adobe/css-tools@4.5.0': {} @@ -7016,6 +7062,19 @@ snapshots: package-manager-detector: 1.6.0 tinyexec: 1.1.2 + '@anthropic-ai/claude-agent-sdk@0.3.251(@anthropic-ai/sdk@0.122.0(zod@4.5.4))(@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4))(zod@4.5.4)': + dependencies: + '@anthropic-ai/sdk': 0.122.0(zod@4.5.4) + '@modelcontextprotocol/sdk': 1.30.0(supports-color@7.2.0)(zod@4.5.4) + zod: 4.5.4 + + '@anthropic-ai/sdk@0.122.0(zod@4.5.4)': + dependencies: + json-schema-to-ts: 3.1.1 + standardwebhooks: 1.1.1 + optionalDependencies: + zod: 4.5.4 + '@babel/code-frame@7.29.7': dependencies: '@babel/helper-validator-identifier': 7.29.7 @@ -7669,6 +7728,28 @@ snapshots: dependencies: '@chevrotain/types': 11.1.2 + '@modelcontextprotocol/sdk@1.30.0(supports-color@7.2.0)(zod@4.5.4)': + dependencies: + '@hono/node-server': 2.1.0(hono@4.13.0) + ajv: 8.20.0 + ajv-formats: 3.0.1(ajv@8.20.0) + content-type: 1.0.5 + cors: 2.8.6 + cross-spawn: 7.0.6 + eventsource: 3.0.7 + eventsource-parser: 3.0.8 + express: 5.2.1(supports-color@7.2.0) + express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0)) + hono: 4.13.0 + jose: 6.2.3 + json-schema-typed: 8.0.2 + pkce-challenge: 5.0.1 + raw-body: 3.0.2 + zod: 4.5.4 + zod-to-json-schema: 3.25.2(zod@4.5.4) + transitivePeerDependencies: + - supports-color + '@modelcontextprotocol/sdk@1.30.0(zod@3.25.76)': dependencies: '@hono/node-server': 2.1.0(hono@4.13.0) @@ -7679,8 +7760,8 @@ snapshots: cross-spawn: 7.0.6 eventsource: 3.0.7 eventsource-parser: 3.0.8 - express: 5.2.1 - express-rate-limit: 8.5.2(express@5.2.1) + express: 5.2.1(supports-color@7.2.0) + express-rate-limit: 8.5.2(express@5.2.1(supports-color@7.2.0)) hono: 4.13.0 jose: 6.2.3 json-schema-typed: 8.0.2 @@ -8917,6 +8998,8 @@ snapshots: '@sindresorhus/merge-streams@4.0.0': {} + '@stablelib/base64@1.0.1': {} + '@stablyai/playwright-base@2.1.14(@playwright/test@1.59.1)(zod@4.5.4)': dependencies: '@playwright/test': 1.59.1 @@ -9923,7 +10006,7 @@ snapshots: bluebird@3.7.2: {} - body-parser@2.3.0: + body-parser@2.3.0(supports-color@7.2.0): dependencies: bytes: 3.1.2 content-type: 2.0.0 @@ -10779,15 +10862,15 @@ snapshots: exponential-backoff@3.1.3: {} - express-rate-limit@8.5.2(express@5.2.1): + express-rate-limit@8.5.2(express@5.2.1(supports-color@7.2.0)): dependencies: - express: 5.2.1 + express: 5.2.1(supports-color@7.2.0) ip-address: 10.4.0 - express@5.2.1: + express@5.2.1(supports-color@7.2.0): dependencies: accepts: 2.0.0 - body-parser: 2.3.0 + body-parser: 2.3.0(supports-color@7.2.0) content-disposition: 1.1.0 content-type: 1.0.5 cookie: 0.7.2 @@ -10797,7 +10880,7 @@ snapshots: encodeurl: 2.0.0 escape-html: 1.0.3 etag: 1.8.1 - finalhandler: 2.1.1 + finalhandler: 2.1.1(supports-color@7.2.0) fresh: 2.0.0 http-errors: 2.0.1 merge-descriptors: 2.0.0 @@ -10808,9 +10891,9 @@ snapshots: proxy-addr: 2.0.7 qs: 6.15.2 range-parser: 1.2.1 - router: 2.2.0 - send: 1.2.1 - serve-static: 2.2.1 + router: 2.2.0(supports-color@7.2.0) + send: 1.2.1(supports-color@7.2.0) + serve-static: 2.2.1(supports-color@7.2.0) statuses: 2.0.2 type-is: 2.1.0 vary: 1.1.2 @@ -10831,6 +10914,8 @@ snapshots: merge2: 1.4.1 micromatch: 4.0.8 + fast-sha256@1.3.0: {} + fast-string-truncated-width@3.0.3: {} fast-string-width@3.0.2: @@ -10871,7 +10956,7 @@ snapshots: dependencies: to-regex-range: 5.0.1 - finalhandler@2.1.1: + finalhandler@2.1.1(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) encodeurl: 2.0.0 @@ -11445,6 +11530,11 @@ snapshots: json-parse-even-better-errors@2.3.1: {} + json-schema-to-ts@3.1.1: + dependencies: + '@babel/runtime': 7.29.7 + ts-algebra: 2.0.0 + json-schema-traverse@1.0.0: {} json-schema-typed@8.0.2: {} @@ -13037,7 +13127,7 @@ snapshots: points-on-curve: 0.2.0 points-on-path: 0.2.1 - router@2.2.0: + router@2.2.0(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) depd: 2.0.0 @@ -13080,7 +13170,7 @@ snapshots: semver@7.8.1: {} - send@1.2.1: + send@1.2.1(supports-color@7.2.0): dependencies: debug: 4.4.3(supports-color@7.2.0) encodeurl: 2.0.0 @@ -13107,12 +13197,12 @@ snapshots: transitivePeerDependencies: - typescript - serve-static@2.2.1: + serve-static@2.2.1(supports-color@7.2.0): dependencies: encodeurl: 2.0.0 escape-html: 1.0.3 parseurl: 1.3.3 - send: 1.2.1 + send: 1.2.1(supports-color@7.2.0) transitivePeerDependencies: - supports-color @@ -13267,6 +13357,11 @@ snapshots: stackback@0.0.2: {} + standardwebhooks@1.1.1: + dependencies: + '@stablelib/base64': 1.0.1 + fast-sha256: 1.3.0 + stat-mode@1.0.0: {} state-local@1.0.7: {} @@ -13437,6 +13532,8 @@ snapshots: dependencies: utf8-byte-length: 1.0.5 + ts-algebra@2.0.0: {} + ts-dedent@2.2.0: {} ts-morph@26.0.0: @@ -13786,6 +13883,10 @@ snapshots: dependencies: zod: 3.25.76 + zod-to-json-schema@3.25.2(zod@4.5.4): + dependencies: + zod: 4.5.4 + zod@3.25.76: {} zod@4.5.4: {} diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index d241459f884..97920f087c5 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -12,6 +12,20 @@ minimumReleaseAgeExclude: - zod@4.5.4 shamefullyHoist: true +# Orca always launches the user's own resolved Claude CLI via +# pathToClaudeCodeExecutable, so the SDK's bundled ~95 MB-per-platform CLI +# binaries must never be installed. Excluding them is what makes the path +# override mandatory rather than merely preferred. +ignoredOptionalDependencies: + - '@anthropic-ai/claude-agent-sdk-darwin-arm64' + - '@anthropic-ai/claude-agent-sdk-darwin-x64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64' + - '@anthropic-ai/claude-agent-sdk-linux-arm64-musl' + - '@anthropic-ai/claude-agent-sdk-linux-x64' + - '@anthropic-ai/claude-agent-sdk-linux-x64-musl' + - '@anthropic-ai/claude-agent-sdk-win32-arm64' + - '@anthropic-ai/claude-agent-sdk-win32-x64' + supportedArchitectures: os: - current diff --git a/src/main/claude-accounts/claude-structured-auth-policy.test.ts b/src/main/claude-accounts/claude-structured-auth-policy.test.ts new file mode 100644 index 00000000000..a2d7c7d5da8 --- /dev/null +++ b/src/main/claude-accounts/claude-structured-auth-policy.test.ts @@ -0,0 +1,168 @@ +import { describe, expect, it } from 'vitest' +import type { GlobalSettings } from '../../shared/global-settings-types' +import type { ClaudeManagedAccount } from '../../shared/managed-account-types' +import { + CLAUDE_AUTH_ENV_VARS, + hasClaudeAuthEnvConflict, + shouldStripClaudeAuthEnvForAccount +} from './environment' +import { + normalizeTuiAgentEnvRecord, + resolveTuiAgentLaunchEnv +} from '../../shared/tui-agent-launch-defaults' +import { claudeStructuredAuthPolicyForSettings } from './claude-structured-auth-policy' + +const HOST_ACCOUNT = { id: 'host-a', managedAuthRuntime: 'host' } as ClaudeManagedAccount +const WSL_ACCOUNT = { id: 'wsl-b', managedAuthRuntime: 'wsl' } as ClaudeManagedAccount +const LEGACY_ACCOUNT = { id: 'legacy-c' } as ClaudeManagedAccount + +function settings( + overrides: Partial< + Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' + > + > +): Parameters[0] { + return { + claudeManagedAccounts: [HOST_ACCOUNT, WSL_ACCOUNT, LEGACY_ACCOUNT], + activeClaudeManagedAccountId: null, + ...overrides + } as Parameters[0] +} + +// The predicate now backs BOTH transports (runtime-auth-preparation.ts and the +// structured wiring), so it needs a test of its own: forcing it to a constant used +// to leave ~1000 tests green. +describe('shouldStripClaudeAuthEnvForAccount', () => { + it('does not strip when no managed account is selected', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], null)).toBe(false) + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], undefined)).toBe(false) + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], '')).toBe(false) + }) + + it('strips for a host-managed account', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT, WSL_ACCOUNT], 'host-a')).toBe(true) + }) + + it('strips for an account with no explicit runtime (the legacy host shape)', () => { + expect(shouldStripClaudeAuthEnvForAccount([LEGACY_ACCOUNT], 'legacy-c')).toBe(true) + }) + + it('does not strip for a WSL-managed account, matching runtime-auth-preparation', () => { + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT, WSL_ACCOUNT], 'wsl-b')).toBe(false) + }) + + it('strips for a selected id no account list explains', () => { + // Fail-safe: an id we cannot resolve is treated as a pinned account, never as + // "no account", so an unreadable settings blob cannot open the strip. + expect(shouldStripClaudeAuthEnvForAccount([HOST_ACCOUNT], 'deleted-d')).toBe(true) + expect(shouldStripClaudeAuthEnvForAccount(undefined, 'deleted-d')).toBe(true) + expect(shouldStripClaudeAuthEnvForAccount([], 'deleted-d')).toBe(true) + }) +}) + +describe('claudeStructuredAuthPolicyForSettings', () => { + it('reads the host runtime selection, not the legacy flat field alone', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ + activeClaudeManagedAccountId: 'host-a', + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } + }) + ) + ).toEqual({ stripAuthEnv: true }) + }) + + it('strips when a host account is pinned by runtime selection', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ activeClaudeManagedAccountIdsByRuntime: { host: 'host-a', wsl: {} } }) + ) + ).toEqual({ stripAuthEnv: true }) + }) + + it('does not strip for system auth, so an API-key-only user keeps their sign-in', () => { + expect(claudeStructuredAuthPolicyForSettings(settings({}))).toEqual({ stripAuthEnv: false }) + }) + + it('ignores a WSL-only selection: the structured child is always a native host process', () => { + expect( + claudeStructuredAuthPolicyForSettings( + settings({ + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-b' } } + }) + ) + ).toEqual({ stripAuthEnv: false }) + }) +}) + +describe('the strip vocabulary the policy governs', () => { + it('covers every Anthropic auth variable the terminal path knows about', () => { + // A new auth var added to the list without a matching refusal/strip path is the + // shape of the leak this lane already shipped once. + expect([...CLAUDE_AUTH_ENV_VARS]).toEqual([ + 'ANTHROPIC_API_KEY', + 'ANTHROPIC_AUTH_TOKEN', + 'CLAUDE_CODE_OAUTH_TOKEN', + 'AWS_BEARER_TOKEN_BEDROCK' + ]) + }) +}) + +// The refusal has to cover exactly what the strip removes. Anything narrower lets an +// override reach the child that applyClaudeEnvPatch would have deleted. +describe('hasClaudeAuthEnvConflict matches the strip it guards', () => { + it('refuses each Anthropic auth variable', () => { + for (const key of CLAUDE_AUTH_ENV_VARS) { + expect(hasClaudeAuthEnvConflict({ [key]: 'v' }, 'linux')).toBe(true) + } + }) + + // `ANTHROPIC_API_KEY=` in the agent env box is how a user blanks a variable, and the + // settings pipeline preserves the empty value (agent-default-env-draft.ts assigns + // everything after the `=`; normalizeTuiAgentEnvRecord drops empty KEYS only). An + // empty value cannot beat the pinned account and the strip removes the name anyway, + // so refusing it would break a terminal launch that works today for no security gain. + it('admits an override whose value is empty, the documented way to blank a variable', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_API_KEY: '' }, 'linux')).toBe(false) + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: '' }, 'win32')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_CUSTOM_HEADERS: '' }, 'linux')).toBe(false) + }) + + it('still refuses the same names once they carry a value', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_API_KEY: 'sk-ant' }, 'linux')).toBe(true) + }) + + // The end-to-end shape the regression actually took: settings text -> normalized + // record -> launch env -> the predicate the terminal preflight gates on. + it('admits a blanked variable all the way from the settings record', () => { + const configured = normalizeTuiAgentEnvRecord({ claude: { ANTHROPIC_API_KEY: '' } }) + const launchEnv = resolveTuiAgentLaunchEnv('claude', configured) + + expect(launchEnv).toEqual({ ANTHROPIC_API_KEY: '' }) + expect(hasClaudeAuthEnvConflict(launchEnv, 'linux')).toBe(false) + }) + + it('folds case on win32, where the OS does', () => { + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: 'sk-lower' }, 'win32')).toBe(true) + expect(hasClaudeAuthEnvConflict({ Anthropic_Custom_Headers: 'x-api-key: v' }, 'win32')).toBe( + true + ) + }) + + it('keeps env names case-sensitive off win32', () => { + expect(hasClaudeAuthEnvConflict({ anthropic_api_key: 'sk-lower' }, 'linux')).toBe(false) + }) + + it('admits non-auth Anthropic settings on both platforms', () => { + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_BASE_URL: 'https://gw.test' }, 'linux')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_BASE_URL: 'https://gw.test' }, 'win32')).toBe(false) + expect(hasClaudeAuthEnvConflict({ ANTHROPIC_CUSTOM_HEADERS: 'X-Trace: 1' }, 'linux')).toBe( + false + ) + expect(hasClaudeAuthEnvConflict(undefined, 'linux')).toBe(false) + }) +}) diff --git a/src/main/claude-accounts/claude-structured-auth-policy.ts b/src/main/claude-accounts/claude-structured-auth-policy.ts new file mode 100644 index 00000000000..c30cd69b827 --- /dev/null +++ b/src/main/claude-accounts/claude-structured-auth-policy.ts @@ -0,0 +1,37 @@ +import type { GlobalSettings } from '../../shared/global-settings-types' +import { shouldStripClaudeAuthEnvForAccount } from './environment' +import { getSelectedClaudeAccountIdForTarget } from './runtime-selection' + +/** The structured mirror of the terminal preflight's `prepareClaudeAuth` result: + * the one field a launch resolution needs from the managed-account state. */ +export type ClaudeStructuredAuthPolicy = { + stripAuthEnv: boolean +} + +/** + * The only supported way to build a structured launch's auth policy. + * + * It exists as a named function rather than an inline object at the wiring site so + * that the settings-to-policy mapping is testable on its own: the one production + * wiring lives in a `@ts-nocheck` file, where neither the compiler nor a type test + * can see a dropped field. + * + * Structured Claude always spawns a native local-host child — the launch resolver + * refuses any record with a remote execution host or a WSL distro — so the host + * selection, not the platform default target, owns its auth. + */ +export function claudeStructuredAuthPolicyForSettings( + settings: Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' + > +): ClaudeStructuredAuthPolicy { + return { + stripAuthEnv: shouldStripClaudeAuthEnvForAccount( + settings.claudeManagedAccounts, + getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }) + ) + } +} diff --git a/src/main/claude-accounts/environment.ts b/src/main/claude-accounts/environment.ts index 83fe3b40209..b85dd60a854 100644 --- a/src/main/claude-accounts/environment.ts +++ b/src/main/claude-accounts/environment.ts @@ -1,3 +1,5 @@ +import type { ClaudeManagedAccount } from '../../shared/managed-account-types' + export const CLAUDE_AUTH_ENV_VARS = [ 'ANTHROPIC_API_KEY', 'ANTHROPIC_AUTH_TOKEN', @@ -13,14 +15,21 @@ export type ClaudeEnvPatch = { export function applyClaudeEnvPatch( baseEnv: Record, patch: ClaudeEnvPatch, - options?: { stripAuthEnv?: boolean } + options?: { stripAuthEnv?: boolean; platform?: NodeJS.Platform } ): Record { if (options?.stripAuthEnv) { for (const key of CLAUDE_AUTH_ENV_VARS) { delete baseEnv[key] } - if (isAuthLikeCustomHeaders(baseEnv.ANTHROPIC_CUSTOM_HEADERS)) { - delete baseEnv.ANTHROPIC_CUSTOM_HEADERS + const platform = options.platform ?? process.platform + for (const key of Object.keys(baseEnv)) { + const normalized = platform === 'win32' ? key.toUpperCase() : key + if ( + (platform === 'win32' && CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized)) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(baseEnv[key])) + ) { + delete baseEnv[key] + } } } @@ -34,16 +43,94 @@ export function applyClaudeEnvPatch( return baseEnv } -export function hasClaudeAuthEnvConflict(env: Record | undefined): boolean { - if (!env) { +/** One string for every transport, so a terminal launch and a structured launch + * cannot drift into telling the user two different things about one refusal. */ +export const CLAUDE_AUTH_ENV_CONFLICT_MESSAGE = + 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' + +export const CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE = + 'A Claude account switch is in progress. Try again after it finishes.' + +/** + * Whether a launch on the host runtime must drop inherited Anthropic auth. + * + * Only a pinned host-managed account owns the credential, so only it may strip: + * with no managed account the user's own `ANTHROPIC_*` is their sign-in, and + * removing it signs them out of a CLI that would otherwise have worked. + */ +export function shouldStripClaudeAuthEnvForAccount( + accounts: readonly ClaudeManagedAccount[] | undefined, + activeAccountId: string | null | undefined +): boolean { + if (!activeAccountId) { return false } return ( - CLAUDE_AUTH_ENV_VARS.some((key) => Boolean(env[key])) || - isAuthLikeCustomHeaders(env.ANTHROPIC_CUSTOM_HEADERS) + (accounts ?? []).find((account) => account.id === activeAccountId)?.managedAuthRuntime !== 'wsl' ) } +/** + * Whether a launch's explicit env carries Anthropic auth a managed account must own. + * + * The key comparison mirrors applyClaudeEnvPatch's strip exactly: case-insensitive on + * win32, where the OS folds env names so `anthropic_api_key` is an effective + * `ANTHROPIC_API_KEY`, and case-sensitive elsewhere. A refusal narrower than the strip + * lets an override through that the strip would have removed. + * + * A non-empty value is what makes it a conflict. `ANTHROPIC_API_KEY=` in the agent env + * box is how a user blanks a variable — the settings pipeline preserves that empty value + * (normalizeTuiAgentEnvRecord drops empty KEYS only) — and an empty override can neither + * authenticate nor beat the pinned account, while the strip removes the name regardless. + * Refusing it would break a terminal launch that works today for no security gain. + */ +/** + * The inherited Anthropic auth a non-stripping launch has to carry forward explicitly. + * + * applyClaudeEnvPatch always strips the inherited half of a child env, and the + * configured half is what overrides it — so a system-auth user's own key only survives + * if the caller puts it back deliberately. Returns the exact keys present, so a + * win32 `anthropic_api_key` is carried under the name the OS actually has. + */ +export function claudeAuthEnvCarriedForward( + inherited: NodeJS.ProcessEnv, + platform: NodeJS.Platform = process.platform +): Record { + const carried: Record = {} + for (const [key, value] of Object.entries(inherited)) { + if (value === undefined) { + continue + } + const normalized = platform === 'win32' ? key.toUpperCase() : key + if ( + CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(value)) + ) { + carried[key] = value + } + } + return carried +} + +export function hasClaudeAuthEnvConflict( + env: Record | undefined, + platform: NodeJS.Platform = process.platform +): boolean { + if (!env) { + return false + } + for (const [key, value] of Object.entries(env)) { + const normalized = platform === 'win32' ? key.toUpperCase() : key + if (value && CLAUDE_AUTH_ENV_VARS.some((authKey) => authKey === normalized)) { + return true + } + if (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && isAuthLikeCustomHeaders(value)) { + return true + } + } + return false +} + function isAuthLikeCustomHeaders(value: string | undefined): boolean { if (!value) { return false diff --git a/src/main/claude-accounts/live-pty-gate.ts b/src/main/claude-accounts/live-pty-gate.ts index 9e30b621924..caab66c3430 100644 --- a/src/main/claude-accounts/live-pty-gate.ts +++ b/src/main/claude-accounts/live-pty-gate.ts @@ -5,6 +5,13 @@ const liveClaudePtyIds = new Set() // survived the app restart inside the daemon. const seededUnconfirmedPtyIds = new Set() let switchInProgress = false +// Woken by endClaudeAuthSwitch so a caller past the point of no return can wait the +// swap out instead of refusing. See whenClaudeAuthSwitchSettles. +const switchSettledListeners = new Set<() => void>() + +/** A managed account swap is a credential-file rewrite, not a network round trip; + * anything past this is a wedged switch, and refusing beats waiting forever. */ +export const CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS = 15_000 export type ClaudeLivePtyPersistence = { addClaudeLivePtySessionId(sessionId: string): void @@ -81,6 +88,35 @@ export function markClaudePtyExited(ptyId: string): void { notifyDrainedOnTransition(hadLivePtys) } +/** + * Register a structured Claude child with the same gate the terminal path uses. + * + * The gate is what makes the managed OAuth refresh defer instead of rotating a + * single-use refresh token out from under a running Claude (runtime-auth-sync.ts). + * A structured session's child is as much a live Claude as a PTY's is, so it has to + * hold the gate too — otherwise a refresh mid-turn breaks its next API call while an + * identical terminal session is protected. + * + * Deliberately not persisted, unlike markClaudePtySpawned: these children are direct + * children of this process and cannot survive a restart, so seeding them back on the + * next launch would hold the gate closed for a process that is provably gone. + */ +export function markClaudeStructuredChildSpawned(childKey: string): void { + liveClaudePtyIds.add(structuredChildGateId(childKey)) +} + +export function markClaudeStructuredChildExited(childKey: string): void { + const hadLivePtys = liveClaudePtyIds.size > 0 + liveClaudePtyIds.delete(structuredChildGateId(childKey)) + notifyDrainedOnTransition(hadLivePtys) +} + +// Namespaced so a structured child can never collide with a daemon PTY session id, +// which confirmSeededClaudeLivePtys reconciles against the daemon's own list. +function structuredChildGateId(childKey: string): string { + return `claude-structured:${childKey}` +} + export function hasLiveClaudePtys(): boolean { return liveClaudePtyIds.size > 0 } @@ -93,7 +129,44 @@ export function beginClaudeAuthSwitch(): void { } export function endClaudeAuthSwitch(): void { + const wasInProgress = switchInProgress switchInProgress = false + if (!wasInProgress) { + return + } + // Each listener removes itself as it settles; Set iteration is defined over that. + for (const listener of switchSettledListeners) { + listener() + } +} + +/** + * Resolves `true` once no account switch is running, `false` if one is still running + * at the deadline. + * + * Exists for callers that have already done irreversible work — a structured acquire + * has closed the old child by the time it resolves its launch, so turning a switch + * into a refusal there strands the user with a dead session and no replacement. + * Waiting for the swap and then launching against it is the recoverable answer; + * refusing is only correct when nothing has been torn down yet. + */ +export function whenClaudeAuthSwitchSettles( + timeoutMs = CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS +): Promise { + if (!switchInProgress) { + return Promise.resolve(true) + } + return new Promise((resolve) => { + const settle = (settled: boolean): void => { + switchSettledListeners.delete(listener) + clearTimeout(timer) + resolve(settled) + } + const listener = (): void => settle(true) + switchSettledListeners.add(listener) + const timer = setTimeout(() => settle(false), timeoutMs) + timer.unref?.() + }) } export function isClaudeAuthSwitchInProgress(): boolean { diff --git a/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts b/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts index ae79c4c7bbb..dabcd9d472f 100644 --- a/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts +++ b/src/main/claude-accounts/runtime-auth/runtime-auth-preparation.ts @@ -2,6 +2,7 @@ import { join } from 'node:path' import type { ClaudeManagedAccount } from '../../../shared/managed-account-types' import { resolveLocalAccountRuntimeTarget } from '../../../shared/local-account-runtime' import { parseWslUncPath } from '../../../shared/wsl-paths' +import { shouldStripClaudeAuthEnvForAccount } from '../environment' import { getDefaultWslDistro, getWslHome } from '../../wsl' import { getSelectedClaudeAccountIdForTarget, @@ -69,7 +70,10 @@ export class ClaudeRuntimeAuthPreparationService extends ClaudeRuntimeAuthSnapsh wslDistro: null, wslLinuxConfigDir: null, envPatch: paths.envPatch, - stripAuthEnv: Boolean(activeAccountId && activeAccount?.managedAuthRuntime !== 'wsl'), + stripAuthEnv: shouldStripClaudeAuthEnvForAccount( + settings.claudeManagedAccounts, + activeAccountId + ), managedRefreshDeferredByLivePty: Boolean( activeAccountId && activeAccount?.managedAuthRuntime !== 'wsl' && diff --git a/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs b/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs new file mode 100644 index 00000000000..4f2a09425fd --- /dev/null +++ b/src/main/claude/__fixtures__/claude-agent-sdk-scripted-cli.mjs @@ -0,0 +1,144 @@ +// Scripted stand-in for the Claude Code CLI, driven by the SDK contract-pin +// tests. It speaks just enough stream-json to satisfy the SDK: it answers every +// inbound control_request with a success control_response, records everything it +// observes to a report file, and plays back the steps listed in a scenario file. +// +// Env contract (set by the test): +// ORCA_SDK_CONTRACT_SCENARIO_PATH — JSON file +// { steps: Step[], controlResponses?: { [subtype]: } } where a Step is +// { emit: } | { awaitUserMessage: true } | { stderr: } | +// { awaitControlResponse: } | { delayMs: } | { exit: } +// ORCA_SDK_CONTRACT_REPORT_PATH — where argv/env observations are written +// ORCA_SDK_CONTRACT_IGNORE_SIGTERM — trap SIGTERM/SIGINT and outlive stdin close +// ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS — record control requests but never answer +// ORCA_SDK_CONTRACT_DESCENDANT — fork an idle grandchild and report its pid +import { spawn } from 'node:child_process' +import { readFileSync, writeFileSync } from 'node:fs' +import { createInterface } from 'node:readline' + +const scenarioPath = process.env.ORCA_SDK_CONTRACT_SCENARIO_PATH +const reportPath = process.env.ORCA_SDK_CONTRACT_REPORT_PATH + +const report = { + argv: process.argv.slice(1), + execPath: process.execPath, + controlRequests: [], + controlResponses: [], + userMessages: [], + descendantPid: null +} +const writeReport = () => { + if (reportPath) { + writeFileSync(reportPath, JSON.stringify(report)) + } +} +// Written immediately so a test can prove which script the SDK executed even if +// the session dies before the scenario completes. +writeReport() + +const scenario = scenarioPath ? JSON.parse(readFileSync(scenarioPath, 'utf8')) : { steps: [] } + +if (process.env.ORCA_SDK_CONTRACT_IGNORE_SIGTERM) { + process.on('SIGTERM', () => {}) + process.on('SIGINT', () => {}) + setInterval(() => {}, 1_000_000) +} +if (process.env.ORCA_SDK_CONTRACT_DESCENDANT) { + const descendant = spawn(process.execPath, ['-e', 'setInterval(() => {}, 1000000)'], { + stdio: 'ignore' + }) + descendant.unref() + report.descendantPid = descendant.pid ?? null + writeReport() +} + +const emit = (frame) => process.stdout.write(`${JSON.stringify(frame)}\n`) + +const waiters = [] +const settle = (kind, requestId) => { + for (let i = waiters.length - 1; i >= 0; i--) { + const waiter = waiters[i] + if ( + waiter.kind === kind && + (waiter.requestId === undefined || waiter.requestId === requestId) + ) { + waiters.splice(i, 1) + waiter.resolve() + } + } +} +const waitFor = (kind, requestId) => { + if (kind === 'user' && report.userMessages.length > 0) { + return Promise.resolve() + } + if ( + kind === 'control_response' && + report.controlResponses.some((frame) => frame.response?.request_id === requestId) + ) { + return Promise.resolve() + } + return new Promise((resolve) => waiters.push({ kind, requestId, resolve })) +} + +createInterface({ input: process.stdin }).on('line', (line) => { + let frame + try { + frame = JSON.parse(line) + } catch { + return + } + if (frame.type === 'control_request') { + report.controlRequests.push(frame) + writeReport() + if (process.env.ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS) { + return + } + emit({ + type: 'control_response', + response: { + subtype: 'success', + request_id: frame.request_id, + response: scenario.controlResponses?.[frame.request?.subtype] ?? { + commands: [], + models: [] + } + } + }) + return + } + if (frame.type === 'control_response') { + report.controlResponses.push(frame) + writeReport() + settle('control_response', frame.response?.request_id) + return + } + if (frame.type === 'user') { + report.userMessages.push(frame) + writeReport() + settle('user') + } +}) + +// Never outlive a wedged test: the readline subscription would otherwise hold +// this process open forever if the SDK side stops driving the scenario. +setTimeout(() => process.exit(3), 20_000).unref() + +for (const step of scenario.steps) { + if (step.emit) { + emit(step.emit) + } else if (step.stderr !== undefined) { + process.stderr.write(step.stderr) + } else if (step.awaitUserMessage) { + await waitFor('user') + } else if (step.awaitControlResponse !== undefined) { + await waitFor('control_response', step.awaitControlResponse) + } else if (step.delayMs) { + await new Promise((resolve) => setTimeout(resolve, step.delayMs)) + } else if (step.exit !== undefined) { + // A CLI that refuses to start: leave with its own status, stderr already written. + writeReport() + process.exit(step.exit) + } +} +writeReport() +process.exit(0) diff --git a/src/main/claude/claude-agent-sdk-contract-pins.test.ts b/src/main/claude/claude-agent-sdk-contract-pins.test.ts new file mode 100644 index 00000000000..46bcb7219d3 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-contract-pins.test.ts @@ -0,0 +1,519 @@ +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { createRequire } from 'node:module' +import { tmpdir } from 'node:os' +import { dirname, join } from 'node:path' +import { + query, + type CanUseTool, + type Options, + type SDKUserMessage, + type SpawnedProcess as SdkSpawnedProcess, + type SpawnOptions as SdkSpawnOptions +} from '@anthropic-ai/claude-agent-sdk' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { claudeQuerySettingsReader } from './claude-agent-sdk-control-requests' +import { createClaudeStructuredLaunchResolver } from './claude-structured-launch-resolution' + +// Contract pins for @anthropic-ai/claude-agent-sdk, run against the real SDK +// driving a scripted fake CLI (never the real Claude binary). These tests exist +// to catch a future SDK version drifting under Orca: unknown-frame pass-through, +// spawner env fidelity, argument parity with the pre-SDK argv, +// permission-callback semantics, and executable-path override. + +const FAKE_CLI = join(__dirname, '__fixtures__', 'claude-agent-sdk-scripted-cli.mjs') +const SESSION_ID = '5348c19f-6a54-4c2e-9c68-9c2b1a3d4e5f' +const LEAF_UUID = 'ad0f7c9e-1b2c-4d3e-8f90-abc123def456' +const PINNED_SDK_VERSION = '0.3.251' +const SDK_PLATFORM_PACKAGE_BASENAMES = [ + 'claude-agent-sdk-darwin-arm64', + 'claude-agent-sdk-darwin-x64', + 'claude-agent-sdk-linux-arm64', + 'claude-agent-sdk-linux-arm64-musl', + 'claude-agent-sdk-linux-x64', + 'claude-agent-sdk-linux-x64-musl', + 'claude-agent-sdk-win32-arm64', + 'claude-agent-sdk-win32-x64' +] + +/** + * The exact argv the hand-rolled transport built before the SDK swap. Frozen here + * as the parity oracle: CLAUDE_STRUCTURED_BASE_OPTIONS has to keep producing it. + */ +const PRE_SDK_ARGV = [ + '-p', + '--input-format', + 'stream-json', + '--output-format', + 'stream-json', + '--include-partial-messages', + '--verbose', + '--replay-user-messages', + '--permission-prompt-tool', + 'stdio', + '--setting-sources', + 'user,project,local' +] + +const RESULT_FRAME = { + type: 'result', + subtype: 'success', + is_error: false, + duration_ms: 1, + duration_api_ms: 1, + num_turns: 1, + result: 'ok', + session_id: SESSION_ID, + total_cost_usd: 0, + usage: { input_tokens: 1, output_tokens: 1 }, + uuid: 'uuid-result-1' +} + +type ScenarioStep = Record +type SpawnSeen = { + command: string + args: string[] + cwd: string | undefined + env: Record +} +type ScriptedCliReport = { + argv: string[] + execPath: string + controlRequests: { request_id: string; request: { subtype: string } }[] + controlResponses: { response: { request_id: string; response?: Record } }[] + userMessages: Record[] +} + +const scratchDirs: string[] = [] +afterEach(() => { + vi.unstubAllEnvs() + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } +}) + +function scriptScenario( + steps: ScenarioStep[], + controlResponses: Record = {} +): { + scenarioPath: string + reportPath: string + cwd: string + readReport: () => ScriptedCliReport +} { + const dir = mkdtempSync(join(tmpdir(), 'claude-sdk-contract-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + const reportPath = join(dir, 'report.json') + writeFileSync(scenarioPath, JSON.stringify({ steps, controlResponses })) + return { + scenarioPath, + reportPath, + cwd: dir, + readReport: () => JSON.parse(readFileSync(reportPath, 'utf8')) as ScriptedCliReport + } +} + +function scenarioEnv(scenario: { scenarioPath: string; reportPath: string }) { + return { + PATH: process.env.PATH, + ORCA_SDK_CONTRACT_SCENARIO_PATH: scenario.scenarioPath, + ORCA_SDK_CONTRACT_REPORT_PATH: scenario.reportPath + } +} + +function recordingSpawner(spawns: SpawnSeen[]) { + return (opts: SdkSpawnOptions): SdkSpawnedProcess => { + spawns.push({ + command: opts.command, + args: [...opts.args], + cwd: opts.cwd, + env: { ...opts.env } + }) + return spawnProcess({ + program: opts.command, + args: opts.args, + cwd: opts.cwd, + env: opts.env as NodeJS.ProcessEnv, + signal: opts.signal + }) as unknown as SdkSpawnedProcess + } +} + +function resolvedLaunch(launchArgs: string[]) { + const record = { + sessionId: 'contract-pin-session', + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [], + launchArgs + } as unknown as AgentSessionRecord + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => FAKE_CLI, + resolveAuthPolicy: () => ({ stripAuthEnv: true }) + })({ identity: { sessionId: record.sessionId } as never }) +} + +function singleUserTurn(): AsyncIterable { + return (async function* () { + yield { + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + } as SDKUserMessage + // Hold input open; the stream ends when the scripted CLI exits, and an + // unresolved bare promise does not keep the event loop alive. + await new Promise(() => {}) + })() +} + +async function drainQuery(options: Options): Promise[]> { + const messages: Record[] = [] + for await (const message of query({ prompt: singleUserTurn(), options })) { + messages.push(message as unknown as Record) + } + return messages +} + +/** Expand `--flag=value` argv entries so both SDK spellings compare equal. */ +function normalizeArgv(args: string[]): string[] { + return args.flatMap((arg) => { + if (!arg.startsWith('--')) { + return [arg] + } + const eq = arg.indexOf('=') + return eq === -1 ? [arg] : [arg.slice(0, eq), arg.slice(eq + 1)] + }) +} + +/** Group the pre-SDK argv into flag/value pairs. */ +function flagTable(args: readonly string[]): { flag: string; value: string | null }[] { + const table: { flag: string; value: string | null }[] = [] + for (let i = 0; i < args.length; i++) { + const flag = args[i]! + const next = args[i + 1] + if (next !== undefined && !next.startsWith('-')) { + table.push({ flag, value: next }) + i++ + } else { + table.push({ flag, value: null }) + } + } + return table +} + +describe('Claude Agent SDK contract pins', () => { + it('yields unknown types, unknown fields and unknown content blocks verbatim, and consumes keep_alive', async () => { + const unknownTopLevel = { + type: 'message_kind_from_the_future', + session_id: SESSION_ID, + uuid: 'uuid-unknown-1', + payload: { alpha: 1, nested: { flags: ['a', 'b'] } } + } + const assistantWithUnknowns = { + type: 'assistant', + message: { + id: 'msg-1', + type: 'message', + role: 'assistant', + model: 'claude-x', + content: [ + { type: 'text', text: 'hello back' }, + { type: 'content_block_from_the_future', payload: { depth: 3 } } + ], + stop_reason: null, + stop_sequence: null, + usage: { input_tokens: 1, output_tokens: 2 } + }, + parent_tool_use_id: null, + uuid: 'uuid-assistant-1', + session_id: SESSION_ID, + field_from_the_future: 'preserved' + } + const scenario = scriptScenario([ + { awaitUserMessage: true }, + { emit: { type: 'keep_alive' } }, + { emit: unknownTopLevel }, + { emit: assistantWithUnknowns }, + { emit: RESULT_FRAME } + ]) + const spawns: SpawnSeen[] = [] + const messages = await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(messages.find((m) => m.uuid === 'uuid-unknown-1')).toEqual(unknownTopLevel) + expect(messages.find((m) => m.uuid === 'uuid-assistant-1')).toEqual(assistantWithUnknowns) + // The SDK intercepts keep_alive internally — a liveness signal must never + // be derived from it reaching the consumer, because it does not. + expect(messages.some((m) => m.type === 'keep_alive')).toBe(false) + expect(messages.some((m) => m.type === 'result')).toBe(true) + }) + + it('hands the custom spawner exactly the caller-supplied env, plus the two pinned SDK mutations', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'ambient-key-must-not-leak') + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: { + ...scenarioEnv(scenario), + CLAUDE_CONFIG_DIR: '/pinned/claude-config', + ORCA_AGENT_SESSION_SPAWN_TOKEN: 'spawn-token-1', + NODE_OPTIONS: '--max-old-space-size=64' + }, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + const env = spawns[0]!.env + // Supplied values arrive verbatim: the config-dir pin and spawn token are + // observable at this boundary, so Orca's auth scrubbing stays assertable. + expect(env.CLAUDE_CONFIG_DIR).toBe('/pinned/claude-config') + expect(env.ORCA_AGENT_SESSION_SPAWN_TOKEN).toBe('spawn-token-1') + // Ambient process.env is NOT merged in when env is supplied. + expect(env.ANTHROPIC_API_KEY).toBeUndefined() + // The SDK's two documented mutations, pinned so a change is noticed. + expect(env.CLAUDE_CODE_ENTRYPOINT).toBe('sdk-ts') + expect('NODE_OPTIONS' in env).toBe(false) + }) + + it('inherits process.env into the child when env is omitted — the ambient-auth sharp edge', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + vi.stubEnv('ORCA_SDK_CONTRACT_SCENARIO_PATH', scenario.scenarioPath) + vi.stubEnv('ORCA_SDK_CONTRACT_REPORT_PATH', scenario.reportPath) + vi.stubEnv('ORCA_SDK_CONTRACT_AMBIENT_CANARY', 'inherited-from-process-env') + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + // Omitting env reproduces the ambient-auth-leak failure mode: the child + // sees everything in process.env. Orca must therefore always pass an + // explicit, fully-constructed env. + expect(spawns[0]!.env.ORCA_SDK_CONTRACT_AMBIENT_CANARY).toBe('inherited-from-process-env') + }) + + it('emits --replay-user-messages only through extraArgs, never on its own', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const bareSpawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + spawnClaudeCodeProcess: recordingSpawner(bareSpawns) + }) + expect(bareSpawns[0]!.args).not.toContain('--replay-user-messages') + + const replayScenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const replaySpawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: replayScenario.cwd, + env: scenarioEnv(replayScenario), + extraArgs: { 'replay-user-messages': null }, + spawnClaudeCodeProcess: recordingSpawner(replaySpawns) + }) + const replayArgs = replaySpawns[0]!.args + expect(replayArgs.filter((arg) => arg === '--replay-user-messages')).toHaveLength(1) + }) + + it('produces a matching CLI flag for every pre-SDK argv entry', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + // Driven by the real resolver, so the argv walk covers the durable-launchArgs + // translation and its merge order, not a hand-written options literal. + const launch = await resolvedLaunch(['--model', 'claude-sonnet-4-5', '--effort', 'high']) + await drainQuery({ + ...launch.options, + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + canUseTool: (async () => ({ behavior: 'deny', message: 'unused' })) as CanUseTool, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(spawns).toHaveLength(1) + const argv = normalizeArgv(spawns[0]!.args) + // Typed-first translation must not also spell the flag through extraArgs. + for (const flag of ['--model', '--effort']) { + expect( + argv.filter((arg) => arg === flag), + `${flag} occurrences` + ).toHaveLength(1) + } + expect(argv[argv.indexOf('--model') + 1]).toBe('claude-sonnet-4-5') + expect(argv[argv.indexOf('--effort') + 1]).toBe('high') + // Headless print mode is the SDK's only mode; `query()` never passes `-p`, + // and if the SDK ever started passing it this pin would notice. + const impliedByHeadlessQuery = new Set(['-p']) + for (const entry of flagTable(PRE_SDK_ARGV)) { + if (impliedByHeadlessQuery.has(entry.flag)) { + expect(argv, `${entry.flag} is implied, never spelled`).not.toContain(entry.flag) + continue + } + const at = argv.indexOf(entry.flag) + expect(at, `SDK argv is missing ${entry.flag}`).toBeGreaterThanOrEqual(0) + if (entry.value !== null) { + expect(argv[at + 1], `value of ${entry.flag}`).toBe(entry.value) + } + } + // The launch resolver always carries one of --session-id / --resume. + const sessionAt = argv.indexOf('--session-id') + expect(sessionAt).toBeGreaterThanOrEqual(0) + expect(argv[sessionAt + 1]).toBe(launch.providerSessionId) + }) + + it('still exposes the runtime get_settings reader the auth diagnostic depends on', async () => { + // 0.3.251 ships getSettings() but redacts it from the Query declaration. This pin + // is the drift alarm: if a bump drops or reshapes it, the diagnostic degrades and + // this test says so instead of the degradation shipping silently. + const settings = { env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } } + const scenario = scriptScenario([{ delayMs: 3_000 }], { get_settings: settings }) + const session = query({ + prompt: singleUserTurn(), + options: { + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario) + } + }) + try { + const read = claudeQuerySettingsReader(session) + expect(read, 'the SDK no longer exposes get_settings at runtime').not.toBeNull() + await expect(read?.()).resolves.toEqual(settings) + } finally { + await session.return(undefined) + } + }) + + it('maps resume identity to --resume and --resume-session-at', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + resume: SESSION_ID, + resumeSessionAt: LEAF_UUID, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + const argv = normalizeArgv(spawns[0]!.args) + const resumeAt = argv.indexOf('--resume') + expect(resumeAt).toBeGreaterThanOrEqual(0) + expect(argv[resumeAt + 1]).toBe(SESSION_ID) + const leafAt = argv.indexOf('--resume-session-at') + expect(leafAt).toBeGreaterThanOrEqual(0) + expect(argv[leafAt + 1]).toBe(LEAF_UUID) + }) + + it('gives canUseTool the wire request_id and fires its abort signal on control_cancel_request', async () => { + const scenario = scriptScenario([ + { awaitUserMessage: true }, + { + emit: { + type: 'control_request', + request_id: 'perm-421', + request: { + subtype: 'can_use_tool', + tool_name: 'Bash', + input: { command: 'echo hi' }, + tool_use_id: 'tool-use-9' + } + } + }, + { delayMs: 120 }, + { emit: { type: 'control_cancel_request', request_id: 'perm-421' } }, + { awaitControlResponse: 'perm-421' }, + { emit: RESULT_FRAME } + ]) + const seen: { toolName: string; requestId: string; toolUseID: string }[] = [] + let abortFired = false + const canUseTool: CanUseTool = (toolName, _input, { signal, requestId, toolUseID }) => { + seen.push({ toolName, requestId, toolUseID }) + return new Promise((resolve) => { + signal.addEventListener('abort', () => { + abortFired = true + resolve({ behavior: 'deny', message: 'cancelled by test' }) + }) + }) + } + const spawns: SpawnSeen[] = [] + await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario), + canUseTool, + spawnClaudeCodeProcess: recordingSpawner(spawns) + }) + + expect(seen).toEqual([{ toolName: 'Bash', requestId: 'perm-421', toolUseID: 'tool-use-9' }]) + expect(abortFired).toBe(true) + // The callback's settlement is written back onto the wire against the same id. + const settled = scenario + .readReport() + .controlResponses.find((frame) => frame.response.request_id === 'perm-421') + expect(settled?.response.response?.behavior).toBe('deny') + // Exactly one process spawn per query, control traffic included. + expect(spawns).toHaveLength(1) + }) + + it('runs the executable given via pathToClaudeCodeExecutable under the default spawner', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: RESULT_FRAME }]) + const messages = await drainQuery({ + pathToClaudeCodeExecutable: FAKE_CLI, + cwd: scenario.cwd, + env: scenarioEnv(scenario) + }) + + expect(messages.some((m) => m.type === 'result')).toBe(true) + const report = scenario.readReport() + // The SDK executed exactly the script we pointed it at — no bundled binary. + expect(report.argv[0]).toBe(FAKE_CLI) + expect(report.execPath).toContain('node') + // And the streaming handshake went to it: the SDK sent its initialize + // control request to our script. + expect(report.controlRequests.some((frame) => frame.request.subtype === 'initialize')).toBe( + true + ) + }) + + it('pins the SDK version the contract was verified against', () => { + const sdkEntry = createRequire(__filename).resolve('@anthropic-ai/claude-agent-sdk') + const manifest = JSON.parse(readFileSync(join(dirname(sdkEntry), 'package.json'), 'utf8')) as { + version: string + } + expect(manifest.version).toBe(PINNED_SDK_VERSION) + }) + + it('keeps the eight bundled CLI platform binaries out of the install', () => { + const sdkEntry = createRequire(__filename).resolve('@anthropic-ai/claude-agent-sdk') + // The SDK's own scoped directory is where pnpm would link its optional + // platform packages; ignoredOptionalDependencies must keep them all absent. + const scopeDir = dirname(dirname(sdkEntry)) + for (const basename of SDK_PLATFORM_PACKAGE_BASENAMES) { + expect( + existsSync(join(scopeDir, basename, 'package.json')), + `${basename} must not be installed` + ).toBe(false) + } + }) +}) diff --git a/src/main/claude/claude-agent-sdk-control-requests.ts b/src/main/claude/claude-agent-sdk-control-requests.ts new file mode 100644 index 00000000000..6bd396413fa --- /dev/null +++ b/src/main/claude/claude-agent-sdk-control-requests.ts @@ -0,0 +1,154 @@ +import type { + PermissionMode, + Query, + SDKControlInterruptResponse +} from '@anthropic-ai/claude-agent-sdk' + +export class ClaudeControlRequestError extends Error { + constructor( + readonly subtype: string, + message: string + ) { + super(message) + this.name = 'ClaudeControlRequestError' + } +} + +export const CLAUDE_DEFAULT_REQUEST_TIMEOUT_MS = 30_000 + +/** The SDK closes a query out from under an in-flight control request with this exact message. */ +const QUERY_CLOSED_MESSAGE = 'Query closed before response received' + +/** 0.3.251 ships getSettings() but redacts it from the Query declaration; the typeof guard below is its degradation path. */ +type ClaudeQuerySettingsReader = { getSettings?: () => Promise } + +export function claudeQuerySettingsReader(query: Query): (() => Promise) | null { + const reader = (query as unknown as ClaudeQuerySettingsReader).getSettings + return typeof reader === 'function' ? reader.bind(query) : null +} + +/** + * cancel_async_message is a runtime Query method the shipped 0.3.251 declaration omits; + * it withdraws a single still-queued async user message by uuid so an interrupted turn + * cannot spawn a later unexpected turn. The typeof guard is its degradation path. + */ +type ClaudeQueryAsyncCanceller = { cancelAsyncMessage?: (uuid: string) => Promise } + +export function claudeQueryAsyncCanceller( + query: Query +): ((uuid: string) => Promise) | null { + const cancel = (query as unknown as ClaudeQueryAsyncCanceller).cancelAsyncMessage + return typeof cancel === 'function' ? cancel.bind(query) : null +} + +export type ClaudeControlOptions = { timeoutMs?: number } + +/** + * Run one native Query control method under Orca's deadline and error classification. + * + * The SDK owns correlation but applies no deadline, so the timeout stays here — and its + * message is load-bearing: the init proof matches on `claude initialize request timed out`. + * A closed query is a transport failure, not the CLI rejecting the request, so only the + * latter is re-thrown as a `ClaudeControlRequestError` a caller may surface as a rejection. + */ +export function runClaudeControl( + subtype: string, + run: () => Promise, + timeoutMs: number = CLAUDE_DEFAULT_REQUEST_TIMEOUT_MS +): Promise { + let timer: ReturnType | null = null + const deadline = new Promise((_resolve, reject) => { + timer = setTimeout(() => reject(new Error(`claude ${subtype} request timed out`)), timeoutMs) + timer.unref?.() + }) + return Promise.race([ + Promise.resolve() + .then(run) + .catch((error: unknown) => { + const message = error instanceof Error ? error.message : String(error) + if (error instanceof ClaudeControlRequestError || message === QUERY_CLOSED_MESSAGE) { + throw error + } + throw new ClaudeControlRequestError(subtype, message) + }), + deadline + ]).finally(() => { + if (timer) { + clearTimeout(timer) + } + }) +} + +/** The native control surface Orca drives, one method per Query control request. */ +export type ClaudeControlSurface = { + interrupt: ( + options?: ClaudeControlOptions & { cancelQueued?: boolean } + ) => Promise + cancelAsyncMessage: (uuid: string, options?: ClaudeControlOptions) => Promise + setModel: (model: string | undefined, options?: ClaudeControlOptions) => Promise + setPermissionMode: (mode: PermissionMode, options?: ClaudeControlOptions) => Promise + applyFlagSettings: ( + settings: Parameters[0], + options?: ClaudeControlOptions + ) => Promise + supportedModels: (options?: ClaudeControlOptions) => Promise + initializationResult: (options?: ClaudeControlOptions) => Promise + getSettings: (options?: ClaudeControlOptions) => Promise +} + +type InterruptingQuery = { + interrupt: (options?: { + cancelQueued?: boolean + }) => Promise +} + +export function createClaudeControlSurface(query: Query): ClaudeControlSurface { + return { + interrupt: (options) => + runClaudeControl( + 'interrupt', + () => + (query as unknown as InterruptingQuery).interrupt( + options?.cancelQueued ? { cancelQueued: true } : undefined + ), + options?.timeoutMs + ), + cancelAsyncMessage: (uuid, options) => { + const cancel = claudeQueryAsyncCanceller(query) + return cancel + ? runClaudeControl('cancel_async_message', () => cancel(uuid), options?.timeoutMs).then( + () => {} + ) + : Promise.resolve() + }, + setModel: (model, options) => + runClaudeControl('set_model', () => query.setModel(model), options?.timeoutMs).then(() => {}), + setPermissionMode: (mode, options) => + runClaudeControl( + 'set_permission_mode', + () => query.setPermissionMode(mode), + options?.timeoutMs + ).then(() => {}), + applyFlagSettings: (settings, options) => + runClaudeControl( + 'apply_flag_settings', + () => query.applyFlagSettings(settings), + options?.timeoutMs + ).then(() => {}), + supportedModels: (options) => + runClaudeControl('list_models', () => query.supportedModels(), options?.timeoutMs), + initializationResult: (options) => + runClaudeControl('initialize', () => query.initializationResult(), options?.timeoutMs), + getSettings: (options) => { + const read = claudeQuerySettingsReader(query) + return read + ? runClaudeControl('get_settings', read, options?.timeoutMs) + : Promise.reject( + new ClaudeControlRequestError( + 'get_settings', + 'this SDK exposes no get_settings request' + ) + ) + } + } +} diff --git a/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts b/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts new file mode 100644 index 00000000000..04d58067353 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof-identity.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, it, vi } from 'vitest' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { collectDescendantRows } from '../pty-descendant-termination' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' +import { mergeClaudeCapturedTrees } from './claude-child-tree-snapshot' + +function posixSnapshot(capturedAtMs: number): DescendantSnapshot { + return { + root: { pid: 100, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 100, + descendants: [{ pid: 200, ppid: 100, pgid: 100, startedAt: 'Mon Jan 1 00:00:01 2026' }], + capturedAtMs + } +} + +function windowsSnapshot(): WindowsDescendantSnapshot { + return { + root: { pid: 100, creationTimeMs: 5 }, + descendants: [{ pid: 200, creationTimeMs: 7 }], + unidentifiedCount: 0, + capturedAtMs: 1 + } +} + +describe('Claude child root identity', () => { + it('keeps a retained row boundary when a refresh observes no new descendants', () => { + const previous = posixSnapshot(1_700_000_000_900) + const next = posixSnapshot(1_700_000_002_100) + + expect( + mergeClaudeCapturedTrees( + { platform: 'posix', tree: previous }, + { platform: 'posix', tree: next } + ) + ).toEqual({ + platform: 'posix', + tree: { ...next, capturedAtMsByPid: { '200': previous.capturedAtMs } } + }) + }) + + it('keeps the descendant verdict when a POSIX root probe is unavailable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => posixSnapshot(1)), + terminateDescendants, + verifyRootIdentity: vi.fn(async () => false) + }) + + // POSIX runs no bare-pid root operation, so a declined probe withholds + // nothing: the handle kill still lands and the verification still speaks. + await expect(tree.reap()).resolves.toBe('exited') + expect(terminateDescendants).toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('rejects mixed old and recycled root rows instead of making the tree killable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => + collectDescendantRows( + 100, + [ + { pid: 100, ppid: 1, pgid: 100, startedAt: 'Mon Jan 1 00:00:00 2026' }, + { pid: 100, ppid: 1, pgid: 101, startedAt: 'Mon Jan 1 00:00:01 2026' }, + { pid: 200, ppid: 100, pgid: 200, startedAt: 'Mon Jan 1 00:00:00 2026' } + ], + 1 + ) + ), + terminateDescendants, + verifyRootIdentity: vi.fn(async () => true) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // No admissible snapshot means no row may be signalled from its number, but + // the root still leaves through the handle Node owns. + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('fails closed when Windows root identity revalidation is unavailable', async () => { + const child = { pid: 100, kill: vi.fn(() => true) } + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshot()), + terminateWindowsTree, + terminateWindowsDescendants, + verifyRootIdentity: vi.fn(async () => false) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // taskkill /T /F addresses a bare pid and stays gated; the handle does not. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-exit-proof.test.ts b/src/main/claude/claude-agent-sdk-exit-proof.test.ts new file mode 100644 index 00000000000..3f15f8e8fca --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof.test.ts @@ -0,0 +1,934 @@ +import { execFileSync } from 'node:child_process' +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import { spawnProcess, type SpawnedProcess } from '../../shared/child-process/run-process' +import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { + createClaudeChildTreeReaper as createClaudeChildTreeReaperImpl, + proveClaudeChildExit, + type ClaudeChildTreeReaper +} from './claude-agent-sdk-exit-proof' + +// The descendant models an MCP server: it either cooperates or, when it traps +// SIGTERM, only a forced, verified sweep can reach it. The root either traps +// SIGTERM too, or leaves promptly on stdin end the way a healthy CLI does — +// which is the path that used to skip descendant proof entirely. +function childWithDescendantScript(input: { + rootTrapsSigterm: boolean + descendantTrapsSigterm: boolean +}): string { + const descendantScript = `${input.descendantTrapsSigterm ? 'process.on("SIGTERM", () => {}); ' : ''}setInterval(() => {}, 1000000)` + const rootBehaviour = input.rootTrapsSigterm + ? `process.on('SIGTERM', () => {}) +process.on('SIGINT', () => {}) +setInterval(() => {}, 1000000)` + : `process.stdin.on('end', () => process.exit(0)) +process.stdin.resume()` + return ` +const descendant = require('node:child_process').spawn( + process.execPath, + ['-e', ${JSON.stringify(descendantScript)}], + { stdio: 'ignore' } +) +descendant.unref() +process.stdout.write(JSON.stringify({ descendantPid: descendant.pid }) + '\\n') +${rootBehaviour} +` +} + +const COOPERATIVE_CHILD = ` +process.stdin.on('end', () => process.exit(0)) +process.stdin.resume() +process.stdout.write('ready\\n') +` + +/** + * Sampled synchronously so it reads the exact moment the close boundary is + * crossed. A zombie has exited (its parent just has not reaped it yet), so a + * kill(pid, 0) probe would misreport it as running. + */ +function descendantState(pid: number): 'running' | 'exited' { + let state: string + try { + state = execFileSync('ps', ['-o', 'state=', '-p', String(pid)], { + encoding: 'utf8', + env: { ...process.env, LANG: 'C', LC_ALL: 'C' } + }).trim() + } catch (error) { + // ps exits 1 when no process matches; anything else is a failed probe, not an answer. + if ((error as { status?: number }).status !== 1) { + throw error + } + return 'exited' + } + return state.startsWith('Z') ? 'exited' : 'running' +} + +/** + * ps lstart is second-resolution, so the identity-safe sweep only SIGKILLs a row + * born strictly before the second the snapshot was captured in. The snapshot is + * armed the moment close begins, so a descendant born in that same second can + * only be asked, never forced — the same bound an MCP server spawned within a + * second of the user closing the chat would hit. + */ +function ageDescendantPastTheCaptureSecond(): Promise { + return new Promise((resolve) => setTimeout(resolve, 1_000 - (Date.now() % 1_000) + 20)) +} + +/** + * The close ladder as production drives it: `closeProcessRegistry` retries an + * unproven close, and each retry re-verifies the retained snapshot. A loaded + * host can spend one attempt's whole window inside `ps`, and reporting false + * there is the honest verdict — the requirement is that TRUE never outruns the + * observation, which the caller asserts at whichever boundary returns it. + */ +async function proveExitWithRetries( + input: Parameters[0], + attempts = 3 +): Promise { + for (let attempt = 1; attempt < attempts; attempt += 1) { + if (await proveClaudeChildExit(input)) { + return true + } + } + return proveClaudeChildExit(input) +} + +function spawnScript(script: string): ReturnType { + return spawnProcess({ + program: process.execPath, + args: ['-e', script], + stdio: ['pipe', 'pipe', 'pipe'] + }) +} + +function firstStdoutLine(child: ReturnType): Promise { + return new Promise((resolve) => { + child.stdout.setEncoding('utf8').once('data', (chunk: string) => resolve(chunk.trim())) + }) +} + +function observeExit(child: EventEmitter): { exitPromise: Promise; exited: () => boolean } { + let exited = false + const exitPromise = new Promise((resolve) => { + child.once('exit', () => { + exited = true + resolve() + }) + }) + return { exitPromise, exited: () => exited } +} + +/** `null` models a spawn that failed before a pid existed. */ +function mockChild( + pid: number | null = 424242 +): EventEmitter & + Pick & { kill: ReturnType } { + const child = new EventEmitter() + return Object.assign(child, { + pid: pid ?? undefined, + stdin: new PassThrough(), + kill: vi.fn(() => true) + }) as never +} + +/** A tree whose verdict is scripted per reap, recording when it was armed. */ +function mockTree(verdicts: DescendantTreeVerdict[]): ClaudeChildTreeReaper & { + capture: ReturnType + reap: ReturnType +} { + let treeVerdict: DescendantTreeVerdict = 'unverifiable' + return { + capture: vi.fn(async () => {}), + reap: vi.fn(async () => { + treeVerdict = verdicts.shift() ?? treeVerdict + return treeVerdict + }), + get treeVerdict() { + return treeVerdict + } + } +} + +function windowsSnapshotOf(descendantPid: number): WindowsDescendantSnapshot { + return { + root: { pid: 424242, creationTimeMs: 1_700_000_000_001 }, + descendants: [{ pid: descendantPid, creationTimeMs: 1_700_000_000_000 }], + unidentifiedCount: 0, + capturedAtMs: 1 + } +} + +function snapshotOf(descendantPid: number): DescendantSnapshot { + return { + root: { pid: 424242, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 1, + descendants: [ + { pid: descendantPid, ppid: 424242, pgid: 1, startedAt: 'Mon Jan 1 00:00:00 2026' } + ], + capturedAtMs: 1 + } +} + +// Unit tests use synthetic process ids; production always supplies the fresh +// identity probe, so the harness explicitly models a matching probe. +function createClaudeChildTreeReaper( + child: Parameters[0], + deps: Parameters[1] = {} +): ReturnType { + return createClaudeChildTreeReaperImpl(child, { + verifyRootIdentity: async () => true, + ...deps + }) +} + +describe('claude child exit proof', () => { + it.runIf(process.platform !== 'win32')( + 'reports a proven exit only once a SIGTERM-resistant descendant is gone at the close boundary', + async () => { + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: true, descendantTrapsSigterm: true }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + expect(descendantState(descendantPid)).toBe('running') + await ageDescendantPastTheCaptureSecond() + + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + // Evaluated AT the boundary, not by polling until a deferred sweep timer + // wins: true releases the lease, so a descendant still running here is + // exactly the orphan the proof exists to prevent. False would be the + // honest verdict for a tree that outlived the bounded ladder. + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + // Failure-safe only: the assertion above owns the requirement, this just + // stops a failing run from leaking a process. + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it.runIf(process.platform !== 'win32')( + 'proves a promptly exiting root only once its stubborn descendant is gone too', + async () => { + // The ordinary healthy close: the root leaves on stdin end within the graceful + // window. Its descendant must still be proven gone, not assumed gone with it. + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: false, descendantTrapsSigterm: true }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + expect(descendantState(descendantPid)).toBe('running') + await ageDescendantPastTheCaptureSecond() + + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it.runIf(process.platform !== 'win32')( + 'still proves a stubborn child whose descendant honours SIGTERM', + async () => { + const child = spawnScript( + childWithDescendantScript({ rootTrapsSigterm: true, descendantTrapsSigterm: false }) + ) + const { descendantPid } = JSON.parse(await firstStdoutLine(child)) as { + descendantPid: number + } + try { + const proven = await proveExitWithRetries({ child, ...observeExit(child) }) + expect({ proven, descendant: descendantState(descendantPid) }).toEqual({ + proven: true, + descendant: 'exited' + }) + } finally { + try { + process.kill(descendantPid, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it('arms the snapshot before stdin closes and verifies it after a clean exit', async () => { + const child = spawnScript(COOPERATIVE_CHILD) + expect(await firstStdoutLine(child)).toBe('ready') + const exit = observeExit(child) + const tree = mockTree(['exited']) + let exitedWhenArmed: boolean | null = null + tree.capture.mockImplementation(async () => { + exitedWhenArmed = exit.exited() + }) + + await expect(proveClaudeChildExit({ child, ...exit, tree })).resolves.toBe(true) + // The snapshot is the only proof that survives the root: taken while it lived, + // verified once it left. A reap before the exit would have been the forced ladder. + expect(exitedWhenArmed).toBe(false) + expect(tree.reap).toHaveBeenCalledTimes(1) + expect(exit.exited()).toBe(true) + }, 20_000) + + it('proves a clean close of a childless root with one snapshot and no signal', async () => { + const child = spawnScript(COOPERATIVE_CHILD) + expect(await firstStdoutLine(child)).toBe('ready') + + await expect(proveClaudeChildExit({ child, ...observeExit(child) })).resolves.toBe(true) + }, 20_000) + + it('reports an unprovable exit as false rather than assuming the child died', async () => { + const child = mockChild() + const tree = mockTree(['exited']) + + await expect( + proveClaudeChildExit({ + child, + exitPromise: new Promise(() => {}), + exited: () => false, + tree + }) + ).resolves.toBe(false) + expect(tree.reap).toHaveBeenCalledTimes(1) + }, 20_000) + + it('reports false when the root exit was observed but a descendant was seen alive', async () => { + const child = mockChild() + const exit = observeExit(child) + const tree = mockTree(['live']) + tree.reap.mockImplementation(async () => { + child.emit('exit', null, 'SIGKILL') + return 'live' + }) + + await expect(proveClaudeChildExit({ child, ...exit, tree })).resolves.toBe(false) + expect(exit.exited()).toBe(true) + // One verification per attempt: the retried close re-verifies, this one does not. + expect(tree.reap).toHaveBeenCalledTimes(1) + }, 20_000) + + it('re-verifies an unproven tree on a retried close instead of trusting the dead root', async () => { + const child = mockChild() + const tree = mockTree(['exited']) + + await expect( + proveClaudeChildExit({ child, exitPromise: Promise.resolve(), exited: () => true, tree }) + ).resolves.toBe(true) + expect(tree.reap).toHaveBeenCalledTimes(1) + }) + + it('stays unproven for a root that left before any snapshot could be armed', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const terminateDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'darwin', + exited: () => true, + captureDescendants, + terminateDescendants + }) + + await expect( + proveClaudeChildExit({ child, exitPromise: Promise.resolve(), exited: () => true, tree }) + ).resolves.toBe(false) + // A dead root's descendants have reparented: walking its pid now could only + // sweep a stranger, so no walk is attempted and nothing is proven. + expect(captureDescendants).not.toHaveBeenCalled() + expect(terminateDescendants).not.toHaveBeenCalled() + expect(tree.treeVerdict).toBe('unverifiable') + }) +}) + +describe('claude child tree reaper', () => { + it('kills the root while verification runs and never stops it first', async () => { + const child = mockChild() + const release = Promise.withResolvers() + const terminateDescendants = vi.fn(() => release.promise) + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const tree = createClaudeChildTreeReaper(child, { + platform: 'darwin', + captureDescendants, + terminateDescendants + }) + + const first = tree.reap() + const second = tree.reap() + await vi.waitFor(() => expect(terminateDescendants).toHaveBeenCalledTimes(1)) + // A stopped root cannot verify: its killed children stay zombie rows in ps. + expect(child.kill.mock.calls).toEqual([['SIGKILL']]) + expect(tree.treeVerdict).toBe('unverifiable') + + release.resolve('exited') + await expect(Promise.all([first, second])).resolves.toEqual(['exited', 'exited']) + expect(captureDescendants).toHaveBeenCalledTimes(1) + expect(tree.treeVerdict).toBe('exited') + }) + + it('re-verifies the retained snapshot on a later reap rather than re-walking a dead root', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => snapshotOf(4243)) + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('live') + .mockResolvedValueOnce('exited') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(1) + expect(terminateDescendants).toHaveBeenNthCalledWith(2, snapshotOf(4243)) + expect(tree.treeVerdict).toBe('exited') + }) + + it('keeps an observed exit when a later re-read cannot see the table', async () => { + const child = mockChild() + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('exited') + .mockResolvedValueOnce('unverifiable') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => snapshotOf(4243)), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('exited') + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(tree.treeVerdict).toBe('exited') + }) + + it('keeps an observed live descendant when a later re-read cannot see the table', async () => { + const child = mockChild() + // Reap #1 completed and saw a descendant alive at its deadline; the root then + // left on its own and the re-verification on a loaded host could not read the + // table. "Could not look" must not erase "was seen alive": the lease release + // gate is exactly the pair this distinguishes. + const terminateDescendants = vi + .fn() + .mockResolvedValueOnce('live') + .mockResolvedValueOnce('unverifiable') + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => snapshotOf(4243)), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('live') + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(tree.treeVerdict).toBe('live') + }) + + it('treats an unreadable process table as unproven and re-walks the live root', async () => { + const child = mockChild() + // A loaded host can miss the table's deadline; while the root still lives + // that is a retryable read, not evidence that it has no descendants. + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(snapshotOf(4243)) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateDescendants).not.toHaveBeenCalled() + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(2) + }) + + it('does not latch a missing root while it is still live', async () => { + const child = mockChild() + const captureDescendants = vi + .fn() + .mockResolvedValueOnce({ rootPgid: null, descendants: [], capturedAtMs: 1 }) + .mockResolvedValueOnce(snapshotOf(4243)) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(snapshotOf(4243)) + }) + + it('refreshes the live snapshot at close time so late descendants are included', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const second = { + ...first, + descendants: [...first.descendants, { ...first.descendants[0], pid: 4244 }] + } + const captureDescendants = vi.fn().mockResolvedValueOnce(first).mockResolvedValueOnce(second) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(second) + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('keeps the original capture boundary for retained POSIX rows', async () => { + const child = mockChild() + const first = { + ...snapshotOf(4243), + capturedAtMs: 1_700_000_000_900 + } + const refreshed = { + ...first, + capturedAtMs: 1_700_000_002_100, + descendants: [ + ...first.descendants, + { + pid: 4244, + ppid: 424242, + pgid: 1, + startedAt: 'Tue Jan 2 00:00:00 2026' + } + ] + } + const captureDescendants = vi.fn().mockResolvedValueOnce(first).mockResolvedValueOnce(refreshed) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(terminateDescendants).toHaveBeenCalledWith({ + ...refreshed, + // The retained 4243 row was first observed in the earlier displayed + // second. Its per-row boundary must not advance with the refresh. + capturedAtMsByPid: { + '4243': first.capturedAtMs, + '4244': refreshed.capturedAtMs + } + }) + }) + + it('fails closed when a POSIX refresh reuses a PID with a new identity', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const replacement = { + ...first, + descendants: [ + { + ...first.descendants[0], + pgid: 9, + startedAt: 'Tue Jan 2 00:00:00 2026' + } + ] + } + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + + await expect(tree.reap()).resolves.toBe('unverifiable') + // The descendant evidence is discarded; the root's identity never was in doubt. + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('fails closed when a Windows refresh reuses a PID with a new creation time', async () => { + const child = mockChild() + const first = windowsSnapshotOf(4243) + const replacement = { + ...first, + descendants: [{ pid: 4243, creationTimeMs: first.descendants[0].creationTimeMs + 1 }] + } + const captureWindowsDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants, + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + await tree.refresh?.() + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('queues a fresh boundary behind an output-triggered capture already in flight', async () => { + const child = mockChild() + const firstDone = Promise.withResolvers() + const first = snapshotOf(4243) + const second = { + ...first, + descendants: [...first.descendants, { ...first.descendants[0], pid: 4244 }] + } + const captureDescendants = vi + .fn() + .mockImplementationOnce(async () => { + await firstDone.promise + return first + }) + .mockResolvedValueOnce(second) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + const outputCapture = tree.refresh!() + await vi.waitFor(() => expect(captureDescendants).toHaveBeenCalledTimes(1)) + const closeCapture = tree.refresh!() + await Promise.resolve() + expect(captureDescendants).toHaveBeenCalledTimes(1) + + firstDone.resolve() + await closeCapture + await tree.reap() + + expect(captureDescendants).toHaveBeenCalledTimes(2) + expect(terminateDescendants).toHaveBeenCalledWith(second) + await outputCapture + }) + + it('retains a replacement descendant when the prior identity exited', async () => { + const child = mockChild() + const first = snapshotOf(4243) + const replacement = snapshotOf(4244) + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce(replacement) + const terminateDescendants = vi.fn(async (snapshot: DescendantSnapshot) => + snapshot.descendants.some((row) => row.pid === 4244) ? ('live' as const) : ('exited' as const) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await expect(tree.reap()).resolves.toBe('live') + + expect(terminateDescendants).toHaveBeenCalledWith({ + ...replacement, + descendants: [...first.descendants, ...replacement.descendants] + }) + }) + + it('retains a Windows replacement descendant while preserving unidentified rows', async () => { + const child = mockChild() + const first = windowsSnapshotOf(4243) + const replacement = { + ...windowsSnapshotOf(4244), + unidentifiedCount: 0 + } + const captureWindowsDescendants = vi + .fn() + .mockResolvedValueOnce({ ...first, unidentifiedCount: 1 }) + .mockResolvedValueOnce(replacement) + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async (snapshot: WindowsDescendantSnapshot) => + snapshot.descendants.some((row) => row.pid === 4244) ? ('live' as const) : ('exited' as const) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants, + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + await tree.refresh?.() + await expect(tree.reap()).resolves.toBe('live') + + expect(terminateWindowsDescendants).toHaveBeenCalledWith({ + ...replacement, + descendants: [...first.descendants, ...replacement.descendants], + unidentifiedCount: 1 + }) + }) + + it('retains the prior identity-safe snapshot when a refresh is partial', async () => { + const child = mockChild() + const first = { + ...snapshotOf(4243), + descendants: [ + ...snapshotOf(4243).descendants, + { ...snapshotOf(4243).descendants[0], pid: 4244 } + ] + } + const captureDescendants = vi + .fn() + .mockResolvedValueOnce(first) + .mockResolvedValueOnce({ + ...first, + descendants: first.descendants.slice(0, 1) + }) + const terminateDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants + }) + + await tree.capture() + await tree.refresh?.() + await tree.reap() + + expect(terminateDescendants).toHaveBeenCalledWith(first) + }) + + it('stops re-walking once the root is gone, however the table behaved', async () => { + const child = mockChild() + let exited = false + const captureDescendants = vi.fn(async () => null) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => exited, + captureDescendants, + terminateDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // An unreadable table costs the snapshot, never the kill on the live root. + expect(child.kill).toHaveBeenCalledTimes(1) + exited = true + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(captureDescendants).toHaveBeenCalledTimes(1) + // The second attempt observes a dead root: Node has dropped the handle, so + // there is nothing left to signal and no recycled pid to reach. + expect(child.kill).toHaveBeenCalledTimes(1) + }) + + it('discards a walk that found no root instead of proving an empty tree', async () => { + const child = mockChild() + const captureDescendants = vi.fn(async () => ({ + rootPgid: null, + descendants: [], + capturedAtMs: 1 + })) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants, + terminateDescendants: vi.fn() + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('unverifiable') + // A vacuous walk remains retryable while the root is live; no empty-tree + // verdict is latched from a missing root row. + expect(captureDescendants).toHaveBeenCalledTimes(2) + }) + + it('discards a walk that raced the root exit instead of proving an empty tree', async () => { + const child = mockChild() + let exited = false + const captureDescendants = vi.fn(async () => { + exited = true + return { rootPgid: 1, descendants: [], capturedAtMs: 1 } + }) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => exited, + captureDescendants, + terminateDescendants: vi.fn() + }) + + await tree.capture() + await expect(tree.reap()).resolves.toBe('unverifiable') + }) + + it('proves a childless snapshot without signalling anything', async () => { + const child = mockChild() + const terminateDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + captureDescendants: vi.fn(async () => ({ + root: { pid: 424242, startedAt: 'Mon Jan 1 00:00:00 2026' }, + rootPgid: 1, + descendants: [], + capturedAtMs: 1 + })), + terminateDescendants + }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(terminateDescendants).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('waits for the Windows tree kill before releasing the root', async () => { + const child = mockChild() + const release = Promise.withResolvers() + const terminateWindowsTree = vi.fn(() => release.promise) + const captureDescendants = vi.fn() + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureDescendants, + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants + }) + + const reap = tree.reap() + await vi.waitFor(() => + expect(terminateWindowsTree).toHaveBeenCalledWith({ + pid: 424242, + creationTimeMs: 1_700_000_000_001 + }) + ) + expect(child.kill).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + release.resolve() + await expect(reap).resolves.toBe('exited') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + expect(terminateWindowsDescendants).toHaveBeenCalledWith(windowsSnapshotOf(4243)) + expect(captureDescendants).not.toHaveBeenCalled() + }) + + it('stays unproven on Windows when taskkill fails and a descendant is still observed', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree: vi.fn(async () => { + throw new Error('taskkill: access denied') + }), + terminateWindowsDescendants: vi.fn(async () => 'live' as const) + }) + + // taskkill's own outcome is not the proof; the table read after it is. + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('stays unproven on Windows when taskkill resolves but a descendant survives it', async () => { + const child = mockChild() + const terminateWindowsTree = vi.fn(async () => {}) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants: vi.fn(async () => 'live' as const) + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(terminateWindowsTree).toHaveBeenCalledTimes(1) + expect(tree.treeVerdict).toBe('live') + }) + + it('never taskkills a Windows root that already exited, but still verifies its snapshot', async () => { + const child = mockChild() + let exited = false + const terminateWindowsTree = vi.fn(async () => {}) + const terminateWindowsDescendants = vi.fn(async () => 'exited' as const) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => exited, + captureWindowsDescendants: vi.fn(async () => windowsSnapshotOf(4243)), + terminateWindowsTree, + terminateWindowsDescendants + }) + + await tree.capture() + exited = true + await expect(tree.reap()).resolves.toBe('exited') + // A dead root's pid may already belong to a stranger: taskkill /T /F on it + // would take down an unrelated tree. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(terminateWindowsDescendants).toHaveBeenCalledWith(windowsSnapshotOf(4243)) + }) + + it('treats an unreadable Windows table as unproven', async () => { + const child = mockChild() + const terminateWindowsDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + captureWindowsDescendants: vi.fn(async () => null), + terminateWindowsTree: vi.fn(async () => {}), + terminateWindowsDescendants + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(terminateWindowsDescendants).not.toHaveBeenCalled() + // A host that cannot supply creation times blocks taskkill, not the root kill. + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('has nothing to reap for a child that never spawned', async () => { + const child = mockChild(null) + const captureDescendants = vi.fn() + const tree = createClaudeChildTreeReaper(child, { platform: 'linux', captureDescendants }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(captureDescendants).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-agent-sdk-exit-proof.ts b/src/main/claude/claude-agent-sdk-exit-proof.ts new file mode 100644 index 00000000000..17533f87a70 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-exit-proof.ts @@ -0,0 +1,366 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import { + terminateDescendantSnapshotWithVerdict, + type DescendantTreeVerdict +} from '../pty-descendant-exit-verification' +import { + captureDescendantSnapshot, + type DescendantSnapshot, + type PosixProcessIdentity +} from '../pty-descendant-termination' +import { + captureWindowsDescendantSnapshot, + terminateIdentifiedWindowsProcessTree, + verifyWindowsDescendantSnapshotExit, + verifyWindowsProcessIdentity, + type WindowsDescendantSnapshot, + type WindowsProcessIdentity +} from '../windows-descendant-exit-verification' +import { mergeClaudeCapturedTrees, type ClaudeCapturedTree } from './claude-child-tree-snapshot' +import { terminateClaudeRoot, terminateClaudeWindowsRoot } from './claude-child-root-termination' +import { + proveClaudeChildExitWithReaper, + type ClaudeChildExitProofInput +} from './claude-child-exit-proof-ladder' + +/** + * A later reap may only raise the latched verdict. An observed exit is final, and + * a descendant seen alive at a deadline is never forgotten by a later look that + * could not read the table: the lease gate discriminates on exactly that pair. + */ +const TREE_VERDICT_TRUST: Record = { + unverifiable: 0, + live: 1, + exited: 2 +} + +type ReapableChild = Pick + +/** + * A walk is only admissible while the root it walked was alive. A POSIX walk + * that found no root says so with a null pgid; either platform's walk can also + * have raced the root's death. Both can only have missed descendants that + * already reparented away, so neither is evidence about the tree. + */ +function admissibleTree( + captured: DescendantSnapshot | WindowsDescendantSnapshot | null, + platform: NodeJS.Platform, + exited: boolean +): ClaudeCapturedTree | null { + if (!captured || exited) { + return null + } + if (platform === 'win32') { + return { platform: 'win32', tree: captured as WindowsDescendantSnapshot } + } + const tree = captured as DescendantSnapshot + return tree.rootPgid === null ? null : { platform: 'posix', tree } +} + +export type ClaudeChildTreeReaperDeps = { + platform?: NodeJS.Platform + /** Whether the root's exit has been observed; only a live root can be walked. */ + exited?: () => boolean + captureDescendants?: (rootPid: number) => Promise + terminateDescendants?: (snapshot: DescendantSnapshot) => Promise + terminateWindowsTree?: (root: WindowsProcessIdentity) => Promise + captureWindowsDescendants?: (rootPid: number) => Promise + terminateWindowsDescendants?: ( + snapshot: WindowsDescendantSnapshot + ) => Promise + /** Identity probe for the bare-pid tree kill; only Windows has one to gate. */ + verifyRootIdentity?: (root: PosixProcessIdentity | WindowsProcessIdentity) => Promise +} + +export type ClaudeChildTreeReaper = { + /** + * Snapshot the root's live descendants. The moment the root dies they reparent + * and no table walk can find them again, so this has to run before anything + * gives the root a reason to leave. Held once; later calls are no-ops. + */ + capture(): Promise + /** Refresh a live root's snapshot at the close boundary; a failed refresh keeps the prior proof. */ + refresh?: () => Promise + /** + * Kill the child's whole tree and report what the bounded verification + * observed. Concurrent calls share one reap, and a later call re-verifies the + * same snapshot rather than trusting a root that has since died on its own. + */ + reap(): Promise + /** + * `unverifiable` until a reap observes otherwise. `exited` is the only verdict + * that lets a close release the lease; `live` names a descendant that was seen + * still running, which no later caller may collapse into "unknown". + */ + readonly treeVerdict: DescendantTreeVerdict +} + +/** + * The same shared primitives the Codex structured provider composes: a raw + * pipe child owns no PTY job, so there is nothing for the PTY job sweep to + * terminate on Windows and no unref'd timer is allowed to outlive the proof. + * + * The proof is unproven by default. `treeVerdict` is assigned in exactly one + * place, from the verdict of `judgeTree`, so a code path that never reaches a + * verification cannot report the tree gone by omission. + */ +export function createClaudeChildTreeReaper( + child: ReapableChild, + deps: ClaudeChildTreeReaperDeps = {} +): ClaudeChildTreeReaper { + const platform = deps.platform ?? process.platform + const exited = deps.exited ?? (() => false) + // Undefined until captured; null when no admissible snapshot exists — the root + // was already gone, or the table could not be read while it was alive — which + // no later read can make up for. + let snapshot: ClaudeCapturedTree | null | undefined + let capturing: Promise | null = null + let refreshing: Promise | null = null + let queuedRefresh: Promise | null = null + let inFlight: Promise | null = null + let treeVerdict: DescendantTreeVerdict = 'unverifiable' + + // Consulted only on win32: POSIX signals descendants by revalidated identity + // and reaches the root solely through Node's handle, so neither needs a probe. + const verifyRoot = + deps.verifyRootIdentity ?? + ((root: PosixProcessIdentity | WindowsProcessIdentity) => + verifyWindowsProcessIdentity(root as WindowsProcessIdentity)) + + function captureOnce(): Promise { + if (refreshing) { + const pending = refreshing + return pending.then(() => queuedRefresh ?? undefined) + } + if (snapshot !== undefined) { + return Promise.resolve() + } + if (capturing) { + const pending = capturing + return pending.then(() => queuedRefresh ?? undefined) + } + const rootPid = child.pid + if (!rootPid || exited()) { + // Only the root's death makes a missing snapshot final: its descendants + // have reparented, and no later walk can reach them. + snapshot = exited() ? null : snapshot + return Promise.resolve() + } + const capture = + platform === 'win32' + ? (deps.captureWindowsDescendants ?? captureWindowsDescendantSnapshot) + : (deps.captureDescendants ?? captureDescendantSnapshot) + capturing = capture(rootPid) + .catch(() => null) + .then((captured) => { + // A walk that found no root, or that raced the root's death, can only + // have missed descendants that already reparented away. A table that + // could not be read in time is not an answer at all: while the root + // still lives the walk is simply retried, rather than latching a failed + // read as proof that there was nothing to find. + const rootExited = exited() + const tree = admissibleTree(captured, platform, rootExited) + if (tree) { + snapshot = tree + } else if (rootExited) { + // Once the root has exited its descendants may have reparented; no + // later table read can make an absent snapshot safe to signal. + snapshot = null + } else { + // A failed read or a walk that did not observe the live root is + // retryable while the root remains alive. Never latch a vacuous null. + snapshot = undefined + } + }) + .finally(() => { + capturing = null + }) + return capturing + } + + function startRefresh(): Promise { + if (exited()) { + return Promise.resolve() + } + const rootPid = child.pid + if (!rootPid) { + return Promise.resolve() + } + const capture = + platform === 'win32' + ? (deps.captureWindowsDescendants ?? captureWindowsDescendantSnapshot) + : (deps.captureDescendants ?? captureDescendantSnapshot) + const operation = (async () => { + const captured = await capture(rootPid).catch(() => null) + if (exited()) { + return + } + const tree = admissibleTree(captured, platform, false) + if (!tree) { + return + } + if (snapshot === undefined) { + snapshot = tree + return + } + if (snapshot !== null) { + // A merge that returns null saw a same-PID identity change: a + // recycle/replace decision, not an absent descendant, so no row here may + // be signalled from its number. Only the descendant evidence is lost — + // the root still leaves through the handle no recycled pid can reach. + snapshot = mergeClaudeCapturedTrees(snapshot, tree) + } + // Keep an earlier admissible snapshot when this close-boundary read fails; + // it remains the only identity-safe evidence after root exit. + })() + refreshing = operation + const clearRefreshing = (): void => { + if (refreshing === operation) { + refreshing = null + } + } + void operation.then(clearRefreshing, clearRefreshing) + return operation + } + + function queueRefreshAfter(pending: Promise): Promise { + if (queuedRefresh) { + return queuedRefresh + } + const operation = pending.then(() => { + if (exited()) { + return + } + return startRefresh() + }) + queuedRefresh = operation + const clearQueuedRefresh = (): void => { + if (queuedRefresh === operation) { + queuedRefresh = null + } + } + void operation.then(clearQueuedRefresh, clearQueuedRefresh) + return operation + } + + async function refresh(): Promise { + const pending = capturing ?? refreshing + if (pending) { + await queueRefreshAfter(pending) + return + } + if (queuedRefresh) { + await queuedRefresh + return + } + try { + await startRefresh() + } catch { + // A refresh is advisory; capture failures leave the prior proof intact. + } + } + + /** The only source of a tree verdict: every `exited` here is an observation. */ + async function judgeTree(): Promise { + const killRoot = (): boolean => terminateClaudeRoot({ child, exited }) + const rootPid = child.pid + if (!rootPid) { + // Never spawned, so the OS never created a tree to orphan. + return 'exited' + } + await captureOnce() + if (platform === 'win32') { + // Why taskkill's own outcome is never the verdict: it resolves identically + // on a timeout, an access denial, a recycled root and a real kill. + const { rootVerified } = await terminateClaudeWindowsRoot({ + snapshot: snapshot?.platform === 'win32' ? snapshot.tree : null, + exited, + verifyRoot: (root) => verifyRoot(root), + terminateTree: (root) => + deps.terminateWindowsTree + ? deps.terminateWindowsTree(root) + : terminateIdentifiedWindowsProcessTree(root, { + ownsRoot: () => !exited() + }).then(() => undefined), + killRoot + }) + if (!rootVerified && !exited()) { + return 'unverifiable' + } + return snapshot?.platform === 'win32' + ? await (deps.terminateWindowsDescendants ?? verifyWindowsDescendantSnapshotExit)( + snapshot.tree + ) + : 'unverifiable' + } + if (snapshot?.platform !== 'posix') { + killRoot() + return 'unverifiable' + } + if (snapshot.tree.descendants.length === 0) { + // Read while the root was alive and childless: a later table read has no + // row it could match, so it would add nothing to this observation. + killRoot() + return 'exited' + } + // Why the root is killed while verification is already running, and never + // SIGSTOPped first the way the Codex non-group path does: measured on macOS, a + // killed child of a stopped parent stays a zombie row in ps with its lstart + // and pgid intact, so verification cannot pass until the root is dead. The + // descendants are signalled by the verifier as soon as it revalidates their + // identities; the root's death then reparents any zombies to init, which + // reaps them. After a root exit the kill is a no-op: Node drops the handle + // on exit and never signals a possibly recycled pid. + const verdictPromise = deps.terminateDescendants + ? deps.terminateDescendants(snapshot.tree) + : terminateDescendantSnapshotWithVerdict(snapshot.tree, { + requireIdentityBeforeSignal: true + }) + killRoot() + // What the verification observed is the verdict: a kill that reports no + // signal means the handle was already gone, never that the tree survived. + return verdictPromise + } + + return { + capture: captureOnce, + refresh, + reap() { + if (inFlight) { + return inFlight + } + const attempt = judgeTree() + .catch((): DescendantTreeVerdict => 'unverifiable') + .then((verdict) => { + treeVerdict = + TREE_VERDICT_TRUST[verdict] > TREE_VERDICT_TRUST[treeVerdict] ? verdict : treeVerdict + return verdict + }) + inFlight = attempt + void attempt.finally(() => { + if (inFlight === attempt) { + inFlight = null + } + }) + return attempt + }, + get treeVerdict() { + return treeVerdict + } + } +} + +/** + * Orca's own shutdown ladder on the child it spawned, kept because the SDK's + * close path returns no proof and Orca never releases a lease on an assumed exit. + * + * Resolves true only after the child actually emitted exit and its snapshotted + * descendants were observed gone; false is unproven. A root that left on its + * own before a snapshot could be armed stays unproven: its descendants had + * already reparented out of reach when the ladder first looked. + */ +export function proveClaudeChildExit(input: ClaudeChildExitProofInput): Promise { + return proveClaudeChildExitWithReaper(input, () => + createClaudeChildTreeReaper(input.child, { exited: input.exited }) + ) +} diff --git a/src/main/claude/claude-agent-sdk-import-boundary.test.ts b/src/main/claude/claude-agent-sdk-import-boundary.test.ts new file mode 100644 index 00000000000..f1a38466d41 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-import-boundary.test.ts @@ -0,0 +1,154 @@ +import { existsSync, readFileSync, statSync } from 'node:fs' +import { dirname, join, relative, resolve } from 'node:path' +import { describe, expect, it } from 'vitest' +import { spawnProcess } from '../../shared/child-process/run-process' + +/** + * Keep the agent SDK on the structured-Claude side of the toggle. + * + * A user who never leaves the terminal/TUI Claude path must not pay for the SDK: + * importing it evaluates a package that rewrites + * `process.env.NoDefaultCurrentDirectoryInExePath`, changing how Windows resolves + * executables for every later subprocess, and a missing or incompatible install + * would take normal runtime startup down with it. The ordinary + * `OrcaRuntimeService` graph reaches the Claude transport module, so only a + * deferred import keeps that boundary — and only a walk of the real import graph + * keeps the next static import from quietly restoring it. + */ +const SDK_PACKAGE = '@anthropic-ai/claude-agent-sdk' +const REPO_ROOT = resolve(__dirname, '..', '..', '..') + +/** The Electron main entry: everything the app loads before any session exists. */ +const ROOT = 'src/main/index.ts' +/** Proof the walk goes all the way into the Claude transport rather than stopping short. */ +const TRANSPORT_MODULE = 'src/main/claude/claude-stream-json-connection.ts' + +/** + * Static, value-carrying specifiers only, read statement by statement so a + * multi-line `import { ... } from '...'` counts. `import type` is erased before + * the module ever loads and a bare `import(...)` is the deferral this guards, so + * neither is an edge the runtime traverses at load time. + */ +const STATEMENT_START = /^\s*(?:import|export)\b/ +const TYPE_ONLY = /^\s*(?:import|export)\s+type\b/ +const FROM_SPECIFIER = /(?:^|\s)from\s*['"]([^'"]+)['"]/ +const SIDE_EFFECT_IMPORT = /^\s*import\s*['"]([^'"]+)['"]/ +/** An import statement never spans more lines than its longest specifier list. */ +const MAX_STATEMENT_LINES = 60 + +function readSpecifiers(source: string): string[] { + const lines = source.split('\n') + const found: string[] = [] + for (let index = 0; index < lines.length; index += 1) { + const first = lines[index] as string + if (!STATEMENT_START.test(first) || TYPE_ONLY.test(first)) { + continue + } + const sideEffect = SIDE_EFFECT_IMPORT.exec(first) + if (sideEffect) { + found.push(sideEffect[1] as string) + continue + } + for (let scan = index; scan < Math.min(lines.length, index + MAX_STATEMENT_LINES); scan += 1) { + if (scan > index && STATEMENT_START.test(lines[scan] as string)) { + break + } + const specifier = FROM_SPECIFIER.exec(lines[scan] as string) + if (specifier) { + found.push(specifier[1] as string) + break + } + } + } + return found +} + +/** Resolve a relative specifier the way the bundler does; unresolvable means not a module. */ +function resolveRelative(fromFile: string, specifier: string): string | null { + const base = join(dirname(fromFile), specifier) + for (const candidate of [base, `${base}.ts`, `${base}.tsx`, join(base, 'index.ts')]) { + if (existsSync(candidate) && statSync(candidate).isFile()) { + return candidate + } + } + return null +} + +function walkStaticImports(rootFile: string): { visited: Set; sdkImporters: string[] } { + const visited = new Set() + const sdkImporters: string[] = [] + const queue = [resolve(REPO_ROOT, rootFile)] + while (queue.length > 0) { + const file = queue.pop() as string + const key = relative(REPO_ROOT, file).split('\\').join('/') + if (visited.has(key)) { + continue + } + visited.add(key) + for (const specifier of readSpecifiers(readFileSync(file, 'utf8'))) { + if (specifier === SDK_PACKAGE || specifier.startsWith(`${SDK_PACKAGE}/`)) { + sdkImporters.push(key) + continue + } + if (!specifier.startsWith('.')) { + continue + } + const target = resolveRelative(file, specifier) + if (target) { + queue.push(target) + } + } + } + return { visited, sdkImporters } +} + +describe('claude agent SDK import boundary', () => { + const walk = walkStaticImports(ROOT) + + it('walks a graph deep enough to reach the Claude transport', () => { + // Without this the guard passes for the wrong reason the moment the walk breaks. + expect(walk.visited.size).toBeGreaterThan(500) + expect([...walk.visited]).toContain(TRANSPORT_MODULE) + }) + + it('never reaches the SDK through a static import from the main entry', () => { + expect( + walk.sdkImporters, + `${SDK_PACKAGE} must stay behind the structured-Claude boundary. Load it with a deferred import inside the session path instead.` + ).toEqual([]) + }) + + it('leaves the Windows executable-search environment alone when the runtime loads', async () => { + // A vitest file runs in its own fork, so this is a clean process; the ambient + // value is cleared first because the developer's own shell may carry one. + delete process.env.NoDefaultCurrentDirectoryInExePath + await import('../runtime/structured-agent-session-runtime') + + expect(process.env.NoDefaultCurrentDirectoryInExePath).toBeUndefined() + }) + + it('still lets the SDK set it, so the guard above is not measuring nothing', async () => { + // A separate process, not this fork: the assertion has to be about a first + // evaluation of the package, which a cached module registry cannot give. + const { NoDefaultCurrentDirectoryInExePath: _cleared, ...env } = process.env + const probe = spawnProcess({ + program: process.execPath, + args: [ + '-e', + `import(${JSON.stringify(SDK_PACKAGE)}).then(() => console.log(String(process.env.NoDefaultCurrentDirectoryInExePath)))` + ], + cwd: REPO_ROOT, + env: env as Record, + stdio: ['ignore', 'pipe', 'ignore'] + }) + const observed = await new Promise((settle) => { + let output = '' + probe.stdout?.setEncoding('utf8').on('data', (chunk: string) => { + output += chunk + }) + probe.once('close', () => settle(output.trim())) + }) + + expect(observed).toBe('1') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-process-spawn.test.ts b/src/main/claude/claude-agent-sdk-process-spawn.test.ts new file mode 100644 index 00000000000..cd3520cf6d5 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-process-spawn.test.ts @@ -0,0 +1,107 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import type { SpawnOptions as SdkSpawnOptions } from '@anthropic-ai/claude-agent-sdk' +import { resolveSpawn, type spawnProcess } from '../../shared/child-process/run-process' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' + +type FakeChild = EventEmitter & { + pid: number + stdin: PassThrough + stdout: PassThrough + stderr: PassThrough + kill: ReturnType +} + +function fakeSpawn() { + const child = new EventEmitter() as FakeChild + child.pid = 4321 + child.stdin = new PassThrough() + child.stdout = new PassThrough() + child.stderr = new PassThrough() + child.kill = vi.fn(() => true) + const specs: ProcessSpec[] = [] + const spawnImpl = ((spec: ProcessSpec) => { + specs.push(spec) + return child + }) as unknown as typeof spawnProcess + return { child, spawnImpl, specs } +} + +function sdkOptions(overrides: Partial = {}): SdkSpawnOptions { + return { + command: '/usr/local/bin/claude', + args: ['--output-format', 'stream-json'], + cwd: '/work/repo', + env: { PATH: '/usr/bin', CLAUDE_CONFIG_DIR: '/accounts/one', UNSET: undefined }, + signal: new AbortController().signal, + ...overrides + } +} + +describe('claude agent SDK process spawn', () => { + it('routes the SDK spawn through Orca and retains the pid the lease adjudicates on', () => { + const process = fakeSpawn() + const spawn = createClaudeCodeProcessSpawn(process.spawnImpl) + + expect(spawn.pid).toBeUndefined() + expect(spawn.child).toBeNull() + const child = spawn.spawn(sdkOptions()) + + expect(child).toBe(process.child) + expect(spawn.child).toBe(process.child) + expect(spawn.pid).toBe(4321) + expect(process.specs[0]).toEqual({ + program: '/usr/local/bin/claude', + args: ['--output-format', 'stream-json'], + cwd: '/work/repo', + env: { PATH: '/usr/bin', CLAUDE_CONFIG_DIR: '/accounts/one' }, + stdio: ['pipe', 'pipe', 'pipe'] + }) + }) + + it('keeps the child out of the SDK abort path so exit proof stays Orca-owned', () => { + const process = fakeSpawn() + const controller = new AbortController() + createClaudeCodeProcessSpawn(process.spawnImpl).spawn(sdkOptions({ signal: controller.signal })) + + // Node's spawn({signal}) kills the child on abort; Orca's ladder must be the + // only thing that can end this process, or close() would report an assumed exit. + expect(process.specs[0]).not.toHaveProperty('signal') + }) + + it('drains stderr into a bounded tail so an exit error still carries it', async () => { + const process = fakeSpawn() + const spawn = createClaudeCodeProcessSpawn(process.spawnImpl) + spawn.spawn(sdkOptions()) + + process.child.stderr.write('x'.repeat(9000)) + process.child.stderr.write('claude: not signed in') + await new Promise((resolve) => setImmediate(resolve)) + + expect(spawn.stderrTail).toMatch(/claude: not signed in$/) + expect(spawn.stderrTail.length).toBe(8192) + }) + + it('hands a Windows .cmd shim to Orca\u2019s argument encoder', () => { + const process = fakeSpawn() + createClaudeCodeProcessSpawn(process.spawnImpl).spawn( + sdkOptions({ + command: 'C:\\Users\\dev\\AppData\\npm\\claude.cmd', + args: ['--setting-sources=user,project,local', '--session-id', 'a b&c'] + }) + ) + + // The spec the spawner builds is what Orca's Windows branch encodes; the SDK's + // own spawn would hand `.cmd` straight to Node and mangle the argument. + const resolved = resolveSpawn(process.specs[0] as ProcessSpec, 'win32') + expect(resolved.file.toLowerCase()).toContain('cmd.exe') + expect(resolved.options.windowsVerbatimArguments).toBe(true) + expect(resolved.args).toHaveLength(1) + // `/v:off` plus the quoted argument is what keeps `&` from splitting the line. + expect(resolved.args[0]).toContain('/v:off') + expect(resolved.args[0]).toContain('"a b&c"') + expect(resolved.args[0]).toContain('"--setting-sources=user,project,local"') + }) +}) diff --git a/src/main/claude/claude-agent-sdk-process-spawn.ts b/src/main/claude/claude-agent-sdk-process-spawn.ts new file mode 100644 index 00000000000..a2b1ad7f158 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-process-spawn.ts @@ -0,0 +1,69 @@ +import type { SpawnOptions as ClaudeAgentSdkSpawnOptions } from '@anthropic-ai/claude-agent-sdk' +import { spawnProcess } from '../../shared/child-process/run-process' + +/** Derived rather than imported: only src/shared/child-process may name node:child_process. */ +type ClaudeCodeChild = ReturnType + +const STDERR_TAIL_MAX_BYTES = 8192 + +export type ClaudeCodeProcessSpawn = { + /** Pass as the SDK's `spawnClaudeCodeProcess`; the SDK never learns the pid because it never owns it. */ + spawn: (options: ClaudeAgentSdkSpawnOptions) => ClaudeCodeChild + /** The retained child, so Orca keeps its own tree-kill and exit-proof ladder. Null until the SDK spawns. */ + readonly child: ClaudeCodeChild | null + /** Ownership proof: the durable lease adjudicates on this pid plus start time plus the spawn token. */ + readonly pid: number | undefined + readonly stderrTail: string +} + +function definedEnv(env: Record): Record { + const next: Record = {} + for (const [key, value] of Object.entries(env)) { + if (value !== undefined) { + next[key] = value + } + } + return next +} + +/** + * Orca supplies the Claude Code child rather than letting the SDK spawn it. + * + * Two independent reasons: the SDK's `SpawnedProcess` has no pid, and Orca's + * spawner is the only path that encodes `.cmd` arguments safely on Windows. + */ +export function createClaudeCodeProcessSpawn( + spawnImpl: typeof spawnProcess = spawnProcess +): ClaudeCodeProcessSpawn { + let child: ClaudeCodeChild | null = null + let stderrTail = '' + return { + spawn: (options) => { + // Why `options.signal` is dropped: it would let the SDK kill the child outside + // Orca's ladder, and close() may never report an exit it did not observe. + const spawned = spawnImpl({ + program: options.command, + args: [...options.args], + ...(options.cwd === undefined ? {} : { cwd: options.cwd }), + env: definedEnv(options.env), + stdio: ['pipe', 'pipe', 'pipe'] + }) + child = spawned + // The SDK drains stderr only for its own local spawn, so a custom spawner must: + // otherwise the child blocks on a full pipe and exit errors lose their tail. + spawned.stderr.setEncoding('utf8').on('data', (chunk: string) => { + stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_MAX_BYTES) + }) + return spawned + }, + get child() { + return child + }, + get pid() { + return child?.pid + }, + get stderrTail() { + return stderrTail + } + } +} diff --git a/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts b/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts new file mode 100644 index 00000000000..9f843527e30 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-root-kill-fallback.test.ts @@ -0,0 +1,190 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import { describe, expect, it, vi } from 'vitest' +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' +import { mergeClaudeCapturedTrees } from './claude-child-tree-snapshot' + +const ROOT_PID = 424242 +const ROOT_STARTED_AT = 'Mon Jan 1 00:00:00 2026' +const ROOT_FORK_MS = Date.parse(ROOT_STARTED_AT) + +function mockChild(): EventEmitter & + Pick & { kill: ReturnType } { + return Object.assign(new EventEmitter(), { + pid: ROOT_PID, + stdin: new PassThrough(), + kill: vi.fn(() => true) + }) as never +} + +function posixSnapshot(input: { + capturedAtMs: number + descendants?: DescendantSnapshot['descendants'] +}): DescendantSnapshot { + return { + root: { pid: ROOT_PID, startedAt: ROOT_STARTED_AT }, + rootPgid: ROOT_PID, + descendants: input.descendants ?? [], + capturedAtMs: input.capturedAtMs + } +} + +function windowsSnapshot(capturedAtMs = 1): WindowsDescendantSnapshot { + return { + root: { pid: ROOT_PID, creationTimeMs: 1_700_000_000_001 }, + descendants: [{ pid: 4243, creationTimeMs: 1_700_000_000_000 }], + unidentifiedCount: 0, + capturedAtMs + } +} + +describe('Claude root kill fallback', () => { + it('kills the root when the first capture landed in the fork second', async () => { + // The production POSIX verifier declines a root born in its capture second, + // and that verdict must not cost the tree the kill on Node's own handle. + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => posixSnapshot({ capturedAtMs: ROOT_FORK_MS + 300 })) + }) + + await expect(tree.reap()).resolves.toBe('exited') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the root after a recycled descendant pid voided the snapshot', async () => { + const child = mockChild() + const captureDescendants = vi + .fn() + .mockResolvedValueOnce( + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 5_000, + descendants: [{ pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: ROOT_STARTED_AT }] + }) + ) + .mockResolvedValueOnce( + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 6_000, + descendants: [ + { pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: 'Mon Jan 1 00:00:30 2026' } + ] + }) + ) + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants, + terminateDescendants: vi.fn(async () => 'exited' as const), + verifyRootIdentity: vi.fn(async () => true) + }) + + await tree.capture() + await tree.refresh?.() + // The descendant evidence is rightly discarded; the root's never was in doubt. + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('keeps an observed live descendant when the root identity probe declined', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => + posixSnapshot({ + capturedAtMs: ROOT_FORK_MS + 5_000, + descendants: [{ pid: 100, ppid: ROOT_PID, pgid: ROOT_PID, startedAt: ROOT_STARTED_AT }] + }) + ), + terminateDescendants: vi.fn(async () => 'live' as const), + verifyRootIdentity: vi.fn(async () => false) + }) + + await expect(tree.reap()).resolves.toBe('live') + expect(tree.treeVerdict).toBe('live') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('reports a Windows taskkill that worked as exited, not unverifiable', async () => { + const child = mockChild() + // Probe 1 gates taskkill; a later probe correctly finds the root already dead. + const verifyRootIdentity = vi.fn().mockResolvedValueOnce(true).mockResolvedValue(false) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => false, + captureWindowsDescendants: vi.fn(async () => windowsSnapshot()), + terminateWindowsTree: vi.fn(async () => {}), + terminateWindowsDescendants: vi.fn(async () => 'exited' as const), + verifyRootIdentity + }) + + await expect(tree.reap()).resolves.toBe('exited') + }) + + it('kills the root when no POSIX snapshot could be read', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => false, + captureDescendants: vi.fn(async () => null), + terminateDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('kills the root when the Windows process table is unreadable', async () => { + const child = mockChild() + const terminateWindowsTree = vi.fn(async () => {}) + const tree = createClaudeChildTreeReaper(child, { + platform: 'win32', + exited: () => false, + captureWindowsDescendants: vi.fn(async () => null), + terminateWindowsTree, + terminateWindowsDescendants: vi.fn() + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + // No identity means no bare-pid tree kill, but the owned handle is still ours. + expect(terminateWindowsTree).not.toHaveBeenCalled() + expect(child.kill).toHaveBeenCalledWith('SIGKILL') + }) + + it('never signals a root the reaper already saw exit', async () => { + const child = mockChild() + const tree = createClaudeChildTreeReaper(child, { + platform: 'linux', + exited: () => true, + captureDescendants: vi.fn(async () => null) + }) + + await expect(tree.reap()).resolves.toBe('unverifiable') + expect(child.kill).not.toHaveBeenCalled() + }) + + it('chains per-pid Windows boundaries across a second merge', async () => { + const first = windowsSnapshot(1_000) + const second: WindowsDescendantSnapshot = { + ...windowsSnapshot(2_000), + descendants: [ + { pid: 4243, creationTimeMs: 1_700_000_000_000 }, + { pid: 4244, creationTimeMs: 1_700_000_000_002 } + ] + } + const third: WindowsDescendantSnapshot = { ...second, capturedAtMs: 3_000 } + + const merged = mergeClaudeCapturedTrees( + { platform: 'win32', tree: first }, + { platform: 'win32', tree: second } + ) + expect(merged?.tree.capturedAtMsByPid).toEqual({ '4243': 1_000, '4244': 2_000 }) + const rechained = mergeClaudeCapturedTrees(merged!, { platform: 'win32', tree: third }) + + expect(rechained?.tree.capturedAtMsByPid).toEqual({ '4243': 1_000, '4244': 2_000 }) + }) +}) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.test.ts b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts new file mode 100644 index 00000000000..62fbd8cf203 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-user-message-queue.test.ts @@ -0,0 +1,65 @@ +import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' +import { describe, expect, it } from 'vitest' +import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' + +/** + * The SDK's input pump is `for await (const frame of prompt) { await transport.write(frame) }`. + * A rejected write — or an abort — ends that loop abruptly, which calls the + * generator's `return()`. Everything below drives that exact shape, because the + * frame the pump already pulled is the one nothing else can reach. + */ +const frame = (text: string): SDKUserMessage => + ({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text }] } + }) as unknown as SDKUserMessage + +const settled = (promise: Promise): Promise<'settled' | 'pending'> => + Promise.race([ + promise.then( + () => 'settled' as const, + () => 'settled' as const + ), + new Promise<'pending'>((resolve) => setTimeout(() => resolve('pending'), 100)) + ]) + +describe('claude user message queue', () => { + it('rejects the frame the SDK pulled but abandoned without writing', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + await pump.next() + await pump.return?.(undefined) + + await expect(settled(sent)).resolves.toBe('settled') + await expect(sent).rejects.toThrow( + 'claude stream-json input ended before the frame was written' + ) + }) + + it('rejects an in-flight frame from fail() when the SDK never resumes the pump', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + await pump.next() + queue.fail(new Error('claude stream-json exited: child died')) + + await expect(settled(sent)).resolves.toBe('settled') + await expect(sent).rejects.toThrow('claude stream-json exited: child died') + }) + + it('still settles a written frame only once the pump asks for the next one', async () => { + const queue = createClaudeUserMessageQueue() + const pump = queue.messages[Symbol.asyncIterator]() + const sent = queue.push(frame('hello')) + + const pulled = await pump.next() + expect(pulled.value).toMatchObject({ type: 'user' }) + // The write proof is the pump coming back for more, exactly as before. + await expect(settled(sent)).resolves.toBe('pending') + void pump.next() + await expect(sent).resolves.toBeUndefined() + }) +}) diff --git a/src/main/claude/claude-agent-sdk-user-message-queue.ts b/src/main/claude/claude-agent-sdk-user-message-queue.ts new file mode 100644 index 00000000000..87fa6660159 --- /dev/null +++ b/src/main/claude/claude-agent-sdk-user-message-queue.ts @@ -0,0 +1,100 @@ +import type { SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' + +type QueuedMessage = { + message: SDKUserMessage + resolve: () => void + reject: (error: Error) => void +} + +export type ClaudeUserMessageQueue = { + /** The SDK's streaming-input prompt; it stays open until `end`. */ + messages: AsyncIterable + /** Resolves once the SDK has finished writing the frame to the child. */ + push: (message: SDKUserMessage) => Promise + /** Reject every unwritten frame, in-flight included; a caller waiting on a send must not hang past the exit. */ + fail: (error: Error) => void + end: () => void +} + +/** The rejection an abandoned frame carries when nothing else has named a cause yet. */ +const UNWRITTEN_FRAME_MESSAGE = 'claude stream-json input ended before the frame was written' + +export function createClaudeUserMessageQueue(): ClaudeUserMessageQueue { + const queued: QueuedMessage[] = [] + // The frame the SDK has taken but not yet acknowledged. It is out of `queued`, + // so it is unreachable from anywhere else and would otherwise never settle. + let inFlight: QueuedMessage | null = null + let wake: (() => void) | null = null + let ended = false + let failure: Error | null = null + const notify = (): void => { + wake?.() + wake = null + } + const rejectInFlight = (error: Error): void => { + const abandoned = inFlight + inFlight = null + abandoned?.reject(error) + } + + async function* drain(): AsyncGenerator { + for (;;) { + const next = queued.shift() + if (next) { + inFlight = next + let written = false + try { + yield next.message + written = true + } finally { + // The SDK's input pump abandons this iterator when its + // `await transport.write(...)` rejects or the query aborts, and the code + // after a `yield` never runs on that path. Settling here is the only + // place a frame it already took can be reached. + if (written) { + inFlight = null + // Resumed only after the SDK's `await transport.write(...)` settled, so this + // is the same "the frame reached the child" proof the hand-rolled write gave. + next.resolve() + } else { + rejectInFlight(failure ?? new Error(UNWRITTEN_FRAME_MESSAGE)) + } + } + continue + } + if (ended || failure) { + return + } + await new Promise((resolve) => { + wake = resolve + }) + } + } + + return { + messages: drain(), + push: (message) => + new Promise((resolve, reject) => { + if (failure) { + reject(failure) + return + } + queued.push({ message, resolve, reject }) + notify() + }), + fail: (error) => { + failure ??= error + for (const entry of queued.splice(0)) { + entry.reject(error) + } + // A pump that never resumes cannot run the generator's cleanup, so the + // exit path has to reach the in-flight frame itself. + rejectInFlight(error) + notify() + }, + end: () => { + ended = true + notify() + } + } +} diff --git a/src/main/claude/claude-child-exit-proof-ladder.ts b/src/main/claude/claude-child-exit-proof-ladder.ts new file mode 100644 index 00000000000..85ed629f1b9 --- /dev/null +++ b/src/main/claude/claude-child-exit-proof-ladder.ts @@ -0,0 +1,41 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import { waitForProcessExitUntil } from '../codex/codex-process-exit-deadline' +import type { ClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' + +const GRACEFUL_EXIT_MS = 1_500 +const FORCED_EXIT_MS = 1_000 + +export type ClaudeChildExitProofInput = { + child: Pick + exitPromise: Promise + exited: () => boolean + tree?: ClaudeChildTreeReaper +} + +export async function proveClaudeChildExitWithReaper( + input: ClaudeChildExitProofInput, + createTree: () => ClaudeChildTreeReaper +): Promise { + const tree = input.tree ?? createTree() + // Arm before stdin closes: only a live root can identify its descendants. + await tree.capture() + try { + input.child.stdin?.end() + } catch { + // The reap below still owns the process. + } + let reaped = false + if (!input.exited()) { + await waitForProcessExitUntil(input.exitPromise, GRACEFUL_EXIT_MS) + if (!input.exited()) { + reaped = true + await tree.refresh?.() + await tree.reap() + await waitForProcessExitUntil(input.exitPromise, FORCED_EXIT_MS) + } + } + if (!reaped && input.exited() && tree.treeVerdict !== 'exited') { + await tree.reap() + } + return input.exited() && tree.treeVerdict === 'exited' +} diff --git a/src/main/claude/claude-child-process-environment.test.ts b/src/main/claude/claude-child-process-environment.test.ts new file mode 100644 index 00000000000..8299fdcdd61 --- /dev/null +++ b/src/main/claude/claude-child-process-environment.test.ts @@ -0,0 +1,63 @@ +import { describe, expect, it } from 'vitest' +import { applyClaudeEnvPatch } from '../claude-accounts/environment' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' + +describe('Claude child process environment', () => { + it('strips case-insensitive auth headers through the shared env patch on Windows', () => { + expect( + applyClaudeEnvPatch( + { + anthropic_api_key: 'inherited-key', + Anthropic_Custom_Headers: 'Authorization: inherited', + SAFE_VALUE: 'preserved' + }, + {}, + { stripAuthEnv: true, platform: 'win32' } + ) + ).toEqual({ SAFE_VALUE: 'preserved' }) + }) + + it('strips case-insensitive inherited auth and session stamps on Windows', () => { + const env = buildClaudeChildProcessEnv( + { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + Claude_Code_Session_Id: 'configured-session' + }, + { + platform: 'win32', + inheritedEnv: { + anthropic_api_key: 'inherited-key', + Anthropic_Custom_Headers: 'Authorization: inherited', + claude_code_child_session: '1', + CLAUDE_CODE_SESSION_ID: 'inherited-session', + SAFE_VALUE: 'preserved' + } + } + ) + + expect(env).toEqual({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + Claude_Code_Session_Id: 'configured-session', + SAFE_VALUE: 'preserved' + }) + }) + + it('can strip child-session stamps reintroduced by a full SDK launch overlay', () => { + expect( + buildClaudeChildProcessEnv( + { + CLAUDE_CODE_CHILD_SESSION: 'configured-child-session', + CLAUDE_CODE_SESSION_ID: 'configured-session', + CLAUDE_CODE_BRIDGE_SESSION_ID: 'configured-bridge-session' + }, + { + scrubConfiguredChildSessionStamps: true, + inheritedEnv: { + CLAUDE_CODE_CHILD_SESSION: 'inherited-child-session', + SAFE_VALUE: 'preserved' + } + } + ) + ).toEqual({ SAFE_VALUE: 'preserved' }) + }) +}) diff --git a/src/main/claude/claude-child-process-environment.ts b/src/main/claude/claude-child-process-environment.ts new file mode 100644 index 00000000000..58f00c5c1b8 --- /dev/null +++ b/src/main/claude/claude-child-process-environment.ts @@ -0,0 +1,69 @@ +import { CLAUDE_AUTH_ENV_VARS, applyClaudeEnvPatch } from '../claude-accounts/environment' + +const CLAUDE_CHILD_SESSION_STAMP_ENV_KEYS = [ + 'CLAUDE_CODE_CHILD_SESSION', + 'CLAUDE_CODE_SESSION_ID', + 'CLAUDE_CODE_BRIDGE_SESSION_ID' +] as const + +function cloneProcessEnv(source: NodeJS.ProcessEnv): Record { + const env: Record = {} + for (const [key, value] of Object.entries(source)) { + if (value !== undefined) { + env[key] = value + } + } + return env +} + +function stripClaudeChildSessionStamps( + env: Record, + platform: NodeJS.Platform +): Record { + for (const key of CLAUDE_CHILD_SESSION_STAMP_ENV_KEYS) { + for (const envKey of Object.keys(env)) { + if (envKey === key || (platform === 'win32' && envKey.toUpperCase() === key)) { + delete env[envKey] + } + } + } + return env +} + +export function buildClaudeChildProcessEnv( + configuredEnv: Record = {}, + options: { + inheritedEnv?: NodeJS.ProcessEnv + platform?: NodeJS.Platform + scrubConfiguredChildSessionStamps?: boolean + } = {} +): Record { + const inheritedEnv = options.inheritedEnv ?? process.env + const platform = options.platform ?? process.platform + const env = applyClaudeEnvPatch( + cloneProcessEnv(inheritedEnv), + {}, + { + stripAuthEnv: true, + platform + } + ) + if (platform === 'win32') { + const authKeys = new Set(CLAUDE_AUTH_ENV_VARS.map((key) => key.toUpperCase())) + for (const [key, value] of Object.entries(env)) { + const normalized = key.toUpperCase() + if ( + authKeys.has(normalized) || + (normalized === 'ANTHROPIC_CUSTOM_HEADERS' && + /authorization|x-api-key|api-key|bearer/i.test(value)) + ) { + delete env[key] + } + } + } + if (options.scrubConfiguredChildSessionStamps) { + return stripClaudeChildSessionStamps({ ...env, ...configuredEnv }, platform) + } + stripClaudeChildSessionStamps(env, platform) + return { ...env, ...configuredEnv } +} diff --git a/src/main/claude/claude-child-root-termination.ts b/src/main/claude/claude-child-root-termination.ts new file mode 100644 index 00000000000..bed422532e1 --- /dev/null +++ b/src/main/claude/claude-child-root-termination.ts @@ -0,0 +1,54 @@ +import type { SpawnedProcess } from '../../shared/child-process/run-process' +import type { PosixProcessIdentity } from '../pty-descendant-termination' +import type { + WindowsDescendantSnapshot, + WindowsProcessIdentity +} from '../windows-descendant-exit-verification' + +export type ClaudeRootIdentity = PosixProcessIdentity | WindowsProcessIdentity + +type RootTerminationInput = { + child: Pick + exited: () => boolean +} + +/** + * Kills the root through the handle Node owns rather than through its pid, which + * is why no identity probe gates it: libuv drops that handle in the same turn it + * reaps, so the signal either reaches the process Orca spawned or reaches + * nothing. A probe here could only let an unreadable process table cost the tree + * the one fallback that still works once every table read has failed. + * + * False means no signal was sent, because the root had already left. + */ +export function terminateClaudeRoot(input: RootTerminationInput): boolean { + return input.exited() ? false : input.child.kill('SIGKILL') +} + +type WindowsRootTerminationInput = { + snapshot: WindowsDescendantSnapshot | null + exited: () => boolean + verifyRoot: (root: WindowsProcessIdentity) => Promise + terminateTree: (root: WindowsProcessIdentity) => Promise + killRoot: () => boolean +} + +/** + * `taskkill /T /F` addresses a bare pid, so a dead root's pid may already belong + * to a stranger whose whole tree it would take down: that one is identity-gated. + * The direct root kill after it runs however the probe decided. + */ +export async function terminateClaudeWindowsRoot( + input: WindowsRootTerminationInput +): Promise<{ rootVerified: boolean }> { + const { snapshot, exited, verifyRoot, terminateTree, killRoot } = input + let rootVerified = false + if (!exited() && snapshot) { + rootVerified = await verifyRoot(snapshot.root).catch(() => false) + if (rootVerified && !exited()) { + await terminateTree(snapshot.root).catch(() => {}) + } + } + killRoot() + return { rootVerified } +} diff --git a/src/main/claude/claude-child-tree-snapshot.ts b/src/main/claude/claude-child-tree-snapshot.ts new file mode 100644 index 00000000000..e0955648b02 --- /dev/null +++ b/src/main/claude/claude-child-tree-snapshot.ts @@ -0,0 +1,128 @@ +import type { DescendantSnapshot } from '../pty-descendant-termination' +import type { WindowsDescendantSnapshot } from '../windows-descendant-exit-verification' + +/** One platform's descendant tree, tagged so neither verifier can be handed the other's rows. */ +export type ClaudeCapturedTree = + | { platform: 'posix'; tree: DescendantSnapshot } + | { platform: 'win32'; tree: WindowsDescendantSnapshot } + +/** + * Process-table reads are not atomic: a refresh can omit a still-live row, but + * it can also observe a new process after the old row exited. Retain rows absent + * from the refresh, but reject a PID whose identity changed between reads. + */ +function mergeRowsByPid( + previous: readonly Row[], + next: readonly Row[], + sameIdentity: (previous: Row, next: Row) => boolean, + previousBoundary: (row: Row) => number, + nextBoundary: (row: Row) => number, + refreshBoundary: number +): { rows: Row[]; capturedAtMsByPid?: Readonly> } | null { + const merged = new Map() + const capturedAtMsByPid: Record = {} + for (const row of previous) { + const prior = merged.get(row.pid) + if (prior && !sameIdentity(prior, row)) { + return null + } + merged.set(row.pid, row) + capturedAtMsByPid[String(row.pid)] = previousBoundary(row) + } + for (const row of next) { + const prior = merged.get(row.pid) + if (prior && !sameIdentity(prior, row)) { + return null + } + if (!prior) { + capturedAtMsByPid[String(row.pid)] = nextBoundary(row) + } + merged.set(row.pid, row) + } + const boundaries = Object.values(capturedAtMsByPid) + const needsBoundaryMap = + new Set(boundaries).size > 1 || boundaries.some((boundary) => boundary !== refreshBoundary) + return { + rows: [...merged.values()], + ...(needsBoundaryMap ? { capturedAtMsByPid } : {}) + } +} + +export function mergeClaudeCapturedTrees( + previous: ClaudeCapturedTree, + next: ClaudeCapturedTree +): ClaudeCapturedTree | null { + if (previous.platform !== next.platform) { + return null + } + if (previous.platform === 'posix' && next.platform === 'posix') { + if (previous.tree.rootPgid !== next.tree.rootPgid) { + return null + } + // A refresh cannot repair an earlier capture that lacked root identity; + // retaining those rows would permit a later numeric-pid kill without proof. + if (!previous.tree.root || !next.tree.root) { + return null + } + if ( + previous.tree.root.pid !== next.tree.root.pid || + previous.tree.root.startedAt !== next.tree.root.startedAt + ) { + return null + } + const descendants = mergeRowsByPid( + previous.tree.descendants, + next.tree.descendants, + (left, right) => left.pgid === right.pgid && left.startedAt === right.startedAt, + (row) => previous.tree.capturedAtMsByPid?.[String(row.pid)] ?? previous.tree.capturedAtMs, + (row) => next.tree.capturedAtMsByPid?.[String(row.pid)] ?? next.tree.capturedAtMs, + next.tree.capturedAtMs + ) + if (!descendants) { + return null + } + return { + platform: 'posix', + tree: { + ...next.tree, + // Retained rows keep their earlier boundary; new rows use the refresh + // boundary. The scalar remains the latest scan for legacy consumers. + descendants: descendants.rows, + ...(descendants.capturedAtMsByPid + ? { capturedAtMsByPid: descendants.capturedAtMsByPid } + : {}) + } + } + } + if (previous.platform === 'win32' && next.platform === 'win32') { + if ( + previous.tree.root.pid !== next.tree.root.pid || + previous.tree.root.creationTimeMs !== next.tree.root.creationTimeMs + ) { + return null + } + const descendants = mergeRowsByPid( + previous.tree.descendants, + next.tree.descendants, + (left, right) => left.creationTimeMs === right.creationTimeMs, + (row) => previous.tree.capturedAtMsByPid?.[String(row.pid)] ?? previous.tree.capturedAtMs, + (row) => next.tree.capturedAtMsByPid?.[String(row.pid)] ?? next.tree.capturedAtMs, + next.tree.capturedAtMs + ) + if (!descendants) { + return null + } + return { + platform: 'win32', + tree: { + ...next.tree, + descendants: descendants.rows, + ...(descendants.capturedAtMsByPid + ? { capturedAtMsByPid: descendants.capturedAtMsByPid } + : {}), + unidentifiedCount: Math.max(previous.tree.unidentifiedCount, next.tree.unidentifiedCount) + } + } + } + return null +} diff --git a/src/main/claude/claude-command-lifecycle-frames.test.ts b/src/main/claude/claude-command-lifecycle-frames.test.ts new file mode 100644 index 00000000000..8126a546ed8 --- /dev/null +++ b/src/main/claude/claude-command-lifecycle-frames.test.ts @@ -0,0 +1,115 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: () => {}, + publish: vi.fn() + } + return { sink, items } +} + +function providerFrameKinds(items: { body: AgentJournalItemBody }[]): string[] { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) +} + +/** + * The queue-bookkeeping frame Claude Code 2.1.258 emits for every uuid-stamped + * command: `command_uuid` plus a state, and no content of its own. Shape and + * states taken from the CLI's own emission sites. + */ +function commandLifecycle(state: 'started' | 'completed' | 'cancelled', uuid: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'command_lifecycle', + command_uuid: 'command-1', + state, + uuid, + session_id: 'claude-session' + } + } +} + +function userTurn(uuid: string, text: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: text } + } + } +} + +function assistantReply(uuid: string, text: string) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { role: 'assistant', content: [{ type: 'text', text }] } + } + } +} + +describe('Claude command_lifecycle frames', () => { + it('keeps queue bookkeeping off the transcript for a whole turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userTurn('user-1', 'Reply with exactly PROBE_OK and nothing else.')) + translator.handle(commandLifecycle('started', 'lifecycle-1')) + translator.handle(assistantReply('assistant-1', 'PROBE_OK')) + translator.handle(commandLifecycle('completed', 'lifecycle-2')) + translator.handle(commandLifecycle('completed', 'lifecycle-3')) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { + type: 'result', + subtype: 'success', + uuid: 'result-1', + session_id: 'claude-session', + is_error: false, + result: 'PROBE_OK', + terminal_reason: 'completed' + } + }) + + expect(providerFrameKinds(state.items)).toEqual([]) + // The turn's real content is untouched. + expect( + state.items.flatMap((item) => + item.body.kind === 'message' && item.body.role === 'assistant' ? [item.body.blocks] : [] + ) + ).toEqual([[{ type: 'text', text: 'PROBE_OK' }]]) + }) + + it('keeps a cancelled command off the transcript too', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(commandLifecycle('cancelled', 'lifecycle-4')) + + expect(providerFrameKinds(state.items)).toEqual([]) + }) +}) diff --git a/src/main/claude/claude-config-dir-pin.test.ts b/src/main/claude/claude-config-dir-pin.test.ts new file mode 100644 index 00000000000..c7be40a6f38 --- /dev/null +++ b/src/main/claude/claude-config-dir-pin.test.ts @@ -0,0 +1,34 @@ +import { homedir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { claudeConfigDirEnvPatch, defaultClaudeConfigDir } from './claude-config-dir-pin' + +describe('claude config dir pin', () => { + it('does not pin the CLI default home, so the macOS Keychain stays reachable', () => { + expect(claudeConfigDirEnvPatch(join(homedir(), '.claude'), { env: {} })).toEqual({}) + expect(claudeConfigDirEnvPatch(`${join(homedir(), '.claude')}/`, { env: {} })).toEqual({}) + expect(claudeConfigDirEnvPatch(' ', { env: {} })).toEqual({}) + }) + + it('pins a managed account home the CLI would not find on its own', () => { + expect(claudeConfigDirEnvPatch('/accounts/claude/managed', { env: {} })).toEqual({ + CLAUDE_CONFIG_DIR: '/accounts/claude/managed' + }) + }) + + it('treats an inherited CLAUDE_CONFIG_DIR as the default the CLI already resolves', () => { + const env = { CLAUDE_CONFIG_DIR: '/inherited/home' } + expect(defaultClaudeConfigDir(env)).toBe('/inherited/home') + expect(claudeConfigDirEnvPatch('/inherited/home', { env })).toEqual({}) + expect(claudeConfigDirEnvPatch('/other/home', { env })).toEqual({ + CLAUDE_CONFIG_DIR: '/other/home' + }) + }) + + it('compares Windows homes case-insensitively', () => { + const env = { CLAUDE_CONFIG_DIR: 'C:\\Users\\Work\\.claude' } + expect(claudeConfigDirEnvPatch('c:\\users\\work\\.claude', { env, platform: 'win32' })).toEqual( + {} + ) + }) +}) diff --git a/src/main/claude/claude-config-dir-pin.ts b/src/main/claude/claude-config-dir-pin.ts new file mode 100644 index 00000000000..d5cc0b8b186 --- /dev/null +++ b/src/main/claude/claude-config-dir-pin.ts @@ -0,0 +1,37 @@ +import { homedir } from 'node:os' +import { join, resolve } from 'node:path' + +/** The config dir the Claude CLI resolves for itself when nothing pins one. */ +export function defaultClaudeConfigDir(env: NodeJS.ProcessEnv = process.env): string { + return env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') +} + +function samePath(a: string, b: string, platform: NodeJS.Platform): boolean { + const left = resolve(a) + const right = resolve(b) + return platform === 'win32' ? left.toLowerCase() === right.toLowerCase() : left === right +} + +/** + * An explicit CLAUDE_CONFIG_DIR moves the Claude CLI off the default Keychain item onto + * one derived from the pinned path, so a claude.ai OAuth login stops working even when + * the pin names the CLI's own default. Pin only a home the CLI would not find on its + * own — the same rule the legacy PTY path applies via `ClaudeRuntimePathResolver`. + * + * The pinned value is the account home verbatim: the CLI keys its credential lookup on + * the literal string, so re-spelling an equivalent path (absolute vs `~`, trailing + * separator) selects a different identity. Normalization here is for the equality test + * only and must never reach the env. + */ +export function claudeConfigDirEnvPatch( + accountHome: string, + options: { env?: NodeJS.ProcessEnv; platform?: NodeJS.Platform } = {} +): { CLAUDE_CONFIG_DIR?: string } { + const env = options.env ?? process.env + const platform = options.platform ?? process.platform + const resolved = accountHome.trim() + if (!resolved || samePath(resolved, defaultClaudeConfigDir(env), platform)) { + return {} + } + return { CLAUDE_CONFIG_DIR: resolved } +} diff --git a/src/main/claude/claude-descendant-escalation-boundary.test.ts b/src/main/claude/claude-descendant-escalation-boundary.test.ts new file mode 100644 index 00000000000..55459df0c7f --- /dev/null +++ b/src/main/claude/claude-descendant-escalation-boundary.test.ts @@ -0,0 +1,124 @@ +import { describe, expect, it, vi } from 'vitest' +import { terminateDescendantSnapshotWithVerdict } from '../pty-descendant-exit-verification' +import { + collectDescendantRows, + type DescendantSnapshot, + type ProcessTableRow +} from '../pty-descendant-termination' +import { createClaudeChildTreeReaper } from './claude-agent-sdk-exit-proof' + +const ROOT_PID = 500 +const ORCA_PGID = 400 +const ROOT_STARTED_AT = 'Thu Sep 3 18:04:50 2026' +/** The second both close-time walks land in. */ +const WALK_SECOND = 'Thu Sep 3 18:05:04 2026' +const WALK_MS = Date.parse(WALK_SECOND) +const EARLIER_SECOND = 'Thu Sep 3 18:05:03 2026' + +/** The measured split: `s20` at :03.946 died, `s21` at :04.042 leaked. */ +const EARLIER_BORN = [700, 701, 702] +const WALK_SECOND_BORN = [721, 722, 723, 724] + +type Cohort = { pids: number[]; startedAt: string } + +const LIVE_TREE: Cohort[] = [ + { pids: EARLIER_BORN, startedAt: EARLIER_SECOND }, + { pids: WALK_SECOND_BORN, startedAt: WALK_SECOND } +] + +function rowsFor(cohorts: Cohort[]): ProcessTableRow[] { + return [ + { pid: ROOT_PID, ppid: 1, pgid: ORCA_PGID, startedAt: ROOT_STARTED_AT }, + ...cohorts.flatMap((cohort) => + cohort.pids.map((pid) => ({ + pid, + ppid: ROOT_PID, + pgid: ORCA_PGID, + startedAt: cohort.startedAt + })) + ) + ] +} + +/** A real ppid walk from the root, exactly as production captures one. */ +function walk(capturedAtMs: number, cohorts: Cohort[] = LIVE_TREE): DescendantSnapshot { + return collectDescendantRows(ROOT_PID, rowsFor(cohorts), capturedAtMs) +} + +function killedPids(calls: [number, NodeJS.Signals][]): number[] { + return calls.flatMap(([pid, signal]) => (signal === 'SIGKILL' ? [pid] : [])).sort((a, b) => a - b) +} + +function signalledPids(calls: [number, NodeJS.Signals][]): number[] { + return calls.flatMap(([pid, signal]) => (signal === 'SIGTERM' ? [pid] : [])).sort((a, b) => a - b) +} + +/** + * Drives the real reaper and the real verifier against a process table where + * every descendant traps SIGTERM, so only a forced sweep can end them. The root + * is alive for both walks and gone by the sweep, which is the measured teardown. + */ +async function sweep( + captures: DescendantSnapshot[], + liveTree: Cohort[] = LIVE_TREE +): Promise<[number, NodeJS.Signals][]> { + const calls: [number, NodeJS.Signals][] = [] + const captureDescendants = vi.fn() + for (const capture of captures) { + captureDescendants.mockResolvedValueOnce(capture) + } + const tree = createClaudeChildTreeReaper( + { pid: ROOT_PID, kill: vi.fn(() => true) }, + { + platform: 'linux', + exited: () => false, + captureDescendants, + terminateDescendants: (snapshot) => + terminateDescendantSnapshotWithVerdict(snapshot, { + requireIdentityBeforeSignal: true, + graceMs: 0, + verifyMs: 120, + sendSignal: (pid, signal) => calls.push([pid, signal]), + readTable: async () => ({ rows: rowsFor(liveTree), capturedAtMs: Date.now() }) + }) + } + ) + // The close ladder's shape: arm, then re-walk the live root at the boundary. + await tree.capture() + await tree.refresh?.() + await tree.reap() + return calls +} + +describe('Claude descendant forced-sweep fence', () => { + it('escalates a descendant forked in the same second as both close walks', async () => { + // Both walks land inside second :04, one ps duration apart, and the root is + // gone before a third could run. A descendant born at :04.042 is no less + // ours than its sibling born 96ms earlier at :03.946. + const calls = await sweep([walk(WALK_MS + 42), walk(WALK_MS + 140)]) + + expect(signalledPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + expect(killedPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + }) + + it('still escalates descendants born before the walk that first saw them', async () => { + const onlyEarlier = [{ pids: EARLIER_BORN, startedAt: EARLIER_SECOND }] + const calls = await sweep([walk(WALK_MS + 42, onlyEarlier)], onlyEarlier) + + expect(killedPids(calls)).toEqual(EARLIER_BORN) + }) + + it('withholds the sweep from a row no walk re-derived, on its start second alone', async () => { + // 900 was seen once, in its own birth second, and the refresh did not find + // it. The merge retains the row, but nothing re-proved it belongs to us, so + // the second-resolution fence is all there is and it still says no. + const retained = { pids: [900], startedAt: WALK_SECOND } + const firstWalk = walk(WALK_MS + 42, [...LIVE_TREE, retained]) + const refresh = walk(WALK_MS + 140) + + const calls = await sweep([firstWalk, refresh], [...LIVE_TREE, retained]) + + expect(signalledPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN, 900]) + expect(killedPids(calls)).toEqual([...EARLIER_BORN, ...WALK_SECOND_BORN]) + }) +}) diff --git a/src/main/claude/claude-stream-json-connection-close.test.ts b/src/main/claude/claude-stream-json-connection-close.test.ts new file mode 100644 index 00000000000..e824139a776 --- /dev/null +++ b/src/main/claude/claude-stream-json-connection-close.test.ts @@ -0,0 +1,126 @@ +import { EventEmitter } from 'node:events' +import { PassThrough } from 'node:stream' +import type { ChildProcessWithoutNullStreams } from 'node:child_process' +import { describe, expect, it, vi } from 'vitest' +import type { query } from '@anthropic-ai/claude-agent-sdk' +import { + openClaudeStreamJsonConnection, + type ClaudeStreamJsonLaunch +} from './claude-stream-json-connection' + +const mocks = vi.hoisted(() => { + const refresh = vi.fn() + const proveClaudeChildExit = vi.fn() + const tree = { + capture: vi.fn(async () => {}), + refresh: (...args: unknown[]) => refresh(...args), + reap: vi.fn(async () => 'exited' as const), + treeVerdict: 'unverifiable' as const + } + return { proveClaudeChildExit, refresh, tree } +}) + +vi.mock('./claude-agent-sdk-exit-proof', () => ({ + createClaudeChildTreeReaper: vi.fn(() => mocks.tree), + proveClaudeChildExit: (...args: unknown[]) => mocks.proveClaudeChildExit(...args) +})) + +function fakeChild(): ChildProcessWithoutNullStreams { + const child = new EventEmitter() + return Object.assign(child, { + pid: 424242, + stdin: new PassThrough(), + stdout: new PassThrough(), + stderr: new PassThrough(), + kill: vi.fn() + }) as unknown as ChildProcessWithoutNullStreams +} + +describe('Claude stream-json close ordering', () => { + it('waits for the live tree refresh before ending stdin', async () => { + const refreshDone = Promise.withResolvers() + mocks.refresh.mockReturnValueOnce(refreshDone.promise) + mocks.proveClaudeChildExit.mockResolvedValueOnce(true) + const child = fakeChild() + const launch: ClaudeStreamJsonLaunch = { + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo' + } + const queryImpl = ((params: Parameters[0]) => { + if (!params.options) { + throw new Error('missing SDK options') + } + params.options.spawnClaudeCodeProcess?.({ + command: 'claude', + args: [], + env: {}, + signal: new AbortController().signal + }) + void (async () => { + for await (const _message of params.prompt) { + // The SDK owns the transport write; the close test only needs its EOF boundary. + } + child.stdin.end() + })() + return (async function* () {})() + }) as typeof query + const connection = await openClaudeStreamJsonConnection(launch, {}, () => child, queryImpl) + + const closing = connection.close() + await new Promise((resolve) => setImmediate(resolve)) + expect(child.stdin.writableEnded).toBe(false) + + refreshDone.resolve() + await expect(closing).resolves.toBe(true) + expect(child.stdin.writableEnded).toBe(true) + }) + + it('requests a fresh close boundary after an output capture starts', async () => { + mocks.refresh.mockReset() + mocks.proveClaudeChildExit.mockReset() + const outputCapture = Promise.withResolvers() + const closeCapture = Promise.withResolvers() + mocks.refresh + .mockReturnValueOnce(outputCapture.promise) + .mockReturnValueOnce(closeCapture.promise) + mocks.proveClaudeChildExit.mockResolvedValueOnce(true) + const child = fakeChild() + const launch: ClaudeStreamJsonLaunch = { + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo' + } + const queryImpl = ((params: Parameters[0]) => { + params.options?.spawnClaudeCodeProcess?.({ + command: 'claude', + args: [], + env: {}, + signal: new AbortController().signal + }) + void (async () => { + for await (const _message of params.prompt) { + // The SDK owns the transport write; the close test only needs its EOF boundary. + } + child.stdin.end() + })() + return (async function* () {})() + }) as typeof query + const connection = await openClaudeStreamJsonConnection(launch, {}, () => child, queryImpl) + + child.stderr.emit('data', 'output') + await vi.waitFor(() => expect(mocks.refresh).toHaveBeenCalledTimes(1)) + const closing = connection.close() + await Promise.resolve() + + expect(mocks.refresh).toHaveBeenCalledTimes(2) + expect(child.stdin.writableEnded).toBe(false) + + outputCapture.resolve() + await Promise.resolve() + expect(child.stdin.writableEnded).toBe(false) + closeCapture.resolve() + await expect(closing).resolves.toBe(true) + expect(child.stdin.writableEnded).toBe(true) + }) +}) diff --git a/src/main/claude/claude-stream-json-connection.test.ts b/src/main/claude/claude-stream-json-connection.test.ts new file mode 100644 index 00000000000..c4f1f9a6fca --- /dev/null +++ b/src/main/claude/claude-stream-json-connection.test.ts @@ -0,0 +1,768 @@ +import { execFileSync } from 'node:child_process' +import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { spawnProcess, type SpawnedProcess } from '../../shared/child-process/run-process' +import { hasLiveClaudePtys } from '../claude-accounts/live-pty-gate' +import type { ProcessSpec } from '../../shared/child-process/process-spec' +import { query, type CanUseTool, type Options } from '@anthropic-ai/claude-agent-sdk' +import { + openClaudeStreamJsonConnection, + type ClaudeStreamJsonConnection, + type ClaudeStreamJsonLaunch +} from './claude-stream-json-connection' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createDeferredStructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { claudeAuthDiagnostic } from './claude-structured-init-proof' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import { readClaudeStructuredSessionOptions } from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' + +// These drive the real SDK against the scripted fake CLI, so every assertion is +// about the environment, argv and frames a real child actually saw. +const FAKE_CLI = join(__dirname, '__fixtures__', 'claude-agent-sdk-scripted-cli.mjs') +const SESSION_ID = '5348c19f-6a54-4c2e-9c68-9c2b1a3d4e5f' +const HOLD_OPEN = { delayMs: 10_000 } + +type ScriptedCliReport = { + argv: string[] + controlRequests: { request_id: string; request: { subtype: string } }[] + controlResponses: { response: { request_id: string; response?: unknown } }[] + userMessages: Record[] + descendantPid: number | null +} + +const scratchDirs: string[] = [] +const openConnections: ClaudeStreamJsonConnection[] = [] + +afterEach(async () => { + for (const connection of openConnections.splice(0)) { + await connection.close() + } + for (const dir of scratchDirs.splice(0)) { + rmSync(dir, { recursive: true, force: true }) + } + spawned.splice(0) + spawnedChildren.splice(0) + vi.unstubAllEnvs() +}) + +function scriptScenario( + steps: Record[], + controlResponses: Record = {} +) { + const dir = mkdtempSync(join(tmpdir(), 'claude-sdk-connection-')) + scratchDirs.push(dir) + const scenarioPath = join(dir, 'scenario.json') + const reportPath = join(dir, 'report.json') + writeFileSync(scenarioPath, JSON.stringify({ steps, controlResponses })) + return { + cwd: dir, + env: { + PATH: process.env.PATH ?? '', + ORCA_SDK_CONTRACT_SCENARIO_PATH: scenarioPath, + ORCA_SDK_CONTRACT_REPORT_PATH: reportPath + }, + readReport: () => JSON.parse(readFileSync(reportPath, 'utf8')) as ScriptedCliReport + } +} + +function launchFor( + scenario: { cwd: string; env: Record }, + env: Record = {} +): ClaudeStreamJsonLaunch { + return { + pathToClaudeCodeExecutable: FAKE_CLI, + options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, sessionId: SESSION_ID }, + cwd: scenario.cwd, + env: { ...scenario.env, ...env } + } +} + +/** The derived child environment, captured where Orca actually hands it to the OS. */ +const spawned: ProcessSpec[] = [] +/** The retained child, so a test can end it the way a crashing CLI would. */ +const spawnedChildren: SpawnedProcess[] = [] + +async function open( + launch: ClaudeStreamJsonLaunch, + handlers: Parameters[1] = {}, + queryImpl?: typeof query +): Promise { + const connection = await openClaudeStreamJsonConnection( + launch, + handlers, + (spec) => { + spawned.push(spec) + const child = spawnProcess(spec) + spawnedChildren.push(child) + return child + }, + queryImpl + ) + openConnections.push(connection) + return connection +} + +function childEnv(): Record { + return (spawned.at(-1)?.env ?? {}) as Record +} + +async function until(read: () => T | null | undefined, label: string): Promise { + for (let attempt = 0; attempt < 400; attempt++) { + const value = read() + if (value !== null && value !== undefined) { + return value + } + await new Promise((resolve) => setTimeout(resolve, 25)) + } + throw new Error(`timed out waiting for ${label}`) +} + +function readReportSafely(scenario: { readReport: () => ScriptedCliReport }) { + try { + return scenario.readReport() + } catch { + return null + } +} + +function processState(pid: number): 'running' | 'exited' { + try { + const state = execFileSync('ps', ['-o', 'state=', '-p', String(pid)], { + encoding: 'utf8', + env: { ...process.env, LANG: 'C', LC_ALL: 'C' } + }).trim() + return state.startsWith('Z') ? 'exited' : 'running' + } catch (error) { + if ((error as { status?: number }).status === 1) { + return 'exited' + } + throw error + } +} + +describe('Claude stream-json connection', () => { + it('passes the Claude Code system-prompt preset through to SDK query', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + let captured: Options | undefined + await open(launchFor(scenario), {}, (params) => { + captured = params.options + return query(params) + }) + + expect(captured?.systemPrompt).toEqual({ type: 'preset', preset: 'claude_code' }) + }) + + it('hands the child a derived environment, the resolved CLI path, and keeps the pid', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + vi.stubEnv('CLAUDE_CODE_CHILD_SESSION', '1') + vi.stubEnv('NODE_OPTIONS', '--require=/tmp/inject.js') + // An inherited value wins over the SDK's default, so clear it to pin the default. + vi.stubEnv('CLAUDE_CODE_ENTRYPOINT', undefined) + vi.stubEnv('ORCA_CONNECTION_MARKER', 'inherited') + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open( + launchFor(scenario, { + CLAUDE_CONFIG_DIR: '/accounts/managed/home', + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ORCA_AGENT_SESSION_SPAWN_TOKEN: 'spawn-9', + CLAUDE_CODE_CHILD_SESSION: 'configured-child-session', + CLAUDE_CODE_SESSION_ID: 'configured-session', + CLAUDE_CODE_BRIDGE_SESSION_ID: 'configured-bridge-session' + }) + ) + + // Ownership proof: the pid is a real live process, not a value the SDK reported. + expect(connection.pid).toEqual(expect.any(Number)) + expect(() => process.kill(connection.pid as number, 0)).not.toThrow() + const env = childEnv() + // The managed home is pinned verbatim: the CLI keys credential lookup on the literal string. + expect(env.CLAUDE_CONFIG_DIR).toBe('/accounts/managed/home') + expect(env.ANTHROPIC_AUTH_TOKEN).toBe('configured-token') + expect(env.ORCA_AGENT_SESSION_SPAWN_TOKEN).toBe('spawn-9') + expect(env.ORCA_CONNECTION_MARKER).toBe('inherited') + expect(env.ANTHROPIC_API_KEY).toBeUndefined() + expect(env.CLAUDE_CODE_CHILD_SESSION).toBeUndefined() + expect(env.CLAUDE_CODE_SESSION_ID).toBeUndefined() + expect(env.CLAUDE_CODE_BRIDGE_SESSION_ID).toBeUndefined() + // Two SDK mutations of the child env, pinned so a bump cannot change them unseen. + expect(env.CLAUDE_CODE_ENTRYPOINT).toBe('sdk-ts') + expect(env.NODE_OPTIONS).toBeUndefined() + // The bundled binary is excluded from the install, so the resolved path is mandatory. + const report = await until(() => readReportSafely(scenario), 'the scripted CLI report') + expect(report.argv[0]).toBe(FAKE_CLI) + // The .mjs fixture makes the SDK run it under node; a real CLI path is the program + // itself. Either way the resolved path is what Orca's spawner is asked to execute. + expect([spawned.at(-1)?.program, ...(spawned.at(-1)?.args ?? [])]).toContain(FAKE_CLI) + expect(report.argv).toContain('--replay-user-messages') + expect(report.argv).toContain(`--session-id=${SESSION_ID}`) + }) + + it('leaves the default CLI home unpinned so macOS Keychain OAuth keeps working', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + await open(launchFor(scenario)) + + await until(() => readReportSafely(scenario), 'the scripted CLI report') + expect(childEnv().CLAUDE_CONFIG_DIR).toBeUndefined() + }) + + it('settles a send only once the frame reached the child, and replays reach onMessage', async () => { + const replay = { + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + isReplay: true, + session_id: SESSION_ID, + uuid: 'uuid-replay-1' + } + const scenario = scriptScenario([{ awaitUserMessage: true }, { emit: replay }, HOLD_OPEN]) + const messages: Record[] = [] + const connection = await open(launchFor(scenario), { + onMessage: (message) => messages.push(message) + }) + + await connection.send({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + }) + // The report exists from the child's first line of work, so poll for the frame + // itself: `send` settles on the SDK's completed write, and the child still has + // to read that line before it can record it. + const report = await until( + () => (readReportSafely(scenario)?.userMessages.length ? readReportSafely(scenario) : null), + 'the user frame recorded by the child' + ) + expect(report.userMessages).toHaveLength(1) + + await until(() => messages.find((message) => message.uuid === 'uuid-replay-1'), 'the replay') + // The replay is delivered verbatim, so the dispatch acknowledgement still binds on it. + expect(messages.find((message) => message.uuid === 'uuid-replay-1')).toEqual(replay) + }) + + it('rejects a send the SDK pulled but could not write to a terminated child', async () => { + const scenario = scriptScenario([{ awaitUserMessage: true }, HOLD_OPEN]) + const connection = await open(launchFor(scenario)) + const child = spawnedChildren.at(-1) + + // Same tick as the send, so the liveness guard still passes and the frame + // reaches the SDK's input pump: its `transport.write` is what fails, which is + // the window a child crashing mid-send actually opens. + child?.kill('SIGKILL') + const sent = connection.send({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'hello' }] }, + parent_tool_use_id: null, + session_id: SESSION_ID + }) + + await expect(sent).rejects.toThrow() + expect(readReportSafely(scenario)?.userMessages ?? []).toHaveLength(0) + }) + + it('delivers an unmodeled frame verbatim so the provider-fallback row survives', async () => { + const unknown = { + type: 'frame_kind_from_the_future', + session_id: SESSION_ID, + uuid: 'uuid-unknown-1', + payload: { nested: { flags: ['a', 'b'] } } + } + const scenario = scriptScenario([{ emit: unknown }, HOLD_OPEN]) + const messages: Record[] = [] + await open(launchFor(scenario), { onMessage: (message) => messages.push(message) }) + + await until(() => messages.find((message) => message.uuid === 'uuid-unknown-1'), 'the frame') + expect(messages.find((message) => message.uuid === 'uuid-unknown-1')).toEqual(unknown) + }) + + it('commits the real partial-message cadence as one assistant item through the translator', async () => { + // The frame order and per-frame uuids are the ones Claude Code 2.1.258 emits + // under --include-partial-messages: every stream_event and the block's final + // assistant frame each carry their own uuid; only message.id ties them. + const stream = (uuid: string, event: Record) => ({ + type: 'stream_event', + uuid, + session_id: SESSION_ID, + parent_tool_use_id: null, + event + }) + const frames = [ + stream('uuid-message-start', { + type: 'message_start', + message: { id: 'msg_01', role: 'assistant', content: [] } + }), + stream('uuid-block-start', { + type: 'content_block_start', + index: 0, + content_block: { type: 'text', text: '' } + }), + stream('uuid-delta-1', { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: 'ST' } + }), + stream('uuid-delta-2', { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: 'REAMOK_ELEC_64E632' } + }), + { + type: 'assistant', + uuid: 'uuid-assistant-final', + session_id: SESSION_ID, + parent_tool_use_id: null, + message: { + id: 'msg_01', + role: 'assistant', + content: [{ type: 'text', text: 'STREAMOK_ELEC_64E632' }], + stop_reason: null + } + }, + stream('uuid-block-stop', { type: 'content_block_stop', index: 0 }), + stream('uuid-message-delta', { type: 'message_delta', delta: { stop_reason: 'end_turn' } }), + stream('uuid-message-stop', { type: 'message_stop' }), + { + type: 'result', + subtype: 'success', + is_error: false, + duration_ms: 1, + duration_api_ms: 1, + num_turns: 1, + result: 'STREAMOK_ELEC_64E632', + stop_reason: 'end_turn', + session_id: SESSION_ID, + uuid: 'uuid-result' + } + ] + const scenario = scriptScenario([...frames.map((frame) => ({ emit: frame })), HOLD_OPEN]) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: SESSION_ID, leafUuid: 'leaf-1' } + }, + journalDir: join(scenario.cwd, 'journal'), + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ journal, fence: 1, publish: vi.fn() }) + const translator = createClaudeJournalTranslator({ sink: deferred.sink }) + let settled = false + await open(launchFor(scenario), { + onMessage: (message) => { + translator.handle({ type: 'message', sessionId: 'session-1', message }) + settled ||= message.type === 'result' + } + }) + + await until(() => (settled ? true : null), 'the result frame') + await deferred.drained() + const items = journal.snapshot().items + const assistant = items.filter( + (item) => item.body.kind === 'message' && item.body.role === 'assistant' + ) + expect(assistant.map((item) => item.body)).toEqual([ + { + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: 'STREAMOK_ELEC_64E632' }] + } + ]) + expect(assistant.map((item) => item.itemId)).toEqual([`claude:${SESSION_ID}:uuid-block-start`]) + expect( + items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) + ).toEqual([]) + // The journal owns a SQLite connection now; afterEach removes this temp root and an open + // handle blocks that on Windows. + await journal.close() + }) + + it('feeds an inbound permission request to canUseTool and writes its answer back on the same id', async () => { + const scenario = scriptScenario([ + { + emit: { + type: 'control_request', + request_id: 'perm-421', + request: { + subtype: 'can_use_tool', + tool_name: 'Bash', + input: { command: 'ls' }, + tool_use_id: 'toolu_1', + permission_suggestions: [{ type: 'addRules' }] + } + } + }, + { awaitControlResponse: 'perm-421' }, + HOLD_OPEN + ]) + const seen: { toolName: string; requestId: string; toolUseID: string; suggestions: unknown }[] = + [] + const canUseTool: CanUseTool = (toolName, _input, options) => { + seen.push({ + toolName, + requestId: options.requestId, + toolUseID: options.toolUseID, + suggestions: options.suggestions + }) + return Promise.resolve({ behavior: 'deny', message: 'No', toolUseID: options.toolUseID }) + } + await open(launchFor(scenario), { canUseTool }) + + await until(() => (seen.length > 0 ? seen : null), 'the inbound permission request') + expect(seen).toEqual([ + { + toolName: 'Bash', + requestId: 'perm-421', + toolUseID: 'toolu_1', + suggestions: [{ type: 'addRules' }] + } + ]) + const written = await until( + () => + readReportSafely(scenario)?.controlResponses.find( + (frame) => frame.response.request_id === 'perm-421' + ), + 'the permission answer' + ) + expect(written.response.response).toMatchObject({ behavior: 'deny', message: 'No' }) + }) + + it('drives Orca control methods onto the SDK and times out with the init proof message', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { models: [{ value: 'sonnet' }], account: { tokenSource: 'oauth' } }, + get_settings: { env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } } + }) + const connection = await open(launchFor(scenario)) + + await expect(connection.initializationResult()).resolves.toMatchObject({ + models: [{ value: 'sonnet' }] + }) + await expect(connection.getSettings()).resolves.toEqual({ + env: { ANTHROPIC_BASE_URL: 'https://settings.example.test' } + }) + await expect(connection.setModel('opus')).resolves.toBeUndefined() + const requests = await until( + () => + readReportSafely(scenario)?.controlRequests.find( + (frame) => frame.request.subtype === 'set_model' + ), + 'the set_model control request' + ) + expect(requests.request.subtype).toBe('set_model') + }) + + it('reads supportedModels from the catalog the running CLI reported', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { + models: [ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { + value: 'opus', + displayName: 'Opus 5', + description: 'The live row, not the seed', + resolvedModel: 'claude-opus-5', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + } + ] + } + }) + const connection = await open(launchFor(scenario)) + + await expect(connection.supportedModels()).resolves.toMatchObject([ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { value: 'opus', displayName: 'Opus 5', supportedEffortLevels: ['low', 'high'] } + ]) + }) + + it('serves the picker the live catalog rather than falling back to the static seed', async () => { + const scenario = scriptScenario([HOLD_OPEN], { + initialize: { + models: [ + { value: 'default', resolvedModel: 'claude-opus-5' }, + { + value: 'opus', + displayName: 'Opus 5', + description: 'The live row, not the seed', + resolvedModel: 'claude-opus-5', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + } + ] + } + }) + const connection = await open(launchFor(scenario)) + const session = { + connection, + options: new Map(), + reportedOptions: {} + } as unknown as ClaudeSession + + const options = await readClaudeStructuredSessionOptions(session, 5_000) + + // The seed carries neither this description nor a two-level effort list, so + // both can only have come from the child. + expect(options.models).toContainEqual({ + id: 'opus', + label: 'Opus 5', + description: 'The live row, not the seed', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }) + expect(options.current.model).toBe('opus') + }) + + it('feeds the auth diagnostic from the settings the running child reports', async () => { + for (const key of ['ANTHROPIC_BASE_URL', 'ANTHROPIC_AUTH_TOKEN', 'ANTHROPIC_API_KEY']) { + vi.stubEnv(key, undefined) + } + const scenario = scriptScenario([HOLD_OPEN], { + get_settings: { + env: { + ANTHROPIC_BASE_URL: 'https://settings.example.test', + ANTHROPIC_AUTH_TOKEN: 'secret' + } + } + }) + const connection = await open(launchFor(scenario)) + const init = { providerSessionId: SESSION_ID, uuid: null, model: null, message: {} } + + // With no ambient auth, every true below can only have come from the CLI's settings. + expect(claudeAuthDiagnostic(init, null)).toMatchObject({ + baseUrlConfigured: false, + authTokenConfigured: false + }) + const diagnostic = claudeAuthDiagnostic(init, await connection.getSettings()) + expect(diagnostic).toMatchObject({ + baseUrlConfigured: true, + authTokenConfigured: true, + apiKeyConfigured: false + }) + expect(JSON.stringify(diagnostic)).not.toContain('secret') + }) + + it('reports an unauthenticated start through the init deadline instead of hanging', async () => { + // The scripted CLI never answers, which is the shape of a silently unauthenticated CLI. + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open({ + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_IGNORE_CONTROL_REQUESTS: '1' } + }) + + await expect(connection.initializationResult({ timeoutMs: 200 })).rejects.toThrow( + 'claude initialize request timed out' + ) + }) + + it('reports a self-exit with its status and stderr, and leaves its tree unverifiable', async () => { + const scenario = scriptScenario([{ stderr: 'claude: not signed in\n' }, { exit: 1 }]) + let exit: Error | null = null + const connection = await open(launchFor(scenario), { + onExit: (error) => { + exit = error + } + }) + + await until(() => exit, 'the exit error') + // The status and stderr are the only diagnostic a refused start leaves behind. + expect((exit as unknown as Error).message).toMatch(/exited \(code 1\): claude: not signed in/) + expect(connection.closed).toBe(true) + // The root's exit is first-hand, but it left before a descendant snapshot + // could be armed, so close() has no tree proof to offer and says so. + await expect(connection.close()).resolves.toBe(false) + expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'unverifiable' }) + }) + + it.runIf(process.platform !== 'win32')( + 'proves a natural SDK exit and cleans up its descendant before recovery', + async () => { + const scenario = scriptScenario([ + { stderr: 'claude: natural exit\n' }, + { delayMs: 500 }, + { exit: 1 } + ]) + let exit: Error | null = null + const connection = await open( + { + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_DESCENDANT: '1' } + }, + { onExit: (error) => (exit = error) } + ) + const report = await until(() => { + const current = readReportSafely(scenario) + return current?.descendantPid ? current : null + }, 'the descendant report') + await until(() => exit, 'the natural exit error') + try { + await expect(connection.close()).resolves.toBe(true) + expect(connection.exitVerdict).toEqual({ root: 'exited', tree: 'exited' }) + expect(processState(report.descendantPid as number)).toBe('exited') + } finally { + try { + process.kill(report.descendantPid as number, 'SIGKILL') + } catch { + // Already gone. + } + } + }, + 20_000 + ) + + it('settles a spawn error followed by close as processless and closes idempotently', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + const missingCli = join(scenario.cwd, 'claude-that-does-not-exist') + let fault: Error | null = null + let exit: Error | null = null + const connection = await open( + { ...launchFor(scenario), pathToClaudeCodeExecutable: missingCli }, + { + onFault: (error) => { + fault = error + }, + onExit: (error) => { + exit = error + } + } + ) + + await until( + () => (connection.exitVerdict.root === 'processless' ? connection.exitVerdict : null), + 'the processless spawn settlement' + ) + expect(connection.pid).toBeUndefined() + expect(fault).toBeInstanceOf(Error) + expect(exit).toBeNull() + await expect(Promise.all([connection.close(), connection.close()])).resolves.toEqual([ + true, + true + ]) + await expect(connection.close()).resolves.toBe(true) + expect(connection.exitVerdict).toEqual({ root: 'processless', tree: 'exited' }) + }) + + it('does not treat a child error event as first-hand root exit proof', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + let exit: Error | null = null + const connection = await open(launchFor(scenario), { + onExit: (error) => { + exit = error + } + }) + const child = spawnedChildren.at(-1) + expect(child).toBeDefined() + + child?.emit('error', new Error('child transport fault')) + + expect(exit).toBeNull() + expect(connection.exitVerdict.root).toBe('live') + await until(() => exit, 'the distinct child exit') + expect(connection.exitVerdict.root).toBe('exited') + }) + + it('proves the exit of a child that ignores a graceful shutdown', async () => { + const scenario = scriptScenario([HOLD_OPEN]) + const connection = await open({ + ...launchFor(scenario), + env: { ...launchFor(scenario).env, ORCA_SDK_CONTRACT_IGNORE_SIGTERM: '1' } + }) + + // Keep the lstart capture boundary outside the child's displayed start second. + await new Promise((resolve) => setTimeout(resolve, 1_100)) + await expect(connection.close()).resolves.toBe(true) + }, 20_000) +}) + +// A structured Claude child owns the account's credentials while it runs, exactly as +// a Claude PTY does. The gate is what makes runtime-auth-sync defer the managed OAuth +// refresh instead of rotating the single-use token out from under a live session, and +// structured sessions used to be invisible to it. +describe('the managed-auth live gate', () => { + it('holds while a structured child runs and releases when it ends', async () => { + // The gate is a process-wide singleton and a sibling test's release lands on its + // child's 'close' event, which can settle after that test's close() resolved. + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + const connection = await open(launchFor(scenario)) + + expect(hasLiveClaudePtys()).toBe(true) + + await connection.close() + + await until(() => (hasLiveClaudePtys() ? null : true), 'the auth gate to drain') + expect(hasLiveClaudePtys()).toBe(false) + }, 30_000) + + it('releases when the child dies on its own rather than through close()', async () => { + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + await open(launchFor(scenario)) + expect(hasLiveClaudePtys()).toBe(true) + + spawnedChildren.at(-1)?.kill('SIGKILL') + + await until(() => (hasLiveClaudePtys() ? null : true), 'the auth gate to drain') + expect(hasLiveClaudePtys()).toBe(false) + }, 30_000) + + // The gate entry is deliberately unpersisted, so confirmSeededClaudeLivePtys can never + // reconcile a stray one: a leak here defers the managed OAuth refresh for the life of + // the process. Entering the gate only after the release handlers are attached makes + // that unreachable regardless of what the setup in between does. + it('leaks no gate entry when setup throws between spawn and handler attachment', async () => { + await until(() => (hasLiveClaudePtys() ? null : true), 'a drained auth gate') + const scenario = scriptScenario([ + { emit: { type: 'system', subtype: 'init', session_id: SESSION_ID, uuid: 'init-1' } }, + { wait: HOLD_OPEN } + ]) + let started: SpawnedProcess | null = null + + try { + await expect( + openClaudeStreamJsonConnection(launchFor(scenario), {}, (spec) => { + const child = spawnProcess(spec) + started = child + const attach = child.stderr.on.bind(child.stderr) + // Measured attach order: the SDK binds stderr 'data' from inside query(), + // before the child is even assigned. The SECOND bind is this connection's own + // armTreeOnOutput — the first statement that runs after the child exists and + // before its 'exit'/'close' release handlers. Throwing on the first is + // vacuous: it escapes before any gate entry could have happened. + let dataAttaches = 0 + child.stderr.on = ((event: string, listener: (...args: unknown[]) => void) => { + if (event === 'data') { + dataAttaches += 1 + if (dataAttaches === 2) { + throw new Error('stderr listener attach failed') + } + } + return attach(event, listener) + }) as typeof child.stderr.on + return child + }) + ).rejects.toThrow('stderr listener attach failed') + + expect(hasLiveClaudePtys()).toBe(false) + } finally { + ;(started as SpawnedProcess | null)?.kill('SIGKILL') + } + }, 30_000) +}) diff --git a/src/main/claude/claude-stream-json-connection.ts b/src/main/claude/claude-stream-json-connection.ts new file mode 100644 index 00000000000..dd6bbc8a5eb --- /dev/null +++ b/src/main/claude/claude-stream-json-connection.ts @@ -0,0 +1,283 @@ +import { randomUUID } from 'node:crypto' +import type * as ClaudeAgentSdk from '@anthropic-ai/claude-agent-sdk' +import type { CanUseTool, OnUserDialog, SDKUserMessage } from '@anthropic-ai/claude-agent-sdk' +import { spawnProcess } from '../../shared/child-process/run-process' +import { + markClaudeStructuredChildExited, + markClaudeStructuredChildSpawned +} from '../claude-accounts/live-pty-gate' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' +import { + ClaudeControlRequestError, + createClaudeControlSurface, + type ClaudeControlSurface +} from './claude-agent-sdk-control-requests' +import { createClaudeChildTreeReaper, proveClaudeChildExit } from './claude-agent-sdk-exit-proof' +import type { DescendantTreeVerdict } from '../pty-descendant-exit-verification' +import { createClaudeCodeProcessSpawn } from './claude-agent-sdk-process-spawn' +import { createClaudeUserMessageQueue } from './claude-agent-sdk-user-message-queue' +import type { ClaudeStructuredSdkOptions } from './claude-structured-launch-resolution' + +export { ClaudeControlRequestError } + +/** + * The SDK is loaded at the structured-Claude boundary rather than by this module's + * import. The ordinary runtime's class graph statically reaches this file, and the + * SDK sets `process.env.NoDefaultCurrentDirectoryInExePath` at import time — a + * Windows executable-search change that a user who never leaves the terminal/TUI + * path never opted into, and a missing SDK would fail runtime startup. Memoized, + * so a session pays the import once per process rather than once per connection. + */ +let claudeAgentSdk: Promise | null = null + +function loadClaudeAgentSdk(): Promise { + claudeAgentSdk ??= import('@anthropic-ai/claude-agent-sdk') + return claudeAgentSdk +} + +export type ClaudeStreamJsonLaunch = { + /** Orca's resolved user CLI; the SDK falls back to a bundled binary that is not installed. */ + pathToClaudeCodeExecutable: string + options: ClaudeStructuredSdkOptions + cwd: string + env?: Record +} + +export type ClaudeStreamJsonConnectionHandlers = { + onMessage?: (message: Record) => void + /** + * The SDK owns inbound permission control: it hands `can_use_tool` to this callback with + * a stable requestId and an abort signal, dedups duplicate delivery, and matches the + * response by request_id itself. Setting it makes the SDK pass `--permission-prompt-tool + * stdio` automatically; it must not be paired with `permissionPromptToolName`. + */ + canUseTool?: CanUseTool + /** `request_user_dialog` control; the CLI only emits kinds declared in `supportedDialogKinds`. */ + onUserDialog?: OnUserDialog + /** A transport/process fault that is not itself first-hand root exit proof. */ + onFault?: (error: Error) => void + onExit?: (error: Error) => void +} + +/** + * Two questions with their own evidence. The root's verdict is first-hand: Orca's + * own child handle reported exit, or reported error then close before it ever had + * a pid. The tree's comes from bounded descendant verification, and `unverifiable` + * is never collapsed into either neighbour. + */ +export type ClaudeChildExitVerdict = { + root: 'exited' | 'live' | 'processless' + tree: DescendantTreeVerdict +} + +export type ClaudeStreamJsonConnection = ClaudeControlSurface & { + readonly pid: number | undefined + readonly closed: boolean + /** What the ladder has observed so far; read after a `close()` that returned false. */ + readonly exitVerdict: ClaudeChildExitVerdict + send: (message: Record) => Promise + /** Resolves true after processless settlement, or root exit plus observed tree exit. */ + close: () => Promise +} + +type ExitStatus = { code: number | null; signal: NodeJS.Signals | null } + +function exitError(stderrTail: string, status: ExitStatus | null, cause?: Error): Error { + const detail = stderrTail.trim() + // The status is the diagnostic a signed-out or refused start leaves behind; + // it has to survive every wrapper between here and the user. + const how = + status?.signal !== null && status?.signal !== undefined + ? ` (signal ${status.signal})` + : status?.code !== null && status?.code !== undefined + ? ` (code ${status.code})` + : '' + const message = `claude stream-json exited${how}${detail ? `: ${detail}` : ''}` + return cause ? new Error(message, { cause }) : new Error(message) +} + +export async function openClaudeStreamJsonConnection( + launch: ClaudeStreamJsonLaunch, + handlers: ClaudeStreamJsonConnectionHandlers = {}, + spawnImpl: typeof spawnProcess = spawnProcess, + queryImpl?: typeof ClaudeAgentSdk.query +): Promise { + const { query } = await loadClaudeAgentSdk() + const spawner = createClaudeCodeProcessSpawn(spawnImpl) + const inbox = createClaudeUserMessageQueue() + const session = (queryImpl ?? query)({ + prompt: inbox.messages, + options: { + ...launch.options, + cwd: launch.cwd, + // Why env is never omitted: the SDK inherits process.env when it is, which is + // exactly the ambient ANTHROPIC_* auth leak this lane already shipped once. + env: buildClaudeChildProcessEnv(launch.env, { scrubConfiguredChildSessionStamps: true }), + pathToClaudeCodeExecutable: launch.pathToClaudeCodeExecutable, + spawnClaudeCodeProcess: spawner.spawn, + ...(handlers.canUseTool ? { canUseTool: handlers.canUseTool } : {}), + ...(handlers.onUserDialog ? { onUserDialog: handlers.onUserDialog } : {}) + } + }) + const child = spawner.child + if (!child) { + throw new Error('the claude agent SDK returned without spawning a child') + } + // This child owns the account's credentials for as long as it runs, exactly as a + // Claude PTY does — hold the OAuth-refresh gate so a managed refresh cannot rotate + // the single-use token out from under it mid-turn. Entered below, once a release + // path exists. + const authGateKey = randomUUID() + const releaseAuthGate = (): void => markClaudeStructuredChildExited(authGateKey) + let exited = false + let exitStatus: ExitStatus | null = null + let closing = false + let processless = false + let prePidSpawnError = false + let terminalError: Error | null = null + let faultReported = false + let exitReported = false + let closePromise: Promise | null = null + // One reaper per child: every close attempt and error-path reap shares its proof. + const rootSettled = (): boolean => exited || processless + const tree = createClaudeChildTreeReaper(child, { exited: rootSettled }) + + // Arm lazily on actual child output instead of issuing a process-table scan for + // every session at startup. A natural SDK exit can race a later close, while + // output-triggered observation still catches the usual live-child window. + let outputObservationArmed = false + const armTreeOnOutput = (): void => { + if (outputObservationArmed) { + return + } + outputObservationArmed = true + void (tree.refresh?.() ?? tree.capture()) + } + child.stderr.on('data', armTreeOnOutput) + // The SDK may synchronously spawn the CLI and consume an early stderr chunk + // before this connection can attach its listener; the bounded tail preserves + // that observation for the same lazy arm. + if (spawner.stderrTail.length > 0) { + armTreeOnOutput() + } + + let settleExit = (): void => {} + const exitPromise = new Promise((resolve) => { + settleExit = resolve + }) + const markExited = (): void => { + exited = true + releaseAuthGate() + settleExit() + } + child.on('exit', (code, signal) => { + exitStatus = { code, signal } + markExited() + handleUnexpectedEnd() + }) + + const handleUnexpectedEnd = (cause?: Error): void => { + terminalError ??= exitError(spawner.stderrTail, exitStatus, cause) + inbox.fail(terminalError) + if (!closing && !faultReported) { + faultReported = true + handlers.onFault?.(terminalError) + } + if (!closing && exited && !exitReported) { + exitReported = true + handlers.onExit?.(terminalError) + } + } + + void (async () => { + for await (const message of session) { + handlers.onMessage?.(message as unknown as Record) + } + })().catch((error: unknown) => { + // The SDK ends its generator in error when the child dies or the transport + // fails; a transport failure with a live child still has to reap the tree. + if (!closing && !exited) { + void tree.reap() + } + handleUnexpectedEnd(error instanceof Error ? error : new Error(String(error))) + }) + + child.on('error', (error) => { + if (spawner.pid === undefined) { + prePidSpawnError = true + } + if (!closing && !exited) { + void tree.reap() + } + handleUnexpectedEnd(error) + }) + child.on('close', () => { + // Covers the spawn-failure path too, where no 'exit' ever arrives. + releaseAuthGate() + if (prePidSpawnError && spawner.pid === undefined) { + processless = true + settleExit() + } + handleUnexpectedEnd() + }) + child.stdin.on('error', (error) => { + if (!closing) { + void tree.reap() + handleUnexpectedEnd(error) + } + }) + // Why here and not at spawn: a structured gate entry is deliberately unpersisted, so + // confirmSeededClaudeLivePtys can never reconcile a stray one and a leak defers the + // managed OAuth refresh for the life of the process. Entering only after 'exit' and + // 'close' are attached makes that unreachable — any later throw still leaves a + // listener that releases. Nothing between spawn and here can yield, so the child + // cannot end before the gate is entered. + markClaudeStructuredChildSpawned(authGateKey) + + const send = (message: Record): Promise => { + if (closing || exited || terminalError || child.stdin.destroyed || !child.stdin.writable) { + return Promise.reject(terminalError ?? new Error('claude stream-json connection is closed')) + } + return inbox.push(message as unknown as SDKUserMessage) + } + + const close = (): Promise => { + closePromise ??= (async () => { + closing = true + // Arm the descendant proof before ending stdin. The SDK may exit the root + // immediately; a post-exit walk cannot recover descendants that reparented. + await (tree.refresh?.() ?? tree.capture()) + inbox.end() + const proven = await proveClaudeChildExit({ + child, + exitPromise, + exited: rootSettled, + tree + }) + inbox.fail(new Error('claude stream-json connection closed')) + if (!proven) { + closePromise = null + } + return proven + })() + return closePromise + } + + return { + ...createClaudeControlSurface(session), + get pid() { + return spawner.pid + }, + get closed() { + return closing || exited || terminalError !== null + }, + get exitVerdict() { + return { + root: processless ? 'processless' : exited ? 'exited' : 'live', + tree: tree.treeVerdict + } as const + }, + send, + close + } +} diff --git a/src/main/claude/claude-streamed-block-identity.ts b/src/main/claude/claude-streamed-block-identity.ts new file mode 100644 index 00000000000..5cbf6674159 --- /dev/null +++ b/src/main/claude/claude-streamed-block-identity.ts @@ -0,0 +1,110 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +// Under --include-partial-messages every stream_event frame carries its own +// uuid, and the block's final `assistant` frame carries yet another; only +// `message.id` ties them together. The block's first stream frame mints the +// journal identity, and the final frame lands on it in block order instead of +// appending a duplicate under its own uuid. + +export type ClaudeStreamedTextDelta = { identity: AgentJournalItemIdentity; text: string } + +type StreamedMessage = { + messageId: string | null + blocks: Map + /** Streamed text blocks whose final assistant frame has not arrived, in block order. */ + awaitingFinal: AgentJournalItemIdentity[] +} + +export type ClaudeStreamedBlockRegistry = { + /** Text a stream_event frame appends to its block, or null when it carries none. */ + observe: (frame: Record) => ClaudeStreamedTextDelta | null + /** The streamed identity a final assistant frame reconciles onto, if its block streamed. */ + reconcile: (frame: { + sessionId: string + parentToolUseId: string | null + messageId: string | null + }) => AgentJournalItemIdentity | null + clear: () => void +} + +function scopeKey(sessionId: string, parentToolUseId: string | null): string { + return `${sessionId}/${parentToolUseId ?? ''}` +} + +export function createClaudeStreamedBlockRegistry(): ClaudeStreamedBlockRegistry { + const messages = new Map() + + const messageFor = (scope: string): StreamedMessage => { + let streamed = messages.get(scope) + if (!streamed) { + streamed = { messageId: null, blocks: new Map(), awaitingFinal: [] } + messages.set(scope, streamed) + } + return streamed + } + + const mint = ( + streamed: StreamedMessage, + sessionId: string, + index: number, + uuid: string + ): AgentJournalItemIdentity => { + const identity: AgentJournalItemIdentity = { provider: 'claude', sessionId, uuid } + streamed.blocks.set(index, identity) + streamed.awaitingFinal.push(identity) + return identity + } + + return { + observe: (frame) => { + const event = claudeRecord(frame.event) + const sessionId = claudeText(frame.session_id) + const uuid = claudeText(frame.uuid) + if (frame.type !== 'stream_event' || !event || !sessionId || !uuid) { + return null + } + const scope = scopeKey(sessionId, claudeText(frame.parent_tool_use_id)) + if (event.type === 'message_start') { + messages.set(scope, { + messageId: claudeText(claudeRecord(event.message)?.id), + blocks: new Map(), + awaitingFinal: [] + }) + return null + } + const index = typeof event.index === 'number' ? event.index : 0 + if (event.type === 'content_block_start') { + const block = claudeRecord(event.content_block) + if (block?.type !== 'text') { + return null + } + const identity = mint(messageFor(scope), sessionId, index, uuid) + const text = claudeText(block.text) + return text ? { identity, text } : null + } + if (event.type !== 'content_block_delta') { + return null + } + const delta = claudeRecord(event.delta) + const text = delta?.type === 'text_delta' ? claudeText(delta.text) : null + if (!text) { + return null + } + const streamed = messageFor(scope) + const identity = streamed.blocks.get(index) ?? mint(streamed, sessionId, index, uuid) + return { identity, text } + }, + reconcile: (frame) => { + const streamed = messages.get(scopeKey(frame.sessionId, frame.parentToolUseId)) + if ( + !streamed || + (frame.messageId && streamed.messageId && frame.messageId !== streamed.messageId) + ) { + return null + } + return streamed.awaitingFinal.shift() ?? null + }, + clear: () => messages.clear() + } +} diff --git a/src/main/claude/claude-streamed-text-checkpoints.test.ts b/src/main/claude/claude-streamed-text-checkpoints.test.ts new file mode 100644 index 00000000000..00a0bc0edd6 --- /dev/null +++ b/src/main/claude/claude-streamed-text-checkpoints.test.ts @@ -0,0 +1,93 @@ +import { describe, expect, it } from 'vitest' +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' + +function identityOf(uuid: string): AgentJournalItemIdentity { + return { provider: 'claude', sessionId: 'claude-session', uuid } +} + +function checkpoints() { + const rows: { uuid: string; text: string }[] = [] + let scheduled: (() => void) | null = null + const store = createClaudeStreamedTextCheckpoints({ + persist: (identity, text) => { + rows.push({ uuid: 'uuid' in identity ? identity.uuid : '', text }) + }, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + return { + store, + rows, + runWindow: () => { + const run = scheduled as (() => void) | null + run?.() + } + } +} + +describe('claude streamed text checkpoints', () => { + it('rewrites a block row with the full text accumulated so far', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'hel') + store.append(identityOf('block-1'), 'lo') + runWindow() + + expect(rows).toEqual([{ uuid: 'block-1', text: 'hello' }]) + expect(store.pending).toBe(1) + }) + + it('drops every block still awaiting its final frame at settlement', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'partial answer') + runWindow() + store.settle() + + expect(store.pending).toBe(0) + // The row written before settlement stays; nothing is rewritten afterwards. + store.flush() + expect(rows).toEqual([{ uuid: 'block-1', text: 'partial answer' }]) + }) + + it('keeps a block whose final frame arrived out of the settlement sweep', () => { + const { store } = checkpoints() + + store.append(identityOf('block-1'), 'one') + store.append(identityOf('block-2'), 'two') + store.forget('claude:claude-session:block-1') + + expect(store.pending).toBe(1) + store.settle() + expect(store.pending).toBe(0) + }) + + it('flushes text the widening checkpoint interval has not written yet', () => { + const { store, rows } = checkpoints() + + store.append(identityOf('block-1'), 'x') + store.flush() + + expect(rows).toEqual([{ uuid: 'block-1', text: 'x' }]) + // Already at the row's length: a second flush has nothing to write. + store.flush() + expect(rows).toHaveLength(1) + }) + + it('stops persisting once disposed', () => { + const { store, rows, runWindow } = checkpoints() + + store.append(identityOf('block-1'), 'text') + store.dispose() + runWindow() + store.flush() + + expect(rows).toEqual([]) + expect(store.pending).toBe(0) + }) +}) diff --git a/src/main/claude/claude-streamed-text-checkpoints.ts b/src/main/claude/claude-streamed-text-checkpoints.ts new file mode 100644 index 00000000000..348ecd99558 --- /dev/null +++ b/src/main/claude/claude-streamed-text-checkpoints.ts @@ -0,0 +1,105 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { + createAgentSessionDeltaCoalescer, + type AgentSessionDeltaCoalescerDeps +} from '../native-chat/agent-session-wire/agent-session-delta-coalescer' + +export type ClaudeStreamedTextCheckpointDeps = { + /** Rewrites the block's journal row with the text accumulated so far. */ + persist: (identity: AgentJournalItemIdentity, text: string) => void + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] +} + +export type ClaudeStreamedTextCheckpoints = { + /** Accumulate a delta; the row is rewritten on the coalescer's own cadence. */ + append: (identity: AgentJournalItemIdentity, text: string) => void + /** Write every block whose row is behind the text received for it. */ + flush: () => void + /** Drop one block's state, for a block whose final frame has now landed. */ + forget: (key: string) => void + /** + * Drop every block still awaiting its final frame, at turn settlement. Their + * text is already journaled by the flush that precedes settlement; keeping it + * live would grow with every interrupted turn for the life of the session. + */ + settle: () => void + /** Blocks still awaiting a final frame. A settled turn must leave none. */ + readonly pending: number + dispose: () => void +} + +/** + * Growth of a streamed block's row between its deltas and its final frame. + * + * The row is rewritten on a widening interval rather than per delta: a 200-line + * reply would otherwise rewrite the same journal row once per token. + */ +export function createClaudeStreamedTextCheckpoints( + deps: ClaudeStreamedTextCheckpointDeps +): ClaudeStreamedTextCheckpoints { + const identities = new Map() + const latestText = new Map() + const checkpointLengths = new Map() + + const persist = (key: string, text: string, force: boolean): void => { + latestText.set(key, text) + const checkpointLength = checkpointLengths.get(key) ?? 0 + const nextLength = Math.max(checkpointLength + 32, Math.ceil(checkpointLength * 1.125)) + if (!force && checkpointLength > 0 && text.length < nextLength) { + return + } + const identity = identities.get(key) + if (!identity) { + return + } + checkpointLengths.set(key, text.length) + deps.persist(identity, text) + } + + const coalescer = createAgentSessionDeltaCoalescer({ + ...(deps.coalesceMs === undefined ? {} : { windowMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + emit: (key, text) => persist(key, text, false) + }) + + const drop = (key: string): void => { + coalescer.forget(key) + identities.delete(key) + latestText.delete(key) + checkpointLengths.delete(key) + } + + return { + append: (identity, text) => { + const key = agentJournalItemKey(identity) + identities.set(key, identity) + coalescer.append(key, text) + }, + flush: () => { + coalescer.flushAll() + for (const [key, text] of latestText) { + if (checkpointLengths.get(key) !== text.length) { + persist(key, text, true) + } + } + }, + forget: drop, + settle: () => { + // Map iteration tolerates deletion of the entry just visited. + for (const key of identities.keys()) { + drop(key) + } + }, + get pending() { + return identities.size + }, + dispose: () => { + coalescer.dispose() + identities.clear() + latestText.clear() + checkpointLengths.clear() + } + } +} diff --git a/src/main/claude/claude-structured-acquisition-release.ts b/src/main/claude/claude-structured-acquisition-release.ts new file mode 100644 index 00000000000..1b633553e89 --- /dev/null +++ b/src/main/claude/claude-structured-acquisition-release.ts @@ -0,0 +1,43 @@ +import { + closeClaudeSession, + claudeAcquisitionCleanupError +} from './claude-structured-session-close' +import type { + ClaudeAcquisitionRegistry, + ClaudeSession, + ClaudeSessionExit, + ClaudeStructuredSessionAdapterDeps +} from './claude-structured-session-state' + +/** + * Cleanup for an acquisition the host could not commit or prove. A session that + * a first-hand exit already removed is not an absence to report as proven: the + * ladder on its connection still answers, and that answer is classified exactly + * as a start-time failure would be. + */ +export async function releaseClaudeAcquisition(input: { + sessionId: string + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + onExitProven?: (sessionId: string, exit: ClaudeSessionExit) => Promise + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] + onEvent?: ClaudeStructuredSessionAdapterDeps['onEvent'] +}): Promise { + const exit = input.exits.get(input.sessionId) + if (!exit || input.sessions.has(input.sessionId) || input.acquisitions.get(input.sessionId)) { + return closeClaudeSession(input) + } + const firstProof = exit.closePromise ? await exit.closePromise : false + // A failed exit-path proof is retained as evidence, not as a terminal result; + // a release retry must drive a fresh tree verification on the same connection. + const retriedProof = firstProof || (await exit.connection.close()) + if (retriedProof) { + await input.onExitProven?.(input.sessionId, exit) + // Keep the first-hand exit evidence indexed until the tree proof succeeds; + // a failed close must be retryable and cannot look like an absent session. + input.exits.delete(input.sessionId) + return true + } + throw claudeAcquisitionCleanupError(exit.connection, exit.error) +} diff --git a/src/main/claude/claude-structured-auth-parity.test.ts b/src/main/claude/claude-structured-auth-parity.test.ts new file mode 100644 index 00000000000..ddc69366aad --- /dev/null +++ b/src/main/claude/claude-structured-auth-parity.test.ts @@ -0,0 +1,235 @@ +import { afterEach, describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { beginClaudeAuthSwitch, endClaudeAuthSwitch } from '../claude-accounts/live-pty-gate' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE +} from '../claude-accounts/environment' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { createClaudeStructuredLaunchResolver } from './claude-structured-launch-resolution' +import { ClaudeStructuredSessionAdapter } from './claude-structured-session-adapter' +import { + PROVIDER_SESSION_ID, + adapterFor, + fakeClaude, + identityFor +} from './claude-structured-session-test-support' + +const SESSION_ID = 'orca-session-auth' +const IDENTITY = { sessionId: SESSION_ID } as Parameters< + ReturnType +>[0]['identity'] + +function record(): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [] + } as unknown as AgentSessionRecord +} + +function resolverFor(options: { + stripAuthEnv: boolean + overlay?: Record + authSwitchSettleTimeoutMs?: number +}): ReturnType { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: options.stripAuthEnv }), + authSwitchSettleTimeoutMs: options.authSwitchSettleTimeoutMs ?? 20, + ...(options.overlay ? { resolveEnv: () => options.overlay as Record } : {}) + }) +} + +/** + * An adapter driven by the REAL launch resolver, not the stub in the shared test + * support — the stub has no auth guard at all, so a teardown-window test built on it + * would pass whatever the guard did. + */ +function realResolverAdapter( + claude: ReturnType, + authSwitchSettleTimeoutMs: number +): ClaudeStructuredSessionAdapter { + const resumable = { + ...record(), + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null } } + ] + } as unknown as AgentSessionRecord + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: createClaudeStructuredLaunchResolver({ + store: { getRecord: () => resumable } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + authSwitchSettleTimeoutMs + }), + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + persistHandle: async () => {} + }) +} + +function withAmbientAuth(value: string, run: () => Promise): Promise { + const restore = process.env.ANTHROPIC_API_KEY + process.env.ANTHROPIC_API_KEY = value + return run().finally(() => { + if (restore === undefined) { + delete process.env.ANTHROPIC_API_KEY + } else { + process.env.ANTHROPIC_API_KEY = restore + } + }) +} + +describe('claude structured auth parity with the terminal preflight', () => { + afterEach(() => { + endClaudeAuthSwitch() + }) + + // Task 1 — the terminal preflight refuses this at spawn-env.ts:25 and + // runtime/spawn-preflight.ts:139; the structured path used to let the override win. + it('refuses an explicit Anthropic auth override while a managed account is pinned', async () => { + await expect( + resolverFor({ stripAuthEnv: true, overlay: { ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' } })({ + identity: IDENTITY + }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) + + it('refuses an auth-like ANTHROPIC_CUSTOM_HEADERS override while a managed account is pinned', async () => { + await expect( + resolverFor({ + stripAuthEnv: true, + overlay: { ANTHROPIC_CUSTOM_HEADERS: 'Authorization: Bearer sk-ant-CONFIGURED' } + })({ identity: IDENTITY }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) + + it('still admits a non-auth env overlay under a managed account', async () => { + const launch = await resolverFor({ + stripAuthEnv: true, + overlay: { ANTHROPIC_BASE_URL: 'https://gateway.example.test' } + })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_BASE_URL).toBe('https://gateway.example.test') + }) + + // Task 2 — legacy computes stripAuthEnv at runtime-auth-preparation.ts:72, so a + // system-auth user's own shell key is their sign-in and must survive. + it('passes an ambient Anthropic key through when no managed account is active', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ stripAuthEnv: false })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-SHELL') + }) + }) + + it('lets an explicit overlay override the ambient key when no managed account is active', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ + stripAuthEnv: false, + overlay: { ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' } + })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-CONFIGURED') + }) + }) + + it('still strips the ambient Anthropic key when a managed account is pinned', async () => { + await withAmbientAuth('sk-ant-SHELL', async () => { + const launch = await resolverFor({ stripAuthEnv: true })({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBeUndefined() + }) + }) + + // Task 3 — the terminal preflight guards this at four sites; the structured path had none. + it('refuses launch resolution when an account switch never settles', async () => { + beginClaudeAuthSwitch() + + await expect( + resolverFor({ stripAuthEnv: true, authSwitchSettleTimeoutMs: 20 })({ identity: IDENTITY }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + }) + + it('waits a settling account switch out rather than refusing a resolved launch', async () => { + beginClaudeAuthSwitch() + setTimeout(() => endClaudeAuthSwitch(), 20) + + const launch = await resolverFor({ + stripAuthEnv: true, + authSwitchSettleTimeoutMs: 5_000 + })({ identity: IDENTITY }) + + expect(launch.claudeConfigDir).toBe('/home/work/.claude') + }) + + it('refuses an acquire before it tears the previous session down', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude) + beginClaudeAuthSwitch() + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + // Nothing was spawned, so the refusal must not have opened a connection. + expect(claude.connections).toHaveLength(0) + }) + + // The teardown between the entry guard and launch resolution closes the live child + // and proves its tree — seconds, not milliseconds. A switch that begins inside it + // has already cost the user their session, so refusing there produces exactly the + // outcome the entry guard advertises against: a dead chat and no replacement. + it('replaces the session when a switch begins inside the acquire teardown', async () => { + const claude = fakeClaude() + const adapter = realResolverAdapter(claude, 5_000) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const live = claude.connections[0]! + const closeWithSwitch = live.close + live.close = async () => { + beginClaudeAuthSwitch() + setTimeout(() => endClaudeAuthSwitch(), 20) + return closeWithSwitch() + } + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).resolves.toMatchObject({ process: { spawnToken: 'spawn-10' } }) + expect(live.closed).toBe(true) + // The replacement child exists: the user's chat came back. + expect(claude.connections).toHaveLength(2) + expect(claude.connections[1]!.closed).toBe(false) + await adapter.closeAll() + }) + + it('still refuses a mid-teardown switch that never settles, leaving nothing half-open', async () => { + const claude = fakeClaude() + const adapter = realResolverAdapter(claude, 20) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const live = claude.connections[0]! + const closeWithSwitch = live.close + live.close = async () => { + beginClaudeAuthSwitch() + return closeWithSwitch() + } + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).rejects.toThrow(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + // No replacement child was opened, so nothing is left running unowned. + expect(claude.connections).toHaveLength(1) + await adapter.closeAll() + }) +}) diff --git a/src/main/claude/claude-structured-content-parts.test.ts b/src/main/claude/claude-structured-content-parts.test.ts new file mode 100644 index 00000000000..d2142150937 --- /dev/null +++ b/src/main/claude/claude-structured-content-parts.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity +} from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: () => {}, + publish: vi.fn() + } + return { sink, items } +} + +function providerRows(items: { body: AgentJournalItemBody }[]) { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame + ? [{ kind: item.body.providerFrame.kind, text: item.body.text }] + : [] + ) +} + +function userMessageWith(part: unknown) { + return { + type: 'message' as const, + sessionId: 'orca-session', + startsTurn: true as const, + message: { + type: 'user', + uuid: 'user-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: [{ type: 'text', text: 'look at this' }, part] } + } + } +} + +/** Exactly what claudeDispatchMessageContent sends for a local attachment. */ +const BASE64_IMAGE = { + type: 'image', + source: { type: 'base64', media_type: 'image/png', data: 'iVBORw0KGgoAAAANSUhEUg==' } +} + +describe('Claude message content parts', () => { + it('does not leak a wire kind for a locally attached image', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userMessageWith(BASE64_IMAGE)) + + expect(providerRows(state.items)).toEqual([]) + }) + + it('still renders an image the CLI sends by url', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + userMessageWith({ type: 'image', source: { type: 'url', url: 'https://x.test/a.png' } }) + ) + + expect(providerRows(state.items)).toEqual([]) + expect( + state.items.flatMap((item) => (item.body.kind === 'message' ? item.body.blocks : [])) + ).toContainEqual({ type: 'image-ref', url: 'https://x.test/a.png' }) + }) + + it('says what is true for a content part it cannot render, not the wire kind', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(userMessageWith({ type: 'some_future_part', payload: { a: 1 } })) + + const rows = providerRows(state.items) + expect(rows).toHaveLength(1) + // The kind stays on the row for debugging, behind the disclosure. + expect(rows[0].kind).toBe('message:user:content:some_future_part') + // ...but the visible text is a sentence, not the opcode. + expect(rows[0].text).not.toContain('message:user:content') + expect(rows[0].text.toLowerCase()).toContain('claude') + }) + + it('prefers a readable sentence the part carries over the placeholder', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + userMessageWith({ type: 'some_future_part', message: 'the server refused the upload' }) + ) + + expect(providerRows(state.items)[0].text).toBe('the server refused the upload') + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.test.ts b/src/main/claude/claude-structured-control-actions.test.ts new file mode 100644 index 00000000000..c471a8aca80 --- /dev/null +++ b/src/main/claude/claude-structured-control-actions.test.ts @@ -0,0 +1,113 @@ +import { describe, expect, it, vi } from 'vitest' +import { cancelClaudeTurn, answerClaudePrompt } from './claude-structured-control-actions' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeSession } from './claude-structured-session-state' + +type InterruptResult = Awaited> + +function sessionWith(input: { + capabilities?: string[] + interrupt: (options?: { cancelQueued?: boolean; timeoutMs?: number }) => Promise + cancelAsyncMessage?: (uuid: string) => Promise + prompts?: ClaudePromptRegistry +}): { + session: ClaudeSession + interrupt: ReturnType + cancelAsyncMessage: ReturnType +} { + const interrupt = vi.fn(input.interrupt) + const cancelAsyncMessage = vi.fn(input.cancelAsyncMessage ?? (async () => {})) + const session = { + capabilities: input.capabilities ?? [], + prompts: input.prompts ?? new ClaudePromptRegistry(), + connection: { interrupt, cancelAsyncMessage } + } as unknown as ClaudeSession + return { session, interrupt, cancelAsyncMessage } +} + +describe('cancelClaudeTurn', () => { + it('interrupts without a receipt on an older CLI and reports the turn cancelled', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + interrupt: async () => undefined + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(interrupt).toHaveBeenCalledWith({ timeoutMs: 5_000 }) + expect(cancelAsyncMessage).not.toHaveBeenCalled() + }) + + it('withdraws every still-queued message a plain interrupt receipt reports', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + capabilities: ['interrupt_receipt_v1'], + interrupt: async () => ({ still_queued: ['queued-1', 'queued-2'] }) + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + // No cancel_queued capability, so the queue is swept one uuid at a time. + expect(interrupt).toHaveBeenCalledWith({ timeoutMs: 5_000 }) + expect(cancelAsyncMessage.mock.calls.map((call) => call[0])).toEqual(['queued-1', 'queued-2']) + }) + + it('sends cancel_queued and never sweeps when the CLI advertises the capability', async () => { + const { session, interrupt, cancelAsyncMessage } = sessionWith({ + capabilities: ['interrupt_receipt_v1', 'interrupt_cancel_queued_v1'], + interrupt: async () => ({ still_queued: [], cancelled: ['queued-1'] }) + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: true }) + expect(interrupt).toHaveBeenCalledWith({ cancelQueued: true, timeoutMs: 5_000 }) + expect(cancelAsyncMessage).not.toHaveBeenCalled() + }) + + it('reports a not-running interrupt as not cancelled without throwing', async () => { + const { session } = sessionWith({ + interrupt: async () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + }) + + await expect(cancelClaudeTurn(session, 5_000)).resolves.toEqual({ cancelled: false }) + }) + + it('propagates a transport failure such as an interrupt timeout', async () => { + const { session } = sessionWith({ + interrupt: async () => { + throw new Error('claude interrupt request timed out') + } + }) + + await expect(cancelClaudeTurn(session, 5_000)).rejects.toThrow('timed out') + }) +}) + +describe('answerClaudePrompt', () => { + it('settles the pending prompt callback and forgets it', async () => { + const prompts = new ClaudePromptRegistry() + const settle = vi.fn() + const prompt = prompts.register({ + requestId: 'perm-1', + toolName: 'Bash', + toolUseId: 'tool-1', + input: { command: 'ls' }, + suggestions: [], + settle + })! + prompts.bindJournalItemId('journal-1', prompt.promptKey) + const { session } = sessionWith({ interrupt: async () => undefined, prompts }) + + await answerClaudePrompt(session, { itemId: 'journal-1', kind: 'approval', optionId: 'allow' }) + + expect(settle).toHaveBeenCalledWith( + expect.objectContaining({ behavior: 'allow', toolUseID: 'tool-1' }) + ) + expect(prompts.find('journal-1')).toBeNull() + }) + + it('refuses an answer for a prompt Claude is no longer waiting on', async () => { + const { session } = sessionWith({ interrupt: async () => undefined }) + await expect( + answerClaudePrompt(session, { itemId: 'missing', kind: 'approval', optionId: 'allow' }) + ).rejects.toThrow(/no longer waiting/) + }) +}) diff --git a/src/main/claude/claude-structured-control-actions.ts b/src/main/claude/claude-structured-control-actions.ts new file mode 100644 index 00000000000..d4484963aae --- /dev/null +++ b/src/main/claude/claude-structured-control-actions.ts @@ -0,0 +1,60 @@ +import { applyClaudePromptAnswer } from './claude-structured-prompt-replies' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import type { ClaudeSession } from './claude-structured-session-state' + +const INTERRUPT_CANCEL_QUEUED_CAPABILITY = 'interrupt_cancel_queued_v1' + +export type ClaudeTurnCancellationGuard = () => boolean + +/** + * Interrupt the running turn, then make sure no queued async user message survives to spawn a + * later unexpected turn. On a CLI advertising `interrupt_cancel_queued_v1` one round trip + * cancels the queue alongside the abort; otherwise the interrupt receipt lists `still_queued` + * uuids, and each is withdrawn best-effort with `cancel_async_message`. Older CLIs resolve no + * receipt, so there is nothing to sweep. + */ +export async function cancelClaudeTurn( + session: ClaudeSession, + timeoutMs: number | undefined, + isCurrent: ClaudeTurnCancellationGuard = () => true +): Promise<{ cancelled: boolean }> { + // The SDK interrupt is session-scoped. Re-check the caller's turn/fence + // immediately before issuing it so a delayed request cannot stop a later turn. + if (!isCurrent()) { + return { cancelled: false } + } + const cancelQueued = session.capabilities.includes(INTERRUPT_CANCEL_QUEUED_CAPABILITY) + try { + const receipt = await session.connection.interrupt({ + ...(cancelQueued ? { cancelQueued: true } : {}), + timeoutMs + }) + if (!cancelQueued) { + for (const uuid of receipt?.still_queued ?? []) { + await session.connection.cancelAsyncMessage(uuid, { timeoutMs }).catch(() => {}) + } + } + return { cancelled: true } + } catch (error) { + if (error instanceof ClaudeControlRequestError) { + return { cancelled: false } + } + throw error + } +} + +export async function answerClaudePrompt( + session: ClaudeSession, + input: { itemId: string; kind: 'approval' | 'question'; optionId: string } +): Promise { + const found = session.prompts.find(input.itemId) + if (!found || found.prompt.kind !== input.kind) { + throw new Error(`claude is no longer waiting on ${input.itemId}`) + } + const response = applyClaudePromptAnswer(found, input.optionId) + if (response === null) { + return + } + session.prompts.forget(found.prompt) + found.prompt.settle(response) +} diff --git a/src/main/claude/claude-structured-dispatch-content.ts b/src/main/claude/claude-structured-dispatch-content.ts new file mode 100644 index 00000000000..71f180bc3ac --- /dev/null +++ b/src/main/claude/claude-structured-dispatch-content.ts @@ -0,0 +1,165 @@ +import { createHash } from 'node:crypto' +import { open } from 'node:fs/promises' +import { extname } from 'node:path' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import type { NativeChatBlock } from '../../shared/native-chat-types' + +const MAX_IMAGE_BYTES = 5 * 1024 * 1024 +const MAX_IMAGE_COUNT = 20 +const MAX_TOTAL_IMAGE_BYTES = 20 * 1024 * 1024 +const MAX_REPLAY_CONTENT_KEY_BYTES = 256 + +type ImageBudget = { + count: number + localBytes: number +} + +export async function readClaudeImage(path: string, openImpl: typeof open = open): Promise { + const file = await openImpl(path, 'r') + try { + const invalidImage = (): Error => + new Error(`Claude image must be a non-empty file no larger than ${MAX_IMAGE_BYTES} bytes`) + const info = await file.stat() + if (!info.isFile()) { + throw new Error('Claude image must be a file') + } + if (info.size > MAX_IMAGE_BYTES) { + throw invalidImage() + } + const buffer = Buffer.allocUnsafe(info.size + 1) + let bytesRead = 0 + while (bytesRead < buffer.length) { + const result = await file.read(buffer, bytesRead, buffer.length - bytesRead, bytesRead) + if (result.bytesRead === 0) { + break + } + bytesRead += result.bytesRead + } + // A file can grow after the initial stat and after the final read returns + // zero. Prove the descriptor's size matches what was copied before sending. + const finalInfo = await file.stat() + if (bytesRead === 0 || bytesRead > MAX_IMAGE_BYTES || finalInfo.size !== bytesRead) { + throw invalidImage() + } + return buffer.subarray(0, bytesRead) + } finally { + await file.close() + } +} + +const IMAGE_MIME_BY_EXTENSION: Record = { + '.gif': 'image/gif', + '.jpeg': 'image/jpeg', + '.jpg': 'image/jpeg', + '.png': 'image/png', + '.webp': 'image/webp' +} + +async function imageContent( + block: Extract, + budget: ImageBudget +): Promise { + budget.count += 1 + if (budget.count > MAX_IMAGE_COUNT) { + throw new Error(`Claude messages support at most ${MAX_IMAGE_COUNT} images`) + } + if (block.url) { + return { type: 'image', source: { type: 'url', url: block.url } } + } + if (!block.path) { + throw new Error('image reference has neither a path nor a URL') + } + const data = await readClaudeImage(block.path) + budget.localBytes += data.byteLength + if (budget.localBytes > MAX_TOTAL_IMAGE_BYTES) { + throw new Error(`Claude images must total no more than ${MAX_TOTAL_IMAGE_BYTES} bytes`) + } + const mediaType = IMAGE_MIME_BY_EXTENSION[extname(block.path).toLowerCase()] + if (!mediaType) { + throw new Error(`Claude does not support the image type ${extname(block.path)}`) + } + return { + type: 'image', + source: { + type: 'base64', + media_type: mediaType, + data: data.toString('base64') + } + } +} + +export async function claudeDispatchMessageContent( + body: AgentJournalMessageItem +): Promise { + if (body.role !== 'user') { + throw new Error('Claude dispatch accepts only user messages') + } + const content: unknown[] = [] + const imageBudget: ImageBudget = { count: 0, localBytes: 0 } + for (const block of body.blocks as NativeChatBlock[]) { + if (block.type === 'text' && block.text.length > 0) { + content.push({ type: 'text', text: block.text }) + } else if (block.type === 'image-ref') { + content.push(await imageContent(block, imageBudget)) + } + } + if (content.length === 0) { + throw new Error('Claude dispatch requires text or an image') + } + return content +} + +/** + * Keep waiter metadata bounded even when a dispatch contains large base64 images. + * The digest is only diagnostic: replay acknowledgement must use provider identity. + */ +export function claudeDispatchContentKey(content: readonly unknown[]): string { + const digest = createHash('sha256') + const summary = content + .map((part) => { + const record = + typeof part === 'object' && part !== null && !Array.isArray(part) + ? (part as Record) + : null + const type = typeof record?.type === 'string' ? record.type : 'unknown' + if (type === 'text') { + return `text:${typeof record?.text === 'string' ? record.text.length : 0}` + } + const source = + typeof record?.source === 'object' && record.source !== null + ? (record.source as Record) + : null + if (type === 'image' && source?.type === 'base64') { + return `image:${typeof source.media_type === 'string' ? source.media_type : ''}:${typeof source.data === 'string' ? source.data.length : 0}` + } + return type + }) + .join(',') + for (const [index, part] of content.entries()) { + const record = + typeof part === 'object' && part !== null && !Array.isArray(part) + ? (part as Record) + : null + const type = typeof record?.type === 'string' ? record.type : 'unknown' + digest.update(`${index}:${type}:`) + if (type === 'text' && typeof record?.text === 'string') { + digest.update(record.text) + continue + } + const source = + typeof record?.source === 'object' && record.source !== null + ? (record.source as Record) + : null + if (type === 'image' && source?.type === 'base64') { + digest.update(typeof source.media_type === 'string' ? source.media_type : '') + digest.update(':') + if (typeof source.data === 'string') { + digest.update(source.data) + } + continue + } + digest.update(JSON.stringify(part)) + } + const key = `v1:${summary.slice(0, 128)}:${digest.digest('hex')}` + return key.slice(0, MAX_REPLAY_CONTENT_KEY_BYTES) +} diff --git a/src/main/claude/claude-structured-dispatch.test.ts b/src/main/claude/claude-structured-dispatch.test.ts new file mode 100644 index 00000000000..4e8289a89e3 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch.test.ts @@ -0,0 +1,598 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import { dispatchClaudeTurn, resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { readClaudeImage } from './claude-structured-dispatch-content' +import type { ClaudeSession } from './claude-structured-session-state' + +function sessionFor(send = vi.fn().mockResolvedValue(undefined)): ClaudeSession { + return { + connection: { send } as unknown as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +function userMessage(blocks: AgentJournalMessageItem['blocks']): AgentJournalMessageItem { + return { kind: 'message', role: 'user', blocks } +} + +function userReplayFrame(uuid: string, text: string): Record { + return { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid, + message: { role: 'user', content: [{ type: 'text', text }] } + } +} + +describe('Claude structured dispatch image limits', () => { + it('recovers the active identity when a timed-out replay arrives late', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(dispatched).resolves.toMatchObject({ state: 'unknown' }) + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid!, 'one'))).toBe(true) + expect(session.activeTurnId).toBe(sentUuid) + expect(session.activeTurnSequence).toBe(session.dispatchSequence) + }) + + it('never lets a late replay for dispatch A resolve dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const firstUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + + expect(resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one'))).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + expect(resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid!, 'two'))).toBe(true) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let an identical late replay for dispatch A resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect(resolveClaudeReplayWaiter(session, userReplayFrame('provider-a', 'same prompt'))).toBe( + false + ) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let a fresh-UUID replay for an evicted dispatch resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid + + const fillerDispatches = await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn( + session, + { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: 'same prompt' }]) + }, + 5 + ) + ) + ) + expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(session.retiredDispatchWaiters).toHaveLength(64) + expect(session.replayContentFallbackBlocked).toBe(true) + expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( + false + ) + + while (session.retiredDispatchWaiters.length > 0) { + const sentUuid = session.retiredDispatchWaiters[0]!.sentUuid + resolveClaudeReplayWaiter(session, userReplayFrame(sentUuid, 'same prompt')) + } + expect(session.retiredDispatchWaiters).toHaveLength(0) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'same prompt' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, userReplayFrame('provider-a-late', 'same prompt')) + ).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + resolveClaudeReplayWaiter(session, userReplayFrame(secondUuid, 'same prompt')) + await expect(second).resolves.toMatchObject({ providerIdentity: { uuid: secondUuid } }) + }) + + it('does not let a fresh-UUID result for an evicted slash dispatch resolve active dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + const firstUuid = session.retiredDispatchWaiters[0]!.sentUuid + + const fillerDispatches = await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn( + session, + { + clientMessageId: `filler-${index}`, + body: userMessage([{ type: 'text', text: '/permissions' }]) + }, + 5 + ) + ) + ) + expect(fillerDispatches.every((outcome) => outcome.state === 'unknown')).toBe(true) + expect(session.retiredDispatchWaiters).toHaveLength(64) + expect(session.replayContentFallbackBlocked).toBe(true) + expect(session.retiredDispatchWaiters.some((waiter) => waiter.sentUuid === firstUuid)).toBe( + false + ) + + while (session.retiredDispatchWaiters.length > 0) { + const sentUuid = session.retiredDispatchWaiters[0]!.sentUuid + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: `result-${sentUuid}`, + user_message_uuid: sentUuid + }) + ).toBe(false) + } + expect(session.retiredDispatchWaiters).toHaveLength(0) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-a-late' + }) + ).toBe(false) + expect(session.dispatchWaiters[0]).toMatchObject({ sentUuid: secondUuid }) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ + providerIdentity: { uuid: 'result-b' } + }) + }) + + it('does not let a legacy result for timed-out ordinary dispatch A resolve slash dispatch B', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'ordinary' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'legacy-result-a' + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ state: 'unknown' }) + }) + + it('removes only its own waiter when a later send fails', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const firstWaiter = session.dispatchWaiters[0] + session.connection.send = vi.fn().mockRejectedValue(new Error('broken pipe')) + + await expect( + dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: 'two' }]) }, + 100 + ) + ).resolves.toMatchObject({ state: 'unknown', reason: 'broken pipe' }) + expect(session.dispatchWaiters).toEqual([firstWaiter]) + + const firstUuid = (firstWaiter as { sentUuid?: string }).sentUuid + resolveClaudeReplayWaiter(session, userReplayFrame(firstUuid!, 'one')) + await expect(first).resolves.toMatchObject({ providerIdentity: { uuid: firstUuid } }) + }) + + it('keeps a replay accepted before its send reports failure', async () => { + let session!: ClaudeSession + const send = vi.fn(async (message: Record) => { + resolveClaudeReplayWaiter(session, { ...message, uuid: 'turn-race' }) + throw new Error('write raced provider acknowledgement') + }) + session = sessionFor(send) + + await expect( + dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'one' }]) }, + 100 + ) + ).resolves.toMatchObject({ state: 'accepted', providerIdentity: { uuid: 'turn-race' } }) + expect(session.dispatchWaiters).toHaveLength(0) + }) + + it('accepts a slash command from its result receipt when Claude omits the user replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'command-result-uuid' + }) + ).toBe(false) + + await expect(dispatched).resolves.toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'command-result-uuid' + } + }) + }) + + it('correlates a later slash-command result by user_message_uuid despite a timed-out slash waiter', async () => { + const session = sessionFor() + const first = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + await expect(first).resolves.toMatchObject({ state: 'unknown' }) + + const second = dispatchClaudeTurn( + session, + { clientMessageId: 'client-2', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 500 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const secondUuid = session.dispatchWaiters[0]!.sentUuid + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + subtype: 'success', + session_id: 'provider-session', + uuid: 'result-b', + user_message_uuid: secondUuid + }) + ).toBe(false) + await expect(second).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'result-b' } + }) + }) + + it('does not mistake a normal turn result for its missing user replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: 'hello' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + expect( + resolveClaudeReplayWaiter(session, { + type: 'result', + session_id: 'provider-session', + uuid: 'unrelated-result-uuid' + }) + ).toBe(false) + expect(session.dispatchWaiters).toHaveLength(1) + expect( + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: 'hello' }] + } + }) + ).toBe(true) + + await expect(dispatched).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'user-replay-uuid' } + }) + }) + + it('ignores a top-level tool-result user frame while waiting for a slash command replay', async () => { + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'text', text: '/permissions' }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'tool-result-uuid', + message: { + role: 'user', + content: [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'done' }] + } + }) + expect(session.dispatchWaiters).toHaveLength(1) + + resolveClaudeReplayWaiter(session, { + type: 'user', + parent_tool_use_id: null, + session_id: 'provider-session', + uuid: 'user-replay-uuid', + message: { + role: 'user', + content: [{ type: 'text', text: '/permissions' }] + } + }) + + await expect(dispatched).resolves.toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: 'provider-session', + uuid: 'user-replay-uuid' + } + }) + }) + + it('rejects more than twenty URL images before sending', async () => { + const session = sessionFor() + const body = userMessage( + Array.from({ length: 21 }, (_, index) => ({ + type: 'image-ref' as const, + url: `https://example.test/${index}.png` + })) + ) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ state: 'rejected', reason: 'Claude messages support at most 20 images' }) + expect(session.connection.send).not.toHaveBeenCalled() + }) + + it('rejects local images whose aggregate size exceeds twenty MiB', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-images-')) + try { + const paths = await Promise.all( + Array.from({ length: 5 }, async (_, index) => { + const path = join(directory, `${index}.png`) + await writeFile(path, Buffer.alloc(5 * 1024 * 1024)) + return path + }) + ) + const session = sessionFor() + const body = userMessage(paths.map((path) => ({ type: 'image-ref' as const, path }))) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ + state: 'rejected', + reason: `Claude images must total no more than ${20 * 1024 * 1024} bytes` + }) + expect(session.connection.send).not.toHaveBeenCalled() + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('rejects a local image by actual bytes read beyond the per-image cap', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + try { + const path = join(directory, 'oversized.png') + await writeFile(path, Buffer.alloc(5 * 1024 * 1024 + 1)) + const session = sessionFor() + const body = userMessage([{ type: 'image-ref', path }]) + + await expect( + dispatchClaudeTurn(session, { clientMessageId: 'client-1', body }, 1) + ).resolves.toEqual({ + state: 'rejected', + reason: `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` + }) + expect(session.connection.send).not.toHaveBeenCalled() + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('allocates local image reads from the file size, not the maximum cap', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + const allocUnsafe = vi.spyOn(Buffer, 'allocUnsafe') + try { + const path = join(directory, 'small.png') + await writeFile(path, Buffer.alloc(64)) + const session = sessionFor() + const dispatched = dispatchClaudeTurn( + session, + { clientMessageId: 'client-1', body: userMessage([{ type: 'image-ref', path }]) }, + 100 + ) + await vi.waitFor(() => expect(session.dispatchWaiters).toHaveLength(1)) + const sentUuid = (session.dispatchWaiters[0] as { sentUuid?: string }).sentUuid + resolveClaudeReplayWaiter(session, { + ...userReplayFrame(sentUuid!, ''), + message: { + role: 'user', + content: [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: '' } } + ] + } + }) + await expect(dispatched).resolves.toMatchObject({ state: 'accepted' }) + expect(allocUnsafe).toHaveBeenCalled() + expect(allocUnsafe.mock.calls.some(([size]) => size === 64 + 1)).toBe(true) + expect(allocUnsafe.mock.calls.some(([size]) => size >= 5 * 1024 * 1024)).toBe(false) + } finally { + allocUnsafe.mockRestore() + await rm(directory, { recursive: true, force: true }) + } + }) + + it('bounds retained waiter identity bytes when image dispatches time out', async () => { + const directory = await mkdtemp(join(tmpdir(), 'orca-claude-image-')) + try { + const path = join(directory, 'large.png') + await writeFile(path, Buffer.alloc(64 * 1024)) + const session = sessionFor() + const body = userMessage([{ type: 'image-ref', path }]) + await Promise.all( + Array.from({ length: 64 }, (_, index) => + dispatchClaudeTurn(session, { clientMessageId: `client-${index}`, body }, 1) + ) + ) + + expect(session.retiredDispatchWaiters).toHaveLength(64) + const retainedKeyBytes = session.retiredDispatchWaiters.reduce( + (total, waiter) => total + waiter.replayContentKey.length, + 0 + ) + expect(retainedKeyBytes).toBeLessThan(64 * 512) + expect( + session.retiredDispatchWaiters.every((waiter) => waiter.replayContentKey.length < 512) + ).toBe(true) + } finally { + await rm(directory, { recursive: true, force: true }) + } + }) + + it('rejects a local image when it grows after the initial stat', async () => { + const stat = vi + .fn() + .mockResolvedValueOnce({ isFile: () => true, size: 64 }) + .mockResolvedValueOnce({ isFile: () => true, size: 128 }) + const read = vi.fn(async (buffer: Buffer, offset: number) => { + if (read.mock.calls.length === 1) { + buffer.fill(1, offset, offset + 64) + return { bytesRead: 64, buffer } + } + return { bytesRead: 0, buffer } + }) + const open = vi.fn().mockResolvedValue({ + stat, + read, + close: vi.fn().mockResolvedValue(undefined) + } as never) + await expect(readClaudeImage('/controlled/growing.png', open)).rejects.toThrow( + `Claude image must be a non-empty file no larger than ${5 * 1024 * 1024} bytes` + ) + }) +}) diff --git a/src/main/claude/claude-structured-dispatch.ts b/src/main/claude/claude-structured-dispatch.ts new file mode 100644 index 00000000000..96271e41d71 --- /dev/null +++ b/src/main/claude/claude-structured-dispatch.ts @@ -0,0 +1,264 @@ +import { randomUUID } from 'node:crypto' +import type { AgentJournalMessageItem } from '../../shared/agent-session-journal-types' +import type { AgentSessionDispatchOutcome } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + claudeHasReplayContent, + readClaudeMessageEnvelope +} from './claude-structured-item-translation' +import type { ClaudeDispatchWaiter, ClaudeSession } from './claude-structured-session-state' +import { readClaudeFrameString } from './claude-structured-init-proof' +import { + claudeDispatchContentKey, + claudeDispatchMessageContent +} from './claude-structured-dispatch-content' + +const MAX_RETIRED_DISPATCH_WAITERS = 64 + +export function resolveClaudeReplayWaiter( + session: ClaudeSession, + message: Record +): boolean { + const envelope = readClaudeMessageEnvelope(message) + const isUserReplay = + envelope?.role === 'user' && + message.parent_tool_use_id === null && + claudeHasReplayContent(envelope) + const isCompletedCommand = message.type === 'result' + if ( + (!isUserReplay && !isCompletedCommand) || + readClaudeFrameString(message, 'session_id') !== session.providerSessionId + ) { + return false + } + const uuid = readClaudeFrameString(message, 'uuid') + if (!uuid) { + return false + } + + // Newer SDK frames carry the client uuid that caused a turn. A correlation + // value is authoritative: never fall back to queue order or content, since + // identical prompts may be in flight across a timeout boundary. + const userMessageUuid = readClaudeFrameString(message, 'user_message_uuid') + if (userMessageUuid) { + const exact = session.dispatchWaiters.find( + (candidate) => candidate.sentUuid === userMessageUuid + ) + if (exact) { + settleWaiter(session, exact, uuid) + return isUserReplay && exact.dispatchSequence === session.dispatchSequence + } + const retired = session.retiredDispatchWaiters.find( + (candidate) => candidate.sentUuid === userMessageUuid + ) + if (retired) { + forgetRetiredWaiter(session, retired) + return recoverLateIdentity(session, retired, uuid, isUserReplay) + } + return false + } + + const exact = session.dispatchWaiters.find((candidate) => candidate.sentUuid === uuid) + if (exact) { + settleWaiter(session, exact, uuid) + return isUserReplay && exact.dispatchSequence === session.dispatchSequence + } + const retired = session.retiredDispatchWaiters.find((candidate) => candidate.sentUuid === uuid) + if (retired) { + forgetRetiredWaiter(session, retired) + return recoverLateIdentity(session, retired, uuid, isUserReplay) + } + + if (isUserReplay) { + // Compatibility CLIs may mint a new replay uuid instead of echoing the + // client uuid. Content is an acceptable join only when it is the sole + // candidate on one side of the timeout boundary; with active and retired + // candidates present, identical prompts are intentionally left unknown. + const replayContentKey = claudeDispatchContentKey(envelope.content) + if (!session.replayContentFallbackBlocked && session.retiredDispatchWaiters.length === 0) { + const compatible = session.dispatchWaiters.filter( + (candidate) => candidate.replayContentKey === replayContentKey + ) + if (compatible.length === 1) { + settleWaiter(session, compatible[0]!, uuid) + return compatible[0]!.dispatchSequence === session.dispatchSequence + } + } else if (!session.replayContentFallbackBlocked && session.dispatchWaiters.length === 0) { + const lateCompatible = session.retiredDispatchWaiters.filter( + (candidate) => candidate.replayContentKey === replayContentKey + ) + if (lateCompatible.length === 1) { + const [candidate] = lateCompatible + forgetRetiredWaiter(session, candidate!) + return recoverLateIdentity(session, candidate!, uuid, true) + } + } + return false + } + const current = session.dispatchWaiters[0] + if (isCompletedCommand && !current?.acceptsResult) { + return false + } + // A legacy result has no dispatch correlation. Any retired waiter makes queue order ambiguous, + // even when the retired dispatch was an ordinary turn rather than a slash command. + if (isCompletedCommand && session.retiredDispatchWaiters.length > 0) { + return false + } + // Once an eviction occurred, a fresh result uuid cannot be joined to a waiter by queue order. + if (isCompletedCommand && session.replayContentFallbackBlocked) { + return false + } + const waiter = uuid ? session.dispatchWaiters.shift() : undefined + if (waiter && uuid) { + clearTimeout(waiter.timer) + waiter.settledUuid = uuid + waiter.resolve(uuid) + return isUserReplay + } + return false +} + +function settleWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter, uuid: string): void { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + clearTimeout(waiter.timer) + waiter.settledUuid = uuid + waiter.resolve(uuid) +} + +function forgetRetiredWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + const index = session.retiredDispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.retiredDispatchWaiters.splice(index, 1) + } +} + +function recoverLateIdentity( + session: ClaudeSession, + waiter: ClaudeDispatchWaiter, + uuid: string, + isUserReplay: boolean +): boolean { + if (!isUserReplay && !waiter.acceptsResult) { + return false + } + if (waiter.dispatchSequence === session.dispatchSequence) { + session.activeTurnId = uuid + session.activeTurnSequence = waiter.dispatchSequence + } + return isUserReplay && waiter.dispatchSequence === session.dispatchSequence +} + +function waitForReplay( + session: ClaudeSession, + timeoutMs: number, + acceptsResult: boolean, + sentUuid: string, + replayContentKey: string +): { waiter: ClaudeDispatchWaiter; promise: Promise } { + let waiter!: ClaudeDispatchWaiter + const promise = new Promise((resolve) => { + waiter = { + acceptsResult, + sentUuid, + dispatchSequence: session.dispatchSequence, + replayContentKey, + resolve, + timer: setTimeout(() => { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + retireWaiter(session, waiter) + resolve(null) + }, timeoutMs) + } + waiter.timer.unref?.() + session.dispatchWaiters.push(waiter) + }) + return { waiter, promise } +} + +function retireWaiter(session: ClaudeSession, waiter: ClaudeDispatchWaiter): void { + const index = session.dispatchWaiters.indexOf(waiter) + if (index !== -1) { + session.dispatchWaiters.splice(index, 1) + } + clearTimeout(waiter.timer) + if (!waiter.retired) { + waiter.retired = true + session.retiredDispatchWaiters.push(waiter) + if (session.retiredDispatchWaiters.length > MAX_RETIRED_DISPATCH_WAITERS) { + session.replayContentFallbackBlocked = true + session.retiredDispatchWaiters.splice( + 0, + session.retiredDispatchWaiters.length - MAX_RETIRED_DISPATCH_WAITERS + ) + } + } +} + +export async function dispatchClaudeTurn( + session: ClaudeSession, + input: { clientMessageId: string; body: AgentJournalMessageItem }, + timeoutMs: number +): Promise { + let content: unknown[] + try { + content = await claudeDispatchMessageContent(input.body) + } catch (error) { + return { state: 'rejected', reason: (error as Error).message } + } + const dispatchSequence = ++session.dispatchSequence + const acceptsResult = input.body.blocks.some( + (block) => block.type === 'text' && block.text.trimStart().startsWith('/') + ) + const sentUuid = randomUUID() + const replay = waitForReplay( + session, + timeoutMs, + acceptsResult, + sentUuid, + claudeDispatchContentKey(content) + ) + const replayed = replay.promise + try { + await session.connection.send({ + type: 'user', + uuid: sentUuid, + message: { role: 'user', content }, + parent_tool_use_id: null, + session_id: session.providerSessionId + }) + } catch (error) { + const waiter = replay.waiter + if (waiter.settledUuid) { + const uuid = await replayed + if (uuid) { + session.activeTurnId = uuid + session.activeTurnSequence = dispatchSequence + return { + state: 'accepted', + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + } + } + } + if (!waiter.retired) { + retireWaiter(session, waiter) + waiter.resolve(null) + } + return { state: 'unknown', reason: (error as Error).message } + } + const uuid = await replayed + if (uuid) { + session.activeTurnId = uuid + session.activeTurnSequence = dispatchSequence + } + return uuid + ? { + state: 'accepted', + providerIdentity: { provider: 'claude', sessionId: session.providerSessionId, uuid } + } + : { state: 'unknown', reason: 'claude accepted a message but did not replay its uuid in time' } +} diff --git a/src/main/claude/claude-structured-effort-reporting.test.ts b/src/main/claude/claude-structured-effort-reporting.test.ts new file mode 100644 index 00000000000..be022d86956 --- /dev/null +++ b/src/main/claude/claude-structured-effort-reporting.test.ts @@ -0,0 +1,257 @@ +import { describe, expect, it } from 'vitest' +import { AgentSessionOptionRejectedError } from '../native-chat/agent-session-wire/structured-agent-session-option-error' +import { + restoreClaudeStructuredSessionOptions, + setClaudeStructuredOption +} from './claude-structured-options' +import { readClaudeSettingsEffort } from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-adapter' +import { acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim from Claude Code 2.1.258's get_settings response. */ +const REAL_SETTINGS = { + applied: { model: 'claude-opus-5[1m]', effort: 'high', advisor: null, ultracode: false }, + effective: { model: 'claude-opus-5[1m]', effortLevel: 'high', env: {} }, + sources: {} +} + +function sessionWith( + reported: string | null, + calls: string[] = [], + listed?: { model: string; catalog: readonly Record[] } +) { + return { + session: { + options: new Map(listed ? [['model', listed.model]] : []), + reportedOptions: {} as { model?: string; effort?: string }, + optionMutationSequence: 0, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + connection: { + supportedModels: async () => { + calls.push('list_models') + return [...(listed?.catalog ?? [])] + }, + setModel: async (model: string) => { + calls.push(`set_model:${model}`) + }, + applyFlagSettings: async (settings: { effortLevel?: string }) => { + // The measured behaviour: an unknown effort is accepted and ignored. + calls.push(`apply:${settings.effortLevel}`) + }, + getSettings: async () => { + calls.push('get_settings') + return reported === null + ? { applied: {}, effective: {}, sources: {} } + : { applied: { effort: reported }, effective: { effortLevel: reported }, sources: {} } + } + } + } as unknown as ClaudeSession, + calls + } +} + +describe('Claude effort reporting', () => { + it('reads the effort get_settings reports', () => { + expect(readClaudeSettingsEffort(REAL_SETTINGS)).toBe('high') + }) + + it.each([ + [ + 'the provider stops reporting it', + { applied: { effort: 'high' }, effective: {}, sources: {} } + ], + ['the payload carries no effective block', { applied: { effort: 'high' } }], + ['the request failed outright', null] + ])('reports no effort when %s', (_case, settings) => { + // Never defaulted: an effort nothing measured would be worse than a blank + // pill, and this is the assertion that goes red if the key is renamed. + expect(readClaudeSettingsEffort(settings)).toBeNull() + }) + + it('publishes the effort from get_settings, which system/init never carries', async () => { + const claude = fakeClaude({ settings: REAL_SETTINGS }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { effort: 'high' } + }) + }) + + it('leaves the effort unreported when the session never learns one', async () => { + const claude = fakeClaude({ settings: { applied: {}, effective: {}, sources: {} } }) + const adapter = await acquired(claude) + + const options = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(options.current.effort).toBeUndefined() + expect(options.current.model).toBeTruthy() + }) + + it('keeps the init fixture free of an effort the real frame never sends', async () => { + const events: ClaudeStructuredSessionEvent[] = [] + await acquired(fakeClaude(), {}, events) + const init = events.flatMap((event) => + event.type === 'message' && event.message.subtype === 'init' ? [event.message] : [] + ) + + expect(init).toHaveLength(1) + expect(init[0]).toHaveProperty('model') + // The regression that hid this defect: a fixture inventing `effortLevel` + // kept every gate green over a value that is always empty in production. + expect(Object.keys(init[0])).not.toContain('effortLevel') + }) +}) + +describe('Claude effort readback', () => { + it('records an effort the child did not adopt without vouching for it', async () => { + const { session, calls } = sessionWith('high') + + // The disagreement stops the confirmation, not the write: no other client + // vetoes here, and the pre-flight catalog guard already refuses the levels + // the model cannot run. + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'bogus-effort-xyz' }, undefined) + ).resolves.toEqual({ effort: 'bogus-effort-xyz' }) + expect(session.confirmedOptions.has('effort')).toBe(false) + // The child's own answer is kept rather than discarded with the refusal. + expect(session.reportedOptions.effort).toBe('high') + expect(calls).toEqual(['apply:bogus-effort-xyz', 'get_settings']) + }) + + it('records an effort the child confirms', async () => { + const { session } = sessionWith('low') + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + ).resolves.toEqual({ effort: 'low' }) + }) + + it('records the request when the readback is unavailable', async () => { + // No evidence of a refusal is not evidence of one; the apply itself succeeded. + const { session } = sessionWith(null) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + ).resolves.toEqual({ effort: 'low' }) + }) +}) + +describe('Claude effort against the model that must run it', () => { + const HAIKU = { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } + const SONNET = { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + } + + it('refuses an effort the current model advertises no control for', async () => { + const { session, calls } = sessionWith('high', [], { model: 'haiku', catalog: [HAIKU, SONNET] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + // Measured on Claude Code 2.1.260: apply_flag_settings stores `high` on a + // haiku session and get_settings reads it straight back, so a send here is + // never undone. The refusal has to land before the write. + expect(calls).toEqual(['list_models']) + expect(session.options.has('effort')).toBe(false) + }) + + it('refuses a level outside the ones the current model advertises', async () => { + const { session } = sessionWith('high', [], { + model: 'sonnet', + catalog: [{ ...SONNET, supportedEffortLevels: ['low', 'medium'] }] + }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + }) + + it('sends an effort the current model advertises', async () => { + const { session, calls } = sessionWith('high', [], { + model: 'sonnet', + catalog: [HAIKU, SONNET] + }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + expect(session.confirmedOptions.has('effort')).toBe(true) + }) + + it('sends `max`, which the readback cannot report, when the model advertises it', async () => { + // UNREPORTED_EFFORTS still governs: no get_settings, so no false disagreement. + const { session, calls } = sessionWith('high', [], { model: 'sonnet', catalog: [SONNET] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'max' }) + expect(calls).toEqual(['list_models', 'apply:max']) + expect(session.confirmedOptions.has('effort')).toBe(false) + }) + + it('sends the effort when the model is not in the catalog the CLI listed', async () => { + // An unlisted model is an unknown one, not one that refuses effort. + const { session, calls } = sessionWith('high', [], { model: 'sonnet', catalog: [HAIKU] }) + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + }) + + it('sends the effort when list_models is unavailable', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [] }) + session.connection.supportedModels = async () => { + calls.push('list_models') + throw new Error('this CLI predates list_models') + } + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'high' }) + expect(calls).toEqual(['list_models', 'apply:high', 'get_settings']) + }) + + it('matches the model the init frame reported, not just the id the user picked', async () => { + const { session } = sessionWith('high', [], { model: 'sonnet', catalog: [HAIKU, SONNET] }) + session.options.delete('model') + session.reportedOptions.model = 'claude-haiku-4-5-20251001' + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'high' }, undefined) + ).rejects.toBeInstanceOf(AgentSessionOptionRejectedError) + }) + + it('keeps a disagreeing effort through restore instead of skipping it', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [SONNET] }) + session.options.set('effort', 'low') + + await restoreClaudeStructuredSessionOptions(session, undefined) + + expect(session.options.get('effort')).toBe('low') + expect(session.restoreSkippedOptions.has('effort')).toBe(false) + expect(session.confirmedOptions.has('effort')).toBe(false) + }) + + it('drops a stale effort on restore instead of replaying it onto the new model', async () => { + const calls: string[] = [] + const { session } = sessionWith('high', calls, { model: 'sonnet', catalog: [HAIKU, SONNET] }) + session.options.set('model', 'haiku') + session.options.set('effort', 'high') + + await restoreClaudeStructuredSessionOptions(session, undefined) + + expect(session.options.has('effort')).toBe(false) + expect(session.restoreSkippedOptions.has('effort')).toBe(true) + expect(calls.filter((call) => call.startsWith('apply:'))).toEqual([]) + }) +}) diff --git a/src/main/claude/claude-structured-inbound-control.test.ts b/src/main/claude/claude-structured-inbound-control.test.ts new file mode 100644 index 00000000000..07be4bbb516 --- /dev/null +++ b/src/main/claude/claude-structured-inbound-control.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it, vi } from 'vitest' +import type { CanUseTool } from '@anthropic-ai/claude-agent-sdk' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { + buildClaudePermissionCallbacks, + CLAUDE_BLOCKING_CONTROL_CALLBACKS, + CLAUDE_CAN_USE_TOOL_SUBTYPE, + CLAUDE_REQUEST_USER_DIALOG_SUBTYPE +} from './claude-structured-inbound-control' + +type CanUseToolOptions = Parameters[2] + +function permissionOptions( + requestId: string, + toolUseID: string, + signal: AbortSignal, + suggestions?: unknown[] +): CanUseToolOptions { + return { + requestId, + toolUseID, + signal, + ...(suggestions ? { suggestions } : {}) + } as unknown as CanUseToolOptions +} + +function callbacksFor() { + const prompts = new ClaudePromptRegistry() + const emit = vi.fn() + const { canUseTool, onUserDialog } = buildClaudePermissionCallbacks({ + sessionId: 'session-1', + prompts, + emit + }) + return { prompts, emit, canUseTool, onUserDialog } +} + +describe('Claude permission callbacks', () => { + it('registers a decodable can_use_tool as a durable prompt and settles it from the registry', async () => { + const control = callbacksFor() + const answered = control.canUseTool( + 'Bash', + { command: 'git status' }, + permissionOptions('perm-1', 'tool-1', new AbortController().signal, [{ type: 'addRules' }]) + ) + + expect(control.emit).toHaveBeenCalledWith( + expect.objectContaining({ + type: 'prompt', + sessionId: 'session-1', + prompt: expect.objectContaining({ promptKey: 'perm-1', toolName: 'Bash', kind: 'approval' }) + }) + ) + const found = control.prompts.find('perm-1') + expect(found?.prompt.suggestions).toEqual([{ type: 'addRules' }]) + // The prompt's settle is the SDK callback's own resolve — answering resolves this promise. + found?.prompt.settle({ behavior: 'allow', toolUseID: 'tool-1' }) + await expect(answered).resolves.toEqual({ behavior: 'allow', toolUseID: 'tool-1' }) + }) + + it('denies a malformed permission request without registering a prompt', async () => { + const control = callbacksFor() + const answered = control.canUseTool( + '', + {}, + permissionOptions('perm-2', 'tool-2', new AbortController().signal) + ) + + await expect(answered).resolves.toEqual({ + behavior: 'deny', + message: 'Orca could not decode this permission request.', + toolUseID: 'tool-2' + }) + expect(control.prompts.find('perm-2')).toBeNull() + expect(control.emit).not.toHaveBeenCalled() + }) + + it('settles a pending prompt with null and forgets it when the abort signal fires', async () => { + const control = callbacksFor() + const controller = new AbortController() + const answered = control.canUseTool( + 'Bash', + { command: 'ls' }, + permissionOptions('perm-3', 'tool-3', controller.signal) + ) + expect(control.prompts.find('perm-3')).not.toBeNull() + + controller.abort() + + await expect(answered).resolves.toBeNull() + expect(control.emit).toHaveBeenLastCalledWith( + expect.objectContaining({ type: 'prompt-cancelled', promptKey: 'perm-3' }) + ) + // Forgotten: a late answer can no longer find the prompt to authorize the wrong tool. + expect(control.prompts.find('perm-3')).toBeNull() + }) + + it('cancels a request whose abort raced ahead of delivery without emitting a prompt', async () => { + const control = callbacksFor() + const controller = new AbortController() + controller.abort() + + const answered = control.canUseTool( + 'Bash', + { command: 'ls' }, + permissionOptions('perm-4', 'tool-4', controller.signal) + ) + + await expect(answered).resolves.toBeNull() + expect(control.prompts.find('perm-4')).toBeNull() + expect(control.emit).toHaveBeenCalledTimes(1) + expect(control.emit).toHaveBeenCalledWith( + expect.objectContaining({ type: 'prompt-cancelled', promptKey: 'perm-4' }) + ) + }) + + it('settles every in-flight prompt with null when the registry is cleared', async () => { + const control = callbacksFor() + const first = control.canUseTool( + 'Bash', + { command: 'a' }, + permissionOptions('perm-5', 'tool-5', new AbortController().signal) + ) + const second = control.canUseTool( + 'Bash', + { command: 'b' }, + permissionOptions('perm-6', 'tool-6', new AbortController().signal) + ) + + // What session close does: settle each pending callback so no promise dangles. + for (const prompt of control.prompts.clear()) { + prompt.settle(null) + } + + await expect(first).resolves.toBeNull() + await expect(second).resolves.toBeNull() + }) + + it('answers a user dialog deny-safe', async () => { + const control = callbacksFor() + await expect( + control.onUserDialog( + { dialogKind: 'refusal_fallback_prompt', payload: {} }, + { signal: new AbortController().signal, requestId: 'dialog-1' } + ) + ).resolves.toEqual({ behavior: 'cancelled' }) + }) + + it('enumerates every blocking control request and wires a callback for each', () => { + // The stable surface of controls a turn can block on. Adding one here without wiring its + // callback below fails this test rather than silently leaving a control unhandled. + expect(new Set(Object.keys(CLAUDE_BLOCKING_CONTROL_CALLBACKS))).toEqual( + new Set([CLAUDE_CAN_USE_TOOL_SUBTYPE, CLAUDE_REQUEST_USER_DIALOG_SUBTYPE]) + ) + const callbacks = buildClaudePermissionCallbacks({ + sessionId: 'session-1', + prompts: new ClaudePromptRegistry(), + emit: vi.fn() + }) as unknown as Record + for (const callbackName of Object.values(CLAUDE_BLOCKING_CONTROL_CALLBACKS)) { + expect(typeof callbacks[callbackName], `${callbackName} must be wired`).toBe('function') + } + }) +}) diff --git a/src/main/claude/claude-structured-inbound-control.ts b/src/main/claude/claude-structured-inbound-control.ts new file mode 100644 index 00000000000..343e76d4ea5 --- /dev/null +++ b/src/main/claude/claude-structured-inbound-control.ts @@ -0,0 +1,91 @@ +import type { CanUseTool, OnUserDialog, PermissionResult } from '@anthropic-ai/claude-agent-sdk' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +export const CLAUDE_CAN_USE_TOOL_SUBTYPE = 'can_use_tool' +export const CLAUDE_REQUEST_USER_DIALOG_SUBTYPE = 'request_user_dialog' + +/** + * The blocking control requests Orca answers, each mapped to the SDK consumer callback that + * answers it. This is the stable surface a real turn can block on: `can_use_tool` through + * `canUseTool` and `request_user_dialog` through `onUserDialog`. Every other control-request + * subtype the SDK routes (elicitation, oauth/host token refresh, mcp_message, hook_callback) + * is either not surfaced to this consumer or fails closed inside the SDK; adding a new + * blocking control Orca must answer means adding its callback here, and the catalog test + * fails if a named callback is missing. + */ +export const CLAUDE_BLOCKING_CONTROL_CALLBACKS = { + [CLAUDE_CAN_USE_TOOL_SUBTYPE]: 'canUseTool', + [CLAUDE_REQUEST_USER_DIALOG_SUBTYPE]: 'onUserDialog' +} as const + +export type ClaudeBlockingControlSubtype = keyof typeof CLAUDE_BLOCKING_CONTROL_CALLBACKS + +export type ClaudePermissionCallbackDeps = { + sessionId: string + prompts: ClaudePromptRegistry + emit: (event: ClaudeStructuredSessionEvent) => void +} + +function denySafeResult(toolUseId: string | undefined): PermissionResult { + return { + behavior: 'deny', + message: 'Orca could not decode this permission request.', + ...(toolUseId ? { toolUseID: toolUseId } : {}) + } +} + +/** + * Build the SDK permission callbacks from the durable prompt registry. + * + * A decodable `can_use_tool` becomes a durable prompt whose `settle` resolves this callback; + * a malformed one is denied without registering. The SDK's abort signal fires on + * `control_cancel_request` (a cancelled turn), which forgets the prompt and settles it with + * `null` — never authorizing a tool. A late answer after abort finds no prompt and is refused + * by `answerClaudePrompt`. `onUserDialog` is deny-safe; the CLI only emits dialog kinds Orca + * declares in `supportedDialogKinds`, which is empty. + */ +export function buildClaudePermissionCallbacks(deps: ClaudePermissionCallbackDeps): { + canUseTool: CanUseTool + onUserDialog: OnUserDialog +} { + const canUseTool: CanUseTool = (toolName, input, options) => + new Promise((resolve) => { + const prompt = deps.prompts.register({ + requestId: options.requestId, + toolName, + toolUseId: options.toolUseID, + input, + suggestions: options.suggestions ?? [], + settle: resolve as (response: Record | null) => void + }) + if (!prompt) { + resolve(denySafeResult(options.toolUseID)) + return + } + const cancel = (): void => { + if (deps.prompts.forgetIfPending(prompt)) { + deps.emit({ + type: 'prompt-cancelled', + sessionId: deps.sessionId, + promptKey: prompt.promptKey + }) + // Null is the SDK's "no response written" sentinel: a cancelled request must not + // be answered, only forgotten. + resolve(null) + } + } + if (options.signal.aborted) { + // No abort event can still fire, so registering a listener would park the callback + // forever behind a prompt nothing will answer. + cancel() + return + } + options.signal.addEventListener('abort', cancel, { once: true }) + deps.emit({ type: 'prompt', sessionId: deps.sessionId, prompt }) + }) + + const onUserDialog: OnUserDialog = () => Promise.resolve({ behavior: 'cancelled' }) + + return { canUseTool, onUserDialog } +} diff --git a/src/main/claude/claude-structured-init-deadline.ts b/src/main/claude/claude-structured-init-deadline.ts new file mode 100644 index 00000000000..f3acd6c3af9 --- /dev/null +++ b/src/main/claude/claude-structured-init-deadline.ts @@ -0,0 +1,68 @@ +import type { ClaudeInitObservation } from './claude-structured-init-proof' +import { claudeInitializationAuthError } from './claude-structured-init-proof' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { AgentSessionAcquisitionRefusal } from '../native-chat/agent-session-wire/structured-agent-session-adapter' + +export type ClaudeInitDeadline = { + promise: Promise + resolve: (init: ClaudeInitObservation) => void + reject: (error: Error) => void + start: () => void + clear: () => void +} + +export function claudeInitTimeoutError( + sessionId: string, + timeoutMs: number +): AgentSessionAcquisitionRefusal { + return new AgentSessionAcquisitionRefusal( + `Claude did not finish starting session ${sessionId} within ${Math.ceil(timeoutMs / 1000)} seconds. Verify the selected Claude account is signed in and CLAUDE_CONFIG_DIR contains valid credentials, then retry; no SessionStart or system/init proof arrived.` + ) +} + +export async function requestClaudeInitialization( + connection: ClaudeStreamJsonConnection, + sessionId: string, + timeoutMs: number +): Promise { + try { + const result = await connection.initializationResult({ timeoutMs }) + const authError = claudeInitializationAuthError(result) + if (authError) { + throw authError + } + return result + } catch (error) { + if (error instanceof Error && error.message === 'claude initialize request timed out') { + throw claudeInitTimeoutError(sessionId, timeoutMs) + } + throw error + } +} + +export function createClaudeInitDeadline(sessionId: string, timeoutMs: number): ClaudeInitDeadline { + let resolve = (_init: ClaudeInitObservation): void => {} + let reject = (_error: Error): void => {} + const promise = new Promise((resolvePromise, rejectPromise) => { + resolve = resolvePromise + reject = rejectPromise + }) + void promise.catch(() => {}) + let timer: ReturnType | null = null + + return { + promise, + resolve, + reject, + start: () => { + timer = setTimeout(() => reject(claudeInitTimeoutError(sessionId, timeoutMs)), timeoutMs) + timer.unref?.() + }, + clear: () => { + if (timer) { + clearTimeout(timer) + timer = null + } + } + } +} diff --git a/src/main/claude/claude-structured-init-proof.ts b/src/main/claude/claude-structured-init-proof.ts new file mode 100644 index 00000000000..c29cb2d4715 --- /dev/null +++ b/src/main/claude/claude-structured-init-proof.ts @@ -0,0 +1,88 @@ +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import type { ClaudeAuthDiagnostic } from './claude-structured-session-state' +import { AgentSessionAcquisitionRefusal } from '../native-chat/agent-session-wire/structured-agent-session-adapter' + +export type ClaudeInitObservation = { + providerSessionId: string + uuid: string | null + /** The resolved model id the CLI reports it is running; only `system/init` carries it. */ + model: string | null + message: Record +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +export function readClaudeFrameString(source: Record, key: string): string | null { + const value = source[key] + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readClaudeInit(message: Record): ClaudeInitObservation | null { + const hookName = readClaudeFrameString(message, 'hook_name') + const isInit = message.type === 'system' && message.subtype === 'init' + const isSessionStart = + message.type === 'system' && + (message.subtype === 'hook_started' || message.subtype === 'hook_response') && + hookName?.startsWith('SessionStart:') === true + if (!isInit && !isSessionStart) { + return null + } + const providerSessionId = readClaudeFrameString(message, 'session_id') + return providerSessionId + ? { + providerSessionId, + uuid: isInit ? readClaudeFrameString(message, 'uuid') : null, + model: isInit ? readClaudeFrameString(message, 'model') : null, + message + } + : null +} + +export function readClaudeModels(initialization: unknown): unknown[] { + return isRecord(initialization) && Array.isArray(initialization.models) + ? initialization.models + : [] +} + +/** CLI capabilities advertised on the initialize result or the yielded system/init frame. */ +export function readClaudeCapabilities( + init: ClaudeInitObservation, + initialization: unknown +): string[] { + const fromResult = isRecord(initialization) ? initialization.capabilities : undefined + const fromFrame = init.message.capabilities + const source = Array.isArray(fromResult) ? fromResult : Array.isArray(fromFrame) ? fromFrame : [] + return source.filter((value): value is string => typeof value === 'string') +} + +export function claudeInitializationAuthError( + initialization: unknown +): AgentSessionAcquisitionRefusal | null { + const account = + isRecord(initialization) && isRecord(initialization.account) ? initialization.account : null + return readClaudeFrameString(account ?? {}, 'tokenSource') === 'none' + ? new AgentSessionAcquisitionRefusal( + 'Claude is not signed in for the selected account. Sign in with the Claude CLI for this CLAUDE_CONFIG_DIR, then retry.' + ) + : null +} + +export function claudeAuthDiagnostic( + init: ClaudeInitObservation, + settings: unknown +): ClaudeAuthDiagnostic { + const env = isRecord(settings) && isRecord(settings.env) ? settings.env : {} + const apiKeySource = readClaudeFrameString(init.message, 'apiKeySource') + const configured = (key: string): boolean => + (typeof env[key] === 'string' && (env[key] as string).trim().length > 0) || + Boolean(process.env[key]?.trim()) + return { + apiKeySourceConfigured: apiKeySource !== null && apiKeySource !== 'none', + baseUrlConfigured: configured('ANTHROPIC_BASE_URL'), + authTokenConfigured: configured('ANTHROPIC_AUTH_TOKEN'), + apiKeyConfigured: configured('ANTHROPIC_API_KEY'), + settingSources: CLAUDE_DEFAULT_SETTING_SOURCES + } +} diff --git a/src/main/claude/claude-structured-item-translation.ts b/src/main/claude/claude-structured-item-translation.ts new file mode 100644 index 00000000000..d86093ee0a5 --- /dev/null +++ b/src/main/claude/claude-structured-item-translation.ts @@ -0,0 +1,179 @@ +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentJournalMessageItem +} from '../../shared/agent-session-journal-types' +import type { NativeChatBlock } from '../../shared/native-chat-types' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' + +export type ClaudeMessageEnvelope = { + sessionId: string + uuid: string + role: 'assistant' | 'user' + content: unknown[] + /** Messages API id shared by every frame of one streamed assistant message. */ + messageId: string | null + parentToolUseId: string | null +} + +export type ClaudeToolUse = { id: string; name: string; input: unknown } +export type ClaudeToolResult = { toolUseId: string; output: string; failed: boolean } + +export function claudeRecord(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +export function claudeText(value: unknown): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +export function readClaudeMessageEnvelope( + frame: Record +): ClaudeMessageEnvelope | null { + if (frame.type !== 'assistant' && frame.type !== 'user') { + return null + } + const message = claudeRecord(frame.message) + const sessionId = claudeText(frame.session_id) + const uuid = claudeText(frame.uuid) + const role = message?.role + return sessionId && uuid && (role === 'assistant' || role === 'user') + ? { + sessionId, + uuid, + role, + content: messageContent(message?.content), + messageId: claudeText(message?.id), + parentToolUseId: claudeText(frame.parent_tool_use_id) + } + : null +} + +// A user replay may carry its text as a bare string (MessageParam), not blocks. +function messageContent(content: unknown): unknown[] { + if (Array.isArray(content)) { + return content + } + const text = claudeText(content) + return text ? [{ type: 'text', text }] : [] +} + +export function claudeMessageIdentity( + envelope: Pick +): AgentJournalItemIdentity { + return { provider: 'claude', sessionId: envelope.sessionId, uuid: envelope.uuid } +} + +function messageBlocks(envelope: ClaudeMessageEnvelope): NativeChatBlock[] { + const blocks: NativeChatBlock[] = [] + for (const value of envelope.content) { + const part = claudeRecord(value) + const text = claudeText(part?.text) + if (part?.type === 'text' && text) { + blocks.push({ type: 'text', text }) + continue + } + const source = claudeRecord(part?.source) + const url = claudeText(source?.url) + if (part?.type === 'image' && source?.type === 'url' && url) { + blocks.push({ type: 'image-ref', url }) + } + } + return blocks +} + +export function claudeMessageBody(envelope: ClaudeMessageEnvelope): AgentJournalMessageItem | null { + const blocks = messageBlocks(envelope) + return blocks.length > 0 ? { kind: 'message', role: envelope.role, blocks } : null +} + +export function claudeHasReplayContent(envelope: ClaudeMessageEnvelope): boolean { + return envelope.content.some((value) => { + const part = claudeRecord(value) + return part !== null && part.type !== 'tool_result' + }) +} + +export function claudeToolUses(envelope: ClaudeMessageEnvelope): ClaudeToolUse[] { + return envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const id = claudeText(part?.id) + const name = claudeText(part?.name) + return part?.type === 'tool_use' && id && name ? [{ id, name, input: part.input ?? null }] : [] + }) +} + +function resultText(value: unknown): string { + if (typeof value === 'string') { + return value + } + if (!Array.isArray(value)) { + return value === undefined ? '' : JSON.stringify(value) + } + return value + .flatMap((entry) => { + if (typeof entry === 'string') { + return [entry] + } + const part = claudeRecord(entry) + return part?.type === 'text' && typeof part.text === 'string' ? [part.text] : [] + }) + .join('\n') +} + +export function claudeToolResults(envelope: ClaudeMessageEnvelope): ClaudeToolResult[] { + return envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const toolUseId = claudeText(part?.tool_use_id) + return part?.type === 'tool_result' && toolUseId + ? [ + { + toolUseId, + output: resultText(part.content), + failed: part.is_error === true + } + ] + : [] + }) +} + +export function claudeThinkingText(envelope: ClaudeMessageEnvelope): string | null { + const parts = envelope.content.flatMap((value) => { + const part = claudeRecord(value) + const thinking = claudeText(part?.thinking) + return part?.type === 'thinking' && thinking ? [thinking] : [] + }) + return parts.length > 0 ? parts.join('\n') : null +} + +export function claudeToolBody(input: { + tool: ClaudeToolUse + result?: ClaudeToolResult +}): AgentJournalItemBody { + return { + kind: 'tool-call', + name: input.tool.name, + input: input.tool.input, + state: input.result ? (input.result.failed ? 'failed' : 'completed') : 'running', + ...(input.result + ? { output: boundInlineText(input.result.output, DEFAULT_JOURNAL_PAYLOAD_LIMITS).bounded } + : {}) + } +} + +export function claudeStreamingMessageBody(text: string): AgentJournalMessageItem { + return { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text }] } +} + +export function claudeToolIdentity(sessionId: string, toolUseId: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-tool:${sessionId}:${toolUseId}` } +} + +export function claudeThinkingIdentity(sessionId: string, uuid: string): AgentJournalItemIdentity { + return { provider: 'orca', clientMessageId: `claude-thinking:${sessionId}:${uuid}` } +} diff --git a/src/main/claude/claude-structured-journal-translation.test.ts b/src/main/claude/claude-structured-journal-translation.test.ts new file mode 100644 index 00000000000..f403313dae8 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation.test.ts @@ -0,0 +1,811 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentJournalItemIdentity, + AgentJournalRenderItem, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { activeStructuredAgentSessionTurnId } from '../../shared/structured-agent-session-projection' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { + createDeferredStructuredAgentSessionEventSink, + type StructuredAgentSessionEventSink +} from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import type { ClaudePendingPrompt } from './claude-structured-prompt-replies' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' + +function sinkState() { + const items: { identity: AgentJournalItemIdentity; body: AgentJournalItemBody }[] = [] + const tombstones: AgentJournalItemIdentity[] = [] + const sink: StructuredAgentSessionEventSink = { + appendItem: (identity, body) => items.push({ identity, body }), + appendTombstone: (identity) => tombstones.push(identity), + publish: vi.fn() + } + return { sink, items, tombstones } +} + +function message( + type: 'assistant' | 'user', + uuid: string, + content: unknown[], + parentToolUseId: string | null = null +) { + return { + type: 'message' as const, + sessionId: 'orca-session', + ...(type === 'user' && parentToolUseId === null ? { startsTurn: true as const } : {}), + message: { + type, + uuid, + session_id: 'claude-session', + parent_tool_use_id: parentToolUseId, + message: { role: type, content } + } + } +} + +// Frames below follow the Claude Code 2.1.258 / SDK 0.3.251 partial-message +// cadence captured from the real CLI: every stream_event carries its own uuid, +// the final assistant frame for a block carries yet another, and only +// message.id ties them together. +function streamEvent(uuid: string, event: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'stream_event', + uuid, + session_id: 'claude-session', + parent_tool_use_id: null, + event + } + } +} + +function resultFrame(subtype: string, fields: Record) { + return { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'result', + subtype, + duration_ms: 1200, + duration_api_ms: 1100, + num_turns: 1, + session_id: 'claude-session', + uuid: `result-${subtype}`, + ...fields + } + } +} + +/** One streamed text turn in wire order: message_start, the block's start frame, + * one delta per chunk, the block's final assistant frame, the stop frames and + * the success result. */ +function streamedTextTurn(input: { + messageId: string + startUuid: string + finalUuid: string + chunks: string[] +}) { + const text = input.chunks.join('') + return { + start: [ + streamEvent(`${input.messageId}-message-start`, { + type: 'message_start', + message: { id: input.messageId, role: 'assistant', content: [] } + }), + streamEvent(input.startUuid, { + type: 'content_block_start', + index: 0, + content_block: { type: 'text', text: '' } + }) + ], + deltas: input.chunks.map((chunk, index) => + streamEvent(`${input.messageId}-delta-${index}`, { + type: 'content_block_delta', + index: 0, + delta: { type: 'text_delta', text: chunk } + }) + ), + final: { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'assistant', + uuid: input.finalUuid, + session_id: 'claude-session', + parent_tool_use_id: null, + message: { + id: input.messageId, + role: 'assistant', + content: [{ type: 'text', text }], + stop_reason: null + } + } + }, + stop: [ + streamEvent(`${input.messageId}-block-stop`, { type: 'content_block_stop', index: 0 }), + streamEvent(`${input.messageId}-message-delta`, { + type: 'message_delta', + delta: { stop_reason: 'end_turn' } + }), + streamEvent(`${input.messageId}-message-stop`, { type: 'message_stop' }), + resultFrame('success', { + is_error: false, + result: text, + stop_reason: 'end_turn', + terminal_reason: 'completed' + }) + ], + text + } +} + +function assistantMessages(items: T[]): T[] { + return items.filter((item) => item.body.kind === 'message' && item.body.role === 'assistant') +} + +function providerFrameKinds(items: { body: AgentJournalItemBody }[]): string[] { + return items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame.kind] : [] + ) +} + +const JOURNAL_IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'claude-session', leafUuid: 'leaf-1' } +} + +let journalRoot = '' + +beforeEach(async () => { + journalRoot = await mkdtemp(join(tmpdir(), 'orca-claude-journal-translation-')) +}) + +afterEach(async () => { + await rm(journalRoot, { recursive: true, force: true }) +}) + +describe('Claude structured journal translation', () => { + it('coalesces partial deltas onto the block identity and reconciles the final frame onto it', () => { + const state = sinkState() + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: state.sink, + schedule: (run, delay) => { + expect(delay).toBe(60) + scheduled = run + return () => { + scheduled = null + } + } + }) + const turn = streamedTextTurn({ + messageId: 'msg_01', + startUuid: 'block-start-1', + finalUuid: 'assistant-final-1', + chunks: ['ST', 'REAMOK_ELEC_64E632'] + }) + const streamedIdentity = { + provider: 'claude', + sessionId: 'claude-session', + uuid: 'block-start-1' + } + + for (const event of turn.start) { + translator.handle(event) + } + for (const delta of turn.deltas) { + translator.handle(delta) + } + expect(state.items).toEqual([]) + + const run = scheduled as (() => void) | null + run?.() + expect(state.items.at(-1)).toEqual({ + identity: streamedIdentity, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: turn.text }] } + }) + + translator.handle(turn.final) + for (const event of turn.stop) { + translator.handle(event) + } + const assistant = assistantMessages(state.items) + expect(assistant.at(-1)).toEqual({ + identity: streamedIdentity, + body: { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: turn.text }] } + }) + expect(new Set(assistant.map((item) => agentJournalItemKey(item.identity))).size).toBe(1) + expect(providerFrameKinds(state.items)).toEqual([]) + }) + + it('journals a count-to-200 stream as one assistant item carrying the complete reply', async () => { + const journal = await openAgentSessionJournal({ + identity: JOURNAL_IDENTITY, + journalDir: journalRoot, + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ journal, fence: 1, publish: vi.fn() }) + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: deferred.sink, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + const numbers = Array.from({ length: 200 }, (_, index) => String(index + 1)) + // The chunk boundaries the real CLI produced for this prompt. + const boundaries = [0, 1, 45, 93, 141, 189, 200] + const chunks = boundaries.slice(1).map((end, index) => { + const slice = numbers.slice(boundaries[index], end).join('\n') + return index === 0 ? slice : `\n${slice}` + }) + const turn = streamedTextTurn({ + messageId: 'msg_count', + startUuid: 'count-start', + finalUuid: 'count-final', + chunks + }) + + for (const event of turn.start) { + translator.handle(event) + } + for (const delta of turn.deltas) { + translator.handle(delta) + // Each chunk lands in its own coalescing window, as it did on the wire. + const run = scheduled as (() => void) | null + run?.() + } + translator.handle(turn.final) + for (const event of turn.stop) { + translator.handle(event) + } + await deferred.drained() + + const items: AgentJournalRenderItem[] = journal.snapshot().items + const assistant = assistantMessages(items) + expect(assistant.map((item) => item.itemId)).toEqual(['claude:claude-session:count-start']) + expect(assistant[0]?.body).toEqual({ + kind: 'message', + role: 'assistant', + blocks: [{ type: 'text', text: numbers.join('\n') }] + }) + expect(providerFrameKinds(items)).toEqual([]) + }) + + it('settles result frames, empty thinking and string user replays without painting a row', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + startsTurn: true, + message: { + type: 'user', + uuid: 'user-replay-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + timestamp: '2026-09-01T00:00:00.000Z', + message: { role: 'user', content: 'Reply with exactly PROBE_OK_1 and nothing else.' } + } + }) + translator.handle( + message('assistant', 'assistant-thinking-empty', [ + { type: 'thinking', thinking: '', signature: 'CAQS6QcKEAgRGAI4AUIIdGhpbmtpbmc' } + ]) + ) + translator.handle( + resultFrame('success', { + is_error: false, + result: 'PROBE_OK_1', + stop_reason: 'end_turn', + terminal_reason: 'completed' + }) + ) + translator.handle( + message('user', 'user-interrupt', [{ type: 'text', text: '[Request interrupted by user]' }]) + ) + translator.handle( + resultFrame('error_during_execution', { + is_error: true, + errors: ['[ede_diagnostic] result_type=user last_content_type=n/a stop_reason=null'], + stop_reason: null, + terminal_reason: 'aborted_streaming', + permission_denials: [] + }) + ) + translator.handle(message('user', 'control-only', [])) + + expect(providerFrameKinds(state.items)).toEqual([]) + expect( + state.items.flatMap((item) => + item.body.kind === 'message' && item.body.role === 'user' ? [item.body.blocks] : [] + ) + ).toEqual([ + [{ type: 'text', text: 'Reply with exactly PROBE_OK_1 and nothing else.' }], + [{ type: 'text', text: '[Request interrupted by user]' }] + ]) + expect( + state.items.some((item) => item.body.kind === 'status' && !item.body.turnLifecycle) + ).toBe(false) + expect( + state.tombstones.flatMap((identity) => + identity.provider === 'legacy' ? [identity.recordId] : [] + ) + ).toEqual(['turn-lifecycle:user-replay-1', 'turn-lifecycle:user-interrupt']) + }) + + it('does not reopen a completed turn when the SDK replays its user row after restart', () => { + const live = sinkState() + const liveTranslator = createClaudeJournalTranslator({ sink: live.sink }) + const replay = { + type: 'message' as const, + sessionId: 'orca-session', + message: { + type: 'user', + uuid: 'picker-command-1', + session_id: 'claude-session', + parent_tool_use_id: null, + isReplay: true, + message: { role: 'user', content: '/model' } + } + } + + liveTranslator.handle({ ...replay, startsTurn: true }) + liveTranslator.handle(resultFrame('success', { is_error: false, result: '' })) + expect(live.tombstones).toContainEqual({ + provider: 'legacy', + agent: 'claude', + sessionId: 'claude-session', + recordId: 'turn-lifecycle:picker-command-1' + }) + liveTranslator.dispose() + + const restarted = sinkState() + const restartedTranslator = createClaudeJournalTranslator({ sink: restarted.sink }) + restartedTranslator.handle(replay) + + expect( + activeStructuredAgentSessionTurnId( + restarted.items.map((item, sequence) => ({ + itemId: agentJournalItemKey(item.identity), + revision: 1, + body: item.body, + sequence, + observedAt: sequence + })) + ) + ).toBeNull() + }) + + it('surfaces an API error carried by a success-subtype result with no assistant frame', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'user-1', [{ type: 'text', text: 'summarize this' }])) + // The SDK models this as a SUCCESS-subtype result whose `result` string is the + // user-facing API error. Suppressing it as ordinary turn bookkeeping ends the + // turn with nothing shown at all. + translator.handle( + resultFrame('success', { + is_error: true, + result: 'API Error: 529 upstream overloaded', + stop_reason: null, + terminal_reason: 'api_error' + }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:success']) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'status', + text: 'API Error: 529 upstream overloaded' + }) + // The turn still settles: the error is an extra row, not a stuck lifecycle. + expect( + state.tombstones.flatMap((identity) => + identity.provider === 'legacy' ? [identity.recordId] : [] + ) + ).toEqual(['turn-lifecycle:user-1']) + }) + + it('drops the stream state of turns that ended without their final frame', () => { + const state = sinkState() + let scheduled: (() => void) | null = null + const translator = createClaudeJournalTranslator({ + sink: state.sink, + schedule: (run) => { + scheduled = run + return () => { + scheduled = null + } + } + }) + for (let turn = 0; turn < 3; turn += 1) { + const aborted = streamedTextTurn({ + messageId: `msg_abort_${turn}`, + startUuid: `abort-start-${turn}`, + finalUuid: `abort-final-${turn}`, + chunks: ['x'.repeat(4_000)] + }) + for (const event of [...aborted.start, ...aborted.deltas]) { + translator.handle(event) + } + const run = scheduled as (() => void) | null + run?.() + // The user interrupts: the result arrives with no final assistant frame, + // so nothing ever reconciles these blocks. + translator.handle( + resultFrame('error_during_execution', { + is_error: true, + terminal_reason: 'aborted_streaming' + }) + ) + // The partial text is already journaled; only the live state is dropped. + expect(translator.pendingStreamedBlocks).toBe(0) + } + + expect(assistantMessages(state.items)).toHaveLength(3) + }) + + it('keeps an ordinary successful result off the timeline', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(resultFrame('success', { is_error: false, result: 'done', errors: [] })) + + expect(providerFrameKinds(state.items)).toEqual([]) + }) + + it('surfaces the reason an error-subtype result stopped the turn', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + resultFrame('error_max_turns', { is_error: true, errors: ['turn limit reached'] }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:error_max_turns']) + }) + + it('keeps an unmodeled result subtype on the bounded provider fallback', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + resultFrame('error_from_the_future', { is_error: true, errors: ['budget exhausted'] }) + ) + + expect(providerFrameKinds(state.items)).toEqual(['message:result:error_from_the_future']) + }) + + it('journals turn lifecycle and updates one tool row through its result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'user-1', [{ type: 'text', text: 'List files' }])) + translator.handle( + message('assistant', 'assistant-tool', [ + { type: 'tool_use', id: 'tool-1', name: 'Bash', input: { command: 'ls' } } + ]) + ) + translator.handle( + message( + 'user', + 'tool-result-1', + [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'a.ts\nb.ts' }], + 'tool-1' + ) + ) + + const keyed = new Map( + state.items.map((item) => [agentJournalItemKey(item.identity), item.body]) + ) + expect(keyed.get('claude:claude-session:user-1')).toMatchObject({ + kind: 'message', + role: 'user' + }) + expect(keyed.get('orca:claude-tool%3Aclaude-session%3Atool-1')).toMatchObject({ + kind: 'tool-call', + name: 'Bash', + state: 'completed', + output: { head: 'a.ts\nb.ts', truncated: false } + }) + expect( + state.items.some( + (item) => item.body.kind === 'status' && item.body.turnLifecycle?.turnId === 'user-1' + ) + ).toBe(true) + + translator.handle( + message( + 'user', + 'tool-result-2', + [{ type: 'tool_result', tool_use_id: 'tool-1', content: 'done again' }], + 'tool-1' + ) + ) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'tool-call', + name: 'tool', + input: null, + output: { head: 'done again' } + }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'result', session_id: 'claude-session', uuid: 'result-1' } + }) + expect(state.tombstones.at(-1)).toMatchObject({ + provider: 'legacy', + agent: 'claude', + recordId: 'turn-lifecycle:user-1' + }) + }) + + it('bounds persisted thinking text to the shared journal payload limit', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + const thinking = 'considering '.repeat(20_000) + + translator.handle(message('assistant', 'assistant-thinking', [{ type: 'thinking', thinking }])) + + expect(state.items.at(-1)?.body).toEqual({ + kind: 'status', + text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + }) + }) + + it('starts a cancellable lifecycle for image-only root user replays', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + message('user', 'user-image', [ + { type: 'image', source: { type: 'base64', media_type: 'image/png', data: 'AA==' } } + ]) + ) + + expect(state.items.at(-1)?.body).toEqual({ + kind: 'status', + text: 'Claude is working…', + turnLifecycle: { turnId: 'user-image', state: 'running' } + }) + }) + + it('does not start a lifecycle for a top-level user tool result', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle( + message('user', 'tool-result-only', [ + { type: 'tool_result', tool_use_id: 'tool-1', content: 'done' } + ]) + ) + + expect(state.items.map((item) => agentJournalItemKey(item.identity))).toEqual([ + 'orca:claude-tool%3Aclaude-session%3Atool-1' + ]) + expect(state.items[0]?.body).toMatchObject({ + kind: 'tool-call', + state: 'completed', + output: { head: 'done' } + }) + expect( + state.items.some( + (item) => item.body.kind === 'status' && item.body.turnLifecycle !== undefined + ) + ).toBe(false) + }) + + it('paints nothing for a user frame that carries no content', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle(message('user', 'control-only', [])) + + expect(state.items).toEqual([]) + expect(state.tombstones).toEqual([]) + }) + + it('renders unmodeled substantive Claude frames as bounded provider rows', () => { + const state = sinkState() + const translator = createClaudeJournalTranslator({ sink: state.sink }) + + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'local_command_output', summary: 'x'.repeat(100_000) } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'hook_response', hook_name: 'PostToolUse' } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'system', subtype: 'command_started', command: '/compact' } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'result', usage: { input_tokens: 12 }, total_cost_usd: 0.01 } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'tool_progress', tool_use_id: 'tool-1', elapsed_time_seconds: 2 } + }) + translator.handle({ + type: 'message', + sessionId: 'orca-session', + message: { type: 'prompt_suggestion', suggestion: '/compact' } + }) + translator.handle( + message('user', 'attachment-1', [ + { type: 'document', source: { type: 'base64', media_type: 'application/pdf' } } + ]) + ) + translator.handle({ + type: 'provider-frame', + sessionId: 'orca-session', + kind: 'control_request:future_control', + payload: { subtype: 'future_control' } + }) + + const frames = state.items.flatMap((item) => + item.body.kind === 'status' && item.body.providerFrame ? [item.body.providerFrame] : [] + ) + expect(frames.map((frame) => frame.kind)).toEqual( + expect.arrayContaining([ + 'message:system:local_command_output', + 'message:system:command_started', + 'message:result', + 'message:user:content:document', + 'control_request:future_control' + ]) + ) + expect(frames.map((frame) => frame.kind)).not.toEqual( + expect.arrayContaining([ + 'message:system:hook_response', + 'message:tool_progress', + 'message:prompt_suggestion' + ]) + ) + expect( + frames.find((frame) => frame.kind === 'message:system:local_command_output')?.payload + ).toEqual(expect.objectContaining({ truncated: true, byteLength: expect.any(Number) })) + }) + + it('preserves a question group as one addressable prompt and cancels it durably', () => { + const state = sinkState() + const bindings: unknown[][] = [] + const translator = createClaudeJournalTranslator({ + sink: state.sink, + bindPromptItemId: (...args) => bindings.push(args) + }) + const approval = prompt({ + requestId: 'permission-1', + promptKey: 'permission-1', + toolUseId: 'tool-1', + toolName: 'Bash', + kind: 'approval', + input: { command: 'git status' }, + questionIds: [] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: approval }) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'approval', + title: 'Allow Bash?', + options: expect.arrayContaining([{ id: 'allow', label: 'Allow' }]) + }) + expect(bindings[0]).toEqual([ + 'orca:claude-prompt%3Aorca-session%3Apermission-1', + 'permission-1' + ]) + + const questions = prompt({ + requestId: 'questions-1', + promptKey: 'questions-1', + toolUseId: 'tool-q', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { question: 'Library?', options: [{ label: 'Luxon' }] }, + { question: 'Ship?', options: [{ label: 'Yes' }] } + ] + }, + questionIds: ['Library?', 'Ship?'] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: questions }) + expect(state.items.filter((item) => item.body.kind === 'question')).toHaveLength(1) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'question', + questions: [ + { id: 'q1', question: 'Library?', multiSelect: false }, + { id: 'q2', question: 'Ship?', multiSelect: false } + ] + }) + expect(bindings.at(-1)).toEqual([ + 'orca:claude-prompt%3Aorca-session%3Aquestions-1', + 'questions-1' + ]) + + const multiSelect = prompt({ + requestId: 'questions-multi', + promptKey: 'questions-multi', + toolUseId: 'tool-multi', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { + question: 'Libraries?', + multiSelect: true, + options: [{ label: 'Luxon' }, { label: 'Temporal' }] + } + ] + }, + questionIds: ['Libraries?'] + }) + translator.handle({ type: 'prompt', sessionId: 'orca-session', prompt: multiSelect }) + expect(state.items.at(-1)?.body).toMatchObject({ + kind: 'question', + question: '1 grouped question from Claude', + options: [], + questions: [ + { + id: 'q1', + question: 'Libraries?', + multiSelect: true, + options: [{ label: 'Luxon' }, { label: 'Temporal' }], + freeTextQuestionId: 'q1' + } + ] + }) + + translator.handle({ + type: 'prompt-cancelled', + sessionId: 'orca-session', + promptKey: 'questions-1' + }) + expect(state.tombstones).toHaveLength(1) + }) +}) + +function prompt( + input: Pick< + ClaudePendingPrompt, + 'requestId' | 'promptKey' | 'toolUseId' | 'toolName' | 'kind' | 'input' | 'questionIds' + > +): ClaudePendingPrompt { + return { + ...input, + suggestions: [], + answers: new Map(), + settle: () => {} + } +} diff --git a/src/main/claude/claude-structured-journal-translation.ts b/src/main/claude/claude-structured-journal-translation.ts new file mode 100644 index 00000000000..ffaad4da570 --- /dev/null +++ b/src/main/claude/claude-structured-journal-translation.ts @@ -0,0 +1,287 @@ +import type { AgentJournalItemIdentity } from '../../shared/agent-session-journal-types' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import type { AgentSessionDeltaCoalescerDeps } from '../native-chat/agent-session-wire/agent-session-delta-coalescer' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' +import { + claudeMessageBody, + claudeMessageIdentity, + claudeHasReplayContent, + claudeRecord, + claudeStreamingMessageBody, + claudeText, + claudeThinkingIdentity, + claudeThinkingText, + claudeToolBody, + claudeToolIdentity, + claudeToolResults, + claudeToolUses, + readClaudeMessageEnvelope, + type ClaudeToolUse +} from './claude-structured-item-translation' +import { + claudeApprovalItem, + claudePromptIdentity, + claudeQuestionItems +} from './claude-structured-prompt-items' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { readableProviderFrameText } from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { + CLAUDE_UNRENDERABLE_CONTENT_TEXT, + claudeProviderFrameKind, + claudeResultFailure, + createClaudeProviderFrameFallback, + isModeledClaudeContent, + isSettledClaudeResultKind +} from './claude-structured-provider-fallback' +import { createClaudeStreamedBlockRegistry } from './claude-streamed-block-identity' +import { createClaudeStreamedTextCheckpoints } from './claude-streamed-text-checkpoints' + +export type ClaudeJournalTranslatorDeps = { + sink: StructuredAgentSessionEventSink + bindPromptItemId?: (journalItemId: string, promptKey: string, questionId?: string) => void + coalesceMs?: number + schedule?: AgentSessionDeltaCoalescerDeps['schedule'] + fallbackIdPrefix?: string +} + +export type ClaudeJournalTranslator = { + handle: (event: ClaudeStructuredSessionEvent) => void + flush: () => void + /** Streamed blocks still awaiting a final frame. A settled turn leaves none. */ + readonly pendingStreamedBlocks: number + dispose: () => void +} + +export function createClaudeSessionJournalTranslator( + sink: StructuredAgentSessionEventSink | undefined, + prompts: ClaudePromptRegistry, + fallbackIdPrefix: string +): ClaudeJournalTranslator | null { + return sink + ? createClaudeJournalTranslator({ + sink, + fallbackIdPrefix, + bindPromptItemId: (itemId, promptKey, questionId) => + prompts.bindJournalItemId(itemId, promptKey, questionId) + }) + : null +} + +function lifecycleIdentity(sessionId: string, turnId: string): AgentJournalItemIdentity { + return { + provider: 'legacy', + agent: 'claude', + sessionId, + recordId: `turn-lifecycle:${turnId}` + } +} + +export function createClaudeJournalTranslator( + deps: ClaudeJournalTranslatorDeps +): ClaudeJournalTranslator { + const tools = new Map() + const promptItems = new Map() + const streamedBlocks = createClaudeStreamedBlockRegistry() + let currentTurn: { sessionId: string; turnId: string } | null = null + const providerFallback = createClaudeProviderFrameFallback( + deps.sink, + deps.fallbackIdPrefix ?? 'acquisition' + ) + const streamedText = createClaudeStreamedTextCheckpoints({ + ...(deps.coalesceMs === undefined ? {} : { coalesceMs: deps.coalesceMs }), + ...(deps.schedule ? { schedule: deps.schedule } : {}), + persist: (identity, text) => { + deps.sink.appendItem(identity, claudeStreamingMessageBody(text)) + deps.sink.publish() + } + }) + + const publishLifecycle = (sessionId: string, turnId: string, running: boolean): void => { + const identity = lifecycleIdentity(sessionId, turnId) + if (running) { + deps.sink.appendItem(identity, { + kind: 'status', + text: 'Claude is working…', + turnLifecycle: { turnId, state: 'running' } + }) + } else { + deps.sink.appendTombstone(identity) + } + deps.sink.publish() + } + + const handleStream = (message: Record): boolean => { + const delta = streamedBlocks.observe(message) + if (!delta) { + return false + } + streamedText.append(delta.identity, delta.text) + return true + } + + const handleMessage = (message: Record, startsTurn: boolean): boolean => { + const envelope = readClaudeMessageEnvelope(message) + if (!envelope) { + return false + } + let changed = false + const body = claudeMessageBody(envelope) + // The final frame of a streamed block lands on the block's identity, not its own uuid. + const identity = + (body && envelope.role === 'assistant' ? streamedBlocks.reconcile(envelope) : null) ?? + claudeMessageIdentity(envelope) + streamedText.forget(agentJournalItemKey(identity)) + if (body) { + deps.sink.appendItem(identity, body) + changed = true + } + for (const tool of claudeToolUses(envelope)) { + tools.set(tool.id, tool) + deps.sink.appendItem( + claudeToolIdentity(envelope.sessionId, tool.id), + claudeToolBody({ tool }) + ) + changed = true + } + for (const result of claudeToolResults(envelope)) { + const tool = tools.get(result.toolUseId) ?? { + id: result.toolUseId, + name: 'tool', + input: null + } + deps.sink.appendItem( + claudeToolIdentity(envelope.sessionId, result.toolUseId), + claudeToolBody({ tool, result }) + ) + // Tool inputs are only needed until their matching result arrives. + tools.delete(result.toolUseId) + changed = true + } + const thinking = claudeThinkingText(envelope) + if (thinking) { + deps.sink.appendItem(claudeThinkingIdentity(envelope.sessionId, envelope.uuid), { + kind: 'status', + text: boundInlineText(thinking, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + }) + changed = true + } + const unhandledContent = envelope.content.filter((part) => !isModeledClaudeContent(part)) + for (const part of unhandledContent) { + const partType = claudeText(claudeRecord(part)?.type) ?? 'unknown' + providerFallback.append( + `message:${envelope.role}:content:${partType}`, + part, + readableProviderFrameText(part) ?? CLAUDE_UNRENDERABLE_CONTENT_TEXT + ) + changed = true + } + // An empty user frame is a replay with nothing to show, not an unknown kind. + if (envelope.content.length === 0 && envelope.role === 'assistant') { + providerFallback.append(`message:${envelope.role}:empty`, message) + changed = true + } + if ( + envelope.role === 'user' && + startsTurn && + claudeHasReplayContent(envelope) && + message.parent_tool_use_id === null + ) { + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + } + currentTurn = { sessionId: envelope.sessionId, turnId: envelope.uuid } + publishLifecycle(envelope.sessionId, envelope.uuid, true) + } + if (changed) { + deps.sink.publish() + } + return true + } + + const handlePrompt = (event: Extract): void => { + const identities: AgentJournalItemIdentity[] = [] + if (event.prompt.kind === 'question') { + for (const question of claudeQuestionItems({ + sessionId: event.sessionId, + prompt: event.prompt + })) { + identities.push(question.identity) + deps.sink.appendItem(question.identity, question.body) + deps.bindPromptItemId?.(agentJournalItemKey(question.identity), event.prompt.promptKey) + } + } else { + const identity = claudePromptIdentity({ + sessionId: event.sessionId, + promptKey: event.prompt.promptKey + }) + identities.push(identity) + deps.sink.appendItem(identity, claudeApprovalItem(event.prompt)) + deps.bindPromptItemId?.(agentJournalItemKey(identity), event.prompt.promptKey) + } + promptItems.set(event.prompt.promptKey, identities) + deps.sink.publish() + } + + return { + handle: (event) => { + if (event.type === 'ended') { + streamedText.flush() + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + currentTurn = null + } + return + } + if (event.type === 'message' && handleStream(event.message)) { + return + } + streamedText.flush() + if (event.type === 'prompt') { + handlePrompt(event) + } else if (event.type === 'prompt-cancelled') { + for (const identity of promptItems.get(event.promptKey) ?? []) { + deps.sink.appendTombstone(identity) + } + promptItems.delete(event.promptKey) + deps.sink.publish() + } else if (event.type === 'message' && event.message.type === 'result') { + if (currentTurn) { + publishLifecycle(currentTurn.sessionId, currentTurn.turnId, false) + currentTurn = null + } + // The turn is over. A block still awaiting its final keeps the text the + // flush above journaled, but its live state goes: an interrupted turn + // would otherwise retain that text for the life of the session. + streamedBlocks.clear() + streamedText.settle() + const kind = claudeProviderFrameKind(event.message) + // Ordinary turn bookkeeping stays suppressed; a reported failure never does. + const failure = claudeResultFailure(event.message) + if (failure || !isSettledClaudeResultKind(kind)) { + providerFallback.append(kind, event.message, failure?.text) + } + } else if (event.type === 'message') { + if (!handleMessage(event.message, event.startsTurn === true)) { + providerFallback.append(claudeProviderFrameKind(event.message), event.message) + } + } else if (event.type === 'provider-frame') { + providerFallback.append(event.kind, event.payload) + } + }, + flush: streamedText.flush, + get pendingStreamedBlocks() { + return streamedText.pending + }, + dispose: () => { + streamedText.dispose() + tools.clear() + promptItems.clear() + streamedBlocks.clear() + } + } +} diff --git a/src/main/claude/claude-structured-launch-resolution.test.ts b/src/main/claude/claude-structured-launch-resolution.test.ts new file mode 100644 index 00000000000..650947cffa1 --- /dev/null +++ b/src/main/claude/claude-structured-launch-resolution.test.ts @@ -0,0 +1,392 @@ +import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' +import { + CLAUDE_DEFAULT_SETTING_SOURCES, + CLAUDE_STRUCTURED_BASE_OPTIONS, + claudeSdkOptionsForLaunchArgs, + claudeSessionIdForOrcaSession, + createClaudeStructuredLaunchResolver +} from './claude-structured-launch-resolution' + +const SESSION_ID = 'orca-session-1' +const IDENTITY = { sessionId: SESSION_ID } as Parameters< + ReturnType +>[0]['identity'] + +function record(overrides: Partial = {}): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: LOCAL_EXECUTION_HOST_ID, + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/work/.claude' }, + providerHandleChain: [], + ...overrides + } as AgentSessionRecord +} + +function identityAt(leafUuid: string | null): typeof IDENTITY { + return { + ...IDENTITY, + providerHandle: { kind: 'claude', sessionId: 'provider-current', leafUuid } + } +} + +function makeExecutable(path: string): void { + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, '') + if (process.platform !== 'win32') { + chmodSync(path, 0o755) + } +} + +function resolverFor( + value: AgentSessionRecord | null, + resolveEnv?: () => Record, + stripAuthEnv = false +) { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => value } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv }), + ...(resolveEnv ? { resolveEnv } : {}) + }) +} + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +/** The normalized steady state of a Windows user whose only Claude account is WSL-managed: the + * prune drops the WSL account out of the host slot and persists that. */ +const WSL_ONLY_NORMALIZED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +const RESUMABLE = record({ + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: 'provider-current', leafUuid: 'leaf-current' } } + ] as AgentSessionRecord['providerHandleChain'] +}) + +describe('claude structured launch resolution', () => { + it('pre-mints a stable provider id and pins interactive setting sources', async () => { + const first = await resolverFor(record())({ identity: IDENTITY }) + const second = await resolverFor(record())({ identity: IDENTITY }) + + expect(first.providerSessionId).toBe(claudeSessionIdForOrcaSession(SESSION_ID)) + expect(second.providerSessionId).toBe(first.providerSessionId) + expect(first).toMatchObject({ + pathToClaudeCodeExecutable: '/usr/local/bin/claude', + cwd: '/repos/workspace-1', + claudeConfigDir: '/home/work/.claude', + resumeLeafUuid: null, + resumed: false + }) + expect(first.options).toEqual({ + includePartialMessages: true, + settingSources: [...CLAUDE_DEFAULT_SETTING_SOURCES], + supportedDialogKinds: [], + extraArgs: { 'replay-user-messages': null }, + systemPrompt: { type: 'preset', preset: 'claude_code' }, + sessionId: first.providerSessionId + }) + expect(first.options.resume).toBeUndefined() + expect(CLAUDE_STRUCTURED_BASE_OPTIONS.includePartialMessages).toBe(true) + }) + + it('resumes the session and leaf at the durable chain head', async () => { + const launch = await resolverFor( + record({ + providerHandleChain: [ + { handle: { provider: 'claude', sessionId: 'provider-old', leafUuid: 'leaf-old' } }, + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: 'leaf-current' + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + )({ identity: identityAt('leaf-current') }) + + expect(launch).toMatchObject({ + providerSessionId: 'provider-current', + resumeLeafUuid: 'leaf-current', + resumed: true + }) + expect(launch.options.resume).toBe('provider-current') + expect(launch.options.resumeSessionAt).toBe('leaf-current') + expect(launch.options.sessionId).toBeUndefined() + }) + + it('refuses a durable journal leaf that diverged before resume resolution', async () => { + const resolve = resolverFor( + record({ + providerHandleChain: [ + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: 'leaf-current' + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + ) + + await expect(resolve({ identity: identityAt('leaf-stale') })).rejects.toThrow( + 'durable resume identity changed before spawn' + ) + }) + + it('keeps session-only resume when the durable handle has no leaf', async () => { + const launch = await resolverFor( + record({ + providerHandleChain: [ + { + handle: { + provider: 'claude', + sessionId: 'provider-current', + leafUuid: null + } + } + ] as AgentSessionRecord['providerHandleChain'] + }) + )({ identity: identityAt(null) }) + + expect(launch.options.resume).toBe('provider-current') + expect(launch.options.resumeSessionAt).toBeUndefined() + }) + + it('preserves durable Claude launch arguments as typed options and extraArgs', async () => { + const launch = await resolverFor( + record({ + launchArgs: [ + '--model', + 'claude-sonnet-4-5', + '--effort', + 'high', + '--dangerously-skip-permissions' + ] + }) + )({ identity: IDENTITY }) + + expect(launch.options.model).toBe('claude-sonnet-4-5') + expect(launch.options.effort).toBe('high') + expect(launch.options.extraArgs).toEqual({ + 'dangerously-skip-permissions': null, + 'replay-user-messages': null + }) + }) + + it('routes durable launch arguments to a typed option first and refuses what neither can carry', () => { + // The catalog's own output: each flag lands in exactly one place, so the SDK + // cannot emit it twice with two different values. + expect(claudeSdkOptionsForLaunchArgs(['--model', 'opus', '--effort', 'xhigh'])).toEqual({ + model: 'opus', + effort: 'xhigh' + }) + // An effort the SDK's union does not name still reaches the CLI, unchanged. + expect(claudeSdkOptionsForLaunchArgs(['--effort', 'ultra'])).toEqual({ + extraArgs: { effort: 'ultra' } + }) + expect(claudeSdkOptionsForLaunchArgs(['--settings=/tmp/s.json'])).toEqual({ + extraArgs: { settings: '/tmp/s.json' } + }) + expect(() => claudeSdkOptionsForLaunchArgs(['-m', 'opus'])).toThrow(/no SDK option/) + }) + + it('keeps the session launch environment pinned after account settings change', async () => { + const resolver = resolverFor(record(), () => ({ + ANTHROPIC_AUTH_TOKEN: 'rotated-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + })) + + expect((await resolver({ identity: IDENTITY })).env).toMatchObject({ + ANTHROPIC_AUTH_TOKEN: 'rotated-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + }) + expect((await resolver({ identity: IDENTITY })).env?.ANTHROPIC_AUTH_TOKEN).toBe('rotated-token') + }) + + // Stripping is the managed-account rule the terminal preflight computes at + // runtime-auth-preparation.ts:72; claude-structured-auth-parity.test.ts covers + // the system-auth half, where the user's own key has to survive. + it('strips ambient Anthropic auth under a managed account but keeps the rest of the env', async () => { + const restore = { + ANTHROPIC_API_KEY: process.env.ANTHROPIC_API_KEY, + ANTHROPIC_AUTH_TOKEN: process.env.ANTHROPIC_AUTH_TOKEN, + CLAUDE_CODE_OAUTH_TOKEN: process.env.CLAUDE_CODE_OAUTH_TOKEN, + ORCA_LAUNCH_RESOLUTION_MARKER: process.env.ORCA_LAUNCH_RESOLUTION_MARKER + } + process.env.ANTHROPIC_API_KEY = 'sk-ant-SHELL-LEAK' + process.env.ANTHROPIC_AUTH_TOKEN = 'tok-SHELL-LEAK' + process.env.CLAUDE_CODE_OAUTH_TOKEN = 'oauth-SHELL-LEAK' + process.env.ORCA_LAUNCH_RESOLUTION_MARKER = 'inherited' + try { + const launch = await resolverFor(record(), undefined, true)({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBeUndefined() + expect(launch.env?.ANTHROPIC_AUTH_TOKEN).toBeUndefined() + expect(launch.env?.CLAUDE_CODE_OAUTH_TOKEN).toBeUndefined() + // The inherited env is still the base — only auth is removed from it. + expect(launch.env?.ORCA_LAUNCH_RESOLUTION_MARKER).toBe('inherited') + expect(launch.env?.PATH ?? launch.env?.Path).toBeTruthy() + } finally { + for (const [key, value] of Object.entries(restore)) { + if (value === undefined) { + delete process.env[key] + } else { + process.env[key] = value + } + } + } + }) + + it('lets an explicit Claude env overlay override ambient auth under system auth', async () => { + const restore = process.env.ANTHROPIC_API_KEY + process.env.ANTHROPIC_API_KEY = 'sk-ant-SHELL-LEAK' + try { + const launch = await resolverFor(record(), () => ({ + ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' + }))({ identity: IDENTITY }) + + expect(launch.env?.ANTHROPIC_API_KEY).toBe('sk-ant-CONFIGURED') + } finally { + if (restore === undefined) { + delete process.env.ANTHROPIC_API_KEY + } else { + process.env.ANTHROPIC_API_KEY = restore + } + } + }) + + it('pairs a resolved Claude CLI with its sibling Node runtime', async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-claude-launch-')) + const binDir = join(root, 'bin') + const claudeCommand = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + const nodeCommand = join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node') + makeExecutable(claudeCommand) + makeExecutable(nodeCommand) + + const launch = await createClaudeStructuredLaunchResolver({ + store: { getRecord: () => record() } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => claudeCommand, + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ + PATH: '/usr/bin', + CLAUDE_CONFIG_DIR: '/accounts/selected/home' + }) + })({ identity: IDENTITY }) + + expect((launch.env?.PATH ?? launch.env?.Path)?.split(delimiter)[0]).toBe(binDir) + }) + + it('refuses other hosts, WSL, providers, and account-home variables', async () => { + await expect( + resolverFor(record({ location: { ...record().location, executionHostId: 'ssh:build' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/local host/) + await expect( + resolverFor(record({ location: { ...record().location, wslDistro: 'Ubuntu' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/local host/) + await expect( + resolverFor(record({ provider: 'codex' } as Partial))({ + identity: IDENTITY + }) + ).rejects.toThrow(/codex session/) + await expect( + resolverFor(record({ accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex' } }))({ + identity: IDENTITY + }) + ).rejects.toThrow(/CLAUDE_CONFIG_DIR/) + }) + + /** The account state can change while a session lives, and a reacquire after an unexpected child + * exit re-resolves the launch. Without the gate here, that reacquire spawns under whatever the + * account state has become. */ + describe('managed-account gate on every acquisition', () => { + function resolverWithGate(read: () => ClaudeManagedAccountGateSettings | null) { + return createClaudeStructuredLaunchResolver({ + store: { getRecord: () => RESUMABLE } as unknown as AgentSessionRecordStore, + resolveWorkspacePath: async (id) => `/repos/${id}`, + resolveCommand: () => '/usr/local/bin/claude', + // Derived, not a literal: the gate and the policy must read the SAME account state, so a + // hardcoded value could assert a pairing production cannot produce. + resolveAuthPolicy: () => { + const settings = read() + if (!settings) { + throw new Error('the gate refuses before the auth policy is computed') + } + return claudeStructuredAuthPolicyForSettings(settings) + }, + readManagedAccountGate: read + }) + } + + it('refuses a reacquire once the account state becomes the refused shape', async () => { + let gate: ClaudeManagedAccountGateSettings | null = HOST_SELECTED + const resolve = resolverWithGate(() => gate) + + // Created while supported: the launch resolves and would spawn. + await expect(resolve({ identity: identityAt('leaf-current') })).resolves.toMatchObject({ + providerSessionId: 'provider-current' + }) + + gate = WSL_ONLY_NORMALIZED + + // Reacquire after the account state changed: refused before anything spawns. + await expect(resolve({ identity: identityAt('leaf-current') })).rejects.toBeInstanceOf( + AgentSessionPreSpawnError + ) + }) + + it('fails closed when the account state cannot be read', async () => { + await expect( + resolverWithGate(() => null)({ identity: identityAt('leaf-current') }) + ).rejects.toBeInstanceOf(AgentSessionPreSpawnError) + }) + + it('keeps resolving when no gate is wired, so other embedders are unaffected', async () => { + await expect( + resolverFor(RESUMABLE)({ identity: identityAt('leaf-current') }) + ).resolves.toMatchObject({ providerSessionId: 'provider-current' }) + }) + }) +}) diff --git a/src/main/claude/claude-structured-launch-resolution.ts b/src/main/claude/claude-structured-launch-resolution.ts new file mode 100644 index 00000000000..4f28f14ad65 --- /dev/null +++ b/src/main/claude/claude-structured-launch-resolution.ts @@ -0,0 +1,273 @@ +import { createHash } from 'node:crypto' +import type { EffortLevel, Options as ClaudeAgentSdkOptions } from '@anthropic-ai/claude-agent-sdk' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { agentSessionProviderHandleChainHead } from '../../shared/agent-session-provider-handle' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + applyClaudeEnvPatch, + hasClaudeAuthEnvConflict +} from '../claude-accounts/environment' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { + CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS, + whenClaudeAuthSwitchSettles +} from '../claude-accounts/live-pty-gate' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' +import { resolveClaudeCommand } from '../codex-cli/command' +import type { AgentSessionRecordStore } from '../runtime/agent-session-record-store' + +export const CLAUDE_DEFAULT_SETTING_SOURCES = ['user', 'project', 'local'] as const + +export type ClaudeStructuredSdkOptions = Pick< + ClaudeAgentSdkOptions, + | 'includePartialMessages' + | 'systemPrompt' + | 'settingSources' + | 'supportedDialogKinds' + | 'extraArgs' + | 'model' + | 'effort' + | 'sessionId' + | 'resume' + | 'resumeSessionAt' +> + +/** + * The options translation of the flags this transport used to build by hand. + * + * `-p`, `--input-format`, `--output-format` and `--verbose` are implied by + * `query()`; `--permission-prompt-tool stdio` is emitted because a `canUseTool` + * callback is supplied. `--replay-user-messages` has no option — the SDK never + * emits it — and Orca's send acknowledgement depends on the replay. + */ +export const CLAUDE_STRUCTURED_BASE_OPTIONS: ClaudeStructuredSdkOptions = { + includePartialMessages: true, + // Keep the SDK on Claude Code's own system-prompt contract. + systemPrompt: { type: 'preset', preset: 'claude_code' }, + settingSources: [...CLAUDE_DEFAULT_SETTING_SOURCES], + supportedDialogKinds: [], + extraArgs: { 'replay-user-messages': null } +} + +const EFFORT_LEVELS: readonly string[] = ['low', 'medium', 'high', 'xhigh', 'max'] + +function cloneDefinedEnv(env: NodeJS.ProcessEnv | Record): Record { + const next: Record = {} + for (const [key, value] of Object.entries(env)) { + if (value !== undefined) { + next[key] = value + } + } + return next +} + +/** + * Translate the record's durable launch arguments into SDK options. + * + * Typed option first so a flag is never emitted twice; `extraArgs` carries + * anything without one. A token expressible neither way is refused rather than + * dropped — a silent drop is how this lane loses launch flags. + */ +export function claudeSdkOptionsForLaunchArgs( + args: readonly string[] +): Pick { + let model: string | undefined + let effort: EffortLevel | undefined + const extraArgs: Record = {} + for (let index = 0; index < args.length; index += 1) { + const token = args[index] ?? '' + if (!token.startsWith('--') || token.length <= 2) { + throw new Error( + `claude launch argument ${token} has no SDK option; refusing rather than dropping it` + ) + } + const equals = token.indexOf('=') + const flag = equals === -1 ? token : token.slice(0, equals) + let value = equals === -1 ? null : token.slice(equals + 1) + if (value === null) { + const next = args[index + 1] + if (next !== undefined && !next.startsWith('-')) { + value = next + index += 1 + } + } + if (flag === '--model' && value !== null) { + model = value + } else if (flag === '--effort' && value !== null && EFFORT_LEVELS.includes(value)) { + effort = value as EffortLevel + } else { + extraArgs[flag.slice(2)] = value + } + } + return { + ...(model === undefined ? {} : { model }), + ...(effort === undefined ? {} : { effort }), + ...(Object.keys(extraArgs).length > 0 ? { extraArgs } : {}) + } +} + +export type ClaudeStructuredLaunch = { + /** Always Orca's resolved user CLI: the SDK's bundled binaries are excluded from the install. */ + pathToClaudeCodeExecutable: string + options: ClaudeStructuredSdkOptions + cwd: string + env?: Record + claudeConfigDir: string + providerSessionId: string + resumeLeafUuid: string | null + resumed: boolean +} + +export type ClaudeStructuredLaunchResolverDeps = { + store: AgentSessionRecordStore + resolveWorkspacePath: (workspaceId: string) => Promise + resolveCommand?: () => string + resolveEnv?: () => + | Promise | undefined> + | Record + | undefined + /** + * Required, and deliberately not defaulted. `stripAuthEnv` used to be a literal + * `true` here, so a missing dependency could not under-strip. Now it can, and the + * failure is silent — so every caller states the account's policy rather than + * inherit a guess. Build it with claudeStructuredAuthPolicyForSettings. + */ + resolveAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + /** How long an in-flight account switch may hold a launch before it is refused. */ + authSwitchSettleTimeoutMs?: number + /** Account state for the managed-account gate; null when it cannot be read, which refuses. */ + readManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null +} + +/** + * Wait a running account switch out, and refuse only if it never settles. + * + * Launch resolution is reached from `acquireClaudeSession` *after* the old child has + * been closed and proved, so a plain refusal here would leave the user with a dead + * chat and no replacement — the very harm the acquire-entry guard exists to prevent. + * The entry guard still refuses outright, because nothing has been torn down yet. + */ +export async function assertClaudeAuthSwitchSettled( + timeoutMs = CLAUDE_AUTH_SWITCH_SETTLE_TIMEOUT_MS +): Promise { + if (!(await whenClaudeAuthSwitchSettles(timeoutMs))) { + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) + } +} + +export function claudeSessionIdForOrcaSession(sessionId: string): string { + const bytes = createHash('sha256').update(`orca-claude:${sessionId}`).digest().subarray(0, 16) + bytes[6] = ((bytes[6] ?? 0) & 0x0f) | 0x40 + bytes[8] = ((bytes[8] ?? 0) & 0x3f) | 0x80 + const hex = bytes.toString('hex') + return `${hex.slice(0, 8)}-${hex.slice(8, 12)}-${hex.slice(12, 16)}-${hex.slice(16, 20)}-${hex.slice(20)}` +} + +export function createClaudeStructuredLaunchResolver( + deps: ClaudeStructuredLaunchResolverDeps +): (input: { identity: AgentSessionJournalIdentity }) => Promise { + return async ({ identity }) => { + await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) + const record = deps.store.getRecord(identity.sessionId) + if (!record) { + throw new Error(`no durable agent-session record for ${identity.sessionId}`) + } + if (record.provider !== 'claude') { + throw new Error(`session ${identity.sessionId} is a ${record.provider} session`) + } + if ( + record.location.executionHostId !== LOCAL_EXECUTION_HOST_ID || + record.location.wslDistro !== null + ) { + throw new Error( + `claude structured sessions run on the local host, not ${record.location.executionHostId}` + ) + } + if (record.accountHome.variable !== 'CLAUDE_CONFIG_DIR') { + throw new Error(`claude sessions pin CLAUDE_CONFIG_DIR, not ${record.accountHome.variable}`) + } + // Every acquisition, not just the first: the account state can change under a live session, and + // a reacquire after an unexpected exit would otherwise spawn under whatever it has become. + // Codex has no gate here — it resolves its account on a different path. + if ( + deps.readManagedAccountGate && + !structuredClaudeMatchesActiveManagedAccount(deps.readManagedAccountGate()) + ) { + throw new AgentSessionPreSpawnError( + 'structured Claude is not offered under the active managed Claude account' + ) + } + const head = agentSessionProviderHandleChainHead(record.providerHandleChain) + if ( + head?.handle.provider === 'claude' && + (identity.providerHandle.kind !== 'claude' || + identity.providerHandle.sessionId !== head.handle.sessionId || + identity.providerHandle.leafUuid !== head.handle.leafUuid) + ) { + throw new Error('claude durable resume identity changed before spawn') + } + const providerSessionId = + head?.handle.provider === 'claude' + ? head.handle.sessionId + : claudeSessionIdForOrcaSession(identity.sessionId) + const durable = claudeSdkOptionsForLaunchArgs(record.launchArgs ?? []) + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const auth = await deps.resolveAuthPolicy() + const overlay = await deps.resolveEnv?.() + // A switch can begin while the policy and overlay resolve, exactly as it can + // during the terminal preflight's prepareClaudeAuth — recheck after the awaits. + await assertClaudeAuthSwitchSettled(deps.authSwitchSettleTimeoutMs) + // Under a managed account the pinned credential is the only auth this launch may + // use, so an explicit override is refused rather than silently beating the pin. + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(overlay)) { + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + } + // Why the overlay merges onto the inherited env rather than replacing it: the child + // still needs PATH and the rest of the shell environment, and withCliRuntimeOnPath + // derives PATH from what it is handed. Ambient Anthropic auth is stripped from the + // inherited half only when a managed account owns the credential; a system-auth + // user's own key is their sign-in and must reach the child. + const env = withCliRuntimeOnPath( + command, + { + ...applyClaudeEnvPatch( + cloneDefinedEnv(process.env), + {}, + { + stripAuthEnv: auth.stripAuthEnv, + platform: process.platform + } + ), + ...(overlay ? cloneDefinedEnv(overlay) : {}) + }, + { platform: process.platform } + ) + return { + pathToClaudeCodeExecutable: command, + options: { + ...durable, + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...durable.extraArgs, ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs }, + ...(head?.handle.provider === 'claude' + ? { + resume: providerSessionId, + ...(head.handle.leafUuid === null ? {} : { resumeSessionAt: head.handle.leafUuid }) + } + : { sessionId: providerSessionId }) + }, + cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + env, + claudeConfigDir: record.accountHome.path, + providerSessionId, + resumeLeafUuid: head?.handle.provider === 'claude' ? head.handle.leafUuid : null, + resumed: head?.handle.provider === 'claude' + } + } +} diff --git a/src/main/claude/claude-structured-location-support.test.ts b/src/main/claude/claude-structured-location-support.test.ts new file mode 100644 index 00000000000..1106667d544 --- /dev/null +++ b/src/main/claude/claude-structured-location-support.test.ts @@ -0,0 +1,91 @@ +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { + __setWindowsProcessTreeLoaderForTests, + resetWindowsProcessTableForTests +} from '../windows/windows-process-table' +import { supportsClaudeStructuredLocation } from './claude-structured-location-support' + +function setPlatform(platform: NodeJS.Platform): PropertyDescriptor | undefined { + const previous = Object.getOwnPropertyDescriptor(process, 'platform') + Object.defineProperty(process, 'platform', { configurable: true, value: platform }) + return previous +} + +describe('supportsClaudeStructuredLocation', () => { + let previousPlatform: PropertyDescriptor | undefined + + beforeEach(() => { + previousPlatform = setPlatform('darwin') + __setWindowsProcessTreeLoaderForTests() + }) + + afterEach(() => { + __setWindowsProcessTreeLoaderForTests() + resetWindowsProcessTableForTests() + if (previousPlatform) { + Object.defineProperty(process, 'platform', previousPlatform) + } + }) + + it('allows local non-WSL locations on macOS and Linux', () => { + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(true) + }) + + it('rejects Windows local locations until creation-time proof is available', () => { + previousPlatform = setPlatform('win32') + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2 }, + getAllProcesses: () => undefined + })) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + }) + + it('accepts Windows local locations once creation-time proof is available', () => { + previousPlatform = setPlatform('win32') + __setWindowsProcessTreeLoaderForTests(() => ({ + ProcessDataFlag: { None: 0, Memory: 1, CommandLine: 2, CreationTime: 4 }, + getAllProcesses: () => undefined + })) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(true) + }) + + it('rejects WSL and remote locations', () => { + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'local', + wslDistro: 'Ubuntu', + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + expect( + supportsClaudeStructuredLocation({ + executionHostId: 'runtime:env-1', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }) + ).toBe(false) + }) +}) diff --git a/src/main/claude/claude-structured-location-support.ts b/src/main/claude/claude-structured-location-support.ts new file mode 100644 index 00000000000..9c784a91325 --- /dev/null +++ b/src/main/claude/claude-structured-location-support.ts @@ -0,0 +1,11 @@ +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { isWindowsProcessStartTimeAvailable } from '../windows/windows-process-table' + +export function supportsClaudeStructuredLocation(location: AgentSessionExecutionLocation): boolean { + return ( + location.executionHostId === LOCAL_EXECUTION_HOST_ID && + location.wslDistro === null && + (process.platform !== 'win32' || isWindowsProcessStartTimeAvailable()) + ) +} diff --git a/src/main/claude/claude-structured-model-confirmation.test.ts b/src/main/claude/claude-structured-model-confirmation.test.ts new file mode 100644 index 00000000000..7bd8f629119 --- /dev/null +++ b/src/main/claude/claude-structured-model-confirmation.test.ts @@ -0,0 +1,209 @@ +import { describe, expect, it } from 'vitest' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { PROVIDER_SESSION_ID, acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim rows from Claude Code 2.1.258's list_models response. */ +const CATALOG = [ + { + value: 'default', + resolvedModel: 'claude-opus-5[1m]', + displayName: 'Default (recommended)', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } +] + +function initFrame(model: string): Record { + // Keys mirror the real per-turn system/init frame: it carries `model` as the + // resolved id, and no effort of any kind. + return { + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION_ID, + uuid: 'turn-init-uuid', + model, + apiKeySource: 'none' + } +} + +describe('Claude model confirmation', () => { + it('adopts the model a later turn reports when nothing was set since', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet' } + }) + + // The CLI's own report of what it is running — the only channel that carries + // it, since set_model answers success for a model it never resolves. + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-haiku-4-5-20251001')) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'haiku' } + }) + }) + + it('keeps a just-set model until the next turn reports one', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // No turn has run, so the acquisition-time report is older than the write and + // must not flip the pill back to the model the session started on. + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'haiku' } + }) + }) + + it('corrects the record when the turn runs a different model than was set', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet' } + }) + }) + + it('guards an effort against the model the turn reported, not the one that was set', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + // set_model answered success for a model it never resolved; the turn runs sonnet. + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + // The picker offers sonnet's levels, so refusing one under haiku — a model the + // pill does not show and the child is not running — is the false positive. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + ).resolves.toMatchObject({ effort: 'high' }) + }) + + it('keeps guarding against the reported model across a second effort write', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + + // The effort write bumps the option fence but does not change what the child + // runs, so the sonnet report is still current and still governs the guard. + // `max` skips the settings readback by contract, so only the catalog gates it: + // sonnet advertises it, haiku advertises no effort control at all. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'max', fence: 7 }) + ).resolves.toMatchObject({ effort: 'max' }) + }) + + it('guards an effort against a just-set model no turn has reported yet', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // The acquisition-time report predates the write, so haiku — which advertises + // no effort control — is still the model the guard must answer for. + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + ).rejects.toThrow('claude model haiku does not accept effort high') + }) + + it('stops vouching for a confirmed effort once the model changes', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'high', fence: 7 }) + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + // The readback was taken under sonnet; nothing has reported haiku holding it. + const options = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(options.current.effort).toBe('high') + expect(options.current.confirmed).toBeUndefined() + }) +}) + +describe('Claude effort the settings readback cannot report', () => { + function sessionWith( + reported: string, + calls: string[] = [] + ): { session: ClaudeSession; calls: string[] } { + return { + session: { + options: new Map([['model', 'sonnet']]), + reportedOptions: {}, + optionMutationSequence: 0, + confirmedOptions: new Set(), + connection: { + supportedModels: async () => { + calls.push('list_models') + return CATALOG + }, + applyFlagSettings: async (settings: { effortLevel?: string }) => { + calls.push(`apply:${settings.effortLevel}`) + }, + getSettings: async () => { + calls.push('get_settings') + return { + applied: { effort: reported }, + effective: { effortLevel: reported }, + sources: {} + } + } + } + } as unknown as ClaudeSession, + calls + } + } + + it('records a session-scoped effort the persisted settings never carry', async () => { + // `max` applies for the session and is deliberately excluded from the + // persisted effortLevel, so the readback reporting `high` is an absence of + // evidence, not a refusal — and the CLI offers `max` in its own catalog. + const { session, calls } = sessionWith('high') + + await expect( + setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + ).resolves.toEqual({ model: 'sonnet', effort: 'max' }) + expect(calls).toEqual(['list_models', 'apply:max']) + }) +}) diff --git a/src/main/claude/claude-structured-option-confirmation.test.ts b/src/main/claude/claude-structured-option-confirmation.test.ts new file mode 100644 index 00000000000..ca7b8b70f1c --- /dev/null +++ b/src/main/claude/claude-structured-option-confirmation.test.ts @@ -0,0 +1,193 @@ +import { describe, expect, it } from 'vitest' +import { + applyStructuredAgentSessionOptions, + createStructuredAgentSessionOptionState, + structuredAgentSessionOptionSnapshot +} from '../../shared/structured-agent-session-options' +import { CLAUDE_SESSION_OPTION_CATALOG } from '../../shared/agent-session-option-catalog-claude-codex' +import type { AgentSessionOptionsResult } from '../../shared/agent-session-wire' +import type { SessionOptionDescriptor } from '../../shared/native-chat-session-options' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' +import { PROVIDER_SESSION_ID, acquired, fakeClaude } from './claude-structured-session-test-support' + +/** Verbatim rows from Claude Code 2.1.260's list_models response: `haiku` really + * does omit both effort keys, which is what makes an effort under it refusable. */ +const CATALOG = [ + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet', + supportsEffort: true, + supportedEffortLevels: ['low', 'medium', 'high', 'xhigh', 'max'] + }, + { value: 'haiku', resolvedModel: 'claude-haiku-4-5-20251001', displayName: 'Haiku' } +] + +function initFrame(model: string): Record { + return { + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION_ID, + uuid: 'turn-init-uuid', + model, + apiKeySource: 'none' + } +} + +function modelPill(result: AgentSessionOptionsResult): SessionOptionDescriptor | undefined { + const state = applyStructuredAgentSessionOptions( + createStructuredAgentSessionOptionState('claude'), + CLAUDE_SESSION_OPTION_CATALOG, + result + ) + return structuredAgentSessionOptionSnapshot(state).find((d) => d.category === 'model') +} + +/** Provenance the record keeps. Nothing renders it — the pill shows the value + * either way, and a report that disagrees is what corrects it. */ +function modelSource(result: AgentSessionOptionsResult): string | undefined { + return modelPill(result)?.valueSource +} + +function modelValue(result: AgentSessionOptionsResult): string | undefined { + const kind = modelPill(result)?.kind + return kind?.type === 'select' ? kind.currentValue : undefined +} + +describe('structured option confirmation reaches the pill', () => { + it('shows a just-set model before any turn reports it', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed ?? []).not.toContain('model') + expect(modelSource(result)).toBe('dispatched') + }) + + it('marks the model reported once the provider names it back', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-haiku-4-5-20251001')) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed).toContain('model') + expect(modelSource(result)).toBe('reported') + }) + + it('records an effort the readback could not take without confirming it', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + settings: { applied: {}, effective: {}, sources: {} }, + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + // `max` is session-scoped and absent from the persisted settings, so it records + // without a readback — recorded, never vouched for. + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'max', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.effort).toBe('max') + expect(result.current.confirmed ?? []).not.toContain('effort') + }) + + it('confirms an effort the readback agreed with', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + settings: { applied: { effort: 'low' }, effective: { effortLevel: 'low' }, sources: {} }, + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'effort', value: 'low', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(result.current.confirmed).toContain('effort') + }) + + it('treats a host that reports no confirmation as unconfirmed', () => { + // Wire compatibility: an older host omits `confirmed` entirely. Absence must + // read as unconfirmed provenance, and the pill still shows the host's value. + const result = { + models: [{ id: 'haiku', label: 'Haiku', isDefault: false, efforts: [] }], + current: { model: 'haiku' } + } + expect(modelSource(result)).toBe('dispatched') + expect(modelValue(result)).toBe('haiku') + }) +}) + +describe('the provider report corrects the pill', () => { + it('moves the pill to the model the turn actually ran', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + expect(modelValue(await adapter.readOptions({ sessionId: 'session-1', fence: 7 }))).toBe( + 'haiku' + ) + + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + + const corrected = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(modelValue(corrected)).toBe('sonnet') + expect(corrected.current.confirmed).toContain('model') + }) + + it('lets a newer write outrank the report it precedes', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { list_models: () => CATALOG } + }) + const adapter = await acquired(claude) + claude.connections[0]!.handlers.onMessage?.(initFrame('claude-sonnet-5')) + await adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'haiku', fence: 7 }) + + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + expect(modelValue(result)).toBe('haiku') + expect(result.current.confirmed ?? []).not.toContain('model') + }) +}) + +describe('confirmation never outlives the write it belongs to', () => { + it('drops an earlier effort confirmation when the value changes', async () => { + const calls: string[] = [] + let reported = 'low' + const session = { + options: new Map([['model', 'sonnet']]), + reportedOptions: {}, + optionMutationSequence: 0, + confirmedOptions: new Set(), + connection: { + supportedModels: async () => CATALOG, + applyFlagSettings: async (s: { effortLevel?: string }) => { + calls.push(`apply:${s.effortLevel}`) + }, + getSettings: async () => ({ + applied: { effort: reported }, + effective: { effortLevel: reported }, + sources: {} + }) + } + } as unknown as ClaudeSession + + await setClaudeStructuredOption(session, { key: 'effort', value: 'low' }, undefined) + expect(session.confirmedOptions.has('effort')).toBe(true) + + // The provider now reports a level it cannot represent; the stale confirmation + // must not survive into the new value. + await setClaudeStructuredOption(session, { key: 'effort', value: 'max' }, undefined) + expect(session.options.get('effort')).toBe('max') + expect(session.confirmedOptions.has('effort')).toBe(false) + expect(calls).toEqual(['apply:low', 'apply:max']) + }) +}) diff --git a/src/main/claude/claude-structured-options.test.ts b/src/main/claude/claude-structured-options.test.ts new file mode 100644 index 00000000000..bc18a589e10 --- /dev/null +++ b/src/main/claude/claude-structured-options.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it, vi } from 'vitest' +import { setClaudeStructuredOption } from './claude-structured-options' +import type { ClaudeSession } from './claude-structured-session-state' + +function sessionFor(setModel: ClaudeSession['connection']['setModel']): ClaudeSession { + return { + connection: { setModel } as ClaudeSession['connection'], + providerSessionId: 'provider-session', + claudeConfigDir: '/accounts/claude', + leafUuid: null, + fence: 1, + acquisitionGeneration: 'generation-1', + prompts: {} as ClaudeSession['prompts'], + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(), + reportedOptions: {}, + reportedModelMutation: 0, + confirmedOptions: new Set(), + restoreSkippedOptions: new Set(), + capabilities: [], + events: undefined, + translator: null + } +} + +describe('Claude structured option mutation fencing', () => { + it('does not let a delayed earlier apply overwrite a later option', async () => { + let releaseFirst!: () => void + const firstApply = new Promise((resolve) => { + releaseFirst = resolve + }) + const setModel = vi + .fn() + .mockReturnValueOnce(firstApply) + .mockResolvedValue(undefined) + const session = sessionFor(setModel) + + const first = setClaudeStructuredOption(session, { key: 'model', value: 'old' }, undefined) + await vi.waitFor(() => expect(setModel).toHaveBeenCalledTimes(1)) + const second = setClaudeStructuredOption(session, { key: 'model', value: 'new' }, undefined) + await expect(second).resolves.toEqual({ model: 'new' }) + + releaseFirst() + await expect(first).resolves.toEqual({ model: 'new' }) + expect(session.options).toEqual(new Map([['model', 'new']])) + }) +}) diff --git a/src/main/claude/claude-structured-options.ts b/src/main/claude/claude-structured-options.ts new file mode 100644 index 00000000000..3d1377b12c6 --- /dev/null +++ b/src/main/claude/claude-structured-options.ts @@ -0,0 +1,147 @@ +import type { EffortLevel, PermissionMode } from '@anthropic-ai/claude-agent-sdk' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { + AgentSessionOptionRejectedError, + isAgentSessionOptionRejectedError +} from '../native-chat/agent-session-wire/structured-agent-session-option-error' +import { + readClaudeCurrentModel, + readClaudeModelEffortLevels, + readClaudeSettingsEffort +} from './claude-structured-session-options' +import type { ClaudeSession } from './claude-structured-session-state' + +const OPTION_ORDER = ['model', 'effort', 'permissionMode'] as const + +/** + * Efforts the settings readback cannot report. `max` applies for the rest of the + * session and is excluded from the persisted `effortLevel` by contract, so + * `get_settings` answers with the level underneath it — an absence of evidence + * that must not be read as the child refusing a level its own catalog offers. + */ +const UNREPORTED_EFFORTS: ReadonlySet = new Set(['max']) + +export function restoredClaudeStructuredSessionOptions( + options: Readonly> | undefined +): Map { + return new Map( + OPTION_ORDER.flatMap((key) => { + const value = options?.[key] + return value ? [[key, value] as const] : [] + }) + ) +} + +export async function setClaudeStructuredOption( + session: ClaudeSession, + input: { key: string; value: string }, + timeoutMs: number | undefined +): Promise>> { + const apply = + input.key === 'model' + ? () => session.connection.setModel(input.value, { timeoutMs }) + : input.key === 'permissionMode' + ? () => session.connection.setPermissionMode(input.value as PermissionMode, { timeoutMs }) + : input.key === 'effort' + ? () => + session.connection.applyFlagSettings( + { effortLevel: input.value as EffortLevel }, + { timeoutMs } + ) + : null + if (!apply) { + throw new AgentSessionOptionRejectedError( + `claude stream-json has no session option named ${input.key}` + ) + } + // The child stores an effort its model has no control for and keeps it across + // every later model switch and restore, so refuse before the write rather than + // read the acceptance back as adoption. Refused here, restore drops the stale + // value instead of replaying it onto a model that cannot use it. + if (input.key === 'effort') { + const { modelId, levels } = await readClaudeModelEffortLevels(session, timeoutMs) + if (levels && !levels.has(input.value)) { + throw new AgentSessionOptionRejectedError( + `claude model ${modelId} does not accept effort ${input.value}` + ) + } + } + const modelWasConfirmed = readClaudeCurrentModel(session).confirmed + const mutationSequence = ++session.optionMutationSequence + // Only a model write can stale the model report — an effort or permission-mode + // write does not change what the child is running. Leaving the stamp behind + // would drop the session back to the written model and refuse, on the next + // effort write, a level the model actually running advertises. + if (modelWasConfirmed && input.key !== 'model') { + session.reportedModelMutation = mutationSequence + } + try { + await apply() + } catch (error) { + if (error instanceof ClaudeControlRequestError) { + throw new AgentSessionOptionRejectedError(error) + } + throw error + } + // apply_flag_settings answers `success` for an effort it then ignores, so the + // absence of a throw proves nothing. Ask what the child actually holds. + const adopted = + input.key === 'effort' && !UNREPORTED_EFFORTS.has(input.value) + ? await session.connection + .getSettings({ timeoutMs }) + .then(readClaudeSettingsEffort) + .catch(() => null) + : null + if (mutationSequence !== session.optionMutationSequence) { + return Object.fromEntries(session.options) + } + // A disagreement stops main vouching for the value, it does not veto the write: + // the pre-flight guard already refuses levels the model advertises no control + // for, and no other client refuses on a readback. Keep the child's own answer so + // the disagreement survives as the level a later read falls back to. + if (adopted !== null && adopted !== input.value) { + session.reportedOptions.effort = adopted + } + session.options.set(input.key, input.value) + // Only a readback that agreed is adoption evidence; one that disagreed or could + // not be taken records the value but must not also claim the provider vouched for it. + if (adopted !== null && adopted === input.value) { + session.confirmedOptions.add(input.key) + } else { + session.confirmedOptions.delete(input.key) + } + // The effort readback was taken under the old model, so a model switch retires + // it: the child keeps the value but nothing has reported the new model holding + // it, and vouching for it would show a confirmed effort no readback covers. + if (input.key === 'model') { + session.confirmedOptions.delete('effort') + } + return Object.fromEntries(session.options) +} + +export async function restoreClaudeStructuredSessionOptions( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise { + // Any write that was already in flight belongs to the previous acquisition + // state and must not repopulate this map after restore starts. + session.optionMutationSequence += 1 + // The fence bump is not a write, so the report the session already holds is still + // current as of this instant; leaving the stamp behind would make every restored + // session read as unconfirmed until its next turn. + session.reportedModelMutation = session.optionMutationSequence + const options = [...session.options.entries()] + session.options.clear() + for (const [key, value] of options) { + try { + await setClaudeStructuredOption(session, { key, value }, timeoutMs) + } catch (error) { + if (!isAgentSessionOptionRejectedError(error)) { + throw error + } + // A stale or unavailable preference must not poison every future acquire; + // the provider's current value remains authoritative and is re-persisted. + session.restoreSkippedOptions.add(key) + } + } +} diff --git a/src/main/claude/claude-structured-owner-identity.test.ts b/src/main/claude/claude-structured-owner-identity.test.ts new file mode 100644 index 00000000000..592b61df338 --- /dev/null +++ b/src/main/claude/claude-structured-owner-identity.test.ts @@ -0,0 +1,39 @@ +import { describe, expect, it, vi } from 'vitest' +import { CLAUDE_SPAWN_TOKEN_ENV, claudeProcessIdentity } from './claude-structured-owner-identity' + +const IDENTITY = { + sessionId: 'session-identity', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude' as const, + providerHandle: { kind: 'claude' as const, sessionId: 'session-1', leafUuid: 'leaf-1' } +} + +describe('claude structured owner identity', () => { + it('exports the spawn token env and records the observed process identity', async () => { + expect(CLAUDE_SPAWN_TOKEN_ENV).toBe('ORCA_AGENT_SESSION_SPAWN_TOKEN') + await expect( + claudeProcessIdentity( + { identity: IDENTITY, spawnToken: 'spawn-a', pid: 4242 }, + async () => 123 + ) + ).resolves.toEqual({ + hostId: 'local', + pid: 4242, + processStartTimeMs: 123, + spawnToken: 'spawn-a' + }) + }) + + it('retries a failed start-time read before giving up', async () => { + const readStartTime = vi + .fn<(pid: number) => Promise>() + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(null) + .mockResolvedValueOnce(456) + await expect( + claudeProcessIdentity({ identity: IDENTITY, spawnToken: 'spawn-a', pid: 4242 }, readStartTime) + ).resolves.toMatchObject({ processStartTimeMs: 456 }) + expect(readStartTime).toHaveBeenCalledTimes(3) + }) +}) diff --git a/src/main/claude/claude-structured-owner-identity.ts b/src/main/claude/claude-structured-owner-identity.ts index 1d13e6ec7c2..e8f251d41f9 100644 --- a/src/main/claude/claude-structured-owner-identity.ts +++ b/src/main/claude/claude-structured-owner-identity.ts @@ -1,4 +1,7 @@ +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' import type { AgentSessionProviderHandleLink } from '../../shared/agent-session-provider-handle' +import type { AgentSessionProcessIdentity } from '../../shared/agent-session-record' +import { readProcessStartTimeMs } from '../runtime/agent-session-process-identity-probe' export function claudeProviderHandleLink(input: { sessionId: string @@ -19,3 +22,41 @@ export function claudeProviderHandleLink(input: { observedAt: input.observedAt } } + +/** The child echoes its spawn token here so the owner probe can tell a live + * child of this reservation from a same-pid stranger. */ +export const CLAUDE_SPAWN_TOKEN_ENV = 'ORCA_AGENT_SESSION_SPAWN_TOKEN' + +const START_TIME_READ_ATTEMPTS = 3 + +export async function claudeProcessIdentity( + input: { + identity: AgentSessionJournalIdentity + spawnToken: string + pid: number | undefined + }, + readStartTime: (pid: number) => Promise = readProcessStartTimeMs +): Promise { + if (input.pid === undefined) { + throw new Error('claude app-server started without a pid') + } + let processStartTimeMs: number | null = null + for ( + let attempt = 0; + attempt < START_TIME_READ_ATTEMPTS && processStartTimeMs === null; + attempt += 1 + ) { + processStartTimeMs = await readStartTime(input.pid) + } + if (processStartTimeMs === null) { + // Why: recording null makes every later owner probe indeterminate — a durable latch. + // Failing here reaps the child and leaves a retryable refusal instead. + throw new Error(`claude app-server start time for pid ${input.pid} could not be read`) + } + return { + hostId: input.identity.hostId, + pid: input.pid, + processStartTimeMs, + spawnToken: input.spawnToken + } +} diff --git a/src/main/claude/claude-structured-prompt-items.test.ts b/src/main/claude/claude-structured-prompt-items.test.ts new file mode 100644 index 00000000000..79916d6a507 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-items.test.ts @@ -0,0 +1,110 @@ +import { describe, expect, it } from 'vitest' +import { agentJournalItemKey } from '../../shared/agent-session-journal-item-key' +import { encodeAgentSessionQuestionAnswers } from '../../shared/agent-session-question-answer' +import { claudeQuestionItems } from './claude-structured-prompt-items' +import { + applyClaudePromptAnswer, + encodeClaudeQuestionOptionId, + type ClaudePendingPrompt +} from './claude-structured-prompt-replies' + +describe('Claude structured question addressing', () => { + it('keeps wire IDs bounded while returning the original question and choice', () => { + const questionId = 'Which option? '.repeat(100) + const label = 'A detailed choice '.repeat(100) + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { questions: [{ question: questionId, options: [{ label }] }] }, + suggestions: [], + questionIds: [questionId], + answers: new Map(), + settle: () => {} + } + + const item = claudeQuestionItems({ sessionId: 'session-1', prompt })[0]! + expect(agentJournalItemKey(item.identity).length).toBeLessThan(512) + expect(item.body.options[0]!.id.length).toBeLessThan(512) + expect(item.body.freeTextQuestionId).toBe('q1') + expect(applyClaudePromptAnswer({ prompt }, item.body.options[0]!.id)).toMatchObject({ + updatedInput: { answers: { [questionId]: label } } + }) + }) + + it('preserves colon-containing free-text answers', () => { + const questionId = 'Where should this run?' + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { questions: [{ question: questionId }] }, + suggestions: [], + questionIds: [questionId], + answers: new Map(), + settle: () => {} + } + const answer = 'https://example.test:8443/path' + + expect( + applyClaudePromptAnswer({ prompt }, encodeClaudeQuestionOptionId('q1', answer)) + ).toMatchObject({ + updatedInput: { answers: { [questionId]: answer } } + }) + }) + + it('returns arrays for multi-select and preserves mixed single and Other answers', () => { + const multiQuestion = 'Which targets?' + const singleQuestion = 'Which mode?' + const otherQuestion = 'Where should it run?' + const prompt: ClaudePendingPrompt = { + requestId: 'question-1', + promptKey: 'question-1', + toolUseId: 'tool-1', + toolName: 'AskUserQuestion', + kind: 'question', + input: { + questions: [ + { + question: multiQuestion, + multiSelect: true, + options: [{ label: 'frontend' }, { label: 'backend' }] + }, + { + question: singleQuestion, + options: [{ label: 'fast' }, { label: 'safe' }] + }, + { question: otherQuestion, options: [] } + ] + }, + suggestions: [], + questionIds: [multiQuestion, singleQuestion, otherQuestion], + answers: new Map(), + settle: () => {} + } + const item = claudeQuestionItems({ sessionId: 'session-1', prompt })[0]! + const questions = item.body.questions! + const encoded = encodeAgentSessionQuestionAnswers([ + { + questionId: 'q1', + optionIds: [questions[0]!.options[0]!.id, questions[0]!.options[1]!.id] + }, + { questionId: 'q2', optionIds: [questions[1]!.options[1]!.id] }, + { questionId: 'q3', optionIds: [], other: 'remote host' } + ]) + + expect(applyClaudePromptAnswer({ prompt }, encoded)).toMatchObject({ + updatedInput: { + answers: { + [multiQuestion]: ['frontend', 'backend'], + [singleQuestion]: 'safe', + [otherQuestion]: 'remote host' + } + } + }) + }) +}) diff --git a/src/main/claude/claude-structured-prompt-items.ts b/src/main/claude/claude-structured-prompt-items.ts new file mode 100644 index 00000000000..3bdf8ab6091 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-items.ts @@ -0,0 +1,134 @@ +import type { + AgentJournalApprovalItem, + AgentJournalItemIdentity, + AgentJournalPromptOption, + AgentJournalQuestion, + AgentJournalQuestionItem +} from '../../shared/agent-session-journal-types' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import { claudeRecord, claudeText } from './claude-structured-item-translation' +import { + CLAUDE_APPROVAL_DECISIONS, + encodeClaudeQuestionOptionId, + type ClaudeApprovalDecision, + type ClaudePendingPrompt +} from './claude-structured-prompt-replies' + +const APPROVAL_LABELS: Record = { + allow: 'Allow', + allowForSession: 'Allow for this session', + deny: 'Deny', + cancel: 'Stop' +} + +const PENDING = { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null +} as const + +export function claudePromptIdentity(input: { + sessionId: string + promptKey: string + questionId?: string +}): AgentJournalItemIdentity { + const suffix = input.questionId ? `:${input.questionId}` : '' + return { + provider: 'orca', + clientMessageId: `claude-prompt:${input.sessionId}:${input.promptKey}${suffix}` + } +} + +export function claudeApprovalItem(prompt: ClaudePendingPrompt): AgentJournalApprovalItem { + const serialized = JSON.stringify(prompt.input) + return { + kind: 'approval', + title: `Allow ${prompt.toolName}?`, + detail: serialized ? boundInlineText(serialized, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text : null, + options: CLAUDE_APPROVAL_DECISIONS.map((decision) => ({ + id: decision, + label: APPROVAL_LABELS[decision] + })), + resolution: { ...PENDING } + } +} + +export type ClaudeQuestionItem = { + identity: AgentJournalItemIdentity + body: AgentJournalQuestionItem +} + +function questionOptions( + question: Record, + questionAddress: string +): AgentJournalPromptOption[] { + if (!Array.isArray(question.options)) { + return [] + } + return question.options.flatMap((value, index) => { + const option = claudeRecord(value) + const label = claudeText(option?.label) + const description = claudeText(option?.description) + return label + ? [ + { + id: encodeClaudeQuestionOptionId(questionAddress, `choice-${index + 1}`), + label, + ...(description ? { description } : {}) + } + ] + : [] + }) +} + +export function claudeQuestionItems(input: { + sessionId: string + prompt: ClaudePendingPrompt +}): ClaudeQuestionItem[] { + const values = Array.isArray(input.prompt.input.questions) ? input.prompt.input.questions : [] + const questions = values.flatMap((value, index): AgentJournalQuestion[] => { + const question = claudeRecord(value) + const questionAddress = `q${index + 1}` + const text = claudeText(question?.question) ?? claudeText(question?.header) + const header = claudeText(question?.header) + return question && input.prompt.questionIds[index] && text + ? [ + { + id: questionAddress, + question: text, + ...(header ? { header } : {}), + options: questionOptions(question, questionAddress), + multiSelect: question.multiSelect === true, + freeTextQuestionId: questionAddress + } + ] + : [] + }) + if (questions.length === 0) { + return [] + } + const legacyCompatible = questions.length === 1 && questions[0]?.multiSelect === false + const first = questions[0]! + return [ + { + identity: claudePromptIdentity({ + sessionId: input.sessionId, + promptKey: input.prompt.promptKey + }), + body: { + kind: 'question', + question: legacyCompatible + ? first.question + : `${questions.length} grouped question${questions.length === 1 ? '' : 's'} from Claude`, + options: legacyCompatible ? first.options : [], + ...(legacyCompatible ? { freeTextQuestionId: first.freeTextQuestionId } : {}), + questions, + resolution: { ...PENDING } + } + } + ] +} diff --git a/src/main/claude/claude-structured-prompt-replies.ts b/src/main/claude/claude-structured-prompt-replies.ts new file mode 100644 index 00000000000..deec74b7308 --- /dev/null +++ b/src/main/claude/claude-structured-prompt-replies.ts @@ -0,0 +1,297 @@ +import { decodeAgentSessionQuestionAnswers } from '../../shared/agent-session-question-answer' + +export const CLAUDE_APPROVAL_DECISIONS = ['allow', 'allowForSession', 'deny', 'cancel'] as const +export type ClaudeApprovalDecision = (typeof CLAUDE_APPROVAL_DECISIONS)[number] + +/** Settles the SDK's `canUseTool` promise; `null` is the SDK's "no response written" sentinel. */ +export type ClaudePromptSettle = (response: Record | null) => void + +export type ClaudePendingPrompt = { + requestId: string + promptKey: string + toolUseId: string + toolName: string + kind: 'approval' | 'question' + input: Record + suggestions: unknown[] + questionIds: readonly string[] + answers: Map + settle: ClaudePromptSettle +} + +export type ClaudePromptRegistration = { + requestId: string + toolName: string + toolUseId: string + input: Record + suggestions: unknown[] + settle: ClaudePromptSettle +} + +type PromptBinding = { + address: string + questionId?: string +} + +function isRecord(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +function readString(value: unknown): string | null { + return typeof value === 'string' && value.trim().length > 0 ? value : null +} + +function questionsFrom(input: Record): Record[] { + return Array.isArray(input.questions) ? input.questions.filter(isRecord) : [] +} + +function questionIdFromAddress(prompt: ClaudePendingPrompt, address: string): string | null { + const match = /^q([1-9]\d*)$/.exec(address) + const index = match ? Number(match[1]) - 1 : -1 + return index >= 0 ? (prompt.questionIds[index] ?? null) : null +} + +function questionAnswer(prompt: ClaudePendingPrompt, questionId: string, optionId: string): string { + const decoded = decodeClaudeQuestionOptionId(optionId) + if (!decoded) { + return optionId + } + const questionIndex = prompt.questionIds.indexOf(questionId) + if (questionIndex === -1) { + return optionId + } + const choice = /^choice-([1-9]\d*)$/.exec(decoded.answer) + const optionIndex = choice ? Number(choice[1]) - 1 : -1 + const question = questionsFrom(prompt.input)[questionIndex] + const options = Array.isArray(question?.options) ? question.options : [] + const option = options[optionIndex] + const label = isRecord(option) ? readString(option.label) : null + if (decoded.questionId === `q${questionIndex + 1}` && label) { + return label + } + if (decoded.questionId === `q${questionIndex + 1}`) { + return decoded.answer + } + const legacyChoice = options.some( + (candidate) => isRecord(candidate) && readString(candidate.label) === decoded.answer + ) + return decoded.questionId === questionId && (legacyChoice || decoded.answer.trim().length > 0) + ? decoded.answer + : optionId +} + +function questionId(question: Record, index: number): string { + return readString(question.question) ?? readString(question.header) ?? `question-${index + 1}` +} + +export function encodeClaudeQuestionOptionId(questionId: string, answer: string): string { + return `${encodeURIComponent(questionId)}:${encodeURIComponent(answer)}` +} + +export function decodeClaudeQuestionOptionId( + optionId: string +): { questionId: string; answer: string } | null { + const separator = optionId.indexOf(':') + if (separator <= 0) { + return null + } + try { + return { + questionId: decodeURIComponent(optionId.slice(0, separator)), + answer: decodeURIComponent(optionId.slice(separator + 1)) + } + } catch { + return null + } +} + +export class ClaudePromptRegistry { + private readonly prompts = new Map() + private readonly journalBindings = new Map() + + register(registration: ClaudePromptRegistration): ClaudePendingPrompt | null { + const toolUseId = readString(registration.toolUseId) + const toolName = readString(registration.toolName) + const input = isRecord(registration.input) ? registration.input : null + if (!toolUseId || !toolName || !input) { + return null + } + const questions = toolName === 'AskUserQuestion' ? questionsFrom(input) : [] + const prompt: ClaudePendingPrompt = { + requestId: registration.requestId, + promptKey: registration.requestId, + toolUseId, + toolName, + kind: questions.length > 0 ? 'question' : 'approval', + input, + suggestions: Array.isArray(registration.suggestions) ? registration.suggestions : [], + questionIds: questions.map(questionId), + answers: new Map(), + settle: registration.settle + } + this.prompts.set(prompt.promptKey, prompt) + return prompt + } + + /** True only if the prompt was still pending; lets an abort and an answer race settle once. */ + forgetIfPending(prompt: ClaudePendingPrompt): boolean { + if (!this.prompts.has(prompt.promptKey)) { + return false + } + this.forget(prompt) + return true + } + + bindJournalItemId(journalItemId: string, promptKey: string, questionIdForItem?: string): void { + this.journalBindings.set(journalItemId, { + address: promptKey, + ...(questionIdForItem ? { questionId: questionIdForItem } : {}) + }) + } + + find(itemId: string): { prompt: ClaudePendingPrompt; questionId?: string } | null { + const binding = this.journalBindings.get(itemId) + const prompt = this.prompts.get(binding?.address ?? itemId) + return prompt + ? { prompt, ...(binding?.questionId ? { questionId: binding.questionId } : {}) } + : null + } + + cancel(requestId: string): ClaudePendingPrompt | null { + const prompt = this.prompts.get(requestId) ?? null + if (prompt) { + this.forget(prompt) + } + return prompt + } + + forget(prompt: ClaudePendingPrompt): void { + this.prompts.delete(prompt.promptKey) + for (const [itemId, binding] of this.journalBindings) { + if (binding.address === prompt.promptKey) { + this.journalBindings.delete(itemId) + } + } + } + + clear(): ClaudePendingPrompt[] { + const pending = [...this.prompts.values()] + this.prompts.clear() + this.journalBindings.clear() + return pending + } +} + +function approvalResponse(prompt: ClaudePendingPrompt, optionId: string): Record { + if (!(CLAUDE_APPROVAL_DECISIONS as readonly string[]).includes(optionId)) { + throw new Error(`${optionId} is not a Claude approval decision`) + } + const decision = optionId as ClaudeApprovalDecision + if (decision === 'allow' || decision === 'allowForSession') { + return { + behavior: 'allow', + updatedInput: prompt.input, + ...(decision === 'allowForSession' && prompt.suggestions.length > 0 + ? { updatedPermissions: prompt.suggestions } + : {}), + toolUseID: prompt.toolUseId + } + } + return { + behavior: 'deny', + message: decision === 'cancel' ? 'User stopped this turn.' : 'User denied this action.', + ...(decision === 'cancel' ? { interrupt: true } : {}), + toolUseID: prompt.toolUseId + } +} + +function questionResponse( + prompt: ClaudePendingPrompt, + optionId: string, + boundQuestionId?: string +): Record | null { + const decoded = decodeClaudeQuestionOptionId(optionId) + const decodedQuestionId = decoded + ? (questionIdFromAddress(prompt, decoded.questionId) ?? + (prompt.questionIds.includes(decoded.questionId) ? decoded.questionId : null)) + : null + const selectedQuestionId = + boundQuestionId ?? + decodedQuestionId ?? + (prompt.questionIds.length === 1 ? prompt.questionIds[0] : null) + if (!selectedQuestionId || !prompt.questionIds.includes(selectedQuestionId)) { + throw new Error(`${optionId} does not name a question on Claude prompt ${prompt.promptKey}`) + } + const answer = questionAnswer(prompt, selectedQuestionId, optionId) + prompt.answers.set(selectedQuestionId, answer) + if (prompt.questionIds.some((id) => !prompt.answers.has(id))) { + return null + } + const answers: Record = {} + for (const id of prompt.questionIds) { + answers[id] = prompt.answers.get(id) as string + } + return { + behavior: 'allow', + updatedInput: { ...prompt.input, answers }, + toolUseID: prompt.toolUseId + } +} + +function groupedQuestionResponse( + prompt: ClaudePendingPrompt, + optionId: string +): Record | null { + const grouped = decodeAgentSessionQuestionAnswers(optionId) + if (!grouped) { + return null + } + const questions = questionsFrom(prompt.input) + if (grouped.length !== prompt.questionIds.length) { + throw new Error(`Grouped answer does not match Claude prompt ${prompt.promptKey}`) + } + const answers: Record = {} + for (let index = 0; index < questions.length; index += 1) { + const question = questions[index]! + const providerQuestionId = prompt.questionIds[index] + const answer = grouped.find((entry) => entry.questionId === `q${index + 1}`) + if (!providerQuestionId || !answer) { + throw new Error(`Grouped answer does not name question ${index + 1}`) + } + const selected = answer.optionIds.map((selectedId) => + questionAnswer(prompt, providerQuestionId, selectedId) + ) + const other = answer.other?.trim() + if (question.multiSelect === true) { + const values = [...selected, ...(other ? [other] : [])] + if (values.length === 0) { + throw new Error(`Grouped answer leaves question ${index + 1} empty`) + } + answers[providerQuestionId] = values + } else { + const value = other || selected[0] + if (!value || selected.length > 1) { + throw new Error(`Grouped answer is invalid for question ${index + 1}`) + } + answers[providerQuestionId] = value + } + } + return { + behavior: 'allow', + updatedInput: { ...prompt.input, answers }, + toolUseID: prompt.toolUseId + } +} + +export function applyClaudePromptAnswer( + found: { prompt: ClaudePendingPrompt; questionId?: string }, + optionId: string +): Record | null { + if (found.prompt.kind === 'approval') { + return approvalResponse(found.prompt, optionId) + } + return ( + groupedQuestionResponse(found.prompt, optionId) ?? + questionResponse(found.prompt, optionId, found.questionId) + ) +} diff --git a/src/main/claude/claude-structured-provider-fallback.test.ts b/src/main/claude/claude-structured-provider-fallback.test.ts new file mode 100644 index 00000000000..e8b27114da4 --- /dev/null +++ b/src/main/claude/claude-structured-provider-fallback.test.ts @@ -0,0 +1,117 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import type { + AgentJournalItemBody, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import { openAgentSessionJournal } from '../native-chat/agent-session-journal/journal-store-factory' +import { createDeferredStructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { createClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeStructuredSessionEvent } from './claude-structured-session-state' + +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-1', + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: 'provider-1', leafUuid: 'leaf-1' } +} + +let root = '' + +function message( + role: 'assistant' | 'user', + uuid: string, + content: unknown[] +): ClaudeStructuredSessionEvent { + return { + type: 'message', + sessionId: 'orca-session', + message: { + type: role, + uuid, + session_id: 'provider-1', + message: { role, content } + } + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-claude-provider-fallback-')) +}) + +afterEach(async () => { + await rm(root, { recursive: true, force: true }) +}) + +describe('Claude provider fallback', () => { + it('drops suppressed init frames instead of dereferencing a null translation', () => { + const items: { identity: unknown; body: AgentJournalItemBody }[] = [] + const sink = { + appendItem: (identity: unknown, body: AgentJournalItemBody) => { + items.push({ identity, body }) + }, + appendTombstone: vi.fn(), + publish: vi.fn() + } + const translator = createClaudeJournalTranslator({ sink }) + const initEvent: ClaudeStructuredSessionEvent = { + type: 'message', + sessionId: 'orca-session', + message: { + type: 'system', + subtype: 'init', + session_id: 'provider-1', + uuid: 'init-1' + } + } + + expect(() => translator.handle(initEvent)).not.toThrow() + expect(items).toEqual([]) + }) + + it('keeps provider-fallback rows distinct across acquisitions', async () => { + const journal = await openAgentSessionJournal({ + identity: IDENTITY, + journalDir: root, + now: () => 1_700_000_000_000, + mintEpoch: () => 'epoch-1' + }) + const deferred = createDeferredStructuredAgentSessionEventSink() + deferred.bind({ + journal, + fence: 1, + publish: vi.fn() + }) + + const first = createClaudeJournalTranslator({ sink: deferred.sink, fallbackIdPrefix: '1' }) + const second = createClaudeJournalTranslator({ sink: deferred.sink, fallbackIdPrefix: '2' }) + + first.handle(message('assistant', 'assistant-1', [{ type: 'future_event', message: 'first' }])) + await deferred.drained() + second.handle( + message('assistant', 'assistant-2', [{ type: 'future_event', message: 'second' }]) + ) + await deferred.drained() + + const fallbackRows = journal + .snapshot() + .items.filter( + (item) => + item.body.kind === 'status' && + item.body.providerFrame?.kind === 'message:assistant:content:future_event' + ) + + expect(fallbackRows).toHaveLength(2) + expect(fallbackRows.map(statusText)).toEqual(['first', 'second']) + }) +}) + +function statusText(row: { body: AgentJournalItemBody }): string { + if (row.body.kind !== 'status') { + throw new Error('expected status row') + } + return row.body.text +} diff --git a/src/main/claude/claude-structured-provider-fallback.ts b/src/main/claude/claude-structured-provider-fallback.ts new file mode 100644 index 00000000000..2528ac027df --- /dev/null +++ b/src/main/claude/claude-structured-provider-fallback.ts @@ -0,0 +1,125 @@ +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + boundInlineText, + DEFAULT_JOURNAL_PAYLOAD_LIMITS +} from '../native-chat/agent-session-journal/journal-payload-bounds' +import { CLAUDE_STREAM_JSON_FRAME_KINDS } from '../native-chat/agent-session-wire/claude-stream-json-frame-schema' +import { unhandledProviderFrameJournalItem } from '../native-chat/agent-session-wire/unhandled-provider-frame' +import { claudeRecord, claudeText } from './claude-structured-item-translation' + +export function claudeProviderFrameKind(message: Record): string { + const type = claudeText(message.type) ?? 'unknown' + const subtype = claudeText(message.subtype) + const eventType = claudeText(claudeRecord(message.event)?.type) + return ['message', type, subtype ?? eventType].filter(Boolean).join(':') +} + +const SETTLED_RESULT_KINDS: ReadonlySet = new Set( + CLAUDE_STREAM_JSON_FRAME_KINDS.filter((kind) => kind.startsWith('message:result:')) +) + +/** A catalogued result subtype is the turn-complete signal the translator settles + * itself; only an unmodeled subtype still needs the provider-fallback row. */ +export function isSettledClaudeResultKind(kind: string): boolean { + return SETTLED_RESULT_KINDS.has(kind) +} + +/** + * The failure a result frame carries that the turn's own frames never showed. + * + * Suppression is by meaning, not by kind. The SDK models an API failure as a + * SUCCESS-subtype result whose `result` string IS the error text and which has + * no assistant frame behind it, so keying on the subtype tombstones the turn and + * shows the user a completed, empty reply. A turn the user aborted is the + * opposite: its interrupt frame already says so, and the diagnostic in `errors` + * would only be noise. + */ +export function claudeResultFailure( + message: Record +): { text: string | null } | null { + if (message.is_error !== true) { + return null + } + const terminalReason = claudeText(message.terminal_reason) + if (terminalReason === 'aborted_streaming' || terminalReason === 'aborted_tools') { + return null + } + const result = claudeText(message.result)?.trim() + if (result) { + return { text: result } + } + const errors = Array.isArray(message.errors) + ? message.errors.flatMap((entry) => { + const text = claudeText(entry)?.trim() + return text ? [text] : [] + }) + : [] + // Nothing readable to lead with, but a reported failure still gets its row. + return { text: errors.length > 0 ? errors.join('\n') : null } +} + +/** + * What a message part that Orca cannot render says for itself. The kinds under + * `message::content:*` are synthesised from whatever `part.type` the CLI + * sends, so they can never be catalogued ahead of time; printing one is leaking + * wire vocabulary at a user who cannot act on it. The frame stays on the row's + * disclosure, so nothing is dropped and the next reader can still name it. + */ +export const CLAUDE_UNRENDERABLE_CONTENT_TEXT = 'Claude sent content Orca cannot display yet' + +export function isModeledClaudeContent(value: unknown): boolean { + const part = claudeRecord(value) + if (!part) { + return false + } + if (part.type === 'text') { + return claudeText(part.text) !== null + } + if (part.type === 'image') { + const source = claudeRecord(part.source) + if (source?.type === 'url') { + return claudeText(source.url) !== null + } + // A local attachment is replayed as the base64 (or file) source Orca itself + // sent, so it is content we recognise -- not an unknown part to surface. + return source?.type === 'base64' || source?.type === 'file' + } + if (part.type === 'tool_use') { + return claudeText(part.id) !== null && claudeText(part.name) !== null + } + if (part.type === 'tool_result') { + return claudeText(part.tool_use_id) !== null + } + // Redacted thinking arrives as an empty string plus a signature. + return part.type === 'thinking' || part.type === 'redacted_thinking' +} + +export function createClaudeProviderFrameFallback( + sink: StructuredAgentSessionEventSink, + acquisitionId: string +): { + /** `displayText` leads the row when Claude knows the sentence the frame itself does not name. */ + append: (kind: string, payload: unknown, displayText?: string | null) => void +} { + let sequence = 0 + return { + append: (kind, payload, displayText) => { + sequence += 1 + const translated = unhandledProviderFrameJournalItem('claude', kind, payload) + if (!translated) { + return + } + const bounded = displayText + ? boundInlineText(displayText, DEFAULT_JOURNAL_PAYLOAD_LIMITS).text + : null + sink.appendItem( + { + provider: 'orca', + clientMessageId: `provider-frame:claude:${acquisitionId}:${sequence}` + }, + bounded ? { ...translated.body, text: bounded } : translated.body + ) + sink.publish() + } + } +} diff --git a/src/main/claude/claude-structured-real-cli.test.ts b/src/main/claude/claude-structured-real-cli.test.ts new file mode 100644 index 00000000000..0f22c175cc6 --- /dev/null +++ b/src/main/claude/claude-structured-real-cli.test.ts @@ -0,0 +1,299 @@ +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtemp, rm } from 'node:fs/promises' +import { homedir, tmpdir } from 'node:os' +import { basename, join, relative } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { resolveClaudeCommand } from '../codex-cli/command' +import { resolveSessionFilePath } from '../native-chat/session-file-resolver' +import { getSpawnArgsForWindows } from '../win32-utils' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' + +const command = resolveClaudeCommand() +const versionLaunch = getSpawnArgsForWindows(command, ['--version']) +const realClaudeAvailable = + spawnSync(versionLaunch.spawnCmd, versionLaunch.spawnArgs, { + stdio: 'ignore', + windowsHide: true, + timeout: 5_000 + }).status === 0 +const authStatusLaunch = getSpawnArgsForWindows(command, ['auth', 'status', '--json']) +/** The CLI's own account report — the only source of truth for where it writes that + * is not derived from Orca's own path expressions. */ +const realClaudeAuthStatus = (() => { + if (!realClaudeAvailable) { + return null + } + const result = spawnSync(authStatusLaunch.spawnCmd, authStatusLaunch.spawnArgs, { + encoding: 'utf8', + windowsHide: true, + timeout: 5_000 + }) + if (result.status !== 0) { + return null + } + try { + return JSON.parse(result.stdout) as { loggedIn?: boolean; projectsDirectory?: string } + } catch { + return null + } +})() +const realClaudeAuthenticated = realClaudeAuthStatus?.loggedIn === true + +function realAdapter( + providerSessionId: string, + claudeConfigDir: string, + events: ClaudeStructuredSessionEvent[] = [] +): ClaudeStructuredSessionAdapter { + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: command, + options: { ...CLAUDE_STRUCTURED_BASE_OPTIONS, sessionId: providerSessionId }, + cwd: process.cwd(), + claudeConfigDir, + providerSessionId, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1, + now: () => 2, + initTimeoutMs: 5_000 + }) +} + +function identity(providerSessionId: string): AgentSessionJournalIdentity { + return { + sessionId: 'real-cli-handshake', + workspaceId: 'real-cli-workspace', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: providerSessionId, leafUuid: null } + } +} + +/** The CLI flushes its transcript on its own schedule; poll rather than race it. */ +async function waitForResolvedTranscript( + providerSessionId: string, + timeoutMs = 15_000 +): Promise { + const deadline = Date.now() + timeoutMs + for (;;) { + // No options: the exact call transcript-read-cache.ts makes for mobile. + const resolved = await resolveSessionFilePath('claude', providerSessionId) + if (resolved || Date.now() >= deadline) { + return resolved + } + await new Promise((resolve) => setTimeout(resolve, 250)) + } +} + +describe.skipIf(!realClaudeAvailable)('Claude structured real CLI handshake', () => { + it.skipIf(!realClaudeAuthenticated)( + 'proves a pre-minted session before the first user message', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + const acquisition = await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli' + }) + const observedSubtypes = events.flatMap((event) => + event.type === 'message' ? [event.message.subtype] : [] + ) + + expect(acquisition.link.handle).toMatchObject({ + provider: 'claude', + sessionId: providerSessionId, + // Init/SessionStart UUIDs are protocol frames, not resumable + // main-transcript leaves; no cursor exists before the first user turn. + leafUuid: null + }) + expect(observedSubtypes).toContain('hook_started') + } finally { + await adapter.closeAll() + } + }, + 10_000 + ) + + // Unit tests can only pin the shape we read, which is exactly how the blank + // Effort pill survived every gate: the fixture invented an `effortLevel` on a + // frame the CLI does not send. This asserts both halves against the live + // binary — that get_settings reports the effort, and that init does not. + it.skipIf(!realClaudeAuthenticated)( + 'reports the current effort through get_settings and never on the init frame', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-effort' + }) + const published = events.flatMap((event) => + event.type === 'message' ? [event.message] : [] + ) + const options = await adapter.readOptions({ sessionId: 'real-cli-handshake', fence: 1 }) + + expect(published.length).toBeGreaterThan(0) + // Not just the init frame: no frame the CLI publishes carries an effort + // at all. Goes red the day one does, which is when the simpler fix + // becomes available. Which frame proves the session varies by host, so + // this asserts over all of them rather than picking one. + expect(published.filter((frame) => 'effortLevel' in frame)).toEqual([]) + // Goes red if `effective.effortLevel` is renamed or dropped, which no + // fixture-backed test can see. + expect(options.current.effort).toEqual(expect.any(String)) + } finally { + await adapter.closeAll() + } + }, + 15_000 + ) + + // Mobile native chat never reads the structured journal — it reads the CLI's own + // transcript through native-chat/session-file-resolver.ts. So this resolves the way + // transcript-read-cache.ts:104 does, with NO root override, and checks the answer + // against the root the CLI itself reports. Deriving the expected root from Orca's own + // `CLAUDE_CONFIG_DIR || ~/.claude` expression — the same one the code under test uses — + // would move both sides together and stay green in exactly the environment that + // blacks mobile out. + // The turn is what creates the file: an init-only handshake writes nothing. + it.skipIf(!realClaudeAuthenticated || !realClaudeAuthStatus?.projectsDirectory)( + 'writes its transcript where the mobile session-file resolver looks for it', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const adapter = realAdapter(providerSessionId, claudeConfigDir) + const cliProjectsDir = realClaudeAuthStatus?.projectsDirectory as string + + let transcriptPath: string | null = null + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-transcript' + }) + await adapter.dispatch({ + sessionId: 'real-cli-handshake', + clientMessageId: 'real-cli-transcript-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'hi' }] }, + fence: 1 + }) + transcriptPath = await waitForResolvedTranscript(providerSessionId) + } finally { + await adapter.closeAll() + } + + expect(transcriptPath).not.toBeNull() + expect(basename(transcriptPath ?? '')).toBe(`${providerSessionId}.jsonl`) + // `//.jsonl` + expect(relative(cliProjectsDir, transcriptPath ?? '').split(/[\\/]/)).toHaveLength(2) + // And the pinned account home is that same root, so the host-side leaf recovery + // (structured-claude-runtime-adapter.ts:64) and mobile agree. + expect(join(claudeConfigDir, 'projects')).toBe(cliProjectsDir) + }, + 45_000 + ) + + // The model half of the same lesson: a fixture can only pin the shape we read. + // set_model answers success for a model it never resolves — a nonexistent id is + // accepted and only fails once a turn runs — so the CLI's own report is the only + // adoption evidence, and it arrives on the init frame that opens each turn. This + // asserts that frame carries the resolved model against the live binary; it goes + // red the day the CLI stops reporting it, which is the day the confirmation + // silently degrades to echoing back whatever Orca sent. + it.skipIf(!realClaudeAuthenticated)( + 'reports the model it adopted on the init frame that opens each turn', + async () => { + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = realAdapter(providerSessionId, claudeConfigDir, events) + + try { + await adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-model' + }) + await adapter.setOption({ + sessionId: 'real-cli-handshake', + key: 'model', + value: 'haiku', + fence: 1 + }) + const before = events.length + await adapter.dispatch({ + sessionId: 'real-cli-handshake', + clientMessageId: 'real-cli-model-1', + body: { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'Say ok' }] }, + fence: 1 + }) + const deadline = Date.now() + 60_000 + let frames: Record[] = [] + for (;;) { + frames = events + .slice(before) + .flatMap((event) => + event.type === 'message' && + event.message.type === 'system' && + event.message.subtype === 'init' + ? [event.message] + : [] + ) + if (frames.length > 0 || Date.now() >= deadline) { + break + } + await new Promise((resolve) => setTimeout(resolve, 250)) + } + + expect(frames).not.toHaveLength(0) + // Both halves: the field exists, and it names the model the picker asked + // for in the catalog's resolved shape rather than the id Orca sent. + expect(frames[0]?.model).toEqual(expect.any(String)) + expect(frames[0]?.model).toBe('claude-haiku-4-5-20251001') + await expect( + adapter.readOptions({ sessionId: 'real-cli-handshake', fence: 1 }) + ).resolves.toMatchObject({ current: { model: 'haiku' } }) + } finally { + await adapter.closeAll() + } + }, + 90_000 + ) + + it('turns a real silent unauthenticated startup into sign-in guidance', async () => { + const claudeConfigDir = await mkdtemp(join(tmpdir(), 'orca-claude-no-auth-')) + const providerSessionId = randomUUID() + const adapter = realAdapter(providerSessionId, claudeConfigDir) + + try { + await expect( + adapter.acquire({ + identity: identity(providerSessionId), + fence: 1, + spawnToken: 'real-cli-no-auth' + }) + ).rejects.toThrow(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + } finally { + await adapter.closeAll() + await rm(claudeConfigDir, { recursive: true, force: true }) + } + }, 10_000) +}) diff --git a/src/main/claude/claude-structured-session-acquisition-processless.test.ts b/src/main/claude/claude-structured-session-acquisition-processless.test.ts new file mode 100644 index 00000000000..e0996477e84 --- /dev/null +++ b/src/main/claude/claude-structured-session-acquisition-processless.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import { AgentSessionPreSpawnError } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { + ClaudeStreamJsonConnection, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import { ClaudeStructuredSessionAdapter } from './claude-structured-session-adapter' + +const PROVIDER_SESSION_ID = '819cf9f8-e43c-4ad7-b50f-54aa158a726a' +const IDENTITY: AgentSessionJournalIdentity = { + sessionId: 'session-processless', + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'opaque', agent: 'claude', value: 'pending' } +} + +describe('Claude structured processless acquisition', () => { + it('classifies pre-pid error and close as processless with idempotent cleanup', async () => { + const fault = new Error('spawn claude ENOENT') + const close = vi.fn(async () => true) + const openConnection: typeof openClaudeStreamJsonConnection = async ( + _launch, + handlers = {} + ) => { + const connection: ClaudeStreamJsonConnection = { + pid: undefined, + closed: true, + exitVerdict: { root: 'processless', tree: 'exited' }, + initializationResult: async () => { + handlers.onFault?.(fault) + throw fault + }, + getSettings: async () => ({}), + supportedModels: async () => [], + interrupt: async () => undefined, + cancelAsyncMessage: async () => {}, + setModel: async () => {}, + setPermissionMode: async () => {}, + applyFlagSettings: async () => {}, + send: async () => {}, + close + } + return connection + } + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + }), + openConnection + }) + + const error = await adapter + .acquire({ identity: IDENTITY, fence: 7, spawnToken: 'spawn-9' }) + .catch((cause: unknown) => cause) + + expect(error).toBeInstanceOf(AgentSessionPreSpawnError) + expect(error).toMatchObject({ message: fault.message }) + expect(close).toHaveBeenCalledOnce() + await expect(adapter.releaseAcquisition({ sessionId: IDENTITY.sessionId })).resolves.toBe(true) + expect(close).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/claude/claude-structured-session-acquisition.ts b/src/main/claude/claude-structured-session-acquisition.ts new file mode 100644 index 00000000000..e8d09bd78d9 --- /dev/null +++ b/src/main/claude/claude-structured-session-acquisition.ts @@ -0,0 +1,298 @@ +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionPreSpawnError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { + AgentSessionAcquisition, + StructuredAgentSessionAcquireInput +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE } from '../claude-accounts/environment' +import { isClaudeAuthSwitchInProgress } from '../claude-accounts/live-pty-gate' +import { openClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { buildClaudePermissionCallbacks } from './claude-structured-inbound-control' +import { resolveClaudeReplayWaiter } from './claude-structured-dispatch' +import { + claudeAuthDiagnostic, + readClaudeCapabilities, + readClaudeFrameString, + readClaudeInit, + readClaudeModels +} from './claude-structured-init-proof' +import { + createClaudeInitDeadline, + requestClaudeInitialization +} from './claude-structured-init-deadline' +import { claudeConfigDirEnvPatch } from './claude-config-dir-pin' +import { CLAUDE_SPAWN_TOKEN_ENV, claudeProcessIdentity } from './claude-structured-owner-identity' +import { + restoreClaudeStructuredSessionOptions, + restoredClaudeStructuredSessionOptions +} from './claude-structured-options' +import { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { createClaudeSessionJournalTranslator } from './claude-structured-journal-translation' +import { readClaudeSettingsEffort } from './claude-structured-session-options' +import { createClaudeSessionPublication } from './claude-structured-session-publication' +import { + cancelClaudeAcquisitionAttempt, + mintClaudeAcquisitionGeneration, + type ClaudeAcquisitionRegistry, + type ClaudeSession, + type ClaudeSessionExit, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeAcquireCallbacks +} from './claude-structured-session-state' +import { + closeClaudePublishedSessionForDeps, + claudeAcquisitionCleanupError +} from './claude-structured-session-close' +import { readClaudeTranscriptEntryUuid } from './claude-tui-exit' + +export const CLAUDE_STRUCTURED_INIT_TIMEOUT_MS = 10_000 + +export async function acquireClaudeSession({ + input, + deps, + sessions, + acquisitions, + exits, + callbacks +}: { + input: StructuredAgentSessionAcquireInput + deps: ClaudeStructuredSessionAdapterDeps + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + callbacks: ClaudeAcquireCallbacks +}): Promise { + // A managed-account switch is mid-swap of the pinned credential home; refuse here, + // before this acquisition cancels the previous attempt and closes the live session. + if (isClaudeAuthSwitchInProgress()) { + throw new AgentSessionPreSpawnError(new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE)) + } + const sessionId = input.identity.sessionId + const prompts = new ClaudePromptRegistry() + const translator = createClaudeSessionJournalTranslator( + input.events, + prompts, + String(input.fence) + ) + const { previous, attempt } = acquisitions.start(sessionId, prompts) + let liveSession: ClaudeSession | null = null + let observedLeafUuid: string | null = null, + expectedProviderSessionId: string | null = null + // Frames are admitted only after launch resolution proves the provider session + // this acquisition owns. Keep the check ahead of every stateful consumer. + const initTimeoutMs = deps.initTimeoutMs ?? CLAUDE_STRUCTURED_INIT_TIMEOUT_MS + const initDeadline = createClaudeInitDeadline(sessionId, initTimeoutMs) + + const onMessage = (message: Record): void => { + const init = readClaudeInit(message) + if (readClaudeFrameString(message, 'session_id') !== expectedProviderSessionId) { + // An init proof for another (or unnamed) provider must fail acquisition + // promptly, while ordinary foreign frames stay quarantined silently. + if (init || (message.type === 'system' && message.subtype === 'init')) { + initDeadline.reject(new Error('claude provider session expected')) + } + return + } + if (init) { + initDeadline.resolve(init) + // Every turn opens with an init frame naming the model the CLI is actually + // running; set_model answers success for a model it never resolves, so this + // report is the session's only adoption evidence. + if (liveSession && init.model) { + liveSession.reportedOptions.model = init.model + liveSession.reportedModelMutation = liveSession.optionMutationSequence + } + } + observedLeafUuid = readClaudeTranscriptEntryUuid(message) ?? observedLeafUuid + if (liveSession) { + liveSession.leafUuid = observedLeafUuid + } + const startsTurn = liveSession ? resolveClaudeReplayWaiter(liveSession, message) : false + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { + type: 'message', + sessionId, + message, + ...(startsTurn ? { startsTurn: true } : {}) + }) + ) + } + const { canUseTool, onUserDialog } = buildClaudePermissionCallbacks({ + sessionId, + prompts, + emit: (event) => + callbacks.deliver(attempt, sessionId, () => callbacks.emit(liveSession, input.events, event)) + }) + + try { + if (previous && !(await cancelClaudeAcquisitionAttempt(previous))) { + acquisitions.restoreIfCurrent(sessionId, attempt, previous) + throw new AgentSessionAcquisitionExitUnprovenError( + new Error(`claude acquisition for session ${sessionId} could not be stopped`) + ) + } + acquisitions.assertCurrent(sessionId, attempt) + let resumeSession = sessions.get(sessionId) + if (!(await closeClaudePublishedSessionForDeps(sessions, sessionId, deps))) { + throw new AgentSessionAcquisitionExitUnprovenError( + new Error(`claude session ${sessionId} could not be stopped`) + ) + } + // A first-hand exit that has not yet proved its full tree still owns a cleanup + // obligation; never let a new acquisition hide that evidence by omission. + const retainedExit = exits.get(sessionId) + if (retainedExit) { + const firstProof = retainedExit.closePromise ? await retainedExit.closePromise : false + const proven = firstProof || (await retainedExit.connection.close().catch(() => false)) + if (!proven) { + throw claudeAcquisitionCleanupError(retainedExit.connection, retainedExit.error) + } + // The old child is superseded by this acquisition. Settle its lifecycle + // before discarding the retained proof so its cursor and callbacks are + // cleaned up exactly once. + await callbacks.settleExit(sessionId, retainedExit) + resumeSession ??= retainedExit.session + } + acquisitions.assertCurrent(sessionId, attempt) + // Both close paths persist their final leaf, so launch validates that durable head. + const launchIdentity = resumeSession + ? { + ...input.identity, + providerHandle: { + kind: 'claude' as const, + sessionId: resumeSession.providerSessionId, + leafUuid: resumeSession.leafUuid + } + } + : input.identity + const launch = await deps + .resolveLaunch({ identity: launchIdentity }) + .catch((error: unknown) => { + throw error instanceof AgentSessionPreSpawnError + ? error + : new AgentSessionPreSpawnError(error) + }) + expectedProviderSessionId = launch.providerSessionId + observedLeafUuid = launch.resumeLeafUuid + acquisitions.assertCurrent(sessionId, attempt) + const open = deps.openConnection ?? openClaudeStreamJsonConnection + const connection = await open( + { + pathToClaudeCodeExecutable: launch.pathToClaudeCodeExecutable, + options: launch.options, + cwd: launch.cwd, + env: { + ...launch.env, + [CLAUDE_SPAWN_TOKEN_ENV]: input.spawnToken, + // Compared against what the child would otherwise inherit, so the record's + // account home still wins over a diverging overlay without a needless pin. + // (`process` is shadowed by a local later in this function, so it is not named here.) + ...claudeConfigDirEnvPatch(launch.claudeConfigDir, launch.env ? { env: launch.env } : {}) + } + }, + { + onMessage, + canUseTool, + onUserDialog, + onFault: (error) => { + if (!attempt.published) { + initDeadline.reject(error) + } + }, + onExit: (error) => { + if (!attempt.published) { + initDeadline.reject(error) + } + callbacks.handleExit(sessionId, attempt, error) + } + } + ) + attempt.connection = connection + acquisitions.assertCurrent(sessionId, attempt) + initDeadline.start() + const [initialization, init] = await Promise.all([ + requestClaudeInitialization(connection, sessionId, initTimeoutMs), + initDeadline.promise + ]) + const models = readClaudeModels(initialization) + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { type: 'options', sessionId, models }) + ) + initDeadline.clear() + acquisitions.assertCurrent(sessionId, attempt) + if (init.providerSessionId !== launch.providerSessionId) { + throw new Error( + `claude proved session ${init.providerSessionId}, expected ${launch.providerSessionId}` + ) + } + const settings = await connection + .getSettings({ timeoutMs: deps.requestTimeoutMs }) + .catch(() => null) + callbacks.deliver(attempt, sessionId, () => + callbacks.emit(liveSession, input.events, { + type: 'auth-diagnostic', + sessionId, + diagnostic: claudeAuthDiagnostic(init, settings) + }) + ) + const process = await claudeProcessIdentity( + { ...input, pid: connection.pid }, + deps.readProcessStartTime + ) + acquisitions.assertCurrent(sessionId, attempt) + if (connection.closed) { + throw new Error(`claude stream-json for session ${sessionId} exited while being acquired`) + } + const publication = createClaudeSessionPublication({ + connection, + init, + claudeConfigDir: launch.claudeConfigDir, + leafUuid: observedLeafUuid, + fence: input.fence, + effort: readClaudeSettingsEffort(settings), + resumed: launch.resumed, + prompts, + translator, + events: input.events, + process, + acquisitionGeneration: mintClaudeAcquisitionGeneration(deps), + options: restoredClaudeStructuredSessionOptions(input.options), + capabilities: readClaudeCapabilities(init, initialization), + ...(deps.mintLinkId ? { linkId: deps.mintLinkId() } : {}), + observedAt: deps.now?.() ?? Date.now() + }) + const acquired: AgentSessionAcquisition = publication.acquisition + liveSession = publication.session + await restoreClaudeStructuredSessionOptions(liveSession, deps.requestTimeoutMs) + acquisitions.assertCurrent(sessionId, attempt) + acquisitions.deleteIfCurrent(sessionId, attempt) + sessions.set(sessionId, liveSession) + attempt.published = true + for (const event of attempt.buffered.splice(0)) { + event() + } + return acquired + } catch (error) { + initDeadline.clear() + let acquisitionError = error + if (sessions.get(sessionId)?.connection !== attempt.connection) { + translator?.dispose() + // Settle any callback that fired before the failure so no SDK promise dangles. + for (const prompt of prompts.clear()) { + prompt.settle(null) + } + const closed = (await attempt.connection?.close()) ?? true + if (attempt.connection?.exitVerdict.root === 'processless') { + acquisitionError = new AgentSessionPreSpawnError(error) + } else if (!closed) { + acquisitionError = claudeAcquisitionCleanupError(attempt.connection, error) + } + } + acquisitions.deleteIfCurrent(sessionId, attempt) + throw acquisitionError + } finally { + attempt.finish() + } +} diff --git a/src/main/claude/claude-structured-session-adapter.test.ts b/src/main/claude/claude-structured-session-adapter.test.ts new file mode 100644 index 00000000000..859ba9a8ed8 --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter.test.ts @@ -0,0 +1,891 @@ +import { homedir } from 'node:os' +import { join } from 'node:path' +import { describe, expect, it, vi } from 'vitest' +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRefusal, + AgentSessionAcquisitionRootExitObservedError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { ClaudeControlRequestError } from './claude-stream-json-connection' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' +import { encodeClaudeQuestionOptionId } from './claude-structured-prompt-replies' +import { + CLAUDE_STRUCTURED_INIT_TIMEOUT_MS, + type ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { + acquired, + adapterFor, + fakeClaude, + identityFor, + invokeCanUseTool, + PROVIDER_SESSION_ID, + tick, + USER_MESSAGE, + type FakeConnection +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter.acquire', () => { + it('finishes its startup deadline before the paired mobile request deadline', () => { + expect(CLAUDE_STRUCTURED_INIT_TIMEOUT_MS).toBeLessThan(30_000) + }) + + it('pins the account and proves init without treating the system-frame uuid as a chain leaf', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + + const acquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9' + }) + + expect(claude.connections[0].launch).toMatchObject({ + cwd: '/work/repo', + env: { + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9', + CLAUDE_CONFIG_DIR: '/accounts/claude' + } + }) + // supportedDialogKinds is now a query() launch option, not an initialize request param. + expect(claude.connections[0].calls.slice(0, 2)).toEqual([ + { subtype: 'initialize' }, + { subtype: 'get_settings' } + ]) + expect(acquisition.process).toEqual({ + hostId: 'host-1', + pid: 4321, + processStartTimeMs: 1_700_000_000_000, + spawnToken: 'spawn-9' + }) + expect(acquisition.link).toEqual({ + linkId: `claude-7-${PROVIDER_SESSION_ID}-empty`, + handle: { provider: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null }, + origin: 'created', + mintedAtFence: 7, + observedAt: 1_700_000_000_500 + }) + expect(events[0]).toMatchObject({ type: 'message', message: { subtype: 'init' } }) + }) + + it('restores persisted model and effort before publishing a reacquired session', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { resumed: true }) + + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options: { model: 'opus', effort: 'high' } + }) + + expect(claude.connections[0].calls.slice(-4)).toEqual([ + { subtype: 'set_model', params: { model: 'opus' } }, + // The restored model's advertised levels gate the replay, so a stale effort + // is dropped rather than re-applied to a model with no effort control. + { subtype: 'list_models' }, + { subtype: 'apply_flag_settings', params: { settings: { effortLevel: 'high' } } }, + // The effort is only recorded once the child reports having adopted it. + { subtype: 'get_settings' } + ]) + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toMatchObject({ + current: { model: 'opus', effort: 'high' } + }) + }) + + it.each([ + ['model', 'set_model', { model: 'retired-model' }], + ['effort', 'apply_flag_settings', { effort: 'retired-effort' }], + ['permissionMode', 'set_permission_mode', { permissionMode: 'retired-mode' }] + ] as const)( + 'self-heals a persisted %s rejected during restore', + async (key, subtype, options) => { + const claude = fakeClaude({ + routes: { + [subtype]: () => { + throw new ClaudeControlRequestError(subtype, 'value is no longer available') + } + } + }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options + }) + ).resolves.toBeDefined() + expect(adapter.readOptionRestoreFailures('session-1')).toEqual([key]) + } + ) + + it('does not treat a transport timeout while restoring an option as recoverable', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new Error('claude set_model request timed out') + } + } + }) + const adapter = adapterFor(claude) + const input = { + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + options: { model: 'temporarily-unavailable' } + } + + await expect(adapter.acquire(input)).rejects.toThrow('claude set_model request timed out') + expect(claude.connections[0]?.closeCount).toBe(1) + }) + + it('recovers a cancellable lifecycle when a timed-out replay arrives late', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + const sent = claude.connections[0]!.sent[0]! + claude.connections[0]!.handlers.onMessage?.({ + ...sent, + uuid: 'late-turn-1' + }) + + expect(events).toContainEqual( + expect.objectContaining({ + type: 'message', + startsTurn: true, + message: expect.objectContaining({ uuid: 'late-turn-1' }) + }) + ) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'late-turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + }) + + it('quarantines SDK frames without the acquired session identity', async () => { + const claude = fakeClaude({ replayUuid: null }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const connection = claude.connections[0]! + + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'foreign-leaf', + session_id: 'foreign-provider-session', + message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } + }) + connection.handlers.onMessage?.({ + type: 'assistant', + uuid: 'missing-session-leaf', + message: { role: 'assistant', content: [{ type: 'text', text: 'do not admit' }] } + }) + + const dispatch = adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + await Promise.resolve() + expect(connection.sent).toHaveLength(1) + connection.handlers.onMessage?.({ + ...connection.sent[0], + uuid: 'foreign-replay', + session_id: 'foreign-provider-session' + }) + await Promise.resolve() + expect(events.filter((event) => event.type === 'message')).toHaveLength(1) + + connection.handlers.onMessage?.({ + ...connection.sent[0], + session_id: PROVIDER_SESSION_ID + }) + await expect(dispatch).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: connection.sent[0]!.uuid } + }) + }) + + it('forwards configured launch environment while keeping ownership pins authoritative', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { + env: { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: '/wrong/account', + [CLAUDE_SPAWN_TOKEN_ENV]: 'wrong-token' + } + }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + expect(claude.connections[0].launch.env).toEqual({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: '/accounts/claude', + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' + }) + }) + + it('leaves CLAUDE_CONFIG_DIR unset when the account home is the CLI default', async () => { + const claude = fakeClaude() + const adapter = adapterFor(claude, { claudeConfigDir: join(homedir(), '.claude'), env: {} }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + // Pinning the CLI's own default suppresses the macOS Keychain and breaks claude.ai login. + expect(claude.connections[0].launch.env).toEqual({ [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' }) + }) + + it('re-pins the account home when the launch env would send the child elsewhere', async () => { + const claude = fakeClaude() + const accountHome = join(homedir(), '.claude') + const adapter = adapterFor(claude, { + claudeConfigDir: accountHome, + env: { CLAUDE_CONFIG_DIR: '/other/account' } + }) + + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + expect(claude.connections[0].launch.env).toEqual({ + CLAUDE_CONFIG_DIR: accountHome, + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-9' + }) + }) + + it('accepts SessionStart as pre-turn proof without treating its system uuid as a leaf', async () => { + const claude = fakeClaude({ initProof: 'session-start', initUuid: 'session-start-uuid' }) + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + + const acquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9' + }) + + expect(acquisition.link.handle).toEqual({ + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: null + }) + expect(events[0]).toMatchObject({ + type: 'message', + message: { subtype: 'hook_started', hook_name: 'SessionStart:startup' } + }) + }) + + it('records only non-secret effective auth-lane diagnostics', async () => { + const claude = fakeClaude({ + settings: { + env: { + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + ANTHROPIC_AUTH_TOKEN: 'secret' + } + } + }) + const events: ClaudeStructuredSessionEvent[] = [] + await acquired(claude, {}, events) + + const diagnostic = events.find((event) => event.type === 'auth-diagnostic') + expect(diagnostic).toEqual({ + type: 'auth-diagnostic', + sessionId: 'session-1', + diagnostic: { + apiKeySourceConfigured: false, + baseUrlConfigured: true, + authTokenConfigured: true, + apiKeyConfigured: false, + settingSources: ['user', 'project', 'local'] + } + }) + expect(JSON.stringify(diagnostic)).not.toContain('secret') + expect(JSON.stringify(diagnostic)).not.toContain('gateway.example.test') + }) + + it('resumes the same provider id and refuses an init proof for another session', async () => { + const resumedClaude = fakeClaude() + const resumed = adapterFor(resumedClaude, { + resumed: true, + resumeLeafUuid: 'leaf-before' + }) + const acquisition = await resumed.acquire({ + identity: identityFor(), + fence: 9, + spawnToken: 'spawn-9' + }) + expect(acquisition.link.origin).toBe('resumed') + expect(acquisition.link.handle).toEqual({ + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'leaf-before' + }) + + const wrongClaude = fakeClaude({ initSessionId: 'different-session' }) + const wrong = adapterFor(wrongClaude) + await expect( + wrong.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(/expected/) + expect(wrongClaude.connections[0].closeCount).toBe(1) + }) + + it('surfaces a CLI startup failure instead of waiting for the init deadline', async () => { + const claude = fakeClaude({ exitBeforeInit: 'Claude login required' }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow('Claude login required') + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('closes a silent unauthenticated startup with actionable account guidance', async () => { + const claude = fakeClaude({ initProof: 'none' }) + const adapter = adapterFor(claude, {}, [], [], 20) + + const error = await adapter + .acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + .catch((cause: unknown) => cause) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRefusal) + expect(error).toMatchObject({ + message: expect.stringMatching(/selected Claude account is signed in.*CLAUDE_CONFIG_DIR/s) + }) + expect(claude.connections[0].calls[0]).toEqual({ subtype: 'initialize' }) + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('refuses an unauthenticated initialize response even when SessionStart runs', async () => { + const claude = fakeClaude({ + initProof: 'session-start', + initAccount: { apiProvider: 'firstParty', tokenSource: 'none' } + }) + const adapter = adapterFor(claude) + + await expect( + adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + ).rejects.toThrow(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + expect(claude.connections[0].closeCount).toBe(1) + }) +}) + +describe('ClaudeStructuredSessionAdapter turns and controls', () => { + it('accepts a dispatch only after Claude replays its provider uuid', async () => { + const claude = fakeClaude({ replayUuid: 'user-provider-uuid' }) + const adapter = await acquired(claude) + + const result = await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + + expect(result).toEqual({ + state: 'accepted', + providerIdentity: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + uuid: 'user-provider-uuid' + } + }) + expect(claude.connections[0].sent[0]).toMatchObject({ + type: 'user', + message: { role: 'user', content: [{ type: 'text', text: 'ship it' }] }, + session_id: PROVIDER_SESSION_ID + }) + }) + + it('leaves delivery unconfirmed when no replay uuid arrives', async () => { + const adapter = await acquired(fakeClaude({ replayUuid: null })) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + }) + + it('requires an acknowledged interrupt and supports controlled options', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-1', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'sonnet', fence: 7 }) + ).resolves.toEqual({ model: 'sonnet' }) + expect(claude.connections[0].calls.slice(-2)).toEqual([ + { subtype: 'interrupt', params: {} }, + { subtype: 'set_model', params: { model: 'sonnet' } } + ]) + + claude.routes.interrupt = () => { + throw new ClaudeControlRequestError('interrupt', 'not running') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-2', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + + claude.routes.interrupt = () => { + throw new Error('claude interrupt request timed out') + } + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-3', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('does not let a delayed cancellation for an earlier turn interrupt the later turn', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', 'turn-U'] }) + const adapter = await acquired(claude) + + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + await adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 6 }) + ).resolves.toEqual({ cancelled: false }) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-U', fence: 7 }) + ).resolves.toEqual({ cancelled: true }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 1 + ) + }) + + it('does not cancel an acknowledged turn after a later dispatch returns unknown', async () => { + const claude = fakeClaude({ replayUuids: ['turn-T', null] }) + const adapter = await acquired(claude) + + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-T', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ + state: 'accepted', + providerIdentity: { uuid: 'turn-T' } + }) + await expect( + adapter.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-U', + body: USER_MESSAGE, + fence: 7 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + expect(claude.connections[0].sent).toHaveLength(2) + + await expect( + adapter.cancelTurn({ sessionId: 'session-1', turnId: 'turn-T', fence: 7 }) + ).resolves.toEqual({ cancelled: false }) + expect(claude.connections[0].calls.filter((call) => call.subtype === 'interrupt')).toHaveLength( + 0 + ) + }) + + it('classifies provider-declined options without treating timeouts as settled', async () => { + const claude = fakeClaude({ + routes: { + set_model: () => { + throw new ClaudeControlRequestError('set_model', 'model unavailable') + } + } + }) + const adapter = await acquired(claude) + + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'fable', fence: 7 }) + ).rejects.toMatchObject({ name: 'AgentSessionOptionRejectedError' }) + claude.routes.set_model = () => { + throw new Error('claude set_model request timed out') + } + await expect( + adapter.setOption({ sessionId: 'session-1', key: 'model', value: 'opus', fence: 7 }) + ).rejects.toThrow('timed out') + }) + + it('hydrates live model choices and maps the resolved current model to its CLI id', async () => { + const claude = fakeClaude({ + initModel: 'claude-sonnet-5', + routes: { + list_models: () => [ + { value: 'default', resolvedModel: 'claude-opus-5', displayName: 'Default' }, + { + value: 'opus', + resolvedModel: 'claude-opus-5', + displayName: 'Opus', + supportsEffort: true, + supportedEffortLevels: ['low', 'high'] + }, + { + value: 'sonnet', + resolvedModel: 'claude-sonnet-5', + displayName: 'Sonnet' + } + ] + } + }) + const adapter = await acquired(claude) + + await expect(adapter.readOptions({ sessionId: 'session-1', fence: 7 })).resolves.toEqual({ + models: [ + { + id: 'opus', + label: 'Opus', + isDefault: true, + efforts: [ + { value: 'low', label: 'Low' }, + { value: 'high', label: 'High' } + ] + }, + { id: 'sonnet', label: 'Sonnet', isDefault: false, efforts: [] } + ], + current: { model: 'sonnet', effort: 'high', confirmed: ['model', 'effort'] } + }) + }) + + it('keeps the shared Claude seed when live model discovery is unavailable', async () => { + const claude = fakeClaude({ + initModel: 'custom-model', + routes: { + list_models: () => { + throw new Error('unsupported') + } + } + }) + const adapter = await acquired(claude) + const result = await adapter.readOptions({ sessionId: 'session-1', fence: 7 }) + + expect(result.models.map((model) => model.id)).toEqual([ + 'fable', + 'opus', + 'sonnet', + 'haiku', + 'custom-model' + ]) + expect(result.current).toEqual({ + model: 'custom-model', + effort: 'high', + confirmed: ['model', 'effort'] + }) + }) +}) + +describe('ClaudeStructuredSessionAdapter acquisition cleanup', () => { + /** A start that fails after the child self-exited, with its close verdict scripted. */ + function failedStart( + unprovenCloseVerdict: ClaudeStreamJsonConnection['exitVerdict'] + ): Promise { + const claude = fakeClaude({ + exitBeforeInit: 'claude stream-json exited (code 1): not logged in', + unprovenCloseVerdict + }) + return adapterFor(claude) + .acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + .catch((error: unknown) => error) + } + + it('releases on a first-hand root exit while still carrying the CLI diagnostic', async () => { + // The root's pid and start time are the lease's identity, and they are + // provably dead: latching the session would strand a signed-out user. + const error = await failedStart({ root: 'exited', tree: 'unverifiable' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): not logged in') + }) + + it('never releases while a descendant was observed alive', async () => { + const error = await failedStart({ root: 'exited', tree: 'live' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + it('never releases for a root Orca never saw leave', async () => { + const error = await failedStart({ root: 'live', tree: 'unverifiable' }) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + /** A published session whose CLI then exits first-hand, with the verdict its ladder holds. */ + async function exitedAfterPublish( + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + ): Promise<{ adapter: ClaudeStructuredSessionAdapter; connection: FakeConnection }> { + const claude = fakeClaude({ unprovenCloseVerdict: exitVerdict }) + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + return { adapter, connection } + } + + it('classifies cleanup after a first-hand exit removed the session as a root exit, never as proven', async () => { + // The host may still be committing or proving the lease when the child dies; + // its cleanup must find the exit the ladder observed, not an absence. + const { adapter, connection } = await exitedAfterPublish({ + root: 'exited', + tree: 'unverifiable' + }) + const error = await adapter.releaseAcquisition({ sessionId: 'session-1' }).catch((e) => e) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): crashed') + expect(connection.closeCount).toBe(2) + }) + + it('never releases after an exit that left a descendant observed alive', async () => { + const { adapter } = await exitedAfterPublish({ root: 'exited', tree: 'live' }) + const error = await adapter.releaseAcquisition({ sessionId: 'session-1' }).catch((e) => e) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + }) + + it('forgets a retained exit once the session is acquired again', async () => { + const options: Parameters[0] = {} + const claude = fakeClaude(options) + const adapter = await acquired(claude) + const first = claude.connections[0] + first.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + first.exitVerdict = { root: 'exited', tree: 'unverifiable' } + first.close = async () => false + options.exitBeforeInit = 'claude stream-json exited (code 1): not logged in' + + await expect( + adapter.acquire({ identity: identityFor(), fence: 8, spawnToken: 'spawn-10' }) + ).rejects.toThrow('not logged in') + // The second start's own proven close is the answer; the first exit is stale. + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + expect(first.closeCount).toBe(1) + }) + + it('reports unproven published-session cleanup so callers can retry safely', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as FakeConnection['close'] + + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(false) + expect(await adapter.readOptions({ sessionId: 'session-1', fence: 7 })).toMatchObject({ + current: { model: 'claude-sonnet-5' } + }) + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + expect(() => adapter.readOptions({ sessionId: 'session-1', fence: 7 })).toThrow( + 'no live claude stream-json session' + ) + }) + + it('does not report a second release as successful while retained exit evidence is unproven', async () => { + const claude = fakeClaude({ unprovenCloseVerdict: { root: 'exited', tree: 'unverifiable' } }) + const adapter = await acquired(claude) + const connection = claude.connections[0] + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + connection.close = vi.fn().mockResolvedValue(false) as unknown as FakeConnection['close'] + + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBeInstanceOf( + AgentSessionAcquisitionRootExitObservedError + ) + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBeInstanceOf( + AgentSessionAcquisitionRootExitObservedError + ) + expect(connection.close).toHaveBeenCalledTimes(2) + }) + + it('keeps shutdown pending until a retained unexpected-exit proof settles', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const connection = claude.connections[0] + const proof = Promise.withResolvers() + connection.close = vi + .fn<() => Promise>() + .mockImplementationOnce(() => proof.promise) + .mockResolvedValueOnce(true) as unknown as FakeConnection['close'] + + connection.handlers.onExit?.(new Error('crashed')) + await tick() + let settled = false + const closing = adapter.closeAll().then(() => { + settled = true + }) + await tick() + expect(settled).toBe(false) + + proof.resolve(false) + await expect(closing).resolves.toBeUndefined() + expect(connection.close).toHaveBeenCalledTimes(2) + }) + + it('does not claim shutdown success for a retained false exit proof', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const connection = claude.connections[0] + connection.close = vi + .fn<() => Promise>() + .mockResolvedValue(false) as unknown as FakeConnection['close'] + + connection.handlers.onExit?.(new Error('crashed')) + await tick() + + await expect(adapter.closeAll()).rejects.toThrow( + 'claude structured session shutdown could not prove every child stopped' + ) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + expect(connection.close).toHaveBeenCalledTimes(4) + }) +}) + +describe('ClaudeStructuredSessionAdapter prompts', () => { + it('turns can_use_tool into an addressable durable approval that settles the SDK callback', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-1', 'tool-1', { + input: { command: 'git status' }, + suggestions: [{ type: 'addRules' }] + }) + expect(events.at(-1)).toMatchObject({ + type: 'prompt', + prompt: { kind: 'approval', toolName: 'Bash', promptKey: 'permission-1' } + }) + + adapter.bindPromptItemId('session-1', 'journal-approval', 'permission-1') + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-approval', + kind: 'approval', + optionId: 'allowForSession', + fence: 7 + }) + // The answer resolves the SDK's own callback promise; the SDK writes the wire response. + await expect(answered.promise).resolves.toEqual({ + behavior: 'allow', + updatedInput: { command: 'git status' }, + updatedPermissions: [{ type: 'addRules' }], + toolUseID: 'tool-1' + }) + }) + + it('collects every AskUserQuestion card before settling the one callback', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const answered = invokeCanUseTool( + claude.connections[0], + 'AskUserQuestion', + 'question-1', + 'tool-question', + { + input: { + questions: [ + { question: 'Library?', options: [{ label: 'Luxon' }] }, + { question: 'Ship now?', options: [{ label: 'Yes' }] } + ] + } + } + ) + adapter.bindPromptItemId('session-1', 'journal-q1', 'question-1', 'Library?') + adapter.bindPromptItemId('session-1', 'journal-q2', 'question-1', 'Ship now?') + + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-q1', + kind: 'question', + optionId: encodeClaudeQuestionOptionId('Library?', 'Luxon'), + fence: 7 + }) + await tick() + expect(answered.settled()).toBe(false) + await adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-q2', + kind: 'question', + optionId: encodeClaudeQuestionOptionId('Ship now?', 'Yes'), + fence: 7 + }) + await expect(answered.promise).resolves.toMatchObject({ + behavior: 'allow', + updatedInput: { answers: { 'Library?': 'Luxon', 'Ship now?': 'Yes' } }, + toolUseID: 'tool-question' + }) + }) + + it('leaves a prompt cancelled and unanswerable once the SDK abort signal fires', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = await acquired(claude, {}, events) + const controller = new AbortController() + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-9', 'tool-9', { + input: { command: 'rm -rf /' }, + signal: controller.signal + }) + adapter.bindPromptItemId('session-1', 'journal-9', 'permission-9') + + controller.abort() + // A cancelled request is forgotten and settled with null — never an authorization. + await expect(answered.promise).resolves.toBeNull() + expect(events.at(-1)).toMatchObject({ type: 'prompt-cancelled', promptKey: 'permission-9' }) + // A late answer after the abort must not authorize the wrong tool. + await expect( + adapter.answerPrompt({ + sessionId: 'session-1', + itemId: 'journal-9', + kind: 'approval', + optionId: 'allow', + fence: 7 + }) + ).rejects.toThrow(/no longer waiting/) + }) + + it('settles an in-flight permission callback when the session closes, leaving no dangling promise', async () => { + const claude = fakeClaude() + const adapter = await acquired(claude) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-close', 'tool-c', { + input: { command: 'ls' } + }) + await tick() + expect(answered.settled()).toBe(false) + + await adapter.closeSession('session-1') + + await expect(answered.promise).resolves.toBeNull() + }) +}) diff --git a/src/main/claude/claude-structured-session-adapter.ts b/src/main/claude/claude-structured-session-adapter.ts new file mode 100644 index 00000000000..f28b6e37f8f --- /dev/null +++ b/src/main/claude/claude-structured-session-adapter.ts @@ -0,0 +1,246 @@ +import type { + AgentSessionAcquisition, + StructuredAgentSessionAcquireInput, + StructuredAgentSessionAdapter +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { answerClaudePrompt, cancelClaudeTurn } from './claude-structured-control-actions' +import { dispatchClaudeTurn } from './claude-structured-dispatch' +import { releaseClaudeAcquisition } from './claude-structured-acquisition-release' +import { acquireClaudeSession } from './claude-structured-session-acquisition' +export { CLAUDE_STRUCTURED_INIT_TIMEOUT_MS } from './claude-structured-session-acquisition' +import { supportsClaudeStructuredLocation } from './claude-structured-location-support' +import { setClaudeStructuredOption } from './claude-structured-options' +import { readClaudeStructuredSessionOptions } from './claude-structured-session-options' +import { + ClaudeAcquisitionRegistry, + type ClaudeAcquisitionAttempt, + type ClaudeSession, + type ClaudeSessionExit, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-state' +import { + closeAllClaudeSessions, + closeClaudeSession, + settleClaudeExitedSession +} from './claude-structured-session-close' +import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' + +export type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' +export type { + ClaudeAuthDiagnostic, + ClaudeStructuredSessionAdapterDeps, + ClaudeStructuredSessionEvent +} from './claude-structured-session-state' + +const DISPATCH_ACK_TIMEOUT_MS = 10_000 + +export class ClaudeStructuredSessionAdapter implements StructuredAgentSessionAdapter { + private readonly sessions = new Map() + private readonly acquisitions = new ClaudeAcquisitionRegistry() + private readonly exits = new Map() + + constructor(private readonly deps: ClaudeStructuredSessionAdapterDeps) {} + + supportsLocation = supportsClaudeStructuredLocation + + acquire = (input: StructuredAgentSessionAcquireInput): Promise => + acquireClaudeSession({ + input, + deps: this.deps, + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + callbacks: { + deliver: (attempt, sessionId, event) => this.deliver(attempt, sessionId, event), + emit: (session, events, event) => this.emit(session, events, event), + handleExit: (sessionId, attempt, error) => this.handleExit(sessionId, attempt, error), + settleExit: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit) + } + }) + + private deliver(attempt: ClaudeAcquisitionAttempt, sessionId: string, event: () => void): void { + if (!attempt.published) { + attempt.buffered.push(event) + return + } + if (this.sessions.get(sessionId)?.connection === attempt.connection) { + event() + } + } + + private handleExit(sessionId: string, attempt: ClaudeAcquisitionAttempt, error: Error): void { + const session = this.sessions.get(sessionId) + if (!session || session.connection !== attempt.connection) { + return + } + this.sessions.delete(sessionId) + // Re-enter the provider's close ladder before publishing lifecycle recovery. + // An exit callback is root evidence only; the retained tree proof must run + // before the host releases and reacquires this exact child. + const closePromise = session.connection.close().catch(() => false) + const exit: ClaudeSessionExit = { + connection: session.connection, + session, + error, + closePromise + } + this.exits.set(sessionId, exit) + void closePromise + .then((proven) => (proven ? this.settleUnexpectedExit(sessionId, exit) : undefined)) + .catch(() => undefined) + } + + /** Lifecycle recovery is published only after the child tree proof is true. */ + private settleUnexpectedExit(sessionId: string, exit: ClaudeSessionExit): Promise { + exit.settlementPromise ??= (async () => { + if (this.exits.get(sessionId) !== exit) { + settleClaudeExitedSession(exit.session) + return + } + // Persist the transcript-derived cursor before publishing the lifecycle + // event that lets the host release and reacquire this exact child. + await this.persistSessionHandle(sessionId, exit.session).catch(() => undefined) + if (this.exits.get(sessionId) !== exit) { + settleClaudeExitedSession(exit.session) + return + } + this.exits.delete(sessionId) + const ended: ClaudeStructuredSessionEvent = { + type: 'ended', + sessionId, + reason: exit.error.message, + cause: 'unexpected-exit', + fence: exit.session.fence, + acquisitionGeneration: exit.session.acquisitionGeneration + } + try { + this.emit(exit.session, exit.session.events, ended) + } finally { + settleClaudeExitedSession(exit.session) + } + })() + return exit.settlementPromise + } + + private async persistSessionHandle(sessionId: string, session: ClaudeSession): Promise { + try { + const transcriptLeaf = this.deps.readTranscriptLeaf + ? await readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf: this.deps.readTranscriptLeaf, + providerSessionId: session.providerSessionId, + previousLeafUuid: session.leafUuid, + claudeConfigDir: session.claudeConfigDir + }) + : null + if (transcriptLeaf) { + session.leafUuid = transcriptLeaf + } + } catch { + // A stale or unavailable tail must not overwrite the last observed leaf. + } + await this.deps.persistHandle?.({ + sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + } + + private emit( + _session: ClaudeSession | null, + _events: StructuredAgentSessionEventSink | undefined, + event: ClaudeStructuredSessionEvent + ): void { + _session?.translator?.handle(event) + this.deps.onEvent?.(event) + } + + bindPromptItemId( + sessionId: string, + journalItemId: string, + promptKey: string, + questionId?: string + ): void { + this.sessions.get(sessionId)?.prompts.bindJournalItemId(journalItemId, promptKey, questionId) + } + + dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => + dispatchClaudeTurn( + this.session(input.sessionId), + input, + this.deps.dispatchAckTimeoutMs ?? DISPATCH_ACK_TIMEOUT_MS + ) + + cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => { + const session = this.session(input.sessionId) + const acquisitionGeneration = session.acquisitionGeneration + return cancelClaudeTurn(session, this.deps.requestTimeoutMs, () => { + // Keep every ownership check adjacent to the provider interrupt. The + // session map check fences a replaced child; the turn check fences a + // delayed cancel after a newer turn was admitted on the same child. + return ( + this.sessions.get(input.sessionId) === session && + session.fence === input.fence && + session.acquisitionGeneration === acquisitionGeneration && + (session.activeTurnId === undefined + ? session.dispatchSequence === 0 + : session.activeTurnId === input.turnId && + session.activeTurnSequence === session.dispatchSequence) + ) + }) + } + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => + answerClaudePrompt(this.session(input.sessionId), input) + setOption: StructuredAgentSessionAdapter['setOption'] = (input) => + setClaudeStructuredOption(this.session(input.sessionId), input, this.deps.requestTimeoutMs) + readOptions = (input: { sessionId: string; fence: number }) => + readClaudeStructuredSessionOptions(this.session(input.sessionId), this.deps.requestTimeoutMs) + + readOptionRestoreFailures = (sessionId: string): readonly string[] => [ + ...(this.sessions.get(sessionId)?.restoreSkippedOptions ?? []) + ] + + releaseAcquisition = (input: { sessionId: string }): Promise => + releaseClaudeAcquisition({ + sessionId: input.sessionId, + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + onExitProven: (sessionId, exit) => this.settleUnexpectedExit(sessionId, exit), + ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) + }) + + closeSession = (sessionId: string): Promise => { + if (this.exits.has(sessionId)) { + return this.releaseAcquisition({ sessionId }) + } + return closeClaudeSession({ + sessionId, + sessions: this.sessions, + acquisitions: this.acquisitions, + ...(this.deps.persistHandle ? { persistHandle: this.deps.persistHandle } : {}), + ...(this.deps.readTranscriptLeaf ? { readTranscriptLeaf: this.deps.readTranscriptLeaf } : {}), + ...(this.deps.onEvent ? { onEvent: this.deps.onEvent } : {}) + }) + } + + closeAll = (): Promise => + closeAllClaudeSessions({ + sessions: this.sessions, + acquisitions: this.acquisitions, + exits: this.exits, + closeSession: this.closeSession, + closeExit: (sessionId) => this.releaseAcquisition({ sessionId }) + }) + + private session(sessionId: string): ClaudeSession { + const session = this.sessions.get(sessionId) + if (!session) { + throw new Error(`no live claude stream-json session for ${sessionId}`) + } + return session + } +} diff --git a/src/main/claude/claude-structured-session-close.test.ts b/src/main/claude/claude-structured-session-close.test.ts new file mode 100644 index 00000000000..f92975e9ef4 --- /dev/null +++ b/src/main/claude/claude-structured-session-close.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { + ClaudeStructuredSessionAdapterDeps, + ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { adapterFor, fakeClaude, identityFor } from './claude-structured-session-test-support' + +describe('Claude published session close lifecycle', () => { + it('ends the session even when the durable handle write rejects', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistenceError = new Error('store unavailable') + const persistHandle = vi + .fn>() + .mockRejectedValueOnce(persistenceError) + .mockResolvedValueOnce(undefined) + const adapter = adapterFor(claude, {}, events, [], undefined, undefined, persistHandle) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + await expect(adapter.closeSession('session-1')).rejects.toBe(persistenceError) + // The child is provably dead; a failed cursor write may not suppress the end. + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(0) + expect(disposeTranslator).toHaveBeenCalledOnce() + + await expect(adapter.closeSession('session-1')).resolves.toBe(true) + expect(persistHandle).toHaveBeenCalledTimes(2) + // The retry persists the same cursor without a second lifecycle end. + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) +}) diff --git a/src/main/claude/claude-structured-session-close.ts b/src/main/claude/claude-structured-session-close.ts new file mode 100644 index 00000000000..d37d3917796 --- /dev/null +++ b/src/main/claude/claude-structured-session-close.ts @@ -0,0 +1,256 @@ +import type { + ClaudeAcquisitionRegistry, + ClaudeSession, + ClaudeSessionExit, + ClaudeStructuredSessionEvent +} from './claude-structured-session-state' +import { cancelClaudeAcquisitionAttempt } from './claude-structured-session-state' +import { + AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, + AgentSessionPreSpawnError +} from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeStreamJsonConnection } from './claude-stream-json-connection' +import { closeProcessRegistry } from '../../shared/child-process/close-process-registry' +import { readClaudeTranscriptLeafWithReproof } from './claude-transcript-branch-proof' + +export function claudeAcquisitionCleanupError( + connection: ClaudeStreamJsonConnection | null | undefined, + cause: unknown +): Error { + const verdict = connection?.exitVerdict + if (verdict?.root === 'processless') { + return new AgentSessionPreSpawnError(cause) + } + return verdict?.root === 'exited' && verdict.tree === 'unverifiable' + ? new AgentSessionAcquisitionRootExitObservedError(cause) + : new AgentSessionAcquisitionExitUnprovenError(cause) +} + +export function settleClaudeDispatchWaiters(session: ClaudeSession): void { + for (const waiter of session.dispatchWaiters.splice(0)) { + clearTimeout(waiter.timer) + waiter.resolve(null) + } +} + +export function settleClaudeExitedSession(session: ClaudeSession): void { + settleClaudeDispatchWaiters(session) + for (const prompt of session.prompts.clear()) { + prompt.settle(null) + } + session.translator?.dispose() +} + +type CloseClaudePublishedSessionInput = { + sessions: Map + sessionId: string + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise +} + +async function finalizeClaudePublishedSession( + input: CloseClaudePublishedSessionInput, + session: ClaudeSession +): Promise { + settleClaudeDispatchWaiters(session) + // Settle every in-flight permission callback so closing leaves no dangling promise; `null` + // writes no response, and the SDK ignores any post-cleanup answer regardless. + for (const prompt of session.prompts.clear()) { + prompt.settle(null) + } + if ((await session.connection.close()) !== true) { + return false + } + try { + const transcriptLeaf = input.readTranscriptLeaf + ? await readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf: input.readTranscriptLeaf, + providerSessionId: session.providerSessionId, + previousLeafUuid: session.leafUuid, + claudeConfigDir: session.claudeConfigDir + }) + : null + if (transcriptLeaf) { + session.leafUuid = transcriptLeaf + } + } catch { + // Keep the last observed main-transcript frame when the durable tail is + // unavailable or proves a stale/divergent branch. + } + const persistence = + session.closePersistence ?? + (session.closePersistence = (async () => { + await input.persistHandle?.({ + sessionId: input.sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + })()) + const ended = { + type: 'ended', + sessionId: input.sessionId, + reason: 'claude session closed' + } as const + let callbackError: unknown + let callbackThrew = false + const deliver = (event: ClaudeStructuredSessionEvent): void => { + try { + input.onEvent?.(event) + } catch (error) { + callbackThrew = true + callbackError ??= error + } + } + let persistenceError: unknown + try { + await persistence + session.closeFinalized = true + input.sessions.delete(input.sessionId) + deliver({ + type: 'handle', + sessionId: input.sessionId, + providerSessionId: session.providerSessionId, + leafUuid: session.leafUuid, + fence: session.fence + }) + } catch (error) { + // Keep the closed session indexed so a retry can persist the same cursor. + // Removing it first would turn a durable-write failure into a no-op retry. + if (session.closePersistence === persistence) { + session.closePersistence = undefined + } + persistenceError = error + } + // The connection already proved the child dead, so the session has ended + // whatever the durable write did: withholding it would strand the renderer on + // a session nothing re-drives. Emitted once, so a retry only re-persists. + if (!session.closeEnded) { + session.closeEnded = true + try { + try { + session.translator?.handle(ended) + } catch (error) { + callbackThrew = true + callbackError ??= error + } + deliver(ended) + } finally { + session.translator?.dispose() + } + } + if (persistenceError) { + throw persistenceError + } + if (callbackThrew) { + throw callbackError + } + return true +} + +export async function closeClaudePublishedSession( + input: CloseClaudePublishedSessionInput +): Promise { + const session = input.sessions.get(input.sessionId) + if (!session) { + return true + } + if (session.closeFinalized) { + return true + } + if (session.closeFinalization) { + return session.closeFinalization + } + const finalization = finalizeClaudePublishedSession(input, session) + session.closeFinalization = finalization + try { + return await finalization + } finally { + if (session.closeFinalization === finalization && !session.closeFinalized) { + session.closeFinalization = undefined + } + } +} + +export function closeClaudePublishedSessionForDeps( + sessions: Map, + sessionId: string, + deps: { + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise + } +): Promise { + return closeClaudePublishedSession({ sessions, sessionId, ...deps }) +} + +export async function closeClaudeSession(input: { + sessionId: string + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + persistHandle?: (handle: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise +}): Promise { + const attempt = input.acquisitions.get(input.sessionId) + if (!(await cancelClaudeAcquisitionAttempt(attempt))) { + return false + } + if (attempt) { + input.acquisitions.deleteIfCurrent(input.sessionId, attempt) + } + return closeClaudePublishedSession(input) +} + +export async function closeAllClaudeSessions(input: { + sessions: Map + acquisitions: ClaudeAcquisitionRegistry + exits: Map + closeSession: (sessionId: string) => Promise + closeExit: (sessionId: string) => Promise +}): Promise { + input.acquisitions.close() + await closeProcessRegistry({ + attempts: 3, + hasEntries: () => + input.sessions.size > 0 || input.acquisitions.size > 0 || input.exits.size > 0, + entryIds: () => + new Set([ + ...input.sessions.keys(), + ...input.acquisitions.sessionIds(), + ...input.exits.keys() + ]), + closeEntry: async (sessionId) => + input.exits.has(sessionId) ? input.closeExit(sessionId) : input.closeSession(sessionId), + failureMessage: 'claude structured session shutdown could not prove every child stopped' + }) +} diff --git a/src/main/claude/claude-structured-session-options.ts b/src/main/claude/claude-structured-session-options.ts new file mode 100644 index 00000000000..afb4fd65076 --- /dev/null +++ b/src/main/claude/claude-structured-session-options.ts @@ -0,0 +1,183 @@ +import type { + AgentSessionModelOption, + AgentSessionOptionChoice, + AgentSessionOptionsResult +} from '../../shared/agent-session-wire' +import { CLAUDE_SESSION_OPTION_CATALOG } from '../../shared/agent-session-option-catalog-claude-codex' +import type { CatalogModel } from '../../shared/agent-session-option-catalog-types' +import type { ClaudeSession } from './claude-structured-session-state' + +type ListedModel = AgentSessionModelOption & { resolvedModel: string | null } + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +function text(value: unknown): string | null { + return typeof value === 'string' && value.trim() ? value : null +} + +/** + * The session's current effort, which only `get_settings` reports: the + * `system/init` frame carries `model` but has never carried an effort of any + * kind. Null when the provider stops reporting it, so the pill goes empty + * rather than showing an effort nothing measured. + */ +export function readClaudeSettingsEffort(settings: unknown): string | null { + return text(record(record(settings)?.effective)?.effortLevel) +} + +function effortLabel(value: string): string { + return value === 'xhigh' ? 'Extra high' : `${value.charAt(0).toUpperCase()}${value.slice(1)}` +} + +function listedEfforts(row: Record): AgentSessionOptionChoice[] { + return row.supportsEffort === true && Array.isArray(row.supportedEffortLevels) + ? row.supportedEffortLevels.flatMap((value) => { + const effort = text(value) + return effort ? [{ value: effort, label: effortLabel(effort) }] : [] + }) + : [] +} + +function listedModels(value: unknown): ListedModel[] { + const response = record(value) + const rows = Array.isArray(response?.models) + ? response.models.map(record).filter((row): row is Record => row !== null) + : [] + const defaultRow = rows.find((row) => text(row.value) === 'default') + const defaultResolvedModel = text(defaultRow?.resolvedModel) + const seen = new Set() + return rows.flatMap((row) => { + const id = text(row.value) + if (!id || id === 'default' || seen.has(id)) { + return [] + } + seen.add(id) + const resolvedModel = text(row.resolvedModel) + const description = text(row.description) + return [ + { + id, + label: text(row.displayName) ?? id, + ...(description ? { description } : {}), + isDefault: resolvedModel !== null && resolvedModel === defaultResolvedModel, + efforts: listedEfforts(row), + resolvedModel + } + ] + }) +} + +function seedEfforts(model: CatalogModel): AgentSessionOptionChoice[] { + const effort = model.options.find((option) => option.id === 'effort') + return effort?.kind.type === 'select' ? effort.kind.choices : [] +} + +function seedModels(): ListedModel[] { + return CLAUDE_SESSION_OPTION_CATALOG.models.map((model) => ({ + id: model.id, + label: model.label, + ...(model.description ? { description: model.description } : {}), + isDefault: model.isDefault === true, + efforts: seedEfforts(model), + resolvedModel: null + })) +} + +function currentModelId(models: ListedModel[], reportedModel: string | undefined): string { + const matched = reportedModel + ? models.find((model) => model.id === reportedModel || model.resolvedModel === reportedModel) + : undefined + return ( + matched?.id ?? reportedModel ?? models.find((model) => model.isDefault)?.id ?? models[0]!.id + ) +} + +/** + * The model the session is running. A report the CLI made after the last write + * outranks the write: it names the model the session ran. An older one does not + * — a model set between turns has no report yet, and deferring to the previous + * turn's would flip the pill back. + * + * Sole resolver of that question: every surface that acts on "the current model" + * — the pill, the effort guard, the rejection it names — reads it here, so two + * of them cannot answer it differently and offer an effort a third then refuses. + */ +export function readClaudeCurrentModel(session: ClaudeSession): { + id: string | undefined + confirmed: boolean +} { + const confirmed = + session.reportedModelMutation === session.optionMutationSequence && + session.reportedOptions.model !== undefined + return { + id: confirmed + ? session.reportedOptions.model + : (session.options.get('model') ?? session.reportedOptions.model), + confirmed + } +} + +/** + * The effort levels the session's current model advertises, with the catalog id + * that matched so a refusal names the model the pill shows. Levels are null when + * nothing identified the model: `apply_flag_settings` accepts and stores any + * level for a model with no effort control, so the catalog is the only evidence + * of a refusal — and an absent or unlisted one is not evidence, or a live CLI + * that predates `list_models` would have every effort refused under it. + */ +export async function readClaudeModelEffortLevels( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise<{ modelId: string | undefined; levels: ReadonlySet | null }> { + const modelId = readClaudeCurrentModel(session).id + if (!modelId) { + return { modelId, levels: null } + } + const catalog = await session.connection.supportedModels({ timeoutMs }).catch(() => null) + const matched = catalog + ? listedModels({ models: catalog }).find( + (model) => model.id === modelId || model.resolvedModel === modelId + ) + : undefined + return { + modelId: matched?.id ?? modelId, + levels: matched ? new Set(matched.efforts.map((choice) => choice.value)) : null + } +} + +export async function readClaudeStructuredSessionOptions( + session: ClaudeSession, + timeoutMs: number | undefined +): Promise { + const catalog = await session.connection.supportedModels({ timeoutMs }).catch(() => null) + const discovered = listedModels(catalog ? { models: catalog } : null) + const models = discovered.length > 0 ? discovered : seedModels() + const current = readClaudeCurrentModel(session) + const model = currentModelId(models, current.id) + if (!models.some((entry) => entry.id === model)) { + models.push({ id: model, label: model, isDefault: false, efforts: [], resolvedModel: null }) + } + const effort = session.options.get('effort') ?? session.reportedOptions.effort + const confirmed = [ + ...(current.confirmed ? ['model'] : []), + ...(effort && session.confirmedOptions.has('effort') ? ['effort'] : []) + ] + return { + models: models.map((entry) => ({ + id: entry.id, + label: entry.label, + ...(entry.description ? { description: entry.description } : {}), + isDefault: entry.isDefault, + efforts: entry.efforts + })), + current: { + model, + ...(effort ? { effort } : {}), + ...(confirmed.length > 0 ? { confirmed } : {}) + } + } +} diff --git a/src/main/claude/claude-structured-session-publication.ts b/src/main/claude/claude-structured-session-publication.ts new file mode 100644 index 00000000000..29d1113c814 --- /dev/null +++ b/src/main/claude/claude-structured-session-publication.ts @@ -0,0 +1,68 @@ +import type { AgentSessionAcquisition } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import type { ClaudeInitObservation } from './claude-structured-init-proof' +import { claudeProviderHandleLink } from './claude-structured-owner-identity' +import type { ClaudePromptRegistry } from './claude-structured-prompt-replies' +import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudeSession } from './claude-structured-session-state' + +export function createClaudeSessionPublication(input: { + connection: ClaudeSession['connection'] + init: ClaudeInitObservation + claudeConfigDir: string + leafUuid: string | null + fence: number + acquisitionGeneration: string + resumed: boolean + prompts: ClaudePromptRegistry + translator: ClaudeJournalTranslator | null + events: ClaudeSession['events'] + process: AgentSessionAcquisition['process'] + linkId?: string + observedAt: number + options?: ReadonlyMap + capabilities: readonly string[] + /** Read from `get_settings`; `system/init` never reports an effort. */ + effort: string | null +}): { acquisition: AgentSessionAcquisition; session: ClaudeSession } { + const model = input.init.model + const effort = input.effort + return { + acquisition: { + process: input.process, + link: claudeProviderHandleLink({ + sessionId: input.init.providerSessionId, + leafUuid: input.leafUuid, + resumed: input.resumed, + fence: input.fence, + ...(input.linkId ? { linkId: input.linkId } : {}), + observedAt: input.observedAt + }), + acquisitionGeneration: input.acquisitionGeneration + }, + session: { + connection: input.connection, + providerSessionId: input.init.providerSessionId, + claudeConfigDir: input.claudeConfigDir, + leafUuid: input.leafUuid, + fence: input.fence, + acquisitionGeneration: input.acquisitionGeneration, + prompts: input.prompts, + dispatchWaiters: [], + retiredDispatchWaiters: [], + replayContentFallbackBlocked: false, + dispatchSequence: 0, + optionMutationSequence: 0, + options: new Map(input.options), + capabilities: input.capabilities, + reportedOptions: { + ...(model ? { model } : {}), + ...(effort ? { effort } : {}) + }, + reportedModelMutation: 0, + confirmedOptions: new Set(effort ? ['effort'] : []), + restoreSkippedOptions: new Set(), + translator: input.translator, + events: input.events + } + } +} diff --git a/src/main/claude/claude-structured-session-recovery.test.ts b/src/main/claude/claude-structured-session-recovery.test.ts new file mode 100644 index 00000000000..5bfe56cf156 --- /dev/null +++ b/src/main/claude/claude-structured-session-recovery.test.ts @@ -0,0 +1,619 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { ClaudeTranscriptPreviousCursorMissingError } from './claude-transcript-branch-proof' +import { + adapterFor, + fakeClaude, + identityFor, + invokeCanUseTool, + PROVIDER_SESSION_ID, + tick +} from './claude-structured-session-test-support' + +describe('ClaudeStructuredSessionAdapter transcript-derived recovery', () => { + it('shares concurrent close finalization and emits lifecycle once', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistence = Promise.withResolvers() + const persistHandle = vi.fn(() => persistence.promise) + const adapter = adapterFor(claude, {}, events, [], undefined, undefined, persistHandle) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + const first = adapter.closeSession('session-1') + const second = adapter.closeSession('session-1') + await tick() + expect(persistHandle).toHaveBeenCalledOnce() + expect(claude.connections[0].closeCount).toBe(1) + + persistence.resolve() + await expect(Promise.all([first, second])).resolves.toEqual([true, true]) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) + + it('still emits ended and disposes state when handle delivery throws', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const callbackError = new Error('handle delivery failed') + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => { + events.push(event) + if (event.type === 'handle') { + throw callbackError + } + }, + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + persistHandle: vi.fn(async () => undefined) + }) + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const session = ( + adapter as unknown as { + sessions: Map void } | null }> + } + ).sessions.get('session-1') + const disposeTranslator = vi.spyOn(session!.translator!, 'dispose') + + await expect(adapter.closeSession('session-1')).rejects.toBe(callbackError) + expect(events.filter((event) => event.type === 'handle')).toHaveLength(1) + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + expect(disposeTranslator).toHaveBeenCalledOnce() + }) + + it('retains a closed session until its durable cursor persistence succeeds', async () => { + const claude = fakeClaude() + const persistenceError = new Error('store unavailable') + const persistHandle = vi + .fn>() + .mockRejectedValueOnce(persistenceError) + .mockResolvedValueOnce(undefined) + const adapter = adapterFor(claude, {}, [], [], undefined, undefined, persistHandle) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + await expect(adapter.closeSession('session-1')).rejects.toBe(persistenceError) + expect(persistHandle).toHaveBeenCalledTimes(1) + await expect(adapter.closeSession('session-1')).resolves.toBe(true) + expect(persistHandle).toHaveBeenCalledTimes(2) + }) + + it('persists only the last transcript-entry uuid before graceful close', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'assistant-leaf' + }) + claude.connections[0].handlers.onMessage?.({ + type: 'result', + session_id: PROVIDER_SESSION_ID, + uuid: 'result-frame-uuid' + }) + claude.connections[0].handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION_ID, + uuid: 'stream-event-frame-uuid' + }) + + await adapter.closeSession('session-1') + + expect(persistedHandles).toEqual([ + { + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'assistant-leaf', + fence: 7 + } + ]) + expect(events.at(-2)).toEqual({ + type: 'handle', + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'assistant-leaf', + fence: 7 + }) + expect(claude.connections[0].closeCount).toBe(1) + }) + + it('prefers a validated durable transcript leaf at graceful close', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-tail') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'durable-tail' }) + }) + + it('passes the pinned Claude account home to transcript validation', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-tail') + const adapter = adapterFor( + claude, + { claudeConfigDir: '/accounts/selected' }, + [], + persistedHandles, + undefined, + readTranscriptLeaf + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/selected' + }) + }) + + it('re-proves from the transcript root when the observed cursor is missing', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValueOnce(new ClaudeTranscriptPreviousCursorMissingError()) + .mockResolvedValueOnce('reproved-main-leaf') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(1, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(2, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: null, + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'reproved-main-leaf' }) + }) + + it('keeps the observed leaf when transcript validation proves a sibling branch', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValue(new Error('latest marker is on a sibling branch')) + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-tail' + }) + + await adapter.closeSession('session-1') + + expect(readTranscriptLeaf).toHaveBeenCalledTimes(1) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'observed-tail' }) + }) + + it('persists the last transcript leaf before an unexpected first-hand exit', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'crash-leaf' + }) + + claude.connections[0].handlers.onExit?.( + new Error('claude stream-json exited (code 1): crashed unexpectedly') + ) + await tick() + + expect(persistedHandles).toContainEqual({ + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'crash-leaf', + fence: 7 + }) + expect(events.at(-1)).toMatchObject({ + type: 'ended', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: expect.any(String) + }) + }) + + it('derives the crash cursor from the validated transcript tail', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const adapter = adapterFor( + claude, + {}, + [], + persistedHandles, + undefined, + vi.fn().mockResolvedValue('durable-crash-leaf') + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'stale-observed-tail' + }) + claude.connections[0].handlers.onExit?.( + new Error('claude stream-json exited (signal SIGKILL): crashed') + ) + await tick() + + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'durable-crash-leaf' }) + }) + + it('re-proves a first-hand crash cursor from the transcript root after stale validation', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValueOnce(new ClaudeTranscriptPreviousCursorMissingError()) + .mockResolvedValueOnce('reproved-crash-leaf') + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'stale-observed-tail' + }) + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(1, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'stale-observed-tail', + claudeConfigDir: '/accounts/claude' + }) + expect(readTranscriptLeaf).toHaveBeenNthCalledWith(2, { + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: null, + claudeConfigDir: '/accounts/claude' + }) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'reproved-crash-leaf' }) + }) + + it('keeps the observed crash leaf when transcript validation proves a sibling branch', async () => { + const claude = fakeClaude() + const persistedHandles: unknown[] = [] + const readTranscriptLeaf = vi + .fn() + .mockRejectedValue(new Error('latest marker is on a sibling branch')) + const adapter = adapterFor(claude, {}, [], persistedHandles, undefined, readTranscriptLeaf) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-crash-tail' + }) + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(readTranscriptLeaf).toHaveBeenCalledTimes(1) + expect(persistedHandles.at(-1)).toMatchObject({ leafUuid: 'observed-crash-tail' }) + }) + + it('publishes lifecycle recovery even when crash-cursor persistence fails', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor( + claude, + {}, + events, + [], + undefined, + undefined, + vi.fn().mockRejectedValue(new Error('store unavailable')) + ) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(events.at(-1)).toMatchObject({ type: 'ended', cause: 'unexpected-exit' }) + }) + + it('runs the child close proof before publishing unexpected-exit recovery', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const close = vi.spyOn(claude.connections[0], 'close').mockResolvedValue(true) + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(close).toHaveBeenCalledOnce() + expect(events.at(-1)).toMatchObject({ type: 'ended', cause: 'unexpected-exit' }) + }) + + it('does not publish recovery while an unexpected-exit close proof is false', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const adapter = adapterFor(claude, {}, events, persistedHandles) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + expect(persistedHandles).toEqual([]) + }) + + it('retains pending prompts while an unexpected-exit proof is unproven', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + const answered = invokeCanUseTool(claude.connections[0], 'Bash', 'permission-1', 'tool-1') + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValue(false) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + + expect(answered.settled()).toBe(false) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + }) + + it('publishes unexpected recovery exactly once after a retained proof retries successfully', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = adapterFor(claude, {}, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + claude.connections[0].close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof claude.connections)[0]['close'] + + claude.connections[0].handlers.onExit?.(new Error('crashed')) + await tick() + await expect(adapter.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(true) + await tick() + + expect(events.filter((event) => event.type === 'ended')).toHaveLength(1) + }) + + it('launches the first replacement from the settled retained transcript cursor', async () => { + const claude = fakeClaude() + const events: ClaudeStructuredSessionEvent[] = [] + const persistedHandles: unknown[] = [] + const journalSink: StructuredAgentSessionEventSink = { + appendItem: () => {}, + appendTombstone: () => {}, + publish: () => {} + } + const readTranscriptLeaf = vi.fn().mockResolvedValue('durable-retained-leaf') + let durableLeafUuid: string | null = null + const resolveLaunch = vi.fn(async ({ identity }) => { + if ( + identity.providerHandle.kind !== 'claude' || + identity.providerHandle.sessionId !== PROVIDER_SESSION_ID || + identity.providerHandle.leafUuid !== durableLeafUuid + ) { + throw new Error('claude durable resume identity changed before spawn') + } + if (durableLeafUuid === null) { + return { + pathToClaudeCodeExecutable: 'claude', + options: { sessionId: PROVIDER_SESSION_ID }, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false + } + } + return { + pathToClaudeCodeExecutable: 'claude', + options: { resume: PROVIDER_SESSION_ID, resumeSessionAt: durableLeafUuid }, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: durableLeafUuid, + resumed: true + } + }) + const persistHandle = vi.fn>( + async (handle) => { + durableLeafUuid = handle.leafUuid + persistedHandles.push(handle) + } + ) + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch, + openConnection: claude.openConnection, + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + readTranscriptLeaf, + persistHandle + }) + const firstAcquisition = await adapter.acquire({ + identity: identityFor(), + fence: 7, + spawnToken: 'spawn-9', + events: journalSink + }) + const first = claude.connections[0] + const oldPrompt = invokeCanUseTool(first, 'Bash', 'permission-retained', 'tool-retained') + const oldSession = ( + adapter as unknown as { + sessions: Map< + string, + { + translator: { dispose: () => void } | null + prompts: { + find: (itemId: string) => { prompt: { settle: (value: unknown) => void } } | null + } + } + > + } + ).sessions.get('session-1') + expect(oldSession?.translator).not.toBeNull() + const disposeTranslator = vi.spyOn(oldSession!.translator!, 'dispose') + const pendingPrompt = oldSession?.prompts.find('permission-retained') + expect(pendingPrompt).not.toBeNull() + const settlePrompt = vi.spyOn(pendingPrompt!.prompt, 'settle') + first.handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION_ID, + uuid: 'observed-retained-leaf' + }) + first.close = vi + .fn<() => Promise>() + .mockResolvedValueOnce(false) + .mockResolvedValueOnce(true) as unknown as (typeof first)['close'] + first.handlers.onExit?.(new Error('crashed before replacement')) + await tick() + + expect(oldPrompt.settled()).toBe(false) + expect(events.filter((event) => event.type === 'ended')).toEqual([]) + + const replacement = await adapter.acquire({ + identity: { + ...identityFor(), + providerHandle: { + kind: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'observed-retained-leaf' + } + }, + fence: 8, + spawnToken: 'spawn-10', + events: journalSink + }) + + expect(disposeTranslator).toHaveBeenCalledOnce() + expect(settlePrompt).toHaveBeenCalledOnce() + expect(settlePrompt).toHaveBeenCalledWith(null) + expect(persistHandle).toHaveBeenCalledOnce() + expect(persistedHandles).toEqual([ + { + sessionId: 'session-1', + providerSessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf', + fence: 7 + } + ]) + expect(readTranscriptLeaf).toHaveBeenCalledOnce() + expect(readTranscriptLeaf).toHaveBeenCalledWith({ + providerSessionId: PROVIDER_SESSION_ID, + previousLeafUuid: 'observed-retained-leaf', + claudeConfigDir: '/accounts/claude' + }) + expect(resolveLaunch).toHaveBeenNthCalledWith(2, { + identity: { + ...identityFor(), + providerHandle: { + kind: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf' + } + } + }) + expect(oldPrompt.settled()).toBe(true) + expect(events.filter((event) => event.type === 'ended')).toEqual([ + { + type: 'ended', + sessionId: 'session-1', + reason: 'crashed before replacement', + cause: 'unexpected-exit', + fence: 7, + acquisitionGeneration: firstAcquisition.acquisitionGeneration + } + ]) + expect(replacement.link).toMatchObject({ + handle: { + provider: 'claude', + sessionId: PROVIDER_SESSION_ID, + leafUuid: 'durable-retained-leaf' + }, + origin: 'resumed', + mintedAtFence: 8 + }) + expect(claude.connections[1]?.launch.options).toMatchObject({ + resume: PROVIDER_SESSION_ID, + resumeSessionAt: 'durable-retained-leaf' + }) + expect(claude.connections).toHaveLength(2) + }) +}) diff --git a/src/main/claude/claude-structured-session-state.ts b/src/main/claude/claude-structured-session-state.ts new file mode 100644 index 00000000000..346ff686f76 --- /dev/null +++ b/src/main/claude/claude-structured-session-state.ts @@ -0,0 +1,276 @@ +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { StructuredAgentSessionEventSink } from '../native-chat/agent-session-wire/structured-agent-session-event-sink' +import type { + ClaudeStreamJsonConnection, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import type { ClaudeStructuredLaunch } from './claude-structured-launch-resolution' +import type { ClaudeJournalTranslator } from './claude-structured-journal-translation' +import type { ClaudePendingPrompt, ClaudePromptRegistry } from './claude-structured-prompt-replies' +import { cancelProcessAcquisition } from '../../shared/child-process/cancel-process-acquisition' +import { randomUUID } from 'node:crypto' + +export type ClaudeAuthDiagnostic = { + apiKeySourceConfigured: boolean + baseUrlConfigured: boolean + authTokenConfigured: boolean + apiKeyConfigured: boolean + settingSources: readonly string[] +} + +export type ClaudeStructuredSessionEvent = + | { + type: 'message' + sessionId: string + message: Record + /** Present only when this replay acknowledged Orca's in-flight dispatch. */ + startsTurn?: true + } + | { type: 'provider-frame'; sessionId: string; kind: string; payload: unknown } + | { type: 'prompt'; sessionId: string; prompt: ClaudePendingPrompt } + | { type: 'prompt-cancelled'; sessionId: string; promptKey: string } + | { type: 'options'; sessionId: string; models: unknown[] } + | { + type: 'handle' + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + } + | { type: 'auth-diagnostic'; sessionId: string; diagnostic: ClaudeAuthDiagnostic } + | { + type: 'ended' + sessionId: string + reason: string + /** Present for first-hand child exits so the host can fence recovery. */ + cause?: 'unexpected-exit' | 'requested-close' + fence?: number + acquisitionGeneration?: string + settlementRetryRequired?: boolean + } + +export type ClaudeStructuredSessionAdapterDeps = { + resolveLaunch: (input: { + identity: AgentSessionJournalIdentity + }) => Promise + onEvent?: (event: ClaudeStructuredSessionEvent) => void + openConnection?: typeof openClaudeStreamJsonConnection + readProcessStartTime?: (pid: number) => Promise + mintLinkId?: () => string + mintAcquisitionGeneration?: () => string + now?: () => number + requestTimeoutMs?: number + initTimeoutMs?: number + dispatchAckTimeoutMs?: number + persistHandle?: (input: { + sessionId: string + providerSessionId: string + leafUuid: string | null + fence: number + }) => Promise + /** Read the durable transcript branch after a child has flushed its final rows. */ + readTranscriptLeaf?: (input: { + providerSessionId: string + previousLeafUuid: string | null + /** Account-scoped Claude config root that owns this provider session. */ + claudeConfigDir: string + }) => Promise +} + +export type ClaudeDispatchWaiter = { + resolve: (uuid: string | null) => void + timer: ReturnType + acceptsResult: boolean + /** Client uuid echoed by Claude so a replay is tied to its own dispatch. */ + sentUuid: string + /** Sequence used to fence a late identity from a newer dispatch. */ + dispatchSequence: number + /** Set when the provider replay settled this waiter before send returned. */ + settledUuid?: string + /** The waiter timed out or its write failed, but its replay may still arrive. */ + retired?: boolean + /** Bounded digest/summary for compatibility CLIs that mint UUIDs. */ + replayContentKey: string +} + +export type ClaudeSession = { + connection: ClaudeStreamJsonConnection + providerSessionId: string + /** Durable transcript files live under this account's `projects` directory. */ + claudeConfigDir: string + leafUuid: string | null + fence: number + acquisitionGeneration: string + prompts: ClaudePromptRegistry + dispatchWaiters: ClaudeDispatchWaiter[] + /** Bounded identities for dispatches whose ack was unknown when they returned. */ + retiredDispatchWaiters: ClaudeDispatchWaiter[] + /** Once a retired waiter is evicted, legacy content-only replay matching is unsafe. */ + replayContentFallbackBlocked: boolean + options: Map + reportedOptions: { model?: string; effort?: string } + /** `optionMutationSequence` when `reportedOptions.model` was last observed, so a + * write still awaiting its first turn outranks the report it will replace. */ + reportedModelMutation: number + /** Options whose recorded value the provider reported, not merely accepted. */ + confirmedOptions: Set + restoreSkippedOptions: Set + /** CLI-advertised protocol capabilities from init; gates interrupt-receipt handling. */ + capabilities: readonly string[] + /** Provider uuid of the most recently admitted turn, if one is active. */ + activeTurnId?: string + /** Monotonic fence advanced when a dispatch starts, including unresolved dispatches. */ + dispatchSequence: number + /** Dispatch sequence that admitted activeTurnId. */ + activeTurnSequence?: number + /** Fences overlapping option writes so a late completion cannot restore stale state. */ + optionMutationSequence: number + /** Shared durable-close write; a failed write clears this for a retry. */ + closePersistence?: Promise + /** Shared full close/finalization operation; a failed operation clears this for a retry. */ + closeFinalization?: Promise + /** Set only after the durable close write succeeds, before lifecycle emission. */ + closeFinalized?: boolean + /** Set once `ended` has been emitted, so a persistence retry cannot repeat it. */ + closeEnded?: boolean + translator: ClaudeJournalTranslator | null + events: StructuredAgentSessionEventSink | undefined +} + +export function mintClaudeAcquisitionGeneration(deps: ClaudeStructuredSessionAdapterDeps): string { + return deps.mintAcquisitionGeneration?.() ?? randomUUID() +} + +/** + * The first-hand exit that removed a published session. Kept until the session + * is acquired again so acquisition cleanup that arrives after the exit finds + * what the ladder observed, not an absence it would otherwise report as proven. + */ +export type ClaudeSessionExit = { + connection: ClaudeStreamJsonConnection + /** Full session identity retained until its child tree is proven gone. */ + session: ClaudeSession + error: Error + /** The exit path's first proof attempt; retries must observe this result. */ + closePromise?: Promise + /** Shared lifecycle settlement for concurrent proof retries. */ + settlementPromise?: Promise +} + +export type ClaudeAcquisitionAttempt = { + connection: ClaudeStreamJsonConnection | null + prompts: ClaudePromptRegistry + buffered: (() => void)[] + published: boolean + cancelled: boolean + exitProven: boolean + finished: Promise + finish: () => void +} + +export function createClaudeAcquisitionAttempt( + prompts: ClaudePromptRegistry +): ClaudeAcquisitionAttempt { + let finish = (): void => {} + const finished = new Promise((resolve) => { + finish = resolve + }) + return { + connection: null, + prompts, + buffered: [], + published: false, + cancelled: false, + exitProven: false, + finished, + finish + } +} + +export class ClaudeAcquisitionRegistry { + private readonly attempts = new Map() + private closing = false + + get size(): number { + return this.attempts.size + } + + start( + sessionId: string, + prompts: ClaudePromptRegistry + ): { + previous: ClaudeAcquisitionAttempt | undefined + attempt: ClaudeAcquisitionAttempt + } { + if (this.closing) { + throw new Error('claude structured session adapter is closing') + } + const previous = this.attempts.get(sessionId) + const attempt = createClaudeAcquisitionAttempt(prompts) + this.attempts.set(sessionId, attempt) + return { previous, attempt } + } + + assertCurrent(sessionId: string, attempt: ClaudeAcquisitionAttempt): void { + if (this.closing || attempt.cancelled || this.attempts.get(sessionId) !== attempt) { + throw new Error(`claude session ${sessionId} was superseded while being acquired`) + } + } + + get(sessionId: string): ClaudeAcquisitionAttempt | undefined { + return this.attempts.get(sessionId) + } + + deleteIfCurrent(sessionId: string, attempt: ClaudeAcquisitionAttempt): void { + if (this.attempts.get(sessionId) === attempt) { + this.attempts.delete(sessionId) + } + } + + restoreIfCurrent( + sessionId: string, + replacement: ClaudeAcquisitionAttempt, + previous: ClaudeAcquisitionAttempt + ): void { + if (this.attempts.get(sessionId) === replacement) { + this.attempts.set(sessionId, previous) + } + } + + sessionIds(): IterableIterator { + return this.attempts.keys() + } + + close(): void { + this.closing = true + } +} + +export async function cancelClaudeAcquisitionAttempt( + attempt: ClaudeAcquisitionAttempt | undefined +): Promise { + if (!attempt) { + return true + } + return cancelProcessAcquisition({ + cancel: () => { + attempt.cancelled = true + }, + connection: () => attempt.connection, + exitProven: () => attempt.exitProven, + finished: attempt.finished + }) +} + +/** What an acquisition hands back to the adapter that owns the session map: + * event delivery ordered against publication, and the two exit settlements. */ +export type ClaudeAcquireCallbacks = { + deliver: (attempt: ClaudeAcquisitionAttempt, sessionId: string, event: () => void) => void + emit: ( + session: ClaudeSession | null, + events: StructuredAgentSessionEventSink | undefined, + event: ClaudeStructuredSessionEvent + ) => void + handleExit: (sessionId: string, attempt: ClaudeAcquisitionAttempt, error: Error) => void + settleExit: (sessionId: string, exit: ClaudeSessionExit) => Promise +} diff --git a/src/main/claude/claude-structured-session-test-support.ts b/src/main/claude/claude-structured-session-test-support.ts new file mode 100644 index 00000000000..6b0768b5134 --- /dev/null +++ b/src/main/claude/claude-structured-session-test-support.ts @@ -0,0 +1,260 @@ +import type { + AgentJournalMessageItem, + AgentSessionJournalIdentity +} from '../../shared/agent-session-journal-types' +import type { + ClaudeStreamJsonConnection, + ClaudeStreamJsonConnectionHandlers, + ClaudeStreamJsonLaunch, + openClaudeStreamJsonConnection +} from './claude-stream-json-connection' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps, + type ClaudeStructuredLaunch, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' + +export const PROVIDER_SESSION_ID = '819cf9f8-e43c-4ad7-b50f-54aa158a726a' + +export const USER_MESSAGE: AgentJournalMessageItem = { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'ship it' }] +} + +export function identityFor(sessionId = 'session-1'): AgentSessionJournalIdentity { + return { + sessionId, + workspaceId: 'workspace-1', + hostId: 'host-1', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: PROVIDER_SESSION_ID, leafUuid: null } + } +} + +type Route = (params: Record | undefined) => unknown + +export type FakeConnection = Omit & { + closed: boolean + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + launch: ClaudeStreamJsonLaunch + handlers: ClaudeStreamJsonConnectionHandlers + calls: { subtype: string; params?: Record }[] + sent: Record[] + closeCount: number +} + +export function fakeClaude( + options: { + initSessionId?: string + initUuid?: string + initModel?: string + initProof?: 'init' | 'session-start' | 'none' + initAccount?: unknown + exitBeforeInit?: string + settings?: unknown + replayUuid?: string | null + replayUuids?: (string | null)[] + capabilities?: string[] + unprovenCloseVerdict?: ClaudeStreamJsonConnection['exitVerdict'] + routes?: Record + } = {} +): { + connections: FakeConnection[] + openConnection: typeof openClaudeStreamJsonConnection + routes: Record +} { + const connections: FakeConnection[] = [] + const routes = options.routes ?? {} + let replayIndex = 0 + const routed = (subtype: string, params?: Record): unknown => { + const route = routes[subtype] + return route ? route(params) : undefined + } + const openConnection = (async (launch, handlers = {}) => { + const connection: FakeConnection = { + launch, + handlers, + calls: [], + sent: [], + closeCount: 0, + pid: 4321, + closed: false, + initializationResult: async () => { + connection.calls.push({ subtype: 'initialize' }) + if (options.exitBeforeInit) { + handlers.onExit?.(new Error(options.exitBeforeInit)) + return { models: [] } + } + if (options.initProof === 'session-start') { + handlers.onMessage?.({ + type: 'system', + subtype: 'hook_started', + hook_name: 'SessionStart:startup', + session_id: options.initSessionId ?? PROVIDER_SESSION_ID, + uuid: options.initUuid ?? 'init-uuid' + }) + } else if (options.initProof !== 'none') { + // Keys mirror the real system/init frame, which carries `model` but no + // effort of any kind: the current effort only comes back from + // get_settings. Never add a field the CLI does not send. + handlers.onMessage?.({ + type: 'system', + subtype: 'init', + session_id: options.initSessionId ?? PROVIDER_SESSION_ID, + uuid: options.initUuid ?? 'init-uuid', + model: options.initModel ?? 'claude-sonnet-5', + apiKeySource: 'none', + ...(options.capabilities ? { capabilities: options.capabilities } : {}) + }) + } + return { + models: [{ value: 'claude-sonnet', displayName: 'Sonnet' }], + ...(options.initAccount === undefined ? {} : { account: options.initAccount }) + } + }, + getSettings: async () => { + connection.calls.push({ subtype: 'get_settings' }) + // Shape measured from Claude Code 2.1.258: {applied, effective, sources}, + // and the only place the session's current effort is reported. + return ( + options.settings ?? { + applied: { model: 'claude-sonnet-5', effort: 'high', advisor: null, ultracode: false }, + effective: { model: 'claude-sonnet-5', effortLevel: 'high', env: {} }, + sources: {} + } + ) + }, + supportedModels: async () => { + connection.calls.push({ subtype: 'list_models' }) + return (routed('list_models') as unknown[] | undefined) ?? [] + }, + setModel: async (model) => { + connection.calls.push({ subtype: 'set_model', params: { model } }) + routed('set_model', { model }) + }, + setPermissionMode: async (mode) => { + connection.calls.push({ subtype: 'set_permission_mode', params: { mode } }) + routed('set_permission_mode', { mode }) + }, + applyFlagSettings: async (settings) => { + connection.calls.push({ subtype: 'apply_flag_settings', params: { settings } }) + routed('apply_flag_settings', { settings }) + }, + interrupt: async (interruptOptions) => { + connection.calls.push({ + subtype: 'interrupt', + params: interruptOptions?.cancelQueued ? { cancelQueued: true } : {} + }) + return routed('interrupt', interruptOptions) as + | Awaited> + | undefined + }, + cancelAsyncMessage: async (uuid) => { + connection.calls.push({ subtype: 'cancel_async_message', params: { uuid } }) + routed('cancel_async_message', { uuid }) + }, + send: async (message) => { + connection.sent.push(message) + if (message.type === 'user' && options.replayUuid !== null) { + const configuredReplayUuid = options.replayUuids + ? options.replayUuids[replayIndex++] + : options.replayUuid + const replayUuid = + configuredReplayUuid === undefined ? `user-uuid-${replayIndex}` : configuredReplayUuid + if (replayUuid !== null) { + handlers.onMessage?.({ + ...message, + uuid: replayUuid + }) + } + } + }, + exitVerdict: options.unprovenCloseVerdict ?? { root: 'live', tree: 'unverifiable' }, + close: async () => { + connection.closeCount += 1 + connection.closed = true + return options.unprovenCloseVerdict === undefined + } + } + connections.push(connection) + return connection + }) as typeof openClaudeStreamJsonConnection + return { connections, openConnection, routes } +} + +export function adapterFor( + claude: ReturnType, + launch: Partial = {}, + events: ClaudeStructuredSessionEvent[] = [], + persistedHandles: unknown[] = [], + initTimeoutMs?: number, + readTranscriptLeaf?: ClaudeStructuredSessionAdapterDeps['readTranscriptLeaf'], + persistHandle?: ClaudeStructuredSessionAdapterDeps['persistHandle'] +): ClaudeStructuredSessionAdapter { + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: 'claude', + options: {}, + cwd: '/work/repo', + claudeConfigDir: '/accounts/claude', + providerSessionId: PROVIDER_SESSION_ID, + resumeLeafUuid: null, + resumed: false, + ...launch + }), + onEvent: (event) => events.push(event), + openConnection: claude.openConnection, + readProcessStartTime: async () => 1_700_000_000_000, + now: () => 1_700_000_000_500, + ...(initTimeoutMs === undefined ? {} : { initTimeoutMs }), + dispatchAckTimeoutMs: 10, + persistHandle: + persistHandle ?? + (async (handle) => { + persistedHandles.push(handle) + }), + ...(readTranscriptLeaf ? { readTranscriptLeaf } : {}) + }) +} + +export async function acquired( + claude: ReturnType, + launch: Partial = {}, + events: ClaudeStructuredSessionEvent[] = [] +): Promise { + const adapter = adapterFor(claude, launch, events) + await adapter.acquire({ identity: identityFor(), fence: 7, spawnToken: 'spawn-9' }) + return adapter +} + +export function tick(): Promise { + return new Promise((resolve) => setImmediate(resolve)) +} + +export function invokeCanUseTool( + connection: FakeConnection, + toolName: string, + requestId: string, + toolUseID: string, + extra: { + input?: Record + suggestions?: unknown[] + signal?: AbortSignal + } = {} +): { promise: Promise; settled: () => boolean } { + const options = { + requestId, + toolUseID, + signal: extra.signal ?? new AbortController().signal, + ...(extra.suggestions ? { suggestions: extra.suggestions } : {}) + } as unknown as Parameters>[2] + let done = false + const promise = Promise.resolve( + connection.handlers.canUseTool?.(toolName, extra.input ?? {}, options) + ).finally(() => { + done = true + }) + return { promise, settled: () => done } +} diff --git a/src/main/claude/claude-transcript-branch-proof.ts b/src/main/claude/claude-transcript-branch-proof.ts index d7065caa275..605f619eb92 100644 --- a/src/main/claude/claude-transcript-branch-proof.ts +++ b/src/main/claude/claude-transcript-branch-proof.ts @@ -5,6 +5,10 @@ const MAX_CLAUDE_TRANSCRIPT_ANCESTRY = 10_000 type TranscriptNode = { parentUuid: string | null sessionId: string | null + /** First line where this UUID was observed in the append-only transcript. */ + lineIndex: number + /** UUIDs from result/init/stream frames and sidechains are never leaves. */ + disallowedLeaf: boolean } export type ClaudeTranscriptBranchProof = { @@ -27,6 +31,54 @@ export class ClaudeTranscriptTailIncompleteError extends Error { } } +/** The sampled cursor is no longer present, so a root proof may still recover safely. */ +export class ClaudeTranscriptPreviousCursorMissingError extends Error { + constructor() { + super( + 'Claude transcript branch proof failed: previous cursor is missing from the session graph' + ) + this.name = 'ClaudeTranscriptPreviousCursorMissingError' + } +} + +function proveMainLineAncestry( + nodes: Map, + startUuid: string, + providerSessionId: string +): void { + const visited = new Set() + let cursor: string | null = startUuid + for (let depth = 0; cursor !== null && depth < MAX_CLAUDE_TRANSCRIPT_ANCESTRY; depth += 1) { + if (visited.has(cursor)) { + throw transcriptError('cycle in parentUuid ancestry') + } + visited.add(cursor) + const node = nodes.get(cursor) + if (!node || node.sessionId !== providerSessionId) { + throw transcriptError(`missing ancestor ${cursor}`) + } + if (node.disallowedLeaf) { + throw transcriptError(`ancestor ${cursor} is not on the main transcript`) + } + cursor = node.parentUuid + } + if (cursor !== null) { + throw transcriptError('ancestry exceeds the bounded proof limit') + } +} + +function proveAppendOrder(nodes: Map): void { + for (const node of nodes.values()) { + if (!node.parentUuid) { + continue + } + const parent = nodes.get(node.parentUuid) + if (parent && parent.lineIndex >= node.lineIndex) { + throw transcriptError('parent row follows descendant') + } + } +} + export function proveClaudeTranscriptBranchFromJsonl(input: { contents: string providerSessionId: string @@ -34,6 +86,7 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { }): ClaudeTranscriptBranchProof { const nodes = new Map() let leafUuid: string | null = null + let leafMarkerLineIndex = -1 const lines = input.contents.split('\n') for (const [index, line] of lines.entries()) { if (!line.trim()) { @@ -59,6 +112,7 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { throw transcriptError('invalid last-prompt marker') } leafUuid = markerLeaf + leafMarkerLineIndex = index } const uuid = nonEmptyString(row.uuid) if (!uuid) { @@ -70,27 +124,60 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { } const sessionId = nonEmptyString(row.sessionId) const existing = nodes.get(uuid) - if (existing && (existing.parentUuid !== parentUuid || existing.sessionId !== sessionId)) { + const disallowedLeaf = + row.isSidechain === true || + row.parent_tool_use_id != null || + row.type === 'result' || + row.type === 'stream_event' || + (row.type === 'system' && row.subtype === 'init') + if ( + existing && + (existing.parentUuid !== parentUuid || + existing.sessionId !== sessionId || + existing.disallowedLeaf !== disallowedLeaf) + ) { throw transcriptError(`record ${uuid} has conflicting ancestry`) } - nodes.set(uuid, { parentUuid, sessionId }) + nodes.set(uuid, { + parentUuid, + sessionId, + lineIndex: existing?.lineIndex ?? index, + disallowedLeaf + }) } if (!leafUuid) { throw transcriptError('missing last-prompt marker') } const leaf = nodes.get(leafUuid) - if (!leaf || leaf.sessionId !== input.providerSessionId) { + if (!leaf || leaf.sessionId !== input.providerSessionId || leaf.disallowedLeaf) { throw transcriptError('marker leaf is missing from the session graph') } + if (leaf.lineIndex > leafMarkerLineIndex) { + throw transcriptError('marker precedes its leaf record') + } const previousLeafUuid = input.previousLeafUuid if (!previousLeafUuid) { + proveMainLineAncestry(nodes, leafUuid, input.providerSessionId) + // A branch proof is based on an append-only snapshot. A child that appears + // before its claimed parent is not a post-snapshot descendant observation; + // accepting that graph would turn reordered/torn rows into durable ancestry. + proveAppendOrder(nodes) return { leafUuid, relation: 'initial' } } const previous = nodes.get(previousLeafUuid) - if (!previous || previous.sessionId !== input.providerSessionId) { - throw transcriptError('previous cursor is missing from the session graph') + if (!previous) { + throw new ClaudeTranscriptPreviousCursorMissingError() } + if (previous.sessionId !== input.providerSessionId || previous.disallowedLeaf) { + throw transcriptError('previous cursor is not on the main transcript') + } + // The latest marker can be equal to, or descend from, a sampled cursor. In + // either case prove the sampled cursor's own ancestry before accepting it; + // otherwise a cursor that descended through a parent-tool-use sidechain + // could be persisted and resumed as if it were on the main transcript. + proveMainLineAncestry(nodes, previousLeafUuid, input.providerSessionId) if (leafUuid === previousLeafUuid) { + proveAppendOrder(nodes) return { leafUuid, relation: 'same' } } const visited = new Set() @@ -104,8 +191,12 @@ export function proveClaudeTranscriptBranchFromJsonl(input: { if (!node || node.sessionId !== input.providerSessionId) { throw transcriptError(`missing ancestor ${cursor}`) } + if (node.disallowedLeaf) { + throw transcriptError(`ancestor ${cursor} is not on the main transcript`) + } cursor = node.parentUuid if (cursor === previousLeafUuid) { + proveAppendOrder(nodes) return { leafUuid, relation: 'descendant' } } } @@ -126,3 +217,37 @@ export async function proveClaudeTranscriptBranch(input: { previousLeafUuid: input.previousLeafUuid }) } + +/** Re-run a durable branch proof from the transcript root when a sampled cursor is stale. */ +export async function readClaudeTranscriptLeafWithReproof(input: { + readTranscriptLeaf: (input: { + providerSessionId: string + previousLeafUuid: string | null + claudeConfigDir: string + }) => Promise + claudeConfigDir: string + providerSessionId: string + previousLeafUuid: string | null +}): Promise { + try { + return await input.readTranscriptLeaf({ + providerSessionId: input.providerSessionId, + previousLeafUuid: input.previousLeafUuid, + claudeConfigDir: input.claudeConfigDir + }) + } catch (error) { + // A missing cursor can be stale after compaction and is safe to re-prove from the root. A torn + // tail is still being written; dropping the cursor would make a later sibling look admissible. + if ( + input.previousLeafUuid === null || + !(error instanceof ClaudeTranscriptPreviousCursorMissingError) + ) { + throw error + } + return input.readTranscriptLeaf({ + providerSessionId: input.providerSessionId, + previousLeafUuid: null, + claudeConfigDir: input.claudeConfigDir + }) + } +} diff --git a/src/main/claude/claude-tui-exit.test.ts b/src/main/claude/claude-tui-exit.test.ts new file mode 100644 index 00000000000..6e1140b0f4d --- /dev/null +++ b/src/main/claude/claude-tui-exit.test.ts @@ -0,0 +1,159 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + completeClaudeTuiExit, + readClaudeTranscriptEntryUuid, + readClaudeTranscriptLeafUuid +} from './claude-tui-exit' + +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('Claude TUI exit', () => { + it('does not sample UUIDs from subagent stdout frames with a parent tool use', () => { + expect( + readClaudeTranscriptEntryUuid({ + type: 'assistant', + uuid: 'subagent-assistant', + parent_tool_use_id: 'parent-tool' + }) + ).toBeNull() + expect( + readClaudeTranscriptEntryUuid({ + type: 'assistant', + uuid: 'main-assistant', + parent_tool_use_id: null + }) + ).toBe('main-assistant') + }) + + it('reads the authoritative last-prompt leaf from a transcript tail', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-exit-')) + roots.push(root) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'user', uuid: 'user-one' }, + { type: 'assistant', uuid: 'assistant-one' }, + { type: 'last-prompt', leafUuid: 'chain-head' }, + { type: 'file-history-snapshot', snapshot: {} } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('chain-head') + }) + + it('falls back to the last persisted message when last-prompt metadata is absent', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-exit-')) + roots.push(root) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'user', uuid: 'user-one' }, + { type: 'assistant', uuid: 'assistant-one' }, + { type: 'system', subtype: 'init', uuid: 'init-frame' }, + { type: 'result', uuid: 'result-frame' }, + { type: 'stream_event', uuid: 'stream-event-frame' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('assistant-one') + }) + + it('ignores sidechain messages when selecting a fallback transcript leaf', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-sidechain-leaf-')) + const transcriptPath = join(root, 'session.jsonl') + await writeFile( + transcriptPath, + [ + { type: 'assistant', uuid: 'main-assistant' }, + { type: 'assistant', uuid: 'subagent-assistant', isSidechain: true }, + { type: 'result', uuid: 'result-frame' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcriptPath)).resolves.toBe('main-assistant') + }) + + it('persists the resumed chain head only after the exact Claude child exits', async () => { + let resolveExit!: (exit: { + pid: number + exitCode: number | null + signal: string | null + }) => void + const exitPromise = new Promise<{ + pid: number + exitCode: number | null + signal: string | null + }>((resolve) => { + resolveExit = resolve + }) + const persistHandle = vi.fn(async () => undefined) + const completion = completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: () => exitPromise, + sessionId: 'provider-session', + transcriptPath: '/accounts/claude/session.jsonl', + fence: 7, + persistHandle, + readLeafUuid: async () => 'tui-leaf', + linkId: 'tui-resumed-link', + now: () => 12 + }) + + expect(persistHandle).not.toHaveBeenCalled() + resolveExit({ pid: 4210, exitCode: 0, signal: null }) + + await expect(completion).resolves.toMatchObject({ + link: { + linkId: 'tui-resumed-link', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'tui-leaf' }, + origin: 'resumed', + mintedAtFence: 7, + observedAt: 12 + } + }) + expect(persistHandle).toHaveBeenCalledTimes(1) + }) + + it('refuses another process exit and a missing transcript leaf', async () => { + const persistHandle = vi.fn(async () => undefined) + await expect( + completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: async () => ({ pid: 4211, exitCode: 0, signal: null }), + sessionId: 'provider-session', + transcriptPath: '/session.jsonl', + fence: 2, + persistHandle, + readLeafUuid: async () => 'leaf' + }) + ).rejects.toThrow(/did not belong to the Claude child/) + await expect( + completeClaudeTuiExit({ + childPid: 4210, + waitForChildExit: async () => ({ pid: 4210, exitCode: 1, signal: null }), + sessionId: 'provider-session', + transcriptPath: '/session.jsonl', + fence: 2, + persistHandle, + readLeafUuid: async () => null + }) + ).rejects.toThrow(/resumable transcript leaf/) + expect(persistHandle).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/claude/claude-tui-exit.ts b/src/main/claude/claude-tui-exit.ts new file mode 100644 index 00000000000..3772e9c5e52 --- /dev/null +++ b/src/main/claude/claude-tui-exit.ts @@ -0,0 +1,119 @@ +import { open } from 'node:fs/promises' +import type { AgentSessionProviderHandleLink } from '../../shared/agent-session-provider-handle' +import { claudeProviderHandleLink } from './claude-structured-owner-identity' + +const TRANSCRIPT_TAIL_CHUNK_BYTES = 64 * 1024 +const TRANSCRIPT_TAIL_READ_LIMIT_BYTES = 4 * 1024 * 1024 + +type TranscriptLeafCandidate = { leafUuid: string; authoritative: boolean } + +function validLeafUuid(value: unknown): string | null { + if (typeof value !== 'string' || value.length === 0 || value.length > 512) { + return null + } + const hasControlCharacter = [...value].some((character) => { + const code = character.codePointAt(0) ?? 0 + return code <= 0x1f || code === 0x7f + }) + return value === value.trim() && !hasControlCharacter ? value : null +} + +export function readClaudeTranscriptEntryUuid(value: Record): string | null { + return value.isSidechain === true || + value.parent_tool_use_id != null || + (value.type !== 'user' && value.type !== 'assistant') + ? null + : validLeafUuid(value.uuid) +} + +function readLeafCandidate(line: string): TranscriptLeafCandidate | null { + try { + const value = JSON.parse(line) as Record + const lastPromptLeaf = value.type === 'last-prompt' ? validLeafUuid(value.leafUuid) : null + if (lastPromptLeaf) { + return { leafUuid: lastPromptLeaf, authoritative: true } + } + const messageLeaf = readClaudeTranscriptEntryUuid(value) + return messageLeaf ? { leafUuid: messageLeaf, authoritative: false } : null + } catch { + return null + } +} + +export async function readClaudeTranscriptLeafUuid(transcriptPath: string): Promise { + const file = await open(transcriptPath, 'r') + try { + const { size } = await file.stat() + let position = size + let suffix = '' + let fallback: string | null = null + let scanned = 0 + while (position > 0 && scanned < TRANSCRIPT_TAIL_READ_LIMIT_BYTES) { + const length = Math.min(TRANSCRIPT_TAIL_CHUNK_BYTES, position) + position -= length + scanned += length + const buffer = Buffer.alloc(length) + await file.read(buffer, 0, length, position) + const lines = `${buffer.toString('utf8')}${suffix}`.split(/\r?\n/) + suffix = position > 0 ? (lines.shift() ?? '') : '' + for (let index = lines.length - 1; index >= 0; index -= 1) { + const line = lines[index]?.trim() + if (!line) { + continue + } + const candidate = readLeafCandidate(line) + if (!candidate) { + continue + } + if (candidate.authoritative) { + return candidate.leafUuid + } + fallback ??= candidate.leafUuid + } + } + return fallback + } finally { + await file.close() + } +} + +export type ClaudeTuiChildExit = { + pid: number + exitCode: number | null + signal: string | null +} + +export async function completeClaudeTuiExit(input: { + childPid: number + waitForChildExit: () => Promise + sessionId: string + transcriptPath: string + fence: number + persistHandle: (link: AgentSessionProviderHandleLink) => Promise + readLeafUuid?: (transcriptPath: string) => Promise + linkId?: string + now?: () => number +}): Promise<{ + exit: ClaudeTuiChildExit + transcriptPath: string + link: AgentSessionProviderHandleLink +}> { + const exit = await input.waitForChildExit() + if (exit.pid !== input.childPid) { + throw new Error('The observed process exit did not belong to the Claude child.') + } + const leafUuid = await (input.readLeafUuid ?? readClaudeTranscriptLeafUuid)(input.transcriptPath) + if (!leafUuid) { + throw new Error('The exited Claude TUI did not persist a resumable transcript leaf.') + } + const link = claudeProviderHandleLink({ + sessionId: input.sessionId, + leafUuid, + resumed: true, + fence: input.fence, + ...(input.linkId ? { linkId: input.linkId } : {}), + observedAt: input.now?.() ?? Date.now() + }) + await input.persistHandle(link) + return { exit, transcriptPath: input.transcriptPath, link } +} diff --git a/src/main/claude/claude-tui-resume-launch.test.ts b/src/main/claude/claude-tui-resume-launch.test.ts new file mode 100644 index 00000000000..f907d1fcb4e --- /dev/null +++ b/src/main/claude/claude-tui-resume-launch.test.ts @@ -0,0 +1,227 @@ +import { chmodSync, mkdtempSync, mkdirSync, writeFileSync } from 'node:fs' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { CLAUDE_AUTH_ENV_CONFLICT_MESSAGE } from '../claude-accounts/environment' +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' +import { createClaudeTuiResumeLaunchBuilder } from './claude-tui-resume-launch' + +function record(overrides: Partial = {}): AgentSessionRecord { + return { + sessionId: 'orca-session-1', + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-folder', + workspaceKind: 'folder' + }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/accounts/claude-one' }, + providerHandleChain: [ + { + linkId: 'created', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'leaf-one' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + ...overrides + } as AgentSessionRecord +} + +function makeExecutable(path: string): void { + mkdirSync(join(path, '..'), { recursive: true }) + writeFileSync(path, '') + if (process.platform !== 'win32') { + chmodSync(path, 0o755) + } +} + +describe('Claude TUI resume launch', () => { + it('pins the workspace, account home, setting sources, and launch identity', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async (workspaceId) => `/workspaces/${workspaceId}`, + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ + SELECTED_ACCOUNT: 'one', + ANTHROPIC_AUTH_TOKEN: 'selected-account-token' + }), + inheritedEnv: { + ANTHROPIC_API_KEY: 'inherited-gateway-key', + ANTHROPIC_BASE_URL: 'https://inherited-gateway.invalid', + CLAUDE_CODE_SESSION_ID: 'parent-session', + SAFE_PARENT: 'kept' + } + }) + + const launch = await build({ record: record(), spawnToken: 'spawn-one' }) + + expect(launch).toMatchObject({ + command: '/usr/local/bin/claude', + args: [ + '--setting-sources', + CLAUDE_DEFAULT_SETTING_SOURCES.join(','), + '--resume', + 'provider-session' + ], + cwd: '/workspaces/workspace-folder', + providerSessionId: 'provider-session', + resumeLeafUuid: 'leaf-one' + }) + expect(launch.env).toMatchObject({ + SAFE_PARENT: 'kept', + SELECTED_ACCOUNT: 'one', + CLAUDE_CONFIG_DIR: '/accounts/claude-one', + ORCA_AGENT_LAUNCH_TOKEN: 'spawn-one', + [CLAUDE_SPAWN_TOKEN_ENV]: 'spawn-one', + ANTHROPIC_AUTH_TOKEN: 'selected-account-token' + }) + // System auth (the only state an explicit ANTHROPIC_AUTH_TOKEN overlay is legal in): + // the user's own inherited key is their sign-in and survives. The managed-account + // half — where it is stripped — is covered by 'structured-to-TUI handoff auth'. + expect(launch.env.ANTHROPIC_API_KEY).toBe('inherited-gateway-key') + // Endpoint selection is not credential material; the existing adapter pinning preserves it. + expect(launch.env.ANTHROPIC_BASE_URL).toBe('https://inherited-gateway.invalid') + expect(launch.env.CLAUDE_CODE_SESSION_ID).toBeUndefined() + }) + + it('pairs the resumed Claude CLI with its sibling Node runtime', async () => { + const root = mkdtempSync(join(tmpdir(), 'orca-claude-resume-')) + const binDir = join(root, 'bin') + const claudeCommand = join(binDir, process.platform === 'win32' ? 'claude.cmd' : 'claude') + const nodeCommand = join(binDir, process.platform === 'win32' ? 'node.cmd' : 'node') + makeExecutable(claudeCommand) + makeExecutable(nodeCommand) + + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => claudeCommand, + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + resolveEnv: () => ({ PATH: '/usr/bin' }), + inheritedEnv: {} + }) + + const launch = await build({ record: record(), spawnToken: 'spawn' }) + + expect((launch.env.PATH ?? launch.env.Path)?.split(delimiter)[0]).toBe(binDir) + }) + + it('uses the durable session environment instead of current account settings', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + resolveEnv: () => ({ ANTHROPIC_AUTH_TOKEN: 'pinned-token' }), + inheritedEnv: {} + }) + + const launch = await build({ record: record(), spawnToken: 'spawn' }) + + expect(launch.env.ANTHROPIC_AUTH_TOKEN).toBe('pinned-token') + }) + + it('resolves the durable chain head instead of an earlier Claude leaf', async () => { + const nextRecord = record({ + providerHandleChain: [ + ...record().providerHandleChain, + { + linkId: 'resumed', + handle: { provider: 'claude', sessionId: 'provider-session', leafUuid: 'leaf-two' }, + origin: 'resumed', + mintedAtFence: 2, + observedAt: 2 + } + ] + }) + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + await expect(build({ record: nextRecord, spawnToken: 'spawn-two' })).resolves.toMatchObject({ + providerSessionId: 'provider-session', + resumeLeafUuid: 'leaf-two' + }) + }) + + it('preserves durable Claude launch arguments before resume defaults', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + const launch = await build({ + record: record({ launchArgs: ['--model', 'claude-sonnet-4-5'] }), + spawnToken: 'spawn' + }) + + expect(launch.args.slice(0, 3)).toEqual(['--model', 'claude-sonnet-4-5', '--setting-sources']) + }) + + it('rejects missing Claude handles and unpinned account homes', async () => { + const build = createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/workspace', + resolveCommand: () => 'claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: {} + }) + + await expect( + build({ record: record({ providerHandleChain: [] }), spawnToken: 'spawn' }) + ).rejects.toThrow('claude_tui_resume_handle_required') + await expect( + build({ + record: record({ accountHome: { variable: 'CODEX_HOME', path: '/wrong' } }), + spawnToken: 'spawn' + }) + ).rejects.toThrow(/CLAUDE_CONFIG_DIR/) + }) +}) + +// buildClaudeChildProcessEnv strips its inherited half unconditionally, so this module +// would have signed a system-auth user out of the session the structured path had just +// honoured. It is not wired up yet; the required policy is what stops the next caller +// from inheriting that. +describe('structured-to-TUI handoff auth', () => { + it('carries a system-auth user their own inherited credential', async () => { + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: false }), + inheritedEnv: { ANTHROPIC_API_KEY: 'sk-ant-SHELL', PATH: '/usr/bin' } + })({ record: record(), spawnToken: 'token-1' }) + + expect(launch.env.ANTHROPIC_API_KEY).toBe('sk-ant-SHELL') + }) + + it('still strips it once a managed account owns the credential', async () => { + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + inheritedEnv: { ANTHROPIC_API_KEY: 'sk-ant-SHELL', PATH: '/usr/bin' } + })({ record: record(), spawnToken: 'token-1' }) + + expect(launch.env.ANTHROPIC_API_KEY).toBeUndefined() + }) + + it('refuses a configured override of a pinned managed account, as the terminal path does', async () => { + await expect( + createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => '/repos/workspace-1', + resolveCommand: () => '/usr/local/bin/claude', + resolveAuthPolicy: () => ({ stripAuthEnv: true }), + resolveEnv: () => ({ ANTHROPIC_API_KEY: 'sk-ant-CONFIGURED' }), + inheritedEnv: {} + })({ record: record(), spawnToken: 'token-1' }) + ).rejects.toThrow(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + }) +}) diff --git a/src/main/claude/claude-tui-resume-launch.ts b/src/main/claude/claude-tui-resume-launch.ts new file mode 100644 index 00000000000..8c09335fd9d --- /dev/null +++ b/src/main/claude/claude-tui-resume-launch.ts @@ -0,0 +1,101 @@ +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { agentSessionProviderHandleChainHead } from '../../shared/agent-session-provider-handle' +import { withCliRuntimeOnPath } from '../../shared/node-cli-command-resolution' +import { resolveClaudeCommand } from '../codex-cli/command' +import { getSpawnArgsForWindows } from '../win32-utils' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + claudeAuthEnvCarriedForward, + hasClaudeAuthEnvConflict +} from '../claude-accounts/environment' +import { buildClaudeChildProcessEnv } from './claude-child-process-environment' +import { claudeConfigDirEnvPatch } from './claude-config-dir-pin' +import { CLAUDE_DEFAULT_SETTING_SOURCES } from './claude-structured-launch-resolution' +import { CLAUDE_SPAWN_TOKEN_ENV } from './claude-structured-owner-identity' + +export const CLAUDE_TUI_RESUME_BASE_ARGS = [ + '--setting-sources', + CLAUDE_DEFAULT_SETTING_SOURCES.join(',') +] as const + +export type ClaudeTuiResumeLaunch = { + command: string + args: string[] + cwd: string + env: Record + providerSessionId: string + resumeLeafUuid: string | null +} + +export type ClaudeTuiResumeLaunchBuilderDeps = { + resolveWorkspacePath: (workspaceId: string) => Promise + resolveCommand?: () => string + resolveEnv?: () => Record + inheritedEnv?: NodeJS.ProcessEnv + /** + * Required so whoever wires this module up has to answer the question rather than + * inherit the wrong default: buildClaudeChildProcessEnv strips its inherited half + * unconditionally, which would sign out a system-auth user whose own ANTHROPIC_* + * is their only credential. Build it with claudeStructuredAuthPolicyForSettings. + */ + resolveAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy +} + +export function createClaudeTuiResumeLaunchBuilder( + deps: ClaudeTuiResumeLaunchBuilderDeps +): (input: { record: AgentSessionRecord; spawnToken: string }) => Promise { + return async ({ record, spawnToken }) => { + if (record.provider !== 'claude') { + throw new Error(`session ${record.sessionId} is a ${record.provider} session`) + } + if (record.accountHome.variable !== 'CLAUDE_CONFIG_DIR') { + throw new Error(`claude sessions pin CLAUDE_CONFIG_DIR, not ${record.accountHome.variable}`) + } + const head = agentSessionProviderHandleChainHead(record.providerHandleChain) + if (head?.handle.provider !== 'claude') { + throw new Error('claude_tui_resume_handle_required') + } + + const command = (deps.resolveCommand ?? resolveClaudeCommand)() + const { spawnCmd, spawnArgs } = getSpawnArgsForWindows(command, [ + ...(record.launchArgs ?? []), + ...CLAUDE_TUI_RESUME_BASE_ARGS, + '--resume', + head.handle.sessionId + ]) + const auth = await deps.resolveAuthPolicy() + const configuredEnv = deps.resolveEnv?.() ?? {} + if (auth.stripAuthEnv && hasClaudeAuthEnvConflict(configuredEnv)) { + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) + } + // The inherited half is always stripped downstream, so a system-auth user's own + // credential only reaches the resumed TUI if it is carried in the configured half. + const carriedAuth = auth.stripAuthEnv + ? {} + : claudeAuthEnvCarriedForward(deps.inheritedEnv ?? process.env) + // Compared against what the child would otherwise inherit, so the record's account + // home still wins over a diverging overlay without a needless pin. + const inheritedEnv = { ...(deps.inheritedEnv ?? process.env), ...configuredEnv } + const env = buildClaudeChildProcessEnv( + { + ...carriedAuth, + ...configuredEnv, + ...claudeConfigDirEnvPatch(record.accountHome.path, { env: inheritedEnv }), + ORCA_AGENT_LAUNCH_TOKEN: spawnToken, + [CLAUDE_SPAWN_TOKEN_ENV]: spawnToken + }, + { inheritedEnv: deps.inheritedEnv } + ) + const pairedEnv = withCliRuntimeOnPath(command, env, { platform: process.platform }) + + return { + command: spawnCmd, + args: spawnArgs, + cwd: await deps.resolveWorkspacePath(record.location.workspaceId), + env: pairedEnv, + providerSessionId: head.handle.sessionId, + resumeLeafUuid: head.handle.leafUuid + } + } +} diff --git a/src/main/claude/claude-tui-resume-proof.test.ts b/src/main/claude/claude-tui-resume-proof.test.ts new file mode 100644 index 00000000000..6d2e99d4b00 --- /dev/null +++ b/src/main/claude/claude-tui-resume-proof.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from 'vitest' +import { proveClaudeTuiResume, readClaudeTuiSessionStartEvidence } from './claude-tui-resume-proof' + +const SESSION = '91deba8d-a398-4b69-a05d-35041536fe8e' +const TRANSCRIPT = '/accounts/claude/projects/workspace/transcript.jsonl' + +function envelope(overrides: Record = {}): Record { + return { + launchToken: 'spawn-one', + payload: JSON.stringify({ + hook_event_name: 'SessionStart', + source: 'resume', + session_id: SESSION, + transcript_path: TRANSCRIPT, + ...overrides + }) + } +} + +describe('Claude TUI resume proof', () => { + it('reads SessionStart identity from the hook envelope', () => { + expect(readClaudeTuiSessionStartEvidence(envelope())).toEqual({ + hookEventName: 'SessionStart', + source: 'resume', + sessionId: SESSION, + transcriptPath: TRANSCRIPT, + launchToken: 'spawn-one' + }) + }) + + it('proves the exact launched session and transcript without terminal output', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-one', + waitForSessionStart: async () => envelope() + }) + ).resolves.toMatchObject({ sessionId: SESSION, transcriptPath: TRANSCRIPT }) + }) + + it.each([ + ['source', { source: 'startup' }, /resume SessionStart/], + ['session', { session_id: 'other-session' }, /different Claude session/], + ['transcript', { transcript_path: '/other/transcript.jsonl' }, /different Claude transcript/] + ])('rejects a mismatched %s', async (_name, overrides, expected) => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-one', + waitForSessionStart: async () => envelope(overrides) + }) + ).rejects.toThrow(expected) + }) + + it('rejects a SessionStart from another launched process', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: TRANSCRIPT, + expectedLaunchToken: 'spawn-two', + waitForSessionStart: async () => envelope() + }) + ).rejects.toThrow(/different launched process/) + }) + + it('compares Windows paths using host path semantics', async () => { + await expect( + proveClaudeTuiResume({ + expectedSessionId: SESSION, + expectedTranscriptPath: 'C:\\Users\\Dev\\session.jsonl', + expectedLaunchToken: 'spawn-one', + platform: 'win32', + waitForSessionStart: async () => + envelope({ transcript_path: 'c:\\users\\dev\\session.jsonl' }) + }) + ).resolves.toMatchObject({ sessionId: SESSION }) + }) +}) diff --git a/src/main/claude/claude-tui-resume-proof.ts b/src/main/claude/claude-tui-resume-proof.ts new file mode 100644 index 00000000000..f352423116e --- /dev/null +++ b/src/main/claude/claude-tui-resume-proof.ts @@ -0,0 +1,111 @@ +import { posix, win32 } from 'node:path' + +export type ClaudeTuiSessionStartEvidence = { + hookEventName: 'SessionStart' + source: 'resume' + sessionId: string + transcriptPath: string + launchToken: string +} + +function record(value: unknown): Record | null { + return typeof value === 'object' && value !== null && !Array.isArray(value) + ? (value as Record) + : null +} + +function nonEmptyString(value: unknown): string | null { + return typeof value === 'string' && value.length > 0 ? value : null +} + +function hookPayload(envelope: Record): Record | null { + if (typeof envelope.payload === 'string') { + try { + return record(JSON.parse(envelope.payload)) + } catch { + return null + } + } + return record(envelope.payload) ?? envelope +} + +export function readClaudeTuiSessionStartEvidence( + value: unknown +): ClaudeTuiSessionStartEvidence | null { + const envelope = record(value) + if (!envelope) { + return null + } + const payload = hookPayload(envelope) + if (!payload) { + return null + } + const hookEventName = nonEmptyString(payload.hook_event_name ?? payload.hookEventName) + const source = nonEmptyString(payload.source) + const sessionId = nonEmptyString(payload.session_id ?? payload.sessionId) + const transcriptPath = nonEmptyString(payload.transcript_path ?? payload.transcriptPath) + const launchToken = nonEmptyString(envelope.launchToken ?? payload.launchToken) + return hookEventName === 'SessionStart' && + source === 'resume' && + sessionId && + transcriptPath && + launchToken + ? { hookEventName, source, sessionId, transcriptPath, launchToken } + : null +} + +function comparablePath(value: string, platform: NodeJS.Platform): string | null { + if (value.includes('\0')) { + return null + } + const path = platform === 'win32' ? win32 : posix + if (!path.isAbsolute(value)) { + return null + } + const normalized = path.normalize(value) + return platform === 'win32' ? normalized.toLowerCase() : normalized +} + +export async function proveClaudeTuiResume(input: { + expectedSessionId: string + expectedTranscriptPath: string + expectedLaunchToken: string + waitForSessionStart: () => Promise + timeoutMs?: number + platform?: NodeJS.Platform +}): Promise { + const timeoutMs = input.timeoutMs ?? 15_000 + let timer: ReturnType | undefined + try { + const evidence = readClaudeTuiSessionStartEvidence( + await Promise.race([ + input.waitForSessionStart(), + new Promise((_resolve, reject) => { + timer = setTimeout( + () => reject(new Error('The agent terminal did not prove the expected Claude resume.')), + timeoutMs + ) + timer.unref?.() + }) + ]) + ) + if (!evidence) { + throw new Error('The agent terminal did not emit a Claude resume SessionStart proof.') + } + if (evidence.launchToken !== input.expectedLaunchToken) { + throw new Error('The Claude resume proof came from a different launched process.') + } + if (evidence.sessionId !== input.expectedSessionId) { + throw new Error('The agent terminal resumed a different Claude session.') + } + const platform = input.platform ?? process.platform + const expectedPath = comparablePath(input.expectedTranscriptPath, platform) + const observedPath = comparablePath(evidence.transcriptPath, platform) + if (!expectedPath || !observedPath || observedPath !== expectedPath) { + throw new Error('The agent terminal resumed a different Claude transcript.') + } + return evidence + } finally { + clearTimeout(timer) + } +} diff --git a/src/main/claude/claude-tui-resume-real-binary.integration.test.ts b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts new file mode 100644 index 00000000000..9ba3daf2285 --- /dev/null +++ b/src/main/claude/claude-tui-resume-real-binary.integration.test.ts @@ -0,0 +1,279 @@ +import { spawnSync } from 'node:child_process' +import { randomUUID } from 'node:crypto' +import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises' +import { homedir, tmpdir } from 'node:os' +import { join } from 'node:path' +import * as pty from 'node-pty' +import { afterEach, describe, expect, it } from 'vitest' +import type { AgentSessionJournalIdentity } from '../../shared/agent-session-journal-types' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { resolveClaudeCommand } from '../codex-cli/command' +import { readStructuredTuiProcessIdentity } from '../runtime/structured-tui-process-identity' +import { getSpawnArgsForWindows } from '../win32-utils' +import { CLAUDE_STRUCTURED_BASE_OPTIONS } from './claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionEvent +} from './claude-structured-session-adapter' +import { createClaudeTuiResumeLaunchBuilder } from './claude-tui-resume-launch' +import { proveClaudeTuiResume } from './claude-tui-resume-proof' + +const command = resolveClaudeCommand() +const claudeAvailable = + spawnSync(command, ['--version'], { stdio: 'ignore', timeout: 5_000 }).status === 0 +const authStatusLaunch = getSpawnArgsForWindows(command, ['auth', 'status', '--json']) +const claudeAuthenticated = (() => { + if (!claudeAvailable) { + return false + } + const result = spawnSync(authStatusLaunch.spawnCmd, authStatusLaunch.spawnArgs, { + encoding: 'utf8', + windowsHide: true, + timeout: 5_000 + }) + return result.status === 0 && /"loggedIn"\s*:\s*true/.test(result.stdout) +})() +const roots: string[] = [] +const transcripts: string[] = [] + +function shellQuote(value: string): string { + return process.platform === 'win32' + ? `"${value.replace(/"/g, '""')}"` + : `'${value.replace(/'/g, `'"'"'`)}'` +} + +async function installCaptureHook( + root: string +): Promise<{ eventsPath: string; settingsPath: string }> { + const scriptPath = join(root, 'capture-session-start.cjs') + const eventsPath = join(root, 'session-start.jsonl') + const settingsPath = join(root, 'settings.json') + await writeFile( + scriptPath, + [ + "const { appendFileSync } = require('node:fs')", + "let input = ''", + "process.stdin.setEncoding('utf8')", + "process.stdin.on('data', (chunk) => { input += chunk })", + "process.stdin.on('end', () => {", + ' const payload = JSON.parse(input)', + ' payload.launchToken = process.env.ORCA_AGENT_LAUNCH_TOKEN', + ' appendFileSync(process.argv[2], `${JSON.stringify(payload)}\\n`)', + '})', + '' + ].join('\n') + ) + await writeFile( + settingsPath, + JSON.stringify({ + theme: 'dark', + hooks: { + SessionStart: [ + { + hooks: [ + { + type: 'command', + command: [process.execPath, scriptPath, eventsPath].map(shellQuote).join(' ') + } + ] + } + ] + } + }) + ) + return { eventsPath, settingsPath } +} + +async function waitForHook( + eventsPath: string, + source: 'startup' | 'resume' +): Promise> { + const deadline = Date.now() + 15_000 + while (Date.now() < deadline) { + const contents = await readFile(eventsPath, 'utf8').catch(() => '') + for (const line of contents.split(/\r?\n/)) { + if (!line.trim()) { + continue + } + const event = JSON.parse(line) as Record + if (event.hook_event_name === 'SessionStart' && event.source === source) { + return event + } + } + await new Promise((resolve) => setTimeout(resolve, 50)) + } + throw new Error(`Claude did not emit a ${source} SessionStart hook`) +} + +type RunningTui = { proc: pty.IPty; exited: Promise } + +function spawnResumeTui(args: string[], env: Record): RunningTui { + const direct = process.platform === 'win32' + const proc = pty.spawn( + direct ? command : process.env.SHELL || '/bin/zsh', + direct ? args : ['-l'], + { + name: 'xterm-256color', + cols: 100, + rows: 30, + cwd: process.cwd(), + env: { ...env, TERM: 'xterm-256color' } + } + ) + if (!direct) { + setTimeout(() => { + proc.write(`${[command, ...args].map(shellQuote).join(' ')}\r`) + }, 100).unref() + } + return { proc, exited: new Promise((resolve) => proc.onExit(() => resolve())) } +} + +function structuredIdentity(providerSessionId: string): AgentSessionJournalIdentity { + return { + sessionId: 'orca-real-claude-resume', + workspaceId: 'workspace-real', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: providerSessionId, leafUuid: null } + } +} + +async function waitForStructuredResult(events: ClaudeStructuredSessionEvent[]): Promise { + const deadline = Date.now() + 30_000 + while (Date.now() < deadline) { + if (events.some((event) => event.type === 'message' && event.message.type === 'result')) { + return + } + await new Promise((resolve) => setTimeout(resolve, 50)) + } + throw new Error('Claude structured session did not finish its product-path turn') +} + +async function stopTui(tui: RunningTui): Promise { + try { + tui.proc.kill('SIGKILL') + } catch { + return + } + await Promise.race([ + tui.exited, + new Promise((_resolve, reject) => + setTimeout(() => reject(new Error('Claude TUI did not exit after cleanup')), 5_000) + ) + ]) +} + +afterEach(async () => { + await Promise.all(transcripts.splice(0).map((path) => rm(path, { force: true }))) + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe.skipIf(!claudeAuthenticated)('real Claude TUI resume proof', () => { + it('resumes a product-created structured session and proves its exact child', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-claude-tui-resume-')) + roots.push(root) + const { eventsPath, settingsPath } = await installCaptureHook(root) + const providerSessionId = randomUUID() + const claudeConfigDir = process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude') + const events: ClaudeStructuredSessionEvent[] = [] + const adapter = new ClaudeStructuredSessionAdapter({ + resolveLaunch: async () => ({ + pathToClaudeCodeExecutable: command, + options: { + ...CLAUDE_STRUCTURED_BASE_OPTIONS, + extraArgs: { ...CLAUDE_STRUCTURED_BASE_OPTIONS.extraArgs, settings: settingsPath }, + sessionId: providerSessionId + }, + cwd: process.cwd(), + claudeConfigDir, + providerSessionId, + resumeLeafUuid: null, + resumed: false + }), + onEvent: (event) => events.push(event), + readProcessStartTime: async () => 1 + }) + let resumed: RunningTui | null = null + try { + const acquisition = await adapter.acquire({ + identity: structuredIdentity(providerSessionId), + fence: 1, + spawnToken: 'real-create' + }) + await expect( + adapter.dispatch({ + sessionId: 'orca-real-claude-resume', + clientMessageId: 'real-product-turn', + fence: 1, + body: { + kind: 'message', + role: 'user', + blocks: [{ type: 'text', text: 'Reply only with ORCA_RESUME_READY.' }] + } + }) + ).resolves.toMatchObject({ state: 'accepted' }) + await waitForStructuredResult(events) + const started = await waitForHook(eventsPath, 'startup') + const transcriptPath = String(started.transcript_path) + transcripts.push(transcriptPath) + expect(started.session_id).toBe(providerSessionId) + await adapter.closeAll() + + const record = { + sessionId: 'orca-real-claude-resume', + provider: 'claude', + location: { workspaceId: 'workspace-real' }, + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: claudeConfigDir }, + providerHandleChain: [ + { + linkId: 'created-real', + handle: acquisition.link.handle, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ] + } as AgentSessionRecord + const launch = await createClaudeTuiResumeLaunchBuilder({ + resolveWorkspacePath: async () => process.cwd(), + resolveCommand: () => command, + // The real binary authenticates from the developer's own environment here, + // which is the system-auth case: stripping it would sign the resume out. + resolveAuthPolicy: () => ({ stripAuthEnv: false }) + })({ record, spawnToken: 'real-resume' }) + resumed = spawnResumeTui([...launch.args, '--settings', settingsPath], launch.env) + let resumedOutput = '' + resumed.proc.onData((data) => { + resumedOutput = `${resumedOutput}${data}`.slice(-4_000) + }) + + const [processIdentity, proof] = await Promise.all([ + readStructuredTuiProcessIdentity({ + hostId: 'local', + rootPid: resumed.proc.pid, + spawnToken: 'real-resume', + agent: 'claude' + }), + proveClaudeTuiResume({ + expectedSessionId: providerSessionId, + expectedTranscriptPath: transcriptPath, + expectedLaunchToken: 'real-resume', + waitForSessionStart: () => waitForHook(eventsPath, 'resume') + }).catch((error) => { + throw new Error(`${String(error)}\nClaude output: ${resumedOutput}`) + }) + ]) + expect(processIdentity).toMatchObject({ + hostId: 'local', + spawnToken: 'real-resume', + pid: expect.any(Number) + }) + expect(proof).toMatchObject({ sessionId: providerSessionId, transcriptPath }) + } finally { + await adapter.closeAll() + if (resumed) { + await stopTui(resumed) + } + } + }, 30_000) +}) diff --git a/src/main/codex/codex-structured-session-close.test.ts b/src/main/codex/codex-structured-session-close.test.ts index 4238c75de9e..b04e7bc2540 100644 --- a/src/main/codex/codex-structured-session-close.test.ts +++ b/src/main/codex/codex-structured-session-close.test.ts @@ -11,6 +11,8 @@ import { } from './codex-structured-session-adapter' import { handleCodexSessionExit } from './codex-structured-session-close' import type { CodexSession } from './codex-structured-session-state' +import type { StructuredAgentSessionAdapter } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' const THREAD = 'thread-1' @@ -60,6 +62,16 @@ function adapterFixture() { return { adapter, connections, events } } +function claudeAdapterStub(): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async () => ({ process: { pid: 1 } }) as never), + dispatch: vi.fn(), + cancelTurn: vi.fn(), + answerPrompt: vi.fn(), + setOption: vi.fn() + } +} + describe('Codex structured session close lifecycle', () => { it('forwards a one-shot exit when lifecycle admission is rejected', () => { const connection: CodexAppServerConnection = { @@ -164,4 +176,27 @@ describe('Codex structured session close lifecycle', () => { { cause: 'unexpected-exit', reason: 'sink failed', fence: 7 } ]) }) + + it('routes Codex sink-failure recovery through force-close and preserves unexpected-exit settlement', async () => { + const { adapter, connections, events } = adapterFixture() + const router = new StructuredAgentSessionAdapterRouter( + { claude: claudeAdapterStub(), codex: adapter }, + async () => {} + ) + await router.acquire({ identity: identity('session-1'), fence: 7, spawnToken: 'spawn-1' }) + const current = connections[0] + if (!current) { + throw new Error('missing connection') + } + current.connection.close = async () => { + current.handlers.onExit?.(new Error('journal sink failed')) + return true + } + + const forceCloseSession = router.forceCloseSession + await expect(forceCloseSession('session-1')).resolves.toBe(true) + expect(events.filter((event) => event.type === 'ended')).toMatchObject([ + { cause: 'unexpected-exit', reason: 'journal sink failed', fence: 7 } + ]) + }) }) diff --git a/src/main/ipc/pty/ipc/spawn-env.ts b/src/main/ipc/pty/ipc/spawn-env.ts index af5da3858bd..94f1acf363e 100644 --- a/src/main/ipc/pty/ipc/spawn-env.ts +++ b/src/main/ipc/pty/ipc/spawn-env.ts @@ -7,7 +7,11 @@ import { isRemoteAgentHooksEnabled } from '../../../../shared/agent-hook-relay' import { isOpaqueRemintedPaneKey } from '../../../../shared/pane-key-alias' import { isValidTerminalTabId } from '../../../../shared/terminal-tab-id' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' -import { hasClaudeAuthEnvConflict } from '../../../claude-accounts/environment' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + hasClaudeAuthEnvConflict +} from '../../../claude-accounts/environment' import { LocalPtyProvider } from '../../../providers/local-pty-provider' import { resolvePathEnvKey } from '../../../pty/windows-environment-path' import { routesFreshSpawnsToLocalProvider } from '../host-env/fresh-spawn-routing' @@ -20,12 +24,10 @@ import { assemblePtyIpcSpawnCodexEnv } from './spawn-env-codex' export async function assemblePtyIpcSpawnEnv(ctx: PtyIpcSpawnState): Promise { const args = ctx.args if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } if (ctx.claudeAuth?.stripAuthEnv && hasClaudeAuthEnvConflict(args.env)) { - throw new Error( - 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' - ) + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) } // Why: the daemon-backed provider skips LocalPtyProvider's buildSpawnEnv, so assemble the same host-local env here for parity. // Safety: skip entirely for SSH — every injection is a loopback secret or a local path that leaks or misleads on the remote host. diff --git a/src/main/ipc/pty/ipc/spawn-preflight.ts b/src/main/ipc/pty/ipc/spawn-preflight.ts index f40b46e5dd5..f885d6e2f6f 100644 --- a/src/main/ipc/pty/ipc/spawn-preflight.ts +++ b/src/main/ipc/pty/ipc/spawn-preflight.ts @@ -4,6 +4,7 @@ import { } from '../../../../shared/local-windows-terminal-runtime' import { isWslUncPath, toWindowsWslPath } from '../../../../shared/wsl-paths' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' +import { CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE } from '../../../claude-accounts/environment' import { mintPtySessionId } from '../../../daemon/pty-session-id' import { resolveWslSessionContext } from '../../../daemon/wsl-session-context' import { LocalPtyProvider } from '../../../providers/local-pty-provider' @@ -193,7 +194,7 @@ export async function preparePtyIpcSpawnPreflight(ctx: PtyIpcSpawnState): Promis ctx.isClaudeLaunch = !ctx.preAdoptedStablePane && !args.connectionId && isClaudeLaunchCommand(args.command) if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } ctx.terminalRuntimeOptions = process.platform === 'win32' && !args.connectionId diff --git a/src/main/ipc/pty/runtime/spawn-preflight.ts b/src/main/ipc/pty/runtime/spawn-preflight.ts index aed89b44df8..43fd2778119 100644 --- a/src/main/ipc/pty/runtime/spawn-preflight.ts +++ b/src/main/ipc/pty/runtime/spawn-preflight.ts @@ -23,7 +23,11 @@ import { import { stripRemotePaneEnvWhenHooksDisabled } from '../provider/liveness' import { isTuiAgent } from '../../../../shared/tui-agent-config' import { isClaudeAuthSwitchInProgress } from '../../../claude-accounts/live-pty-gate' -import { hasClaudeAuthEnvConflict } from '../../../claude-accounts/environment' +import { + CLAUDE_AUTH_ENV_CONFLICT_MESSAGE, + CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE, + hasClaudeAuthEnvConflict +} from '../../../claude-accounts/environment' import { isSafePtySessionId, mintPtySessionId, @@ -65,7 +69,7 @@ export async function prepareRuntimePtySpawn( ctx.isClaudeLaunch = !ctx.preAdoptedStablePane && !args.connectionId && isClaudeLaunchCommand(args.command) if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } // Why: runtime-created terminals carry no renderer-computed projectRuntime; resolve from worktreeId to honor the project's Windows runtime. ctx.terminalRuntimeOptions = @@ -134,12 +138,10 @@ export async function prepareRuntimePtySpawn( ? await ctx.deps.prepareClaudeAuth(ctx.codexSelectionTarget) : null if (ctx.isClaudeLaunch && isClaudeAuthSwitchInProgress()) { - throw new Error('A Claude account switch is in progress. Try again after it finishes.') + throw new Error(CLAUDE_AUTH_SWITCH_IN_PROGRESS_MESSAGE) } if (ctx.claudeAuth?.stripAuthEnv && hasClaudeAuthEnvConflict(args.env)) { - throw new Error( - 'This Claude launch defines explicit Anthropic auth environment variables. Remove those overrides before using a managed Claude account.' - ) + throw new Error(CLAUDE_AUTH_ENV_CONFLICT_MESSAGE) } ctx.shouldPersistHostSessionBinding = args.persistHostSessionBinding === true diff --git a/src/main/ipc/runtime.test.ts b/src/main/ipc/runtime.test.ts index dc7f8a01cd7..07010087363 100644 --- a/src/main/ipc/runtime.test.ts +++ b/src/main/ipc/runtime.test.ts @@ -136,6 +136,40 @@ describe('registerRuntimeHandlers', () => { }) }) + it('projects Claude structured tabs to the same-version desktop client', async () => { + const claudeTab = { + type: 'agent-session', + id: 'agent-session:claude-1', + title: 'Claude Chat', + sessionId: 'claude-1', + agent: 'claude', + isActive: true + } + const runtime = { + getRuntimeId: vi.fn().mockReturnValue('runtime-1'), + restoreStructuredAgentSessionTabs: vi.fn(async () => undefined), + listMobileSessionTabs: vi.fn(async () => ({ + worktree: 'workspace-1', + publicationEpoch: 'epoch-1', + snapshotVersion: 1, + activeGroupId: 'group-1', + activeTabId: claudeTab.id, + activeTabType: 'agent-session', + tabGroups: [{ id: 'group-1', activeTabId: claudeTab.id, tabOrder: [claudeTab.id] }], + tabs: [claudeTab] + })) + } + + registerRuntimeHandlers(runtime as never) + const callRegistration = handleMock.mock.calls.find(([channel]) => channel === 'runtime:call') + const result = await callRegistration![1](runtimeCallEvent(), { + method: 'session.tabs.list', + params: { worktree: 'id:workspace-1' } + }) + + expect(result).toMatchObject({ ok: true, result: { tabs: [claudeTab] } }) + }) + it('registers project group runtime RPC methods for local desktop callers', async () => { const runtime = { syncWindowGraph: vi.fn(), diff --git a/src/main/ipc/runtime.ts b/src/main/ipc/runtime.ts index 901d14bfce6..3901d8b1ffa 100644 --- a/src/main/ipc/runtime.ts +++ b/src/main/ipc/runtime.ts @@ -10,7 +10,10 @@ import type { import type { RuntimeRpcResponse } from '../../shared/runtime-rpc-envelope' import type { ClientHostedBrowserRowsEvent } from '../../shared/client-hosted-browser-rows' import { TERMINAL_FIT_RESTORE_DEADLINE_MS } from '../../shared/terminal-fit-restore-deadline' -import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' +import { + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY +} from '../../shared/protocol-version' import { RpcDispatcher } from '../runtime/rpc/dispatcher' import { ALL_RPC_METHODS } from '../runtime/rpc/methods' import { DesktopRuntimeSenderLifecycle } from './desktop-runtime-sender-lifecycle' @@ -76,7 +79,10 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientId: 'desktop-renderer', clientKind: 'runtime', connectionId: desktopSenders.connectionIdFor(event.sender), - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] } )) as RuntimeRpcResponse } @@ -121,7 +127,10 @@ export function registerRuntimeHandlers(runtime: OrcaRuntimeService): void { clientId: 'desktop-renderer', clientKind: 'runtime', connectionId, - clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] + clientCapabilities: [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] } ) .finally(stop) diff --git a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts index 4f3ef118af5..4f21285e417 100644 --- a/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts +++ b/src/main/native-chat/agent-session-wire/claude-stream-json-frame-schema.ts @@ -1,4 +1,4 @@ -// SDKMessage discriminators from Claude Agent SDK 0.3.231 / Claude Code 2.1.231. +// SDKMessage discriminators from Claude Agent SDK 0.3.251 / Claude Code 2.1.258. export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:assistant', 'message:user', @@ -42,7 +42,15 @@ export const CLAUDE_STREAM_JSON_FRAME_KINDS = [ 'message:prompt_suggestion', 'message:system:mirror_error', 'message:system:informational', - 'message:conversation_reset' + 'message:conversation_reset', + // Queue bookkeeping the CLI emits per client-supplied command uuid. Absent + // from the SDK's SDKMessage union, which is why it reached users as raw JSON. + 'message:command_lifecycle', + 'message:result:success', + 'message:result:error_during_execution', + 'message:result:error_max_turns', + 'message:result:error_max_budget_usd', + 'message:result:error_max_structured_output_retries' ] as const export type ClaudeStreamJsonFrameKind = (typeof CLAUDE_STREAM_JSON_FRAME_KINDS)[number] diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts index ad9ca66c52a..22bd645d8a6 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.test.ts @@ -65,6 +65,29 @@ describe('provider frame classification catalog', () => { ).toBe('error-surface') }) + it('keeps command queue bookkeeping off the transcript without hiding a failed one', () => { + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'started' + }) + ).toBe('status-chrome') + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'cancelled' + }) + ).toBe('status-chrome') + // Payload inspection outranks the catalogue, so suppressing the kind cannot + // swallow a state the provider reports as a failure. + expect( + classifyProviderFrame('claude', 'message:command_lifecycle', { + command_uuid: 'command-1', + state: 'failed' + }) + ).toBe('error-surface') + }) + it('keeps unknown future frames on the substantive bounded fallback path', () => { expect(classifyProviderFrame('codex', 'notification:future/event', {})).toBe( 'timeline-substantive' diff --git a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts index 8d11df995a6..474b1385a4f 100644 --- a/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts +++ b/src/main/native-chat/agent-session-wire/provider-frame-disposition.ts @@ -131,7 +131,18 @@ export const PROVIDER_FRAME_CLASSIFICATIONS = { 'message:prompt_suggestion': 'status-chrome', 'message:system:mirror_error': 'error-surface', 'message:system:informational': 'timeline-substantive', - 'message:conversation_reset': 'status-chrome' + 'message:conversation_reset': 'status-chrome', + // A `started`/`completed`/`cancelled` state for one queued command uuid and + // nothing else; the CLI keeps it out of its own transcript too. A state that + // reads as a failure still surfaces, via the payload check in classify. + 'message:command_lifecycle': 'status-chrome', + // The turn-complete signal: lifecycle, never a transcript row. Error subtypes + // included — the turn's assistant frames already carry any user-facing text. + 'message:result:success': 'status-chrome', + 'message:result:error_during_execution': 'status-chrome', + 'message:result:error_max_turns': 'status-chrome', + 'message:result:error_max_budget_usd': 'status-chrome', + 'message:result:error_max_structured_output_retries': 'status-chrome' } } as const satisfies ProviderFrameClassificationTable diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts new file mode 100644 index 00000000000..c6566083eac --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionAdapterRouter } from './structured-agent-session-adapter-router' + +function adapterOf( + releaseAcquisition: StructuredAgentSessionAdapter['releaseAcquisition'] +): StructuredAgentSessionAdapter { + return { + acquire: vi.fn(async () => ({ process: { pid: 1 } }) as never), + releaseAcquisition, + dispatch: vi.fn(), + cancelTurn: vi.fn(), + answerPrompt: vi.fn(), + setOption: vi.fn() + } as unknown as StructuredAgentSessionAdapter +} + +describe('StructuredAgentSessionAdapterRouter.releaseAcquisition', () => { + it('drops the owner even when its release reports a typed failure', async () => { + const failure = new Error('root exited') + const claude = adapterOf(vi.fn().mockRejectedValueOnce(failure).mockResolvedValue(false)) + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + + await expect(router.releaseAcquisition({ sessionId: 'session-1' })).rejects.toBe(failure) + // With no owner left, a later release asks every adapter instead of the stale one. + await expect(router.releaseAcquisition({ sessionId: 'session-1' })).resolves.toBe(false) + expect(claude.releaseAcquisition).toHaveBeenCalledTimes(2) + expect(codex.releaseAcquisition).toHaveBeenCalledTimes(1) + }) +}) + +describe('StructuredAgentSessionAdapterRouter.closeSession', () => { + it('retains the owner after an unproven close so a later retry reaches the same adapter', async () => { + const claude = adapterOf(vi.fn(async () => true)) + const closeSession = vi.fn().mockResolvedValueOnce(false).mockResolvedValueOnce(true) + const dispatch = vi.fn().mockResolvedValue({ state: 'unknown', reason: 'test' }) + claude.closeSession = closeSession + claude.dispatch = dispatch + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + + await expect(router.closeSession('session-1')).resolves.toBe(false) + await expect( + router.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: {} as never, + fence: 1 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + await expect(router.closeSession('session-1')).resolves.toBe(true) + expect(closeSession).toHaveBeenCalledTimes(2) + expect(dispatch).toHaveBeenCalledTimes(1) + }) +}) + +describe('StructuredAgentSessionAdapterRouter optional lifecycle methods', () => { + it.each([ + ['forceCloseSession', 'forceCloseSession'], + ['disposeSession', 'disposeSession'] + ] as const)( + '%s forwards to the owner and retains it until proven stopped', + async (_label, method) => { + const claude = adapterOf(vi.fn(async () => true)) + const stop = vi.fn().mockResolvedValueOnce(false).mockResolvedValueOnce(true) + claude[method] = stop + const dispatch = vi.fn().mockResolvedValue({ state: 'unknown', reason: 'test' }) + claude.dispatch = dispatch + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + const identity = { sessionId: 'session-1', agent: 'claude' } as never + await router.acquire({ identity, fence: 1, spawnToken: 'spawn-1' }) + const stopSession = router[method] + + await expect(stopSession('session-1')).resolves.toBe(false) + await expect( + router.dispatch({ + sessionId: 'session-1', + clientMessageId: 'client-1', + body: {} as never, + fence: 1 + }) + ).resolves.toMatchObject({ state: 'unknown' }) + await expect(stopSession('session-1')).resolves.toBe(true) + expect(stop).toHaveBeenCalledTimes(2) + expect(dispatch).toHaveBeenCalledOnce() + } + ) + + it.each(['forceCloseSession', 'disposeSession'] as const)( + 'falls back to closeSession when an owner lacks %s', + async (method) => { + const closeSession = vi.fn().mockResolvedValue(true) + const claude = adapterOf(vi.fn(async () => true)) + claude.closeSession = closeSession + const codex = adapterOf(vi.fn(async () => false)) + const router = new StructuredAgentSessionAdapterRouter({ claude, codex }, async () => {}) + await router.acquire({ + identity: { sessionId: 'session-1', agent: 'claude' } as never, + fence: 1, + spawnToken: 'spawn-1' + }) + const stopSession = router[method] + + await expect(stopSession('session-1')).resolves.toBe(true) + expect(closeSession).toHaveBeenCalledWith('session-1') + } + ) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts new file mode 100644 index 00000000000..6ac0c8e0fbf --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter-router.ts @@ -0,0 +1,124 @@ +import type { AgentSessionJournalIdentity } from '../../../shared/agent-session-journal-types' +import type { AgentSessionExecutionLocation } from '../../../shared/agent-session-record' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' + +type RoutedAgent = 'claude' | 'codex' + +export class StructuredAgentSessionAdapterRouter implements StructuredAgentSessionAdapter { + private readonly owners = new Map() + + constructor( + private readonly adapters: Record, + private readonly closeAdapters: () => Promise + ) {} + + supportsCreate = (location: AgentSessionExecutionLocation, agent: string): boolean => { + const adapter = this.adapterForAgent(agent) + return adapter ? (adapter.supportsLocation?.(location) ?? false) : false + } + + supportsLocation = (location: AgentSessionExecutionLocation): boolean => + Object.values(this.adapters).some((adapter) => adapter.supportsLocation?.(location) ?? false) + + async acquire(input: Parameters[0]) { + const adapter = this.requireAgent(input.identity) + const acquired = await adapter.acquire(input) + this.owners.set(input.identity.sessionId, adapter) + return acquired + } + + async releaseAcquisition(input: { sessionId: string }): Promise { + const adapter = this.owners.get(input.sessionId) + if (adapter) { + try { + return (await adapter.releaseAcquisition?.(input)) === true + } finally { + this.owners.delete(input.sessionId) + } + } + let released = false + for (const candidate of Object.values(this.adapters)) { + released = (await candidate.releaseAcquisition?.(input)) === true || released + } + return released + } + + dispatch: StructuredAgentSessionAdapter['dispatch'] = (input) => + this.owner(input.sessionId).dispatch(input) + + cancelTurn: StructuredAgentSessionAdapter['cancelTurn'] = (input) => + this.owner(input.sessionId).cancelTurn(input) + + answerPrompt: StructuredAgentSessionAdapter['answerPrompt'] = (input) => + this.owner(input.sessionId).answerPrompt(input) + + setOption: StructuredAgentSessionAdapter['setOption'] = (input) => + this.owner(input.sessionId).setOption(input) + + readOptions = (input: { sessionId: string; fence: number }) => { + const reader = this.owner(input.sessionId).readOptions + if (!reader) { + throw new Error(`structured session ${input.sessionId} does not report options`) + } + return reader(input) + } + + readOptionRestoreFailures = (sessionId: string): readonly string[] => + this.owner(sessionId).readOptionRestoreFailures?.(sessionId) ?? [] + + historyFilePath = (input: { identity: AgentSessionJournalIdentity }) => + this.requireAgent(input.identity).historyFilePath?.(input) ?? Promise.resolve(null) + + closeSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.closeSession) + + forceCloseSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.forceCloseSession ?? adapter.closeSession) + + disposeSession = (sessionId: string): Promise => + this.stopSession(sessionId, (adapter) => adapter.disposeSession ?? adapter.closeSession) + + private async stopSession( + sessionId: string, + selectStop: ( + adapter: StructuredAgentSessionAdapter + ) => NonNullable | undefined + ): Promise { + const adapter = this.owners.get(sessionId) + if (!adapter) { + return false + } + const stop = selectStop(adapter) + const stopped = await stop?.call(adapter, sessionId) + if (stopped === true) { + this.owners.delete(sessionId) + return true + } + return false + } + + async closeAll(): Promise { + this.owners.clear() + await this.closeAdapters() + } + + private owner(sessionId: string): StructuredAgentSessionAdapter { + const adapter = this.owners.get(sessionId) + if (!adapter) { + throw new Error(`no live structured adapter owns ${sessionId}`) + } + return adapter + } + + private requireAgent(identity: AgentSessionJournalIdentity): StructuredAgentSessionAdapter { + const adapter = this.adapterForAgent(identity.agent) + if (!adapter) { + throw new Error(`structured sessions do not support ${identity.agent}`) + } + return adapter + } + + private adapterForAgent(agent: string): StructuredAgentSessionAdapter | null { + return agent === 'claude' || agent === 'codex' ? this.adapters[agent] : null + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts index 5f67240c1c6..77cce9153f5 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it, vi } from 'vitest' import { AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, rethrowAfterAgentSessionAcquisitionCleanup } from './structured-agent-session-adapter' @@ -28,6 +29,27 @@ describe('failed agent-session acquisition cleanup', () => { ).rejects.toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) }) + it('keeps a first-hand root exit that cleanup observed, with the provider diagnostic', async () => { + const cause = new Error('proof failed') + const exit = new AgentSessionAcquisitionRootExitObservedError( + new Error('claude stream-json exited (code 1): crashed') + ) + const error = await rethrowAfterAgentSessionAcquisitionCleanup( + { + releaseAcquisition: vi.fn(async () => { + throw exit + }) + }, + 'session-1', + cause + ).catch((thrown: unknown) => thrown) + + expect(error).toBeInstanceOf(AgentSessionAcquisitionRootExitObservedError) + expect(error).not.toBeInstanceOf(AgentSessionAcquisitionExitUnprovenError) + expect((error as Error).message).toBe('claude stream-json exited (code 1): crashed') + expect((error as Error).cause).toMatchObject({ errors: [cause, exit] }) + }) + it('reports unproven exit when cleanup throws', async () => { const error = await rethrowAfterAgentSessionAcquisitionCleanup( { diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts index cccc8ce6f13..01c16a60e55 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-adapter.ts @@ -32,6 +32,20 @@ export class AgentSessionAcquisitionRefusal extends Error { } } +/** + * The provider's own root process was observed to exit, but its descendant tree + * could not be verified. The lease keys on the root's pid and start time, so its + * observed death releases the reservation; nothing is claimed about descendants. + * Never thrown when a descendant was observed still alive — that stays unproven. + */ +export class AgentSessionAcquisitionRootExitObservedError extends Error { + constructor(cause: unknown) { + // The provider's own diagnostic is the only thing the user can act on. + super(cause instanceof Error ? cause.message : String(cause), { cause }) + this.name = 'AgentSessionAcquisitionRootExitObservedError' + } +} + export class AgentSessionAcquisitionExitUnprovenError extends Error { constructor(cause: unknown) { super('agent_session_acquisition_exit_unproven', { cause }) @@ -49,7 +63,7 @@ export type AgentSessionAcquisition = { acquisitionGeneration?: string } -/** Acquisition validation failed before the adapter attempted to spawn. */ +/** Acquisition failed with first-hand proof that no provider process existed. */ export class AgentSessionPreSpawnError extends Error { constructor(cause: unknown) { super(cause instanceof Error ? cause.message : String(cause), { cause }) @@ -105,7 +119,9 @@ export type StructuredAgentSessionAdapter = { * at — the store rejects a link minted at any other fence. */ acquire(input: StructuredAgentSessionAcquireInput): Promise /** Reaps an acquired provider when the host cannot commit or prove its lease. - * Returns true only after provider child exit is proven. */ + * Returns true only after provider child exit is proven. Throws + * `AgentSessionAcquisitionRootExitObservedError` when the provider root's own + * exit was observed first-hand but its descendants could not be verified. */ releaseAcquisition?(input: { sessionId: string }): Promise dispatch(input: { sessionId: string @@ -133,6 +149,8 @@ export type StructuredAgentSessionAdapter = { input: StructuredAgentSessionSetOptionInput ): Promise>> readOptions?(input: { sessionId: string; fence: number }): Promise + /** Option keys skipped after a provider rejected their persisted restore value. */ + readOptionRestoreFailures?(sessionId: string): readonly string[] /** Transcript path for journal recovery. Omit to let the existing session-file * resolver discover it from the provider session id. */ historyFilePath?(input: { identity: AgentSessionJournalIdentity }): Promise @@ -154,9 +172,15 @@ export async function rethrowAfterAgentSessionAcquisitionCleanup( try { released = (await adapter.releaseAcquisition?.({ sessionId })) === true } catch (cleanupError) { - throw new AgentSessionAcquisitionExitUnprovenError( - new AggregateError([cause, cleanupError], 'agent session acquisition cleanup failed') - ) + // A root exit the cleanup observed first-hand keeps its classification and its + // provider diagnostic; the failure that triggered cleanup rides along as cause. + throw cleanupError instanceof AgentSessionAcquisitionRootExitObservedError + ? new AgentSessionAcquisitionRootExitObservedError( + new AggregateError([cause, cleanupError], cleanupError.message) + ) + : new AgentSessionAcquisitionExitUnprovenError( + new AggregateError([cause, cleanupError], 'agent session acquisition cleanup failed') + ) } if (released) { throw cause diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts index 05c3d8c9e5e..7113be8d54b 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-context.ts @@ -5,7 +5,6 @@ import type { AgentSessionWireRefusal } from '../../../shared/agent-session-wire' import type { AgentJournalResetReason } from '../../../shared/agent-session-journal-types' -import type { AgentSessionAttachParams } from './structured-agent-session-attach' import type { AgentSessionJournal } from '../agent-session-journal/journal-store' import type { StructuredAgentSessionHostDeps, @@ -30,8 +29,6 @@ export type StructuredAgentSessionAttachContext = { } tasks: StructuredAgentSessionTaskQueue reconcileLeases: (sessionId: string) => Promise - /** Retries a durable provider-exit journal settlement before a new owner is reserved. */ - retryPendingSettlement?: (sessionId: string, params: AgentSessionAttachParams) => Promise serialize: (sessionId: string, task: () => Promise) => Promise now: () => number } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts index a21edcee35c..b08a56ea4d9 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-flow.ts @@ -26,6 +26,7 @@ import type { AgentSessionRecordStore } from '../../runtime/agent-session-record import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { AgentSessionAcquisitionExitUnprovenError, + AgentSessionAcquisitionRootExitObservedError, AgentSessionAcquisitionRefusal, AgentSessionPreSpawnError, isAgentSessionPreSpawnError, @@ -119,7 +120,9 @@ export async function performAttach( ? 'processless' : error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' - : 'exit-proven' + : error instanceof AgentSessionAcquisitionRootExitObservedError + ? 'root-exit-observed' + : 'exit-proven' const outcome = error instanceof AgentSessionAcquisitionExitUnprovenError ? { @@ -209,13 +212,17 @@ async function settlePostAcquisitionAttachFailure( cause: unknown ): Promise { let cleanupError: unknown = cause - let exitProof: 'exit-proven' | 'unproven' = 'unproven' + let exitProof: 'exit-proven' | 'root-exit-observed' | 'unproven' = 'unproven' try { await rethrowAfterAgentSessionAcquisitionCleanup(input.adapter, record.sessionId, cause) } catch (error) { cleanupError = error exitProof = - error instanceof AgentSessionAcquisitionExitUnprovenError ? 'unproven' : 'exit-proven' + error instanceof AgentSessionAcquisitionExitUnprovenError + ? 'unproven' + : error instanceof AgentSessionAcquisitionRootExitObservedError + ? 'root-exit-observed' + : 'exit-proven' } // Why: the close is awaited so the map entry is gone only once its handle is // released, but a failed close must not also cost the store settlement below. diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts index f4551ef9313..a22bbdcbb3e 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach-orchestration.ts @@ -17,6 +17,7 @@ import { pinnedAgentSessionLaunchEnv } from './structured-agent-session-launch-env' import { refuseAgentSessionMutation } from './structured-agent-session-mutation-admission' +import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import type { StructuredAgentSessionAttachContext } from './structured-agent-session-attach-context' import type { DeferredStructuredAgentSessionEventSink } from './structured-agent-session-event-sink' import { agentSessionJournalCloseRetries } from '../agent-session-journal/journal-close-retry' @@ -41,14 +42,20 @@ export function attachStructuredAgentSession( return refuseAgentSessionMutation(unreconciled) } await context.runtimeState.resolveRecovery(sessionId) - if (context.retryPendingSettlement) { - const settled = await context.retryPendingSettlement(sessionId, params) - if (!settled) { - return refuseAgentSessionMutation({ - code: 'agent_session_ownership_unknown', - message: 'The provider-exit terminal journal settlement is still pending; retry attach.' - }) - } + // Retries a durable provider-exit journal settlement before a new owner is reserved. Answers + // settled when the record has none pending, so every attach can ask unconditionally. + const settled = await retryPendingStructuredAgentSessionSettlement({ + deps: context.deps, + sessions: context.sessions, + sessionId, + params, + now: () => context.now() + }) + if (!settled) { + return refuseAgentSessionMutation({ + code: 'agent_session_ownership_unknown', + message: 'The provider-exit terminal journal settlement is still pending; retry attach.' + }) } const eventSink = context.runtimeState.eventSinkFor(sessionId) const attached = await performAttach({ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts index 83766f25fc5..ce58e31b4ee 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-attach.ts @@ -5,7 +5,6 @@ // the record store's compare-and-swap, which also owns the idempotency row, so // a retried attach replays instead of reserving a second owner. -import type { AgentType } from '../../../shared/agent-status-types' import type { AgentSessionJournalIdentity, AgentSessionProviderHandle @@ -52,7 +51,7 @@ export type AgentSessionAttachParams = { envelope: AgentSessionMutationEnvelope location: AgentSessionExecutionLocation provider: AgentSessionHandleProvider - agent: AgentType + agent: AgentSessionHandleProvider accountHome: AgentSessionAccountHome runtimeKind: AgentSessionOwnerRuntimeKind /** Omitted only for create-by-intent; the adapter proves the durable handle. */ diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts new file mode 100644 index 00000000000..87bc33bc4b9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-claude-options-round-trip.test.ts @@ -0,0 +1,182 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffRequest, + AgentSessionMutationEnvelope +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-claude' } +const CLAUDE_SESSION = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' +const DEFAULT_MODEL = 'sonnet' +const PICKED_MODEL = 'opus' + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let activeModel: string +let transcriptPath: string + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function handoff(direction: AgentSessionHandoffDirection): AgentSessionHandoffRequest { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { envelope: envelope('agentSession.requestHandoff', fields), ...fields } +} + +function owner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { + handle: 'term-claude', + tabId: 'tab-claude', + paneKey: 'pane-claude', + ptyId: 'pty-claude' + }, + process: { hostId: 'local', pid: 5200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `claude-tui-${fence}`, + handle: { provider: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'tui-leaf' }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + transcriptPath + } +} + +function transport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ fence, spawnToken }) => owner(fence, spawnToken), + reproveTuiOwner: async ({ owner: current }) => current, + recoverTuiOwner: async (record) => + owner( + record.lease.runtimeFence, + record.lease.ownerProcess?.spawnToken ?? record.lease.reservedSpawnToken ?? 'recovered' + ), + stopRecoveredOwner: async () => undefined, + waitForTuiExit: async (current) => ({ transcriptPath: current.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + acquire = vi.fn(async ({ fence, spawnToken, options }) => { + activeModel = options?.model ?? DEFAULT_MODEL + return { + process: { hostId: 'local', pid: 4200, processStartTimeMs: NOW, spawnToken }, + link: { + linkId: `claude-native-${fence}`, + handle: { provider: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'native-leaf' }, + origin: acquire.mock.calls.length === 1 ? 'created' : 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } + }) + return { + acquire, + dispatch: vi.fn(), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async ({ value }) => { + activeModel = value + return { model: value } + }), + readOptions: vi.fn(async () => ({ current: { model: activeModel }, models: [] })), + closeSession: vi.fn(async () => { + activeModel = DEFAULT_MODEL + return true + }) + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-claude-handoff-options-')) + resetHostTestOperationIds() + activeModel = DEFAULT_MODEL + transcriptPath = join(root, 'claude.jsonl') + await writeFile(transcriptPath, '', 'utf8') + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-claude', + handoffTransport: transport(), + now: () => NOW + }) + expect( + await host.attach( + CALLER, + hostTestAttachParams(null, { + provider: 'claude', + agent: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: join(root, 'claude-home') }, + providerHandle: { kind: 'claude', sessionId: CLAUDE_SESSION, leafUuid: 'native-leaf' } + }) + ) + ).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + await new Promise((resolve) => setTimeout(resolve, 100)) + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true, maxRetries: 3, retryDelay: 50 }) +}) + +describe('Claude structured session handoff options', () => { + it('keeps a directly selected model through chat to TUI to chat', async () => { + const fields = { key: 'model', value: PICKED_MODEL } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', fields), + ...fields + }) + ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) + + expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + ) + expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + ) + + expect(acquire.mock.calls[1]?.[0].options).toEqual({ model: PICKED_MODEL }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + expect(activeModel).toBe(PICKED_MODEL) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts new file mode 100644 index 00000000000..6082ab074f5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-grouped-prompt.test.ts @@ -0,0 +1,166 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionMutationEnvelope } from '../../../shared/agent-session-wire' +import { encodeAgentSessionQuestionAnswers } from '../../../shared/agent-session-question-answer' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { journalDirectoryFor } from '../agent-session-journal/journal-paths' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import type { + AgentSessionDispatchOutcome, + StructuredAgentSessionAdapter +} from './structured-agent-session-adapter' +import type { AgentSessionAttachParams } from './structured-agent-session-attach' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' + +const CALLER = { callerKey: 'client-1' } + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +const attachParams = (): AgentSessionAttachParams => hostTestAttachParams(null) + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let answerPrompt: Mock +let ordinal = 0 + +function adapter(): StructuredAgentSessionAdapter { + const dispatch = vi.fn(async (): Promise => { + ordinal += 1 + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal } + } + }) + return { + acquire, + releaseAcquisition: vi.fn(async () => true), + dispatch, + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt, + setOption: vi.fn(async () => undefined) + } +} + +async function seedGroupedQuestion(): Promise<{ itemId: string; revision: number }> { + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: journalDirectoryFor(root, { workspaceId: 'workspace-1', sessionId: SESSION }) + }) + const appended = await journal.appendItem( + { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 100 }, + { + kind: 'question', + question: '2 grouped questions from Claude', + options: [], + questions: [ + { + id: 'q1', + question: 'Targets', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ] + }, + { + id: 'q2', + question: 'Host', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ], + resolution: { state: 'pending', selectedOptionId: null, resolvedBy: null, resolvedAt: null } + }, + { fence: 1 } + ) + return { itemId: appended.itemId, revision: appended.revision } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-wire-grouped-')) + resetHostTestOperationIds() + ordinal = 0 + acquire = vi.fn(async ({ fence }) => ({ + process: { + hostId: 'local', + pid: 4242, + processStartTimeMs: 1_700_000_000_000, + spawnToken: store.getRecord(SESSION)?.lease.reservedSpawnToken ?? 'spawn-a' + }, + link: { + linkId: `link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: store.getRecord(SESSION)?.providerHandleChain.length ? 'resumed' : 'created', + mintedAtFence: fence, + observedAt: NOW + } + })) + answerPrompt = vi.fn(async () => undefined) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-a', + now: () => NOW + }) +}) + +afterEach(async () => { + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('grouped question admission', () => { + it('admits renderer question-group payloads with child ids and multi-select answers', async () => { + const prompt = await seedGroupedQuestion() + const attached = await host.attach(CALLER, attachParams()) + expect(attached.ok).toBe(true) + const optionId = encodeAgentSessionQuestionAnswers([ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ]) + const fields = { itemId: prompt.itemId, expectedRevision: prompt.revision, optionId } + const result = await host.respondToPrompt(CALLER, { + envelope: envelope('agentSession.respondTo:question', fields), + kind: 'question', + ...fields + }) + expect(result).toMatchObject({ ok: true, value: { resolution: { state: 'resolved' } } }) + expect(answerPrompt).toHaveBeenCalledWith( + expect.objectContaining({ itemId: prompt.itemId, optionId }) + ) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts new file mode 100644 index 00000000000..b881d55e771 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-admission.ts @@ -0,0 +1,138 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionOperationOutcome } from '../../../shared/agent-session-operation-ledger' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffResult, + AgentSessionHandoffStatus, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import { AGENT_SESSION_WIRE_REFUSAL_CODES } from '../../../shared/agent-session-wire' +import { + agentSessionFingerprintConflict, + computeAgentSessionPayloadFingerprint +} from '../../../shared/agent-session-mutation-envelope' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export type StructuredHandoffAdmission = + | { decision: 'continue'; record: AgentSessionRecord; fingerprint: string } + | { decision: 'replay'; outcome: AgentSessionOperationOutcome } + | { decision: 'refused'; refusal: AgentSessionWireRefusal } + +export async function admitStructuredHandoffRequest(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + record: AgentSessionRecord + status?: AgentSessionHandoffStatus +}): Promise { + const action = input.params.action ?? 'start' + const requestFingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: input.record.sessionId, + fields: { direction: input.params.direction, mode: input.params.mode, action } + }) + const conflict = agentSessionFingerprintConflict(input.params.envelope, requestFingerprint) + if (conflict) { + return { decision: 'refused', refusal: conflict } + } + const fingerprint = computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff.operation', + sessionId: input.record.sessionId, + fields: { direction: input.params.direction } + }) + const operation = await input.operationGuard.check({ + callerKey: input.callerKey, + sessionId: input.record.sessionId, + operationId: input.params.envelope.clientOperationId, + fingerprint, + action, + ...(input.status ? { status: input.status } : {}), + now: input.deps.now() + }) + if (operation.decision === 'replay') { + return { decision: 'replay', outcome: operation.outcome } + } + if (operation.decision === 'refused') { + return { + decision: 'refused', + refusal: { + code: operation.code as 'agent_session_operation_conflict', + message: 'This handoff operation could not be admitted.' + } + } + } + if (input.params.envelope.expectedRuntimeFence !== input.record.lease.runtimeFence) { + await input.deps.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_checkpoint_stale' } + }) + input.operationGuard.finish(input.record.sessionId, input.params.envelope.clientOperationId) + return { + decision: 'refused', + refusal: { + code: 'agent_session_checkpoint_stale', + message: 'The session owner changed before the handoff request arrived.', + currentFence: input.record.lease.runtimeFence + } + } + } + return { decision: 'continue', record: input.record, fingerprint } +} + +export function replayedStructuredHandoffRefusal( + outcome: AgentSessionOperationOutcome +): AgentSessionWireRefusal | null { + if ( + outcome.status !== 'failed' || + !AGENT_SESSION_WIRE_REFUSAL_CODES.includes( + outcome.code as (typeof AGENT_SESSION_WIRE_REFUSAL_CODES)[number] + ) + ) { + return null + } + return { + code: outcome.code as (typeof AGENT_SESSION_WIRE_REFUSAL_CODES)[number], + message: 'This handoff request was previously refused.' + } +} + +export async function refuseAdmittedStructuredHandoff(input: { + deps: StructuredAgentSessionHandoffDeps + callerKey: string + params: AgentSessionHandoffRequest + refusal: AgentSessionWireRefusal +}): Promise> { + await input.deps.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.params.envelope.clientOperationId, + outcome: { status: 'failed', code: input.refusal.code } + }) + return { ok: false, refusal: input.refusal } +} + +export function structuredHandoffRetryIsAdmissible( + status: AgentSessionHandoffStatus, + params: AgentSessionHandoffRequest +): boolean { + return ( + status.phase === 'failed' && + status.direction === params.direction && + status.operationId === params.envelope.clientOperationId && + status.error?.recoverableOwner !== 'none' + ) +} + +export function structuredHandoffRetryResumesStoppedOwner( + record: AgentSessionRecord, + params: AgentSessionHandoffRequest +): boolean { + return ( + record.lease.claimStatus === 'released' && + record.lease.handoffStage === 'old-owner-stopped' && + record.lease.handoffOperationId === params.envelope.clientOperationId + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts new file mode 100644 index 00000000000..7f2bec98962 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.test.ts @@ -0,0 +1,98 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { StructuredAgentSessionHandoffFlowRunner } from './structured-agent-session-handoff-flow-runner' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-flow-runner-outcome-write-failure' +const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' +const OPERATION = `${NOW}-00000000000000000000000000000002` +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('structured handoff flow runner outcome-write failure', () => { + it('still reports the flow failure when the failed-outcome ledger write throws', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-handoff-flow-runner-')) + roots.push(root) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + // Materialize the store file so its later disappearance reads as corruption, + // making every subsequent ledger write reject. + await store.admitOperation({ + callerKey: 'seed', + operationId: `${NOW}-00000000000000000000000000000009`, + fingerprint: 'seed', + now: NOW + }) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: join(root, 'journal') + }) + await rm(join(root, 'store'), { recursive: true, force: true }) + const failures: unknown[] = [] + const fields = { + direction: 'to-native' as const, + mode: 'now' as const, + action: 'retry' as const + } + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: SESSION, + fields + }) + }, + ...fields + } + const runner = new StructuredAgentSessionHandoffFlowRunner({ + deps: { + store, + claimKeyId: 'key-1', + session: () => ({ journal, fence: 1 }), + suspendNative: async () => ({ state: 'stopped' as const }), + acquireNative: async () => { + throw new Error('unused') + }, + importTuiHistory: async () => {}, + publish: () => {}, + schedule: async () => { + throw new Error('scheduling failed') + }, + now: () => NOW + }, + operationGuard: new StructuredAgentSessionHandoffOperationGuard(store), + flowContext: (): StructuredAgentSessionHandoffFlowContext => { + throw new Error('unreachable: scheduling rejects before the flow needs context') + }, + fail: (_params, error) => { + failures.push(error) + } + }) + runner.begin({ callerKey: 'client-1', params, turnId: null, fingerprint: 'fp' }) + await runner.drain() + expect(failures).toHaveLength(1) + expect((failures[0] as Error).message).toBe('scheduling failed') + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts new file mode 100644 index 00000000000..7278502ce1f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-flow-runner.ts @@ -0,0 +1,110 @@ +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { stopStructuredNativeTurn } from './structured-agent-session-handoff-flow-context' +import { handoffStructuredSessionToTui } from './structured-agent-session-handoff-forward' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { assertScheduledStructuredHandoffIsAdmissible } from './structured-agent-session-handoff-revalidation' +import { handoffStructuredSessionToNative } from './structured-agent-session-handoff-reverse' +import { structuredTuiStatus } from './structured-agent-session-handoff-status' +import type { + StructuredAgentSessionHandoffDeps, + StructuredAgentSessionHandoffFlowContext +} from './structured-agent-session-handoff-types' + +export class StructuredAgentSessionHandoffFlowRunner { + private readonly active = new Set>() + + constructor( + private readonly input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + flowContext: () => StructuredAgentSessionHandoffFlowContext + fail: (params: AgentSessionHandoffRequest, error: unknown) => void + } + ) {} + + async drain(): Promise { + await Promise.allSettled(this.active) + } + + track(task: Promise): void { + this.active.add(task) + void task.finally(() => this.active.delete(task)) + } + + begin(input: { + callerKey: string + params: AgentSessionHandoffRequest + turnId: string | null + fingerprint: string + tuiAlreadyExited?: boolean + }): void { + const { callerKey, params, turnId, fingerprint, tuiAlreadyExited = false } = input + const sessionId = params.envelope.sessionId + const journalSequence = this.input.deps.session(sessionId).journal.cursor().sequence + this.input.operationGuard.start(sessionId, { + callerKey, + operationId: params.envelope.clientOperationId, + fingerprint + }) + const flow = this.run(params, turnId, tuiAlreadyExited, journalSequence) + .then(() => { + this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId) + return this.input.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId } + }) + }) + .catch(async (error) => { + try { + await this.input.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_handoff_failed' } + }) + } catch { + // Best-effort: a store write failure must not suppress the client's failure + // notification or leak the flow as an unhandled rejection. + } + this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId) + this.input.fail(params, error) + }) + .finally(() => this.input.operationGuard.finish(sessionId, params.envelope.clientOperationId)) + this.track(flow) + } + + private run( + params: AgentSessionHandoffRequest, + turnId: string | null, + tuiAlreadyExited: boolean, + journalSequence: number + ): Promise { + const sessionId = params.envelope.sessionId + return this.input.deps.schedule(sessionId, async () => { + const context = this.input.flowContext() + assertScheduledStructuredHandoffIsAdmissible({ + record: context.requireRecord(sessionId), + journal: this.input.deps.session(sessionId).journal, + params, + turnId, + journalSequence, + tuiAlreadyExited, + tuiStatus: structuredTuiStatus(context.owner(sessionId), this.input.deps.transport) + }) + if (turnId && params.mode === 'stop-turn') { + const stopped = await stopStructuredNativeTurn(this.input.deps, sessionId, turnId) + if (!stopped) { + throw new Error('The current turn did not acknowledge cancellation.') + } + } + await (params.direction === 'to-tui' + ? handoffStructuredSessionToTui(context, params, params.action === 'retry') + : handoffStructuredSessionToNative( + context, + params, + params.action === 'retry', + tuiAlreadyExited + )) + }) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts new file mode 100644 index 00000000000..7c77806ad91 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.test.ts @@ -0,0 +1,221 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { + agentSessionLeaseFixture, + agentSessionRecordFixture +} from '../../../shared/agent-session-record.test-fixture' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { + queuedStructuredHandoffCanBegin, + StructuredAgentSessionHandoffQueue +} from './structured-agent-session-handoff-queue' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { assertScheduledStructuredHandoffIsAdmissible } from './structured-agent-session-handoff-revalidation' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-alpha-1' +const OPERATION_A = `${NOW}-00000000000000000000000000000001` +const OPERATION_B = `${NOW}-00000000000000000000000000000002` + +let root: string | null = null + +afterEach(async () => { + if (root) { + await rm(root, { recursive: true, force: true }) + root = null + } +}) + +async function createGuard() { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-operation-guard-')) + const store = await AgentSessionRecordStore.open({ directory: root, hostId: 'local' }) + return { guard: new StructuredAgentSessionHandoffOperationGuard(store), store } +} + +function status(phase: 'switching' | 'queued' | 'idle'): AgentSessionHandoffStatus { + return { + owner: phase === 'idle' ? 'native' : 'none', + direction: phase === 'idle' ? null : 'to-tui', + phase, + stage: phase === 'switching' ? 'preparing' : null, + operationId: phase === 'idle' ? null : OPERATION_A + } +} + +describe('structured handoff operation ownership', () => { + it('reserves one winner across concurrent admissions', async () => { + const { guard } = await createGuard() + const check = (operationId: string) => + guard.check({ + callerKey: operationId, + sessionId: SESSION, + operationId, + fingerprint: operationId, + action: 'start', + now: NOW + }) + + const decisions = await Promise.all([check(OPERATION_A), check(OPERATION_B)]) + + expect(decisions.map(({ decision }) => decision).sort()).toEqual(['new', 'refused']) + }) + + it.each(['switching', 'queued'] as const)( + 'durably refuses a distinct operation while the %s operation owns the session', + async (phase) => { + const { guard } = await createGuard() + guard.start(SESSION, { + callerKey: 'client-a', + operationId: OPERATION_A, + fingerprint: 'fingerprint-a' + }) + + expect( + await guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'start', + status: status(phase), + now: NOW + }) + ).toEqual({ decision: 'refused', code: 'agent_session_operation_conflict' }) + + guard.finish(SESSION, OPERATION_A) + expect( + await guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'start', + status: status('idle'), + now: NOW + }) + ).toMatchObject({ + decision: 'replay', + outcome: { status: 'failed', code: 'agent_session_operation_conflict' } + }) + } + ) + + it('admits only cancellation beside a queued operation', async () => { + const { guard } = await createGuard() + guard.start(SESSION, { + callerKey: 'client-a', + operationId: OPERATION_A, + fingerprint: 'fingerprint-a' + }) + + await expect( + guard.check({ + callerKey: 'client-b', + sessionId: SESSION, + operationId: OPERATION_B, + fingerprint: 'fingerprint-b', + action: 'cancel-queued', + status: status('queued'), + now: NOW + }) + ).resolves.toEqual({ decision: 'new' }) + }) +}) + +describe('queued handoff fence revalidation', () => { + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION_A, + expectedRuntimeFence: 7, + payloadFingerprint: 'fingerprint' + }, + direction: 'to-tui', + mode: 'after-turn', + action: 'start' + } + const queued = status('queued') + + it('accepts the same live owner and fence', () => { + const record = agentSessionRecordFixture( + agentSessionLeaseFixture({ runtimeKind: 'native', ownerProcess: null }) + ) + expect(queuedStructuredHandoffCanBegin(record, queued, params)).toBe(true) + }) + + it.each([ + agentSessionLeaseFixture({ runtimeKind: 'native', runtimeFence: 8, ownerProcess: null }), + agentSessionLeaseFixture({ runtimeKind: 'tui' }), + agentSessionLeaseFixture({ + runtimeKind: 'native', + ownerProcess: null, + handoffStage: 'preparing' + }) + ])('refuses a changed durable owner or fence', (lease) => { + expect(queuedStructuredHandoffCanBegin(agentSessionRecordFixture(lease), queued, params)).toBe( + false + ) + }) + + it('cannot cancel after the idle waiter claims the queued operation', async () => { + const queue = new StructuredAgentSessionHandoffQueue() + const ready = vi.fn() + queue.enqueue(SESSION, () => true, ready) + await vi.waitFor(() => expect(ready).toHaveBeenCalledOnce()) + expect(queue.cancel(SESSION)).toBe(false) + }) +}) + +describe('scheduled handoff revalidation', () => { + it('refuses a native turn accepted ahead of the scheduled handoff', async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-revalidation-')) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'claude', + providerHandle: { kind: 'claude', sessionId: SESSION, leafUuid: null } + }, + journalDir: join(root, 'journal') + }) + const journalSequence = journal.cursor().sequence + await journal.appendItem( + { provider: 'orca', clientMessageId: 'turn-running' }, + { kind: 'status', text: 'running', turnLifecycle: { turnId: 'turn-1', state: 'running' } }, + { fence: 7 } + ) + const params: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION_A, + expectedRuntimeFence: 7, + payloadFingerprint: 'fingerprint' + }, + direction: 'to-tui', + mode: 'now', + action: 'start' + } + + expect(() => + assertScheduledStructuredHandoffIsAdmissible({ + record: agentSessionRecordFixture( + agentSessionLeaseFixture({ runtimeKind: 'native', ownerProcess: null }) + ), + journal, + params, + turnId: null, + journalSequence, + tuiAlreadyExited: false, + tuiStatus: 'busy' + }) + ).toThrow('session changed') + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts new file mode 100644 index 00000000000..da8d2eaad3d --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-operation-guard.ts @@ -0,0 +1,125 @@ +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { + AgentSessionOperationOutcome, + AgentSessionOperationRefusalCode +} from '../../../shared/agent-session-operation-ledger' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' + +type ActiveOperation = { callerKey: string; operationId: string; fingerprint: string } + +export type HandoffOperationDecision = + | { decision: 'new' } + | { decision: 'replay'; outcome: AgentSessionOperationOutcome } + | { decision: 'retry' } + | { decision: 'refused'; code: AgentSessionOperationRefusalCode } + +export class StructuredAgentSessionHandoffOperationGuard { + private readonly activeBySession = new Map() + + constructor(private readonly store: AgentSessionRecordStore) {} + + async check(input: { + callerKey: string + sessionId: string + operationId: string + fingerprint: string + action: 'start' | 'cancel-queued' | 'retry' | 'recover' + status?: AgentSessionHandoffStatus + now: number + }): Promise { + const ledger = await this.store.admitOperation({ + callerKey: input.callerKey, + operationId: input.operationId, + fingerprint: input.fingerprint, + now: input.now + }) + if (ledger.decision === 'refused') { + return { decision: 'refused', code: ledger.code } + } + const active = this.activeBySession.get(input.sessionId) + const queuedCancellation = + input.action === 'cancel-queued' && + input.status?.phase === 'queued' && + input.status.operationId === active?.operationId + const activeConflict = Boolean( + active && + ((active.operationId === input.operationId && + (active.fingerprint !== input.fingerprint || active.callerKey !== input.callerKey)) || + (active.operationId !== input.operationId && !queuedCancellation)) + ) + const queuedConflict = Boolean( + !active && + input.status?.phase === 'queued' && + input.status.operationId !== input.operationId && + input.action !== 'cancel-queued' + ) + if (activeConflict || queuedConflict) { + if (ledger.decision === 'admit') { + await this.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.operationId, + outcome: { status: 'failed', code: 'agent_session_operation_conflict' } + }) + } + return { decision: 'refused', code: 'agent_session_operation_conflict' } + } + if (ledger.decision === 'admit') { + this.reserve(input) + return { decision: 'new' } + } + if (input.action === 'retry' && ledger.row.outcome.status === 'failed') { + await this.store.recordOperationOutcome({ + callerKey: input.callerKey, + operationId: input.operationId, + outcome: { status: 'pending' } + }) + this.reserve(input) + return { decision: 'retry' } + } + if ( + ledger.row.outcome.status === 'pending' && + !active && + input.status?.operationId !== input.operationId + ) { + this.reserve(input) + return { decision: 'new' } + } + return { decision: 'replay', outcome: ledger.row.outcome } + } + + start(sessionId: string, operation: ActiveOperation): void { + this.activeBySession.set(sessionId, operation) + } + + private reserve(input: { + action: 'start' | 'cancel-queued' | 'retry' | 'recover' + callerKey: string + sessionId: string + operationId: string + fingerprint: string + }): void { + if (input.action !== 'cancel-queued') { + this.start(input.sessionId, input) + } + } + + finish(sessionId: string, operationId: string): void { + if (this.activeBySession.get(sessionId)?.operationId === operationId) { + this.activeBySession.delete(sessionId) + } + } + + async settle( + sessionId: string, + operationId: string, + outcome: AgentSessionOperationOutcome + ): Promise { + const active = this.activeBySession.get(sessionId) + await this.store.recordOperationOutcome({ + ...(active?.operationId === operationId ? { callerKey: active.callerKey } : {}), + operationId, + outcome + }) + this.finish(sessionId, operationId) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts new file mode 100644 index 00000000000..e5ee4f7ca9b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.test.ts @@ -0,0 +1,286 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi, type Mock } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffRequest, + AgentSessionMutationEnvelope +} from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' +import { AgentSessionOptionRejectedError } from './structured-agent-session-option-error' +import { StructuredAgentSessionHost } from './structured-agent-session-host' +import { + HOST_TEST_NOW as NOW, + HOST_TEST_SESSION as SESSION, + HOST_TEST_THREAD as THREAD, + hostTestAttachParams, + hostTestMessage, + hostTestOperationId, + resetHostTestOperationIds +} from './structured-agent-session-host-test-data' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +const CALLER = { callerKey: 'client-1' } +const DEFAULT_MODEL = 'gpt-default' +const PICKED_MODEL = 'gpt-picked' +const PICKED_EFFORT = 'medium' + +let root: string +let store: AgentSessionRecordStore +let host: StructuredAgentSessionHost +let acquire: Mock +let activeModel: string +let activeEffort: string | null +let transcriptPath: string +let optionFailure: Error | null +const dispatchedModels: string[] = [] +const launchedOptions: (Readonly> | undefined)[] = [] +const closedTuiOwners: StructuredTuiOwner[] = [] + +function envelope(method: string, fields: Record): AgentSessionMutationEnvelope { + return { + sessionId: SESSION, + clientOperationId: hostTestOperationId(), + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function handoff(direction: AgentSessionHandoffDirection): AgentSessionHandoffRequest { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { envelope: envelope('agentSession.requestHandoff', fields), ...fields } +} + +function tuiOwner(fence: number, spawnToken: string): StructuredTuiOwner { + return { + terminal: { handle: 'term-tui', tabId: 'tab-tui', paneKey: 'pane-tui', ptyId: 'pty-tui' }, + process: { + hostId: 'local', + pid: 5200, + processStartTimeMs: NOW, + spawnToken + }, + link: { + linkId: `tui-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + transcriptPath + } +} + +function handoffTransport(): StructuredAgentSessionHandoffTransport { + return { + hostLabel: 'Test host', + launchTui: async ({ record, fence, spawnToken }) => { + launchedOptions.push(record.options) + return tuiOwner(fence, spawnToken) + }, + reproveTuiOwner: async ({ owner }) => owner, + recoverTuiOwner: async (record) => + tuiOwner( + record.lease.runtimeFence, + record.lease.ownerProcess?.spawnToken ?? record.lease.reservedSpawnToken ?? 'recovered' + ), + stopRecoveredOwner: async () => undefined, + closeTuiOwner: async (owner) => { + closedTuiOwners.push(owner) + return { transcriptPath: owner.transcriptPath } + }, + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle' + } +} + +function adapter(): StructuredAgentSessionAdapter { + acquire = vi.fn(async ({ fence, spawnToken, options }) => { + activeModel = options?.model ?? DEFAULT_MODEL + activeEffort = options?.effort ?? null + return { + process: { + hostId: 'local', + pid: 4200 + acquire.mock.calls.length, + processStartTimeMs: NOW, + spawnToken + }, + link: { + linkId: `native-link-${fence}`, + handle: { provider: 'codex', threadId: THREAD }, + origin: acquire.mock.calls.length === 1 ? 'created' : 'resumed', + mintedAtFence: fence, + observedAt: NOW + } + } + }) + return { + acquire, + dispatch: vi.fn(async () => { + dispatchedModels.push(activeModel) + return { + state: 'accepted', + providerIdentity: { provider: 'codex', threadId: THREAD, turnId: 'turn-1', ordinal: 1 } + } + }), + cancelTurn: vi.fn(async () => ({ cancelled: true })), + answerPrompt: vi.fn(async () => undefined), + setOption: vi.fn(async ({ key, value }) => { + if (optionFailure) { + const error = optionFailure + optionFailure = null + throw error + } + if (key === 'model') { + activeModel = value + } else if (key === 'effort') { + activeEffort = value + } + return { + model: activeModel, + ...(activeEffort ? { effort: activeEffort } : {}) + } + }), + readOptions: vi.fn(async () => ({ + current: { model: activeModel, ...(activeEffort ? { effort: activeEffort } : {}) }, + models: [] + })), + closeSession: vi.fn(async () => { + activeModel = DEFAULT_MODEL + return true + }) + } +} + +beforeEach(async () => { + root = await mkdtemp(join(tmpdir(), 'orca-handoff-options-')) + resetHostTestOperationIds() + activeModel = DEFAULT_MODEL + activeEffort = null + optionFailure = null + dispatchedModels.length = 0 + launchedOptions.length = 0 + closedTuiOwners.length = 0 + const accountHome = join(root, 'codex-home') + const sessionsDir = join(accountHome, 'sessions', '2026', '08', '12') + transcriptPath = join(sessionsDir, `rollout-2026-08-12T10-00-00-${THREAD}.jsonl`) + await mkdir(sessionsDir, { recursive: true }) + await writeFile( + transcriptPath, + `${JSON.stringify({ + type: 'session_meta', + timestamp: '2026-08-12T10:00:00.000Z', + payload: { id: THREAD, session_id: THREAD } + })}\n`, + 'utf8' + ) + store = await AgentSessionRecordStore.open({ directory: join(root, 'store'), hostId: 'local' }) + host = new StructuredAgentSessionHost({ + store, + adapter: adapter(), + journalRoot: root, + claimKeyId: 'key-1', + mintSpawnToken: () => 'spawn-native', + handoffTransport: handoffTransport(), + now: () => NOW + }) + const attached = await host.attach( + CALLER, + hostTestAttachParams(null, { accountHome: { variable: 'CODEX_HOME', path: accountHome } }) + ) + expect(attached).toMatchObject({ ok: true }) +}) + +afterEach(async () => { + await host.flushAllStreamedEvents() + await rm(root, { recursive: true, force: true }) +}) + +describe('structured session handoff options', () => { + it('settles a pre-mutation rejection so a fresh retry can succeed', async () => { + optionFailure = new AgentSessionOptionRejectedError('model list unavailable') + const fields = { key: 'model', value: PICKED_MODEL } + const rejected = { + envelope: envelope('agentSession.setOption', fields), + ...fields + } + + expect(await host.setOption(CALLER, rejected)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_invalid', message: 'model list unavailable' } + }) + expect(await host.setOption(CALLER, rejected)).toMatchObject({ + ok: false, + refusal: { code: 'agent_session_operation_invalid' } + }) + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', fields), + ...fields + }) + ).toMatchObject({ ok: true, value: { options: { model: PICKED_MODEL } } }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + }) + + it('keeps a picked model through a native to TUI to native round trip', async () => { + const optionFields = { key: 'model', value: PICKED_MODEL } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', optionFields), + ...optionFields + }) + ).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ model: PICKED_MODEL }) + + const effortFields = { key: 'effort', value: PICKED_EFFORT } + expect( + await host.setOption(CALLER, { + envelope: envelope('agentSession.setOption', effortFields), + ...effortFields + }) + ).toMatchObject({ ok: true }) + expect(store.getRecord(SESSION)?.options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + + expect(await host.requestHandoff(CALLER, handoff('to-tui'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui' }) + ) + expect(await host.requestHandoff(CALLER, handoff('to-native'))).toMatchObject({ ok: true }) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native' }) + ) + + expect(launchedOptions).toEqual([{ model: PICKED_MODEL, effort: PICKED_EFFORT }]) + expect(closedTuiOwners).toHaveLength(1) + expect(acquire.mock.calls[1]?.[0].options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + expect(store.getRecord(SESSION)?.options).toEqual({ + model: PICKED_MODEL, + effort: PICKED_EFFORT + }) + const body = hostTestMessage('use the selected model') + expect( + await host.send(CALLER, { + envelope: envelope('agentSession.send', { body }), + body + }) + ).toMatchObject({ ok: true }) + expect(dispatchedModels).toEqual([PICKED_MODEL]) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts new file mode 100644 index 00000000000..a5afd9891a1 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-options.ts @@ -0,0 +1,23 @@ +import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' + +export async function readNativeHandoffSessionOptions(input: { + adapter: Pick + sessionId: string + fence: number + priorOptions?: Readonly> +}): Promise> | undefined> { + const { adapter, sessionId, fence, priorOptions } = input + const reported = await adapter.readOptions?.({ + sessionId, + fence + }) + if (!reported) { + return undefined + } + const { model: _model, effort: _effort, ...restored } = priorOptions ?? {} + return { + ...restored, + model: reported.current.model, + ...(reported.current.effort ? { effort: reported.current.effort } : {}) + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts new file mode 100644 index 00000000000..f8d6db2bace --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue-start.ts @@ -0,0 +1,46 @@ +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { StructuredAgentSessionHandoffQueue } from './structured-agent-session-handoff-queue' +import type { + StructuredAgentSessionHandoffDeps, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +export function queueStructuredHandoffAfterTurn(input: { + callerKey: string + params: AgentSessionHandoffRequest + deps: StructuredAgentSessionHandoffDeps + queue: StructuredAgentSessionHandoffQueue + owner: (sessionId: string) => StructuredTuiOwner | undefined + setStatus: ( + sessionId: string, + status: Parameters[1] + ) => void + begin: (callerKey: string, params: AgentSessionHandoffRequest, tuiAlreadyExited?: boolean) => void +}): void { + const { callerKey, params, deps, queue, owner, setStatus, begin } = input + const sessionId = params.envelope.sessionId + let tuiReadiness: 'idle' | 'exited' | null = null + setStatus(sessionId, { + owner: params.direction === 'to-tui' ? 'native' : 'tui', + direction: params.direction, + phase: 'queued', + stage: null, + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + const tuiOwner = owner(sessionId) + queue.enqueue( + sessionId, + async (signal) => { + if (params.direction === 'to-tui') { + return !activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items) + } + tuiReadiness = tuiOwner + ? ((await deps.transport?.waitForTuiIdleOrExit(tuiOwner, signal)) ?? null) + : null + return tuiReadiness !== null + }, + () => begin(callerKey, { ...params, mode: 'now' }, tuiReadiness === 'exited') + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts new file mode 100644 index 00000000000..9ea3d7ff08f --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-queue.ts @@ -0,0 +1,133 @@ +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { + StructuredAgentSessionHandoffDeps, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +export class StructuredAgentSessionHandoffQueue { + private readonly controllers = new Map() + + cancel(sessionId: string): boolean { + const controller = this.controllers.get(sessionId) + controller?.abort() + this.controllers.delete(sessionId) + return controller !== undefined + } + + enqueue( + sessionId: string, + isIdle: (signal: AbortSignal) => boolean | Promise, + onReady: () => void + ): void { + this.cancel(sessionId) + const controller = new AbortController() + this.controllers.set(sessionId, controller) + void this.waitUntilIdle(sessionId, controller, isIdle).then((ready) => { + if (ready) { + onReady() + } + }) + } + + private async waitUntilIdle( + sessionId: string, + controller: AbortController, + isIdle: (signal: AbortSignal) => boolean | Promise + ): Promise { + while (this.controllers.get(sessionId) === controller && !controller.signal.aborted) { + try { + if (await isIdle(controller.signal)) { + this.controllers.delete(sessionId) + return true + } + } catch { + if (controller.signal.aborted) { + return false + } + } + await new Promise((resolve) => setTimeout(resolve, 150)) + } + return false + } +} + +export function queuedStructuredHandoffCanBegin( + record: AgentSessionRecord, + status: AgentSessionHandoffStatus, + params: AgentSessionHandoffRequest +): boolean { + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + return ( + record.sessionId === params.envelope.sessionId && + status.phase === 'queued' && + status.direction === params.direction && + status.operationId === params.envelope.clientOperationId && + record.lease.runtimeFence === params.envelope.expectedRuntimeFence && + record.lease.runtimeKind === expectedOwner && + record.lease.claimStatus === 'live' && + record.lease.handoffStage === null && + !record.lease.unreconciled + ) +} + +export function enqueueStructuredHandoffAfterTurn(input: { + deps: StructuredAgentSessionHandoffDeps + queue: StructuredAgentSessionHandoffQueue + params: AgentSessionHandoffRequest + tuiOwner: StructuredTuiOwner | undefined + status: () => AgentSessionHandoffStatus + requireRecord: () => AgentSessionRecord + setStatus: (status: AgentSessionHandoffStatus) => void + begin: (params: AgentSessionHandoffRequest, tuiAlreadyExited: boolean) => void + refuse: (record: AgentSessionRecord) => void +}): void { + const { deps, params, queue, tuiOwner } = input + const sessionId = params.envelope.sessionId + let tuiReadiness: 'idle' | 'exited' | null = null + let observedTuiQueue = false + input.setStatus({ + owner: params.direction === 'to-tui' ? 'native' : 'tui', + direction: params.direction, + phase: 'queued', + stage: null, + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + queue.enqueue( + sessionId, + async (signal) => { + if (params.direction === 'to-tui') { + return !activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items) + } + if (!observedTuiQueue) { + observedTuiQueue = true + return false + } + tuiReadiness = tuiOwner + ? ((await deps.transport?.waitForTuiIdleOrExit(tuiOwner, signal)) ?? null) + : null + if (tuiReadiness === 'exited') { + return true + } + if (!activeStructuredAgentSessionTurnId(deps.session(sessionId).journal.snapshot().items)) { + tuiReadiness = 'idle' + return true + } + return false + }, + () => { + const record = input.requireRecord() + const status = input.status() + if (!queuedStructuredHandoffCanBegin(record, status, params)) { + input.refuse(record) + return + } + input.begin(params, tuiReadiness === 'exited') + } + ) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts new file mode 100644 index 00000000000..0aab4f0335b --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-recover.ts @@ -0,0 +1,30 @@ +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { + beginStructuredManualRecovery, + structuredManualRecoveryIsAdmissible +} from './structured-agent-session-manual-recovery' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' + +export async function requestStructuredManualRecovery(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + fingerprint: string + record: AgentSessionRecord + status: AgentSessionHandoffStatus + requireRecord: (sessionId: string) => AgentSessionRecord + restore: (sessionId: string) => Promise + setStatus: (sessionId: string, status: AgentSessionHandoffStatus) => void +}): Promise { + if (!structuredManualRecoveryIsAdmissible(input.record, input.status)) { + return false + } + beginStructuredManualRecovery(input) + return true +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts new file mode 100644 index 00000000000..fe480d6359c --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-result.ts @@ -0,0 +1,32 @@ +import type { + AgentSessionHandoffResult, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export function structuredHandoffRefusal( + code: AgentSessionWireRefusal['code'], + message: string +): AgentSessionWireRefusal { + return { code, message } +} + +export function structuredHandoffSuccess( + deps: StructuredAgentSessionHandoffDeps, + sessionId: string, + replayed: boolean, + status: AgentSessionHandoffResult['status'] +): AgentSessionMutationResult { + const record = deps.store.getRecord(sessionId) + if (!record) { + throw new Error('agent_session_identity_required') + } + return { + ok: true, + replayed, + fence: record.lease.runtimeFence, + cursor: deps.session(sessionId).journal.cursor(), + value: { status } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts new file mode 100644 index 00000000000..66005959378 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-revalidation.ts @@ -0,0 +1,52 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { structuredHandoffRetryResumesStoppedOwner } from './structured-agent-session-handoff-admission' +import { structuredSessionHasPendingPrompt } from './structured-agent-session-handoff-status' + +export function assertScheduledStructuredHandoffIsAdmissible(input: { + record: AgentSessionRecord + journal: AgentSessionJournal + params: AgentSessionHandoffRequest + turnId: string | null + journalSequence: number + tuiAlreadyExited: boolean + tuiStatus: 'idle' | 'busy' +}): void { + const { params, record } = input + if (params.action === 'retry' && structuredHandoffRetryResumesStoppedOwner(record, params)) { + return + } + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + if ( + record.lease.runtimeFence !== params.envelope.expectedRuntimeFence || + record.lease.runtimeKind !== expectedOwner || + record.lease.claimStatus !== 'live' || + record.lease.handoffStage !== null || + record.lease.unreconciled + ) { + throw new Error('agent_session_checkpoint_stale') + } + if (structuredSessionHasPendingPrompt(input.journal)) { + throw new Error('Resolve the pending question or approval before switching.') + } + if (params.mode !== 'stop-turn' && input.journal.cursor().sequence !== input.journalSequence) { + throw new Error('The session changed before the handoff started.') + } + const activeTurn = activeStructuredAgentSessionTurnId(input.journal.snapshot().items) + if (params.direction === 'to-tui') { + const expectedTurn = params.mode === 'stop-turn' ? input.turnId : null + if (activeTurn !== expectedTurn) { + throw new Error('The native turn changed before the handoff started.') + } + return + } + if ( + !input.tuiAlreadyExited && + input.tuiStatus !== 'idle' && + (params.mode !== 'after-turn' || activeTurn !== null) + ) { + throw new Error('The agent terminal became busy before the handoff started.') + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts new file mode 100644 index 00000000000..91c6163bd17 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.test.ts @@ -0,0 +1,100 @@ +import { describe, expect, it, vi } from 'vitest' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import { handoffStructuredSessionToNative } from './structured-agent-session-handoff-reverse' +import type { StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' + +const OPERATION_ID = 'operation-1' +const SESSION_ID = 'session-1' + +vi.mock('../../runtime/agent-session-handoff-record-transitions', () => ({ + abandonStoredAgentSessionHandoffAttempt: vi.fn(async () => undefined), + reserveStoredAgentSessionHandoffOwner: vi.fn(async () => record()), + rollbackStoredAgentSessionHandoffPreparation: vi.fn(async () => undefined), + stopStoredAgentSessionOwnerForHandoff: vi.fn(async () => record()) +})) + +function record(): AgentSessionRecord { + return { + sessionId: SESSION_ID, + provider: 'claude', + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + lease: { + runtimeFence: 3, + handoffStage: 'old-owner-stopped', + handoffOperationId: OPERATION_ID + } + } as unknown as AgentSessionRecord +} + +function contextWith( + revealNativeSession: () => Promise, + statuses: AgentSessionHandoffStatus[] +): StructuredAgentSessionHandoffFlowContext { + return { + deps: { + store: {} as never, + claimKeyId: 'key-1', + now: () => 1_800_000_000_000, + importTuiHistory: vi.fn(async () => undefined), + acquireNative: vi.fn(async () => record()), + transport: { revealNativeSession } + } as never, + owner: () => undefined, + retainOwner: vi.fn(), + releaseOwner: vi.fn(), + setStatus: (_sessionId, status) => statuses.push(status), + enterPreparing: vi.fn(async () => undefined), + publishStage: vi.fn(), + requireRecord: () => record() + } +} + +// Why this ordering matters: releaseOwner has already run by the time the reveal fires, +// so a reveal that rejects before the status flip leaves the session released but never +// marked native — a stuck chat with no owner on either side. +describe('handoffStructuredSessionToNative', () => { + it('marks the session native before revealing it', async () => { + const statuses: AgentSessionHandoffStatus[] = [] + const order: string[] = [] + const context = contextWith(async () => { + order.push('reveal') + }, statuses) + const setStatus = context.setStatus + context.setStatus = (sessionId, status) => { + order.push('status') + setStatus(sessionId, status) + } + + await handoffStructuredSessionToNative( + context, + { envelope: { sessionId: SESSION_ID, clientOperationId: OPERATION_ID } } as never, + true + ) + + expect(order).toEqual(['status', 'reveal']) + expect(statuses.at(-1)).toMatchObject({ owner: 'native', direction: null, phase: 'idle' }) + }) + + it('still leaves the session marked native when the reveal rejects', async () => { + const statuses: AgentSessionHandoffStatus[] = [] + const context = contextWith(async () => { + throw new Error('publish failed') + }, statuses) + + await expect( + handoffStructuredSessionToNative( + context, + { envelope: { sessionId: SESSION_ID, clientOperationId: OPERATION_ID } } as never, + true + ) + ).rejects.toThrow('publish failed') + + expect(statuses.at(-1)).toMatchObject({ owner: 'native' }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts index f59f7735245..ebfca81525c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-reverse.ts @@ -130,12 +130,8 @@ export async function handoffStructuredSessionToNative( throw error } context.releaseOwner(sessionId) - await deps.transport?.revealNativeSession?.({ - workspaceId: record.location.workspaceId, - sessionId, - agent: record.provider, - ...(owner?.adoptedTerminal ? { adoptedTerminal: true } : {}) - }) + // Why status lands before the reveal: the native owner is already proven here, and a + // reveal that rejects must not leave the session released but never marked native. context.setStatus(sessionId, { owner: 'native', direction: null, @@ -143,4 +139,10 @@ export async function handoffStructuredSessionToNative( stage: record.lease.handoffStage, operationId: record.lease.handoffOperationId }) + await deps.transport?.revealNativeSession?.({ + workspaceId: record.location.workspaceId, + sessionId, + agent: record.provider, + ...(owner?.adoptedTerminal ? { adoptedTerminal: true } : {}) + }) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts new file mode 100644 index 00000000000..e5dd478f719 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-coordinator.ts @@ -0,0 +1,80 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import type { AgentSessionJournal } from '../agent-session-journal/journal-store' +import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from './structured-agent-session-handoff-types' + +type TestCoordinatorInput = { + store: AgentSessionRecordStore + journal: AgentSessionJournal + sessionId: string + provider: 'claude' | 'codex' + claudeSessionId: string + codexThreadId: string + now: number + launchTui: StructuredAgentSessionHandoffTransport['launchTui'] + reproveTuiOwner: StructuredAgentSessionHandoffTransport['reproveTuiOwner'] + stopRecoveredOwner: StructuredAgentSessionHandoffTransport['stopRecoveredOwner'] + closeTuiOwner: NonNullable + waitForTuiExit: StructuredAgentSessionHandoffTransport['waitForTuiExit'] + waitForTuiIdleOrExit: StructuredAgentSessionHandoffTransport['waitForTuiIdleOrExit'] + stopFailedTuiLaunch: NonNullable + recoverTuiOwner: (record: AgentSessionRecord) => Promise + tuiStatus: () => 'idle' | 'busy' + acquireNative: (input: { + sessionId: string + fence: number + spawnToken: string + }) => Promise + acquireNativeStop: (turnId: string) => Promise + takeImportFailure: () => Error | null + statuses: AgentSessionHandoffStatus[] +} + +export function createStructuredAgentSessionHandoffTestCoordinator( + input: TestCoordinatorInput +): StructuredAgentSessionHandoffCoordinator { + return new StructuredAgentSessionHandoffCoordinator({ + store: input.store, + claimKeyId: 'key-1', + transport: { + hostLabel: 'Test host', + launchTui: input.launchTui, + reproveTuiOwner: input.reproveTuiOwner, + recoverTuiOwner: input.recoverTuiOwner, + stopRecoveredOwner: input.stopRecoveredOwner, + closeTuiOwner: input.closeTuiOwner, + waitForTuiExit: input.waitForTuiExit, + waitForTuiIdleOrExit: input.waitForTuiIdleOrExit, + tuiStatus: input.tuiStatus, + stopFailedTuiLaunch: input.stopFailedTuiLaunch + }, + session: () => ({ + journal: input.journal, + fence: input.store.getRecord(input.sessionId)?.lease.runtimeFence ?? 1 + }), + suspendNative: async () => ({ state: 'stopped' as const }), + acquireNative: input.acquireNative, + acquireNativeStop: (_sessionId, turnId) => input.acquireNativeStop(turnId), + importTuiHistory: async ({ fence }) => { + const importFailure = input.takeImportFailure() + if (importFailure) { + throw importFailure + } + await input.journal.appendItem( + input.provider === 'claude' + ? { provider: 'claude', sessionId: input.claudeSessionId, uuid: 'tui-turn' } + : { provider: 'codex', threadId: input.codexThreadId, turnId: 'tui-turn', ordinal: 0 }, + { kind: 'message', role: 'assistant', blocks: [{ type: 'text', text: 'from tui' }] }, + { fence, recovered: true } + ) + }, + publish: (_sessionId, status) => input.statuses.push(status), + schedule: async (_sessionId, task) => task(), + now: () => input.now + }) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts new file mode 100644 index 00000000000..ca33e486999 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-identities.ts @@ -0,0 +1,40 @@ +export type StructuredHandoffProviderCase = { + provider: 'claude' | 'codex' + accountHome: { variable: 'CLAUDE_CONFIG_DIR' | 'CODEX_HOME'; pathName: string } +} + +export const STRUCTURED_HANDOFF_PROVIDER_CASES: StructuredHandoffProviderCase[] = [ + { provider: 'codex', accountHome: { variable: 'CODEX_HOME', pathName: 'codex-home' } }, + { + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', pathName: 'claude-home' } + } +] + +export function structuredHandoffTestProcess(now: number, spawnToken: string, pid: number) { + return { hostId: 'local', pid, processStartTimeMs: now - 1_000, spawnToken } +} + +export function structuredHandoffTestLink(input: { + provider: 'claude' | 'codex' + fence: number + id: string + now: number + claudeSessionId: string + codexThreadId: string +}) { + return { + linkId: input.id, + handle: + input.provider === 'claude' + ? ({ + provider: 'claude' as const, + sessionId: input.claudeSessionId, + leafUuid: input.id.startsWith('native-link') ? 'tui-exit-leaf' : 'current-leaf' + } as const) + : ({ provider: 'codex' as const, threadId: input.codexThreadId } as const), + origin: 'resumed' as const, + mintedAtFence: input.fence, + observedAt: input.now + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts new file mode 100644 index 00000000000..550ebded0da --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff-test-requests.ts @@ -0,0 +1,53 @@ +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffAction, + AgentSessionHandoffMode, + AgentSessionHandoffRequest +} from '../../../shared/agent-session-wire' + +export type StructuredHandoffTestRequestOptions = { + action?: AgentSessionHandoffAction + operationId?: string +} + +export class StructuredHandoffTestRequests { + private operations = 0 + + constructor( + private readonly now: number, + private readonly sessionId: string, + private readonly readFence: () => number + ) {} + + reset(): void { + this.operations = 0 + } + + operationId(): string { + this.operations += 1 + return `${this.now}-${this.operations.toString(16).padStart(32, '0')}` + } + + request( + direction: AgentSessionHandoffDirection, + mode: AgentSessionHandoffMode, + options: StructuredHandoffTestRequestOptions = {} + ): AgentSessionHandoffRequest { + const action = options.action ?? 'start' + const fields = { direction, mode, action } + return { + envelope: { + sessionId: this.sessionId, + clientOperationId: options.operationId ?? this.operationId(), + expectedRuntimeFence: this.readFence(), + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: this.sessionId, + fields + }) + }, + ...fields + } + } +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts index 0e213921609..57333d1c90c 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-handoff.ts @@ -1,63 +1,286 @@ import type { AgentSessionRecord } from '../../../shared/agent-session-record' -import type { AgentSessionHandoffStatus } from '../../../shared/agent-session-wire' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffResult, + AgentSessionHandoffStatus, + AgentSessionMutationResult, + AgentSessionWireRefusal +} from '../../../shared/agent-session-wire' +import { activeStructuredAgentSessionTurnId } from '../../../shared/structured-agent-session-projection' +import { + admitStructuredHandoffRequest, + refuseAdmittedStructuredHandoff, + replayedStructuredHandoffRefusal, + structuredHandoffRetryIsAdmissible +} from './structured-agent-session-handoff-admission' import { createStructuredHandoffFlowContext, requireStructuredHandoffRecord } from './structured-agent-session-handoff-flow-context' -import { restoreStructuredAgentSessionHandoff } from './structured-agent-session-handoff-restart' +import { StructuredAgentSessionHandoffFlowRunner } from './structured-agent-session-handoff-flow-runner' +import { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { StructuredAgentSessionHandoffQueue } from './structured-agent-session-handoff-queue' +import { queueStructuredHandoffAfterTurn } from './structured-agent-session-handoff-queue-start' import { closeRetainedTuiOwner } from './structured-agent-session-handoff-owner-close' +import { requestStructuredManualRecovery } from './structured-agent-session-handoff-recover' +import { restoreStructuredAgentSessionHandoff } from './structured-agent-session-handoff-restart' +import { + structuredHandoffRefusal as refusal, + structuredHandoffSuccess +} from './structured-agent-session-handoff-result' +import { + failedStructuredHandoffStatus, + idleStructuredHandoffStatus, + structuredSessionHasPendingPrompt, + structuredTuiStatus +} from './structured-agent-session-handoff-status' import type { StructuredAgentSessionHandoffDeps, StructuredAgentSessionHandoffFlowContext } from './structured-agent-session-handoff-types' import { StructuredAgentSessionHandoffState } from './structured-agent-session-handoff-state' - export class StructuredAgentSessionHandoffCoordinator { private readonly state: StructuredAgentSessionHandoffState - + private readonly queue = new StructuredAgentSessionHandoffQueue() + private readonly operationGuard: StructuredAgentSessionHandoffOperationGuard + private readonly flowRunner: StructuredAgentSessionHandoffFlowRunner constructor(private readonly deps: StructuredAgentSessionHandoffDeps) { - // oxfmt-ignore - this.state = new StructuredAgentSessionHandoffState({ requireRecord: (sessionId) => this.requireRecord(sessionId), publish: deps.publish, hostLabel: deps.transport?.hostLabel }) - } - - status = (sessionId: string) => this.state.status(sessionId) - - closeRetainedTuiOwner = (sessionId: string): Promise => - closeRetainedTuiOwner({ - sessionId, - deps: this.deps, - owner: this.state.owner, - requireRecord: this.requireRecord, - releaseOwner: this.state.releaseOwner + this.state = new StructuredAgentSessionHandoffState({ + requireRecord: (sessionId) => this.requireRecord(sessionId), + publish: deps.publish, + hostLabel: deps.transport?.hostLabel }) - + this.operationGuard = new StructuredAgentSessionHandoffOperationGuard(deps.store) + this.flowRunner = new StructuredAgentSessionHandoffFlowRunner({ + deps, + operationGuard: this.operationGuard, + flowContext: () => this.flowContext(), + fail: (params, error) => this.fail(params, error) + }) + } + status = (sessionId: string): AgentSessionHandoffStatus => this.state.status(sessionId) + drain = (): Promise => this.flowRunner.drain() + closeRetainedTuiOwner = (sessionId: string): Promise => + this.closeRetainedOwner(sessionId) setStatus = (sessionId: string, status: AgentSessionHandoffStatus): void => this.state.setStatus(sessionId, status) - + async request( + callerKey: string, + params: AgentSessionHandoffRequest + ): Promise> { + const record = this.requireRecord(params.envelope.sessionId) + const currentStatus = this.state.cachedStatus(record.sessionId) + const admission = await admitStructuredHandoffRequest({ + deps: this.deps, + operationGuard: this.operationGuard, + callerKey, + params, + record, + ...(currentStatus ? { status: currentStatus } : {}) + }) + if (admission.decision === 'replay') { + const replayedRefusal = replayedStructuredHandoffRefusal(admission.outcome) + if (replayedRefusal) { + return { ok: false, refusal: replayedRefusal } + } + return this.success(record.sessionId, true) + } + if (admission.decision === 'refused') { + return { ok: false, refusal: admission.refusal } + } + const { fingerprint } = admission + const action = params.action ?? 'start' + if (action === 'cancel-queued') { + if (currentStatus?.phase !== 'queued' || currentStatus?.direction !== params.direction) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'No matching queued handoff exists.' + ) + } + this.queue.cancel(record.sessionId) + this.setStatus(record.sessionId, idleStructuredHandoffStatus(record)) + await this.deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId: record.sessionId } + }) + return this.success(record.sessionId, false) + } + if (!this.deps.transport) { + return this.refuseAdmitted( + callerKey, + params, + 'structured_agent_session_unsupported', + 'Agent TUI handoff is unavailable on this host.' + ) + } + if (action === 'recover') { + const status = this.status(record.sessionId) + const started = await requestStructuredManualRecovery({ + deps: this.deps, + operationGuard: this.operationGuard, + callerKey, + params, + fingerprint, + record, + status, + requireRecord: this.requireRecord, + restore: this.restore, + setStatus: this.setStatus + }) + if (!started) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'This handoff is no longer eligible for proof recovery.' + ) + } + return this.success(record.sessionId, false) + } + if (action === 'retry') { + if (!structuredHandoffRetryIsAdmissible(this.status(record.sessionId), params)) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_operation_conflict', + 'This handoff is no longer retryable.' + ) + } + this.begin(callerKey, params, null, fingerprint) + return this.success(record.sessionId, false) + } + const expectedOwner = params.direction === 'to-tui' ? 'native' : 'tui' + if (record.lease.runtimeKind !== expectedOwner || record.lease.claimStatus !== 'live') { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + `The ${expectedOwner} runtime does not own this session.` + ) + } + if (structuredSessionHasPendingPrompt(this.deps.session(record.sessionId).journal)) { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + 'Resolve the pending question or approval before switching.' + ) + } + const turnId = activeStructuredAgentSessionTurnId( + this.deps.session(record.sessionId).journal.snapshot().items + ) + const tuiOwner = this.state.owner(record.sessionId) + const busy = + expectedOwner === 'native' + ? turnId !== null + : structuredTuiStatus(tuiOwner, this.deps.transport) !== 'idle' + if (busy && params.mode === 'now') { + return this.refuseAdmitted( + callerKey, + params, + 'agent_session_conflict', + 'The current turn must finish before switching.' + ) + } + if (busy && params.mode === 'after-turn') { + queueStructuredHandoffAfterTurn({ + callerKey, + params, + deps: this.deps, + queue: this.queue, + owner: (sessionId) => this.state.owner(sessionId), + setStatus: this.setStatus, + begin: (key, next, tuiAlreadyExited) => + this.begin(key, next, null, fingerprint, tuiAlreadyExited) + }) + return this.success(record.sessionId, false) + } + if (busy && expectedOwner === 'tui' && params.mode === 'stop-turn') { + return this.refuseAdmitted( + callerKey, + params, + 'structured_agent_session_unsupported', + 'Exit the agent terminal after this turn to continue in chat.' + ) + } + this.begin(callerKey, params, turnId, fingerprint) + return this.success(record.sessionId, false) + } async restore(sessionId: string): Promise { await restoreStructuredAgentSessionHandoff( { deps: this.deps, requireRecord: (id) => this.requireRecord(id), flowContext: () => this.flowContext(), - retainOwner: this.state.retainOwner, - setStatus: this.state.setStatus + retainOwner: (id, owner) => this.state.retainOwner(id, owner), + setStatus: (id, status) => this.state.setStatus(id, status) }, sessionId ) } - + private refuseAdmitted( + callerKey: string, + params: AgentSessionHandoffRequest, + code: AgentSessionWireRefusal['code'], + message: string + ): Promise> { + return refuseAdmittedStructuredHandoff({ + deps: this.deps, + callerKey, + params, + refusal: refusal(code, message) + }) + } + private success( + sessionId: string, + replayed: boolean + ): AgentSessionMutationResult { + return structuredHandoffSuccess(this.deps, sessionId, replayed, this.status(sessionId)) + } + private begin( + callerKey: string, + params: AgentSessionHandoffRequest, + turnId: string | null, + fingerprint: string, + tuiAlreadyExited = false + ): void { + this.flowRunner.begin({ + callerKey, + params, + turnId, + fingerprint, + tuiAlreadyExited + }) + } private flowContext(): StructuredAgentSessionHandoffFlowContext { return createStructuredHandoffFlowContext({ deps: this.deps, - owner: this.state.owner, - retainOwner: this.state.retainOwner, - releaseOwner: this.state.releaseOwner, - setStatus: this.state.setStatus, + owner: (sessionId) => this.state.owner(sessionId), + retainOwner: (sessionId, owner) => this.state.retainOwner(sessionId, owner), + releaseOwner: (sessionId) => this.state.releaseOwner(sessionId), + setStatus: (sessionId, status) => this.state.setStatus(sessionId, status), requireRecord: (sessionId) => this.requireRecord(sessionId) }) } - + private fail(params: AgentSessionHandoffRequest, error: unknown): void { + const record = this.requireRecord(params.envelope.sessionId) + this.setStatus( + record.sessionId, + failedStructuredHandoffStatus(record, params, error, this.deps.transport?.hostLabel) + ) + } + private closeRetainedOwner(sessionId: string): Promise { + return closeRetainedTuiOwner({ + sessionId, + deps: this.deps, + owner: this.state.owner, + requireRecord: this.requireRecord, + releaseOwner: this.state.releaseOwner + }) + } private requireRecord = (sessionId: string): AgentSessionRecord => requireStructuredHandoffRecord(this.deps, sessionId) } diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts index 99d6c212c10..c38f64c3318 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-host.ts @@ -6,6 +6,8 @@ import type { AgentSessionAttachResult, AgentSessionHistoryRequest, AgentSessionHistoryResult, + AgentSessionHandoffRequest, + AgentSessionHandoffResult, AgentSessionHandoffStatus, AgentSessionMutationResult, AgentSessionOptionsResult, @@ -56,7 +58,6 @@ import type { StructuredAgentSessionHostSession } from './structured-agent-session-host-types' import { readStructuredAgentSessionHistoryResult } from './structured-agent-session-history-result' -import { retryPendingStructuredAgentSessionSettlement } from './structured-agent-session-settlement-retry' import { StructuredAgentSessionEventRecovery } from './structured-agent-session-event-recovery' export type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' export class StructuredAgentSessionHost { @@ -181,14 +182,6 @@ export class StructuredAgentSessionHost { subscribers: this.subscribers, tasks: this.tasks, reconcileLeases: (sessionId) => this.reconcileLeases(sessionId), - retryPendingSettlement: (sessionId, params) => - retryPendingStructuredAgentSessionSettlement({ - deps: this.deps, - sessions: this.sessions, - sessionId, - params, - now: () => this.now() - }), serialize: (sessionId, task) => this.serialize(sessionId, task), now: () => this.now() } @@ -299,6 +292,12 @@ export class StructuredAgentSessionHost { ): ReturnType => setStructuredAgentSessionOption(this.mutationContext(), caller, params) + requestHandoff = ( + caller: StructuredAgentSessionCaller, + params: AgentSessionHandoffRequest + ): Promise> => + this.handoffs.request(caller.callerKey, params) + readOptions = (sessionId: string): Promise => readStructuredAgentSessionOptions(this.mutationContext(), sessionId) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts new file mode 100644 index 00000000000..91be1a15aa5 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-manual-recovery.ts @@ -0,0 +1,103 @@ +import type { AgentSessionRecord } from '../../../shared/agent-session-record' +import type { + AgentSessionHandoffRequest, + AgentSessionHandoffStatus +} from '../../../shared/agent-session-wire' +import { setStoredAgentSessionHandoffStage } from '../../runtime/agent-session-handoff-record-transitions' +import type { StructuredAgentSessionHandoffOperationGuard } from './structured-agent-session-handoff-operation-guard' +import { idleStructuredHandoffStatus } from './structured-agent-session-handoff-status' +import type { StructuredAgentSessionHandoffDeps } from './structured-agent-session-handoff-types' + +export function structuredManualRecoveryIsAdmissible( + record: AgentSessionRecord, + status: AgentSessionHandoffStatus | undefined +): boolean { + return ( + record.lease.handoffStage === 'manual-recovery' && + record.lease.runtimeKind === 'tui' && + record.lease.ownerProcess !== null && + status?.error?.canRetryProof === true + ) +} + +export function beginStructuredManualRecovery(input: { + deps: StructuredAgentSessionHandoffDeps + operationGuard: StructuredAgentSessionHandoffOperationGuard + callerKey: string + params: AgentSessionHandoffRequest + fingerprint: string + requireRecord: (sessionId: string) => AgentSessionRecord + restore: (sessionId: string) => Promise + setStatus: (sessionId: string, status: AgentSessionHandoffStatus) => void +}): Promise { + const { + callerKey, + deps, + fingerprint, + operationGuard, + params, + requireRecord, + restore, + setStatus + } = input + const sessionId = params.envelope.sessionId + operationGuard.start(sessionId, { + callerKey, + operationId: params.envelope.clientOperationId, + fingerprint + }) + setStatus(sessionId, { + owner: 'none', + direction: params.direction, + phase: 'switching', + stage: 'recovering', + operationId: params.envelope.clientOperationId, + hostLabel: deps.transport?.hostLabel + }) + return deps + .schedule(sessionId, async () => { + let record = requireRecord(sessionId) + if (record.lease.claimStatus === 'reserved' && record.lease.handoffOperationId !== null) { + record = await setStoredAgentSessionHandoffStage(deps.store, { + sessionId, + fence: record.lease.runtimeFence, + stage: 'new-owner-proving', + handoffOperationId: record.lease.handoffOperationId, + now: deps.now() + }) + } + await restore(record.sessionId) + if (requireRecord(sessionId).lease.handoffStage === 'manual-recovery') { + throw new Error('The TUI owner proof is still unavailable.') + } + }) + .then(() => { + operationGuard.finish(sessionId, params.envelope.clientOperationId) + return deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'succeeded', sessionId } + }) + }) + .catch(async (error) => { + await deps.store.recordOperationOutcome({ + callerKey, + operationId: params.envelope.clientOperationId, + outcome: { status: 'failed', code: 'agent_session_handoff_failed' } + }) + operationGuard.finish(sessionId, params.envelope.clientOperationId) + const status = idleStructuredHandoffStatus(requireRecord(sessionId)) + setStatus(sessionId, { + ...status, + ...(status.error + ? { + error: { + ...status.error, + details: error instanceof Error ? error.message : String(error) + } + } + : {}) + }) + }) + .finally(() => operationGuard.finish(sessionId, params.envelope.clientOperationId)) +} diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts index b9ba03ff327..6e71f822170 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-option-restoration.ts @@ -1,7 +1,7 @@ import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' export async function readNativeSessionOptions(input: { - adapter: Pick + adapter: Pick sessionId: string fence: number priorOptions?: Readonly> @@ -11,7 +11,13 @@ export async function readNativeSessionOptions(input: { if (!reported) { return undefined } - const { model: _model, effort: _effort, ...restored } = priorOptions ?? {} + const skipped = new Set(input.adapter.readOptionRestoreFailures?.(sessionId) ?? []) + const restored = priorOptions ? { ...priorOptions } : {} + delete restored.model + delete restored.effort + for (const key of skipped) { + delete restored[key] + } return { ...restored, model: reported.current.model, diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts new file mode 100644 index 00000000000..3caba894cb9 --- /dev/null +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-proven-dead-retry.test.ts @@ -0,0 +1,178 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, describe, expect, it, vi } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../../shared/agent-session-mutation-envelope' +import type { AgentSessionHandoffRequest } from '../../../shared/agent-session-wire' +import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' +import { recoverStoredDeadTuiOwnerForHandoff } from '../../runtime/agent-session-handoff-record-transitions' +import { openAgentSessionJournal } from '../agent-session-journal/journal-store-factory' +import { StructuredAgentSessionHandoffCoordinator } from './structured-agent-session-handoff' +import type { StructuredAgentSessionHandoffTransport } from './structured-agent-session-handoff-types' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-proven-dead-retry' +const THREAD = '019fd532-7c11-7a90-b6de-4e1a2c3d5f60' +const CREATE_OPERATION = `${NOW}-00000000000000000000000000000000` +const OPERATION = `${NOW}-00000000000000000000000000000001` +const roots: string[] = [] + +afterEach(async () => { + await Promise.all(roots.splice(0).map((root) => rm(root, { recursive: true, force: true }))) +}) + +describe('structured session proven-dead TUI retry', () => { + it('acquires native ownership without trying to close the dead TUI again', async () => { + const root = await mkdtemp(join(tmpdir(), 'orca-handoff-dead-retry-')) + roots.push(root) + const store = await AgentSessionRecordStore.open({ + directory: join(root, 'store'), + hostId: 'local' + }) + const reserved = await store.reserveOwner({ + sessionId: SESSION, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'folder' + }, + provider: 'codex', + accountHome: { variable: 'CODEX_HOME', path: join(root, 'codex-home') }, + runtimeKind: 'tui', + expectedFence: null, + spawnToken: 'tui-spawn', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'reservation-unused' }, + operation: { callerKey: 'test', operationId: CREATE_OPERATION, fingerprint: 'create' }, + now: NOW + }) + const tuiFence = reserved.record.lease.runtimeFence + await store.commitProcessIdentity({ + sessionId: SESSION, + fence: tuiFence, + process: { + hostId: 'local', + pid: 4200, + processStartTimeMs: NOW - 1_000, + spawnToken: 'tui-spawn' + }, + now: NOW + }) + await store.proveOwner({ + sessionId: SESSION, + fence: tuiFence, + link: { + linkId: 'tui-link', + handle: { provider: 'codex', threadId: THREAD }, + origin: 'created', + mintedAtFence: tuiFence, + observedAt: NOW + }, + now: NOW + }) + await recoverStoredDeadTuiOwnerForHandoff(store, { + sessionId: SESSION, + expectedFence: tuiFence, + operationId: OPERATION, + probe: { outcome: 'pid-absent' }, + now: NOW + }) + const journal = await openAgentSessionJournal({ + identity: { + sessionId: SESSION, + workspaceId: 'workspace-1', + hostId: 'local', + agent: 'codex', + providerHandle: { kind: 'codex', threadId: THREAD } + }, + journalDir: join(root, 'journal') + }) + const closeTuiOwner = + vi.fn>() + const coordinator = new StructuredAgentSessionHandoffCoordinator({ + store, + claimKeyId: 'key-1', + transport: { + hostLabel: 'Test host', + launchTui: vi.fn(), + reproveTuiOwner: vi.fn(), + recoverTuiOwner: vi.fn(), + stopRecoveredOwner: vi.fn(), + closeTuiOwner, + waitForTuiExit: vi.fn(), + waitForTuiIdleOrExit: vi.fn(), + tuiStatus: () => 'busy' + }, + session: () => ({ journal, fence: store.getRecord(SESSION)?.lease.runtimeFence ?? 1 }), + suspendNative: vi.fn(), + acquireNative: async ({ fence, spawnToken }) => { + await store.commitProcessIdentity({ + sessionId: SESSION, + fence, + process: { + hostId: 'local', + pid: 4300, + processStartTimeMs: NOW, + spawnToken + }, + now: NOW + }) + return store.proveOwner({ + sessionId: SESSION, + fence, + link: { + linkId: 'native-link', + handle: { provider: 'codex', threadId: THREAD }, + origin: 'resumed', + mintedAtFence: fence, + observedAt: NOW + }, + now: NOW + }) + }, + acquireNativeStop: vi.fn(async () => true), + importTuiHistory: vi.fn(), + publish: vi.fn(), + schedule: async (_sessionId, task) => task(), + now: () => NOW + }) + const fields = { + direction: 'to-native' as const, + mode: 'now' as const, + action: 'retry' as const + } + const request: AgentSessionHandoffRequest = { + envelope: { + sessionId: SESSION, + clientOperationId: OPERATION, + expectedRuntimeFence: store.getRecord(SESSION)?.lease.runtimeFence ?? null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.requestHandoff', + sessionId: SESSION, + fields + }) + }, + ...fields + } + + expect(coordinator.status(SESSION)).toMatchObject({ phase: 'failed', owner: 'tui' }) + expect( + await ( + coordinator as { + request: (callerKey: string, params: AgentSessionHandoffRequest) => Promise + } + ).request('client-1', request) + ).toMatchObject({ ok: true }) + await vi.waitFor(() => expect(coordinator.status(SESSION).owner).toBe('native')) + // Settle the flow's trailing outcome write before afterEach removes the store root. + await coordinator.drain() + expect(closeTuiOwner).not.toHaveBeenCalled() + expect(store.getRecord(SESSION)?.lease).toMatchObject({ + runtimeKind: 'native', + claimStatus: 'live', + handoffStage: null + }) + }) +}) diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts index a5927dc0c14..5cd888cc5cc 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-recovery-exits.test.ts @@ -8,7 +8,7 @@ import { spawnProcess } from '../../../shared/child-process/run-process' import { CODEX_SPAWN_TOKEN_ENV } from '../../codex/codex-structured-owner-identity' import { AgentSessionRecordStore } from '../../runtime/agent-session-record-store' import { readProcessStartTimeMs } from '../../runtime/agent-session-process-identity-probe' -import { createStructuredAgentSessionOwnerProbe } from '../../runtime/structured-agent-session-runtime' +import { createStructuredAgentSessionOwnerProbe } from '../../runtime/structured-agent-session-owner-probe' import type { StructuredAgentSessionAdapter } from './structured-agent-session-adapter' import { StructuredAgentSessionHost } from './structured-agent-session-host' import type { StructuredAgentSessionHostDeps } from './structured-agent-session-host-types' diff --git a/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts b/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts index 9d16e19a7f7..26ad85b5cfb 100644 --- a/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts +++ b/src/main/native-chat/agent-session-wire/structured-agent-session-turns-prompt.ts @@ -1,6 +1,11 @@ import { parseAgentJournalItemKey } from '../../../shared/agent-session-journal-item-key' +import { + decodeAgentSessionQuestionAnswers, + isValidAgentSessionQuestionAnswers +} from '../../../shared/agent-session-question-answer' import type { AgentJournalItemBody, + AgentJournalQuestion, AgentJournalResolution } from '../../../shared/agent-session-journal-types' import type { AgentSessionPromptResult } from '../../../shared/agent-session-wire' @@ -14,6 +19,7 @@ function invalid(message: string): TurnOutcome { function promptBodyOf(body: AgentJournalItemBody): { options: readonly { id: string }[] freeTextQuestionId?: string + questions?: AgentJournalQuestion[] resolution: AgentJournalResolution } | null { return body.kind === 'approval' || body.kind === 'question' ? body : null @@ -64,7 +70,19 @@ export async function performPrompt( prompt.freeTextQuestionId !== undefined && freeText?.questionId === prompt.freeTextQuestionId && freeText.answer.trim().length > 0 - if (!acceptsFreeText && !prompt.options.some((option) => option.id === input.optionId)) { + const grouped = + item.body.kind === 'question' && prompt.questions + ? decodeAgentSessionQuestionAnswers(input.optionId) + : null + const acceptsGrouped = + grouped !== null && + prompt.questions !== undefined && + isValidAgentSessionQuestionAnswers(prompt.questions, grouped) + if ( + !acceptsFreeText && + !acceptsGrouped && + !prompt.options.some((option) => option.id === input.optionId) + ) { return invalid(`Option ${input.optionId} is not offered by item ${input.itemId}.`) } const identity = parseAgentJournalItemKey(input.itemId) diff --git a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts index 804d2900a22..60f34707ac9 100644 --- a/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts +++ b/src/main/native-chat/agent-session-wire/unhandled-provider-frame.ts @@ -54,7 +54,8 @@ function directReadableMessage(payload: unknown): string | null { return null } -function readableMessage(payload: unknown): string | null { +/** The provider's own sentence for a frame, when it carries one. */ +export function readableProviderFrameText(payload: unknown): string | null { const direct = directReadableMessage(payload) if (direct || typeof payload !== 'object' || payload === null || Array.isArray(payload)) { return direct @@ -89,7 +90,7 @@ export function unhandledProviderFrameJournalItem( // Why: the opcode alone ("codex · notification:warning") tells the user nothing // and reads as protocol noise. Lead with the provider's own sentence when it has // one; the raw frame stays behind the row's disclosure either way. - const message = readableMessage(payload) + const message = readableProviderFrameText(payload) const display = message ? boundInlineText(message, limits) : null return { body: { diff --git a/src/main/native-chat/claude-structured-managed-account-support.test.ts b/src/main/native-chat/claude-structured-managed-account-support.test.ts new file mode 100644 index 00000000000..f647579d03e --- /dev/null +++ b/src/main/native-chat/claude-structured-managed-account-support.test.ts @@ -0,0 +1,154 @@ +import { describe, expect, it } from 'vitest' +import { getSelectedClaudeAccountIdForTarget } from '../claude-accounts/runtime-selection' +import { + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from './claude-structured-managed-account-support' + +function account(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +function settings( + overrides: Partial +): ClaudeManagedAccountGateSettings { + return { claudeManagedAccounts: [], activeClaudeManagedAccountId: null, ...overrides } +} + +describe('structuredClaudeMatchesActiveManagedAccount', () => { + it('allows an unmanaged install, where nothing claims an identity', () => { + expect(structuredClaudeMatchesActiveManagedAccount(settings({}))).toBe(true) + }) + + it('allows a selected host account, which the runtime syncs into the ambient config', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } + }) + ) + ).toBe(true) + }) + + it('refuses a WSL-only managed account, which never reaches the ambient config', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } + }) + ) + ).toBe(false) + }) + + it('refuses when a host selection names an account that is WSL-bound or missing', () => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountIdsByRuntime: { host: 'wsl-1', wsl: {} } + }) + ) + ).toBe(false) + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountIdsByRuntime: { host: 'gone', wsl: {} } + }) + ) + ).toBe(false) + }) + + /** Absent and empty are the same answer: this user has no managed Claude accounts, so nothing + * claims an identity and the ambient path is legitimate. Only settings that cannot be READ are + * unknown. Treating a missing key as unknown strands profiles that simply never wrote it — the + * auth policy's own predicate takes `(accounts ?? [])` for exactly this reason. */ + it('treats an absent account list the same as an empty one', () => { + expect( + structuredClaudeMatchesActiveManagedAccount(settings({ claudeManagedAccounts: [] })) + ).toBe(true) + expect( + structuredClaudeMatchesActiveManagedAccount({ + activeClaudeManagedAccountId: null + } as unknown as ClaudeManagedAccountGateSettings) + ).toBe(true) + }) + + it('fails closed when the settings cannot be read at all', () => { + expect(structuredClaudeMatchesActiveManagedAccount(null)).toBe(false) + expect(structuredClaudeMatchesActiveManagedAccount(undefined)).toBe(false) + }) + + /** The four states this gate exists to tell apart, pinned together so a change to one is visible + * against the others. */ + it.each([ + ['no managed accounts', [], null, true], + ['accounts present, none active, no WSL account', [account('host-1', 'host')], null, true], + ['host account selected', [account('host-1', 'host')], 'host-1', true], + ['WSL-only, normalized to no host selection', [account('wsl-1', 'wsl')], null, false] + ] as const)('resolves %s', (_name, claudeManagedAccounts, activeId, expected) => { + expect( + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: [...claudeManagedAccounts], + activeClaudeManagedAccountIdsByRuntime: { host: activeId, wsl: {} } + }) + ) + ).toBe(expected) + }) + + /** THE discriminator, and the whole of this rule. With nothing selected for the host runtime the + * settings alone cannot distinguish honest deselection from the WSL-only steady state, because + * `pruneInvalidClaudeRuntimeSelection` empties the host slot in the second case and persists it. + * So the presence of ANY WSL-bound account decides. Simplifying this to "none active -> + * supported" re-opens the auth-identity misrepresentation this gate exists to prevent. */ + it('splits none-active on whether a WSL-bound account exists at all', () => { + const noneActive = (accounts: ReturnType[]) => + structuredClaudeMatchesActiveManagedAccount( + settings({ + claudeManagedAccounts: accounts, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } + }) + ) + + expect(noneActive([account('host-1', 'host')])).toBe(true) + expect(noneActive([account('host-1', 'host'), account('host-2', 'host')])).toBe(true) + expect(noneActive([account('wsl-1', 'wsl')])).toBe(false) + // Mixed list still refuses: the WSL account is present and nothing is selected. + expect(noneActive([account('host-1', 'host'), account('wsl-1', 'wsl')])).toBe(false) + }) + + /** The gate and the auth policy must resolve the SAME account. A legacy settings blob carries the + * selection only in the flat `activeClaudeManagedAccountId`, which is where the accessor's + * fall-through lives — reading the runtime map directly silently disagrees with the policy. */ + it('resolves the same account as the auth policy on a legacy flat selection', () => { + const legacy = settings({ + claudeManagedAccounts: [account('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1' + }) + + expect(getSelectedClaudeAccountIdForTarget(legacy, { runtime: 'host' })).toBe('host-1') + expect(structuredClaudeMatchesActiveManagedAccount(legacy)).toBe(true) + }) + + it('agrees with the auth policy that a legacy flat WSL selection is refused', () => { + const legacy = settings({ + claudeManagedAccounts: [account('wsl-1', 'wsl')], + activeClaudeManagedAccountId: 'wsl-1' + }) + + expect(getSelectedClaudeAccountIdForTarget(legacy, { runtime: 'host' })).toBe('wsl-1') + expect(structuredClaudeMatchesActiveManagedAccount(legacy)).toBe(false) + }) +}) diff --git a/src/main/native-chat/claude-structured-managed-account-support.ts b/src/main/native-chat/claude-structured-managed-account-support.ts new file mode 100644 index 00000000000..dccf6216bda --- /dev/null +++ b/src/main/native-chat/claude-structured-managed-account-support.ts @@ -0,0 +1,61 @@ +import type { GlobalSettings } from '../../shared/global-settings-types' +import { getSelectedClaudeAccountIdForTarget } from '../claude-accounts/runtime-selection' + +export type ClaudeManagedAccountGateSettings = Pick< + GlobalSettings, + | 'claudeManagedAccounts' + | 'activeClaudeManagedAccountId' + | 'activeClaudeManagedAccountIdsByRuntime' +> + +/** + * A structured Claude session launches against the ambient Claude config, which the account service + * keeps in sync with the selected HOST account. A WSL-bound managed account lives inside the distro + * and is never synced there, so such a session would authenticate as whatever the ambient identity + * happens to be while the UI names the WSL account — the user is told one identity and given + * another. Refuse the structured path there and let the terminal-backed one, which resolves the + * account per runtime, handle that account shape. + * + * Reads the selection through the same accessor the auth policy uses. Resolving it any other way + * lets the two disagree, and a session admitted by this gate would then run under a policy computed + * from a different account than the one approved here. + * + * Unknown answers refuse, and only genuinely unknown ones: settings that cannot be read at all, or + * an active selection this cannot resolve. An install with no managed accounts — the list empty or + * never written — claims no identity and is fine. + */ +export function structuredClaudeMatchesActiveManagedAccount( + settings: ClaudeManagedAccountGateSettings | null | undefined +): boolean { + if (!settings) { + return false + } + // Absent is the same answer as empty — this user has no managed Claude accounts, so nothing + // claims an identity and ambient auth is the truth. Only settings that cannot be READ are + // unknown, and those refuse above. The auth policy reads the list the same way. + const accounts = settings.claudeManagedAccounts ?? [] + if (accounts.length === 0) { + return true + } + const activeHostId = getSelectedClaudeAccountIdForTarget(settings, { runtime: 'host' }) + if (!activeHostId) { + // Nothing selected for the host runtime is two different states that the settings cannot tell + // apart after the fact: honest deselection, where ambient auth is the truth and the UI names no + // identity, and the WSL-only case, where the prune emptied the host slot and persisted null + // while the UI still names the WSL account. The presence of any WSL-bound account decides. + return !accounts.some((candidate) => candidate.managedAuthRuntime === 'wsl') + } + const active = accounts.find((candidate) => candidate.id === activeHostId) + return active ? active.managedAuthRuntime !== 'wsl' : false +} + +/** Reads the gate's settings, answering null when they cannot be read so callers refuse. */ +export function readClaudeManagedAccountGateSettings( + getSettings: () => ClaudeManagedAccountGateSettings +): ClaudeManagedAccountGateSettings | null { + try { + return getSettings() + } catch { + return null + } +} diff --git a/src/main/native-chat/session-file-resolver-claude-roots.test.ts b/src/main/native-chat/session-file-resolver-claude-roots.test.ts new file mode 100644 index 00000000000..87ebd570280 --- /dev/null +++ b/src/main/native-chat/session-file-resolver-claude-roots.test.ts @@ -0,0 +1,97 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const scanned = vi.hoisted(() => ({ dirs: [] as string[], hits: {} as Record })) +vi.mock('../ai-vault/session-scanner-discovery', () => ({ + walkSessionFiles: async (dir: string) => { + scanned.dirs.push(dir) + const hit = scanned.hits[dir] + return hit ? [hit] : [] + } +})) + +import { homedir } from 'node:os' +import { join } from 'node:path' +import { resolveSessionFilePath } from './session-file-resolver' + +const DEFAULT_ROOT = join(homedir(), '.claude', 'projects') +const CONFIG_DIR = '/opt/claude-home' +const CONFIG_ROOT = join(CONFIG_DIR, 'projects') + +let previousConfigDir: string | undefined + +beforeEach(() => { + previousConfigDir = process.env.CLAUDE_CONFIG_DIR + scanned.dirs = [] + scanned.hits = {} +}) + +afterEach(() => { + if (previousConfigDir === undefined) { + delete process.env.CLAUDE_CONFIG_DIR + } else { + process.env.CLAUDE_CONFIG_DIR = previousConfigDir + } +}) + +/** + * Honouring CLAUDE_CONFIG_DIR fixed new sessions but would otherwise hide every + * transcript written before the user adopted the variable. The Codex resolver in this + * same file already searches managed-then-default and de-dupes; Claude does the same. + */ +describe('claude transcript roots', () => { + it('searches the config-dir root first, then the default home', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([CONFIG_ROOT, DEFAULT_ROOT]) + }) + + it('still finds history written before CLAUDE_CONFIG_DIR was adopted', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + const legacy = join(DEFAULT_ROOT, '-repos-old', 'session-1.jsonl') + scanned.hits[DEFAULT_ROOT] = legacy + + await expect(resolveSessionFilePath('claude', 'session-1')).resolves.toBe(legacy) + }) + + it('prefers the config-dir root when both hold the session', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + scanned.hits[CONFIG_ROOT] = join(CONFIG_ROOT, '-repos-new', 'session-1.jsonl') + scanned.hits[DEFAULT_ROOT] = join(DEFAULT_ROOT, '-repos-old', 'session-1.jsonl') + + await expect(resolveSessionFilePath('claude', 'session-1')).resolves.toBe( + scanned.hits[CONFIG_ROOT] + ) + // The default root is never reached, so the common case pays for one scan. + expect(scanned.dirs).toEqual([CONFIG_ROOT]) + }) + + it('scans one root when the variable is unset', async () => { + delete process.env.CLAUDE_CONFIG_DIR + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([DEFAULT_ROOT]) + }) + + it('de-dupes when CLAUDE_CONFIG_DIR names the default home', async () => { + process.env.CLAUDE_CONFIG_DIR = join(homedir(), '.claude') + + await resolveSessionFilePath('claude', 'session-1') + + expect(scanned.dirs).toEqual([DEFAULT_ROOT]) + }) + + it('honours an explicit root override without adding fallbacks', async () => { + process.env.CLAUDE_CONFIG_DIR = CONFIG_DIR + // The account-home callers (structured-claude-runtime-adapter, the host handoff) + // know the exact tree their session pinned; a fallback there could resolve a + // different account's transcript. + await resolveSessionFilePath('claude', 'session-1', { + claudeProjectsDir: '/accounts/pinned/projects' + }) + + expect(scanned.dirs).toEqual(['/accounts/pinned/projects']) + }) +}) diff --git a/src/main/native-chat/session-file-resolver.test.ts b/src/main/native-chat/session-file-resolver.test.ts index 584d8a25a9d..04946f5b464 100644 --- a/src/main/native-chat/session-file-resolver.test.ts +++ b/src/main/native-chat/session-file-resolver.test.ts @@ -1,9 +1,12 @@ import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' import { tmpdir } from 'node:os' import { dirname, join } from 'node:path' -import { afterEach, describe, expect, it } from 'vitest' +import { afterEach, describe, expect, it, vi } from 'vitest' -import { ClaudeTranscriptTailIncompleteError } from '../claude/claude-transcript-branch-proof' +import { + ClaudeTranscriptTailIncompleteError, + readClaudeTranscriptLeafWithReproof +} from '../claude/claude-transcript-branch-proof' import { readClaudeTranscriptLeafUuid, resolveSessionFilePath } from './session-file-resolver' let tempRoots: string[] = [] @@ -139,6 +142,263 @@ describe('resolveSessionFilePath', () => { ) }) + it('rejects non-transcript and sidechain UUIDs as the durable leaf', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-leaf-filter-') + const transcript = join(root, 'session.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'sidechain-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + isSidechain: true + }, + { type: 'result', uuid: 'result-frame', parentUuid: 'main-user', sessionId: 'session-1' }, + { + type: 'system', + subtype: 'init', + uuid: 'init-frame', + parentUuid: null, + sessionId: 'session-1' + }, + { type: 'stream_event', uuid: 'stream-frame', parentUuid: null, sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'sidechain-assistant', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'marker leaf is missing from the session graph' + ) + }) + + it('rejects a main leaf whose ancestry crosses a subagent sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-sidechain-ancestry-') + const transcript = join(root, 'session.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'sidechain-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + isSidechain: true + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'sidechain-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'not on the main transcript' + ) + }) + + it('rejects a main leaf whose ancestry crosses a parent-tool-use sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-ancestry-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1')).rejects.toThrow( + 'not on the main transcript' + ) + }) + + it('rejects a previous cursor descended from a parent-tool-use sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-cursor-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'main-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect( + readClaudeTranscriptLeafUuid(transcript, 'session-1', 'main-after-sidechain') + ).rejects.toThrow('not on the main transcript') + }) + + it('rejects a latest marker descended from a parent-tool-use cursor sidechain', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-parent-tool-cursor-descendant-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'main-user', parentUuid: null, sessionId: 'session-1' }, + { + type: 'assistant', + uuid: 'subagent-assistant', + parentUuid: 'main-user', + sessionId: 'session-1', + parent_tool_use_id: 'tool-use-1' + }, + { + type: 'assistant', + uuid: 'main-after-sidechain', + parentUuid: 'subagent-assistant', + sessionId: 'session-1' + }, + { + type: 'assistant', + uuid: 'latest-after-sidechain', + parentUuid: 'main-after-sidechain', + sessionId: 'session-1' + }, + { type: 'last-prompt', leafUuid: 'latest-after-sidechain', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect( + readClaudeTranscriptLeafUuid(transcript, 'session-1', 'main-after-sidechain') + ).rejects.toThrow('not on the main transcript') + }) + + it('rejects a post-snapshot descendant whose parent row was observed later', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-post-snapshot-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { + type: 'assistant', + uuid: 'descendant', + parentUuid: 'previous', + sessionId: 'session-1' + }, + { type: 'assistant', uuid: 'previous', parentUuid: null, sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'descendant', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1', 'previous')).rejects.toThrow( + 'parent row follows descendant' + ) + }) + + it('does not re-prove a divergent sibling after the sampled cursor rejects', async () => { + const root = await makeRoot('orca-native-chat-resolve-claude-sibling-reproof-') + const transcript = join(root, 'transcript.jsonl') + await writeFile( + transcript, + [ + { type: 'user', uuid: 'root', parentUuid: null, sessionId: 'session-1' }, + { type: 'assistant', uuid: 'old', parentUuid: 'root', sessionId: 'session-1' }, + { type: 'assistant', uuid: 'new', parentUuid: 'root', sessionId: 'session-1' }, + { type: 'last-prompt', leafUuid: 'new', sessionId: 'session-1' } + ] + .map((record) => JSON.stringify(record)) + .join('\n'), + 'utf8' + ) + const calls: (string | null)[] = [] + const readTranscriptLeaf = async ({ + previousLeafUuid + }: { + previousLeafUuid: string | null + }) => { + calls.push(previousLeafUuid) + return readClaudeTranscriptLeafUuid(transcript, 'session-1', previousLeafUuid) + } + + await expect(readClaudeTranscriptLeafUuid(transcript, 'session-1', 'old')).rejects.toThrow( + 'sibling branch' + ) + + await expect( + readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf, + claudeConfigDir: '/accounts/claude', + providerSessionId: 'session-1', + previousLeafUuid: 'old' + }) + ).rejects.toThrow('sibling branch') + expect(calls).toEqual(['old']) + }) + + it('does not accept a divergent sibling after a truncated-tail reproof', async () => { + const calls: (string | null)[] = [] + const readTranscriptLeaf = vi.fn( + async ({ previousLeafUuid }: { previousLeafUuid: string | null }) => { + calls.push(previousLeafUuid) + if (calls.length === 1) { + throw new ClaudeTranscriptTailIncompleteError() + } + return 'divergent-sibling' + } + ) + + await expect( + readClaudeTranscriptLeafWithReproof({ + readTranscriptLeaf, + claudeConfigDir: '/accounts/claude', + providerSessionId: 'session-1', + previousLeafUuid: 'old' + }) + ).rejects.toBeInstanceOf(ClaudeTranscriptTailIncompleteError) + expect(calls).toEqual(['old']) + }) + it('globs Claude project subdirs for .jsonl', async () => { const root = await makeRoot('orca-native-chat-resolve-claude-') const claudeProjectsDir = join(root, 'claude-projects') @@ -410,3 +670,42 @@ describe('resolveSessionFilePath', () => { expect(resolved).toBe(target) }) }) + +// Mobile native chat resolves with no root override (transcript-read-cache.ts:104), +// while the account home a structured Claude session pins is +// `CLAUDE_CONFIG_DIR || ~/.claude` (runtime-paths.ts:15). When the two disagree the +// CLI writes one place and mobile reads another, and the chat goes dark with no +// wire-level error — so the default root has to honour the same variable. +describe('the default Claude transcript root mobile falls back to', () => { + it('follows CLAUDE_CONFIG_DIR, the same variable the pinned account home follows', async () => { + const configDir = await makeRoot('orca-native-chat-claude-config-dir-') + const slugDir = join(configDir, 'projects', '-repos-workspace-1') + await mkdir(slugDir, { recursive: true }) + const transcript = join(slugDir, 'session-under-config-dir.jsonl') + await writeFile(transcript, '', 'utf8') + const previous = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = configDir + + try { + // No `claudeProjectsDir` override: exactly the call mobile makes. + await expect(resolveSessionFilePath('claude', 'session-under-config-dir')).resolves.toBe( + transcript + ) + } finally { + restoreEnv('CLAUDE_CONFIG_DIR', previous) + } + }) + + it('ignores a blank CLAUDE_CONFIG_DIR rather than resolving against the filesystem root', async () => { + const previous = process.env.CLAUDE_CONFIG_DIR + process.env.CLAUDE_CONFIG_DIR = ' ' + + try { + await expect( + resolveSessionFilePath('claude', 'session-that-does-not-exist') + ).resolves.toBeNull() + } finally { + restoreEnv('CLAUDE_CONFIG_DIR', previous) + } + }) +}) diff --git a/src/main/native-chat/session-file-resolver.ts b/src/main/native-chat/session-file-resolver.ts index 0ee73fddc8d..12d2e615742 100644 --- a/src/main/native-chat/session-file-resolver.ts +++ b/src/main/native-chat/session-file-resolver.ts @@ -31,8 +31,20 @@ import { proveClaudeTranscriptBranch } from '../claude/claude-transcript-branch- // the remote main resolves its local home, so we never hardcode an absolute // user path — homedir()/CODEX_HOME resolution stays runtime-relative and is // computed per call (not at module load) so it tracks the live home. -function claudeProjectsDir(): string { - return join(homedir(), '.claude', 'projects') +// Why CLAUDE_CONFIG_DIR and not just homedir(): a structured Claude session pins its +// account home to `CLAUDE_CONFIG_DIR || ~/.claude` (claude-accounts/runtime-paths.ts), +// and the CLI writes its transcript under whatever home it was given. Mobile native chat +// resolves with no root override, so a default that ignored the variable read a different +// tree than the CLI wrote — a silent blackout, not an error. +// Why both roots and not just that one: adopting the variable would otherwise hide every +// transcript written before it was set. Same managed-then-default shape as +// codexSessionsDirs below, de-duped so the usual case still scans once. +function claudeProjectsDirs(): string[] { + const candidates = [ + join(process.env.CLAUDE_CONFIG_DIR?.trim() || join(homedir(), '.claude'), 'projects'), + join(homedir(), '.claude', 'projects') + ] + return candidates.filter((dir, index) => candidates.indexOf(dir) === index) } // Why: Orca launches Codex with ORCA_CODEX_HOME pointing at its own managed @@ -173,9 +185,11 @@ async function resolveSessionFileById( } if (transcriptAgent === 'claude') { + // An explicit root is the caller naming the exact account tree its session pinned; + // adding a fallback there could resolve a different account's transcript. return resolveClaudeSessionFile( trimmedId, - options.claudeProjectsDir ?? claudeProjectsDir(), + options.claudeProjectsDir ? [options.claudeProjectsDir] : claudeProjectsDirs(), signal ) } @@ -205,16 +219,22 @@ async function resolveSessionFileById( async function resolveClaudeSessionFile( sessionId: string, - projectsDir: string, + projectsDirs: readonly string[], signal?: AbortSignal ): Promise { const targetName = `${sessionId}.jsonl` - const files = await walkSessionFiles(projectsDir, 'claude', [], { - extensions: new Set(['.jsonl']), - filePredicate: (path) => basename(path) === targetName, - signal - }) - return files[0] ?? null + for (const projectsDir of projectsDirs) { + // No existence pre-check: walkSessionFiles already yields [] for a missing root. + const files = await walkSessionFiles(projectsDir, 'claude', [], { + extensions: new Set(['.jsonl']), + filePredicate: (path) => basename(path) === targetName, + signal + }) + if (files[0]) { + return files[0] + } + } + return null } async function resolveCodexSessionFile( diff --git a/src/main/native-chat/structured-agent-session-create-support.test.ts b/src/main/native-chat/structured-agent-session-create-support.test.ts new file mode 100644 index 00000000000..ab19d369bac --- /dev/null +++ b/src/main/native-chat/structured-agent-session-create-support.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, it } from 'vitest' +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import type { ClaudeManagedAccountGateSettings } from './claude-structured-managed-account-support' +import { resolveStructuredAgentSessionCreateSupport } from './structured-agent-session-create-support' + +const LOCAL: AgentSessionExecutionLocation = { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' +} + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +const WSL_ONLY: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +function support( + overrides: Partial[0]> = {} +) { + return resolveStructuredAgentSessionCreateSupport({ + agent: 'claude', + location: LOCAL, + adapterSupportsCreate: true, + getSettings: () => HOST_SELECTED, + ...overrides + }) +} + +describe('resolveStructuredAgentSessionCreateSupport', () => { + it('supports Claude under a selected host account', () => { + expect(support()).toEqual({ supported: true }) + }) + + it('refuses Claude under a WSL-only managed account', () => { + expect(support({ getSettings: () => WSL_ONLY })).toEqual({ supported: false, reason: 'wsl' }) + }) + + it('fails closed for Claude when the settings throw', () => { + expect( + support({ + getSettings: () => { + throw new Error('no store') + } + }) + ).toEqual({ supported: false, reason: 'wsl' }) + }) + + it('leaves Codex to the adapter answer under the same WSL-only account', () => { + expect(support({ agent: 'codex', getSettings: () => WSL_ONLY })).toEqual({ supported: true }) + }) + + it.each([ + ['remote', { ...LOCAL, executionHostId: 'ssh:host-a' }, 'remote'], + ['wsl workspace', { ...LOCAL, wslDistro: 'Ubuntu' }, 'wsl'], + ['unsupported agent', LOCAL, 'agent'] + ] as const)('keeps the adapter refusal reason for %s', (_name, location, reason) => { + expect(support({ adapterSupportsCreate: false, location })).toEqual({ + supported: false, + reason + }) + }) +}) diff --git a/src/main/native-chat/structured-agent-session-create-support.ts b/src/main/native-chat/structured-agent-session-create-support.ts new file mode 100644 index 00000000000..9b96a1af4be --- /dev/null +++ b/src/main/native-chat/structured-agent-session-create-support.ts @@ -0,0 +1,48 @@ +import type { AgentSessionExecutionLocation } from '../../shared/agent-session-record' +import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' +import { + readClaudeManagedAccountGateSettings, + structuredClaudeMatchesActiveManagedAccount, + type ClaudeManagedAccountGateSettings +} from './claude-structured-managed-account-support' + +export type StructuredAgentSessionCreateSupport = { + supported: boolean + reason?: 'agent' | 'remote' | 'wsl' +} + +/** + * The create-support verdict, kept out of the runtime class file because that file is `@ts-nocheck` + * — a call site there is not typechecked, so an auth-identity decision written inline would compile + * however wrong it was. The runtime hands over the two facts it owns and this decides. + */ +export function resolveStructuredAgentSessionCreateSupport(input: { + agent: 'claude' | 'codex' + location: AgentSessionExecutionLocation + adapterSupportsCreate: boolean + getSettings: () => ClaudeManagedAccountGateSettings +}): StructuredAgentSessionCreateSupport { + if (!input.adapterSupportsCreate) { + return { + supported: false, + reason: + input.location.executionHostId !== LOCAL_EXECUTION_HOST_ID + ? 'remote' + : input.location.wslDistro + ? 'wsl' + : 'agent' + } + } + // Claude only: Codex resolves its account on a different path, so its answer is untouched here. + // `wsl` is the closest existing reason — the cause is a WSL-bound account rather than a WSL + // workspace — and no client reads the field, so it stays as-is. + if ( + input.agent === 'claude' && + !structuredClaudeMatchesActiveManagedAccount( + readClaudeManagedAccountGateSettings(input.getSettings) + ) + ) { + return { supported: false, reason: 'wsl' } + } + return { supported: true } +} diff --git a/src/main/providers/windows-foreground-process-rows.ts b/src/main/providers/windows-foreground-process-rows.ts index 5f462649e6c..e8320a6d00a 100644 --- a/src/main/providers/windows-foreground-process-rows.ts +++ b/src/main/providers/windows-foreground-process-rows.ts @@ -108,6 +108,26 @@ export async function queryWindowsPaneProcessInventory( } } +/** + * The descendant walk over rows the caller already read. + * + * Why exported: a caller that needs a field this module's projection drops — + * process creation time, for a PID-reuse-safe teardown snapshot — would + * otherwise read the whole table a second time to get it. + * Null when the root is absent, which is a stale or filtered snapshot rather + * than a root with no descendants. + */ +export function windowsDescendantsFromRows( + rows: Row[], + rootPid: number +): (Row & { depth: number })[] | null { + const index = getProcessTableIndex(rows) + if (!index.byPid.has(rootPid)) { + return null + } + return collectDescendantsFromIndex(index, rootPid).sort((a, b) => b.depth - a.depth) +} + /** Test-only: clear the shared snapshot so one case's rows never serve the next. */ export function resetWindowsProcessRowsSnapshotForTests(): void { resetWindowsProcessTableForTests() diff --git a/src/main/pty-descendant-exit-verification.ts b/src/main/pty-descendant-exit-verification.ts index c0ed223704a..4c8471fa955 100644 --- a/src/main/pty-descendant-exit-verification.ts +++ b/src/main/pty-descendant-exit-verification.ts @@ -21,50 +21,143 @@ function waitForDelay(ms: number): Promise { function matchingSnapshotRows( snapshot: DescendantSnapshot, - table: readonly ProcessTableRow[] + table: readonly ProcessTableRow[], + rejectDuplicatePids = false ): ProcessTableRow[] { const expected = new Map(snapshot.descendants.map((row) => [row.pid, row])) - return table.filter((live) => { - const row = expected.get(live.pid) - return row?.startedAt === live.startedAt && row.pgid === live.pgid + const rowsByPid = new Map() + for (const live of table) { + const rows = rowsByPid.get(live.pid) + if (rows) { + rows.push(live) + } else { + rowsByPid.set(live.pid, [live]) + } + } + return [...expected.entries()].flatMap(([pid, row]) => { + const rows = rowsByPid.get(pid) + if (rejectDuplicatePids && rows?.length !== 1) { + // Duplicate PID rows make this non-atomic process-table read ambiguous; + // never signal or count either identity as proof of liveness. + return [] + } + return (rows ?? []).filter((live) => live.startedAt === row.startedAt && live.pgid === row.pgid) }) } +function hasDuplicateSnapshotPids( + snapshot: DescendantSnapshot, + table: readonly ProcessTableRow[] +): boolean { + const expected = new Set(snapshot.descendants.map((row) => row.pid)) + const counts = new Map() + for (const live of table) { + if (expected.has(live.pid)) { + counts.set(live.pid, (counts.get(live.pid) ?? 0) + 1) + } + } + return [...counts.values()].some((count) => count > 1) +} + type VerificationDeps = TerminateDeps & { verifyMs?: number + /** Revalidate identities before signaling; used by Claude's close proof. */ + requireIdentityBeforeSignal?: boolean } +/** + * Orca's verdict vocabulary for a snapshotted tree, with no synonyms: `live` is + * an identity-matched descendant still observed at the deadline; `unverifiable` + * is a table that could not be read, which is never evidence either way. + */ +export type DescendantTreeVerdict = 'exited' | 'live' | 'unverifiable' + /** An unreadable process table is never proof that a stopped descendant exited. */ export async function terminateDescendantSnapshotAndWait( snapshot: DescendantSnapshot, deps: VerificationDeps = {} ): Promise { + return (await terminateDescendantSnapshotWithVerdict(snapshot, deps)) === 'exited' +} + +/** Signals the snapshot, then reports what the last table read observed. */ +export async function terminateDescendantSnapshotWithVerdict( + snapshot: DescendantSnapshot, + deps: VerificationDeps = {} +): Promise { const sendSignal = deps.sendSignal ?? sendDescendantSignal const readTable = deps.readTable ?? readProcessTable const graceMs = deps.graceMs ?? DESCENDANT_KILL_GRACE_MS const verifyMs = deps.verifyMs ?? DESCENDANT_KILL_VERIFY_MS const deadline = Date.now() + verifyMs - for (const row of snapshot.descendants) { - sendSignal(row.pid, 'SIGTERM') - } let forced = false + let signalled = !deps.requireIdentityBeforeSignal + let missingObservations = 0 + if (signalled) { + for (const row of snapshot.descendants) { + sendSignal(row.pid, 'SIGTERM') + } + } while (Date.now() < deadline) { const capture = await readProcessTableBeforeDeadline( readTable, deps.timeoutMs ?? DESCENDANT_SNAPSHOT_TIMEOUT_MS ) - if (!capture) { - return false - } - const live = matchingSnapshotRows(snapshot, capture.rows) - if (live.length === 0) { - return true - } - if (!forced && Date.now() >= deadline - verifyMs + graceMs) { - forced = true - for (const row of live) { - if (hasUnambiguousStartIdentity(row, snapshot.capturedAtMs)) { - sendSignal(row.pid, 'SIGKILL') + // A read that missed its own deadline is not an answer, and surrendering on + // the first slow one spends none of the window this verification was given: + // on a loaded host that reported a tree unverifiable without ever seeing it. + if (capture) { + if (deps.requireIdentityBeforeSignal && hasDuplicateSnapshotPids(snapshot, capture.rows)) { + // A duplicate target pid is an ambiguous non-atomic read. Do not signal + // either row and do not turn that uncertainty into an exited verdict. + await waitForDelay(50) + continue + } + const live = matchingSnapshotRows(snapshot, capture.rows, deps.requireIdentityBeforeSignal) + if (live.length === 0) { + // Before a signal has been sent, an empty identity match means the + // snapshotted descendants already exited or were replaced. Signalling + // those old numeric pids would be unsafe. + if (deps.requireIdentityBeforeSignal) { + // A single process-table read can race a fork or return a partial + // view; require two bounded absences before claiming the tree gone. + missingObservations += 1 + if (missingObservations < 2) { + await waitForDelay(50) + continue + } + } + return 'exited' + } + missingObservations = 0 + if (!signalled) { + // Revalidate every identity immediately before the first signal. A PID + // can be recycled between the original walk and close, so never signal + // from the stale snapshot alone. + for (const row of live) { + sendSignal(row.pid, 'SIGTERM') + } + signalled = true + } + if (!forced && Date.now() >= deadline - verifyMs + graceMs) { + forced = true + for (const row of live) { + // A row a walk re-derived from a live root is ours whatever second it + // was born in, which start time alone can never establish for one born + // in its own capture second. Rows no walk re-derived still answer to + // the second-resolution fence, which is all the evidence they have. + // Scoped to the identity-revalidating callers; the same argument holds + // for the rest, but widening it is a deliberate change of its own. + if ( + (deps.requireIdentityBeforeSignal === true && + snapshot.reDerivedPids?.has(row.pid) === true) || + hasUnambiguousStartIdentity( + row, + snapshot.capturedAtMsByPid?.[String(row.pid)] ?? snapshot.capturedAtMs + ) + ) { + sendSignal(row.pid, 'SIGKILL') + } } } } @@ -74,5 +167,19 @@ export async function terminateDescendantSnapshotAndWait( readTable, deps.timeoutMs ?? DESCENDANT_SNAPSHOT_TIMEOUT_MS ) - return finalCapture !== null && matchingSnapshotRows(snapshot, finalCapture.rows).length === 0 + if (!finalCapture) { + return 'unverifiable' + } + if (deps.requireIdentityBeforeSignal && hasDuplicateSnapshotPids(snapshot, finalCapture.rows)) { + return 'unverifiable' + } + const finalLive = matchingSnapshotRows( + snapshot, + finalCapture.rows, + deps.requireIdentityBeforeSignal + ) + if (finalLive.length > 0) { + return 'live' + } + return deps.requireIdentityBeforeSignal && missingObservations < 2 ? 'unverifiable' : 'exited' } diff --git a/src/main/pty-descendant-termination.test.ts b/src/main/pty-descendant-termination.test.ts index e1255a678d8..0c4bea81306 100644 --- a/src/main/pty-descendant-termination.test.ts +++ b/src/main/pty-descendant-termination.test.ts @@ -15,7 +15,10 @@ import { type ProcessTableCapture, type ProcessTableRow } from './pty-descendant-termination' -import { terminateDescendantSnapshotAndWait } from './pty-descendant-exit-verification' +import { + terminateDescendantSnapshotAndWait, + terminateDescendantSnapshotWithVerdict +} from './pty-descendant-exit-verification' const CAPTURED_AT_MS = Date.parse('Tue Jul 14 12:00:00 2026') @@ -53,7 +56,14 @@ function snapshot( rootPgid: number | null = 10, capturedAtMs = CAPTURED_AT_MS ) { - return { rootPgid, descendants, capturedAtMs } + return { + ...(rootPgid === null ? {} : { root: { pid: 10, startedAt: 'Mon Jul 13 12:54:47 2026' } }), + rootPgid, + descendants, + capturedAtMs, + // Everything a walk returns was re-derived by it. + ...(rootPgid === null ? {} : { reDerivedPids: new Set(descendants.map((row) => row.pid)) }) + } } describe('parseProcessTable', () => { @@ -298,6 +308,32 @@ describe('terminateDescendantSnapshot', () => { expect(sendSignal).not.toHaveBeenCalled() expect(vi.getTimerCount()).toBe(0) }) + + it("uses each row's capture boundary when escalating a merged snapshot", async () => { + const oldBoundary = CAPTURED_AT_MS + 900 + const refreshBoundary = CAPTURED_AT_MS + 2_100 + const retained = row(20, 10, 20, 'Tue Jul 14 12:00:00 2026') + const fresh = row(30, 10, 30, 'Tue Jul 14 12:00:01 2026') + const sendSignal = vi.fn() + terminateDescendantSnapshot( + { + ...snapshot([retained, fresh], 10, refreshBoundary), + capturedAtMsByPid: { '20': oldBoundary, '30': refreshBoundary } + }, + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([retained, fresh])) + } + ) + sendSignal.mockClear() + + await vi.advanceTimersByTimeAsync(DESCENDANT_KILL_GRACE_MS) + + // PID 20 was retained from the earlier capture and is still in its + // capture second; PID 30 was newly observed by the refresh and is old + // enough for a bounded forced cleanup. + expect(sendSignal.mock.calls).toEqual([[30, 'SIGKILL']]) + }) }) describe('terminateDescendantSnapshotAndWait', () => { @@ -333,14 +369,114 @@ describe('terminateDescendantSnapshotAndWait', () => { it('does not claim exit when the verification table is unavailable', async () => { const sendSignal = vi.fn() - const result = await terminateDescendantSnapshotAndWait(snapshot([row(20, 10, 20)]), { + const pending = terminateDescendantSnapshotAndWait(snapshot([row(20, 10, 20)]), { sendSignal, - readTable: vi.fn().mockRejectedValue(new Error('ps exploded')) + readTable: vi.fn().mockRejectedValue(new Error('ps exploded')), + verifyMs: 200 }) + await vi.advanceTimersByTimeAsync(400) - expect(result).toBe(false) + await expect(pending).resolves.toBe(false) expect(sendSignal).toHaveBeenCalledWith(20, 'SIGTERM') }) + + it('keeps polling past a read that missed its deadline rather than surrendering', async () => { + const survivor = row(20, 10, 20) + const readTable = vi + .fn() + // A loaded host can miss one read's deadline with the window still open. + .mockRejectedValueOnce(new Error('ps timed out')) + .mockResolvedValueOnce(tableCapture([survivor])) + .mockResolvedValueOnce(tableCapture([])) + const sendSignal = vi.fn() + + const pending = terminateDescendantSnapshotWithVerdict(snapshot([survivor]), { + sendSignal, + readTable, + graceMs: 0, + verifyMs: 2_000 + }) + await vi.advanceTimersByTimeAsync(500) + + await expect(pending).resolves.toBe('exited') + expect(sendSignal.mock.calls).toEqual([ + [20, 'SIGTERM'], + [20, 'SIGKILL'] + ]) + }) + + it('names a survivor seen at the deadline live, never unverifiable', async () => { + const survivor = row(20, 10, 20) + const pending = terminateDescendantSnapshotWithVerdict(snapshot([survivor]), { + sendSignal: vi.fn(), + readTable: vi.fn().mockResolvedValue(tableCapture([survivor])), + graceMs: 0, + verifyMs: 100 + }) + await vi.advanceTimersByTimeAsync(200) + + await expect(pending).resolves.toBe('live') + }) + + it('names an unreadable verification table unverifiable', async () => { + const pending = terminateDescendantSnapshotWithVerdict(snapshot([row(20, 10, 20)]), { + sendSignal: vi.fn(), + readTable: vi.fn().mockRejectedValue(new Error('ps exploded')), + verifyMs: 200 + }) + await vi.advanceTimersByTimeAsync(400) + + await expect(pending).resolves.toBe('unverifiable') + }) + + it('does not signal a recycled descendant when identity validation is required', async () => { + const sendSignal = vi.fn() + const recycled = row(20, 10, 20, 'Tue Jul 14 13:00:00 2026') + const pending = terminateDescendantSnapshotWithVerdict( + snapshot([row(20, 10, 20, 'Tue Jul 14 12:00:00 2026')]), + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([recycled])), + requireIdentityBeforeSignal: true, + verifyMs: 100 + } + ) + + await vi.advanceTimersByTimeAsync(200) + await expect(pending).resolves.toBe('exited') + expect(sendSignal).not.toHaveBeenCalled() + }) + + it('uses row-scoped boundaries for forced cleanup in the exit verifier', async () => { + const oldBoundary = CAPTURED_AT_MS + 900 + const refreshBoundary = CAPTURED_AT_MS + 2_100 + const retained = row(20, 10, 20, 'Tue Jul 14 12:00:00 2026') + const fresh = row(30, 10, 30, 'Tue Jul 14 12:00:01 2026') + const sendSignal = vi.fn() + const pending = terminateDescendantSnapshotWithVerdict( + { + ...snapshot([retained, fresh], 10, refreshBoundary), + capturedAtMsByPid: { '20': oldBoundary, '30': refreshBoundary }, + // What a merge produces: only the refresh re-derived 30; 20 is retained. + reDerivedPids: new Set([30]) + }, + { + sendSignal, + readTable: vi.fn().mockResolvedValue(tableCapture([retained, fresh])), + requireIdentityBeforeSignal: true, + graceMs: 0, + verifyMs: 100 + } + ) + await vi.advanceTimersByTimeAsync(200) + + await expect(pending).resolves.toBe('live') + expect(sendSignal.mock.calls).toEqual([ + [20, 'SIGTERM'], + [30, 'SIGTERM'], + [30, 'SIGKILL'] + ]) + }) }) describe('createProcessTableSnapshotReader', () => { diff --git a/src/main/pty-descendant-termination.ts b/src/main/pty-descendant-termination.ts index bf254d03b56..4f56c3e977b 100644 --- a/src/main/pty-descendant-termination.ts +++ b/src/main/pty-descendant-termination.ts @@ -21,12 +21,25 @@ export type ProcessTableRow = { startedAt: string } +export type PosixProcessIdentity = Pick + export type DescendantSnapshot = { + /** Identity of the root observed in the same process-table capture. */ + root?: PosixProcessIdentity rootPgid: number | null descendants: ProcessTableRow[] - /** Wall-clock boundary for deciding whether ps's second-resolution lstart - * can safely distinguish this process from a later PID reuse. */ + /** Wall-clock boundary for an unmerged snapshot (or legacy callers). */ capturedAtMs: number + /** Per-PID identity boundaries for merged captures. */ + capturedAtMsByPid?: Readonly> + /** + * PIDs this walk re-derived from a live root. A ppid walk only reaches what + * the root actually parents, so membership is proof of ownership that owes + * nothing to `lstart`'s one-second resolution: a stranger would have to have + * been forked into our own tree, and then it is not a stranger. Rows a merge + * retained from an earlier walk are absent, and still answer to start time. + */ + reDerivedPids?: ReadonlySet } export type ProcessTableCapture = { @@ -156,9 +169,13 @@ export function collectDescendantRows( ): DescendantSnapshot { const childrenByPpid = new Map() let rootRow: ProcessTableRow | null = null + let duplicateRoot = false for (const row of table) { if (row.pid === rootPid) { - rootRow = row + // A non-atomic process-table read can contain both an old and a recycled + // root row. There is no safe identity to retain in that case. + duplicateRoot = rootRow !== null + rootRow ??= row continue } const siblings = childrenByPpid.get(row.ppid) @@ -172,7 +189,7 @@ export function collectDescendantRows( // An absent root has already exited — its real descendants reparent to pid 1 and // become unreachable by ppid, so any rows still pointing at the vacated PID are a // PID-reuse coincidence. Sweeping them could signal an unrelated process, so bail. - if (!rootRow) { + if (!rootRow || duplicateRoot) { return { rootPgid: null, descendants: [], capturedAtMs } } const descendants: ProcessTableRow[] = [] @@ -191,7 +208,13 @@ export function collectDescendantRows( queue.push(child.pid) } } - return { rootPgid: rootRow.pgid, descendants, capturedAtMs } + return { + root: { pid: rootRow.pid, startedAt: rootRow.startedAt }, + rootPgid: rootRow.pgid, + descendants, + capturedAtMs, + reDerivedPids: new Set(descendants.map((row) => row.pid)) + } } type SnapshotDeps = { @@ -316,7 +339,11 @@ export type TerminateDeps = { } export function hasUnambiguousStartIdentity(row: ProcessTableRow, capturedAtMs: number): boolean { - const startedAtMs = Date.parse(row.startedAt) + return hasUnambiguousStartTime(row.startedAt, capturedAtMs) +} + +export function hasUnambiguousStartTime(startedAt: string, capturedAtMs: number): boolean { + const startedAtMs = Date.parse(startedAt) if (!Number.isFinite(startedAtMs)) { return false } @@ -364,7 +391,10 @@ export function terminateDescendantSnapshot( for (const row of snapshot.descendants) { const live = liveTargets.get(row.pid) if ( - hasUnambiguousStartIdentity(row, snapshot.capturedAtMs) && + hasUnambiguousStartIdentity( + row, + snapshot.capturedAtMsByPid?.[String(row.pid)] ?? snapshot.capturedAtMs + ) && live?.startedAt === row.startedAt && live.pgid === row.pgid ) { diff --git a/src/main/runtime/agent-session-acquisition-failure-settlement.ts b/src/main/runtime/agent-session-acquisition-failure-settlement.ts index 7ad20397813..c790a149f31 100644 --- a/src/main/runtime/agent-session-acquisition-failure-settlement.ts +++ b/src/main/runtime/agent-session-acquisition-failure-settlement.ts @@ -4,10 +4,28 @@ import { type AgentSessionOperationOutcome } from '../../shared/agent-session-operation-ledger' import { nextAgentSessionFence } from '../../shared/agent-session-next-fence' -import type { AgentSessionRecord } from '../../shared/agent-session-record' +import type { + AgentSessionDeathEvidence, + AgentSessionRecord +} from '../../shared/agent-session-record' import { assertFence, withLease } from './agent-session-lease-transitions' import type { AgentSessionStoreState } from './agent-session-record-store-file' +/** + * How the failed attempt's provider process was accounted for. + * - `exit-proven`: cleanup observed the whole tree gone. + * - `root-exit-observed`: the owner root's exit was observed first-hand, so the + * identity this lease is keyed on is dead, but its descendants could not be + * verified. Releases the lease and says exactly that, claiming nothing more. + * - `processless`: the attempt failed before a process existed. + * - `unproven`: nothing about the process was observed; the reservation latches. + */ +export type AgentSessionAcquisitionExitProof = + | 'exit-proven' + | 'root-exit-observed' + | 'processless' + | 'unproven' + export type AgentSessionFailedAcquisitionSettlement = { sessionId: string fence: number @@ -15,7 +33,7 @@ export type AgentSessionFailedAcquisitionSettlement = { callerKey: string operationId: string outcome: Extract - exitProof: 'exit-proven' | 'processless' | 'unproven' + exitProof: AgentSessionAcquisitionExitProof now: number } @@ -83,11 +101,18 @@ export function settleFailedAgentSessionPostAcquisitionAttachment( claimStatus: 'released', lastRenewedAt: args.now, handoffOperationId: null, - deathEvidence: { - kind: 'exit-observed', - detail: 'post-acquisition cleanup proved no provider child remains', - observedAt: args.now - } + deathEvidence: + args.exitProof === 'root-exit-observed' + ? { + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable', + observedAt: args.now + } + : { + kind: 'exit-observed', + detail: 'post-acquisition cleanup proved no provider child remains', + observedAt: args.now + } }) state.records.set(args.sessionId, next) state.operations = settleAgentSessionOperation(state.operations, args) @@ -127,18 +152,29 @@ function settleFailedLease( claimStatus: 'released', lastRenewedAt: args.now, handoffOperationId: null, - deathEvidence: - args.exitProof === 'processless' - ? { - kind: 'pid-absent', - detail: 'reservation failed before spawn', - observedAt: args.now - } - : { - // Cleanup proved no child of this attempt remains; it may never have spawned. - kind: 'exit-observed', - detail: 'acquisition cleanup proved no provider child remains', - observedAt: args.now - } + deathEvidence: acquisitionDeathEvidence(args.exitProof, args.now) }) } + +/** Records only what was observed: never a tree claim the cleanup did not make. */ +function acquisitionDeathEvidence( + exitProof: AgentSessionAcquisitionExitProof, + observedAt: number +): AgentSessionDeathEvidence { + if (exitProof === 'processless') { + return { kind: 'pid-absent', detail: 'reservation failed before spawn', observedAt } + } + if (exitProof === 'root-exit-observed') { + return { + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable', + observedAt + } + } + // Cleanup proved no child of this attempt remains; it may never have spawned. + return { + kind: 'exit-observed', + detail: 'acquisition cleanup proved no provider child remains', + observedAt + } +} diff --git a/src/main/runtime/agent-session-launch-env-backfill.test.ts b/src/main/runtime/agent-session-launch-env-backfill.test.ts new file mode 100644 index 00000000000..c95705f2aa5 --- /dev/null +++ b/src/main/runtime/agent-session-launch-env-backfill.test.ts @@ -0,0 +1,90 @@ +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' +import { AgentSessionRecordStore } from './agent-session-record-store' +import type { AgentSessionReserveRequest } from './agent-session-reservation-admission' + +const NOW = 1_800_000_000_000 +const SESSION = 'session-launch-env' +let directory: string + +function request(overrides: Partial = {}): AgentSessionReserveRequest { + return { + sessionId: SESSION, + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' + }, + provider: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/home/dev/.claude' }, + runtimeKind: 'native', + expectedFence: null, + spawnToken: 'spawn-a', + claimKeyId: 'key-1', + handoffOperationId: null, + probe: { outcome: 'reservation-unused' }, + operation: { + callerKey: 'client-1', + operationId: `${NOW}-00000000000000000000000000000001`, + fingerprint: 'fp-1' + }, + now: NOW, + ...overrides + } +} + +beforeEach(async () => { + directory = await mkdtemp(join(tmpdir(), 'orca-agent-session-launch-env-')) +}) + +afterEach(async () => { + await rm(directory, { recursive: true, force: true }) +}) + +describe('legacy agent session launch environment', () => { + it('durably pins the first environment resolved by a current reservation', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + await store.reserveOwner(request()) + await store.reserveOwner( + request({ + expectedFence: 1, + spawnToken: 'spawn-b', + launchEnv: { ANTHROPIC_AUTH_TOKEN: 'pinned-token' }, + operation: { + callerKey: 'client-1', + operationId: `${NOW}-00000000000000000000000000000002`, + fingerprint: 'fp-2' + } + }) + ) + + const reopened = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + expect( + (reopened.getRecord(SESSION) as { launchEnv?: Record } | null)?.launchEnv + ).toBeUndefined() + }) + + it('rejects an environment that could not be reloaded before writing it', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + const launchEnv = Object.fromEntries( + Array.from({ length: 257 }, (_, index) => [`KEY_${index}`, 'value']) + ) + + await expect(store.reserveOwner(request({ launchEnv }))).rejects.toThrow( + 'agent_session_launch_env_invalid' + ) + expect(store.getRecord(SESSION)).toBeNull() + }) + + it('rejects an overlong environment key before writing it', async () => { + const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) + + await expect( + store.reserveOwner(request({ launchEnv: { ['K'.repeat(513)]: 'value' } })) + ).rejects.toThrow('agent_session_launch_env_invalid') + expect(store.getRecord(SESSION)).toBeNull() + }) +}) diff --git a/src/main/runtime/agent-session-record-options.test.ts b/src/main/runtime/agent-session-record-options.test.ts index a1dfb9ccdef..5795763d96c 100644 --- a/src/main/runtime/agent-session-record-options.test.ts +++ b/src/main/runtime/agent-session-record-options.test.ts @@ -31,6 +31,20 @@ it('fails option hydration before ownership can be proved', async () => { ).rejects.toThrow('model list unavailable') }) +it('drops provider-rejected persisted options before the next owner proof', async () => { + await expect( + readNativeSessionOptions({ + adapter: { + readOptions: async () => ({ models: [], current: { model: 'provider-model' } }), + readOptionRestoreFailures: () => ['permissionMode'] + }, + sessionId: SESSION, + fence: 2, + priorOptions: { permissionMode: 'retired-mode', other: 'keep' } + }) + ).resolves.toEqual({ model: 'provider-model', other: 'keep' }) +}) + it('persists resumed provider options atomically with owner proof', async () => { const store = await AgentSessionRecordStore.open({ directory, hostId: 'local' }) const reserved = await store.reserveOwner({ diff --git a/src/main/runtime/agent-session-resume-args.test.ts b/src/main/runtime/agent-session-resume-args.test.ts new file mode 100644 index 00000000000..db4d0b07b8d --- /dev/null +++ b/src/main/runtime/agent-session-resume-args.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest' +import { resolveAgentSessionResumeArgs } from './agent-session-resume-args' + +describe('agent session resume arguments', () => { + it('keeps the session creation arguments after mutable defaults change', () => { + expect( + resolveAgentSessionResumeArgs({ + persistedArgs: ['--model', 'claude-created'], + defaultArgs: '--model claude-current', + shell: 'posix' + }) + ).toBe("'--model' 'claude-created'") + }) + + it('keeps an explicit empty snapshot when defaults are toggled off', () => { + expect( + resolveAgentSessionResumeArgs({ + persistedArgs: [], + defaultArgs: '--dangerously-skip-permissions', + shell: 'posix' + }) + ).toBe('') + }) + + it('uses current defaults for legacy records without a snapshot', () => { + expect( + resolveAgentSessionResumeArgs({ + defaultArgs: '--dangerously-skip-permissions', + shell: 'posix' + }) + ).toBe('--dangerously-skip-permissions') + }) +}) diff --git a/src/main/runtime/agent-session-resume-args.ts b/src/main/runtime/agent-session-resume-args.ts new file mode 100644 index 00000000000..dc276726dc6 --- /dev/null +++ b/src/main/runtime/agent-session-resume-args.ts @@ -0,0 +1,17 @@ +import type { AgentSessionLaunchArgs } from '../../shared/agent-session-record' +import { quoteStartupArg, type AgentStartupShell } from '../../shared/tui-agent-startup-shell' + +export function resolveAgentSessionResumeArgs(input: { + requestArgs?: string | null + persistedArgs?: AgentSessionLaunchArgs + defaultArgs?: string | null + shell: AgentStartupShell +}): string | null | undefined { + if (input.requestArgs !== undefined) { + return input.requestArgs + } + if (input.persistedArgs !== undefined) { + return input.persistedArgs.map((arg) => quoteStartupArg(arg, input.shell)).join(' ') + } + return input.defaultArgs +} diff --git a/src/main/runtime/claude-structured-session-integration.test.ts b/src/main/runtime/claude-structured-session-integration.test.ts new file mode 100644 index 00000000000..712543d0428 --- /dev/null +++ b/src/main/runtime/claude-structured-session-integration.test.ts @@ -0,0 +1,747 @@ +import { mkdir, mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { computeAgentSessionPayloadFingerprint } from '../../shared/agent-session-mutation-envelope' +import type { AgentJournalRenderItem } from '../../shared/agent-session-journal-types' +import type { AgentSessionSubscribeEvent } from '../../shared/agent-session-wire' +import { STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../shared/protocol-version' +import type { + ClaudeStreamJsonConnection, + ClaudeStreamJsonConnectionHandlers, + ClaudeStreamJsonLaunch, + openClaudeStreamJsonConnection +} from '../claude/claude-stream-json-connection' +import { claudeSessionIdForOrcaSession } from '../claude/claude-structured-launch-resolution' +import { + CLAUDE_SPAWN_TOKEN_ENV, + claudeProviderHandleLink +} from '../claude/claude-structured-owner-identity' +import { attachFingerprintFields } from '../native-chat/agent-session-wire/structured-agent-session-attach' +import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import type { + StructuredAgentSessionHandoffTransport, + StructuredTuiOwner +} from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' +import type { OrcaRuntimeService } from './orca-runtime' +import type { RpcRequest, RpcResponse } from './rpc/core' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { RpcDispatcher } from './rpc/dispatcher' +import { STRUCTURED_AGENT_SESSION_METHODS } from './rpc/methods/structured-agent-session' +import { + ensureStructuredAgentSessionHost, + stopStructuredAgentSessionRuntime +} from './structured-agent-session-runtime' + +const SESSION = 'claude-integration-1' +const PROVIDER_SESSION = claudeSessionIdForOrcaSession(SESSION) +const WORKSPACE = 'workspace-claude' +// Why 'runtime': this file exercises the Claude structured integration over agentSession.*, not the +// mobile surface — nothing here asserts anything mobile-specific, and its sibling integration +// suites use 'runtime' too. Mobile additionally requires the experimental structured-chat setting, +// which structured-agent-session.test.ts pins in both its satisfied and refused states. +const CLIENT = { + clientKind: 'runtime' as const, + clientCapabilities: [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY] +} + +const { readClaudeTranscriptLeafUuid, resolveSessionFilePath } = vi.hoisted(() => ({ + readClaudeTranscriptLeafUuid: vi.fn(), + resolveSessionFilePath: vi.fn() +})) + +vi.mock('../native-chat/session-file-resolver', () => ({ + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +})) + +type FakeClaudeConnection = Omit & { + closed: boolean + exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] + launch: ClaudeStreamJsonLaunch + handlers: ClaudeStreamJsonConnectionHandlers + calls: { subtype: string; params?: Record }[] + sent: Record[] +} + +function fakeClaude() { + const connections: FakeClaudeConnection[] = [] + let initializeAccount: unknown + /** A child that dies during start, with the close verdict its ladder observed. */ + let selfExit: { message: string; exitVerdict: ClaudeStreamJsonConnection['exitVerdict'] } | null = + null + const openConnection = (async (launch, handlers = {}) => { + const connection: FakeClaudeConnection = { + launch, + handlers, + calls: [], + sent: [], + pid: 4321 + connections.length, + closed: false, + initializationResult: async () => { + connection.calls.push({ subtype: 'initialize' }) + if (selfExit) { + handlers.onExit?.(new Error(selfExit.message)) + return { models: [] } + } + handlers.onMessage?.({ + type: 'system', + subtype: 'init', + session_id: PROVIDER_SESSION, + ...(connections.length === 0 ? { uuid: 'init-leaf' } : {}), + model: 'claude-sonnet-5', + apiKeySource: 'none' + }) + return { + models: [{ value: 'sonnet', displayName: 'Sonnet' }], + ...(initializeAccount === undefined ? {} : { account: initializeAccount }) + } + }, + getSettings: async () => { + connection.calls.push({ subtype: 'get_settings' }) + return { env: {} } + }, + supportedModels: async () => { + connection.calls.push({ subtype: 'list_models' }) + return [{ value: 'sonnet', displayName: 'Sonnet' }] + }, + setModel: async (model) => { + connection.calls.push({ subtype: 'set_model', params: { model } }) + }, + setPermissionMode: async (mode) => { + connection.calls.push({ subtype: 'set_permission_mode', params: { mode } }) + }, + applyFlagSettings: async (settings) => { + connection.calls.push({ subtype: 'apply_flag_settings', params: { settings } }) + }, + interrupt: async () => { + connection.calls.push({ subtype: 'interrupt', params: {} }) + return undefined + }, + cancelAsyncMessage: async () => {}, + send: async (message) => { + connection.sent.push(message) + if (message.type === 'user') { + handlers.onMessage?.({ ...message, uuid: 'user-1' }) + } + }, + exitVerdict: selfExit?.exitVerdict ?? { root: 'live', tree: 'unverifiable' }, + close: async () => { + connection.closed = true + return selfExit === null + } + } + connections.push(connection) + return connection + }) as typeof openClaudeStreamJsonConnection + const live = (): FakeClaudeConnection => { + const connection = connections.at(-1) + if (!connection) { + throw new Error('no Claude connection') + } + return connection + } + return { + connections, + openConnection, + live, + setInitializeAccount: (account: unknown) => { + initializeAccount = account + }, + setSelfExit: (exit: typeof selfExit) => { + selfExit = exit + } + } +} + +let operations = 0 +// Keep IDs unique without making each assertion depend on a wall-clock tick. +const TEST_OPERATION_TIMESTAMP = Date.now().toString() + +function operationId(): string { + operations += 1 + return `${TEST_OPERATION_TIMESTAMP}-${operations.toString(16).padStart(32, '0')}` +} + +function envelope(method: string, fields: Record, fence: number | null) { + return { + sessionId: SESSION, + clientOperationId: operationId(), + expectedRuntimeFence: fence, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method, + sessionId: SESSION, + fields + }) + } +} + +function createIntentParams() { + const worktree = `id:${WORKSPACE}` + const fields = { worktree, agent: 'claude' } + return { envelope: envelope('agentSession.create', fields, null), ...fields } +} + +function ensureParams(fence: number) { + const params = { + location: { + executionHostId: 'local', + wslDistro: null, + workspaceId: WORKSPACE, + workspaceKind: 'git-worktree' as const + }, + provider: 'claude' as const, + agent: 'claude', + accountHome: { variable: 'CLAUDE_CONFIG_DIR' as const, path: join(root, 'claude-home') }, + runtimeKind: 'native' as const, + providerHandle: { + kind: 'claude' as const, + sessionId: PROVIDER_SESSION, + leafUuid: 'assistant-leaf' + } + } + const base = { + sessionId: SESSION, + clientOperationId: operationId(), + expectedRuntimeFence: fence, + payloadFingerprint: '' + } + return { + ...params, + envelope: { + ...base, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.attach', + sessionId: SESSION, + fields: attachFingerprintFields({ ...params, envelope: base } as never) + }) + } + } +} + +function leaseOf(sessionId: string): { + claimStatus: string + runtimeFence: number + handoffStage: string | null + deathEvidence: { kind: string; detail: string } | null +} { + const host = getStructuredAgentSessionHost() as unknown as { + deps: { store: { getRecord: (id: string) => { lease: ReturnType } } } + } + return host.deps.store.getRecord(sessionId).lease +} + +function handoffParams(direction: 'to-native' | 'to-tui', fence: number) { + const fields = { direction, mode: 'now' as const, action: 'start' as const } + return { + envelope: envelope('agentSession.requestHandoff', fields, fence), + ...fields + } +} + +let claude: ReturnType +let root: string +let dispatcher: RpcDispatcher +let cleanups: Map void> +let tuiOwner: StructuredTuiOwner | null +let transcriptPath: string +/** Managed-account state and configured overlay this host installs, per test. */ +let claudeAuthPolicy: ClaudeStructuredAuthPolicy +let claudeLaunchEnv: Record + +async function call(method: string, params: unknown): Promise { + const replies: RpcResponse[] = [] + const request: RpcRequest = { id: `req-${operations}`, authToken: 'token', method, params } + await dispatcher.dispatchStreaming(request, (raw) => replies.push(JSON.parse(raw)), CLIENT) + if (!replies[0]) { + throw new Error(`no reply for ${method}`) + } + return replies[0] +} + +async function ok(method: string, params: unknown): Promise { + const response = await call(method, params) + expect(response, JSON.stringify(response)).toMatchObject({ ok: true }) + const result = (response as { result: { ok: boolean; value?: T } }).result + expect(result).toMatchObject({ ok: true }) + return result.value as T +} + +async function subscribe(): Promise { + const frames: AgentSessionSubscribeEvent[] = [] + await dispatcher.dispatchStreaming( + { + id: 'subscribe-1', + authToken: 'token', + method: 'agentSession.subscribe', + params: { sessionId: SESSION } + }, + (raw) => { + const response = JSON.parse(raw) as { ok: boolean; result?: AgentSessionSubscribeEvent } + if (response.ok && response.result) { + frames.push(response.result) + } + }, + CLIENT + ) + return frames +} + +function itemsOf(frames: AgentSessionSubscribeEvent[]): AgentJournalRenderItem[] { + const items = new Map() + for (const frame of frames) { + const rows = + frame.type === 'snapshot' || frame.type === 'reset' + ? frame.page.items + : frame.type === 'batch' + ? frame.batch.items + : [] + for (const row of rows) { + items.set(row.itemId, row) + } + } + return [...items.values()] +} + +function textOf(item: AgentJournalRenderItem): string { + return item.body?.kind === 'message' + ? item.body.blocks.map((block) => (block.type === 'text' ? block.text : '')).join('') + : '' +} + +beforeEach(async () => { + operations = 0 + claudeAuthPolicy = { stripAuthEnv: false } + claudeLaunchEnv = { + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test' + } + root = await mkdtemp(join(tmpdir(), 'orca-claude-structured-integration-')) + transcriptPath = join(root, 'claude-home', 'projects', 'workspace', `${PROVIDER_SESSION}.jsonl`) + await mkdir(join(root, 'claude-home', 'projects', 'workspace'), { recursive: true }) + resolveSessionFilePath.mockResolvedValue(transcriptPath) + // The production branch proof returns the latest descendant of the prior + // cursor; mirror that contract so structured close does not regress to a + // stale mocked head. + readClaudeTranscriptLeafUuid.mockImplementation( + async (_path: string, _providerSessionId: string, previousLeafUuid?: string | null) => + previousLeafUuid ?? 'init-leaf' + ) + claude = fakeClaude() + tuiOwner = null + cleanups = new Map() + const handoffTransport: StructuredAgentSessionHandoffTransport = { + hostLabel: 'Scripted Claude host', + launchTui: async ({ record, fence, spawnToken }) => { + const head = record.providerHandleChain.at(-1)?.handle + tuiOwner = { + terminal: { + handle: 'term-claude-tui', + tabId: 'tab-claude-tui', + paneKey: 'tab-claude-tui:leaf-claude-tui', + ptyId: 'pty-claude-tui' + }, + process: { + hostId: 'local', + pid: 7331, + processStartTimeMs: 100, + spawnToken + }, + link: claudeProviderHandleLink({ + sessionId: PROVIDER_SESSION, + leafUuid: head?.provider === 'claude' ? head.leafUuid : null, + resumed: true, + fence, + observedAt: 1 + }), + transcriptPath + } + return tuiOwner + }, + reproveTuiOwner: async ({ owner }) => { + if (owner.link.handle.provider !== 'claude' || !owner.transcriptPath) { + return owner + } + return { + ...owner, + link: claudeProviderHandleLink({ + sessionId: owner.link.handle.sessionId, + leafUuid: await readClaudeTranscriptLeafUuid(owner.transcriptPath), + resumed: true, + fence: owner.link.mintedAtFence, + observedAt: 1 + }) + } + }, + recoverTuiOwner: async () => { + if (!tuiOwner) { + throw new Error('scripted TUI owner missing') + } + return tuiOwner + }, + stopRecoveredOwner: async () => {}, + waitForTuiExit: async (owner) => ({ transcriptPath: owner.transcriptPath }), + waitForTuiIdleOrExit: async () => 'idle', + tuiStatus: () => 'idle', + stopFailedTuiLaunch: async () => {} + } + const runtime = { + getRuntimeId: () => 'runtime-1', + getStructuredAgentSessionCreateSupport: async () => ({ supported: true }), + resolveStructuredAgentSessionCreateIntent: async (input: { envelope: unknown }) => ({ + ...ensureParams(1), + envelope: input.envelope, + providerHandle: undefined + }), + publishStructuredAgentSessionTab: vi.fn(), + ensureStructuredAgentSessionHost: () => + ensureStructuredAgentSessionHost({ + stateDirectory: root, + hostId: 'local', + claimKeyId: 'key-1', + resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, + resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeCommand: () => '/usr/local/bin/claude', + readProcessStartTime: async (pid: number) => pid * 10, + resolveClaudeLaunchEnv: () => claudeLaunchEnv, + resolveClaudeAuthPolicy: () => claudeAuthPolicy, + openClaudeConnection: claude.openConnection, + handoffTransport + }).then(() => undefined), + registerSubscriptionCleanup: (id: string, dispose: () => void) => cleanups.set(id, dispose), + cleanupSubscription: (id: string) => cleanups.get(id)?.(), + cleanupSubscriptionsByPrefix: () => {} + } + dispatcher = new RpcDispatcher({ + runtime: runtime as unknown as OrcaRuntimeService, + methods: STRUCTURED_AGENT_SESSION_METHODS + }) +}) + +afterEach(async () => { + vi.unstubAllEnvs() + await stopStructuredAgentSessionRuntime() + await rm(root, { recursive: true, force: true }) +}) + +describe('a structured Claude session over agentSession.*', () => { + it('strips ambient Anthropic auth from the child once a managed account is pinned', async () => { + claudeAuthPolicy = { stripAuthEnv: true } + claudeLaunchEnv = { ANTHROPIC_BASE_URL: 'https://gateway.example.test' } + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + vi.stubEnv('ANTHROPIC_AUTH_TOKEN', 'tok-SHELL-LEAK') + + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + + const env = claude.live().launch.env + expect(env).not.toHaveProperty('ANTHROPIC_API_KEY') + expect(env).not.toHaveProperty('ANTHROPIC_AUTH_TOKEN') + expect(env).toMatchObject({ + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: join(root, 'claude-home') + }) + }) + + it('refuses a create whose configured env overrides the pinned managed account auth', async () => { + claudeAuthPolicy = { stripAuthEnv: true } + // The default overlay carries ANTHROPIC_AUTH_TOKEN, which the terminal path + // refuses at spawn-env.ts:25 rather than letting it beat the pinned account. + const refused = await call('agentSession.create', createIntentParams()) + + expect(JSON.stringify(refused)).toContain('explicit Anthropic auth environment') + // Refused before spawn: no provider child was ever opened. + expect(claude.connections).toHaveLength(0) + }) + + it('durably returns actionable sign-in guidance when initialization has no credentials', async () => { + claude.setInitializeAccount({ apiProvider: 'firstParty', tokenSource: 'none' }) + const params = createIntentParams() + + const first = await call('agentSession.create', params) + const retry = await call('agentSession.create', params) + + expect(first).toMatchObject({ + ok: true, + result: { + ok: false, + refusal: { + code: 'agent_session_operation_invalid', + message: expect.stringMatching(/not signed in.*Claude CLI.*CLAUDE_CONFIG_DIR/s) + } + } + }) + expect((retry as { result: unknown }).result).toEqual((first as { result: unknown }).result) + expect(claude.connections).toHaveLength(1) + }) + + it('releases a session whose CLI self-exited during create, with its diagnostic intact', async () => { + claude.setSelfExit({ + message: 'claude stream-json exited (code 1): claude: not signed in', + // The root's death is first-hand; its descendants were never snapshottable. + exitVerdict: { root: 'exited', tree: 'unverifiable' } + }) + + const failed = await call('agentSession.create', createIntentParams()) + + expect(JSON.stringify(failed)).toContain('claude: not signed in') + const lease = leaseOf(SESSION) + // Latching here would refuse every later attach with agent_session_ownership_unknown, + // wedging a user who only needs to sign in. + expect(lease).toMatchObject({ claimStatus: 'released', handoffStage: null }) + expect(lease.deathEvidence).toMatchObject({ + kind: 'exit-observed', + detail: 'the provider process exited; its descendants were not verifiable' + }) + + claude.setSelfExit(null) + // Signing in and reopening the chat works: the reservation was not latched. + await ok<{ fence: number }>('agentSession.ensure', ensureParams(lease.runtimeFence)) + }) + + it('keeps a session reserved when a descendant of the failed start was seen alive', async () => { + claude.setSelfExit({ + message: 'claude stream-json exited (code 1): claude: not signed in', + exitVerdict: { root: 'exited', tree: 'live' } + }) + + await call('agentSession.create', createIntentParams()) + + // A live descendant still holds the provider session: releasing would hand a + // second writer to it. + expect(leaseOf(SESSION)).toMatchObject({ + claimStatus: 'reserved', + handoffStage: 'manual-recovery' + }) + claude.setSelfExit(null) + }) + + it('routes a published Claude first-hand exit through fenced host reconciliation', async () => { + await ok<{ fence: number }>('agentSession.create', createIntentParams()) + const connection = claude.live() + connection.exitVerdict = { root: 'exited', tree: 'unverifiable' } + connection.handlers.onExit?.(new Error('claude stream-json exited (code 1): crashed')) + + for ( + let attempt = 0; + attempt < 20 && leaseOf(SESSION).claimStatus !== 'released'; + attempt += 1 + ) { + await new Promise((resolve) => setTimeout(resolve, 5)) + } + expect(leaseOf(SESSION)).toMatchObject({ claimStatus: 'released', handoffStage: null }) + }) + + it('creates, sends, streams, approves, interrupts, and resumes from the chain head', async () => { + vi.stubEnv('ANTHROPIC_API_KEY', 'sk-ant-SHELL-LEAK') + const created = await ok<{ fence: number }>('agentSession.create', createIntentParams()) + expect(claude.live().launch.options).toMatchObject({ sessionId: PROVIDER_SESSION }) + expect(claude.live().launch.options.resume).toBeUndefined() + expect(claude.live().launch.env).toMatchObject({ + ANTHROPIC_AUTH_TOKEN: 'configured-token', + ANTHROPIC_BASE_URL: 'https://gateway.example.test', + CLAUDE_CONFIG_DIR: join(root, 'claude-home'), + [CLAUDE_SPAWN_TOKEN_ENV]: expect.any(String) + }) + // System auth: the user's own shell key is their sign-in, exactly as on the + // terminal path, and the configured overlay still wins over it. + expect(claude.live().launch.env).toMatchObject({ ANTHROPIC_API_KEY: 'sk-ant-SHELL-LEAK' }) + expect(claude.live().launch.env?.PATH ?? claude.live().launch.env?.Path).toBeTruthy() + const history = await call('agentSession.history', { + sessionId: SESSION, + direction: 'tail', + limit: 1 + }) + expect(history).toMatchObject({ + ok: true, + result: { providerSession: { key: 'session_id', id: PROVIDER_SESSION } } + }) + const stream = await subscribe() + + const body = { kind: 'message', role: 'user', blocks: [{ type: 'text', text: 'List files' }] } + const sent = await ok<{ + submission: { dispatchState: string; providerItemId: string | null } + }>('agentSession.send', { + envelope: envelope('agentSession.send', { body }, created.fence), + body + }) + expect(sent.submission).toMatchObject({ + dispatchState: 'accepted', + providerItemId: `claude:${PROVIDER_SESSION}:user-1` + }) + + claude.live().handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION, + uuid: 'assistant-leaf', + event: { type: 'content_block_delta', delta: { type: 'text_delta', text: 'Two files.' } } + }) + claude.live().handlers.onMessage?.({ + type: 'assistant', + session_id: PROVIDER_SESSION, + uuid: 'assistant-leaf', + parent_tool_use_id: null, + message: { role: 'assistant', content: [{ type: 'text', text: 'Two files.' }] } + }) + claude.live().handlers.onMessage?.({ + type: 'result', + subtype: 'success', + session_id: PROVIDER_SESSION, + uuid: 'result-frame-uuid' + }) + claude.live().handlers.onMessage?.({ + type: 'stream_event', + session_id: PROVIDER_SESSION, + uuid: 'stream-event-frame-uuid', + event: { type: 'message_stop' } + }) + await getStructuredAgentSessionHost()?.flushStreamedEvents(SESSION) + expect(itemsOf(stream).find((item) => textOf(item) === 'Two files.')?.itemId).toBe( + `claude:${PROVIDER_SESSION}:assistant-leaf` + ) + + const answeredPermission = Promise.resolve( + claude.live().handlers.canUseTool?.('Bash', { command: 'ls' }, { + requestId: 'permission-1', + toolUseID: 'tool-1', + signal: new AbortController().signal + } as never) + ) + await getStructuredAgentSessionHost()?.flushStreamedEvents(SESSION) + const approval = itemsOf(stream).find((item) => item.body?.kind === 'approval') + expect(approval?.body).toMatchObject({ title: 'Allow Bash?', detail: '{"command":"ls"}' }) + await ok('agentSession.respondToApproval', { + envelope: envelope( + 'agentSession.respondTo:approval', + { + itemId: approval?.itemId, + expectedRevision: approval?.revision, + optionId: 'allow' + }, + created.fence + ), + itemId: approval?.itemId, + expectedRevision: approval?.revision, + optionId: 'allow' + }) + // Answering resolves the SDK's own canUseTool callback with the allow decision. + await expect(answeredPermission).resolves.toMatchObject({ + behavior: 'allow', + toolUseID: 'tool-1' + }) + + await expect( + ok('agentSession.cancel', { + envelope: envelope('agentSession.cancel', { turnId: 'user-1' }, created.fence), + turnId: 'user-1' + }) + ).resolves.toMatchObject({ turnId: 'user-1', cancelled: true }) + expect(claude.live().calls.at(-1)).toMatchObject({ subtype: 'interrupt' }) + + const host = getStructuredAgentSessionHost() as unknown as { + deps: { + store: { + getRecord: (sessionId: string) => { + providerHandleChain: { handle: { provider: string; leafUuid?: string | null } }[] + } + } + } + } + expect(host.deps.store.getRecord(SESSION).providerHandleChain.at(-1)?.handle).toMatchObject({ + provider: 'claude', + leafUuid: null + }) + const old = claude.live() + const resumed = await ok<{ fence: number }>('agentSession.ensure', ensureParams(created.fence)) + expect(resumed.fence).toBe(created.fence + 1) + expect(old.closed).toBe(true) + expect(resolveSessionFilePath).toHaveBeenCalledWith('claude', PROVIDER_SESSION, { + claudeProjectsDir: join(root, 'claude-home', 'projects') + }) + expect(claude.live().launch.options).toMatchObject({ + resume: PROVIDER_SESSION, + resumeSessionAt: 'assistant-leaf' + }) + expect(host.deps.store.getRecord(SESSION).providerHandleChain.at(-1)).toMatchObject({ + handle: { + provider: 'claude', + sessionId: PROVIDER_SESSION, + leafUuid: 'assistant-leaf' + }, + origin: 'resumed' + }) + }) + + it('completes a scripted native to TUI to native cycle with provider-history rehydration', async () => { + const created = await ok<{ fence: number }>('agentSession.create', createIntentParams()) + await writeFile( + transcriptPath, + [ + { + type: 'user', + uuid: 'native-user', + message: { role: 'user', content: [{ type: 'text', text: 'NATIVE_USER' }] } + }, + { + type: 'assistant', + uuid: 'native-assistant', + message: { role: 'assistant', content: [{ type: 'text', text: 'NATIVE_ASSISTANT' }] } + }, + { + type: 'user', + uuid: 'tui-user', + message: { role: 'user', content: [{ type: 'text', text: 'TUI_USER' }] } + }, + { + type: 'assistant', + uuid: 'tui-assistant', + message: { role: 'assistant', content: [{ type: 'text', text: 'TUI_ASSISTANT' }] } + }, + { type: 'last-prompt', leafUuid: 'tui-assistant' } + ] + .map((entry) => JSON.stringify(entry)) + .join('\n') + ) + + await ok('agentSession.requestHandoff', handoffParams('to-tui', created.fence)) + const host = getStructuredAgentSessionHost()! + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'tui', phase: 'idle' }) + ) + expect(claude.connections[0]?.closed).toBe(true) + + const tuiFence = ( + host as unknown as { + deps: { store: { getRecord: (id: string) => { lease: { runtimeFence: number } } } } + } + ).deps.store.getRecord(SESSION).lease.runtimeFence + readClaudeTranscriptLeafUuid.mockResolvedValueOnce('tui-assistant') + await ok('agentSession.requestHandoff', handoffParams('to-native', tuiFence)) + await vi.waitFor(async () => + expect(await host.handoffStatus(SESSION)).toMatchObject({ owner: 'native', phase: 'idle' }) + ) + + const frames = await subscribe() + const texts = itemsOf(frames).map(textOf).filter(Boolean) + expect(texts).toEqual( + expect.arrayContaining(['NATIVE_USER', 'NATIVE_ASSISTANT', 'TUI_USER', 'TUI_ASSISTANT']) + ) + expect(new Set(texts).size).toBe(texts.length) + expect(claude.connections).toHaveLength(2) + expect(claude.live().launch.options).toMatchObject({ resume: PROVIDER_SESSION }) + const record = ( + host as unknown as { + deps: { + store: { + getRecord: (id: string) => { + providerHandleChain: { handle: { provider: string; leafUuid?: string | null } }[] + } + } + } + } + ).deps.store.getRecord(SESSION) + expect(record.providerHandleChain.at(-1)?.handle).toMatchObject({ + provider: 'claude', + leafUuid: 'tui-assistant' + }) + }) +}) diff --git a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts index afbb70fc286..f70cf033718 100644 --- a/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts +++ b/src/main/runtime/orca-runtime-get-agent-session-execution-namespace.ts @@ -17,6 +17,9 @@ import { resolveTuiAgentLaunchArgs, resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' +import type { AgentSessionLaunchArgs } from '../../shared/agent-session-record' +import { resolveStartupShell } from '../../shared/tui-agent-startup-shell' +import { resolveAgentSessionResumeArgs } from './agent-session-resume-args' export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntimeWithResolveWorktreeRemovalTarget { protected getAgentSessionExecutionNamespace( @@ -88,7 +91,12 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim async ensureAgentSession( request: RuntimeEnsureAgentSessionRequest, _caller: RuntimeAgentSessionRpcCaller = {}, - handoffAuthority?: { spawnToken: string; providerRoot: string; sessionId: string } + handoffAuthority?: { + spawnToken: string + providerRoot: string + sessionId: string + launchArgs?: AgentSessionLaunchArgs + } ): Promise { if (request.kind === 'automatic') { // Legacy renderer sleep records are migration evidence, not host authority. @@ -134,10 +142,12 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim agent: request.agent, providerSession: identity.providerSession, cmdOverrides: settings.agentCmdOverrides ?? {}, - agentArgs: - request.agentArgs !== undefined - ? request.agentArgs - : resolveTuiAgentLaunchArgs(request.agent, settings.agentDefaultArgs), + agentArgs: resolveAgentSessionResumeArgs({ + requestArgs: request.agentArgs, + persistedArgs: handoffAuthority?.launchArgs, + defaultArgs: resolveTuiAgentLaunchArgs(request.agent, settings.agentDefaultArgs), + shell: resolveStartupShell(platform, shell) + }), agentEnv: { ...resolveTuiAgentLaunchEnv(request.agent, settings.agentDefaultEnv), ...(handoffAuthority && request.agent === 'codex' @@ -148,6 +158,7 @@ export class OrcaRuntimeWithGetAgentSessionExecutionNamespace extends OrcaRuntim }, ompResumeFilePath: request.ompResumeFilePath, sessionOptions: this.toAgentSessionOptions(request.launchPreferences), + sessionOptionsOverrideAgentArgs: Boolean(request.launchPreferences), platform, shell, isRemote diff --git a/src/main/runtime/orca-runtime-get-worktree-ps.ts b/src/main/runtime/orca-runtime-get-worktree-ps.ts index 42c9c7ff6d3..06391e2ae83 100644 --- a/src/main/runtime/orca-runtime-get-worktree-ps.ts +++ b/src/main/runtime/orca-runtime-get-worktree-ps.ts @@ -10,6 +10,7 @@ import { } from './runtime-worktree-ps-activity' import { attachRuntimeWorktreeAgentRows } from './runtime-worktree-agent-rows' import { compareWorktreePs } from './runtime-worktree-status-projection' +import type { AgentSessionRecord } from '../../shared/agent-session-record' import type { Repo } from '../../shared/repo-types' import { enrichMissingRepoGitRemoteIdentities } from '../repo-git-remote-identity-enrichment' import { ensureStructuredAgentSessionHost as installStructuredAgentSessionHost } from './structured-agent-session-runtime' @@ -21,9 +22,11 @@ import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' import { resolveLocalWindowsAgentStartupShell } from '../../shared/windows-terminal-shell' +import { resolveStartupShell, tokenizeStartupCommand } from '../../shared/tui-agent-startup-shell' import { resolveCodexStructuredAppServerArgs } from '../codex/codex-structured-app-server-args' import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' import { hostname } from 'node:os' +import { claudeStructuredAuthPolicyForSettings } from '../claude-accounts/claude-structured-auth-policy' import { probeAgentSessionProcessIdentity } from './agent-session-process-identity-probe' import { structuredAgentSessionTabId } from '../../shared/structured-agent-session-projection' @@ -144,13 +147,47 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent // in a plain folder lands in the folder rather than failing to resolve. resolveWorkspacePath: async (workspaceId) => (await this.resolveRuntimeFileTarget(`id:${workspaceId}`)).worktree.path, - resolveLaunchArgs: () => this.resolveConfiguredCodexStructuredArgs(), + resolveLaunchArgs: (provider) => this.resolveConfiguredStructuredLaunchArgs(provider), resolveLaunchEnvOverlay: () => resolveTuiAgentLaunchEnv('codex', this.requireStore().getSettings().agentDefaultEnv), + resolveClaudeLaunchEnv: () => + resolveTuiAgentLaunchEnv('claude', this.requireStore().getSettings().agentDefaultEnv), + resolveClaudeAuthPolicy: () => + claudeStructuredAuthPolicyForSettings(this.requireStore().getSettings()), + // Same gate and same settings as agentSession.createSupport, re-read on every acquisition. + getClaudeManagedAccountGateSettings: () => this.requireStore().getSettings(), handoffTransport: this.createStructuredAgentSessionHandoffTransport() }) } + // Why the provider is honoured rather than assumed: Codex app-server flags are not + // Claude CLI flags, and prepending them to `claude` makes it exit on an unknown option. + protected resolveConfiguredStructuredLaunchArgs( + provider: AgentSessionRecord['provider'] + ): string[] { + if (provider === 'claude') { + return this.resolveConfiguredClaudeStructuredArgs() + } + return this.resolveConfiguredCodexStructuredArgs() + } + + protected resolveConfiguredClaudeStructuredArgs(): string[] { + const settings = this.requireStore().getSettings() + const shell = resolveStartupShell( + process.platform, + resolveLocalWindowsAgentStartupShell({ + platform: process.platform, + isRemote: false, + terminalWindowsShell: settings.terminalWindowsShell + }) + ) + const tokenized = tokenizeStartupCommand( + resolveTuiAgentLaunchArgs('claude', settings.agentDefaultArgs), + shell + ) + return tokenized.ok ? tokenized.tokens : [] + } + protected resolveConfiguredCodexStructuredArgs(): string[] { const settings = this.requireStore().getSettings() const shell = resolveLocalWindowsAgentStartupShell({ @@ -195,7 +232,7 @@ export class OrcaRuntimeWithGetWorktreePs extends OrcaRuntimeWithStructuredAgent tuiStatus: (owner) => this.structuredTuiStatus(owner), closeTuiOwner: (owner) => this.closeStructuredTuiOwner(owner), revealNativeSession: async ({ workspaceId, sessionId, agent = 'codex', adoptedTerminal }) => { - if (adoptedTerminal || agent !== 'codex') { + if (adoptedTerminal || (agent !== 'codex' && agent !== 'claude')) { return } await this.publishStructuredAgentSessionTab({ diff --git a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts index 5700b71d9a5..cfb8b421966 100644 --- a/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts +++ b/src/main/runtime/orca-runtime-resolve-recovered-structured-tui-transcript.ts @@ -4,6 +4,7 @@ import type { AgentSessionOwnerBinding } from '../../shared/agent-session-host-a import { agentSessionOwnerBindingsEqual } from '../../shared/claimed-agent-pty-owner-snapshot' import { resolvePinnedCodexRolloutProof } from '../codex/codex-tui-rollout-proof' import { getStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { resolveStructuredAgentSessionCreateSupport } from '../native-chat/structured-agent-session-create-support' import { LOCAL_EXECUTION_HOST_ID } from '../../shared/execution-host' import type { AgentStatusIpcPayload } from '../../shared/agent-status-types' import { getLocalProjectWorktreeGitOptions } from '../project-runtime-git-options' @@ -12,6 +13,8 @@ import { getSystemCodexHomePath } from '../codex/codex-home-paths' import { resolveTuiAgentLaunchEnv } from '../../shared/tui-agent-launch-defaults' import { hasPersistedStructuredAgentSessionStore as hasPersistedStructuredAgentSessionStoreOnDisk } from './structured-agent-session-runtime' import { getProfileUserDataPath } from '../orca-profiles/profile-storage-paths' +import { homedir } from 'node:os' +import { join } from 'node:path' export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends OrcaRuntimeWithStopStructuredSessionProcess { protected async resolveRecoveredStructuredTuiTranscript(input: { @@ -45,22 +48,18 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca async getStructuredAgentSessionCreateSupport( worktreeSelector: string, - agent: 'codex' + agent: 'claude' | 'codex' ): Promise<{ supported: boolean; reason?: 'agent' | 'remote' | 'wsl' }> { const location = await this.resolveStructuredAgentSessionLocation(worktreeSelector) await this.ensureStructuredAgentSessionHost() - if (getStructuredAgentSessionHost()?.supportsCreate(location, agent)) { - return { supported: true } - } - return { - supported: false, - reason: - location.executionHostId !== LOCAL_EXECUTION_HOST_ID - ? 'remote' - : location.wslDistro - ? 'wsl' - : 'agent' - } + // The verdict lives in a typechecked module; this file is @ts-nocheck. + return resolveStructuredAgentSessionCreateSupport({ + agent, + location, + adapterSupportsCreate: + getStructuredAgentSessionHost()?.supportsCreate(location, agent) === true, + getSettings: () => this.requireStore().getSettings() + }) } protected hasProviderSessionObservationSource(): boolean { @@ -108,8 +107,23 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca async resolveStructuredAgentSessionCreateIntent(input: { envelope: { sessionId: string; clientOperationId: string } worktree: string - agent: 'codex' + agent: 'claude' | 'codex' }): Promise { + if (input.agent === 'claude') { + return this.resolveStructuredAgentSessionIntent(input, async ({ launchEnv, location }) => { + return ( + launchEnv.CLAUDE_CONFIG_DIR?.trim() || + this.accounts + .getClaudeConfigDirectory( + location.wslDistro + ? { runtime: 'wsl', wslDistro: location.wslDistro } + : { runtime: 'host' } + ) + ?.trim() || + join(homedir(), '.claude') + ) + }) + } return this.resolveStructuredAgentSessionIntent(input, async ({ workspacePath, launchEnv }) => { // A create has no process yet, so the current selection is what it must follow. const preparedHome = await this.prepareCodexStructuredLaunchFn?.({ workspacePath, launchEnv }) @@ -126,11 +140,17 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca input: { envelope: { sessionId: string; clientOperationId: string } worktree: string - agent: 'codex' + agent: 'claude' | 'codex' }, resolveAccountHomePath: (context: { workspacePath: string launchEnv: NodeJS.ProcessEnv + location: { + executionHostId: string + wslDistro: string | null + workspaceId: string + workspaceKind: 'folder' | 'git-worktree' + } }) => string | Promise ): Promise { const support = await this.getStructuredAgentSessionCreateSupport(input.worktree, input.agent) @@ -152,8 +172,8 @@ export class OrcaRuntimeWithResolveRecoveredStructuredTuiTranscript extends Orca provider: input.agent, agent: input.agent, accountHome: { - variable: 'CODEX_HOME', - path: await resolveAccountHomePath({ workspacePath, launchEnv }) + variable: input.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', + path: await resolveAccountHomePath({ workspacePath, launchEnv, location }) }, runtimeKind: 'native' } diff --git a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts index 4466594b0dc..5b2c160f2c6 100644 --- a/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts +++ b/src/main/runtime/orca-runtime-restore-structured-agent-session-tabs-once.ts @@ -45,7 +45,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu } this.hydrateHeadlessMobileSessionTabsFromWorkspaceSession() for (const session of host?.listSessionTabs() ?? []) { - if (session.agent !== 'codex') { + if (session.agent !== 'codex' && session.agent !== 'claude') { continue } let sessionId = session.sessionId @@ -54,7 +54,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu } await this.publishStructuredAgentSessionTab({ ...session, - agent: 'codex', + agent: session.agent, sessionId, activate: false, notify: false @@ -65,7 +65,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu async publishStructuredAgentSessionTab(input: { workspaceId: string sessionId: string - agent: 'codex' + agent: 'claude' | 'codex' activate: boolean notify?: boolean }): Promise { @@ -105,7 +105,7 @@ export class OrcaRuntimeWithRestoreStructuredAgentSessionTabsOnce extends OrcaRu const tab: RuntimeMobileSessionAgentTab = { type: 'agent-session', id, - title: 'Codex Chat', + title: input.agent === 'claude' ? 'Claude Chat' : 'Codex Chat', sessionId: input.sessionId, agent: input.agent, isActive: input.activate diff --git a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts index 083c677e39f..af1fbc20372 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-create-intent.test.ts @@ -52,4 +52,108 @@ describe('structured agent-session create intent', () => { path: '/accounts/selected/home' }) }) + + it('pins the configured Claude launch home without Codex launch preparation', async () => { + const prepareCodexStructuredLaunch = vi.fn() + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + agentDefaultEnv: { + claude: { CLAUDE_CONFIG_DIR: '/configured/claude-home' } + } + }) + } as never, + undefined, + { prepareCodexStructuredLaunch } + ) + vi.spyOn(runtime, 'getStructuredAgentSessionCreateSupport').mockResolvedValue({ + supported: true + }) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise<{ + executionHostId: string + wslDistro: null + workspaceId: string + workspaceKind: 'git-worktree' + }> + resolveRuntimeFileTarget: (selector: string) => Promise<{ + worktree: { path: string } + }> + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + internal.resolveRuntimeFileTarget = vi.fn(async () => ({ + worktree: { path: '/repos/workspace-1' } + })) + + const intent = await runtime.resolveStructuredAgentSessionCreateIntent({ + envelope: { sessionId: 'session-1', clientOperationId: 'operation-1' }, + worktree: 'id:workspace-1', + agent: 'claude' + }) + + expect(prepareCodexStructuredLaunch).not.toHaveBeenCalled() + expect(intent.accountHome).toEqual({ + variable: 'CLAUDE_CONFIG_DIR', + path: '/configured/claude-home' + }) + }) + + it('uses the managed Claude launch home before falling back to ~/.claude', async () => { + const prepareCodexStructuredLaunch = vi.fn() + const getRuntimeConfigDir = vi.fn(() => '/accounts/managed/claude-home') + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + agentDefaultEnv: { claude: {} } + }) + } as never, + undefined, + { prepareCodexStructuredLaunch } + ) + runtime.setAccountServices({ + claudeAccounts: { getRuntimeConfigDir } as never, + codexAccounts: {} as never, + rateLimits: {} as never + }) + vi.spyOn(runtime, 'getStructuredAgentSessionCreateSupport').mockResolvedValue({ + supported: true + }) + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise<{ + executionHostId: string + wslDistro: null + workspaceId: string + workspaceKind: 'git-worktree' + }> + resolveRuntimeFileTarget: (selector: string) => Promise<{ + worktree: { path: string } + }> + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + internal.resolveRuntimeFileTarget = vi.fn(async () => ({ + worktree: { path: '/repos/workspace-1' } + })) + + const intent = await runtime.resolveStructuredAgentSessionCreateIntent({ + envelope: { sessionId: 'session-1', clientOperationId: 'operation-1' }, + worktree: 'id:workspace-1', + agent: 'claude' + }) + + expect(getRuntimeConfigDir).toHaveBeenCalledTimes(1) + expect(intent.accountHome).toEqual({ + variable: 'CLAUDE_CONFIG_DIR', + path: '/accounts/managed/claude-home' + }) + }) }) diff --git a/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts b/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts new file mode 100644 index 00000000000..c4333fd0a4d --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-agent-session-launch-args.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' + +type InstalledDeps = { + resolveLaunchArgs: (provider: 'claude' | 'codex') => Promise | string[] + resolveLaunchEnvOverlay: () => Record + resolveClaudeLaunchEnv?: () => Record +} + +const { installStructuredAgentSessionHost } = vi.hoisted(() => ({ + installStructuredAgentSessionHost: vi.fn(async (_deps: unknown) => ({}) as never) +})) + +vi.mock('./structured-agent-session-runtime', async (importOriginal) => ({ + ...(await importOriginal()), + ensureStructuredAgentSessionHost: installStructuredAgentSessionHost +})) + +function runtimeWith(settings: Record): OrcaRuntimeService { + return new OrcaRuntimeService({ getSettings: () => settings } as never) +} + +async function installedDeps(settings: Record): Promise { + installStructuredAgentSessionHost.mockClear() + await runtimeWith(settings).ensureStructuredAgentSessionHost() + return installStructuredAgentSessionHost.mock.calls[0]?.[0] as InstalledDeps +} + +describe('structured agent-session launch args wiring', () => { + it('resolves Claude launch args from the Claude agent defaults, not Codex flags', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { + claude: '--dangerously-skip-permissions --model opus', + codex: '--dangerously-bypass-approvals-and-sandbox' + }, + agentDefaultEnv: {} + }) + + expect(await deps.resolveLaunchArgs('claude')).toEqual([ + '--dangerously-skip-permissions', + '--model', + 'opus' + ]) + }) + + it('still resolves Codex app-server args for a Codex session', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { + claude: '--dangerously-skip-permissions', + codex: '--dangerously-bypass-approvals-and-sandbox' + }, + agentDefaultEnv: {} + }) + + const codexArgs = await deps.resolveLaunchArgs('codex') + expect(codexArgs).not.toContain('--dangerously-skip-permissions') + expect(codexArgs.length).toBeGreaterThan(0) + }) + + it('never lets a broken Codex args configuration block a Claude session', async () => { + const deps = await installedDeps({ + agentDefaultArgs: { claude: '--model opus', codex: '--not-a-real-codex-flag' }, + agentDefaultEnv: {} + }) + + expect(await deps.resolveLaunchArgs('claude')).toEqual(['--model', 'opus']) + expect(() => deps.resolveLaunchArgs('codex')).toThrow() + }) + + it('supplies the Claude env overlay so the launch resolver does not fall back to process.env', async () => { + const deps = await installedDeps({ + agentDefaultArgs: {}, + agentDefaultEnv: { + claude: { ORCA_CLAUDE_OVERLAY: 'claude-value' }, + codex: { ORCA_CODEX_OVERLAY: 'codex-value' } + } + }) + + expect(deps.resolveClaudeLaunchEnv).toBeTypeOf('function') + expect(deps.resolveClaudeLaunchEnv?.()).toMatchObject({ + ORCA_CLAUDE_OVERLAY: 'claude-value' + }) + expect(deps.resolveClaudeLaunchEnv?.()).not.toHaveProperty('ORCA_CODEX_OVERLAY') + expect(deps.resolveLaunchEnvOverlay()).toMatchObject({ ORCA_CODEX_OVERLAY: 'codex-value' }) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts b/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts index 9e70602f91b..89836d1e027 100644 --- a/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts +++ b/src/main/runtime/orca-runtime-structured-agent-session-launch-tui.ts @@ -28,7 +28,12 @@ export class OrcaRuntimeWithStructuredAgentSessionLaunchTui extends OrcaRuntimeW presentation: 'background' }, {}, - { spawnToken, providerRoot: record.accountHome.path, sessionId: record.sessionId } + { + spawnToken, + providerRoot: record.accountHome.path, + sessionId: record.sessionId, + ...(record.launchArgs !== undefined ? { launchArgs: record.launchArgs } : {}) + } ) const terminal = launched.terminal let spawnedOwner: StructuredTuiOwner | null = null diff --git a/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts b/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts new file mode 100644 index 00000000000..64b451e9ea9 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-claude-account-gate.test.ts @@ -0,0 +1,112 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' +import { OrcaRuntimeService } from './orca-runtime' +import { setStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import type { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +function managedAccount(id: string, managedAuthRuntime: 'host' | 'wsl') { + return { + id, + email: `${id}@example.com`, + managedAuthPath: `/managed/${id}`, + managedAuthRuntime, + authMethod: 'subscription-oauth' as const, + createdAt: 0, + updatedAt: 0, + lastAuthenticatedAt: 0 + } +} + +const WSL_ONLY: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('wsl-1', 'wsl')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: { Ubuntu: 'wsl-1' } } +} + +/** Registered Claude accounts with none selected: ambient auth, and the UI names no host identity, + * so this must reach structured rather than silently falling back to a terminal session. */ +const ACCOUNTS_PRESENT_NONE_ACTIVE: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host'), managedAccount('host-2', 'host')], + activeClaudeManagedAccountId: null, + activeClaudeManagedAccountIdsByRuntime: { host: null, wsl: {} } +} + +const HOST_SELECTED: ClaudeManagedAccountGateSettings = { + claudeManagedAccounts: [managedAccount('host-1', 'host')], + activeClaudeManagedAccountId: 'host-1', + activeClaudeManagedAccountIdsByRuntime: { host: 'host-1', wsl: {} } +} + +function runtimeWithAccounts(claude: ClaudeManagedAccountGateSettings | null): OrcaRuntimeService { + // No store at all is the unreadable-settings case the gate must fail closed on. + const runtime = claude + ? new OrcaRuntimeService({ getSettings: () => claude } as never) + : new OrcaRuntimeService() + const internal = runtime as unknown as { + resolveStructuredAgentSessionLocation: (selector: string) => Promise + ensureStructuredAgentSessionHost: () => Promise + } + internal.resolveStructuredAgentSessionLocation = vi.fn(async () => ({ + executionHostId: 'local', + wslDistro: null, + workspaceId: 'workspace-1', + workspaceKind: 'git-worktree' as const + })) + // The adapter's own location answer is irrelevant here; pin it supported so only the account + // gate can refuse. + internal.ensureStructuredAgentSessionHost = vi.fn(async () => {}) + setStructuredAgentSessionHost({ + supportsCreate: () => true + } as unknown as StructuredAgentSessionHost) + return runtime +} + +afterEach(() => { + setStructuredAgentSessionHost(null) +}) + +describe('structured Claude managed-account gate', () => { + it('refuses Claude under a WSL-only managed account', async () => { + const runtime = runtimeWithAccounts(WSL_ONLY) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: false }) + }) + + it('supports Claude when accounts are registered but none is selected', async () => { + const runtime = runtimeWithAccounts(ACCOUNTS_PRESENT_NONE_ACTIVE) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: true }) + }) + + it('still supports Claude under a selected host managed account', async () => { + const runtime = runtimeWithAccounts(HOST_SELECTED) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: true }) + }) + + it('fails closed for Claude when the account runtime cannot be determined', async () => { + const runtime = runtimeWithAccounts(null) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'claude') + ).resolves.toMatchObject({ supported: false }) + }) + + /** The gate is Claude's alone: Codex resolves its account separately and this lane must not + * change any Codex answer. */ + it('leaves Codex supported under the same WSL-only Claude account', async () => { + const runtime = runtimeWithAccounts(WSL_ONLY) + await expect( + runtime.getStructuredAgentSessionCreateSupport('id:workspace-1', 'codex') + ).resolves.toMatchObject({ supported: true }) + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts b/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts new file mode 100644 index 00000000000..0d9b12f7c52 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-claude-gate-wiring.test.ts @@ -0,0 +1,64 @@ +import { describe, expect, it, vi } from 'vitest' + +const installed = vi.hoisted(() => ({ deps: null as Record | null })) + +vi.mock('electron', () => ({ + BrowserWindow: { fromId: vi.fn(() => null) }, + webContents: { fromId: vi.fn(() => null) }, + ipcMain: { on: vi.fn(), removeListener: vi.fn() }, + app: { getPath: vi.fn(() => '/tmp') } +})) + +vi.mock('./structured-agent-session-runtime', () => ({ + ensureStructuredAgentSessionHost: vi.fn(async (deps: Record) => { + installed.deps = deps + }) +})) + +import { OrcaRuntimeService } from './orca-runtime' +import { + readClaudeManagedAccountGateSettings, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' + +const SETTINGS = { + claudeManagedAccounts: [], + activeClaudeManagedAccountId: null, + agentDefaultEnv: {}, + agentDefaultArgs: {} +} as unknown as ClaudeManagedAccountGateSettings + +function gateSettingsGetter(): (() => ClaudeManagedAccountGateSettings) | undefined { + const deps: Record = installed.deps ?? {} + const get = deps['getClaudeManagedAccountGateSettings'] + return typeof get === 'function' ? (get as () => ClaudeManagedAccountGateSettings) : undefined +} + +/** The runtime class this wiring lives on does not typecheck its own `this` calls, so a broken or + * missing gate hookup compiles clean. Pin it behaviourally instead. */ +describe('structured Claude managed-account gate wiring', () => { + it('hands the host a gate reader that resolves the live settings', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService({ getSettings: () => SETTINGS } as never) + + await runtime.ensureStructuredAgentSessionHost() + + const get = gateSettingsGetter() + expect(typeof get).toBe('function') + expect(get?.()).toBe(SETTINGS) + }) + + /** The installer composes this getter with the fail-closed reader, which is the shape the + * resolver consumes; pin that composition end to end. */ + it('composes into a null answer instead of throwing when settings cannot be read', async () => { + installed.deps = null + const runtime = new OrcaRuntimeService() + + await runtime.ensureStructuredAgentSessionHost() + + const get = gateSettingsGetter() + expect(typeof get).toBe('function') + expect(() => get?.()).toThrow() + expect(readClaudeManagedAccountGateSettings(get!)).toBeNull() + }) +}) diff --git a/src/main/runtime/orca-runtime-structured-session-restore.test.ts b/src/main/runtime/orca-runtime-structured-session-restore.test.ts index 9e752330445..d6ec1e22782 100644 --- a/src/main/runtime/orca-runtime-structured-session-restore.test.ts +++ b/src/main/runtime/orca-runtime-structured-session-restore.test.ts @@ -325,6 +325,56 @@ describe('structured session cold restoration', () => { expect(closed.tabGroups?.[0]?.tabOrder).toEqual(['terminal-tab']) }) + it('publishes restored Claude tabs with the Claude title', async () => { + const runtime = new OrcaRuntimeService() + const publish = vi.spyOn(runtime, 'publishStructuredAgentSessionTab') + const internal = runtime as unknown as { + hasPersistedStructuredAgentSessionStore(): boolean + getKnownWorkspaceSessionWorktreeIds(): Set + hydrateHeadlessMobileSessionTabsFromWorkspaceSession(): Set + refreshMobileSessionPtyRecords(): Promise | null> + ensureStructuredAgentSessionHost(): Promise + } + internal.hasPersistedStructuredAgentSessionStore = () => true + internal.getKnownWorkspaceSessionWorktreeIds = () => new Set() + internal.hydrateHeadlessMobileSessionTabsFromWorkspaceSession = () => new Set() + internal.refreshMobileSessionPtyRecords = async () => new Set() + internal.ensureStructuredAgentSessionHost = async () => undefined + setStructuredAgentSessionHost({ + reconcileRestartLeases: async () => undefined, + restoreReadableSessions: async () => undefined, + listSessionTabs: () => [ + { + sessionId: 'agent-session:agent-session:restored-claude', + workspaceId: 'workspace-1', + agent: 'claude' + } + ] + } as never) + + await runtime.restoreStructuredAgentSessionTabs() + + expect(publish).toHaveBeenCalledWith({ + workspaceId: 'workspace-1', + sessionId: 'restored-claude', + agent: 'claude', + activate: false, + notify: false + }) + + const restored = await runtime.listMobileSessionTabs('id:workspace-1') + expect(restored.tabs).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + type: 'agent-session', + id: 'agent-session:restored-claude', + title: 'Claude Chat', + agent: 'claude' + }) + ]) + ) + }) + it('commits the host close when the renderer already removed the structured tab', async () => { const runtime = new OrcaRuntimeService() runtime.setNotifier({ diff --git a/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts b/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts new file mode 100644 index 00000000000..d15f754b244 --- /dev/null +++ b/src/main/runtime/orca-runtime-structured-tui-tab-binding.test.ts @@ -0,0 +1,730 @@ +import { createHash } from 'node:crypto' +import { describe, expect, it, vi } from 'vitest' +import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' +import { createEphemeralAgentSessionClaimSigner } from './agent-session-claim-identity' +import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' +import { OrcaRuntimeService } from './orca-runtime' + +const { + probeAgentSessionProcessIdentity, + proveCodexTuiRollout, + readClaudeTranscriptLeafUuid, + readStructuredTuiProcessIdentity, + resolveSessionFilePath, + resolvePinnedCodexRolloutProof +} = vi.hoisted(() => ({ + probeAgentSessionProcessIdentity: vi.fn(), + proveCodexTuiRollout: vi.fn(), + readClaudeTranscriptLeafUuid: vi.fn(), + readStructuredTuiProcessIdentity: vi.fn(), + resolveSessionFilePath: vi.fn(), + resolvePinnedCodexRolloutProof: vi.fn() +})) + +vi.mock('./structured-tui-process-identity', () => ({ readStructuredTuiProcessIdentity })) +vi.mock('../codex/codex-tui-rollout-proof', () => ({ + proveCodexTuiRollout, + resolvePinnedCodexRolloutProof +})) +vi.mock('../native-chat/session-file-resolver', () => ({ + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +})) +vi.mock('./agent-session-process-identity-probe', async (importOriginal) => ({ + ...(await importOriginal()), + probeAgentSessionProcessIdentity +})) + +const WORKTREE_ID = 'repo-1::/tmp/structured-handoff' + +function notifier(revealTerminalSession: ReturnType) { + return { + worktreesChanged: vi.fn(), + reposChanged: vi.fn(), + activateWorktree: vi.fn(), + createTerminal: vi.fn(), + revealTerminalSession, + splitTerminal: vi.fn(), + renameTerminal: vi.fn(), + focusTerminal: vi.fn(), + closeTerminal: vi.fn(), + sleepWorktree: vi.fn(), + terminalFitOverrideChanged: vi.fn(), + terminalDriverChanged: vi.fn() + } +} + +describe('structured TUI launch tab binding', () => { + it('recovers a live TUI from durable owner inventory in a fresh runtime', async () => { + const namespace = { + machine: 'native:test', + principal: 'uid:1', + container: 'native', + providerRoot: '/tmp/codex-home' + } + const signer = createEphemeralAgentSessionClaimSigner('profile-test') + const claim = signer.createClaim({ + namespace, + identity: { agent: 'codex', providerSession: { key: 'session_id', id: 'thread-1' } }, + canonicalWorktreeId: WORKTREE_ID + }) + const terminalHandle = 'term_cold_owner' + const leafId = '23013912-13f8-44e5-818f-d40a1ff4e8c5' + resolvePinnedCodexRolloutProof.mockResolvedValue('/tmp/codex-home/sessions/thread-1.jsonl') + const writeAgentSessionProof = vi.fn(() => false) + const runtime = new OrcaRuntimeService(undefined, undefined, { + agentSessionClaimSigner: signer + }) + runtime.setPtyController({ + listProcesses: vi.fn(async () => [ + { + id: 'pty-cold-owner', + incarnationId: 'incarnation-1', + cwd: '/tmp/structured-handoff', + title: 'codex', + worktreeId: WORKTREE_ID, + terminalHandle, + agentSessionOwners: [ + { + claim, + generation: 'generation-1', + phase: 'live' as const, + ptyId: 'pty-cold-owner', + surface: { + worktreeId: WORKTREE_ID, + tabId: 'tab-cold-owner', + leafId, + terminalHandle + } + } + ] + } + ]), + write: () => true, + kill: () => true, + writeAgentSessionProof, + getForegroundProcess: async () => null + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + refreshMobileSessionPtyRecords(): Promise | null> + listResolvedWorktrees(): Promise + resolveTerminalWorkspaceLaunchScope(): Promise<{ + id: string + path: string + connectionId: null + repo: null + folderWorkspace: null + }> + getAgentSessionExecutionNamespace(): typeof namespace + ptysById: Map< + string, + { + launchToken: string | null + launchAgent: string | null + agentSessionOwners: unknown[] + tabId?: string | null + paneKey?: string | null + } + > + } + internal.listResolvedWorktrees = vi.fn(async () => [ + { id: WORKTREE_ID, repoId: 'repo-1', path: '/tmp/structured-handoff' } + ]) + internal.resolveTerminalWorkspaceLaunchScope = vi.fn(async () => ({ + id: WORKTREE_ID, + path: '/tmp/structured-handoff', + connectionId: null, + repo: null, + folderWorkspace: null + })) + internal.getAgentSessionExecutionNamespace = () => namespace + proveCodexTuiRollout.mockResolvedValueOnce({ + transcriptPath: '/tmp/codex-home/sessions/thread-1.jsonl' + }) + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + await internal.refreshMobileSessionPtyRecords() + const coldPty = internal.ptysById.get('pty-cold-owner')! + expect(coldPty).toMatchObject({ launchToken: null, launchAgent: null }) + expect(coldPty.agentSessionOwners).toHaveLength(1) + const runtimeId = (runtime as unknown as { runtimeId: string }).runtimeId + ;( + runtime as unknown as { + handles: Map< + string, + { + handle: string + runtimeId: string + rendererGraphEpoch: number + worktreeId: string + tabId: string + leafId: string + ptyId: string + ptyGeneration: number + } + > + } + ).handles.set(terminalHandle, { + handle: terminalHandle, + runtimeId, + rendererGraphEpoch: 0, + worktreeId: WORKTREE_ID, + tabId: 'pty:pty-cold-owner', + leafId: 'pty:pty-cold-owner', + ptyId: 'pty-cold-owner', + ptyGeneration: 0 + }) + coldPty.tabId = 'tab-cold-owner' + coldPty.paneKey = `tab-cold-owner:${leafId}` + coldPty.launchToken = 'spawn-token' + coldPty.launchAgent = 'codex' + + const owner = await internal.createStructuredAgentSessionHandoffTransport().recoverTuiOwner({ + sessionId: 'session-1', + location: { workspaceId: WORKTREE_ID, executionHostId: 'local' }, + accountHome: { variable: 'CODEX_HOME', path: namespace.providerRoot }, + providerHandleChain: [{ handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 }], + lease: { + ownerProcess: { + hostId: 'local', + pid: 4243, + processStartTimeMs: 10, + spawnToken: 'spawn-token' + }, + runtimeFence: 3 + } + } as never) + + expect(owner.terminal).toEqual({ + handle: terminalHandle, + tabId: 'tab-cold-owner', + paneKey: `tab-cold-owner:${leafId}`, + ptyId: 'pty-cold-owner' + }) + expect(proveCodexTuiRollout).toHaveBeenCalledWith( + expect.objectContaining({ + codexHome: namespace.providerRoot, + threadId: 'thread-1', + readOutput: expect.any(Function), + write: expect.any(Function) + }) + ) + expect(resolvePinnedCodexRolloutProof).not.toHaveBeenCalled() + expect(writeAgentSessionProof).not.toHaveBeenCalled() + expect(agentSessionPtyWriteGate.boundSessionId('pty-cold-owner')).toBe('session-1') + agentSessionPtyWriteGate.unbindPty('pty-cold-owner') + }) + + it('rebuilds a Claude proving link from current launch-token-bound hook evidence', async () => { + const paneKey = 'tab-claude:leaf-claude' + const spawnToken = 'claude-restart-token' + const sessionId = '019fd532-7c11-7a90-b6de-4e1a2c3d5f61' + const transcriptPath = '/tmp/claude-home/projects/worktree/session.jsonl' + const attestAgentHookCompatibilityAuthority = vi.fn(() => ({ + paneKey, + source: 'hydrated_commitment' as const + })) + const runtime = new OrcaRuntimeService(null, undefined, { + attestAgentHookCompatibilityAuthority, + getAgentProviderSessionRowsForPane: () => [ + { + paneKey, + connectionId: null, + state: 'done', + prompt: '', + agentType: 'claude', + receivedAt: Date.now() + 1000, + stateStartedAt: 10, + providerSession: { key: 'session_id', id: sessionId, transcriptPath } + } + ] + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + ptysById: Map + restoredOrchestrationAuthorityByPtyId: Map + } + internal.ptysById.set('pty-claude', { + ptyId: 'pty-claude', + worktreeId: WORKTREE_ID, + connectionId: null, + tabId: 'tab-claude', + paneKey, + launchToken: spawnToken, + launchAgent: 'claude', + connected: true + }) + resolveSessionFilePath.mockResolvedValue('/tmp/claude-home/projects/worktree/session.jsonl') + readClaudeTranscriptLeafUuid.mockResolvedValue('leaf-before-resume') + const record = { + sessionId: 'session-1', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/tmp/claude-home' }, + providerHandleChain: [ + { + linkId: 'claude-old', + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-before-resume' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + lease: { + runtimeFence: 3, + ownerProcess: { + hostId: 'local', + pid: 4343, + processStartTimeMs: 20, + spawnToken + }, + provenHandleLinkId: null + } + } as never + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + const transport = internal.createStructuredAgentSessionHandoffTransport() + const recovered = await transport.recoverTuiOwner(record) + expect(recovered).toMatchObject({ + transcriptPath, + link: { + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-before-resume' }, + origin: 'resumed', + mintedAtFence: 3 + } + }) + expect(recovered.link.linkId).not.toBe('claude-old') + + const pty = internal.ptysById.get('pty-claude') as { + launchToken: string | null + } + pty.launchToken = null + const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(recovered.terminal.handle)! + internal.restoredOrchestrationAuthorityByPtyId.set('pty-claude', { + ptyId: 'pty-claude', + worktreeId: WORKTREE_ID, + terminalHandle: recovered.terminal.handle, + paneKey: recovered.terminal.paneKey, + processIncarnation: dispatchAuthority.processIncarnation, + hostScope: dispatchAuthority.hostScope + }) + + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toMatchObject({ + paneKey, + terminalHandle: recovered.terminal.handle, + processIncarnation: dispatchAuthority.processIncarnation + }) + expect(attestAgentHookCompatibilityAuthority).toHaveBeenCalledWith({ + paneKey, + launchTokenHash: createHash('sha256').update(spawnToken).digest('hex'), + connectionId: null, + terminalProvenance: 'restored' + }) + attestAgentHookCompatibilityAuthority.mockReturnValueOnce(null as never) + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toBeNull() + expect(agentSessionPtyWriteGate.boundSessionId('pty-claude')).toBe('session-1') + agentSessionPtyWriteGate.unbindPty('pty-claude') + }) + + it('requires restored hook attestation after the runtime restarts', async () => { + const paneKey = 'tab-restored:leaf-restored' + const spawnToken = 'restored-token' + const sessionId = '019fd532-7c11-7a90-b6de-4e1a2c3d5f62' + const transcriptPath = '/tmp/claude-home/projects/worktree/restored.jsonl' + const attestAgentHookCompatibilityAuthority = vi.fn(() => ({ + paneKey, + source: 'hydrated_commitment' as const + })) + const runtime = new OrcaRuntimeService(null, undefined, { + attestAgentHookCompatibilityAuthority, + getAgentProviderSessionRowsForPane: () => [ + { + paneKey, + connectionId: null, + state: 'done', + prompt: '', + agentType: 'claude', + receivedAt: Date.now() + 1000, + stateStartedAt: 10, + providerSession: { key: 'session_id', id: sessionId, transcriptPath } + } + ] + }) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + ptysById: Map + restoredOrchestrationAuthorityByPtyId: Map + } + internal.ptysById.set('pty-restored', { + ptyId: 'pty-restored', + worktreeId: WORKTREE_ID, + connectionId: null, + tabId: 'tab-restored', + paneKey, + launchToken: spawnToken, + launchAgent: 'claude', + connected: true + }) + resolveSessionFilePath.mockResolvedValue('/tmp/claude-home/projects/worktree/restored.jsonl') + readClaudeTranscriptLeafUuid.mockResolvedValue('leaf-restored') + const record = { + sessionId: 'session-restored', + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/tmp/claude-home' }, + providerHandleChain: [ + { + linkId: 'claude-restored', + handle: { provider: 'claude', sessionId, leafUuid: 'leaf-restored' }, + origin: 'created', + mintedAtFence: 1, + observedAt: 1 + } + ], + lease: { + runtimeFence: 4, + ownerProcess: { + hostId: 'local', + pid: 4545, + processStartTimeMs: 30, + spawnToken + }, + provenHandleLinkId: null + } + } as never + + const recovered = await internal + .createStructuredAgentSessionHandoffTransport() + .recoverTuiOwner(record) + const restoredPty = internal.ptysById.get('pty-restored') as { + launchToken: string | null + } + restoredPty.launchToken = null + const dispatchAuthority = runtime.getOrchestrationDispatchAuthority(recovered.terminal.handle)! + internal.restoredOrchestrationAuthorityByPtyId.set('pty-restored', { + ptyId: 'pty-restored', + worktreeId: WORKTREE_ID, + terminalHandle: recovered.terminal.handle, + paneKey: recovered.terminal.paneKey, + processIncarnation: dispatchAuthority.processIncarnation, + hostScope: dispatchAuthority.hostScope + }) + + expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toMatchObject({ + paneKey, + terminalHandle: recovered.terminal.handle, + processIncarnation: dispatchAuthority.processIncarnation + }) + expect(attestAgentHookCompatibilityAuthority).toHaveBeenCalledWith({ + paneKey, + launchTokenHash: createHash('sha256').update(spawnToken).digest('hex'), + connectionId: null, + terminalProvenance: 'restored' + }) + attestAgentHookCompatibilityAuthority.mockReturnValueOnce(null as never) + await expect( + runtime.verifyOrchestrationCompatibilityCaller({ + terminalHandle: recovered.terminal.handle, + paneKey, + launchToken: spawnToken + }) + ).toBeNull() + agentSessionPtyWriteGate.unbindPty('pty-restored') + }) + + it('proves the published launch tab before returning its revealed renderer binding', async () => { + let explicitStatus: { + state: 'working' | 'done' + prompt: string + receivedAt: number + stateStartedAt: number + paneKey: string + terminalHandle: string + } | null = null + const revealTerminalSession = vi.fn( + (_worktreeId: string, _options: { tabId?: string; leafId?: string; ptyId?: string }) => + Promise.resolve({ tabId: 'tab-renderer' }) + ) + const runtime = new OrcaRuntimeService( + { + getSettings: () => ({ + disabledTuiAgents: [], + agentCmdOverrides: {}, + agentDefaultArgs: { + codex: '-m gpt-5.6-sol -c model_reasoning_effort=high' + }, + agentDefaultEnv: {} + }) + } as never, + undefined, + { + getAgentStatusSnapshot: () => (explicitStatus ? [explicitStatus as never] : []) + } + ) + runtime.setNotifier(notifier(revealTerminalSession) as never) + const spawn = vi.fn().mockResolvedValue({ id: 'pty-structured', pid: 4242 }) + runtime.setPtyController({ + spawn, + write: () => true, + kill: () => true, + getForegroundProcess: async () => null + }) + + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + resolveTerminalWorkspaceLaunchScope(): Promise<{ + id: string + path: string + connectionId: null + repo: null + folderWorkspace: null + }> + markLocalWorkspaceTrustedForAgent(): void + waitForTerminal(): Promise + waitForAdoptedStructuredTuiProof(): Promise<{ transcriptPath?: string }> + waitForStructuredTuiPtyExit(): Promise + closeTerminal(handle: string): Promise + handles: Map< + string, + { + rendererGraphEpoch: number + tabId: string + leafId: string + } + > + graphStatus: 'ready' + } + internal.resolveTerminalWorkspaceLaunchScope = vi.fn(async () => ({ + id: WORKTREE_ID, + path: '/tmp/structured-handoff', + connectionId: null, + repo: null, + folderWorkspace: null + })) + internal.markLocalWorkspaceTrustedForAgent = vi.fn() + const waitForTerminal = vi.fn(async () => ({})) + internal.waitForTerminal = waitForTerminal + const waitForAdoptedStructuredTuiProof = vi.fn(async () => { + const snapshot = await runtime.listMobileSessionTabs(`id:${WORKTREE_ID}`) + expect(snapshot.tabs).toContainEqual( + expect.objectContaining({ + type: 'terminal', + parentTabId: expect.any(String), + leafId: expect.any(String), + ptyId: 'pty-structured', + terminal: expect.any(String) + }) + ) + expect(revealTerminalSession).not.toHaveBeenCalled() + return { transcriptPath: '/tmp/rollout.jsonl' } + }) + internal.waitForAdoptedStructuredTuiProof = waitForAdoptedStructuredTuiProof + const waitForStructuredTuiPtyExit = vi.fn(async () => {}) + internal.waitForStructuredTuiPtyExit = waitForStructuredTuiPtyExit + const closeTerminal = vi.fn(async () => undefined) + internal.closeTerminal = closeTerminal + readStructuredTuiProcessIdentity.mockResolvedValue({ + hostId: 'local', + pid: 4243, + processStartTimeMs: 10, + spawnToken: 'spawn-token' + }) + probeAgentSessionProcessIdentity.mockResolvedValue({ + outcome: 'identity-matched', + matchedOn: ['process-start-time'] + }) + + const transport = internal.createStructuredAgentSessionHandoffTransport() + const onSpawned = vi.fn(async () => {}) + const owner = await transport.launchTui({ + record: { + sessionId: 'session-1', + location: { workspaceId: WORKTREE_ID, executionHostId: 'local' }, + accountHome: { variable: 'CODEX_HOME', path: '/tmp/codex-home' }, + launchArgs: ['--search'], + options: { model: 'gpt-5.6-terra', effort: 'medium' }, + providerHandleChain: [ + { handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 } + ] + } as never, + fence: 3, + spawnToken: 'spawn-token', + onSpawned + }) + + const reveal = revealTerminalSession.mock.calls[0]?.[1] as { + tabId: string + leafId: string + } + expect(owner.terminal).toMatchObject({ + tabId: 'tab-renderer', + paneKey: `${reveal.tabId}:${reveal.leafId}`, + ptyId: 'pty-structured' + }) + expect(waitForTerminal).toHaveBeenCalledWith( + expect.any(String), + expect.objectContaining({ condition: 'tui-idle' }) + ) + expect(waitForAdoptedStructuredTuiProof).toHaveBeenCalledOnce() + expect(onSpawned).toHaveBeenCalledWith( + expect.objectContaining({ + terminal: expect.objectContaining({ ptyId: 'pty-structured' }), + process: expect.objectContaining({ spawnToken: 'spawn-token' }) + }) + ) + expect(onSpawned.mock.invocationCallOrder[0]).toBeLessThan( + waitForTerminal.mock.invocationCallOrder[0]! + ) + expect(waitForAdoptedStructuredTuiProof.mock.invocationCallOrder[0]).toBeLessThan( + revealTerminalSession.mock.invocationCallOrder[0]! + ) + const launchCommand = spawn.mock.calls[0]?.[0]?.command + expect(launchCommand).toContain("'-m' 'gpt-5.6-terra'") + expect(launchCommand).toContain("'-c' 'model_reasoning_effort=medium'") + expect(launchCommand).toContain("'--search'") + expect(launchCommand).not.toContain('gpt-5.6-sol') + expect(launchCommand).not.toContain('model_reasoning_effort=high') + + Object.assign(internal.handles.get(owner.terminal.handle)!, { + rendererGraphEpoch: -1, + tabId: 'tab-retired', + leafId: 'leaf-retired' + }) + internal.graphStatus = 'ready' + + explicitStatus = { + state: 'working', + prompt: '', + receivedAt: Date.now(), + stateStartedAt: Date.now(), + paneKey: owner.terminal.paneKey, + terminalHandle: owner.terminal.handle + } + expect(transport.tuiStatus(owner)).toBe('busy') + await expect( + transport.waitForTuiIdleOrExit(owner, new AbortController().signal) + ).resolves.toBeNull() + + explicitStatus = { ...explicitStatus, state: 'done', receivedAt: Date.now() } + expect(transport.tuiStatus(owner)).toBe('idle') + await expect(transport.waitForTuiIdleOrExit(owner, new AbortController().signal)).resolves.toBe( + 'idle' + ) + + explicitStatus = null + const livePty = ( + runtime as unknown as { + ptysById: Map< + string, + { + tailBuffer: string[] + tailPartialLine: string + preview: string + lastAgentStatus: null + lastAgentStatusObservedLive: boolean + } + > + } + ).ptysById.get('pty-structured')! + Object.assign(livePty, { + tailBuffer: [ + 'OpenAI Codex (v0.147.0)', + 'model: gpt-5.6-terra', + 'directory: /tmp/structured-handoff' + ], + tailPartialLine: '', + preview: '', + lastAgentStatus: null, + lastAgentStatusObservedLive: false + }) + expect(transport.tuiStatus(owner)).toBe('idle') + await expect(transport.waitForTuiIdleOrExit(owner, new AbortController().signal)).resolves.toBe( + 'idle' + ) + + const pty = ( + runtime as unknown as { + ptysById: Map + } + ).ptysById.get('pty-structured')! + pty.launchToken = null + const persistedRecord = { + sessionId: 'session-1', + providerHandleChain: [{ handle: { provider: 'codex', threadId: 'thread-1' }, observedAt: 1 }], + lease: { ownerProcess: owner.process, provenHandleLinkId: owner.link.linkId } + } as never + + const rebound = await transport.reproveTuiOwner({ record: persistedRecord, owner }) + expect(rebound.terminal).toMatchObject({ + ptyId: 'pty-structured', + tabId: owner.terminal.tabId, + paneKey: owner.terminal.paneKey + }) + expect(rebound.terminal.handle).not.toBe(owner.terminal.handle) + await transport.waitForTuiExit(rebound) + expect(waitForStructuredTuiPtyExit).toHaveBeenCalledWith('pty-structured') + expect(waitForAdoptedStructuredTuiProof).toHaveBeenCalledOnce() + + await expect(transport.closeTuiOwner?.(rebound)).resolves.toEqual({ + transcriptPath: '/tmp/rollout.jsonl' + }) + expect(closeTerminal).toHaveBeenCalledWith(rebound.terminal.handle) + + explicitStatus = null + pty.connected = false + await expect( + transport.waitForTuiIdleOrExit(rebound, new AbortController().signal) + ).resolves.toBe('exited') + await expect(transport.stopFailedTuiLaunch?.(rebound)).resolves.toBeUndefined() + }) + + it('reveals Claude structured native sessions into the mobile graph', async () => { + const runtime = new OrcaRuntimeService() + const publish = vi.spyOn(runtime, 'publishStructuredAgentSessionTab') + const focusEditorTab = vi.fn() + runtime.setNotifier({ focusEditorTab } as never) + const internal = runtime as unknown as { + createStructuredAgentSessionHandoffTransport(): StructuredAgentSessionHandoffTransport + } + + await internal.createStructuredAgentSessionHandoffTransport().revealNativeSession?.({ + workspaceId: WORKTREE_ID, + sessionId: 'session-claude', + agent: 'claude' + }) + + expect(publish).toHaveBeenCalledWith( + expect.objectContaining({ + workspaceId: WORKTREE_ID, + sessionId: 'session-claude', + agent: 'claude', + activate: false + }) + ) + expect(focusEditorTab).toHaveBeenCalledWith( + 'structured-agent-session-session-claude', + WORKTREE_ID + ) + }) +}) diff --git a/src/main/runtime/rpc/e2ee-channel-v2.test.ts b/src/main/runtime/rpc/e2ee-channel-v2.test.ts index f9e602ced41..b26b057abaf 100644 --- a/src/main/runtime/rpc/e2ee-channel-v2.test.ts +++ b/src/main/runtime/rpc/e2ee-channel-v2.test.ts @@ -156,6 +156,25 @@ describe('E2EEChannel v2', () => { }) }) + it('forwards post-auth capability-shaped frames without mutating authenticated capabilities', () => { + const ctx = setup() + const { schedule } = startV2(ctx) + const onMessage = vi.fn() + ctx.channel.onMessage(onMessage) + authenticate(ctx, schedule) + + const capabilityFrame = JSON.stringify({ + type: 'e2ee_client_capabilities', + v: 1, + clientCapabilities: ['agent-session.structured.v1'] + }) + ctx.channel.handleRawMessage(clientText(capabilityFrame, schedule, 1n)) + + expect(ctx.channel.clientCapabilities).toEqual([]) + expect(onMessage).toHaveBeenCalledOnce() + expect(onMessage.mock.calls[0]?.[0]).toBe(capabilityFrame) + }) + it('rejects legacy downgrade and runtime-only capability metadata when mobile v2 is required', () => { const legacy = setup() legacy.channel.handleRawMessage( diff --git a/src/main/runtime/rpc/methods/clipboard.test.ts b/src/main/runtime/rpc/methods/clipboard.test.ts index 118b21c766a..c0d224b85c6 100644 --- a/src/main/runtime/rpc/methods/clipboard.test.ts +++ b/src/main/runtime/rpc/methods/clipboard.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' import { RpcDispatcher } from '../dispatcher' -import type { RpcRequest } from '../core' +import type { RpcRequest, RpcResponse } from '../core' import type { OrcaRuntimeService } from '../../orca-runtime' import { CLIPBOARD_IMAGE_MAX_BASE64_CHARS, @@ -21,6 +21,10 @@ import { CLIPBOARD_METHODS, resetClipboardImageUploadsForTest } from './clipboard' +import { + hasMobileClipboardImagePath, + resetMobileClipboardImageProvenanceForTest +} from '../mobile-clipboard-image-provenance' function makeRequest(method: string, params?: unknown): RpcRequest { return { id: 'req-1', authToken: 'tok', method, params } @@ -31,15 +35,36 @@ function makeDispatcher(): RpcDispatcher { return new RpcDispatcher({ runtime, methods: CLIPBOARD_METHODS }) } +async function callMobile( + dispatcher: RpcDispatcher, + method: string, + params: unknown, + clientId = 'device-a' +): Promise { + const replies: RpcResponse[] = [] + await dispatcher.dispatchStreaming( + makeRequest(method, params), + (raw) => replies.push(JSON.parse(raw) as RpcResponse), + { clientKind: 'mobile', clientId } + ) + const response = replies[0] + if (!response) { + throw new Error(`no reply for ${method}`) + } + return response +} + describe('clipboard RPC methods', () => { beforeEach(() => { saveClipboardImageBufferAsTempFile.mockReset() resetClipboardImageUploadsForTest() + resetMobileClipboardImageProvenanceForTest() }) afterEach(() => { vi.useRealTimers() resetClipboardImageUploadsForTest() + resetMobileClipboardImageProvenanceForTest() }) it('saves browser-provided clipboard image bytes on the runtime host', async () => { @@ -64,6 +89,37 @@ describe('clipboard RPC methods', () => { }) }) + it('records a successful direct mobile upload for only the authenticated client', async () => { + const path = '/tmp/orca-paste-image.png' + saveClipboardImageBufferAsTempFile.mockResolvedValue(path) + const dispatcher = makeDispatcher() + + await expect( + callMobile(dispatcher, 'clipboard.saveImageAsTempFile', { + contentBase64: Buffer.from('png-bytes').toString('base64'), + connectionId: null + }) + ).resolves.toMatchObject({ ok: true, result: path }) + + expect(hasMobileClipboardImagePath('device-a', path)).toBe(true) + expect(hasMobileClipboardImagePath('device-b', path)).toBe(false) + }) + + it('does not authorize a remote-host clipboard path for local structured delivery', async () => { + const path = '/tmp/orca-paste-image.png' + saveClipboardImageBufferAsTempFile.mockResolvedValue(path) + const dispatcher = makeDispatcher() + + await expect( + callMobile(dispatcher, 'clipboard.saveImageAsTempFile', { + contentBase64: Buffer.from('png-bytes').toString('base64'), + connectionId: 'ssh-1' + }) + ).resolves.toMatchObject({ ok: true, result: path }) + + expect(hasMobileClipboardImagePath('device-a', path)).toBe(false) + }) + it('rejects non-base64 clipboard image payloads', async () => { const dispatcher = makeDispatcher() @@ -140,6 +196,48 @@ describe('clipboard RPC methods', () => { expect(saveClipboardImageBufferAsTempFile).toHaveBeenCalledWith(Buffer.from('png-bytes'), { connectionId: 'ssh-1' }) + expect(hasMobileClipboardImagePath('device-a', '/tmp/orca-paste-image.png')).toBe(false) + }) + + it('binds chunk mutation and provenance to the mobile client that started the upload', async () => { + saveClipboardImageBufferAsTempFile.mockResolvedValue('/tmp/orca-paste-image.png') + const dispatcher = makeDispatcher() + const contentBase64 = Buffer.from('png-bytes').toString('base64') + const start = await callMobile(dispatcher, 'clipboard.startImageUpload', { + expectedBase64Length: contentBase64.length, + connectionId: null + }) + const uploadId = (start.ok ? start.result : null) as { uploadId: string } + + for (const method of [ + 'clipboard.appendImageUploadChunk', + 'clipboard.commitImageUpload', + 'clipboard.abortImageUpload' + ]) { + const params = + method === 'clipboard.appendImageUploadChunk' + ? { uploadId: uploadId.uploadId, offset: 0, contentBase64 } + : { uploadId: uploadId.uploadId } + await expect(callMobile(dispatcher, method, params, 'device-b')).resolves.toMatchObject({ + ok: false + }) + } + + await expect( + callMobile(dispatcher, 'clipboard.appendImageUploadChunk', { + uploadId: uploadId.uploadId, + offset: 0, + contentBase64 + }) + ).resolves.toMatchObject({ + ok: true, + result: { receivedBase64Length: contentBase64.length } + }) + await expect( + callMobile(dispatcher, 'clipboard.commitImageUpload', { uploadId: uploadId.uploadId }) + ).resolves.toMatchObject({ ok: true, result: '/tmp/orca-paste-image.png' }) + expect(hasMobileClipboardImagePath('device-a', '/tmp/orca-paste-image.png')).toBe(true) + expect(hasMobileClipboardImagePath('device-b', '/tmp/orca-paste-image.png')).toBe(false) }) it('rejects out-of-order chunk offsets', async () => { diff --git a/src/main/runtime/rpc/methods/clipboard.ts b/src/main/runtime/rpc/methods/clipboard.ts index 3d5212c7a52..e6b487d7761 100644 --- a/src/main/runtime/rpc/methods/clipboard.ts +++ b/src/main/runtime/rpc/methods/clipboard.ts @@ -1,11 +1,12 @@ import { z } from 'zod' -import { defineMethod, type RpcMethod } from '../core' +import { defineMethod, type RpcContext, type RpcMethod } from '../core' import { saveClipboardImageBufferAsTempFile } from '../../../window/clipboard-image-temp-file' import { randomUUID } from 'node:crypto' import { CLIPBOARD_IMAGE_MAX_BASE64_CHARS, CLIPBOARD_IMAGE_TOO_LARGE_ERROR } from '../../../../shared/clipboard-image' +import { recordMobileClipboardImagePath } from '../mobile-clipboard-image-provenance' const MAX_CLIPBOARD_IMAGE_BASE64_CHARS = CLIPBOARD_IMAGE_MAX_BASE64_CHARS export const CLIPBOARD_IMAGE_UPLOAD_CHUNK_BASE64_CHARS = 512 * 1024 @@ -16,6 +17,7 @@ const BASE64_PATTERN = /^[A-Za-z0-9+/]*={0,2}$/ type ClipboardImageUpload = { expectedBase64Length: number connectionId?: string | null + mobileClientId?: string chunks: string[] receivedBase64Length: number expiresAt: number @@ -69,6 +71,28 @@ function getUpload(uploadId: string): ClipboardImageUpload { return upload } +function mobileClientId(ctx: RpcContext): string | undefined { + if (ctx.clientKind !== 'mobile') { + return undefined + } + const clientId = ctx.clientId?.trim() + if (!clientId) { + throw new Error('Clipboard image upload requires an authenticated mobile client') + } + return clientId +} + +function assertMobileUploadOwner( + upload: ClipboardImageUpload, + ctx: RpcContext +): string | undefined { + const clientId = mobileClientId(ctx) + if (clientId && upload.mobileClientId !== clientId) { + throw new Error('Clipboard image upload was not found') + } + return clientId +} + function assertValidBase64Content(value: string): void { if (!isValidBase64(value)) { throw new Error('Clipboard image content must be base64') @@ -131,15 +155,24 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.saveImageAsTempFile', params: SaveImageAsTempFile, - handler: async (params) => - saveClipboardImageBufferAsTempFile(Buffer.from(params.contentBase64, 'base64'), { - connectionId: params.connectionId - }) + handler: async (params, ctx) => { + const clientId = mobileClientId(ctx) + const path = await saveClipboardImageBufferAsTempFile( + Buffer.from(params.contentBase64, 'base64'), + { + connectionId: params.connectionId + } + ) + if (clientId && !params.connectionId) { + recordMobileClipboardImagePath(clientId, path) + } + return path + } }), defineMethod({ name: 'clipboard.startImageUpload', params: StartImageUpload, - handler: (params) => { + handler: (params, ctx) => { pruneExpiredUploads() if (clipboardImageUploads.size >= CLIPBOARD_IMAGE_UPLOAD_MAX_CONCURRENT) { throw new Error('Too many clipboard image uploads are in progress') @@ -148,6 +181,7 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ clipboardImageUploads.set(uploadId, { expectedBase64Length: params.expectedBase64Length, connectionId: params.connectionId, + mobileClientId: mobileClientId(ctx), chunks: [], receivedBase64Length: 0, expiresAt: Date.now() + CLIPBOARD_IMAGE_UPLOAD_TTL_MS, @@ -159,8 +193,9 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.appendImageUploadChunk', params: AppendImageUploadChunk, - handler: (params) => { + handler: (params, ctx) => { const upload = getUpload(params.uploadId) + assertMobileUploadOwner(upload, ctx) if (params.offset !== upload.receivedBase64Length) { throw new Error('Clipboard image chunk offset is out of order') } @@ -177,17 +212,25 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.commitImageUpload', params: CommitImageUpload, - handler: async (params) => { + handler: async (params, ctx) => { const upload = getUpload(params.uploadId) + const clientId = assertMobileUploadOwner(upload, ctx) try { if (upload.receivedBase64Length !== upload.expectedBase64Length) { throw new Error('Clipboard image upload is incomplete') } const contentBase64 = upload.chunks.join('') assertValidBase64Content(contentBase64) - return await saveClipboardImageBufferAsTempFile(Buffer.from(contentBase64, 'base64'), { - connectionId: upload.connectionId - }) + const path = await saveClipboardImageBufferAsTempFile( + Buffer.from(contentBase64, 'base64'), + { + connectionId: upload.connectionId + } + ) + if (clientId && !upload.connectionId) { + recordMobileClipboardImagePath(clientId, path) + } + return path } finally { // Why: failed SSH or filesystem commits must not leave bounded upload // memory pinned until TTL cleanup. @@ -198,7 +241,12 @@ export const CLIPBOARD_METHODS: RpcMethod[] = [ defineMethod({ name: 'clipboard.abortImageUpload', params: AbortImageUpload, - handler: (params) => { + handler: (params, ctx) => { + pruneExpiredUploads() + const upload = clipboardImageUploads.get(params.uploadId) + if (upload) { + assertMobileUploadOwner(upload, ctx) + } deleteUpload(params.uploadId) return { aborted: true } } diff --git a/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts new file mode 100644 index 00000000000..dcba8b7b64e --- /dev/null +++ b/src/main/runtime/rpc/methods/mobile-markdown-tab-methods.ts @@ -0,0 +1,22 @@ +import { defineMethod, type RpcAnyMethod } from '../core' +import { ActivateTab, SaveMarkdownTab } from './session-tabs-schemas' + +export const MOBILE_MARKDOWN_TAB_METHODS: RpcAnyMethod[] = [ + defineMethod({ + name: 'markdown.readTab', + params: ActivateTab, + handler: async (params, { runtime }) => + runtime.readMobileMarkdownTab(params.worktree, params.tabId) + }), + defineMethod({ + name: 'markdown.saveTab', + params: SaveMarkdownTab, + handler: async (params, { runtime }) => + runtime.saveMobileMarkdownTab( + params.worktree, + params.tabId, + params.baseVersion, + params.content + ) + }) +] diff --git a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts index 226277f6ebd..a2a93533b6f 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-capability-mutations.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from 'vitest' import { + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' @@ -67,13 +68,24 @@ describe('session tab structured capability mutations', () => { expect(fixture.calls[method.runtimeMethod]).toHaveBeenCalledOnce() }) - it(`rejects ${method.name} for a legacy Claude row`, async () => { + it(`rejects ${method.name} on a Claude row the client never negotiated`, async () => { const fixture = createFixture([STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]) const response = await fixture.dispatch(method.name, method.params('claude-session')) expect(response.ok).toBe(false) expect(fixture.calls[method.runtimeMethod]).not.toHaveBeenCalled() }) + + it(`allows ${method.name} for a client that negotiated Claude rows`, async () => { + const fixture = createFixture([ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ]) + const response = await fixture.dispatch(method.name, method.params('claude-session')) + + expect(response.ok).toBe(true) + expect(fixture.calls[method.runtimeMethod]).toHaveBeenCalledOnce() + }) } it.each(['session.tabs.close', 'session.tabs.closeLifecycle'] as const)( diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts index 4a61a99bc20..7128f756d6e 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from 'vitest' import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY } from '../../../../shared/protocol-version' import type { RuntimeMobileSessionTabsSnapshot } from '../../../../shared/runtime-types' @@ -119,44 +120,105 @@ describe('projectSessionTabAgentStatus', () => { expect(capable).toBe(snapshot) }) - it('withholds legacy Claude rows from paired structured clients', () => { - const snapshot = { - ...makeSnapshot(false), - tabs: [ - { - type: 'agent-session', - id: 'agent-session:codex', - title: 'Codex Chat', - sessionId: 'codex', - agent: 'codex', - isActive: true - }, - { - type: 'agent-session', - id: 'agent-session:claude', - title: 'Claude Chat', - sessionId: 'claude', - agent: 'claude', - isActive: false - } - ], - activeTabId: 'agent-session:codex', - activeTabType: 'agent-session' + const claudeSnapshot = { + ...makeSnapshot(false), + tabs: [ + { + type: 'agent-session', + id: 'agent-session:codex', + title: 'Codex Chat', + sessionId: 'codex', + agent: 'codex', + isActive: true + }, + { + type: 'agent-session', + id: 'agent-session:claude', + title: 'Claude Chat', + sessionId: 'claude', + agent: 'claude', + isActive: false + } + ], + activeGroupId: 'group-a', + activeTabId: 'agent-session:codex', + activeTabType: 'agent-session', + tabGroups: [ + { id: 'group-a', activeTabId: 'agent-session:codex', tabOrder: ['agent-session:codex'] }, + { id: 'group-b', activeTabId: 'agent-session:claude', tabOrder: ['agent-session:claude'] } + ], + tabGroupLayout: { + type: 'split', + direction: 'horizontal', + first: { type: 'leaf', groupId: 'group-a' }, + second: { type: 'leaf', groupId: 'group-b' } + } + } as unknown as RuntimeMobileSessionTabsSnapshot + + const structuredMobile = [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] + + it.each([ + ['mobile', 'mobile' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]], + ['runtime', 'runtime' as const, [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY]] + ])( + 'withholds Claude rows from a paired %s client that never negotiated them', + (_name, clientKind, capabilities) => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + + expect(projected.tabs.map((tab) => tab.id)).toEqual(['agent-session:codex']) + // A row pruned from `tabs` but left in the layout is its own dead tab. + expect(projected.tabGroups?.map((group) => group.id)).toEqual(['group-a']) + expect(projected.tabGroupLayout).toEqual({ type: 'leaf', groupId: 'group-a' }) + expect(projected.activeGroupId).toBe('group-a') + expect(projected.activeTabId).toBe('agent-session:codex') + expect(projected.activeTabType).toBe('agent-session') + } + ) + + it.each([ + ['mobile', 'mobile' as const, structuredMobile], + [ + 'runtime', + 'runtime' as const, + [ + STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY + ] + ] + ])( + 'publishes Claude rows to a paired %s client that negotiated them', + (_name, clientKind, capabilities) => { + const projected = projectSessionTabAgentStatus(claudeSnapshot, clientKind, capabilities, true) + + expect(projected).toBe(claudeSnapshot) + expect(projected.tabGroupLayout).toEqual(claudeSnapshot.tabGroupLayout) + } + ) + + it('keeps Claude rows on the local renderer, which negotiates nothing', () => { + expect(projectSessionTabAgentStatus(claudeSnapshot, undefined, undefined)).toBe(claudeSnapshot) + expect(projectSessionTabAgentStatus(claudeSnapshot, undefined, [])).toBe(claudeSnapshot) + }) + + it('leaves Codex rows untouched whether or not the Claude capability is present', () => { + const codexOnly = { + ...claudeSnapshot, + tabs: claudeSnapshot.tabs.filter((tab) => tab.id !== 'agent-session:claude'), + tabGroups: claudeSnapshot.tabGroups?.filter((group) => group.id !== 'group-b'), + tabGroupLayout: { type: 'leaf', groupId: 'group-a' } } as unknown as RuntimeMobileSessionTabsSnapshot - expect( - projectSessionTabAgentStatus(snapshot, 'runtime', [ - STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY - ]).tabs.map((tab) => tab.id) - ).toEqual(['agent-session:codex']) - expect( - projectSessionTabAgentStatus( - snapshot, - 'mobile', - [STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], - true - ).tabs.map((tab) => tab.id) - ).toEqual(['agent-session:codex']) + for (const capabilities of [[STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY], structuredMobile]) { + for (const clientKind of ['mobile', 'runtime'] as const) { + expect(projectSessionTabAgentStatus(codexOnly, clientKind, capabilities, true)).toBe( + codexOnly + ) + } + } + expect(projectSessionTabAgentStatus(codexOnly, undefined, undefined)).toBe(codexOnly) }) it('withholds session boundaries from legacy paired clients', () => { diff --git a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts index 375b3b499d5..0e0d9c716a5 100644 --- a/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts +++ b/src/main/runtime/rpc/methods/session-tab-agent-status-projection.ts @@ -1,5 +1,6 @@ import { AGENT_SESSION_BOUNDARY_RUNTIME_CAPABILITY, + CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY, type RuntimeCapability } from '../../../../shared/protocol-version' import type { @@ -24,7 +25,14 @@ export function projectSessionTabAgentStatus true) - if (structuredVisible && clientKind !== undefined) { + // Why: a paired client renders only codex structured tabs unless it says otherwise + // (mobile's resolveMobileNativeChat returns null for every other agent), so an + // ungated row would list and select into a pane that shows neither chat nor terminal. + if ( + structuredVisible && + clientKind !== undefined && + !clientCapabilities?.includes(CLAUDE_STRUCTURED_AGENT_SESSION_RUNTIME_CAPABILITY) + ) { projected = projectAgentSessionTabsOut(projected, (tab) => tab.agent !== 'codex') } // Why: only paired runtimes have legacy `done` completion side effects; mobile must keep its row without changing the exact v2 auth shape. diff --git a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts index 6a9372ed2c1..4f7c1b20cc3 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session-schemas.ts @@ -98,7 +98,7 @@ export const CreateIntentParams = z .object({ envelope: MutationEnvelope, worktree: Identifier('Invalid worktree selector'), - agent: z.literal('codex') + agent: z.enum(['claude', 'codex']) }) .strict() @@ -107,7 +107,7 @@ export const CreateParams = z.union([AttachParams, CreateIntentParams]) export const CreateSupportParams = z .object({ worktree: Identifier('Invalid worktree selector'), - agent: z.literal('codex') + agent: z.enum(['claude', 'codex']) }) .strict() @@ -170,6 +170,15 @@ export const SetOptionParams = z }) .strict() +export const HandoffParams = z + .object({ + envelope: MutationEnvelope, + direction: z.enum(['to-tui', 'to-native']), + mode: z.enum(['now', 'after-turn', 'stop-turn']), + action: z.enum(['start', 'cancel-queued', 'retry', 'recover']).optional() + }) + .strict() + export const OptionsParams = z.object({ sessionId: SessionId }).strict() /** One surface's claim on one session. The id names the surface, not the client: two chat views diff --git a/src/main/runtime/rpc/methods/structured-agent-session.test.ts b/src/main/runtime/rpc/methods/structured-agent-session.test.ts index b65e6eff825..155d0aa6768 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.test.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.test.ts @@ -99,6 +99,22 @@ function hostStub(): StructuredAgentSessionHost { setSessionTabVisibility: vi.fn(async () => undefined), respondToPrompt: vi.fn(async () => ({ ok: true, replayed: false })), setOption: vi.fn(async () => ({ ok: true, replayed: false })), + requestHandoff: vi.fn(async () => ({ + ok: true, + replayed: false, + fence: 1, + cursor: { epoch: 'epoch-a', sequence: 0 }, + value: { + status: { + owner: 'native', + direction: null, + phase: 'idle', + stage: null, + operationId: null + } + } + })), + supportsCreate: vi.fn(() => true), handoffStatus: vi.fn(async () => ({ owner: 'native' })), readOptions: vi.fn(async () => ({ models: [{ id: 'gpt-live', label: 'GPT Live', isDefault: true, efforts: [] }], @@ -122,9 +138,12 @@ function dispatcher(runtimeOverrides: Record = {}): RpcDispatch workspaceId: 'workspace-1', workspaceKind: 'git-worktree' }, - provider: 'codex', - agent: 'codex', - accountHome: { variable: 'CODEX_HOME', path: '/host/.codex' }, + provider: params.agent, + agent: params.agent, + accountHome: { + variable: params.agent === 'claude' ? 'CLAUDE_CONFIG_DIR' : 'CODEX_HOME', + path: params.agent === 'claude' ? '/host/.claude' : '/host/.codex' + }, runtimeKind: 'native' })), publishStructuredAgentSessionTab: vi.fn() @@ -221,7 +240,7 @@ describe('capability gating', () => { } // Bump deliberately: the whole agentSession.* surface is behind the structured capability, // so an additive method is invisible to old clients and needs no protocol bump. - expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(16) + expect(STRUCTURED_AGENT_SESSION_METHODS).toHaveLength(17) }) it('hides the surface from a declared client that did not advertise it', async () => { @@ -329,6 +348,49 @@ describe('method routing', () => { ) }) + it('routes Claude create support and create through the provider-aware runtime', async () => { + const worktree = 'id:workspace-1' + const support = await call( + 'agentSession.createSupport', + { worktree, agent: 'claude' }, + STRUCTURED_CLIENT + ) + expect(support).toMatchObject({ ok: true, result: { supported: true } }) + expect(runtimeCalls.getStructuredAgentSessionCreateSupport).toHaveBeenCalledWith( + worktree, + 'claude' + ) + + const params = { + envelope: envelope({ + expectedRuntimeFence: null, + payloadFingerprint: computeAgentSessionPayloadFingerprint({ + method: 'agentSession.create', + sessionId: SESSION, + fields: { worktree, agent: 'claude' } + }) + }), + worktree, + agent: 'claude' + } + const created = await call('agentSession.create', params, STRUCTURED_CLIENT) + expect(created).toMatchObject({ ok: true, result: { ok: true } }) + expect(runtimeCalls.resolveStructuredAgentSessionCreateIntent).toHaveBeenCalledWith(params) + expect(hostCalls.attach).toHaveBeenCalledWith( + expect.anything(), + expect.objectContaining({ + accountHome: { variable: 'CLAUDE_CONFIG_DIR', path: '/host/.claude' } + }) + ) + expect(runtimeCalls.publishStructuredAgentSessionTab).toHaveBeenCalledWith( + expect.objectContaining({ + sessionId: SESSION, + activate: true, + agent: 'claude' + }) + ) + }) + it('reports an unknown create outcome when attach commits before tab publication fails', async () => { const worktree = 'id:workspace-1' const params = { @@ -371,6 +433,25 @@ describe('method routing', () => { expect(ensured).toMatchObject({ ok: true }) }) + /** A client-supplied location skips the worktree-resolving support check, so both attach-shaped + * entries must ask the executing host directly or a host that cannot fence a provider child + * would create one anyway. */ + it.each(['agentSession.create', 'agentSession.ensure'])( + 'refuses %s for a client-supplied location the executing host does not support', + async (method) => { + hostCalls.supportsCreate.mockReturnValue(false) + + const refused = await call(method, attachParams()) + + expect(refused).toMatchObject({ + ok: false, + error: { message: expect.stringContaining('structured_agent_session_unsupported') } + }) + expect(hostCalls.attach).not.toHaveBeenCalled() + expect(hostCalls.supportsCreate).toHaveBeenCalledWith(attachParams().location, 'codex') + } + ) + it('tags the prompt kind from the method name, not from the client', async () => { const params = { envelope: envelope(), @@ -386,7 +467,7 @@ describe('method routing', () => { ]) }) - it('does not register the structured handoff mutation', async () => { + it('routes the structured handoff mutation through the host', async () => { const response = await call('agentSession.requestHandoff', { envelope: envelope(), direction: 'to-tui', @@ -394,7 +475,11 @@ describe('method routing', () => { action: 'start' }) - expect(response).toMatchObject({ ok: false, error: { code: 'method_not_found' } }) + expect(response).toMatchObject({ ok: true }) + expect(hostCalls.requestHandoff).toHaveBeenCalledWith( + expect.anything(), + expect.objectContaining({ direction: 'to-tui', mode: 'now', action: 'start' }) + ) }) }) @@ -434,25 +519,6 @@ describe('parameter validation', () => { ) }) - it('rejects Claude structured create shapes', async () => { - await rejects('agentSession.createSupport', { - worktree: 'id:workspace-1', - agent: 'claude' - }) - const fields = { worktree: 'id:workspace-1', agent: 'claude' } - await rejects('agentSession.create', { - envelope: envelope({ - expectedRuntimeFence: null, - payloadFingerprint: computeAgentSessionPayloadFingerprint({ - method: 'agentSession.create', - sessionId: SESSION, - fields - }) - }), - ...fields - }) - }) - it('requires a sha256 fingerprint and a positive fence', async () => { await rejects( 'agentSession.send', diff --git a/src/main/runtime/rpc/methods/structured-agent-session.ts b/src/main/runtime/rpc/methods/structured-agent-session.ts index ffd23499a3e..3b18f6b0ef1 100644 --- a/src/main/runtime/rpc/methods/structured-agent-session.ts +++ b/src/main/runtime/rpc/methods/structured-agent-session.ts @@ -10,6 +10,7 @@ import { agentSessionFingerprintConflict, computeAgentSessionPayloadFingerprint } from '../../../../shared/agent-session-mutation-envelope' +import type { z } from 'zod' import { defineMethod, defineStreamingMethod, type RpcAnyMethod, type RpcContext } from '../core' import { ensureStructuredHostInstalled as ensureHostInstalled, @@ -18,6 +19,7 @@ import { structuredCallerFor as callerFor, supportsStructuredSessions } from './structured-agent-session-gate' +import type { AgentSessionAttachParams } from '../../../native-chat/agent-session-wire/structured-agent-session-attach' import { STRUCTURED_AGENT_SESSION_HOLD_METHODS } from './structured-agent-session-hold' import { AttachParams, @@ -25,6 +27,7 @@ import { CreateParams, CreateSupportParams, HistoryParams, + HandoffParams, HandoffStatusParams, OptionsParams, RespondParams, @@ -43,6 +46,29 @@ function subscriptionIdFor(ctx: RpcContext, sessionId: string): string { return ctx.requestId ? `${base}:${ctx.requestId}` : base } +/** + * The attach-shaped entries take the location from the client instead of resolving it from a + * worktree, so they never reach the worktree-resolving create-support check. Ask the executing + * host the same question directly: the answer includes host-measured facts the client cannot see + * or forge, such as whether this machine can read a provider child's process start time. + */ +async function attachClientSuppliedLocation( + params: z.infer, + ctx: RpcContext +): Promise { + await ensureHostInstalled(ctx) + const host = requireHost(ctx) + if (!host.supportsCreate(params.location, params.agent)) { + throw new Error('structured_agent_session_unsupported') + } + const { agent: _attachAgent, provider: _attachProvider, ...attachWithoutAgent } = params + return host.attach(callerFor(ctx), { + ...attachWithoutAgent, + provider: params.provider as 'claude' | 'codex', + agent: params.agent as 'claude' | 'codex' + } as AgentSessionAttachParams) +} + export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ defineMethod({ name: 'agentSession.createSupport', @@ -86,16 +112,20 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ } }) await ensureHostInstalled(ctx) - const result = await requireHost(ctx).attach(callerFor(ctx), { - ...resolved, + const { agent: _resolvedAgent, provider: _resolvedProvider, ...resolvedAttach } = resolved + const attachParams: AgentSessionAttachParams = { + ...resolvedAttach, + provider: resolved.provider as 'claude' | 'codex', + agent: resolved.agent as 'claude' | 'codex', envelope: { ...params.envelope, payloadFingerprint: hostFingerprint } - }) - if (result.ok && resolved.agent === 'codex') { + } + const result = await requireHost(ctx).attach(callerFor(ctx), attachParams) + if (result.ok) { try { await ctx.runtime.publishStructuredAgentSessionTab({ workspaceId: resolved.location.workspaceId, sessionId: result.value.sessionId, - agent: 'codex', + agent: resolved.agent as 'claude' | 'codex', activate: true }) } catch (error) { @@ -104,24 +134,20 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ ok: false, refusal: { code: 'agent_session_operation_unknown', - message: 'The Codex chat may have been created, but its tab could not be confirmed.' + message: 'The chat may have been created, but its tab could not be confirmed.' } } } } return result } - await ensureHostInstalled(ctx) - return requireHost(ctx).attach(callerFor(ctx), params) + return attachClientSuppliedLocation(params, ctx) } }), defineMethod({ name: 'agentSession.ensure', params: AttachParams, - handler: async (params, ctx) => { - await ensureHostInstalled(ctx) - return requireHost(ctx).attach(callerFor(ctx), params) - } + handler: async (params, ctx) => attachClientSuppliedLocation(params, ctx) }), defineMethod({ name: 'agentSession.send', @@ -165,6 +191,11 @@ export const STRUCTURED_AGENT_SESSION_METHODS: RpcAnyMethod[] = [ params: SetOptionParams, handler: async (params, ctx) => requireHost(ctx).setOption(callerFor(ctx), params) }), + defineMethod({ + name: 'agentSession.requestHandoff', + params: HandoffParams, + handler: async (params, ctx) => requireHost(ctx).requestHandoff(callerFor(ctx), params) + }), defineMethod({ name: 'agentSession.handoffStatus', params: HandoffStatusParams, diff --git a/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts b/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts new file mode 100644 index 00000000000..aa1709fcecb --- /dev/null +++ b/src/main/runtime/rpc/mobile-clipboard-image-provenance.test.ts @@ -0,0 +1,53 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' +import { + hasMobileClipboardImagePath, + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES, + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS, + mobileClipboardImageProvenanceSizeForTest, + recordMobileClipboardImagePath, + resetMobileClipboardImageProvenanceForTest +} from './mobile-clipboard-image-provenance' + +describe('mobile clipboard image provenance', () => { + beforeEach(() => { + vi.useFakeTimers() + vi.setSystemTime(new Date('2026-01-01T00:00:00Z')) + resetMobileClipboardImageProvenanceForTest() + }) + + afterEach(() => { + resetMobileClipboardImageProvenanceForTest() + vi.useRealTimers() + }) + + it('expires records without consuming them on repeated checks', () => { + recordMobileClipboardImagePath('device-a', '/tmp/image.png') + + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(true) + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(true) + vi.advanceTimersByTime(MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS + 1) + expect(hasMobileClipboardImagePath('device-a', '/tmp/image.png')).toBe(false) + expect(mobileClipboardImageProvenanceSizeForTest()).toBe(0) + }) + + it('evicts the oldest record at the global bound and supports test cleanup', () => { + for (let index = 0; index <= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES; index++) { + recordMobileClipboardImagePath(`device-${index}`, `/tmp/image-${index}.png`) + vi.advanceTimersByTime(1) + } + + expect(mobileClipboardImageProvenanceSizeForTest()).toBe( + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES + ) + expect(hasMobileClipboardImagePath('device-0', '/tmp/image-0.png')).toBe(false) + expect( + hasMobileClipboardImagePath( + `device-${MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES}`, + `/tmp/image-${MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES}.png` + ) + ).toBe(true) + + resetMobileClipboardImageProvenanceForTest() + expect(mobileClipboardImageProvenanceSizeForTest()).toBe(0) + }) +}) diff --git a/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts b/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts new file mode 100644 index 00000000000..117455593c9 --- /dev/null +++ b/src/main/runtime/rpc/mobile-clipboard-image-provenance.ts @@ -0,0 +1,90 @@ +export const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS = 60 * 60 * 1000 +export const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES = 256 +const MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_PER_CLIENT = 64 + +const pathsByClient = new Map>() +let entryCount = 0 + +function deletePath(clientId: string, path: string): void { + const paths = pathsByClient.get(clientId) + if (!paths?.delete(path)) { + return + } + entryCount-- + if (paths.size === 0) { + pathsByClient.delete(clientId) + } +} + +function pruneExpired(now: number): void { + for (const [clientId, paths] of pathsByClient) { + for (const [path, expiresAt] of paths) { + if (expiresAt <= now) { + deletePath(clientId, path) + } + } + } +} + +function deleteOldestEntry(): void { + let oldest: { clientId: string; path: string; expiresAt: number } | null = null + for (const [clientId, paths] of pathsByClient) { + for (const [path, expiresAt] of paths) { + if (!oldest || expiresAt < oldest.expiresAt) { + oldest = { clientId, path, expiresAt } + } + } + } + if (oldest) { + deletePath(oldest.clientId, oldest.path) + } +} + +export function recordMobileClipboardImagePath(clientId: string | undefined, path: string): void { + const owner = clientId?.trim() + if (!owner) { + return + } + const now = Date.now() + pruneExpired(now) + let paths = pathsByClient.get(owner) + if (!paths) { + paths = new Map() + pathsByClient.set(owner, paths) + } + if (paths.delete(path)) { + entryCount-- + } + while (paths.size >= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_PER_CLIENT) { + const oldestPath = paths.keys().next().value + if (typeof oldestPath !== 'string') { + break + } + deletePath(owner, oldestPath) + } + while (entryCount >= MOBILE_CLIPBOARD_IMAGE_PROVENANCE_MAX_ENTRIES) { + deleteOldestEntry() + } + paths = pathsByClient.get(owner) ?? new Map() + pathsByClient.set(owner, paths) + paths.set(path, now + MOBILE_CLIPBOARD_IMAGE_PROVENANCE_TTL_MS) + entryCount++ +} + +export function hasMobileClipboardImagePath(clientId: string | undefined, path: string): boolean { + const owner = clientId?.trim() + if (!owner) { + return false + } + pruneExpired(Date.now()) + return pathsByClient.get(owner)?.has(path) ?? false +} + +export function resetMobileClipboardImageProvenanceForTest(): void { + pathsByClient.clear() + entryCount = 0 +} + +export function mobileClipboardImageProvenanceSizeForTest(): number { + return entryCount +} diff --git a/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts b/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts new file mode 100644 index 00000000000..3834a417ddb --- /dev/null +++ b/src/main/runtime/rpc/mobile-e2ee-v2-client-capabilities.ts @@ -0,0 +1,21 @@ +import type { RuntimeCapability } from '../../../shared/protocol-version' +import { parseRemoteRuntimeJsonText } from '../../../shared/remote-runtime-request-frames' +import { parseRuntimeClientCapabilities } from './runtime-client-capabilities' + +export function parseMobileE2EEV2ClientCapabilities( + plaintext: string +): readonly RuntimeCapability[] | null { + try { + const message = parseRemoteRuntimeJsonText(plaintext) as Record + if ( + Object.keys(message).sort().join(',') !== 'clientCapabilities,type,v' || + message.type !== 'e2ee_client_capabilities' || + message.v !== 1 + ) { + return null + } + return parseRuntimeClientCapabilities(message.clientCapabilities) + } catch { + return null + } +} diff --git a/src/main/runtime/rpc/mobile-socket-wiring.test.ts b/src/main/runtime/rpc/mobile-socket-wiring.test.ts index 505cf63feb6..14d5beb8cd0 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.test.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.test.ts @@ -348,4 +348,86 @@ describe('MobileSocketWiring', () => { expect(transport.setClientId).not.toHaveBeenCalled() expect(ws.close).toHaveBeenCalledWith(4001, 'Unauthorized') }) + + it('keeps post-auth v2 capability-shaped frames on the RPC path', () => { + const desktop = nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(1)) + const phone = nacl.box.keyPair.fromSecretKey(new Uint8Array(32).fill(2)) + const ws = new FakeSocket() + const transport = new FakeTransport() + const onText = vi.fn() + const metadata: MobileSocketTransportMetadata = { + transport: 'relay', + relayHostId: 'AbCdEf0123_-xyZ9', + relayDeviceId: 'device-1', + basisConnId: 'connection-1', + credentialKind: 'resume' + } + const wiring = new MobileSocketWiring({ + deviceRegistry: registryFor('device-1', 'valid-token'), + e2eeKeypair: { + publicKey: desktop.publicKey, + secretKey: desktop.secretKey, + publicKeyB64: Buffer.from(desktop.publicKey).toString('base64') + }, + onText, + onBinary: vi.fn(), + onClose: vi.fn() + }) + wiring.attachTransport(transport, () => metadata) + const hello: MobileE2EEV2Hello = { + type: 'e2ee_hello', + v: 2, + clientPublicKeyB64: Buffer.from(phone.publicKey).toString('base64'), + clientNonceB64: Buffer.from(new Uint8Array(32).fill(3)).toString('base64'), + capabilities: { framing: [2], payloadKinds: ['text', 'binary'] }, + context: { + protocol: 'orca-mobile-e2ee', + initiator: 'mobile', + responder: 'desktop', + transport: 'relay', + relayHostId: metadata.relayHostId + } + } + transport.receive(ws, JSON.stringify(hello)) + const ready = JSON.parse(ws.sent[0]!.toString()) as MobileE2EEV2Ready + const handshake = validateMobileE2EEV2Handshake(hello, ready)! + const schedule = deriveMobileE2EEV2KeySchedule({ + sharedSecret: deriveSharedKey(phone.secretKey, desktop.publicKey), + transcript: encodeMobileE2EEV2Transcript(handshake), + clientNonce: handshake.clientNonce, + desktopNonce: handshake.desktopNonce + }) + const send = (value: unknown, counter: bigint): void => { + const frame = sealMobileE2EEV2Frame({ + payload: new TextEncoder().encode(JSON.stringify(value)), + key: schedule.mobileToDesktopKey, + sessionId: schedule.sessionId, + direction: 'mobile-to-desktop', + payloadKind: 'text', + counter + }) + transport.receive(ws, Buffer.from(frame).toString('base64')) + } + send( + { + type: 'e2ee_auth', + v: 2, + transcriptHashB64: Buffer.from(schedule.transcriptHash).toString('base64'), + deviceToken: 'valid-token' + }, + 0n + ) + const capabilityFrame = { + type: 'e2ee_client_capabilities', + v: 1, + clientCapabilities: ['agent-session.structured.v1'] + } + send(capabilityFrame, 1n) + send({ id: 'rpc-1', method: 'agentSession.history', params: {} }, 2n) + + expect(onText).toHaveBeenCalledTimes(2) + expect(onText.mock.calls[0]?.[0].clientCapabilities).toEqual([]) + expect(JSON.parse(onText.mock.calls[0]?.[1] ?? '')).toEqual(capabilityFrame) + expect(onText.mock.calls[1]?.[0].clientCapabilities).toEqual([]) + }) }) diff --git a/src/main/runtime/rpc/mobile-socket-wiring.ts b/src/main/runtime/rpc/mobile-socket-wiring.ts index 43004be4582..384f403e9de 100644 --- a/src/main/runtime/rpc/mobile-socket-wiring.ts +++ b/src/main/runtime/rpc/mobile-socket-wiring.ts @@ -165,7 +165,9 @@ export class MobileSocketWiring { ws, connectionId, device, - clientCapabilities: channel.clientCapabilities, + get clientCapabilities() { + return channel.clientCapabilities + }, transport: metadata } this.authenticatedSockets.set(ws, socket) diff --git a/src/main/runtime/structured-agent-session-integration-replay.test.ts b/src/main/runtime/structured-agent-session-integration-replay.test.ts index 4edec30499b..e5baa032341 100644 --- a/src/main/runtime/structured-agent-session-integration-replay.test.ts +++ b/src/main/runtime/structured-agent-session-integration-replay.test.ts @@ -259,6 +259,7 @@ beforeEach(async () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => { bootEnvironmentReads += 1 return { @@ -320,6 +321,7 @@ describe('a structured codex session over agentSession.*', () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), openCodexConnection: codex.openConnection, readProcessStartTime: async () => 1_700_000_000_000 }) diff --git a/src/main/runtime/structured-agent-session-integration.test.ts b/src/main/runtime/structured-agent-session-integration.test.ts index aa5f819e1fb..2982a6530b2 100644 --- a/src/main/runtime/structured-agent-session-integration.test.ts +++ b/src/main/runtime/structured-agent-session-integration.test.ts @@ -307,6 +307,7 @@ beforeEach(async () => { claimKeyId: 'key-1', resolveWorkspacePath: async (workspaceId) => `/repos/${workspaceId}`, resolveCodexCommand: () => '/usr/local/bin/codex', + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => { bootEnvironmentReads += 1 return { diff --git a/src/main/runtime/structured-agent-session-owner-probe.ts b/src/main/runtime/structured-agent-session-owner-probe.ts new file mode 100644 index 00000000000..47f92a08ca6 --- /dev/null +++ b/src/main/runtime/structured-agent-session-owner-probe.ts @@ -0,0 +1,108 @@ +import type { AgentSessionOwnerProbe } from '../../shared/agent-session-lease-adjudication' +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { + probeAgentSessionProcessIdentities, + probeAgentSessionProcessIdentity, + probeAgentSessionReservation +} from './agent-session-process-identity-probe' +import { findAgentSessionSpawnTokenProcesses } from './agent-session-spawn-token-process-scan' +import { readEchoedAgentSessionSpawnToken } from './agent-session-spawn-token-readback' + +/** + * The lease's only source of truth about a previous owner. Everything it cannot + * answer PID-reuse-safely reports `indeterminate`. An exact owner stays fenced in `recovering`; + * an ownerless, unattributable reservation enters `manual-recovery`. + */ +export function createStructuredAgentSessionOwnerProbe( + hostId: string, + probe = probeAgentSessionProcessIdentity, + findSpawnTokenProcesses = findAgentSessionSpawnTokenProcesses +): (record: AgentSessionRecord) => Promise { + return async (record) => { + const owner = record.lease.ownerProcess + if (!owner) { + if (record.lease.processlessAt !== undefined && record.lease.processlessAt !== null) { + return { outcome: 'reservation-unused' } + } + const spawnToken = record.lease.reservedSpawnToken + if (spawnToken === null) { + if (record.lease.claimStatus === 'reserved') { + return { + outcome: 'indeterminate', + reason: 'reservation recorded no spawn token to scan for' + } + } + // The token is minted before the child and is the only thing a child could be carrying. + // No owner and no token means nothing on any host can be holding this lease — answering + // `indeterminate` here is what latches an already-free record into recovery forever. + return { outcome: 'reservation-unused' } + } + // Freeing a reservation needs positive proof that nothing spawned under its token. The scan + // answers null where the platform cannot read another process's environment. + return probeAgentSessionReservation({ + spawnToken, + findProcessesWithSpawnToken: (token) => findSpawnTokenProcesses(token), + hasProviderActivitySinceReservation: async () => + agentSessionReservationTouchedProvider(record) + }) + } + if (owner.hostId !== hostId) { + // Checking a remote host's pid against this machine's process table is + // exactly how a live owner gets declared dead. + return { + outcome: 'indeterminate', + reason: `owner runs on ${owner.hostId}, which this host cannot probe` + } + } + // The env read-back answers on hosts that expose it and null elsewhere, giving the + // probe a PID-reuse-safe element even when no start time was recorded. + return probe({ + identity: owner, + deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } + }) + } +} + +export function createStructuredAgentSessionOwnerProbes( + hostId: string, + probeMany: typeof probeAgentSessionProcessIdentities = probeAgentSessionProcessIdentities, + probeOne = createStructuredAgentSessionOwnerProbe(hostId) +): (records: readonly AgentSessionRecord[]) => Promise> { + return async (records) => { + const results = new Map() + const localOwners: { + record: AgentSessionRecord + owner: NonNullable + }[] = [] + for (const record of records) { + const owner = record.lease.ownerProcess + if (owner?.hostId === hostId) { + localOwners.push({ record, owner }) + } else { + results.set(record.sessionId, await probeOne(record)) + } + } + const probes = await probeMany({ + identities: localOwners.map(({ owner }) => owner), + deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } + }) + for (const [index, { record }] of localOwners.entries()) { + results.set( + record.sessionId, + probes[index] ?? { outcome: 'indeterminate', reason: 'owner probe returned no result' } + ) + } + return results + } +} + +/** + * The only provider-side trace a reservation can leave in its own record: a handle link minted at + * this fence. `proveAgentSessionOwner` refuses to append one before an identity is committed, so a + * link at the reservation's fence means a child got far enough to resume the provider thread. It + * cannot see activity the child produced without proving a handle, which is why it is paired with + * the token scan rather than trusted alone. + */ +function agentSessionReservationTouchedProvider(record: AgentSessionRecord): boolean { + return record.providerHandleChain.at(-1)?.mintedAtFence === record.lease.runtimeFence +} diff --git a/src/main/runtime/structured-agent-session-runtime-exit.test.ts b/src/main/runtime/structured-agent-session-runtime-exit.test.ts index a8419176357..5c6e43c2bc0 100644 --- a/src/main/runtime/structured-agent-session-runtime-exit.test.ts +++ b/src/main/runtime/structured-agent-session-runtime-exit.test.ts @@ -84,6 +84,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, @@ -180,6 +181,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, @@ -260,6 +262,7 @@ describe('structured session runtime provider-exit wiring', () => { hostId: 'local', claimKeyId: 'key-1', resolveWorkspacePath: async () => root!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveCodexCommand: () => 'codex', resolveEnvironment: async () => ({ PATH: process.env.PATH }), openCodexConnection: openConnection, diff --git a/src/main/runtime/structured-agent-session-runtime.test.ts b/src/main/runtime/structured-agent-session-runtime.test.ts index b03d17cdb3f..3b69a0a4be3 100644 --- a/src/main/runtime/structured-agent-session-runtime.test.ts +++ b/src/main/runtime/structured-agent-session-runtime.test.ts @@ -13,7 +13,9 @@ import type { } from '../../shared/agent-session-record' import { createStructuredAgentSessionOwnerProbe, - createStructuredAgentSessionOwnerProbes, + createStructuredAgentSessionOwnerProbes +} from './structured-agent-session-owner-probe' +import { ensureStructuredAgentSessionHost, hasPersistedStructuredAgentSessionStore, stopStructuredAgentSessionRuntime @@ -226,6 +228,7 @@ describe('structured agent-session runtime install', () => { hostId: HOST_ID, claimKeyId: 'key-1', resolveWorkspacePath: async () => stateDirectory!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => ({}), reapOrphanChildren, onError @@ -252,6 +255,7 @@ describe('structured agent-session runtime install', () => { hostId: HOST_ID, claimKeyId: 'key-1', resolveWorkspacePath: async () => stateDirectory!, + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }), resolveEnvironment: async () => ({}), reapOrphanChildren: async () => { throw failure @@ -301,7 +305,8 @@ describe('a teardown that fails is retried by the next stop', () => { claimKeyId: 'key-1', resolveWorkspacePath: async () => directory!, resolveEnvironment: async () => ({}), - reapOrphanChildren: async () => [] + reapOrphanChildren: async () => [], + resolveClaudeAuthPolicy: () => ({ stripAuthEnv: true }) }) const journalDir = join(directory, 'stubborn-journal') diff --git a/src/main/runtime/structured-agent-session-runtime.ts b/src/main/runtime/structured-agent-session-runtime.ts index 52a3bd81818..923d5627f72 100644 --- a/src/main/runtime/structured-agent-session-runtime.ts +++ b/src/main/runtime/structured-agent-session-runtime.ts @@ -9,29 +9,33 @@ import { existsSync } from 'node:fs' import { join } from 'node:path' -import type { AgentSessionOwnerProbe } from '../../shared/agent-session-lease-adjudication' import type { AgentSessionRecord } from '../../shared/agent-session-record' import { createCodexStructuredLaunchResolver } from '../codex/codex-structured-launch-resolution' import { CodexStructuredSessionAdapter, type CodexStructuredSessionAdapterDeps } from '../codex/codex-structured-session-adapter' +import type { ClaudeStructuredSessionAdapterDeps } from '../claude/claude-structured-session-adapter' import { StructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-host' +import { StructuredAgentSessionAdapterRouter } from '../native-chat/agent-session-wire/structured-agent-session-adapter-router' import type { StructuredAgentSessionHandoffTransport } from '../native-chat/agent-session-wire/structured-agent-session-handoff-types' import { setStructuredAgentSessionHost } from '../native-chat/agent-session-wire/structured-agent-session-registry' +import { + readClaudeManagedAccountGateSettings, + type ClaudeManagedAccountGateSettings +} from '../native-chat/claude-structured-managed-account-support' import { AgentSessionRecordStore } from './agent-session-record-store' import { agentSessionStorePath } from './agent-session-record-store-file' import { stopOrphanAgentSessionChildren } from './agent-session-orphan-child-reaper' import { - probeAgentSessionProcessIdentities, - probeAgentSessionProcessIdentity, - probeAgentSessionReservation -} from './agent-session-process-identity-probe' -import { findAgentSessionSpawnTokenProcesses } from './agent-session-spawn-token-process-scan' -import { readEchoedAgentSessionSpawnToken } from './agent-session-spawn-token-readback' + createStructuredAgentSessionOwnerProbe, + createStructuredAgentSessionOwnerProbes +} from './structured-agent-session-owner-probe' import { agentSessionPtyWriteGate } from './agent-session-pty-write-gate' import { resolveLoginShellEnvironment } from '../startup/login-shell-environment' import { recordAgentSessionProviderHandle } from './agent-session-provider-handle-transition' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { createStructuredClaudeRuntimeAdapter } from './structured-claude-runtime-adapter' /** Sibling of the journal tree rather than inside it: one file adjudicates every * session's lease, while a journal is per session. */ @@ -55,13 +59,20 @@ export type StructuredAgentSessionRuntimeDeps = { claimKeyId: string resolveWorkspacePath: (workspaceId: string) => Promise resolveCodexCommand?: (options?: { pathEnv?: string | null; homePath?: string }) => string + resolveClaudeCommand?: () => string /** Provider transports are overridden only to drive the runtime against scripted children. */ openCodexConnection?: CodexStructuredSessionAdapterDeps['openConnection'] + openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] /** Scripted app-servers carry fake pids the real start-time read cannot answer for. */ readProcessStartTime?: CodexStructuredSessionAdapterDeps['readProcessStartTime'] resolveLaunchArgs?: (provider: AgentSessionRecord['provider']) => Promise | string[] resolveLaunchEnv?: () => Promise resolveLaunchEnvOverlay?: () => Promise> | Record + resolveClaudeLaunchEnv?: () => Promise> | Record + /** Required, and asserted at install time — an absent policy must not degrade to a guess. */ + resolveClaudeAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + /** Raw settings getter; the reader that fails closed around it is built here, in checked code. */ + getClaudeManagedAccountGateSettings?: () => ClaudeManagedAccountGateSettings resolveEnvironment?: () => Promise resolveCodexOverrides?: () => NodeJS.ProcessEnv onError?: (input: { scope: string; error: unknown }) => void @@ -71,13 +82,17 @@ export type StructuredAgentSessionRuntimeDeps = { type InstalledRuntime = { host: StructuredAgentSessionHost - adapter: CodexStructuredSessionAdapter + adapter: { closeAll(): Promise } /** Resolves after every adapter-exit recovery callback has settled. */ waitForRecovery: () => Promise } let installing: Promise | null = null +/** Thrown when the host is installed without a Claude auth policy resolver. */ +export const CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED = + 'structured agent-session host requires a Claude auth policy resolver' + /** * Runtimes whose teardown did not finish. `installing` is cleared regardless so * nothing new attaches, but dropping the runtime as well would strand every @@ -147,8 +162,14 @@ async function tearDownRuntime(installed: InstalledRuntime): Promise { } async function install(deps: StructuredAgentSessionRuntimeDeps): Promise { + // Why thrown rather than defaulted: the caller is `@ts-nocheck`, so a dropped + // field arrives here as `undefined`. Refusing to install is loud; guessing a + // policy is the silent under-strip this assertion exists to prevent. + if (typeof deps.resolveClaudeAuthPolicy !== 'function') { + throw new Error(CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED) + } const bootEnvironment = (deps.resolveEnvironment ?? resolveLoginShellEnvironment)() - const resolveEnvironment = async (): Promise => ({ + const resolveCodexEnvironment = async (): Promise => ({ ...(await bootEnvironment), ...(await deps.resolveLaunchEnv?.()), ...(await deps.resolveLaunchEnvOverlay?.()), @@ -181,7 +202,7 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise + readClaudeManagedAccountGateSettings(deps.getClaudeManagedAccountGateSettings!) + } + : {}), + onUnexpectedExit: (event) => { + recoveryChain = recoveryChain.then(async () => { + try { + await host?.handleAdapterEvent(event) + } catch (error) { + deps.onError?.({ scope: `structured-agent-session-exit:${event.sessionId}`, error }) + } + }) + }, + ...(deps.openClaudeConnection ? { openClaudeConnection: deps.openClaudeConnection } : {}), + ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) + }) + const adapter = new StructuredAgentSessionAdapterRouter({ codex, claude }, async () => { + await Promise.all([codex.closeAll(), claude.closeAll()]) + }) host = new StructuredAgentSessionHost({ store, adapter, @@ -246,102 +295,3 @@ async function install(deps: StructuredAgentSessionRuntimeDeps): Promise Promise { - return async (record) => { - const owner = record.lease.ownerProcess - if (!owner) { - if (record.lease.processlessAt !== undefined && record.lease.processlessAt !== null) { - return { outcome: 'reservation-unused' } - } - const spawnToken = record.lease.reservedSpawnToken - if (spawnToken === null) { - if (record.lease.claimStatus === 'reserved') { - return { - outcome: 'indeterminate', - reason: 'reservation recorded no spawn token to scan for' - } - } - // The token is minted before the child and is the only thing a child could be carrying. - // No owner and no token means nothing on any host can be holding this lease — answering - // `indeterminate` here is what latches an already-free record into recovery forever. - return { outcome: 'reservation-unused' } - } - // Freeing a reservation needs positive proof that nothing spawned under its token. The scan - // answers null where the platform cannot read another process's environment. - return probeAgentSessionReservation({ - spawnToken, - findProcessesWithSpawnToken: (token) => findSpawnTokenProcesses(token), - hasProviderActivitySinceReservation: async () => - agentSessionReservationTouchedProvider(record) - }) - } - if (owner.hostId !== hostId) { - // Checking a remote host's pid against this machine's process table is - // exactly how a live owner gets declared dead. - return { - outcome: 'indeterminate', - reason: `owner runs on ${owner.hostId}, which this host cannot probe` - } - } - // The env read-back answers on hosts that expose it and null elsewhere, giving the - // probe a PID-reuse-safe element even when no start time was recorded. - return probe({ - identity: owner, - deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } - }) - } -} - -export function createStructuredAgentSessionOwnerProbes( - hostId: string, - probeMany: typeof probeAgentSessionProcessIdentities = probeAgentSessionProcessIdentities, - probeOne = createStructuredAgentSessionOwnerProbe(hostId) -): (records: readonly AgentSessionRecord[]) => Promise> { - return async (records) => { - const results = new Map() - const localOwners: { - record: AgentSessionRecord - owner: NonNullable - }[] = [] - for (const record of records) { - const owner = record.lease.ownerProcess - if (owner?.hostId === hostId) { - localOwners.push({ record, owner }) - } else { - results.set(record.sessionId, await probeOne(record)) - } - } - const probes = await probeMany({ - identities: localOwners.map(({ owner }) => owner), - deps: { readEchoedSpawnToken: readEchoedAgentSessionSpawnToken } - }) - for (const [index, { record }] of localOwners.entries()) { - results.set( - record.sessionId, - probes[index] ?? { outcome: 'indeterminate', reason: 'owner probe returned no result' } - ) - } - return results - } -} - -/** - * The only provider-side trace a reservation can leave in its own record: a handle link minted at - * this fence. `proveAgentSessionOwner` refuses to append one before an identity is committed, so a - * link at the reservation's fence means a child got far enough to resume the provider thread. It - * cannot see activity the child produced without proving a handle, which is why it is paired with - * the token scan rather than trusted alone. - */ -function agentSessionReservationTouchedProvider(record: AgentSessionRecord): boolean { - return record.providerHandleChain.at(-1)?.mintedAtFence === record.lease.runtimeFence -} diff --git a/src/main/runtime/structured-claude-auth-policy-wiring.test.ts b/src/main/runtime/structured-claude-auth-policy-wiring.test.ts new file mode 100644 index 00000000000..f018dfd30da --- /dev/null +++ b/src/main/runtime/structured-claude-auth-policy-wiring.test.ts @@ -0,0 +1,59 @@ +import { readFileSync } from 'node:fs' +import { join } from 'node:path' +import { mkdtemp, rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { afterEach, describe, expect, it } from 'vitest' +import { + CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED, + ensureStructuredAgentSessionHost, + stopStructuredAgentSessionRuntime +} from './structured-agent-session-runtime' + +/** + * The structured host's Claude auth policy has exactly one production wiring, and it + * lives in `orca-runtime-get-worktree-ps.ts` — a `@ts-nocheck` file, so neither the + * compiler nor a type test can see the field disappear. Deleting that wiring used to + * leave ~1000 tests green while every `ANTHROPIC_*` variable in the shell reached the + * child, because `stripAuthEnv` silently fell back to `false`. + * + * Two independent guards replace that silence, and this file pins both. + */ +describe('structured Claude auth policy wiring', () => { + // The behavioural version of this assertion — importing the runtime class and + // capturing the installed deps — costs 35s of module transform for the whole + // OrcaRuntime chain (measured), so the wiring itself is pinned by source and the + // policy's meaning by claude-structured-auth-policy.test.ts. + it('passes a settings-derived Claude auth policy to the host installer', () => { + const source = readFileSync(join(__dirname, 'orca-runtime-get-worktree-ps.ts'), 'utf8') + + expect(source).toContain('claudeStructuredAuthPolicyForSettings') + expect(source).toMatch( + /resolveClaudeAuthPolicy:\s*\(\)\s*=>\s*\n?\s*claudeStructuredAuthPolicyForSettings\(/ + ) + }) + + describe('installing without one', () => { + let stateDirectory: string | null = null + + afterEach(async () => { + await stopStructuredAgentSessionRuntime() + if (stateDirectory) { + await rm(stateDirectory, { recursive: true, force: true }) + stateDirectory = null + } + }) + + it('refuses loudly rather than defaulting to a guess', async () => { + stateDirectory = await mkdtemp(join(tmpdir(), 'orca-auth-policy-wiring-')) + + await expect( + ensureStructuredAgentSessionHost({ + stateDirectory, + hostId: 'local', + claimKeyId: 'key-1', + resolveWorkspacePath: async () => stateDirectory as string + } as unknown as Parameters[0]) + ).rejects.toThrow(CLAUDE_STRUCTURED_AUTH_POLICY_REQUIRED) + }) + }) +}) diff --git a/src/main/runtime/structured-claude-runtime-adapter.ts b/src/main/runtime/structured-claude-runtime-adapter.ts new file mode 100644 index 00000000000..398288562b9 --- /dev/null +++ b/src/main/runtime/structured-claude-runtime-adapter.ts @@ -0,0 +1,98 @@ +import type { AgentSessionRecord } from '../../shared/agent-session-record' +import { join } from 'node:path' +import { resolveClaudeCommand } from '../codex-cli/command' +import type { ClaudeStructuredAuthPolicy } from '../claude-accounts/claude-structured-auth-policy' +import { createClaudeStructuredLaunchResolver } from '../claude/claude-structured-launch-resolution' +import { + ClaudeStructuredSessionAdapter, + type ClaudeStructuredSessionAdapterDeps +} from '../claude/claude-structured-session-adapter' +import { claudeProviderHandleLink } from '../claude/claude-structured-owner-identity' +import type { StructuredAgentSessionLifecycleEvent } from '../native-chat/agent-session-wire/structured-agent-session-adapter' +import { + readClaudeTranscriptLeafUuid, + resolveSessionFilePath +} from '../native-chat/session-file-resolver' +import { recordAgentSessionProviderHandle } from './agent-session-provider-handle-transition' +import type { ClaudeManagedAccountGateSettings } from '../native-chat/claude-structured-managed-account-support' +import type { AgentSessionRecordStore } from './agent-session-record-store' + +export type StructuredClaudeRuntimeAdapterDeps = { + store: AgentSessionRecordStore + resolveWorkspacePath: (workspaceId: string) => Promise + resolveClaudeCommand?: () => string + resolveClaudeLaunchEnv?: () => Promise> | Record + /** Managed-account auth state for a Claude launch, mirroring the terminal preflight. + * Required: an absent policy is what silently under-strips. */ + resolveClaudeAuthPolicy: () => Promise | ClaudeStructuredAuthPolicy + readClaudeManagedAccountGate?: () => ClaudeManagedAccountGateSettings | null + openClaudeConnection?: ClaudeStructuredSessionAdapterDeps['openConnection'] + readProcessStartTime?: ClaudeStructuredSessionAdapterDeps['readProcessStartTime'] + onUnexpectedExit: (event: StructuredAgentSessionLifecycleEvent) => void +} + +export function createStructuredClaudeRuntimeAdapter( + deps: StructuredClaudeRuntimeAdapterDeps +): ClaudeStructuredSessionAdapter { + const { store } = deps + return new ClaudeStructuredSessionAdapter({ + resolveLaunch: createClaudeStructuredLaunchResolver({ + store, + resolveWorkspacePath: deps.resolveWorkspacePath, + resolveCommand: deps.resolveClaudeCommand ?? resolveClaudeCommand, + ...(deps.resolveClaudeLaunchEnv ? { resolveEnv: deps.resolveClaudeLaunchEnv } : {}), + resolveAuthPolicy: deps.resolveClaudeAuthPolicy, + ...(deps.readClaudeManagedAccountGate + ? { readManagedAccountGate: deps.readClaudeManagedAccountGate } + : {}) + }), + persistHandle: async ({ sessionId, providerSessionId, leafUuid, fence }) => { + const currentFence = store.getRecord(sessionId)?.lease.runtimeFence ?? fence + const observedAt = Date.now() + await store.transitionHandoff(sessionId, (record: AgentSessionRecord) => + recordAgentSessionProviderHandle({ + record, + fence: currentFence, + link: claudeProviderHandleLink({ + sessionId: providerSessionId, + leafUuid, + resumed: true, + fence: currentFence, + observedAt + }), + now: observedAt + }) + ) + }, + readTranscriptLeaf: async ({ providerSessionId, previousLeafUuid, claudeConfigDir }) => { + const transcriptPath = await resolveSessionFilePath('claude', providerSessionId, { + claudeProjectsDir: join(claudeConfigDir, 'projects') + }) + return transcriptPath + ? await readClaudeTranscriptLeafUuid(transcriptPath, providerSessionId, previousLeafUuid) + : null + }, + onEvent: (event) => { + if ( + event.type === 'ended' && + event.cause === 'unexpected-exit' && + event.fence !== undefined && + event.acquisitionGeneration + ) { + deps.onUnexpectedExit({ + type: 'ended', + sessionId: event.sessionId, + reason: event.reason, + cause: event.cause, + fence: event.fence, + acquisitionGeneration: event.acquisitionGeneration, + ...(event.settlementRetryRequired + ? { settlementRetryRequired: event.settlementRetryRequired } + : {}) + }) + } + }, + ...(deps.openClaudeConnection ? { openConnection: deps.openClaudeConnection } : {}), + ...(deps.readProcessStartTime ? { readProcessStartTime: deps.readProcessStartTime } : {}) + }) +} diff --git a/src/main/runtime/structured-tui-process-identity.test.ts b/src/main/runtime/structured-tui-process-identity.test.ts index 4324f8c5a26..fb5919824e5 100644 --- a/src/main/runtime/structured-tui-process-identity.test.ts +++ b/src/main/runtime/structured-tui-process-identity.test.ts @@ -238,6 +238,41 @@ describe('structured TUI process identity', () => { } }) + it('does not call a child absent after a single look that outlasted the budget', async () => { + // Measured on a 2,085-process host under load: one whole-machine `ps` took 6.2s while the + // shell-delivered child landed at ~3.5s. `ps` reads the table when it STARTS, so that one + // capture reported a t=0 machine and returned with the 5s budget already spent -- the loop + // answered "no exact child" without ever looking again. + let clockMs = 0 + let captures = 0 + await expect( + readStructuredTuiProcessIdentity({ + hostId: 'local', + rootPid: 100, + spawnToken: 'spawn-slow-ps', + agent: 'claude', + platform: 'darwin', + readPosixRows: async () => { + captures += 1 + const observedAtMs = clockMs + clockMs += 6_200 + return [ + { pid: 100, ppid: 1, stat: 'Ss', command: '/bin/zsh' }, + ...(observedAtMs >= 3_500 + ? [{ pid: 101, ppid: 100, stat: 'S+', command: 'claude --resume session-1' }] + : []) + ] + }, + readStartTime: async () => 1_700_000_000_000, + now: () => clockMs, + sleep: async (delayMs) => { + clockMs += delayMs + } + }) + ).resolves.toMatchObject({ pid: 101, spawnToken: 'spawn-slow-ps' }) + expect(captures).toBe(2) + }) + it('fails closed when the process snapshot omitted the PTY root', async () => { await expect( readStructuredTuiProcessIdentity({ diff --git a/src/main/runtime/structured-tui-process-identity.ts b/src/main/runtime/structured-tui-process-identity.ts index 0014351d781..f5ee019882b 100644 --- a/src/main/runtime/structured-tui-process-identity.ts +++ b/src/main/runtime/structured-tui-process-identity.ts @@ -20,6 +20,12 @@ const STRUCTURED_TUI_PROCESS_POLL_MS = 50 // window the added latency is bounded by one interval. const STRUCTURED_TUI_PROCESS_FAST_POLL_WINDOW_MS = 1_000 const STRUCTURED_TUI_PROCESS_MAX_POLL_MS = 500 +// Why a floor and not just the deadline: the first capture races the spawn it is looking for, +// so a null from it is absence of the child's arrival, not evidence the child is missing. The +// budget above assumes a look is nearly free, but one whole-machine `ps` measured 6.2s on a +// 2,085-process host under load -- long enough to spend the entire budget before the child +// (observed landing at ~3.5s) could exist, and answer "no exact child" after a single look. +const STRUCTURED_TUI_PROCESS_MIN_CAPTURES = 2 function descendants(rows: ProcessRow[], rootPid: number): (ProcessRow & { depth: number })[] { const children = new Map() @@ -175,6 +181,7 @@ export async function readStructuredTuiProcessIdentity(input: { const startedAtMs = now() const deadline = startedAtMs + (input.timeoutMs ?? STRUCTURED_TUI_PROCESS_WAIT_MS) let pollDelayMs = input.pollIntervalMs ?? STRUCTURED_TUI_PROCESS_POLL_MS + let captures = 0 while (true) { const rows: ProcessRow[] = @@ -186,6 +193,7 @@ export async function readStructuredTuiProcessIdentity(input: { foreground: false })) : posixRows(await (input.readPosixRows ?? getFreshProcessTableSnapshot)()) + captures += 1 let rootPresent = false for (const row of rows) { if (row.pid === input.rootPid) { @@ -218,11 +226,11 @@ export async function readStructuredTuiProcessIdentity(input: { } } const remainingMs = deadline - now() - if (remainingMs <= 0) { + if (remainingMs <= 0 && captures >= STRUCTURED_TUI_PROCESS_MIN_CAPTURES) { const label = input.agent === 'codex' ? 'Codex' : 'Claude' throw new Error(`The resumed terminal did not expose one exact ${label} child process.`) } - await sleep(Math.min(pollDelayMs, remainingMs)) + await sleep(Math.max(0, Math.min(pollDelayMs, remainingMs))) if (now() - startedAtMs >= STRUCTURED_TUI_PROCESS_FAST_POLL_WINDOW_MS) { // Never below the caller's interval, so an explicitly slow poll stays slow. pollDelayMs = Math.max( diff --git a/src/main/windows-descendant-exit-verification.test.ts b/src/main/windows-descendant-exit-verification.test.ts new file mode 100644 index 00000000000..392c44399e7 --- /dev/null +++ b/src/main/windows-descendant-exit-verification.test.ts @@ -0,0 +1,175 @@ +import { describe, expect, it, vi } from 'vitest' +import { + captureWindowsDescendantSnapshot, + terminateIdentifiedWindowsProcessTree, + verifyWindowsDescendantSnapshotExit, + type WindowsDescendantSnapshot +} from './windows-descendant-exit-verification' + +function snapshot( + descendants: { pid: number; creationTimeMs: number }[], + unidentifiedCount = 0 +): WindowsDescendantSnapshot { + return { + root: { pid: 100, creationTimeMs: 5 }, + descendants, + unidentifiedCount, + capturedAtMs: 1_700_000_000_000 + } +} + +describe('captureWindowsDescendantSnapshot', () => { + it('walks the whole subtree and keeps only rows a later read can re-identify', async () => { + const captured = await captureWindowsDescendantSnapshot(100, { + // 400 is a grandchild; 300 denied a creation-time query, so no later read + // could tell it from a recycled pid and signalling it would risk a stranger. + readTable: vi.fn(async () => [ + { pid: 100, ppid: 1, creationTimeMs: 5 }, + { pid: 200, ppid: 100, creationTimeMs: 7 }, + { pid: 300, ppid: 100 }, + { pid: 400, ppid: 200, creationTimeMs: 9 }, + { pid: 500, ppid: 1, creationTimeMs: 11 } + ]), + now: () => 42 + }) + + expect(captured).toEqual({ + root: { pid: 100, creationTimeMs: 5 }, + descendants: [ + { pid: 400, creationTimeMs: 9 }, + { pid: 200, creationTimeMs: 7 } + ], + // Seen but not re-identifiable: counted, so no later read can prove it gone. + unidentifiedCount: 1, + capturedAtMs: 42 + }) + }) + + it('reports an unreadable or rootless table as no snapshot rather than an empty one', async () => { + await expect( + captureWindowsDescendantSnapshot(100, { + readTable: vi.fn(async () => { + throw new Error('table unavailable') + }) + }) + ).resolves.toBeNull() + // A snapshot without the root is stale or filtered; only an observed root + // can authoritatively have no descendants. + await expect( + captureWindowsDescendantSnapshot(100, { + readTable: vi.fn(async () => [{ pid: 999, ppid: 1, creationTimeMs: 5 }]) + }) + ).resolves.toBeNull() + }) + + it('refuses an invalid root pid', async () => { + const readTable = vi.fn() + await expect(captureWindowsDescendantSnapshot(0, { readTable })).resolves.toBeNull() + expect(readTable).not.toHaveBeenCalled() + }) +}) + +describe('verifyWindowsDescendantSnapshotExit', () => { + it('proves an empty tree without reading the table', async () => { + const readTable = vi.fn() + await expect(verifyWindowsDescendantSnapshotExit(snapshot([]), { readTable })).resolves.toBe( + 'exited' + ) + expect(readTable).not.toHaveBeenCalled() + }) + + it('never proves a tree that held a descendant it could not identify', async () => { + // A descendant that denied the creation-time query was seen in the table; + // being unable to re-identify it is "could not look", never "it is gone". + const readTable = vi.fn() + await expect(verifyWindowsDescendantSnapshotExit(snapshot([], 1), { readTable })).resolves.toBe( + 'unverifiable' + ) + expect(readTable).not.toHaveBeenCalled() + + // The identified sibling leaving proves nothing about the unidentified one. + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }], 1), { + readTable: vi.fn(async () => []), + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(1) + }) + ).resolves.toBe('unverifiable') + }) + + it('reports exited once no identity-matched row remains', async () => { + const readTable = vi + .fn() + .mockResolvedValueOnce([{ pid: 200, ppid: 100, creationTimeMs: 7 }]) + // The pid came back on a different process; that is a recycle, not a survivor. + .mockResolvedValueOnce([{ pid: 200, ppid: 100, creationTimeMs: 99 }]) + + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable, + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(1) + }) + ).resolves.toBe('exited') + expect(readTable).toHaveBeenCalledTimes(2) + }) + + it('reports live for a descendant still matched at the deadline', async () => { + let clock = 0 + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable: vi.fn(async () => [{ pid: 200, ppid: 100, creationTimeMs: 7 }]), + wait: async () => { + clock += 100 + }, + now: () => clock, + verifyMs: 250 + }) + ).resolves.toBe('live') + }) + + it('reports unverifiable when the table cannot be read at the deadline', async () => { + await expect( + verifyWindowsDescendantSnapshotExit(snapshot([{ pid: 200, creationTimeMs: 7 }]), { + readTable: vi.fn(async () => { + throw new Error('table unavailable') + }), + wait: async () => {}, + now: vi.fn().mockReturnValueOnce(0).mockReturnValue(9_999) + }) + ).resolves.toBe('unverifiable') + }) +}) + +describe('terminateIdentifiedWindowsProcessTree', () => { + it('never taskkills a replacement that reused the captured root pid', async () => { + const terminateTree = vi.fn(async () => {}) + + await expect( + terminateIdentifiedWindowsProcessTree( + { pid: 100, creationTimeMs: 5 }, + { + readTable: vi.fn(async () => [{ pid: 100, ppid: 1, creationTimeMs: 99 }]), + terminateTree + } + ) + ).resolves.toBe(false) + expect(terminateTree).not.toHaveBeenCalled() + }) + + it('rechecks retained-child ownership after the identity read settles', async () => { + const terminateTree = vi.fn(async () => {}) + + await expect( + terminateIdentifiedWindowsProcessTree( + { pid: 100, creationTimeMs: 5 }, + { + readTable: vi.fn(async () => [{ pid: 100, ppid: 1, creationTimeMs: 5 }]), + ownsRoot: () => false, + terminateTree + } + ) + ).resolves.toBe(false) + expect(terminateTree).not.toHaveBeenCalled() + }) +}) diff --git a/src/main/windows-descendant-exit-verification.ts b/src/main/windows-descendant-exit-verification.ts new file mode 100644 index 00000000000..079833a2bd6 --- /dev/null +++ b/src/main/windows-descendant-exit-verification.ts @@ -0,0 +1,156 @@ +import type { DescendantTreeVerdict } from './pty-descendant-exit-verification' +import { windowsDescendantsFromRows } from './providers/windows-foreground-process-rows' +import { readWindowsProcessTableFresh } from './windows/windows-process-table' +import { terminateWindowsProcessTree } from './windows-process-tree-kill' + +export const WINDOWS_DESCENDANT_KILL_VERIFY_MS = 3_500 +const WINDOWS_DESCENDANT_POLL_MS = 100 + +/** + * A Windows descendant tree captured while its root was alive, with the + * PID-reuse guard the POSIX snapshot gets from ps lstart: a row only counts as + * the same process when its creation time still matches. Rows without a + * creation time are never signalled, because a bare pid cannot be re-identified, + * but they are counted: a descendant that was seen and denied identification + * is one no later read can prove gone. + */ +export type WindowsProcessIdentity = { pid: number; creationTimeMs: number } + +export type WindowsDescendantSnapshot = { + root: WindowsProcessIdentity + descendants: WindowsProcessIdentity[] + /** Descendants seen in the walk that denied the creation-time query. */ + unidentifiedCount: number + capturedAtMs: number + /** Per-PID boundaries retained when close refreshes merge snapshots. */ + capturedAtMsByPid?: Readonly> +} + +export type WindowsDescendantVerificationDeps = { + readTable?: () => Promise<{ pid: number; ppid: number; creationTimeMs?: number }[]> + now?: () => number + wait?: (ms: number) => Promise + verifyMs?: number +} + +/** Revalidate a Windows PID/creation-time identity immediately before a kill. */ +export async function verifyWindowsProcessIdentity( + target: WindowsProcessIdentity, + deps: Pick = {} +): Promise { + if (!Number.isInteger(target.pid) || target.pid <= 0 || !Number.isFinite(target.creationTimeMs)) { + return false + } + const table = await (deps.readTable ?? readWindowsProcessTableFresh)().catch(() => null) + const current = table?.filter((row) => row.pid === target.pid) ?? [] + return current.length === 1 && current[0]?.creationTimeMs === target.creationTimeMs +} + +function delay(ms: number): Promise { + return new Promise((resolve) => { + const timer = setTimeout(resolve, ms) + timer.unref?.() + }) +} + +/** + * Snapshot a Windows root's descendants while it is still alive. Resolves null + * (never rejects) when the table is unreadable or the root is absent — the same + * contract as the POSIX walk, because "cannot see" is never "nothing is there". + */ +export async function captureWindowsDescendantSnapshot( + rootPid: number, + deps: WindowsDescendantVerificationDeps = {} +): Promise { + if (!Number.isInteger(rootPid) || rootPid <= 0) { + return null + } + const capturedAtMs = (deps.now ?? Date.now)() + // One table read, not a walk plus an identity read: each is bounded in + // seconds, and this runs inside the close ladder's budget. + const table = await (deps.readTable ?? readWindowsProcessTableFresh)().catch(() => null) + const descendants = table && windowsDescendantsFromRows(table, rootPid) + const root = table?.find((row) => row.pid === rootPid) + if (!descendants || typeof root?.creationTimeMs !== 'number') { + return null + } + return { + root: { pid: root.pid, creationTimeMs: root.creationTimeMs }, + descendants: descendants.flatMap((row) => + // A descendant that denied a creation-time query cannot be told from a + // recycled pid later, so it is never signalled on a bare pid. + typeof row.creationTimeMs === 'number' + ? [{ pid: row.pid, creationTimeMs: row.creationTimeMs }] + : [] + ), + unidentifiedCount: descendants.filter((row) => typeof row.creationTimeMs !== 'number').length, + capturedAtMs + } +} + +export type IdentifiedWindowsTreeTerminationDeps = { + readTable?: WindowsDescendantVerificationDeps['readTable'] + terminateTree?: (target: WindowsProcessIdentity) => Promise + ownsRoot?: () => boolean +} + +/** Revalidate the captured root at the last async boundary before taskkill. */ +export async function terminateIdentifiedWindowsProcessTree( + target: WindowsProcessIdentity, + deps: IdentifiedWindowsTreeTerminationDeps = {} +): Promise { + if (!(await verifyWindowsProcessIdentity(target, { readTable: deps.readTable }))) { + return false + } + if (deps.ownsRoot?.() === false) { + return false + } + await ( + deps.terminateTree ?? + ((identified: WindowsProcessIdentity) => terminateWindowsProcessTree(identified.pid)) + )(target) + return true +} + +/** + * Whether a snapshotted Windows tree is gone, polled to a bounded deadline. + * + * Why a verification pass at all: `taskkill /T /F` resolves the same way on a + * timeout, an access denial and a recycled root as it does on a successful + * kill, so its completion is never evidence. Only a table read that no longer + * shows an identity-matched row is. + */ +export async function verifyWindowsDescendantSnapshotExit( + snapshot: WindowsDescendantSnapshot, + deps: WindowsDescendantVerificationDeps = {} +): Promise { + // The most a read can prove: a descendant that denied identification was seen + // and can never be matched gone, so "could not look" caps the verdict. + const proven: DescendantTreeVerdict = snapshot.unidentifiedCount > 0 ? 'unverifiable' : 'exited' + if (snapshot.descendants.length === 0) { + return proven + } + const now = deps.now ?? Date.now + const readTable = deps.readTable ?? readWindowsProcessTableFresh + const deadline = now() + (deps.verifyMs ?? WINDOWS_DESCENDANT_KILL_VERIFY_MS) + let verdict: DescendantTreeVerdict = 'unverifiable' + do { + const table = await readTable().catch(() => null) + if (!table) { + verdict = 'unverifiable' + } else { + const live = new Map(table.map((row) => [row.pid, row.creationTimeMs])) + verdict = snapshot.descendants.some((row) => live.get(row.pid) === row.creationTimeMs) + ? 'live' + : proven + if (verdict === proven) { + return verdict + } + } + if (now() >= deadline) { + return verdict + } + await (deps.wait ?? delay)(WINDOWS_DESCENDANT_POLL_MS) + } while (now() < deadline) + return verdict +} diff --git a/src/relay/pty-handler-ownership-attestation.test.ts b/src/relay/pty-handler-ownership-attestation.test.ts index 94f462c46d2..ee1c144df09 100644 --- a/src/relay/pty-handler-ownership-attestation.test.ts +++ b/src/relay/pty-handler-ownership-attestation.test.ts @@ -33,7 +33,8 @@ import { endPtyHandlerTest, type MockDispatcher } from './pty-handler-test-harness' -import { PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS } from '../shared/process-table-snapshot-reader' +import * as processTableSnapshotReader from '../shared/process-table-snapshot-reader' +import { RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS } from '../shared/ssh-relay-pty-ownership-proof' const PANE_KEY = 'tab-agent:22222222-2222-4222-8222-222222222222' @@ -139,16 +140,41 @@ describe('PtyHandler publishes host-attested PTY ownership', () => { expect(entry?.ownerClientInstanceId).toBe('client-A') }) - it('dates the foreground observation instead of stamping it fresh', async () => { - // `capturedAgeMs` used to be a hardcoded 0 with no reader anywhere, so the one field that - // exists to bound staleness asserted the evidence was never stale. It now carries the - // actual age of the TTL-shared capture the record was derived from. - const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + it('publishes the age the capture reported, rather than restamping it fresh', async () => { + // This assertion used to read `capturedAgeMs <= PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS`, which + // could not fail: `beginPtyHandlerTest` installs fake timers, so `Date.now()` is frozen, the + // real reader reports exactly +0, and `0 <= 500` held identically for a hardcoded zero, for + // completion-stamping and for start-stamping. The one test guarding this field was blind to + // every change to it, while the real reader on a 2,002-process host returns thousands of ms. + // + // So drive a real age in from the reader. That the reader MEASURES the age correctly is + // pinned separately, against a controllable clock, by process-table-snapshot.test.ts; what + // belongs here is that the handler publishes what it was given instead of restamping. + const capturedAgeMs = 6_140 + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockResolvedValue({ rows: [], capturedAgeMs }) + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) const entry = (await listProcesses()).find((process) => process.id === id) - expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBeLessThanOrEqual( - PROCESS_TABLE_SNAPSHOT_MAX_STALENESS_MS + expect(snapshot).toHaveBeenCalled() + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBe(capturedAgeMs) + }) + + it('publishes an age a destructive consumer will refuse, rather than one it will trust', async () => { + // The point of the field, stated as the consumer sees it: an observation this old cannot + // authorize a stop, and the whole bug was that it used to arrive claiming it could. + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockResolvedValue({ rows: [], capturedAgeMs: 6_140 }) + + const { id } = await spawnFrom(7, { env: { ORCA_PANE_KEY: PANE_KEY } }) + const entry = (await listProcesses()).find((process) => process.id === id) + + expect(snapshot).toHaveBeenCalled() + expect(entry?.foregroundProcessEvidence?.capturedAgeMs).toBeGreaterThan( + RELAY_PTY_SWEEP_MAX_EVIDENCE_AGE_MS ) }) }) diff --git a/src/relay/pty-handler-spawn-admission.test.ts b/src/relay/pty-handler-spawn-admission.test.ts index 045ee2e412d..07fe8e9d88f 100644 --- a/src/relay/pty-handler-spawn-admission.test.ts +++ b/src/relay/pty-handler-spawn-admission.test.ts @@ -126,6 +126,46 @@ describe('PtyHandler', () => { expect(hasChildren).toHaveBeenLastCalledWith(mockPtyInstance.pid, { fresh: true }) }) + it('does not re-enter the shared capture after the evidence read gave up on it', async () => { + // The budget is worthless if the compatibility fields answer by joining the very capture the + // evidence read just abandoned: `inspectPtyChildProcesses` and `getForegroundProcessName` + // read the same TTL-shared table with no budget of their own, so on a slow host this call + // would still block for the whole capture -- once, then once per managed PTY in the listing. + const snapshot = vi + .spyOn(processTableSnapshotReader, 'getStrictProcessTableSnapshotWithAge') + .mockRejectedValue(new Error('process table unreadable: capture_over_budget')) + const hasChildren = vi.spyOn(ptyChildProcessInspection, 'inspectPtyChildProcesses') + const foregroundName = vi.spyOn(ptyShellUtils, 'getForegroundProcessName') + + const { id } = (await spawnPty({ cols: 80, rows: 24 })) as { id: string } + hasChildren.mockClear() + foregroundName.mockClear() + + const inspection = (await dispatcher.callRequest('pty.inspectProcess', { id })) as { + hasChildProcesses: boolean + childProcessEvidence?: string + foregroundProcessEvidence?: { verdict: string; reason?: string } + } + + expect(snapshot).toHaveBeenCalled() + expect(hasChildren).not.toHaveBeenCalled() + // The verdict the gates already handle, reached promptly instead of late. + expect(inspection.foregroundProcessEvidence?.verdict).toBe('unverifiable') + expect(inspection.foregroundProcessEvidence?.reason).toBe('process_table_unreadable') + // The honest verdict rather than a fabricated negative, reached without the wait. The + // compatibility boolean still spells `unverifiable` as `false` for older clients. + expect(inspection.childProcessEvidence).toBe('unverifiable') + expect(inspection.hasChildProcesses).toBe(false) + + const listing = (await dispatcher.callRequest('pty.listProcesses', {})) as { + id: string + title: string + }[] + + expect(foregroundName).not.toHaveBeenCalled() + expect(listing.find((entry) => entry.id === id)?.title).toBeTruthy() + }) + it('rejects strict process inspection for a missing relay PTY', async () => { await expect(dispatcher.callRequest('pty.inspectProcess', { id: 'missing' })).rejects.toThrow( 'terminal_gone' diff --git a/src/relay/pty-handler.ts b/src/relay/pty-handler.ts index 190bcabb3f9..4b7c6dac2d6 100644 --- a/src/relay/pty-handler.ts +++ b/src/relay/pty-handler.ts @@ -2663,6 +2663,9 @@ export class PtyHandler { } } let rows: readonly ProcessTableRow[] | null = null + // Set only when the budgeted evidence read gave up, so the compatibility fields below do not + // turn around and ask the same unreadable table again with no budget at all. + let tableUnavailable = false let evidence: RemoteForegroundEvidence | undefined if (process.platform === 'win32') { // Why SSH-to-Windows is always unverifiable: POSIX has a real foreground primitive @@ -2701,6 +2704,7 @@ export class PtyHandler { rows ) } catch { + tableUnavailable = true evidence = { authorityGeneration: this.ptyIdMintEpoch, observationEpoch: ++this.foregroundEvidenceEpoch, @@ -2727,13 +2731,19 @@ export class PtyHandler { // 1.36s CIM scan, and polling that would reinstate exactly the fork storm the shared table // exists to prevent (#15209, #15036). Close and cleanup decisions ask for the scan by name; // a poll gets the honest `unverifiable` instead of a fabricated negative. + // Why `tableUnavailable` first: it means the budgeted evidence read already gave up. Without + // this arm `inspectPtyChildProcesses` re-enters `getProcessTableSnapshot()` and joins the very + // capture this call just abandoned, blocking for all of it and spending the whole latency the + // budget exists to avoid. The destructive `pty.hasChildProcesses` RPC keeps its fresh probe. const childProcessEvidence: PtyChildProcessVerdict = rows ? rows.some((row) => row.ppid === managed.pty.pid) ? 'children' : 'no-children' - : process.platform === 'win32' && params.scanChildProcesses !== true + : tableUnavailable ? 'unverifiable' - : await inspectPtyChildProcesses(managed.pty.pid) + : process.platform === 'win32' && params.scanChildProcesses !== true + ? 'unverifiable' + : await inspectPtyChildProcesses(managed.pty.pid) return { foregroundProcess, // `unverifiable` keeps spelling itself `false` on the compatibility field, which is what @@ -2758,6 +2768,10 @@ export class PtyHandler { // process-table work on the host. const includeForegroundProcessEvidence = params.includeForegroundProcessEvidence !== false let evidenceRows: readonly ProcessTableRow[] | null = null + // Same reason as `inspectProcess`: once the budgeted read has given up, the per-PTY title + // fallback below must not re-enter the same capture without a budget -- and here it would do + // so once per managed PTY. + let evidenceTableUnavailable = false let evidenceResults: BatchedForegroundProcessResult[] = [] const evidenceEpoch = ++this.foregroundEvidenceEpoch // Worst-case capture time for the snapshot below, not the instant its await settled: the @@ -2783,6 +2797,7 @@ export class PtyHandler { } catch { // An unreadable capture is represented as unverifiable evidence below; // existing inventory fields remain available for old clients. + evidenceTableUnavailable = true } } for (const [entryIndex, [id, managed]] of managedEntries.entries()) { @@ -2797,7 +2812,7 @@ export class PtyHandler { const title = (evidenceRows ? (evidenceResults[entryIndex]?.processName ?? managed.pty.process ?? null) - : includeForegroundProcessEvidence + : includeForegroundProcessEvidence && !evidenceTableUnavailable ? await getForegroundProcessName(managed.pty.pid, managed.pty.process || null) : managed.pty.process || null) || 'shell' const foregroundProcessEvidence = diff --git a/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx b/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx index b6748fe7e82..9b1a2967682 100644 --- a/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatQuestionCard.test.tsx @@ -28,7 +28,7 @@ afterEach(() => { function render( prompt: AskPrompt, onAnswer: (s: AskAnswerSelection[]) => void, - allowOther = true + allowOther: boolean | readonly boolean[] = true ): void { act(() => { root.render( @@ -155,4 +155,71 @@ describe('NativeChatQuestionCard', () => { expect(container.querySelector('input')).toBeNull() expect(container.textContent).not.toContain('Type your answer') }) + + it('applies free-text capability per question in a grouped prompt', () => { + render( + { + questions: [ + { + header: 'Listed', + question: 'Pick a listed value', + multiSelect: false, + options: [{ label: 'One' }] + }, + { + header: 'Custom', + question: 'Provide a custom value', + multiSelect: false, + options: [] + } + ] + }, + vi.fn(), + [false, true] + ) + + expect(container.querySelector('input')).toBeNull() + clickAction('Skip') + expect(container.querySelector('input')).not.toBeNull() + }) + + it('submits grouped multi-select and free-text answers together', () => { + const onAnswer = vi.fn() + render( + { + questions: [ + { + header: 'Targets', + question: 'Which targets?', + multiSelect: true, + options: [{ label: 'Web' }, { label: 'Mobile' }] + }, + { + header: 'Notes', + question: 'Anything else?', + multiSelect: false, + options: [] + } + ] + }, + onAnswer, + [false, true] + ) + + clickOption('Web') + clickOption('Mobile') + clickAction('Next') + const input = container.querySelector('input')! + act(() => { + const setter = Object.getOwnPropertyDescriptor(HTMLInputElement.prototype, 'value')!.set! + setter.call(input, 'SSH host') + input.dispatchEvent(new Event('input', { bubbles: true })) + }) + clickAction('Submit') + + expect(onAnswer).toHaveBeenCalledWith([ + { indices: [0, 1], other: '' }, + { indices: [], other: 'SSH host' } + ]) + }) }) diff --git a/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx b/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx index 4bc881ee3e1..1b1ce5a3547 100644 --- a/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx +++ b/src/renderer/src/components/native-chat/NativeChatQuestionCard.tsx @@ -10,7 +10,7 @@ export type NativeChatQuestionCardProps = { isSubmitting?: boolean /** Deliver the chosen answer (per-question option indices + free text). */ onAnswer: (selections: AskAnswerSelection[]) => void - allowOther?: boolean + allowOther?: boolean | readonly boolean[] /** Dismiss the prompt (sends Escape to the agent). */ onCancel: () => void /** Exposes the free-text row so pane-level Paste can target it while the @@ -42,6 +42,7 @@ export function NativeChatQuestionCard({ const total = prompt.questions.length const isLast = index === total - 1 const q = prompt.questions[index]! + const questionAllowsOther = Array.isArray(allowOther) ? (allowOther[index] ?? false) : allowOther const setOther = (qi: number, value: string): void => { setOtherText((prev) => { @@ -186,7 +187,7 @@ export function NativeChatQuestionCard({ /> ))}
    - {allowOther ? ( + {questionAllowsOther ? ( <> diff --git a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx index d53e8f536d2..031ce4bcd15 100644 --- a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.test.tsx @@ -165,6 +165,7 @@ function model(overrides: Partial = {}): SessionOptionD ] }, valueSource: 'applied', + transport: 'catalog', settable: true, ...overrides } @@ -183,6 +184,7 @@ const effort: SessionOptionDescriptor = { ] }, valueSource: 'applied', + transport: 'catalog', settable: true } @@ -192,6 +194,7 @@ const fast: SessionOptionDescriptor = { category: 'mode', kind: { type: 'boolean', currentValue: true }, valueSource: 'applied', + transport: 'catalog', settable: true } @@ -342,17 +345,48 @@ describe('NativeChatSessionOptionPickers', () => { expect(screen.queryByRole('button', { name: /^Effort/ })).toBeNull() }) - it('shows the unconfirmed hint for dispatched values', () => { + // The terminal transport typed the value at the agent and has not read it back, + // so the pill says so; the structured transport's own per-turn report is the + // confirmation, which makes the same hedge transient noise there. + it('hedges a dispatched value the terminal transport produced', () => { render( ) - expect(screen.getByText('Sent to the agent — not confirmed')).not.toBeNull() + expect(screen.getByText('Model')).not.toBeNull() + expect(screen.getAllByText('Sent to the agent — not confirmed').length).toBeGreaterThan(0) }) + it('does not hedge a dispatched value the structured transport produced', () => { + render( + + ) + expect(screen.getByText('Model')).not.toBeNull() + expect(screen.queryByText(/not confirmed/)).toBeNull() + }) + + it.each(['catalog', 'agent-session'] as const)( + 'does not hedge a reported value on the %s transport', + (transport) => { + render( + + ) + expect(screen.getByText('Model')).not.toBeNull() + expect(screen.queryByText(/not confirmed/)).toBeNull() + } + ) + it('renders agent-picker routes as one action instead of radio choices', async () => { const invokeAction = vi.fn().mockResolvedValue({ snapshot: [] }) const liveSurface = { ...surface, invokeAction } @@ -449,6 +483,7 @@ describe('NativeChatSessionOptionPickers', () => { category: 'mode', kind: { type: 'boolean' }, valueSource: 'unknown', + transport: 'catalog', settable: true } ]} @@ -463,26 +498,7 @@ describe('NativeChatSessionOptionPickers', () => { await waitFor(() => expect(setOption).toHaveBeenCalledWith('thinking', false)) }) - it('does not show unconfirmed for applied flip-only booleans', () => { - render( - - ) - expect(screen.queryByText('Sent to the agent — not confirmed')).toBeNull() - }) - - it('shows unconfirmed for confirmable dispatched booleans', () => { + it('tooltips a dispatched option pill with the category alone', () => { render( { category: 'mode', kind: { type: 'boolean', currentValue: true }, valueSource: 'dispatched', + transport: 'catalog', settable: true } ]} isWorking={false} /> ) - expect(screen.getByText('Sent to the agent — not confirmed')).not.toBeNull() + expect(screen.getAllByText('Thinking').length).toBeGreaterThan(0) + expect(screen.getAllByText('Sent to the agent — not confirmed').length).toBeGreaterThan(0) }) }) diff --git a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx index 87a860662d2..31ff2cbdc4e 100644 --- a/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx +++ b/src/renderer/src/components/native-chat/NativeChatSessionOptionPickers.tsx @@ -15,10 +15,11 @@ import { import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip' import { translate } from '@/i18n/i18n' import { sortNativeChatSessionOptions } from '../../../../shared/native-chat-session-option-snapshot' -import type { - SessionOptionDescriptor, - SessionOptionsSurface, - SessionOptionValue +import { + sessionOptionDispatchUnconfirmed, + type SessionOptionDescriptor, + type SessionOptionsSurface, + type SessionOptionValue } from '../../../../shared/native-chat-session-options' import { nativeChatModelPillLabel, @@ -250,7 +251,7 @@ function NativeChatSessionOptionPickersInner({ tooltipLabel={optionsTooltip} disabled={isWorking || pendingId !== null} disabledReason={optionsReason} - dispatched={options.some((descriptor) => descriptor.valueSource === 'dispatched')} + dispatched={options.some(sessionOptionDispatchUnconfirmed)} /> {options.map((descriptor, index) => { @@ -283,7 +284,7 @@ function NativeChatSessionOptionPickersInner({ tooltipLabel={modelTooltip} disabled={isWorking || pendingId !== null} disabledReason={modelReason} - dispatched={model.valueSource === 'dispatched'} + dispatched={sessionOptionDispatchUnconfirmed(model)} /> {modelReason && !model.settable ? ( diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx index bfb3dd1ef52..d3c13a48ef1 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.test.tsx @@ -3,6 +3,9 @@ import { act, cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' import React, { forwardRef, useImperativeHandle } from 'react' import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentJournalRenderItem } from '../../../../shared/agent-session-journal-types' +import { decodeAgentSessionQuestionAnswers } from '../../../../shared/agent-session-question-answer' +import type { NativeChatQuestionCardProps } from './NativeChatQuestionCard' const mocks = vi.hoisted(() => ({ call: vi.fn(), @@ -13,6 +16,9 @@ const mocks = vi.hoisted(() => ({ onLinkClick?: (...args: unknown[]) => void }, composerProps: null as null | { structuredTransport?: Record }, + questionCardProps: null as NativeChatQuestionCardProps | null, + promptItems: [] as AgentJournalRenderItem[], + respond: vi.fn(), handlePasteEvent: vi.fn(), pasteFromClipboard: vi.fn(), submissions: [] as unknown[] @@ -53,7 +59,7 @@ vi.mock('./use-structured-agent-session', async () => { hasOlder: false, loadingOlder: false, loadOlder: vi.fn(), - prompts: [], + prompts: mocks.promptItems, outbox: outbox.outbox, blockedClientMessageId: outbox.blockedClientMessageId, send: outbox.send, @@ -61,7 +67,7 @@ vi.mock('./use-structured-agent-session', async () => { isWorking: false, turnId: null, cancel: vi.fn(), - respond: vi.fn(), + respond: mocks.respond, optionSnapshot: [ { id: 'model', @@ -125,7 +131,12 @@ vi.mock('./NativeChatComposer', () => ({ })) vi.mock('./NativeChatEmptyState', () => ({ NativeChatEmptyState: () => null })) vi.mock('./NativeChatApprovalCard', () => ({ NativeChatApprovalCard: () => null })) -vi.mock('./NativeChatQuestionCard', () => ({ NativeChatQuestionCard: () => null })) +vi.mock('./NativeChatQuestionCard', () => ({ + NativeChatQuestionCard: (props: NativeChatQuestionCardProps) => { + mocks.questionCardProps = props + return null + } +})) import { NativeChatStructuredSession } from './NativeChatStructuredSession' @@ -136,6 +147,9 @@ describe('NativeChatStructuredSession', () => { mocks.mode = 'static' mocks.messageListProps = null mocks.composerProps = null + mocks.questionCardProps = null + mocks.promptItems = [] + mocks.respond.mockReset() mocks.handlePasteEvent.mockReset() mocks.pasteFromClipboard.mockReset() mocks.submissions = [] @@ -546,4 +560,131 @@ describe('NativeChatStructuredSession', () => { vi.useRealTimers() } }, 30000) + + it('passes Claude grouped questions and one shared answer through the card', () => { + mocks.promptItems = [ + { + itemId: 'question-item', + revision: 1, + sequence: 1, + observedAt: 1, + body: { + kind: 'question', + question: '2 grouped questions from Claude', + options: [], + questions: [ + { + id: 'q1', + header: 'Targets', + question: 'Which targets?', + multiSelect: true, + options: [ + { id: 'target-web', label: 'Web' }, + { id: 'target-mobile', label: 'Mobile' } + ], + freeTextQuestionId: 'q1' + }, + { + id: 'q2', + header: 'Host', + question: 'Where should it run?', + multiSelect: false, + options: [], + freeTextQuestionId: 'q2' + } + ], + resolution: { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null + } + } + } + ] + + render( + + ) + + const card = mocks.questionCardProps + if (!card) { + throw new Error('question card was not rendered') + } + expect(card.prompt.questions).toHaveLength(2) + expect(card.prompt.questions[0]).toMatchObject({ + question: 'Which targets?', + multiSelect: true, + options: [{ label: 'Web' }, { label: 'Mobile' }] + }) + expect(card.allowOther).toEqual([true, true]) + + card.onAnswer([ + { indices: [0, 1], other: '' }, + { indices: [], other: 'SSH host' } + ]) + const encoded = mocks.respond.mock.calls[0]?.[1] + expect(decodeAgentSessionQuestionAnswers(encoded)).toEqual([ + { questionId: 'q1', optionIds: ['target-web', 'target-mobile'] }, + { questionId: 'q2', optionIds: [], other: 'SSH host' } + ]) + }) + + it('keeps legacy single-question option ids and free text behavior', () => { + mocks.promptItems = [ + { + itemId: 'legacy-question-item', + revision: 1, + sequence: 1, + observedAt: 1, + body: { + kind: 'question', + question: 'Pick a library', + options: [ + { id: 'q1:choice-1', label: 'React' }, + { id: 'q1:choice-2', label: 'Vue' } + ], + freeTextQuestionId: 'q1', + resolution: { + state: 'pending', + selectedOptionId: null, + resolvedBy: null, + resolvedAt: null + } + } + } + ] + + render( + + ) + + const card = mocks.questionCardProps + if (!card) { + throw new Error('question card was not rendered') + } + expect(card.prompt.questions).toEqual([ + { + question: 'Pick a library', + multiSelect: false, + options: [{ label: 'React' }, { label: 'Vue' }] + } + ]) + card.onAnswer([{ indices: [1], other: '' }]) + expect(mocks.respond).toHaveBeenCalledWith(mocks.promptItems[0], 'q1:choice-2') + }) }) diff --git a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx index f7d2fd62663..d6464c27760 100644 --- a/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx +++ b/src/renderer/src/components/native-chat/NativeChatStructuredSession.tsx @@ -4,6 +4,7 @@ import type { AgentStatusOrchestrationContext, AgentType } from '../../../../shared/agent-status-types' +import { encodeAgentSessionQuestionAnswers } from '../../../../shared/agent-session-question-answer' import { dispatchStructuredAgentSessionComposerCommand } from '../../../../shared/structured-agent-session-composer' import { structuredAgentSessionPaneKey } from '../../../../shared/structured-agent-session-projection' import type { NativeChatLiveSession } from './use-native-chat-live-session' @@ -85,6 +86,21 @@ export function NativeChatStructuredSession(props: { const fileLinkClick = useNativeChatFileLinkClick(props.allowFileUriLinks ? fileLinkContext : null) const prompt = controller.prompts[0] ?? null const questionBody = prompt?.body.kind === 'question' ? prompt.body : null + const questions = + questionBody?.questions ?? + (questionBody + ? [ + { + id: questionBody.freeTextQuestionId ?? 'q1', + question: questionBody.question, + options: questionBody.options, + multiSelect: false, + ...(questionBody.freeTextQuestionId + ? { freeTextQuestionId: questionBody.freeTextQuestionId } + : {}) + } + ] + : []) const retryableOutboxEntry = controller.outbox.find((entry) => entry.state === 'unconfirmed') ?? controller.outbox.find( @@ -166,17 +182,39 @@ export function NativeChatStructuredSession(props: { ) : null} {prompt && questionBody ? ( ({ label: option.label })) - } - ] + questions: questions.map((question) => ({ + question: question.question, + ...(question.header ? { header: question.header } : {}), + multiSelect: question.multiSelect, + options: question.options.map((option) => ({ + label: option.label, + ...(option.description ? { description: option.description } : {}) + })) + })) }} - allowOther={Boolean(questionBody.freeTextQuestionId)} + allowOther={questions.map((question) => Boolean(question.freeTextQuestionId))} onAnswer={(answers) => { + if (questionBody.questions) { + const grouped = questions.map((question, questionIndex) => { + const answer = answers[questionIndex] + const other = answer?.other?.trim() + const optionIds = (answer?.indices ?? []).flatMap((optionIndex) => { + const optionId = question.options[optionIndex]?.id + return optionId ? [optionId] : [] + }) + return { + questionId: question.id, + optionIds: question.multiSelect || !other ? optionIds : [], + ...(other ? { other } : {}) + } + }) + if (grouped.every((answer) => answer.optionIds.length > 0 || answer.other)) { + void controller.respond(prompt, encodeAgentSessionQuestionAnswers(grouped)) + } + return + } const index = answers[0]?.indices[0] const other = answers[0]?.other?.trim() const optionId = diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx new file mode 100644 index 00000000000..4a185a3d630 --- /dev/null +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.test.tsx @@ -0,0 +1,58 @@ +// @vitest-environment happy-dom + +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, describe, expect, it, vi } from 'vitest' +import type { AgentSessionHandoffStatus } from '../../../../shared/agent-session-wire' +import { StructuredAgentSessionHandoffChrome } from './StructuredAgentSessionHandoffChrome' + +const IDLE_NATIVE: AgentSessionHandoffStatus = { + owner: 'native', + direction: null, + phase: 'idle', + stage: null, + operationId: null +} + +afterEach(cleanup) + +describe('StructuredAgentSessionHandoffChrome', () => { + it('uses queued-safe admission when the native view still appears idle', () => { + const onRequest = vi.fn() + render( + + ) + + fireEvent.click(screen.getByRole('button', { name: 'Open agent TUI' })) + + expect(onRequest).toHaveBeenCalledWith('to-tui', 'after-turn') + }) + + it('offers one Retry action for a recoverable dead TUI owner', () => { + const onRequest = vi.fn() + render( + + ) + + expect(screen.queryByRole('button', { name: 'Return to chat' })).toBeNull() + fireEvent.click(screen.getByRole('button', { name: 'Retry' })) + expect(onRequest).toHaveBeenCalledWith('to-native', 'now', 'retry') + }) +}) diff --git a/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx new file mode 100644 index 00000000000..840039564d8 --- /dev/null +++ b/src/renderer/src/components/native-chat/StructuredAgentSessionHandoffChrome.tsx @@ -0,0 +1,225 @@ +import type { + AgentSessionHandoffDirection, + AgentSessionHandoffMode, + AgentSessionHandoffStatus +} from '../../../../shared/agent-session-wire' +import { Button } from '@/components/ui/button' +import { Badge } from '@/components/ui/badge' +import { translate } from '@/i18n/i18n' + +type Props = { + status: AgentSessionHandoffStatus | null + isWorking: boolean + onRequest: ( + direction: AgentSessionHandoffDirection, + mode: AgentSessionHandoffMode, + action?: 'start' | 'cancel-queued' | 'retry' | 'recover' + ) => void +} + +function handoffStageCopy(status: AgentSessionHandoffStatus): string { + if (status.stage === 'preparing') { + return status.direction === 'to-tui' + ? translate('components.native-chat.handoff.stage.finishingChat', 'Finishing chat session…') + : translate( + 'components.native-chat.handoff.stage.finishingTerminal', + 'Finishing agent terminal…' + ) + } + if (status.stage === 'old-owner-stopped') { + return status.direction === 'to-tui' + ? translate('components.native-chat.handoff.stage.openingTerminal', 'Opening agent terminal…') + : translate('components.native-chat.handoff.stage.resumingChat', 'Resuming chat session…') + } + if (status.stage === 'new-owner-proving') { + return status.direction === 'to-tui' + ? translate( + 'components.native-chat.handoff.stage.verifyingTerminal', + 'Verifying agent terminal…' + ) + : translate('components.native-chat.handoff.stage.verifyingChat', 'Verifying chat session…') + } + if (status.stage === 'recovering') { + return translate('components.native-chat.handoff.stage.recovering', 'Recovering agent session…') + } + if (status.stage === 'manual-recovery') { + return translate( + 'components.native-chat.handoff.stage.manualRecovery', + 'Agent session needs recovery' + ) + } + return translate('components.native-chat.handoff.switchingOwner', 'Switching session owner…') +} + +export function StructuredAgentSessionHandoffChrome({ + status, + isWorking, + onRequest +}: Props): React.JSX.Element | null { + if (!status) { + return null + } + const owner = status?.owner ?? 'native' + const phase = status?.phase ?? 'idle' + const switching = phase === 'switching' || phase === 'waiting-for-exit' + return ( + <> +
    + + {switching + ? translate('components.native-chat.handoff.mode.switching', 'Switching') + : owner === 'tui' + ? translate('components.native-chat.handoff.mode.terminal', 'Terminal') + : translate('components.native-chat.handoff.mode.chat', 'Chat')} + +
    + {phase === 'queued' && status?.direction ? ( + <> + + {status.direction === 'to-tui' + ? translate( + 'components.native-chat.handoff.switchingAfterTurn', + 'Switching after this turn' + ) + : translate( + 'components.native-chat.handoff.returningAfterTurn', + 'Returning after this turn' + )} + + + + ) : owner === 'native' && phase === 'idle' ? ( + isWorking ? ( + <> + + + + ) : ( + + ) + ) : owner === 'tui' && phase === 'idle' ? ( + + ) : null} +
    +
    + {owner === 'tui' && phase === 'idle' ? ( +
    + + {status?.hostLabel + ? translate( + 'components.native-chat.handoff.agentOpenOnHost', + 'Agent is open in terminal on {{value0}}.', + { value0: status.hostLabel } + ) + : translate('components.native-chat.handoff.agentOpen', 'Agent is open in terminal.')} + + +
    + ) : null} + {switching ? ( +
    + {phase === 'waiting-for-exit' + ? translate( + 'components.native-chat.handoff.exitTerminal', + 'Exit the agent terminal to continue in chat.' + ) + : status?.stage + ? handoffStageCopy(status) + : translate( + 'components.native-chat.handoff.switchingOwner', + 'Switching session owner…' + )} +
    + ) : null} + {phase === 'failed' && status?.error ? ( +
    +
    + {status.error.message} + {status.direction && status.error.canRetryProof ? ( + + ) : status.direction && status.error.recoverableOwner !== 'none' ? ( + + ) : null} +
    + {status.error.details ? ( +
    + {translate('components.native-chat.handoff.details', 'Details')} +

    {status.error.details}

    +
    + ) : null} +
    + ) : null} + + ) +} diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts index 9dae22da197..8cf82b69ca1 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.test.ts @@ -117,6 +117,7 @@ describe('native chat PTY session options', () => { expect(effortResult.snapshot.map(({ id }) => id)).toEqual(['model', 'effort', 'fastMode']) expect(effortResult.snapshot.find(({ id }) => id === 'effort')).toMatchObject({ valueSource: 'dispatched', + transport: 'catalog', kind: { currentValue: 'high' } }) expect(listener).toHaveBeenCalledOnce() diff --git a/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts b/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts index aa7562b5944..3e3587e68d1 100644 --- a/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts +++ b/src/renderer/src/components/native-chat/native-chat-pty-session-options.ts @@ -98,7 +98,8 @@ export function createNativeChatPtySessionOptions( catalog, models: activeModels(), record, - mode: args.mode + mode: args.mode, + liveTransport: 'catalog' }) const listeners = new Set<(value: SessionOptionDescriptor[]) => void>() @@ -108,7 +109,8 @@ export function createNativeChatPtySessionOptions( catalog, models: activeModels(), record, - mode: args.mode + mode: args.mode, + liveTransport: 'catalog' }) for (const listener of listeners) { listener(snapshot) diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts b/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts index 60d0ffe6aba..7cdf621b272 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-labels.test.ts @@ -18,6 +18,7 @@ function modelDescriptor( id: 'model', label: 'Model', valueSource, + transport: 'catalog', settable: true, kind: { type: 'select', diff --git a/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts b/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts index 9010aa66f27..4acb1a1932a 100644 --- a/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts +++ b/src/renderer/src/components/native-chat/native-chat-session-option-snapshot.ts @@ -7,6 +7,7 @@ import { buildNativeChatSessionOptionSnapshot as buildSharedSnapshot, resolveEffectiveNativeChatModelId, withTrackedNativeChatModel, + type NativeChatLiveOptionTransport, type NativeChatSessionOptionMode } from '../../../../shared/native-chat-session-option-snapshot' import { @@ -15,7 +16,7 @@ import { } from '../../../../shared/native-chat-session-option-state' import { translate } from '@/i18n/i18n' -export type { NativeChatSessionOptionMode } +export type { NativeChatLiveOptionTransport, NativeChatSessionOptionMode } export { flattenNativeChatSessionOptionRecord, resolveEffectiveNativeChatModelId, @@ -27,6 +28,7 @@ export function buildNativeChatSessionOptionSnapshot(args: { models: readonly CatalogModel[] record: NativeChatSessionOptionRecord mode: NativeChatSessionOptionMode + liveTransport: NativeChatLiveOptionTransport }): SessionOptionDescriptor[] { return buildSharedSnapshot({ ...args, diff --git a/src/renderer/src/components/native-chat/use-structured-agent-session.ts b/src/renderer/src/components/native-chat/use-structured-agent-session.ts index 5bea1af8c50..4f8c37116af 100644 --- a/src/renderer/src/components/native-chat/use-structured-agent-session.ts +++ b/src/renderer/src/components/native-chat/use-structured-agent-session.ts @@ -136,6 +136,11 @@ export function useStructuredAgentSession(args: { [sessionId, target] ) + // Turns are what confirm an option: the provider names the model it is running + // on the frame that opens each one, so re-read the options as a turn changes + // rather than leaving the last write unconfirmed for the life of the session. + const turnId = activeStructuredAgentSessionTurnId(state.items) + useEffect(() => { if (!isVisible || !optionCatalog) { return @@ -157,7 +162,7 @@ export function useStructuredAgentSession(args: { return () => { stale = true } - }, [isVisible, optionCatalog, sessionId, state.fence, target]) + }, [isVisible, optionCatalog, sessionId, state.fence, target, turnId]) const optionSnapshot = useMemo( () => structuredAgentSessionOptionSnapshot(optionState), @@ -219,7 +224,6 @@ export function useStructuredAgentSession(args: { (item.body.kind === 'approval' || item.body.kind === 'question') && item.body.resolution.state === 'pending' ) - const turnId = activeStructuredAgentSessionTurnId(state.items) return { messages: projectStructuredAgentSessionMessages( state.items, diff --git a/src/renderer/src/components/settings/ExperimentalPane.test.tsx b/src/renderer/src/components/settings/ExperimentalPane.test.tsx index b421b77bfc2..ba8e1518921 100644 --- a/src/renderer/src/components/settings/ExperimentalPane.test.tsx +++ b/src/renderer/src/components/settings/ExperimentalPane.test.tsx @@ -258,8 +258,12 @@ describe('ExperimentalPane', () => { }) expect(container.textContent).toContain('Use updated structured native chat') + // The one opt-in gates both providers, so its copy must not name only Codex. expect(container.textContent).toContain( - 'Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.' + 'Opt in to the host-owned structured chat runtime for Codex and Claude.' + ) + expect(container.textContent).toContain( + 'Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.' ) expect(container.textContent).toContain('Default view') root.unmount() diff --git a/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx b/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx index 93c4b1c899d..85d27dae2e3 100644 --- a/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx +++ b/src/renderer/src/components/settings/NativeChatExperimentalSetting.tsx @@ -126,13 +126,13 @@ export function NativeChatExperimentalSetting({

    {translate( 'auto.components.settings.ExperimentalPane.nativeChat.structuredCopy', - 'Opt in to the host-owned structured Codex runtime. Off keeps the existing terminal-backed chat path.' + 'Opt in to the host-owned structured chat runtime for Codex and Claude. Off keeps the existing terminal-backed chat path.' )}

    {translate( 'auto.components.settings.ExperimentalPane.nativeChat.structuredScope', - 'Local macOS and Linux sessions only for now. Windows, WSL, and remote execution hosts (including SSH) continue to use terminal chat.' + 'Local sessions only for now. WSL and remote execution hosts (including SSH) continue to use terminal chat, and Windows falls back to it unless Orca can read process start times.' )}

    diff --git a/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx b/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx index 5709dc87f90..13f3a0c883d 100644 --- a/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx +++ b/src/renderer/src/components/sidebar/NonGitFolderDialog.tsx @@ -17,8 +17,9 @@ import { markOnboardingProjectAdded } from '@/lib/onboarding-project-checklist' import { translate } from '@/i18n/i18n' import { upsertAddedRepoWithProjectHostSetup } from './add-repo-store-upsert' import { worktreeRefreshOptions } from './add-repo-runtime-owner' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' const NonGitFolderDialog = React.memo(function NonGitFolderDialog() { const activeModal = useAppStore((s) => s.activeModal) @@ -101,8 +102,11 @@ const NonGitFolderDialog = React.memo(function NonGitFolderDialog() { ...(launch.startup ? { startup: launch.startup } : {}), ...(launch.route === 'structured-native-chat' ? { providesInitialSurface: true } : {}) }) - if (launch.route === 'structured-native-chat' && launch.agent === 'codex') { - const structured = startStructuredCodexLaunch(folderWorktree.id) + if ( + launch.route === 'structured-native-chat' && + isAgentSessionHandleProvider(launch.agent) + ) { + const structured = startStructuredAgentLaunch(folderWorktree.id, launch.agent) const fallback = structured.claimDefinitiveRefusalFallback(() => { activateAndRevealWorktree(folderWorktree.id, { sidebarRevealBehavior: 'auto', diff --git a/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts b/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts index 661bf623880..b6f1a1654f3 100644 --- a/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts +++ b/src/renderer/src/components/sidebar/folder-workspace-composer-submit.ts @@ -28,8 +28,9 @@ import { resolveAgentLaunchRoute } from '@/lib/agent-launch-routing' import { readLocalRuntimeCapabilities } from '@/runtime/local-runtime-capabilities' -import { startStructuredCodexLaunch } from '@/lib/structured-agent-session-launch' -import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-codex-session' +import { startStructuredAgentLaunch } from '@/lib/structured-agent-session-launch' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' +import { StructuredAgentSessionCreateRefusalError } from '@/lib/launch-structured-agent-session' import { useAppStore } from '@/store' import { buildFolderWorkspaceLinkedStartupPlan, @@ -232,8 +233,8 @@ export async function submitFolderWorkspaceCreate({ runtimeEnvironmentId }) let structuredLaunchAccepted = structuredLaunch - if (structuredLaunch && quickAgent === 'codex') { - const launch = startStructuredCodexLaunch(folderWorkspaceKey(workspace.id), { + if (structuredLaunch && isAgentSessionHandleProvider(quickAgent)) { + const launch = startStructuredAgentLaunch(folderWorkspaceKey(workspace.id), quickAgent, { prompt: launchDraftPrompt ?? note }) const refusalFallback = launch.claimDefinitiveRefusalFallback(async () => { diff --git a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx index 85d978e3bf6..6d5b523c2af 100644 --- a/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx +++ b/src/renderer/src/components/tab-bar/QuickLaunchButton.tsx @@ -8,6 +8,7 @@ import { useAgentDetectionTargetForWorktree } from '@/hooks/useAgentDetectionTar import { useDetectedAgents } from '@/hooks/useDetectedAgents' import { useOptionalShortcutLabel } from '@/hooks/useShortcutLabel' import { launchAgentInNewTab } from '@/lib/launch-agent-in-new-tab' +import { isAgentSessionHandleProvider } from '../../../../shared/agent-session-provider-handle' import type { TuiAgent } from '../../../../shared/tui-agent' import type { LaunchSource } from '../../../../shared/telemetry-events' import { @@ -15,7 +16,7 @@ import { filterEnabledTuiAgents } from '../../../../shared/tui-agent-selection' import { translate } from '@/i18n/i18n' -import { useStructuredCodexLaunchStatus } from '@/lib/structured-agent-session-launch' +import { useStructuredAgentLaunchStatus } from '@/lib/structured-agent-session-launch' export type QuickLaunchAgentMenuItemsProps = { worktreeId: string @@ -117,7 +118,12 @@ function QuickLaunchAgentMenuItemsInner({ const openSettingsPage = useAppStore((s) => s.openSettingsPage) const openSettingsTarget = useAppStore((s) => s.openSettingsTarget) const newAgentShortcut = useOptionalShortcutLabel('tab.newAgent') - const structuredCodexLaunchStatus = useStructuredCodexLaunchStatus(worktreeId) + // One hook per structured provider: the launch registry is keyed by agent, and hooks cannot run + // inside the agent list's render loop. + const structuredLaunchStatusByAgent = { + claude: useStructuredAgentLaunchStatus(worktreeId, 'claude'), + codex: useStructuredAgentLaunchStatus(worktreeId, 'codex') + } const openAgentSettings = useCallback(() => { openSettingsTarget({ pane: 'agents', repoId: null }) @@ -199,26 +205,33 @@ function QuickLaunchAgentMenuItemsInner({ {agents.map((agent) => { const entry = getCatalogEntry(agent) const label = entry?.label ?? agent - const isStructuredCodexPending = - agent === 'codex' && structuredCodexLaunchStatus === 'pending' - const menuLabel = isStructuredCodexPending ? 'Starting Codex chat…' : label + const isStructuredLaunchPending = + isAgentSessionHandleProvider(agent) && structuredLaunchStatusByAgent[agent] === 'pending' + const pendingLabel = translate( + 'components.native-chat.structuredSessionLaunchPending', + 'Starting {{value0}} chat…', + { value0: label } + ) + const menuLabel = isStructuredLaunchPending ? pendingLabel : label const showsDefaultAgentShortcut = newAgentShortcut !== null && defaultAgent !== 'blank' && agent === defaultAgent return ( runLaunch(agent)} className="gap-2 rounded-[7px] px-2 py-1.5 text-[12px] leading-5 font-medium" - title={translate( - 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', - isStructuredCodexPending - ? 'Starting Codex chat…' - : 'Launch {{value0}} in a new terminal', - isStructuredCodexPending ? undefined : { value0: label } - )} + title={ + isStructuredLaunchPending + ? pendingLabel + : translate( + 'auto.components.tab.bar.QuickLaunchButton.ec2adf093e', + 'Launch {{value0}} in a new terminal', + { value0: label } + ) + } > - {isStructuredCodexPending ? ( + {isStructuredLaunchPending ? (